Skip to main content

detcore/
lib.rs

1/*
2 * Copyright (c) Meta Platforms, Inc. and affiliates.
3 * All rights reserved.
4 *
5 * This source code is licensed under the BSD-style license found in the
6 * LICENSE file in the root directory of this source tree.
7 */
8
9//! Detcore is a Reverie tool that determinizes the execution of a process.
10//!
11//! # Backend-abstraction commandment
12//!
13//! Detcore is a *tool* written against Reverie's **abstract** instrumentation
14//! interface (the `reverie` crate). It depends only on those traits and types
15//! and is deliberately ignorant of how a guest is actually traced.
16//!
17//! Detcore MUST NEVER depend on or import a concrete Reverie backend or support
18//! crate -- any `reverie-*` crate other than the abstract `reverie-core`
19//! interface. Choosing and instantiating a backend, and running a detcore tool
20//! against it, is the sole responsibility of the `hermit-cli` package. There
21//! are no backend-specific hacks in detcore: any tracing-mechanism-specific
22//! behavior belongs behind the Reverie abstraction, not here.
23//!
24//! Why: Hermit follows Reverie's abstract model. A backend dependency in
25//! detcore would couple the determinism engine to one tracing mechanism and
26//! break the clean abstraction boundary that lets the same tool run over any
27//! backend.
28//!
29//! The one allowed exception is test-only: detcore's own integration tests
30//! (under `detcore/tests/`, wired via the `reverie-ptrace` **dev-dependency**)
31//! drive a real tracer to exercise the tool. That coupling never reaches the
32//! shipped library. This invariant is enforced in CI by
33//! `scripts/check-detcore-backend-abstraction.sh`.
34
35#![deny(clippy::all)]
36#![deny(missing_docs)]
37#![allow(clippy::uninlined_format_args)]
38
39mod config;
40mod consts;
41mod cpuid;
42mod digest;
43mod dirents;
44/// Schedule-alignment and edit-distance algorithms shared by Hermit tools.
45#[allow(missing_docs)]
46pub mod edit_distance;
47mod fd;
48mod io_buffers;
49mod iovecs;
50#[allow(unused)]
51mod ivar;
52pub mod logdiff;
53mod memory;
54pub mod netlink_route;
55mod procfs;
56mod procmaps;
57pub mod random;
58mod record_or_replay;
59mod resources;
60mod scheduler;
61mod sock_diag;
62mod stat;
63mod syscall_classification;
64mod syscall_time;
65mod syscalls;
66mod tool_global;
67mod tool_local;
68pub mod util;
69
70pub mod detlog;
71pub mod preemptions;
72pub mod types;
73use std::fs::File;
74use std::io::Write;
75use std::os::unix::io::RawFd;
76use std::sync::Arc;
77use std::sync::Mutex;
78use std::time::Duration;
79
80pub use config::BlockingMode;
81pub use config::CONFIG_FINGERPRINT_ENV;
82pub use config::Config;
83pub use config::RunsPostFork;
84pub use config::SchedHeuristic;
85pub use config::config_wire_fingerprint;
86// AUTONOMOUS-BOT-IMPLEMENTED
87// TODO-HUMAN-REVIEW(PR-1120): Review the public canonical Detcore root identity.
88pub use consts::ROOT_DETPID;
89pub use digest::Digest;
90#[cfg(test)]
91use rand::RngExt as _;
92use raw_cpuid::CpuIdResult;
93use raw_cpuid::cpuid;
94pub use record_or_replay::RecordOrReplay;
95use reverie::Error;
96use reverie::ExitStatus;
97use reverie::GlobalRPC;
98use reverie::Guest;
99use reverie::Pid;
100use reverie::Rdtsc;
101use reverie::RdtscResult;
102use reverie::RegDisplay;
103use reverie::Signal;
104use reverie::Subscription;
105use reverie::Tid;
106use reverie::TimerSchedule;
107use reverie::Tool;
108pub use reverie::process::Namespace;
109use reverie::syscalls::CloneFlags;
110use reverie::syscalls::Displayable;
111use reverie::syscalls::EpollCreate1;
112use reverie::syscalls::Errno;
113use reverie::syscalls::InotifyInit1;
114use reverie::syscalls::MemoryAccess;
115use reverie::syscalls::Syscall;
116use reverie::syscalls::SyscallInfo;
117use reverie::syscalls::Sysno;
118pub use scheduler::Priority;
119pub use scheduler::runqueue::DEFAULT_PRIORITY;
120pub use scheduler::runqueue::FIRST_PRIORITY;
121pub use scheduler::runqueue::LAST_PRIORITY;
122pub use tool_global::BackendFailureCleanup;
123pub use tool_global::GlobalState;
124use tool_global::ThreadDeregistration;
125use tool_global::acknowledge_robust_list_exit_time;
126use tool_global::create_child_thread;
127use tool_global::create_vfork_child_thread;
128use tool_global::deregister_thread;
129pub use tool_global::format_unsupported_syscall_warning;
130pub use tool_global::prepare_exec;
131use tool_global::report_unsupported_syscall;
132use tool_global::robust_list_wakes_after_exit;
133
134fn select_thread_exit_detpid(
135    thread_detpid: Option<DetPid>,
136    process_detpid: DetPid,
137) -> (DetPid, bool) {
138    match thread_detpid {
139        Some(detpid) => (detpid, false),
140        None => (process_detpid, true),
141    }
142}
143
144fn select_thread_start_detpid(thread_detpid: Option<DetPid>, guest_pid: Pid) -> DetPid {
145    thread_detpid.unwrap_or_else(|| DetPid::from_raw(guest_pid.into()))
146}
147
148fn is_root_thread_start(is_root_process: bool, dettid: DetTid, detpid: DetPid) -> bool {
149    is_root_process && dettid == detpid
150}
151
152// AUTONOMOUS-BOT-IMPLEMENTED
153// TODO-HUMAN-REVIEW(PR-644): Review the typed fail-closed backend signal.
154/// Identifies an unsupported syscall that a backend must terminate without unwinding.
155#[derive(Debug)]
156pub struct UnsupportedSyscallError(pub Sysno);
157
158impl std::fmt::Display for UnsupportedSyscallError {
159    fn fmt(&self, formatter: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
160        write!(formatter, "unsupported syscall: {:?}", self.0)
161    }
162}
163
164impl std::error::Error for UnsupportedSyscallError {}
165pub use tool_local::Detcore;
166pub use tool_local::FileMetadata;
167/// Returns whether the audited runtime policy classifies `sysno` as unsupported.
168// AUTONOMOUS-BOT-IMPLEMENTED
169// TODO-HUMAN-REVIEW(PR-644): Review the copied-DBT-child classification surface.
170pub fn is_unsupported_syscall(sysno: Sysno) -> bool {
171    matches!(
172        syscall_classification::classify_syscall(sysno),
173        syscall_classification::SyscallClassification::Unsupported
174    )
175}
176
177/// Every syscall in the pinned x86_64 table, including the final entry.
178///
179/// Backends that sweep the classification table must use this rather than
180/// `Sysno::iter()`, which stops one short and silently drops the last row.
181pub fn all_pinned_syscalls() -> impl Iterator<Item = Sysno> {
182    syscall_classification::all_pinned_syscalls()
183}
184
185/// Returns whether the audited runtime policy classifies `sysno` as
186/// `Determinized` — that is, Detcore either models the syscall with a handler
187/// or applies an explicit deterministic refusal policy to it.
188///
189/// This is the complement of the refusal boundary below. A backend that
190/// executes guest syscalls outside `handle_syscall_event` needs BOTH: the
191/// refusal set tells it which syscalls to answer with a fixed errno, and this
192/// predicate tells it which syscalls Detcore claims to determinize at all.
193/// Running a `Determinized` syscall natively is a determinism hole even when it
194/// is not in the refusal set, because the modelling that makes it deterministic
195/// lives in a handler the backend never entered.
196pub fn is_determinized_syscall(sysno: Sysno) -> bool {
197    matches!(
198        syscall_classification::classify_syscall(sysno),
199        syscall_classification::SyscallClassification::Determinized
200    )
201}
202
203/// Returns whether `sysno` is a kernel-keyring syscall (`add_key`,
204/// `request_key`, `keyctl`) that Detcore hides behind a deterministic
205/// `CONFIG_KEYS`-absent boundary under the default fail-closed policy.
206// AUTONOMOUS-BOT-IMPLEMENTED
207// TODO-HUMAN-REVIEW(PR-916): Exposed so the copied-DBT-child policy can preserve
208// the same keyring isolation boundary that the reclassification (PR-848) moved
209// out of the Unsupported set.
210pub fn is_kernel_keyring_syscall(sysno: Sysno) -> bool {
211    syscall_classification::is_kernel_keyring_syscall(sysno)
212}
213
214/// Returns whether Detcore deterministically refuses `sysno` with a fixed
215/// errno when the fail-closed policy is active, without consulting the host.
216///
217/// This is the boundary backends that execute guest syscalls outside Detcore's
218/// `handle_syscall_event` dispatcher (the DBT copied-child fast path and the
219/// KVM executor) consult to enforce the same fixed refusal the ptrace path
220/// enforces. It deliberately excludes emulated / no-op / host-forwarding
221/// families (credential no-ops, `timer_create`, AF_UNIX autobind, `openat2`,
222/// `copy_file_range`), because fail-closing a copied child for those would
223/// diverge from the ptrace path rather than match it.
224// AUTONOMOUS-BOT-IMPLEMENTED
225// TODO-HUMAN-REVIEW(PR-978): Review the copied-DBT-child deterministic-refusal surface.
226pub fn is_deterministically_refused_syscall(sysno: Sysno) -> bool {
227    syscall_classification::is_deterministically_refused_syscall(sysno)
228}
229
230/// Returns whether `sysno` is refused by the default fail-closed policy but
231/// forwarded under the explicit compatibility opt-out. The legacy
232/// `strict_only` name is retained for API compatibility.
233pub fn is_strict_only_deterministic_refusal_syscall(sysno: Sysno) -> bool {
234    syscall_classification::is_strict_only_deterministic_refusal_syscall(sysno)
235}
236
237use tool_local::PosixTimers;
238use tool_local::ProcessCpuTime;
239pub use tool_local::ThreadState;
240pub use tool_local::ThreadStats;
241pub use tool_local::thread_rng_from_parent;
242use tracing::debug;
243use tracing::error;
244use tracing::info;
245use tracing::trace;
246use tracing::warn;
247pub use types::DetTid;
248use types::*;
249pub use util::punch_out_print;
250
251use crate::resources::Permission;
252use crate::resources::ResourceID;
253use crate::syscall_classification::SyscallClassification;
254use crate::syscall_classification::classify_syscall;
255use crate::syscall_classification::is_credential_identity_noop_syscall;
256use crate::syscall_classification::is_futex2_enosys_syscall;
257use crate::syscall_classification::is_host_kernel_probe_syscall;
258use crate::syscall_classification::is_host_security_identity_probe_syscall;
259use crate::syscall_classification::is_landlock_sandbox_syscall;
260use crate::syscall_classification::is_mount_introspection_enosys_syscall;
261use crate::syscall_classification::is_mount_ns_admin_refused_syscall;
262use crate::syscall_classification::is_optional_memory_feature_syscall;
263use crate::syscall_classification::is_ownership_change_noop_syscall;
264use crate::syscall_classification::is_perf_event_enosys_syscall;
265use crate::syscall_classification::is_privileged_admin_refused_syscall;
266use crate::syscall_classification::is_privileged_observation_refused_syscall;
267use crate::syscall_classification::is_process_isolation_refused_syscall;
268use crate::syscall_classification::is_remap_file_pages_enosys_syscall;
269use crate::syscall_classification::is_unimplemented_enosys_syscall;
270use crate::syscall_classification::is_unsupported_async_ipc_syscall;
271use crate::syscall_classification::is_zero_copy_pipe_syscall;
272use crate::syscalls::helpers::with_guest_rip;
273use crate::syscalls::helpers::with_guest_time;
274use crate::syscalls::time::guest_clock_time;
275use crate::tool_global::resource_request;
276use crate::tool_global::trace_schedevent;
277use crate::tool_global::unrecoverable_shutdown;
278use crate::types::SigWrapper;
279
280#[macro_use]
281extern crate bitflags;
282
283#[cold]
284fn report_rcb_overshoot(
285    panic_on_rcb_overshoot: bool,
286    clock_value: u64,
287    delta_rcbs: u64,
288    last_timer: u64,
289) {
290    let message = format!(
291        "{} prehook: PMU RCB overshoot! Clock_value: {}. Stepped forward {} RCBs, but should have trapped at {}",
292        reverie::SKID_OVERSHOOT_MARKER,
293        clock_value,
294        delta_rcbs,
295        last_timer
296    );
297    if panic_on_rcb_overshoot {
298        panic!("{}", message);
299    }
300    reverie::record_skid_overshoot();
301    error!("{}", message);
302}
303
304fn rcb_timer_overshot(delta_rcbs: u64, last_timer: u64) -> bool {
305    delta_rcbs > last_timer
306}
307
308fn choose_rcb_timer(
309    max_rcbs_remaining: u64,
310    current_rcbs: u64,
311    next_interrupt: Option<u64>,
312) -> (u64, bool) {
313    if let Some(next_interrupt) = next_interrupt {
314        let interrupt_rcbs = next_interrupt - current_rcbs;
315        if interrupt_rcbs < max_rcbs_remaining {
316            return (interrupt_rcbs, false);
317        }
318    }
319    (max_rcbs_remaining, true)
320}
321
322impl<T: RecordOrReplay> Detcore<T> {
323    /// Registers a child whose native backend executed the clone syscall.
324    ///
325    /// The caller must initialize the child's local thread state from the same
326    /// parent state and clone flags before the child enters its start hook.
327    // TODO-HUMAN-REVIEW(PR-743): Review the backend-neutral native child registration API.
328    pub async fn register_external_child<G: Guest<Self>>(
329        &self,
330        guest: &mut G,
331        child_tid: Tid,
332        child_tid_addr: usize,
333        flags: CloneFlags,
334        exit_signal: libc::c_int,
335        physical_ids: Option<(i32, i32)>,
336    ) {
337        let child_dettid = DetTid::from_raw(child_tid.into());
338        guest.thread_state_mut().clone_flags = Some(flags);
339        if !flags.contains(CloneFlags::CLONE_THREAD) {
340            guest
341                .thread_state()
342                .prepare_child_process_cpu_time(child_dettid);
343        }
344        let parent_dettid = guest.thread_state().dettid;
345        let parent_pedigree = &mut guest.thread_state_mut().pedigree;
346        let child_pedigree = parent_pedigree.fork_mut();
347        debug!(
348            "[dtid {}] after registering external child (tid {}, pedigree {}) parent's pedigree becomes {}",
349            parent_dettid, child_dettid, child_pedigree, parent_pedigree,
350        );
351        // The kernel clone has already succeeded before an external backend
352        // calls this method, so these parent updates are not speculative. If
353        // registration observes a retired parent, create_child_thread exits
354        // that parent by tail injection; no continuing caller can retry or
355        // roll the successful clone back.
356        tool_global::create_child_thread(
357            guest,
358            child_dettid,
359            tool_global::child_tid_clear_address(flags, child_tid_addr),
360            Some(flags),
361            exit_signal,
362            physical_ids,
363        )
364        .await;
365        guest.thread_state_mut().clone_flags = None;
366    }
367    async fn passthrough<G: Guest<Self>>(
368        &self,
369        guest: &mut G,
370        call: Syscall,
371    ) -> Result<i64, Error> {
372        self.record_or_replay_preserving_tool_errors(guest, call)
373            .await
374    }
375
376    // AUTONOMOUS-BOT-IMPLEMENTED
377    // TODO-HUMAN-REVIEW(PR-643): Review unsupported-syscall reporting and fail-fast behavior.
378    /// Applies the legacy policy to an explicitly listed but unsupported syscall.
379    async fn handle_unsupported_syscall<G: Guest<Self>>(
380        &self,
381        guest: &mut G,
382        call: Syscall,
383        dettid: DetTid,
384        panic_on_unsupported_syscalls: bool,
385    ) -> Result<i64, Error> {
386        if panic_on_unsupported_syscalls {
387            error!(
388                "[detcore, dtid {}] unsupported syscall: {} = ?",
389                dettid,
390                call.display(&guest.memory()),
391            );
392            if guest.config().shutdown_on_unsupported_syscall {
393                // A fail-closed policy decision: the run named a syscall hermit
394                // cannot service and `shutdown_on_unsupported_syscall` says stop.
395                unrecoverable_shutdown(guest, detcore_model::HERMIT_POLICY_REFUSAL_EXIT).await;
396            }
397            if guest.config().exit_on_unsupported_syscall {
398                return Err(Error::Tool(anyhow::Error::new(UnsupportedSyscallError(
399                    call.number(),
400                ))));
401            }
402            panic!("unsupported syscall: {:?}", call);
403        }
404        report_unsupported_syscall(guest, call.number()).await;
405        self.passthrough(guest, call).await
406    }
407
408    /// Defense-in-depth determinism for the registers the syscall instruction
409    /// clobbers.
410    ///
411    /// On x86-64 the `syscall` instruction destroys `%rcx` (which the CPU loads
412    /// with the return instruction pointer) and `%r11` (the saved `RFLAGS`).
413    /// After a syscall these are architecturally "undefined", so hermit must not
414    /// assume a well-behaved guest ignores them: a misbehaving guest that reads
415    /// `%rcx`/`%r11` must still observe deterministic values. Reverie's
416    /// injected-syscall path can otherwise leave its *private trampoline page's*
417    /// RIP/RFLAGS in these registers, which is both nondeterministic and an
418    /// information leak of tracer internals.
419    ///
420    /// This forces both registers to the guest's own (deterministic) RIP and
421    /// RFLAGS, which is exactly what a faithful `SYSRET` would leave there. It is
422    /// a no-op when they already hold the canonical values (the common path), so
423    /// it only writes registers when something diverged. Register-preserved
424    /// arguments (`%rdi`..`%r9`, callee-saved) are deliberately left untouched:
425    /// the Linux ABI preserves them, so zeroing them would break faithful,
426    /// well-behaved programs.
427    #[cfg(target_arch = "x86_64")]
428    async fn canonicalize_syscall_clobbers<G: Guest<Self>>(&self, guest: &mut G) {
429        let mut regs = guest.regs().await;
430        // A faithful SYSRET leaves the return RIP in %rcx and RFLAGS in %r11.
431        if regs.rcx != regs.rip || regs.r11 != regs.eflags {
432            regs.rcx = regs.rip;
433            regs.r11 = regs.eflags;
434            if let Err(err) = guest.set_regs(regs).await {
435                // Best-effort: some backends cannot write registers. Do not fail
436                // the syscall over a defense-in-depth hardening step.
437                debug!(
438                    "canonicalize_syscall_clobbers: set_regs unsupported/failed: {}",
439                    err
440                );
441            }
442        }
443    }
444
445    /// No-op on architectures without the x86-64 `%rcx`/`%r11` syscall clobber.
446    #[cfg(not(target_arch = "x86_64"))]
447    async fn canonicalize_syscall_clobbers<G: Guest<Self>>(&self, _guest: &mut G) {}
448
449    /// Update logical thread time with any outstanding ticks of the Reverie clock.  Returns a list
450    /// of corresponding Branch/OtherInstructions events if schedule recording is enabled.
451    ///
452    /// # Arguments
453    ///
454    /// * `precise_branch`: if true, there were no non-branch instructions since the last recorded branch instruction.
455    async fn update_logical_time_rcbs<G: Guest<Self>>(
456        &self,
457        guest: &mut G,
458        precise_branch: bool,
459    ) -> Option<Vec<SchedEvent>> {
460        if self.cfg.max_timeslice.is_some() {
461            let precise_timers = !guest.config().imprecise_timers;
462            // TODO(T86591083): we might need to not always increment as a hack fix
463            // for deterministic virtual time without sequentialize threads.
464            let clock_value = guest.read_clock().expect("Couldn't read clock");
465            // N.B. clock_value does not yet include any updates for the inbound
466            // syscall/instruction because this function is the very first thing that
467            // happens in each type of handler.
468            let thread_state = guest.thread_state_mut();
469            let dettid = thread_state.dettid;
470            assert!(thread_state.committed_clock_value <= clock_value);
471            let delta_rcbs: u64 = clock_value - thread_state.committed_clock_value;
472            if self.cfg.use_rcb_time() {
473                // AUTONOMOUS-BOT-IMPLEMENTED
474                // TODO-HUMAN-REVIEW(PR-1151)
475                if thread_state.chaos_slowdown_active {
476                    let factor = thread_state.rcb_time_multiplier();
477                    thread_state
478                        .thread_logical_time
479                        .add_rcbs_with_multiplier(delta_rcbs, factor);
480                } else {
481                    thread_state.thread_logical_time.add_rcbs(delta_rcbs);
482                }
483            }
484            thread_state.account_process_cpu_time();
485            thread_state.committed_clock_value = clock_value;
486
487            if thread_state.end_of_timeslice.is_some() {
488                if let Some(last_timer) = thread_state.last_rcb_timer
489                    && rcb_timer_overshot(delta_rcbs, last_timer)
490                    && precise_timers
491                {
492                    report_rcb_overshoot(
493                        self.cfg.panic_on_rcb_overshoot,
494                        clock_value,
495                        delta_rcbs,
496                        last_timer,
497                    );
498                    // Preserve timer state. `pre_handler_hook` will yield through the normal
499                    // scheduler path if the slice expired; `post_handler_hook` will otherwise
500                    // re-arm an overshot `interrupt_at` timer.
501                }
502                // Otherwise we're very early, at the prehook of handle_thread_start.
503            } else {
504                panic!(
505                    "Invariant violation: end_of_timeslice is None during update_logical_time_rcbs..."
506                )
507            }
508
509            trace!(
510                "[dtid {}] updated rcb clock, new logical time: {:?}, i.e. {}, timeslice end: {}, local rcb clock_value {:?}",
511                dettid,
512                &thread_state.thread_logical_time,
513                thread_state.thread_logical_time.as_nanos(),
514                thread_state
515                    .end_of_timeslice
516                    .map_or_else(|| "".to_string(), |x| format!("{}", x)),
517                clock_value,
518            );
519            if self.cfg.use_rcb_time() && self.cfg.should_trace_schedevent() {
520                let mut vec = Vec::new();
521                let ev = with_guest_time(
522                    guest,
523                    SchedEvent::branches(
524                        dettid,
525                        delta_rcbs
526                            .try_into()
527                            .expect("should not have more than 2^32 branches at once"),
528                    ),
529                );
530                let ev = if precise_branch {
531                    with_guest_rip(guest, ev).await
532                } else {
533                    ev
534                };
535
536                if delta_rcbs > 0 {
537                    // We don't fill the end_rip here, because the current rip is NOT precisely the
538                    // end of this block of branch events.  Other instructions may have occured
539                    // since the last branch.
540                    vec.push(ev)
541                } else {
542                    trace!(
543                        "[detcore, dtid {}] Refusing to record zero-braches event: {:?}",
544                        &ev.dettid, ev
545                    );
546                }
547                if !precise_branch {
548                    // This will ALWAYS record, even if the branches above are zero.
549                    let ev2 = with_guest_time(
550                        guest,
551                        SchedEvent {
552                            dettid,
553                            op: Op::OtherInstructions,
554                            count: 1,
555                            start_rip: None,
556                            end_rip: None,
557                            end_time: None,
558                        },
559                    );
560                    // Fill in end_rip because current rip represents the end of this event.
561                    let ev2 = with_guest_rip(guest, ev2).await;
562                    vec.push(ev2);
563                }
564                Some(vec)
565            } else {
566                None
567            }
568        } else {
569            None
570        }
571    }
572
573    /// A common hook called at the start of *every* handler, just after we receive
574    /// control from the guest.
575    async fn pre_handler_hook<G: Guest<Self>>(&self, guest: &mut G, precise_branch: bool) {
576        // A handler that left early (an error return) may not have cleared
577        // this; no request made by this new handler belongs to that syscall.
578        guest.thread_state_mut().in_uncharged_bootstrap_syscall = false;
579        let dettid = guest.thread_state().dettid;
580        let evs = self.update_logical_time_rcbs(guest, precise_branch).await;
581
582        if guest.thread_state().guest_past_first_execve() {
583            detlog_debug!(
584                "(pre) registers [dtid {}][rcbs {}]. {}",
585                dettid,
586                guest.thread_state().thread_logical_time.rcbs(),
587                guest.regs().await.display()
588            );
589        }
590        trace!(
591            "prehook [dtid {}] Updating rcbs and checking time remaining.",
592            dettid
593        );
594        if let Some(vec) = evs {
595            for ev in vec {
596                trace_schedevent(guest, ev, false).await;
597            }
598        }
599
600        self.end_timeslice_if_needed(guest).await;
601    }
602
603    // AUTONOMOUS-BOT-IMPLEMENTED
604    /// Yield when accumulated logical time reaches the syscall-boundary target deadline.
605    async fn end_timeslice_if_needed<G: Guest<Self>>(&self, guest: &mut G) {
606        let thread_state = guest.thread_state();
607        let Some(slice_end) = thread_state.end_of_timeslice else {
608            return;
609        };
610        if !thread_state.timeslice_expired() {
611            return;
612        }
613
614        trace!(
615            "[dtid {}] logical time {} reached timeslice target {}",
616            thread_state.dettid,
617            thread_state.thread_logical_time.as_nanos(),
618            slice_end
619        );
620        self.end_timeslice(guest).await;
621    }
622
623    /// A common hook called at the end of *every* handler, just before returning control
624    /// to the guest. This enforces the logical target and resets the PMU maximum timer.
625    ///
626    /// However, note that the thread's timeslice (turn) may have expired DURING this handler.
627    /// Therefore the timeslice can end in the posthook as well as in the prehook.
628    async fn post_handler_hook<G: Guest<Self>>(&self, guest: &mut G) {
629        self.end_timeslice_if_needed(guest).await;
630
631        let dettid = guest.thread_state().dettid;
632        let mut current_time = guest.thread_state().thread_logical_time.as_nanos();
633
634        if let Some(mut max_timeslice_end) = guest.thread_state().max_timeslice_end {
635            assert!(guest.config().max_timeslice.is_some());
636            let mut replay_rcb_end = guest.thread_state().replay_rcb_end;
637            // TODO: get rid of fractional NANOS_PER_RCB so it's clear that this does not lose precision:
638            // AUTONOMOUS-BOT-IMPLEMENTED
639            // TODO-HUMAN-REVIEW(PR-1151)
640            let clock_multiplier = guest.config().clock_multiplier.unwrap_or(1.0)
641                * guest.thread_state().rcb_time_multiplier().as_f64();
642            let epsilon = Duration::from_nanos((NANOS_PER_RCB * clock_multiplier).ceil() as u64);
643
644            if replay_rcb_end.is_none() && current_time + epsilon > max_timeslice_end {
645                trace!(
646                    "posthook [dtid {}] less than one RCB remains before PMU maximum {}; ending slice",
647                    dettid, max_timeslice_end
648                );
649                self.end_timeslice(guest).await;
650                max_timeslice_end = guest
651                    .thread_state()
652                    .max_timeslice_end
653                    .expect("ending a PMU-backed timeslice must install a new maximum");
654                current_time = guest.thread_state().thread_logical_time.as_nanos();
655                replay_rcb_end = guest.thread_state().replay_rcb_end;
656            }
657            if replay_rcb_end.is_none() && current_time + epsilon > max_timeslice_end {
658                panic!(
659                    "Ended time slice, but current time {} is still beyond PMU maximum {}",
660                    current_time, max_timeslice_end
661                );
662            }
663
664            let current_rcbs = guest.thread_state().thread_logical_time.rcbs();
665            let current_pmu_rcbs = guest.thread_state().committed_clock_value;
666            let (ns_remaining, max_rcbs_remaining) = if let Some(replay_rcb_end) = replay_rcb_end {
667                assert!(
668                    replay_rcb_end > current_pmu_rcbs,
669                    "recorded PMU RCB deadline {} is not ahead of current {}",
670                    replay_rcb_end,
671                    current_pmu_rcbs
672                );
673                let logical_remaining = if max_timeslice_end > current_time {
674                    max_timeslice_end - current_time
675                } else {
676                    LogicalTime::ZERO
677                };
678                (logical_remaining, replay_rcb_end - current_pmu_rcbs)
679            } else {
680                let logical_remaining = max_timeslice_end - current_time;
681                (
682                    logical_remaining,
683                    logical_remaining.into_rcbs_with_multiplier(clock_multiplier),
684                )
685            };
686            let next_interrupt = self
687                .cfg
688                .use_rcb_time()
689                .then(|| {
690                    guest
691                        .thread_state()
692                        .interrupt_at
693                        .range((current_rcbs + 1)..)
694                        .next()
695                        .copied()
696                })
697                .flatten();
698            let (rcbs_remaining, timer_is_max) =
699                choose_rcb_timer(max_rcbs_remaining, current_rcbs, next_interrupt);
700            if let Some(next_interrupt) = next_interrupt {
701                debug!(
702                    "[dtid: {}] current rcbs: {}, next interrupt_at: {}",
703                    dettid, current_rcbs, next_interrupt
704                )
705            }
706
707            trace!(
708                "posthook [dtid {}] {} remaining before PMU maximum ({} rcbs).",
709                dettid, ns_remaining, rcbs_remaining,
710            );
711
712            if replay_rcb_end.is_none() && ns_remaining.is_zero() {
713                panic!(
714                    "Timer invariant broken: we should not exit a handler with 0 timeslice remaining."
715                );
716            }
717            assert!(rcbs_remaining > 0);
718            trace!(
719                "posthook [dtid {}] Resetting timer to {:?} RCBs in the future (current {})",
720                dettid,
721                rcbs_remaining,
722                guest.thread_state().thread_logical_time.rcbs()
723            );
724            {
725                let thread_state = guest.thread_state_mut();
726                thread_state.last_rcb_timer = Some(rcbs_remaining);
727                thread_state.last_rcb_timer_is_max = timer_is_max;
728            }
729
730            if guest.config().imprecise_timers {
731                guest
732                    .set_timer(TimerSchedule::Rcbs(rcbs_remaining))
733                    .expect("Failed to set timer");
734            } else {
735                guest
736                    .set_timer_precise(TimerSchedule::Rcbs(rcbs_remaining))
737                    .expect("Failed to set timer");
738            }
739        } else {
740            assert!(guest.config().max_timeslice.is_none());
741            guest.thread_state_mut().last_rcb_timer = None;
742            guest.thread_state_mut().last_rcb_timer_is_max = false;
743        }
744
745        if guest.thread_state().guest_past_first_execve() {
746            detlog_debug!(
747                "(post) registers [dtid {}][rcbs {}]. {}",
748                dettid,
749                guest.thread_state().thread_logical_time.rcbs(),
750                guest.regs().await.display(),
751            );
752        }
753    }
754
755    /// End this logical timeslice and talk to the scheduler before continuing.
756    ///
757    /// Effects
758    ///  - ends timeslice (mutating thread stats and both deadlines)
759    ///  - priority change / yield RPC
760    async fn end_timeslice<G: Guest<Self>>(&self, guest: &mut G) {
761        self.end_timeslice_with_sched_yield(guest, false).await;
762    }
763
764    async fn end_timeslice_for_sched_yield<G: Guest<Self>>(&self, guest: &mut G) {
765        self.end_timeslice_with_sched_yield(guest, true).await;
766    }
767
768    async fn end_timeslice_with_sched_yield<G: Guest<Self>>(
769        &self,
770        guest: &mut G,
771        mut explicit_sched_yield: bool,
772    ) {
773        let chaos = guest.config().chaos;
774        loop {
775            let thread_state = guest.thread_state();
776            let dettid = thread_state.dettid;
777            let end_time = thread_state.thread_logical_time.as_nanos();
778            info!(
779                "[detcore, dtid {}] ending timeslice T{}. {} syscalls and {} signals this timeslice.",
780                dettid,
781                thread_state.stats.timeslice_count,
782                thread_state.stats.timeslice_syscall_count,
783                thread_state.stats.timeslice_signal_count,
784            );
785            let maybe_prio = guest.thread_state_mut().next_timeslice(&self.cfg);
786
787            // Depending on chaos mode, a received timer event is either a preemption or a changepoint
788            let req = if let Some(prio) = maybe_prio {
789                Self::priority_changepoint_request(guest, end_time, prio)
790            } else if chaos {
791                Self::random_priority_changepoint_request(guest, end_time)
792            } else if explicit_sched_yield && self.cfg.replay_schedule_from.is_none() {
793                Self::sched_yield_request(guest)
794            } else {
795                Self::yield_request(guest)
796            };
797            resource_request(guest, req).await;
798
799            // AUTONOMOUS-BOT-IMPLEMENTED
800            // TODO-HUMAN-REVIEW(PR-1151)
801            // Multiple scheduler commits can occur without an intervening
802            // conditional branch. Exact-RCB replay represents those as
803            // adjacent zero-RCB slices, which must be consumed before the
804            // guest resumes.
805            if !guest.thread_state().timeslice_expired() {
806                break;
807            }
808            explicit_sched_yield = false;
809        }
810    }
811
812    /// Hash the guest REGISTER FILE and log it.
813    ///
814    /// # The sampling boundary: GUEST-LOGICAL CONTROL, never handler interior
815    ///
816    /// This is called from exactly one place -- immediately after a syscall has finished and its
817    /// result has been written back, before the guest resumes. At that instant the guest
818    /// LOGICALLY HAS CONTROL: the architectural state is what the guest itself would observe at
819    /// its own RIP, and it is the same instant at which stack and heap are already hashed.
820    ///
821    /// It is deliberately NOT sampled anywhere inside a tool handler. A backend that runs its
822    /// handler IN-GUEST (sabre, liteinst, e9patch) executes instructions the ptrace reference
823    /// never executes, using guest registers as scratch while it does. Registers there
824    /// legitimately differ across backends, so comparing them would report correct behaviour as a
825    /// divergence and burn the prefix-depth ratchet on artifacts. Handler-interior state is out of
826    /// the domain, not excluded from it by a filter -- the same "define the domain" rule the heap
827    /// definition follows.
828    ///
829    /// # What is in the hash, and what is deliberately not
830    ///
831    /// Included: the general-purpose registers the guest can observe, `rip`, `rsp`, `rflags`,
832    /// `orig_rax`, and the TLS bases `fs_base`/`gs_base`.
833    ///
834    /// EXCLUDED, with reasons rather than by convenience:
835    /// * `rcx` and `r11` -- architecturally clobbered by the `SYSCALL` instruction, which stores
836    ///   the return RIP and RFLAGS in them. They carry no information beyond `rip`/`eflags`, which
837    ///   ARE hashed, and a patching backend that reaches the kernel by some route other than a
838    ///   bare `SYSCALL` will leave different values there for a reason that is not a determinism
839    ///   defect.
840    /// * The segment selectors `cs`/`ss`/`ds`/`es`/`fs`/`gs` -- constant for a 64-bit userspace
841    ///   guest, so they add no signal; the TLS BASES are what a guest actually observes and those
842    ///   are hashed.
843    fn detlog_registers<G: Guest<Self>>(
844        &self,
845        guest: &mut G,
846        regs: &libc::user_regs_struct,
847        seq: u64,
848    ) {
849        if !self.cfg.detlog_regs {
850            return;
851        }
852        // COST TIER: cadence 1 == full (every control point); N > 1 == spot-check every Nth.
853        // The cadence index is a PER-THREAD counter, NOT a shared one: a global atomic would be
854        // incremented in whatever order threads happen to reach it, so the cadence -- and
855        // therefore which points got sampled -- would itself be nondeterministic. A determinism
856        // instrument must not have a nondeterministic sampling schedule.
857        //
858        // It is `stats.syscall_count`, which starts at ZERO, rather than the syscall ORDINAL used
859        // in the log (which starts at 2). With the ordinal, a guest whose control points never
860        // land on a multiple of the cadence emitted NOTHING and the run still reported PASS -- a
861        // spot-tier green backed by zero samples. Indexing from zero makes the first control point
862        // of every thread always sampled, so a spot-tier run can never be silently empty.
863        let cadence = self.cfg.detlog_regs_cadence.max(1);
864        let index = {
865            let stats = &mut guest.thread_state_mut().stats;
866            let i = stats.regs_sample_index;
867            stats.regs_sample_index = i.saturating_add(1);
868            i
869        };
870        let _ = seq;
871        if !index.is_multiple_of(cadence) {
872            return;
873        }
874        let tier = if cadence == 1 {
875            "full".to_string()
876        } else {
877            format!("spot-1/{cadence}")
878        };
879        let mut bytes = Vec::with_capacity(19 * 8);
880        for v in [
881            regs.rax,
882            regs.rbx,
883            regs.rdx,
884            regs.rsi,
885            regs.rdi,
886            regs.rbp,
887            regs.rsp,
888            regs.r8,
889            regs.r9,
890            regs.r10,
891            regs.r12,
892            regs.r13,
893            regs.r14,
894            regs.r15,
895            regs.rip,
896            regs.eflags,
897            regs.orig_rax,
898            regs.fs_base,
899            regs.gs_base,
900        ] {
901            bytes.extend_from_slice(&v.to_le_bytes());
902        }
903        detlog!(
904            "[registers][dtid {}] control_point=syscall-exit tier={} {}",
905            guest.thread_state().dettid,
906            tier,
907            Digest::new(&bytes)
908        );
909    }
910
911    fn detlog_memory_maps<G: Guest<Self>>(&self, guest: &mut G) -> Result<(), reverie::Error> {
912        if !(self.cfg.detlog_stack || self.cfg.detlog_heap) {
913            // Don't incur the *significant* performance penalty for reading
914            // /proc/maps unless one of these flags is enabled.
915            return Ok(());
916        }
917        // ...and don't incur it when nothing would observe the record either.
918        //
919        // The hash below is an argument to `detlog!`, so `tracing` already skips
920        // it when INFO is disabled. Enumerating the maps is NOT: it happens
921        // before the macro, on every syscall, whether or not a record is ever
922        // written. Measured on a QEMU/Linux boot with `RUST_LOG` unset, where
923        // each run emitted 123 bytes of log in total: no flag 43.71s,
924        // `--detlog-stack` 190.37s (4.36x), `--detlog-heap` 207.90s (4.76x).
925        // `--detlog-regs` was already inert because everything it does before
926        // its own `detlog!` is trivial; this restores the same property here.
927        //
928        // Skipping is invisible to the guest: enumerating maps and hashing guest
929        // memory are host-side observations of the tracee that neither issue a
930        // guest syscall nor advance virtual time, so a run that skips them
931        // executes the same guest instruction stream as one that does not.
932        if !detlog_observed!() {
933            return Ok(());
934        }
935        // Out-of-process backends (e.g. KVM) report their guest memory regions
936        // directly, because `guest.pid()` is the host VMM process there and its
937        // `/proc/<pid>/maps` describes the VMM, not the guest address space.
938        // Reading those host addresses through `guest.memory()` (guest-address
939        // space) would fault and abort the syscall. When the backend supplies
940        // regions, hash those guest ranges; otherwise fall back to the ptrace
941        // path of parsing `/proc/<pid>/maps`.
942        if let Some(regions) = guest.detlog_memory_regions() {
943            for region in regions {
944                let want = match region.kind {
945                    reverie::DetlogRegionKind::Stack => self.cfg.detlog_stack,
946                    reverie::DetlogRegionKind::Heap => self.cfg.detlog_heap,
947                };
948                if !want {
949                    continue;
950                }
951                let dettid = guest.thread_state().dettid;
952                detlog!(
953                    "[memory][dtid {}] {:?} {:#x}-{:#x}->{}",
954                    dettid,
955                    region.kind,
956                    region.start,
957                    region.end,
958                    procmaps::compute_hash_range(guest, region.start, region.end)?
959                )
960            }
961            return Ok(());
962        }
963        let mut labelled_heap = false;
964        for mmap in procmaps::from_pid(guest.pid(), |map| match map.pathname {
965            procmaps::MMapPath::Stack if self.cfg.detlog_stack => true,
966            procmaps::MMapPath::Heap if self.cfg.detlog_heap => true,
967            _ => false,
968        })? {
969            labelled_heap |= matches!(mmap.pathname, procmaps::MMapPath::Heap);
970            let dettid = guest.thread_state().dettid;
971            detlog!(
972                "[memory][dtid {}] {}->{}",
973                dettid,
974                procmaps::display(&mmap),
975                procmaps::compute_hash(guest, &mmap)?
976            )
977        }
978
979        // The kernel labels `[heap]` only for `[mm->start_brk, mm->brk)`. Under a
980        // backend that loads the guest itself (DynamoRIO) that break belongs to
981        // the loader, so the guest's heap is an unlabelled anonymous mapping and
982        // the filter above matches nothing. Emitting no record there is worse
983        // than a wrong one: downstream a zero-record heap comparison reads as
984        // "compared and matched" rather than "never measured". Fall back to the
985        // brk range Detcore observed, which identifies the heap on every backend.
986        if self.cfg.detlog_heap && !labelled_heap {
987            self.detlog_brk_heap(guest)?;
988        }
989        Ok(())
990    }
991
992    /// Emit the `[heap]` record from the observed program break, for backends
993    /// where the kernel does not label the guest's heap.
994    ///
995    /// Reports `[start_brk, brk)` -- the range Detcore observed -- taking the
996    /// non-address columns from the enclosing anonymous mapping, so the record
997    /// is textually comparable with the labelled record another backend
998    /// produces for the same guest.
999    ///
1000    /// ⚠️ THE EXTENT IS THE OBSERVED BREAK, NOT THE MAPPING THAT CONTAINS IT,
1001    /// and the two are not the same region. The selector below admits any
1002    /// anonymous mapping with `address.0 <= start && end <= address.1`, which
1003    /// is a SUPERSET by construction. Reporting that mapping instead would make
1004    /// the comparability claim above true only when the arena happens to
1005    /// coincide with the break, and would fold non-heap bytes into the digest.
1006    /// On the backend this path exists for, the premise is that the LOADER owns
1007    /// the break, so those extra bytes are loader arena -- the least
1008    /// reproducible region in the process. A silently empty record would become
1009    /// a loudly divergent one, for a reason that is not the guest's heap.
1010    ///
1011    /// The labelled path above hashes its mapping directly, which is correct
1012    /// there because the kernel defines `[heap]` as exactly `[start_brk, brk)`.
1013    /// Both paths therefore report the same quantity.
1014    fn detlog_brk_heap<G: Guest<Self>>(&self, guest: &mut G) -> Result<(), reverie::Error> {
1015        let Some((start, end)) = guest
1016            .thread_state()
1017            .memory_metadata
1018            .lock()
1019            .expect("memory metadata mutex poisoned")
1020            .brk_heap_range()
1021        else {
1022            return Ok(());
1023        };
1024        let enclosing = procmaps::from_pid(guest.pid(), |map| {
1025            matches!(map.pathname, procmaps::MMapPath::Anonymous)
1026                && map.address.0 <= start
1027                && end <= map.address.1
1028        })?;
1029        let Some(mmap) = enclosing.into_iter().next() else {
1030            return Ok(());
1031        };
1032        let dettid = guest.thread_state().dettid;
1033        detlog!(
1034            "[memory][dtid {}] {}->{}",
1035            dettid,
1036            procmaps::display_range_as(&mmap, start, end, "[heap]"),
1037            procmaps::compute_hash_range(guest, start, end)?
1038        );
1039        Ok(())
1040    }
1041}
1042
1043/// Render a finished syscall for the `finish syscall` DETLOG line.
1044///
1045/// Output pointers are dereferenced only when the kernel can have written
1046/// them. Linux leaves the output buffer of the syscalls listed in
1047/// [`failure_leaves_outputs_unwritten`] untouched when they fail with any errno
1048/// other than EFAULT, so rendering it would publish whatever the guest
1049/// happened to have there -- typically uninitialized stack -- as if it were a
1050/// result, and two otherwise identical runs would diverge on it
1051/// (<https://github.com/rrnewton/hermit/issues/3153>). For those syscalls such
1052/// a failure renders the pointer arguments without their pointees; the errno
1053/// is printed next to this rendering from the result itself.
1054///
1055/// EFAULT keeps the outputs rendered, because it is also the error of a copy
1056/// that faulted part way: `copy_to_user()` may already have stored a prefix of
1057/// the struct, which the guest can read and DETLOG must show. A tool error is
1058/// not a guest errno and proves nothing about the buffer, so it keeps them
1059/// rendered too.
1060///
1061/// Every other syscall keeps rendering its outputs on failure, because some
1062/// Linux syscalls do write an output on an error return (for example
1063/// `gettimeofday` stores `tv` before it faults on a bad `tz`). Hiding such an
1064/// output would remove real evidence from DETLOG.
1065fn display_syscall_finished<'a, M: MemoryAccess>(
1066    syscall: &'a Syscall,
1067    memory: &'a M,
1068    result: &Result<i64, Error>,
1069) -> reverie::syscalls::Display<'a, M, Syscall> {
1070    match syscall {
1071        Syscall::Fstat(_) => syscall.display(memory), //FIXME: T136880615 - fstat structure isn't fully deterministic yet
1072        _ if failure_proves_outputs_unwritten(result)
1073            && failure_leaves_outputs_unwritten(syscall) =>
1074        {
1075            syscall.display(memory)
1076        }
1077        _ => syscall.display_with_outputs(memory),
1078    }
1079}
1080
1081/// Syscalls whose output buffer Linux does not write when the syscall fails
1082/// with an errno other than EFAULT, restricted to those whose output the pinned Reverie formatter dereferences.
1083///
1084/// This is deliberately a list of syscalls PROVEN not to write on failure,
1085/// rather than a list of exceptions that do: a syscall missing from it keeps
1086/// its outputs rendered, which can at worst show stale guest memory, whereas a
1087/// syscall wrongly missing from an exception list would silently hide a value
1088/// the kernel really wrote.
1089///
1090/// Kernel behaviour (Linux `fs/stat.c`, `kernel/time/posix-timers.c`):
1091///
1092/// - `stat`, `lstat`, `fstat`, `newfstatat`: `vfs_stat()`, `vfs_lstat()`,
1093///   `vfs_fstat()` and `vfs_fstatat()` return their error before
1094///   `cp_new_stat()` is called, so the `struct stat` is never copied out.
1095///   Hermit's `handle_stat_family` likewise returns the injected syscall's
1096///   error before it rewrites the buffer.
1097/// - `statx`: `do_statx()` returns the `vfs_statx()` error before
1098///   `cp_statx()` copies the `struct statx` out.
1099/// - `clock_gettime`: `put_timespec64()` runs only when
1100///   `kc->clock_get_timespec()` succeeded (`if (!error && put_timespec64(..))`),
1101///   and an unknown clock returns `-EINVAL` before that. Hermit's
1102///   `handle_clock_gettime` writes `tp` only on its success path.
1103///
1104/// In every case the one failure that follows a copy attempt is the `-EFAULT`
1105/// from the copy itself, where `copy_to_user()` may have stored a prefix of
1106/// the struct before faulting. [`failure_proves_outputs_unwritten`] therefore
1107/// keeps rendering EFAULT failures, so that partial prefix stays visible.
1108///
1109/// `gettimeofday` is intentionally absent: `SYSCALL_DEFINE2(gettimeofday)` in
1110/// `kernel/time/time.c` stores `tv` and then returns `-EFAULT` if copying
1111/// `tz` faults, so its rendered output can be kernel-written on failure.
1112fn failure_leaves_outputs_unwritten(syscall: &Syscall) -> bool {
1113    matches!(
1114        syscall,
1115        Syscall::Stat(_)
1116            | Syscall::Lstat(_)
1117            | Syscall::Fstat(_)
1118            | Syscall::Newfstatat(_)
1119            | Syscall::Statx(_)
1120            | Syscall::ClockGettime(_)
1121    )
1122}
1123
1124/// Whether `result` is a failure that, for a syscall listed in
1125/// [`failure_leaves_outputs_unwritten`], proves Linux stored no output: a
1126/// guest errno other than EFAULT. EFAULT can follow a partial copy, and a
1127/// tool error is not the syscall's result.
1128fn failure_proves_outputs_unwritten(result: &Result<i64, Error>) -> bool {
1129    matches!(result, Err(Error::Errno(errno)) if *errno != Errno::EFAULT)
1130}
1131
1132#[reverie::tool]
1133impl<T: RecordOrReplay> Tool for Detcore<T> {
1134    type GlobalState = GlobalState;
1135    type ThreadState = ThreadState<T::ThreadState>;
1136
1137    fn observe_signal_dequeues(config: &Config) -> bool {
1138        config.kvm_shared_dequeue_timers && config.sequentialize_threads && config.backend_is_kvm
1139    }
1140
1141    async fn handle_signal_dequeue<G: Guest<Self>>(
1142        &self,
1143        guest: &mut G,
1144        dequeue: reverie::SignalDequeue,
1145    ) -> Result<(), Errno> {
1146        tool_global::signal_dequeued(guest, dequeue).await
1147    }
1148
1149    /// Constructor for Detcore process-local state.
1150    fn new(pid: Pid, cfg: &Config) -> Self {
1151        let detpid = DetPid::from_raw(pid.into()); // TODO(T78538674): virtualize pid.
1152        cfg.validate_invariants();
1153        Self {
1154            detpid,
1155            cfg: cfg.clone(),
1156            record_or_replay: T::new(pid, cfg),
1157        }
1158    }
1159
1160    /// NOTE: these subscriptions are used ONLY for hermit run mode.  Hermit record has its own
1161    /// subscriptions specified in recorder/mod.rs.
1162    fn subscriptions(config: &Config) -> Subscription {
1163        let do_sched =
1164            config.sched_heuristic != SchedHeuristic::None || config.sequentialize_threads;
1165
1166        if !config.passthru_opt {
1167            // Fail closed by default in every build profile. Besides allowing syscall-specific
1168            // handlers to run, interception is what charges generic syscall logical time.
1169            //
1170            // `cpuid` is the one exception. With CPUID virtualization off the handler only
1171            // re-executes the host instruction, so trapping does not change what the guest
1172            // sees, but the subscription makes Reverie probe and enable CPUID faulting and log
1173            // an ERROR for each exec on hosts that lack it
1174            // (https://github.com/rrnewton/hermit/issues/3460). A backend that installs its
1175            // own CPUID table (KVM) still needs the trap for the guest to see host values.
1176            let mut subscription = Subscription::all_syscalls();
1177            subscription.rdtsc();
1178            if config.virtualize_cpuid || config.cpuid_virtualized_by_backend {
1179                subscription.cpuid();
1180            }
1181            subscription
1182        } else {
1183            // Explicit performance opt-in: unlisted syscalls bypass Detcore entirely. Keep this
1184            // path separate so its allow-list can be tightened without weakening the default.
1185            let mut subscription = Subscription::none();
1186            subscription.syscalls([
1187                Sysno::write,
1188                // AUTONOMOUS-BOT-IMPLEMENTED
1189                // TODO-HUMAN-REVIEW(#547)
1190                Sysno::writev,
1191                // Timer-slack procfs virtualization must see every scalar and
1192                // vectored read/write form that Linux accepts for the file.
1193                Sysno::readv,
1194                Sysno::preadv,
1195                Sysno::preadv2,
1196                Sysno::pwritev,
1197                Sysno::pwritev2,
1198                // AUTONOMOUS-BOT-IMPLEMENTED
1199                // TODO-HUMAN-REVIEW(#683)
1200                Sysno::pwrite64,
1201                Sysno::openat,
1202                Sysno::open,
1203                Sysno::creat,
1204                Sysno::close,
1205                Sysno::read,
1206                Sysno::pread64,
1207                Sysno::lseek,
1208                Sysno::fadvise64,
1209                Sysno::mmap,
1210                Sysno::madvise,
1211                Sysno::munmap,
1212                Sysno::mremap,
1213                Sysno::fcntl,
1214                Sysno::arch_prctl,
1215                // AUTONOMOUS-BOT-IMPLEMENTED
1216                // TODO-HUMAN-REVIEW(PR-2150): Timer-slack prctl state is
1217                // virtual and must not bypass Detcore under passthru_opt.
1218                Sysno::prctl,
1219                Sysno::ioctl,
1220                Sysno::futex,
1221                Sysno::clone,
1222                Sysno::clone3,
1223                Sysno::fork,
1224                Sysno::vfork,
1225                Sysno::wait4,
1226                Sysno::waitid,
1227                Sysno::setsid,
1228                Sysno::uname,
1229                Sysno::exit_group,
1230                Sysno::exit,
1231                // AUTONOMOUS-BOT-IMPLEMENTED
1232                // The scheduler must observe changes to the address used for
1233                // its modeled CHILD_CLEARTID wake, including record/replay's
1234                // passthrough optimization.
1235                Sysno::set_tid_address,
1236                // AUTONOMOUS-BOT-IMPLEMENTED
1237                // Rare (once per thread) but load-bearing: without it the exit
1238                // hook cannot replay `exit_robust_list()` and robust-mutex
1239                // waiters are never woken.
1240                Sysno::set_robust_list,
1241                Sysno::dup,
1242                Sysno::dup2,
1243                Sysno::dup3,
1244                Sysno::pipe,
1245                Sysno::pipe2,
1246                Sysno::getrandom,
1247                Sysno::utime,
1248                Sysno::utimes,
1249                Sysno::utimensat,
1250                Sysno::futimesat,
1251                Sysno::socket,
1252                Sysno::socketpair,
1253                Sysno::eventfd,
1254                Sysno::eventfd2,
1255                Sysno::sched_getaffinity,
1256                Sysno::sched_setaffinity,
1257                Sysno::signalfd,
1258                Sysno::signalfd4,
1259                Sysno::timerfd_create,
1260                Sysno::timerfd_settime,
1261                Sysno::timerfd_gettime,
1262                Sysno::inotify_init,
1263                Sysno::inotify_init1,
1264                Sysno::inotify_add_watch,
1265                Sysno::inotify_rm_watch,
1266                Sysno::memfd_create,
1267                // AUTONOMOUS-BOT-IMPLEMENTED
1268                // TODO-HUMAN-REVIEW(PR-862): Keep modeled pidfd creation intercepted.
1269                Sysno::pidfd_open,
1270                // AUTONOMOUS-BOT-IMPLEMENTED
1271                // TODO-HUMAN-REVIEW(PR-1175): Keep pidfd signal/get-fd
1272                // determinization from being bypassed under the passthru opt-in.
1273                Sysno::pidfd_send_signal,
1274                Sysno::pidfd_getfd,
1275                Sysno::userfaultfd,
1276                Sysno::io_uring_setup,
1277                Sysno::io_uring_enter,
1278                Sysno::io_uring_register,
1279                Sysno::accept,
1280                Sysno::accept4,
1281                Sysno::nanosleep,
1282                Sysno::clock_nanosleep,
1283                Sysno::sched_yield,
1284                Sysno::poll,
1285                Sysno::ppoll,
1286                Sysno::prlimit64,
1287                Sysno::epoll_create,
1288                Sysno::epoll_create1,
1289                Sysno::epoll_ctl,
1290                Sysno::epoll_pwait,
1291                Sysno::epoll_wait,
1292                Sysno::epoll_wait_old,
1293                Sysno::epoll_ctl_old,
1294                Sysno::recvfrom,
1295                Sysno::rt_sigsuspend,
1296                Sysno::rt_sigtimedwait,
1297                Sysno::execve,
1298                Sysno::execveat,
1299                Sysno::rseq,
1300                Sysno::getpid,
1301                Sysno::gettid,
1302                Sysno::getcpu,
1303                Sysno::rt_sigprocmask,
1304                Sysno::rt_sigaction,
1305                Sysno::getrusage,
1306                Sysno::sysinfo,
1307                // AUTONOMOUS-BOT-IMPLEMENTED
1308                // TODO-HUMAN-REVIEW(#686): Review scratch fd sets and scheduler polling.
1309                Sysno::pselect6,
1310                // `select` is included by the Determinized sweep below;
1311                // T137258824 tracks improving its implementation.
1312            ]);
1313
1314            if do_sched {
1315                subscription.syscalls([
1316                    // TODO: some of the above could probably move to this bucket.
1317                    Sysno::alarm,
1318                    Sysno::pause,
1319                ]);
1320            }
1321
1322            if config.virtualize_metadata {
1323                subscription.syscalls([
1324                    Sysno::getdents,
1325                    Sysno::getdents64,
1326                    Sysno::stat,
1327                    Sysno::lstat,
1328                    Sysno::fstat,
1329                    Sysno::newfstatat,
1330                    Sysno::statx,
1331                ]);
1332            }
1333
1334            if true
1335            // TODO: could introduce a flag for this:
1336            /* config.virtualize_keys */
1337            {
1338                subscription.syscalls([Sysno::add_key, Sysno::request_key, Sysno::keyctl]);
1339            }
1340
1341            if do_sched {
1342                subscription.syscall(Sysno::connect);
1343            }
1344            if do_sched || config.warn_non_zero_binds {
1345                subscription.syscall(Sysno::bind);
1346            }
1347
1348            if config.warn_non_zero_binds {
1349                subscription.syscall(Sysno::bind);
1350            }
1351
1352            if config.virtualize_time {
1353                subscription.rdtsc();
1354                subscription.syscalls([
1355                    Sysno::gettimeofday,
1356                    Sysno::time,
1357                    Sysno::clock_gettime,
1358                    Sysno::clock_getres,
1359                ]);
1360            }
1361
1362            if config.virtualize_cpuid {
1363                subscription.cpuid();
1364            }
1365
1366            // AUTONOMOUS-BOT-IMPLEMENTED
1367            // TODO-HUMAN-REVIEW(PR-978): Keep the passthru-opt allow-list in sync
1368            // with the complete Determinized classification AND with the
1369            // Unsupported set. Even under the performance opt-in, Detcore must
1370            // see every syscall that its audited policy says it models or
1371            // deterministically refuses. Otherwise record/replay can bypass a
1372            // Detcore handler and execute the syscall natively against the host.
1373            //
1374            // `Determinized` and `Unsupported` are disjoint classifications, so
1375            // the determinized filter alone leaves the Unsupported set
1376            // unsubscribed: record/replay would silently execute an unsupported
1377            // syscall on the live host instead of invalidating its determinism
1378            // claim. Both halves are required; neither implies the other.
1379            // NOTE: `all_pinned_syscalls()`, not `Sysno::iter()`. The latter
1380            // stops one short of the end of the table, which silently dropped
1381            // `lsm_list_modules` out of this sweep.
1382            subscription.syscalls(crate::all_pinned_syscalls().filter(|sysno| {
1383                crate::is_determinized_syscall(*sysno) || crate::is_unsupported_syscall(*sysno)
1384            }));
1385
1386            // Make sure we also intercept everything that the record-or-replay tool
1387            // wants.
1388            subscription | T::subscriptions(config)
1389        }
1390    }
1391
1392    async fn handle_cpuid_event<G: Guest<Self>>(
1393        &self,
1394        guest: &mut G,
1395        eax: u32,
1396        ecx: u32,
1397    ) -> Result<CpuIdResult, Errno> {
1398        trace!("handle_cpuid_event: eax: {}, ecx: {}", eax, ecx);
1399        self.pre_handler_hook(guest, false).await;
1400        let res = if self.cfg.virtualize_cpuid {
1401            let dettid = guest.thread_state().dettid;
1402            let time = &mut guest.thread_state_mut().thread_logical_time;
1403            let intercepted = cpuid::InterceptedCpuid::new();
1404            time.add_cpuid();
1405            let nanos = time.as_nanos();
1406            trace!(
1407                "[dtid {}] inbound cpuid, new logical time: {:?}",
1408                dettid, time
1409            );
1410            if self.cfg.should_trace_schedevent() {
1411                trace_schedevent(
1412                    guest,
1413                    SchedEvent {
1414                        dettid,
1415                        op: Op::Cpuid,
1416                        count: 1,
1417                        start_rip: None,
1418                        end_rip: None,
1419                        end_time: Some(nanos),
1420                    },
1421                    true,
1422                )
1423                .await;
1424            }
1425            intercepted.cpuid(eax, ecx).unwrap_or_else(|| {
1426                warn!(
1427                    "[dtid {}] cpuid leaf 0x{:x} subleaf 0x{:x} not in deterministic table; returning zero result",
1428                    dettid, eax, ecx
1429                );
1430                CpuIdResult {
1431                    eax: 0,
1432                    ebx: 0,
1433                    ecx: 0,
1434                    edx: 0,
1435                }
1436            })
1437        } else {
1438            cpuid!(eax, ecx)
1439        };
1440        self.post_handler_hook(guest).await;
1441        Ok(res)
1442    }
1443
1444    async fn handle_rdtsc_event<G: Guest<Self>>(
1445        &self,
1446        guest: &mut G,
1447        request: Rdtsc,
1448    ) -> Result<RdtscResult, Errno> {
1449        trace!("handle_rdtsc_event: {:?}", request);
1450        self.pre_handler_hook(guest, false).await;
1451        let result = if guest.config().virtualize_time {
1452            let dettid = guest.thread_state().dettid;
1453            guest.thread_state_mut().thread_logical_time.add_rdtsc();
1454            info!(
1455                "[dtid {}] inbound rdtsc, new logical time: {:?}",
1456                dettid,
1457                guest.thread_state().thread_logical_time
1458            );
1459            if self.cfg.should_trace_schedevent() {
1460                let ev = with_guest_time(
1461                    guest,
1462                    SchedEvent {
1463                        dettid,
1464                        op: Op::Rdtsc,
1465                        count: 1,
1466                        start_rip: None,
1467                        end_rip: None,
1468                        end_time: None,
1469                    },
1470                );
1471                trace_schedevent(guest, ev, true).await;
1472            }
1473            // The guest TSC must name the same instant as `clock_gettime`. Both
1474            // now read the coordinator's clock through the shared per-process
1475            // floor, so a guest comparing the two -- a clocksource watchdog, a
1476            // delay loop calibrated against a device timer, a second vCPU
1477            // reading the TSC -- sees one time base instead of two.
1478            //
1479            // The `add_rdtsc()` charge above is folded into global time by the
1480            // same RPC that reads it back (`send_and_update_time` updates the
1481            // coordinator before dispatching the request), so consecutive reads
1482            // from one thread still advance.
1483            let tsc = guest_clock_time(guest).await;
1484            Ok(RdtscResult {
1485                // We treat virtual cycles as equivalent to virtual nanoseconds.
1486                tsc: tsc.as_nanos(),
1487                aux: None,
1488            })
1489        } else {
1490            self.record_or_replay
1491                .handle_rdtsc_event(&mut guest.into_guest(), request)
1492                .await
1493        };
1494        self.post_handler_hook(guest).await;
1495        result
1496    }
1497
1498    // Note: we will not see SIGSTKFLT used for timers.
1499    async fn handle_signal_event<G: Guest<Self>>(
1500        &self,
1501        guest: &mut G,
1502        signal: Signal,
1503    ) -> Result<Option<Signal>, Errno> {
1504        if signal == Signal::SIGINT && self.cfg.sigint_instakill {
1505            warn!("Fatal: Exiting hermit container immediately upon SIGINT");
1506            // ⚠️ NOT A REFUSAL. The operator interrupted the run; hermit examined
1507            // nothing and decided nothing. Reporting `128 + SIGINT` is what every
1508            // other tool reports for this, and it keeps 122 meaning one thing.
1509            unrecoverable_shutdown(guest, detcore_model::HERMIT_SIGINT_DEATH_EXIT).await
1510        } else {
1511            self.pre_handler_hook(guest, false).await;
1512
1513            let dettid = guest.thread_state().dettid;
1514            let mycount = guest.thread_state().stats.signal_count;
1515            info!(
1516                "[dtid {}] handling inbound signal (#{}) {}",
1517                dettid, mycount, signal
1518            );
1519            guest.thread_state_mut().stats.count_signal();
1520            let time = &guest.thread_state().thread_logical_time;
1521            let nanos = time.as_nanos();
1522
1523            if self.cfg.sequentialize_threads && self.cfg.should_trace_schedevent() {
1524                trace_schedevent(
1525                    guest,
1526                    SchedEvent {
1527                        dettid,
1528                        op: Op::SignalReceived(signal.into()),
1529                        count: 1,
1530                        start_rip: None,
1531                        end_rip: None,
1532                        end_time: Some(nanos),
1533                    },
1534                    true,
1535                )
1536                .await;
1537            }
1538
1539            let request = guest.thread_state_mut().mk_request(
1540                ResourceID::InboundSignal(SigWrapper::from(signal)),
1541                Permission::RW,
1542            );
1543            resource_request(guest, request).await;
1544            // A delivered signal may be caught, ignored, or terminate the
1545            // thread group. Reading the lists is harmless; the collected
1546            // wakes remain inert unless the ptrace exit callback later reports
1547            // that this exact signal caused physical exit.
1548            self.stage_thread_group_robust_list_wakes(
1549                guest,
1550                tool_local::RobustListExit::Signal(signal as i32),
1551            )
1552            .await;
1553            info!(
1554                "[dtid {}] finish delivering signal (#{}) {}",
1555                dettid, mycount, signal
1556            );
1557
1558            self.post_handler_hook(guest).await;
1559            Ok(Some(signal))
1560        }
1561    }
1562
1563    fn init_thread_state(
1564        &self,
1565        tid: Tid,
1566        parent: Option<(Tid, &Self::ThreadState)>,
1567    ) -> Self::ThreadState {
1568        trace!("[tid {}] detcore init new thread state", tid);
1569
1570        let record_or_replay = self
1571            .record_or_replay
1572            .init_thread_state(tid, parent.map(|(ptid, ts)| (ptid, ts.as_ref())));
1573
1574        // TODO(T78538674): virtualize tid, extend tid<=>dettid mapping here.
1575        match parent {
1576            None => ThreadState::new(DetPid::from_raw(tid.into()), &self.cfg, record_or_replay),
1577            Some(pts) => {
1578                let clone_flags = pts
1579                    .1
1580                    .clone_flags
1581                    .expect("clone_flags must be set by parent");
1582                let dettid = DetPid::from_raw(tid.into());
1583
1584                // If we had mutable access to the parent state, we could update it here, but
1585                // instead we leave that to the clone/fork handling.
1586                let (_next_parent_pedigree, child_pedigree) = pts.1.pedigree.fork();
1587                let child_logical_time = pts.1.thread_logical_time.clone_for_child();
1588                let last_accounted_user_time = child_logical_time.user_cpu_time();
1589                let last_accounted_system_time = child_logical_time.system_cpu_time();
1590                if !clone_flags.contains(CloneFlags::CLONE_THREAD) {
1591                    pts.1.prepare_child_process_cpu_time(dettid);
1592                }
1593                let guest_clock = Arc::clone(&pts.1.guest_clock);
1594
1595                ThreadState {
1596                    dettid,
1597                    detpid: None, // Initialized later.
1598                    thread_start_entered: false,
1599                    physical_tid: None,
1600                    signal_task_identity: None,
1601                    open_file_creator: None,
1602                    mm_id: MmId::for_clone(
1603                        pts.1.mm_id,
1604                        dettid,
1605                        clone_flags.contains(CloneFlags::CLONE_VM),
1606                    ),
1607                    memory_metadata: if clone_flags.contains(CloneFlags::CLONE_VM) {
1608                        Arc::clone(&pts.1.memory_metadata)
1609                    } else {
1610                        Arc::new(Mutex::new(
1611                            pts.1
1612                                .memory_metadata
1613                                .lock()
1614                                .expect("memory metadata mutex poisoned")
1615                                .clone(),
1616                        ))
1617                    },
1618                    pedigree: child_pedigree.clone(),
1619                    initialized_random_auxv: None,
1620                    stats: ThreadStats::new(),
1621                    file_metadata: {
1622                        debug!(
1623                            "[init_thread-state, parent dtid = {}] child thread {}, clone_flags = {:x?}",
1624                            pts.0, tid, clone_flags
1625                        );
1626                        if clone_flags.contains(CloneFlags::CLONE_FILES) {
1627                            pts.1.file_metadata.clone()
1628                        } else {
1629                            Arc::new(Mutex::new(
1630                                pts.1.file_metadata.lock().unwrap().fork_for(dettid),
1631                            ))
1632                        }
1633                    },
1634                    discover_live_file_metadata: pts.1.discover_live_file_metadata,
1635                    // Linux copies the creating thread's current timer slack
1636                    // into both fields of every new task (thread or process).
1637                    timer_slack_ns: pts.1.timer_slack_ns,
1638                    default_timer_slack_ns: pts.1.timer_slack_ns,
1639                    // POSIX timers are shared among threads of a process but are
1640                    // NOT inherited across fork(2). Share the table for a new
1641                    // thread (CLONE_THREAD); give a new process a fresh, empty
1642                    // one.
1643                    posix_timers: if clone_flags.contains(CloneFlags::CLONE_THREAD) {
1644                        Arc::clone(&pts.1.posix_timers)
1645                    } else {
1646                        Arc::new(Mutex::new(PosixTimers::default()))
1647                    },
1648                    // Resource limits are process state: threads share them,
1649                    // while a forked process inherits a snapshot.
1650                    resource_limits: if clone_flags.contains(CloneFlags::CLONE_THREAD) {
1651                        Arc::clone(&pts.1.resource_limits)
1652                    } else {
1653                        Arc::new(Mutex::new(
1654                            pts.1
1655                                .resource_limits
1656                                .lock()
1657                                .expect("resource limits mutex poisoned")
1658                                .clone(),
1659                        ))
1660                    },
1661                    process_cpu_time: if clone_flags.contains(CloneFlags::CLONE_THREAD) {
1662                        Arc::clone(&pts.1.process_cpu_time)
1663                    } else {
1664                        Arc::new(Mutex::new(ProcessCpuTime::default()))
1665                    },
1666                    // Wall time belongs to the traced process tree, not to an
1667                    // individual process. Forked processes and cloned threads
1668                    // therefore retain one monotonic view of raw logical time.
1669                    guest_clock,
1670                    parent_process_cpu_time: if clone_flags.contains(CloneFlags::CLONE_THREAD) {
1671                        pts.1.parent_process_cpu_time.clone()
1672                    } else {
1673                        Some(Arc::clone(&pts.1.process_cpu_time))
1674                    },
1675                    last_accounted_user_time,
1676                    last_accounted_system_time,
1677                    thread_cpu_start_user_time: last_accounted_user_time,
1678                    thread_cpu_start_system_time: last_accounted_system_time,
1679                    clone_flags: None,
1680                    pending_vfork: pts.1.pending_vfork.clone(),
1681
1682                    // Child RNG identity follows the deterministic creation
1683                    // pedigree, never the backend/host Tid. Guest-visible IDs
1684                    // and scheduler targeting continue to use `dettid`.
1685                    prng: tool_local::thread_rng_from_parent_pedigree(
1686                        "USER RAND",
1687                        &pts.1.prng,
1688                        &child_pedigree,
1689                        tool_local::ChildRngStream::User,
1690                    ),
1691                    chaos_prng: tool_local::thread_rng_from_parent_pedigree(
1692                        "CHAOSRAND",
1693                        &pts.1.chaos_prng,
1694                        &child_pedigree,
1695                        tool_local::ChildRngStream::Chaos,
1696                    ),
1697
1698                    // For comparing progress to other threads, it is important that our
1699                    // child thread start at a sensible place, rather than starting back
1700                    // at zero:
1701                    thread_logical_time: child_logical_time,
1702                    // A new thread gets a new clock, so we've committed 0 ticks
1703                    committed_clock_value: 0,
1704                    // A new thread or process is never the backend runtime's
1705                    // bootstrapping thread, so it starts outside any window.
1706                    uncharged_bootstrap_syscalls: 0,
1707                    in_uncharged_bootstrap_syscall: false,
1708
1709                    end_of_timeslice: None,
1710                    replay_rcb_end: None,
1711                    // AUTONOMOUS-BOT-IMPLEMENTED
1712                    // TODO-HUMAN-REVIEW(PR-1151)
1713                    chaos_epoch: tool_local::chaos_epoch_sentinel(),
1714                    chaos_slowdown_factor: RcbTimeMultiplier::ONE,
1715                    chaos_slowdown_active: false,
1716                    pending_chaos_epochs: Vec::new(),
1717                    max_timeslice_end: None,
1718                    last_rcb_timer: None,
1719                    last_rcb_timer_is_max: false,
1720
1721                    record_or_replay,
1722                    preemption_points: None,
1723
1724                    // We only get to the point of creating child threads if we're past the first execve.
1725                    past_global_first_execve: true,
1726                    interrupt_at: self.cfg.interrupts_for_thread(dettid),
1727
1728                    // `copy_process()` sets `p->robust_list = NULL` for every
1729                    // new task, thread or process alike. The child re-registers
1730                    // its own head before it can own a robust futex.
1731                    robust_list_head: None,
1732                    robust_list_process: if clone_flags.contains(CloneFlags::CLONE_THREAD) {
1733                        Arc::clone(&pts.1.robust_list_process)
1734                    } else {
1735                        Arc::new(Mutex::new(tool_local::RobustListProcessState::default()))
1736                    },
1737                }
1738            }
1739        }
1740    }
1741
1742    async fn handle_thread_start<G: Guest<Self>>(&self, guest: &mut G) -> Result<(), Error> {
1743        guest.thread_state_mut().thread_start_entered = true;
1744        let new_dettid = DetTid::from_raw(guest.tid().into()); // TODO(T78538674): virtualize pid/tid:
1745        assert_eq!(new_dettid, guest.thread_state().dettid);
1746        let detpid = select_thread_start_detpid(guest.thread_state().detpid, guest.pid());
1747        let is_root_thread = is_root_thread_start(guest.is_root_process(), new_dettid, detpid);
1748        trace!(
1749            "[tid {}] detcore handle_thread_start, pid={}",
1750            guest.tid(),
1751            detpid
1752        );
1753
1754        // Delayed initialization of thread_state for this new thread:
1755        let thread_state = guest.thread_state_mut();
1756        thread_state.detpid = Some(detpid);
1757        if thread_state.recover_process_mm_id(detpid) {
1758            debug!(
1759                "[detcore, dtid {}] recovered process memory identity {} for unparented thread state",
1760                new_dettid, detpid
1761            );
1762        }
1763
1764        if let Some(vfork) = guest.thread_state_mut().pending_vfork.take() {
1765            create_vfork_child_thread(guest, new_dettid, vfork).await;
1766        } else if is_root_thread {
1767            // There is no fork event to catch for the root thread.
1768            debug!(
1769                "[detcore, dtid {}] root thread start, scheduling.. full config:\n {:?}",
1770                &new_dettid,
1771                guest.config()
1772            );
1773            let physical_ids = if guest
1774                .config()
1775                .backend_requires_thread_directed_process_signals
1776            {
1777                Some((
1778                    guest.pid().as_raw(),
1779                    guest
1780                        .thread_state()
1781                        .physical_tid
1782                        .expect("backend requires a host thread ID before registration"),
1783                ))
1784            } else {
1785                None
1786            };
1787            if let Some(post_exec_mm) =
1788                create_child_thread(guest, new_dettid, 0, None, libc::SIGCHLD, physical_ids).await
1789            {
1790                guest.thread_state_mut().mm_id = post_exec_mm;
1791            }
1792        }
1793
1794        // Except for the root task, let's block until it's our turn to go:
1795        let th = tool_global::thread_start_request(&self.cfg, guest, detpid).await;
1796
1797        // Finish the delayed initialization of the full threadstate:
1798        {
1799            let ts = guest.thread_state_mut();
1800            ts.preemption_points = th.map(|x| x.into_iter());
1801            ts.next_timeslice(&self.cfg); // Must be after preemption_points is set.
1802        }
1803
1804        // The prehook is a noop for a thread just starting.  Can't end the timeslice.  There's no
1805        // RCB progress to record.  However, we call it for consistency with all the other handlers.
1806        self.pre_handler_hook(guest, true).await;
1807        // ^ precise_branch=true: There should have been ZERO prior instructions before this,
1808        // because the thread hasn't done anything yet.
1809
1810        self.record_or_replay
1811            .handle_thread_start(&mut guest.into_guest())
1812            .await?;
1813
1814        self.post_handler_hook(guest).await;
1815        Ok(())
1816    }
1817
1818    async fn handle_post_exec<G: Guest<Self>>(&self, guest: &mut G) -> Result<(), Errno> {
1819        // A nonleader exec preserves the survivor's state and PMU counter, but
1820        // Linux changes its TID. Bind that state before any image callback RPC.
1821        match tool_global::reconnect_exec(guest).await {
1822            Err(Errno::EOPNOTSUPP) => {
1823                let message = "unsupported: preemption recording and replay across nonleader exec";
1824                error!("{message}");
1825                // Report even when logging is disabled or redirected. The
1826                // ordinary refusal helper preserves the policy exit class;
1827                // reconnect already bound identity, so its diagnostic RPC is
1828                // authenticated and cannot silently retire an unbound owner.
1829                let _ = writeln!(crate::util::RetryingStderr, "{message}");
1830                tool_global::unrecoverable_shutdown(
1831                    guest,
1832                    detcore_model::HERMIT_POLICY_REFUSAL_EXIT,
1833                )
1834                .await;
1835            }
1836            result => result?,
1837        }
1838        guest.thread_state_mut().past_global_first_execve = true;
1839        // Only a successful exec reaches this callback. Delete the old image's
1840        // POSIX timer IDs while exec still owns its scheduler turn; the global
1841        // notification below cancels their deadlines before the pre-handler
1842        // can yield. alarm()/ITIMER_REAL and the continuous clock survive exec.
1843        guest
1844            .thread_state()
1845            .posix_timers
1846            .lock()
1847            .unwrap()
1848            .clear_for_exec();
1849        // A successful exec clears the kernel's clear_child_tid registration.
1850        // Mirror that reset before any replacement-image syscall can run.
1851        if guest.config().sequentialize_threads {
1852            tool_global::set_child_tid_address(guest, 0).await;
1853        }
1854
1855        tool_global::mark_past_first_execve(guest).await;
1856        self.pre_handler_hook(guest, false).await;
1857
1858        let auxv = guest.auxv();
1859        let initialized = guest
1860            .thread_state_mut()
1861            .complete_initial_random_auxv(auxv.at_random().map(|p| p.as_raw()))
1862            .expect("authenticated initial random handoff no longer matches this image");
1863        // The successful early write already emitted the ordinary auxv INFO
1864        // record. Consume only its fact here: no second draw, write or event.
1865        if !initialized && let Some(ptr) = auxv.at_random() {
1866            // It is safe to mutate this address since libc has not yet had a
1867            // chance to modify or copy the auxv table.
1868            let memory = guest.memory();
1869            let dettid = guest.thread_state().dettid;
1870            let ptr = unsafe { ptr.into_mut() };
1871            random::initialize_auxv(
1872                guest.thread_state_mut().thread_prng(),
1873                memory,
1874                ptr.cast(),
1875                dettid,
1876            )?;
1877        }
1878
1879        // Successful exec never returns through handle_syscall_event, so the
1880        // nested recorder/replayer needs this callback to commit or retire its
1881        // pending exec state before the replacement image issues another exec.
1882        self.record_or_replay
1883            .handle_post_exec(&mut guest.into_guest())
1884            .await?;
1885
1886        self.post_handler_hook(guest).await;
1887        Ok(())
1888    }
1889
1890    /// A timer fires to preempt the guest and give other threads a turn.
1891    async fn handle_timer_event<G: Guest<Self>>(&self, guest: &mut G) {
1892        info!(
1893            "[detcore, dtid {}] inbound timer preemption event",
1894            guest.thread_state().dettid
1895        );
1896        if guest.config().preemption_stacktrace {
1897            let mut file_writer: Box<dyn Write> =
1898                match &guest.config().preemption_stacktrace_log_file {
1899                    Some(path) => Box::new(
1900                        File::create(path).expect("Failed to open preemption stacktrace log file"),
1901                    ),
1902                    None => Box::new(std::io::stderr()),
1903                };
1904            let ts = guest.thread_state();
1905            writeln!(
1906                file_writer,
1907                "\n>>> Guest tid {} preempted at thread time {} with stack trace:",
1908                ts.dettid,
1909                ts.thread_logical_time.as_nanos(),
1910            )
1911            .unwrap();
1912            if let Some(backtrace) = guest.backtrace() {
1913                if let Ok(pbt) = backtrace.pretty() {
1914                    writeln!(file_writer, "{}", pbt).unwrap();
1915                } else {
1916                    writeln!(file_writer, "{}", backtrace).unwrap();
1917                }
1918            } else {
1919                warn!("Could not read backtrace!");
1920            }
1921        }
1922        // This may LOOK like a noop, but actually all of the logic for ending the timeslice is in
1923        // the prehook.  All the timer has to do is interrupt the guest and generate an extra call
1924        // to this prehook.
1925        self.pre_handler_hook(guest, true).await;
1926        if guest.config().no_rcb_time && guest.thread_state().last_rcb_timer_is_max {
1927            let max_timeslice_end = guest
1928                .thread_state()
1929                .max_timeslice_end
1930                .expect("PMU maximum timer requires a deadline");
1931            guest
1932                .thread_state_mut()
1933                .thread_logical_time
1934                .advance_to(max_timeslice_end);
1935            if self.cfg.should_trace_schedevent() {
1936                let dettid = guest.thread_state().dettid;
1937                let ev = with_guest_time(
1938                    guest,
1939                    SchedEvent {
1940                        dettid,
1941                        op: Op::OtherInstructions,
1942                        count: 1,
1943                        start_rip: None,
1944                        end_rip: None,
1945                        end_time: None,
1946                    },
1947                );
1948                let ev = with_guest_rip(guest, ev).await;
1949                trace_schedevent(guest, ev, false).await;
1950            }
1951            if self.cfg.replay_schedule_from.is_some() {
1952                let fallback_deadline = max_timeslice_end
1953                    + Duration::from_nanos(u64::from(
1954                        self.cfg
1955                            .max_timeslice
1956                            .expect("PMU maximum must be configured"),
1957                    ));
1958                let thread_state = guest.thread_state_mut();
1959                let replay_deadline = thread_state
1960                    .end_of_timeslice
1961                    .filter(|deadline| *deadline > max_timeslice_end)
1962                    .unwrap_or(fallback_deadline);
1963                thread_state.end_of_timeslice = Some(replay_deadline);
1964                thread_state.max_timeslice_end = Some(replay_deadline);
1965                thread_state.last_rcb_timer = None;
1966                thread_state.last_rcb_timer_is_max = false;
1967                thread_state.stats.reset_timeslice();
1968            } else {
1969                self.end_timeslice(guest).await;
1970            }
1971        }
1972        self.post_handler_hook(guest).await;
1973    }
1974
1975    async fn handle_syscall_event<G: Guest<Self>>(
1976        &self,
1977        guest: &mut G,
1978        call: Syscall,
1979    ) -> Result<i64, Error> {
1980        self.pre_handler_hook(guest, false).await;
1981
1982        let dettid = guest.thread_state().dettid;
1983
1984        if guest.thread_state().guest_past_first_execve() {
1985            detlog!(
1986                event = crate::detlog::DetLogEvent::Syscall;
1987                "[syscall][detcore, dtid {}] inbound syscall: {} = ?",
1988                dettid,
1989                call.display(&guest.memory())
1990            );
1991        }
1992
1993        // The hot syscall path only reads a few Copy flags from the config, so copy
1994        // just those out instead of cloning the entire Config on every intercepted
1995        // syscall (previously flagged inline as an unnecessary copy). guest.config()
1996        // borrows guest immutably; bind the flags in a tight scope so the borrow ends
1997        // before the later thread_state_mut()/&mut guest borrows below.
1998        let (sequentialize_threads, virtualize_time, panic_on_unsupported_syscalls) = {
1999            let config = guest.config();
2000            (
2001                config.sequentialize_threads,
2002                config.virtualize_time,
2003                config.panic_on_unsupported_syscalls,
2004            )
2005        };
2006
2007        if sequentialize_threads && self.cfg.should_trace_schedevent() {
2008            trace_schedevent(
2009                guest,
2010                with_guest_time(
2011                    guest,
2012                    SchedEvent::syscall(dettid, call.number(), SyscallPhase::Prehook),
2013                ),
2014                true,
2015            )
2016            .await;
2017        }
2018
2019        let syscall_cost_ns = syscall_time::cost_ns(call.number());
2020        // The backend-runtime bootstrap window. A backend-resident runtime (the
2021        // LiteInst preload constructor) issues hundreds to thousands of syscalls
2022        // on one thread (315 for a small C program, 6,970 for emacs; see
2023        // `syscall_time::MAX_UNCHARGED_BOOTSTRAP_SYSCALLS`) between its validated
2024        // begin trap and the trap that ends its bootstrap: the ready report, or
2025        // the report that preparation failed.
2026        // Reverie reports that window only for the bootstrapping thread; every
2027        // other thread and every forked process sees no window.
2028        //
2029        // What is withheld: the per-syscall cost below, for the bootstrapping
2030        // thread only, for each syscall it makes inside the window that does not
2031        // observe virtual time, up to
2032        // `syscall_time::MAX_UNCHARGED_BOOTSTRAP_SYSCALLS` per window. That
2033        // includes syscalls made by guest code the runtime calls into, such as
2034        // an interposed malloc or open64. Charging the runtime's own syscalls
2035        // would make guest-visible virtual time (uptime, CLOCK_MONOTONIC, CPU
2036        // time) depend on which backend ran the program. Every withheld syscall
2037        // is still counted and fully handled below, so Detcore keeps tracking
2038        // the fds, mappings and inodes it creates.
2039        //
2040        // What is always charged, inside the window too: syscalls that observe
2041        // virtual time (`syscall_time::observes_virtual_time`), so two clock
2042        // reads on the bootstrapping thread differ by at least one syscall cost
2043        // exactly as outside the window; and every syscall past the cap, so a
2044        // loop inside the window, or a runtime that never reports ready, again
2045        // advances the clock at each syscall.
2046        //
2047        // When it matters: with a syscall-driven clock (no PMU, or
2048        // --max-timeslice=disabled, which Hermit selects when perf counters are
2049        // unavailable, or --no-rcb-time) these charges are the thread's only
2050        // per-syscall progress, and with sequentialized threads the 500x no-RCB
2051        // multiplier makes each one large. With RCB time on a PMU host, retired
2052        // branches still advance the thread's clock inside the window and only
2053        // the syscall costs are withheld.
2054        //
2055        // The decision depends only on the window the backend reports (opened
2056        // and closed by trap instructions the runtime executes), the syscall
2057        // number and the count of earlier uncharged syscalls in the same
2058        // window: all functions of the guest's own execution. Record/replay and
2059        // both --verify runs therefore make the same decisions. The clock is
2060        // never reset: the first syscall after the window continues from the
2061        // value the window left. See https://github.com/rrnewton/hermit/issues/3338
2062        // and https://github.com/rrnewton/hermit/pull/3430#issuecomment-5928691696.
2063        //
2064        // The scheduler turns such a syscall needs are withheld the same way.
2065        // Handling an uncharged syscall can still commit scheduler turns (for
2066        // example the file resources of a read of /proc/self/maps), and each
2067        // committed turn normally advances global time by the per-turn
2068        // scheduler cost. While the syscall is being handled the thread is
2069        // marked `in_uncharged_bootstrap_syscall`, `tool_global::resource_request`
2070        // copies that mark into the request, and the scheduler does not
2071        // advance global time for a marked turn unless it is an IO-polling
2072        // retry, whose time enforces timeouts. Without this, a LiteInst run of
2073        // the clock-trajectory fixture reached its first clock read 11.5 ms of
2074        // virtual time later than with it, and each exec added 12.0 ms more
2075        // (measured locally), which was enough for its sysinfo uptime to
2076        // differ from the ptrace backend's on the hosted runner
2077        // (https://github.com/rrnewton/hermit/issues/3517). A charged syscall
2078        // (one that observes virtual time, or any past the cap) is not marked,
2079        // so its turns advance the clock as everywhere else.
2080        let in_backend_runtime_bootstrap = guest.is_backend_runtime_bootstrap();
2081        let new_count = {
2082            // which results from not being able to borrow guest twice.
2083            let thread_state = guest.thread_state_mut();
2084            thread_state.stats.count_syscall();
2085
2086            // Every intercepted syscall advances logical time, including configurations that do
2087            // not serialize threads. This keeps virtual clocks productive during syscall loops.
2088            // The one exception is the bootstrap window described above: on the bootstrapping
2089            // thread, up to MAX_UNCHARGED_BOOTSTRAP_SYSCALLS syscalls per window that do not
2090            // observe virtual time are left uncharged.
2091            let charged =
2092                thread_state.charge_syscall_time(in_backend_runtime_bootstrap, call.number());
2093            if charged {
2094                thread_state
2095                    .thread_logical_time
2096                    .add_syscall_with_cost(syscall_cost_ns);
2097            }
2098            thread_state.in_uncharged_bootstrap_syscall = !charged;
2099            // This only folds the thread's new user and system time into the process total.
2100            // An uncharged syscall added none, so it needs no guard.
2101            thread_state.account_process_cpu_time();
2102            thread_state.stats.syscall_count
2103        };
2104
2105        // Happens-before enforcement checkpoint. When the run carries a
2106        // happens-before program with syscall-count anchors, every intercepted
2107        // syscall checks in with the scheduler at its prehook, carrying this
2108        // thread's running syscall count. The scheduler fires any anchor at
2109        // `Position::SyscallCount(new_count)` on this thread and parks the thread
2110        // (out of the run queue) when that anchor is the AFTER endpoint of a Hard
2111        // edge whose BEFORE endpoint has not fired yet. This is the gate that
2112        // makes an authored partial order reproduce a known race deterministically
2113        // (see detcore-model `happens_before`). It requires sequentialized
2114        // threads (enforced by the CLI) so the scheduler owns ordering.
2115        if guest
2116            .config()
2117            .happens_before
2118            .as_ref()
2119            .is_some_and(|p| p.has_syscall_count_anchors())
2120        {
2121            let request = guest.thread_state().mk_request(
2122                ResourceID::HappensBeforeCheckpoint(new_count),
2123                Permission::R,
2124            );
2125            resource_request(guest, request).await;
2126        }
2127
2128        // Only an emulated RNG readv supplies authoritative imported geometry.
2129        // A generic pre-dispatch snapshot would become stale across pipe waits.
2130        let mut rng_readv_output = None;
2131        let res = match classify_syscall(call.number()) {
2132            // Rseq is not type-safe in the pinned Reverie revision. Dispatch by Sysno so a
2133            // future typed representation preserves this explicit policy.
2134            SyscallClassification::Determinized if call.number() == Sysno::rseq => {
2135                if panic_on_unsupported_syscalls {
2136                    Err(Error::Errno(Errno::ENOSYS))
2137                } else {
2138                    self.passthrough(guest, call).await
2139                }
2140            }
2141            // AUTONOMOUS-BOT-IMPLEMENTED
2142            // TODO-HUMAN-REVIEW(#663)
2143            // The pinned Reverie revision exposes process_madvise only as a raw call.
2144            SyscallClassification::Determinized if call.number() == Sysno::process_madvise => {
2145                match call {
2146                    Syscall::Other(_, args) => Self::handle_process_madvise(args.arg0, args.arg4),
2147                    _ => unreachable!("process_madvise unexpectedly gained a typed variant"),
2148                }
2149            }
2150            // AUTONOMOUS-BOT-IMPLEMENTED
2151            // TODO-HUMAN-REVIEW(PR-1175): The pinned Reverie revision exposes
2152            // pidfd_send_signal/pidfd_getfd only as raw calls, so dispatch on the
2153            // Sysno. See the handlers in syscalls/files.rs for the determinism
2154            // argument.
2155            SyscallClassification::Determinized if call.number() == Sysno::pidfd_send_signal => {
2156                match call {
2157                    Syscall::Other(_, args) => {
2158                        self.handle_pidfd_send_signal(
2159                            guest,
2160                            call,
2161                            args.arg0 as RawFd,
2162                            args.arg3 as u32,
2163                        )
2164                        .await
2165                    }
2166                    _ => unreachable!("pidfd_send_signal unexpectedly gained a typed variant"),
2167                }
2168            }
2169            // AUTONOMOUS-BOT-IMPLEMENTED
2170            // TODO-HUMAN-REVIEW(PR-1175): pidfd_getfd, likewise untyped.
2171            SyscallClassification::Determinized if call.number() == Sysno::pidfd_getfd => {
2172                match call {
2173                    Syscall::Other(_, args) => {
2174                        self.handle_pidfd_getfd(
2175                            guest,
2176                            call,
2177                            args.arg0 as RawFd,
2178                            args.arg1 as RawFd,
2179                            args.arg2 as u32,
2180                        )
2181                        .await
2182                    }
2183                    _ => unreachable!("pidfd_getfd unexpectedly gained a typed variant"),
2184                }
2185            }
2186            // AUTONOMOUS-BOT-IMPLEMENTED
2187            // TODO-HUMAN-REVIEW(#715): Deterministic ENOSYS for syscalls the pinned
2188            // x86_64 kernel leaves unimplemented (sys_ni_syscall). A fixed -ENOSYS is
2189            // deterministic by construction and identical to the modern kernel's own
2190            // return, so no guest-visible behavior changes versus the legacy
2191            // pass-through. These are untyped (Syscall::Other) in the pinned Reverie,
2192            // so dispatch on the Sysno before the typed match below.
2193            SyscallClassification::Determinized
2194                if is_unimplemented_enosys_syscall(call.number()) =>
2195            {
2196                Err(Error::Errno(Errno::ENOSYS))
2197            }
2198            // AUTONOMOUS-BOT-IMPLEMENTED
2199            // TODO-HUMAN-REVIEW(PR-852): Review the futex2 fallback contract.
2200            // Detcore models legacy futex but not the newer vector/sized futex2
2201            // ABI. Match a kernel without futex2 so runtimes take their
2202            // established legacy-futex fallback without consulting the host.
2203            SyscallClassification::Determinized if is_futex2_enosys_syscall(call.number()) => {
2204                Err(Error::Errno(Errno::ENOSYS))
2205            }
2206            // AUTONOMOUS-BOT-IMPLEMENTED
2207            // TODO-HUMAN-REVIEW(PR-836): Host filesystem and mount
2208            // introspection are outside the deterministic model. Return the
2209            // portable feature-absence errno so callers use /proc fallbacks.
2210            // AUTONOMOUS-BOT-IMPLEMENTED
2211            // TODO-HUMAN-REVIEW(PR-859): Extend this boundary to obsolete ustat
2212            // host-filesystem capacity counters.
2213            SyscallClassification::Determinized
2214                if is_mount_introspection_enosys_syscall(call.number()) =>
2215            {
2216                Err(Error::Errno(Errno::ENOSYS))
2217            }
2218            // AUTONOMOUS-BOT-IMPLEMENTED
2219            // TODO-HUMAN-REVIEW(PR-848): Hide unmodeled shared keyrings and
2220            // request-key upcalls behind the portable CONFIG_KEYS-absent errno.
2221            // TODO-HUMAN-REVIEW(PR-916): Fail closed whenever the
2222            // panic-on-unsupported policy is active; ordinary runs select that
2223            // policy by default. The explicit compatibility opt-out keeps the
2224            // pre-848 host pass-through so the guest observes a real working
2225            // keyring, restoring the enabled rr `keyctl` compatibility test.
2226            // Under fail-closed execution the deterministic ENOSYS boundary is
2227            // preserved.
2228            SyscallClassification::Determinized if is_kernel_keyring_syscall(call.number()) => {
2229                if panic_on_unsupported_syscalls {
2230                    Err(Error::Errno(Errno::ENOSYS))
2231                } else {
2232                    self.passthrough(guest, call).await
2233                }
2234            }
2235            // AUTONOMOUS-BOT-IMPLEMENTED
2236            // TODO-HUMAN-REVIEW(PR-855): Fail-closed runs cannot expose
2237            // unmodeled pipe-buffer ownership or vmsplice page pinning. Return
2238            // ENOSYS so callers use read/write fallbacks, but preserve host
2239            // pass-through under the explicit compatibility opt-out used by the
2240            // existing rr splice test.
2241            SyscallClassification::Determinized if is_zero_copy_pipe_syscall(call.number()) => {
2242                if panic_on_unsupported_syscalls {
2243                    Err(Error::Errno(Errno::ENOSYS))
2244                } else {
2245                    self.passthrough(guest, call).await
2246                }
2247            }
2248            // AUTONOMOUS-BOT-IMPLEMENTED
2249            // TODO-HUMAN-REVIEW(PR-860): Host LSM attributes are outside
2250            // Detcore's model. Present a stable feature-absence boundary
2251            // instead of forwarding probes.
2252            SyscallClassification::Determinized
2253                if is_host_security_identity_probe_syscall(call.number()) =>
2254            {
2255                Err(Error::Errno(Errno::ENOSYS))
2256            }
2257            // AUTONOMOUS-BOT-IMPLEMENTED
2258            // TODO-HUMAN-REVIEW(#722): Deterministic EPERM for privileged
2259            // system-administration syscalls (module load/unload, kexec, reboot,
2260            // swap, raw I/O ports, root-mount pivot, host/domain name, tty
2261            // hangup, disk quotas). The deterministic guest does not hold the
2262            // capabilities these require against the host kernel, so a fixed
2263            // -EPERM matches the unprivileged errno, never perturbs global host
2264            // state, and is identical across --verify and record/replay. These
2265            // are untyped (Syscall::Other) in the pinned Reverie, so dispatch on
2266            // the Sysno before the typed match below.
2267            SyscallClassification::Determinized
2268                if is_privileged_admin_refused_syscall(call.number()) =>
2269            {
2270                Err(Error::Errno(Errno::EPERM))
2271            }
2272            // AUTONOMOUS-BOT-IMPLEMENTED
2273            // TODO-HUMAN-REVIEW(PR-844): Enforce a deterministic boundary
2274            // around host-global process accounting and cross-process memory.
2275            SyscallClassification::Determinized
2276                if is_process_isolation_refused_syscall(call.number()) =>
2277            {
2278                Err(Error::Errno(Errno::EPERM))
2279            }
2280            // AUTONOMOUS-BOT-IMPLEMENTED
2281            // TODO-HUMAN-REVIEW(PR-876): Guest performance events
2282            // expose host PMU availability, policy, and asynchronous counter
2283            // state that Detcore does not model. A fixed ENOSYS preserves the
2284            // portable feature-probe fallback without creating an untracked fd.
2285            SyscallClassification::Determinized if is_perf_event_enosys_syscall(call.number()) => {
2286                Err(Error::Errno(Errno::ENOSYS))
2287            }
2288            // AUTONOMOUS-BOT-IMPLEMENTED
2289            // TODO-HUMAN-REVIEW(PR-853): Refuse nested tracing, host-object
2290            // comparison at the deterministic boundary.
2291            SyscallClassification::Determinized
2292                if is_privileged_observation_refused_syscall(call.number()) =>
2293            {
2294                Err(Error::Errno(Errno::EPERM))
2295            }
2296            // AUTONOMOUS-BOT-IMPLEMENTED
2297            // TODO-HUMAN-REVIEW(#720): set_mempolicy_home_node is untyped in the
2298            // pinned Reverie revision. Hermit exposes a single virtual NUMA node,
2299            // so setting a memory range's home node has no observable effect: a
2300            // deterministic no-op.
2301            SyscallClassification::Determinized
2302                if call.number() == Sysno::set_mempolicy_home_node =>
2303            {
2304                Ok(0)
2305            }
2306            // AUTONOMOUS-BOT-IMPLEMENTED
2307            // TODO-HUMAN-REVIEW(#724): Deterministic EPERM for privileged mount
2308            // and namespace administration syscalls (mount/umount2/mount_setattr/
2309            // move_mount/open_tree/fsopen/fsmount/fsconfig/fspick, unshare, setns,
2310            // open_by_handle_at, fanotify_init/fanotify_mark, settimeofday). A
2311            // deterministic container pins the guest's namespaces, mount
2312            // hierarchy, and virtual clock for the whole run, so these are
2313            // refused with a fixed -EPERM: the unprivileged errno for the
2314            // capability-gated operations and a deliberate deterministic refusal
2315            // otherwise. Never forwarded to the host; identical across --verify
2316            // and record/replay. Untyped (Syscall::Other) in the pinned Reverie,
2317            // so dispatch on the Sysno before the typed match below.
2318            SyscallClassification::Determinized
2319                if is_mount_ns_admin_refused_syscall(call.number()) =>
2320            {
2321                Err(Error::Errno(Errno::EPERM))
2322            }
2323            // AUTONOMOUS-BOT-IMPLEMENTED
2324            // TODO-HUMAN-REVIEW(#731): Deterministic ENOSYS for the
2325            // asynchronous and message-passing I/O and IPC interfaces Detcore
2326            // does not model: Linux native AIO (io_setup/io_destroy/io_submit/
2327            // io_cancel/io_getevents/io_pgetevents), POSIX message queues
2328            // (mq_*), and System V message queues (msg*). AIO completion is
2329            // kernel-driven and lives outside logical time; the message-queue
2330            // families operate on global, key/name-addressed kernel objects
2331            // shared with the whole host. A fixed -ENOSYS is the errno a kernel
2332            // built without AIO/CONFIG_POSIX_MQUEUE/CONFIG_SYSVIPC returns, is
2333            // never forwarded to the host, and is identical across --verify and
2334            // record/replay (mirrors the io_uring refusal). Untyped
2335            // (Syscall::Other) in the pinned Reverie, so dispatch on the Sysno
2336            // before the typed match below.
2337            // AUTONOMOUS-BOT-IMPLEMENTED
2338            // TODO-HUMAN-REVIEW(PR-859): Include System V semaphore and shared-
2339            // memory objects in the existing CONFIG_SYSVIPC refusal boundary.
2340            SyscallClassification::Determinized
2341                if is_unsupported_async_ipc_syscall(call.number()) =>
2342            {
2343                Err(Error::Errno(Errno::ENOSYS))
2344            }
2345            // AUTONOMOUS-BOT-IMPLEMENTED
2346            // TODO-HUMAN-REVIEW(PR-882): Legacy nonlinear page
2347            // remapping has host-dependent kernel support and VMA behavior that
2348            // Detcore does not model. Preserve the documented mmap fallback.
2349            SyscallClassification::Determinized
2350                if is_remap_file_pages_enosys_syscall(call.number()) =>
2351            {
2352                Err(Error::Errno(Errno::ENOSYS))
2353            }
2354            // AUTONOMOUS-BOT-IMPLEMENTED
2355            // TODO-HUMAN-REVIEW(#787): BATCH 38. openat2 is untyped (Syscall::Other)
2356            // in the pinned Reverie revision. It is a superset of openat whose
2357            // callers must fall back to openat when it returns ENOSYS (kernels
2358            // before 5.6 lack openat2), so a fixed -ENOSYS routes them onto the
2359            // already-determinized openat path with no host dependency and behavior
2360            // identical across --verify and record/replay.
2361            SyscallClassification::Determinized if call.number() == Sysno::openat2 => {
2362                Err(Error::Errno(Errno::ENOSYS))
2363            }
2364            // AUTONOMOUS-BOT-IMPLEMENTED
2365            // TODO-HUMAN-REVIEW(#787): BATCH 38. The credential-setting family
2366            // (setuid/setgid and their re-/res-/fs- variants, and setgroups) is
2367            // untyped (Syscall::Other) in the pinned Reverie. Detcore presents a
2368            // fixed virtual-root identity (getuid/geteuid/getgid/getegid are
2369            // virtualized to 0) and never tracks a credential change, so these
2370            // succeed as deterministic no-ops returning 0 -- the value a real root
2371            // process gets for a permitted credential change (and the previous
2372            // fs-id, virtual 0, for setfsuid/setfsgid). That lets privilege-
2373            // dropping programs proceed instead of fail-closing and is identical
2374            // across --verify and record/replay.
2375            SyscallClassification::Determinized
2376                if is_credential_identity_noop_syscall(call.number()) =>
2377            {
2378                Ok(0)
2379            }
2380            // AUTONOMOUS-BOT-IMPLEMENTED
2381            // TODO-HUMAN-REVIEW(#1851): The file-ownership mutation family
2382            // (chown/fchown/fchownat/lchown) completes the fixed virtual-root
2383            // identity that the credential query (#1549) and credential set
2384            // (#787) families already implement. A real root process's chown
2385            // succeeds for any uid, so 0 is the value the virtual identity must
2386            // observe; forwarding instead returned the errno of whatever host
2387            // identity the backend happened to run under (EPERM with no user
2388            // namespace, EINVAL for an unmapped uid inside a one-uid map, and
2389            // backend-dependent for in-process backends).
2390            //
2391            // The emulation covers the IDENTITY half only. Root privilege
2392            // waives the ownership permission check; it does not waive pathname,
2393            // descriptor, or flag errors, so handle_ownership_change_noop
2394            // translates the target arguments into a side-effect-free metadata
2395            // lookup and returns 0 only if that validation succeeds. ENOENT,
2396            // EBADF, EFAULT, ENOTDIR and the fchownat flag EINVAL therefore
2397            // still reach the guest; the host-identity-dependent EPERM/EINVAL
2398            // cannot be produced at all. No setattr is attempted, so host
2399            // ownership, mode bits, and timestamps are never modified, and
2400            // Detcore does not model per-file ownership, so the success is not
2401            // observable through a later stat -- see
2402            // is_ownership_change_noop_syscall and handle_ownership_change_noop
2403            // for the full boundary, and
2404            // hermit-cli/tests/chown_virtual_root_identity.rs for the bracket
2405            // that fails if this arm's RESULT regresses.
2406            SyscallClassification::Determinized
2407                if is_ownership_change_noop_syscall(call.number()) =>
2408            {
2409                self.handle_ownership_change_noop(guest, call).await
2410            }
2411            // AUTONOMOUS-BOT-IMPLEMENTED
2412            // TODO-HUMAN-REVIEW(#827): Deterministic ENOSYS for the Landlock
2413            // unprivileged-sandbox syscalls (landlock_create_ruleset,
2414            // landlock_add_rule, landlock_restrict_self). Landlock availability
2415            // and ABI version depend on the host kernel build
2416            // (CONFIG_SECURITY_LANDLOCK) and runtime LSM stacking, so forwarding
2417            // them (the legacy pass-through) is host-dependent and, because a
2418            // ruleset restricts the whole thread tree, a global-state isolation
2419            // hole. A fixed -ENOSYS is the errno a kernel built without Landlock
2420            // returns, so the guest sees a consistent "sandbox unavailable"
2421            // answer regardless of host; never forwarded to the host and
2422            // bitwise-identical across --verify and record/replay. Untyped
2423            // (Syscall::Other) in the pinned Reverie, so dispatch on the Sysno
2424            // before the typed match below.
2425            SyscallClassification::Determinized if is_landlock_sandbox_syscall(call.number()) => {
2426                Err(Error::Errno(Errno::ENOSYS))
2427            }
2428            // AUTONOMOUS-BOT-IMPLEMENTED
2429            // TODO-HUMAN-REVIEW(PR-847): Refuse unmodeled host-kernel probes
2430            // with a fixed ENOSYS so guest behavior does not depend on BPF/LSM
2431            // configuration or mutable page-cache state.
2432            SyscallClassification::Determinized if is_host_kernel_probe_syscall(call.number()) => {
2433                Err(Error::Errno(Errno::ENOSYS))
2434            }
2435            // AUTONOMOUS-BOT-IMPLEMENTED
2436            // TODO-HUMAN-REVIEW(PR-838): Review close_range descriptor-table
2437            // synchronization. The pinned Reverie exposes close_range as a raw
2438            // call, so dispatch by Sysno before the typed match.
2439            SyscallClassification::Determinized if call.number() == Sysno::close_range => {
2440                self.handle_close_range(guest, call).await
2441            }
2442            // AUTONOMOUS-BOT-IMPLEMENTED
2443            // TODO-HUMAN-REVIEW(PR-839): Optional modern memory APIs vary with
2444            // host kernel configuration, CET support, and pidfd lifecycle.
2445            // Present the portable feature-absence result instead.
2446            SyscallClassification::Determinized
2447                if is_optional_memory_feature_syscall(call.number()) =>
2448            {
2449                Err(Error::Errno(Errno::ENOSYS))
2450            }
2451            // AUTONOMOUS-BOT-IMPLEMENTED
2452            // TODO-HUMAN-REVIEW(#773): epoll_pwait2 is untyped (Syscall::Other)
2453            // in the pinned Reverie revision. It is epoll_pwait with a
2454            // `struct timespec *` timeout; recent glibc routes epoll_wait/
2455            // epoll_pwait through it. Handled identically to epoll_pwait
2456            // (scheduler yield + record/replay forwarding).
2457            SyscallClassification::Determinized if call.number() == Sysno::epoll_pwait2 => {
2458                self.handle_epoll_pwait2(guest, call).await
2459            }
2460            SyscallClassification::Determinized => match call {
2461                Syscall::Write(w) => self.handle_write(guest, w).await,
2462                // AUTONOMOUS-BOT-IMPLEMENTED
2463                // TODO-HUMAN-REVIEW(#547)
2464                Syscall::Writev(w) => self.handle_writev(guest, w).await,
2465                Syscall::Openat(o) => self.handle_openat(guest, o).await,
2466                Syscall::Open(o) => self.handle_openat(guest, o.into()).await,
2467                Syscall::Creat(o) => self.handle_openat(guest, o.into()).await,
2468                Syscall::Close(s) => self.handle_close(guest, s).await,
2469                Syscall::Read(s) if self.sock_diag_reply_fd(guest, s.fd()) => {
2470                    self.handle_sock_diag_read(guest, s).await
2471                }
2472                Syscall::Read(s) => self.handle_read(guest, s).await,
2473                Syscall::Pread64(s) => self.handle_pread64(guest, s).await,
2474                Syscall::Lseek(s) => self.handle_lseek(guest, s).await,
2475                // AUTONOMOUS-BOT-IMPLEMENTED
2476                // TODO-HUMAN-REVIEW(PR-838): Review regular-file sendfile mediation.
2477                Syscall::Sendfile(s) => self.handle_sendfile(guest, s).await,
2478                // AUTONOMOUS-BOT-IMPLEMENTED
2479                // TODO-HUMAN-REVIEW(PR-887): Present a stable pre-4.5-kernel
2480                // boundary so callers use determinized read/write copying.
2481                Syscall::CopyFileRange(_) => Err(Error::Errno(Errno::ENOSYS)),
2482                // TODO-HUMAN-REVIEW(#794): vectored scatter/gather I/O, mirroring
2483                // read/pread64/pwrite64/writev.
2484                Syscall::Readv(s) if self.sock_diag_reply_fd(guest, s.fd()) => {
2485                    self.handle_sock_diag_readv(guest, s).await
2486                }
2487                Syscall::Readv(s) => {
2488                    self.handle_readv_with_output(guest, s, &mut rng_readv_output)
2489                        .await
2490                }
2491                Syscall::Preadv(s) => {
2492                    self.handle_preadv_with_output(guest, s, &mut rng_readv_output)
2493                        .await
2494                }
2495                Syscall::Preadv2(s) => {
2496                    self.handle_preadv2_with_output(guest, s, &mut rng_readv_output)
2497                        .await
2498                }
2499                Syscall::Pwritev(s) => self.handle_pwritev(guest, s).await,
2500                Syscall::Pwritev2(s) => self.handle_pwritev2(guest, s).await,
2501                // AUTONOMOUS-BOT-IMPLEMENTED
2502                // TODO-HUMAN-REVIEW(#683)
2503                Syscall::Pwrite64(s) => self.handle_pwrite64(guest, s).await,
2504                // This syscall is advisory; fixed success preserves its API contract.
2505                Syscall::Fadvise64(_) => Ok(0),
2506                Syscall::Mmap(s) => self.handle_mmap(guest, s).await,
2507                Syscall::Madvise(s) => self.handle_madvise(guest, s).await,
2508                // AUTONOMOUS-BOT-IMPLEMENTED
2509                // TODO-HUMAN-REVIEW(#775)
2510                Syscall::Mincore(s) => self.handle_mincore(guest, s).await,
2511                Syscall::Munmap(s) => self.handle_munmap(guest, s).await,
2512                Syscall::Mremap(s) => self.handle_mremap(guest, s).await,
2513                Syscall::Stat(s) => self.handle_stat_family(guest, s.into()).await,
2514                Syscall::Lstat(s) => self.handle_stat_family(guest, s.into()).await,
2515                Syscall::Fstat(s) => self.handle_stat_family(guest, s.into()).await,
2516                Syscall::Newfstatat(s) => self.handle_stat_family(guest, s.into()).await,
2517                Syscall::Statx(s) => self.handle_statx(guest, s).await,
2518                // AUTONOMOUS-BOT-IMPLEMENTED
2519                // TODO-HUMAN-REVIEW(#877)
2520                Syscall::Readlink(s) => self.handle_readlink(guest, s).await,
2521                // AUTONOMOUS-BOT-IMPLEMENTED
2522                Syscall::Readlinkat(s) => self.handle_readlinkat(guest, s).await,
2523                Syscall::Fcntl(s) => self.handle_fcntl(guest, s).await,
2524                // AUTONOMOUS-BOT-IMPLEMENTED
2525                // TODO-HUMAN-REVIEW(PR-912)
2526                Syscall::Ioctl(s)
2527                    if syscalls::socket_timestamp_ioctl::is_socket_timestamp_ioctl(s) =>
2528                {
2529                    self.handle_socket_timestamp_ioctl(guest, s).await
2530                }
2531                Syscall::Ioctl(s) => self.handle_ioctl(guest, s).await,
2532                Syscall::Futex(s) => self.handle_futex(guest, s).await,
2533
2534                Syscall::Clone(s) => self.handle_clone_family(guest, s.into()).await,
2535                Syscall::Clone3(s) => self.handle_clone_family(guest, s.into()).await,
2536                Syscall::Fork(s) => self.handle_clone_family(guest, s.into()).await,
2537
2538                // Forward vfork as vfork (rather than rewriting to fork) so the
2539                // kernel enforces the CLONE_VFORK parent-blocking contract while the
2540                // child registers itself and runs to exec/exit.
2541                Syscall::Vfork(s) => self.handle_clone_family(guest, s.into()).await,
2542                Syscall::Wait4(s) => self.handle_wait4(guest, s).await,
2543                Syscall::Waitid(s) => self.handle_waitid(guest, s).await,
2544
2545                Syscall::Setpgid(s) => self.handle_setpgid(guest, s).await,
2546                Syscall::Setsid(s) => self.handle_setsid(guest, s).await,
2547                // Without virtual time the guest reads the host clock. Each of these
2548                // must have a record/replay handler: a recording captures the value
2549                // the guest observed, and replay returns it without reading the host.
2550                Syscall::Gettimeofday(s) => {
2551                    if virtualize_time {
2552                        self.handle_gettimeofday(guest, s).await
2553                    } else {
2554                        self.passthrough(guest, call).await
2555                    }
2556                }
2557                Syscall::Time(s) => {
2558                    if virtualize_time {
2559                        self.handle_time(guest, s).await
2560                    } else {
2561                        self.passthrough(guest, call).await
2562                    }
2563                }
2564                Syscall::ClockGettime(s) => {
2565                    if virtualize_time {
2566                        self.handle_clock_gettime(guest, s).await
2567                    } else {
2568                        self.passthrough(guest, call).await
2569                    }
2570                }
2571                Syscall::ClockGetres(s) => {
2572                    if virtualize_time {
2573                        self.handle_clock_getres(guest, s).await
2574                    } else {
2575                        self.passthrough(guest, call).await
2576                    }
2577                }
2578                // AUTONOMOUS-BOT-IMPLEMENTED
2579                // TODO-HUMAN-REVIEW(#663)
2580                Syscall::ClockSettime(_) => Err(Error::Errno(Errno::EPERM)),
2581                // AUTONOMOUS-BOT-IMPLEMENTED
2582                // TODO-HUMAN-REVIEW(PR-892)
2583                Syscall::Getitimer(s) => self.handle_getitimer(guest, s).await,
2584                // AUTONOMOUS-BOT-IMPLEMENTED
2585                // TODO-HUMAN-REVIEW(#663)
2586                Syscall::Setitimer(s) => self.handle_setitimer(guest, s).await,
2587                // AUTONOMOUS-BOT-IMPLEMENTED
2588                // TODO-HUMAN-REVIEW(PR-857): Virtual NTP query and fixed mutation refusal.
2589                Syscall::Adjtimex(s) => {
2590                    if virtualize_time {
2591                        self.handle_adjtimex(guest, s).await
2592                    } else {
2593                        self.handle_unsupported_syscall(
2594                            guest,
2595                            call,
2596                            dettid,
2597                            panic_on_unsupported_syscalls,
2598                        )
2599                        .await
2600                    }
2601                }
2602                // AUTONOMOUS-BOT-IMPLEMENTED
2603                // TODO-HUMAN-REVIEW(PR-857): Clock-id form of virtual NTP query.
2604                Syscall::ClockAdjtime(s) => {
2605                    if virtualize_time {
2606                        self.handle_clock_adjtime(guest, s).await
2607                    } else {
2608                        self.handle_unsupported_syscall(
2609                            guest,
2610                            call,
2611                            dettid,
2612                            panic_on_unsupported_syscalls,
2613                        )
2614                        .await
2615                    }
2616                }
2617                // AUTONOMOUS-BOT-IMPLEMENTED
2618                // TODO-HUMAN-REVIEW(PR-857): Empty virtual kernel ring buffer.
2619                Syscall::Syslog(s) => self.handle_syslog(guest, s).await,
2620                Syscall::ArchPrctl(s) => self.handle_arch_prctl(guest, s).await,
2621                // AUTONOMOUS-BOT-IMPLEMENTED
2622                Syscall::Seccomp(s) => self.handle_seccomp(guest, s).await,
2623                // AUTONOMOUS-BOT-IMPLEMENTED
2624                // TODO-HUMAN-REVIEW(#663)
2625                Syscall::Prctl(s) => self.handle_prctl(guest, s).await,
2626                // AUTONOMOUS-BOT-IMPLEMENTED
2627                // TODO-HUMAN-REVIEW(#663)
2628                Syscall::Getpriority(s) => self.handle_getpriority(guest, s).await,
2629                // AUTONOMOUS-BOT-IMPLEMENTED
2630                // TODO-HUMAN-REVIEW(#663)
2631                Syscall::Setpriority(s) => self.handle_setpriority(guest, s).await,
2632                Syscall::Uname(s) => self.handle_uname(guest, s).await,
2633                Syscall::ExitGroup(s) => self.handle_exit_group(guest, s).await,
2634                Syscall::Exit(s) => self.handle_exit(guest, s).await,
2635
2636                Syscall::Dup(w) => self.handle_dup(guest, w).await.map_err(Into::into),
2637                Syscall::Dup2(w) => self.handle_dup2(guest, w).await.map_err(Into::into),
2638                Syscall::Dup3(w) => self.handle_dup3(guest, w).await.map_err(Into::into),
2639                Syscall::Pipe(w) => self.handle_pipe2(guest, w.into()).await,
2640                Syscall::Pipe2(w) => self.handle_pipe2(guest, w).await,
2641                Syscall::Getrandom(s) => self.handle_getrandom(guest, s).await,
2642                Syscall::Utime(s) => self.handle_utime(guest, s).await.map_err(Into::into),
2643                Syscall::Utimes(s) => self.handle_utimes(guest, s).await.map_err(Into::into),
2644                // NB: lutimes is a libc function not a syscall
2645                Syscall::Utimensat(s) => self.handle_utimensat(guest, s).await.map_err(Into::into),
2646                // NB: futimes/futimens are libc functions not a syscall,
2647                // futimesat is obsolete, return -ENOSYS for simplicity.
2648                Syscall::Futimesat(_s) => Err(Error::Errno(Errno::ENOSYS)),
2649                // io_uring completion and memory-sharing semantics are not deterministic.
2650                Syscall::IoUringSetup(_)
2651                | Syscall::IoUringEnter(_)
2652                | Syscall::IoUringRegister(_) => Err(Error::Errno(Errno::ENOSYS)),
2653                Syscall::Socket(s) => self.handle_socket(guest, s).await,
2654                Syscall::Socketpair(s) => self.handle_socketpair(guest, s).await,
2655                Syscall::Connect(s) => self.handle_connect(guest, s).await,
2656                Syscall::Bind(s) => self.handle_bind(guest, s).await,
2657                // AUTONOMOUS-BOT-IMPLEMENTED
2658                // TODO-HUMAN-REVIEW(#663)
2659                Syscall::Setsockopt(s) => self.handle_setsockopt(guest, s).await,
2660                // AUTONOMOUS-BOT-IMPLEMENTED
2661                // TODO-HUMAN-REVIEW(#663)
2662                Syscall::Listen(s) => self.handle_listen(guest, s).await,
2663                // AUTONOMOUS-BOT-IMPLEMENTED
2664                // TODO-HUMAN-REVIEW(#663)
2665                Syscall::Getsockname(s) => self.handle_getsockname(guest, s).await,
2666                // AUTONOMOUS-BOT-IMPLEMENTED
2667                // TODO-HUMAN-REVIEW(#663)
2668                Syscall::Getpeername(s) => self.handle_getpeername(guest, s).await,
2669                // AUTONOMOUS-BOT-IMPLEMENTED
2670                // TODO-HUMAN-REVIEW(#663)
2671                Syscall::Getsockopt(s) => self.handle_getsockopt(guest, s).await,
2672                // AUTONOMOUS-BOT-IMPLEMENTED
2673                // TODO-HUMAN-REVIEW(#818): shutdown is the lone remaining
2674                // socket-family syscall; half-closes a tracked socket and
2675                // forwards via record_or_replay (KVM ratchet round 12).
2676                Syscall::Shutdown(s) => self.handle_shutdown(guest, s).await,
2677                Syscall::Eventfd(s) => self.handle_eventfd2(guest, s.into()).await,
2678                Syscall::Eventfd2(s) => self.handle_eventfd2(guest, s).await,
2679                Syscall::Signalfd(s) => self.handle_signalfd4(guest, s.into()).await,
2680                Syscall::Signalfd4(s) => self.handle_signalfd4(guest, s).await,
2681                Syscall::TimerfdCreate(s) => self.handle_timerfd_create(guest, s).await,
2682                Syscall::TimerfdSettime(s) => self.handle_timerfd_settime(guest, s).await,
2683                Syscall::TimerfdGettime(s) => self.handle_timerfd_gettime(guest, s).await,
2684                Syscall::InotifyInit(s) => {
2685                    self.handle_inotify_init1(guest, InotifyInit1::from(s))
2686                        .await
2687                }
2688                Syscall::InotifyInit1(s) => self.handle_inotify_init1(guest, s).await,
2689                Syscall::InotifyAddWatch(s) => self.handle_inotify_add_watch(guest, s).await,
2690                Syscall::InotifyRmWatch(s) => self.handle_inotify_rm_watch(guest, s).await,
2691                Syscall::MemfdCreate(s) => self.handle_memfd_create(guest, s).await,
2692                // AUTONOMOUS-BOT-IMPLEMENTED
2693                // TODO-HUMAN-REVIEW(PR-862): Record/replay and register pidfds.
2694                Syscall::PidfdOpen(s) => self.handle_pidfd_open(guest, s).await,
2695                // AUTONOMOUS-BOT-IMPLEMENTED
2696                // TODO-HUMAN-REVIEW(PR-899): Host object handles and mount IDs
2697                // are outside Detcore's filesystem identity model.
2698                Syscall::NameToHandleAt(_) => Err(Error::Errno(Errno::EOPNOTSUPP)),
2699                Syscall::Userfaultfd(s) => self.handle_userfaultfd(guest, s).await,
2700                Syscall::Accept(s) => self.handle_accept4(guest, s.into()).await,
2701                Syscall::Accept4(s) => self.handle_accept4(guest, s).await,
2702
2703                Syscall::Nanosleep(s) => self.handle_nanosleep_family(guest, s.into()).await,
2704                Syscall::ClockNanosleep(s) => self.handle_nanosleep_family(guest, s.into()).await,
2705                Syscall::SchedYield(s) => self.handle_sched_yield(guest, s).await,
2706
2707                // NB: getdents is not recommended, (g)libc should call getdents64 only
2708                // see: sysdeps/unix/sysv/linux/getdents.c.
2709                Syscall::Getdents(s) => self.handle_getdents(guest, s).await,
2710                Syscall::Getdents64(s) => self.handle_getdents64(guest, s).await,
2711
2712                Syscall::Poll(s) => self.handle_poll(guest, s).await,
2713                // AUTONOMOUS-BOT-IMPLEMENTED
2714                // TODO-HUMAN-REVIEW(#686): Review scratch fd sets and scheduler polling.
2715                Syscall::Pselect6(s) => self.handle_pselect6(guest, s).await,
2716                // AUTONOMOUS-BOT-IMPLEMENTED
2717                // TODO-HUMAN-REVIEW(#800): select is the timeval sibling of pselect6.
2718                Syscall::Select(s) => self.handle_select(guest, s).await,
2719                // AUTONOMOUS-BOT-IMPLEMENTED
2720                Syscall::Ppoll(s) => self.handle_ppoll(guest, s).await,
2721                Syscall::EpollCreate(s) => {
2722                    self.handle_epoll_create1(guest, EpollCreate1::from(s))
2723                        .await
2724                }
2725                Syscall::EpollCreate1(s) => self.handle_epoll_create1(guest, s).await,
2726                Syscall::EpollCtl(s) => self.handle_epoll_ctl(guest, s).await,
2727                Syscall::EpollPwait(s) => self.handle_epoll_pwait(guest, s).await,
2728                Syscall::EpollWait(s) => self.handle_epoll_wait(guest, s).await,
2729                Syscall::EpollWaitOld(s) => panic!(
2730                    "Not handling deprecated syscall: {}",
2731                    s.display(&guest.memory())
2732                ),
2733                // AUTONOMOUS-BOT-IMPLEMENTED
2734                // TODO-HUMAN-REVIEW(#549)
2735                // The obsolete x86_64 entry point is absent from modern Linux kernels.
2736                Syscall::EpollCtlOld(_) => Err(Error::Errno(Errno::ENOSYS)),
2737
2738                Syscall::SchedGetaffinity(s) => self.handle_sched_getaffinity(guest, s).await,
2739                Syscall::SchedSetaffinity(s) => self.handle_sched_setaffinity(guest, s).await,
2740
2741                // ===== BATCH 3: NUMA memory-placement and Linux CPU-scheduling
2742                // policy. Hermit exposes a single virtual NUMA node and replaces
2743                // the Linux scheduler with Detcore, so these are inoperative and
2744                // are virtualized to fixed, host-independent results (see the
2745                // determinism argument in syscall_classification.rs). Setters and
2746                // count-returning calls are no-ops; getters emulate a default
2747                // single-node / SCHED_OTHER answer.
2748                // AUTONOMOUS-BOT-IMPLEMENTED
2749                // TODO-HUMAN-REVIEW(#720)
2750                Syscall::Mbind(_) => Ok(0),
2751                Syscall::SetMempolicy(_) => Ok(0),
2752                Syscall::GetMempolicy(s) => self.handle_get_mempolicy(guest, s).await,
2753                Syscall::MigratePages(_) => Ok(0),
2754                Syscall::MovePages(s) => self.handle_move_pages(guest, s).await,
2755                Syscall::SchedSetscheduler(_) => Ok(0),
2756                Syscall::SchedSetparam(_) => Ok(0),
2757                // Report the fixed default policy SCHED_OTHER (0).
2758                Syscall::SchedGetscheduler(_) => Ok(0),
2759                Syscall::SchedGetparam(s) => self.handle_sched_getparam(guest, s).await,
2760                Syscall::SchedRrGetInterval(s) => self.handle_sched_rr_get_interval(guest, s).await,
2761
2762                // ===== BATCH 51: fail-closed utility syscalls, re-enabling chrt,
2763                // ionice, and flock under --strict. Detcore replaces the Linux
2764                // scheduler, exposes a single virtual CPU, and serializes guest
2765                // threads, so a thread's Linux scheduling attributes (sched_getattr)
2766                // and I/O priority (ioprio_set) are inert: those two have no
2767                // deterministic effect and are emulated to fixed, host-independent
2768                // results (see syscall_classification.rs).
2769                //
2770                // flock is NOT in that inert group and is not emulated. This comment
2771                // used to claim it "is never contended inside the serialized
2772                // container"; that was measured false -- two open file descriptions
2773                // in ONE process both held the same LOCK_EX under the old no-op,
2774                // where native Linux excluded the second -- and the no-op was
2775                // removed. Serializing THREADS does not make a whole-file lock
2776                // uncontended, because flock conflicts are between OPEN FILE
2777                // DESCRIPTIONS. flock is now forwarded to the kernel, like fcntl's
2778                // POSIX record locks; handle_flock carries the determinism argument
2779                // and the one case Detcore refuses (a contended BLOCKING request,
2780                // which it cannot park a thread on deterministically).
2781                // AUTONOMOUS-BOT-IMPLEMENTED
2782                // TODO-HUMAN-REVIEW(#791)
2783                Syscall::SchedGetattr(s) => self.handle_sched_getattr(guest, s).await,
2784                // AUTONOMOUS-BOT-IMPLEMENTED
2785                // TODO-HUMAN-REVIEW(PR-841): Review virtual sched_setattr no-op policy.
2786                Syscall::SchedSetattr(s) => self.handle_sched_setattr(guest, s).await,
2787                // AUTONOMOUS-BOT-IMPLEMENTED
2788                // TODO-HUMAN-REVIEW(#791)
2789                Syscall::IoprioSet(s) => self.handle_ioprio_set(guest, s).await,
2790                // AUTONOMOUS-BOT-IMPLEMENTED
2791                // TODO-HUMAN-REVIEW(PR-881): Review virtual ioprio_get defaults.
2792                Syscall::IoprioGet(s) => self.handle_ioprio_get(guest, s).await,
2793                // AUTONOMOUS-BOT-IMPLEMENTED
2794                // TODO-HUMAN-REVIEW(#2373)
2795                Syscall::Flock(s) => self.handle_flock(guest, s).await,
2796
2797                // TODO-HUMAN-REVIEW(PR-1064): recvfrom/read/readv/recvmmsg reach
2798                // a NETLINK_SOCK_DIAG dump exactly as recvmsg does. Until they
2799                // were routed through the same sanitizer, four of the five
2800                // usable receive syscalls returned raw host socket inode
2801                // numbers, which made the determinization optional from the
2802                // guest's point of view: `socket.recv()` alone was enough to
2803                // skip it. Non-socket-diag descriptors take the same path as
2804                // before; the predicate is checked inside.
2805                Syscall::Recvfrom(s) if self.sock_diag_reply_fd(guest, s.fd()) => {
2806                    self.handle_sock_diag_recvfrom(guest, s).await
2807                }
2808                Syscall::Recvfrom(s) => self.handle_socket_receive(guest, s, s.fd(), true).await,
2809                // AUTONOMOUS-BOT-IMPLEMENTED
2810                // TODO-HUMAN-REVIEW(PR-901)
2811                Syscall::Recvmsg(s) => self.handle_recvmsg(guest, s).await,
2812                Syscall::Sendto(s) => self.handle_sendrecv(guest, s).await,
2813                Syscall::Sendmsg(s) => self.handle_sendmsg(guest, s).await,
2814                Syscall::Sendmmsg(s) => self.handle_sendmmsg(guest, s).await,
2815
2816                // AUTONOMOUS-BOT-IMPLEMENTED
2817                // TODO-HUMAN-REVIEW(#788): recvmmsg is the multi-message form of
2818                // recvmsg and shares its NonblockableSyscall impl. The fd is made
2819                // temporarily nonblocking, the kernel fills the mmsghdr array
2820                // atomically, and the Detcore scheduler owns any blocking, so the
2821                // timeout argument (deliberately ignored, see helpers.rs) does not
2822                // introduce nondeterminism.
2823                // TODO-HUMAN-REVIEW(PR-901): Review batched ancillary timestamp rewriting.
2824                Syscall::Recvmmsg(s) if self.sock_diag_reply_fd(guest, s.fd()) => {
2825                    self.handle_sock_diag_recvmmsg(guest, s).await
2826                }
2827                Syscall::Recvmmsg(s) => self.handle_recvmmsg(guest, s).await,
2828                Syscall::RtSigtimedwait(s) => self.handle_rt_sigtimedwait(guest, s).await,
2829                Syscall::RtSigsuspend(s) => self.handle_rt_sigsuspend(guest, s).await,
2830                // AUTONOMOUS-BOT-IMPLEMENTED
2831                // TODO-HUMAN-REVIEW(#663)
2832                Syscall::RtSigpending(s) => self.handle_rt_sigpending(guest, s).await,
2833                // AUTONOMOUS-BOT-IMPLEMENTED
2834                // TODO-HUMAN-REVIEW(#663)
2835                Syscall::Kill(s) => self.handle_kill(guest, s).await,
2836                // AUTONOMOUS-BOT-IMPLEMENTED
2837                // TODO-HUMAN-REVIEW(#663)
2838                Syscall::Tgkill(s) => self.handle_tgkill(guest, s).await,
2839                // AUTONOMOUS-BOT-IMPLEMENTED
2840                // TODO-HUMAN-REVIEW(#812)
2841                Syscall::Tkill(s) => self.handle_tkill(guest, s).await,
2842                // AUTONOMOUS-BOT-IMPLEMENTED
2843                // TODO-HUMAN-REVIEW(#812)
2844                Syscall::RtSigqueueinfo(s) => self.handle_rt_sigqueueinfo(guest, s).await,
2845                // AUTONOMOUS-BOT-IMPLEMENTED
2846                // TODO-HUMAN-REVIEW(#812)
2847                Syscall::RtTgsigqueueinfo(s) => self.handle_rt_tgsigqueueinfo(guest, s).await,
2848
2849                Syscall::Execve(s) => self.handle_execveat(guest, s.into()).await,
2850                Syscall::Execveat(s) => self.handle_execveat(guest, s).await,
2851
2852                Syscall::Getcpu(s) => self.handle_getcpu(guest, s).await,
2853
2854                // AUTONOMOUS-BOT-IMPLEMENTED
2855                // TODO-HUMAN-REVIEW(#1549): Credential-query
2856                // family emulated to the fixed virtual-root identity (0). See
2857                // syscall_classification.rs for the determinism rationale.
2858                // getuid/geteuid/getgid/getegid return the constant directly;
2859                // getresuid/getresgid write the constant to each provided result
2860                // pointer. Never forwarded to the host, so the answer no longer
2861                // depends on whether the backend runs the guest in a
2862                // CLONE_NEWUSER namespace.
2863                Syscall::Getuid(_)
2864                | Syscall::Geteuid(_)
2865                | Syscall::Getgid(_)
2866                | Syscall::Getegid(_) => Ok(0),
2867                Syscall::Getresuid(s) => self.handle_getresuid(guest, s).await,
2868                Syscall::Getresgid(s) => self.handle_getresgid(guest, s).await,
2869                Syscall::RtSigprocmask(s) => self.handle_rt_sigprocmask(guest, s).await,
2870                Syscall::RtSigaction(s) => self.handle_rt_sigaction(guest, s).await,
2871                Syscall::Alarm(s) => self.handle_alarm(guest, s).await,
2872                Syscall::Pause(s) => self.handle_pause(guest, s).await,
2873
2874                Syscall::Getrusage(s) => self.handle_getrusage(guest, s).await,
2875                Syscall::Sysinfo(s) => self.handle_sysinfo(guest, s).await,
2876                // AUTONOMOUS-BOT-IMPLEMENTED
2877                Syscall::Times(s) => self.handle_times(guest, s).await,
2878                Syscall::Prlimit64(s) => self.handle_prlimit64(guest, s).await,
2879                // AUTONOMOUS-BOT-IMPLEMENTED
2880                // TODO-HUMAN-REVIEW(#663)
2881                Syscall::Getrlimit(s) => self.handle_getrlimit(guest, s).await,
2882                // AUTONOMOUS-BOT-IMPLEMENTED
2883                // TODO-HUMAN-REVIEW(#663)
2884                Syscall::Setrlimit(s) => self.handle_setrlimit(guest, s).await,
2885
2886                // POSIX per-process timers use the virtual clock and scheduler
2887                // for deterministic arming and supported signal delivery.
2888                Syscall::TimerCreate(s) => self.handle_timer_create(guest, s).await,
2889                Syscall::TimerSettime(s) => self.handle_timer_settime(guest, s).await,
2890                Syscall::TimerGettime(s) => self.handle_timer_gettime(guest, s).await,
2891                Syscall::TimerGetoverrun(s) => self.handle_timer_getoverrun(guest, s).await,
2892                Syscall::TimerDelete(s) => self.handle_timer_delete(guest, s).await,
2893
2894                // Serialized threads share a total memory order, so process-wide
2895                // memory barriers are trivially satisfied and can be no-ops.
2896                Syscall::Membarrier(s) => self.handle_membarrier(guest, s).await,
2897
2898                // Filesystem statistics: passthrough is record/replay-aware so the
2899                // (otherwise host-dependent) result is captured and reproduced.
2900                // statfs/fstatfs run the real syscall, then canonicalize the
2901                // host-varying fields (free blocks/inodes, fsid) so the result is
2902                // deterministic under --verify (a bare passthrough diverged, e.g.
2903                // for tar).
2904                Syscall::Statfs(s) => self.handle_statfs(guest, s).await,
2905                Syscall::Fstatfs(s) => self.handle_fstatfs(guest, s).await,
2906
2907                unexpected => {
2908                    self.handle_unsupported_syscall(
2909                        guest,
2910                        unexpected,
2911                        dettid,
2912                        panic_on_unsupported_syscalls,
2913                    )
2914                    .await
2915                }
2916            },
2917            // AUTONOMOUS-BOT-IMPLEMENTED
2918            // TODO-HUMAN-REVIEW(PR-2985): Review scheduler tracking of set_tid_address.
2919            // Linux and the backend own the pass-through call. Detcore also
2920            // mirrors its successful registration into the scheduler because
2921            // the scheduler supplies the logical CHILD_CLEARTID wake.
2922            SyscallClassification::PassThrough if call.number() == Sysno::set_tid_address => {
2923                match call {
2924                    Syscall::SetTidAddress(s) => self.handle_set_tid_address(guest, s).await,
2925                    _ => unreachable!("set_tid_address unexpectedly lost its typed variant"),
2926                }
2927            }
2928            // AUTONOMOUS-BOT-IMPLEMENTED
2929            // TODO-HUMAN-REVIEW(PR-2223): Review observing
2930            // the robust-list registration without changing its pass-through
2931            // classification. This is still a pass-through — Linux owns the
2932            // registration and supplies its result — but Detcore remembers the
2933            // head address so thread exit can replay `exit_robust_list()`
2934            // against its own futex waiter pool.
2935            SyscallClassification::PassThrough if call.number() == Sysno::set_robust_list => {
2936                match call {
2937                    Syscall::SetRobustList(s) => self.handle_set_robust_list(guest, s).await,
2938                    _ => self.passthrough(guest, call).await,
2939                }
2940            }
2941            // faccessat2 and fchmodat2 are untyped in the pinned Reverie revision; the
2942            // reviewed classification table routes them, and every other reviewed
2943            // PassThrough syscall, through the blanket arm below.
2944            // AUTONOMOUS-BOT-IMPLEMENTED
2945            // TODO-HUMAN-REVIEW(PR-644): Keep dispatch aligned with the reviewed classification.
2946            SyscallClassification::PassThrough => self.passthrough(guest, call).await,
2947            SyscallClassification::Unsupported => {
2948                self.handle_unsupported_syscall(guest, call, dettid, panic_on_unsupported_syscalls)
2949                    .await
2950            }
2951        };
2952
2953        // A copy may already have changed guest memory before reporting a
2954        // terminal backend error. Do not perform even the syscall-result
2955        // display's memory reads, or later observers/posthooks/timer effects.
2956        // Random-device reads have completed their release RPC before returning;
2957        // getrandom acquired no file resource. Physical cleanup belongs to the
2958        // backend failure owner, not to this observer fence.
2959        if res.as_ref().is_err_and(crate::random::is_copy_failure) {
2960            return res;
2961        }
2962
2963        detlog!(
2964            event = crate::detlog::DetLogEvent::SyscallResult {
2965                finished_syscall_number: new_count,
2966            };
2967            "[syscall][detcore, dtid {}] finish syscall #{}: {} = {:?}",
2968            dettid,
2969            new_count,
2970            display_syscall_finished(&call, &guest.memory(), &res),
2971            res
2972        );
2973
2974        // Same guest-logical-control point that already anchors the stack/heap hashes: the
2975        // syscall is complete and its result written back, so the guest logically has control.
2976        // Reading registers is itself backend work, so keep the disabled path inert.  In
2977        // particular, a run that does not request register evidence must not be perturbed by
2978        // collecting data that will immediately be discarded.
2979        if self.cfg.detlog_regs {
2980            let control_point_regs = guest.regs().await;
2981            let regs_seq = guest.thread_state().stats.syscall_count;
2982            self.detlog_registers(guest, &control_point_regs, regs_seq);
2983        }
2984
2985        // brk is PassThrough, so nothing else records where the guest's heap is.
2986        // Both brk(NULL) and brk(addr) return the break in effect afterwards.
2987        if let Syscall::Brk(_) = &call
2988            && let Ok(brk) = res
2989            && brk > 0
2990        {
2991            guest
2992                .thread_state()
2993                .memory_metadata
2994                .lock()
2995                .expect("memory metadata mutex poisoned")
2996                .observe_brk(brk as u64);
2997        }
2998
2999        self.detlog_memory_maps(guest)?;
3000        // Same control point again, for the bytes this syscall moved through a guest buffer.
3001        // Unlike the two mapping hashes above, the extent comes from the syscall's OWN
3002        // arguments, so it does not matter whether the buffer lives on the stack, in the brk
3003        // heap, in BSS or in an anonymous mmap -- the last two of which neither mapping hash
3004        // can see. Only successful calls moved anything.
3005        if let Ok(ret) = &res
3006            && self.cfg.detlog_io_buffers
3007        {
3008            io_buffers::detlog_io_buffers(guest, &call, *ret, dettid, rng_readv_output.as_deref())?;
3009        }
3010
3011        if sequentialize_threads && self.cfg.should_trace_schedevent() {
3012            trace_schedevent(
3013                guest,
3014                with_guest_time(
3015                    guest,
3016                    SchedEvent::syscall(dettid, call.number(), SyscallPhase::Posthook),
3017                ),
3018                true,
3019            )
3020            .await;
3021        }
3022
3023        // The syscall is finished; a turn the post-hook takes (a timeslice
3024        // end) is the thread's own and advances global time as usual.
3025        guest.thread_state_mut().in_uncharged_bootstrap_syscall = false;
3026        self.post_handler_hook(guest).await;
3027
3028        // Defense-in-depth: unless the backend already owns this guarantee,
3029        // force the syscall-clobbered registers (%rcx/%r11 on x86-64) to
3030        // deterministic values before returning to the guest.
3031        if !self.cfg.syscall_clobbers_virtualized_by_backend {
3032            self.canonicalize_syscall_clobbers(guest).await;
3033        }
3034
3035        res
3036    }
3037
3038    async fn on_exit_thread<G: GlobalRPC<Self::GlobalState>>(
3039        &self,
3040        tid: Tid,
3041        global_state: &G,
3042        mut thread_state: Self::ThreadState,
3043        exit_status: ExitStatus,
3044    ) -> Result<(), Error> {
3045        let dettid = thread_state.dettid;
3046        // Cancellation can consume the backend's transferred state before the
3047        // successful-exec callback has rebound its logical identity.
3048        let current = DetTid::from_raw(tid.as_raw());
3049        let transferred_exec = current != dettid;
3050        debug!(
3051            "[detcore, dtid {}] thread exit hook, deregistering from scheduler.",
3052            dettid
3053        );
3054        // Close the final in-progress timeslice so this thread contributes its
3055        // last (partial) slice to the run report, even if it never exhausted a
3056        // full slice.
3057        let now = thread_state.thread_logical_time.as_nanos();
3058        thread_state.stats.close_final_timeslice(now);
3059        // Reverie invokes this callback while the backend still owns the exit
3060        // event, before the guest parent can consume it with wait. Ptrace also
3061        // guarantees that the process leader exits after the other threads, so
3062        // the final published aggregate is complete when wait returns.
3063        // DETERMINISTIC RECOVERY (TODO-HUMAN-REVIEW(PR-1147)). `ThreadState::detpid`
3064        // is `Option` and starts as `None` ("Initialized later" at the clone site),
3065        // so a thread that reaches the exit hook before its per-thread identity is
3066        // populated used to `.expect()` here. That panic fires inside a Reverie
3067        // teardown callback, while the backend still owns the exit event, which is
3068        // the worst place to abort: it can wedge the supervisor rather than fail one
3069        // thread. Fall back to the PROCESS-level `self.detpid`, which is
3070        // non-optional and preserves the identity source selected for this
3071        // backend: virtual for the current DBT ABI, physical for ABI v1. Warn so
3072        // the exceptional window is observable instead of silently papered over.
3073        //
3074        // CORRECTED BY #2348, and stated rather than quietly dropped: the normal
3075        // current-ABI DBT path pre-populates `thread_state.detpid` with the
3076        // client-published virtual process identity, so thread start and exit
3077        // agree on that value. This recovery is only for a thread that exits
3078        // before that initialization. In that exceptional window thread start
3079        // would fall back to `guest.pid()` (the physical host pid for `DbtGuest`),
3080        // while this exit path uses `self.detpid`. ABI v1 also intentionally keeps
3081        // physical callback identities. The warning makes either compatibility
3082        // or recovery path observable; this comment does not claim those fallback
3083        // identities are virtual or independently deterministic.
3084        let (detpid, used_process_detpid) =
3085            select_thread_exit_detpid(thread_state.detpid, self.detpid);
3086        if used_process_detpid {
3087            tracing::warn!(
3088                "[detcore, dtid {}] thread exited before its per-thread detpid was \
3089                 initialized; falling back to the process detpid {}",
3090                dettid,
3091                self.detpid
3092            );
3093        }
3094        // The transferred survivor is the final leader even if cancellation
3095        // precedes local rebinding. Its snapshot includes the worker's tail;
3096        // the earlier displaced-leader snapshot is only a prefix.
3097        if current == detpid {
3098            thread_state.record_exited_child_process_cpu_time(detpid);
3099        } else {
3100            thread_state.account_process_cpu_time();
3101        }
3102        let mm_id = thread_state.mm_id;
3103        let exit_signal = match &exit_status {
3104            ExitStatus::Signaled(signal, _) => Some(*signal as i32),
3105            ExitStatus::Exited(_) => None,
3106        };
3107        // Publish each owner's final clock under its own RPC identity before
3108        // counting that owner in the complete physical-exit barrier. Otherwise
3109        // the group's maximum could be charged to whichever callback finishes
3110        // last, followed by a backwards update from its real deregistration.
3111        // Exec preparation already cleared the old image's robust list. An
3112        // unbound transfer must authenticate its consuming cleanup before any
3113        // ordinary RPC can publish the survivor's clock under the leader TID.
3114        let exit_time_accounted = !transferred_exec
3115            && (!thread_state.has_matching_robust_list_exit(exit_signal)
3116                || acknowledge_robust_list_exit_time(
3117                    thread_state.thread_logical_time.clone(),
3118                    global_state,
3119                    mm_id,
3120                )
3121                .await);
3122        if !exit_time_accounted {
3123            // Preserve benign cleanup for a retired incarnation without
3124            // acknowledging an unaccounted owner or releasing its staged wakes.
3125            thread_state.record_robust_list_head(None);
3126        }
3127        if exit_time_accounted
3128            && let Some((_group_time, ready)) = thread_state.take_robust_list_wakes_after_exit(
3129                exit_signal,
3130                thread_state.thread_logical_time.clone(),
3131            )
3132        {
3133            let identities: Vec<_> = ready
3134                .iter()
3135                .map(|(owner, wake)| (*owner, wake.futex))
3136                .collect();
3137            let counts = robust_list_wakes_after_exit(
3138                thread_state.thread_logical_time.clone(),
3139                global_state,
3140                mm_id,
3141                ready,
3142            )
3143            .await;
3144            for ((owner, futex), count) in identities.into_iter().zip(counts) {
3145                info!(
3146                    "[detcore, dtid {}] robust-list owner death woke {} waiter(s) on futex {:?} after physical exit",
3147                    owner, count, futex,
3148                );
3149            }
3150        }
3151        let pending_chaos_epochs = thread_state.take_pending_chaos_epochs();
3152        let deregistration = ThreadDeregistration {
3153            dettid,
3154            detpid,
3155            mm: mm_id,
3156            thread_start_entered: thread_state.thread_start_entered,
3157            timeslice_stats: thread_state.stats.timeslice_stats,
3158            syscall_count: thread_state.stats.syscall_count,
3159            chaos_epochs: pending_chaos_epochs,
3160        };
3161        if transferred_exec {
3162            tool_global::retire_exec(
3163                thread_state.thread_logical_time.clone(),
3164                global_state,
3165                deregistration,
3166                exit_status.signal().is_some(),
3167            )
3168            .await?;
3169        } else {
3170            deregister_thread(
3171                thread_state.thread_logical_time.clone(),
3172                &self.cfg,
3173                global_state,
3174                deregistration,
3175            )
3176            .await;
3177        }
3178
3179        self.record_or_replay
3180            .on_exit_thread(
3181                tid,
3182                global_state,
3183                thread_state.record_or_replay,
3184                exit_status,
3185            )
3186            .await?;
3187
3188        Ok(())
3189    }
3190}
3191
3192#[cfg(test)]
3193mod subscription_tests {
3194    use super::*;
3195
3196    fn strict_config(passthru_opt: bool) -> Config {
3197        Config {
3198            sequentialize_threads: true,
3199            deterministic_io: true,
3200            passthru_opt,
3201            ..Default::default()
3202        }
3203    }
3204
3205    /// The last row of the pinned table is the one a `Sysno::iter()` sweep
3206    /// drops. `lsm_list_modules` is Determinized AND deterministically refused,
3207    /// so before this was fixed it executed natively against the host under
3208    /// `--passthru-opt` — which is the default for `hermit record` and
3209    /// `hermit replay` — instead of receiving its fixed refusal.
3210    #[test]
3211    fn passthru_opt_covers_the_final_row_of_the_pinned_table() {
3212        let last = Sysno::last();
3213        // Guard the premise: if the table endpoint moves, this test must be
3214        // re-derived rather than silently passing on a different syscall.
3215        assert_eq!(last, Sysno::lsm_list_modules);
3216        assert!(crate::is_determinized_syscall(last));
3217        assert!(crate::is_deterministically_refused_syscall(last));
3218        // The bug this pins: the final row is absent from `Sysno::iter()`.
3219        assert!(!Sysno::iter().any(|sysno| sysno == last));
3220
3221        let subscriptions = <Detcore as Tool>::subscriptions(&strict_config(true));
3222        assert!(
3223            subscriptions.iter_syscalls().any(|sysno| sysno == last),
3224            "{last} must be intercepted under passthru_opt; it is deterministically refused"
3225        );
3226    }
3227
3228    #[test]
3229    fn passthru_opt_intercepts_every_unsupported_syscall() {
3230        let subscriptions = <Detcore as Tool>::subscriptions(&strict_config(true));
3231        let unsupported: Vec<Sysno> = crate::all_pinned_syscalls()
3232            .filter(|sysno| crate::is_unsupported_syscall(*sysno))
3233            .collect();
3234
3235        assert_eq!(unsupported, [Sysno::restart_syscall]);
3236        for syscall in unsupported {
3237            assert!(
3238                subscriptions
3239                    .iter_syscalls()
3240                    .any(|subscribed| subscribed == syscall),
3241                "passthru_opt allowed unsupported {syscall} to bypass Detcore"
3242            );
3243        }
3244    }
3245
3246    /// `passthru_opt` is not a niche flag: `record_or_replay_config` turns it on
3247    /// for every `hermit record` / `hermit replay`, so this covers the record
3248    /// and replay subscription too.
3249    #[test]
3250    fn passthru_opt_subscribes_every_determinized_syscall() {
3251        let determinized: Vec<Sysno> = crate::all_pinned_syscalls()
3252            .filter(|sysno| crate::is_determinized_syscall(*sysno))
3253            .collect();
3254        let subscriptions = <Detcore as Tool>::subscriptions(&strict_config(true));
3255        let delivered: Vec<Sysno> = subscriptions.iter_syscalls().collect();
3256        let missing = determinized
3257            .iter()
3258            .filter(|sysno| !delivered.contains(sysno))
3259            .copied()
3260            .collect::<Vec<_>>();
3261
3262        assert!(
3263            missing.is_empty(),
3264            "passthru_opt let Determinized syscalls bypass Detcore: {}",
3265            missing
3266                .iter()
3267                .map(|sysno| sysno.to_string())
3268                .collect::<Vec<_>>()
3269                .join(" ")
3270        );
3271        assert!(
3272            delivered.contains(&Sysno::syslog),
3273            "syslog must reach its deterministic Detcore handler"
3274        );
3275    }
3276
3277    #[test]
3278    fn passthru_opt_intercepts_thread_exit_registrations() {
3279        let subscriptions = <Detcore as Tool>::subscriptions(&strict_config(true));
3280        for syscall in [Sysno::set_tid_address, Sysno::set_robust_list] {
3281            assert!(
3282                subscriptions
3283                    .iter_syscalls()
3284                    .any(|subscribed| subscribed == syscall),
3285                "passthru_opt allowed {syscall} to bypass Detcore's thread-exit state"
3286            );
3287        }
3288    }
3289
3290    #[test]
3291    fn passthru_opt_leaves_unlisted_passthrough_syscalls_unsubscribed() {
3292        assert_eq!(
3293            syscall_classification::classify_syscall(Sysno::chdir),
3294            syscall_classification::SyscallClassification::PassThrough
3295        );
3296
3297        let subscriptions = <Detcore as Tool>::subscriptions(&strict_config(true));
3298        assert!(
3299            !subscriptions
3300                .iter_syscalls()
3301                .any(|sysno| sysno == Sysno::chdir),
3302            "chdir is PassThrough and must remain outside the partial subscription"
3303        );
3304    }
3305
3306    /// Io-buffer hashing can only hash a syscall Detcore is subscribed to.
3307    /// Under `--passthru-opt` the subscription narrows, so the check's coverage
3308    /// narrows with it -- silently, because a syscall that never reaches
3309    /// Detcore produces no record and no record is indistinguishable from a
3310    /// syscall that moved no bytes.
3311    ///
3312    /// Measured 2026-08-21 on `f05bf04e4f`, one probe calling `getcwd`,
3313    /// `recvmsg`, `readv` and `readlink`:
3314    ///
3315    /// ```text
3316    /// hermit --log=info run                --detlog-io-buffers -- ./probe   12 [iobuf] records
3317    /// hermit --log=info run --passthru-opt --detlog-io-buffers -- ./probe   11 [iobuf] records
3318    /// ```
3319    ///
3320    /// The single lost syscall is `getcwd`, and the reason is structural: 9 of
3321    /// the 20 are in the literal allow-list and 10 more arrive via the
3322    /// unconditional Determinized sweep at the end of `subscriptions`, but
3323    /// `getcwd` is classified `PassThrough`, so neither path picks it up.
3324    ///
3325    /// WHY THIS IS A TEST AND NOT A RUNTIME WARNING. The set is stable, so a
3326    /// warning would print the same sentence on every `--passthru-opt` run
3327    /// forever and be tuned out within a week. The exposure is not today's
3328    /// one-syscall gap; it is that 19 of 20 holds only because two
3329    /// INDEPENDENTLY MAINTAINED lists happen to agree -- the classification
3330    /// table in `syscall_classification.rs` and the match arms in
3331    /// `io_buffers.rs`. Reclassifying any one of those ten from `Determinized`
3332    /// to `PassThrough` would drop it out of the sweep and out of io-buffers'
3333    /// reach with nothing failing. This asserts the relationship so that edit
3334    /// cannot land quietly.
3335    ///
3336    /// It is deliberately an EQUALITY, not a subset check, so it fails in both
3337    /// directions: a nineteenth syscall going missing, and `getcwd` becoming
3338    /// covered while this expectation still claims it is not.
3339    #[test]
3340    fn passthru_opt_leaves_io_buffer_hashing_blind_only_for_getcwd() {
3341        let subscribed: Vec<Sysno> = <Detcore as Tool>::subscriptions(&strict_config(true))
3342            .iter_syscalls()
3343            .collect();
3344        let unreachable: Vec<Sysno> = crate::io_buffers::HASHED_SYSCALLS
3345            .iter()
3346            .copied()
3347            .filter(|sysno| !subscribed.contains(sysno))
3348            .collect();
3349
3350        assert_eq!(
3351            unreachable,
3352            vec![Sysno::getcwd],
3353            "--passthru-opt changes which syscalls io-buffer hashing can reach, and the set \
3354             moved. {} of {} reachable. If a syscall was RECLASSIFIED out of Determinized, that \
3355             silently shrank an enabled determinism check -- re-derive rather than editing this \
3356             expectation to match.",
3357            crate::io_buffers::HASHED_SYSCALLS.len() - unreachable.len(),
3358            crate::io_buffers::HASHED_SYSCALLS.len()
3359        );
3360    }
3361
3362    /// The other direction, and the reason the check above is worth having:
3363    /// with the default subscription every buffer-carrying syscall is
3364    /// reachable, so there is nothing to report on an ordinary run.
3365    #[test]
3366    fn the_default_subscription_reaches_every_io_buffer_syscall() {
3367        let subscriptions = <Detcore as Tool>::subscriptions(&strict_config(false));
3368        let unreachable: Vec<Sysno> = crate::io_buffers::HASHED_SYSCALLS
3369            .iter()
3370            .copied()
3371            .filter(|sysno| !subscriptions.iter_syscalls().any(|s| s == *sysno))
3372            .collect();
3373
3374        assert!(
3375            unreachable.is_empty(),
3376            "without --passthru-opt every io-buffer syscall must be reachable; missing {unreachable:?}"
3377        );
3378    }
3379
3380    #[test]
3381    fn strict_subscriptions_intercept_every_event_by_default() {
3382        let subscriptions = <Detcore as Tool>::subscriptions(&strict_config(false));
3383
3384        assert_eq!(subscriptions, Subscription::all());
3385        assert!(
3386            subscriptions
3387                .iter_syscalls()
3388                .any(|sysno| sysno == Sysno::ppoll)
3389        );
3390
3391        // On a backend without its own CPUID table (ptrace), `--no-virtualize-cpuid`
3392        // must not subscribe to `cpuid` in either mode. The
3393        // subscription is what makes Reverie probe and enable CPUID faulting, and
3394        // a host without faulting then logs an ERROR for each traced exec
3395        // (https://github.com/rrnewton/hermit/issues/3460). Everything else stays
3396        // as intercepted as it is with CPUID virtualization on.
3397        for passthru_opt in [false, true] {
3398            let virtualized = <Detcore as Tool>::subscriptions(&strict_config(passthru_opt));
3399            let config = Config {
3400                virtualize_cpuid: false,
3401                ..strict_config(passthru_opt)
3402            };
3403            let subscriptions = <Detcore as Tool>::subscriptions(&config);
3404
3405            assert!(virtualized.has_cpuid(), "passthru_opt={passthru_opt}");
3406            assert!(!subscriptions.has_cpuid(), "passthru_opt={passthru_opt}");
3407            assert!(subscriptions.has_rdtsc(), "passthru_opt={passthru_opt}");
3408            assert!(
3409                subscriptions
3410                    .iter_syscalls()
3411                    .eq(virtualized.iter_syscalls()),
3412                "passthru_opt={passthru_opt}: the syscall set must not depend on CPUID virtualization"
3413            );
3414        }
3415
3416        // KVM installs its own CPUID table, so with virtualization off the guest
3417        // sees host values only through the trap: keep subscribing there.
3418        let kvm_host_cpuid = Config {
3419            virtualize_cpuid: false,
3420            cpuid_virtualized_by_backend: true,
3421            ..strict_config(false)
3422        };
3423        assert_eq!(
3424            <Detcore as Tool>::subscriptions(&kvm_host_cpuid),
3425            Subscription::all()
3426        );
3427    }
3428
3429    #[test]
3430    fn passthru_opt_uses_the_partial_subscription_set() {
3431        let subscriptions = <Detcore as Tool>::subscriptions(&strict_config(true));
3432
3433        assert_ne!(subscriptions, Subscription::all());
3434        assert!(
3435            subscriptions
3436                .iter_syscalls()
3437                .any(|sysno| sysno == Sysno::clock_gettime)
3438        );
3439        assert!(
3440            subscriptions
3441                .iter_syscalls()
3442                .any(|sysno| sysno == Sysno::rt_sigsuspend)
3443        );
3444        assert!(
3445            subscriptions
3446                .iter_syscalls()
3447                .any(|sysno| sysno == Sysno::ppoll)
3448        );
3449        assert!(
3450            subscriptions
3451                .iter_syscalls()
3452                .any(|sysno| sysno == Sysno::madvise)
3453        );
3454        assert!(
3455            subscriptions
3456                .iter_syscalls()
3457                .any(|sysno| sysno == Sysno::arch_prctl)
3458        );
3459        assert!(
3460            subscriptions
3461                .iter_syscalls()
3462                .any(|sysno| sysno == Sysno::writev)
3463        );
3464        for sysno in [
3465            Sysno::read,
3466            Sysno::write,
3467            Sysno::pread64,
3468            Sysno::pwrite64,
3469            Sysno::readv,
3470            Sysno::writev,
3471            Sysno::preadv,
3472            Sysno::preadv2,
3473            Sysno::pwritev,
3474            Sysno::pwritev2,
3475            Sysno::prctl,
3476        ] {
3477            assert!(
3478                subscriptions
3479                    .iter_syscalls()
3480                    .any(|subscribed| subscribed == sysno),
3481                "timer-slack mediation requires {sysno:?}"
3482            );
3483        }
3484        assert!(
3485            subscriptions
3486                .iter_syscalls()
3487                .any(|sysno| sysno == Sysno::pwrite64)
3488        );
3489        assert!(
3490            subscriptions
3491                .iter_syscalls()
3492                .any(|sysno| sysno == Sysno::pidfd_open)
3493        );
3494    }
3495}
3496
3497#[cfg(test)]
3498mod rcb_overshoot_tests {
3499    use std::fmt::Write;
3500    use std::sync::Arc;
3501    use std::sync::Mutex;
3502
3503    use tracing::Event;
3504    use tracing::Id;
3505    use tracing::Level;
3506    use tracing::Metadata;
3507    use tracing::Subscriber;
3508    use tracing::field::Field;
3509    use tracing::field::Visit;
3510    use tracing::span::Attributes;
3511    use tracing::span::Record;
3512    use tracing::subscriber::with_default;
3513
3514    use super::rcb_timer_overshot;
3515    use super::report_rcb_overshoot;
3516
3517    struct ErrorSubscriber(Arc<Mutex<Option<String>>>);
3518
3519    struct EventVisitor(String);
3520
3521    impl Visit for EventVisitor {
3522        fn record_debug(&mut self, field: &Field, value: &dyn std::fmt::Debug) {
3523            let _ = write!(self.0, "{}={:?}", field.name(), value);
3524        }
3525    }
3526
3527    impl Subscriber for ErrorSubscriber {
3528        fn enabled(&self, metadata: &Metadata<'_>) -> bool {
3529            *metadata.level() == Level::ERROR
3530        }
3531
3532        fn new_span(&self, _span: &Attributes<'_>) -> Id {
3533            Id::from_u64(1)
3534        }
3535
3536        fn record(&self, _span: &Id, _values: &Record<'_>) {}
3537
3538        fn record_follows_from(&self, _span: &Id, _follows: &Id) {}
3539
3540        fn event(&self, event: &Event<'_>) {
3541            if *event.metadata().level() == Level::ERROR {
3542                let mut visitor = EventVisitor(String::new());
3543                event.record(&mut visitor);
3544                *self.0.lock().unwrap() = Some(visitor.0);
3545            }
3546        }
3547
3548        fn enter(&self, _span: &Id) {}
3549
3550        fn exit(&self, _span: &Id) {}
3551    }
3552
3553    #[test]
3554    fn default_overshoot_policy_emits_error_and_returns() {
3555        let _ = reverie::take_skid_overshoot_count();
3556        let error = Arc::new(Mutex::new(None));
3557        with_default(ErrorSubscriber(error.clone()), || {
3558            report_rcb_overshoot(false, 16_249, 139, 100);
3559        });
3560
3561        let error = error.lock().unwrap().take().expect("missing ERROR event");
3562        assert!(error.contains(reverie::SKID_OVERSHOOT_MARKER), "{error}");
3563        assert!(error.contains("PMU RCB overshoot"), "{error}");
3564        assert!(error.contains("16249"), "{error}");
3565        assert!(error.contains("139"), "{error}");
3566        assert!(error.contains("100"), "{error}");
3567        assert_eq!(
3568            reverie::take_skid_overshoot_count(),
3569            1,
3570            "the log-and-continue path must feed the supervisor's structural count"
3571        );
3572    }
3573
3574    #[test]
3575    fn exact_rcb_timer_hit_is_not_an_overshoot() {
3576        assert!(!rcb_timer_overshot(100, 100));
3577        assert!(!rcb_timer_overshot(99, 100));
3578        assert!(rcb_timer_overshot(101, 100));
3579    }
3580
3581    #[test]
3582    #[should_panic(expected = "PMU RCB overshoot")]
3583    fn opt_in_overshoot_policy_panics() {
3584        report_rcb_overshoot(true, 16_249, 139, 100);
3585    }
3586}
3587
3588#[cfg(test)]
3589mod timeslice_timer_tests {
3590    use super::*;
3591
3592    #[test]
3593    fn manual_interrupts_can_shorten_but_not_extend_maximum() {
3594        assert_eq!(choose_rcb_timer(100, 100, Some(150)), (50, false));
3595        assert_eq!(choose_rcb_timer(100, 100, Some(250)), (100, true));
3596        assert_eq!(choose_rcb_timer(100, 100, None), (100, true));
3597    }
3598
3599    #[test]
3600    fn pmu_duration_conversion_applies_clock_multiplier() {
3601        let duration = crate::types::LogicalTime::from_nanos(100);
3602        assert_eq!(duration.into_rcbs_with_multiplier(2.0), 5);
3603        assert_eq!(duration.into_rcbs_with_multiplier(0.5), 20);
3604        assert_eq!(
3605            crate::types::LogicalTime::from_nanos(101).into_rcbs_with_multiplier(2.0),
3606            5
3607        );
3608    }
3609
3610    #[test]
3611    #[should_panic(expected = "max_timeslice must be at least one RCB")]
3612    fn detcore_constructor_validates_programmatic_config() {
3613        let config = Config {
3614            max_timeslice: std::num::NonZeroU64::new(1),
3615            ..Default::default()
3616        };
3617
3618        let _ = <Detcore as Tool>::new(Pid::from_raw(1), &config);
3619    }
3620}
3621
3622#[cfg(test)]
3623mod process_tree_guest_clock_tests {
3624    use super::*;
3625
3626    #[test]
3627    fn forked_process_shares_guest_clock_domain() {
3628        let config = Config::default();
3629        let tool = <Detcore as Tool>::new(Pid::from_raw(1), &config);
3630        let mut parent = ThreadState::new(DetPid::from_raw(1), &config, ());
3631        parent.clone_flags = Some(CloneFlags::empty());
3632
3633        let child = <Detcore as Tool>::init_thread_state(
3634            &tool,
3635            Tid::from_raw(2),
3636            Some((Tid::from_raw(1), &parent)),
3637        );
3638
3639        assert!(Arc::ptr_eq(&parent.guest_clock, &child.guest_clock));
3640    }
3641}
3642
3643#[cfg(test)]
3644mod child_rng_identity_tests {
3645    use super::*;
3646
3647    fn child_state(
3648        tool: &Detcore,
3649        parent: &mut ThreadState<()>,
3650        host_tid: i32,
3651        clone_flags: CloneFlags,
3652    ) -> ThreadState<()> {
3653        parent.clone_flags = Some(clone_flags);
3654        <Detcore as Tool>::init_thread_state(
3655            tool,
3656            Tid::from_raw(host_tid),
3657            Some((Tid::from_raw(parent.dettid.as_raw()), parent)),
3658        )
3659    }
3660
3661    fn rng_sample(child: &ThreadState<()>) -> [u64; 4] {
3662        let mut rng = child.prng.clone();
3663        std::array::from_fn(|_| rng.random())
3664    }
3665
3666    fn chaos_rng_sample(child: &ThreadState<()>) -> [u64; 4] {
3667        let mut rng = child.chaos_prng.clone();
3668        std::array::from_fn(|_| rng.random())
3669    }
3670
3671    #[test]
3672    fn common_child_rng_identity_ignores_backend_tid_for_every_clone_shape() {
3673        let config = Config::default();
3674        let tool = <Detcore as Tool>::new(Pid::from_raw(1), &config);
3675        for (label, clone_flags) in [
3676            ("fork", CloneFlags::empty()),
3677            ("vfork", CloneFlags::CLONE_VM | CloneFlags::CLONE_VFORK),
3678            ("clone-process", CloneFlags::CLONE_VM),
3679            (
3680                "clone-thread",
3681                CloneFlags::CLONE_VM | CloneFlags::CLONE_THREAD,
3682            ),
3683        ] {
3684            let mut low_tid_parent = ThreadState::new(DetPid::from_raw(1), &config, ());
3685            let mut high_tid_parent = ThreadState::new(DetPid::from_raw(1), &config, ());
3686            let low_tid_child = child_state(&tool, &mut low_tid_parent, 2, clone_flags);
3687            let high_tid_child = child_state(&tool, &mut high_tid_parent, 42_002, clone_flags);
3688
3689            assert_eq!(format!("{}", low_tid_child.pedigree), "C", "{label}");
3690            assert_eq!(low_tid_child.dettid.as_raw(), 2, "{label}");
3691            assert_eq!(high_tid_child.dettid.as_raw(), 42_002, "{label}");
3692            assert_eq!(
3693                rng_sample(&low_tid_child),
3694                rng_sample(&high_tid_child),
3695                "{label} child RNG depended on the backend Tid"
3696            );
3697            assert_eq!(
3698                chaos_rng_sample(&low_tid_child),
3699                chaos_rng_sample(&high_tid_child),
3700                "{label} child chaos RNG depended on the backend Tid"
3701            );
3702            assert_ne!(
3703                rng_sample(&low_tid_child),
3704                chaos_rng_sample(&low_tid_child),
3705                "{label} child guest and chaos RNG streams were coupled"
3706            );
3707        }
3708    }
3709
3710    #[test]
3711    fn serialized_siblings_receive_distinct_child_rng_streams() {
3712        let config = Config::default();
3713        let tool = <Detcore as Tool>::new(Pid::from_raw(1), &config);
3714        let mut parent = ThreadState::new(DetPid::from_raw(1), &config, ());
3715
3716        let first = child_state(&tool, &mut parent, 2, CloneFlags::empty());
3717        let first_pedigree = parent.pedigree.fork_mut();
3718        assert_eq!(format!("{}", first.pedigree), format!("{first_pedigree}"));
3719        let second = child_state(&tool, &mut parent, 3, CloneFlags::empty());
3720
3721        assert_eq!(format!("{}", second.pedigree), "PC");
3722        assert_ne!(rng_sample(&first), rng_sample(&second));
3723    }
3724}
3725
3726#[cfg(test)]
3727mod thread_exit_identity_tests {
3728    use super::*;
3729
3730    #[test]
3731    fn initialized_thread_exit_identity_is_preserved() {
3732        let thread_detpid = DetPid::from_raw(41);
3733        let process_detpid = DetPid::from_raw(7);
3734        assert_eq!(
3735            select_thread_exit_detpid(Some(thread_detpid), process_detpid),
3736            (thread_detpid, false)
3737        );
3738    }
3739
3740    #[test]
3741    fn missing_thread_exit_identity_uses_process_identity() {
3742        let process_detpid = DetPid::from_raw(7);
3743        assert_eq!(
3744            select_thread_exit_detpid(None, process_detpid),
3745            (process_detpid, true)
3746        );
3747    }
3748}
3749
3750#[cfg(test)]
3751mod thread_start_identity_tests {
3752    use super::*;
3753
3754    #[test]
3755    fn backend_process_identity_is_preserved() {
3756        let backend_detpid = DetPid::from_raw(3);
3757        assert_eq!(
3758            select_thread_start_detpid(Some(backend_detpid), Pid::from_raw(42_001)),
3759            backend_detpid
3760        );
3761        assert_eq!(
3762            select_thread_start_detpid(None, Pid::from_raw(42_001)),
3763            DetPid::from_raw(42_001)
3764        );
3765    }
3766
3767    #[test]
3768    fn root_thread_uses_deterministic_process_identity() {
3769        let detpid = DetPid::from_raw(3);
3770        assert!(is_root_thread_start(true, DetTid::from_raw(3), detpid));
3771        assert!(!is_root_thread_start(false, DetTid::from_raw(3), detpid));
3772        assert!(!is_root_thread_start(true, DetTid::from_raw(4), detpid));
3773    }
3774}
3775
3776#[cfg(test)]
3777mod thread_cpu_time_tests {
3778    use super::*;
3779
3780    #[test]
3781    fn cloned_threads_and_processes_keep_absolute_time_without_inheriting_work() {
3782        for clone_flags in [
3783            CloneFlags::CLONE_THREAD,
3784            CloneFlags::empty(),
3785            CloneFlags::CLONE_VFORK | CloneFlags::CLONE_VM,
3786        ] {
3787            let config = Config::default();
3788            let tool = <Detcore as Tool>::new(Pid::from_raw(1), &config);
3789            let mut parent = ThreadState::new(DetPid::from_raw(1), &config, ());
3790            parent.thread_logical_time.add_rcbs(7);
3791            parent.thread_logical_time.add_syscall_with_cost(101);
3792            parent.clone_flags = Some(clone_flags);
3793            let child = <Detcore as Tool>::init_thread_state(
3794                &tool,
3795                Tid::from_raw(2),
3796                Some((Tid::from_raw(1), &parent)),
3797            );
3798            assert_eq!(
3799                child.thread_logical_time.as_nanos(),
3800                parent.thread_logical_time.as_nanos()
3801            );
3802            assert_eq!(
3803                child.thread_logical_time.inherited_nanos(),
3804                LogicalTime::from_nanos(171)
3805            );
3806            let mut global = GlobalTime::new(&config);
3807            let epoch = global.as_nanos();
3808            for thread in [&parent, &child] {
3809                global.update_global_time(
3810                    thread.dettid,
3811                    thread.thread_logical_time.as_nanos(),
3812                    thread.thread_logical_time.inherited_nanos(),
3813                );
3814            }
3815            assert_eq!(global.as_nanos(), epoch + LogicalTime::from_nanos(171));
3816        }
3817    }
3818
3819    #[test]
3820    fn cloned_thread_and_fork_child_start_with_zero_thread_cpu() {
3821        for clone_flags in [CloneFlags::CLONE_THREAD, CloneFlags::empty()] {
3822            let config = Config::default();
3823            let tool = <Detcore as Tool>::new(Pid::from_raw(1), &config);
3824            let mut parent = ThreadState::new(DetPid::from_raw(1), &config, ());
3825
3826            // Give the parent nonzero CPU before the child exists. The child's
3827            // absolute logical clock inherits this position for scheduler
3828            // ordering, but Linux per-thread CPU accounting must not.
3829            parent.thread_logical_time.add_rcbs(200);
3830            parent.thread_logical_time.add_syscall();
3831            parent.clone_flags = Some(clone_flags);
3832
3833            let mut child = <Detcore as Tool>::init_thread_state(
3834                &tool,
3835                Tid::from_raw(2),
3836                Some((Tid::from_raw(1), &parent)),
3837            );
3838            assert_eq!(
3839                child.thread_cpu_time(),
3840                (LogicalTime::ZERO, LogicalTime::ZERO),
3841                "clone flags {clone_flags:?} inherited pre-creation CPU"
3842            );
3843
3844            child.thread_logical_time.add_rcbs(4);
3845            child.thread_logical_time.add_syscall();
3846            let (user, system) = child.thread_cpu_time();
3847            assert!(user > LogicalTime::ZERO);
3848            assert!(system > LogicalTime::ZERO);
3849            assert!(user < parent.thread_logical_time.user_cpu_time());
3850            assert!(system <= parent.thread_logical_time.system_cpu_time());
3851        }
3852    }
3853}
3854
3855/// Regression tests for <https://github.com/rrnewton/hermit/issues/3153>: a
3856/// failed syscall must not render an output buffer the kernel never wrote.
3857///
3858/// Each buffer is pre-filled with sentinel values standing in for the
3859/// uninitialized guest stack the issue observed, and the rendering is checked
3860/// through the same `finish syscall` format the DETLOG line uses.
3861#[cfg(test)]
3862mod finished_syscall_display_tests {
3863    use std::ffi::CString;
3864
3865    use reverie::syscalls::AddrMut;
3866    use reverie::syscalls::AtFlags;
3867    use reverie::syscalls::ClockGettime;
3868    use reverie::syscalls::ClockId;
3869    use reverie::syscalls::Gettimeofday;
3870    use reverie::syscalls::LocalMemory;
3871    use reverie::syscalls::Newfstatat;
3872    use reverie::syscalls::PathPtr;
3873    use reverie::syscalls::StatPtr;
3874    use reverie::syscalls::Statx;
3875    use reverie::syscalls::StatxMask;
3876    use reverie::syscalls::StatxPtr;
3877    use reverie::syscalls::Timespec;
3878    use reverie::syscalls::TimespecMutPtr;
3879    use reverie::syscalls::Timeval;
3880    use reverie::syscalls::TimevalMutPtr;
3881
3882    use super::*;
3883
3884    /// A value no real `st_size` or timestamp in these tests can take.
3885    const SENTINEL: i64 = 0x5EED_0BAD_F00D;
3886
3887    /// Render exactly what the `finish syscall` DETLOG line renders.
3888    fn finish_line(syscall: &Syscall, result: Result<i64, Error>) -> String {
3889        let memory = LocalMemory::new();
3890        format!(
3891            "{} = {:?}",
3892            display_syscall_finished(syscall, &memory, &result),
3893            result
3894        )
3895    }
3896
3897    fn sentinel_stat() -> libc::stat {
3898        // SAFETY: all-zero is a valid `struct stat`.
3899        let mut stat: libc::stat = unsafe { std::mem::zeroed() };
3900        stat.st_mode = libc::S_IFREG | 0o644;
3901        stat.st_size = SENTINEL;
3902        stat
3903    }
3904
3905    fn sentinel_statx() -> libc::statx {
3906        // SAFETY: all-zero is a valid `struct statx`.
3907        let mut statx: libc::statx = unsafe { std::mem::zeroed() };
3908        statx.stx_mode = (libc::S_IFREG | 0o644) as u16;
3909        statx.stx_size = SENTINEL as u64;
3910        statx
3911    }
3912
3913    fn newfstatat(path: &CString, stat: &libc::stat) -> Syscall {
3914        Syscall::Newfstatat(
3915            Newfstatat::new()
3916                .with_dirfd(libc::AT_FDCWD)
3917                .with_path(PathPtr::from_ptr(path.as_ptr()))
3918                .with_stat(StatPtr::from_ptr(stat as *const libc::stat))
3919                .with_flags(AtFlags::empty()),
3920        )
3921    }
3922
3923    fn statx(path: &CString, statx: &libc::statx) -> Syscall {
3924        Syscall::Statx(
3925            Statx::new()
3926                .with_dirfd(libc::AT_FDCWD)
3927                .with_path(PathPtr::from_ptr(path.as_ptr()))
3928                .with_flags(AtFlags::empty())
3929                .with_mask(StatxMask::STATX_BASIC_STATS)
3930                .with_statx(StatxPtr::from_ptr(statx as *const libc::statx)),
3931        )
3932    }
3933
3934    fn assert_no_struct_rendered(line: &str) {
3935        assert!(!line.contains("st_mode"), "rendered st_mode: {line}");
3936        assert!(!line.contains("st_size"), "rendered st_size: {line}");
3937        assert!(
3938            !line.contains(&SENTINEL.to_string()),
3939            "rendered the unwritten sentinel: {line}"
3940        );
3941    }
3942
3943    #[test]
3944    fn failed_newfstatat_renders_pointer_and_errno_but_not_the_buffer() {
3945        let path = CString::new("/nonexistent/issue-3153").unwrap();
3946        let stat = sentinel_stat();
3947        let line = finish_line(&newfstatat(&path, &stat), Err(Errno::ENOENT.into()));
3948
3949        assert_no_struct_rendered(&line);
3950        assert!(
3951            line.contains(&format!("{:p}", &stat as *const libc::stat)),
3952            "stat pointer missing: {line}"
3953        );
3954        assert!(
3955            line.contains("\"/nonexistent/issue-3153\""),
3956            "path missing: {line}"
3957        );
3958        assert!(line.contains("ENOENT"), "errno missing: {line}");
3959    }
3960
3961    #[test]
3962    fn successful_newfstatat_still_renders_the_buffer() {
3963        let path = CString::new("/dev/null").unwrap();
3964        let stat = sentinel_stat();
3965        let line = finish_line(&newfstatat(&path, &stat), Ok(0));
3966
3967        assert!(
3968            line.contains(&format!(
3969                "{:p} -> {{st_mode=SFlag(S_IFREG) | 0644, st_size={}, ...}}",
3970                &stat as *const libc::stat, SENTINEL
3971            )),
3972            "stat struct missing: {line}"
3973        );
3974    }
3975
3976    #[test]
3977    fn failed_statx_renders_pointer_and_errno_but_not_the_buffer() {
3978        let path = CString::new("/nonexistent/issue-3153").unwrap();
3979        let buf = sentinel_statx();
3980        let line = finish_line(&statx(&path, &buf), Err(Errno::ENOENT.into()));
3981
3982        assert_no_struct_rendered(&line);
3983        assert!(
3984            line.contains(&format!("{:p}", &buf as *const libc::statx)),
3985            "statx pointer missing: {line}"
3986        );
3987        assert!(line.contains("ENOENT"), "errno missing: {line}");
3988    }
3989
3990    #[test]
3991    fn successful_statx_still_renders_the_buffer() {
3992        let path = CString::new("/dev/null").unwrap();
3993        let buf = sentinel_statx();
3994        let line = finish_line(&statx(&path, &buf), Ok(0));
3995
3996        assert!(
3997            line.contains(&format!(
3998                "{:p} -> {{st_mode=SFlag(S_IFREG) | 0644, st_size={}, ...}}",
3999                &buf as *const libc::statx, SENTINEL
4000            )),
4001            "statx struct missing: {line}"
4002        );
4003    }
4004
4005    #[test]
4006    fn failed_clock_gettime_does_not_render_the_timespec() {
4007        let tp = Timespec {
4008            tv_sec: SENTINEL,
4009            tv_nsec: 0,
4010        };
4011        let call = Syscall::ClockGettime(
4012            ClockGettime::new()
4013                .with_clockid(ClockId::CLOCK_MONOTONIC)
4014                .with_tp(Some(TimespecMutPtr(
4015                    AddrMut::from_ptr(&tp as *const Timespec).unwrap(),
4016                ))),
4017        );
4018
4019        let failed = finish_line(&call, Err(Errno::EINVAL.into()));
4020        assert!(!failed.contains("tv_sec"), "rendered tv_sec: {failed}");
4021        assert!(
4022            failed.contains(&format!("{:p}", &tp as *const Timespec)),
4023            "tp pointer missing: {failed}"
4024        );
4025        assert!(failed.contains("EINVAL"), "errno missing: {failed}");
4026
4027        let succeeded = finish_line(&call, Ok(0));
4028        assert!(
4029            succeeded.contains(&format!("tv_sec: {SENTINEL}")),
4030            "timespec missing on success: {succeeded}"
4031        );
4032    }
4033
4034    /// EFAULT is also the error of a copy-out that faulted part way, after
4035    /// `copy_to_user()` stored a prefix the guest can read, so it must keep
4036    /// the buffer rendered for every allowlisted syscall.
4037    #[test]
4038    fn efault_failures_still_render_a_possible_partial_copy() {
4039        let path = CString::new("/dev/null").unwrap();
4040        let stat = sentinel_stat();
4041        let line = finish_line(&newfstatat(&path, &stat), Err(Errno::EFAULT.into()));
4042        assert!(
4043            line.contains(&format!(
4044                "{:p} -> {{st_mode=SFlag(S_IFREG) | 0644, st_size={}, ...}}",
4045                &stat as *const libc::stat, SENTINEL
4046            )),
4047            "partial stat copy hidden on EFAULT: {line}"
4048        );
4049        assert!(line.contains("EFAULT"), "errno missing: {line}");
4050
4051        let buf = sentinel_statx();
4052        let line = finish_line(&statx(&path, &buf), Err(Errno::EFAULT.into()));
4053        assert!(
4054            line.contains(&format!(
4055                "{:p} -> {{st_mode=SFlag(S_IFREG) | 0644, st_size={}, ...}}",
4056                &buf as *const libc::statx, SENTINEL
4057            )),
4058            "partial statx copy hidden on EFAULT: {line}"
4059        );
4060
4061        let tp = Timespec {
4062            tv_sec: SENTINEL,
4063            tv_nsec: 0,
4064        };
4065        let call = Syscall::ClockGettime(
4066            ClockGettime::new()
4067                .with_clockid(ClockId::CLOCK_MONOTONIC)
4068                .with_tp(Some(TimespecMutPtr(
4069                    AddrMut::from_ptr(&tp as *const Timespec).unwrap(),
4070                ))),
4071        );
4072        let line = finish_line(&call, Err(Errno::EFAULT.into()));
4073        assert!(
4074            line.contains(&format!("tv_sec: {SENTINEL}")),
4075            "partial timespec copy hidden on EFAULT: {line}"
4076        );
4077    }
4078
4079    /// A tool error is not the syscall's result and proves nothing about the
4080    /// buffer, so only a guest errno other than EFAULT hides it.
4081    #[test]
4082    fn only_non_efault_guest_errnos_prove_the_buffer_unwritten() {
4083        for errno in [Errno::ENOENT, Errno::EINVAL, Errno::EBADF, Errno::ENOMEM] {
4084            assert!(failure_proves_outputs_unwritten(&Err(errno.into())));
4085        }
4086        assert!(!failure_proves_outputs_unwritten(
4087            &Err(Errno::EFAULT.into())
4088        ));
4089        assert!(!failure_proves_outputs_unwritten(&Ok(0)));
4090        assert!(!failure_proves_outputs_unwritten(&Err(Error::Tool(
4091            anyhow::anyhow!("tool failure")
4092        ))));
4093
4094        let path = CString::new("/dev/null").unwrap();
4095        let stat = sentinel_stat();
4096        let line = finish_line(
4097            &newfstatat(&path, &stat),
4098            Err(Error::Tool(anyhow::anyhow!("tool failure"))),
4099        );
4100        assert!(
4101            line.contains(&format!("st_size={SENTINEL}")),
4102            "buffer hidden on a tool error: {line}"
4103        );
4104    }
4105
4106    /// `gettimeofday` stores `tv` before it can fail with `EFAULT` on a bad
4107    /// `tz`, so a failed call can carry a kernel-written output. It must stay
4108    /// rendered: suppressing it would hide real evidence from DETLOG.
4109    #[test]
4110    fn failed_gettimeofday_still_renders_the_timeval() {
4111        let tv = Timeval {
4112            tv_sec: SENTINEL,
4113            tv_usec: 0,
4114        };
4115        let call = Syscall::Gettimeofday(
4116            Gettimeofday::new()
4117                .with_tv(Some(TimevalMutPtr(
4118                    AddrMut::from_ptr(&tv as *const Timeval).unwrap(),
4119                )))
4120                .with_tz(None),
4121        );
4122
4123        let line = finish_line(&call, Err(Errno::EFAULT.into()));
4124        assert!(
4125            line.contains(&format!("tv_sec: {SENTINEL}")),
4126            "kernel-written timeval hidden on failure: {line}"
4127        );
4128        assert!(line.contains("EFAULT"), "errno missing: {line}");
4129    }
4130
4131    /// `fstat` never renders its buffer (T136880615); the result must not
4132    /// change that in either direction.
4133    #[test]
4134    fn fstat_never_renders_the_buffer() {
4135        let stat = sentinel_stat();
4136        let call = Syscall::Fstat(
4137            reverie::syscalls::Fstat::new()
4138                .with_fd(3)
4139                .with_stat(StatPtr::from_ptr(&stat as *const libc::stat)),
4140        );
4141        for result in [Ok(0), Err(Errno::EBADF.into())] {
4142            assert_no_struct_rendered(&finish_line(&call, result));
4143        }
4144    }
4145}