detcore/lib.rs
1/*
2 * Copyright (c) Meta Platforms, Inc. and affiliates.
3 * All rights reserved.
4 *
5 * This source code is licensed under the BSD-style license found in the
6 * LICENSE file in the root directory of this source tree.
7 */
8
9//! Detcore is a Reverie tool that determinizes the execution of a process.
10//!
11//! # Backend-abstraction commandment
12//!
13//! Detcore is a *tool* written against Reverie's **abstract** instrumentation
14//! interface (the `reverie` crate). It depends only on those traits and types
15//! and is deliberately ignorant of how a guest is actually traced.
16//!
17//! Detcore MUST NEVER depend on or import a concrete Reverie backend or support
18//! crate -- any `reverie-*` crate other than the abstract `reverie-core`
19//! interface. Choosing and instantiating a backend, and running a detcore tool
20//! against it, is the sole responsibility of the `hermit-cli` package. There
21//! are no backend-specific hacks in detcore: any tracing-mechanism-specific
22//! behavior belongs behind the Reverie abstraction, not here.
23//!
24//! Why: Hermit follows Reverie's abstract model. A backend dependency in
25//! detcore would couple the determinism engine to one tracing mechanism and
26//! break the clean abstraction boundary that lets the same tool run over any
27//! backend.
28//!
29//! The one allowed exception is test-only: detcore's own integration tests
30//! (under `detcore/tests/`, wired via the `reverie-ptrace` **dev-dependency**)
31//! drive a real tracer to exercise the tool. That coupling never reaches the
32//! shipped library. This invariant is enforced in CI by
33//! `scripts/check-detcore-backend-abstraction.sh`.
34
35#![deny(clippy::all)]
36#![deny(missing_docs)]
37#![allow(clippy::uninlined_format_args)]
38
39mod config;
40mod consts;
41mod cpuid;
42mod digest;
43mod dirents;
44/// Schedule-alignment and edit-distance algorithms shared by Hermit tools.
45#[allow(missing_docs)]
46pub mod edit_distance;
47mod fd;
48mod io_buffers;
49mod iovecs;
50#[allow(unused)]
51mod ivar;
52pub mod logdiff;
53mod memory;
54pub mod netlink_route;
55mod procfs;
56mod procmaps;
57pub mod random;
58mod record_or_replay;
59mod resources;
60mod scheduler;
61mod sock_diag;
62mod stat;
63mod syscall_classification;
64mod syscall_time;
65mod syscalls;
66mod tool_global;
67mod tool_local;
68pub mod util;
69
70pub mod detlog;
71pub mod preemptions;
72pub mod types;
73use std::fs::File;
74use std::io::Write;
75use std::os::unix::io::RawFd;
76use std::sync::Arc;
77use std::sync::Mutex;
78use std::time::Duration;
79
80pub use config::BlockingMode;
81pub use config::CONFIG_FINGERPRINT_ENV;
82pub use config::Config;
83pub use config::RunsPostFork;
84pub use config::SchedHeuristic;
85pub use config::config_wire_fingerprint;
86// AUTONOMOUS-BOT-IMPLEMENTED
87// TODO-HUMAN-REVIEW(PR-1120): Review the public canonical Detcore root identity.
88pub use consts::ROOT_DETPID;
89pub use digest::Digest;
90#[cfg(test)]
91use rand::RngExt as _;
92use raw_cpuid::CpuIdResult;
93use raw_cpuid::cpuid;
94pub use record_or_replay::RecordOrReplay;
95use reverie::Error;
96use reverie::ExitStatus;
97use reverie::GlobalRPC;
98use reverie::Guest;
99use reverie::Pid;
100use reverie::Rdtsc;
101use reverie::RdtscResult;
102use reverie::RegDisplay;
103use reverie::Signal;
104use reverie::Subscription;
105use reverie::Tid;
106use reverie::TimerSchedule;
107use reverie::Tool;
108pub use reverie::process::Namespace;
109use reverie::syscalls::CloneFlags;
110use reverie::syscalls::Displayable;
111use reverie::syscalls::EpollCreate1;
112use reverie::syscalls::Errno;
113use reverie::syscalls::InotifyInit1;
114use reverie::syscalls::MemoryAccess;
115use reverie::syscalls::Syscall;
116use reverie::syscalls::SyscallInfo;
117use reverie::syscalls::Sysno;
118pub use scheduler::Priority;
119pub use scheduler::runqueue::DEFAULT_PRIORITY;
120pub use scheduler::runqueue::FIRST_PRIORITY;
121pub use scheduler::runqueue::LAST_PRIORITY;
122pub use tool_global::BackendFailureCleanup;
123pub use tool_global::GlobalState;
124use tool_global::ThreadDeregistration;
125use tool_global::acknowledge_robust_list_exit_time;
126use tool_global::create_child_thread;
127use tool_global::create_vfork_child_thread;
128use tool_global::deregister_thread;
129pub use tool_global::format_unsupported_syscall_warning;
130pub use tool_global::prepare_exec;
131use tool_global::report_unsupported_syscall;
132use tool_global::robust_list_wakes_after_exit;
133
134fn select_thread_exit_detpid(
135 thread_detpid: Option<DetPid>,
136 process_detpid: DetPid,
137) -> (DetPid, bool) {
138 match thread_detpid {
139 Some(detpid) => (detpid, false),
140 None => (process_detpid, true),
141 }
142}
143
144fn select_thread_start_detpid(thread_detpid: Option<DetPid>, guest_pid: Pid) -> DetPid {
145 thread_detpid.unwrap_or_else(|| DetPid::from_raw(guest_pid.into()))
146}
147
148fn is_root_thread_start(is_root_process: bool, dettid: DetTid, detpid: DetPid) -> bool {
149 is_root_process && dettid == detpid
150}
151
152// AUTONOMOUS-BOT-IMPLEMENTED
153// TODO-HUMAN-REVIEW(PR-644): Review the typed fail-closed backend signal.
154/// Identifies an unsupported syscall that a backend must terminate without unwinding.
155#[derive(Debug)]
156pub struct UnsupportedSyscallError(pub Sysno);
157
158impl std::fmt::Display for UnsupportedSyscallError {
159 fn fmt(&self, formatter: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
160 write!(formatter, "unsupported syscall: {:?}", self.0)
161 }
162}
163
164impl std::error::Error for UnsupportedSyscallError {}
165pub use tool_local::Detcore;
166pub use tool_local::FileMetadata;
167/// Returns whether the audited runtime policy classifies `sysno` as unsupported.
168// AUTONOMOUS-BOT-IMPLEMENTED
169// TODO-HUMAN-REVIEW(PR-644): Review the copied-DBT-child classification surface.
170pub fn is_unsupported_syscall(sysno: Sysno) -> bool {
171 matches!(
172 syscall_classification::classify_syscall(sysno),
173 syscall_classification::SyscallClassification::Unsupported
174 )
175}
176
177/// Every syscall in the pinned x86_64 table, including the final entry.
178///
179/// Backends that sweep the classification table must use this rather than
180/// `Sysno::iter()`, which stops one short and silently drops the last row.
181pub fn all_pinned_syscalls() -> impl Iterator<Item = Sysno> {
182 syscall_classification::all_pinned_syscalls()
183}
184
185/// Returns whether the audited runtime policy classifies `sysno` as
186/// `Determinized` — that is, Detcore either models the syscall with a handler
187/// or applies an explicit deterministic refusal policy to it.
188///
189/// This is the complement of the refusal boundary below. A backend that
190/// executes guest syscalls outside `handle_syscall_event` needs BOTH: the
191/// refusal set tells it which syscalls to answer with a fixed errno, and this
192/// predicate tells it which syscalls Detcore claims to determinize at all.
193/// Running a `Determinized` syscall natively is a determinism hole even when it
194/// is not in the refusal set, because the modelling that makes it deterministic
195/// lives in a handler the backend never entered.
196pub fn is_determinized_syscall(sysno: Sysno) -> bool {
197 matches!(
198 syscall_classification::classify_syscall(sysno),
199 syscall_classification::SyscallClassification::Determinized
200 )
201}
202
203/// Returns whether `sysno` is a kernel-keyring syscall (`add_key`,
204/// `request_key`, `keyctl`) that Detcore hides behind a deterministic
205/// `CONFIG_KEYS`-absent boundary under the default fail-closed policy.
206// AUTONOMOUS-BOT-IMPLEMENTED
207// TODO-HUMAN-REVIEW(PR-916): Exposed so the copied-DBT-child policy can preserve
208// the same keyring isolation boundary that the reclassification (PR-848) moved
209// out of the Unsupported set.
210pub fn is_kernel_keyring_syscall(sysno: Sysno) -> bool {
211 syscall_classification::is_kernel_keyring_syscall(sysno)
212}
213
214/// Returns whether Detcore deterministically refuses `sysno` with a fixed
215/// errno when the fail-closed policy is active, without consulting the host.
216///
217/// This is the boundary backends that execute guest syscalls outside Detcore's
218/// `handle_syscall_event` dispatcher (the DBT copied-child fast path and the
219/// KVM executor) consult to enforce the same fixed refusal the ptrace path
220/// enforces. It deliberately excludes emulated / no-op / host-forwarding
221/// families (credential no-ops, `timer_create`, AF_UNIX autobind, `openat2`,
222/// `copy_file_range`), because fail-closing a copied child for those would
223/// diverge from the ptrace path rather than match it.
224// AUTONOMOUS-BOT-IMPLEMENTED
225// TODO-HUMAN-REVIEW(PR-978): Review the copied-DBT-child deterministic-refusal surface.
226pub fn is_deterministically_refused_syscall(sysno: Sysno) -> bool {
227 syscall_classification::is_deterministically_refused_syscall(sysno)
228}
229
230/// Returns whether `sysno` is refused by the default fail-closed policy but
231/// forwarded under the explicit compatibility opt-out. The legacy
232/// `strict_only` name is retained for API compatibility.
233pub fn is_strict_only_deterministic_refusal_syscall(sysno: Sysno) -> bool {
234 syscall_classification::is_strict_only_deterministic_refusal_syscall(sysno)
235}
236
237use tool_local::PosixTimers;
238use tool_local::ProcessCpuTime;
239pub use tool_local::ThreadState;
240pub use tool_local::ThreadStats;
241pub use tool_local::thread_rng_from_parent;
242use tracing::debug;
243use tracing::error;
244use tracing::info;
245use tracing::trace;
246use tracing::warn;
247pub use types::DetTid;
248use types::*;
249pub use util::punch_out_print;
250
251use crate::resources::Permission;
252use crate::resources::ResourceID;
253use crate::syscall_classification::SyscallClassification;
254use crate::syscall_classification::classify_syscall;
255use crate::syscall_classification::is_credential_identity_noop_syscall;
256use crate::syscall_classification::is_futex2_enosys_syscall;
257use crate::syscall_classification::is_host_kernel_probe_syscall;
258use crate::syscall_classification::is_host_security_identity_probe_syscall;
259use crate::syscall_classification::is_landlock_sandbox_syscall;
260use crate::syscall_classification::is_mount_introspection_enosys_syscall;
261use crate::syscall_classification::is_mount_ns_admin_refused_syscall;
262use crate::syscall_classification::is_optional_memory_feature_syscall;
263use crate::syscall_classification::is_ownership_change_noop_syscall;
264use crate::syscall_classification::is_perf_event_enosys_syscall;
265use crate::syscall_classification::is_privileged_admin_refused_syscall;
266use crate::syscall_classification::is_privileged_observation_refused_syscall;
267use crate::syscall_classification::is_process_isolation_refused_syscall;
268use crate::syscall_classification::is_remap_file_pages_enosys_syscall;
269use crate::syscall_classification::is_unimplemented_enosys_syscall;
270use crate::syscall_classification::is_unsupported_async_ipc_syscall;
271use crate::syscall_classification::is_zero_copy_pipe_syscall;
272use crate::syscalls::helpers::with_guest_rip;
273use crate::syscalls::helpers::with_guest_time;
274use crate::syscalls::time::guest_clock_time;
275use crate::tool_global::resource_request;
276use crate::tool_global::trace_schedevent;
277use crate::tool_global::unrecoverable_shutdown;
278use crate::types::SigWrapper;
279
280#[macro_use]
281extern crate bitflags;
282
283#[cold]
284fn report_rcb_overshoot(
285 panic_on_rcb_overshoot: bool,
286 clock_value: u64,
287 delta_rcbs: u64,
288 last_timer: u64,
289) {
290 let message = format!(
291 "{} prehook: PMU RCB overshoot! Clock_value: {}. Stepped forward {} RCBs, but should have trapped at {}",
292 reverie::SKID_OVERSHOOT_MARKER,
293 clock_value,
294 delta_rcbs,
295 last_timer
296 );
297 if panic_on_rcb_overshoot {
298 panic!("{}", message);
299 }
300 reverie::record_skid_overshoot();
301 error!("{}", message);
302}
303
304fn rcb_timer_overshot(delta_rcbs: u64, last_timer: u64) -> bool {
305 delta_rcbs > last_timer
306}
307
308fn choose_rcb_timer(
309 max_rcbs_remaining: u64,
310 current_rcbs: u64,
311 next_interrupt: Option<u64>,
312) -> (u64, bool) {
313 if let Some(next_interrupt) = next_interrupt {
314 let interrupt_rcbs = next_interrupt - current_rcbs;
315 if interrupt_rcbs < max_rcbs_remaining {
316 return (interrupt_rcbs, false);
317 }
318 }
319 (max_rcbs_remaining, true)
320}
321
322impl<T: RecordOrReplay> Detcore<T> {
323 /// Registers a child whose native backend executed the clone syscall.
324 ///
325 /// The caller must initialize the child's local thread state from the same
326 /// parent state and clone flags before the child enters its start hook.
327 // TODO-HUMAN-REVIEW(PR-743): Review the backend-neutral native child registration API.
328 pub async fn register_external_child<G: Guest<Self>>(
329 &self,
330 guest: &mut G,
331 child_tid: Tid,
332 child_tid_addr: usize,
333 flags: CloneFlags,
334 exit_signal: libc::c_int,
335 physical_ids: Option<(i32, i32)>,
336 ) {
337 let child_dettid = DetTid::from_raw(child_tid.into());
338 guest.thread_state_mut().clone_flags = Some(flags);
339 if !flags.contains(CloneFlags::CLONE_THREAD) {
340 guest
341 .thread_state()
342 .prepare_child_process_cpu_time(child_dettid);
343 }
344 let parent_dettid = guest.thread_state().dettid;
345 let parent_pedigree = &mut guest.thread_state_mut().pedigree;
346 let child_pedigree = parent_pedigree.fork_mut();
347 debug!(
348 "[dtid {}] after registering external child (tid {}, pedigree {}) parent's pedigree becomes {}",
349 parent_dettid, child_dettid, child_pedigree, parent_pedigree,
350 );
351 // The kernel clone has already succeeded before an external backend
352 // calls this method, so these parent updates are not speculative. If
353 // registration observes a retired parent, create_child_thread exits
354 // that parent by tail injection; no continuing caller can retry or
355 // roll the successful clone back.
356 tool_global::create_child_thread(
357 guest,
358 child_dettid,
359 tool_global::child_tid_clear_address(flags, child_tid_addr),
360 Some(flags),
361 exit_signal,
362 physical_ids,
363 )
364 .await;
365 guest.thread_state_mut().clone_flags = None;
366 }
367 async fn passthrough<G: Guest<Self>>(
368 &self,
369 guest: &mut G,
370 call: Syscall,
371 ) -> Result<i64, Error> {
372 self.record_or_replay_preserving_tool_errors(guest, call)
373 .await
374 }
375
376 // AUTONOMOUS-BOT-IMPLEMENTED
377 // TODO-HUMAN-REVIEW(PR-643): Review unsupported-syscall reporting and fail-fast behavior.
378 /// Applies the legacy policy to an explicitly listed but unsupported syscall.
379 async fn handle_unsupported_syscall<G: Guest<Self>>(
380 &self,
381 guest: &mut G,
382 call: Syscall,
383 dettid: DetTid,
384 panic_on_unsupported_syscalls: bool,
385 ) -> Result<i64, Error> {
386 if panic_on_unsupported_syscalls {
387 error!(
388 "[detcore, dtid {}] unsupported syscall: {} = ?",
389 dettid,
390 call.display(&guest.memory()),
391 );
392 if guest.config().shutdown_on_unsupported_syscall {
393 // A fail-closed policy decision: the run named a syscall hermit
394 // cannot service and `shutdown_on_unsupported_syscall` says stop.
395 unrecoverable_shutdown(guest, detcore_model::HERMIT_POLICY_REFUSAL_EXIT).await;
396 }
397 if guest.config().exit_on_unsupported_syscall {
398 return Err(Error::Tool(anyhow::Error::new(UnsupportedSyscallError(
399 call.number(),
400 ))));
401 }
402 panic!("unsupported syscall: {:?}", call);
403 }
404 report_unsupported_syscall(guest, call.number()).await;
405 self.passthrough(guest, call).await
406 }
407
408 /// Defense-in-depth determinism for the registers the syscall instruction
409 /// clobbers.
410 ///
411 /// On x86-64 the `syscall` instruction destroys `%rcx` (which the CPU loads
412 /// with the return instruction pointer) and `%r11` (the saved `RFLAGS`).
413 /// After a syscall these are architecturally "undefined", so hermit must not
414 /// assume a well-behaved guest ignores them: a misbehaving guest that reads
415 /// `%rcx`/`%r11` must still observe deterministic values. Reverie's
416 /// injected-syscall path can otherwise leave its *private trampoline page's*
417 /// RIP/RFLAGS in these registers, which is both nondeterministic and an
418 /// information leak of tracer internals.
419 ///
420 /// This forces both registers to the guest's own (deterministic) RIP and
421 /// RFLAGS, which is exactly what a faithful `SYSRET` would leave there. It is
422 /// a no-op when they already hold the canonical values (the common path), so
423 /// it only writes registers when something diverged. Register-preserved
424 /// arguments (`%rdi`..`%r9`, callee-saved) are deliberately left untouched:
425 /// the Linux ABI preserves them, so zeroing them would break faithful,
426 /// well-behaved programs.
427 #[cfg(target_arch = "x86_64")]
428 async fn canonicalize_syscall_clobbers<G: Guest<Self>>(&self, guest: &mut G) {
429 let mut regs = guest.regs().await;
430 // A faithful SYSRET leaves the return RIP in %rcx and RFLAGS in %r11.
431 if regs.rcx != regs.rip || regs.r11 != regs.eflags {
432 regs.rcx = regs.rip;
433 regs.r11 = regs.eflags;
434 if let Err(err) = guest.set_regs(regs).await {
435 // Best-effort: some backends cannot write registers. Do not fail
436 // the syscall over a defense-in-depth hardening step.
437 debug!(
438 "canonicalize_syscall_clobbers: set_regs unsupported/failed: {}",
439 err
440 );
441 }
442 }
443 }
444
445 /// No-op on architectures without the x86-64 `%rcx`/`%r11` syscall clobber.
446 #[cfg(not(target_arch = "x86_64"))]
447 async fn canonicalize_syscall_clobbers<G: Guest<Self>>(&self, _guest: &mut G) {}
448
449 /// Update logical thread time with any outstanding ticks of the Reverie clock. Returns a list
450 /// of corresponding Branch/OtherInstructions events if schedule recording is enabled.
451 ///
452 /// # Arguments
453 ///
454 /// * `precise_branch`: if true, there were no non-branch instructions since the last recorded branch instruction.
455 async fn update_logical_time_rcbs<G: Guest<Self>>(
456 &self,
457 guest: &mut G,
458 precise_branch: bool,
459 ) -> Option<Vec<SchedEvent>> {
460 if self.cfg.max_timeslice.is_some() {
461 let precise_timers = !guest.config().imprecise_timers;
462 // TODO(T86591083): we might need to not always increment as a hack fix
463 // for deterministic virtual time without sequentialize threads.
464 let clock_value = guest.read_clock().expect("Couldn't read clock");
465 // N.B. clock_value does not yet include any updates for the inbound
466 // syscall/instruction because this function is the very first thing that
467 // happens in each type of handler.
468 let thread_state = guest.thread_state_mut();
469 let dettid = thread_state.dettid;
470 assert!(thread_state.committed_clock_value <= clock_value);
471 let delta_rcbs: u64 = clock_value - thread_state.committed_clock_value;
472 if self.cfg.use_rcb_time() {
473 // AUTONOMOUS-BOT-IMPLEMENTED
474 // TODO-HUMAN-REVIEW(PR-1151)
475 if thread_state.chaos_slowdown_active {
476 let factor = thread_state.rcb_time_multiplier();
477 thread_state
478 .thread_logical_time
479 .add_rcbs_with_multiplier(delta_rcbs, factor);
480 } else {
481 thread_state.thread_logical_time.add_rcbs(delta_rcbs);
482 }
483 }
484 thread_state.account_process_cpu_time();
485 thread_state.committed_clock_value = clock_value;
486
487 if thread_state.end_of_timeslice.is_some() {
488 if let Some(last_timer) = thread_state.last_rcb_timer
489 && rcb_timer_overshot(delta_rcbs, last_timer)
490 && precise_timers
491 {
492 report_rcb_overshoot(
493 self.cfg.panic_on_rcb_overshoot,
494 clock_value,
495 delta_rcbs,
496 last_timer,
497 );
498 // Preserve timer state. `pre_handler_hook` will yield through the normal
499 // scheduler path if the slice expired; `post_handler_hook` will otherwise
500 // re-arm an overshot `interrupt_at` timer.
501 }
502 // Otherwise we're very early, at the prehook of handle_thread_start.
503 } else {
504 panic!(
505 "Invariant violation: end_of_timeslice is None during update_logical_time_rcbs..."
506 )
507 }
508
509 trace!(
510 "[dtid {}] updated rcb clock, new logical time: {:?}, i.e. {}, timeslice end: {}, local rcb clock_value {:?}",
511 dettid,
512 &thread_state.thread_logical_time,
513 thread_state.thread_logical_time.as_nanos(),
514 thread_state
515 .end_of_timeslice
516 .map_or_else(|| "".to_string(), |x| format!("{}", x)),
517 clock_value,
518 );
519 if self.cfg.use_rcb_time() && self.cfg.should_trace_schedevent() {
520 let mut vec = Vec::new();
521 let ev = with_guest_time(
522 guest,
523 SchedEvent::branches(
524 dettid,
525 delta_rcbs
526 .try_into()
527 .expect("should not have more than 2^32 branches at once"),
528 ),
529 );
530 let ev = if precise_branch {
531 with_guest_rip(guest, ev).await
532 } else {
533 ev
534 };
535
536 if delta_rcbs > 0 {
537 // We don't fill the end_rip here, because the current rip is NOT precisely the
538 // end of this block of branch events. Other instructions may have occured
539 // since the last branch.
540 vec.push(ev)
541 } else {
542 trace!(
543 "[detcore, dtid {}] Refusing to record zero-braches event: {:?}",
544 &ev.dettid, ev
545 );
546 }
547 if !precise_branch {
548 // This will ALWAYS record, even if the branches above are zero.
549 let ev2 = with_guest_time(
550 guest,
551 SchedEvent {
552 dettid,
553 op: Op::OtherInstructions,
554 count: 1,
555 start_rip: None,
556 end_rip: None,
557 end_time: None,
558 },
559 );
560 // Fill in end_rip because current rip represents the end of this event.
561 let ev2 = with_guest_rip(guest, ev2).await;
562 vec.push(ev2);
563 }
564 Some(vec)
565 } else {
566 None
567 }
568 } else {
569 None
570 }
571 }
572
573 /// A common hook called at the start of *every* handler, just after we receive
574 /// control from the guest.
575 async fn pre_handler_hook<G: Guest<Self>>(&self, guest: &mut G, precise_branch: bool) {
576 // A handler that left early (an error return) may not have cleared
577 // this; no request made by this new handler belongs to that syscall.
578 guest.thread_state_mut().in_uncharged_bootstrap_syscall = false;
579 let dettid = guest.thread_state().dettid;
580 let evs = self.update_logical_time_rcbs(guest, precise_branch).await;
581
582 if guest.thread_state().guest_past_first_execve() {
583 detlog_debug!(
584 "(pre) registers [dtid {}][rcbs {}]. {}",
585 dettid,
586 guest.thread_state().thread_logical_time.rcbs(),
587 guest.regs().await.display()
588 );
589 }
590 trace!(
591 "prehook [dtid {}] Updating rcbs and checking time remaining.",
592 dettid
593 );
594 if let Some(vec) = evs {
595 for ev in vec {
596 trace_schedevent(guest, ev, false).await;
597 }
598 }
599
600 self.end_timeslice_if_needed(guest).await;
601 }
602
603 // AUTONOMOUS-BOT-IMPLEMENTED
604 /// Yield when accumulated logical time reaches the syscall-boundary target deadline.
605 async fn end_timeslice_if_needed<G: Guest<Self>>(&self, guest: &mut G) {
606 let thread_state = guest.thread_state();
607 let Some(slice_end) = thread_state.end_of_timeslice else {
608 return;
609 };
610 if !thread_state.timeslice_expired() {
611 return;
612 }
613
614 trace!(
615 "[dtid {}] logical time {} reached timeslice target {}",
616 thread_state.dettid,
617 thread_state.thread_logical_time.as_nanos(),
618 slice_end
619 );
620 self.end_timeslice(guest).await;
621 }
622
623 /// A common hook called at the end of *every* handler, just before returning control
624 /// to the guest. This enforces the logical target and resets the PMU maximum timer.
625 ///
626 /// However, note that the thread's timeslice (turn) may have expired DURING this handler.
627 /// Therefore the timeslice can end in the posthook as well as in the prehook.
628 async fn post_handler_hook<G: Guest<Self>>(&self, guest: &mut G) {
629 self.end_timeslice_if_needed(guest).await;
630
631 let dettid = guest.thread_state().dettid;
632 let mut current_time = guest.thread_state().thread_logical_time.as_nanos();
633
634 if let Some(mut max_timeslice_end) = guest.thread_state().max_timeslice_end {
635 assert!(guest.config().max_timeslice.is_some());
636 let mut replay_rcb_end = guest.thread_state().replay_rcb_end;
637 // TODO: get rid of fractional NANOS_PER_RCB so it's clear that this does not lose precision:
638 // AUTONOMOUS-BOT-IMPLEMENTED
639 // TODO-HUMAN-REVIEW(PR-1151)
640 let clock_multiplier = guest.config().clock_multiplier.unwrap_or(1.0)
641 * guest.thread_state().rcb_time_multiplier().as_f64();
642 let epsilon = Duration::from_nanos((NANOS_PER_RCB * clock_multiplier).ceil() as u64);
643
644 if replay_rcb_end.is_none() && current_time + epsilon > max_timeslice_end {
645 trace!(
646 "posthook [dtid {}] less than one RCB remains before PMU maximum {}; ending slice",
647 dettid, max_timeslice_end
648 );
649 self.end_timeslice(guest).await;
650 max_timeslice_end = guest
651 .thread_state()
652 .max_timeslice_end
653 .expect("ending a PMU-backed timeslice must install a new maximum");
654 current_time = guest.thread_state().thread_logical_time.as_nanos();
655 replay_rcb_end = guest.thread_state().replay_rcb_end;
656 }
657 if replay_rcb_end.is_none() && current_time + epsilon > max_timeslice_end {
658 panic!(
659 "Ended time slice, but current time {} is still beyond PMU maximum {}",
660 current_time, max_timeslice_end
661 );
662 }
663
664 let current_rcbs = guest.thread_state().thread_logical_time.rcbs();
665 let current_pmu_rcbs = guest.thread_state().committed_clock_value;
666 let (ns_remaining, max_rcbs_remaining) = if let Some(replay_rcb_end) = replay_rcb_end {
667 assert!(
668 replay_rcb_end > current_pmu_rcbs,
669 "recorded PMU RCB deadline {} is not ahead of current {}",
670 replay_rcb_end,
671 current_pmu_rcbs
672 );
673 let logical_remaining = if max_timeslice_end > current_time {
674 max_timeslice_end - current_time
675 } else {
676 LogicalTime::ZERO
677 };
678 (logical_remaining, replay_rcb_end - current_pmu_rcbs)
679 } else {
680 let logical_remaining = max_timeslice_end - current_time;
681 (
682 logical_remaining,
683 logical_remaining.into_rcbs_with_multiplier(clock_multiplier),
684 )
685 };
686 let next_interrupt = self
687 .cfg
688 .use_rcb_time()
689 .then(|| {
690 guest
691 .thread_state()
692 .interrupt_at
693 .range((current_rcbs + 1)..)
694 .next()
695 .copied()
696 })
697 .flatten();
698 let (rcbs_remaining, timer_is_max) =
699 choose_rcb_timer(max_rcbs_remaining, current_rcbs, next_interrupt);
700 if let Some(next_interrupt) = next_interrupt {
701 debug!(
702 "[dtid: {}] current rcbs: {}, next interrupt_at: {}",
703 dettid, current_rcbs, next_interrupt
704 )
705 }
706
707 trace!(
708 "posthook [dtid {}] {} remaining before PMU maximum ({} rcbs).",
709 dettid, ns_remaining, rcbs_remaining,
710 );
711
712 if replay_rcb_end.is_none() && ns_remaining.is_zero() {
713 panic!(
714 "Timer invariant broken: we should not exit a handler with 0 timeslice remaining."
715 );
716 }
717 assert!(rcbs_remaining > 0);
718 trace!(
719 "posthook [dtid {}] Resetting timer to {:?} RCBs in the future (current {})",
720 dettid,
721 rcbs_remaining,
722 guest.thread_state().thread_logical_time.rcbs()
723 );
724 {
725 let thread_state = guest.thread_state_mut();
726 thread_state.last_rcb_timer = Some(rcbs_remaining);
727 thread_state.last_rcb_timer_is_max = timer_is_max;
728 }
729
730 if guest.config().imprecise_timers {
731 guest
732 .set_timer(TimerSchedule::Rcbs(rcbs_remaining))
733 .expect("Failed to set timer");
734 } else {
735 guest
736 .set_timer_precise(TimerSchedule::Rcbs(rcbs_remaining))
737 .expect("Failed to set timer");
738 }
739 } else {
740 assert!(guest.config().max_timeslice.is_none());
741 guest.thread_state_mut().last_rcb_timer = None;
742 guest.thread_state_mut().last_rcb_timer_is_max = false;
743 }
744
745 if guest.thread_state().guest_past_first_execve() {
746 detlog_debug!(
747 "(post) registers [dtid {}][rcbs {}]. {}",
748 dettid,
749 guest.thread_state().thread_logical_time.rcbs(),
750 guest.regs().await.display(),
751 );
752 }
753 }
754
755 /// End this logical timeslice and talk to the scheduler before continuing.
756 ///
757 /// Effects
758 /// - ends timeslice (mutating thread stats and both deadlines)
759 /// - priority change / yield RPC
760 async fn end_timeslice<G: Guest<Self>>(&self, guest: &mut G) {
761 self.end_timeslice_with_sched_yield(guest, false).await;
762 }
763
764 async fn end_timeslice_for_sched_yield<G: Guest<Self>>(&self, guest: &mut G) {
765 self.end_timeslice_with_sched_yield(guest, true).await;
766 }
767
768 async fn end_timeslice_with_sched_yield<G: Guest<Self>>(
769 &self,
770 guest: &mut G,
771 mut explicit_sched_yield: bool,
772 ) {
773 let chaos = guest.config().chaos;
774 loop {
775 let thread_state = guest.thread_state();
776 let dettid = thread_state.dettid;
777 let end_time = thread_state.thread_logical_time.as_nanos();
778 info!(
779 "[detcore, dtid {}] ending timeslice T{}. {} syscalls and {} signals this timeslice.",
780 dettid,
781 thread_state.stats.timeslice_count,
782 thread_state.stats.timeslice_syscall_count,
783 thread_state.stats.timeslice_signal_count,
784 );
785 let maybe_prio = guest.thread_state_mut().next_timeslice(&self.cfg);
786
787 // Depending on chaos mode, a received timer event is either a preemption or a changepoint
788 let req = if let Some(prio) = maybe_prio {
789 Self::priority_changepoint_request(guest, end_time, prio)
790 } else if chaos {
791 Self::random_priority_changepoint_request(guest, end_time)
792 } else if explicit_sched_yield && self.cfg.replay_schedule_from.is_none() {
793 Self::sched_yield_request(guest)
794 } else {
795 Self::yield_request(guest)
796 };
797 resource_request(guest, req).await;
798
799 // AUTONOMOUS-BOT-IMPLEMENTED
800 // TODO-HUMAN-REVIEW(PR-1151)
801 // Multiple scheduler commits can occur without an intervening
802 // conditional branch. Exact-RCB replay represents those as
803 // adjacent zero-RCB slices, which must be consumed before the
804 // guest resumes.
805 if !guest.thread_state().timeslice_expired() {
806 break;
807 }
808 explicit_sched_yield = false;
809 }
810 }
811
812 /// Hash the guest REGISTER FILE and log it.
813 ///
814 /// # The sampling boundary: GUEST-LOGICAL CONTROL, never handler interior
815 ///
816 /// This is called from exactly one place -- immediately after a syscall has finished and its
817 /// result has been written back, before the guest resumes. At that instant the guest
818 /// LOGICALLY HAS CONTROL: the architectural state is what the guest itself would observe at
819 /// its own RIP, and it is the same instant at which stack and heap are already hashed.
820 ///
821 /// It is deliberately NOT sampled anywhere inside a tool handler. A backend that runs its
822 /// handler IN-GUEST (sabre, liteinst, e9patch) executes instructions the ptrace reference
823 /// never executes, using guest registers as scratch while it does. Registers there
824 /// legitimately differ across backends, so comparing them would report correct behaviour as a
825 /// divergence and burn the prefix-depth ratchet on artifacts. Handler-interior state is out of
826 /// the domain, not excluded from it by a filter -- the same "define the domain" rule the heap
827 /// definition follows.
828 ///
829 /// # What is in the hash, and what is deliberately not
830 ///
831 /// Included: the general-purpose registers the guest can observe, `rip`, `rsp`, `rflags`,
832 /// `orig_rax`, and the TLS bases `fs_base`/`gs_base`.
833 ///
834 /// EXCLUDED, with reasons rather than by convenience:
835 /// * `rcx` and `r11` -- architecturally clobbered by the `SYSCALL` instruction, which stores
836 /// the return RIP and RFLAGS in them. They carry no information beyond `rip`/`eflags`, which
837 /// ARE hashed, and a patching backend that reaches the kernel by some route other than a
838 /// bare `SYSCALL` will leave different values there for a reason that is not a determinism
839 /// defect.
840 /// * The segment selectors `cs`/`ss`/`ds`/`es`/`fs`/`gs` -- constant for a 64-bit userspace
841 /// guest, so they add no signal; the TLS BASES are what a guest actually observes and those
842 /// are hashed.
843 fn detlog_registers<G: Guest<Self>>(
844 &self,
845 guest: &mut G,
846 regs: &libc::user_regs_struct,
847 seq: u64,
848 ) {
849 if !self.cfg.detlog_regs {
850 return;
851 }
852 // COST TIER: cadence 1 == full (every control point); N > 1 == spot-check every Nth.
853 // The cadence index is a PER-THREAD counter, NOT a shared one: a global atomic would be
854 // incremented in whatever order threads happen to reach it, so the cadence -- and
855 // therefore which points got sampled -- would itself be nondeterministic. A determinism
856 // instrument must not have a nondeterministic sampling schedule.
857 //
858 // It is `stats.syscall_count`, which starts at ZERO, rather than the syscall ORDINAL used
859 // in the log (which starts at 2). With the ordinal, a guest whose control points never
860 // land on a multiple of the cadence emitted NOTHING and the run still reported PASS -- a
861 // spot-tier green backed by zero samples. Indexing from zero makes the first control point
862 // of every thread always sampled, so a spot-tier run can never be silently empty.
863 let cadence = self.cfg.detlog_regs_cadence.max(1);
864 let index = {
865 let stats = &mut guest.thread_state_mut().stats;
866 let i = stats.regs_sample_index;
867 stats.regs_sample_index = i.saturating_add(1);
868 i
869 };
870 let _ = seq;
871 if !index.is_multiple_of(cadence) {
872 return;
873 }
874 let tier = if cadence == 1 {
875 "full".to_string()
876 } else {
877 format!("spot-1/{cadence}")
878 };
879 let mut bytes = Vec::with_capacity(19 * 8);
880 for v in [
881 regs.rax,
882 regs.rbx,
883 regs.rdx,
884 regs.rsi,
885 regs.rdi,
886 regs.rbp,
887 regs.rsp,
888 regs.r8,
889 regs.r9,
890 regs.r10,
891 regs.r12,
892 regs.r13,
893 regs.r14,
894 regs.r15,
895 regs.rip,
896 regs.eflags,
897 regs.orig_rax,
898 regs.fs_base,
899 regs.gs_base,
900 ] {
901 bytes.extend_from_slice(&v.to_le_bytes());
902 }
903 detlog!(
904 "[registers][dtid {}] control_point=syscall-exit tier={} {}",
905 guest.thread_state().dettid,
906 tier,
907 Digest::new(&bytes)
908 );
909 }
910
911 fn detlog_memory_maps<G: Guest<Self>>(&self, guest: &mut G) -> Result<(), reverie::Error> {
912 if !(self.cfg.detlog_stack || self.cfg.detlog_heap) {
913 // Don't incur the *significant* performance penalty for reading
914 // /proc/maps unless one of these flags is enabled.
915 return Ok(());
916 }
917 // ...and don't incur it when nothing would observe the record either.
918 //
919 // The hash below is an argument to `detlog!`, so `tracing` already skips
920 // it when INFO is disabled. Enumerating the maps is NOT: it happens
921 // before the macro, on every syscall, whether or not a record is ever
922 // written. Measured on a QEMU/Linux boot with `RUST_LOG` unset, where
923 // each run emitted 123 bytes of log in total: no flag 43.71s,
924 // `--detlog-stack` 190.37s (4.36x), `--detlog-heap` 207.90s (4.76x).
925 // `--detlog-regs` was already inert because everything it does before
926 // its own `detlog!` is trivial; this restores the same property here.
927 //
928 // Skipping is invisible to the guest: enumerating maps and hashing guest
929 // memory are host-side observations of the tracee that neither issue a
930 // guest syscall nor advance virtual time, so a run that skips them
931 // executes the same guest instruction stream as one that does not.
932 if !detlog_observed!() {
933 return Ok(());
934 }
935 // Out-of-process backends (e.g. KVM) report their guest memory regions
936 // directly, because `guest.pid()` is the host VMM process there and its
937 // `/proc/<pid>/maps` describes the VMM, not the guest address space.
938 // Reading those host addresses through `guest.memory()` (guest-address
939 // space) would fault and abort the syscall. When the backend supplies
940 // regions, hash those guest ranges; otherwise fall back to the ptrace
941 // path of parsing `/proc/<pid>/maps`.
942 if let Some(regions) = guest.detlog_memory_regions() {
943 for region in regions {
944 let want = match region.kind {
945 reverie::DetlogRegionKind::Stack => self.cfg.detlog_stack,
946 reverie::DetlogRegionKind::Heap => self.cfg.detlog_heap,
947 };
948 if !want {
949 continue;
950 }
951 let dettid = guest.thread_state().dettid;
952 detlog!(
953 "[memory][dtid {}] {:?} {:#x}-{:#x}->{}",
954 dettid,
955 region.kind,
956 region.start,
957 region.end,
958 procmaps::compute_hash_range(guest, region.start, region.end)?
959 )
960 }
961 return Ok(());
962 }
963 let mut labelled_heap = false;
964 for mmap in procmaps::from_pid(guest.pid(), |map| match map.pathname {
965 procmaps::MMapPath::Stack if self.cfg.detlog_stack => true,
966 procmaps::MMapPath::Heap if self.cfg.detlog_heap => true,
967 _ => false,
968 })? {
969 labelled_heap |= matches!(mmap.pathname, procmaps::MMapPath::Heap);
970 let dettid = guest.thread_state().dettid;
971 detlog!(
972 "[memory][dtid {}] {}->{}",
973 dettid,
974 procmaps::display(&mmap),
975 procmaps::compute_hash(guest, &mmap)?
976 )
977 }
978
979 // The kernel labels `[heap]` only for `[mm->start_brk, mm->brk)`. Under a
980 // backend that loads the guest itself (DynamoRIO) that break belongs to
981 // the loader, so the guest's heap is an unlabelled anonymous mapping and
982 // the filter above matches nothing. Emitting no record there is worse
983 // than a wrong one: downstream a zero-record heap comparison reads as
984 // "compared and matched" rather than "never measured". Fall back to the
985 // brk range Detcore observed, which identifies the heap on every backend.
986 if self.cfg.detlog_heap && !labelled_heap {
987 self.detlog_brk_heap(guest)?;
988 }
989 Ok(())
990 }
991
992 /// Emit the `[heap]` record from the observed program break, for backends
993 /// where the kernel does not label the guest's heap.
994 ///
995 /// Reports `[start_brk, brk)` -- the range Detcore observed -- taking the
996 /// non-address columns from the enclosing anonymous mapping, so the record
997 /// is textually comparable with the labelled record another backend
998 /// produces for the same guest.
999 ///
1000 /// ⚠️ THE EXTENT IS THE OBSERVED BREAK, NOT THE MAPPING THAT CONTAINS IT,
1001 /// and the two are not the same region. The selector below admits any
1002 /// anonymous mapping with `address.0 <= start && end <= address.1`, which
1003 /// is a SUPERSET by construction. Reporting that mapping instead would make
1004 /// the comparability claim above true only when the arena happens to
1005 /// coincide with the break, and would fold non-heap bytes into the digest.
1006 /// On the backend this path exists for, the premise is that the LOADER owns
1007 /// the break, so those extra bytes are loader arena -- the least
1008 /// reproducible region in the process. A silently empty record would become
1009 /// a loudly divergent one, for a reason that is not the guest's heap.
1010 ///
1011 /// The labelled path above hashes its mapping directly, which is correct
1012 /// there because the kernel defines `[heap]` as exactly `[start_brk, brk)`.
1013 /// Both paths therefore report the same quantity.
1014 fn detlog_brk_heap<G: Guest<Self>>(&self, guest: &mut G) -> Result<(), reverie::Error> {
1015 let Some((start, end)) = guest
1016 .thread_state()
1017 .memory_metadata
1018 .lock()
1019 .expect("memory metadata mutex poisoned")
1020 .brk_heap_range()
1021 else {
1022 return Ok(());
1023 };
1024 let enclosing = procmaps::from_pid(guest.pid(), |map| {
1025 matches!(map.pathname, procmaps::MMapPath::Anonymous)
1026 && map.address.0 <= start
1027 && end <= map.address.1
1028 })?;
1029 let Some(mmap) = enclosing.into_iter().next() else {
1030 return Ok(());
1031 };
1032 let dettid = guest.thread_state().dettid;
1033 detlog!(
1034 "[memory][dtid {}] {}->{}",
1035 dettid,
1036 procmaps::display_range_as(&mmap, start, end, "[heap]"),
1037 procmaps::compute_hash_range(guest, start, end)?
1038 );
1039 Ok(())
1040 }
1041}
1042
1043/// Render a finished syscall for the `finish syscall` DETLOG line.
1044///
1045/// Output pointers are dereferenced only when the kernel can have written
1046/// them. Linux leaves the output buffer of the syscalls listed in
1047/// [`failure_leaves_outputs_unwritten`] untouched when they fail with any errno
1048/// other than EFAULT, so rendering it would publish whatever the guest
1049/// happened to have there -- typically uninitialized stack -- as if it were a
1050/// result, and two otherwise identical runs would diverge on it
1051/// (<https://github.com/rrnewton/hermit/issues/3153>). For those syscalls such
1052/// a failure renders the pointer arguments without their pointees; the errno
1053/// is printed next to this rendering from the result itself.
1054///
1055/// EFAULT keeps the outputs rendered, because it is also the error of a copy
1056/// that faulted part way: `copy_to_user()` may already have stored a prefix of
1057/// the struct, which the guest can read and DETLOG must show. A tool error is
1058/// not a guest errno and proves nothing about the buffer, so it keeps them
1059/// rendered too.
1060///
1061/// Every other syscall keeps rendering its outputs on failure, because some
1062/// Linux syscalls do write an output on an error return (for example
1063/// `gettimeofday` stores `tv` before it faults on a bad `tz`). Hiding such an
1064/// output would remove real evidence from DETLOG.
1065fn display_syscall_finished<'a, M: MemoryAccess>(
1066 syscall: &'a Syscall,
1067 memory: &'a M,
1068 result: &Result<i64, Error>,
1069) -> reverie::syscalls::Display<'a, M, Syscall> {
1070 match syscall {
1071 Syscall::Fstat(_) => syscall.display(memory), //FIXME: T136880615 - fstat structure isn't fully deterministic yet
1072 _ if failure_proves_outputs_unwritten(result)
1073 && failure_leaves_outputs_unwritten(syscall) =>
1074 {
1075 syscall.display(memory)
1076 }
1077 _ => syscall.display_with_outputs(memory),
1078 }
1079}
1080
1081/// Syscalls whose output buffer Linux does not write when the syscall fails
1082/// with an errno other than EFAULT, restricted to those whose output the pinned Reverie formatter dereferences.
1083///
1084/// This is deliberately a list of syscalls PROVEN not to write on failure,
1085/// rather than a list of exceptions that do: a syscall missing from it keeps
1086/// its outputs rendered, which can at worst show stale guest memory, whereas a
1087/// syscall wrongly missing from an exception list would silently hide a value
1088/// the kernel really wrote.
1089///
1090/// Kernel behaviour (Linux `fs/stat.c`, `kernel/time/posix-timers.c`):
1091///
1092/// - `stat`, `lstat`, `fstat`, `newfstatat`: `vfs_stat()`, `vfs_lstat()`,
1093/// `vfs_fstat()` and `vfs_fstatat()` return their error before
1094/// `cp_new_stat()` is called, so the `struct stat` is never copied out.
1095/// Hermit's `handle_stat_family` likewise returns the injected syscall's
1096/// error before it rewrites the buffer.
1097/// - `statx`: `do_statx()` returns the `vfs_statx()` error before
1098/// `cp_statx()` copies the `struct statx` out.
1099/// - `clock_gettime`: `put_timespec64()` runs only when
1100/// `kc->clock_get_timespec()` succeeded (`if (!error && put_timespec64(..))`),
1101/// and an unknown clock returns `-EINVAL` before that. Hermit's
1102/// `handle_clock_gettime` writes `tp` only on its success path.
1103///
1104/// In every case the one failure that follows a copy attempt is the `-EFAULT`
1105/// from the copy itself, where `copy_to_user()` may have stored a prefix of
1106/// the struct before faulting. [`failure_proves_outputs_unwritten`] therefore
1107/// keeps rendering EFAULT failures, so that partial prefix stays visible.
1108///
1109/// `gettimeofday` is intentionally absent: `SYSCALL_DEFINE2(gettimeofday)` in
1110/// `kernel/time/time.c` stores `tv` and then returns `-EFAULT` if copying
1111/// `tz` faults, so its rendered output can be kernel-written on failure.
1112fn failure_leaves_outputs_unwritten(syscall: &Syscall) -> bool {
1113 matches!(
1114 syscall,
1115 Syscall::Stat(_)
1116 | Syscall::Lstat(_)
1117 | Syscall::Fstat(_)
1118 | Syscall::Newfstatat(_)
1119 | Syscall::Statx(_)
1120 | Syscall::ClockGettime(_)
1121 )
1122}
1123
1124/// Whether `result` is a failure that, for a syscall listed in
1125/// [`failure_leaves_outputs_unwritten`], proves Linux stored no output: a
1126/// guest errno other than EFAULT. EFAULT can follow a partial copy, and a
1127/// tool error is not the syscall's result.
1128fn failure_proves_outputs_unwritten(result: &Result<i64, Error>) -> bool {
1129 matches!(result, Err(Error::Errno(errno)) if *errno != Errno::EFAULT)
1130}
1131
1132#[reverie::tool]
1133impl<T: RecordOrReplay> Tool for Detcore<T> {
1134 type GlobalState = GlobalState;
1135 type ThreadState = ThreadState<T::ThreadState>;
1136
1137 fn observe_signal_dequeues(config: &Config) -> bool {
1138 config.kvm_shared_dequeue_timers && config.sequentialize_threads && config.backend_is_kvm
1139 }
1140
1141 async fn handle_signal_dequeue<G: Guest<Self>>(
1142 &self,
1143 guest: &mut G,
1144 dequeue: reverie::SignalDequeue,
1145 ) -> Result<(), Errno> {
1146 tool_global::signal_dequeued(guest, dequeue).await
1147 }
1148
1149 /// Constructor for Detcore process-local state.
1150 fn new(pid: Pid, cfg: &Config) -> Self {
1151 let detpid = DetPid::from_raw(pid.into()); // TODO(T78538674): virtualize pid.
1152 cfg.validate_invariants();
1153 Self {
1154 detpid,
1155 cfg: cfg.clone(),
1156 record_or_replay: T::new(pid, cfg),
1157 }
1158 }
1159
1160 /// NOTE: these subscriptions are used ONLY for hermit run mode. Hermit record has its own
1161 /// subscriptions specified in recorder/mod.rs.
1162 fn subscriptions(config: &Config) -> Subscription {
1163 let do_sched =
1164 config.sched_heuristic != SchedHeuristic::None || config.sequentialize_threads;
1165
1166 if !config.passthru_opt {
1167 // Fail closed by default in every build profile. Besides allowing syscall-specific
1168 // handlers to run, interception is what charges generic syscall logical time.
1169 //
1170 // `cpuid` is the one exception. With CPUID virtualization off the handler only
1171 // re-executes the host instruction, so trapping does not change what the guest
1172 // sees, but the subscription makes Reverie probe and enable CPUID faulting and log
1173 // an ERROR for each exec on hosts that lack it
1174 // (https://github.com/rrnewton/hermit/issues/3460). A backend that installs its
1175 // own CPUID table (KVM) still needs the trap for the guest to see host values.
1176 let mut subscription = Subscription::all_syscalls();
1177 subscription.rdtsc();
1178 if config.virtualize_cpuid || config.cpuid_virtualized_by_backend {
1179 subscription.cpuid();
1180 }
1181 subscription
1182 } else {
1183 // Explicit performance opt-in: unlisted syscalls bypass Detcore entirely. Keep this
1184 // path separate so its allow-list can be tightened without weakening the default.
1185 let mut subscription = Subscription::none();
1186 subscription.syscalls([
1187 Sysno::write,
1188 // AUTONOMOUS-BOT-IMPLEMENTED
1189 // TODO-HUMAN-REVIEW(#547)
1190 Sysno::writev,
1191 // Timer-slack procfs virtualization must see every scalar and
1192 // vectored read/write form that Linux accepts for the file.
1193 Sysno::readv,
1194 Sysno::preadv,
1195 Sysno::preadv2,
1196 Sysno::pwritev,
1197 Sysno::pwritev2,
1198 // AUTONOMOUS-BOT-IMPLEMENTED
1199 // TODO-HUMAN-REVIEW(#683)
1200 Sysno::pwrite64,
1201 Sysno::openat,
1202 Sysno::open,
1203 Sysno::creat,
1204 Sysno::close,
1205 Sysno::read,
1206 Sysno::pread64,
1207 Sysno::lseek,
1208 Sysno::fadvise64,
1209 Sysno::mmap,
1210 Sysno::madvise,
1211 Sysno::munmap,
1212 Sysno::mremap,
1213 Sysno::fcntl,
1214 Sysno::arch_prctl,
1215 // AUTONOMOUS-BOT-IMPLEMENTED
1216 // TODO-HUMAN-REVIEW(PR-2150): Timer-slack prctl state is
1217 // virtual and must not bypass Detcore under passthru_opt.
1218 Sysno::prctl,
1219 Sysno::ioctl,
1220 Sysno::futex,
1221 Sysno::clone,
1222 Sysno::clone3,
1223 Sysno::fork,
1224 Sysno::vfork,
1225 Sysno::wait4,
1226 Sysno::waitid,
1227 Sysno::setsid,
1228 Sysno::uname,
1229 Sysno::exit_group,
1230 Sysno::exit,
1231 // AUTONOMOUS-BOT-IMPLEMENTED
1232 // The scheduler must observe changes to the address used for
1233 // its modeled CHILD_CLEARTID wake, including record/replay's
1234 // passthrough optimization.
1235 Sysno::set_tid_address,
1236 // AUTONOMOUS-BOT-IMPLEMENTED
1237 // Rare (once per thread) but load-bearing: without it the exit
1238 // hook cannot replay `exit_robust_list()` and robust-mutex
1239 // waiters are never woken.
1240 Sysno::set_robust_list,
1241 Sysno::dup,
1242 Sysno::dup2,
1243 Sysno::dup3,
1244 Sysno::pipe,
1245 Sysno::pipe2,
1246 Sysno::getrandom,
1247 Sysno::utime,
1248 Sysno::utimes,
1249 Sysno::utimensat,
1250 Sysno::futimesat,
1251 Sysno::socket,
1252 Sysno::socketpair,
1253 Sysno::eventfd,
1254 Sysno::eventfd2,
1255 Sysno::sched_getaffinity,
1256 Sysno::sched_setaffinity,
1257 Sysno::signalfd,
1258 Sysno::signalfd4,
1259 Sysno::timerfd_create,
1260 Sysno::timerfd_settime,
1261 Sysno::timerfd_gettime,
1262 Sysno::inotify_init,
1263 Sysno::inotify_init1,
1264 Sysno::inotify_add_watch,
1265 Sysno::inotify_rm_watch,
1266 Sysno::memfd_create,
1267 // AUTONOMOUS-BOT-IMPLEMENTED
1268 // TODO-HUMAN-REVIEW(PR-862): Keep modeled pidfd creation intercepted.
1269 Sysno::pidfd_open,
1270 // AUTONOMOUS-BOT-IMPLEMENTED
1271 // TODO-HUMAN-REVIEW(PR-1175): Keep pidfd signal/get-fd
1272 // determinization from being bypassed under the passthru opt-in.
1273 Sysno::pidfd_send_signal,
1274 Sysno::pidfd_getfd,
1275 Sysno::userfaultfd,
1276 Sysno::io_uring_setup,
1277 Sysno::io_uring_enter,
1278 Sysno::io_uring_register,
1279 Sysno::accept,
1280 Sysno::accept4,
1281 Sysno::nanosleep,
1282 Sysno::clock_nanosleep,
1283 Sysno::sched_yield,
1284 Sysno::poll,
1285 Sysno::ppoll,
1286 Sysno::prlimit64,
1287 Sysno::epoll_create,
1288 Sysno::epoll_create1,
1289 Sysno::epoll_ctl,
1290 Sysno::epoll_pwait,
1291 Sysno::epoll_wait,
1292 Sysno::epoll_wait_old,
1293 Sysno::epoll_ctl_old,
1294 Sysno::recvfrom,
1295 Sysno::rt_sigsuspend,
1296 Sysno::rt_sigtimedwait,
1297 Sysno::execve,
1298 Sysno::execveat,
1299 Sysno::rseq,
1300 Sysno::getpid,
1301 Sysno::gettid,
1302 Sysno::getcpu,
1303 Sysno::rt_sigprocmask,
1304 Sysno::rt_sigaction,
1305 Sysno::getrusage,
1306 Sysno::sysinfo,
1307 // AUTONOMOUS-BOT-IMPLEMENTED
1308 // TODO-HUMAN-REVIEW(#686): Review scratch fd sets and scheduler polling.
1309 Sysno::pselect6,
1310 // `select` is included by the Determinized sweep below;
1311 // T137258824 tracks improving its implementation.
1312 ]);
1313
1314 if do_sched {
1315 subscription.syscalls([
1316 // TODO: some of the above could probably move to this bucket.
1317 Sysno::alarm,
1318 Sysno::pause,
1319 ]);
1320 }
1321
1322 if config.virtualize_metadata {
1323 subscription.syscalls([
1324 Sysno::getdents,
1325 Sysno::getdents64,
1326 Sysno::stat,
1327 Sysno::lstat,
1328 Sysno::fstat,
1329 Sysno::newfstatat,
1330 Sysno::statx,
1331 ]);
1332 }
1333
1334 if true
1335 // TODO: could introduce a flag for this:
1336 /* config.virtualize_keys */
1337 {
1338 subscription.syscalls([Sysno::add_key, Sysno::request_key, Sysno::keyctl]);
1339 }
1340
1341 if do_sched {
1342 subscription.syscall(Sysno::connect);
1343 }
1344 if do_sched || config.warn_non_zero_binds {
1345 subscription.syscall(Sysno::bind);
1346 }
1347
1348 if config.warn_non_zero_binds {
1349 subscription.syscall(Sysno::bind);
1350 }
1351
1352 if config.virtualize_time {
1353 subscription.rdtsc();
1354 subscription.syscalls([
1355 Sysno::gettimeofday,
1356 Sysno::time,
1357 Sysno::clock_gettime,
1358 Sysno::clock_getres,
1359 ]);
1360 }
1361
1362 if config.virtualize_cpuid {
1363 subscription.cpuid();
1364 }
1365
1366 // AUTONOMOUS-BOT-IMPLEMENTED
1367 // TODO-HUMAN-REVIEW(PR-978): Keep the passthru-opt allow-list in sync
1368 // with the complete Determinized classification AND with the
1369 // Unsupported set. Even under the performance opt-in, Detcore must
1370 // see every syscall that its audited policy says it models or
1371 // deterministically refuses. Otherwise record/replay can bypass a
1372 // Detcore handler and execute the syscall natively against the host.
1373 //
1374 // `Determinized` and `Unsupported` are disjoint classifications, so
1375 // the determinized filter alone leaves the Unsupported set
1376 // unsubscribed: record/replay would silently execute an unsupported
1377 // syscall on the live host instead of invalidating its determinism
1378 // claim. Both halves are required; neither implies the other.
1379 // NOTE: `all_pinned_syscalls()`, not `Sysno::iter()`. The latter
1380 // stops one short of the end of the table, which silently dropped
1381 // `lsm_list_modules` out of this sweep.
1382 subscription.syscalls(crate::all_pinned_syscalls().filter(|sysno| {
1383 crate::is_determinized_syscall(*sysno) || crate::is_unsupported_syscall(*sysno)
1384 }));
1385
1386 // Make sure we also intercept everything that the record-or-replay tool
1387 // wants.
1388 subscription | T::subscriptions(config)
1389 }
1390 }
1391
1392 async fn handle_cpuid_event<G: Guest<Self>>(
1393 &self,
1394 guest: &mut G,
1395 eax: u32,
1396 ecx: u32,
1397 ) -> Result<CpuIdResult, Errno> {
1398 trace!("handle_cpuid_event: eax: {}, ecx: {}", eax, ecx);
1399 self.pre_handler_hook(guest, false).await;
1400 let res = if self.cfg.virtualize_cpuid {
1401 let dettid = guest.thread_state().dettid;
1402 let time = &mut guest.thread_state_mut().thread_logical_time;
1403 let intercepted = cpuid::InterceptedCpuid::new();
1404 time.add_cpuid();
1405 let nanos = time.as_nanos();
1406 trace!(
1407 "[dtid {}] inbound cpuid, new logical time: {:?}",
1408 dettid, time
1409 );
1410 if self.cfg.should_trace_schedevent() {
1411 trace_schedevent(
1412 guest,
1413 SchedEvent {
1414 dettid,
1415 op: Op::Cpuid,
1416 count: 1,
1417 start_rip: None,
1418 end_rip: None,
1419 end_time: Some(nanos),
1420 },
1421 true,
1422 )
1423 .await;
1424 }
1425 intercepted.cpuid(eax, ecx).unwrap_or_else(|| {
1426 warn!(
1427 "[dtid {}] cpuid leaf 0x{:x} subleaf 0x{:x} not in deterministic table; returning zero result",
1428 dettid, eax, ecx
1429 );
1430 CpuIdResult {
1431 eax: 0,
1432 ebx: 0,
1433 ecx: 0,
1434 edx: 0,
1435 }
1436 })
1437 } else {
1438 cpuid!(eax, ecx)
1439 };
1440 self.post_handler_hook(guest).await;
1441 Ok(res)
1442 }
1443
1444 async fn handle_rdtsc_event<G: Guest<Self>>(
1445 &self,
1446 guest: &mut G,
1447 request: Rdtsc,
1448 ) -> Result<RdtscResult, Errno> {
1449 trace!("handle_rdtsc_event: {:?}", request);
1450 self.pre_handler_hook(guest, false).await;
1451 let result = if guest.config().virtualize_time {
1452 let dettid = guest.thread_state().dettid;
1453 guest.thread_state_mut().thread_logical_time.add_rdtsc();
1454 info!(
1455 "[dtid {}] inbound rdtsc, new logical time: {:?}",
1456 dettid,
1457 guest.thread_state().thread_logical_time
1458 );
1459 if self.cfg.should_trace_schedevent() {
1460 let ev = with_guest_time(
1461 guest,
1462 SchedEvent {
1463 dettid,
1464 op: Op::Rdtsc,
1465 count: 1,
1466 start_rip: None,
1467 end_rip: None,
1468 end_time: None,
1469 },
1470 );
1471 trace_schedevent(guest, ev, true).await;
1472 }
1473 // The guest TSC must name the same instant as `clock_gettime`. Both
1474 // now read the coordinator's clock through the shared per-process
1475 // floor, so a guest comparing the two -- a clocksource watchdog, a
1476 // delay loop calibrated against a device timer, a second vCPU
1477 // reading the TSC -- sees one time base instead of two.
1478 //
1479 // The `add_rdtsc()` charge above is folded into global time by the
1480 // same RPC that reads it back (`send_and_update_time` updates the
1481 // coordinator before dispatching the request), so consecutive reads
1482 // from one thread still advance.
1483 let tsc = guest_clock_time(guest).await;
1484 Ok(RdtscResult {
1485 // We treat virtual cycles as equivalent to virtual nanoseconds.
1486 tsc: tsc.as_nanos(),
1487 aux: None,
1488 })
1489 } else {
1490 self.record_or_replay
1491 .handle_rdtsc_event(&mut guest.into_guest(), request)
1492 .await
1493 };
1494 self.post_handler_hook(guest).await;
1495 result
1496 }
1497
1498 // Note: we will not see SIGSTKFLT used for timers.
1499 async fn handle_signal_event<G: Guest<Self>>(
1500 &self,
1501 guest: &mut G,
1502 signal: Signal,
1503 ) -> Result<Option<Signal>, Errno> {
1504 if signal == Signal::SIGINT && self.cfg.sigint_instakill {
1505 warn!("Fatal: Exiting hermit container immediately upon SIGINT");
1506 // ⚠️ NOT A REFUSAL. The operator interrupted the run; hermit examined
1507 // nothing and decided nothing. Reporting `128 + SIGINT` is what every
1508 // other tool reports for this, and it keeps 122 meaning one thing.
1509 unrecoverable_shutdown(guest, detcore_model::HERMIT_SIGINT_DEATH_EXIT).await
1510 } else {
1511 self.pre_handler_hook(guest, false).await;
1512
1513 let dettid = guest.thread_state().dettid;
1514 let mycount = guest.thread_state().stats.signal_count;
1515 info!(
1516 "[dtid {}] handling inbound signal (#{}) {}",
1517 dettid, mycount, signal
1518 );
1519 guest.thread_state_mut().stats.count_signal();
1520 let time = &guest.thread_state().thread_logical_time;
1521 let nanos = time.as_nanos();
1522
1523 if self.cfg.sequentialize_threads && self.cfg.should_trace_schedevent() {
1524 trace_schedevent(
1525 guest,
1526 SchedEvent {
1527 dettid,
1528 op: Op::SignalReceived(signal.into()),
1529 count: 1,
1530 start_rip: None,
1531 end_rip: None,
1532 end_time: Some(nanos),
1533 },
1534 true,
1535 )
1536 .await;
1537 }
1538
1539 let request = guest.thread_state_mut().mk_request(
1540 ResourceID::InboundSignal(SigWrapper::from(signal)),
1541 Permission::RW,
1542 );
1543 resource_request(guest, request).await;
1544 // A delivered signal may be caught, ignored, or terminate the
1545 // thread group. Reading the lists is harmless; the collected
1546 // wakes remain inert unless the ptrace exit callback later reports
1547 // that this exact signal caused physical exit.
1548 self.stage_thread_group_robust_list_wakes(
1549 guest,
1550 tool_local::RobustListExit::Signal(signal as i32),
1551 )
1552 .await;
1553 info!(
1554 "[dtid {}] finish delivering signal (#{}) {}",
1555 dettid, mycount, signal
1556 );
1557
1558 self.post_handler_hook(guest).await;
1559 Ok(Some(signal))
1560 }
1561 }
1562
1563 fn init_thread_state(
1564 &self,
1565 tid: Tid,
1566 parent: Option<(Tid, &Self::ThreadState)>,
1567 ) -> Self::ThreadState {
1568 trace!("[tid {}] detcore init new thread state", tid);
1569
1570 let record_or_replay = self
1571 .record_or_replay
1572 .init_thread_state(tid, parent.map(|(ptid, ts)| (ptid, ts.as_ref())));
1573
1574 // TODO(T78538674): virtualize tid, extend tid<=>dettid mapping here.
1575 match parent {
1576 None => ThreadState::new(DetPid::from_raw(tid.into()), &self.cfg, record_or_replay),
1577 Some(pts) => {
1578 let clone_flags = pts
1579 .1
1580 .clone_flags
1581 .expect("clone_flags must be set by parent");
1582 let dettid = DetPid::from_raw(tid.into());
1583
1584 // If we had mutable access to the parent state, we could update it here, but
1585 // instead we leave that to the clone/fork handling.
1586 let (_next_parent_pedigree, child_pedigree) = pts.1.pedigree.fork();
1587 let child_logical_time = pts.1.thread_logical_time.clone_for_child();
1588 let last_accounted_user_time = child_logical_time.user_cpu_time();
1589 let last_accounted_system_time = child_logical_time.system_cpu_time();
1590 if !clone_flags.contains(CloneFlags::CLONE_THREAD) {
1591 pts.1.prepare_child_process_cpu_time(dettid);
1592 }
1593 let guest_clock = Arc::clone(&pts.1.guest_clock);
1594
1595 ThreadState {
1596 dettid,
1597 detpid: None, // Initialized later.
1598 thread_start_entered: false,
1599 physical_tid: None,
1600 signal_task_identity: None,
1601 open_file_creator: None,
1602 mm_id: MmId::for_clone(
1603 pts.1.mm_id,
1604 dettid,
1605 clone_flags.contains(CloneFlags::CLONE_VM),
1606 ),
1607 memory_metadata: if clone_flags.contains(CloneFlags::CLONE_VM) {
1608 Arc::clone(&pts.1.memory_metadata)
1609 } else {
1610 Arc::new(Mutex::new(
1611 pts.1
1612 .memory_metadata
1613 .lock()
1614 .expect("memory metadata mutex poisoned")
1615 .clone(),
1616 ))
1617 },
1618 pedigree: child_pedigree.clone(),
1619 initialized_random_auxv: None,
1620 stats: ThreadStats::new(),
1621 file_metadata: {
1622 debug!(
1623 "[init_thread-state, parent dtid = {}] child thread {}, clone_flags = {:x?}",
1624 pts.0, tid, clone_flags
1625 );
1626 if clone_flags.contains(CloneFlags::CLONE_FILES) {
1627 pts.1.file_metadata.clone()
1628 } else {
1629 Arc::new(Mutex::new(
1630 pts.1.file_metadata.lock().unwrap().fork_for(dettid),
1631 ))
1632 }
1633 },
1634 discover_live_file_metadata: pts.1.discover_live_file_metadata,
1635 // Linux copies the creating thread's current timer slack
1636 // into both fields of every new task (thread or process).
1637 timer_slack_ns: pts.1.timer_slack_ns,
1638 default_timer_slack_ns: pts.1.timer_slack_ns,
1639 // POSIX timers are shared among threads of a process but are
1640 // NOT inherited across fork(2). Share the table for a new
1641 // thread (CLONE_THREAD); give a new process a fresh, empty
1642 // one.
1643 posix_timers: if clone_flags.contains(CloneFlags::CLONE_THREAD) {
1644 Arc::clone(&pts.1.posix_timers)
1645 } else {
1646 Arc::new(Mutex::new(PosixTimers::default()))
1647 },
1648 // Resource limits are process state: threads share them,
1649 // while a forked process inherits a snapshot.
1650 resource_limits: if clone_flags.contains(CloneFlags::CLONE_THREAD) {
1651 Arc::clone(&pts.1.resource_limits)
1652 } else {
1653 Arc::new(Mutex::new(
1654 pts.1
1655 .resource_limits
1656 .lock()
1657 .expect("resource limits mutex poisoned")
1658 .clone(),
1659 ))
1660 },
1661 process_cpu_time: if clone_flags.contains(CloneFlags::CLONE_THREAD) {
1662 Arc::clone(&pts.1.process_cpu_time)
1663 } else {
1664 Arc::new(Mutex::new(ProcessCpuTime::default()))
1665 },
1666 // Wall time belongs to the traced process tree, not to an
1667 // individual process. Forked processes and cloned threads
1668 // therefore retain one monotonic view of raw logical time.
1669 guest_clock,
1670 parent_process_cpu_time: if clone_flags.contains(CloneFlags::CLONE_THREAD) {
1671 pts.1.parent_process_cpu_time.clone()
1672 } else {
1673 Some(Arc::clone(&pts.1.process_cpu_time))
1674 },
1675 last_accounted_user_time,
1676 last_accounted_system_time,
1677 thread_cpu_start_user_time: last_accounted_user_time,
1678 thread_cpu_start_system_time: last_accounted_system_time,
1679 clone_flags: None,
1680 pending_vfork: pts.1.pending_vfork.clone(),
1681
1682 // Child RNG identity follows the deterministic creation
1683 // pedigree, never the backend/host Tid. Guest-visible IDs
1684 // and scheduler targeting continue to use `dettid`.
1685 prng: tool_local::thread_rng_from_parent_pedigree(
1686 "USER RAND",
1687 &pts.1.prng,
1688 &child_pedigree,
1689 tool_local::ChildRngStream::User,
1690 ),
1691 chaos_prng: tool_local::thread_rng_from_parent_pedigree(
1692 "CHAOSRAND",
1693 &pts.1.chaos_prng,
1694 &child_pedigree,
1695 tool_local::ChildRngStream::Chaos,
1696 ),
1697
1698 // For comparing progress to other threads, it is important that our
1699 // child thread start at a sensible place, rather than starting back
1700 // at zero:
1701 thread_logical_time: child_logical_time,
1702 // A new thread gets a new clock, so we've committed 0 ticks
1703 committed_clock_value: 0,
1704 // A new thread or process is never the backend runtime's
1705 // bootstrapping thread, so it starts outside any window.
1706 uncharged_bootstrap_syscalls: 0,
1707 in_uncharged_bootstrap_syscall: false,
1708
1709 end_of_timeslice: None,
1710 replay_rcb_end: None,
1711 // AUTONOMOUS-BOT-IMPLEMENTED
1712 // TODO-HUMAN-REVIEW(PR-1151)
1713 chaos_epoch: tool_local::chaos_epoch_sentinel(),
1714 chaos_slowdown_factor: RcbTimeMultiplier::ONE,
1715 chaos_slowdown_active: false,
1716 pending_chaos_epochs: Vec::new(),
1717 max_timeslice_end: None,
1718 last_rcb_timer: None,
1719 last_rcb_timer_is_max: false,
1720
1721 record_or_replay,
1722 preemption_points: None,
1723
1724 // We only get to the point of creating child threads if we're past the first execve.
1725 past_global_first_execve: true,
1726 interrupt_at: self.cfg.interrupts_for_thread(dettid),
1727
1728 // `copy_process()` sets `p->robust_list = NULL` for every
1729 // new task, thread or process alike. The child re-registers
1730 // its own head before it can own a robust futex.
1731 robust_list_head: None,
1732 robust_list_process: if clone_flags.contains(CloneFlags::CLONE_THREAD) {
1733 Arc::clone(&pts.1.robust_list_process)
1734 } else {
1735 Arc::new(Mutex::new(tool_local::RobustListProcessState::default()))
1736 },
1737 }
1738 }
1739 }
1740 }
1741
1742 async fn handle_thread_start<G: Guest<Self>>(&self, guest: &mut G) -> Result<(), Error> {
1743 guest.thread_state_mut().thread_start_entered = true;
1744 let new_dettid = DetTid::from_raw(guest.tid().into()); // TODO(T78538674): virtualize pid/tid:
1745 assert_eq!(new_dettid, guest.thread_state().dettid);
1746 let detpid = select_thread_start_detpid(guest.thread_state().detpid, guest.pid());
1747 let is_root_thread = is_root_thread_start(guest.is_root_process(), new_dettid, detpid);
1748 trace!(
1749 "[tid {}] detcore handle_thread_start, pid={}",
1750 guest.tid(),
1751 detpid
1752 );
1753
1754 // Delayed initialization of thread_state for this new thread:
1755 let thread_state = guest.thread_state_mut();
1756 thread_state.detpid = Some(detpid);
1757 if thread_state.recover_process_mm_id(detpid) {
1758 debug!(
1759 "[detcore, dtid {}] recovered process memory identity {} for unparented thread state",
1760 new_dettid, detpid
1761 );
1762 }
1763
1764 if let Some(vfork) = guest.thread_state_mut().pending_vfork.take() {
1765 create_vfork_child_thread(guest, new_dettid, vfork).await;
1766 } else if is_root_thread {
1767 // There is no fork event to catch for the root thread.
1768 debug!(
1769 "[detcore, dtid {}] root thread start, scheduling.. full config:\n {:?}",
1770 &new_dettid,
1771 guest.config()
1772 );
1773 let physical_ids = if guest
1774 .config()
1775 .backend_requires_thread_directed_process_signals
1776 {
1777 Some((
1778 guest.pid().as_raw(),
1779 guest
1780 .thread_state()
1781 .physical_tid
1782 .expect("backend requires a host thread ID before registration"),
1783 ))
1784 } else {
1785 None
1786 };
1787 if let Some(post_exec_mm) =
1788 create_child_thread(guest, new_dettid, 0, None, libc::SIGCHLD, physical_ids).await
1789 {
1790 guest.thread_state_mut().mm_id = post_exec_mm;
1791 }
1792 }
1793
1794 // Except for the root task, let's block until it's our turn to go:
1795 let th = tool_global::thread_start_request(&self.cfg, guest, detpid).await;
1796
1797 // Finish the delayed initialization of the full threadstate:
1798 {
1799 let ts = guest.thread_state_mut();
1800 ts.preemption_points = th.map(|x| x.into_iter());
1801 ts.next_timeslice(&self.cfg); // Must be after preemption_points is set.
1802 }
1803
1804 // The prehook is a noop for a thread just starting. Can't end the timeslice. There's no
1805 // RCB progress to record. However, we call it for consistency with all the other handlers.
1806 self.pre_handler_hook(guest, true).await;
1807 // ^ precise_branch=true: There should have been ZERO prior instructions before this,
1808 // because the thread hasn't done anything yet.
1809
1810 self.record_or_replay
1811 .handle_thread_start(&mut guest.into_guest())
1812 .await?;
1813
1814 self.post_handler_hook(guest).await;
1815 Ok(())
1816 }
1817
1818 async fn handle_post_exec<G: Guest<Self>>(&self, guest: &mut G) -> Result<(), Errno> {
1819 // A nonleader exec preserves the survivor's state and PMU counter, but
1820 // Linux changes its TID. Bind that state before any image callback RPC.
1821 match tool_global::reconnect_exec(guest).await {
1822 Err(Errno::EOPNOTSUPP) => {
1823 let message = "unsupported: preemption recording and replay across nonleader exec";
1824 error!("{message}");
1825 // Report even when logging is disabled or redirected. The
1826 // ordinary refusal helper preserves the policy exit class;
1827 // reconnect already bound identity, so its diagnostic RPC is
1828 // authenticated and cannot silently retire an unbound owner.
1829 let _ = writeln!(crate::util::RetryingStderr, "{message}");
1830 tool_global::unrecoverable_shutdown(
1831 guest,
1832 detcore_model::HERMIT_POLICY_REFUSAL_EXIT,
1833 )
1834 .await;
1835 }
1836 result => result?,
1837 }
1838 guest.thread_state_mut().past_global_first_execve = true;
1839 // Only a successful exec reaches this callback. Delete the old image's
1840 // POSIX timer IDs while exec still owns its scheduler turn; the global
1841 // notification below cancels their deadlines before the pre-handler
1842 // can yield. alarm()/ITIMER_REAL and the continuous clock survive exec.
1843 guest
1844 .thread_state()
1845 .posix_timers
1846 .lock()
1847 .unwrap()
1848 .clear_for_exec();
1849 // A successful exec clears the kernel's clear_child_tid registration.
1850 // Mirror that reset before any replacement-image syscall can run.
1851 if guest.config().sequentialize_threads {
1852 tool_global::set_child_tid_address(guest, 0).await;
1853 }
1854
1855 tool_global::mark_past_first_execve(guest).await;
1856 self.pre_handler_hook(guest, false).await;
1857
1858 let auxv = guest.auxv();
1859 let initialized = guest
1860 .thread_state_mut()
1861 .complete_initial_random_auxv(auxv.at_random().map(|p| p.as_raw()))
1862 .expect("authenticated initial random handoff no longer matches this image");
1863 // The successful early write already emitted the ordinary auxv INFO
1864 // record. Consume only its fact here: no second draw, write or event.
1865 if !initialized && let Some(ptr) = auxv.at_random() {
1866 // It is safe to mutate this address since libc has not yet had a
1867 // chance to modify or copy the auxv table.
1868 let memory = guest.memory();
1869 let dettid = guest.thread_state().dettid;
1870 let ptr = unsafe { ptr.into_mut() };
1871 random::initialize_auxv(
1872 guest.thread_state_mut().thread_prng(),
1873 memory,
1874 ptr.cast(),
1875 dettid,
1876 )?;
1877 }
1878
1879 // Successful exec never returns through handle_syscall_event, so the
1880 // nested recorder/replayer needs this callback to commit or retire its
1881 // pending exec state before the replacement image issues another exec.
1882 self.record_or_replay
1883 .handle_post_exec(&mut guest.into_guest())
1884 .await?;
1885
1886 self.post_handler_hook(guest).await;
1887 Ok(())
1888 }
1889
1890 /// A timer fires to preempt the guest and give other threads a turn.
1891 async fn handle_timer_event<G: Guest<Self>>(&self, guest: &mut G) {
1892 info!(
1893 "[detcore, dtid {}] inbound timer preemption event",
1894 guest.thread_state().dettid
1895 );
1896 if guest.config().preemption_stacktrace {
1897 let mut file_writer: Box<dyn Write> =
1898 match &guest.config().preemption_stacktrace_log_file {
1899 Some(path) => Box::new(
1900 File::create(path).expect("Failed to open preemption stacktrace log file"),
1901 ),
1902 None => Box::new(std::io::stderr()),
1903 };
1904 let ts = guest.thread_state();
1905 writeln!(
1906 file_writer,
1907 "\n>>> Guest tid {} preempted at thread time {} with stack trace:",
1908 ts.dettid,
1909 ts.thread_logical_time.as_nanos(),
1910 )
1911 .unwrap();
1912 if let Some(backtrace) = guest.backtrace() {
1913 if let Ok(pbt) = backtrace.pretty() {
1914 writeln!(file_writer, "{}", pbt).unwrap();
1915 } else {
1916 writeln!(file_writer, "{}", backtrace).unwrap();
1917 }
1918 } else {
1919 warn!("Could not read backtrace!");
1920 }
1921 }
1922 // This may LOOK like a noop, but actually all of the logic for ending the timeslice is in
1923 // the prehook. All the timer has to do is interrupt the guest and generate an extra call
1924 // to this prehook.
1925 self.pre_handler_hook(guest, true).await;
1926 if guest.config().no_rcb_time && guest.thread_state().last_rcb_timer_is_max {
1927 let max_timeslice_end = guest
1928 .thread_state()
1929 .max_timeslice_end
1930 .expect("PMU maximum timer requires a deadline");
1931 guest
1932 .thread_state_mut()
1933 .thread_logical_time
1934 .advance_to(max_timeslice_end);
1935 if self.cfg.should_trace_schedevent() {
1936 let dettid = guest.thread_state().dettid;
1937 let ev = with_guest_time(
1938 guest,
1939 SchedEvent {
1940 dettid,
1941 op: Op::OtherInstructions,
1942 count: 1,
1943 start_rip: None,
1944 end_rip: None,
1945 end_time: None,
1946 },
1947 );
1948 let ev = with_guest_rip(guest, ev).await;
1949 trace_schedevent(guest, ev, false).await;
1950 }
1951 if self.cfg.replay_schedule_from.is_some() {
1952 let fallback_deadline = max_timeslice_end
1953 + Duration::from_nanos(u64::from(
1954 self.cfg
1955 .max_timeslice
1956 .expect("PMU maximum must be configured"),
1957 ));
1958 let thread_state = guest.thread_state_mut();
1959 let replay_deadline = thread_state
1960 .end_of_timeslice
1961 .filter(|deadline| *deadline > max_timeslice_end)
1962 .unwrap_or(fallback_deadline);
1963 thread_state.end_of_timeslice = Some(replay_deadline);
1964 thread_state.max_timeslice_end = Some(replay_deadline);
1965 thread_state.last_rcb_timer = None;
1966 thread_state.last_rcb_timer_is_max = false;
1967 thread_state.stats.reset_timeslice();
1968 } else {
1969 self.end_timeslice(guest).await;
1970 }
1971 }
1972 self.post_handler_hook(guest).await;
1973 }
1974
1975 async fn handle_syscall_event<G: Guest<Self>>(
1976 &self,
1977 guest: &mut G,
1978 call: Syscall,
1979 ) -> Result<i64, Error> {
1980 self.pre_handler_hook(guest, false).await;
1981
1982 let dettid = guest.thread_state().dettid;
1983
1984 if guest.thread_state().guest_past_first_execve() {
1985 detlog!(
1986 event = crate::detlog::DetLogEvent::Syscall;
1987 "[syscall][detcore, dtid {}] inbound syscall: {} = ?",
1988 dettid,
1989 call.display(&guest.memory())
1990 );
1991 }
1992
1993 // The hot syscall path only reads a few Copy flags from the config, so copy
1994 // just those out instead of cloning the entire Config on every intercepted
1995 // syscall (previously flagged inline as an unnecessary copy). guest.config()
1996 // borrows guest immutably; bind the flags in a tight scope so the borrow ends
1997 // before the later thread_state_mut()/&mut guest borrows below.
1998 let (sequentialize_threads, virtualize_time, panic_on_unsupported_syscalls) = {
1999 let config = guest.config();
2000 (
2001 config.sequentialize_threads,
2002 config.virtualize_time,
2003 config.panic_on_unsupported_syscalls,
2004 )
2005 };
2006
2007 if sequentialize_threads && self.cfg.should_trace_schedevent() {
2008 trace_schedevent(
2009 guest,
2010 with_guest_time(
2011 guest,
2012 SchedEvent::syscall(dettid, call.number(), SyscallPhase::Prehook),
2013 ),
2014 true,
2015 )
2016 .await;
2017 }
2018
2019 let syscall_cost_ns = syscall_time::cost_ns(call.number());
2020 // The backend-runtime bootstrap window. A backend-resident runtime (the
2021 // LiteInst preload constructor) issues hundreds to thousands of syscalls
2022 // on one thread (315 for a small C program, 6,970 for emacs; see
2023 // `syscall_time::MAX_UNCHARGED_BOOTSTRAP_SYSCALLS`) between its validated
2024 // begin trap and the trap that ends its bootstrap: the ready report, or
2025 // the report that preparation failed.
2026 // Reverie reports that window only for the bootstrapping thread; every
2027 // other thread and every forked process sees no window.
2028 //
2029 // What is withheld: the per-syscall cost below, for the bootstrapping
2030 // thread only, for each syscall it makes inside the window that does not
2031 // observe virtual time, up to
2032 // `syscall_time::MAX_UNCHARGED_BOOTSTRAP_SYSCALLS` per window. That
2033 // includes syscalls made by guest code the runtime calls into, such as
2034 // an interposed malloc or open64. Charging the runtime's own syscalls
2035 // would make guest-visible virtual time (uptime, CLOCK_MONOTONIC, CPU
2036 // time) depend on which backend ran the program. Every withheld syscall
2037 // is still counted and fully handled below, so Detcore keeps tracking
2038 // the fds, mappings and inodes it creates.
2039 //
2040 // What is always charged, inside the window too: syscalls that observe
2041 // virtual time (`syscall_time::observes_virtual_time`), so two clock
2042 // reads on the bootstrapping thread differ by at least one syscall cost
2043 // exactly as outside the window; and every syscall past the cap, so a
2044 // loop inside the window, or a runtime that never reports ready, again
2045 // advances the clock at each syscall.
2046 //
2047 // When it matters: with a syscall-driven clock (no PMU, or
2048 // --max-timeslice=disabled, which Hermit selects when perf counters are
2049 // unavailable, or --no-rcb-time) these charges are the thread's only
2050 // per-syscall progress, and with sequentialized threads the 500x no-RCB
2051 // multiplier makes each one large. With RCB time on a PMU host, retired
2052 // branches still advance the thread's clock inside the window and only
2053 // the syscall costs are withheld.
2054 //
2055 // The decision depends only on the window the backend reports (opened
2056 // and closed by trap instructions the runtime executes), the syscall
2057 // number and the count of earlier uncharged syscalls in the same
2058 // window: all functions of the guest's own execution. Record/replay and
2059 // both --verify runs therefore make the same decisions. The clock is
2060 // never reset: the first syscall after the window continues from the
2061 // value the window left. See https://github.com/rrnewton/hermit/issues/3338
2062 // and https://github.com/rrnewton/hermit/pull/3430#issuecomment-5928691696.
2063 //
2064 // The scheduler turns such a syscall needs are withheld the same way.
2065 // Handling an uncharged syscall can still commit scheduler turns (for
2066 // example the file resources of a read of /proc/self/maps), and each
2067 // committed turn normally advances global time by the per-turn
2068 // scheduler cost. While the syscall is being handled the thread is
2069 // marked `in_uncharged_bootstrap_syscall`, `tool_global::resource_request`
2070 // copies that mark into the request, and the scheduler does not
2071 // advance global time for a marked turn unless it is an IO-polling
2072 // retry, whose time enforces timeouts. Without this, a LiteInst run of
2073 // the clock-trajectory fixture reached its first clock read 11.5 ms of
2074 // virtual time later than with it, and each exec added 12.0 ms more
2075 // (measured locally), which was enough for its sysinfo uptime to
2076 // differ from the ptrace backend's on the hosted runner
2077 // (https://github.com/rrnewton/hermit/issues/3517). A charged syscall
2078 // (one that observes virtual time, or any past the cap) is not marked,
2079 // so its turns advance the clock as everywhere else.
2080 let in_backend_runtime_bootstrap = guest.is_backend_runtime_bootstrap();
2081 let new_count = {
2082 // which results from not being able to borrow guest twice.
2083 let thread_state = guest.thread_state_mut();
2084 thread_state.stats.count_syscall();
2085
2086 // Every intercepted syscall advances logical time, including configurations that do
2087 // not serialize threads. This keeps virtual clocks productive during syscall loops.
2088 // The one exception is the bootstrap window described above: on the bootstrapping
2089 // thread, up to MAX_UNCHARGED_BOOTSTRAP_SYSCALLS syscalls per window that do not
2090 // observe virtual time are left uncharged.
2091 let charged =
2092 thread_state.charge_syscall_time(in_backend_runtime_bootstrap, call.number());
2093 if charged {
2094 thread_state
2095 .thread_logical_time
2096 .add_syscall_with_cost(syscall_cost_ns);
2097 }
2098 thread_state.in_uncharged_bootstrap_syscall = !charged;
2099 // This only folds the thread's new user and system time into the process total.
2100 // An uncharged syscall added none, so it needs no guard.
2101 thread_state.account_process_cpu_time();
2102 thread_state.stats.syscall_count
2103 };
2104
2105 // Happens-before enforcement checkpoint. When the run carries a
2106 // happens-before program with syscall-count anchors, every intercepted
2107 // syscall checks in with the scheduler at its prehook, carrying this
2108 // thread's running syscall count. The scheduler fires any anchor at
2109 // `Position::SyscallCount(new_count)` on this thread and parks the thread
2110 // (out of the run queue) when that anchor is the AFTER endpoint of a Hard
2111 // edge whose BEFORE endpoint has not fired yet. This is the gate that
2112 // makes an authored partial order reproduce a known race deterministically
2113 // (see detcore-model `happens_before`). It requires sequentialized
2114 // threads (enforced by the CLI) so the scheduler owns ordering.
2115 if guest
2116 .config()
2117 .happens_before
2118 .as_ref()
2119 .is_some_and(|p| p.has_syscall_count_anchors())
2120 {
2121 let request = guest.thread_state().mk_request(
2122 ResourceID::HappensBeforeCheckpoint(new_count),
2123 Permission::R,
2124 );
2125 resource_request(guest, request).await;
2126 }
2127
2128 // Only an emulated RNG readv supplies authoritative imported geometry.
2129 // A generic pre-dispatch snapshot would become stale across pipe waits.
2130 let mut rng_readv_output = None;
2131 let res = match classify_syscall(call.number()) {
2132 // Rseq is not type-safe in the pinned Reverie revision. Dispatch by Sysno so a
2133 // future typed representation preserves this explicit policy.
2134 SyscallClassification::Determinized if call.number() == Sysno::rseq => {
2135 if panic_on_unsupported_syscalls {
2136 Err(Error::Errno(Errno::ENOSYS))
2137 } else {
2138 self.passthrough(guest, call).await
2139 }
2140 }
2141 // AUTONOMOUS-BOT-IMPLEMENTED
2142 // TODO-HUMAN-REVIEW(#663)
2143 // The pinned Reverie revision exposes process_madvise only as a raw call.
2144 SyscallClassification::Determinized if call.number() == Sysno::process_madvise => {
2145 match call {
2146 Syscall::Other(_, args) => Self::handle_process_madvise(args.arg0, args.arg4),
2147 _ => unreachable!("process_madvise unexpectedly gained a typed variant"),
2148 }
2149 }
2150 // AUTONOMOUS-BOT-IMPLEMENTED
2151 // TODO-HUMAN-REVIEW(PR-1175): The pinned Reverie revision exposes
2152 // pidfd_send_signal/pidfd_getfd only as raw calls, so dispatch on the
2153 // Sysno. See the handlers in syscalls/files.rs for the determinism
2154 // argument.
2155 SyscallClassification::Determinized if call.number() == Sysno::pidfd_send_signal => {
2156 match call {
2157 Syscall::Other(_, args) => {
2158 self.handle_pidfd_send_signal(
2159 guest,
2160 call,
2161 args.arg0 as RawFd,
2162 args.arg3 as u32,
2163 )
2164 .await
2165 }
2166 _ => unreachable!("pidfd_send_signal unexpectedly gained a typed variant"),
2167 }
2168 }
2169 // AUTONOMOUS-BOT-IMPLEMENTED
2170 // TODO-HUMAN-REVIEW(PR-1175): pidfd_getfd, likewise untyped.
2171 SyscallClassification::Determinized if call.number() == Sysno::pidfd_getfd => {
2172 match call {
2173 Syscall::Other(_, args) => {
2174 self.handle_pidfd_getfd(
2175 guest,
2176 call,
2177 args.arg0 as RawFd,
2178 args.arg1 as RawFd,
2179 args.arg2 as u32,
2180 )
2181 .await
2182 }
2183 _ => unreachable!("pidfd_getfd unexpectedly gained a typed variant"),
2184 }
2185 }
2186 // AUTONOMOUS-BOT-IMPLEMENTED
2187 // TODO-HUMAN-REVIEW(#715): Deterministic ENOSYS for syscalls the pinned
2188 // x86_64 kernel leaves unimplemented (sys_ni_syscall). A fixed -ENOSYS is
2189 // deterministic by construction and identical to the modern kernel's own
2190 // return, so no guest-visible behavior changes versus the legacy
2191 // pass-through. These are untyped (Syscall::Other) in the pinned Reverie,
2192 // so dispatch on the Sysno before the typed match below.
2193 SyscallClassification::Determinized
2194 if is_unimplemented_enosys_syscall(call.number()) =>
2195 {
2196 Err(Error::Errno(Errno::ENOSYS))
2197 }
2198 // AUTONOMOUS-BOT-IMPLEMENTED
2199 // TODO-HUMAN-REVIEW(PR-852): Review the futex2 fallback contract.
2200 // Detcore models legacy futex but not the newer vector/sized futex2
2201 // ABI. Match a kernel without futex2 so runtimes take their
2202 // established legacy-futex fallback without consulting the host.
2203 SyscallClassification::Determinized if is_futex2_enosys_syscall(call.number()) => {
2204 Err(Error::Errno(Errno::ENOSYS))
2205 }
2206 // AUTONOMOUS-BOT-IMPLEMENTED
2207 // TODO-HUMAN-REVIEW(PR-836): Host filesystem and mount
2208 // introspection are outside the deterministic model. Return the
2209 // portable feature-absence errno so callers use /proc fallbacks.
2210 // AUTONOMOUS-BOT-IMPLEMENTED
2211 // TODO-HUMAN-REVIEW(PR-859): Extend this boundary to obsolete ustat
2212 // host-filesystem capacity counters.
2213 SyscallClassification::Determinized
2214 if is_mount_introspection_enosys_syscall(call.number()) =>
2215 {
2216 Err(Error::Errno(Errno::ENOSYS))
2217 }
2218 // AUTONOMOUS-BOT-IMPLEMENTED
2219 // TODO-HUMAN-REVIEW(PR-848): Hide unmodeled shared keyrings and
2220 // request-key upcalls behind the portable CONFIG_KEYS-absent errno.
2221 // TODO-HUMAN-REVIEW(PR-916): Fail closed whenever the
2222 // panic-on-unsupported policy is active; ordinary runs select that
2223 // policy by default. The explicit compatibility opt-out keeps the
2224 // pre-848 host pass-through so the guest observes a real working
2225 // keyring, restoring the enabled rr `keyctl` compatibility test.
2226 // Under fail-closed execution the deterministic ENOSYS boundary is
2227 // preserved.
2228 SyscallClassification::Determinized if is_kernel_keyring_syscall(call.number()) => {
2229 if panic_on_unsupported_syscalls {
2230 Err(Error::Errno(Errno::ENOSYS))
2231 } else {
2232 self.passthrough(guest, call).await
2233 }
2234 }
2235 // AUTONOMOUS-BOT-IMPLEMENTED
2236 // TODO-HUMAN-REVIEW(PR-855): Fail-closed runs cannot expose
2237 // unmodeled pipe-buffer ownership or vmsplice page pinning. Return
2238 // ENOSYS so callers use read/write fallbacks, but preserve host
2239 // pass-through under the explicit compatibility opt-out used by the
2240 // existing rr splice test.
2241 SyscallClassification::Determinized if is_zero_copy_pipe_syscall(call.number()) => {
2242 if panic_on_unsupported_syscalls {
2243 Err(Error::Errno(Errno::ENOSYS))
2244 } else {
2245 self.passthrough(guest, call).await
2246 }
2247 }
2248 // AUTONOMOUS-BOT-IMPLEMENTED
2249 // TODO-HUMAN-REVIEW(PR-860): Host LSM attributes are outside
2250 // Detcore's model. Present a stable feature-absence boundary
2251 // instead of forwarding probes.
2252 SyscallClassification::Determinized
2253 if is_host_security_identity_probe_syscall(call.number()) =>
2254 {
2255 Err(Error::Errno(Errno::ENOSYS))
2256 }
2257 // AUTONOMOUS-BOT-IMPLEMENTED
2258 // TODO-HUMAN-REVIEW(#722): Deterministic EPERM for privileged
2259 // system-administration syscalls (module load/unload, kexec, reboot,
2260 // swap, raw I/O ports, root-mount pivot, host/domain name, tty
2261 // hangup, disk quotas). The deterministic guest does not hold the
2262 // capabilities these require against the host kernel, so a fixed
2263 // -EPERM matches the unprivileged errno, never perturbs global host
2264 // state, and is identical across --verify and record/replay. These
2265 // are untyped (Syscall::Other) in the pinned Reverie, so dispatch on
2266 // the Sysno before the typed match below.
2267 SyscallClassification::Determinized
2268 if is_privileged_admin_refused_syscall(call.number()) =>
2269 {
2270 Err(Error::Errno(Errno::EPERM))
2271 }
2272 // AUTONOMOUS-BOT-IMPLEMENTED
2273 // TODO-HUMAN-REVIEW(PR-844): Enforce a deterministic boundary
2274 // around host-global process accounting and cross-process memory.
2275 SyscallClassification::Determinized
2276 if is_process_isolation_refused_syscall(call.number()) =>
2277 {
2278 Err(Error::Errno(Errno::EPERM))
2279 }
2280 // AUTONOMOUS-BOT-IMPLEMENTED
2281 // TODO-HUMAN-REVIEW(PR-876): Guest performance events
2282 // expose host PMU availability, policy, and asynchronous counter
2283 // state that Detcore does not model. A fixed ENOSYS preserves the
2284 // portable feature-probe fallback without creating an untracked fd.
2285 SyscallClassification::Determinized if is_perf_event_enosys_syscall(call.number()) => {
2286 Err(Error::Errno(Errno::ENOSYS))
2287 }
2288 // AUTONOMOUS-BOT-IMPLEMENTED
2289 // TODO-HUMAN-REVIEW(PR-853): Refuse nested tracing, host-object
2290 // comparison at the deterministic boundary.
2291 SyscallClassification::Determinized
2292 if is_privileged_observation_refused_syscall(call.number()) =>
2293 {
2294 Err(Error::Errno(Errno::EPERM))
2295 }
2296 // AUTONOMOUS-BOT-IMPLEMENTED
2297 // TODO-HUMAN-REVIEW(#720): set_mempolicy_home_node is untyped in the
2298 // pinned Reverie revision. Hermit exposes a single virtual NUMA node,
2299 // so setting a memory range's home node has no observable effect: a
2300 // deterministic no-op.
2301 SyscallClassification::Determinized
2302 if call.number() == Sysno::set_mempolicy_home_node =>
2303 {
2304 Ok(0)
2305 }
2306 // AUTONOMOUS-BOT-IMPLEMENTED
2307 // TODO-HUMAN-REVIEW(#724): Deterministic EPERM for privileged mount
2308 // and namespace administration syscalls (mount/umount2/mount_setattr/
2309 // move_mount/open_tree/fsopen/fsmount/fsconfig/fspick, unshare, setns,
2310 // open_by_handle_at, fanotify_init/fanotify_mark, settimeofday). A
2311 // deterministic container pins the guest's namespaces, mount
2312 // hierarchy, and virtual clock for the whole run, so these are
2313 // refused with a fixed -EPERM: the unprivileged errno for the
2314 // capability-gated operations and a deliberate deterministic refusal
2315 // otherwise. Never forwarded to the host; identical across --verify
2316 // and record/replay. Untyped (Syscall::Other) in the pinned Reverie,
2317 // so dispatch on the Sysno before the typed match below.
2318 SyscallClassification::Determinized
2319 if is_mount_ns_admin_refused_syscall(call.number()) =>
2320 {
2321 Err(Error::Errno(Errno::EPERM))
2322 }
2323 // AUTONOMOUS-BOT-IMPLEMENTED
2324 // TODO-HUMAN-REVIEW(#731): Deterministic ENOSYS for the
2325 // asynchronous and message-passing I/O and IPC interfaces Detcore
2326 // does not model: Linux native AIO (io_setup/io_destroy/io_submit/
2327 // io_cancel/io_getevents/io_pgetevents), POSIX message queues
2328 // (mq_*), and System V message queues (msg*). AIO completion is
2329 // kernel-driven and lives outside logical time; the message-queue
2330 // families operate on global, key/name-addressed kernel objects
2331 // shared with the whole host. A fixed -ENOSYS is the errno a kernel
2332 // built without AIO/CONFIG_POSIX_MQUEUE/CONFIG_SYSVIPC returns, is
2333 // never forwarded to the host, and is identical across --verify and
2334 // record/replay (mirrors the io_uring refusal). Untyped
2335 // (Syscall::Other) in the pinned Reverie, so dispatch on the Sysno
2336 // before the typed match below.
2337 // AUTONOMOUS-BOT-IMPLEMENTED
2338 // TODO-HUMAN-REVIEW(PR-859): Include System V semaphore and shared-
2339 // memory objects in the existing CONFIG_SYSVIPC refusal boundary.
2340 SyscallClassification::Determinized
2341 if is_unsupported_async_ipc_syscall(call.number()) =>
2342 {
2343 Err(Error::Errno(Errno::ENOSYS))
2344 }
2345 // AUTONOMOUS-BOT-IMPLEMENTED
2346 // TODO-HUMAN-REVIEW(PR-882): Legacy nonlinear page
2347 // remapping has host-dependent kernel support and VMA behavior that
2348 // Detcore does not model. Preserve the documented mmap fallback.
2349 SyscallClassification::Determinized
2350 if is_remap_file_pages_enosys_syscall(call.number()) =>
2351 {
2352 Err(Error::Errno(Errno::ENOSYS))
2353 }
2354 // AUTONOMOUS-BOT-IMPLEMENTED
2355 // TODO-HUMAN-REVIEW(#787): BATCH 38. openat2 is untyped (Syscall::Other)
2356 // in the pinned Reverie revision. It is a superset of openat whose
2357 // callers must fall back to openat when it returns ENOSYS (kernels
2358 // before 5.6 lack openat2), so a fixed -ENOSYS routes them onto the
2359 // already-determinized openat path with no host dependency and behavior
2360 // identical across --verify and record/replay.
2361 SyscallClassification::Determinized if call.number() == Sysno::openat2 => {
2362 Err(Error::Errno(Errno::ENOSYS))
2363 }
2364 // AUTONOMOUS-BOT-IMPLEMENTED
2365 // TODO-HUMAN-REVIEW(#787): BATCH 38. The credential-setting family
2366 // (setuid/setgid and their re-/res-/fs- variants, and setgroups) is
2367 // untyped (Syscall::Other) in the pinned Reverie. Detcore presents a
2368 // fixed virtual-root identity (getuid/geteuid/getgid/getegid are
2369 // virtualized to 0) and never tracks a credential change, so these
2370 // succeed as deterministic no-ops returning 0 -- the value a real root
2371 // process gets for a permitted credential change (and the previous
2372 // fs-id, virtual 0, for setfsuid/setfsgid). That lets privilege-
2373 // dropping programs proceed instead of fail-closing and is identical
2374 // across --verify and record/replay.
2375 SyscallClassification::Determinized
2376 if is_credential_identity_noop_syscall(call.number()) =>
2377 {
2378 Ok(0)
2379 }
2380 // AUTONOMOUS-BOT-IMPLEMENTED
2381 // TODO-HUMAN-REVIEW(#1851): The file-ownership mutation family
2382 // (chown/fchown/fchownat/lchown) completes the fixed virtual-root
2383 // identity that the credential query (#1549) and credential set
2384 // (#787) families already implement. A real root process's chown
2385 // succeeds for any uid, so 0 is the value the virtual identity must
2386 // observe; forwarding instead returned the errno of whatever host
2387 // identity the backend happened to run under (EPERM with no user
2388 // namespace, EINVAL for an unmapped uid inside a one-uid map, and
2389 // backend-dependent for in-process backends).
2390 //
2391 // The emulation covers the IDENTITY half only. Root privilege
2392 // waives the ownership permission check; it does not waive pathname,
2393 // descriptor, or flag errors, so handle_ownership_change_noop
2394 // translates the target arguments into a side-effect-free metadata
2395 // lookup and returns 0 only if that validation succeeds. ENOENT,
2396 // EBADF, EFAULT, ENOTDIR and the fchownat flag EINVAL therefore
2397 // still reach the guest; the host-identity-dependent EPERM/EINVAL
2398 // cannot be produced at all. No setattr is attempted, so host
2399 // ownership, mode bits, and timestamps are never modified, and
2400 // Detcore does not model per-file ownership, so the success is not
2401 // observable through a later stat -- see
2402 // is_ownership_change_noop_syscall and handle_ownership_change_noop
2403 // for the full boundary, and
2404 // hermit-cli/tests/chown_virtual_root_identity.rs for the bracket
2405 // that fails if this arm's RESULT regresses.
2406 SyscallClassification::Determinized
2407 if is_ownership_change_noop_syscall(call.number()) =>
2408 {
2409 self.handle_ownership_change_noop(guest, call).await
2410 }
2411 // AUTONOMOUS-BOT-IMPLEMENTED
2412 // TODO-HUMAN-REVIEW(#827): Deterministic ENOSYS for the Landlock
2413 // unprivileged-sandbox syscalls (landlock_create_ruleset,
2414 // landlock_add_rule, landlock_restrict_self). Landlock availability
2415 // and ABI version depend on the host kernel build
2416 // (CONFIG_SECURITY_LANDLOCK) and runtime LSM stacking, so forwarding
2417 // them (the legacy pass-through) is host-dependent and, because a
2418 // ruleset restricts the whole thread tree, a global-state isolation
2419 // hole. A fixed -ENOSYS is the errno a kernel built without Landlock
2420 // returns, so the guest sees a consistent "sandbox unavailable"
2421 // answer regardless of host; never forwarded to the host and
2422 // bitwise-identical across --verify and record/replay. Untyped
2423 // (Syscall::Other) in the pinned Reverie, so dispatch on the Sysno
2424 // before the typed match below.
2425 SyscallClassification::Determinized if is_landlock_sandbox_syscall(call.number()) => {
2426 Err(Error::Errno(Errno::ENOSYS))
2427 }
2428 // AUTONOMOUS-BOT-IMPLEMENTED
2429 // TODO-HUMAN-REVIEW(PR-847): Refuse unmodeled host-kernel probes
2430 // with a fixed ENOSYS so guest behavior does not depend on BPF/LSM
2431 // configuration or mutable page-cache state.
2432 SyscallClassification::Determinized if is_host_kernel_probe_syscall(call.number()) => {
2433 Err(Error::Errno(Errno::ENOSYS))
2434 }
2435 // AUTONOMOUS-BOT-IMPLEMENTED
2436 // TODO-HUMAN-REVIEW(PR-838): Review close_range descriptor-table
2437 // synchronization. The pinned Reverie exposes close_range as a raw
2438 // call, so dispatch by Sysno before the typed match.
2439 SyscallClassification::Determinized if call.number() == Sysno::close_range => {
2440 self.handle_close_range(guest, call).await
2441 }
2442 // AUTONOMOUS-BOT-IMPLEMENTED
2443 // TODO-HUMAN-REVIEW(PR-839): Optional modern memory APIs vary with
2444 // host kernel configuration, CET support, and pidfd lifecycle.
2445 // Present the portable feature-absence result instead.
2446 SyscallClassification::Determinized
2447 if is_optional_memory_feature_syscall(call.number()) =>
2448 {
2449 Err(Error::Errno(Errno::ENOSYS))
2450 }
2451 // AUTONOMOUS-BOT-IMPLEMENTED
2452 // TODO-HUMAN-REVIEW(#773): epoll_pwait2 is untyped (Syscall::Other)
2453 // in the pinned Reverie revision. It is epoll_pwait with a
2454 // `struct timespec *` timeout; recent glibc routes epoll_wait/
2455 // epoll_pwait through it. Handled identically to epoll_pwait
2456 // (scheduler yield + record/replay forwarding).
2457 SyscallClassification::Determinized if call.number() == Sysno::epoll_pwait2 => {
2458 self.handle_epoll_pwait2(guest, call).await
2459 }
2460 SyscallClassification::Determinized => match call {
2461 Syscall::Write(w) => self.handle_write(guest, w).await,
2462 // AUTONOMOUS-BOT-IMPLEMENTED
2463 // TODO-HUMAN-REVIEW(#547)
2464 Syscall::Writev(w) => self.handle_writev(guest, w).await,
2465 Syscall::Openat(o) => self.handle_openat(guest, o).await,
2466 Syscall::Open(o) => self.handle_openat(guest, o.into()).await,
2467 Syscall::Creat(o) => self.handle_openat(guest, o.into()).await,
2468 Syscall::Close(s) => self.handle_close(guest, s).await,
2469 Syscall::Read(s) if self.sock_diag_reply_fd(guest, s.fd()) => {
2470 self.handle_sock_diag_read(guest, s).await
2471 }
2472 Syscall::Read(s) => self.handle_read(guest, s).await,
2473 Syscall::Pread64(s) => self.handle_pread64(guest, s).await,
2474 Syscall::Lseek(s) => self.handle_lseek(guest, s).await,
2475 // AUTONOMOUS-BOT-IMPLEMENTED
2476 // TODO-HUMAN-REVIEW(PR-838): Review regular-file sendfile mediation.
2477 Syscall::Sendfile(s) => self.handle_sendfile(guest, s).await,
2478 // AUTONOMOUS-BOT-IMPLEMENTED
2479 // TODO-HUMAN-REVIEW(PR-887): Present a stable pre-4.5-kernel
2480 // boundary so callers use determinized read/write copying.
2481 Syscall::CopyFileRange(_) => Err(Error::Errno(Errno::ENOSYS)),
2482 // TODO-HUMAN-REVIEW(#794): vectored scatter/gather I/O, mirroring
2483 // read/pread64/pwrite64/writev.
2484 Syscall::Readv(s) if self.sock_diag_reply_fd(guest, s.fd()) => {
2485 self.handle_sock_diag_readv(guest, s).await
2486 }
2487 Syscall::Readv(s) => {
2488 self.handle_readv_with_output(guest, s, &mut rng_readv_output)
2489 .await
2490 }
2491 Syscall::Preadv(s) => {
2492 self.handle_preadv_with_output(guest, s, &mut rng_readv_output)
2493 .await
2494 }
2495 Syscall::Preadv2(s) => {
2496 self.handle_preadv2_with_output(guest, s, &mut rng_readv_output)
2497 .await
2498 }
2499 Syscall::Pwritev(s) => self.handle_pwritev(guest, s).await,
2500 Syscall::Pwritev2(s) => self.handle_pwritev2(guest, s).await,
2501 // AUTONOMOUS-BOT-IMPLEMENTED
2502 // TODO-HUMAN-REVIEW(#683)
2503 Syscall::Pwrite64(s) => self.handle_pwrite64(guest, s).await,
2504 // This syscall is advisory; fixed success preserves its API contract.
2505 Syscall::Fadvise64(_) => Ok(0),
2506 Syscall::Mmap(s) => self.handle_mmap(guest, s).await,
2507 Syscall::Madvise(s) => self.handle_madvise(guest, s).await,
2508 // AUTONOMOUS-BOT-IMPLEMENTED
2509 // TODO-HUMAN-REVIEW(#775)
2510 Syscall::Mincore(s) => self.handle_mincore(guest, s).await,
2511 Syscall::Munmap(s) => self.handle_munmap(guest, s).await,
2512 Syscall::Mremap(s) => self.handle_mremap(guest, s).await,
2513 Syscall::Stat(s) => self.handle_stat_family(guest, s.into()).await,
2514 Syscall::Lstat(s) => self.handle_stat_family(guest, s.into()).await,
2515 Syscall::Fstat(s) => self.handle_stat_family(guest, s.into()).await,
2516 Syscall::Newfstatat(s) => self.handle_stat_family(guest, s.into()).await,
2517 Syscall::Statx(s) => self.handle_statx(guest, s).await,
2518 // AUTONOMOUS-BOT-IMPLEMENTED
2519 // TODO-HUMAN-REVIEW(#877)
2520 Syscall::Readlink(s) => self.handle_readlink(guest, s).await,
2521 // AUTONOMOUS-BOT-IMPLEMENTED
2522 Syscall::Readlinkat(s) => self.handle_readlinkat(guest, s).await,
2523 Syscall::Fcntl(s) => self.handle_fcntl(guest, s).await,
2524 // AUTONOMOUS-BOT-IMPLEMENTED
2525 // TODO-HUMAN-REVIEW(PR-912)
2526 Syscall::Ioctl(s)
2527 if syscalls::socket_timestamp_ioctl::is_socket_timestamp_ioctl(s) =>
2528 {
2529 self.handle_socket_timestamp_ioctl(guest, s).await
2530 }
2531 Syscall::Ioctl(s) => self.handle_ioctl(guest, s).await,
2532 Syscall::Futex(s) => self.handle_futex(guest, s).await,
2533
2534 Syscall::Clone(s) => self.handle_clone_family(guest, s.into()).await,
2535 Syscall::Clone3(s) => self.handle_clone_family(guest, s.into()).await,
2536 Syscall::Fork(s) => self.handle_clone_family(guest, s.into()).await,
2537
2538 // Forward vfork as vfork (rather than rewriting to fork) so the
2539 // kernel enforces the CLONE_VFORK parent-blocking contract while the
2540 // child registers itself and runs to exec/exit.
2541 Syscall::Vfork(s) => self.handle_clone_family(guest, s.into()).await,
2542 Syscall::Wait4(s) => self.handle_wait4(guest, s).await,
2543 Syscall::Waitid(s) => self.handle_waitid(guest, s).await,
2544
2545 Syscall::Setpgid(s) => self.handle_setpgid(guest, s).await,
2546 Syscall::Setsid(s) => self.handle_setsid(guest, s).await,
2547 // Without virtual time the guest reads the host clock. Each of these
2548 // must have a record/replay handler: a recording captures the value
2549 // the guest observed, and replay returns it without reading the host.
2550 Syscall::Gettimeofday(s) => {
2551 if virtualize_time {
2552 self.handle_gettimeofday(guest, s).await
2553 } else {
2554 self.passthrough(guest, call).await
2555 }
2556 }
2557 Syscall::Time(s) => {
2558 if virtualize_time {
2559 self.handle_time(guest, s).await
2560 } else {
2561 self.passthrough(guest, call).await
2562 }
2563 }
2564 Syscall::ClockGettime(s) => {
2565 if virtualize_time {
2566 self.handle_clock_gettime(guest, s).await
2567 } else {
2568 self.passthrough(guest, call).await
2569 }
2570 }
2571 Syscall::ClockGetres(s) => {
2572 if virtualize_time {
2573 self.handle_clock_getres(guest, s).await
2574 } else {
2575 self.passthrough(guest, call).await
2576 }
2577 }
2578 // AUTONOMOUS-BOT-IMPLEMENTED
2579 // TODO-HUMAN-REVIEW(#663)
2580 Syscall::ClockSettime(_) => Err(Error::Errno(Errno::EPERM)),
2581 // AUTONOMOUS-BOT-IMPLEMENTED
2582 // TODO-HUMAN-REVIEW(PR-892)
2583 Syscall::Getitimer(s) => self.handle_getitimer(guest, s).await,
2584 // AUTONOMOUS-BOT-IMPLEMENTED
2585 // TODO-HUMAN-REVIEW(#663)
2586 Syscall::Setitimer(s) => self.handle_setitimer(guest, s).await,
2587 // AUTONOMOUS-BOT-IMPLEMENTED
2588 // TODO-HUMAN-REVIEW(PR-857): Virtual NTP query and fixed mutation refusal.
2589 Syscall::Adjtimex(s) => {
2590 if virtualize_time {
2591 self.handle_adjtimex(guest, s).await
2592 } else {
2593 self.handle_unsupported_syscall(
2594 guest,
2595 call,
2596 dettid,
2597 panic_on_unsupported_syscalls,
2598 )
2599 .await
2600 }
2601 }
2602 // AUTONOMOUS-BOT-IMPLEMENTED
2603 // TODO-HUMAN-REVIEW(PR-857): Clock-id form of virtual NTP query.
2604 Syscall::ClockAdjtime(s) => {
2605 if virtualize_time {
2606 self.handle_clock_adjtime(guest, s).await
2607 } else {
2608 self.handle_unsupported_syscall(
2609 guest,
2610 call,
2611 dettid,
2612 panic_on_unsupported_syscalls,
2613 )
2614 .await
2615 }
2616 }
2617 // AUTONOMOUS-BOT-IMPLEMENTED
2618 // TODO-HUMAN-REVIEW(PR-857): Empty virtual kernel ring buffer.
2619 Syscall::Syslog(s) => self.handle_syslog(guest, s).await,
2620 Syscall::ArchPrctl(s) => self.handle_arch_prctl(guest, s).await,
2621 // AUTONOMOUS-BOT-IMPLEMENTED
2622 Syscall::Seccomp(s) => self.handle_seccomp(guest, s).await,
2623 // AUTONOMOUS-BOT-IMPLEMENTED
2624 // TODO-HUMAN-REVIEW(#663)
2625 Syscall::Prctl(s) => self.handle_prctl(guest, s).await,
2626 // AUTONOMOUS-BOT-IMPLEMENTED
2627 // TODO-HUMAN-REVIEW(#663)
2628 Syscall::Getpriority(s) => self.handle_getpriority(guest, s).await,
2629 // AUTONOMOUS-BOT-IMPLEMENTED
2630 // TODO-HUMAN-REVIEW(#663)
2631 Syscall::Setpriority(s) => self.handle_setpriority(guest, s).await,
2632 Syscall::Uname(s) => self.handle_uname(guest, s).await,
2633 Syscall::ExitGroup(s) => self.handle_exit_group(guest, s).await,
2634 Syscall::Exit(s) => self.handle_exit(guest, s).await,
2635
2636 Syscall::Dup(w) => self.handle_dup(guest, w).await.map_err(Into::into),
2637 Syscall::Dup2(w) => self.handle_dup2(guest, w).await.map_err(Into::into),
2638 Syscall::Dup3(w) => self.handle_dup3(guest, w).await.map_err(Into::into),
2639 Syscall::Pipe(w) => self.handle_pipe2(guest, w.into()).await,
2640 Syscall::Pipe2(w) => self.handle_pipe2(guest, w).await,
2641 Syscall::Getrandom(s) => self.handle_getrandom(guest, s).await,
2642 Syscall::Utime(s) => self.handle_utime(guest, s).await.map_err(Into::into),
2643 Syscall::Utimes(s) => self.handle_utimes(guest, s).await.map_err(Into::into),
2644 // NB: lutimes is a libc function not a syscall
2645 Syscall::Utimensat(s) => self.handle_utimensat(guest, s).await.map_err(Into::into),
2646 // NB: futimes/futimens are libc functions not a syscall,
2647 // futimesat is obsolete, return -ENOSYS for simplicity.
2648 Syscall::Futimesat(_s) => Err(Error::Errno(Errno::ENOSYS)),
2649 // io_uring completion and memory-sharing semantics are not deterministic.
2650 Syscall::IoUringSetup(_)
2651 | Syscall::IoUringEnter(_)
2652 | Syscall::IoUringRegister(_) => Err(Error::Errno(Errno::ENOSYS)),
2653 Syscall::Socket(s) => self.handle_socket(guest, s).await,
2654 Syscall::Socketpair(s) => self.handle_socketpair(guest, s).await,
2655 Syscall::Connect(s) => self.handle_connect(guest, s).await,
2656 Syscall::Bind(s) => self.handle_bind(guest, s).await,
2657 // AUTONOMOUS-BOT-IMPLEMENTED
2658 // TODO-HUMAN-REVIEW(#663)
2659 Syscall::Setsockopt(s) => self.handle_setsockopt(guest, s).await,
2660 // AUTONOMOUS-BOT-IMPLEMENTED
2661 // TODO-HUMAN-REVIEW(#663)
2662 Syscall::Listen(s) => self.handle_listen(guest, s).await,
2663 // AUTONOMOUS-BOT-IMPLEMENTED
2664 // TODO-HUMAN-REVIEW(#663)
2665 Syscall::Getsockname(s) => self.handle_getsockname(guest, s).await,
2666 // AUTONOMOUS-BOT-IMPLEMENTED
2667 // TODO-HUMAN-REVIEW(#663)
2668 Syscall::Getpeername(s) => self.handle_getpeername(guest, s).await,
2669 // AUTONOMOUS-BOT-IMPLEMENTED
2670 // TODO-HUMAN-REVIEW(#663)
2671 Syscall::Getsockopt(s) => self.handle_getsockopt(guest, s).await,
2672 // AUTONOMOUS-BOT-IMPLEMENTED
2673 // TODO-HUMAN-REVIEW(#818): shutdown is the lone remaining
2674 // socket-family syscall; half-closes a tracked socket and
2675 // forwards via record_or_replay (KVM ratchet round 12).
2676 Syscall::Shutdown(s) => self.handle_shutdown(guest, s).await,
2677 Syscall::Eventfd(s) => self.handle_eventfd2(guest, s.into()).await,
2678 Syscall::Eventfd2(s) => self.handle_eventfd2(guest, s).await,
2679 Syscall::Signalfd(s) => self.handle_signalfd4(guest, s.into()).await,
2680 Syscall::Signalfd4(s) => self.handle_signalfd4(guest, s).await,
2681 Syscall::TimerfdCreate(s) => self.handle_timerfd_create(guest, s).await,
2682 Syscall::TimerfdSettime(s) => self.handle_timerfd_settime(guest, s).await,
2683 Syscall::TimerfdGettime(s) => self.handle_timerfd_gettime(guest, s).await,
2684 Syscall::InotifyInit(s) => {
2685 self.handle_inotify_init1(guest, InotifyInit1::from(s))
2686 .await
2687 }
2688 Syscall::InotifyInit1(s) => self.handle_inotify_init1(guest, s).await,
2689 Syscall::InotifyAddWatch(s) => self.handle_inotify_add_watch(guest, s).await,
2690 Syscall::InotifyRmWatch(s) => self.handle_inotify_rm_watch(guest, s).await,
2691 Syscall::MemfdCreate(s) => self.handle_memfd_create(guest, s).await,
2692 // AUTONOMOUS-BOT-IMPLEMENTED
2693 // TODO-HUMAN-REVIEW(PR-862): Record/replay and register pidfds.
2694 Syscall::PidfdOpen(s) => self.handle_pidfd_open(guest, s).await,
2695 // AUTONOMOUS-BOT-IMPLEMENTED
2696 // TODO-HUMAN-REVIEW(PR-899): Host object handles and mount IDs
2697 // are outside Detcore's filesystem identity model.
2698 Syscall::NameToHandleAt(_) => Err(Error::Errno(Errno::EOPNOTSUPP)),
2699 Syscall::Userfaultfd(s) => self.handle_userfaultfd(guest, s).await,
2700 Syscall::Accept(s) => self.handle_accept4(guest, s.into()).await,
2701 Syscall::Accept4(s) => self.handle_accept4(guest, s).await,
2702
2703 Syscall::Nanosleep(s) => self.handle_nanosleep_family(guest, s.into()).await,
2704 Syscall::ClockNanosleep(s) => self.handle_nanosleep_family(guest, s.into()).await,
2705 Syscall::SchedYield(s) => self.handle_sched_yield(guest, s).await,
2706
2707 // NB: getdents is not recommended, (g)libc should call getdents64 only
2708 // see: sysdeps/unix/sysv/linux/getdents.c.
2709 Syscall::Getdents(s) => self.handle_getdents(guest, s).await,
2710 Syscall::Getdents64(s) => self.handle_getdents64(guest, s).await,
2711
2712 Syscall::Poll(s) => self.handle_poll(guest, s).await,
2713 // AUTONOMOUS-BOT-IMPLEMENTED
2714 // TODO-HUMAN-REVIEW(#686): Review scratch fd sets and scheduler polling.
2715 Syscall::Pselect6(s) => self.handle_pselect6(guest, s).await,
2716 // AUTONOMOUS-BOT-IMPLEMENTED
2717 // TODO-HUMAN-REVIEW(#800): select is the timeval sibling of pselect6.
2718 Syscall::Select(s) => self.handle_select(guest, s).await,
2719 // AUTONOMOUS-BOT-IMPLEMENTED
2720 Syscall::Ppoll(s) => self.handle_ppoll(guest, s).await,
2721 Syscall::EpollCreate(s) => {
2722 self.handle_epoll_create1(guest, EpollCreate1::from(s))
2723 .await
2724 }
2725 Syscall::EpollCreate1(s) => self.handle_epoll_create1(guest, s).await,
2726 Syscall::EpollCtl(s) => self.handle_epoll_ctl(guest, s).await,
2727 Syscall::EpollPwait(s) => self.handle_epoll_pwait(guest, s).await,
2728 Syscall::EpollWait(s) => self.handle_epoll_wait(guest, s).await,
2729 Syscall::EpollWaitOld(s) => panic!(
2730 "Not handling deprecated syscall: {}",
2731 s.display(&guest.memory())
2732 ),
2733 // AUTONOMOUS-BOT-IMPLEMENTED
2734 // TODO-HUMAN-REVIEW(#549)
2735 // The obsolete x86_64 entry point is absent from modern Linux kernels.
2736 Syscall::EpollCtlOld(_) => Err(Error::Errno(Errno::ENOSYS)),
2737
2738 Syscall::SchedGetaffinity(s) => self.handle_sched_getaffinity(guest, s).await,
2739 Syscall::SchedSetaffinity(s) => self.handle_sched_setaffinity(guest, s).await,
2740
2741 // ===== BATCH 3: NUMA memory-placement and Linux CPU-scheduling
2742 // policy. Hermit exposes a single virtual NUMA node and replaces
2743 // the Linux scheduler with Detcore, so these are inoperative and
2744 // are virtualized to fixed, host-independent results (see the
2745 // determinism argument in syscall_classification.rs). Setters and
2746 // count-returning calls are no-ops; getters emulate a default
2747 // single-node / SCHED_OTHER answer.
2748 // AUTONOMOUS-BOT-IMPLEMENTED
2749 // TODO-HUMAN-REVIEW(#720)
2750 Syscall::Mbind(_) => Ok(0),
2751 Syscall::SetMempolicy(_) => Ok(0),
2752 Syscall::GetMempolicy(s) => self.handle_get_mempolicy(guest, s).await,
2753 Syscall::MigratePages(_) => Ok(0),
2754 Syscall::MovePages(s) => self.handle_move_pages(guest, s).await,
2755 Syscall::SchedSetscheduler(_) => Ok(0),
2756 Syscall::SchedSetparam(_) => Ok(0),
2757 // Report the fixed default policy SCHED_OTHER (0).
2758 Syscall::SchedGetscheduler(_) => Ok(0),
2759 Syscall::SchedGetparam(s) => self.handle_sched_getparam(guest, s).await,
2760 Syscall::SchedRrGetInterval(s) => self.handle_sched_rr_get_interval(guest, s).await,
2761
2762 // ===== BATCH 51: fail-closed utility syscalls, re-enabling chrt,
2763 // ionice, and flock under --strict. Detcore replaces the Linux
2764 // scheduler, exposes a single virtual CPU, and serializes guest
2765 // threads, so a thread's Linux scheduling attributes (sched_getattr)
2766 // and I/O priority (ioprio_set) are inert: those two have no
2767 // deterministic effect and are emulated to fixed, host-independent
2768 // results (see syscall_classification.rs).
2769 //
2770 // flock is NOT in that inert group and is not emulated. This comment
2771 // used to claim it "is never contended inside the serialized
2772 // container"; that was measured false -- two open file descriptions
2773 // in ONE process both held the same LOCK_EX under the old no-op,
2774 // where native Linux excluded the second -- and the no-op was
2775 // removed. Serializing THREADS does not make a whole-file lock
2776 // uncontended, because flock conflicts are between OPEN FILE
2777 // DESCRIPTIONS. flock is now forwarded to the kernel, like fcntl's
2778 // POSIX record locks; handle_flock carries the determinism argument
2779 // and the one case Detcore refuses (a contended BLOCKING request,
2780 // which it cannot park a thread on deterministically).
2781 // AUTONOMOUS-BOT-IMPLEMENTED
2782 // TODO-HUMAN-REVIEW(#791)
2783 Syscall::SchedGetattr(s) => self.handle_sched_getattr(guest, s).await,
2784 // AUTONOMOUS-BOT-IMPLEMENTED
2785 // TODO-HUMAN-REVIEW(PR-841): Review virtual sched_setattr no-op policy.
2786 Syscall::SchedSetattr(s) => self.handle_sched_setattr(guest, s).await,
2787 // AUTONOMOUS-BOT-IMPLEMENTED
2788 // TODO-HUMAN-REVIEW(#791)
2789 Syscall::IoprioSet(s) => self.handle_ioprio_set(guest, s).await,
2790 // AUTONOMOUS-BOT-IMPLEMENTED
2791 // TODO-HUMAN-REVIEW(PR-881): Review virtual ioprio_get defaults.
2792 Syscall::IoprioGet(s) => self.handle_ioprio_get(guest, s).await,
2793 // AUTONOMOUS-BOT-IMPLEMENTED
2794 // TODO-HUMAN-REVIEW(#2373)
2795 Syscall::Flock(s) => self.handle_flock(guest, s).await,
2796
2797 // TODO-HUMAN-REVIEW(PR-1064): recvfrom/read/readv/recvmmsg reach
2798 // a NETLINK_SOCK_DIAG dump exactly as recvmsg does. Until they
2799 // were routed through the same sanitizer, four of the five
2800 // usable receive syscalls returned raw host socket inode
2801 // numbers, which made the determinization optional from the
2802 // guest's point of view: `socket.recv()` alone was enough to
2803 // skip it. Non-socket-diag descriptors take the same path as
2804 // before; the predicate is checked inside.
2805 Syscall::Recvfrom(s) if self.sock_diag_reply_fd(guest, s.fd()) => {
2806 self.handle_sock_diag_recvfrom(guest, s).await
2807 }
2808 Syscall::Recvfrom(s) => self.handle_socket_receive(guest, s, s.fd(), true).await,
2809 // AUTONOMOUS-BOT-IMPLEMENTED
2810 // TODO-HUMAN-REVIEW(PR-901)
2811 Syscall::Recvmsg(s) => self.handle_recvmsg(guest, s).await,
2812 Syscall::Sendto(s) => self.handle_sendrecv(guest, s).await,
2813 Syscall::Sendmsg(s) => self.handle_sendmsg(guest, s).await,
2814 Syscall::Sendmmsg(s) => self.handle_sendmmsg(guest, s).await,
2815
2816 // AUTONOMOUS-BOT-IMPLEMENTED
2817 // TODO-HUMAN-REVIEW(#788): recvmmsg is the multi-message form of
2818 // recvmsg and shares its NonblockableSyscall impl. The fd is made
2819 // temporarily nonblocking, the kernel fills the mmsghdr array
2820 // atomically, and the Detcore scheduler owns any blocking, so the
2821 // timeout argument (deliberately ignored, see helpers.rs) does not
2822 // introduce nondeterminism.
2823 // TODO-HUMAN-REVIEW(PR-901): Review batched ancillary timestamp rewriting.
2824 Syscall::Recvmmsg(s) if self.sock_diag_reply_fd(guest, s.fd()) => {
2825 self.handle_sock_diag_recvmmsg(guest, s).await
2826 }
2827 Syscall::Recvmmsg(s) => self.handle_recvmmsg(guest, s).await,
2828 Syscall::RtSigtimedwait(s) => self.handle_rt_sigtimedwait(guest, s).await,
2829 Syscall::RtSigsuspend(s) => self.handle_rt_sigsuspend(guest, s).await,
2830 // AUTONOMOUS-BOT-IMPLEMENTED
2831 // TODO-HUMAN-REVIEW(#663)
2832 Syscall::RtSigpending(s) => self.handle_rt_sigpending(guest, s).await,
2833 // AUTONOMOUS-BOT-IMPLEMENTED
2834 // TODO-HUMAN-REVIEW(#663)
2835 Syscall::Kill(s) => self.handle_kill(guest, s).await,
2836 // AUTONOMOUS-BOT-IMPLEMENTED
2837 // TODO-HUMAN-REVIEW(#663)
2838 Syscall::Tgkill(s) => self.handle_tgkill(guest, s).await,
2839 // AUTONOMOUS-BOT-IMPLEMENTED
2840 // TODO-HUMAN-REVIEW(#812)
2841 Syscall::Tkill(s) => self.handle_tkill(guest, s).await,
2842 // AUTONOMOUS-BOT-IMPLEMENTED
2843 // TODO-HUMAN-REVIEW(#812)
2844 Syscall::RtSigqueueinfo(s) => self.handle_rt_sigqueueinfo(guest, s).await,
2845 // AUTONOMOUS-BOT-IMPLEMENTED
2846 // TODO-HUMAN-REVIEW(#812)
2847 Syscall::RtTgsigqueueinfo(s) => self.handle_rt_tgsigqueueinfo(guest, s).await,
2848
2849 Syscall::Execve(s) => self.handle_execveat(guest, s.into()).await,
2850 Syscall::Execveat(s) => self.handle_execveat(guest, s).await,
2851
2852 Syscall::Getcpu(s) => self.handle_getcpu(guest, s).await,
2853
2854 // AUTONOMOUS-BOT-IMPLEMENTED
2855 // TODO-HUMAN-REVIEW(#1549): Credential-query
2856 // family emulated to the fixed virtual-root identity (0). See
2857 // syscall_classification.rs for the determinism rationale.
2858 // getuid/geteuid/getgid/getegid return the constant directly;
2859 // getresuid/getresgid write the constant to each provided result
2860 // pointer. Never forwarded to the host, so the answer no longer
2861 // depends on whether the backend runs the guest in a
2862 // CLONE_NEWUSER namespace.
2863 Syscall::Getuid(_)
2864 | Syscall::Geteuid(_)
2865 | Syscall::Getgid(_)
2866 | Syscall::Getegid(_) => Ok(0),
2867 Syscall::Getresuid(s) => self.handle_getresuid(guest, s).await,
2868 Syscall::Getresgid(s) => self.handle_getresgid(guest, s).await,
2869 Syscall::RtSigprocmask(s) => self.handle_rt_sigprocmask(guest, s).await,
2870 Syscall::RtSigaction(s) => self.handle_rt_sigaction(guest, s).await,
2871 Syscall::Alarm(s) => self.handle_alarm(guest, s).await,
2872 Syscall::Pause(s) => self.handle_pause(guest, s).await,
2873
2874 Syscall::Getrusage(s) => self.handle_getrusage(guest, s).await,
2875 Syscall::Sysinfo(s) => self.handle_sysinfo(guest, s).await,
2876 // AUTONOMOUS-BOT-IMPLEMENTED
2877 Syscall::Times(s) => self.handle_times(guest, s).await,
2878 Syscall::Prlimit64(s) => self.handle_prlimit64(guest, s).await,
2879 // AUTONOMOUS-BOT-IMPLEMENTED
2880 // TODO-HUMAN-REVIEW(#663)
2881 Syscall::Getrlimit(s) => self.handle_getrlimit(guest, s).await,
2882 // AUTONOMOUS-BOT-IMPLEMENTED
2883 // TODO-HUMAN-REVIEW(#663)
2884 Syscall::Setrlimit(s) => self.handle_setrlimit(guest, s).await,
2885
2886 // POSIX per-process timers use the virtual clock and scheduler
2887 // for deterministic arming and supported signal delivery.
2888 Syscall::TimerCreate(s) => self.handle_timer_create(guest, s).await,
2889 Syscall::TimerSettime(s) => self.handle_timer_settime(guest, s).await,
2890 Syscall::TimerGettime(s) => self.handle_timer_gettime(guest, s).await,
2891 Syscall::TimerGetoverrun(s) => self.handle_timer_getoverrun(guest, s).await,
2892 Syscall::TimerDelete(s) => self.handle_timer_delete(guest, s).await,
2893
2894 // Serialized threads share a total memory order, so process-wide
2895 // memory barriers are trivially satisfied and can be no-ops.
2896 Syscall::Membarrier(s) => self.handle_membarrier(guest, s).await,
2897
2898 // Filesystem statistics: passthrough is record/replay-aware so the
2899 // (otherwise host-dependent) result is captured and reproduced.
2900 // statfs/fstatfs run the real syscall, then canonicalize the
2901 // host-varying fields (free blocks/inodes, fsid) so the result is
2902 // deterministic under --verify (a bare passthrough diverged, e.g.
2903 // for tar).
2904 Syscall::Statfs(s) => self.handle_statfs(guest, s).await,
2905 Syscall::Fstatfs(s) => self.handle_fstatfs(guest, s).await,
2906
2907 unexpected => {
2908 self.handle_unsupported_syscall(
2909 guest,
2910 unexpected,
2911 dettid,
2912 panic_on_unsupported_syscalls,
2913 )
2914 .await
2915 }
2916 },
2917 // AUTONOMOUS-BOT-IMPLEMENTED
2918 // TODO-HUMAN-REVIEW(PR-2985): Review scheduler tracking of set_tid_address.
2919 // Linux and the backend own the pass-through call. Detcore also
2920 // mirrors its successful registration into the scheduler because
2921 // the scheduler supplies the logical CHILD_CLEARTID wake.
2922 SyscallClassification::PassThrough if call.number() == Sysno::set_tid_address => {
2923 match call {
2924 Syscall::SetTidAddress(s) => self.handle_set_tid_address(guest, s).await,
2925 _ => unreachable!("set_tid_address unexpectedly lost its typed variant"),
2926 }
2927 }
2928 // AUTONOMOUS-BOT-IMPLEMENTED
2929 // TODO-HUMAN-REVIEW(PR-2223): Review observing
2930 // the robust-list registration without changing its pass-through
2931 // classification. This is still a pass-through — Linux owns the
2932 // registration and supplies its result — but Detcore remembers the
2933 // head address so thread exit can replay `exit_robust_list()`
2934 // against its own futex waiter pool.
2935 SyscallClassification::PassThrough if call.number() == Sysno::set_robust_list => {
2936 match call {
2937 Syscall::SetRobustList(s) => self.handle_set_robust_list(guest, s).await,
2938 _ => self.passthrough(guest, call).await,
2939 }
2940 }
2941 // faccessat2 and fchmodat2 are untyped in the pinned Reverie revision; the
2942 // reviewed classification table routes them, and every other reviewed
2943 // PassThrough syscall, through the blanket arm below.
2944 // AUTONOMOUS-BOT-IMPLEMENTED
2945 // TODO-HUMAN-REVIEW(PR-644): Keep dispatch aligned with the reviewed classification.
2946 SyscallClassification::PassThrough => self.passthrough(guest, call).await,
2947 SyscallClassification::Unsupported => {
2948 self.handle_unsupported_syscall(guest, call, dettid, panic_on_unsupported_syscalls)
2949 .await
2950 }
2951 };
2952
2953 // A copy may already have changed guest memory before reporting a
2954 // terminal backend error. Do not perform even the syscall-result
2955 // display's memory reads, or later observers/posthooks/timer effects.
2956 // Random-device reads have completed their release RPC before returning;
2957 // getrandom acquired no file resource. Physical cleanup belongs to the
2958 // backend failure owner, not to this observer fence.
2959 if res.as_ref().is_err_and(crate::random::is_copy_failure) {
2960 return res;
2961 }
2962
2963 detlog!(
2964 event = crate::detlog::DetLogEvent::SyscallResult {
2965 finished_syscall_number: new_count,
2966 };
2967 "[syscall][detcore, dtid {}] finish syscall #{}: {} = {:?}",
2968 dettid,
2969 new_count,
2970 display_syscall_finished(&call, &guest.memory(), &res),
2971 res
2972 );
2973
2974 // Same guest-logical-control point that already anchors the stack/heap hashes: the
2975 // syscall is complete and its result written back, so the guest logically has control.
2976 // Reading registers is itself backend work, so keep the disabled path inert. In
2977 // particular, a run that does not request register evidence must not be perturbed by
2978 // collecting data that will immediately be discarded.
2979 if self.cfg.detlog_regs {
2980 let control_point_regs = guest.regs().await;
2981 let regs_seq = guest.thread_state().stats.syscall_count;
2982 self.detlog_registers(guest, &control_point_regs, regs_seq);
2983 }
2984
2985 // brk is PassThrough, so nothing else records where the guest's heap is.
2986 // Both brk(NULL) and brk(addr) return the break in effect afterwards.
2987 if let Syscall::Brk(_) = &call
2988 && let Ok(brk) = res
2989 && brk > 0
2990 {
2991 guest
2992 .thread_state()
2993 .memory_metadata
2994 .lock()
2995 .expect("memory metadata mutex poisoned")
2996 .observe_brk(brk as u64);
2997 }
2998
2999 self.detlog_memory_maps(guest)?;
3000 // Same control point again, for the bytes this syscall moved through a guest buffer.
3001 // Unlike the two mapping hashes above, the extent comes from the syscall's OWN
3002 // arguments, so it does not matter whether the buffer lives on the stack, in the brk
3003 // heap, in BSS or in an anonymous mmap -- the last two of which neither mapping hash
3004 // can see. Only successful calls moved anything.
3005 if let Ok(ret) = &res
3006 && self.cfg.detlog_io_buffers
3007 {
3008 io_buffers::detlog_io_buffers(guest, &call, *ret, dettid, rng_readv_output.as_deref())?;
3009 }
3010
3011 if sequentialize_threads && self.cfg.should_trace_schedevent() {
3012 trace_schedevent(
3013 guest,
3014 with_guest_time(
3015 guest,
3016 SchedEvent::syscall(dettid, call.number(), SyscallPhase::Posthook),
3017 ),
3018 true,
3019 )
3020 .await;
3021 }
3022
3023 // The syscall is finished; a turn the post-hook takes (a timeslice
3024 // end) is the thread's own and advances global time as usual.
3025 guest.thread_state_mut().in_uncharged_bootstrap_syscall = false;
3026 self.post_handler_hook(guest).await;
3027
3028 // Defense-in-depth: unless the backend already owns this guarantee,
3029 // force the syscall-clobbered registers (%rcx/%r11 on x86-64) to
3030 // deterministic values before returning to the guest.
3031 if !self.cfg.syscall_clobbers_virtualized_by_backend {
3032 self.canonicalize_syscall_clobbers(guest).await;
3033 }
3034
3035 res
3036 }
3037
3038 async fn on_exit_thread<G: GlobalRPC<Self::GlobalState>>(
3039 &self,
3040 tid: Tid,
3041 global_state: &G,
3042 mut thread_state: Self::ThreadState,
3043 exit_status: ExitStatus,
3044 ) -> Result<(), Error> {
3045 let dettid = thread_state.dettid;
3046 // Cancellation can consume the backend's transferred state before the
3047 // successful-exec callback has rebound its logical identity.
3048 let current = DetTid::from_raw(tid.as_raw());
3049 let transferred_exec = current != dettid;
3050 debug!(
3051 "[detcore, dtid {}] thread exit hook, deregistering from scheduler.",
3052 dettid
3053 );
3054 // Close the final in-progress timeslice so this thread contributes its
3055 // last (partial) slice to the run report, even if it never exhausted a
3056 // full slice.
3057 let now = thread_state.thread_logical_time.as_nanos();
3058 thread_state.stats.close_final_timeslice(now);
3059 // Reverie invokes this callback while the backend still owns the exit
3060 // event, before the guest parent can consume it with wait. Ptrace also
3061 // guarantees that the process leader exits after the other threads, so
3062 // the final published aggregate is complete when wait returns.
3063 // DETERMINISTIC RECOVERY (TODO-HUMAN-REVIEW(PR-1147)). `ThreadState::detpid`
3064 // is `Option` and starts as `None` ("Initialized later" at the clone site),
3065 // so a thread that reaches the exit hook before its per-thread identity is
3066 // populated used to `.expect()` here. That panic fires inside a Reverie
3067 // teardown callback, while the backend still owns the exit event, which is
3068 // the worst place to abort: it can wedge the supervisor rather than fail one
3069 // thread. Fall back to the PROCESS-level `self.detpid`, which is
3070 // non-optional and preserves the identity source selected for this
3071 // backend: virtual for the current DBT ABI, physical for ABI v1. Warn so
3072 // the exceptional window is observable instead of silently papered over.
3073 //
3074 // CORRECTED BY #2348, and stated rather than quietly dropped: the normal
3075 // current-ABI DBT path pre-populates `thread_state.detpid` with the
3076 // client-published virtual process identity, so thread start and exit
3077 // agree on that value. This recovery is only for a thread that exits
3078 // before that initialization. In that exceptional window thread start
3079 // would fall back to `guest.pid()` (the physical host pid for `DbtGuest`),
3080 // while this exit path uses `self.detpid`. ABI v1 also intentionally keeps
3081 // physical callback identities. The warning makes either compatibility
3082 // or recovery path observable; this comment does not claim those fallback
3083 // identities are virtual or independently deterministic.
3084 let (detpid, used_process_detpid) =
3085 select_thread_exit_detpid(thread_state.detpid, self.detpid);
3086 if used_process_detpid {
3087 tracing::warn!(
3088 "[detcore, dtid {}] thread exited before its per-thread detpid was \
3089 initialized; falling back to the process detpid {}",
3090 dettid,
3091 self.detpid
3092 );
3093 }
3094 // The transferred survivor is the final leader even if cancellation
3095 // precedes local rebinding. Its snapshot includes the worker's tail;
3096 // the earlier displaced-leader snapshot is only a prefix.
3097 if current == detpid {
3098 thread_state.record_exited_child_process_cpu_time(detpid);
3099 } else {
3100 thread_state.account_process_cpu_time();
3101 }
3102 let mm_id = thread_state.mm_id;
3103 let exit_signal = match &exit_status {
3104 ExitStatus::Signaled(signal, _) => Some(*signal as i32),
3105 ExitStatus::Exited(_) => None,
3106 };
3107 // Publish each owner's final clock under its own RPC identity before
3108 // counting that owner in the complete physical-exit barrier. Otherwise
3109 // the group's maximum could be charged to whichever callback finishes
3110 // last, followed by a backwards update from its real deregistration.
3111 // Exec preparation already cleared the old image's robust list. An
3112 // unbound transfer must authenticate its consuming cleanup before any
3113 // ordinary RPC can publish the survivor's clock under the leader TID.
3114 let exit_time_accounted = !transferred_exec
3115 && (!thread_state.has_matching_robust_list_exit(exit_signal)
3116 || acknowledge_robust_list_exit_time(
3117 thread_state.thread_logical_time.clone(),
3118 global_state,
3119 mm_id,
3120 )
3121 .await);
3122 if !exit_time_accounted {
3123 // Preserve benign cleanup for a retired incarnation without
3124 // acknowledging an unaccounted owner or releasing its staged wakes.
3125 thread_state.record_robust_list_head(None);
3126 }
3127 if exit_time_accounted
3128 && let Some((_group_time, ready)) = thread_state.take_robust_list_wakes_after_exit(
3129 exit_signal,
3130 thread_state.thread_logical_time.clone(),
3131 )
3132 {
3133 let identities: Vec<_> = ready
3134 .iter()
3135 .map(|(owner, wake)| (*owner, wake.futex))
3136 .collect();
3137 let counts = robust_list_wakes_after_exit(
3138 thread_state.thread_logical_time.clone(),
3139 global_state,
3140 mm_id,
3141 ready,
3142 )
3143 .await;
3144 for ((owner, futex), count) in identities.into_iter().zip(counts) {
3145 info!(
3146 "[detcore, dtid {}] robust-list owner death woke {} waiter(s) on futex {:?} after physical exit",
3147 owner, count, futex,
3148 );
3149 }
3150 }
3151 let pending_chaos_epochs = thread_state.take_pending_chaos_epochs();
3152 let deregistration = ThreadDeregistration {
3153 dettid,
3154 detpid,
3155 mm: mm_id,
3156 thread_start_entered: thread_state.thread_start_entered,
3157 timeslice_stats: thread_state.stats.timeslice_stats,
3158 syscall_count: thread_state.stats.syscall_count,
3159 chaos_epochs: pending_chaos_epochs,
3160 };
3161 if transferred_exec {
3162 tool_global::retire_exec(
3163 thread_state.thread_logical_time.clone(),
3164 global_state,
3165 deregistration,
3166 exit_status.signal().is_some(),
3167 )
3168 .await?;
3169 } else {
3170 deregister_thread(
3171 thread_state.thread_logical_time.clone(),
3172 &self.cfg,
3173 global_state,
3174 deregistration,
3175 )
3176 .await;
3177 }
3178
3179 self.record_or_replay
3180 .on_exit_thread(
3181 tid,
3182 global_state,
3183 thread_state.record_or_replay,
3184 exit_status,
3185 )
3186 .await?;
3187
3188 Ok(())
3189 }
3190}
3191
3192#[cfg(test)]
3193mod subscription_tests {
3194 use super::*;
3195
3196 fn strict_config(passthru_opt: bool) -> Config {
3197 Config {
3198 sequentialize_threads: true,
3199 deterministic_io: true,
3200 passthru_opt,
3201 ..Default::default()
3202 }
3203 }
3204
3205 /// The last row of the pinned table is the one a `Sysno::iter()` sweep
3206 /// drops. `lsm_list_modules` is Determinized AND deterministically refused,
3207 /// so before this was fixed it executed natively against the host under
3208 /// `--passthru-opt` — which is the default for `hermit record` and
3209 /// `hermit replay` — instead of receiving its fixed refusal.
3210 #[test]
3211 fn passthru_opt_covers_the_final_row_of_the_pinned_table() {
3212 let last = Sysno::last();
3213 // Guard the premise: if the table endpoint moves, this test must be
3214 // re-derived rather than silently passing on a different syscall.
3215 assert_eq!(last, Sysno::lsm_list_modules);
3216 assert!(crate::is_determinized_syscall(last));
3217 assert!(crate::is_deterministically_refused_syscall(last));
3218 // The bug this pins: the final row is absent from `Sysno::iter()`.
3219 assert!(!Sysno::iter().any(|sysno| sysno == last));
3220
3221 let subscriptions = <Detcore as Tool>::subscriptions(&strict_config(true));
3222 assert!(
3223 subscriptions.iter_syscalls().any(|sysno| sysno == last),
3224 "{last} must be intercepted under passthru_opt; it is deterministically refused"
3225 );
3226 }
3227
3228 #[test]
3229 fn passthru_opt_intercepts_every_unsupported_syscall() {
3230 let subscriptions = <Detcore as Tool>::subscriptions(&strict_config(true));
3231 let unsupported: Vec<Sysno> = crate::all_pinned_syscalls()
3232 .filter(|sysno| crate::is_unsupported_syscall(*sysno))
3233 .collect();
3234
3235 assert_eq!(unsupported, [Sysno::restart_syscall]);
3236 for syscall in unsupported {
3237 assert!(
3238 subscriptions
3239 .iter_syscalls()
3240 .any(|subscribed| subscribed == syscall),
3241 "passthru_opt allowed unsupported {syscall} to bypass Detcore"
3242 );
3243 }
3244 }
3245
3246 /// `passthru_opt` is not a niche flag: `record_or_replay_config` turns it on
3247 /// for every `hermit record` / `hermit replay`, so this covers the record
3248 /// and replay subscription too.
3249 #[test]
3250 fn passthru_opt_subscribes_every_determinized_syscall() {
3251 let determinized: Vec<Sysno> = crate::all_pinned_syscalls()
3252 .filter(|sysno| crate::is_determinized_syscall(*sysno))
3253 .collect();
3254 let subscriptions = <Detcore as Tool>::subscriptions(&strict_config(true));
3255 let delivered: Vec<Sysno> = subscriptions.iter_syscalls().collect();
3256 let missing = determinized
3257 .iter()
3258 .filter(|sysno| !delivered.contains(sysno))
3259 .copied()
3260 .collect::<Vec<_>>();
3261
3262 assert!(
3263 missing.is_empty(),
3264 "passthru_opt let Determinized syscalls bypass Detcore: {}",
3265 missing
3266 .iter()
3267 .map(|sysno| sysno.to_string())
3268 .collect::<Vec<_>>()
3269 .join(" ")
3270 );
3271 assert!(
3272 delivered.contains(&Sysno::syslog),
3273 "syslog must reach its deterministic Detcore handler"
3274 );
3275 }
3276
3277 #[test]
3278 fn passthru_opt_intercepts_thread_exit_registrations() {
3279 let subscriptions = <Detcore as Tool>::subscriptions(&strict_config(true));
3280 for syscall in [Sysno::set_tid_address, Sysno::set_robust_list] {
3281 assert!(
3282 subscriptions
3283 .iter_syscalls()
3284 .any(|subscribed| subscribed == syscall),
3285 "passthru_opt allowed {syscall} to bypass Detcore's thread-exit state"
3286 );
3287 }
3288 }
3289
3290 #[test]
3291 fn passthru_opt_leaves_unlisted_passthrough_syscalls_unsubscribed() {
3292 assert_eq!(
3293 syscall_classification::classify_syscall(Sysno::chdir),
3294 syscall_classification::SyscallClassification::PassThrough
3295 );
3296
3297 let subscriptions = <Detcore as Tool>::subscriptions(&strict_config(true));
3298 assert!(
3299 !subscriptions
3300 .iter_syscalls()
3301 .any(|sysno| sysno == Sysno::chdir),
3302 "chdir is PassThrough and must remain outside the partial subscription"
3303 );
3304 }
3305
3306 /// Io-buffer hashing can only hash a syscall Detcore is subscribed to.
3307 /// Under `--passthru-opt` the subscription narrows, so the check's coverage
3308 /// narrows with it -- silently, because a syscall that never reaches
3309 /// Detcore produces no record and no record is indistinguishable from a
3310 /// syscall that moved no bytes.
3311 ///
3312 /// Measured 2026-08-21 on `f05bf04e4f`, one probe calling `getcwd`,
3313 /// `recvmsg`, `readv` and `readlink`:
3314 ///
3315 /// ```text
3316 /// hermit --log=info run --detlog-io-buffers -- ./probe 12 [iobuf] records
3317 /// hermit --log=info run --passthru-opt --detlog-io-buffers -- ./probe 11 [iobuf] records
3318 /// ```
3319 ///
3320 /// The single lost syscall is `getcwd`, and the reason is structural: 9 of
3321 /// the 20 are in the literal allow-list and 10 more arrive via the
3322 /// unconditional Determinized sweep at the end of `subscriptions`, but
3323 /// `getcwd` is classified `PassThrough`, so neither path picks it up.
3324 ///
3325 /// WHY THIS IS A TEST AND NOT A RUNTIME WARNING. The set is stable, so a
3326 /// warning would print the same sentence on every `--passthru-opt` run
3327 /// forever and be tuned out within a week. The exposure is not today's
3328 /// one-syscall gap; it is that 19 of 20 holds only because two
3329 /// INDEPENDENTLY MAINTAINED lists happen to agree -- the classification
3330 /// table in `syscall_classification.rs` and the match arms in
3331 /// `io_buffers.rs`. Reclassifying any one of those ten from `Determinized`
3332 /// to `PassThrough` would drop it out of the sweep and out of io-buffers'
3333 /// reach with nothing failing. This asserts the relationship so that edit
3334 /// cannot land quietly.
3335 ///
3336 /// It is deliberately an EQUALITY, not a subset check, so it fails in both
3337 /// directions: a nineteenth syscall going missing, and `getcwd` becoming
3338 /// covered while this expectation still claims it is not.
3339 #[test]
3340 fn passthru_opt_leaves_io_buffer_hashing_blind_only_for_getcwd() {
3341 let subscribed: Vec<Sysno> = <Detcore as Tool>::subscriptions(&strict_config(true))
3342 .iter_syscalls()
3343 .collect();
3344 let unreachable: Vec<Sysno> = crate::io_buffers::HASHED_SYSCALLS
3345 .iter()
3346 .copied()
3347 .filter(|sysno| !subscribed.contains(sysno))
3348 .collect();
3349
3350 assert_eq!(
3351 unreachable,
3352 vec![Sysno::getcwd],
3353 "--passthru-opt changes which syscalls io-buffer hashing can reach, and the set \
3354 moved. {} of {} reachable. If a syscall was RECLASSIFIED out of Determinized, that \
3355 silently shrank an enabled determinism check -- re-derive rather than editing this \
3356 expectation to match.",
3357 crate::io_buffers::HASHED_SYSCALLS.len() - unreachable.len(),
3358 crate::io_buffers::HASHED_SYSCALLS.len()
3359 );
3360 }
3361
3362 /// The other direction, and the reason the check above is worth having:
3363 /// with the default subscription every buffer-carrying syscall is
3364 /// reachable, so there is nothing to report on an ordinary run.
3365 #[test]
3366 fn the_default_subscription_reaches_every_io_buffer_syscall() {
3367 let subscriptions = <Detcore as Tool>::subscriptions(&strict_config(false));
3368 let unreachable: Vec<Sysno> = crate::io_buffers::HASHED_SYSCALLS
3369 .iter()
3370 .copied()
3371 .filter(|sysno| !subscriptions.iter_syscalls().any(|s| s == *sysno))
3372 .collect();
3373
3374 assert!(
3375 unreachable.is_empty(),
3376 "without --passthru-opt every io-buffer syscall must be reachable; missing {unreachable:?}"
3377 );
3378 }
3379
3380 #[test]
3381 fn strict_subscriptions_intercept_every_event_by_default() {
3382 let subscriptions = <Detcore as Tool>::subscriptions(&strict_config(false));
3383
3384 assert_eq!(subscriptions, Subscription::all());
3385 assert!(
3386 subscriptions
3387 .iter_syscalls()
3388 .any(|sysno| sysno == Sysno::ppoll)
3389 );
3390
3391 // On a backend without its own CPUID table (ptrace), `--no-virtualize-cpuid`
3392 // must not subscribe to `cpuid` in either mode. The
3393 // subscription is what makes Reverie probe and enable CPUID faulting, and
3394 // a host without faulting then logs an ERROR for each traced exec
3395 // (https://github.com/rrnewton/hermit/issues/3460). Everything else stays
3396 // as intercepted as it is with CPUID virtualization on.
3397 for passthru_opt in [false, true] {
3398 let virtualized = <Detcore as Tool>::subscriptions(&strict_config(passthru_opt));
3399 let config = Config {
3400 virtualize_cpuid: false,
3401 ..strict_config(passthru_opt)
3402 };
3403 let subscriptions = <Detcore as Tool>::subscriptions(&config);
3404
3405 assert!(virtualized.has_cpuid(), "passthru_opt={passthru_opt}");
3406 assert!(!subscriptions.has_cpuid(), "passthru_opt={passthru_opt}");
3407 assert!(subscriptions.has_rdtsc(), "passthru_opt={passthru_opt}");
3408 assert!(
3409 subscriptions
3410 .iter_syscalls()
3411 .eq(virtualized.iter_syscalls()),
3412 "passthru_opt={passthru_opt}: the syscall set must not depend on CPUID virtualization"
3413 );
3414 }
3415
3416 // KVM installs its own CPUID table, so with virtualization off the guest
3417 // sees host values only through the trap: keep subscribing there.
3418 let kvm_host_cpuid = Config {
3419 virtualize_cpuid: false,
3420 cpuid_virtualized_by_backend: true,
3421 ..strict_config(false)
3422 };
3423 assert_eq!(
3424 <Detcore as Tool>::subscriptions(&kvm_host_cpuid),
3425 Subscription::all()
3426 );
3427 }
3428
3429 #[test]
3430 fn passthru_opt_uses_the_partial_subscription_set() {
3431 let subscriptions = <Detcore as Tool>::subscriptions(&strict_config(true));
3432
3433 assert_ne!(subscriptions, Subscription::all());
3434 assert!(
3435 subscriptions
3436 .iter_syscalls()
3437 .any(|sysno| sysno == Sysno::clock_gettime)
3438 );
3439 assert!(
3440 subscriptions
3441 .iter_syscalls()
3442 .any(|sysno| sysno == Sysno::rt_sigsuspend)
3443 );
3444 assert!(
3445 subscriptions
3446 .iter_syscalls()
3447 .any(|sysno| sysno == Sysno::ppoll)
3448 );
3449 assert!(
3450 subscriptions
3451 .iter_syscalls()
3452 .any(|sysno| sysno == Sysno::madvise)
3453 );
3454 assert!(
3455 subscriptions
3456 .iter_syscalls()
3457 .any(|sysno| sysno == Sysno::arch_prctl)
3458 );
3459 assert!(
3460 subscriptions
3461 .iter_syscalls()
3462 .any(|sysno| sysno == Sysno::writev)
3463 );
3464 for sysno in [
3465 Sysno::read,
3466 Sysno::write,
3467 Sysno::pread64,
3468 Sysno::pwrite64,
3469 Sysno::readv,
3470 Sysno::writev,
3471 Sysno::preadv,
3472 Sysno::preadv2,
3473 Sysno::pwritev,
3474 Sysno::pwritev2,
3475 Sysno::prctl,
3476 ] {
3477 assert!(
3478 subscriptions
3479 .iter_syscalls()
3480 .any(|subscribed| subscribed == sysno),
3481 "timer-slack mediation requires {sysno:?}"
3482 );
3483 }
3484 assert!(
3485 subscriptions
3486 .iter_syscalls()
3487 .any(|sysno| sysno == Sysno::pwrite64)
3488 );
3489 assert!(
3490 subscriptions
3491 .iter_syscalls()
3492 .any(|sysno| sysno == Sysno::pidfd_open)
3493 );
3494 }
3495}
3496
3497#[cfg(test)]
3498mod rcb_overshoot_tests {
3499 use std::fmt::Write;
3500 use std::sync::Arc;
3501 use std::sync::Mutex;
3502
3503 use tracing::Event;
3504 use tracing::Id;
3505 use tracing::Level;
3506 use tracing::Metadata;
3507 use tracing::Subscriber;
3508 use tracing::field::Field;
3509 use tracing::field::Visit;
3510 use tracing::span::Attributes;
3511 use tracing::span::Record;
3512 use tracing::subscriber::with_default;
3513
3514 use super::rcb_timer_overshot;
3515 use super::report_rcb_overshoot;
3516
3517 struct ErrorSubscriber(Arc<Mutex<Option<String>>>);
3518
3519 struct EventVisitor(String);
3520
3521 impl Visit for EventVisitor {
3522 fn record_debug(&mut self, field: &Field, value: &dyn std::fmt::Debug) {
3523 let _ = write!(self.0, "{}={:?}", field.name(), value);
3524 }
3525 }
3526
3527 impl Subscriber for ErrorSubscriber {
3528 fn enabled(&self, metadata: &Metadata<'_>) -> bool {
3529 *metadata.level() == Level::ERROR
3530 }
3531
3532 fn new_span(&self, _span: &Attributes<'_>) -> Id {
3533 Id::from_u64(1)
3534 }
3535
3536 fn record(&self, _span: &Id, _values: &Record<'_>) {}
3537
3538 fn record_follows_from(&self, _span: &Id, _follows: &Id) {}
3539
3540 fn event(&self, event: &Event<'_>) {
3541 if *event.metadata().level() == Level::ERROR {
3542 let mut visitor = EventVisitor(String::new());
3543 event.record(&mut visitor);
3544 *self.0.lock().unwrap() = Some(visitor.0);
3545 }
3546 }
3547
3548 fn enter(&self, _span: &Id) {}
3549
3550 fn exit(&self, _span: &Id) {}
3551 }
3552
3553 #[test]
3554 fn default_overshoot_policy_emits_error_and_returns() {
3555 let _ = reverie::take_skid_overshoot_count();
3556 let error = Arc::new(Mutex::new(None));
3557 with_default(ErrorSubscriber(error.clone()), || {
3558 report_rcb_overshoot(false, 16_249, 139, 100);
3559 });
3560
3561 let error = error.lock().unwrap().take().expect("missing ERROR event");
3562 assert!(error.contains(reverie::SKID_OVERSHOOT_MARKER), "{error}");
3563 assert!(error.contains("PMU RCB overshoot"), "{error}");
3564 assert!(error.contains("16249"), "{error}");
3565 assert!(error.contains("139"), "{error}");
3566 assert!(error.contains("100"), "{error}");
3567 assert_eq!(
3568 reverie::take_skid_overshoot_count(),
3569 1,
3570 "the log-and-continue path must feed the supervisor's structural count"
3571 );
3572 }
3573
3574 #[test]
3575 fn exact_rcb_timer_hit_is_not_an_overshoot() {
3576 assert!(!rcb_timer_overshot(100, 100));
3577 assert!(!rcb_timer_overshot(99, 100));
3578 assert!(rcb_timer_overshot(101, 100));
3579 }
3580
3581 #[test]
3582 #[should_panic(expected = "PMU RCB overshoot")]
3583 fn opt_in_overshoot_policy_panics() {
3584 report_rcb_overshoot(true, 16_249, 139, 100);
3585 }
3586}
3587
3588#[cfg(test)]
3589mod timeslice_timer_tests {
3590 use super::*;
3591
3592 #[test]
3593 fn manual_interrupts_can_shorten_but_not_extend_maximum() {
3594 assert_eq!(choose_rcb_timer(100, 100, Some(150)), (50, false));
3595 assert_eq!(choose_rcb_timer(100, 100, Some(250)), (100, true));
3596 assert_eq!(choose_rcb_timer(100, 100, None), (100, true));
3597 }
3598
3599 #[test]
3600 fn pmu_duration_conversion_applies_clock_multiplier() {
3601 let duration = crate::types::LogicalTime::from_nanos(100);
3602 assert_eq!(duration.into_rcbs_with_multiplier(2.0), 5);
3603 assert_eq!(duration.into_rcbs_with_multiplier(0.5), 20);
3604 assert_eq!(
3605 crate::types::LogicalTime::from_nanos(101).into_rcbs_with_multiplier(2.0),
3606 5
3607 );
3608 }
3609
3610 #[test]
3611 #[should_panic(expected = "max_timeslice must be at least one RCB")]
3612 fn detcore_constructor_validates_programmatic_config() {
3613 let config = Config {
3614 max_timeslice: std::num::NonZeroU64::new(1),
3615 ..Default::default()
3616 };
3617
3618 let _ = <Detcore as Tool>::new(Pid::from_raw(1), &config);
3619 }
3620}
3621
3622#[cfg(test)]
3623mod process_tree_guest_clock_tests {
3624 use super::*;
3625
3626 #[test]
3627 fn forked_process_shares_guest_clock_domain() {
3628 let config = Config::default();
3629 let tool = <Detcore as Tool>::new(Pid::from_raw(1), &config);
3630 let mut parent = ThreadState::new(DetPid::from_raw(1), &config, ());
3631 parent.clone_flags = Some(CloneFlags::empty());
3632
3633 let child = <Detcore as Tool>::init_thread_state(
3634 &tool,
3635 Tid::from_raw(2),
3636 Some((Tid::from_raw(1), &parent)),
3637 );
3638
3639 assert!(Arc::ptr_eq(&parent.guest_clock, &child.guest_clock));
3640 }
3641}
3642
3643#[cfg(test)]
3644mod child_rng_identity_tests {
3645 use super::*;
3646
3647 fn child_state(
3648 tool: &Detcore,
3649 parent: &mut ThreadState<()>,
3650 host_tid: i32,
3651 clone_flags: CloneFlags,
3652 ) -> ThreadState<()> {
3653 parent.clone_flags = Some(clone_flags);
3654 <Detcore as Tool>::init_thread_state(
3655 tool,
3656 Tid::from_raw(host_tid),
3657 Some((Tid::from_raw(parent.dettid.as_raw()), parent)),
3658 )
3659 }
3660
3661 fn rng_sample(child: &ThreadState<()>) -> [u64; 4] {
3662 let mut rng = child.prng.clone();
3663 std::array::from_fn(|_| rng.random())
3664 }
3665
3666 fn chaos_rng_sample(child: &ThreadState<()>) -> [u64; 4] {
3667 let mut rng = child.chaos_prng.clone();
3668 std::array::from_fn(|_| rng.random())
3669 }
3670
3671 #[test]
3672 fn common_child_rng_identity_ignores_backend_tid_for_every_clone_shape() {
3673 let config = Config::default();
3674 let tool = <Detcore as Tool>::new(Pid::from_raw(1), &config);
3675 for (label, clone_flags) in [
3676 ("fork", CloneFlags::empty()),
3677 ("vfork", CloneFlags::CLONE_VM | CloneFlags::CLONE_VFORK),
3678 ("clone-process", CloneFlags::CLONE_VM),
3679 (
3680 "clone-thread",
3681 CloneFlags::CLONE_VM | CloneFlags::CLONE_THREAD,
3682 ),
3683 ] {
3684 let mut low_tid_parent = ThreadState::new(DetPid::from_raw(1), &config, ());
3685 let mut high_tid_parent = ThreadState::new(DetPid::from_raw(1), &config, ());
3686 let low_tid_child = child_state(&tool, &mut low_tid_parent, 2, clone_flags);
3687 let high_tid_child = child_state(&tool, &mut high_tid_parent, 42_002, clone_flags);
3688
3689 assert_eq!(format!("{}", low_tid_child.pedigree), "C", "{label}");
3690 assert_eq!(low_tid_child.dettid.as_raw(), 2, "{label}");
3691 assert_eq!(high_tid_child.dettid.as_raw(), 42_002, "{label}");
3692 assert_eq!(
3693 rng_sample(&low_tid_child),
3694 rng_sample(&high_tid_child),
3695 "{label} child RNG depended on the backend Tid"
3696 );
3697 assert_eq!(
3698 chaos_rng_sample(&low_tid_child),
3699 chaos_rng_sample(&high_tid_child),
3700 "{label} child chaos RNG depended on the backend Tid"
3701 );
3702 assert_ne!(
3703 rng_sample(&low_tid_child),
3704 chaos_rng_sample(&low_tid_child),
3705 "{label} child guest and chaos RNG streams were coupled"
3706 );
3707 }
3708 }
3709
3710 #[test]
3711 fn serialized_siblings_receive_distinct_child_rng_streams() {
3712 let config = Config::default();
3713 let tool = <Detcore as Tool>::new(Pid::from_raw(1), &config);
3714 let mut parent = ThreadState::new(DetPid::from_raw(1), &config, ());
3715
3716 let first = child_state(&tool, &mut parent, 2, CloneFlags::empty());
3717 let first_pedigree = parent.pedigree.fork_mut();
3718 assert_eq!(format!("{}", first.pedigree), format!("{first_pedigree}"));
3719 let second = child_state(&tool, &mut parent, 3, CloneFlags::empty());
3720
3721 assert_eq!(format!("{}", second.pedigree), "PC");
3722 assert_ne!(rng_sample(&first), rng_sample(&second));
3723 }
3724}
3725
3726#[cfg(test)]
3727mod thread_exit_identity_tests {
3728 use super::*;
3729
3730 #[test]
3731 fn initialized_thread_exit_identity_is_preserved() {
3732 let thread_detpid = DetPid::from_raw(41);
3733 let process_detpid = DetPid::from_raw(7);
3734 assert_eq!(
3735 select_thread_exit_detpid(Some(thread_detpid), process_detpid),
3736 (thread_detpid, false)
3737 );
3738 }
3739
3740 #[test]
3741 fn missing_thread_exit_identity_uses_process_identity() {
3742 let process_detpid = DetPid::from_raw(7);
3743 assert_eq!(
3744 select_thread_exit_detpid(None, process_detpid),
3745 (process_detpid, true)
3746 );
3747 }
3748}
3749
3750#[cfg(test)]
3751mod thread_start_identity_tests {
3752 use super::*;
3753
3754 #[test]
3755 fn backend_process_identity_is_preserved() {
3756 let backend_detpid = DetPid::from_raw(3);
3757 assert_eq!(
3758 select_thread_start_detpid(Some(backend_detpid), Pid::from_raw(42_001)),
3759 backend_detpid
3760 );
3761 assert_eq!(
3762 select_thread_start_detpid(None, Pid::from_raw(42_001)),
3763 DetPid::from_raw(42_001)
3764 );
3765 }
3766
3767 #[test]
3768 fn root_thread_uses_deterministic_process_identity() {
3769 let detpid = DetPid::from_raw(3);
3770 assert!(is_root_thread_start(true, DetTid::from_raw(3), detpid));
3771 assert!(!is_root_thread_start(false, DetTid::from_raw(3), detpid));
3772 assert!(!is_root_thread_start(true, DetTid::from_raw(4), detpid));
3773 }
3774}
3775
3776#[cfg(test)]
3777mod thread_cpu_time_tests {
3778 use super::*;
3779
3780 #[test]
3781 fn cloned_threads_and_processes_keep_absolute_time_without_inheriting_work() {
3782 for clone_flags in [
3783 CloneFlags::CLONE_THREAD,
3784 CloneFlags::empty(),
3785 CloneFlags::CLONE_VFORK | CloneFlags::CLONE_VM,
3786 ] {
3787 let config = Config::default();
3788 let tool = <Detcore as Tool>::new(Pid::from_raw(1), &config);
3789 let mut parent = ThreadState::new(DetPid::from_raw(1), &config, ());
3790 parent.thread_logical_time.add_rcbs(7);
3791 parent.thread_logical_time.add_syscall_with_cost(101);
3792 parent.clone_flags = Some(clone_flags);
3793 let child = <Detcore as Tool>::init_thread_state(
3794 &tool,
3795 Tid::from_raw(2),
3796 Some((Tid::from_raw(1), &parent)),
3797 );
3798 assert_eq!(
3799 child.thread_logical_time.as_nanos(),
3800 parent.thread_logical_time.as_nanos()
3801 );
3802 assert_eq!(
3803 child.thread_logical_time.inherited_nanos(),
3804 LogicalTime::from_nanos(171)
3805 );
3806 let mut global = GlobalTime::new(&config);
3807 let epoch = global.as_nanos();
3808 for thread in [&parent, &child] {
3809 global.update_global_time(
3810 thread.dettid,
3811 thread.thread_logical_time.as_nanos(),
3812 thread.thread_logical_time.inherited_nanos(),
3813 );
3814 }
3815 assert_eq!(global.as_nanos(), epoch + LogicalTime::from_nanos(171));
3816 }
3817 }
3818
3819 #[test]
3820 fn cloned_thread_and_fork_child_start_with_zero_thread_cpu() {
3821 for clone_flags in [CloneFlags::CLONE_THREAD, CloneFlags::empty()] {
3822 let config = Config::default();
3823 let tool = <Detcore as Tool>::new(Pid::from_raw(1), &config);
3824 let mut parent = ThreadState::new(DetPid::from_raw(1), &config, ());
3825
3826 // Give the parent nonzero CPU before the child exists. The child's
3827 // absolute logical clock inherits this position for scheduler
3828 // ordering, but Linux per-thread CPU accounting must not.
3829 parent.thread_logical_time.add_rcbs(200);
3830 parent.thread_logical_time.add_syscall();
3831 parent.clone_flags = Some(clone_flags);
3832
3833 let mut child = <Detcore as Tool>::init_thread_state(
3834 &tool,
3835 Tid::from_raw(2),
3836 Some((Tid::from_raw(1), &parent)),
3837 );
3838 assert_eq!(
3839 child.thread_cpu_time(),
3840 (LogicalTime::ZERO, LogicalTime::ZERO),
3841 "clone flags {clone_flags:?} inherited pre-creation CPU"
3842 );
3843
3844 child.thread_logical_time.add_rcbs(4);
3845 child.thread_logical_time.add_syscall();
3846 let (user, system) = child.thread_cpu_time();
3847 assert!(user > LogicalTime::ZERO);
3848 assert!(system > LogicalTime::ZERO);
3849 assert!(user < parent.thread_logical_time.user_cpu_time());
3850 assert!(system <= parent.thread_logical_time.system_cpu_time());
3851 }
3852 }
3853}
3854
3855/// Regression tests for <https://github.com/rrnewton/hermit/issues/3153>: a
3856/// failed syscall must not render an output buffer the kernel never wrote.
3857///
3858/// Each buffer is pre-filled with sentinel values standing in for the
3859/// uninitialized guest stack the issue observed, and the rendering is checked
3860/// through the same `finish syscall` format the DETLOG line uses.
3861#[cfg(test)]
3862mod finished_syscall_display_tests {
3863 use std::ffi::CString;
3864
3865 use reverie::syscalls::AddrMut;
3866 use reverie::syscalls::AtFlags;
3867 use reverie::syscalls::ClockGettime;
3868 use reverie::syscalls::ClockId;
3869 use reverie::syscalls::Gettimeofday;
3870 use reverie::syscalls::LocalMemory;
3871 use reverie::syscalls::Newfstatat;
3872 use reverie::syscalls::PathPtr;
3873 use reverie::syscalls::StatPtr;
3874 use reverie::syscalls::Statx;
3875 use reverie::syscalls::StatxMask;
3876 use reverie::syscalls::StatxPtr;
3877 use reverie::syscalls::Timespec;
3878 use reverie::syscalls::TimespecMutPtr;
3879 use reverie::syscalls::Timeval;
3880 use reverie::syscalls::TimevalMutPtr;
3881
3882 use super::*;
3883
3884 /// A value no real `st_size` or timestamp in these tests can take.
3885 const SENTINEL: i64 = 0x5EED_0BAD_F00D;
3886
3887 /// Render exactly what the `finish syscall` DETLOG line renders.
3888 fn finish_line(syscall: &Syscall, result: Result<i64, Error>) -> String {
3889 let memory = LocalMemory::new();
3890 format!(
3891 "{} = {:?}",
3892 display_syscall_finished(syscall, &memory, &result),
3893 result
3894 )
3895 }
3896
3897 fn sentinel_stat() -> libc::stat {
3898 // SAFETY: all-zero is a valid `struct stat`.
3899 let mut stat: libc::stat = unsafe { std::mem::zeroed() };
3900 stat.st_mode = libc::S_IFREG | 0o644;
3901 stat.st_size = SENTINEL;
3902 stat
3903 }
3904
3905 fn sentinel_statx() -> libc::statx {
3906 // SAFETY: all-zero is a valid `struct statx`.
3907 let mut statx: libc::statx = unsafe { std::mem::zeroed() };
3908 statx.stx_mode = (libc::S_IFREG | 0o644) as u16;
3909 statx.stx_size = SENTINEL as u64;
3910 statx
3911 }
3912
3913 fn newfstatat(path: &CString, stat: &libc::stat) -> Syscall {
3914 Syscall::Newfstatat(
3915 Newfstatat::new()
3916 .with_dirfd(libc::AT_FDCWD)
3917 .with_path(PathPtr::from_ptr(path.as_ptr()))
3918 .with_stat(StatPtr::from_ptr(stat as *const libc::stat))
3919 .with_flags(AtFlags::empty()),
3920 )
3921 }
3922
3923 fn statx(path: &CString, statx: &libc::statx) -> Syscall {
3924 Syscall::Statx(
3925 Statx::new()
3926 .with_dirfd(libc::AT_FDCWD)
3927 .with_path(PathPtr::from_ptr(path.as_ptr()))
3928 .with_flags(AtFlags::empty())
3929 .with_mask(StatxMask::STATX_BASIC_STATS)
3930 .with_statx(StatxPtr::from_ptr(statx as *const libc::statx)),
3931 )
3932 }
3933
3934 fn assert_no_struct_rendered(line: &str) {
3935 assert!(!line.contains("st_mode"), "rendered st_mode: {line}");
3936 assert!(!line.contains("st_size"), "rendered st_size: {line}");
3937 assert!(
3938 !line.contains(&SENTINEL.to_string()),
3939 "rendered the unwritten sentinel: {line}"
3940 );
3941 }
3942
3943 #[test]
3944 fn failed_newfstatat_renders_pointer_and_errno_but_not_the_buffer() {
3945 let path = CString::new("/nonexistent/issue-3153").unwrap();
3946 let stat = sentinel_stat();
3947 let line = finish_line(&newfstatat(&path, &stat), Err(Errno::ENOENT.into()));
3948
3949 assert_no_struct_rendered(&line);
3950 assert!(
3951 line.contains(&format!("{:p}", &stat as *const libc::stat)),
3952 "stat pointer missing: {line}"
3953 );
3954 assert!(
3955 line.contains("\"/nonexistent/issue-3153\""),
3956 "path missing: {line}"
3957 );
3958 assert!(line.contains("ENOENT"), "errno missing: {line}");
3959 }
3960
3961 #[test]
3962 fn successful_newfstatat_still_renders_the_buffer() {
3963 let path = CString::new("/dev/null").unwrap();
3964 let stat = sentinel_stat();
3965 let line = finish_line(&newfstatat(&path, &stat), Ok(0));
3966
3967 assert!(
3968 line.contains(&format!(
3969 "{:p} -> {{st_mode=SFlag(S_IFREG) | 0644, st_size={}, ...}}",
3970 &stat as *const libc::stat, SENTINEL
3971 )),
3972 "stat struct missing: {line}"
3973 );
3974 }
3975
3976 #[test]
3977 fn failed_statx_renders_pointer_and_errno_but_not_the_buffer() {
3978 let path = CString::new("/nonexistent/issue-3153").unwrap();
3979 let buf = sentinel_statx();
3980 let line = finish_line(&statx(&path, &buf), Err(Errno::ENOENT.into()));
3981
3982 assert_no_struct_rendered(&line);
3983 assert!(
3984 line.contains(&format!("{:p}", &buf as *const libc::statx)),
3985 "statx pointer missing: {line}"
3986 );
3987 assert!(line.contains("ENOENT"), "errno missing: {line}");
3988 }
3989
3990 #[test]
3991 fn successful_statx_still_renders_the_buffer() {
3992 let path = CString::new("/dev/null").unwrap();
3993 let buf = sentinel_statx();
3994 let line = finish_line(&statx(&path, &buf), Ok(0));
3995
3996 assert!(
3997 line.contains(&format!(
3998 "{:p} -> {{st_mode=SFlag(S_IFREG) | 0644, st_size={}, ...}}",
3999 &buf as *const libc::statx, SENTINEL
4000 )),
4001 "statx struct missing: {line}"
4002 );
4003 }
4004
4005 #[test]
4006 fn failed_clock_gettime_does_not_render_the_timespec() {
4007 let tp = Timespec {
4008 tv_sec: SENTINEL,
4009 tv_nsec: 0,
4010 };
4011 let call = Syscall::ClockGettime(
4012 ClockGettime::new()
4013 .with_clockid(ClockId::CLOCK_MONOTONIC)
4014 .with_tp(Some(TimespecMutPtr(
4015 AddrMut::from_ptr(&tp as *const Timespec).unwrap(),
4016 ))),
4017 );
4018
4019 let failed = finish_line(&call, Err(Errno::EINVAL.into()));
4020 assert!(!failed.contains("tv_sec"), "rendered tv_sec: {failed}");
4021 assert!(
4022 failed.contains(&format!("{:p}", &tp as *const Timespec)),
4023 "tp pointer missing: {failed}"
4024 );
4025 assert!(failed.contains("EINVAL"), "errno missing: {failed}");
4026
4027 let succeeded = finish_line(&call, Ok(0));
4028 assert!(
4029 succeeded.contains(&format!("tv_sec: {SENTINEL}")),
4030 "timespec missing on success: {succeeded}"
4031 );
4032 }
4033
4034 /// EFAULT is also the error of a copy-out that faulted part way, after
4035 /// `copy_to_user()` stored a prefix the guest can read, so it must keep
4036 /// the buffer rendered for every allowlisted syscall.
4037 #[test]
4038 fn efault_failures_still_render_a_possible_partial_copy() {
4039 let path = CString::new("/dev/null").unwrap();
4040 let stat = sentinel_stat();
4041 let line = finish_line(&newfstatat(&path, &stat), Err(Errno::EFAULT.into()));
4042 assert!(
4043 line.contains(&format!(
4044 "{:p} -> {{st_mode=SFlag(S_IFREG) | 0644, st_size={}, ...}}",
4045 &stat as *const libc::stat, SENTINEL
4046 )),
4047 "partial stat copy hidden on EFAULT: {line}"
4048 );
4049 assert!(line.contains("EFAULT"), "errno missing: {line}");
4050
4051 let buf = sentinel_statx();
4052 let line = finish_line(&statx(&path, &buf), Err(Errno::EFAULT.into()));
4053 assert!(
4054 line.contains(&format!(
4055 "{:p} -> {{st_mode=SFlag(S_IFREG) | 0644, st_size={}, ...}}",
4056 &buf as *const libc::statx, SENTINEL
4057 )),
4058 "partial statx copy hidden on EFAULT: {line}"
4059 );
4060
4061 let tp = Timespec {
4062 tv_sec: SENTINEL,
4063 tv_nsec: 0,
4064 };
4065 let call = Syscall::ClockGettime(
4066 ClockGettime::new()
4067 .with_clockid(ClockId::CLOCK_MONOTONIC)
4068 .with_tp(Some(TimespecMutPtr(
4069 AddrMut::from_ptr(&tp as *const Timespec).unwrap(),
4070 ))),
4071 );
4072 let line = finish_line(&call, Err(Errno::EFAULT.into()));
4073 assert!(
4074 line.contains(&format!("tv_sec: {SENTINEL}")),
4075 "partial timespec copy hidden on EFAULT: {line}"
4076 );
4077 }
4078
4079 /// A tool error is not the syscall's result and proves nothing about the
4080 /// buffer, so only a guest errno other than EFAULT hides it.
4081 #[test]
4082 fn only_non_efault_guest_errnos_prove_the_buffer_unwritten() {
4083 for errno in [Errno::ENOENT, Errno::EINVAL, Errno::EBADF, Errno::ENOMEM] {
4084 assert!(failure_proves_outputs_unwritten(&Err(errno.into())));
4085 }
4086 assert!(!failure_proves_outputs_unwritten(
4087 &Err(Errno::EFAULT.into())
4088 ));
4089 assert!(!failure_proves_outputs_unwritten(&Ok(0)));
4090 assert!(!failure_proves_outputs_unwritten(&Err(Error::Tool(
4091 anyhow::anyhow!("tool failure")
4092 ))));
4093
4094 let path = CString::new("/dev/null").unwrap();
4095 let stat = sentinel_stat();
4096 let line = finish_line(
4097 &newfstatat(&path, &stat),
4098 Err(Error::Tool(anyhow::anyhow!("tool failure"))),
4099 );
4100 assert!(
4101 line.contains(&format!("st_size={SENTINEL}")),
4102 "buffer hidden on a tool error: {line}"
4103 );
4104 }
4105
4106 /// `gettimeofday` stores `tv` before it can fail with `EFAULT` on a bad
4107 /// `tz`, so a failed call can carry a kernel-written output. It must stay
4108 /// rendered: suppressing it would hide real evidence from DETLOG.
4109 #[test]
4110 fn failed_gettimeofday_still_renders_the_timeval() {
4111 let tv = Timeval {
4112 tv_sec: SENTINEL,
4113 tv_usec: 0,
4114 };
4115 let call = Syscall::Gettimeofday(
4116 Gettimeofday::new()
4117 .with_tv(Some(TimevalMutPtr(
4118 AddrMut::from_ptr(&tv as *const Timeval).unwrap(),
4119 )))
4120 .with_tz(None),
4121 );
4122
4123 let line = finish_line(&call, Err(Errno::EFAULT.into()));
4124 assert!(
4125 line.contains(&format!("tv_sec: {SENTINEL}")),
4126 "kernel-written timeval hidden on failure: {line}"
4127 );
4128 assert!(line.contains("EFAULT"), "errno missing: {line}");
4129 }
4130
4131 /// `fstat` never renders its buffer (T136880615); the result must not
4132 /// change that in either direction.
4133 #[test]
4134 fn fstat_never_renders_the_buffer() {
4135 let stat = sentinel_stat();
4136 let call = Syscall::Fstat(
4137 reverie::syscalls::Fstat::new()
4138 .with_fd(3)
4139 .with_stat(StatPtr::from_ptr(&stat as *const libc::stat)),
4140 );
4141 for result in [Ok(0), Err(Errno::EBADF.into())] {
4142 assert_no_struct_rendered(&finish_line(&call, result));
4143 }
4144 }
4145}