Skip to main content

detcore/syscalls/
misc.rs

1/*
2 * Copyright (c) Meta Platforms, Inc. and affiliates.
3 * All rights reserved.
4 *
5 * This source code is licensed under the BSD-style license found in the
6 * LICENSE file in the root directory of this source tree.
7 */
8
9//! Miscellaneous virtualized syscalls.
10
11use std::collections::hash_map::DefaultHasher;
12use std::hash::Hash;
13use std::hash::Hasher;
14
15use reverie::Error;
16use reverie::Guest;
17use reverie::syscalls;
18use reverie::syscalls::AddrMut;
19use reverie::syscalls::ArchPrctlCmd;
20use reverie::syscalls::Errno;
21use reverie::syscalls::MemoryAccess;
22
23use crate::consts::DEFAULT_HOSTNAME;
24use crate::detlog;
25#[cfg(test)]
26use crate::random::GETRANDOM_MAX_BYTES;
27use crate::random::RANDOM_FILL_CHUNK_BYTES;
28#[cfg(test)]
29use crate::random::getrandom_request_len;
30#[cfg(test)]
31use crate::random::validate_getrandom_flags;
32use crate::random::write_random_chunk;
33use crate::record_or_replay::RecordOrReplay;
34use crate::tool_global::create_session;
35use crate::tool_global::set_process_group;
36use crate::tool_local::Detcore;
37use crate::types::DetPid;
38
39const ARCH_GET_XCOMP_SUPP: libc::c_int = 0x1021;
40const ARCH_GET_XCOMP_PERM: libc::c_int = 0x1022;
41const ARCH_REQ_XCOMP_PERM: libc::c_int = 0x1023;
42const ARCH_GET_XCOMP_GUEST_PERM: libc::c_int = 0x1024;
43const ARCH_REQ_XCOMP_GUEST_PERM: libc::c_int = 0x1025;
44
45const ARCH_SHSTK_ENABLE: libc::c_int = 0x5001;
46const ARCH_SHSTK_DISABLE: libc::c_int = 0x5002;
47const ARCH_SHSTK_LOCK: libc::c_int = 0x5003;
48const ARCH_SHSTK_UNLOCK: libc::c_int = 0x5004;
49const ARCH_SHSTK_STATUS: libc::c_int = 0x5005;
50const ARCH_SHSTK_VALID_MASK: usize = 0b11;
51
52const SECCOMP_SET_MODE_STRICT: u32 = 0;
53const SECCOMP_SET_MODE_FILTER: u32 = 1;
54const SECCOMP_GET_ACTION_AVAIL: u32 = 2;
55const SECCOMP_GET_NOTIF_SIZES: u32 = 3;
56const SECCOMP_FILTER_FLAG_TSYNC: u32 = 1;
57
58fn seccomp_result(op: u32, flags: u32, has_args: bool) -> Result<i64, Errno> {
59    if op > SECCOMP_GET_NOTIF_SIZES {
60        return Err(Errno::EINVAL);
61    }
62    if op == SECCOMP_SET_MODE_STRICT && (flags != 0 || has_args) {
63        return Err(Errno::EINVAL);
64    }
65    if op == SECCOMP_SET_MODE_FILTER && flags & !SECCOMP_FILTER_FLAG_TSYNC != 0 {
66        return Err(Errno::EINVAL);
67    }
68    if matches!(
69        op,
70        SECCOMP_SET_MODE_FILTER | SECCOMP_GET_ACTION_AVAIL | SECCOMP_GET_NOTIF_SIZES
71    ) && !has_args
72    {
73        return Err(Errno::EFAULT);
74    }
75
76    // Hermit cannot enforce a guest-installed BPF policy in every backend.
77    // Report the operation as unsupported instead of claiming a filter exists.
78    Err(Errno::EOPNOTSUPP)
79}
80
81fn is_supported_prctl_option(option: libc::c_int) -> bool {
82    matches!(
83        option,
84        libc::PR_SET_NAME
85            | libc::PR_GET_NAME
86            | libc::PR_SET_THP_DISABLE
87            | libc::PR_GET_THP_DISABLE
88            // AUTONOMOUS-BOT-IMPLEMENTED
89            // TODO-HUMAN-REVIEW(#919)
90            //
91            // Dumpability is per-process state initialized deterministically by
92            // ordinary exec and subsequently controlled only by the guest. Keep
93            // Linux's 0/1 validation and state transitions rather than refusing
94            // applications that explicitly disable or restore core-dump access.
95            | libc::PR_SET_DUMPABLE
96            | libc::PR_GET_DUMPABLE
97            // PR_SET_TIMERSLACK and PR_GET_TIMERSLACK are deliberately absent:
98            // `handle_prctl` emulates them against per-thread Detcore state.
99            // AUTONOMOUS-BOT-IMPLEMENTED
100            // TODO-HUMAN-REVIEW(#802)
101            //
102            // PR_{SET,GET}_KEEPCAPS only read/toggle the calling thread's
103            // "keep capabilities across a UID change" flag. The result is a pure
104            // function of the guest's own prior prctl calls (0/1), never host
105            // state, so passthrough is deterministic and bitwise-identical across
106            // runs. Supporting it lets `setpriv` (and the `date`/privilege-drop
107            // wrappers that call it) run under --strict instead of aborting with
108            // "keep process capabilities failed: Function not implemented".
109            | libc::PR_SET_KEEPCAPS
110            | libc::PR_GET_KEEPCAPS
111            // AUTONOMOUS-BOT-IMPLEMENTED
112            // TODO-HUMAN-REVIEW(#824)
113            //
114            // PR_{SET,GET}_PDEATHSIG only read/toggle the calling thread's
115            // parent-death-signal attribute. PR_SET_PDEATHSIG validates its
116            // signal argument (a valid signal or 0 succeeds, anything else
117            // faults EINVAL) and PR_GET_PDEATHSIG reports the value the guest
118            // previously set. The result is a pure function of the guest's own
119            // prior prctl calls and its argument, never host state, so
120            // passthrough is deterministic and bitwise-identical across runs.
121            // The registered signal only ever fires on parent death, which is a
122            // deterministically scheduled event under Hermit. Supporting it lets
123            // `setpriv --pdeathsig` run under --strict instead of aborting with
124            // "set parent death signal failed: Function not implemented".
125            | libc::PR_SET_PDEATHSIG
126            | libc::PR_GET_PDEATHSIG
127    )
128}
129
130// AUTONOMOUS-BOT-IMPLEMENTED
131// TODO-HUMAN-REVIEW(PR-1125): Review KVM capability-control prctl forwarding.
132fn is_backend_virtualized_capability_prctl(option: libc::c_int) -> bool {
133    matches!(option, libc::PR_CAPBSET_DROP | libc::PR_CAP_AMBIENT)
134}
135
136/// Is `which` one of the Linux `PRIO_*` target selectors for get/setpriority?
137fn is_valid_prio_which(which: i32) -> bool {
138    which == libc::PRIO_PROCESS as i32
139        || which == libc::PRIO_PGRP as i32
140        || which == libc::PRIO_USER as i32
141}
142
143// AUTONOMOUS-BOT-IMPLEMENTED
144// TODO-HUMAN-REVIEW(#806)
145/// Deterministic raw `getpriority(2)` result under Hermit's inert-nice model.
146///
147/// Returns `20 - nice` (nice 0 -> 20) for any valid target regardless of `who`,
148/// because nice is inert under the virtualized scheduler and `getpriority` never
149/// checks permissions; an unknown `which` faults with `EINVAL`, matching Linux.
150fn getpriority_result(which: i32) -> Result<i64, Errno> {
151    if is_valid_prio_which(which) {
152        Ok(20)
153    } else {
154        Err(Errno::EINVAL)
155    }
156}
157
158// AUTONOMOUS-BOT-IMPLEMENTED
159// TODO-HUMAN-REVIEW(#806)
160/// Deterministic raw `setpriority(2)` result: accept any priority change for a
161/// valid target as an inert no-op (returns 0), `EINVAL` for an unknown `which`.
162fn setpriority_result(which: i32) -> Result<i64, Errno> {
163    if is_valid_prio_which(which) {
164        Ok(0)
165    } else {
166        Err(Errno::EINVAL)
167    }
168}
169
170// AUTONOMOUS-BOT-IMPLEMENTED
171// TODO-HUMAN-REVIEW(PR-857): Deterministic empty kernel-log action policy.
172fn deterministic_syslog_result(action: i32, len: usize) -> Result<i64, Errno> {
173    const SYSLOG_ACTION_CONSOLE_LEVEL: i32 = 8;
174
175    match action {
176        0..=7 | 9..=10 => Ok(0),
177        SYSLOG_ACTION_CONSOLE_LEVEL if (1..=8).contains(&len) => Ok(0),
178        _ => Err(Errno::EINVAL),
179    }
180}
181
182fn from_str(s: &str) -> [i8; 65] {
183    let mut ret: [i8; 65] = [0; 65];
184    for (i, ch) in s.bytes().take(64).enumerate() {
185        ret[i] = ch as i8;
186    }
187    ret
188}
189
190const RANDOM_DEVICE_BYTE_STRIDE: u8 = 73;
191const RANDOM_DEVICE_FIRST_BYTE: u8 = 41;
192
193// AUTONOMOUS-BOT-IMPLEMENTED
194// TODO-HUMAN-REVIEW(PR-1096): Review the backend-independent random-device stream.
195fn canonical_random_device_byte(seed: u64, index: u64) -> u8 {
196    let seed_byte = seed.rotate_right(((index % 8) * 8) as u32) as u8;
197    (index as u8)
198        .wrapping_mul(RANDOM_DEVICE_BYTE_STRIDE)
199        .wrapping_add(RANDOM_DEVICE_FIRST_BYTE)
200        ^ seed_byte
201}
202
203/// Scatter one stream through an already-imported array. Scratch space is
204/// independent of request size; only EFAULT becomes a successful copied prefix.
205fn fill_canonical_random_iovecs(
206    memory: &mut impl MemoryAccess,
207    iovecs: &[crate::iovecs::ImportedIovec],
208    seed: u64,
209    stream_offset: u64,
210    hasher: &mut DefaultHasher,
211) -> Result<usize, Error> {
212    let mut local_words = [0_u64; RANDOM_FILL_CHUNK_BYTES / std::mem::size_of::<u64>()];
213    let mut written = 0_usize;
214    for iov in iovecs {
215        let mut segment_written = 0;
216        while segment_written < iov.len {
217            let remote_chunk = match iov
218                .base
219                .checked_add(segment_written)
220                .and_then(AddrMut::<u8>::from_raw)
221            {
222                Some(address) => address,
223                None if written == 0 => return Err(Errno::EFAULT.into()),
224                None => return Ok(written),
225            };
226            let chunk_len = (iov.len - segment_written).min(RANDOM_FILL_CHUNK_BYTES);
227            let local_buf = unsafe {
228                std::slice::from_raw_parts_mut(local_words.as_mut_ptr().cast::<u8>(), chunk_len)
229            };
230            for (index, byte) in local_buf.iter_mut().enumerate() {
231                *byte = canonical_random_device_byte(
232                    seed,
233                    stream_offset
234                        .saturating_add(written as u64)
235                        .saturating_add(index as u64),
236                );
237            }
238            let n = match write_random_chunk(memory, remote_chunk, local_buf) {
239                Ok(n) => n,
240                Err(Errno::EFAULT) if written > 0 => return Ok(written),
241                Err(error) => return Err(crate::random::copy_error(error)),
242            };
243            if n == 0 && written == 0 {
244                return Err(Errno::EFAULT.into());
245            }
246            if cfg!(debug_assertions) {
247                Hash::hash_slice(&local_buf[..n], hasher);
248            }
249            written += n;
250            segment_written += n;
251            if n < chunk_len {
252                return Ok(written);
253            }
254        }
255    }
256    Ok(written)
257}
258
259impl<T: RecordOrReplay> Detcore<T> {
260    /// Validates seccomp capability probes without installing guest filters.
261    // TODO-HUMAN-REVIEW(PR-874): Review deterministic seccomp probe validation.
262    pub async fn handle_seccomp<G: Guest<Self>>(
263        &self,
264        _guest: &mut G,
265        call: syscalls::Seccomp,
266    ) -> Result<i64, Error> {
267        seccomp_result(call.op(), call.flags(), call.args().is_some()).map_err(Into::into)
268    }
269
270    fn write_arch_prctl_u64<G: Guest<Self>>(
271        &self,
272        guest: &mut G,
273        raw_addr: usize,
274        value: u64,
275    ) -> Result<i64, Error> {
276        let addr = AddrMut::<u64>::from_raw(raw_addr).ok_or(Errno::EFAULT)?;
277        guest.memory().write_value(addr, &value)?;
278        Ok(0)
279    }
280
281    // AUTONOMOUS-BOT-IMPLEMENTED
282    // TODO-HUMAN-REVIEW(#539): Confirm the virtual arch_prctl control policy.
283    /// Preserve thread-local bases while hiding host CPU feature controls.
284    pub async fn handle_arch_prctl<G: Guest<Self>>(
285        &self,
286        guest: &mut G,
287        call: syscalls::ArchPrctl,
288    ) -> Result<i64, Error> {
289        let cpuid_uses_backend_policy =
290            self.cfg.virtualize_cpuid && self.cfg.cpuid_virtualized_by_backend;
291        let cpuid_uses_faulting = self.cfg.virtualize_cpuid && guest.has_cpuid_interception();
292        match call.cmd() {
293            ArchPrctlCmd::ARCH_SET_FS(_)
294            | ArchPrctlCmd::ARCH_SET_GS(_)
295            | ArchPrctlCmd::ARCH_GET_FS(_)
296            | ArchPrctlCmd::ARCH_GET_GS(_) => Ok(guest.inject(call).await?),
297
298            // KVM installs a deterministic CPUID table while leaving the instruction enabled.
299            ArchPrctlCmd::ARCH_GET_CPUID(_) if cpuid_uses_backend_policy => Ok(1),
300            ArchPrctlCmd::ARCH_SET_CPUID(value) if cpuid_uses_backend_policy => {
301                if value == 0 {
302                    Err(Errno::EPERM.into())
303                } else {
304                    Ok(0)
305                }
306            }
307
308            // When Reverie successfully disables native CPUID, Detcore answers its fault from a
309            // fixed table. Preserve that backend state and reject attempts to re-enable CPUID.
310            ArchPrctlCmd::ARCH_GET_CPUID(_) if cpuid_uses_faulting => Ok(0),
311            ArchPrctlCmd::ARCH_SET_CPUID(value) if cpuid_uses_faulting => {
312                if value == 0 {
313                    Ok(0)
314                } else {
315                    Err(Errno::EPERM.into())
316                }
317            }
318            // Reverie cannot faithfully deliver a CPUID fault requested by the tracee. In
319            // explicit host-CPUID mode, expose a fixed enabled control state and reject disable.
320            ArchPrctlCmd::ARCH_GET_CPUID(_) if !self.cfg.virtualize_cpuid => Ok(1),
321            ArchPrctlCmd::ARCH_SET_CPUID(value) if !self.cfg.virtualize_cpuid => {
322                if value == 0 {
323                    Err(Errno::EPERM.into())
324                } else {
325                    Ok(0)
326                }
327            }
328
329            // Ptrace hosts without CPUID-faulting support retain the kernel's honest state.
330            ArchPrctlCmd::ARCH_GET_CPUID(_) | ArchPrctlCmd::ARCH_SET_CPUID(_) => {
331                Ok(guest.inject(call).await?)
332            }
333
334            // Expose a conservative virtual CPU with no optional extended-state permissions.
335            ArchPrctlCmd::Other(
336                ARCH_GET_XCOMP_SUPP | ARCH_GET_XCOMP_PERM | ARCH_GET_XCOMP_GUEST_PERM,
337                addr,
338            ) => self.write_arch_prctl_u64(guest, addr, 0),
339            ArchPrctlCmd::Other(ARCH_REQ_XCOMP_PERM | ARCH_REQ_XCOMP_GUEST_PERM, _) => {
340                Err(Errno::EINVAL.into())
341            }
342
343            // Keep shadow stacks disabled in the virtual policy. Disabling an already-disabled
344            // feature is idempotent; enable/lock/unlock requests cannot be honored.
345            ArchPrctlCmd::Other(ARCH_SHSTK_STATUS, addr) => {
346                self.write_arch_prctl_u64(guest, addr, 0)
347            }
348            ArchPrctlCmd::Other(ARCH_SHSTK_DISABLE, features)
349                if features != 0 && features & !ARCH_SHSTK_VALID_MASK == 0 =>
350            {
351                Ok(0)
352            }
353            ArchPrctlCmd::Other(ARCH_SHSTK_DISABLE, _)
354            | ArchPrctlCmd::Other(ARCH_SHSTK_ENABLE | ARCH_SHSTK_LOCK | ARCH_SHSTK_UNLOCK, _) => {
355                Err(Errno::EINVAL.into())
356            }
357
358            ArchPrctlCmd::Other(_, _) => Err(Errno::EINVAL.into()),
359        }
360    }
361
362    // AUTONOMOUS-BOT-IMPLEMENTED
363    // TODO-HUMAN-REVIEW(#663)
364    /// Preserve deterministic Ruby thread controls, report the container's fixed
365    /// capability bounding set, and reject options that expose unmodeled process
366    /// or host state.
367    pub async fn handle_prctl<G: Guest<Self>>(
368        &self,
369        guest: &mut G,
370        call: syscalls::Prctl,
371    ) -> Result<i64, Error> {
372        match call.option() {
373            // The capability bounding set is fixed by the container launch policy.
374            libc::PR_CAPBSET_READ => Ok(self.record_or_replay(guest, call).await?),
375            // AUTONOMOUS-BOT-IMPLEMENTED
376            // TODO-HUMAN-REVIEW(PR-2150): Timer slack is shared with the
377            // virtual `/proc/<tid>/timerslack_ns` channel.  Never pass either
378            // setter through to the physical tracee: a large physical slack
379            // changes wake timing on Detcore's remaining host-timed waits.
380            libc::PR_SET_TIMERSLACK => {
381                let requested = call.arg2();
382                let state = guest.thread_state_mut();
383                // Linux treats zero as reset-to-default. Hermit virtualizes the
384                // Linux scheduling policy as SCHED_OTHER, so the kernel's
385                // RT/DL no-op branch is unreachable in the guest model.
386                state.timer_slack_ns = if requested == 0 {
387                    state.default_timer_slack_ns
388                } else {
389                    requested
390                };
391                Ok(0)
392            }
393            libc::PR_GET_TIMERSLACK => Ok(guest.thread_state().timer_slack_ns as i64),
394            option
395                if guest.config().backend_virtualizes_capability_prctls
396                    && is_backend_virtualized_capability_prctl(option) =>
397            {
398                self.passthrough(guest, call.into()).await
399            }
400            option if is_supported_prctl_option(option) => {
401                self.passthrough(guest, call.into()).await
402            }
403            _ => Err(Errno::ENOSYS.into()),
404        }
405    }
406
407    // AUTONOMOUS-BOT-IMPLEMENTED
408    // TODO-HUMAN-REVIEW(#806)
409    /// Report the deterministic default nice value for any scheduling target.
410    ///
411    /// Under Hermit the Linux nice value is inert: the scheduler is virtualized
412    /// and guest threads are serialized onto one virtual CPU, so a process's,
413    /// group's, or user's scheduling priority never affects guest-visible
414    /// computation. Report the deterministic default nice (0) for every valid
415    /// target regardless of `who` — real tools such as `renice -p <pid>` always
416    /// pass an explicit pid, and the raw `getpriority(2)` never checks
417    /// permissions on a read, so it must never return `EPERM`. An unknown
418    /// `which` still faults with `EINVAL`, matching Linux.
419    pub async fn handle_getpriority<G: Guest<Self>>(
420        &self,
421        _guest: &mut G,
422        call: syscalls::Getpriority,
423    ) -> Result<i64, Error> {
424        Ok(getpriority_result(call.which())?)
425    }
426
427    // AUTONOMOUS-BOT-IMPLEMENTED
428    // TODO-HUMAN-REVIEW(PR-857): Deterministic syslog(2) virtualization.
429    /// Present an empty kernel ring buffer. Reads and size queries return zero,
430    /// controls are inert, and invalid actions preserve Linux's EINVAL boundary.
431    pub async fn handle_syslog<G: Guest<Self>>(
432        &self,
433        _guest: &mut G,
434        call: syscalls::Syslog,
435    ) -> Result<i64, Error> {
436        Ok(deterministic_syslog_result(call.priority(), call.len())?)
437    }
438
439    // AUTONOMOUS-BOT-IMPLEMENTED
440    // TODO-HUMAN-REVIEW(#806)
441    /// Accept any priority change as a deterministic no-op.
442    ///
443    /// Nice values are inert under Hermit's virtualized, serialized scheduler,
444    /// so accept the request without touching host scheduling. The guest runs as
445    /// a single uid-0 container principal, so a real `setpriority(2)` from the
446    /// caller would succeed anyway; never fabricate `EPERM` for tools such as
447    /// `nice -n 5 <cmd>`, `renice -p <pid>`, or Python's `os.nice`. An unknown
448    /// `which` still faults with `EINVAL`, matching Linux.
449    pub async fn handle_setpriority<G: Guest<Self>>(
450        &self,
451        _guest: &mut G,
452        call: syscalls::Setpriority,
453    ) -> Result<i64, Error> {
454        Ok(setpriority_result(call.which())?)
455    }
456
457    // AUTONOMOUS-BOT-IMPLEMENTED
458    // TODO-HUMAN-REVIEW(#663)
459    /// Reject cross-process memory advice without consulting host process state.
460    pub fn handle_process_madvise(pidfd: usize, flags: usize) -> Result<i64, Error> {
461        if flags != 0 {
462            return Err(Errno::EINVAL.into());
463        }
464
465        // Linux interprets pidfd as an int. Preserve its deterministic invalid-fd
466        // rejection, but never let a valid host pidfd alter another process's memory.
467        if (pidfd as libc::c_int) < 0 {
468            Err(Errno::EBADF.into())
469        } else {
470            Err(Errno::EPERM.into())
471        }
472    }
473
474    // AUTONOMOUS-BOT-IMPLEMENTED
475    // TODO-HUMAN-REVIEW(PR-1096): Review the backend-independent random-device stream.
476    /// Fill guest memory from the canonical stream used by every backend's
477    /// `/dev/random` and `/dev/urandom` virtualization.
478    pub(super) fn fill_random_device_bytes<G: Guest<Self>>(
479        &self,
480        guest: &mut G,
481        remote_buf: AddrMut<u8>,
482        len: usize,
483        stream_offset: u64,
484    ) -> Result<usize, Error> {
485        self.fill_random_device_iovecs(
486            guest,
487            &[crate::iovecs::ImportedIovec {
488                base: remote_buf.as_raw(),
489                len,
490            }],
491            stream_offset,
492        )
493    }
494
495    pub(super) fn fill_random_device_iovecs<G: Guest<Self>>(
496        &self,
497        guest: &mut G,
498        iovecs: &[crate::iovecs::ImportedIovec],
499        stream_offset: u64,
500    ) -> Result<usize, Error> {
501        let seed = guest.config().rng_seed();
502        let mut hasher = DefaultHasher::new();
503        let written = fill_canonical_random_iovecs(
504            &mut guest.memory(),
505            iovecs,
506            seed,
507            stream_offset,
508            &mut hasher,
509        )?;
510        if cfg!(debug_assertions) {
511            detlog!(
512                "[dtid {}] USER RAND [/dev/[u]random] Filled guest memory with {} canonical random bytes at offset {}, hash of bytes: {}",
513                guest.thread_state().dettid,
514                written,
515                stream_offset,
516                hasher.finish()
517            );
518        }
519        Ok(written)
520    }
521
522    /// uname syscall
523    pub async fn handle_uname<G: Guest<Self>>(
524        &self,
525        guest: &mut G,
526        call: syscalls::Uname,
527    ) -> Result<i64, Error> {
528        let ret = self.record_or_replay(guest, call).await?;
529        if let Some(buf) = call.buf() {
530            let mut un = guest.memory().read_value(buf)?;
531            // Keep this in configured UTC: `Local` initializes libc TLS, which is unavailable
532            // while a DynamoRIO application thread is executing a client callback.
533            let epoch = guest.config().epoch;
534
535            if !guest.config().has_uts_namespace {
536                // FIXME: It should be possible to remove this once all tests
537                // are also using namespaces.
538                un.nodename = from_str(DEFAULT_HOSTNAME);
539                un.domainname = from_str(DEFAULT_HOSTNAME.split('.').next_back().unwrap_or(""));
540            }
541
542            un.release = from_str("5.2.0");
543            un.version = from_str(&format!("#1 SMP {}", epoch.format("%a %b %d %T %Z %Y")));
544            guest.memory().write_value(buf, &un)?;
545        }
546
547        Ok(ret)
548    }
549
550    /// Fill `getrandom(2)` requests from the current thread's seeded deterministic PRNG.
551    /// Supported blocking/source-selection flags share that always-ready stream; invalid Linux
552    /// flag combinations are rejected before guest memory is touched.
553    pub async fn handle_getrandom<G: Guest<Self>>(
554        &self,
555        guest: &mut G,
556        call: syscalls::Getrandom,
557    ) -> Result<i64, Error> {
558        let memory = guest.memory();
559        let dettid = guest.thread_state().dettid;
560        crate::random::getrandom(guest.thread_state_mut().thread_prng(), memory, dettid, call)
561    }
562
563    /// setsid system call
564    pub async fn handle_setsid<G: Guest<Self>>(
565        &self,
566        guest: &mut G,
567        call: syscalls::Setsid,
568    ) -> Result<i64, Error> {
569        let res = guest.inject(call).await?;
570        let process = guest.thread_state().detpid.expect("detpid unset");
571        let _ = create_session(guest, process).await;
572
573        // task is trying to become a daemon process. for more details
574        // see: https://notes.shichao.io/apue/ch13/
575        if guest.config().kill_daemons {
576            guest.daemonize().await;
577        }
578        Ok(res)
579    }
580
581    /// setpgid system call. The kernel remains authoritative for validation;
582    /// after success Detcore mirrors the guest-visible process-group change so
583    /// group-selecting waits do not consult host `/proc` state.
584    pub async fn handle_setpgid<G: Guest<Self>>(
585        &self,
586        guest: &mut G,
587        call: syscalls::Setpgid,
588    ) -> Result<i64, Error> {
589        let res = guest.inject(call).await?;
590        let caller = guest.thread_state().detpid.expect("detpid unset");
591        let process = if call.pid() == 0 {
592            caller
593        } else {
594            DetPid::from_raw(call.pid())
595        };
596        let group = if call.pgid() == 0 {
597            process
598        } else {
599            DetPid::from_raw(call.pgid())
600        };
601        let _ = set_process_group(guest, process, group).await;
602        Ok(res)
603    }
604
605    /// membarrier (system call).
606    ///
607    /// `membarrier(2)` issues process-wide memory barriers so that userspace can
608    /// use asymmetric fences (e.g. CPython's QSBR, RCU-style reclamation).
609    /// Detcore serializes all guest threads onto a single logical CPU with a
610    /// total memory order, so any requested barrier is *already* satisfied and
611    /// every command is a deterministic no-op. For `MEMBARRIER_CMD_QUERY` we
612    /// report the set of commands we emulate so the guest stays on this
613    /// controlled path instead of a host-dependent fallback; every other command
614    /// returns success without doing anything.
615    pub async fn handle_membarrier<G: Guest<Self>>(
616        &self,
617        guest: &mut G,
618        call: syscalls::Membarrier,
619    ) -> Result<i64, Error> {
620        // Values from <linux/membarrier.h>.
621        const MEMBARRIER_CMD_QUERY: i32 = 0;
622        const MEMBARRIER_CMD_GLOBAL: i32 = 1 << 0;
623        const MEMBARRIER_CMD_GLOBAL_EXPEDITED: i32 = 1 << 1;
624        const MEMBARRIER_CMD_REGISTER_GLOBAL_EXPEDITED: i32 = 1 << 2;
625        const MEMBARRIER_CMD_PRIVATE_EXPEDITED: i32 = 1 << 3;
626        const MEMBARRIER_CMD_REGISTER_PRIVATE_EXPEDITED: i32 = 1 << 4;
627        const SUPPORTED: i32 = MEMBARRIER_CMD_GLOBAL
628            | MEMBARRIER_CMD_GLOBAL_EXPEDITED
629            | MEMBARRIER_CMD_REGISTER_GLOBAL_EXPEDITED
630            | MEMBARRIER_CMD_PRIVATE_EXPEDITED
631            | MEMBARRIER_CMD_REGISTER_PRIVATE_EXPEDITED;
632
633        let cmd = call.cmd();
634        if cmd == MEMBARRIER_CMD_QUERY {
635            detlog!(
636                "[dtid {}] membarrier(QUERY) => reporting emulated commands {:#x}",
637                guest.thread_state().dettid,
638                SUPPORTED,
639            );
640            Ok(SUPPORTED as i64)
641        } else {
642            detlog!(
643                "[dtid {}] membarrier(cmd={}) no-op (threads are serialized on one CPU)",
644                guest.thread_state().dettid,
645                cmd,
646            );
647            Ok(0)
648        }
649    }
650
651    /// getcpu system call
652    pub async fn handle_getcpu<G: Guest<Self>>(
653        &self,
654        guest: &mut G,
655        call: syscalls::Getcpu,
656    ) -> Result<i64, Error> {
657        // Always set the CPU to 0.
658        if let Some(cpu) = call.cpu() {
659            guest.memory().write_value(cpu, &0)?;
660        }
661
662        // Always set the NUMA node to 0.
663        if let Some(node) = call.node() {
664            guest.memory().write_value(node, &0)?;
665        }
666
667        Ok(0)
668    }
669
670    /// getresuid under Hermit. Detcore presents a fixed virtual-root identity, so
671    /// the real, effective, and saved user IDs are all the constant 0. Under the
672    /// ptrace backend the guest runs inside a CLONE_NEWUSER namespace that already
673    /// maps the host uid to 0; emulating the same constant here makes in-process
674    /// backends (DBT) agree with that golden reference instead of leaking the host
675    /// uid, and the fully emulated result is bitwise-identical across --verify and
676    /// record/replay.
677    // AUTONOMOUS-BOT-IMPLEMENTED
678    // TODO-HUMAN-REVIEW(#1549)
679    pub async fn handle_getresuid<G: Guest<Self>>(
680        &self,
681        guest: &mut G,
682        call: syscalls::Getresuid,
683    ) -> Result<i64, Error> {
684        if let Some(ruid) = call.ruid() {
685            guest.memory().write_value(ruid, &0)?;
686        }
687        if let Some(euid) = call.euid() {
688            guest.memory().write_value(euid, &0)?;
689        }
690        if let Some(suid) = call.suid() {
691            guest.memory().write_value(suid, &0)?;
692        }
693        Ok(0)
694    }
695
696    /// getresgid under Hermit. The group-ID counterpart of `handle_getresuid`:
697    /// the real, effective, and saved group IDs are all the fixed virtual-root
698    /// constant 0, matching the ptrace CLONE_NEWUSER identity and deterministic
699    /// across --verify and record/replay.
700    // AUTONOMOUS-BOT-IMPLEMENTED
701    // TODO-HUMAN-REVIEW(#1549)
702    pub async fn handle_getresgid<G: Guest<Self>>(
703        &self,
704        guest: &mut G,
705        call: syscalls::Getresgid,
706    ) -> Result<i64, Error> {
707        if let Some(rgid) = call.rgid() {
708            guest.memory().write_value(rgid, &0)?;
709        }
710        if let Some(egid) = call.egid() {
711            guest.memory().write_value(egid, &0)?;
712        }
713        if let Some(sgid) = call.sgid() {
714            guest.memory().write_value(sgid, &0)?;
715        }
716        Ok(0)
717    }
718
719    /// get_mempolicy under Hermit. The container exposes a single virtual NUMA
720    /// node, so the effective policy is always the default and every address
721    /// resolves to node 0. The result is fully emulated (never injected), so it
722    /// is bitwise-identical across the two --verify runs and under record/replay,
723    /// removing the host-NUMA-topology dependence a passthrough would introduce.
724    // AUTONOMOUS-BOT-IMPLEMENTED
725    // TODO-HUMAN-REVIEW(#720)
726    pub async fn handle_get_mempolicy<G: Guest<Self>>(
727        &self,
728        guest: &mut G,
729        call: syscalls::GetMempolicy,
730    ) -> Result<i64, Error> {
731        // Report MPOL_DEFAULT (0) for the current policy / node when requested.
732        // The nodemask output is left untouched: reverie exposes it as an
733        // immutable pointer, and MPOL_DEFAULT carries no node set.
734        if let Some(policy) = call.policy() {
735            guest.memory().write_value(policy, &0)?;
736        }
737        Ok(0)
738    }
739
740    /// move_pages under Hermit. On a single virtual NUMA node nothing can be
741    /// relocated, so report every page as residing on node 0 and succeed. The
742    /// answer is a fixed constant, so it is deterministic across --verify and
743    /// record/replay.
744    // AUTONOMOUS-BOT-IMPLEMENTED
745    // TODO-HUMAN-REVIEW(#720)
746    pub async fn handle_move_pages<G: Guest<Self>>(
747        &self,
748        guest: &mut G,
749        call: syscalls::MovePages,
750    ) -> Result<i64, Error> {
751        // When a status buffer is supplied (either a real move request or a
752        // location query with nodes == NULL), report node 0 for every page.
753        if let Some(status) = call.status() {
754            let count = call.nr_pages() as usize;
755            let zeros = vec![0i32; count];
756            guest.memory().write_values(status, &zeros)?;
757        }
758        Ok(0)
759    }
760}
761
762#[cfg(test)]
763mod tests {
764    use super::*;
765
766    struct ScatterMemory {
767        bytes: Vec<u8>,
768        outcomes: std::collections::VecDeque<Result<usize, Errno>>,
769        writes: usize,
770    }
771
772    impl ScatterMemory {
773        fn new(size: usize, outcomes: impl IntoIterator<Item = Result<usize, Errno>>) -> Self {
774            Self {
775                bytes: vec![0xa5; size],
776                outcomes: outcomes.into_iter().collect(),
777                writes: 0,
778            }
779        }
780    }
781
782    impl MemoryAccess for ScatterMemory {
783        fn read_vectored(
784            &self,
785            _: &[std::io::IoSlice],
786            _: &mut [std::io::IoSliceMut],
787        ) -> Result<usize, Errno> {
788            panic!("canonical scatter must not read guest bytes")
789        }
790        fn write_vectored(
791            &mut self,
792            _: &[std::io::IoSlice],
793            _: &mut [std::io::IoSliceMut],
794        ) -> Result<usize, Errno> {
795            panic!("canonical scatter must use the user-access capability")
796        }
797        fn write_with_user_access(
798            &mut self,
799            address: AddrMut<u8>,
800            bytes: &[u8],
801        ) -> Result<usize, Errno> {
802            self.writes += 1;
803            let count = self.outcomes.pop_front().unwrap_or(Ok(bytes.len()))?;
804            assert!(count <= bytes.len());
805            let offset = address.as_raw() - 0x1000;
806            self.bytes[offset..offset + count].copy_from_slice(&bytes[..count]);
807            Ok(count)
808        }
809    }
810
811    fn assert_copy_failure(error: Error, expected: Errno) {
812        let Error::Tool(error) = error else {
813            panic!("copy failure became a guest errno: {error:?}")
814        };
815        assert_eq!(
816            error
817                .downcast_ref::<crate::random::RandomCopyFailure>()
818                .expect("typed copy failure")
819                .errno(),
820            expected
821        );
822    }
823
824    fn scatter(memory: &mut ScatterMemory, lengths: &[usize], offset: u64) -> Result<usize, Error> {
825        let mut base = 0x1000;
826        let vectors: Vec<_> = lengths
827            .iter()
828            .map(|&len| {
829                let vector = crate::iovecs::ImportedIovec { base, len };
830                base += len;
831                vector
832            })
833            .collect();
834        fill_canonical_random_iovecs(memory, &vectors, 0, offset, &mut DefaultHasher::new())
835    }
836
837    #[test]
838    fn random_scatter_distinguishes_fault_prefixes_from_backend_errors() {
839        // Exercise both an earlier complete iovec and an earlier complete chunk.
840        for lengths in [vec![3, 5], vec![RANDOM_FILL_CHUNK_BYTES + 5]] {
841            let prefix = if lengths.len() == 2 {
842                3
843            } else {
844                RANDOM_FILL_CHUNK_BYTES
845            };
846            for error in [Errno::EFAULT, Errno::EIO, Errno::ENOMEM] {
847                let mut memory = ScatterMemory::new(prefix + 5, [Ok(prefix), Err(error)]);
848                let result = scatter(&mut memory, &lengths, 7);
849                match error {
850                    Errno::EFAULT => assert_eq!(result.unwrap(), prefix),
851                    _ => assert_copy_failure(result.unwrap_err(), error),
852                }
853                let expected: Vec<_> = (7..7 + prefix as u64)
854                    .map(|index| canonical_random_device_byte(0, index))
855                    .collect();
856                assert_eq!(&memory.bytes[..prefix], expected);
857                assert_eq!(&memory.bytes[prefix..], &[0xa5; 5]);
858                assert_eq!(memory.writes, 2);
859            }
860        }
861    }
862
863    #[test]
864    fn random_scatter_short_or_zero_copy_stops_without_touching_later_segments() {
865        for first in [0, 2] {
866            let mut memory = ScatterMemory::new(10, [Ok(first)]);
867            let result = scatter(&mut memory, &[5, 5], 0);
868            if first == 0 {
869                assert!(matches!(result, Err(Error::Errno(Errno::EFAULT))));
870            } else {
871                assert_eq!(result.unwrap(), first);
872                assert_eq!(&memory.bytes[..2], &[41, 114]);
873            }
874            assert!(memory.bytes[first..].iter().all(|&byte| byte == 0xa5));
875            assert_eq!(memory.writes, 1);
876        }
877        let mut memory = ScatterMemory::new(10, [Ok(5), Ok(0)]);
878        assert_eq!(scatter(&mut memory, &[5, 5], 0).unwrap(), 5);
879        assert!(memory.bytes[5..].iter().all(|&byte| byte == 0xa5));
880        assert_eq!(memory.writes, 2);
881    }
882
883    #[test]
884    fn random_scatter_saturated_bytes_match_partitioned_calls() {
885        for (offset, expected) in [(u64::MAX - 2, [78, 151, 224, 224]), (u64::MAX, [224; 4])] {
886            let mut single = ScatterMemory::new(4, []);
887            assert_eq!(scatter(&mut single, &[4], offset).unwrap(), 4);
888            let mut partitioned = ScatterMemory::new(4, []);
889            let mut hash = DefaultHasher::new();
890            assert_eq!(
891                fill_canonical_random_iovecs(
892                    &mut partitioned,
893                    &[crate::iovecs::ImportedIovec {
894                        base: 0x1000,
895                        len: 1
896                    }],
897                    0,
898                    offset,
899                    &mut hash
900                )
901                .unwrap(),
902                1
903            );
904            assert_eq!(
905                fill_canonical_random_iovecs(
906                    &mut partitioned,
907                    &[crate::iovecs::ImportedIovec {
908                        base: 0x1001,
909                        len: 3
910                    }],
911                    0,
912                    offset.saturating_add(1),
913                    &mut hash
914                )
915                .unwrap(),
916                3
917            );
918            assert_eq!(single.bytes, expected);
919            assert_eq!(partitioned.bytes, expected);
920        }
921    }
922
923    #[test]
924    fn random_scatter_backend_error_rolls_back_shared_cursor_after_eight_byte_prefix() {
925        let fd = crate::fd::DetFd::new(
926            3,
927            nix::fcntl::OFlag::O_RDONLY,
928            crate::fd::FdType::Rng,
929            crate::types::OpenFileId::new(crate::types::DetTid::from_raw(1), 0),
930        );
931        for error in [Errno::EFAULT, Errno::EIO] {
932            let mut memory = ScatterMemory::new(8, [Ok(4), Err(error)]);
933            let result = fd.with_random_device_stream(|offset| scatter(&mut memory, &[8], offset));
934            if error == Errno::EFAULT {
935                assert_eq!(result.unwrap(), 4);
936            } else {
937                assert_copy_failure(result.unwrap_err(), Errno::EIO);
938            }
939            // First iteration commits four; the EIO iteration must not commit
940            // its physically copied prefix or fabricate a successful result.
941            assert_eq!(fd.random_device_offset(), 4);
942            assert_eq!(&memory.bytes[4..], &[0xa5; 4]);
943            assert_eq!(memory.writes, 2);
944        }
945    }
946
947    #[test]
948    fn prctl_support_covers_deterministic_controls() {
949        for option in [
950            libc::PR_SET_NAME,
951            libc::PR_GET_NAME,
952            libc::PR_SET_THP_DISABLE,
953            libc::PR_GET_THP_DISABLE,
954            // Deterministic per-process dumpability state.
955            libc::PR_SET_DUMPABLE,
956            libc::PR_GET_DUMPABLE,
957            // Deterministic per-thread capability-retention flag used by setpriv.
958            libc::PR_SET_KEEPCAPS,
959            libc::PR_GET_KEEPCAPS,
960            // Deterministic per-thread parent-death-signal flag used by setpriv.
961            libc::PR_SET_PDEATHSIG,
962            libc::PR_GET_PDEATHSIG,
963        ] {
964            assert!(is_supported_prctl_option(option));
965        }
966
967        assert!(!is_supported_prctl_option(libc::PR_SET_NO_NEW_PRIVS));
968        assert!(!is_supported_prctl_option(libc::PR_SET_TIMERSLACK));
969        assert!(!is_supported_prctl_option(libc::PR_GET_TIMERSLACK));
970    }
971
972    #[test]
973    fn backend_virtualized_prctl_support_is_capability_scoped() {
974        for option in [libc::PR_CAPBSET_DROP, libc::PR_CAP_AMBIENT] {
975            assert!(is_backend_virtualized_capability_prctl(option));
976        }
977        for option in [libc::PR_SET_KEEPCAPS, libc::PR_SET_SECUREBITS] {
978            assert!(!is_backend_virtualized_capability_prctl(option));
979        }
980    }
981
982    #[test]
983    fn getrandom_accepts_linux_flags() {
984        for flags in [
985            0,
986            libc::GRND_NONBLOCK as usize,
987            libc::GRND_RANDOM as usize,
988            (libc::GRND_NONBLOCK | libc::GRND_RANDOM) as usize,
989            libc::GRND_INSECURE as usize,
990            (libc::GRND_NONBLOCK | libc::GRND_INSECURE) as usize,
991            1_usize << 32,
992        ] {
993            assert!(
994                validate_getrandom_flags(flags).is_ok(),
995                "valid flags rejected: {flags:#x}"
996            );
997        }
998    }
999
1000    #[test]
1001    fn getrandom_rejects_invalid_flags() {
1002        for flags in [
1003            0x8000_0000,
1004            (1_usize << 32) | 0x8000_0000,
1005            (libc::GRND_RANDOM | libc::GRND_INSECURE) as usize,
1006        ] {
1007            assert_eq!(validate_getrandom_flags(flags), Err(Errno::EINVAL));
1008        }
1009    }
1010
1011    #[test]
1012    fn getrandom_caps_requests_at_linux_max_rw_count() {
1013        assert_eq!(getrandom_request_len(16), 16);
1014        assert_eq!(getrandom_request_len(usize::MAX), GETRANDOM_MAX_BYTES);
1015    }
1016
1017    #[test]
1018    fn canonical_random_device_stream_matches_kvm_root_contract() {
1019        let first: Vec<_> = (0..8)
1020            .map(|index| canonical_random_device_byte(0, index))
1021            .collect();
1022        assert_eq!(first, [41, 114, 187, 4, 77, 150, 223, 40]);
1023
1024        let continued: Vec<_> = (8..16)
1025            .map(|index| canonical_random_device_byte(0, index))
1026            .collect();
1027        assert_eq!(continued, [113, 186, 3, 76, 149, 222, 39, 112]);
1028
1029        let seeded: Vec<_> = (0..16)
1030            .map(|index| canonical_random_device_byte(17, index))
1031            .collect();
1032        assert_eq!(
1033            seeded,
1034            [
1035                56, 114, 187, 4, 77, 150, 223, 40, 96, 186, 3, 76, 149, 222, 39, 112
1036            ]
1037        );
1038        assert_ne!(seeded, [first, continued].concat());
1039    }
1040
1041    #[test]
1042    fn getpriority_reports_default_nice_for_every_target() {
1043        // Every valid PRIO_* selector reports the default nice (raw 20 = nice 0),
1044        // regardless of `who`. Real tools such as `renice -p <pid>` pass an
1045        // explicit pid, and getpriority never checks permissions, so this must
1046        // never be EPERM (the pre-fix stub only accepted PRIO_PROCESS/who==0).
1047        for which in [libc::PRIO_PROCESS, libc::PRIO_PGRP, libc::PRIO_USER] {
1048            assert_eq!(getpriority_result(which as i32), Ok(20));
1049        }
1050    }
1051
1052    #[test]
1053    fn setpriority_accepts_any_change_for_valid_target() {
1054        // Nice is inert under Hermit, so any priority change for a valid target
1055        // succeeds as a no-op — including nonzero nice (`nice -n 5`, os.nice).
1056        for which in [libc::PRIO_PROCESS, libc::PRIO_PGRP, libc::PRIO_USER] {
1057            assert_eq!(setpriority_result(which as i32), Ok(0));
1058        }
1059    }
1060
1061    #[test]
1062    fn get_and_set_priority_reject_unknown_which_with_einval() {
1063        // Match Linux: an unknown target selector faults with EINVAL, not EPERM.
1064        for which in [-1, 3, 42] {
1065            assert_eq!(getpriority_result(which), Err(Errno::EINVAL));
1066            assert_eq!(setpriority_result(which), Err(Errno::EINVAL));
1067        }
1068    }
1069
1070    #[test]
1071    fn seccomp_tsync_null_probe_matches_linux_validation() {
1072        assert_eq!(
1073            seccomp_result(SECCOMP_SET_MODE_FILTER, SECCOMP_FILTER_FLAG_TSYNC, false,),
1074            Err(Errno::EFAULT)
1075        );
1076        assert_eq!(
1077            seccomp_result(SECCOMP_SET_MODE_FILTER, 1 << 31, false),
1078            Err(Errno::EINVAL)
1079        );
1080        assert_eq!(
1081            seccomp_result(SECCOMP_SET_MODE_FILTER, 0, true),
1082            Err(Errno::EOPNOTSUPP)
1083        );
1084    }
1085
1086    #[test]
1087    fn process_madvise_is_rejected_deterministically() {
1088        assert!(matches!(
1089            Detcore::<crate::record_or_replay::NoopTool>::handle_process_madvise(
1090                (-10_000_i32) as usize,
1091                0
1092            ),
1093            Err(Error::Errno(Errno::EBADF))
1094        ));
1095        assert!(matches!(
1096            Detcore::<crate::record_or_replay::NoopTool>::handle_process_madvise(3, 1),
1097            Err(Error::Errno(Errno::EINVAL))
1098        ));
1099        assert!(matches!(
1100            Detcore::<crate::record_or_replay::NoopTool>::handle_process_madvise(3, 0),
1101            Err(Error::Errno(Errno::EPERM))
1102        ));
1103    }
1104
1105    #[test]
1106    fn syslog_exposes_an_empty_log_and_validates_actions() {
1107        for action in 0..=7 {
1108            assert_eq!(deterministic_syslog_result(action, 0), Ok(0));
1109        }
1110        for action in [9, 10] {
1111            assert_eq!(deterministic_syslog_result(action, 0), Ok(0));
1112        }
1113        assert_eq!(deterministic_syslog_result(8, 1), Ok(0));
1114        assert_eq!(deterministic_syslog_result(8, 8), Ok(0));
1115        assert_eq!(deterministic_syslog_result(8, 0), Err(Errno::EINVAL));
1116        assert_eq!(deterministic_syslog_result(8, 9), Err(Errno::EINVAL));
1117        assert_eq!(deterministic_syslog_result(11, 0), Err(Errno::EINVAL));
1118    }
1119}