Skip to main content

detcore/syscalls/
helpers.rs

1/*
2 * Copyright (c) Meta Platforms, Inc. and affiliates.
3 * All rights reserved.
4 *
5 * This source code is licensed under the BSD-style license found in the
6 * LICENSE file in the root directory of this source tree.
7 */
8
9use std::num::NonZeroUsize;
10use std::time::Duration;
11
12use async_trait::async_trait;
13use reverie::Errno;
14use reverie::Error;
15use reverie::Guest;
16use reverie::Stack;
17use reverie::syscalls;
18use reverie::syscalls::Addr;
19use reverie::syscalls::Displayable;
20use reverie::syscalls::MapFlags;
21use reverie::syscalls::MemoryAccess;
22use reverie::syscalls::ProtFlags;
23use reverie::syscalls::Syscall;
24use reverie::syscalls::SyscallInfo;
25use reverie::syscalls::Sysno;
26use reverie::syscalls::Timespec;
27use reverie::syscalls::WaitPidFlag;
28
29use crate::fd::FdType;
30use crate::record_or_replay::RecordOrReplay;
31use crate::resources::ExternalOpId;
32use crate::resources::Permission;
33use crate::resources::ResourceID;
34use crate::resources::Resources;
35use crate::syscalls::threads::KernelSigaction;
36use crate::syscalls::threads::KernelSigset;
37use crate::syscalls::threads::WaitSignalDisposition;
38use crate::syscalls::threads::block_signals_for_disposition;
39use crate::syscalls::threads::blocked_signal_mask;
40use crate::syscalls::threads::restore_signals_after_disposition;
41use crate::syscalls::threads::wait_signal_disposition;
42use crate::tool_global::ResumeStatus;
43use crate::tool_global::resource_request;
44use crate::tool_global::thread_observe_time;
45use crate::tool_global::trace_schedevent;
46use crate::tool_local::Detcore;
47use crate::tool_local::finish_partial_record_or_replay_write;
48use crate::types::DetTid;
49use crate::types::LogicalTime;
50use crate::types::OpenFileId;
51use crate::types::SchedEvent;
52use crate::types::SyscallPhase;
53
54impl<T: RecordOrReplay> Detcore<T> {
55    // AUTONOMOUS-BOT-IMPLEMENTED
56    // TODO-HUMAN-REVIEW(#2373)
57    /// Apply the established unsupported-syscall refusal policy to a *supported*
58    /// syscall whose particular operation Detcore cannot serve deterministically.
59    ///
60    /// This exists so such a refusal cannot invent its own policy. Three separate
61    /// config knobs govern how a fail-closed run dies -- `shutdown_on_unsupported_syscall`
62    /// (hard `exit(1)` through `unrecoverable_shutdown`), `exit_on_unsupported_syscall`
63    /// (a typed `UnsupportedSyscallError` the backend terminates on without
64    /// unwinding), and the `panic!` fallback -- and `Detcore::handle_unsupported_syscall`
65    /// consults all three in that order. The normal `hermit run` CLI happens to set
66    /// `shutdown_on_unsupported_syscall = panic_on_unsupported_syscalls`, so reading
67    /// only the latter looks equivalent, but that coupling is a CLI default, not an
68    /// invariant: an embedder that sets just `exit_on_unsupported_syscall` would get a
69    /// process-wide `exit(1)` from a bespoke call site where the standard path returns
70    /// a catchable error.
71    ///
72    /// When the run is *not* fail-closed, the caller's `fallback` errno is returned,
73    /// because the operation itself is legal and the guest is entitled to a normal
74    /// failure code (`handle_unsupported_syscall` passes the call through instead,
75    /// which is not available here -- passing through is precisely the thing the
76    /// caller has determined it cannot do).
77    pub(crate) async fn refuse_unserviceable_operation<G: Guest<Self>>(
78        &self,
79        guest: &mut G,
80        sysno: reverie::syscalls::Sysno,
81        fallback: Errno,
82    ) -> Result<i64, Error> {
83        if !self.cfg.panic_on_unsupported_syscalls {
84            return Err(fallback.into());
85        }
86        if guest.config().shutdown_on_unsupported_syscall {
87            // Fail-closed policy: the operation is unserviceable and the config
88            // forbids passing it through.
89            crate::tool_global::unrecoverable_shutdown(
90                guest,
91                detcore_model::HERMIT_POLICY_REFUSAL_EXIT,
92            )
93            .await;
94        }
95        if guest.config().exit_on_unsupported_syscall {
96            return Err(Error::Tool(anyhow::Error::new(
97                crate::UnsupportedSyscallError(sysno),
98            )));
99        }
100        panic!("unserviceable operation on syscall: {sysno:?}");
101    }
102
103    /// Record or replay a BLOCKING syscall without stalling the current thread (and thus
104    /// deadlocking).  This uses a protocol of an extra resource request before/after the
105    /// syscall to inform the scheduler that the thread is leaving/rejoining the runnable
106    /// threads pool.
107    ///
108    /// This is only valid to use (1) in hermit record/replay modes, or (2)
109    /// when we're in "hermit run", but we're NOT sequentializing threads, because in
110    /// that case it's ok to use the blocking versions of system calls.
111    pub async fn record_or_replay_blocking<G: Guest<Self>>(
112        &self,
113        guest: &mut G,
114        call: Syscall,
115    ) -> Result<i64, Error> {
116        let dettid = guest.thread_state().dettid;
117        let op_id = ExternalOpId::new(dettid, guest.thread_state().stats.syscall_count);
118        self.record_or_replay_blocking_resource(guest, call, ResourceID::BlockingExternalIO(op_id))
119            .await
120    }
121
122    /// Execute the real `rt_sigsuspend` outside the runnable set while preserving
123    /// its signal-only completion condition for the scheduler.
124    pub async fn record_or_replay_rt_sigsuspend<G: Guest<Self>>(
125        &self,
126        guest: &mut G,
127        call: syscalls::RtSigsuspend,
128    ) -> Result<i64, Error> {
129        let dettid = guest.thread_state().dettid;
130        let op_id = ExternalOpId::new(dettid, guest.thread_state().stats.syscall_count);
131        self.record_or_replay_blocking_resource(
132            guest,
133            call.into(),
134            ResourceID::BlockingRtSigsuspend(op_id),
135        )
136        .await
137    }
138
139    async fn record_or_replay_blocking_resource<G: Guest<Self>>(
140        &self,
141        guest: &mut G,
142        call: Syscall,
143        blocking_resource: ResourceID,
144    ) -> Result<i64, Error> {
145        let dettid = guest.thread_state().dettid;
146        let op_id = match &blocking_resource {
147            ResourceID::BlockingExternalIO(op_id) | ResourceID::BlockingRtSigsuspend(op_id) => {
148                *op_id
149            }
150            _ => unreachable!("blocking syscall helper requires a blocking resource"),
151        };
152        // Internal-vs-external fd classification happens at the call sites that hold the
153        // typed, nonblockize-able syscall (see execute_nonblockable_fd_syscall):
154        // container-internal pipes are routed to the InternalIOPolling nonblockize-retry
155        // path and must NOT reach this external-blocking protocol. BlockingExternalIO
156        // deschedules the thread to run in the background and rejoin nondeterministically,
157        // which is unsafe for a pipe whose reader and writer are interdependent -- doing
158        // so is the root cause of the record/replay pipe deadlock. The remaining callers
159        // (external poll, wait4) are external by construction (their fd is not a single
160        // extractable internal pipe). Guard the invariant in debug builds while the
161        // deterministic scheduler is active. With thread sequentialization disabled,
162        // resource requests are no-ops and internal pipes intentionally use a blocking
163        // host syscall, as documented by this method.
164        debug_assert!(
165            !self.cfg.sequentialize_threads || !syscall_targets_internal_fd(guest, call),
166            "record_or_replay_blocking (BlockingExternalIO) reached for an internal pipe fd \
167             on syscall {}; internal fds must use the InternalIOPolling path",
168            call.name()
169        );
170        {
171            let mut rsrcs = Resources::new(dettid);
172            // With sequentialization enabled, only truly EXTERNAL endpoints reach here.
173            // Without it, resource_request is a no-op and internal fds may block directly.
174            rsrcs.insert(blocking_resource, Permission::RW);
175            rsrcs.fyi(call.name());
176            resource_request(guest, rsrcs).await;
177        }
178        tracing::trace!(
179            "Guest proceeding to execute potentially blocking call {}...",
180            call.name()
181        );
182        let res = self
183            .record_or_replay_preserving_tool_errors(guest, call)
184            .await;
185        // N.B. BlockingExternalIO is a "oneshot" resource, so no need to release
186        // explicitly here:
187        {
188            let mut rsrcs = Resources::new(dettid);
189            rsrcs.insert(ResourceID::BlockedExternalContinue(op_id), Permission::RW);
190            rsrcs.fyi(call.name());
191            resource_request(guest, rsrcs).await;
192        }
193        res
194    }
195
196    /// Executes a nonblockable syscall according to the following strategy:
197    /// - Record mode: Execute possibly blocking syscall
198    /// - Run mode: Transform the syscall to nonblocking if required before executing
199    ///
200    /// These are fd-oriented syscalls in the sense that whether they block or not depends
201    /// on whether NONBLOCK was set on the corresponding file descriptor.
202    pub async fn execute_nonblockable_fd_syscall<
203        G: Guest<Self>,
204        C: SyscallInfo + NonblockableSyscall + Into<Syscall>,
205    >(
206        &self,
207        guest: &mut G,
208        call: C,
209    ) -> Result<i64, Error> {
210        let wrapped: Syscall = call.into();
211
212        let action = match ioaction_based_on_fd_status(guest, call) {
213            Ok(action) => action,
214            Err(errno) => {
215                // Descriptor metadata is advisory for choosing an execution strategy. If the
216                // descriptor is invalid or cannot be classified, execute through the
217                // scheduler-safe blocking path and let the kernel provide the syscall errno.
218                // Returning the metadata error (or panicking on it) can change Linux error
219                // precedence, for example connect(-1, invalid_sockaddr, ...).
220                tracing::trace!(
221                    "NonblockableSyscall: fd classification failed with {}; executing kernel-authoritatively: {}",
222                    errno,
223                    call.name()
224                );
225                return self.record_or_replay_blocking(guest, wrapped).await;
226            }
227        };
228
229        // Is this operation on a container-INTERNAL fd (currently: pipes)? Internal
230        // pipes are made physically nonblocking even in record/replay (see
231        // handle_pipe2), so they can take the deterministic InternalIOPolling
232        // nonblockize-and-retry path. They must NOT be forced onto the
233        // BlockingExternalIO path in R/R: a pipe reader and its paired writer are not
234        // independent, so descheduling the reader as "external blocking IO" deadlocks
235        // the sequentialized scheduler (the documented R/R pipe hang). Truly external
236        // endpoints (host fds, network sockets) still use BlockingExternalIO. Sockets
237        // are left external for now: there is no internal-vs-external socket detection
238        // yet (see the handle_accept4 comment).
239        let internal_fd = syscall_targets_internal_fd(guest, wrapped);
240
241        if !self.cfg.sequentialize_threads
242            || (self.cfg.recordreplay_modes && !internal_fd)
243            || action == IOAction::Blocking
244        {
245            tracing::trace!(
246                "NonblockableSyscall: executing in blocking mode after all: {}",
247                call.name()
248            );
249            // We let these have nondeterminstic timing in record mode:
250            Ok(self.record_or_replay_blocking(guest, wrapped).await?)
251            // If in the future we want to record EXTERNAL network traffic only, we have a
252            // challenge to overcome.  We don't know if we need to record until after the
253            // accept completes, so we need an API for *post-facto* recording.
254        } else if action == IOAction::NonblockizeRetry {
255            tracing::trace!(
256                "NonblockableSyscall: converting to nonblocking syscall (internal polling): {}",
257                call.name()
258            );
259            let mut rsrc = Resources::new(guest.thread_state().dettid);
260            rsrc.insert(ResourceID::InternalIOPolling, Permission::W);
261            rsrc.fyi(call.name());
262            // In record/replay mode, route an internal-fd (pipe) read/write through the
263            // record/replay subtool so its data is captured on record and reproduced on
264            // replay (see retry_nonblocking_syscall). In plain `hermit run` there is no
265            // recorder, so execute directly (subtool = None).
266            let subtool = (self.cfg.recordreplay_modes && internal_fd).then_some(self);
267            Ok(retry_nonblocking_syscall(guest, call, rsrc, subtool).await?)
268        } else {
269            assert!(action == IOAction::PassThru);
270            tracing::trace!(
271                "NonblockableSyscall: just passing it through: {}",
272                call.name()
273            );
274            // Otherwise, the socket was already nonblocking, so we can safely execute it just once.
275            self.record_or_replay_preserving_tool_errors(guest, wrapped)
276                .await
277        }
278    }
279
280    // AUTONOMOUS-BOT-IMPLEMENTED
281    // TODO-HUMAN-REVIEW(#547)
282    /// Complete a logically blocking pipe writev after Hermit has made the pipe physically
283    /// nonblocking. A positive short write is an implementation artifact here: without
284    /// O_NONBLOCK, Linux blocks until the full vector is written unless a signal or error
285    /// interrupts it. Atomic vectors retain a private iovec snapshot for every retry; larger
286    /// vectors advance a positive short-write remainder through scalar writes.
287    pub async fn execute_blocking_pipe_writev<G: Guest<Self>>(
288        &self,
289        guest: &mut G,
290        call: syscalls::Writev,
291        expected_open_file: OpenFileId,
292    ) -> Result<i64, Error> {
293        const MAX_IOVECS: usize = 1024;
294        // Linux limits a single vectored transfer to INT_MAX rounded down to a page.
295        const MAX_RW_COUNT: usize = 0x7fff_f000;
296        // Linux guarantees pipe writes through this size are atomic.
297        const PIPE_BUF: usize = 4096;
298        // Every backend provides at least 512 bytes of tool scratch. Linux's
299        // own fast-iovec path is smaller; this covers common vectors.
300        const STACK_IOVECS: usize = 32;
301
302        let Some(iov_addr) = call.iov() else {
303            return self.execute_nonblockable_fd_syscall(guest, call).await;
304        };
305        if call.len() == 0 || call.len() > MAX_IOVECS {
306            return self.execute_nonblockable_fd_syscall(guest, call).await;
307        }
308
309        let iovecs: Vec<(usize, usize)> = {
310            let mut raw_iovecs = vec![
311                libc::iovec {
312                    iov_base: std::ptr::null_mut(),
313                    iov_len: 0,
314                };
315                call.len()
316            ];
317            guest.memory().read_values(iov_addr, &mut raw_iovecs)?;
318            raw_iovecs
319                .into_iter()
320                .map(|iovec| (iovec.iov_base as usize, iovec.iov_len))
321                .collect()
322        };
323        let requested = iovecs.iter().try_fold(0usize, |total, (_, length)| {
324            total.checked_add(*length).ok_or(Errno::EINVAL)
325        })?;
326        if requested > isize::MAX as usize {
327            return Err(Errno::EINVAL.into());
328        }
329        let target = requested.min(MAX_RW_COUNT);
330        if target == 0 {
331            return self.execute_nonblockable_fd_syscall(guest, call).await;
332        }
333
334        let atomic_pipe_write = target <= PIPE_BUF;
335
336        tracing::trace!(
337            "NonblockableSyscall: converting to nonblocking syscall (internal polling): writev"
338        );
339        let mut resources = pipe_writev_resources(guest.thread_state().dettid, call);
340        let subtool = self.cfg.recordreplay_modes.then_some(self);
341        let mut current = Syscall::Writev(call);
342        let mut written_total = 0usize;
343
344        // Keep ordinary signals pending while the scheduler and the target-side
345        // disposition check decide whether a wakeup interrupts this write. The
346        // ptrace-owned backends need the explicit mask while inspecting the
347        // target's current disposition.
348        let blocked_mask = blocked_signal_mask();
349        let mut stack = guest.stack().await;
350        let atomic_scratch_iov = if atomic_pipe_write && iovecs.len() <= STACK_IOVECS {
351            let mut raw_iovecs = [libc::iovec {
352                iov_base: std::ptr::null_mut(),
353                iov_len: 0,
354            }; STACK_IOVECS];
355            for (raw, (base, length)) in raw_iovecs.iter_mut().zip(&iovecs) {
356                raw.iov_base = *base as *mut libc::c_void;
357                raw.iov_len = *length;
358            }
359            let scratch_iov: Addr<libc::iovec> = stack.push(raw_iovecs).cast();
360            Some(scratch_iov.as_raw())
361        } else {
362            None
363        };
364        let blocked_mask_addr = stack.push(blocked_mask);
365        let old_mask_addr = stack.reserve::<KernelSigset>();
366        let action_addr = stack.reserve::<KernelSigaction>();
367        let _mask_guard = stack.commit()?;
368        let guest_signal_mask =
369            block_signals_for_disposition(guest, blocked_mask_addr, old_mask_addr).await?;
370
371        let result: Result<i64, Error> = loop {
372            // A cross-task signal can replace the initial scheduler request,
373            // before the first physical attempt, as well as a later retry.
374            let status = resource_request(guest, resources.clone()).await;
375            let disposition = match wait_signal_disposition(
376                guest,
377                status.clone(),
378                &guest_signal_mask,
379                action_addr,
380                true,
381            )
382            .await
383            {
384                Ok(disposition) => disposition,
385                Err(error) => break Err(error),
386            };
387            if matches!(status, ResumeStatus::Signaled(_))
388                && let Some(result) = interrupted_write_result(&call, written_total, disposition)
389            {
390                break result;
391            }
392
393            if resources.poll_attempt > 0
394                && !guest
395                    .thread_state()
396                    .with_detfd(call.fd(), |detfd| {
397                        detfd.open_file_id() == expected_open_file
398                    })
399                    .unwrap_or(false)
400            {
401                break if written_total > 0 {
402                    Ok(written_total as i64)
403                } else {
404                    self.refuse_unserviceable_operation(guest, Sysno::writev, Errno::EOPNOTSUPP)
405                        .await
406                };
407            }
408
409            let result = if atomic_pipe_write {
410                self.execute_atomic_pipe_writev_attempt(guest, call, &iovecs, atomic_scratch_iov)
411                    .await
412            } else {
413                match subtool {
414                    Some(detcore) => {
415                        detcore
416                            .record_or_replay_preserving_tool_errors(guest, current)
417                            .await
418                    }
419                    None => guest.inject_with_retry(current).await.map_err(Error::from),
420                }
421            };
422            match result {
423                Ok(written) if written > 0 => {
424                    let written = match usize::try_from(written) {
425                        Ok(written) => written,
426                        Err(_) => break Err(Errno::EIO.into()),
427                    };
428                    written_total = match written_total.checked_add(written) {
429                        Some(written_total) => written_total,
430                        None => break Err(Errno::EIO.into()),
431                    };
432                    if written_total >= target {
433                        break Ok(written_total as i64);
434                    }
435                    if atomic_pipe_write {
436                        break Ok(written_total as i64);
437                    }
438                    current = match remaining_writev_segment(
439                        call.fd(),
440                        &iovecs,
441                        written_total,
442                        target - written_total,
443                    ) {
444                        Ok(Some(write)) => Syscall::Write(write),
445                        Ok(None) => break Ok(written_total as i64),
446                        Err(_) => break Ok(written_total as i64),
447                    };
448                }
449                Ok(0) => break Ok(written_total as i64),
450                Err(Error::Errno(Errno::EAGAIN)) => {
451                    if !atomic_pipe_write && matches!(current, Syscall::Writev(_)) {
452                        current = match remaining_writev_segment(call.fd(), &iovecs, 0, target) {
453                            Ok(Some(write)) => Syscall::Write(write),
454                            Ok(None) => break Ok(0),
455                            Err(error) => break Err(error.into()),
456                        };
457                    }
458                }
459                Err(error) => {
460                    break finish_partial_record_or_replay_write(written_total as i64, error);
461                }
462                Ok(_) => break Err(Errno::EIO.into()),
463            }
464
465            resources.poll_attempt += 1;
466            tracing::trace!(
467                "Retry #{} for {}blocking pipe writev after EAGAIN: {}",
468                resources.poll_attempt,
469                if atomic_pipe_write { "atomic " } else { "" },
470                call.display(&guest.memory())
471            );
472            record_retry_event(guest, call).await;
473        };
474
475        restore_signals_after_disposition(guest, old_mask_addr).await?;
476        result
477    }
478
479    // AUTONOMOUS-BOT-IMPLEMENTED
480    // TODO-HUMAN-REVIEW(#2176): Review scalar blocking-pipe write completion and
481    // the fail-closed descriptor-replacement boundary.
482    /// Complete a logically blocking scalar pipe write after Hermit has made the pipe
483    /// physically nonblocking.
484    ///
485    /// Linux may return a positive short write for a request larger than `PIPE_BUF`, then
486    /// continue blocking for the remainder. Hermit's physical `O_NONBLOCK` is an internal
487    /// scheduler mechanism, so exposing that first short write changes guest behavior. Retry
488    /// the unconsumed suffix until the logical write completes, a signal arrives, or a real
489    /// error occurs. A signal or error after progress returns the partial byte count, matching
490    /// Linux.
491    ///
492    /// A concurrent close/dup2 can replace the numeric fd while this helper is yielded. Linux
493    /// keeps the original open-file description alive inside a blocking syscall, but Reverie
494    /// does not yet expose a backend-neutral retained-fd handle. Detect replacement before a
495    /// retry and fail closed rather than writing the suffix into an unrelated object.
496    pub async fn execute_blocking_pipe_write<G: Guest<Self>>(
497        &self,
498        guest: &mut G,
499        call: syscalls::Write,
500        expected_open_file: OpenFileId,
501    ) -> Result<i64, Error> {
502        const MAX_RW_COUNT: usize = 0x7fff_f000;
503
504        let target = call.len().min(MAX_RW_COUNT);
505        if target == 0 {
506            return self.execute_nonblockable_fd_syscall(guest, call).await;
507        }
508
509        tracing::trace!(
510            "NonblockableSyscall: converting to nonblocking syscall (internal polling): write"
511        );
512        let mut resources = Resources::new(guest.thread_state().dettid);
513        resources.insert(ResourceID::InternalIOPolling, Permission::W);
514        resources.fyi(call.name());
515        let subtool = self.cfg.recordreplay_modes.then_some(self);
516        let mut current = call;
517        let mut written_total = 0usize;
518
519        loop {
520            if resources.poll_attempt > 0
521                && matches!(
522                    resource_request(guest, resources.clone()).await,
523                    ResumeStatus::Signaled(_)
524                )
525            {
526                break if written_total > 0 {
527                    Ok(written_total as i64)
528                } else {
529                    Err(call.signal_interrupt_errno().into())
530                };
531            }
532
533            if resources.poll_attempt > 0
534                && !guest
535                    .thread_state()
536                    .with_detfd(call.fd(), |detfd| {
537                        detfd.open_file_id() == expected_open_file
538                    })
539                    .unwrap_or(false)
540            {
541                break if written_total > 0 {
542                    Ok(written_total as i64)
543                } else {
544                    self.refuse_unserviceable_operation(guest, Sysno::write, Errno::EOPNOTSUPP)
545                        .await
546                };
547            }
548
549            let result = match subtool {
550                Some(detcore) => {
551                    detcore
552                        .record_or_replay_preserving_tool_errors(guest, current)
553                        .await
554                }
555                None => guest.inject_with_retry(current).await.map_err(Error::from),
556            };
557            match result {
558                Ok(written) if written > 0 => {
559                    let written = usize::try_from(written).map_err(|_| Errno::EIO)?;
560                    let remaining = target.checked_sub(written_total).ok_or(Errno::EIO)?;
561                    if written > remaining {
562                        break Err(Errno::EIO.into());
563                    }
564                    written_total = written_total.checked_add(written).ok_or(Errno::EIO)?;
565                    if written_total == target {
566                        break Ok(written_total as i64);
567                    }
568                    let Some(buffer) = call.buf() else {
569                        break Err(Errno::EFAULT.into());
570                    };
571                    let Some(next_buffer) = buffer
572                        .as_raw()
573                        .checked_add(written_total)
574                        .and_then(Addr::<u8>::from_raw)
575                    else {
576                        break finish_partial_record_or_replay_write(
577                            written_total as i64,
578                            Errno::EFAULT.into(),
579                        );
580                    };
581                    current = call
582                        .with_buf(Some(next_buffer))
583                        .with_len(target - written_total);
584                }
585                Ok(0) => break Ok(written_total as i64),
586                Err(Error::Errno(Errno::EAGAIN)) => {}
587                Err(error) => {
588                    break finish_partial_record_or_replay_write(written_total as i64, error);
589                }
590                Ok(_) => break Err(Errno::EIO.into()),
591            }
592
593            resources.poll_attempt += 1;
594            tracing::trace!(
595                "Retry #{} for blocking pipe write: {}",
596                resources.poll_attempt,
597                call.display(&guest.memory())
598            );
599            record_retry_event(guest, call).await;
600        }
601    }
602
603    async fn execute_atomic_pipe_writev_attempt<G: Guest<Self>>(
604        &self,
605        guest: &mut G,
606        call: syscalls::Writev,
607        iovecs: &[(usize, usize)],
608        stack_iov: Option<usize>,
609    ) -> Result<i64, Error> {
610        if let Some(stack_iov) = stack_iov {
611            let scratch_call = call.with_iov(Addr::from_raw(stack_iov));
612            return if self.cfg.recordreplay_modes {
613                self.record_or_replay_preserving_tool_errors(guest, scratch_call)
614                    .await
615            } else {
616                guest
617                    .inject_with_retry(scratch_call)
618                    .await
619                    .map_err(Error::from)
620            };
621        }
622
623        let mapping_len = iovecs
624            .len()
625            .checked_mul(std::mem::size_of::<libc::iovec>())
626            .expect("validated iovec count cannot overflow scratch length");
627        let mapped = guest
628            .inject_with_retry(Syscall::Mmap(
629                syscalls::Mmap::new()
630                    .with_addr(None)
631                    .with_len(mapping_len)
632                    .with_prot(ProtFlags::PROT_READ | ProtFlags::PROT_WRITE)
633                    .with_flags(MapFlags::MAP_PRIVATE | MapFlags::MAP_ANONYMOUS)
634                    .with_fd(-1)
635                    .with_offset(0),
636            ))
637            .await
638            .unwrap_or_else(|error| panic!("failed to map atomic writev scratch: {error}"));
639        let mapped = usize::try_from(mapped)
640            .unwrap_or_else(|_| panic!("atomic writev scratch mmap returned {mapped}"));
641        let scratch_iov = Addr::<libc::iovec>::from_raw(mapped)
642            .unwrap_or_else(|| panic!("atomic writev scratch mmap returned a null address"));
643        let mapping_addr: Addr<libc::c_void> = scratch_iov.cast();
644        let write_result = {
645            let raw_iovecs: Vec<libc::iovec> = iovecs
646                .iter()
647                .map(|(base, length)| libc::iovec {
648                    iov_base: *base as *mut libc::c_void,
649                    iov_len: *length,
650                })
651                .collect();
652            // SAFETY: the injected anonymous mapping is exclusively owned scratch space.
653            guest
654                .memory()
655                .write_values(unsafe { scratch_iov.into_mut() }, &raw_iovecs)
656        };
657        if let Err(write_error) = write_result {
658            guest
659                .inject_with_retry(Syscall::Munmap(
660                    syscalls::Munmap::new()
661                        .with_addr(Some(mapping_addr))
662                        .with_len(mapping_len),
663                ))
664                .await
665                .unwrap_or_else(|cleanup_error| {
666                    panic!(
667                        "failed to populate atomic writev scratch ({write_error}); cleanup failed ({cleanup_error})"
668                    )
669                });
670            panic!("failed to populate atomic writev scratch: {write_error}");
671        }
672
673        let scratch_call = call.with_iov(Some(scratch_iov));
674        let result = if self.cfg.recordreplay_modes {
675            self.record_or_replay_preserving_tool_errors(guest, scratch_call)
676                .await
677        } else {
678            guest
679                .inject_with_retry(scratch_call)
680                .await
681                .map_err(Error::from)
682        };
683        guest
684            .inject_with_retry(Syscall::Munmap(
685                syscalls::Munmap::new()
686                    .with_addr(Some(mapping_addr))
687                    .with_len(mapping_len),
688            ))
689            .await
690            .unwrap_or_else(|error| panic!("failed to unmap atomic writev scratch: {error}"));
691        result
692    }
693
694    /// Override physically_nonblocking to true for the file descriptor, if appropriate.
695    pub fn maybe_set_nonblocking_fd<G: Guest<Self>>(&self, guest: &G, fd: i32) {
696        if self.cfg.sequentialize_threads && !self.cfg.debug_externalize_sockets {
697            guest
698                .thread_state()
699                .with_detfd(fd, |detfd| {
700                    detfd.set_physically_nonblocking();
701                })
702                .unwrap();
703        }
704    }
705}
706
707fn remaining_writev_segment(
708    fd: i32,
709    iovecs: &[(usize, usize)],
710    mut consumed: usize,
711    remaining_limit: usize,
712) -> Result<Option<syscalls::Write>, Errno> {
713    for (base, length) in iovecs {
714        if consumed >= *length {
715            consumed -= *length;
716            continue;
717        }
718        let base = base.checked_add(consumed).ok_or(Errno::EFAULT)?;
719        let buffer = Addr::<u8>::from_raw(base).ok_or(Errno::EFAULT)?;
720        return Ok(Some(
721            syscalls::Write::new()
722                .with_fd(fd)
723                .with_buf(Some(buffer))
724                .with_len((*length - consumed).min(remaining_limit)),
725        ));
726    }
727    Ok(None)
728}
729
730/// A blocking syscall that involves a fail descriptor may be handled in these three ways:
731#[derive(PartialEq, Eq, Debug)]
732pub enum IOAction {
733    /// It may physically block and we can't change that.  Treat it as ExternalBlockingIO.
734    Blocking,
735    /// We can nonblockize and retry the call.
736    NonblockizeRetry,
737    /// The call is nonblocking already, and safe to execute.
738    PassThru,
739}
740
741/// Returns the strategy for an FD-based call that may block when executed.
742///
743/// Failure means descriptor metadata could not classify the call; the caller must preserve the
744/// kernel's authority over the syscall result rather than exposing this advisory lookup error.
745pub fn ioaction_based_on_fd_status<
746    G: Guest<Detcore<T>>,
747    T: RecordOrReplay,
748    C: SyscallInfo + Into<Syscall>,
749>(
750    guest: &mut G,
751    call: C,
752) -> Result<IOAction, Errno> {
753    let wrapped: Syscall = call.into();
754    let fd = get_fd(wrapped).unwrap_or_else(|| panic!("Failed to get fd for {}", call.name()));
755    let (phys, virt) = guest.thread_state().with_detfd(fd, |detfd| {
756        (detfd.physically_nonblocking(), detfd.is_nonblocking())
757    })?;
758    tracing::trace!(
759        "Checking FD {} for nonblocking: physical {} / virtual {}",
760        fd,
761        phys,
762        virt
763    );
764    if virt && !phys {
765        // TF: simulate nonblocking on top of physically blocking? How?
766        panic!(
767            "Invariant violation, fd {}: we cannot simulate nonblocking behavior when set to blocking mode in the kernel.",
768            fd
769        );
770    } else if !virt && !phys {
771        // FF: logically blocking, physically blocking, this could only work with BlockingExternalIO.
772        Ok(IOAction::Blocking)
773    } else if virt && phys {
774        // TT: both nonblocking, so firing once is sufficient
775        Ok(IOAction::PassThru)
776    } else {
777        // FT: Need to simulate blocking on top of nonblocking.
778        Ok(IOAction::NonblockizeRetry)
779    }
780}
781
782/// Does this single-fd syscall operate on a container-INTERNAL file descriptor?
783///
784/// Currently this recognizes pipes, whose two endpoints are always both owned by guest
785/// processes inside the deterministic container. Internal pipes are made physically
786/// nonblocking (see `handle_pipe2`) so a potentially-blocking op on them can use the
787/// deterministic `InternalIOPolling` nonblockize-and-retry strategy instead of
788/// `BlockingExternalIO`. Treating an internal pipe as external blocking IO deadlocks
789/// the sequentialized scheduler in record/replay, because a pipe reader and its paired
790/// writer are not independent.
791///
792/// Sockets are intentionally NOT classified as internal here: there is no reliable
793/// internal-vs-external socket detection yet (loopback / AF_UNIX-to-another-guest vs a
794/// real host peer), so sockets conservatively remain external. Syscalls whose fd is not
795/// directly extractable (e.g. poll/ppoll, which carry a pointer to an fd array) return
796/// false and keep their existing handling.
797pub fn syscall_targets_internal_fd<G: Guest<Detcore<T>>, T: RecordOrReplay>(
798    guest: &mut G,
799    call: Syscall,
800) -> bool {
801    match get_fd(call) {
802        Some(fd) => guest
803            .thread_state()
804            .with_detfd(fd, |detfd| matches!(detfd.ty(), FdType::Pipe))
805            .unwrap_or(false),
806        None => false,
807    }
808}
809
810/// A large subset of system calls have a single, unique file descriptor argument.  This
811/// is a convenience function for grabbing that argument.
812///
813/// It does not cover system calls with multiple fd arguments, with pointers to heap
814/// structures that contain fds.
815pub(crate) fn get_fd(s: Syscall) -> Option<i32> {
816    match s {
817        Syscall::Recvfrom(s) => Some(s.fd()),
818        Syscall::Recvmsg(s) => Some(s.sockfd()),
819        Syscall::Recvmmsg(s) => Some(s.fd()),
820        Syscall::Sendto(s) => Some(s.fd()),
821        Syscall::Sendmsg(s) => Some(s.fd()),
822        Syscall::Sendmmsg(s) => Some(s.sockfd()),
823        Syscall::Accept(s) => Some(s.sockfd()),
824        Syscall::Accept4(s) => Some(s.sockfd()),
825        Syscall::Connect(s) => Some(s.fd()),
826        Syscall::Bind(s) => Some(s.fd()),
827        Syscall::Listen(s) => Some(s.fd()),
828        Syscall::Getsockname(s) => Some(s.fd()),
829        Syscall::Getpeername(s) => Some(s.fd()),
830        Syscall::Setsockopt(s) => Some(s.fd()),
831        Syscall::Getsockopt(s) => Some(s.fd()),
832
833        Syscall::Read(s) => Some(s.fd()),
834        Syscall::Write(s) => Some(s.fd()),
835        Syscall::Close(s) => Some(s.fd()),
836        Syscall::Fstat(s) => Some(s.fd()),
837        Syscall::Lseek(s) => Some(s.fd()),
838        Syscall::Mmap(s) => Some(s.fd()),
839        Syscall::Ioctl(s) => Some(s.fd()),
840        Syscall::Pread64(s) => Some(s.fd()),
841        Syscall::Pwrite64(s) => Some(s.fd()),
842        Syscall::Readv(s) => Some(s.fd()),
843        Syscall::Writev(s) => Some(s.fd()),
844
845        Syscall::Shutdown(s) => Some(s.fd()),
846        Syscall::Fcntl(s) => Some(s.fd()),
847        Syscall::Flock(s) => Some(s.fd()),
848        Syscall::Fsync(s) => Some(s.fd()),
849        Syscall::Fdatasync(s) => Some(s.fd()),
850        Syscall::Ftruncate(s) => Some(s.fd()),
851        Syscall::Fchdir(s) => Some(s.fd()),
852        Syscall::Fchmod(s) => Some(s.fd()),
853        Syscall::Fchown(s) => Some(s.fd()),
854        Syscall::Fstatfs(s) => Some(s.fd()),
855        Syscall::Readahead(s) => Some(s.fd()),
856        Syscall::Fsetxattr(s) => Some(s.fd()),
857        Syscall::Fgetxattr(s) => Some(s.fd()),
858        Syscall::Flistxattr(s) => Some(s.fd()),
859        Syscall::Fremovexattr(s) => Some(s.fd()),
860        Syscall::Fadvise64(s) => Some(s.fd()),
861        Syscall::InotifyAddWatch(s) => Some(s.fd()),
862        Syscall::InotifyRmWatch(s) => Some(s.fd()),
863        Syscall::SyncFileRange(s) => Some(s.fd()),
864        Syscall::Vmsplice(s) => Some(s.fd()),
865        Syscall::Utimensat(s) => Some(s.dirfd()),
866        Syscall::Signalfd(s) => Some(s.fd()),
867        Syscall::Fallocate(s) => Some(s.fd()),
868        Syscall::TimerfdSettime(s) => Some(s.fd()),
869        Syscall::TimerfdGettime(s) => Some(s.fd()),
870        Syscall::Signalfd4(s) => Some(s.fd()),
871        Syscall::Preadv(s) => Some(s.fd()),
872        Syscall::Pwritev(s) => Some(s.fd()),
873        Syscall::Syncfs(s) => Some(s.fd()),
874        Syscall::Setns(s) => Some(s.fd()),
875        Syscall::FinitModule(s) => Some(s.fd()),
876        Syscall::Preadv2(s) => Some(s.fd()),
877        Syscall::Pwritev2(s) => Some(s.fd()),
878
879        Syscall::Openat(s) => Some(s.dirfd()),
880        Syscall::Mkdirat(s) => Some(s.dirfd()),
881        Syscall::Mknodat(s) => Some(s.dirfd()),
882        Syscall::Fchownat(s) => Some(s.dirfd()),
883        Syscall::Futimesat(s) => Some(s.dirfd()),
884        Syscall::Newfstatat(s) => Some(s.dirfd()),
885        Syscall::Unlinkat(s) => Some(s.dirfd()),
886        Syscall::Readlinkat(s) => Some(s.dirfd()),
887        Syscall::Fchmodat(s) => Some(s.dirfd()),
888        Syscall::Faccessat(s) => Some(s.dirfd()),
889        Syscall::NameToHandleAt(s) => Some(s.dirfd()),
890        Syscall::Execveat(s) => Some(s.dirfd()),
891        Syscall::Statx(s) => Some(s.dirfd()),
892        Syscall::Symlinkat(s) => Some(s.newdirfd()),
893        Syscall::PerfEventOpen(s) => Some(s.group_fd()),
894        Syscall::OpenByHandleAt(s) => Some(s.mount_fd()),
895
896        Syscall::EpollCtl(s) => Some(s.epfd()),
897        Syscall::EpollWait(s) => Some(s.epfd()),
898        Syscall::EpollPwait(s) => Some(s.epfd()),
899
900        // Ambiguous, 2 fds, no answer:
901        Syscall::Dup2(_) => None,
902        // Ambiguous, 2 fds, no answer:
903        Syscall::Sendfile(_) => None,
904        // Ambiguous, 2 fds, no answer:
905        Syscall::Renameat(_) => None,
906        // Ambiguous, 2 fds, no answer:
907        Syscall::Linkat(_) => None,
908        // Ambiguous, 2 fds, no answer:
909        Syscall::FanotifyMark(_) => None,
910        // Ambiguous, 2 fds, no answer:
911        Syscall::Renameat2(_) => None,
912        // Ambiguous, 2 fds, no answer:
913        Syscall::Dup3(_) => None,
914        // Ambiguous, 2 fds, no answer:
915        Syscall::KexecLoad(_) => None,
916
917        // Takes a pointer to fd, not directly accessible:
918        Syscall::Poll(_) => None,
919        // Takes a pointer to fd, not directly accessible:
920        Syscall::Ppoll(_) => None,
921
922        _ => None,
923    }
924}
925
926/// A system call which may or may not block, but which can be MADE nonblocking.
927// `async_trait` rewrites `into_nonblocking` to return a boxed future and marks it
928// `#[must_use]`; the future is already `#[must_use]` in its own right, so clippy sees a double
929// annotation. Both are generated, so the lint's own suggestion -- give the `must_use` an explicit
930// reason -- cannot be applied at this source. Allowed at the item rather than crate-wide so any
931// hand-written double `must_use` elsewhere still fails `#![deny(clippy::all)]` (detcore/src/lib.rs:35).
932// Appeared with nightly-2026-08-08 (rustc 1.99.0-nightly 1a98b1e13) against an unchanged tree.
933#[allow(clippy::double_must_use)]
934#[async_trait]
935pub trait NonblockableSyscall: SyscallInfo {
936    /// Convert the system call to a nonblocking version of itself.  Sometimes this means
937    /// setting a zero timeout, and sometimes it means something else.
938    ///
939    /// This may need to stack allocate, so it returns a StackGuard.
940    async fn into_nonblocking<T: RecordOrReplay, G: Guest<Detcore<T>>>(
941        self,
942        guest: &mut G,
943    ) -> (Self, Option<<G::Stack as Stack>::StackGuard>);
944
945    /// Check if the result (in nonblocking mode) is analogous to blocking in blocking mode.
946    /// I.e. the result means "try again".
947    fn syscall_would_have_blocked(&self, res: Result<i64, Errno>) -> bool {
948        res == Ok(0)
949    }
950
951    /// Return the errno used when a signal interrupts this internally polled syscall.
952    /// Most blocking I/O is restartable when its handler uses `SA_RESTART`.
953    fn signal_interrupt_errno(&self) -> Errno {
954        Errno::ERESTARTSYS
955    }
956
957    /// Convert a physical nonblocking completion into the result expected by the guest.
958    /// `retried` is true after a prior result was classified as blocked.
959    fn normalize_nonblocking_result(
960        &self,
961        res: Result<i64, Errno>,
962        _retried: bool,
963    ) -> Result<i64, Errno> {
964        res
965    }
966}
967
968/// A system call which can logically timeout and then would return a given value
969/// indicating that timeout.
970pub trait TimeoutableSyscall: SyscallInfo {
971    /// What would the syscall return IF it timed out?
972    fn timeout_return_val(&self) -> Result<i64, Errno>;
973}
974
975fn interrupted_write_result<C: NonblockableSyscall>(
976    call: &C,
977    written_total: usize,
978    disposition: Option<WaitSignalDisposition>,
979) -> Option<Result<i64, Error>> {
980    match disposition {
981        None => None,
982        Some(_) if written_total > 0 => Some(Ok(written_total as i64)),
983        Some(_) => Some(Err(call.signal_interrupt_errno().into())),
984    }
985}
986
987fn pipe_writev_resources(dettid: DetTid, call: reverie::syscalls::Writev) -> Resources {
988    let mut resources = Resources::new(dettid);
989    resources.insert(ResourceID::InternalIOPolling, Permission::W);
990    resources.fyi(call.name());
991    resources.set_signal_interrupt_errno(call.signal_interrupt_errno());
992    resources
993}
994
995#[async_trait]
996impl NonblockableSyscall for reverie::syscalls::Poll {
997    async fn into_nonblocking<T: RecordOrReplay, G: Guest<Detcore<T>>>(
998        self,
999        _guest: &mut G,
1000    ) -> (Self, Option<<G::Stack as Stack>::StackGuard>) {
1001        (self.with_timeout(0), None)
1002    }
1003
1004    fn signal_interrupt_errno(&self) -> Errno {
1005        Errno::EINTR
1006    }
1007}
1008
1009impl TimeoutableSyscall for reverie::syscalls::Poll {
1010    fn timeout_return_val(&self) -> Result<i64, Errno> {
1011        Ok(0)
1012    }
1013}
1014
1015#[async_trait]
1016impl NonblockableSyscall for reverie::syscalls::Ppoll {
1017    async fn into_nonblocking<T: RecordOrReplay, G: Guest<Detcore<T>>>(
1018        self,
1019        guest: &mut G,
1020    ) -> (Self, Option<<G::Stack as Stack>::StackGuard>) {
1021        let (tp, guard) = zero_timespec(guest).await;
1022        // SAFETY: `tp` points to exclusively owned scratch storage kept alive by `guard`.
1023        let tp = unsafe { tp.into_mut() };
1024        (self.with_timeout(Some(tp)), Some(guard))
1025    }
1026
1027    fn signal_interrupt_errno(&self) -> Errno {
1028        Errno::EINTR
1029    }
1030}
1031
1032impl TimeoutableSyscall for reverie::syscalls::Ppoll {
1033    fn timeout_return_val(&self) -> Result<i64, Errno> {
1034        Ok(0)
1035    }
1036}
1037
1038#[async_trait]
1039impl NonblockableSyscall for reverie::syscalls::EpollWait {
1040    async fn into_nonblocking<T: RecordOrReplay, G: Guest<Detcore<T>>>(
1041        self,
1042        _guest: &mut G,
1043    ) -> (Self, Option<<G::Stack as Stack>::StackGuard>) {
1044        (self.with_timeout(0), None)
1045    }
1046
1047    fn signal_interrupt_errno(&self) -> Errno {
1048        Errno::EINTR
1049    }
1050}
1051
1052impl TimeoutableSyscall for reverie::syscalls::EpollWait {
1053    fn timeout_return_val(&self) -> Result<i64, Errno> {
1054        Ok(0)
1055    }
1056}
1057
1058// `epoll_pwait` was the one member of the
1059// poll/epoll family with no nonblocking form, even though `Ppoll` -- the
1060// sigmask variant of `poll` -- has had one all along. Programs that issue
1061// `epoll_pwait` DIRECTLY (libuv does, which is how the `cmake` hang surfaced)
1062// therefore reached an unhandled path while the `EpollWait` impl above went
1063// unused by them. Note this is NOT glibc's `epoll_wait(2)` on x86_64: glibc
1064// only spells it `epoll_pwait` where `__NR_epoll_wait` is absent, which x86_64
1065// is not. With a NULL sigmask the two calls are semantically identical, so the
1066// nonblocking form is the same: timeout 0, EINTR, and a 0 (no events) timeout
1067// return.
1068#[async_trait]
1069impl NonblockableSyscall for reverie::syscalls::EpollPwait {
1070    async fn into_nonblocking<T: RecordOrReplay, G: Guest<Detcore<T>>>(
1071        self,
1072        _guest: &mut G,
1073    ) -> (Self, Option<<G::Stack as Stack>::StackGuard>) {
1074        (self.with_timeout(0), None)
1075    }
1076
1077    fn signal_interrupt_errno(&self) -> Errno {
1078        Errno::EINTR
1079    }
1080}
1081
1082impl TimeoutableSyscall for reverie::syscalls::EpollPwait {
1083    fn timeout_return_val(&self) -> Result<i64, Errno> {
1084        Ok(0)
1085    }
1086}
1087
1088async fn zero_timespec<'stack, T: RecordOrReplay, G: Guest<Detcore<T>>>(
1089    guest: &mut G,
1090) -> (Addr<'stack, Timespec>, <G::Stack as Stack>::StackGuard) {
1091    let mut stack = guest.stack().await;
1092    let tp_val = Timespec {
1093        tv_sec: 0,
1094        tv_nsec: 0,
1095    };
1096    let tp = stack.push(tp_val);
1097    let guard = stack.commit().expect("stack.commit to succeed");
1098    (tp, guard)
1099}
1100
1101#[async_trait]
1102impl NonblockableSyscall for reverie::syscalls::Wait4 {
1103    async fn into_nonblocking<T: RecordOrReplay, G: Guest<Detcore<T>>>(
1104        self,
1105        _guest: &mut G,
1106    ) -> (Self, Option<<G::Stack as Stack>::StackGuard>) {
1107        let call2 = self.with_options(self.options() | WaitPidFlag::WNOHANG);
1108        (call2, None)
1109    }
1110
1111    // Child has not changed state yet, so we go to the scheduler and wait to poll again.
1112    // In scenarios with lots of outstanding waits, this polling strategy can change the asymptotic
1113    // complexity of the program. Ideally, we would model the blocking `wait4` (and process state
1114    // transitions) directly in the scheduler, and execute it only when we know it will complete.
1115    //
1116    // The polling backoff strategy mitigates this problem however.
1117    fn syscall_would_have_blocked(&self, res: Result<i64, Errno>) -> bool {
1118        res == Ok(0)
1119    }
1120}
1121
1122#[async_trait]
1123/// Used only for FUTEX_WAIT
1124impl NonblockableSyscall for reverie::syscalls::Futex {
1125    async fn into_nonblocking<T: RecordOrReplay, G: Guest<Detcore<T>>>(
1126        self,
1127        guest: &mut G,
1128    ) -> (Self, Option<<G::Stack as Stack>::StackGuard>) {
1129        let (tp, guard) = zero_timespec(guest).await;
1130        (self.with_timeout(Some(tp)), Some(guard))
1131    }
1132
1133    fn syscall_would_have_blocked(&self, res: Result<i64, Errno>) -> bool {
1134        // EAGAIN can mean the futex wait's compare-and-block failed and we should return that to
1135        // the guest.  With timeout=0, the timeout is what shows that it would have blocked.
1136        res == Err(Errno::ETIMEDOUT)
1137    }
1138}
1139
1140impl TimeoutableSyscall for reverie::syscalls::Futex {
1141    fn timeout_return_val(&self) -> Result<i64, Errno> {
1142        Err(Errno::ETIMEDOUT)
1143    }
1144}
1145
1146#[async_trait]
1147impl NonblockableSyscall for reverie::syscalls::RtSigtimedwait {
1148    async fn into_nonblocking<T: RecordOrReplay, G: Guest<Detcore<T>>>(
1149        self,
1150        guest: &mut G,
1151    ) -> (Self, Option<<G::Stack as Stack>::StackGuard>) {
1152        // This is a bit more complicated because we need a new timespec to point to in
1153        // the guest memory.
1154        let (tp, guard) = zero_timespec(guest).await;
1155        (self.with_timeout(Some(tp)), Some(guard))
1156    }
1157
1158    fn syscall_would_have_blocked(&self, res: Result<i64, Errno>) -> bool {
1159        res == Err(Errno::EAGAIN)
1160    }
1161
1162    fn signal_interrupt_errno(&self) -> Errno {
1163        Errno::EINTR
1164    }
1165}
1166
1167impl TimeoutableSyscall for reverie::syscalls::RtSigtimedwait {
1168    fn timeout_return_val(&self) -> Result<i64, Errno> {
1169        Err(Errno::EAGAIN)
1170    }
1171}
1172
1173/// While the read syscall is quite general, this nonblocking capacity is used
1174/// ONLY for sockets and pipes.
1175#[async_trait]
1176impl NonblockableSyscall for reverie::syscalls::Read {
1177    async fn into_nonblocking<T: RecordOrReplay, G: Guest<Detcore<T>>>(
1178        self,
1179        guest: &mut G,
1180    ) -> (Self, Option<<G::Stack as Stack>::StackGuard>) {
1181        network_comm_syscall(self, guest)
1182    }
1183
1184    fn syscall_would_have_blocked(&self, res: Result<i64, Errno>) -> bool {
1185        // A return value of Ok(0) indicates end of file.
1186        // Note that we've ruled out 0-count reads before this point.
1187        res == Err(Errno::EAGAIN) || res == Err(Errno::EWOULDBLOCK)
1188    }
1189}
1190
1191/// While the read syscall is quite general, this nonblocking capacity is used
1192/// ONLY for sockets and pipes.
1193#[async_trait]
1194impl NonblockableSyscall for reverie::syscalls::Write {
1195    async fn into_nonblocking<T: RecordOrReplay, G: Guest<Detcore<T>>>(
1196        self,
1197        guest: &mut G,
1198    ) -> (Self, Option<<G::Stack as Stack>::StackGuard>) {
1199        network_comm_syscall(self, guest)
1200    }
1201
1202    fn syscall_would_have_blocked(&self, res: Result<i64, Errno>) -> bool {
1203        // A return value of Ok(0) indicates end of file.
1204        // Note that we've ruled out 0-count reads before this point.
1205        res == Err(Errno::EAGAIN) || res == Err(Errno::EWOULDBLOCK)
1206    }
1207}
1208
1209// AUTONOMOUS-BOT-IMPLEMENTED
1210// TODO-HUMAN-REVIEW(#794)
1211/// Vectored reads have the same blocking behavior as scalar reads on pipes and sockets.
1212#[async_trait]
1213impl NonblockableSyscall for reverie::syscalls::Readv {
1214    async fn into_nonblocking<T: RecordOrReplay, G: Guest<Detcore<T>>>(
1215        self,
1216        guest: &mut G,
1217    ) -> (Self, Option<<G::Stack as Stack>::StackGuard>) {
1218        network_comm_syscall(self, guest)
1219    }
1220
1221    fn syscall_would_have_blocked(&self, res: Result<i64, Errno>) -> bool {
1222        res == Err(Errno::EAGAIN) || res == Err(Errno::EWOULDBLOCK)
1223    }
1224}
1225
1226// AUTONOMOUS-BOT-IMPLEMENTED
1227// TODO-HUMAN-REVIEW(#547)
1228/// Vectored writes have the same blocking behavior as scalar writes on pipes and sockets.
1229#[async_trait]
1230impl NonblockableSyscall for reverie::syscalls::Writev {
1231    async fn into_nonblocking<T: RecordOrReplay, G: Guest<Detcore<T>>>(
1232        self,
1233        guest: &mut G,
1234    ) -> (Self, Option<<G::Stack as Stack>::StackGuard>) {
1235        network_comm_syscall(self, guest)
1236    }
1237
1238    fn syscall_would_have_blocked(&self, res: Result<i64, Errno>) -> bool {
1239        res == Err(Errno::EAGAIN) || res == Err(Errno::EWOULDBLOCK)
1240    }
1241}
1242
1243/// A common helper shared among several network syscalls.
1244/// We can't actually CONVERT these syscalls into nonblocking, but we can assert that they are by
1245/// checking the status of their file descriptor.
1246fn network_comm_syscall<T: RecordOrReplay, G: Guest<Detcore<T>>, C: SyscallInfo + Into<Syscall>>(
1247    call: C,
1248    guest: &mut G,
1249) -> (C, Option<<G::Stack as Stack>::StackGuard>) {
1250    // Already nonblocking because we've assured the socket is.
1251    let fd = get_fd(call.into()).unwrap_or_else(|| {
1252        panic!(
1253            "network_comm_syscall called on invalid syscall / unknown fd: {}",
1254            call.name()
1255        );
1256    });
1257    guest
1258        .thread_state()
1259        .with_detfd(fd, |detfd| {
1260            assert!(
1261                detfd.physically_nonblocking(),
1262                "expecting sockets/pipes to be physically nonblocking"
1263            );
1264        })
1265        .unwrap();
1266    (call, None)
1267}
1268
1269#[async_trait]
1270impl NonblockableSyscall for reverie::syscalls::Accept4 {
1271    async fn into_nonblocking<T: RecordOrReplay, G: Guest<Detcore<T>>>(
1272        self,
1273        guest: &mut G,
1274    ) -> (Self, Option<<G::Stack as Stack>::StackGuard>) {
1275        network_comm_syscall(self, guest)
1276    }
1277
1278    fn syscall_would_have_blocked(&self, res: Result<i64, Errno>) -> bool {
1279        res == Err(Errno::EAGAIN) || res == Err(Errno::EWOULDBLOCK)
1280    }
1281}
1282
1283impl TimeoutableSyscall for reverie::syscalls::Accept4 {
1284    fn timeout_return_val(&self) -> Result<i64, Errno> {
1285        Ok(0)
1286    }
1287}
1288
1289#[async_trait]
1290impl NonblockableSyscall for reverie::syscalls::Recvfrom {
1291    async fn into_nonblocking<T: RecordOrReplay, G: Guest<Detcore<T>>>(
1292        self,
1293        guest: &mut G,
1294    ) -> (Self, Option<<G::Stack as Stack>::StackGuard>) {
1295        network_comm_syscall(self, guest)
1296    }
1297
1298    fn syscall_would_have_blocked(&self, res: Result<i64, Errno>) -> bool {
1299        res == Err(Errno::EAGAIN) || res == Err(Errno::EWOULDBLOCK)
1300    }
1301}
1302
1303#[async_trait]
1304impl NonblockableSyscall for reverie::syscalls::Recvmsg {
1305    async fn into_nonblocking<T: RecordOrReplay, G: Guest<Detcore<T>>>(
1306        self,
1307        guest: &mut G,
1308    ) -> (Self, Option<<G::Stack as Stack>::StackGuard>) {
1309        network_comm_syscall(self, guest)
1310    }
1311
1312    fn syscall_would_have_blocked(&self, res: Result<i64, Errno>) -> bool {
1313        res == Err(Errno::EAGAIN) || res == Err(Errno::EWOULDBLOCK)
1314    }
1315}
1316
1317#[async_trait]
1318impl NonblockableSyscall for reverie::syscalls::Recvmmsg {
1319    // This system call has a timeout argument, but we ignore it because the underlying
1320    // socket is nonblocking anyway (in runs where we call this).
1321    async fn into_nonblocking<T: RecordOrReplay, G: Guest<Detcore<T>>>(
1322        self,
1323        guest: &mut G,
1324    ) -> (Self, Option<<G::Stack as Stack>::StackGuard>) {
1325        network_comm_syscall(self, guest)
1326    }
1327
1328    fn syscall_would_have_blocked(&self, res: Result<i64, Errno>) -> bool {
1329        res == Err(Errno::EAGAIN) || res == Err(Errno::EWOULDBLOCK)
1330    }
1331}
1332
1333#[async_trait]
1334impl NonblockableSyscall for reverie::syscalls::Sendto {
1335    async fn into_nonblocking<T: RecordOrReplay, G: Guest<Detcore<T>>>(
1336        self,
1337        guest: &mut G,
1338    ) -> (Self, Option<<G::Stack as Stack>::StackGuard>) {
1339        network_comm_syscall(self, guest)
1340    }
1341
1342    fn syscall_would_have_blocked(&self, res: Result<i64, Errno>) -> bool {
1343        res == Err(Errno::EAGAIN) || res == Err(Errno::EWOULDBLOCK)
1344    }
1345}
1346
1347#[async_trait]
1348impl NonblockableSyscall for reverie::syscalls::Sendmmsg {
1349    async fn into_nonblocking<T: RecordOrReplay, G: Guest<Detcore<T>>>(
1350        self,
1351        guest: &mut G,
1352    ) -> (Self, Option<<G::Stack as Stack>::StackGuard>) {
1353        network_comm_syscall(self, guest)
1354    }
1355
1356    fn syscall_would_have_blocked(&self, res: Result<i64, Errno>) -> bool {
1357        res == Err(Errno::EAGAIN) || res == Err(Errno::EWOULDBLOCK)
1358    }
1359}
1360
1361#[async_trait]
1362impl NonblockableSyscall for reverie::syscalls::Sendmsg {
1363    async fn into_nonblocking<T: RecordOrReplay, G: Guest<Detcore<T>>>(
1364        self,
1365        guest: &mut G,
1366    ) -> (Self, Option<<G::Stack as Stack>::StackGuard>) {
1367        network_comm_syscall(self, guest)
1368    }
1369
1370    fn syscall_would_have_blocked(&self, res: Result<i64, Errno>) -> bool {
1371        res == Err(Errno::EAGAIN) || res == Err(Errno::EWOULDBLOCK)
1372    }
1373}
1374
1375#[async_trait]
1376impl NonblockableSyscall for reverie::syscalls::Connect {
1377    async fn into_nonblocking<T: RecordOrReplay, G: Guest<Detcore<T>>>(
1378        self,
1379        guest: &mut G,
1380    ) -> (Self, Option<<G::Stack as Stack>::StackGuard>) {
1381        network_comm_syscall(self, guest)
1382    }
1383
1384    fn syscall_would_have_blocked(&self, res: Result<i64, Errno>) -> bool {
1385        res == Err(Errno::EAGAIN)
1386            || res == Err(Errno::EWOULDBLOCK)
1387            || res == Err(Errno::EINPROGRESS)
1388            || res == Err(Errno::EALREADY)
1389    }
1390
1391    fn normalize_nonblocking_result(
1392        &self,
1393        res: Result<i64, Errno>,
1394        retried: bool,
1395    ) -> Result<i64, Errno> {
1396        match (retried, res) {
1397            (true, Err(Errno::EISCONN)) => Ok(0),
1398            (_, res) => res,
1399        }
1400    }
1401}
1402
1403/// Transform a syscall to nonblocking, then retry it until it returns a successful result.
1404/// Retry a nonblockizable syscall (e.g. a pipe/socket read or write) until it succeeds.
1405///
1406/// `subtool` selects how each poll iteration executes the underlying syscall. Pass
1407/// `Some(detcore)` in record/replay mode for a container-INTERNAL fd (currently pipes):
1408/// each iteration is then routed through `Detcore::record_or_replay`, so the read's
1409/// bytes (and every intervening `EAGAIN`) are captured in the recording and reproduced
1410/// verbatim on replay. Without this, an internal-pipe read on the InternalIOPolling path
1411/// bypasses the recorder and replay reads live from a pipe whose cross-process writer
1412/// schedule is not reproduced -- the reader sees EOF instead of the recorded data and
1413/// replay desyncs. Pass `None` for plain `hermit run` (no recording) or for external fds.
1414pub async fn retry_nonblocking_syscall<T, G, C>(
1415    guest: &mut G,
1416    call: C,
1417    rsrc: Resources,
1418    subtool: Option<&Detcore<T>>,
1419) -> Result<i64, Error>
1420where
1421    C: NonblockableSyscall + Into<Syscall>,
1422    T: RecordOrReplay,
1423    G: Guest<Detcore<T>>,
1424{
1425    // Bogus 99 return value is dead code below:
1426    retry_nonblocking_syscall_helper(guest, call, rsrc, None, subtool).await
1427}
1428
1429/// Retry a non-blocking syscall until it succeeds. Set the timeout to zero for the actual
1430/// syscalls (retries), while monitoring the clock to see if/when the logical timeout
1431/// should trigger.  Timeout is passed as an ABSOLUTE TIME (not duration).
1432pub async fn retry_nonblocking_syscall_with_timeout<T, G, C>(
1433    guest: &mut G,
1434    call: C,
1435    rsrc: Resources,
1436    // Logical timeout:
1437    maybe_timeout: Option<LogicalTime>,
1438) -> Result<i64, Error>
1439where
1440    C: NonblockableSyscall + TimeoutableSyscall + Into<Syscall>,
1441    T: RecordOrReplay,
1442    G: Guest<Detcore<T>>,
1443{
1444    let maybe_tup = maybe_timeout.map(|t| (t, call.timeout_return_val()));
1445    // poll/epoll_wait/futex/rt_sigtimedwait keep their existing execution (raw
1446    // inject_with_retry): their record/replay handling is out of scope for the internal
1447    // pipe data-ordering fix, and their fds are not necessarily internal pipes.
1448    retry_nonblocking_syscall_helper(guest, call, rsrc, maybe_tup, None).await
1449}
1450
1451// Private helper.
1452async fn retry_nonblocking_syscall_helper<T, G, C>(
1453    guest: &mut G,
1454    call0: C,
1455    rsrc: Resources,
1456    maybe_timeout: Option<(LogicalTime, Result<i64, Errno>)>,
1457    subtool: Option<&Detcore<T>>,
1458) -> Result<i64, Error>
1459where
1460    C: NonblockableSyscall + Into<Syscall>,
1461    T: RecordOrReplay,
1462    G: Guest<Detcore<T>>,
1463{
1464    // The stack-allocated memory here needs to live across the loop, which means
1465    // surviving multiple syscall injections:
1466    let (call, _maybe_stackguard) = call0.into_nonblocking(guest).await;
1467    let mut rsrc = rsrc.clone();
1468
1469    loop {
1470        let resumed = match call.into() {
1471            Syscall::Read(read) if maybe_timeout.is_none() && _maybe_stackguard.is_none() => {
1472                crate::tool_global::polled_read_request(guest, read, rsrc.clone()).await
1473            }
1474            _ => resource_request(guest, rsrc.clone()).await,
1475        };
1476        if matches!(resumed, ResumeStatus::Signaled(_)) {
1477            let errno = call.signal_interrupt_errno();
1478            tracing::trace!(
1479                "retry_nonblocking_syscall: interrupted by signal before retrying {}: {:?}",
1480                call.display(&guest.memory()),
1481                errno
1482            );
1483            return Err(errno.into());
1484        }
1485        // Route through the record/replay subtool for internal pipes so each poll (an
1486        // EAGAIN, or the final data-bearing read) becomes one recorded event that replay
1487        // reproduces deterministically; otherwise execute the syscall directly.
1488        let res = match subtool {
1489            Some(detcore) => {
1490                detcore
1491                    .record_or_replay_preserving_tool_errors(guest, call)
1492                    .await
1493            }
1494            None => guest.inject_with_retry(call).await.map_err(Error::from),
1495        };
1496        let syscall_result = match res {
1497            Ok(value) => Ok(value),
1498            Err(Error::Errno(error)) => Err(error),
1499            Err(error) => return Err(error),
1500        };
1501        if call.syscall_would_have_blocked(syscall_result) {
1502            rsrc.poll_attempt += 1;
1503            if let Some((timeout, timeout_result)) = maybe_timeout {
1504                let new_time = thread_observe_time(guest).await;
1505                if new_time >= timeout {
1506                    tracing::trace!(
1507                        "Timing out syscall after #{} retries: {}",
1508                        rsrc.poll_attempt - 1,
1509                        call.display(&guest.memory())
1510                    );
1511                    return timeout_result.map_err(|e| e.into());
1512                } else {
1513                    tracing::trace!(
1514                        "Retry #{} for syscall due to result {:?}, {} from timeout: {}",
1515                        rsrc.poll_attempt,
1516                        syscall_result,
1517                        timeout - new_time,
1518                        call.display(&guest.memory())
1519                    );
1520                    record_retry_event(guest, call).await;
1521                }
1522            } else {
1523                tracing::trace!(
1524                    "Retry #{} for syscall due to result {:?}: {}",
1525                    rsrc.poll_attempt,
1526                    syscall_result,
1527                    call.display(&guest.memory())
1528                );
1529                record_retry_event(guest, call).await;
1530            }
1531        } else {
1532            let res = call
1533                .normalize_nonblocking_result(syscall_result, rsrc.poll_attempt > 0)
1534                .map_err(|e| e.into());
1535            tracing::trace!(
1536                "retry_nonblocking_syscall: syscall completed after {} retries: {} = {:?}",
1537                rsrc.poll_attempt,
1538                call.display(&guest.memory()),
1539                res
1540            );
1541            return res;
1542        }
1543    }
1544}
1545
1546pub(crate) async fn record_retry_event<G, C, T>(guest: &mut G, call: C)
1547where
1548    C: SyscallInfo,
1549    T: RecordOrReplay,
1550    G: Guest<Detcore<T>>,
1551{
1552    let dettid = guest.thread_state().dettid;
1553    let cfg = &guest.config();
1554    if cfg.sequentialize_threads && cfg.should_trace_schedevent() {
1555        trace_schedevent(
1556            guest,
1557            with_guest_time(
1558                guest,
1559                SchedEvent::syscall(dettid, call.number(), SyscallPhase::Polling),
1560            ),
1561            true,
1562        )
1563        .await;
1564    }
1565}
1566
1567// A helper function for enriching the schedevent with local information.
1568pub fn with_guest_time<G, T>(guest: &G, event: SchedEvent) -> SchedEvent
1569where
1570    G: Guest<Detcore<T>>,
1571    T: RecordOrReplay,
1572{
1573    let dettime = &guest.thread_state().thread_logical_time;
1574    event.with_dettime(dettime)
1575}
1576
1577// Enrich the event with the RIP register from the current guest state, but only if it is unset.
1578pub async fn with_guest_rip<G, T>(guest: &mut G, mut event: SchedEvent) -> SchedEvent
1579where
1580    G: Guest<Detcore<T>>,
1581    T: RecordOrReplay,
1582{
1583    assert!(event.end_rip.is_none());
1584
1585    let regs = guest.regs().await;
1586    let end_rip = NonZeroUsize::new(regs.rip.try_into().unwrap()).unwrap();
1587    event.end_rip = Some(end_rip);
1588    event
1589}
1590
1591// Convert to absolute logical time point for the timeout.
1592// 0 duration means no timeout, and this will return None.
1593pub async fn millis_duration_to_absolute_timeout<G: Guest<Detcore<T>>, T: RecordOrReplay>(
1594    guest: &mut G,
1595    timeout_millis: i32,
1596) -> Option<LogicalTime> {
1597    match positive_millis_as_nanos(timeout_millis) {
1598        Some(timeout_nanos) => nanos_duration_to_absolute_timeout(guest, timeout_nanos).await,
1599        None => None,
1600    }
1601}
1602
1603/// Milliseconds to nanoseconds for a strictly positive timeout; `None` for the
1604/// non-positive values Linux treats as "return immediately" (0) or "wait
1605/// forever" (-1), neither of which is a deadline.
1606///
1607/// Kept as a separate, unit-bracketed function on purpose. This conversion was
1608/// previously inlined as `* 1000` instead of `* 1_000_000`, which made every
1609/// finite deadline 1000x too short (a 1 ms timeout expired after 1 us). That
1610/// is invisible in an end-to-end test that only checks a syscall's return
1611/// value, so the arithmetic is pinned here by `millis_to_nanos_conversion`.
1612fn positive_millis_as_nanos(timeout_millis: i32) -> Option<u128> {
1613    (timeout_millis > 0).then(|| (timeout_millis as u128) * 1_000_000)
1614}
1615
1616// Convert to absolute logical time point for the timeout.
1617// 0 duration means no timeout, and this will return None.
1618pub async fn nanos_duration_to_absolute_timeout<G: Guest<Detcore<T>>, T: RecordOrReplay>(
1619    guest: &mut G,
1620    timeout_nanos: u128,
1621) -> Option<LogicalTime> {
1622    if timeout_nanos > 0 {
1623        let ns_delta = Duration::from_nanos(timeout_nanos as u64);
1624        let base_time = thread_observe_time(guest).await;
1625        let target_time = base_time + ns_delta;
1626        Some(target_time)
1627    } else {
1628        None
1629    }
1630}
1631
1632#[cfg(test)]
1633mod tests {
1634    use super::*;
1635
1636    /// Bracket the millisecond-to-nanosecond timeout conversion in both
1637    /// directions: the non-positive values that are NOT deadlines, and the
1638    /// positive values whose magnitude must be exactly 1e6 ns per ms.
1639    ///
1640    /// The 1 ms case is the specific regression guard: an inlined `* 1000`
1641    /// yields 1_000 here instead of 1_000_000, i.e. a deadline 1000x too
1642    /// short. Asserting the exact value (not merely "nonzero" or "greater
1643    /// than") is what makes that failure visible.
1644    #[test]
1645    fn millis_to_nanos_conversion() {
1646        // Not deadlines: -1 is "infinite", 0 is "return immediately".
1647        assert_eq!(positive_millis_as_nanos(-1), None);
1648        assert_eq!(positive_millis_as_nanos(i32::MIN), None);
1649        assert_eq!(positive_millis_as_nanos(0), None);
1650
1651        // Positive timeouts: exactly 1e6 nanoseconds per millisecond.
1652        assert_eq!(positive_millis_as_nanos(1), Some(1_000_000));
1653        assert_eq!(positive_millis_as_nanos(1_000), Some(1_000_000_000));
1654        assert_eq!(
1655            positive_millis_as_nanos(i32::MAX),
1656            Some(i32::MAX as u128 * 1_000_000)
1657        );
1658
1659        // The scale itself, stated independently of any single case so a
1660        // future refactor cannot satisfy the above by coincidence.
1661        for millis in [1, 2, 7, 250, 1_000, 86_400_000] {
1662            assert_eq!(
1663                positive_millis_as_nanos(millis),
1664                Some(millis as u128 * 1_000_000),
1665                "1 ms must convert to 1_000_000 ns, not 1_000"
1666            );
1667        }
1668    }
1669
1670    #[test]
1671    fn connect_nonblocking_results() {
1672        let call = reverie::syscalls::Connect::new();
1673        assert!(call.syscall_would_have_blocked(Err(Errno::EINPROGRESS)));
1674        assert!(call.syscall_would_have_blocked(Err(Errno::EALREADY)));
1675        assert_eq!(
1676            call.normalize_nonblocking_result(Err(Errno::EISCONN), true),
1677            Ok(0)
1678        );
1679        assert_eq!(
1680            call.normalize_nonblocking_result(Err(Errno::EISCONN), false),
1681            Err(Errno::EISCONN)
1682        );
1683    }
1684
1685    #[test]
1686    fn signal_interruption_errno_matches_linux_restart_policy() {
1687        assert_eq!(
1688            reverie::syscalls::Poll::new().signal_interrupt_errno(),
1689            Errno::EINTR
1690        );
1691        assert_eq!(
1692            reverie::syscalls::Ppoll::new().signal_interrupt_errno(),
1693            Errno::EINTR
1694        );
1695        assert_eq!(
1696            reverie::syscalls::EpollWait::new().signal_interrupt_errno(),
1697            Errno::EINTR
1698        );
1699        let sigtimedwait = reverie::syscalls::RtSigtimedwait::new();
1700        assert_eq!(sigtimedwait.signal_interrupt_errno(), Errno::EINTR);
1701        assert!(sigtimedwait.syscall_would_have_blocked(Err(Errno::EAGAIN)));
1702        assert_eq!(sigtimedwait.timeout_return_val(), Err(Errno::EAGAIN));
1703        assert_eq!(
1704            reverie::syscalls::Read::new().signal_interrupt_errno(),
1705            Errno::ERESTARTSYS
1706        );
1707        // A zero-progress writev interruption returns this internal errno on
1708        // ptrace so Linux applies the handler's SA_RESTART policy.
1709        assert_eq!(
1710            reverie::syscalls::Writev::new().signal_interrupt_errno(),
1711            Errno::ERESTARTSYS
1712        );
1713        assert_eq!(
1714            reverie::syscalls::Futex::new().signal_interrupt_errno(),
1715            Errno::ERESTARTSYS
1716        );
1717    }
1718
1719    #[test]
1720    fn writev_signal_result_uses_disposition_and_progress() {
1721        let call = reverie::syscalls::Writev::new();
1722        for disposition in [
1723            WaitSignalDisposition::Interrupt,
1724            WaitSignalDisposition::Restart,
1725        ] {
1726            assert!(matches!(
1727                interrupted_write_result(&call, 0, Some(disposition)),
1728                Some(Err(Error::Errno(Errno::ERESTARTSYS)))
1729            ));
1730            assert!(matches!(
1731                interrupted_write_result(&call, 17, Some(disposition)),
1732                Some(Ok(17))
1733            ));
1734        }
1735        assert!(interrupted_write_result(&call, 0, None).is_none());
1736        assert!(interrupted_write_result(&call, 17, None).is_none());
1737    }
1738
1739    #[test]
1740    fn pipe_writev_requests_signal_disposition() {
1741        let dettid = DetTid::from_raw(42);
1742        let call = reverie::syscalls::Writev::new();
1743        let request = pipe_writev_resources(dettid, call);
1744
1745        assert_eq!(
1746            request.signal_interrupt_errno(),
1747            Some(Errno::ERESTARTSYS.into_raw())
1748        );
1749        assert_eq!(request.resources.len(), 1);
1750    }
1751}