reverie/tool.rs
1/*
2 * Copyright (c) Meta Platforms, Inc. and affiliates.
3 * All rights reserved.
4 *
5 * This source code is licensed under the BSD-style license found in the
6 * LICENSE file in the root directory of this source tree.
7 */
8
9//! The API that a Reverie tool (client) should implement.
10//!
11//! Reverie tools consist of two portions: the global and local (per-guest
12//! thread) instrumentation, though in some backends these will execute in the
13//! same process.
14
15use async_trait::async_trait;
16use reverie_syscalls::Syscall;
17use serde::Serialize;
18use serde::de::DeserializeOwned;
19
20use crate::ExitStatus;
21use crate::Pid;
22use crate::Signal;
23use crate::SignalEvent;
24use crate::Subscription;
25use crate::Tid;
26use crate::error::Errno;
27use crate::error::Error;
28use crate::guest::Guest;
29#[cfg(target_arch = "x86_64")]
30use crate::rdtsc::Rdtsc;
31#[cfg(target_arch = "x86_64")]
32use crate::rdtsc::RdtscResult;
33
34/// Who owns a guest thread: the single axis that governs *both* how the thread
35/// executes *and* who owns its thread-synchronization primitives (`futex`,
36/// `CLONE_CHILD_CLEARTID`).
37///
38/// These two concerns must never disagree. If a thread executes under the Tool
39/// (registered in the Tool's scheduler) while its `futex` is serviced by the
40/// host — or vice versa — a `pthread_join` deadlocks: the joiner's `FUTEX_WAIT`
41/// waits in one domain while the exiting thread's `CLEARTID` wake fires in the
42/// other, so the wake never reaches the waiter. Collapsing both concerns onto
43/// this single enum makes that split-brain state *unrepresentable*: a thread is
44/// either wholly `Tool`-owned or wholly `Host`-owned, never half of each.
45///
46/// A backend that runs child threads (e.g. the KVM backend, where each guest
47/// thread runs on its own vCPU) selects the ownership for a tool's threads from
48/// [`Tool::thread_ownership`] and uses the *same* value to decide (a) whether to
49/// drive the thread through the Tool loop and (b) whether its `futex` routes to
50/// the Tool. Backends that do not run child threads (e.g. ptrace, which already
51/// routes every subscribed `futex` to the Tool) may ignore this value.
52#[derive(
53 Debug,
54 Clone,
55 Copy,
56 PartialEq,
57 Eq,
58 Default,
59 Serialize,
60 serde::Deserialize
61)]
62pub enum ThreadOwnership {
63 /// The Tool owns the thread: it is driven through the Tool loop (so the Tool
64 /// observes every one of its syscalls and schedules it) and its `futex` /
65 /// `CLEARTID` synchronization is serviced by the Tool. This is the safe,
66 /// "follow children" default — it matches the golden ptrace backend, where
67 /// the Tool (Detcore) owns every futex and no thread synchronization touches
68 /// the host. Determinism can only be guaranteed for `Tool`-owned threads.
69 #[default]
70 Tool,
71 /// The host owns the thread: it runs uninstrumented on the backend's direct
72 /// execution personality and its `futex` / `CLEARTID` synchronization uses
73 /// real host futex words.
74 ///
75 /// This is the `unmonitored_` opt-out. It is internally consistent (host
76 /// execution + host futex, so it does not by itself deadlock a join), but it
77 /// is a determinism/coverage hazard, **not** a correctness shortcut:
78 ///
79 /// * The Tool never sees the thread's syscalls, so it cannot sanitize,
80 /// record, or schedule them — determinism is **not** guaranteed for it or
81 /// for anything ordered against it.
82 /// * Mixing `Host`-owned threads into a tool that expects to schedule the
83 /// whole thread group (e.g. Detcore) breaks that tool's model.
84 ///
85 /// It is deliberately not named `unsafe_`: it cannot cause undefined
86 /// behavior, only nondeterminism and missed instrumentation.
87 Host,
88}
89
90impl ThreadOwnership {
91 /// Whether a thread with this ownership executes under the Tool loop (as
92 /// opposed to the backend's direct host execution personality).
93 pub fn executes_on_tool(self) -> bool {
94 matches!(self, ThreadOwnership::Tool)
95 }
96
97 /// Whether a thread with this ownership executes on the backend's direct
98 /// host execution personality (as opposed to the Tool loop). The exact
99 /// inverse of [`Self::executes_on_tool`].
100 pub fn executes_on_host(self) -> bool {
101 matches!(self, ThreadOwnership::Host)
102 }
103
104 /// Whether the thread's `futex` / `CLEARTID` synchronization is serviced by
105 /// the host rather than the Tool. This is the exact inverse of
106 /// [`Self::executes_on_tool`]; the two are derived from the same value so
107 /// execution and synchronization ownership can never disagree.
108 pub fn futex_is_host_owned(self) -> bool {
109 matches!(self, ThreadOwnership::Host)
110 }
111}
112
113/// The global half of a complete Reverie tool.
114///
115/// One global instance of this type will exist at runtime (singleton). This
116/// global state is shared by the tool across the whole process tree being
117/// instrumented.
118#[async_trait]
119pub trait GlobalTool: Send + Sync + Default {
120 /// The message to send to the global tool.
121 type Request: Serialize + DeserializeOwned + Send;
122
123 /// The result of sending the message.
124 type Response: Serialize + DeserializeOwned + Send;
125
126 /// Static, read-only configuration data that is available everywhere the
127 /// tool runs code.
128 type Config: Serialize + DeserializeOwned + Send + Sync + Clone + Default;
129
130 /// Initialize the tool, allocating the global state.
131 async fn init_global_state(_cfg: &Self::Config) -> Self {
132 Default::default()
133 }
134
135 /// Install one run-scoped backend capability before the first guest hook.
136 /// The default preserves backend-owned selection. A Tool that requires
137 /// controlled process signals must reject missing capabilities here.
138 fn install_backend_signal_control(
139 &self,
140 _control: Option<crate::BackendSignalControl>,
141 ) -> Result<crate::BackendSignalControlMode, Error> {
142 Ok(crate::BackendSignalControlMode::Unchanged)
143 }
144
145 /// Authorize the current real user-return boundary. The backend calls
146 /// this outside signal locks; host callback arrival is not authorization.
147 fn authorize_backend_signal_boundary(
148 &self,
149 _task: crate::SignalTaskIdentity,
150 ) -> Result<Option<crate::SignalDeliveryPermit>, Error> {
151 Ok(None)
152 }
153
154 /// Consume a real signal boundary before user entry. This hook must retain
155 /// ownership across cancellation; it is not an ordinary grant or syscall.
156 async fn on_backend_signal_boundary(
157 &self,
158 _receipt: crate::SignalBoundaryReceipt,
159 ) -> Result<(), Error> {
160 Ok(())
161 }
162
163 /// Reports final process cleanup after the leader has joined every guest
164 /// thread and released their descriptor references, including its own.
165 /// Only a backend offering `BackendSignalControl` sends this callback. A
166 /// Tool may retain a fence awaiting it only with that capability installed
167 /// in `ToolControlled` mode; a backend offering that mode must complete the
168 /// callback after successful process retirement. A terminal boundary receipt
169 /// can establish this fence and must remain able to cancel peer RPCs.
170 /// The fence then keeps other processes from observing host-timed EOF.
171 ///
172 /// There need not be a preceding terminal boundary receipt: synchronous
173 /// hardware faults can terminate a process without a delivery permit.
174 /// Consumers must distinguish an exact known process with no outstanding
175 /// controlled boundary from an early callback for a still-pending boundary.
176 /// This notification alone does not order an otherwise unfenced exit.
177 ///
178 /// KVM emits this once from a successfully retired process leader, before
179 /// its consuming hooks or joins of independent child processes. It is not
180 /// a child-wait publication or an ordinary scheduling request. The callback
181 /// must settle its exact process generation synchronously and must not wait
182 /// for guest progress. A backend failure ends the run instead of emitting
183 /// a successful retirement notification.
184 fn on_backend_process_retired(&self, _event: BackendProcessRetirement) -> Result<(), Error> {
185 Ok(())
186 }
187
188 /// Receive a (potentially) inter-process upcall on the global state object.
189 /// This intended to be IPC, inter-process communication, in some backends,
190 /// and a local method call in others, but never truly a communication
191 /// between different machines.
192 ///
193 /// It receives a shared reference to the global state object, which must
194 /// manage its own synchronization.
195 ///
196 /// On a fatal KVM or ordinary-ptrace run failure, an in-flight Tool callback and its inline
197 /// RPC future may be dropped at any await point. RPC implementations must
198 /// leave shared state safe for concurrent consuming cleanup when dropped;
199 /// a normal response is not manufactured to complete the abandoned RPC.
200 /// The terminal transition in `report_backend_failure` must also make
201 /// cleanup possible for requests that were admitted but did not complete.
202 async fn receive_rpc(&self, _from: Tid, _message: Self::Request) -> Self::Response;
203
204 /// Reports a fatal backend failure before cleanup can wait on another Tool
205 /// callback. This is a failed run, not a guest exit, signal, or RPC reply.
206 /// Implementations must finish their terminal transition synchronously,
207 /// including making concurrent consuming cleanup safe, before returning.
208 /// Complete that transition before waking any failure subscriber. Several
209 /// workers may report distinct errors in one run, so this method must be
210 /// idempotent and preserve the first terminal cause. It must not wait for
211 /// Tool callbacks or physical worker joins that depend on that transition.
212 fn report_backend_failure(&self, _event: BackendFailure) {}
213
214 /// Waits until this run cannot continue faithfully. Each call must subscribe
215 /// independently: multiple Tool callbacks and the scheduler may be waiting.
216 /// The default preserves Tools that do not own a scheduler.
217 /// Returning allows the backend to drop in-flight callbacks, including
218 /// `receive_rpc`, and proceed to consuming exit hooks. Shared state must
219 /// already support that cleanup; returning is not an ordinary RPC reply.
220 /// The ptrace backend polls the returned future again only after the waker
221 /// it was last polled with fires, so a pending future must arrange that
222 /// wake rather than rely on being polled again for another reason.
223 async fn wait_for_backend_failure(&self) {
224 std::future::pending::<()>().await
225 }
226
227 /// Reports that a backend observed a child transition and committed its
228 /// waitability for the parent process.
229 ///
230 /// Backends that model child lifecycle outside the host kernel should
231 /// invoke this at that boundary rather than inferring state from signal
232 /// delivery. Backends whose host kernel owns child waitability may retain
233 /// the default no-op.
234 ///
235 /// KVM polls this callback through its first suspension before making the
236 /// status visible to a parent wait. A Tool-controlled signal scheduler must
237 /// commit its publication or suppression decision in that synchronous
238 /// prefix: it must neither reach an `.await` nor otherwise block on parent
239 /// progress. The backend may already be holding a concurrent parent wait
240 /// across the whole prefix, so waiting for that parent would deadlock.
241 /// Work after that admission point may await parent progress; the backend
242 /// retains and finishes the same pinned future after publishing waitability.
243 /// If that synchronous prefix makes a concurrent parent runnable, the
244 /// backend fences its wait until publication completes; the parent cannot
245 /// observe the callback decision while still receiving a no-child-ready
246 /// result.
247 ///
248 /// No callback is emitted when the exact parent generation is already
249 /// terminal. Such a child is run-teardown state rather than a new waitable
250 /// transition, and the backend auto-reaps its status. A terminal transitive
251 /// ancestor does not suppress a child event while the direct parent remains
252 /// logically live; that parent retains its exact wait semantics.
253 /// For a live parent, callback admission only controls when waitability is
254 /// exposed. It does not reap the backend status: a Tool-controlled wait must
255 /// still be injected into the Guest before Tool shadow state is consumed.
256 async fn on_backend_child_wait_event(
257 &self,
258 _event: BackendChildWaitEvent,
259 ) -> Result<(), Error> {
260 Ok(())
261 }
262}
263
264/// Final descriptor cleanup and worker joins for one exact process lifetime.
265#[derive(Clone, Copy, Debug, Eq, PartialEq)]
266pub struct BackendProcessRetirement {
267 /// Process identity retained after its last live task has retired.
268 pub process: crate::SignalProcessId,
269 /// Authoritative process status after all guest threads have exited.
270 pub status: ExitStatus,
271}
272
273/// The location of a fatal backend failure. The backend retains its typed cause;
274/// this notification only ends dependent waits and must not invent guest status.
275/// Host-side ordinary-ptrace capture failures use the run root's PID/TID with
276/// a `ptrace stdout capture` or `ptrace stderr capture` phase. Those locations
277/// identify the host run owner, not an inferred guest writer or guest failure.
278#[derive(Clone, Copy, Debug, Eq, PartialEq)]
279pub struct BackendFailure {
280 /// Guest process owning the failed operation.
281 pub pid: Pid,
282 /// Guest thread owning the failed operation.
283 pub tid: Tid,
284 /// Backend operation that failed.
285 pub phase: &'static str,
286}
287
288/// A child state and waitability decision observed by an execution backend.
289#[derive(Clone, Copy, Debug, Eq, PartialEq)]
290pub enum BackendChildWaitState {
291 /// The child terminated with this exit status.
292 Exited {
293 /// The terminal status reported by the backend.
294 status: ExitStatus,
295 /// Whether the parent may consume this status with a wait syscall.
296 /// Explicit `SIGCHLD` ignore and `SA_NOCLDWAIT` make this false.
297 waitable: bool,
298 /// Virtual child uid reported through `siginfo_t`.
299 uid: u32,
300 /// Child user CPU time in signed Linux clock ticks.
301 user_ticks: i64,
302 /// Child system CPU time in signed Linux clock ticks.
303 system_ticks: i64,
304 },
305 /// The child entered a job-control stop for this signal number.
306 Stopped(i32),
307 /// A previously stopped child resumed.
308 Continued,
309}
310
311/// A backend-observed child waitability decision.
312#[derive(Clone, Copy, Debug, Eq, PartialEq)]
313pub struct BackendChildWaitEvent {
314 /// Exact process lifetime whose wait syscalls may observe the transition.
315 pub parent: crate::SignalProcessId,
316 /// Exact child process lifetime that changed state.
317 pub child: crate::SignalProcessId,
318 /// The observed child state and whether it remains waitable.
319 pub state: BackendChildWaitState,
320}
321
322impl BackendChildWaitEvent {
323 /// Returns the complete terminal publication payload when the receiving
324 /// parent remains inside the traced process tree.
325 pub fn child_exit_completion(self) -> Option<crate::ChildExitCompletion> {
326 let BackendChildWaitState::Exited {
327 status,
328 waitable,
329 uid,
330 user_ticks,
331 system_ticks,
332 } = self.state
333 else {
334 return None;
335 };
336 Some(crate::ChildExitCompletion {
337 parent: self.parent,
338 child: self.child,
339 status,
340 waitable,
341 uid,
342 user_ticks,
343 system_ticks,
344 })
345 }
346}
347
348#[async_trait]
349impl GlobalTool for () {
350 type Request = ();
351 type Response = ();
352 type Config = ();
353
354 async fn receive_rpc(&self, _from: Tid, _message: ()) {}
355}
356
357/// A trait that every Reverie *tool* must implement. The primary function of the
358/// tool specifies how syscalls and signals are handled.
359///
360/// The type that a `Tool` is implemented for represents the process-level state.
361/// That is, one runtime instance of this type will be created for each guest
362/// process. This type is in turn a factory for *thread level states*, which are
363/// allocated dynamically upon guest thread creation. Instances of the thread
364/// state are also managed by Reverie.
365///
366/// During fatal KVM run cleanup, asynchronous event callbacks may be dropped
367/// at any await point. Their thread state is then passed to `on_exit_thread`
368/// even if the start callback was never entered or did not finish. Exit hooks
369/// must consume partially initialized state without requiring guest execution
370/// or a normal response from an abandoned callback. This contract covers
371/// returned runtime errors; arbitrary panic unwinding is not guaranteed to
372/// invoke consuming hooks.
373///
374/// The ordinary ptrace backend owns execution-control state and wait statuses.
375/// Tools must use Guest/backend APIs for resumes, stepping, detach/attach,
376/// tracing options, wait/reap, and mutations of registers or signal information
377/// (including PTRACE_SETSIGINFO). Raw operations outside those APIs invalidate
378/// its current-stop ownership contract. Read-only ptrace/memory observations
379/// are permitted; supported Guest injection can replace the current stop.
380///
381/// For ordinary non-syscall Errno-only callbacks, the return type erases causal
382/// provenance. If a same-generation observation justifies yielding to the
383/// original lifecycle owner, the callback's errno is retained as a diagnostic
384/// while that owner supplies actual exit or exec status. A live callback error
385/// remains fatal; an actually received Error::Tool or Error::Io is always fatal.
386/// This is cancellation/death precedence, not attribution of an errno to a
387/// memory access. The ptrace completion API exposes these records; successful
388/// legacy waits project them away. No host timeout is used to choose death.
389///
390/// # Example
391///
392/// Here is an example of a tool that simply counts the number of syscalls
393/// intercepted for each thread:
394/// ```
395/// use reverie::syscalls::*;
396/// use reverie::*;
397///
398/// /// Our process-level state.
399/// #[derive(Debug, Default, Clone)]
400/// struct MyTool;
401///
402/// #[reverie::tool]
403/// impl Tool for MyTool {
404/// /// The global state type.
405/// type GlobalState = ();
406/// /// Count of syscalls.
407/// type ThreadState = u64;
408///
409/// async fn handle_syscall_event<T: Guest<Self>>(
410/// &self,
411/// guest: &mut T,
412/// syscall: Syscall,
413/// ) -> Result<i64, Error> {
414/// *guest.thread_state_mut() += 1;
415///
416/// // Inject the syscall. If we don't do this, the syscall will be
417/// // supressed.
418/// let ret = guest.inject(syscall).await?;
419///
420/// Ok(ret)
421/// }
422/// }
423/// ```
424#[async_trait]
425pub trait Tool: Send + Sync + Default {
426 /// The type of the global half that goes along with this Local tool. By
427 /// including this type, the Tool is actually a complete specification for an
428 /// instrumentation tool.
429 type GlobalState: GlobalTool;
430
431 /// Tool-state specific to each guest thread. If unset, this defaults to the
432 /// unit type `()`, indicating that the tool does not have thread-level
433 /// state.
434 ///
435 /// Both thread-local and process-local state may have to be migrated between
436 /// address spaces by a Reverie backend. Hence the `ThreadState` type must
437 /// implement [`Serialize`] and [`DeserializeOwned`].
438 ///
439 /// The thread-local storage must be in a good, consistent state when each
440 /// handler returns, and also when handlers yield.
441 ///
442 /// [`Serialize`]: serde::Serialize
443 /// [`DeserializeOwned`]: serde::de::DeserializeOwned
444 type ThreadState: Serialize + DeserializeOwned + Default + Send + Sync;
445
446 /// A common constructor that initializes state when a process is created,
447 /// including the guest's initial, root process. Of course, every process
448 /// includes at least one thread, but the process level state is allocated
449 /// before thread level-state for the process's main thread is allocated.
450 ///
451 /// For now this method assumes access to the global state, but that may
452 /// change.
453 fn new(_pid: Pid, _cfg: &<Self::GlobalState as GlobalTool>::Config) -> Self {
454 Default::default()
455 }
456
457 /// Events the tool subscribes to. This is only called *once* for the entire
458 /// tree. By default, all syscalls are traced (but CPUID/RDTSC instructions
459 /// are not).
460 fn subscriptions(_cfg: &<Self::GlobalState as GlobalTool>::Config) -> Subscription {
461 Subscription::all_syscalls()
462 }
463
464 /// How this tool's guest threads (children created via `CLONE_THREAD`) are
465 /// owned. See [`ThreadOwnership`]. Called once per tree, like
466 /// [`Tool::subscriptions`].
467 ///
468 /// The default is [`ThreadOwnership::Tool`] — the tool follows its children:
469 /// every guest thread is driven through the Tool loop and its `futex`
470 /// synchronization is serviced by the Tool. This is what a determinizing
471 /// tool such as Detcore needs (it owns every futex and schedules every
472 /// thread, matching the golden ptrace backend), and it is the safe default
473 /// for any tool. Override it to return [`ThreadOwnership::Host`] only to opt
474 /// a tool's child threads *out* of instrumentation, accepting the
475 /// determinism/coverage hazard documented on that variant.
476 ///
477 /// Only backends that themselves run child threads (e.g. KVM) consult this;
478 /// backends like ptrace that already route every subscribed `futex` to the
479 /// Tool ignore it.
480 fn thread_ownership(_cfg: &<Self::GlobalState as GlobalTool>::Config) -> ThreadOwnership {
481 ThreadOwnership::Tool
482 }
483
484 /// A guest process creates additional threads, which need their tool state
485 /// initialized. This method returns a newly-allocated thread state. This
486 /// method necessarily runs before the first instruction of a newly created
487 /// guest thread.
488 ///
489 /// If the parent thread is running a handler which injects a fork, this
490 /// callback executes on behalf of the child and may observe the parent's
491 /// thread-local state just this one time. It is important to know WHEN that
492 /// view into the parent's thread-state occurs. We currently guarantee that
493 /// this is *immediately* upon the `.inject()` call that creates the child
494 /// thread. Any later point of execution for `init_thread_state` could delay
495 /// the creation of the child arbitrarily long, waiting for the parent to
496 /// relinquish its hold on its own thread-local state.
497 ///
498 /// The parent Tid always refers to the thread-ID that called
499 /// fork/clone/vfork in order to create the new guest thread. Access to the
500 /// parent's state allows the child state to be defined in terms of modifying
501 /// the parent's, such as tracking the depth in a tree of threads.
502 ///
503 /// # Arguments
504 ///
505 /// * `&self`: a handle on the process-level state.
506 /// * `child`: the new child thread's ID.
507 /// * `parent`: A tuple of the parent thread ID and a snapshot of the
508 /// parent's thread-local state. This is `None` if the current thread is the
509 /// root of the guest process tree.
510 fn init_thread_state(
511 &self,
512 _child: Tid,
513 _parent: Option<(Tid, &Self::ThreadState)>,
514 ) -> Self::ThreadState {
515 Default::default()
516 }
517
518 /// Similar to `handle_syscall_event`, except this traps the first
519 /// instruction executed by a new thread. Typical uses of this method include
520 /// delaying thread execution or running initialization actions (injections
521 /// or rpcs).
522 ///
523 /// `init_thread_state` runs once for every constructed thread state. This
524 /// callback runs once when that thread is allowed to start; cancellation
525 /// before admission can consume the state without entering this callback.
526 /// Fatal run failure may also drop it before completion. This callback is
527 /// guaranteed to run independently from the parent. It does not view the
528 /// parents state, and this handler runs in its own asynchronous task.
529 /// Blocking this task on an `.await` will not interfere with the progress of
530 /// the parent thread.
531 ///
532 /// # Arguments
533 ///
534 /// * `&self`: The process-level state for this thread.
535 /// * `guest`: A handle to the guest thread.
536 async fn handle_thread_start<T: Guest<Self>>(&self, _guest: &mut T) -> Result<(), Error> {
537 Ok(())
538 }
539
540 /// Called upon a *successful* execve. In `handle_syscall_event`, after
541 /// injecting `execve`, it is not possible to run code after a successful
542 /// `execve` because it never returns.
543 ///
544 /// NOTE: Thread and process state are unchanged across this execve boundary.
545 /// Thus, this can be useful for doing something like counting the number of
546 /// times a process successfully calls `execve`.
547 async fn handle_post_exec<T: Guest<Self>>(&self, _guest: &mut T) -> Result<(), Errno> {
548 Ok(())
549 }
550
551 /// The tool receives an event from the guest, via the Reverie program
552 /// instrumentation. A Reverie syscall handler fires in the moment *before* a
553 /// guest syscall executes (like a "prehook").
554 ///
555 /// After the event is trapped, control transfers to `handle_syscall_event`
556 /// which is put in temporary control of the guest thread. Via `guest`, we
557 /// can directly access the thread/process local state, and we can also
558 /// remotely access (1) the global state and (2) the memory/registers of the
559 /// guest thread it controls.
560 ///
561 /// NOTE: Only syscalls we have subscribed to [`Tool::subscriptions`] will
562 /// have this handler invoked.
563 async fn handle_syscall_event<T: Guest<Self>>(
564 &self,
565 guest: &mut T,
566 c: Syscall,
567 ) -> Result<i64, Error> {
568 guest.tail_inject(c).await
569 }
570
571 /// CPUID is trapped, the tool should implement this function to return
572 /// `[eax, ebx, ecx, edx]`.
573 ///
574 /// NOTE:
575 /// * This is never called by default unless cpuid events are subscribed
576 /// to.
577 /// * This is only available on x86_64.
578 #[cfg(target_arch = "x86_64")]
579 async fn handle_cpuid_event<T: Guest<Self>>(
580 &self,
581 _guest: &mut T,
582 eax: u32,
583 ecx: u32,
584 ) -> Result<raw_cpuid::CpuIdResult, Errno> {
585 Ok(raw_cpuid::cpuid!(eax, ecx))
586 }
587
588 /// rdtsc/rdtscp is trapped, the tool should implement this function to
589 /// return the counter.
590 ///
591 /// NOTE:
592 /// * This is never called by default unless rdtsc events are subscribed
593 /// to.
594 /// * This is only available on x86_64.
595 #[cfg(target_arch = "x86_64")]
596 async fn handle_rdtsc_event<T: Guest<Self>>(
597 &self,
598 _guest: &mut T,
599 request: Rdtsc,
600 ) -> Result<RdtscResult, Errno> {
601 Ok(RdtscResult::new(request))
602 }
603
604 /// Handles a guest's signal before it is delivered to guest.
605 ///
606 /// # Return value
607 /// - `Some(sig)`: The signal `sig` will be delivered to guest.
608 /// - `None`: The signal is supressed and never delivered to the guest.
609 async fn handle_signal_event<T: Guest<Self>>(
610 &self,
611 _guest: &mut T,
612 signal: Signal,
613 ) -> Result<Option<Signal>, Errno> {
614 Ok(Some(signal))
615 }
616
617 /// Enables acknowledgment of real KVM pending removals before thread start.
618 /// Other Tools retain their existing pending-state behavior by default.
619 /// The static-ELF runner requires effective [`ThreadOwnership::Tool`],
620 /// including caller overrides, and rejects an incompatible Host choice
621 /// before initializing GlobalState or consuming/executing the installed ELF.
622 fn observe_signal_dequeues(_config: &<Self::GlobalState as GlobalTool>::Config) -> bool {
623 false
624 }
625
626 /// Acknowledges one irreversible pending removal before any later Tool/guest work.
627 /// The backend retains the journal entry until this returns success. An error
628 /// is terminal; it is never a rollback or an ordinary guest syscall errno.
629 /// Notifications are process-wide FIFO, but each runs on its removing Guest.
630 /// Another owner can wait here before posting its next Tool scheduler request.
631 /// An opted-in Tool must complete this acknowledgment without requiring that
632 /// waiting owner to make progress or relinquish its scheduler token. FIFO
633 /// sequencing alone does not establish deterministic event membership.
634 async fn handle_signal_dequeue<G: Guest<Self>>(
635 &self,
636 _guest: &mut G,
637 _dequeue: crate::SignalDequeue,
638 ) -> Result<(), Errno> {
639 Ok(())
640 }
641
642 /// Handles a structured guest signal immediately before a virtual backend
643 /// delivers it.
644 ///
645 /// The default is a compatibility bridge to [`Self::handle_signal_event`]
646 /// for signal numbers represented by [`Signal`]. It preserves the complete
647 /// event when the legacy hook keeps it and preserves suppression when that
648 /// hook drops it. Legacy signal-number replacement and raw real-time signal
649 /// numbers are rejected with `ENOSYS`, because neither path can produce
650 /// coherent replacement `siginfo_t`; a structured override must do that
651 /// rather than silently losing information in the legacy enum.
652 async fn handle_structured_signal_event<T: Guest<Self>>(
653 &self,
654 guest: &mut T,
655 event: SignalEvent,
656 ) -> Result<Option<SignalEvent>, Errno> {
657 let signal = Signal::try_from(event.signal()).map_err(|_| Errno::ENOSYS)?;
658 match self.handle_signal_event(guest, signal).await? {
659 None => Ok(None),
660 Some(replacement) if replacement as i32 == event.signal() => Ok(Some(event)),
661 // Linux's ptrace reinjection path clears the old siginfo and
662 // synthesizes SI_USER with the tracer parent's credentials when a
663 // tracer changes the signal number. A virtual backend has no
664 // faithful guest-visible identity for that host tracer. Refuse the
665 // lossy legacy replacement; a structured-hook override can return
666 // a new SignalEvent with coherent replacement metadata.
667 Some(_) => Err(Errno::ENOSYS),
668 }
669 }
670
671 /// Handles a timer event generated by a call to `Guest::set_timer`
672 async fn handle_timer_event<T: Guest<Self>>(&self, _guest: &mut T) {}
673
674 /// Called when a thread will exit shortly or has exited. That means there
675 /// will be no more intercepted events on this thread.
676 ///
677 /// Serves as a "destructor" for the thread state, and thus takes it by move.
678 /// KVM cleanup consumes each constructed state once on returned runtime
679 /// errors or cancellation, including states whose `handle_thread_start`
680 /// was never entered or completed. No further guest event is started to
681 /// perform this cleanup. Panic unwinding may bypass the hook.
682 async fn on_exit_thread<G: GlobalRPC<Self::GlobalState>>(
683 &self,
684 _tid: Tid,
685 _global_state: &G,
686 _thread_state: Self::ThreadState,
687 _exit_status: ExitStatus,
688 ) -> Result<(), Error> {
689 Ok(())
690 }
691
692 /// Called when a process will exit shortly or has exited. That means there
693 /// will be no more intercepted events on from any thread within this
694 /// process.
695 ///
696 /// Serves as a "destructor" for the process state (`self`), and thus takes
697 /// it by move.
698 /// On KVM this also consumes a constructed process cancelled before its
699 /// initial thread starts, after the thread states have been consumed. It
700 /// must support cleanup after fatal failure without ordinary guest RPC
701 /// progress. Panic unwinding may bypass the hook.
702 async fn on_exit_process<G: GlobalRPC<Self::GlobalState>>(
703 self,
704 _pid: Pid,
705 _global_state: &G,
706 _exit_status: ExitStatus,
707 ) -> Result<(), Error> {
708 Ok(())
709 }
710}
711
712/// A "noop" tool that doesn't do anything.
713impl Tool for () {
714 type GlobalState = ();
715 type ThreadState = ();
716
717 fn subscriptions(_cfg: &()) -> Subscription {
718 Subscription::none()
719 }
720}
721
722/// A handle to send messages to the global state (potentially a remote,
723/// inter-process communication).
724#[async_trait]
725pub trait GlobalRPC<G: GlobalTool>: Sync {
726 /// Send an RPC message to wherever the global state is stored, synchronously
727 /// blocks the current thread until a response is received.
728 async fn send_rpc(&self, message: G::Request) -> G::Response;
729
730 /// Return the read-only tool configuration
731 fn config(&self) -> &G::Config;
732}