Skip to main content

henad_compute/gpu/
mod.rs

1//! GPU engine machinery, the sibling of [`crate::cpu`], for models whose state lives in GPU
2//! buffers and never round-trips to the CPU.
3//!
4//! Nothing here ever *creates* a `wgpu::Device`. A host passes one in through a [`GpuContext`].
5//! A model contributes shaders, seed data and metadata, and every wgpu object is built here.
6
7pub mod agent_engine;
8pub mod capacity;
9mod contracts;
10pub mod fault;
11pub mod grid_engine;
12pub mod limits;
13pub mod primitives;
14pub mod sim_thread;
15#[cfg(not(target_arch = "wasm32"))]
16pub mod stepping;
17pub mod timing;
18pub mod view;
19
20#[cfg(test)]
21mod tests;
22
23pub use agent_engine::GpuAgentState;
24pub use capacity::Demand;
25pub use grid_engine::GpuGridState;
26pub use limits::GpuNeeds;
27pub use primitives::readback::StatsPoll;
28pub use primitives::spatial_hash::{GpuSpatialHash, HashGrid};
29pub use sim_thread::{GpuSimState, GpuStats};
30pub use view::agents::GpuAgents;
31pub use view::display::{DisplayTarget, GpuDisplay};
32/// The wgpu release Henad builds on. Its types sit in [`GpuContext`]'s fields and in device requests, and a caller
33/// refers to them through this path instead of its own `wgpu` dependency.
34pub use wgpu;
35
36#[cfg(test)]
37use tests::support::headless_context;
38
39pub use sim_thread::GpuSimThread;
40
41use std::sync::Arc;
42use std::sync::atomic::{AtomicBool, Ordering};
43
44use crate::fault::{Fault, FaultSink};
45use crate::runtime_info::RuntimeInfo;
46
47/// Largest number of steps one command buffer holds.
48///
49/// Enough passes in one submission trip the OS GPU watchdog, with no error, and every later readback reads zero.
50pub const MAX_STEPS_PER_SUBMISSION: u32 = 64;
51
52/// GPU handles that a host passes to the engines, with the sink for errors no scope catches.
53///
54/// A clone is cheap, and shares the fault sink and the record of a lost device.
55#[derive(Debug, Clone)]
56pub struct GpuContext {
57    /// Device every model builds on.
58    pub device: wgpu::Device,
59    /// Queue the engine submits its work to.
60    pub queue: wgpu::Queue,
61    /// Colour format of the target a display pipeline renders into.
62    ///
63    /// A model builds its display render pipeline once, at construction, and a pipeline is tied to its colour target
64    /// format.
65    pub target_format: wgpu::TextureFormat,
66    /// Sink for errors that nothing else caught. See [`GpuContext::new`].
67    pub faults: FaultSink,
68    /// Set once the device is lost.
69    lost: Arc<AtomicBool>,
70    /// Facts about the host and the adapter, when whoever acquired the device attached them.
71    runtime_info: Option<Arc<RuntimeInfo>>,
72}
73
74impl GpuContext {
75    /// Creates a context over `device` and `queue`, and takes over the device's error handling. Left to wgpu, every
76    /// error is fatal.
77    ///
78    /// Errors raised inside a [`fault::catching_on`] go to that scope. Everything else, including
79    /// egui's own rendering on the same device, is stored in `faults` for the host to pick up. On the
80    /// web this is the only route. [`fault::catching_on`] pushes no scopes there.
81    ///
82    /// A `GPUInternalError` still ends the web build. wgpu converts an error with
83    /// `Error::from_js`. Anything other than a `GPUValidationError` or a `GPUOutOfMemoryError`
84    /// panics there. A model provokes those two, and the handler reports them normally.
85    ///
86    /// The context also records the loss of the device. [`Self::is_lost`] reports it.
87    ///
88    /// Note that a second `new` on the same device takes over the error handler from the first context, and on native
89    /// targets the lost callback too. From then on only the newest context receives the device's unscoped errors, and
90    /// on native targets its loss. In a browser every context on the device records the loss. A clone shares the fault
91    /// sink and the loss record with the context it came from.
92    ///
93    /// The context carries no [`RuntimeInfo`]. A host that passes its context to a sweep attaches one with
94    /// `.with_runtime_info(RuntimeInfo::collect(&adapter, &device))`. Without it the sweep's manifest records no
95    /// adapter.
96    pub fn new(
97        device: wgpu::Device,
98        queue: wgpu::Queue,
99        target_format: wgpu::TextureFormat,
100        faults: FaultSink,
101    ) -> Self {
102        let sink = faults.clone();
103        // wgpu requires `Send + Sync` here even where nothing can be sent. Under atomics a
104        // `wgpu::Error` is neither, and the sink holds one. `SendWrapper` panics the moment a
105        // second thread touches it. On the web that check is the whole guarantee.
106        #[cfg(all(target_arch = "wasm32", target_feature = "atomics"))]
107        let sink = send_wrapper::SendWrapper::new(sink);
108        device.on_uncaptured_error(std::sync::Arc::new(move |error: wgpu::Error| {
109            log::error!("unhandled GPU error: {error}");
110            sink.set_once(Fault::device("running on the GPU", error));
111        }));
112        let lost = Arc::new(AtomicBool::new(false));
113        let callback_lost = Arc::clone(&lost);
114        device.set_device_lost_callback(move |reason, message| {
115            log::error!("GPU device lost ({reason:?}): {message}");
116            callback_lost.store(true, Ordering::Release);
117        });
118        Self {
119            device,
120            queue,
121            target_format,
122            faults,
123            lost,
124            runtime_info: None,
125        }
126    }
127
128    /// Returns the context with `info` attached, as [`Self::runtime_info`] reads it back.
129    pub fn with_runtime_info(self, info: RuntimeInfo) -> Self {
130        Self {
131            runtime_info: Some(Arc::new(info)),
132            ..self
133        }
134    }
135
136    /// Facts about the host and the adapter, `None` when nothing attached them.
137    pub fn runtime_info(&self) -> Option<&RuntimeInfo> {
138        self.runtime_info.as_deref()
139    }
140
141    /// Returns whether the device is lost. A lost device runs no more work.
142    ///
143    /// Note that a destroyed device counts as lost once a poll finds its queue empty.
144    pub fn is_lost(&self) -> bool {
145        self.lost.load(Ordering::Acquire)
146    }
147}