henad_compute/gpu/mod.rs
1//! GPU engine machinery, the sibling of [`crate::cpu`], for models whose state lives in GPU
2//! buffers and never round-trips to the CPU.
3//!
4//! Nothing here ever *creates* a `wgpu::Device`. A host passes one in through a [`GpuContext`].
5//! A model contributes shaders, seed data and metadata, and every wgpu object is built here.
6
7pub mod agent_engine;
8pub mod capacity;
9mod contracts;
10pub mod fault;
11pub mod grid_engine;
12pub mod limits;
13pub mod primitives;
14pub mod sim_thread;
15#[cfg(not(target_arch = "wasm32"))]
16pub mod stepping;
17pub mod timing;
18pub mod view;
19
20#[cfg(test)]
21mod tests;
22
23pub use agent_engine::GpuAgentState;
24pub use capacity::Demand;
25pub use grid_engine::GpuGridState;
26pub use limits::GpuNeeds;
27pub use primitives::readback::StatsPoll;
28pub use primitives::spatial_hash::{GpuSpatialHash, HashGrid};
29pub use sim_thread::{GpuSimState, GpuStats};
30pub use view::agents::GpuAgents;
31pub use view::display::{DisplayTarget, GpuDisplay};
32/// The wgpu release Henad builds on. Its types sit in [`GpuContext`]'s fields and in device requests, and a caller
33/// refers to them through this path instead of its own `wgpu` dependency.
34pub use wgpu;
35
36#[cfg(test)]
37use tests::support::headless_context;
38
39pub use sim_thread::GpuSimThread;
40
41use std::sync::Arc;
42use std::sync::atomic::{AtomicBool, Ordering};
43
44use crate::fault::{Fault, FaultSink};
45use crate::runtime_info::RuntimeInfo;
46
47/// Largest number of steps one command buffer holds.
48///
49/// Enough passes in one submission trip the OS GPU watchdog, with no error, and every later readback reads zero.
50pub const MAX_STEPS_PER_SUBMISSION: u32 = 64;
51
52/// GPU handles that a host passes to the engines, with the sink for errors no scope catches.
53///
54/// A clone is cheap, and shares the fault sink and the record of a lost device.
55#[derive(Debug, Clone)]
56pub struct GpuContext {
57 /// Device every model builds on.
58 pub device: wgpu::Device,
59 /// Queue the engine submits its work to.
60 pub queue: wgpu::Queue,
61 /// Colour format of the target a display pipeline renders into.
62 ///
63 /// A model builds its display render pipeline once, at construction, and a pipeline is tied to its colour target
64 /// format.
65 pub target_format: wgpu::TextureFormat,
66 /// Sink for errors that nothing else caught. See [`GpuContext::new`].
67 pub faults: FaultSink,
68 /// Set once the device is lost.
69 lost: Arc<AtomicBool>,
70 /// Facts about the host and the adapter, when whoever acquired the device attached them.
71 runtime_info: Option<Arc<RuntimeInfo>>,
72}
73
74impl GpuContext {
75 /// Creates a context over `device` and `queue`, and takes over the device's error handling. Left to wgpu, every
76 /// error is fatal.
77 ///
78 /// Errors raised inside a [`fault::catching_on`] go to that scope. Everything else, including
79 /// egui's own rendering on the same device, is stored in `faults` for the host to pick up. On the
80 /// web this is the only route. [`fault::catching_on`] pushes no scopes there.
81 ///
82 /// A `GPUInternalError` still ends the web build. wgpu converts an error with
83 /// `Error::from_js`. Anything other than a `GPUValidationError` or a `GPUOutOfMemoryError`
84 /// panics there. A model provokes those two, and the handler reports them normally.
85 ///
86 /// The context also records the loss of the device. [`Self::is_lost`] reports it.
87 ///
88 /// Note that a second `new` on the same device takes over the error handler from the first context, and on native
89 /// targets the lost callback too. From then on only the newest context receives the device's unscoped errors, and
90 /// on native targets its loss. In a browser every context on the device records the loss. A clone shares the fault
91 /// sink and the loss record with the context it came from.
92 ///
93 /// The context carries no [`RuntimeInfo`]. A host that passes its context to a sweep attaches one with
94 /// `.with_runtime_info(RuntimeInfo::collect(&adapter, &device))`. Without it the sweep's manifest records no
95 /// adapter.
96 pub fn new(
97 device: wgpu::Device,
98 queue: wgpu::Queue,
99 target_format: wgpu::TextureFormat,
100 faults: FaultSink,
101 ) -> Self {
102 let sink = faults.clone();
103 // wgpu requires `Send + Sync` here even where nothing can be sent. Under atomics a
104 // `wgpu::Error` is neither, and the sink holds one. `SendWrapper` panics the moment a
105 // second thread touches it. On the web that check is the whole guarantee.
106 #[cfg(all(target_arch = "wasm32", target_feature = "atomics"))]
107 let sink = send_wrapper::SendWrapper::new(sink);
108 device.on_uncaptured_error(std::sync::Arc::new(move |error: wgpu::Error| {
109 log::error!("unhandled GPU error: {error}");
110 sink.set_once(Fault::device("running on the GPU", error));
111 }));
112 let lost = Arc::new(AtomicBool::new(false));
113 let callback_lost = Arc::clone(&lost);
114 device.set_device_lost_callback(move |reason, message| {
115 log::error!("GPU device lost ({reason:?}): {message}");
116 callback_lost.store(true, Ordering::Release);
117 });
118 Self {
119 device,
120 queue,
121 target_format,
122 faults,
123 lost,
124 runtime_info: None,
125 }
126 }
127
128 /// Returns the context with `info` attached, as [`Self::runtime_info`] reads it back.
129 pub fn with_runtime_info(self, info: RuntimeInfo) -> Self {
130 Self {
131 runtime_info: Some(Arc::new(info)),
132 ..self
133 }
134 }
135
136 /// Facts about the host and the adapter, `None` when nothing attached them.
137 pub fn runtime_info(&self) -> Option<&RuntimeInfo> {
138 self.runtime_info.as_deref()
139 }
140
141 /// Returns whether the device is lost. A lost device runs no more work.
142 ///
143 /// Note that a destroyed device counts as lost once a poll finds its queue empty.
144 pub fn is_lost(&self) -> bool {
145 self.lost.load(Ordering::Acquire)
146 }
147}