frust_gpu/arena.rs
1//! The per-frame bump arena every small GPU upload goes through:
2//! [`HostBuffer`], the [`BufferSlice`] it hands back, and the
3//! [`BufferUploader`] seam that keeps both host-testable.
4//!
5//! A frame produces many small pieces of data a shader has to read —
6//! per-draw uniforms, a handful of vertices, a transform block. Giving each
7//! one its own `wgpu::Buffer` means an allocation, a bind group and a
8//! separate upload per draw. The arena replaces that with one buffer: every
9//! [`HostBuffer::alloc`] appends to a CPU-side staging vector and returns
10//! the `(offset, size)` slice the caller will bind at, one
11//! [`HostBuffer::flush`] uploads the whole frame's bytes with a single
12//! `queue.write_buffer`, and [`HostBuffer::reset`] rewinds the bump pointer
13//! at the start of the next frame.
14//!
15//! # Alignment
16//!
17//! Every slice is aligned to at least the adapter's
18//! [`TierCaps::min_uniform_buffer_offset_alignment`], so any slice is
19//! legal as a uniform binding offset — including a dynamic one. That value
20//! is 256 under the GLES-3.0/WebGL2 profile and on the iOS Simulator (which
21//! misreports its own alignment; [`crate::context`] forces the device
22//! request back up to 256 there), which is the strictest target this
23//! workspace has, so an arena built against those caps satisfies every
24//! looser adapter as well. A caller may ask for *more* alignment for a
25//! specific slice — a vertex stride, say — and never gets less.
26//!
27//! # Grow-only, with a high-water mark
28//!
29//! The GPU buffer is created on the first flush that has bytes and then
30//! reused. It grows geometrically when a frame outgrows it and never
31//! shrinks, so the steady state after a few frames is zero buffer
32//! allocations per frame; [`HostBuffer::high_water_mark`] reports the
33//! largest frame seen, which is what a host tunes an initial size against.
34//!
35//! # Why there is no 4-deep ring
36//!
37//! A hand-rolled Vulkan/Metal arena double- or quadruple-buffers, because
38//! the CPU writes into persistently mapped memory the GPU may still be
39//! reading from an in-flight frame; the ring is what keeps this frame's
40//! writes off last frame's bytes.
41//!
42//! `wgpu::Queue::write_buffer` has no such hazard to guard. The bytes are
43//! copied out of the caller's slice into a queue-owned staging allocation at
44//! call time, and the actual device-side copy is recorded ahead of the
45//! command buffers submitted after it, in queue order — so a write issued
46//! for frame N lands after frame N-1's commands have already run, and
47//! `wgpu` tracks and reclaims the staging memory itself once the GPU is done
48//! with it. The caller never holds a pointer into memory the GPU is reading,
49//! which is the only thing a ring exists to arrange. Growing is safe for the
50//! same reason: `wgpu` refcounts a replaced buffer until every submission
51//! referencing it has retired.
52//!
53//! What the caller must still respect is ordering within its own frame:
54//! rewind with [`HostBuffer::reset`] at frame start, allocate, flush once
55//! before submitting the frame's commands. Rewinding a frame whose commands
56//! are already recorded but not yet flushed would overwrite the bytes those
57//! commands are going to read.
58
59use crate::caps::TierCaps;
60use crate::context::DeviceHandle;
61
62/// The usages every arena buffer is created with: the two binding kinds a
63/// frame's transient data is read as, plus the copy destination
64/// `write_buffer` needs.
65pub const ARENA_USAGE: wgpu::BufferUsages = wgpu::BufferUsages::UNIFORM
66 .union(wgpu::BufferUsages::VERTEX)
67 .union(wgpu::BufferUsages::COPY_DST);
68
69/// The smallest GPU buffer the arena ever creates. A first frame with one
70/// 64-byte uniform in it should not lead to a second allocation on the
71/// second frame.
72pub const MIN_ARENA_CAPACITY: u64 = 64 * 1024;
73
74/// Default debug label for the arena's GPU buffer.
75const DEFAULT_LABEL: &str = "frust-gpu host arena";
76
77/// Where a [`HostBuffer::alloc`] call's bytes ended up: a byte range within
78/// the arena's single GPU buffer.
79///
80/// Not `wgpu::BufferSlice` — this is a plain pair of numbers with no
81/// borrow of a buffer, so it can be stored in a display list, sorted, or
82/// handed to a bind-group builder long after the allocation that produced
83/// it.
84#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)]
85pub struct BufferSlice {
86 /// Byte offset into the arena buffer. Always a multiple of the
87 /// effective alignment, so it is legal as a uniform binding offset.
88 pub offset: u64,
89 /// Length in bytes of the data written, excluding any alignment padding
90 /// that precedes it.
91 pub size: u64,
92}
93
94/// Creates and writes the arena's GPU buffer.
95///
96/// Exists so [`HostBuffer`] takes an upload *capability* rather than a
97/// `wgpu::Device`/`wgpu::Queue` pair: a host test implements it with
98/// counters and asserts that a second frame reuses the same buffer and
99/// issues exactly one write, with no GPU. [`DeviceHandle`] is the only
100/// production implementation.
101pub trait BufferUploader {
102 /// The created buffer — `wgpu::Buffer` in production.
103 type Buffer;
104
105 /// Creates an uninitialized buffer of `size` bytes with `usage`.
106 fn allocate_buffer(
107 &self,
108 label: Option<&str>,
109 size: u64,
110 usage: wgpu::BufferUsages,
111 ) -> Self::Buffer;
112
113 /// Copies `bytes` into `buffer` at `offset`.
114 fn write_buffer(&self, buffer: &Self::Buffer, offset: u64, bytes: &[u8]);
115}
116
117impl BufferUploader for DeviceHandle {
118 type Buffer = wgpu::Buffer;
119
120 fn allocate_buffer(
121 &self,
122 label: Option<&str>,
123 size: u64,
124 usage: wgpu::BufferUsages,
125 ) -> wgpu::Buffer {
126 self.device.create_buffer(&wgpu::BufferDescriptor {
127 label,
128 size,
129 usage,
130 mapped_at_creation: false,
131 })
132 }
133
134 fn write_buffer(&self, buffer: &wgpu::Buffer, offset: u64, bytes: &[u8]) {
135 self.queue.write_buffer(buffer, offset, bytes);
136 }
137}
138
139/// One growable GPU buffer plus the CPU-side staging vector a frame bump-
140/// allocates into.
141///
142/// Generic over the buffer type (`B`) for the same reason
143/// [`crate::texture::Texture`] is generic over its texture: the bump
144/// pointer, alignment, growth and flush behavior are exercised on a host
145/// with a stand-in value. The engine always uses the default
146/// `HostBuffer<wgpu::Buffer>`.
147#[derive(Debug)]
148pub struct HostBuffer<B = wgpu::Buffer> {
149 staging: Vec<u8>,
150 buffer: Option<B>,
151 capacity: u64,
152 high_water_mark: u64,
153 min_alignment: u32,
154 label: String,
155 buffer_allocations: u64,
156 flushes: u64,
157}
158
159impl<B> HostBuffer<B> {
160 /// An empty arena aligned for `caps`' adapter. No GPU buffer exists
161 /// until the first [`Self::flush`] that has bytes to upload.
162 pub fn new(caps: &TierCaps) -> Self {
163 Self::with_label(caps, DEFAULT_LABEL)
164 }
165
166 /// [`Self::new`] with a caller-chosen debug label on the GPU buffer,
167 /// for a host that runs more than one arena and wants to tell them
168 /// apart in a graphics debugger.
169 pub fn with_label(caps: &TierCaps, label: &str) -> Self {
170 Self {
171 staging: Vec::new(),
172 buffer: None,
173 capacity: 0,
174 high_water_mark: 0,
175 // An adapter that reported 0 would make every offset alignment
176 // a no-op; 1 keeps the arithmetic total and the bytes packed.
177 min_alignment: caps.min_uniform_buffer_offset_alignment.max(1),
178 label: label.to_string(),
179 buffer_allocations: 0,
180 flushes: 0,
181 }
182 }
183
184 /// Appends `bytes` to this frame's data and returns the slice they
185 /// occupy.
186 ///
187 /// The offset is rounded up to `max(align, min_uniform_buffer_offset_alignment)`,
188 /// so a caller passes the alignment its *own* use needs (a vertex
189 /// stride, `1` for "no opinion") and never has to know the adapter's.
190 /// Padding bytes are zero-filled.
191 pub fn alloc(&mut self, bytes: &[u8], align: u32) -> BufferSlice {
192 let align = align.max(self.min_alignment).max(1) as usize;
193 let offset = self.staging.len().next_multiple_of(align);
194 self.staging.resize(offset, 0);
195 self.staging.extend_from_slice(bytes);
196 self.high_water_mark = self.high_water_mark.max(self.staging.len() as u64);
197 BufferSlice {
198 offset: offset as u64,
199 size: bytes.len() as u64,
200 }
201 }
202
203 /// Rewinds the bump pointer for a new frame, keeping the GPU buffer and
204 /// its capacity.
205 ///
206 /// Every [`BufferSlice`] handed out before this call is stale
207 /// afterwards — the next frame's allocations reuse those offsets.
208 pub fn reset(&mut self) {
209 self.staging.clear();
210 }
211
212 /// Uploads this frame's bytes in one `write_buffer`, growing (or
213 /// creating) the GPU buffer first if the frame outgrew it, and returns
214 /// the buffer the frame's [`BufferSlice`] offsets refer to.
215 ///
216 /// Answers `None` only when nothing has ever been allocated, so there is
217 /// no buffer to name. Call once per frame, after the frame's last
218 /// [`Self::alloc`] and before submitting commands that read it.
219 pub fn flush<U>(&mut self, uploader: &U) -> Option<&B>
220 where
221 U: BufferUploader<Buffer = B>,
222 {
223 // `write_buffer` copies whole 4-byte units, so the tail of an
224 // odd-length frame is zero-padded up to that boundary rather than
225 // dropped.
226 let len = self
227 .staging
228 .len()
229 .next_multiple_of(wgpu::COPY_BUFFER_ALIGNMENT as usize);
230 self.staging.resize(len, 0);
231 if len == 0 {
232 return self.buffer.as_ref();
233 }
234
235 if self.buffer.is_none() || self.capacity < len as u64 {
236 let capacity = grown_capacity(self.capacity, len as u64);
237 self.buffer = Some(uploader.allocate_buffer(Some(&self.label), capacity, ARENA_USAGE));
238 self.capacity = capacity;
239 self.buffer_allocations += 1;
240 }
241 self.flushes += 1;
242
243 let buffer = self
244 .buffer
245 .as_ref()
246 .expect("the arena buffer exists once a frame has bytes");
247 uploader.write_buffer(buffer, 0, &self.staging);
248 Some(buffer)
249 }
250
251 /// The GPU buffer backing the arena, once one exists.
252 pub fn buffer(&self) -> Option<&B> {
253 self.buffer.as_ref()
254 }
255
256 /// Bytes allocated so far this frame, padding included.
257 pub fn used(&self) -> u64 {
258 self.staging.len() as u64
259 }
260
261 /// The GPU buffer's current size in bytes, `0` before the first flush.
262 pub fn capacity(&self) -> u64 {
263 self.capacity
264 }
265
266 /// The largest single frame the arena has ever held, in bytes — what a
267 /// host would size an initial capacity against.
268 pub fn high_water_mark(&self) -> u64 {
269 self.high_water_mark
270 }
271
272 /// How many times a GPU buffer has been created: one for the first
273 /// non-empty frame, plus one per growth. A steady-state frame loop
274 /// leaves this constant.
275 pub fn buffer_allocations(&self) -> u64 {
276 self.buffer_allocations
277 }
278
279 /// How many uploads the arena has issued — one per non-empty flush.
280 pub fn flush_count(&self) -> u64 {
281 self.flushes
282 }
283
284 /// The alignment floor every slice is rounded up to, from the adapter's
285 /// `min_uniform_buffer_offset_alignment`.
286 pub fn min_alignment(&self) -> u32 {
287 self.min_alignment
288 }
289}
290
291/// The capacity a buffer of `current` bytes grows to in order to hold
292/// `required`.
293///
294/// Geometric (next power of two, never below [`MIN_ARENA_CAPACITY`]) so a
295/// frame that creeps upward by a few bytes each frame does not reallocate
296/// every frame, and never smaller than `current` — the arena is grow-only,
297/// and a buffer that shrank would invalidate offsets a previous frame's
298/// in-flight commands still read. A `required` beyond the adapter's
299/// `max_buffer_size` is the device's error to report, not something the
300/// arena can round away.
301fn grown_capacity(current: u64, required: u64) -> u64 {
302 current.max(
303 required
304 .checked_next_power_of_two()
305 .unwrap_or(required)
306 .max(MIN_ARENA_CAPACITY),
307 )
308}
309
310#[cfg(test)]
311mod tests {
312 use super::*;
313 use crate::caps::DownlevelProfile;
314
315 /// A [`BufferUploader`] with no GPU behind it: the buffer is an id, and
316 /// every create/write is counted so growth and per-frame upload counts
317 /// are assertable.
318 #[derive(Debug, Default)]
319 struct FakeUploader {
320 created: std::cell::Cell<u32>,
321 writes: std::cell::Cell<u32>,
322 last_write_len: std::cell::Cell<usize>,
323 }
324
325 impl BufferUploader for FakeUploader {
326 type Buffer = u32;
327
328 fn allocate_buffer(&self, _label: Option<&str>, _size: u64, _u: wgpu::BufferUsages) -> u32 {
329 self.created.set(self.created.get() + 1);
330 self.created.get()
331 }
332
333 fn write_buffer(&self, _buffer: &u32, _offset: u64, bytes: &[u8]) {
334 self.writes.set(self.writes.get() + 1);
335 self.last_write_len.set(bytes.len());
336 }
337 }
338
339 fn arena() -> HostBuffer<u32> {
340 HostBuffer::new(&TierCaps::fake(DownlevelProfile::WebGl2))
341 }
342
343 #[test]
344 fn alloc_aligns_to_the_adapter_minimum() {
345 let mut arena = arena();
346 assert_eq!(arena.min_alignment(), 256);
347 let first = arena.alloc(&[1u8; 4], 1);
348 let second = arena.alloc(&[2u8; 300], 1);
349 let third = arena.alloc(&[3u8; 1], 1);
350 assert_eq!(first, BufferSlice { offset: 0, size: 4 });
351 assert_eq!(
352 second,
353 BufferSlice {
354 offset: 256,
355 size: 300
356 }
357 );
358 assert_eq!(
359 third,
360 BufferSlice {
361 offset: 768,
362 size: 1
363 }
364 );
365 }
366
367 #[test]
368 fn alloc_honors_a_larger_caller_alignment() {
369 let mut arena = arena();
370 arena.alloc(&[0u8; 1], 1);
371 let wide = arena.alloc(&[0u8; 8], 1024);
372 assert_eq!(wide.offset, 1024);
373 }
374
375 #[test]
376 fn reset_rewinds_and_keeps_the_high_water_mark() {
377 let mut arena = arena();
378 arena.alloc(&[0u8; 512], 1);
379 assert_eq!(arena.used(), 512);
380 assert_eq!(arena.high_water_mark(), 512);
381
382 arena.reset();
383 assert_eq!(arena.used(), 0);
384 assert_eq!(arena.high_water_mark(), 512);
385
386 arena.alloc(&[0u8; 16], 1);
387 assert_eq!(arena.alloc(&[0u8; 16], 1).offset, 256);
388 assert_eq!(arena.high_water_mark(), 512);
389 }
390
391 #[test]
392 fn flush_with_nothing_allocated_creates_no_buffer() {
393 let mut arena = arena();
394 let uploader = FakeUploader::default();
395 assert_eq!(arena.flush(&uploader), None);
396 assert_eq!(uploader.created.get(), 0);
397 assert_eq!(uploader.writes.get(), 0);
398 }
399
400 #[test]
401 fn flush_pads_the_upload_to_the_copy_alignment() {
402 let mut arena = arena();
403 let uploader = FakeUploader::default();
404 arena.alloc(&[7u8; 5], 1);
405 arena.flush(&uploader);
406 assert_eq!(uploader.last_write_len.get(), 8);
407 }
408
409 #[test]
410 fn growth_is_geometric_and_never_shrinks() {
411 assert_eq!(grown_capacity(0, 1), MIN_ARENA_CAPACITY);
412 assert_eq!(grown_capacity(0, MIN_ARENA_CAPACITY), MIN_ARENA_CAPACITY);
413 assert_eq!(
414 grown_capacity(MIN_ARENA_CAPACITY, MIN_ARENA_CAPACITY + 1),
415 2 * MIN_ARENA_CAPACITY
416 );
417 // A smaller frame leaves the capacity where it is.
418 assert_eq!(
419 grown_capacity(4 * MIN_ARENA_CAPACITY, 8),
420 4 * MIN_ARENA_CAPACITY
421 );
422 }
423}