frust_gpu/diag.rs
1//! GPU-side frame diagnostics: [`TimestampRing`], real GPU time per pass.
2//!
3//! A frame's `encode_us`/`submit_us` are CPU wall-clock spans around calls that
4//! only *record* and *queue* work — they say how long the CPU spent, never how
5//! long the GPU took. This module answers the second question directly, by
6//! having the GPU stamp its own clock at each pass boundary.
7//!
8//! # Pass boundaries only
9//!
10//! Every timestamp this ring takes is written through
11//! `wgpu::RenderPassDescriptor::timestamp_writes`, which needs only
12//! `wgpu::Features::TIMESTAMP_QUERY`. `CommandEncoder::write_timestamp` and
13//! `RenderPass::write_timestamp` are deliberately never used: they need
14//! `TIMESTAMP_QUERY_INSIDE_ENCODERS`/`TIMESTAMP_QUERY_INSIDE_PASSES`, which
15//! `wgpu` documents as unavailable on tile-based GPUs — Adreno, and Apple's own
16//! Metal families — i.e. exactly the mobile adapters this tier exists for. A
17//! probe that only works on desktop would answer the wrong question.
18//!
19//! # Inert unless the device really has the feature
20//!
21//! `TIMESTAMP_QUERY` is requested only by a `perf-trace` build, and only when
22//! the adapter offers it ([`crate::context`]'s `required_features`). This ring
23//! makes no assumption about that: it asks the **device** it is handed what it
24//! was created with, and when the answer is no it becomes inert — no query set,
25//! no buffers, no per-frame work, and [`TimestampRing::latest`] answering
26//! `None` forever. An inert ring is what a caller reports as `gpu_q=0`.
27//!
28//! # The ring
29//!
30//! Reading a timestamp back means mapping a buffer, which is only ready some
31//! frames after the submit that wrote it. So the ring holds [`RING_FRAMES`]
32//! independent slots — query set, resolve buffer, readback buffer — and a
33//! frame's slot is not read until the ring comes back around to it, by which
34//! time its map has long since completed. Nothing on the frame path ever
35//! blocks: the map is polled, never waited on, and a slot whose map has not
36//! landed simply goes untimed for that frame rather than stalling it.
37//!
38//! # A pass that draws nothing may measure nothing
39//!
40//! Metal samples a render pass's counters at the vertex/fragment *stage*
41//! boundaries, so a pass that runs neither stage — a pure clear, an empty
42//! pass — can leave its query pair unwritten, and the ring drops the pair
43//! rather than reporting a garbage span (see `span_duration`). This is a
44//! measurement gap, never a correctness one: an untimed pass draws exactly
45//! what it always did. A span made of several passes, which is the shape every
46//! caller here uses, absorbs it — the drawing passes still report.
47//!
48//! One frame's shape, in call order:
49//!
50//! 1. [`TimestampRing::begin_frame`] — advance to the next slot and harvest
51//! whatever the frame that last used it left mapped.
52//! 2. [`TimestampRing::pass_writes`] once per timed pass, naming the span the
53//! pass belongs to. Each call takes a *fresh* query pair, so several passes
54//! may share one span and their durations add up.
55//! 3. [`TimestampRing::resolve`] into the same encoder the passes were recorded
56//! into, before the caller submits it.
57//! 4. [`TimestampRing::end_frame`] after that submit (or
58//! [`TimestampRing::abandon_frame`] when the frame was refused and nothing
59//! was submitted).
60
61use std::sync::Arc;
62use std::sync::atomic::{AtomicU8, AtomicU32, Ordering};
63use std::time::Duration;
64
65/// How many frames-in-flight the ring keeps independent slots for.
66///
67/// Four is the depth a triple-buffered swapchain plus one in-flight submit can
68/// reach, so a slot's map has always completed by the time the ring returns to
69/// it — nothing on the frame path ever waits on a map.
70pub const RING_FRAMES: usize = 4;
71
72/// The largest number of named spans one frame can be split into.
73///
74/// A fixed ceiling rather than a `Vec` so [`GpuFrameSpans`] stays `Copy` and a
75/// caller can carry one through a frame record without allocating.
76pub const MAX_SPANS: usize = 8;
77
78/// How many individual passes one frame can time by default.
79///
80/// Each timed pass costs its own query pair, so this is what fixes the query
81/// set's `count` at `2 * PASSES`. A frame with more passes than this times the
82/// first `DEFAULT_PASS_CAPACITY` and leaves the rest untimed — draw-correct
83/// either way, since a pass with no `timestamp_writes` is an ordinary pass.
84pub const DEFAULT_PASS_CAPACITY: usize = 24;
85
86/// The hard ceiling on a slot's query count, from `wgpu`'s own
87/// `QUERY_SET_MAX_QUERIES`. Halved because every pass takes a *pair*.
88const MAX_PASS_CAPACITY: usize = (wgpu::QUERY_SET_MAX_QUERIES / 2) as usize;
89
90/// Bytes one resolved timestamp query occupies (`wgpu::QUERY_SIZE`).
91const QUERY_BYTES: u64 = wgpu::QUERY_SIZE as u64;
92
93/// A single pass span longer than this is discarded rather than reported.
94///
95/// A GPU pass measured in whole seconds is not a slow frame, it is a garbage
96/// tick pair: an unwritten query, a driver that reset its clock across a
97/// submit, or a counter wrap. Reporting it would poison every percentile
98/// computed off the raw series, so the span is dropped and its frame simply
99/// reads a little low.
100const IMPLAUSIBLE_SPAN_NANOS: u64 = 1_000_000_000;
101
102/// The `AtomicU8` states a slot's map callback moves through.
103mod map_state {
104 /// The map was requested and has not answered yet.
105 pub const PENDING: u8 = 0;
106 /// The map completed and the buffer holds this frame's resolved queries.
107 pub const READY: u8 = 1;
108 /// The map failed; the slot is recycled without producing a reading.
109 pub const FAILED: u8 = 2;
110}
111
112/// One frame's GPU span durations, in the caller's own span order.
113///
114/// The span *names* are the caller's business — this crate knows only how many
115/// there are and which index each pass was charged to. `frust-engine`'s
116/// `diag::EngineSpan` is the naming this repo's engine tier uses.
117#[derive(Debug, Clone, Copy, PartialEq, Eq)]
118pub struct GpuFrameSpans {
119 durations: [Duration; MAX_SPANS],
120 len: usize,
121}
122
123impl Default for GpuFrameSpans {
124 fn default() -> Self {
125 Self {
126 durations: [Duration::ZERO; MAX_SPANS],
127 len: 0,
128 }
129 }
130}
131
132impl GpuFrameSpans {
133 /// An all-zero reading over `len` spans (clamped to [`MAX_SPANS`]).
134 #[must_use]
135 pub fn zeroed(len: usize) -> Self {
136 Self {
137 durations: [Duration::ZERO; MAX_SPANS],
138 len: len.min(MAX_SPANS),
139 }
140 }
141
142 /// How many spans this reading covers.
143 #[must_use]
144 pub fn len(&self) -> usize {
145 self.len
146 }
147
148 /// Whether this reading covers no spans at all.
149 #[must_use]
150 pub fn is_empty(&self) -> bool {
151 self.len == 0
152 }
153
154 /// The duration charged to `index`, or [`Duration::ZERO`] past the end —
155 /// a span no pass was charged to is genuinely zero, so an out-of-range
156 /// index answering the same is the honest reading rather than a silent
157 /// failure mode.
158 #[must_use]
159 pub fn span(&self, index: usize) -> Duration {
160 self.durations.get(index).copied().unwrap_or(Duration::ZERO)
161 }
162
163 /// Every span in order.
164 #[must_use]
165 pub fn as_slice(&self) -> &[Duration] {
166 &self.durations[..self.len]
167 }
168
169 /// The sum of every span — the frame's total *attributed* GPU pass time.
170 ///
171 /// Deliberately a sum rather than last-end-minus-first-begin: the gaps
172 /// between a frame's passes are queue and driver time this ring cannot
173 /// attribute to any pass, and folding them into a "total" would report
174 /// work the breakdown below it does not account for.
175 #[must_use]
176 pub fn total(&self) -> Duration {
177 self.as_slice()
178 .iter()
179 .copied()
180 .fold(Duration::ZERO, |acc, span| acc.saturating_add(span))
181 }
182
183 /// Charges `span` with `duration` on top of whatever it already holds.
184 fn add(&mut self, span: usize, duration: Duration) {
185 if let Some(slot) = self.durations.get_mut(span) {
186 *slot = slot.saturating_add(duration);
187 }
188 }
189}
190
191/// One frame-in-flight's own query set and its two staging buffers.
192#[derive(Debug)]
193struct RingFrame {
194 queries: wgpu::QuerySet,
195 /// `QUERY_RESOLVE | COPY_SRC` — what `resolve_query_set` writes into. It
196 /// cannot also be `MAP_READ`, which is why the readback below exists.
197 resolve: wgpu::Buffer,
198 /// `COPY_DST | MAP_READ` — the mappable copy the harvest reads.
199 readback: wgpu::Buffer,
200 /// Next free pass pair, bumped by [`TimestampRing::pass_writes`].
201 ///
202 /// Atomic rather than a `Cell` so the ring stays `Sync`: a renderer that
203 /// records its passes off the thread that owns the ring is a shape this
204 /// type should not rule out, and the counter is uncontended in every
205 /// current caller anyway.
206 cursor: AtomicU32,
207 /// Which span each pass pair was charged to, indexed by pair.
208 owners: Vec<AtomicU8>,
209 state: SlotState,
210}
211
212/// Where one ring slot is in the record → resolve → map → harvest cycle.
213#[derive(Debug)]
214enum SlotState {
215 /// Holds nothing; free to record into.
216 Idle,
217 /// This frame's passes are being recorded into it.
218 Recording,
219 /// Submitted and mapping; `used` pass pairs are staged in `readback`.
220 Mapping { used: usize, status: Arc<AtomicU8> },
221}
222
223/// A ring of per-frame GPU timestamp query sets — see the module header.
224#[derive(Debug)]
225pub struct TimestampRing {
226 /// How many named spans a frame is split into.
227 spans: usize,
228 /// How many passes one frame can time.
229 passes: usize,
230 /// Nanoseconds per timestamp tick, from `queue.get_timestamp_period()`.
231 period_ns: f32,
232 /// Empty when inert — the one thing every method keys off.
233 frames: Vec<RingFrame>,
234 /// Which slot the frame in progress is recording into.
235 current: usize,
236 /// Whether the current frame's slot was actually free to record into.
237 armed: bool,
238 /// The most recently harvested reading.
239 latest: Option<GpuFrameSpans>,
240 /// How many frames have been harvested, so a consumer can tell a fresh
241 /// reading from the one it already reported.
242 harvested: u64,
243}
244
245impl TimestampRing {
246 /// A ring timing `spans` named spans per frame, over
247 /// [`DEFAULT_PASS_CAPACITY`] passes.
248 ///
249 /// Inert — no GPU resource created at all — when `device` was not created
250 /// with `wgpu::Features::TIMESTAMP_QUERY`, which is every build that did
251 /// not compile `perf-trace` in and every adapter that does not offer the
252 /// feature.
253 #[must_use]
254 pub fn new(device: &wgpu::Device, queue: &wgpu::Queue, spans: usize, label: &str) -> Self {
255 Self::with_capacity(device, queue, spans, DEFAULT_PASS_CAPACITY, label)
256 }
257
258 /// [`Self::new`] with an explicit per-frame pass capacity.
259 #[must_use]
260 pub fn with_capacity(
261 device: &wgpu::Device,
262 queue: &wgpu::Queue,
263 spans: usize,
264 passes: usize,
265 label: &str,
266 ) -> Self {
267 let spans = spans.min(MAX_SPANS);
268 let passes = passes.min(MAX_PASS_CAPACITY);
269 if spans == 0 || passes == 0 || !device.features().contains(wgpu::Features::TIMESTAMP_QUERY)
270 {
271 return Self::inert(spans);
272 }
273
274 // A non-finite or non-positive period would scale every tick delta
275 // into nonsense, and there is no sound reading to fall back on — an
276 // inert ring reporting nothing beats a ring reporting garbage.
277 let period_ns = queue.get_timestamp_period();
278 if !period_ns.is_finite() || period_ns <= 0.0 {
279 log::warn!(
280 "frust-gpu: timestamp period {period_ns} is unusable; GPU pass timing is off"
281 );
282 return Self::inert(spans);
283 }
284
285 let query_count = (passes * 2) as u32;
286 let staged_bytes = u64::from(query_count) * QUERY_BYTES;
287 // `resolve_query_set`'s destination offset is 256-byte aligned; the
288 // buffer is rounded up to the same grid so a future non-zero offset
289 // needs no re-sizing rule of its own.
290 let buffer_bytes = staged_bytes.next_multiple_of(wgpu::QUERY_RESOLVE_BUFFER_ALIGNMENT);
291
292 let frames = (0..RING_FRAMES)
293 .map(|slot| RingFrame {
294 queries: device.create_query_set(&wgpu::QuerySetDescriptor {
295 label: Some(&format!("{label} queries {slot}")),
296 ty: wgpu::QueryType::Timestamp,
297 count: query_count,
298 }),
299 resolve: device.create_buffer(&wgpu::BufferDescriptor {
300 label: Some(&format!("{label} resolve {slot}")),
301 size: buffer_bytes,
302 usage: wgpu::BufferUsages::QUERY_RESOLVE | wgpu::BufferUsages::COPY_SRC,
303 mapped_at_creation: false,
304 }),
305 readback: device.create_buffer(&wgpu::BufferDescriptor {
306 label: Some(&format!("{label} readback {slot}")),
307 size: buffer_bytes,
308 usage: wgpu::BufferUsages::COPY_DST | wgpu::BufferUsages::MAP_READ,
309 mapped_at_creation: false,
310 }),
311 cursor: AtomicU32::new(0),
312 owners: (0..passes).map(|_| AtomicU8::new(0)).collect(),
313 state: SlotState::Idle,
314 })
315 .collect();
316
317 Self {
318 spans,
319 passes,
320 period_ns,
321 frames,
322 // The first `begin_frame` advances onto slot 0.
323 current: RING_FRAMES - 1,
324 armed: false,
325 latest: None,
326 harvested: 0,
327 }
328 }
329
330 /// A ring that measures nothing, for a device without `TIMESTAMP_QUERY`
331 /// and for a caller that wants the `gpu_q=0` shape without a GPU at all.
332 #[must_use]
333 pub fn inert(spans: usize) -> Self {
334 Self {
335 spans: spans.min(MAX_SPANS),
336 passes: 0,
337 period_ns: 0.0,
338 frames: Vec::new(),
339 current: 0,
340 armed: false,
341 latest: None,
342 harvested: 0,
343 }
344 }
345
346 /// Whether this ring holds real query sets, i.e. whether a frame line
347 /// should report `gpu_q=1`.
348 #[must_use]
349 pub fn is_active(&self) -> bool {
350 !self.frames.is_empty()
351 }
352
353 /// How many named spans a frame is split into.
354 #[must_use]
355 pub fn spans(&self) -> usize {
356 self.spans
357 }
358
359 /// Nanoseconds per timestamp tick, `0.0` on an inert ring.
360 #[must_use]
361 pub fn timestamp_period_ns(&self) -> f32 {
362 self.period_ns
363 }
364
365 /// The most recently harvested frame's spans, or `None` before the first
366 /// reading lands (and forever on an inert ring).
367 ///
368 /// The reading lags the CPU frame that asks for it by up to
369 /// [`RING_FRAMES`] frames — the price of never blocking on a map. In
370 /// steady state one fresh reading lands per frame, so the lag is a
371 /// constant offset rather than a gap in the series; [`Self::harvested`]
372 /// is how a consumer tells a repeat from a fresh reading.
373 #[must_use]
374 pub fn latest(&self) -> Option<GpuFrameSpans> {
375 self.latest
376 }
377
378 /// How many frame readings this ring has harvested.
379 #[must_use]
380 pub fn harvested(&self) -> u64 {
381 self.harvested
382 }
383
384 /// Advances onto the next slot and harvests whatever the frame that last
385 /// used it left mapped.
386 ///
387 /// Polls `device` once, without blocking: a map that has not landed leaves
388 /// its slot alone and the frame goes untimed rather than stalling on it.
389 pub fn begin_frame(&mut self, device: &wgpu::Device) {
390 if self.frames.is_empty() {
391 return;
392 }
393 self.current = (self.current + 1) % self.frames.len();
394
395 // One non-blocking poll, which is what actually invokes any map
396 // callback whose copy has completed. A poll failure is not worth
397 // reporting per frame: the slot simply stays pending and this frame
398 // goes untimed.
399 let _ = device.poll(wgpu::PollType::Poll);
400
401 let spans = self.spans;
402 let period_ns = self.period_ns;
403 let Some(slot) = self.frames.get_mut(self.current) else {
404 self.armed = false;
405 return;
406 };
407
408 if let SlotState::Mapping { used, status } = &slot.state {
409 match status.load(Ordering::Acquire) {
410 map_state::PENDING => {
411 // Still in flight: leave the slot exactly as it is (its
412 // readback buffer is mapped-in-progress and must not be
413 // written again) and skip timing this frame.
414 self.armed = false;
415 return;
416 }
417 map_state::READY => {
418 let reading = harvest(slot, *used, spans, period_ns);
419 slot.readback.unmap();
420 self.latest = Some(reading);
421 self.harvested = self.harvested.saturating_add(1);
422 }
423 // A failed map leaves `wgpu`'s own map context marked, so the
424 // slot is unmapped anyway before it is recorded into again.
425 _ => slot.readback.unmap(),
426 }
427 }
428
429 slot.cursor.store(0, Ordering::Relaxed);
430 slot.state = SlotState::Recording;
431 self.armed = true;
432 }
433
434 /// The `timestamp_writes` for one pass belonging to `span`.
435 ///
436 /// Each call takes a fresh query pair, so a span made of several passes
437 /// (the frame's own strip passes, a layer's page rounds) sums correctly.
438 /// `None` — an untimed pass — when the ring is inert, the slot was not
439 /// free this frame, `span` is out of range, or the frame has already used
440 /// its pass capacity.
441 #[must_use]
442 pub fn pass_writes(&self, span: usize) -> Option<wgpu::RenderPassTimestampWrites<'_>> {
443 if !self.armed || span >= self.spans {
444 return None;
445 }
446 let slot = self.frames.get(self.current)?;
447 let pair = slot.cursor.fetch_add(1, Ordering::Relaxed) as usize;
448 if pair >= self.passes {
449 return None;
450 }
451 slot.owners.get(pair)?.store(span as u8, Ordering::Relaxed);
452 let first = (pair * 2) as u32;
453 Some(wgpu::RenderPassTimestampWrites {
454 query_set: &slot.queries,
455 beginning_of_pass_write_index: Some(first),
456 end_of_pass_write_index: Some(first + 1),
457 })
458 }
459
460 /// Records this frame's query resolve and its copy into the mappable
461 /// readback buffer, into the same encoder the timed passes were recorded
462 /// into. The caller submits that encoder.
463 ///
464 /// Only the pairs actually taken are resolved: an untouched query resolves
465 /// to whatever the driver left there, and reading it back would be
466 /// inventing a span rather than measuring one.
467 pub fn resolve(&self, encoder: &mut wgpu::CommandEncoder) {
468 if !self.armed {
469 return;
470 }
471 let Some(slot) = self.frames.get(self.current) else {
472 return;
473 };
474 let used = self.used_pairs(slot);
475 if used == 0 {
476 return;
477 }
478 let queries = (used * 2) as u32;
479 encoder.resolve_query_set(&slot.queries, 0..queries, &slot.resolve, 0);
480 encoder.copy_buffer_to_buffer(
481 &slot.resolve,
482 0,
483 &slot.readback,
484 0,
485 u64::from(queries) * QUERY_BYTES,
486 );
487 }
488
489 /// Requests the map that a later [`Self::begin_frame`] harvests. Call
490 /// after submitting the encoder [`Self::resolve`] was recorded into.
491 pub fn end_frame(&mut self) {
492 if !self.armed {
493 return;
494 }
495 self.armed = false;
496 let Some(slot) = self.frames.get_mut(self.current) else {
497 return;
498 };
499 let used = (slot.cursor.load(Ordering::Relaxed) as usize).min(self.passes);
500 if used == 0 {
501 slot.state = SlotState::Idle;
502 return;
503 }
504 let status = Arc::new(AtomicU8::new(map_state::PENDING));
505 let callback = Arc::clone(&status);
506 let bytes = (used * 2) as u64 * QUERY_BYTES;
507 slot.readback
508 .slice(0..bytes)
509 .map_async(wgpu::MapMode::Read, move |result| {
510 let state = if result.is_ok() {
511 map_state::READY
512 } else {
513 map_state::FAILED
514 };
515 callback.store(state, Ordering::Release);
516 });
517 slot.state = SlotState::Mapping { used, status };
518 }
519
520 /// Closes a frame that was refused before its encoder was submitted.
521 ///
522 /// Nothing was recorded and nothing will be, so the slot goes straight
523 /// back to idle without a map — mapping a buffer no copy ever reached
524 /// would harvest the previous frame's ticks as if they were this one's.
525 pub fn abandon_frame(&mut self) {
526 if !self.armed {
527 return;
528 }
529 self.armed = false;
530 if let Some(slot) = self.frames.get_mut(self.current) {
531 slot.cursor.store(0, Ordering::Relaxed);
532 slot.state = SlotState::Idle;
533 }
534 }
535
536 /// How many pass pairs the frame in progress actually took.
537 fn used_pairs(&self, slot: &RingFrame) -> usize {
538 (slot.cursor.load(Ordering::Relaxed) as usize).min(self.passes)
539 }
540}
541
542/// Reads `used` resolved query pairs out of `slot`'s mapped readback buffer and
543/// charges each one to the span its pass named.
544fn harvest(slot: &RingFrame, used: usize, spans: usize, period_ns: f32) -> GpuFrameSpans {
545 let mut reading = GpuFrameSpans::zeroed(spans);
546 let bytes = (used * 2) as u64 * QUERY_BYTES;
547 let mapped =
548 slot.readback.slice(0..bytes).get_mapped_range().expect(
549 "frust-gpu diag: the harvested slot's readback is mapped over the resolved range",
550 );
551 let stride = (QUERY_BYTES * 2) as usize;
552 for pair in 0..used {
553 let at = pair * stride;
554 let Some(chunk) = mapped.get(at..at + stride) else {
555 break;
556 };
557 let (begin, end) = tick_pair(chunk);
558 let Some(duration) = span_duration(begin, end, period_ns) else {
559 continue;
560 };
561 let owner = slot
562 .owners
563 .get(pair)
564 .map_or(usize::MAX, |span| span.load(Ordering::Relaxed) as usize);
565 reading.add(owner, duration);
566 }
567 drop(mapped);
568 reading
569}
570
571/// The `(begin, end)` tick pair a 16-byte resolved chunk holds, little-endian
572/// as `wgpu` resolves it. A short chunk answers `(0, 0)`, which
573/// [`span_duration`] then rejects — the one shape that cannot be a real span.
574fn tick_pair(chunk: &[u8]) -> (u64, u64) {
575 fn tick(bytes: Option<&[u8]>) -> u64 {
576 bytes
577 .and_then(|slice| <[u8; 8]>::try_from(slice).ok())
578 .map_or(0, u64::from_le_bytes)
579 }
580 (tick(chunk.get(0..8)), tick(chunk.get(8..16)))
581}
582
583/// One pass's duration from its tick pair, or `None` when the pair cannot be a
584/// real measurement.
585///
586/// Rejected: a non-increasing pair (an unwritten query, or a counter that
587/// wrapped across the pass) and anything past [`IMPLAUSIBLE_SPAN_NANOS`] (see
588/// that constant). Both are dropped rather than clamped — a clamped garbage
589/// value is still reported as a measurement.
590fn span_duration(begin: u64, end: u64, period_ns: f32) -> Option<Duration> {
591 if end <= begin || !period_ns.is_finite() || period_ns <= 0.0 {
592 return None;
593 }
594 let nanos = (end - begin) as f64 * f64::from(period_ns);
595 if !nanos.is_finite() || nanos >= IMPLAUSIBLE_SPAN_NANOS as f64 {
596 return None;
597 }
598 Some(Duration::from_nanos(nanos as u64))
599}
600
601#[cfg(test)]
602mod tests {
603 use super::*;
604
605 #[test]
606 fn zeroed_reading_reports_its_span_count_and_no_time() {
607 let reading = GpuFrameSpans::zeroed(4);
608 assert_eq!(reading.len(), 4);
609 assert!(!reading.is_empty());
610 assert_eq!(reading.as_slice().len(), 4);
611 assert_eq!(reading.total(), Duration::ZERO);
612 for index in 0..4 {
613 assert_eq!(reading.span(index), Duration::ZERO);
614 }
615 }
616
617 #[test]
618 fn span_count_is_clamped_to_the_ceiling() {
619 assert_eq!(GpuFrameSpans::zeroed(MAX_SPANS + 9).len(), MAX_SPANS);
620 }
621
622 #[test]
623 fn an_index_past_the_end_reads_zero_rather_than_panicking() {
624 let reading = GpuFrameSpans::zeroed(2);
625 assert_eq!(reading.span(7), Duration::ZERO);
626 assert_eq!(reading.span(usize::MAX), Duration::ZERO);
627 }
628
629 #[test]
630 fn several_passes_charged_to_one_span_add_up() {
631 let mut reading = GpuFrameSpans::zeroed(3);
632 reading.add(1, Duration::from_micros(400));
633 reading.add(1, Duration::from_micros(250));
634 reading.add(2, Duration::from_micros(90));
635 assert_eq!(reading.span(0), Duration::ZERO);
636 assert_eq!(reading.span(1), Duration::from_micros(650));
637 assert_eq!(reading.span(2), Duration::from_micros(90));
638 assert_eq!(reading.total(), Duration::from_micros(740));
639 }
640
641 #[test]
642 fn a_charge_past_the_span_ceiling_is_dropped_not_wrapped() {
643 let mut reading = GpuFrameSpans::zeroed(2);
644 reading.add(MAX_SPANS, Duration::from_secs(1));
645 assert_eq!(reading.total(), Duration::ZERO);
646 }
647
648 #[test]
649 fn span_duration_scales_ticks_by_the_period() {
650 // 1000 ticks at 1ns/tick, and the same 1000 ticks on an adapter whose
651 // tick is 38.4ns (a real Adreno-class period) — the same pair must
652 // read as two different durations, which is the whole point of
653 // scaling by the queue's own period rather than assuming nanoseconds.
654 assert_eq!(span_duration(0, 1_000, 1.0), Some(Duration::from_micros(1)));
655 assert_eq!(
656 span_duration(500, 1_500, 38.4),
657 Some(Duration::from_nanos(38_400))
658 );
659 }
660
661 #[test]
662 fn a_non_increasing_tick_pair_is_no_measurement() {
663 assert_eq!(span_duration(0, 0, 1.0), None);
664 assert_eq!(span_duration(900, 900, 1.0), None);
665 assert_eq!(span_duration(1_000, 999, 1.0), None);
666 }
667
668 #[test]
669 fn an_implausible_span_is_dropped_rather_than_reported() {
670 assert_eq!(span_duration(0, IMPLAUSIBLE_SPAN_NANOS, 1.0), None);
671 assert_eq!(span_duration(0, u64::MAX, 1.0), None);
672 // Just under the ceiling still reads.
673 assert!(span_duration(0, IMPLAUSIBLE_SPAN_NANOS - 1, 1.0).is_some());
674 }
675
676 #[test]
677 fn an_unusable_period_yields_no_measurement() {
678 assert_eq!(span_duration(0, 1_000, 0.0), None);
679 assert_eq!(span_duration(0, 1_000, -1.0), None);
680 assert_eq!(span_duration(0, 1_000, f32::NAN), None);
681 assert_eq!(span_duration(0, 1_000, f32::INFINITY), None);
682 }
683
684 #[test]
685 fn a_tick_pair_reads_little_endian_and_a_short_chunk_reads_zero() {
686 let mut chunk = [0_u8; 16];
687 chunk[..8].copy_from_slice(&7_u64.to_le_bytes());
688 chunk[8..].copy_from_slice(&19_u64.to_le_bytes());
689 assert_eq!(tick_pair(&chunk), (7, 19));
690 assert_eq!(tick_pair(&chunk[..12]), (7, 0));
691 assert_eq!(tick_pair(&[]), (0, 0));
692 }
693
694 #[test]
695 fn an_inert_ring_measures_nothing_and_never_reports_a_reading() {
696 // The `gpu_q=0` shape a device without TIMESTAMP_QUERY produces,
697 // exercised with no GPU in the loop at all.
698 let mut ring = TimestampRing::inert(4);
699 assert!(!ring.is_active());
700 assert_eq!(ring.spans(), 4);
701 assert_eq!(ring.timestamp_period_ns(), 0.0);
702 assert!(ring.pass_writes(0).is_none());
703 assert!(ring.latest().is_none());
704 ring.end_frame();
705 ring.abandon_frame();
706 assert!(ring.latest().is_none());
707 assert_eq!(ring.harvested(), 0);
708 }
709
710 /// A full-screen triangle drawn by the real-adapter test below, so each
711 /// timed pass runs the vertex and fragment stages Metal samples its
712 /// counters at — see that test's own comment.
713 const RING_TEST_WGSL: &str = "\
714@vertex
715fn vs_main(@builtin(vertex_index) index: u32) -> @builtin(position) vec4<f32> {
716 let uv = vec2<f32>(f32((index << 1u) & 2u), f32(index & 2u));
717 return vec4<f32>(uv * 2.0 - 1.0, 0.0, 1.0);
718}
719
720@fragment
721fn fs_main() -> @location(0) vec4<f32> {
722 return vec4<f32>(0.2, 0.4, 0.8, 1.0);
723}
724";
725
726 /// Blocks on `future` by polling it to completion — `wgpu`'s native
727 /// adapter and device requests resolve without an executor driving them,
728 /// and this crate has no async runtime of its own.
729 fn block_on<F: std::future::Future>(future: F) -> F::Output {
730 use std::task::{Context, Poll, Waker};
731
732 let waker = Waker::noop();
733 let mut cx = Context::from_waker(waker);
734 let mut future = std::pin::pin!(future);
735 loop {
736 match future.as_mut().poll(&mut cx) {
737 Poll::Ready(value) => return value,
738 Poll::Pending => std::thread::yield_now(),
739 }
740 }
741 }
742
743 #[test]
744 #[ignore = "needs a real GPU adapter offering TIMESTAMP_QUERY; run with \
745 `cargo test -p frust-gpu -- --ignored` (pin the adapter on a \
746 multi-GPU host with WGPU_BACKEND / WGPU_ADAPTER_NAME)"]
747 fn a_real_adapter_reports_plausible_per_pass_gpu_time() {
748 block_on(async {
749 let instance = wgpu::Instance::new(
750 wgpu::InstanceDescriptor::new_without_display_handle_from_env(),
751 );
752 let adapter = wgpu::util::initialize_adapter_from_env_or_default(&instance, None)
753 .await
754 .expect("no compatible GPU adapter");
755 println!("frust-gpu timestamp-ring adapter: {:?}", adapter.get_info());
756 if !adapter.features().contains(wgpu::Features::TIMESTAMP_QUERY) {
757 println!("adapter offers no TIMESTAMP_QUERY — the inert path is the whole story");
758 return;
759 }
760 let (device, queue) = adapter
761 .request_device(&wgpu::DeviceDescriptor {
762 label: Some("frust-gpu timestamp ring test device"),
763 required_features: wgpu::Features::TIMESTAMP_QUERY,
764 required_limits: wgpu::Limits::default(),
765 ..Default::default()
766 })
767 .await
768 .expect("failed to create the device");
769
770 // Two spans, three passes charged across them: two into span 0 and
771 // one into span 1, so the reading is per-span rather than per-pass
772 // and both spans must come back nonzero.
773 //
774 // Every pass DRAWS. A draw-less pass is not a shortcut here: Metal
775 // samples a render pass's counters at the vertex/fragment stage
776 // boundaries (`setStartOfVertexSampleIndex`/
777 // `setEndOfFragmentSampleIndex`), so a pass that runs neither stage
778 // can leave its query pair unwritten — the case
779 // [`span_duration`] drops rather than reports.
780 let mut ring = TimestampRing::new(&device, &queue, 2, "frust-gpu ring test");
781 assert!(
782 ring.is_active(),
783 "a TIMESTAMP_QUERY device must arm the ring"
784 );
785 let format = wgpu::TextureFormat::Rgba8Unorm;
786 let target = crate::HeadlessTarget::new(&device, 512, 512, format);
787 let shader = device.create_shader_module(wgpu::ShaderModuleDescriptor {
788 label: Some("frust-gpu ring test shader"),
789 source: wgpu::ShaderSource::Wgsl(RING_TEST_WGSL.into()),
790 });
791 let pipeline = device.create_render_pipeline(&wgpu::RenderPipelineDescriptor {
792 label: Some("frust-gpu ring test pipeline"),
793 layout: None,
794 vertex: wgpu::VertexState {
795 module: &shader,
796 entry_point: Some("vs_main"),
797 compilation_options: wgpu::PipelineCompilationOptions::default(),
798 buffers: &[],
799 },
800 fragment: Some(wgpu::FragmentState {
801 module: &shader,
802 entry_point: Some("fs_main"),
803 compilation_options: wgpu::PipelineCompilationOptions::default(),
804 targets: &[Some(format.into())],
805 }),
806 primitive: wgpu::PrimitiveState::default(),
807 depth_stencil: None,
808 multisample: wgpu::MultisampleState::default(),
809 multiview_mask: None,
810 cache: None,
811 });
812
813 let mut readings = 0_u64;
814 for frame in 0..(RING_FRAMES * 3) {
815 ring.begin_frame(&device);
816 let mut encoder = device.create_command_encoder(&wgpu::CommandEncoderDescriptor {
817 label: Some("frust-gpu ring test frame"),
818 });
819 for (span, instances) in [(0_usize, 64_u32), (0, 64), (1, 64)] {
820 let mut pass = encoder.begin_render_pass(&wgpu::RenderPassDescriptor {
821 label: Some("frust-gpu ring test pass"),
822 color_attachments: &[Some(wgpu::RenderPassColorAttachment {
823 view: target.view(),
824 depth_slice: None,
825 resolve_target: None,
826 ops: wgpu::Operations {
827 load: wgpu::LoadOp::Clear(wgpu::Color::BLACK),
828 store: wgpu::StoreOp::Store,
829 },
830 })],
831 depth_stencil_attachment: None,
832 timestamp_writes: ring.pass_writes(span),
833 occlusion_query_set: None,
834 multiview_mask: None,
835 });
836 pass.set_pipeline(&pipeline);
837 pass.draw(0..3, 0..instances);
838 }
839 ring.resolve(&mut encoder);
840 queue.submit([encoder.finish()]);
841 ring.end_frame();
842 // The frame path never waits on a map; a test may, and does,
843 // so the harvest is deterministic rather than timing-dependent.
844 let _ = device.poll(wgpu::PollType::wait_indefinitely());
845 if ring.harvested() > readings {
846 readings = ring.harvested();
847 println!("frame {frame}: {:?}", ring.latest());
848 }
849 }
850
851 let reading = ring
852 .latest()
853 .expect("a TIMESTAMP_QUERY device must have produced at least one reading");
854 assert_eq!(reading.len(), 2);
855 assert!(
856 reading.span(0) > Duration::ZERO,
857 "span 0 covered two real render passes: {reading:?}"
858 );
859 assert!(
860 reading.span(1) > Duration::ZERO,
861 "span 1 covered a real render pass: {reading:?}"
862 );
863 assert_eq!(reading.total(), reading.span(0) + reading.span(1));
864 assert!(
865 reading.total() < Duration::from_millis(100),
866 "three 256x256 clears cannot plausibly take {:?}",
867 reading.total()
868 );
869 });
870 }
871}