Skip to main content

frust_gpu/
diag.rs

1//! GPU-side frame diagnostics: [`TimestampRing`], real GPU time per pass.
2//!
3//! A frame's `encode_us`/`submit_us` are CPU wall-clock spans around calls that
4//! only *record* and *queue* work — they say how long the CPU spent, never how
5//! long the GPU took. This module answers the second question directly, by
6//! having the GPU stamp its own clock at each pass boundary.
7//!
8//! # Pass boundaries only
9//!
10//! Every timestamp this ring takes is written through
11//! `wgpu::RenderPassDescriptor::timestamp_writes`, which needs only
12//! `wgpu::Features::TIMESTAMP_QUERY`. `CommandEncoder::write_timestamp` and
13//! `RenderPass::write_timestamp` are deliberately never used: they need
14//! `TIMESTAMP_QUERY_INSIDE_ENCODERS`/`TIMESTAMP_QUERY_INSIDE_PASSES`, which
15//! `wgpu` documents as unavailable on tile-based GPUs — Adreno, and Apple's own
16//! Metal families — i.e. exactly the mobile adapters this tier exists for. A
17//! probe that only works on desktop would answer the wrong question.
18//!
19//! # Inert unless the device really has the feature
20//!
21//! `TIMESTAMP_QUERY` is requested only by a `perf-trace` build, and only when
22//! the adapter offers it ([`crate::context`]'s `required_features`). This ring
23//! makes no assumption about that: it asks the **device** it is handed what it
24//! was created with, and when the answer is no it becomes inert — no query set,
25//! no buffers, no per-frame work, and [`TimestampRing::latest`] answering
26//! `None` forever. An inert ring is what a caller reports as `gpu_q=0`.
27//!
28//! # The ring
29//!
30//! Reading a timestamp back means mapping a buffer, which is only ready some
31//! frames after the submit that wrote it. So the ring holds [`RING_FRAMES`]
32//! independent slots — query set, resolve buffer, readback buffer — and a
33//! frame's slot is not read until the ring comes back around to it, by which
34//! time its map has long since completed. Nothing on the frame path ever
35//! blocks: the map is polled, never waited on, and a slot whose map has not
36//! landed simply goes untimed for that frame rather than stalling it.
37//!
38//! # A pass that draws nothing may measure nothing
39//!
40//! Metal samples a render pass's counters at the vertex/fragment *stage*
41//! boundaries, so a pass that runs neither stage — a pure clear, an empty
42//! pass — can leave its query pair unwritten, and the ring drops the pair
43//! rather than reporting a garbage span (see `span_duration`). This is a
44//! measurement gap, never a correctness one: an untimed pass draws exactly
45//! what it always did. A span made of several passes, which is the shape every
46//! caller here uses, absorbs it — the drawing passes still report.
47//!
48//! One frame's shape, in call order:
49//!
50//! 1. [`TimestampRing::begin_frame`] — advance to the next slot and harvest
51//!    whatever the frame that last used it left mapped.
52//! 2. [`TimestampRing::pass_writes`] once per timed pass, naming the span the
53//!    pass belongs to. Each call takes a *fresh* query pair, so several passes
54//!    may share one span and their durations add up.
55//! 3. [`TimestampRing::resolve`] into the same encoder the passes were recorded
56//!    into, before the caller submits it.
57//! 4. [`TimestampRing::end_frame`] after that submit (or
58//!    [`TimestampRing::abandon_frame`] when the frame was refused and nothing
59//!    was submitted).
60
61use std::sync::Arc;
62use std::sync::atomic::{AtomicU8, AtomicU32, Ordering};
63use std::time::Duration;
64
65/// How many frames-in-flight the ring keeps independent slots for.
66///
67/// Four is the depth a triple-buffered swapchain plus one in-flight submit can
68/// reach, so a slot's map has always completed by the time the ring returns to
69/// it — nothing on the frame path ever waits on a map.
70pub const RING_FRAMES: usize = 4;
71
72/// The largest number of named spans one frame can be split into.
73///
74/// A fixed ceiling rather than a `Vec` so [`GpuFrameSpans`] stays `Copy` and a
75/// caller can carry one through a frame record without allocating.
76pub const MAX_SPANS: usize = 8;
77
78/// How many individual passes one frame can time by default.
79///
80/// Each timed pass costs its own query pair, so this is what fixes the query
81/// set's `count` at `2 * PASSES`. A frame with more passes than this times the
82/// first `DEFAULT_PASS_CAPACITY` and leaves the rest untimed — draw-correct
83/// either way, since a pass with no `timestamp_writes` is an ordinary pass.
84pub const DEFAULT_PASS_CAPACITY: usize = 24;
85
86/// The hard ceiling on a slot's query count, from `wgpu`'s own
87/// `QUERY_SET_MAX_QUERIES`. Halved because every pass takes a *pair*.
88const MAX_PASS_CAPACITY: usize = (wgpu::QUERY_SET_MAX_QUERIES / 2) as usize;
89
90/// Bytes one resolved timestamp query occupies (`wgpu::QUERY_SIZE`).
91const QUERY_BYTES: u64 = wgpu::QUERY_SIZE as u64;
92
93/// A single pass span longer than this is discarded rather than reported.
94///
95/// A GPU pass measured in whole seconds is not a slow frame, it is a garbage
96/// tick pair: an unwritten query, a driver that reset its clock across a
97/// submit, or a counter wrap. Reporting it would poison every percentile
98/// computed off the raw series, so the span is dropped and its frame simply
99/// reads a little low.
100const IMPLAUSIBLE_SPAN_NANOS: u64 = 1_000_000_000;
101
102/// The `AtomicU8` states a slot's map callback moves through.
103mod map_state {
104    /// The map was requested and has not answered yet.
105    pub const PENDING: u8 = 0;
106    /// The map completed and the buffer holds this frame's resolved queries.
107    pub const READY: u8 = 1;
108    /// The map failed; the slot is recycled without producing a reading.
109    pub const FAILED: u8 = 2;
110}
111
112/// One frame's GPU span durations, in the caller's own span order.
113///
114/// The span *names* are the caller's business — this crate knows only how many
115/// there are and which index each pass was charged to. `frust-engine`'s
116/// `diag::EngineSpan` is the naming this repo's engine tier uses.
117#[derive(Debug, Clone, Copy, PartialEq, Eq)]
118pub struct GpuFrameSpans {
119    durations: [Duration; MAX_SPANS],
120    len: usize,
121}
122
123impl Default for GpuFrameSpans {
124    fn default() -> Self {
125        Self {
126            durations: [Duration::ZERO; MAX_SPANS],
127            len: 0,
128        }
129    }
130}
131
132impl GpuFrameSpans {
133    /// An all-zero reading over `len` spans (clamped to [`MAX_SPANS`]).
134    #[must_use]
135    pub fn zeroed(len: usize) -> Self {
136        Self {
137            durations: [Duration::ZERO; MAX_SPANS],
138            len: len.min(MAX_SPANS),
139        }
140    }
141
142    /// How many spans this reading covers.
143    #[must_use]
144    pub fn len(&self) -> usize {
145        self.len
146    }
147
148    /// Whether this reading covers no spans at all.
149    #[must_use]
150    pub fn is_empty(&self) -> bool {
151        self.len == 0
152    }
153
154    /// The duration charged to `index`, or [`Duration::ZERO`] past the end —
155    /// a span no pass was charged to is genuinely zero, so an out-of-range
156    /// index answering the same is the honest reading rather than a silent
157    /// failure mode.
158    #[must_use]
159    pub fn span(&self, index: usize) -> Duration {
160        self.durations.get(index).copied().unwrap_or(Duration::ZERO)
161    }
162
163    /// Every span in order.
164    #[must_use]
165    pub fn as_slice(&self) -> &[Duration] {
166        &self.durations[..self.len]
167    }
168
169    /// The sum of every span — the frame's total *attributed* GPU pass time.
170    ///
171    /// Deliberately a sum rather than last-end-minus-first-begin: the gaps
172    /// between a frame's passes are queue and driver time this ring cannot
173    /// attribute to any pass, and folding them into a "total" would report
174    /// work the breakdown below it does not account for.
175    #[must_use]
176    pub fn total(&self) -> Duration {
177        self.as_slice()
178            .iter()
179            .copied()
180            .fold(Duration::ZERO, |acc, span| acc.saturating_add(span))
181    }
182
183    /// Charges `span` with `duration` on top of whatever it already holds.
184    fn add(&mut self, span: usize, duration: Duration) {
185        if let Some(slot) = self.durations.get_mut(span) {
186            *slot = slot.saturating_add(duration);
187        }
188    }
189}
190
191/// One frame-in-flight's own query set and its two staging buffers.
192#[derive(Debug)]
193struct RingFrame {
194    queries: wgpu::QuerySet,
195    /// `QUERY_RESOLVE | COPY_SRC` — what `resolve_query_set` writes into. It
196    /// cannot also be `MAP_READ`, which is why the readback below exists.
197    resolve: wgpu::Buffer,
198    /// `COPY_DST | MAP_READ` — the mappable copy the harvest reads.
199    readback: wgpu::Buffer,
200    /// Next free pass pair, bumped by [`TimestampRing::pass_writes`].
201    ///
202    /// Atomic rather than a `Cell` so the ring stays `Sync`: a renderer that
203    /// records its passes off the thread that owns the ring is a shape this
204    /// type should not rule out, and the counter is uncontended in every
205    /// current caller anyway.
206    cursor: AtomicU32,
207    /// Which span each pass pair was charged to, indexed by pair.
208    owners: Vec<AtomicU8>,
209    state: SlotState,
210}
211
212/// Where one ring slot is in the record → resolve → map → harvest cycle.
213#[derive(Debug)]
214enum SlotState {
215    /// Holds nothing; free to record into.
216    Idle,
217    /// This frame's passes are being recorded into it.
218    Recording,
219    /// Submitted and mapping; `used` pass pairs are staged in `readback`.
220    Mapping { used: usize, status: Arc<AtomicU8> },
221}
222
223/// A ring of per-frame GPU timestamp query sets — see the module header.
224#[derive(Debug)]
225pub struct TimestampRing {
226    /// How many named spans a frame is split into.
227    spans: usize,
228    /// How many passes one frame can time.
229    passes: usize,
230    /// Nanoseconds per timestamp tick, from `queue.get_timestamp_period()`.
231    period_ns: f32,
232    /// Empty when inert — the one thing every method keys off.
233    frames: Vec<RingFrame>,
234    /// Which slot the frame in progress is recording into.
235    current: usize,
236    /// Whether the current frame's slot was actually free to record into.
237    armed: bool,
238    /// The most recently harvested reading.
239    latest: Option<GpuFrameSpans>,
240    /// How many frames have been harvested, so a consumer can tell a fresh
241    /// reading from the one it already reported.
242    harvested: u64,
243}
244
245impl TimestampRing {
246    /// A ring timing `spans` named spans per frame, over
247    /// [`DEFAULT_PASS_CAPACITY`] passes.
248    ///
249    /// Inert — no GPU resource created at all — when `device` was not created
250    /// with `wgpu::Features::TIMESTAMP_QUERY`, which is every build that did
251    /// not compile `perf-trace` in and every adapter that does not offer the
252    /// feature.
253    #[must_use]
254    pub fn new(device: &wgpu::Device, queue: &wgpu::Queue, spans: usize, label: &str) -> Self {
255        Self::with_capacity(device, queue, spans, DEFAULT_PASS_CAPACITY, label)
256    }
257
258    /// [`Self::new`] with an explicit per-frame pass capacity.
259    #[must_use]
260    pub fn with_capacity(
261        device: &wgpu::Device,
262        queue: &wgpu::Queue,
263        spans: usize,
264        passes: usize,
265        label: &str,
266    ) -> Self {
267        let spans = spans.min(MAX_SPANS);
268        let passes = passes.min(MAX_PASS_CAPACITY);
269        if spans == 0 || passes == 0 || !device.features().contains(wgpu::Features::TIMESTAMP_QUERY)
270        {
271            return Self::inert(spans);
272        }
273
274        // A non-finite or non-positive period would scale every tick delta
275        // into nonsense, and there is no sound reading to fall back on — an
276        // inert ring reporting nothing beats a ring reporting garbage.
277        let period_ns = queue.get_timestamp_period();
278        if !period_ns.is_finite() || period_ns <= 0.0 {
279            log::warn!(
280                "frust-gpu: timestamp period {period_ns} is unusable; GPU pass timing is off"
281            );
282            return Self::inert(spans);
283        }
284
285        let query_count = (passes * 2) as u32;
286        let staged_bytes = u64::from(query_count) * QUERY_BYTES;
287        // `resolve_query_set`'s destination offset is 256-byte aligned; the
288        // buffer is rounded up to the same grid so a future non-zero offset
289        // needs no re-sizing rule of its own.
290        let buffer_bytes = staged_bytes.next_multiple_of(wgpu::QUERY_RESOLVE_BUFFER_ALIGNMENT);
291
292        let frames = (0..RING_FRAMES)
293            .map(|slot| RingFrame {
294                queries: device.create_query_set(&wgpu::QuerySetDescriptor {
295                    label: Some(&format!("{label} queries {slot}")),
296                    ty: wgpu::QueryType::Timestamp,
297                    count: query_count,
298                }),
299                resolve: device.create_buffer(&wgpu::BufferDescriptor {
300                    label: Some(&format!("{label} resolve {slot}")),
301                    size: buffer_bytes,
302                    usage: wgpu::BufferUsages::QUERY_RESOLVE | wgpu::BufferUsages::COPY_SRC,
303                    mapped_at_creation: false,
304                }),
305                readback: device.create_buffer(&wgpu::BufferDescriptor {
306                    label: Some(&format!("{label} readback {slot}")),
307                    size: buffer_bytes,
308                    usage: wgpu::BufferUsages::COPY_DST | wgpu::BufferUsages::MAP_READ,
309                    mapped_at_creation: false,
310                }),
311                cursor: AtomicU32::new(0),
312                owners: (0..passes).map(|_| AtomicU8::new(0)).collect(),
313                state: SlotState::Idle,
314            })
315            .collect();
316
317        Self {
318            spans,
319            passes,
320            period_ns,
321            frames,
322            // The first `begin_frame` advances onto slot 0.
323            current: RING_FRAMES - 1,
324            armed: false,
325            latest: None,
326            harvested: 0,
327        }
328    }
329
330    /// A ring that measures nothing, for a device without `TIMESTAMP_QUERY`
331    /// and for a caller that wants the `gpu_q=0` shape without a GPU at all.
332    #[must_use]
333    pub fn inert(spans: usize) -> Self {
334        Self {
335            spans: spans.min(MAX_SPANS),
336            passes: 0,
337            period_ns: 0.0,
338            frames: Vec::new(),
339            current: 0,
340            armed: false,
341            latest: None,
342            harvested: 0,
343        }
344    }
345
346    /// Whether this ring holds real query sets, i.e. whether a frame line
347    /// should report `gpu_q=1`.
348    #[must_use]
349    pub fn is_active(&self) -> bool {
350        !self.frames.is_empty()
351    }
352
353    /// How many named spans a frame is split into.
354    #[must_use]
355    pub fn spans(&self) -> usize {
356        self.spans
357    }
358
359    /// Nanoseconds per timestamp tick, `0.0` on an inert ring.
360    #[must_use]
361    pub fn timestamp_period_ns(&self) -> f32 {
362        self.period_ns
363    }
364
365    /// The most recently harvested frame's spans, or `None` before the first
366    /// reading lands (and forever on an inert ring).
367    ///
368    /// The reading lags the CPU frame that asks for it by up to
369    /// [`RING_FRAMES`] frames — the price of never blocking on a map. In
370    /// steady state one fresh reading lands per frame, so the lag is a
371    /// constant offset rather than a gap in the series; [`Self::harvested`]
372    /// is how a consumer tells a repeat from a fresh reading.
373    #[must_use]
374    pub fn latest(&self) -> Option<GpuFrameSpans> {
375        self.latest
376    }
377
378    /// How many frame readings this ring has harvested.
379    #[must_use]
380    pub fn harvested(&self) -> u64 {
381        self.harvested
382    }
383
384    /// Advances onto the next slot and harvests whatever the frame that last
385    /// used it left mapped.
386    ///
387    /// Polls `device` once, without blocking: a map that has not landed leaves
388    /// its slot alone and the frame goes untimed rather than stalling on it.
389    pub fn begin_frame(&mut self, device: &wgpu::Device) {
390        if self.frames.is_empty() {
391            return;
392        }
393        self.current = (self.current + 1) % self.frames.len();
394
395        // One non-blocking poll, which is what actually invokes any map
396        // callback whose copy has completed. A poll failure is not worth
397        // reporting per frame: the slot simply stays pending and this frame
398        // goes untimed.
399        let _ = device.poll(wgpu::PollType::Poll);
400
401        let spans = self.spans;
402        let period_ns = self.period_ns;
403        let Some(slot) = self.frames.get_mut(self.current) else {
404            self.armed = false;
405            return;
406        };
407
408        if let SlotState::Mapping { used, status } = &slot.state {
409            match status.load(Ordering::Acquire) {
410                map_state::PENDING => {
411                    // Still in flight: leave the slot exactly as it is (its
412                    // readback buffer is mapped-in-progress and must not be
413                    // written again) and skip timing this frame.
414                    self.armed = false;
415                    return;
416                }
417                map_state::READY => {
418                    let reading = harvest(slot, *used, spans, period_ns);
419                    slot.readback.unmap();
420                    self.latest = Some(reading);
421                    self.harvested = self.harvested.saturating_add(1);
422                }
423                // A failed map leaves `wgpu`'s own map context marked, so the
424                // slot is unmapped anyway before it is recorded into again.
425                _ => slot.readback.unmap(),
426            }
427        }
428
429        slot.cursor.store(0, Ordering::Relaxed);
430        slot.state = SlotState::Recording;
431        self.armed = true;
432    }
433
434    /// The `timestamp_writes` for one pass belonging to `span`.
435    ///
436    /// Each call takes a fresh query pair, so a span made of several passes
437    /// (the frame's own strip passes, a layer's page rounds) sums correctly.
438    /// `None` — an untimed pass — when the ring is inert, the slot was not
439    /// free this frame, `span` is out of range, or the frame has already used
440    /// its pass capacity.
441    #[must_use]
442    pub fn pass_writes(&self, span: usize) -> Option<wgpu::RenderPassTimestampWrites<'_>> {
443        if !self.armed || span >= self.spans {
444            return None;
445        }
446        let slot = self.frames.get(self.current)?;
447        let pair = slot.cursor.fetch_add(1, Ordering::Relaxed) as usize;
448        if pair >= self.passes {
449            return None;
450        }
451        slot.owners.get(pair)?.store(span as u8, Ordering::Relaxed);
452        let first = (pair * 2) as u32;
453        Some(wgpu::RenderPassTimestampWrites {
454            query_set: &slot.queries,
455            beginning_of_pass_write_index: Some(first),
456            end_of_pass_write_index: Some(first + 1),
457        })
458    }
459
460    /// Records this frame's query resolve and its copy into the mappable
461    /// readback buffer, into the same encoder the timed passes were recorded
462    /// into. The caller submits that encoder.
463    ///
464    /// Only the pairs actually taken are resolved: an untouched query resolves
465    /// to whatever the driver left there, and reading it back would be
466    /// inventing a span rather than measuring one.
467    pub fn resolve(&self, encoder: &mut wgpu::CommandEncoder) {
468        if !self.armed {
469            return;
470        }
471        let Some(slot) = self.frames.get(self.current) else {
472            return;
473        };
474        let used = self.used_pairs(slot);
475        if used == 0 {
476            return;
477        }
478        let queries = (used * 2) as u32;
479        encoder.resolve_query_set(&slot.queries, 0..queries, &slot.resolve, 0);
480        encoder.copy_buffer_to_buffer(
481            &slot.resolve,
482            0,
483            &slot.readback,
484            0,
485            u64::from(queries) * QUERY_BYTES,
486        );
487    }
488
489    /// Requests the map that a later [`Self::begin_frame`] harvests. Call
490    /// after submitting the encoder [`Self::resolve`] was recorded into.
491    pub fn end_frame(&mut self) {
492        if !self.armed {
493            return;
494        }
495        self.armed = false;
496        let Some(slot) = self.frames.get_mut(self.current) else {
497            return;
498        };
499        let used = (slot.cursor.load(Ordering::Relaxed) as usize).min(self.passes);
500        if used == 0 {
501            slot.state = SlotState::Idle;
502            return;
503        }
504        let status = Arc::new(AtomicU8::new(map_state::PENDING));
505        let callback = Arc::clone(&status);
506        let bytes = (used * 2) as u64 * QUERY_BYTES;
507        slot.readback
508            .slice(0..bytes)
509            .map_async(wgpu::MapMode::Read, move |result| {
510                let state = if result.is_ok() {
511                    map_state::READY
512                } else {
513                    map_state::FAILED
514                };
515                callback.store(state, Ordering::Release);
516            });
517        slot.state = SlotState::Mapping { used, status };
518    }
519
520    /// Closes a frame that was refused before its encoder was submitted.
521    ///
522    /// Nothing was recorded and nothing will be, so the slot goes straight
523    /// back to idle without a map — mapping a buffer no copy ever reached
524    /// would harvest the previous frame's ticks as if they were this one's.
525    pub fn abandon_frame(&mut self) {
526        if !self.armed {
527            return;
528        }
529        self.armed = false;
530        if let Some(slot) = self.frames.get_mut(self.current) {
531            slot.cursor.store(0, Ordering::Relaxed);
532            slot.state = SlotState::Idle;
533        }
534    }
535
536    /// How many pass pairs the frame in progress actually took.
537    fn used_pairs(&self, slot: &RingFrame) -> usize {
538        (slot.cursor.load(Ordering::Relaxed) as usize).min(self.passes)
539    }
540}
541
542/// Reads `used` resolved query pairs out of `slot`'s mapped readback buffer and
543/// charges each one to the span its pass named.
544fn harvest(slot: &RingFrame, used: usize, spans: usize, period_ns: f32) -> GpuFrameSpans {
545    let mut reading = GpuFrameSpans::zeroed(spans);
546    let bytes = (used * 2) as u64 * QUERY_BYTES;
547    let mapped =
548        slot.readback.slice(0..bytes).get_mapped_range().expect(
549            "frust-gpu diag: the harvested slot's readback is mapped over the resolved range",
550        );
551    let stride = (QUERY_BYTES * 2) as usize;
552    for pair in 0..used {
553        let at = pair * stride;
554        let Some(chunk) = mapped.get(at..at + stride) else {
555            break;
556        };
557        let (begin, end) = tick_pair(chunk);
558        let Some(duration) = span_duration(begin, end, period_ns) else {
559            continue;
560        };
561        let owner = slot
562            .owners
563            .get(pair)
564            .map_or(usize::MAX, |span| span.load(Ordering::Relaxed) as usize);
565        reading.add(owner, duration);
566    }
567    drop(mapped);
568    reading
569}
570
571/// The `(begin, end)` tick pair a 16-byte resolved chunk holds, little-endian
572/// as `wgpu` resolves it. A short chunk answers `(0, 0)`, which
573/// [`span_duration`] then rejects — the one shape that cannot be a real span.
574fn tick_pair(chunk: &[u8]) -> (u64, u64) {
575    fn tick(bytes: Option<&[u8]>) -> u64 {
576        bytes
577            .and_then(|slice| <[u8; 8]>::try_from(slice).ok())
578            .map_or(0, u64::from_le_bytes)
579    }
580    (tick(chunk.get(0..8)), tick(chunk.get(8..16)))
581}
582
583/// One pass's duration from its tick pair, or `None` when the pair cannot be a
584/// real measurement.
585///
586/// Rejected: a non-increasing pair (an unwritten query, or a counter that
587/// wrapped across the pass) and anything past [`IMPLAUSIBLE_SPAN_NANOS`] (see
588/// that constant). Both are dropped rather than clamped — a clamped garbage
589/// value is still reported as a measurement.
590fn span_duration(begin: u64, end: u64, period_ns: f32) -> Option<Duration> {
591    if end <= begin || !period_ns.is_finite() || period_ns <= 0.0 {
592        return None;
593    }
594    let nanos = (end - begin) as f64 * f64::from(period_ns);
595    if !nanos.is_finite() || nanos >= IMPLAUSIBLE_SPAN_NANOS as f64 {
596        return None;
597    }
598    Some(Duration::from_nanos(nanos as u64))
599}
600
601#[cfg(test)]
602mod tests {
603    use super::*;
604
605    #[test]
606    fn zeroed_reading_reports_its_span_count_and_no_time() {
607        let reading = GpuFrameSpans::zeroed(4);
608        assert_eq!(reading.len(), 4);
609        assert!(!reading.is_empty());
610        assert_eq!(reading.as_slice().len(), 4);
611        assert_eq!(reading.total(), Duration::ZERO);
612        for index in 0..4 {
613            assert_eq!(reading.span(index), Duration::ZERO);
614        }
615    }
616
617    #[test]
618    fn span_count_is_clamped_to_the_ceiling() {
619        assert_eq!(GpuFrameSpans::zeroed(MAX_SPANS + 9).len(), MAX_SPANS);
620    }
621
622    #[test]
623    fn an_index_past_the_end_reads_zero_rather_than_panicking() {
624        let reading = GpuFrameSpans::zeroed(2);
625        assert_eq!(reading.span(7), Duration::ZERO);
626        assert_eq!(reading.span(usize::MAX), Duration::ZERO);
627    }
628
629    #[test]
630    fn several_passes_charged_to_one_span_add_up() {
631        let mut reading = GpuFrameSpans::zeroed(3);
632        reading.add(1, Duration::from_micros(400));
633        reading.add(1, Duration::from_micros(250));
634        reading.add(2, Duration::from_micros(90));
635        assert_eq!(reading.span(0), Duration::ZERO);
636        assert_eq!(reading.span(1), Duration::from_micros(650));
637        assert_eq!(reading.span(2), Duration::from_micros(90));
638        assert_eq!(reading.total(), Duration::from_micros(740));
639    }
640
641    #[test]
642    fn a_charge_past_the_span_ceiling_is_dropped_not_wrapped() {
643        let mut reading = GpuFrameSpans::zeroed(2);
644        reading.add(MAX_SPANS, Duration::from_secs(1));
645        assert_eq!(reading.total(), Duration::ZERO);
646    }
647
648    #[test]
649    fn span_duration_scales_ticks_by_the_period() {
650        // 1000 ticks at 1ns/tick, and the same 1000 ticks on an adapter whose
651        // tick is 38.4ns (a real Adreno-class period) — the same pair must
652        // read as two different durations, which is the whole point of
653        // scaling by the queue's own period rather than assuming nanoseconds.
654        assert_eq!(span_duration(0, 1_000, 1.0), Some(Duration::from_micros(1)));
655        assert_eq!(
656            span_duration(500, 1_500, 38.4),
657            Some(Duration::from_nanos(38_400))
658        );
659    }
660
661    #[test]
662    fn a_non_increasing_tick_pair_is_no_measurement() {
663        assert_eq!(span_duration(0, 0, 1.0), None);
664        assert_eq!(span_duration(900, 900, 1.0), None);
665        assert_eq!(span_duration(1_000, 999, 1.0), None);
666    }
667
668    #[test]
669    fn an_implausible_span_is_dropped_rather_than_reported() {
670        assert_eq!(span_duration(0, IMPLAUSIBLE_SPAN_NANOS, 1.0), None);
671        assert_eq!(span_duration(0, u64::MAX, 1.0), None);
672        // Just under the ceiling still reads.
673        assert!(span_duration(0, IMPLAUSIBLE_SPAN_NANOS - 1, 1.0).is_some());
674    }
675
676    #[test]
677    fn an_unusable_period_yields_no_measurement() {
678        assert_eq!(span_duration(0, 1_000, 0.0), None);
679        assert_eq!(span_duration(0, 1_000, -1.0), None);
680        assert_eq!(span_duration(0, 1_000, f32::NAN), None);
681        assert_eq!(span_duration(0, 1_000, f32::INFINITY), None);
682    }
683
684    #[test]
685    fn a_tick_pair_reads_little_endian_and_a_short_chunk_reads_zero() {
686        let mut chunk = [0_u8; 16];
687        chunk[..8].copy_from_slice(&7_u64.to_le_bytes());
688        chunk[8..].copy_from_slice(&19_u64.to_le_bytes());
689        assert_eq!(tick_pair(&chunk), (7, 19));
690        assert_eq!(tick_pair(&chunk[..12]), (7, 0));
691        assert_eq!(tick_pair(&[]), (0, 0));
692    }
693
694    #[test]
695    fn an_inert_ring_measures_nothing_and_never_reports_a_reading() {
696        // The `gpu_q=0` shape a device without TIMESTAMP_QUERY produces,
697        // exercised with no GPU in the loop at all.
698        let mut ring = TimestampRing::inert(4);
699        assert!(!ring.is_active());
700        assert_eq!(ring.spans(), 4);
701        assert_eq!(ring.timestamp_period_ns(), 0.0);
702        assert!(ring.pass_writes(0).is_none());
703        assert!(ring.latest().is_none());
704        ring.end_frame();
705        ring.abandon_frame();
706        assert!(ring.latest().is_none());
707        assert_eq!(ring.harvested(), 0);
708    }
709
710    /// A full-screen triangle drawn by the real-adapter test below, so each
711    /// timed pass runs the vertex and fragment stages Metal samples its
712    /// counters at — see that test's own comment.
713    const RING_TEST_WGSL: &str = "\
714@vertex
715fn vs_main(@builtin(vertex_index) index: u32) -> @builtin(position) vec4<f32> {
716    let uv = vec2<f32>(f32((index << 1u) & 2u), f32(index & 2u));
717    return vec4<f32>(uv * 2.0 - 1.0, 0.0, 1.0);
718}
719
720@fragment
721fn fs_main() -> @location(0) vec4<f32> {
722    return vec4<f32>(0.2, 0.4, 0.8, 1.0);
723}
724";
725
726    /// Blocks on `future` by polling it to completion — `wgpu`'s native
727    /// adapter and device requests resolve without an executor driving them,
728    /// and this crate has no async runtime of its own.
729    fn block_on<F: std::future::Future>(future: F) -> F::Output {
730        use std::task::{Context, Poll, Waker};
731
732        let waker = Waker::noop();
733        let mut cx = Context::from_waker(waker);
734        let mut future = std::pin::pin!(future);
735        loop {
736            match future.as_mut().poll(&mut cx) {
737                Poll::Ready(value) => return value,
738                Poll::Pending => std::thread::yield_now(),
739            }
740        }
741    }
742
743    #[test]
744    #[ignore = "needs a real GPU adapter offering TIMESTAMP_QUERY; run with \
745                `cargo test -p frust-gpu -- --ignored` (pin the adapter on a \
746                multi-GPU host with WGPU_BACKEND / WGPU_ADAPTER_NAME)"]
747    fn a_real_adapter_reports_plausible_per_pass_gpu_time() {
748        block_on(async {
749            let instance = wgpu::Instance::new(
750                wgpu::InstanceDescriptor::new_without_display_handle_from_env(),
751            );
752            let adapter = wgpu::util::initialize_adapter_from_env_or_default(&instance, None)
753                .await
754                .expect("no compatible GPU adapter");
755            println!("frust-gpu timestamp-ring adapter: {:?}", adapter.get_info());
756            if !adapter.features().contains(wgpu::Features::TIMESTAMP_QUERY) {
757                println!("adapter offers no TIMESTAMP_QUERY — the inert path is the whole story");
758                return;
759            }
760            let (device, queue) = adapter
761                .request_device(&wgpu::DeviceDescriptor {
762                    label: Some("frust-gpu timestamp ring test device"),
763                    required_features: wgpu::Features::TIMESTAMP_QUERY,
764                    required_limits: wgpu::Limits::default(),
765                    ..Default::default()
766                })
767                .await
768                .expect("failed to create the device");
769
770            // Two spans, three passes charged across them: two into span 0 and
771            // one into span 1, so the reading is per-span rather than per-pass
772            // and both spans must come back nonzero.
773            //
774            // Every pass DRAWS. A draw-less pass is not a shortcut here: Metal
775            // samples a render pass's counters at the vertex/fragment stage
776            // boundaries (`setStartOfVertexSampleIndex`/
777            // `setEndOfFragmentSampleIndex`), so a pass that runs neither stage
778            // can leave its query pair unwritten — the case
779            // [`span_duration`] drops rather than reports.
780            let mut ring = TimestampRing::new(&device, &queue, 2, "frust-gpu ring test");
781            assert!(
782                ring.is_active(),
783                "a TIMESTAMP_QUERY device must arm the ring"
784            );
785            let format = wgpu::TextureFormat::Rgba8Unorm;
786            let target = crate::HeadlessTarget::new(&device, 512, 512, format);
787            let shader = device.create_shader_module(wgpu::ShaderModuleDescriptor {
788                label: Some("frust-gpu ring test shader"),
789                source: wgpu::ShaderSource::Wgsl(RING_TEST_WGSL.into()),
790            });
791            let pipeline = device.create_render_pipeline(&wgpu::RenderPipelineDescriptor {
792                label: Some("frust-gpu ring test pipeline"),
793                layout: None,
794                vertex: wgpu::VertexState {
795                    module: &shader,
796                    entry_point: Some("vs_main"),
797                    compilation_options: wgpu::PipelineCompilationOptions::default(),
798                    buffers: &[],
799                },
800                fragment: Some(wgpu::FragmentState {
801                    module: &shader,
802                    entry_point: Some("fs_main"),
803                    compilation_options: wgpu::PipelineCompilationOptions::default(),
804                    targets: &[Some(format.into())],
805                }),
806                primitive: wgpu::PrimitiveState::default(),
807                depth_stencil: None,
808                multisample: wgpu::MultisampleState::default(),
809                multiview_mask: None,
810                cache: None,
811            });
812
813            let mut readings = 0_u64;
814            for frame in 0..(RING_FRAMES * 3) {
815                ring.begin_frame(&device);
816                let mut encoder = device.create_command_encoder(&wgpu::CommandEncoderDescriptor {
817                    label: Some("frust-gpu ring test frame"),
818                });
819                for (span, instances) in [(0_usize, 64_u32), (0, 64), (1, 64)] {
820                    let mut pass = encoder.begin_render_pass(&wgpu::RenderPassDescriptor {
821                        label: Some("frust-gpu ring test pass"),
822                        color_attachments: &[Some(wgpu::RenderPassColorAttachment {
823                            view: target.view(),
824                            depth_slice: None,
825                            resolve_target: None,
826                            ops: wgpu::Operations {
827                                load: wgpu::LoadOp::Clear(wgpu::Color::BLACK),
828                                store: wgpu::StoreOp::Store,
829                            },
830                        })],
831                        depth_stencil_attachment: None,
832                        timestamp_writes: ring.pass_writes(span),
833                        occlusion_query_set: None,
834                        multiview_mask: None,
835                    });
836                    pass.set_pipeline(&pipeline);
837                    pass.draw(0..3, 0..instances);
838                }
839                ring.resolve(&mut encoder);
840                queue.submit([encoder.finish()]);
841                ring.end_frame();
842                // The frame path never waits on a map; a test may, and does,
843                // so the harvest is deterministic rather than timing-dependent.
844                let _ = device.poll(wgpu::PollType::wait_indefinitely());
845                if ring.harvested() > readings {
846                    readings = ring.harvested();
847                    println!("frame {frame}: {:?}", ring.latest());
848                }
849            }
850
851            let reading = ring
852                .latest()
853                .expect("a TIMESTAMP_QUERY device must have produced at least one reading");
854            assert_eq!(reading.len(), 2);
855            assert!(
856                reading.span(0) > Duration::ZERO,
857                "span 0 covered two real render passes: {reading:?}"
858            );
859            assert!(
860                reading.span(1) > Duration::ZERO,
861                "span 1 covered a real render pass: {reading:?}"
862            );
863            assert_eq!(reading.total(), reading.span(0) + reading.span(1));
864            assert!(
865                reading.total() < Duration::from_millis(100),
866                "three 256x256 clears cannot plausibly take {:?}",
867                reading.total()
868            );
869        });
870    }
871}