Skip to main content

gpu_handle_types/
sync.rs

1// SPDX-License-Identifier: MIT OR Apache-2.0
2
3use core::future::Future;
4use core::pin::Pin;
5use std::any::Any;
6use std::ffi::c_void;
7use std::sync::Arc;
8use std::time::Duration;
9
10#[cfg(feature = "wgpu")]
11use crate::SliceOutcome;
12use crate::{BackendKind, Error, GlBackend};
13
14/// Boxed dyn-future return shape used by `SyncWaiter::wait_async` so the
15/// trait stays object-safe through `Arc<dyn SyncWaiter>`. Native
16/// `async fn` on the trait would be RPITIT and dyn-incompatible by
17/// default; callers dispatch through `Arc<dyn SyncWaiter>`, so boxing the
18/// future is mandatory.
19// `Send` off wasm; dropped on wasm, where the `SyncWaiter`
20// the future borrows is thread-affine (`&dyn SyncWaiter` is not `Send`
21// once the waiter is `!Sync`). An explicit `+ Send` on a `dyn Future`
22// cannot be spelled with a non-auto marker trait, so the alias is
23// cfg-split directly.
24#[cfg(not(target_family = "wasm"))]
25pub type BoxFuture<'a, T> = Pin<Box<dyn Future<Output = T> + Send + 'a>>;
26#[cfg(target_family = "wasm")]
27pub type BoxFuture<'a, T> = Pin<Box<dyn Future<Output = T> + 'a>>;
28
29/// Default wait budget for `SyncPoint::wait()` and
30/// `SyncPoint::wait_blocking()` — 10 seconds.
31pub const DEFAULT_WAIT_TIMEOUT: Duration = Duration::from_secs(10);
32
33/// A far-future but always-representable wait cap. `Instant + Duration` overflows
34/// the platform's representable range only for a `timeout` so large it is
35/// effectively infinite (near `Duration::MAX` — an `Instant` spans hundreds of
36/// billions of years), and `Instant` has no `saturating_add`. Rather than let such
37/// a `timeout` silently drop its deadline — turning a bounded wait into a hang —
38/// [`wait_deadline`] clamps to `now +` this. It sits far beyond any real GPU wait
39/// (budgets are seconds), so a legitimately huge finite wait still resolves against
40/// it as intended.
41const SATURATED_WAIT_CAP: Duration = Duration::from_secs(60 * 60 * 24 * 365); // ~1 year
42
43/// The absolute deadline a `wait(timeout)` poll loop should honour, or `None` for
44/// the [`SyncWaiter::wait`] contract's `Duration::MAX` "wait forever" sentinel.
45///
46/// - `timeout == Duration::MAX` → `None`: the caller asked to wait forever, so the
47///   loop runs with no deadline (matching every backend's explicit `Duration::MAX`
48///   handling, which maps to the native "infinite": `vkWaitSemaphores(UINT64_MAX)`,
49///   `WaitForSingleObject(INFINITE)`, `cuEventSynchronize`, …).
50/// - any other `timeout` → `Some(now + timeout)`, **saturated** to a bounded,
51///   always-representable [`Instant`](std::time::Instant) on `Instant + Duration`
52///   overflow — clamped to `now + 1 year`, far beyond any real GPU wait budget.
53///
54/// This keeps the two meanings of "no deadline" distinct: a returned `None`
55/// encodes ONLY the explicit forever sentinel, never an accidental `checked_add`
56/// overflow — which would otherwise silently convert a bounded wait into an
57/// unbounded hang.
58pub fn wait_deadline(timeout: Duration) -> Option<std::time::Instant> {
59    if timeout == Duration::MAX {
60        return None;
61    }
62    let now = std::time::Instant::now();
63    Some(now.checked_add(timeout).unwrap_or_else(|| now.checked_add(SATURATED_WAIT_CAP).unwrap_or(now)))
64}
65
66/// Backend-specific wait dispatch + lifetime anchor for a [`SyncPoint`].
67///
68/// Every non-trivial `SyncPoint` variant carries an `Arc<dyn SyncWaiter>`
69/// that:
70///
71/// 1. **Owns the underlying primitive's lifetime** — the timeline
72///    semaphore, D3D12 fence, `MTLSharedEvent`, `cl_event`, `GLsync`,
73///    etc. Cloning the `SyncPoint` clones the `Arc`; the primitive lives
74///    as long as any clone outlives. This rules out the "consumer holds
75///    a `*mut c_void` after the producer dropped" footgun.
76///
77/// 2. **Provides the wait implementation** — `SyncPoint::wait` /
78///    `is_signaled` / `backend` dispatch through this trait, with no
79///    global registries and no `OnceLock<fn>` runtime callbacks.
80///
81/// Implementations live in the producer crate (typically an interop
82/// layer's per-backend module) and are constructed at the same site that
83/// mints the `SyncPoint`.
84// `MaybeSendSync` = `Send + Sync` off wasm; empty on wasm,
85// where a waiter can hold a thread-affine `Arc<wgpu::Device>` /
86// `web_sys` handle. See [`crate::MaybeSendSync`].
87pub trait SyncWaiter: crate::MaybeSendSync + 'static {
88    /// Block until the GPU work this sync point represents has
89    /// completed, or until `timeout` elapses.
90    ///
91    /// `Duration::MAX` means "wait forever" — implementations should
92    /// translate this to the platform's native "infinite" sentinel
93    /// (`UINT64_MAX` for `vkWaitSemaphores`, `INFINITE` for
94    /// `WaitForSingleObject`, etc.).
95    ///
96    /// # Timeout granularity
97    ///
98    /// Native wait APIs vary in granularity, and implementations
99    /// preserve the caller's intent at the expense of *requested*
100    /// (not realised) latency on sub-API-tick timeouts:
101    ///
102    /// - **Metal** (`MTLSharedEvent::waitUntilSignaledValue:timeoutMS:`)
103    ///   accepts millisecond-granular `u64`. Sub-millisecond positive
104    ///   timeouts (e.g. `Duration::from_micros(100)`) round **up**
105    ///   to 1 ms — truncation to 0 would silently behave as
106    ///   "do not wait". A high-frequency progress poll using
107    ///   `Duration::from_micros(100)` therefore observes ~10× the
108    ///   requested latency on a not-yet-signaled event. For
109    ///   non-blocking probes use [`SyncWaiter::is_signaled`] (which
110    ///   passes `Duration::ZERO` and short-circuits on the
111    ///   `signaledValue() >= value` accessor — driver-side, no
112    ///   timer).
113    /// - **Win32** (`WaitForSingleObject`) is also millisecond-
114    ///   granular; the same round-up applies.
115    /// - **Vulkan** (`vkWaitSemaphores`), **D3D12 fence**, **CUDA
116    ///   external semaphore**, **OpenCL** are nanosecond-granular and
117    ///   honour the caller's `Duration` exactly.
118    ///
119    /// Callers that need *both* a deterministic poll cadence under
120    /// 1 ms *and* a real wait fallback should compose the two —
121    /// e.g. `is_signaled` in a hot loop with their own `Instant`
122    /// budget, then a single coarse `wait(Duration::from_millis(N))`
123    /// when the budget is exhausted.
124    fn wait(&self, timeout: Duration) -> Result<(), Error>;
125
126    /// Async sibling of [`wait`](Self::wait). Returns a boxed future so
127    /// the trait stays object-safe through `Arc<dyn SyncWaiter>` — native
128    /// async-fn-in-trait is RPITIT and dyn-incompatible by default; callers
129    /// dispatch through `Arc<dyn SyncWaiter>` everywhere, so the explicit
130    /// `Pin<Box<...>>` is mandatory.
131    ///
132    /// # Default implementation
133    ///
134    /// Backends without a hand-tuned implementation get a two-phase
135    /// hybrid: a short cooperative-yield spin (catches sub-ms waits
136    /// without burning a thread) followed by an iteration-bounded
137    /// `is_signaled` probe loop. Per-backend impls SHOULD override with
138    /// their native blocking-with-timeout primitive on a dedicated
139    /// waiter thread for waits the spin loop didn't catch.
140    ///
141    /// The default implementation is correct (resolves when signalled,
142    /// returns `Error::Timeout` after `timeout`, never blocks the
143    /// executor) but not optimal: it busy-polls inside the spin loop
144    /// and falls back to a `Duration::ZERO` probe loop after that. For
145    /// production paths, override with a backend-specific impl.
146    fn wait_async<'a>(&'a self, timeout: Duration) -> BoxFuture<'a, Result<(), Error>> {
147        Box::pin(async move {
148            // Fast-path yield-spin — bounded cooperative-yield probe. Caps iteration
149            // count rather than wall-clock time so executors that re-
150            // poll immediately on `wake_by_ref` (pollster) don't burn
151            // CPU here.
152            const SPIN_ITERATIONS: usize = 64;
153            // `None` iff `timeout == Duration::MAX` (the wait-forever sentinel); every
154            // finite timeout gets a bounded, always-representable deadline — an
155            // `Instant + Duration` overflow cannot silently drop it and hang.
156            let deadline = wait_deadline(timeout);
157            for _ in 0..SPIN_ITERATIONS {
158                match self.is_signaled() {
159                    Ok(true) => return Ok(()),
160                    Ok(false) => {}
161                    Err(e) => return Err(e),
162                }
163                if let Some(d) = deadline
164                    && std::time::Instant::now() >= d
165                {
166                    return Err(Error::Timeout);
167                }
168                yield_once().await;
169            }
170            // Waiter-thread fallback — default impl probes `is_signaled` in a
171            // yielding loop. Per-backend impls SHOULD override to hand
172            // off to a dedicated waiter thread with the native
173            // blocking-with-timeout primitive.
174            loop {
175                match self.is_signaled() {
176                    Ok(true) => return Ok(()),
177                    Ok(false) => {}
178                    Err(e) => return Err(e),
179                }
180                if let Some(d) = deadline
181                    && std::time::Instant::now() >= d
182                {
183                    return Err(Error::Timeout);
184                }
185                yield_once().await;
186            }
187        })
188    }
189
190    /// Non-blocking probe.
191    ///
192    /// - `Ok(true)` — the sync point has been reached.
193    /// - `Ok(false)` — work is still in flight (the underlying poll
194    ///   timed out at zero).
195    /// - `Err(_)` — driver-level failure (device lost / TDR /
196    ///   `wgpu::PollError::WrongSubmissionIndex` / equivalent on other
197    ///   backends). Callers polling in a loop **must** break on `Err` —
198    ///   folding errors into `Ok(false)` would loop forever on a TDR'd
199    ///   device.
200    ///
201    /// Default implementation calls `wait(Duration::ZERO)` and maps
202    /// `Err(Error::Timeout)` → `Ok(false)`; every other `Err` is
203    /// propagated. Backends override when they have a more direct
204    /// "is this primitive currently signaled" probe (e.g.
205    /// `vkGetSemaphoreCounterValue`, `ID3D12Fence::GetCompletedValue`)
206    /// that distinguishes "not signaled" from real errors without
207    /// going through the wait path.
208    fn is_signaled(&self) -> Result<bool, Error> {
209        match self.wait(Duration::ZERO) {
210            Ok(()) => Ok(true),
211            Err(Error::Timeout) => Ok(false),
212            Err(other) => Err(other),
213        }
214    }
215
216    /// Backend identity for routing decisions on the consumer side.
217    fn backend(&self) -> BackendKind;
218
219    /// Downcast hook for cross-API bridges that need access to the
220    /// concrete waiter type — e.g. a Vulkan→CUDA bridge that wants to
221    /// pull the `VkDevice` out of the `VulkanWaiter` to issue its own
222    /// `cuImportExternalSemaphore`. Most callers use the per-variant
223    /// raw fields on `SyncPoint` directly and never need this.
224    fn as_any(&self) -> &dyn Any;
225
226    /// [`CudaEventWaiter`] view of this waiter, when the concrete type
227    /// implements it. Default `None`.
228    ///
229    /// This is the trait-object route to the CUDA-only
230    /// [`CudaEventWaiter::wait_on_foreign_stream`] extension: `Any` can
231    /// only downcast to *concrete* types, which consumers in other
232    /// crates cannot name — so producers whose waiter implements
233    /// [`CudaEventWaiter`] override this with `Some(self)` and
234    /// consumers (e.g. a CUDA import path gating its private copy stream
235    /// on a producer's `SyncPoint::CudaEvent`) reach the extension method
236    /// without knowing the concrete type.
237    fn as_cuda_event_waiter(&self) -> Option<&dyn CudaEventWaiter> {
238        None
239    }
240
241    /// Rebuild this waiter bound to `value` instead of the value it was
242    /// constructed with, reusing the same underlying primitive (and the
243    /// same lifetime anchors).
244    ///
245    /// # Why this exists
246    ///
247    /// [`wait`](Self::wait) / [`is_signaled`](Self::is_signaled) take no
248    /// value argument — the waiter *embeds* the value it resolves at,
249    /// captured when it was built. So a [`SyncPoint`] whose `value`
250    /// field was substituted while its `waiter` was cloned verbatim has
251    /// two surfaces that disagree: the GPU side (a consumer reading
252    /// `SyncPoint::*.value` to stage a queue wait) waits the new value,
253    /// while the CPU side silently resolves at the old one — reporting
254    /// "already signalled" for a signal the producer has not emitted.
255    ///
256    /// Any code that mints a `SyncPoint` at a value other than the one a
257    /// template was built with MUST route the waiter through this method
258    /// rather than cloning it, so the two surfaces cannot diverge.
259    ///
260    /// # Contract
261    ///
262    /// - `Some(w)` — `w` waits `value` on the same primitive this waiter
263    ///   waits on, and holds the same keep-alive chain. Implementations
264    ///   MUST NOT carry over state that is only valid for the original
265    ///   value (e.g. a captured submission index that retires at it).
266    /// - `None` (the default) — the underlying primitive carries no
267    ///   value, so there is nothing to rebind: binary semaphores,
268    ///   fences, `GLsync`, `cl_semaphore_khr`. A `None` here is not a
269    ///   failure; it means the CPU surface has no value to disagree
270    ///   about.
271    fn rebind_to_value(&self, value: u64) -> Option<Arc<dyn SyncWaiter>> {
272        let _ = value;
273        None
274    }
275}
276
277/// CUDA-event-specific extension trait for the
278/// [`SyncPoint::CudaEvent`] variant. Supertrait of [`SyncWaiter`] so a
279/// `CudaEventWaiter` always satisfies the generic [`SyncPoint::wait`]
280/// dispatch path; adds the CUDA-only `wait_on_foreign_stream` extension
281/// method that cross-API bridges downcast to via
282/// [`SyncWaiter::as_any`].
283///
284/// # Drop discipline
285///
286/// Implementations MUST push the event's owning `CUcontext` before
287/// calling `cuEventDestroy_v2`, and MUST pop iff the push succeeded
288/// (pop-without-push would silently consume whatever the calling
289/// thread had on top of its context stack). `cuEventDestroy_v2`
290/// requires the owning context to still exist; the push is defensive
291/// against driver-internal cleanup paths that probe `cuCtxGetCurrent`
292/// and emit confusing diagnostics when no context is current.
293///
294/// # Cross-context wait
295///
296/// `cuEventRecord` is strict-same-context — a `CUevent` cannot be
297/// recorded against a stream from a different `CUcontext`. But
298/// `cuStreamWaitEvent` is cross-context (NVIDIA Driver API contract).
299/// `wait_on_foreign_stream` is the cross-context wait entry point:
300/// the consumer's CUDA stream may live in a different `CUcontext`
301/// than the event, and the waiter issues `cuStreamWaitEvent` so the
302/// foreign stream gates on this event's completion without a CPU
303/// bounce.
304///
305/// A CUDA-driver-backed waiter (e.g. one built on the `mini-cuda`
306/// driver loader) implements this trait concretely; this trait only
307/// defines the contract.
308pub trait CudaEventWaiter: SyncWaiter {
309    /// Issue `cuStreamWaitEvent(foreign_stream, self.event, 0)` so
310    /// `foreign_stream` gates on this event's completion. The stream
311    /// may belong to a different `CUcontext` than the event's owning
312    /// context — `cuStreamWaitEvent` is cross-context per the NVIDIA
313    /// Driver API.
314    ///
315    /// `foreign_stream` is a raw `CUstream` pointer (the null pointer
316    /// resolves to the calling thread's default stream).
317    ///
318    /// Returns [`Error::NotSupported`] if the CUDA driver loader is
319    /// unavailable or the wait dispatch returns a driver error; the
320    /// concrete error wording is implementation-defined.
321    fn wait_on_foreign_stream(&self, foreign_stream: *mut c_void) -> Result<(), Error>;
322}
323
324/// Discriminator for [`SyncPoint::Vulkan`]'s `VkSemaphore` flavour.
325///
326/// `VkSemaphore` ships in two shapes — binary (one-shot, signal-and-
327/// reset) and timeline (monotonic 64-bit payload, multi-waiter,
328/// concurrent-signal-safe). A consumer staging a wait must route
329/// through `VkTimelineSemaphoreSubmitInfo` for timeline semaphores and
330/// must NOT for binary semaphores — passing the wrong shape risks
331/// validation errors, driver-quirk-dependent wrong waits, and (on
332/// already-consumed binary signals) hard deadlock at submit time.
333///
334/// The Vulkan API has no public probe that safely distinguishes the
335/// two (`vkGetSemaphoreCounterValue` is undefined on binaries per
336/// `VUID-vkGetSemaphoreCounterValue-semaphore-03255`), so the producer
337/// must declare intent at mint time.
338#[derive(Copy, Clone, Debug, PartialEq, Eq)]
339#[non_exhaustive]
340pub enum VulkanSemaphoreKind {
341    /// `vk::SemaphoreType::TIMELINE`. `value` is a payload coordinate;
342    /// any number of waiters may target the same value, and signals
343    /// with strictly-increasing payloads can be queued concurrently.
344    Timeline,
345    /// `vk::SemaphoreType::BINARY`. Single-shot — exactly one signal
346    /// must pair with exactly one wait. The `value` field is ignored
347    /// (set to `0` by convention). Producers carrying binaries here
348    /// MUST guarantee the wait is the only consumer of the signal.
349    Binary,
350}
351
352/// Shared, mutate-once slot the producer fills in after the caller
353/// submits the encoded work. Used by [`SyncPoint::DeferredWgpu`].
354///
355/// The slot is initialised to `None` at mint time and transitions to
356/// `Some(idx)` exactly once via [`Self::set`]; subsequent calls are
357/// rejected with [`Error::InvalidArgument`] (single-shot semantics).
358#[cfg(feature = "wgpu")]
359#[derive(Debug, Default)]
360pub struct DeferredWgpuSlot {
361    cell: std::sync::OnceLock<wgpu::SubmissionIndex>,
362}
363
364#[cfg(feature = "wgpu")]
365impl DeferredWgpuSlot {
366    pub fn new() -> Self {
367        Self::default()
368    }
369
370    /// Commit the post-submit `SubmissionIndex`. Single-shot — a second
371    /// call returns [`Error::InvalidArgument`] without overwriting the
372    /// committed value.
373    pub fn set(&self, idx: wgpu::SubmissionIndex) -> Result<(), Error> {
374        self.cell.set(idx).map_err(|_| Error::InvalidArgument("DeferredWgpuSlot already committed".into()))
375    }
376
377    /// Return the committed `SubmissionIndex`, or `None` if the producer
378    /// has not yet submitted.
379    pub fn get(&self) -> Option<&wgpu::SubmissionIndex> {
380        self.cell.get()
381    }
382}
383
384// Defensive — the `OnceLock<wgpu::SubmissionIndex>` design relies on
385// `SubmissionIndex` being `Clone + Send + Sync`. If a wgpu upgrade
386// silently dropped any of these bounds, the OnceLock-storing
387// `DeferredWgpuSlot` would surface a confusing trait-bound error far
388// from the cause; this assertion fails compilation here, with a clear
389// message, instead.
390#[cfg(feature = "wgpu")]
391static_assertions::assert_impl_all!(wgpu::SubmissionIndex: Clone, Send, Sync);
392
393/// GPU sync primitive. Produced by every submit-side path; consumed by
394/// anything that needs to serialise on a prior GPU submission.
395///
396/// Cross-backend bridges are deliberately absent from this leaf crate —
397/// those belong in an interop layer (such as `wgpu-interop`), where the
398/// target-device context is available.
399///
400/// # Lifetime
401///
402/// The per-variant `waiter: Arc<dyn SyncWaiter>` field anchors the
403/// lifetime of any raw primitive the variant exposes (`*mut c_void`
404/// fence pointers, `u64` semaphore handles, `GLsync` opaques, etc.).
405/// Cloning a `SyncPoint` is `Arc::clone` on the waiter — cheap, no
406/// driver round-trip. The raw fields are guaranteed to remain valid
407/// for as long as any clone is alive.
408#[derive(Clone)]
409#[non_exhaustive]
410pub enum SyncPoint {
411    /// Vulkan semaphore. `device` is the `VkDevice` the semaphore was
412    /// created on, exposed for bridges that need to re-import it. The
413    /// `waiter` owns an `Arc<TimelineSemaphore>` (or equivalent) that
414    /// anchors the `VkSemaphore`'s lifetime.
415    ///
416    /// `kind` distinguishes timeline vs binary semaphores —
417    /// see [`VulkanSemaphoreKind`]. `value` is the timeline payload
418    /// coordinate when `kind == Timeline`; for `kind == Binary` the
419    /// field is meaningless (set to `0` by convention) and consumers
420    /// route through wgpu-hal's binary-wait path
421    /// (`add_wait_semaphore(_, None, _)`).
422    Vulkan { semaphore: u64, kind: VulkanSemaphoreKind, value: u64, device: *mut c_void, waiter: Arc<dyn SyncWaiter> },
423    /// D3D12 fence sync. `fence` is an `ID3D12Fence*` (COM pointer);
424    /// `value` is the monotonic signal value the consumer waits on
425    /// (`fence->GetCompletedValue() >= value`). The `waiter` owns the
426    /// `Arc<ExportFence>` (or equivalent) that holds the COM ref.
427    ///
428    /// `value` is monotonically signaled by the producer. Bridges that
429    /// reuse a single shared fence across calls serialise the
430    /// (mint + `Signal`) pair under a value-lock so concurrent bridges
431    /// always sequence in monotonic order.
432    /// A consumer waiting on `value=N` is guaranteed to pass once a
433    /// later `value=M >= N` has been signaled, even if the producer's
434    /// counter has since advanced past `M`.
435    D3D12 { fence: *mut c_void, value: u64, waiter: Arc<dyn SyncWaiter> },
436    /// D3D11 keyed-mutex sync. `keyed_mutex` is an `IDXGIKeyedMutex*`;
437    /// `key` is the integer key the producer released to. The
438    /// `waiter` owns the `Arc<...>` that anchors the COM ref.
439    D3D11 { keyed_mutex: *mut c_void, key: u64, waiter: Arc<dyn SyncWaiter> },
440    /// CUDA external semaphore. `event` is a `CUexternalSemaphore`;
441    /// `value` is `Some(v)` for timeline waits — including the products
442    /// of Vulkan→CUDA / D3D12→CUDA bridges, which carry the producer's
443    /// timeline value so the consumer can enqueue
444    /// `cuWaitExternalSemaphoresAsync` on whichever CUDA stream their
445    /// work runs on. `None` is reserved for setup-metadata SyncPoints
446    /// where the consumer mints the actual signal value at emission time
447    /// and doesn't need a resolvable wait at construction.
448    Cuda { event: *mut c_void, value: Option<u64>, waiter: Arc<dyn SyncWaiter> },
449    /// Intra-CUDA event. `event` is a `CUevent` recorded on the
450    /// producing stream via `cuEventRecord`. Consumers wait via
451    /// `cuStreamWaitEvent` (chained CUDA streams — cross-context
452    /// supported per NVIDIA Driver API) or `cuEventSynchronize`
453    /// (blocking CPU wait).
454    ///
455    /// Distinct from [`Self::Cuda`] — that variant carries a
456    /// `CUexternalSemaphore` waited via `cuWaitExternalSemaphoresAsync`.
457    /// `CUevent` and `CUexternalSemaphore` are different opaque types
458    /// in the CUDA Driver API; passing a `CUevent` into the external-
459    /// semaphore wait path returns `CUDA_ERROR_INVALID_HANDLE`.
460    ///
461    /// Cross-API consumers cannot wait on a `CUevent` directly; they
462    /// need a `CUexternalSemaphore` (the [`Self::Cuda`] shape); an
463    /// interop layer performs that conversion when the chain spans a
464    /// backend boundary.
465    ///
466    /// `context` is the event's **owning** `CUcontext` — bound at
467    /// `cuEventCreate` time per the NVIDIA Driver API ("creates an event
468    /// for the current context"). The owning context must still exist
469    /// at destroy time or `cuEventDestroy_v2` returns
470    /// `CUDA_ERROR_INVALID_CONTEXT`. There is no `cuEventGetCtx` API —
471    /// the metadata must travel with the event.
472    ///
473    /// `device` is the CUDA ordinal the owning context was created on
474    /// (derivable from `context` via `cuCtxGetDevice` but cheap to
475    /// carry and matches the device-keyed bridge shape used by
476    /// [`Self::Cuda`]).
477    ///
478    /// `waiter` implements [`CudaEventWaiter`] — a supertrait of
479    /// [`SyncWaiter`] that adds the CUDA-specific
480    /// `wait_on_foreign_stream` extension method and the push-owning-
481    /// context-before-destroy Drop discipline. The leaf crate stores
482    /// the upcasted `Arc<dyn SyncWaiter>` so the generic [`Self::waiter`]
483    /// dispatch path is uniform with every other variant; cross-API
484    /// bridges that need the CUDA extension methods downcast through
485    /// [`SyncWaiter::as_any`].
486    CudaEvent { event: *mut c_void, context: *mut c_void, device: i32, waiter: Arc<dyn SyncWaiter> },
487    /// OpenCL sync. `event` is either a `cl_event` (binary,
488    /// signaled-on-completion; `value` is `None`) or a
489    /// `cl_semaphore_khr` aliasing a timeline (`value` is the
490    /// caller-signaled timeline value). The `waiter` owns the
491    /// `Arc<...>` that holds the producing context's lifetime.
492    OpenCl { event: *mut c_void, value: Option<u64>, waiter: Arc<dyn SyncWaiter> },
493    /// Metal shared event. `event` is an `MTLSharedEvent*`; `value` is
494    /// the signal value the consumer waits on. The `waiter` owns the
495    /// `Arc<MTLSharedEvent>` keeping the event alive.
496    Metal { event: *mut c_void, value: u64, waiter: Arc<dyn SyncWaiter> },
497    /// OpenGL semaphore (`GLuint` produced by `glGenSemaphoresEXT`,
498    /// imported from a cross-API OS handle via
499    /// `glImportSemaphoreFdEXT` / `glImportSemaphoreWin32HandleEXT`).
500    ///
501    /// `value` is `Some(_)` for timeline-equivalent semaphores
502    /// (D3D12_FENCE_EXT handle types); `None` for plain binary
503    /// (OPAQUE_FD / OPAQUE_WIN32) — the GL driver latches the D3D12
504    /// fence value at wait/signal time via
505    /// `glSemaphoreParameterui64vEXT(sem, GL_D3D12_FENCE_VALUE_EXT, &v)`.
506    ///
507    /// `context` is the caller's `EGLContext` / `HGLRC` / `CGLContextObj`
508    /// this semaphore was created against. Waits must be dispatched on
509    /// a thread where the same context is current.
510    OpenGL { semaphore: u32, value: Option<u64>, context: *mut c_void, waiter: Arc<dyn SyncWaiter> },
511    /// GL fence sync. `glsync` is a `GLsync` opaque pointer produced
512    /// by `glFenceSync(GL_SYNC_GPU_COMMANDS_COMPLETE, 0)`.
513    ///
514    /// Distinct from [`Self::OpenGL`] — that one carries an *imported*
515    /// GL semaphore (`GLuint`); this one is the raw GL-native fence.
516    ///
517    /// `backend` records the [`GlBackend`] flavour of the GL context that
518    /// produced this `GLsync`. Cross-API consumers (an OpenGL→X bridge)
519    /// route through it to pick the matching loader (`wglGetProcAddress`
520    /// vs `eglGetProcAddress`) when resolving `glClientWaitSync` for the
521    /// CPU-fallback path.
522    /// Without it the consumer would hard-code `Desktop`, which silently
523    /// resolves the wrong proc-address on EGL / ANGLE / Android hosts and
524    /// stalls forever on a never-signalled handle.
525    OpenGLSync { glsync: *mut c_void, backend: GlBackend, waiter: Arc<dyn SyncWaiter> },
526    /// `wgpu::Queue::submit` return value. Wait dispatches through the
527    /// `waiter`, which holds an `Arc<wgpu::Device>` and calls
528    /// `device.poll(PollType::Wait { submission_index: Some(..), .. })`.
529    #[cfg(feature = "wgpu")]
530    Wgpu { submission_index: wgpu::SubmissionIndex, waiter: Arc<dyn SyncWaiter> },
531    /// `wgpu::Queue::submit` index *not yet known* at SyncPoint mint
532    /// time. Used where the caller owns the encoder and submits after the
533    /// producer has returned; `slot.set(idx)` is called once after
534    /// `queue.submit(...)`, and until then the waiter falls back to
535    /// `device.poll(PollType::Wait { submission_index: None, .. })`
536    /// (full device drain). After commit, waits resolve precisely on
537    /// the recorded `SubmissionIndex`.
538    #[cfg(feature = "wgpu")]
539    DeferredWgpu { slot: Arc<DeferredWgpuSlot>, device: Arc<wgpu::Device>, waiter: Arc<dyn SyncWaiter> },
540    /// CPU-only work — no GPU sync needed. `wait` returns `Ok(())`
541    /// immediately.
542    Cpu,
543    /// No-op sync — used for pipelines where the consumer side handles
544    /// ordering through a separate channel (e.g. SurfaceControl
545    /// transactions). Behaves identically to `Cpu` for waits.
546    Noop,
547}
548
549// SAFETY: the `*mut c_void` / `*mut wgpu...` raw fields are immutable
550// after construction and dereferenced only by the per-variant `waiter`,
551// whose `SyncWaiter: Send + Sync` bound asserts platform-correct sharing
552// for whatever the pointer references. Per-variant doc-comments specify
553// the additional caller-asserted preconditions (e.g. D3D11 requires the
554// device's multithread-protect flag).
555unsafe impl Send for SyncPoint {}
556unsafe impl Sync for SyncPoint {}
557
558impl SyncPoint {
559    /// Async wait. Default 10-second timeout.
560    ///
561    /// Async is the default. Use [`wait_blocking`]
562    /// when the caller is on a synchronous code path (sync-loop
563    /// renderers, test harnesses, JNI dispatcher Drop paths).
564    ///
565    /// [`wait_blocking`]: Self::wait_blocking
566    pub async fn wait(&self) -> Result<(), Error> {
567        self.wait_with_timeout_async(DEFAULT_WAIT_TIMEOUT).await
568    }
569
570    /// Async wait with explicit timeout.
571    pub async fn wait_with_timeout_async(&self, timeout: Duration) -> Result<(), Error> {
572        // Binary Vulkan CPU-wait rejection.
573        if let Some(err) = self.binary_vk_cpu_wait_error() {
574            return Err(err);
575        }
576        match self.waiter() {
577            Some(w) => w.wait_async(timeout).await,
578            None => Ok(()),
579        }
580    }
581
582    /// Synchronous wait with the default 10-second timeout.
583    pub fn wait_blocking(&self) -> Result<(), Error> {
584        self.wait_with_timeout(DEFAULT_WAIT_TIMEOUT)
585    }
586
587    /// Synchronous wait with an explicit timeout.
588    ///
589    /// Timeout granularity is backend-dependent — see
590    /// [`SyncWaiter::wait`] for the per-backend rounding contract.
591    /// In particular, Metal and Win32 are millisecond-granular and
592    /// round sub-ms positive timeouts up to 1 ms; Vulkan / D3D12 /
593    /// CUDA / OpenCL are nanosecond-granular. For sub-ms polls use
594    /// [`Self::is_signaled`] in a loop with your own `Instant` budget.
595    pub fn wait_with_timeout(&self, timeout: Duration) -> Result<(), Error> {
596        // Binary Vulkan CPU-wait rejection — `vkWaitSemaphores` is
597        // undefined on binary semaphores, and the Khronos spec gives
598        // binary semaphores no CPU-side wait entry point. Producers
599        // that need CPU-waitable binary semaphores MUST attach a
600        // fence-equivalent fallback (a `wgpu::SubmissionIndex` from the
601        // same submit). When the fallback is absent, the backend's waiter
602        // fails both `wait_blocking` and the async `wait()` with
603        // `Error::NotSupported` (see `binary_vk_cpu_wait_error`).
604        if let Some(err) = self.binary_vk_cpu_wait_error() {
605            return Err(err);
606        }
607        match self.waiter() {
608            Some(w) => w.wait(timeout),
609            None => Ok(()),
610        }
611    }
612
613    /// Leaf-crate hook for rejecting a CPU wait on a binary
614    /// `SyncPoint::Vulkan` whose producer did not attach a
615    /// fence-equivalent fallback. Always `None`: whether the fallback is
616    /// attached is visible only to the backend's waiter, which is where
617    /// that half of the binary-Vulkan CPU-wait contract is enforced.
618    fn binary_vk_cpu_wait_error(&self) -> Option<Error> {
619        // The fence-equivalent fallback lives on the per-backend Vulkan
620        // waiter impl. From the leaf crate's perspective we only see
621        // `Arc<dyn SyncWaiter>` — the contract is that `wait` /
622        // `wait_async` on a binary semaphore without fallback return
623        // `Error::NotSupported`. Backends MUST honour that contract; we
624        // cannot enforce the "fallback present" check here, so this
625        // returns `None` rather than gate behaviour it can't observe.
626        let _ = matches!(self, Self::Vulkan { kind: VulkanSemaphoreKind::Binary, .. });
627        None
628    }
629
630    /// Sequence two sync points. Returns a `SyncPoint` whose wait
631    /// completes only after both `self` and `then` have completed.
632    ///
633    /// Implemented by a small adapter waiter that
634    /// drives both inner waits in sequence; cheap (one heap alloc for
635    /// the adapter `Arc`).
636    pub fn chain(self, then: SyncPoint) -> SyncPoint {
637        // Trivial cases — skip the adapter entirely.
638        if matches!(self, Self::Cpu | Self::Noop) {
639            return then;
640        }
641        if matches!(then, Self::Cpu | Self::Noop) {
642            return self;
643        }
644        let backend = self.backend();
645        let waiter: Arc<dyn SyncWaiter> = Arc::new(ChainWaiter { first: self, then, backend });
646        // The chain adapter looks like an opaque waiter; we expose it
647        // through the `Cuda` shape (see `with_chain_waiter`) only because
648        // every variant needs *some* per-variant carrier. The event/value
649        // fields are unused — only the `waiter` is consulted.
650        //
651        // A dedicated `SyncPoint::Chained` variant would be cleaner.
652        // Callers should treat the result as opaque (only use
653        // `.wait()`, `.is_signaled()`, `.backend()`).
654        Self::Cpu // intentionally fall back; ChainWaiter is used below
655            .with_chain_waiter(waiter)
656    }
657
658    /// Internal helper to splice a `ChainWaiter` into a placeholder
659    /// `SyncPoint::Cpu`. Used only by `chain`.
660    fn with_chain_waiter(self, waiter: Arc<dyn SyncWaiter>) -> SyncPoint {
661        // Encode the chain waiter as a `Cuda` variant with a sentinel
662        // event pointer. The variant choice is arbitrary — consumers
663        // dispatch on `.waiter().backend()` for routing decisions, so
664        // the variant tag is invisible.
665        //
666        // This is a pragmatic placeholder in lieu of a dedicated
667        // `SyncPoint::Chained` variant.
668        let _ = self;
669        Self::Cuda { event: core::ptr::null_mut(), value: None, waiter }
670    }
671}
672
673// ─────────────────────────────────────────────────────────────────────────────
674// Raw-handle reconstruction (cross-C / cross-instance transport carriers)
675// ─────────────────────────────────────────────────────────────────────────────
676
677/// A [`SyncWaiter`] for a [`SyncPoint`] **reconstructed from a raw foreign fence
678/// handle** that has not yet been imported onto a waiting device.
679///
680/// A D3D12 fence / Vulkan semaphore / Metal shared event that crossed an API or
681/// process boundary as a raw handle (e.g. a `#[repr(C)]` descriptor over a C ABI)
682/// carries no device the leaf crate can drive a wait on — the real CPU/GPU wait
683/// only becomes possible once a consumer **imports** the handle onto its own device
684/// (the importer stages the device-side `vkWaitSemaphores` /
685/// `ID3D12CommandQueue::Wait` / `encodeWaitForEvent` off the carried raw fields).
686///
687/// So a `SyncPoint` reconstructed via [`SyncPoint::from_raw_d3d12_fence`] /
688/// [`SyncPoint::from_raw_vulkan_semaphore`] / [`SyncPoint::from_raw_metal_event`] is
689/// a **transport carrier**: its raw fields are consumed by the importer, and its
690/// `waiter` is this carrier. Calling [`SyncPoint::wait_blocking`] / `wait` /
691/// `is_signaled` on the carrier (before import) returns [`Error::NotSupported`] —
692/// fail-closed, never a silent "already signaled" — because the leaf crate cannot
693/// wait a fence it has no device for. Import the handle first; the import path
694/// produces the real, waitable `SyncPoint`.
695#[derive(Debug)]
696pub struct RawFenceCarrierWaiter {
697    backend: BackendKind,
698}
699
700impl RawFenceCarrierWaiter {
701    fn new(backend: BackendKind) -> Self {
702        Self { backend }
703    }
704
705    fn not_imported() -> Error {
706        Error::NotSupported(std::borrow::Cow::Borrowed(
707            "SyncPoint reconstructed from a raw foreign fence handle is a transport \
708                 carrier — import it onto a device before waiting",
709        ))
710    }
711}
712
713impl SyncWaiter for RawFenceCarrierWaiter {
714    fn wait(&self, _timeout: Duration) -> Result<(), Error> {
715        Err(Self::not_imported())
716    }
717
718    fn is_signaled(&self) -> Result<bool, Error> {
719        Err(Self::not_imported())
720    }
721
722    fn backend(&self) -> BackendKind {
723        self.backend
724    }
725
726    fn as_any(&self) -> &dyn Any {
727        self
728    }
729}
730
731impl SyncPoint {
732    /// Reconstruct a [`SyncPoint::D3D12`] transport carrier from a raw
733    /// `ID3D12Fence*` pointer + signal value that crossed an API / process boundary
734    /// (e.g. a `#[repr(C)]` fence descriptor over a C ABI).
735    ///
736    /// `fence` is an `ID3D12Fence*` COM pointer the caller guarantees is live for
737    /// the carrier's lifetime; `value` is the monotonic value the consumer waits for
738    /// (`fence->GetCompletedValue() >= value`). The returned `SyncPoint` is a
739    /// **carrier** ([`RawFenceCarrierWaiter`]) — its raw `(fence, value)` fields feed
740    /// a device-side import (e.g. an interop layer's acquire-sync path); a CPU
741    /// `wait()` on it before import returns [`Error::NotSupported`].
742    pub fn from_raw_d3d12_fence(fence: *mut c_void, value: u64) -> SyncPoint {
743        let waiter: Arc<dyn SyncWaiter> = Arc::new(RawFenceCarrierWaiter::new(BackendKind::D3D12));
744        SyncPoint::D3D12 { fence, value, waiter }
745    }
746
747    /// Reconstruct a [`SyncPoint::Vulkan`] transport carrier from a raw
748    /// `VkSemaphore` + `VkDevice` that crossed an API / process boundary.
749    ///
750    /// `semaphore` is the raw `VkSemaphore` (a 64-bit non-dispatchable handle);
751    /// `kind` distinguishes timeline vs binary (see [`VulkanSemaphoreKind`]);
752    /// `value` is the timeline coordinate (ignored for `Binary`); `device` is the
753    /// `VkDevice*` the semaphore lives on (needed by an importer that re-exports it
754    /// for a cross-device wait). The returned `SyncPoint` is a **carrier** — its raw
755    /// fields feed a device-side import; a CPU `wait()` before import returns
756    /// [`Error::NotSupported`].
757    pub fn from_raw_vulkan_semaphore(
758        semaphore: u64,
759        kind: VulkanSemaphoreKind,
760        value: u64,
761        device: *mut c_void,
762    ) -> SyncPoint {
763        let waiter: Arc<dyn SyncWaiter> = Arc::new(RawFenceCarrierWaiter::new(BackendKind::Vulkan));
764        SyncPoint::Vulkan { semaphore, kind, value, device, waiter }
765    }
766
767    /// Reconstruct a [`SyncPoint::Metal`] transport carrier from a raw
768    /// `MTLSharedEvent*` + signal value that crossed an API / process boundary.
769    ///
770    /// `event` is an `MTLSharedEvent*` the caller guarantees is live for the
771    /// carrier's lifetime; `value` is the value the consumer waits for. The returned
772    /// `SyncPoint` is a **carrier** — its raw `(event, value)` fields feed a
773    /// device-side import; a CPU `wait()` before import returns
774    /// [`Error::NotSupported`].
775    pub fn from_raw_metal_event(event: *mut c_void, value: u64) -> SyncPoint {
776        let waiter: Arc<dyn SyncWaiter> = Arc::new(RawFenceCarrierWaiter::new(BackendKind::Metal));
777        SyncPoint::Metal { event, value, waiter }
778    }
779}
780
781/// Per-process waiter thread for [`SyncPoint::DeferredWgpu`].
782///
783/// The waiter thread is stateless — every per-instance datum (device
784/// handle, deferred slot) travels inside the `SliceFn` closure — so one
785/// thread serves every deferred sync point in the process.
786#[cfg(feature = "wgpu")]
787fn deferred_wgpu_waiter_thread() -> &'static crate::WaiterThread {
788    static THREAD: std::sync::OnceLock<crate::WaiterThread> = std::sync::OnceLock::new();
789    THREAD.get_or_init(|| crate::WaiterThread::new("wgpu-deferred"))
790}
791
792/// Backend-internal waiter for [`SyncPoint::DeferredWgpu`]. Until the
793/// producer commits a `SubmissionIndex` via [`DeferredWgpuSlot::set`],
794/// `wait` falls back to `device.poll(PollType::Wait { submission_index:
795/// None, .. })` (full device drain). After commit, it passes
796/// `submission_index: Some(idx)` and resolves on that precise index.
797///
798/// `wait_async` routes through [`run_hybrid_wait`] — the fast-path
799/// yield-spin spins on `is_signaled` for sub-millisecond signals, the
800/// waiter-thread fallback hands off to the
801/// per-process waiter thread which issues blocking-with-timeout slices
802/// of [`WAITER_SLICE`] on a dedicated thread.
803#[cfg(feature = "wgpu")]
804pub(crate) struct DeferredWgpuWaiter {
805    slot: Arc<DeferredWgpuSlot>,
806    device: Arc<wgpu::Device>,
807}
808
809#[cfg(feature = "wgpu")]
810impl DeferredWgpuWaiter {
811    pub(crate) fn new(slot: Arc<DeferredWgpuSlot>, device: Arc<wgpu::Device>) -> Self {
812        Self { slot, device }
813    }
814
815    /// Single `device.poll(Wait { .. })` issuance keyed off the
816    /// current commit state. Shared between `wait`, `is_signaled`, and
817    /// the per-slice closure produced by `wait_async`.
818    fn poll_slice(&self, timeout: Option<Duration>) -> SliceOutcome {
819        let submission_index = self.slot.get().cloned();
820        match self.device.poll(wgpu::PollType::Wait { submission_index, timeout }) {
821            Ok(_) => SliceOutcome::Signaled,
822            Err(wgpu::PollError::Timeout) => SliceOutcome::TimedOut,
823            Err(e) => {
824                log::warn!("SyncPoint::DeferredWgpu: device.poll returned {e:?}");
825                SliceOutcome::Failed(Error::NotSupported("SyncPoint::DeferredWgpu: device.poll failed".into()))
826            }
827        }
828    }
829}
830
831#[cfg(feature = "wgpu")]
832impl SyncWaiter for DeferredWgpuWaiter {
833    fn wait(&self, timeout: Duration) -> Result<(), Error> {
834        let timeout_arg = if timeout == Duration::MAX { None } else { Some(timeout) };
835        match self.poll_slice(timeout_arg) {
836            SliceOutcome::Signaled => Ok(()),
837            SliceOutcome::TimedOut => Err(Error::Timeout),
838            SliceOutcome::Failed(e) => Err(e),
839        }
840    }
841
842    fn wait_async<'a>(&'a self, timeout: Duration) -> BoxFuture<'a, Result<(), Error>> {
843        let slot = self.slot.clone();
844        let device = self.device.clone();
845        let make_slice = move || -> crate::SliceFn {
846            Box::new(move |slice: Duration| -> SliceOutcome {
847                let submission_index = slot.get().cloned();
848                match device.poll(wgpu::PollType::Wait { submission_index, timeout: Some(slice) }) {
849                    Ok(_) => SliceOutcome::Signaled,
850                    Err(wgpu::PollError::Timeout) => SliceOutcome::TimedOut,
851                    Err(e) => {
852                        log::warn!("SyncPoint::DeferredWgpu::wait_async: device.poll returned {e:?}");
853                        SliceOutcome::Failed(Error::NotSupported(
854                            "SyncPoint::DeferredWgpu::wait_async: device.poll failed".into(),
855                        ))
856                    }
857                }
858            })
859        };
860        Box::pin(crate::run_hybrid_wait(move || self.is_signaled(), deferred_wgpu_waiter_thread(), timeout, make_slice))
861    }
862
863    fn is_signaled(&self) -> Result<bool, Error> {
864        match self.poll_slice(Some(Duration::ZERO)) {
865            SliceOutcome::Signaled => Ok(true),
866            SliceOutcome::TimedOut => Ok(false),
867            SliceOutcome::Failed(e) => Err(e),
868        }
869    }
870
871    fn backend(&self) -> BackendKind {
872        BackendKind::Wgpu
873    }
874
875    fn as_any(&self) -> &dyn Any {
876        self
877    }
878}
879
880/// Construct a [`SyncPoint::DeferredWgpu`] complete with its waiter.
881///
882/// For producers that hand out a sync point *before* they submit: the
883/// returned SyncPoint's clones all share the same `Arc<DeferredWgpuSlot>`,
884/// so the producer-side `slot.set(idx)` at submit time is visible to every
885/// clone.
886#[cfg(feature = "wgpu")]
887pub fn make_deferred_wgpu_sync_point(slot: Arc<DeferredWgpuSlot>, device: Arc<wgpu::Device>) -> SyncPoint {
888    let waiter: Arc<dyn SyncWaiter> = Arc::new(DeferredWgpuWaiter::new(slot.clone(), device.clone()));
889    SyncPoint::DeferredWgpu { slot, device, waiter }
890}
891
892struct ChainWaiter {
893    first: SyncPoint,
894    then: SyncPoint,
895    backend: BackendKind,
896}
897
898impl SyncWaiter for ChainWaiter {
899    fn wait(&self, timeout: Duration) -> Result<(), Error> {
900        let start = std::time::Instant::now();
901        self.first.wait_with_timeout(timeout)?;
902        let elapsed = start.elapsed();
903        let remaining = timeout.saturating_sub(elapsed);
904        self.then.wait_with_timeout(remaining)
905    }
906
907    fn wait_async<'a>(&'a self, timeout: Duration) -> BoxFuture<'a, Result<(), Error>> {
908        Box::pin(async move {
909            let start = std::time::Instant::now();
910            self.first.wait_with_timeout_async(timeout).await?;
911            let elapsed = start.elapsed();
912            let remaining = timeout.saturating_sub(elapsed);
913            self.then.wait_with_timeout_async(remaining).await
914        })
915    }
916
917    fn is_signaled(&self) -> Result<bool, Error> {
918        Ok(self.first.is_signaled()? && self.then.is_signaled()?)
919    }
920
921    fn backend(&self) -> BackendKind {
922        self.backend
923    }
924
925    fn as_any(&self) -> &dyn Any {
926        self
927    }
928}
929
930impl SyncPoint {
931    /// Non-blocking probe.
932    ///
933    /// - `Ok(true)`  — sync point reached (or trivial `Cpu`/`Noop`).
934    /// - `Ok(false)` — work still in flight.
935    /// - `Err(_)`    — driver-level failure (device lost / TDR /
936    ///   wrong-submission-index). See [`SyncWaiter::is_signaled`]
937    ///   for the rationale on surfacing errors instead of folding
938    ///   them into `false`.
939    pub fn is_signaled(&self) -> Result<bool, Error> {
940        match self.waiter() {
941            Some(w) => w.is_signaled(),
942            None => Ok(true),
943        }
944    }
945
946    /// Backend identity. For the trivial variants (`Cpu`, `Noop`) this
947    /// returns [`BackendKind::Cpu`]; otherwise it dispatches to the
948    /// per-variant waiter.
949    pub fn backend(&self) -> BackendKind {
950        match self.waiter() {
951            Some(w) => w.backend(),
952            None => BackendKind::Cpu,
953        }
954    }
955
956    /// Borrow the per-variant waiter, if any. `Cpu` and `Noop` return
957    /// `None`.
958    pub fn waiter(&self) -> Option<&Arc<dyn SyncWaiter>> {
959        match self {
960            Self::Vulkan { waiter, .. }
961            | Self::D3D12 { waiter, .. }
962            | Self::D3D11 { waiter, .. }
963            | Self::Cuda { waiter, .. }
964            | Self::CudaEvent { waiter, .. }
965            | Self::OpenCl { waiter, .. }
966            | Self::Metal { waiter, .. }
967            | Self::OpenGL { waiter, .. }
968            | Self::OpenGLSync { waiter, .. } => Some(waiter),
969            #[cfg(feature = "wgpu")]
970            Self::Wgpu { waiter, .. } => Some(waiter),
971            #[cfg(feature = "wgpu")]
972            Self::DeferredWgpu { waiter, .. } => Some(waiter),
973            Self::Cpu | Self::Noop => None,
974        }
975    }
976}
977
978impl core::fmt::Debug for SyncPoint {
979    fn fmt(&self, f: &mut core::fmt::Formatter<'_>) -> core::fmt::Result {
980        match self {
981            Self::Vulkan { semaphore, kind, value, .. } => f
982                .debug_struct("Vulkan")
983                .field("semaphore", semaphore)
984                .field("kind", kind)
985                .field("value", value)
986                .finish(),
987            Self::D3D12 { value, .. } => f.debug_struct("D3D12").field("value", value).finish(),
988            Self::D3D11 { key, .. } => f.debug_struct("D3D11").field("key", key).finish(),
989            Self::Cuda { value, .. } => f.debug_struct("Cuda").field("value", value).finish(),
990            Self::CudaEvent { event, device, .. } => {
991                f.debug_struct("CudaEvent").field("event", &(*event as usize)).field("device", device).finish()
992            }
993            Self::OpenCl { event, value, .. } => {
994                f.debug_struct("OpenCl").field("event", &(*event as usize)).field("value", value).finish()
995            }
996            Self::Metal { value, .. } => f.debug_struct("Metal").field("value", value).finish(),
997            Self::OpenGL { semaphore, value, .. } => {
998                f.debug_struct("OpenGL").field("semaphore", semaphore).field("value", value).finish()
999            }
1000            Self::OpenGLSync { glsync, backend, .. } => {
1001                f.debug_struct("OpenGLSync").field("glsync", &(*glsync as usize)).field("backend", backend).finish()
1002            }
1003            #[cfg(feature = "wgpu")]
1004            Self::Wgpu { .. } => f.write_str("SyncPoint::Wgpu"),
1005            #[cfg(feature = "wgpu")]
1006            Self::DeferredWgpu { slot, .. } => {
1007                f.debug_struct("DeferredWgpu").field("committed", &slot.get().is_some()).finish()
1008            }
1009            Self::Cpu => f.write_str("SyncPoint::Cpu"),
1010            Self::Noop => f.write_str("SyncPoint::Noop"),
1011        }
1012    }
1013}
1014
1015/// Runtime-agnostic single-yield future. Returns `Pending` once then
1016/// resolves to `Ready(())` on the next poll, with `wake_by_ref` driving
1017/// the re-poll. Works under any executor (`pollster`, `tokio`, `smol`,
1018/// `futures::executor`).
1019///
1020/// Used by the default `SyncWaiter::wait_async` impl and by per-backend
1021/// overrides for their fast-path yield-spin loops.
1022pub async fn yield_once() {
1023    let mut yielded = false;
1024    core::future::poll_fn(|cx| {
1025        if yielded {
1026            core::task::Poll::Ready(())
1027        } else {
1028            yielded = true;
1029            cx.waker().wake_by_ref();
1030            core::task::Poll::Pending
1031        }
1032    })
1033    .await
1034}
1035
1036/// Helper: convert a `Duration` to nanoseconds, saturating at
1037/// `u64::MAX` for "wait forever". Matches `vkWaitSemaphores`'
1038/// `UINT64_MAX` infinite-wait sentinel.
1039pub fn duration_to_ns(t: Duration) -> u64 {
1040    u64::try_from(t.as_nanos()).unwrap_or(u64::MAX)
1041}
1042
1043/// Helper: convert a `Duration` to milliseconds, saturating at
1044/// `u32::MAX` for the Win32 `INFINITE` sentinel
1045/// (`WaitForSingleObject`, `WaitForMultipleObjects`).
1046pub fn duration_to_ms_u32(t: Duration) -> u32 {
1047    u32::try_from(t.as_millis()).unwrap_or(u32::MAX)
1048}