Skip to main content

gpu_handle_types/
sync.rs

1// SPDX-License-Identifier: MIT OR Apache-2.0
2
3use core::future::Future;
4use core::pin::Pin;
5use std::any::Any;
6use std::ffi::c_void;
7use std::sync::Arc;
8use std::time::Duration;
9
10#[cfg(feature = "wgpu")]
11use crate::SliceOutcome;
12use crate::{BackendKind, Error, GlBackend};
13
14/// Boxed dyn-future return shape used by `SyncWaiter::wait_async` so the
15/// trait stays object-safe through `Arc<dyn SyncWaiter>`. Native
16/// `async fn` on the trait would be RPITIT and dyn-incompatible by
17/// default; callers dispatch through `Arc<dyn SyncWaiter>`, so boxing the
18/// future is mandatory.
19// `Send` off wasm; dropped on wasm, where the `SyncWaiter`
20// the future borrows is thread-affine (`&dyn SyncWaiter` is not `Send`
21// once the waiter is `!Sync`). An explicit `+ Send` on a `dyn Future`
22// cannot be spelled with a non-auto marker trait, so the alias is
23// cfg-split directly.
24#[cfg(not(target_family = "wasm"))]
25pub type BoxFuture<'a, T> = Pin<Box<dyn Future<Output = T> + Send + 'a>>;
26#[cfg(target_family = "wasm")]
27pub type BoxFuture<'a, T> = Pin<Box<dyn Future<Output = T> + 'a>>;
28
29/// Default wait budget for `SyncPoint::wait()` and
30/// `SyncPoint::wait_blocking()` — 10 seconds.
31pub const DEFAULT_WAIT_TIMEOUT: Duration = Duration::from_secs(10);
32
33/// A far-future but always-representable wait cap. `Instant + Duration` overflows
34/// the platform's representable range only for a `timeout` so large it is
35/// effectively infinite (near `Duration::MAX` — an `Instant` spans hundreds of
36/// billions of years), and `Instant` has no `saturating_add`. Rather than let such
37/// a `timeout` silently drop its deadline — turning a bounded wait into a hang —
38/// [`wait_deadline`] clamps to `now +` this. It sits far beyond any real GPU wait
39/// (budgets are seconds), so a legitimately huge finite wait still resolves against
40/// it as intended.
41const SATURATED_WAIT_CAP: Duration = Duration::from_secs(60 * 60 * 24 * 365); // ~1 year
42
43/// The absolute deadline a `wait(timeout)` poll loop should honour, or `None` for
44/// the [`SyncWaiter::wait`] contract's `Duration::MAX` "wait forever" sentinel.
45///
46/// - `timeout == Duration::MAX` → `None`: the caller asked to wait forever, so the
47///   loop runs with no deadline (matching every backend's explicit `Duration::MAX`
48///   handling, which maps to the native "infinite": `vkWaitSemaphores(UINT64_MAX)`,
49///   `WaitForSingleObject(INFINITE)`, `cuEventSynchronize`, …).
50/// - any other `timeout` → `Some(now + timeout)`, **saturated** to a bounded,
51///   always-representable [`Instant`](std::time::Instant) on `Instant + Duration`
52///   overflow — clamped to `now + 1 year`, far beyond any real GPU wait budget.
53///
54/// This keeps the two meanings of "no deadline" distinct: a returned `None`
55/// encodes ONLY the explicit forever sentinel, never an accidental `checked_add`
56/// overflow — which would otherwise silently convert a bounded wait into an
57/// unbounded hang.
58pub fn wait_deadline(timeout: Duration) -> Option<std::time::Instant> {
59    if timeout == Duration::MAX {
60        return None;
61    }
62    let now = std::time::Instant::now();
63    Some(now.checked_add(timeout).unwrap_or_else(|| now.checked_add(SATURATED_WAIT_CAP).unwrap_or(now)))
64}
65
66/// Backend-specific wait dispatch + lifetime anchor for a [`SyncPoint`].
67///
68/// Every non-trivial `SyncPoint` variant carries an `Arc<dyn SyncWaiter>`
69/// that:
70///
71/// 1. **Owns the underlying primitive's lifetime** — the timeline
72///    semaphore, D3D12 fence, `MTLSharedEvent`, `cl_event`, `GLsync`,
73///    etc. Cloning the `SyncPoint` clones the `Arc`; the primitive lives
74///    as long as any clone outlives. This rules out the "consumer holds
75///    a `*mut c_void` after the producer dropped" footgun.
76///
77/// 2. **Provides the wait implementation** — `SyncPoint::wait` /
78///    `is_signaled` / `backend` dispatch through this trait, with no
79///    global registries and no `OnceLock<fn>` runtime callbacks.
80///
81/// Implementations live in the producer crate (typically an interop
82/// layer's per-backend module) and are constructed at the same site that
83/// mints the `SyncPoint`.
84// `MaybeSendSync` = `Send + Sync` off wasm; empty on wasm,
85// where a waiter can hold a thread-affine `wgpu::Device` /
86// `web_sys` handle. See [`crate::MaybeSendSync`].
87pub trait SyncWaiter: crate::MaybeSendSync + 'static {
88    /// Block until the GPU work this sync point represents has
89    /// completed, or until `timeout` elapses.
90    ///
91    /// `Duration::MAX` means "wait forever" — implementations should
92    /// translate this to the platform's native "infinite" sentinel
93    /// (`UINT64_MAX` for `vkWaitSemaphores`, `INFINITE` for
94    /// `WaitForSingleObject`, etc.).
95    ///
96    /// # Timeout granularity
97    ///
98    /// Native wait APIs vary in granularity, and implementations
99    /// preserve the caller's intent at the expense of *requested*
100    /// (not realised) latency on sub-API-tick timeouts:
101    ///
102    /// - **Metal** (`MTLSharedEvent::waitUntilSignaledValue:timeoutMS:`)
103    ///   accepts millisecond-granular `u64`. Sub-millisecond positive
104    ///   timeouts (e.g. `Duration::from_micros(100)`) round **up**
105    ///   to 1 ms — truncation to 0 would silently behave as
106    ///   "do not wait". A high-frequency progress poll using
107    ///   `Duration::from_micros(100)` therefore observes ~10× the
108    ///   requested latency on a not-yet-signaled event. For
109    ///   non-blocking probes use [`SyncWaiter::is_signaled`] (which
110    ///   passes `Duration::ZERO` and short-circuits on the
111    ///   `signaledValue() >= value` accessor — driver-side, no
112    ///   timer).
113    /// - **Win32** (`WaitForSingleObject`) is also millisecond-
114    ///   granular; the same round-up applies.
115    /// - **Vulkan** (`vkWaitSemaphores`), **D3D12 fence**, **CUDA
116    ///   external semaphore**, **OpenCL** are nanosecond-granular and
117    ///   honour the caller's `Duration` exactly.
118    ///
119    /// Callers that need *both* a deterministic poll cadence under
120    /// 1 ms *and* a real wait fallback should compose the two —
121    /// e.g. `is_signaled` in a hot loop with their own `Instant`
122    /// budget, then a single coarse `wait(Duration::from_millis(N))`
123    /// when the budget is exhausted.
124    fn wait(&self, timeout: Duration) -> Result<(), Error>;
125
126    /// Async sibling of [`wait`](Self::wait). Returns a boxed future so
127    /// the trait stays object-safe through `Arc<dyn SyncWaiter>` — native
128    /// async-fn-in-trait is RPITIT and dyn-incompatible by default; callers
129    /// dispatch through `Arc<dyn SyncWaiter>` everywhere, so the explicit
130    /// `Pin<Box<...>>` is mandatory.
131    ///
132    /// # Default implementation
133    ///
134    /// Backends without a hand-tuned implementation get a two-phase
135    /// hybrid: a short cooperative-yield spin (catches sub-ms waits
136    /// without burning a thread) followed by an iteration-bounded
137    /// `is_signaled` probe loop. Per-backend impls SHOULD override with
138    /// their native blocking-with-timeout primitive on a dedicated
139    /// waiter thread for waits the spin loop didn't catch.
140    ///
141    /// The default implementation is correct (resolves when signalled,
142    /// returns `Error::Timeout` after `timeout`, never blocks the
143    /// executor) but not optimal: it busy-polls inside the spin loop
144    /// and falls back to a `Duration::ZERO` probe loop after that. For
145    /// production paths, override with a backend-specific impl.
146    fn wait_async<'a>(&'a self, timeout: Duration) -> BoxFuture<'a, Result<(), Error>> {
147        Box::pin(async move {
148            // Fast-path yield-spin — bounded cooperative-yield probe. Caps iteration
149            // count rather than wall-clock time so executors that re-
150            // poll immediately on `wake_by_ref` (pollster) don't burn
151            // CPU here.
152            const SPIN_ITERATIONS: usize = 64;
153            // `None` iff `timeout == Duration::MAX` (the wait-forever sentinel); every
154            // finite timeout gets a bounded, always-representable deadline — an
155            // `Instant + Duration` overflow cannot silently drop it and hang.
156            let deadline = wait_deadline(timeout);
157            for _ in 0..SPIN_ITERATIONS {
158                match self.is_signaled() {
159                    Ok(true) => return Ok(()),
160                    Ok(false) => {}
161                    Err(e) => return Err(e),
162                }
163                if let Some(d) = deadline
164                    && std::time::Instant::now() >= d
165                {
166                    return Err(Error::Timeout);
167                }
168                yield_once().await;
169            }
170            // Waiter-thread fallback — default impl probes `is_signaled` in a
171            // yielding loop. Per-backend impls SHOULD override to hand
172            // off to a dedicated waiter thread with the native
173            // blocking-with-timeout primitive.
174            loop {
175                match self.is_signaled() {
176                    Ok(true) => return Ok(()),
177                    Ok(false) => {}
178                    Err(e) => return Err(e),
179                }
180                if let Some(d) = deadline
181                    && std::time::Instant::now() >= d
182                {
183                    return Err(Error::Timeout);
184                }
185                yield_once().await;
186            }
187        })
188    }
189
190    /// Non-blocking probe.
191    ///
192    /// - `Ok(true)` — the sync point has been reached.
193    /// - `Ok(false)` — work is still in flight (the underlying poll
194    ///   timed out at zero).
195    /// - `Err(_)` — driver-level failure (device lost / TDR /
196    ///   `wgpu::PollError::WrongSubmissionIndex` / equivalent on other
197    ///   backends). Callers polling in a loop **must** break on `Err` —
198    ///   folding errors into `Ok(false)` would loop forever on a TDR'd
199    ///   device.
200    ///
201    /// Default implementation calls `wait(Duration::ZERO)` and maps
202    /// `Err(Error::Timeout)` → `Ok(false)`; every other `Err` is
203    /// propagated. Backends override when they have a more direct
204    /// "is this primitive currently signaled" probe (e.g.
205    /// `vkGetSemaphoreCounterValue`, `ID3D12Fence::GetCompletedValue`)
206    /// that distinguishes "not signaled" from real errors without
207    /// going through the wait path.
208    fn is_signaled(&self) -> Result<bool, Error> {
209        match self.wait(Duration::ZERO) {
210            Ok(()) => Ok(true),
211            Err(Error::Timeout) => Ok(false),
212            Err(other) => Err(other),
213        }
214    }
215
216    /// Backend identity for routing decisions on the consumer side.
217    fn backend(&self) -> BackendKind;
218
219    /// Downcast hook for cross-API bridges that need access to the
220    /// concrete waiter type — e.g. a Vulkan→CUDA bridge that wants to
221    /// pull the `VkDevice` out of the `VulkanWaiter` to issue its own
222    /// `cuImportExternalSemaphore`. Most callers use the per-variant
223    /// raw fields on `SyncPoint` directly and never need this.
224    fn as_any(&self) -> &dyn Any;
225
226    /// [`CudaEventWaiter`] view of this waiter, when the concrete type
227    /// implements it. Default `None`.
228    ///
229    /// This is the trait-object route to the CUDA-only
230    /// [`CudaEventWaiter::wait_on_foreign_stream`] extension: `Any` can
231    /// only downcast to *concrete* types, which consumers in other
232    /// crates cannot name — so producers whose waiter implements
233    /// [`CudaEventWaiter`] override this with `Some(self)` and
234    /// consumers (e.g. a CUDA import path gating its private copy stream
235    /// on a producer's `SyncPoint::CudaEvent`) reach the extension method
236    /// without knowing the concrete type.
237    fn as_cuda_event_waiter(&self) -> Option<&dyn CudaEventWaiter> {
238        None
239    }
240
241    /// Rebuild this waiter bound to `value` instead of the value it was
242    /// constructed with, reusing the same underlying primitive (and the
243    /// same lifetime anchors).
244    ///
245    /// # Why this exists
246    ///
247    /// [`wait`](Self::wait) / [`is_signaled`](Self::is_signaled) take no
248    /// value argument — the waiter *embeds* the value it resolves at,
249    /// captured when it was built. So a [`SyncPoint`] whose `value`
250    /// field was substituted while its `waiter` was cloned verbatim has
251    /// two surfaces that disagree: the GPU side (a consumer reading
252    /// `SyncPoint::*.value` to stage a queue wait) waits the new value,
253    /// while the CPU side silently resolves at the old one — reporting
254    /// "already signalled" for a signal the producer has not emitted.
255    ///
256    /// Any code that mints a `SyncPoint` at a value other than the one a
257    /// template was built with MUST route the waiter through this method
258    /// rather than cloning it, so the two surfaces cannot diverge.
259    ///
260    /// # Contract
261    ///
262    /// - `Some(w)` — `w` waits `value` on the same primitive this waiter
263    ///   waits on, and holds the same keep-alive chain. Implementations
264    ///   MUST NOT carry over state that is only valid for the original
265    ///   value (e.g. a captured submission index that retires at it).
266    /// - `None` (the default) — the underlying primitive carries no
267    ///   value, so there is nothing to rebind: binary semaphores,
268    ///   fences, `GLsync`, `cl_semaphore_khr`. A `None` here is not a
269    ///   failure; it means the CPU surface has no value to disagree
270    ///   about.
271    fn rebind_to_value(&self, value: u64) -> Option<Arc<dyn SyncWaiter>> {
272        let _ = value;
273        None
274    }
275}
276
277/// CUDA-event-specific extension trait for the
278/// [`SyncPoint::CudaEvent`] variant. Supertrait of [`SyncWaiter`] so a
279/// `CudaEventWaiter` always satisfies the generic [`SyncPoint::wait`]
280/// dispatch path; adds the CUDA-only `wait_on_foreign_stream` extension
281/// method that cross-API bridges downcast to via
282/// [`SyncWaiter::as_any`].
283///
284/// # Drop discipline
285///
286/// Implementations MUST push the event's owning `CUcontext` before
287/// calling `cuEventDestroy_v2`, and MUST pop iff the push succeeded
288/// (pop-without-push would silently consume whatever the calling
289/// thread had on top of its context stack). `cuEventDestroy_v2`
290/// requires the owning context to still exist; the push is defensive
291/// against driver-internal cleanup paths that probe `cuCtxGetCurrent`
292/// and emit confusing diagnostics when no context is current.
293///
294/// # Cross-context wait
295///
296/// `cuEventRecord` is strict-same-context — a `CUevent` cannot be
297/// recorded against a stream from a different `CUcontext`. But
298/// `cuStreamWaitEvent` is cross-context (NVIDIA Driver API contract).
299/// `wait_on_foreign_stream` is the cross-context wait entry point:
300/// the consumer's CUDA stream may live in a different `CUcontext`
301/// than the event, and the waiter issues `cuStreamWaitEvent` so the
302/// foreign stream gates on this event's completion without a CPU
303/// bounce.
304///
305/// A waiter backed by the CUDA driver API implements this trait
306/// concretely; this trait only defines the contract.
307pub trait CudaEventWaiter: SyncWaiter {
308    /// Issue `cuStreamWaitEvent(foreign_stream, self.event, 0)` so
309    /// `foreign_stream` gates on this event's completion. The stream
310    /// may belong to a different `CUcontext` than the event's owning
311    /// context — `cuStreamWaitEvent` is cross-context per the NVIDIA
312    /// Driver API.
313    ///
314    /// `foreign_stream` is a raw `CUstream` pointer (the null pointer
315    /// resolves to the calling thread's default stream).
316    ///
317    /// Returns [`Error::NotSupported`] if the CUDA driver loader is
318    /// unavailable or the wait dispatch returns a driver error; the
319    /// concrete error wording is implementation-defined.
320    fn wait_on_foreign_stream(&self, foreign_stream: *mut c_void) -> Result<(), Error>;
321}
322
323/// Discriminator for [`SyncPoint::Vulkan`]'s `VkSemaphore` flavour.
324///
325/// `VkSemaphore` ships in two shapes — binary (one-shot, signal-and-
326/// reset) and timeline (monotonic 64-bit payload, multi-waiter,
327/// concurrent-signal-safe). A consumer staging a wait must route
328/// through `VkTimelineSemaphoreSubmitInfo` for timeline semaphores and
329/// must NOT for binary semaphores — passing the wrong shape risks
330/// validation errors, driver-quirk-dependent wrong waits, and (on
331/// already-consumed binary signals) hard deadlock at submit time.
332///
333/// The Vulkan API has no public probe that safely distinguishes the
334/// two (`vkGetSemaphoreCounterValue` is undefined on binaries per
335/// `VUID-vkGetSemaphoreCounterValue-semaphore-03255`), so the producer
336/// must declare intent at mint time.
337#[derive(Copy, Clone, Debug, PartialEq, Eq)]
338#[non_exhaustive]
339pub enum VulkanSemaphoreKind {
340    /// `vk::SemaphoreType::TIMELINE`. `value` is a payload coordinate;
341    /// any number of waiters may target the same value, and signals
342    /// with strictly-increasing payloads can be queued concurrently.
343    Timeline,
344    /// `vk::SemaphoreType::BINARY`. Single-shot — exactly one signal
345    /// must pair with exactly one wait. The `value` field is ignored
346    /// (set to `0` by convention). Producers carrying binaries here
347    /// MUST guarantee the wait is the only consumer of the signal.
348    Binary,
349}
350
351/// Shared, mutate-once slot the producer fills in after the caller
352/// submits the encoded work. Used by [`SyncPoint::DeferredWgpu`].
353///
354/// The slot is initialised to `None` at mint time and transitions to
355/// `Some(idx)` exactly once via [`Self::set`]; subsequent calls are
356/// rejected with [`Error::InvalidArgument`] (single-shot semantics).
357#[cfg(feature = "wgpu")]
358#[derive(Debug, Default)]
359pub struct DeferredWgpuSlot {
360    cell: std::sync::OnceLock<wgpu::SubmissionIndex>,
361}
362
363#[cfg(feature = "wgpu")]
364impl DeferredWgpuSlot {
365    pub fn new() -> Self {
366        Self::default()
367    }
368
369    /// Commit the post-submit `SubmissionIndex`. Single-shot — a second
370    /// call returns [`Error::InvalidArgument`] without overwriting the
371    /// committed value.
372    pub fn set(&self, idx: wgpu::SubmissionIndex) -> Result<(), Error> {
373        self.cell.set(idx).map_err(|_| Error::InvalidArgument("DeferredWgpuSlot already committed".into()))
374    }
375
376    /// Return the committed `SubmissionIndex`, or `None` if the producer
377    /// has not yet submitted.
378    pub fn get(&self) -> Option<&wgpu::SubmissionIndex> {
379        self.cell.get()
380    }
381}
382
383// Defensive — the `OnceLock<wgpu::SubmissionIndex>` design relies on
384// `SubmissionIndex` being `Clone + Send + Sync`. If a wgpu upgrade
385// silently dropped any of these bounds, the OnceLock-storing
386// `DeferredWgpuSlot` would surface a confusing trait-bound error far
387// from the cause; this assertion fails compilation here, with a clear
388// message, instead.
389#[cfg(feature = "wgpu")]
390static_assertions::assert_impl_all!(wgpu::SubmissionIndex: Clone, Send, Sync);
391
392/// GPU sync primitive. Produced by every submit-side path; consumed by
393/// anything that needs to serialise on a prior GPU submission.
394///
395/// Cross-backend bridges are deliberately absent from this leaf crate —
396/// those belong in an interop layer, where the target-device context is
397/// available.
398///
399/// # Lifetime
400///
401/// The per-variant `waiter: Arc<dyn SyncWaiter>` field anchors the
402/// lifetime of any raw primitive the variant exposes (`*mut c_void`
403/// fence pointers, `u64` semaphore handles, `GLsync` opaques, etc.).
404/// Cloning a `SyncPoint` is `Arc::clone` on the waiter — cheap, no
405/// driver round-trip. The raw fields are guaranteed to remain valid
406/// for as long as any clone is alive.
407#[derive(Clone)]
408#[non_exhaustive]
409pub enum SyncPoint {
410    /// Vulkan semaphore. `device` is the `VkDevice` the semaphore was
411    /// created on, exposed for bridges that need to re-import it. The
412    /// `waiter` owns whatever anchors the `VkSemaphore`'s lifetime.
413    ///
414    /// `kind` distinguishes timeline vs binary semaphores —
415    /// see [`VulkanSemaphoreKind`]. `value` is the timeline payload
416    /// coordinate when `kind == Timeline`; for `kind == Binary` the
417    /// field is meaningless (set to `0` by convention) and consumers
418    /// route through a binary wait (with wgpu-hal,
419    /// `vulkan::Queue::add_wait_semaphore(_, None, _)`).
420    Vulkan { semaphore: u64, kind: VulkanSemaphoreKind, value: u64, device: *mut c_void, waiter: Arc<dyn SyncWaiter> },
421    /// D3D12 fence sync. `fence` is an `ID3D12Fence*` (COM pointer);
422    /// `value` is the monotonic signal value the consumer waits on
423    /// (`fence->GetCompletedValue() >= value`). The `waiter` holds the
424    /// COM reference that keeps the fence alive.
425    ///
426    /// `value` is monotonically signaled by the producer. Bridges that
427    /// reuse a single shared fence across calls serialise the
428    /// (mint + `Signal`) pair under a value-lock so concurrent bridges
429    /// always sequence in monotonic order.
430    /// A consumer waiting on `value=N` is guaranteed to pass once a
431    /// later `value=M >= N` has been signaled, even if the producer's
432    /// counter has since advanced past `M`.
433    D3D12 { fence: *mut c_void, value: u64, waiter: Arc<dyn SyncWaiter> },
434    /// D3D11 keyed-mutex sync. `keyed_mutex` is an `IDXGIKeyedMutex*`;
435    /// `key` is the integer key the producer released to. The
436    /// `waiter` holds the COM reference that keeps the mutex alive.
437    D3D11 { keyed_mutex: *mut c_void, key: u64, waiter: Arc<dyn SyncWaiter> },
438    /// CUDA external semaphore. `event` is a `CUexternalSemaphore`;
439    /// `value` is `Some(v)` for timeline waits — including the products
440    /// of Vulkan→CUDA / D3D12→CUDA bridges, which carry the producer's
441    /// timeline value so the consumer can enqueue
442    /// `cuWaitExternalSemaphoresAsync` on whichever CUDA stream their
443    /// work runs on. `None` is reserved for setup-metadata SyncPoints
444    /// where the consumer mints the actual signal value at emission time
445    /// and doesn't need a resolvable wait at construction.
446    Cuda { event: *mut c_void, value: Option<u64>, waiter: Arc<dyn SyncWaiter> },
447    /// Intra-CUDA event. `event` is a `CUevent` recorded on the
448    /// producing stream via `cuEventRecord`. Consumers wait via
449    /// `cuStreamWaitEvent` (chained CUDA streams — cross-context
450    /// supported per NVIDIA Driver API) or `cuEventSynchronize`
451    /// (blocking CPU wait).
452    ///
453    /// Distinct from [`Self::Cuda`] — that variant carries a
454    /// `CUexternalSemaphore` waited via `cuWaitExternalSemaphoresAsync`.
455    /// `CUevent` and `CUexternalSemaphore` are different opaque types
456    /// in the CUDA Driver API; passing a `CUevent` into the external-
457    /// semaphore wait path returns `CUDA_ERROR_INVALID_HANDLE`.
458    ///
459    /// Cross-API consumers cannot wait on a `CUevent` directly; they
460    /// need a `CUexternalSemaphore` (the [`Self::Cuda`] shape); an
461    /// interop layer performs that conversion when the chain spans a
462    /// backend boundary.
463    ///
464    /// `context` is the event's **owning** `CUcontext` — bound at
465    /// `cuEventCreate` time per the NVIDIA Driver API ("creates an event
466    /// for the current context"). The owning context must still exist
467    /// at destroy time or `cuEventDestroy_v2` returns
468    /// `CUDA_ERROR_INVALID_CONTEXT`. There is no `cuEventGetCtx` API —
469    /// the metadata must travel with the event.
470    ///
471    /// `device` is the CUDA ordinal the owning context was created on
472    /// (derivable from `context` via `cuCtxGetDevice` but cheap to
473    /// carry and matches the device-keyed bridge shape used by
474    /// [`Self::Cuda`]).
475    ///
476    /// `waiter` implements [`CudaEventWaiter`] — a supertrait of
477    /// [`SyncWaiter`] that adds the CUDA-specific
478    /// `wait_on_foreign_stream` extension method and the push-owning-
479    /// context-before-destroy Drop discipline. The leaf crate stores
480    /// the upcasted `Arc<dyn SyncWaiter>` so the generic [`Self::waiter`]
481    /// dispatch path is uniform with every other variant; cross-API
482    /// bridges that need the CUDA extension methods downcast through
483    /// [`SyncWaiter::as_any`].
484    CudaEvent { event: *mut c_void, context: *mut c_void, device: i32, waiter: Arc<dyn SyncWaiter> },
485    /// OpenCL sync. `event` is either a `cl_event` (binary,
486    /// signaled-on-completion; `value` is `None`) or a
487    /// `cl_semaphore_khr` aliasing a timeline (`value` is the
488    /// caller-signaled timeline value). The `waiter` keeps the
489    /// producing context alive.
490    OpenCl { event: *mut c_void, value: Option<u64>, waiter: Arc<dyn SyncWaiter> },
491    /// Metal shared event. `event` is an `MTLSharedEvent*`; `value` is
492    /// the signal value the consumer waits on. The `waiter` holds the
493    /// reference that keeps the event alive.
494    Metal { event: *mut c_void, value: u64, waiter: Arc<dyn SyncWaiter> },
495    /// OpenGL semaphore (`GLuint` produced by `glGenSemaphoresEXT`,
496    /// imported from a cross-API OS handle via
497    /// `glImportSemaphoreFdEXT` / `glImportSemaphoreWin32HandleEXT`).
498    ///
499    /// `value` is `Some(_)` for timeline-equivalent semaphores
500    /// (D3D12_FENCE_EXT handle types); `None` for plain binary
501    /// (OPAQUE_FD / OPAQUE_WIN32) — the GL driver latches the D3D12
502    /// fence value at wait/signal time via
503    /// `glSemaphoreParameterui64vEXT(sem, GL_D3D12_FENCE_VALUE_EXT, &v)`.
504    ///
505    /// `context` is the caller's `EGLContext` / `HGLRC` / `CGLContextObj`
506    /// this semaphore was created against. Waits must be dispatched on
507    /// a thread where the same context is current.
508    OpenGL { semaphore: u32, value: Option<u64>, context: *mut c_void, waiter: Arc<dyn SyncWaiter> },
509    /// GL fence sync. `glsync` is a `GLsync` opaque pointer produced
510    /// by `glFenceSync(GL_SYNC_GPU_COMMANDS_COMPLETE, 0)`.
511    ///
512    /// Distinct from [`Self::OpenGL`] — that one carries an *imported*
513    /// GL semaphore (`GLuint`); this one is the raw GL-native fence.
514    ///
515    /// `backend` records the [`GlBackend`] flavour of the GL context that
516    /// produced this `GLsync`. Cross-API consumers (an OpenGL→X bridge)
517    /// route through it to pick the matching loader (`wglGetProcAddress`
518    /// vs `eglGetProcAddress`) when resolving `glClientWaitSync` for the
519    /// CPU-fallback path.
520    /// Without it the consumer would hard-code `Desktop`, which silently
521    /// resolves the wrong proc-address on EGL / ANGLE / Android hosts and
522    /// stalls forever on a never-signalled handle.
523    OpenGLSync { glsync: *mut c_void, backend: GlBackend, waiter: Arc<dyn SyncWaiter> },
524    /// `wgpu::Queue::submit` return value. Wait dispatches through the
525    /// `waiter`, which holds a `wgpu::Device` clone and calls
526    /// `device.poll(PollType::Wait { submission_index: Some(..), .. })`.
527    #[cfg(feature = "wgpu")]
528    Wgpu { submission_index: wgpu::SubmissionIndex, waiter: Arc<dyn SyncWaiter> },
529    /// `wgpu::Queue::submit` index *not yet known* at SyncPoint mint
530    /// time. Used where the caller owns the encoder and submits after the
531    /// producer has returned; `slot.set(idx)` is called once after
532    /// `queue.submit(...)`, and until then the waiter falls back to
533    /// `device.poll(PollType::Wait { submission_index: None, .. })`
534    /// (full device drain). After commit, waits resolve precisely on
535    /// the recorded `SubmissionIndex`.
536    #[cfg(feature = "wgpu")]
537    DeferredWgpu { slot: Arc<DeferredWgpuSlot>, device: wgpu::Device, waiter: Arc<dyn SyncWaiter> },
538    /// CPU-only work — no GPU sync needed. `wait` returns `Ok(())`
539    /// immediately.
540    Cpu,
541    /// No-op sync — used for pipelines where the consumer side handles
542    /// ordering through a separate channel (e.g. SurfaceControl
543    /// transactions). Behaves identically to `Cpu` for waits.
544    Noop,
545}
546
547// SAFETY: the `*mut c_void` raw fields are immutable after
548// construction and dereferenced only by the per-variant `waiter`,
549// whose `SyncWaiter: Send + Sync` bound asserts platform-correct sharing
550// for whatever the pointer references. Per-variant doc-comments specify
551// the additional caller-asserted preconditions (e.g. D3D11 requires the
552// device's multithread-protect flag).
553unsafe impl Send for SyncPoint {}
554unsafe impl Sync for SyncPoint {}
555
556impl SyncPoint {
557    /// Async wait. Default 10-second timeout.
558    ///
559    /// Async is the default. Use [`wait_blocking`]
560    /// when the caller is on a synchronous code path (sync-loop
561    /// renderers, test harnesses, JNI dispatcher Drop paths).
562    ///
563    /// [`wait_blocking`]: Self::wait_blocking
564    pub async fn wait(&self) -> Result<(), Error> {
565        self.wait_with_timeout_async(DEFAULT_WAIT_TIMEOUT).await
566    }
567
568    /// Async wait with explicit timeout.
569    pub async fn wait_with_timeout_async(&self, timeout: Duration) -> Result<(), Error> {
570        // Binary Vulkan CPU-wait rejection.
571        if let Some(err) = self.binary_vk_cpu_wait_error() {
572            return Err(err);
573        }
574        match self.waiter() {
575            Some(w) => w.wait_async(timeout).await,
576            None => Ok(()),
577        }
578    }
579
580    /// Synchronous wait with the default 10-second timeout.
581    pub fn wait_blocking(&self) -> Result<(), Error> {
582        self.wait_with_timeout(DEFAULT_WAIT_TIMEOUT)
583    }
584
585    /// Synchronous wait with an explicit timeout.
586    ///
587    /// Timeout granularity is backend-dependent — see
588    /// [`SyncWaiter::wait`] for the per-backend rounding contract.
589    /// In particular, Metal and Win32 are millisecond-granular and
590    /// round sub-ms positive timeouts up to 1 ms; Vulkan / D3D12 /
591    /// CUDA / OpenCL are nanosecond-granular. For sub-ms polls use
592    /// [`Self::is_signaled`] in a loop with your own `Instant` budget.
593    pub fn wait_with_timeout(&self, timeout: Duration) -> Result<(), Error> {
594        // Binary Vulkan CPU-wait rejection — `vkWaitSemaphores` is
595        // undefined on binary semaphores, and the Khronos spec gives
596        // binary semaphores no CPU-side wait entry point. Producers
597        // that need CPU-waitable binary semaphores MUST attach a
598        // fence-equivalent fallback (a `wgpu::SubmissionIndex` from the
599        // same submit). When the fallback is absent, the backend's waiter
600        // fails both `wait_blocking` and the async `wait()` with
601        // `Error::NotSupported` (see `binary_vk_cpu_wait_error`).
602        if let Some(err) = self.binary_vk_cpu_wait_error() {
603            return Err(err);
604        }
605        match self.waiter() {
606            Some(w) => w.wait(timeout),
607            None => Ok(()),
608        }
609    }
610
611    /// Leaf-crate hook for rejecting a CPU wait on a binary
612    /// `SyncPoint::Vulkan` whose producer did not attach a
613    /// fence-equivalent fallback. Always `None`: whether the fallback is
614    /// attached is visible only to the backend's waiter, which is where
615    /// that half of the binary-Vulkan CPU-wait contract is enforced.
616    fn binary_vk_cpu_wait_error(&self) -> Option<Error> {
617        // The fence-equivalent fallback lives on the per-backend Vulkan
618        // waiter impl; this crate only sees `Arc<dyn SyncWaiter>`. The
619        // contract is that `wait` / `wait_async` on a binary semaphore
620        // without a fallback return `Error::NotSupported`. Backends MUST
621        // honour it; the "fallback present" check is not observable here,
622        // so this returns `None` rather than gate on it.
623        let _ = matches!(self, Self::Vulkan { kind: VulkanSemaphoreKind::Binary, .. });
624        None
625    }
626
627    /// Sequence two sync points. Returns a `SyncPoint` whose wait
628    /// completes only after both `self` and `then` have completed.
629    ///
630    /// If either side is `Cpu` / `Noop`, the other is returned unchanged.
631    /// Otherwise the result is an **opaque** carrier: a small adapter
632    /// waiter drives both inner waits in sequence (one heap allocation),
633    /// and it is stored in the [`SyncPoint::Cuda`] variant with a null
634    /// `event` and no `value`. Only use the result through
635    /// [`wait`](Self::wait), [`wait_blocking`](Self::wait_blocking),
636    /// [`is_signaled`](Self::is_signaled), [`backend`](Self::backend)
637    /// (the first sync point's backend) and [`waiter`](Self::waiter);
638    /// its raw fields carry nothing.
639    pub fn chain(self, then: SyncPoint) -> SyncPoint {
640        // Trivial cases — skip the adapter entirely.
641        if matches!(self, Self::Cpu | Self::Noop) {
642            return then;
643        }
644        if matches!(then, Self::Cpu | Self::Noop) {
645            return self;
646        }
647        let backend = self.backend();
648        let waiter: Arc<dyn SyncWaiter> = Arc::new(ChainWaiter { first: self, then, backend });
649        Self::Cpu.with_chain_waiter(waiter)
650    }
651
652    /// Wraps a `ChainWaiter` in the carrier variant `chain` documents.
653    /// Used only by `chain`.
654    fn with_chain_waiter(self, waiter: Arc<dyn SyncWaiter>) -> SyncPoint {
655        // Any variant with a `waiter` field would do: every caller-facing
656        // method dispatches through the waiter, so the event / value
657        // fields are never read.
658        let _ = self;
659        Self::Cuda { event: core::ptr::null_mut(), value: None, waiter }
660    }
661}
662
663// ─────────────────────────────────────────────────────────────────────────────
664// Raw-handle reconstruction (cross-C / cross-instance transport carriers)
665// ─────────────────────────────────────────────────────────────────────────────
666
667/// A [`SyncWaiter`] for a [`SyncPoint`] **reconstructed from a raw foreign fence
668/// handle** that has not yet been imported onto a waiting device.
669///
670/// A D3D12 fence / Vulkan semaphore / Metal shared event that crossed an API or
671/// process boundary as a raw handle (e.g. a `#[repr(C)]` descriptor over a C ABI)
672/// carries no device the leaf crate can drive a wait on — the real CPU/GPU wait
673/// only becomes possible once a consumer **imports** the handle onto its own device
674/// (the importer stages the device-side `vkWaitSemaphores` /
675/// `ID3D12CommandQueue::Wait` / `encodeWaitForEvent` off the carried raw fields).
676///
677/// So a `SyncPoint` reconstructed via [`SyncPoint::from_raw_d3d12_fence`] /
678/// [`SyncPoint::from_raw_vulkan_semaphore`] / [`SyncPoint::from_raw_metal_event`] is
679/// a **transport carrier**: its raw fields are consumed by the importer, and its
680/// `waiter` is this carrier. Calling [`SyncPoint::wait_blocking`] / `wait` /
681/// `is_signaled` on the carrier (before import) returns [`Error::NotSupported`] —
682/// fail-closed, never a silent "already signaled" — because the leaf crate cannot
683/// wait a fence it has no device for. Import the handle first; the import path
684/// produces the real, waitable `SyncPoint`.
685#[derive(Debug)]
686pub struct RawFenceCarrierWaiter {
687    backend: BackendKind,
688}
689
690impl RawFenceCarrierWaiter {
691    fn new(backend: BackendKind) -> Self {
692        Self { backend }
693    }
694
695    fn not_imported() -> Error {
696        Error::NotSupported(std::borrow::Cow::Borrowed(
697            "SyncPoint reconstructed from a raw foreign fence handle is a transport \
698                 carrier — import it onto a device before waiting",
699        ))
700    }
701}
702
703impl SyncWaiter for RawFenceCarrierWaiter {
704    fn wait(&self, _timeout: Duration) -> Result<(), Error> {
705        Err(Self::not_imported())
706    }
707
708    fn is_signaled(&self) -> Result<bool, Error> {
709        Err(Self::not_imported())
710    }
711
712    fn backend(&self) -> BackendKind {
713        self.backend
714    }
715
716    fn as_any(&self) -> &dyn Any {
717        self
718    }
719}
720
721impl SyncPoint {
722    /// Reconstruct a [`SyncPoint::D3D12`] transport carrier from a raw
723    /// `ID3D12Fence*` pointer + signal value that crossed an API / process boundary
724    /// (e.g. a `#[repr(C)]` fence descriptor over a C ABI).
725    ///
726    /// `fence` is an `ID3D12Fence*` COM pointer the caller guarantees is live for
727    /// the carrier's lifetime; `value` is the monotonic value the consumer waits for
728    /// (`fence->GetCompletedValue() >= value`). The returned `SyncPoint` is a
729    /// **carrier** ([`RawFenceCarrierWaiter`]) — its raw `(fence, value)` fields feed
730    /// a device-side import (e.g. an interop layer's acquire-sync path); a CPU
731    /// `wait()` on it before import returns [`Error::NotSupported`].
732    pub fn from_raw_d3d12_fence(fence: *mut c_void, value: u64) -> SyncPoint {
733        let waiter: Arc<dyn SyncWaiter> = Arc::new(RawFenceCarrierWaiter::new(BackendKind::D3D12));
734        SyncPoint::D3D12 { fence, value, waiter }
735    }
736
737    /// Reconstruct a [`SyncPoint::Vulkan`] transport carrier from a raw
738    /// `VkSemaphore` + `VkDevice` that crossed an API / process boundary.
739    ///
740    /// `semaphore` is the raw `VkSemaphore` (a 64-bit non-dispatchable handle);
741    /// `kind` distinguishes timeline vs binary (see [`VulkanSemaphoreKind`]);
742    /// `value` is the timeline coordinate (ignored for `Binary`); `device` is the
743    /// `VkDevice*` the semaphore lives on (needed by an importer that re-exports it
744    /// for a cross-device wait). The returned `SyncPoint` is a **carrier** — its raw
745    /// fields feed a device-side import; a CPU `wait()` before import returns
746    /// [`Error::NotSupported`].
747    pub fn from_raw_vulkan_semaphore(
748        semaphore: u64,
749        kind: VulkanSemaphoreKind,
750        value: u64,
751        device: *mut c_void,
752    ) -> SyncPoint {
753        let waiter: Arc<dyn SyncWaiter> = Arc::new(RawFenceCarrierWaiter::new(BackendKind::Vulkan));
754        SyncPoint::Vulkan { semaphore, kind, value, device, waiter }
755    }
756
757    /// Reconstruct a [`SyncPoint::Metal`] transport carrier from a raw
758    /// `MTLSharedEvent*` + signal value that crossed an API / process boundary.
759    ///
760    /// `event` is an `MTLSharedEvent*` the caller guarantees is live for the
761    /// carrier's lifetime; `value` is the value the consumer waits for. The returned
762    /// `SyncPoint` is a **carrier** — its raw `(event, value)` fields feed a
763    /// device-side import; a CPU `wait()` before import returns
764    /// [`Error::NotSupported`].
765    pub fn from_raw_metal_event(event: *mut c_void, value: u64) -> SyncPoint {
766        let waiter: Arc<dyn SyncWaiter> = Arc::new(RawFenceCarrierWaiter::new(BackendKind::Metal));
767        SyncPoint::Metal { event, value, waiter }
768    }
769}
770
771/// Per-process waiter thread for [`SyncPoint::DeferredWgpu`].
772///
773/// The waiter thread is stateless — every per-instance datum (device
774/// handle, deferred slot) travels inside the `SliceFn` closure — so one
775/// thread serves every deferred sync point in the process.
776#[cfg(feature = "wgpu")]
777fn deferred_wgpu_waiter_thread() -> &'static crate::WaiterThread {
778    static THREAD: std::sync::OnceLock<crate::WaiterThread> = std::sync::OnceLock::new();
779    THREAD.get_or_init(|| crate::WaiterThread::new("wgpu-deferred"))
780}
781
782/// Backend-internal waiter for [`SyncPoint::DeferredWgpu`]. Until the
783/// producer commits a `SubmissionIndex` via [`DeferredWgpuSlot::set`],
784/// `wait` falls back to `device.poll(PollType::Wait { submission_index:
785/// None, .. })` (full device drain). After commit, it passes
786/// `submission_index: Some(idx)` and resolves on that precise index.
787///
788/// `wait_async` routes through [`run_hybrid_wait`] — the fast-path
789/// yield-spin spins on `is_signaled` for sub-millisecond signals, the
790/// waiter-thread fallback hands off to the
791/// per-process waiter thread which issues blocking-with-timeout slices
792/// of [`WAITER_SLICE`] on a dedicated thread.
793#[cfg(feature = "wgpu")]
794pub(crate) struct DeferredWgpuWaiter {
795    slot: Arc<DeferredWgpuSlot>,
796    device: wgpu::Device,
797}
798
799#[cfg(feature = "wgpu")]
800impl DeferredWgpuWaiter {
801    pub(crate) fn new(slot: Arc<DeferredWgpuSlot>, device: wgpu::Device) -> Self {
802        Self { slot, device }
803    }
804
805    /// Single `device.poll(Wait { .. })` issuance keyed off the
806    /// current commit state. Shared between `wait`, `is_signaled`, and
807    /// the per-slice closure produced by `wait_async`.
808    fn poll_slice(&self, timeout: Option<Duration>) -> SliceOutcome {
809        let submission_index = self.slot.get().cloned();
810        match self.device.poll(wgpu::PollType::Wait { submission_index, timeout }) {
811            Ok(_) => SliceOutcome::Signaled,
812            Err(wgpu::PollError::Timeout) => SliceOutcome::TimedOut,
813            Err(e) => {
814                log::warn!("SyncPoint::DeferredWgpu: device.poll returned {e:?}");
815                SliceOutcome::Failed(Error::NotSupported("SyncPoint::DeferredWgpu: device.poll failed".into()))
816            }
817        }
818    }
819}
820
821#[cfg(feature = "wgpu")]
822impl SyncWaiter for DeferredWgpuWaiter {
823    fn wait(&self, timeout: Duration) -> Result<(), Error> {
824        let timeout_arg = if timeout == Duration::MAX { None } else { Some(timeout) };
825        match self.poll_slice(timeout_arg) {
826            SliceOutcome::Signaled => Ok(()),
827            SliceOutcome::TimedOut => Err(Error::Timeout),
828            SliceOutcome::Failed(e) => Err(e),
829        }
830    }
831
832    fn wait_async<'a>(&'a self, timeout: Duration) -> BoxFuture<'a, Result<(), Error>> {
833        let slot = self.slot.clone();
834        let device = self.device.clone();
835        let make_slice = move || -> crate::SliceFn {
836            Box::new(move |slice: Duration| -> SliceOutcome {
837                let submission_index = slot.get().cloned();
838                match device.poll(wgpu::PollType::Wait { submission_index, timeout: Some(slice) }) {
839                    Ok(_) => SliceOutcome::Signaled,
840                    Err(wgpu::PollError::Timeout) => SliceOutcome::TimedOut,
841                    Err(e) => {
842                        log::warn!("SyncPoint::DeferredWgpu::wait_async: device.poll returned {e:?}");
843                        SliceOutcome::Failed(Error::NotSupported(
844                            "SyncPoint::DeferredWgpu::wait_async: device.poll failed".into(),
845                        ))
846                    }
847                }
848            })
849        };
850        Box::pin(crate::run_hybrid_wait(move || self.is_signaled(), deferred_wgpu_waiter_thread(), timeout, make_slice))
851    }
852
853    fn is_signaled(&self) -> Result<bool, Error> {
854        match self.poll_slice(Some(Duration::ZERO)) {
855            SliceOutcome::Signaled => Ok(true),
856            SliceOutcome::TimedOut => Ok(false),
857            SliceOutcome::Failed(e) => Err(e),
858        }
859    }
860
861    fn backend(&self) -> BackendKind {
862        BackendKind::Wgpu
863    }
864
865    fn as_any(&self) -> &dyn Any {
866        self
867    }
868}
869
870/// Construct a [`SyncPoint::DeferredWgpu`] complete with its waiter.
871///
872/// For producers that hand out a sync point *before* they submit: the
873/// returned SyncPoint's clones all share the same `Arc<DeferredWgpuSlot>`,
874/// so the producer-side `slot.set(idx)` at submit time is visible to every
875/// clone.
876#[cfg(feature = "wgpu")]
877pub fn make_deferred_wgpu_sync_point(slot: Arc<DeferredWgpuSlot>, device: wgpu::Device) -> SyncPoint {
878    let waiter: Arc<dyn SyncWaiter> = Arc::new(DeferredWgpuWaiter::new(slot.clone(), device.clone()));
879    SyncPoint::DeferredWgpu { slot, device, waiter }
880}
881
882struct ChainWaiter {
883    first: SyncPoint,
884    then: SyncPoint,
885    backend: BackendKind,
886}
887
888impl SyncWaiter for ChainWaiter {
889    fn wait(&self, timeout: Duration) -> Result<(), Error> {
890        let start = std::time::Instant::now();
891        self.first.wait_with_timeout(timeout)?;
892        let elapsed = start.elapsed();
893        let remaining = timeout.saturating_sub(elapsed);
894        self.then.wait_with_timeout(remaining)
895    }
896
897    fn wait_async<'a>(&'a self, timeout: Duration) -> BoxFuture<'a, Result<(), Error>> {
898        Box::pin(async move {
899            let start = std::time::Instant::now();
900            self.first.wait_with_timeout_async(timeout).await?;
901            let elapsed = start.elapsed();
902            let remaining = timeout.saturating_sub(elapsed);
903            self.then.wait_with_timeout_async(remaining).await
904        })
905    }
906
907    fn is_signaled(&self) -> Result<bool, Error> {
908        Ok(self.first.is_signaled()? && self.then.is_signaled()?)
909    }
910
911    fn backend(&self) -> BackendKind {
912        self.backend
913    }
914
915    fn as_any(&self) -> &dyn Any {
916        self
917    }
918}
919
920impl SyncPoint {
921    /// Non-blocking probe.
922    ///
923    /// - `Ok(true)`  — sync point reached (or trivial `Cpu`/`Noop`).
924    /// - `Ok(false)` — work still in flight.
925    /// - `Err(_)`    — driver-level failure (device lost / TDR /
926    ///   wrong-submission-index). See [`SyncWaiter::is_signaled`]
927    ///   for the rationale on surfacing errors instead of folding
928    ///   them into `false`.
929    pub fn is_signaled(&self) -> Result<bool, Error> {
930        match self.waiter() {
931            Some(w) => w.is_signaled(),
932            None => Ok(true),
933        }
934    }
935
936    /// Backend identity. For the trivial variants (`Cpu`, `Noop`) this
937    /// returns [`BackendKind::Cpu`]; otherwise it dispatches to the
938    /// per-variant waiter.
939    pub fn backend(&self) -> BackendKind {
940        match self.waiter() {
941            Some(w) => w.backend(),
942            None => BackendKind::Cpu,
943        }
944    }
945
946    /// Borrow the per-variant waiter, if any. `Cpu` and `Noop` return
947    /// `None`.
948    pub fn waiter(&self) -> Option<&Arc<dyn SyncWaiter>> {
949        match self {
950            Self::Vulkan { waiter, .. }
951            | Self::D3D12 { waiter, .. }
952            | Self::D3D11 { waiter, .. }
953            | Self::Cuda { waiter, .. }
954            | Self::CudaEvent { waiter, .. }
955            | Self::OpenCl { waiter, .. }
956            | Self::Metal { waiter, .. }
957            | Self::OpenGL { waiter, .. }
958            | Self::OpenGLSync { waiter, .. } => Some(waiter),
959            #[cfg(feature = "wgpu")]
960            Self::Wgpu { waiter, .. } => Some(waiter),
961            #[cfg(feature = "wgpu")]
962            Self::DeferredWgpu { waiter, .. } => Some(waiter),
963            Self::Cpu | Self::Noop => None,
964        }
965    }
966}
967
968impl core::fmt::Debug for SyncPoint {
969    fn fmt(&self, f: &mut core::fmt::Formatter<'_>) -> core::fmt::Result {
970        match self {
971            Self::Vulkan { semaphore, kind, value, .. } => f
972                .debug_struct("Vulkan")
973                .field("semaphore", semaphore)
974                .field("kind", kind)
975                .field("value", value)
976                .finish(),
977            Self::D3D12 { value, .. } => f.debug_struct("D3D12").field("value", value).finish(),
978            Self::D3D11 { key, .. } => f.debug_struct("D3D11").field("key", key).finish(),
979            Self::Cuda { value, .. } => f.debug_struct("Cuda").field("value", value).finish(),
980            Self::CudaEvent { event, device, .. } => {
981                f.debug_struct("CudaEvent").field("event", &(*event as usize)).field("device", device).finish()
982            }
983            Self::OpenCl { event, value, .. } => {
984                f.debug_struct("OpenCl").field("event", &(*event as usize)).field("value", value).finish()
985            }
986            Self::Metal { value, .. } => f.debug_struct("Metal").field("value", value).finish(),
987            Self::OpenGL { semaphore, value, .. } => {
988                f.debug_struct("OpenGL").field("semaphore", semaphore).field("value", value).finish()
989            }
990            Self::OpenGLSync { glsync, backend, .. } => {
991                f.debug_struct("OpenGLSync").field("glsync", &(*glsync as usize)).field("backend", backend).finish()
992            }
993            #[cfg(feature = "wgpu")]
994            Self::Wgpu { .. } => f.write_str("SyncPoint::Wgpu"),
995            #[cfg(feature = "wgpu")]
996            Self::DeferredWgpu { slot, .. } => {
997                f.debug_struct("DeferredWgpu").field("committed", &slot.get().is_some()).finish()
998            }
999            Self::Cpu => f.write_str("SyncPoint::Cpu"),
1000            Self::Noop => f.write_str("SyncPoint::Noop"),
1001        }
1002    }
1003}
1004
1005/// Runtime-agnostic single-yield future. Returns `Pending` once then
1006/// resolves to `Ready(())` on the next poll, with `wake_by_ref` driving
1007/// the re-poll. Works under any executor (`pollster`, `tokio`, `smol`,
1008/// `futures::executor`).
1009///
1010/// Used by the default `SyncWaiter::wait_async` impl and by per-backend
1011/// overrides for their fast-path yield-spin loops.
1012pub async fn yield_once() {
1013    let mut yielded = false;
1014    core::future::poll_fn(|cx| {
1015        if yielded {
1016            core::task::Poll::Ready(())
1017        } else {
1018            yielded = true;
1019            cx.waker().wake_by_ref();
1020            core::task::Poll::Pending
1021        }
1022    })
1023    .await
1024}
1025
1026/// Helper: convert a `Duration` to nanoseconds, saturating at
1027/// `u64::MAX` for "wait forever". Matches `vkWaitSemaphores`'
1028/// `UINT64_MAX` infinite-wait sentinel.
1029pub fn duration_to_ns(t: Duration) -> u64 {
1030    u64::try_from(t.as_nanos()).unwrap_or(u64::MAX)
1031}
1032
1033/// Helper: convert a `Duration` to milliseconds, saturating at
1034/// `u32::MAX` for the Win32 `INFINITE` sentinel
1035/// (`WaitForSingleObject`, `WaitForMultipleObjects`).
1036pub fn duration_to_ms_u32(t: Duration) -> u32 {
1037    u32::try_from(t.as_millis()).unwrap_or(u32::MAX)
1038}