gpu_handle_types/sync.rs
1// SPDX-License-Identifier: MIT OR Apache-2.0
2
3use core::future::Future;
4use core::pin::Pin;
5use std::any::Any;
6use std::ffi::c_void;
7use std::sync::Arc;
8use std::time::Duration;
9
10#[cfg(feature = "wgpu")]
11use crate::SliceOutcome;
12use crate::{BackendKind, Error, GlBackend};
13
14/// Boxed dyn-future return shape used by `SyncWaiter::wait_async` so the
15/// trait stays object-safe through `Arc<dyn SyncWaiter>`. Native
16/// `async fn` on the trait would be RPITIT and dyn-incompatible by
17/// default; callers dispatch through `Arc<dyn SyncWaiter>`, so boxing the
18/// future is mandatory.
19// `Send` off wasm; dropped on wasm, where the `SyncWaiter`
20// the future borrows is thread-affine (`&dyn SyncWaiter` is not `Send`
21// once the waiter is `!Sync`). An explicit `+ Send` on a `dyn Future`
22// cannot be spelled with a non-auto marker trait, so the alias is
23// cfg-split directly.
24#[cfg(not(target_family = "wasm"))]
25pub type BoxFuture<'a, T> = Pin<Box<dyn Future<Output = T> + Send + 'a>>;
26#[cfg(target_family = "wasm")]
27pub type BoxFuture<'a, T> = Pin<Box<dyn Future<Output = T> + 'a>>;
28
29/// Default wait budget for `SyncPoint::wait()` and
30/// `SyncPoint::wait_blocking()` — 10 seconds.
31pub const DEFAULT_WAIT_TIMEOUT: Duration = Duration::from_secs(10);
32
33/// A far-future but always-representable wait cap. `Instant + Duration` overflows
34/// the platform's representable range only for a `timeout` so large it is
35/// effectively infinite (near `Duration::MAX` — an `Instant` spans hundreds of
36/// billions of years), and `Instant` has no `saturating_add`. Rather than let such
37/// a `timeout` silently drop its deadline — turning a bounded wait into a hang —
38/// [`wait_deadline`] clamps to `now +` this. It sits far beyond any real GPU wait
39/// (budgets are seconds), so a legitimately huge finite wait still resolves against
40/// it as intended.
41const SATURATED_WAIT_CAP: Duration = Duration::from_secs(60 * 60 * 24 * 365); // ~1 year
42
43/// The absolute deadline a `wait(timeout)` poll loop should honour, or `None` for
44/// the [`SyncWaiter::wait`] contract's `Duration::MAX` "wait forever" sentinel.
45///
46/// - `timeout == Duration::MAX` → `None`: the caller asked to wait forever, so the
47/// loop runs with no deadline (matching every backend's explicit `Duration::MAX`
48/// handling, which maps to the native "infinite": `vkWaitSemaphores(UINT64_MAX)`,
49/// `WaitForSingleObject(INFINITE)`, `cuEventSynchronize`, …).
50/// - any other `timeout` → `Some(now + timeout)`, **saturated** to a bounded,
51/// always-representable [`Instant`](std::time::Instant) on `Instant + Duration`
52/// overflow — clamped to `now + 1 year`, far beyond any real GPU wait budget.
53///
54/// This keeps the two meanings of "no deadline" distinct: a returned `None`
55/// encodes ONLY the explicit forever sentinel, never an accidental `checked_add`
56/// overflow — which would otherwise silently convert a bounded wait into an
57/// unbounded hang.
58pub fn wait_deadline(timeout: Duration) -> Option<std::time::Instant> {
59 if timeout == Duration::MAX {
60 return None;
61 }
62 let now = std::time::Instant::now();
63 Some(now.checked_add(timeout).unwrap_or_else(|| now.checked_add(SATURATED_WAIT_CAP).unwrap_or(now)))
64}
65
66/// Backend-specific wait dispatch + lifetime anchor for a [`SyncPoint`].
67///
68/// Every non-trivial `SyncPoint` variant carries an `Arc<dyn SyncWaiter>`
69/// that:
70///
71/// 1. **Owns the underlying primitive's lifetime** — the timeline
72/// semaphore, D3D12 fence, `MTLSharedEvent`, `cl_event`, `GLsync`,
73/// etc. Cloning the `SyncPoint` clones the `Arc`; the primitive lives
74/// as long as any clone outlives. This rules out the "consumer holds
75/// a `*mut c_void` after the producer dropped" footgun.
76///
77/// 2. **Provides the wait implementation** — `SyncPoint::wait` /
78/// `is_signaled` / `backend` dispatch through this trait, with no
79/// global registries and no `OnceLock<fn>` runtime callbacks.
80///
81/// Implementations live in the producer crate (typically an interop
82/// layer's per-backend module) and are constructed at the same site that
83/// mints the `SyncPoint`.
84// `MaybeSendSync` = `Send + Sync` off wasm; empty on wasm,
85// where a waiter can hold a thread-affine `wgpu::Device` /
86// `web_sys` handle. See [`crate::MaybeSendSync`].
87pub trait SyncWaiter: crate::MaybeSendSync + 'static {
88 /// Block until the GPU work this sync point represents has
89 /// completed, or until `timeout` elapses.
90 ///
91 /// `Duration::MAX` means "wait forever" — implementations should
92 /// translate this to the platform's native "infinite" sentinel
93 /// (`UINT64_MAX` for `vkWaitSemaphores`, `INFINITE` for
94 /// `WaitForSingleObject`, etc.).
95 ///
96 /// # Timeout granularity
97 ///
98 /// Native wait APIs vary in granularity, and implementations
99 /// preserve the caller's intent at the expense of *requested*
100 /// (not realised) latency on sub-API-tick timeouts:
101 ///
102 /// - **Metal** (`MTLSharedEvent::waitUntilSignaledValue:timeoutMS:`)
103 /// accepts millisecond-granular `u64`. Sub-millisecond positive
104 /// timeouts (e.g. `Duration::from_micros(100)`) round **up**
105 /// to 1 ms — truncation to 0 would silently behave as
106 /// "do not wait". A high-frequency progress poll using
107 /// `Duration::from_micros(100)` therefore observes ~10× the
108 /// requested latency on a not-yet-signaled event. For
109 /// non-blocking probes use [`SyncWaiter::is_signaled`] (which
110 /// passes `Duration::ZERO` and short-circuits on the
111 /// `signaledValue() >= value` accessor — driver-side, no
112 /// timer).
113 /// - **Win32** (`WaitForSingleObject`) is also millisecond-
114 /// granular; the same round-up applies.
115 /// - **Vulkan** (`vkWaitSemaphores`), **D3D12 fence**, **CUDA
116 /// external semaphore**, **OpenCL** are nanosecond-granular and
117 /// honour the caller's `Duration` exactly.
118 ///
119 /// Callers that need *both* a deterministic poll cadence under
120 /// 1 ms *and* a real wait fallback should compose the two —
121 /// e.g. `is_signaled` in a hot loop with their own `Instant`
122 /// budget, then a single coarse `wait(Duration::from_millis(N))`
123 /// when the budget is exhausted.
124 fn wait(&self, timeout: Duration) -> Result<(), Error>;
125
126 /// Async sibling of [`wait`](Self::wait). Returns a boxed future so
127 /// the trait stays object-safe through `Arc<dyn SyncWaiter>` — native
128 /// async-fn-in-trait is RPITIT and dyn-incompatible by default; callers
129 /// dispatch through `Arc<dyn SyncWaiter>` everywhere, so the explicit
130 /// `Pin<Box<...>>` is mandatory.
131 ///
132 /// # Default implementation
133 ///
134 /// Backends without a hand-tuned implementation get a two-phase
135 /// hybrid: a short cooperative-yield spin (catches sub-ms waits
136 /// without burning a thread) followed by an iteration-bounded
137 /// `is_signaled` probe loop. Per-backend impls SHOULD override with
138 /// their native blocking-with-timeout primitive on a dedicated
139 /// waiter thread for waits the spin loop didn't catch.
140 ///
141 /// The default implementation is correct (resolves when signalled,
142 /// returns `Error::Timeout` after `timeout`, never blocks the
143 /// executor) but not optimal: it busy-polls inside the spin loop
144 /// and falls back to a `Duration::ZERO` probe loop after that. For
145 /// production paths, override with a backend-specific impl.
146 fn wait_async<'a>(&'a self, timeout: Duration) -> BoxFuture<'a, Result<(), Error>> {
147 Box::pin(async move {
148 // Fast-path yield-spin — bounded cooperative-yield probe. Caps iteration
149 // count rather than wall-clock time so executors that re-
150 // poll immediately on `wake_by_ref` (pollster) don't burn
151 // CPU here.
152 const SPIN_ITERATIONS: usize = 64;
153 // `None` iff `timeout == Duration::MAX` (the wait-forever sentinel); every
154 // finite timeout gets a bounded, always-representable deadline — an
155 // `Instant + Duration` overflow cannot silently drop it and hang.
156 let deadline = wait_deadline(timeout);
157 for _ in 0..SPIN_ITERATIONS {
158 match self.is_signaled() {
159 Ok(true) => return Ok(()),
160 Ok(false) => {}
161 Err(e) => return Err(e),
162 }
163 if let Some(d) = deadline
164 && std::time::Instant::now() >= d
165 {
166 return Err(Error::Timeout);
167 }
168 yield_once().await;
169 }
170 // Waiter-thread fallback — default impl probes `is_signaled` in a
171 // yielding loop. Per-backend impls SHOULD override to hand
172 // off to a dedicated waiter thread with the native
173 // blocking-with-timeout primitive.
174 loop {
175 match self.is_signaled() {
176 Ok(true) => return Ok(()),
177 Ok(false) => {}
178 Err(e) => return Err(e),
179 }
180 if let Some(d) = deadline
181 && std::time::Instant::now() >= d
182 {
183 return Err(Error::Timeout);
184 }
185 yield_once().await;
186 }
187 })
188 }
189
190 /// Non-blocking probe.
191 ///
192 /// - `Ok(true)` — the sync point has been reached.
193 /// - `Ok(false)` — work is still in flight (the underlying poll
194 /// timed out at zero).
195 /// - `Err(_)` — driver-level failure (device lost / TDR /
196 /// `wgpu::PollError::WrongSubmissionIndex` / equivalent on other
197 /// backends). Callers polling in a loop **must** break on `Err` —
198 /// folding errors into `Ok(false)` would loop forever on a TDR'd
199 /// device.
200 ///
201 /// Default implementation calls `wait(Duration::ZERO)` and maps
202 /// `Err(Error::Timeout)` → `Ok(false)`; every other `Err` is
203 /// propagated. Backends override when they have a more direct
204 /// "is this primitive currently signaled" probe (e.g.
205 /// `vkGetSemaphoreCounterValue`, `ID3D12Fence::GetCompletedValue`)
206 /// that distinguishes "not signaled" from real errors without
207 /// going through the wait path.
208 fn is_signaled(&self) -> Result<bool, Error> {
209 match self.wait(Duration::ZERO) {
210 Ok(()) => Ok(true),
211 Err(Error::Timeout) => Ok(false),
212 Err(other) => Err(other),
213 }
214 }
215
216 /// Backend identity for routing decisions on the consumer side.
217 fn backend(&self) -> BackendKind;
218
219 /// Downcast hook for cross-API bridges that need access to the
220 /// concrete waiter type — e.g. a Vulkan→CUDA bridge that wants to
221 /// pull the `VkDevice` out of the `VulkanWaiter` to issue its own
222 /// `cuImportExternalSemaphore`. Most callers use the per-variant
223 /// raw fields on `SyncPoint` directly and never need this.
224 fn as_any(&self) -> &dyn Any;
225
226 /// [`CudaEventWaiter`] view of this waiter, when the concrete type
227 /// implements it. Default `None`.
228 ///
229 /// This is the trait-object route to the CUDA-only
230 /// [`CudaEventWaiter::wait_on_foreign_stream`] extension: `Any` can
231 /// only downcast to *concrete* types, which consumers in other
232 /// crates cannot name — so producers whose waiter implements
233 /// [`CudaEventWaiter`] override this with `Some(self)` and
234 /// consumers (e.g. a CUDA import path gating its private copy stream
235 /// on a producer's `SyncPoint::CudaEvent`) reach the extension method
236 /// without knowing the concrete type.
237 fn as_cuda_event_waiter(&self) -> Option<&dyn CudaEventWaiter> {
238 None
239 }
240
241 /// Rebuild this waiter bound to `value` instead of the value it was
242 /// constructed with, reusing the same underlying primitive (and the
243 /// same lifetime anchors).
244 ///
245 /// # Why this exists
246 ///
247 /// [`wait`](Self::wait) / [`is_signaled`](Self::is_signaled) take no
248 /// value argument — the waiter *embeds* the value it resolves at,
249 /// captured when it was built. So a [`SyncPoint`] whose `value`
250 /// field was substituted while its `waiter` was cloned verbatim has
251 /// two surfaces that disagree: the GPU side (a consumer reading
252 /// `SyncPoint::*.value` to stage a queue wait) waits the new value,
253 /// while the CPU side silently resolves at the old one — reporting
254 /// "already signalled" for a signal the producer has not emitted.
255 ///
256 /// Any code that mints a `SyncPoint` at a value other than the one a
257 /// template was built with MUST route the waiter through this method
258 /// rather than cloning it, so the two surfaces cannot diverge.
259 ///
260 /// # Contract
261 ///
262 /// - `Some(w)` — `w` waits `value` on the same primitive this waiter
263 /// waits on, and holds the same keep-alive chain. Implementations
264 /// MUST NOT carry over state that is only valid for the original
265 /// value (e.g. a captured submission index that retires at it).
266 /// - `None` (the default) — the underlying primitive carries no
267 /// value, so there is nothing to rebind: binary semaphores,
268 /// fences, `GLsync`, `cl_semaphore_khr`. A `None` here is not a
269 /// failure; it means the CPU surface has no value to disagree
270 /// about.
271 fn rebind_to_value(&self, value: u64) -> Option<Arc<dyn SyncWaiter>> {
272 let _ = value;
273 None
274 }
275}
276
277/// CUDA-event-specific extension trait for the
278/// [`SyncPoint::CudaEvent`] variant. Supertrait of [`SyncWaiter`] so a
279/// `CudaEventWaiter` always satisfies the generic [`SyncPoint::wait`]
280/// dispatch path; adds the CUDA-only `wait_on_foreign_stream` extension
281/// method that cross-API bridges downcast to via
282/// [`SyncWaiter::as_any`].
283///
284/// # Drop discipline
285///
286/// Implementations MUST push the event's owning `CUcontext` before
287/// calling `cuEventDestroy_v2`, and MUST pop iff the push succeeded
288/// (pop-without-push would silently consume whatever the calling
289/// thread had on top of its context stack). `cuEventDestroy_v2`
290/// requires the owning context to still exist; the push is defensive
291/// against driver-internal cleanup paths that probe `cuCtxGetCurrent`
292/// and emit confusing diagnostics when no context is current.
293///
294/// # Cross-context wait
295///
296/// `cuEventRecord` is strict-same-context — a `CUevent` cannot be
297/// recorded against a stream from a different `CUcontext`. But
298/// `cuStreamWaitEvent` is cross-context (NVIDIA Driver API contract).
299/// `wait_on_foreign_stream` is the cross-context wait entry point:
300/// the consumer's CUDA stream may live in a different `CUcontext`
301/// than the event, and the waiter issues `cuStreamWaitEvent` so the
302/// foreign stream gates on this event's completion without a CPU
303/// bounce.
304///
305/// A waiter backed by the CUDA driver API implements this trait
306/// concretely; this trait only defines the contract.
307pub trait CudaEventWaiter: SyncWaiter {
308 /// Issue `cuStreamWaitEvent(foreign_stream, self.event, 0)` so
309 /// `foreign_stream` gates on this event's completion. The stream
310 /// may belong to a different `CUcontext` than the event's owning
311 /// context — `cuStreamWaitEvent` is cross-context per the NVIDIA
312 /// Driver API.
313 ///
314 /// `foreign_stream` is a raw `CUstream` pointer (the null pointer
315 /// resolves to the calling thread's default stream).
316 ///
317 /// Returns [`Error::NotSupported`] if the CUDA driver loader is
318 /// unavailable or the wait dispatch returns a driver error; the
319 /// concrete error wording is implementation-defined.
320 fn wait_on_foreign_stream(&self, foreign_stream: *mut c_void) -> Result<(), Error>;
321}
322
323/// Discriminator for [`SyncPoint::Vulkan`]'s `VkSemaphore` flavour.
324///
325/// `VkSemaphore` ships in two shapes — binary (one-shot, signal-and-
326/// reset) and timeline (monotonic 64-bit payload, multi-waiter,
327/// concurrent-signal-safe). A consumer staging a wait must route
328/// through `VkTimelineSemaphoreSubmitInfo` for timeline semaphores and
329/// must NOT for binary semaphores — passing the wrong shape risks
330/// validation errors, driver-quirk-dependent wrong waits, and (on
331/// already-consumed binary signals) hard deadlock at submit time.
332///
333/// The Vulkan API has no public probe that safely distinguishes the
334/// two (`vkGetSemaphoreCounterValue` is undefined on binaries per
335/// `VUID-vkGetSemaphoreCounterValue-semaphore-03255`), so the producer
336/// must declare intent at mint time.
337#[derive(Copy, Clone, Debug, PartialEq, Eq)]
338#[non_exhaustive]
339pub enum VulkanSemaphoreKind {
340 /// `vk::SemaphoreType::TIMELINE`. `value` is a payload coordinate;
341 /// any number of waiters may target the same value, and signals
342 /// with strictly-increasing payloads can be queued concurrently.
343 Timeline,
344 /// `vk::SemaphoreType::BINARY`. Single-shot — exactly one signal
345 /// must pair with exactly one wait. The `value` field is ignored
346 /// (set to `0` by convention). Producers carrying binaries here
347 /// MUST guarantee the wait is the only consumer of the signal.
348 Binary,
349}
350
351/// Shared, mutate-once slot the producer fills in after the caller
352/// submits the encoded work. Used by [`SyncPoint::DeferredWgpu`].
353///
354/// The slot is initialised to `None` at mint time and transitions to
355/// `Some(idx)` exactly once via [`Self::set`]; subsequent calls are
356/// rejected with [`Error::InvalidArgument`] (single-shot semantics).
357#[cfg(feature = "wgpu")]
358#[derive(Debug, Default)]
359pub struct DeferredWgpuSlot {
360 cell: std::sync::OnceLock<wgpu::SubmissionIndex>,
361}
362
363#[cfg(feature = "wgpu")]
364impl DeferredWgpuSlot {
365 pub fn new() -> Self {
366 Self::default()
367 }
368
369 /// Commit the post-submit `SubmissionIndex`. Single-shot — a second
370 /// call returns [`Error::InvalidArgument`] without overwriting the
371 /// committed value.
372 pub fn set(&self, idx: wgpu::SubmissionIndex) -> Result<(), Error> {
373 self.cell.set(idx).map_err(|_| Error::InvalidArgument("DeferredWgpuSlot already committed".into()))
374 }
375
376 /// Return the committed `SubmissionIndex`, or `None` if the producer
377 /// has not yet submitted.
378 pub fn get(&self) -> Option<&wgpu::SubmissionIndex> {
379 self.cell.get()
380 }
381}
382
383// Defensive — the `OnceLock<wgpu::SubmissionIndex>` design relies on
384// `SubmissionIndex` being `Clone + Send + Sync`. If a wgpu upgrade
385// silently dropped any of these bounds, the OnceLock-storing
386// `DeferredWgpuSlot` would surface a confusing trait-bound error far
387// from the cause; this assertion fails compilation here, with a clear
388// message, instead.
389#[cfg(feature = "wgpu")]
390static_assertions::assert_impl_all!(wgpu::SubmissionIndex: Clone, Send, Sync);
391
392/// GPU sync primitive. Produced by every submit-side path; consumed by
393/// anything that needs to serialise on a prior GPU submission.
394///
395/// Cross-backend bridges are deliberately absent from this leaf crate —
396/// those belong in an interop layer, where the target-device context is
397/// available.
398///
399/// # Lifetime
400///
401/// The per-variant `waiter: Arc<dyn SyncWaiter>` field anchors the
402/// lifetime of any raw primitive the variant exposes (`*mut c_void`
403/// fence pointers, `u64` semaphore handles, `GLsync` opaques, etc.).
404/// Cloning a `SyncPoint` is `Arc::clone` on the waiter — cheap, no
405/// driver round-trip. The raw fields are guaranteed to remain valid
406/// for as long as any clone is alive.
407#[derive(Clone)]
408#[non_exhaustive]
409pub enum SyncPoint {
410 /// Vulkan semaphore. `device` is the `VkDevice` the semaphore was
411 /// created on, exposed for bridges that need to re-import it. The
412 /// `waiter` owns whatever anchors the `VkSemaphore`'s lifetime.
413 ///
414 /// `kind` distinguishes timeline vs binary semaphores —
415 /// see [`VulkanSemaphoreKind`]. `value` is the timeline payload
416 /// coordinate when `kind == Timeline`; for `kind == Binary` the
417 /// field is meaningless (set to `0` by convention) and consumers
418 /// route through a binary wait (with wgpu-hal,
419 /// `vulkan::Queue::add_wait_semaphore(_, None, _)`).
420 Vulkan { semaphore: u64, kind: VulkanSemaphoreKind, value: u64, device: *mut c_void, waiter: Arc<dyn SyncWaiter> },
421 /// D3D12 fence sync. `fence` is an `ID3D12Fence*` (COM pointer);
422 /// `value` is the monotonic signal value the consumer waits on
423 /// (`fence->GetCompletedValue() >= value`). The `waiter` holds the
424 /// COM reference that keeps the fence alive.
425 ///
426 /// `value` is monotonically signaled by the producer. Bridges that
427 /// reuse a single shared fence across calls serialise the
428 /// (mint + `Signal`) pair under a value-lock so concurrent bridges
429 /// always sequence in monotonic order.
430 /// A consumer waiting on `value=N` is guaranteed to pass once a
431 /// later `value=M >= N` has been signaled, even if the producer's
432 /// counter has since advanced past `M`.
433 D3D12 { fence: *mut c_void, value: u64, waiter: Arc<dyn SyncWaiter> },
434 /// D3D11 keyed-mutex sync. `keyed_mutex` is an `IDXGIKeyedMutex*`;
435 /// `key` is the integer key the producer released to. The
436 /// `waiter` holds the COM reference that keeps the mutex alive.
437 D3D11 { keyed_mutex: *mut c_void, key: u64, waiter: Arc<dyn SyncWaiter> },
438 /// CUDA external semaphore. `event` is a `CUexternalSemaphore`;
439 /// `value` is `Some(v)` for timeline waits — including the products
440 /// of Vulkan→CUDA / D3D12→CUDA bridges, which carry the producer's
441 /// timeline value so the consumer can enqueue
442 /// `cuWaitExternalSemaphoresAsync` on whichever CUDA stream their
443 /// work runs on. `None` is reserved for setup-metadata SyncPoints
444 /// where the consumer mints the actual signal value at emission time
445 /// and doesn't need a resolvable wait at construction.
446 Cuda { event: *mut c_void, value: Option<u64>, waiter: Arc<dyn SyncWaiter> },
447 /// Intra-CUDA event. `event` is a `CUevent` recorded on the
448 /// producing stream via `cuEventRecord`. Consumers wait via
449 /// `cuStreamWaitEvent` (chained CUDA streams — cross-context
450 /// supported per NVIDIA Driver API) or `cuEventSynchronize`
451 /// (blocking CPU wait).
452 ///
453 /// Distinct from [`Self::Cuda`] — that variant carries a
454 /// `CUexternalSemaphore` waited via `cuWaitExternalSemaphoresAsync`.
455 /// `CUevent` and `CUexternalSemaphore` are different opaque types
456 /// in the CUDA Driver API; passing a `CUevent` into the external-
457 /// semaphore wait path returns `CUDA_ERROR_INVALID_HANDLE`.
458 ///
459 /// Cross-API consumers cannot wait on a `CUevent` directly; they
460 /// need a `CUexternalSemaphore` (the [`Self::Cuda`] shape); an
461 /// interop layer performs that conversion when the chain spans a
462 /// backend boundary.
463 ///
464 /// `context` is the event's **owning** `CUcontext` — bound at
465 /// `cuEventCreate` time per the NVIDIA Driver API ("creates an event
466 /// for the current context"). The owning context must still exist
467 /// at destroy time or `cuEventDestroy_v2` returns
468 /// `CUDA_ERROR_INVALID_CONTEXT`. There is no `cuEventGetCtx` API —
469 /// the metadata must travel with the event.
470 ///
471 /// `device` is the CUDA ordinal the owning context was created on
472 /// (derivable from `context` via `cuCtxGetDevice` but cheap to
473 /// carry and matches the device-keyed bridge shape used by
474 /// [`Self::Cuda`]).
475 ///
476 /// `waiter` implements [`CudaEventWaiter`] — a supertrait of
477 /// [`SyncWaiter`] that adds the CUDA-specific
478 /// `wait_on_foreign_stream` extension method and the push-owning-
479 /// context-before-destroy Drop discipline. The leaf crate stores
480 /// the upcasted `Arc<dyn SyncWaiter>` so the generic [`Self::waiter`]
481 /// dispatch path is uniform with every other variant; cross-API
482 /// bridges that need the CUDA extension methods downcast through
483 /// [`SyncWaiter::as_any`].
484 CudaEvent { event: *mut c_void, context: *mut c_void, device: i32, waiter: Arc<dyn SyncWaiter> },
485 /// OpenCL sync. `event` is either a `cl_event` (binary,
486 /// signaled-on-completion; `value` is `None`) or a
487 /// `cl_semaphore_khr` aliasing a timeline (`value` is the
488 /// caller-signaled timeline value). The `waiter` keeps the
489 /// producing context alive.
490 OpenCl { event: *mut c_void, value: Option<u64>, waiter: Arc<dyn SyncWaiter> },
491 /// Metal shared event. `event` is an `MTLSharedEvent*`; `value` is
492 /// the signal value the consumer waits on. The `waiter` holds the
493 /// reference that keeps the event alive.
494 Metal { event: *mut c_void, value: u64, waiter: Arc<dyn SyncWaiter> },
495 /// OpenGL semaphore (`GLuint` produced by `glGenSemaphoresEXT`,
496 /// imported from a cross-API OS handle via
497 /// `glImportSemaphoreFdEXT` / `glImportSemaphoreWin32HandleEXT`).
498 ///
499 /// `value` is `Some(_)` for timeline-equivalent semaphores
500 /// (D3D12_FENCE_EXT handle types); `None` for plain binary
501 /// (OPAQUE_FD / OPAQUE_WIN32) — the GL driver latches the D3D12
502 /// fence value at wait/signal time via
503 /// `glSemaphoreParameterui64vEXT(sem, GL_D3D12_FENCE_VALUE_EXT, &v)`.
504 ///
505 /// `context` is the caller's `EGLContext` / `HGLRC` / `CGLContextObj`
506 /// this semaphore was created against. Waits must be dispatched on
507 /// a thread where the same context is current.
508 OpenGL { semaphore: u32, value: Option<u64>, context: *mut c_void, waiter: Arc<dyn SyncWaiter> },
509 /// GL fence sync. `glsync` is a `GLsync` opaque pointer produced
510 /// by `glFenceSync(GL_SYNC_GPU_COMMANDS_COMPLETE, 0)`.
511 ///
512 /// Distinct from [`Self::OpenGL`] — that one carries an *imported*
513 /// GL semaphore (`GLuint`); this one is the raw GL-native fence.
514 ///
515 /// `backend` records the [`GlBackend`] flavour of the GL context that
516 /// produced this `GLsync`. Cross-API consumers (an OpenGL→X bridge)
517 /// route through it to pick the matching loader (`wglGetProcAddress`
518 /// vs `eglGetProcAddress`) when resolving `glClientWaitSync` for the
519 /// CPU-fallback path.
520 /// Without it the consumer would hard-code `Desktop`, which silently
521 /// resolves the wrong proc-address on EGL / ANGLE / Android hosts and
522 /// stalls forever on a never-signalled handle.
523 OpenGLSync { glsync: *mut c_void, backend: GlBackend, waiter: Arc<dyn SyncWaiter> },
524 /// `wgpu::Queue::submit` return value. Wait dispatches through the
525 /// `waiter`, which holds a `wgpu::Device` clone and calls
526 /// `device.poll(PollType::Wait { submission_index: Some(..), .. })`.
527 #[cfg(feature = "wgpu")]
528 Wgpu { submission_index: wgpu::SubmissionIndex, waiter: Arc<dyn SyncWaiter> },
529 /// `wgpu::Queue::submit` index *not yet known* at SyncPoint mint
530 /// time. Used where the caller owns the encoder and submits after the
531 /// producer has returned; `slot.set(idx)` is called once after
532 /// `queue.submit(...)`, and until then the waiter falls back to
533 /// `device.poll(PollType::Wait { submission_index: None, .. })`
534 /// (full device drain). After commit, waits resolve precisely on
535 /// the recorded `SubmissionIndex`.
536 #[cfg(feature = "wgpu")]
537 DeferredWgpu { slot: Arc<DeferredWgpuSlot>, device: wgpu::Device, waiter: Arc<dyn SyncWaiter> },
538 /// CPU-only work — no GPU sync needed. `wait` returns `Ok(())`
539 /// immediately.
540 Cpu,
541 /// No-op sync — used for pipelines where the consumer side handles
542 /// ordering through a separate channel (e.g. SurfaceControl
543 /// transactions). Behaves identically to `Cpu` for waits.
544 Noop,
545}
546
547// SAFETY: the `*mut c_void` raw fields are immutable after
548// construction and dereferenced only by the per-variant `waiter`,
549// whose `SyncWaiter: Send + Sync` bound asserts platform-correct sharing
550// for whatever the pointer references. Per-variant doc-comments specify
551// the additional caller-asserted preconditions (e.g. D3D11 requires the
552// device's multithread-protect flag).
553unsafe impl Send for SyncPoint {}
554unsafe impl Sync for SyncPoint {}
555
556impl SyncPoint {
557 /// Async wait. Default 10-second timeout.
558 ///
559 /// Async is the default. Use [`wait_blocking`]
560 /// when the caller is on a synchronous code path (sync-loop
561 /// renderers, test harnesses, JNI dispatcher Drop paths).
562 ///
563 /// [`wait_blocking`]: Self::wait_blocking
564 pub async fn wait(&self) -> Result<(), Error> {
565 self.wait_with_timeout_async(DEFAULT_WAIT_TIMEOUT).await
566 }
567
568 /// Async wait with explicit timeout.
569 pub async fn wait_with_timeout_async(&self, timeout: Duration) -> Result<(), Error> {
570 // Binary Vulkan CPU-wait rejection.
571 if let Some(err) = self.binary_vk_cpu_wait_error() {
572 return Err(err);
573 }
574 match self.waiter() {
575 Some(w) => w.wait_async(timeout).await,
576 None => Ok(()),
577 }
578 }
579
580 /// Synchronous wait with the default 10-second timeout.
581 pub fn wait_blocking(&self) -> Result<(), Error> {
582 self.wait_with_timeout(DEFAULT_WAIT_TIMEOUT)
583 }
584
585 /// Synchronous wait with an explicit timeout.
586 ///
587 /// Timeout granularity is backend-dependent — see
588 /// [`SyncWaiter::wait`] for the per-backend rounding contract.
589 /// In particular, Metal and Win32 are millisecond-granular and
590 /// round sub-ms positive timeouts up to 1 ms; Vulkan / D3D12 /
591 /// CUDA / OpenCL are nanosecond-granular. For sub-ms polls use
592 /// [`Self::is_signaled`] in a loop with your own `Instant` budget.
593 pub fn wait_with_timeout(&self, timeout: Duration) -> Result<(), Error> {
594 // Binary Vulkan CPU-wait rejection — `vkWaitSemaphores` is
595 // undefined on binary semaphores, and the Khronos spec gives
596 // binary semaphores no CPU-side wait entry point. Producers
597 // that need CPU-waitable binary semaphores MUST attach a
598 // fence-equivalent fallback (a `wgpu::SubmissionIndex` from the
599 // same submit). When the fallback is absent, the backend's waiter
600 // fails both `wait_blocking` and the async `wait()` with
601 // `Error::NotSupported` (see `binary_vk_cpu_wait_error`).
602 if let Some(err) = self.binary_vk_cpu_wait_error() {
603 return Err(err);
604 }
605 match self.waiter() {
606 Some(w) => w.wait(timeout),
607 None => Ok(()),
608 }
609 }
610
611 /// Leaf-crate hook for rejecting a CPU wait on a binary
612 /// `SyncPoint::Vulkan` whose producer did not attach a
613 /// fence-equivalent fallback. Always `None`: whether the fallback is
614 /// attached is visible only to the backend's waiter, which is where
615 /// that half of the binary-Vulkan CPU-wait contract is enforced.
616 fn binary_vk_cpu_wait_error(&self) -> Option<Error> {
617 // The fence-equivalent fallback lives on the per-backend Vulkan
618 // waiter impl; this crate only sees `Arc<dyn SyncWaiter>`. The
619 // contract is that `wait` / `wait_async` on a binary semaphore
620 // without a fallback return `Error::NotSupported`. Backends MUST
621 // honour it; the "fallback present" check is not observable here,
622 // so this returns `None` rather than gate on it.
623 let _ = matches!(self, Self::Vulkan { kind: VulkanSemaphoreKind::Binary, .. });
624 None
625 }
626
627 /// Sequence two sync points. Returns a `SyncPoint` whose wait
628 /// completes only after both `self` and `then` have completed.
629 ///
630 /// If either side is `Cpu` / `Noop`, the other is returned unchanged.
631 /// Otherwise the result is an **opaque** carrier: a small adapter
632 /// waiter drives both inner waits in sequence (one heap allocation),
633 /// and it is stored in the [`SyncPoint::Cuda`] variant with a null
634 /// `event` and no `value`. Only use the result through
635 /// [`wait`](Self::wait), [`wait_blocking`](Self::wait_blocking),
636 /// [`is_signaled`](Self::is_signaled), [`backend`](Self::backend)
637 /// (the first sync point's backend) and [`waiter`](Self::waiter);
638 /// its raw fields carry nothing.
639 pub fn chain(self, then: SyncPoint) -> SyncPoint {
640 // Trivial cases — skip the adapter entirely.
641 if matches!(self, Self::Cpu | Self::Noop) {
642 return then;
643 }
644 if matches!(then, Self::Cpu | Self::Noop) {
645 return self;
646 }
647 let backend = self.backend();
648 let waiter: Arc<dyn SyncWaiter> = Arc::new(ChainWaiter { first: self, then, backend });
649 Self::Cpu.with_chain_waiter(waiter)
650 }
651
652 /// Wraps a `ChainWaiter` in the carrier variant `chain` documents.
653 /// Used only by `chain`.
654 fn with_chain_waiter(self, waiter: Arc<dyn SyncWaiter>) -> SyncPoint {
655 // Any variant with a `waiter` field would do: every caller-facing
656 // method dispatches through the waiter, so the event / value
657 // fields are never read.
658 let _ = self;
659 Self::Cuda { event: core::ptr::null_mut(), value: None, waiter }
660 }
661}
662
663// ─────────────────────────────────────────────────────────────────────────────
664// Raw-handle reconstruction (cross-C / cross-instance transport carriers)
665// ─────────────────────────────────────────────────────────────────────────────
666
667/// A [`SyncWaiter`] for a [`SyncPoint`] **reconstructed from a raw foreign fence
668/// handle** that has not yet been imported onto a waiting device.
669///
670/// A D3D12 fence / Vulkan semaphore / Metal shared event that crossed an API or
671/// process boundary as a raw handle (e.g. a `#[repr(C)]` descriptor over a C ABI)
672/// carries no device the leaf crate can drive a wait on — the real CPU/GPU wait
673/// only becomes possible once a consumer **imports** the handle onto its own device
674/// (the importer stages the device-side `vkWaitSemaphores` /
675/// `ID3D12CommandQueue::Wait` / `encodeWaitForEvent` off the carried raw fields).
676///
677/// So a `SyncPoint` reconstructed via [`SyncPoint::from_raw_d3d12_fence`] /
678/// [`SyncPoint::from_raw_vulkan_semaphore`] / [`SyncPoint::from_raw_metal_event`] is
679/// a **transport carrier**: its raw fields are consumed by the importer, and its
680/// `waiter` is this carrier. Calling [`SyncPoint::wait_blocking`] / `wait` /
681/// `is_signaled` on the carrier (before import) returns [`Error::NotSupported`] —
682/// fail-closed, never a silent "already signaled" — because the leaf crate cannot
683/// wait a fence it has no device for. Import the handle first; the import path
684/// produces the real, waitable `SyncPoint`.
685#[derive(Debug)]
686pub struct RawFenceCarrierWaiter {
687 backend: BackendKind,
688}
689
690impl RawFenceCarrierWaiter {
691 fn new(backend: BackendKind) -> Self {
692 Self { backend }
693 }
694
695 fn not_imported() -> Error {
696 Error::NotSupported(std::borrow::Cow::Borrowed(
697 "SyncPoint reconstructed from a raw foreign fence handle is a transport \
698 carrier — import it onto a device before waiting",
699 ))
700 }
701}
702
703impl SyncWaiter for RawFenceCarrierWaiter {
704 fn wait(&self, _timeout: Duration) -> Result<(), Error> {
705 Err(Self::not_imported())
706 }
707
708 fn is_signaled(&self) -> Result<bool, Error> {
709 Err(Self::not_imported())
710 }
711
712 fn backend(&self) -> BackendKind {
713 self.backend
714 }
715
716 fn as_any(&self) -> &dyn Any {
717 self
718 }
719}
720
721impl SyncPoint {
722 /// Reconstruct a [`SyncPoint::D3D12`] transport carrier from a raw
723 /// `ID3D12Fence*` pointer + signal value that crossed an API / process boundary
724 /// (e.g. a `#[repr(C)]` fence descriptor over a C ABI).
725 ///
726 /// `fence` is an `ID3D12Fence*` COM pointer the caller guarantees is live for
727 /// the carrier's lifetime; `value` is the monotonic value the consumer waits for
728 /// (`fence->GetCompletedValue() >= value`). The returned `SyncPoint` is a
729 /// **carrier** ([`RawFenceCarrierWaiter`]) — its raw `(fence, value)` fields feed
730 /// a device-side import (e.g. an interop layer's acquire-sync path); a CPU
731 /// `wait()` on it before import returns [`Error::NotSupported`].
732 pub fn from_raw_d3d12_fence(fence: *mut c_void, value: u64) -> SyncPoint {
733 let waiter: Arc<dyn SyncWaiter> = Arc::new(RawFenceCarrierWaiter::new(BackendKind::D3D12));
734 SyncPoint::D3D12 { fence, value, waiter }
735 }
736
737 /// Reconstruct a [`SyncPoint::Vulkan`] transport carrier from a raw
738 /// `VkSemaphore` + `VkDevice` that crossed an API / process boundary.
739 ///
740 /// `semaphore` is the raw `VkSemaphore` (a 64-bit non-dispatchable handle);
741 /// `kind` distinguishes timeline vs binary (see [`VulkanSemaphoreKind`]);
742 /// `value` is the timeline coordinate (ignored for `Binary`); `device` is the
743 /// `VkDevice*` the semaphore lives on (needed by an importer that re-exports it
744 /// for a cross-device wait). The returned `SyncPoint` is a **carrier** — its raw
745 /// fields feed a device-side import; a CPU `wait()` before import returns
746 /// [`Error::NotSupported`].
747 pub fn from_raw_vulkan_semaphore(
748 semaphore: u64,
749 kind: VulkanSemaphoreKind,
750 value: u64,
751 device: *mut c_void,
752 ) -> SyncPoint {
753 let waiter: Arc<dyn SyncWaiter> = Arc::new(RawFenceCarrierWaiter::new(BackendKind::Vulkan));
754 SyncPoint::Vulkan { semaphore, kind, value, device, waiter }
755 }
756
757 /// Reconstruct a [`SyncPoint::Metal`] transport carrier from a raw
758 /// `MTLSharedEvent*` + signal value that crossed an API / process boundary.
759 ///
760 /// `event` is an `MTLSharedEvent*` the caller guarantees is live for the
761 /// carrier's lifetime; `value` is the value the consumer waits for. The returned
762 /// `SyncPoint` is a **carrier** — its raw `(event, value)` fields feed a
763 /// device-side import; a CPU `wait()` before import returns
764 /// [`Error::NotSupported`].
765 pub fn from_raw_metal_event(event: *mut c_void, value: u64) -> SyncPoint {
766 let waiter: Arc<dyn SyncWaiter> = Arc::new(RawFenceCarrierWaiter::new(BackendKind::Metal));
767 SyncPoint::Metal { event, value, waiter }
768 }
769}
770
771/// Per-process waiter thread for [`SyncPoint::DeferredWgpu`].
772///
773/// The waiter thread is stateless — every per-instance datum (device
774/// handle, deferred slot) travels inside the `SliceFn` closure — so one
775/// thread serves every deferred sync point in the process.
776#[cfg(feature = "wgpu")]
777fn deferred_wgpu_waiter_thread() -> &'static crate::WaiterThread {
778 static THREAD: std::sync::OnceLock<crate::WaiterThread> = std::sync::OnceLock::new();
779 THREAD.get_or_init(|| crate::WaiterThread::new("wgpu-deferred"))
780}
781
782/// Backend-internal waiter for [`SyncPoint::DeferredWgpu`]. Until the
783/// producer commits a `SubmissionIndex` via [`DeferredWgpuSlot::set`],
784/// `wait` falls back to `device.poll(PollType::Wait { submission_index:
785/// None, .. })` (full device drain). After commit, it passes
786/// `submission_index: Some(idx)` and resolves on that precise index.
787///
788/// `wait_async` routes through [`run_hybrid_wait`] — the fast-path
789/// yield-spin spins on `is_signaled` for sub-millisecond signals, the
790/// waiter-thread fallback hands off to the
791/// per-process waiter thread which issues blocking-with-timeout slices
792/// of [`WAITER_SLICE`] on a dedicated thread.
793#[cfg(feature = "wgpu")]
794pub(crate) struct DeferredWgpuWaiter {
795 slot: Arc<DeferredWgpuSlot>,
796 device: wgpu::Device,
797}
798
799#[cfg(feature = "wgpu")]
800impl DeferredWgpuWaiter {
801 pub(crate) fn new(slot: Arc<DeferredWgpuSlot>, device: wgpu::Device) -> Self {
802 Self { slot, device }
803 }
804
805 /// Single `device.poll(Wait { .. })` issuance keyed off the
806 /// current commit state. Shared between `wait`, `is_signaled`, and
807 /// the per-slice closure produced by `wait_async`.
808 fn poll_slice(&self, timeout: Option<Duration>) -> SliceOutcome {
809 let submission_index = self.slot.get().cloned();
810 match self.device.poll(wgpu::PollType::Wait { submission_index, timeout }) {
811 Ok(_) => SliceOutcome::Signaled,
812 Err(wgpu::PollError::Timeout) => SliceOutcome::TimedOut,
813 Err(e) => {
814 log::warn!("SyncPoint::DeferredWgpu: device.poll returned {e:?}");
815 SliceOutcome::Failed(Error::NotSupported("SyncPoint::DeferredWgpu: device.poll failed".into()))
816 }
817 }
818 }
819}
820
821#[cfg(feature = "wgpu")]
822impl SyncWaiter for DeferredWgpuWaiter {
823 fn wait(&self, timeout: Duration) -> Result<(), Error> {
824 let timeout_arg = if timeout == Duration::MAX { None } else { Some(timeout) };
825 match self.poll_slice(timeout_arg) {
826 SliceOutcome::Signaled => Ok(()),
827 SliceOutcome::TimedOut => Err(Error::Timeout),
828 SliceOutcome::Failed(e) => Err(e),
829 }
830 }
831
832 fn wait_async<'a>(&'a self, timeout: Duration) -> BoxFuture<'a, Result<(), Error>> {
833 let slot = self.slot.clone();
834 let device = self.device.clone();
835 let make_slice = move || -> crate::SliceFn {
836 Box::new(move |slice: Duration| -> SliceOutcome {
837 let submission_index = slot.get().cloned();
838 match device.poll(wgpu::PollType::Wait { submission_index, timeout: Some(slice) }) {
839 Ok(_) => SliceOutcome::Signaled,
840 Err(wgpu::PollError::Timeout) => SliceOutcome::TimedOut,
841 Err(e) => {
842 log::warn!("SyncPoint::DeferredWgpu::wait_async: device.poll returned {e:?}");
843 SliceOutcome::Failed(Error::NotSupported(
844 "SyncPoint::DeferredWgpu::wait_async: device.poll failed".into(),
845 ))
846 }
847 }
848 })
849 };
850 Box::pin(crate::run_hybrid_wait(move || self.is_signaled(), deferred_wgpu_waiter_thread(), timeout, make_slice))
851 }
852
853 fn is_signaled(&self) -> Result<bool, Error> {
854 match self.poll_slice(Some(Duration::ZERO)) {
855 SliceOutcome::Signaled => Ok(true),
856 SliceOutcome::TimedOut => Ok(false),
857 SliceOutcome::Failed(e) => Err(e),
858 }
859 }
860
861 fn backend(&self) -> BackendKind {
862 BackendKind::Wgpu
863 }
864
865 fn as_any(&self) -> &dyn Any {
866 self
867 }
868}
869
870/// Construct a [`SyncPoint::DeferredWgpu`] complete with its waiter.
871///
872/// For producers that hand out a sync point *before* they submit: the
873/// returned SyncPoint's clones all share the same `Arc<DeferredWgpuSlot>`,
874/// so the producer-side `slot.set(idx)` at submit time is visible to every
875/// clone.
876#[cfg(feature = "wgpu")]
877pub fn make_deferred_wgpu_sync_point(slot: Arc<DeferredWgpuSlot>, device: wgpu::Device) -> SyncPoint {
878 let waiter: Arc<dyn SyncWaiter> = Arc::new(DeferredWgpuWaiter::new(slot.clone(), device.clone()));
879 SyncPoint::DeferredWgpu { slot, device, waiter }
880}
881
882struct ChainWaiter {
883 first: SyncPoint,
884 then: SyncPoint,
885 backend: BackendKind,
886}
887
888impl SyncWaiter for ChainWaiter {
889 fn wait(&self, timeout: Duration) -> Result<(), Error> {
890 let start = std::time::Instant::now();
891 self.first.wait_with_timeout(timeout)?;
892 let elapsed = start.elapsed();
893 let remaining = timeout.saturating_sub(elapsed);
894 self.then.wait_with_timeout(remaining)
895 }
896
897 fn wait_async<'a>(&'a self, timeout: Duration) -> BoxFuture<'a, Result<(), Error>> {
898 Box::pin(async move {
899 let start = std::time::Instant::now();
900 self.first.wait_with_timeout_async(timeout).await?;
901 let elapsed = start.elapsed();
902 let remaining = timeout.saturating_sub(elapsed);
903 self.then.wait_with_timeout_async(remaining).await
904 })
905 }
906
907 fn is_signaled(&self) -> Result<bool, Error> {
908 Ok(self.first.is_signaled()? && self.then.is_signaled()?)
909 }
910
911 fn backend(&self) -> BackendKind {
912 self.backend
913 }
914
915 fn as_any(&self) -> &dyn Any {
916 self
917 }
918}
919
920impl SyncPoint {
921 /// Non-blocking probe.
922 ///
923 /// - `Ok(true)` — sync point reached (or trivial `Cpu`/`Noop`).
924 /// - `Ok(false)` — work still in flight.
925 /// - `Err(_)` — driver-level failure (device lost / TDR /
926 /// wrong-submission-index). See [`SyncWaiter::is_signaled`]
927 /// for the rationale on surfacing errors instead of folding
928 /// them into `false`.
929 pub fn is_signaled(&self) -> Result<bool, Error> {
930 match self.waiter() {
931 Some(w) => w.is_signaled(),
932 None => Ok(true),
933 }
934 }
935
936 /// Backend identity. For the trivial variants (`Cpu`, `Noop`) this
937 /// returns [`BackendKind::Cpu`]; otherwise it dispatches to the
938 /// per-variant waiter.
939 pub fn backend(&self) -> BackendKind {
940 match self.waiter() {
941 Some(w) => w.backend(),
942 None => BackendKind::Cpu,
943 }
944 }
945
946 /// Borrow the per-variant waiter, if any. `Cpu` and `Noop` return
947 /// `None`.
948 pub fn waiter(&self) -> Option<&Arc<dyn SyncWaiter>> {
949 match self {
950 Self::Vulkan { waiter, .. }
951 | Self::D3D12 { waiter, .. }
952 | Self::D3D11 { waiter, .. }
953 | Self::Cuda { waiter, .. }
954 | Self::CudaEvent { waiter, .. }
955 | Self::OpenCl { waiter, .. }
956 | Self::Metal { waiter, .. }
957 | Self::OpenGL { waiter, .. }
958 | Self::OpenGLSync { waiter, .. } => Some(waiter),
959 #[cfg(feature = "wgpu")]
960 Self::Wgpu { waiter, .. } => Some(waiter),
961 #[cfg(feature = "wgpu")]
962 Self::DeferredWgpu { waiter, .. } => Some(waiter),
963 Self::Cpu | Self::Noop => None,
964 }
965 }
966}
967
968impl core::fmt::Debug for SyncPoint {
969 fn fmt(&self, f: &mut core::fmt::Formatter<'_>) -> core::fmt::Result {
970 match self {
971 Self::Vulkan { semaphore, kind, value, .. } => f
972 .debug_struct("Vulkan")
973 .field("semaphore", semaphore)
974 .field("kind", kind)
975 .field("value", value)
976 .finish(),
977 Self::D3D12 { value, .. } => f.debug_struct("D3D12").field("value", value).finish(),
978 Self::D3D11 { key, .. } => f.debug_struct("D3D11").field("key", key).finish(),
979 Self::Cuda { value, .. } => f.debug_struct("Cuda").field("value", value).finish(),
980 Self::CudaEvent { event, device, .. } => {
981 f.debug_struct("CudaEvent").field("event", &(*event as usize)).field("device", device).finish()
982 }
983 Self::OpenCl { event, value, .. } => {
984 f.debug_struct("OpenCl").field("event", &(*event as usize)).field("value", value).finish()
985 }
986 Self::Metal { value, .. } => f.debug_struct("Metal").field("value", value).finish(),
987 Self::OpenGL { semaphore, value, .. } => {
988 f.debug_struct("OpenGL").field("semaphore", semaphore).field("value", value).finish()
989 }
990 Self::OpenGLSync { glsync, backend, .. } => {
991 f.debug_struct("OpenGLSync").field("glsync", &(*glsync as usize)).field("backend", backend).finish()
992 }
993 #[cfg(feature = "wgpu")]
994 Self::Wgpu { .. } => f.write_str("SyncPoint::Wgpu"),
995 #[cfg(feature = "wgpu")]
996 Self::DeferredWgpu { slot, .. } => {
997 f.debug_struct("DeferredWgpu").field("committed", &slot.get().is_some()).finish()
998 }
999 Self::Cpu => f.write_str("SyncPoint::Cpu"),
1000 Self::Noop => f.write_str("SyncPoint::Noop"),
1001 }
1002 }
1003}
1004
1005/// Runtime-agnostic single-yield future. Returns `Pending` once then
1006/// resolves to `Ready(())` on the next poll, with `wake_by_ref` driving
1007/// the re-poll. Works under any executor (`pollster`, `tokio`, `smol`,
1008/// `futures::executor`).
1009///
1010/// Used by the default `SyncWaiter::wait_async` impl and by per-backend
1011/// overrides for their fast-path yield-spin loops.
1012pub async fn yield_once() {
1013 let mut yielded = false;
1014 core::future::poll_fn(|cx| {
1015 if yielded {
1016 core::task::Poll::Ready(())
1017 } else {
1018 yielded = true;
1019 cx.waker().wake_by_ref();
1020 core::task::Poll::Pending
1021 }
1022 })
1023 .await
1024}
1025
1026/// Helper: convert a `Duration` to nanoseconds, saturating at
1027/// `u64::MAX` for "wait forever". Matches `vkWaitSemaphores`'
1028/// `UINT64_MAX` infinite-wait sentinel.
1029pub fn duration_to_ns(t: Duration) -> u64 {
1030 u64::try_from(t.as_nanos()).unwrap_or(u64::MAX)
1031}
1032
1033/// Helper: convert a `Duration` to milliseconds, saturating at
1034/// `u32::MAX` for the Win32 `INFINITE` sentinel
1035/// (`WaitForSingleObject`, `WaitForMultipleObjects`).
1036pub fn duration_to_ms_u32(t: Duration) -> u32 {
1037 u32::try_from(t.as_millis()).unwrap_or(u32::MAX)
1038}