gpu-handle-types 0.2.0

Typed, owned native GPU resource handles (Vulkan, D3D11/12, Metal, OpenGL, CUDA, OpenCL, DMA-BUF, IOSurface, AHardwareBuffer, WebGPU, ...), cross-API sync points and video pixel formats, for passing GPU resources between libraries.
Documentation
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
826
827
828
829
830
831
832
833
834
835
836
837
838
839
840
841
842
843
844
845
846
847
848
849
850
851
852
853
854
855
856
857
858
859
860
861
862
863
864
865
866
867
868
869
870
871
872
873
874
875
876
877
878
879
880
881
882
883
884
885
886
887
888
889
890
891
892
893
894
895
896
897
898
899
900
901
902
903
904
905
906
907
908
909
910
911
912
913
914
915
916
917
918
919
920
921
922
923
924
925
926
927
928
929
930
931
932
933
934
935
936
937
938
939
940
941
942
943
944
945
946
947
948
949
950
951
952
953
954
955
956
957
958
959
960
961
962
963
964
965
966
967
968
969
970
971
972
973
974
975
976
977
978
979
980
981
982
983
984
985
986
987
988
989
990
991
992
993
994
995
996
997
998
999
1000
1001
1002
1003
1004
1005
1006
1007
1008
1009
1010
1011
1012
1013
1014
1015
1016
1017
1018
1019
1020
1021
1022
1023
1024
1025
1026
1027
1028
1029
1030
1031
1032
1033
1034
1035
1036
1037
1038
// SPDX-License-Identifier: MIT OR Apache-2.0

use core::future::Future;
use core::pin::Pin;
use std::any::Any;
use std::ffi::c_void;
use std::sync::Arc;
use std::time::Duration;

#[cfg(feature = "wgpu")]
use crate::SliceOutcome;
use crate::{BackendKind, Error, GlBackend};

/// Boxed dyn-future return shape used by `SyncWaiter::wait_async` so the
/// trait stays object-safe through `Arc<dyn SyncWaiter>`. Native
/// `async fn` on the trait would be RPITIT and dyn-incompatible by
/// default; callers dispatch through `Arc<dyn SyncWaiter>`, so boxing the
/// future is mandatory.
// `Send` off wasm; dropped on wasm, where the `SyncWaiter`
// the future borrows is thread-affine (`&dyn SyncWaiter` is not `Send`
// once the waiter is `!Sync`). An explicit `+ Send` on a `dyn Future`
// cannot be spelled with a non-auto marker trait, so the alias is
// cfg-split directly.
#[cfg(not(target_family = "wasm"))]
pub type BoxFuture<'a, T> = Pin<Box<dyn Future<Output = T> + Send + 'a>>;
#[cfg(target_family = "wasm")]
pub type BoxFuture<'a, T> = Pin<Box<dyn Future<Output = T> + 'a>>;

/// Default wait budget for `SyncPoint::wait()` and
/// `SyncPoint::wait_blocking()` — 10 seconds.
pub const DEFAULT_WAIT_TIMEOUT: Duration = Duration::from_secs(10);

/// A far-future but always-representable wait cap. `Instant + Duration` overflows
/// the platform's representable range only for a `timeout` so large it is
/// effectively infinite (near `Duration::MAX` — an `Instant` spans hundreds of
/// billions of years), and `Instant` has no `saturating_add`. Rather than let such
/// a `timeout` silently drop its deadline — turning a bounded wait into a hang —
/// [`wait_deadline`] clamps to `now +` this. It sits far beyond any real GPU wait
/// (budgets are seconds), so a legitimately huge finite wait still resolves against
/// it as intended.
const SATURATED_WAIT_CAP: Duration = Duration::from_secs(60 * 60 * 24 * 365); // ~1 year

/// The absolute deadline a `wait(timeout)` poll loop should honour, or `None` for
/// the [`SyncWaiter::wait`] contract's `Duration::MAX` "wait forever" sentinel.
///
/// - `timeout == Duration::MAX` → `None`: the caller asked to wait forever, so the
///   loop runs with no deadline (matching every backend's explicit `Duration::MAX`
///   handling, which maps to the native "infinite": `vkWaitSemaphores(UINT64_MAX)`,
///   `WaitForSingleObject(INFINITE)`, `cuEventSynchronize`, …).
/// - any other `timeout` → `Some(now + timeout)`, **saturated** to a bounded,
///   always-representable [`Instant`](std::time::Instant) on `Instant + Duration`
///   overflow — clamped to `now + 1 year`, far beyond any real GPU wait budget.
///
/// This keeps the two meanings of "no deadline" distinct: a returned `None`
/// encodes ONLY the explicit forever sentinel, never an accidental `checked_add`
/// overflow — which would otherwise silently convert a bounded wait into an
/// unbounded hang.
pub fn wait_deadline(timeout: Duration) -> Option<std::time::Instant> {
    if timeout == Duration::MAX {
        return None;
    }
    let now = std::time::Instant::now();
    Some(now.checked_add(timeout).unwrap_or_else(|| now.checked_add(SATURATED_WAIT_CAP).unwrap_or(now)))
}

/// Backend-specific wait dispatch + lifetime anchor for a [`SyncPoint`].
///
/// Every non-trivial `SyncPoint` variant carries an `Arc<dyn SyncWaiter>`
/// that:
///
/// 1. **Owns the underlying primitive's lifetime** — the timeline
///    semaphore, D3D12 fence, `MTLSharedEvent`, `cl_event`, `GLsync`,
///    etc. Cloning the `SyncPoint` clones the `Arc`; the primitive lives
///    as long as any clone outlives. This rules out the "consumer holds
///    a `*mut c_void` after the producer dropped" footgun.
///
/// 2. **Provides the wait implementation** — `SyncPoint::wait` /
///    `is_signaled` / `backend` dispatch through this trait, with no
///    global registries and no `OnceLock<fn>` runtime callbacks.
///
/// Implementations live in the producer crate (typically an interop
/// layer's per-backend module) and are constructed at the same site that
/// mints the `SyncPoint`.
// `MaybeSendSync` = `Send + Sync` off wasm; empty on wasm,
// where a waiter can hold a thread-affine `wgpu::Device` /
// `web_sys` handle. See [`crate::MaybeSendSync`].
pub trait SyncWaiter: crate::MaybeSendSync + 'static {
    /// Block until the GPU work this sync point represents has
    /// completed, or until `timeout` elapses.
    ///
    /// `Duration::MAX` means "wait forever" — implementations should
    /// translate this to the platform's native "infinite" sentinel
    /// (`UINT64_MAX` for `vkWaitSemaphores`, `INFINITE` for
    /// `WaitForSingleObject`, etc.).
    ///
    /// # Timeout granularity
    ///
    /// Native wait APIs vary in granularity, and implementations
    /// preserve the caller's intent at the expense of *requested*
    /// (not realised) latency on sub-API-tick timeouts:
    ///
    /// - **Metal** (`MTLSharedEvent::waitUntilSignaledValue:timeoutMS:`)
    ///   accepts millisecond-granular `u64`. Sub-millisecond positive
    ///   timeouts (e.g. `Duration::from_micros(100)`) round **up**
    ///   to 1 ms — truncation to 0 would silently behave as
    ///   "do not wait". A high-frequency progress poll using
    ///   `Duration::from_micros(100)` therefore observes ~10× the
    ///   requested latency on a not-yet-signaled event. For
    ///   non-blocking probes use [`SyncWaiter::is_signaled`] (which
    ///   passes `Duration::ZERO` and short-circuits on the
    ///   `signaledValue() >= value` accessor — driver-side, no
    ///   timer).
    /// - **Win32** (`WaitForSingleObject`) is also millisecond-
    ///   granular; the same round-up applies.
    /// - **Vulkan** (`vkWaitSemaphores`), **D3D12 fence**, **CUDA
    ///   external semaphore**, **OpenCL** are nanosecond-granular and
    ///   honour the caller's `Duration` exactly.
    ///
    /// Callers that need *both* a deterministic poll cadence under
    /// 1 ms *and* a real wait fallback should compose the two —
    /// e.g. `is_signaled` in a hot loop with their own `Instant`
    /// budget, then a single coarse `wait(Duration::from_millis(N))`
    /// when the budget is exhausted.
    fn wait(&self, timeout: Duration) -> Result<(), Error>;

    /// Async sibling of [`wait`](Self::wait). Returns a boxed future so
    /// the trait stays object-safe through `Arc<dyn SyncWaiter>` — native
    /// async-fn-in-trait is RPITIT and dyn-incompatible by default; callers
    /// dispatch through `Arc<dyn SyncWaiter>` everywhere, so the explicit
    /// `Pin<Box<...>>` is mandatory.
    ///
    /// # Default implementation
    ///
    /// Backends without a hand-tuned implementation get a two-phase
    /// hybrid: a short cooperative-yield spin (catches sub-ms waits
    /// without burning a thread) followed by an iteration-bounded
    /// `is_signaled` probe loop. Per-backend impls SHOULD override with
    /// their native blocking-with-timeout primitive on a dedicated
    /// waiter thread for waits the spin loop didn't catch.
    ///
    /// The default implementation is correct (resolves when signalled,
    /// returns `Error::Timeout` after `timeout`, never blocks the
    /// executor) but not optimal: it busy-polls inside the spin loop
    /// and falls back to a `Duration::ZERO` probe loop after that. For
    /// production paths, override with a backend-specific impl.
    fn wait_async<'a>(&'a self, timeout: Duration) -> BoxFuture<'a, Result<(), Error>> {
        Box::pin(async move {
            // Fast-path yield-spin — bounded cooperative-yield probe. Caps iteration
            // count rather than wall-clock time so executors that re-
            // poll immediately on `wake_by_ref` (pollster) don't burn
            // CPU here.
            const SPIN_ITERATIONS: usize = 64;
            // `None` iff `timeout == Duration::MAX` (the wait-forever sentinel); every
            // finite timeout gets a bounded, always-representable deadline — an
            // `Instant + Duration` overflow cannot silently drop it and hang.
            let deadline = wait_deadline(timeout);
            for _ in 0..SPIN_ITERATIONS {
                match self.is_signaled() {
                    Ok(true) => return Ok(()),
                    Ok(false) => {}
                    Err(e) => return Err(e),
                }
                if let Some(d) = deadline
                    && std::time::Instant::now() >= d
                {
                    return Err(Error::Timeout);
                }
                yield_once().await;
            }
            // Waiter-thread fallback — default impl probes `is_signaled` in a
            // yielding loop. Per-backend impls SHOULD override to hand
            // off to a dedicated waiter thread with the native
            // blocking-with-timeout primitive.
            loop {
                match self.is_signaled() {
                    Ok(true) => return Ok(()),
                    Ok(false) => {}
                    Err(e) => return Err(e),
                }
                if let Some(d) = deadline
                    && std::time::Instant::now() >= d
                {
                    return Err(Error::Timeout);
                }
                yield_once().await;
            }
        })
    }

    /// Non-blocking probe.
    ///
    /// - `Ok(true)` — the sync point has been reached.
    /// - `Ok(false)` — work is still in flight (the underlying poll
    ///   timed out at zero).
    /// - `Err(_)` — driver-level failure (device lost / TDR /
    ///   `wgpu::PollError::WrongSubmissionIndex` / equivalent on other
    ///   backends). Callers polling in a loop **must** break on `Err` —
    ///   folding errors into `Ok(false)` would loop forever on a TDR'd
    ///   device.
    ///
    /// Default implementation calls `wait(Duration::ZERO)` and maps
    /// `Err(Error::Timeout)` → `Ok(false)`; every other `Err` is
    /// propagated. Backends override when they have a more direct
    /// "is this primitive currently signaled" probe (e.g.
    /// `vkGetSemaphoreCounterValue`, `ID3D12Fence::GetCompletedValue`)
    /// that distinguishes "not signaled" from real errors without
    /// going through the wait path.
    fn is_signaled(&self) -> Result<bool, Error> {
        match self.wait(Duration::ZERO) {
            Ok(()) => Ok(true),
            Err(Error::Timeout) => Ok(false),
            Err(other) => Err(other),
        }
    }

    /// Backend identity for routing decisions on the consumer side.
    fn backend(&self) -> BackendKind;

    /// Downcast hook for cross-API bridges that need access to the
    /// concrete waiter type — e.g. a Vulkan→CUDA bridge that wants to
    /// pull the `VkDevice` out of the `VulkanWaiter` to issue its own
    /// `cuImportExternalSemaphore`. Most callers use the per-variant
    /// raw fields on `SyncPoint` directly and never need this.
    fn as_any(&self) -> &dyn Any;

    /// [`CudaEventWaiter`] view of this waiter, when the concrete type
    /// implements it. Default `None`.
    ///
    /// This is the trait-object route to the CUDA-only
    /// [`CudaEventWaiter::wait_on_foreign_stream`] extension: `Any` can
    /// only downcast to *concrete* types, which consumers in other
    /// crates cannot name — so producers whose waiter implements
    /// [`CudaEventWaiter`] override this with `Some(self)` and
    /// consumers (e.g. a CUDA import path gating its private copy stream
    /// on a producer's `SyncPoint::CudaEvent`) reach the extension method
    /// without knowing the concrete type.
    fn as_cuda_event_waiter(&self) -> Option<&dyn CudaEventWaiter> {
        None
    }

    /// Rebuild this waiter bound to `value` instead of the value it was
    /// constructed with, reusing the same underlying primitive (and the
    /// same lifetime anchors).
    ///
    /// # Why this exists
    ///
    /// [`wait`](Self::wait) / [`is_signaled`](Self::is_signaled) take no
    /// value argument — the waiter *embeds* the value it resolves at,
    /// captured when it was built. So a [`SyncPoint`] whose `value`
    /// field was substituted while its `waiter` was cloned verbatim has
    /// two surfaces that disagree: the GPU side (a consumer reading
    /// `SyncPoint::*.value` to stage a queue wait) waits the new value,
    /// while the CPU side silently resolves at the old one — reporting
    /// "already signalled" for a signal the producer has not emitted.
    ///
    /// Any code that mints a `SyncPoint` at a value other than the one a
    /// template was built with MUST route the waiter through this method
    /// rather than cloning it, so the two surfaces cannot diverge.
    ///
    /// # Contract
    ///
    /// - `Some(w)` — `w` waits `value` on the same primitive this waiter
    ///   waits on, and holds the same keep-alive chain. Implementations
    ///   MUST NOT carry over state that is only valid for the original
    ///   value (e.g. a captured submission index that retires at it).
    /// - `None` (the default) — the underlying primitive carries no
    ///   value, so there is nothing to rebind: binary semaphores,
    ///   fences, `GLsync`, `cl_semaphore_khr`. A `None` here is not a
    ///   failure; it means the CPU surface has no value to disagree
    ///   about.
    fn rebind_to_value(&self, value: u64) -> Option<Arc<dyn SyncWaiter>> {
        let _ = value;
        None
    }
}

/// CUDA-event-specific extension trait for the
/// [`SyncPoint::CudaEvent`] variant. Supertrait of [`SyncWaiter`] so a
/// `CudaEventWaiter` always satisfies the generic [`SyncPoint::wait`]
/// dispatch path; adds the CUDA-only `wait_on_foreign_stream` extension
/// method that cross-API bridges downcast to via
/// [`SyncWaiter::as_any`].
///
/// # Drop discipline
///
/// Implementations MUST push the event's owning `CUcontext` before
/// calling `cuEventDestroy_v2`, and MUST pop iff the push succeeded
/// (pop-without-push would silently consume whatever the calling
/// thread had on top of its context stack). `cuEventDestroy_v2`
/// requires the owning context to still exist; the push is defensive
/// against driver-internal cleanup paths that probe `cuCtxGetCurrent`
/// and emit confusing diagnostics when no context is current.
///
/// # Cross-context wait
///
/// `cuEventRecord` is strict-same-context — a `CUevent` cannot be
/// recorded against a stream from a different `CUcontext`. But
/// `cuStreamWaitEvent` is cross-context (NVIDIA Driver API contract).
/// `wait_on_foreign_stream` is the cross-context wait entry point:
/// the consumer's CUDA stream may live in a different `CUcontext`
/// than the event, and the waiter issues `cuStreamWaitEvent` so the
/// foreign stream gates on this event's completion without a CPU
/// bounce.
///
/// A waiter backed by the CUDA driver API implements this trait
/// concretely; this trait only defines the contract.
pub trait CudaEventWaiter: SyncWaiter {
    /// Issue `cuStreamWaitEvent(foreign_stream, self.event, 0)` so
    /// `foreign_stream` gates on this event's completion. The stream
    /// may belong to a different `CUcontext` than the event's owning
    /// context — `cuStreamWaitEvent` is cross-context per the NVIDIA
    /// Driver API.
    ///
    /// `foreign_stream` is a raw `CUstream` pointer (the null pointer
    /// resolves to the calling thread's default stream).
    ///
    /// Returns [`Error::NotSupported`] if the CUDA driver loader is
    /// unavailable or the wait dispatch returns a driver error; the
    /// concrete error wording is implementation-defined.
    fn wait_on_foreign_stream(&self, foreign_stream: *mut c_void) -> Result<(), Error>;
}

/// Discriminator for [`SyncPoint::Vulkan`]'s `VkSemaphore` flavour.
///
/// `VkSemaphore` ships in two shapes — binary (one-shot, signal-and-
/// reset) and timeline (monotonic 64-bit payload, multi-waiter,
/// concurrent-signal-safe). A consumer staging a wait must route
/// through `VkTimelineSemaphoreSubmitInfo` for timeline semaphores and
/// must NOT for binary semaphores — passing the wrong shape risks
/// validation errors, driver-quirk-dependent wrong waits, and (on
/// already-consumed binary signals) hard deadlock at submit time.
///
/// The Vulkan API has no public probe that safely distinguishes the
/// two (`vkGetSemaphoreCounterValue` is undefined on binaries per
/// `VUID-vkGetSemaphoreCounterValue-semaphore-03255`), so the producer
/// must declare intent at mint time.
#[derive(Copy, Clone, Debug, PartialEq, Eq)]
#[non_exhaustive]
pub enum VulkanSemaphoreKind {
    /// `vk::SemaphoreType::TIMELINE`. `value` is a payload coordinate;
    /// any number of waiters may target the same value, and signals
    /// with strictly-increasing payloads can be queued concurrently.
    Timeline,
    /// `vk::SemaphoreType::BINARY`. Single-shot — exactly one signal
    /// must pair with exactly one wait. The `value` field is ignored
    /// (set to `0` by convention). Producers carrying binaries here
    /// MUST guarantee the wait is the only consumer of the signal.
    Binary,
}

/// Shared, mutate-once slot the producer fills in after the caller
/// submits the encoded work. Used by [`SyncPoint::DeferredWgpu`].
///
/// The slot is initialised to `None` at mint time and transitions to
/// `Some(idx)` exactly once via [`Self::set`]; subsequent calls are
/// rejected with [`Error::InvalidArgument`] (single-shot semantics).
#[cfg(feature = "wgpu")]
#[derive(Debug, Default)]
pub struct DeferredWgpuSlot {
    cell: std::sync::OnceLock<wgpu::SubmissionIndex>,
}

#[cfg(feature = "wgpu")]
impl DeferredWgpuSlot {
    pub fn new() -> Self {
        Self::default()
    }

    /// Commit the post-submit `SubmissionIndex`. Single-shot — a second
    /// call returns [`Error::InvalidArgument`] without overwriting the
    /// committed value.
    pub fn set(&self, idx: wgpu::SubmissionIndex) -> Result<(), Error> {
        self.cell.set(idx).map_err(|_| Error::InvalidArgument("DeferredWgpuSlot already committed".into()))
    }

    /// Return the committed `SubmissionIndex`, or `None` if the producer
    /// has not yet submitted.
    pub fn get(&self) -> Option<&wgpu::SubmissionIndex> {
        self.cell.get()
    }
}

// Defensive — the `OnceLock<wgpu::SubmissionIndex>` design relies on
// `SubmissionIndex` being `Clone + Send + Sync`. If a wgpu upgrade
// silently dropped any of these bounds, the OnceLock-storing
// `DeferredWgpuSlot` would surface a confusing trait-bound error far
// from the cause; this assertion fails compilation here, with a clear
// message, instead.
#[cfg(feature = "wgpu")]
static_assertions::assert_impl_all!(wgpu::SubmissionIndex: Clone, Send, Sync);

/// GPU sync primitive. Produced by every submit-side path; consumed by
/// anything that needs to serialise on a prior GPU submission.
///
/// Cross-backend bridges are deliberately absent from this leaf crate —
/// those belong in an interop layer, where the target-device context is
/// available.
///
/// # Lifetime
///
/// The per-variant `waiter: Arc<dyn SyncWaiter>` field anchors the
/// lifetime of any raw primitive the variant exposes (`*mut c_void`
/// fence pointers, `u64` semaphore handles, `GLsync` opaques, etc.).
/// Cloning a `SyncPoint` is `Arc::clone` on the waiter — cheap, no
/// driver round-trip. The raw fields are guaranteed to remain valid
/// for as long as any clone is alive.
#[derive(Clone)]
#[non_exhaustive]
pub enum SyncPoint {
    /// Vulkan semaphore. `device` is the `VkDevice` the semaphore was
    /// created on, exposed for bridges that need to re-import it. The
    /// `waiter` owns whatever anchors the `VkSemaphore`'s lifetime.
    ///
    /// `kind` distinguishes timeline vs binary semaphores —
    /// see [`VulkanSemaphoreKind`]. `value` is the timeline payload
    /// coordinate when `kind == Timeline`; for `kind == Binary` the
    /// field is meaningless (set to `0` by convention) and consumers
    /// route through a binary wait (with wgpu-hal,
    /// `vulkan::Queue::add_wait_semaphore(_, None, _)`).
    Vulkan { semaphore: u64, kind: VulkanSemaphoreKind, value: u64, device: *mut c_void, waiter: Arc<dyn SyncWaiter> },
    /// D3D12 fence sync. `fence` is an `ID3D12Fence*` (COM pointer);
    /// `value` is the monotonic signal value the consumer waits on
    /// (`fence->GetCompletedValue() >= value`). The `waiter` holds the
    /// COM reference that keeps the fence alive.
    ///
    /// `value` is monotonically signaled by the producer. Bridges that
    /// reuse a single shared fence across calls serialise the
    /// (mint + `Signal`) pair under a value-lock so concurrent bridges
    /// always sequence in monotonic order.
    /// A consumer waiting on `value=N` is guaranteed to pass once a
    /// later `value=M >= N` has been signaled, even if the producer's
    /// counter has since advanced past `M`.
    D3D12 { fence: *mut c_void, value: u64, waiter: Arc<dyn SyncWaiter> },
    /// D3D11 keyed-mutex sync. `keyed_mutex` is an `IDXGIKeyedMutex*`;
    /// `key` is the integer key the producer released to. The
    /// `waiter` holds the COM reference that keeps the mutex alive.
    D3D11 { keyed_mutex: *mut c_void, key: u64, waiter: Arc<dyn SyncWaiter> },
    /// CUDA external semaphore. `event` is a `CUexternalSemaphore`;
    /// `value` is `Some(v)` for timeline waits — including the products
    /// of Vulkan→CUDA / D3D12→CUDA bridges, which carry the producer's
    /// timeline value so the consumer can enqueue
    /// `cuWaitExternalSemaphoresAsync` on whichever CUDA stream their
    /// work runs on. `None` is reserved for setup-metadata SyncPoints
    /// where the consumer mints the actual signal value at emission time
    /// and doesn't need a resolvable wait at construction.
    Cuda { event: *mut c_void, value: Option<u64>, waiter: Arc<dyn SyncWaiter> },
    /// Intra-CUDA event. `event` is a `CUevent` recorded on the
    /// producing stream via `cuEventRecord`. Consumers wait via
    /// `cuStreamWaitEvent` (chained CUDA streams — cross-context
    /// supported per NVIDIA Driver API) or `cuEventSynchronize`
    /// (blocking CPU wait).
    ///
    /// Distinct from [`Self::Cuda`] — that variant carries a
    /// `CUexternalSemaphore` waited via `cuWaitExternalSemaphoresAsync`.
    /// `CUevent` and `CUexternalSemaphore` are different opaque types
    /// in the CUDA Driver API; passing a `CUevent` into the external-
    /// semaphore wait path returns `CUDA_ERROR_INVALID_HANDLE`.
    ///
    /// Cross-API consumers cannot wait on a `CUevent` directly; they
    /// need a `CUexternalSemaphore` (the [`Self::Cuda`] shape); an
    /// interop layer performs that conversion when the chain spans a
    /// backend boundary.
    ///
    /// `context` is the event's **owning** `CUcontext` — bound at
    /// `cuEventCreate` time per the NVIDIA Driver API ("creates an event
    /// for the current context"). The owning context must still exist
    /// at destroy time or `cuEventDestroy_v2` returns
    /// `CUDA_ERROR_INVALID_CONTEXT`. There is no `cuEventGetCtx` API —
    /// the metadata must travel with the event.
    ///
    /// `device` is the CUDA ordinal the owning context was created on
    /// (derivable from `context` via `cuCtxGetDevice` but cheap to
    /// carry and matches the device-keyed bridge shape used by
    /// [`Self::Cuda`]).
    ///
    /// `waiter` implements [`CudaEventWaiter`] — a supertrait of
    /// [`SyncWaiter`] that adds the CUDA-specific
    /// `wait_on_foreign_stream` extension method and the push-owning-
    /// context-before-destroy Drop discipline. The leaf crate stores
    /// the upcasted `Arc<dyn SyncWaiter>` so the generic [`Self::waiter`]
    /// dispatch path is uniform with every other variant; cross-API
    /// bridges that need the CUDA extension methods downcast through
    /// [`SyncWaiter::as_any`].
    CudaEvent { event: *mut c_void, context: *mut c_void, device: i32, waiter: Arc<dyn SyncWaiter> },
    /// OpenCL sync. `event` is either a `cl_event` (binary,
    /// signaled-on-completion; `value` is `None`) or a
    /// `cl_semaphore_khr` aliasing a timeline (`value` is the
    /// caller-signaled timeline value). The `waiter` keeps the
    /// producing context alive.
    OpenCl { event: *mut c_void, value: Option<u64>, waiter: Arc<dyn SyncWaiter> },
    /// Metal shared event. `event` is an `MTLSharedEvent*`; `value` is
    /// the signal value the consumer waits on. The `waiter` holds the
    /// reference that keeps the event alive.
    Metal { event: *mut c_void, value: u64, waiter: Arc<dyn SyncWaiter> },
    /// OpenGL semaphore (`GLuint` produced by `glGenSemaphoresEXT`,
    /// imported from a cross-API OS handle via
    /// `glImportSemaphoreFdEXT` / `glImportSemaphoreWin32HandleEXT`).
    ///
    /// `value` is `Some(_)` for timeline-equivalent semaphores
    /// (D3D12_FENCE_EXT handle types); `None` for plain binary
    /// (OPAQUE_FD / OPAQUE_WIN32) — the GL driver latches the D3D12
    /// fence value at wait/signal time via
    /// `glSemaphoreParameterui64vEXT(sem, GL_D3D12_FENCE_VALUE_EXT, &v)`.
    ///
    /// `context` is the caller's `EGLContext` / `HGLRC` / `CGLContextObj`
    /// this semaphore was created against. Waits must be dispatched on
    /// a thread where the same context is current.
    OpenGL { semaphore: u32, value: Option<u64>, context: *mut c_void, waiter: Arc<dyn SyncWaiter> },
    /// GL fence sync. `glsync` is a `GLsync` opaque pointer produced
    /// by `glFenceSync(GL_SYNC_GPU_COMMANDS_COMPLETE, 0)`.
    ///
    /// Distinct from [`Self::OpenGL`] — that one carries an *imported*
    /// GL semaphore (`GLuint`); this one is the raw GL-native fence.
    ///
    /// `backend` records the [`GlBackend`] flavour of the GL context that
    /// produced this `GLsync`. Cross-API consumers (an OpenGL→X bridge)
    /// route through it to pick the matching loader (`wglGetProcAddress`
    /// vs `eglGetProcAddress`) when resolving `glClientWaitSync` for the
    /// CPU-fallback path.
    /// Without it the consumer would hard-code `Desktop`, which silently
    /// resolves the wrong proc-address on EGL / ANGLE / Android hosts and
    /// stalls forever on a never-signalled handle.
    OpenGLSync { glsync: *mut c_void, backend: GlBackend, waiter: Arc<dyn SyncWaiter> },
    /// `wgpu::Queue::submit` return value. Wait dispatches through the
    /// `waiter`, which holds a `wgpu::Device` clone and calls
    /// `device.poll(PollType::Wait { submission_index: Some(..), .. })`.
    #[cfg(feature = "wgpu")]
    Wgpu { submission_index: wgpu::SubmissionIndex, waiter: Arc<dyn SyncWaiter> },
    /// `wgpu::Queue::submit` index *not yet known* at SyncPoint mint
    /// time. Used where the caller owns the encoder and submits after the
    /// producer has returned; `slot.set(idx)` is called once after
    /// `queue.submit(...)`, and until then the waiter falls back to
    /// `device.poll(PollType::Wait { submission_index: None, .. })`
    /// (full device drain). After commit, waits resolve precisely on
    /// the recorded `SubmissionIndex`.
    #[cfg(feature = "wgpu")]
    DeferredWgpu { slot: Arc<DeferredWgpuSlot>, device: wgpu::Device, waiter: Arc<dyn SyncWaiter> },
    /// CPU-only work — no GPU sync needed. `wait` returns `Ok(())`
    /// immediately.
    Cpu,
    /// No-op sync — used for pipelines where the consumer side handles
    /// ordering through a separate channel (e.g. SurfaceControl
    /// transactions). Behaves identically to `Cpu` for waits.
    Noop,
}

// SAFETY: the `*mut c_void` raw fields are immutable after
// construction and dereferenced only by the per-variant `waiter`,
// whose `SyncWaiter: Send + Sync` bound asserts platform-correct sharing
// for whatever the pointer references. Per-variant doc-comments specify
// the additional caller-asserted preconditions (e.g. D3D11 requires the
// device's multithread-protect flag).
unsafe impl Send for SyncPoint {}
unsafe impl Sync for SyncPoint {}

impl SyncPoint {
    /// Async wait. Default 10-second timeout.
    ///
    /// Async is the default. Use [`wait_blocking`]
    /// when the caller is on a synchronous code path (sync-loop
    /// renderers, test harnesses, JNI dispatcher Drop paths).
    ///
    /// [`wait_blocking`]: Self::wait_blocking
    pub async fn wait(&self) -> Result<(), Error> {
        self.wait_with_timeout_async(DEFAULT_WAIT_TIMEOUT).await
    }

    /// Async wait with explicit timeout.
    pub async fn wait_with_timeout_async(&self, timeout: Duration) -> Result<(), Error> {
        // Binary Vulkan CPU-wait rejection.
        if let Some(err) = self.binary_vk_cpu_wait_error() {
            return Err(err);
        }
        match self.waiter() {
            Some(w) => w.wait_async(timeout).await,
            None => Ok(()),
        }
    }

    /// Synchronous wait with the default 10-second timeout.
    pub fn wait_blocking(&self) -> Result<(), Error> {
        self.wait_with_timeout(DEFAULT_WAIT_TIMEOUT)
    }

    /// Synchronous wait with an explicit timeout.
    ///
    /// Timeout granularity is backend-dependent — see
    /// [`SyncWaiter::wait`] for the per-backend rounding contract.
    /// In particular, Metal and Win32 are millisecond-granular and
    /// round sub-ms positive timeouts up to 1 ms; Vulkan / D3D12 /
    /// CUDA / OpenCL are nanosecond-granular. For sub-ms polls use
    /// [`Self::is_signaled`] in a loop with your own `Instant` budget.
    pub fn wait_with_timeout(&self, timeout: Duration) -> Result<(), Error> {
        // Binary Vulkan CPU-wait rejection — `vkWaitSemaphores` is
        // undefined on binary semaphores, and the Khronos spec gives
        // binary semaphores no CPU-side wait entry point. Producers
        // that need CPU-waitable binary semaphores MUST attach a
        // fence-equivalent fallback (a `wgpu::SubmissionIndex` from the
        // same submit). When the fallback is absent, the backend's waiter
        // fails both `wait_blocking` and the async `wait()` with
        // `Error::NotSupported` (see `binary_vk_cpu_wait_error`).
        if let Some(err) = self.binary_vk_cpu_wait_error() {
            return Err(err);
        }
        match self.waiter() {
            Some(w) => w.wait(timeout),
            None => Ok(()),
        }
    }

    /// Leaf-crate hook for rejecting a CPU wait on a binary
    /// `SyncPoint::Vulkan` whose producer did not attach a
    /// fence-equivalent fallback. Always `None`: whether the fallback is
    /// attached is visible only to the backend's waiter, which is where
    /// that half of the binary-Vulkan CPU-wait contract is enforced.
    fn binary_vk_cpu_wait_error(&self) -> Option<Error> {
        // The fence-equivalent fallback lives on the per-backend Vulkan
        // waiter impl; this crate only sees `Arc<dyn SyncWaiter>`. The
        // contract is that `wait` / `wait_async` on a binary semaphore
        // without a fallback return `Error::NotSupported`. Backends MUST
        // honour it; the "fallback present" check is not observable here,
        // so this returns `None` rather than gate on it.
        let _ = matches!(self, Self::Vulkan { kind: VulkanSemaphoreKind::Binary, .. });
        None
    }

    /// Sequence two sync points. Returns a `SyncPoint` whose wait
    /// completes only after both `self` and `then` have completed.
    ///
    /// If either side is `Cpu` / `Noop`, the other is returned unchanged.
    /// Otherwise the result is an **opaque** carrier: a small adapter
    /// waiter drives both inner waits in sequence (one heap allocation),
    /// and it is stored in the [`SyncPoint::Cuda`] variant with a null
    /// `event` and no `value`. Only use the result through
    /// [`wait`](Self::wait), [`wait_blocking`](Self::wait_blocking),
    /// [`is_signaled`](Self::is_signaled), [`backend`](Self::backend)
    /// (the first sync point's backend) and [`waiter`](Self::waiter);
    /// its raw fields carry nothing.
    pub fn chain(self, then: SyncPoint) -> SyncPoint {
        // Trivial cases — skip the adapter entirely.
        if matches!(self, Self::Cpu | Self::Noop) {
            return then;
        }
        if matches!(then, Self::Cpu | Self::Noop) {
            return self;
        }
        let backend = self.backend();
        let waiter: Arc<dyn SyncWaiter> = Arc::new(ChainWaiter { first: self, then, backend });
        Self::Cpu.with_chain_waiter(waiter)
    }

    /// Wraps a `ChainWaiter` in the carrier variant `chain` documents.
    /// Used only by `chain`.
    fn with_chain_waiter(self, waiter: Arc<dyn SyncWaiter>) -> SyncPoint {
        // Any variant with a `waiter` field would do: every caller-facing
        // method dispatches through the waiter, so the event / value
        // fields are never read.
        let _ = self;
        Self::Cuda { event: core::ptr::null_mut(), value: None, waiter }
    }
}

// ─────────────────────────────────────────────────────────────────────────────
// Raw-handle reconstruction (cross-C / cross-instance transport carriers)
// ─────────────────────────────────────────────────────────────────────────────

/// A [`SyncWaiter`] for a [`SyncPoint`] **reconstructed from a raw foreign fence
/// handle** that has not yet been imported onto a waiting device.
///
/// A D3D12 fence / Vulkan semaphore / Metal shared event that crossed an API or
/// process boundary as a raw handle (e.g. a `#[repr(C)]` descriptor over a C ABI)
/// carries no device the leaf crate can drive a wait on — the real CPU/GPU wait
/// only becomes possible once a consumer **imports** the handle onto its own device
/// (the importer stages the device-side `vkWaitSemaphores` /
/// `ID3D12CommandQueue::Wait` / `encodeWaitForEvent` off the carried raw fields).
///
/// So a `SyncPoint` reconstructed via [`SyncPoint::from_raw_d3d12_fence`] /
/// [`SyncPoint::from_raw_vulkan_semaphore`] / [`SyncPoint::from_raw_metal_event`] is
/// a **transport carrier**: its raw fields are consumed by the importer, and its
/// `waiter` is this carrier. Calling [`SyncPoint::wait_blocking`] / `wait` /
/// `is_signaled` on the carrier (before import) returns [`Error::NotSupported`] —
/// fail-closed, never a silent "already signaled" — because the leaf crate cannot
/// wait a fence it has no device for. Import the handle first; the import path
/// produces the real, waitable `SyncPoint`.
#[derive(Debug)]
pub struct RawFenceCarrierWaiter {
    backend: BackendKind,
}

impl RawFenceCarrierWaiter {
    fn new(backend: BackendKind) -> Self {
        Self { backend }
    }

    fn not_imported() -> Error {
        Error::NotSupported(std::borrow::Cow::Borrowed(
            "SyncPoint reconstructed from a raw foreign fence handle is a transport \
                 carrier — import it onto a device before waiting",
        ))
    }
}

impl SyncWaiter for RawFenceCarrierWaiter {
    fn wait(&self, _timeout: Duration) -> Result<(), Error> {
        Err(Self::not_imported())
    }

    fn is_signaled(&self) -> Result<bool, Error> {
        Err(Self::not_imported())
    }

    fn backend(&self) -> BackendKind {
        self.backend
    }

    fn as_any(&self) -> &dyn Any {
        self
    }
}

impl SyncPoint {
    /// Reconstruct a [`SyncPoint::D3D12`] transport carrier from a raw
    /// `ID3D12Fence*` pointer + signal value that crossed an API / process boundary
    /// (e.g. a `#[repr(C)]` fence descriptor over a C ABI).
    ///
    /// `fence` is an `ID3D12Fence*` COM pointer the caller guarantees is live for
    /// the carrier's lifetime; `value` is the monotonic value the consumer waits for
    /// (`fence->GetCompletedValue() >= value`). The returned `SyncPoint` is a
    /// **carrier** ([`RawFenceCarrierWaiter`]) — its raw `(fence, value)` fields feed
    /// a device-side import (e.g. an interop layer's acquire-sync path); a CPU
    /// `wait()` on it before import returns [`Error::NotSupported`].
    pub fn from_raw_d3d12_fence(fence: *mut c_void, value: u64) -> SyncPoint {
        let waiter: Arc<dyn SyncWaiter> = Arc::new(RawFenceCarrierWaiter::new(BackendKind::D3D12));
        SyncPoint::D3D12 { fence, value, waiter }
    }

    /// Reconstruct a [`SyncPoint::Vulkan`] transport carrier from a raw
    /// `VkSemaphore` + `VkDevice` that crossed an API / process boundary.
    ///
    /// `semaphore` is the raw `VkSemaphore` (a 64-bit non-dispatchable handle);
    /// `kind` distinguishes timeline vs binary (see [`VulkanSemaphoreKind`]);
    /// `value` is the timeline coordinate (ignored for `Binary`); `device` is the
    /// `VkDevice*` the semaphore lives on (needed by an importer that re-exports it
    /// for a cross-device wait). The returned `SyncPoint` is a **carrier** — its raw
    /// fields feed a device-side import; a CPU `wait()` before import returns
    /// [`Error::NotSupported`].
    pub fn from_raw_vulkan_semaphore(
        semaphore: u64,
        kind: VulkanSemaphoreKind,
        value: u64,
        device: *mut c_void,
    ) -> SyncPoint {
        let waiter: Arc<dyn SyncWaiter> = Arc::new(RawFenceCarrierWaiter::new(BackendKind::Vulkan));
        SyncPoint::Vulkan { semaphore, kind, value, device, waiter }
    }

    /// Reconstruct a [`SyncPoint::Metal`] transport carrier from a raw
    /// `MTLSharedEvent*` + signal value that crossed an API / process boundary.
    ///
    /// `event` is an `MTLSharedEvent*` the caller guarantees is live for the
    /// carrier's lifetime; `value` is the value the consumer waits for. The returned
    /// `SyncPoint` is a **carrier** — its raw `(event, value)` fields feed a
    /// device-side import; a CPU `wait()` before import returns
    /// [`Error::NotSupported`].
    pub fn from_raw_metal_event(event: *mut c_void, value: u64) -> SyncPoint {
        let waiter: Arc<dyn SyncWaiter> = Arc::new(RawFenceCarrierWaiter::new(BackendKind::Metal));
        SyncPoint::Metal { event, value, waiter }
    }
}

/// Per-process waiter thread for [`SyncPoint::DeferredWgpu`].
///
/// The waiter thread is stateless — every per-instance datum (device
/// handle, deferred slot) travels inside the `SliceFn` closure — so one
/// thread serves every deferred sync point in the process.
#[cfg(feature = "wgpu")]
fn deferred_wgpu_waiter_thread() -> &'static crate::WaiterThread {
    static THREAD: std::sync::OnceLock<crate::WaiterThread> = std::sync::OnceLock::new();
    THREAD.get_or_init(|| crate::WaiterThread::new("wgpu-deferred"))
}

/// Backend-internal waiter for [`SyncPoint::DeferredWgpu`]. Until the
/// producer commits a `SubmissionIndex` via [`DeferredWgpuSlot::set`],
/// `wait` falls back to `device.poll(PollType::Wait { submission_index:
/// None, .. })` (full device drain). After commit, it passes
/// `submission_index: Some(idx)` and resolves on that precise index.
///
/// `wait_async` routes through [`run_hybrid_wait`] — the fast-path
/// yield-spin spins on `is_signaled` for sub-millisecond signals, the
/// waiter-thread fallback hands off to the
/// per-process waiter thread which issues blocking-with-timeout slices
/// of [`WAITER_SLICE`] on a dedicated thread.
#[cfg(feature = "wgpu")]
pub(crate) struct DeferredWgpuWaiter {
    slot: Arc<DeferredWgpuSlot>,
    device: wgpu::Device,
}

#[cfg(feature = "wgpu")]
impl DeferredWgpuWaiter {
    pub(crate) fn new(slot: Arc<DeferredWgpuSlot>, device: wgpu::Device) -> Self {
        Self { slot, device }
    }

    /// Single `device.poll(Wait { .. })` issuance keyed off the
    /// current commit state. Shared between `wait`, `is_signaled`, and
    /// the per-slice closure produced by `wait_async`.
    fn poll_slice(&self, timeout: Option<Duration>) -> SliceOutcome {
        let submission_index = self.slot.get().cloned();
        match self.device.poll(wgpu::PollType::Wait { submission_index, timeout }) {
            Ok(_) => SliceOutcome::Signaled,
            Err(wgpu::PollError::Timeout) => SliceOutcome::TimedOut,
            Err(e) => {
                log::warn!("SyncPoint::DeferredWgpu: device.poll returned {e:?}");
                SliceOutcome::Failed(Error::NotSupported("SyncPoint::DeferredWgpu: device.poll failed".into()))
            }
        }
    }
}

#[cfg(feature = "wgpu")]
impl SyncWaiter for DeferredWgpuWaiter {
    fn wait(&self, timeout: Duration) -> Result<(), Error> {
        let timeout_arg = if timeout == Duration::MAX { None } else { Some(timeout) };
        match self.poll_slice(timeout_arg) {
            SliceOutcome::Signaled => Ok(()),
            SliceOutcome::TimedOut => Err(Error::Timeout),
            SliceOutcome::Failed(e) => Err(e),
        }
    }

    fn wait_async<'a>(&'a self, timeout: Duration) -> BoxFuture<'a, Result<(), Error>> {
        let slot = self.slot.clone();
        let device = self.device.clone();
        let make_slice = move || -> crate::SliceFn {
            Box::new(move |slice: Duration| -> SliceOutcome {
                let submission_index = slot.get().cloned();
                match device.poll(wgpu::PollType::Wait { submission_index, timeout: Some(slice) }) {
                    Ok(_) => SliceOutcome::Signaled,
                    Err(wgpu::PollError::Timeout) => SliceOutcome::TimedOut,
                    Err(e) => {
                        log::warn!("SyncPoint::DeferredWgpu::wait_async: device.poll returned {e:?}");
                        SliceOutcome::Failed(Error::NotSupported(
                            "SyncPoint::DeferredWgpu::wait_async: device.poll failed".into(),
                        ))
                    }
                }
            })
        };
        Box::pin(crate::run_hybrid_wait(move || self.is_signaled(), deferred_wgpu_waiter_thread(), timeout, make_slice))
    }

    fn is_signaled(&self) -> Result<bool, Error> {
        match self.poll_slice(Some(Duration::ZERO)) {
            SliceOutcome::Signaled => Ok(true),
            SliceOutcome::TimedOut => Ok(false),
            SliceOutcome::Failed(e) => Err(e),
        }
    }

    fn backend(&self) -> BackendKind {
        BackendKind::Wgpu
    }

    fn as_any(&self) -> &dyn Any {
        self
    }
}

/// Construct a [`SyncPoint::DeferredWgpu`] complete with its waiter.
///
/// For producers that hand out a sync point *before* they submit: the
/// returned SyncPoint's clones all share the same `Arc<DeferredWgpuSlot>`,
/// so the producer-side `slot.set(idx)` at submit time is visible to every
/// clone.
#[cfg(feature = "wgpu")]
pub fn make_deferred_wgpu_sync_point(slot: Arc<DeferredWgpuSlot>, device: wgpu::Device) -> SyncPoint {
    let waiter: Arc<dyn SyncWaiter> = Arc::new(DeferredWgpuWaiter::new(slot.clone(), device.clone()));
    SyncPoint::DeferredWgpu { slot, device, waiter }
}

struct ChainWaiter {
    first: SyncPoint,
    then: SyncPoint,
    backend: BackendKind,
}

impl SyncWaiter for ChainWaiter {
    fn wait(&self, timeout: Duration) -> Result<(), Error> {
        let start = std::time::Instant::now();
        self.first.wait_with_timeout(timeout)?;
        let elapsed = start.elapsed();
        let remaining = timeout.saturating_sub(elapsed);
        self.then.wait_with_timeout(remaining)
    }

    fn wait_async<'a>(&'a self, timeout: Duration) -> BoxFuture<'a, Result<(), Error>> {
        Box::pin(async move {
            let start = std::time::Instant::now();
            self.first.wait_with_timeout_async(timeout).await?;
            let elapsed = start.elapsed();
            let remaining = timeout.saturating_sub(elapsed);
            self.then.wait_with_timeout_async(remaining).await
        })
    }

    fn is_signaled(&self) -> Result<bool, Error> {
        Ok(self.first.is_signaled()? && self.then.is_signaled()?)
    }

    fn backend(&self) -> BackendKind {
        self.backend
    }

    fn as_any(&self) -> &dyn Any {
        self
    }
}

impl SyncPoint {
    /// Non-blocking probe.
    ///
    /// - `Ok(true)`  — sync point reached (or trivial `Cpu`/`Noop`).
    /// - `Ok(false)` — work still in flight.
    /// - `Err(_)`    — driver-level failure (device lost / TDR /
    ///   wrong-submission-index). See [`SyncWaiter::is_signaled`]
    ///   for the rationale on surfacing errors instead of folding
    ///   them into `false`.
    pub fn is_signaled(&self) -> Result<bool, Error> {
        match self.waiter() {
            Some(w) => w.is_signaled(),
            None => Ok(true),
        }
    }

    /// Backend identity. For the trivial variants (`Cpu`, `Noop`) this
    /// returns [`BackendKind::Cpu`]; otherwise it dispatches to the
    /// per-variant waiter.
    pub fn backend(&self) -> BackendKind {
        match self.waiter() {
            Some(w) => w.backend(),
            None => BackendKind::Cpu,
        }
    }

    /// Borrow the per-variant waiter, if any. `Cpu` and `Noop` return
    /// `None`.
    pub fn waiter(&self) -> Option<&Arc<dyn SyncWaiter>> {
        match self {
            Self::Vulkan { waiter, .. }
            | Self::D3D12 { waiter, .. }
            | Self::D3D11 { waiter, .. }
            | Self::Cuda { waiter, .. }
            | Self::CudaEvent { waiter, .. }
            | Self::OpenCl { waiter, .. }
            | Self::Metal { waiter, .. }
            | Self::OpenGL { waiter, .. }
            | Self::OpenGLSync { waiter, .. } => Some(waiter),
            #[cfg(feature = "wgpu")]
            Self::Wgpu { waiter, .. } => Some(waiter),
            #[cfg(feature = "wgpu")]
            Self::DeferredWgpu { waiter, .. } => Some(waiter),
            Self::Cpu | Self::Noop => None,
        }
    }
}

impl core::fmt::Debug for SyncPoint {
    fn fmt(&self, f: &mut core::fmt::Formatter<'_>) -> core::fmt::Result {
        match self {
            Self::Vulkan { semaphore, kind, value, .. } => f
                .debug_struct("Vulkan")
                .field("semaphore", semaphore)
                .field("kind", kind)
                .field("value", value)
                .finish(),
            Self::D3D12 { value, .. } => f.debug_struct("D3D12").field("value", value).finish(),
            Self::D3D11 { key, .. } => f.debug_struct("D3D11").field("key", key).finish(),
            Self::Cuda { value, .. } => f.debug_struct("Cuda").field("value", value).finish(),
            Self::CudaEvent { event, device, .. } => {
                f.debug_struct("CudaEvent").field("event", &(*event as usize)).field("device", device).finish()
            }
            Self::OpenCl { event, value, .. } => {
                f.debug_struct("OpenCl").field("event", &(*event as usize)).field("value", value).finish()
            }
            Self::Metal { value, .. } => f.debug_struct("Metal").field("value", value).finish(),
            Self::OpenGL { semaphore, value, .. } => {
                f.debug_struct("OpenGL").field("semaphore", semaphore).field("value", value).finish()
            }
            Self::OpenGLSync { glsync, backend, .. } => {
                f.debug_struct("OpenGLSync").field("glsync", &(*glsync as usize)).field("backend", backend).finish()
            }
            #[cfg(feature = "wgpu")]
            Self::Wgpu { .. } => f.write_str("SyncPoint::Wgpu"),
            #[cfg(feature = "wgpu")]
            Self::DeferredWgpu { slot, .. } => {
                f.debug_struct("DeferredWgpu").field("committed", &slot.get().is_some()).finish()
            }
            Self::Cpu => f.write_str("SyncPoint::Cpu"),
            Self::Noop => f.write_str("SyncPoint::Noop"),
        }
    }
}

/// Runtime-agnostic single-yield future. Returns `Pending` once then
/// resolves to `Ready(())` on the next poll, with `wake_by_ref` driving
/// the re-poll. Works under any executor (`pollster`, `tokio`, `smol`,
/// `futures::executor`).
///
/// Used by the default `SyncWaiter::wait_async` impl and by per-backend
/// overrides for their fast-path yield-spin loops.
pub async fn yield_once() {
    let mut yielded = false;
    core::future::poll_fn(|cx| {
        if yielded {
            core::task::Poll::Ready(())
        } else {
            yielded = true;
            cx.waker().wake_by_ref();
            core::task::Poll::Pending
        }
    })
    .await
}

/// Helper: convert a `Duration` to nanoseconds, saturating at
/// `u64::MAX` for "wait forever". Matches `vkWaitSemaphores`'
/// `UINT64_MAX` infinite-wait sentinel.
pub fn duration_to_ns(t: Duration) -> u64 {
    u64::try_from(t.as_nanos()).unwrap_or(u64::MAX)
}

/// Helper: convert a `Duration` to milliseconds, saturating at
/// `u32::MAX` for the Win32 `INFINITE` sentinel
/// (`WaitForSingleObject`, `WaitForMultipleObjects`).
pub fn duration_to_ms_u32(t: Duration) -> u32 {
    u32::try_from(t.as_millis()).unwrap_or(u32::MAX)
}