Skip to main content

baracuda_cusolver_sys/
lib.rs

1//! Raw FFI + dynamic loader for NVIDIA cuSOLVER (Dense + Sparse + Refactor).
2//!
3//! `baracuda-cusolver` wraps this with a safe, typed API. Use this
4//! crate directly only if you need a function that the safe layer
5//! hasn't wrapped yet (in which case please file a bug).
6
7#![allow(non_camel_case_types, non_snake_case, non_upper_case_globals)]
8#![warn(missing_debug_implementations)]
9
10use core::ffi::{c_int, c_void};
11use std::sync::OnceLock;
12
13use baracuda_core::{Library, LoaderError, platform};
14use baracuda_cuda_sys::runtime::cudaStream_t;
15use baracuda_types::CudaStatus;
16
17// ---- handles --------------------------------------------------------------
18
19/// Opaque dense cuSOLVER handle.
20pub type cusolverDnHandle_t = *mut c_void;
21/// Opaque sparse cuSOLVER handle.
22pub type cusolverSpHandle_t = *mut c_void;
23/// Opaque sparse-refactor cuSOLVER handle.
24pub type cusolverRfHandle_t = *mut c_void;
25/// Opaque generic-API parameter object for `cusolverDnX*` routines.
26pub type cusolverDnParams_t = *mut c_void;
27/// Opaque parameter object for iterative-refinement solvers.
28pub type cusolverDnIRSParams_t = *mut c_void;
29/// Opaque info object for iterative-refinement solvers.
30pub type cusolverDnIRSInfos_t = *mut c_void;
31/// Opaque control object for Jacobi-based symmetric eigendecomposition.
32pub type syevjInfo_t = *mut c_void;
33/// Opaque control object for Jacobi-based SVD.
34pub type gesvdjInfo_t = *mut c_void;
35
36// ---- enums ----------------------------------------------------------------
37
38/// Transpose selector — same values as `cublasOperation_t` (N=0, T=1, C=2).
39#[repr(i32)]
40#[derive(Copy, Clone, Debug, Eq, PartialEq)]
41pub enum cublasOperation_t {
42    /// No transpose.
43    N = 0,
44    /// Transpose.
45    T = 1,
46    /// Conjugate transpose.
47    C = 2,
48}
49
50/// Triangular fill mode (matches the cuBLAS enum).
51#[repr(i32)]
52#[derive(Copy, Clone, Debug, Eq, PartialEq)]
53pub enum cublasFillMode_t {
54    /// Lower-triangular.
55    Lower = 0,
56    /// Upper-triangular.
57    Upper = 1,
58    /// Full / dense matrix.
59    Full = 2,
60}
61
62/// Side selector for one-sided operators (apply from left or right).
63#[repr(i32)]
64#[derive(Copy, Clone, Debug, Eq, PartialEq)]
65pub enum cublasSideMode_t {
66    /// Apply from the left.
67    Left = 0,
68    /// Apply from the right.
69    Right = 1,
70}
71
72/// Diagonal-unit selector (matches the cuBLAS enum).
73#[repr(i32)]
74#[derive(Copy, Clone, Debug, Eq, PartialEq)]
75pub enum cublasDiagType_t {
76    /// Non-unit diagonal.
77    NonUnit = 0,
78    /// Unit diagonal.
79    Unit = 1,
80}
81
82/// Generalized eigenproblem variant for `sygv*` / `hegv*` routines.
83#[repr(i32)]
84#[derive(Copy, Clone, Debug, Eq, PartialEq)]
85pub enum cusolverEigType_t {
86    /// Type-1 generalized eigenproblem: `A*x = lambda*B*x`.
87    Type1 = 1,
88    /// Type-2 generalized eigenproblem: `A*B*x = lambda*x`.
89    Type2 = 2,
90    /// Type-3 generalized eigenproblem: `B*A*x = lambda*x`.
91    Type3 = 3,
92}
93
94/// Eigenvalue-only vs eigenvalue-and-vector selector for eigensolvers.
95#[repr(i32)]
96#[derive(Copy, Clone, Debug, Eq, PartialEq)]
97pub enum cusolverEigMode_t {
98    /// Compute eigenvalues only.
99    NoVector = 0,
100    /// Compute eigenvalues and eigenvectors.
101    Vector = 1,
102}
103
104/// Subset-of-spectrum selector for partial eigensolvers.
105#[repr(i32)]
106#[derive(Copy, Clone, Debug, Eq, PartialEq)]
107pub enum cusolverEigRange_t {
108    /// Compute all eigenvalues.
109    All = 1001,
110    /// Compute eigenvalues with indices in `[il, iu]`.
111    I = 1002,
112    /// Compute eigenvalues in the half-open interval `(vl, vu]`.
113    V = 1003,
114}
115
116/// Element-dtype selector used by the generic-API (`cusolverDnX*`) routines.
117#[repr(i32)]
118#[derive(Copy, Clone, Debug, Eq, PartialEq)]
119pub enum cudaDataType {
120    /// 32-bit real (`f32`).
121    R_32F = 0,
122    /// 64-bit real (`f64`).
123    R_64F = 1,
124    /// 16-bit real (IEEE half / `f16`).
125    R_16F = 2,
126    /// 32-bit complex (`Complex<f32>`).
127    C_32F = 4,
128    /// 64-bit complex (`Complex<f64>`).
129    C_64F = 5,
130    /// 16-bit real bfloat16.
131    R_16BF = 14,
132}
133
134// ---- status ---------------------------------------------------------------
135
136/// Status / error code returned by cuSOLVER FFI calls.
137#[derive(Copy, Clone, Debug, Eq, PartialEq, Ord, PartialOrd, Hash)]
138#[repr(transparent)]
139pub struct cusolverStatus_t(pub i32);
140
141impl cusolverStatus_t {
142    /// Status: success.
143    pub const SUCCESS: Self = Self(0);
144    /// Status: not initialized.
145    pub const NOT_INITIALIZED: Self = Self(1);
146    /// Status: alloc failed.
147    pub const ALLOC_FAILED: Self = Self(2);
148    /// Status: invalid value.
149    pub const INVALID_VALUE: Self = Self(3);
150    /// Status: arch mismatch.
151    pub const ARCH_MISMATCH: Self = Self(4);
152    /// Status: execution failed.
153    pub const EXECUTION_FAILED: Self = Self(6);
154    /// Status: internal error.
155    pub const INTERNAL_ERROR: Self = Self(7);
156    /// Status: not supported.
157    pub const NOT_SUPPORTED: Self = Self(9);
158    /// Status: zero pivot.
159    pub const ZERO_PIVOT: Self = Self(10);
160
161    /// Returns `true` when this is the success status code.
162    pub const fn is_success(self) -> bool {
163        self.0 == 0
164    }
165}
166
167impl CudaStatus for cusolverStatus_t {
168    fn code(self) -> i32 {
169        self.0
170    }
171    fn name(self) -> &'static str {
172        match self.0 {
173            0 => "CUSOLVER_STATUS_SUCCESS",
174            1 => "CUSOLVER_STATUS_NOT_INITIALIZED",
175            2 => "CUSOLVER_STATUS_ALLOC_FAILED",
176            3 => "CUSOLVER_STATUS_INVALID_VALUE",
177            6 => "CUSOLVER_STATUS_EXECUTION_FAILED",
178            7 => "CUSOLVER_STATUS_INTERNAL_ERROR",
179            9 => "CUSOLVER_STATUS_NOT_SUPPORTED",
180            10 => "CUSOLVER_STATUS_ZERO_PIVOT",
181            _ => "CUSOLVER_STATUS_UNRECOGNIZED",
182        }
183    }
184    fn description(self) -> &'static str {
185        match self.0 {
186            0 => "success",
187            1 => "cuSOLVER not initialized",
188            6 => "execution failed on device",
189            10 => "factorization produced a zero pivot",
190            _ => "unrecognized cuSOLVER status code",
191        }
192    }
193    fn is_success(self) -> bool {
194        cusolverStatus_t::is_success(self)
195    }
196    fn library(self) -> &'static str {
197        "cusolver"
198    }
199}
200
201// ---- complex types (alias to plain 2-element arrays) ---------------------
202
203/// Single-precision complex number, ABI-compatible with cuBLAS / cuSOLVER `cuComplex`.
204#[repr(C)]
205#[derive(Copy, Clone, Debug)]
206pub struct cuComplex {
207    /// Real component (`f32`).
208    pub x: f32,
209    /// Imaginary component (`f32`).
210    pub y: f32,
211}
212
213/// Double-precision complex number, ABI-compatible with cuBLAS / cuSOLVER `cuDoubleComplex`.
214#[repr(C)]
215#[derive(Copy, Clone, Debug)]
216pub struct cuDoubleComplex {
217    /// Real component (`f64`).
218    pub x: f64,
219    /// Imaginary component (`f64`).
220    pub y: f64,
221}
222
223// ---- PFN type declaration macros -----------------------------------------
224
225/// `getrf_bufferSize(handle, m, n, a, lda, lwork) -> status`
226macro_rules! dn_getrf_bufsize {
227    ($(#[$attr:meta])* $name:ident, $t:ty) => {
228        $(#[$attr])*
229        pub type $name = unsafe extern "C" fn(
230            handle: cusolverDnHandle_t,
231            m: c_int,
232            n: c_int,
233            a: *mut $t,
234            lda: c_int,
235            lwork: *mut c_int,
236        ) -> cusolverStatus_t;
237    };
238}
239
240macro_rules! dn_getrf {
241    ($(#[$attr:meta])* $name:ident, $t:ty) => {
242        $(#[$attr])*
243        pub type $name = unsafe extern "C" fn(
244            handle: cusolverDnHandle_t,
245            m: c_int,
246            n: c_int,
247            a: *mut $t,
248            lda: c_int,
249            workspace: *mut $t,
250            ipiv: *mut c_int,
251            info: *mut c_int,
252        ) -> cusolverStatus_t;
253    };
254}
255
256macro_rules! dn_getrs {
257    ($(#[$attr:meta])* $name:ident, $t:ty) => {
258        $(#[$attr])*
259        pub type $name = unsafe extern "C" fn(
260            handle: cusolverDnHandle_t,
261            trans: cublasOperation_t,
262            n: c_int,
263            nrhs: c_int,
264            a: *const $t,
265            lda: c_int,
266            ipiv: *const c_int,
267            b: *mut $t,
268            ldb: c_int,
269            info: *mut c_int,
270        ) -> cusolverStatus_t;
271    };
272}
273
274macro_rules! dn_geqrf_bufsize {
275    ($(#[$attr:meta])* $name:ident, $t:ty) => {
276        $(#[$attr])*
277        pub type $name = unsafe extern "C" fn(
278            handle: cusolverDnHandle_t,
279            m: c_int,
280            n: c_int,
281            a: *mut $t,
282            lda: c_int,
283            lwork: *mut c_int,
284        ) -> cusolverStatus_t;
285    };
286}
287
288macro_rules! dn_geqrf {
289    ($(#[$attr:meta])* $name:ident, $t:ty) => {
290        $(#[$attr])*
291        pub type $name = unsafe extern "C" fn(
292            handle: cusolverDnHandle_t,
293            m: c_int,
294            n: c_int,
295            a: *mut $t,
296            lda: c_int,
297            tau: *mut $t,
298            workspace: *mut $t,
299            lwork: c_int,
300            info: *mut c_int,
301        ) -> cusolverStatus_t;
302    };
303}
304
305macro_rules! dn_potrf_bufsize {
306    ($(#[$attr:meta])* $name:ident, $t:ty) => {
307        $(#[$attr])*
308        pub type $name = unsafe extern "C" fn(
309            handle: cusolverDnHandle_t,
310            uplo: cublasFillMode_t,
311            n: c_int,
312            a: *mut $t,
313            lda: c_int,
314            lwork: *mut c_int,
315        ) -> cusolverStatus_t;
316    };
317}
318
319macro_rules! dn_potrf {
320    ($(#[$attr:meta])* $name:ident, $t:ty) => {
321        $(#[$attr])*
322        pub type $name = unsafe extern "C" fn(
323            handle: cusolverDnHandle_t,
324            uplo: cublasFillMode_t,
325            n: c_int,
326            a: *mut $t,
327            lda: c_int,
328            workspace: *mut $t,
329            lwork: c_int,
330            info: *mut c_int,
331        ) -> cusolverStatus_t;
332    };
333}
334
335macro_rules! dn_potrs {
336    ($(#[$attr:meta])* $name:ident, $t:ty) => {
337        $(#[$attr])*
338        pub type $name = unsafe extern "C" fn(
339            handle: cusolverDnHandle_t,
340            uplo: cublasFillMode_t,
341            n: c_int,
342            nrhs: c_int,
343            a: *const $t,
344            lda: c_int,
345            b: *mut $t,
346            ldb: c_int,
347            info: *mut c_int,
348        ) -> cusolverStatus_t;
349    };
350}
351
352macro_rules! dn_gesvd_bufsize {
353    ($(#[$attr:meta])* $name:ident) => {
354        $(#[$attr])*
355        pub type $name = unsafe extern "C" fn(
356            handle: cusolverDnHandle_t,
357            m: c_int,
358            n: c_int,
359            lwork: *mut c_int,
360        ) -> cusolverStatus_t;
361    };
362}
363
364macro_rules! dn_gesvd_real {
365    ($(#[$attr:meta])* $name:ident, $t:ty) => {
366        $(#[$attr])*
367        pub type $name = unsafe extern "C" fn(
368            handle: cusolverDnHandle_t,
369            jobu: u8,
370            jobvt: u8,
371            m: c_int,
372            n: c_int,
373            a: *mut $t,
374            lda: c_int,
375            s: *mut $t,
376            u: *mut $t,
377            ldu: c_int,
378            vt: *mut $t,
379            ldvt: c_int,
380            work: *mut $t,
381            lwork: c_int,
382            rwork: *mut $t,
383            info: *mut c_int,
384        ) -> cusolverStatus_t;
385    };
386}
387
388macro_rules! dn_gesvd_complex {
389    ($(#[$attr:meta])* $name:ident, $t:ty, $real:ty) => {
390        $(#[$attr])*
391        pub type $name = unsafe extern "C" fn(
392            handle: cusolverDnHandle_t,
393            jobu: u8,
394            jobvt: u8,
395            m: c_int,
396            n: c_int,
397            a: *mut $t,
398            lda: c_int,
399            s: *mut $real,
400            u: *mut $t,
401            ldu: c_int,
402            vt: *mut $t,
403            ldvt: c_int,
404            work: *mut $t,
405            lwork: c_int,
406            rwork: *mut $real,
407            info: *mut c_int,
408        ) -> cusolverStatus_t;
409    };
410}
411
412macro_rules! dn_syevd_bufsize {
413    ($(#[$attr:meta])* $name:ident, $t:ty, $real:ty) => {
414        $(#[$attr])*
415        pub type $name = unsafe extern "C" fn(
416            handle: cusolverDnHandle_t,
417            jobz: cusolverEigMode_t,
418            uplo: cublasFillMode_t,
419            n: c_int,
420            a: *const $t,
421            lda: c_int,
422            w: *const $real,
423            lwork: *mut c_int,
424        ) -> cusolverStatus_t;
425    };
426}
427
428macro_rules! dn_syevd {
429    ($(#[$attr:meta])* $name:ident, $t:ty, $real:ty) => {
430        $(#[$attr])*
431        pub type $name = unsafe extern "C" fn(
432            handle: cusolverDnHandle_t,
433            jobz: cusolverEigMode_t,
434            uplo: cublasFillMode_t,
435            n: c_int,
436            a: *mut $t,
437            lda: c_int,
438            w: *mut $real,
439            work: *mut $t,
440            lwork: c_int,
441            info: *mut c_int,
442        ) -> cusolverStatus_t;
443    };
444}
445
446// ---- core Dn handle ------------------------------------------------------
447
448/// cuSOLVER: create a dense cuSOLVER handle. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
449pub type PFN_cusolverDnCreate =
450    unsafe extern "C" fn(handle: *mut cusolverDnHandle_t) -> cusolverStatus_t;
451/// cuSOLVER: destroy a dense cuSOLVER handle. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
452pub type PFN_cusolverDnDestroy =
453    unsafe extern "C" fn(handle: cusolverDnHandle_t) -> cusolverStatus_t;
454/// cuSOLVER: bind a CUDA stream to a dense cuSOLVER handle. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
455pub type PFN_cusolverDnSetStream =
456    unsafe extern "C" fn(handle: cusolverDnHandle_t, stream: cudaStream_t) -> cusolverStatus_t;
457/// cuSOLVER: query the CUDA stream bound to a dense cuSOLVER handle. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
458pub type PFN_cusolverDnGetStream =
459    unsafe extern "C" fn(handle: cusolverDnHandle_t, stream: *mut cudaStream_t) -> cusolverStatus_t;
460
461/// cuSOLVER: return the cuSOLVER library version. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
462pub type PFN_cusolverGetVersion = unsafe extern "C" fn(version: *mut c_int) -> cusolverStatus_t;
463
464// ---- LU factorization (getrf / getrs) — S/D/C/Z --------------------------
465
466dn_getrf_bufsize!(
467    #[doc = "cuSOLVER: single-precision workspace-size query for LU factorization with partial pivoting. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
468    PFN_cusolverDnSgetrf_bufferSize,
469    f32
470);
471dn_getrf_bufsize!(
472    #[doc = "cuSOLVER: double-precision workspace-size query for LU factorization with partial pivoting. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
473    PFN_cusolverDnDgetrf_bufferSize,
474    f64
475);
476dn_getrf_bufsize!(
477    #[doc = "cuSOLVER: single-precision complex workspace-size query for LU factorization with partial pivoting. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
478    PFN_cusolverDnCgetrf_bufferSize,
479    cuComplex
480);
481dn_getrf_bufsize!(
482    #[doc = "cuSOLVER: double-precision complex workspace-size query for LU factorization with partial pivoting. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
483    PFN_cusolverDnZgetrf_bufferSize,
484    cuDoubleComplex
485);
486
487dn_getrf!(
488    #[doc = "cuSOLVER: single-precision LU factorization with partial pivoting. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
489    PFN_cusolverDnSgetrf,
490    f32
491);
492dn_getrf!(
493    #[doc = "cuSOLVER: double-precision LU factorization with partial pivoting. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
494    PFN_cusolverDnDgetrf,
495    f64
496);
497dn_getrf!(
498    #[doc = "cuSOLVER: single-precision complex LU factorization with partial pivoting. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
499    PFN_cusolverDnCgetrf,
500    cuComplex
501);
502dn_getrf!(
503    #[doc = "cuSOLVER: double-precision complex LU factorization with partial pivoting. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
504    PFN_cusolverDnZgetrf,
505    cuDoubleComplex
506);
507
508dn_getrs!(
509    #[doc = "cuSOLVER: single-precision solve linear system using LU factors. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
510    PFN_cusolverDnSgetrs,
511    f32
512);
513dn_getrs!(
514    #[doc = "cuSOLVER: double-precision solve linear system using LU factors. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
515    PFN_cusolverDnDgetrs,
516    f64
517);
518dn_getrs!(
519    #[doc = "cuSOLVER: single-precision complex solve linear system using LU factors. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
520    PFN_cusolverDnCgetrs,
521    cuComplex
522);
523dn_getrs!(
524    #[doc = "cuSOLVER: double-precision complex solve linear system using LU factors. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
525    PFN_cusolverDnZgetrs,
526    cuDoubleComplex
527);
528
529// ---- QR factorization (geqrf) — S/D/C/Z ----------------------------------
530
531dn_geqrf_bufsize!(
532    #[doc = "cuSOLVER: single-precision workspace-size query for QR factorization (Householder). See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
533    PFN_cusolverDnSgeqrf_bufferSize,
534    f32
535);
536dn_geqrf_bufsize!(
537    #[doc = "cuSOLVER: double-precision workspace-size query for QR factorization (Householder). See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
538    PFN_cusolverDnDgeqrf_bufferSize,
539    f64
540);
541dn_geqrf_bufsize!(
542    #[doc = "cuSOLVER: single-precision complex workspace-size query for QR factorization (Householder). See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
543    PFN_cusolverDnCgeqrf_bufferSize,
544    cuComplex
545);
546dn_geqrf_bufsize!(
547    #[doc = "cuSOLVER: double-precision complex workspace-size query for QR factorization (Householder). See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
548    PFN_cusolverDnZgeqrf_bufferSize,
549    cuDoubleComplex
550);
551
552dn_geqrf!(
553    #[doc = "cuSOLVER: single-precision QR factorization (Householder). See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
554    PFN_cusolverDnSgeqrf,
555    f32
556);
557dn_geqrf!(
558    #[doc = "cuSOLVER: double-precision QR factorization (Householder). See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
559    PFN_cusolverDnDgeqrf,
560    f64
561);
562dn_geqrf!(
563    #[doc = "cuSOLVER: single-precision complex QR factorization (Householder). See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
564    PFN_cusolverDnCgeqrf,
565    cuComplex
566);
567dn_geqrf!(
568    #[doc = "cuSOLVER: double-precision complex QR factorization (Householder). See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
569    PFN_cusolverDnZgeqrf,
570    cuDoubleComplex
571);
572
573// ---- Cholesky (potrf / potrs) — S/D/C/Z ----------------------------------
574
575dn_potrf_bufsize!(
576    #[doc = "cuSOLVER: single-precision workspace-size query for Cholesky factorization. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
577    PFN_cusolverDnSpotrf_bufferSize,
578    f32
579);
580dn_potrf_bufsize!(
581    #[doc = "cuSOLVER: double-precision workspace-size query for Cholesky factorization. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
582    PFN_cusolverDnDpotrf_bufferSize,
583    f64
584);
585dn_potrf_bufsize!(
586    #[doc = "cuSOLVER: single-precision complex workspace-size query for Cholesky factorization. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
587    PFN_cusolverDnCpotrf_bufferSize,
588    cuComplex
589);
590dn_potrf_bufsize!(
591    #[doc = "cuSOLVER: double-precision complex workspace-size query for Cholesky factorization. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
592    PFN_cusolverDnZpotrf_bufferSize,
593    cuDoubleComplex
594);
595
596dn_potrf!(
597    #[doc = "cuSOLVER: single-precision Cholesky factorization. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
598    PFN_cusolverDnSpotrf,
599    f32
600);
601dn_potrf!(
602    #[doc = "cuSOLVER: double-precision Cholesky factorization. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
603    PFN_cusolverDnDpotrf,
604    f64
605);
606dn_potrf!(
607    #[doc = "cuSOLVER: single-precision complex Cholesky factorization. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
608    PFN_cusolverDnCpotrf,
609    cuComplex
610);
611dn_potrf!(
612    #[doc = "cuSOLVER: double-precision complex Cholesky factorization. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
613    PFN_cusolverDnZpotrf,
614    cuDoubleComplex
615);
616
617dn_potrs!(
618    #[doc = "cuSOLVER: single-precision solve linear system using Cholesky factors. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
619    PFN_cusolverDnSpotrs,
620    f32
621);
622dn_potrs!(
623    #[doc = "cuSOLVER: double-precision solve linear system using Cholesky factors. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
624    PFN_cusolverDnDpotrs,
625    f64
626);
627dn_potrs!(
628    #[doc = "cuSOLVER: single-precision complex solve linear system using Cholesky factors. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
629    PFN_cusolverDnCpotrs,
630    cuComplex
631);
632dn_potrs!(
633    #[doc = "cuSOLVER: double-precision complex solve linear system using Cholesky factors. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
634    PFN_cusolverDnZpotrs,
635    cuDoubleComplex
636);
637
638// ---- SVD — S/D/C/Z -------------------------------------------------------
639
640dn_gesvd_bufsize!(
641    #[doc = "cuSOLVER: single-precision workspace-size query for singular value decomposition. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
642    PFN_cusolverDnSgesvd_bufferSize
643);
644dn_gesvd_bufsize!(
645    #[doc = "cuSOLVER: double-precision workspace-size query for singular value decomposition. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
646    PFN_cusolverDnDgesvd_bufferSize
647);
648dn_gesvd_bufsize!(
649    #[doc = "cuSOLVER: single-precision complex workspace-size query for singular value decomposition. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
650    PFN_cusolverDnCgesvd_bufferSize
651);
652dn_gesvd_bufsize!(
653    #[doc = "cuSOLVER: double-precision complex workspace-size query for singular value decomposition. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
654    PFN_cusolverDnZgesvd_bufferSize
655);
656
657dn_gesvd_real!(
658    #[doc = "cuSOLVER: single-precision singular value decomposition. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
659    PFN_cusolverDnSgesvd,
660    f32
661);
662dn_gesvd_real!(
663    #[doc = "cuSOLVER: double-precision singular value decomposition. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
664    PFN_cusolverDnDgesvd,
665    f64
666);
667dn_gesvd_complex!(
668    #[doc = "cuSOLVER: single-precision complex singular value decomposition. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
669    PFN_cusolverDnCgesvd,
670    cuComplex,
671    f32
672);
673dn_gesvd_complex!(
674    #[doc = "cuSOLVER: double-precision complex singular value decomposition. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
675    PFN_cusolverDnZgesvd,
676    cuDoubleComplex,
677    f64
678);
679
680// ---- Symmetric/Hermitian eigendecomposition (syevd/heevd) --------------
681
682dn_syevd_bufsize!(
683    #[doc = "cuSOLVER: single-precision workspace-size query for symmetric eigendecomposition (divide-and-conquer). See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
684    PFN_cusolverDnSsyevd_bufferSize,
685    f32,
686    f32
687);
688dn_syevd_bufsize!(
689    #[doc = "cuSOLVER: double-precision workspace-size query for symmetric eigendecomposition (divide-and-conquer). See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
690    PFN_cusolverDnDsyevd_bufferSize,
691    f64,
692    f64
693);
694dn_syevd_bufsize!(
695    #[doc = "cuSOLVER: single-precision complex workspace-size query for Hermitian eigendecomposition (divide-and-conquer). See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
696    PFN_cusolverDnCheevd_bufferSize,
697    cuComplex,
698    f32
699);
700dn_syevd_bufsize!(
701    #[doc = "cuSOLVER: double-precision complex workspace-size query for Hermitian eigendecomposition (divide-and-conquer). See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
702    PFN_cusolverDnZheevd_bufferSize,
703    cuDoubleComplex,
704    f64
705);
706
707dn_syevd!(
708    #[doc = "cuSOLVER: single-precision symmetric eigendecomposition (divide-and-conquer). See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
709    PFN_cusolverDnSsyevd,
710    f32,
711    f32
712);
713dn_syevd!(
714    #[doc = "cuSOLVER: double-precision symmetric eigendecomposition (divide-and-conquer). See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
715    PFN_cusolverDnDsyevd,
716    f64,
717    f64
718);
719dn_syevd!(
720    #[doc = "cuSOLVER: single-precision complex Hermitian eigendecomposition (divide-and-conquer). See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
721    PFN_cusolverDnCheevd,
722    cuComplex,
723    f32
724);
725dn_syevd!(
726    #[doc = "cuSOLVER: double-precision complex Hermitian eigendecomposition (divide-and-conquer). See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
727    PFN_cusolverDnZheevd,
728    cuDoubleComplex,
729    f64
730);
731
732// ---- Generic 64-bit / mixed-precision API (cusolverDnX…) ----------------
733
734/// cuSOLVER: create a generic-API parameter object. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
735pub type PFN_cusolverDnCreateParams =
736    unsafe extern "C" fn(params: *mut cusolverDnParams_t) -> cusolverStatus_t;
737/// cuSOLVER: destroy a generic-API parameter object. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
738pub type PFN_cusolverDnDestroyParams =
739    unsafe extern "C" fn(params: cusolverDnParams_t) -> cusolverStatus_t;
740
741/// cuSOLVER: generic-API workspace-size query for LU factorization with partial pivoting (dtype configurable via `cudaDataType`). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
742pub type PFN_cusolverDnXgetrf_bufferSize = unsafe extern "C" fn(
743    handle: cusolverDnHandle_t,
744    params: cusolverDnParams_t,
745    m: i64,
746    n: i64,
747    data_type_a: cudaDataType,
748    a: *const c_void,
749    lda: i64,
750    compute_type: cudaDataType,
751    workspace_in_bytes_on_device: *mut usize,
752    workspace_in_bytes_on_host: *mut usize,
753) -> cusolverStatus_t;
754
755/// cuSOLVER: generic-API LU factorization with partial pivoting (dtype configurable via `cudaDataType`). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
756pub type PFN_cusolverDnXgetrf = unsafe extern "C" fn(
757    handle: cusolverDnHandle_t,
758    params: cusolverDnParams_t,
759    m: i64,
760    n: i64,
761    data_type_a: cudaDataType,
762    a: *mut c_void,
763    lda: i64,
764    ipiv: *mut i64,
765    compute_type: cudaDataType,
766    bufferondevice: *mut c_void,
767    workspace_in_bytes_on_device: usize,
768    bufferonhost: *mut c_void,
769    workspace_in_bytes_on_host: usize,
770    info: *mut c_int,
771) -> cusolverStatus_t;
772
773/// cuSOLVER: generic-API solve linear system using LU factors (dtype configurable via `cudaDataType`). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
774pub type PFN_cusolverDnXgetrs = unsafe extern "C" fn(
775    handle: cusolverDnHandle_t,
776    params: cusolverDnParams_t,
777    trans: cublasOperation_t,
778    n: i64,
779    nrhs: i64,
780    data_type_a: cudaDataType,
781    a: *const c_void,
782    lda: i64,
783    ipiv: *const i64,
784    data_type_b: cudaDataType,
785    b: *mut c_void,
786    ldb: i64,
787    info: *mut c_int,
788) -> cusolverStatus_t;
789
790/// cuSOLVER: generic-API workspace-size query for QR factorization (Householder) (dtype configurable via `cudaDataType`). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
791pub type PFN_cusolverDnXgeqrf_bufferSize = unsafe extern "C" fn(
792    handle: cusolverDnHandle_t,
793    params: cusolverDnParams_t,
794    m: i64,
795    n: i64,
796    data_type_a: cudaDataType,
797    a: *const c_void,
798    lda: i64,
799    data_type_tau: cudaDataType,
800    tau: *const c_void,
801    compute_type: cudaDataType,
802    workspace_in_bytes_on_device: *mut usize,
803    workspace_in_bytes_on_host: *mut usize,
804) -> cusolverStatus_t;
805
806/// cuSOLVER: generic-API QR factorization (Householder) (dtype configurable via `cudaDataType`). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
807pub type PFN_cusolverDnXgeqrf = unsafe extern "C" fn(
808    handle: cusolverDnHandle_t,
809    params: cusolverDnParams_t,
810    m: i64,
811    n: i64,
812    data_type_a: cudaDataType,
813    a: *mut c_void,
814    lda: i64,
815    data_type_tau: cudaDataType,
816    tau: *mut c_void,
817    compute_type: cudaDataType,
818    bufferondevice: *mut c_void,
819    workspace_in_bytes_on_device: usize,
820    bufferonhost: *mut c_void,
821    workspace_in_bytes_on_host: usize,
822    info: *mut c_int,
823) -> cusolverStatus_t;
824
825/// cuSOLVER: generic-API workspace-size query for Cholesky factorization (dtype configurable via `cudaDataType`). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
826pub type PFN_cusolverDnXpotrf_bufferSize = unsafe extern "C" fn(
827    handle: cusolverDnHandle_t,
828    params: cusolverDnParams_t,
829    uplo: cublasFillMode_t,
830    n: i64,
831    data_type_a: cudaDataType,
832    a: *const c_void,
833    lda: i64,
834    compute_type: cudaDataType,
835    workspace_in_bytes_on_device: *mut usize,
836    workspace_in_bytes_on_host: *mut usize,
837) -> cusolverStatus_t;
838
839/// cuSOLVER: generic-API Cholesky factorization (dtype configurable via `cudaDataType`). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
840pub type PFN_cusolverDnXpotrf = unsafe extern "C" fn(
841    handle: cusolverDnHandle_t,
842    params: cusolverDnParams_t,
843    uplo: cublasFillMode_t,
844    n: i64,
845    data_type_a: cudaDataType,
846    a: *mut c_void,
847    lda: i64,
848    compute_type: cudaDataType,
849    bufferondevice: *mut c_void,
850    workspace_in_bytes_on_device: usize,
851    bufferonhost: *mut c_void,
852    workspace_in_bytes_on_host: usize,
853    info: *mut c_int,
854) -> cusolverStatus_t;
855
856/// cuSOLVER: generic-API solve linear system using Cholesky factors (dtype configurable via `cudaDataType`). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
857pub type PFN_cusolverDnXpotrs = unsafe extern "C" fn(
858    handle: cusolverDnHandle_t,
859    params: cusolverDnParams_t,
860    uplo: cublasFillMode_t,
861    n: i64,
862    nrhs: i64,
863    data_type_a: cudaDataType,
864    a: *const c_void,
865    lda: i64,
866    data_type_b: cudaDataType,
867    b: *mut c_void,
868    ldb: i64,
869    info: *mut c_int,
870) -> cusolverStatus_t;
871
872/// cuSOLVER: generic-API workspace-size query for symmetric eigendecomposition (divide-and-conquer) (dtype configurable via `cudaDataType`). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
873pub type PFN_cusolverDnXsyevd_bufferSize = unsafe extern "C" fn(
874    handle: cusolverDnHandle_t,
875    params: cusolverDnParams_t,
876    jobz: cusolverEigMode_t,
877    uplo: cublasFillMode_t,
878    n: i64,
879    data_type_a: cudaDataType,
880    a: *const c_void,
881    lda: i64,
882    data_type_w: cudaDataType,
883    w: *const c_void,
884    compute_type: cudaDataType,
885    device_bytes: *mut usize,
886    host_bytes: *mut usize,
887) -> cusolverStatus_t;
888
889/// cuSOLVER: generic-API symmetric eigendecomposition (divide-and-conquer) (dtype configurable via `cudaDataType`). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
890pub type PFN_cusolverDnXsyevd = unsafe extern "C" fn(
891    handle: cusolverDnHandle_t,
892    params: cusolverDnParams_t,
893    jobz: cusolverEigMode_t,
894    uplo: cublasFillMode_t,
895    n: i64,
896    data_type_a: cudaDataType,
897    a: *mut c_void,
898    lda: i64,
899    data_type_w: cudaDataType,
900    w: *mut c_void,
901    compute_type: cudaDataType,
902    bufferondevice: *mut c_void,
903    device_bytes: usize,
904    bufferonhost: *mut c_void,
905    host_bytes: usize,
906    info: *mut c_int,
907) -> cusolverStatus_t;
908
909// ==========================================================================
910// Jacobi-based eigendecompositions + SVD
911// ==========================================================================
912
913/// cuSOLVER: create a Jacobi-eigendecomposition control object. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
914pub type PFN_cusolverDnCreateSyevjInfo =
915    unsafe extern "C" fn(info: *mut syevjInfo_t) -> cusolverStatus_t;
916/// cuSOLVER: destroy a Jacobi-eigendecomposition control object. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
917pub type PFN_cusolverDnDestroySyevjInfo =
918    unsafe extern "C" fn(info: syevjInfo_t) -> cusolverStatus_t;
919/// cuSOLVER: set convergence tolerance on a Jacobi-eigendecomposition control object. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
920pub type PFN_cusolverDnXsyevjSetTolerance =
921    unsafe extern "C" fn(info: syevjInfo_t, tolerance: f64) -> cusolverStatus_t;
922/// cuSOLVER: set the maximum number of sweeps on a Jacobi-eigendecomposition control object. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
923pub type PFN_cusolverDnXsyevjSetMaxSweeps =
924    unsafe extern "C" fn(info: syevjInfo_t, max_sweeps: c_int) -> cusolverStatus_t;
925
926macro_rules! dn_syevj_bufsize {
927    ($(#[$attr:meta])* $name:ident, $t:ty, $real:ty) => {
928        $(#[$attr])*
929        pub type $name = unsafe extern "C" fn(
930            handle: cusolverDnHandle_t,
931            jobz: cusolverEigMode_t,
932            uplo: cublasFillMode_t,
933            n: c_int,
934            a: *const $t,
935            lda: c_int,
936            w: *const $real,
937            lwork: *mut c_int,
938            params: syevjInfo_t,
939        ) -> cusolverStatus_t;
940    };
941}
942dn_syevj_bufsize!(
943    #[doc = "cuSOLVER: single-precision workspace-size query for symmetric eigendecomposition (Jacobi). See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
944    PFN_cusolverDnSsyevj_bufferSize,
945    f32,
946    f32
947);
948dn_syevj_bufsize!(
949    #[doc = "cuSOLVER: double-precision workspace-size query for symmetric eigendecomposition (Jacobi). See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
950    PFN_cusolverDnDsyevj_bufferSize,
951    f64,
952    f64
953);
954dn_syevj_bufsize!(
955    #[doc = "cuSOLVER: single-precision complex workspace-size query for Hermitian eigendecomposition (Jacobi). See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
956    PFN_cusolverDnCheevj_bufferSize,
957    cuComplex,
958    f32
959);
960dn_syevj_bufsize!(
961    #[doc = "cuSOLVER: double-precision complex workspace-size query for Hermitian eigendecomposition (Jacobi). See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
962    PFN_cusolverDnZheevj_bufferSize,
963    cuDoubleComplex,
964    f64
965);
966
967macro_rules! dn_syevj {
968    ($(#[$attr:meta])* $name:ident, $t:ty, $real:ty) => {
969        $(#[$attr])*
970        pub type $name = unsafe extern "C" fn(
971            handle: cusolverDnHandle_t,
972            jobz: cusolverEigMode_t,
973            uplo: cublasFillMode_t,
974            n: c_int,
975            a: *mut $t,
976            lda: c_int,
977            w: *mut $real,
978            work: *mut $t,
979            lwork: c_int,
980            info: *mut c_int,
981            params: syevjInfo_t,
982        ) -> cusolverStatus_t;
983    };
984}
985dn_syevj!(
986    #[doc = "cuSOLVER: single-precision symmetric eigendecomposition (Jacobi). See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
987    PFN_cusolverDnSsyevj,
988    f32,
989    f32
990);
991dn_syevj!(
992    #[doc = "cuSOLVER: double-precision symmetric eigendecomposition (Jacobi). See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
993    PFN_cusolverDnDsyevj,
994    f64,
995    f64
996);
997dn_syevj!(
998    #[doc = "cuSOLVER: single-precision complex Hermitian eigendecomposition (Jacobi). See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
999    PFN_cusolverDnCheevj,
1000    cuComplex,
1001    f32
1002);
1003dn_syevj!(
1004    #[doc = "cuSOLVER: double-precision complex Hermitian eigendecomposition (Jacobi). See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
1005    PFN_cusolverDnZheevj,
1006    cuDoubleComplex,
1007    f64
1008);
1009
1010/// cuSOLVER: create a Jacobi-SVD control object. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1011pub type PFN_cusolverDnCreateGesvdjInfo =
1012    unsafe extern "C" fn(info: *mut gesvdjInfo_t) -> cusolverStatus_t;
1013/// cuSOLVER: destroy a Jacobi-SVD control object. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1014pub type PFN_cusolverDnDestroyGesvdjInfo =
1015    unsafe extern "C" fn(info: gesvdjInfo_t) -> cusolverStatus_t;
1016
1017macro_rules! dn_gesvdj_bufsize {
1018    ($(#[$attr:meta])* $name:ident, $t:ty, $real:ty) => {
1019        $(#[$attr])*
1020        pub type $name = unsafe extern "C" fn(
1021            handle: cusolverDnHandle_t,
1022            jobz: cusolverEigMode_t,
1023            econ: c_int,
1024            m: c_int,
1025            n: c_int,
1026            a: *const $t,
1027            lda: c_int,
1028            s: *const $real,
1029            u: *const $t,
1030            ldu: c_int,
1031            v: *const $t,
1032            ldv: c_int,
1033            lwork: *mut c_int,
1034            params: gesvdjInfo_t,
1035        ) -> cusolverStatus_t;
1036    };
1037}
1038dn_gesvdj_bufsize!(
1039    #[doc = "cuSOLVER: single-precision workspace-size query for Jacobi-method singular value decomposition. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
1040    PFN_cusolverDnSgesvdj_bufferSize,
1041    f32,
1042    f32
1043);
1044dn_gesvdj_bufsize!(
1045    #[doc = "cuSOLVER: double-precision workspace-size query for Jacobi-method singular value decomposition. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
1046    PFN_cusolverDnDgesvdj_bufferSize,
1047    f64,
1048    f64
1049);
1050dn_gesvdj_bufsize!(
1051    #[doc = "cuSOLVER: single-precision complex workspace-size query for Jacobi-method singular value decomposition. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
1052    PFN_cusolverDnCgesvdj_bufferSize,
1053    cuComplex,
1054    f32
1055);
1056dn_gesvdj_bufsize!(
1057    #[doc = "cuSOLVER: double-precision complex workspace-size query for Jacobi-method singular value decomposition. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
1058    PFN_cusolverDnZgesvdj_bufferSize,
1059    cuDoubleComplex,
1060    f64
1061);
1062
1063macro_rules! dn_gesvdj {
1064    ($(#[$attr:meta])* $name:ident, $t:ty, $real:ty) => {
1065        $(#[$attr])*
1066        pub type $name = unsafe extern "C" fn(
1067            handle: cusolverDnHandle_t,
1068            jobz: cusolverEigMode_t,
1069            econ: c_int,
1070            m: c_int,
1071            n: c_int,
1072            a: *mut $t,
1073            lda: c_int,
1074            s: *mut $real,
1075            u: *mut $t,
1076            ldu: c_int,
1077            v: *mut $t,
1078            ldv: c_int,
1079            work: *mut $t,
1080            lwork: c_int,
1081            info: *mut c_int,
1082            params: gesvdjInfo_t,
1083        ) -> cusolverStatus_t;
1084    };
1085}
1086dn_gesvdj!(
1087    #[doc = "cuSOLVER: single-precision Jacobi-method singular value decomposition. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
1088    PFN_cusolverDnSgesvdj,
1089    f32,
1090    f32
1091);
1092dn_gesvdj!(
1093    #[doc = "cuSOLVER: double-precision Jacobi-method singular value decomposition. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
1094    PFN_cusolverDnDgesvdj,
1095    f64,
1096    f64
1097);
1098dn_gesvdj!(
1099    #[doc = "cuSOLVER: single-precision complex Jacobi-method singular value decomposition. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
1100    PFN_cusolverDnCgesvdj,
1101    cuComplex,
1102    f32
1103);
1104dn_gesvdj!(
1105    #[doc = "cuSOLVER: double-precision complex Jacobi-method singular value decomposition. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
1106    PFN_cusolverDnZgesvdj,
1107    cuDoubleComplex,
1108    f64
1109);
1110
1111// ==========================================================================
1112// Apply Q from QR (orgqr / ormqr)
1113// ==========================================================================
1114
1115macro_rules! dn_orgqr_bufsize {
1116    ($(#[$attr:meta])* $name:ident, $t:ty) => {
1117        $(#[$attr])*
1118        pub type $name = unsafe extern "C" fn(
1119            handle: cusolverDnHandle_t,
1120            m: c_int,
1121            n: c_int,
1122            k: c_int,
1123            a: *const $t,
1124            lda: c_int,
1125            tau: *const $t,
1126            lwork: *mut c_int,
1127        ) -> cusolverStatus_t;
1128    };
1129}
1130dn_orgqr_bufsize!(
1131    #[doc = "cuSOLVER: single-precision workspace-size query for generate the explicit Q from a QR factorization. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
1132    PFN_cusolverDnSorgqr_bufferSize,
1133    f32
1134);
1135dn_orgqr_bufsize!(
1136    #[doc = "cuSOLVER: double-precision workspace-size query for generate the explicit Q from a QR factorization. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
1137    PFN_cusolverDnDorgqr_bufferSize,
1138    f64
1139);
1140dn_orgqr_bufsize!(
1141    #[doc = "cuSOLVER: single-precision complex workspace-size query for generate the explicit unitary Q from a QR factorization. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
1142    PFN_cusolverDnCungqr_bufferSize,
1143    cuComplex
1144);
1145dn_orgqr_bufsize!(
1146    #[doc = "cuSOLVER: double-precision complex workspace-size query for generate the explicit unitary Q from a QR factorization. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
1147    PFN_cusolverDnZungqr_bufferSize,
1148    cuDoubleComplex
1149);
1150
1151macro_rules! dn_orgqr {
1152    ($(#[$attr:meta])* $name:ident, $t:ty) => {
1153        $(#[$attr])*
1154        pub type $name = unsafe extern "C" fn(
1155            handle: cusolverDnHandle_t,
1156            m: c_int,
1157            n: c_int,
1158            k: c_int,
1159            a: *mut $t,
1160            lda: c_int,
1161            tau: *const $t,
1162            work: *mut $t,
1163            lwork: c_int,
1164            info: *mut c_int,
1165        ) -> cusolverStatus_t;
1166    };
1167}
1168dn_orgqr!(
1169    #[doc = "cuSOLVER: single-precision generate the explicit Q from a QR factorization. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
1170    PFN_cusolverDnSorgqr,
1171    f32
1172);
1173dn_orgqr!(
1174    #[doc = "cuSOLVER: double-precision generate the explicit Q from a QR factorization. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
1175    PFN_cusolverDnDorgqr,
1176    f64
1177);
1178dn_orgqr!(
1179    #[doc = "cuSOLVER: single-precision complex generate the explicit unitary Q from a QR factorization. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
1180    PFN_cusolverDnCungqr,
1181    cuComplex
1182);
1183dn_orgqr!(
1184    #[doc = "cuSOLVER: double-precision complex generate the explicit unitary Q from a QR factorization. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
1185    PFN_cusolverDnZungqr,
1186    cuDoubleComplex
1187);
1188
1189macro_rules! dn_ormqr_bufsize {
1190    ($(#[$attr:meta])* $name:ident, $t:ty) => {
1191        $(#[$attr])*
1192        pub type $name = unsafe extern "C" fn(
1193            handle: cusolverDnHandle_t,
1194            side: c_int,
1195            trans: cublasOperation_t,
1196            m: c_int,
1197            n: c_int,
1198            k: c_int,
1199            a: *const $t,
1200            lda: c_int,
1201            tau: *const $t,
1202            c: *const $t,
1203            ldc: c_int,
1204            lwork: *mut c_int,
1205        ) -> cusolverStatus_t;
1206    };
1207}
1208dn_ormqr_bufsize!(
1209    #[doc = "cuSOLVER: single-precision workspace-size query for apply Q (or Q^T) from a QR factorization to a matrix. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
1210    PFN_cusolverDnSormqr_bufferSize,
1211    f32
1212);
1213dn_ormqr_bufsize!(
1214    #[doc = "cuSOLVER: double-precision workspace-size query for apply Q (or Q^T) from a QR factorization to a matrix. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
1215    PFN_cusolverDnDormqr_bufferSize,
1216    f64
1217);
1218dn_ormqr_bufsize!(
1219    #[doc = "cuSOLVER: single-precision complex workspace-size query for apply Q (or Q^H) from a QR factorization to a matrix. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
1220    PFN_cusolverDnCunmqr_bufferSize,
1221    cuComplex
1222);
1223dn_ormqr_bufsize!(
1224    #[doc = "cuSOLVER: double-precision complex workspace-size query for apply Q (or Q^H) from a QR factorization to a matrix. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
1225    PFN_cusolverDnZunmqr_bufferSize,
1226    cuDoubleComplex
1227);
1228
1229macro_rules! dn_ormqr {
1230    ($(#[$attr:meta])* $name:ident, $t:ty) => {
1231        $(#[$attr])*
1232        pub type $name = unsafe extern "C" fn(
1233            handle: cusolverDnHandle_t,
1234            side: c_int,
1235            trans: cublasOperation_t,
1236            m: c_int,
1237            n: c_int,
1238            k: c_int,
1239            a: *const $t,
1240            lda: c_int,
1241            tau: *const $t,
1242            c: *mut $t,
1243            ldc: c_int,
1244            work: *mut $t,
1245            lwork: c_int,
1246            info: *mut c_int,
1247        ) -> cusolverStatus_t;
1248    };
1249}
1250dn_ormqr!(
1251    #[doc = "cuSOLVER: single-precision apply Q (or Q^T) from a QR factorization to a matrix. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
1252    PFN_cusolverDnSormqr,
1253    f32
1254);
1255dn_ormqr!(
1256    #[doc = "cuSOLVER: double-precision apply Q (or Q^T) from a QR factorization to a matrix. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
1257    PFN_cusolverDnDormqr,
1258    f64
1259);
1260dn_ormqr!(
1261    #[doc = "cuSOLVER: single-precision complex apply Q (or Q^H) from a QR factorization to a matrix. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
1262    PFN_cusolverDnCunmqr,
1263    cuComplex
1264);
1265dn_ormqr!(
1266    #[doc = "cuSOLVER: double-precision complex apply Q (or Q^H) from a QR factorization to a matrix. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
1267    PFN_cusolverDnZunmqr,
1268    cuDoubleComplex
1269);
1270
1271// ---- Sparse cuSOLVER -----------------------------------------------------
1272
1273/// cuSOLVER: create a sparse cuSOLVER handle. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1274pub type PFN_cusolverSpCreate =
1275    unsafe extern "C" fn(handle: *mut cusolverSpHandle_t) -> cusolverStatus_t;
1276/// cuSOLVER: destroy a sparse cuSOLVER handle. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1277pub type PFN_cusolverSpDestroy =
1278    unsafe extern "C" fn(handle: cusolverSpHandle_t) -> cusolverStatus_t;
1279/// cuSOLVER: bind a CUDA stream to a sparse cuSOLVER handle. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1280pub type PFN_cusolverSpSetStream =
1281    unsafe extern "C" fn(handle: cusolverSpHandle_t, stream: cudaStream_t) -> cusolverStatus_t;
1282
1283/// cuSOLVER: single-precision sparse linear solve via Cholesky on a CSR matrix. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1284pub type PFN_cusolverSpScsrlsvchol = unsafe extern "C" fn(
1285    handle: cusolverSpHandle_t,
1286    m: c_int,
1287    nnz: c_int,
1288    descr_a: *mut c_void,
1289    csr_val: *const f32,
1290    csr_row_ptr: *const c_int,
1291    csr_col_ind: *const c_int,
1292    b: *const f32,
1293    tol: f32,
1294    reorder: c_int,
1295    x: *mut f32,
1296    singularity: *mut c_int,
1297) -> cusolverStatus_t;
1298
1299/// cuSOLVER: double-precision sparse linear solve via Cholesky on a CSR matrix. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1300pub type PFN_cusolverSpDcsrlsvchol = unsafe extern "C" fn(
1301    handle: cusolverSpHandle_t,
1302    m: c_int,
1303    nnz: c_int,
1304    descr_a: *mut c_void,
1305    csr_val: *const f64,
1306    csr_row_ptr: *const c_int,
1307    csr_col_ind: *const c_int,
1308    b: *const f64,
1309    tol: f64,
1310    reorder: c_int,
1311    x: *mut f64,
1312    singularity: *mut c_int,
1313) -> cusolverStatus_t;
1314
1315/// cuSOLVER: single-precision sparse linear solve via QR on a CSR matrix. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1316pub type PFN_cusolverSpScsrlsvqr = unsafe extern "C" fn(
1317    handle: cusolverSpHandle_t,
1318    m: c_int,
1319    nnz: c_int,
1320    descr_a: *mut c_void,
1321    csr_val: *const f32,
1322    csr_row_ptr: *const c_int,
1323    csr_col_ind: *const c_int,
1324    b: *const f32,
1325    tol: f32,
1326    reorder: c_int,
1327    x: *mut f32,
1328    singularity: *mut c_int,
1329) -> cusolverStatus_t;
1330
1331/// cuSOLVER: double-precision sparse linear solve via QR on a CSR matrix. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1332pub type PFN_cusolverSpDcsrlsvqr = unsafe extern "C" fn(
1333    handle: cusolverSpHandle_t,
1334    m: c_int,
1335    nnz: c_int,
1336    descr_a: *mut c_void,
1337    csr_val: *const f64,
1338    csr_row_ptr: *const c_int,
1339    csr_col_ind: *const c_int,
1340    b: *const f64,
1341    tol: f64,
1342    reorder: c_int,
1343    x: *mut f64,
1344    singularity: *mut c_int,
1345) -> cusolverStatus_t;
1346
1347// ---- Refactor (Rf) sparse LU refactorization -----------------------------
1348
1349/// cuSOLVER: create a sparse-refactor cuSOLVER handle. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1350pub type PFN_cusolverRfCreate =
1351    unsafe extern "C" fn(handle: *mut cusolverRfHandle_t) -> cusolverStatus_t;
1352/// cuSOLVER: destroy a sparse-refactor cuSOLVER handle. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1353pub type PFN_cusolverRfDestroy =
1354    unsafe extern "C" fn(handle: cusolverRfHandle_t) -> cusolverStatus_t;
1355/// cuSOLVER: supply sparse triangular factors and pivot vectors to the refactor handle. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1356pub type PFN_cusolverRfSetupDevice = unsafe extern "C" fn(
1357    n: c_int,
1358    nnz_a: c_int,
1359    h_csr_row_ptr_a: *mut c_int,
1360    h_csr_col_ind_a: *mut c_int,
1361    h_csr_val_a: *mut f64,
1362    nnz_l: c_int,
1363    h_csr_row_ptr_l: *mut c_int,
1364    h_csr_col_ind_l: *mut c_int,
1365    h_csr_val_l: *mut f64,
1366    nnz_u: c_int,
1367    h_csr_row_ptr_u: *mut c_int,
1368    h_csr_col_ind_u: *mut c_int,
1369    h_csr_val_u: *mut f64,
1370    p: *mut c_int,
1371    q: *mut c_int,
1372    handle: cusolverRfHandle_t,
1373) -> cusolverStatus_t;
1374/// cuSOLVER: analyze the sparsity pattern of the refactor handle. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1375pub type PFN_cusolverRfAnalyze =
1376    unsafe extern "C" fn(handle: cusolverRfHandle_t) -> cusolverStatus_t;
1377/// cuSOLVER: refactor with new numerical values reusing the analyzed pattern. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1378pub type PFN_cusolverRfRefactor =
1379    unsafe extern "C" fn(handle: cusolverRfHandle_t) -> cusolverStatus_t;
1380/// cuSOLVER: solve linear systems using the refactor handle. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1381pub type PFN_cusolverRfSolve = unsafe extern "C" fn(
1382    handle: cusolverRfHandle_t,
1383    p: *mut c_int,
1384    q: *mut c_int,
1385    nrhs: c_int,
1386    temp: *mut f64,
1387    ld_temp: c_int,
1388    xf: *mut f64,
1389    ld_xf: c_int,
1390) -> cusolverStatus_t;
1391
1392// ==========================================================================
1393// Least-squares (gels): A*X = B → X
1394// ==========================================================================
1395
1396macro_rules! dn_gels_bufsize {
1397    ($(#[$attr:meta])* $name:ident, $t:ty) => {
1398        $(#[$attr])*
1399        pub type $name = unsafe extern "C" fn(
1400            handle: cusolverDnHandle_t,
1401            m: c_int,
1402            n: c_int,
1403            nrhs: c_int,
1404            d_a: *mut $t,
1405            lda: c_int,
1406            d_b: *mut $t,
1407            ldb: c_int,
1408            d_x: *mut $t,
1409            ldx: c_int,
1410            d_work: *mut c_void,
1411            lwork_bytes: *mut usize,
1412        ) -> cusolverStatus_t;
1413    };
1414}
1415dn_gels_bufsize!(
1416    #[doc = "cuSOLVER: single-precision workspace-size query for least-squares solver (A*X = B). See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
1417    PFN_cusolverDnSSgels_bufferSize,
1418    f32
1419);
1420dn_gels_bufsize!(
1421    #[doc = "cuSOLVER: double-precision workspace-size query for least-squares solver (A*X = B). See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
1422    PFN_cusolverDnDDgels_bufferSize,
1423    f64
1424);
1425dn_gels_bufsize!(
1426    #[doc = "cuSOLVER: single-precision complex workspace-size query for least-squares solver (A*X = B). See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
1427    PFN_cusolverDnCCgels_bufferSize,
1428    cuComplex
1429);
1430dn_gels_bufsize!(
1431    #[doc = "cuSOLVER: double-precision complex workspace-size query for least-squares solver (A*X = B). See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
1432    PFN_cusolverDnZZgels_bufferSize,
1433    cuDoubleComplex
1434);
1435
1436macro_rules! dn_gels {
1437    ($(#[$attr:meta])* $name:ident, $t:ty) => {
1438        $(#[$attr])*
1439        pub type $name = unsafe extern "C" fn(
1440            handle: cusolverDnHandle_t,
1441            m: c_int,
1442            n: c_int,
1443            nrhs: c_int,
1444            d_a: *mut $t,
1445            lda: c_int,
1446            d_b: *mut $t,
1447            ldb: c_int,
1448            d_x: *mut $t,
1449            ldx: c_int,
1450            d_work: *mut c_void,
1451            lwork_bytes: usize,
1452            iter: *mut c_int,
1453            d_info: *mut c_int,
1454        ) -> cusolverStatus_t;
1455    };
1456}
1457dn_gels!(
1458    #[doc = "cuSOLVER: single-precision least-squares solver (A*X = B). See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
1459    PFN_cusolverDnSSgels,
1460    f32
1461);
1462dn_gels!(
1463    #[doc = "cuSOLVER: double-precision least-squares solver (A*X = B). See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
1464    PFN_cusolverDnDDgels,
1465    f64
1466);
1467dn_gels!(
1468    #[doc = "cuSOLVER: single-precision complex least-squares solver (A*X = B). See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
1469    PFN_cusolverDnCCgels,
1470    cuComplex
1471);
1472dn_gels!(
1473    #[doc = "cuSOLVER: double-precision complex least-squares solver (A*X = B). See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
1474    PFN_cusolverDnZZgels,
1475    cuDoubleComplex
1476);
1477
1478// ==========================================================================
1479// Inverse from Cholesky (potri)
1480// ==========================================================================
1481
1482macro_rules! dn_potri_bufsize {
1483    ($(#[$attr:meta])* $name:ident, $t:ty) => {
1484        $(#[$attr])*
1485        pub type $name = unsafe extern "C" fn(
1486            handle: cusolverDnHandle_t,
1487            uplo: cublasFillMode_t,
1488            n: c_int,
1489            a: *mut $t,
1490            lda: c_int,
1491            lwork: *mut c_int,
1492        ) -> cusolverStatus_t;
1493    };
1494}
1495dn_potri_bufsize!(
1496    #[doc = "cuSOLVER: single-precision workspace-size query for matrix inverse from Cholesky factors. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
1497    PFN_cusolverDnSpotri_bufferSize,
1498    f32
1499);
1500dn_potri_bufsize!(
1501    #[doc = "cuSOLVER: double-precision workspace-size query for matrix inverse from Cholesky factors. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
1502    PFN_cusolverDnDpotri_bufferSize,
1503    f64
1504);
1505dn_potri_bufsize!(
1506    #[doc = "cuSOLVER: single-precision complex workspace-size query for matrix inverse from Cholesky factors. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
1507    PFN_cusolverDnCpotri_bufferSize,
1508    cuComplex
1509);
1510dn_potri_bufsize!(
1511    #[doc = "cuSOLVER: double-precision complex workspace-size query for matrix inverse from Cholesky factors. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
1512    PFN_cusolverDnZpotri_bufferSize,
1513    cuDoubleComplex
1514);
1515
1516macro_rules! dn_potri {
1517    ($(#[$attr:meta])* $name:ident, $t:ty) => {
1518        $(#[$attr])*
1519        pub type $name = unsafe extern "C" fn(
1520            handle: cusolverDnHandle_t,
1521            uplo: cublasFillMode_t,
1522            n: c_int,
1523            a: *mut $t,
1524            lda: c_int,
1525            work: *mut $t,
1526            lwork: c_int,
1527            info: *mut c_int,
1528        ) -> cusolverStatus_t;
1529    };
1530}
1531dn_potri!(
1532    #[doc = "cuSOLVER: single-precision matrix inverse from Cholesky factors. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
1533    PFN_cusolverDnSpotri,
1534    f32
1535);
1536dn_potri!(
1537    #[doc = "cuSOLVER: double-precision matrix inverse from Cholesky factors. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
1538    PFN_cusolverDnDpotri,
1539    f64
1540);
1541dn_potri!(
1542    #[doc = "cuSOLVER: single-precision complex matrix inverse from Cholesky factors. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
1543    PFN_cusolverDnCpotri,
1544    cuComplex
1545);
1546dn_potri!(
1547    #[doc = "cuSOLVER: double-precision complex matrix inverse from Cholesky factors. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
1548    PFN_cusolverDnZpotri,
1549    cuDoubleComplex
1550);
1551
1552// ==========================================================================
1553// Batched Jacobi eigen / SVD
1554// ==========================================================================
1555
1556macro_rules! dn_syevj_batched_bufsize {
1557    ($(#[$attr:meta])* $name:ident, $t:ty, $real:ty) => {
1558        $(#[$attr])*
1559        pub type $name = unsafe extern "C" fn(
1560            handle: cusolverDnHandle_t,
1561            jobz: cusolverEigMode_t,
1562            uplo: cublasFillMode_t,
1563            n: c_int,
1564            a: *const $t,
1565            lda: c_int,
1566            w: *const $real,
1567            lwork: *mut c_int,
1568            params: syevjInfo_t,
1569            batch_size: c_int,
1570        ) -> cusolverStatus_t;
1571    };
1572}
1573dn_syevj_batched_bufsize!(
1574    #[doc = "cuSOLVER: single-precision workspace-size query for batched symmetric eigendecomposition (Jacobi). See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
1575    PFN_cusolverDnSsyevjBatched_bufferSize,
1576    f32,
1577    f32
1578);
1579dn_syevj_batched_bufsize!(
1580    #[doc = "cuSOLVER: double-precision workspace-size query for batched symmetric eigendecomposition (Jacobi). See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
1581    PFN_cusolverDnDsyevjBatched_bufferSize,
1582    f64,
1583    f64
1584);
1585dn_syevj_batched_bufsize!(
1586    #[doc = "cuSOLVER: single-precision complex workspace-size query for batched Hermitian eigendecomposition (Jacobi). See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
1587    PFN_cusolverDnCheevjBatched_bufferSize,
1588    cuComplex,
1589    f32
1590);
1591dn_syevj_batched_bufsize!(
1592    #[doc = "cuSOLVER: double-precision complex workspace-size query for batched Hermitian eigendecomposition (Jacobi). See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
1593    PFN_cusolverDnZheevjBatched_bufferSize,
1594    cuDoubleComplex,
1595    f64
1596);
1597
1598macro_rules! dn_syevj_batched {
1599    ($(#[$attr:meta])* $name:ident, $t:ty, $real:ty) => {
1600        $(#[$attr])*
1601        pub type $name = unsafe extern "C" fn(
1602            handle: cusolverDnHandle_t,
1603            jobz: cusolverEigMode_t,
1604            uplo: cublasFillMode_t,
1605            n: c_int,
1606            a: *mut $t,
1607            lda: c_int,
1608            w: *mut $real,
1609            work: *mut $t,
1610            lwork: c_int,
1611            info: *mut c_int,
1612            params: syevjInfo_t,
1613            batch_size: c_int,
1614        ) -> cusolverStatus_t;
1615    };
1616}
1617dn_syevj_batched!(
1618    #[doc = "cuSOLVER: single-precision batched symmetric eigendecomposition (Jacobi). See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
1619    PFN_cusolverDnSsyevjBatched,
1620    f32,
1621    f32
1622);
1623dn_syevj_batched!(
1624    #[doc = "cuSOLVER: double-precision batched symmetric eigendecomposition (Jacobi). See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
1625    PFN_cusolverDnDsyevjBatched,
1626    f64,
1627    f64
1628);
1629dn_syevj_batched!(
1630    #[doc = "cuSOLVER: single-precision complex batched Hermitian eigendecomposition (Jacobi). See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
1631    PFN_cusolverDnCheevjBatched,
1632    cuComplex,
1633    f32
1634);
1635dn_syevj_batched!(
1636    #[doc = "cuSOLVER: double-precision complex batched Hermitian eigendecomposition (Jacobi). See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
1637    PFN_cusolverDnZheevjBatched,
1638    cuDoubleComplex,
1639    f64
1640);
1641
1642macro_rules! dn_gesvdj_batched_bufsize {
1643    ($(#[$attr:meta])* $name:ident, $t:ty, $real:ty) => {
1644        $(#[$attr])*
1645        pub type $name = unsafe extern "C" fn(
1646            handle: cusolverDnHandle_t,
1647            jobz: cusolverEigMode_t,
1648            m: c_int,
1649            n: c_int,
1650            a: *const $t,
1651            lda: c_int,
1652            s: *const $real,
1653            u: *const $t,
1654            ldu: c_int,
1655            v: *const $t,
1656            ldv: c_int,
1657            lwork: *mut c_int,
1658            params: gesvdjInfo_t,
1659            batch_size: c_int,
1660        ) -> cusolverStatus_t;
1661    };
1662}
1663dn_gesvdj_batched_bufsize!(
1664    #[doc = "cuSOLVER: single-precision workspace-size query for batched Jacobi-method singular value decomposition. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
1665    PFN_cusolverDnSgesvdjBatched_bufferSize,
1666    f32,
1667    f32
1668);
1669dn_gesvdj_batched_bufsize!(
1670    #[doc = "cuSOLVER: double-precision workspace-size query for batched Jacobi-method singular value decomposition. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
1671    PFN_cusolverDnDgesvdjBatched_bufferSize,
1672    f64,
1673    f64
1674);
1675dn_gesvdj_batched_bufsize!(
1676    #[doc = "cuSOLVER: single-precision complex workspace-size query for batched Jacobi-method singular value decomposition. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
1677    PFN_cusolverDnCgesvdjBatched_bufferSize,
1678    cuComplex,
1679    f32
1680);
1681dn_gesvdj_batched_bufsize!(
1682    #[doc = "cuSOLVER: double-precision complex workspace-size query for batched Jacobi-method singular value decomposition. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
1683    PFN_cusolverDnZgesvdjBatched_bufferSize,
1684    cuDoubleComplex,
1685    f64
1686);
1687
1688macro_rules! dn_gesvdj_batched {
1689    ($(#[$attr:meta])* $name:ident, $t:ty, $real:ty) => {
1690        $(#[$attr])*
1691        pub type $name = unsafe extern "C" fn(
1692            handle: cusolverDnHandle_t,
1693            jobz: cusolverEigMode_t,
1694            m: c_int,
1695            n: c_int,
1696            a: *mut $t,
1697            lda: c_int,
1698            s: *mut $real,
1699            u: *mut $t,
1700            ldu: c_int,
1701            v: *mut $t,
1702            ldv: c_int,
1703            work: *mut $t,
1704            lwork: c_int,
1705            info: *mut c_int,
1706            params: gesvdjInfo_t,
1707            batch_size: c_int,
1708        ) -> cusolverStatus_t;
1709    };
1710}
1711dn_gesvdj_batched!(
1712    #[doc = "cuSOLVER: single-precision batched Jacobi-method singular value decomposition. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
1713    PFN_cusolverDnSgesvdjBatched,
1714    f32,
1715    f32
1716);
1717dn_gesvdj_batched!(
1718    #[doc = "cuSOLVER: double-precision batched Jacobi-method singular value decomposition. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
1719    PFN_cusolverDnDgesvdjBatched,
1720    f64,
1721    f64
1722);
1723dn_gesvdj_batched!(
1724    #[doc = "cuSOLVER: single-precision complex batched Jacobi-method singular value decomposition. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
1725    PFN_cusolverDnCgesvdjBatched,
1726    cuComplex,
1727    f32
1728);
1729dn_gesvdj_batched!(
1730    #[doc = "cuSOLVER: double-precision complex batched Jacobi-method singular value decomposition. See <https://docs.nvidia.com/cuda/cusolver/index.html>."]
1731    PFN_cusolverDnZgesvdjBatched,
1732    cuDoubleComplex,
1733    f64
1734);
1735
1736// ==========================================================================
1737// cuSOLVERMg — multi-GPU dense solvers (separate library libcusolverMg)
1738// ==========================================================================
1739
1740/// Opaque multi-GPU dense cuSOLVERMg handle.
1741pub type cusolverMgHandle_t = *mut c_void;
1742/// Opaque multi-GPU matrix descriptor.
1743pub type cudaLibMgMatrixDesc_t = *mut c_void;
1744/// Opaque multi-GPU device grid.
1745pub type cudaLibMgGrid_t = *mut c_void;
1746
1747/// cuSOLVER: create a multi-GPU dense cuSOLVERMg handle. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1748pub type PFN_cusolverMgCreate =
1749    unsafe extern "C" fn(handle: *mut cusolverMgHandle_t) -> cusolverStatus_t;
1750/// cuSOLVER: destroy a multi-GPU dense cuSOLVERMg handle. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1751pub type PFN_cusolverMgDestroy =
1752    unsafe extern "C" fn(handle: cusolverMgHandle_t) -> cusolverStatus_t;
1753/// cuSOLVER: select GPU devices for a cuSOLVERMg handle. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1754pub type PFN_cusolverMgDeviceSelect = unsafe extern "C" fn(
1755    handle: cusolverMgHandle_t,
1756    n_devices: c_int,
1757    device_id: *const c_int,
1758) -> cusolverStatus_t;
1759
1760/// cuSOLVER: create a multi-GPU device grid. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1761pub type PFN_cusolverMgCreateDeviceGrid = unsafe extern "C" fn(
1762    grid: *mut cudaLibMgGrid_t,
1763    num_row_devices: i32,
1764    num_col_devices: i32,
1765    device_id: *const i32,
1766    mapping: i32,
1767) -> cusolverStatus_t;
1768
1769/// cuSOLVER: destroy a multi-GPU device grid. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1770pub type PFN_cusolverMgDestroyGrid =
1771    unsafe extern "C" fn(grid: cudaLibMgGrid_t) -> cusolverStatus_t;
1772
1773/// cuSOLVER: create a multi-GPU matrix descriptor. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1774pub type PFN_cusolverMgCreateMatrixDesc = unsafe extern "C" fn(
1775    desc: *mut cudaLibMgMatrixDesc_t,
1776    num_rows: i64,
1777    num_cols: i64,
1778    row_block_size: i64,
1779    col_block_size: i64,
1780    data_type: cudaDataType,
1781    grid: cudaLibMgGrid_t,
1782) -> cusolverStatus_t;
1783
1784/// cuSOLVER: destroy a multi-GPU matrix descriptor. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1785pub type PFN_cusolverMgDestroyMatrixDesc =
1786    unsafe extern "C" fn(desc: cudaLibMgMatrixDesc_t) -> cusolverStatus_t;
1787
1788/// cuSOLVER: multi-GPU workspace-size query for LU factorization with partial pivoting. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1789pub type PFN_cusolverMgGetrf_bufferSize = unsafe extern "C" fn(
1790    handle: cusolverMgHandle_t,
1791    m: c_int,
1792    n: c_int,
1793    array_d_a: *mut *mut c_void,
1794    ia: c_int,
1795    ja: c_int,
1796    desc_a: cudaLibMgMatrixDesc_t,
1797    array_d_ipiv: *mut *mut c_int,
1798    compute_type: cudaDataType,
1799    lwork: *mut i64,
1800) -> cusolverStatus_t;
1801
1802/// cuSOLVER: multi-GPU LU factorization with partial pivoting. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1803pub type PFN_cusolverMgGetrf = unsafe extern "C" fn(
1804    handle: cusolverMgHandle_t,
1805    m: c_int,
1806    n: c_int,
1807    array_d_a: *mut *mut c_void,
1808    ia: c_int,
1809    ja: c_int,
1810    desc_a: cudaLibMgMatrixDesc_t,
1811    array_d_ipiv: *mut *mut c_int,
1812    compute_type: cudaDataType,
1813    array_d_work: *mut *mut c_void,
1814    lwork: i64,
1815    info: *mut c_int,
1816) -> cusolverStatus_t;
1817
1818/// cuSOLVER: multi-GPU workspace-size query for Cholesky factorization. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1819pub type PFN_cusolverMgPotrf_bufferSize = unsafe extern "C" fn(
1820    handle: cusolverMgHandle_t,
1821    uplo: cublasFillMode_t,
1822    n: c_int,
1823    array_d_a: *mut *mut c_void,
1824    ia: c_int,
1825    ja: c_int,
1826    desc_a: cudaLibMgMatrixDesc_t,
1827    compute_type: cudaDataType,
1828    lwork: *mut i64,
1829) -> cusolverStatus_t;
1830
1831/// cuSOLVER: multi-GPU Cholesky factorization. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1832pub type PFN_cusolverMgPotrf = unsafe extern "C" fn(
1833    handle: cusolverMgHandle_t,
1834    uplo: cublasFillMode_t,
1835    n: c_int,
1836    array_d_a: *mut *mut c_void,
1837    ia: c_int,
1838    ja: c_int,
1839    desc_a: cudaLibMgMatrixDesc_t,
1840    compute_type: cudaDataType,
1841    array_d_work: *mut *mut c_void,
1842    lwork: i64,
1843    info: *mut c_int,
1844) -> cusolverStatus_t;
1845
1846/// cuSOLVER: multi-GPU workspace-size query for symmetric eigendecomposition (divide-and-conquer). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1847pub type PFN_cusolverMgSyevd_bufferSize = unsafe extern "C" fn(
1848    handle: cusolverMgHandle_t,
1849    jobz: cusolverEigMode_t,
1850    uplo: cublasFillMode_t,
1851    n: c_int,
1852    array_d_a: *mut *mut c_void,
1853    ia: c_int,
1854    ja: c_int,
1855    desc_a: cudaLibMgMatrixDesc_t,
1856    w: *mut c_void,
1857    data_type_w: cudaDataType,
1858    compute_type: cudaDataType,
1859    lwork: *mut i64,
1860) -> cusolverStatus_t;
1861
1862/// cuSOLVER: multi-GPU symmetric eigendecomposition (divide-and-conquer). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1863pub type PFN_cusolverMgSyevd = unsafe extern "C" fn(
1864    handle: cusolverMgHandle_t,
1865    jobz: cusolverEigMode_t,
1866    uplo: cublasFillMode_t,
1867    n: c_int,
1868    array_d_a: *mut *mut c_void,
1869    ia: c_int,
1870    ja: c_int,
1871    desc_a: cudaLibMgMatrixDesc_t,
1872    w: *mut c_void,
1873    data_type_w: cudaDataType,
1874    compute_type: cudaDataType,
1875    array_d_work: *mut *mut c_void,
1876    lwork: i64,
1877    info: *mut c_int,
1878) -> cusolverStatus_t;
1879
1880// ---- loader --------------------------------------------------------------
1881
1882fn cusolver_candidates() -> Vec<String> {
1883    platform::versioned_library_candidates("cusolver", &["13", "12", "11"])
1884}
1885
1886macro_rules! cusolver_fns {
1887    ($($(#[$m:meta])* $name:ident as $sym:literal : $pfn:ty);* $(;)?) => {
1888        /// Loaded cuSOLVER shared library plus a per-symbol `OnceLock` of function pointers.
1889        pub struct Cusolver {
1890            lib: Library,
1891            $($name: OnceLock<$pfn>,)*
1892        }
1893        impl core::fmt::Debug for Cusolver {
1894            fn fmt(&self, f: &mut core::fmt::Formatter<'_>) -> core::fmt::Result {
1895                f.debug_struct("Cusolver").field("lib", &self.lib).finish_non_exhaustive()
1896            }
1897        }
1898        impl Cusolver {
1899            $(
1900                $(#[$m])*
1901                pub fn $name(&self) -> Result<$pfn, LoaderError> {
1902                    if let Some(&p) = self.$name.get() { return Ok(p); }
1903                    let raw: *mut () = unsafe { self.lib.raw_symbol($sym)? };
1904                    let p: $pfn = unsafe { core::mem::transmute_copy::<*mut (), $pfn>(&raw) };
1905                    let _ = self.$name.set(p);
1906                    Ok(p)
1907                }
1908            )*
1909            fn empty(lib: Library) -> Self {
1910                Self { lib, $($name: OnceLock::new(),)* }
1911            }
1912        }
1913    };
1914}
1915
1916cusolver_fns! {
1917    // Dn handle
1918    /// cuSOLVER: create a dense cuSOLVER handle. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1919    cusolver_dn_create as "cusolverDnCreate": PFN_cusolverDnCreate;
1920    /// cuSOLVER: destroy a dense cuSOLVER handle. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1921    cusolver_dn_destroy as "cusolverDnDestroy": PFN_cusolverDnDestroy;
1922    /// cuSOLVER: bind a CUDA stream to a dense cuSOLVER handle. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1923    cusolver_dn_set_stream as "cusolverDnSetStream": PFN_cusolverDnSetStream;
1924    /// cuSOLVER: query the CUDA stream bound to a dense cuSOLVER handle. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1925    cusolver_dn_get_stream as "cusolverDnGetStream": PFN_cusolverDnGetStream;
1926    /// cuSOLVER: return the cuSOLVER library version. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1927    cusolver_get_version as "cusolverGetVersion": PFN_cusolverGetVersion;
1928    // LU (getrf/getrs) S/D/C/Z
1929    /// cuSOLVER: single-precision workspace-size query for LU factorization with partial pivoting. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1930    cusolver_dn_sgetrf_buffer_size as "cusolverDnSgetrf_bufferSize": PFN_cusolverDnSgetrf_bufferSize;
1931    /// cuSOLVER: double-precision workspace-size query for LU factorization with partial pivoting. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1932    cusolver_dn_dgetrf_buffer_size as "cusolverDnDgetrf_bufferSize": PFN_cusolverDnDgetrf_bufferSize;
1933    /// cuSOLVER: single-precision complex workspace-size query for LU factorization with partial pivoting. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1934    cusolver_dn_cgetrf_buffer_size as "cusolverDnCgetrf_bufferSize": PFN_cusolverDnCgetrf_bufferSize;
1935    /// cuSOLVER: double-precision complex workspace-size query for LU factorization with partial pivoting. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1936    cusolver_dn_zgetrf_buffer_size as "cusolverDnZgetrf_bufferSize": PFN_cusolverDnZgetrf_bufferSize;
1937    /// cuSOLVER: single-precision LU factorization with partial pivoting. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1938    cusolver_dn_sgetrf as "cusolverDnSgetrf": PFN_cusolverDnSgetrf;
1939    /// cuSOLVER: double-precision LU factorization with partial pivoting. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1940    cusolver_dn_dgetrf as "cusolverDnDgetrf": PFN_cusolverDnDgetrf;
1941    /// cuSOLVER: single-precision complex LU factorization with partial pivoting. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1942    cusolver_dn_cgetrf as "cusolverDnCgetrf": PFN_cusolverDnCgetrf;
1943    /// cuSOLVER: double-precision complex LU factorization with partial pivoting. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1944    cusolver_dn_zgetrf as "cusolverDnZgetrf": PFN_cusolverDnZgetrf;
1945    /// cuSOLVER: single-precision solve linear system using LU factors. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1946    cusolver_dn_sgetrs as "cusolverDnSgetrs": PFN_cusolverDnSgetrs;
1947    /// cuSOLVER: double-precision solve linear system using LU factors. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1948    cusolver_dn_dgetrs as "cusolverDnDgetrs": PFN_cusolverDnDgetrs;
1949    /// cuSOLVER: single-precision complex solve linear system using LU factors. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1950    cusolver_dn_cgetrs as "cusolverDnCgetrs": PFN_cusolverDnCgetrs;
1951    /// cuSOLVER: double-precision complex solve linear system using LU factors. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1952    cusolver_dn_zgetrs as "cusolverDnZgetrs": PFN_cusolverDnZgetrs;
1953    // QR (geqrf) S/D/C/Z
1954    /// cuSOLVER: single-precision workspace-size query for QR factorization (Householder). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1955    cusolver_dn_sgeqrf_buffer_size as "cusolverDnSgeqrf_bufferSize": PFN_cusolverDnSgeqrf_bufferSize;
1956    /// cuSOLVER: double-precision workspace-size query for QR factorization (Householder). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1957    cusolver_dn_dgeqrf_buffer_size as "cusolverDnDgeqrf_bufferSize": PFN_cusolverDnDgeqrf_bufferSize;
1958    /// cuSOLVER: single-precision complex workspace-size query for QR factorization (Householder). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1959    cusolver_dn_cgeqrf_buffer_size as "cusolverDnCgeqrf_bufferSize": PFN_cusolverDnCgeqrf_bufferSize;
1960    /// cuSOLVER: double-precision complex workspace-size query for QR factorization (Householder). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1961    cusolver_dn_zgeqrf_buffer_size as "cusolverDnZgeqrf_bufferSize": PFN_cusolverDnZgeqrf_bufferSize;
1962    /// cuSOLVER: single-precision QR factorization (Householder). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1963    cusolver_dn_sgeqrf as "cusolverDnSgeqrf": PFN_cusolverDnSgeqrf;
1964    /// cuSOLVER: double-precision QR factorization (Householder). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1965    cusolver_dn_dgeqrf as "cusolverDnDgeqrf": PFN_cusolverDnDgeqrf;
1966    /// cuSOLVER: single-precision complex QR factorization (Householder). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1967    cusolver_dn_cgeqrf as "cusolverDnCgeqrf": PFN_cusolverDnCgeqrf;
1968    /// cuSOLVER: double-precision complex QR factorization (Householder). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1969    cusolver_dn_zgeqrf as "cusolverDnZgeqrf": PFN_cusolverDnZgeqrf;
1970    // Cholesky (potrf/potrs) S/D/C/Z
1971    /// cuSOLVER: single-precision workspace-size query for Cholesky factorization. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1972    cusolver_dn_spotrf_buffer_size as "cusolverDnSpotrf_bufferSize": PFN_cusolverDnSpotrf_bufferSize;
1973    /// cuSOLVER: double-precision workspace-size query for Cholesky factorization. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1974    cusolver_dn_dpotrf_buffer_size as "cusolverDnDpotrf_bufferSize": PFN_cusolverDnDpotrf_bufferSize;
1975    /// cuSOLVER: single-precision complex workspace-size query for Cholesky factorization. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1976    cusolver_dn_cpotrf_buffer_size as "cusolverDnCpotrf_bufferSize": PFN_cusolverDnCpotrf_bufferSize;
1977    /// cuSOLVER: double-precision complex workspace-size query for Cholesky factorization. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1978    cusolver_dn_zpotrf_buffer_size as "cusolverDnZpotrf_bufferSize": PFN_cusolverDnZpotrf_bufferSize;
1979    /// cuSOLVER: single-precision Cholesky factorization. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1980    cusolver_dn_spotrf as "cusolverDnSpotrf": PFN_cusolverDnSpotrf;
1981    /// cuSOLVER: double-precision Cholesky factorization. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1982    cusolver_dn_dpotrf as "cusolverDnDpotrf": PFN_cusolverDnDpotrf;
1983    /// cuSOLVER: single-precision complex Cholesky factorization. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1984    cusolver_dn_cpotrf as "cusolverDnCpotrf": PFN_cusolverDnCpotrf;
1985    /// cuSOLVER: double-precision complex Cholesky factorization. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1986    cusolver_dn_zpotrf as "cusolverDnZpotrf": PFN_cusolverDnZpotrf;
1987    /// cuSOLVER: single-precision solve linear system using Cholesky factors. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1988    cusolver_dn_spotrs as "cusolverDnSpotrs": PFN_cusolverDnSpotrs;
1989    /// cuSOLVER: double-precision solve linear system using Cholesky factors. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1990    cusolver_dn_dpotrs as "cusolverDnDpotrs": PFN_cusolverDnDpotrs;
1991    /// cuSOLVER: single-precision complex solve linear system using Cholesky factors. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1992    cusolver_dn_cpotrs as "cusolverDnCpotrs": PFN_cusolverDnCpotrs;
1993    /// cuSOLVER: double-precision complex solve linear system using Cholesky factors. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1994    cusolver_dn_zpotrs as "cusolverDnZpotrs": PFN_cusolverDnZpotrs;
1995    // SVD S/D/C/Z
1996    /// cuSOLVER: single-precision workspace-size query for singular value decomposition. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1997    cusolver_dn_sgesvd_buffer_size as "cusolverDnSgesvd_bufferSize": PFN_cusolverDnSgesvd_bufferSize;
1998    /// cuSOLVER: double-precision workspace-size query for singular value decomposition. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
1999    cusolver_dn_dgesvd_buffer_size as "cusolverDnDgesvd_bufferSize": PFN_cusolverDnDgesvd_bufferSize;
2000    /// cuSOLVER: single-precision complex workspace-size query for singular value decomposition. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2001    cusolver_dn_cgesvd_buffer_size as "cusolverDnCgesvd_bufferSize": PFN_cusolverDnCgesvd_bufferSize;
2002    /// cuSOLVER: double-precision complex workspace-size query for singular value decomposition. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2003    cusolver_dn_zgesvd_buffer_size as "cusolverDnZgesvd_bufferSize": PFN_cusolverDnZgesvd_bufferSize;
2004    /// cuSOLVER: single-precision singular value decomposition. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2005    cusolver_dn_sgesvd as "cusolverDnSgesvd": PFN_cusolverDnSgesvd;
2006    /// cuSOLVER: double-precision singular value decomposition. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2007    cusolver_dn_dgesvd as "cusolverDnDgesvd": PFN_cusolverDnDgesvd;
2008    /// cuSOLVER: single-precision complex singular value decomposition. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2009    cusolver_dn_cgesvd as "cusolverDnCgesvd": PFN_cusolverDnCgesvd;
2010    /// cuSOLVER: double-precision complex singular value decomposition. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2011    cusolver_dn_zgesvd as "cusolverDnZgesvd": PFN_cusolverDnZgesvd;
2012    // syevd / heevd
2013    /// cuSOLVER: single-precision workspace-size query for symmetric eigendecomposition (divide-and-conquer). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2014    cusolver_dn_ssyevd_buffer_size as "cusolverDnSsyevd_bufferSize": PFN_cusolverDnSsyevd_bufferSize;
2015    /// cuSOLVER: double-precision workspace-size query for symmetric eigendecomposition (divide-and-conquer). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2016    cusolver_dn_dsyevd_buffer_size as "cusolverDnDsyevd_bufferSize": PFN_cusolverDnDsyevd_bufferSize;
2017    /// cuSOLVER: single-precision complex workspace-size query for Hermitian eigendecomposition (divide-and-conquer). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2018    cusolver_dn_cheevd_buffer_size as "cusolverDnCheevd_bufferSize": PFN_cusolverDnCheevd_bufferSize;
2019    /// cuSOLVER: double-precision complex workspace-size query for Hermitian eigendecomposition (divide-and-conquer). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2020    cusolver_dn_zheevd_buffer_size as "cusolverDnZheevd_bufferSize": PFN_cusolverDnZheevd_bufferSize;
2021    /// cuSOLVER: single-precision symmetric eigendecomposition (divide-and-conquer). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2022    cusolver_dn_ssyevd as "cusolverDnSsyevd": PFN_cusolverDnSsyevd;
2023    /// cuSOLVER: double-precision symmetric eigendecomposition (divide-and-conquer). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2024    cusolver_dn_dsyevd as "cusolverDnDsyevd": PFN_cusolverDnDsyevd;
2025    /// cuSOLVER: single-precision complex Hermitian eigendecomposition (divide-and-conquer). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2026    cusolver_dn_cheevd as "cusolverDnCheevd": PFN_cusolverDnCheevd;
2027    /// cuSOLVER: double-precision complex Hermitian eigendecomposition (divide-and-conquer). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2028    cusolver_dn_zheevd as "cusolverDnZheevd": PFN_cusolverDnZheevd;
2029    // Generic 64-bit X… API
2030    /// cuSOLVER: create a generic-API parameter object. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2031    cusolver_dn_create_params as "cusolverDnCreateParams": PFN_cusolverDnCreateParams;
2032    /// cuSOLVER: destroy a generic-API parameter object. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2033    cusolver_dn_destroy_params as "cusolverDnDestroyParams": PFN_cusolverDnDestroyParams;
2034    /// cuSOLVER: generic-API workspace-size query for LU factorization with partial pivoting (dtype configurable via `cudaDataType`). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2035    cusolver_dn_xgetrf_buffer_size as "cusolverDnXgetrf_bufferSize": PFN_cusolverDnXgetrf_bufferSize;
2036    /// cuSOLVER: generic-API LU factorization with partial pivoting (dtype configurable via `cudaDataType`). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2037    cusolver_dn_xgetrf as "cusolverDnXgetrf": PFN_cusolverDnXgetrf;
2038    /// cuSOLVER: generic-API solve linear system using LU factors (dtype configurable via `cudaDataType`). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2039    cusolver_dn_xgetrs as "cusolverDnXgetrs": PFN_cusolverDnXgetrs;
2040    /// cuSOLVER: generic-API workspace-size query for QR factorization (Householder) (dtype configurable via `cudaDataType`). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2041    cusolver_dn_xgeqrf_buffer_size as "cusolverDnXgeqrf_bufferSize": PFN_cusolverDnXgeqrf_bufferSize;
2042    /// cuSOLVER: generic-API QR factorization (Householder) (dtype configurable via `cudaDataType`). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2043    cusolver_dn_xgeqrf as "cusolverDnXgeqrf": PFN_cusolverDnXgeqrf;
2044    /// cuSOLVER: generic-API workspace-size query for Cholesky factorization (dtype configurable via `cudaDataType`). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2045    cusolver_dn_xpotrf_buffer_size as "cusolverDnXpotrf_bufferSize": PFN_cusolverDnXpotrf_bufferSize;
2046    /// cuSOLVER: generic-API Cholesky factorization (dtype configurable via `cudaDataType`). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2047    cusolver_dn_xpotrf as "cusolverDnXpotrf": PFN_cusolverDnXpotrf;
2048    /// cuSOLVER: generic-API solve linear system using Cholesky factors (dtype configurable via `cudaDataType`). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2049    cusolver_dn_xpotrs as "cusolverDnXpotrs": PFN_cusolverDnXpotrs;
2050    /// cuSOLVER: generic-API workspace-size query for symmetric eigendecomposition (divide-and-conquer) (dtype configurable via `cudaDataType`). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2051    cusolver_dn_xsyevd_buffer_size as "cusolverDnXsyevd_bufferSize": PFN_cusolverDnXsyevd_bufferSize;
2052    /// cuSOLVER: generic-API symmetric eigendecomposition (divide-and-conquer) (dtype configurable via `cudaDataType`). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2053    cusolver_dn_xsyevd as "cusolverDnXsyevd": PFN_cusolverDnXsyevd;
2054    // Jacobi eigen
2055    /// cuSOLVER: create a Jacobi-eigendecomposition control object. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2056    cusolver_dn_create_syevj_info as "cusolverDnCreateSyevjInfo": PFN_cusolverDnCreateSyevjInfo;
2057    /// cuSOLVER: destroy a Jacobi-eigendecomposition control object. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2058    cusolver_dn_destroy_syevj_info as "cusolverDnDestroySyevjInfo": PFN_cusolverDnDestroySyevjInfo;
2059    /// cuSOLVER: set convergence tolerance on a Jacobi-eigendecomposition control object. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2060    cusolver_dn_xsyevj_set_tolerance as "cusolverDnXsyevjSetTolerance": PFN_cusolverDnXsyevjSetTolerance;
2061    /// cuSOLVER: set the maximum number of sweeps on a Jacobi-eigendecomposition control object. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2062    cusolver_dn_xsyevj_set_max_sweeps as "cusolverDnXsyevjSetMaxSweeps": PFN_cusolverDnXsyevjSetMaxSweeps;
2063    /// cuSOLVER: single-precision workspace-size query for symmetric eigendecomposition (Jacobi). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2064    cusolver_dn_ssyevj_buffer_size as "cusolverDnSsyevj_bufferSize": PFN_cusolverDnSsyevj_bufferSize;
2065    /// cuSOLVER: double-precision workspace-size query for symmetric eigendecomposition (Jacobi). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2066    cusolver_dn_dsyevj_buffer_size as "cusolverDnDsyevj_bufferSize": PFN_cusolverDnDsyevj_bufferSize;
2067    /// cuSOLVER: single-precision complex workspace-size query for Hermitian eigendecomposition (Jacobi). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2068    cusolver_dn_cheevj_buffer_size as "cusolverDnCheevj_bufferSize": PFN_cusolverDnCheevj_bufferSize;
2069    /// cuSOLVER: double-precision complex workspace-size query for Hermitian eigendecomposition (Jacobi). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2070    cusolver_dn_zheevj_buffer_size as "cusolverDnZheevj_bufferSize": PFN_cusolverDnZheevj_bufferSize;
2071    /// cuSOLVER: single-precision symmetric eigendecomposition (Jacobi). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2072    cusolver_dn_ssyevj as "cusolverDnSsyevj": PFN_cusolverDnSsyevj;
2073    /// cuSOLVER: double-precision symmetric eigendecomposition (Jacobi). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2074    cusolver_dn_dsyevj as "cusolverDnDsyevj": PFN_cusolverDnDsyevj;
2075    /// cuSOLVER: single-precision complex Hermitian eigendecomposition (Jacobi). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2076    cusolver_dn_cheevj as "cusolverDnCheevj": PFN_cusolverDnCheevj;
2077    /// cuSOLVER: double-precision complex Hermitian eigendecomposition (Jacobi). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2078    cusolver_dn_zheevj as "cusolverDnZheevj": PFN_cusolverDnZheevj;
2079    // Jacobi SVD
2080    /// cuSOLVER: create a Jacobi-SVD control object. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2081    cusolver_dn_create_gesvdj_info as "cusolverDnCreateGesvdjInfo": PFN_cusolverDnCreateGesvdjInfo;
2082    /// cuSOLVER: destroy a Jacobi-SVD control object. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2083    cusolver_dn_destroy_gesvdj_info as "cusolverDnDestroyGesvdjInfo": PFN_cusolverDnDestroyGesvdjInfo;
2084    /// cuSOLVER: single-precision workspace-size query for Jacobi-method singular value decomposition. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2085    cusolver_dn_sgesvdj_buffer_size as "cusolverDnSgesvdj_bufferSize": PFN_cusolverDnSgesvdj_bufferSize;
2086    /// cuSOLVER: double-precision workspace-size query for Jacobi-method singular value decomposition. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2087    cusolver_dn_dgesvdj_buffer_size as "cusolverDnDgesvdj_bufferSize": PFN_cusolverDnDgesvdj_bufferSize;
2088    /// cuSOLVER: single-precision complex workspace-size query for Jacobi-method singular value decomposition. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2089    cusolver_dn_cgesvdj_buffer_size as "cusolverDnCgesvdj_bufferSize": PFN_cusolverDnCgesvdj_bufferSize;
2090    /// cuSOLVER: double-precision complex workspace-size query for Jacobi-method singular value decomposition. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2091    cusolver_dn_zgesvdj_buffer_size as "cusolverDnZgesvdj_bufferSize": PFN_cusolverDnZgesvdj_bufferSize;
2092    /// cuSOLVER: single-precision Jacobi-method singular value decomposition. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2093    cusolver_dn_sgesvdj as "cusolverDnSgesvdj": PFN_cusolverDnSgesvdj;
2094    /// cuSOLVER: double-precision Jacobi-method singular value decomposition. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2095    cusolver_dn_dgesvdj as "cusolverDnDgesvdj": PFN_cusolverDnDgesvdj;
2096    /// cuSOLVER: single-precision complex Jacobi-method singular value decomposition. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2097    cusolver_dn_cgesvdj as "cusolverDnCgesvdj": PFN_cusolverDnCgesvdj;
2098    /// cuSOLVER: double-precision complex Jacobi-method singular value decomposition. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2099    cusolver_dn_zgesvdj as "cusolverDnZgesvdj": PFN_cusolverDnZgesvdj;
2100    // orgqr / ormqr (apply/generate Q from QR)
2101    /// cuSOLVER: single-precision workspace-size query for generate the explicit Q from a QR factorization. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2102    cusolver_dn_sorgqr_buffer_size as "cusolverDnSorgqr_bufferSize": PFN_cusolverDnSorgqr_bufferSize;
2103    /// cuSOLVER: double-precision workspace-size query for generate the explicit Q from a QR factorization. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2104    cusolver_dn_dorgqr_buffer_size as "cusolverDnDorgqr_bufferSize": PFN_cusolverDnDorgqr_bufferSize;
2105    /// cuSOLVER: single-precision complex workspace-size query for generate the explicit unitary Q from a QR factorization. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2106    cusolver_dn_cungqr_buffer_size as "cusolverDnCungqr_bufferSize": PFN_cusolverDnCungqr_bufferSize;
2107    /// cuSOLVER: double-precision complex workspace-size query for generate the explicit unitary Q from a QR factorization. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2108    cusolver_dn_zungqr_buffer_size as "cusolverDnZungqr_bufferSize": PFN_cusolverDnZungqr_bufferSize;
2109    /// cuSOLVER: single-precision generate the explicit Q from a QR factorization. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2110    cusolver_dn_sorgqr as "cusolverDnSorgqr": PFN_cusolverDnSorgqr;
2111    /// cuSOLVER: double-precision generate the explicit Q from a QR factorization. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2112    cusolver_dn_dorgqr as "cusolverDnDorgqr": PFN_cusolverDnDorgqr;
2113    /// cuSOLVER: single-precision complex generate the explicit unitary Q from a QR factorization. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2114    cusolver_dn_cungqr as "cusolverDnCungqr": PFN_cusolverDnCungqr;
2115    /// cuSOLVER: double-precision complex generate the explicit unitary Q from a QR factorization. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2116    cusolver_dn_zungqr as "cusolverDnZungqr": PFN_cusolverDnZungqr;
2117    /// cuSOLVER: single-precision workspace-size query for apply Q (or Q^T) from a QR factorization to a matrix. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2118    cusolver_dn_sormqr_buffer_size as "cusolverDnSormqr_bufferSize": PFN_cusolverDnSormqr_bufferSize;
2119    /// cuSOLVER: double-precision workspace-size query for apply Q (or Q^T) from a QR factorization to a matrix. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2120    cusolver_dn_dormqr_buffer_size as "cusolverDnDormqr_bufferSize": PFN_cusolverDnDormqr_bufferSize;
2121    /// cuSOLVER: single-precision complex workspace-size query for apply Q (or Q^H) from a QR factorization to a matrix. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2122    cusolver_dn_cunmqr_buffer_size as "cusolverDnCunmqr_bufferSize": PFN_cusolverDnCunmqr_bufferSize;
2123    /// cuSOLVER: double-precision complex workspace-size query for apply Q (or Q^H) from a QR factorization to a matrix. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2124    cusolver_dn_zunmqr_buffer_size as "cusolverDnZunmqr_bufferSize": PFN_cusolverDnZunmqr_bufferSize;
2125    /// cuSOLVER: single-precision apply Q (or Q^T) from a QR factorization to a matrix. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2126    cusolver_dn_sormqr as "cusolverDnSormqr": PFN_cusolverDnSormqr;
2127    /// cuSOLVER: double-precision apply Q (or Q^T) from a QR factorization to a matrix. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2128    cusolver_dn_dormqr as "cusolverDnDormqr": PFN_cusolverDnDormqr;
2129    /// cuSOLVER: single-precision complex apply Q (or Q^H) from a QR factorization to a matrix. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2130    cusolver_dn_cunmqr as "cusolverDnCunmqr": PFN_cusolverDnCunmqr;
2131    /// cuSOLVER: double-precision complex apply Q (or Q^H) from a QR factorization to a matrix. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2132    cusolver_dn_zunmqr as "cusolverDnZunmqr": PFN_cusolverDnZunmqr;
2133    // Sparse
2134    /// cuSOLVER: create a sparse cuSOLVER handle. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2135    cusolver_sp_create as "cusolverSpCreate": PFN_cusolverSpCreate;
2136    /// cuSOLVER: destroy a sparse cuSOLVER handle. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2137    cusolver_sp_destroy as "cusolverSpDestroy": PFN_cusolverSpDestroy;
2138    /// cuSOLVER: bind a CUDA stream to a sparse cuSOLVER handle. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2139    cusolver_sp_set_stream as "cusolverSpSetStream": PFN_cusolverSpSetStream;
2140    /// cuSOLVER: single-precision sparse linear solve via Cholesky on a CSR matrix. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2141    cusolver_sp_scsrlsvchol as "cusolverSpScsrlsvchol": PFN_cusolverSpScsrlsvchol;
2142    /// cuSOLVER: double-precision sparse linear solve via Cholesky on a CSR matrix. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2143    cusolver_sp_dcsrlsvchol as "cusolverSpDcsrlsvchol": PFN_cusolverSpDcsrlsvchol;
2144    /// cuSOLVER: single-precision sparse linear solve via QR on a CSR matrix. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2145    cusolver_sp_scsrlsvqr as "cusolverSpScsrlsvqr": PFN_cusolverSpScsrlsvqr;
2146    /// cuSOLVER: double-precision sparse linear solve via QR on a CSR matrix. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2147    cusolver_sp_dcsrlsvqr as "cusolverSpDcsrlsvqr": PFN_cusolverSpDcsrlsvqr;
2148    // Refactor
2149    /// cuSOLVER: create a sparse-refactor cuSOLVER handle. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2150    cusolver_rf_create as "cusolverRfCreate": PFN_cusolverRfCreate;
2151    /// cuSOLVER: destroy a sparse-refactor cuSOLVER handle. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2152    cusolver_rf_destroy as "cusolverRfDestroy": PFN_cusolverRfDestroy;
2153    /// cuSOLVER: supply sparse triangular factors and pivot vectors to the refactor handle. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2154    cusolver_rf_setup_device as "cusolverRfSetupDevice": PFN_cusolverRfSetupDevice;
2155    /// cuSOLVER: analyze the sparsity pattern of the refactor handle. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2156    cusolver_rf_analyze as "cusolverRfAnalyze": PFN_cusolverRfAnalyze;
2157    /// cuSOLVER: refactor with new numerical values reusing the analyzed pattern. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2158    cusolver_rf_refactor as "cusolverRfRefactor": PFN_cusolverRfRefactor;
2159    /// cuSOLVER: solve linear systems using the refactor handle. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2160    cusolver_rf_solve as "cusolverRfSolve": PFN_cusolverRfSolve;
2161    // Least squares (gels) S/D/C/Z
2162    /// cuSOLVER: single-precision workspace-size query for least-squares solver (A*X = B). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2163    cusolver_dn_ssgels_buffer_size as "cusolverDnSSgels_bufferSize": PFN_cusolverDnSSgels_bufferSize;
2164    /// cuSOLVER: double-precision workspace-size query for least-squares solver (A*X = B). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2165    cusolver_dn_ddgels_buffer_size as "cusolverDnDDgels_bufferSize": PFN_cusolverDnDDgels_bufferSize;
2166    /// cuSOLVER: single-precision complex workspace-size query for least-squares solver (A*X = B). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2167    cusolver_dn_ccgels_buffer_size as "cusolverDnCCgels_bufferSize": PFN_cusolverDnCCgels_bufferSize;
2168    /// cuSOLVER: double-precision complex workspace-size query for least-squares solver (A*X = B). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2169    cusolver_dn_zzgels_buffer_size as "cusolverDnZZgels_bufferSize": PFN_cusolverDnZZgels_bufferSize;
2170    /// cuSOLVER: single-precision least-squares solver (A*X = B). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2171    cusolver_dn_ssgels as "cusolverDnSSgels": PFN_cusolverDnSSgels;
2172    /// cuSOLVER: double-precision least-squares solver (A*X = B). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2173    cusolver_dn_ddgels as "cusolverDnDDgels": PFN_cusolverDnDDgels;
2174    /// cuSOLVER: single-precision complex least-squares solver (A*X = B). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2175    cusolver_dn_ccgels as "cusolverDnCCgels": PFN_cusolverDnCCgels;
2176    /// cuSOLVER: double-precision complex least-squares solver (A*X = B). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2177    cusolver_dn_zzgels as "cusolverDnZZgels": PFN_cusolverDnZZgels;
2178    // potri (inverse from Cholesky) S/D/C/Z
2179    /// cuSOLVER: single-precision workspace-size query for matrix inverse from Cholesky factors. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2180    cusolver_dn_spotri_buffer_size as "cusolverDnSpotri_bufferSize": PFN_cusolverDnSpotri_bufferSize;
2181    /// cuSOLVER: double-precision workspace-size query for matrix inverse from Cholesky factors. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2182    cusolver_dn_dpotri_buffer_size as "cusolverDnDpotri_bufferSize": PFN_cusolverDnDpotri_bufferSize;
2183    /// cuSOLVER: single-precision complex workspace-size query for matrix inverse from Cholesky factors. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2184    cusolver_dn_cpotri_buffer_size as "cusolverDnCpotri_bufferSize": PFN_cusolverDnCpotri_bufferSize;
2185    /// cuSOLVER: double-precision complex workspace-size query for matrix inverse from Cholesky factors. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2186    cusolver_dn_zpotri_buffer_size as "cusolverDnZpotri_bufferSize": PFN_cusolverDnZpotri_bufferSize;
2187    /// cuSOLVER: single-precision matrix inverse from Cholesky factors. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2188    cusolver_dn_spotri as "cusolverDnSpotri": PFN_cusolverDnSpotri;
2189    /// cuSOLVER: double-precision matrix inverse from Cholesky factors. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2190    cusolver_dn_dpotri as "cusolverDnDpotri": PFN_cusolverDnDpotri;
2191    /// cuSOLVER: single-precision complex matrix inverse from Cholesky factors. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2192    cusolver_dn_cpotri as "cusolverDnCpotri": PFN_cusolverDnCpotri;
2193    /// cuSOLVER: double-precision complex matrix inverse from Cholesky factors. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2194    cusolver_dn_zpotri as "cusolverDnZpotri": PFN_cusolverDnZpotri;
2195    // Batched Jacobi eigen
2196    /// cuSOLVER: single-precision workspace-size query for batched symmetric eigendecomposition (Jacobi). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2197    cusolver_dn_ssyevj_batched_buffer_size as "cusolverDnSsyevjBatched_bufferSize": PFN_cusolverDnSsyevjBatched_bufferSize;
2198    /// cuSOLVER: double-precision workspace-size query for batched symmetric eigendecomposition (Jacobi). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2199    cusolver_dn_dsyevj_batched_buffer_size as "cusolverDnDsyevjBatched_bufferSize": PFN_cusolverDnDsyevjBatched_bufferSize;
2200    /// cuSOLVER: single-precision complex workspace-size query for batched Hermitian eigendecomposition (Jacobi). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2201    cusolver_dn_cheevj_batched_buffer_size as "cusolverDnCheevjBatched_bufferSize": PFN_cusolverDnCheevjBatched_bufferSize;
2202    /// cuSOLVER: double-precision complex workspace-size query for batched Hermitian eigendecomposition (Jacobi). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2203    cusolver_dn_zheevj_batched_buffer_size as "cusolverDnZheevjBatched_bufferSize": PFN_cusolverDnZheevjBatched_bufferSize;
2204    /// cuSOLVER: single-precision batched symmetric eigendecomposition (Jacobi). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2205    cusolver_dn_ssyevj_batched as "cusolverDnSsyevjBatched": PFN_cusolverDnSsyevjBatched;
2206    /// cuSOLVER: double-precision batched symmetric eigendecomposition (Jacobi). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2207    cusolver_dn_dsyevj_batched as "cusolverDnDsyevjBatched": PFN_cusolverDnDsyevjBatched;
2208    /// cuSOLVER: single-precision complex batched Hermitian eigendecomposition (Jacobi). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2209    cusolver_dn_cheevj_batched as "cusolverDnCheevjBatched": PFN_cusolverDnCheevjBatched;
2210    /// cuSOLVER: double-precision complex batched Hermitian eigendecomposition (Jacobi). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2211    cusolver_dn_zheevj_batched as "cusolverDnZheevjBatched": PFN_cusolverDnZheevjBatched;
2212    // Batched Jacobi SVD
2213    /// cuSOLVER: single-precision workspace-size query for batched Jacobi-method singular value decomposition. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2214    cusolver_dn_sgesvdj_batched_buffer_size as "cusolverDnSgesvdjBatched_bufferSize": PFN_cusolverDnSgesvdjBatched_bufferSize;
2215    /// cuSOLVER: double-precision workspace-size query for batched Jacobi-method singular value decomposition. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2216    cusolver_dn_dgesvdj_batched_buffer_size as "cusolverDnDgesvdjBatched_bufferSize": PFN_cusolverDnDgesvdjBatched_bufferSize;
2217    /// cuSOLVER: single-precision complex workspace-size query for batched Jacobi-method singular value decomposition. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2218    cusolver_dn_cgesvdj_batched_buffer_size as "cusolverDnCgesvdjBatched_bufferSize": PFN_cusolverDnCgesvdjBatched_bufferSize;
2219    /// cuSOLVER: double-precision complex workspace-size query for batched Jacobi-method singular value decomposition. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2220    cusolver_dn_zgesvdj_batched_buffer_size as "cusolverDnZgesvdjBatched_bufferSize": PFN_cusolverDnZgesvdjBatched_bufferSize;
2221    /// cuSOLVER: single-precision batched Jacobi-method singular value decomposition. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2222    cusolver_dn_sgesvdj_batched as "cusolverDnSgesvdjBatched": PFN_cusolverDnSgesvdjBatched;
2223    /// cuSOLVER: double-precision batched Jacobi-method singular value decomposition. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2224    cusolver_dn_dgesvdj_batched as "cusolverDnDgesvdjBatched": PFN_cusolverDnDgesvdjBatched;
2225    /// cuSOLVER: single-precision complex batched Jacobi-method singular value decomposition. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2226    cusolver_dn_cgesvdj_batched as "cusolverDnCgesvdjBatched": PFN_cusolverDnCgesvdjBatched;
2227    /// cuSOLVER: double-precision complex batched Jacobi-method singular value decomposition. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2228    cusolver_dn_zgesvdj_batched as "cusolverDnZgesvdjBatched": PFN_cusolverDnZgesvdjBatched;
2229}
2230
2231/// Lazy-load the cuSOLVER dense / sparse / refactor shared library and return its function-pointer table.
2232pub fn cusolver() -> Result<&'static Cusolver, LoaderError> {
2233    static CUSOLVER: OnceLock<Cusolver> = OnceLock::new();
2234    if let Some(c) = CUSOLVER.get() {
2235        return Ok(c);
2236    }
2237    let candidates: Vec<&'static str> = cusolver_candidates()
2238        .into_iter()
2239        .map(|s| Box::leak(s.into_boxed_str()) as &'static str)
2240        .collect();
2241    let candidates_leaked: &'static [&'static str] = Box::leak(candidates.into_boxed_slice());
2242    let lib = Library::open("cusolver", candidates_leaked)?;
2243    let c = Cusolver::empty(lib);
2244    let _ = CUSOLVER.set(c);
2245    Ok(CUSOLVER.get().expect("OnceLock set or lost race"))
2246}
2247
2248// ==========================================================================
2249// cuSOLVERMg — multi-GPU solver, ships in libcusolverMg (separate library)
2250// ==========================================================================
2251
2252fn cusolver_mg_candidates() -> Vec<String> {
2253    platform::versioned_library_candidates("cusolverMg", &["13", "12", "11"])
2254}
2255
2256macro_rules! cusolver_mg_fns {
2257    ($($(#[$m:meta])* $name:ident as $sym:literal : $pfn:ty);* $(;)?) => {
2258        /// Loaded cuSOLVERMg shared library plus a per-symbol `OnceLock` of function pointers.
2259        pub struct CusolverMg {
2260            lib: Library,
2261            $($name: OnceLock<$pfn>,)*
2262        }
2263        impl core::fmt::Debug for CusolverMg {
2264            fn fmt(&self, f: &mut core::fmt::Formatter<'_>) -> core::fmt::Result {
2265                f.debug_struct("CusolverMg").field("lib", &self.lib).finish_non_exhaustive()
2266            }
2267        }
2268        impl CusolverMg {
2269            $(
2270                $(#[$m])*
2271                pub fn $name(&self) -> Result<$pfn, LoaderError> {
2272                    if let Some(&p) = self.$name.get() { return Ok(p); }
2273                    let raw: *mut () = unsafe { self.lib.raw_symbol($sym)? };
2274                    let p: $pfn = unsafe { core::mem::transmute_copy::<*mut (), $pfn>(&raw) };
2275                    let _ = self.$name.set(p);
2276                    Ok(p)
2277                }
2278            )*
2279            fn empty(lib: Library) -> Self {
2280                Self { lib, $($name: OnceLock::new(),)* }
2281            }
2282        }
2283    };
2284}
2285
2286cusolver_mg_fns! {
2287    /// cuSOLVER: create a multi-GPU dense cuSOLVERMg handle. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2288    cusolver_mg_create as "cusolverMgCreate": PFN_cusolverMgCreate;
2289    /// cuSOLVER: destroy a multi-GPU dense cuSOLVERMg handle. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2290    cusolver_mg_destroy as "cusolverMgDestroy": PFN_cusolverMgDestroy;
2291    /// cuSOLVER: select GPU devices for a cuSOLVERMg handle. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2292    cusolver_mg_device_select as "cusolverMgDeviceSelect": PFN_cusolverMgDeviceSelect;
2293    /// cuSOLVER: create a multi-GPU device grid. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2294    cusolver_mg_create_device_grid as "cusolverMgCreateDeviceGrid": PFN_cusolverMgCreateDeviceGrid;
2295    /// cuSOLVER: destroy a multi-GPU device grid. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2296    cusolver_mg_destroy_grid as "cusolverMgDestroyGrid": PFN_cusolverMgDestroyGrid;
2297    /// cuSOLVER: create a multi-GPU matrix descriptor. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2298    cusolver_mg_create_matrix_desc as "cusolverMgCreateMatrixDesc": PFN_cusolverMgCreateMatrixDesc;
2299    /// cuSOLVER: destroy a multi-GPU matrix descriptor. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2300    cusolver_mg_destroy_matrix_desc as "cusolverMgDestroyMatrixDesc": PFN_cusolverMgDestroyMatrixDesc;
2301    /// cuSOLVER: multi-GPU workspace-size query for LU factorization with partial pivoting. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2302    cusolver_mg_getrf_buffer_size as "cusolverMgGetrf_bufferSize": PFN_cusolverMgGetrf_bufferSize;
2303    /// cuSOLVER: multi-GPU LU factorization with partial pivoting. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2304    cusolver_mg_getrf as "cusolverMgGetrf": PFN_cusolverMgGetrf;
2305    /// cuSOLVER: multi-GPU workspace-size query for Cholesky factorization. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2306    cusolver_mg_potrf_buffer_size as "cusolverMgPotrf_bufferSize": PFN_cusolverMgPotrf_bufferSize;
2307    /// cuSOLVER: multi-GPU Cholesky factorization. See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2308    cusolver_mg_potrf as "cusolverMgPotrf": PFN_cusolverMgPotrf;
2309    /// cuSOLVER: multi-GPU workspace-size query for symmetric eigendecomposition (divide-and-conquer). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2310    cusolver_mg_syevd_buffer_size as "cusolverMgSyevd_bufferSize": PFN_cusolverMgSyevd_bufferSize;
2311    /// cuSOLVER: multi-GPU symmetric eigendecomposition (divide-and-conquer). See <https://docs.nvidia.com/cuda/cusolver/index.html>.
2312    cusolver_mg_syevd as "cusolverMgSyevd": PFN_cusolverMgSyevd;
2313}
2314
2315/// Lazy-load the cuSOLVERMg multi-GPU shared library and return its function-pointer table.
2316pub fn cusolver_mg() -> Result<&'static CusolverMg, LoaderError> {
2317    static MG: OnceLock<CusolverMg> = OnceLock::new();
2318    if let Some(c) = MG.get() {
2319        return Ok(c);
2320    }
2321    let candidates: Vec<&'static str> = cusolver_mg_candidates()
2322        .into_iter()
2323        .map(|s| Box::leak(s.into_boxed_str()) as &'static str)
2324        .collect();
2325    let leaked: &'static [&'static str] = Box::leak(candidates.into_boxed_slice());
2326    let lib = Library::open("cusolverMg", leaked)?;
2327    let _ = MG.set(CusolverMg::empty(lib));
2328    Ok(MG.get().expect("OnceLock set or lost race"))
2329}