Skip to main content

themis/topology/
gpu.rs

1//! GPU topology query types.
2
3use core::num::{NonZeroU32, NonZeroU64, NonZeroUsize};
4
5use super::types::GpuDeviceProperties;
6use crate::law::{MemoryTier, TopologyEpoch};
7
8/// GPU device topology snapshot (atlas ADR 0002).
9///
10/// Provider-fed: themis stays stateless law, so there is no `detect()` here —
11/// device backends (hephaestus) construct this from wgpu adapter limits or
12/// CUDA device attributes via [`GpuTopology::from_provider`]. Consumers:
13/// moirai's occupancy planner (warp-aware launch shaping) and mnemosyne's
14/// kernel resource budgets read these capacities; the `Registers`/`SharedMem`
15/// figures are budget vocabulary, never host-allocatable (see
16/// [`MemoryTier::is_host_allocatable`]). Every capacity accessor returns
17/// `None` when the provider's API did not report it — unknowability is
18/// type-level, never a sentinel zero.
19#[derive(Debug, Clone, PartialEq, Eq)]
20pub struct GpuTopology {
21    epoch: TopologyEpoch,
22    properties: GpuDeviceProperties,
23}
24
25impl GpuTopology {
26    /// Construct a snapshot from provider-reported device properties.
27    #[must_use]
28    pub const fn from_provider(properties: GpuDeviceProperties) -> Self {
29        Self {
30            epoch: TopologyEpoch::INITIAL,
31            properties,
32        }
33    }
34
35    /// Snapshot epoch.
36    #[must_use]
37    #[inline]
38    pub const fn epoch(&self) -> TopologyEpoch {
39        self.epoch
40    }
41
42    /// Streaming-multiprocessor / compute-unit count, when reported.
43    #[must_use]
44    #[inline]
45    pub const fn compute_units(&self) -> Option<NonZeroU32> {
46        self.properties.compute_units
47    }
48
49    /// Warp / wavefront / subgroup width in lanes, when reported.
50    #[must_use]
51    #[inline]
52    pub const fn warp_width(&self) -> Option<NonZeroU32> {
53        self.properties.warp_width
54    }
55
56    /// Maximum resident threads per compute unit, when reported.
57    #[must_use]
58    #[inline]
59    pub const fn max_threads_per_unit(&self) -> Option<NonZeroU32> {
60        self.properties.max_threads_per_unit
61    }
62
63    /// 32-bit registers per compute unit (budgeted `Registers` tier), when
64    /// reported.
65    #[must_use]
66    #[inline]
67    pub const fn registers_per_unit(&self) -> Option<NonZeroU32> {
68        self.properties.registers_per_unit
69    }
70
71    /// Shared/local memory bytes per compute unit (budgeted `SharedMem`
72    /// tier), when reported.
73    #[must_use]
74    #[inline]
75    pub const fn shared_mem_per_unit_bytes(&self) -> Option<NonZeroUsize> {
76        self.properties.shared_mem_per_unit_bytes
77    }
78
79    /// Device L2 cache size in bytes, when reported.
80    #[must_use]
81    #[inline]
82    pub const fn l2_bytes(&self) -> Option<NonZeroUsize> {
83        self.properties.l2_bytes
84    }
85
86    /// Device global-memory tier.
87    #[must_use]
88    #[inline]
89    pub const fn memory_tier(&self) -> MemoryTier {
90        self.properties.memory_tier
91    }
92
93    /// Device global-memory capacity in bytes, when reported.
94    #[must_use]
95    #[inline]
96    pub const fn memory_bytes(&self) -> Option<NonZeroU64> {
97        self.properties.memory_bytes
98    }
99
100    /// Total resident warps at theoretical full occupancy:
101    /// `compute_units · max_threads_per_unit / warp_width`, when all three
102    /// capacities are reported.
103    #[must_use]
104    #[inline]
105    pub const fn max_resident_warps(&self) -> Option<u64> {
106        match (
107            self.properties.compute_units,
108            self.properties.max_threads_per_unit,
109            self.properties.warp_width,
110        ) {
111            (Some(units), Some(threads), Some(width)) => {
112                Some((units.get() as u64) * (threads.get() as u64) / (width.get() as u64))
113            }
114            _ => None,
115        }
116    }
117}