1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
use super::core::Space;
use super::tile::{IntoTile, TileConfig};
use super::tuner::SpatialAutoTuner;
use parsu::profile::HardwareProfile;
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub(crate) struct ResolvedSpatialDispatch {
pub global_size: (u32, u32, u32),
pub local_size: (u32, u32, u32),
pub dispatch_groups: (u32, u32, u32),
}
impl Space {
/// Creates a 1D GPU execution domain with the specified global problem size.
///
/// By default, tile dimensions are automatically factored into optimal workgroups
#[inline(always)]
pub fn gpu_x(size_x: usize) -> Self {
Self {
x: 0,
y: 0,
z: 0,
cell_x: 0,
cell_y: 0,
cell_z: 0,
tile_x: 0,
tile_y: 0,
tile_z: 0,
size_x: size_x.max(1),
size_y: 1,
size_z: 1,
tile_size_x: 0,
tile_size_y: 0,
tile_size_z: 0,
}
}
/// Creates a 2D GPU execution domain with the specified width and height.
///
/// By default, tile dimensions are automatically factored into optimal workgroups.
#[inline(always)]
pub fn gpu_xy(size_x: usize, size_y: usize) -> Self {
Self {
x: 0,
y: 0,
z: 0,
cell_x: 0,
cell_y: 0,
cell_z: 0,
tile_x: 0,
tile_y: 0,
tile_z: 0,
size_x: size_x.max(1),
size_y: size_y.max(1),
size_z: 1,
tile_size_x: 0,
tile_size_y: 0,
tile_size_z: 0,
}
}
/// Creates a 3D volumetric GPU execution domain with the specified width, height, and depth.
///
/// By default, tile dimensions are automatically factored into optimal workgroups.
#[inline(always)]
pub fn gpu_xyz(size_x: usize, size_y: usize, size_z: usize) -> Self {
Self {
x: 0,
y: 0,
z: 0,
cell_x: 0,
cell_y: 0,
cell_z: 0,
tile_x: 0,
tile_y: 0,
tile_z: 0,
size_x: size_x.max(1),
size_y: size_y.max(1),
size_z: size_z.max(1),
tile_size_x: 0,
tile_size_y: 0,
tile_size_z: 0,
}
}
/// Overrides automatic tile tuning with an explicit, fixed tile configuration.
///
/// Accepts scalar values for 1D, tuples/arrays for 2D, and triplets for 3D.
///
/// # Examples
/// ```rust
/// use enki::Space;
///
/// let space_1d = Space::gpu_x(10_000_000).tile(256);
/// let space_2d = Space::gpu_xy(1920, 1080).tile([16, 16]);
/// let space_3d = Space::gpu_xyz(256, 256, 128).tile([8, 8, 4]);
/// ```
#[inline(always)]
pub fn tile<T: IntoTile>(mut self, tile: T) -> Self {
match tile.into_tile() {
TileConfig::Auto => {
self.tile_size_x = 0;
self.tile_size_y = 0;
self.tile_size_z = 0;
}
TileConfig::Custom(tx, ty, tz) => {
self.tile_size_x = tx as usize;
self.tile_size_y = ty as usize;
self.tile_size_z = tz as usize;
if self.tile_size_x > 0 {
self.cell_x = self.x % self.tile_size_x;
self.tile_x = self.x / self.tile_size_x;
}
if self.tile_size_y > 0 {
self.cell_y = self.y % self.tile_size_y;
self.tile_y = self.y / self.tile_size_y;
}
if self.tile_size_z > 0 {
self.cell_z = self.z % self.tile_size_z;
self.tile_z = self.z / self.tile_size_z;
}
}
}
self
}
pub(crate) fn resolve_dispatch(&self, profile: &HardwareProfile) -> ResolvedSpatialDispatch {
let global_size = (self.size_x as u32, self.size_y as u32, self.size_z as u32);
let config = if self.tile_size_x == 0 {
TileConfig::Auto
} else {
TileConfig::Custom(
self.tile_size_x as u32,
self.tile_size_y.max(1) as u32,
self.tile_size_z.max(1) as u32,
)
};
let local_size = SpatialAutoTuner::resolve_tile(config, global_size, profile);
let group_x = global_size.0.div_ceil(local_size.0);
let group_y = global_size.1.div_ceil(local_size.1);
let group_z = global_size.2.div_ceil(local_size.2);
ResolvedSpatialDispatch {
global_size,
local_size,
dispatch_groups: (group_x, group_y, group_z),
}
}
}