This file is a merged representation of the entire codebase, combined into a single document by Repomix.
<file_summary>
This section contains a summary of this file.
<purpose>
This file contains a packed representation of the entire repository's contents.
It is designed to be easily consumable by AI systems for analysis, code review,
or other automated processes.
</purpose>
<file_format>
The content is organized as follows:
1. This summary section
2. Repository information
3. Directory structure
4. Repository files (if enabled)
5. Multiple file entries, each consisting of:
- File path as an attribute
- Full contents of the file
</file_format>
<usage_guidelines>
- This file should be treated as read-only. Any changes should be made to the
original repository files, not this packed version.
- When processing this file, use the file path to distinguish
between different files in the repository.
- Be aware that this file may contain sensitive information. Handle it with
the same level of security as you would the original repository.
</usage_guidelines>
<notes>
- Some files may have been excluded based on .gitignore rules and Repomix's configuration
- Binary files are not included in this packed representation. Please refer to the Repository Structure section for a complete list of file paths, including binary files
- Files matching patterns in .gitignore are excluded
- Files matching default ignore patterns are excluded
- Files are sorted by Git change count (files with more changes are at the bottom)
</notes>
</file_summary>
<directory_structure>
enki_api/
context/
ambient.rs
contract.rs
errors.rs
flow.rs
instance.rs
instant.rs
mod.rs
resources/
slice/
dispatch.rs
format.rs
mod.rs
range.rs
ro.rs
rw.rs
vec/
core.rs
dispatch.rs
format.rs
lifecycle.rs
mod.rs
ops.rs
state.rs
transfer.rs
atomic.rs
mod.rs
param.rs
tile_mem.rs
space/
builder.rs
core.rs
cpu.rs
mod.rs
tile.rs
tuner.rs
type_safety/
arg_match.rs
mod.rs
nam_run.rs
mod.rs
enki_cli/
src/
bin/
cargo-enki.rs
enki.rs
args.rs
decoder.rs
lib.rs
runner.rs
Cargo.toml
README.md
enki_macros/
src/
lib.rs
Cargo.toml
README.md
lib.rs
</directory_structure>
<files>
This section contains the contents of the repository's files.
<file path="enki_api/context/ambient.rs">
use std::cell::Cell;
use std::panic::Location;
use super::errors::emit_and_abort;
use super::flow::Flow;
thread_local! {
static ACTIVE_FLOW_PTR: Cell<*mut Flow<'static>> = const { Cell::new(std::ptr::null_mut()) };
static ACTIVE_FLOW_CALLER: Cell<Option<&'static Location<'static>>> = const { Cell::new(None) };
}
pub(crate) struct ActiveFlowGuard;
impl ActiveFlowGuard {
#[track_caller]
pub fn enter(flow: &mut Flow<'_>) -> Self {
let caller = Location::caller();
ACTIVE_FLOW_PTR.with(|cell| {
if !cell.get().is_null() {
let diag = anu::diagnostics::rt::nested_frame(caller);
emit_and_abort(&diag);
}
cell.set(flow as *mut Flow<'_> as *mut Flow<'static>);
});
ACTIVE_FLOW_CALLER.with(|cell| {
cell.set(Some(caller));
});
Self
}
}
impl Drop for ActiveFlowGuard {
fn drop(&mut self) {
ACTIVE_FLOW_PTR.with(|cell| {
cell.set(std::ptr::null_mut());
});
ACTIVE_FLOW_CALLER.with(|cell| {
cell.set(None);
});
}
}
#[inline(always)]
pub(crate) fn is_inside_active_flow() -> bool {
ACTIVE_FLOW_PTR.with(|cell| !cell.get().is_null())
}
#[track_caller]
pub(crate) fn with_active_flow<R, F>(f: F) -> R
where
F: FnOnce(&mut Flow<'_>) -> R,
{
let caller = Location::caller();
ACTIVE_FLOW_PTR.with(|cell| {
let ptr = cell.get();
if ptr.is_null() {
let diag = anu::diagnostics::rt::outside_frame(caller);
emit_and_abort(&diag);
}
// SAFETY: الـ Pointer صالح طالما أن الحارس ActiveFlowGuard لم يسقط
let flow = unsafe { &mut *ptr };
f(flow)
})
}
#[inline(always)]
pub(crate) fn resolve_user_caller(
caller: &'static Location<'static>,
) -> &'static Location<'static> {
let file = caller.file();
let is_internal = file.contains("enki_api")
|| file.contains("format.rs")
|| file.contains("state.rs")
|| file.contains("core.rs")
|| file.contains("ambient.rs")
|| file.contains("flow.rs");
if is_internal {
ACTIVE_FLOW_CALLER.with(|c| c.get()).unwrap_or(caller)
} else {
caller
}
}
</file>
<file path="enki_api/context/contract.rs">
use anu::nam_args_api::NamSignatureContract;
use anyhow::{Result, anyhow};
#[inline]
pub fn resolve_nam_name<F: 'static>() -> Result<String> {
let full_name = std::any::type_name::<F>();
extract_nam_name(full_name)
}
pub fn extract_nam_name(full_name: &str) -> Result<String> {
let base = match full_name.find('<') {
Some(idx) => &full_name[..idx],
None => full_name,
};
let last_segment = match base.rfind("::") {
Some(idx) => &base[idx + 2..],
None => base,
};
if last_segment.is_empty() {
return Err(anyhow!(
"[Enki Contract] Failed to extract nam name from identifier: '{full_name}'"
));
}
Ok(last_segment.to_string())
}
pub fn lookup_signature_contract(nam_name: &str) -> Option<&'static NamSignatureContract> {
let symbol_name = format!("__ENKI_CONTRACT_{nam_name}\0");
#[cfg(unix)]
unsafe {
unsafe extern "C" {
fn dlsym(
handle: *mut std::ffi::c_void,
symbol: *const std::os::raw::c_char,
) -> *mut std::ffi::c_void;
}
let ptr = dlsym(
std::ptr::null_mut(),
symbol_name.as_ptr() as *const std::os::raw::c_char,
);
if !ptr.is_null() {
return Some(&*(ptr as *const NamSignatureContract));
}
}
#[cfg(windows)]
unsafe {
unsafe extern "system" {
fn GetModuleHandleA(lpModuleName: *const std::os::raw::c_char)
-> *mut std::ffi::c_void;
fn GetProcAddress(
hModule: *mut std::ffi::c_void,
lpProcName: *const std::os::raw::c_char,
) -> *mut std::ffi::c_void;
}
let module = GetModuleHandleA(std::ptr::null());
if !module.is_null() {
let ptr = GetProcAddress(module, symbol_name.as_ptr() as *const std::os::raw::c_char);
if !ptr.is_null() {
return Some(&*(ptr as *const NamSignatureContract));
}
}
}
None
}
</file>
<file path="enki_api/context/errors.rs">
use std::io::Write;
fn executable_name() -> String {
std::env::current_exe()
.ok()
.and_then(|p| p.file_name().map(|n| n.to_string_lossy().to_string()))
.unwrap_or_else(|| "enki_app".to_string())
}
pub(crate) fn emit_and_abort(diag: &anu::diagnostics::Diagnostic) -> ! {
let _ = std::io::stdout().flush();
eprintln!();
let formatted = anu::diagnostics::emit_diagnostic(diag);
eprint!("{formatted}");
let bin_name = executable_name();
eprintln!(
"\x1b[1;91merror\x1b[0m: aborting execution of `{bin_name}` due to 1 previous error\n"
);
let _ = std::io::stderr().flush();
std::process::exit(101);
}
pub(crate) fn handle_execution_error(err: &anyhow::Error) -> ! {
let err_str = err.root_cause().to_string();
eprint!("{err_str}");
if !err_str.ends_with('\n') {
eprintln!();
}
if !err_str.contains("could not compile") {
let bin_name = executable_name();
eprintln!(
"\x1b[1;91merror\x1b[0m: execution halted in `{bin_name}` due to 1 previous error\n"
);
}
let _ = std::io::stderr().flush();
std::process::exit(101);
}
pub(crate) fn emit_warning(diag: &anu::diagnostics::Diagnostic) {
let formatted = anu::diagnostics::emit_diagnostic(diag);
eprint!("{formatted}");
let _ = std::io::stderr().flush();
}
</file>
<file path="enki_api/context/flow.rs">
use anyhow::{Context, Result};
use ash::vk;
use std::sync::Mutex;
use std::sync::atomic::{AtomicBool, Ordering};
use crate::enki_api::context::contract::{lookup_signature_contract, resolve_nam_name};
use crate::enki_api::context::errors::emit_and_abort;
use crate::enki_api::context::instance::Enki;
use crate::enki_api::resources::{GpuVec, Slice};
use crate::enki_api::space::Space;
use super::instant::GpuInstant;
use anu::diagnostics::source::Span;
use anu::nam_args_api::NamDispatchMap;
use anu::pipeline_synthesis::ComputeSynthesisInput;
use anu::recording::queue::TaskQueue;
use anu::recording::recipe::CompiledExecutionRecipe;
use anu::recording::task::{ComputeTask, PresentBufferTask, RawTask};
use anu::validation::{BorrowEngine, FrameBorrowLedger};
use utu::GpuWindow;
/// An active, coherent GPU compute and presentation execution stream.
///
/// A `Flow` batches nam (kernel) dispatches and display presentation tasks into an atomic
/// command buffer sequence synchronized by timeline semaphores.
pub struct Flow<'a> {
pub enki: &'a Enki,
pub queue: TaskQueue<'a>,
pub cmd: vk::CommandBuffer,
pub slot_idx: usize,
pub timeline_value: u64,
pub submitted: AtomicBool,
pub sticky_error: Option<anyhow::Error>,
pub borrow_ledger: FrameBorrowLedger,
pub timestamp_count: u32,
}
impl<'a> Flow<'a> {
/// Initializes a new Flow, acquiring a command buffer from the ring and resetting query pools.
pub fn new(enki: &'a Enki) -> Self {
let engine = &enki.engine;
let current_gpu_value = engine.timeline_semaphore.get_timeline_value().unwrap_or(0);
engine.allocator.reclaim_resources(current_gpu_value);
let timeline_value = engine.timeline_counter.fetch_add(1, Ordering::SeqCst) + 1;
engine
.allocator
.current_timeline_value
.store(timeline_value, Ordering::Release);
let (cmd, slot_idx) = {
let mut ring = engine.command_ring.lock().unwrap();
ring.acquire_next_cmd(engine.raw_device(), engine.timeline_semaphore.handle)
.context("[Flow] Failed to acquire command buffer from ring")
.unwrap()
};
let max_queries = engine.max_timestamp_queries;
let start_query = (slot_idx as u32) * max_queries;
unsafe {
let begin_info = vk::CommandBufferBeginInfo::default()
.flags(vk::CommandBufferUsageFlags::ONE_TIME_SUBMIT);
engine
.raw_device()
.begin_command_buffer(cmd, &begin_info)
.unwrap();
engine.raw_device().cmd_reset_query_pool(
cmd,
engine.query_pool,
start_query,
max_queries,
);
}
Self {
enki,
queue: TaskQueue::new(),
cmd,
slot_idx,
timeline_value,
submitted: AtomicBool::new(false),
sticky_error: None,
borrow_ledger: FrameBorrowLedger::new(),
timestamp_count: 0,
}
}
/// Queues an owned GPU vector for direct presentation to the display window.
///
/// # Temporal Presentation Hazard
/// Once queued for presentation, any subsequent dispatch (`.run()` or `.run_unchecked()`) within the same flow that
/// attempts to mutate (`&mut`) this buffer halts execution with diagnostic **`error[E1010]`**.
/// Read-only access (`&`) after presentation remains permitted.
///
/// # Diagnostics
/// Halts execution with diagnostic **`error[E2002]`** if element count is less than `width * height`.
#[track_caller]
pub fn present<T: Copy + Send + Sync + 'static>(&mut self, pixels: &GpuVec<T>) {
self.validate_and_present_raw(
pixels.slot_index,
pixels._inner.buffer(),
pixels.offset as u64,
pixels.len(),
pixels.stride(),
);
}
/// Queues a contiguous GPU slice for direct presentation to the display window.
///
/// # Temporal Presentation Hazard
/// Once queued for presentation, any subsequent dispatch (`.run()` or `.run_unchecked()`) within the same flow that
/// attempts to mutate (`&mut`) this buffer halts execution with diagnostic **`error[E1010]`**.
/// Read-only access (`&`) after presentation remains permitted.
///
/// # Diagnostics
/// Halts execution with diagnostic **`error[E2002]`** if element count is less than `width * height`. #[track_caller]
pub fn present_slice<T: Copy + Send + Sync + 'static>(&mut self, pixels: &Slice<T>) {
self.validate_and_present_raw(
pixels.slot_index,
pixels._inner.buffer(),
pixels.offset as u64,
pixels.len(),
pixels.stride(),
);
}
#[track_caller]
fn validate_and_present_raw(
&mut self,
buffer_id: u32,
buffer: ash::vk::Buffer,
offset: u64,
element_count: usize,
_element_stride: usize,
) {
self.borrow_ledger
.mark_queued_for_presentation(buffer_id as usize);
let (width, height) = if let Some(window_mutex) = &self.enki.gpu_window {
let window = window_mutex.lock().unwrap();
(window.extent.width, window.extent.height)
} else {
(0, 0)
};
if width == 0 || height == 0 {
return;
}
let required_elements = (width * height) as usize;
if element_count < required_elements {
let caller = std::panic::Location::caller();
let diag = anu::diagnostics::rt::present_dimension_mismatch(
width,
height,
element_count,
caller,
);
emit_and_abort(&diag);
}
let task = PresentBufferTask {
buffer_id,
buffer,
offset,
width,
height,
};
self.push_task(RawTask::PresentBuffer(task));
}
pub(crate) fn nam_impl_direct<F>(
&mut self,
space: &Space,
args_ctx: anu::nam_args_api::IngressContext<'static>,
mut map: NamDispatchMap,
) -> Result<()>
where
F: 'static,
{
let engine = &self.enki.engine;
let nam_name = resolve_nam_name::<F>()?;
map.nam_name = nam_name.clone();
if let Some(contract) = lookup_signature_contract(&nam_name) {
map.expected_contract = Some(contract);
}
let required_param_bytes = args_ctx.pack().len() as u64;
let arena_capacity = engine.param_arena.size_bytes();
if required_param_bytes > arena_capacity {
let diag =
anu::diagnostics::hw::param_arena_overflow(required_param_bytes, arena_capacity);
crate::enki_api::context::errors::emit_and_abort(&diag);
}
if let Err(violation) = BorrowEngine::validate_dispatch(&map, &mut self.borrow_ledger) {
let diag = anu::diagnostics::ContractDiagnosticBuilder::from_violation(violation, &map);
emit_and_abort(&diag);
}
let dispatch = space.resolve_dispatch(&engine.hardware_profile);
let input = ComputeSynthesisInput {
nam_name: nam_name.clone(),
local_size: dispatch.local_size,
host_manifest_dir: std::env::var("CARGO_MANIFEST_DIR").ok(),
caller_file_path: map.call_site.map(|(f, _, _)| f.to_string()),
arg_descriptors: args_ctx.descriptors.clone(),
};
let artifact = engine
.synthesizer
.synthesize_compute(engine, &input)
.context("[Flow] JIT synthesis failed for nam")?;
let total_threads = (dispatch.global_size.0 as u64)
* (dispatch.global_size.1 as u64)
* (dispatch.global_size.2 as u64);
let required_stack_bytes = (artifact.stack_size_per_thread as u64) * total_threads;
let (stack_bda, stack_buffer) = if required_stack_bytes > 0 {
match apsu::GpuStackBuffer::allocate(engine.allocator.clone(), required_stack_bytes) {
Ok(Some(buf)) => {
let bda = buf.device_address();
(bda, Some(buf))
}
Ok(None) => (0, None),
Err(alloc_err) => {
let mut diag = anu::diagnostics::hw::stack_overflow(
&alloc_err,
dispatch.global_size,
artifact.stack_size_per_thread,
None,
);
if let Some((file, line, col)) = map.call_site {
diag.add_span(Span::primary(file, line as usize, col as usize, 1));
}
emit_and_abort(&diag);
}
}
} else {
(0, None)
};
let task = ComputeTask {
pipeline: artifact.pipeline,
layout: artifact.layout,
grid_size: dispatch.global_size,
local_size: dispatch.local_size,
args_ctx,
stack_bda,
_stack_buffer: stack_buffer,
};
self.push_task(RawTask::Compute(task));
Ok(())
}
/// Finalizes and submits the recorded flow to the GPU queue, halting on failure.
pub fn end_flow(self) {
if let Err(e) = self.try_end_flow() {
crate::enki_api::context::errors::handle_execution_error(&e);
}
}
/// Finalizes and submits the recorded flow to the GPU queue, returning a `Result`.
pub fn try_end_flow(self) -> Result<()> {
if self.submitted.swap(true, Ordering::SeqCst) {
return Ok(());
}
let engine = &self.enki.engine;
let recipe = engine.compile_recipe(&self.queue);
let has_present = self
.queue
.tasks
.iter()
.any(|t| matches!(t, RawTask::PresentBuffer(_)));
if has_present && let Some(window_mutex) = &self.enki.gpu_window {
self.submit_windowed(&recipe, window_mutex)?;
} else {
self.submit_headless(&recipe)?;
}
{
let mut ring = engine.command_ring.lock().unwrap();
ring.update_slot_timeline(self.slot_idx, self.timeline_value);
}
Ok(())
}
fn submit_headless(&self, recipe: &CompiledExecutionRecipe) -> Result<()> {
let engine = &self.enki.engine;
let device = engine.raw_device();
recipe
.replay(
engine,
self.cmd,
&self.queue,
self.slot_idx,
engine.query_pool,
)
.context("[Flow] Recipe replay failed in headless submit")?;
unsafe {
device.end_command_buffer(self.cmd)?;
}
let cmd_buffers = [self.cmd];
let signal_semaphores = [engine.timeline_semaphore.handle];
let signal_values = [self.timeline_value];
let mut timeline_info =
vk::TimelineSemaphoreSubmitInfo::default().signal_semaphore_values(&signal_values);
let submit_info = vk::SubmitInfo::default()
.push_next(&mut timeline_info)
.command_buffers(&cmd_buffers)
.signal_semaphores(&signal_semaphores);
unsafe {
device.queue_submit(engine.queue.handle, &[submit_info], vk::Fence::null())?;
engine
.timeline_semaphore
.wait_timeline(self.timeline_value, std::time::Duration::from_secs(5))?;
}
Ok(())
}
fn submit_windowed(
&self,
recipe: &CompiledExecutionRecipe,
window_mutex: &Mutex<GpuWindow>,
) -> Result<()> {
let engine = &self.enki.engine;
let device = engine.raw_device();
let mut window_lock = window_mutex.lock().unwrap();
let (image_index, _) = window_lock
.acquire_next_image(std::time::Duration::from_secs(5))
.context("[Flow] Failed to acquire next swapchain image")?;
recipe
.replay(
engine,
self.cmd,
&self.queue,
self.slot_idx,
engine.query_pool,
)
.context("[Flow] Recipe replay failed in windowed submit")?;
for task in &self.queue.tasks {
if let RawTask::PresentBuffer(p) = task {
window_lock.cmd_copy_buffer_to_image(
self.cmd,
image_index,
p.buffer,
p.offset,
p.width,
p.height,
);
}
}
unsafe {
device.end_command_buffer(self.cmd)?;
}
let wait_semaphores = [window_lock.current_image_acquired_semaphore()];
let wait_stages =
[vk::PipelineStageFlags::COLOR_ATTACHMENT_OUTPUT | vk::PipelineStageFlags::TRANSFER];
let signal_semaphores = [
engine.timeline_semaphore.handle,
window_lock.current_render_finished_semaphore(),
];
let signal_values = [self.timeline_value, 0];
let mut timeline_info =
vk::TimelineSemaphoreSubmitInfo::default().signal_semaphore_values(&signal_values);
let cmd_buffers = [self.cmd];
let submit_info = vk::SubmitInfo::default()
.push_next(&mut timeline_info)
.wait_semaphores(&wait_semaphores)
.wait_dst_stage_mask(&wait_stages)
.command_buffers(&cmd_buffers)
.signal_semaphores(&signal_semaphores);
let in_flight_fence = window_lock.current_in_flight_fence();
unsafe {
device.queue_submit(engine.queue.handle, &[submit_info], in_flight_fence)?;
}
window_lock
.present_image(engine.queue.handle, image_index)
.context("[Flow] Failed to present swapchain image")?;
Ok(())
}
/// Records a hardware timestamp query mark on the GPU timeline for profiling.
#[track_caller]
pub fn mark(&mut self) -> GpuInstant {
let caller = std::panic::Location::caller();
let engine = &self.enki.engine;
let max_queries = engine.max_timestamp_queries;
if self.timestamp_count >= max_queries {
let diag = anu::diagnostics::hw::timestamp_queries_exceeded(
self.timestamp_count + 1,
max_queries,
Some(caller),
);
emit_and_abort(&diag);
}
let global_query_slot = (self.slot_idx as u32) * max_queries + self.timestamp_count;
self.timestamp_count += 1;
self.write_timestamp(global_query_slot, vk::PipelineStageFlags2::ALL_COMMANDS);
GpuInstant {
query_slot: global_query_slot,
timeline_value: self.timeline_value,
timestamp_period: engine.timestamp_period,
}
}
/// Records a timestamp query into the active flow.
pub fn write_timestamp(&mut self, query_index: u32, stage: vk::PipelineStageFlags2) {
self.push_task(RawTask::WriteTimestamp { query_index, stage });
}
/// Pushes a low-level task onto the internal execution queue.
pub fn push_task(&mut self, task: RawTask<'a>) {
self.queue.push(task);
}
}
impl<'a> Drop for Flow<'a> {
fn drop(&mut self) {
if !self.submitted.load(Ordering::SeqCst) {
let _ = self.enki.engine.wait_idle();
}
}
}
</file>
<file path="enki_api/context/instance.rs">
use anyhow::{Context, Result};
use std::sync::{Arc, Mutex, OnceLock};
use anu::context::{EngineConfig, EnkiEngine, EnkiEngineBuilder};
use utu::GpuWindow;
use super::ambient::ActiveFlowGuard;
use super::errors::handle_execution_error;
use super::flow::Flow;
static ACTIVE_ENKI: OnceLock<Arc<Enki>> = OnceLock::new();
/// Sets the global active Enki context instance.
pub fn set_active_enki(enki: Arc<Enki>) {
let _ = ACTIVE_ENKI.set(enki);
}
/// Retrieves a clone of the globally active Enki context.
pub fn active_enki() -> Arc<Enki> {
ACTIVE_ENKI
.get()
.cloned()
.expect("[Enki] No active Enki context found. Did you call 'Enki::init()'?")
}
/// Retrieves a clone of the underlying Vulkan compute engine (`EnkiEngine`).
pub fn active_engine() -> Arc<EnkiEngine> {
active_enki().engine.clone()
}
/// Builder for configuring engine settings before runtime initialization.
pub struct EnkiBuilder {
config: EngineConfig,
}
impl EnkiBuilder {
pub fn new() -> Self {
Self {
config: EngineConfig::default(),
}
}
/// Sets the application name reported to the Vulkan driver.
pub fn app_name(mut self, name: impl Into<String>) -> Self {
self.config.app_name = name.into();
self
}
/// Sets the maximum capacity in bytes allocated for uniform parameter uploads (ParamArena).
pub fn param_arena_size(mut self, size_bytes: u64) -> Self {
self.config.max_param_arena_size = Some(size_bytes);
self
}
// /// Configures the maximum number of bindless sampled image descriptors.
// pub fn max_sampled_images(mut self, count: u32) -> Self {
// self.config.max_sampled_images = count;
// self
// }
// /// Configures the maximum number of bindless storage image descriptors.
// pub fn max_storage_images(mut self, count: u32) -> Self {
// self.config.max_storage_images = count;
// self
// }
// /// Configures the maximum number of bindless sampler descriptors.
// pub fn max_samplers(mut self, count: u32) -> Self {
// self.config.max_samplers = count;
// self
// }
/// Finalizes configuration and initializes a headless GPU context.
#[track_caller]
pub fn init(self) -> Arc<Enki> {
let caller = std::panic::Location::caller();
let mut config = self.config;
config.headless = true;
config.caller_location = Some((caller.file(), caller.line(), caller.column()));
match Enki::new_headless_with_config(config) {
Ok(enki) => enki,
Err(e) => handle_execution_error(&e),
}
}
/// Finalizes configuration and initializes a windowed presentation GPU context.
#[track_caller]
pub fn init_windowed<W>(self, window: Arc<W>, width: u32, height: u32) -> Arc<Enki>
where
W: raw_window_handle::HasWindowHandle
+ raw_window_handle::HasDisplayHandle
+ Send
+ Sync
+ 'static,
{
let caller = std::panic::Location::caller();
let mut config = self.config;
config.headless = false;
config.caller_location = Some((caller.file(), caller.line(), caller.column()));
match Enki::new_windowed_with_config(config, window, width, height) {
Ok(enki) => enki,
Err(e) => handle_execution_error(&e),
}
}
/// Sets the maximum number of hardware timestamp profiling queries per flow.
pub fn max_timestamp_queries(mut self, count: u32) -> Self {
self.config.max_timestamp_queries = count;
self
}
}
impl Default for EnkiBuilder {
fn default() -> Self {
Self::new()
}
}
/// The central handle to the Enki heterogeneous GPU runtime.
///
/// Manages Vulkan instance lifecycle, hardware devices, memory allocation heaps,
/// and execution flows across host CPU and GPU silicon.
pub struct Enki {
pub window_context: Option<Arc<dyn std::any::Any + Send + Sync>>,
pub gpu_window: Option<Mutex<GpuWindow>>,
pub engine: Arc<EnkiEngine>,
}
impl Enki {
/// Returns a new `EnkiBuilder` to configure runtime options before initialization.
pub fn builder() -> EnkiBuilder {
EnkiBuilder::new()
}
/// Initializes a headless GPU compute context with default settings.
pub fn init() -> Arc<Self> {
Self::builder().init()
}
/// Initializes a windowed GPU context with swapchain presentation support.
pub fn init_windowed<W>(window: Arc<W>, width: u32, height: u32) -> Arc<Self>
where
W: raw_window_handle::HasWindowHandle
+ raw_window_handle::HasDisplayHandle
+ Send
+ Sync
+ 'static,
{
Self::builder().init_windowed(window, width, height)
}
fn new_headless_with_config(config: EngineConfig) -> Result<Arc<Self>> {
let engine = EnkiEngineBuilder::new(config)
.build()
.context("[Enki Core] Failed to build headless EnkiEngine")?;
let enki = Arc::new(Self {
window_context: None,
gpu_window: None,
engine: Arc::from(engine),
});
set_active_enki(enki.clone());
Ok(enki)
}
fn new_windowed_with_config<W>(
mut config: EngineConfig,
window: Arc<W>,
width: u32,
height: u32,
) -> Result<Arc<Self>>
where
W: raw_window_handle::HasWindowHandle
+ raw_window_handle::HasDisplayHandle
+ Send
+ Sync
+ 'static,
{
let display_handle = window
.display_handle()
.map_err(|_| {
let diag = anu::diagnostics::hw::headless_display_mismatch();
anyhow::anyhow!("{}", anu::diagnostics::emit_diagnostic(&diag))
})?
.as_raw();
let surface_extensions = ash_window::enumerate_required_extensions(display_handle)
.map_err(|_| {
let diag = anu::diagnostics::hw::headless_display_mismatch();
anyhow::anyhow!("{}", anu::diagnostics::emit_diagnostic(&diag))
})?;
for &ext_ptr in surface_extensions {
let cstr = unsafe { std::ffi::CStr::from_ptr(ext_ptr) };
config.required_instance_extensions.push(cstr.to_owned());
}
config
.required_device_extensions
.push(std::ffi::CString::from(ash::khr::swapchain::NAME));
let engine = EnkiEngineBuilder::new(config)
.build()
.context("[Enki Core] Failed to build windowed EnkiEngine")?;
let engine = Arc::new(engine);
let gpu_window = GpuWindow::new(
&engine.instance,
engine.raw_physical_device(),
&engine.device,
window.as_ref(),
window.as_ref(),
width,
height,
3,
)
.context("[Enki Core] Failed to create GpuWindow context")?;
let enki = Arc::new(Self {
window_context: Some(window as Arc<dyn std::any::Any + Send + Sync>),
gpu_window: Some(Mutex::new(gpu_window)),
engine,
});
set_active_enki(enki.clone());
Ok(enki)
}
/// Returns an `Arc` clone of the currently active global Enki runtime instance.
pub fn active() -> Arc<Self> {
active_enki()
}
pub fn begin_flow(&self) -> Flow<'_> {
Flow::new(self)
}
/// Records and executes an atomic compute and presentation flow, halting on execution error.
#[track_caller]
#[inline(always)]
pub fn flow<R, F>(&self, f: F) -> R
where
F: FnOnce(&mut Flow<'_>) -> R,
{
match self.try_flow(f) {
Ok(output) => output,
Err(e) => handle_execution_error(&e),
}
}
/// Records and executes an atomic flow, returning an explicit `Result` on failure.
#[track_caller]
pub fn try_flow<R, F>(&self, f: F) -> Result<R>
where
F: FnOnce(&mut Flow<'_>) -> R,
{
let mut flow = self.begin_flow();
let output = {
let _guard = ActiveFlowGuard::enter(&mut flow);
f(&mut flow)
};
if let Some(err) = flow.sticky_error.take() {
return Err(err);
}
flow.try_end_flow()?;
Ok(output)
}
/// Resizes the underlying presentation window swapchain dimensions.
pub fn resize(&self, width: u32, height: u32) -> Result<()> {
if let Some(window_mutex) = &self.gpu_window {
let mut window = window_mutex.lock().unwrap();
window
.recreate(self.engine.raw_physical_device(), width, height)
.context("[Enki Core] Swapchain recreation failed")?;
}
Ok(())
}
/// Blocks the host CPU until all in-flight GPU timeline tasks are completely idle.
pub fn wait_idle(&self) -> Result<()> {
self.engine.wait_idle()
}
}
</file>
<file path="enki_api/context/instant.rs">
use std::ops::Sub;
use std::time::Duration;
use crate::enki_api::context::active_engine;
use crate::enki_api::context::ambient::{is_inside_active_flow, resolve_user_caller};
use crate::enki_api::context::errors::emit_and_abort;
/// A hardware-accurate timestamp query recorded directly on the GPU timeline.
#[derive(Debug, Clone, Copy, PartialEq)]
pub struct GpuInstant {
pub(crate) query_slot: u32,
pub(crate) timeline_value: u64,
pub(crate) timestamp_period: f32,
}
impl GpuInstant {
#[track_caller]
fn assert_cpu_readable(&self, operation: &'static str) {
if is_inside_active_flow() {
let raw_caller = std::panic::Location::caller();
let caller = resolve_user_caller(raw_caller);
let diag = anu::diagnostics::rt::phase_violation(operation, caller);
emit_and_abort(&diag);
}
}
/// Computes the elapsed duration between this timestamp and an earlier GPU instant.
///
/// # Diagnostics
/// Halts execution if called inside an active `flow` before the GPU has retired the query.
#[track_caller]
pub fn duration_since(&self, earlier: &GpuInstant) -> Duration {
self.assert_cpu_readable("duration_since on `GpuInstant`");
let engine = active_engine();
let max_timeline = self.timeline_value.max(earlier.timeline_value);
let current_gpu = engine.timeline_semaphore.get_timeline_value().unwrap_or(0);
if current_gpu < max_timeline {
engine
.timeline_semaphore
.wait_timeline(max_timeline, Duration::from_secs(5))
.expect(
"[GpuInstant] Timeout waiting for GPU to finish work before reading timestamps",
);
}
let ticks_end = engine
.get_timestamp_query_result(self.query_slot)
.expect("[GpuInstant] Failed to read end timestamp query");
let ticks_start = engine
.get_timestamp_query_result(earlier.query_slot)
.expect("[GpuInstant] Failed to read start timestamp query");
let delta_ticks = ticks_end.saturating_sub(ticks_start);
let nanos = (delta_ticks as f64 * self.timestamp_period as f64) as u64;
Duration::from_nanos(nanos)
}
/// Computes the elapsed duration from this instant until a later GPU instant.
#[inline(always)]
#[track_caller]
pub fn elapsed_until(&self, later: &GpuInstant) -> Duration {
later.duration_since(self)
}
}
impl Sub for GpuInstant {
type Output = Duration;
#[inline(always)]
#[track_caller]
fn sub(self, other: GpuInstant) -> Duration {
self.duration_since(&other)
}
}
impl Sub<&GpuInstant> for &GpuInstant {
type Output = Duration;
#[inline(always)]
#[track_caller]
fn sub(self, other: &GpuInstant) -> Duration {
self.duration_since(other)
}
}
</file>
<file path="enki_api/context/mod.rs">
pub(crate) mod ambient;
pub(crate) mod contract;
pub(crate) mod errors;
pub mod flow;
pub mod instance;
pub mod instant;
// Public Facade Exports
pub use flow::Flow;
pub use instance::{Enki, EnkiBuilder, active_engine, active_enki, set_active_enki};
pub use instant::GpuInstant;
</file>
<file path="enki_api/resources/slice/dispatch.rs">
use super::ro::Slice;
use super::rw::SliceMut;
use crate::enki_api::type_safety::GpuTypeMatch;
use anu::nam_args_api::{
AccessIntent, ArgDescriptor, ArgValue, GpuBufferTarget, GpuType, IngressContext,
InputResourceRecord, ResourceAccessKind,
};
impl<T: Copy + Send + Sync + 'static> GpuType for &Slice<'_, T> {
fn describe() -> ArgDescriptor {
ArgDescriptor::global_slice_read::<T>(false)
}
fn collect<'a>(&self, ctx: &mut IngressContext<'a>) {
let desc = <Self as GpuType>::describe();
ctx.push_arg(
ArgValue::SliceBDA {
bda: self.device_address,
count: self.len as u64,
},
desc,
);
ctx.push_resource_binding(
format!("Slice_Slot_{}", self.slot_index),
self.slot_index,
self.state.clone(),
AccessIntent::Read,
);
}
}
impl<T: Copy + Send + Sync + 'static> GpuType for &mut SliceMut<'_, T> {
fn describe() -> ArgDescriptor {
ArgDescriptor::global_slice_read_write::<T>(false)
}
fn collect<'a>(&self, ctx: &mut IngressContext<'a>) {
let desc = <Self as GpuType>::describe();
ctx.push_arg(
ArgValue::SliceBDA {
bda: self.device_address,
count: self.len as u64,
},
desc,
);
ctx.push_resource_binding(
format!("SliceMut_Slot_{}", self.slot_index),
self.slot_index,
self.state.clone(),
AccessIntent::ReadWrite,
);
}
}
impl<T: Copy + Send + Sync + 'static> GpuBufferTarget<T> for Slice<'_, T> {}
impl<T: Copy + Send + Sync + 'static> GpuBufferTarget<T> for SliceMut<'_, T> {}
impl<'target, T: Copy + Send + Sync + 'static> GpuTypeMatch<'target> for &'target Slice<'_, T> {
type Target = &'target [T];
fn describe() -> ArgDescriptor {
ArgDescriptor::global_slice_read::<T>(false)
}
fn collect<'a>(&self, ctx: &mut IngressContext<'a>) {
let desc = <Self as GpuTypeMatch>::describe();
ctx.push_arg(
ArgValue::SliceBDA {
bda: self.device_address(),
count: self.len as u64,
},
desc,
);
ctx.push_resource_binding(
format!("Slice_Slot_{}", self.slot_index),
self.slot_index,
self.state.clone(),
AccessIntent::Read,
);
}
fn as_input_record(&self, arg_index: usize) -> InputResourceRecord {
InputResourceRecord {
arg_index,
root_id: Some(self.root_id()),
access_kind: ResourceAccessKind::Slice {
element_range: self.element_range(),
total_container_len: self.len,
},
is_mutable: false,
element_type_name: std::any::type_name::<T>(),
}
}
}
impl<'target, T: Copy + Send + Sync + 'static> GpuTypeMatch<'target>
for &'target mut SliceMut<'_, T>
{
type Target = &'target mut [T];
fn describe() -> ArgDescriptor {
ArgDescriptor::global_slice_read_write::<T>(false)
}
fn collect<'a>(&self, ctx: &mut IngressContext<'a>) {
let desc = <Self as GpuTypeMatch>::describe();
ctx.push_arg(
ArgValue::SliceBDA {
bda: self.device_address(),
count: self.len as u64,
},
desc,
);
ctx.push_resource_binding(
format!("SliceMut_Slot_{}", self.slot_index),
self.slot_index,
self.state.clone(),
AccessIntent::ReadWrite,
);
}
fn as_input_record(&self, arg_index: usize) -> InputResourceRecord {
InputResourceRecord {
arg_index,
root_id: Some(self.root_id()),
access_kind: ResourceAccessKind::Slice {
element_range: self.element_range(),
total_container_len: self.len,
},
is_mutable: true,
element_type_name: std::any::type_name::<T>(),
}
}
}
impl<'target, T: Copy + Send + Sync + 'static> GpuTypeMatch<'target> for &'target SliceMut<'_, T> {
type Target = &'target [T];
fn describe() -> ArgDescriptor {
ArgDescriptor::global_slice_read::<T>(false)
}
fn collect<'a>(&self, ctx: &mut IngressContext<'a>) {
let desc = <Self as GpuTypeMatch>::describe();
ctx.push_arg(
ArgValue::SliceBDA {
bda: self.device_address(),
count: self.len as u64,
},
desc,
);
ctx.push_resource_binding(
format!("SliceMut_Slot_{}", self.slot_index),
self.slot_index,
self.state.clone(),
AccessIntent::Read,
);
}
fn as_input_record(&self, arg_index: usize) -> InputResourceRecord {
InputResourceRecord {
arg_index,
root_id: Some(self.root_id()),
access_kind: ResourceAccessKind::Slice {
element_range: self.element_range(),
total_container_len: self.len,
},
is_mutable: false,
element_type_name: std::any::type_name::<T>(),
}
}
}
</file>
<file path="enki_api/resources/slice/format.rs">
use super::ro::Slice;
use super::rw::SliceMut;
use crate::enki_api::context::ambient::{is_inside_active_flow, resolve_user_caller};
use crate::enki_api::context::errors::emit_and_abort;
use std::fmt;
impl<'a, T: Copy + Send + Sync + 'static> Slice<'a, T> {
#[track_caller]
fn assert_host_printable(&self, operation: &'static str) {
if is_inside_active_flow() {
let raw_caller = std::panic::Location::caller();
let caller = resolve_user_caller(raw_caller);
let diag = anu::diagnostics::rt::phase_violation(operation, caller);
emit_and_abort(&diag);
}
}
#[track_caller]
fn format_sample<F>(&self, f: &mut fmt::Formatter<'_>, mut format_elem: F) -> fmt::Result
where
F: FnMut(&T, &mut fmt::Formatter<'_>) -> fmt::Result,
{
if self.len == 0 {
return write!(f, "[]");
}
self.assert_host_printable("printing of `Slice`");
const MAX_FULL_DISPLAY: usize = 10_000;
const EDGE_SAMPLE_COUNT: usize = 64;
if self.len <= MAX_FULL_DISPLAY {
let data = self.to_vec();
write!(f, "[")?;
for (i, item) in data.iter().enumerate() {
if i > 0 {
write!(f, ", ")?;
}
format_elem(item, f)?;
}
write!(f, "]")
} else {
write!(f, "[")?;
for i in 0..EDGE_SAMPLE_COUNT {
if let Some(item) = self.get(i) {
if i > 0 {
write!(f, ", ")?;
}
format_elem(&item, f)?;
}
}
let omitted = self.len - (EDGE_SAMPLE_COUNT * 2);
write!(f, ", ... ({} elements omitted) ..., ", omitted)?;
for i in (self.len - EDGE_SAMPLE_COUNT)..self.len {
if let Some(item) = self.get(i) {
if i > (self.len - EDGE_SAMPLE_COUNT) {
write!(f, ", ")?;
}
format_elem(&item, f)?;
}
}
write!(f, "]")
}
}
}
impl<'a, T: Copy + Send + Sync + fmt::Debug + 'static> fmt::Debug for Slice<'a, T> {
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
self.assert_host_printable("debug formatting of `Slice`");
if f.alternate() {
struct DataPreview<'a, 'b, T: Copy + Send + Sync + fmt::Debug + 'static>(
&'b Slice<'a, T>,
);
impl<'a, 'b, T: Copy + Send + Sync + fmt::Debug + 'static> fmt::Debug for DataPreview<'a, 'b, T> {
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
self.0
.format_sample(f, |item, formatter| write!(formatter, "{:?}", item))
}
}
f.debug_struct("Slice")
.field("root_id", &self.root_id())
.field("element_range", &self.element_range())
.field("len", &self.len)
.field("stride", &self.stride)
.field("data", &DataPreview(self))
.finish()
} else {
self.format_sample(f, |item, formatter| write!(formatter, "{:?}", item))
}
}
}
impl<'a, T: Copy + Send + Sync + fmt::Debug + 'static> fmt::Debug for SliceMut<'a, T> {
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
self.as_slice()
.assert_host_printable("debug formatting of `SliceMut`");
if f.alternate() {
struct DataPreview<'a, 'b, T: Copy + Send + Sync + fmt::Debug + 'static>(
&'b SliceMut<'a, T>,
);
impl<'a, 'b, T: Copy + Send + Sync + fmt::Debug + 'static> fmt::Debug for DataPreview<'a, 'b, T> {
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
self.0
.as_slice()
.format_sample(f, |item, formatter| write!(formatter, "{:?}", item))
}
}
f.debug_struct("SliceMut")
.field("root_id", &self.root_id())
.field("element_range", &self.element_range())
.field("len", &self.len)
.field("stride", &self.stride)
.field("data", &DataPreview(self))
.finish()
} else {
self.as_slice()
.format_sample(f, |item, formatter| write!(formatter, "{:?}", item))
}
}
}
</file>
<file path="enki_api/resources/slice/mod.rs">
//! Module: enki_api/resources/slice/mod.rs
//!
//! Zero-allocation GPU Slicing system.
//! Separates safe read-only slices (`Slice`) from free-indexing mutable slices (`SliceMut`).
pub mod dispatch;
pub mod format;
pub(crate) mod range;
pub mod ro;
pub mod rw;
pub use self::ro::Slice;
pub use self::rw::SliceMut;
</file>
<file path="enki_api/resources/slice/range.rs">
use std::ops::{Bound, RangeBounds};
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub struct ResolvedRange {
pub start: usize,
pub count: usize,
pub end: usize,
}
impl ResolvedRange {
#[track_caller]
pub(crate) fn resolve<R: RangeBounds<usize>>(
range: R,
total_len: usize,
caller: &'static std::panic::Location<'static>,
) -> Self {
let start = match range.start_bound() {
Bound::Included(&s) => s,
Bound::Excluded(&s) => s
.checked_add(1)
.expect("[Slice Range] Start bound integer overflow"),
Bound::Unbounded => 0,
};
let end = match range.end_bound() {
Bound::Included(&e) => e
.checked_add(1)
.expect("[Slice Range] End bound integer overflow"),
Bound::Excluded(&e) => e,
Bound::Unbounded => total_len,
};
if start > end || end > total_len {
let diag = anu::diagnostics::rt::slice_out_of_bounds(caller);
let formatted = anu::diagnostics::emit_diagnostic(&diag);
eprint!("{formatted}");
let bin_name = std::env::current_exe()
.ok()
.and_then(|p| p.file_name().map(|n| n.to_string_lossy().to_string()))
.unwrap_or_else(|| "enki_app".to_string());
eprintln!(
"\x1b[1;91merror\x1b[0m: aborting execution of `{bin_name}` due to 1 previous error\n"
);
std::process::exit(101);
}
let count = end - start;
Self { start, count, end }
}
}
</file>
<file path="enki_api/resources/slice/ro.rs">
use std::marker::PhantomData;
use std::mem::MaybeUninit;
use std::ops::{Range, RangeBounds};
use std::sync::Arc;
use std::sync::atomic::AtomicU8;
use super::range::ResolvedRange;
use crate::enki_api::context::active_engine;
use crate::enki_api::context::ambient::is_inside_active_flow;
use crate::enki_api::resources::GpuVec;
use apsu::GpuDeviceBuffer;
/// An immutable, zero-allocation borrowed view into a contiguous sub-range of GPU memory.
///
/// Unlike [`GpuVec`], a `Slice` does not own or allocate physical VRAM; it represents
/// a lightweight view window with a 64-bit Buffer Device Address (BDA) and element bounds.
///
/// # Nam Dispatch Semantics
/// Passing `&Slice<T>` to a `#[nam]` function binds as a global read-only slice **`&[T]`**,
/// granting parallel GPU threads indexed read access across the entire slice bounds.
///
/// Unlike mutable slices, multiple immutable `Slice` views from the same parent buffer
/// are explicitly permitted to overlap during kernel dispatches.
pub struct Slice<'a, T: Copy + Send + Sync + 'static> {
pub(crate) slot_index: u32,
pub(crate) offset: u32,
pub(crate) element_offset: usize,
pub(crate) stride: u32,
pub(crate) len: usize,
pub(crate) device_address: u64,
pub(crate) state: Arc<AtomicU8>,
pub(crate) _inner: Arc<GpuDeviceBuffer>,
pub(crate) _phantom: PhantomData<&'a T>,
}
impl<'a, T: Copy + Send + Sync + 'static> Clone for Slice<'a, T> {
fn clone(&self) -> Self {
Self {
slot_index: self.slot_index,
offset: self.offset,
element_offset: self.element_offset,
stride: self.stride,
len: self.len,
device_address: self.device_address,
state: self.state.clone(),
_inner: self._inner.clone(),
_phantom: PhantomData,
}
}
}
impl<'a, T: Copy + Send + Sync + 'static> Slice<'a, T> {
/// Returns the number of elements inside this slice.
#[inline(always)]
pub const fn len(&self) -> usize {
self.len
}
/// Returns `true` if the slice contains no elements.
#[inline(always)]
pub const fn is_empty(&self) -> bool {
self.len == 0
}
/// Returns the unique identity of the underlying parent buffer allocation in VRAM.
#[inline(always)]
pub const fn root_id(&self) -> usize {
self.slot_index as usize
}
/// Returns the concrete element range `[start..end)` relative to the parent buffer.
#[inline(always)]
pub fn element_range(&self) -> Range<usize> {
self.element_offset..(self.element_offset + self.len)
}
/// Returns the 64-bit GPU Buffer Device Address (BDA) pointing to the start of this sub-range.
#[inline(always)]
pub const fn device_address(&self) -> u64 {
self.device_address
}
/// Returns the internal engine slot index of the underlying physical buffer.
#[inline(always)]
pub const fn slot_index(&self) -> u32 {
self.slot_index
}
/// Returns the size in bytes of a single element `T`.
#[inline(always)]
pub const fn stride(&self) -> usize {
self.stride as usize
}
/// Sub-slices this view into a smaller contiguous sub-range.
///
/// # Diagnostics
/// Halts execution with diagnostic `error[E2005]` if `range` exceeds the slice bounds.
#[track_caller]
pub fn slice<R: RangeBounds<usize>>(&self, range: R) -> Slice<'a, T> {
let caller = std::panic::Location::caller();
let resolved = ResolvedRange::resolve(range, self.len, caller);
let byte_offset = (resolved.start * self.stride as usize) as u64;
Slice {
slot_index: self.slot_index,
offset: self.offset + byte_offset as u32,
element_offset: self.element_offset + resolved.start,
stride: self.stride,
len: resolved.count,
device_address: self.device_address + byte_offset,
state: self.state.clone(),
_inner: self._inner.clone(),
_phantom: PhantomData,
}
}
/// Splits the slice into two disjoint sub-slices at index `mid`.
///
/// # Panics
/// Panics if `mid > self.len()`.
pub fn split_at(&self, mid: usize) -> (Slice<'a, T>, Slice<'a, T>) {
(self.slice(..mid), self.slice(mid..))
}
/// Reads back a single element from this GPU slice to the host CPU at the specified index.
///
/// Automatically synchronizes with the GPU timeline if the underlying buffer is currently in-flight.
///
/// # Diagnostics
/// Halts execution with diagnostic `error[E2001]` if called inside an active `flow`.
#[track_caller]
pub fn get(&self, index: usize) -> Option<T> {
if index >= self.len {
return None;
}
if is_inside_active_flow() {
panic!(
"\n\x1b[1;31merror[E0701]\x1b[0m\x1b[1m: illegal host read from `Slice` during GPU command recording phase\x1b[0m\n\
\x1b[1;34m --> \x1b[0mSlice (Slot: {})\n\
\x1b[1;34m = help\x1b[0m: Move your readback operation OUTSIDE the `enki.frame` closure.\n",
self.slot_index
);
}
let engine = active_engine();
let target_timeline = engine
.timeline_counter
.load(std::sync::atomic::Ordering::SeqCst);
let current_gpu = engine.timeline_semaphore.get_timeline_value().unwrap_or(0);
if current_gpu < target_timeline {
engine
.timeline_semaphore
.wait_timeline(target_timeline, std::time::Duration::from_secs(5))
.expect("[Slice Transfer] Timeline wait timed out before reading element");
}
let element_size = self.stride as usize;
let byte_offset = self.offset as u64 + (index * element_size) as u64;
let mut out_val = MaybeUninit::<T>::uninit();
let byte_slice: &mut [u8] = unsafe {
std::slice::from_raw_parts_mut(out_val.as_mut_ptr() as *mut u8, element_size)
};
engine
.transfer_manager
.read_buffer(&self._inner, byte_offset, byte_slice)
.expect("[Slice Transfer] Failed to read element from GPU VRAM");
unsafe { Some(out_val.assume_init()) }
}
/// Reads all elements in this GPU slice back to a newly allocated host `Vec<T>`.
///
/// # Diagnostics
/// Halts execution with diagnostic `error[E2001]` if called inside an active `flow`.
#[track_caller]
pub fn to_vec(&self) -> Vec<T> {
if self.len == 0 {
return Vec::new();
}
if is_inside_active_flow() {
panic!(
"\n\x1b[1;31merror[E0701]\x1b[0m\x1b[1m: illegal host `to_vec()` on `Slice` during GPU command recording phase\x1b[0m\n\
\x1b[1;34m --> \x1b[0mSlice (Slot: {})\n\
\x1b[1;34m = help\x1b[0m: Move your readback operation OUTSIDE the `enki.frame` closure.\n",
self.slot_index
);
}
let engine = active_engine();
let target_timeline = engine
.timeline_counter
.load(std::sync::atomic::Ordering::SeqCst);
let current_gpu = engine.timeline_semaphore.get_timeline_value().unwrap_or(0);
if current_gpu < target_timeline {
engine
.timeline_semaphore
.wait_timeline(target_timeline, std::time::Duration::from_secs(10))
.expect("[Slice Transfer] Timeline wait timed out before reading slice");
}
let element_size = self.stride as usize;
let total_bytes = self.len * element_size;
let mut uninit_vec: Vec<MaybeUninit<T>> = Vec::with_capacity(self.len);
unsafe {
uninit_vec.set_len(self.len);
}
let byte_slice: &mut [u8] = unsafe {
std::slice::from_raw_parts_mut(uninit_vec.as_mut_ptr() as *mut u8, total_bytes)
};
engine
.transfer_manager
.read_buffer(&self._inner, self.offset as u64, byte_slice)
.expect("[Slice Transfer] Failed to read slice bytes from GPU VRAM");
unsafe {
let mut manual_vec = std::mem::ManuallyDrop::new(uninit_vec);
Vec::from_raw_parts(
manual_vec.as_mut_ptr() as *mut T,
manual_vec.len(),
manual_vec.capacity(),
)
}
}
/// Clones the contents of this slice into an independent, physical [`GpuVec<T>`] in VRAM.
///
/// Matches `std::slice::to_vec` but allocates physical VRAM instead of host memory.
///
/// # Diagnostics
/// Halts execution with diagnostic `error[E2001]` if called inside an active `flow`.
pub fn to_gpu_vec(&self) -> GpuVec<T> {
let cpu_data = self.to_vec();
GpuVec::from_slice(&cpu_data)
}
}
</file>
<file path="enki_api/resources/slice/rw.rs">
use std::marker::PhantomData;
use std::ops::{Range, RangeBounds};
use std::sync::Arc;
use std::sync::atomic::AtomicU8;
use crate::enki_api::context::active_engine;
use crate::enki_api::context::ambient::{is_inside_active_flow, resolve_user_caller};
use crate::enki_api::context::errors::emit_and_abort;
use super::range::ResolvedRange;
use super::ro::Slice;
use crate::enki_api::resources::GpuVec;
use apsu::GpuDeviceBuffer;
/// An exclusive, mutable borrowed view into a contiguous sub-range of GPU memory.
///
/// Enforces Rust's exclusive reference semantics over a VRAM sub-range.
/// Does not implement [`Clone`] to guarantee single-writer exclusivity.
///
/// # Nam Dispatch & Safety Policies
/// Passing `&mut SliceMut<T>` to a `#[nam]` function binds as a global read-write slice **`&mut [T]`**.
///
/// - **Parallel Safe Mode (`.run()`):** If the execution domain has more than one thread
/// (`size_x * size_y * size_z > 1`), passing a mutable slice is strictly forbidden and halts
/// execution with diagnostic **`error[E1009]`** because unconstrained concurrent writes
/// cannot be statically proven race-free.
/// - **Unchecked Dispatch (`.run_unchecked()`):** Required to pass `SliceMut` across parallel
/// threads when writes are manually coordinated (e.g., via atomic slot indexing or disjoint offsets).
/// - **Spatial Disjointness:** In all dispatch modes, if multiple slices from the same root
/// buffer are passed where at least one is mutable, any overlapping range halts execution
/// with diagnostic **`error[E1007]`**.
pub struct SliceMut<'a, T: Copy + Send + Sync + 'static> {
pub(crate) slot_index: u32,
pub(crate) offset: u32,
pub(crate) element_offset: usize,
pub(crate) stride: u32,
pub(crate) len: usize,
pub(crate) device_address: u64,
pub(crate) state: Arc<AtomicU8>,
pub(crate) _inner: Arc<GpuDeviceBuffer>,
pub(crate) _phantom: PhantomData<&'a mut T>,
}
impl<'a, T: Copy + Send + Sync + 'static> SliceMut<'a, T> {
/// Returns the number of elements inside this slice.
#[inline(always)]
pub const fn len(&self) -> usize {
self.len
}
/// Returns `true` if the slice contains no elements.
#[inline(always)]
pub const fn is_empty(&self) -> bool {
self.len == 0
}
/// Returns the unique identity of the underlying parent buffer allocation in VRAM.
#[inline(always)]
pub const fn root_id(&self) -> usize {
self.slot_index as usize
}
/// Returns the concrete element range `[start..end)` relative to the parent buffer.
#[inline(always)]
pub fn element_range(&self) -> Range<usize> {
self.element_offset..(self.element_offset + self.len)
}
/// Returns the 64-bit GPU Buffer Device Address (BDA) pointing to the start of this sub-range.
#[inline(always)]
pub const fn device_address(&self) -> u64 {
self.device_address
}
/// Returns the internal engine slot index of the underlying physical buffer.
#[inline(always)]
pub const fn slot_index(&self) -> u32 {
self.slot_index
}
/// Returns the size in bytes of a single element `T`.
#[inline(always)]
pub const fn stride(&self) -> usize {
self.stride as usize
}
#[track_caller]
fn assert_host_writable(&self, operation: &'static str) {
if is_inside_active_flow() {
let raw_caller = std::panic::Location::caller();
let caller = resolve_user_caller(raw_caller);
let diag = anu::diagnostics::rt::phase_violation(operation, caller);
emit_and_abort(&diag);
}
}
/// Splits this mutable slice into two disjoint mutable sub-slices at index `mid`.
///
/// Guarantees that the two returned slices are mathematically non-overlapping.
///
/// # Panics
/// Panics if `mid > self.len()`.
pub fn split_at_mut(&mut self, mid: usize) -> (SliceMut<'_, T>, SliceMut<'_, T>) {
assert!(
mid <= self.len,
"\n\x1b[1;31m[Enki Slice Error]\x1b[0m: `split_at_mut` index {} out of bounds for slice of len {}\n",
mid,
self.len
);
let byte_split = (mid * self.stride as usize) as u64;
let left = SliceMut {
slot_index: self.slot_index,
offset: self.offset,
element_offset: self.element_offset,
stride: self.stride,
len: mid,
device_address: self.device_address,
state: self.state.clone(),
_inner: self._inner.clone(),
_phantom: PhantomData,
};
let right = SliceMut {
slot_index: self.slot_index,
offset: self.offset + byte_split as u32,
element_offset: self.element_offset + mid,
stride: self.stride,
len: self.len - mid,
device_address: self.device_address + byte_split,
state: self.state.clone(),
_inner: self._inner.clone(),
_phantom: PhantomData,
};
(left, right)
}
/// Reborrows a smaller contiguous mutable sub-range from this slice.
///
/// # Diagnostics
/// Halts execution with diagnostic **`error[E2005]`** if `range` exceeds the slice bounds.
#[track_caller]
pub fn slice_mut<R: RangeBounds<usize>>(&mut self, range: R) -> SliceMut<'_, T> {
let caller = std::panic::Location::caller();
let resolved = ResolvedRange::resolve(range, self.len, caller);
let byte_offset = (resolved.start * self.stride as usize) as u64;
SliceMut {
slot_index: self.slot_index,
offset: self.offset + byte_offset as u32,
element_offset: self.element_offset + resolved.start,
stride: self.stride,
len: resolved.count,
device_address: self.device_address + byte_offset,
state: self.state.clone(),
_inner: self._inner.clone(),
_phantom: PhantomData,
}
}
/// Reborrows this mutable slice as an immutable, read-only [`Slice`].
#[inline(always)]
pub fn as_slice(&self) -> Slice<'_, T> {
Slice {
slot_index: self.slot_index,
offset: self.offset,
element_offset: self.element_offset,
stride: self.stride,
len: self.len,
device_address: self.device_address,
state: self.state.clone(),
_inner: self._inner.clone(),
_phantom: PhantomData,
}
}
/// Consumes this mutable slice and converts it into an immutable [`Slice`].
#[inline(always)]
pub fn into_slice(self) -> Slice<'a, T> {
Slice {
slot_index: self.slot_index,
offset: self.offset,
element_offset: self.element_offset,
stride: self.stride,
len: self.len,
device_address: self.device_address,
state: self.state.clone(),
_inner: self._inner,
_phantom: PhantomData,
}
}
/// Sets a single value from the host CPU into the GPU slice at the specified index.
///
/// # Diagnostics
/// Halts execution with diagnostic `error[E2001]` if called inside an active `flow`.
#[track_caller]
pub fn set(&mut self, index: usize, value: T) -> bool {
if index >= self.len {
return false;
}
self.copy_from_slice_at(index, std::slice::from_ref(&value));
true
}
/// Overwrites the entire contents of this GPU slice with data from a host CPU slice.
///
/// # Panics
/// Panics if `self.len() != data.len()`.
///
/// # Diagnostics
/// Halts execution with diagnostic `error[E2001]` if called inside an active `flow`.
#[track_caller]
pub fn copy_from_slice(&mut self, data: &[T]) {
assert_eq!(
self.len,
data.len(),
"[SliceMut Transfer] `copy_from_slice` length mismatch: destination has len {} but source has len {}",
self.len,
data.len()
);
self.copy_from_slice_at(0, data);
}
/// Overwrites a sub-range of this GPU slice starting at element `offset` with host data.
///
/// # Panics
/// Panics if `offset + data.len() > self.len()`.
///
/// # Diagnostics
/// Halts execution with diagnostic `error[E2001]` if called inside an active `flow`.
#[track_caller]
pub fn copy_from_slice_at(&mut self, offset: usize, data: &[T]) {
assert!(
offset + data.len() <= self.len,
"[SliceMut Transfer] `copy_from_slice_at` out of bounds: offset {} + len {} exceeds slice len {}",
offset,
data.len(),
self.len
);
self.assert_host_writable("write to `SliceMut`");
if data.is_empty() {
return;
}
let engine = active_engine();
let target_timeline = engine
.timeline_counter
.load(std::sync::atomic::Ordering::SeqCst);
let current_gpu = engine.timeline_semaphore.get_timeline_value().unwrap_or(0);
if current_gpu < target_timeline {
engine
.timeline_semaphore
.wait_timeline(target_timeline, std::time::Duration::from_secs(5))
.expect("[SliceMut Transfer] Timeline wait timed out before writing");
}
let element_size = self.stride as usize;
let byte_offset = self.offset as u64 + (offset * element_size) as u64;
let size_in_bytes = data.len() * element_size;
let byte_data: &[u8] =
unsafe { std::slice::from_raw_parts(data.as_ptr() as *const u8, size_in_bytes) };
engine
.transfer_manager
.write_buffer(&self._inner, byte_offset, byte_data)
.expect("[SliceMut Transfer] Failed to write slice into GPU VRAM");
}
/// Reads all elements in this GPU slice back to a newly allocated host `Vec<T>`.
///
/// # Diagnostics
/// Halts execution with diagnostic `error[E2001]` if called inside an active `flow`.
#[track_caller]
pub fn to_vec(&self) -> Vec<T> {
self.as_slice().to_vec()
}
/// Clones the contents of this slice into an independent, physical [`GpuVec<T>`] in VRAM.
///
/// # Diagnostics
/// Halts execution with diagnostic `error[E2001]` if called inside an active `flow`.
pub fn to_gpu_vec(&self) -> GpuVec<T> {
self.as_slice().to_gpu_vec()
}
}
</file>
<file path="enki_api/resources/vec/core.rs">
use std::marker::PhantomData;
use std::sync::Arc;
use std::sync::atomic::AtomicU8;
use crate::enki_api::context::active_engine;
use apsu::{BufferUsage, GpuDeviceBuffer};
/// A contiguous, growable array allocated in GPU VRAM with exclusive ownership semantics.
///
/// `GpuVec<T>` mirrors standard Rust `Vec<T>` ergonomics while residing entirely
/// in GPU memory via 64-bit Buffer Device Addresses (BDA).
///
/// # Behavior & Invariants
/// - **Exclusive Ownership:** Follows Rust's move semantics. Calling `clone()` performs
/// a physical VRAM-to-VRAM deep copy with a distinct memory range.
/// - **SPMD Kernel Dispatch:** Passing `&GpuVec<T>` or `&mut GpuVec<T>` to a `#[nam]`
/// binds per-thread as `&T` or `&mut T` (1:1 cell access). The vector's `len()` must
/// be greater than or equal to the total execution domain (`space.size_x * space.size_y * space.size_z`),
/// otherwise dispatch halts with diagnostic **`error[E1008]`** (`SpaceDomainOverflow`).
/// - **Zero-Sized Types:** Zero-sized types (ZST) are strictly forbidden on GPU hardware.
pub struct GpuVec<T: Copy + Send + Sync + 'static> {
pub(crate) slot_index: u32,
pub(crate) offset: u32,
pub(crate) stride: u32,
pub(crate) len: usize,
pub(crate) capacity: usize,
pub(crate) device_address: u64,
pub(crate) size_in_bytes: u64,
pub(crate) state: Arc<AtomicU8>,
pub(crate) _inner: Arc<GpuDeviceBuffer>,
pub(crate) _phantom: PhantomData<T>,
}
impl<T: Copy + Send + Sync + 'static> GpuVec<T> {
pub(crate) fn allocate_raw(capacity: usize) -> (Arc<GpuDeviceBuffer>, u64, u32, u64) {
let element_size = std::mem::size_of::<T>();
assert!(
element_size > 0,
"[GpuVec] Zero-Sized Types (ZST) are not supported on GPU hardware!"
);
let engine = active_engine();
let allocator = engine.allocator.clone();
let usage = BufferUsage::STORAGE_BUFFER
| BufferUsage::SHADER_DEVICE_ADDRESS
| BufferUsage::TRANSFER_DST
| BufferUsage::TRANSFER_SRC;
let alloc_capacity = capacity.max(1);
let size_in_bytes = (alloc_capacity * element_size) as u64;
let device_buffer = match GpuDeviceBuffer::new(allocator, size_in_bytes, usage) {
Ok(buf) => buf,
Err(_) => {
let diag = anu::diagnostics::hw::out_of_vram(None);
crate::enki_api::context::errors::emit_and_abort(&diag);
}
};
let device_address = device_buffer.device_address();
let slot_index = device_buffer.id;
(
Arc::new(device_buffer),
device_address,
slot_index,
size_in_bytes,
)
}
/// Constructs a new, empty `GpuVec<T>` without pre-allocated VRAM capacity.
#[inline(always)]
pub fn new() -> Self {
Self::with_capacity(0)
}
/// Creates an empty `GpuVec<T>` with pre-allocated VRAM capacity for `capacity` elements.
pub fn with_capacity(capacity: usize) -> Self {
let element_size = std::mem::size_of::<T>() as u32;
let (buffer, device_address, slot_index, size_in_bytes) = Self::allocate_raw(capacity);
Self {
slot_index,
offset: 0,
stride: element_size,
len: 0,
capacity,
device_address,
size_in_bytes,
state: buffer.state.clone(),
_inner: buffer,
_phantom: PhantomData,
}
}
/// Creates a new `GpuVec<T>` initialized with data copied from a host CPU slice.
pub fn from_slice(data: &[T]) -> Self {
let count = data.len();
let element_size = std::mem::size_of::<T>();
assert!(
element_size > 0,
"[GpuVec] Zero-Sized Types (ZST) are not supported on GPU hardware."
);
let (buffer, device_address, slot_index, size_in_bytes) = Self::allocate_raw(count);
if count > 0 {
let engine = active_engine();
let bytes: &[u8] = unsafe {
std::slice::from_raw_parts(data.as_ptr() as *const u8, size_of_val(data))
};
engine
.transfer_manager
.write_buffer(&buffer, 0, bytes)
.expect("[GpuVec] Failed to upload initial data to GPU VRAM");
}
Self {
slot_index,
offset: 0,
stride: element_size as u32,
len: count,
capacity: count,
device_address,
size_in_bytes,
state: buffer.state.clone(),
_inner: buffer,
_phantom: PhantomData,
}
}
/// Creates a `GpuVec<T>` of size `count` with each element initialized to `value`.
pub fn from_elem(value: T, count: usize) -> Self {
let host_data = vec![value; count];
Self::from_slice(&host_data)
}
/// Creates a `GpuVec<T>` of size `count` initialized with zeroed VRAM memory.
pub fn zeroed(count: usize) -> Self {
let element_size = std::mem::size_of::<T>();
let (buffer, device_address, slot_index, size_in_bytes) = Self::allocate_raw(count);
if count > 0 {
let engine = active_engine();
let zero_bytes = vec![0u8; count * element_size];
engine
.transfer_manager
.write_buffer(&buffer, 0, &zero_bytes)
.expect("[GpuVec] Failed to zero-initialize GPU VRAM");
}
Self {
slot_index,
offset: 0,
stride: element_size as u32,
len: count,
capacity: count,
device_address,
size_in_bytes,
state: buffer.state.clone(),
_inner: buffer,
_phantom: PhantomData,
}
}
/// Returns the number of active elements in the vector.
#[inline(always)]
pub const fn len(&self) -> usize {
self.len
}
/// Returns the maximum number of elements the vector can hold without reallocating VRAM.
#[inline(always)]
pub const fn capacity(&self) -> usize {
self.capacity
}
/// Returns `true` if the vector contains no elements.
#[inline(always)]
pub const fn is_empty(&self) -> bool {
self.len == 0
}
/// Returns the size in bytes of a single element `T`.
#[inline(always)]
pub const fn stride(&self) -> usize {
self.stride as usize
}
/// Returns the 64-bit GPU Buffer Device Address (BDA) pointing directly to this VRAM allocation.
#[inline(always)]
pub const fn device_address(&self) -> u64 {
self.device_address
}
/// Returns the internal engine slot index used for synchronization and resource tracking.
#[inline(always)]
pub const fn slot_index(&self) -> u32 {
self.slot_index
}
/// Returns the total physical size allocated in VRAM in bytes.
#[inline(always)]
pub const fn size_in_bytes(&self) -> u64 {
self.size_in_bytes
}
}
impl<T: Copy + Send + Sync + 'static> Default for GpuVec<T> {
#[inline(always)]
fn default() -> Self {
Self::new()
}
}
</file>
<file path="enki_api/resources/vec/dispatch.rs">
use super::core::GpuVec;
use crate::enki_api::type_safety::GpuTypeMatch;
use anu::nam_args_api::{
AccessIntent, ArgDescriptor, ArgValue, GpuBufferTarget, GpuType, IngressContext,
InputResourceRecord, ResourceAccessKind,
};
impl<T: Copy + Send + Sync + 'static> GpuType for &GpuVec<T> {
fn describe() -> ArgDescriptor {
ArgDescriptor::spmd_cell_const::<T>(false)
}
fn collect<'a>(&self, ctx: &mut IngressContext<'a>) {
let desc = <Self as GpuType>::describe();
ctx.push_arg(ArgValue::BufferBDA(self.device_address), desc);
ctx.push_resource_binding(
format!("GpuVec_Slot_{}", self.slot_index),
self.slot_index,
self.state.clone(),
AccessIntent::Read,
);
}
}
impl<T: Copy + Send + Sync + 'static> GpuType for &mut GpuVec<T> {
fn describe() -> ArgDescriptor {
ArgDescriptor::spmd_cell_mut::<T>(false)
}
fn collect<'a>(&self, ctx: &mut IngressContext<'a>) {
let desc = <Self as GpuType>::describe();
ctx.push_arg(ArgValue::BufferBDA(self.device_address), desc);
ctx.push_resource_binding(
format!("GpuVec_Slot_{}", self.slot_index),
self.slot_index,
self.state.clone(),
AccessIntent::ReadWrite,
);
}
}
impl<T: Copy + Send + Sync + 'static> GpuBufferTarget<T> for GpuVec<T> {}
impl<'target, T: Copy + Send + Sync + 'static> GpuTypeMatch<'target> for &'target GpuVec<T> {
type Target = &'target T;
fn describe() -> ArgDescriptor {
ArgDescriptor::spmd_cell_const::<T>(false)
}
fn collect<'a>(&self, ctx: &mut IngressContext<'a>) {
let desc = <Self as GpuTypeMatch>::describe();
ctx.push_arg(ArgValue::BufferBDA(self.device_address()), desc);
ctx.push_resource_binding(
format!("GpuVec_Slot_{}", self.slot_index()),
self.slot_index(),
self.state.clone(),
AccessIntent::Read,
);
}
fn as_input_record(&self, arg_index: usize) -> InputResourceRecord {
InputResourceRecord {
arg_index,
root_id: Some(self.slot_index() as usize),
access_kind: ResourceAccessKind::PerCell {
element_count: self.len,
},
is_mutable: false,
element_type_name: std::any::type_name::<T>(),
}
}
}
impl<'target, T: Copy + Send + Sync + 'static> GpuTypeMatch<'target> for &'target mut GpuVec<T> {
type Target = &'target mut T;
fn describe() -> ArgDescriptor {
ArgDescriptor::spmd_cell_mut::<T>(false)
}
fn collect<'a>(&self, ctx: &mut IngressContext<'a>) {
let desc = <Self as GpuTypeMatch>::describe();
ctx.push_arg(ArgValue::BufferBDA(self.device_address()), desc);
ctx.push_resource_binding(
format!("GpuVec_Slot_{}", self.slot_index()),
self.slot_index(),
self.state.clone(),
AccessIntent::ReadWrite,
);
}
fn as_input_record(&self, arg_index: usize) -> InputResourceRecord {
InputResourceRecord {
arg_index,
root_id: Some(self.slot_index as usize),
access_kind: ResourceAccessKind::PerCell {
element_count: self.len,
},
is_mutable: true,
element_type_name: std::any::type_name::<T>(),
}
}
}
</file>
<file path="enki_api/resources/vec/format.rs">
use super::core::GpuVec;
use std::fmt;
impl<T: Copy + Send + Sync + 'static> GpuVec<T> {
fn format_sample<F>(&self, f: &mut fmt::Formatter<'_>, mut format_elem: F) -> fmt::Result
where
F: FnMut(&T, &mut fmt::Formatter<'_>) -> fmt::Result,
{
if self.len == 0 {
return write!(f, "[]");
}
const MAX_FULL_DISPLAY: usize = 10_000;
const EDGE_SAMPLE_COUNT: usize = 64;
if self.len <= MAX_FULL_DISPLAY {
let data = self.to_vec();
write!(f, "[")?;
for (i, item) in data.iter().enumerate() {
if i > 0 {
write!(f, ", ")?;
}
format_elem(item, f)?;
}
write!(f, "]")
} else {
write!(f, "[")?;
for i in 0..EDGE_SAMPLE_COUNT {
if let Some(item) = self.get(i) {
if i > 0 {
write!(f, ", ")?;
}
format_elem(&item, f)?;
}
}
let omitted = self.len - (EDGE_SAMPLE_COUNT * 2);
write!(f, ", ... ({} elements omitted) ..., ", omitted)?;
for i in (self.len - EDGE_SAMPLE_COUNT)..self.len {
if let Some(item) = self.get(i) {
if i > (self.len - EDGE_SAMPLE_COUNT) {
write!(f, ", ")?;
}
format_elem(&item, f)?;
}
}
write!(f, "]")
}
}
}
impl<T: Copy + Send + Sync + fmt::Debug + 'static> fmt::Debug for GpuVec<T> {
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
self.assert_host_readable("Debug");
if f.alternate() {
struct DataPreview<'a, T: Copy + Send + Sync + fmt::Debug + 'static>(&'a GpuVec<T>);
impl<'a, T: Copy + Send + Sync + fmt::Debug + 'static> fmt::Debug for DataPreview<'a, T> {
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
self.0
.format_sample(f, |item, formatter| write!(formatter, "{:?}", item))
}
}
f.debug_struct("GpuVec")
.field("slot", &self.slot_index)
.field("len", &self.len)
.field("capacity", &self.capacity)
.field(
"device_address",
&format_args!("0x{:X}", self.device_address),
)
.field("vram_size", &format_args!("{} B", self.size_in_bytes))
.field("data", &DataPreview(self))
.finish()
} else {
self.format_sample(f, |item, formatter| write!(formatter, "{:?}", item))
}
}
}
</file>
<file path="enki_api/resources/vec/lifecycle.rs">
use super::core::GpuVec;
use super::state::BufferPhase;
use crate::enki_api::context::active_engine;
use std::marker::PhantomData;
impl<T: Copy + Send + Sync + 'static> GpuVec<T> {
pub(crate) fn wait_idle_if_in_flight(&self) {
if let BufferPhase::InFlight { target_timeline } = self.current_phase() {
let engine = active_engine();
engine
.timeline_semaphore
.wait_timeline(target_timeline, std::time::Duration::from_secs(5))
.expect("[GpuVec Lifecycle] Timeout waiting for GPU to finish in-flight work");
}
}
pub(crate) fn reallocate(&mut self, new_capacity: usize) {
self.assert_host_readable("reallocate");
self.wait_idle_if_in_flight();
assert!(
new_capacity >= self.len,
"[GpuVec Lifecycle] New capacity must be greater than or equal to current len"
);
let (new_buffer, new_bda, new_slot, new_size_bytes) = Self::allocate_raw(new_capacity);
if self.len > 0 {
let engine = active_engine();
let bytes_to_copy = (self.len * self.stride as usize) as u64;
engine
.transfer_manager
.execute_copy_command(
self._inner.buffer(),
new_buffer.buffer(),
0,
0,
bytes_to_copy,
)
.expect("[GpuVec Lifecycle] Failed to copy elements during VRAM reallocation");
}
self.slot_index = new_slot;
self.capacity = new_capacity;
self.device_address = new_bda;
self.size_in_bytes = new_size_bytes;
self.state = new_buffer.state.clone();
self._inner = new_buffer;
}
/// Reserves capacity for at least `additional` more elements to be inserted.
///
/// # Diagnostics
/// Halts execution with diagnostic `error[E2001]` if called inside an active `flow`.
pub fn reserve(&mut self, additional: usize) {
let required_capacity = self
.len
.checked_add(additional)
.expect("[GpuVec] Capacity overflow");
if required_capacity > self.capacity {
let new_capacity = (self.capacity * 2).max(required_capacity).max(4);
self.reallocate(new_capacity);
}
}
}
impl<T: Copy + Send + Sync + 'static> Clone for GpuVec<T> {
/// Creates an independent physical duplicate of the vector in VRAM.
///
/// Allocates a new buffer with a distinct BDA and unique engine slot index,
/// executing a direct VRAM-to-VRAM copy command.
fn clone(&self) -> Self {
self.assert_host_readable("clone");
self.wait_idle_if_in_flight();
let (new_buffer, new_bda, new_slot, new_size_bytes) = Self::allocate_raw(self.capacity);
if self.len > 0 {
let engine = active_engine();
let bytes_to_copy = (self.len * self.stride as usize) as u64;
engine
.transfer_manager
.execute_copy_command(
self._inner.buffer(),
new_buffer.buffer(),
0,
0,
bytes_to_copy,
)
.expect("[GpuVec Lifecycle] VRAM-to-VRAM copy failed during clone");
}
Self {
slot_index: new_slot,
offset: self.offset,
stride: self.stride,
len: self.len,
capacity: self.capacity,
device_address: new_bda,
size_in_bytes: new_size_bytes,
state: new_buffer.state.clone(),
_inner: new_buffer,
_phantom: PhantomData,
}
}
}
</file>
<file path="enki_api/resources/vec/mod.rs">
pub mod core;
pub mod dispatch;
pub mod format;
pub mod lifecycle;
pub mod ops;
pub mod state;
pub mod transfer;
pub use self::core::GpuVec;
pub use self::state::{BufferPhase, PhaseViolationError};
impl<T: Copy + Send + Sync + 'static> From<GpuVec<T>> for Vec<T> {
#[inline(always)]
fn from(gpu_vec: GpuVec<T>) -> Self {
gpu_vec.to_vec()
}
}
impl<T: Copy + Send + Sync + 'static> From<&[T]> for GpuVec<T> {
#[inline(always)]
fn from(slice: &[T]) -> Self {
GpuVec::from_slice(slice)
}
}
impl<T: Copy + Send + Sync + 'static> From<Vec<T>> for GpuVec<T> {
#[inline(always)]
fn from(vec: Vec<T>) -> Self {
GpuVec::from_slice(&vec)
}
}
/// Creates a [`GpuVec`] containing the provided arguments.
///
/// Matches standard Rust `vec!` syntax:
/// ```rust
/// use enki::gpu_vec;
///
/// // Explicit elements
/// let v1 = gpu_vec![1u32, 2, 3, 4];
///
/// // Repeating elements
/// let v2 = gpu_vec![0.0f32; 1024];
/// ```
#[macro_export]
macro_rules! gpu_vec {
() => {
$crate::enki_api::resources::GpuVec::new()
};
($elem:expr; $n:expr) => {
$crate::enki_api::resources::GpuVec::from_elem($elem, $n)
};
($($x:expr),* $(,)?) => {
$crate::enki_api::resources::GpuVec::from_slice(&[$($x),*])
};
}
</file>
<file path="enki_api/resources/vec/ops.rs">
use super::core::GpuVec;
use crate::enki_api::context::active_engine;
use crate::enki_api::resources::slice::range::ResolvedRange;
use crate::enki_api::resources::slice::{Slice, SliceMut};
use std::mem::MaybeUninit;
use std::ops::RangeBounds;
impl<T: Copy + Send + Sync + 'static> GpuVec<T> {
/// Appends an element to the back of the vector in VRAM, expanding capacity if needed.
///
/// # Diagnostics
/// Halts execution with diagnostic `error[E2001]` if called inside an active `flow`.
pub fn push(&mut self, value: T) {
self.assert_host_readable("push");
self.wait_idle_if_in_flight();
if self.len == self.capacity {
self.reserve(1);
}
let engine = active_engine();
let element_size = std::mem::size_of::<T>();
let byte_offset = self.offset as u64 + (self.len * element_size) as u64;
let byte_data: &[u8] =
unsafe { std::slice::from_raw_parts(&value as *const T as *const u8, element_size) };
engine
.transfer_manager
.write_buffer(&self._inner, byte_offset, byte_data)
.expect("[GpuVec Ops] Failed to push element into GPU VRAM");
self.len += 1;
}
/// Removes the last element from the vector in VRAM and transfers it back to the host.
///
/// # Diagnostics
/// Halts execution with diagnostic `error[E2001]` if called inside an active `flow`.
pub fn pop(&mut self) -> Option<T> {
self.assert_host_readable("pop");
self.wait_idle_if_in_flight();
if self.len == 0 {
return None;
}
self.len -= 1;
let last_index = self.len;
let engine = active_engine();
let element_size = std::mem::size_of::<T>();
let byte_offset = self.offset as u64 + (last_index * element_size) as u64;
let mut out_val = MaybeUninit::<T>::uninit();
let byte_slice: &mut [u8] = unsafe {
std::slice::from_raw_parts_mut(out_val.as_mut_ptr() as *mut u8, element_size)
};
engine
.transfer_manager
.read_buffer(&self._inner, byte_offset, byte_slice)
.expect("[GpuVec Ops] Failed to read popped element from GPU VRAM");
unsafe { Some(out_val.assume_init()) }
}
/// Clears all elements from the vector without deallocating VRAM capacity.
///
/// # Diagnostics
/// Halts execution with diagnostic `error[E2001]` if called inside an active `flow`.
#[inline(always)]
pub fn clear(&mut self) {
self.assert_host_readable("clear");
self.wait_idle_if_in_flight();
self.len = 0;
}
/// Shortens the vector, keeping the first `len` elements and dropping the rest.
pub fn truncate(&mut self, len: usize) {
self.assert_host_readable("truncate");
self.wait_idle_if_in_flight();
if len < self.len {
self.len = len;
}
}
/// Resizes the vector in-place in VRAM so that `len` equals `new_len`.
///
/// # Diagnostics
/// Halts execution with diagnostic `error[E2001]` if called inside an active `flow`.
pub fn resize(&mut self, new_len: usize, value: T) {
self.assert_host_readable("resize");
self.wait_idle_if_in_flight();
if new_len > self.len {
let additional = new_len - self.len;
self.reserve(additional);
let old_len = self.len;
let host_fill = vec![value; additional];
let element_size = std::mem::size_of::<T>();
let byte_offset = self.offset as u64 + (old_len * element_size) as u64;
let size_in_bytes = additional * element_size;
let engine = active_engine();
let byte_data: &[u8] = unsafe {
std::slice::from_raw_parts(host_fill.as_ptr() as *const u8, size_in_bytes)
};
engine
.transfer_manager
.write_buffer(&self._inner, byte_offset, byte_data)
.expect("[GpuVec Ops] Failed to write filled elements during resize");
self.len = new_len;
} else {
self.truncate(new_len);
}
}
/// Overwrites all active elements in the vector with `value`.
///
/// # Diagnostics
/// Halts execution with diagnostic `error[E2001]` if called inside an active `flow`.
pub fn fill(&mut self, value: T) {
if self.len == 0 {
return;
}
self.assert_host_readable("fill");
self.wait_idle_if_in_flight();
let host_fill = vec![value; self.len];
let element_size = std::mem::size_of::<T>();
let size_in_bytes = self.len * element_size;
let engine = active_engine();
let byte_data: &[u8] =
unsafe { std::slice::from_raw_parts(host_fill.as_ptr() as *const u8, size_in_bytes) };
engine
.transfer_manager
.write_buffer(&self._inner, self.offset as u64, byte_data)
.expect("[GpuVec Ops] Failed to fill vector in GPU VRAM");
}
}
impl<T: Copy + Send + Sync + 'static> GpuVec<T> {
/// Borrows the entire vector as an immutable, read-only [`Slice`].
#[inline(always)]
pub fn as_slice(&self) -> Slice<'_, T> {
self.slice(..)
}
/// Borrows a contiguous sub-range of this vector as an immutable [`Slice`].
#[track_caller]
pub fn slice<R: RangeBounds<usize>>(&self, range: R) -> Slice<'_, T> {
let caller = std::panic::Location::caller();
let resolved = ResolvedRange::resolve(range, self.len, caller);
let byte_offset = (resolved.start * self.stride as usize) as u64;
Slice {
slot_index: self.slot_index,
offset: self.offset + byte_offset as u32,
element_offset: resolved.start,
stride: self.stride,
len: resolved.count,
device_address: self.device_address + byte_offset,
state: self.state.clone(),
_inner: self._inner.clone(),
_phantom: std::marker::PhantomData,
}
}
/// Borrows a contiguous sub-range of this vector as an exclusive mutable [`SliceMut`].
#[inline(always)]
pub fn as_mut_slice(&mut self) -> SliceMut<'_, T> {
self.slice_mut(..)
}
/// Borrows a contiguous sub-range of this vector as an exclusive mutable [`SliceMut`].
pub fn slice_mut<R: RangeBounds<usize>>(&mut self, range: R) -> SliceMut<'_, T> {
let caller = std::panic::Location::caller();
let resolved = ResolvedRange::resolve(range, self.len, caller);
let byte_offset = (resolved.start * self.stride as usize) as u64;
SliceMut {
slot_index: self.slot_index,
offset: self.offset + byte_offset as u32,
element_offset: resolved.start,
stride: self.stride,
len: resolved.count,
device_address: self.device_address + byte_offset,
state: self.state.clone(),
_inner: self._inner.clone(),
_phantom: std::marker::PhantomData,
}
}
/// Splits this vector into two disjoint mutable sub-slices at the given index.
#[inline(always)]
pub fn split_at_mut(&mut self, mid: usize) -> (SliceMut<'_, T>, SliceMut<'_, T>) {
assert!(
mid <= self.len,
"\n\x1b[1;31m[Enki GpuVec Error]\x1b[0m: `split_at_mut` index {} out of bounds for vector of len {}\n",
mid,
self.len
);
let byte_split = (mid * self.stride as usize) as u64;
let left = SliceMut {
slot_index: self.slot_index,
offset: self.offset,
element_offset: 0,
stride: self.stride,
len: mid,
device_address: self.device_address,
state: self.state.clone(),
_inner: self._inner.clone(),
_phantom: std::marker::PhantomData,
};
let right = SliceMut {
slot_index: self.slot_index,
offset: self.offset + byte_split as u32,
element_offset: mid,
stride: self.stride,
len: self.len - mid,
device_address: self.device_address + byte_split,
state: self.state.clone(),
_inner: self._inner.clone(),
_phantom: std::marker::PhantomData,
};
(left, right)
}
/// Borrows the vector as a mutable [`SliceMut`] without taking an exclusive `&mut self` borrow.
///
/// # Safety
/// Bypasses compile-time vector exclusivity on the host. When dispatched to GPU Nams:
/// - In safe mode (`.run()`), passing this slice across parallel threads (`size_x * size_y * size_z > 1`)
/// will halt with diagnostic **`error[E1009]`**.
/// - In unchecked mode (`.run_unchecked()`), the runtime `BorrowEngine` will still actively
/// enforce range disjointness against other active slices of the same buffer (**`error[E1007]`**).
#[inline(always)]
pub unsafe fn as_mut_slice_unchecked(&self) -> SliceMut<'_, T> {
unsafe { self.slice_mut_unchecked(..) }
}
/// Borrows a sub-range as a mutable [`SliceMut`] without taking an exclusive `&mut self` borrow.
///
/// # Safety
/// Bypasses compile-time vector exclusivity on the host. Range disjointness against other
/// active slices of the same buffer is actively enforced by the runtime `BorrowEngine` (**`error[E1007]`**).
#[track_caller]
pub unsafe fn slice_mut_unchecked<R: RangeBounds<usize>>(&self, range: R) -> SliceMut<'_, T> {
let caller = std::panic::Location::caller();
let resolved = ResolvedRange::resolve(range, self.len, caller);
let byte_offset = (resolved.start * self.stride as usize) as u64;
SliceMut {
slot_index: self.slot_index,
offset: self.offset + byte_offset as u32,
element_offset: resolved.start,
stride: self.stride,
len: resolved.count,
device_address: self.device_address + byte_offset,
state: self.state.clone(),
_inner: self._inner.clone(),
_phantom: std::marker::PhantomData,
}
}
}
</file>
<file path="enki_api/resources/vec/state.rs">
use super::core::GpuVec;
use crate::enki_api::context::active_engine;
use crate::enki_api::context::ambient::{is_inside_active_flow, resolve_user_caller};
use crate::enki_api::context::errors::emit_and_abort;
/// The temporal execution phase of a GPU buffer relative to the command timeline.
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum BufferPhase {
/// Buffer is idle and safe for immediate host manipulation.
Idle,
/// Buffer is currently involved in recording commands within an active flow.
Recording,
/// Buffer has commands submitted and executing on the GPU hardware.
InFlight { target_timeline: u64 },
}
#[derive(Debug)]
pub struct PhaseViolationError {
pub root_id: usize,
pub phase: BufferPhase,
pub operation: &'static str,
}
impl std::fmt::Display for PhaseViolationError {
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
write!(
f,
"illegal host {op} on GpuVec (root id: {root}) during GPU command recording phase",
op = self.operation,
root = self.root_id,
)
}
}
impl std::error::Error for PhaseViolationError {}
impl<T: Copy + Send + Sync + 'static> GpuVec<T> {
/// Queries the real-time execution phase of this buffer.
pub fn current_phase(&self) -> BufferPhase {
if is_inside_active_flow() {
return BufferPhase::Recording;
}
let engine = active_engine();
let target_timeline = engine
.timeline_counter
.load(std::sync::atomic::Ordering::SeqCst);
let current_gpu_timeline = engine.timeline_semaphore.get_timeline_value().unwrap_or(0);
if current_gpu_timeline < target_timeline {
BufferPhase::InFlight { target_timeline }
} else {
BufferPhase::Idle
}
}
#[track_caller]
#[inline(always)]
pub(crate) fn assert_host_readable(&self, operation: &'static str) {
if let BufferPhase::Recording = self.current_phase() {
let raw_caller = std::panic::Location::caller();
let caller = resolve_user_caller(raw_caller);
let diag = anu::diagnostics::rt::phase_violation(operation, caller);
emit_and_abort(&diag);
}
}
/// Returns `true` if the buffer is idle and safe for host access without waiting.
#[inline(always)]
pub fn is_idle(&self) -> bool {
matches!(self.current_phase(), BufferPhase::Idle)
}
/// Returns `true` if the buffer is currently involved in an active flow recording.
#[inline(always)]
pub fn is_recording(&self) -> bool {
matches!(self.current_phase(), BufferPhase::Recording)
}
}
</file>
<file path="enki_api/resources/vec/transfer.rs">
use super::core::GpuVec;
use crate::enki_api::context::active_engine;
use std::mem::MaybeUninit;
impl<T: Copy + Send + Sync + 'static> GpuVec<T> {
/// Reads a single element from VRAM back to the host CPU at the specified index.
///
/// Automatically synchronizes with the GPU timeline if the buffer is currently in-flight.
///
/// # Diagnostics
/// Halts execution with diagnostic `error[E02001]` if called inside an active `flow`.
#[track_caller]
pub fn get(&self, index: usize) -> Option<T> {
if index >= self.len {
return None;
}
self.assert_host_readable("get");
self.wait_idle_if_in_flight();
let engine = active_engine();
let element_size = std::mem::size_of::<T>();
let byte_offset = self.offset as u64 + (index * element_size) as u64;
let mut out_val = MaybeUninit::<T>::uninit();
let byte_slice: &mut [u8] = unsafe {
std::slice::from_raw_parts_mut(out_val.as_mut_ptr() as *mut u8, element_size)
};
engine
.transfer_manager
.read_buffer(&self._inner, byte_offset, byte_slice)
.expect("[GpuVec Transfer] Failed to read element from GPU VRAM");
unsafe { Some(out_val.assume_init()) }
}
/// Copies all active elements from VRAM into a newly allocated host `Vec<T>`.
///
/// # Diagnostics
/// Halts execution with diagnostic `error[E2001]` if called inside an active `flow`.
#[track_caller]
pub fn to_vec(&self) -> Vec<T> {
self.assert_host_readable("to_vec");
self.wait_idle_if_in_flight();
if self.len == 0 {
return Vec::new();
}
let engine = active_engine();
let element_size = std::mem::size_of::<T>();
let total_bytes = self.len * element_size;
let mut uninit_vec: Vec<MaybeUninit<T>> = Vec::with_capacity(self.len);
unsafe {
uninit_vec.set_len(self.len);
}
let byte_slice: &mut [u8] = unsafe {
std::slice::from_raw_parts_mut(uninit_vec.as_mut_ptr() as *mut u8, total_bytes)
};
engine
.transfer_manager
.read_buffer(&self._inner, self.offset as u64, byte_slice)
.expect("[GpuVec Transfer] Failed to read bytes from GPU VRAM");
unsafe {
let mut manual_vec = std::mem::ManuallyDrop::new(uninit_vec);
Vec::from_raw_parts(
manual_vec.as_mut_ptr() as *mut T,
manual_vec.len(),
manual_vec.capacity(),
)
}
}
/// Sets a single value from the host CPU to the VRAM buffer at the given index.
///
/// # Diagnostics
/// Halts execution with diagnostic `error[E2001]` if called inside an active `flow`.
#[track_caller]
pub fn set(&mut self, index: usize, value: T) -> bool {
if index >= self.len {
return false;
}
self.copy_from_slice_at(index, std::slice::from_ref(&value));
true
}
/// Overwrites the active contents of this vector with data from a host CPU slice.
///
/// # Panics
/// Panics if `self.len() != data.len()`.
///
/// # Diagnostics
/// Halts execution with diagnostic `error[E2001]` if called inside an active `flow`.
#[track_caller]
pub fn copy_from_slice(&mut self, data: &[T]) {
assert_eq!(
self.len,
data.len(),
"[GpuVec Transfer] `copy_from_slice` length mismatch: destination has len {} but source has len {}",
self.len,
data.len()
);
self.copy_from_slice_at(0, data);
}
/// Overwrites a sub-range of VRAM starting at element `offset` with data from a CPU slice.
///
/// # Panics
/// Panics if `offset + data.len() > self.len()`.
///
/// # Diagnostics
/// Halts execution with diagnostic `error[E2001]` if called inside an active `flow`.
#[track_caller]
pub fn copy_from_slice_at(&mut self, offset: usize, data: &[T]) {
assert!(
offset + data.len() <= self.len,
"[GpuVec Transfer] `copy_from_slice_at` out of bounds: offset {} + len {} exceeds vector len {}",
offset,
data.len(),
self.len
);
self.assert_host_readable("copy_from_slice_at");
self.wait_idle_if_in_flight();
if data.is_empty() {
return;
}
let engine = active_engine();
let element_size = std::mem::size_of::<T>();
let byte_offset = self.offset as u64 + (offset * element_size) as u64;
let size_in_bytes = data.len() * element_size;
let byte_data: &[u8] =
unsafe { std::slice::from_raw_parts(data.as_ptr() as *const u8, size_in_bytes) };
engine
.transfer_manager
.write_buffer(&self._inner, byte_offset, byte_data)
.expect("[GpuVec Transfer] Failed to write slice into GPU VRAM");
}
/// Appends all elements from a host CPU slice to the end of the vector in VRAM.
///
/// Reallocates VRAM capacity via exponential growth if necessary.
///
/// # Diagnostics
/// Halts execution with diagnostic `error[E2001]` if called inside an active `flow`.
#[track_caller]
pub fn extend_from_slice(&mut self, data: &[T]) {
if data.is_empty() {
return;
}
self.assert_host_readable("extend_from_slice");
self.wait_idle_if_in_flight();
let old_len = self.len;
self.reserve(data.len());
let engine = active_engine();
let element_size = std::mem::size_of::<T>();
let byte_offset = self.offset as u64 + (old_len * element_size) as u64;
let size_in_bytes = data.len() * element_size;
let byte_data: &[u8] =
unsafe { std::slice::from_raw_parts(data.as_ptr() as *const u8, size_in_bytes) };
engine
.transfer_manager
.write_buffer(&self._inner, byte_offset, byte_data)
.expect("[GpuVec Transfer] Failed to append slice into GPU VRAM");
self.len += data.len();
}
}
</file>
<file path="enki_api/resources/atomic.rs">
use std::marker::PhantomData;
use std::sync::Arc;
use std::sync::atomic::{AtomicU8, Ordering};
use anyhow::Context;
use apsu::{BufferUsage, GpuDeviceBuffer};
use anu::nam_args_api::{
AccessIntent, ArgDescriptor, ArgValue, GpuBufferTarget, GpuType, IngressContext,
};
use crate::enki_api::context::active_engine;
/// Marks integer primitive types natively supported by GPU hardware atomic instructions.
///
/// Maps host scalar types (`u32`, `i32`, `u64`, etc.) to their standard library
/// atomic equivalents (`AtomicU32`, `AtomicI32`, etc.) inside `#[nam]` functions.
pub trait GpuAtomicTarget: Copy + Send + Sync + 'static {
/// The corresponding standard library atomic type exposed inside GPU compute nam.
type NamTarget: 'static;
}
impl GpuAtomicTarget for u32 {
type NamTarget = core::sync::atomic::AtomicU32;
}
impl GpuAtomicTarget for i32 {
type NamTarget = core::sync::atomic::AtomicI32;
}
impl GpuAtomicTarget for u64 {
type NamTarget = core::sync::atomic::AtomicU64;
}
impl GpuAtomicTarget for i64 {
type NamTarget = core::sync::atomic::AtomicI64;
}
impl GpuAtomicTarget for usize {
type NamTarget = core::sync::atomic::AtomicUsize;
}
impl GpuAtomicTarget for isize {
type NamTarget = core::sync::atomic::AtomicIsize;
}
/// A single, isolated hardware atomic variable residing in GPU VRAM.
///
/// Allocated as a GPU device storage buffer with direct 64-bit Buffer Device Address (BDA) support.
///
/// # Nam Dispatch Semantics
/// Passing `&GpuAtomic<T>` or `&mut GpuAtomic<T>` to a `#[nam]` function binds as a shared reference
/// to the standard library atomic type **`&T::NamTarget`** (e.g., `&AtomicU32`), permitting
/// atomic operations such as `.fetch_add()`, `.load()`, and `.store()`.
#[derive(Clone)]
pub struct GpuAtomic<T: GpuAtomicTarget> {
pub slot_index: u32,
pub device_address: u64,
pub state: Arc<AtomicU8>,
pub _inner: Arc<GpuDeviceBuffer>,
_phantom: PhantomData<T>,
}
impl<T: GpuAtomicTarget> GpuAtomic<T> {
/// Allocates a single hardware atomic variable in VRAM initialized with the given value.
pub fn new(initial_value: T) -> Self {
let engine = active_engine();
let allocator = engine.allocator.clone();
let transfer_manager = &engine.transfer_manager;
let usage = BufferUsage::STORAGE_BUFFER
| BufferUsage::SHADER_DEVICE_ADDRESS
| BufferUsage::TRANSFER_DST
| BufferUsage::TRANSFER_SRC;
let size_bytes = std::mem::size_of::<T>() as u64;
let device_buffer = GpuDeviceBuffer::new(allocator, size_bytes, usage)
.map_err(|e| anyhow::anyhow!("[GpuAtomic] Failed to allocate device buffer: {}", e))
.unwrap();
let slot_index = device_buffer.id;
let device_address = device_buffer.device_address();
let state = device_buffer.state.clone();
let bytes: &[u8] = unsafe {
std::slice::from_raw_parts(&initial_value as *const T as *const u8, size_bytes as usize)
};
transfer_manager
.write_buffer(&device_buffer, 0, bytes)
.map_err(|e| anyhow::anyhow!("[GpuAtomic] Failed to upload initial value: {}", e))
.unwrap();
Self {
slot_index,
device_address,
state,
_inner: Arc::new(device_buffer),
_phantom: PhantomData,
}
}
/// Reads the current value of the atomic variable from VRAM back to the host CPU.
///
/// Automatically waits for in-flight GPU execution on the timeline before reading.
///
/// # Diagnostics
/// Halts execution with diagnostic **`error[E2001]`** if called inside an active `flow`.
pub fn get(&self) -> T {
self.try_get()
.expect("[GpuAtomic] Failed to read atomic value from GPU VRAM")
}
fn try_get(&self) -> anyhow::Result<T> {
let engine = active_engine();
let last_value = engine.timeline_counter.load(Ordering::SeqCst);
if last_value > 0 {
engine
.timeline_semaphore
.wait_timeline(last_value, std::time::Duration::from_secs(5))
.context("[GpuAtomic] Timeline wait failed before reading")?;
}
let mut out_val = unsafe { std::mem::zeroed::<T>() };
let byte_data: &mut [u8] = unsafe {
std::slice::from_raw_parts_mut(
&mut out_val as *mut T as *mut u8,
std::mem::size_of::<T>(),
)
};
engine
.transfer_manager
.read_buffer(&self._inner, 0, byte_data)
.context("[GpuAtomic] Failed to read atomic value")?;
Ok(out_val)
}
/// Overwrites the atomic variable in VRAM from the host CPU.
///
/// # Diagnostics
/// Halts execution with diagnostic **`error[E2001]`** if called inside an active `flow`.
pub fn set(&self, value: T) {
self.try_set(value)
.expect("[GpuAtomic] Failed to write atomic value to GPU VRAM");
}
fn try_set(&self, value: T) -> anyhow::Result<()> {
let engine = active_engine();
let byte_data: &[u8] = unsafe {
std::slice::from_raw_parts(&value as *const T as *const u8, std::mem::size_of::<T>())
};
engine
.transfer_manager
.write_buffer(&self._inner, 0, byte_data)
.map_err(|e| anyhow::anyhow!("[GpuAtomic] set failed: {}", e))
}
}
impl<T: GpuAtomicTarget> GpuType for &GpuAtomic<T> {
fn describe() -> ArgDescriptor {
ArgDescriptor::atomic_cell::<T>()
}
fn collect<'a>(&self, ctx: &mut IngressContext<'a>) {
let desc = Self::describe();
ctx.push_arg(ArgValue::BufferBDA(self.device_address), desc);
ctx.push_resource_binding(
format!("GpuAtomic_Slot_{}", self.slot_index),
self.slot_index,
self.state.clone(),
AccessIntent::ReadWrite,
);
}
}
/// A contiguous array of hardware atomic variables residing in GPU VRAM.
///
/// Designed for parallel GPU coordination algorithms such as spatial binning,
/// global work distribution, histogram calculations, and concurrent bucket accumulation.
///
/// # Nam Dispatch Semantics
/// Passing `&GpuAtomicVec<T>` or `&mut GpuAtomicVec<T>` to a `#[nam]` function binds as a slice of
/// standard library atomics **`&[T::NamTarget]`** (e.g., `&[AtomicU32]`).
///
/// In safe dispatch mode, the runtime `BorrowEngine` permits concurrent multi-threaded writes
/// because hardware atomic operations are safe across parallel threads by definition.
#[derive(Clone)]
pub struct GpuAtomicVec<T: GpuAtomicTarget> {
pub slot_index: u32,
pub element_count: usize,
pub stride: u32,
pub device_address: u64,
pub state: Arc<AtomicU8>,
pub _inner: Arc<GpuDeviceBuffer>,
_phantom: PhantomData<T>,
}
impl<T: GpuAtomicTarget> GpuAtomicVec<T> {
/// Allocates an array of hardware atomic variables in VRAM initialized with a host slice.
pub fn new(data: &[T]) -> Self {
let engine = active_engine();
let allocator = engine.allocator.clone();
let transfer_manager = &engine.transfer_manager;
let usage = BufferUsage::STORAGE_BUFFER
| BufferUsage::SHADER_DEVICE_ADDRESS
| BufferUsage::TRANSFER_DST
| BufferUsage::TRANSFER_SRC;
let element_count = data.len();
let element_size = std::mem::size_of::<T>();
let size_bytes = std::mem::size_of_val(data) as u64;
let device_buffer = GpuDeviceBuffer::new(allocator, size_bytes, usage)
.map_err(|e| anyhow::anyhow!("[GpuAtomicVec] Failed to allocate device buffer: {}", e))
.unwrap();
let slot_index = device_buffer.id;
let device_address = device_buffer.device_address();
let state = device_buffer.state.clone();
let bytes: &[u8] = unsafe {
std::slice::from_raw_parts(data.as_ptr() as *const u8, std::mem::size_of_val(data))
};
transfer_manager
.write_buffer(&device_buffer, 0, bytes)
.map_err(|e| anyhow::anyhow!("[GpuAtomicVec] Failed to upload initial data: {}", e))
.unwrap();
Self {
slot_index,
element_count,
stride: element_size as u32,
device_address,
state,
_inner: Arc::new(device_buffer),
_phantom: PhantomData,
}
}
/// Allocates an uninitialized array of hardware atomic variables with the specified capacity.
pub fn with_capacity(capacity: usize) -> Self {
let engine = active_engine();
let allocator = engine.allocator.clone();
let usage = BufferUsage::STORAGE_BUFFER
| BufferUsage::SHADER_DEVICE_ADDRESS
| BufferUsage::TRANSFER_DST
| BufferUsage::TRANSFER_SRC;
let element_size = std::mem::size_of::<T>();
let size_bytes = (capacity * element_size) as u64;
let device_buffer = GpuDeviceBuffer::new(allocator, size_bytes, usage)
.map_err(|e| anyhow::anyhow!("[GpuAtomicVec] Allocation failed: {}", e))
.unwrap();
let slot_index = device_buffer.id;
let device_address = device_buffer.device_address();
let state = device_buffer.state.clone();
Self {
slot_index,
element_count: capacity,
stride: element_size as u32,
device_address,
state,
_inner: Arc::new(device_buffer),
_phantom: PhantomData,
}
}
/// Returns the number of atomic variables in the buffer.
#[inline(always)]
pub const fn len(&self) -> usize {
self.element_count
}
/// Returns `true` if the buffer contains zero elements.
#[inline(always)]
pub const fn is_empty(&self) -> bool {
self.element_count == 0
}
/// Reads all atomic elements from VRAM back to a host `Vec<T>`.
///
/// Automatically waits for in-flight GPU execution on the timeline before reading.
///
/// # Diagnostics
/// Halts execution with diagnostic **`error[E2001]`** if called inside an active `flow`.
pub fn to_vec(&self) -> anyhow::Result<Vec<T>> {
let engine = active_engine();
let last_value = engine.timeline_counter.load(Ordering::SeqCst);
if last_value > 0 {
engine
.timeline_semaphore
.wait_timeline(last_value, std::time::Duration::from_secs(5))
.context("[GpuAtomicVec] Timeline wait failed before reading")?;
}
let mut out_data = vec![unsafe { std::mem::zeroed::<T>() }; self.element_count];
let byte_data: &mut [u8] = unsafe {
std::slice::from_raw_parts_mut(
out_data.as_mut_ptr() as *mut u8,
out_data.len() * std::mem::size_of::<T>(),
)
};
engine
.transfer_manager
.read_buffer(&self._inner, 0, byte_data)
.context("[GpuAtomicVec] Failed to read atomic buffer data")?;
Ok(out_data)
}
}
impl<T: GpuAtomicTarget> GpuType for &GpuAtomicVec<T> {
fn describe() -> ArgDescriptor {
ArgDescriptor::atomic_slice::<T>()
}
fn collect<'a>(&self, ctx: &mut IngressContext<'a>) {
let desc = Self::describe();
ctx.push_arg(
ArgValue::SliceBDA {
bda: self.device_address,
count: self.element_count as u64,
},
desc,
);
ctx.push_resource_binding(
format!("GpuAtomicVec_Slot_{}", self.slot_index),
self.slot_index,
self.state.clone(),
AccessIntent::ReadWrite,
);
}
}
impl<T: GpuAtomicTarget> GpuType for &mut GpuAtomic<T> {
fn describe() -> ArgDescriptor {
ArgDescriptor::atomic_cell::<T>()
}
fn collect<'a>(&self, ctx: &mut IngressContext<'a>) {
let desc = Self::describe();
ctx.push_arg(ArgValue::BufferBDA(self.device_address), desc);
ctx.push_resource_binding(
format!("GpuAtomic_Slot_{}", self.slot_index),
self.slot_index,
self.state.clone(),
AccessIntent::ReadWrite,
);
}
}
impl<T: GpuAtomicTarget> GpuType for &mut GpuAtomicVec<T> {
fn describe() -> ArgDescriptor {
ArgDescriptor::atomic_slice::<T>()
}
fn collect<'a>(&self, ctx: &mut IngressContext<'a>) {
let desc = Self::describe();
ctx.push_arg(
ArgValue::SliceBDA {
bda: self.device_address,
count: self.element_count as u64,
},
desc,
);
ctx.push_resource_binding(
format!("GpuAtomicVec_Slot_{}", self.slot_index),
self.slot_index,
self.state.clone(),
AccessIntent::ReadWrite,
);
}
}
impl<T: GpuAtomicTarget> GpuBufferTarget<T> for GpuAtomic<T> {}
impl<T: GpuAtomicTarget> GpuBufferTarget<T> for GpuAtomicVec<T> {}
</file>
<file path="enki_api/resources/mod.rs">
pub mod atomic;
pub mod param;
pub mod slice;
pub mod tile_mem;
pub mod vec;
pub use atomic::{GpuAtomic, GpuAtomicTarget, GpuAtomicVec};
pub use param::GpuParam;
pub use slice::{Slice, SliceMut};
pub use tile_mem::GpuTileMem;
pub use vec::GpuVec;
</file>
<file path="enki_api/resources/param.rs">
use anu::nam_args_api::{ArgDescriptor, ArgValue, GpuType, IngressContext};
use std::ops::{Deref, DerefMut};
/// Container for uniforms and small configuration structs passed directly by value into GPU nam functions.
///
/// Unlike buffer resources, `GpuParam<T>` does not allocate an independent VRAM buffer.
/// Its byte payload is packed directly into the engine's uniform `ParamArena` per dispatch.
///
/// # Nam Dispatch Semantics
/// Passing `GpuParam<T>` or `&GpuParam<T>` to a `#[nam]` function binds directly by-value
/// as **`T`** inside the nam function signature (e.g., `dt: f32`, `cam: CameraParams`).
#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
#[repr(transparent)]
pub struct GpuParam<T> {
/// The inner uniform value.
pub value: T,
}
impl<T> GpuParam<T> {
/// Creates a new `GpuParam` wrapping the provided value.
#[inline(always)]
pub const fn new(value: T) -> Self {
Self { value }
}
/// Updates the inner value.
#[inline(always)]
pub fn set(&mut self, value: T) {
self.value = value;
}
/// Gets an immutable reference to the inner value.
#[inline(always)]
pub const fn get(&self) -> &T {
&self.value
}
}
impl<T> From<T> for GpuParam<T> {
#[inline(always)]
fn from(value: T) -> Self {
Self::new(value)
}
}
impl<T> Deref for GpuParam<T> {
type Target = T;
#[inline(always)]
fn deref(&self) -> &Self::Target {
&self.value
}
}
impl<T> DerefMut for GpuParam<T> {
#[inline(always)]
fn deref_mut(&mut self) -> &mut Self::Target {
&mut self.value
}
}
impl<T: Copy + Send + Sync + 'static> GpuType for GpuParam<T> {
fn describe() -> ArgDescriptor {
ArgDescriptor::by_value::<T>()
}
fn collect<'a>(&self, ctx: &mut IngressContext<'a>) {
let size = std::mem::size_of::<T>();
let bytes =
unsafe { std::slice::from_raw_parts(&self.value as *const T as *const u8, size) }
.to_vec();
let desc = Self::describe();
ctx.push_arg(ArgValue::Payload(bytes), desc);
}
}
impl<T: Copy + Send + Sync + 'static> GpuType for &GpuParam<T> {
fn describe() -> ArgDescriptor {
ArgDescriptor::by_value::<T>()
}
fn collect<'a>(&self, ctx: &mut IngressContext<'a>) {
GpuType::collect(*self, ctx);
}
}
</file>
<file path="enki_api/resources/tile_mem.rs">
use anu::nam_args_api::{ArgDescriptor, ArgValue, GpuType, IngressContext};
use std::marker::PhantomData;
/// A zero-cost hardware token used to allocate on-chip shared scratchpad memory for a spatial tile.
///
/// `GpuTileMem` does not allocate physical VRAM and has a zero-byte footprint in the `ParamArena`.
/// On GPU hardware, it maps directly to high-speed on-chip SRAM (`Workgroup` storage class / LDS).
///
/// # Nam Dispatch Semantics
/// Passing `GpuTileMem<T, N>` or `&GpuTileMem<T, N>` to a `#[nam]` function binds as an exclusive,
/// shared array reference **`&mut [T; N]`** across all threads within the same spatial tile.
///
/// # Synchronization & Memory Barrier
/// Threads within the tile coordinate access to this shared scratchpad via [`Space::sync()`].
///
/// # Diagnostics
/// Halts execution with diagnostic **`error[E1006]`** if total byte size (`N * size_of::<T>()`)
/// exceeds the physical GPU's maximum compute workgroup shared memory capacity.
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub struct GpuTileMem<T, const N: usize> {
_phantom: PhantomData<T>,
}
impl<T, const N: usize> GpuTileMem<T, N> {
/// Creates a new `GpuTileMem` token representing an on-chip scratchpad of `N` elements of type `T`.
#[inline(always)]
pub const fn new() -> Self {
Self {
_phantom: PhantomData,
}
}
/// Returns the compile-time element capacity `N` of this tile scratchpad.
#[inline(always)]
pub const fn len(&self) -> usize {
N
}
/// Returns `true` if the scratchpad element capacity is zero.
#[inline(always)]
pub const fn is_empty(&self) -> bool {
N == 0
}
}
impl<T: Copy + Send + Sync + 'static, const N: usize> GpuType for GpuTileMem<T, N> {
fn describe() -> ArgDescriptor {
ArgDescriptor::workgroup_scratchpad::<T, N>()
}
fn collect<'a>(&self, ctx: &mut IngressContext<'a>) {
let desc = Self::describe();
ctx.push_arg(ArgValue::ZeroFootprint, desc);
}
}
impl<T: Copy + Send + Sync + 'static, const N: usize> GpuType for &GpuTileMem<T, N> {
fn describe() -> ArgDescriptor {
ArgDescriptor::workgroup_scratchpad::<T, N>()
}
fn collect<'a>(&self, ctx: &mut IngressContext<'a>) {
GpuType::collect(*self, ctx);
}
}
</file>
<file path="enki_api/space/builder.rs">
use super::core::Space;
use super::tile::{IntoTile, TileConfig};
use super::tuner::SpatialAutoTuner;
use parsu::profile::HardwareProfile;
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub(crate) struct ResolvedSpatialDispatch {
pub global_size: (u32, u32, u32),
pub local_size: (u32, u32, u32),
pub dispatch_groups: (u32, u32, u32),
}
impl Space {
/// Creates a 1D GPU execution domain with the specified global problem size.
///
/// By default, tile dimensions are automatically factored into optimal workgroups
#[inline(always)]
pub fn gpu_x(size_x: usize) -> Self {
Self {
x: 0,
y: 0,
z: 0,
cell_x: 0,
cell_y: 0,
cell_z: 0,
tile_x: 0,
tile_y: 0,
tile_z: 0,
size_x: size_x.max(1),
size_y: 1,
size_z: 1,
tile_size_x: 0,
tile_size_y: 0,
tile_size_z: 0,
}
}
/// Creates a 2D GPU execution domain with the specified width and height.
///
/// By default, tile dimensions are automatically factored into optimal workgroups.
#[inline(always)]
pub fn gpu_xy(size_x: usize, size_y: usize) -> Self {
Self {
x: 0,
y: 0,
z: 0,
cell_x: 0,
cell_y: 0,
cell_z: 0,
tile_x: 0,
tile_y: 0,
tile_z: 0,
size_x: size_x.max(1),
size_y: size_y.max(1),
size_z: 1,
tile_size_x: 0,
tile_size_y: 0,
tile_size_z: 0,
}
}
/// Creates a 3D volumetric GPU execution domain with the specified width, height, and depth.
///
/// By default, tile dimensions are automatically factored into optimal workgroups.
#[inline(always)]
pub fn gpu_xyz(size_x: usize, size_y: usize, size_z: usize) -> Self {
Self {
x: 0,
y: 0,
z: 0,
cell_x: 0,
cell_y: 0,
cell_z: 0,
tile_x: 0,
tile_y: 0,
tile_z: 0,
size_x: size_x.max(1),
size_y: size_y.max(1),
size_z: size_z.max(1),
tile_size_x: 0,
tile_size_y: 0,
tile_size_z: 0,
}
}
/// Overrides automatic tile tuning with an explicit, fixed tile configuration.
///
/// Accepts scalar values for 1D, tuples/arrays for 2D, and triplets for 3D.
///
/// # Examples
/// ```rust
/// use enki::Space;
///
/// let space_1d = Space::gpu_x(10_000_000).tile(256);
/// let space_2d = Space::gpu_xy(1920, 1080).tile([16, 16]);
/// let space_3d = Space::gpu_xyz(256, 256, 128).tile([8, 8, 4]);
/// ```
#[inline(always)]
pub fn tile<T: IntoTile>(mut self, tile: T) -> Self {
match tile.into_tile() {
TileConfig::Auto => {
self.tile_size_x = 0;
self.tile_size_y = 0;
self.tile_size_z = 0;
}
TileConfig::Custom(tx, ty, tz) => {
self.tile_size_x = tx as usize;
self.tile_size_y = ty as usize;
self.tile_size_z = tz as usize;
if self.tile_size_x > 0 {
self.cell_x = self.x % self.tile_size_x;
self.tile_x = self.x / self.tile_size_x;
}
if self.tile_size_y > 0 {
self.cell_y = self.y % self.tile_size_y;
self.tile_y = self.y / self.tile_size_y;
}
if self.tile_size_z > 0 {
self.cell_z = self.z % self.tile_size_z;
self.tile_z = self.z / self.tile_size_z;
}
}
}
self
}
pub(crate) fn resolve_dispatch(&self, profile: &HardwareProfile) -> ResolvedSpatialDispatch {
let global_size = (self.size_x as u32, self.size_y as u32, self.size_z as u32);
let config = if self.tile_size_x == 0 {
TileConfig::Auto
} else {
TileConfig::Custom(
self.tile_size_x as u32,
self.tile_size_y.max(1) as u32,
self.tile_size_z.max(1) as u32,
)
};
let local_size = SpatialAutoTuner::resolve_tile(config, global_size, profile);
let group_x = global_size.0.div_ceil(local_size.0);
let group_y = global_size.1.div_ceil(local_size.1);
let group_z = global_size.2.div_ceil(local_size.2);
ResolvedSpatialDispatch {
global_size,
local_size,
dispatch_groups: (group_x, group_y, group_z),
}
}
}
</file>
<file path="enki_api/space/core.rs">
/// Spatial execution context provided to every invocation of a `#[nam]` function.
///
/// Encapsulates global thread coordinates, local tile topology, domain boundaries,
/// flat memory indexing helpers, and intra-tile synchronization barriers.
#[derive(Copy, Clone, Debug, PartialEq, Eq, Default)]
#[repr(C)]
pub struct Space {
/// Global execution coordinate along the X axis.
pub x: usize,
/// Global execution coordinate along the Y axis.
pub y: usize,
/// Global execution coordinate along the Z axis.
pub z: usize,
/// Local cell coordinate along the X axis inside the current tile.
pub cell_x: usize,
/// Local cell coordinate along the Y axis inside the current tile.
pub cell_y: usize,
/// Local cell coordinate along the Z axis inside the current tile.
pub cell_z: usize,
/// Tile index along the X axis in the global tile grid.
pub tile_x: usize,
/// Tile index along the Y axis in the global tile grid.
pub tile_y: usize,
/// Tile index along the Z axis in the global tile grid.
pub tile_z: usize,
/// Total global problem size along the X axis.
pub size_x: usize,
/// Total global problem size along the Y axis.
pub size_y: usize,
/// Total global problem size along the Z axis.
pub size_z: usize,
/// Local tile capacity dimension along the X axis.
pub tile_size_x: usize,
/// Local tile capacity dimension along the Y axis.
pub tile_size_y: usize,
/// Local tile capacity dimension along the Z axis.
pub tile_size_z: usize,
}
impl Space {
/// Synchronizes all execution threads within the current spatial tile.
///
/// # Diagnostics
///
/// Triggers a compile-error `error[E0009]` if all cells in a tile do not reach `space.sync()` concurrently.
///
/// Does nothing if the `#[nam]` function is called on the CPU.
#[inline(always)]
pub fn sync(&self) {
__enki_sync();
}
/// Returns the global coordinate along the X axis.
#[inline(always)]
pub const fn pos_x(&self) -> usize {
self.x
}
/// Returns the global (x, y) coordinates as a 2D tuple.
#[inline(always)]
pub const fn pos_xy(&self) -> (usize, usize) {
(self.x, self.y)
}
/// Returns the global (x, y, z) coordinates as a 3D tuple.
#[inline(always)]
pub const fn pos_xyz(&self) -> (usize, usize, usize) {
(self.x, self.y, self.z)
}
/// Returns the local cell coordinate along the X axis inside the tile.
#[inline(always)]
pub const fn cell_x(&self) -> usize {
self.cell_x
}
/// Returns the local cell (cell_x, cell_y) coordinates inside the tile.
#[inline(always)]
pub const fn cell_xy(&self) -> (usize, usize) {
(self.cell_x, self.cell_y)
}
/// Returns the local cell (cell_x, cell_y, cell_z) coordinates inside the tile.
#[inline(always)]
pub const fn cell_xyz(&self) -> (usize, usize, usize) {
(self.cell_x, self.cell_y, self.cell_z)
}
/// Returns the tile index along the X axis in the global tile grid.
#[inline(always)]
pub const fn tile_x(&self) -> usize {
self.tile_x
}
/// Returns the tile index (tile_x, tile_y) in the global tile grid.
#[inline(always)]
pub const fn tile_xy(&self) -> (usize, usize) {
(self.tile_x, self.tile_y)
}
/// Returns the tile index (tile_x, tile_y, tile_z) in the global tile grid.
#[inline(always)]
pub const fn tile_xyz(&self) -> (usize, usize, usize) {
(self.tile_x, self.tile_y, self.tile_z)
}
/// Returns the global domain size along the X axis.
#[inline(always)]
pub const fn size_x(&self) -> usize {
self.size_x
}
/// Returns the global domain (size_x, size_y) as a 2D tuple.
#[inline(always)]
pub const fn size_xy(&self) -> (usize, usize) {
(self.size_x, self.size_y)
}
/// Returns the global domain (size_x, size_y, size_z) as a 3D tuple.
#[inline(always)]
pub const fn size_xyz(&self) -> (usize, usize, usize) {
(self.size_x, self.size_y, self.size_z)
}
/// Returns the tile dimension along the X axis.
#[inline(always)]
pub const fn tile_size_x(&self) -> usize {
self.tile_size_x
}
/// Returns the tile dimensions (tile_size_x, tile_size_y) as a 2D tuple.
#[inline(always)]
pub const fn tile_size_xy(&self) -> (usize, usize) {
(self.tile_size_x, self.tile_size_y)
}
/// Returns the tile dimensions (tile_size_x, tile_size_y, tile_size_z) as a 3D tuple.
#[inline(always)]
pub const fn tile_size_xyz(&self) -> (usize, usize, usize) {
(self.tile_size_x, self.tile_size_y, self.tile_size_z)
}
/// Returns the number of tiles along the X axis in the global grid.
#[inline(always)]
pub fn tile_count_x(&self) -> usize {
self.size_x.div_ceil(self.tile_size_x.max(1))
}
/// Returns the number of tiles along the Y axis in the global grid.
#[inline(always)]
pub fn tile_count_y(&self) -> usize {
self.size_y.div_ceil(self.tile_size_y.max(1))
}
/// Returns the number of tiles along the Z axis in the global grid.
#[inline(always)]
pub fn tile_count_z(&self) -> usize {
self.size_z.div_ceil(self.tile_size_z.max(1))
}
/// Returns the number of tiles (count_x, count_y) in the global grid.
#[inline(always)]
pub fn tile_count_xy(&self) -> (usize, usize) {
let tx = self.size_x.div_ceil(self.tile_size_x.max(1));
let ty = self.size_y.div_ceil(self.tile_size_y.max(1));
(tx, ty)
}
/// Returns the number of tiles (count_x, count_y, count_z) in the global grid.
#[inline(always)]
pub fn tile_count_xyz(&self) -> (usize, usize, usize) {
let tx = self.size_x.div_ceil(self.tile_size_x.max(1));
let ty = self.size_y.div_ceil(self.tile_size_y.max(1));
let tz = self.size_z.div_ceil(self.tile_size_z.max(1));
(tx, ty, tz)
}
/// Returns the total number of tiles across all dimensions in the grid.
#[inline(always)]
pub fn total_tiles(&self) -> usize {
let (tx, ty, tz) = self.tile_count_xyz();
tx * ty * tz
}
/// Computes the 1D linear flat memory index for the current invocation.
///
/// Evaluates `x + y * size_x + z * size_x * size_y`.
/// Works symmetrically across 1D, 2D, and 3D execution domains.
#[inline(always)]
pub const fn index(&self) -> usize {
self.x + self.y * self.size_x + self.z * self.size_x * self.size_y
}
/// Computes the 1D linear cell index of this invocation inside the current tile.
///
/// Evaluates `cell_x + cell_y * tile_size_x + cell_z * tile_size_x * tile_size_y`.
#[inline(always)]
pub const fn cell_index(&self) -> usize {
self.cell_x
+ self.cell_y * self.tile_size_x
+ self.cell_z * self.tile_size_x * self.tile_size_y
}
/// Computes the 1D linear index of the current tile in the global tile grid.
#[inline(always)]
pub fn tile_index(&self) -> usize {
let (tx, ty) = self.tile_count_xy();
self.tile_x + self.tile_y * tx + self.tile_z * tx * ty
}
/// Returns the total execution domain size (`size_x * size_y * size_z`).
#[inline(always)]
pub const fn total_cells(&self) -> usize {
self.size_x * self.size_y * self.size_z
}
/// Returns `true` if the current invocation is within 1D bounds (`x < size_x`).
#[inline(always)]
pub const fn in_bounds_x(&self) -> bool {
self.x < self.size_x
}
/// Returns `true` if the current invocation is within 2D bounds (`x < size_x && y < size_y`).
#[inline(always)]
pub const fn in_bounds_xy(&self) -> bool {
self.x < self.size_x && self.y < self.size_y
}
/// Returns `true` if the current invocation is within 3D bounds (`x < size_x && y < size_y && z < size_z`).
#[inline(always)]
pub const fn in_bounds_xyz(&self) -> bool {
self.x < self.size_x && self.y < self.size_y && self.z < self.size_z
}
/// Returns `true` if the current invocation is within all active domain dimensions.
#[inline(always)]
pub const fn in_bounds(&self) -> bool {
self.in_bounds_xyz()
}
/// Returns `true` if this invocation is the leader of the current tile (`cell == (0, 0, 0)`).
///
/// Used in intra-tile reductions to perform single-thread writes to global memory after [`Self::sync`].
#[inline(always)]
pub const fn is_tile_leader(&self) -> bool {
self.cell_x == 0 && self.cell_y == 0 && self.cell_z == 0
}
/// Returns `true` if this invocation is the global origin of the entire dispatch (`pos == (0, 0, 0)`).
#[inline(always)]
pub const fn is_global_leader(&self) -> bool {
self.x == 0 && self.y == 0 && self.z == 0
}
/// Computes 2D normalized UV coordinates in `[0.0, 1.0]` with pixel-center alignment (`+0.5`).
#[inline(always)]
pub fn uv(&self) -> (f32, f32) {
(
(self.x as f32 + 0.5) / self.size_x as f32,
(self.y as f32 + 0.5) / self.size_y.max(1) as f32,
)
}
/// Computes 3D normalized volumetric UVW coordinates in `[0.0, 1.0]` with voxel-center alignment.
#[inline(always)]
pub fn uvw(&self) -> (f32, f32, f32) {
(
(self.x as f32 + 0.5) / self.size_x as f32,
(self.y as f32 + 0.5) / self.size_y.max(1) as f32,
(self.z as f32 + 0.5) / self.size_z.max(1) as f32,
)
}
/// Computes Normalized Device Coordinates (NDC) in `[-1.0, 1.0]` for ray generation.
#[inline(always)]
pub fn ndc(&self) -> (f32, f32) {
let (u, v) = self.uv();
(u * 2.0 - 1.0, 1.0 - v * 2.0)
}
/// Computes the aspect ratio of the 2D domain (`size_x / size_y`).
#[inline(always)]
pub fn aspect_ratio(&self) -> f32 {
self.size_x as f32 / self.size_y.max(1) as f32
}
/// Encodes normalized floating-point RGBA components into a packed `u32` integer (`0xAABBGGRR` / `0xAARRGGBB`).
#[inline(always)]
pub fn set_rgba_color(&self, r: f32, g: f32, b: f32, a: f32) -> u32 {
let a = (a.clamp(0.0, 1.0) * 255.0) as u32;
let r = (r.clamp(0.0, 1.0) * 255.0) as u32;
let g = (g.clamp(0.0, 1.0) * 255.0) as u32;
let b = (b.clamp(0.0, 1.0) * 255.0) as u32;
(a << 24) | (r << 16) | (g << 8) | b
}
/// Encodes normalized RGB components into a packed `u32` integer with full alpha (`a = 1.0`).
#[inline(always)]
pub fn set_rgb_color(&self, r: f32, g: f32, b: f32) -> u32 {
self.set_rgba_color(r, g, b, 1.0)
}
/// Encodes normalized RG components into a packed `u32` integer (`b = 0.0, a = 1.0`).
#[inline(always)]
pub fn set_rg_color(&self, r: f32, g: f32) -> u32 {
self.set_rgba_color(r, g, 0.0, 1.0)
}
/// Encodes normalized red component into a packed `u32` integer (`g = 0.0, b = 0.0, a = 1.0`).
#[inline(always)]
pub fn set_r_color(&self, r: f32) -> u32 {
self.set_rgba_color(r, 0.0, 0.0, 1.0)
}
/// Encodes normalized green component into a packed `u32` integer (`r = 0.0, b = 0.0, a = 1.0`).
#[inline(always)]
pub fn set_g_color(&self, g: f32) -> u32 {
self.set_rgba_color(0.0, g, 0.0, 1.0)
}
/// Encodes normalized blue component into a packed `u32` integer (`r = 0.0, g = 0.0, a = 1.0`).
#[inline(always)]
pub fn set_b_color(&self, b: f32) -> u32 {
self.set_rgba_color(0.0, 0.0, b, 1.0)
}
/// Encodes normalized alpha component into a packed `u32` integer with black color.
#[inline(always)]
pub fn set_a_color(&self, a: f32) -> u32 {
self.set_rgba_color(0.0, 0.0, 0.0, a)
}
}
#[unsafe(no_mangle)]
#[inline(never)]
#[cold]
pub extern "C" fn __enki_sync() {
std::sync::atomic::compiler_fence(std::sync::atomic::Ordering::SeqCst);
core::hint::black_box(());
}
</file>
<file path="enki_api/space/cpu.rs">
use super::core::Space;
impl Space {
/// Creates a 1D CPU execution point for coordinate `x` within domain size `size_x`.
///
/// Can be chained with [`.tile()`](Self::tile) to configure local tile coordinates.
#[inline(always)]
pub fn cpu_x(x: usize, size_x: usize) -> Self {
Self {
x,
y: 0,
z: 0,
cell_x: 0,
cell_y: 0,
cell_z: 0,
tile_x: x,
tile_y: 0,
tile_z: 0,
size_x: size_x.max(1),
size_y: 1,
size_z: 1,
tile_size_x: 1,
tile_size_y: 1,
tile_size_z: 1,
}
}
/// Creates a 2D CPU execution point for coordinate `(x, y)` within domain size `(size_x, size_y)`.
///
/// Can be chained with [`.tile()`](Self::tile) to configure local tile coordinates.
#[inline(always)]
pub fn cpu_xy(x: usize, y: usize, size_x: usize, size_y: usize) -> Self {
Self {
x,
y,
z: 0,
cell_x: 0,
cell_y: 0,
cell_z: 0,
tile_x: x,
tile_y: y,
tile_z: 0,
size_x: size_x.max(1),
size_y: size_y.max(1),
size_z: 1,
tile_size_x: 1,
tile_size_y: 1,
tile_size_z: 1,
}
}
/// Creates a 3D volumetric CPU execution point for coordinate `(x, y, z)` within domain size `(size_x, size_y, size_z)`.
///
/// Can be chained with [`.tile()`](Self::tile) to configure local tile coordinates.
#[inline(always)]
pub fn cpu_xyz(
x: usize,
y: usize,
z: usize,
size_x: usize,
size_y: usize,
size_z: usize,
) -> Self {
Self {
x,
y,
z,
cell_x: 0,
cell_y: 0,
cell_z: 0,
tile_x: x,
tile_y: y,
tile_z: z,
size_x: size_x.max(1),
size_y: size_y.max(1),
size_z: size_z.max(1),
tile_size_x: 1,
tile_size_y: 1,
tile_size_z: 1,
}
}
}
</file>
<file path="enki_api/space/mod.rs">
pub(crate) mod builder;
pub(crate) mod core;
pub(crate) mod cpu;
pub(crate) mod tile;
pub(crate) mod tuner;
pub use self::core::Space;
pub use self::tile::{IntoTile, TileConfig};
</file>
<file path="enki_api/space/tile.rs">
/// Configuration strategy for partitioning execution domains into spatial tiles.
#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Default)]
pub enum TileConfig {
/// Automatic tile sizing derived dynamically from device hardware specifications.
#[default]
Auto,
/// Explicit tile dimensions along the X, Y, and Z axes: `(tile_x, tile_y, tile_z)`.
Custom(u32, u32, u32),
}
impl TileConfig {
/// Creates an explicit 1D tile configuration along the X axis.
#[inline(always)]
pub const fn custom_x(x: u32) -> Self {
Self::Custom(x, 1, 1)
}
/// Creates an explicit 2D tile configuration along the X and Y axes.
#[inline(always)]
pub const fn custom_xy(x: u32, y: u32) -> Self {
Self::Custom(x, y, 1)
}
/// Creates an explicit 3D volumetric tile configuration along the X, Y, and Z axes.
#[inline(always)]
pub const fn custom_xyz(x: u32, y: u32, z: u32) -> Self {
Self::Custom(x, y, z)
}
/// Alias for [`Self::custom_x`].
#[inline(always)]
pub const fn custom_1d(x: u32) -> Self {
Self::custom_x(x)
}
/// Alias for [`Self::custom_xy`].
#[inline(always)]
pub const fn custom_2d(x: u32, y: u32) -> Self {
Self::custom_xy(x, y)
}
/// Alias for [`Self::custom_xyz`].
#[inline(always)]
pub const fn custom_3d(x: u32, y: u32, z: u32) -> Self {
Self::custom_xyz(x, y, z)
}
/// Returns `true` if this configuration is set to automatic hardware tuning.
#[inline(always)]
pub const fn is_auto(&self) -> bool {
matches!(self, Self::Auto)
}
}
/// Conversion trait enabling ergonomic construction of [`TileConfig`] from numbers, tuples, and arrays.
pub trait IntoTile {
/// Converts this value into a concrete [`TileConfig`].
fn into_tile(self) -> TileConfig;
}
impl IntoTile for TileConfig {
#[inline(always)]
fn into_tile(self) -> TileConfig {
self
}
}
impl IntoTile for u32 {
#[inline(always)]
fn into_tile(self) -> TileConfig {
TileConfig::Custom(self.max(1), 1, 1)
}
}
impl IntoTile for i32 {
#[inline(always)]
fn into_tile(self) -> TileConfig {
TileConfig::Custom(self.max(1) as u32, 1, 1)
}
}
impl IntoTile for usize {
#[inline(always)]
fn into_tile(self) -> TileConfig {
TileConfig::Custom((self as u32).max(1), 1, 1)
}
}
impl IntoTile for [u32; 2] {
#[inline(always)]
fn into_tile(self) -> TileConfig {
TileConfig::Custom(self[0].max(1), self[1].max(1), 1)
}
}
impl IntoTile for (u32, u32) {
#[inline(always)]
fn into_tile(self) -> TileConfig {
TileConfig::Custom(self.0.max(1), self.1.max(1), 1)
}
}
impl IntoTile for [i32; 2] {
#[inline(always)]
fn into_tile(self) -> TileConfig {
TileConfig::Custom(self[0].max(1) as u32, self[1].max(1) as u32, 1)
}
}
impl IntoTile for (i32, i32) {
#[inline(always)]
fn into_tile(self) -> TileConfig {
TileConfig::Custom(self.0.max(1) as u32, self.1.max(1) as u32, 1)
}
}
impl IntoTile for [usize; 2] {
#[inline(always)]
fn into_tile(self) -> TileConfig {
TileConfig::Custom((self[0] as u32).max(1), (self[1] as u32).max(1), 1)
}
}
impl IntoTile for (usize, usize) {
#[inline(always)]
fn into_tile(self) -> TileConfig {
TileConfig::Custom((self.0 as u32).max(1), (self.1 as u32).max(1), 1)
}
}
impl IntoTile for [u32; 3] {
#[inline(always)]
fn into_tile(self) -> TileConfig {
TileConfig::Custom(self[0].max(1), self[1].max(1), self[2].max(1))
}
}
impl IntoTile for (u32, u32, u32) {
#[inline(always)]
fn into_tile(self) -> TileConfig {
TileConfig::Custom(self.0.max(1), self.1.max(1), self.2.max(1))
}
}
impl IntoTile for [i32; 3] {
#[inline(always)]
fn into_tile(self) -> TileConfig {
TileConfig::Custom(
self[0].max(1) as u32,
self[1].max(1) as u32,
self[2].max(1) as u32,
)
}
}
impl IntoTile for (i32, i32, i32) {
#[inline(always)]
fn into_tile(self) -> TileConfig {
TileConfig::Custom(
self.0.max(1) as u32,
self.1.max(1) as u32,
self.2.max(1) as u32,
)
}
}
impl IntoTile for [usize; 3] {
#[inline(always)]
fn into_tile(self) -> TileConfig {
TileConfig::Custom(
(self[0] as u32).max(1),
(self[1] as u32).max(1),
(self[2] as u32).max(1),
)
}
}
impl IntoTile for (usize, usize, usize) {
#[inline(always)]
fn into_tile(self) -> TileConfig {
TileConfig::Custom(
(self.0 as u32).max(1),
(self.1 as u32).max(1),
(self.2 as u32).max(1),
)
}
}
</file>
<file path="enki_api/space/tuner.rs">
//! Module: enki_api/space/tuner.rs
//!
//! Hardware-agnostic mathematical engine for deriving optimal spatial tile dimensions.
use super::tile::TileConfig;
use parsu::profile::HardwareProfile;
pub struct SpatialAutoTuner;
impl SpatialAutoTuner {
/// Resolves the final (tile_x, tile_y, tile_z) based on the requested TileConfig,
/// problem dimensions, and the queried hardware profile.
pub fn resolve_tile(
config: TileConfig,
global_size: (u32, u32, u32),
profile: &HardwareProfile,
) -> (u32, u32, u32) {
let (gx, gy, gz) = (
global_size.0.max(1),
global_size.1.max(1),
global_size.2.max(1),
);
match config {
TileConfig::Custom(cx, cy, cz) => Self::validate_and_clamp_custom(cx, cy, cz, profile),
TileConfig::Auto => {
let s = if profile.subgroup_size == 0 {
32
} else {
profile.subgroup_size
};
let m = if profile.max_compute_workgroup_invocations == 0 {
1024
} else {
profile.max_compute_workgroup_invocations
};
if gz > 1 {
Self::tune_3d(gx, gy, gz, s, m, profile.is_integrated)
} else if gy > 1 {
Self::tune_2d(gx, gy, s, m, profile.is_integrated)
} else {
Self::tune_1d(gx, s, m, profile.is_integrated)
}
}
}
}
/// Computes the target invocation count per workgroup (T) based on subgroup size.
/// Latency hiding sweet spot is between 4 and 8 subgroups, clamped by hardware limit (M).
#[inline(always)]
fn target_invocations(s: u32, m: u32, is_integrated: bool) -> u32 {
if is_integrated {
(2 * s).min(m).min(64)
} else {
let ideal = (8 * s).min(m);
if s <= 32 {
ideal.min(256)
} else {
ideal.min(512)
}
}
}
/// 1D Auto-tuning: Aligns to subgroup size (S) and avoids trailing inactive threads for small N.
fn tune_1d(n: u32, s: u32, m: u32, is_integrated: bool) -> (u32, u32, u32) {
let target = Self::target_invocations(s, m, is_integrated);
// If the problem size N is smaller than the target, allocate the smallest multiple of S
let needed_subgroups = (n + s - 1) / s;
let needed_threads = (needed_subgroups * s).max(s);
let final_x = target.min(needed_threads).min(m);
(final_x, 1, 1)
}
/// 2D Auto-tuning: Geometric 2D factorization (Lx * Ly = T) optimizing L1/L2 Spatial Locality.
fn tune_2d(_w: u32, _h: u32, s: u32, m: u32, is_integrated: bool) -> (u32, u32, u32) {
let target = Self::target_invocations(s, m, is_integrated);
// Decompose target T into (Lx, Ly) power-of-two rectangles where Lx >= Ly
let ly = 1u32 << (target.ilog2() / 2);
let lx = target / ly;
(lx, ly, 1)
}
/// 3D Auto-tuning: Volumetric factorization (Lx * Ly * Lz = T).
fn tune_3d(_x: u32, _y: u32, _z: u32, s: u32, m: u32, is_integrated: bool) -> (u32, u32, u32) {
let target = Self::target_invocations(s, m, is_integrated);
let lz = 1u32 << (target.ilog2() / 3);
let remainder = target / lz;
let ly = 1u32 << (remainder.ilog2() / 2);
let lx = remainder / ly;
(lx, ly, lz)
}
/// Validates user custom tiles against physical device limits.
fn validate_and_clamp_custom(
cx: u32,
cy: u32,
cz: u32,
profile: &HardwareProfile,
) -> (u32, u32, u32) {
let x = cx.max(1);
let y = cy.max(1);
let z = cz.max(1);
let total = (x as u64) * (y as u64) * (z as u64);
let m = if profile.max_compute_workgroup_invocations == 0 {
1024
} else {
profile.max_compute_workgroup_invocations as u64
};
if total > m {
let diag = anu::diagnostics::hw::tile_invocations_exceeded(None);
crate::enki_api::context::errors::emit_warning(&diag);
let clamped_x = ((m / ((y * z) as u64)) as u32).max(1);
(clamped_x, y, z)
} else {
(x, y, z)
}
}
}
</file>
<file path="enki_api/type_safety/arg_match.rs">
use std::mem::size_of;
use crate::enki_api::resources::atomic::GpuAtomicTarget;
use crate::enki_api::resources::{GpuAtomic, GpuAtomicVec, GpuParam, GpuTileMem};
use anu::nam_args_api::{
AccessIntent, ArgDescriptor, ArgValue, IngressContext, InputResourceRecord, ResourceAccessKind,
};
pub trait GpuTypeMatch<'target> {
type Target;
fn describe() -> ArgDescriptor;
fn collect<'a>(&self, ctx: &mut IngressContext<'a>);
fn as_input_record(&self, arg_index: usize) -> InputResourceRecord;
}
// 1. GpuParam
impl<'target, T: Copy + Send + Sync + 'static> GpuTypeMatch<'target> for GpuParam<T> {
type Target = T;
fn describe() -> ArgDescriptor {
ArgDescriptor::by_value::<T>()
}
fn collect<'a>(&self, ctx: &mut IngressContext<'a>) {
let size = size_of::<T>();
let bytes =
unsafe { std::slice::from_raw_parts(&self.value as *const T as *const u8, size) }
.to_vec();
ctx.push_arg(ArgValue::Payload(bytes), Self::describe());
}
fn as_input_record(&self, arg_index: usize) -> InputResourceRecord {
InputResourceRecord {
arg_index,
root_id: None,
access_kind: ResourceAccessKind::ValueUniform {
byte_size: size_of::<T>(),
},
is_mutable: false,
element_type_name: std::any::type_name::<T>(),
}
}
}
impl<'target, T: Copy + Send + Sync + 'static> GpuTypeMatch<'target> for &GpuParam<T> {
type Target = T;
fn describe() -> ArgDescriptor {
ArgDescriptor::by_value::<T>()
}
fn collect<'a>(&self, ctx: &mut IngressContext<'a>) {
GpuTypeMatch::collect(*self, ctx);
}
fn as_input_record(&self, arg_index: usize) -> InputResourceRecord {
GpuTypeMatch::as_input_record(*self, arg_index)
}
}
// 2. GpuTileMem
impl<'target, T: Copy + Send + Sync + 'static, const N: usize> GpuTypeMatch<'target>
for GpuTileMem<T, N>
{
type Target = &'target mut [T; N];
fn describe() -> ArgDescriptor {
ArgDescriptor::workgroup_scratchpad::<T, N>()
}
fn collect<'a>(&self, ctx: &mut IngressContext<'a>) {
ctx.push_arg(ArgValue::ZeroFootprint, Self::describe());
}
fn as_input_record(&self, arg_index: usize) -> InputResourceRecord {
InputResourceRecord {
arg_index,
root_id: None,
access_kind: ResourceAccessKind::TileScratchpad { element_count: N },
is_mutable: true,
element_type_name: std::any::type_name::<T>(),
}
}
}
impl<'target, T: Copy + Send + Sync + 'static, const N: usize> GpuTypeMatch<'target>
for &'target GpuTileMem<T, N>
{
type Target = &'target mut [T; N];
fn describe() -> ArgDescriptor {
ArgDescriptor::workgroup_scratchpad::<T, N>()
}
fn collect<'a>(&self, ctx: &mut IngressContext<'a>) {
GpuTypeMatch::collect(*self, ctx);
}
fn as_input_record(&self, arg_index: usize) -> InputResourceRecord {
GpuTypeMatch::as_input_record(*self, arg_index)
}
}
// 3. Atomics
impl<'target, T: GpuAtomicTarget> GpuTypeMatch<'target> for &'target GpuAtomic<T> {
type Target = &'target T::NamTarget;
fn describe() -> ArgDescriptor {
ArgDescriptor::atomic_cell::<T>()
}
fn collect<'a>(&self, ctx: &mut IngressContext<'a>) {
ctx.push_arg(ArgValue::BufferBDA(self.device_address), Self::describe());
ctx.push_resource_binding(
format!("GpuAtomic_Slot_{}", self.slot_index),
self.slot_index,
self.state.clone(),
AccessIntent::ReadWrite,
);
}
fn as_input_record(&self, arg_index: usize) -> InputResourceRecord {
InputResourceRecord {
arg_index,
root_id: Some(self.slot_index as usize),
access_kind: ResourceAccessKind::Atomic { element_count: 1 },
is_mutable: true,
element_type_name: std::any::type_name::<T>(),
}
}
}
impl<'target, T: GpuAtomicTarget> GpuTypeMatch<'target> for &'target mut GpuAtomic<T> {
type Target = &'target T::NamTarget;
fn describe() -> ArgDescriptor {
ArgDescriptor::atomic_cell::<T>()
}
fn collect<'a>(&self, ctx: &mut IngressContext<'a>) {
ctx.push_arg(ArgValue::BufferBDA(self.device_address), Self::describe());
ctx.push_resource_binding(
format!("GpuAtomic_Slot_{}", self.slot_index),
self.slot_index,
self.state.clone(),
AccessIntent::ReadWrite,
);
}
fn as_input_record(&self, arg_index: usize) -> InputResourceRecord {
InputResourceRecord {
arg_index,
root_id: Some(self.slot_index as usize),
access_kind: ResourceAccessKind::Atomic { element_count: 1 },
is_mutable: true,
element_type_name: std::any::type_name::<T>(),
}
}
}
impl<'target, T: GpuAtomicTarget> GpuTypeMatch<'target> for &'target GpuAtomicVec<T> {
type Target = &'target [T::NamTarget];
fn describe() -> ArgDescriptor {
ArgDescriptor::atomic_slice::<T>()
}
fn collect<'a>(&self, ctx: &mut IngressContext<'a>) {
ctx.push_arg(
ArgValue::SliceBDA {
bda: self.device_address,
count: self.element_count as u64,
},
Self::describe(),
);
ctx.push_resource_binding(
format!("GpuAtomicVec_Slot_{}", self.slot_index),
self.slot_index,
self.state.clone(),
AccessIntent::ReadWrite,
);
}
fn as_input_record(&self, arg_index: usize) -> InputResourceRecord {
InputResourceRecord {
arg_index,
root_id: Some(self.slot_index as usize),
access_kind: ResourceAccessKind::Atomic {
element_count: self.element_count,
},
is_mutable: true,
element_type_name: std::any::type_name::<T>(),
}
}
}
impl<'target, T: GpuAtomicTarget> GpuTypeMatch<'target> for &'target mut GpuAtomicVec<T> {
type Target = &'target [T::NamTarget];
fn describe() -> ArgDescriptor {
ArgDescriptor::atomic_slice::<T>()
}
fn collect<'a>(&self, ctx: &mut IngressContext<'a>) {
ctx.push_arg(
ArgValue::SliceBDA {
bda: self.device_address,
count: self.element_count as u64,
},
Self::describe(),
);
ctx.push_resource_binding(
format!("GpuAtomicVec_Slot_{}", self.slot_index),
self.slot_index,
self.state.clone(),
AccessIntent::ReadWrite,
);
}
fn as_input_record(&self, arg_index: usize) -> InputResourceRecord {
InputResourceRecord {
arg_index,
root_id: Some(self.slot_index as usize),
access_kind: ResourceAccessKind::Atomic {
element_count: self.element_count,
},
is_mutable: true,
element_type_name: std::any::type_name::<T>(),
}
}
}
</file>
<file path="enki_api/type_safety/mod.rs">
pub mod arg_match;
pub mod nam_run;
pub use arg_match::GpuTypeMatch;
pub use nam_run::*;
</file>
<file path="enki_api/type_safety/nam_run.rs">
use super::arg_match::GpuTypeMatch;
use crate::enki_api::context::ambient::with_active_flow;
use crate::enki_api::space::Space;
use anu::nam_args_api::{DispatchSafetyMode, NamDispatchMap, SpaceContract};
macro_rules! define_nam_run_traits {
(NamRun0) => {
/// Execution trait for 0-argument `#[nam]` functions.
pub trait NamRun0 {
/// Executes the nam function across the specified spatial domain in safe mode.
fn run(self, space: &Space);
/// Executes the nam function across the specified spatial domain in unchecked mode.
unsafe fn run_unchecked(self, space: &Space);
}
impl<F> NamRun0 for F
where
F: Fn(&Space) + 'static,
{
#[track_caller]
#[inline(always)]
fn run(self, space: &Space) {
let caller = std::panic::Location::caller();
dispatch_nam_internal::<F>(self, space, caller, DispatchSafetyMode::Safe, Vec::new(), anu::nam_args_api::IngressContext::new());
}
#[track_caller]
#[inline(always)]
unsafe fn run_unchecked(self, space: &Space) {
let caller = std::panic::Location::caller();
dispatch_nam_internal::<F>(self, space, caller, DispatchSafetyMode::Unchecked, Vec::new(), anu::nam_args_api::IngressContext::new());
}
}
};
($(($Trait:ident, $($Arg:ident),*));* $(;)?) => {
$(
/// Execution trait for `#[nam]` functions with arguments.
pub trait $Trait<'a, $($Arg),*> {
/// Executes the nam function across the specified spatial domain in safe mode.
///
/// # Active Verification
/// Enforces parameter contracts (`E1001`–`E1004`), domain bounds (`E1008`),
/// slice range disjointness (`E1007`), temporal presentation mutability (`E1010`),
/// and forbids unconstrained parallel mutable slices (`E1009`).
#[allow(non_snake_case)]
fn run(self, space: &Space, $($Arg: $Arg),*);
/// Executes the nam function across the specified spatial domain in unchecked mode.
///
/// Bypasses the parallel mutable slice policy (`E1009`), permitting [`SliceMut`]
/// arguments across multiple threads when writes are coordinated manually.
///
/// # Safety & Active Verification
/// Does **not** disable all validation: type contracts (`E1001`–`E1004`),
/// domain bounds (`E1008`), slice range collisions (`E1007`), and temporal presentation
/// hazards (`E1010`) remain actively enforced by the runtime `BorrowEngine`.
#[allow(non_snake_case)]
unsafe fn run_unchecked(self, space: &Space, $($Arg: $Arg),*);
}
impl<'a, 'target, F, $($Arg),*> $Trait<'a, $($Arg),*> for F
where
$($Arg: GpuTypeMatch<'target>),*,
F: Fn(&Space, $($Arg::Target),*) + 'static,
{
#[track_caller]
#[inline(always)]
#[allow(non_snake_case)]
#[allow(unused_assignments)]
fn run(self, space: &Space, $($Arg: $Arg),*) {
let caller = std::panic::Location::caller();
let mut ctx = anu::nam_args_api::IngressContext::new();
let mut inputs = Vec::new();
let mut arg_idx = 0;
$(
ctx.current_arg_index = arg_idx;
$Arg.collect(&mut ctx);
inputs.push($Arg.as_input_record(arg_idx));
arg_idx += 1;
)*
dispatch_nam_internal::<F>(self, space, caller, DispatchSafetyMode::Safe, inputs, ctx);
}
#[track_caller]
#[inline(always)]
#[allow(non_snake_case)]
#[allow(unused_assignments)]
unsafe fn run_unchecked(self, space: &Space, $($Arg: $Arg),*) {
let caller = std::panic::Location::caller();
let mut ctx = anu::nam_args_api::IngressContext::new();
let mut inputs = Vec::new();
let mut arg_idx = 0;
$(
ctx.current_arg_index = arg_idx;
$Arg.collect(&mut ctx);
inputs.push($Arg.as_input_record(arg_idx));
arg_idx += 1;
)*
dispatch_nam_internal::<F>(self, space, caller, DispatchSafetyMode::Unchecked, inputs, ctx);
}
}
)*
};
}
#[inline(always)]
fn dispatch_nam_internal<F: 'static>(
_func: F,
space: &Space,
caller: &'static std::panic::Location<'static>,
safety_mode: DispatchSafetyMode,
inputs: Vec<anu::nam_args_api::InputResourceRecord>,
ctx: anu::nam_args_api::IngressContext<'static>,
) {
with_active_flow(|flow| {
if flow.sticky_error.is_some() {
return;
}
let space_contract = SpaceContract::new(space.size_x, space.size_y, space.size_z);
let call_site = Some((caller.file(), caller.line(), caller.column()));
let mut map = NamDispatchMap::new(String::new(), safety_mode, space_contract, call_site);
map.inputs = inputs;
if let Err(e) = flow.nam_impl_direct::<F>(space, ctx, map) {
flow.sticky_error = Some(e);
}
});
}
define_nam_run_traits!(NamRun0);
define_nam_run_traits! {
(NamRun1, T0);
(NamRun2, T0, T1);
(NamRun3, T0, T1, T2);
(NamRun4, T0, T1, T2, T3);
(NamRun5, T0, T1, T2, T3, T4);
(NamRun6, T0, T1, T2, T3, T4, T5);
(NamRun7, T0, T1, T2, T3, T4, T5, T6);
(NamRun8, T0, T1, T2, T3, T4, T5, T6, T7);
(NamRun9, T0, T1, T2, T3, T4, T5, T6, T7, T8);
(NamRun10, T0, T1, T2, T3, T4, T5, T6, T7, T8, T9);
(NamRun11, T0, T1, T2, T3, T4, T5, T6, T7, T8, T9, T10);
(NamRun12, T0, T1, T2, T3, T4, T5, T6, T7, T8, T9, T10, T11);
(NamRun13, T0, T1, T2, T3, T4, T5, T6, T7, T8, T9, T10, T11, T12);
(NamRun14, T0, T1, T2, T3, T4, T5, T6, T7, T8, T9, T10, T11, T12, T13);
(NamRun15, T0, T1, T2, T3, T4, T5, T6, T7, T8, T9, T10, T11, T12, T13, T14);
(NamRun16, T0, T1, T2, T3, T4, T5, T6, T7, T8, T9, T10, T11, T12, T13, T14, T15);
}
</file>
<file path="enki_api/mod.rs">
pub mod context;
pub mod resources;
pub mod space;
pub mod type_safety;
pub use crate::gpu_vec;
pub use context::{Enki, Flow, GpuInstant};
pub use resources::{GpuAtomic, GpuAtomicVec, GpuParam, GpuTileMem, GpuVec, Slice, SliceMut};
pub use space::Space;
pub use type_safety::*;
</file>
<file path="enki_cli/src/bin/cargo-enki.rs">
fn main() {
cargo_enki::run();
}
</file>
<file path="enki_cli/src/bin/enki.rs">
fn main() {
cargo_enki::run();
}
</file>
<file path="enki_cli/src/args.rs">
#[derive(Debug, PartialEq, Eq)]
pub enum Action {
Run { cargo_args: Vec<String> },
Debug { token: String },
Help,
Version,
Unknown(String),
}
pub fn parse() -> Action {
let mut raw_args: Vec<String> = std::env::args().skip(1).collect();
if let Some(first) = raw_args.first() {
if first == "enki" {
raw_args.remove(0);
}
}
if raw_args.is_empty() {
return Action::Help;
}
let command = &raw_args[0];
match command.as_str() {
"run" => {
let cargo_args = raw_args[1..].to_vec();
Action::Run { cargo_args }
}
"debug" | "decode" => {
let token = if raw_args.len() > 1 {
raw_args[1..].join(" ")
} else {
println!("Paste the Enki Crash Token below (then press Enter):");
let mut buffer = String::new();
let _ = std::io::stdin().read_line(&mut buffer);
buffer.trim().to_string()
};
Action::Debug { token }
}
"-h" | "--help" | "help" => Action::Help,
"-V" | "--version" | "version" => Action::Version,
other => Action::Unknown(other.to_string()),
}
}
</file>
<file path="enki_cli/src/decoder.rs">
use base64::Engine;
use sha2::{Digest, Sha256};
use std::io::{self, Write};
const MASTER_PIN_HASH: &str = "218e19b44b9de9e57aad8f3520a0605838eb58af2fa1ac49a4a080acf35ee421";
const SALT: &str = "ENKI_DEV_SECRET_SALT_2026";
const PUBLIC_KEY_SEED: [u8; 32] = [
0x45, 0x6E, 0x6B, 0x69, 0x5F, 0x50, 0x61, 0x72, 0x73, 0x75, 0x5F, 0x53, 0x65, 0x63, 0x72, 0x65,
0x74, 0x4B, 0x65, 0x79, 0x5F, 0x32, 0x30, 0x32, 0x36, 0x5F, 0x41, 0x6C, 0x70, 0x68, 0x61, 0x21,
];
pub fn execute_debug(raw_token: &str) {
print!("\x1b[1;94m --> \x1b[0m\x1b[1mEnter Developer Passcode:\x1b[0m ");
let _ = io::stdout().flush();
let mut input = String::new();
if io::stdin().read_line(&mut input).is_err() {
eprintln!("\n\x1b[1;91merror\x1b[0m: failed to read input.");
std::process::exit(1);
}
let entered_pin = input.trim();
let mut hasher = Sha256::new();
hasher.update(entered_pin.as_bytes());
hasher.update(SALT.as_bytes());
let computed_hash = format!("{:x}", hasher.finalize());
if computed_hash != MASTER_PIN_HASH {
eprintln!("\n\x1b[1;91merror\x1b[0m: access denied: invalid developer passcode.");
std::process::exit(1);
}
let b64_clean: String = raw_token
.lines()
.map(|l| l.trim())
.filter(|l| !l.starts_with("-----") && !l.is_empty())
.collect();
if b64_clean.is_empty() {
eprintln!("\n\x1b[1;91merror\x1b[0m: empty or invalid crash token payload.");
std::process::exit(1);
}
let mut data = match base64::engine::general_purpose::STANDARD.decode(&b64_clean) {
Ok(d) => d,
Err(e) => {
eprintln!("\n\x1b[1;91merror\x1b[0m: Base64 decoding failed: {e}");
std::process::exit(1);
}
};
for i in 0..data.len() {
let b = data[i];
let unrotated = (b >> 3) | (b << 5);
let k = PUBLIC_KEY_SEED[i % 32];
let unxored = unrotated ^ (k.wrapping_add((i & 0xFF) as u8));
data[i] = unxored;
}
let json_str = match String::from_utf8(data) {
Ok(s) => s,
Err(e) => {
eprintln!("\n\x1b[1;91merror\x1b[0m: decrypted payload is not valid UTF-8: {e}");
std::process::exit(1);
}
};
let report: serde_json::Value = match serde_json::from_str(&json_str) {
Ok(v) => v,
Err(e) => {
eprintln!("\n\x1b[1;91merror\x1b[0m: failed to parse decrypted JSON: {e}");
std::process::exit(1);
}
};
println!(
"\x1b[1;32m====================================================================\x1b[0m"
);
println!("\x1b[1;32m>>> ENKI PARSU CRASH FLIGHT RECORDER DECRYPTED REPORT <<<\x1b[0m");
println!(
"\x1b[1;33mTarget Nam:\x1b[0m {}",
report["nam"].as_str().unwrap_or("unknown")
);
println!(
"\x1b[1;33mCompiler Phase:\x1b[0m {}",
report["phase"].as_str().unwrap_or("unknown")
);
println!(
"\x1b[1;33mOS & Platform:\x1b[0m {}",
report["os"].as_str().unwrap_or("unknown")
);
println!(
"\x1b[1;33mWorkgroup Size:\x1b[0m {}",
report["workgroup"]
);
println!(
"\x1b[1;33mArgs Count:\x1b[0m {}",
report["args_count"]
);
println!(
"\x1b[1;91mFailed Source:\x1b[0m {}:{}",
report["inv_file"].as_str().unwrap_or("?"),
report["inv_line"]
);
println!(
"\x1b[1;91mFunction:\x1b[0m {}",
report["inv_func"].as_str().unwrap_or("?")
);
println!(
"\x1b[1;91mInvariant Message:\x1b[0m {}",
report["message"].as_str().unwrap_or("?")
);
let type_a = report["type_a"].as_str().unwrap_or("none");
if type_a != "none" {
println!("\x1b[1;36mType A:\x1b[0m {type_a}");
}
let type_b = report["type_b"].as_str().unwrap_or("none");
if type_b != "none" {
println!("\x1b[1;36mType B:\x1b[0m {type_b}");
}
println!(
"\x1b[1;32m====================================================================\x1b[0m\n"
);
}
</file>
<file path="enki_cli/src/lib.rs">
pub mod args;
pub mod decoder;
pub mod runner;
use args::Action;
pub fn run() {
let action = args::parse();
match action {
Action::Run { cargo_args } => {
runner::execute_run(&cargo_args);
}
Action::Debug { token } => {
decoder::execute_debug(&token);
}
Action::Help => {
print_help();
}
Action::Version => {
println!("enki {}", env!("CARGO_PKG_VERSION"));
}
Action::Unknown(subcmd) => {
eprintln!("\x1b[1;91merror\x1b[0m: no such subcommand: `{subcmd}`");
eprintln!("\n For more information, try `enki --help`\n");
std::process::exit(1);
}
}
}
fn print_help() {
println!(
"\
enki {version}
The command-line interface for the Enki heterogeneous GPU compute platform
USAGE:
enki [OPTIONS] [SUBCOMMAND]
cargo enki [OPTIONS] [SUBCOMMAND]
SUBCOMMANDS:
run Compile and run the current package with GPU JIT support
OPTIONS:
-h, --help Print help information
-V, --version Print version information
EXAMPLES:
enki run
enki run --release
enki run --bin my_app
enki run --example demo -- --custom-arg
",
version = env!("CARGO_PKG_VERSION")
);
}
</file>
<file path="enki_cli/src/runner.rs">
use std::process::{Command, Stdio};
pub fn execute_run(cargo_args: &[String]) -> ! {
let mut cmd = Command::new("cargo");
cmd.arg("run");
cmd.arg("--config").arg("profile.dev.opt-level=2");
cmd.arg("--config")
.arg("profile.dev.package.\"*\".opt-level=2");
for arg in cargo_args {
cmd.arg(arg);
}
let current_rustflags = std::env::var("RUSTFLAGS").unwrap_or_default();
let enki_rustflags = if current_rustflags.is_empty() {
"--emit=llvm-bc".to_string()
} else if current_rustflags.contains("--emit=llvm-bc") {
current_rustflags
} else {
format!("{current_rustflags} --emit=llvm-bc")
};
cmd.env("RUSTFLAGS", enki_rustflags);
cmd.stdin(Stdio::inherit())
.stdout(Stdio::inherit())
.stderr(Stdio::inherit());
match cmd.status() {
Ok(status) => {
let code = status.code().unwrap_or(1);
std::process::exit(code);
}
Err(err) => {
eprintln!("\x1b[1;91merror\x1b[0m: failed to execute `cargo run`: {err}");
std::process::exit(1);
}
}
}
</file>
<file path="enki_cli/Cargo.toml">
[package]
name = "cargo-enki"
version = "0.1.0"
edition = "2024"
default-run = "enki"
authors = ["Enki Dev <enkiruntime@proton.me>"]
license = "MIT OR Apache-2.0"
description = "CLI toolchain and runner for the Enki heterogeneous GPU compute platform"
repository = "https://github.com/enkiruntime/enki"
keywords = ["gpu", "cli", "cargo-subcommand", "vulkan", "enki"]
categories = ["command-line-utilities", "development-tools::cargo-plugins"]
readme = "README.md"
[dependencies]
sha2 = "0.10"
base64 = "0.22"
serde_json = "1.0"
</file>
<file path="enki_cli/README.md">
# cargo-enki
CLI toolchain and runner for the Enki heterogeneous GPU compute platform.
Part of the [Enki](https://github.com/enkiruntime/enki) ecosystem.
## Installation
```bash
cargo install cargo-enki
```
## Usage
```bash
# Run with GPU JIT support
cargo enki run
# Or directly
enki run
enki run --release
```
## Crash Token Decoder
```bash
enki debug <TOKEN>
```
</file>
<file path="enki_macros/src/lib.rs">
extern crate proc_macro;
use proc_macro::TokenStream;
use quote::quote;
use syn::{FnArg, ItemFn, Pat, Type};
#[proc_macro_attribute]
pub fn nam(_attr: TokenStream, item: TokenStream) -> TokenStream {
let parsed_fn = match syn::parse::<ItemFn>(item.clone()) {
Ok(f) => f,
Err(_) => return item,
};
let fn_name = &parsed_fn.sig.ident;
let fn_name_str = fn_name.to_string();
let mut param_metas = Vec::new();
for (i, input) in parsed_fn.sig.inputs.iter().enumerate() {
if let FnArg::Typed(pat_type) = input {
let param_name = match &*pat_type.pat {
Pat::Ident(pi) => pi.ident.to_string(),
_ => format!("arg_{i}"),
};
if param_name == "space" || i == 0 {
continue;
}
let type_str = quote!(#pat_type.ty).to_string();
let is_mutable = match &*pat_type.ty {
Type::Reference(tr) => tr.mutability.is_some(),
_ => false,
};
let is_slice = type_str.contains('[') || type_str.contains("Slice");
let is_atomic = type_str.contains("Atomic");
let is_tile_mem = type_str.contains("TileMem");
let is_ref = matches!(&*pat_type.ty, Type::Reference(_));
let kind_token = if is_atomic {
quote!(::enki::ExpectedParamKind::Atomic)
} else if is_tile_mem {
quote!(::enki::ExpectedParamKind::TileScratchpad)
} else if is_slice {
quote!(::enki::ExpectedParamKind::Slice)
} else if !is_ref {
quote!(::enki::ExpectedParamKind::ByValue)
} else {
quote!(::enki::ExpectedParamKind::PerCell)
};
param_metas.push(quote! {
::enki::NamParamMeta {
name: #param_name,
type_str: #type_str,
line: line!(),
column: column!(),
is_mutable: #is_mutable,
expected_kind: #kind_token,
}
});
}
}
let contract_ident = quote::format_ident!("__ENKI_CONTRACT_{}", fn_name);
let expanded = quote! {
#[unsafe(no_mangle)]
#[allow(non_snake_case)]
#parsed_fn
#[doc(hidden)]
#[allow(non_upper_case_globals)]
#[unsafe(no_mangle)]
pub static #contract_ident: ::enki::NamSignatureContract = ::enki::NamSignatureContract {
nam_name: #fn_name_str,
file_path: file!(),
line: line!(),
params: &[
#(#param_metas),*
],
};
};
TokenStream::from(expanded)
}
</file>
<file path="enki_macros/Cargo.toml">
[package]
name = "enki_macros"
version = "0.1.0"
edition = "2024"
authors = ["Enki Dev <enkiruntime@proton.me>"]
license = "MIT OR Apache-2.0"
description = "Procedural macros for Enki GPU kernel"
repository = "https://github.com/enkiruntime/enki"
keywords = ["gpu", "proc-macro", "macros", "vulkan", "enki"]
categories = ["development-tools::procedural-macro-helpers"]
readme = "README.md"
[lib]
proc-macro = true
[dependencies]
syn = { version = "2.0", features = ["full"] }
quote = "1.0"
</file>
<file path="enki_macros/README.md">
# enki_macros
Procedural attribute macro `#[nam]` for the Enki heterogeneous compute platform
Part of the [Enki](https://github.com/enkiruntime/enki) ecosystem
</file>
<file path="lib.rs">
//! # Enki
//!
//! Pure Rust heterogeneous GPU compute platform with Just-In-Time (JIT) compilation.
//!
//! Enki enables writing high-performance compute and graphics algorithms using standard,
//! idiomatic Rust (structs, traits, pattern matching, iters, atomics, enums and payloads, macros, and `crates.io` dependencies)
//! and executes them across GPU and CPU hardware with zero-cost abstractions.
pub mod enki_api;
// Core Execution Context
pub use enki_api::context::{
Enki, EnkiBuilder, Flow, GpuInstant, active_engine, active_enki, set_active_enki,
};
// Spatial Domain & Topology
pub use enki_api::space::{IntoTile, Space, TileConfig};
// GPU Resources & Memory Types
pub use enki_api::resources::{
GpuAtomic, GpuAtomicTarget, GpuAtomicVec, GpuParam, GpuTileMem, GpuVec, Slice, SliceMut,
};
// Safe Nam Execution Traits & Ingress Reflection
pub use enki_api::type_safety::*;
// Re-export contract types directly at root for macros
pub use anu::nam_args_api::{ExpectedParamKind, NamParamMeta, NamSignatureContract};
// Procedural Macro for GPU Nam Functions
pub use enki_macros::nam;
/// Convenient re-exports of common Enki types and traits.
pub mod prelude {
pub use super::enki_api::context::{
Enki, EnkiBuilder, Flow, GpuInstant, active_engine, active_enki, set_active_enki,
};
pub use super::enki_api::resources::{
GpuAtomic, GpuAtomicTarget, GpuAtomicVec, GpuParam, GpuTileMem, GpuVec, Slice, SliceMut,
};
pub use super::enki_api::space::{IntoTile, Space, TileConfig};
pub use super::enki_api::type_safety::*;
pub use super::gpu_vec;
pub use super::nam;
pub use super::{ExpectedParamKind, NamParamMeta, NamSignatureContract};
}
</file>
</files>