Skip to main content

ruprim_host/simd/
mod.rs

1//! SIMD-optimized kernels for tensor operations.
2//!
3//! Provides portable SIMD implementations via `macerator` with automatic
4//! dispatch to the best available instruction set:
5//! - aarch64: NEON
6//! - x86_64: AVX2, AVX512, SSE
7//! - wasm32: SIMD128
8//! - Other: Scalar fallback
9//!
10//! Enable with the `simd` feature flag (enabled by default).
11
12// Portable SIMD kernels using macerator (reductions, scatter-add)
13#[cfg(feature = "simd")]
14pub mod kernels;
15
16// SIMD-aligned memory allocation
17#[cfg(feature = "simd")]
18pub mod aligned;
19
20// When simd feature enabled: use portable macerator for binary/comparison/bool ops
21#[cfg(feature = "simd")]
22mod portable;
23
24#[cfg(feature = "simd")]
25pub use portable::{
26    CmpOp, abs_inplace_f32, add_inplace_f32, add_shared_row_inplace_f32, bool_and_inplace_u8,
27    bool_and_u8, bool_not_inplace_u8, bool_not_u8, bool_or_inplace_u8, bool_or_u8,
28    bool_xor_inplace_u8, bool_xor_u8, cmp_f32, cmp_scalar_f32, div_inplace_f32,
29    div_shared_row_inplace_f32, mask_fill_f32, mask_fill_f64, mask_fill_i64, mask_fill_u8,
30    mask_where_f32, mask_where_f64, mask_where_i64, mask_where_u8, mul_inplace_f32,
31    mul_shared_row_inplace_f32, recip_inplace_f32, sub_inplace_f32, sub_shared_row_inplace_f32,
32};
33
34// When simd feature disabled: use scalar fallback (bool ops + CmpOp only)
35#[cfg(not(feature = "simd"))]
36mod scalar;
37
38#[cfg(not(feature = "simd"))]
39pub use scalar::{
40    CmpOp, bool_and_inplace_u8, bool_and_u8, bool_not_inplace_u8, bool_not_u8, bool_or_inplace_u8,
41    bool_or_u8, bool_xor_inplace_u8, bool_xor_u8,
42};