Skip to main content

strided_basic/
lib.rs

1#![doc = include_str!("../README.md")]
2
3//! Cache-optimized kernels for strided multidimensional array operations.
4//!
5//! This crate is a Rust port of Julia's [Strided.jl](https://github.com/Jutho/Strided.jl)
6//! and [StridedViews.jl](https://github.com/Jutho/StridedViews.jl) libraries, providing
7//! efficient operations on strided multidimensional array views.
8//!
9//! # Core Types
10//!
11//! - [`StridedView`] / [`StridedViewMut`]: Dynamic-rank strided views over existing data
12//! - [`StridedArray`]: Owned strided multidimensional array
13//! - [`ElementOp`] trait and implementations ([`Identity`], [`Conj`], [`Transpose`], [`Adjoint`]):
14//!   Type-level element operations applied lazily on access
15//! - [`ExecutionPolicy`] / [`with_execution_policy`]: optional bounds on
16//!   strided-owned CPU fanout without creating a Rayon pool
17//!
18//! # Primary API (view-based, Julia-compatible)
19//!
20//! ## Map Operations
21//!
22//! - [`map_into`]: Apply a function element-wise from source to destination
23//! - [`zip_map2_into`], [`zip_map3_into`], [`zip_map4_into`]: Multi-array element-wise operations
24//!
25//! ## Reduce Operations
26//!
27//! - [`reduce`]: Full reduction with map function
28//! - [`reduce_axis`]: Reduce along a single axis
29//!
30//! ## Basic Operations
31//!
32//! - [`copy_into`]: Copy array contents
33//! - [`add`], [`mul`]: Element-wise arithmetic
34//! - [`axpy`]: y = alpha*x + y (array version)
35//! - [`sum`], [`dot`]: Reductions
36//! - [`symmetrize_into`], [`symmetrize_conj_into`]: Matrix symmetrization
37//!
38//! # Example
39//!
40//! ```rust
41//! use strided_basic::{StridedView, StridedViewMut, StridedArray, Identity, map_into};
42//!
43//! // Create a column-major array (Julia default)
44//! let src = StridedArray::<f64>::from_fn_col_major(&[2, 3], |idx| {
45//!     (idx[0] * 10 + idx[1]) as f64
46//! });
47//! let mut dest = StridedArray::<f64>::col_major(&[2, 3]);
48//!
49//! // Map with view-based API
50//! map_into(&mut dest.view_mut(), &src.view(), |x| x * 2.0).unwrap();
51//! assert_eq!(dest.get(&[1, 2]), 24.0); // (1*10 + 2) * 2
52//! ```
53//!
54//! # Cache Optimization
55//!
56//! The library uses Julia's blocking strategy for cache efficiency:
57//! - Dimensions are sorted by stride magnitude for optimal memory access
58//! - Operations are blocked into tiles fitting L1 cache ([`BLOCK_MEMORY_SIZE`] = 32KB)
59//! - Contiguous arrays use fast paths bypassing the blocking machinery
60
61mod block;
62mod copy_plan;
63mod dense_update;
64mod erased;
65mod erased_common;
66mod exec_context;
67pub mod execution;
68mod execution_policy;
69mod fuse;
70mod kernel;
71mod layout_check;
72mod map_view;
73mod maybe_sync;
74mod ops_view;
75mod order;
76mod raw_ops;
77mod reduce_view;
78mod simd;
79mod threading;
80pub use copy_plan::CopyPlan;
81pub use dense_update::{axpby_accum, embed_diagonal_into_uninit, triangular_mask_into_uninit};
82pub use erased::{ErasedConcatenatePlan, ErasedCopyPlan, ErasedReducePlan, ReduceOp};
83pub use exec_context::ExecContext;
84pub use execution_policy::{with_execution_policy, ExecutionPolicy};
85pub use map_view::{
86    broadcast_mul_into, broadcast_mul_into_uninit, compare_into, compare_into_uninit, map_into,
87    mul_into, mul_into_uninit, zip_map2_into, zip_map3_into, zip_map4_into, CompareOp,
88};
89pub use maybe_sync::{MaybeSend, MaybeSendSync, MaybeSync};
90pub use ops_view::{
91    add, axpy, copy_conj, copy_into, copy_into_col_major, copy_into_uninit, copy_scale,
92    copy_transpose_scale_into, dot, fma, mul, sum, symmetrize_conj_into, symmetrize_into,
93};
94pub use raw_ops::{
95    axpy_conj_raw, axpy_raw, copy_scale_conj_raw, copy_scale_raw, RAW_FUSED_RANK_LIMIT,
96};
97pub use reduce_view::{reduce, reduce_axis};
98pub use simd::MaybeSimdOps;
99pub use strided_view::view;
100pub use strided_view::*;
101/// Block memory size for cache-optimized iteration (L1 cache target).
102///
103/// Operations are blocked into tiles that fit within this size to maximize cache hits.
104/// Default: 32KB (typical L1 data cache size).
105pub const BLOCK_MEMORY_SIZE: usize = 32 * 1024;
106
107/// Cache line size in bytes.
108///
109/// Used for memory region calculations in block size computation.
110pub const CACHE_LINE_SIZE: usize = 64;
111
112mod gather_plan;
113mod outer_product;
114mod static_indexing_plan;
115pub use gather_plan::{
116    DynamicSlicePlan, DynamicUpdateSlicePlan, GatherIndex, GatherPlan, GatherSpec, ScatterPlan,
117    ScatterSpec,
118};
119pub use outer_product::{
120    batched_outer_product_into, batched_outer_product_into_uninit, plan_lazy_outer_product,
121    LazyOuterProductLayout,
122};
123pub use static_indexing_plan::{ConcatenatePlan, PadPlan, ReversePlan, SlicePlan};