Skip to main content

strided_basic/
lib.rs

1#![doc = include_str!("../README.md")]
2
3//! Cache-optimized kernels for strided multidimensional array operations.
4//!
5//! This crate is a Rust port of Julia's [Strided.jl](https://github.com/Jutho/Strided.jl)
6//! and [StridedViews.jl](https://github.com/Jutho/StridedViews.jl) libraries, providing
7//! efficient operations on strided multidimensional array views.
8//!
9//! # Core Types
10//!
11//! - [`StridedView`] / [`StridedViewMut`]: Dynamic-rank strided views over existing data
12//! - [`StridedArray`]: Owned strided multidimensional array
13//! - [`ElementOp`] trait and implementations ([`Identity`], [`Conj`], [`Transpose`], [`Adjoint`]):
14//!   Type-level element operations applied lazily on access
15//! - [`ExecutionPolicy`] / [`with_execution_policy`]: optional bounds on
16//!   strided-owned CPU fanout without creating a Rayon pool
17//!
18//! # Primary API (view-based, Julia-compatible)
19//!
20//! ## Map Operations
21//!
22//! - [`map_into`]: Apply a function element-wise from source to destination
23//! - [`zip_map2_into`], [`zip_map3_into`], [`zip_map4_into`]: Multi-array element-wise operations
24//!
25//! ## In-place Update Operations
26//!
27//! - [`map_update_into`], [`zip_update2_into`], [`zip_update3_into`]: `dest[i] = f(dest[i], inputs...)`,
28//!   reading the old destination without a second view of it
29//!
30//! ## Reduce Operations
31//!
32//! - [`reduce`]: Full reduction with map function
33//! - [`reduce_axis`]: Reduce along a single axis
34//!
35//! ## Basic Operations
36//!
37//! - [`copy_into`]: Copy array contents
38//! - [`add`], [`mul`]: Element-wise arithmetic
39//! - [`axpy`]: y = alpha*x + y (array version)
40//! - [`sum`], [`dot`]: Reductions
41//! - [`symmetrize_into`], [`symmetrize_conj_into`]: Matrix symmetrization
42//!
43//! # Example
44//!
45//! ```rust
46//! use strided_basic::{StridedView, StridedViewMut, StridedArray, Identity, map_into};
47//!
48//! // Create a column-major array (Julia default)
49//! let src = StridedArray::<f64>::from_fn_col_major(&[2, 3], |idx| {
50//!     (idx[0] * 10 + idx[1]) as f64
51//! });
52//! let mut dest = StridedArray::<f64>::col_major(&[2, 3]);
53//!
54//! // Map with view-based API
55//! map_into(&mut dest.view_mut(), &src.view(), |x| x * 2.0).unwrap();
56//! assert_eq!(dest.get(&[1, 2]), 24.0); // (1*10 + 2) * 2
57//! ```
58//!
59//! # Cache Optimization
60//!
61//! The library uses Julia's blocking strategy for cache efficiency:
62//! - Dimensions are sorted by stride magnitude for optimal memory access
63//! - Operations are blocked into tiles fitting L1 cache ([`BLOCK_MEMORY_SIZE`] = 32KB)
64//! - Contiguous arrays use fast paths bypassing the blocking machinery
65
66mod block;
67mod complex_div;
68mod copy_plan;
69mod dense_update;
70mod erased;
71mod erased_common;
72mod exec_context;
73pub mod execution;
74mod execution_policy;
75mod fuse;
76mod kernel;
77mod layout_check;
78mod map_view;
79mod maybe_sync;
80mod ops_view;
81mod order;
82mod raw_ops;
83mod reduce_view;
84mod simd;
85mod threading;
86mod update_view;
87pub use complex_div::{robust_complex_divide_f32, robust_complex_divide_f64};
88pub use copy_plan::CopyPlan;
89pub use dense_update::{axpby_accum, embed_diagonal_into_uninit, triangular_mask_into_uninit};
90pub use erased::{ErasedConcatenatePlan, ErasedCopyPlan, ErasedReducePlan, ReduceOp};
91pub use exec_context::ExecContext;
92pub use execution_policy::{with_execution_policy, ExecutionPolicy};
93pub use map_view::{
94    broadcast_mul_into, broadcast_mul_into_uninit, compare_into, compare_into_uninit, map_into,
95    mul_into, mul_into_uninit, zip_map2_into, zip_map3_into, zip_map4_into, CompareOp,
96};
97pub use maybe_sync::{MaybeSend, MaybeSendSync, MaybeSync};
98pub use ops_view::{
99    add, axpy, copy_conj, copy_into, copy_into_col_major, copy_into_uninit, copy_scale,
100    copy_transpose_scale_into, dot, fma, mul, sum, symmetrize_conj_into, symmetrize_into,
101};
102pub use raw_ops::{
103    axpy_conj_raw, axpy_raw, copy_scale_conj_raw, copy_scale_raw, RAW_FUSED_RANK_LIMIT,
104};
105pub use reduce_view::{reduce, reduce_axis};
106pub use simd::MaybeSimdOps;
107pub use strided_view::view;
108pub use strided_view::*;
109pub use update_view::{map_update_into, zip_update2_into, zip_update3_into};
110/// Block memory size for cache-optimized iteration (L1 cache target).
111///
112/// Operations are blocked into tiles that fit within this size to maximize cache hits.
113/// Default: 32KB (typical L1 data cache size).
114pub const BLOCK_MEMORY_SIZE: usize = 32 * 1024;
115
116/// Cache line size in bytes.
117///
118/// Used for memory region calculations in block size computation.
119pub const CACHE_LINE_SIZE: usize = 64;
120
121mod gather_plan;
122mod outer_product;
123mod static_indexing_plan;
124pub use gather_plan::{
125    DynamicSlicePlan, DynamicUpdateSlicePlan, GatherIndex, GatherPlan, GatherSpec, ScatterPlan,
126    ScatterSpec,
127};
128pub use outer_product::{
129    batched_outer_product_into, batched_outer_product_into_uninit, plan_lazy_outer_product,
130    LazyOuterProductLayout,
131};
132pub use static_indexing_plan::{ConcatenatePlan, PadPlan, ReversePlan, SlicePlan};