Skip to main content

strided_basic/
lib.rs

1#![doc = include_str!("../README.md")]
2
3//! Cache-optimized kernels for strided multidimensional array operations.
4//!
5//! This crate is a Rust port of Julia's [Strided.jl](https://github.com/Jutho/Strided.jl)
6//! and [StridedViews.jl](https://github.com/Jutho/StridedViews.jl) libraries, providing
7//! efficient operations on strided multidimensional array views.
8//!
9//! # Core Types
10//!
11//! - [`StridedView`] / [`StridedViewMut`]: Dynamic-rank strided views over existing data
12//! - [`StridedArray`]: Owned strided multidimensional array
13//! - [`ElementOp`] trait and implementations ([`Identity`], [`Conj`], [`Transpose`], [`Adjoint`]):
14//!   Type-level element operations applied lazily on access
15//! - [`ExecutionPolicy`] / [`with_execution_policy`]: optional bounds on
16//!   strided-owned CPU fanout without creating a Rayon pool
17//!
18//! # Primary API (view-based, Julia-compatible)
19//!
20//! ## Map Operations
21//!
22//! - [`map_into`]: Apply a function element-wise from source to destination
23//! - [`zip_map2_into`], [`zip_map3_into`], [`zip_map4_into`]: Multi-array element-wise operations
24//!
25//! ## In-place Update Operations
26//!
27//! - [`map_update_into`], [`zip_update2_into`], [`zip_update3_into`]: `dest[i] = f(dest[i], inputs...)`,
28//!   reading the old destination without a second view of it
29//!
30//! ## Reduce Operations
31//!
32//! - [`reduce`]: Full reduction with map function
33//! - [`reduce_axis`]: Reduce along a single axis
34//!
35//! ## Basic Operations
36//!
37//! - [`copy_into`]: Copy array contents
38//! - [`add`], [`mul`]: Element-wise arithmetic
39//! - [`axpy`]: y = alpha*x + y (array version)
40//! - [`sum`], [`dot`]: Reductions
41//! - [`symmetrize_into`], [`symmetrize_conj_into`]: Matrix symmetrization
42//!
43//! # Example
44//!
45//! ```rust
46//! use strided_basic::{StridedView, StridedViewMut, StridedArray, Identity, map_into};
47//!
48//! // Create a column-major array (Julia default)
49//! let src = StridedArray::<f64>::from_fn_col_major(&[2, 3], |idx| {
50//!     (idx[0] * 10 + idx[1]) as f64
51//! });
52//! let mut dest = StridedArray::<f64>::col_major(&[2, 3]);
53//!
54//! // Map with view-based API
55//! map_into(&mut dest.view_mut(), &src.view(), |x| x * 2.0).unwrap();
56//! assert_eq!(dest.get(&[1, 2]), 24.0); // (1*10 + 2) * 2
57//! ```
58//!
59//! # Cache Optimization
60//!
61//! The library uses Julia's blocking strategy for cache efficiency:
62//! - Dimensions are sorted by stride magnitude for optimal memory access
63//! - Operations are blocked into tiles fitting L1 cache ([`BLOCK_MEMORY_SIZE`] = 32KB)
64//! - Contiguous arrays use fast paths bypassing the blocking machinery
65
66mod block;
67mod complex_div;
68mod copy_plan;
69mod dense_update;
70mod erased;
71mod erased_common;
72mod exec_context;
73pub mod execution;
74mod execution_policy;
75mod fuse;
76mod kernel;
77mod layout_check;
78mod map_view;
79mod maybe_sync;
80mod ops_view;
81mod order;
82mod raw_ops;
83mod reduce_view;
84mod simd;
85mod threading;
86mod update_view;
87pub use complex_div::{robust_complex_divide_f32, robust_complex_divide_f64};
88pub use copy_plan::CopyPlan;
89pub use dense_update::{axpby_accum, embed_diagonal_into_uninit, triangular_mask_into_uninit};
90pub use erased::{
91    ArgReduceOp, ErasedArgReducePlan, ErasedConcatenatePlan, ErasedCopyPlan, ErasedNormPlan,
92    ErasedReducePlan, ErasedScanPlan, NormKind, NormSpec, ReduceOp, ScanOp, ScanOptions,
93};
94pub use exec_context::ExecContext;
95pub use execution_policy::{with_execution_policy, ExecutionPolicy};
96pub use map_view::{
97    broadcast_mul_into, broadcast_mul_into_uninit, compare_into, compare_into_uninit, map_into,
98    mul_into, mul_into_uninit, zip_map2_into, zip_map3_into, zip_map4_into, CompareOp,
99};
100pub use maybe_sync::{MaybeSend, MaybeSendSync, MaybeSync};
101pub use ops_view::{
102    add, axpy, copy_conj, copy_into, copy_into_col_major, copy_into_uninit, copy_scale,
103    copy_transpose_scale_into, dot, fma, mul, sum, symmetrize_conj_into, symmetrize_into,
104};
105pub use raw_ops::{
106    axpy_conj_raw, axpy_raw, copy_scale_conj_raw, copy_scale_raw, RAW_FUSED_RANK_LIMIT,
107};
108pub use reduce_view::{reduce, reduce_axis};
109pub use simd::MaybeSimdOps;
110pub use strided_view::view;
111pub use strided_view::*;
112pub use update_view::{map_update_into, zip_update2_into, zip_update3_into};
113/// Block memory size for cache-optimized iteration (L1 cache target).
114///
115/// Operations are blocked into tiles that fit within this size to maximize cache hits.
116/// Default: 32KB (typical L1 data cache size).
117pub const BLOCK_MEMORY_SIZE: usize = 32 * 1024;
118
119/// Cache line size in bytes.
120///
121/// Used for memory region calculations in block size computation.
122pub const CACHE_LINE_SIZE: usize = 64;
123
124mod gather_plan;
125mod outer_product;
126mod static_indexing_plan;
127pub use gather_plan::{
128    DynamicSlicePlan, DynamicUpdateSlicePlan, GatherIndex, GatherPlan, GatherSpec, ScatterPlan,
129    ScatterSpec,
130};
131pub use outer_product::{
132    batched_outer_product_into, batched_outer_product_into_uninit, plan_lazy_outer_product,
133    LazyOuterProductLayout,
134};
135pub use static_indexing_plan::{ConcatenatePlan, PadPlan, ReversePlan, SlicePlan};