ruda-optim 0.21.13

Ruda optimizer updates, gradient state, clipping and learning-rate schedules.
Documentation
// SPDX-License-Identifier: Apache-2.0
//! Opt-in, out-of-place AdamW updates with FP32 master parameters and moments.
//!
//! `fused-adamw` provides configuration and a CPU **reference**, not a fallback.
//! `fused-adamw-device` adds a single-kernel device path with F32/F16/BF16 gradients.
//! Existing [`crate::AdamW`] dispatch and checkpoint formats are unchanged.
//!
//! This is a low-level optimizer primitive, not a differentiable optimizer or an
//! automatic replacement for the model optimizer adaptor. See
//! `docs/en/fused-adamw.md` for the numerical, asynchronous and benchmark contract.

mod config;
pub use config::{AdamWOptions, FusedAdamWError, StepCoefficients, StepControl};

/// An independent, explicitly invoked CPU numerical oracle for tests.
pub mod reference;

#[cfg(feature = "fused-adamw-device")]
mod kernel;
#[cfg(feature = "fused-adamw-device")]
mod device;
/// Unsafe preallocated-storage kernels for native framework adapters.
#[cfg(feature = "fused-adamw-device")]
pub mod storage;
/// Bounded hierarchical gradient-statistics workspace layout.
pub mod stats_plan;
#[cfg(feature = "fused-adamw-device")]
pub use device::{AdamWState, AdamWUpdate, adamw_step};

#[cfg(test)]
mod existing_optimizer_tests;

/// Explicit local-group gradient diagnostics and clipping policy.
#[cfg(feature = "gradient-guard")]
pub mod gradient_norm;

#[cfg(feature = "gradient-guard-device")]
mod guard_kernel;
#[cfg(feature = "gradient-guard-device")]
mod guard_device;
#[cfg(feature = "gradient-guard-device")]
pub use guard_device::{
    AdamWEntry, GuardedAdamWUpdate, GradientStatsReport, SkipReason,
    gradient_stats_sync, guarded_adamw_step,
};