Skip to main content

ruda_tensor/
expert_projection.rs

1//! Original native floating expert projections and storage-rounded SwiGLU input VJPs.
2use crate::{Backend,moe::MoeExpertStrategy,tensor::{FloatTensor,IntTensor}};
3use core::fmt;
4
5/// Explicit original grouped execution, independent of quantized base execution choices.
6#[derive(Clone,Copy,Debug,PartialEq,Eq)]
7pub struct ExpertProjectionOptions {
8    /// First global expert ID represented by actual original `[E,N,K]` weights.
9    pub expert_start:usize,
10    /// Original actual floating grouped forward policy.
11    pub forward:MoeExpertStrategy,
12    /// Original actual independent floating grouped backward policy.
13    pub backward:MoeExpertStrategy,
14}
15/// Requested original native input/cube derivatives.
16#[derive(Clone,Copy,Debug,PartialEq,Eq)]
17pub struct ExpertProjectionSelection {pub input:bool,pub weights:bool}
18/// Actual original optional VJP outputs. Native weight derivatives retain FP32.
19#[derive(Debug)]
20pub struct ExpertProjectionBackward<B:Backend> {pub input:Option<FloatTensor<B>>,pub weights:Option<FloatTensor<B>>}
21/// Original actual grouped projection, including real trainable expert adapter matrices.
22pub trait ExpertProjectionOps:Backend {
23    /// Original native row/projection or first-order differentiation failure.
24    type ExpertProjectionError:fmt::Debug;
25    /// Original actual floating cube and native private COPY row mapping.
26    type ExpertProjectionState:Clone+Send+fmt::Debug+'static;
27    /// Execute actual U32 assigned expert projections and restore original incoming row order.
28    fn expert_projection_forward(input:FloatTensor<Self>,global_ids:IntTensor<Self>,weights:FloatTensor<Self>,options:ExpertProjectionOptions)
29        -> Result<(FloatTensor<Self>,Self::ExpertProjectionState),Self::ExpertProjectionError>;
30    /// Native first-order selected derivatives, without allocations for omitted cube/input gradients.
31    fn expert_projection_backward(state:Self::ExpertProjectionState,gradient:FloatTensor<Self>,selection:ExpertProjectionSelection)
32        -> Result<ExpertProjectionBackward<Self>,Self::ExpertProjectionError>;
33}
34/// Actual requested native storage-rounded activation input derivatives.
35#[derive(Clone,Copy,Debug,PartialEq,Eq)]
36pub struct NativeSwiGluSelection {pub gate:bool,pub up:bool}
37/// Original actual optional activation input VJPs.
38#[derive(Debug)]
39pub struct NativeSwiGluBackward<B:Backend> {pub gate:Option<FloatTensor<B>>,pub up:Option<FloatTensor<B>>}
40/// Original ruDNN activation arithmetic, not an alternate unfused AD derivative formula.
41pub trait NativeSwiGluOps:Backend {
42    /// Original native activation or first-order differentiation failure.
43    type SwiGluError:fmt::Debug;
44    /// Round stored SiLU before multiplication, retaining original activation storage and geometry.
45    fn native_swiglu(gate:FloatTensor<Self>,up:FloatTensor<Self>) -> Result<FloatTensor<Self>,Self::SwiGluError>;
46    /// Actual original selected first-order input VJPs, including intermediate storage rounding.
47    fn native_swiglu_backward(gate:FloatTensor<Self>,up:FloatTensor<Self>,gradient:FloatTensor<Self>,selection:NativeSwiGluSelection)
48        -> Result<NativeSwiGluBackward<Self>,Self::SwiGluError>;
49}