use alloc::vec::Vec;
use core::fmt::Debug;
use crate::{Backend,moe::{MoeOps,MoeOptions,MoeExpertStrategy,MoeCombineGradientStrategy},tensor::{FloatTensor,IntTensor}};
#[derive(Debug)]
pub struct MoeDispatched<B:MoeDispatchOps> {
pub values:FloatTensor<B>,
pub weights:FloatTensor<B>,
pub selected_experts:IntTensor<B>,
pub row_experts:IntTensor<B>,
pub state:B::MoeDispatchState,
}
#[derive(Clone,Copy,Debug,PartialEq,Eq)]
pub struct MoeCombineSelection {
pub experts:bool,
pub weights:bool,
}
#[derive(Debug)]
pub struct MoeCombineBackward<B:Backend> {
pub experts:Option<FloatTensor<B>>,
pub weights:Option<FloatTensor<B>>,
}
pub trait MoeDispatchOps:MoeOps {
type MoeDispatchState:Clone+Send+Debug+'static;
fn moe_dispatch(input:FloatTensor<Self>,logits:FloatTensor<Self>,bias:Option<FloatTensor<Self>>,options:MoeOptions)
-> Result<MoeDispatched<Self>,Self::MoeError>;
fn moe_dispatch_counts(state:&Self::MoeDispatchState,expert_prefix:&[usize]) -> Result<Vec<usize>,Self::MoeError>;
fn moe_dispatch_backward(state:Self::MoeDispatchState,gradient:FloatTensor<Self>) -> Result<FloatTensor<Self>,Self::MoeError>;
fn moe_dispatch_weights_backward(state:Self::MoeDispatchState,gradient:FloatTensor<Self>) -> Result<FloatTensor<Self>,Self::MoeError>;
fn moe_combine(state:Self::MoeDispatchState,expert_values:FloatTensor<Self>,weights:FloatTensor<Self>,backward_strategy:MoeCombineGradientStrategy)
-> Result<FloatTensor<Self>,Self::MoeError>;
fn moe_combine_backward(state:Self::MoeDispatchState,expert_values:FloatTensor<Self>,weights:FloatTensor<Self>,gradient:FloatTensor<Self>,
strategy:MoeCombineGradientStrategy,selection:MoeCombineSelection) -> Result<MoeCombineBackward<Self>,Self::MoeError>;
}
#[derive(Clone,Copy,Debug,PartialEq,Eq)]
pub struct MoeReceivedOptions {
pub expert_start:usize,
pub forward:MoeExpertStrategy,
pub backward:MoeExpertStrategy,
}
#[derive(Clone,Copy,Debug,PartialEq,Eq)]
pub struct MoeReceivedSelection {
pub input:bool,
pub gate:bool,
pub up:bool,
pub down:bool,
}
#[derive(Debug)]
pub struct MoeReceivedBackward<B:Backend> {
pub input:Option<FloatTensor<B>>,
pub gate:Option<FloatTensor<B>>,
pub up:Option<FloatTensor<B>>,
pub down:Option<FloatTensor<B>>,
}
pub trait MoeReceivedOps:MoeOps {
type MoeReceivedState:Clone+Send+Debug+'static;
fn moe_received_forward(input:FloatTensor<Self>,global_ids:IntTensor<Self>,gate:FloatTensor<Self>,up:FloatTensor<Self>,down:FloatTensor<Self>,
options:MoeReceivedOptions,selection:MoeReceivedSelection) -> Result<(FloatTensor<Self>,Self::MoeReceivedState),Self::MoeError>;
fn moe_received_inference(input:FloatTensor<Self>,global_ids:IntTensor<Self>,gate:FloatTensor<Self>,up:FloatTensor<Self>,down:FloatTensor<Self>,options:MoeReceivedOptions)
-> Result<FloatTensor<Self>,Self::MoeError> {
Self::moe_received_forward(input,global_ids,gate,up,down,options,MoeReceivedSelection {input:false,gate:false,up:false,down:false}).map(|(output,_)|output)
}
fn moe_received_backward(state:Self::MoeReceivedState,gradient:FloatTensor<Self>,selection:MoeReceivedSelection)
-> Result<MoeReceivedBackward<Self>,Self::MoeError>;
}