use ruda_model::config::Config;
use ruda_model::record::Record;
use ruda_model::tensor::backend::Backend;
use ruda_model::tensor::{ElementConversion, Tensor};
#[derive(Config, Debug)]
pub struct MomentumConfig {
#[config(default = 0.9)]
pub momentum: f64,
#[config(default = 0.1)]
pub dampening: f64,
#[config(default = false)]
pub nesterov: bool,
}
#[derive(Record, Clone, new)]
pub struct MomentumState<B: Backend, const D: usize> {
velocity: Tensor<B, D>,
}
impl<B:Backend,const D:usize> super::OptimizerCheckpointBuffers<B,D> for MomentumState<B,D> {
fn visit_checkpoint_buffers<F:FnMut(&Tensor<B,D>)>(&self,visit:&mut F) {visit(&self.velocity);}
fn map_checkpoint_buffers<F:FnMut(Tensor<B,D>)->Tensor<B,D>>(self,map:&mut F) -> Self {Self {velocity:map(self.velocity)}}
}
impl<B:Backend,const D:usize> super::FlatOptimizerCheckpointState<B,D> for MomentumState<B,D> {
type FlatState=MomentumState<B,1>;
fn into_flat_shard(self,shard:&super::FlatOptimizerTensorShard) -> Result<Self::FlatState,super::OptimizerShardError> {
Ok(MomentumState {velocity:shard.partition(self.velocity)?})
}
}
impl<B:Backend,const D:usize> super::OptimizerCheckpointScalars for MomentumState<B,D> {
type Scalars=();
fn checkpoint_scalars(&self) {}
}
#[derive(Clone)]
pub struct Momentum<B: Backend> {
momentum: B::FloatElem,
dampening: f64,
nesterov: bool,
}
impl<B: Backend> Momentum<B> {
pub fn new(config: &MomentumConfig) -> Self {
Self {
momentum: config.momentum.elem(),
dampening: config.dampening,
nesterov: config.nesterov,
}
}
pub fn transform<const D: usize>(
&self,
grad: Tensor<B, D>,
state: Option<MomentumState<B, D>>,
) -> (Tensor<B, D>, MomentumState<B, D>) {
let velocity = if let Some(state) = state {
grad.clone()
.mul_scalar(1.0 - self.dampening)
.add(state.velocity.mul_scalar(self.momentum))
} else {
grad.clone()
};
let grad = match self.nesterov {
true => velocity.clone().mul_scalar(self.momentum).add(grad),
false => velocity.clone(),
};
(grad, MomentumState::new(velocity))
}
}
impl<B: Backend, const D: usize> MomentumState<B, D> {
pub fn velocity(&self) -> &Tensor<B, D> {
&self.velocity
}
pub fn to_device(mut self, device: &B::Device) -> Self {
self.velocity = self.velocity.to_device(device);
self
}
}