use super::*;
use ruda_model::{module::ModuleDisplay,tensor::IntegerTensorCollective};
use crate::transformer::{TransformerProjectionShape,Nf4TransformerProjection,MixedTransformerProjection,
AwqTransformerProjection,ProjectedTransformerStack,ProjectedTransformerModel};
pub trait ShardTransformerProjection<B:Backend>:TransformerProjectionShape<B> {
type Sharded:FullyShardedModule<B>+ModuleDisplay;
fn shard(self,context:&mut ShardingContext<B>) -> Self::Sharded;
}
pub trait GatherTransformerProjection<AB:Backend,B:Backend>:FullyShardedModule<AB>+ModuleDisplay {
type Gathered:TransformerProjectionShape<AB>;
fn gather_projection<C:IntegerTensorCollective<B>>(&self,communicator:C) -> Result<Self::Gathered,C::Error>;
}
macro_rules! shard_original_projection {
($source:ident,$target:ident,$method:ident) => {
impl<B:Backend> ShardTransformerProjection<B> for crate::$source<B> {
type Sharded=$target<B>;
fn shard(self,context:&mut ShardingContext<B>) -> Self::Sharded {context.$method(self)}
}
impl<B:Backend> GatherTransformerProjection<B,B> for $target<B> {
type Gathered=crate::$source<B>;
fn gather_projection<C:IntegerTensorCollective<B>>(&self,communicator:C) -> Result<Self::Gathered,C::Error> {self.gather_inference(communicator)}
}
impl<B:Backend,S:CheckpointStrategy> GatherTransformerProjection<Autodiff<B,S>,B> for $target<Autodiff<B,S>> {
type Gathered=crate::$source<Autodiff<B,S>>;
fn gather_projection<C:IntegerTensorCollective<B>>(&self,communicator:C) -> Result<Self::Gathered,C::Error> {self.gather(communicator)}
}
};
}
shard_original_projection!(Linear,FullyShardedLinear,linear);
shard_original_projection!(LoRALinear,FullyShardedLoRALinear,lora);
shard_original_projection!(FrozenAwqLinear,FullyShardedAwqLinear,awq);
shard_original_projection!(AwqLoRALinear,FullyShardedAwqLoRALinear,awq_lora);
shard_original_projection!(FrozenNf4Linear,FullyShardedNf4Linear,nf4);
shard_original_projection!(Nf4LoRALinear,FullyShardedNf4LoRALinear,nf4_lora);
impl<B:Backend> ShardTransformerProjection<B> for AwqTransformerProjection<B> {
type Sharded=FullyShardedAwqProjection<B>;
fn shard(self,context:&mut ShardingContext<B>) -> Self::Sharded {
match self {Self::Dense(layer)=>FullyShardedAwqProjection::Dense(context.linear(layer)),
Self::LoRA(layer)=>FullyShardedAwqProjection::LoRA(context.lora(layer)),
Self::Awq(layer)=>FullyShardedAwqProjection::Awq(context.awq(layer)),
Self::AwqLoRA(layer)=>FullyShardedAwqProjection::AwqLoRA(context.awq_lora(layer))}
}
}
impl<B:Backend> GatherTransformerProjection<B,B> for FullyShardedAwqProjection<B> {
type Gathered=AwqTransformerProjection<B>;
fn gather_projection<C:IntegerTensorCollective<B>>(&self,communicator:C) -> Result<Self::Gathered,C::Error> {self.gather_inference(communicator)}
}
impl<B:Backend,S:CheckpointStrategy> GatherTransformerProjection<Autodiff<B,S>,B> for FullyShardedAwqProjection<Autodiff<B,S>> {
type Gathered=AwqTransformerProjection<Autodiff<B,S>>;
fn gather_projection<C:IntegerTensorCollective<B>>(&self,communicator:C) -> Result<Self::Gathered,C::Error> {self.gather(communicator)}
}
#[derive(Module,Debug)]
pub enum FullyShardedNf4Projection<B:Backend> {
Dense(FullyShardedLinear<B>),
LoRA(FullyShardedLoRALinear<B>),
Nf4(FullyShardedNf4Linear<B>),
Nf4LoRA(FullyShardedNf4LoRALinear<B>),
}
impl<B:Backend> ShardTransformerProjection<B> for Nf4TransformerProjection<B> {
type Sharded=FullyShardedNf4Projection<B>;
fn shard(self,context:&mut ShardingContext<B>) -> Self::Sharded {
match self {Self::Dense(layer)=>FullyShardedNf4Projection::Dense(context.linear(layer)),
Self::LoRA(layer)=>FullyShardedNf4Projection::LoRA(context.lora(layer)),
Self::Nf4(layer)=>FullyShardedNf4Projection::Nf4(context.nf4(layer)),
Self::Nf4LoRA(layer)=>FullyShardedNf4Projection::Nf4LoRA(context.nf4_lora(layer))}
}
}
#[derive(Module,Debug)]
pub enum FullyShardedMixedProjection<B:Backend> {
Dense(FullyShardedLinear<B>),
LoRA(FullyShardedLoRALinear<B>),
Awq(FullyShardedAwqLinear<B>),
AwqLoRA(FullyShardedAwqLoRALinear<B>),
Nf4(FullyShardedNf4Linear<B>),
Nf4LoRA(FullyShardedNf4LoRALinear<B>),
}
impl<B:Backend> ShardTransformerProjection<B> for MixedTransformerProjection<B> {
type Sharded=FullyShardedMixedProjection<B>;
fn shard(self,context:&mut ShardingContext<B>) -> Self::Sharded {
match self {Self::Dense(layer)=>FullyShardedMixedProjection::Dense(context.linear(layer)),
Self::LoRA(layer)=>FullyShardedMixedProjection::LoRA(context.lora(layer)),
Self::Awq(layer)=>FullyShardedMixedProjection::Awq(context.awq(layer)),
Self::AwqLoRA(layer)=>FullyShardedMixedProjection::AwqLoRA(context.awq_lora(layer)),
Self::Nf4(layer)=>FullyShardedMixedProjection::Nf4(context.nf4(layer)),
Self::Nf4LoRA(layer)=>FullyShardedMixedProjection::Nf4LoRA(context.nf4_lora(layer))}
}
}
macro_rules! gather_selected_projection {
($source:ident,$target:ident,[$($variant:ident),+],$backend:ty,[$($generics:tt)*],$gather:ident) => {
impl<$($generics)*> GatherTransformerProjection<$backend,B> for $source<$backend> {
type Gathered=$target<$backend>;
fn gather_projection<C:IntegerTensorCollective<B>>(&self,communicator:C) -> Result<Self::Gathered,C::Error> {
match self {$(Self::$variant(layer)=>layer.$gather(communicator).map($target::$variant),)+}
}
}
};
}
gather_selected_projection!(FullyShardedNf4Projection,Nf4TransformerProjection,[Dense,LoRA,Nf4,Nf4LoRA],B,[B:Backend],gather_inference);
gather_selected_projection!(FullyShardedNf4Projection,Nf4TransformerProjection,[Dense,LoRA,Nf4,Nf4LoRA],Autodiff<B,S>,[B:Backend,S:CheckpointStrategy],gather);
gather_selected_projection!(FullyShardedMixedProjection,MixedTransformerProjection,[Dense,LoRA,Awq,AwqLoRA,Nf4,Nf4LoRA],B,[B:Backend],gather_inference);
gather_selected_projection!(FullyShardedMixedProjection,MixedTransformerProjection,[Dense,LoRA,Awq,AwqLoRA,Nf4,Nf4LoRA],Autodiff<B,S>,[B:Backend,S:CheckpointStrategy],gather);
pub type FullyShardedProjectedAttention<B,P> = FullyShardedAwqAttention<B,P>;
pub type FullyShardedProjectedFeedForward<B,P> = FullyShardedAwqFeedForward<B,P>;
pub type FullyShardedProjectedTransformerBlock<B,P> = FullyShardedAwqTransformerBlock<B,P>;
pub type FullyShardedProjectedTransformerStack<B,P> = FullyShardedAwqTransformerStack<B,P>;
pub type FullyShardedProjectedTransformerHead<B,P> = FullyShardedAwqTransformerHead<B,P>;
pub type FullyShardedProjectedTransformerModel<B,P> = FullyShardedAwqTransformerModel<B,P>;
pub type FullyShardedNf4TransformerModel<B> = FullyShardedProjectedTransformerModel<B,FullyShardedNf4Projection<B>>;
pub type FullyShardedMixedTransformerModel<B> = FullyShardedProjectedTransformerModel<B,FullyShardedMixedProjection<B>>;
pub type FullyShardedProjectedError<C,Q> = FullyShardedAwqError<C,Q>;
pub type FullyShardedProjectedTrainingError<C,Q> = FullyShardedAwqTrainingError<C,Q>;
impl<B:Backend> ShardingContext<B> {
pub fn projected_transformer_stack<P:ShardTransformerProjection<B>>(&mut self,stack:ProjectedTransformerStack<B,P>)
-> FullyShardedProjectedTransformerStack<B,P::Sharded> {self.awq_transformer_stack(stack)}
pub fn projected_transformer_model<P:ShardTransformerProjection<B>>(&mut self,model:ProjectedTransformerModel<B,P>)
-> FullyShardedProjectedTransformerModel<B,P::Sharded> {self.awq_transformer_model(model)}
}