use super::*;
use ruda_model::tensor::{Bool,IntegerTensorCollective};
use crate::transformer::{TransformerProjection,BackendProjection};
use crate::{attention::{DenseAttentionMask,DenseAttentionOptions,PackedSequenceLayout,PackedAttentionOptions,PackedDocumentAttentionMask},
cache::{ProjectedKvCache,TransformerKvCache},
transformer::{AwqTransformerProjection,AwqGroupedQueryAttention,AwqFeedForward,AwqTransformerBlock,AwqTransformerStack}};
#[derive(Module,Debug)]
pub enum FullyShardedAwqProjection<B:Backend> {
Dense(FullyShardedLinear<B>),
LoRA(FullyShardedLoRALinear<B>),
Awq(FullyShardedAwqLinear<B>),
AwqLoRA(FullyShardedAwqLoRALinear<B>),
}
#[derive(Module,Debug)]
pub struct FullyShardedAwqAttention<B:Backend,P:Module<B>=FullyShardedAwqProjection<B>> {
pub query:BackendProjection<B,P>,
pub key:BackendProjection<B,P>,
pub value:BackendProjection<B,P>,
pub output:BackendProjection<B,P>,
pub dropout:crate::Dropout,
pub query_heads:usize,
pub kv_heads:usize,
pub head_dimension:usize,
}
#[derive(Module,Debug)]
pub struct FullyShardedAwqFeedForward<B:Backend,P:Module<B>=FullyShardedAwqProjection<B>> {
pub up:P,
pub gate:Option<P>,
pub down:P,
pub activation:FullyShardedActivation<B>,
pub dropout:crate::Dropout,
}
#[derive(Module,Debug)]
pub struct FullyShardedAwqTransformerBlock<B:Backend,P:Module<B>=FullyShardedAwqProjection<B>> {
pub attention:FullyShardedAwqAttention<B,P>,
pub feed_forward:FullyShardedAwqFeedForward<B,P>,
pub attention_norm:FullyShardedTransformerNorm<B>,
pub feed_forward_norm:FullyShardedTransformerNorm<B>,
pub residual_dropout:crate::Dropout,
pub norm_first:bool,
}
#[derive(Module,Debug)]
pub struct FullyShardedAwqTransformerStack<B:Backend,P:Module<B>=FullyShardedAwqProjection<B>> {
pub blocks:Vec<FullyShardedAwqTransformerBlock<B,P>>,
}
impl<B:Backend> ShardingContext<B> {
pub fn awq_projection<P:ShardTransformerProjection<B>>(&mut self,projection:P) -> P::Sharded {projection.shard(self)}
pub fn awq_attention<P:ShardTransformerProjection<B>>(&mut self,attention:AwqGroupedQueryAttention<B,P>) -> FullyShardedAwqAttention<B,P::Sharded> {
FullyShardedAwqAttention {query:self.awq_projection(attention.query),key:self.awq_projection(attention.key),
value:self.awq_projection(attention.value),output:self.awq_projection(attention.output),dropout:attention.dropout,
query_heads:attention.query_heads,kv_heads:attention.kv_heads,head_dimension:attention.head_dimension}
}
pub fn awq_feed_forward<P:ShardTransformerProjection<B>>(&mut self,feed:AwqFeedForward<B,P>) -> FullyShardedAwqFeedForward<B,P::Sharded> {
FullyShardedAwqFeedForward {up:self.awq_projection(feed.up),gate:feed.gate.map(|gate|self.awq_projection(gate)),
down:self.awq_projection(feed.down),activation:self.activation(feed.activation),dropout:feed.dropout}
}
pub fn awq_transformer<P:ShardTransformerProjection<B>>(&mut self,block:AwqTransformerBlock<B,P>) -> FullyShardedAwqTransformerBlock<B,P::Sharded> {
FullyShardedAwqTransformerBlock {attention:self.awq_attention(block.attention),feed_forward:self.awq_feed_forward(block.feed_forward),
attention_norm:self.normalization(block.attention_norm),feed_forward_norm:self.normalization(block.feed_forward_norm),
residual_dropout:block.residual_dropout,norm_first:block.norm_first}
}
pub fn awq_transformer_stack<P:ShardTransformerProjection<B>>(&mut self,stack:AwqTransformerStack<B,P>) -> FullyShardedAwqTransformerStack<B,P::Sharded> {
FullyShardedAwqTransformerStack {blocks:stack.blocks.into_iter().map(|block|self.awq_transformer(block)).collect()}
}
}
macro_rules! gather_awq_components {
($backend:ty,[$($generics:tt)*],$gather:ident) => {
impl<$($generics)*> FullyShardedAwqProjection<$backend> {
pub fn $gather<C:IntegerTensorCollective<B>>(&self,communicator:C) -> Result<AwqTransformerProjection<$backend>,C::Error> {
match self {
Self::Dense(layer)=>layer.$gather(communicator).map(AwqTransformerProjection::Dense),
Self::LoRA(layer)=>layer.$gather(communicator).map(AwqTransformerProjection::LoRA),
Self::Awq(layer)=>layer.$gather(communicator).map(AwqTransformerProjection::Awq),
Self::AwqLoRA(layer)=>layer.$gather(communicator).map(AwqTransformerProjection::AwqLoRA),
}
}
}
impl<$($generics)*,P:GatherTransformerProjection<$backend,B>> FullyShardedAwqAttention<$backend,P> {
pub fn $gather<C:IntegerTensorCollective<B>>(&self,communicator:C) -> Result<AwqGroupedQueryAttention<$backend,P::Gathered>,C::Error> {
Ok(AwqGroupedQueryAttention::from_projections(self.query.gather_projection(communicator.clone())?,self.key.gather_projection(communicator.clone())?,
self.value.gather_projection(communicator.clone())?,self.output.gather_projection(communicator)?,self.query_heads,self.kv_heads,self.head_dimension,self.dropout.clone()))
}
}
impl<$($generics)*,P:GatherTransformerProjection<$backend,B>> FullyShardedAwqFeedForward<$backend,P> {
pub fn $gather<C:IntegerTensorCollective<B>>(&self,communicator:C) -> Result<AwqFeedForward<$backend,P::Gathered>,C::Error> {
Ok(AwqFeedForward::from_projections(self.up.gather_projection(communicator.clone())?,
self.gate.as_ref().map(|gate|gate.gather_projection(communicator.clone())).transpose()?,self.down.gather_projection(communicator.clone())?,
self.activation.$gather(communicator)?,self.dropout.clone()))
}
}
impl<$($generics)*,P:GatherTransformerProjection<$backend,B>> FullyShardedAwqTransformerBlock<$backend,P> {
pub fn $gather<C:IntegerTensorCollective<B>>(&self,communicator:C) -> Result<AwqTransformerBlock<$backend,P::Gathered>,C::Error> {
Ok(AwqTransformerBlock {attention:self.attention.$gather(communicator.clone())?,feed_forward:self.feed_forward.$gather(communicator.clone())?,
attention_norm:self.attention_norm.$gather(communicator.clone())?,feed_forward_norm:self.feed_forward_norm.$gather(communicator)?,
residual_dropout:self.residual_dropout.clone(),norm_first:self.norm_first})
}
}
};
}
gather_awq_components!(B,[B:Backend],gather_inference);
gather_awq_components!(Autodiff<B,S>,[B:Backend,S:CheckpointStrategy],gather);
impl<B:Backend,P:Module<B>> FullyShardedAwqTransformerStack<B,P> {
pub fn from_full<Q:ShardTransformerProjection<B,Sharded=P>>(stack:AwqTransformerStack<B,Q>,rank:usize,world:usize) -> Self {ShardingContext::new(rank,world).awq_transformer_stack(stack)}
pub fn new_kv_cache(&self,initial_capacity:usize) -> TransformerKvCache<B> {TransformerKvCache::new(self.blocks.len(),initial_capacity)}
}
macro_rules! execute_awq_blocks {
($backend:ty,[$($generics:tt)*],$gather:ident,$forward:ident,$packed:ident) => {
impl<$($generics)*,P:GatherTransformerProjection<$backend,B>> FullyShardedAwqTransformerBlock<$backend,P>
where P::Gathered:TransformerProjection<$backend> {
pub fn $forward<C,F>(&self,input:Tensor<$backend,3>,masks:DenseAttentionMask<$backend>,options:DenseAttentionOptions,communicator:C,positions:F)
-> Result<Tensor<$backend,3>,FullyShardedAwqError<C::Error,<P::Gathered as TransformerProjection<$backend>>::Error>>
where C:IntegerTensorCollective<B>,F:FnOnce(Tensor<$backend,4>,Tensor<$backend,4>)->(Tensor<$backend,4>,Tensor<$backend,4>) {
self.$gather(communicator).map_err(FullyShardedAwqError::Collective)?.forward_with_positions(input,masks,options,positions)
.map_err(FullyShardedAwqError::Projection)
}
pub fn $packed<C,F>(&self,input:Tensor<$backend,2>,layout:&PackedSequenceLayout,masks:Option<&[PackedDocumentAttentionMask<$backend>]>,
options:PackedAttentionOptions,communicator:C,positions:F)
-> Result<Tensor<$backend,2>,FullyShardedAwqError<C::Error,<P::Gathered as TransformerProjection<$backend>>::Error>>
where C:IntegerTensorCollective<B>,F:FnOnce(Tensor<$backend,3>,Tensor<$backend,3>)->(Tensor<$backend,3>,Tensor<$backend,3>) {
self.$gather(communicator).map_err(FullyShardedAwqError::Collective)?.forward_packed_with_positions(input,layout,masks,options,positions)
.map_err(FullyShardedAwqError::Projection)
}
}
impl<$($generics)*,P:GatherTransformerProjection<$backend,B>> FullyShardedAwqTransformerStack<$backend,P>
where P::Gathered:TransformerProjection<$backend> {
pub fn $forward<C,F>(&self,mut input:Tensor<$backend,3>,masks:DenseAttentionMask<$backend>,options:DenseAttentionOptions,communicator:C,mut positions:F)
-> Result<Tensor<$backend,3>,FullyShardedAwqError<C::Error,<P::Gathered as TransformerProjection<$backend>>::Error>>
where C:IntegerTensorCollective<B>,F:FnMut(usize,Tensor<$backend,4>,Tensor<$backend,4>)->(Tensor<$backend,4>,Tensor<$backend,4>) {
for (index,block) in self.blocks.iter().enumerate() {input=block.$forward(input,masks.clone(),options,communicator.clone(),|query,key|positions(index,query,key))?;}
Ok(input)
}
pub fn $packed<C,F>(&self,mut input:Tensor<$backend,2>,layout:&PackedSequenceLayout,masks:Option<&[PackedDocumentAttentionMask<$backend>]>,
options:PackedAttentionOptions,communicator:C,mut positions:F)
-> Result<Tensor<$backend,2>,FullyShardedAwqError<C::Error,<P::Gathered as TransformerProjection<$backend>>::Error>>
where C:IntegerTensorCollective<B>,F:FnMut(usize,Tensor<$backend,3>,Tensor<$backend,3>)->(Tensor<$backend,3>,Tensor<$backend,3>) {
for (index,block) in self.blocks.iter().enumerate() {input=block.$packed(input,layout,masks,options,communicator.clone(),|query,key|positions(index,query,key))?;}
Ok(input)
}
}
};
}
execute_awq_blocks!(B,[B:Backend],gather_inference,forward_inference,forward_packed_inference);
execute_awq_blocks!(Autodiff<B,S>,[B:Backend,S:CheckpointStrategy],gather,forward,forward_packed);
impl<B:Backend,P:GatherTransformerProjection<B,B>> FullyShardedAwqTransformerBlock<B,P>
where P::Gathered:TransformerProjection<B> {
pub fn forward_cached_inference<C,F>(&self,input:Tensor<B,3>,new_visible:Option<Tensor<B,2,Bool>>,cache:&mut ProjectedKvCache<B>,
masks:DenseAttentionMask<B>,options:DenseAttentionOptions,communicator:C,positions:F)
-> Result<Tensor<B,3>,FullyShardedAwqError<C::Error,<P::Gathered as TransformerProjection<B>>::Error>>
where C:IntegerTensorCollective<B>,F:FnOnce(Tensor<B,4>,Tensor<B,4>,usize)->(Tensor<B,4>,Tensor<B,4>) {
self.gather_inference(communicator).map_err(FullyShardedAwqError::Collective)?
.forward_cached_with_positions(input,new_visible,cache,masks,options,positions).map_err(FullyShardedAwqError::Projection)
}
}