pub struct KvCacheQuant<B: BackendKvDtype<K>, K: KvDtypeKind> {Show 13 fields
pub k: <B as BackendKvDtype<K>>::KvBuffer,
pub v: <B as BackendKvDtype<K>>::KvBuffer,
pub k_scales: <B as BackendKvDtype<K>>::KvScales,
pub v_scales: <B as BackendKvDtype<K>>::KvScales,
pub len: usize,
pub capacity: usize,
pub num_kv_heads: usize,
pub head_dim: usize,
pub block_size: usize,
pub block_table: Option<B::Buffer>,
pub context_lens: Option<B::Buffer>,
pub paged_block_indices: Vec<u32>,
pub _kv_dtype: PhantomData<K>,
}Expand description
Quantized-KV cache (Dim 5 INT8 / future FP8 paths). Sibling of
KvCache for backends that store K/V in a non-FP16 element type
plus per-token per-kv-head scales.
Why a separate struct: the FP16 KvCache<B, K> uses B::Buffer
uniformly, which is FP16 on every concrete backend. Stuffing INT8
storage into that buffer would require unsafe transmutes; making
the FP16 struct generic over the storage type would force every
existing call site (4 model files, ~20 functions) to pick up an
equality-bound on the associated type. Keeping a parallel struct
for INT8 is the cheaper trade — the kernel launchers in
[crate::int8_kv] take cudarc primitives directly anyway.
KStorage and ScaleStorage come from BackendKvDtype<K>::KvBuffer
and BackendKvDtype<K>::KvScales. On CUDA they wrap CudaSlice<i8>
and CudaSlice<f16>.
Fields§
§k: <B as BackendKvDtype<K>>::KvBuffer§v: <B as BackendKvDtype<K>>::KvBuffer§k_scales: <B as BackendKvDtype<K>>::KvScales§v_scales: <B as BackendKvDtype<K>>::KvScales§len: usize§capacity: usize§num_kv_heads: usize§head_dim: usize§block_size: usize§block_table: Option<B::Buffer>§context_lens: Option<B::Buffer>§paged_block_indices: Vec<u32>§_kv_dtype: PhantomData<K>Auto Trait Implementations§
impl<B, K> Freeze for KvCacheQuant<B, K>where
<B as BackendKvDtype<K>>::KvBuffer: Freeze,
<B as BackendKvDtype<K>>::KvScales: Freeze,
<B as Backend>::Buffer: Freeze,
impl<B, K> RefUnwindSafe for KvCacheQuant<B, K>where
<B as BackendKvDtype<K>>::KvBuffer: RefUnwindSafe,
<B as BackendKvDtype<K>>::KvScales: RefUnwindSafe,
<B as Backend>::Buffer: RefUnwindSafe,
K: RefUnwindSafe,
impl<B, K> Send for KvCacheQuant<B, K>
impl<B, K> Sync for KvCacheQuant<B, K>
impl<B, K> Unpin for KvCacheQuant<B, K>where
<B as BackendKvDtype<K>>::KvBuffer: Unpin,
<B as BackendKvDtype<K>>::KvScales: Unpin,
<B as Backend>::Buffer: Unpin,
K: Unpin,
impl<B, K> UnsafeUnpin for KvCacheQuant<B, K>where
<B as BackendKvDtype<K>>::KvBuffer: UnsafeUnpin,
<B as BackendKvDtype<K>>::KvScales: UnsafeUnpin,
<B as Backend>::Buffer: UnsafeUnpin,
impl<B, K> UnwindSafe for KvCacheQuant<B, K>where
<B as BackendKvDtype<K>>::KvBuffer: UnwindSafe,
<B as BackendKvDtype<K>>::KvScales: UnwindSafe,
<B as Backend>::Buffer: UnwindSafe,
K: UnwindSafe,
Blanket Implementations§
Source§impl<T> BorrowMut<T> for Twhere
T: ?Sized,
impl<T> BorrowMut<T> for Twhere
T: ?Sized,
Source§fn borrow_mut(&mut self) -> &mut T
fn borrow_mut(&mut self) -> &mut T
Source§impl<T> Instrument for T
impl<T> Instrument for T
Source§fn instrument(self, span: Span) -> Instrumented<Self> ⓘ
fn instrument(self, span: Span) -> Instrumented<Self> ⓘ
Source§fn in_current_span(self) -> Instrumented<Self> ⓘ
fn in_current_span(self) -> Instrumented<Self> ⓘ
Source§impl<T> IntoEither for T
impl<T> IntoEither for T
Source§fn into_either(self, into_left: bool) -> Either<Self, Self> ⓘ
fn into_either(self, into_left: bool) -> Either<Self, Self> ⓘ
self into a Left variant of Either<Self, Self>
if into_left is true.
Converts self into a Right variant of Either<Self, Self>
otherwise. Read moreSource§fn into_either_with<F>(self, into_left: F) -> Either<Self, Self> ⓘ
fn into_either_with<F>(self, into_left: F) -> Either<Self, Self> ⓘ
self into a Left variant of Either<Self, Self>
if into_left(&self) returns true.
Converts self into a Right variant of Either<Self, Self>
otherwise. Read more