pub(crate) mod admission;
pub mod batched_body;
pub mod batched_head;
pub mod forward_gpu;
pub mod gpu_ffn;
pub mod gpu_full_attn;
pub mod io_heads;
pub mod kv_cache;
pub mod kv_persist;
pub mod model;
pub mod native_matrix;
mod native_storage;
mod native_embedding;
pub mod profile;
pub mod tokenizer;
pub use kv_cache::{
DecodeRegime, DenseKvBuffers, GemmaLcpLayerKv, HbKvBuffers, HybridKvBuffers, MlxKvCache,
};
pub use model::{MlxModelWeights, MultiSeqPrefillOutput};
pub use profile::{KernelTypeProfile, ProfileAccumulator, TokenProfile};
#[cfg(test)]
mod admission_tests;
#[cfg(test)]
mod native_io_tests;