aprender-serve 0.70.2

Pure Rust ML inference engine built from scratch - model serving for GGUF and safetensors
//! GGUF (GPT-Generated Unified Format) parser
//!
//! Pure Rust implementation of GGUF binary format reader.
//! Used by llama.cpp, Ollama, and compatible tools.
//!
//! Format specification: <https://github.com/ggerganov/ggml/blob/master/docs/gguf.md>
//!
//! ## Module Structure
//!
//! This module is being incrementally refactored from a 54K-line monolith
//! into focused submodules for better testability and coverage.

// GGUF Module Structure
//
// Incremental shatter of src/gguf.rs (54K lines) into domain modules.
// Each module should be ≤800 lines for testability.
//
// Shatter Plan (19 modules from 54K lines):
// 🚧 types.rs: Additional tests for constants (~50 lines)
// - header.rs: GGUFHeader, TensorInfo
// - model.rs: GGUFModel, MappedGGUFModel
// - config.rs: GGUFConfig
// - transformer.rs: GGUFTransformer, GGUFTransformerLayer
// - quantized.rs: Quantized tensor types
// - owned.rs: OwnedQuantized* types
// - cached.rs: Cached model variants
// - batching.rs: Batch processing
// - scheduling.rs: Request scheduling
// - gpu_buffer.rs: GPU buffer management
// - prefix_cache.rs: Prefix caching
// - kv_cache.rs: KV cache types
// - inference.rs: OwnedQuantizedModel inference impl
// - cuda.rs: CUDA-specific code
//
// Migration Strategy: Include monolith, gradually extract, re-export all

// Modular structure
mod batch_scheduler;
mod config;
#[cfg(feature = "cuda")]
mod cuda;
#[cfg(feature = "cuda")]
mod cuda_model;
/// #3432: the one ggml `type_traits` table (block size + bytes per block).
pub mod ggml_type_table;
mod inference;
mod inference_types;
mod io;
pub(crate) mod keys;
mod loader;
mod model;
mod owned;
#[cfg(feature = "cuda")]
pub mod parity;
mod quantized;
pub mod qwen35_load;
pub mod qwen3_moe_load;
mod runtime;
mod transformer;
mod types;
pub(crate) mod utils;
#[cfg(feature = "gpu")]
mod wgpu_backend;
#[cfg(feature = "gpu")]
mod wgpu_model;

// Pure math operations (shared between CPU and GPU paths)
// UCBD §4: pub for re-export of rms_norm at crate root
/// #3726: canonical byte-level BPE (pre-tokenizer + ranked merges) for `gpt2` vocabularies.
pub mod byte_level_bpe;
pub mod ops;

// Test helpers module - shared utilities for GGUF tests
#[cfg(test)]
pub(crate) mod test_helpers;

// Test factory module - synthesize valid GGUF files in memory
#[cfg(test)]
pub(crate) mod test_factory;

// Rosetta format factory - synthesize all model formats (GGUF, SafeTensors, APR)
#[cfg(test)]
pub(crate) mod format_factory;

// Re-export types from organized modules
pub use batch_scheduler::*;
pub use config::*;
#[cfg(feature = "cuda")]
pub use cuda::{
    BatchedDecodeState, CudaBackend, CudaInitError, Qwen35CudaModel, Qwen35CudaState,
    Qwen3MoeCudaModel, Qwen3MoeCudaState, Qwen3MoeShape,
};
#[cfg(feature = "cuda")]
pub use cuda_model::*;
pub use model::*;
// PMAT-785: single-source-of-truth GPU quant whitelist predicate, shared between
// the primary `apr run`/`apr serve` gate (infer::is_legacy_gguf_quant) and the
// construction-time gate (OwnedQuantizedModel::has_gpu_unsupported_quant).
// `loader.rs` include!()s `dtype.rs`, where the predicate is defined.
pub(crate) use loader::gpu_unsupported_quant_qtype;
pub use quantized::*;
pub use runtime::*;
#[cfg(feature = "gpu")]
pub use wgpu_model::*;
pub mod logprobs;
pub use logprobs::*;
pub use transformer::*;
pub use types::*;

// Re-export inference types
pub use inference_types::*;

// Re-export cached model types from inference module
#[cfg(any(feature = "gpu", feature = "cuda"))]
pub use inference::{
    DequantizedFFNWeights, DequantizedWeightCache, OwnedQuantizedModelCached,
    OwnedQuantizedModelCachedSync,
};

// Tests module - shattered from monolith into focused part files
#[cfg(test)]
mod format_factory_tests;
#[cfg(test)]
mod inference_types_tests;
#[cfg(test)]
mod io_tests;
#[cfg(test)]
mod quantized_tests;
#[cfg(test)]
mod tests;

/// The dense (llama/qwen2/qwen3/...) forward behind the one engine (#4268):
/// `apr run`, `run --batch`, `chat` and `serve` drive a dense GGUF through
/// [`crate::session::Session`] on the CPU or the CUDA backend.
#[path = "inference/forward/dense_session.rs"]
pub mod dense_session;
/// The dense CUDA forward over a borrowed model: serve's scheduler turn (#4280).
#[cfg(feature = "cuda")]
#[path = "inference/forward/dense_session_borrowed.rs"]
pub mod dense_session_borrowed;
/// #3604: the F2 hybrid guard's receipt. CUDA-free on purpose, so its decision
/// table is tested on every build.
#[path = "inference/forward/f2_receipt.rs"]
pub mod f2_receipt;
/// Qwen3.5 / Qwen3.8 hybrid (Gated `DeltaNet` + gated attention) CPU forward (#3091).
#[path = "inference/forward/forward_qwen35.rs"]
pub mod forward_qwen35;
/// PMAT-4269 (M1): the Qwen3-MoE CPU forward behind the one engine.
#[path = "inference/forward/moe_session.rs"]
pub mod moe_session;
/// The Qwen3.5 hybrid held resident across calls — one build, one F2 guard, a
/// decode state that outlives the turn (#3595 `apr chat`, #3571 `apr serve`).
#[path = "inference/forward/qwen35_session.rs"]
pub mod qwen35_session;