use crate::error::{RealizarError, Result};
use super::config::{GGUFConfig, ValidatedModelConfig};
use super::quantized::{QKVWeights, QuantizedTensorRef};
use super::types::{
GGUFModel, GGUF_TYPE_F32, GGUF_TYPE_Q2_K, GGUF_TYPE_Q4_0, GGUF_TYPE_Q4_1, GGUF_TYPE_Q5_0,
GGUF_TYPE_Q8_0,
};
#[cfg(test)]
use super::types::{GGUF_TYPE_BF16, GGUF_TYPE_F16, GGUF_TYPE_Q4_K, GGUF_TYPE_Q5_K, GGUF_TYPE_Q6_K};
pub struct QuantizedGGUFTransformerLayer {
pub attn_norm_weight: Vec<f32>,
pub attn_norm_bias: Option<Vec<f32>>,
pub qkv_weight: QKVWeights,
pub qkv_bias: Option<Vec<f32>>,
pub attn_output_weight: QuantizedTensorRef,
pub attn_output_bias: Option<Vec<f32>>,
pub ffn_up_weight: QuantizedTensorRef,
pub ffn_up_bias: Option<Vec<f32>>,
pub ffn_down_weight: QuantizedTensorRef,
pub ffn_down_bias: Option<Vec<f32>>,
pub ffn_gate_weight: Option<QuantizedTensorRef>,
pub ffn_gate_bias: Option<Vec<f32>>,
pub ffn_norm_weight: Option<Vec<f32>>,
pub ffn_norm_bias: Option<Vec<f32>>,
pub attn_q_norm_weight: Option<Vec<f32>>,
pub attn_k_norm_weight: Option<Vec<f32>>,
pub post_attn_norm_weight: Option<Vec<f32>>,
pub post_ffw_norm_weight: Option<Vec<f32>>,
}
pub fn unsupported_architecture_reason<'n>(
architecture: &str,
tensor_names: impl IntoIterator<Item = &'n str>,
) -> Option<String> {
let ssm_tensor = tensor_names
.into_iter()
.find(|name| name.contains("ssm_") || name.contains("ssm."));
let tensor_name = ssm_tensor?;
if hybrid_forward_handles(architecture) {
return None;
}
Some(format!(
"Architecture '{architecture}' uses SSM/Gated Delta Net layers (detected tensor '{tensor_name}'). \
There is no CPU or GPU forward for architecture '{architecture}' with SSM tensors. \
Tracking issues: #3090 (GPU) and #3091 (CPU) — both implement the Qwen3.5 hybrid ('qwen35'), not this architecture. \
Use a standard transformer model (e.g., Qwen2.5, LLaMA, Mistral) or wait for SSM support in a future release."
))
}
#[must_use]
pub fn hybrid_forward_handles(architecture: &str) -> bool {
architecture == "qwen35"
}
#[must_use]
pub fn moe_forward_handles(architecture: &str) -> bool {
crate::tensor_names::normalize_architecture(architecture) == "qwen3_moe"
}
pub(crate) fn dense_loader_refusal<'n>(
architecture: &str,
tensor_names: impl IntoIterator<Item = &'n str>,
) -> Option<String> {
if !hybrid_forward_handles(architecture) {
return None;
}
let tensor_name = tensor_names
.into_iter()
.find(|name| name.contains("ssm_") || name.contains("ssm."))?;
Some(format!(
"Architecture '{architecture}' is the Qwen3.5 hybrid (Gated Delta Net, detected tensor '{tensor_name}'): \
it runs through `Qwen35Model` (realizar::gguf::forward_qwen35, CPU #3091) or `Qwen35CudaModel` (GPU #3090), \
NOT through the dense QuantizedGGUFTransformer loader, whose per-layer tensor names this file does not carry. \
Load it with `apr run` / `apr chat`, which dispatch this architecture to the hybrid forward."
))
}
pub fn hybrid_gpu_quant_refusal<'n>(
architecture: &str,
tensors: impl IntoIterator<Item = (&'n str, u32)>,
) -> Option<String> {
if !hybrid_forward_handles(architecture) {
return None;
}
let (name, qtype) = super::loader::hybrid_gpu_unsupported_quant_tensor(tensors)?;
Some(format!(
"Architecture '{architecture}': tensor '{name}' has GGML type {qtype}, which has no verified GPU GEMV kernel. \
The GPU weight upload would decode it as Q4_K and produce garbage logits (PMAT-781/783/785). \
Run this model on the CPU (`apr run --no-gpu`), or requantize to Q4_0/Q4_1/Q5_0/Q8_0/Q4_K/Q5_K/Q6_K."
))
}
pub struct QuantizedGGUFTransformer<'a> {
pub config: GGUFConfig,
pub data: &'a [u8],
pub token_embedding: Vec<f32>,
pub position_embedding: Option<Vec<f32>>,
pub layers: Vec<QuantizedGGUFTransformerLayer>,
pub moe_layers: Vec<Option<crate::gguf::qwen3_moe_load::Qwen3MoeQuantizedLayer>>,
pub output_norm_weight: Vec<f32>,
pub output_norm_bias: Option<Vec<f32>>,
pub lm_head_weight: QuantizedTensorRef,
pub lm_head_bias: Option<Vec<f32>>,
}
impl<'a> QuantizedGGUFTransformer<'a> {
pub fn from_gguf(model: &GGUFModel, data: &'a [u8]) -> Result<Self> {
let config = ValidatedModelConfig::from_gguf(model)?.into_inner();
if let Some(reason) = unsupported_architecture_reason(
&config.architecture,
model.tensors.iter().map(|t| t.name.as_str()),
) {
return Err(crate::RealizarError::FormatError { reason });
}
if let Some(reason) = dense_loader_refusal(
&config.architecture,
model.tensors.iter().map(|t| t.name.as_str()),
) {
return Err(crate::RealizarError::FormatError { reason });
}
let canonical_arch = crate::tensor_names::normalize_architecture(&config.architecture);
if canonical_arch == "qwen3_moe" {
return Self::from_gguf_for_moe(model, data);
}
let token_embedding = model.get_tensor_f32("token_embd.weight", data)?;
let position_embedding = model
.get_tensor_f32("position_embd.weight", data)
.or_else(|_| model.get_tensor_f32("token_pos_embd.weight", data))
.or_else(|_| model.get_tensor_f32("model.position_embedding.weight", data))
.ok();
let mut layers = Vec::with_capacity(config.num_layers);
for layer_idx in 0..config.num_layers {
let layer = Self::load_quantized_layer(model, data, layer_idx)?;
layers.push(layer);
}
let output_norm_weight = model.get_tensor_f32("output_norm.weight", data)?;
let output_norm_bias = model
.get_tensor_f32("output_norm.bias", data)
.or_else(|_| model.get_tensor_f32("model.norm.bias", data))
.ok();
let lm_head_weight = Self::get_tensor_ref(model, data, "output.weight")
.or_else(|_| Self::get_tensor_ref(model, data, "token_embd.weight"))?;
let lm_head_bias = model.get_tensor_f32("output.bias", data).ok();
Ok(Self {
config,
data,
token_embedding,
position_embedding,
layers,
moe_layers: Vec::new(),
output_norm_weight,
output_norm_bias,
lm_head_weight,
lm_head_bias,
})
}
pub fn from_gguf_for_moe(model: &GGUFModel, data: &'a [u8]) -> Result<Self> {
let config = ValidatedModelConfig::from_gguf(model)?.into_inner();
let canonical_arch = crate::tensor_names::normalize_architecture(&config.architecture);
if canonical_arch != "qwen3_moe" {
return Err(crate::error::RealizarError::InvalidShape {
reason: format!(
"from_gguf_for_moe: architecture '{}' (canonical '{}') is not qwen3_moe — \
caller should dispatch to from_gguf instead",
config.architecture, canonical_arch
),
});
}
let has_ssm = model
.tensors
.iter()
.any(|t| t.name.contains("ssm_") || t.name.contains("ssm."));
if has_ssm {
return Err(crate::RealizarError::FormatError {
reason: format!(
"Architecture '{}' has both qwen3_moe arch tag AND SSM tensors — \
unsupported hybrid configuration",
config.architecture
),
});
}
let token_embedding = model.get_tensor_f32("token_embd.weight", data)?;
let position_embedding = model
.get_tensor_f32("position_embd.weight", data)
.or_else(|_| model.get_tensor_f32("token_pos_embd.weight", data))
.or_else(|_| model.get_tensor_f32("model.position_embedding.weight", data))
.ok();
let mut layers = Vec::with_capacity(config.num_layers);
let mut moe_layers = Vec::with_capacity(config.num_layers);
for layer_idx in 0..config.num_layers {
layers.push(Self::load_quantized_layer_moe_skeleton(
model, data, layer_idx,
)?);
moe_layers.push(Some(crate::gguf::qwen3_moe_load::load_qwen3_moe_layer(
model, data, layer_idx,
)?));
}
let output_norm_weight = model.get_tensor_f32("output_norm.weight", data)?;
let output_norm_bias = model
.get_tensor_f32("output_norm.bias", data)
.or_else(|_| model.get_tensor_f32("model.norm.bias", data))
.ok();
let lm_head_weight = Self::get_tensor_ref(model, data, "output.weight")
.or_else(|_| Self::get_tensor_ref(model, data, "token_embd.weight"))?;
let lm_head_bias = model.get_tensor_f32("output.bias", data).ok();
Ok(Self {
config,
data,
token_embedding,
position_embedding,
layers,
moe_layers,
output_norm_weight,
output_norm_bias,
lm_head_weight,
lm_head_bias,
})
}
fn load_quantized_layer_moe_skeleton(
model: &GGUFModel,
data: &[u8],
layer_idx: usize,
) -> Result<QuantizedGGUFTransformerLayer> {
let prefix = format!("blk.{layer_idx}");
let attn_norm_weight = model.get_tensor_f32(&format!("{prefix}.attn_norm.weight"), data)?;
let attn_norm_bias = model
.get_tensor_f32(&format!("{prefix}.attn_norm.bias"), data)
.or_else(|_| model.get_tensor_f32(&format!("{prefix}.input_layernorm.bias"), data))
.ok();
let q = Self::get_tensor_ref(model, data, &format!("{prefix}.attn_q.weight"))?;
let k = Self::get_tensor_ref(model, data, &format!("{prefix}.attn_k.weight"))?;
let v = Self::get_tensor_ref(model, data, &format!("{prefix}.attn_v.weight"))?;
let q_bias = model
.get_tensor_f32(&format!("{prefix}.attn_q.bias"), data)
.ok();
let k_bias = model
.get_tensor_f32(&format!("{prefix}.attn_k.bias"), data)
.ok();
let v_bias = model
.get_tensor_f32(&format!("{prefix}.attn_v.bias"), data)
.ok();
let qkv_bias = match (q_bias, k_bias, v_bias) {
(Some(qb), Some(kb), Some(vb)) => {
let mut combined = Vec::with_capacity(qb.len() + kb.len() + vb.len());
combined.extend_from_slice(&qb);
combined.extend_from_slice(&kb);
combined.extend_from_slice(&vb);
Some(combined)
},
_ => None,
};
let qkv_weight = QKVWeights::Separate { q, k, v };
let attn_output_weight =
Self::get_tensor_ref(model, data, &format!("{prefix}.attn_output.weight"))?;
let attn_output_bias = model
.get_tensor_f32(&format!("{prefix}.attn_output.bias"), data)
.ok();
let dense_ffn_placeholder = QuantizedTensorRef {
offset: 0,
byte_size: 0,
num_elements: 0,
qtype: GGUF_TYPE_F32,
};
let ffn_norm_weight = model
.get_tensor_f32(&format!("{prefix}.ffn_norm.weight"), data)
.ok();
let ffn_norm_bias = model
.get_tensor_f32(&format!("{prefix}.ffn_norm.bias"), data)
.or_else(|_| {
model.get_tensor_f32(&format!("{prefix}.post_attention_layernorm.bias"), data)
})
.ok();
let attn_q_norm_weight = model
.get_tensor_f32(&format!("{prefix}.attn_q_norm.weight"), data)
.ok();
let attn_k_norm_weight = model
.get_tensor_f32(&format!("{prefix}.attn_k_norm.weight"), data)
.ok();
let post_attn_norm_weight = model
.get_tensor_f32(&format!("{prefix}.post_attention_norm.weight"), data)
.ok();
let post_ffw_norm_weight = model
.get_tensor_f32(&format!("{prefix}.post_ffw_norm.weight"), data)
.ok();
Ok(QuantizedGGUFTransformerLayer {
attn_norm_weight,
attn_norm_bias,
qkv_weight,
qkv_bias,
attn_output_weight,
attn_output_bias,
ffn_up_weight: dense_ffn_placeholder.clone(),
ffn_up_bias: None,
ffn_down_weight: dense_ffn_placeholder,
ffn_down_bias: None,
ffn_gate_weight: None,
ffn_gate_bias: None,
ffn_norm_weight,
ffn_norm_bias,
attn_q_norm_weight,
attn_k_norm_weight,
post_attn_norm_weight,
post_ffw_norm_weight,
})
}
fn tensor_byte_size(qtype: u32, num_elements: usize, dims: &[u64]) -> Result<usize> {
const LEGACY_FLAT_SIZED: &[u32] = &[
GGUF_TYPE_Q4_0,
GGUF_TYPE_Q4_1,
GGUF_TYPE_Q5_0,
GGUF_TYPE_Q8_0,
GGUF_TYPE_Q2_K,
];
let scalar = super::ggml_type_table::traits(qtype).is_some_and(|t| t.blck_size == 1);
let sized = if scalar || LEGACY_FLAT_SIZED.contains(&qtype) {
super::ggml_type_table::flat_byte_size(qtype, num_elements)
} else {
super::ggml_type_table::byte_size(qtype, dims)
};
sized.map_err(|reason| RealizarError::UnsupportedOperation {
operation: "tensor_byte_size".to_string(),
reason,
})
}
fn resolve_qtype(
name: &str,
claimed_qtype: u32,
byte_size: usize,
num_elements: usize,
offset: usize,
data_len: usize,
) -> (usize, u32) {
if offset + byte_size <= data_len {
return (byte_size, claimed_qtype);
}
let avail = data_len.saturating_sub(offset);
let q4_0_size = num_elements.div_ceil(32) * 18;
if q4_0_size <= avail && q4_0_size > 0 {
eprintln!(
"[PAR-058-RESOLVED] Tensor '{name}' qtype mismatch: header says {claimed_qtype} but byte size suggests Q4_0. Using Q4_0."
);
return (q4_0_size, GGUF_TYPE_Q4_0);
}
let q8_0_size = num_elements.div_ceil(32) * 34;
if q8_0_size <= avail && q8_0_size > 0 {
eprintln!(
"[PAR-058-RESOLVED] Tensor '{name}' qtype mismatch: header says {claimed_qtype} but byte size suggests Q8_0. Using Q8_0."
);
return (q8_0_size, GGUF_TYPE_Q8_0);
}
(byte_size, claimed_qtype)
}
pub(crate) fn get_tensor_ref(
model: &GGUFModel,
data: &[u8],
name: &str,
) -> Result<QuantizedTensorRef> {
let tensor = model
.tensors
.iter()
.find(|t| t.name == name)
.ok_or_else(|| RealizarError::InvalidShape {
reason: format!("Tensor '{}' not found", name),
})?;
let num_elements: usize = tensor.dims.iter().map(|&d| d as usize).product();
let offset = model.tensor_data_start + tensor.offset as usize;
let byte_size = Self::tensor_byte_size(tensor.qtype, num_elements, &tensor.dims)?;
let (byte_size, actual_qtype) = Self::resolve_qtype(
name,
tensor.qtype,
byte_size,
num_elements,
offset,
data.len(),
);
if offset + byte_size > data.len() {
return Err(RealizarError::InvalidShape {
reason: format!(
"Tensor '{}' data range [{}, {}) exceeds file size {}",
name,
offset,
offset + byte_size,
data.len()
),
});
}
Ok(QuantizedTensorRef {
offset,
byte_size,
num_elements,
qtype: actual_qtype,
})
}
fn load_quantized_layer(
model: &GGUFModel,
data: &[u8],
layer_idx: usize,
) -> Result<QuantizedGGUFTransformerLayer> {
let prefix = format!("blk.{}", layer_idx);
let attn_norm_weight =
model.get_tensor_f32(&format!("{}.attn_norm.weight", prefix), data)?;
let attn_norm_bias = model
.get_tensor_f32(&format!("{}.attn_norm.bias", prefix), data)
.or_else(|_| model.get_tensor_f32(&format!("{}.input_layernorm.bias", prefix), data))
.ok();
let (qkv_weight, qkv_bias) = if let Ok(fused) =
Self::get_tensor_ref(model, data, &format!("{}.attn_qkv.weight", prefix))
{
let bias = model
.get_tensor_f32(&format!("{}.attn_qkv.bias", prefix), data)
.ok();
(QKVWeights::Fused(fused), bias)
} else {
let q = Self::get_tensor_ref(model, data, &format!("{}.attn_q.weight", prefix))?;
let k = Self::get_tensor_ref(model, data, &format!("{}.attn_k.weight", prefix))?;
let v = Self::get_tensor_ref(model, data, &format!("{}.attn_v.weight", prefix))?;
let q_bias = model
.get_tensor_f32(&format!("{}.attn_q.bias", prefix), data)
.ok();
let k_bias = model
.get_tensor_f32(&format!("{}.attn_k.bias", prefix), data)
.ok();
let v_bias = model
.get_tensor_f32(&format!("{}.attn_v.bias", prefix), data)
.ok();
let bias = match (q_bias, k_bias, v_bias) {
(Some(qb), Some(kb), Some(vb)) => {
let mut combined = Vec::with_capacity(qb.len() + kb.len() + vb.len());
combined.extend_from_slice(&qb);
combined.extend_from_slice(&kb);
combined.extend_from_slice(&vb);
Some(combined)
},
_ => None,
};
(QKVWeights::Separate { q, k, v }, bias)
};
let attn_output_weight =
Self::get_tensor_ref(model, data, &format!("{}.attn_output.weight", prefix))?;
let attn_output_bias = model
.get_tensor_f32(&format!("{}.attn_output.bias", prefix), data)
.ok();
let ffn_up_weight =
Self::get_tensor_ref(model, data, &format!("{}.ffn_up.weight", prefix))?;
let ffn_up_bias = model
.get_tensor_f32(&format!("{}.ffn_up.bias", prefix), data)
.or_else(|_| model.get_tensor_f32(&format!("{}.mlp.up_proj.bias", prefix), data))
.ok();
let ffn_down_weight =
Self::get_tensor_ref(model, data, &format!("{}.ffn_down.weight", prefix))?;
let ffn_down_bias = model
.get_tensor_f32(&format!("{}.ffn_down.bias", prefix), data)
.or_else(|_| model.get_tensor_f32(&format!("{}.mlp.down_proj.bias", prefix), data))
.ok();
let ffn_gate_weight =
Self::get_tensor_ref(model, data, &format!("{}.ffn_gate.weight", prefix)).ok();
let ffn_gate_bias = model
.get_tensor_f32(&format!("{}.ffn_gate.bias", prefix), data)
.ok();
let ffn_norm_weight = model
.get_tensor_f32(&format!("{}.ffn_norm.weight", prefix), data)
.ok();
let ffn_norm_bias = model
.get_tensor_f32(&format!("{}.ffn_norm.bias", prefix), data)
.or_else(|_| {
model.get_tensor_f32(&format!("{}.post_attention_layernorm.bias", prefix), data)
})
.ok();
let attn_q_norm_weight = model
.get_tensor_f32(&format!("{}.attn_q_norm.weight", prefix), data)
.ok();
let attn_k_norm_weight = model
.get_tensor_f32(&format!("{}.attn_k_norm.weight", prefix), data)
.ok();
let post_attn_norm_weight = model
.get_tensor_f32(&format!("{}.post_attention_norm.weight", prefix), data)
.ok();
let post_ffw_norm_weight = model
.get_tensor_f32(&format!("{}.post_ffw_norm.weight", prefix), data)
.ok();
Ok(QuantizedGGUFTransformerLayer {
attn_norm_weight,
attn_norm_bias,
qkv_weight,
qkv_bias,
attn_output_weight,
attn_output_bias,
ffn_up_weight,
ffn_up_bias,
ffn_down_weight,
ffn_down_bias,
ffn_gate_weight,
ffn_gate_bias,
ffn_norm_weight,
ffn_norm_bias,
attn_q_norm_weight,
attn_k_norm_weight,
post_attn_norm_weight,
post_ffw_norm_weight,
})
}
}
#[cfg(test)]
mod tensor_byte_size_tests {
use super::*;
use crate::gguf::types::GGUF_TYPE_Q3_K;
#[test]
fn f16_byte_size_is_two_bytes_per_element() {
let n = 1024;
let got = QuantizedGGUFTransformer::tensor_byte_size(GGUF_TYPE_F16, n, &[1024])
.expect("F16 must have a known byte size (PMAT-788)");
assert_eq!(got, n * 2, "F16 is 2 bytes/element");
}
#[test]
fn bf16_byte_size_is_two_bytes_per_element() {
let n = 768;
let got = QuantizedGGUFTransformer::tensor_byte_size(GGUF_TYPE_BF16, n, &[768])
.expect("BF16 must have a known byte size");
assert_eq!(got, n * 2);
}
#[test]
fn q3_k_byte_size_is_the_ggml_super_block_size() {
let got = QuantizedGGUFTransformer::tensor_byte_size(GGUF_TYPE_Q3_K, 256, &[256])
.expect("Q3_K has a defined ggml block size (#3432)");
assert_eq!(got, 110, "Q3_K is 110 bytes per 256-element super-block");
}
#[test]
fn iq_quants_used_by_real_unsloth_files_are_sized() {
let dims = [4u64, 512];
let n = 4 * 512;
let iq2_xxs = QuantizedGGUFTransformer::tensor_byte_size(16, n, &dims)
.expect("IQ2_XXS (type 16) must be sized — Qwen3.5-0.8B-UD-IQ2_XXS.gguf (#3432)");
assert_eq!(iq2_xxs, 4 * 2 * 66, "IQ2_XXS: 66 bytes per super-block");
let iq4_xs = QuantizedGGUFTransformer::tensor_byte_size(23, n, &dims)
.expect("IQ4_XS (type 23) must be sized — Qwen3.5-0.8B-IQ4_XS.gguf (#3432)");
assert_eq!(iq4_xs, 4 * 2 * 136, "IQ4_XS: 136 bytes per super-block");
}
#[test]
fn an_unknown_type_id_is_still_refused() {
let err = QuantizedGGUFTransformer::tensor_byte_size(9_999, 256, &[256])
.expect_err("an id outside the ggml table must not be sized");
let msg = err.to_string();
assert!(
msg.contains("9999"),
"the refusal must name the id it was given, got: {msg}"
);
}
#[test]
fn previously_supported_ids_keep_their_byte_sizes() {
let dims = [4u64, 1000];
let n = 4000usize;
let cases: [(u32, usize); 11] = [
(GGUF_TYPE_F32, 16_000),
(GGUF_TYPE_F16, 8_000),
(GGUF_TYPE_BF16, 8_000),
(GGUF_TYPE_Q4_0, 2_250),
(GGUF_TYPE_Q4_1, 2_500),
(GGUF_TYPE_Q5_0, 2_750),
(GGUF_TYPE_Q8_0, 4_250),
(GGUF_TYPE_Q2_K, 1_344),
(GGUF_TYPE_Q4_K, 2_304),
(GGUF_TYPE_Q5_K, 2_816),
(GGUF_TYPE_Q6_K, 3_360),
];
for (qtype, expected) in cases {
let got = QuantizedGGUFTransformer::tensor_byte_size(qtype, n, &dims)
.expect("previously supported qtype must stay supported");
assert_eq!(got, expected, "qtype {qtype} byte size changed");
}
}
#[test]
fn previously_supported_ids_keep_their_1d_byte_sizes() {
let dims = [1000u64];
let n = 1000usize;
let cases: [(u32, usize); 4] = [
(GGUF_TYPE_F32, 4_000),
(GGUF_TYPE_Q4_0, 32 * 18),
(GGUF_TYPE_Q2_K, 4 * 84),
(GGUF_TYPE_Q4_K, 4 * 144),
];
for (qtype, expected) in cases {
let got = QuantizedGGUFTransformer::tensor_byte_size(qtype, n, &dims)
.expect("previously supported qtype must stay supported");
assert_eq!(got, expected, "qtype {qtype} 1-D byte size changed");
}
}
}
#[cfg(test)]
mod unsupported_architecture_tests {
use super::{
dense_loader_refusal, hybrid_forward_handles, hybrid_gpu_quant_refusal,
unsupported_architecture_reason,
};
#[test]
fn qwen35_gated_delta_net_tensors_are_admitted() {
assert_eq!(
unsupported_architecture_reason(
"qwen35",
[
"token_embd.weight",
"blk.0.ssm_conv1d.weight",
"blk.0.attn_q.weight",
],
),
None,
"the hybrid forward runs qwen35 on both backends (#3090/#3091)"
);
assert_eq!(
unsupported_architecture_reason("qwen35", ["blk.0.ssm.a"]),
None,
"the dotted `ssm.` spelling is the same architecture"
);
}
#[test]
fn an_ssm_architecture_with_no_forward_is_still_refused() {
let reason = unsupported_architecture_reason(
"mamba",
[
"token_embd.weight",
"blk.0.ssm_conv1d.weight",
"blk.0.attn_q.weight",
],
)
.expect("an SSM architecture with no forward must be refused");
assert!(
reason.contains("mamba") && reason.contains("SSM/Gated Delta Net"),
"refusal must name the architecture and the reason, got: {reason}"
);
assert!(reason.contains("#3090"), "must mention GPU #3090");
assert!(reason.contains("#3091"), "must mention CPU #3091");
assert!(
reason.contains("blk.0.ssm_conv1d.weight"),
"must mention tensor name"
);
assert!(
!reason.contains("NEITHER the CPU nor the GPU"),
"the 'neither backend implements Gated DeltaNet' claim is false since #3091: {reason}"
);
assert!(
unsupported_architecture_reason("mamba", ["blk.0.ssm.a"]).is_some(),
"`ssm.` spelling must be refused too"
);
}
#[test]
fn unsupported_architecture_handled_set_is_the_runtime_dispatch_literal() {
assert!(hybrid_forward_handles("qwen35"));
for other in ["qwen3_5", "qwen3.5", "QWEN35", "mamba", "qwen2", "llama"] {
assert!(
!hybrid_forward_handles(other),
"'{other}' reaches no hybrid forward"
);
assert!(
unsupported_architecture_reason(other, ["blk.0.ssm_a"]).is_some(),
"'{other}' carries SSM tensors and has no forward — it must be refused"
);
}
}
#[test]
fn unsupported_architecture_dense_loader_still_refuses_the_hybrid() {
let reason = dense_loader_refusal("qwen35", ["token_embd.weight", "blk.0.ssm_a"])
.expect("the dense loader cannot load a hybrid GGUF");
assert!(
reason.contains("Qwen35Model") && reason.contains("forward_qwen35"),
"the refusal must name the loader that DOES work: {reason}"
);
assert!(
!reason.contains("NEITHER the CPU nor the GPU"),
"the dense loader's limits are not the runtime's: {reason}"
);
assert_eq!(
dense_loader_refusal("qwen2", ["blk.0.attn_q.weight"]),
None,
"a dense model must still load through the dense loader"
);
assert_eq!(
dense_loader_refusal("mamba", ["blk.0.ssm_a"]),
None,
"an architecture no forward handles is refused by the shared predicate, not here"
);
}
#[test]
fn unsupported_architecture_hybrid_quant_gate_judges_the_deltanet_tensors() {
let reason = hybrid_gpu_quant_refusal(
"qwen35",
[
("token_embd.weight", 12u32),
("blk.0.ssm_alpha.weight", 11), ("blk.0.ffn_down.weight", 12),
],
)
.expect("a Q3_K DeltaNet projection has no GPU kernel — it must be refused");
assert!(
reason.contains("blk.0.ssm_alpha.weight") && reason.contains("11"),
"the refusal must name the tensor and its GGML type: {reason}"
);
assert_eq!(
hybrid_gpu_quant_refusal(
"qwen35",
[
("blk.0.attn_qkv.weight", 8u32),
("blk.0.ssm_alpha.weight", 12),
("blk.0.ssm_out.weight", 14),
],
),
None,
"Q8_0/Q4_K/Q6_K all have verified GPU GEMV kernels"
);
assert_eq!(
hybrid_gpu_quant_refusal("qwen2", [("blk.0.attn_q.weight", 11u32)]),
None,
"a dense model is judged by the dense quant gate, not this one"
);
}
#[test]
fn standard_transformer_tensors_are_accepted() {
assert_eq!(
unsupported_architecture_reason(
"qwen2",
[
"token_embd.weight",
"blk.0.attn_q.weight",
"blk.0.ffn_down.weight",
"output.weight",
],
),
None,
"a dense transformer must load"
);
}
}
include!("transformer_quantized_layer_field.rs");
#[cfg(test)]
mod moe_forward_handles_tests {
use super::{hybrid_forward_handles, moe_forward_handles};
#[test]
fn moe_forward_handles_exactly_what_the_runtime_dispatches() {
for arch in ["qwen3moe", "qwen3_moe"] {
assert!(moe_forward_handles(arch), "{arch} runs the MoE forward");
}
for arch in [
"qwen35moe",
"qwen3_5moe",
"qwen3",
"qwen35",
"qwen2",
"llama",
"mixtral",
] {
assert!(
!moe_forward_handles(arch),
"{arch} is not the qwen3moe forward"
);
}
for arch in ["qwen3moe", "qwen3_moe", "qwen35"] {
assert!(!(moe_forward_handles(arch) && hybrid_forward_handles(arch)));
}
}
}