#[cfg(test)]
#[cfg(feature = "cuda")]
#[allow(clippy::unwrap_used, clippy::expect_used, clippy::panic)]
mod iq4_xs_device_ab_tests {
use super::*;
use crate::quantize::iq4_xs::{dequantize_iq4_xs, GGML_TYPE_IQ4_XS, IQ4_XS_BLOCK_BYTES};
const QK: usize = 256;
const TOL: f32 = 1e-5;
fn create_executor() -> Option<CudaExecutor> {
CudaExecutor::new(0).ok()
}
fn iq4_xs_weights(n: usize, k: usize, seed: u32) -> Vec<u8> {
let blocks = n * (k / QK);
let mut data = Vec::with_capacity(blocks * IQ4_XS_BLOCK_BYTES);
let mut state = seed;
for _ in 0..blocks {
data.push(0x00);
data.push(0x3c); for _ in 0..(IQ4_XS_BLOCK_BYTES - 2) {
state = state.wrapping_mul(1_664_525).wrapping_add(1_013_904_223);
data.push((state >> 24) as u8);
}
}
data
}
fn ab(exec: &mut CudaExecutor, label: &str, weights: &[u8], k: usize, n: usize) -> f32 {
ab_with_device_bytes(exec, label, weights, weights, k, n)
}
fn ab_with_device_bytes(
exec: &mut CudaExecutor,
label: &str,
weights: &[u8],
device: &[u8],
k: usize,
n: usize,
) -> f32 {
assert_eq!(k % QK, 0, "{label}: k={k} is not a multiple of 256");
assert_eq!(
weights.len(),
n * (k / QK) * IQ4_XS_BLOCK_BYTES,
"{label}: byte count"
);
let input: Vec<f32> = (0..k)
.map(|i| (((i * 7 + 3) % 17) as f32 - 8.0) / 4.0)
.collect();
let expected = crate::quantize::iq_parallel_matvec(GGML_TYPE_IQ4_XS, weights, &input, k, n)
.expect("CPU IQ4_XS matvec");
let dense = dequantize_iq4_xs(weights).expect("CPU IQ4_XS dequant");
assert_eq!(dense.len(), n * k, "{label}: dequant length");
let w_buf = GpuBuffer::from_host(&exec.context, device).unwrap();
let x_buf = GpuBuffer::from_host(&exec.context, &input).unwrap();
let y_buf = GpuBuffer::from_host(&exec.context, &vec![f32::NAN; n]).unwrap();
exec.iq4_xs_gemv_into(
w_buf.as_ptr(),
&x_buf,
&y_buf,
u32::try_from(n).unwrap(),
u32::try_from(k).unwrap(),
)
.expect("IQ4_XS GEMV launch");
exec.stream.synchronize().unwrap();
let mut got = vec![0.0f32; n];
y_buf.copy_to_host(&mut got).unwrap();
let mut worst = 0.0f32;
let mut worst_row = 0usize;
let mut nonfinite = 0usize;
for row in 0..n {
let bound: f32 = dense[row * k..(row + 1) * k]
.iter()
.zip(&input)
.map(|(w, x)| (w * x).abs())
.sum();
if !got[row].is_finite() {
nonfinite += 1;
continue;
}
let err = (got[row] - expected[row]).abs() / bound.max(1e-6);
if err > worst {
worst = err;
worst_row = row;
}
}
assert_eq!(
nonfinite, 0,
"{label} k={k} n={n}: {nonfinite} rows came back non-finite — the output buffer is \
pre-filled with NaN, so these rows were never written"
);
let nonzero = expected.iter().filter(|v| v.abs() > 1e-6).count();
assert!(
nonzero >= n / 2,
"{label}: only {nonzero}/{n} reference rows non-zero — vacuous"
);
eprintln!(
"#3951 {label:<22} k={k:5} n={n:5} blocks/row={:2} worst={worst:.3e} (row {worst_row}: GPU {} CPU {})",
k / QK,
got[worst_row],
expected[worst_row]
);
worst
}
#[test]
fn the_iq4_xs_kernel_agrees_with_the_cpu_decoder_at_every_shape_the_failing_model_uses() {
let Some(mut exec) = create_executor() else {
eprintln!("SKIP: no CUDA device — this is the one check that needs one");
return;
};
let shapes: [(&str, usize, usize); 9] = [
("ADMITTED 4B ffn", 2560, 9216),
("attn_gate", 1024, 2048),
("attn_qkv", 1024, 6144),
("ffn_down", 3584, 1024),
("ffn_gate/up", 1024, 3584),
("attn_k", 1024, 512),
("attn_output", 2048, 1024),
("attn_q", 1024, 4096),
("single block row", 256, 64),
];
let mut failures = Vec::new();
for (i, (label, k, n)) in shapes.iter().enumerate() {
let w = iq4_xs_weights(*n, *k, 0x9E37_79B9 ^ (i as u32).wrapping_mul(0x85EB_CA6B));
let worst = ab(&mut exec, label, &w, *k, *n);
if worst > TOL {
failures.push(format!("{label} k={k} n={n}: worst {worst:.3e}"));
}
}
assert!(
failures.is_empty(),
"#3951: the IQ4_XS GPU kernel disagrees with the CPU decoder beyond {TOL:e} of the \
row bound at: {failures:?}. The CPU decoder is bit-exact vs gguf-py (#3947)."
);
}
#[test]
fn the_iq4_xs_kernel_agrees_with_the_cpu_decoder_on_real_tensors() {
let Ok(dir) = std::env::var("APR_IQ4XS_DUMP_DIR") else {
eprintln!("SKIP: APR_IQ4XS_DUMP_DIR unset");
return;
};
let Some(mut exec) = create_executor() else {
eprintln!("SKIP: no CUDA device");
return;
};
let mut entries: Vec<_> = std::fs::read_dir(&dir)
.expect("dump dir")
.filter_map(Result::ok)
.map(|e| e.path())
.filter(|p| p.extension().is_some_and(|x| x == "bin"))
.collect();
entries.sort();
assert!(
!entries.is_empty(),
"no .bin tensors in {dir} — would pass vacuously"
);
let mut failures = Vec::new();
for p in &entries {
let stem = p.file_stem().unwrap().to_string_lossy().into_owned();
let parts: Vec<&str> = stem.split("__").collect();
let k: usize = parts[1].trim_start_matches('k').parse().unwrap();
let n: usize = parts[2].trim_start_matches('n').parse().unwrap();
let bytes = std::fs::read(p).unwrap();
let worst = ab(&mut exec, parts[0], &bytes, k, n);
if worst > TOL {
failures.push(format!("{} k={k} n={n}: worst {worst:.3e}", parts[0]));
}
}
assert!(
failures.is_empty(),
"#3951 real tensors disagree: {failures:?}"
);
}
#[test]
fn the_harness_goes_red_on_a_single_corrupted_block() {
let Some(mut exec) = create_executor() else {
eprintln!("SKIP: no CUDA device");
return;
};
let (k, n) = (3584usize, 1024usize);
let weights = iq4_xs_weights(n, k, 0x5EED_1234);
let mut device = weights.clone();
let row = 517usize;
let block_in_row = 9usize;
let at = (row * (k / QK) + block_in_row) * IQ4_XS_BLOCK_BYTES;
for b in &mut device[at + 8..at + IQ4_XS_BLOCK_BYTES] {
*b = !*b; }
let worst = ab_with_device_bytes(&mut exec, "PLANTED 1 block", &weights, &device, k, n);
assert!(
worst > TOL,
"#3951 negative control: one corrupted block in {} (row {row}) produced worst \
{worst:.3e} <= {TOL:e}. The harness cannot see a single-block defect, so its \
greens license nothing.",
n * (k / QK)
);
}
}