#[test]
#[ignore = "one-shot probe: needs Q2K_PROBE_DIR from the gguf-py dump"]
fn probe_dump_q2_k_decodes_for_gguf_py_comparison_3960() {
let Ok(dir) = std::env::var("Q2K_PROBE_DIR") else {
panic!("PROBE: Q2K_PROBE_DIR is unset -- this probe did NOT run");
};
let mut done = 0usize;
for entry in std::fs::read_dir(&dir).expect("probe dir") {
let path = entry.expect("dir entry").path();
if path.extension().and_then(|e| e.to_str()) != Some("bin") {
continue;
}
let bytes = std::fs::read(&path).expect("read bin");
let out = super::dequant::dequantize_q2_k(&bytes).expect("dequantize_q2_k");
let mut buf = Vec::with_capacity(out.len() * 4);
for v in &out {
buf.extend_from_slice(&v.to_le_bytes());
}
let dst = path.with_extension("apr.f32");
std::fs::write(&dst, &buf).expect("write apr.f32");
eprintln!(
"PROBE {}: {} bytes -> {} values",
path.display(),
bytes.len(),
out.len()
);
done += 1;
}
assert!(
done > 0,
"PROBE: no .bin files in {dir} -- nothing was compared"
);
}
const GOLDEN_BLOCKS: &[u8] = include_bytes!("fixtures/q2k_gguf_py_blocks.bin");
const GOLDEN_EXPECTED: &[u8] = include_bytes!("fixtures/q2k_gguf_py_expected.f32");
const Q2K_BLOCK_BYTES: usize = 84;
fn golden_expected() -> Vec<f32> {
GOLDEN_EXPECTED
.chunks_exact(4)
.map(|b| f32::from_le_bytes([b[0], b[1], b[2], b[3]]))
.collect()
}
#[test]
fn the_q2_k_decoder_is_bitwise_identical_to_gguf_py_3960() {
assert_eq!(
GOLDEN_BLOCKS.len(),
12 * Q2K_BLOCK_BYTES,
"fixture: 12 super-blocks of 84 bytes"
);
let want = golden_expected();
assert_eq!(want.len(), 12 * 256);
let got = super::dequant::dequantize_q2_k(GOLDEN_BLOCKS).expect("decode golden blocks");
assert_eq!(got.len(), want.len());
for (i, (g, w)) in got.iter().zip(&want).enumerate() {
assert_eq!(
g.to_bits(),
w.to_bits(),
"value {i} (block {}, lane {}): aprender {g} vs gguf-py {w}",
i / 256,
i % 256
);
}
assert!(want.iter().filter(|v| **v != 0.0).count() > want.len() * 9 / 10);
assert!(want.iter().any(|v| *v < 0.0) && want.iter().any(|v| *v > 0.0));
}
pub(crate) fn q2_k_block_by_lanes(block: &[u8]) -> [f32; 256] {
let f16 = |lo: u8, hi: u8| half::f16::from_le_bytes([lo, hi]).to_f32();
let (d, dmin) = (f16(block[80], block[81]), f16(block[82], block[83]));
let mut out = [0.0f32; 256];
for lane in 0..32usize {
let (g, s, h) = (lane / 16, (lane % 16) / 4, (lane % 4) / 2);
let sc = block[8 * g + 2 * s + h];
let dl = d * f32::from(sc & 0x0F);
let ml = dmin * f32::from(sc >> 4);
let qbase = 16 + 32 * g + 16 * h + 8 * (lane % 2);
for j in 0..8 {
let q = (block[qbase + j] >> (2 * s)) & 0x03;
out[8 * lane + j] = dl * f32::from(q) - ml;
}
}
out
}
#[test]
fn the_kernel_lane_mapping_reproduces_the_decoder_bitwise_3960() {
let want = golden_expected();
for (b, blk) in GOLDEN_BLOCKS.chunks_exact(Q2K_BLOCK_BYTES).enumerate() {
let got = q2_k_block_by_lanes(blk);
for (lane_j, (g, w)) in got.iter().zip(&want[b * 256..(b + 1) * 256]).enumerate() {
assert_eq!(
g.to_bits(),
w.to_bits(),
"block {b} output {lane_j} (lane {}, j {}): lane-math {g} vs gguf-py {w}",
lane_j / 8,
lane_j % 8
);
}
}
}
#[test]
fn q2_k_geometry_the_kernel_relies_on_3960() {
assert_eq!(Q2K_BLOCK_BYTES, 16 + 64 + 2 + 2, "scales[16] qs[64] d dmin");
assert_eq!(
Q2K_BLOCK_BYTES % 4,
0,
"block bases must stay 4-aligned for u32 qs loads"
);
for lane in 0..32usize {
let (g, h) = (lane / 16, (lane % 4) / 2);
let off = 16 + 32 * g + 16 * h + 8 * (lane % 2);
assert_eq!(off % 4, 0, "lane {lane}: qs run at +{off} is not 4-aligned");
assert!(off + 8 <= 80, "lane {lane}: qs run overlaps d/dmin");
}
let mut uses = [0u32; 16];
for lane in 0..32usize {
uses[8 * (lane / 16) + 2 * ((lane % 16) / 4) + (lane % 4) / 2] += 1;
}
assert!(
uses.iter().all(|u| *u == 2),
"scale-byte use per lane mapping: {uses:?}"
);
}