pub enum QTensor {
F32 {
data: Vec<f32>,
rows: usize,
cols: usize,
},
Mapped {
model: Arc<CmfModel>,
idx: usize,
dtype: TensorDtype,
rows: usize,
cols: usize,
row_scale: Vec<f32>,
col_field: Vec<f32>,
vbit_offsets: Vec<usize>,
repack: Vec<u8>,
},
}Variants§
F32
Mapped
Fields
dtype: TensorDtypevbit_offsets: Vec<usize>Vbit only: byte offset of each row’s packed data within the
tensor blob ([rows + 1], computed once at load — the per-
matvec prefix scan over row bit-widths was O(rows) each call).
repack: Vec<u8>q8-family decode repack (load-time, optional): rows in groups of 4, interleaved in 16-byte units — one 64-byte line per iteration feeds all 4 sdot lanes, ONE sequential weight stream per worker instead of four (this is where llama.cpp’s repacked Q8 kernels get their bandwidth). Empty = off (CMF_REPACK=0, non-SDOT arch, or an ineligible shape). Trades an anonymous copy of the quants for mmap pages that go cold.
Implementations§
Source§impl QTensor
impl QTensor
pub fn from_f32(data: Vec<f32>, rows: usize, cols: usize) -> Self
Sourcepub fn from_model(model: &Arc<CmfModel>, name: &str) -> Result<Self, String>
pub fn from_model(model: &Arc<CmfModel>, name: &str) -> Result<Self, String>
Wrap a directory tensor without dequantizing the payload. Falls back to dequantized f32 for dtypes without a fused kernel.
Examples found in repository?
154fn main() -> Result<(), Box<dyn Error>> {
155 let mut args = std::env::args_os().skip(1);
156 let dir = PathBuf::from(args.next().ok_or("missing raw fixture directory")?);
157 let layer = dir
158 .file_name()
159 .and_then(|s| s.to_str())
160 .and_then(|s| s.strip_prefix("raw-layer-"))
161 .ok_or("fixture directory must be named raw-layer-N")?;
162 let prefix = format!("model.layers.{layer}.engram");
163 let cmf = dir.join("engram-component.cmf");
164 write_fixture(&dir, &cmf)?;
165 let model = Arc::new(CmfModel::open(&cmf)?);
166 let embed = RawFp8Rows::from_model(
167 &model,
168 &format!("{prefix}.embed.weight"),
169 &format!("{prefix}.embed.scale"),
170 )?;
171 let wkv = QTensor::from_model(&model, &format!("{prefix}.wkv.weight"))?;
172 let q_weight = f32_tensor(&model, &format!("{prefix}.q_weight"))?;
173 let k_weight = f32_tensor(&model, &format!("{prefix}.k_weight"))?;
174 let engram = Dsv41Engram {
175 embed,
176 wkv,
177 q_weight,
178 k_weight,
179 };
180 let cfg = cfg();
181 let meta: Value = serde_json::from_slice(&fs::read(dir.join("manifest.json"))?)?;
182 let mut max_abs = 0.0f32;
183 let mut sum_abs = 0.0f64;
184 let mut count = 0usize;
185 let mut cases = 0usize;
186 for case in 0..3 {
187 let prefix = format!("case{case}");
188 let input_shape = meta[&format!("{prefix}.input")]["shape"]
189 .as_array()
190 .unwrap();
191 let seq = input_shape[1].as_u64().unwrap() as usize;
192 let input = read_bf16(&dir.join(format!("{prefix}_input.bin")))?;
193 let expected = read_bf16(&dir.join(format!("{prefix}_output.bin")))?;
194 let index_bytes = fs::read(dir.join(format!("{prefix}_indices.bin")))?;
195 let mut indices = Vec::with_capacity(index_bytes.len() / 8);
196 for b in index_bytes.chunks_exact(8) {
197 indices.push(u64::from_le_bytes(b.try_into().unwrap()) as usize);
198 }
199 let token_mask: Vec<bool> = if case == 1 {
200 vec![
201 true, true, true, true, true, false, false, false, true, true, true, true, true,
202 true, true, true, true, true, true, true,
203 ]
204 } else {
205 vec![true; seq]
206 };
207 for token in 0..seq {
208 let mut h = input[token * 4 * 5120..(token + 1) * 4 * 5120].to_vec();
209 let hashes = &indices[token * 24..(token + 1) * 24];
210 dsv41_apply_engram_for_test(&engram, &mut h, hashes, &cfg, token_mask[token]);
211 for (&got, &want) in h
212 .iter()
213 .zip(&expected[token * 4 * 5120..(token + 1) * 4 * 5120])
214 {
215 let d = (got - want).abs();
216 max_abs = max_abs.max(d);
217 sum_abs += d as f64;
218 count += 1;
219 }
220 }
221 cases += 1;
222 println!("case={case} seq={seq} cumulative_max_abs={max_abs:.8e}",);
223 }
224 println!(
225 "summary cases={cases} values={count} max_abs={max_abs:.8e} mean_abs={:.8e}",
226 sum_abs / count.max(1) as f64
227 );
228 Ok(())
229}More examples
21fn main() {
22 let mut args = std::env::args().skip(1);
23 let path = args
24 .next()
25 .expect("usage: matvec_bw <model.cmf> [sweep|one <tensor>]");
26 let mode = args.next().unwrap_or_else(|| "sweep".to_string());
27 let model = Arc::new(cortiq_core::CmfModel::open(&path).expect("open model"));
28
29 // Real LM activations carry a few heavy channels (>8·rms); measured
30 // mean on this model is ~3.7. NOUT models that distribution.
31 let nout: usize = std::env::var("NOUT")
32 .ok()
33 .and_then(|v| v.parse().ok())
34 .unwrap_or(4);
35 let mk_x = |cols: usize| -> Vec<f32> {
36 let mut x: Vec<f32> = (0..cols).map(|i| ((i % 17) as f32 - 8.0) / 8.0).collect();
37 for k in 0..nout {
38 x[k * 37 % cols] = 40.0;
39 }
40 x
41 };
42
43 if mode == "one" {
44 let name = args
45 .next()
46 .unwrap_or_else(|| "model.embed_tokens.weight".to_string());
47 let entry = model.tensor(&name).expect("tensor not found");
48 let (rows, cols) = (entry.shape[0], entry.shape[1]);
49 let nbytes = entry.nbytes as f64;
50 println!(
51 "tensor {name}: {rows}x{cols} {:?} = {:.1} MB, NOUT={nout}",
52 entry.dtype,
53 nbytes / 1e6
54 );
55 let t = QTensor::from_model(&model, &name).expect("wrap");
56 let x = mk_x(cols);
57 let mut out = vec![0f32; rows];
58 t.matvec(&x, &mut out, None);
59 for nt in [1usize, 2, 4, 6, 8, 10] {
60 let pool = if nt == 1 { None } else { Some(Pool::new(nt)) };
61 let iters = 8;
62 t.matvec(&x, &mut out, pool.as_ref());
63 let t0 = Instant::now();
64 for _ in 0..iters {
65 t.matvec(&x, &mut out, pool.as_ref());
66 }
67 let el = t0.elapsed().as_secs_f64();
68 println!(
69 "threads={nt:2} {:6.2} ms/matvec {:6.1} GB/s (sink {:.3})",
70 el / iters as f64 * 1e3,
71 nbytes * iters as f64 / el / 1e9,
72 out[0]
73 );
74 }
75 return;
76 }
77
78 // sweep: every 2-D q8_2f tensor once = one decode's worth of weights.
79 let names: Vec<String> = model
80 .tensors
81 .iter()
82 .filter(|t| t.dtype == TensorDtype::Q8_2f && t.shape.len() == 2)
83 .map(|t| t.name.clone())
84 .collect();
85 let total_bytes: f64 = model
86 .tensors
87 .iter()
88 .filter(|t| t.dtype == TensorDtype::Q8_2f && t.shape.len() == 2)
89 .map(|t| t.nbytes as f64)
90 .sum();
91 println!(
92 "sweep: {} q8_2f tensors, {:.2} GB total (= weights streamed per decode token), NOUT={nout}",
93 names.len(),
94 total_bytes / 1e9
95 );
96
97 let tensors: Vec<(QTensor, Vec<f32>, Vec<f32>)> = names
98 .iter()
99 .map(|n| {
100 let e = model.tensor(n).unwrap();
101 let (rows, cols) = (e.shape[0], e.shape[1]);
102 (
103 QTensor::from_model(&model, n).expect("wrap"),
104 mk_x(cols),
105 vec![0f32; rows],
106 )
107 })
108 .collect();
109 let mut tensors = tensors;
110
111 // Whole-model residency pass BEFORE any timing: the first touch of a
112 // 4.2 GB mmap faults ~260k pages, which would otherwise be charged to
113 // whichever thread count happens to run first. REVERSE=1 flips the
114 // order as a check that no first-touch cost is left in the table.
115 for _ in 0..2 {
116 for (t, x, out) in tensors.iter_mut() {
117 t.matvec(x, out, None);
118 }
119 }
120 let mut counts = vec![1usize, 2, 4, 6, 8, 10];
121 if std::env::var("REVERSE").is_ok() {
122 counts.reverse();
123 }
124 for nt in counts {
125 let pool = if nt == 1 { None } else { Some(Pool::new(nt)) };
126 // one warm pass, then two measured
127 for (t, x, out) in tensors.iter_mut() {
128 t.matvec(x, out, pool.as_ref());
129 }
130 let iters = 2;
131 let t0 = Instant::now();
132 for _ in 0..iters {
133 for (t, x, out) in tensors.iter_mut() {
134 t.matvec(x, out, pool.as_ref());
135 }
136 }
137 let el = t0.elapsed().as_secs_f64();
138 let per_tok = el / iters as f64;
139 println!(
140 "threads={nt:2} {:7.1} ms/sweep {:6.1} GB/s -> weight-path-only ceiling {:5.1} tok/s",
141 per_tok * 1e3,
142 total_bytes * iters as f64 / el / 1e9,
143 1.0 / per_tok
144 );
145 }
146}329fn verify_component_layer(
330 root: &Path,
331 projected_root: &Path,
332 layer: usize,
333 reference_hashes: &[Vec<Vec<Vec<usize>>>],
334) -> Result<(), Box<dyn Error>> {
335 let dir = root.join(format!("raw-layer-{layer}"));
336 let layer_meta: Value =
337 serde_json::from_slice(&fs::read(root.join(format!("layer-{layer}.json")))?)?;
338 let original_rows: Vec<usize> = serde_json::from_value(layer_meta["original_row_ids"].clone())?;
339 let row_to_compact: HashMap<usize, usize> = original_rows
340 .iter()
341 .copied()
342 .enumerate()
343 .map(|(compact, original)| (original, compact))
344 .collect();
345 let prefix = format!("model.layers.{layer}.engram");
346 let cmf = dir.join("engram-component.cmf");
347 let model = Arc::new(CmfModel::open(&cmf)?);
348 let embed = RawFp8Rows::from_model(
349 &model,
350 &format!("{prefix}.embed.weight"),
351 &format!("{prefix}.embed.scale"),
352 )?;
353 let wkv = QTensor::from_model(&model, &format!("{prefix}.wkv.weight"))?;
354 let q_weight = f32_tensor(&model, &format!("{prefix}.q_weight"))?;
355 let k_weight = f32_tensor(&model, &format!("{prefix}.k_weight"))?;
356 let engram = Dsv41Engram {
357 embed,
358 wkv,
359 q_weight,
360 k_weight,
361 };
362 let cases = layer_meta["cases"]
363 .as_array()
364 .ok_or("layer cases missing")?;
365 for (case_no, case) in cases.iter().enumerate() {
366 let original_hashes: Vec<Vec<usize>> =
367 serde_json::from_value(case["original_hash_ids"].clone())?;
368 if original_hashes.len() != reference_hashes[case_no].len() {
369 return Err(format!("layer {layer} case {case_no} sequence length mismatch").into());
370 }
371 let mut expected_compact = Vec::with_capacity(original_hashes.len() * HASH_COLS);
372 for (pos, rows) in original_hashes.iter().enumerate() {
373 if rows.len() != HASH_COLS || reference_hashes[case_no][pos][0].len() != HASH_COLS {
374 return Err(format!("layer {layer} case {case_no} hash width mismatch").into());
375 }
376 let layer_index = if layer == 1 { 0 } else { 1 };
377 if rows != &reference_hashes[case_no][pos][layer_index] {
378 return Err(format!(
379 "layer {layer} case {case_no} original indices differ from full hash oracle at position {pos}"
380 )
381 .into());
382 }
383 for &row in rows {
384 expected_compact.push(
385 *row_to_compact
386 .get(&row)
387 .ok_or_else(|| format!("layer {layer} missing compact row for {row}"))?,
388 );
389 }
390 }
391 let got_compact = read_i64(&dir.join(format!("case{case_no}_indices.bin")))?;
392 if got_compact != expected_compact {
393 let first = got_compact
394 .iter()
395 .zip(&expected_compact)
396 .enumerate()
397 .find(|(_, (a, b))| a != b);
398 return Err(format!(
399 "layer {layer} case {case_no} compact indices mismatch: {first:?}"
400 )
401 .into());
402 }
403
404 let seq = original_hashes.len();
405 let mut gathered = vec![0.0f32; seq * HASH_COLS * 256];
406 for token in 0..seq {
407 for col in 0..HASH_COLS {
408 engram.embed.row_into(
409 got_compact[token * HASH_COLS + col],
410 &mut gathered
411 [(token * HASH_COLS + col) * 256..(token * HASH_COLS + col + 1) * 256],
412 );
413 }
414 }
415 gathered.iter_mut().for_each(|v| *v = bf16_roundtrip(*v));
416 compare_bf16(
417 &format!("layer={layer} case={case_no} gathered"),
418 &gathered,
419 &read_bf16(&dir.join(format!("case{case_no}_gathered.bin")))?,
420 )?;
421
422 let mut projected = Vec::with_capacity(seq * (DIM * (HC_MULT + 1)));
423 for token in 0..seq {
424 let input = &gathered[token * HASH_COLS * 256..(token + 1) * HASH_COLS * 256];
425 let mut row = vec![0.0f32; DIM * (HC_MULT + 1)];
426 engram.wkv.matvec(input, &mut row, None);
427 row.iter_mut().for_each(|v| *v = bf16_roundtrip(*v));
428 projected.extend_from_slice(&row);
429 }
430 compare_bf16(
431 &format!("layer={layer} case={case_no} projected"),
432 &projected,
433 &read_bf16(&projected_root.join(format!("layer-{layer}/case{case_no}_projected.bin")))?,
434 )?;
435
436 let input = read_bf16(&dir.join(format!("case{case_no}_input.bin")))?;
437 let expected_output = read_bf16(&dir.join(format!("case{case_no}_output.bin")))?;
438 let mask: Option<Vec<bool>> = if case["token_mask"].is_null() {
439 None
440 } else {
441 Some(serde_json::from_value(case["token_mask"].clone())?)
442 };
443 let mut output = Vec::with_capacity(input.len());
444 for token in 0..seq {
445 let mut h = input[token * HC_MULT * DIM..(token + 1) * HC_MULT * DIM].to_vec();
446 dsv41_apply_engram_for_test(
447 &engram,
448 &mut h,
449 &got_compact[token * HASH_COLS..(token + 1) * HASH_COLS],
450 &cfg(),
451 mask.as_ref().map(|m| m[token]).unwrap_or(true),
452 );
453 output.extend_from_slice(&h);
454 }
455 compare_bf16(
456 &format!("layer={layer} case={case_no} output"),
457 &output,
458 &expected_output,
459 )?;
460 println!(
461 "component layer={} case={} name={} compact_indices=true gathered=true projected=true output=true",
462 layer,
463 case_no,
464 case["name"].as_str().unwrap_or("?")
465 );
466 }
467 Ok(())
468}Sourcepub fn model_dtype(&self) -> Option<TensorDtype>
pub fn model_dtype(&self) -> Option<TensorDtype>
The layout this tensor is stored in, when it is mapped from a model. The frames branch on it — a q2tp gate against a q4tp down is a real combination in the 2-bit profile and needs a different kernel.
Sourcepub fn model_idx(&self) -> Option<usize>
pub fn model_idx(&self) -> Option<usize>
The tensor’s index in the model directory, when it is mapped from one. The GPU frames bind by index rather than by name — a name lookup per layer per token is not free, and the index is what the device cache is keyed on anyway.
Sourcepub fn model_arc(&self) -> Option<Arc<CmfModel>>
pub fn model_arc(&self) -> Option<Arc<CmfModel>>
The model this tensor is mapped from, when it is mapped at all. The GPU frames need the container to reach the bytes; a QTensor already holds it, and threading a second handle down every call site to say the same thing invites the two to disagree.
pub fn rows(&self) -> usize
Sourcepub fn mapped_q4tp(&self) -> Option<(&Arc<CmfModel>, usize)>
pub fn mapped_q4tp(&self) -> Option<(&Arc<CmfModel>, usize)>
Same slot as mapped_q4t for a q4tp tensor — the fused DiT FFN picks
its kernels by which of the two answers.
Sourcepub fn mapped_device_gemm(&self) -> Option<(&Arc<CmfModel>, usize)>
pub fn mapped_device_gemm(&self) -> Option<(&Arc<CmfModel>, usize)>
(model, tensor idx) for a mapped weight in ANY codec the fused device paths can run — four-bit tiled or either int8 layout.
The fused DiT chains asked for mapped_q4tp by name, so an eight-bit
container never reached them and rendered through per-op GEMMs even
after those kernels learned its codec. The gate is what the codec has
a device GEMM for, not which codec it is.
Sourcepub fn mapped_q2tp(&self) -> Option<(&Arc<CmfModel>, usize)>
pub fn mapped_q2tp(&self) -> Option<(&Arc<CmfModel>, usize)>
(model, tensor idx) for a q2tp mapped weight — the 2-bit twin of
mapped_q4tp, used by the mixed MoE profile.
pub fn cols(&self) -> usize
Sourcepub fn mapped_q1(&self) -> Option<(&Arc<CmfModel>, usize)>
pub fn mapped_q1(&self) -> Option<(&Arc<CmfModel>, usize)>
(model, tensor idx) for a q1 mapped weight — the wgpu token graph keys its resident VRAM cache by idx. None for any other dtype/kind.
Sourcepub fn graph_weight(&self) -> Option<(&Arc<CmfModel>, usize, u8, &[f32])>
pub fn graph_weight(&self) -> Option<(&Arc<CmfModel>, usize, u8, &[f32])>
(model, idx, kind, row_scale) for a graph-capable mapped weight.
kind: 0=q8_row (per-row scales), 1=q1, 2=q4_block, 3=q1t
(tile-embedded, no rs), 5=q4_tiled, 6=q4tp, 7=q8_2f (both scale
planes live inside the tensor). None only for vbit.
The old comment here claimed q4_block was unhandled while the arm right below mapped it, and it named q8_2f as unhandled after that stopped being true — a stale comment on this function is how a model silently loses the graph, so it is worth keeping honest.
Sourcepub fn as_f32(&self) -> Option<&[f32]>
pub fn as_f32(&self) -> Option<&[f32]>
Dense f32 view — only for owned tensors. Masked/sparse execution paths require it; quantized weights don’t support masks yet.
Sourcepub fn row_f32(&self, r: usize, dst: &mut [f32])
pub fn row_f32(&self, r: usize, dst: &mut [f32])
Dequantize one row into dst (embedding lookup).
Sourcepub fn sparse_col_ok(&self) -> bool
pub fn sparse_col_ok(&self) -> bool
Can this tensor’s columns be read cheaply (for sparse down_proj)? True for F32/Q8Row/Q8_2f (per-row scale, direct strided access); false for group-packed q4/vbit (column access would unpack whole groups — sparse execution falls back to f32 for those).
Sourcepub fn add_col_scaled(&self, c: usize, w: f32, out: &mut [f32])
pub fn add_col_scaled(&self, c: usize, w: f32, out: &mut [f32])
down_proj [hidden, inter]: accumulate w · col(c) into out
[hidden] — reads ONLY column c (one neuron) from the mmap,
no full-matrix dequant. out[k] += w · down[k, c].
Sourcepub fn prefetch_row(&self, r: usize)
pub fn prefetch_row(&self, r: usize)
Touch the head of row r so the DRAM latency of the next
neuron’s weights overlaps the current one’s arithmetic.
Scattered rows are what per-token sparsity reads, and a 2 KB stride is past what the hardware prefetcher follows: without this every row starts with a cold miss that nothing hides. One touch per 512 bytes is enough — the rest of the row is a sequential run the prefetcher does pick up.
Sourcepub fn add_row_scaled(
&self,
r: usize,
w: f32,
out: &mut [f32],
scratch: &mut [f32],
)
pub fn add_row_scaled( &self, r: usize, w: f32, out: &mut [f32], scratch: &mut [f32], )
out += w · row(r) — the transposed twin of add_col_scaled.
A neuron’s down weights are a COLUMN of [hidden, inter], and a
column is strided: reading one costs a cache line per element, so
per-neuron dynamic sparsity saves arithmetic and no bytes. Stored
transposed (down_proj.t.weight, [inter, hidden]) the same
weights are a contiguous ROW, and this accumulate reads exactly
the neurons the token asked for.
Sourcepub fn row_dot(&self, r: usize, x: &[f32], scratch: &mut [f32]) -> f32
pub fn row_dot(&self, r: usize, x: &[f32], scratch: &mut [f32]) -> f32
Dot of row r with x (gate/up active-neuron path). Reads only
row r from the mmap — no full dequant. q4/vbit dequant the row
into scratch first (rare for active-FFN weights).
Sourcepub fn matvec(&self, x: &[f32], out: &mut [f32], pool: Option<&Pool>)
pub fn matvec(&self, x: &[f32], out: &mut [f32], pool: Option<&Pool>)
out = W · x (row-major). F32 delegates to the historical
bit-exact path; Mapped runs the fused int8 kernel.
Examples found in repository?
21fn main() {
22 let mut args = std::env::args().skip(1);
23 let path = args
24 .next()
25 .expect("usage: matvec_bw <model.cmf> [sweep|one <tensor>]");
26 let mode = args.next().unwrap_or_else(|| "sweep".to_string());
27 let model = Arc::new(cortiq_core::CmfModel::open(&path).expect("open model"));
28
29 // Real LM activations carry a few heavy channels (>8·rms); measured
30 // mean on this model is ~3.7. NOUT models that distribution.
31 let nout: usize = std::env::var("NOUT")
32 .ok()
33 .and_then(|v| v.parse().ok())
34 .unwrap_or(4);
35 let mk_x = |cols: usize| -> Vec<f32> {
36 let mut x: Vec<f32> = (0..cols).map(|i| ((i % 17) as f32 - 8.0) / 8.0).collect();
37 for k in 0..nout {
38 x[k * 37 % cols] = 40.0;
39 }
40 x
41 };
42
43 if mode == "one" {
44 let name = args
45 .next()
46 .unwrap_or_else(|| "model.embed_tokens.weight".to_string());
47 let entry = model.tensor(&name).expect("tensor not found");
48 let (rows, cols) = (entry.shape[0], entry.shape[1]);
49 let nbytes = entry.nbytes as f64;
50 println!(
51 "tensor {name}: {rows}x{cols} {:?} = {:.1} MB, NOUT={nout}",
52 entry.dtype,
53 nbytes / 1e6
54 );
55 let t = QTensor::from_model(&model, &name).expect("wrap");
56 let x = mk_x(cols);
57 let mut out = vec![0f32; rows];
58 t.matvec(&x, &mut out, None);
59 for nt in [1usize, 2, 4, 6, 8, 10] {
60 let pool = if nt == 1 { None } else { Some(Pool::new(nt)) };
61 let iters = 8;
62 t.matvec(&x, &mut out, pool.as_ref());
63 let t0 = Instant::now();
64 for _ in 0..iters {
65 t.matvec(&x, &mut out, pool.as_ref());
66 }
67 let el = t0.elapsed().as_secs_f64();
68 println!(
69 "threads={nt:2} {:6.2} ms/matvec {:6.1} GB/s (sink {:.3})",
70 el / iters as f64 * 1e3,
71 nbytes * iters as f64 / el / 1e9,
72 out[0]
73 );
74 }
75 return;
76 }
77
78 // sweep: every 2-D q8_2f tensor once = one decode's worth of weights.
79 let names: Vec<String> = model
80 .tensors
81 .iter()
82 .filter(|t| t.dtype == TensorDtype::Q8_2f && t.shape.len() == 2)
83 .map(|t| t.name.clone())
84 .collect();
85 let total_bytes: f64 = model
86 .tensors
87 .iter()
88 .filter(|t| t.dtype == TensorDtype::Q8_2f && t.shape.len() == 2)
89 .map(|t| t.nbytes as f64)
90 .sum();
91 println!(
92 "sweep: {} q8_2f tensors, {:.2} GB total (= weights streamed per decode token), NOUT={nout}",
93 names.len(),
94 total_bytes / 1e9
95 );
96
97 let tensors: Vec<(QTensor, Vec<f32>, Vec<f32>)> = names
98 .iter()
99 .map(|n| {
100 let e = model.tensor(n).unwrap();
101 let (rows, cols) = (e.shape[0], e.shape[1]);
102 (
103 QTensor::from_model(&model, n).expect("wrap"),
104 mk_x(cols),
105 vec![0f32; rows],
106 )
107 })
108 .collect();
109 let mut tensors = tensors;
110
111 // Whole-model residency pass BEFORE any timing: the first touch of a
112 // 4.2 GB mmap faults ~260k pages, which would otherwise be charged to
113 // whichever thread count happens to run first. REVERSE=1 flips the
114 // order as a check that no first-touch cost is left in the table.
115 for _ in 0..2 {
116 for (t, x, out) in tensors.iter_mut() {
117 t.matvec(x, out, None);
118 }
119 }
120 let mut counts = vec![1usize, 2, 4, 6, 8, 10];
121 if std::env::var("REVERSE").is_ok() {
122 counts.reverse();
123 }
124 for nt in counts {
125 let pool = if nt == 1 { None } else { Some(Pool::new(nt)) };
126 // one warm pass, then two measured
127 for (t, x, out) in tensors.iter_mut() {
128 t.matvec(x, out, pool.as_ref());
129 }
130 let iters = 2;
131 let t0 = Instant::now();
132 for _ in 0..iters {
133 for (t, x, out) in tensors.iter_mut() {
134 t.matvec(x, out, pool.as_ref());
135 }
136 }
137 let el = t0.elapsed().as_secs_f64();
138 let per_tok = el / iters as f64;
139 println!(
140 "threads={nt:2} {:7.1} ms/sweep {:6.1} GB/s -> weight-path-only ceiling {:5.1} tok/s",
141 per_tok * 1e3,
142 total_bytes * iters as f64 / el / 1e9,
143 1.0 / per_tok
144 );
145 }
146}More examples
329fn verify_component_layer(
330 root: &Path,
331 projected_root: &Path,
332 layer: usize,
333 reference_hashes: &[Vec<Vec<Vec<usize>>>],
334) -> Result<(), Box<dyn Error>> {
335 let dir = root.join(format!("raw-layer-{layer}"));
336 let layer_meta: Value =
337 serde_json::from_slice(&fs::read(root.join(format!("layer-{layer}.json")))?)?;
338 let original_rows: Vec<usize> = serde_json::from_value(layer_meta["original_row_ids"].clone())?;
339 let row_to_compact: HashMap<usize, usize> = original_rows
340 .iter()
341 .copied()
342 .enumerate()
343 .map(|(compact, original)| (original, compact))
344 .collect();
345 let prefix = format!("model.layers.{layer}.engram");
346 let cmf = dir.join("engram-component.cmf");
347 let model = Arc::new(CmfModel::open(&cmf)?);
348 let embed = RawFp8Rows::from_model(
349 &model,
350 &format!("{prefix}.embed.weight"),
351 &format!("{prefix}.embed.scale"),
352 )?;
353 let wkv = QTensor::from_model(&model, &format!("{prefix}.wkv.weight"))?;
354 let q_weight = f32_tensor(&model, &format!("{prefix}.q_weight"))?;
355 let k_weight = f32_tensor(&model, &format!("{prefix}.k_weight"))?;
356 let engram = Dsv41Engram {
357 embed,
358 wkv,
359 q_weight,
360 k_weight,
361 };
362 let cases = layer_meta["cases"]
363 .as_array()
364 .ok_or("layer cases missing")?;
365 for (case_no, case) in cases.iter().enumerate() {
366 let original_hashes: Vec<Vec<usize>> =
367 serde_json::from_value(case["original_hash_ids"].clone())?;
368 if original_hashes.len() != reference_hashes[case_no].len() {
369 return Err(format!("layer {layer} case {case_no} sequence length mismatch").into());
370 }
371 let mut expected_compact = Vec::with_capacity(original_hashes.len() * HASH_COLS);
372 for (pos, rows) in original_hashes.iter().enumerate() {
373 if rows.len() != HASH_COLS || reference_hashes[case_no][pos][0].len() != HASH_COLS {
374 return Err(format!("layer {layer} case {case_no} hash width mismatch").into());
375 }
376 let layer_index = if layer == 1 { 0 } else { 1 };
377 if rows != &reference_hashes[case_no][pos][layer_index] {
378 return Err(format!(
379 "layer {layer} case {case_no} original indices differ from full hash oracle at position {pos}"
380 )
381 .into());
382 }
383 for &row in rows {
384 expected_compact.push(
385 *row_to_compact
386 .get(&row)
387 .ok_or_else(|| format!("layer {layer} missing compact row for {row}"))?,
388 );
389 }
390 }
391 let got_compact = read_i64(&dir.join(format!("case{case_no}_indices.bin")))?;
392 if got_compact != expected_compact {
393 let first = got_compact
394 .iter()
395 .zip(&expected_compact)
396 .enumerate()
397 .find(|(_, (a, b))| a != b);
398 return Err(format!(
399 "layer {layer} case {case_no} compact indices mismatch: {first:?}"
400 )
401 .into());
402 }
403
404 let seq = original_hashes.len();
405 let mut gathered = vec![0.0f32; seq * HASH_COLS * 256];
406 for token in 0..seq {
407 for col in 0..HASH_COLS {
408 engram.embed.row_into(
409 got_compact[token * HASH_COLS + col],
410 &mut gathered
411 [(token * HASH_COLS + col) * 256..(token * HASH_COLS + col + 1) * 256],
412 );
413 }
414 }
415 gathered.iter_mut().for_each(|v| *v = bf16_roundtrip(*v));
416 compare_bf16(
417 &format!("layer={layer} case={case_no} gathered"),
418 &gathered,
419 &read_bf16(&dir.join(format!("case{case_no}_gathered.bin")))?,
420 )?;
421
422 let mut projected = Vec::with_capacity(seq * (DIM * (HC_MULT + 1)));
423 for token in 0..seq {
424 let input = &gathered[token * HASH_COLS * 256..(token + 1) * HASH_COLS * 256];
425 let mut row = vec![0.0f32; DIM * (HC_MULT + 1)];
426 engram.wkv.matvec(input, &mut row, None);
427 row.iter_mut().for_each(|v| *v = bf16_roundtrip(*v));
428 projected.extend_from_slice(&row);
429 }
430 compare_bf16(
431 &format!("layer={layer} case={case_no} projected"),
432 &projected,
433 &read_bf16(&projected_root.join(format!("layer-{layer}/case{case_no}_projected.bin")))?,
434 )?;
435
436 let input = read_bf16(&dir.join(format!("case{case_no}_input.bin")))?;
437 let expected_output = read_bf16(&dir.join(format!("case{case_no}_output.bin")))?;
438 let mask: Option<Vec<bool>> = if case["token_mask"].is_null() {
439 None
440 } else {
441 Some(serde_json::from_value(case["token_mask"].clone())?)
442 };
443 let mut output = Vec::with_capacity(input.len());
444 for token in 0..seq {
445 let mut h = input[token * HC_MULT * DIM..(token + 1) * HC_MULT * DIM].to_vec();
446 dsv41_apply_engram_for_test(
447 &engram,
448 &mut h,
449 &got_compact[token * HASH_COLS..(token + 1) * HASH_COLS],
450 &cfg(),
451 mask.as_ref().map(|m| m[token]).unwrap_or(true),
452 );
453 output.extend_from_slice(&h);
454 }
455 compare_bf16(
456 &format!("layer={layer} case={case_no} output"),
457 &output,
458 &expected_output,
459 )?;
460 println!(
461 "component layer={} case={} name={} compact_indices=true gathered=true projected=true output=true",
462 layer,
463 case_no,
464 case["name"].as_str().unwrap_or("?")
465 );
466 }
467 Ok(())
468}Source§impl QTensor
impl QTensor
Sourcepub fn q4tp_mapped(&self) -> Option<(&Arc<CmfModel>, usize)>
pub fn q4tp_mapped(&self) -> Option<(&Arc<CmfModel>, usize)>
Batched matvec (prefill-GEMM): xs — row-major [b, cols],
out — row-major [b, rows]. Element-wise semantics are IDENTICAL
to b matvec calls (same dot kernels in the same order); the win —
the weight row streams from DRAM once per batch, not b times.
(model, index) when this is a memory-mapped q4tp tensor — the
identity a device-resident chain needs to hand tp_matmat the
weight without going through this struct’s own dispatch.
pub fn matmat( &self, xs_all: &[f32], b: usize, out: &mut [f32], pool: Option<&Pool>, )
Source§impl QTensor
impl QTensor
Sourcepub fn device_matmat(&self, xs: &[f32], b: usize, out: &mut [f32]) -> bool
pub fn device_matmat(&self, xs: &[f32], b: usize, out: &mut [f32]) -> bool
The device GEMM this tensor would take, run once on the caller’s data — the startup parity probe’s arm, and the one place that knows which entry point each codec has.
It exists because the probe used to look for a q4tp weight by
name AND dtype, and a container packed any other way was declared
“host path” for the whole render even though its codec had a device
GEMM of its own. A gate that only recognizes one codec is a gate
that silently downgrades every other one.
Sourcepub fn matvec_many<const N: usize>(
ts: [&QTensor; N],
x: &[f32],
outs: [&mut [f32]; N],
pool: Option<&Pool>,
)
pub fn matvec_many<const N: usize>( ts: [&QTensor; N], x: &[f32], outs: [&mut [f32]; N], pool: Option<&Pool>, )
Multi-matrix job (roadmap §3 P0): N tensors sharing one input
run under a SINGLE pool dispatch — QKV or gate+up cost one
barrier instead of N. Per-row math is the exact same kernel as
matvec (bit-identical outputs); only the dispatch is fused.
Falls back to N sequential matvecs when the set is not a uniform
q8-family/F32 group or there is no pool.
Source§impl QTensor
impl QTensor
Sourcepub fn matvec2_many<const N: usize>(
ts: [&QTensor; N],
x1: &[f32],
x2: &[f32],
o1s: [&mut [f32]; N],
o2s: [&mut [f32]; N],
pool: Option<&Pool>,
)
pub fn matvec2_many<const N: usize>( ts: [&QTensor; N], x1: &[f32], x2: &[f32], o1s: [&mut [f32]; N], o2s: [&mut [f32]; N], pool: Option<&Pool>, )
Pair-input multi-matrix job: N tensors × 2 shared inputs under a
single pool dispatch — the MTP/pair decode path publishes one job
for Q/K/V (and one for gate+up) instead of one per tensor.
Per-row math is exactly matvec2’s kernels; bit-identical.
Sourcepub fn matvec_silu_mul(
gate: &QTensor,
up: &QTensor,
x: &[f32],
out: &mut [f32],
pool: Option<&Pool>,
) -> bool
pub fn matvec_silu_mul( gate: &QTensor, up: &QTensor, x: &[f32], out: &mut [f32], pool: Option<&Pool>, ) -> bool
Fused gate+up matvec with SiLU·mul: for each row r, computes
silu(gate·x) * (up·x) and writes to out[r]. ONE pool dispatch,
no intermediate g/u buffers, no separate silu pass. Falls back
(returns false) for unsupported dtype combos.
Sourcepub fn moe_gate_up_many(
pairs: &[(&QTensor, &QTensor)],
x: &[f32],
outs: &mut [Vec<f32>],
pool: Option<&Pool>,
) -> bool
pub fn moe_gate_up_many( pairs: &[(&QTensor, &QTensor)], x: &[f32], outs: &mut [Vec<f32>], pool: Option<&Pool>, ) -> bool
Every routed expert’s fused gate/up/SiLU under ONE pool dispatch.
The per-expert path pays a pool barrier per expert per stage: at 9
experts over 40 layers that is ~720 barriers a token, and a decode
profile of Qwen3.6-35B-A3B showed the pool parked in
psynch_cvwait about twice as long as it spent computing. Laying
every expert’s rows end-to-end in one virtual row space collapses
the stage to a single dispatch. The per-row body is the
single-expert q4tp arm verbatim, so outputs are bit-identical.
false = something is outside the fused q4tp kernel (dtype, shape,
or the CMF_SDOT=0 exact contract); the caller walks the ordinary
per-expert path.
Sourcepub fn moe_down_many(
downs: &[&QTensor],
gs: &[Vec<f32>],
weights: &[f32],
out: &mut [f32],
pool: Option<&Pool>,
) -> bool
pub fn moe_down_many( downs: &[&QTensor], gs: &[Vec<f32>], weights: &[f32], out: &mut [f32], pool: Option<&Pool>, ) -> bool
Every routed expert’s down projection, weighted and summed into
out, under ONE pool dispatch.
Partitioned by OUTPUT row rather than by expert: each row is owned
by a single worker, so the experts are summed in the caller’s order
— the same sequence of f32 adds the serial out[i] += w·eo[i] loop
performs, hence bit-identical. Partitioning by expert instead would
race on the shared accumulator.