aprender-serve 0.70.2

Pure Rust ML inference engine built from scratch - model serving for GGUF and safetensors
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
//! Generic Parallel Matrix-Vector Multiplication (Contract: quantized-dot-product-v1.yaml)
//!
//! Replaces ~300 lines of duplicated parallel matvec code (Q4_K, Q5_K, Q6_K variants)
//! with a single generic implementation parameterized by `QuantBlockFormat`.
//!
//! ## Design
//!
//! The generic matvec delegates to format-specific SIMD dot products for the inner
//! loop (those remain hand-optimized). The outer loop — validation, padding,
//! parallel dispatch, tiling — is format-independent and encoded here once.
//!
//! ## Tuning Constants (from existing code)
//!
//! - `PARALLEL_THRESHOLD = 256` — below this, use sequential path (PAR-126)
//! - `MIDI_TILE_M = 64` — TCB-style midi-tile for L2 cache reuse

use super::format_trait::QuantBlockFormat;
use crate::error::{RealizarError, Result};
use std::borrow::Cow;

/// Parallel threshold: use sequential path below this out_dim (PAR-126)
const PARALLEL_THRESHOLD: usize = 256;

/// TCB-style midi-tile size for parallel chunking (L2 cache reuse)
const MIDI_TILE_M: usize = 64;

/// Pad activations to super-block boundary when `in_dim % elements_per_superblock != 0`.
///
/// GH-202 FIX: Quantized weights are stored with per-row padding to super-block
/// boundaries. The fused dot kernels expect activations to match the padded length.
#[inline]
fn pad_activations_generic(activations: &[f32], padded_len: usize) -> Cow<'_, [f32]> {
    if activations.len() == padded_len {
        Cow::Borrowed(activations)
    } else {
        let mut padded = vec![0.0f32; padded_len];
        padded[..activations.len()].copy_from_slice(activations);
        Cow::Owned(padded)
    }
}

/// Type alias for format-specific SIMD dot product function
///
/// The generic matvec calls this per-row. Each format provides its own
/// optimized SIMD implementation (or scalar fallback).
pub type FusedDotFn = fn(&[u8], &[f32]) -> Result<f32>;

/// Generic parallel fused matrix-vector multiply for any blocked quantization format.
///
/// Writes results into a pre-allocated output buffer (zero-allocation hot path).
///
/// # Arguments
///
/// * `weight_data` - Raw quantized weight data, row-major [out_dim × bytes_per_row]
/// * `activations` - Input activations [in_dim]
/// * `in_dim` - Input dimension
/// * `out_dim` - Output dimension
/// * `output` - Pre-allocated output buffer [out_dim]
/// * `dot_fn` - Format-specific SIMD dot product function
///
/// # Errors
///
/// Returns error if:
/// - Weight data is too small for the given dimensions
/// - Activation length doesn't match input dimension
/// - Output buffer length is less than out_dim
#[allow(clippy::similar_names)]
pub fn generic_parallel_matvec_into<F: QuantBlockFormat>(
    weight_data: &[u8],
    activations: &[f32],
    in_dim: usize,
    out_dim: usize,
    output: &mut [f32],
    dot_fn: FusedDotFn,
) -> Result<()> {
    let super_blocks_per_row = in_dim.div_ceil(F::ELEMENTS_PER_SUPERBLOCK);
    let bytes_per_row = super_blocks_per_row * F::SUPERBLOCK_BYTES;

    // Validate weight data size
    let expected_weight_bytes = out_dim * bytes_per_row;
    if weight_data.len() < expected_weight_bytes {
        return Err(RealizarError::InvalidShape {
            reason: format!(
                "{} weight data too small: need {} bytes for {}x{}, have {}",
                F::FORMAT_ID,
                expected_weight_bytes,
                out_dim,
                in_dim,
                weight_data.len()
            ),
        });
    }

    // Validate activation length
    if activations.len() != in_dim {
        return Err(RealizarError::InvalidShape {
            reason: format!(
                "Activation length {} doesn't match in_dim {}",
                activations.len(),
                in_dim
            ),
        });
    }

    // Validate output buffer
    if output.len() < out_dim {
        return Err(RealizarError::InvalidShape {
            reason: format!(
                "Output buffer too small: need {}, have {}",
                out_dim,
                output.len()
            ),
        });
    }

    // GH-202 FIX: Pad activations to super-block boundary
    let padded_in_dim = super_blocks_per_row * F::ELEMENTS_PER_SUPERBLOCK;
    let acts = pad_activations_generic(activations, padded_in_dim);

    if out_dim < PARALLEL_THRESHOLD {
        // Sequential path: avoids rayon overhead for small matrices
        for o in 0..out_dim {
            let row_start = o * bytes_per_row;
            let row_end = row_start + bytes_per_row;
            let row_data = &weight_data[row_start..row_end];
            output[o] = dot_fn(row_data, &acts).unwrap_or(0.0);
        }
    } else {
        // Parallel path with TCB-style midi-tile chunking
        use rayon::prelude::*;

        output[..out_dim]
            .par_chunks_mut(MIDI_TILE_M)
            .enumerate()
            .for_each(|(midi_idx, midi_chunk)| {
                let midi_start = midi_idx * MIDI_TILE_M;

                for (local_idx, out) in midi_chunk.iter_mut().enumerate() {
                    let row = midi_start + local_idx;
                    let row_start = row * bytes_per_row;
                    let row_end = row_start + bytes_per_row;
                    let row_data = &weight_data[row_start..row_end];
                    *out = dot_fn(row_data, &acts).unwrap_or(0.0);
                }
            });
    }

    Ok(())
}

/// Generic parallel fused matrix-vector multiply (allocating variant).
///
/// Convenience wrapper that allocates the output buffer.
///
/// # Errors
///
/// Same as `generic_parallel_matvec_into`.
#[allow(clippy::similar_names)]
pub fn generic_parallel_matvec<F: QuantBlockFormat>(
    weight_data: &[u8],
    activations: &[f32],
    in_dim: usize,
    out_dim: usize,
    dot_fn: FusedDotFn,
) -> Result<Vec<f32>> {
    let mut output = vec![0.0f32; out_dim];
    generic_parallel_matvec_into::<F>(
        weight_data,
        activations,
        in_dim,
        out_dim,
        &mut output,
        dot_fn,
    )?;
    Ok(output)
}

/// Multi-row variant of [`generic_parallel_matvec_into`] (#4228): `input` is `m` token rows of
/// `in_dim`, `output` is `m` rows of `out_dim`, token-major.
///
/// Rayon splits the weight into row tiles sized to stay L2-resident; each tile runs every token
/// against its rows, so the weight streams from DRAM once per call instead of once per token.
/// Every output is the same `dot_fn(row, padded_token)` the single-row function computes, so
/// the result is bit-identical to calling it once per token.
///
/// # Errors
/// Mis-sized weight, input or output buffers.
pub fn generic_multirow_matmul_into<F: QuantBlockFormat>(
    weight_data: &[u8],
    input: &[f32],
    m: usize,
    in_dim: usize,
    out_dim: usize,
    output: &mut [f32],
    dot_fn: FusedDotFn,
) -> Result<()> {
    use rayon::prelude::*;

    let super_blocks_per_row = in_dim.div_ceil(F::ELEMENTS_PER_SUPERBLOCK);
    let bytes_per_row = super_blocks_per_row * F::SUPERBLOCK_BYTES;
    if weight_data.len() < out_dim * bytes_per_row
        || input.len() != m * in_dim
        || output.len() < m * out_dim
    {
        return Err(RealizarError::InvalidShape {
            reason: format!(
                "{} multirow: weight {} < {out_dim}x{bytes_per_row}, input {} != {m}x{in_dim} \
                 or output {} < {m}x{out_dim}",
                F::FORMAT_ID,
                weight_data.len(),
                input.len(),
                output.len()
            ),
        });
    }
    if m == 0 || out_dim == 0 {
        return Ok(());
    }
    let padded = super_blocks_per_row * F::ELEMENTS_PER_SUPERBLOCK;
    let acts: Vec<Cow<'_, [f32]>> = input
        .chunks_exact(in_dim)
        .map(|row| pad_activations_generic(row, padded))
        .collect();

    // L2-sized row tile, as the Q4_K multi-row kernel: 256 KiB of weight, 4..=64 rows.
    let tile = ((256 * 1024) / bytes_per_row.max(1)).clamp(4, 64) / 4 * 4;
    // Row-major [out_dim][m] so each tile owns a contiguous chunk; transposed below.
    let mut by_row = vec![0.0f32; out_dim * m];
    by_row
        .par_chunks_mut(tile * m)
        .enumerate()
        .for_each(|(ti, chunk)| {
            let row0 = ti * tile;
            let rows = chunk.len() / m;
            for (t, act) in acts.iter().enumerate() {
                for r in 0..rows {
                    let at = (row0 + r) * bytes_per_row;
                    chunk[r * m + t] =
                        dot_fn(&weight_data[at..at + bytes_per_row], act).unwrap_or(0.0);
                }
            }
        });
    for (row, vals) in by_row.chunks_exact(m).enumerate() {
        for (t, &v) in vals.iter().enumerate() {
            output[t * out_dim + row] = v;
        }
    }
    Ok(())
}

#[cfg(test)]
mod tests {
    use super::*;
    use crate::quantize::format_trait::{Q4K, Q6K};
    use crate::quantize::generic_dot::generic_fused_dot_scalar;

    /// Helper: create Q4_K weight data (zeros with valid structure)
    fn create_q4k_test_weights(out_dim: usize, in_dim: usize) -> Vec<u8> {
        let super_blocks_per_row = in_dim.div_ceil(256);
        let bytes_per_row = super_blocks_per_row * 144;
        vec![0u8; out_dim * bytes_per_row]
    }

    /// Scalar Q4K dot wrapper for testing
    fn q4k_scalar_dot(data: &[u8], acts: &[f32]) -> Result<f32> {
        generic_fused_dot_scalar::<Q4K>(data, acts)
    }

    /// Scalar Q6K dot wrapper for testing
    fn q6k_scalar_dot(data: &[u8], acts: &[f32]) -> Result<f32> {
        generic_fused_dot_scalar::<Q6K>(data, acts)
    }

    #[test]
    fn test_generic_matvec_q4k_basic() {
        let in_dim = 256;
        let out_dim = 64;
        let weights = create_q4k_test_weights(out_dim, in_dim);
        let acts = vec![1.0f32; in_dim];
        let mut output = vec![0.0f32; out_dim];

        let result = generic_parallel_matvec_into::<Q4K>(
            &weights,
            &acts,
            in_dim,
            out_dim,
            &mut output,
            q4k_scalar_dot,
        );
        assert!(result.is_ok());
        // All zero weights → all zero outputs
        assert!(output.iter().all(|&v| v == 0.0));
    }

    #[test]
    fn test_generic_matvec_q6k_basic() {
        let in_dim: usize = 256;
        let out_dim: usize = 32;
        let super_blocks_per_row = in_dim.div_ceil(256);
        let bytes_per_row = super_blocks_per_row * 210;
        let weights = vec![0u8; out_dim * bytes_per_row];
        let acts = vec![1.0f32; in_dim];
        let mut output = vec![0.0f32; out_dim];

        let result = generic_parallel_matvec_into::<Q6K>(
            &weights,
            &acts,
            in_dim,
            out_dim,
            &mut output,
            q6k_scalar_dot,
        );
        assert!(result.is_ok());
    }

    #[test]
    fn test_generic_matvec_weight_too_small() {
        let weights = vec![0u8; 100]; // Too small
        let acts = vec![1.0f32; 256];
        let mut output = vec![0.0f32; 64];

        let result = generic_parallel_matvec_into::<Q4K>(
            &weights,
            &acts,
            256,
            64,
            &mut output,
            q4k_scalar_dot,
        );
        assert!(result.is_err());
    }

    #[test]
    fn test_generic_matvec_activation_mismatch() {
        let weights = create_q4k_test_weights(64, 256);
        let acts = vec![1.0f32; 128]; // Wrong length
        let mut output = vec![0.0f32; 64];

        let result = generic_parallel_matvec_into::<Q4K>(
            &weights,
            &acts,
            256,
            64,
            &mut output,
            q4k_scalar_dot,
        );
        assert!(result.is_err());
    }

    #[test]
    fn test_generic_matvec_output_too_small() {
        let weights = create_q4k_test_weights(64, 256);
        let acts = vec![1.0f32; 256];
        let mut output = vec![0.0f32; 32]; // Too small

        let result = generic_parallel_matvec_into::<Q4K>(
            &weights,
            &acts,
            256,
            64,
            &mut output,
            q4k_scalar_dot,
        );
        assert!(result.is_err());
    }

    #[test]
    fn test_generic_matvec_allocating_variant() {
        let weights = create_q4k_test_weights(64, 256);
        let acts = vec![1.0f32; 256];

        let result = generic_parallel_matvec::<Q4K>(&weights, &acts, 256, 64, q4k_scalar_dot);
        assert!(result.is_ok());
        assert_eq!(result.expect("should succeed").len(), 64);
    }

    #[test]
    fn test_generic_matvec_parallel_threshold() {
        // Above threshold (512 > 256) — takes parallel path
        let in_dim = 256;
        let out_dim = 512;
        let weights = create_q4k_test_weights(out_dim, in_dim);
        let acts = vec![1.0f32; in_dim];
        let mut output = vec![0.0f32; out_dim];

        let result = generic_parallel_matvec_into::<Q4K>(
            &weights,
            &acts,
            in_dim,
            out_dim,
            &mut output,
            q4k_scalar_dot,
        );
        assert!(result.is_ok());
    }

    #[test]
    fn test_generic_matvec_padding() {
        // in_dim not a multiple of 256 — requires padding (GH-202)
        let in_dim: usize = 200;
        let out_dim: usize = 16;
        let super_blocks_per_row = in_dim.div_ceil(256);
        let bytes_per_row = super_blocks_per_row * 144;
        let weights = vec![0u8; out_dim * bytes_per_row];
        let acts = vec![1.0f32; in_dim];
        let mut output = vec![0.0f32; out_dim];

        let result = generic_parallel_matvec_into::<Q4K>(
            &weights,
            &acts,
            in_dim,
            out_dim,
            &mut output,
            q4k_scalar_dot,
        );
        assert!(result.is_ok());
    }

    /// Each of the three size checks in `generic_multirow_matmul_into` rejects on its own.
    #[test]
    fn test_multirow_rejects_each_mis_sized_buffer_alone() {
        let (m, in_dim, out_dim) = (2usize, 256usize, 3usize);
        let bytes_per_row = 144;
        let need = out_dim * bytes_per_row;
        let acts = vec![1.0f32; m * in_dim];
        let run = |weight: usize, input: usize, output: usize| {
            let weights = vec![0u8; weight];
            let acts = vec![1.0f32; input];
            let mut out = vec![0.0f32; output];
            generic_multirow_matmul_into::<Q4K>(
                &weights,
                &acts,
                m,
                in_dim,
                out_dim,
                &mut out,
                q4k_scalar_dot,
            )
        };
        assert!(
            run(need, acts.len(), m * out_dim).is_ok(),
            "exact sizes fit"
        );
        assert!(
            run(need - 1, acts.len(), m * out_dim).is_err(),
            "weight one byte short"
        );
        assert!(
            run(need + 1, acts.len(), m * out_dim).is_ok(),
            "a longer weight is fine"
        );
        assert!(
            run(need, acts.len() - 1, m * out_dim).is_err(),
            "input one short"
        );
        assert!(
            run(need, acts.len() + 1, m * out_dim).is_err(),
            "input one long"
        );
        assert!(
            run(need, acts.len(), m * out_dim - 1).is_err(),
            "output one short"
        );
        assert!(
            run(need, acts.len(), m * out_dim + 1).is_ok(),
            "a longer output is fine"
        );
    }

    /// No tokens or no rows is an empty product: `Ok`, and the output is never written. Either
    /// alone must return early — a zero `m` would otherwise ask rayon for zero-sized chunks.
    #[test]
    fn test_multirow_zero_tokens_or_zero_rows_is_an_untouched_ok() {
        let in_dim = 256usize;
        for (m, out_dim) in [(0usize, 3usize), (2, 0), (0, 0)] {
            let weights = vec![0u8; out_dim * 144];
            let acts = vec![1.0f32; m * in_dim];
            let mut out = vec![7.0f32; 4];
            let r = generic_multirow_matmul_into::<Q4K>(
                &weights,
                &acts,
                m,
                in_dim,
                out_dim,
                &mut out,
                q4k_scalar_dot,
            );
            assert!(r.is_ok(), "m={m} out_dim={out_dim}");
            assert_eq!(
                out,
                vec![7.0f32; 4],
                "m={m} out_dim={out_dim} wrote the output"
            );
        }
    }

    /// Weight base and the weight row each `dot_fn` call reads, in call order.
    static TILE_TRACE: std::sync::Mutex<(usize, usize, Vec<usize>)> =
        std::sync::Mutex::new((0, 1, Vec::new()));

    fn tracing_dot(data: &[u8], _acts: &[f32]) -> Result<f32> {
        let mut g = TILE_TRACE.lock().expect("trace lock");
        let row = (data.as_ptr() as usize - g.0) / g.1;
        g.2.push(row);
        Ok(0.0)
    }

    /// The row tile is `clamp(256 KiB / bytes_per_row, 4, 64) / 4 * 4` rows: the tile is
    /// invisible in the output, so read it off the order of `dot_fn` calls on one thread
    /// (a tile runs every token over its rows before the next tile starts).
    #[test]
    fn test_multirow_row_tile_is_l2_sized_and_a_multiple_of_four() {
        let (m, in_dim, out_dim) = (2usize, 256 * 70, 96usize);
        let bytes_per_row = 70 * 144; // 10080: 262144 / 10080 = 26 -> 26 / 4 * 4 = 24
        let weights = vec![0u8; out_dim * bytes_per_row];
        let input = vec![0.0f32; m * in_dim];
        let mut output = vec![1.0f32; m * out_dim];
        let pool = rayon::ThreadPoolBuilder::new()
            .num_threads(1)
            .build()
            .expect("one-thread pool");
        let trace = {
            let mut g = TILE_TRACE.lock().expect("trace lock");
            *g = (weights.as_ptr() as usize, bytes_per_row, Vec::new());
            drop(g);
            pool.install(|| {
                generic_multirow_matmul_into::<Q4K>(
                    &weights,
                    &input,
                    m,
                    in_dim,
                    out_dim,
                    &mut output,
                    tracing_dot,
                )
                .expect("multirow");
            });
            let t = TILE_TRACE.lock().expect("trace lock").2.clone();
            t
        };
        assert_eq!(trace.len(), m * out_dim);
        let first_run = trace[1..]
            .iter()
            .position(|&r| r <= trace[0])
            .map(|p| p + 1)
            .expect("the first tile's second token restarts at its first row");
        assert_eq!(first_run, 24, "row tile size");
    }
}