Skip to main content

QuantizedMatrixQ4

Struct QuantizedMatrixQ4 

Source
pub struct QuantizedMatrixQ4 {
    pub data: Vec<u8>,
    pub scales: Vec<f32>,
    pub n: usize,
    pub k: usize,
}
Expand description

A weight matrix quantized to symmetric int4, packed two values per byte.

Layout mirrors crate::int8::QuantizedMatrix: [n, k] row-major in the checkpoint’s own nn.Linear orientation, one f32 scale per output channel, so no transpose is ever materialized.

Fields§

§data: Vec<u8>

n * k.div_ceil(2) bytes: row-major, two biased nibbles per byte, low nibble first.

§scales: Vec<f32>

One scale per output channel.

§n: usize§k: usize

Implementations§

Source§

impl QuantizedMatrixQ4

Source

pub fn quantize(weight: &[f32], n: usize, k: usize) -> Self

Quantizes an [n, k] f32 weight matrix.

§Panics

If weight.len() != n * k, or a weight is non-finite — a NaN reaching the quantizer means the graph upstream is already corrupt, and refusing loudly beats baking it into an artifact.

Examples found in repository?
examples/int4_speed_gate.rs (line 84)
45fn main() {
46    // (label, n, k) for one microdecoder layer, at decode geometry m = 1.
47    const HIDDEN: usize = 1024;
48    const INTERMEDIATE: usize = 3072;
49    const Q_WIDTH: usize = 16 * 128;
50    const KV_WIDTH: usize = 8 * 128;
51    let projections: &[(&str, usize, usize)] = &[
52        ("q_proj", Q_WIDTH, HIDDEN),
53        ("k_proj", KV_WIDTH, HIDDEN),
54        ("v_proj", KV_WIDTH, HIDDEN),
55        ("o_proj", HIDDEN, Q_WIDTH),
56        ("gate_up", INTERMEDIATE * 2, HIDDEN),
57        ("down_proj", HIDDEN, INTERMEDIATE),
58    ];
59
60    // Fifteen depths x five layers is how often this body is walked per frame.
61    const ROUNDS: usize = 15 * 5;
62
63    let dispatched = Int8Tier::dispatch();
64    println!("dispatched int8 route: {}", dispatched.as_str());
65    println!("shapes: microdecoder layer at m=1, {ROUNDS} rounds (15 depths x 5 layers)\n");
66    // The ratio columns are SPEED ratios (q4's throughput relative to the q8 variant):
67    // t_q8 / t_q4, so 1.0 means parity and 0.05 means q4 runs at 5% of q8's speed.
68    println!(
69        "{:<10} {:>6} {:>6} {:>10} {:>10} {:>10} {:>9} {:>9}",
70        "proj", "n", "k", "q8-scalar", "q4-scalar", "q8-route", "spd/q8scl", "spd/route"
71    );
72
73    let mut total_q8_scalar = 0.0_f64;
74    let mut total_q4 = 0.0_f64;
75    let mut total_q8_route = 0.0_f64;
76
77    for &(label, n, k) in projections {
78        let weight = deterministic(n * k, 0x51ED_0000 + n as u64);
79        let activation = deterministic(k, 0xA0C7_0000 + k as u64);
80        let mut x_q = vec![0_i8; k];
81        let scale = quantize_row_q8(&activation, &mut x_q);
82
83        let q8 = QuantizedMatrix::quantize(&weight, n, k);
84        let q4 = QuantizedMatrixQ4::quantize(&weight, n, k);
85        let mut out = vec![0.0_f32; n];
86
87        // Warm the caches for whichever side runs first, so the ordering does not decide it.
88        linear_q8(&x_q, &[scale], &q8, None, 1, &mut out, Int8Tier::Scalar);
89        linear_q4(&x_q, &[scale], &q4, None, 1, &mut out);
90
91        // INTERLEAVED repeats, reporting the minimum of each variant.
92        //
93        // A first attempt timed the three variants in sequential blocks and produced 98.60 ms for
94        // k_proj against 5.14 ms for v_proj — identical 1024x1024 shapes, 19x apart. That is
95        // scheduler and thermal noise, and it is precisely what doctrine #8's same-thermal-window
96        // rule exists to prevent. Interleaving puts all three variants in the same window on every
97        // repeat, and the minimum is the least noise-contaminated estimator of a deterministic
98        // kernel: noise only ever adds time.
99        const REPEATS: usize = 7;
100        let mut q8_scalar = f64::MAX;
101        let mut q4_ms = f64::MAX;
102        let mut q8_route = f64::MAX;
103        for _ in 0..REPEATS {
104            let started = Instant::now();
105            for _ in 0..ROUNDS {
106                linear_q8(&x_q, &[scale], &q8, None, 1, &mut out, Int8Tier::Scalar);
107            }
108            q8_scalar = q8_scalar.min(started.elapsed().as_secs_f64() * 1000.0);
109
110            let started = Instant::now();
111            for _ in 0..ROUNDS {
112                linear_q4(&x_q, &[scale], &q4, None, 1, &mut out);
113            }
114            q4_ms = q4_ms.min(started.elapsed().as_secs_f64() * 1000.0);
115
116            let started = Instant::now();
117            for _ in 0..ROUNDS {
118                linear_q8(&x_q, &[scale], &q8, None, 1, &mut out, dispatched);
119            }
120            q8_route = q8_route.min(started.elapsed().as_secs_f64() * 1000.0);
121        }
122
123        total_q8_scalar += q8_scalar;
124        total_q4 += q4_ms;
125        total_q8_route += q8_route;
126
127        println!(
128            "{label:<10} {n:>6} {k:>6} {q8_scalar:>9.2}m {q4_ms:>9.2}m {q8_route:>9.2}m {:>8.2}x {:>8.2}x",
129            q8_scalar / q4_ms,
130            q8_route / q4_ms
131        );
132    }
133
134    println!(
135        "\ntotals (one layer, {ROUNDS} rounds): q8-scalar {total_q8_scalar:.1} ms, \
136         q4-scalar {total_q4:.1} ms, q8-route {total_q8_route:.1} ms"
137    );
138    println!(
139        "int4 vs scalar int8 : {:.2}x  ({})",
140        total_q8_scalar / total_q4,
141        if total_q4 < total_q8_scalar {
142            "FASTER - the halved-bytes thesis holds at equal implementation quality"
143        } else {
144            "SLOWER - unpack cost exceeds the traffic saving even against scalar"
145        }
146    );
147    println!(
148        "int4 vs shipping int8: {:.2}x  ({})",
149        total_q8_route / total_q4,
150        if total_q4 < total_q8_route {
151            "FASTER - gate (a) PASSES; the listening gate is now worth running"
152        } else {
153            "SLOWER - gate (a) FAILS as built; int4 needs an in-register SIMD unpack to compete"
154        }
155    );
156
157    // Bytes are the thesis; report them so the ratio can be read against the traffic it saves.
158    let q8_bytes: usize = projections.iter().map(|(_, n, k)| n * k).sum();
159    println!(
160        "\nweight bytes per layer: q8 {:.1} MB, q4 {:.1} MB",
161        q8_bytes as f64 / 1e6,
162        q8_bytes as f64 / 2e6
163    );
164}
Source

pub fn dequantize_row(&self, row: usize) -> Vec<f32>

Dequantizes one output channel back to f32, for parity comparison against the f32 weights.

Source

pub fn packed_bytes(&self) -> usize

Bytes of weight storage, the number this lever exists to shrink.

Trait Implementations§

Source§

impl Clone for QuantizedMatrixQ4

Source§

fn clone(&self) -> QuantizedMatrixQ4

Returns a duplicate of the value. Read more
1.0.0 (const: unstable) · Source§

fn clone_from(&mut self, source: &Self)

Performs copy-assignment from source. Read more
Source§

impl Debug for QuantizedMatrixQ4

Source§

fn fmt(&self, f: &mut Formatter<'_>) -> Result

Formats the value using the given formatter. Read more
Source§

impl PartialEq for QuantizedMatrixQ4

Source§

fn eq(&self, other: &QuantizedMatrixQ4) -> bool

Equality operator ==. Read more
1.0.0 (const: unstable) · Source§

fn ne(&self, other: &Rhs) -> bool

Inequality operator !=. Read more
Source§

impl StructuralPartialEq for QuantizedMatrixQ4

Auto Trait Implementations§

Blanket Implementations§

Source§

impl<T> Any for T
where T: 'static + ?Sized,

Source§

fn type_id(&self) -> TypeId

Gets the TypeId of self. Read more
Source§

impl<T> Borrow<T> for T
where T: ?Sized,

Source§

fn borrow(&self) -> &T

Immutably borrows from an owned value. Read more
Source§

impl<T> BorrowMut<T> for T
where T: ?Sized,

Source§

fn borrow_mut(&mut self) -> &mut T

Mutably borrows from an owned value. Read more
Source§

impl<T> CloneToUninit for T
where T: Clone,

Source§

unsafe fn clone_to_uninit(&self, dest: *mut u8)

🔬This is a nightly-only experimental API. (clone_to_uninit)
Performs copy-assignment from self to dest. Read more
Source§

impl<T> From<T> for T

Source§

fn from(t: T) -> T

Returns the argument unchanged.

Source§

impl<T, U> Into<U> for T
where U: From<T>,

Source§

fn into(self) -> U

Calls U::from(self).

That is, this conversion is whatever the implementation of From<T> for U chooses to do.

Source§

impl<T> ToOwned for T
where T: Clone,

Source§

type Owned = T

The resulting type after obtaining ownership.
Source§

fn to_owned(&self) -> T

Creates owned data from borrowed data, usually by cloning. Read more
Source§

fn clone_into(&self, target: &mut T)

Uses borrowed data to replace owned data, usually by cloning. Read more
Source§

impl<T, U> TryFrom<U> for T
where U: Into<T>,

Source§

type Error = Infallible

The type returned in the event of a conversion error.
Source§

fn try_from(value: U) -> Result<T, <T as TryFrom<U>>::Error>

Performs the conversion.
Source§

impl<T, U> TryInto<U> for T
where U: TryFrom<T>,

Source§

type Error = <U as TryFrom<T>>::Error

The type returned in the event of a conversion error.
Source§

fn try_into(self) -> Result<U, <U as TryFrom<T>>::Error>

Performs the conversion.