1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
impl GgufReader {
/// Extract a tensor as F32 data (dequantizing if needed)
///
/// Postcondition: data.len() == shape.iter().product()
#[ensures(ret.as_ref().map_or(true, |(data, shape)| data.len() == shape.iter().product::<usize>()))]
pub fn get_tensor_f32(&self, name: &str) -> Result<(Vec<f32>, Vec<usize>)> {
let meta = self
.tensors
.iter()
.find(|t| t.name == name)
.ok_or_else(|| AprenderError::FormatError {
message: format!("Tensor '{name}' not found in GGUF"),
})?;
let shape: Vec<usize> = meta.dims.iter().map(|&d| d as usize).collect();
// BUG-GGUF-002 FIX: Use checked multiplication to prevent integer overflow
let num_elements = shape
.iter()
.try_fold(1usize, |acc, &dim| acc.checked_mul(dim))
.ok_or_else(|| AprenderError::FormatError {
message: format!(
"Tensor '{}' shape {:?} causes integer overflow (malicious file?)",
name, shape
),
})?;
// BUG-GGUF-002 FIX: Validate total elements against reasonable limit
if num_elements > MAX_TENSOR_ELEMENTS {
return Err(AprenderError::FormatError {
message: format!(
"Tensor '{}' has {} elements, exceeds max {} (possible malicious file)",
name, num_elements, MAX_TENSOR_ELEMENTS
),
});
}
let tensor_start = self.data_offset + meta.offset as usize;
let data = match meta.dtype {
0 => {
// F32 - direct copy
// BUG-GGUF-002 FIX: Use checked_mul for byte size calculation
let byte_size =
num_elements
.checked_mul(4)
.ok_or_else(|| AprenderError::FormatError {
message: format!("Tensor '{}' byte size calculation overflow", name),
})?;
if tensor_start + byte_size > self.data.len() {
return Err(AprenderError::FormatError {
message: format!("Tensor '{name}' data exceeds file size"),
});
}
let bytes = &self.data[tensor_start..tensor_start + byte_size];
bytes
.as_chunks::<4>()
.0
.iter()
.map(|c| f32::from_le_bytes([c[0], c[1], c[2], c[3]]))
.collect()
}
1 => {
// F16 - convert to F32
// BUG-GGUF-002 FIX: Use checked_mul for byte size calculation
let byte_size =
num_elements
.checked_mul(2)
.ok_or_else(|| AprenderError::FormatError {
message: format!("Tensor '{}' byte size calculation overflow", name),
})?;
if tensor_start + byte_size > self.data.len() {
return Err(AprenderError::FormatError {
message: format!("Tensor '{name}' data exceeds file size"),
});
}
let bytes = &self.data[tensor_start..tensor_start + byte_size];
bytes
.as_chunks::<2>()
.0
.iter()
.map(|c| f16_to_f32(u16::from_le_bytes([c[0], c[1]])))
.collect()
}
// GGML dtype values (from ggml.h):
// 0=F32, 1=F16, 2=Q4_0, 3=Q4_1, 6=Q5_0, 7=Q5_1, 8=Q8_0, 9=Q8_1
// 10=Q2_K, 11=Q3_K, 12=Q4_K, 13=Q5_K, 14=Q6_K
// 16+=IQ variants
2 => {
// Q4_0 - dequantize
super::dequantize_q4_0(&self.data, tensor_start, num_elements)?
}
3 => {
// Q4_1 - dequantize (blocks of 32 with scale and min)
super::dequantize_q4_1(&self.data, tensor_start, num_elements)?
}
6 => {
// Q5_0 - dequantize (blocks of 32 with 5-bit quants)
super::dequantize_q5_0(&self.data, tensor_start, num_elements)?
}
7 => {
// Q5_1 - dequantize (blocks of 32 with 5-bit quants + min)
dequantize_q5_1(&self.data, tensor_start, num_elements)?
}
8 => {
// Q8_0 - dequantize
super::dequantize_q8_0(&self.data, tensor_start, num_elements)?
}
10 => {
// Q2_K - dequantize (super blocks of 256)
dequantize_q2_k(&self.data, tensor_start, num_elements)?
}
11 => {
// Q3_K - dequantize (super blocks of 256)
dequantize_q3_k(&self.data, tensor_start, num_elements)?
}
12 => {
// Q4_K - dequantize (super blocks of 256 elements, 144 bytes/block)
dequantize_q4_k(&self.data, tensor_start, num_elements)?
}
13 => {
// Q5_K - dequantize (super blocks of 256 elements, 176 bytes/block)
dequantize_q5_k(&self.data, tensor_start, num_elements)?
}
14 => {
// Q6_K - dequantize (super blocks of 256 elements, 210 bytes/block)
dequantize_q6_k(&self.data, tensor_start, num_elements)?
}
// #3947: IQ4_NL (20), IQ3_S (21) and IQ4_XS (23) have real decoders in
// trueno_quant, bit-exact against gguf-py on every IQ tensor of the three
// #3947 release models. Every other IQ type still reaches the refusal below.
20 | 21 | 23 => {
trueno_quant::dequantize_iq_to_f32(
meta.dtype,
self.data.get(tensor_start..).unwrap_or(&[]),
num_elements,
)
.map_err(|e| AprenderError::FormatError {
message: format!("GGUF tensor '{name}': {e}"),
})?
}
// #3656: IQ types (16..=23) used to go to `dequantize_iq_approximate`, which
// mapped each raw byte to `(b - 128) * 0.01` and returned Ok — `apr convert`
// wrote those invented weights (std 36.6x the real tensor) and exited 0. With
// no real dequantizer here, the only honest answer is a refusal.
_ => {
let type_name = trueno_quant::GgmlType::from_id(meta.dtype)
.map_or("a type ggml does not define", trueno_quant::GgmlType::as_str);
return Err(AprenderError::FormatError {
message: format!(
"GGUF tensor '{name}' is {type_name} (ggml type {}): aprender-core has \
no dequantizer for it, so it cannot be read as F32. Refusing rather \
than approximating, which would invent its weights",
meta.dtype
),
});
}
};
Ok((data, shape))
}
/// Get all tensors as F32
pub fn get_all_tensors_f32(&self) -> Result<TensorDataMap> {
self.get_all_tensors_f32_with_progress(|_, _, _| {})
}
/// Get all tensors as F32 with per-tensor progress callback.
///
/// Contract: GH-692 — progress feedback for large GGUF dequantization.
/// Callback receives (current_index, total_count, tensor_name).
pub fn get_all_tensors_f32_with_progress(
&self,
progress: impl Fn(usize, usize, &str),
) -> Result<TensorDataMap> {
// #3947 quorum (PR #3958): IQ4_NL / IQ3_S / IQ4_XS decode per tensor so
// inspection (`apr qa` tensor_contract, validate) can read them, but the ticket
// asked for inspection only. Every whole-model F32 load (`apr import`'s GH-375
// fallback, `apr convert`) goes through here, so it keeps refusing them, as it
// did under #3656. Checked before any decoding so nothing partial is produced.
if let Some(meta) = self.tensors.iter().find(|t| matches!(t.dtype, 20 | 21 | 23)) {
let type_name = trueno_quant::GgmlType::from_id(meta.dtype)
.map_or("an IQ type", trueno_quant::GgmlType::as_str);
return Err(AprenderError::FormatError {
message: format!(
"GGUF tensor '{}' is {type_name} (ggml type {}): aprender-core decodes it \
for inspection only and does not convert IQ models to F32 (#3947)",
meta.name, meta.dtype
),
});
}
let total = self.tensors.len();
let mut result = BTreeMap::new();
for (i, meta) in self.tensors.iter().enumerate() {
progress(i + 1, total, &meta.name);
let (data, shape) = self.get_tensor_f32(&meta.name)?;
result.insert(meta.name.clone(), (data, shape));
}
Ok(result)
}
/// Get raw tensor bytes without dequantization (preserves Q4K/Q6K)
///
/// Returns (raw_bytes, shape, ggml_dtype) where dtype is the GGML type id. Every
/// type live in upstream ggml is sized (`trueno_quant::TRAITS`); an id upstream
/// removed, or one newer than that table, is refused by name.
pub fn get_tensor_raw(&self, name: &str) -> Result<(Vec<u8>, Vec<usize>, u32)> {
let meta = self
.tensors
.iter()
.find(|t| t.name == name)
.ok_or_else(|| AprenderError::FormatError {
message: format!("Tensor '{name}' not found in GGUF"),
})?;
let shape: Vec<usize> = meta.dims.iter().map(|&d| d as usize).collect();
// BUG-GGUF-002 FIX: Use checked multiplication to prevent integer overflow
let num_elements = shape
.iter()
.try_fold(1usize, |acc, &dim| acc.checked_mul(dim))
.ok_or_else(|| AprenderError::FormatError {
message: format!(
"Tensor '{}' shape {:?} causes integer overflow (malicious file?)",
name, shape
),
})?;
// BUG-GGUF-002 FIX: Validate total elements against reasonable limit
if num_elements > MAX_TENSOR_ELEMENTS {
return Err(AprenderError::FormatError {
message: format!(
"Tensor '{}' has {} elements, exceeds max {} (possible malicious file)",
name, num_elements, MAX_TENSOR_ELEMENTS
),
});
}
let tensor_start = self.data_offset + meta.offset as usize;
// #3601: size from ggml's own `type_traits` (`trueno_quant::TRAITS`, extracted
// from upstream and fixture-checked by PMAT-3430) rather than a hand-typed match.
// That match knew 15 of ggml's 35 live ids, so every IQ*/TQ* tensor was refused
// as "Unsupported dtype 23 for raw extraction" — which read as a corrupt file.
// Sizing a type is not a claim that anything here can dequantize it.
let ggml_type = trueno_quant::GgmlType::try_from_id(meta.dtype)
.map_err(|e| unsized_ggml_type_error(name, &e))?;
// Whole blocks, rounding DOWN: the arithmetic the hand-typed arms used, kept so
// no id that was already sized changes its byte count.
// BUG-GGUF-002 FIX: Use checked arithmetic to prevent overflow in byte size calc
let byte_size = (num_elements / ggml_type.block_size())
.checked_mul(ggml_type.block_bytes())
.ok_or_else(|| AprenderError::FormatError {
message: format!(
"Tensor '{}' byte size calculation overflow (dtype: {})",
name, meta.dtype
),
})?;
if tensor_start + byte_size > self.data.len() {
return Err(AprenderError::FormatError {
message: format!("Tensor '{name}' data exceeds file size"),
});
}
let bytes = self.data[tensor_start..tensor_start + byte_size].to_vec();
Ok((bytes, shape, meta.dtype))
}
/// Get all tensors as raw bytes (preserves quantization)
///
/// Returns BTreeMap of name -> (raw_bytes, shape, ggml_dtype)
pub fn get_all_tensors_raw(&self) -> Result<BTreeMap<String, (Vec<u8>, Vec<usize>, u32)>> {
let mut result = BTreeMap::new();
for meta in &self.tensors {
let (data, shape, dtype) = self.get_tensor_raw(&meta.name)?;
result.insert(meta.name.clone(), (data, shape, dtype));
}
Ok(result)
}
}
/// #3601: the refusal for a tensor whose ggml type id has no size in `trueno_quant::TRAITS`.
///
/// Every id live upstream is sized, so this is reached only by an id upstream REMOVED or one
/// newer than the pinned table. The message says which — "Unsupported dtype 23" sent readers
/// looking for file corruption when the gap was in apr.
fn unsized_ggml_type_error(tensor: &str, e: &trueno_quant::GgmlTypeError) -> AprenderError {
let why = match e {
trueno_quant::GgmlTypeError::Removed { .. } => {
"no current ggml defines a layout for it, so the file must be re-quantized from \
its source weights"
}
trueno_quant::GgmlTypeError::Unknown { .. } => {
"it is newer than the ggml type table this apr was built with (pinned by \
scripts/llama_pin.toml) — a gap in apr, not a corrupt file; please report it at \
https://github.com/paiml/aprender/issues"
}
};
AprenderError::FormatError {
message: format!(
"GGUF tensor '{tensor}': {e}; {why}. The ids apr sizes are listed in \
contracts/ggml-type-v1.yaml"
),
}
}