structio 0.2.2

High performance JSON and BEVE for Rust structs. No dependencies, no proc-macros, no intermediate representation.
Documentation
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
//! Float parsing: Eisel-Lemire, with an exact fast path and a fallback.
//!
//! Three tiers, in the order they are attempted:
//!
//! 1. **Exact.** A mantissa under 2^53 with a small decimal exponent is one
//!    hardware multiply or divide away from the correct answer, because both
//!    operands and the result are exactly representable.
//! 2. **Eisel-Lemire.** A 64x128 multiply against a table of truncated powers
//!    of five decides the rounding for essentially every real input. See
//!    <https://arxiv.org/abs/2101.11408>.
//! 3. **Fallback.** For the handful of inputs where 128 bits cannot resolve a
//!    tie, defer to the standard library, which does full big-integer
//!    arithmetic. This is correct by construction and effectively never runs.

use super::table::{LARGEST_POWER_OF_FIVE, POWER_OF_FIVE_128, SMALLEST_POWER_OF_FIVE};
use crate::error::{ErrorCode, PResult};
use crate::num::atoi::is_digit;

/// The parts of a decimal literal, before any rounding decision.
struct Decimal {
    /// Up to 19 significant digits, as an integer.
    mantissa: u64,
    /// Power of ten to apply to `mantissa`.
    exp10: i64,
    negative: bool,
    /// Set when significant digits were dropped, which makes `mantissa` a
    /// lower bound rather than the exact value.
    truncated: bool,
}

/// Per-width constants for [`compute_float`].
pub(crate) trait RawFloat: Copy {
    const MANTISSA_EXPLICIT_BITS: i32;
    const MINIMUM_EXPONENT: i32;
    const INFINITE_POWER: i32;
    const SMALLEST_POWER_OF_TEN: i32;
    const LARGEST_POWER_OF_TEN: i32;
    const MIN_EXPONENT_ROUND_TO_EVEN: i32;
    const MAX_EXPONENT_ROUND_TO_EVEN: i32;
    /// Largest power of ten that is exactly representable.
    const MAX_EXACT_POW10: i32;
    /// Largest integer mantissa that is exactly representable.
    const MAX_EXACT_MANTISSA: u64;

    fn from_bits(mantissa: u64, exponent: i32) -> Self;
    fn from_u64_exact(v: u64) -> Self;
    fn pow10_exact(i: usize) -> Self;
    fn mul(self, rhs: Self) -> Self;
    fn div(self, rhs: Self) -> Self;
    fn neg(self) -> Self;
    fn parse_fallback(s: &str) -> Self;
}

const F64_POW10: [f64; 23] = [
    1e0, 1e1, 1e2, 1e3, 1e4, 1e5, 1e6, 1e7, 1e8, 1e9, 1e10, 1e11, 1e12, 1e13, 1e14, 1e15, 1e16,
    1e17, 1e18, 1e19, 1e20, 1e21, 1e22,
];

const F32_POW10: [f32; 11] = [1e0, 1e1, 1e2, 1e3, 1e4, 1e5, 1e6, 1e7, 1e8, 1e9, 1e10];

impl RawFloat for f64 {
    const MANTISSA_EXPLICIT_BITS: i32 = 52;
    const MINIMUM_EXPONENT: i32 = -1023;
    const INFINITE_POWER: i32 = 0x7FF;
    const SMALLEST_POWER_OF_TEN: i32 = -342;
    const LARGEST_POWER_OF_TEN: i32 = 308;
    const MIN_EXPONENT_ROUND_TO_EVEN: i32 = -4;
    const MAX_EXPONENT_ROUND_TO_EVEN: i32 = 23;
    const MAX_EXACT_POW10: i32 = 22;
    const MAX_EXACT_MANTISSA: u64 = 1 << 53;

    #[inline(always)]
    fn from_bits(mantissa: u64, exponent: i32) -> Self {
        f64::from_bits(mantissa | ((exponent as u64) << 52))
    }
    #[inline(always)]
    fn from_u64_exact(v: u64) -> Self {
        v as f64
    }
    #[inline(always)]
    fn pow10_exact(i: usize) -> Self {
        F64_POW10[i]
    }
    #[inline(always)]
    fn mul(self, rhs: Self) -> Self {
        self * rhs
    }
    #[inline(always)]
    fn div(self, rhs: Self) -> Self {
        self / rhs
    }
    #[inline(always)]
    fn neg(self) -> Self {
        -self
    }
    #[inline(never)]
    fn parse_fallback(s: &str) -> Self {
        // The scanner accepts a strict subset of Rust's float grammar, so this
        // cannot fail. Panicking beats returning a silently wrong number from
        // the path that exists precisely because the other two can be wrong.
        s.parse::<f64>()
            .expect("scanner accepts only valid float syntax")
    }
}

impl RawFloat for f32 {
    const MANTISSA_EXPLICIT_BITS: i32 = 23;
    const MINIMUM_EXPONENT: i32 = -127;
    const INFINITE_POWER: i32 = 0xFF;
    const SMALLEST_POWER_OF_TEN: i32 = -65;
    const LARGEST_POWER_OF_TEN: i32 = 38;
    const MIN_EXPONENT_ROUND_TO_EVEN: i32 = -17;
    const MAX_EXPONENT_ROUND_TO_EVEN: i32 = 10;
    const MAX_EXACT_POW10: i32 = 10;
    const MAX_EXACT_MANTISSA: u64 = 1 << 24;

    #[inline(always)]
    fn from_bits(mantissa: u64, exponent: i32) -> Self {
        f32::from_bits((mantissa as u32) | ((exponent as u32) << 23))
    }
    #[inline(always)]
    fn from_u64_exact(v: u64) -> Self {
        v as f32
    }
    #[inline(always)]
    fn pow10_exact(i: usize) -> Self {
        F32_POW10[i]
    }
    #[inline(always)]
    fn mul(self, rhs: Self) -> Self {
        self * rhs
    }
    #[inline(always)]
    fn div(self, rhs: Self) -> Self {
        self / rhs
    }
    #[inline(always)]
    fn neg(self) -> Self {
        -self
    }
    #[inline(never)]
    fn parse_fallback(s: &str) -> Self {
        // The scanner accepts a strict subset of Rust's float grammar, so this
        // cannot fail. Panicking beats returning a silently wrong number from
        // the path that exists precisely because the other two can be wrong.
        s.parse::<f32>()
            .expect("scanner accepts only valid float syntax")
    }
}

/// Binary exponent of `10^q`, exact for the range the table covers.
/// `217706 / 65536` approximates `log2(10)`.
#[inline(always)]
const fn power(q: i32) -> i32 {
    ((q.wrapping_mul(217_706)) >> 16) + 63
}

#[inline(always)]
const fn full_mul(a: u64, b: u64) -> (u64, u64) {
    let r = (a as u128) * (b as u128);
    (r as u64, (r >> 64) as u64)
}

/// Top 128 bits of `w * 5^q`, or close enough to decide the rounding.
///
/// The 192-bit product is `(w * hi) << 64 + (w * lo)`; only the leading 128
/// bits matter. The second multiply is skipped unless the low bits of the
/// leading word sit right on a rounding boundary.
#[inline(always)]
fn product_approx(q: i64, w: u64, precision: i32) -> (u64, u64) {
    let mask: u64 = if precision < 64 {
        u64::MAX >> precision
    } else {
        u64::MAX
    };
    debug_assert!(q >= SMALLEST_POWER_OF_FIVE as i64 && q <= LARGEST_POWER_OF_FIVE as i64);
    let index = (q - SMALLEST_POWER_OF_FIVE as i64) as usize;
    let (blo, bhi) = POWER_OF_FIVE_128[index];

    let (mut lo, mut hi) = full_mul(w, bhi);
    if hi & mask == mask {
        let (_, second_hi) = full_mul(w, blo);
        lo = lo.wrapping_add(second_hi);
        if second_hi > lo {
            hi += 1;
        }
    }
    (lo, hi)
}

/// Biased mantissa and exponent, or `None` when 128 bits could not decide.
fn compute_float<F: RawFloat>(q: i64, mut w: u64) -> Option<(u64, i32)> {
    if w == 0 || q < F::SMALLEST_POWER_OF_TEN as i64 {
        return Some((0, 0));
    }
    if q > F::LARGEST_POWER_OF_TEN as i64 {
        return Some((0, F::INFINITE_POWER));
    }

    let lz = w.leading_zeros() as i32;
    w <<= lz;

    let (lo, hi) = product_approx(q, w, F::MANTISSA_EXPLICIT_BITS + 3);
    if lo == u64::MAX {
        // The approximation is saturated, so adding one could carry across the
        // rounding boundary. Outside this exponent window that cannot happen,
        // because the product is exact.
        let inside_safe_exponent = (-27..=55).contains(&q);
        if !inside_safe_exponent {
            return None;
        }
    }

    let upperbit = (hi >> 63) as i32;
    let shift = upperbit + 64 - F::MANTISSA_EXPLICIT_BITS - 3;
    let mut mantissa = hi >> shift;
    let mut power2 = power(q as i32) + upperbit - lz - F::MINIMUM_EXPONENT;

    if power2 <= 0 {
        if -power2 + 1 >= 64 {
            // More than 64 bits below the smallest subnormal.
            return Some((0, 0));
        }
        mantissa >>= -power2 + 1;
        mantissa += mantissa & 1;
        mantissa >>= 1;
        // Rounding up out of the subnormal range lands on the smallest normal,
        // which the exponent field encodes as 1.
        let e = (mantissa >= (1u64 << F::MANTISSA_EXPLICIT_BITS)) as i32;
        return Some((mantissa, e));
    }

    // A tie with an even mantissa rounds down, not up. This can only be reached
    // when the product was exact, which is why the exponent window is checked.
    if lo <= 1
        && q >= F::MIN_EXPONENT_ROUND_TO_EVEN as i64
        && q <= F::MAX_EXPONENT_ROUND_TO_EVEN as i64
        && mantissa & 3 == 1
        && (mantissa << shift) == hi
    {
        mantissa &= !1u64;
    }

    mantissa += mantissa & 1;
    mantissa >>= 1;
    if mantissa >= (2u64 << F::MANTISSA_EXPLICIT_BITS) {
        mantissa = 1u64 << F::MANTISSA_EXPLICIT_BITS;
        power2 += 1;
    }
    mantissa &= !(1u64 << F::MANTISSA_EXPLICIT_BITS);
    if power2 >= F::INFINITE_POWER {
        return Some((0, F::INFINITE_POWER));
    }
    Some((mantissa, power2))
}

/// Scan a JSON number literal. Leaves `*i` on the first byte after the token.
fn scan(buf: &[u8], i: &mut usize) -> PResult<Decimal> {
    let n = buf.len();
    let mut idx = *i;

    let negative = idx < n && buf[idx] == b'-';
    if negative {
        idx += 1;
    }

    let mut mantissa: u64 = 0;
    let mut digits = 0i32;
    let mut exp10: i64 = 0;
    let mut truncated = false;

    if idx >= n || !is_digit(buf[idx]) {
        return Err(ErrorCode::ExpectedNumber);
    }

    // Integer part. JSON forbids a leading zero followed by more digits.
    if buf[idx] == b'0' {
        idx += 1;
        if idx < n && is_digit(buf[idx]) {
            return Err(ErrorCode::InvalidNumber);
        }
    } else {
        while idx < n && is_digit(buf[idx]) {
            let d = buf[idx] - b'0';
            if digits < 19 {
                mantissa = mantissa * 10 + d as u64;
                digits += 1;
            } else {
                // Past 19 digits the value no longer fits; keep the magnitude
                // by growing the exponent instead.
                exp10 += 1;
                truncated |= d != 0;
            }
            idx += 1;
        }
    }

    // Fraction.
    if idx < n && buf[idx] == b'.' {
        idx += 1;
        if idx >= n || !is_digit(buf[idx]) {
            return Err(ErrorCode::InvalidNumber);
        }
        while idx < n && is_digit(buf[idx]) {
            let d = buf[idx] - b'0';
            if mantissa == 0 && d == 0 {
                // A leading zero after the point shifts the exponent without
                // contributing a significant digit.
                exp10 -= 1;
            } else if digits < 19 {
                mantissa = mantissa * 10 + d as u64;
                digits += 1;
                exp10 -= 1;
            } else {
                truncated |= d != 0;
            }
            idx += 1;
        }
    }

    // Exponent.
    if idx < n && (buf[idx] | 0x20) == b'e' {
        idx += 1;
        let mut exp_neg = false;
        if idx < n && (buf[idx] == b'+' || buf[idx] == b'-') {
            exp_neg = buf[idx] == b'-';
            idx += 1;
        }
        if idx >= n || !is_digit(buf[idx]) {
            return Err(ErrorCode::InvalidNumber);
        }
        let mut e: i64 = 0;
        while idx < n && is_digit(buf[idx]) {
            // Saturate rather than overflow; anything past this is already
            // zero or infinity.
            if e < 0x10_0000 {
                e = e * 10 + (buf[idx] - b'0') as i64;
            }
            idx += 1;
        }
        exp10 += if exp_neg { -e } else { e };
    }

    *i = idx;
    Ok(Decimal {
        mantissa,
        exp10,
        negative,
        truncated,
    })
}

/// Parse a JSON number into `F`, advancing `*i` past the token.
#[inline]
pub(crate) fn parse_float<F: RawFloat>(buf: &[u8], i: &mut usize) -> PResult<F> {
    let start = *i;
    let d = scan(buf, i)?;
    let end = *i;

    if d.mantissa == 0 {
        // Preserve the sign of zero: `-0.0` is a distinct value.
        let z = F::from_bits(0, 0);
        return Ok(if d.negative { z.neg() } else { z });
    }

    // Tier 1: both operands exactly representable, so one operation is exact.
    if !d.truncated
        && d.mantissa <= F::MAX_EXACT_MANTISSA
        && d.exp10 >= -(F::MAX_EXACT_POW10 as i64)
        && d.exp10 <= F::MAX_EXACT_POW10 as i64
    {
        let m = F::from_u64_exact(d.mantissa);
        let v = if d.exp10 < 0 {
            m.div(F::pow10_exact((-d.exp10) as usize))
        } else {
            m.mul(F::pow10_exact(d.exp10 as usize))
        };
        return Ok(if d.negative { v.neg() } else { v });
    }

    // Tier 2: Eisel-Lemire. When digits were dropped the true mantissa lies in
    // `[m, m+1]`, so both ends must round the same way for the result to be
    // certain.
    let resolved = match compute_float::<F>(d.exp10, d.mantissa) {
        Some(a) => {
            if d.truncated {
                match compute_float::<F>(d.exp10, d.mantissa + 1) {
                    Some(b) if a == b => Some(a),
                    _ => None,
                }
            } else {
                Some(a)
            }
        }
        None => None,
    };

    if let Some((mantissa, exponent)) = resolved {
        let v = F::from_bits(mantissa, exponent);
        return Ok(if d.negative { v.neg() } else { v });
    }

    // Tier 3: exact big-integer arithmetic, via the standard library. This is
    // reached for well under 0.1% of inputs even when they are chosen to be
    // hostile, so the UTF-8 check costs nothing measurable and buys the float
    // path its way out of `unsafe` entirely.
    let text = core::str::from_utf8(&buf[start..end]).map_err(|_| ErrorCode::InvalidNumber)?;
    Ok(F::parse_fallback(text))
}

/// Walk a JSON number literal without converting it, leaving `*i` on the first
/// byte after the token.
///
/// The same grammar [`parse_float`] holds its input to, because it is the same
/// walk: [`Parser::read_number_str`](crate::json::Parser::read_number_str)
/// hands back the digits for a type this crate cannot convert to, and a token
/// it accepted must be one every other reader would have accepted too.
#[inline]
pub(crate) fn scan_number(buf: &[u8], i: &mut usize) -> PResult<()> {
    scan(buf, i)?;
    Ok(())
}

/// Whether `s` is one JSON number literal and nothing else.
///
/// The check behind
/// [`Writer::write_number_str`](crate::json::Writer::write_number_str), which
/// only runs it under `debug_assertions`.
pub(crate) fn is_number(s: &str) -> bool {
    let mut i = 0;
    scan_number(s.as_bytes(), &mut i).is_ok() && i == s.len()
}