Skip to main content

deser_yaml/
resolve.rs

1//! Resolution of scalars to atoms.
2//!
3//! In YAML the type of a plain (unquoted) scalar is determined by its
4//! content: `true` is a boolean, `42` an integer and so on.  The rules for
5//! this differ between YAML versions.  Quoted and block scalars are always
6//! strings unless they have an explicit tag.
7use std::borrow::Cow;
8
9use deser_core::adapters::{Base64, BytesEncoding};
10use deser_core::ext::{Date, Datetime, ExtValue, Offset, Time};
11use deser_core::{Atom, Bytes, Implicit, ImplicitValue};
12
13/// The YAML version that determines how plain scalars are resolved.
14///
15/// A document can declare its version with a `%YAML` directive.  The
16/// version configured with
17/// [`DeserializerConfig::set_version`](crate::DeserializerConfig::set_version) applies
18/// to documents that do not declare a version.
19#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
20#[non_exhaustive]
21pub enum Version {
22    /// YAML 1.1.
23    ///
24    /// In addition to what YAML 1.2 supports, `y`, `yes`, `on` (and their
25    /// negations) are booleans, integers can be written in binary (`0b1010`),
26    /// octal (`0777`) and base 60 (`1:30`) and numbers can contain `_` as
27    /// digit separators.  Unlike YAML 1.2, `0o777` is a string.
28    V1_1,
29    /// YAML 1.2 with the core schema.
30    #[default]
31    V1_2,
32}
33
34const TAG_PREFIX: &str = "tag:yaml.org,2002:";
35
36/// How a tag affects a scalar.
37pub(crate) enum ScalarTag<'t> {
38    /// The scalar is a string (the `!` tag or a quoted scalar).
39    Str,
40    /// A standard tag that determines the type.
41    Standard(&'t str),
42    /// A tag that is not known.  The value is a string, the tag is passed on.
43    Custom,
44}
45
46/// Classifies a tag.
47pub(crate) fn classify_tag(tag: &str) -> ScalarTag<'_> {
48    if tag == "!" {
49        return ScalarTag::Str;
50    }
51    match tag.strip_prefix(TAG_PREFIX) {
52        Some(
53            name @ ("str" | "int" | "float" | "bool" | "null" | "binary" | "timestamp" | "seq"
54            | "map"),
55        ) => ScalarTag::Standard(name),
56        _ => ScalarTag::Custom,
57    }
58}
59
60/// Returns `true` if the tag is a standard tag for a collection.
61pub(crate) fn is_collection_tag(tag: &str, is_map: bool) -> Result<bool, &'static str> {
62    match classify_tag(tag) {
63        ScalarTag::Str => Ok(true),
64        ScalarTag::Standard("seq") if !is_map => Ok(true),
65        ScalarTag::Standard("map") if is_map => Ok(true),
66        ScalarTag::Standard(_) => Err("tag does not apply to this kind of node"),
67        _ => Ok(false),
68    }
69}
70
71/// Resolves a plain scalar without tag.
72#[inline]
73pub(crate) fn resolve_plain(value: Cow<'_, str>, version: Version) -> Atom<'_> {
74    match resolve_plain_str(&value, version) {
75        Some(atom) => atom,
76        None => Atom::Str(value.into()),
77    }
78}
79
80/// Resolves a plain scalar without tag into the atom that is emitted.
81///
82/// The type of plain scalars is inferred from their text.  Scalars that
83/// are not strings are emitted as [`Atom::Implicit`] so that types which
84/// expect strings receive the text (`1.10` is `"1.10"` for a `String` and
85/// `1.1` for an `f64`).  Integers which do not fit into 64 bits are emitted
86/// as their value.
87#[inline]
88pub(crate) fn resolve_plain_implicit(value: Cow<'_, str>, version: Version) -> Atom<'_> {
89    match resolve_plain_str(&value, version) {
90        Some(atom) => match ImplicitValue::from_atom(&atom) {
91            Some(resolved) => Atom::Implicit(Implicit::new(value, resolved)),
92            None => atom,
93        },
94        None => Atom::Str(value.into()),
95    }
96}
97
98/// Returns `true` if a plain scalar is a string in the given version.
99pub(crate) fn is_plain_str(s: &str, version: Version) -> bool {
100    resolve_plain_str(s, version).is_none()
101}
102
103/// Returns `true` if YAML 1.1 readers resolve a plain scalar implicitly to
104/// something that the resolution of this crate does not cover.
105///
106/// Common YAML 1.1 readers (such as PyYAML) resolve timestamps without a
107/// tag and treat `=` as the value key.  The syntax of timestamps is matched
108/// without validating the date as these readers fail on invalid dates.
109pub(crate) fn is_yaml11_implicit(s: &str) -> bool {
110    s == "=" || is_yaml11_timestamp_syntax(s)
111}
112
113/// Matches the regular expression of YAML 1.1 timestamps.
114fn is_yaml11_timestamp_syntax(s: &str) -> bool {
115    let bytes = s.as_bytes();
116    let mut pos = 0;
117    let digits = |pos: &mut usize, min: usize, max: usize| -> bool {
118        let len = bytes[*pos..]
119            .iter()
120            .take(max)
121            .take_while(|x| x.is_ascii_digit())
122            .count();
123        *pos += len;
124        len >= min
125    };
126    let expect = |pos: &mut usize, c: u8| -> bool {
127        let rv = bytes.get(*pos) == Some(&c);
128        *pos += rv as usize;
129        rv
130    };
131    if !digits(&mut pos, 4, 4) || !expect(&mut pos, b'-') {
132        return false;
133    }
134    // `YYYY-MM-DD` or `YYYY-M-D` followed by a time
135    let date_start = pos;
136    if !digits(&mut pos, 1, 2) || !expect(&mut pos, b'-') || !digits(&mut pos, 1, 2) {
137        return false;
138    }
139    if pos == bytes.len() {
140        return pos - date_start == 5;
141    }
142    match bytes[pos] {
143        b'T' | b't' => pos += 1,
144        b' ' | b'\t' => {
145            while matches!(bytes.get(pos), Some(b' ' | b'\t')) {
146                pos += 1;
147            }
148        }
149        _ => return false,
150    }
151    if !digits(&mut pos, 1, 2)
152        || !expect(&mut pos, b':')
153        || !digits(&mut pos, 2, 2)
154        || !expect(&mut pos, b':')
155        || !digits(&mut pos, 2, 2)
156    {
157        return false;
158    }
159    if expect(&mut pos, b'.') {
160        digits(&mut pos, 0, usize::MAX);
161    }
162    while matches!(bytes.get(pos), Some(b' ' | b'\t')) {
163        pos += 1;
164    }
165    match bytes.get(pos) {
166        None => true,
167        Some(b'Z') => pos + 1 == bytes.len(),
168        Some(b'+' | b'-') => {
169            pos += 1;
170            if !digits(&mut pos, 1, 2) {
171                return false;
172            }
173            if expect(&mut pos, b':') && !digits(&mut pos, 2, 2) {
174                return false;
175            }
176            pos == bytes.len()
177        }
178        _ => false,
179    }
180}
181
182#[test]
183fn test_yaml11_timestamp_syntax() {
184    for s in [
185        "2001-12-14",
186        "2001-12-14t21:59:43.10-05:00",
187        "2001-12-14 21:59:43.10 -5",
188        "2001-12-15 2:59:43.10",
189        "2002-12-14T21:59:43Z",
190        "2001-02-30",
191    ] {
192        assert!(is_yaml11_timestamp_syntax(s), "{}", s);
193    }
194    for s in [
195        "2001-12",
196        "2001-1-1",
197        "20011-12-14",
198        "2001-12-14 foo",
199        "hello",
200    ] {
201        assert!(!is_yaml11_timestamp_syntax(s), "{}", s);
202    }
203}
204
205/// Returns `true` if the text of an implicit value can be written as plain
206/// scalar.
207///
208/// This is the case if readers of all versions from `compat` on read the
209/// text as the same value.  Empty text is not written as it's only null
210/// in some places.
211pub(crate) fn writes_as_plain(value: &Implicit, compat: Version) -> bool {
212    let text = value.text().as_str();
213    let reads_as = |version| {
214        resolve_plain_str(text, version)
215            .and_then(|atom| ImplicitValue::from_atom(&atom))
216            .is_some_and(|resolved| resolved.is_same(value.value()))
217    };
218    !text.is_empty()
219        && reads_as(Version::V1_2)
220        && (compat == Version::V1_2 || reads_as(Version::V1_1))
221}
222
223fn resolve_plain_str(s: &str, version: Version) -> Option<Atom<'static>> {
224    let first = match s.as_bytes().first() {
225        Some(&first) => first,
226        None => return Some(Atom::Null),
227    };
228    match first {
229        b'0'..=b'9' | b'+' | b'-' | b'.' => match version {
230            Version::V1_2 => parse_core_number(s),
231            Version::V1_1 => parse_yaml11_int(s).or_else(|| parse_yaml11_float(s)),
232        },
233        b'~' if s.len() == 1 => Some(Atom::Null),
234        b'n' | b'N' | b't' | b'T' | b'f' | b'F' | b'y' | b'Y' | b'o' | b'O' => {
235            parse_null(s).or_else(|| parse_bool(s, version))
236        }
237        _ => None,
238    }
239}
240
241/// Resolves a scalar with a standard tag.
242pub(crate) fn resolve_standard<'x>(
243    name: &str,
244    value: Cow<'x, str>,
245    version: Version,
246) -> Result<Atom<'x>, &'static str> {
247    let s = &*value;
248    match name {
249        "str" => Ok(Atom::Str(value.into())),
250        "null" => parse_null(s).ok_or("invalid !!null value"),
251        "bool" => parse_bool(s, version).ok_or("invalid !!bool value"),
252        "int" => match version {
253            Version::V1_2 => parse_core_int(s),
254            Version::V1_1 => parse_yaml11_int(s),
255        }
256        .ok_or("invalid !!int value"),
257        "float" => match version {
258            Version::V1_2 => parse_core_float(s).or_else(|| parse_core_int(s).map(int_to_float)),
259            Version::V1_1 => {
260                parse_yaml11_float(s).or_else(|| parse_yaml11_int(s).map(int_to_float))
261            }
262        }
263        .ok_or("invalid !!float value"),
264        "binary" => decode_base64(s)
265            .map(|bytes| Atom::Bytes(Bytes::new(bytes)))
266            .ok_or("invalid !!binary value"),
267        "timestamp" => parse_timestamp(s)
268            .map(|value| Atom::Ext(ExtValue::owned(value)))
269            .ok_or("invalid !!timestamp value"),
270        _ => Err("tag does not apply to scalars"),
271    }
272}
273
274/// Parses a YAML timestamp (<https://yaml.org/type/timestamp.html>).
275///
276/// Dates without time are local dates, timestamps without time zone are in
277/// UTC.
278pub(crate) fn parse_timestamp(s: &str) -> Option<Datetime> {
279    let bytes = s.as_bytes();
280    let mut pos = 0;
281    // reads between `min` and `max` digits
282    let number = |pos: &mut usize, min: usize, max: usize| -> Option<u32> {
283        let len = bytes[*pos..]
284            .iter()
285            .take(max)
286            .take_while(|x| x.is_ascii_digit())
287            .count();
288        if len < min {
289            return None;
290        }
291        let rv = s[*pos..*pos + len].parse().ok()?;
292        *pos += len;
293        Some(rv)
294    };
295
296    let year = number(&mut pos, 4, 4)? as u16;
297    let date_only = bytes.len() == 10;
298    let expect =
299        |pos: &mut usize, c: u8| -> Option<()> { (bytes.get(*pos) == Some(&c)).then(|| *pos += 1) };
300    expect(&mut pos, b'-')?;
301    let (min, max) = if date_only { (2, 2) } else { (1, 2) };
302    let month = number(&mut pos, min, max)? as u8;
303    expect(&mut pos, b'-')?;
304    let day = number(&mut pos, min, max)? as u8;
305    let date = Date { year, month, day };
306    if !date.is_valid() {
307        return None;
308    }
309    if date_only {
310        return Some(Datetime::from(date));
311    }
312
313    match bytes.get(pos)? {
314        b'T' | b't' => pos += 1,
315        b' ' | b'\t' => {
316            while let Some(b' ' | b'\t') = bytes.get(pos) {
317                pos += 1;
318            }
319        }
320        _ => return None,
321    }
322    let hour = number(&mut pos, 1, 2)? as u8;
323    expect(&mut pos, b':')?;
324    let minute = number(&mut pos, 2, 2)? as u8;
325    expect(&mut pos, b':')?;
326    let second = number(&mut pos, 2, 2)? as u8;
327    let mut nanosecond = 0;
328    if bytes.get(pos) == Some(&b'.') {
329        pos += 1;
330        let start = pos;
331        while bytes.get(pos).is_some_and(u8::is_ascii_digit) {
332            pos += 1;
333        }
334        // digits beyond nanoseconds are truncated
335        let digits = &s[start..pos.min(start + 9)];
336        if !digits.is_empty() {
337            nanosecond = digits.parse::<u32>().ok()? * 10u32.pow(9 - digits.len() as u32);
338        }
339    }
340    let time = Time {
341        hour,
342        minute,
343        second,
344        nanosecond,
345    };
346    if !time.is_valid() {
347        return None;
348    }
349
350    while let Some(b' ' | b'\t') = bytes.get(pos) {
351        pos += 1;
352    }
353    let offset = match bytes.get(pos) {
354        // timestamps without time zone are in UTC
355        None => Offset::Z,
356        Some(b'Z') => {
357            pos += 1;
358            Offset::Z
359        }
360        Some(&sign @ (b'+' | b'-')) => {
361            pos += 1;
362            let hours = number(&mut pos, 1, 2)?;
363            let minutes = if bytes.get(pos) == Some(&b':') {
364                pos += 1;
365                number(&mut pos, 2, 2)?
366            } else {
367                0
368            };
369            if hours > 23 || minutes > 59 {
370                return None;
371            }
372            let minutes = (hours * 60 + minutes) as i16;
373            Offset::Custom {
374                minutes: if sign == b'-' { -minutes } else { minutes },
375            }
376        }
377        _ => return None,
378    };
379    if pos != bytes.len() {
380        return None;
381    }
382    Some(Datetime {
383        date: Some(date),
384        time: Some(time),
385        offset: Some(offset),
386    })
387}
388
389#[test]
390fn test_parse_timestamp() {
391    let ts = |s: &str| parse_timestamp(s).map(|x| x.to_string());
392    assert_eq!(ts("2002-12-14").as_deref(), Some("2002-12-14"));
393    assert_eq!(
394        ts("2001-12-14t21:59:43.10-05:00").as_deref(),
395        Some("2001-12-14T21:59:43.1-05:00")
396    );
397    assert_eq!(
398        ts("2001-12-14 21:59:43.10 -5").as_deref(),
399        Some("2001-12-14T21:59:43.1-05:00")
400    );
401    assert_eq!(
402        ts("2001-12-15 2:59:43.10").as_deref(),
403        Some("2001-12-15T02:59:43.1Z")
404    );
405    assert_eq!(
406        ts("2001-12-15T02:59:43.1Z").as_deref(),
407        Some("2001-12-15T02:59:43.1Z")
408    );
409    assert_eq!(
410        ts("2001-1-5 02:59:43").as_deref(),
411        Some("2001-01-05T02:59:43Z")
412    );
413    for invalid in [
414        "",
415        "2002-12-1",
416        "2002-13-14",
417        "2002-12-14 ",
418        "2002-12-14T25:00:00",
419        "2002-12-14T02:59",
420        "2002-12-14T02:59:43X",
421        "2002-12-14T02:59:43+24",
422        "02002-12-14",
423    ] {
424        assert!(parse_timestamp(invalid).is_none(), "{}", invalid);
425    }
426}
427
428fn int_to_float(atom: Atom) -> Atom<'static> {
429    Atom::F64(match atom {
430        Atom::U64(value) => value as f64,
431        Atom::I64(value) => value as f64,
432        Atom::F64(value) => value,
433        Atom::Ext(ref ext) => match (ext.downcast_ref::<u128>(), ext.downcast_ref::<i128>()) {
434            (Some(&value), _) => value as f64,
435            (_, Some(&value)) => value as f64,
436            _ => unreachable!(),
437        },
438        _ => unreachable!(),
439    })
440}
441
442fn parse_null(s: &str) -> Option<Atom<'static>> {
443    match s {
444        "" | "~" | "null" | "Null" | "NULL" => Some(Atom::Null),
445        _ => None,
446    }
447}
448
449fn parse_bool(s: &str, version: Version) -> Option<Atom<'static>> {
450    match s {
451        "true" | "True" | "TRUE" => Some(Atom::Bool(true)),
452        "false" | "False" | "FALSE" => Some(Atom::Bool(false)),
453        "y" | "Y" | "yes" | "Yes" | "YES" | "on" | "On" | "ON" if version == Version::V1_1 => {
454            Some(Atom::Bool(true))
455        }
456        "n" | "N" | "no" | "No" | "NO" | "off" | "Off" | "OFF" if version == Version::V1_1 => {
457            Some(Atom::Bool(false))
458        }
459        _ => None,
460    }
461}
462
463/// Splits off an optional sign.  Returns `true` for negative numbers.
464fn split_sign(s: &str) -> (bool, &str) {
465    match s.as_bytes().first() {
466        Some(b'-') => (true, &s[1..]),
467        Some(b'+') => (false, &s[1..]),
468        _ => (false, s),
469    }
470}
471
472/// Accumulates digits of a radix, skipping `_` if allowed.
473///
474/// Returns `None` if there are no digits or an invalid character.  The
475/// value saturates to `None` in the magnitude if it overflows 128 bits, in
476/// which case the approximate float value is returned as well.
477fn accumulate(digits: &str, radix: u32, underscores: bool) -> Option<Magnitude> {
478    let mut value = Some(0u128);
479    let mut approx = 0f64;
480    let mut seen = false;
481    for c in digits.chars() {
482        if c == '_' && underscores {
483            continue;
484        }
485        let digit = c.to_digit(radix)?;
486        seen = true;
487        value = value
488            .and_then(|v| v.checked_mul(radix as u128))
489            .and_then(|v| v.checked_add(digit as u128));
490        approx = approx * radix as f64 + digit as f64;
491    }
492    if value.is_none() && radix == 10 {
493        // parsing the text rounds correctly
494        approx = digits.replace('_', "").parse().unwrap_or(approx);
495    }
496    if seen {
497        Some(Magnitude { value, approx })
498    } else {
499        None
500    }
501}
502
503struct Magnitude {
504    value: Option<u128>,
505    approx: f64,
506}
507
508fn make_int(negative: bool, magnitude: Magnitude) -> Atom<'static> {
509    match magnitude.value {
510        Some(value) if !negative => match u64::try_from(value) {
511            Ok(value) => Atom::U64(value),
512            Err(_) => Atom::Ext(ExtValue::owned(value)),
513        },
514        Some(value) if value <= 1u128 << 63 => Atom::I64((value as i128).wrapping_neg() as i64),
515        Some(value) if value <= 1u128 << 127 => {
516            Atom::Ext(ExtValue::owned((value as i128).wrapping_neg()))
517        }
518        // integers that do not even fit into 128 bits are approximated
519        _ => Atom::F64(if negative {
520            -magnitude.approx
521        } else {
522            magnitude.approx
523        }),
524    }
525}
526
527/// YAML 1.2 core schema integers: `[-+]?[0-9]+`, `0o[0-7]+`, `0x[0-9a-fA-F]+`.
528fn parse_core_int(s: &str) -> Option<Atom<'static>> {
529    if let Some(rest) = s.strip_prefix("0o") {
530        return accumulate(rest, 8, false).map(|m| make_int(false, m));
531    }
532    if let Some(rest) = s.strip_prefix("0x") {
533        return accumulate(rest, 16, false).map(|m| make_int(false, m));
534    }
535    let (negative, digits) = split_sign(s);
536    accumulate(digits, 10, false).map(|m| make_int(negative, m))
537}
538
539/// YAML 1.2 core schema integers and floats.
540///
541/// Decimal integers that fit into 64 bits are parsed in a single pass and
542/// decimal floats skip the attempt to parse them as integer.  All other
543/// numbers go through [`parse_core_int`] and [`parse_core_float`].
544fn parse_core_number(s: &str) -> Option<Atom<'static>> {
545    let bytes = s.as_bytes();
546    let negative = bytes.first() == Some(&b'-');
547    let digits_start = usize::from(matches!(bytes.first(), Some(b'-' | b'+')));
548    let mut pos = digits_start;
549    let mut value = Some(0u64);
550    while let Some(&b @ b'0'..=b'9') = bytes.get(pos) {
551        value = value
552            .and_then(|v| v.checked_mul(10))
553            .and_then(|v| v.checked_add(u64::from(b - b'0')));
554        pos += 1;
555    }
556    match (bytes.get(pos), value) {
557        (None, Some(value)) if pos > digits_start => Some(make_int(
558            negative,
559            Magnitude {
560                value: Some(value.into()),
561                approx: value as f64,
562            },
563        )),
564        (Some(b'.' | b'e' | b'E'), _) => parse_core_float(s),
565        _ => parse_core_int(s).or_else(|| parse_core_float(s)),
566    }
567}
568
569/// YAML 1.1 integers: binary, octal, decimal, hexadecimal and base 60, all
570/// with an optional sign and `_` separators.
571fn parse_yaml11_int(s: &str) -> Option<Atom<'static>> {
572    let (negative, rest) = split_sign(s);
573    let magnitude = if let Some(digits) = rest.strip_prefix("0b") {
574        accumulate(digits, 2, true)?
575    } else if let Some(digits) = rest.strip_prefix("0x") {
576        accumulate(digits, 16, true)?
577    } else if rest == "0" {
578        accumulate(rest, 10, false)?
579    } else if let Some(digits) = rest.strip_prefix('0') {
580        accumulate(digits, 8, true)?
581    } else if rest
582        .as_bytes()
583        .first()
584        .is_some_and(|b| (b'1'..=b'9').contains(b))
585    {
586        if rest.contains(':') {
587            parse_base60(rest)?
588        } else {
589            accumulate(rest, 10, true)?
590        }
591    } else {
592        return None;
593    };
594    Some(make_int(negative, magnitude))
595}
596
597/// Parses the base 60 integer `[1-9][0-9_]*(:[0-5]?[0-9])+`.
598fn parse_base60(s: &str) -> Option<Magnitude> {
599    let mut parts = s.split(':');
600    let mut rv = accumulate(parts.next()?, 10, true)?;
601    for part in parts {
602        if !is_base60_digit(part) {
603            return None;
604        }
605        let digit = part.parse::<u8>().ok()?;
606        rv.value = rv
607            .value
608            .and_then(|v| v.checked_mul(60))
609            .and_then(|v| v.checked_add(digit as u128));
610        rv.approx = rv.approx * 60.0 + digit as f64;
611    }
612    Some(rv)
613}
614
615fn is_base60_digit(s: &str) -> bool {
616    matches!(s.as_bytes(), [b'0'..=b'9'] | [b'0'..=b'5', b'0'..=b'9'])
617}
618
619fn parse_special_float(s: &str) -> Option<Atom<'static>> {
620    match s {
621        ".nan" | ".NaN" | ".NAN" => return Some(Atom::F64(f64::NAN)),
622        _ => {}
623    }
624    let (negative, rest) = split_sign(s);
625    match rest {
626        ".inf" | ".Inf" | ".INF" => Some(Atom::F64(if negative {
627            f64::NEG_INFINITY
628        } else {
629            f64::INFINITY
630        })),
631        _ => None,
632    }
633}
634
635/// Skips ASCII digits (and `_` if allowed) and returns the number of
636/// digits.
637fn skip_digits(bytes: &[u8], pos: &mut usize, underscores: bool) -> usize {
638    let mut count = 0;
639    while let Some(&b) = bytes.get(*pos) {
640        if b.is_ascii_digit() {
641            count += 1;
642        } else if !(b == b'_' && underscores) {
643            break;
644        }
645        *pos += 1;
646    }
647    count
648}
649
650/// Skips an exponent (`[eE][-+]?[0-9]+`) if there is one.  Returns `false`
651/// if the exponent is malformed.
652fn skip_exponent(bytes: &[u8], pos: &mut usize, sign_required: bool) -> bool {
653    if !matches!(bytes.get(*pos), Some(b'e' | b'E')) {
654        return true;
655    }
656    *pos += 1;
657    if matches!(bytes.get(*pos), Some(b'-' | b'+')) {
658        *pos += 1;
659    } else if sign_required {
660        return false;
661    }
662    skip_digits(bytes, pos, false) > 0
663}
664
665fn parse_float_text(s: &str, underscores: bool) -> Option<Atom<'static>> {
666    let value = if underscores && s.contains('_') {
667        s.replace('_', "").parse()
668    } else {
669        s.parse()
670    };
671    value.ok().map(Atom::F64)
672}
673
674/// YAML 1.2 core schema floats:
675/// `[-+]?(\.[0-9]+|[0-9]+(\.[0-9]*)?)([eE][-+]?[0-9]+)?` and the special
676/// values.
677fn parse_core_float(s: &str) -> Option<Atom<'static>> {
678    if let Some(atom) = parse_special_float(s) {
679        return Some(atom);
680    }
681    let bytes = s.as_bytes();
682    let mut pos = usize::from(matches!(bytes.first(), Some(b'-' | b'+')));
683    let int_digits = skip_digits(bytes, &mut pos, false);
684    let mut frac_digits = 0;
685    if bytes.get(pos) == Some(&b'.') {
686        pos += 1;
687        frac_digits = skip_digits(bytes, &mut pos, false);
688    } else if int_digits == 0 {
689        return None;
690    }
691    if int_digits == 0 && frac_digits == 0 {
692        return None;
693    }
694    if !skip_exponent(bytes, &mut pos, false) || pos != bytes.len() {
695        return None;
696    }
697    parse_float_text(s, false)
698}
699
700/// YAML 1.1 floats: `[-+]?[0-9][0-9_]*\.[0-9_]*([eE][-+][0-9]+)?`,
701/// `[-+]?\.[0-9][0-9_]*([eE][-+][0-9]+)?`, base 60 floats and the special
702/// values.
703fn parse_yaml11_float(s: &str) -> Option<Atom<'static>> {
704    if let Some(atom) = parse_special_float(s) {
705        return Some(atom);
706    }
707    let (negative, rest) = split_sign(s);
708    let bytes = rest.as_bytes();
709    let mut pos = 0;
710    match bytes.first() {
711        Some(b'0'..=b'9') => {
712            skip_digits(bytes, &mut pos, true);
713            if bytes.get(pos) == Some(&b':') {
714                return parse_base60_float(negative, rest);
715            }
716            if bytes.get(pos) != Some(&b'.') {
717                return None;
718            }
719            pos += 1;
720            skip_digits(bytes, &mut pos, true);
721        }
722        Some(b'.') if matches!(bytes.get(1), Some(b'0'..=b'9')) => {
723            pos += 1;
724            skip_digits(bytes, &mut pos, true);
725        }
726        _ => return None,
727    }
728    if !skip_exponent(bytes, &mut pos, true) || pos != bytes.len() {
729        return None;
730    }
731    parse_float_text(s, true)
732}
733
734/// Parses `[0-9][0-9_]*(:[0-5]?[0-9])+\.[0-9_]*`.
735fn parse_base60_float(negative: bool, s: &str) -> Option<Atom<'static>> {
736    let (int_part, frac_part) = s.split_once('.')?;
737    let mut parts = int_part.split(':');
738    let mut value = accumulate(parts.next()?, 10, true)?.approx;
739    let mut count = 0;
740    for part in parts {
741        if !is_base60_digit(part) {
742            return None;
743        }
744        value = value * 60.0 + part.parse::<u8>().ok()? as f64;
745        count += 1;
746    }
747    if count == 0 || !frac_part.bytes().all(|b| b.is_ascii_digit() || b == b'_') {
748        return None;
749    }
750    let frac: f64 = format!("0.{}", frac_part.replace('_', "")).parse().ok()?;
751    value += frac;
752    Some(Atom::F64(if negative { -value } else { value }))
753}
754
755/// Decodes base64 as used by `!!binary`.  Whitespace (the line breaks of
756/// block scalars) is ignored, otherwise it decodes like other bytes
757/// (leniently, see [`deser::adapters::Base64`](deser_core::adapters::Base64)).
758fn decode_base64(s: &str) -> Option<Vec<u8>> {
759    if s.bytes().any(|b| b.is_ascii_whitespace()) {
760        let s: String = s.chars().filter(|c| !c.is_ascii_whitespace()).collect();
761        Base64::decode(&s).ok()
762    } else {
763        Base64::decode(s).ok()
764    }
765}
766
767#[test]
768fn test_base64() {
769    assert_eq!(decode_base64("").unwrap(), b"");
770    assert_eq!(decode_base64("Zg==").unwrap(), b"f");
771    assert_eq!(decode_base64("Zm8=").unwrap(), b"fo");
772    assert_eq!(decode_base64("Zm9v").unwrap(), b"foo");
773    assert_eq!(decode_base64("Zm9v\n YmFy").unwrap(), b"foobar");
774    assert_eq!(decode_base64("Zm9v\n YmE=\n").unwrap(), b"fooba");
775    // lenient like other bytes
776    assert_eq!(decode_base64("Zm8").unwrap(), b"fo");
777    assert_eq!(decode_base64("-_8=").unwrap(), b"\xfb\xff");
778    assert_eq!(decode_base64("Zm9"), None);
779    assert_eq!(decode_base64("Z=9v"), None);
780    assert_eq!(decode_base64("Zm9!"), None);
781    assert_eq!(decode_base64("Zm8=="), None);
782    // unused bits have to be zero
783    assert_eq!(decode_base64("Zh=="), None);
784}
785
786#[test]
787fn test_big_ints() {
788    assert_eq!(
789        parse_core_int("18446744073709551616"),
790        Some(Atom::Ext(ExtValue::owned(18446744073709551616u128)))
791    );
792    assert_eq!(
793        parse_core_int("-9223372036854775808"),
794        Some(Atom::I64(i64::MIN))
795    );
796    assert_eq!(
797        parse_core_int("-9223372036854775809"),
798        Some(Atom::Ext(ExtValue::owned(-9223372036854775809i128)))
799    );
800    assert_eq!(
801        parse_core_int("-170141183460469231731687303715884105728"),
802        Some(Atom::Ext(ExtValue::owned(i128::MIN)))
803    );
804    assert_eq!(
805        parse_core_int("1000000000000000000000000000000000000000000"),
806        Some(Atom::F64(1e42))
807    );
808}
809
810#[test]
811fn test_core_number_fast_path() {
812    let tokens = [
813        "0",
814        "1",
815        "+1",
816        "-1",
817        "-0",
818        "007",
819        "123456789",
820        "9223372036854775807",
821        "9223372036854775808",
822        "-9223372036854775808",
823        "-9223372036854775809",
824        "18446744073709551615",
825        "18446744073709551616",
826        "-18446744073709551616",
827        "340282366920938463463374607431768211456",
828        "1.5",
829        "-1.5",
830        "+1.5",
831        "1.",
832        ".5",
833        "-.5",
834        "1e5",
835        "1E+5",
836        "1e-5",
837        "1.5e-5",
838        "1e",
839        "1.e5",
840        "1e400",
841        "0x1f",
842        "0o17",
843        "-0x1f",
844        ".inf",
845        "-.inf",
846        ".nan",
847        "1_000",
848        "1-2",
849        "-",
850        "+",
851        "12a",
852        "1:30",
853    ];
854    for token in tokens {
855        let expected = parse_core_int(token).or_else(|| parse_core_float(token));
856        assert_eq!(
857            format!("{:?}", parse_core_number(token)),
858            format!("{:?}", expected),
859            "{}",
860            token
861        );
862    }
863}