use alloc::format;
use alloc::string::String;
use alloc::vec::Vec;
use core::str;
use deser_core::__format::extend;
use deser_core::Text;
use deser_core::de::DeserializeDriver;
use deser_core::ext::{ExtValue, Number as ExactNumber};
use deser_core::{Atom, Error, ErrorKind, Event, State};
use crate::scan::skip_to_escape_single;
use crate::scan::{EscapeScanner, is_ascii, skip_to_escape, validate_utf8_slice};
pub(crate) enum Str<'a, 'b> {
Borrowed(&'a str),
Scratch(&'b str),
}
pub(crate) enum Number<'a> {
I64(i64),
BigInt(&'a str),
U64(u64),
F64(f64),
Literal(f64),
U128(u128),
I128(i128),
}
impl Number<'_> {
fn into_f64(self) -> f64 {
match self {
Number::F64(value) | Number::Literal(value) => value,
_ => unreachable!("not a float"),
}
}
}
macro_rules! overflow {
($a:ident * 10 + $b:ident, $c:expr) => {
$a >= $c / 10 && ($a > $c / 10 || $b > $c % 10)
};
}
pub(crate) trait Out<'i> {
fn state_mut(&mut self) -> &mut State;
fn emit<'e, E: Into<Event<'e>>>(&mut self, event: E) -> Result<(), Error>;
fn emit_input(&mut self, atom: Atom<'i>) -> Result<(), Error>;
}
#[inline(always)]
fn key_atom(key: &str) -> Atom<'_> {
Atom::Lexical(Text::borrowed(key))
}
pub(crate) struct Borrowing<'a, 'd, 'i>(pub &'a mut DeserializeDriver<'d, 'i>);
impl<'i> Out<'i> for Borrowing<'_, '_, 'i> {
#[inline(always)]
fn state_mut(&mut self) -> &mut State {
self.0.state_mut()
}
#[inline(always)]
fn emit<'e, E: Into<Event<'e>>>(&mut self, event: E) -> Result<(), Error> {
self.0.emit(event)
}
#[inline(always)]
fn emit_input(&mut self, atom: Atom<'i>) -> Result<(), Error> {
self.0.emit_borrowed(atom)
}
}
pub(crate) struct Copying<'a, 'd, 'de>(pub &'a mut DeserializeDriver<'d, 'de>);
impl<'i> Out<'i> for Copying<'_, '_, '_> {
#[inline(always)]
fn state_mut(&mut self) -> &mut State {
self.0.state_mut()
}
#[inline(always)]
fn emit<'e, E: Into<Event<'e>>>(&mut self, event: E) -> Result<(), Error> {
self.0.emit(event)
}
#[inline(always)]
fn emit_input(&mut self, atom: Atom<'i>) -> Result<(), Error> {
self.0.emit(atom)
}
}
pub(crate) struct Discard(pub State);
impl<'i> Out<'i> for Discard {
#[inline(always)]
fn state_mut(&mut self) -> &mut State {
&mut self.0
}
#[inline(always)]
fn emit<'e, E: Into<Event<'e>>>(&mut self, _event: E) -> Result<(), Error> {
Ok(())
}
#[inline(always)]
fn emit_input(&mut self, _atom: Atom<'i>) -> Result<(), Error> {
Ok(())
}
}
#[derive(Clone, Copy)]
pub(crate) struct Options {
pub validate_utf8: bool,
pub exact_numbers: bool,
}
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub(crate) enum Progress {
Done(usize),
NeedMore(usize),
}
#[derive(Clone, Copy, PartialEq, Eq, Debug)]
enum Container {
Top,
Seq,
Map,
}
#[derive(Clone, Copy, PartialEq, Eq, Debug)]
enum Expect {
Value,
ValueOrEnd,
Key,
KeyOrEnd,
Colon,
AfterValue,
}
#[derive(Clone, Copy, Debug)]
struct PartialString {
scanned: usize,
copied: usize,
}
#[derive(Debug)]
pub(crate) struct Parser {
stack: Vec<Container>,
container: Container,
expect: Expect,
scratch: Vec<u8>,
partial: Option<PartialString>,
recoverable: Option<usize>,
}
impl Default for Parser {
fn default() -> Parser {
Parser {
stack: Vec::new(),
container: Container::Top,
expect: Expect::Value,
scratch: Vec::new(),
partial: None,
recoverable: None,
}
}
}
macro_rules! emit {
($out:expr, $base:expr, $start:expr, $end:expr, $event:expr) => {{
$out.state_mut()
.set_input_range($base + $start, $base + $end);
$out.emit($event)
}};
}
impl Parser {
pub(crate) fn is_idle(&self) -> bool {
self.expect == Expect::Value && self.container == Container::Top && self.partial.is_none()
}
pub(crate) fn reset(&mut self) {
self.stack.clear();
self.container = Container::Top;
self.expect = Expect::Value;
self.partial = None;
self.recoverable = None;
}
pub(crate) fn recoverable(&self) -> Option<usize> {
self.recoverable
}
#[inline(always)]
pub(crate) fn parse<'i, O: Out<'i>>(
&mut self,
input: &'i [u8],
pos: usize,
eof: bool,
base: usize,
options: Options,
out: &mut O,
) -> Result<Progress, Error> {
self.recoverable = None;
let mut cur = Cursor {
input,
pos,
validate_utf8: options.validate_utf8,
hit_end: false,
partial: (0, 0),
number_start: 0,
truncated: false,
eof,
};
if self.scratch.capacity() == 0 {
take_scratch(&mut self.scratch, out.state_mut());
}
let rv = match self.run(&mut cur, eof, base, options.exact_numbers, out) {
Ok(progress) => Ok(progress),
Err(err) if err.offset().is_none() => Err(err.with_offset(base + cur.pos)),
Err(err) => Err(err),
};
if self.partial.is_none() && self.scratch.capacity() != 0 {
put_scratch(&mut self.scratch, out.state_mut());
}
rv
}
#[inline(always)]
fn run<'i, O: Out<'i>>(
&mut self,
cur: &mut Cursor<'i>,
eof: bool,
base: usize,
exact_numbers: bool,
out: &mut O,
) -> Result<Progress, Error> {
let input = cur.input;
let stack = &mut self.stack;
let scratch = &mut self.scratch;
let mut container = self.container;
let mut partial = self.partial.take();
macro_rules! suspend {
($consumed:expr, $expect:expr) => {{
self.container = container;
self.expect = $expect;
return Ok(Progress::NeedMore($consumed));
}};
}
macro_rules! sink {
($rv:expr, $expect:expr) => {
if let Err(err) = $rv {
self.container = container;
self.expect = $expect;
self.recoverable = Some(cur.pos);
return Err(err);
}
};
}
macro_rules! next_byte {
($expect:expr) => {
match cur.parse_whitespace() {
Some(byte) => byte,
None if eof => return Err(eof_error()),
None => suspend!(cur.pos, $expect),
}
};
}
macro_rules! string {
($start:expr, $expect:expr) => {{
cur.hit_end = false;
let rv = cur.parse_str(scratch, $start, partial.take());
if cur.hit_end && !eof {
self.partial = Some(PartialString {
scanned: cur.partial.0 - $start,
copied: cur.partial.1 - $start,
});
suspend!($start, $expect)
}
rv?
}};
}
macro_rules! close {
($start:expr, $event:expr) => {{
container = stack.pop().unwrap_or(Container::Top);
sink!(
emit!(out, base, $start, cur.pos, $event),
Expect::AfterValue
);
}};
}
macro_rules! key {
($byte:expr) => {{
let start = cur.pos;
let key = match $byte {
b'"' => {
cur.bump();
string!(start, Expect::Key)
}
b'\'' => {
cur.bump();
string!(start, Expect::Key)
}
b'a'..=b'z' | b'A'..=b'Z' | b'_' | b'$' | b'\\' | 0x80..=0xff => {
cur.hit_end = false;
let rv = cur.parse_identifier(scratch);
if cur.hit_end && !eof {
suspend!(start, Expect::Key)
}
rv?
}
_ => return Err(token_error(base + start, "expected map key")),
};
match key {
Str::Borrowed(key) => {
out.state_mut()
.set_input_range(base + start, base + cur.pos);
sink!(out.emit_input(key_atom(key)), Expect::Colon)
}
Str::Scratch(key) => sink!(
emit!(out, base, start, cur.pos, key_atom(key)),
Expect::Colon
),
}
colon!();
}};
}
macro_rules! colon {
() => {
match next_byte!(Expect::Colon) {
b':' => cur.bump(),
_ => return Err(Error::new(ErrorKind::Unexpected, "expected colon")),
}
};
}
macro_rules! open_map {
() => {{
let byte = if partial.is_some() {
b'"'
} else {
next_byte!(Expect::KeyOrEnd)
};
if byte == b'}' {
let start = cur.pos;
cur.bump();
close!(start, Event::MapEnd);
false
} else {
key!(byte);
true
}
}};
}
macro_rules! open_seq {
() => {{
if next_byte!(Expect::ValueOrEnd) == b']' {
let start = cur.pos;
cur.bump();
close!(start, Event::SeqEnd);
false
} else {
true
}
}};
}
macro_rules! string_value {
($start:expr) => {
match string!($start, Expect::Value) {
Str::Borrowed(val) => {
out.state_mut()
.set_input_range(base + $start, base + cur.pos);
sink!(
out.emit_input(Atom::Str(Text::borrowed(val))),
Expect::AfterValue
)
}
Str::Scratch(val) => sink!(
emit!(out, base, $start, cur.pos, Event::from(val)),
Expect::AfterValue
),
}
};
}
macro_rules! number {
($byte:expr, $start:expr) => {{
cur.number_start = $start;
cur.truncated = false;
let rv = match $byte {
b'-' => {
let first_digit = cur.next_or_nul();
cur.parse_integer(false, first_digit)
}
b'+' => {
let first_digit = cur.next_or_nul();
cur.parse_integer(true, first_digit)
}
byte => cur.parse_integer(true, byte),
};
if cur.hit_end && !eof {
suspend!($start, Expect::Value)
}
let number = rv?;
out.state_mut()
.set_input_range(base + $start, base + cur.pos);
sink!(
emit_number(out, number, input, exact_numbers, $start, cur.pos),
Expect::AfterValue
)
}};
}
let mut skip_value = match self.expect {
Expect::Value => false,
Expect::ValueOrEnd => !open_seq!(),
Expect::KeyOrEnd => !open_map!(),
Expect::Key => {
let byte = if partial.is_some() {
b'"'
} else {
next_byte!(Expect::Key)
};
key!(byte);
false
}
Expect::Colon => {
colon!();
false
}
Expect::AfterValue => true,
};
'value: loop {
if !skip_value {
let byte = if partial.is_some() {
b'"'
} else {
next_byte!(Expect::Value)
};
let start = cur.pos;
cur.bump();
cur.hit_end = false;
match byte {
b'"' => string_value!(start),
b'\'' => string_value!(start),
b'0'..=b'9' | b'-' => number!(byte, start),
b'+' | b'.' | b'I' | b'N' => number!(byte, start),
b'n' | b't' | b'f' => {
let (rest, event): (&[u8], _) = match byte {
b'n' => (b"ull", Event::Atom(Atom::Null)),
b't' => (b"rue", Event::from(true)),
_ => (b"alse", Event::from(false)),
};
let rv = cur.parse_ident(rest);
if cur.hit_end && !eof {
suspend!(start, Expect::Value)
}
rv?;
sink!(emit!(out, base, start, cur.pos, event), Expect::AfterValue)
}
b'{' => {
stack.push(container);
container = Container::Map;
sink!(
emit!(out, base, start, cur.pos, Event::map_start()),
Expect::KeyOrEnd
);
if open_map!() {
continue 'value;
}
}
b'[' => {
stack.push(container);
container = Container::Seq;
sink!(
emit!(out, base, start, cur.pos, Event::seq_start()),
Expect::ValueOrEnd
);
if open_seq!() {
continue 'value;
}
}
b',' => return Err(token_error(base + start, "unexpected comma")),
b':' => return Err(token_error(base + start, "unexpected colon")),
b']' | b'}' => return Err(token_error(base + start, "expected a value")),
_ => return Err(token_error(base + start, "unexpected character")),
}
}
skip_value = false;
loop {
let close = match container {
Container::Top => {
self.container = Container::Top;
self.expect = Expect::Value;
return Ok(Progress::Done(cur.pos));
}
Container::Map => b'}',
Container::Seq => b']',
};
match next_byte!(Expect::AfterValue) {
b',' => {
cur.bump();
let more = if container == Container::Map {
open_map!()
} else {
open_seq!()
};
if more {
continue 'value;
}
}
byte if byte == close => {
let start = cur.pos;
cur.bump();
close!(
start,
if close == b'}' {
Event::MapEnd
} else {
Event::SeqEnd
}
);
}
b']' | b'}' => {
return Err(Error::new(
ErrorKind::Unexpected,
if container == Container::Map {
"unexpected end of seq"
} else {
"unexpected end of map"
},
));
}
_ => {
return Err(Error::new(ErrorKind::Unexpected, "expected a comma"));
}
}
}
}
}
}
#[inline(never)]
fn take_scratch(scratch: &mut Vec<u8>, state: &mut State) {
*scratch = state.__private_take_scratch();
}
#[inline(never)]
fn put_scratch(scratch: &mut Vec<u8>, state: &mut State) {
state.__private_put_scratch(core::mem::take(scratch));
}
#[cold]
fn eof_error() -> Error {
Error::new(ErrorKind::EndOfFile, "unexpected end of file")
}
pub(crate) struct Cursor<'a> {
pub(crate) input: &'a [u8],
pub(crate) pos: usize,
validate_utf8: bool,
hit_end: bool,
partial: (usize, usize),
number_start: usize,
truncated: bool,
eof: bool,
}
enum Comment {
End(usize),
Incomplete,
Invalid(usize),
}
impl<'a> Cursor<'a> {
pub(crate) fn new(input: &'a [u8], pos: usize) -> Cursor<'a> {
Cursor {
input,
pos,
validate_utf8: true,
hit_end: false,
partial: (0, 0),
number_start: 0,
truncated: false,
eof: true,
}
}
pub(crate) fn new_partial(input: &'a [u8], pos: usize, eof: bool) -> Cursor<'a> {
Cursor {
eof,
..Cursor::new(input, pos)
}
}
#[inline(always)]
fn parse_str<'b>(
&mut self,
buffer: &'b mut Vec<u8>,
start: usize,
resume: Option<PartialString>,
) -> Result<Str<'a, 'b>, Error> {
let plain = resume.is_none() && self.input[start] != b'\'';
if plain {
let end = skip_to_escape(self.input, self.pos);
if self.input.get(end) == Some(&b'"') {
let bytes = &self.input[self.pos..end];
if !self.validate_utf8 || is_ascii(bytes) || validate_utf8_slice(bytes) {
self.pos = end + 1;
return Ok(Str::Borrowed(unsafe { str::from_utf8_unchecked(bytes) }));
}
}
}
self.parse_str_slow(buffer, start, resume)
}
#[inline(never)]
fn parse_str_slow<'b>(
&mut self,
buffer: &'b mut Vec<u8>,
start: usize,
resume: Option<PartialString>,
) -> Result<Str<'a, 'b>, Error> {
let validate_utf8 = self.validate_utf8;
fn result(validate_utf8: bool, bytes: &[u8]) -> Result<&str, Error> {
if validate_utf8 && !is_ascii(bytes) && !validate_utf8_slice(bytes) {
return Err(Error::new(ErrorKind::Unexpected, "invalid utf-8 in string"));
}
Ok(unsafe { str::from_utf8_unchecked(bytes) })
}
let mut copied = match resume {
Some(resume) => {
self.pos = start + resume.scanned;
start + resume.copied
}
None => {
buffer.clear();
self.pos
}
};
let single = self.input[start] == b'\'';
let mut escapes = EscapeScanner::new();
loop {
self.pos = if single {
skip_to_escape_single(self.input, self.pos)
} else {
escapes.next(self.input, self.pos)
};
if self.pos == self.input.len() {
self.hit_end = true;
self.partial = (self.pos, copied);
return Err(Error::new(
ErrorKind::Unexpected,
"unexpected end of string",
));
}
let byte = self.input[self.pos];
let byte = if single && byte == b'\'' { b'"' } else { byte };
match byte {
b'"' => {
if buffer.is_empty() {
let input = self.input;
let borrowed = &input[copied..self.pos];
self.pos += 1;
return result(validate_utf8, borrowed).map(Str::Borrowed);
} else {
extend(buffer, &self.input[copied..self.pos]);
self.pos += 1;
return result(validate_utf8, buffer).map(Str::Scratch);
}
}
b'\\' => {
extend(buffer, &self.input[copied..self.pos]);
if let Some(&byte) = self.input.get(self.pos + 1)
&& let unescaped @ 1.. = UNESCAPE[usize::from(byte)]
{
buffer.push(unescaped);
self.pos += 2;
copied = self.pos;
continue;
}
let escape = self.pos;
self.pos += 1;
if let Err(err) = self.parse_escape(buffer) {
self.partial = (escape, escape);
return Err(err);
}
copied = self.pos;
}
byte if byte != b'\n' && byte != b'\r' => self.pos += 1,
_ => {
return Err(Error::new(
ErrorKind::Unexpected,
"unexpected character in string",
));
}
}
}
}
#[inline]
fn next(&mut self) -> Option<u8> {
if self.pos < self.input.len() {
let ch = self.input[self.pos];
self.pos += 1;
Some(ch)
} else {
self.hit_end = true;
None
}
}
fn next_or_nul(&mut self) -> u8 {
self.next().unwrap_or(b'\0')
}
#[inline]
fn peek(&mut self) -> Option<u8> {
if self.pos < self.input.len() {
Some(self.input[self.pos])
} else {
self.hit_end = true;
None
}
}
fn peek_or_nul(&mut self) -> u8 {
self.peek().unwrap_or(b'\0')
}
#[inline(always)]
fn digits(&mut self) -> Option<(u64, usize)> {
let bytes = self.input.get(self.pos..self.pos + 8)?;
let value = u64::from_le_bytes(bytes.try_into().unwrap());
let digits = value.wrapping_sub(0x3030_3030_3030_3030);
let above = value.wrapping_add(0x4646_4646_4646_4646);
let other = (digits | above) & 0x8080_8080_8080_8080;
if other == 0 {
self.pos += 8;
return Some((combine_digits(digits), 8));
}
if other & 0x80 != 0 {
return Some((0, 0));
}
let count = (other.trailing_zeros() / 8) as usize;
self.pos += count;
Some((combine_digits(digits << (64 - 8 * count)), count))
}
#[inline]
fn bump(&mut self) {
self.pos += 1;
}
fn next_or_eof(&mut self) -> Result<u8, Error> {
self.next()
.ok_or_else(|| Error::new(ErrorKind::EndOfFile, "unexpected end of file"))
}
fn parse_escape(&mut self, buffer: &mut Vec<u8>) -> Result<(), Error> {
let ch = self.next_or_eof()?;
match ch {
b'"' => buffer.push(b'"'),
b'\\' => buffer.push(b'\\'),
b'/' => buffer.push(b'/'),
b'b' => buffer.push(b'\x08'),
b'f' => buffer.push(b'\x0c'),
b'n' => buffer.push(b'\n'),
b'r' => buffer.push(b'\r'),
b't' => buffer.push(b'\t'),
b'u' => {
let c = match self.decode_hex_escape()? {
0xDC00..=0xDFFF => return Err(lone_surrogate()),
n1 @ 0xD800..=0xDBFF => {
if self.next_or_eof()? != b'\\' || self.next_or_eof()? != b'u' {
return Err(lone_surrogate());
}
let n2 = self.decode_hex_escape()?;
if !(0xDC00..=0xDFFF).contains(&n2) {
return Err(lone_surrogate());
}
let n = (u32::from(n1 - 0xD800) << 10 | u32::from(n2 - 0xDC00)) + 0x1_0000;
match char::from_u32(n) {
Some(c) => c,
None => return Err(lone_surrogate()),
}
}
n => match char::from_u32(u32::from(n)) {
Some(c) => c,
None => return Err(lone_surrogate()),
},
};
buffer.extend_from_slice(c.encode_utf8(&mut [0_u8; 4]).as_bytes());
}
b'\'' => buffer.push(b'\''),
b'v' => buffer.push(b'\x0b'),
b'0' => match self.peek() {
Some(b'0'..=b'9') => return Err(invalid_escape()),
Some(_) => buffer.push(b'\0'),
None => return Err(eof_error()),
},
b'x' => {
let mut n = 0;
for _ in 0..2 {
match char::from(self.next_or_eof()?).to_digit(16) {
Some(digit) => n = n * 16 + digit,
None => {
return Err(Error::new(ErrorKind::Unexpected, "invalid hex escape"));
}
}
}
let c = char::from_u32(n).unwrap();
buffer.extend_from_slice(c.encode_utf8(&mut [0_u8; 4]).as_bytes());
}
b'\n' => {}
b'\r' => match self.peek() {
Some(b'\n') => self.bump(),
Some(_) => {}
None => return Err(eof_error()),
},
0xe2 => match self.input.get(self.pos..self.pos + 2) {
Some([0x80, 0xa8 | 0xa9]) => self.pos += 2,
Some(_) => buffer.push(0xe2),
None => {
self.hit_end = true;
return Err(eof_error());
}
},
b'1'..=b'9' => return Err(invalid_escape()),
byte => buffer.push(byte),
}
Ok(())
}
fn parse_identifier<'b>(&mut self, buffer: &'b mut Vec<u8>) -> Result<Str<'a, 'b>, Error> {
let start = self.pos;
let mut copied = None;
loop {
let first = self.pos == start;
match self.peek() {
Some(b'a'..=b'z' | b'A'..=b'Z' | b'_' | b'$') => self.bump(),
Some(b'0'..=b'9') if !first => self.bump(),
Some(b'\\') => {
let from = match copied {
Some(from) => from,
None => {
buffer.clear();
start
}
};
buffer.extend_from_slice(&self.input[from..self.pos]);
let escape = self.pos;
self.bump();
let c = match self.next_or_eof()? {
b'u' => char::from_u32(u32::from(self.decode_hex_escape()?)),
_ => None,
};
let Some(c) = c.filter(|&c| is_identifier_char(c, first)) else {
self.pos = escape;
return Err(invalid_identifier());
};
buffer.extend_from_slice(c.encode_utf8(&mut [0; 4]).as_bytes());
copied = Some(self.pos);
}
Some(byte @ 0x80..=0xff) => {
let len = match byte {
0xc0..=0xdf => 2,
0xe0..=0xef => 3,
_ => 4,
};
let Some(bytes) = self.input.get(self.pos..self.pos + len) else {
self.hit_end = true;
return Err(eof_error());
};
match str::from_utf8(bytes).ok().and_then(|s| s.chars().next()) {
Some(c) if is_identifier_char(c, first) => self.pos += len,
Some(_) => break,
None => return Err(Error::new(ErrorKind::Unexpected, "invalid utf-8")),
}
}
_ => break,
}
}
if self.pos == start {
return Err(Error::new(ErrorKind::Unexpected, "expected map key"));
}
match copied {
Some(from) => {
buffer.extend_from_slice(&self.input[from..self.pos]);
Ok(Str::Scratch(unsafe { str::from_utf8_unchecked(buffer) }))
}
None => Ok(Str::Borrowed(unsafe {
str::from_utf8_unchecked(&self.input[start..self.pos])
})),
}
}
fn decode_hex_escape(&mut self) -> Result<u16, Error> {
let mut n = 0;
for _ in 0..4 {
n = match self.next_or_eof()? {
c @ b'0'..=b'9' => n * 16_u16 + u16::from(c - b'0'),
b'a' | b'A' => n * 16_u16 + 10_u16,
b'b' | b'B' => n * 16_u16 + 11_u16,
b'c' | b'C' => n * 16_u16 + 12_u16,
b'd' | b'D' => n * 16_u16 + 13_u16,
b'e' | b'E' => n * 16_u16 + 14_u16,
b'f' | b'F' => n * 16_u16 + 15_u16,
_ => {
return Err(Error::new(ErrorKind::Unexpected, "invalid hex escape"));
}
};
}
Ok(n)
}
#[inline(always)]
pub(crate) fn parse_whitespace(&mut self) -> Option<u8> {
const SPACES: u64 = u64::from_ne_bytes([b' '; 8]);
let input = self.input;
let mut pos = self.pos;
loop {
if pos + 8 <= input.len() {
let mut bytes = [0u8; 8];
bytes.copy_from_slice(&input[pos..pos + 8]);
if u64::from_ne_bytes(bytes) == SPACES {
pos += 8;
continue;
}
}
match input.get(pos) {
Some(b' ' | b'\n' | b'\t' | b'\r') => pos += 1,
Some(0x0b | 0x0c) => pos += 1,
Some(b'/') => match self.skip_comment(pos) {
Comment::End(end) => pos = end,
Comment::Incomplete => {
self.pos = pos;
self.hit_end = true;
return None;
}
Comment::Invalid(pos) => {
self.pos = pos;
return Some(input[pos]);
}
},
Some(&byte) if byte >= 0x80 => match unicode_whitespace(&input[pos..]) {
Some(len) => pos += len,
None if input.len() - pos < 3 && !self.eof => {
self.pos = pos;
self.hit_end = true;
return None;
}
None => {
self.pos = pos;
return Some(byte);
}
},
Some(&byte) => {
self.pos = pos;
return Some(byte);
}
None => {
self.pos = pos;
self.hit_end = true;
return None;
}
}
}
}
#[cold]
fn skip_comment(&self, pos: usize) -> Comment {
let input = self.input;
let end = match input.get(pos + 1) {
Some(b'/') => match line_comment_len(&input[pos + 2..]) {
Some(len) => pos + 2 + len,
None if self.eof => input.len(),
None => return Comment::Incomplete,
},
Some(b'*') => match input[pos + 2..].windows(2).position(|w| w == b"*/") {
Some(index) => pos + 2 + index + 2,
None => return Comment::Incomplete,
},
Some(_) => return Comment::Invalid(pos),
None if self.eof => return Comment::Invalid(pos),
None => return Comment::Incomplete,
};
let comment = &input[pos..end];
if self.validate_utf8
&& !is_ascii(comment)
&& let Err(err) = str::from_utf8(comment)
{
return Comment::Invalid(pos + err.valid_up_to());
}
Comment::End(end)
}
fn parse_ident(&mut self, ident: &[u8]) -> Result<(), Error> {
for expected in ident {
match self.next() {
None => {
return Err(Error::new(ErrorKind::EndOfFile, "unexpected end of file"));
}
Some(next) => {
if next != *expected {
return Err(Error::new(ErrorKind::Unexpected, "unexpected character"));
}
}
}
}
Ok(())
}
fn parse_integer(&mut self, nonnegative: bool, first_digit: u8) -> Result<Number<'a>, Error> {
match first_digit {
b'0' => match self.peek_or_nul() {
b'0'..=b'9' => Err(Error::new(
ErrorKind::Unexpected,
"only a single leading 0 is allowed",
)),
b'x' | b'X' => self.parse_hex(nonnegative),
_ => self.parse_number(nonnegative, 0),
},
b'.' => match self.peek_or_nul() {
b'0'..=b'9' => {
self.pos -= 1;
self.parse_decimal(nonnegative, 0, 0)
}
_ => Err(Error::new(ErrorKind::Unexpected, "expected a digit")),
},
b'I' => {
self.parse_ident(b"nfinity")?;
Ok(Number::F64(if nonnegative {
f64::INFINITY
} else {
f64::NEG_INFINITY
}))
}
b'N' => {
self.parse_ident(b"aN")?;
Ok(Number::F64(f64::NAN))
}
c @ b'1'..=b'9' => {
let mut res = u64::from(c - b'0');
while res < EIGHT_DIGITS_LIMIT
&& let Some((digits, count)) = self.digits()
{
res = res * POW10_U64[count] + digits;
if count < 8 {
return self.parse_number(nonnegative, res);
}
}
loop {
match self.peek_or_nul() {
c @ b'0'..=b'9' => {
self.bump();
let digit = u64::from(c - b'0');
if overflow!(res * 10 + digit, u64::MAX) {
return self.parse_overflowing_integer(nonnegative, res);
}
res = res * 10 + digit;
}
_ => {
return self.parse_number(nonnegative, res);
}
}
}
}
_ => Err(Error::new(ErrorKind::Unexpected, "invalid integer")),
}
}
fn number_text(&self, nonnegative: bool) -> &'a str {
let input = self.input;
let mut start = self.pos;
while start > 0 && input[start - 1].is_ascii_digit() {
start -= 1;
}
if !nonnegative {
start -= 1;
}
str::from_utf8(&input[start..self.pos]).unwrap()
}
#[cold]
fn parse_overflowing_integer(
&mut self,
nonnegative: bool,
significand: u64,
) -> Result<Number<'a>, Error> {
let digits_start = self.pos - 1;
self.truncated = true;
let float = self.parse_long_integer(
nonnegative,
significand,
1, )?;
let is_integer = self.input[digits_start..self.pos]
.iter()
.all(|c| c.is_ascii_digit());
if is_integer {
let text = self.number_text(nonnegative);
let fits = if nonnegative {
text.parse::<u128>().is_ok()
} else {
text.parse::<i128>().is_ok()
};
if fits {
return Ok(Number::BigInt(text));
}
}
Ok(Number::Literal(float))
}
fn parse_long_integer(
&mut self,
nonnegative: bool,
significand: u64,
mut exponent: i32,
) -> Result<f64, Error> {
loop {
match self.peek_or_nul() {
b'0'..=b'9' => {
self.bump();
exponent += 1;
}
b'.' => {
return self
.parse_decimal(nonnegative, significand, exponent)
.map(Number::into_f64);
}
b'e' | b'E' => {
return self.parse_exponent(nonnegative, significand, exponent);
}
_ => {
return self.float(nonnegative, significand, exponent);
}
}
}
}
fn parse_number(&mut self, nonnegative: bool, significand: u64) -> Result<Number<'a>, Error> {
match self.peek_or_nul() {
b'.' => self.parse_decimal(nonnegative, significand, 0),
b'e' | b'E' => self
.parse_exponent(nonnegative, significand, 0)
.map(Number::Literal),
_ => {
Ok(if nonnegative {
Number::U64(significand)
} else {
let neg = (significand as i64).wrapping_neg();
if neg > 0 {
Number::BigInt(self.number_text(false))
} else {
Number::I64(neg)
}
})
}
}
}
fn parse_decimal(
&mut self,
nonnegative: bool,
mut significand: u64,
starting_exp: i32,
) -> Result<Number<'a>, Error> {
self.bump();
let mut exponent = starting_exp;
let mut overflowed = false;
while significand < EIGHT_DIGITS_LIMIT
&& let Some((digits, count)) = self.digits()
{
significand = significand * POW10_U64[count] + digits;
exponent -= count as i32;
if count < 8 {
break;
}
}
while let c @ b'0'..=b'9' = self.peek_or_nul() {
self.bump();
let digit = u64::from(c - b'0');
if overflow!(significand * 10 + digit, u64::MAX) {
while let b'0'..=b'9' = self.peek_or_nul() {
self.bump();
}
overflowed = true;
self.truncated = true;
break;
}
significand = significand * 10 + digit;
exponent -= 1;
}
match self.peek_or_nul() {
b'e' | b'E' => self
.parse_exponent(nonnegative, significand, exponent)
.map(Number::Literal),
_ => {
let value = self.float(nonnegative, significand, exponent)?;
Ok(
if !overflowed
&& starting_exp == 0
&& is_shortest_repr(significand, exponent.unsigned_abs())
{
Number::F64(value)
} else {
Number::Literal(value)
},
)
}
}
}
fn parse_hex(&mut self, nonnegative: bool) -> Result<Number<'a>, Error> {
self.bump();
let mut value = 0u128;
let mut dropped = 0u64;
let mut sticky = false;
let mut digits = 0;
while let Some(digit) = char::from(self.peek_or_nul()).to_digit(16) {
self.bump();
digits += 1;
match value.checked_mul(16) {
Some(shifted) if dropped == 0 => value = shifted | u128::from(digit),
_ => {
dropped += 1;
sticky |= digit != 0;
}
}
}
if digits == 0 {
return Err(Error::new(ErrorKind::Unexpected, "expected a hex digit"));
}
if dropped > 0 {
let scale = f64::from_bits((1023 + 4 * dropped.min(256)) << 52);
let float = (value | u128::from(sticky)) as f64 * scale;
if float.is_infinite() {
return Err(number_out_of_range());
}
return Ok(Number::F64(if nonnegative { float } else { -float }));
}
Ok(match value {
value if nonnegative => match u64::try_from(value) {
Ok(value) => Number::U64(value),
Err(_) => Number::U128(value),
},
value if value <= 1 << 63 => Number::I64((value as i64).wrapping_neg()),
value if value <= 1 << 127 => Number::I128((value as i128).wrapping_neg()),
value => Number::F64(-(value as f64)),
})
}
fn parse_exponent(
&mut self,
nonnegative: bool,
significand: u64,
starting_exp: i32,
) -> Result<f64, Error> {
self.bump();
let positive_exp = match self.peek_or_nul() {
b'+' => {
self.bump();
true
}
b'-' => {
self.bump();
false
}
_ => true,
};
let mut exp = match self.next_or_nul() {
c @ b'0'..=b'9' => i32::from(c - b'0'),
_ => {
return Err(Error::new(
ErrorKind::Unexpected,
"expected digit after exponent",
));
}
};
while let c @ b'0'..=b'9' = self.peek_or_nul() {
self.bump();
let digit = i32::from(c - b'0');
if overflow!(exp * 10 + digit, i32::MAX) {
return self.parse_exponent_overflow(nonnegative, significand, positive_exp);
}
exp = exp * 10 + digit;
}
let final_exp = if positive_exp {
starting_exp.saturating_add(exp)
} else {
starting_exp.saturating_sub(exp)
};
self.float(nonnegative, significand, final_exp)
}
#[inline]
fn float(&self, nonnegative: bool, significand: u64, exponent: i32) -> Result<f64, Error> {
if !self.truncated {
let value = match POW10.get(exponent.unsigned_abs() as usize) {
Some(&pow) if significand <= 1 << 53 => Some(if exponent >= 0 {
significand as f64 * pow
} else {
significand as f64 / pow
}),
Some(_) => eisel_lemire(significand, exponent),
None if significand == 0 => Some(0.0),
None => None,
};
if let Some(value) = value {
return Ok(if nonnegative { value } else { -value });
}
}
self.parse_float_text()
}
#[cold]
#[inline(never)]
fn parse_float_text(&self) -> Result<f64, Error> {
let text = unsafe { str::from_utf8_unchecked(&self.input[self.number_start..self.pos]) };
match text.parse::<f64>() {
Ok(value) if value.is_finite() => Ok(value),
_ => Err(number_out_of_range()),
}
}
#[cold]
#[inline(never)]
fn parse_exponent_overflow(
&mut self,
nonnegative: bool,
significand: u64,
positive_exp: bool,
) -> Result<f64, Error> {
if significand != 0 && positive_exp {
return Err(Error::new(ErrorKind::Unexpected, "infinity takes no sign"));
}
while let b'0'..=b'9' = self.peek_or_nul() {
self.bump();
}
Ok(if nonnegative { 0.0 } else { -0.0 })
}
}
const EIGHT_DIGITS_LIMIT: u64 = (u64::MAX - 99_999_999) / 100_000_000;
#[inline(always)]
fn combine_digits(digits: u64) -> u64 {
let pairs = digits.wrapping_mul(10).wrapping_add(digits >> 8);
let low = (pairs & 0x0000_00ff_0000_00ff).wrapping_mul(0x000f_4240_0000_0064);
let high = ((pairs >> 16) & 0x0000_00ff_0000_00ff).wrapping_mul(0x0000_2710_0000_0001);
u64::from((low.wrapping_add(high) >> 32) as u32)
}
static UNESCAPE: [u8; 256] = {
let mut table = [0; 256];
table[b'"' as usize] = b'"';
table[b'\\' as usize] = b'\\';
table[b'/' as usize] = b'/';
table[b'b' as usize] = b'\x08';
table[b'f' as usize] = b'\x0c';
table[b'n' as usize] = b'\n';
table[b'r' as usize] = b'\r';
table[b't' as usize] = b'\t';
table
};
static POW10_U64: [u64; 20] = {
let mut table = [1; 20];
let mut idx = 1;
while idx < table.len() {
table[idx] = table[idx - 1] * 10;
idx += 1;
}
table
};
static POW10: [f64; 23] = [
1e0, 1e1, 1e2, 1e3, 1e4, 1e5, 1e6, 1e7, 1e8, 1e9, 1e10, 1e11, 1e12, 1e13, 1e14, 1e15, 1e16,
1e17, 1e18, 1e19, 1e20, 1e21, 1e22,
];
static POW5: [(u64, u64); 45] = {
let mut table = [(0, 0); 45];
let mut q = -22i32;
while q <= 22 {
let pow = 5u128.pow(q.unsigned_abs());
let value = if q >= 0 {
pow << pow.leading_zeros()
} else {
let z = 128 - pow.leading_zeros();
let mut quotient = 0u128;
let mut rem = 1u128;
let mut bit = 0;
while bit < z + 127 {
rem <<= 1;
quotient <<= 1;
if rem >= pow {
rem -= pow;
quotient |= 1;
}
bit += 1;
}
quotient + 1
};
table[(q + 22) as usize] = (value as u64, (value >> 64) as u64);
q += 1;
}
table
};
fn eisel_lemire(significand: u64, exponent: i32) -> Option<f64> {
const MANTISSA_BITS: i32 = 52;
const PRECISION_MASK: u64 = u64::MAX >> (MANTISSA_BITS + 3);
let leading_zeros = significand.leading_zeros();
let significand = significand << leading_zeros;
let (pow_lo, pow_hi) = POW5[(exponent + 22) as usize];
let product = u128::from(significand) * u128::from(pow_hi);
let (mut lo, mut hi) = (product as u64, (product >> 64) as u64);
if hi & PRECISION_MASK == PRECISION_MASK {
let carry = ((u128::from(significand) * u128::from(pow_lo)) >> 64) as u64;
lo = lo.wrapping_add(carry);
if carry > lo {
hi += 1;
}
}
let upper_bit = (hi >> 63) as i32;
let shift = upper_bit + 64 - MANTISSA_BITS - 3;
let mut mantissa = hi >> shift;
let mut power2 = ((exponent.wrapping_mul(152_170 + 65536) >> 16) + 63) + upper_bit
- leading_zeros as i32
+ 1023;
if power2 <= 0 {
return None;
}
if lo <= 1 && (-4..=23).contains(&exponent) && mantissa & 3 == 1 && mantissa << shift == hi {
mantissa &= !1;
}
mantissa += mantissa & 1;
mantissa >>= 1;
if mantissa >= 2 << MANTISSA_BITS {
mantissa = 1 << MANTISSA_BITS;
power2 += 1;
}
mantissa &= !(1 << MANTISSA_BITS);
if power2 >= 0x7ff {
return None;
}
Some(f64::from_bits(mantissa | (power2 as u64) << MANTISSA_BITS))
}
#[cold]
fn token_error(offset: usize, msg: &'static str) -> Error {
Error::new(ErrorKind::Unexpected, msg).with_offset(offset)
}
#[inline]
fn is_shortest_repr(digits: u64, frac_len: u32) -> bool {
const MAX: u64 = 1_000_000_000_000_000;
digits < MAX
&& (frac_len == 1 || !digits.is_multiple_of(10))
&& (frac_len <= 4
|| digits == 0
|| (frac_len <= 19 && digits >= POW10_U64[frac_len as usize - 4]))
}
#[inline]
fn emit_number<'i, O: Out<'i>>(
out: &mut O,
number: Number,
input: &[u8],
exact_numbers: bool,
start: usize,
end: usize,
) -> Result<(), Error> {
match number {
Number::U64(val) => out.emit(Event::from(val)),
Number::I64(val) => out.emit(Event::from(val)),
Number::F64(val) => out.emit(Event::from(val)),
Number::Literal(val) if exact_numbers => emit_literal(out, input, val, start, end),
Number::Literal(val) => out.emit(Event::from(val)),
Number::BigInt(val) => emit_big_int(out, val),
Number::U128(val) => out.emit(Atom::Ext(ExtValue::borrowed(&val))),
Number::I128(val) => out.emit(Atom::Ext(ExtValue::borrowed(&val))),
}
}
#[inline(never)]
fn emit_literal<'i, O: Out<'i>>(
out: &mut O,
input: &[u8],
value: f64,
start: usize,
end: usize,
) -> Result<(), Error> {
let text = unsafe { str::from_utf8_unchecked(&input[start..end]) };
let normalized = json_number(text);
let text = normalized.as_deref().unwrap_or(text);
let number = ExactNumber::new(text, value);
out.emit(Atom::Ext(ExtValue::borrowed_value::<ExactNumber>(&number)))
}
fn json_number(text: &str) -> Option<String> {
let (sign, rest) = match text.as_bytes()[0] {
b'+' => ("", &text[1..]),
b'-' => ("-", &text[1..]),
_ => ("", text),
};
let (int, frac) = rest.split_once('.').unwrap_or((rest, ""));
let exp_start = frac.find(['e', 'E']).unwrap_or(frac.len());
let (frac, exp) = frac.split_at(exp_start);
if !text.starts_with('+') && !int.is_empty() && (!frac.is_empty() || !rest.contains('.')) {
return None;
}
let int = if int.is_empty() { "0" } else { int };
let dot = if frac.is_empty() { "" } else { "." };
Some(format!("{sign}{int}{dot}{frac}{exp}"))
}
#[cold]
fn emit_big_int<'i, O: Out<'i>>(out: &mut O, text: &str) -> Result<(), Error> {
if text.starts_with('-') {
let value: i128 = text.parse().unwrap();
out.emit(Atom::Ext(ExtValue::borrowed(&value)))
} else {
let value: u128 = text.parse().unwrap();
out.emit(Atom::Ext(ExtValue::borrowed(&value)))
}
}
fn line_comment_len(bytes: &[u8]) -> Option<usize> {
let mut index = 0;
while index < bytes.len() {
match bytes[index] {
b'\n' | b'\r' => return Some(index + 1),
0xe2 if matches!(bytes.get(index + 1..index + 3), Some([0x80, 0xa8 | 0xa9])) => {
return Some(index + 3);
}
_ => index += 1,
}
}
None
}
fn unicode_whitespace(bytes: &[u8]) -> Option<usize> {
match bytes {
[0xc2, 0xa0, ..] => Some(2),
[0xe1, 0x9a, 0x80, ..]
| [0xe2, 0x80, 0x80..=0x8a | 0xa8 | 0xa9 | 0xaf, ..]
| [0xe2, 0x81, 0x9f, ..]
| [0xe3, 0x80, 0x80, ..]
| [0xef, 0xbb, 0xbf, ..] => Some(3),
_ => None,
}
}
fn is_identifier_char(c: char, first: bool) -> bool {
matches!(c, '$' | '_')
|| unicode_ident::is_xid_start(c)
|| (!first && (unicode_ident::is_xid_continue(c) || matches!(c, '\u{200c}' | '\u{200d}')))
}
#[cold]
fn invalid_identifier() -> Error {
Error::new(ErrorKind::Unexpected, "invalid escape in identifier")
}
#[cold]
fn invalid_escape() -> Error {
Error::new(ErrorKind::Unexpected, "invalid escape in string")
}
#[cold]
fn lone_surrogate() -> Error {
Error::new(
ErrorKind::Unexpected,
"lone surrogate in unicode escape in string",
)
}
#[cold]
fn number_out_of_range() -> Error {
Error::new(ErrorKind::OutOfRange, "number out of range")
}
#[cfg(test)]
mod tests {
use deser_core::de::Recording;
use super::*;
const OPTIONS: Options = Options {
validate_utf8: true,
exact_numbers: true,
};
fn events(value: Recording) -> Vec<Event<'static>> {
value.events().cloned().collect()
}
fn chunk_sizes(len: usize) -> impl Iterator<Item = usize> {
(1..=len).filter(move |&size| !cfg!(miri) || matches!(size, 1 | 3 | 8) || size == len)
}
fn parse_complete(input: &str) -> Result<Vec<Event<'static>>, String> {
let mut out = None::<Recording>;
let mut parser = Parser::default();
{
let mut driver = DeserializeDriver::new(&mut out);
parser
.parse(
input.as_bytes(),
0,
true,
0,
OPTIONS,
&mut Borrowing(&mut driver),
)
.map_err(|err| err.message().to_string())?;
}
Ok(events(out.unwrap()))
}
fn parse_chunked(input: &str, size: usize) -> Result<Vec<Event<'static>>, String> {
let mut out = None::<Recording>;
let mut parser = Parser::default();
{
let mut driver = DeserializeDriver::new(&mut out);
let mut buffer = Vec::new();
let mut base = 0;
let mut rest = input.as_bytes();
loop {
let len = size.min(rest.len());
buffer.extend_from_slice(&rest[..len]);
rest = &rest[len..];
let eof = rest.is_empty();
match parser
.parse(&buffer, 0, eof, base, OPTIONS, &mut Copying(&mut driver))
.map_err(|err| err.message().to_string())?
{
Progress::Done(_) => break,
Progress::NeedMore(consumed) => {
assert!(!eof, "more input needed at the end");
buffer.drain(..consumed);
base += consumed;
}
}
}
}
Ok(events(out.unwrap()))
}
#[test]
fn test_chunks() {
let long = "x".repeat(300);
let inputs = [
r#"{"a": [1, -2, 3.5, 1e10, -0.25e-3, 12345678901234567890123], "b": {"c": null}}"#.to_string(),
r#"["plain", "esc\"aped\\", "\u00e4\ud83d\ude00", "ä", true, false, null, [], {}, [[]]]"#.to_string(),
format!(r#"{{"{long}": "{long}\n{long}", "k\u0041": [{{}}, {{"x": [1]}}]}}"#),
" 42 ".to_string(),
"\"top\"".to_string(),
"123".to_string(),
" [ 1 , 2 ] ".to_string(),
];
for input in inputs {
let expected = parse_complete(&input).unwrap();
for size in chunk_sizes(input.len()) {
assert_eq!(
parse_chunked(&input, size).unwrap(),
expected,
"{input} size {size}"
);
}
}
}
#[test]
fn test_errors_in_chunks() {
for input in [
"[1, 2",
"[1 2]",
"{\"a\" 1}",
"[tru]",
"\"abc",
"{\"a\": \"\\x\"}",
] {
let expected = parse_complete(input).unwrap_err();
for size in chunk_sizes(input.len()) {
assert_eq!(
parse_chunked(input, size).unwrap_err(),
expected,
"{input} size {size}"
);
}
}
}
#[test]
fn test_digit_runs() {
fn parse(text: &str) -> Vec<f64> {
let mut out = None::<Vec<f64>>;
let mut driver = DeserializeDriver::new(&mut out);
Parser::default()
.parse(
text.as_bytes(),
0,
true,
0,
OPTIONS,
&mut Borrowing(&mut driver),
)
.unwrap();
drop(driver);
out.unwrap()
}
let digits = "12345678901234567890123";
let step = if cfg!(miri) { 4 } else { 1 };
for int_len in (1..=digits.len()).step_by(step) {
for frac_len in (0..=digits.len()).step_by(step) {
let mut number = digits[..int_len].to_string();
if frac_len > 0 {
number.push('.');
number.push_str(&digits[digits.len() - frac_len..]);
}
let value: f64 = number.parse().unwrap();
for text in [
format!("[{number}]"),
format!("[{number},-{number}e1 , {number}]"),
format!("[{number} ]"),
] {
let values = parse(&text);
assert_eq!(values[0].to_bits(), value.to_bits(), "{text}");
assert_eq!(values.last().unwrap().to_bits(), value.to_bits(), "{text}");
}
}
}
}
#[test]
fn test_integers() {
fn parse(text: &str) -> Result<u64, String> {
let mut out = None::<u64>;
let mut driver = DeserializeDriver::new(&mut out);
Parser::default()
.parse(
text.as_bytes(),
0,
true,
0,
OPTIONS,
&mut Borrowing(&mut driver),
)
.map_err(|err| err.message().to_string())?;
drop(driver);
Ok(out.unwrap())
}
let mut value = 0u64;
for digit in (1..=20).map(|x| x % 10) {
value = value.wrapping_mul(10).wrapping_add(digit);
let text = value.to_string();
assert_eq!(parse(&text), Ok(value), "{text}");
}
for value in [
u64::MAX,
u64::MAX - 1,
u64::MAX / 10,
99_999_999,
100_000_000,
EIGHT_DIGITS_LIMIT,
EIGHT_DIGITS_LIMIT * 100_000_000 + 99_999_999,
(EIGHT_DIGITS_LIMIT + 1) * 100_000_000,
] {
assert_eq!(parse(&value.to_string()), Ok(value), "{value}");
}
}
#[test]
fn test_floats_are_rounded_correctly() {
fn parse(text: &str) -> Result<f64, String> {
let mut out = None::<f64>;
let mut driver = DeserializeDriver::new(&mut out);
let options = Options {
validate_utf8: false,
exact_numbers: false,
};
Parser::default()
.parse(
text.as_bytes(),
0,
true,
0,
options,
&mut Borrowing(&mut driver),
)
.map_err(|err| err.message().to_string())?;
drop(driver);
Ok(out.unwrap())
}
let check = |text: &str| match text.parse::<f64>() {
Ok(value) if value.is_finite() => {
assert_eq!(parse(text).map(f64::to_bits), Ok(value.to_bits()), "{text}");
}
_ => assert_eq!(parse(text), Err("number out of range".into()), "{text}"),
};
for text in [
"2e-23",
"0.1",
"9007199254740993",
"9007199254740993.0",
"1e23",
"8.98846567431158e307",
"1.7976931348623157e308",
"1.7976931348623159e308",
"2.2250738585072011e-308",
"4.9e-324",
"2.4703282292062328e-324",
"1e-400",
"0e999",
"-0.0e-999",
"123456789012345678901234567890e-10",
"0.000000000000000000000000000001234567890123456789",
] {
check(text);
}
let mut rng: u64 = 0x2545_f491_4f6c_dd1d;
let mut next = |n: u64| {
rng ^= rng << 13;
rng ^= rng >> 7;
rng ^= rng << 17;
rng % n
};
for _ in 0..if cfg!(miri) { 100 } else { 100_000 } {
let mut text = String::new();
if next(2) == 0 {
text.push('-');
}
text.push(char::from(b'1' + next(9) as u8));
for _ in 0..next(25) {
text.push(char::from(b'0' + next(10) as u8));
}
if next(2) == 0 {
text.push('.');
for _ in 0..1 + next(25) {
text.push(char::from(b'0' + next(10) as u8));
}
}
if next(2) == 0 {
text.push_str(&format!("e{}", next(700) as i32 - 350));
}
check(&text);
}
for _ in 0..if cfg!(miri) { 100 } else { 100_000 } {
let digits = 16 + next(4) as usize;
let mut text = String::new();
text.push(char::from(b'1' + next(9) as u8));
for _ in 1..digits {
text.push(char::from(b'0' + next(10) as u8));
}
if next(2) == 0 {
text.insert(1 + next(digits as u64 - 1) as usize, '.');
}
if next(2) == 0 {
text.push_str(&format!("e{}", next(30) as i32 - 15));
}
check(&text);
}
for _ in 0..if cfg!(miri) { 10 } else { 10_000 } {
let m = (1u64 << 52) + next(1 << 52);
check(&format!("{}.5", m));
check(&format!("{}.4999999999", m));
let half = (2 * m + 1) << 10;
for value in [half - 1, half, half + 1] {
check(&format!("{value}e0"));
check(&format!("{}e-1", u128::from(value) * 10));
}
}
}
fn assert_like_json(inputs: &[(&str, &str)]) {
for (input, json) in inputs {
let expected = parse_complete(json).unwrap();
assert_eq!(parse_complete(input), Ok(expected.clone()), "{input}");
for size in chunk_sizes(input.len()) {
assert_eq!(
parse_chunked(input, size),
Ok(expected.clone()),
"{input} size {size}"
);
}
}
}
fn assert_errors(inputs: &[(&str, &str)]) {
for (input, msg) in inputs {
assert_eq!(parse_complete(input).unwrap_err(), *msg, "{input}");
for size in chunk_sizes(input.len()) {
assert_eq!(
parse_chunked(input, size).unwrap_err(),
*msg,
"{input} size {size}"
);
}
}
}
#[test]
fn test_comments() {
assert_like_json(&[
("// c\n1", "1"),
("/* c */ 1 /* c */", "1"),
("1 // c", "1"),
("[1, /* a\n * b */ 2 // c\n, 3]", "[1, 2, 3]"),
("{/**/\"a\"/**/:/**/1/**/}", "{\"a\": 1}"),
(
"[\"// no comment\", \"/* no comment */\"]",
"[\"// no comment\", \"/* no comment */\"]",
),
("/* ä */ [/* 😀 */]", "[]"),
]);
assert_errors(&[
("[1 /* c", "unexpected end of file"),
("[1 / 2]", "expected a comma"),
("[/", "unexpected character"),
("[1 /", "expected a comma"),
("/x", "unexpected character"),
]);
}
#[test]
fn test_trailing_commas() {
assert_like_json(&[
("[1,]", "[1]"),
("[1, 2 , ]", "[1, 2]"),
("{\"a\": 1,}", "{\"a\": 1}"),
("[[1,],{\"a\":[],},]", "[[1],{\"a\":[]}]"),
]);
assert_errors(&[
("[,]", "unexpected comma"),
("[1,,]", "unexpected comma"),
("{,}", "expected map key"),
("{\"a\": 1,,}", "expected map key"),
]);
}
#[test]
fn test_json5() {
assert_like_json(&[
("{a: 1, $b_2: 2, _: 3}", r#"{"a": 1, "$b_2": 2, "_": 3}"#),
(
"{Infinity: 1, null: 2, true: 3}",
r#"{"Infinity": 1, "null": 2, "true": 3}"#,
),
("{ä: 1, a\u{200d}b: 2}", "{\"ä\": 1, \"a\u{200d}b\": 2}"),
(
"{u\u{308}ber: 1, x\u{665}: 2}",
"{\"u\u{308}ber\": 1, \"x\u{665}\": 2}",
),
(
r"{sig\u03A3ma: 1, \u0061b: 2, a\u0062: 3}",
r#"{"sigΣma": 1, "ab": 2, "ab": 3}"#,
),
("{'a': 1}", r#"{"a": 1}"#),
("'a\"b'", r#""a\"b""#),
("\"a'b\"", r#""a'b""#),
(r"'a\'b'", r#""a'b""#),
(r"'\v\0\x41\a\ä'", r#""\u000b\u0000Aaä""#),
("'a\\\nb\\\r\nc\\\rd\\\u{2028}e'", r#""abcde""#),
("'a\tb'", r#""a\tb""#),
(
"[0x1F, 0XfF, -0x10, +1, +1.5, .5, -.5, 5., 5.e1]",
"[31, 255, -16, 1, 1.5, 0.5, -0.5, 5.0, 5e1]",
),
(
"[0xFFFFFFFFFFFFFFFF, -0x8000000000000000]",
"[18446744073709551615, -9223372036854775808]",
),
(
"[0x10000000000000000, -0x8000000000000001]",
"[18446744073709551616, -9223372036854775809]",
),
(
"[0xFFFFFFFFFFFFFFFFFFFFFFFFFFFFFFFF, -0x80000000000000000000000000000000]",
"[340282366920938463463374607431768211455, \
-170141183460469231731687303715884105728]",
),
("\u{feff}[1,\u{a0}2\u{2028}\u{3000}\x0b\x0c]", "[1, 2]"),
("[1, // a\u{2028}2, // b\u{2029}3]", "[1, 2, 3]"),
]);
assert_errors(&[
("{1: 2}", "expected map key"),
("{\u{308}u: 1}", "expected map key"),
("{a b: 1}", "expected colon"),
(r"{\u0031: 1}", "invalid escape in identifier"),
(r"{a\u0020: 1}", "invalid escape in identifier"),
(r"{a\x41: 1}", "invalid escape in identifier"),
(r"{a\uD800: 1}", "invalid escape in identifier"),
(r"{a\u00: 1}", "invalid hex escape"),
("[a]", "unexpected character"),
("'a", "unexpected end of string"),
("'a\nb'", "unexpected character in string"),
(r"'\01'", "invalid escape in string"),
(r"'\1'", "invalid escape in string"),
(r"'\xZ0'", "invalid hex escape"),
("[.]", "expected a digit"),
("[0x]", "expected a hex digit"),
("[Inf]", "unexpected character"),
("\u{2029}x", "unexpected character"),
]);
}
fn pow2(exp: u64) -> f64 {
f64::from_bits((1023 + exp) << 52)
}
#[test]
fn test_json5_hex_floats_are_rounded_correctly() {
let value = |text: &str| match parse_complete(text).unwrap()[..] {
[Event::Atom(Atom::F64(value))] => value,
ref events => panic!("{events:?}"),
};
let half = format!("0x20000000000001{}", "0".repeat(20));
let above = format!("0x20000000000001{}1", "0".repeat(19));
let next = pow2(133) + pow2(81);
assert_eq!(value(&half), pow2(133));
assert_eq!(value(&above), next);
assert_eq!(value(&format!("-{above}")), -next);
}
#[test]
fn test_json5_large_hex() {
let events = parse_complete(&format!("[0x1{}, -0x1{0}]", "0".repeat(32))).unwrap();
assert_eq!(
events[1..3],
[Event::from(pow2(128)), Event::from(-pow2(128))]
);
assert_errors(&[(&format!("0x1{}", "0".repeat(256)), "number out of range")]);
}
#[test]
fn test_line_scan() {
fn lines(input: &str) -> Vec<&str> {
let mut rv = Vec::new();
let mut start = 0;
let mut scan = crate::scan::LineScan::default();
while let Some(end) = scan.find_end(input.as_bytes(), start) {
rv.push(&input[start..end]);
start = end + 1;
}
rv.push(&input[start..]);
rv
}
assert_eq!(lines("1\n2"), ["1", "2"]);
assert_eq!(lines("1 /* a\nb */\n2"), ["1 /* a\nb */", "2"]);
assert_eq!(lines("1 // a\n2"), ["1 // a", "2"]);
assert_eq!(lines("\"/*\"\n2 */"), ["\"/*\"", "2 */"]);
assert_eq!(lines("\"a\n\"b"), ["\"a", "\"b"]);
assert_eq!(lines("1 / 2\n3"), ["1 / 2", "3"]);
assert_eq!(lines("1 /\"a\n\"\n2"), ["1 /\"a", "\"", "2"]);
assert_eq!(lines("'/*'\n2 */"), ["'/*'", "2 */"]);
assert_eq!(
lines("'a\\\nb'\n'c\\\r\nd'\n2"),
["'a\\\nb'", "'c\\\r\nd'", "2"]
);
}
#[test]
fn test_json5_number_text() {
for (text, json) in [
("+1.5", Some("1.5")),
("-.5e3", Some("-0.5e3")),
(".5", Some("0.5")),
("5.", Some("5")),
("5.e1", Some("5e1")),
("+5.E1", Some("5E1")),
("1.5e3", None),
("-1", None),
] {
assert_eq!(json_number(text).as_deref(), json, "{text}");
}
}
#[test]
fn test_json5_non_finite() {
let events = parse_complete("[Infinity, -Infinity, +Infinity, NaN, -NaN]").unwrap();
let floats: Vec<f64> = events
.iter()
.filter_map(|event| match event {
Event::Atom(Atom::F64(value)) => Some(*value),
_ => None,
})
.collect();
assert_eq!(floats.len(), 5);
assert_eq!(
floats[..3],
[f64::INFINITY, f64::NEG_INFINITY, f64::INFINITY]
);
assert!(floats[3].is_nan() && floats[4].is_nan());
}
}