Skip to main content

rich/
json.rs

1//! JSON pretty-printing.
2//!
3//! Port of upstream `rich/json.py`. [`Json`] parses a JSON string and renders it
4//! with 2-space indentation (matching Python's `json.dumps(indent=2)`) and the
5//! default JSON highlight colors.
6//!
7//! Non-ASCII strings render as UTF-8, matching upstream (`rich.json.JSON`
8//! defaults to `ensure_ascii=False`); object keys keep input order, and a
9//! repeated key keeps its first position but its last value — what both
10//! `dict` and serde_json's `preserve_order` do. Number formatting follows
11//! `json.dumps` too: floats are written
12//! with Python's `float.__repr__` ([`crate::pyformat::float_repr`]).
13//! [`JsonOptions`] carries upstream's `indent`, `sort_keys`, `ensure_ascii`,
14//! `allow_nan` and `highlight`; highlighting is upstream's `JSONHighlighter`.
15//!
16//! ## Why the parser is hand-written
17//!
18//! Upstream's parser is Python's `json`, which differs from `serde_json` in two
19//! important ways this module reproduces:
20//!
21//! * `json.loads` accepts (and `json.dumps(allow_nan=True)` emits) the
22//!   non-finite literals `NaN`, `Infinity` and `-Infinity`. `serde_json` has no
23//!   `Value` that can hold them and rejects the documents outright.
24//! * `serde_json` caps nesting at 128 levels, so a 200-deep document — which
25//!   CPython parses without complaint — came back as "invalid JSON".
26//!
27//! Raising a recursion limit only moves the failure to a stack overflow, so
28//! parsing, rendering and *dropping* the tree here are all iterative: nesting
29//! depth costs heap, never stack. String decoding uses `serde_json`; integer
30//! tokens retain their exact digits, finite floats use Python's `repr`, and
31//! overflowing exponents become signed Infinity as in Python.
32//!
33//! Nesting is still bounded, at [`MAX_DEPTH`] levels, where CPython raises
34//! `RecursionError`: `json.loads` recurses in C, and CPython 3.12+ stops just
35//! short of 10 000 levels (3.11 at its 1 000-frame recursion limit). Rendering
36//! indents every line by its depth, so an unbounded document costs quadratic
37//! memory: a 100 000-deep array asked for tens of gigabytes and was killed
38//! rather than reporting an error. The limit accepts every document CPython
39//! 3.12+ accepts; the few levels between its exact (stack-dependent) cutoff
40//! and `MAX_DEPTH` are the only difference.
41
42use std::collections::HashMap;
43
44/// The deepest nesting [`Json::new`] accepts (DIVERGENCES #25). Deeper
45/// documents fail with the error CPython's `json.loads` raises
46/// (`RecursionError`); see the module docs.
47pub const MAX_DEPTH: usize = 10_000;
48
49use crate::console::{Console, ConsoleOptions};
50use crate::errors::{Result, RichError};
51use crate::measure::Measurement;
52use crate::protocol::Renderable;
53use crate::segment::Segment;
54use crate::style::StyleType;
55use crate::text::{Span, Text};
56
57/// A parsed JSON document, rendered with syntax highlighting. Mirrors `rich.json.JSON`.
58///
59/// Like upstream, the document is re-serialized (`json.dumps`) with the
60/// [`JsonOptions`] and highlighted with `JSONHighlighter` when it is built;
61/// [`text`](Json::text) is that highlighted `Text`.
62pub struct Json {
63    value: Node,
64    options: JsonOptions,
65    /// The highlighted document, one `Text` per physical line (upstream
66    /// wraps a `Text` line by line, so this is how it renders). Built on
67    /// first use: a document near [`MAX_DEPTH`] formats to hundreds of
68    /// megabytes of indentation, and one that is only parsed and dropped
69    /// should not pay for that.
70    lines: std::sync::OnceLock<Vec<Text>>,
71    /// See [`Json::no_wrap`].
72    no_wrap: bool,
73    #[cfg(feature = "json-escape-safe")]
74    escape_safe: bool,
75}
76
77/// The keyword arguments of `rich.json.JSON`, for [`Json::with_options`].
78///
79/// `skip_keys` and `check_circular` are accepted for parity but cannot change
80/// the output of a document parsed from a string: its keys are all strings
81/// and it has no cycles. Upstream's `default=` callable only applies to
82/// `JSON.from_data`, which serializes arbitrary (non-JSON) objects.
83#[derive(Clone, Debug, PartialEq, Eq)]
84pub struct JsonOptions {
85    /// `indent=`: the indent string, or `None` for a single line. An integer
86    /// indent `n` is `n` spaces (see [`JsonOptions::indent_spaces`]).
87    pub indent: Option<String>,
88    /// `highlight=`: apply `JSONHighlighter` (default on).
89    pub highlight: bool,
90    /// `skip_keys=` (no effect on a parsed document).
91    pub skip_keys: bool,
92    /// `ensure_ascii=`: escape every non-ASCII character (default off).
93    pub ensure_ascii: bool,
94    /// `check_circular=` (no effect on a parsed document).
95    pub check_circular: bool,
96    /// `allow_nan=`: with it off, `NaN` and the infinities are an error
97    /// (default on).
98    pub allow_nan: bool,
99    /// `sort_keys=`: sort object keys (default off).
100    pub sort_keys: bool,
101}
102
103impl Default for JsonOptions {
104    fn default() -> Self {
105        JsonOptions {
106            indent: Some("  ".to_string()),
107            highlight: true,
108            skip_keys: false,
109            ensure_ascii: false,
110            check_circular: true,
111            allow_nan: true,
112            sort_keys: false,
113        }
114    }
115}
116
117impl JsonOptions {
118    /// An integer `indent=`: `n` spaces, as `json.dumps` turns an int into
119    /// `" " * n` (so zero or a negative number puts each item on its own line
120    /// with no indentation).
121    pub fn indent_spaces(indent: i64) -> Option<String> {
122        Some(" ".repeat(usize::try_from(indent).unwrap_or(0)))
123    }
124}
125
126/// A Python `str` as generalized UTF-8 (WTF-8): UTF-8, except that a lone
127/// surrogate from a `\uD800`-style escape (which Python's `json` accepts)
128/// is kept as its three-byte encoding, as `surrogatepass` writes it. Byte
129/// order is still code point order, so `sort_keys` sorts as Python does.
130type PyStr = Vec<u8>;
131
132/// A parsed JSON value.
133///
134/// Scalars keep the form they will be printed in: numbers are stored already
135/// normalized where needed, strings already decoded (upstream re-encodes them
136/// through `json.dumps`, so `"A"` prints as `"A"`).
137#[derive(Debug)]
138enum Node {
139    Null,
140    Bool(bool),
141    Number(String),
142    /// `NaN`, `Infinity` or `-Infinity`. Python's `json` round-trips these;
143    /// rich's `JSONHighlighter` has no rule that matches them, so they print
144    /// unstyled.
145    NonFinite(&'static str),
146    Str(PyStr),
147    Array(Vec<Node>),
148    Object(Vec<(PyStr, Node)>),
149}
150
151impl Drop for Node {
152    /// Dismantle the tree with an explicit stack.
153    ///
154    /// The compiler's drop glue recurses once per nesting level, so a document
155    /// deep enough to parse (parsing has no depth limit here) would overflow
156    /// the stack on the way out — a crash with no error message at all, which
157    /// is worse than the rejection this module used to hand out.
158    fn drop(&mut self) {
159        let mut pending: Vec<Node> = Vec::new();
160        take_children(self, &mut pending);
161        while let Some(mut node) = pending.pop() {
162            take_children(&mut node, &mut pending);
163            // `node` drops here with its children already moved out, so this
164            // same `drop` runs against an empty container and stops.
165        }
166    }
167}
168
169/// Move a node's children into `out`, leaving the node childless.
170fn take_children(node: &mut Node, out: &mut Vec<Node>) {
171    match node {
172        Node::Array(items) => out.append(items),
173        Node::Object(entries) => out.extend(entries.drain(..).map(|(_, value)| value)),
174        _ => {}
175    }
176}
177
178impl Json {
179    /// Parse `text` as JSON with upstream's default options. Returns an error
180    /// if it is not valid JSON.
181    pub fn new(text: &str) -> Result<Self> {
182        Json::with_options(text, &JsonOptions::default())
183    }
184
185    /// Parse `text` as JSON and format it with `options`. Port of
186    /// `rich.json.JSON(json, indent=…, highlight=…, …)`.
187    pub fn with_options(text: &str, options: &JsonOptions) -> Result<Self> {
188        let value = Parser::new(text).parse_document()?;
189        // `json.dumps(allow_nan=False)` raises here, at construction.
190        if !options.allow_nan {
191            if let Some(literal) = first_non_finite(&value) {
192                return Err(RichError::Json(format!(
193                    "Out of range float values are not JSON compliant: {}",
194                    match literal {
195                        "NaN" => "nan",
196                        "Infinity" => "inf",
197                        _ => "-inf",
198                    }
199                )));
200            }
201        }
202        Ok(Json {
203            value,
204            options: options.clone(),
205            lines: std::sync::OnceLock::new(),
206            no_wrap: false,
207            #[cfg(feature = "json-escape-safe")]
208            escape_safe: false,
209        })
210    }
211
212    /// The formatted document, one highlighted `Text` per line.
213    fn lines(&self) -> &[Text] {
214        self.lines.get_or_init(|| {
215            let dumped = dumps(&self.value, &self.options);
216            let text = if !self.options.highlight {
217                Text::new(dumped.json)
218            } else if dumped.exact {
219                let mut text = Text::new(dumped.json);
220                let mut spans = dumped.spans;
221                spans.extend(dumped.keys);
222                text.set_spans(spans);
223                text
224            } else {
225                json_highlight(dumped.json)
226            };
227            if text.plain().contains('\n') {
228                text.split("\n", false, true)
229            } else {
230                vec![text]
231            }
232        })
233    }
234
235    /// The formatted, highlighted document. Port of the `JSON.text`
236    /// attribute (upstream's `__rich__` returns it).
237    pub fn text(&self) -> Text {
238        match self.lines() {
239            [line] => line.clone(),
240            lines => Text::new("\n").join(lines),
241        }
242    }
243
244    /// Opt in to escape-aware display boundaries (requires `json-escape-safe`).
245    /// Cropping omits partial escapes. Folding preserves escapes that fit the
246    /// width; narrower widths split oversized escapes to avoid losing content.
247    /// The default remains Python rich's ordinary word folding/cropping.
248    #[cfg(feature = "json-escape-safe")]
249    pub fn escape_safe(mut self, enabled: bool) -> Self {
250        self.escape_safe = enabled;
251        self
252    }
253
254    /// Keep `rich.json.JSON`'s `self.text.no_wrap = True`, which **crops** each
255    /// line at the available width instead of wrapping it.
256    ///
257    /// Defaults to `false`, because that is what a *top-level*
258    /// `Console.print(JSON(...))` produces and that is how this renderable is
259    /// normally reached. Upstream's flag really is set, but `Console.print`
260    /// never renders the `Text` it is set on: `_collect_renderables` sees a
261    /// `Text`, buffers it, and `check_text` hands back
262    /// `Text(sep, justify=…, end=…).join(buffered)` — and `Text.join` starts
263    /// from `self.blank_copy()`, i.e. from the *separator's* metadata. The
264    /// separator has `no_wrap=None`, so the copy that actually renders wraps
265    /// with the default `fold` overflow — or with the `no_wrap` and
266    /// `overflow` that `print` was given, which reach it as
267    /// [`ConsoleOptions`] and which this renderable honours while the flag is
268    /// off.
269    ///
270    /// Turn it on whenever the document is **nested** inside another
271    /// renderable — a `Panel`, `Padding`, `Styled`, `Constrain`, or rich-cli's
272    /// `ForceWidth`. Those are `ConsoleRenderable`s, so `_collect_renderables`
273    /// appends them untouched and the inner `Text` reaches
274    /// `Text.__rich_console__` with the flag intact; `Text.wrap` then skips
275    /// `divide_line` and only `truncate`s to the width.
276    ///
277    /// The two are not interchangeable. Wrapping keeps every character;
278    /// cropping discards what does not fit, which for JSON means the printed
279    /// document no longer parses — so the choice has to follow upstream's,
280    /// not taste.
281    #[must_use]
282    pub fn no_wrap(mut self, no_wrap: bool) -> Self {
283        self.no_wrap = no_wrap;
284        self
285    }
286
287    /// The document's lines rendered without wrapping, joined by newlines.
288    #[cfg(feature = "json-escape-safe")]
289    fn render_unwrapped(&self, console: &Console) -> Vec<Segment> {
290        let mut out = Vec::new();
291        for (index, line) in self.lines().iter().enumerate() {
292            if index > 0 {
293                out.push(Segment::line());
294            }
295            out.extend(line.render(console.theme(), console.base_style()));
296        }
297        out
298    }
299}
300
301/// A serialized document, with the spans `JSONHighlighter` would give it.
302struct Dumped {
303    json: String,
304    /// Token spans (`json.brace`, `json.str`, …) in document order.
305    spans: Vec<Span>,
306    /// `json.key` spans, which upstream appends after every token span.
307    keys: Vec<Span>,
308    /// Whether `spans` + `keys` are exactly what the highlighter's regexes
309    /// produce. They are for a whitespace indent and strings whose closing
310    /// quote does not follow a backslash; otherwise the regexes run.
311    exact: bool,
312}
313
314impl Dumped {
315    fn styled(&mut self, token: &str, style: &'static str) {
316        let start = self.json.len();
317        self.json.push_str(token);
318        self.spans.push(Span {
319            start,
320            end: self.json.len(),
321            style: StyleType::Name(style.to_string()),
322        });
323    }
324
325    /// A quoted string. `JSON_STR`'s lazy `".*?(?<!\\)"` ends at the first
326    /// quote not preceded by a backslash, which is this string's own closing
327    /// quote unless the string ends in an escaped backslash.
328    fn string(&mut self, quoted: &str, key: bool) {
329        if quoted.as_bytes().get(quoted.len().wrapping_sub(2)) == Some(&b'\\') {
330            self.exact = false;
331        }
332        let start = self.json.len();
333        self.styled(quoted, "json.str");
334        if key {
335            self.keys.push(Span {
336                start,
337                end: self.json.len(),
338                style: StyleType::Name("json.key".to_string()),
339            });
340        }
341    }
342}
343
344/// Serialize `value` as Python's `json.dumps` does with `options`, iteratively
345/// so nesting depth costs heap rather than stack.
346fn dumps(value: &Node, options: &JsonOptions) -> Dumped {
347    use std::borrow::Cow;
348    let indent = options.indent.as_deref();
349    // `json.dumps` separators: `(', ', ': ')` on one line, `(',', ': ')` with
350    // an indent.
351    let item_separator = if indent.is_some() { "," } else { ", " };
352    let newline_indent = |level: usize| -> Cow<'static, str> {
353        match indent {
354            Some(indent) => Cow::Owned(format!("\n{}", indent.repeat(level))),
355            None => Cow::Borrowed(""),
356        }
357    };
358    let quote_string = |string: &[u8]| -> String { quote_py(string, options.ensure_ascii) };
359
360    /// One entry of the serializer's work list, consumed newest-first.
361    enum Task<'a> {
362        /// Serialize this value, `usize` levels deep.
363        Value(&'a Node, usize),
364        /// Write this unstyled text.
365        Plain(Cow<'a, str>),
366        /// Write a closing brace.
367        Close(&'static str),
368        /// Write an object key and its `": "`.
369        Key(&'a [u8]),
370    }
371
372    let mut dumped = Dumped {
373        json: String::new(),
374        spans: Vec::new(),
375        keys: Vec::new(),
376        // The highlighter's patterns could match inside any other indent.
377        exact: indent.is_none_or(|indent| indent.chars().all(|c| c == ' ' || c == '\t')),
378    };
379    let mut stack = vec![Task::Value(value, 0)];
380    while let Some(task) = stack.pop() {
381        let (node, level) = match task {
382            Task::Plain(text) => {
383                dumped.json.push_str(&text);
384                continue;
385            }
386            Task::Close(brace) => {
387                dumped.styled(brace, "json.brace");
388                continue;
389            }
390            Task::Key(key) => {
391                dumped.string(&quote_string(key), true);
392                dumped.json.push_str(": ");
393                continue;
394            }
395            Task::Value(node, level) => (node, level),
396        };
397        match node {
398            Node::Null => dumped.styled("null", "json.null"),
399            Node::Bool(true) => dumped.styled("true", "json.bool_true"),
400            Node::Bool(false) => dumped.styled("false", "json.bool_false"),
401            Node::Number(number) => dumped.styled(number, "json.number"),
402            // No `JSONHighlighter` pattern matches `NaN` or the infinities.
403            Node::NonFinite(literal) => dumped.json.push_str(literal),
404            Node::Str(string) => dumped.string(&quote_string(string), false),
405            Node::Array(items) => {
406                dumped.styled("[", "json.brace");
407                if items.is_empty() {
408                    dumped.styled("]", "json.brace");
409                    continue;
410                }
411                // Pushed back-to-front, so they pop in document order.
412                stack.push(Task::Close("]"));
413                stack.push(Task::Plain(newline_indent(level)));
414                let last = items.len() - 1;
415                for (index, item) in items.iter().enumerate().rev() {
416                    if index != last {
417                        stack.push(Task::Plain(Cow::Borrowed(item_separator)));
418                    }
419                    stack.push(Task::Value(item, level + 1));
420                    stack.push(Task::Plain(newline_indent(level + 1)));
421                }
422            }
423            Node::Object(entries) => {
424                dumped.styled("{", "json.brace");
425                if entries.is_empty() {
426                    dumped.styled("}", "json.brace");
427                    continue;
428                }
429                let mut order: Vec<&(PyStr, Node)> = entries.iter().collect();
430                if options.sort_keys {
431                    // `sorted(dct.items())`: keys are unique, so this is by key,
432                    // in code point order (which UTF-8 byte order preserves).
433                    order.sort_by(|a, b| a.0.cmp(&b.0));
434                }
435                stack.push(Task::Close("}"));
436                stack.push(Task::Plain(newline_indent(level)));
437                let last = order.len() - 1;
438                for (index, (key, item)) in order.into_iter().enumerate().rev() {
439                    if index != last {
440                        stack.push(Task::Plain(Cow::Borrowed(item_separator)));
441                    }
442                    stack.push(Task::Value(item, level + 1));
443                    stack.push(Task::Key(key));
444                    stack.push(Task::Plain(newline_indent(level + 1)));
445                }
446            }
447        }
448    }
449    dumped
450}
451
452/// The first non-finite literal in the document, if any, found iteratively.
453fn first_non_finite(value: &Node) -> Option<&'static str> {
454    let mut pending = vec![value];
455    while let Some(node) = pending.pop() {
456        match node {
457            Node::NonFinite(literal) => return Some(literal),
458            Node::Array(items) => pending.extend(items.iter()),
459            Node::Object(entries) => pending.extend(entries.iter().map(|(_, value)| value)),
460            _ => {}
461        }
462    }
463    None
464}
465
466/// `ensure_ascii=True`: re-escape every character outside `' '..='~'` in an
467/// already-quoted JSON string as `\uXXXX` (a surrogate pair above the BMP),
468/// as `py_encode_basestring_ascii` does. The quoted form's own escapes are
469/// ASCII already.
470fn ascii_escape(quoted: &str) -> String {
471    if quoted.bytes().all(|b| (b' '..=b'~').contains(&b)) {
472        return quoted.to_string();
473    }
474    let mut out = String::with_capacity(quoted.len() + 16);
475    for c in quoted.chars() {
476        if (' '..='~').contains(&c) {
477            out.push(c);
478            continue;
479        }
480        let mut units = [0u16; 2];
481        for unit in c.encode_utf16(&mut units) {
482            out.push_str(&format!("\\u{unit:04x}"));
483        }
484    }
485    out
486}
487
488/// Upstream's `JSONHighlighter` regexes: the combined token pattern, and the
489/// string pattern its key pass re-scans with.
490fn json_patterns() -> &'static (fancy_regex::Regex, fancy_regex::Regex) {
491    static PATTERNS: std::sync::OnceLock<(fancy_regex::Regex, fancy_regex::Regex)> =
492        std::sync::OnceLock::new();
493    PATTERNS.get_or_init(|| {
494        const JSON_STR: &str = r#"(?<![\\\w])(?P<str>b?".*?(?<!\\)")"#;
495        let combined = [
496            r"(?P<brace>[\{\[\(\)\]\}])",
497            r"\b(?P<bool_true>true)\b|\b(?P<bool_false>false)\b|\b(?P<null>null)\b",
498            r"(?P<number>(?<!\w)\-?[0-9]+\.?[0-9]*(e[\-\+]?\d+?)?\b|0x[0-9a-fA-F]*)",
499            JSON_STR,
500        ]
501        .join("|");
502        (
503            fancy_regex::Regex::new(&combined).expect("valid JSON highlighter pattern"),
504            fancy_regex::Regex::new(JSON_STR).expect("valid JSON string pattern"),
505        )
506    })
507}
508
509/// Highlight a formatted document. Port of `JSONHighlighter.highlight`: the
510/// `json.`-prefixed token spans, then a `json.key` span on every string that
511/// is followed (past whitespace) by a colon.
512fn json_highlight(json: String) -> Text {
513    let (combined, string) = json_patterns();
514    let mut text = Text::new(json);
515    text.highlight_with_regex(combined, None, "json.");
516    let plain = text.plain();
517    let mut keys = Vec::new();
518    for found in string.find_iter(plain) {
519        let Ok(found) = found else { break };
520        for &byte in &plain.as_bytes()[found.end()..] {
521            match byte {
522                b':' => keys.push((found.start(), found.end())),
523                b' ' | b'\n' | b'\r' | b'\t' => continue,
524                _ => {}
525            }
526            break;
527        }
528    }
529    for (start, end) in keys {
530        text.push_span(Span {
531            start,
532            end,
533            style: StyleType::Name("json.key".to_string()),
534        });
535    }
536    text
537}
538
539impl Renderable for Json {
540    /// Upstream `JSON.__rich__` returns its highlighted `Text`, so measuring a
541    /// `JSON` is `Text.__rich_measure__` over the formatted document.
542    fn measure(&self, _console: &Console, _options: &ConsoleOptions) -> Measurement {
543        let mut minimum = 0;
544        let mut maximum = 0;
545        for line in self.lines() {
546            let (line_minimum, line_maximum) = line.measurement();
547            minimum = minimum.max(line_minimum);
548            maximum = maximum.max(line_maximum);
549        }
550        Measurement::new(minimum, maximum)
551    }
552
553    /// `Text.__rich_console__` of the document: each line wraps (or, with
554    /// `no_wrap`, is truncated) to the width with the options' `justify` and
555    /// `overflow` — the document's own `overflow` is `None`.
556    fn rich_render(&self, console: &Console, options: &ConsoleOptions) -> Vec<Segment> {
557        #[cfg(feature = "json-escape-safe")]
558        if self.escape_safe {
559            let segments = self.render_unwrapped(console);
560            return escape_safe_lines(&segments, options.max_width, self.no_wrap);
561        }
562        let no_wrap = self.no_wrap || options.no_wrap.unwrap_or(false);
563        let overflow = options.overflow.unwrap_or_default();
564        let tab_size = match console.tab_size() {
565            0 => crate::text::DEFAULT_TAB_SIZE,
566            tab_size => tab_size,
567        };
568        let mut out = Vec::new();
569        let mut first = true;
570        for line in self.lines() {
571            for rendered in line.render_lines_wrapped_tabs(
572                console.theme(),
573                console.base_style(),
574                Some(options.max_width),
575                options.justify,
576                overflow,
577                no_wrap,
578                tab_size,
579            ) {
580                if !first {
581                    out.push(Segment::line());
582                }
583                first = false;
584                out.extend(rendered);
585            }
586        }
587        out
588    }
589}
590
591/// Tokenize each physical line into JSON escapes and ordinary graphemes, then
592/// choose every boundary from the space actually remaining. No stale absolute
593/// wrap points survive an adjusted escape boundary (#98).
594#[cfg(feature = "json-escape-safe")]
595fn escape_safe_lines(segments: &[Segment], width: usize, crop: bool) -> Vec<Segment> {
596    if width == 0 {
597        return Vec::new();
598    }
599    let lines = Segment::split_lines(segments);
600    let last = lines.len().saturating_sub(1);
601    let mut out = Vec::new();
602    for (line_index, line) in lines.into_iter().enumerate() {
603        let plain: String = line
604            .iter()
605            .filter(|s| !s.control)
606            .map(|s| s.text.as_str())
607            .collect();
608        // Keep upstream's exact behavior when there is no escape to protect.
609        if !plain.contains('\\') {
610            out.extend(if crop {
611                Segment::crop_lines(&line, width)
612            } else {
613                Segment::fold_lines_words(&line, width)
614            });
615        } else {
616            let (spans, _) = crate::cells::split_graphemes(&plain);
617            let mut atoms = Vec::new();
618            let mut index = 0;
619            while index < spans.len() {
620                let (start, mut end, mut cells) = spans[index];
621                if plain.as_bytes()[start] == b'\\' {
622                    let escape_end = start
623                        + if plain.as_bytes().get(start + 1) == Some(&b'u') {
624                            6
625                        } else {
626                            2
627                        };
628                    while end < escape_end && index + 1 < spans.len() {
629                        index += 1;
630                        end = spans[index].1;
631                        cells += spans[index].2;
632                    }
633                    if !crop && cells > width {
634                        // An atom wider than the whole console cannot both fit
635                        // and stay atomic. Split its ASCII spelling rather than
636                        // overrun into Console's final crop and lose characters.
637                        for offset in start..escape_end {
638                            atoms.push((offset, offset + 1, 1));
639                        }
640                        if end > escape_end {
641                            atoms.push((escape_end, end, 0));
642                        }
643                        index += 1;
644                        continue;
645                    }
646                }
647                atoms.push((start, end, cells));
648                index += 1;
649            }
650            let mut breaks = Vec::new();
651            let mut cells = 0;
652            let mut stop = plain.len();
653            for (start, _, atom_width) in atoms {
654                if cells + atom_width > width {
655                    if crop {
656                        stop = start;
657                        break;
658                    }
659                    if cells > 0 {
660                        breaks.push(start);
661                        cells = 0;
662                    }
663                }
664                cells += atom_width;
665            }
666            let mut position = 0;
667            let mut next = 0;
668            for segment in line {
669                if segment.control {
670                    out.push(segment);
671                    continue;
672                }
673                let mut buffer = String::new();
674                for ch in segment.text.chars() {
675                    if position >= stop {
676                        break;
677                    }
678                    if breaks.get(next) == Some(&position) {
679                        if !buffer.is_empty() {
680                            out.push(Segment::new(
681                                std::mem::take(&mut buffer),
682                                segment.style.clone(),
683                            ));
684                        }
685                        out.push(Segment::line());
686                        next += 1;
687                    }
688                    buffer.push(ch);
689                    position += ch.len_utf8();
690                }
691                if !buffer.is_empty() {
692                    out.push(Segment::new(buffer, segment.style));
693                }
694            }
695        }
696        if line_index != last {
697            out.push(Segment::line());
698        }
699    }
700    out
701}
702
703/// Serialize a string as a JSON string literal (quoted + escaped).
704/// `json.dumps` of a [`PyStr`]. A lone surrogate is `\udXXX` under
705/// `ensure_ascii`; otherwise Python emits the surrogate itself, which a Rust
706/// string cannot hold, so it becomes U+FFFD (see docs/DIVERGENCES.md).
707fn quote_py(string: &[u8], ensure_ascii: bool) -> String {
708    let escape = |run: &str| -> String {
709        let quoted = quote(run);
710        if ensure_ascii {
711            ascii_escape(&quoted)
712        } else {
713            quoted
714        }
715    };
716    if let Ok(run) = std::str::from_utf8(string) {
717        return escape(run);
718    }
719    let mut out = String::from("\"");
720    let mut rest = string;
721    while !rest.is_empty() {
722        let valid = match std::str::from_utf8(rest) {
723            Ok(_) => rest.len(),
724            Err(error) => error.valid_up_to(),
725        };
726        let (run, tail) = rest.split_at(valid);
727        let quoted = escape(std::str::from_utf8(run).unwrap_or_default());
728        out.push_str(&quoted[1..quoted.len() - 1]);
729        rest = tail;
730        if let [0xED, b1 @ 0xA0..=0xBF, b2, tail @ ..] = rest {
731            let unit = 0xD000 | (u32::from(b1 & 0x3F) << 6) | u32::from(b2 & 0x3F);
732            if ensure_ascii {
733                out.push_str(&format!("\\u{unit:04x}"));
734            } else {
735                out.push(char::REPLACEMENT_CHARACTER);
736            }
737            rest = tail;
738        } else if !rest.is_empty() {
739            // Unreachable for parser output; never loop forever.
740            out.push(char::REPLACEMENT_CHARACTER);
741            rest = &rest[1..];
742        }
743    }
744    out.push('"');
745    out
746}
747
748/// Decode a JSON string token as Python's `json` does, keeping unpaired
749/// `\uD800`-`\uDFFF` escapes as surrogates in the result. `None` when the
750/// token is invalid for another reason (the caller reports `serde_json`'s
751/// error for it).
752fn decode_with_surrogates(token: &str) -> Option<PyStr> {
753    let inner = token.strip_prefix('"')?.strip_suffix('"')?;
754    let mut out = Vec::with_capacity(inner.len());
755    let mut chars = inner.chars().peekable();
756    let hex4 = |chars: &mut std::iter::Peekable<std::str::Chars<'_>>| -> Option<u32> {
757        let mut value = 0;
758        for _ in 0..4 {
759            value = value * 16 + chars.next()?.to_digit(16)?;
760        }
761        Some(value)
762    };
763    while let Some(c) = chars.next() {
764        if c < ' ' {
765            return None;
766        }
767        if c != '\\' {
768            let mut buf = [0u8; 4];
769            out.extend_from_slice(c.encode_utf8(&mut buf).as_bytes());
770            continue;
771        }
772        let decoded = match chars.next()? {
773            '"' => '"',
774            '\\' => '\\',
775            '/' => '/',
776            'b' => '\u{8}',
777            'f' => '\u{c}',
778            'n' => '\n',
779            'r' => '\r',
780            't' => '\t',
781            'u' => {
782                let mut unit = hex4(&mut chars)?;
783                if (0xD800..0xDC00).contains(&unit) {
784                    let mut lookahead = chars.clone();
785                    if lookahead.next() == Some('\\') && lookahead.next() == Some('u') {
786                        if let Some(low @ 0xDC00..=0xDFFF) = hex4(&mut lookahead) {
787                            unit = 0x10000 + ((unit - 0xD800) << 10) + (low - 0xDC00);
788                            chars = lookahead;
789                        }
790                    }
791                }
792                match char::from_u32(unit) {
793                    Some(c) => c,
794                    None => {
795                        // A lone surrogate: its generalized UTF-8 encoding.
796                        out.push(0xE0 | (unit >> 12) as u8);
797                        out.push(0x80 | ((unit >> 6) & 0x3F) as u8);
798                        out.push(0x80 | (unit & 0x3F) as u8);
799                        continue;
800                    }
801                }
802            }
803            _ => return None,
804        };
805        let mut buf = [0u8; 4];
806        out.extend_from_slice(decoded.encode_utf8(&mut buf).as_bytes());
807    }
808    Some(out)
809}
810
811fn quote(string: &str) -> String {
812    serde_json::to_string(string).unwrap_or_else(|_| format!("{string:?}"))
813}
814
815/// A container being filled in, held on the parser's explicit stack.
816enum Frame {
817    Array(Vec<Node>),
818    Object {
819        entries: Vec<(PyStr, Node)>,
820        /// Key -> position in `entries`, so a repeated key overwrites in place
821        /// (`{"a": 1, "a": 2}` is one entry) without an O(n^2) rescan. Dropped
822        /// with the frame, so only the objects on the current path pay for it.
823        seen: HashMap<PyStr, usize>,
824        /// The key whose value is currently being parsed.
825        key: PyStr,
826    },
827}
828
829/// A non-recursive JSON reader. Structure is walked with an explicit stack;
830/// scalar tokens are handed to `serde_json` for decoding so escapes, number
831/// formats and their rejections stay identical to the rest of the workspace.
832struct Parser<'a> {
833    src: &'a str,
834    bytes: &'a [u8],
835    pos: usize,
836}
837
838impl<'a> Parser<'a> {
839    fn new(src: &'a str) -> Self {
840        Parser {
841            src,
842            bytes: src.as_bytes(),
843            pos: 0,
844        }
845    }
846
847    fn parse_document(&mut self) -> Result<Node> {
848        let value = self.parse_value()?;
849        self.skip_whitespace();
850        if self.pos != self.bytes.len() {
851            return Err(self.error("trailing characters"));
852        }
853        Ok(value)
854    }
855
856    /// Parse one value, descending into containers with an explicit stack.
857    fn parse_value(&mut self) -> Result<Node> {
858        let mut stack: Vec<Frame> = Vec::new();
859        let mut node: Node;
860
861        'descend: loop {
862            self.skip_whitespace();
863            // CPython's scanner enters a recursive call per container, empty
864            // ones included, and raises once the recursion limit is reached.
865            if let Some(kind @ (b'[' | b'{')) = self.peek() {
866                if stack.len() >= MAX_DEPTH {
867                    let what = if kind == b'[' { "array" } else { "object" };
868                    return Err(self.error(&format!(
869                        "maximum recursion depth exceeded while decoding a JSON {what} \
870                         from a unicode string"
871                    )));
872                }
873            }
874            match self.peek() {
875                Some(b'[') => {
876                    self.pos += 1;
877                    self.skip_whitespace();
878                    if self.peek() == Some(b']') {
879                        self.pos += 1;
880                        node = Node::Array(Vec::new());
881                    } else {
882                        stack.push(Frame::Array(Vec::new()));
883                        continue 'descend;
884                    }
885                }
886                Some(b'{') => {
887                    self.pos += 1;
888                    self.skip_whitespace();
889                    if self.peek() == Some(b'}') {
890                        self.pos += 1;
891                        node = Node::Object(Vec::new());
892                    } else {
893                        let key = self.parse_key()?;
894                        stack.push(Frame::Object {
895                            entries: Vec::new(),
896                            seen: HashMap::new(),
897                            key,
898                        });
899                        continue 'descend;
900                    }
901                }
902                _ => node = self.parse_scalar()?,
903            }
904
905            // `node` is finished: hand it to its parent, then close as many
906            // containers as end here.
907            loop {
908                let Some(frame) = stack.last_mut() else {
909                    return Ok(node);
910                };
911                let closing = match frame {
912                    Frame::Array(items) => {
913                        items.push(node);
914                        b']'
915                    }
916                    Frame::Object { entries, seen, key } => {
917                        let key = std::mem::take(key);
918                        match seen.get(&key) {
919                            Some(&at) => entries[at].1 = node,
920                            None => {
921                                seen.insert(key.clone(), entries.len());
922                                entries.push((key, node));
923                            }
924                        }
925                        b'}'
926                    }
927                };
928                self.skip_whitespace();
929                match self.peek() {
930                    Some(b',') => {
931                        self.pos += 1;
932                        if closing == b'}' {
933                            let next_key = self.parse_key()?;
934                            if let Some(Frame::Object { key, .. }) = stack.last_mut() {
935                                *key = next_key;
936                            }
937                        }
938                        continue 'descend;
939                    }
940                    Some(byte) if byte == closing => {
941                        self.pos += 1;
942                        node = match stack.pop() {
943                            Some(Frame::Array(items)) => Node::Array(items),
944                            Some(Frame::Object { entries, .. }) => Node::Object(entries),
945                            None => unreachable!("the frame was just borrowed"),
946                        };
947                    }
948                    _ if closing == b']' => return Err(self.error("expected `,` or `]`")),
949                    _ => return Err(self.error("expected `,` or `}`")),
950                }
951            }
952        }
953    }
954
955    /// Parse `"key" :`, leaving the parser on the value.
956    fn parse_key(&mut self) -> Result<PyStr> {
957        self.skip_whitespace();
958        if self.peek() != Some(b'"') {
959            return Err(self.error("key must be a string"));
960        }
961        let key = self.parse_string()?;
962        self.skip_whitespace();
963        if self.peek() != Some(b':') {
964            return Err(self.error("expected `:`"));
965        }
966        self.pos += 1;
967        Ok(key)
968    }
969
970    fn parse_scalar(&mut self) -> Result<Node> {
971        match self.peek() {
972            Some(b'"') => Ok(Node::Str(self.parse_string()?)),
973            Some(b't') => {
974                self.expect_literal("true")?;
975                Ok(Node::Bool(true))
976            }
977            Some(b'f') => {
978                self.expect_literal("false")?;
979                Ok(Node::Bool(false))
980            }
981            Some(b'n') => {
982                self.expect_literal("null")?;
983                Ok(Node::Null)
984            }
985            // Python's json emits and accepts these three (`allow_nan=True` is
986            // the default both ways), so upstream renders documents containing
987            // them instead of rejecting the file.
988            Some(b'N') => {
989                self.expect_literal("NaN")?;
990                Ok(Node::NonFinite("NaN"))
991            }
992            Some(b'I') => {
993                self.expect_literal("Infinity")?;
994                Ok(Node::NonFinite("Infinity"))
995            }
996            Some(b'-') if self.src[self.pos..].starts_with("-Infinity") => {
997                self.pos += "-Infinity".len();
998                Ok(Node::NonFinite("-Infinity"))
999            }
1000            Some(b'-' | b'0'..=b'9') => self.parse_number(),
1001            Some(_) => Err(self.error("expected value")),
1002            None => Err(self.error("EOF while parsing a value")),
1003        }
1004    }
1005
1006    /// Read a string token and decode it with `serde_json`, so escapes, lone
1007    /// surrogates and raw control characters behave exactly as before.
1008    fn parse_string(&mut self) -> Result<PyStr> {
1009        let start = self.pos;
1010        let mut end = self.pos + 1;
1011        loop {
1012            match self.bytes.get(end) {
1013                None => return Err(self.error_at(self.bytes.len(), "EOF while parsing a string")),
1014                Some(b'"') => {
1015                    end += 1;
1016                    break;
1017                }
1018                Some(b'\\') => {
1019                    end += 1;
1020                    // Step over the escaped character whole. A multi-byte
1021                    // character after a backslash is invalid JSON, but `end`
1022                    // must still land on a UTF-8 boundary or slicing panics
1023                    // before `serde_json` gets to reject it.
1024                    match self.src[end..].chars().next() {
1025                        Some(ch) => end += ch.len_utf8(),
1026                        None => {
1027                            return Err(
1028                                self.error_at(self.bytes.len(), "EOF while parsing a string")
1029                            )
1030                        }
1031                    }
1032                }
1033                // Continuation bytes are never `"` or `\`, so scanning byte by
1034                // byte cannot mistake one for a delimiter.
1035                Some(_) => end += 1,
1036            }
1037        }
1038        let token = &self.src[start..end];
1039        let decoded = match serde_json::from_str::<String>(token) {
1040            Ok(decoded) => decoded.into_bytes(),
1041            // `serde_json` refuses lone surrogates, which Python's `json`
1042            // accepts; anything else it refuses stays an error.
1043            Err(error) => decode_with_surrogates(token)
1044                .ok_or_else(|| self.error_at(start, &describe(&error)))?,
1045        };
1046        self.pos = end;
1047        Ok(decoded)
1048    }
1049
1050    /// Validate JSON's number grammar before decoding, preserving arbitrary-size
1051    /// integers and Python's float overflow to Infinity (#74).
1052    fn parse_number(&mut self) -> Result<Node> {
1053        let start = self.pos;
1054        let mut end = start;
1055        while matches!(
1056            self.bytes.get(end),
1057            Some(b'-' | b'+' | b'.' | b'e' | b'E' | b'0'..=b'9')
1058        ) {
1059            end += 1;
1060        }
1061        let token = &self.src[start..end];
1062        let digits = token.as_bytes();
1063        let mut i = usize::from(digits.first() == Some(&b'-'));
1064        if digits.get(i) == Some(&b'0') {
1065            i += 1;
1066        } else {
1067            let first = i;
1068            while digits.get(i).is_some_and(u8::is_ascii_digit) {
1069                i += 1;
1070            }
1071            if i == first {
1072                return Err(self.error_at(start, "invalid number"));
1073            }
1074        }
1075        let mut floating = false;
1076        if digits.get(i) == Some(&b'.') {
1077            floating = true;
1078            i += 1;
1079            let first = i;
1080            while digits.get(i).is_some_and(u8::is_ascii_digit) {
1081                i += 1;
1082            }
1083            if i == first {
1084                return Err(self.error_at(start, "invalid number"));
1085            }
1086        }
1087        if matches!(digits.get(i), Some(b'e' | b'E')) {
1088            floating = true;
1089            i += 1;
1090            if matches!(digits.get(i), Some(b'+' | b'-')) {
1091                i += 1;
1092            }
1093            let first = i;
1094            while digits.get(i).is_some_and(u8::is_ascii_digit) {
1095                i += 1;
1096            }
1097            if i == first {
1098                return Err(self.error_at(start, "invalid number"));
1099            }
1100        }
1101        if i != digits.len() {
1102            return Err(self.error_at(start, "invalid number"));
1103        }
1104        self.pos = end;
1105        if !floating {
1106            return Ok(Node::Number(
1107                if token == "-0" { "0" } else { token }.to_string(),
1108            ));
1109        }
1110        let value: f64 = token
1111            .parse()
1112            .map_err(|_| self.error_at(start, "invalid number"))?;
1113        if value.is_infinite() {
1114            return Ok(Node::NonFinite(if value.is_sign_negative() {
1115                "-Infinity"
1116            } else {
1117                "Infinity"
1118            }));
1119        }
1120        // `json.dumps` writes a float with `float.__repr__`.
1121        Ok(Node::Number(crate::pyformat::float_repr(value)))
1122    }
1123
1124    fn expect_literal(&mut self, literal: &str) -> Result<()> {
1125        if self.src[self.pos..].starts_with(literal) {
1126            self.pos += literal.len();
1127            Ok(())
1128        } else {
1129            Err(self.error("expected value"))
1130        }
1131    }
1132
1133    fn peek(&self) -> Option<u8> {
1134        self.bytes.get(self.pos).copied()
1135    }
1136
1137    fn skip_whitespace(&mut self) {
1138        while matches!(self.peek(), Some(b' ' | b'\t' | b'\n' | b'\r')) {
1139            self.pos += 1;
1140        }
1141    }
1142
1143    fn error(&self, message: &str) -> RichError {
1144        self.error_at(self.pos, message)
1145    }
1146
1147    fn error_at(&self, pos: usize, message: &str) -> RichError {
1148        let (line, column) = self.line_column(pos);
1149        RichError::Json(format!("{message} at line {line} column {column}"))
1150    }
1151
1152    fn line_column(&self, pos: usize) -> (usize, usize) {
1153        let mut pos = pos.min(self.src.len());
1154        while !self.src.is_char_boundary(pos) {
1155            pos -= 1;
1156        }
1157        let before = &self.src[..pos];
1158        let line = 1 + before.matches('\n').count();
1159        let column = before
1160            .rsplit('\n')
1161            .next()
1162            .map_or(0, |tail| tail.chars().count())
1163            + 1;
1164        (line, column)
1165    }
1166}
1167
1168/// `serde_json`'s message without its own `at line … column …` suffix, which
1169/// counts from the start of the token slice rather than the document.
1170fn describe(error: &serde_json::Error) -> String {
1171    let text = error.to_string();
1172    match text.find(" at line ") {
1173        Some(at) => text[..at].to_string(),
1174        None => text,
1175    }
1176}
1177
1178#[cfg(test)]
1179mod tests {
1180    use super::*;
1181    use crate::color::ColorSystem;
1182
1183    /// The spans recorded while serializing are exactly the ones upstream's
1184    /// `JSONHighlighter` regexes find, wherever `dumps` claims they are.
1185    #[test]
1186    fn recorded_spans_match_the_highlighter_regexes() {
1187        let documents = [
1188            r#"{"name": "Alice", "age": 30, "admin": true, "tags": ["a", "b"], "meta": null}"#,
1189            r#"[1e20, -0.0, 1.5e-7, 12345678901234567890, false, [], {}, [[]], {"a": {}}]"#,
1190            r#"{"true": "false", "x:y": "a\"b", "q\"": ": null", "n": [NaN, Infinity, -Infinity]}"#,
1191            r#"{"caf\u00e9": "\u2764 \ud83d\ude00", "ctl": "\u0001\t\n", "": ""}"#,
1192            r#"["[{(1)}]", "0x1F", "b\"x", "\\\"", "tail\\x"]"#,
1193        ];
1194        let indents = [None, Some("  "), Some("\t"), Some("")];
1195        for document in documents {
1196            for indent in indents {
1197                for ensure_ascii in [false, true] {
1198                    let options = JsonOptions {
1199                        indent: indent.map(str::to_string),
1200                        ensure_ascii,
1201                        sort_keys: true,
1202                        ..JsonOptions::default()
1203                    };
1204                    let value = Parser::new(document).parse_document().unwrap();
1205                    let dumped = dumps(&value, &options);
1206                    assert!(dumped.exact, "{document} / {indent:?}");
1207                    let mut recorded = dumped.spans.clone();
1208                    recorded.extend(dumped.keys.clone());
1209                    let expected = json_highlight(dumped.json.clone());
1210                    assert_eq!(recorded, expected.spans(), "{document} / {indent:?}");
1211                }
1212            }
1213        }
1214        // A string ending in a backslash defeats `JSON_STR`, so the regexes run.
1215        let value = Parser::new(r#"{"k": "a\\"}"#).parse_document().unwrap();
1216        assert!(!dumps(&value, &JsonOptions::default()).exact);
1217        let value = Parser::new("[1]").parse_document().unwrap();
1218        let options = JsonOptions {
1219            indent: Some("1".to_string()),
1220            ..JsonOptions::default()
1221        };
1222        assert!(!dumps(&value, &options).exact);
1223    }
1224
1225    fn render(text: &str) -> String {
1226        let console = Console::builder()
1227            .force_terminal(true)
1228            .color_system(Some(ColorSystem::Truecolor))
1229            .width(40)
1230            .build();
1231        console.render_to_string(&Json::new(text).unwrap())
1232    }
1233
1234    fn render_plain(text: &str, width: usize) -> String {
1235        let console = Console::builder().width(width).color_system(None).build();
1236        console.render_to_string(&Json::new(text).expect("valid json"))
1237    }
1238
1239    #[test]
1240    fn empty_collections_stay_inline() {
1241        assert_eq!(render("{}"), "\x1b[1m{\x1b[0m\x1b[1m}\x1b[0m");
1242        assert_eq!(render("[]"), "\x1b[1m[\x1b[0m\x1b[1m]\x1b[0m");
1243    }
1244
1245    #[test]
1246    fn object_with_scalars() {
1247        assert_eq!(
1248            render(r#"{"ok": false}"#),
1249            "\x1b[1m{\x1b[0m\n  \x1b[1;34m\"ok\"\x1b[0m: \x1b[3;91mfalse\x1b[0m\n\x1b[1m}\x1b[0m"
1250        );
1251    }
1252
1253    #[test]
1254    fn invalid_json_errors() {
1255        assert!(Json::new("{not json}").is_err());
1256    }
1257
1258    #[test]
1259    fn non_ascii_stays_utf8_in_input_order() {
1260        // Upstream's JSON defaults to ensure_ascii=False, so accented/symbol
1261        // characters render as UTF-8 (not \uXXXX), and keys keep input order.
1262        // (Byte-parity is guaranteed by the `json_unicode` golden.)
1263        let out = render("{\"name\": \"caf\u{e9}\", \"emoji\": \"\u{2764}\"}");
1264        assert!(out.contains("caf\u{e9}"), "café stays UTF-8: {out:?}");
1265        assert!(out.contains('\u{2764}'), "heart stays UTF-8");
1266        let name_at = out.find("name").expect("name key present");
1267        let emoji_at = out.find("emoji").expect("emoji key present");
1268        assert!(name_at < emoji_at, "keys keep input order");
1269    }
1270
1271    #[test]
1272    fn a_long_value_is_wrapped_rather_than_cropped() {
1273        // A long string value used to be cut mid-token, so the printed document
1274        // was missing data -- and, for JSON, no longer parseable -- at exit 0.
1275        let payload = format!("{{\"k\": \"{}\"}}", "y".repeat(120));
1276        let out = render_plain(&payload, 40);
1277        assert_eq!(
1278            out.matches('y').count(),
1279            120,
1280            "characters were dropped:
1281{out}"
1282        );
1283    }
1284
1285    /// Upstream wraps the JSON at word boundaries: `JSON.text` asks for
1286    /// `no_wrap`, but `Console.print` re-joins it through `Text(sep).join(...)`
1287    /// and the joined copy carries neither the flag nor an overflow, so the
1288    /// default `fold` word wrap applies. Character folding split words
1289    /// (`over t` / `he lazy`), which no upstream output ever shows.
1290    ///
1291    /// Captured from rich 15.0.0:
1292    /// `Console(width=40).print(JSON(...))`.
1293    #[test]
1294    fn wrapping_breaks_at_word_boundaries() {
1295        let payload = r#"{"k": "the quick brown fox jumps over the lazy dog and keeps running for a very long time indeed"}"#;
1296        assert_eq!(
1297            render_plain(payload, 40),
1298            "{\n  \"k\": \"the quick brown fox jumps over \n\
1299             the lazy dog and keeps running for a \n\
1300             very long time indeed\"\n}"
1301        );
1302    }
1303
1304    #[cfg(feature = "json-escape-safe")]
1305    #[test]
1306    fn adjusted_escape_boundaries_preserve_the_remaining_payload() {
1307        let input = format!(r#"{{"v":"aa\u0001{}"}}"#, "b".repeat(40));
1308        for width in 6..=20 {
1309            let output = Console::builder()
1310                .width(width)
1311                .force_terminal(false)
1312                .build()
1313                .render_to_string(&Json::new(&input).unwrap().escape_safe(true));
1314            assert_eq!(output.matches('b').count(), 40, "width {width}: {output:?}");
1315        }
1316    }
1317
1318    #[cfg(feature = "json-escape-safe")]
1319    #[test]
1320    fn wrapping_keeps_json_escapes_atomic_at_narrow_widths() {
1321        let payload = r#"{"v":"a\"b\\c\nd\u0001e"}"#;
1322        for width in 8..=14 {
1323            let output = Console::builder()
1324                .width(width)
1325                .force_terminal(false)
1326                .build()
1327                .render_to_string(&Json::new(payload).unwrap().escape_safe(true));
1328            for line in output.lines() {
1329                let bytes = line.as_bytes();
1330                let mut index = 0;
1331                while index < bytes.len() {
1332                    if bytes[index] != b'\\' {
1333                        index += 1;
1334                        continue;
1335                    }
1336                    assert!(
1337                        index + 1 < bytes.len(),
1338                        "split escape at width {width}: {output:?}"
1339                    );
1340                    if bytes[index + 1] == b'u' {
1341                        assert!(
1342                            index + 6 <= bytes.len(),
1343                            "split unicode escape at width {width}: {output:?}"
1344                        );
1345                        index += 6;
1346                    } else {
1347                        index += 2;
1348                    }
1349                }
1350            }
1351        }
1352    }
1353
1354    #[cfg(feature = "json-escape-safe")]
1355    #[test]
1356    fn escape_folding_preserves_bytes_even_below_the_escape_width() {
1357        let payload = r#"{"v":"a\"b\\c\nd\u0001eeeeeeeeeeee"}"#;
1358        let wide = Console::builder()
1359            .width(100)
1360            .force_terminal(false)
1361            .build()
1362            .render_to_string(&Json::new(payload).unwrap().escape_safe(true))
1363            .replace('\n', "");
1364        for width in 1..=20 {
1365            let output = Console::builder()
1366                .width(width)
1367                .force_terminal(false)
1368                .build()
1369                .render_to_string(&Json::new(payload).unwrap().escape_safe(true));
1370            assert_eq!(output.replace('\n', ""), wide, "width {width}");
1371            assert!(output
1372                .lines()
1373                .all(|line| crate::cells::cell_len(line) <= width));
1374        }
1375    }
1376
1377    #[cfg(feature = "json-escape-safe")]
1378    #[test]
1379    fn escape_cropping_never_emits_a_partial_escape() {
1380        let payload = r#""a\"b\\c\nd\u0001eeee""#;
1381        for width in 1..=24 {
1382            let output = Console::builder()
1383                .width(width)
1384                .force_terminal(false)
1385                .build()
1386                .render_to_string(&Json::new(payload).unwrap().no_wrap(true).escape_safe(true));
1387            let mut chars = output.chars();
1388            while let Some(c) = chars.next() {
1389                if c == '\\' {
1390                    let next = chars.next().expect("complete short escape");
1391                    if next == 'u' {
1392                        for _ in 0..4 {
1393                            assert!(chars.next().is_some_and(|c| c.is_ascii_hexdigit()));
1394                        }
1395                    }
1396                }
1397            }
1398        }
1399    }
1400
1401    /// Nested inside another renderable, `JSON.text.no_wrap` survives and each
1402    /// line is **cropped** at the width rather than wrapped — see
1403    /// [`Json::no_wrap`]. Wrapping here instead was silent content loss: with
1404    /// `rich -j doc.json -w 120` on an 80-column console the document was laid
1405    /// out at 120 and then cropped to 80 by `Console.print`, so whole runs
1406    /// vanished and the surviving text read as if it were contiguous.
1407    ///
1408    /// Captured from rich-cli 1.8.1 driven by rich 15.0.0:
1409    /// `COLUMNS=80 rich -j long.json -w 40`.
1410    #[test]
1411    fn a_nested_document_is_cropped_rather_than_wrapped() {
1412        let payload = r#"{"k": "the quick brown fox jumps over the lazy dog and keeps running for a very long time indeed"}"#;
1413        let console = Console::builder().width(40).color_system(None).build();
1414        let json = Json::new(payload).expect("valid json").no_wrap(true);
1415        assert_eq!(
1416            console.render_to_string(&json),
1417            "{\n  \"k\": \"the quick brown fox jumps over t\n}"
1418        );
1419
1420        // …and the wrap is still the default, because a bare
1421        // `Console.print(JSON(...))` loses the flag in `Text.join`.
1422        assert_eq!(
1423            render_plain(payload, 40),
1424            "{\n  \"k\": \"the quick brown fox jumps over \n\
1425             the lazy dog and keeps running for a \n\
1426             very long time indeed\"\n}"
1427        );
1428    }
1429
1430    /// A crop must not cut a double-width character in half: `set_cell_size`
1431    /// drops the straddling character and the line comes out one cell short,
1432    /// never one cell over.
1433    #[test]
1434    fn cropping_never_splits_a_wide_character() {
1435        let console = Console::builder().width(12).color_system(None).build();
1436        let json = Json::new("{\"k\": \"\u{1f306}\u{1f306}\u{1f306}\"}")
1437            .expect("valid json")
1438            .no_wrap(true);
1439        for line in console.render_to_string(&json).lines() {
1440            assert!(
1441                crate::cells::cell_len(line) <= 12,
1442                "line {line:?} overflows the crop"
1443            );
1444        }
1445    }
1446
1447    /// serde_json's default float parser takes a fast path that can land 1 ULP
1448    /// from the value in the file, so the rendered number parsed back to a
1449    /// *different* double. The `float_roundtrip` feature makes parsing exact.
1450    #[test]
1451    fn floats_round_trip_exactly() {
1452        for literal in [
1453            "-938371.9565467801",
1454            "0.1",
1455            "1.7976931348623157e308",
1456            "5e-324",
1457            "3.141592653589793",
1458        ] {
1459            let out = render_plain(&format!("{{\"v\": {literal}}}"), 120);
1460            let rendered: String = out
1461                .split(':')
1462                .nth(1)
1463                .expect("a value after the key")
1464                .trim()
1465                .trim_end_matches(['}', ' ', '\n'])
1466                .to_string();
1467            let want: f64 = literal.parse().expect("literal parses");
1468            let got: f64 = rendered
1469                .parse()
1470                .unwrap_or_else(|_| panic!("rendered {rendered:?}"));
1471            assert_eq!(
1472                got.to_bits(),
1473                want.to_bits(),
1474                "{literal} rendered as {rendered} — a different double"
1475            );
1476        }
1477    }
1478
1479    /// serde_json stops at 128 levels, so a 200-deep document — which CPython
1480    /// parses without complaint — was reported as invalid JSON and the CLI
1481    /// exited 1 on a file upstream renders.
1482    #[test]
1483    fn deep_nesting_is_not_rejected() {
1484        for depth in [128, 129, 200, 1000] {
1485            let payload = format!("{}1{}", "[".repeat(depth), "]".repeat(depth));
1486            let json = Json::new(&payload)
1487                .unwrap_or_else(|error| panic!("depth {depth} rejected: {error}"));
1488            let console = Console::builder()
1489                .width(4 * depth + 8)
1490                .color_system(None)
1491                .build();
1492            let out = console.render_to_string(&json);
1493            assert_eq!(
1494                out.matches('[').count(),
1495                depth,
1496                "depth {depth} did not render every level"
1497            );
1498        }
1499    }
1500
1501    /// Parsing and dropping must cost heap, not stack: a recursive parser (or
1502    /// the compiler's recursive drop glue) turns a deep document into a stack
1503    /// overflow, which kills the process without even an error message.
1504    #[test]
1505    fn very_deep_nesting_does_not_overflow_the_stack() {
1506        let payload = format!("{}1{}", "[".repeat(MAX_DEPTH), "]".repeat(MAX_DEPTH));
1507        let json = Json::new(&payload).expect("deep document parses");
1508        drop(json);
1509        // Drop glue on a tree far past the parse limit stays iterative too.
1510        let mut node = Node::Number("1".to_string());
1511        for _ in 0..100_000 {
1512            node = Node::Array(vec![node]);
1513        }
1514        drop(node);
1515    }
1516
1517    /// Past `MAX_DEPTH` the document is refused with CPython's
1518    /// `RecursionError` message instead of being rendered: its indentation
1519    /// alone is quadratic in the depth, and a 100 000-deep array exhausted
1520    /// memory and was killed.
1521    #[test]
1522    fn nesting_past_the_limit_is_an_error() {
1523        for (open, close, what) in [("[", "]", "array"), ("{\"a\":", "}", "object")] {
1524            for depth in [MAX_DEPTH + 1, 100_000] {
1525                let payload = format!("{}1{}", open.repeat(depth), close.repeat(depth));
1526                let error = Json::new(&payload).err().expect("too deep");
1527                assert!(
1528                    error.to_string().contains(&format!(
1529                        "maximum recursion depth exceeded while decoding a JSON {what}"
1530                    )),
1531                    "{error}"
1532                );
1533            }
1534        }
1535        // An empty container counts as a level, as a call in CPython's scanner.
1536        let payload = format!("{}[]{}", "[".repeat(MAX_DEPTH), "]".repeat(MAX_DEPTH));
1537        assert!(Json::new(&payload).is_err());
1538        let payload = format!(
1539            "{}[]{}",
1540            "[".repeat(MAX_DEPTH - 1),
1541            "]".repeat(MAX_DEPTH - 1)
1542        );
1543        assert!(Json::new(&payload).is_ok());
1544    }
1545
1546    /// Python's json accepts and emits `NaN` / `Infinity` / `-Infinity`
1547    /// (`allow_nan=True` is the default), and rich's JSONHighlighter has no
1548    /// rule that matches them, so upstream prints them *unstyled*. serde_json
1549    /// rejected the whole document.
1550    ///
1551    /// Captured from rich 15.0.0:
1552    /// `Console(width=40, force_terminal=True).print(JSON(...))`.
1553    #[test]
1554    fn non_finite_numbers_render_unstyled() {
1555        assert_eq!(
1556            render(r#"{"a": NaN, "b": Infinity, "c": -Infinity, "d": 1.5}"#),
1557            "\x1b[1m{\x1b[0m\n  \x1b[1;34m\"a\"\x1b[0m: NaN,\n  \x1b[1;34m\"b\"\x1b[0m: \
1558             Infinity,\n  \x1b[1;34m\"c\"\x1b[0m: -Infinity,\n  \x1b[1;34m\"d\"\x1b[0m: \
1559             \x1b[1;36m1.5\x1b[0m\n\x1b[1m}\x1b[0m"
1560        );
1561        // Python spells them with those exact capitalisations and nothing else.
1562        for rejected in [r#"{"a": nan}"#, r#"{"a": inf}"#, r#"{"a": -inf}"#] {
1563            assert!(Json::new(rejected).is_err(), "{rejected} should not parse");
1564        }
1565    }
1566
1567    /// The hand-written reader must accept and reject exactly what serde_json
1568    /// does for bounded numbers — Python also accepts overflowing exponents.
1569    #[test]
1570    fn acceptance_matches_serde_json() {
1571        let samples = [
1572            "{}",
1573            "[]",
1574            "  {\t\"a\" :\n1 }  ",
1575            r#"{"a": 1, "a": 2}"#,
1576            r#"{"a": [1, {"b": null}], "c": "x"}"#,
1577            "0",
1578            "-0",
1579            "0.0",
1580            "1e10",
1581            "1E+10",
1582            "1e-7",
1583            "12345678901234567890",
1584            "-12345678901234567890123456789012345",
1585            "01",
1586            "1.",
1587            ".1",
1588            "+1",
1589            "1e",
1590            "-",
1591            "--1",
1592            "-i",
1593            "-Inf",
1594            "Infinit",
1595            "NAN",
1596            "1 2",
1597            "",
1598            "   ",
1599            "{",
1600            "[",
1601            "]",
1602            "}",
1603            "[,]",
1604            "[1,]",
1605            r#"{"a": 1,}"#,
1606            r#"{a: 1}"#,
1607            r#"{'a': 1}"#,
1608            r#"{"a" 1}"#,
1609            "[1 2]",
1610            "truex",
1611            "tru",
1612            "nul",
1613            r#""unterminated"#,
1614            r#""\q""#,
1615            r#""é""#,
1616            r#""😀""#,
1617            "\"raw\nnewline\"",
1618            r#""café ❤""#,
1619            "\"\u{e9}\\\"",
1620            "\u{feff}{}",
1621            "[[[[1]]]]",
1622        ];
1623        for sample in samples {
1624            let ours = Json::new(sample).is_ok();
1625            let theirs = serde_json::from_str::<serde_json::Value>(sample).is_ok();
1626            assert_eq!(ours, theirs, "disagreed about {sample:?}");
1627        }
1628    }
1629
1630    #[test]
1631    fn python_numbers_preserve_large_integers_and_overflow() {
1632        for number in [
1633            "1234567890123456789012345678901234567890",
1634            "-1234567890123456789012345678901234567890",
1635        ] {
1636            assert_eq!(render_plain(number, 100), number);
1637        }
1638        assert_eq!(render_plain("-0", 100), "0");
1639        assert_eq!(render_plain("1e400", 100), "Infinity");
1640        assert_eq!(render_plain("-1e999", 100), "-Infinity");
1641        for invalid in ["01", "-01", "1.e2", "1e+", "1e400x", "--1", "1+2", ".1"] {
1642            assert!(Json::new(invalid).is_err(), "accepted {invalid}");
1643        }
1644    }
1645
1646    /// And the tree it builds must be the tree serde_json would have built.
1647    /// `serde_json::to_string_pretty` happens to use the very layout upstream's
1648    /// `json.dumps(indent=2)` does, so it doubles as a reference dump: key
1649    /// order, repeated-key collapsing, string escaping and number formatting
1650    /// all have to agree.
1651    #[test]
1652    fn the_parsed_tree_matches_serde_json() {
1653        let samples = [
1654            r#"{"name": "Alice", "age": 30, "admin": true, "tags": ["a", "b"], "meta": null}"#,
1655            r#"{"a": 1, "b": 2, "a": 3}"#,
1656            r#"{"a": {"b": {"c": [1, [], {}, [[2]]]}}}"#,
1657            r#"{"k": "A\t\"x\"A\\\/é"}"#,
1658            r#"[0, 12345678901234567890, -7]"#,
1659            r#"{"café": "❤", "": ""}"#,
1660            "[]",
1661            "{}",
1662            "\"top level\"",
1663            "1234",
1664        ];
1665        for sample in samples {
1666            let reference: serde_json::Value =
1667                serde_json::from_str(sample).expect("sample is valid JSON");
1668            assert_eq!(
1669                render_plain(sample, 10_000),
1670                serde_json::to_string_pretty(&reference).expect("value re-serialises"),
1671                "diverged on {sample}"
1672            );
1673        }
1674    }
1675
1676    /// Floats are written with Python's `float.__repr__`, as `json.dumps` does,
1677    /// not serde_json's spelling (`1e-7`, `10000000000.0`).
1678    #[test]
1679    fn floats_render_as_python_repr() {
1680        let out = render_plain(
1681            "[-0.5, 1e10, 1E+10, 1e-7, 1.7976931348623157e308, 1e16]",
1682            10_000,
1683        );
1684        let values: Vec<&str> = out
1685            .lines()
1686            .filter_map(|line| line.trim().strip_suffix(',').or(Some(line.trim())))
1687            .filter(|value| !matches!(*value, "[" | "]"))
1688            .collect();
1689        assert_eq!(
1690            values,
1691            [
1692                "-0.5",
1693                "10000000000.0",
1694                "10000000000.0",
1695                "1e-07",
1696                "1.7976931348623157e+308",
1697                "1e+16"
1698            ]
1699        );
1700    }
1701
1702    /// A repeated key collapses to one entry — first position, last value —
1703    /// which is what both `dict` and serde_json's `preserve_order` produce.
1704    #[test]
1705    fn a_repeated_key_keeps_its_position_and_last_value() {
1706        assert_eq!(
1707            render_plain(r#"{"a": 1, "b": 2, "a": 3}"#, 40),
1708            "{\n  \"a\": 3,\n  \"b\": 2\n}"
1709        );
1710    }
1711
1712    /// Escapes are decoded and re-encoded, because upstream re-serialises the
1713    /// parsed data with `json.dumps`.
1714    #[test]
1715    fn escapes_are_re_encoded_like_dumps() {
1716        assert_eq!(
1717            render_plain(r#"{"k": "A\t\"x\""}"#, 60),
1718            "{\n  \"k\": \"A\\t\\\"x\\\"\"\n}"
1719        );
1720    }
1721}