Skip to main content

vole_document/adapter/pdf/
lexer.rs

1//! Byte-authoritative PDF lexical scanner.
2//!
3//! `lex` partitions an input into ordered [`Span`]s such that every byte belongs
4//! to exactly one span. It performs no structural interpretation: keywords,
5//! numbers, references, and object boundaries remain inside `Regular` and
6//! `LiteralString` spans for the Phase 3.2 parser to resolve. In particular a
7//! literal string is a single opaque span, so occurrences of `obj`, `endobj`, or
8//! `stream` *inside* a string can never be mistaken for structure.
9//!
10//! A `stream` keyword followed by an EOL switches to an **opaque payload** span
11//! that runs to the next `endstream` keyword. This is essential: compressed
12//! stream bytes are high-entropy and routinely contain unbalanced string
13//! delimiters; tokenizing them would swallow all later structure. The payload is
14//! emitted as one `Regular` span (its exact bounds are re-derived from `/Length`
15//! by the physical scanner), and `endstream` is emitted as a `Regular` keyword so
16//! the object loop can resume.
17//!
18//! Unterminated constructs are not errors here: they extend to EOF and are
19//! reported as non-fatal [`LexIssue`]s, because a byte cover must still be
20//! produced for hostile or truncated input.
21
22use crate::error::{Error, Result};
23use crate::limits::Limits;
24
25use super::span::{Span, SpanKind, SpanSet};
26
27/// A tolerated lexical anomaly. Coverage is still complete when one is present.
28#[derive(Debug, Clone, Copy, PartialEq, Eq)]
29pub enum LexIssue {
30    /// A `(` literal string reached EOF before its matching `)`.
31    UnterminatedLiteralString { at: u64 },
32    /// A `<` hex string reached EOF before its closing `>`.
33    UnterminatedHexString { at: u64 },
34    /// A `%` comment reached EOF without a closing CR or LF.
35    UnterminatedComment { at: u64 },
36}
37
38/// The lexical cover of an input plus any non-fatal issues encountered.
39#[derive(Debug, Clone, PartialEq, Eq)]
40pub struct LexResult {
41    /// The exact byte cover.
42    pub spans: SpanSet,
43    /// Anomalies tolerated while building the cover, in source order.
44    pub issues: Vec<LexIssue>,
45}
46
47/// PDF whitespace: NUL, HT, LF, FF, CR, SP.
48const fn is_whitespace(b: u8) -> bool {
49    matches!(b, 0x00 | 0x09 | 0x0A | 0x0C | 0x0D | 0x20)
50}
51
52/// PDF delimiters, none of which may appear inside a name or regular token.
53const fn is_delimiter(b: u8) -> bool {
54    matches!(
55        b,
56        b'(' | b')' | b'<' | b'>' | b'[' | b']' | b'{' | b'}' | b'/' | b'%'
57    )
58}
59
60/// A regular byte is neither whitespace nor a delimiter.
61const fn is_regular(b: u8) -> bool {
62    !is_whitespace(b) && !is_delimiter(b)
63}
64
65/// Lex `input` under `limits`, returning a complete byte cover.
66///
67/// Every byte of `input` appears in exactly one span, in ascending order. The
68/// cover is validated against the input length before returning, so an internal
69/// scanner bug surfaces as [`crate::ErrorClass::CoverageViolation`] rather than
70/// silent data loss. Exceeding `limits.max_pdf_spans` returns
71/// [`crate::ErrorClass::ResourceLimit`].
72pub fn lex(input: &[u8], limits: Limits) -> Result<LexResult> {
73    let n = input.len();
74    let max = limits.max_pdf_spans;
75    let mut spans: Vec<Span> = Vec::new();
76    let mut issues: Vec<LexIssue> = Vec::new();
77    let mut pos: usize = 0;
78
79    while pos < n {
80        let b = input[pos];
81        if is_whitespace(b) {
82            let start = pos;
83            while pos < n && is_whitespace(input[pos]) {
84                pos += 1;
85            }
86            push(&mut spans, start, pos - start, SpanKind::Whitespace, max)?;
87        } else if b == b'%' {
88            let start = pos;
89            pos += 1;
90            while pos < n && input[pos] != b'\r' && input[pos] != b'\n' {
91                pos += 1;
92            }
93            if pos >= n {
94                issues.push(LexIssue::UnterminatedComment { at: start as u64 });
95            }
96            push(&mut spans, start, pos - start, SpanKind::Comment, max)?;
97        } else if b == b'(' {
98            let start = pos;
99            pos = scan_literal_string(input, start, &mut issues);
100            push(&mut spans, start, pos - start, SpanKind::LiteralString, max)?;
101        } else if b == b'<' {
102            if pos + 1 < n && input[pos + 1] == b'<' {
103                push(&mut spans, pos, 2, SpanKind::DictOpen, max)?;
104                pos += 2;
105            } else {
106                let start = pos;
107                pos += 1;
108                while pos < n && input[pos] != b'>' {
109                    pos += 1;
110                }
111                if pos < n {
112                    pos += 1; // include the closing '>'
113                } else {
114                    issues.push(LexIssue::UnterminatedHexString { at: start as u64 });
115                }
116                push(&mut spans, start, pos - start, SpanKind::HexString, max)?;
117            }
118        } else if b == b'>' {
119            if pos + 1 < n && input[pos + 1] == b'>' {
120                push(&mut spans, pos, 2, SpanKind::DictClose, max)?;
121                pos += 2;
122            } else {
123                // A lone '>' is a delimiter with no lexical role; keep it as a
124                // single-byte token so the cover stays complete.
125                push(&mut spans, pos, 1, SpanKind::Regular, max)?;
126                pos += 1;
127            }
128        } else if b == b'[' {
129            push(&mut spans, pos, 1, SpanKind::ArrayOpen, max)?;
130            pos += 1;
131        } else if b == b']' {
132            push(&mut spans, pos, 1, SpanKind::ArrayClose, max)?;
133            pos += 1;
134        } else if b == b'{' {
135            push(&mut spans, pos, 1, SpanKind::BraceOpen, max)?;
136            pos += 1;
137        } else if b == b'}' {
138            push(&mut spans, pos, 1, SpanKind::BraceClose, max)?;
139            pos += 1;
140        } else if b == b'/' {
141            let start = pos;
142            pos += 1;
143            while pos < n && is_regular(input[pos]) {
144                pos += 1;
145            }
146            push(&mut spans, start, pos - start, SpanKind::Name, max)?;
147        } else if b == b')' {
148            // Unmatched ')' cannot open a literal string; keep it as a
149            // single-byte token to preserve coverage.
150            push(&mut spans, pos, 1, SpanKind::Regular, max)?;
151            pos += 1;
152        } else {
153            let start = pos;
154            while pos < n && is_regular(input[pos]) {
155                pos += 1;
156            }
157            push(&mut spans, start, pos - start, SpanKind::Regular, max)?;
158            // A `stream` keyword followed by an EOL introduces an opaque payload.
159            if input[start..pos] == *b"stream"
160                && let Some(eol_len) = post_stream_eol_len(input, pos)
161            {
162                let eol_start = pos;
163                pos += eol_len;
164                push(&mut spans, eol_start, eol_len, SpanKind::Whitespace, max)?;
165                let data_start = pos;
166                match find_endstream(input, data_start) {
167                    Some(es) => {
168                        if es > data_start {
169                            push(
170                                &mut spans,
171                                data_start,
172                                es - data_start,
173                                SpanKind::Regular,
174                                max,
175                            )?;
176                        }
177                        push(&mut spans, es, b"endstream".len(), SpanKind::Regular, max)?;
178                        pos = es + b"endstream".len();
179                    }
180                    None => {
181                        // No terminating keyword: the rest is one opaque span.
182                        if data_start < n {
183                            push(
184                                &mut spans,
185                                data_start,
186                                n - data_start,
187                                SpanKind::Regular,
188                                max,
189                            )?;
190                        }
191                        pos = n;
192                    }
193                }
194            }
195        }
196    }
197
198    let spans = SpanSet { spans };
199    spans.validate(n as u64)?;
200    Ok(LexResult { spans, issues })
201}
202
203/// Length of the EOL directly after a `stream` keyword: `LF` (1) or `CRLF` (2).
204/// A lone `CR` is not a valid stream EOL, matching the physical scanner.
205fn post_stream_eol_len(input: &[u8], pos: usize) -> Option<usize> {
206    match input.get(pos) {
207        Some(b'\n') => Some(1),
208        Some(b'\r') if input.get(pos + 1) == Some(&b'\n') => Some(2),
209        _ => None,
210    }
211}
212
213/// First offset of the `endstream` keyword at or after `from`, or `None` if there
214/// is none.
215///
216/// A keyword is terminated on the right, so this requires the byte immediately
217/// after `endstream` to be PDF whitespace, a PDF delimiter, or EOF; the byte
218/// *before* may be anything (it is the last payload byte). Requiring a preceding
219/// EOL is wrong for real producers: Ghostscript 10.00.0 (and others) emit the
220/// stream payload immediately followed by `endstream` with no intervening EOL,
221/// so an EOL-preceded search over-reads the payload to a *later* `endstream` and
222/// corrupts all subsequent structure.
223///
224/// The false-positive risk is low: payload bytes would have to contain the
225/// literal 10-byte run `endstream` followed by a delimiter or whitespace, and the
226/// physical scanner additionally re-derives the exact bounds from `/Length` when
227/// one is present. The scan is bounded by `input.len()` and returns the first
228/// such occurrence.
229fn find_endstream(input: &[u8], from: usize) -> Option<usize> {
230    let needle = b"endstream";
231    let mut i = from;
232    while i + needle.len() <= input.len() {
233        if &input[i..i + needle.len()] == needle {
234            let terminated = match input.get(i + needle.len()) {
235                None => true,
236                Some(&b) => is_whitespace(b) || is_delimiter(b),
237            };
238            if terminated {
239                return Some(i);
240            }
241        }
242        i += 1;
243    }
244    None
245}
246
247/// Consume a `(` literal string starting at `start`; returns the first offset
248/// after the span. `\` escapes the next byte, and `\` before a CR, LF, or CRLF
249/// is a line continuation that also swallows the EOL.
250fn scan_literal_string(input: &[u8], start: usize, issues: &mut Vec<LexIssue>) -> usize {
251    let n = input.len();
252    let mut pos = start;
253    let mut depth: u64 = 0;
254    loop {
255        if pos >= n {
256            issues.push(LexIssue::UnterminatedLiteralString { at: start as u64 });
257            return pos;
258        }
259        let c = input[pos];
260        if c == b'\\' {
261            pos += 1;
262            if pos < n {
263                let escaped = input[pos];
264                pos += 1;
265                if escaped == b'\r' && pos < n && input[pos] == b'\n' {
266                    pos += 1;
267                }
268            }
269        } else if c == b'(' {
270            depth = depth.saturating_add(1);
271            pos += 1;
272        } else if c == b')' {
273            depth = depth.saturating_sub(1);
274            pos += 1;
275            if depth == 0 {
276                return pos;
277            }
278        } else {
279            pos += 1;
280        }
281    }
282}
283
284/// Append a span, enforcing the span-count bound.
285fn push(spans: &mut Vec<Span>, start: usize, len: usize, kind: SpanKind, max: u32) -> Result<()> {
286    if spans.len() as u64 >= max as u64 {
287        return Err(Error::resource_limit(format!(
288            "pdf span count exceeds limit {max}"
289        )));
290    }
291    spans.push(Span {
292        start: start as u64,
293        len: len as u64,
294        kind,
295    });
296    Ok(())
297}
298
299#[cfg(test)]
300mod tests {
301    use super::*;
302    use crate::error::ErrorClass;
303
304    fn run(input: &[u8]) -> LexResult {
305        lex(input, Limits::DEFAULT).expect("lex must succeed")
306    }
307
308    fn kinds(r: &LexResult) -> Vec<SpanKind> {
309        r.spans.spans.iter().map(|s| s.kind).collect()
310    }
311
312    #[test]
313    fn empty_input() {
314        let r = run(b"");
315        assert!(r.spans.spans.is_empty());
316        assert!(r.issues.is_empty());
317        assert!(r.spans.validate(0).is_ok());
318    }
319
320    #[test]
321    fn single_whitespace() {
322        let r = run(b" ");
323        assert_eq!(r.spans.spans.len(), 1);
324        assert_eq!(r.spans.spans[0].kind, SpanKind::Whitespace);
325        assert_eq!(r.spans.spans[0].len, 1);
326    }
327
328    #[test]
329    fn whitespace_run_coalesces() {
330        let input = b" \t\r\n\x0c\x00 ";
331        let r = run(input);
332        assert_eq!(r.spans.spans.len(), 1);
333        assert_eq!(r.spans.spans[0].kind, SpanKind::Whitespace);
334        assert_eq!(r.spans.spans[0].len, input.len() as u64);
335    }
336
337    #[test]
338    fn comment_to_eol_excludes_eol() {
339        let r = run(b"%hello\n");
340        assert_eq!(kinds(&r), vec![SpanKind::Comment, SpanKind::Whitespace]);
341        assert_eq!(r.spans.spans[0].start, 0);
342        assert_eq!(r.spans.spans[0].len, 6);
343        assert_eq!(r.spans.spans[1].start, 6);
344        assert_eq!(r.spans.spans[1].len, 1);
345        assert!(r.issues.is_empty());
346    }
347
348    #[test]
349    fn comment_stops_before_cr() {
350        let r = run(b"%x\r\n");
351        assert_eq!(kinds(&r), vec![SpanKind::Comment, SpanKind::Whitespace]);
352        assert_eq!(r.spans.spans[0].len, 2);
353        assert_eq!(r.spans.spans[1].len, 2);
354        assert!(r.issues.is_empty());
355    }
356
357    #[test]
358    fn comment_to_eof_is_issue() {
359        let r = run(b"%abc");
360        assert_eq!(kinds(&r), vec![SpanKind::Comment]);
361        assert_eq!(r.spans.spans[0].len, 4);
362        assert_eq!(r.issues, vec![LexIssue::UnterminatedComment { at: 0 }]);
363    }
364
365    #[test]
366    fn literal_string_with_escape() {
367        // ( a \ ) b )  -- the escaped ')' does not close the string.
368        let r = run(b"(a\\)b)");
369        assert_eq!(kinds(&r), vec![SpanKind::LiteralString]);
370        assert_eq!(r.spans.spans[0].len, 6);
371        assert!(r.issues.is_empty());
372    }
373
374    #[test]
375    fn literal_string_with_line_continuation() {
376        // ( a \ CR LF b ) -- backslash before CRLF swallows the EOL.
377        let r = run(b"(a\\\r\nb)");
378        assert_eq!(kinds(&r), vec![SpanKind::LiteralString]);
379        assert_eq!(r.spans.spans[0].len, 7);
380        assert!(r.issues.is_empty());
381    }
382
383    #[test]
384    fn literal_string_with_nested_parens() {
385        let r = run(b"(a(b)c)");
386        assert_eq!(kinds(&r), vec![SpanKind::LiteralString]);
387        assert_eq!(r.spans.spans[0].len, 7);
388        assert!(r.issues.is_empty());
389    }
390
391    #[test]
392    fn unterminated_literal_string_is_issue() {
393        let r = run(b"(abc");
394        assert_eq!(kinds(&r), vec![SpanKind::LiteralString]);
395        assert_eq!(r.spans.spans[0].len, 4);
396        assert_eq!(
397            r.issues,
398            vec![LexIssue::UnterminatedLiteralString { at: 0 }]
399        );
400    }
401
402    #[test]
403    fn percent_inside_literal_string_is_not_a_comment() {
404        let r = run(b"( % )");
405        assert_eq!(kinds(&r), vec![SpanKind::LiteralString]);
406        assert_eq!(r.spans.spans[0].len, 5);
407        assert!(r.issues.is_empty());
408    }
409
410    #[test]
411    fn paren_inside_hex_string_is_not_special() {
412        let r = run(b"<4(2>");
413        assert_eq!(kinds(&r), vec![SpanKind::HexString]);
414        assert_eq!(r.spans.spans[0].len, 5);
415        assert!(r.issues.is_empty());
416    }
417
418    #[test]
419    fn hex_string_basic() {
420        let r = run(b"<4142>");
421        assert_eq!(kinds(&r), vec![SpanKind::HexString]);
422        assert_eq!(r.spans.spans[0].len, 6);
423    }
424
425    #[test]
426    fn unterminated_hex_string_is_issue() {
427        let r = run(b"<41");
428        assert_eq!(kinds(&r), vec![SpanKind::HexString]);
429        assert_eq!(r.spans.spans[0].len, 3);
430        assert_eq!(r.issues, vec![LexIssue::UnterminatedHexString { at: 0 }]);
431    }
432
433    #[test]
434    fn dict_delimiters() {
435        let r = run(b"<<>>");
436        assert_eq!(kinds(&r), vec![SpanKind::DictOpen, SpanKind::DictClose]);
437        assert_eq!(r.spans.spans[0].len, 2);
438        assert_eq!(r.spans.spans[1].len, 2);
439    }
440
441    #[test]
442    fn lone_closers_are_regular_tokens() {
443        let r = run(b")>");
444        assert_eq!(kinds(&r), vec![SpanKind::Regular, SpanKind::Regular]);
445        let r = run(b">>>");
446        assert_eq!(kinds(&r), vec![SpanKind::DictClose, SpanKind::Regular]);
447    }
448
449    #[test]
450    fn name_and_empty_name() {
451        let r = run(b"/Name");
452        assert_eq!(kinds(&r), vec![SpanKind::Name]);
453        assert_eq!(r.spans.spans[0].len, 5);
454        let r = run(b"/");
455        assert_eq!(kinds(&r), vec![SpanKind::Name]);
456        assert_eq!(r.spans.spans[0].len, 1);
457    }
458
459    #[test]
460    fn arrays_and_braces() {
461        let r = run(b"[]{}");
462        assert_eq!(
463            kinds(&r),
464            vec![
465                SpanKind::ArrayOpen,
466                SpanKind::ArrayClose,
467                SpanKind::BraceOpen,
468                SpanKind::BraceClose,
469            ]
470        );
471    }
472
473    #[test]
474    fn numbers_and_reference_are_regular_tokens() {
475        let r = run(b"12 0 R");
476        assert_eq!(
477            kinds(&r),
478            vec![
479                SpanKind::Regular,
480                SpanKind::Whitespace,
481                SpanKind::Regular,
482                SpanKind::Whitespace,
483                SpanKind::Regular,
484            ]
485        );
486        let text: Vec<&[u8]> = r
487            .spans
488            .spans
489            .iter()
490            .map(|s| &b"12 0 R"[s.start as usize..(s.start + s.len) as usize])
491            .collect();
492        assert_eq!(text, vec![&b"12"[..], b" ", b"0", b" ", b"R"]);
493    }
494
495    #[test]
496    fn endobj_inside_literal_string_stays_one_span() {
497        let r = run(b"(1 0 obj endobj)5");
498        assert_eq!(kinds(&r), vec![SpanKind::LiteralString, SpanKind::Regular]);
499        assert_eq!(r.spans.spans[0].start, 0);
500        assert_eq!(r.spans.spans[0].len, 16);
501        assert_eq!(r.spans.spans[1].start, 16);
502        assert_eq!(r.spans.spans[1].len, 1);
503    }
504
505    #[test]
506    fn stream_payload_without_trailing_eol_is_one_opaque_span() {
507        // Real producers (e.g. Ghostscript 10.00.0) write the payload directly
508        // before `endstream` with no intervening EOL. The payload here also
509        // contains an unbalanced `(` and `endstream`/`stream`-like runs that must
510        // not be mistaken for the terminating keyword.
511        let payload = b"(unbalanced ( with endstreamZ and streamY bytes";
512        let mut input = Vec::new();
513        input.extend_from_slice(b"stream\n");
514        input.extend_from_slice(payload);
515        input.extend_from_slice(b"endstream\n");
516
517        let r = run(&input);
518        assert_eq!(
519            kinds(&r),
520            vec![
521                SpanKind::Regular,    // stream
522                SpanKind::Whitespace, // \n
523                SpanKind::Regular,    // opaque payload
524                SpanKind::Regular,    // endstream
525                SpanKind::Whitespace, // \n
526            ]
527        );
528        let p = &r.spans.spans[2];
529        assert_eq!(
530            &input[p.start as usize..(p.start + p.len) as usize],
531            payload
532        );
533        let es = &r.spans.spans[3];
534        assert_eq!(
535            &input[es.start as usize..(es.start + es.len) as usize],
536            b"endstream"
537        );
538        assert!(r.issues.is_empty());
539        r.spans.validate(input.len() as u64).unwrap();
540    }
541
542    #[test]
543    fn stream_payload_followed_by_eol_then_endstream_still_works() {
544        let input = b"stream\nhello\nendstream\n";
545        let r = run(input);
546        assert_eq!(
547            kinds(&r),
548            vec![
549                SpanKind::Regular,    // stream
550                SpanKind::Whitespace, // \n
551                SpanKind::Regular,    // opaque payload (hello + trailing EOL)
552                SpanKind::Regular,    // endstream
553                SpanKind::Whitespace, // \n
554            ]
555        );
556        let p = &r.spans.spans[2];
557        assert_eq!(
558            &input[p.start as usize..(p.start + p.len) as usize],
559            b"hello\n"
560        );
561        let es = &r.spans.spans[3];
562        assert_eq!(
563            &input[es.start as usize..(es.start + es.len) as usize],
564            b"endstream"
565        );
566        assert!(r.issues.is_empty());
567        r.spans.validate(input.len() as u64).unwrap();
568    }
569
570    #[test]
571    fn endstream_at_eof_terminates_payload() {
572        let input = b"stream\npayloadendstream";
573        let r = run(input);
574        assert_eq!(
575            kinds(&r),
576            vec![
577                SpanKind::Regular,    // stream
578                SpanKind::Whitespace, // \n
579                SpanKind::Regular,    // payload
580                SpanKind::Regular,    // endstream (EOF-terminated)
581            ]
582        );
583        let es = &r.spans.spans[3];
584        assert_eq!(
585            &input[es.start as usize..(es.start + es.len) as usize],
586            b"endstream"
587        );
588        assert_eq!(es.start + es.len, input.len() as u64);
589        assert!(r.issues.is_empty());
590        r.spans.validate(input.len() as u64).unwrap();
591    }
592
593    #[test]
594    fn endstream_like_run_without_right_terminator_is_not_a_keyword() {
595        // `endstreamZ` (regular byte after) must not terminate the payload, so the
596        // real `endstream\n` is the first recognised keyword.
597        let input = b"stream\nxx endstreamZ yyendstream\n";
598        let r = run(input);
599        assert_eq!(
600            kinds(&r),
601            vec![
602                SpanKind::Regular,    // stream
603                SpanKind::Whitespace, // \n
604                SpanKind::Regular,    // opaque payload
605                SpanKind::Regular,    // endstream
606                SpanKind::Whitespace, // \n
607            ]
608        );
609        let p = &r.spans.spans[2];
610        assert_eq!(
611            &input[p.start as usize..(p.start + p.len) as usize],
612            b"xx endstreamZ yy"
613        );
614        r.spans.validate(input.len() as u64).unwrap();
615    }
616
617    #[test]
618    fn span_at_over_lexed_input() {
619        let r = run(b"12 0 R");
620        assert_eq!(r.spans.span_at(0).map(|s| s.kind), Some(SpanKind::Regular));
621        assert_eq!(r.spans.span_at(1).map(|s| s.kind), Some(SpanKind::Regular));
622        assert_eq!(
623            r.spans.span_at(2).map(|s| s.kind),
624            Some(SpanKind::Whitespace)
625        );
626        assert_eq!(r.spans.span_at(3).map(|s| s.kind), Some(SpanKind::Regular));
627        assert_eq!(
628            r.spans.span_at(4).map(|s| s.kind),
629            Some(SpanKind::Whitespace)
630        );
631        assert_eq!(r.spans.span_at(5).map(|s| s.kind), Some(SpanKind::Regular));
632        assert_eq!(r.spans.span_at(6), None);
633    }
634
635    #[test]
636    fn span_limit_triggers_resource_limit() {
637        let limits = Limits {
638            max_pdf_spans: 2,
639            ..Limits::DEFAULT
640        };
641        let e = lex(b"a b c", limits).unwrap_err();
642        assert_eq!(e.class(), ErrorClass::ResourceLimit);
643    }
644
645    #[test]
646    fn cover_invariant_holds_for_a_battery() {
647        let inputs: [&[u8]; 15] = [
648            b"",
649            b" ",
650            b"%%EOF",
651            b"<< /Type /Catalog >>",
652            b"[1 2.5 -3 (str) <4142> /Name]",
653            b"(unterminated",
654            b"<414243",
655            b"%comment with ( and < and >>",
656            b"()<>[]{}",
657            b"\x00\x09\x0a\x0c\x0d\x20mixed",
658            b"trailing>",
659            b")))",
660            b"<<<<<<",
661            b"(nested (deep (deeper)) end)",
662            b"1 0 obj\n<< /A (x) >>\nendobj",
663        ];
664        for input in inputs {
665            let r = lex(input, Limits::DEFAULT).expect("lex must succeed");
666            r.spans
667                .validate(input.len() as u64)
668                .expect("cover must validate");
669        }
670    }
671
672    fn xorshift64(state: &mut u64) -> u64 {
673        let mut x = *state;
674        x ^= x << 13;
675        x ^= x >> 7;
676        x ^= x << 17;
677        *state = x;
678        x
679    }
680
681    #[test]
682    fn cover_invariant_holds_for_random_bytes() {
683        let mut state: u64 = 0x9E37_79B9_7F4A_7C15;
684        for _ in 0..500 {
685            let len = (xorshift64(&mut state) % 300) as usize;
686            let mut buf = Vec::with_capacity(len);
687            for _ in 0..len {
688                buf.push((xorshift64(&mut state) & 0xFF) as u8);
689            }
690            let r = lex(&buf, Limits::STRICT).expect("lex must not fail on bounded input");
691            r.spans
692                .validate(buf.len() as u64)
693                .expect("random cover must validate");
694            assert!(r.spans.spans.len() <= buf.len());
695        }
696    }
697}