Skip to main content

tabnas/
lexer.rs

1// Copyright (c) 2013-2026 Richard Rodger, MIT License
2
3use crate::error::TabnasError;
4use crate::options::{LexCheck, LexCheckResult, MatchTokenMatcher, Options};
5use crate::token::{
6    Point, Token, TIN_BD, TIN_CM, TIN_LN, TIN_NR, TIN_SP, TIN_ST, TIN_TX, TIN_VL, TIN_ZZ,
7};
8use crate::value::Value;
9use regex::Regex;
10use std::panic::{catch_unwind, AssertUnwindSafe};
11use std::sync::Arc;
12
13pub struct Lexer<'a> {
14    src: &'a str,
15    chars: Vec<char>,
16    byte_indices: Vec<usize>,
17    char_len: usize,
18    idx: usize,
19    ri: usize,
20    ci: usize,
21    options: Arc<Options>,
22    ignore_tins: Vec<crate::Tin>,
23    char_sets: crate::text::CharSets,
24    err: Option<TabnasError>,
25    end_reached: bool,
26    /// `number.exclude`, compiled once per grammar and shared by the
27    /// parser that owns it; see [`compile_number_exclude`].
28    exclude_regex: Option<Arc<Regex>>,
29    want: Option<Vec<crate::Tin>>,
30    standalone: Option<(crate::Rule, crate::Context)>,
31}
32
33#[derive(Clone)]
34pub(crate) struct LexerState {
35    idx: usize,
36    ri: usize,
37    ci: usize,
38    err: Option<TabnasError>,
39    end_reached: bool,
40}
41
42/// Opaque snapshot returned by [`Lexer::relex_for_rule`]. Pass it to
43/// [`Lexer::unrelex`] if the caller later rejects the committed recut.
44#[derive(Clone)]
45pub struct RelexCheckpoint {
46    state: LexerState,
47    replay: std::collections::VecDeque<Token>,
48}
49
50enum CheckFlow {
51    Continue,
52    Skip,
53    Token(Box<Token>),
54}
55
56/// The result the lexer passes through its own internals.
57///
58/// `TabnasError` is 664 bytes, so `LexResult<Token>` is 664 bytes
59/// too: three times the token it carries, and moved on the SUCCESS path
60/// through every frame of the lexer chain. Boxing the error inside the
61/// engine makes those results the size of a token again, and costs an
62/// allocation only when there is an error to report, which is the path that
63/// is already building a 664-byte diagnostic.
64///
65/// The public entry points still hand back an unboxed `TabnasError`, so no
66/// caller of this crate sees the box.
67type LexResult<T> = Result<T, Box<TabnasError>>;
68
69/// The compiled form of `number.exclude`, or `None` when there is no
70/// pattern or it does not compile (an invalid pattern excludes nothing,
71/// as it always has).
72///
73/// Compiling a regex costs about 700K instructions, which is more than
74/// a small document costs to parse, so the parser compiles it once per
75/// grammar and hands every lexer the same automaton through the `Arc`.
76/// Cloning a `Regex` would not do: it shares the automaton but builds a
77/// fresh scratch-cache pool, and the first search from each clone fills
78/// it. Go and TypeScript compile the pattern once, at configuration time.
79pub(crate) fn compile_number_exclude(options: &Options) -> Option<Arc<Regex>> {
80    let pattern = options.number.exclude.as_deref()?;
81    Regex::new(pattern).ok().map(Arc::new)
82}
83
84impl<'a> Lexer<'a> {
85    pub fn new(src: &'a str, mut options: Options) -> Self {
86        if let Err(error) = options.validate_comment_definitions() {
87            panic!("invalid options: {error}");
88        }
89        // A lexer built directly may be handed options nobody has
90        // ordered yet. The parser's own lexer comes through
91        // `with_shared`, whose options were ordered when they were
92        // prepared, and whose exclude pattern was compiled then too.
93        options.sort_for_lexing();
94        let exclude_regex = compile_number_exclude(&options);
95        Self::with_shared(src, Arc::new(options), exclude_regex)
96    }
97
98    /// Lex against options the parser already owns and has ordered, with
99    /// the `number.exclude` pattern it compiled from them.
100    pub(crate) fn with_shared(
101        src: &'a str,
102        options: Arc<Options>,
103        exclude_regex: Option<Arc<Regex>>,
104    ) -> Self {
105        let mut chars = Vec::new();
106        let mut byte_indices = Vec::new();
107        for (b_idx, c) in src.char_indices() {
108            chars.push(c);
109            byte_indices.push(b_idx);
110        }
111        let char_len = chars.len();
112
113        Lexer {
114            src,
115            chars,
116            byte_indices,
117            char_len,
118            idx: 0,
119            ri: 1,
120            ci: 1,
121            ignore_tins: options.ignore_tins(),
122            char_sets: options.char_sets(),
123            options,
124            err: None,
125            end_reached: false,
126            exclude_regex,
127            want: None,
128            standalone: None,
129        }
130    }
131
132    fn current_point(&self) -> Point {
133        Point {
134            len: self.src.len(),
135            site: crate::Site {
136                si: self.byte_position(),
137                pos: self.idx,
138                ri: self.ri,
139                ci: self.ci,
140            },
141        }
142    }
143
144    /// The source from char index `start` up to `end`, clipped to the end
145    /// of the source: the span of a bad token, cut as TypeScript's
146    /// `lex.bad(why, pstart, pend)` cuts it (`ts/src/lexer.ts`).
147    ///
148    /// Only the END is clipped. The caller must supply `start <= end`
149    /// and `start <= self.char_len`, or the indexing panics. Every call
150    /// site below passes `esc_point.site.pos - 1`, which additionally
151    /// needs `1 <= pos`; that holds because `esc_point` is captured
152    /// after the escape character has been consumed. These three are
153    /// the preconditions `rs/verus/lexer_span.rs` ASSUMES: it proves the
154    /// span arithmetic is in bounds given them, and does not model the
155    /// call sites, so nothing machine-checks that they hold here. A
156    /// refactor that captured the point before the advance would still
157    /// pass `rs/verus/run.sh`.
158    fn source_span(&self, start: usize, end: usize) -> String {
159        self.chars[start..end.min(self.char_len)].iter().collect()
160    }
161
162    fn state(&self) -> LexerState {
163        LexerState {
164            idx: self.idx,
165            ri: self.ri,
166            ci: self.ci,
167            err: self.err.clone(),
168            end_reached: self.end_reached,
169        }
170    }
171
172    /// Full immutable source supplied to this lexer.
173    pub fn source(&self) -> &str {
174        self.src
175    }
176
177    /// Source remaining at the live cursor.
178    pub fn remaining(&self) -> &str {
179        &self.src[self.byte_position()..]
180    }
181
182    /// Return at most `max_chars` Unicode scalar values from the live cursor.
183    /// This is the Rust counterpart of the public `lex.fwd`/`Lex.Fwd` helper.
184    pub fn forward(&self, max_chars: usize) -> &str {
185        let remaining = self.remaining();
186        let end = remaining
187            .char_indices()
188            .nth(max_chars)
189            .map_or(remaining.len(), |(index, _)| index);
190        &remaining[..end]
191    }
192
193    /// Snapshot the live cursor for token construction.
194    pub fn point(&self) -> Point {
195        self.current_point()
196    }
197
198    /// Advance by Unicode scalar values. Returns false without moving when
199    /// the requested count extends beyond end-of-source.
200    pub fn advance_chars(&mut self, count: usize) -> bool {
201        if self.idx.saturating_add(count) > self.char_len {
202            return false;
203        }
204        for _ in 0..count {
205            self.advance();
206        }
207        true
208    }
209
210    /// Construct a token from a point captured before cursor advancement.
211    pub fn token(
212        &self,
213        name: impl AsRef<str>,
214        tin: crate::Tin,
215        value: Value,
216        source: impl Into<crate::TokenText>,
217        point: Point,
218    ) -> Token {
219        Token::new(name, tin, value, source, point)
220    }
221
222    /// Resolve or allocate a token identity in this lexer's configuration.
223    pub fn token_tin(&mut self, name: impl Into<String>) -> crate::Tin {
224        Arc::make_mut(&mut self.options).register_token(name)
225    }
226
227    /// Resolve a token identity back to its configured name.
228    pub fn token_name(&self, tin: crate::Tin) -> String {
229        self.options.token_name(tin)
230    }
231
232    /// Whether the options this lexer runs under enable `alt`: not when
233    /// `rule.exclude` names one of its groups, nor when `rule.include`
234    /// lists groups and it declares none of them. The parser skips such an
235    /// alternate, and TypeScript removes it from the rule spec
236    /// (`filterRules`) before anything reads the spec, so a custom matcher
237    /// that reads a rule's alternates, to tell a key position from a value
238    /// one, asks this to see the same alternates TypeScript does.
239    pub fn alt_enabled(&self, alt: &crate::AltSpec) -> bool {
240        crate::parser::groups_enabled(alt, &self.options)
241    }
242
243    /// Construct a bad token at the current cursor.
244    ///
245    /// A custom matcher that returns it reports a lexer fault, which the
246    /// parser handles as it handles its own lexer's: a rule's fetch raises
247    /// it at once with `why` as the code, at this position, or, under
248    /// recovery, records it, coalescing a run of them into one error, and
249    /// moves the cursor past its source; relexing leaves it for an
250    /// alternate to re-cut. The cursor need not be advanced here.
251    pub fn bad(&self, why: impl Into<String>) -> Token {
252        let point = self.current_point();
253        let source = self
254            .peek()
255            .map_or_else(String::new, |character| character.to_string());
256        let mut token = Token::new("#BD", TIN_BD, Value::Undefined, source, point);
257        token.err = crate::TokenCode::from(why.into());
258        token.why = token.err.clone();
259        token
260    }
261
262    /// Construct a bad token whose displayed source is a scalar-indexed span.
263    /// As in TypeScript, the diagnostic point remains the live cursor.
264    /// Returned from a matcher it is handled as [`Lexer::bad`] describes;
265    /// recovery steps past the span, counted from that cursor.
266    pub fn bad_span(&self, why: impl Into<String>, start: usize, end: usize) -> Token {
267        let point = self.current_point();
268        let source = if start <= end && end <= self.char_len {
269            let start_byte = self
270                .byte_indices
271                .get(start)
272                .copied()
273                .unwrap_or(self.src.len());
274            let end_byte = self
275                .byte_indices
276                .get(end)
277                .copied()
278                .unwrap_or(self.src.len());
279            self.src[start_byte..end_byte].to_string()
280        } else {
281            self.peek()
282                .map_or_else(String::new, |character| character.to_string())
283        };
284        let mut token = Token::new("#BD", TIN_BD, Value::Undefined, source, point);
285        token.err = crate::TokenCode::from(why.into());
286        token.why = token.err.clone();
287        token
288    }
289
290    fn byte_position(&self) -> usize {
291        self.byte_indices
292            .get(self.idx)
293            .copied()
294            .unwrap_or(self.src.len())
295    }
296
297    fn advance(&mut self) -> Option<char> {
298        if self.idx < self.char_len {
299            let c = self.chars[self.idx];
300            self.idx += 1;
301            if self.char_sets.row.contains(c) {
302                self.ri += 1;
303                self.ci = 1;
304            } else {
305                self.ci += 1;
306            }
307            Some(c)
308        } else {
309            None
310        }
311    }
312
313    /// Step over one character of a string body, counting a column and
314    /// never a row. `advance` counts a row for any character in
315    /// `line.rowChars`, which is right between tokens and wrong inside a
316    /// string: TypeScript's `buildStringBodySpec` (ts/src/lexer.ts) makes
317    /// a row character plain body there unless the string is multi-line
318    /// and the character is in `line.chars` as well, which is
319    /// [`Lexer::advance_string_line`].
320    fn advance_body(&mut self) -> Option<char> {
321        let c = self.peek()?;
322        self.idx += 1;
323        self.ci += 1;
324        Some(c)
325    }
326
327    /// Step over a line character inside a multi-line string: the column
328    /// resets, and a character in `line.rowChars` counts a row, as
329    /// TypeScript's string-body classes LINE and LINE+ROW do
330    /// (`STRING_BODY_TABLE` in ts/src/lexer.ts) and as Go's
331    /// `BuildStringBodySpec` ports them.
332    fn advance_string_line(&mut self) -> Option<char> {
333        let c = self.peek()?;
334        self.idx += 1;
335        if self.char_sets.row.contains(c) {
336            self.ri += 1;
337        }
338        self.ci = 1;
339        Some(c)
340    }
341
342    fn peek(&self) -> Option<char> {
343        if self.idx < self.char_len {
344            Some(self.chars[self.idx])
345        } else {
346            None
347        }
348    }
349
350    fn peek_at(&self, offset: usize) -> Option<char> {
351        let i = self.idx + offset;
352        if i < self.char_len {
353            Some(self.chars[i])
354        } else {
355            None
356        }
357    }
358
359    fn wants(&self, tin: crate::Tin) -> bool {
360        self.want
361            .as_ref()
362            .is_none_or(|wanted| wanted.contains(&tin))
363    }
364
365    /// Bring a fetch's point and current character up to the live cursor.
366    ///
367    /// A custom matcher, or an imperative check, may move the cursor and
368    /// then decline: xml's steps over a byte-order mark and goes on to
369    /// look for a tag. Every matcher after it starts where it left the
370    /// cursor, as TypeScript's do (each reads `lex.pnt`) and Go's do (each
371    /// reads `l.pnt`), and the current character is `None` once the cursor
372    /// is at the end of the source. Reading the character the fetch began
373    /// on instead put the built-in matchers' tokens at the wrong place,
374    /// sent them looking for a character that was no longer there, and,
375    /// when the matcher had stepped over the LAST character, panicked on
376    /// the missing one.
377    #[inline]
378    fn resync(&self, point: &mut Point, current: &mut Option<char>) {
379        if self.idx != point.site.pos {
380            *point = self.current_point();
381            *current = self.peek();
382        }
383    }
384
385    /// Whether TypeScript would try a built-in matcher at all in a fetch
386    /// that began on `first` and whose cursor has since been moved by a
387    /// matcher that declined.
388    ///
389    /// TypeScript chooses the matchers a fetch tries from a table indexed
390    /// by the character the fetch begins on (`buildLexDispatch` in
391    /// ts/src/utility.ts, read in `Lex.next`): a built-in is listed for a
392    /// Latin-1 character only when a token it makes could start with it,
393    /// unless it carries a check, and every built-in is listed for a
394    /// character from U+0100 up. The table is not consulted again when a
395    /// custom matcher moves the cursor, so this asks it about `first`, and
396    /// the matcher then tests the character under the cursor as usual.
397    /// Before the cursor moves the two are the same character and the
398    /// matcher's own test is the whole answer, so this is `true`.
399    #[inline]
400    fn listed(
401        &self,
402        entry: usize,
403        first: char,
404        has_check: bool,
405        could_start: impl FnOnce(char) -> bool,
406    ) -> bool {
407        self.idx == entry || u32::from(first) >= 256 || has_check || could_start(first)
408    }
409
410    fn run_check(&mut self, check: Option<LexCheck>, point: Point) -> CheckFlow {
411        let Some(check) = check else {
412            return CheckFlow::Continue;
413        };
414        let remaining = &self.src[self.byte_position()..];
415        let result = check
416            .run_imperative(self)
417            .or_else(|| check.run(remaining))
418            .unwrap_or(LexCheckResult::Continue);
419        match result {
420            LexCheckResult::Continue => CheckFlow::Continue,
421            LexCheckResult::Skip => CheckFlow::Skip,
422            LexCheckResult::NativeToken(token) => CheckFlow::Token(token),
423            LexCheckResult::Token(token)
424                if !token.source.is_empty() && remaining.starts_with(&token.source) =>
425            {
426                let tin = if token.tin < 0 {
427                    self.options.token(&token.name).unwrap_or(token.tin)
428                } else {
429                    token.tin
430                };
431                if tin < 0 {
432                    return CheckFlow::Skip;
433                }
434                for _ in token.source.chars() {
435                    self.advance();
436                }
437                CheckFlow::Token(Box::new(Token::new(
438                    token.name,
439                    tin,
440                    token.value,
441                    token.source,
442                    point,
443                )))
444            }
445            LexCheckResult::Token(_) => CheckFlow::Skip,
446        }
447    }
448
449    /// Give any plugin matcher whose order is below `before` its turn.
450    ///
451    /// Nine sites in the lexer call this per token, once at each stage a
452    /// matcher is allowed to intervene. A grammar with no custom matcher,
453    /// which is most of them, was paying nine index lookups per token to be
454    /// told nine times that there is nothing to run. The guard is inline so
455    /// those sites skip the call itself; the walk stays out of line.
456    #[inline]
457    fn run_custom_matchers(
458        &mut self,
459        index: &mut usize,
460        before: f64,
461        plugin: &mut Option<(&mut crate::Rule, &mut crate::Context)>,
462    ) -> Option<Token> {
463        if *index >= self.options.lex.matchers.len() {
464            return None;
465        }
466        self.run_remaining_custom_matchers(index, before, plugin)
467    }
468
469    #[inline(never)]
470    fn run_remaining_custom_matchers(
471        &mut self,
472        index: &mut usize,
473        before: f64,
474        plugin: &mut Option<(&mut crate::Rule, &mut crate::Context)>,
475    ) -> Option<Token> {
476        while let Some(matcher) = self
477            .options
478            .lex
479            .matchers
480            .get_index(*index)
481            .map(|(_, matcher)| matcher)
482            .filter(|matcher| matcher.order < before)
483            .cloned()
484        {
485            *index += 1;
486            // Each matcher starts where the one before it left the cursor:
487            // one may step over a character and decline, as xml's does over
488            // a byte-order mark, and TypeScript's matchers all read the
489            // live `lex.pnt`.
490            let point = self.current_point();
491            let remaining = &self.src[self.byte_position()..];
492            let saved = self.state();
493            let token = if let Some(callback) = matcher.imperative.as_ref() {
494                let Some((rule, context)) = plugin.as_mut() else {
495                    continue;
496                };
497                callback(self, rule, context)
498            } else {
499                matcher
500                    .matcher
501                    .as_ref()
502                    .and_then(|callback| callback(remaining))
503                    .filter(|token| {
504                        !token.source.is_empty() && remaining.starts_with(&token.source)
505                    })
506                    .map(|token| {
507                        Token::new(token.name, token.tin, token.value, token.source, point)
508                    })
509            };
510            let Some(mut token) = token else {
511                if self.want.is_some() {
512                    self.rewind(saved);
513                }
514                continue;
515            };
516
517            // TypeScript and Go run opaque custom matchers speculatively for
518            // a negotiated cut. An unwanted result rolls back locally so a
519            // later matcher can still satisfy the request.
520            let tin = if token.tin < 0 {
521                self.options.token(&token.name).unwrap_or(token.tin)
522            } else {
523                token.tin
524            };
525            if tin < 0 || !self.wants(tin) {
526                self.rewind(saved);
527                continue;
528            }
529            if matcher.imperative.is_none() {
530                for _ in token.src.chars() {
531                    self.advance();
532                }
533            }
534            token.tin = tin;
535            return Some(token);
536        }
537        None
538    }
539
540    fn is_text_delimiter_here(&self) -> bool {
541        self.is_text_delimiter_at(self.idx)
542    }
543
544    fn is_text_delimiter_at(&self, index: usize) -> bool {
545        let Some(ch) = self.chars.get(index).copied() else {
546            return true;
547        };
548        let remaining = &self.src[self.byte_indices[index]..];
549        (self.options.space.lex && self.char_sets.space.contains(ch))
550            || (self.options.fixed.lex
551                && self
552                    .options
553                    .fixed
554                    .tokens
555                    .values()
556                    .any(|token| !token.source.is_empty() && remaining.starts_with(&token.source)))
557            || (self.options.line.lex
558                && (self.char_sets.line_ends.contains(ch) || matches!(ch, '\u{2028}' | '\u{2029}')))
559            || (self.options.comment.lex
560                && self.options.comment.definitions.values().any(|definition| {
561                    definition.lex
562                        && !definition.start.is_empty()
563                        && remaining.starts_with(&definition.start)
564                }))
565            || self
566                .options
567                .ender
568                .iter()
569                .any(|ender| !ender.is_empty() && remaining.starts_with(ender))
570    }
571
572    /// Fetches the next non-IGNORE token (skipping spaces, lines, comments).
573    pub fn next_token(&mut self) -> Result<Token, TabnasError> {
574        let point = self.current_point();
575        let result = match catch_unwind(AssertUnwindSafe(|| {
576            if let Some(ref error) = self.err {
577                return Err(Box::new(error.clone()));
578            }
579
580            loop {
581                let token = self.next_raw(None)?;
582                if !self.ignore_tins.contains(&token.tin) {
583                    return Ok(token);
584                }
585            }
586        })) {
587            Ok(result) => result,
588            Err(payload) => self.record_panic(payload, "Lexer::next_token", point),
589        };
590        // The box is internal to the engine; a caller gets the error itself.
591        result.map_err(|error| *error)
592    }
593
594    /// Fetch the next token without discarding whitespace, line, or comment tokens.
595    pub fn next_raw_token(&mut self) -> Result<Token, TabnasError> {
596        let point = self.current_point();
597        let result = match catch_unwind(AssertUnwindSafe(|| self.next_raw(None))) {
598            Ok(result) => result,
599            Err(payload) => self.record_panic(payload, "Lexer::next_raw_token", point),
600        };
601        result.map_err(|error| *error)
602    }
603
604    /// Fetch one token for an imperative parser callback, preserving ignored
605    /// space/line/comment tokens just like TypeScript's public `lex.next`.
606    /// Replayed tokens produced by `Context::rewind` are served first.
607    pub fn next_raw_for_rule(
608        &mut self,
609        rule: &mut crate::Rule,
610        context: &mut crate::Context,
611    ) -> LexResult<Token> {
612        if let Some(token) = context.next_replay() {
613            Ok(token)
614        } else {
615            self.next_raw_with(None, Some((rule, context)))
616        }
617    }
618
619    /// Fetch the next non-ignored token for an imperative parser callback.
620    pub fn next_for_rule(
621        &mut self,
622        rule: &mut crate::Rule,
623        context: &mut crate::Context,
624    ) -> LexResult<Token> {
625        loop {
626            let token = self.next_raw_for_rule(rule, context)?;
627            if !self.ignore_tins.contains(&token.tin) {
628                return Ok(token);
629            }
630        }
631    }
632
633    /// Public negotiated-relex entry point for native parser callbacks.
634    /// A successful recut commits the lexer cursor and returns an opaque undo
635    /// checkpoint; a failed recut restores all lexer state before returning.
636    pub fn relex_for_rule(
637        &mut self,
638        from: &Token,
639        wanted: &[crate::Tin],
640        rule: &mut crate::Rule,
641        context: &mut crate::Context,
642    ) -> Option<(Token, RelexCheckpoint)> {
643        self.relex(from, wanted, rule, context)
644    }
645
646    /// Undo a committed [`Lexer::relex_for_rule`] operation, including the
647    /// pending tokens hidden while the replacement cut was negotiated.
648    pub fn unrelex(&mut self, checkpoint: RelexCheckpoint, context: &mut crate::Context) {
649        self.restore(checkpoint.state);
650        context.restore_replay(checkpoint.replay);
651    }
652
653    fn record_panic(
654        &mut self,
655        payload: Box<dyn std::any::Any + Send>,
656        api: &str,
657        point: Point,
658    ) -> LexResult<Token> {
659        let error = TabnasError::from_panic(
660            payload,
661            api,
662            self.src,
663            point.site.pos,
664            point.site.ri,
665            point.site.ci,
666            &self.options,
667        );
668        self.err = Some(error.clone());
669        Err(Box::new(error))
670    }
671
672    /// Fetch a raw token while restricting non-eager custom token matchers to
673    /// the exact tins accepted at the parser slot being filled. Builtin and
674    /// fixed-token matchers are unaffected by this gate.
675    pub(crate) fn next_rule_token(
676        &mut self,
677        expected_match_tins: &[crate::Tin],
678        rule: &mut crate::Rule,
679        context: &mut crate::Context,
680    ) -> LexResult<Token> {
681        self.next_raw_with(Some(expected_match_tins), Some((rule, context)))
682    }
683
684    /// Step past a bad token the parser has absorbed or skipped, as
685    /// TypeScript's `advanceLexPast` does (ts/src/rules.ts): a bad token
686    /// does not advance the cursor by itself, so recovery moves it to the
687    /// end of the token's span, never backwards, and, for a fault raised
688    /// inside a compound construct, on past the next row character so
689    /// lexing resumes on a fresh row. The lexer's own faults latch until
690    /// this clears them.
691    pub(crate) fn skip_bad(&mut self, token: &Token, to_line_end: bool) {
692        let span = token.src.chars().count().max(1);
693        let mut target = self.idx.max(token.site.pos.saturating_add(span));
694        if to_line_end {
695            let mut end = target;
696            while end < self.char_len && !self.char_sets.row.contains(self.chars[end]) {
697                end += 1;
698            }
699            target = target.max(self.char_len.min(end + 1));
700        }
701        while self.idx < target && self.idx < self.char_len {
702            self.advance();
703        }
704        self.err = None;
705        if self.idx < self.char_len {
706            self.end_reached = false;
707        }
708    }
709
710    /// Re-cut an already buffered source span, constrained to the token
711    /// identities requested by one alternate. On success the cursor remains
712    /// after the new cut; the returned state can restore the original cut if
713    /// that alternate later fails.
714    pub(crate) fn relex(
715        &mut self,
716        from: &Token,
717        wanted: &[crate::Tin],
718        rule: &mut crate::Rule,
719        context: &mut crate::Context,
720    ) -> Option<(Token, RelexCheckpoint)> {
721        if from.src.is_empty() || from.site.pos > self.char_len || wanted.is_empty() {
722            return None;
723        }
724        // The standing error, if any, is moved into the checkpoint rather
725        // than copied: the cut below clears it anyway, and a copy is the
726        // length of the source (`full_source`) for every cut attempted.
727        let err = self.err.take();
728        let saved = LexerState {
729            err,
730            ..self.state()
731        };
732        // TypeScript temporarily replaces the lexer's pending-token queue
733        // with an empty queue for a negotiated cut. Rust keeps that queue on
734        // Context, so hide it explicitly and preserve it in the checkpoint.
735        let replay = context.take_replay();
736        self.idx = from.site.pos;
737        self.ri = from.site.ri;
738        self.ci = from.site.ci;
739        self.err = None;
740        self.end_reached = false;
741        self.want = Some(wanted.to_vec());
742        // Straight to the matchers, past `next_raw_with`: an error here only
743        // rejects the cut, and the restore below puts back the lexer's own,
744        // so the source it would attach, and the copy it would keep, are
745        // never seen. With them, every rejected cut cost the length of the
746        // source, and a flat stylesheet parsed in quadratic time.
747        let recut = self.next_raw_inner(None, Some((rule, context))).ok();
748        self.want = None;
749        match recut.filter(|token| wanted.contains(&token.tin)) {
750            Some(mut token) => {
751                token.ignored = from.ignored.clone();
752                Some((
753                    token,
754                    RelexCheckpoint {
755                        state: saved,
756                        replay,
757                    },
758                ))
759            }
760            None => {
761                self.restore(saved);
762                // Discard any speculative replay generated by an imperative
763                // matcher and restore the queue that preceded the attempt.
764                context.restore_replay(replay);
765                None
766            }
767        }
768    }
769
770    pub(crate) fn restore(&mut self, state: LexerState) {
771        self.idx = state.idx;
772        self.ri = state.ri;
773        self.ci = state.ci;
774        self.err = state.err;
775        self.end_reached = state.end_reached;
776        self.want = None;
777    }
778
779    /// Put the cursor back after a custom matcher's speculative attempt and
780    /// keep the request being negotiated, as TypeScript's `Lex.speculate`
781    /// restores the point and leaves `want` alone. [`Lexer::restore`] ends
782    /// a negotiation, so the first custom matcher to decline during a
783    /// re-cut used to lift the request for every matcher after it, and a
784    /// string matcher could then cut the `#ST` the re-cut was there to
785    /// avoid.
786    fn rewind(&mut self, state: LexerState) {
787        let want = self.want.take();
788        self.restore(state);
789        self.want = want;
790    }
791
792    fn next_raw(&mut self, expected_match_tins: Option<&[crate::Tin]>) -> LexResult<Token> {
793        // Only a lexer being driven directly needs these, and building
794        // them costs a whole `Options` clone. A parse reaches the lexer
795        // through `next_rule_token`, which brings the real rule and
796        // context with it, so it never wants them at all.
797        let (mut rule, mut context) = match self.standalone.take() {
798            Some(pair) => pair,
799            None => (
800                crate::Rule::new("#NORULE", Value::Undefined),
801                crate::Context::new(
802                    self.options.rewind.history,
803                    self.src,
804                    Value::Undefined,
805                    Arc::clone(&self.options),
806                    crate::InstanceInfo::default(),
807                ),
808            ),
809        };
810        let result = self.next_raw_with(expected_match_tins, Some((&mut rule, &mut context)));
811        self.standalone = Some((rule, context));
812        result
813    }
814
815    fn modify_text_value(
816        &mut self,
817        mut value: Value,
818        plugin: &mut Option<(&mut crate::Rule, &mut crate::Context)>,
819    ) -> Value {
820        if self.options.text.modify.is_empty() {
821            return value;
822        }
823        let modifiers = self.options.text.modify.clone();
824        let options = self.options.clone();
825        let Some((rule, context)) = plugin.as_mut() else {
826            panic!("imperative text modifier requires an active lexer context");
827        };
828        for modifier in modifiers {
829            value = modifier.run(value, self, rule, context, &options);
830        }
831        value
832    }
833
834    fn next_raw_with(
835        &mut self,
836        expected_match_tins: Option<&[crate::Tin]>,
837        plugin: Option<(&mut crate::Rule, &mut crate::Context)>,
838    ) -> LexResult<Token> {
839        let result = self.next_raw_inner(expected_match_tins, plugin);
840        match result {
841            Ok(token) => Ok(token),
842            Err(mut error) => {
843                // The matchers build their errors without the source, and it
844                // is attached here, where an error leaves the lexer: a
845                // negotiated cut (`relex`) rejects many candidates, each an
846                // error nobody sees, and a copy of the whole source apiece
847                // made that quadratic.
848                error.full_source = self.src.to_string();
849                error.apply_options(&self.options);
850                self.err = Some((*error).clone());
851                Err(error)
852            }
853        }
854    }
855
856    fn next_raw_inner(
857        &mut self,
858        expected_match_tins: Option<&[crate::Tin]>,
859        mut plugin: Option<(&mut crate::Rule, &mut crate::Context)>,
860    ) -> LexResult<Token> {
861        if self.end_reached {
862            return Ok(Token::new(
863                "#ZZ",
864                TIN_ZZ,
865                Value::Undefined,
866                "",
867                self.current_point(),
868            ));
869        }
870
871        if self.idx >= self.char_len {
872            self.end_reached = true;
873            return Ok(Token::new(
874                "#ZZ",
875                TIN_ZZ,
876                Value::Undefined,
877                "",
878                self.current_point(),
879            ));
880        }
881
882        // The fetch begins on `first`, at `entry`. `pnt` and `c`, the point
883        // a token starts at and the character under the cursor, follow the
884        // cursor whenever a custom matcher or a check moves it and declines
885        // (`resync`); `c` is `None` once the cursor is at the end.
886        let entry = self.idx;
887        let first = self.chars[entry];
888        let mut pnt = self.current_point();
889        let mut c = Some(first);
890        let mut custom_index = 0;
891
892        if let Some(token) = self.run_custom_matchers(&mut custom_index, 1_000_000.0, &mut plugin) {
893            return Ok(token);
894        }
895        self.resync(&mut pnt, &mut c);
896
897        // User-declared match tokens occupy the 1e6 matcher priority band.
898        let match_skipped = if self.options.match_lex
899            && (!self.options.match_values.is_empty()
900                || self
901                    .options
902                    .match_tokens
903                    .values()
904                    .any(|matcher| self.wants(matcher.tin)))
905        {
906            match self.run_check(self.options.match_check.clone(), pnt) {
907                CheckFlow::Continue => false,
908                CheckFlow::Skip => true,
909                CheckFlow::Token(token) => return Ok(*token),
910            }
911        } else {
912            false
913        };
914        self.resync(&mut pnt, &mut c);
915        let remaining = &self.src[self.byte_position()..];
916        let custom_value = (self.options.match_lex && !match_skipped && self.want.is_none())
917            .then(|| {
918                self.options
919                    .match_values
920                    .values()
921                    .find_map(|matcher| match &matcher.matcher {
922                        MatchTokenMatcher::Callback(callback) => callback(remaining)
923                            .filter(|result| {
924                                !result.source.is_empty() && remaining.starts_with(&result.source)
925                            })
926                            .map(|result| (result.source, result.value)),
927                        MatchTokenMatcher::Regex(regex) => {
928                            let captures = regex.captures(remaining)?;
929                            let found = captures
930                                .get(0)
931                                .filter(|found| found.start() == 0 && !found.as_str().is_empty())?;
932                            let source = found.as_str().to_string();
933                            let value = matcher.transform.as_ref().map_or_else(
934                                || {
935                                    matcher
936                                        .val
937                                        .clone()
938                                        .unwrap_or_else(|| Value::String(source.clone()))
939                                },
940                                |transform| {
941                                    let groups = captures
942                                        .iter()
943                                        .map(|capture| {
944                                            capture.map_or_else(String::new, |value| {
945                                                value.as_str().into()
946                                            })
947                                        })
948                                        .collect::<Vec<_>>();
949                                    transform(&groups)
950                                },
951                            );
952                            Some((source, value))
953                        }
954                    })
955            })
956            .flatten();
957        if let Some((source, value)) = custom_value {
958            for _ in source.chars() {
959                self.advance();
960            }
961            return Ok(Token::new("#VL", TIN_VL, value, source, pnt));
962        }
963
964        let remaining = &self.src[self.byte_position()..];
965        // With no custom matcher there is nothing for the band to do: both
966        // passes walk an empty table and yield nothing, and `fix_len` is
967        // read only by that walk. Most grammars register none, and every
968        // token fetch of theirs paid the eager pass's scan of the fixed
969        // table (one closure call per fixed literal) to arrive at the
970        // `None` this guard now hands over directly. TS `makeMatchMatcher`
971        // returns null on an empty table (ts/src/lexer.ts) and the band is
972        // never installed; Go reaches the same place by defaulting
973        // `MatchLex` off unless `Options.Match` is set. Rust defaults
974        // `match_lex` true as TS does, so the guard is the parity.
975        let custom = (self.options.match_lex
976            && !match_skipped
977            && !self.options.match_tokens.is_empty())
978        .then(|| {
979            // Two passes, position-expected before eager, as go/lexer.go
980            // matchMatch and ts/src/lexer.ts makeMatchMatcher both make.
981            // One tin-ordered pass in which eagerness merely bypassed the
982            // slot gate let an eager matcher EARLIER in tin order win over
983            // an expected one later: with `p = %x31-39` beside
984            // `d = %x30-39`, the `2` of `12` lexed as the narrower class
985            // the `*d` loop never asked for. Eagerness is for firing where
986            // the slot's list is narrower than the grammar, never for
987            // outbidding what the slot names.
988            //
989            // Under a want the alternate's own tin list is the sharper
990            // gate, so one filtered pass is the whole search. With no
991            // expected list at all (a standalone lexer, no rule) nothing
992            // constrains the caller and every matcher is eligible in the
993            // first pass.
994            // The longest FIXED literal this slot expects that matches
995            // here, or 0. Only the eager pass consults it: there, a
996            // literal the slot names beats an eager-only matcher that
997            // cuts no further than it does. Without this, a character
998            // class that CONTAINS a literal the grammar also uses
999            // swallows it wherever the class is eager (`num = "0" /
1000            // posdigit *digit` beside `digit = %x30-39` rejected
1001            // `0.0.0`). LENGTH decides, not mere existence, so a keyword
1002            // literal cannot truncate a longer word: ties go to the
1003            // literal, and an eager matcher that cuts further still
1004            // wins. TS and Go do the same, in makeMatchMatcher and
1005            // matchMatch.
1006            //
1007            // Computed once per fetch and only when a regex matcher in the
1008            // eager pass has something to weigh against it, as TS
1009            // `expectedFixedLen` does (`fixLen = -1` until asked). The
1010            // scan is the whole fixed table against the slot's list; an
1011            // expected matcher that wins in pass 0, or a fetch under a
1012            // want, never needs it. Nothing the scan reads changes
1013            // between the two passes, so lazy equals eager.
1014            let mut fix_len: Option<usize> = None;
1015            let compute_fix_len = || {
1016                if self.want.is_none() && self.options.fixed.lex {
1017                    expected_match_tins.map_or(0, |expected| {
1018                        self.options
1019                            .fixed
1020                            .tokens
1021                            .values()
1022                            .filter(|token| {
1023                                !token.source.is_empty()
1024                                    && expected.contains(&token.tin)
1025                                    && remaining.starts_with(&token.source)
1026                            })
1027                            .map(|token| token.source.len())
1028                            .max()
1029                            .unwrap_or(0)
1030                    })
1031                } else {
1032                    0
1033                }
1034            };
1035            let passes = if self.want.is_some() { 1 } else { 2 };
1036            (0..passes).find_map(|pass| {
1037                self.options.match_tokens.values().find_map(|matcher| {
1038                    if !self.wants(matcher.tin) {
1039                        return None;
1040                    }
1041                    if self.want.is_none() {
1042                        let expected = expected_match_tins
1043                            .is_none_or(|expected| expected.contains(&matcher.tin));
1044                        if pass == 0 {
1045                            if !expected {
1046                                return None;
1047                            }
1048                        } else if expected || !matcher.eager {
1049                            return None;
1050                        }
1051                    }
1052                    let result = match &matcher.matcher {
1053                        MatchTokenMatcher::Regex(regex) => regex
1054                            .find(remaining)
1055                            .filter(|found| found.start() == 0)
1056                            // The eager pass yields to an expected
1057                            // literal it cannot out-cut; the fixed
1058                            // matcher (2e6) runs next and takes it. See
1059                            // `fix_len` above.
1060                            .filter(|found| {
1061                                pass == 0 || {
1062                                    let fix_len = *fix_len.get_or_insert_with(compute_fix_len);
1063                                    fix_len == 0 || found.len() > fix_len
1064                                }
1065                            })
1066                            .map(|found| {
1067                                let source = found.as_str().to_string();
1068                                (source.clone(), Value::String(source))
1069                            }),
1070                        MatchTokenMatcher::Callback(callback) => callback(remaining)
1071                            .filter(|result| {
1072                                !result.source.is_empty() && remaining.starts_with(&result.source)
1073                            })
1074                            .map(|result| (result.source, result.value)),
1075                    };
1076                    result.map(|(source, value)| (matcher.name.clone(), matcher.tin, source, value))
1077                })
1078            })
1079        });
1080        if let Some(Some((name, tin, matched, value))) = custom {
1081            for _ in matched.chars() {
1082                self.advance();
1083            }
1084            return Ok(Token::new(name, tin, value, matched, pnt));
1085        }
1086
1087        if let Some(token) = self.run_custom_matchers(&mut custom_index, 2_000_000.0, &mut plugin) {
1088            return Ok(token);
1089        }
1090        self.resync(&mut pnt, &mut c);
1091
1092        // Fixed literals occupy the 2e6 band and use longest-match wins.
1093        let fixed_skipped = if self.options.fixed.lex {
1094            match self.run_check(self.options.fixed.check.clone(), pnt) {
1095                CheckFlow::Continue => false,
1096                CheckFlow::Skip => true,
1097                CheckFlow::Token(token) => return Ok(*token),
1098            }
1099        } else {
1100            false
1101        };
1102        self.resync(&mut pnt, &mut c);
1103        let fixed_skipped = fixed_skipped
1104            || !self.listed(entry, first, self.options.fixed.check.is_some(), |ch| {
1105                self.options
1106                    .fixed
1107                    .tokens
1108                    .values()
1109                    .any(|token| token.source.starts_with(ch))
1110            });
1111        let remaining = &self.src[self.byte_position()..];
1112        // The winner is carried out of the table as its position, not as a
1113        // copy of its text. `Token::new` takes the name and the source text
1114        // by reference and stores both inline, so the only owned copy the
1115        // token needs is the one inside `Value::String`. Naming the match
1116        // as three owned values cost three `String` allocations per fixed
1117        // token, two of them freed again before the token was built.
1118        // The first byte decides almost every entry. Asking `wants` and then
1119        // `starts_with` of each fixed token in turn ran a tin lookup and a
1120        // `memcmp` per token in the grammar per token in the input, and a
1121        // grammar with fifty fixed tokens pays fifty of each to reject
1122        // forty-nine. One byte answers the same question, and an empty
1123        // source is kept out by its own check, which only entries that
1124        // already matched the byte ever reach.
1125        let first_byte = remaining.as_bytes().first().copied();
1126        let fixed = (self.options.fixed.lex && !fixed_skipped)
1127            .then(|| {
1128                self.options
1129                    .fixed
1130                    .tokens
1131                    .values()
1132                    .enumerate()
1133                    .filter(|(_, token)| {
1134                        token.source.as_bytes().first().copied() == first_byte
1135                            && !token.source.is_empty()
1136                            && self.wants(token.tin)
1137                            && remaining.starts_with(&token.source)
1138                    })
1139                    .max_by_key(|(_, token)| token.source.len())
1140                    .map(|(index, token)| (index, token.source.chars().count()))
1141            })
1142            .flatten();
1143        if let Some((index, source_chars)) = fixed {
1144            for _ in 0..source_chars {
1145                self.advance();
1146            }
1147            let (_, token) = self
1148                .options
1149                .fixed
1150                .tokens
1151                .get_index(index)
1152                .expect("index came from this table, which nothing writes to mid-parse");
1153            return Ok(Token::new(
1154                &token.name,
1155                token.tin,
1156                Value::String(token.source.clone()),
1157                token.source.as_str(),
1158                pnt,
1159            ));
1160        }
1161
1162        if let Some(token) = self.run_custom_matchers(&mut custom_index, 3_000_000.0, &mut plugin) {
1163            return Ok(token);
1164        }
1165        self.resync(&mut pnt, &mut c);
1166
1167        // 1. Whitespace
1168        let space_skipped = if self.options.space.lex && self.wants(TIN_SP) {
1169            match self.run_check(self.options.space.check.clone(), pnt) {
1170                CheckFlow::Continue => false,
1171                CheckFlow::Skip => true,
1172                CheckFlow::Token(token) => return Ok(*token),
1173            }
1174        } else {
1175            false
1176        };
1177        self.resync(&mut pnt, &mut c);
1178        if self.options.space.lex
1179            && !space_skipped
1180            && self.wants(TIN_SP)
1181            && c.is_some_and(|ch| self.char_sets.space.contains(ch))
1182            && self.listed(entry, first, self.options.space.check.is_some(), |ch| {
1183                self.char_sets.space.contains(ch)
1184            })
1185        {
1186            let mut src = String::new();
1187            while let Some(ch) = self.peek() {
1188                if self.char_sets.space.contains(ch) {
1189                    src.push(ch);
1190                    self.advance();
1191                } else {
1192                    break;
1193                }
1194            }
1195            return Ok(Token::new(
1196                "#SP",
1197                TIN_SP,
1198                Value::String(src.clone()),
1199                src,
1200                pnt,
1201            ));
1202        }
1203
1204        if let Some(token) = self.run_custom_matchers(&mut custom_index, 4_000_000.0, &mut plugin) {
1205            return Ok(token);
1206        }
1207        self.resync(&mut pnt, &mut c);
1208
1209        // 2. Line ending
1210        let line_skipped = if self.options.line.lex && self.wants(TIN_LN) {
1211            match self.run_check(self.options.line.check.clone(), pnt) {
1212                CheckFlow::Continue => false,
1213                CheckFlow::Skip => true,
1214                CheckFlow::Token(token) => return Ok(*token),
1215            }
1216        } else {
1217            false
1218        };
1219        self.resync(&mut pnt, &mut c);
1220        let line_skipped = line_skipped
1221            || !self.listed(entry, first, self.options.line.check.is_some(), |ch| {
1222                self.char_sets.line_ends.contains(ch)
1223            });
1224        if self.options.line.lex
1225            && !line_skipped
1226            && self.wants(TIN_LN)
1227            && c.is_some_and(|ch| self.char_sets.line_ends.contains(ch))
1228        {
1229            let mut src = String::new();
1230            let mut seen = std::collections::HashSet::new();
1231            while let Some(ch) = self.peek() {
1232                if !self.char_sets.line_ends.contains(ch) {
1233                    break;
1234                }
1235                if self.options.line.single && !seen.insert(ch) {
1236                    break;
1237                }
1238                src.push(self.advance().expect("peeked character must advance"));
1239            }
1240            self.ci = 1;
1241            return Ok(Token::new(
1242                "#LN",
1243                TIN_LN,
1244                Value::String(src.clone()),
1245                src,
1246                pnt,
1247            ));
1248        }
1249
1250        if let Some(bad_char @ ('\u{2028}' | '\u{2029}')) = c {
1251            if self.options.line.lex && !line_skipped && self.wants(TIN_LN) {
1252                self.advance();
1253                let err = TabnasError::new(
1254                    "unexpected",
1255                    bad_char.to_string(),
1256                    "",
1257                    pnt.site.pos,
1258                    pnt.site.ri,
1259                    pnt.site.ci,
1260                );
1261                self.err = Some(err.clone());
1262                return Err(Box::new(err));
1263            }
1264        }
1265
1266        if let Some(token) = self.run_custom_matchers(&mut custom_index, 5_000_000.0, &mut plugin) {
1267            return Ok(token);
1268        }
1269        self.resync(&mut pnt, &mut c);
1270
1271        // 3. Quoted strings. These precede comments in the canonical matcher
1272        // order, so an overlapping quote/comment opener is a string unless
1273        // string matching explicitly abandons the malformed candidate.
1274        let string_skipped = if self.options.string.lex && self.wants(TIN_ST) {
1275            match self.run_check(self.options.string.check.clone(), pnt) {
1276                CheckFlow::Continue => false,
1277                CheckFlow::Skip => true,
1278                CheckFlow::Token(token) => return Ok(*token),
1279            }
1280        } else {
1281            false
1282        };
1283        self.resync(&mut pnt, &mut c);
1284        let quote = c.filter(|ch| self.char_sets.string.contains(*ch));
1285        if let Some(quote) = quote.filter(|_| {
1286            self.options.string.lex
1287                && !string_skipped
1288                && self.wants(TIN_ST)
1289                && self.listed(entry, first, self.options.string.check.is_some(), |ch| {
1290                    self.char_sets.string.contains(ch)
1291                })
1292        }) {
1293            let start = (self.idx, self.ri, self.ci);
1294            match self.match_string(quote, pnt) {
1295                result @ Ok(_) => return result,
1296                Err(error) if !self.options.string.abandon => return Err(error),
1297                Err(_) => {
1298                    (self.idx, self.ri, self.ci) = start;
1299                    self.err = None;
1300                }
1301            }
1302        }
1303
1304        if let Some(token) = self.run_custom_matchers(&mut custom_index, 6_000_000.0, &mut plugin) {
1305            return Ok(token);
1306        }
1307        self.resync(&mut pnt, &mut c);
1308
1309        // 4. Comments (longest opening marker wins; ties sort by name).
1310        let comment_skipped = if self.options.comment.lex && self.wants(TIN_CM) {
1311            match self.run_check(self.options.comment.check.clone(), pnt) {
1312                CheckFlow::Continue => false,
1313                CheckFlow::Skip => true,
1314                CheckFlow::Token(token) => return Ok(*token),
1315            }
1316        } else {
1317            false
1318        };
1319        self.resync(&mut pnt, &mut c);
1320        if self.options.comment.lex
1321            && !comment_skipped
1322            && self.wants(TIN_CM)
1323            && self.listed(entry, first, self.options.comment.check.is_some(), |ch| {
1324                self.options
1325                    .comment
1326                    .definitions
1327                    .values()
1328                    .any(|definition| definition.start.starts_with(ch))
1329            })
1330        {
1331            if let Some(token) = self.match_comment(pnt)? {
1332                return Ok(token);
1333            }
1334        }
1335
1336        if let Some(token) = self.run_custom_matchers(&mut custom_index, 7_000_000.0, &mut plugin) {
1337            return Ok(token);
1338        }
1339        self.resync(&mut pnt, &mut c);
1340
1341        // 5. Numbers
1342        let number_skipped = if self.options.number.lex && self.wants(TIN_NR) {
1343            match self.run_check(self.options.number.check.clone(), pnt) {
1344                CheckFlow::Continue => false,
1345                CheckFlow::Skip => true,
1346                CheckFlow::Token(token) => return Ok(*token),
1347            }
1348        } else {
1349            false
1350        };
1351        self.resync(&mut pnt, &mut c);
1352        let could_start_number =
1353            |ch: char| ch == '-' || ch == '+' || ch == '.' || ch.is_ascii_digit();
1354        if self.options.number.lex
1355            && !number_skipped
1356            && self.wants(TIN_NR)
1357            && c.is_some_and(could_start_number)
1358            && self.listed(
1359                entry,
1360                first,
1361                self.options.number.check.is_some(),
1362                could_start_number,
1363            )
1364        {
1365            if let Some(tkn) = self.match_number(pnt)? {
1366                return Ok(tkn);
1367            }
1368        }
1369
1370        if let Some(token) = self.run_custom_matchers(&mut custom_index, 8_000_000.0, &mut plugin) {
1371            return Ok(token);
1372        }
1373        self.resync(&mut pnt, &mut c);
1374
1375        // 6. Text and named/regex values share the same delimited run.
1376        // Negotiated lexing gates this combined family by its primary token
1377        // identity (#TX), matching the TypeScript and Go dispatchers. Once
1378        // entered, an exact or regexp value definition may still produce
1379        // #VL; the caller rejects and rolls that cut back when #VL was not
1380        // requested.
1381        let text_matcher_wanted = self.wants(TIN_TX);
1382        let value_lex = self.options.value.lex && text_matcher_wanted;
1383        let text_lex = self.options.text.lex && text_matcher_wanted;
1384        let text_skipped = if text_lex || value_lex {
1385            match self.run_check(self.options.text.check.clone(), pnt) {
1386                CheckFlow::Continue => false,
1387                CheckFlow::Skip => true,
1388                CheckFlow::Token(token) => return Ok(*token),
1389            }
1390        } else {
1391            false
1392        };
1393        self.resync(&mut pnt, &mut c);
1394        if (text_lex || value_lex) && !text_skipped && !self.is_text_delimiter_here() {
1395            let start = (self.idx, self.ri, self.ci);
1396            // Only a `value` definition declaring `consume` looks at the
1397            // rest of the document, and the JSON grammar has none -- but
1398            // this ran for every text token, copying the whole tail of the
1399            // input each time. `self.src` is borrowed from the caller for
1400            // `'a` and is never reassigned, so reading the reference out
1401            // before the scan below gives a slice that does not borrow
1402            // `self` and survives the `&mut self` the scan needs.
1403            let source: &'a str = self.src;
1404            let remaining = &source[self.byte_position()..];
1405            let mut src = String::new();
1406            while let Some(ch) = self.peek() {
1407                if self.is_text_delimiter_here() {
1408                    break;
1409                }
1410                src.push(ch);
1411                self.advance();
1412            }
1413
1414            let mut output = None;
1415            if value_lex {
1416                if let Some(definition) = self
1417                    .options
1418                    .value
1419                    .definitions
1420                    .get(&src)
1421                    .filter(|definition| definition.matcher.is_none())
1422                    .cloned()
1423                {
1424                    output = Some(Token::new(
1425                        "#VL",
1426                        TIN_VL,
1427                        definition
1428                            .val
1429                            .clone()
1430                            .unwrap_or_else(|| Value::String(src.clone())),
1431                        src.clone(),
1432                        pnt,
1433                    ));
1434                }
1435
1436                if output.is_none() {
1437                    let mut definitions: Vec<_> = self
1438                        .options
1439                        .value
1440                        .definitions
1441                        .iter()
1442                        .filter(|(_, definition)| definition.matcher.is_some())
1443                        .map(|(name, definition)| (name.clone(), definition.clone()))
1444                        .collect();
1445                    definitions.sort_by(|(name_a, _), (name_b, _)| name_a.cmp(name_b));
1446                    for (_, definition) in definitions {
1447                        let regex = definition.matcher.as_ref().expect("filtered matcher");
1448                        let target: &str = if definition.consume { remaining } else { &src };
1449                        let Some(captures) = regex.captures(target) else {
1450                            continue;
1451                        };
1452                        let Some(found) = captures.get(0).filter(|found| found.start() == 0) else {
1453                            continue;
1454                        };
1455                        if !definition.consume && found.end() != target.len() {
1456                            continue;
1457                        }
1458                        let matched = found.as_str().to_string();
1459                        let value = definition.transform.as_ref().map_or_else(
1460                            || {
1461                                definition
1462                                    .val
1463                                    .clone()
1464                                    .unwrap_or_else(|| Value::String(matched.clone()))
1465                            },
1466                            |transform| {
1467                                let groups = captures
1468                                    .iter()
1469                                    .map(|capture| {
1470                                        capture
1471                                            .map_or_else(String::new, |value| value.as_str().into())
1472                                    })
1473                                    .collect::<Vec<_>>();
1474                                transform(&groups)
1475                            },
1476                        );
1477                        if definition.consume {
1478                            (self.idx, self.ri, self.ci) = start;
1479                            for _ in matched.chars() {
1480                                self.advance();
1481                            }
1482                        }
1483                        output = Some(Token::new("#VL", TIN_VL, value, matched, pnt));
1484                        break;
1485                    }
1486                }
1487            }
1488
1489            if output.is_none() && (!text_lex || text_skipped) {
1490                (self.idx, self.ri, self.ci) = start;
1491            } else if output.is_none() {
1492                output = Some(Token::new(
1493                    "#TX",
1494                    TIN_TX,
1495                    Value::String(src.clone()),
1496                    src,
1497                    pnt,
1498                ));
1499            }
1500
1501            if let Some(mut token) = output {
1502                let value = std::mem::replace(&mut token.val, Value::Undefined);
1503                token.val = self.modify_text_value(value, &mut plugin);
1504                return Ok(token);
1505            }
1506        }
1507
1508        if let Some(token) = self.run_custom_matchers(&mut custom_index, f64::INFINITY, &mut plugin)
1509        {
1510            return Ok(token);
1511        }
1512        self.resync(&mut pnt, &mut c);
1513
1514        // 7. Unclaimed character -> Error: unexpected, raised where the
1515        // cursor stands, as TypeScript's `Lex.next` builds its #BD at the
1516        // live `pnt` and Go's `nextUnfiltered2` at `l.pnt`. A matcher that
1517        // stepped over the last character and declined leaves no character
1518        // to name: the error then names none, at the end of the source, as
1519        // it does in both. This took a character unconditionally, and so
1520        // panicked on a lone byte-order mark under the xml plugin, whose
1521        // matcher steps over a mark at the start of the source.
1522        let bad_source = self.advance().map_or_else(String::new, String::from);
1523        let err = TabnasError::new(
1524            "unexpected",
1525            bad_source,
1526            "",
1527            pnt.site.pos,
1528            pnt.site.ri,
1529            pnt.site.ci,
1530        );
1531        self.err = Some(err.clone());
1532        Err(Box::new(err))
1533    }
1534
1535    fn match_comment(&mut self, pnt: Point) -> LexResult<Option<Token>> {
1536        let remaining = &self.src[self.byte_position()..];
1537        let mut definitions: Vec<_> = self
1538            .options
1539            .comment
1540            .definitions
1541            .iter()
1542            .filter(|(_, definition)| {
1543                // Same first-byte test as the fixed-token scan above.
1544                definition.start.as_bytes().first().copied()
1545                    == remaining.as_bytes().first().copied()
1546                    && !definition.start.is_empty()
1547                    && definition.lex
1548                    && remaining.starts_with(&definition.start)
1549            })
1550            .collect();
1551        definitions.sort_by(|(name_a, a), (name_b, b)| {
1552            b.start
1553                .len()
1554                .cmp(&a.start.len())
1555                .then_with(|| name_a.cmp(name_b))
1556        });
1557        let Some((_, definition)) = definitions.first() else {
1558            return Ok(None);
1559        };
1560        let definition = (*definition).clone();
1561        let mut src = String::new();
1562        for _ in definition.start.chars() {
1563            src.push(self.advance().expect("comment marker must advance"));
1564        }
1565
1566        let mut terminated_by_suffix = false;
1567        let mut closed = definition.line;
1568        loop {
1569            let remainder = &self.src[self.byte_position()..];
1570            let suffix = definition
1571                .suffixes
1572                .iter()
1573                .filter(|suffix| !suffix.is_empty() && remainder.starts_with(*suffix))
1574                .max_by_key(|suffix| suffix.len())
1575                .cloned();
1576            let suffix = suffix.or_else(|| {
1577                let matcher = definition.suffix_matcher.as_ref()?;
1578                let effect = matcher.run(remainder);
1579                if effect.is_some() {
1580                    return effect;
1581                }
1582                let saved = self.state();
1583                let wanted = self.want.clone();
1584                let token = matcher.run_imperative(self);
1585                self.restore(saved);
1586                self.want = wanted;
1587                token.map(|token| token.src.to_string())
1588            });
1589            let remainder = &self.src[self.byte_position()..];
1590            let suffix =
1591                suffix.filter(|suffix| !suffix.is_empty() && remainder.starts_with(suffix));
1592            if let Some(suffix) = suffix {
1593                for _ in suffix.chars() {
1594                    src.push(self.advance().expect("comment suffix must advance"));
1595                }
1596                terminated_by_suffix = true;
1597                closed = true;
1598                break;
1599            }
1600            if !definition.line
1601                && !definition.end.is_empty()
1602                && remainder.starts_with(&definition.end)
1603            {
1604                for _ in definition.end.chars() {
1605                    src.push(self.advance().expect("comment end must advance"));
1606                }
1607                closed = true;
1608                break;
1609            }
1610            let Some(ch) = self.peek() else {
1611                break;
1612            };
1613            if definition.line && (self.char_sets.line_ends.contains(ch)) {
1614                break;
1615            }
1616            src.push(self.advance().expect("comment body must advance"));
1617        }
1618
1619        if !closed {
1620            let err = TabnasError::new(
1621                "unterminated_comment",
1622                src,
1623                "",
1624                pnt.site.pos,
1625                pnt.site.ri,
1626                pnt.site.ci,
1627            );
1628            self.err = Some(err.clone());
1629            return Err(Box::new(err));
1630        }
1631
1632        if definition.eat_line && !terminated_by_suffix {
1633            while let Some(ch) = self.peek() {
1634                if !self.char_sets.line_ends.contains(ch) {
1635                    break;
1636                }
1637                src.push(self.advance().expect("comment line tail must advance"));
1638            }
1639        }
1640
1641        Ok(Some(Token::new(
1642            "#CM",
1643            TIN_CM,
1644            Value::String(src.clone()),
1645            src,
1646            pnt,
1647        )))
1648    }
1649
1650    fn match_number(&mut self, pnt: Point) -> LexResult<Option<Token>> {
1651        let start_idx = self.idx;
1652        let mut src = String::new();
1653
1654        // Optional sign.
1655        if matches!(self.peek(), Some('-' | '+')) {
1656            src.push(self.advance().unwrap());
1657        }
1658
1659        // Base-prefixed integers are complete at the final valid digit.
1660        if self.peek() == Some('0') {
1661            if let Some(prefix) = self.peek_at(1) {
1662                let radix = match prefix {
1663                    'x' | 'X' if self.options.number.hex => Some(16),
1664                    'o' | 'O' if self.options.number.oct => Some(8),
1665                    'b' | 'B' if self.options.number.bin => Some(2),
1666                    _ => None,
1667                };
1668                if let Some(radix) = radix {
1669                    src.push(self.advance().expect("peeked zero"));
1670                    src.push(self.advance().expect("peeked base prefix"));
1671                    let mut saw_digit = false;
1672                    while let Some(ch) = self.peek() {
1673                        if ch.is_digit(radix) {
1674                            saw_digit = true;
1675                            src.push(self.advance().expect("peeked base digit"));
1676                        } else if self
1677                            .options
1678                            .number
1679                            .sep
1680                            .as_ref()
1681                            .is_some_and(|separator| separator.contains(ch))
1682                        {
1683                            src.push(self.advance().expect("peeked base digit"));
1684                        } else {
1685                            break;
1686                        }
1687                    }
1688                    if saw_digit && self.is_text_delimiter_here() {
1689                        if self
1690                            .exclude_regex
1691                            .as_ref()
1692                            .is_some_and(|regex| regex.is_match(&src))
1693                        {
1694                            self.reset_number(start_idx, pnt);
1695                            return Ok(None);
1696                        }
1697                        if self.options.value.lex {
1698                            if let Some(definition) = self
1699                                .options
1700                                .value
1701                                .definitions
1702                                .get(&src)
1703                                .filter(|definition| definition.matcher.is_none())
1704                            {
1705                                return Ok(Some(Token::new(
1706                                    "#VL",
1707                                    TIN_VL,
1708                                    definition
1709                                        .val
1710                                        .clone()
1711                                        .unwrap_or_else(|| Value::String(src.clone())),
1712                                    src,
1713                                    pnt,
1714                                )));
1715                            }
1716                        }
1717                        // The digits of the literal, prefix, sign and any
1718                        // separators removed, folded as they are read. The
1719                        // fold keeps a bounded head, a digit count and a
1720                        // sticky bit, so a literal of any length costs the
1721                        // same handful of bytes: buffering the digits
1722                        // instead would let one long token multiply the
1723                        // memory the source already holds.
1724                        let mut fold = DigitFold::new(radix.trailing_zeros());
1725                        for ch in src.chars().skip_while(|ch| matches!(ch, '-' | '+')).skip(2) {
1726                            if self
1727                                .options
1728                                .number
1729                                .sep
1730                                .as_ref()
1731                                .is_some_and(|separator| separator.contains(ch))
1732                            {
1733                                continue;
1734                            }
1735                            fold.push(ch.to_digit(radix).expect("validated base digit"));
1736                        }
1737                        let mut value = fold.finish();
1738                        if src.starts_with('-') {
1739                            value = -value;
1740                        }
1741                        return Ok(Some(Token::new(
1742                            "#NR",
1743                            TIN_NR,
1744                            Value::Number(value),
1745                            src,
1746                            pnt,
1747                        )));
1748                    }
1749                    self.reset_number(start_idx, pnt);
1750                    return Ok(None);
1751                }
1752            }
1753        }
1754
1755        let Some(ch) = self.peek() else {
1756            self.reset_number(start_idx, pnt);
1757            return Ok(None);
1758        };
1759        if ch == '.' {
1760            if !self.peek_at(1).is_some_and(|next| next.is_ascii_digit()) {
1761                self.reset_number(start_idx, pnt);
1762                return Ok(None);
1763            }
1764            src.push(self.advance().expect("peeked leading decimal point"));
1765        } else if !ch.is_ascii_digit() {
1766            self.reset_number(start_idx, pnt);
1767            return Ok(None);
1768        }
1769
1770        let (has_digits, edge_separator) = self.scan_number_digits(&mut src);
1771        if !has_digits || edge_separator {
1772            self.reset_number(start_idx, pnt);
1773            return Ok(None);
1774        }
1775
1776        // The canonical regexp admits a trailing decimal point and an
1777        // exponent after it (`2.e3`), but declines `0.a` as one text run.
1778        if self.peek() == Some('.') {
1779            let next = self.peek_at(1);
1780            let exponent_after_dot = matches!(next, Some('e' | 'E'))
1781                && match self.peek_at(2) {
1782                    Some('+' | '-') => self.peek_at(3).is_some_and(|ch| ch.is_ascii_digit()),
1783                    Some(ch) => ch.is_ascii_digit(),
1784                    None => false,
1785                };
1786            if next.is_some_and(|ch| ch.is_ascii_digit()) {
1787                src.push(self.advance().expect("peeked decimal point"));
1788                let (_, edge_separator) = self.scan_number_digits(&mut src);
1789                if edge_separator {
1790                    self.reset_number(start_idx, pnt);
1791                    return Ok(None);
1792                }
1793            } else if next.is_some()
1794                && !self.is_text_delimiter_at(self.idx + 1)
1795                && next != Some('.')
1796                && !exponent_after_dot
1797            {
1798                self.reset_number(start_idx, pnt);
1799                return Ok(None);
1800            } else {
1801                src.push(self.advance().expect("peeked trailing decimal point"));
1802            }
1803        }
1804
1805        if matches!(self.peek(), Some('e' | 'E')) {
1806            let exponent_start = self.idx;
1807            let source_len = src.len();
1808            src.push(self.advance().expect("peeked exponent marker"));
1809            if matches!(self.peek(), Some('+' | '-')) {
1810                src.push(self.advance().expect("peeked exponent sign"));
1811            }
1812            let (has_exponent_digits, edge_separator) = self.scan_number_digits(&mut src);
1813            if edge_separator {
1814                self.reset_number(start_idx, pnt);
1815                return Ok(None);
1816            }
1817            if !has_exponent_digits {
1818                self.idx = exponent_start;
1819                src.truncate(source_len);
1820            }
1821        }
1822
1823        if !self.is_text_delimiter_here() {
1824            self.reset_number(start_idx, pnt);
1825            return Ok(None);
1826        }
1827
1828        // Check exclusion regex (e.g. ^00+)
1829        if let Some(ref re) = self.exclude_regex {
1830            if re.is_match(&src) {
1831                // Number is excluded, backtrack
1832                self.reset_number(start_idx, pnt);
1833                return Ok(None);
1834            }
1835        }
1836
1837        if self.options.value.lex {
1838            if let Some(definition) = self
1839                .options
1840                .value
1841                .definitions
1842                .get(&src)
1843                .filter(|definition| definition.matcher.is_none())
1844            {
1845                return Ok(Some(Token::new(
1846                    "#VL",
1847                    TIN_VL,
1848                    definition
1849                        .val
1850                        .clone()
1851                        .unwrap_or_else(|| Value::String(src.clone())),
1852                    src,
1853                    pnt,
1854                )));
1855            }
1856        }
1857
1858        // Parse float
1859        let parse_src = self.options.number.sep.as_ref().map_or_else(
1860            || src.clone(),
1861            |separator| src.chars().filter(|ch| !separator.contains(*ch)).collect(),
1862        );
1863        match parse_src.parse::<f64>() {
1864            Ok(num) => Ok(Some(Token::new(
1865                "#NR",
1866                TIN_NR,
1867                Value::Number(num),
1868                src,
1869                pnt,
1870            ))),
1871            Err(_) => {
1872                self.reset_number(start_idx, pnt);
1873                Ok(None)
1874            }
1875        }
1876    }
1877
1878    fn reset_number(&mut self, start_idx: usize, pnt: Point) {
1879        self.idx = start_idx;
1880        self.ri = pnt.site.ri;
1881        self.ci = pnt.site.ci;
1882    }
1883
1884    /// Consume a decimal digit/separator run. Separators are legal only
1885    /// between digits; a leading or trailing separator makes the whole run
1886    /// fall through to text, matching the TypeScript regexp and Go scanner.
1887    fn scan_number_digits(&mut self, src: &mut String) -> (bool, bool) {
1888        // The run is measured before any of it is consumed. Advancing
1889        // as it goes would hold `&mut self` across a read of
1890        // `self.options.number.sep`, and the way that used to be settled
1891        // was to clone the separator — an allocation and a free for
1892        // every number in the input, for a value that cannot change
1893        // while one number is being scanned.
1894        let run_start = self.idx;
1895        let mut saw_digit = false;
1896        let mut last_was_separator = false;
1897        let mut end = run_start;
1898        {
1899            let separator = self.options.number.sep.as_deref();
1900            while let Some(ch) = self.chars.get(end).copied() {
1901                if ch.is_ascii_digit() {
1902                    saw_digit = true;
1903                    last_was_separator = false;
1904                } else if separator.is_some_and(|separator| separator.contains(ch)) {
1905                    last_was_separator = true;
1906                } else {
1907                    break;
1908                }
1909                end += 1;
1910            }
1911        }
1912        while self.idx < end {
1913            src.push(self.advance().expect("scanned number character"));
1914        }
1915        let starts_with_separator = self.idx > run_start
1916            && self.options.number.sep.as_deref().is_some_and(|separator| {
1917                self.chars[run_start..self.idx]
1918                    .first()
1919                    .is_some_and(|ch| separator.contains(*ch))
1920            });
1921        (saw_digit, starts_with_separator || last_was_separator)
1922    }
1923
1924    fn match_string(&mut self, quote: char, pnt: Point) -> LexResult<Token> {
1925        let quote_char = self.advance().unwrap();
1926        let mut out_str = String::new();
1927        let mut raw_src = String::new();
1928        raw_src.push(quote_char);
1929
1930        let mut pending_high_surrogate: Option<u16> = None;
1931
1932        // The body classes of TypeScript's `buildStringBodySpec`
1933        // (ts/src/lexer.ts), which Go's `BuildStringBodySpec` ports: a
1934        // line character is LINE or LINE+ROW only inside a multi-line
1935        // string, where it resets the column and, in `line.rowChars`,
1936        // counts a row; anywhere else in a body it is plain content,
1937        // counted as a column, unless it is a control character, which
1938        // stops the body as `unprintable`. This loop used to count a row
1939        // for any row character it stepped over, string body or not, and
1940        // to refuse any line character inside a single-line string: with
1941        // json5's U+2028 and U+2029 as row characters, `y` in
1942        // `"a<U+2028>b" y` sat on row 2 here and on row 1 in TypeScript
1943        // and Go, and with the two in `line.chars` as well the string was
1944        // `unprintable` (tabnas/parser#263).
1945        let multi_line = self.options.string.multi_chars.contains(quote);
1946
1947        while let Some(c) = self.peek() {
1948            if c == quote {
1949                raw_src.push(self.advance().unwrap());
1950                // Rust strings cannot represent a lone UTF-16 surrogate, so
1951                // preserve the Go-port behavior and fold it to U+FFFD.
1952                self.flush_surrogate(&mut pending_high_surrogate, &mut out_str);
1953                return Ok(Token::new(
1954                    "#ST",
1955                    TIN_ST,
1956                    Value::String(out_str),
1957                    raw_src,
1958                    pnt,
1959                ));
1960            }
1961
1962            if let Some(replacement) = self.options.string.replace.get(&c).cloned() {
1963                raw_src.push(self.advance().expect("peeked character must advance"));
1964                self.flush_surrogate(&mut pending_high_surrogate, &mut out_str);
1965                out_str.push_str(&replacement);
1966                continue;
1967            }
1968
1969            if multi_line && self.char_sets.line.contains(c) {
1970                raw_src.push(
1971                    self.advance_string_line()
1972                        .expect("peeked character must advance"),
1973                );
1974                out_str.push(c);
1975                continue;
1976            }
1977
1978            // A control character stops the body: a line character
1979            // always, since a single-line string cannot hold one, and any
1980            // other unless `string.allowControl` admits it. Sited ON the
1981            // character, as TypeScript does (`pnt.sI = sI; pnt.cI = cI`
1982            // before its `bad()` call, ts/src/lexer.ts). `pnt` is the
1983            // opening quote, and reporting that put every embedded
1984            // newline at the start of its string.
1985            if (c as u32) < 32
1986                && (self.char_sets.line.contains(c) || !self.options.string.allow_control)
1987            {
1988                let site = self.current_point().site;
1989                let err =
1990                    TabnasError::new("unprintable", c.to_string(), "", site.pos, site.ri, site.ci);
1991                self.err = Some(err.clone());
1992                return Err(Box::new(err));
1993            }
1994
1995            if c == self.options.string.escape_char {
1996                raw_src.push(self.advance().unwrap());
1997                let esc_point = self.current_point();
1998                if let Some(esc) = self.advance() {
1999                    raw_src.push(esc);
2000                    if let Some(replacement) = self.options.string.escape.get(&esc).cloned() {
2001                        self.flush_surrogate(&mut pending_high_surrogate, &mut out_str);
2002                        out_str.push_str(&replacement);
2003                        continue;
2004                    }
2005                    match esc {
2006                        'u' => {
2007                            // Unicode escape: \uXXXX or \u{X...}. An
2008                            // invalid one is reported on the backslash
2009                            // with the span TypeScript cuts: six source
2010                            // characters for the fixed-width form, four
2011                            // for `\x`, and through the closing brace
2012                            // (or to the end of the source) for the
2013                            // braced form -- clipped, never padded, so a
2014                            // truncated escape at end of input reports
2015                            // exactly the characters that are there.
2016                            if self.peek() == Some('{') && !self.options.string.escape_strict {
2017                                raw_src.push(self.advance().unwrap()); // '{'
2018                                let mut hex = String::new();
2019                                let mut closed = false;
2020                                while let Some(h) = self.peek() {
2021                                    if h == '}' {
2022                                        raw_src.push(self.advance().unwrap());
2023                                        closed = true;
2024                                        break;
2025                                    }
2026                                    raw_src.push(self.advance().unwrap());
2027                                    hex.push(h);
2028                                }
2029
2030                                if !closed
2031                                    || hex.is_empty()
2032                                    || hex.len() > 6
2033                                    || !hex.chars().all(|ch| ch.is_ascii_hexdigit())
2034                                {
2035                                    let err = TabnasError::new(
2036                                        "invalid_unicode",
2037                                        self.source_span(esc_point.site.pos - 1, self.idx),
2038                                        "",
2039                                        esc_point.site.pos - 1,
2040                                        esc_point.site.ri,
2041                                        esc_point.site.ci - 1,
2042                                    );
2043                                    self.err = Some(err.clone());
2044                                    return Err(Box::new(err));
2045                                }
2046
2047                                let cp = match u32::from_str_radix(&hex, 16) {
2048                                    Ok(val) if val <= 0x10FFFF => val,
2049                                    _ => {
2050                                        let err = TabnasError::new(
2051                                            "invalid_unicode",
2052                                            self.source_span(esc_point.site.pos - 1, self.idx),
2053                                            "",
2054                                            esc_point.site.pos - 1,
2055                                            esc_point.site.ri,
2056                                            esc_point.site.ci - 1,
2057                                        );
2058                                        self.err = Some(err.clone());
2059                                        return Err(Box::new(err));
2060                                    }
2061                                };
2062
2063                                self.emit_unicode_escape(
2064                                    cp,
2065                                    &mut pending_high_surrogate,
2066                                    &mut out_str,
2067                                );
2068                            } else {
2069                                // Exactly 4 hex digits: \uXXXX
2070                                let mut hex = String::new();
2071                                for _ in 0..4 {
2072                                    if let Some(h) = self.peek() {
2073                                        if h.is_ascii_hexdigit() {
2074                                            raw_src.push(self.advance().unwrap());
2075                                            hex.push(h);
2076                                        } else {
2077                                            break;
2078                                        }
2079                                    } else {
2080                                        break;
2081                                    }
2082                                }
2083
2084                                if hex.len() != 4 {
2085                                    let err = TabnasError::new(
2086                                        "invalid_unicode",
2087                                        self.source_span(
2088                                            esc_point.site.pos - 1,
2089                                            esc_point.site.pos + 5,
2090                                        ),
2091                                        "",
2092                                        esc_point.site.pos - 1,
2093                                        esc_point.site.ri,
2094                                        esc_point.site.ci - 1,
2095                                    );
2096                                    self.err = Some(err.clone());
2097                                    return Err(Box::new(err));
2098                                }
2099
2100                                let cp = u16::from_str_radix(&hex, 16).map_err(|_| {
2101                                    let err = TabnasError::new(
2102                                        "invalid_unicode",
2103                                        self.source_span(
2104                                            esc_point.site.pos - 1,
2105                                            esc_point.site.pos + 5,
2106                                        ),
2107                                        "",
2108                                        esc_point.site.pos - 1,
2109                                        esc_point.site.ri,
2110                                        esc_point.site.ci - 1,
2111                                    );
2112                                    self.err = Some(err.clone());
2113                                    err
2114                                })?;
2115
2116                                self.emit_unicode_escape(
2117                                    u32::from(cp),
2118                                    &mut pending_high_surrogate,
2119                                    &mut out_str,
2120                                );
2121                            }
2122                        }
2123                        'x' if !self.options.string.escape_strict => {
2124                            let mut hex = String::new();
2125                            for _ in 0..2 {
2126                                if let Some(h) = self.peek() {
2127                                    if h.is_ascii_hexdigit() {
2128                                        raw_src.push(
2129                                            self.advance().expect("peeked character must advance"),
2130                                        );
2131                                        hex.push(h);
2132                                    }
2133                                }
2134                            }
2135                            if hex.len() != 2 {
2136                                let err = TabnasError::new(
2137                                    "invalid_ascii",
2138                                    self.source_span(
2139                                        esc_point.site.pos - 1,
2140                                        esc_point.site.pos + 3,
2141                                    ),
2142                                    "",
2143                                    esc_point.site.pos - 1,
2144                                    esc_point.site.ri,
2145                                    esc_point.site.ci - 1,
2146                                );
2147                                self.err = Some(err.clone());
2148                                return Err(Box::new(err));
2149                            }
2150                            let byte = u8::from_str_radix(&hex, 16).expect("validated ASCII hex");
2151                            self.flush_surrogate(&mut pending_high_surrogate, &mut out_str);
2152                            out_str.push(char::from(byte));
2153                        }
2154                        other => {
2155                            if !self.options.string.allow_unknown {
2156                                // Sited on the escape CHARACTER with a
2157                                // one-character span, as TypeScript
2158                                // (`pnt.sI = sI; pnt.cI = cI; lex.bad(
2159                                // S.unexpected, sI, sI + 1)`) and Go do;
2160                                // the other escape errors sit on the
2161                                // backslash and span the construct.
2162                                let err = TabnasError::new(
2163                                    "unexpected",
2164                                    other.to_string(),
2165                                    "",
2166                                    esc_point.site.pos,
2167                                    esc_point.site.ri,
2168                                    esc_point.site.ci,
2169                                );
2170                                self.err = Some(err.clone());
2171                                return Err(Box::new(err));
2172                            }
2173                            self.flush_surrogate(&mut pending_high_surrogate, &mut out_str);
2174                            out_str.push(other);
2175                        }
2176                    }
2177                } else {
2178                    let err = TabnasError::new(
2179                        "unterminated_string",
2180                        raw_src,
2181                        "",
2182                        pnt.site.pos,
2183                        pnt.site.ri,
2184                        pnt.site.ci,
2185                    );
2186                    self.err = Some(err.clone());
2187                    return Err(Box::new(err));
2188                }
2189            } else {
2190                self.flush_surrogate(&mut pending_high_surrogate, &mut out_str);
2191                raw_src.push(self.advance_body().expect("peeked character must advance"));
2192                out_str.push(c);
2193            }
2194        }
2195
2196        let err = TabnasError::new(
2197            "unterminated_string",
2198            raw_src,
2199            "",
2200            pnt.site.pos,
2201            pnt.site.ri,
2202            pnt.site.ci,
2203        );
2204        self.err = Some(err.clone());
2205        Err(Box::new(err))
2206    }
2207
2208    fn flush_surrogate(&self, pending: &mut Option<u16>, out: &mut String) {
2209        if pending.take().is_some() {
2210            out.push('\u{FFFD}');
2211        }
2212    }
2213
2214    /// Emit one decoded Unicode escape while pairing UTF-16 surrogate code
2215    /// units across both `\\uXXXX` and `\\u{...}` spellings.
2216    fn emit_unicode_escape(&self, cp: u32, pending: &mut Option<u16>, out: &mut String) {
2217        if (0xD800..=0xDBFF).contains(&cp) {
2218            self.flush_surrogate(pending, out);
2219            *pending = Some(cp as u16);
2220        } else if (0xDC00..=0xDFFF).contains(&cp) {
2221            if let Some(high) = pending.take() {
2222                let scalar = 0x10000 + (((u32::from(high)) - 0xD800) << 10) + (cp - 0xDC00);
2223                out.push(char::from_u32(scalar).expect("paired surrogates form a Unicode scalar"));
2224            } else {
2225                out.push('\u{FFFD}');
2226            }
2227        } else {
2228            self.flush_surrogate(pending, out);
2229            out.push(char::from_u32(cp).expect("validated escape is a Unicode scalar"));
2230        }
2231    }
2232}
2233
2234// ---------------------------------------------------------------------------
2235// Base-prefixed integer literals.
2236//
2237// A `0x`, `0o` or `0b` literal is read as an EXACT integer and rounded to
2238// a double ONCE. The obvious fold -- `value = value * radix + digit` in
2239// `f64` -- rounds at every digit, and past the 53-bit exact integer range
2240// those roundings accumulate: `0Xa6f2f78f4f9bf44` came out as
2241// `43a4de5ef1e9f37e` where canonical TypeScript and the Go port both
2242// answer `43a4de5ef1e9f37f`, one unit in the last place low. That is
2243// silently altered data, not a formatting difference.
2244//
2245// TypeScript coerces the literal with unary `+`, whose StringNumericValue
2246// is the exact mathematical value of the digits rounded once, half to
2247// even; Go reads it through `big.Int` and `big.Float.Float64()`, which is
2248// the same rule. These reproduce it. Only the VALUE is affected: which
2249// literals are accepted, and the token they become, are settled by
2250// `match_number` before any of this runs.
2251//
2252// The decimal path needs none of it -- `str::parse::<f64>` is correctly
2253// rounded for a digit string of any length.
2254// ---------------------------------------------------------------------------
2255
2256/// `2^k` for a non-negative `k`, exactly, saturating to infinity above the
2257/// double range. A repeated multiply would round on the way up.
2258fn pow2(k: i64) -> f64 {
2259    debug_assert!(k >= 0, "only non-negative exponents arise here");
2260    if k > 1023 {
2261        f64::INFINITY
2262    } else {
2263        f64::from_bits(((k + 1023) as u64) << 52)
2264    }
2265}
2266
2267/// Folds the digits of a base-prefixed literal into the NEAREST double,
2268/// rounding half to even, without holding the digits.
2269///
2270/// `bits` is the width of one digit, so the base is a power of two: 1 for
2271/// binary, 3 for octal, 4 for hexadecimal. Those are the only bases
2272/// `match_number` reads, which is what lets a `u128` head plus a sticky
2273/// bit stand in for arbitrary-precision arithmetic.
2274///
2275/// Only three things about a literal can change the answer: the top
2276/// `128 / bits` significant digits, how many digits follow them, and
2277/// whether any of those is non-zero. This keeps exactly those, so the
2278/// space it costs does not grow with the literal, however long an
2279/// untrusted document makes one.
2280struct DigitFold {
2281    bits: u32,
2282    /// The significant digits packed so far, at most `head_len` of them.
2283    head: u128,
2284    /// How many significant digits have been pushed, head and tail alike.
2285    len: usize,
2286    /// Whether any digit past the head was non-zero.
2287    sticky: bool,
2288    /// Whether a non-zero digit has been seen. Leading zeros carry no
2289    /// value, and dropping them is what makes the head wider than the 54
2290    /// significant bits the rounding needs.
2291    started: bool,
2292}
2293
2294impl DigitFold {
2295    fn new(bits: u32) -> Self {
2296        debug_assert!(
2297            (1..=4).contains(&bits),
2298            "only the power-of-two bases the lexer reads"
2299        );
2300        DigitFold {
2301            bits,
2302            head: 0,
2303            len: 0,
2304            sticky: false,
2305            started: false,
2306        }
2307    }
2308
2309    /// A `u128` holds exactly this many digits of the base.
2310    fn head_len(&self) -> usize {
2311        (128 / self.bits) as usize
2312    }
2313
2314    fn push(&mut self, digit: u32) {
2315        if !self.started {
2316            if 0 == digit {
2317                return;
2318            }
2319            self.started = true;
2320        }
2321        if self.len < self.head_len() {
2322            self.head = (self.head << self.bits) | u128::from(digit);
2323        } else if 0 != digit {
2324            self.sticky = true;
2325        }
2326        self.len += 1;
2327    }
2328
2329    fn finish(&self) -> f64 {
2330        if !self.started {
2331            return 0.0;
2332        }
2333        let head_len = self.head_len();
2334        if self.len <= head_len {
2335            // A `u128` to `f64` cast rounds to nearest, ties to even, which
2336            // is the rule the canonical runtime follows.
2337            return self.head as f64;
2338        }
2339
2340        // Longer than a u128: the head holds the top `head_len` digits and
2341        // `sticky` remembers whether anything below them was set. Those two
2342        // are all the rounding can depend on. The leading digit is
2343        // non-zero, so the head is at least 121 bits wide in every base
2344        // here and `shift` is comfortably positive.
2345        let dropped = i64::from(self.bits) * (self.len - head_len) as i64;
2346        let shift = 128 - self.head.leading_zeros() - 53;
2347
2348        let mut mantissa = (self.head >> shift) as u64;
2349        let half = (self.head >> (shift - 1)) & 1 == 1;
2350        let sticky = self.head & ((1u128 << (shift - 1)) - 1) != 0 || self.sticky;
2351        if half && (sticky || mantissa & 1 == 1) {
2352            // At most 2^53, which is still an exact double.
2353            mantissa += 1;
2354        }
2355        // The mantissa carries at most 53 significant bits, so the scaling
2356        // is exact inside the double range and overflows to infinity
2357        // outside it.
2358        mantissa as f64 * pow2(dropped + i64::from(shift))
2359    }
2360}