Skip to main content

tabnas/
lexer.rs

1// Copyright (c) 2013-2026 Richard Rodger, MIT License
2
3use crate::error::TabnasError;
4use crate::options::{LexCheck, LexCheckResult, MatchTokenMatcher, Options};
5use crate::token::{
6    Point, Token, TIN_BD, TIN_CM, TIN_LN, TIN_NR, TIN_SP, TIN_ST, TIN_TX, TIN_VL, TIN_ZZ,
7};
8use crate::value::Value;
9use regex::Regex;
10use std::panic::{catch_unwind, AssertUnwindSafe};
11use std::sync::Arc;
12
13pub struct Lexer<'a> {
14    src: &'a str,
15    chars: Vec<char>,
16    byte_indices: Vec<usize>,
17    char_len: usize,
18    idx: usize,
19    ri: usize,
20    ci: usize,
21    options: Arc<Options>,
22    ignore_tins: Vec<crate::Tin>,
23    char_sets: crate::text::CharSets,
24    err: Option<TabnasError>,
25    end_reached: bool,
26    /// `number.exclude`, compiled once per grammar and shared by the
27    /// parser that owns it; see [`compile_number_exclude`].
28    exclude_regex: Option<Arc<Regex>>,
29    want: Option<Vec<crate::Tin>>,
30    standalone: Option<(crate::Rule, crate::Context)>,
31}
32
33#[derive(Clone)]
34pub(crate) struct LexerState {
35    idx: usize,
36    ri: usize,
37    ci: usize,
38    err: Option<TabnasError>,
39    end_reached: bool,
40}
41
42/// Opaque snapshot returned by [`Lexer::relex_for_rule`]. Pass it to
43/// [`Lexer::unrelex`] if the caller later rejects the committed recut.
44#[derive(Clone)]
45pub struct RelexCheckpoint {
46    state: LexerState,
47    replay: std::collections::VecDeque<Token>,
48}
49
50enum CheckFlow {
51    Continue,
52    Skip,
53    Token(Box<Token>),
54}
55
56/// The result the lexer passes through its own internals.
57///
58/// `TabnasError` is 664 bytes, so `LexResult<Token>` is 664 bytes
59/// too: three times the token it carries, and moved on the SUCCESS path
60/// through every frame of the lexer chain. Boxing the error inside the
61/// engine makes those results the size of a token again, and costs an
62/// allocation only when there is an error to report, which is the path that
63/// is already building a 664-byte diagnostic.
64///
65/// The public entry points still hand back an unboxed `TabnasError`, so no
66/// caller of this crate sees the box.
67type LexResult<T> = Result<T, Box<TabnasError>>;
68
69/// The compiled form of `number.exclude`, or `None` when there is no
70/// pattern or it does not compile (an invalid pattern excludes nothing,
71/// as it always has).
72///
73/// Compiling a regex costs about 700K instructions, which is more than
74/// a small document costs to parse, so the parser compiles it once per
75/// grammar and hands every lexer the same automaton through the `Arc`.
76/// Cloning a `Regex` would not do: it shares the automaton but builds a
77/// fresh scratch-cache pool, and the first search from each clone fills
78/// it. Go and TypeScript compile the pattern once, at configuration time.
79pub(crate) fn compile_number_exclude(options: &Options) -> Option<Arc<Regex>> {
80    let pattern = options.number.exclude.as_deref()?;
81    Regex::new(pattern).ok().map(Arc::new)
82}
83
84impl<'a> Lexer<'a> {
85    pub fn new(src: &'a str, mut options: Options) -> Self {
86        if let Err(error) = options.validate_comment_definitions() {
87            panic!("invalid options: {error}");
88        }
89        // A lexer built directly may be handed options nobody has
90        // ordered yet. The parser's own lexer comes through
91        // `with_shared`, whose options were ordered when they were
92        // prepared, and whose exclude pattern was compiled then too.
93        options.sort_for_lexing();
94        let exclude_regex = compile_number_exclude(&options);
95        Self::with_shared(src, Arc::new(options), exclude_regex)
96    }
97
98    /// Lex against options the parser already owns and has ordered, with
99    /// the `number.exclude` pattern it compiled from them.
100    pub(crate) fn with_shared(
101        src: &'a str,
102        options: Arc<Options>,
103        exclude_regex: Option<Arc<Regex>>,
104    ) -> Self {
105        let mut chars = Vec::new();
106        let mut byte_indices = Vec::new();
107        for (b_idx, c) in src.char_indices() {
108            chars.push(c);
109            byte_indices.push(b_idx);
110        }
111        let char_len = chars.len();
112
113        Lexer {
114            src,
115            chars,
116            byte_indices,
117            char_len,
118            idx: 0,
119            ri: 1,
120            ci: 1,
121            ignore_tins: options.ignore_tins(),
122            char_sets: options.char_sets(),
123            options,
124            err: None,
125            end_reached: false,
126            exclude_regex,
127            want: None,
128            standalone: None,
129        }
130    }
131
132    fn current_point(&self) -> Point {
133        Point {
134            len: self.src.len(),
135            site: crate::Site {
136                si: self.byte_position(),
137                pos: self.idx,
138                ri: self.ri,
139                ci: self.ci,
140            },
141        }
142    }
143
144    /// The source from char index `start` up to `end`, clipped to the end
145    /// of the source: the span of a bad token, cut as TypeScript's
146    /// `lex.bad(why, pstart, pend)` cuts it (`ts/src/lexer.ts`).
147    ///
148    /// Only the END is clipped. The caller must supply `start <= end`
149    /// and `start <= self.char_len`, or the indexing panics. Every call
150    /// site below passes `esc_point.site.pos - 1`, which additionally
151    /// needs `1 <= pos`; that holds because `esc_point` is captured
152    /// after the escape character has been consumed. These three are
153    /// the preconditions `rs/verus/lexer_span.rs` ASSUMES: it proves the
154    /// span arithmetic is in bounds given them, and does not model the
155    /// call sites, so nothing machine-checks that they hold here. A
156    /// refactor that captured the point before the advance would still
157    /// pass `rs/verus/run.sh`.
158    fn source_span(&self, start: usize, end: usize) -> String {
159        self.chars[start..end.min(self.char_len)].iter().collect()
160    }
161
162    fn state(&self) -> LexerState {
163        LexerState {
164            idx: self.idx,
165            ri: self.ri,
166            ci: self.ci,
167            err: self.err.clone(),
168            end_reached: self.end_reached,
169        }
170    }
171
172    /// Full immutable source supplied to this lexer.
173    pub fn source(&self) -> &str {
174        self.src
175    }
176
177    /// Source remaining at the live cursor.
178    pub fn remaining(&self) -> &str {
179        &self.src[self.byte_position()..]
180    }
181
182    /// Return at most `max_chars` Unicode scalar values from the live cursor.
183    /// This is the Rust counterpart of the public `lex.fwd`/`Lex.Fwd` helper.
184    pub fn forward(&self, max_chars: usize) -> &str {
185        let remaining = self.remaining();
186        let end = remaining
187            .char_indices()
188            .nth(max_chars)
189            .map_or(remaining.len(), |(index, _)| index);
190        &remaining[..end]
191    }
192
193    /// Snapshot the live cursor for token construction.
194    pub fn point(&self) -> Point {
195        self.current_point()
196    }
197
198    /// Advance by Unicode scalar values. Returns false without moving when
199    /// the requested count extends beyond end-of-source.
200    pub fn advance_chars(&mut self, count: usize) -> bool {
201        if self.idx.saturating_add(count) > self.char_len {
202            return false;
203        }
204        for _ in 0..count {
205            self.advance();
206        }
207        true
208    }
209
210    /// Construct a token from a point captured before cursor advancement.
211    pub fn token(
212        &self,
213        name: impl AsRef<str>,
214        tin: crate::Tin,
215        value: Value,
216        source: impl Into<crate::TokenText>,
217        point: Point,
218    ) -> Token {
219        Token::new(name, tin, value, source, point)
220    }
221
222    /// Resolve or allocate a token identity in this lexer's configuration.
223    pub fn token_tin(&mut self, name: impl Into<String>) -> crate::Tin {
224        Arc::make_mut(&mut self.options).register_token(name)
225    }
226
227    /// Resolve a token identity back to its configured name.
228    pub fn token_name(&self, tin: crate::Tin) -> String {
229        self.options.token_name(tin)
230    }
231
232    /// Whether the options this lexer runs under enable `alt`: not when
233    /// `rule.exclude` names one of its groups, nor when `rule.include`
234    /// lists groups and it declares none of them. The parser skips such an
235    /// alternate, and TypeScript removes it from the rule spec
236    /// (`filterRules`) before anything reads the spec, so a custom matcher
237    /// that reads a rule's alternates, to tell a key position from a value
238    /// one, asks this to see the same alternates TypeScript does.
239    pub fn alt_enabled(&self, alt: &crate::AltSpec) -> bool {
240        crate::parser::groups_enabled(alt, &self.options)
241    }
242
243    /// Construct a bad token at the current cursor.
244    ///
245    /// A custom matcher that returns it reports a lexer fault, which the
246    /// parser handles as it handles its own lexer's: a rule's fetch raises
247    /// it at once with `why` as the code, at this position, or, under
248    /// recovery, records it, coalescing a run of them into one error, and
249    /// moves the cursor past its source; relexing leaves it for an
250    /// alternate to re-cut. The cursor need not be advanced here.
251    pub fn bad(&self, why: impl Into<String>) -> Token {
252        let point = self.current_point();
253        let source = self
254            .peek()
255            .map_or_else(String::new, |character| character.to_string());
256        let mut token = Token::new("#BD", TIN_BD, Value::Undefined, source, point);
257        token.err = crate::TokenCode::from(why.into());
258        token.why = token.err.clone();
259        token
260    }
261
262    /// Construct a bad token whose displayed source is a scalar-indexed span.
263    /// As in TypeScript, the diagnostic point remains the live cursor.
264    /// Returned from a matcher it is handled as [`Lexer::bad`] describes;
265    /// recovery steps past the span, counted from that cursor.
266    pub fn bad_span(&self, why: impl Into<String>, start: usize, end: usize) -> Token {
267        let point = self.current_point();
268        let source = if start <= end && end <= self.char_len {
269            let start_byte = self
270                .byte_indices
271                .get(start)
272                .copied()
273                .unwrap_or(self.src.len());
274            let end_byte = self
275                .byte_indices
276                .get(end)
277                .copied()
278                .unwrap_or(self.src.len());
279            self.src[start_byte..end_byte].to_string()
280        } else {
281            self.peek()
282                .map_or_else(String::new, |character| character.to_string())
283        };
284        let mut token = Token::new("#BD", TIN_BD, Value::Undefined, source, point);
285        token.err = crate::TokenCode::from(why.into());
286        token.why = token.err.clone();
287        token
288    }
289
290    fn byte_position(&self) -> usize {
291        self.byte_indices
292            .get(self.idx)
293            .copied()
294            .unwrap_or(self.src.len())
295    }
296
297    fn advance(&mut self) -> Option<char> {
298        if self.idx < self.char_len {
299            let c = self.chars[self.idx];
300            self.idx += 1;
301            if self.char_sets.row.contains(c) {
302                self.ri += 1;
303                self.ci = 1;
304            } else {
305                self.ci += 1;
306            }
307            Some(c)
308        } else {
309            None
310        }
311    }
312
313    fn peek(&self) -> Option<char> {
314        if self.idx < self.char_len {
315            Some(self.chars[self.idx])
316        } else {
317            None
318        }
319    }
320
321    fn peek_at(&self, offset: usize) -> Option<char> {
322        let i = self.idx + offset;
323        if i < self.char_len {
324            Some(self.chars[i])
325        } else {
326            None
327        }
328    }
329
330    fn wants(&self, tin: crate::Tin) -> bool {
331        self.want
332            .as_ref()
333            .is_none_or(|wanted| wanted.contains(&tin))
334    }
335
336    fn run_check(&mut self, check: Option<LexCheck>, point: Point) -> CheckFlow {
337        let Some(check) = check else {
338            return CheckFlow::Continue;
339        };
340        let remaining = &self.src[self.byte_position()..];
341        let result = check
342            .run_imperative(self)
343            .or_else(|| check.run(remaining))
344            .unwrap_or(LexCheckResult::Continue);
345        match result {
346            LexCheckResult::Continue => CheckFlow::Continue,
347            LexCheckResult::Skip => CheckFlow::Skip,
348            LexCheckResult::NativeToken(token) => CheckFlow::Token(token),
349            LexCheckResult::Token(token)
350                if !token.source.is_empty() && remaining.starts_with(&token.source) =>
351            {
352                let tin = if token.tin < 0 {
353                    self.options.token(&token.name).unwrap_or(token.tin)
354                } else {
355                    token.tin
356                };
357                if tin < 0 {
358                    return CheckFlow::Skip;
359                }
360                for _ in token.source.chars() {
361                    self.advance();
362                }
363                CheckFlow::Token(Box::new(Token::new(
364                    token.name,
365                    tin,
366                    token.value,
367                    token.source,
368                    point,
369                )))
370            }
371            LexCheckResult::Token(_) => CheckFlow::Skip,
372        }
373    }
374
375    /// Give any plugin matcher whose order is below `before` its turn.
376    ///
377    /// Nine sites in the lexer call this per token, once at each stage a
378    /// matcher is allowed to intervene. A grammar with no custom matcher,
379    /// which is most of them, was paying nine index lookups per token to be
380    /// told nine times that there is nothing to run. The guard is inline so
381    /// those sites skip the call itself; the walk stays out of line.
382    #[inline]
383    fn run_custom_matchers(
384        &mut self,
385        index: &mut usize,
386        before: f64,
387        point: Point,
388        plugin: &mut Option<(&mut crate::Rule, &mut crate::Context)>,
389    ) -> Option<Token> {
390        if *index >= self.options.lex.matchers.len() {
391            return None;
392        }
393        self.run_remaining_custom_matchers(index, before, point, plugin)
394    }
395
396    #[inline(never)]
397    fn run_remaining_custom_matchers(
398        &mut self,
399        index: &mut usize,
400        before: f64,
401        point: Point,
402        plugin: &mut Option<(&mut crate::Rule, &mut crate::Context)>,
403    ) -> Option<Token> {
404        while let Some(matcher) = self
405            .options
406            .lex
407            .matchers
408            .get_index(*index)
409            .map(|(_, matcher)| matcher)
410            .filter(|matcher| matcher.order < before)
411            .cloned()
412        {
413            *index += 1;
414            let remaining = &self.src[self.byte_position()..];
415            let saved = self.state();
416            let token = if let Some(callback) = matcher.imperative.as_ref() {
417                let Some((rule, context)) = plugin.as_mut() else {
418                    continue;
419                };
420                callback(self, rule, context)
421            } else {
422                matcher
423                    .matcher
424                    .as_ref()
425                    .and_then(|callback| callback(remaining))
426                    .filter(|token| {
427                        !token.source.is_empty() && remaining.starts_with(&token.source)
428                    })
429                    .map(|token| {
430                        Token::new(token.name, token.tin, token.value, token.source, point)
431                    })
432            };
433            let Some(mut token) = token else {
434                if self.want.is_some() {
435                    self.restore(saved);
436                }
437                continue;
438            };
439
440            // TypeScript and Go run opaque custom matchers speculatively for
441            // a negotiated cut. An unwanted result rolls back locally so a
442            // later matcher can still satisfy the request.
443            let tin = if token.tin < 0 {
444                self.options.token(&token.name).unwrap_or(token.tin)
445            } else {
446                token.tin
447            };
448            if tin < 0 || !self.wants(tin) {
449                self.restore(saved);
450                continue;
451            }
452            if matcher.imperative.is_none() {
453                for _ in token.src.chars() {
454                    self.advance();
455                }
456            }
457            token.tin = tin;
458            return Some(token);
459        }
460        None
461    }
462
463    fn is_text_delimiter_here(&self) -> bool {
464        self.is_text_delimiter_at(self.idx)
465    }
466
467    fn is_text_delimiter_at(&self, index: usize) -> bool {
468        let Some(ch) = self.chars.get(index).copied() else {
469            return true;
470        };
471        let remaining = &self.src[self.byte_indices[index]..];
472        (self.options.space.lex && self.char_sets.space.contains(ch))
473            || (self.options.fixed.lex
474                && self
475                    .options
476                    .fixed
477                    .tokens
478                    .values()
479                    .any(|token| !token.source.is_empty() && remaining.starts_with(&token.source)))
480            || (self.options.line.lex
481                && (self.char_sets.line_ends.contains(ch) || matches!(ch, '\u{2028}' | '\u{2029}')))
482            || (self.options.comment.lex
483                && self.options.comment.definitions.values().any(|definition| {
484                    definition.lex
485                        && !definition.start.is_empty()
486                        && remaining.starts_with(&definition.start)
487                }))
488            || self
489                .options
490                .ender
491                .iter()
492                .any(|ender| !ender.is_empty() && remaining.starts_with(ender))
493    }
494
495    /// Fetches the next non-IGNORE token (skipping spaces, lines, comments).
496    pub fn next_token(&mut self) -> Result<Token, TabnasError> {
497        let point = self.current_point();
498        let result = match catch_unwind(AssertUnwindSafe(|| {
499            if let Some(ref error) = self.err {
500                return Err(Box::new(error.clone()));
501            }
502
503            loop {
504                let token = self.next_raw(None)?;
505                if !self.ignore_tins.contains(&token.tin) {
506                    return Ok(token);
507                }
508            }
509        })) {
510            Ok(result) => result,
511            Err(payload) => self.record_panic(payload, "Lexer::next_token", point),
512        };
513        // The box is internal to the engine; a caller gets the error itself.
514        result.map_err(|error| *error)
515    }
516
517    /// Fetch the next token without discarding whitespace, line, or comment tokens.
518    pub fn next_raw_token(&mut self) -> Result<Token, TabnasError> {
519        let point = self.current_point();
520        let result = match catch_unwind(AssertUnwindSafe(|| self.next_raw(None))) {
521            Ok(result) => result,
522            Err(payload) => self.record_panic(payload, "Lexer::next_raw_token", point),
523        };
524        result.map_err(|error| *error)
525    }
526
527    /// Fetch one token for an imperative parser callback, preserving ignored
528    /// space/line/comment tokens just like TypeScript's public `lex.next`.
529    /// Replayed tokens produced by `Context::rewind` are served first.
530    pub fn next_raw_for_rule(
531        &mut self,
532        rule: &mut crate::Rule,
533        context: &mut crate::Context,
534    ) -> LexResult<Token> {
535        if let Some(token) = context.next_replay() {
536            Ok(token)
537        } else {
538            self.next_raw_with(None, Some((rule, context)))
539        }
540    }
541
542    /// Fetch the next non-ignored token for an imperative parser callback.
543    pub fn next_for_rule(
544        &mut self,
545        rule: &mut crate::Rule,
546        context: &mut crate::Context,
547    ) -> LexResult<Token> {
548        loop {
549            let token = self.next_raw_for_rule(rule, context)?;
550            if !self.ignore_tins.contains(&token.tin) {
551                return Ok(token);
552            }
553        }
554    }
555
556    /// Public negotiated-relex entry point for native parser callbacks.
557    /// A successful recut commits the lexer cursor and returns an opaque undo
558    /// checkpoint; a failed recut restores all lexer state before returning.
559    pub fn relex_for_rule(
560        &mut self,
561        from: &Token,
562        wanted: &[crate::Tin],
563        rule: &mut crate::Rule,
564        context: &mut crate::Context,
565    ) -> Option<(Token, RelexCheckpoint)> {
566        self.relex(from, wanted, rule, context)
567    }
568
569    /// Undo a committed [`Lexer::relex_for_rule`] operation, including the
570    /// pending tokens hidden while the replacement cut was negotiated.
571    pub fn unrelex(&mut self, checkpoint: RelexCheckpoint, context: &mut crate::Context) {
572        self.restore(checkpoint.state);
573        context.restore_replay(checkpoint.replay);
574    }
575
576    fn record_panic(
577        &mut self,
578        payload: Box<dyn std::any::Any + Send>,
579        api: &str,
580        point: Point,
581    ) -> LexResult<Token> {
582        let error = TabnasError::from_panic(
583            payload,
584            api,
585            self.src,
586            point.site.pos,
587            point.site.ri,
588            point.site.ci,
589            &self.options,
590        );
591        self.err = Some(error.clone());
592        Err(Box::new(error))
593    }
594
595    /// Fetch a raw token while restricting non-eager custom token matchers to
596    /// the exact tins accepted at the parser slot being filled. Builtin and
597    /// fixed-token matchers are unaffected by this gate.
598    pub(crate) fn next_rule_token(
599        &mut self,
600        expected_match_tins: &[crate::Tin],
601        rule: &mut crate::Rule,
602        context: &mut crate::Context,
603    ) -> LexResult<Token> {
604        self.next_raw_with(Some(expected_match_tins), Some((rule, context)))
605    }
606
607    /// Step past a bad token the parser has absorbed or skipped, as
608    /// TypeScript's `advanceLexPast` does (ts/src/rules.ts): a bad token
609    /// does not advance the cursor by itself, so recovery moves it to the
610    /// end of the token's span, never backwards, and, for a fault raised
611    /// inside a compound construct, on past the next row character so
612    /// lexing resumes on a fresh row. The lexer's own faults latch until
613    /// this clears them.
614    pub(crate) fn skip_bad(&mut self, token: &Token, to_line_end: bool) {
615        let span = token.src.chars().count().max(1);
616        let mut target = self.idx.max(token.site.pos.saturating_add(span));
617        if to_line_end {
618            let mut end = target;
619            while end < self.char_len && !self.char_sets.row.contains(self.chars[end]) {
620                end += 1;
621            }
622            target = target.max(self.char_len.min(end + 1));
623        }
624        while self.idx < target && self.idx < self.char_len {
625            self.advance();
626        }
627        self.err = None;
628        if self.idx < self.char_len {
629            self.end_reached = false;
630        }
631    }
632
633    /// Re-cut an already buffered source span, constrained to the token
634    /// identities requested by one alternate. On success the cursor remains
635    /// after the new cut; the returned state can restore the original cut if
636    /// that alternate later fails.
637    pub(crate) fn relex(
638        &mut self,
639        from: &Token,
640        wanted: &[crate::Tin],
641        rule: &mut crate::Rule,
642        context: &mut crate::Context,
643    ) -> Option<(Token, RelexCheckpoint)> {
644        if from.src.is_empty() || from.site.pos > self.char_len || wanted.is_empty() {
645            return None;
646        }
647        // The standing error, if any, is moved into the checkpoint rather
648        // than copied: the cut below clears it anyway, and a copy is the
649        // length of the source (`full_source`) for every cut attempted.
650        let err = self.err.take();
651        let saved = LexerState {
652            err,
653            ..self.state()
654        };
655        // TypeScript temporarily replaces the lexer's pending-token queue
656        // with an empty queue for a negotiated cut. Rust keeps that queue on
657        // Context, so hide it explicitly and preserve it in the checkpoint.
658        let replay = context.take_replay();
659        self.idx = from.site.pos;
660        self.ri = from.site.ri;
661        self.ci = from.site.ci;
662        self.err = None;
663        self.end_reached = false;
664        self.want = Some(wanted.to_vec());
665        // Straight to the matchers, past `next_raw_with`: an error here only
666        // rejects the cut, and the restore below puts back the lexer's own,
667        // so the source it would attach, and the copy it would keep, are
668        // never seen. With them, every rejected cut cost the length of the
669        // source, and a flat stylesheet parsed in quadratic time.
670        let recut = self.next_raw_inner(None, Some((rule, context))).ok();
671        self.want = None;
672        match recut.filter(|token| wanted.contains(&token.tin)) {
673            Some(mut token) => {
674                token.ignored = from.ignored.clone();
675                Some((
676                    token,
677                    RelexCheckpoint {
678                        state: saved,
679                        replay,
680                    },
681                ))
682            }
683            None => {
684                self.restore(saved);
685                // Discard any speculative replay generated by an imperative
686                // matcher and restore the queue that preceded the attempt.
687                context.restore_replay(replay);
688                None
689            }
690        }
691    }
692
693    pub(crate) fn restore(&mut self, state: LexerState) {
694        self.idx = state.idx;
695        self.ri = state.ri;
696        self.ci = state.ci;
697        self.err = state.err;
698        self.end_reached = state.end_reached;
699        self.want = None;
700    }
701
702    fn next_raw(&mut self, expected_match_tins: Option<&[crate::Tin]>) -> LexResult<Token> {
703        // Only a lexer being driven directly needs these, and building
704        // them costs a whole `Options` clone. A parse reaches the lexer
705        // through `next_rule_token`, which brings the real rule and
706        // context with it, so it never wants them at all.
707        let (mut rule, mut context) = match self.standalone.take() {
708            Some(pair) => pair,
709            None => (
710                crate::Rule::new("#NORULE", Value::Undefined),
711                crate::Context::new(
712                    self.options.rewind.history,
713                    self.src,
714                    Value::Undefined,
715                    Arc::clone(&self.options),
716                    crate::InstanceInfo::default(),
717                ),
718            ),
719        };
720        let result = self.next_raw_with(expected_match_tins, Some((&mut rule, &mut context)));
721        self.standalone = Some((rule, context));
722        result
723    }
724
725    fn modify_text_value(
726        &mut self,
727        mut value: Value,
728        plugin: &mut Option<(&mut crate::Rule, &mut crate::Context)>,
729    ) -> Value {
730        if self.options.text.modify.is_empty() {
731            return value;
732        }
733        let modifiers = self.options.text.modify.clone();
734        let options = self.options.clone();
735        let Some((rule, context)) = plugin.as_mut() else {
736            panic!("imperative text modifier requires an active lexer context");
737        };
738        for modifier in modifiers {
739            value = modifier.run(value, self, rule, context, &options);
740        }
741        value
742    }
743
744    fn next_raw_with(
745        &mut self,
746        expected_match_tins: Option<&[crate::Tin]>,
747        plugin: Option<(&mut crate::Rule, &mut crate::Context)>,
748    ) -> LexResult<Token> {
749        let result = self.next_raw_inner(expected_match_tins, plugin);
750        match result {
751            Ok(token) => Ok(token),
752            Err(mut error) => {
753                // The matchers build their errors without the source, and it
754                // is attached here, where an error leaves the lexer: a
755                // negotiated cut (`relex`) rejects many candidates, each an
756                // error nobody sees, and a copy of the whole source apiece
757                // made that quadratic.
758                error.full_source = self.src.to_string();
759                error.apply_options(&self.options);
760                self.err = Some((*error).clone());
761                Err(error)
762            }
763        }
764    }
765
766    fn next_raw_inner(
767        &mut self,
768        expected_match_tins: Option<&[crate::Tin]>,
769        mut plugin: Option<(&mut crate::Rule, &mut crate::Context)>,
770    ) -> LexResult<Token> {
771        if self.end_reached {
772            return Ok(Token::new(
773                "#ZZ",
774                TIN_ZZ,
775                Value::Undefined,
776                "",
777                self.current_point(),
778            ));
779        }
780
781        if self.idx >= self.char_len {
782            self.end_reached = true;
783            return Ok(Token::new(
784                "#ZZ",
785                TIN_ZZ,
786                Value::Undefined,
787                "",
788                self.current_point(),
789            ));
790        }
791
792        let pnt = self.current_point();
793        let c = self.peek().unwrap();
794        let mut custom_index = 0;
795
796        if let Some(token) =
797            self.run_custom_matchers(&mut custom_index, 1_000_000.0, pnt, &mut plugin)
798        {
799            return Ok(token);
800        }
801
802        // User-declared match tokens occupy the 1e6 matcher priority band.
803        let match_skipped = if self.options.match_lex
804            && (!self.options.match_values.is_empty()
805                || self
806                    .options
807                    .match_tokens
808                    .values()
809                    .any(|matcher| self.wants(matcher.tin)))
810        {
811            match self.run_check(self.options.match_check.clone(), pnt) {
812                CheckFlow::Continue => false,
813                CheckFlow::Skip => true,
814                CheckFlow::Token(token) => return Ok(*token),
815            }
816        } else {
817            false
818        };
819        let remaining = &self.src[self.byte_position()..];
820        let custom_value = (self.options.match_lex && !match_skipped && self.want.is_none())
821            .then(|| {
822                self.options
823                    .match_values
824                    .values()
825                    .find_map(|matcher| match &matcher.matcher {
826                        MatchTokenMatcher::Callback(callback) => callback(remaining)
827                            .filter(|result| {
828                                !result.source.is_empty() && remaining.starts_with(&result.source)
829                            })
830                            .map(|result| (result.source, result.value)),
831                        MatchTokenMatcher::Regex(regex) => {
832                            let captures = regex.captures(remaining)?;
833                            let found = captures
834                                .get(0)
835                                .filter(|found| found.start() == 0 && !found.as_str().is_empty())?;
836                            let source = found.as_str().to_string();
837                            let value = matcher.transform.as_ref().map_or_else(
838                                || {
839                                    matcher
840                                        .val
841                                        .clone()
842                                        .unwrap_or_else(|| Value::String(source.clone()))
843                                },
844                                |transform| {
845                                    let groups = captures
846                                        .iter()
847                                        .map(|capture| {
848                                            capture.map_or_else(String::new, |value| {
849                                                value.as_str().into()
850                                            })
851                                        })
852                                        .collect::<Vec<_>>();
853                                    transform(&groups)
854                                },
855                            );
856                            Some((source, value))
857                        }
858                    })
859            })
860            .flatten();
861        if let Some((source, value)) = custom_value {
862            for _ in source.chars() {
863                self.advance();
864            }
865            return Ok(Token::new("#VL", TIN_VL, value, source, pnt));
866        }
867
868        let remaining = &self.src[self.byte_position()..];
869        // With no custom matcher there is nothing for the band to do: both
870        // passes walk an empty table and yield nothing, and `fix_len` is
871        // read only by that walk. Most grammars register none, and every
872        // token fetch of theirs paid the eager pass's scan of the fixed
873        // table (one closure call per fixed literal) to arrive at the
874        // `None` this guard now hands over directly. TS `makeMatchMatcher`
875        // returns null on an empty table (ts/src/lexer.ts) and the band is
876        // never installed; Go reaches the same place by defaulting
877        // `MatchLex` off unless `Options.Match` is set. Rust defaults
878        // `match_lex` true as TS does, so the guard is the parity.
879        let custom = (self.options.match_lex
880            && !match_skipped
881            && !self.options.match_tokens.is_empty())
882        .then(|| {
883            // Two passes, position-expected before eager, as go/lexer.go
884            // matchMatch and ts/src/lexer.ts makeMatchMatcher both make.
885            // One tin-ordered pass in which eagerness merely bypassed the
886            // slot gate let an eager matcher EARLIER in tin order win over
887            // an expected one later: with `p = %x31-39` beside
888            // `d = %x30-39`, the `2` of `12` lexed as the narrower class
889            // the `*d` loop never asked for. Eagerness is for firing where
890            // the slot's list is narrower than the grammar, never for
891            // outbidding what the slot names.
892            //
893            // Under a want the alternate's own tin list is the sharper
894            // gate, so one filtered pass is the whole search. With no
895            // expected list at all (a standalone lexer, no rule) nothing
896            // constrains the caller and every matcher is eligible in the
897            // first pass.
898            // The longest FIXED literal this slot expects that matches
899            // here, or 0. Only the eager pass consults it: there, a
900            // literal the slot names beats an eager-only matcher that
901            // cuts no further than it does. Without this, a character
902            // class that CONTAINS a literal the grammar also uses
903            // swallows it wherever the class is eager (`num = "0" /
904            // posdigit *digit` beside `digit = %x30-39` rejected
905            // `0.0.0`). LENGTH decides, not mere existence, so a keyword
906            // literal cannot truncate a longer word: ties go to the
907            // literal, and an eager matcher that cuts further still
908            // wins. TS and Go do the same, in makeMatchMatcher and
909            // matchMatch.
910            //
911            // Computed once per fetch and only when a regex matcher in the
912            // eager pass has something to weigh against it, as TS
913            // `expectedFixedLen` does (`fixLen = -1` until asked). The
914            // scan is the whole fixed table against the slot's list; an
915            // expected matcher that wins in pass 0, or a fetch under a
916            // want, never needs it. Nothing the scan reads changes
917            // between the two passes, so lazy equals eager.
918            let mut fix_len: Option<usize> = None;
919            let compute_fix_len = || {
920                if self.want.is_none() && self.options.fixed.lex {
921                    expected_match_tins.map_or(0, |expected| {
922                        self.options
923                            .fixed
924                            .tokens
925                            .values()
926                            .filter(|token| {
927                                !token.source.is_empty()
928                                    && expected.contains(&token.tin)
929                                    && remaining.starts_with(&token.source)
930                            })
931                            .map(|token| token.source.len())
932                            .max()
933                            .unwrap_or(0)
934                    })
935                } else {
936                    0
937                }
938            };
939            let passes = if self.want.is_some() { 1 } else { 2 };
940            (0..passes).find_map(|pass| {
941                self.options.match_tokens.values().find_map(|matcher| {
942                    if !self.wants(matcher.tin) {
943                        return None;
944                    }
945                    if self.want.is_none() {
946                        let expected = expected_match_tins
947                            .is_none_or(|expected| expected.contains(&matcher.tin));
948                        if pass == 0 {
949                            if !expected {
950                                return None;
951                            }
952                        } else if expected || !matcher.eager {
953                            return None;
954                        }
955                    }
956                    let result = match &matcher.matcher {
957                        MatchTokenMatcher::Regex(regex) => regex
958                            .find(remaining)
959                            .filter(|found| found.start() == 0)
960                            // The eager pass yields to an expected
961                            // literal it cannot out-cut; the fixed
962                            // matcher (2e6) runs next and takes it. See
963                            // `fix_len` above.
964                            .filter(|found| {
965                                pass == 0 || {
966                                    let fix_len = *fix_len.get_or_insert_with(compute_fix_len);
967                                    fix_len == 0 || found.len() > fix_len
968                                }
969                            })
970                            .map(|found| {
971                                let source = found.as_str().to_string();
972                                (source.clone(), Value::String(source))
973                            }),
974                        MatchTokenMatcher::Callback(callback) => callback(remaining)
975                            .filter(|result| {
976                                !result.source.is_empty() && remaining.starts_with(&result.source)
977                            })
978                            .map(|result| (result.source, result.value)),
979                    };
980                    result.map(|(source, value)| (matcher.name.clone(), matcher.tin, source, value))
981                })
982            })
983        });
984        if let Some(Some((name, tin, matched, value))) = custom {
985            for _ in matched.chars() {
986                self.advance();
987            }
988            return Ok(Token::new(name, tin, value, matched, pnt));
989        }
990
991        if let Some(token) =
992            self.run_custom_matchers(&mut custom_index, 2_000_000.0, pnt, &mut plugin)
993        {
994            return Ok(token);
995        }
996
997        // Fixed literals occupy the 2e6 band and use longest-match wins.
998        let fixed_skipped = if self.options.fixed.lex {
999            match self.run_check(self.options.fixed.check.clone(), pnt) {
1000                CheckFlow::Continue => false,
1001                CheckFlow::Skip => true,
1002                CheckFlow::Token(token) => return Ok(*token),
1003            }
1004        } else {
1005            false
1006        };
1007        let remaining = &self.src[self.byte_position()..];
1008        // The winner is carried out of the table as its position, not as a
1009        // copy of its text. `Token::new` takes the name and the source text
1010        // by reference and stores both inline, so the only owned copy the
1011        // token needs is the one inside `Value::String`. Naming the match
1012        // as three owned values cost three `String` allocations per fixed
1013        // token, two of them freed again before the token was built.
1014        // The first byte decides almost every entry. Asking `wants` and then
1015        // `starts_with` of each fixed token in turn ran a tin lookup and a
1016        // `memcmp` per token in the grammar per token in the input, and a
1017        // grammar with fifty fixed tokens pays fifty of each to reject
1018        // forty-nine. One byte answers the same question, and an empty
1019        // source is kept out by its own check, which only entries that
1020        // already matched the byte ever reach.
1021        let first_byte = remaining.as_bytes().first().copied();
1022        let fixed = (self.options.fixed.lex && !fixed_skipped)
1023            .then(|| {
1024                self.options
1025                    .fixed
1026                    .tokens
1027                    .values()
1028                    .enumerate()
1029                    .filter(|(_, token)| {
1030                        token.source.as_bytes().first().copied() == first_byte
1031                            && !token.source.is_empty()
1032                            && self.wants(token.tin)
1033                            && remaining.starts_with(&token.source)
1034                    })
1035                    .max_by_key(|(_, token)| token.source.len())
1036                    .map(|(index, token)| (index, token.source.chars().count()))
1037            })
1038            .flatten();
1039        if let Some((index, source_chars)) = fixed {
1040            for _ in 0..source_chars {
1041                self.advance();
1042            }
1043            let (_, token) = self
1044                .options
1045                .fixed
1046                .tokens
1047                .get_index(index)
1048                .expect("index came from this table, which nothing writes to mid-parse");
1049            return Ok(Token::new(
1050                &token.name,
1051                token.tin,
1052                Value::String(token.source.clone()),
1053                token.source.as_str(),
1054                pnt,
1055            ));
1056        }
1057
1058        if let Some(token) =
1059            self.run_custom_matchers(&mut custom_index, 3_000_000.0, pnt, &mut plugin)
1060        {
1061            return Ok(token);
1062        }
1063
1064        // 1. Whitespace
1065        let space_skipped = if self.options.space.lex && self.wants(TIN_SP) {
1066            match self.run_check(self.options.space.check.clone(), pnt) {
1067                CheckFlow::Continue => false,
1068                CheckFlow::Skip => true,
1069                CheckFlow::Token(token) => return Ok(*token),
1070            }
1071        } else {
1072            false
1073        };
1074        if self.options.space.lex
1075            && !space_skipped
1076            && self.wants(TIN_SP)
1077            && self.char_sets.space.contains(c)
1078        {
1079            let mut src = String::new();
1080            while let Some(ch) = self.peek() {
1081                if self.char_sets.space.contains(ch) {
1082                    src.push(ch);
1083                    self.advance();
1084                } else {
1085                    break;
1086                }
1087            }
1088            return Ok(Token::new(
1089                "#SP",
1090                TIN_SP,
1091                Value::String(src.clone()),
1092                src,
1093                pnt,
1094            ));
1095        }
1096
1097        if let Some(token) =
1098            self.run_custom_matchers(&mut custom_index, 4_000_000.0, pnt, &mut plugin)
1099        {
1100            return Ok(token);
1101        }
1102
1103        // 2. Line ending
1104        let line_skipped = if self.options.line.lex && self.wants(TIN_LN) {
1105            match self.run_check(self.options.line.check.clone(), pnt) {
1106                CheckFlow::Continue => false,
1107                CheckFlow::Skip => true,
1108                CheckFlow::Token(token) => return Ok(*token),
1109            }
1110        } else {
1111            false
1112        };
1113        if self.options.line.lex
1114            && !line_skipped
1115            && self.wants(TIN_LN)
1116            && (self.char_sets.line_ends.contains(c))
1117        {
1118            let mut src = String::new();
1119            let mut seen = std::collections::HashSet::new();
1120            while let Some(ch) = self.peek() {
1121                if !self.char_sets.line_ends.contains(ch) {
1122                    break;
1123                }
1124                if self.options.line.single && !seen.insert(ch) {
1125                    break;
1126                }
1127                src.push(self.advance().expect("peeked character must advance"));
1128            }
1129            self.ci = 1;
1130            return Ok(Token::new(
1131                "#LN",
1132                TIN_LN,
1133                Value::String(src.clone()),
1134                src,
1135                pnt,
1136            ));
1137        }
1138
1139        if self.options.line.lex
1140            && !line_skipped
1141            && self.wants(TIN_LN)
1142            && (c == '\u{2028}' || c == '\u{2029}')
1143        {
1144            let bad_char = self.advance().expect("peeked character must advance");
1145            let err = TabnasError::new(
1146                "unexpected",
1147                bad_char.to_string(),
1148                "",
1149                pnt.site.pos,
1150                pnt.site.ri,
1151                pnt.site.ci,
1152            );
1153            self.err = Some(err.clone());
1154            return Err(Box::new(err));
1155        }
1156
1157        if let Some(token) =
1158            self.run_custom_matchers(&mut custom_index, 5_000_000.0, pnt, &mut plugin)
1159        {
1160            return Ok(token);
1161        }
1162
1163        // 3. Quoted strings. These precede comments in the canonical matcher
1164        // order, so an overlapping quote/comment opener is a string unless
1165        // string matching explicitly abandons the malformed candidate.
1166        let string_skipped = if self.options.string.lex && self.wants(TIN_ST) {
1167            match self.run_check(self.options.string.check.clone(), pnt) {
1168                CheckFlow::Continue => false,
1169                CheckFlow::Skip => true,
1170                CheckFlow::Token(token) => return Ok(*token),
1171            }
1172        } else {
1173            false
1174        };
1175        if self.options.string.lex
1176            && !string_skipped
1177            && self.wants(TIN_ST)
1178            && self.char_sets.string.contains(c)
1179        {
1180            let start = (self.idx, self.ri, self.ci);
1181            match self.match_string(c, pnt) {
1182                result @ Ok(_) => return result,
1183                Err(error) if !self.options.string.abandon => return Err(error),
1184                Err(_) => {
1185                    (self.idx, self.ri, self.ci) = start;
1186                    self.err = None;
1187                }
1188            }
1189        }
1190
1191        if let Some(token) =
1192            self.run_custom_matchers(&mut custom_index, 6_000_000.0, pnt, &mut plugin)
1193        {
1194            return Ok(token);
1195        }
1196
1197        // 4. Comments (longest opening marker wins; ties sort by name).
1198        let comment_skipped = if self.options.comment.lex && self.wants(TIN_CM) {
1199            match self.run_check(self.options.comment.check.clone(), pnt) {
1200                CheckFlow::Continue => false,
1201                CheckFlow::Skip => true,
1202                CheckFlow::Token(token) => return Ok(*token),
1203            }
1204        } else {
1205            false
1206        };
1207        if self.options.comment.lex && !comment_skipped && self.wants(TIN_CM) {
1208            if let Some(token) = self.match_comment(pnt)? {
1209                return Ok(token);
1210            }
1211        }
1212
1213        if let Some(token) =
1214            self.run_custom_matchers(&mut custom_index, 7_000_000.0, pnt, &mut plugin)
1215        {
1216            return Ok(token);
1217        }
1218
1219        // 5. Numbers
1220        let number_skipped = if self.options.number.lex && self.wants(TIN_NR) {
1221            match self.run_check(self.options.number.check.clone(), pnt) {
1222                CheckFlow::Continue => false,
1223                CheckFlow::Skip => true,
1224                CheckFlow::Token(token) => return Ok(*token),
1225            }
1226        } else {
1227            false
1228        };
1229        if self.options.number.lex
1230            && !number_skipped
1231            && self.wants(TIN_NR)
1232            && (c == '-' || c == '+' || c == '.' || c.is_ascii_digit())
1233        {
1234            if let Some(tkn) = self.match_number(pnt)? {
1235                return Ok(tkn);
1236            }
1237        }
1238
1239        if let Some(token) =
1240            self.run_custom_matchers(&mut custom_index, 8_000_000.0, pnt, &mut plugin)
1241        {
1242            return Ok(token);
1243        }
1244
1245        // 6. Text and named/regex values share the same delimited run.
1246        // Negotiated lexing gates this combined family by its primary token
1247        // identity (#TX), matching the TypeScript and Go dispatchers. Once
1248        // entered, an exact or regexp value definition may still produce
1249        // #VL; the caller rejects and rolls that cut back when #VL was not
1250        // requested.
1251        let text_matcher_wanted = self.wants(TIN_TX);
1252        let value_lex = self.options.value.lex && text_matcher_wanted;
1253        let text_lex = self.options.text.lex && text_matcher_wanted;
1254        let text_skipped = if text_lex || value_lex {
1255            match self.run_check(self.options.text.check.clone(), pnt) {
1256                CheckFlow::Continue => false,
1257                CheckFlow::Skip => true,
1258                CheckFlow::Token(token) => return Ok(*token),
1259            }
1260        } else {
1261            false
1262        };
1263        if (text_lex || value_lex) && !text_skipped && !self.is_text_delimiter_here() {
1264            let start = (self.idx, self.ri, self.ci);
1265            // Only a `value` definition declaring `consume` looks at the
1266            // rest of the document, and the JSON grammar has none -- but
1267            // this ran for every text token, copying the whole tail of the
1268            // input each time. `self.src` is borrowed from the caller for
1269            // `'a` and is never reassigned, so reading the reference out
1270            // before the scan below gives a slice that does not borrow
1271            // `self` and survives the `&mut self` the scan needs.
1272            let source: &'a str = self.src;
1273            let remaining = &source[self.byte_position()..];
1274            let mut src = String::new();
1275            while let Some(ch) = self.peek() {
1276                if self.is_text_delimiter_here() {
1277                    break;
1278                }
1279                src.push(ch);
1280                self.advance();
1281            }
1282
1283            let mut output = None;
1284            if value_lex {
1285                if let Some(definition) = self
1286                    .options
1287                    .value
1288                    .definitions
1289                    .get(&src)
1290                    .filter(|definition| definition.matcher.is_none())
1291                    .cloned()
1292                {
1293                    output = Some(Token::new(
1294                        "#VL",
1295                        TIN_VL,
1296                        definition
1297                            .val
1298                            .clone()
1299                            .unwrap_or_else(|| Value::String(src.clone())),
1300                        src.clone(),
1301                        pnt,
1302                    ));
1303                }
1304
1305                if output.is_none() {
1306                    let mut definitions: Vec<_> = self
1307                        .options
1308                        .value
1309                        .definitions
1310                        .iter()
1311                        .filter(|(_, definition)| definition.matcher.is_some())
1312                        .map(|(name, definition)| (name.clone(), definition.clone()))
1313                        .collect();
1314                    definitions.sort_by(|(name_a, _), (name_b, _)| name_a.cmp(name_b));
1315                    for (_, definition) in definitions {
1316                        let regex = definition.matcher.as_ref().expect("filtered matcher");
1317                        let target: &str = if definition.consume { remaining } else { &src };
1318                        let Some(captures) = regex.captures(target) else {
1319                            continue;
1320                        };
1321                        let Some(found) = captures.get(0).filter(|found| found.start() == 0) else {
1322                            continue;
1323                        };
1324                        if !definition.consume && found.end() != target.len() {
1325                            continue;
1326                        }
1327                        let matched = found.as_str().to_string();
1328                        let value = definition.transform.as_ref().map_or_else(
1329                            || {
1330                                definition
1331                                    .val
1332                                    .clone()
1333                                    .unwrap_or_else(|| Value::String(matched.clone()))
1334                            },
1335                            |transform| {
1336                                let groups = captures
1337                                    .iter()
1338                                    .map(|capture| {
1339                                        capture
1340                                            .map_or_else(String::new, |value| value.as_str().into())
1341                                    })
1342                                    .collect::<Vec<_>>();
1343                                transform(&groups)
1344                            },
1345                        );
1346                        if definition.consume {
1347                            (self.idx, self.ri, self.ci) = start;
1348                            for _ in matched.chars() {
1349                                self.advance();
1350                            }
1351                        }
1352                        output = Some(Token::new("#VL", TIN_VL, value, matched, pnt));
1353                        break;
1354                    }
1355                }
1356            }
1357
1358            if output.is_none() && (!text_lex || text_skipped) {
1359                (self.idx, self.ri, self.ci) = start;
1360            } else if output.is_none() {
1361                output = Some(Token::new(
1362                    "#TX",
1363                    TIN_TX,
1364                    Value::String(src.clone()),
1365                    src,
1366                    pnt,
1367                ));
1368            }
1369
1370            if let Some(mut token) = output {
1371                let value = std::mem::replace(&mut token.val, Value::Undefined);
1372                token.val = self.modify_text_value(value, &mut plugin);
1373                return Ok(token);
1374            }
1375        }
1376
1377        if let Some(token) =
1378            self.run_custom_matchers(&mut custom_index, f64::INFINITY, pnt, &mut plugin)
1379        {
1380            return Ok(token);
1381        }
1382
1383        // 7. Unclaimed character -> Error: unexpected
1384        let bad_char = self.advance().unwrap();
1385        let err = TabnasError::new(
1386            "unexpected",
1387            bad_char.to_string(),
1388            "",
1389            pnt.site.pos,
1390            pnt.site.ri,
1391            pnt.site.ci,
1392        );
1393        self.err = Some(err.clone());
1394        Err(Box::new(err))
1395    }
1396
1397    fn match_comment(&mut self, pnt: Point) -> LexResult<Option<Token>> {
1398        let remaining = &self.src[self.byte_position()..];
1399        let mut definitions: Vec<_> = self
1400            .options
1401            .comment
1402            .definitions
1403            .iter()
1404            .filter(|(_, definition)| {
1405                // Same first-byte test as the fixed-token scan above.
1406                definition.start.as_bytes().first().copied()
1407                    == remaining.as_bytes().first().copied()
1408                    && !definition.start.is_empty()
1409                    && definition.lex
1410                    && remaining.starts_with(&definition.start)
1411            })
1412            .collect();
1413        definitions.sort_by(|(name_a, a), (name_b, b)| {
1414            b.start
1415                .len()
1416                .cmp(&a.start.len())
1417                .then_with(|| name_a.cmp(name_b))
1418        });
1419        let Some((_, definition)) = definitions.first() else {
1420            return Ok(None);
1421        };
1422        let definition = (*definition).clone();
1423        let mut src = String::new();
1424        for _ in definition.start.chars() {
1425            src.push(self.advance().expect("comment marker must advance"));
1426        }
1427
1428        let mut terminated_by_suffix = false;
1429        let mut closed = definition.line;
1430        loop {
1431            let remainder = &self.src[self.byte_position()..];
1432            let suffix = definition
1433                .suffixes
1434                .iter()
1435                .filter(|suffix| !suffix.is_empty() && remainder.starts_with(*suffix))
1436                .max_by_key(|suffix| suffix.len())
1437                .cloned();
1438            let suffix = suffix.or_else(|| {
1439                let matcher = definition.suffix_matcher.as_ref()?;
1440                let effect = matcher.run(remainder);
1441                if effect.is_some() {
1442                    return effect;
1443                }
1444                let saved = self.state();
1445                let wanted = self.want.clone();
1446                let token = matcher.run_imperative(self);
1447                self.restore(saved);
1448                self.want = wanted;
1449                token.map(|token| token.src.to_string())
1450            });
1451            let remainder = &self.src[self.byte_position()..];
1452            let suffix =
1453                suffix.filter(|suffix| !suffix.is_empty() && remainder.starts_with(suffix));
1454            if let Some(suffix) = suffix {
1455                for _ in suffix.chars() {
1456                    src.push(self.advance().expect("comment suffix must advance"));
1457                }
1458                terminated_by_suffix = true;
1459                closed = true;
1460                break;
1461            }
1462            if !definition.line
1463                && !definition.end.is_empty()
1464                && remainder.starts_with(&definition.end)
1465            {
1466                for _ in definition.end.chars() {
1467                    src.push(self.advance().expect("comment end must advance"));
1468                }
1469                closed = true;
1470                break;
1471            }
1472            let Some(ch) = self.peek() else {
1473                break;
1474            };
1475            if definition.line && (self.char_sets.line_ends.contains(ch)) {
1476                break;
1477            }
1478            src.push(self.advance().expect("comment body must advance"));
1479        }
1480
1481        if !closed {
1482            let err = TabnasError::new(
1483                "unterminated_comment",
1484                src,
1485                "",
1486                pnt.site.pos,
1487                pnt.site.ri,
1488                pnt.site.ci,
1489            );
1490            self.err = Some(err.clone());
1491            return Err(Box::new(err));
1492        }
1493
1494        if definition.eat_line && !terminated_by_suffix {
1495            while let Some(ch) = self.peek() {
1496                if !self.char_sets.line_ends.contains(ch) {
1497                    break;
1498                }
1499                src.push(self.advance().expect("comment line tail must advance"));
1500            }
1501        }
1502
1503        Ok(Some(Token::new(
1504            "#CM",
1505            TIN_CM,
1506            Value::String(src.clone()),
1507            src,
1508            pnt,
1509        )))
1510    }
1511
1512    fn match_number(&mut self, pnt: Point) -> LexResult<Option<Token>> {
1513        let start_idx = self.idx;
1514        let mut src = String::new();
1515
1516        // Optional sign.
1517        if matches!(self.peek(), Some('-' | '+')) {
1518            src.push(self.advance().unwrap());
1519        }
1520
1521        // Base-prefixed integers are complete at the final valid digit.
1522        if self.peek() == Some('0') {
1523            if let Some(prefix) = self.peek_at(1) {
1524                let radix = match prefix {
1525                    'x' | 'X' if self.options.number.hex => Some(16),
1526                    'o' | 'O' if self.options.number.oct => Some(8),
1527                    'b' | 'B' if self.options.number.bin => Some(2),
1528                    _ => None,
1529                };
1530                if let Some(radix) = radix {
1531                    src.push(self.advance().expect("peeked zero"));
1532                    src.push(self.advance().expect("peeked base prefix"));
1533                    let mut saw_digit = false;
1534                    while let Some(ch) = self.peek() {
1535                        if ch.is_digit(radix) {
1536                            saw_digit = true;
1537                            src.push(self.advance().expect("peeked base digit"));
1538                        } else if self
1539                            .options
1540                            .number
1541                            .sep
1542                            .as_ref()
1543                            .is_some_and(|separator| separator.contains(ch))
1544                        {
1545                            src.push(self.advance().expect("peeked base digit"));
1546                        } else {
1547                            break;
1548                        }
1549                    }
1550                    if saw_digit && self.is_text_delimiter_here() {
1551                        if self
1552                            .exclude_regex
1553                            .as_ref()
1554                            .is_some_and(|regex| regex.is_match(&src))
1555                        {
1556                            self.reset_number(start_idx, pnt);
1557                            return Ok(None);
1558                        }
1559                        if self.options.value.lex {
1560                            if let Some(definition) = self
1561                                .options
1562                                .value
1563                                .definitions
1564                                .get(&src)
1565                                .filter(|definition| definition.matcher.is_none())
1566                            {
1567                                return Ok(Some(Token::new(
1568                                    "#VL",
1569                                    TIN_VL,
1570                                    definition
1571                                        .val
1572                                        .clone()
1573                                        .unwrap_or_else(|| Value::String(src.clone())),
1574                                    src,
1575                                    pnt,
1576                                )));
1577                            }
1578                        }
1579                        // The digits of the literal, prefix, sign and any
1580                        // separators removed, folded as they are read. The
1581                        // fold keeps a bounded head, a digit count and a
1582                        // sticky bit, so a literal of any length costs the
1583                        // same handful of bytes: buffering the digits
1584                        // instead would let one long token multiply the
1585                        // memory the source already holds.
1586                        let mut fold = DigitFold::new(radix.trailing_zeros());
1587                        for ch in src.chars().skip_while(|ch| matches!(ch, '-' | '+')).skip(2) {
1588                            if self
1589                                .options
1590                                .number
1591                                .sep
1592                                .as_ref()
1593                                .is_some_and(|separator| separator.contains(ch))
1594                            {
1595                                continue;
1596                            }
1597                            fold.push(ch.to_digit(radix).expect("validated base digit"));
1598                        }
1599                        let mut value = fold.finish();
1600                        if src.starts_with('-') {
1601                            value = -value;
1602                        }
1603                        return Ok(Some(Token::new(
1604                            "#NR",
1605                            TIN_NR,
1606                            Value::Number(value),
1607                            src,
1608                            pnt,
1609                        )));
1610                    }
1611                    self.reset_number(start_idx, pnt);
1612                    return Ok(None);
1613                }
1614            }
1615        }
1616
1617        let Some(ch) = self.peek() else {
1618            self.reset_number(start_idx, pnt);
1619            return Ok(None);
1620        };
1621        if ch == '.' {
1622            if !self.peek_at(1).is_some_and(|next| next.is_ascii_digit()) {
1623                self.reset_number(start_idx, pnt);
1624                return Ok(None);
1625            }
1626            src.push(self.advance().expect("peeked leading decimal point"));
1627        } else if !ch.is_ascii_digit() {
1628            self.reset_number(start_idx, pnt);
1629            return Ok(None);
1630        }
1631
1632        let (has_digits, edge_separator) = self.scan_number_digits(&mut src);
1633        if !has_digits || edge_separator {
1634            self.reset_number(start_idx, pnt);
1635            return Ok(None);
1636        }
1637
1638        // The canonical regexp admits a trailing decimal point and an
1639        // exponent after it (`2.e3`), but declines `0.a` as one text run.
1640        if self.peek() == Some('.') {
1641            let next = self.peek_at(1);
1642            let exponent_after_dot = matches!(next, Some('e' | 'E'))
1643                && match self.peek_at(2) {
1644                    Some('+' | '-') => self.peek_at(3).is_some_and(|ch| ch.is_ascii_digit()),
1645                    Some(ch) => ch.is_ascii_digit(),
1646                    None => false,
1647                };
1648            if next.is_some_and(|ch| ch.is_ascii_digit()) {
1649                src.push(self.advance().expect("peeked decimal point"));
1650                let (_, edge_separator) = self.scan_number_digits(&mut src);
1651                if edge_separator {
1652                    self.reset_number(start_idx, pnt);
1653                    return Ok(None);
1654                }
1655            } else if next.is_some()
1656                && !self.is_text_delimiter_at(self.idx + 1)
1657                && next != Some('.')
1658                && !exponent_after_dot
1659            {
1660                self.reset_number(start_idx, pnt);
1661                return Ok(None);
1662            } else {
1663                src.push(self.advance().expect("peeked trailing decimal point"));
1664            }
1665        }
1666
1667        if matches!(self.peek(), Some('e' | 'E')) {
1668            let exponent_start = self.idx;
1669            let source_len = src.len();
1670            src.push(self.advance().expect("peeked exponent marker"));
1671            if matches!(self.peek(), Some('+' | '-')) {
1672                src.push(self.advance().expect("peeked exponent sign"));
1673            }
1674            let (has_exponent_digits, edge_separator) = self.scan_number_digits(&mut src);
1675            if edge_separator {
1676                self.reset_number(start_idx, pnt);
1677                return Ok(None);
1678            }
1679            if !has_exponent_digits {
1680                self.idx = exponent_start;
1681                src.truncate(source_len);
1682            }
1683        }
1684
1685        if !self.is_text_delimiter_here() {
1686            self.reset_number(start_idx, pnt);
1687            return Ok(None);
1688        }
1689
1690        // Check exclusion regex (e.g. ^00+)
1691        if let Some(ref re) = self.exclude_regex {
1692            if re.is_match(&src) {
1693                // Number is excluded, backtrack
1694                self.reset_number(start_idx, pnt);
1695                return Ok(None);
1696            }
1697        }
1698
1699        if self.options.value.lex {
1700            if let Some(definition) = self
1701                .options
1702                .value
1703                .definitions
1704                .get(&src)
1705                .filter(|definition| definition.matcher.is_none())
1706            {
1707                return Ok(Some(Token::new(
1708                    "#VL",
1709                    TIN_VL,
1710                    definition
1711                        .val
1712                        .clone()
1713                        .unwrap_or_else(|| Value::String(src.clone())),
1714                    src,
1715                    pnt,
1716                )));
1717            }
1718        }
1719
1720        // Parse float
1721        let parse_src = self.options.number.sep.as_ref().map_or_else(
1722            || src.clone(),
1723            |separator| src.chars().filter(|ch| !separator.contains(*ch)).collect(),
1724        );
1725        match parse_src.parse::<f64>() {
1726            Ok(num) => Ok(Some(Token::new(
1727                "#NR",
1728                TIN_NR,
1729                Value::Number(num),
1730                src,
1731                pnt,
1732            ))),
1733            Err(_) => {
1734                self.reset_number(start_idx, pnt);
1735                Ok(None)
1736            }
1737        }
1738    }
1739
1740    fn reset_number(&mut self, start_idx: usize, pnt: Point) {
1741        self.idx = start_idx;
1742        self.ri = pnt.site.ri;
1743        self.ci = pnt.site.ci;
1744    }
1745
1746    /// Consume a decimal digit/separator run. Separators are legal only
1747    /// between digits; a leading or trailing separator makes the whole run
1748    /// fall through to text, matching the TypeScript regexp and Go scanner.
1749    fn scan_number_digits(&mut self, src: &mut String) -> (bool, bool) {
1750        // The run is measured before any of it is consumed. Advancing
1751        // as it goes would hold `&mut self` across a read of
1752        // `self.options.number.sep`, and the way that used to be settled
1753        // was to clone the separator — an allocation and a free for
1754        // every number in the input, for a value that cannot change
1755        // while one number is being scanned.
1756        let run_start = self.idx;
1757        let mut saw_digit = false;
1758        let mut last_was_separator = false;
1759        let mut end = run_start;
1760        {
1761            let separator = self.options.number.sep.as_deref();
1762            while let Some(ch) = self.chars.get(end).copied() {
1763                if ch.is_ascii_digit() {
1764                    saw_digit = true;
1765                    last_was_separator = false;
1766                } else if separator.is_some_and(|separator| separator.contains(ch)) {
1767                    last_was_separator = true;
1768                } else {
1769                    break;
1770                }
1771                end += 1;
1772            }
1773        }
1774        while self.idx < end {
1775            src.push(self.advance().expect("scanned number character"));
1776        }
1777        let starts_with_separator = self.idx > run_start
1778            && self.options.number.sep.as_deref().is_some_and(|separator| {
1779                self.chars[run_start..self.idx]
1780                    .first()
1781                    .is_some_and(|ch| separator.contains(*ch))
1782            });
1783        (saw_digit, starts_with_separator || last_was_separator)
1784    }
1785
1786    fn match_string(&mut self, quote: char, pnt: Point) -> LexResult<Token> {
1787        let quote_char = self.advance().unwrap();
1788        let mut out_str = String::new();
1789        let mut raw_src = String::new();
1790        raw_src.push(quote_char);
1791
1792        let mut pending_high_surrogate: Option<u16> = None;
1793
1794        while let Some(c) = self.peek() {
1795            if c == quote {
1796                raw_src.push(self.advance().unwrap());
1797                // Rust strings cannot represent a lone UTF-16 surrogate, so
1798                // preserve the Go-port behavior and fold it to U+FFFD.
1799                self.flush_surrogate(&mut pending_high_surrogate, &mut out_str);
1800                return Ok(Token::new(
1801                    "#ST",
1802                    TIN_ST,
1803                    Value::String(out_str),
1804                    raw_src,
1805                    pnt,
1806                ));
1807            }
1808
1809            if let Some(replacement) = self.options.string.replace.get(&c).cloned() {
1810                raw_src.push(self.advance().expect("peeked character must advance"));
1811                self.flush_surrogate(&mut pending_high_surrogate, &mut out_str);
1812                out_str.push_str(&replacement);
1813                continue;
1814            }
1815
1816            if self.char_sets.line.contains(c) {
1817                if self.options.string.multi_chars.contains(quote) {
1818                    raw_src.push(self.advance().expect("peeked character must advance"));
1819                    out_str.push(c);
1820                    continue;
1821                }
1822                // Sited ON the line character, as TypeScript does
1823                // (`pnt.sI = sI; pnt.cI = cI` before its `bad()` call,
1824                // ts/src/lexer.ts) and as the control-character branch
1825                // below already does. `pnt` is the opening quote, and
1826                // reporting that put every embedded newline at the start
1827                // of its string.
1828                let site = self.current_point().site;
1829                let err =
1830                    TabnasError::new("unprintable", c.to_string(), "", site.pos, site.ri, site.ci);
1831                self.err = Some(err.clone());
1832                return Err(Box::new(err));
1833            }
1834
1835            // Check for unprintable unescaped control characters in string (< 32)
1836            if (c as u32) < 32 && !self.options.string.allow_control {
1837                let err = TabnasError::new(
1838                    "unprintable",
1839                    c.to_string(),
1840                    "",
1841                    self.current_point().site.pos,
1842                    self.current_point().site.ri,
1843                    self.current_point().site.ci,
1844                );
1845                self.err = Some(err.clone());
1846                return Err(Box::new(err));
1847            }
1848
1849            if c == self.options.string.escape_char {
1850                raw_src.push(self.advance().unwrap());
1851                let esc_point = self.current_point();
1852                if let Some(esc) = self.advance() {
1853                    raw_src.push(esc);
1854                    if let Some(replacement) = self.options.string.escape.get(&esc).cloned() {
1855                        self.flush_surrogate(&mut pending_high_surrogate, &mut out_str);
1856                        out_str.push_str(&replacement);
1857                        continue;
1858                    }
1859                    match esc {
1860                        'u' => {
1861                            // Unicode escape: \uXXXX or \u{X...}. An
1862                            // invalid one is reported on the backslash
1863                            // with the span TypeScript cuts: six source
1864                            // characters for the fixed-width form, four
1865                            // for `\x`, and through the closing brace
1866                            // (or to the end of the source) for the
1867                            // braced form -- clipped, never padded, so a
1868                            // truncated escape at end of input reports
1869                            // exactly the characters that are there.
1870                            if self.peek() == Some('{') && !self.options.string.escape_strict {
1871                                raw_src.push(self.advance().unwrap()); // '{'
1872                                let mut hex = String::new();
1873                                let mut closed = false;
1874                                while let Some(h) = self.peek() {
1875                                    if h == '}' {
1876                                        raw_src.push(self.advance().unwrap());
1877                                        closed = true;
1878                                        break;
1879                                    }
1880                                    raw_src.push(self.advance().unwrap());
1881                                    hex.push(h);
1882                                }
1883
1884                                if !closed
1885                                    || hex.is_empty()
1886                                    || hex.len() > 6
1887                                    || !hex.chars().all(|ch| ch.is_ascii_hexdigit())
1888                                {
1889                                    let err = TabnasError::new(
1890                                        "invalid_unicode",
1891                                        self.source_span(esc_point.site.pos - 1, self.idx),
1892                                        "",
1893                                        esc_point.site.pos - 1,
1894                                        esc_point.site.ri,
1895                                        esc_point.site.ci - 1,
1896                                    );
1897                                    self.err = Some(err.clone());
1898                                    return Err(Box::new(err));
1899                                }
1900
1901                                let cp = match u32::from_str_radix(&hex, 16) {
1902                                    Ok(val) if val <= 0x10FFFF => val,
1903                                    _ => {
1904                                        let err = TabnasError::new(
1905                                            "invalid_unicode",
1906                                            self.source_span(esc_point.site.pos - 1, self.idx),
1907                                            "",
1908                                            esc_point.site.pos - 1,
1909                                            esc_point.site.ri,
1910                                            esc_point.site.ci - 1,
1911                                        );
1912                                        self.err = Some(err.clone());
1913                                        return Err(Box::new(err));
1914                                    }
1915                                };
1916
1917                                self.emit_unicode_escape(
1918                                    cp,
1919                                    &mut pending_high_surrogate,
1920                                    &mut out_str,
1921                                );
1922                            } else {
1923                                // Exactly 4 hex digits: \uXXXX
1924                                let mut hex = String::new();
1925                                for _ in 0..4 {
1926                                    if let Some(h) = self.peek() {
1927                                        if h.is_ascii_hexdigit() {
1928                                            raw_src.push(self.advance().unwrap());
1929                                            hex.push(h);
1930                                        } else {
1931                                            break;
1932                                        }
1933                                    } else {
1934                                        break;
1935                                    }
1936                                }
1937
1938                                if hex.len() != 4 {
1939                                    let err = TabnasError::new(
1940                                        "invalid_unicode",
1941                                        self.source_span(
1942                                            esc_point.site.pos - 1,
1943                                            esc_point.site.pos + 5,
1944                                        ),
1945                                        "",
1946                                        esc_point.site.pos - 1,
1947                                        esc_point.site.ri,
1948                                        esc_point.site.ci - 1,
1949                                    );
1950                                    self.err = Some(err.clone());
1951                                    return Err(Box::new(err));
1952                                }
1953
1954                                let cp = u16::from_str_radix(&hex, 16).map_err(|_| {
1955                                    let err = TabnasError::new(
1956                                        "invalid_unicode",
1957                                        self.source_span(
1958                                            esc_point.site.pos - 1,
1959                                            esc_point.site.pos + 5,
1960                                        ),
1961                                        "",
1962                                        esc_point.site.pos - 1,
1963                                        esc_point.site.ri,
1964                                        esc_point.site.ci - 1,
1965                                    );
1966                                    self.err = Some(err.clone());
1967                                    err
1968                                })?;
1969
1970                                self.emit_unicode_escape(
1971                                    u32::from(cp),
1972                                    &mut pending_high_surrogate,
1973                                    &mut out_str,
1974                                );
1975                            }
1976                        }
1977                        'x' if !self.options.string.escape_strict => {
1978                            let mut hex = String::new();
1979                            for _ in 0..2 {
1980                                if let Some(h) = self.peek() {
1981                                    if h.is_ascii_hexdigit() {
1982                                        raw_src.push(
1983                                            self.advance().expect("peeked character must advance"),
1984                                        );
1985                                        hex.push(h);
1986                                    }
1987                                }
1988                            }
1989                            if hex.len() != 2 {
1990                                let err = TabnasError::new(
1991                                    "invalid_ascii",
1992                                    self.source_span(
1993                                        esc_point.site.pos - 1,
1994                                        esc_point.site.pos + 3,
1995                                    ),
1996                                    "",
1997                                    esc_point.site.pos - 1,
1998                                    esc_point.site.ri,
1999                                    esc_point.site.ci - 1,
2000                                );
2001                                self.err = Some(err.clone());
2002                                return Err(Box::new(err));
2003                            }
2004                            let byte = u8::from_str_radix(&hex, 16).expect("validated ASCII hex");
2005                            self.flush_surrogate(&mut pending_high_surrogate, &mut out_str);
2006                            out_str.push(char::from(byte));
2007                        }
2008                        other => {
2009                            if !self.options.string.allow_unknown {
2010                                // Sited on the escape CHARACTER with a
2011                                // one-character span, as TypeScript
2012                                // (`pnt.sI = sI; pnt.cI = cI; lex.bad(
2013                                // S.unexpected, sI, sI + 1)`) and Go do;
2014                                // the other escape errors sit on the
2015                                // backslash and span the construct.
2016                                let err = TabnasError::new(
2017                                    "unexpected",
2018                                    other.to_string(),
2019                                    "",
2020                                    esc_point.site.pos,
2021                                    esc_point.site.ri,
2022                                    esc_point.site.ci,
2023                                );
2024                                self.err = Some(err.clone());
2025                                return Err(Box::new(err));
2026                            }
2027                            self.flush_surrogate(&mut pending_high_surrogate, &mut out_str);
2028                            out_str.push(other);
2029                        }
2030                    }
2031                } else {
2032                    let err = TabnasError::new(
2033                        "unterminated_string",
2034                        raw_src,
2035                        "",
2036                        pnt.site.pos,
2037                        pnt.site.ri,
2038                        pnt.site.ci,
2039                    );
2040                    self.err = Some(err.clone());
2041                    return Err(Box::new(err));
2042                }
2043            } else {
2044                self.flush_surrogate(&mut pending_high_surrogate, &mut out_str);
2045                raw_src.push(self.advance().unwrap());
2046                out_str.push(c);
2047            }
2048        }
2049
2050        let err = TabnasError::new(
2051            "unterminated_string",
2052            raw_src,
2053            "",
2054            pnt.site.pos,
2055            pnt.site.ri,
2056            pnt.site.ci,
2057        );
2058        self.err = Some(err.clone());
2059        Err(Box::new(err))
2060    }
2061
2062    fn flush_surrogate(&self, pending: &mut Option<u16>, out: &mut String) {
2063        if pending.take().is_some() {
2064            out.push('\u{FFFD}');
2065        }
2066    }
2067
2068    /// Emit one decoded Unicode escape while pairing UTF-16 surrogate code
2069    /// units across both `\\uXXXX` and `\\u{...}` spellings.
2070    fn emit_unicode_escape(&self, cp: u32, pending: &mut Option<u16>, out: &mut String) {
2071        if (0xD800..=0xDBFF).contains(&cp) {
2072            self.flush_surrogate(pending, out);
2073            *pending = Some(cp as u16);
2074        } else if (0xDC00..=0xDFFF).contains(&cp) {
2075            if let Some(high) = pending.take() {
2076                let scalar = 0x10000 + (((u32::from(high)) - 0xD800) << 10) + (cp - 0xDC00);
2077                out.push(char::from_u32(scalar).expect("paired surrogates form a Unicode scalar"));
2078            } else {
2079                out.push('\u{FFFD}');
2080            }
2081        } else {
2082            self.flush_surrogate(pending, out);
2083            out.push(char::from_u32(cp).expect("validated escape is a Unicode scalar"));
2084        }
2085    }
2086}
2087
2088// ---------------------------------------------------------------------------
2089// Base-prefixed integer literals.
2090//
2091// A `0x`, `0o` or `0b` literal is read as an EXACT integer and rounded to
2092// a double ONCE. The obvious fold -- `value = value * radix + digit` in
2093// `f64` -- rounds at every digit, and past the 53-bit exact integer range
2094// those roundings accumulate: `0Xa6f2f78f4f9bf44` came out as
2095// `43a4de5ef1e9f37e` where canonical TypeScript and the Go port both
2096// answer `43a4de5ef1e9f37f`, one unit in the last place low. That is
2097// silently altered data, not a formatting difference.
2098//
2099// TypeScript coerces the literal with unary `+`, whose StringNumericValue
2100// is the exact mathematical value of the digits rounded once, half to
2101// even; Go reads it through `big.Int` and `big.Float.Float64()`, which is
2102// the same rule. These reproduce it. Only the VALUE is affected: which
2103// literals are accepted, and the token they become, are settled by
2104// `match_number` before any of this runs.
2105//
2106// The decimal path needs none of it -- `str::parse::<f64>` is correctly
2107// rounded for a digit string of any length.
2108// ---------------------------------------------------------------------------
2109
2110/// `2^k` for a non-negative `k`, exactly, saturating to infinity above the
2111/// double range. A repeated multiply would round on the way up.
2112fn pow2(k: i64) -> f64 {
2113    debug_assert!(k >= 0, "only non-negative exponents arise here");
2114    if k > 1023 {
2115        f64::INFINITY
2116    } else {
2117        f64::from_bits(((k + 1023) as u64) << 52)
2118    }
2119}
2120
2121/// Folds the digits of a base-prefixed literal into the NEAREST double,
2122/// rounding half to even, without holding the digits.
2123///
2124/// `bits` is the width of one digit, so the base is a power of two: 1 for
2125/// binary, 3 for octal, 4 for hexadecimal. Those are the only bases
2126/// `match_number` reads, which is what lets a `u128` head plus a sticky
2127/// bit stand in for arbitrary-precision arithmetic.
2128///
2129/// Only three things about a literal can change the answer: the top
2130/// `128 / bits` significant digits, how many digits follow them, and
2131/// whether any of those is non-zero. This keeps exactly those, so the
2132/// space it costs does not grow with the literal, however long an
2133/// untrusted document makes one.
2134struct DigitFold {
2135    bits: u32,
2136    /// The significant digits packed so far, at most `head_len` of them.
2137    head: u128,
2138    /// How many significant digits have been pushed, head and tail alike.
2139    len: usize,
2140    /// Whether any digit past the head was non-zero.
2141    sticky: bool,
2142    /// Whether a non-zero digit has been seen. Leading zeros carry no
2143    /// value, and dropping them is what makes the head wider than the 54
2144    /// significant bits the rounding needs.
2145    started: bool,
2146}
2147
2148impl DigitFold {
2149    fn new(bits: u32) -> Self {
2150        debug_assert!(
2151            (1..=4).contains(&bits),
2152            "only the power-of-two bases the lexer reads"
2153        );
2154        DigitFold {
2155            bits,
2156            head: 0,
2157            len: 0,
2158            sticky: false,
2159            started: false,
2160        }
2161    }
2162
2163    /// A `u128` holds exactly this many digits of the base.
2164    fn head_len(&self) -> usize {
2165        (128 / self.bits) as usize
2166    }
2167
2168    fn push(&mut self, digit: u32) {
2169        if !self.started {
2170            if 0 == digit {
2171                return;
2172            }
2173            self.started = true;
2174        }
2175        if self.len < self.head_len() {
2176            self.head = (self.head << self.bits) | u128::from(digit);
2177        } else if 0 != digit {
2178            self.sticky = true;
2179        }
2180        self.len += 1;
2181    }
2182
2183    fn finish(&self) -> f64 {
2184        if !self.started {
2185            return 0.0;
2186        }
2187        let head_len = self.head_len();
2188        if self.len <= head_len {
2189            // A `u128` to `f64` cast rounds to nearest, ties to even, which
2190            // is the rule the canonical runtime follows.
2191            return self.head as f64;
2192        }
2193
2194        // Longer than a u128: the head holds the top `head_len` digits and
2195        // `sticky` remembers whether anything below them was set. Those two
2196        // are all the rounding can depend on. The leading digit is
2197        // non-zero, so the head is at least 121 bits wide in every base
2198        // here and `shift` is comfortably positive.
2199        let dropped = i64::from(self.bits) * (self.len - head_len) as i64;
2200        let shift = 128 - self.head.leading_zeros() - 53;
2201
2202        let mut mantissa = (self.head >> shift) as u64;
2203        let half = (self.head >> (shift - 1)) & 1 == 1;
2204        let sticky = self.head & ((1u128 << (shift - 1)) - 1) != 0 || self.sticky;
2205        if half && (sticky || mantissa & 1 == 1) {
2206            // At most 2^53, which is still an exact double.
2207            mantissa += 1;
2208        }
2209        // The mantissa carries at most 53 significant bits, so the scaling
2210        // is exact inside the double range and overflows to infinity
2211        // outside it.
2212        mantissa as f64 * pow2(dropped + i64::from(shift))
2213    }
2214}