tabnas/lexer.rs
1// Copyright (c) 2013-2026 Richard Rodger, MIT License
2
3use crate::error::TabnasError;
4use crate::options::{LexCheck, LexCheckResult, MatchTokenMatcher, Options};
5use crate::token::{
6 Point, Token, TIN_BD, TIN_CM, TIN_LN, TIN_NR, TIN_SP, TIN_ST, TIN_TX, TIN_VL, TIN_ZZ,
7};
8use crate::value::Value;
9use regex::Regex;
10use std::panic::{catch_unwind, AssertUnwindSafe};
11use std::sync::Arc;
12
13pub struct Lexer<'a> {
14 src: &'a str,
15 chars: Vec<char>,
16 byte_indices: Vec<usize>,
17 char_len: usize,
18 idx: usize,
19 ri: usize,
20 ci: usize,
21 options: Arc<Options>,
22 ignore_tins: Vec<crate::Tin>,
23 char_sets: crate::text::CharSets,
24 err: Option<TabnasError>,
25 end_reached: bool,
26 /// `number.exclude`, compiled once per grammar and shared by the
27 /// parser that owns it; see [`compile_number_exclude`].
28 exclude_regex: Option<Arc<Regex>>,
29 want: Option<Vec<crate::Tin>>,
30 standalone: Option<(crate::Rule, crate::Context)>,
31}
32
33#[derive(Clone)]
34pub(crate) struct LexerState {
35 idx: usize,
36 ri: usize,
37 ci: usize,
38 err: Option<TabnasError>,
39 end_reached: bool,
40}
41
42/// Opaque snapshot returned by [`Lexer::relex_for_rule`]. Pass it to
43/// [`Lexer::unrelex`] if the caller later rejects the committed recut.
44#[derive(Clone)]
45pub struct RelexCheckpoint {
46 state: LexerState,
47 replay: std::collections::VecDeque<Token>,
48}
49
50enum CheckFlow {
51 Continue,
52 Skip,
53 Token(Box<Token>),
54}
55
56/// The result the lexer passes through its own internals.
57///
58/// `TabnasError` is 664 bytes, so `LexResult<Token>` is 664 bytes
59/// too: three times the token it carries, and moved on the SUCCESS path
60/// through every frame of the lexer chain. Boxing the error inside the
61/// engine makes those results the size of a token again, and costs an
62/// allocation only when there is an error to report, which is the path that
63/// is already building a 664-byte diagnostic.
64///
65/// The public entry points still hand back an unboxed `TabnasError`, so no
66/// caller of this crate sees the box.
67type LexResult<T> = Result<T, Box<TabnasError>>;
68
69/// The compiled form of `number.exclude`, or `None` when there is no
70/// pattern or it does not compile (an invalid pattern excludes nothing,
71/// as it always has).
72///
73/// Compiling a regex costs about 700K instructions, which is more than
74/// a small document costs to parse, so the parser compiles it once per
75/// grammar and hands every lexer the same automaton through the `Arc`.
76/// Cloning a `Regex` would not do: it shares the automaton but builds a
77/// fresh scratch-cache pool, and the first search from each clone fills
78/// it. Go and TypeScript compile the pattern once, at configuration time.
79pub(crate) fn compile_number_exclude(options: &Options) -> Option<Arc<Regex>> {
80 let pattern = options.number.exclude.as_deref()?;
81 Regex::new(pattern).ok().map(Arc::new)
82}
83
84impl<'a> Lexer<'a> {
85 pub fn new(src: &'a str, mut options: Options) -> Self {
86 if let Err(error) = options.validate_comment_definitions() {
87 panic!("invalid options: {error}");
88 }
89 // A lexer built directly may be handed options nobody has
90 // ordered yet. The parser's own lexer comes through
91 // `with_shared`, whose options were ordered when they were
92 // prepared, and whose exclude pattern was compiled then too.
93 options.sort_for_lexing();
94 let exclude_regex = compile_number_exclude(&options);
95 Self::with_shared(src, Arc::new(options), exclude_regex)
96 }
97
98 /// Lex against options the parser already owns and has ordered, with
99 /// the `number.exclude` pattern it compiled from them.
100 pub(crate) fn with_shared(
101 src: &'a str,
102 options: Arc<Options>,
103 exclude_regex: Option<Arc<Regex>>,
104 ) -> Self {
105 let mut chars = Vec::new();
106 let mut byte_indices = Vec::new();
107 for (b_idx, c) in src.char_indices() {
108 chars.push(c);
109 byte_indices.push(b_idx);
110 }
111 let char_len = chars.len();
112
113 Lexer {
114 src,
115 chars,
116 byte_indices,
117 char_len,
118 idx: 0,
119 ri: 1,
120 ci: 1,
121 ignore_tins: options.ignore_tins(),
122 char_sets: options.char_sets(),
123 options,
124 err: None,
125 end_reached: false,
126 exclude_regex,
127 want: None,
128 standalone: None,
129 }
130 }
131
132 fn current_point(&self) -> Point {
133 Point {
134 len: self.src.len(),
135 site: crate::Site {
136 si: self.byte_position(),
137 pos: self.idx,
138 ri: self.ri,
139 ci: self.ci,
140 },
141 }
142 }
143
144 /// The source from char index `start` up to `end`, clipped to the end
145 /// of the source: the span of a bad token, cut as TypeScript's
146 /// `lex.bad(why, pstart, pend)` cuts it (`ts/src/lexer.ts`).
147 ///
148 /// Only the END is clipped. The caller must supply `start <= end`
149 /// and `start <= self.char_len`, or the indexing panics. Every call
150 /// site below passes `esc_point.site.pos - 1`, which additionally
151 /// needs `1 <= pos`; that holds because `esc_point` is captured
152 /// after the escape character has been consumed. These three are
153 /// the preconditions `rs/verus/lexer_span.rs` ASSUMES: it proves the
154 /// span arithmetic is in bounds given them, and does not model the
155 /// call sites, so nothing machine-checks that they hold here. A
156 /// refactor that captured the point before the advance would still
157 /// pass `rs/verus/run.sh`.
158 fn source_span(&self, start: usize, end: usize) -> String {
159 self.chars[start..end.min(self.char_len)].iter().collect()
160 }
161
162 fn state(&self) -> LexerState {
163 LexerState {
164 idx: self.idx,
165 ri: self.ri,
166 ci: self.ci,
167 err: self.err.clone(),
168 end_reached: self.end_reached,
169 }
170 }
171
172 /// Full immutable source supplied to this lexer.
173 pub fn source(&self) -> &str {
174 self.src
175 }
176
177 /// Source remaining at the live cursor.
178 pub fn remaining(&self) -> &str {
179 &self.src[self.byte_position()..]
180 }
181
182 /// Return at most `max_chars` Unicode scalar values from the live cursor.
183 /// This is the Rust counterpart of the public `lex.fwd`/`Lex.Fwd` helper.
184 pub fn forward(&self, max_chars: usize) -> &str {
185 let remaining = self.remaining();
186 let end = remaining
187 .char_indices()
188 .nth(max_chars)
189 .map_or(remaining.len(), |(index, _)| index);
190 &remaining[..end]
191 }
192
193 /// Snapshot the live cursor for token construction.
194 pub fn point(&self) -> Point {
195 self.current_point()
196 }
197
198 /// Advance by Unicode scalar values. Returns false without moving when
199 /// the requested count extends beyond end-of-source.
200 pub fn advance_chars(&mut self, count: usize) -> bool {
201 if self.idx.saturating_add(count) > self.char_len {
202 return false;
203 }
204 for _ in 0..count {
205 self.advance();
206 }
207 true
208 }
209
210 /// Construct a token from a point captured before cursor advancement.
211 pub fn token(
212 &self,
213 name: impl AsRef<str>,
214 tin: crate::Tin,
215 value: Value,
216 source: impl Into<crate::TokenText>,
217 point: Point,
218 ) -> Token {
219 Token::new(name, tin, value, source, point)
220 }
221
222 /// Resolve or allocate a token identity in this lexer's configuration.
223 pub fn token_tin(&mut self, name: impl Into<String>) -> crate::Tin {
224 Arc::make_mut(&mut self.options).register_token(name)
225 }
226
227 /// Resolve a token identity back to its configured name.
228 pub fn token_name(&self, tin: crate::Tin) -> String {
229 self.options.token_name(tin)
230 }
231
232 /// Whether the options this lexer runs under enable `alt`: not when
233 /// `rule.exclude` names one of its groups, nor when `rule.include`
234 /// lists groups and it declares none of them. The parser skips such an
235 /// alternate, and TypeScript removes it from the rule spec
236 /// (`filterRules`) before anything reads the spec, so a custom matcher
237 /// that reads a rule's alternates, to tell a key position from a value
238 /// one, asks this to see the same alternates TypeScript does.
239 pub fn alt_enabled(&self, alt: &crate::AltSpec) -> bool {
240 crate::parser::groups_enabled(alt, &self.options)
241 }
242
243 /// Construct a bad token at the current cursor.
244 ///
245 /// A custom matcher that returns it reports a lexer fault, which the
246 /// parser handles as it handles its own lexer's: a rule's fetch raises
247 /// it at once with `why` as the code, at this position, or, under
248 /// recovery, records it, coalescing a run of them into one error, and
249 /// moves the cursor past its source; relexing leaves it for an
250 /// alternate to re-cut. The cursor need not be advanced here.
251 pub fn bad(&self, why: impl Into<String>) -> Token {
252 let point = self.current_point();
253 let source = self
254 .peek()
255 .map_or_else(String::new, |character| character.to_string());
256 let mut token = Token::new("#BD", TIN_BD, Value::Undefined, source, point);
257 token.err = crate::TokenCode::from(why.into());
258 token.why = token.err.clone();
259 token
260 }
261
262 /// Construct a bad token whose displayed source is a scalar-indexed span.
263 /// As in TypeScript, the diagnostic point remains the live cursor.
264 /// Returned from a matcher it is handled as [`Lexer::bad`] describes;
265 /// recovery steps past the span, counted from that cursor.
266 pub fn bad_span(&self, why: impl Into<String>, start: usize, end: usize) -> Token {
267 let point = self.current_point();
268 let source = if start <= end && end <= self.char_len {
269 let start_byte = self
270 .byte_indices
271 .get(start)
272 .copied()
273 .unwrap_or(self.src.len());
274 let end_byte = self
275 .byte_indices
276 .get(end)
277 .copied()
278 .unwrap_or(self.src.len());
279 self.src[start_byte..end_byte].to_string()
280 } else {
281 self.peek()
282 .map_or_else(String::new, |character| character.to_string())
283 };
284 let mut token = Token::new("#BD", TIN_BD, Value::Undefined, source, point);
285 token.err = crate::TokenCode::from(why.into());
286 token.why = token.err.clone();
287 token
288 }
289
290 fn byte_position(&self) -> usize {
291 self.byte_indices
292 .get(self.idx)
293 .copied()
294 .unwrap_or(self.src.len())
295 }
296
297 fn advance(&mut self) -> Option<char> {
298 if self.idx < self.char_len {
299 let c = self.chars[self.idx];
300 self.idx += 1;
301 if self.char_sets.row.contains(c) {
302 self.ri += 1;
303 self.ci = 1;
304 } else {
305 self.ci += 1;
306 }
307 Some(c)
308 } else {
309 None
310 }
311 }
312
313 /// Step over one character of a string body, counting a column and
314 /// never a row. `advance` counts a row for any character in
315 /// `line.rowChars`, which is right between tokens and wrong inside a
316 /// string: TypeScript's `buildStringBodySpec` (ts/src/lexer.ts) makes
317 /// a row character plain body there unless the string is multi-line
318 /// and the character is in `line.chars` as well, which is
319 /// [`Lexer::advance_string_line`].
320 fn advance_body(&mut self) -> Option<char> {
321 let c = self.peek()?;
322 self.idx += 1;
323 self.ci += 1;
324 Some(c)
325 }
326
327 /// Step over a line character inside a multi-line string: the column
328 /// resets, and a character in `line.rowChars` counts a row, as
329 /// TypeScript's string-body classes LINE and LINE+ROW do
330 /// (`STRING_BODY_TABLE` in ts/src/lexer.ts) and as Go's
331 /// `BuildStringBodySpec` ports them.
332 fn advance_string_line(&mut self) -> Option<char> {
333 let c = self.peek()?;
334 self.idx += 1;
335 if self.char_sets.row.contains(c) {
336 self.ri += 1;
337 }
338 self.ci = 1;
339 Some(c)
340 }
341
342 fn peek(&self) -> Option<char> {
343 if self.idx < self.char_len {
344 Some(self.chars[self.idx])
345 } else {
346 None
347 }
348 }
349
350 fn peek_at(&self, offset: usize) -> Option<char> {
351 let i = self.idx + offset;
352 if i < self.char_len {
353 Some(self.chars[i])
354 } else {
355 None
356 }
357 }
358
359 fn wants(&self, tin: crate::Tin) -> bool {
360 self.want
361 .as_ref()
362 .is_none_or(|wanted| wanted.contains(&tin))
363 }
364
365 /// Bring a fetch's point and current character up to the live cursor.
366 ///
367 /// A custom matcher, or an imperative check, may move the cursor and
368 /// then decline: xml's steps over a byte-order mark and goes on to
369 /// look for a tag. Every matcher after it starts where it left the
370 /// cursor, as TypeScript's do (each reads `lex.pnt`) and Go's do (each
371 /// reads `l.pnt`), and the current character is `None` once the cursor
372 /// is at the end of the source. Reading the character the fetch began
373 /// on instead put the built-in matchers' tokens at the wrong place,
374 /// sent them looking for a character that was no longer there, and,
375 /// when the matcher had stepped over the LAST character, panicked on
376 /// the missing one.
377 #[inline]
378 fn resync(&self, point: &mut Point, current: &mut Option<char>) {
379 if self.idx != point.site.pos {
380 *point = self.current_point();
381 *current = self.peek();
382 }
383 }
384
385 /// Whether TypeScript would try a built-in matcher at all in a fetch
386 /// that began on `first` and whose cursor has since been moved by a
387 /// matcher that declined.
388 ///
389 /// TypeScript chooses the matchers a fetch tries from a table indexed
390 /// by the character the fetch begins on (`buildLexDispatch` in
391 /// ts/src/utility.ts, read in `Lex.next`): a built-in is listed for a
392 /// Latin-1 character only when a token it makes could start with it,
393 /// unless it carries a check, and every built-in is listed for a
394 /// character from U+0100 up. The table is not consulted again when a
395 /// custom matcher moves the cursor, so this asks it about `first`, and
396 /// the matcher then tests the character under the cursor as usual.
397 /// Before the cursor moves the two are the same character and the
398 /// matcher's own test is the whole answer, so this is `true`.
399 #[inline]
400 fn listed(
401 &self,
402 entry: usize,
403 first: char,
404 has_check: bool,
405 could_start: impl FnOnce(char) -> bool,
406 ) -> bool {
407 self.idx == entry || u32::from(first) >= 256 || has_check || could_start(first)
408 }
409
410 fn run_check(&mut self, check: Option<LexCheck>, point: Point) -> CheckFlow {
411 let Some(check) = check else {
412 return CheckFlow::Continue;
413 };
414 let remaining = &self.src[self.byte_position()..];
415 let result = check
416 .run_imperative(self)
417 .or_else(|| check.run(remaining))
418 .unwrap_or(LexCheckResult::Continue);
419 match result {
420 LexCheckResult::Continue => CheckFlow::Continue,
421 LexCheckResult::Skip => CheckFlow::Skip,
422 LexCheckResult::NativeToken(token) => CheckFlow::Token(token),
423 LexCheckResult::Token(token)
424 if !token.source.is_empty() && remaining.starts_with(&token.source) =>
425 {
426 let tin = if token.tin < 0 {
427 self.options.token(&token.name).unwrap_or(token.tin)
428 } else {
429 token.tin
430 };
431 if tin < 0 {
432 return CheckFlow::Skip;
433 }
434 for _ in token.source.chars() {
435 self.advance();
436 }
437 CheckFlow::Token(Box::new(Token::new(
438 token.name,
439 tin,
440 token.value,
441 token.source,
442 point,
443 )))
444 }
445 LexCheckResult::Token(_) => CheckFlow::Skip,
446 }
447 }
448
449 /// Give any plugin matcher whose order is below `before` its turn.
450 ///
451 /// Nine sites in the lexer call this per token, once at each stage a
452 /// matcher is allowed to intervene. A grammar with no custom matcher,
453 /// which is most of them, was paying nine index lookups per token to be
454 /// told nine times that there is nothing to run. The guard is inline so
455 /// those sites skip the call itself; the walk stays out of line.
456 #[inline]
457 fn run_custom_matchers(
458 &mut self,
459 index: &mut usize,
460 before: f64,
461 plugin: &mut Option<(&mut crate::Rule, &mut crate::Context)>,
462 ) -> Option<Token> {
463 if *index >= self.options.lex.matchers.len() {
464 return None;
465 }
466 self.run_remaining_custom_matchers(index, before, plugin)
467 }
468
469 #[inline(never)]
470 fn run_remaining_custom_matchers(
471 &mut self,
472 index: &mut usize,
473 before: f64,
474 plugin: &mut Option<(&mut crate::Rule, &mut crate::Context)>,
475 ) -> Option<Token> {
476 while let Some(matcher) = self
477 .options
478 .lex
479 .matchers
480 .get_index(*index)
481 .map(|(_, matcher)| matcher)
482 .filter(|matcher| matcher.order < before)
483 .cloned()
484 {
485 *index += 1;
486 // Each matcher starts where the one before it left the cursor:
487 // one may step over a character and decline, as xml's does over
488 // a byte-order mark, and TypeScript's matchers all read the
489 // live `lex.pnt`.
490 let point = self.current_point();
491 let remaining = &self.src[self.byte_position()..];
492 let saved = self.state();
493 let token = if let Some(callback) = matcher.imperative.as_ref() {
494 let Some((rule, context)) = plugin.as_mut() else {
495 continue;
496 };
497 callback(self, rule, context)
498 } else {
499 matcher
500 .matcher
501 .as_ref()
502 .and_then(|callback| callback(remaining))
503 .filter(|token| {
504 !token.source.is_empty() && remaining.starts_with(&token.source)
505 })
506 .map(|token| {
507 Token::new(token.name, token.tin, token.value, token.source, point)
508 })
509 };
510 let Some(mut token) = token else {
511 if self.want.is_some() {
512 self.rewind(saved);
513 }
514 continue;
515 };
516
517 // TypeScript and Go run opaque custom matchers speculatively for
518 // a negotiated cut. An unwanted result rolls back locally so a
519 // later matcher can still satisfy the request.
520 let tin = if token.tin < 0 {
521 self.options.token(&token.name).unwrap_or(token.tin)
522 } else {
523 token.tin
524 };
525 if tin < 0 || !self.wants(tin) {
526 self.rewind(saved);
527 continue;
528 }
529 if matcher.imperative.is_none() {
530 for _ in token.src.chars() {
531 self.advance();
532 }
533 }
534 token.tin = tin;
535 return Some(token);
536 }
537 None
538 }
539
540 fn is_text_delimiter_here(&self) -> bool {
541 self.is_text_delimiter_at(self.idx)
542 }
543
544 fn is_text_delimiter_at(&self, index: usize) -> bool {
545 let Some(ch) = self.chars.get(index).copied() else {
546 return true;
547 };
548 let remaining = &self.src[self.byte_indices[index]..];
549 (self.options.space.lex && self.char_sets.space.contains(ch))
550 || (self.options.fixed.lex
551 && self
552 .options
553 .fixed
554 .tokens
555 .values()
556 .any(|token| !token.source.is_empty() && remaining.starts_with(&token.source)))
557 || (self.options.line.lex
558 && (self.char_sets.line_ends.contains(ch) || matches!(ch, '\u{2028}' | '\u{2029}')))
559 || (self.options.comment.lex
560 && self.options.comment.definitions.values().any(|definition| {
561 definition.lex
562 && !definition.start.is_empty()
563 && remaining.starts_with(&definition.start)
564 }))
565 || self
566 .options
567 .ender
568 .iter()
569 .any(|ender| !ender.is_empty() && remaining.starts_with(ender))
570 }
571
572 /// Fetches the next non-IGNORE token (skipping spaces, lines, comments).
573 pub fn next_token(&mut self) -> Result<Token, TabnasError> {
574 let point = self.current_point();
575 let result = match catch_unwind(AssertUnwindSafe(|| {
576 if let Some(ref error) = self.err {
577 return Err(Box::new(error.clone()));
578 }
579
580 loop {
581 let token = self.next_raw(None)?;
582 if !self.ignore_tins.contains(&token.tin) {
583 return Ok(token);
584 }
585 }
586 })) {
587 Ok(result) => result,
588 Err(payload) => self.record_panic(payload, "Lexer::next_token", point),
589 };
590 // The box is internal to the engine; a caller gets the error itself.
591 result.map_err(|error| *error)
592 }
593
594 /// Fetch the next token without discarding whitespace, line, or comment tokens.
595 pub fn next_raw_token(&mut self) -> Result<Token, TabnasError> {
596 let point = self.current_point();
597 let result = match catch_unwind(AssertUnwindSafe(|| self.next_raw(None))) {
598 Ok(result) => result,
599 Err(payload) => self.record_panic(payload, "Lexer::next_raw_token", point),
600 };
601 result.map_err(|error| *error)
602 }
603
604 /// Fetch one token for an imperative parser callback, preserving ignored
605 /// space/line/comment tokens just like TypeScript's public `lex.next`.
606 /// Replayed tokens produced by `Context::rewind` are served first.
607 pub fn next_raw_for_rule(
608 &mut self,
609 rule: &mut crate::Rule,
610 context: &mut crate::Context,
611 ) -> LexResult<Token> {
612 if let Some(token) = context.next_replay() {
613 Ok(token)
614 } else {
615 self.next_raw_with(None, Some((rule, context)))
616 }
617 }
618
619 /// Fetch the next non-ignored token for an imperative parser callback.
620 pub fn next_for_rule(
621 &mut self,
622 rule: &mut crate::Rule,
623 context: &mut crate::Context,
624 ) -> LexResult<Token> {
625 loop {
626 let token = self.next_raw_for_rule(rule, context)?;
627 if !self.ignore_tins.contains(&token.tin) {
628 return Ok(token);
629 }
630 }
631 }
632
633 /// Public negotiated-relex entry point for native parser callbacks.
634 /// A successful recut commits the lexer cursor and returns an opaque undo
635 /// checkpoint; a failed recut restores all lexer state before returning.
636 pub fn relex_for_rule(
637 &mut self,
638 from: &Token,
639 wanted: &[crate::Tin],
640 rule: &mut crate::Rule,
641 context: &mut crate::Context,
642 ) -> Option<(Token, RelexCheckpoint)> {
643 self.relex(from, wanted, rule, context)
644 }
645
646 /// Undo a committed [`Lexer::relex_for_rule`] operation, including the
647 /// pending tokens hidden while the replacement cut was negotiated.
648 pub fn unrelex(&mut self, checkpoint: RelexCheckpoint, context: &mut crate::Context) {
649 self.restore(checkpoint.state);
650 context.restore_replay(checkpoint.replay);
651 }
652
653 fn record_panic(
654 &mut self,
655 payload: Box<dyn std::any::Any + Send>,
656 api: &str,
657 point: Point,
658 ) -> LexResult<Token> {
659 let error = TabnasError::from_panic(
660 payload,
661 api,
662 self.src,
663 point.site.pos,
664 point.site.ri,
665 point.site.ci,
666 &self.options,
667 );
668 self.err = Some(error.clone());
669 Err(Box::new(error))
670 }
671
672 /// Fetch a raw token while restricting non-eager custom token matchers to
673 /// the exact tins accepted at the parser slot being filled. Builtin and
674 /// fixed-token matchers are unaffected by this gate.
675 pub(crate) fn next_rule_token(
676 &mut self,
677 expected_match_tins: &[crate::Tin],
678 rule: &mut crate::Rule,
679 context: &mut crate::Context,
680 ) -> LexResult<Token> {
681 self.next_raw_with(Some(expected_match_tins), Some((rule, context)))
682 }
683
684 /// Step past a bad token the parser has absorbed or skipped, as
685 /// TypeScript's `advanceLexPast` does (ts/src/rules.ts): a bad token
686 /// does not advance the cursor by itself, so recovery moves it to the
687 /// end of the token's span, never backwards, and, for a fault raised
688 /// inside a compound construct, on past the next row character so
689 /// lexing resumes on a fresh row. The lexer's own faults latch until
690 /// this clears them.
691 pub(crate) fn skip_bad(&mut self, token: &Token, to_line_end: bool) {
692 let span = token.src.chars().count().max(1);
693 let mut target = self.idx.max(token.site.pos.saturating_add(span));
694 if to_line_end {
695 let mut end = target;
696 while end < self.char_len && !self.char_sets.row.contains(self.chars[end]) {
697 end += 1;
698 }
699 target = target.max(self.char_len.min(end + 1));
700 }
701 while self.idx < target && self.idx < self.char_len {
702 self.advance();
703 }
704 self.err = None;
705 if self.idx < self.char_len {
706 self.end_reached = false;
707 }
708 }
709
710 /// Re-cut an already buffered source span, constrained to the token
711 /// identities requested by one alternate. On success the cursor remains
712 /// after the new cut; the returned state can restore the original cut if
713 /// that alternate later fails.
714 pub(crate) fn relex(
715 &mut self,
716 from: &Token,
717 wanted: &[crate::Tin],
718 rule: &mut crate::Rule,
719 context: &mut crate::Context,
720 ) -> Option<(Token, RelexCheckpoint)> {
721 if from.src.is_empty() || from.site.pos > self.char_len || wanted.is_empty() {
722 return None;
723 }
724 // The standing error, if any, is moved into the checkpoint rather
725 // than copied: the cut below clears it anyway, and a copy is the
726 // length of the source (`full_source`) for every cut attempted.
727 let err = self.err.take();
728 let saved = LexerState {
729 err,
730 ..self.state()
731 };
732 // TypeScript temporarily replaces the lexer's pending-token queue
733 // with an empty queue for a negotiated cut. Rust keeps that queue on
734 // Context, so hide it explicitly and preserve it in the checkpoint.
735 let replay = context.take_replay();
736 self.idx = from.site.pos;
737 self.ri = from.site.ri;
738 self.ci = from.site.ci;
739 self.err = None;
740 self.end_reached = false;
741 self.want = Some(wanted.to_vec());
742 // Straight to the matchers, past `next_raw_with`: an error here only
743 // rejects the cut, and the restore below puts back the lexer's own,
744 // so the source it would attach, and the copy it would keep, are
745 // never seen. With them, every rejected cut cost the length of the
746 // source, and a flat stylesheet parsed in quadratic time.
747 let recut = self.next_raw_inner(None, Some((rule, context))).ok();
748 self.want = None;
749 match recut.filter(|token| wanted.contains(&token.tin)) {
750 Some(mut token) => {
751 token.ignored = from.ignored.clone();
752 Some((
753 token,
754 RelexCheckpoint {
755 state: saved,
756 replay,
757 },
758 ))
759 }
760 None => {
761 self.restore(saved);
762 // Discard any speculative replay generated by an imperative
763 // matcher and restore the queue that preceded the attempt.
764 context.restore_replay(replay);
765 None
766 }
767 }
768 }
769
770 pub(crate) fn restore(&mut self, state: LexerState) {
771 self.idx = state.idx;
772 self.ri = state.ri;
773 self.ci = state.ci;
774 self.err = state.err;
775 self.end_reached = state.end_reached;
776 self.want = None;
777 }
778
779 /// Put the cursor back after a custom matcher's speculative attempt and
780 /// keep the request being negotiated, as TypeScript's `Lex.speculate`
781 /// restores the point and leaves `want` alone. [`Lexer::restore`] ends
782 /// a negotiation, so the first custom matcher to decline during a
783 /// re-cut used to lift the request for every matcher after it, and a
784 /// string matcher could then cut the `#ST` the re-cut was there to
785 /// avoid.
786 fn rewind(&mut self, state: LexerState) {
787 let want = self.want.take();
788 self.restore(state);
789 self.want = want;
790 }
791
792 fn next_raw(&mut self, expected_match_tins: Option<&[crate::Tin]>) -> LexResult<Token> {
793 // Only a lexer being driven directly needs these, and building
794 // them costs a whole `Options` clone. A parse reaches the lexer
795 // through `next_rule_token`, which brings the real rule and
796 // context with it, so it never wants them at all.
797 let (mut rule, mut context) = match self.standalone.take() {
798 Some(pair) => pair,
799 None => (
800 crate::Rule::new("#NORULE", Value::Undefined),
801 crate::Context::new(
802 self.options.rewind.history,
803 self.src,
804 Value::Undefined,
805 Arc::clone(&self.options),
806 crate::InstanceInfo::default(),
807 ),
808 ),
809 };
810 let result = self.next_raw_with(expected_match_tins, Some((&mut rule, &mut context)));
811 self.standalone = Some((rule, context));
812 result
813 }
814
815 fn modify_text_value(
816 &mut self,
817 mut value: Value,
818 plugin: &mut Option<(&mut crate::Rule, &mut crate::Context)>,
819 ) -> Value {
820 if self.options.text.modify.is_empty() {
821 return value;
822 }
823 let modifiers = self.options.text.modify.clone();
824 let options = self.options.clone();
825 let Some((rule, context)) = plugin.as_mut() else {
826 panic!("imperative text modifier requires an active lexer context");
827 };
828 for modifier in modifiers {
829 value = modifier.run(value, self, rule, context, &options);
830 }
831 value
832 }
833
834 fn next_raw_with(
835 &mut self,
836 expected_match_tins: Option<&[crate::Tin]>,
837 plugin: Option<(&mut crate::Rule, &mut crate::Context)>,
838 ) -> LexResult<Token> {
839 let result = self.next_raw_inner(expected_match_tins, plugin);
840 match result {
841 Ok(token) => Ok(token),
842 Err(mut error) => {
843 // The matchers build their errors without the source, and it
844 // is attached here, where an error leaves the lexer: a
845 // negotiated cut (`relex`) rejects many candidates, each an
846 // error nobody sees, and a copy of the whole source apiece
847 // made that quadratic.
848 error.full_source = self.src.to_string();
849 error.apply_options(&self.options);
850 self.err = Some((*error).clone());
851 Err(error)
852 }
853 }
854 }
855
856 fn next_raw_inner(
857 &mut self,
858 expected_match_tins: Option<&[crate::Tin]>,
859 mut plugin: Option<(&mut crate::Rule, &mut crate::Context)>,
860 ) -> LexResult<Token> {
861 if self.end_reached {
862 return Ok(Token::new(
863 "#ZZ",
864 TIN_ZZ,
865 Value::Undefined,
866 "",
867 self.current_point(),
868 ));
869 }
870
871 if self.idx >= self.char_len {
872 self.end_reached = true;
873 return Ok(Token::new(
874 "#ZZ",
875 TIN_ZZ,
876 Value::Undefined,
877 "",
878 self.current_point(),
879 ));
880 }
881
882 // The fetch begins on `first`, at `entry`. `pnt` and `c`, the point
883 // a token starts at and the character under the cursor, follow the
884 // cursor whenever a custom matcher or a check moves it and declines
885 // (`resync`); `c` is `None` once the cursor is at the end.
886 let entry = self.idx;
887 let first = self.chars[entry];
888 let mut pnt = self.current_point();
889 let mut c = Some(first);
890 let mut custom_index = 0;
891
892 if let Some(token) = self.run_custom_matchers(&mut custom_index, 1_000_000.0, &mut plugin) {
893 return Ok(token);
894 }
895 self.resync(&mut pnt, &mut c);
896
897 // User-declared match tokens occupy the 1e6 matcher priority band.
898 let match_skipped = if self.options.match_lex
899 && (!self.options.match_values.is_empty()
900 || self
901 .options
902 .match_tokens
903 .values()
904 .any(|matcher| self.wants(matcher.tin)))
905 {
906 match self.run_check(self.options.match_check.clone(), pnt) {
907 CheckFlow::Continue => false,
908 CheckFlow::Skip => true,
909 CheckFlow::Token(token) => return Ok(*token),
910 }
911 } else {
912 false
913 };
914 self.resync(&mut pnt, &mut c);
915 let remaining = &self.src[self.byte_position()..];
916 let custom_value = (self.options.match_lex && !match_skipped && self.want.is_none())
917 .then(|| {
918 self.options
919 .match_values
920 .values()
921 .find_map(|matcher| match &matcher.matcher {
922 MatchTokenMatcher::Callback(callback) => callback(remaining)
923 .filter(|result| {
924 !result.source.is_empty() && remaining.starts_with(&result.source)
925 })
926 .map(|result| (result.source, result.value)),
927 MatchTokenMatcher::Regex(regex) => {
928 let captures = regex.captures(remaining)?;
929 let found = captures
930 .get(0)
931 .filter(|found| found.start() == 0 && !found.as_str().is_empty())?;
932 let source = found.as_str().to_string();
933 let value = matcher.transform.as_ref().map_or_else(
934 || {
935 matcher
936 .val
937 .clone()
938 .unwrap_or_else(|| Value::String(source.clone()))
939 },
940 |transform| {
941 let groups = captures
942 .iter()
943 .map(|capture| {
944 capture.map_or_else(String::new, |value| {
945 value.as_str().into()
946 })
947 })
948 .collect::<Vec<_>>();
949 transform(&groups)
950 },
951 );
952 Some((source, value))
953 }
954 })
955 })
956 .flatten();
957 if let Some((source, value)) = custom_value {
958 for _ in source.chars() {
959 self.advance();
960 }
961 return Ok(Token::new("#VL", TIN_VL, value, source, pnt));
962 }
963
964 let remaining = &self.src[self.byte_position()..];
965 // With no custom matcher there is nothing for the band to do: both
966 // passes walk an empty table and yield nothing, and `fix_len` is
967 // read only by that walk. Most grammars register none, and every
968 // token fetch of theirs paid the eager pass's scan of the fixed
969 // table (one closure call per fixed literal) to arrive at the
970 // `None` this guard now hands over directly. TS `makeMatchMatcher`
971 // returns null on an empty table (ts/src/lexer.ts) and the band is
972 // never installed; Go reaches the same place by defaulting
973 // `MatchLex` off unless `Options.Match` is set. Rust defaults
974 // `match_lex` true as TS does, so the guard is the parity.
975 let custom = (self.options.match_lex
976 && !match_skipped
977 && !self.options.match_tokens.is_empty())
978 .then(|| {
979 // Two passes, position-expected before eager, as go/lexer.go
980 // matchMatch and ts/src/lexer.ts makeMatchMatcher both make.
981 // One tin-ordered pass in which eagerness merely bypassed the
982 // slot gate let an eager matcher EARLIER in tin order win over
983 // an expected one later: with `p = %x31-39` beside
984 // `d = %x30-39`, the `2` of `12` lexed as the narrower class
985 // the `*d` loop never asked for. Eagerness is for firing where
986 // the slot's list is narrower than the grammar, never for
987 // outbidding what the slot names.
988 //
989 // Under a want the alternate's own tin list is the sharper
990 // gate, so one filtered pass is the whole search. With no
991 // expected list at all (a standalone lexer, no rule) nothing
992 // constrains the caller and every matcher is eligible in the
993 // first pass.
994 // The longest FIXED literal this slot expects that matches
995 // here, or 0. Only the eager pass consults it: there, a
996 // literal the slot names beats an eager-only matcher that
997 // cuts no further than it does. Without this, a character
998 // class that CONTAINS a literal the grammar also uses
999 // swallows it wherever the class is eager (`num = "0" /
1000 // posdigit *digit` beside `digit = %x30-39` rejected
1001 // `0.0.0`). LENGTH decides, not mere existence, so a keyword
1002 // literal cannot truncate a longer word: ties go to the
1003 // literal, and an eager matcher that cuts further still
1004 // wins. TS and Go do the same, in makeMatchMatcher and
1005 // matchMatch.
1006 //
1007 // Computed once per fetch and only when a regex matcher in the
1008 // eager pass has something to weigh against it, as TS
1009 // `expectedFixedLen` does (`fixLen = -1` until asked). The
1010 // scan is the whole fixed table against the slot's list; an
1011 // expected matcher that wins in pass 0, or a fetch under a
1012 // want, never needs it. Nothing the scan reads changes
1013 // between the two passes, so lazy equals eager.
1014 let mut fix_len: Option<usize> = None;
1015 let compute_fix_len = || {
1016 if self.want.is_none() && self.options.fixed.lex {
1017 expected_match_tins.map_or(0, |expected| {
1018 self.options
1019 .fixed
1020 .tokens
1021 .values()
1022 .filter(|token| {
1023 !token.source.is_empty()
1024 && expected.contains(&token.tin)
1025 && remaining.starts_with(&token.source)
1026 })
1027 .map(|token| token.source.len())
1028 .max()
1029 .unwrap_or(0)
1030 })
1031 } else {
1032 0
1033 }
1034 };
1035 let passes = if self.want.is_some() { 1 } else { 2 };
1036 (0..passes).find_map(|pass| {
1037 self.options.match_tokens.values().find_map(|matcher| {
1038 if !self.wants(matcher.tin) {
1039 return None;
1040 }
1041 if self.want.is_none() {
1042 let expected = expected_match_tins
1043 .is_none_or(|expected| expected.contains(&matcher.tin));
1044 if pass == 0 {
1045 if !expected {
1046 return None;
1047 }
1048 } else if expected || !matcher.eager {
1049 return None;
1050 }
1051 }
1052 let result = match &matcher.matcher {
1053 MatchTokenMatcher::Regex(regex) => regex
1054 .find(remaining)
1055 .filter(|found| found.start() == 0)
1056 // The eager pass yields to an expected
1057 // literal it cannot out-cut; the fixed
1058 // matcher (2e6) runs next and takes it. See
1059 // `fix_len` above.
1060 .filter(|found| {
1061 pass == 0 || {
1062 let fix_len = *fix_len.get_or_insert_with(compute_fix_len);
1063 fix_len == 0 || found.len() > fix_len
1064 }
1065 })
1066 .map(|found| {
1067 let source = found.as_str().to_string();
1068 (source.clone(), Value::String(source))
1069 }),
1070 MatchTokenMatcher::Callback(callback) => callback(remaining)
1071 .filter(|result| {
1072 !result.source.is_empty() && remaining.starts_with(&result.source)
1073 })
1074 .map(|result| (result.source, result.value)),
1075 };
1076 result.map(|(source, value)| (matcher.name.clone(), matcher.tin, source, value))
1077 })
1078 })
1079 });
1080 if let Some(Some((name, tin, matched, value))) = custom {
1081 for _ in matched.chars() {
1082 self.advance();
1083 }
1084 return Ok(Token::new(name, tin, value, matched, pnt));
1085 }
1086
1087 if let Some(token) = self.run_custom_matchers(&mut custom_index, 2_000_000.0, &mut plugin) {
1088 return Ok(token);
1089 }
1090 self.resync(&mut pnt, &mut c);
1091
1092 // Fixed literals occupy the 2e6 band and use longest-match wins.
1093 let fixed_skipped = if self.options.fixed.lex {
1094 match self.run_check(self.options.fixed.check.clone(), pnt) {
1095 CheckFlow::Continue => false,
1096 CheckFlow::Skip => true,
1097 CheckFlow::Token(token) => return Ok(*token),
1098 }
1099 } else {
1100 false
1101 };
1102 self.resync(&mut pnt, &mut c);
1103 let fixed_skipped = fixed_skipped
1104 || !self.listed(entry, first, self.options.fixed.check.is_some(), |ch| {
1105 self.options
1106 .fixed
1107 .tokens
1108 .values()
1109 .any(|token| token.source.starts_with(ch))
1110 });
1111 let remaining = &self.src[self.byte_position()..];
1112 // The winner is carried out of the table as its position, not as a
1113 // copy of its text. `Token::new` takes the name and the source text
1114 // by reference and stores both inline, so the only owned copy the
1115 // token needs is the one inside `Value::String`. Naming the match
1116 // as three owned values cost three `String` allocations per fixed
1117 // token, two of them freed again before the token was built.
1118 // The first byte decides almost every entry. Asking `wants` and then
1119 // `starts_with` of each fixed token in turn ran a tin lookup and a
1120 // `memcmp` per token in the grammar per token in the input, and a
1121 // grammar with fifty fixed tokens pays fifty of each to reject
1122 // forty-nine. One byte answers the same question, and an empty
1123 // source is kept out by its own check, which only entries that
1124 // already matched the byte ever reach.
1125 let first_byte = remaining.as_bytes().first().copied();
1126 let fixed = (self.options.fixed.lex && !fixed_skipped)
1127 .then(|| {
1128 self.options
1129 .fixed
1130 .tokens
1131 .values()
1132 .enumerate()
1133 .filter(|(_, token)| {
1134 token.source.as_bytes().first().copied() == first_byte
1135 && !token.source.is_empty()
1136 && self.wants(token.tin)
1137 && remaining.starts_with(&token.source)
1138 })
1139 .max_by_key(|(_, token)| token.source.len())
1140 .map(|(index, token)| (index, token.source.chars().count()))
1141 })
1142 .flatten();
1143 if let Some((index, source_chars)) = fixed {
1144 for _ in 0..source_chars {
1145 self.advance();
1146 }
1147 let (_, token) = self
1148 .options
1149 .fixed
1150 .tokens
1151 .get_index(index)
1152 .expect("index came from this table, which nothing writes to mid-parse");
1153 return Ok(Token::new(
1154 &token.name,
1155 token.tin,
1156 Value::String(token.source.clone()),
1157 token.source.as_str(),
1158 pnt,
1159 ));
1160 }
1161
1162 if let Some(token) = self.run_custom_matchers(&mut custom_index, 3_000_000.0, &mut plugin) {
1163 return Ok(token);
1164 }
1165 self.resync(&mut pnt, &mut c);
1166
1167 // 1. Whitespace
1168 let space_skipped = if self.options.space.lex && self.wants(TIN_SP) {
1169 match self.run_check(self.options.space.check.clone(), pnt) {
1170 CheckFlow::Continue => false,
1171 CheckFlow::Skip => true,
1172 CheckFlow::Token(token) => return Ok(*token),
1173 }
1174 } else {
1175 false
1176 };
1177 self.resync(&mut pnt, &mut c);
1178 if self.options.space.lex
1179 && !space_skipped
1180 && self.wants(TIN_SP)
1181 && c.is_some_and(|ch| self.char_sets.space.contains(ch))
1182 && self.listed(entry, first, self.options.space.check.is_some(), |ch| {
1183 self.char_sets.space.contains(ch)
1184 })
1185 {
1186 let mut src = String::new();
1187 while let Some(ch) = self.peek() {
1188 if self.char_sets.space.contains(ch) {
1189 src.push(ch);
1190 self.advance();
1191 } else {
1192 break;
1193 }
1194 }
1195 return Ok(Token::new(
1196 "#SP",
1197 TIN_SP,
1198 Value::String(src.clone()),
1199 src,
1200 pnt,
1201 ));
1202 }
1203
1204 if let Some(token) = self.run_custom_matchers(&mut custom_index, 4_000_000.0, &mut plugin) {
1205 return Ok(token);
1206 }
1207 self.resync(&mut pnt, &mut c);
1208
1209 // 2. Line ending
1210 let line_skipped = if self.options.line.lex && self.wants(TIN_LN) {
1211 match self.run_check(self.options.line.check.clone(), pnt) {
1212 CheckFlow::Continue => false,
1213 CheckFlow::Skip => true,
1214 CheckFlow::Token(token) => return Ok(*token),
1215 }
1216 } else {
1217 false
1218 };
1219 self.resync(&mut pnt, &mut c);
1220 let line_skipped = line_skipped
1221 || !self.listed(entry, first, self.options.line.check.is_some(), |ch| {
1222 self.char_sets.line_ends.contains(ch)
1223 });
1224 if self.options.line.lex
1225 && !line_skipped
1226 && self.wants(TIN_LN)
1227 && c.is_some_and(|ch| self.char_sets.line_ends.contains(ch))
1228 {
1229 let mut src = String::new();
1230 let mut seen = std::collections::HashSet::new();
1231 while let Some(ch) = self.peek() {
1232 if !self.char_sets.line_ends.contains(ch) {
1233 break;
1234 }
1235 if self.options.line.single && !seen.insert(ch) {
1236 break;
1237 }
1238 src.push(self.advance().expect("peeked character must advance"));
1239 }
1240 self.ci = 1;
1241 return Ok(Token::new(
1242 "#LN",
1243 TIN_LN,
1244 Value::String(src.clone()),
1245 src,
1246 pnt,
1247 ));
1248 }
1249
1250 if let Some(bad_char @ ('\u{2028}' | '\u{2029}')) = c {
1251 if self.options.line.lex && !line_skipped && self.wants(TIN_LN) {
1252 self.advance();
1253 let err = TabnasError::new(
1254 "unexpected",
1255 bad_char.to_string(),
1256 "",
1257 pnt.site.pos,
1258 pnt.site.ri,
1259 pnt.site.ci,
1260 );
1261 self.err = Some(err.clone());
1262 return Err(Box::new(err));
1263 }
1264 }
1265
1266 if let Some(token) = self.run_custom_matchers(&mut custom_index, 5_000_000.0, &mut plugin) {
1267 return Ok(token);
1268 }
1269 self.resync(&mut pnt, &mut c);
1270
1271 // 3. Quoted strings. These precede comments in the canonical matcher
1272 // order, so an overlapping quote/comment opener is a string unless
1273 // string matching explicitly abandons the malformed candidate.
1274 let string_skipped = if self.options.string.lex && self.wants(TIN_ST) {
1275 match self.run_check(self.options.string.check.clone(), pnt) {
1276 CheckFlow::Continue => false,
1277 CheckFlow::Skip => true,
1278 CheckFlow::Token(token) => return Ok(*token),
1279 }
1280 } else {
1281 false
1282 };
1283 self.resync(&mut pnt, &mut c);
1284 let quote = c.filter(|ch| self.char_sets.string.contains(*ch));
1285 if let Some(quote) = quote.filter(|_| {
1286 self.options.string.lex
1287 && !string_skipped
1288 && self.wants(TIN_ST)
1289 && self.listed(entry, first, self.options.string.check.is_some(), |ch| {
1290 self.char_sets.string.contains(ch)
1291 })
1292 }) {
1293 let start = (self.idx, self.ri, self.ci);
1294 match self.match_string(quote, pnt) {
1295 result @ Ok(_) => return result,
1296 Err(error) if !self.options.string.abandon => return Err(error),
1297 Err(_) => {
1298 (self.idx, self.ri, self.ci) = start;
1299 self.err = None;
1300 }
1301 }
1302 }
1303
1304 if let Some(token) = self.run_custom_matchers(&mut custom_index, 6_000_000.0, &mut plugin) {
1305 return Ok(token);
1306 }
1307 self.resync(&mut pnt, &mut c);
1308
1309 // 4. Comments (longest opening marker wins; ties sort by name).
1310 let comment_skipped = if self.options.comment.lex && self.wants(TIN_CM) {
1311 match self.run_check(self.options.comment.check.clone(), pnt) {
1312 CheckFlow::Continue => false,
1313 CheckFlow::Skip => true,
1314 CheckFlow::Token(token) => return Ok(*token),
1315 }
1316 } else {
1317 false
1318 };
1319 self.resync(&mut pnt, &mut c);
1320 if self.options.comment.lex
1321 && !comment_skipped
1322 && self.wants(TIN_CM)
1323 && self.listed(entry, first, self.options.comment.check.is_some(), |ch| {
1324 self.options
1325 .comment
1326 .definitions
1327 .values()
1328 .any(|definition| definition.start.starts_with(ch))
1329 })
1330 {
1331 if let Some(token) = self.match_comment(pnt)? {
1332 return Ok(token);
1333 }
1334 }
1335
1336 if let Some(token) = self.run_custom_matchers(&mut custom_index, 7_000_000.0, &mut plugin) {
1337 return Ok(token);
1338 }
1339 self.resync(&mut pnt, &mut c);
1340
1341 // 5. Numbers
1342 let number_skipped = if self.options.number.lex && self.wants(TIN_NR) {
1343 match self.run_check(self.options.number.check.clone(), pnt) {
1344 CheckFlow::Continue => false,
1345 CheckFlow::Skip => true,
1346 CheckFlow::Token(token) => return Ok(*token),
1347 }
1348 } else {
1349 false
1350 };
1351 self.resync(&mut pnt, &mut c);
1352 let could_start_number =
1353 |ch: char| ch == '-' || ch == '+' || ch == '.' || ch.is_ascii_digit();
1354 if self.options.number.lex
1355 && !number_skipped
1356 && self.wants(TIN_NR)
1357 && c.is_some_and(could_start_number)
1358 && self.listed(
1359 entry,
1360 first,
1361 self.options.number.check.is_some(),
1362 could_start_number,
1363 )
1364 {
1365 if let Some(tkn) = self.match_number(pnt)? {
1366 return Ok(tkn);
1367 }
1368 }
1369
1370 if let Some(token) = self.run_custom_matchers(&mut custom_index, 8_000_000.0, &mut plugin) {
1371 return Ok(token);
1372 }
1373 self.resync(&mut pnt, &mut c);
1374
1375 // 6. Text and named/regex values share the same delimited run.
1376 // Negotiated lexing gates this combined family by its primary token
1377 // identity (#TX), matching the TypeScript and Go dispatchers. Once
1378 // entered, an exact or regexp value definition may still produce
1379 // #VL; the caller rejects and rolls that cut back when #VL was not
1380 // requested.
1381 let text_matcher_wanted = self.wants(TIN_TX);
1382 let value_lex = self.options.value.lex && text_matcher_wanted;
1383 let text_lex = self.options.text.lex && text_matcher_wanted;
1384 let text_skipped = if text_lex || value_lex {
1385 match self.run_check(self.options.text.check.clone(), pnt) {
1386 CheckFlow::Continue => false,
1387 CheckFlow::Skip => true,
1388 CheckFlow::Token(token) => return Ok(*token),
1389 }
1390 } else {
1391 false
1392 };
1393 self.resync(&mut pnt, &mut c);
1394 if (text_lex || value_lex) && !text_skipped && !self.is_text_delimiter_here() {
1395 let start = (self.idx, self.ri, self.ci);
1396 // Only a `value` definition declaring `consume` looks at the
1397 // rest of the document, and the JSON grammar has none -- but
1398 // this ran for every text token, copying the whole tail of the
1399 // input each time. `self.src` is borrowed from the caller for
1400 // `'a` and is never reassigned, so reading the reference out
1401 // before the scan below gives a slice that does not borrow
1402 // `self` and survives the `&mut self` the scan needs.
1403 let source: &'a str = self.src;
1404 let remaining = &source[self.byte_position()..];
1405 let mut src = String::new();
1406 while let Some(ch) = self.peek() {
1407 if self.is_text_delimiter_here() {
1408 break;
1409 }
1410 src.push(ch);
1411 self.advance();
1412 }
1413
1414 let mut output = None;
1415 if value_lex {
1416 if let Some(definition) = self
1417 .options
1418 .value
1419 .definitions
1420 .get(&src)
1421 .filter(|definition| definition.matcher.is_none())
1422 .cloned()
1423 {
1424 output = Some(Token::new(
1425 "#VL",
1426 TIN_VL,
1427 definition
1428 .val
1429 .clone()
1430 .unwrap_or_else(|| Value::String(src.clone())),
1431 src.clone(),
1432 pnt,
1433 ));
1434 }
1435
1436 if output.is_none() {
1437 let mut definitions: Vec<_> = self
1438 .options
1439 .value
1440 .definitions
1441 .iter()
1442 .filter(|(_, definition)| definition.matcher.is_some())
1443 .map(|(name, definition)| (name.clone(), definition.clone()))
1444 .collect();
1445 definitions.sort_by(|(name_a, _), (name_b, _)| name_a.cmp(name_b));
1446 for (_, definition) in definitions {
1447 let regex = definition.matcher.as_ref().expect("filtered matcher");
1448 let target: &str = if definition.consume { remaining } else { &src };
1449 let Some(captures) = regex.captures(target) else {
1450 continue;
1451 };
1452 let Some(found) = captures.get(0).filter(|found| found.start() == 0) else {
1453 continue;
1454 };
1455 if !definition.consume && found.end() != target.len() {
1456 continue;
1457 }
1458 let matched = found.as_str().to_string();
1459 let value = definition.transform.as_ref().map_or_else(
1460 || {
1461 definition
1462 .val
1463 .clone()
1464 .unwrap_or_else(|| Value::String(matched.clone()))
1465 },
1466 |transform| {
1467 let groups = captures
1468 .iter()
1469 .map(|capture| {
1470 capture
1471 .map_or_else(String::new, |value| value.as_str().into())
1472 })
1473 .collect::<Vec<_>>();
1474 transform(&groups)
1475 },
1476 );
1477 if definition.consume {
1478 (self.idx, self.ri, self.ci) = start;
1479 for _ in matched.chars() {
1480 self.advance();
1481 }
1482 }
1483 output = Some(Token::new("#VL", TIN_VL, value, matched, pnt));
1484 break;
1485 }
1486 }
1487 }
1488
1489 if output.is_none() && (!text_lex || text_skipped) {
1490 (self.idx, self.ri, self.ci) = start;
1491 } else if output.is_none() {
1492 output = Some(Token::new(
1493 "#TX",
1494 TIN_TX,
1495 Value::String(src.clone()),
1496 src,
1497 pnt,
1498 ));
1499 }
1500
1501 if let Some(mut token) = output {
1502 let value = std::mem::replace(&mut token.val, Value::Undefined);
1503 token.val = self.modify_text_value(value, &mut plugin);
1504 return Ok(token);
1505 }
1506 }
1507
1508 if let Some(token) = self.run_custom_matchers(&mut custom_index, f64::INFINITY, &mut plugin)
1509 {
1510 return Ok(token);
1511 }
1512 self.resync(&mut pnt, &mut c);
1513
1514 // 7. Unclaimed character -> Error: unexpected, raised where the
1515 // cursor stands, as TypeScript's `Lex.next` builds its #BD at the
1516 // live `pnt` and Go's `nextUnfiltered2` at `l.pnt`. A matcher that
1517 // stepped over the last character and declined leaves no character
1518 // to name: the error then names none, at the end of the source, as
1519 // it does in both. This took a character unconditionally, and so
1520 // panicked on a lone byte-order mark under the xml plugin, whose
1521 // matcher steps over a mark at the start of the source.
1522 let bad_source = self.advance().map_or_else(String::new, String::from);
1523 let err = TabnasError::new(
1524 "unexpected",
1525 bad_source,
1526 "",
1527 pnt.site.pos,
1528 pnt.site.ri,
1529 pnt.site.ci,
1530 );
1531 self.err = Some(err.clone());
1532 Err(Box::new(err))
1533 }
1534
1535 fn match_comment(&mut self, pnt: Point) -> LexResult<Option<Token>> {
1536 let remaining = &self.src[self.byte_position()..];
1537 let mut definitions: Vec<_> = self
1538 .options
1539 .comment
1540 .definitions
1541 .iter()
1542 .filter(|(_, definition)| {
1543 // Same first-byte test as the fixed-token scan above.
1544 definition.start.as_bytes().first().copied()
1545 == remaining.as_bytes().first().copied()
1546 && !definition.start.is_empty()
1547 && definition.lex
1548 && remaining.starts_with(&definition.start)
1549 })
1550 .collect();
1551 definitions.sort_by(|(name_a, a), (name_b, b)| {
1552 b.start
1553 .len()
1554 .cmp(&a.start.len())
1555 .then_with(|| name_a.cmp(name_b))
1556 });
1557 let Some((_, definition)) = definitions.first() else {
1558 return Ok(None);
1559 };
1560 let definition = (*definition).clone();
1561 let mut src = String::new();
1562 for _ in definition.start.chars() {
1563 src.push(self.advance().expect("comment marker must advance"));
1564 }
1565
1566 let mut terminated_by_suffix = false;
1567 let mut closed = definition.line;
1568 loop {
1569 let remainder = &self.src[self.byte_position()..];
1570 let suffix = definition
1571 .suffixes
1572 .iter()
1573 .filter(|suffix| !suffix.is_empty() && remainder.starts_with(*suffix))
1574 .max_by_key(|suffix| suffix.len())
1575 .cloned();
1576 let suffix = suffix.or_else(|| {
1577 let matcher = definition.suffix_matcher.as_ref()?;
1578 let effect = matcher.run(remainder);
1579 if effect.is_some() {
1580 return effect;
1581 }
1582 let saved = self.state();
1583 let wanted = self.want.clone();
1584 let token = matcher.run_imperative(self);
1585 self.restore(saved);
1586 self.want = wanted;
1587 token.map(|token| token.src.to_string())
1588 });
1589 let remainder = &self.src[self.byte_position()..];
1590 let suffix =
1591 suffix.filter(|suffix| !suffix.is_empty() && remainder.starts_with(suffix));
1592 if let Some(suffix) = suffix {
1593 for _ in suffix.chars() {
1594 src.push(self.advance().expect("comment suffix must advance"));
1595 }
1596 terminated_by_suffix = true;
1597 closed = true;
1598 break;
1599 }
1600 if !definition.line
1601 && !definition.end.is_empty()
1602 && remainder.starts_with(&definition.end)
1603 {
1604 for _ in definition.end.chars() {
1605 src.push(self.advance().expect("comment end must advance"));
1606 }
1607 closed = true;
1608 break;
1609 }
1610 let Some(ch) = self.peek() else {
1611 break;
1612 };
1613 if definition.line && (self.char_sets.line_ends.contains(ch)) {
1614 break;
1615 }
1616 src.push(self.advance().expect("comment body must advance"));
1617 }
1618
1619 if !closed {
1620 let err = TabnasError::new(
1621 "unterminated_comment",
1622 src,
1623 "",
1624 pnt.site.pos,
1625 pnt.site.ri,
1626 pnt.site.ci,
1627 );
1628 self.err = Some(err.clone());
1629 return Err(Box::new(err));
1630 }
1631
1632 if definition.eat_line && !terminated_by_suffix {
1633 while let Some(ch) = self.peek() {
1634 if !self.char_sets.line_ends.contains(ch) {
1635 break;
1636 }
1637 src.push(self.advance().expect("comment line tail must advance"));
1638 }
1639 }
1640
1641 Ok(Some(Token::new(
1642 "#CM",
1643 TIN_CM,
1644 Value::String(src.clone()),
1645 src,
1646 pnt,
1647 )))
1648 }
1649
1650 fn match_number(&mut self, pnt: Point) -> LexResult<Option<Token>> {
1651 let start_idx = self.idx;
1652 let mut src = String::new();
1653
1654 // Optional sign.
1655 if matches!(self.peek(), Some('-' | '+')) {
1656 src.push(self.advance().unwrap());
1657 }
1658
1659 // Base-prefixed integers are complete at the final valid digit.
1660 if self.peek() == Some('0') {
1661 if let Some(prefix) = self.peek_at(1) {
1662 let radix = match prefix {
1663 'x' | 'X' if self.options.number.hex => Some(16),
1664 'o' | 'O' if self.options.number.oct => Some(8),
1665 'b' | 'B' if self.options.number.bin => Some(2),
1666 _ => None,
1667 };
1668 if let Some(radix) = radix {
1669 src.push(self.advance().expect("peeked zero"));
1670 src.push(self.advance().expect("peeked base prefix"));
1671 let mut saw_digit = false;
1672 while let Some(ch) = self.peek() {
1673 if ch.is_digit(radix) {
1674 saw_digit = true;
1675 src.push(self.advance().expect("peeked base digit"));
1676 } else if self
1677 .options
1678 .number
1679 .sep
1680 .as_ref()
1681 .is_some_and(|separator| separator.contains(ch))
1682 {
1683 src.push(self.advance().expect("peeked base digit"));
1684 } else {
1685 break;
1686 }
1687 }
1688 if saw_digit && self.is_text_delimiter_here() {
1689 if self
1690 .exclude_regex
1691 .as_ref()
1692 .is_some_and(|regex| regex.is_match(&src))
1693 {
1694 self.reset_number(start_idx, pnt);
1695 return Ok(None);
1696 }
1697 if self.options.value.lex {
1698 if let Some(definition) = self
1699 .options
1700 .value
1701 .definitions
1702 .get(&src)
1703 .filter(|definition| definition.matcher.is_none())
1704 {
1705 return Ok(Some(Token::new(
1706 "#VL",
1707 TIN_VL,
1708 definition
1709 .val
1710 .clone()
1711 .unwrap_or_else(|| Value::String(src.clone())),
1712 src,
1713 pnt,
1714 )));
1715 }
1716 }
1717 // The digits of the literal, prefix, sign and any
1718 // separators removed, folded as they are read. The
1719 // fold keeps a bounded head, a digit count and a
1720 // sticky bit, so a literal of any length costs the
1721 // same handful of bytes: buffering the digits
1722 // instead would let one long token multiply the
1723 // memory the source already holds.
1724 let mut fold = DigitFold::new(radix.trailing_zeros());
1725 for ch in src.chars().skip_while(|ch| matches!(ch, '-' | '+')).skip(2) {
1726 if self
1727 .options
1728 .number
1729 .sep
1730 .as_ref()
1731 .is_some_and(|separator| separator.contains(ch))
1732 {
1733 continue;
1734 }
1735 fold.push(ch.to_digit(radix).expect("validated base digit"));
1736 }
1737 let mut value = fold.finish();
1738 if src.starts_with('-') {
1739 value = -value;
1740 }
1741 return Ok(Some(Token::new(
1742 "#NR",
1743 TIN_NR,
1744 Value::Number(value),
1745 src,
1746 pnt,
1747 )));
1748 }
1749 self.reset_number(start_idx, pnt);
1750 return Ok(None);
1751 }
1752 }
1753 }
1754
1755 let Some(ch) = self.peek() else {
1756 self.reset_number(start_idx, pnt);
1757 return Ok(None);
1758 };
1759 if ch == '.' {
1760 if !self.peek_at(1).is_some_and(|next| next.is_ascii_digit()) {
1761 self.reset_number(start_idx, pnt);
1762 return Ok(None);
1763 }
1764 src.push(self.advance().expect("peeked leading decimal point"));
1765 } else if !ch.is_ascii_digit() {
1766 self.reset_number(start_idx, pnt);
1767 return Ok(None);
1768 }
1769
1770 let (has_digits, edge_separator) = self.scan_number_digits(&mut src);
1771 if !has_digits || edge_separator {
1772 self.reset_number(start_idx, pnt);
1773 return Ok(None);
1774 }
1775
1776 // The canonical regexp admits a trailing decimal point and an
1777 // exponent after it (`2.e3`), but declines `0.a` as one text run.
1778 if self.peek() == Some('.') {
1779 let next = self.peek_at(1);
1780 let exponent_after_dot = matches!(next, Some('e' | 'E'))
1781 && match self.peek_at(2) {
1782 Some('+' | '-') => self.peek_at(3).is_some_and(|ch| ch.is_ascii_digit()),
1783 Some(ch) => ch.is_ascii_digit(),
1784 None => false,
1785 };
1786 if next.is_some_and(|ch| ch.is_ascii_digit()) {
1787 src.push(self.advance().expect("peeked decimal point"));
1788 let (_, edge_separator) = self.scan_number_digits(&mut src);
1789 if edge_separator {
1790 self.reset_number(start_idx, pnt);
1791 return Ok(None);
1792 }
1793 } else if next.is_some()
1794 && !self.is_text_delimiter_at(self.idx + 1)
1795 && next != Some('.')
1796 && !exponent_after_dot
1797 {
1798 self.reset_number(start_idx, pnt);
1799 return Ok(None);
1800 } else {
1801 src.push(self.advance().expect("peeked trailing decimal point"));
1802 }
1803 }
1804
1805 if matches!(self.peek(), Some('e' | 'E')) {
1806 let exponent_start = self.idx;
1807 let source_len = src.len();
1808 src.push(self.advance().expect("peeked exponent marker"));
1809 if matches!(self.peek(), Some('+' | '-')) {
1810 src.push(self.advance().expect("peeked exponent sign"));
1811 }
1812 let (has_exponent_digits, edge_separator) = self.scan_number_digits(&mut src);
1813 if edge_separator {
1814 self.reset_number(start_idx, pnt);
1815 return Ok(None);
1816 }
1817 if !has_exponent_digits {
1818 self.idx = exponent_start;
1819 src.truncate(source_len);
1820 }
1821 }
1822
1823 if !self.is_text_delimiter_here() {
1824 self.reset_number(start_idx, pnt);
1825 return Ok(None);
1826 }
1827
1828 // Check exclusion regex (e.g. ^00+)
1829 if let Some(ref re) = self.exclude_regex {
1830 if re.is_match(&src) {
1831 // Number is excluded, backtrack
1832 self.reset_number(start_idx, pnt);
1833 return Ok(None);
1834 }
1835 }
1836
1837 if self.options.value.lex {
1838 if let Some(definition) = self
1839 .options
1840 .value
1841 .definitions
1842 .get(&src)
1843 .filter(|definition| definition.matcher.is_none())
1844 {
1845 return Ok(Some(Token::new(
1846 "#VL",
1847 TIN_VL,
1848 definition
1849 .val
1850 .clone()
1851 .unwrap_or_else(|| Value::String(src.clone())),
1852 src,
1853 pnt,
1854 )));
1855 }
1856 }
1857
1858 // Parse float
1859 let parse_src = self.options.number.sep.as_ref().map_or_else(
1860 || src.clone(),
1861 |separator| src.chars().filter(|ch| !separator.contains(*ch)).collect(),
1862 );
1863 match parse_src.parse::<f64>() {
1864 Ok(num) => Ok(Some(Token::new(
1865 "#NR",
1866 TIN_NR,
1867 Value::Number(num),
1868 src,
1869 pnt,
1870 ))),
1871 Err(_) => {
1872 self.reset_number(start_idx, pnt);
1873 Ok(None)
1874 }
1875 }
1876 }
1877
1878 fn reset_number(&mut self, start_idx: usize, pnt: Point) {
1879 self.idx = start_idx;
1880 self.ri = pnt.site.ri;
1881 self.ci = pnt.site.ci;
1882 }
1883
1884 /// Consume a decimal digit/separator run. Separators are legal only
1885 /// between digits; a leading or trailing separator makes the whole run
1886 /// fall through to text, matching the TypeScript regexp and Go scanner.
1887 fn scan_number_digits(&mut self, src: &mut String) -> (bool, bool) {
1888 // The run is measured before any of it is consumed. Advancing
1889 // as it goes would hold `&mut self` across a read of
1890 // `self.options.number.sep`, and the way that used to be settled
1891 // was to clone the separator — an allocation and a free for
1892 // every number in the input, for a value that cannot change
1893 // while one number is being scanned.
1894 let run_start = self.idx;
1895 let mut saw_digit = false;
1896 let mut last_was_separator = false;
1897 let mut end = run_start;
1898 {
1899 let separator = self.options.number.sep.as_deref();
1900 while let Some(ch) = self.chars.get(end).copied() {
1901 if ch.is_ascii_digit() {
1902 saw_digit = true;
1903 last_was_separator = false;
1904 } else if separator.is_some_and(|separator| separator.contains(ch)) {
1905 last_was_separator = true;
1906 } else {
1907 break;
1908 }
1909 end += 1;
1910 }
1911 }
1912 while self.idx < end {
1913 src.push(self.advance().expect("scanned number character"));
1914 }
1915 let starts_with_separator = self.idx > run_start
1916 && self.options.number.sep.as_deref().is_some_and(|separator| {
1917 self.chars[run_start..self.idx]
1918 .first()
1919 .is_some_and(|ch| separator.contains(*ch))
1920 });
1921 (saw_digit, starts_with_separator || last_was_separator)
1922 }
1923
1924 fn match_string(&mut self, quote: char, pnt: Point) -> LexResult<Token> {
1925 let quote_char = self.advance().unwrap();
1926 let mut out_str = String::new();
1927 let mut raw_src = String::new();
1928 raw_src.push(quote_char);
1929
1930 let mut pending_high_surrogate: Option<u16> = None;
1931
1932 // The body classes of TypeScript's `buildStringBodySpec`
1933 // (ts/src/lexer.ts), which Go's `BuildStringBodySpec` ports: a
1934 // line character is LINE or LINE+ROW only inside a multi-line
1935 // string, where it resets the column and, in `line.rowChars`,
1936 // counts a row; anywhere else in a body it is plain content,
1937 // counted as a column, unless it is a control character, which
1938 // stops the body as `unprintable`. This loop used to count a row
1939 // for any row character it stepped over, string body or not, and
1940 // to refuse any line character inside a single-line string: with
1941 // json5's U+2028 and U+2029 as row characters, `y` in
1942 // `"a<U+2028>b" y` sat on row 2 here and on row 1 in TypeScript
1943 // and Go, and with the two in `line.chars` as well the string was
1944 // `unprintable` (tabnas/parser#263).
1945 let multi_line = self.options.string.multi_chars.contains(quote);
1946
1947 while let Some(c) = self.peek() {
1948 if c == quote {
1949 raw_src.push(self.advance().unwrap());
1950 // Rust strings cannot represent a lone UTF-16 surrogate, so
1951 // preserve the Go-port behavior and fold it to U+FFFD.
1952 self.flush_surrogate(&mut pending_high_surrogate, &mut out_str);
1953 return Ok(Token::new(
1954 "#ST",
1955 TIN_ST,
1956 Value::String(out_str),
1957 raw_src,
1958 pnt,
1959 ));
1960 }
1961
1962 if let Some(replacement) = self.options.string.replace.get(&c).cloned() {
1963 raw_src.push(self.advance().expect("peeked character must advance"));
1964 self.flush_surrogate(&mut pending_high_surrogate, &mut out_str);
1965 out_str.push_str(&replacement);
1966 continue;
1967 }
1968
1969 if multi_line && self.char_sets.line.contains(c) {
1970 raw_src.push(
1971 self.advance_string_line()
1972 .expect("peeked character must advance"),
1973 );
1974 out_str.push(c);
1975 continue;
1976 }
1977
1978 // A control character stops the body: a line character
1979 // always, since a single-line string cannot hold one, and any
1980 // other unless `string.allowControl` admits it. Sited ON the
1981 // character, as TypeScript does (`pnt.sI = sI; pnt.cI = cI`
1982 // before its `bad()` call, ts/src/lexer.ts). `pnt` is the
1983 // opening quote, and reporting that put every embedded
1984 // newline at the start of its string.
1985 if (c as u32) < 32
1986 && (self.char_sets.line.contains(c) || !self.options.string.allow_control)
1987 {
1988 let site = self.current_point().site;
1989 let err =
1990 TabnasError::new("unprintable", c.to_string(), "", site.pos, site.ri, site.ci);
1991 self.err = Some(err.clone());
1992 return Err(Box::new(err));
1993 }
1994
1995 if c == self.options.string.escape_char {
1996 raw_src.push(self.advance().unwrap());
1997 let esc_point = self.current_point();
1998 if let Some(esc) = self.advance() {
1999 raw_src.push(esc);
2000 if let Some(replacement) = self.options.string.escape.get(&esc).cloned() {
2001 self.flush_surrogate(&mut pending_high_surrogate, &mut out_str);
2002 out_str.push_str(&replacement);
2003 continue;
2004 }
2005 match esc {
2006 'u' => {
2007 // Unicode escape: \uXXXX or \u{X...}. An
2008 // invalid one is reported on the backslash
2009 // with the span TypeScript cuts: six source
2010 // characters for the fixed-width form, four
2011 // for `\x`, and through the closing brace
2012 // (or to the end of the source) for the
2013 // braced form -- clipped, never padded, so a
2014 // truncated escape at end of input reports
2015 // exactly the characters that are there.
2016 if self.peek() == Some('{') && !self.options.string.escape_strict {
2017 raw_src.push(self.advance().unwrap()); // '{'
2018 let mut hex = String::new();
2019 let mut closed = false;
2020 while let Some(h) = self.peek() {
2021 if h == '}' {
2022 raw_src.push(self.advance().unwrap());
2023 closed = true;
2024 break;
2025 }
2026 raw_src.push(self.advance().unwrap());
2027 hex.push(h);
2028 }
2029
2030 if !closed
2031 || hex.is_empty()
2032 || hex.len() > 6
2033 || !hex.chars().all(|ch| ch.is_ascii_hexdigit())
2034 {
2035 let err = TabnasError::new(
2036 "invalid_unicode",
2037 self.source_span(esc_point.site.pos - 1, self.idx),
2038 "",
2039 esc_point.site.pos - 1,
2040 esc_point.site.ri,
2041 esc_point.site.ci - 1,
2042 );
2043 self.err = Some(err.clone());
2044 return Err(Box::new(err));
2045 }
2046
2047 let cp = match u32::from_str_radix(&hex, 16) {
2048 Ok(val) if val <= 0x10FFFF => val,
2049 _ => {
2050 let err = TabnasError::new(
2051 "invalid_unicode",
2052 self.source_span(esc_point.site.pos - 1, self.idx),
2053 "",
2054 esc_point.site.pos - 1,
2055 esc_point.site.ri,
2056 esc_point.site.ci - 1,
2057 );
2058 self.err = Some(err.clone());
2059 return Err(Box::new(err));
2060 }
2061 };
2062
2063 self.emit_unicode_escape(
2064 cp,
2065 &mut pending_high_surrogate,
2066 &mut out_str,
2067 );
2068 } else {
2069 // Exactly 4 hex digits: \uXXXX
2070 let mut hex = String::new();
2071 for _ in 0..4 {
2072 if let Some(h) = self.peek() {
2073 if h.is_ascii_hexdigit() {
2074 raw_src.push(self.advance().unwrap());
2075 hex.push(h);
2076 } else {
2077 break;
2078 }
2079 } else {
2080 break;
2081 }
2082 }
2083
2084 if hex.len() != 4 {
2085 let err = TabnasError::new(
2086 "invalid_unicode",
2087 self.source_span(
2088 esc_point.site.pos - 1,
2089 esc_point.site.pos + 5,
2090 ),
2091 "",
2092 esc_point.site.pos - 1,
2093 esc_point.site.ri,
2094 esc_point.site.ci - 1,
2095 );
2096 self.err = Some(err.clone());
2097 return Err(Box::new(err));
2098 }
2099
2100 let cp = u16::from_str_radix(&hex, 16).map_err(|_| {
2101 let err = TabnasError::new(
2102 "invalid_unicode",
2103 self.source_span(
2104 esc_point.site.pos - 1,
2105 esc_point.site.pos + 5,
2106 ),
2107 "",
2108 esc_point.site.pos - 1,
2109 esc_point.site.ri,
2110 esc_point.site.ci - 1,
2111 );
2112 self.err = Some(err.clone());
2113 err
2114 })?;
2115
2116 self.emit_unicode_escape(
2117 u32::from(cp),
2118 &mut pending_high_surrogate,
2119 &mut out_str,
2120 );
2121 }
2122 }
2123 'x' if !self.options.string.escape_strict => {
2124 let mut hex = String::new();
2125 for _ in 0..2 {
2126 if let Some(h) = self.peek() {
2127 if h.is_ascii_hexdigit() {
2128 raw_src.push(
2129 self.advance().expect("peeked character must advance"),
2130 );
2131 hex.push(h);
2132 }
2133 }
2134 }
2135 if hex.len() != 2 {
2136 let err = TabnasError::new(
2137 "invalid_ascii",
2138 self.source_span(
2139 esc_point.site.pos - 1,
2140 esc_point.site.pos + 3,
2141 ),
2142 "",
2143 esc_point.site.pos - 1,
2144 esc_point.site.ri,
2145 esc_point.site.ci - 1,
2146 );
2147 self.err = Some(err.clone());
2148 return Err(Box::new(err));
2149 }
2150 let byte = u8::from_str_radix(&hex, 16).expect("validated ASCII hex");
2151 self.flush_surrogate(&mut pending_high_surrogate, &mut out_str);
2152 out_str.push(char::from(byte));
2153 }
2154 other => {
2155 if !self.options.string.allow_unknown {
2156 // Sited on the escape CHARACTER with a
2157 // one-character span, as TypeScript
2158 // (`pnt.sI = sI; pnt.cI = cI; lex.bad(
2159 // S.unexpected, sI, sI + 1)`) and Go do;
2160 // the other escape errors sit on the
2161 // backslash and span the construct.
2162 let err = TabnasError::new(
2163 "unexpected",
2164 other.to_string(),
2165 "",
2166 esc_point.site.pos,
2167 esc_point.site.ri,
2168 esc_point.site.ci,
2169 );
2170 self.err = Some(err.clone());
2171 return Err(Box::new(err));
2172 }
2173 self.flush_surrogate(&mut pending_high_surrogate, &mut out_str);
2174 out_str.push(other);
2175 }
2176 }
2177 } else {
2178 let err = TabnasError::new(
2179 "unterminated_string",
2180 raw_src,
2181 "",
2182 pnt.site.pos,
2183 pnt.site.ri,
2184 pnt.site.ci,
2185 );
2186 self.err = Some(err.clone());
2187 return Err(Box::new(err));
2188 }
2189 } else {
2190 self.flush_surrogate(&mut pending_high_surrogate, &mut out_str);
2191 raw_src.push(self.advance_body().expect("peeked character must advance"));
2192 out_str.push(c);
2193 }
2194 }
2195
2196 let err = TabnasError::new(
2197 "unterminated_string",
2198 raw_src,
2199 "",
2200 pnt.site.pos,
2201 pnt.site.ri,
2202 pnt.site.ci,
2203 );
2204 self.err = Some(err.clone());
2205 Err(Box::new(err))
2206 }
2207
2208 fn flush_surrogate(&self, pending: &mut Option<u16>, out: &mut String) {
2209 if pending.take().is_some() {
2210 out.push('\u{FFFD}');
2211 }
2212 }
2213
2214 /// Emit one decoded Unicode escape while pairing UTF-16 surrogate code
2215 /// units across both `\\uXXXX` and `\\u{...}` spellings.
2216 fn emit_unicode_escape(&self, cp: u32, pending: &mut Option<u16>, out: &mut String) {
2217 if (0xD800..=0xDBFF).contains(&cp) {
2218 self.flush_surrogate(pending, out);
2219 *pending = Some(cp as u16);
2220 } else if (0xDC00..=0xDFFF).contains(&cp) {
2221 if let Some(high) = pending.take() {
2222 let scalar = 0x10000 + (((u32::from(high)) - 0xD800) << 10) + (cp - 0xDC00);
2223 out.push(char::from_u32(scalar).expect("paired surrogates form a Unicode scalar"));
2224 } else {
2225 out.push('\u{FFFD}');
2226 }
2227 } else {
2228 self.flush_surrogate(pending, out);
2229 out.push(char::from_u32(cp).expect("validated escape is a Unicode scalar"));
2230 }
2231 }
2232}
2233
2234// ---------------------------------------------------------------------------
2235// Base-prefixed integer literals.
2236//
2237// A `0x`, `0o` or `0b` literal is read as an EXACT integer and rounded to
2238// a double ONCE. The obvious fold -- `value = value * radix + digit` in
2239// `f64` -- rounds at every digit, and past the 53-bit exact integer range
2240// those roundings accumulate: `0Xa6f2f78f4f9bf44` came out as
2241// `43a4de5ef1e9f37e` where canonical TypeScript and the Go port both
2242// answer `43a4de5ef1e9f37f`, one unit in the last place low. That is
2243// silently altered data, not a formatting difference.
2244//
2245// TypeScript coerces the literal with unary `+`, whose StringNumericValue
2246// is the exact mathematical value of the digits rounded once, half to
2247// even; Go reads it through `big.Int` and `big.Float.Float64()`, which is
2248// the same rule. These reproduce it. Only the VALUE is affected: which
2249// literals are accepted, and the token they become, are settled by
2250// `match_number` before any of this runs.
2251//
2252// The decimal path needs none of it -- `str::parse::<f64>` is correctly
2253// rounded for a digit string of any length.
2254// ---------------------------------------------------------------------------
2255
2256/// `2^k` for a non-negative `k`, exactly, saturating to infinity above the
2257/// double range. A repeated multiply would round on the way up.
2258fn pow2(k: i64) -> f64 {
2259 debug_assert!(k >= 0, "only non-negative exponents arise here");
2260 if k > 1023 {
2261 f64::INFINITY
2262 } else {
2263 f64::from_bits(((k + 1023) as u64) << 52)
2264 }
2265}
2266
2267/// Folds the digits of a base-prefixed literal into the NEAREST double,
2268/// rounding half to even, without holding the digits.
2269///
2270/// `bits` is the width of one digit, so the base is a power of two: 1 for
2271/// binary, 3 for octal, 4 for hexadecimal. Those are the only bases
2272/// `match_number` reads, which is what lets a `u128` head plus a sticky
2273/// bit stand in for arbitrary-precision arithmetic.
2274///
2275/// Only three things about a literal can change the answer: the top
2276/// `128 / bits` significant digits, how many digits follow them, and
2277/// whether any of those is non-zero. This keeps exactly those, so the
2278/// space it costs does not grow with the literal, however long an
2279/// untrusted document makes one.
2280struct DigitFold {
2281 bits: u32,
2282 /// The significant digits packed so far, at most `head_len` of them.
2283 head: u128,
2284 /// How many significant digits have been pushed, head and tail alike.
2285 len: usize,
2286 /// Whether any digit past the head was non-zero.
2287 sticky: bool,
2288 /// Whether a non-zero digit has been seen. Leading zeros carry no
2289 /// value, and dropping them is what makes the head wider than the 54
2290 /// significant bits the rounding needs.
2291 started: bool,
2292}
2293
2294impl DigitFold {
2295 fn new(bits: u32) -> Self {
2296 debug_assert!(
2297 (1..=4).contains(&bits),
2298 "only the power-of-two bases the lexer reads"
2299 );
2300 DigitFold {
2301 bits,
2302 head: 0,
2303 len: 0,
2304 sticky: false,
2305 started: false,
2306 }
2307 }
2308
2309 /// A `u128` holds exactly this many digits of the base.
2310 fn head_len(&self) -> usize {
2311 (128 / self.bits) as usize
2312 }
2313
2314 fn push(&mut self, digit: u32) {
2315 if !self.started {
2316 if 0 == digit {
2317 return;
2318 }
2319 self.started = true;
2320 }
2321 if self.len < self.head_len() {
2322 self.head = (self.head << self.bits) | u128::from(digit);
2323 } else if 0 != digit {
2324 self.sticky = true;
2325 }
2326 self.len += 1;
2327 }
2328
2329 fn finish(&self) -> f64 {
2330 if !self.started {
2331 return 0.0;
2332 }
2333 let head_len = self.head_len();
2334 if self.len <= head_len {
2335 // A `u128` to `f64` cast rounds to nearest, ties to even, which
2336 // is the rule the canonical runtime follows.
2337 return self.head as f64;
2338 }
2339
2340 // Longer than a u128: the head holds the top `head_len` digits and
2341 // `sticky` remembers whether anything below them was set. Those two
2342 // are all the rounding can depend on. The leading digit is
2343 // non-zero, so the head is at least 121 bits wide in every base
2344 // here and `shift` is comfortably positive.
2345 let dropped = i64::from(self.bits) * (self.len - head_len) as i64;
2346 let shift = 128 - self.head.leading_zeros() - 53;
2347
2348 let mut mantissa = (self.head >> shift) as u64;
2349 let half = (self.head >> (shift - 1)) & 1 == 1;
2350 let sticky = self.head & ((1u128 << (shift - 1)) - 1) != 0 || self.sticky;
2351 if half && (sticky || mantissa & 1 == 1) {
2352 // At most 2^53, which is still an exact double.
2353 mantissa += 1;
2354 }
2355 // The mantissa carries at most 53 significant bits, so the scaling
2356 // is exact inside the double range and overflows to infinity
2357 // outside it.
2358 mantissa as f64 * pow2(dropped + i64::from(shift))
2359 }
2360}