Skip to main content

polydat_grammar/
lexer.rs

1// Copyright 2024-2026 Jonathan Shook
2// SPDX-License-Identifier: Apache-2.0
3
4//! Lexer for the Polydat DSL.
5//!
6//! Tokenizes `.polydat` source text into a flat token stream. The grammar
7//! is line-oriented but the lexer doesn't enforce line structure —
8//! that's the parser's job.
9
10/// A source location for error reporting.
11#[derive(Debug, Clone, Copy, PartialEq, Eq)]
12pub struct Span {
13    /// The line, from 1.
14    pub line: usize,
15    /// The column, from 1.
16    pub col: usize,
17}
18
19/// A token with its source location.
20#[derive(Debug, Clone, PartialEq)]
21pub struct Token {
22    /// What the token is.
23    pub kind: TokenKind,
24    /// Where it appears.
25    pub span: Span,
26}
27
28#[derive(Debug, Clone, PartialEq)]
29/// The kinds of token the lexer produces.
30pub enum TokenKind {
31    /// A bare identifier: `cycle`, `hash`, `temp_lut`
32    Ident(String),
33    /// `const` keyword — declares an effectively-const binding.
34    /// Replaces the former `final` / `init` distinction; both
35    /// surfaces collapsed into one. The value is materialized at
36    /// the earliest opportunity (compile-time fold if possible,
37    /// scope-init pull otherwise) and is then immutable for the
38    /// scope's lifetime. Independent of provenance: a `const`
39    /// binding's RHS may reference workload params, iter-vars
40    /// bound by an enclosing comprehension, or other in-scope
41    /// names — whatever the Polydat compiler can resolve.
42    Const,
43    /// `input` keyword — declares one per-cycle kernel input slot.
44    /// Surface: `input <name>[: <type>]` (single) or
45    /// `input (<name>[: <type>], ...)` (tuple sugar). Mirrors the
46    /// module-signature param-list shape.
47    Input,
48    /// `extern` keyword
49    Extern,
50    /// `shared` keyword
51    Shared,
52    /// `volatile` keyword. Wire-coloring modifier excluding the
53    /// binding's value from `hash_const`.
54    Volatile,
55    /// `cursor` keyword
56    Cursor,
57    /// `over` keyword — SRD 71 partition-source binding on
58    /// cursor declarations: `cursor q = range(0, N) over p`.
59    /// Soft keyword: only recognised at statement level after
60    /// a cursor decl's constructor expression; in expression
61    /// position it's a plain identifier.
62    Over,
63    /// `pragma` keyword (module-level directive opening, SRD 15
64    /// §"Module-Level Pragmas"). Followed by an `Ident` naming the
65    /// pragma. Distinct from line comments so the parser sees
66    /// pragmas as first-class statements rather than scraping them
67    /// out of `// @pragma:` text.
68    Pragma,
69    /// `for` followed by its raw comprehension text (SRD 113). The
70    /// lexer captures everything after the keyword up to a `{` at
71    /// nesting depth zero or the end of a line at nesting depth zero
72    /// (a bracketed union runs across lines), whichever comes
73    /// first, so the comprehension grammar stays owned by the
74    /// comprehension parser rather than being re-tokenized here.
75    For(String),
76    /// `tile` keyword (SRD 114). The body after `:=` is captured raw
77    /// into a [`TokenKind::TileBody`] when it is a block or heredoc.
78    Tile,
79    /// A raw tile body and how it was written.
80    TileBody(String, crate::ast::TileBodyKind),
81    /// `.` (field access: `base.ordinal`)
82    Dot,
83    /// Integer literal: `1000`, `0xFF`
84    IntLit(u64),
85    /// Float literal: `72.0`, `3.14`
86    FloatLit(f64),
87    /// String literal (contents only, no quotes): `"hello {name}"`
88    StringLit(String),
89    /// `:=` (binding operator, used by every binding shape:
90    /// cycle bindings, `const`, `shared`, `volatile`)
91    ColonEq,
92    /// `=` — the extern default (`extern name: type = default`)
93    /// and the cursor declaration (`cursor name = expr`); bindings
94    /// use `:=`.
95    Eq,
96    /// `(`
97    LParen,
98    /// `)`
99    RParen,
100    /// `[`
101    LBracket,
102    /// `]`
103    RBracket,
104    /// `{`
105    LBrace,
106    /// `}`
107    RBrace,
108    /// `,`
109    Comma,
110    /// `:`
111    Colon,
112    /// `->`
113    Arrow,
114    /// `+`
115    Plus,
116    /// `-` (both binary subtract and unary negate)
117    Minus,
118    /// `*`
119    Star,
120    /// `/`
121    Slash,
122    /// `%`
123    Percent,
124    /// `^` (bitwise XOR)
125    Caret,
126    /// `**` (power)
127    StarStar,
128    /// `<<` (shift left)
129    ShiftLeft,
130    /// `>>` (shift right)
131    ShiftRight,
132    /// `&` (bitwise AND)
133    Ampersand,
134    /// `&&` (logical AND — SRD-84 Part 1)
135    AmpAmp,
136    /// `|` (bitwise OR)
137    Pipe,
138    /// `||` (logical OR — SRD-84 Part 1)
139    PipePipe,
140    /// `!` (unary bitwise NOT)
141    Bang,
142    /// `<` (less-than comparison)
143    Lt,
144    /// `>` (greater-than comparison)
145    Gt,
146    /// `==` (equal-to comparison)
147    EqEq,
148    /// `!=` (not-equal-to comparison)
149    BangEq,
150    /// `<=` (less-than-or-equal comparison)
151    LtEq,
152    /// `>=` (greater-than-or-equal comparison)
153    GtEq,
154    /// End of input
155    Eof,
156}
157
158/// Lex source text into tokens.
159pub fn lex(source: &str) -> Result<Vec<Token>, String> {
160    let mut tokens = Vec::new();
161    let chars: Vec<char> = source.chars().collect();
162    let mut pos = 0;
163    let mut line = 1;
164    let mut col = 1;
165    // Set by the `tile` keyword; the next `:=` captures a raw body.
166    let mut tile_pending = false;
167
168    while pos < chars.len() {
169        let c = chars[pos];
170
171        // Skip whitespace (but track line/col)
172        if c == '\n' {
173            line += 1;
174            col = 1;
175            pos += 1;
176            // A tile header and its `:=` sit on one line; a header that
177            // ends without one is the parser's error to report, not a
178            // reason to read the next binding's value as a tile body.
179            tile_pending = false;
180            continue;
181        }
182        if c.is_ascii_whitespace() {
183            col += 1;
184            pos += 1;
185            continue;
186        }
187
188        // Skip hash comments (# to end of line, YAML-style)
189        if c == '#' {
190            while pos < chars.len() && chars[pos] != '\n' {
191                pos += 1;
192            }
193            continue;
194        }
195
196        // Skip comments (// and /* */)
197        if c == '/' && pos + 1 < chars.len() {
198            if chars[pos + 1] == '/' {
199                // Line comment (// or ///) — skip to end of line
200                while pos < chars.len() && chars[pos] != '\n' {
201                    pos += 1;
202                }
203                continue;
204            }
205            if chars[pos + 1] == '*' {
206                // Block comment /* ... */ — skip to closing */
207                pos += 2;
208                col += 2;
209                while pos + 1 < chars.len() {
210                    if chars[pos] == '\n' {
211                        line += 1;
212                        col = 1;
213                    }
214                    if chars[pos] == '*' && chars[pos + 1] == '/' {
215                        pos += 2;
216                        col += 2;
217                        break;
218                    }
219                    pos += 1;
220                    col += 1;
221                }
222                continue;
223            }
224        }
225
226        // Arithmetic operator `/` (not a comment start)
227        if c == '/' {
228            let span = Span { line, col };
229            tokens.push(Token {
230                kind: TokenKind::Slash,
231                span,
232            });
233            pos += 1;
234            col += 1;
235            continue;
236        }
237
238        let span = Span { line, col };
239
240        // A tile binds a wire, so its body follows `:=` like any other
241        // binding's value. A block that follows the header directly is
242        // left to the parser, which reports the missing `:=`.
243
244        // A heredoc anywhere else is a string literal: `<<<` ... `>>>`
245        // with one newline trimmed from each end. This is how a
246        // template handed in by a host arrives as an argument, as in
247        // `doc := polytile("json", <<< ... >>>)`.
248        if !tile_pending
249            && chars[pos..].starts_with(&['<', '<', '<'])
250            && let Some((body, _, consumed, newlines, end_col)) =
251                capture_tile_body(&chars, pos, col)?
252        {
253            tokens.push(Token {
254                kind: TokenKind::StringLit(body),
255                span,
256            });
257            pos += consumed;
258            if newlines > 0 {
259                line += newlines;
260                col = end_col;
261            } else {
262                col += consumed;
263            }
264            continue;
265        }
266
267        // Two-character operators
268        if c == ':' && pos + 1 < chars.len() && chars[pos + 1] == '=' {
269            tokens.push(Token {
270                kind: TokenKind::ColonEq,
271                span,
272            });
273            pos += 2;
274            col += 2;
275            if tile_pending {
276                tile_pending = false;
277                if let Some((body, kind, consumed, newlines, end_col)) =
278                    capture_tile_body(&chars, pos, col)?
279                {
280                    tokens.push(Token {
281                        kind: TokenKind::TileBody(body, kind),
282                        span: Span { line, col },
283                    });
284                    pos += consumed;
285                    if newlines > 0 {
286                        line += newlines;
287                        col = end_col;
288                    } else {
289                        col += consumed;
290                    }
291                }
292            }
293            continue;
294        }
295        if c == '-' && pos + 1 < chars.len() && chars[pos + 1] == '>' {
296            tokens.push(Token {
297                kind: TokenKind::Arrow,
298                span,
299            });
300            pos += 2;
301            col += 2;
302            continue;
303        }
304
305        // Number literal (integer or float).
306        // The `-` character is always lexed as `Minus`; negative values
307        // are handled by the parser via `UnaryNeg`.
308        if c.is_ascii_digit() {
309            let start = pos;
310
311            // Check for hex: 0x...
312            if pos + 1 < chars.len()
313                && chars[pos] == '0'
314                && (chars[pos + 1] == 'x' || chars[pos + 1] == 'X')
315            {
316                pos += 2;
317                col += 2;
318                let hex_start = pos;
319                while pos < chars.len() && chars[pos].is_ascii_hexdigit() {
320                    pos += 1;
321                    col += 1;
322                }
323                let hex: String = chars[hex_start..pos].iter().collect();
324                let val = u64::from_str_radix(&hex, 16).map_err(|e| {
325                    format!(
326                        "invalid hex literal at line {}, col {}: {e}",
327                        span.line, span.col
328                    )
329                })?;
330                tokens.push(Token {
331                    kind: TokenKind::IntLit(val),
332                    span,
333                });
334                continue;
335            }
336
337            while pos < chars.len() && chars[pos].is_ascii_digit() {
338                pos += 1;
339                col += 1;
340            }
341
342            // Check for float
343            let mut is_float = false;
344            if pos < chars.len()
345                && chars[pos] == '.'
346                && pos + 1 < chars.len()
347                && chars[pos + 1].is_ascii_digit()
348            {
349                is_float = true;
350                pos += 1;
351                col += 1;
352                while pos < chars.len() && chars[pos].is_ascii_digit() {
353                    pos += 1;
354                    col += 1;
355                }
356                // Scientific notation
357                if pos < chars.len() && (chars[pos] == 'e' || chars[pos] == 'E') {
358                    pos += 1;
359                    col += 1;
360                    if pos < chars.len() && (chars[pos] == '+' || chars[pos] == '-') {
361                        pos += 1;
362                        col += 1;
363                    }
364                    while pos < chars.len() && chars[pos].is_ascii_digit() {
365                        pos += 1;
366                        col += 1;
367                    }
368                }
369            } else if pos < chars.len() && (chars[pos] == 'e' || chars[pos] == 'E') {
370                // Scientific notation without a decimal point: 1e10, 2E-3
371                is_float = true;
372                pos += 1;
373                col += 1;
374                if pos < chars.len() && (chars[pos] == '+' || chars[pos] == '-') {
375                    pos += 1;
376                    col += 1;
377                }
378                while pos < chars.len() && chars[pos].is_ascii_digit() {
379                    pos += 1;
380                    col += 1;
381                }
382            }
383
384            // SRD-18c Layer 6 / SRD-18e Push 4: SI suffix
385            // literals. Suffix attaches to the just-lexed
386            // numeric literal when the next 1-2 chars form
387            // a known suffix AND the char after the suffix
388            // isn't an identifier-continuation character
389            // (so `Kilometers` stays a valid identifier
390            // following an unrelated `1`).
391            let suffix_consumed = match peek_si_suffix(&chars, pos) {
392                Some((multiplier, len, is_subunit)) => {
393                    pos += len;
394                    col += len;
395                    Some((multiplier, is_subunit))
396                }
397                None => None,
398            };
399
400            if is_float {
401                let num: String = chars[start..pos - suffix_len_consumed(suffix_consumed)]
402                    .iter()
403                    .collect();
404                let mut val: f64 = num.parse().map_err(|e| {
405                    format!("invalid float at line {}, col {}: {e}", span.line, span.col)
406                })?;
407                if let Some((mult, is_sub)) = suffix_consumed {
408                    if is_sub {
409                        val /= mult as f64;
410                    } else {
411                        val *= mult as f64;
412                    }
413                }
414                // SI-promoted floats become U64 when the
415                // result is exactly integral (`1.5G` →
416                // 1_500_000_000 → U64) — matches SRD-18c
417                // §"Layer 6" type-resolution rule.
418                if suffix_consumed.is_some()
419                    && val.fract() == 0.0
420                    && val >= 0.0
421                    && val <= u64::MAX as f64
422                {
423                    tokens.push(Token {
424                        kind: TokenKind::IntLit(val as u64),
425                        span,
426                    });
427                } else {
428                    tokens.push(Token {
429                        kind: TokenKind::FloatLit(val),
430                        span,
431                    });
432                }
433            } else {
434                // Skip trailing underscore separators in int literals
435                let num_end = pos - suffix_len_consumed(suffix_consumed);
436                let num: String = chars[start..num_end]
437                    .iter()
438                    .filter(|c| **c != '_')
439                    .collect();
440                let val: u64 = num.parse().map_err(|e| {
441                    format!(
442                        "invalid integer at line {}, col {}: {e}",
443                        span.line, span.col
444                    )
445                })?;
446                match suffix_consumed {
447                    Some((mult, true)) => {
448                        // Sub-unit suffix on integer →
449                        // float (`5m` = 0.005).
450                        let f = val as f64 / mult as f64;
451                        tokens.push(Token {
452                            kind: TokenKind::FloatLit(f),
453                            span,
454                        });
455                    }
456                    Some((mult, false)) => {
457                        let v = val.checked_mul(mult).ok_or_else(|| {
458                            format!(
459                                "integer literal with SI suffix overflows u64 at line {}, col {}",
460                                span.line, span.col
461                            )
462                        })?;
463                        tokens.push(Token {
464                            kind: TokenKind::IntLit(v),
465                            span,
466                        });
467                    }
468                    None => {
469                        tokens.push(Token {
470                            kind: TokenKind::IntLit(val),
471                            span,
472                        });
473                    }
474                }
475            }
476            continue;
477        }
478
479        // Single-character tokens
480        match c {
481            '(' => {
482                tokens.push(Token {
483                    kind: TokenKind::LParen,
484                    span,
485                });
486                pos += 1;
487                col += 1;
488                continue;
489            }
490            ')' => {
491                tokens.push(Token {
492                    kind: TokenKind::RParen,
493                    span,
494                });
495                pos += 1;
496                col += 1;
497                continue;
498            }
499            '[' => {
500                tokens.push(Token {
501                    kind: TokenKind::LBracket,
502                    span,
503                });
504                pos += 1;
505                col += 1;
506                continue;
507            }
508            ']' => {
509                tokens.push(Token {
510                    kind: TokenKind::RBracket,
511                    span,
512                });
513                pos += 1;
514                col += 1;
515                continue;
516            }
517            '{' => {
518                tokens.push(Token {
519                    kind: TokenKind::LBrace,
520                    span,
521                });
522                pos += 1;
523                col += 1;
524                continue;
525            }
526            '}' => {
527                tokens.push(Token {
528                    kind: TokenKind::RBrace,
529                    span,
530                });
531                pos += 1;
532                col += 1;
533                continue;
534            }
535            ',' => {
536                tokens.push(Token {
537                    kind: TokenKind::Comma,
538                    span,
539                });
540                pos += 1;
541                col += 1;
542                continue;
543            }
544            '=' => {
545                if pos + 1 < chars.len() && chars[pos + 1] == '=' {
546                    tokens.push(Token {
547                        kind: TokenKind::EqEq,
548                        span,
549                    });
550                    pos += 2;
551                    col += 2;
552                } else {
553                    tokens.push(Token {
554                        kind: TokenKind::Eq,
555                        span,
556                    });
557                    pos += 1;
558                    col += 1;
559                }
560                continue;
561            }
562            ':' => {
563                tokens.push(Token {
564                    kind: TokenKind::Colon,
565                    span,
566                });
567                pos += 1;
568                col += 1;
569                continue;
570            }
571            '+' => {
572                tokens.push(Token {
573                    kind: TokenKind::Plus,
574                    span,
575                });
576                pos += 1;
577                col += 1;
578                continue;
579            }
580            '-' => {
581                tokens.push(Token {
582                    kind: TokenKind::Minus,
583                    span,
584                });
585                pos += 1;
586                col += 1;
587                continue;
588            }
589            '*' => {
590                if pos + 1 < chars.len() && chars[pos + 1] == '*' {
591                    tokens.push(Token {
592                        kind: TokenKind::StarStar,
593                        span,
594                    });
595                    pos += 2;
596                    col += 2;
597                } else {
598                    tokens.push(Token {
599                        kind: TokenKind::Star,
600                        span,
601                    });
602                    pos += 1;
603                    col += 1;
604                }
605                continue;
606            }
607            '%' => {
608                tokens.push(Token {
609                    kind: TokenKind::Percent,
610                    span,
611                });
612                pos += 1;
613                col += 1;
614                continue;
615            }
616            '^' => {
617                tokens.push(Token {
618                    kind: TokenKind::Caret,
619                    span,
620                });
621                pos += 1;
622                col += 1;
623                continue;
624            }
625            '<' => {
626                if pos + 1 < chars.len() && chars[pos + 1] == '<' {
627                    tokens.push(Token {
628                        kind: TokenKind::ShiftLeft,
629                        span,
630                    });
631                    pos += 2;
632                    col += 2;
633                } else if pos + 1 < chars.len() && chars[pos + 1] == '=' {
634                    tokens.push(Token {
635                        kind: TokenKind::LtEq,
636                        span,
637                    });
638                    pos += 2;
639                    col += 2;
640                } else {
641                    tokens.push(Token {
642                        kind: TokenKind::Lt,
643                        span,
644                    });
645                    pos += 1;
646                    col += 1;
647                }
648                continue;
649            }
650            '>' => {
651                if pos + 1 < chars.len() && chars[pos + 1] == '>' {
652                    tokens.push(Token {
653                        kind: TokenKind::ShiftRight,
654                        span,
655                    });
656                    pos += 2;
657                    col += 2;
658                } else if pos + 1 < chars.len() && chars[pos + 1] == '=' {
659                    tokens.push(Token {
660                        kind: TokenKind::GtEq,
661                        span,
662                    });
663                    pos += 2;
664                    col += 2;
665                } else {
666                    tokens.push(Token {
667                        kind: TokenKind::Gt,
668                        span,
669                    });
670                    pos += 1;
671                    col += 1;
672                }
673                continue;
674            }
675            '.' => {
676                tokens.push(Token {
677                    kind: TokenKind::Dot,
678                    span,
679                });
680                pos += 1;
681                col += 1;
682                continue;
683            }
684            '&' => {
685                if pos + 1 < chars.len() && chars[pos + 1] == '&' {
686                    tokens.push(Token {
687                        kind: TokenKind::AmpAmp,
688                        span,
689                    });
690                    pos += 2;
691                    col += 2;
692                    continue;
693                }
694                tokens.push(Token {
695                    kind: TokenKind::Ampersand,
696                    span,
697                });
698                pos += 1;
699                col += 1;
700                continue;
701            }
702            '|' => {
703                if pos + 1 < chars.len() && chars[pos + 1] == '|' {
704                    tokens.push(Token {
705                        kind: TokenKind::PipePipe,
706                        span,
707                    });
708                    pos += 2;
709                    col += 2;
710                    continue;
711                }
712                tokens.push(Token {
713                    kind: TokenKind::Pipe,
714                    span,
715                });
716                pos += 1;
717                col += 1;
718                continue;
719            }
720            '!' => {
721                if pos + 1 < chars.len() && chars[pos + 1] == '=' {
722                    tokens.push(Token {
723                        kind: TokenKind::BangEq,
724                        span,
725                    });
726                    pos += 2;
727                    col += 2;
728                } else {
729                    tokens.push(Token {
730                        kind: TokenKind::Bang,
731                        span,
732                    });
733                    pos += 1;
734                    col += 1;
735                }
736                continue;
737            }
738            _ => {}
739        }
740
741        // String literal (double-quoted or single-quoted)
742        if c == '"' || c == '\'' {
743            let quote = c;
744            pos += 1;
745            col += 1;
746            let mut s = String::new();
747            while pos < chars.len() && chars[pos] != quote {
748                if chars[pos] == '\\' && pos + 1 < chars.len() {
749                    pos += 1;
750                    col += 1;
751                    match chars[pos] {
752                        'n' => s.push('\n'),
753                        't' => s.push('\t'),
754                        '\\' => s.push('\\'),
755                        c if c == quote => s.push(c),
756                        other => {
757                            s.push('\\');
758                            s.push(other);
759                        }
760                    }
761                } else {
762                    s.push(chars[pos]);
763                }
764                pos += 1;
765                col += 1;
766            }
767            if pos < chars.len() {
768                pos += 1; // skip closing quote
769                col += 1;
770            } else {
771                return Err(format!(
772                    "unterminated string at line {}, col {}",
773                    span.line, span.col
774                ));
775            }
776            tokens.push(Token {
777                kind: TokenKind::StringLit(s),
778                span,
779            });
780            continue;
781        }
782
783        // Identifier or keyword
784        if c.is_ascii_alphabetic() || c == '_' {
785            let start = pos;
786            while pos < chars.len() && (chars[pos].is_ascii_alphanumeric() || chars[pos] == '_') {
787                pos += 1;
788                col += 1;
789            }
790            let word: String = chars[start..pos].iter().collect();
791            let kind = match word.as_str() {
792                "const" => TokenKind::Const,
793                "input" => TokenKind::Input,
794                "extern" => TokenKind::Extern,
795                "shared" => TokenKind::Shared,
796                "volatile" => TokenKind::Volatile,
797                "cursor" => TokenKind::Cursor,
798                "over" => TokenKind::Over,
799                "pragma" => TokenKind::Pragma,
800                "tile" => {
801                    tile_pending = true;
802                    TokenKind::Tile
803                }
804                "for" => {
805                    let (text, consumed) = capture_for_text(&chars, pos);
806                    // A bracketed union runs across lines: the position keeps
807                    // counting them.
808                    for c in &chars[pos..pos + consumed] {
809                        if *c == '\n' {
810                            line += 1;
811                            col = 1;
812                        } else {
813                            col += 1;
814                        }
815                    }
816                    pos += consumed;
817                    tokens.push(Token {
818                        kind: TokenKind::For(text),
819                        span,
820                    });
821                    continue;
822                }
823                _ => TokenKind::Ident(word),
824            };
825            tokens.push(Token { kind, span });
826            continue;
827        }
828
829        return Err(format!(
830            "unexpected character '{}' at line {}, col {}",
831            c, line, col
832        ));
833    }
834
835    tokens.push(Token {
836        kind: TokenKind::Eof,
837        span: Span { line, col },
838    });
839    Ok(tokens)
840}
841
842/// Capture a raw tile body after `tile ... :=` (SRD 114 §2.1).
843///
844/// Skips whitespace, then: a `{` or `[` starts a balanced, string-aware
845/// block that includes its brackets; `<<<` starts a heredoc ending at
846/// `>>>`, with one leading and one trailing newline trimmed. Anything
847/// else, such as a string literal, is left to the main loop and `None`
848/// is returned. On success returns the body, its kind, the chars
849/// consumed from `start`, the newlines crossed, and the column after
850/// the body.
851type TileBodyCapture = (String, crate::ast::TileBodyKind, usize, usize, usize);
852
853fn capture_tile_body(
854    chars: &[char],
855    start: usize,
856    start_col: usize,
857) -> Result<Option<TileBodyCapture>, String> {
858    use crate::ast::TileBodyKind;
859    let mut pos = start;
860    let mut col = start_col;
861    let mut newlines = 0;
862    while pos < chars.len() && chars[pos].is_whitespace() {
863        if chars[pos] == '\n' {
864            newlines += 1;
865            col = 1;
866        } else {
867            col += 1;
868        }
869        pos += 1;
870    }
871    if pos >= chars.len() {
872        return Ok(None);
873    }
874    let body_start = pos;
875    let kind;
876    match chars[pos] {
877        '{' | '[' => {
878            kind = TileBodyKind::Block;
879            let mut depth = 0i32;
880            let mut quote: Option<char> = None;
881            loop {
882                if pos >= chars.len() {
883                    return Err("unterminated tile body: block never closed".to_string());
884                }
885                let c = chars[pos];
886                if c == '\n' {
887                    newlines += 1;
888                    col = 0;
889                }
890                if let Some(q) = quote {
891                    if c == '\\' && pos + 1 < chars.len() {
892                        pos += 2;
893                        col += 2;
894                        continue;
895                    }
896                    if c == q {
897                        quote = None;
898                    }
899                } else {
900                    match c {
901                        '"' => quote = Some(c),
902                        '{' | '[' => depth += 1,
903                        '}' | ']' => depth -= 1,
904                        _ => {}
905                    }
906                }
907                pos += 1;
908                col += 1;
909                if depth == 0 && quote.is_none() {
910                    break;
911                }
912            }
913        }
914        '<' if chars[pos..].starts_with(&['<', '<', '<']) => {
915            kind = TileBodyKind::Heredoc;
916            pos += 3;
917            col += 3;
918            let text_start = pos;
919            loop {
920                if pos + 2 >= chars.len() {
921                    return Err(
922                        "unterminated tile body: heredoc never closed with `>>>`".to_string()
923                    );
924                }
925                if chars[pos..].starts_with(&['>', '>', '>']) {
926                    break;
927                }
928                if chars[pos] == '\n' {
929                    newlines += 1;
930                    col = 0;
931                }
932                pos += 1;
933                col += 1;
934            }
935            let mut text: String = chars[text_start..pos].iter().collect();
936            if let Some(t) = text.strip_prefix('\n') {
937                text = t.to_string();
938            }
939            if let Some(t) = text.strip_suffix('\n') {
940                text = t.to_string();
941            }
942            pos += 3;
943            col += 3;
944            return Ok(Some((text, kind, pos - start, newlines, col)));
945        }
946        _ => return Ok(None),
947    }
948    let text: String = chars[body_start..pos].iter().collect();
949    Ok(Some((
950        dedent_block(&text),
951        kind,
952        pos - start,
953        newlines,
954        col,
955    )))
956}
957
958/// A block body keeps the author's layout but not the indentation of
959/// the statement it sits in: the common leading whitespace of the lines
960/// after the first is removed, so a tile declared inside a `for` body
961/// renders the same bytes as one at top level.
962fn dedent_block(text: &str) -> String {
963    let mut lines = text.split('\n');
964    let first = lines.next().unwrap_or("");
965    let rest: Vec<&str> = lines.collect();
966    let indent = rest
967        .iter()
968        .filter(|l| !l.trim().is_empty())
969        .map(|l| l.chars().take_while(|c| *c == ' ' || *c == '\t').count())
970        .min()
971        .unwrap_or(0);
972    if indent == 0 {
973        return text.to_string();
974    }
975    let mut out = String::with_capacity(text.len());
976    out.push_str(first);
977    for line in rest {
978        out.push('\n');
979        let skip = line
980            .chars()
981            .take_while(|c| *c == ' ' || *c == '\t')
982            .count()
983            .min(indent);
984        out.push_str(&line.chars().skip(skip).collect::<String>());
985    }
986    out
987}
988
989/// Capture the raw comprehension text after a `for` keyword.
990///
991/// Scans from `start` until a `{` at paren/bracket depth zero or a
992/// newline, stopping early at a `//` or `#` comment. String literals
993/// are skipped whole so braces and commas inside them do not count.
994/// Returns the trimmed text and the number of chars consumed; the `{`
995/// and the newline are left for the main loop.
996fn capture_for_text(chars: &[char], start: usize) -> (String, usize) {
997    let mut pos = start;
998    let mut depth = 0usize;
999    let mut text = String::new();
1000    while pos < chars.len() {
1001        let c = chars[pos];
1002        match c {
1003            '\n' if depth == 0 => break,
1004            '\n' => {
1005                // Inside brackets the capture runs on: a union's members
1006                // sit one per line (comprehension_forms.md §8.2).
1007                text.push(' ');
1008                pos += 1;
1009                continue;
1010            }
1011            '{' if depth == 0 => {
1012                // `{name}` inside a `where` predicate is a coordinate
1013                // reference, not the start of the block. A block brace
1014                // is never immediately followed by an identifier and a
1015                // closing brace.
1016                match placeholder_len(chars, pos) {
1017                    Some(n) => {
1018                        text.extend(chars[pos..pos + n].iter());
1019                        pos += n;
1020                        continue;
1021                    }
1022                    None => break,
1023                }
1024            }
1025            '#' if depth == 0 => break,
1026            '/' if depth == 0 && pos + 1 < chars.len() && chars[pos + 1] == '/' => break,
1027            '#' => {
1028                skip_comment(chars, &mut pos);
1029                continue;
1030            }
1031            '/' if pos + 1 < chars.len() && chars[pos + 1] == '/' => {
1032                skip_comment(chars, &mut pos);
1033                continue;
1034            }
1035            '"' | '\'' => {
1036                let quote = c;
1037                text.push(c);
1038                pos += 1;
1039                while pos < chars.len() && chars[pos] != quote && chars[pos] != '\n' {
1040                    if chars[pos] == '\\' && pos + 1 < chars.len() {
1041                        text.push(chars[pos]);
1042                        pos += 1;
1043                    }
1044                    text.push(chars[pos]);
1045                    pos += 1;
1046                }
1047                if pos < chars.len() && chars[pos] == quote {
1048                    text.push(quote);
1049                    pos += 1;
1050                }
1051                continue;
1052            }
1053            '(' | '[' => depth += 1,
1054            ')' | ']' => depth = depth.saturating_sub(1),
1055            _ => {}
1056        }
1057        text.push(c);
1058        pos += 1;
1059    }
1060    (text.trim().to_string(), pos - start)
1061}
1062
1063/// Length of a `{identifier}` placeholder starting at `pos`, if the
1064/// text there is one.
1065fn placeholder_len(chars: &[char], pos: usize) -> Option<usize> {
1066    let mut i = pos + 1;
1067    let first = *chars.get(i)?;
1068    if !(first.is_ascii_alphabetic() || first == '_') {
1069        return None;
1070    }
1071    while i < chars.len() && (chars[i].is_ascii_alphanumeric() || chars[i] == '_') {
1072        i += 1;
1073    }
1074    (chars.get(i) == Some(&'}')).then_some(i + 1 - pos)
1075}
1076
1077/// SRD-18c Layer 6 / SRD-18e Push 4: peek for an SI suffix
1078/// at `pos`. Returns `Some((multiplier, len, is_subunit))`
1079/// on a match, `None` otherwise.
1080///
1081/// Two-char binary suffixes (`Ki`, `Mi`, `Gi`, `Ti`, `Pi`)
1082/// are checked first to avoid `K` greedy-eating the `K` of
1083/// `Ki`. Decimal suffixes (`K`, `M`, `G`, `T`, `P`) and sub-
1084/// unit suffixes (`m`, `u`, `n`) are length-1.
1085///
1086/// A match requires the char *after* the suffix to NOT be
1087/// an identifier-continuation character — so `1Kilometers`
1088/// stays as IntLit(1) + Ident("Kilometers"), not `1000000`.
1089fn peek_si_suffix(chars: &[char], pos: usize) -> Option<(u64, usize, bool)> {
1090    if pos >= chars.len() {
1091        return None;
1092    }
1093    // Try 2-char binary suffixes first.
1094    if pos + 1 < chars.len() && chars[pos + 1] == 'i' {
1095        let mult = match chars[pos] {
1096            'K' => 1u64 << 10,
1097            'M' => 1u64 << 20,
1098            'G' => 1u64 << 30,
1099            'T' => 1u64 << 40,
1100            'P' => 1u64 << 50,
1101            _ => 0,
1102        };
1103        if mult > 0 {
1104            // Check that the suffix isn't part of an identifier.
1105            let next = chars.get(pos + 2);
1106            if !next.is_some_and(|c| c.is_ascii_alphanumeric() || *c == '_') {
1107                return Some((mult, 2, false));
1108            }
1109        }
1110    }
1111    // 1-char decimal suffixes.
1112    let (mult, is_subunit) = match chars[pos] {
1113        'K' => (1_000u64, false),
1114        'M' => (1_000_000u64, false),
1115        'G' => (1_000_000_000u64, false),
1116        'T' => (1_000_000_000_000u64, false),
1117        'P' => (1_000_000_000_000_000u64, false),
1118        'm' => (1_000u64, true),         // 10⁻³
1119        'u' => (1_000_000u64, true),     // 10⁻⁶
1120        'n' => (1_000_000_000u64, true), // 10⁻⁹
1121        _ => return None,
1122    };
1123    let next = chars.get(pos + 1);
1124    if !next.is_some_and(|c| c.is_ascii_alphanumeric() || *c == '_') {
1125        Some((mult, 1, is_subunit))
1126    } else {
1127        None
1128    }
1129}
1130
1131/// Length consumed for the SI suffix, or 0 when no suffix
1132/// was applied. Used by the numeric-literal branch to
1133/// recover the digit-only end-position when slicing the
1134/// literal text out for parsing.
1135fn suffix_len_consumed(suffix: Option<(u64, bool)>) -> usize {
1136    match suffix {
1137        // We applied either a 1-char or 2-char suffix; the
1138        // caller only stored the multiplier and is_subunit
1139        // pair, so we re-derive: binary multipliers (powers
1140        // of 2 from 2^10 up) take 2 chars, others take 1.
1141        Some((m, _)) => {
1142            if m.is_power_of_two() && m >= (1 << 10) {
1143                2
1144            } else {
1145                1
1146            }
1147        }
1148        None => 0,
1149    }
1150}
1151
1152/// Advance `pos` past a line comment inside a bracketed `for` capture:
1153/// to the newline, which the caller then folds into the capture.
1154fn skip_comment(chars: &[char], pos: &mut usize) {
1155    while *pos < chars.len() && chars[*pos] != '\n' {
1156        *pos += 1;
1157    }
1158}
1159
1160#[cfg(test)]
1161mod tests {
1162    use super::*;
1163
1164    #[test]
1165    fn lex_cycle_binding() {
1166        let tokens = lex("seed := hash(cycle)").unwrap();
1167        assert!(matches!(tokens[0].kind, TokenKind::Ident(ref s) if s == "seed"));
1168        assert!(matches!(tokens[1].kind, TokenKind::ColonEq));
1169        assert!(matches!(tokens[2].kind, TokenKind::Ident(ref s) if s == "hash"));
1170        assert!(matches!(tokens[3].kind, TokenKind::LParen));
1171        assert!(matches!(tokens[4].kind, TokenKind::Ident(ref s) if s == "cycle"));
1172        assert!(matches!(tokens[5].kind, TokenKind::RParen));
1173    }
1174
1175    #[test]
1176    fn lex_const_binding() {
1177        let tokens = lex("const lut := dist_normal(72.0, 5.0)").unwrap();
1178        assert!(matches!(tokens[0].kind, TokenKind::Const));
1179        assert!(matches!(tokens[1].kind, TokenKind::Ident(ref s) if s == "lut"));
1180        assert!(matches!(tokens[2].kind, TokenKind::ColonEq));
1181        assert!(matches!(tokens[3].kind, TokenKind::Ident(ref s) if s == "dist_normal"));
1182        assert!(matches!(tokens[5].kind, TokenKind::FloatLit(v) if v == 72.0));
1183        assert!(matches!(tokens[7].kind, TokenKind::FloatLit(v) if v == 5.0));
1184    }
1185
1186    #[test]
1187    fn lex_input_keyword_tuple() {
1188        let tokens = lex("input (cycle: u64, thread: u64)").unwrap();
1189        assert!(matches!(tokens[0].kind, TokenKind::Input));
1190        assert!(matches!(tokens[1].kind, TokenKind::LParen));
1191        assert!(matches!(tokens[2].kind, TokenKind::Ident(ref s) if s == "cycle"));
1192    }
1193
1194    #[test]
1195    fn lex_destructuring() {
1196        let tokens = lex("(a, b, c) := mixed_radix(cycle, 100, 1000, 0)").unwrap();
1197        assert!(matches!(tokens[0].kind, TokenKind::LParen));
1198        assert!(matches!(tokens[1].kind, TokenKind::Ident(ref s) if s == "a"));
1199    }
1200
1201    #[test]
1202    fn lex_string_with_interpolation() {
1203        let tokens = lex(r#"id := "{code}-{seq}""#).unwrap();
1204        // id(0) :=(1) string(2)
1205        assert!(matches!(tokens[2].kind, TokenKind::StringLit(ref s) if s == "{code}-{seq}"));
1206    }
1207
1208    #[test]
1209    fn lex_named_args() {
1210        let tokens = lex("dist_normal(mean: 72.0, stddev: 5.0)").unwrap();
1211        assert!(matches!(tokens[2].kind, TokenKind::Ident(ref s) if s == "mean"));
1212        assert!(matches!(tokens[3].kind, TokenKind::Colon));
1213        assert!(matches!(tokens[4].kind, TokenKind::FloatLit(v) if v == 72.0));
1214    }
1215
1216    #[test]
1217    fn lex_array_literal() {
1218        let tokens = lex("[60.0, 20.0, 15.0, 5.0]").unwrap();
1219        assert!(matches!(tokens[0].kind, TokenKind::LBracket));
1220        assert!(matches!(tokens[1].kind, TokenKind::FloatLit(v) if v == 60.0));
1221        assert!(matches!(tokens[8].kind, TokenKind::RBracket));
1222    }
1223
1224    #[test]
1225    fn lex_comments_stripped() {
1226        let tokens = lex("// this is a comment\nseed := hash(cycle)").unwrap();
1227        assert!(matches!(tokens[0].kind, TokenKind::Ident(ref s) if s == "seed"));
1228    }
1229
1230    #[test]
1231    fn lex_arrow() {
1232        let tokens = lex("(x: u64) -> (y: u64)").unwrap();
1233        // ( x : u64 ) -> ...
1234        // 0 1 2  3  4  5
1235        assert!(matches!(tokens[5].kind, TokenKind::Arrow));
1236    }
1237
1238    #[test]
1239    fn lex_large_int() {
1240        let tokens = lex("1710000000000").unwrap();
1241        assert!(matches!(tokens[0].kind, TokenKind::IntLit(1710000000000)));
1242    }
1243
1244    #[test]
1245    fn lex_hex_int() {
1246        let tokens = lex("0xFF").unwrap();
1247        assert!(matches!(tokens[0].kind, TokenKind::IntLit(255)));
1248    }
1249
1250    #[test]
1251    fn lex_block_comment() {
1252        let tokens = lex("a := /* skip this */ hash(b)").unwrap();
1253        assert!(matches!(tokens[0].kind, TokenKind::Ident(ref s) if s == "a"));
1254        assert!(matches!(tokens[1].kind, TokenKind::ColonEq));
1255        assert!(matches!(tokens[2].kind, TokenKind::Ident(ref s) if s == "hash"));
1256    }
1257
1258    #[test]
1259    fn lex_block_comment_multiline() {
1260        let tokens = lex("a := 42\n/* this\nis\na\nblock */\nb := 7").unwrap();
1261        assert!(matches!(tokens[0].kind, TokenKind::Ident(ref s) if s == "a"));
1262        assert!(matches!(tokens[2].kind, TokenKind::IntLit(42)));
1263        assert!(matches!(tokens[3].kind, TokenKind::Ident(ref s) if s == "b"));
1264    }
1265
1266    #[test]
1267    fn lex_doc_comment() {
1268        // Triple-slash doc comments are stripped like line comments
1269        let tokens = lex("/// doc comment\nseed := hash(cycle)").unwrap();
1270        assert!(matches!(tokens[0].kind, TokenKind::Ident(ref s) if s == "seed"));
1271    }
1272
1273    #[test]
1274    fn lex_input_keyword_bare() {
1275        let tokens = lex("input cycle: u64").unwrap();
1276        assert!(matches!(tokens[0].kind, TokenKind::Input));
1277        assert!(matches!(tokens[1].kind, TokenKind::Ident(ref s) if s == "cycle"));
1278        assert!(matches!(tokens[2].kind, TokenKind::Colon));
1279        assert!(matches!(tokens[3].kind, TokenKind::Ident(ref s) if s == "u64"));
1280    }
1281
1282    #[test]
1283    fn lex_arithmetic_operators() {
1284        let tokens = lex("a + b * 2.0 - c / d % e ^ f").unwrap();
1285        assert!(matches!(tokens[0].kind, TokenKind::Ident(ref s) if s == "a"));
1286        assert!(matches!(tokens[1].kind, TokenKind::Plus));
1287        assert!(matches!(tokens[2].kind, TokenKind::Ident(ref s) if s == "b"));
1288        assert!(matches!(tokens[3].kind, TokenKind::Star));
1289        assert!(matches!(tokens[4].kind, TokenKind::FloatLit(v) if v == 2.0));
1290        assert!(matches!(tokens[5].kind, TokenKind::Minus));
1291        assert!(matches!(tokens[6].kind, TokenKind::Ident(ref s) if s == "c"));
1292        assert!(matches!(tokens[7].kind, TokenKind::Slash));
1293        assert!(matches!(tokens[8].kind, TokenKind::Ident(ref s) if s == "d"));
1294        assert!(matches!(tokens[9].kind, TokenKind::Percent));
1295        assert!(matches!(tokens[10].kind, TokenKind::Ident(ref s) if s == "e"));
1296        assert!(matches!(tokens[11].kind, TokenKind::Caret));
1297        assert!(matches!(tokens[12].kind, TokenKind::Ident(ref s) if s == "f"));
1298    }
1299
1300    #[test]
1301    fn lex_minus_binary_vs_negative_literal() {
1302        // After an identifier, `-` is a binary Minus, not a negative literal.
1303        let tokens = lex("x - 3").unwrap();
1304        assert!(matches!(tokens[0].kind, TokenKind::Ident(ref s) if s == "x"));
1305        assert!(matches!(tokens[1].kind, TokenKind::Minus));
1306        assert!(matches!(tokens[2].kind, TokenKind::IntLit(3)));
1307    }
1308
1309    #[test]
1310    fn lex_negative_via_minus_token() {
1311        // `-3.0` is lexed as Minus + FloatLit(3.0); the parser
1312        // handles negation via UnaryNeg.
1313        let tokens = lex("-3.0").unwrap();
1314        assert!(matches!(tokens[0].kind, TokenKind::Minus));
1315        assert!(matches!(tokens[1].kind, TokenKind::FloatLit(v) if v == 3.0));
1316
1317        // `-3` is Minus + IntLit(3).
1318        let tokens = lex("-3").unwrap();
1319        assert!(matches!(tokens[0].kind, TokenKind::Minus));
1320        assert!(matches!(tokens[1].kind, TokenKind::IntLit(3)));
1321    }
1322
1323    #[test]
1324    fn lex_scientific_notation() {
1325        let tokens = lex("1e10").unwrap();
1326        assert!(matches!(&tokens[0].kind, TokenKind::FloatLit(v) if (*v - 1e10).abs() < 1e5));
1327    }
1328
1329    #[test]
1330    fn lex_scientific_notation_negative_exponent() {
1331        let tokens = lex("1e-10").unwrap();
1332        assert!(matches!(&tokens[0].kind, TokenKind::FloatLit(v) if *v > 0.0 && *v < 1e-5));
1333    }
1334
1335    #[test]
1336    fn lex_scientific_notation_with_decimal() {
1337        let tokens = lex("2.5e3").unwrap();
1338        assert!(matches!(&tokens[0].kind, TokenKind::FloatLit(v) if (*v - 2500.0).abs() < 0.1));
1339    }
1340
1341    #[test]
1342    fn lex_scientific_notation_positive_exponent() {
1343        let tokens = lex("3E+5").unwrap();
1344        assert!(matches!(&tokens[0].kind, TokenKind::FloatLit(v) if (*v - 3e5).abs() < 1.0));
1345    }
1346
1347    #[test]
1348    fn lex_sci_uppercase_e() {
1349        let t = lex("2.5E3").unwrap();
1350        assert!(matches!(&t[0].kind, TokenKind::FloatLit(v) if (*v - 2500.0).abs() < 0.1));
1351    }
1352
1353    #[test]
1354    fn lex_sci_explicit_positive_exp() {
1355        let t = lex("1e+10").unwrap();
1356        assert!(matches!(&t[0].kind, TokenKind::FloatLit(v) if (*v - 1e10).abs() < 1e5));
1357    }
1358
1359    #[test]
1360    fn lex_sci_decimal_negative_exp() {
1361        let t = lex("3.14e-2").unwrap();
1362        assert!(matches!(&t[0].kind, TokenKind::FloatLit(v) if (*v - 0.0314).abs() < 0.001));
1363    }
1364
1365    #[test]
1366    fn lex_sci_zero_exponent() {
1367        let t = lex("0.5e0").unwrap();
1368        assert!(matches!(&t[0].kind, TokenKind::FloatLit(v) if (*v - 0.5).abs() < 0.001));
1369    }
1370
1371    #[test]
1372    fn lex_sci_very_small() {
1373        let t = lex("1e-300").unwrap();
1374        assert!(matches!(&t[0].kind, TokenKind::FloatLit(v) if *v > 0.0 && *v < 1e-299));
1375    }
1376
1377    #[test]
1378    fn lex_sci_uppercase_no_decimal() {
1379        let t = lex("1E10").unwrap();
1380        assert!(matches!(&t[0].kind, TokenKind::FloatLit(v) if (*v - 1e10).abs() < 1e5));
1381    }
1382
1383    #[test]
1384    fn lex_slash_vs_comment() {
1385        // Single `/` is Slash, `//` is a comment.
1386        let tokens = lex("a / b // comment").unwrap();
1387        assert!(matches!(tokens[0].kind, TokenKind::Ident(ref s) if s == "a"));
1388        assert!(matches!(tokens[1].kind, TokenKind::Slash));
1389        assert!(matches!(tokens[2].kind, TokenKind::Ident(ref s) if s == "b"));
1390        assert!(matches!(tokens[3].kind, TokenKind::Eof));
1391    }
1392
1393    // ── SRD-18c Layer 6 / SRD-18e Push 4: SI suffixes ──
1394
1395    #[test]
1396    fn lex_si_decimal_k_m_g_t_p() {
1397        let tokens = lex("1K 1M 1G 1T 1P").unwrap();
1398        assert!(matches!(tokens[0].kind, TokenKind::IntLit(1_000)));
1399        assert!(matches!(tokens[1].kind, TokenKind::IntLit(1_000_000)));
1400        assert!(matches!(tokens[2].kind, TokenKind::IntLit(1_000_000_000)));
1401        assert!(matches!(
1402            tokens[3].kind,
1403            TokenKind::IntLit(1_000_000_000_000)
1404        ));
1405        assert!(matches!(
1406            tokens[4].kind,
1407            TokenKind::IntLit(1_000_000_000_000_000)
1408        ));
1409    }
1410
1411    #[test]
1412    fn lex_si_binary_ki_mi_gi_ti_pi() {
1413        let tokens = lex("1Ki 1Mi 1Gi 1Ti 1Pi").unwrap();
1414        assert!(matches!(tokens[0].kind, TokenKind::IntLit(1024)));
1415        assert!(matches!(tokens[1].kind, TokenKind::IntLit(1_048_576)));
1416        assert!(matches!(tokens[2].kind, TokenKind::IntLit(1_073_741_824)));
1417        assert!(matches!(
1418            tokens[3].kind,
1419            TokenKind::IntLit(1_099_511_627_776)
1420        ));
1421        assert!(matches!(
1422            tokens[4].kind,
1423            TokenKind::IntLit(1_125_899_906_842_624)
1424        ));
1425    }
1426
1427    #[test]
1428    fn lex_si_subunit_m_u_n() {
1429        // Sub-unit suffixes promote integer to float (5m = 0.005).
1430        let tokens = lex("5m 5u 5n").unwrap();
1431        assert!(matches!(tokens[0].kind, TokenKind::FloatLit(v) if (v - 0.005).abs() < 1e-12));
1432        assert!(matches!(tokens[1].kind, TokenKind::FloatLit(v) if (v - 0.000_005).abs() < 1e-15));
1433        assert!(
1434            matches!(tokens[2].kind, TokenKind::FloatLit(v) if (v - 0.000_000_005).abs() < 1e-18)
1435        );
1436    }
1437
1438    #[test]
1439    fn lex_si_float_base_with_decimal_suffix() {
1440        // 1.5K = 1500, integral → IntLit per SRD-18c §"Layer 6"
1441        let tokens = lex("1.5K").unwrap();
1442        assert!(
1443            matches!(tokens[0].kind, TokenKind::IntLit(1_500)),
1444            "1.5K should be IntLit(1500), got {:?}",
1445            tokens[0].kind
1446        );
1447    }
1448
1449    #[test]
1450    fn lex_si_float_base_non_integral_stays_float() {
1451        // 1.5G = 1_500_000_000.0 — integral → IntLit
1452        // 1.5K = 1500 — integral → IntLit
1453        // 1.25K = 1250 — integral → IntLit
1454        // To force float: an irrational-result combination.
1455        // 0.001K = 1.0 → still integral → IntLit. Hmm. Let's
1456        // try a sub-unit float: 1.5m = 0.0015, non-integral.
1457        let tokens = lex("1.5m").unwrap();
1458        assert!(matches!(tokens[0].kind, TokenKind::FloatLit(v) if (v - 0.0015).abs() < 1e-12));
1459    }
1460
1461    #[test]
1462    fn lex_si_kilometers_stays_identifier() {
1463        // The K of `Kilometers` is followed by an
1464        // identifier-cont char; suffix doesn't apply.
1465        let tokens = lex("1 Kilometers").unwrap();
1466        assert!(matches!(tokens[0].kind, TokenKind::IntLit(1)));
1467        assert!(matches!(tokens[1].kind, TokenKind::Ident(ref s) if s == "Kilometers"));
1468    }
1469
1470    #[test]
1471    fn lex_si_overflow_errors_loud() {
1472        // 1P × 1000 doesn't overflow but multiplying by P
1473        // a value already at u64::MAX/1000 + 1 would. Use
1474        // a value that 1P would overflow: 100000P = 10^20 > u64::MAX.
1475        let err = lex("100000P").unwrap_err();
1476        assert!(err.contains("overflows u64"), "{err}");
1477    }
1478
1479    #[test]
1480    fn lex_si_in_range_expression() {
1481        // SI literals work in range positions (SRD-18c
1482        // example `1K..1M..100K`). The lexer doesn't know
1483        // about ranges yet, but should produce the right
1484        // numeric tokens.
1485        let tokens = lex("1K..1M..100K").unwrap();
1486        assert!(matches!(tokens[0].kind, TokenKind::IntLit(1_000)));
1487        // Token 1 is `..` (still parsed as Dot Dot in
1488        // pre-Push-3 lexer, but the numeric values are right).
1489        // We just verify the IntLits at positions 0, 3, 6.
1490        // (Positions depend on how `..` is tokenized today;
1491        // we walk the IntLits.)
1492        let int_lits: Vec<u64> = tokens
1493            .iter()
1494            .filter_map(|t| match &t.kind {
1495                TokenKind::IntLit(v) => Some(*v),
1496                _ => None,
1497            })
1498            .collect();
1499        assert_eq!(int_lits, vec![1_000, 1_000_000, 100_000]);
1500    }
1501
1502    #[test]
1503    fn lex_si_disambiguation_two_char_first() {
1504        // `1Ki` should match the 2-char binary suffix, NOT
1505        // 1-char `K` followed by `i` identifier.
1506        let tokens = lex("1Ki").unwrap();
1507        assert!(matches!(tokens[0].kind, TokenKind::IntLit(1024)));
1508        // No leftover identifier after the suffix.
1509        assert!(matches!(tokens[1].kind, TokenKind::Eof));
1510    }
1511
1512    #[test]
1513    fn lex_si_followed_by_operator_applies_suffix() {
1514        // `1K+5` — `K` followed by `+` (not ident-cont) → suffix applies.
1515        let tokens = lex("1K+5").unwrap();
1516        assert!(matches!(tokens[0].kind, TokenKind::IntLit(1000)));
1517        assert!(matches!(tokens[1].kind, TokenKind::Plus));
1518        assert!(matches!(tokens[2].kind, TokenKind::IntLit(5)));
1519    }
1520}