Skip to main content

bynk_syntax/
lexer.rs

1//! Lexer for Bynk v0.
2//!
3//! Token kinds correspond to the terminals defined in the grammar (spec §3
4//! and §4). Whitespace is skipped; line comments are emitted as `Comment`
5//! tokens so the formatter can preserve them through round-trips (v1.1 LSP
6//! spec §3.5). Doc blocks (`---`) are emitted as `DocBlock` tokens, lexed
7//! outside of logos (see [`tokenize`]).
8
9use logos::Logos;
10
11use crate::error::CompileError;
12use crate::span::{FileId, Span};
13
14/// #1657: the largest `Int`, `Number.MAX_SAFE_INTEGER` (2^53 − 1). An `Int`
15/// is a JS safe integer; the range is symmetric.
16pub const MAX_SAFE_INT: i64 = 9_007_199_254_740_991;
17
18/// v0.142 (ADR 0166): strip `_` digit separators from a numeric literal's lexeme
19/// before it is parsed into a value. The lexer's `IntLit`/`FloatLit` regexes only
20/// admit an `_` between two digit groups, so removing every `_` yields a plain
21/// digit string; the separators are purely visual. Allocates only when the
22/// literal actually carries a separator (the common case does not).
23pub(crate) fn strip_digit_separators(lexeme: &str) -> std::borrow::Cow<'_, str> {
24    if lexeme.as_bytes().contains(&b'_') {
25        std::borrow::Cow::Owned(lexeme.replace('_', ""))
26    } else {
27        std::borrow::Cow::Borrowed(lexeme)
28    }
29}
30
31/// Token kinds. Discriminants without payload data; the lexeme is recovered
32/// from the source string via the token's [`Span`].
33///
34/// Note: `--` line comments and `---` doc block markers are handled outside
35/// logos (see [`tokenize`]), because doc blocks are delimited by `---` lines
36/// containing only the marker and may span multiple source lines.
37#[derive(Logos, Debug, Clone, Copy, PartialEq, Eq)]
38#[logos(skip r"[ \t\r\n]+")]
39pub enum TokenKind {
40    // Keywords
41    #[token("commons")]
42    Commons,
43    #[token("type")]
44    Type,
45    // Events track, slice 0 (spine #936): `event Name = { fields }` — a
46    // context-only item declaring a typed fact. RESERVED_CONTEXTUAL, not
47    // hard, per ADR 0272's `messages` postmortem (a hard keyword broke the
48    // dotted-name case its own worked example used); `event` is at least as
49    // likely a parameter/local name (`on event(event: T)`).
50    #[token("event")]
51    Event,
52    #[token("fn")]
53    Fn,
54    #[token("where")]
55    Where,
56    // #548: the `and` keyword was retired — refinement predicates now join with
57    // `&&` (the one conjunction spelling). `and` is a free identifier again.
58    #[token("true")]
59    True,
60    #[token("false")]
61    False,
62    #[token("Int")]
63    Int,
64    #[token("String")]
65    String,
66    #[token("Bool")]
67    Bool,
68    // v0.21 keyword
69    #[token("Float")]
70    Float,
71    // v0.86 keyword (ADR 0112): the `Duration` base type.
72    #[token("Duration")]
73    Duration,
74    // v0.90 keyword (ADR 0114): the `Instant` base type.
75    #[token("Instant")]
76    Instant,
77    // v0.110 keyword (ADR 0142): the `Bytes` base type.
78    #[token("Bytes")]
79    Bytes,
80    // v0.1 keywords
81    #[token("let")]
82    Let,
83    #[token("if")]
84    If,
85    #[token("else")]
86    Else,
87    #[token("Ok")]
88    Ok,
89    #[token("Err")]
90    Err,
91    #[token("Result")]
92    Result,
93    #[token("ValidationError")]
94    ValidationError,
95    // v0.22b keyword
96    #[token("JsonError")]
97    JsonError,
98    // v0.2 keywords
99    #[token("enum")]
100    Enum,
101    #[token("match")]
102    Match,
103    #[token("Option")]
104    Option,
105    #[token("record")]
106    Record,
107    #[token("self")]
108    Self_,
109    #[token("Some")]
110    Some,
111    #[token("None")]
112    None,
113    #[token("is")]
114    Is,
115    // v0.3 keywords
116    #[token("opaque")]
117    Opaque,
118    #[token("uses")]
119    Uses,
120    // v0.4 keywords
121    #[token("context")]
122    Context,
123    #[token("consumes")]
124    Consumes,
125    #[token("exports")]
126    Exports,
127    #[token("transparent")]
128    Transparent,
129    // v0.6 keywords
130    #[token("as")]
131    As,
132    // v0.7 keywords (v0.112: `assert`→`expect`, `test`→`suite`/`case`;
133    // v0.118: `mocks` retired — test doubles are stubs at a seam; the stub form
134    // moved off the punned `provides` keyword to its own `stub` keyword in the
135    // keyword-hygiene batch, #548)
136    #[token("expect")]
137    Expect,
138    #[token("suite")]
139    Suite,
140    #[token("case")]
141    Case,
142    // Keyword-hygiene batch (#548): the test-scope stub `stub Cap.op(…) <rhs>`,
143    // formerly the third pun on `provides`. `provides` now heads only a provider
144    // declaration / external provider.
145    #[token("stub")]
146    Stub,
147    // v0.114 keyword — generative tests (testing track slice 2). `for` and `all`
148    // are deliberately *not* keywords: `all` is a list combinator (`all(xs, p)`)
149    // and must stay a usable identifier. The `for all` binder is parsed
150    // contextually (two identifiers) inside a `property` body instead.
151    #[token("property")]
152    Property,
153    // v0.17 keywords
154    #[token("adapter")]
155    Adapter,
156    #[token("binding")]
157    Binding,
158    // v0.5 keywords
159    #[token("agent")]
160    Agent,
161    #[token("capability")]
162    Capability,
163    #[token("Effect")]
164    Effect,
165    // v0.146 keyword (ADR 0170): `do e` — an effect-performing expression
166    // statement (the binder-free `let _ <- e` for a unit effect).
167    #[token("do")]
168    Do,
169    #[token("given")]
170    Given,
171    #[token("on")]
172    On,
173    // v0.9 keyword
174    #[token("http")]
175    Http,
176    // v0.10a keyword
177    #[token("cron")]
178    Cron,
179    // v0.10b keyword
180    #[token("queue")]
181    Queue,
182    // v0.44 keywords: `from` heads a service's protocol clause; `protocol` is
183    // reserved (protocols are a closed, compiler-known set — no declaration kind).
184    #[token("from")]
185    From,
186    #[token("protocol")]
187    Protocol,
188    #[token("provides")]
189    Provides,
190    #[token("service")]
191    Service,
192    // v0.45 keywords: `actor` heads a boundary-contract declaration; `by`
193    // heads a handler's actor clause.
194    #[token("actor")]
195    Actor,
196    #[token("by")]
197    By,
198    // v0.80 keywords: `invariant` heads an agent invariant declaration; `implies`
199    // is the directional logical-implication operator (`P implies Q` ≡ `!P || Q`).
200    #[token("invariant")]
201    Invariant,
202    #[token("implies")]
203    Implies,
204    // v0.115 keywords — function contracts (testing track slice 3). `requires`
205    // and `ensures` head a contract clause on a `fn` signature (between the
206    // return type and the body). `result` is deliberately *not* a keyword: it is
207    // the ordinary value name outside a contract, so it stays a usable
208    // identifier; inside an `ensures` predicate it is bound contextually as the
209    // function's return value (parsed by scope, like `for`/`all` in slice 2).
210    // Distinct from ADR 0127's capability `@requires` annotation.
211    #[token("requires")]
212    Requires,
213    #[token("ensures")]
214    Ensures,
215    // v0.116 keyword — step invariants (testing track slice 4). `transition` heads
216    // an agent step-invariant declaration (beside `invariant`), a predicate over
217    // the pre- and post-commit state pair. `old` and `new` are deliberately *not*
218    // keywords: they stay ordinary value names outside a `transition`, and inside a
219    // `transition` predicate they are bound contextually to the old/new state
220    // records (parsed by scope, like `result` in an `ensures`).
221    #[token("transition")]
222    Transition,
223    // message-bundles track, slice 1: `messages <tag> { "code" => "template" }`
224    // — a commons item declaring one locale's message bundle.
225    #[token("messages")]
226    Messages,
227    /// `...` — used in record-spread expressions (v0.5).
228    #[token("...")]
229    DotDotDot,
230    /// `..` — the "rest of the fields" marker on an events subscription
231    /// pattern (Events track slice 1, spine #936): `from Events(E { region:
232    /// Region.Domestic, .. })`. A genuine token, not two adjacent `Dot`s —
233    /// logos maximal-munches `...`/`..`/`.` correctly once all three are
234    /// registered, and a real token keeps this in agreement with
235    /// tree-sitter's grammar (which declares `".."` as one literal), so the
236    /// two parsers cannot diverge on a whitespace-split `. .` the way ADR
237    /// 0253 D4 found a leaking `where`-check divergence once before.
238    #[token("..")]
239    DotDot,
240    /// `<-` — Effect bind operator (v0.5).
241    #[token("<-")]
242    LArrow,
243    /// `~>` — asynchronous fire-and-forget send marker (v0.79). A leading
244    /// statement marker, never on the RHS of a `let`; distinct from `<-` so the
245    /// call site shows whether the caller waits.
246    #[token("~>")]
247    TildeArrow,
248    /// `:=` — Cell write (v0.81, storage track). A handler statement
249    /// `cell := expr`; distinct from `=` (binding) and `:` (annotation). Longer
250    /// than `:`/`=` so logos matches it as one token.
251    #[token(":=")]
252    ColonEq,
253
254    /// A documentation block: `---` line ... `---` line. The token's span
255    /// covers the full block including both `---` markers. The body content
256    /// is recovered from the source via the span (see [`doc_block_content`]).
257    /// Inserted by [`tokenize`]; not lexed by logos directly.
258    DocBlock,
259
260    /// A line comment: `-- ...` running to end of line. The span starts at
261    /// the `--` marker and runs through the last character before the
262    /// terminating newline (exclusive). The trivia body (the text after the
263    /// `--` marker) is recovered from the source via the span. Inserted by
264    /// [`tokenize`]; not lexed by logos directly so it cannot be mistaken
265    /// for an `--` operator sequence.
266    Comment,
267
268    // Identifier
269    #[regex(r"[A-Za-z][A-Za-z0-9_]*")]
270    Ident,
271
272    // Literals. v0.142 (ADR 0166): an `_` digit separator may appear between
273    // digits (`1_048_576`) — never leading, trailing, or doubled (each `_` must
274    // sit between two digit groups). The separators are stripped before the value
275    // is parsed; they are purely visual.
276    #[regex(r"[0-9]+(_[0-9]+)*")]
277    IntLit,
278    // A float literal: fraction with a digit on both sides of the `.`, an
279    // exponent, or both (v0.21 §3). `1.` and `.5` are NOT float literals —
280    // the digit-both-sides rule keeps `2.5.round()` / `1.toFloat()` lexing
281    // as method calls on numeric literals. Digit separators (v0.142) may appear
282    // in any digit group, including the exponent.
283    #[regex(
284        r"[0-9]+(_[0-9]+)*\.[0-9]+(_[0-9]+)*([eE][+-]?[0-9]+(_[0-9]+)*)?|[0-9]+(_[0-9]+)*[eE][+-]?[0-9]+(_[0-9]+)*"
285    )]
286    FloatLit,
287    // A double-quoted string with simple escapes. The body excludes the closing
288    // quote; we accept any non-quote/non-backslash/non-newline char, or a
289    // backslash followed by one of the four allowed escapes.
290    #[regex(r#""([^"\\\n]|\\[nt"\\])*""#)]
291    StrLit,
292    // An interpolated string `"… \(expr) …"` (v0.43). Hand-scanned in
293    // `tokenize` (logos cannot balance the holes' parens), never produced by
294    // the logos lexer — like [`TokenKind::DocBlock`]/[`TokenKind::Comment`].
295    // The span covers the whole `"…"`; the parser splits chunks from holes.
296    InterpStr,
297
298    // Multi-char operators
299    #[token("->")]
300    Arrow,
301    #[token("==")]
302    EqEq,
303    #[token("!=")]
304    BangEq,
305    #[token("<=")]
306    LtEq,
307    #[token(">=")]
308    GtEq,
309    #[token("&&")]
310    AmpAmp,
311    #[token("||")]
312    PipePipe,
313
314    // Single-char operators
315    #[token("+")]
316    Plus,
317    #[token("-")]
318    Minus,
319    #[token("*")]
320    Star,
321    #[token("/")]
322    Slash,
323    #[token("!")]
324    Bang,
325    #[token("=")]
326    Eq,
327    #[token("<")]
328    Lt,
329    #[token(">")]
330    Gt,
331    // v0.1 postfix operator
332    #[token("?")]
333    Question,
334    // v0.2 match-arm arrow
335    #[token("=>")]
336    FatArrow,
337    // v0.2 wildcard pattern (also valid as identifier start; the lexer
338    // prefers identifier for any longer match, so `_foo` is still Ident).
339    #[token("_")]
340    Underscore,
341    // v0.2 sum-type variant separator (also used as future bitwise OR);
342    // single `|` distinct from `||`.
343    #[token("|")]
344    Pipe,
345    /// `@` — storage-annotation marker (v0.85, storage track; ADR 0111). Leads a
346    /// `@name(args)` annotation on a `store` field (`@ttl(…)`/`@indexed(…)`); it
347    /// appears only in store-field-declaration position, never as an expression
348    /// operator.
349    #[token("@")]
350    At,
351
352    // Punctuation
353    #[token("(")]
354    LParen,
355    #[token(")")]
356    RParen,
357    #[token("{")]
358    LBrace,
359    #[token("}")]
360    RBrace,
361    #[token("[")]
362    LBracket,
363    #[token("]")]
364    RBracket,
365    #[token(",")]
366    Comma,
367    #[token(":")]
368    Colon,
369    #[token(".")]
370    Dot,
371}
372
373impl TokenKind {
374    /// Human-readable display name for diagnostics.
375    pub fn describe(self) -> &'static str {
376        use TokenKind::*;
377        match self {
378            Commons => "`commons`",
379            Type => "`type`",
380            Event => "`event`",
381            Fn => "`fn`",
382            Where => "`where`",
383            True => "`true`",
384            False => "`false`",
385            Int => "`Int`",
386            String => "`String`",
387            Bool => "`Bool`",
388            Float => "`Float`",
389            Duration => "`Duration`",
390            Instant => "`Instant`",
391            Bytes => "`Bytes`",
392            Let => "`let`",
393            If => "`if`",
394            Else => "`else`",
395            Ok => "`Ok`",
396            Err => "`Err`",
397            Result => "`Result`",
398            ValidationError => "`ValidationError`",
399            JsonError => "`JsonError`",
400            Enum => "`enum`",
401            Match => "`match`",
402            Option => "`Option`",
403            Record => "`record`",
404            Self_ => "`self`",
405            Some => "`Some`",
406            None => "`None`",
407            Is => "`is`",
408            Opaque => "`opaque`",
409            Uses => "`uses`",
410            Context => "`context`",
411            Consumes => "`consumes`",
412            Exports => "`exports`",
413            Transparent => "`transparent`",
414            As => "`as`",
415            Expect => "`expect`",
416            Suite => "`suite`",
417            Case => "`case`",
418            Property => "`property`",
419            Adapter => "`adapter`",
420            Binding => "`binding`",
421            Agent => "`agent`",
422            Capability => "`capability`",
423            Effect => "`Effect`",
424            Do => "`do`",
425            Given => "`given`",
426            On => "`on`",
427            Http => "`http`",
428            Cron => "`cron`",
429            Queue => "`queue`",
430            From => "`from`",
431            Protocol => "`protocol`",
432            Provides => "`provides`",
433            Stub => "`stub`",
434            Service => "`service`",
435            Actor => "`actor`",
436            By => "`by`",
437            Invariant => "`invariant`",
438            Implies => "`implies`",
439            Requires => "`requires`",
440            Ensures => "`ensures`",
441            Transition => "`transition`",
442            Messages => "`messages`",
443            ColonEq => "`:=`",
444            DotDotDot => "`...`",
445            DotDot => "`..`",
446            LArrow => "`<-`",
447            TildeArrow => "`~>`",
448            DocBlock => "documentation block",
449            Comment => "line comment",
450            Ident => "identifier",
451            IntLit => "integer literal",
452            FloatLit => "float literal",
453            StrLit => "string literal",
454            InterpStr => "interpolated string",
455            Arrow => "`->`",
456            EqEq => "`==`",
457            BangEq => "`!=`",
458            LtEq => "`<=`",
459            GtEq => "`>=`",
460            AmpAmp => "`&&`",
461            PipePipe => "`||`",
462            Plus => "`+`",
463            Minus => "`-`",
464            Star => "`*`",
465            Slash => "`/`",
466            Bang => "`!`",
467            Eq => "`=`",
468            Lt => "`<`",
469            Gt => "`>`",
470            Question => "`?`",
471            FatArrow => "`=>`",
472            Underscore => "`_`",
473            Pipe => "`|`",
474            At => "`@`",
475            LParen => "`(`",
476            RParen => "`)`",
477            LBrace => "`{`",
478            RBrace => "`}`",
479            LBracket => "`[`",
480            RBracket => "`]`",
481            Comma => "`,`",
482            Colon => "`:`",
483            Dot => "`.`",
484        }
485    }
486}
487
488/// A token plus its source span.
489#[derive(Debug, Clone, Copy)]
490pub struct Token {
491    pub kind: TokenKind,
492    pub span: Span,
493}
494
495/// Tokenise a source string with no real file identity — every span's
496/// [`FileId`] defaults to [`FileId::UNKNOWN`]. See [`tokenize_in`] for the
497/// real-identity entry point production callers use.
498pub fn tokenize(source: &str) -> Result<Vec<Token>, CompileError> {
499    tokenize_in(source, FileId::UNKNOWN)
500}
501
502/// Tokenise a source string, stamping every token's span with `file`.
503/// Returns the full token vector or the first lexical error.
504///
505/// Doc blocks (`---` ... `---`) and line comments (`-- ...`) are recognised
506/// outside the logos-generated lexer: we scan the source one segment at a
507/// time, dispatching to logos for ordinary tokens between non-token spans.
508pub fn tokenize_in(source: &str, file: FileId) -> Result<Vec<Token>, CompileError> {
509    let mut tokens = Vec::new();
510    let bytes = source.as_bytes();
511    let mut pos = 0;
512    while pos < bytes.len() {
513        // Detect a `---` doc-block marker at the start of a line (the line may
514        // begin with leading whitespace; the marker itself must be alone on
515        // its line).
516        if let Some(open_end) = doc_block_open_at(source, pos) {
517            // Find the matching closing `---` line.
518            match doc_block_close(source, open_end) {
519                Some((close_start, close_end)) => {
520                    let span = Span::new_in(file, pos, close_end);
521                    tokens.push(Token {
522                        kind: TokenKind::DocBlock,
523                        span,
524                    });
525                    let _ = close_start;
526                    pos = close_end;
527                    continue;
528                }
529                None => {
530                    return Err(CompileError::new(
531                        "bynk.lex.unclosed_doc_block",
532                        Span::new_in(file, pos, open_end),
533                        "documentation block opened but never closed",
534                    )
535                    .with_note(
536                        "a doc block must be terminated by another `---` on a line by itself",
537                    ));
538                }
539            }
540        }
541        // A `--` line comment: emit a `Comment` token covering everything
542        // up to (but not including) the terminating newline. Doc-block
543        // detection above already ruled out a `---` marker at line start
544        // — and once we've consumed past the leading `--`, any further
545        // dashes are part of the comment body. Preserving comments as
546        // trivia tokens lets the parser attach them to declarations so
547        // the formatter can emit them in place (v1.1 LSP spec §3.5).
548        //
549        // #548 (keyword-hygiene batch): a `--` opens a comment only when it is
550        // at the start of input or **preceded by whitespace**. Adjacent to a
551        // preceding token (`a--b`), the `--` is *not* a comment — it lexes as two
552        // `-` operators (`a - -b`), so a subtraction-of-negation is never
553        // silently swallowed as a line comment. This resolves the `a--b`
554        // "comment vs subtraction" ambiguity in favour of subtraction.
555        let comment_eligible = pos == 0 || matches!(bytes[pos - 1], b' ' | b'\t' | b'\r' | b'\n');
556        if comment_eligible && pos + 1 < bytes.len() && bytes[pos] == b'-' && bytes[pos + 1] == b'-'
557        {
558            let start = pos;
559            while pos < bytes.len() && bytes[pos] != b'\n' {
560                pos += 1;
561            }
562            tokens.push(Token {
563                kind: TokenKind::Comment,
564                span: Span::new_in(file, start, pos),
565            });
566            continue;
567        }
568        // Skip ordinary whitespace inline (logos handles it too, but we may
569        // be in the middle of the source between specials).
570        if matches!(bytes[pos], b' ' | b'\t' | b'\r' | b'\n') {
571            pos += 1;
572            continue;
573        }
574        // An interpolated string `"… \(expr) …"` (v0.43): only strings that
575        // actually contain a `\(` hole are hand-scanned here; plain strings
576        // fall through to the logos `StrLit` path unchanged. `\(` is an
577        // invalid escape in the logos grammar, so this never re-routes a
578        // currently-valid literal.
579        if bytes[pos] == b'"' && has_interp_hole(bytes, pos) {
580            let end = scan_str(bytes, source, pos, 0, file)?;
581            tokens.push(Token {
582                kind: TokenKind::InterpStr,
583                span: Span::new_in(file, pos, end),
584            });
585            pos = end;
586            continue;
587        }
588        // Otherwise dispatch a single logos token starting at `pos`.
589        let mut lex = TokenKind::lexer(&source[pos..]);
590        let Some(result) = lex.next() else {
591            // No token at this position; treat as unexpected character so
592            // the user sees something useful.
593            let ch = source[pos..].chars().next().unwrap_or('\0');
594            let span = Span::new_in(file, pos, pos + ch.len_utf8());
595            return Err(CompileError::new(
596                "bynk.lex.unexpected_character",
597                span,
598                format!("unexpected character `{ch}`"),
599            ));
600        };
601        let local = lex.span();
602        let span: Span = Span::new_in(file, pos + local.start, pos + local.end);
603        match result {
604            Ok(kind) => {
605                // #1657 (runtime-semantics track §3.4): an `Int` is a JS safe
606                // integer, ±(2^53 − 1), the range the runtime represents
607                // exactly. A larger literal would round silently
608                // (`9007199254740993 == 9007199254740992`). The sign is a
609                // separate operator, and the bound is symmetric, so the
610                // magnitude decides.
611                if kind == TokenKind::IntLit {
612                    let slice = &source[span.range()];
613                    if !strip_digit_separators(slice)
614                        .parse::<i64>()
615                        .is_ok_and(|n| n <= MAX_SAFE_INT)
616                    {
617                        return Err(CompileError::new(
618                            "bynk.lex.integer_overflow",
619                            span,
620                            format!("integer literal `{slice}` is out of range for an `Int`"),
621                        )
622                        .with_note(
623                            "an `Int` is a safe integer, -(2^53 - 1) to 2^53 - 1 \
624                             (±9007199254740991); use a `Float` or a `String` for a larger value",
625                        ));
626                    }
627                }
628                if kind == TokenKind::FloatLit {
629                    let slice = &source[span.range()];
630                    match strip_digit_separators(slice).parse::<f64>() {
631                        Ok(v) if v.is_finite() => {}
632                        _ => {
633                            return Err(CompileError::new(
634                                "bynk.lex.float_literal_overflow",
635                                span,
636                                format!(
637                                    "float literal `{slice}` is out of range for a 64-bit float"
638                                ),
639                            )
640                            .with_note(
641                                "the literal does not fit a finite IEEE 754 double; \
642                                 the largest finite value is ~1.8e308",
643                            ));
644                        }
645                    }
646                }
647                tokens.push(Token { kind, span });
648                pos = span.end;
649            }
650            Err(()) => {
651                let slice = &source[span.range()];
652                let ch = slice.chars().next().unwrap_or('\0');
653                let err = if ch == '"' {
654                    CompileError::new(
655                        "bynk.lex.unterminated_string",
656                        span,
657                        "unterminated string literal",
658                    )
659                    .with_note(
660                        "string literals must close with `\"` on the same line; \
661                         supported escapes are `\\n`, `\\t`, `\\\"`, `\\\\`",
662                    )
663                } else {
664                    CompileError::new(
665                        "bynk.lex.unexpected_character",
666                        span,
667                        format!("unexpected character `{ch}`"),
668                    )
669                };
670                return Err(err);
671            }
672        }
673    }
674    Ok(tokens)
675}
676
677/// Like [`tokenize`], but with every interpolated-string token replaced by the
678/// tokens of its holes — each hole's bytes re-lexed and its token spans rebased
679/// to absolute source positions (the same rebase [`crate::parser`] applies when
680/// parsing a hole), recursing through nested interpolation. Chunk (literal) text
681/// between holes yields no tokens.
682///
683/// An interpolated string lexes to a single opaque `InterpStr` token, so the
684/// LSP's token-based cursor resolution (hover, go-to-definition, references,
685/// semantic tokens) is otherwise blind to identifiers inside `"… \(name) …"`.
686/// Expanding the holes makes those identifiers visible as ordinary `Ident`
687/// tokens with their real spans. (Issue #473.)
688///
689/// On a malformed interpolation (an `InterpStr` whose holes don't split, or a
690/// hole whose bytes don't re-lex) the offending token is kept opaque rather than
691/// dropped, so resolution degrades to the pre-fix behaviour instead of losing
692/// tokens.
693pub fn tokenize_expanding_holes(source: &str) -> Result<Vec<Token>, CompileError> {
694    tokenize_expanding_holes_in(source, FileId::UNKNOWN)
695}
696
697/// Like [`tokenize_expanding_holes`], but stamping every token's span
698/// (including rebased hole tokens) with `file`.
699pub fn tokenize_expanding_holes_in(source: &str, file: FileId) -> Result<Vec<Token>, CompileError> {
700    let mut out = Vec::new();
701    for tok in tokenize_in(source, file)? {
702        expand_hole_token(source, file, tok, &mut out);
703    }
704    Ok(out)
705}
706
707/// Push `tok` onto `out`, expanding it into its holes' tokens if it is an
708/// `InterpStr` (see [`tokenize_expanding_holes`]); otherwise push it as-is.
709fn expand_hole_token(source: &str, file: FileId, tok: Token, out: &mut Vec<Token>) {
710    if tok.kind != TokenKind::InterpStr {
711        out.push(tok);
712        return;
713    }
714    let Ok(segments) = split_interp(source, tok.span) else {
715        out.push(tok); // malformed interpolation — keep the opaque token
716        return;
717    };
718    for segment in segments {
719        let InterpSegment::Hole(hole) = segment else {
720            continue; // chunk text carries no tokens
721        };
722        let Ok(hole_tokens) = tokenize_in(&source[hole.range()], file) else {
723            continue;
724        };
725        for mut t in hole_tokens {
726            // Rebase the hole's local spans to absolute source positions.
727            t.span = Span::new_in(file, t.span.start + hole.start, t.span.end + hole.start);
728            expand_hole_token(source, file, t, out); // recurse for nested interpolation
729        }
730    }
731}
732
733/// Cheap routing pre-scan (v0.43): does the string opening at `start` contain a
734/// `\(` interpolation hole before it closes (or the line ends)? Decides whether
735/// `tokenize` hand-scans the string as an `InterpStr` or defers to logos for a
736/// plain `StrLit`. Deliberately tolerant — a malformed string with a hole is
737/// routed here so the hole-aware scanner produces the precise error.
738fn has_interp_hole(bytes: &[u8], start: usize) -> bool {
739    let mut i = start + 1;
740    while i < bytes.len() {
741        match bytes[i] {
742            b'\n' | b'"' => return false,
743            b'\\' => {
744                if bytes.get(i + 1) == Some(&b'(') {
745                    return true;
746                }
747                i += 2;
748            }
749            _ => i += 1,
750        }
751    }
752    false
753}
754
755/// Scan a double-quoted string starting at `start` (the opening `"`), returning
756/// the byte offset just past the closing `"`. Recognises the four simple
757/// escapes plus `\(…)` interpolation holes, whose parens are balanced (and
758/// whose nested strings are skipped) by [`scan_hole`]. (v0.43.)
759fn scan_str(
760    bytes: &[u8],
761    source: &str,
762    start: usize,
763    depth: usize,
764    file: FileId,
765) -> Result<usize, CompileError> {
766    debug_assert_eq!(bytes[start], b'"');
767    if depth > crate::MAX_NESTING_DEPTH {
768        // Anchor on the opening `"` of the string that tipped over the limit.
769        return Err(too_deeply_nested_interpolation(Span::new_in(
770            file,
771            start,
772            start + 1,
773        )));
774    }
775    let mut i = start + 1;
776    loop {
777        if i >= bytes.len() || bytes[i] == b'\n' {
778            return Err(CompileError::new(
779                "bynk.lex.unterminated_string",
780                Span::new_in(file, start, i.min(bytes.len())),
781                "unterminated string literal",
782            )
783            .with_note(
784                "string literals must close with `\"` on the same line; \
785                 supported escapes are `\\n`, `\\t`, `\\\"`, `\\\\`, and `\\(…)` interpolation",
786            ));
787        }
788        match bytes[i] {
789            b'"' => return Ok(i + 1),
790            b'\\' => match bytes.get(i + 1) {
791                Some(b'n' | b't' | b'"' | b'\\') => i += 2,
792                Some(b'(') => i = scan_hole(bytes, source, i + 2, depth + 1, file)?,
793                other => {
794                    let shown = other.map(|b| (*b as char).to_string()).unwrap_or_default();
795                    // Cover `\` plus the whole offending char, advanced to a char
796                    // boundary so the span never splits a multibyte codepoint
797                    // (e.g. `\é`) — a fuzz invariant.
798                    let mut end = (i + 2).min(bytes.len());
799                    while end < source.len() && !source.is_char_boundary(end) {
800                        end += 1;
801                    }
802                    return Err(CompileError::new(
803                        "bynk.lex.bad_escape",
804                        Span::new_in(file, i, end),
805                        format!("invalid escape sequence `\\{shown}` in string literal"),
806                    )
807                    .with_note("supported escapes: \\n \\t \\\" \\\\ \\(…)"));
808                }
809            },
810            // Any other byte advances one position. UTF-8 continuation bytes
811            // are all >= 0x80, so they never collide with the ASCII specials.
812            _ => i += 1,
813        }
814    }
815}
816
817/// Scan an interpolation hole body. `start` points just past the `\(`; returns
818/// the offset just past the matching `)`. Tracks paren depth and skips nested
819/// strings (whose own parens must not close the hole), recursing through
820/// [`scan_str`] so nested interpolation nests correctly. (v0.43.)
821fn scan_hole(
822    bytes: &[u8],
823    source: &str,
824    start: usize,
825    nesting: usize,
826    file: FileId,
827) -> Result<usize, CompileError> {
828    if nesting > crate::MAX_NESTING_DEPTH {
829        // Anchor on the `\(` opener that tipped over the limit; it sits two
830        // bytes before `start` and is pure ASCII, so the span stays on char
831        // boundaries (a fuzz invariant).
832        return Err(too_deeply_nested_interpolation(Span::new_in(
833            file,
834            start.saturating_sub(2),
835            start,
836        )));
837    }
838    let mut i = start;
839    let mut depth = 1usize;
840    loop {
841        if i >= bytes.len() || bytes[i] == b'\n' {
842            return Err(CompileError::new(
843                "bynk.lex.unterminated_interpolation",
844                Span::new_in(file, start.saturating_sub(2), i.min(bytes.len())),
845                "unterminated interpolation hole",
846            )
847            .with_note(
848                "an interpolation hole `\\(…)` must close with a matching `)` on the same line",
849            ));
850        }
851        match bytes[i] {
852            b'(' => {
853                depth += 1;
854                i += 1;
855            }
856            b')' => {
857                depth -= 1;
858                i += 1;
859                if depth == 0 {
860                    return Ok(i);
861                }
862            }
863            b'"' => i = scan_str(bytes, source, i, nesting + 1, file)?,
864            _ => i += 1,
865        }
866    }
867}
868
869/// The bounded-depth diagnostic for interpolation that nests past
870/// [`crate::MAX_NESTING_DEPTH`]. `\("\("\(…` mutually recurses
871/// [`scan_str`] ↔ [`scan_hole`], one stack frame per level, so an unbounded
872/// scanner overflows and aborts `tokenize` (#713). `span` anchors the report on
873/// the opener that tipped over the limit (the `"` or the `\(`).
874fn too_deeply_nested_interpolation(span: Span) -> CompileError {
875    CompileError::new(
876        "bynk.lex.interpolation_too_deep",
877        span,
878        format!(
879            "string interpolation nests more than {} levels deep",
880            crate::MAX_NESTING_DEPTH
881        ),
882    )
883    .with_note(
884        "deeply nested `\\(…)` interpolation is rejected to keep the lexer from \
885         overflowing its stack and aborting; flatten or split the string",
886    )
887}
888
889/// One segment of a split interpolated string (v0.43): literal text (escapes
890/// resolved) or the absolute source span of a hole's expression (the bytes
891/// between `\(` and its matching `)`). The parser turns the latter into a real
892/// `Expr`; the lexer owns only the scanning.
893pub(crate) enum InterpSegment {
894    Chunk(String),
895    Hole(Span),
896}
897
898/// Split an `InterpStr` token (its `span` covers the whole `"…"`) into chunks
899/// and hole spans. Escapes in the chunks are resolved here (mirroring
900/// [`parse_string_literal`]); holes are returned as spans for the parser to
901/// re-lex and parse as expressions. (v0.43.)
902pub(crate) fn split_interp(source: &str, span: Span) -> Result<Vec<InterpSegment>, CompileError> {
903    let bytes = source.as_bytes();
904    let inner_end = span.end - 1; // the closing `"`
905    let mut segments = Vec::new();
906    let mut chunk = String::new();
907    let mut i = span.start + 1; // past the opening `"`
908    while i < inner_end {
909        match bytes[i] {
910            b'\\' => match bytes[i + 1] {
911                b'n' => {
912                    chunk.push('\n');
913                    i += 2;
914                }
915                b't' => {
916                    chunk.push('\t');
917                    i += 2;
918                }
919                b'"' => {
920                    chunk.push('"');
921                    i += 2;
922                }
923                b'\\' => {
924                    chunk.push('\\');
925                    i += 2;
926                }
927                b'(' => {
928                    if !chunk.is_empty() {
929                        segments.push(InterpSegment::Chunk(std::mem::take(&mut chunk)));
930                    }
931                    let hole_start = i + 2;
932                    let after = scan_hole(bytes, source, hole_start, 0, span.file)?;
933                    // `after` is one past the matching `)`; the hole body is
934                    // everything up to that `)`.
935                    segments.push(InterpSegment::Hole(Span::new_in(
936                        span.file,
937                        hole_start,
938                        after - 1,
939                    )));
940                    i = after;
941                }
942                // The lexer already validated every escape, so nothing else
943                // can appear here.
944                other => unreachable!("unvalidated escape `\\{}` in InterpStr", other as char),
945            },
946            _ => {
947                let ch = source[i..].chars().next().unwrap();
948                chunk.push(ch);
949                i += ch.len_utf8();
950            }
951        }
952    }
953    if !chunk.is_empty() {
954        segments.push(InterpSegment::Chunk(chunk));
955    }
956    Ok(segments)
957}
958
959/// If a `---` doc-block marker line starts at or shortly after `pos` (which
960/// must be at a line boundary), return the byte offset just past the marker
961/// line (after the terminating newline, or at EOF). The doc-block grammar
962/// requires the marker to be alone on its line; leading horizontal whitespace
963/// is allowed and ignored.
964fn doc_block_open_at(source: &str, pos: usize) -> Option<usize> {
965    let bytes = source.as_bytes();
966    if !at_line_start(source, pos) {
967        return None;
968    }
969    // Skip leading horizontal whitespace.
970    let mut i = pos;
971    while i < bytes.len() && (bytes[i] == b' ' || bytes[i] == b'\t') {
972        i += 1;
973    }
974    if i + 3 > bytes.len() {
975        return None;
976    }
977    if &bytes[i..i + 3] != b"---" {
978        return None;
979    }
980    i += 3;
981    // The marker may have additional trailing dashes (per spec "three or more
982    // consecutive hyphens"). Consume them.
983    while i < bytes.len() && bytes[i] == b'-' {
984        i += 1;
985    }
986    // After the dashes, allow only horizontal whitespace then newline/EOF.
987    while i < bytes.len() && (bytes[i] == b' ' || bytes[i] == b'\t' || bytes[i] == b'\r') {
988        i += 1;
989    }
990    if i == bytes.len() {
991        return Some(i);
992    }
993    if bytes[i] == b'\n' {
994        return Some(i + 1);
995    }
996    None
997}
998
999/// Find the next closing `---` line at or after `pos`. Returns
1000/// `(start_of_line, end_of_line)` (`end_of_line` is just past the
1001/// terminating newline, or at EOF).
1002fn doc_block_close(source: &str, mut pos: usize) -> Option<(usize, usize)> {
1003    let bytes = source.as_bytes();
1004    while pos < bytes.len() {
1005        // Advance pos to the start of a line.
1006        let line_start = pos;
1007        // Find the end of this line.
1008        let mut line_end = line_start;
1009        while line_end < bytes.len() && bytes[line_end] != b'\n' {
1010            line_end += 1;
1011        }
1012        // Check this line.
1013        if let Some(end) = doc_block_open_at(source, line_start) {
1014            return Some((line_start, end));
1015        }
1016        // Move to the next line.
1017        pos = if line_end < bytes.len() {
1018            line_end + 1
1019        } else {
1020            line_end
1021        };
1022    }
1023    None
1024}
1025
1026/// Returns true if byte offset `pos` is at a line start (column 0).
1027fn at_line_start(source: &str, pos: usize) -> bool {
1028    if pos == 0 {
1029        return true;
1030    }
1031    let bytes = source.as_bytes();
1032    bytes[pos - 1] == b'\n'
1033}
1034
1035/// The doc-block body as a byte range into `source` — leading/trailing `---`
1036/// marker lines stripped, no further processing (unlike [`doc_block_content`],
1037/// which additionally strips a common per-line indent — not offset-preserving).
1038/// Callers that need to map a position in the body back to `source` (e.g.
1039/// document-link spans) use this instead of re-deriving it from the string
1040/// `doc_block_content` returns.
1041pub fn doc_block_body_range(source: &str, span: Span) -> Option<std::ops::Range<usize>> {
1042    let slice = &source[span.range()];
1043    // Drop the first line (opening marker).
1044    let after_open_rel = slice.find('\n')? + 1;
1045    let after_open = &slice[after_open_rel..];
1046    let bytes = after_open.as_bytes();
1047    // Trim the trailing closing-marker line.
1048    let mut i = bytes.len();
1049    if i > 0 && bytes[i - 1] == b'\n' {
1050        i -= 1;
1051    }
1052    while i > 0 && matches!(bytes[i - 1], b' ' | b'\t' | b'\r') {
1053        i -= 1;
1054    }
1055    while i > 0 && bytes[i - 1] == b'-' {
1056        i -= 1;
1057    }
1058    if i > 0 && bytes[i - 1] == b'\n' {
1059        i -= 1;
1060    }
1061    let start = span.range().start + after_open_rel;
1062    Some(start..start + i)
1063}
1064
1065/// Extract the body content of a doc-block token from its source span.
1066/// Strips the leading and trailing `---` marker lines and returns the body
1067/// verbatim. If every non-empty content line begins with the same horizontal
1068/// whitespace prefix (e.g., because the doc block sits inside a brace-form
1069/// commons body), that common prefix is removed so the body reads naturally
1070/// when emitted as JSDoc.
1071pub fn doc_block_content(source: &str, span: Span) -> String {
1072    let Some(range) = doc_block_body_range(source, span) else {
1073        return String::new();
1074    };
1075    let body = &source[range];
1076
1077    // Compute the common leading-whitespace prefix across all non-empty lines
1078    // and strip it. This lets writers indent the doc block alongside the
1079    // declaration it documents without bleeding the indent into the JSDoc.
1080    let common: Option<usize> = body
1081        .lines()
1082        .filter(|l| !l.trim().is_empty())
1083        .map(|l| l.bytes().take_while(|&b| b == b' ' || b == b'\t').count())
1084        .min();
1085    let strip = common.unwrap_or(0);
1086    if strip == 0 {
1087        return body.to_string();
1088    }
1089    let mut out = String::with_capacity(body.len());
1090    let mut first = true;
1091    for line in body.lines() {
1092        if !first {
1093            out.push('\n');
1094        }
1095        first = false;
1096        if line.trim().is_empty() {
1097            // Preserve blank lines.
1098            continue;
1099        }
1100        let leading: usize = line
1101            .bytes()
1102            .take_while(|&b| b == b' ' || b == b'\t')
1103            .count();
1104        let drop = strip.min(leading);
1105        out.push_str(&line[drop..]);
1106    }
1107    out
1108}
1109
1110/// Extract the body of a `Comment` trivia token: everything after the
1111/// leading `--` marker, preserving its inline whitespace verbatim. Used by
1112/// the parser when attaching comments to declarations.
1113pub fn comment_body(source: &str, span: Span) -> &str {
1114    let slice = &source[span.range()];
1115    // Strip leading "--" if present (defensive — the lexer always emits
1116    // Comment tokens whose span begins with `--`).
1117    slice.strip_prefix("--").unwrap_or(slice)
1118}
1119
1120/// Returns true if there is a blank line (a line containing only whitespace)
1121/// in `source` strictly between byte offsets `from` (inclusive) and `to`
1122/// (exclusive). Used by the parser to detect orphan doc blocks.
1123///
1124/// A doc-block token's span ends just past the closing-marker line's
1125/// terminating newline. So if the next declaration begins on the immediately
1126/// following line, the substring between contains no newline (only optional
1127/// indentation). Any newline in the substring therefore implies at least one
1128/// entirely-blank line separating the doc from the declaration.
1129pub fn has_blank_line_between(source: &str, from: usize, to: usize) -> bool {
1130    if to <= from {
1131        return false;
1132    }
1133    let bytes = source.as_bytes();
1134    let mut i = from;
1135    while i < to {
1136        if bytes[i] == b'\n' {
1137            return true;
1138        }
1139        if !matches!(bytes[i], b' ' | b'\t' | b'\r') {
1140            return false;
1141        }
1142        i += 1;
1143    }
1144    false
1145}
1146
1147#[cfg(test)]
1148mod tests {
1149    use super::*;
1150
1151    fn kinds(source: &str) -> Vec<TokenKind> {
1152        tokenize(source)
1153            .unwrap()
1154            .into_iter()
1155            .map(|t| t.kind)
1156            .collect()
1157    }
1158
1159    #[test]
1160    fn keywords_and_idents() {
1161        use TokenKind::*;
1162        assert_eq!(
1163            kinds("commons type fn where true false Int String Bool foo bar"),
1164            vec![
1165                Commons, Type, Fn, Where, True, False, Int, String, Bool, Ident, Ident
1166            ],
1167        );
1168        // #548: `and` is no longer a keyword — it lexes as an ordinary identifier.
1169        assert_eq!(kinds("and"), vec![Ident]);
1170    }
1171
1172    #[test]
1173    fn deeply_nested_interpolation_is_bounded_not_overflowed() {
1174        // `"\("\("\(…` mutually recurses scan_str <-> scan_hole, one frame per
1175        // level, and an unbounded scanner overflows `tokenize` and aborts the
1176        // process (#713). Well past the limit it must return a bounded-depth
1177        // diagnostic instead. The holes are left open so the depth guard, not a
1178        // later `)`, stops the scan.
1179        let depth = crate::MAX_NESTING_DEPTH + 8;
1180        let src = format!("\"{}", "\\(\"".repeat(depth));
1181        let err = tokenize(&src).unwrap_err();
1182        assert_eq!(err.category, "bynk.lex.interpolation_too_deep");
1183    }
1184
1185    #[test]
1186    fn integer_and_string_literals() {
1187        use TokenKind::*;
1188        assert_eq!(
1189            kinds(r#"0 42 "hello" "with\nescape""#),
1190            vec![IntLit, IntLit, StrLit, StrLit]
1191        );
1192    }
1193
1194    #[test]
1195    fn operators() {
1196        use TokenKind::*;
1197        assert_eq!(
1198            kinds("-> == != <= >= && || + - * / ! = < > ( ) { } [ ] , : . @"),
1199            vec![
1200                Arrow, EqEq, BangEq, LtEq, GtEq, AmpAmp, PipePipe, Plus, Minus, Star, Slash, Bang,
1201                Eq, Lt, Gt, LParen, RParen, LBrace, RBrace, LBracket, RBracket, Comma, Colon, Dot,
1202                At,
1203            ],
1204        );
1205    }
1206
1207    #[test]
1208    fn dot_family_maximal_munch() {
1209        // Events track slice 1 (spine #936): `..` must lex as one `DotDot`
1210        // token, not two `Dot`s — a real token keeps agreement with
1211        // tree-sitter (which declares `".."` as one literal), so a
1212        // whitespace-split `. .` cannot silently parse where a real `..`
1213        // is required. Also confirms `...`/`..`/`.` don't shadow each other
1214        // regardless of declaration order (logos maximal-munch).
1215        use TokenKind::*;
1216        assert_eq!(
1217            kinds("a .. b ... c . d . ."),
1218            vec![Ident, DotDot, Ident, DotDotDot, Ident, Dot, Ident, Dot, Dot,],
1219        );
1220    }
1221
1222    #[test]
1223    fn line_comments_emitted_as_trivia() {
1224        // v1.1: line comments are preserved as Comment tokens so the
1225        // formatter can attach and re-emit them.
1226        use TokenKind::*;
1227        let src = "-- a comment\ntype X = Int -- trailing\n";
1228        assert_eq!(kinds(src), vec![Comment, Type, Ident, Eq, Int, Comment],);
1229    }
1230
1231    #[test]
1232    fn comment_body_extracts_text_after_marker() {
1233        let toks = tokenize("-- hello world\n").unwrap();
1234        assert_eq!(toks.len(), 1);
1235        assert_eq!(toks[0].kind, TokenKind::Comment);
1236        assert_eq!(
1237            comment_body("-- hello world\n", toks[0].span),
1238            " hello world"
1239        );
1240    }
1241
1242    #[test]
1243    fn comment_does_not_consume_newline() {
1244        // Two adjacent comment lines should produce two distinct tokens
1245        // — the newline between them is not part of either comment's span.
1246        let toks = tokenize("-- one\n-- two\n").unwrap();
1247        assert_eq!(toks.len(), 2);
1248        assert!(toks.iter().all(|t| t.kind == TokenKind::Comment));
1249    }
1250
1251    #[test]
1252    fn dashdash_opens_a_comment_only_when_whitespace_preceded() {
1253        // #548: a `--` opens a comment at the start of input, or when preceded by
1254        // whitespace/line-start. Adjacent to a preceding token it is *not* a
1255        // comment — `a--b` lexes as `a - -b`, never a swallowed line comment.
1256        use TokenKind::*;
1257        assert_eq!(kinds("a--b"), vec![Ident, Minus, Minus, Ident]);
1258        // A trailing decrement-looking `x--` is two operators, not a comment
1259        // that eats the rest of the line — including at end-of-input with no
1260        // trailing newline (the `pos + 1 < len` guard still holds for `x--`).
1261        assert_eq!(kinds("x--\ny"), vec![Ident, Minus, Minus, Ident]);
1262        assert_eq!(kinds("x--"), vec![Ident, Minus, Minus]);
1263        // Whitespace-preceded and start-of-input `--` are still comments.
1264        assert_eq!(kinds("a -- c"), vec![Ident, Comment]);
1265        assert_eq!(kinds("-- c"), vec![Comment]);
1266        // Start of a fresh line (newline-preceded) is a comment.
1267        assert_eq!(kinds("a\n-- c"), vec![Ident, Comment]);
1268        // The comment/doc-block asymmetry: `--` needs only whitespace before it,
1269        // so a mid-line `a ---b` is a *comment* (the leading `-` of the three is
1270        // whitespace-preceded); a `---` doc-block additionally needs line-start,
1271        // which `a ---b` is not.
1272        assert_eq!(kinds("a ---b"), vec![Ident, Comment]);
1273        // A single `-` between terms is unaffected.
1274        assert_eq!(kinds("a - b"), vec![Ident, Minus, Ident]);
1275    }
1276
1277    #[test]
1278    fn unterminated_string_is_error() {
1279        let err = tokenize("\"oops\n").unwrap_err();
1280        assert_eq!(err.category, "bynk.lex.unterminated_string");
1281    }
1282
1283    #[test]
1284    fn integer_overflow_is_error() {
1285        let err = tokenize("99999999999999999999").unwrap_err();
1286        assert_eq!(err.category, "bynk.lex.integer_overflow");
1287    }
1288
1289    #[test]
1290    fn an_int_literal_must_be_a_safe_integer() {
1291        // #1657: the `Int` range is ±(2^53 − 1). The largest safe integer lexes;
1292        // one more, and i64-range values above it, do not.
1293        assert!(tokenize("9007199254740991").is_ok());
1294        assert!(tokenize("9_007_199_254_740_991").is_ok());
1295        for over in [
1296            "9007199254740992",
1297            "9007199254740993",
1298            "9223372036854775807",
1299        ] {
1300            let err = tokenize(over).unwrap_err();
1301            assert_eq!(err.category, "bynk.lex.integer_overflow", "{over}");
1302        }
1303    }
1304
1305    #[test]
1306    fn digit_separators_lex_as_one_number() {
1307        use TokenKind::*;
1308        // v0.142 (ADR 0166): `_` between digit groups keeps the literal a single
1309        // token for both Int and Float.
1310        assert_eq!(kinds("1_048_576"), vec![IntLit]);
1311        assert_eq!(kinds("1_000.500_5"), vec![FloatLit]);
1312        assert_eq!(kinds("1_000e1_0"), vec![FloatLit]);
1313        // A separator-carrying literal that is in range still lexes (the value is
1314        // validated after stripping the separators).
1315        assert!(tokenize("9_007_199_254_740_991").is_ok());
1316        // Overflow is still caught on the separator-free value.
1317        let err = tokenize("9_999_999_999_999_999_999_9").unwrap_err();
1318        assert_eq!(err.category, "bynk.lex.integer_overflow");
1319    }
1320
1321    #[test]
1322    fn strip_digit_separators_removes_underscores() {
1323        assert_eq!(strip_digit_separators("1_048_576"), "1048576");
1324        assert_eq!(strip_digit_separators("42"), "42");
1325    }
1326
1327    #[test]
1328    fn unexpected_character_is_error() {
1329        let err = tokenize("type X = Int $").unwrap_err();
1330        assert_eq!(err.category, "bynk.lex.unexpected_character");
1331    }
1332
1333    #[test]
1334    fn v0_1_keywords() {
1335        use TokenKind::*;
1336        assert_eq!(
1337            kinds("let if else Ok Err Result ValidationError"),
1338            vec![Let, If, Else, Ok, Err, Result, ValidationError],
1339        );
1340    }
1341
1342    #[test]
1343    fn question_token() {
1344        use TokenKind::*;
1345        assert_eq!(kinds("x?"), vec![Ident, Question]);
1346    }
1347
1348    #[test]
1349    fn v0_2_keywords() {
1350        use TokenKind::*;
1351        assert_eq!(
1352            kinds("enum match Option record self Some None is"),
1353            vec![Enum, Match, Option, Record, Self_, Some, None, Is],
1354        );
1355    }
1356
1357    #[test]
1358    fn pipe_and_pipe_pipe_disambiguated() {
1359        use TokenKind::*;
1360        assert_eq!(kinds("| || |"), vec![Pipe, PipePipe, Pipe]);
1361    }
1362
1363    #[test]
1364    fn v0_7_keywords() {
1365        use TokenKind::*;
1366        assert_eq!(kinds("expect suite case"), vec![Expect, Suite, Case],);
1367        // v0.118: `mocks` and `wires` are retired — plain identifiers now.
1368        assert_eq!(kinds("mocks wires"), vec![Ident, Ident]);
1369    }
1370
1371    #[test]
1372    fn fat_arrow_and_underscore() {
1373        use TokenKind::*;
1374        assert_eq!(kinds("_ =>"), vec![Underscore, FatArrow]);
1375    }
1376
1377    // -- v0.43 string interpolation --
1378
1379    #[test]
1380    fn interp_string_is_one_token() {
1381        use TokenKind::*;
1382        assert_eq!(kinds(r#""Hello, \(name)!""#), vec![InterpStr]);
1383        // A plain string (no hole) stays a `StrLit`, via the logos path.
1384        assert_eq!(kinds(r#""Hello, world""#), vec![StrLit]);
1385    }
1386
1387    #[test]
1388    fn interp_balances_nested_parens_and_strings() {
1389        use TokenKind::*;
1390        // The `)` inside `f(x)` must not close the hole early.
1391        assert_eq!(kinds(r#""= \(f(x))""#), vec![InterpStr]);
1392        // A `)` inside a nested string inside the hole is also ignored.
1393        assert_eq!(kinds(r#""= \(label(")"))""#), vec![InterpStr]);
1394        // A nested interpolated string inside a hole.
1395        assert_eq!(kinds(r#""out \("in \(x)")""#), vec![InterpStr]);
1396    }
1397
1398    // Issue #473: hole-expanding tokenisation makes identifiers inside `\(…)`
1399    // visible to the LSP's token-based cursor resolution.
1400    #[test]
1401    fn expanding_holes_exposes_hole_identifiers() {
1402        use TokenKind::*;
1403        let expand = |src: &str| {
1404            tokenize_expanding_holes(src)
1405                .unwrap()
1406                .into_iter()
1407                .map(|t| t.kind)
1408                .collect::<Vec<_>>()
1409        };
1410        // The opaque `InterpStr` is replaced by its hole's tokens; the chunk
1411        // text (`Hello, ` / `!`) carries none.
1412        assert_eq!(expand(r#""Hello, \(name)!""#), vec![Ident]);
1413        // A call hole exposes every token of the call expression.
1414        assert_eq!(expand(r#""= \(f(x))""#), vec![Ident, LParen, Ident, RParen]);
1415        // Nested interpolation recurses to the innermost hole's identifier.
1416        assert_eq!(expand(r#""out \("in \(x)")""#), vec![Ident]);
1417        // A plain (hole-free) string is untouched.
1418        assert_eq!(expand(r#""Hello, world""#), vec![StrLit]);
1419    }
1420
1421    #[test]
1422    fn expanding_holes_rebases_spans_to_absolute() {
1423        let src = r#""Hello, \(name)!""#;
1424        let toks = tokenize_expanding_holes(src).unwrap();
1425        let ident = toks
1426            .iter()
1427            .find(|t| t.kind == TokenKind::Ident)
1428            .expect("the hole identifier is exposed");
1429        // The span points at `name` in the original source, not a hole-local 0.
1430        assert_eq!(&src[ident.span.range()], "name");
1431        assert_eq!(ident.span.start, src.find("name").unwrap());
1432    }
1433
1434    #[test]
1435    fn escaped_open_paren_is_not_a_hole() {
1436        use TokenKind::*;
1437        // `\\(` is a literal backslash followed by `(` — no hole, so the
1438        // string lexes as a plain `StrLit` on the logos path.
1439        assert_eq!(kinds(r#""a \\(b) c""#), vec![StrLit]);
1440    }
1441
1442    #[test]
1443    fn unterminated_hole_is_an_error() {
1444        // The hole runs to end of line without its closing `)`.
1445        let err = tokenize("\"value \\(x + 1\n\"").unwrap_err();
1446        assert_eq!(err.category, "bynk.lex.unterminated_interpolation");
1447    }
1448
1449    #[test]
1450    fn unterminated_interp_string_is_an_error() {
1451        // A hole closes but the string never does (newline before the `"`).
1452        let err = tokenize("\"value \\(x) more\n").unwrap_err();
1453        assert_eq!(err.category, "bynk.lex.unterminated_string");
1454    }
1455
1456    #[test]
1457    fn bad_escape_in_interp_string_is_an_error() {
1458        let err = tokenize(r#""a \q \(x)""#).unwrap_err();
1459        assert_eq!(err.category, "bynk.lex.bad_escape");
1460    }
1461
1462    fn doc_block_span(source: &str) -> Span {
1463        tokenize(source)
1464            .unwrap()
1465            .into_iter()
1466            .find(|t| t.kind == TokenKind::DocBlock)
1467            .expect("a DocBlock token")
1468            .span
1469    }
1470
1471    #[test]
1472    fn doc_block_body_range_slices_to_the_same_bytes_doc_block_content_would_strip() {
1473        let src = "---\nHello there.\n---\nfn f() -> Int = 1\n";
1474        let span = doc_block_span(src);
1475        let range = doc_block_body_range(src, span).unwrap();
1476        assert_eq!(&src[range], "Hello there.");
1477    }
1478
1479    #[test]
1480    fn doc_block_body_range_is_offset_preserving_unlike_doc_block_content() {
1481        // A content line indented relative to its (unindented) markers:
1482        // doc_block_content strips the common indent (not offset-preserving),
1483        // doc_block_body_range does not — its slice still contains the raw
1484        // indentation, so span-based callers can map a position in the raw
1485        // body straight back to `src`.
1486        let src = "---\n  See [Foo].\n---\nfn f() -> Int = 1\n";
1487        let span = doc_block_span(src);
1488        let range = doc_block_body_range(src, span).unwrap();
1489        assert_eq!(&src[range.clone()], "  See [Foo].");
1490        assert_eq!(doc_block_content(src, span), "See [Foo].");
1491        // The raw range still locates `[Foo]` correctly within `src`.
1492        let bracket_rel = src[range.clone()].find('[').unwrap();
1493        assert_eq!(
1494            &src[range.start + bracket_rel..range.start + bracket_rel + 5],
1495            "[Foo]"
1496        );
1497    }
1498
1499    #[test]
1500    fn doc_block_content_and_body_range_agree_on_empty_body() {
1501        let src = "---\n---\nfn f() -> Int = 1\n";
1502        let span = doc_block_span(src);
1503        let range = doc_block_body_range(src, span).unwrap();
1504        assert_eq!(&src[range], "");
1505        assert_eq!(doc_block_content(src, span), "");
1506    }
1507}