bynk_syntax/lexer.rs
1//! Lexer for Bynk v0.
2//!
3//! Token kinds correspond to the terminals defined in the grammar (spec §3
4//! and §4). Whitespace is skipped; line comments are emitted as `Comment`
5//! tokens so the formatter can preserve them through round-trips (v1.1 LSP
6//! spec §3.5). Doc blocks (`---`) are emitted as `DocBlock` tokens, lexed
7//! outside of logos (see [`tokenize`]).
8
9use logos::Logos;
10
11use crate::error::CompileError;
12use crate::span::Span;
13
14/// v0.142 (ADR 0166): strip `_` digit separators from a numeric literal's lexeme
15/// before it is parsed into a value. The lexer's `IntLit`/`FloatLit` regexes only
16/// admit an `_` between two digit groups, so removing every `_` yields a plain
17/// digit string; the separators are purely visual. Allocates only when the
18/// literal actually carries a separator (the common case does not).
19pub(crate) fn strip_digit_separators(lexeme: &str) -> std::borrow::Cow<'_, str> {
20 if lexeme.as_bytes().contains(&b'_') {
21 std::borrow::Cow::Owned(lexeme.replace('_', ""))
22 } else {
23 std::borrow::Cow::Borrowed(lexeme)
24 }
25}
26
27/// Token kinds. Discriminants without payload data; the lexeme is recovered
28/// from the source string via the token's [`Span`].
29///
30/// Note: `--` line comments and `---` doc block markers are handled outside
31/// logos (see [`tokenize`]), because doc blocks are delimited by `---` lines
32/// containing only the marker and may span multiple source lines.
33#[derive(Logos, Debug, Clone, Copy, PartialEq, Eq)]
34#[logos(skip r"[ \t\r\n]+")]
35pub enum TokenKind {
36 // Keywords
37 #[token("commons")]
38 Commons,
39 #[token("type")]
40 Type,
41 // Events track, slice 0 (spine #936): `event Name = { fields }` — a
42 // context-only item declaring a typed fact. RESERVED_CONTEXTUAL, not
43 // hard, per ADR 0272's `messages` postmortem (a hard keyword broke the
44 // dotted-name case its own worked example used); `event` is at least as
45 // likely a parameter/local name (`on event(event: T)`).
46 #[token("event")]
47 Event,
48 #[token("fn")]
49 Fn,
50 #[token("where")]
51 Where,
52 // #548: the `and` keyword was retired — refinement predicates now join with
53 // `&&` (the one conjunction spelling). `and` is a free identifier again.
54 #[token("true")]
55 True,
56 #[token("false")]
57 False,
58 #[token("Int")]
59 Int,
60 #[token("String")]
61 String,
62 #[token("Bool")]
63 Bool,
64 // v0.21 keyword
65 #[token("Float")]
66 Float,
67 // v0.86 keyword (ADR 0112): the `Duration` base type.
68 #[token("Duration")]
69 Duration,
70 // v0.90 keyword (ADR 0114): the `Instant` base type.
71 #[token("Instant")]
72 Instant,
73 // v0.110 keyword (ADR 0142): the `Bytes` base type.
74 #[token("Bytes")]
75 Bytes,
76 // v0.1 keywords
77 #[token("let")]
78 Let,
79 #[token("if")]
80 If,
81 #[token("else")]
82 Else,
83 #[token("Ok")]
84 Ok,
85 #[token("Err")]
86 Err,
87 #[token("Result")]
88 Result,
89 #[token("ValidationError")]
90 ValidationError,
91 // v0.22b keyword
92 #[token("JsonError")]
93 JsonError,
94 // v0.2 keywords
95 #[token("enum")]
96 Enum,
97 #[token("match")]
98 Match,
99 #[token("Option")]
100 Option,
101 #[token("record")]
102 Record,
103 #[token("self")]
104 Self_,
105 #[token("Some")]
106 Some,
107 #[token("None")]
108 None,
109 #[token("is")]
110 Is,
111 // v0.3 keywords
112 #[token("opaque")]
113 Opaque,
114 #[token("uses")]
115 Uses,
116 // v0.4 keywords
117 #[token("context")]
118 Context,
119 #[token("consumes")]
120 Consumes,
121 #[token("exports")]
122 Exports,
123 #[token("transparent")]
124 Transparent,
125 // v0.6 keywords
126 #[token("as")]
127 As,
128 // v0.7 keywords (v0.112: `assert`→`expect`, `test`→`suite`/`case`;
129 // v0.118: `mocks` retired — test doubles are stubs at a seam; the stub form
130 // moved off the punned `provides` keyword to its own `stub` keyword in the
131 // keyword-hygiene batch, #548)
132 #[token("expect")]
133 Expect,
134 #[token("suite")]
135 Suite,
136 #[token("case")]
137 Case,
138 // Keyword-hygiene batch (#548): the test-scope stub `stub Cap.op(…) <rhs>`,
139 // formerly the third pun on `provides`. `provides` now heads only a provider
140 // declaration / external provider.
141 #[token("stub")]
142 Stub,
143 // v0.114 keyword — generative tests (testing track slice 2). `for` and `all`
144 // are deliberately *not* keywords: `all` is a list combinator (`all(xs, p)`)
145 // and must stay a usable identifier. The `for all` binder is parsed
146 // contextually (two identifiers) inside a `property` body instead.
147 #[token("property")]
148 Property,
149 // v0.17 keywords
150 #[token("adapter")]
151 Adapter,
152 #[token("binding")]
153 Binding,
154 // v0.5 keywords
155 #[token("agent")]
156 Agent,
157 #[token("capability")]
158 Capability,
159 #[token("Effect")]
160 Effect,
161 // v0.146 keyword (ADR 0170): `do e` — an effect-performing expression
162 // statement (the binder-free `let _ <- e` for a unit effect).
163 #[token("do")]
164 Do,
165 #[token("given")]
166 Given,
167 #[token("on")]
168 On,
169 // v0.9 keyword
170 #[token("http")]
171 Http,
172 // v0.10a keyword
173 #[token("cron")]
174 Cron,
175 // v0.10b keyword
176 #[token("queue")]
177 Queue,
178 // v0.44 keywords: `from` heads a service's protocol clause; `protocol` is
179 // reserved (protocols are a closed, compiler-known set — no declaration kind).
180 #[token("from")]
181 From,
182 #[token("protocol")]
183 Protocol,
184 #[token("provides")]
185 Provides,
186 #[token("service")]
187 Service,
188 // v0.45 keywords: `actor` heads a boundary-contract declaration; `by`
189 // heads a handler's actor clause.
190 #[token("actor")]
191 Actor,
192 #[token("by")]
193 By,
194 // v0.80 keywords: `invariant` heads an agent invariant declaration; `implies`
195 // is the directional logical-implication operator (`P implies Q` ≡ `!P || Q`).
196 #[token("invariant")]
197 Invariant,
198 #[token("implies")]
199 Implies,
200 // v0.115 keywords — function contracts (testing track slice 3). `requires`
201 // and `ensures` head a contract clause on a `fn` signature (between the
202 // return type and the body). `result` is deliberately *not* a keyword: it is
203 // the ordinary value name outside a contract, so it stays a usable
204 // identifier; inside an `ensures` predicate it is bound contextually as the
205 // function's return value (parsed by scope, like `for`/`all` in slice 2).
206 // Distinct from ADR 0127's capability `@requires` annotation.
207 #[token("requires")]
208 Requires,
209 #[token("ensures")]
210 Ensures,
211 // v0.116 keyword — step invariants (testing track slice 4). `transition` heads
212 // an agent step-invariant declaration (beside `invariant`), a predicate over
213 // the pre- and post-commit state pair. `old` and `new` are deliberately *not*
214 // keywords: they stay ordinary value names outside a `transition`, and inside a
215 // `transition` predicate they are bound contextually to the old/new state
216 // records (parsed by scope, like `result` in an `ensures`).
217 #[token("transition")]
218 Transition,
219 // message-bundles track, slice 1: `messages <tag> { "code" => "template" }`
220 // — a commons item declaring one locale's message bundle.
221 #[token("messages")]
222 Messages,
223 /// `...` — used in record-spread expressions (v0.5).
224 #[token("...")]
225 DotDotDot,
226 /// `..` — the "rest of the fields" marker on an events subscription
227 /// pattern (Events track slice 1, spine #936): `from Events(E { region:
228 /// Region.Domestic, .. })`. A genuine token, not two adjacent `Dot`s —
229 /// logos maximal-munches `...`/`..`/`.` correctly once all three are
230 /// registered, and a real token keeps this in agreement with
231 /// tree-sitter's grammar (which declares `".."` as one literal), so the
232 /// two parsers cannot diverge on a whitespace-split `. .` the way ADR
233 /// 0253 D4 found a leaking `where`-check divergence once before.
234 #[token("..")]
235 DotDot,
236 /// `<-` — Effect bind operator (v0.5).
237 #[token("<-")]
238 LArrow,
239 /// `~>` — asynchronous fire-and-forget send marker (v0.79). A leading
240 /// statement marker, never on the RHS of a `let`; distinct from `<-` so the
241 /// call site shows whether the caller waits.
242 #[token("~>")]
243 TildeArrow,
244 /// `:=` — Cell write (v0.81, storage track). A handler statement
245 /// `cell := expr`; distinct from `=` (binding) and `:` (annotation). Longer
246 /// than `:`/`=` so logos matches it as one token.
247 #[token(":=")]
248 ColonEq,
249
250 /// A documentation block: `---` line ... `---` line. The token's span
251 /// covers the full block including both `---` markers. The body content
252 /// is recovered from the source via the span (see [`doc_block_content`]).
253 /// Inserted by [`tokenize`]; not lexed by logos directly.
254 DocBlock,
255
256 /// A line comment: `-- ...` running to end of line. The span starts at
257 /// the `--` marker and runs through the last character before the
258 /// terminating newline (exclusive). The trivia body (the text after the
259 /// `--` marker) is recovered from the source via the span. Inserted by
260 /// [`tokenize`]; not lexed by logos directly so it cannot be mistaken
261 /// for an `--` operator sequence.
262 Comment,
263
264 // Identifier
265 #[regex(r"[A-Za-z][A-Za-z0-9_]*")]
266 Ident,
267
268 // Literals. v0.142 (ADR 0166): an `_` digit separator may appear between
269 // digits (`1_048_576`) — never leading, trailing, or doubled (each `_` must
270 // sit between two digit groups). The separators are stripped before the value
271 // is parsed; they are purely visual.
272 #[regex(r"[0-9]+(_[0-9]+)*")]
273 IntLit,
274 // A float literal: fraction with a digit on both sides of the `.`, an
275 // exponent, or both (v0.21 §3). `1.` and `.5` are NOT float literals —
276 // the digit-both-sides rule keeps `2.5.round()` / `1.toFloat()` lexing
277 // as method calls on numeric literals. Digit separators (v0.142) may appear
278 // in any digit group, including the exponent.
279 #[regex(
280 r"[0-9]+(_[0-9]+)*\.[0-9]+(_[0-9]+)*([eE][+-]?[0-9]+(_[0-9]+)*)?|[0-9]+(_[0-9]+)*[eE][+-]?[0-9]+(_[0-9]+)*"
281 )]
282 FloatLit,
283 // A double-quoted string with simple escapes. The body excludes the closing
284 // quote; we accept any non-quote/non-backslash/non-newline char, or a
285 // backslash followed by one of the four allowed escapes.
286 #[regex(r#""([^"\\\n]|\\[nt"\\])*""#)]
287 StrLit,
288 // An interpolated string `"… \(expr) …"` (v0.43). Hand-scanned in
289 // `tokenize` (logos cannot balance the holes' parens), never produced by
290 // the logos lexer — like [`TokenKind::DocBlock`]/[`TokenKind::Comment`].
291 // The span covers the whole `"…"`; the parser splits chunks from holes.
292 InterpStr,
293
294 // Multi-char operators
295 #[token("->")]
296 Arrow,
297 #[token("==")]
298 EqEq,
299 #[token("!=")]
300 BangEq,
301 #[token("<=")]
302 LtEq,
303 #[token(">=")]
304 GtEq,
305 #[token("&&")]
306 AmpAmp,
307 #[token("||")]
308 PipePipe,
309
310 // Single-char operators
311 #[token("+")]
312 Plus,
313 #[token("-")]
314 Minus,
315 #[token("*")]
316 Star,
317 #[token("/")]
318 Slash,
319 #[token("!")]
320 Bang,
321 #[token("=")]
322 Eq,
323 #[token("<")]
324 Lt,
325 #[token(">")]
326 Gt,
327 // v0.1 postfix operator
328 #[token("?")]
329 Question,
330 // v0.2 match-arm arrow
331 #[token("=>")]
332 FatArrow,
333 // v0.2 wildcard pattern (also valid as identifier start; the lexer
334 // prefers identifier for any longer match, so `_foo` is still Ident).
335 #[token("_")]
336 Underscore,
337 // v0.2 sum-type variant separator (also used as future bitwise OR);
338 // single `|` distinct from `||`.
339 #[token("|")]
340 Pipe,
341 /// `@` — storage-annotation marker (v0.85, storage track; ADR 0111). Leads a
342 /// `@name(args)` annotation on a `store` field (`@ttl(…)`/`@indexed(…)`); it
343 /// appears only in store-field-declaration position, never as an expression
344 /// operator.
345 #[token("@")]
346 At,
347
348 // Punctuation
349 #[token("(")]
350 LParen,
351 #[token(")")]
352 RParen,
353 #[token("{")]
354 LBrace,
355 #[token("}")]
356 RBrace,
357 #[token("[")]
358 LBracket,
359 #[token("]")]
360 RBracket,
361 #[token(",")]
362 Comma,
363 #[token(":")]
364 Colon,
365 #[token(".")]
366 Dot,
367}
368
369impl TokenKind {
370 /// Human-readable display name for diagnostics.
371 pub fn describe(self) -> &'static str {
372 use TokenKind::*;
373 match self {
374 Commons => "`commons`",
375 Type => "`type`",
376 Event => "`event`",
377 Fn => "`fn`",
378 Where => "`where`",
379 True => "`true`",
380 False => "`false`",
381 Int => "`Int`",
382 String => "`String`",
383 Bool => "`Bool`",
384 Float => "`Float`",
385 Duration => "`Duration`",
386 Instant => "`Instant`",
387 Bytes => "`Bytes`",
388 Let => "`let`",
389 If => "`if`",
390 Else => "`else`",
391 Ok => "`Ok`",
392 Err => "`Err`",
393 Result => "`Result`",
394 ValidationError => "`ValidationError`",
395 JsonError => "`JsonError`",
396 Enum => "`enum`",
397 Match => "`match`",
398 Option => "`Option`",
399 Record => "`record`",
400 Self_ => "`self`",
401 Some => "`Some`",
402 None => "`None`",
403 Is => "`is`",
404 Opaque => "`opaque`",
405 Uses => "`uses`",
406 Context => "`context`",
407 Consumes => "`consumes`",
408 Exports => "`exports`",
409 Transparent => "`transparent`",
410 As => "`as`",
411 Expect => "`expect`",
412 Suite => "`suite`",
413 Case => "`case`",
414 Property => "`property`",
415 Adapter => "`adapter`",
416 Binding => "`binding`",
417 Agent => "`agent`",
418 Capability => "`capability`",
419 Effect => "`Effect`",
420 Do => "`do`",
421 Given => "`given`",
422 On => "`on`",
423 Http => "`http`",
424 Cron => "`cron`",
425 Queue => "`queue`",
426 From => "`from`",
427 Protocol => "`protocol`",
428 Provides => "`provides`",
429 Stub => "`stub`",
430 Service => "`service`",
431 Actor => "`actor`",
432 By => "`by`",
433 Invariant => "`invariant`",
434 Implies => "`implies`",
435 Requires => "`requires`",
436 Ensures => "`ensures`",
437 Transition => "`transition`",
438 Messages => "`messages`",
439 ColonEq => "`:=`",
440 DotDotDot => "`...`",
441 DotDot => "`..`",
442 LArrow => "`<-`",
443 TildeArrow => "`~>`",
444 DocBlock => "documentation block",
445 Comment => "line comment",
446 Ident => "identifier",
447 IntLit => "integer literal",
448 FloatLit => "float literal",
449 StrLit => "string literal",
450 InterpStr => "interpolated string",
451 Arrow => "`->`",
452 EqEq => "`==`",
453 BangEq => "`!=`",
454 LtEq => "`<=`",
455 GtEq => "`>=`",
456 AmpAmp => "`&&`",
457 PipePipe => "`||`",
458 Plus => "`+`",
459 Minus => "`-`",
460 Star => "`*`",
461 Slash => "`/`",
462 Bang => "`!`",
463 Eq => "`=`",
464 Lt => "`<`",
465 Gt => "`>`",
466 Question => "`?`",
467 FatArrow => "`=>`",
468 Underscore => "`_`",
469 Pipe => "`|`",
470 At => "`@`",
471 LParen => "`(`",
472 RParen => "`)`",
473 LBrace => "`{`",
474 RBrace => "`}`",
475 LBracket => "`[`",
476 RBracket => "`]`",
477 Comma => "`,`",
478 Colon => "`:`",
479 Dot => "`.`",
480 }
481 }
482}
483
484/// A token plus its source span.
485#[derive(Debug, Clone, Copy)]
486pub struct Token {
487 pub kind: TokenKind,
488 pub span: Span,
489}
490
491/// Tokenise a source string. Returns the full token vector or the first
492/// lexical error.
493///
494/// Doc blocks (`---` ... `---`) and line comments (`-- ...`) are recognised
495/// outside the logos-generated lexer: we scan the source one segment at a
496/// time, dispatching to logos for ordinary tokens between non-token spans.
497pub fn tokenize(source: &str) -> Result<Vec<Token>, CompileError> {
498 let mut tokens = Vec::new();
499 let bytes = source.as_bytes();
500 let mut pos = 0;
501 while pos < bytes.len() {
502 // Detect a `---` doc-block marker at the start of a line (the line may
503 // begin with leading whitespace; the marker itself must be alone on
504 // its line).
505 if let Some(open_end) = doc_block_open_at(source, pos) {
506 // Find the matching closing `---` line.
507 match doc_block_close(source, open_end) {
508 Some((close_start, close_end)) => {
509 let span = Span::new(pos, close_end);
510 tokens.push(Token {
511 kind: TokenKind::DocBlock,
512 span,
513 });
514 let _ = close_start;
515 pos = close_end;
516 continue;
517 }
518 None => {
519 return Err(CompileError::new(
520 "bynk.lex.unclosed_doc_block",
521 Span::new(pos, open_end),
522 "documentation block opened but never closed",
523 )
524 .with_note(
525 "a doc block must be terminated by another `---` on a line by itself",
526 ));
527 }
528 }
529 }
530 // A `--` line comment: emit a `Comment` token covering everything
531 // up to (but not including) the terminating newline. Doc-block
532 // detection above already ruled out a `---` marker at line start
533 // — and once we've consumed past the leading `--`, any further
534 // dashes are part of the comment body. Preserving comments as
535 // trivia tokens lets the parser attach them to declarations so
536 // the formatter can emit them in place (v1.1 LSP spec §3.5).
537 //
538 // #548 (keyword-hygiene batch): a `--` opens a comment only when it is
539 // at the start of input or **preceded by whitespace**. Adjacent to a
540 // preceding token (`a--b`), the `--` is *not* a comment — it lexes as two
541 // `-` operators (`a - -b`), so a subtraction-of-negation is never
542 // silently swallowed as a line comment. This resolves the `a--b`
543 // "comment vs subtraction" ambiguity in favour of subtraction.
544 let comment_eligible = pos == 0 || matches!(bytes[pos - 1], b' ' | b'\t' | b'\r' | b'\n');
545 if comment_eligible && pos + 1 < bytes.len() && bytes[pos] == b'-' && bytes[pos + 1] == b'-'
546 {
547 let start = pos;
548 while pos < bytes.len() && bytes[pos] != b'\n' {
549 pos += 1;
550 }
551 tokens.push(Token {
552 kind: TokenKind::Comment,
553 span: Span::new(start, pos),
554 });
555 continue;
556 }
557 // Skip ordinary whitespace inline (logos handles it too, but we may
558 // be in the middle of the source between specials).
559 if matches!(bytes[pos], b' ' | b'\t' | b'\r' | b'\n') {
560 pos += 1;
561 continue;
562 }
563 // An interpolated string `"… \(expr) …"` (v0.43): only strings that
564 // actually contain a `\(` hole are hand-scanned here; plain strings
565 // fall through to the logos `StrLit` path unchanged. `\(` is an
566 // invalid escape in the logos grammar, so this never re-routes a
567 // currently-valid literal.
568 if bytes[pos] == b'"' && has_interp_hole(bytes, pos) {
569 let end = scan_str(bytes, source, pos, 0)?;
570 tokens.push(Token {
571 kind: TokenKind::InterpStr,
572 span: Span::new(pos, end),
573 });
574 pos = end;
575 continue;
576 }
577 // Otherwise dispatch a single logos token starting at `pos`.
578 let mut lex = TokenKind::lexer(&source[pos..]);
579 let Some(result) = lex.next() else {
580 // No token at this position; treat as unexpected character so
581 // the user sees something useful.
582 let ch = source[pos..].chars().next().unwrap_or('\0');
583 let span = Span::new(pos, pos + ch.len_utf8());
584 return Err(CompileError::new(
585 "bynk.lex.unexpected_character",
586 span,
587 format!("unexpected character `{ch}`"),
588 ));
589 };
590 let local = lex.span();
591 let span: Span = Span::new(pos + local.start, pos + local.end);
592 match result {
593 Ok(kind) => {
594 if kind == TokenKind::IntLit {
595 let slice = &source[span.range()];
596 if strip_digit_separators(slice).parse::<i64>().is_err() {
597 return Err(CompileError::new(
598 "bynk.lex.integer_overflow",
599 span,
600 format!(
601 "integer literal `{slice}` is out of range for a 64-bit signed integer"
602 ),
603 )
604 .with_note("the range is -2^63 to 2^63 - 1"));
605 }
606 }
607 if kind == TokenKind::FloatLit {
608 let slice = &source[span.range()];
609 match strip_digit_separators(slice).parse::<f64>() {
610 Ok(v) if v.is_finite() => {}
611 _ => {
612 return Err(CompileError::new(
613 "bynk.lex.float_literal_overflow",
614 span,
615 format!(
616 "float literal `{slice}` is out of range for a 64-bit float"
617 ),
618 )
619 .with_note(
620 "the literal does not fit a finite IEEE 754 double; \
621 the largest finite value is ~1.8e308",
622 ));
623 }
624 }
625 }
626 tokens.push(Token { kind, span });
627 pos = span.end;
628 }
629 Err(()) => {
630 let slice = &source[span.range()];
631 let ch = slice.chars().next().unwrap_or('\0');
632 let err = if ch == '"' {
633 CompileError::new(
634 "bynk.lex.unterminated_string",
635 span,
636 "unterminated string literal",
637 )
638 .with_note(
639 "string literals must close with `\"` on the same line; \
640 supported escapes are `\\n`, `\\t`, `\\\"`, `\\\\`",
641 )
642 } else {
643 CompileError::new(
644 "bynk.lex.unexpected_character",
645 span,
646 format!("unexpected character `{ch}`"),
647 )
648 };
649 return Err(err);
650 }
651 }
652 }
653 Ok(tokens)
654}
655
656/// Like [`tokenize`], but with every interpolated-string token replaced by the
657/// tokens of its holes — each hole's bytes re-lexed and its token spans rebased
658/// to absolute source positions (the same rebase [`crate::parser`] applies when
659/// parsing a hole), recursing through nested interpolation. Chunk (literal) text
660/// between holes yields no tokens.
661///
662/// An interpolated string lexes to a single opaque `InterpStr` token, so the
663/// LSP's token-based cursor resolution (hover, go-to-definition, references,
664/// semantic tokens) is otherwise blind to identifiers inside `"… \(name) …"`.
665/// Expanding the holes makes those identifiers visible as ordinary `Ident`
666/// tokens with their real spans. (Issue #473.)
667///
668/// On a malformed interpolation (an `InterpStr` whose holes don't split, or a
669/// hole whose bytes don't re-lex) the offending token is kept opaque rather than
670/// dropped, so resolution degrades to the pre-fix behaviour instead of losing
671/// tokens.
672pub fn tokenize_expanding_holes(source: &str) -> Result<Vec<Token>, CompileError> {
673 let mut out = Vec::new();
674 for tok in tokenize(source)? {
675 expand_hole_token(source, tok, &mut out);
676 }
677 Ok(out)
678}
679
680/// Push `tok` onto `out`, expanding it into its holes' tokens if it is an
681/// `InterpStr` (see [`tokenize_expanding_holes`]); otherwise push it as-is.
682fn expand_hole_token(source: &str, tok: Token, out: &mut Vec<Token>) {
683 if tok.kind != TokenKind::InterpStr {
684 out.push(tok);
685 return;
686 }
687 let Ok(segments) = split_interp(source, tok.span) else {
688 out.push(tok); // malformed interpolation — keep the opaque token
689 return;
690 };
691 for segment in segments {
692 let InterpSegment::Hole(hole) = segment else {
693 continue; // chunk text carries no tokens
694 };
695 let Ok(hole_tokens) = tokenize(&source[hole.range()]) else {
696 continue;
697 };
698 for mut t in hole_tokens {
699 // Rebase the hole's local spans to absolute source positions.
700 t.span = Span::new(t.span.start + hole.start, t.span.end + hole.start);
701 expand_hole_token(source, t, out); // recurse for nested interpolation
702 }
703 }
704}
705
706/// Cheap routing pre-scan (v0.43): does the string opening at `start` contain a
707/// `\(` interpolation hole before it closes (or the line ends)? Decides whether
708/// `tokenize` hand-scans the string as an `InterpStr` or defers to logos for a
709/// plain `StrLit`. Deliberately tolerant — a malformed string with a hole is
710/// routed here so the hole-aware scanner produces the precise error.
711fn has_interp_hole(bytes: &[u8], start: usize) -> bool {
712 let mut i = start + 1;
713 while i < bytes.len() {
714 match bytes[i] {
715 b'\n' | b'"' => return false,
716 b'\\' => {
717 if bytes.get(i + 1) == Some(&b'(') {
718 return true;
719 }
720 i += 2;
721 }
722 _ => i += 1,
723 }
724 }
725 false
726}
727
728/// Scan a double-quoted string starting at `start` (the opening `"`), returning
729/// the byte offset just past the closing `"`. Recognises the four simple
730/// escapes plus `\(…)` interpolation holes, whose parens are balanced (and
731/// whose nested strings are skipped) by [`scan_hole`]. (v0.43.)
732fn scan_str(bytes: &[u8], source: &str, start: usize, depth: usize) -> Result<usize, CompileError> {
733 debug_assert_eq!(bytes[start], b'"');
734 if depth > crate::MAX_NESTING_DEPTH {
735 // Anchor on the opening `"` of the string that tipped over the limit.
736 return Err(too_deeply_nested_interpolation(Span::new(start, start + 1)));
737 }
738 let mut i = start + 1;
739 loop {
740 if i >= bytes.len() || bytes[i] == b'\n' {
741 return Err(CompileError::new(
742 "bynk.lex.unterminated_string",
743 Span::new(start, i.min(bytes.len())),
744 "unterminated string literal",
745 )
746 .with_note(
747 "string literals must close with `\"` on the same line; \
748 supported escapes are `\\n`, `\\t`, `\\\"`, `\\\\`, and `\\(…)` interpolation",
749 ));
750 }
751 match bytes[i] {
752 b'"' => return Ok(i + 1),
753 b'\\' => match bytes.get(i + 1) {
754 Some(b'n' | b't' | b'"' | b'\\') => i += 2,
755 Some(b'(') => i = scan_hole(bytes, source, i + 2, depth + 1)?,
756 other => {
757 let shown = other.map(|b| (*b as char).to_string()).unwrap_or_default();
758 // Cover `\` plus the whole offending char, advanced to a char
759 // boundary so the span never splits a multibyte codepoint
760 // (e.g. `\é`) — a fuzz invariant.
761 let mut end = (i + 2).min(bytes.len());
762 while end < source.len() && !source.is_char_boundary(end) {
763 end += 1;
764 }
765 return Err(CompileError::new(
766 "bynk.lex.bad_escape",
767 Span::new(i, end),
768 format!("invalid escape sequence `\\{shown}` in string literal"),
769 )
770 .with_note("supported escapes: \\n \\t \\\" \\\\ \\(…)"));
771 }
772 },
773 // Any other byte advances one position. UTF-8 continuation bytes
774 // are all >= 0x80, so they never collide with the ASCII specials.
775 _ => i += 1,
776 }
777 }
778}
779
780/// Scan an interpolation hole body. `start` points just past the `\(`; returns
781/// the offset just past the matching `)`. Tracks paren depth and skips nested
782/// strings (whose own parens must not close the hole), recursing through
783/// [`scan_str`] so nested interpolation nests correctly. (v0.43.)
784fn scan_hole(
785 bytes: &[u8],
786 source: &str,
787 start: usize,
788 nesting: usize,
789) -> Result<usize, CompileError> {
790 if nesting > crate::MAX_NESTING_DEPTH {
791 // Anchor on the `\(` opener that tipped over the limit; it sits two
792 // bytes before `start` and is pure ASCII, so the span stays on char
793 // boundaries (a fuzz invariant).
794 return Err(too_deeply_nested_interpolation(Span::new(
795 start.saturating_sub(2),
796 start,
797 )));
798 }
799 let mut i = start;
800 let mut depth = 1usize;
801 loop {
802 if i >= bytes.len() || bytes[i] == b'\n' {
803 return Err(CompileError::new(
804 "bynk.lex.unterminated_interpolation",
805 Span::new(start.saturating_sub(2), i.min(bytes.len())),
806 "unterminated interpolation hole",
807 )
808 .with_note(
809 "an interpolation hole `\\(…)` must close with a matching `)` on the same line",
810 ));
811 }
812 match bytes[i] {
813 b'(' => {
814 depth += 1;
815 i += 1;
816 }
817 b')' => {
818 depth -= 1;
819 i += 1;
820 if depth == 0 {
821 return Ok(i);
822 }
823 }
824 b'"' => i = scan_str(bytes, source, i, nesting + 1)?,
825 _ => i += 1,
826 }
827 }
828}
829
830/// The bounded-depth diagnostic for interpolation that nests past
831/// [`crate::MAX_NESTING_DEPTH`]. `\("\("\(…` mutually recurses
832/// [`scan_str`] ↔ [`scan_hole`], one stack frame per level, so an unbounded
833/// scanner overflows and aborts `tokenize` (#713). `span` anchors the report on
834/// the opener that tipped over the limit (the `"` or the `\(`).
835fn too_deeply_nested_interpolation(span: Span) -> CompileError {
836 CompileError::new(
837 "bynk.lex.interpolation_too_deep",
838 span,
839 format!(
840 "string interpolation nests more than {} levels deep",
841 crate::MAX_NESTING_DEPTH
842 ),
843 )
844 .with_note(
845 "deeply nested `\\(…)` interpolation is rejected to keep the lexer from \
846 overflowing its stack and aborting; flatten or split the string",
847 )
848}
849
850/// One segment of a split interpolated string (v0.43): literal text (escapes
851/// resolved) or the absolute source span of a hole's expression (the bytes
852/// between `\(` and its matching `)`). The parser turns the latter into a real
853/// `Expr`; the lexer owns only the scanning.
854pub(crate) enum InterpSegment {
855 Chunk(String),
856 Hole(Span),
857}
858
859/// Split an `InterpStr` token (its `span` covers the whole `"…"`) into chunks
860/// and hole spans. Escapes in the chunks are resolved here (mirroring
861/// [`parse_string_literal`]); holes are returned as spans for the parser to
862/// re-lex and parse as expressions. (v0.43.)
863pub(crate) fn split_interp(source: &str, span: Span) -> Result<Vec<InterpSegment>, CompileError> {
864 let bytes = source.as_bytes();
865 let inner_end = span.end - 1; // the closing `"`
866 let mut segments = Vec::new();
867 let mut chunk = String::new();
868 let mut i = span.start + 1; // past the opening `"`
869 while i < inner_end {
870 match bytes[i] {
871 b'\\' => match bytes[i + 1] {
872 b'n' => {
873 chunk.push('\n');
874 i += 2;
875 }
876 b't' => {
877 chunk.push('\t');
878 i += 2;
879 }
880 b'"' => {
881 chunk.push('"');
882 i += 2;
883 }
884 b'\\' => {
885 chunk.push('\\');
886 i += 2;
887 }
888 b'(' => {
889 if !chunk.is_empty() {
890 segments.push(InterpSegment::Chunk(std::mem::take(&mut chunk)));
891 }
892 let hole_start = i + 2;
893 let after = scan_hole(bytes, source, hole_start, 0)?;
894 // `after` is one past the matching `)`; the hole body is
895 // everything up to that `)`.
896 segments.push(InterpSegment::Hole(Span::new(hole_start, after - 1)));
897 i = after;
898 }
899 // The lexer already validated every escape, so nothing else
900 // can appear here.
901 other => unreachable!("unvalidated escape `\\{}` in InterpStr", other as char),
902 },
903 _ => {
904 let ch = source[i..].chars().next().unwrap();
905 chunk.push(ch);
906 i += ch.len_utf8();
907 }
908 }
909 }
910 if !chunk.is_empty() {
911 segments.push(InterpSegment::Chunk(chunk));
912 }
913 Ok(segments)
914}
915
916/// If a `---` doc-block marker line starts at or shortly after `pos` (which
917/// must be at a line boundary), return the byte offset just past the marker
918/// line (after the terminating newline, or at EOF). The doc-block grammar
919/// requires the marker to be alone on its line; leading horizontal whitespace
920/// is allowed and ignored.
921fn doc_block_open_at(source: &str, pos: usize) -> Option<usize> {
922 let bytes = source.as_bytes();
923 if !at_line_start(source, pos) {
924 return None;
925 }
926 // Skip leading horizontal whitespace.
927 let mut i = pos;
928 while i < bytes.len() && (bytes[i] == b' ' || bytes[i] == b'\t') {
929 i += 1;
930 }
931 if i + 3 > bytes.len() {
932 return None;
933 }
934 if &bytes[i..i + 3] != b"---" {
935 return None;
936 }
937 i += 3;
938 // The marker may have additional trailing dashes (per spec "three or more
939 // consecutive hyphens"). Consume them.
940 while i < bytes.len() && bytes[i] == b'-' {
941 i += 1;
942 }
943 // After the dashes, allow only horizontal whitespace then newline/EOF.
944 while i < bytes.len() && (bytes[i] == b' ' || bytes[i] == b'\t' || bytes[i] == b'\r') {
945 i += 1;
946 }
947 if i == bytes.len() {
948 return Some(i);
949 }
950 if bytes[i] == b'\n' {
951 return Some(i + 1);
952 }
953 None
954}
955
956/// Find the next closing `---` line at or after `pos`. Returns
957/// `(start_of_line, end_of_line)` (`end_of_line` is just past the
958/// terminating newline, or at EOF).
959fn doc_block_close(source: &str, mut pos: usize) -> Option<(usize, usize)> {
960 let bytes = source.as_bytes();
961 while pos < bytes.len() {
962 // Advance pos to the start of a line.
963 let line_start = pos;
964 // Find the end of this line.
965 let mut line_end = line_start;
966 while line_end < bytes.len() && bytes[line_end] != b'\n' {
967 line_end += 1;
968 }
969 // Check this line.
970 if let Some(end) = doc_block_open_at(source, line_start) {
971 return Some((line_start, end));
972 }
973 // Move to the next line.
974 pos = if line_end < bytes.len() {
975 line_end + 1
976 } else {
977 line_end
978 };
979 }
980 None
981}
982
983/// Returns true if byte offset `pos` is at a line start (column 0).
984fn at_line_start(source: &str, pos: usize) -> bool {
985 if pos == 0 {
986 return true;
987 }
988 let bytes = source.as_bytes();
989 bytes[pos - 1] == b'\n'
990}
991
992/// The doc-block body as a byte range into `source` — leading/trailing `---`
993/// marker lines stripped, no further processing (unlike [`doc_block_content`],
994/// which additionally strips a common per-line indent — not offset-preserving).
995/// Callers that need to map a position in the body back to `source` (e.g.
996/// document-link spans) use this instead of re-deriving it from the string
997/// `doc_block_content` returns.
998pub fn doc_block_body_range(source: &str, span: Span) -> Option<std::ops::Range<usize>> {
999 let slice = &source[span.range()];
1000 // Drop the first line (opening marker).
1001 let after_open_rel = slice.find('\n')? + 1;
1002 let after_open = &slice[after_open_rel..];
1003 let bytes = after_open.as_bytes();
1004 // Trim the trailing closing-marker line.
1005 let mut i = bytes.len();
1006 if i > 0 && bytes[i - 1] == b'\n' {
1007 i -= 1;
1008 }
1009 while i > 0 && matches!(bytes[i - 1], b' ' | b'\t' | b'\r') {
1010 i -= 1;
1011 }
1012 while i > 0 && bytes[i - 1] == b'-' {
1013 i -= 1;
1014 }
1015 if i > 0 && bytes[i - 1] == b'\n' {
1016 i -= 1;
1017 }
1018 let start = span.range().start + after_open_rel;
1019 Some(start..start + i)
1020}
1021
1022/// Extract the body content of a doc-block token from its source span.
1023/// Strips the leading and trailing `---` marker lines and returns the body
1024/// verbatim. If every non-empty content line begins with the same horizontal
1025/// whitespace prefix (e.g., because the doc block sits inside a brace-form
1026/// commons body), that common prefix is removed so the body reads naturally
1027/// when emitted as JSDoc.
1028pub fn doc_block_content(source: &str, span: Span) -> String {
1029 let Some(range) = doc_block_body_range(source, span) else {
1030 return String::new();
1031 };
1032 let body = &source[range];
1033
1034 // Compute the common leading-whitespace prefix across all non-empty lines
1035 // and strip it. This lets writers indent the doc block alongside the
1036 // declaration it documents without bleeding the indent into the JSDoc.
1037 let common: Option<usize> = body
1038 .lines()
1039 .filter(|l| !l.trim().is_empty())
1040 .map(|l| l.bytes().take_while(|&b| b == b' ' || b == b'\t').count())
1041 .min();
1042 let strip = common.unwrap_or(0);
1043 if strip == 0 {
1044 return body.to_string();
1045 }
1046 let mut out = String::with_capacity(body.len());
1047 let mut first = true;
1048 for line in body.lines() {
1049 if !first {
1050 out.push('\n');
1051 }
1052 first = false;
1053 if line.trim().is_empty() {
1054 // Preserve blank lines.
1055 continue;
1056 }
1057 let leading: usize = line
1058 .bytes()
1059 .take_while(|&b| b == b' ' || b == b'\t')
1060 .count();
1061 let drop = strip.min(leading);
1062 out.push_str(&line[drop..]);
1063 }
1064 out
1065}
1066
1067/// Extract the body of a `Comment` trivia token: everything after the
1068/// leading `--` marker, preserving its inline whitespace verbatim. Used by
1069/// the parser when attaching comments to declarations.
1070pub fn comment_body(source: &str, span: Span) -> &str {
1071 let slice = &source[span.range()];
1072 // Strip leading "--" if present (defensive — the lexer always emits
1073 // Comment tokens whose span begins with `--`).
1074 slice.strip_prefix("--").unwrap_or(slice)
1075}
1076
1077/// Returns true if there is a blank line (a line containing only whitespace)
1078/// in `source` strictly between byte offsets `from` (inclusive) and `to`
1079/// (exclusive). Used by the parser to detect orphan doc blocks.
1080///
1081/// A doc-block token's span ends just past the closing-marker line's
1082/// terminating newline. So if the next declaration begins on the immediately
1083/// following line, the substring between contains no newline (only optional
1084/// indentation). Any newline in the substring therefore implies at least one
1085/// entirely-blank line separating the doc from the declaration.
1086pub fn has_blank_line_between(source: &str, from: usize, to: usize) -> bool {
1087 if to <= from {
1088 return false;
1089 }
1090 let bytes = source.as_bytes();
1091 let mut i = from;
1092 while i < to {
1093 if bytes[i] == b'\n' {
1094 return true;
1095 }
1096 if !matches!(bytes[i], b' ' | b'\t' | b'\r') {
1097 return false;
1098 }
1099 i += 1;
1100 }
1101 false
1102}
1103
1104#[cfg(test)]
1105mod tests {
1106 use super::*;
1107
1108 fn kinds(source: &str) -> Vec<TokenKind> {
1109 tokenize(source)
1110 .unwrap()
1111 .into_iter()
1112 .map(|t| t.kind)
1113 .collect()
1114 }
1115
1116 #[test]
1117 fn keywords_and_idents() {
1118 use TokenKind::*;
1119 assert_eq!(
1120 kinds("commons type fn where true false Int String Bool foo bar"),
1121 vec![
1122 Commons, Type, Fn, Where, True, False, Int, String, Bool, Ident, Ident
1123 ],
1124 );
1125 // #548: `and` is no longer a keyword — it lexes as an ordinary identifier.
1126 assert_eq!(kinds("and"), vec![Ident]);
1127 }
1128
1129 #[test]
1130 fn deeply_nested_interpolation_is_bounded_not_overflowed() {
1131 // `"\("\("\(…` mutually recurses scan_str <-> scan_hole, one frame per
1132 // level, and an unbounded scanner overflows `tokenize` and aborts the
1133 // process (#713). Well past the limit it must return a bounded-depth
1134 // diagnostic instead. The holes are left open so the depth guard, not a
1135 // later `)`, stops the scan.
1136 let depth = crate::MAX_NESTING_DEPTH + 8;
1137 let src = format!("\"{}", "\\(\"".repeat(depth));
1138 let err = tokenize(&src).unwrap_err();
1139 assert_eq!(err.category, "bynk.lex.interpolation_too_deep");
1140 }
1141
1142 #[test]
1143 fn integer_and_string_literals() {
1144 use TokenKind::*;
1145 assert_eq!(
1146 kinds(r#"0 42 "hello" "with\nescape""#),
1147 vec![IntLit, IntLit, StrLit, StrLit]
1148 );
1149 }
1150
1151 #[test]
1152 fn operators() {
1153 use TokenKind::*;
1154 assert_eq!(
1155 kinds("-> == != <= >= && || + - * / ! = < > ( ) { } [ ] , : . @"),
1156 vec![
1157 Arrow, EqEq, BangEq, LtEq, GtEq, AmpAmp, PipePipe, Plus, Minus, Star, Slash, Bang,
1158 Eq, Lt, Gt, LParen, RParen, LBrace, RBrace, LBracket, RBracket, Comma, Colon, Dot,
1159 At,
1160 ],
1161 );
1162 }
1163
1164 #[test]
1165 fn dot_family_maximal_munch() {
1166 // Events track slice 1 (spine #936): `..` must lex as one `DotDot`
1167 // token, not two `Dot`s — a real token keeps agreement with
1168 // tree-sitter (which declares `".."` as one literal), so a
1169 // whitespace-split `. .` cannot silently parse where a real `..`
1170 // is required. Also confirms `...`/`..`/`.` don't shadow each other
1171 // regardless of declaration order (logos maximal-munch).
1172 use TokenKind::*;
1173 assert_eq!(
1174 kinds("a .. b ... c . d . ."),
1175 vec![Ident, DotDot, Ident, DotDotDot, Ident, Dot, Ident, Dot, Dot,],
1176 );
1177 }
1178
1179 #[test]
1180 fn line_comments_emitted_as_trivia() {
1181 // v1.1: line comments are preserved as Comment tokens so the
1182 // formatter can attach and re-emit them.
1183 use TokenKind::*;
1184 let src = "-- a comment\ntype X = Int -- trailing\n";
1185 assert_eq!(kinds(src), vec![Comment, Type, Ident, Eq, Int, Comment],);
1186 }
1187
1188 #[test]
1189 fn comment_body_extracts_text_after_marker() {
1190 let toks = tokenize("-- hello world\n").unwrap();
1191 assert_eq!(toks.len(), 1);
1192 assert_eq!(toks[0].kind, TokenKind::Comment);
1193 assert_eq!(
1194 comment_body("-- hello world\n", toks[0].span),
1195 " hello world"
1196 );
1197 }
1198
1199 #[test]
1200 fn comment_does_not_consume_newline() {
1201 // Two adjacent comment lines should produce two distinct tokens
1202 // — the newline between them is not part of either comment's span.
1203 let toks = tokenize("-- one\n-- two\n").unwrap();
1204 assert_eq!(toks.len(), 2);
1205 assert!(toks.iter().all(|t| t.kind == TokenKind::Comment));
1206 }
1207
1208 #[test]
1209 fn dashdash_opens_a_comment_only_when_whitespace_preceded() {
1210 // #548: a `--` opens a comment at the start of input, or when preceded by
1211 // whitespace/line-start. Adjacent to a preceding token it is *not* a
1212 // comment — `a--b` lexes as `a - -b`, never a swallowed line comment.
1213 use TokenKind::*;
1214 assert_eq!(kinds("a--b"), vec![Ident, Minus, Minus, Ident]);
1215 // A trailing decrement-looking `x--` is two operators, not a comment
1216 // that eats the rest of the line — including at end-of-input with no
1217 // trailing newline (the `pos + 1 < len` guard still holds for `x--`).
1218 assert_eq!(kinds("x--\ny"), vec![Ident, Minus, Minus, Ident]);
1219 assert_eq!(kinds("x--"), vec![Ident, Minus, Minus]);
1220 // Whitespace-preceded and start-of-input `--` are still comments.
1221 assert_eq!(kinds("a -- c"), vec![Ident, Comment]);
1222 assert_eq!(kinds("-- c"), vec![Comment]);
1223 // Start of a fresh line (newline-preceded) is a comment.
1224 assert_eq!(kinds("a\n-- c"), vec![Ident, Comment]);
1225 // The comment/doc-block asymmetry: `--` needs only whitespace before it,
1226 // so a mid-line `a ---b` is a *comment* (the leading `-` of the three is
1227 // whitespace-preceded); a `---` doc-block additionally needs line-start,
1228 // which `a ---b` is not.
1229 assert_eq!(kinds("a ---b"), vec![Ident, Comment]);
1230 // A single `-` between terms is unaffected.
1231 assert_eq!(kinds("a - b"), vec![Ident, Minus, Ident]);
1232 }
1233
1234 #[test]
1235 fn unterminated_string_is_error() {
1236 let err = tokenize("\"oops\n").unwrap_err();
1237 assert_eq!(err.category, "bynk.lex.unterminated_string");
1238 }
1239
1240 #[test]
1241 fn integer_overflow_is_error() {
1242 let err = tokenize("99999999999999999999").unwrap_err();
1243 assert_eq!(err.category, "bynk.lex.integer_overflow");
1244 }
1245
1246 #[test]
1247 fn digit_separators_lex_as_one_number() {
1248 use TokenKind::*;
1249 // v0.142 (ADR 0166): `_` between digit groups keeps the literal a single
1250 // token for both Int and Float.
1251 assert_eq!(kinds("1_048_576"), vec![IntLit]);
1252 assert_eq!(kinds("1_000.500_5"), vec![FloatLit]);
1253 assert_eq!(kinds("1_000e1_0"), vec![FloatLit]);
1254 // A separator-carrying literal that is in range still lexes (the value is
1255 // validated after stripping the separators).
1256 assert!(tokenize("9_223_372_036_854_775_807").is_ok());
1257 // Overflow is still caught on the separator-free value.
1258 let err = tokenize("9_999_999_999_999_999_999_9").unwrap_err();
1259 assert_eq!(err.category, "bynk.lex.integer_overflow");
1260 }
1261
1262 #[test]
1263 fn strip_digit_separators_removes_underscores() {
1264 assert_eq!(strip_digit_separators("1_048_576"), "1048576");
1265 assert_eq!(strip_digit_separators("42"), "42");
1266 }
1267
1268 #[test]
1269 fn unexpected_character_is_error() {
1270 let err = tokenize("type X = Int $").unwrap_err();
1271 assert_eq!(err.category, "bynk.lex.unexpected_character");
1272 }
1273
1274 #[test]
1275 fn v0_1_keywords() {
1276 use TokenKind::*;
1277 assert_eq!(
1278 kinds("let if else Ok Err Result ValidationError"),
1279 vec![Let, If, Else, Ok, Err, Result, ValidationError],
1280 );
1281 }
1282
1283 #[test]
1284 fn question_token() {
1285 use TokenKind::*;
1286 assert_eq!(kinds("x?"), vec![Ident, Question]);
1287 }
1288
1289 #[test]
1290 fn v0_2_keywords() {
1291 use TokenKind::*;
1292 assert_eq!(
1293 kinds("enum match Option record self Some None is"),
1294 vec![Enum, Match, Option, Record, Self_, Some, None, Is],
1295 );
1296 }
1297
1298 #[test]
1299 fn pipe_and_pipe_pipe_disambiguated() {
1300 use TokenKind::*;
1301 assert_eq!(kinds("| || |"), vec![Pipe, PipePipe, Pipe]);
1302 }
1303
1304 #[test]
1305 fn v0_7_keywords() {
1306 use TokenKind::*;
1307 assert_eq!(kinds("expect suite case"), vec![Expect, Suite, Case],);
1308 // v0.118: `mocks` and `wires` are retired — plain identifiers now.
1309 assert_eq!(kinds("mocks wires"), vec![Ident, Ident]);
1310 }
1311
1312 #[test]
1313 fn fat_arrow_and_underscore() {
1314 use TokenKind::*;
1315 assert_eq!(kinds("_ =>"), vec![Underscore, FatArrow]);
1316 }
1317
1318 // -- v0.43 string interpolation --
1319
1320 #[test]
1321 fn interp_string_is_one_token() {
1322 use TokenKind::*;
1323 assert_eq!(kinds(r#""Hello, \(name)!""#), vec![InterpStr]);
1324 // A plain string (no hole) stays a `StrLit`, via the logos path.
1325 assert_eq!(kinds(r#""Hello, world""#), vec![StrLit]);
1326 }
1327
1328 #[test]
1329 fn interp_balances_nested_parens_and_strings() {
1330 use TokenKind::*;
1331 // The `)` inside `f(x)` must not close the hole early.
1332 assert_eq!(kinds(r#""= \(f(x))""#), vec![InterpStr]);
1333 // A `)` inside a nested string inside the hole is also ignored.
1334 assert_eq!(kinds(r#""= \(label(")"))""#), vec![InterpStr]);
1335 // A nested interpolated string inside a hole.
1336 assert_eq!(kinds(r#""out \("in \(x)")""#), vec![InterpStr]);
1337 }
1338
1339 // Issue #473: hole-expanding tokenisation makes identifiers inside `\(…)`
1340 // visible to the LSP's token-based cursor resolution.
1341 #[test]
1342 fn expanding_holes_exposes_hole_identifiers() {
1343 use TokenKind::*;
1344 let expand = |src: &str| {
1345 tokenize_expanding_holes(src)
1346 .unwrap()
1347 .into_iter()
1348 .map(|t| t.kind)
1349 .collect::<Vec<_>>()
1350 };
1351 // The opaque `InterpStr` is replaced by its hole's tokens; the chunk
1352 // text (`Hello, ` / `!`) carries none.
1353 assert_eq!(expand(r#""Hello, \(name)!""#), vec![Ident]);
1354 // A call hole exposes every token of the call expression.
1355 assert_eq!(expand(r#""= \(f(x))""#), vec![Ident, LParen, Ident, RParen]);
1356 // Nested interpolation recurses to the innermost hole's identifier.
1357 assert_eq!(expand(r#""out \("in \(x)")""#), vec![Ident]);
1358 // A plain (hole-free) string is untouched.
1359 assert_eq!(expand(r#""Hello, world""#), vec![StrLit]);
1360 }
1361
1362 #[test]
1363 fn expanding_holes_rebases_spans_to_absolute() {
1364 let src = r#""Hello, \(name)!""#;
1365 let toks = tokenize_expanding_holes(src).unwrap();
1366 let ident = toks
1367 .iter()
1368 .find(|t| t.kind == TokenKind::Ident)
1369 .expect("the hole identifier is exposed");
1370 // The span points at `name` in the original source, not a hole-local 0.
1371 assert_eq!(&src[ident.span.range()], "name");
1372 assert_eq!(ident.span.start, src.find("name").unwrap());
1373 }
1374
1375 #[test]
1376 fn escaped_open_paren_is_not_a_hole() {
1377 use TokenKind::*;
1378 // `\\(` is a literal backslash followed by `(` — no hole, so the
1379 // string lexes as a plain `StrLit` on the logos path.
1380 assert_eq!(kinds(r#""a \\(b) c""#), vec![StrLit]);
1381 }
1382
1383 #[test]
1384 fn unterminated_hole_is_an_error() {
1385 // The hole runs to end of line without its closing `)`.
1386 let err = tokenize("\"value \\(x + 1\n\"").unwrap_err();
1387 assert_eq!(err.category, "bynk.lex.unterminated_interpolation");
1388 }
1389
1390 #[test]
1391 fn unterminated_interp_string_is_an_error() {
1392 // A hole closes but the string never does (newline before the `"`).
1393 let err = tokenize("\"value \\(x) more\n").unwrap_err();
1394 assert_eq!(err.category, "bynk.lex.unterminated_string");
1395 }
1396
1397 #[test]
1398 fn bad_escape_in_interp_string_is_an_error() {
1399 let err = tokenize(r#""a \q \(x)""#).unwrap_err();
1400 assert_eq!(err.category, "bynk.lex.bad_escape");
1401 }
1402
1403 fn doc_block_span(source: &str) -> Span {
1404 tokenize(source)
1405 .unwrap()
1406 .into_iter()
1407 .find(|t| t.kind == TokenKind::DocBlock)
1408 .expect("a DocBlock token")
1409 .span
1410 }
1411
1412 #[test]
1413 fn doc_block_body_range_slices_to_the_same_bytes_doc_block_content_would_strip() {
1414 let src = "---\nHello there.\n---\nfn f() -> Int = 1\n";
1415 let span = doc_block_span(src);
1416 let range = doc_block_body_range(src, span).unwrap();
1417 assert_eq!(&src[range], "Hello there.");
1418 }
1419
1420 #[test]
1421 fn doc_block_body_range_is_offset_preserving_unlike_doc_block_content() {
1422 // A content line indented relative to its (unindented) markers:
1423 // doc_block_content strips the common indent (not offset-preserving),
1424 // doc_block_body_range does not — its slice still contains the raw
1425 // indentation, so span-based callers can map a position in the raw
1426 // body straight back to `src`.
1427 let src = "---\n See [Foo].\n---\nfn f() -> Int = 1\n";
1428 let span = doc_block_span(src);
1429 let range = doc_block_body_range(src, span).unwrap();
1430 assert_eq!(&src[range.clone()], " See [Foo].");
1431 assert_eq!(doc_block_content(src, span), "See [Foo].");
1432 // The raw range still locates `[Foo]` correctly within `src`.
1433 let bracket_rel = src[range.clone()].find('[').unwrap();
1434 assert_eq!(
1435 &src[range.start + bracket_rel..range.start + bracket_rel + 5],
1436 "[Foo]"
1437 );
1438 }
1439
1440 #[test]
1441 fn doc_block_content_and_body_range_agree_on_empty_body() {
1442 let src = "---\n---\nfn f() -> Int = 1\n";
1443 let span = doc_block_span(src);
1444 let range = doc_block_body_range(src, span).unwrap();
1445 assert_eq!(&src[range], "");
1446 assert_eq!(doc_block_content(src, span), "");
1447 }
1448}