Skip to main content

bynk_syntax/
parser.rs

1//! Hand-written recursive-descent parser for Bynk v0.
2//!
3//! Token grammar in spec §4. The expression parser uses one function per
4//! precedence level (§4.4). Errors carry spans and short fix-oriented
5//! messages; the parser does not currently attempt synchronisation, which
6//! means at most one parse error is reported per compilation.
7
8use crate::ast::*;
9use crate::error::CompileError;
10use crate::lexer::{Token, TokenKind, comment_body, doc_block_content, has_blank_line_between};
11use crate::span::Span;
12mod declarations;
13mod expressions;
14mod statements;
15mod types;
16
17/// Side-channel store for line-comment trivia (v1.1 LSP spec §3.5).
18///
19/// Built once up-front by [`split_trivia`] from the raw lexer token stream.
20/// Comments are removed from the token stream the parser walks; their text
21/// is filed into `leading` (comments on lines preceding a content token)
22/// and `trailing` (a single comment on the same line as a content token).
23/// The parser consumes entries through [`TriviaTable::take_leading`] and
24/// [`TriviaTable::take_trailing`] as it recognises declarations.
25#[derive(Debug, Default)]
26struct TriviaTable {
27    /// `leading[i]` holds the comment-body texts that appear immediately
28    /// before content token `i` (zero or more `--` lines, in source order,
29    /// not separated from the token by another content token).
30    leading: Vec<Vec<String>>,
31    /// `trailing[i]` holds an optional comment on the same source line as
32    /// content token `i`. Only one trailing comment is recorded per token
33    /// because a single `--` consumes the rest of the line.
34    trailing: Vec<Option<String>>,
35    /// Any pending leading comments at end-of-file (no content token
36    /// followed). Used to preserve file-trailing comments.
37    epilogue: Vec<String>,
38}
39
40impl TriviaTable {
41    fn take_leading(&mut self, index: usize) -> Vec<String> {
42        match self.leading.get_mut(index) {
43            Some(v) => std::mem::take(v),
44            None => Vec::new(),
45        }
46    }
47
48    fn take_trailing(&mut self, index: usize) -> Option<String> {
49        self.trailing.get_mut(index).and_then(|s| s.take())
50    }
51
52    fn take_epilogue(&mut self) -> Vec<String> {
53        std::mem::take(&mut self.epilogue)
54    }
55
56    /// True when every entry has been drained via `take_leading`/
57    /// `take_trailing`/`take_epilogue` — i.e. no comment was silently
58    /// dropped. Each harvest is already a `mem::take`, so anything still
59    /// present here is exactly the set of comments that never reached an
60    /// AST `Trivia` field.
61    ///
62    /// Deliberately **not** wired into a `debug_assert!` in the general parse
63    /// path: expressions carry no per-node trivia (§ "Comment trivia" in the
64    /// 2026-07-27 pipeline review), so an ordinary, valid program with a
65    /// comment inside a `match`/list/record/binop — a common, accepted
66    /// pattern `bynk-fmt`'s own comment-loss guard already handles
67    /// gracefully — would leave `leading`/`trailing` non-empty and trip it on
68    /// every compile, not just on formatting. Instead surfaced through
69    /// [`parse_units_with_drain_check`] (finding #66), whose one caller
70    /// (`bynk-fmt`) *does* care about exactly this signal.
71    /// [`Self::epilogue_is_empty`] is the narrower, safe-to-assert check.
72    fn is_fully_drained(&self) -> bool {
73        self.leading.iter().all(Vec::is_empty)
74            && self.trailing.iter().all(Option::is_none)
75            && self.epilogue.is_empty()
76    }
77
78    /// True when no file-trailing comment was left stranded. Unlike
79    /// [`Self::is_fully_drained`], this is safe to assert unconditionally: a
80    /// clean file's epilogue is empty by construction (nothing pending at
81    /// EOF), and the one shape that legitimately populates it — a top-level
82    /// trailing comment — is drained by every parse path that calls
83    /// `take_epilogue`. A brace-form declaration that forgets to is exactly
84    /// the bug this catches.
85    fn epilogue_is_empty(&self) -> bool {
86        self.epilogue.is_empty()
87    }
88}
89
90/// Remove `Comment` trivia tokens from `tokens` and bin them into a
91/// [`TriviaTable`] keyed against the surviving content tokens. A comment
92/// on the same source line as the preceding content token is recorded as
93/// that token's *trailing* trivia; everything else is *leading* for the
94/// next content token.
95fn split_trivia(tokens: &[Token], source: &str) -> (Vec<Token>, TriviaTable) {
96    let mut filtered: Vec<Token> = Vec::with_capacity(tokens.len());
97    let mut table = TriviaTable::default();
98    let mut pending_leading: Vec<String> = Vec::new();
99    let mut last_content_end: Option<usize> = None;
100    for tok in tokens {
101        if tok.kind == TokenKind::Comment {
102            let body = comment_body(source, tok.span).to_string();
103            // If nothing has been buffered as leading for the next token and
104            // there is no newline between the previous content token and
105            // this comment, it trails that token.
106            if pending_leading.is_empty()
107                && let Some(prev_end) = last_content_end
108                && !source[prev_end..tok.span.start].contains('\n')
109            {
110                let last_idx = filtered.len() - 1;
111                // Only attach if no trailing already recorded (shouldn't
112                // happen because `--` consumes through end-of-line).
113                if table.trailing[last_idx].is_none() {
114                    table.trailing[last_idx] = Some(body);
115                    continue;
116                }
117            }
118            pending_leading.push(body);
119            continue;
120        }
121        filtered.push(*tok);
122        table.leading.push(std::mem::take(&mut pending_leading));
123        table.trailing.push(None);
124        last_content_end = Some(tok.span.end);
125    }
126    table.epilogue = pending_leading;
127    (filtered, table)
128}
129
130/// Parse a token slice into a [`Commons`] AST.
131///
132/// Accepts either form of v0.3 commons file:
133/// - Brace form: `commons name { items... }` (v0–v0.2 compatible).
134/// - Fragment form: `commons name uses... items...` to EOF (v0.3).
135pub fn parse(tokens: &[Token], source: &str) -> Result<Commons, Vec<CompileError>> {
136    parse_with_warnings(tokens, source).map(|(c, _warnings)| c)
137}
138
139/// [`parse`] with the non-fatal diagnostics threaded out alongside the AST
140/// (ADR 0117) — see [`parse_units_with_warnings`].
141pub fn parse_with_warnings(
142    tokens: &[Token],
143    source: &str,
144) -> Result<(Commons, Vec<CompileError>), Vec<CompileError>> {
145    let (unit, warnings) = parse_unit_with_warnings(tokens, source)?;
146    match unit {
147        SourceUnit::Commons(c) => Ok((c, warnings)),
148        SourceUnit::Context(ctx) => Err(vec![
149            CompileError::new(
150                "bynk.parse.unexpected_context",
151                ctx.span,
152                "expected a `commons` declaration but found a `context` declaration",
153            )
154            .with_note(
155                "contexts must be compiled as part of a project — pass the source directory, e.g. `bynkc compile --target bundle --output out src`",
156            ),
157        ]),
158        SourceUnit::Suite(t) => Err(vec![
159            CompileError::new(
160                "bynk.parse.unexpected_suite",
161                t.span,
162                "expected a `commons` declaration but found a `suite` declaration",
163            )
164            .with_note(
165                "tests must be compiled as part of a project — pass the source directory, e.g. `bynkc compile --target bundle --output out src`",
166            ),
167        ]),
168        SourceUnit::Adapter(a) => Err(vec![
169            CompileError::new(
170                "bynk.parse.unexpected_adapter",
171                a.span,
172                "expected a `commons` declaration but found an `adapter` declaration",
173            )
174            .with_note(
175                "adapters must be compiled as part of a project — pass the source directory, e.g. `bynkc compile --target bundle --output out src`",
176            ),
177        ]),
178    }
179}
180
181/// Parse a token slice into a [`SourceUnit`] with error recovery, returning a
182/// best-effort partial AST plus the full list of parse errors and warnings.
183///
184/// Used by the LSP: item-level recovery skips past a malformed declaration to
185/// the next top-level item, so multiple errors are reported per compilation
186/// rather than just the first. Compared to [`parse_unit`], this never bails;
187/// if no SourceUnit could be parsed at all (e.g. the file is empty or the
188/// header itself fails) the returned `Option` is `None`.
189///
190/// Keeps only the *first* unit — v0.113 allows more than one top-level unit
191/// per file (an atomic `commons` + `suite`, DECISION S), and every existing
192/// caller here is keyed on the primary declaration. [`parse_units_with_recovery`]
193/// is the same recovery parse without that narrowing, for the one caller
194/// (finding #29/#30) that needs every unit a file declares.
195pub fn parse_unit_with_recovery(
196    tokens: &[Token],
197    source: &str,
198) -> (Option<SourceUnit>, Vec<CompileError>) {
199    let (units, errors) = parse_units_with_recovery(tokens, source);
200    (units.into_iter().next(), errors)
201}
202
203/// [`parse_unit_with_recovery`], keeping **every** top-level unit instead of
204/// discarding all but the first (finding #29/#30). Used by the IDE's own parse
205/// entry point, which needs to see a trailing `suite` in an atomic
206/// `commons`+`suite` file, not just the primary declaration.
207pub fn parse_units_with_recovery(
208    tokens: &[Token],
209    source: &str,
210) -> (Vec<SourceUnit>, Vec<CompileError>) {
211    let (filtered, trivia) = split_trivia(tokens, source);
212    let mut warnings = Vec::new();
213    let mut p = Parser::new(&filtered, source, trivia, &mut warnings);
214    p.recover_mode = true;
215    let mut units = Vec::new();
216    loop {
217        match p.parse_unit() {
218            Ok(u) => units.push(u),
219            Err(e) => {
220                p.recovered_errors.push(e);
221                break;
222            }
223        }
224        // A genuinely malformed trailing declaration is still surfaced via
225        // recovery — checked *after* each successful parse, matching
226        // `parse_unit`'s own "at least once" attempt on the first unit (an
227        // empty file must still produce its usual unexpected-EOF diagnostic,
228        // not silently yield an empty `units` with no error at all).
229        if p.peek().is_none() {
230            break;
231        }
232    }
233    let mut all_errors = p.recovered_errors;
234    all_errors.append(&mut warnings);
235    (units, all_errors)
236}
237
238/// Parse a token slice into a [`SourceUnit`] — either a commons or a context.
239///
240/// Each `.bynk` file is exactly one declaration of one kind.
241pub fn parse_unit(tokens: &[Token], source: &str) -> Result<SourceUnit, Vec<CompileError>> {
242    parse_unit_with_warnings(tokens, source).map(|(unit, _warnings)| unit)
243}
244
245/// [`parse_unit`] with the non-fatal diagnostics threaded out alongside the
246/// AST (ADR 0117) — see [`parse_units_with_warnings`].
247pub fn parse_unit_with_warnings(
248    tokens: &[Token],
249    source: &str,
250) -> Result<(SourceUnit, Vec<CompileError>), Vec<CompileError>> {
251    let (filtered, trivia) = split_trivia(tokens, source);
252    let mut warnings = Vec::new();
253    let mut p = Parser::new(&filtered, source, trivia, &mut warnings);
254    let result = match p.parse_unit() {
255        Ok(u) => {
256            if let Some(extra) = p.peek() {
257                Err(vec![
258                    CompileError::new(
259                        "bynk.parse.extra_tokens",
260                        extra.span,
261                        "unexpected token after top-level declaration",
262                    )
263                    .with_note(
264                        "a `.bynk` file contains exactly one `commons` or `context` declaration",
265                    ),
266                ])
267            } else {
268                Ok(u)
269            }
270        }
271        Err(e) => Err(vec![e]),
272    };
273    // ADR 0117: warnings (e.g. orphan doc blocks) ride alongside a successful
274    // parse — severity governs gating at the caller, not here.
275    match result {
276        Ok(u) => {
277            // See `parse_units_with_warnings`: a file-trailing comment must
278            // have been drained by `take_epilogue`.
279            debug_assert!(
280                p.trivia.epilogue_is_empty(),
281                "a file-trailing comment was left undrained after a successful parse"
282            );
283            Ok((u, warnings))
284        }
285        Err(mut errs) => {
286            errs.append(&mut warnings);
287            Err(errs)
288        }
289    }
290}
291
292/// Parse a token slice into **all** the top-level [`SourceUnit`]s in one file
293/// (v0.113, testing track slice 1b). A `.bynk` file may hold more than one
294/// top-level declaration — an *atomic* file with `commons`/`context` **and** a
295/// `suite` together (DECISION S) — so the compiler parses a `Vec`, not a single
296/// unit. Test-ness is a property of each declaration, not of the file.
297///
298/// Bails on the first malformed declaration (like [`parse_unit`], not the
299/// recovering LSP path). An empty file is an error.
300pub fn parse_units(tokens: &[Token], source: &str) -> Result<Vec<SourceUnit>, Vec<CompileError>> {
301    parse_units_with_warnings(tokens, source).map(|(units, _warnings)| units)
302}
303
304/// [`parse_units`] with the non-fatal diagnostics threaded out alongside the
305/// AST (ADR 0117): a successful parse returns `Ok((units, warnings))` instead
306/// of hard-failing on a warning-severity diagnostic (an orphan doc block used
307/// to abort file discovery and throw the good AST away). A failed parse still
308/// returns every diagnostic — errors then warnings — in the `Err`.
309pub fn parse_units_with_warnings(
310    tokens: &[Token],
311    source: &str,
312) -> Result<(Vec<SourceUnit>, Vec<CompileError>), Vec<CompileError>> {
313    parse_units_with_drain_check(tokens, source)
314        .map(|(units, warnings, _drained)| (units, warnings))
315}
316
317/// [`parse_units_with_warnings`] plus whether every comment's trivia was
318/// drained into the AST (`TriviaTable::is_fully_drained`). Finding #66:
319/// `bynk-fmt`'s comment-preservation guard re-tokenized its own rendered
320/// output just to diff comment bodies against the input — wasted work in the
321/// overwhelming common case where nothing was left behind. That guard uses
322/// this drain signal, computed from the same parse it already needs for
323/// rendering, as a fast-path: `true` means every comment landed in the AST,
324/// so re-checking the output can be skipped outright. No other caller needs
325/// the signal, so it rides its own entry point rather than widening
326/// [`parse_units_with_warnings`].
327pub fn parse_units_with_drain_check(
328    tokens: &[Token],
329    source: &str,
330) -> Result<(Vec<SourceUnit>, Vec<CompileError>, bool), Vec<CompileError>> {
331    let (filtered, trivia) = split_trivia(tokens, source);
332    let mut warnings = Vec::new();
333    let mut p = Parser::new(&filtered, source, trivia, &mut warnings);
334    let mut units = Vec::new();
335    let mut errors: Vec<CompileError> = Vec::new();
336    while p.peek().is_some() {
337        match p.parse_unit() {
338            Ok(u) => units.push(u),
339            Err(e) => {
340                errors.push(e);
341                break;
342            }
343        }
344    }
345    let eof = p.eof_span();
346    let fully_drained = p.trivia.is_fully_drained();
347    // `p` (and thus its `&mut warnings` borrow) is no longer used past here, so
348    // the local `warnings` are readable again.
349    if !errors.is_empty() {
350        errors.append(&mut warnings);
351        return Err(errors);
352    }
353    if units.is_empty() {
354        return Err(vec![CompileError::new(
355            "bynk.parse.unexpected_eof",
356            eof,
357            "expected `commons`, `context`, or `suite` to start the file, found end of file",
358        )]);
359    }
360    // A file-trailing comment must have been drained by `take_epilogue` — the
361    // one brace-form declarations forgot to call (a live comment-loss bug,
362    // not the fundamentally-unfixed expression-interior case: expressions
363    // carry no trivia at all, so asserting full drainage here would fire on
364    // any ordinary program with a comment inside a `match`/list/record, which
365    // `bynk-fmt`'s own comment-loss guard already handles gracefully rather
366    // than as a hard failure).
367    debug_assert!(
368        p.trivia.epilogue_is_empty(),
369        "a file-trailing comment was left undrained after a successful parse"
370    );
371    Ok((units, warnings, fully_drained))
372}
373
374/// A signed numeric literal in refinement-bound position (v0.21): `InRange`
375/// bounds are either both `Int` or both `Float`.
376enum SignedNumLit {
377    Int(IntBound),
378    Float(FloatBound),
379}
380
381struct Parser<'a> {
382    tokens: &'a [Token],
383    source: &'a str,
384    pos: usize,
385    /// Accumulated non-fatal diagnostics. v0.3 uses this for orphan-doc
386    /// warnings, which are emitted as errors with a distinguishable category.
387    warnings: &'a mut Vec<CompileError>,
388    /// When true, the item-level loops catch errors from individual item
389    /// parses, push them into `recovered_errors`, and skip forward to the
390    /// next top-level item boundary instead of bailing. Used by the LSP via
391    /// [`parse_unit_with_recovery`]; disabled in the normal `parse` path so
392    /// existing single-error behaviour is preserved.
393    recover_mode: bool,
394    /// Errors collected during recovery-mode parsing. Only populated when
395    /// `recover_mode` is true.
396    recovered_errors: Vec<CompileError>,
397    /// Line-comment trivia separated from the token stream. See
398    /// [`TriviaTable`].
399    trivia: TriviaTable,
400    /// Live recursion depth of the three self-recursive parse entry points
401    /// (`parse_expr`, `parse_type_ref`, `parse_pattern`). Incremented on entry
402    /// and decremented on exit by [`Parser::enter_recursion`] so it tracks the
403    /// current stack depth; when it exceeds [`crate::MAX_NESTING_DEPTH`] the
404    /// parser reports a bounded-depth diagnostic instead of overflowing its
405    /// stack (#713).
406    depth: usize,
407    /// When true, a bare `ident {` on the *spine* of the current expression is
408    /// an identifier followed by an unrelated block, never a record
409    /// construction — so an `if`/`match` condition that ends in a bare
410    /// identifier does not swallow the branch/arm block as `Ident { field }`
411    /// (#636). Set only around the condition parse (see [`parse_cond_expr`]);
412    /// `parse_expr` clears it, so the restriction is lifted inside any
413    /// delimited sub-expression (parentheses, call arguments, list, record
414    /// field). Mirrors Rust's `NO_STRUCT_LITERAL` restriction.
415    no_record_literal: bool,
416    /// Running count of unclosed `{` seen so far — maintained solely by
417    /// [`Self::bump`] (the one primitive that advances `self.pos`), so it
418    /// always reflects the true nesting depth no matter which parse function
419    /// is on the call stack. Finding #27/#30: `recover_to_top_item` reads it
420    /// against [`Self::item_loop_baseline`] to tell "the enclosing item
421    /// loop's own closing brace" apart from a still-unclosed nested
422    /// construct's — without it, a sync scan that started partway through
423    /// such a construct (an error deep inside a function body) stopped at the
424    /// first `}` it saw, however deeply nested, and the enclosing item loop
425    /// mistook that for its own body's end.
426    brace_depth: usize,
427    /// Stack of `brace_depth` snapshots, one per active item-loop body
428    /// (commons/context/adapter/suite) — pushed right after that body's own
429    /// `{` is consumed (or at loop entry, for a brace-free fragment form),
430    /// popped at the loop's normal exit. `recover_to_top_item` treats its top
431    /// entry as the depth an `}` must return to before it counts as the
432    /// enclosing body's own closing brace rather than a nested construct's.
433    item_loop_baseline: Vec<usize>,
434}
435
436impl<'a> Parser<'a> {
437    fn new(
438        tokens: &'a [Token],
439        source: &'a str,
440        trivia: TriviaTable,
441        warnings: &'a mut Vec<CompileError>,
442    ) -> Self {
443        Self {
444            tokens,
445            source,
446            pos: 0,
447            warnings,
448            recover_mode: false,
449            recovered_errors: Vec::new(),
450            trivia,
451            depth: 0,
452            no_record_literal: false,
453            brace_depth: 0,
454            item_loop_baseline: Vec::new(),
455        }
456    }
457
458    /// Enter a self-recursive parse step, bumping the live recursion depth and
459    /// failing with a bounded-depth diagnostic if it would exceed
460    /// [`crate::MAX_NESTING_DEPTH`]. The caller pairs a successful entry with a
461    /// matching `self.depth -= 1` on the way out (see `parse_expr` /
462    /// `parse_type_ref`); on the error path the depth is restored here so a
463    /// recovering caller is not left mis-counted. `what` names the construct
464    /// for the message (e.g. "this expression", "this type"). See #713.
465    fn enter_recursion(&mut self, what: &str) -> Result<(), CompileError> {
466        self.depth += 1;
467        if self.depth > crate::MAX_NESTING_DEPTH {
468            self.depth -= 1;
469            let span = self
470                .peek()
471                .map(|t| t.span)
472                .unwrap_or_else(|| self.eof_span());
473            return Err(self.nesting_too_deep(span, what));
474        }
475        Ok(())
476    }
477
478    /// The bounded-depth diagnostic shared by [`enter_recursion`] and
479    /// [`enter_chain_fold`].
480    fn nesting_too_deep(&self, span: Span, what: &str) -> CompileError {
481        CompileError::new(
482            "bynk.parse.nesting_too_deep",
483            span,
484            format!(
485                "{what} nests more than {} levels deep",
486                crate::MAX_NESTING_DEPTH
487            ),
488        )
489        .with_note(
490            "deeply nested source is rejected to keep the parser from overflowing its \
491             stack and aborting; flatten or split the construct",
492        )
493    }
494
495    /// The bounded-depth diagnostic for the *iteratively*-built spines —
496    /// associative operator chains ([`enter_chain_fold`]) and postfix receiver
497    /// chains ([`deepen_spine`]). Same code as [`nesting_too_deep`] (one budget,
498    /// one diagnostic) but phrased for a flat chain, which is long rather than
499    /// *nested*, and points at the idiomatic fix.
500    fn expression_too_long(&self, span: Span) -> CompileError {
501        CompileError::new(
502            "bynk.parse.nesting_too_deep",
503            span,
504            format!(
505                "this expression is more than {} levels deep",
506                crate::MAX_NESTING_DEPTH
507            ),
508        )
509        .with_note(
510            "a long operator or member chain is rejected to keep the compiler from overflowing \
511             its stack; split it across `let` bindings, or reduce a sequence with \
512             `.sum()`/`.fold(...)`",
513        )
514    }
515
516    /// Count one more operand folded onto an associative operator chain against
517    /// the same recursion budget as [`enter_recursion`] (#714).
518    ///
519    /// Associative chains (`+`, `*`, `&&`, `||`) are built *iteratively* in the
520    /// precedence ladder, so — unlike parentheses, calls, or `implies` — they
521    /// never re-enter `parse_expr` and thus slip past the `enter_recursion`
522    /// guard. Yet each fold deepens the left-nested `Expr` tree by one level,
523    /// and a long flat chain (`1 + 1 + … + 1`) overflows every *recursive*
524    /// consumer of that tree downstream — the checker's `type_of`, the
525    /// formatter, the emitter, and the AST's own recursive `Drop` — exactly as
526    /// deeply nested source overflows the parser. Counting each fold on the
527    /// shared `depth` budget bounds the whole expression's height, and because
528    /// it is the *same* budget it composes with the ambient nesting depth, so a
529    /// chain buried inside deeply nested source cannot exceed the bound either.
530    ///
531    /// The caller accumulates `folds` and subtracts them from `depth` before it
532    /// returns, so the live count unwinds as a recursive descent would; on the
533    /// overflow path the whole chain's contribution is restored here so a
534    /// recovering caller is not left mis-counted.
535    fn enter_chain_fold(&mut self, folds: &mut usize, span: Span) -> Result<(), CompileError> {
536        self.depth += 1;
537        *folds += 1;
538        if self.depth > crate::MAX_NESTING_DEPTH {
539            self.depth -= *folds;
540            *folds = 0;
541            return Err(self.expression_too_long(span));
542        }
543        Ok(())
544    }
545
546    /// Count one more level of an iteratively-built postfix receiver spine
547    /// (`a.b.c…`, `f()?.g()…`) against the shared budget (#714). Like
548    /// [`enter_chain_fold`], postfix loops rather than recurses, so a long spine
549    /// escapes [`enter_recursion`] yet grows an arbitrarily deep receiver tree
550    /// that the downstream walks recurse through. `parse_postfix` restores
551    /// `depth` wholesale on the way out (its many error paths make a
552    /// save/restore wrapper cleaner than per-fold unwinding), so this only bumps
553    /// and checks.
554    fn deepen_spine(&mut self, span: Span) -> Result<(), CompileError> {
555        self.depth += 1;
556        if self.depth > crate::MAX_NESTING_DEPTH {
557            return Err(self.expression_too_long(span));
558        }
559        Ok(())
560    }
561
562    /// Comments immediately preceding the current peek position. Consumed
563    /// (the table entry is cleared) so the same comments are not attached
564    /// to two nodes.
565    fn take_leading_trivia(&mut self) -> Vec<String> {
566        self.trivia.take_leading(self.pos)
567    }
568
569    /// Trailing comment, if any, on the same source line as the most
570    /// recently consumed content token. Call AFTER finishing a declaration
571    /// or statement, while `self.pos` points one past its last token.
572    fn take_trailing_trivia(&mut self) -> Option<String> {
573        if self.pos == 0 {
574            return None;
575        }
576        self.trivia.take_trailing(self.pos - 1)
577    }
578
579    /// Handle a per-item parse error. In recovery mode, record the error and
580    /// advance to the next sync point so the item loop can continue; otherwise
581    /// propagate as a hard failure.
582    fn handle_item_err(&mut self, e: CompileError) -> Result<(), CompileError> {
583        if self.recover_mode {
584            self.recovered_errors.push(e);
585            let before = self.pos;
586            self.recover_to_top_item();
587            // The sync target may be the very token that produced the error —
588            // a context-only keyword (`capability`, `service`, …) at item
589            // position in a commons errors *without consuming it*, and it is
590            // itself a sync point. Recovery must always make progress, or the
591            // item loop re-reports the same error until memory runs out
592            // (found by the `parse` fuzz target on a seed input).
593            if self.pos == before {
594                self.bump();
595            }
596            Ok(())
597        } else {
598            Err(e)
599        }
600    }
601
602    /// Skip forward to the next top-level item boundary: either an
603    /// [`is_item_start`] keyword at the enclosing item loop's own nesting
604    /// depth, a closing brace that returns to that depth, or end-of-input.
605    /// Used only in recovery mode.
606    ///
607    /// Finding #27/#30: brace-depth-gated against
608    /// [`Self::item_loop_baseline`], so a `}` deep inside a still-unclosed
609    /// nested construct (an error partway through a function body, itself
610    /// inside a `match` arm) is skipped over rather than mistaken for the
611    /// enclosing body's own closing brace — the old flat scan stopped at
612    /// literally the first `}` it saw, however deep, handing the item loop a
613    /// brace that did not belong to it and making it return with zero items.
614    fn recover_to_top_item(&mut self) {
615        let baseline = self.item_loop_baseline.last().copied().unwrap_or(0);
616        while let Some(t) = self.peek() {
617            match t.kind {
618                TokenKind::RBrace if self.brace_depth == baseline => return,
619                _ if self.brace_depth == baseline && is_item_start(t.kind) => return,
620                _ => {
621                    self.bump();
622                }
623            }
624        }
625    }
626
627    /// Mark the start of a top-level item loop (`declarations.rs`'s
628    /// `parse_commons_brace`/`_fragment`, `parse_context_brace`/`_fragment`,
629    /// `parse_test_brace`/`_fragment`, `parse_adapter_body`) — called right
630    /// after that body's own `{` is consumed (brace form) or at the loop's
631    /// own entry (fragment form, which has no enclosing brace of its own).
632    /// Paired with [`Self::exit_item_loop`] at the loop's normal exit.
633    fn enter_item_loop(&mut self) {
634        self.item_loop_baseline.push(self.brace_depth);
635    }
636
637    /// Pair of [`Self::enter_item_loop`].
638    fn exit_item_loop(&mut self) {
639        self.item_loop_baseline.pop();
640    }
641
642    fn peek(&self) -> Option<Token> {
643        self.tokens.get(self.pos).copied()
644    }
645
646    fn peek_kind(&self) -> Option<TokenKind> {
647        self.peek().map(|t| t.kind)
648    }
649
650    /// The token `n` positions ahead of the cursor (`nth(0)` == `peek()`).
651    fn nth(&self, n: usize) -> Option<Token> {
652        self.tokens.get(self.pos + n).copied()
653    }
654
655    fn nth_kind(&self, n: usize) -> Option<TokenKind> {
656        self.nth(n).map(|t| t.kind)
657    }
658
659    /// The source text of the token `n` positions ahead, or `""` if none.
660    fn nth_text(&self, n: usize) -> &'a str {
661        self.nth(n).map(|t| self.slice(t.span)).unwrap_or("")
662    }
663
664    /// The span of the most recently consumed token (`self.pos - 1`). Falls back
665    /// to the current token's span when nothing has been consumed yet.
666    fn prev_span(&self) -> Span {
667        self.tokens
668            .get(self.pos.wrapping_sub(1))
669            .or_else(|| self.peek_ref())
670            .map(|t| t.span)
671            .unwrap_or_default()
672    }
673
674    fn peek_ref(&self) -> Option<&Token> {
675        self.tokens.get(self.pos)
676    }
677
678    fn bump(&mut self) -> Option<Token> {
679        let t = self.peek();
680        if let Some(t) = t {
681            match t.kind {
682                TokenKind::LBrace => self.brace_depth += 1,
683                TokenKind::RBrace => self.brace_depth = self.brace_depth.saturating_sub(1),
684                _ => {}
685            }
686            self.pos += 1;
687        }
688        t
689    }
690
691    fn eat(&mut self, kind: TokenKind) -> Option<Token> {
692        if self.peek_kind() == Some(kind) {
693            self.bump()
694        } else {
695            None
696        }
697    }
698
699    fn slice(&self, span: Span) -> &'a str {
700        &self.source[span.range()]
701    }
702
703    /// True when the next token sits on a later line than `prev`. Used to
704    /// keep a `[` that opens a new line out of the postfix type-application
705    /// form: `f` followed by `[1, 2]` on the next line is an identifier and
706    /// a list literal, not `f[…]` (v0.20b).
707    fn next_token_on_new_line(&self, prev: Span) -> bool {
708        match self.peek() {
709            Some(t) if prev.end <= t.span.start => {
710                self.source[prev.end..t.span.start].contains('\n')
711            }
712            _ => false,
713        }
714    }
715
716    /// Span pointing at the end of input — used for "unexpected EOF" reports.
717    /// The start backs up to the **start of the final char**, not `len - 1`, so
718    /// the span never splits a multibyte codepoint (an unterminated construct
719    /// whose last line ends in non-ASCII — e.g. a `--` comment ending in `→`).
720    fn eof_span(&self) -> Span {
721        let end = self.source.len();
722        let start = (0..end)
723            .rev()
724            .find(|&i| self.source.is_char_boundary(i))
725            .unwrap_or(0);
726        Span::new(start, end)
727    }
728
729    fn expect(&mut self, kind: TokenKind, ctx: &str) -> Result<Token, CompileError> {
730        match self.peek() {
731            Some(t) if t.kind == kind => {
732                self.bump();
733                Ok(t)
734            }
735            Some(t) => Err(CompileError::new(
736                "bynk.parse.expected_token",
737                t.span,
738                format!(
739                    "expected {} {ctx}, found {}",
740                    kind.describe(),
741                    t.kind.describe()
742                ),
743            )),
744            None => Err(CompileError::new(
745                "bynk.parse.unexpected_eof",
746                self.eof_span(),
747                format!("expected {} {ctx}, found end of file", kind.describe()),
748            )),
749        }
750    }
751
752    fn expect_ident(&mut self, ctx: &str) -> Result<Ident, CompileError> {
753        match self.peek() {
754            Some(t) if t.kind == TokenKind::Ident => {
755                self.bump();
756                Ok(Ident {
757                    name: self.slice(t.span).to_string(),
758                    span: t.span,
759                })
760            }
761            // v0.5 contextual keyword `on` doubles as an identifier in
762            // expression / field-access positions so users can name fields and
763            // parameters using it. It retains its keyword meaning only at
764            // handler-decl-level (`on call(...)`).
765            //
766            // v0.7 / v0.112: `suite` and `case` are contextual too — they
767            // introduce the suite declaration and its cases, but are perfectly
768            // valid commons/context/field names otherwise.
769            //
770            // The tier is single-sourced in `keywords::RESERVED_CONTEXTUAL`:
771            // this arm defers to it rather than hardcoding the token kinds, so
772            // extending that list is enough to admit a new contextual keyword
773            // here. Each of these words lexes only to its own token, so matching
774            // the source text is equivalent to matching the kind.
775            Some(t) if crate::keywords::is_reserved_contextual(self.slice(t.span)) => {
776                self.bump();
777                Ok(Ident {
778                    name: self.slice(t.span).to_string(),
779                    span: t.span,
780                })
781            }
782            Some(t) if is_reserved_keyword(t.kind) => Err(CompileError::new(
783                "bynk.parse.reserved_keyword",
784                t.span,
785                format!(
786                    "expected identifier {ctx}, but `{}` is a reserved keyword",
787                    self.slice(t.span)
788                ),
789            )
790            .with_note("rename the identifier to something that is not a keyword")),
791            Some(t) => Err(CompileError::new(
792                "bynk.parse.expected_token",
793                t.span,
794                format!("expected identifier {ctx}, found {}", t.kind.describe()),
795            )),
796            None => Err(CompileError::new(
797                "bynk.parse.unexpected_eof",
798                self.eof_span(),
799                format!("expected identifier {ctx}, found end of file"),
800            )),
801        }
802    }
803
804    // -- top level --
805
806    /// Consume an optional doc block at the current position, returning the
807    /// (content, end-of-doc span) pair. Returns None if the next token is not
808    /// a doc block.
809    fn take_doc_block(&mut self) -> Option<(String, Span)> {
810        if self.peek_kind() == Some(TokenKind::DocBlock) {
811            let t = self.bump().unwrap();
812            let body = doc_block_content(self.source, t.span);
813            return Some((body, t.span));
814        }
815        None
816    }
817
818    /// Collect all line-comment trivia leading the next declaration plus
819    /// the optional doc block. Comments may appear both *before* and
820    /// *between* the doc and the declaration; the spec canonicalises both
821    /// groups above the doc, so we concatenate them.
822    fn collect_item_lead(&mut self) -> (Vec<String>, Option<(String, Span)>) {
823        let mut leading = self.take_leading_trivia();
824        let doc = self.take_doc_block();
825        if doc.is_some() {
826            leading.extend(self.take_leading_trivia());
827        }
828        (leading, doc)
829    }
830
831    /// Attach a parsed doc block to a following declaration unless a blank
832    /// line separates them, in which case the doc is orphaned (warning).
833    fn finalize_doc(&mut self, doc: Option<(String, Span)>, next_span: Span) -> Option<String> {
834        let (content, doc_span) = doc?;
835        // A blank line between the doc and the next decl orphans the doc.
836        if has_blank_line_between(self.source, doc_span.end, next_span.start) {
837            self.warnings.push(
838                CompileError::new(
839                    "bynk.parse.orphan_doc_block",
840                    doc_span,
841                    "documentation block is separated from the following declaration by a blank line; it will not be attached",
842                )
843                .with_note(
844                    "remove the blank line to attach the doc to the next declaration, \
845                     or remove the doc block if it is not meant to document anything",
846                ),
847            );
848            return None;
849        }
850        Some(content)
851    }
852}
853
854/// Parse the body of a lexed double-quoted string literal (the lexeme,
855/// including surrounding quotes), applying the v0 escape rules.
856fn parse_string_literal(lexeme: &str, span: Span) -> Result<String, CompileError> {
857    let bytes = lexeme.as_bytes();
858    debug_assert!(bytes.first() == Some(&b'"') && bytes.last() == Some(&b'"'));
859    let inner = &lexeme[1..lexeme.len() - 1];
860    let mut out = String::with_capacity(inner.len());
861    let mut chars = inner.chars();
862    while let Some(c) = chars.next() {
863        if c == '\\' {
864            match chars.next() {
865                Some('n') => out.push('\n'),
866                Some('t') => out.push('\t'),
867                Some('"') => out.push('"'),
868                Some('\\') => out.push('\\'),
869                other => {
870                    return Err(CompileError::new(
871                        "bynk.lex.bad_escape",
872                        span,
873                        format!(
874                            "invalid escape sequence `\\{}` in string literal",
875                            other.map(|c| c.to_string()).unwrap_or_default()
876                        ),
877                    )
878                    .with_note("supported escapes: \\n \\t \\\" \\\\"));
879                }
880            }
881        } else {
882            out.push(c);
883        }
884    }
885    Ok(out)
886}
887
888fn is_reserved_keyword(kind: TokenKind) -> bool {
889    use TokenKind::*;
890    matches!(
891        kind,
892        Commons
893            | Type
894            | Fn
895            | Where
896            | True
897            | False
898            | Int
899            | String
900            | Bool
901            | Let
902            | If
903            | Else
904            | Ok
905            | Err
906            | Result
907            | ValidationError
908            | Enum
909            | Match
910            | Option
911            | Record
912            | Self_
913            | Some
914            | None
915            | Is
916            | Opaque
917            | Uses
918            | Context
919            | Consumes
920            | Exports
921            | Transparent
922            | Agent
923            | As
924            | Capability
925            | Effect
926            | Do
927            | Given
928            | On
929            | Http
930            | Provides
931            | Stub
932            | Service
933            | Actor
934            | By
935            | Expect
936            | Suite
937            | Case
938            | Float
939            | Duration
940            | Instant
941            | Bytes
942            | JsonError
943            | Property
944            | Adapter
945            | Binding
946            | Cron
947            | Queue
948            | From
949            | Protocol
950            | Invariant
951            | Implies
952            | Requires
953            | Ensures
954            | Transition
955    )
956}
957
958/// True when `kind` starts a top-level unit (`commons`/`context`/`adapter`/
959/// `suite`) or an item within one of their bodies — every keyword any of
960/// `parse_commons_brace`/`_fragment`, `parse_context_brace`/`_fragment`,
961/// `parse_test_brace`/`_fragment`, or `parse_adapter_body` dispatches on
962/// (`declarations.rs`). The single set [`Parser::recover_to_top_item`]'s sync
963/// scan checks against — finding #27/#30: that scan had drifted from what the
964/// item loops actually recognise (`Property`, `Actor`, `Event`, `Binding`,
965/// and the `adapter` unit keyword itself were all missing), so an error
966/// recovery sync could walk past a real item/unit boundary instead of
967/// stopping there.
968fn is_item_start(kind: TokenKind) -> bool {
969    use TokenKind::*;
970    matches!(
971        kind,
972        // Top-level unit keywords.
973        Commons | Context | Adapter | Suite
974        // Body items shared across commons/context/adapter.
975        | Type | Fn | Messages | Event | Uses
976        // Context/adapter-only body items.
977        | Consumes | Exports | Capability | Provides | Service | Agent | Actor
978        // Adapter-only.
979        | Binding
980        // Suite/test-only body items.
981        | Stub | Case | Property
982    )
983}
984
985#[cfg(test)]
986mod tests {
987    use super::*;
988    use crate::lexer::tokenize;
989
990    fn parse_str(src: &str) -> Result<Commons, Vec<CompileError>> {
991        let toks = tokenize(src).map_err(|e| vec![e])?;
992        parse(&toks, src)
993    }
994
995    fn parse_recover_str(src: &str) -> (Option<SourceUnit>, Vec<CompileError>) {
996        let toks = match tokenize(src) {
997            Ok(t) => t,
998            Err(e) => return (None, vec![e]),
999        };
1000        parse_unit_with_recovery(&toks, src)
1001    }
1002
1003    /// Finding #29/#30: `parse_units_with_recovery` keeps every top-level unit
1004    /// an atomic `commons`+`suite` file declares (v0.113, DECISION S), where
1005    /// `parse_unit_with_recovery` keeps only the first — the defect the IDE's
1006    /// old single-unit parse entry point had (it silently discarded the
1007    /// trailing `suite`).
1008    #[test]
1009    fn parse_units_with_recovery_keeps_every_top_level_unit() {
1010        let src =
1011            "commons m {\n  fn f() -> Int { 1 }\n}\n\nsuite m\n\ncase \"c\" {\n  expect true\n}\n";
1012        let toks = tokenize(src).unwrap();
1013        let (units, errors) = parse_units_with_recovery(&toks, src);
1014        assert!(errors.is_empty(), "{errors:?}");
1015        assert_eq!(units.len(), 2, "expected both units, got {units:?}");
1016        assert!(matches!(units[0], SourceUnit::Commons(_)));
1017        assert!(matches!(units[1], SourceUnit::Suite(_)));
1018
1019        // The singular wrapper still narrows to just the first, unchanged.
1020        let (unit, errors) = parse_unit_with_recovery(&toks, src);
1021        assert!(errors.is_empty(), "{errors:?}");
1022        assert!(matches!(unit, Some(SourceUnit::Commons(_))));
1023    }
1024
1025    /// `parse_unit_with_recovery` always attempts at least one parse, even on
1026    /// empty input — `parse_units_with_recovery`'s loop must preserve that
1027    /// (its `while`-style peek check alone would skip the body entirely and
1028    /// silently return no error), since 16+ existing callers rely on an empty
1029    /// file still producing its usual diagnostic rather than a silent `None`
1030    /// with no error at all.
1031    #[test]
1032    fn empty_input_still_reports_an_error_through_the_plural_entry_point() {
1033        let toks = tokenize("").unwrap();
1034        let (units, errors) = parse_units_with_recovery(&toks, "");
1035        assert!(units.is_empty());
1036        assert!(
1037            !errors.is_empty(),
1038            "an empty file must still produce a diagnostic, not silently no units and no error"
1039        );
1040
1041        let (unit, unit_errors) = parse_unit_with_recovery(&toks, "");
1042        assert!(unit.is_none());
1043        assert_eq!(
1044            errors.len(),
1045            unit_errors.len(),
1046            "the singular wrapper must see the same error(s) as the plural entry point"
1047        );
1048    }
1049
1050    /// Finding #66: `parse_units_with_drain_check`'s `fully_drained` flag is
1051    /// `bynk-fmt`'s signal for whether its comment-loss guard can skip a
1052    /// re-tokenize-and-diff of its own output. It must be `true` for an
1053    /// ordinary file (every comment sits before a declaration/statement or
1054    /// trails one) and `false` the moment a comment sits inside an expression
1055    /// subtree, where `TriviaTable` has no field to attach it to.
1056    #[test]
1057    fn drain_check_reports_expression_interior_comments_as_undrained() {
1058        let ordinary = "commons x {\n-- note\ntype T = Int where Positive\n}\n";
1059        let toks = tokenize(ordinary).unwrap();
1060        let (_, _, drained) = parse_units_with_drain_check(&toks, ordinary).unwrap();
1061        assert!(
1062            drained,
1063            "a declaration-leading comment must be fully drained"
1064        );
1065
1066        let lossy = "commons x {\n  fn f() -> Int {\n    1 + -- note\n    2\n  }\n}\n";
1067        let toks = tokenize(lossy).unwrap();
1068        let (_, _, drained) = parse_units_with_drain_check(&toks, lossy).unwrap();
1069        assert!(
1070            !drained,
1071            "a comment inside a binop expression must be reported as undrained"
1072        );
1073    }
1074
1075    #[test]
1076    fn eof_span_never_splits_a_multibyte_codepoint() {
1077        // An unterminated construct whose final line ends in a non-ASCII char
1078        // (here a `--` comment ending in `→`) once produced an `unexpected_eof`
1079        // span of `len - 1 .. len`, landing on the arrow's last continuation
1080        // byte. Every reported span must sit on char boundaries.
1081        for src in [
1082            "commons x {\n  -- ends with an arrow →",
1083            "agent A {\n  key k: String\n  -- note 🦀",
1084            "commons y {\n  type T = é",
1085        ] {
1086            let (_unit, errors) = parse_recover_str(src);
1087            for e in &errors {
1088                assert!(
1089                    src.is_char_boundary(e.span.start) && src.is_char_boundary(e.span.end),
1090                    "span {:?} splits a codepoint in {src:?}",
1091                    e.span,
1092                );
1093            }
1094        }
1095    }
1096
1097    #[test]
1098    fn reserved_contextual_keywords_readable_in_expression_position() {
1099        // Events track, slice 0 (#939): `expect_ident`'s `RESERVED_CONTEXTUAL`
1100        // exemption (`keywords::RESERVED_CONTEXTUAL`: case/event/messages/on/
1101        // suite) covers *declaring* a binding with one of these names — a
1102        // parameter, a `let` — but the primary-expression parser previously
1103        // only admitted plain `TokenKind::Ident` when *reading one back*.
1104        // Latent since messages/on/case/suite shipped (no fixture happened to
1105        // name a binding after one of them and read it back inside the body);
1106        // surfaced concretely when `event` joined the tier and collided with
1107        // `examples/event-log`'s pre-existing `add(event: Event)` handler.
1108        for kw in ["case", "event", "messages", "on", "suite"] {
1109            let src = format!("commons x\n\nfn f({kw}: Int) -> Int {{\n  {kw}\n}}\n");
1110            let result = parse_str(&src);
1111            assert!(
1112                result.is_ok(),
1113                "a parameter named `{kw}` must be readable in expression position: {:?}",
1114                result.err()
1115            );
1116        }
1117    }
1118
1119    #[test]
1120    fn recovery_skips_garbage_between_decls() {
1121        // Two `type` declarations separated by garbage. Recovery should
1122        // accept both and report one error for the garbage between them.
1123        let src = "commons x {\n\
1124                   type A = Int where NonNegative\n\
1125                   ??? !!!\n\
1126                   type B = String where NonEmpty\n\
1127                   }";
1128        let (unit, errors) = parse_recover_str(src);
1129        let unit = unit.expect("recovery should produce a partial AST");
1130        let SourceUnit::Commons(c) = unit else {
1131            panic!("expected commons")
1132        };
1133        // Both type decls should have been collected despite the garbage.
1134        let names: Vec<_> = c
1135            .items
1136            .iter()
1137            .map(|i| match i {
1138                CommonsItem::Type(t) => t.name.name.clone(),
1139                _ => panic!("expected only types"),
1140            })
1141            .collect();
1142        assert!(
1143            names.contains(&"A".to_string()) && names.contains(&"B".to_string()),
1144            "expected both A and B; got {names:?}",
1145        );
1146        assert!(!errors.is_empty(), "expected at least one parse error");
1147    }
1148
1149    #[test]
1150    fn recovery_handles_bad_first_decl_then_good_second() {
1151        // First decl is malformed (missing `=`); second is well-formed.
1152        let src = "commons x {\n\
1153                   type A Int where NonNegative\n\
1154                   type B = String where NonEmpty\n\
1155                   }";
1156        let (unit, errors) = parse_recover_str(src);
1157        let unit = unit.expect("recovery should produce a partial AST");
1158        let SourceUnit::Commons(c) = unit else {
1159            panic!("expected commons")
1160        };
1161        let names: Vec<_> = c
1162            .items
1163            .iter()
1164            .filter_map(|i| match i {
1165                CommonsItem::Type(t) => Some(t.name.name.clone()),
1166                _ => None,
1167            })
1168            .collect();
1169        assert!(
1170            names.contains(&"B".to_string()),
1171            "B should be parsed after A's failure; got {names:?}"
1172        );
1173        assert!(!errors.is_empty(), "expected at least one parse error");
1174    }
1175
1176    /// Finding #27/#30: an error two levels deep inside `f`'s body (a
1177    /// `match` arm's own block) used to make `recover_to_top_item`'s flat,
1178    /// depth-blind scan stop at the *first* `}` it saw — the arm block's own,
1179    /// not `f`'s. Two more `}` (the match's, then `f`'s) then got consumed one
1180    /// at a time across repeated recovery re-entries, and the outer item loop
1181    /// eventually mistook the commons's *own* closing `}` for having arrived
1182    /// early, returning zero items and a spurious second
1183    /// `bynk.parse.expected_unit_header` error. With brace-depth tracking, `g`
1184    /// is recovered as the sole item and only `f`'s own error is reported.
1185    #[test]
1186    fn recovery_skips_a_nested_blocks_own_closing_brace() {
1187        let src = "commons m {\n  \
1188                   fn f() -> Int {\n    \
1189                   match 1 {\n      \
1190                   is 1 -> { let z = }\n      \
1191                   is _ -> 2\n    \
1192                   }\n  \
1193                   }\n  \
1194                   fn g() -> Int { 2 }\n\
1195                   }\n";
1196        let (unit, errors) = parse_recover_str(src);
1197        let unit = unit.expect("recovery should produce a partial AST");
1198        let SourceUnit::Commons(c) = unit else {
1199            panic!("expected commons")
1200        };
1201        let names: Vec<_> = c
1202            .items
1203            .iter()
1204            .filter_map(|i| match i {
1205                CommonsItem::Fn(f) => match &f.name {
1206                    FnName::Free(id) => Some(id.name.clone()),
1207                    _ => None,
1208                },
1209                _ => None,
1210            })
1211            .collect();
1212        assert_eq!(
1213            names,
1214            vec!["g".to_string()],
1215            "g must still be recovered as an item; got {names:?}"
1216        );
1217        assert!(
1218            !errors
1219                .iter()
1220                .any(|e| e.category == "bynk.parse.expected_unit_header"),
1221            "the outer body's own closing brace must not be mistaken for \
1222             end-of-file: {errors:?}"
1223        );
1224    }
1225
1226    #[test]
1227    fn doc_block_attaches_to_type() {
1228        let c =
1229            parse_str("commons x {\n---\nA descriptive doc.\n---\ntype T = Int where Positive\n}")
1230                .unwrap();
1231        let CommonsItem::Type(t) = &c.items[0] else {
1232            panic!()
1233        };
1234        assert!(t.documentation.is_some());
1235        assert!(
1236            t.documentation
1237                .as_ref()
1238                .unwrap()
1239                .contains("A descriptive doc.")
1240        );
1241    }
1242
1243    #[test]
1244    fn interpolated_string_parses_into_parts() {
1245        // v0.43: `"Hi, \(name)!"` splits into chunk / hole / chunk.
1246        let c = parse_str("commons x\n\nfn f(name: String) -> String {\n  \"Hi, \\(name)!\"\n}\n")
1247            .unwrap();
1248        let CommonsItem::Fn(f) = &c.items[0] else {
1249            panic!("expected fn")
1250        };
1251        let ExprKind::InterpStr(parts) = &f.body.tail.kind else {
1252            panic!("expected InterpStr, got {:?}", f.body.tail.kind)
1253        };
1254        assert_eq!(parts.len(), 3);
1255        assert!(matches!(&parts[0], InterpPart::Chunk(s) if s == "Hi, "));
1256        assert!(
1257            matches!(&parts[1], InterpPart::Hole(h) if matches!(&h.kind, ExprKind::Ident(id) if id.name == "name"))
1258        );
1259        assert!(matches!(&parts[2], InterpPart::Chunk(s) if s == "!"));
1260    }
1261
1262    #[test]
1263    fn interpolated_hole_parses_a_full_expression() {
1264        // A hole holds an arbitrary expression, not just an identifier.
1265        let c =
1266            parse_str("commons x\n\nfn f(a: Int, b: Int) -> String {\n  \"sum = \\(a + b)\"\n}\n")
1267                .unwrap();
1268        let CommonsItem::Fn(f) = &c.items[0] else {
1269            panic!("expected fn")
1270        };
1271        let ExprKind::InterpStr(parts) = &f.body.tail.kind else {
1272            panic!("expected InterpStr")
1273        };
1274        assert!(matches!(&parts[1], InterpPart::Hole(h) if matches!(&h.kind, ExprKind::BinOp(..))));
1275    }
1276
1277    #[test]
1278    fn empty_interpolation_hole_is_rejected() {
1279        let errs = parse_str("commons x\n\nfn f() -> String {\n  \"\\()\"\n}\n").unwrap_err();
1280        assert!(
1281            errs.iter()
1282                .any(|e| e.category == "bynk.parse.empty_interpolation"),
1283            "expected empty_interpolation; got {errs:?}"
1284        );
1285    }
1286
1287    #[test]
1288    fn interpolation_hole_lex_error_span_is_rebased() {
1289        // #716: a lex error inside a `\(…)` hole once carried a span relative to
1290        // the hole substring — never rebased by `hole.start` — so it pointed at
1291        // the file's opening bytes and could split a multibyte char, tripping
1292        // the char-boundary invariant. The error must land on the offending
1293        // bytes within the hole and stay on char boundaries.
1294        let cases = [
1295            // `$` is not a valid token; the error should point at it, not byte 0.
1296            "commons x\n\nfn f() -> String {\n  \"a \\($)\"\n}\n",
1297            // Integer overflow — the reported span must cover the literal itself.
1298            "commons x\n\nfn f() -> String {\n  \"n = \\(99999999999999999999)\"\n}\n",
1299            // A multibyte char before the hole means an un-rebased span could
1300            // land inside the `é`; the rebased span must not.
1301            "commons x\n\nfn f() -> String {\n  \"é \\($)\"\n}\n",
1302        ];
1303        for src in cases {
1304            let errs = parse_str(src).unwrap_err();
1305            assert!(!errs.is_empty(), "expected a lex error for {src:?}");
1306            for e in &errs {
1307                assert!(
1308                    src.is_char_boundary(e.span.start) && src.is_char_boundary(e.span.end),
1309                    "span {:?} splits a codepoint in {src:?}",
1310                    e.span,
1311                );
1312                // The error must point inside the interpolation hole, not at the
1313                // header text that precedes it.
1314                let hole_start = src.find("\\(").expect("case has a hole") + 2;
1315                assert!(
1316                    e.span.start >= hole_start,
1317                    "span {:?} precedes the hole (starts at {hole_start}) in {src:?}",
1318                    e.span,
1319                );
1320            }
1321        }
1322    }
1323
1324    #[test]
1325    fn fragment_form_parses() {
1326        let c = parse_str("commons x.y\n\ntype T = Int where NonNegative\n").unwrap();
1327        assert_eq!(c.form, CommonsForm::Fragment);
1328        assert_eq!(c.items.len(), 1);
1329    }
1330
1331    #[test]
1332    fn uses_parses() {
1333        let c = parse_str("commons x\n\nuses other.lib\n").unwrap();
1334        assert_eq!(c.uses.len(), 1);
1335        assert_eq!(c.uses[0].target.joined(), "other.lib");
1336    }
1337
1338    fn parse_unit_str(src: &str) -> Result<SourceUnit, Vec<CompileError>> {
1339        let toks = tokenize(src).map_err(|e| vec![e])?;
1340        parse_unit(&toks, src)
1341    }
1342
1343    #[test]
1344    fn minimal_context_parses() {
1345        let u = parse_unit_str("context commerce.orders {}").unwrap();
1346        let SourceUnit::Context(c) = u else {
1347            panic!("expected context");
1348        };
1349        assert_eq!(c.name.joined(), "commerce.orders");
1350        assert!(c.items.is_empty());
1351    }
1352
1353    #[test]
1354    fn context_consumes_and_exports_parse() {
1355        let src = "context commerce.orders {\n  uses commerce.money\n  consumes commerce.payment\n  exports opaque { OrderId }\n  exports transparent { OrderError }\n  type OrderId = String where Matches(\"ORD-[0-9]+\")\n  type OrderError = enum { CartEmpty, BadInput }\n}";
1356        let u = parse_unit_str(src).unwrap();
1357        let SourceUnit::Context(c) = u else { panic!() };
1358        assert_eq!(c.uses.len(), 1);
1359        assert_eq!(c.consumes.len(), 1);
1360        assert_eq!(c.exports.len(), 2);
1361        assert_eq!(c.exports[0].kind, ExportKind::Type(Visibility::Opaque));
1362        assert_eq!(c.exports[1].kind, ExportKind::Type(Visibility::Transparent));
1363    }
1364
1365    #[test]
1366    fn context_fragment_form_parses() {
1367        let src = "context x.y\n\nuses other.lib\nconsumes other.ctx\nexports opaque { T }\n\ntype T = Int where NonNegative\n";
1368        let u = parse_unit_str(src).unwrap();
1369        let SourceUnit::Context(c) = u else { panic!() };
1370        assert_eq!(c.form, CommonsForm::Fragment);
1371        assert_eq!(c.uses.len(), 1);
1372        assert_eq!(c.consumes.len(), 1);
1373        assert_eq!(c.exports.len(), 1);
1374    }
1375
1376    #[test]
1377    fn opaque_type_parses() {
1378        let c = parse_str("commons x { type T = opaque Int where NonNegative }").unwrap();
1379        let CommonsItem::Type(t) = &c.items[0] else {
1380            panic!()
1381        };
1382        assert!(matches!(t.body, TypeBody::Opaque { .. }));
1383    }
1384
1385    #[test]
1386    fn empty_commons() {
1387        let c = parse_str("commons fitness.units {}").unwrap();
1388        assert_eq!(c.name.joined(), "fitness.units");
1389        assert!(c.items.is_empty());
1390    }
1391
1392    #[test]
1393    fn one_type_decl() {
1394        let c = parse_str("commons x { type Metres = Int where NonNegative }").unwrap();
1395        assert_eq!(c.items.len(), 1);
1396        let CommonsItem::Type(t) = &c.items[0] else {
1397            panic!()
1398        };
1399        assert_eq!(t.name.name, "Metres");
1400        match &t.body {
1401            TypeBody::Refined {
1402                base, refinement, ..
1403            } => {
1404                assert_eq!(*base, BaseType::Int);
1405                assert!(refinement.is_some());
1406            }
1407            _ => panic!("expected refined body"),
1408        }
1409    }
1410
1411    #[test]
1412    fn function_decl() {
1413        let c = parse_str("commons x { fn add(a: Int, b: Int) -> Int { a + b } }").unwrap();
1414        let CommonsItem::Fn(f) = &c.items[0] else {
1415            panic!()
1416        };
1417        assert_eq!(f.name.ident().name, "add");
1418        assert_eq!(f.params.len(), 2);
1419    }
1420
1421    #[test]
1422    fn chained_comparison_is_error() {
1423        let errs = parse_str("commons x { fn f(a: Int, b: Int, c: Int) -> Bool { a < b < c } }")
1424            .unwrap_err();
1425        assert_eq!(errs[0].category, "bynk.parse.non_associative");
1426    }
1427
1428    #[test]
1429    fn chained_equality_is_error() {
1430        let errs = parse_str("commons x { fn f(a: Int, b: Int, c: Int) -> Bool { a == b == c } }")
1431            .unwrap_err();
1432        assert_eq!(errs[0].category, "bynk.parse.non_associative");
1433    }
1434
1435    /// Run `f` on a thread with a generous stack. The depth-guard tests build
1436    /// source that, *without* the guard, overflows — so if the guard ever
1437    /// regressed we want a clean assertion failure, not a `SIGABRT` that takes
1438    /// the whole test binary down. A large stack also absorbs the fat frames a
1439    /// debug build spends per recursion level (production release frames are
1440    /// ~9 KB/level, so `MAX_NESTING_DEPTH = 64` sits well inside a 1 MB stack;
1441    /// a debug frame is several times larger and would overflow libtest's
1442    /// default 2 MB test thread near the limit even though the guard fires).
1443    fn on_big_stack<T: Send + 'static>(f: impl FnOnce() -> T + Send + 'static) -> T {
1444        std::thread::Builder::new()
1445            .stack_size(64 * 1024 * 1024)
1446            .spawn(f)
1447            .unwrap()
1448            .join()
1449            .unwrap()
1450    }
1451
1452    #[test]
1453    fn deeply_nested_parens_are_bounded_not_overflowed() {
1454        // Without a depth guard the parenthesised-expression recursion
1455        // (`parse_primary` -> `parse_expr` -> …) overflows the stack and aborts
1456        // the process (#713). Well past the limit it must instead report a
1457        // bounded-depth diagnostic. The nesting is left open so the guard, not
1458        // a later `)`, is what stops the descent.
1459        let errs = on_big_stack(|| {
1460            let depth = crate::MAX_NESTING_DEPTH + 8;
1461            let src = format!(
1462                "commons x {{ fn f() -> Int {{ {}0{} }} }}",
1463                "(".repeat(depth),
1464                ")".repeat(depth),
1465            );
1466            parse_str(&src).unwrap_err()
1467        });
1468        assert_eq!(errs[0].category, "bynk.parse.nesting_too_deep");
1469    }
1470
1471    #[test]
1472    fn deeply_nested_types_are_bounded_not_overflowed() {
1473        // The type parser self-recurses through generic type arguments
1474        // (`parse_type_ref` -> `parse_type_atom` -> `parse_type_ref`); the same
1475        // guard bounds it (#713). A right-nested `Result[Int, …]` in parameter
1476        // position drives that recursion.
1477        let errs = on_big_stack(|| {
1478            let depth = crate::MAX_NESTING_DEPTH + 8;
1479            let src = format!(
1480                "commons x {{ fn f(x: {}Int{}) -> Int {{ 0 }} }}",
1481                "Result[Int, ".repeat(depth),
1482                "]".repeat(depth),
1483            );
1484            parse_str(&src).unwrap_err()
1485        });
1486        assert_eq!(errs[0].category, "bynk.parse.nesting_too_deep");
1487    }
1488
1489    #[test]
1490    fn deeply_nested_patterns_are_bounded_not_overflowed() {
1491        // Variant patterns are a third self-recursive descent (`parse_pattern`
1492        // -> `parse_pattern_binding` -> `parse_pattern`) that routes through
1493        // neither `parse_expr` nor `parse_type_ref`; without its own guard a
1494        // nested `Ok(Ok(…))` match arm reproduces the #713 crash.
1495        let errs = on_big_stack(|| {
1496            let depth = crate::MAX_NESTING_DEPTH + 8;
1497            let src = format!(
1498                "commons x {{ fn f(n: Int) -> Int {{ match n {{ {}n{} => 0 }} }} }}",
1499                "Ok(".repeat(depth),
1500                ")".repeat(depth),
1501            );
1502            parse_str(&src).unwrap_err()
1503        });
1504        assert_eq!(errs[0].category, "bynk.parse.nesting_too_deep");
1505    }
1506
1507    #[test]
1508    fn nesting_below_the_limit_still_parses() {
1509        // The guard must not reject ordinary well-nested source: a paren-nested
1510        // expression comfortably under the limit still parses cleanly.
1511        let ok = on_big_stack(|| {
1512            let depth = crate::MAX_NESTING_DEPTH - 8;
1513            let src = format!(
1514                "commons x {{ fn f() -> Int {{ {}0{} }} }}",
1515                "(".repeat(depth),
1516                ")".repeat(depth),
1517            );
1518            parse_str(&src).is_ok()
1519        });
1520        assert!(ok, "well-nested source under the limit should parse");
1521    }
1522
1523    #[test]
1524    fn let_statement_parses() {
1525        let c = parse_str("commons x { fn f(n: Int) -> Int { let y = n + 1\n y } }").unwrap();
1526        let CommonsItem::Fn(f) = &c.items[0] else {
1527            panic!()
1528        };
1529        assert_eq!(f.body.statements.len(), 1);
1530        match &f.body.statements[0] {
1531            Statement::Let(l) => {
1532                assert_eq!(l.name.name, "y");
1533                assert!(l.type_annot.is_none());
1534            }
1535            _ => panic!("expected a pure `let` statement"),
1536        }
1537    }
1538
1539    #[test]
1540    fn let_with_annotation() {
1541        let c = parse_str("commons x { fn f(n: Int) -> Int { let y: Int = n\n y } }").unwrap();
1542        let CommonsItem::Fn(f) = &c.items[0] else {
1543            panic!()
1544        };
1545        match &f.body.statements[0] {
1546            Statement::Let(l) => assert!(l.type_annot.is_some()),
1547            _ => panic!("expected a pure `let` statement"),
1548        }
1549    }
1550
1551    #[test]
1552    fn if_else_parses_as_expression() {
1553        let c = parse_str("commons x { fn f(b: Bool) -> Int { if b { 1 } else { 0 } } }").unwrap();
1554        let CommonsItem::Fn(f) = &c.items[0] else {
1555            panic!()
1556        };
1557        assert!(matches!(f.body.tail.kind, ExprKind::If { .. }));
1558    }
1559
1560    #[test]
1561    fn else_if_chain_parses() {
1562        let c = parse_str(
1563            "commons x { fn f(n: Int) -> Int { if n < 0 { -1 } else if n == 0 { 0 } else { 1 } } }",
1564        )
1565        .unwrap();
1566        let CommonsItem::Fn(f) = &c.items[0] else {
1567            panic!()
1568        };
1569        let ExprKind::If { else_block, .. } = &f.body.tail.kind else {
1570            panic!()
1571        };
1572        // The else-branch is a block whose tail is another `If`.
1573        assert!(else_block.statements.is_empty());
1574        assert!(matches!(else_block.tail.kind, ExprKind::If { .. }));
1575    }
1576
1577    #[test]
1578    fn ok_and_err_parse_as_expressions() {
1579        let c = parse_str("commons x { fn f(n: Int) -> Result[Int, String] { Ok(n) } }").unwrap();
1580        let CommonsItem::Fn(f) = &c.items[0] else {
1581            panic!()
1582        };
1583        assert!(matches!(f.body.tail.kind, ExprKind::Ok(_)));
1584
1585        let c =
1586            parse_str("commons x { fn f(n: Int) -> Result[Int, String] { Err(\"x\") } }").unwrap();
1587        let CommonsItem::Fn(f) = &c.items[0] else {
1588            panic!()
1589        };
1590        assert!(matches!(f.body.tail.kind, ExprKind::Err(_)));
1591    }
1592
1593    #[test]
1594    fn question_postfix_parses() {
1595        let c = parse_str(
1596            "commons x { type T = Int where Positive\n fn f(n: Int) -> Result[T, ValidationError] { let x = T.of(n)?\n Ok(x) } }",
1597        )
1598        .unwrap();
1599        let CommonsItem::Fn(f) = &c.items[1] else {
1600            panic!()
1601        };
1602        let Statement::Let(l) = &f.body.statements[0] else {
1603            panic!("expected a pure `let` statement");
1604        };
1605        assert!(matches!(l.value.kind, ExprKind::Question(_)));
1606    }
1607
1608    #[test]
1609    fn constructor_call_parses() {
1610        let c = parse_str(
1611            "commons x { type T = Int where Positive\n fn f(n: Int) -> Result[T, ValidationError] { T.of(n) } }",
1612        )
1613        .unwrap();
1614        let CommonsItem::Fn(f) = &c.items[1] else {
1615            panic!()
1616        };
1617        // v0.2: T.of(n) parses as a MethodCall with receiver Ident("T"); the
1618        // checker reinterprets it as a static call by noticing T is a type.
1619        let ExprKind::MethodCall {
1620            receiver, method, ..
1621        } = &f.body.tail.kind
1622        else {
1623            panic!("expected MethodCall, got {:?}", f.body.tail.kind)
1624        };
1625        let ExprKind::Ident(id) = &receiver.kind else {
1626            panic!("expected receiver Ident");
1627        };
1628        assert_eq!(id.name, "T");
1629        assert_eq!(method.name, "of");
1630    }
1631
1632    #[test]
1633    fn result_type_ref_parses() {
1634        let c = parse_str("commons x { fn f(n: Int) -> Result[Int, String] { Ok(n) } }").unwrap();
1635        let CommonsItem::Fn(f) = &c.items[0] else {
1636            panic!()
1637        };
1638        assert!(matches!(f.return_type, TypeRef::Result(_, _, _)));
1639    }
1640
1641    #[test]
1642    fn result_missing_arg_count_errors() {
1643        let errs = parse_str("commons x { fn f(n: Int) -> Result[Int] { Ok(n) } }").unwrap_err();
1644        assert_eq!(errs[0].category, "bynk.parse.generic_arg_count");
1645    }
1646
1647    #[test]
1648    fn field_access_parses_in_v0_2() {
1649        // v0.2: field access is supported (the type checker validates the
1650        // field exists on the receiver's type). Parser-level acceptance:
1651        let c =
1652            parse_str("commons x { type R = { foo: Int }\n fn f(r: R) -> Int { r.foo } }").unwrap();
1653        let CommonsItem::Fn(f) = &c.items[1] else {
1654            panic!()
1655        };
1656        assert!(matches!(f.body.tail.kind, ExprKind::FieldAccess { .. }));
1657    }
1658
1659    // -- v1.1 trivia attachment --
1660
1661    #[test]
1662    fn leading_line_comment_attaches_to_next_decl() {
1663        let src = "commons x {\n-- explain the type\ntype T = Int where NonNegative\n}";
1664        let c = parse_str(src).unwrap();
1665        let CommonsItem::Type(t) = &c.items[0] else {
1666            panic!()
1667        };
1668        assert_eq!(t.trivia.leading, vec![" explain the type".to_string()]);
1669        assert!(t.trivia.trailing.is_none());
1670    }
1671
1672    #[test]
1673    fn trailing_line_comment_attaches_to_prev_decl() {
1674        let src = "commons x {\ntype T = Int where NonNegative  -- trailing note\n}";
1675        let c = parse_str(src).unwrap();
1676        let CommonsItem::Type(t) = &c.items[0] else {
1677            panic!()
1678        };
1679        assert!(t.trivia.leading.is_empty());
1680        assert_eq!(t.trivia.trailing.as_deref(), Some(" trailing note"));
1681    }
1682
1683    #[test]
1684    fn grouped_leading_comments_attach_together() {
1685        let src = "commons x {\n-- one\n-- two\n-- three\ntype T = Int where Positive\n}";
1686        let c = parse_str(src).unwrap();
1687        let CommonsItem::Type(t) = &c.items[0] else {
1688            panic!()
1689        };
1690        assert_eq!(
1691            t.trivia.leading,
1692            vec![" one".to_string(), " two".to_string(), " three".to_string()],
1693        );
1694    }
1695
1696    #[test]
1697    fn comment_with_doc_block_keeps_both() {
1698        // Both `-- intro` and the doc block should attach to the type decl.
1699        let src = "commons x {\n-- intro\n---\ndocs\n---\ntype T = Int where Positive\n}";
1700        let c = parse_str(src).unwrap();
1701        let CommonsItem::Type(t) = &c.items[0] else {
1702            panic!()
1703        };
1704        assert_eq!(t.trivia.leading, vec![" intro".to_string()]);
1705        assert_eq!(t.documentation.as_deref(), Some("docs"));
1706    }
1707
1708    #[test]
1709    fn messages_keyword_does_not_collide_with_a_commons_name_segment() {
1710        // `messages` is RESERVED_CONTEXTUAL (like `case`/`on`/`suite`), not a
1711        // hard keyword: `commons app.messages { ... }` — the design's own
1712        // natural naming choice for a bundle commons — must still parse.
1713        // (Caught during slice-1 implementation: a first pass made `messages`
1714        // a plain hard keyword and this exact name broke.)
1715        let src = "commons app.messages {\ntype T = Int where Positive\n}";
1716        let c = parse_str(src).unwrap();
1717        assert_eq!(c.name.joined(), "app.messages");
1718    }
1719
1720    #[test]
1721    fn messages_decl_parses_tag_annotation_and_entries() {
1722        // message-bundles slice 1 (#859): the construct + doc/trivia wiring.
1723        let src = "commons app.messages {\n\
1724                   -- intro\n\
1725                   ---\n\
1726                   docs\n\
1727                   ---\n\
1728                   messages \"en\" @reference {\n\
1729                   \"greeting\" => \"Hello, {name}!\"\n\
1730                   \"farewell\" => \"Bye\"\n\
1731                   } -- trailing\n\
1732                   }";
1733        let c = parse_str(src).unwrap();
1734        let CommonsItem::Messages(m) = &c.items[0] else {
1735            panic!("expected a messages item, got {:?}", c.items[0]);
1736        };
1737        assert_eq!(m.tag, "en");
1738        assert_eq!(m.annotations.len(), 1);
1739        assert_eq!(m.annotations[0].name.name, "reference");
1740        assert!(m.annotations[0].args.is_empty());
1741        assert_eq!(m.entries.len(), 2);
1742        assert_eq!(m.entries[0].code, "greeting");
1743        assert_eq!(m.entries[0].template, "Hello, {name}!");
1744        assert_eq!(m.entries[1].code, "farewell");
1745        assert_eq!(m.entries[1].template, "Bye");
1746        assert_eq!(m.trivia.leading, vec![" intro".to_string()]);
1747        assert_eq!(m.documentation.as_deref(), Some("docs"));
1748        assert_eq!(m.trivia.trailing.as_deref(), Some(" trailing"));
1749    }
1750
1751    #[test]
1752    fn messages_decl_parses_with_no_annotation_and_no_entries() {
1753        // The parser stays permissive on annotation cardinality (zero-or-more)
1754        // — "exactly one `@reference`" is a checker concern (validate.rs), not
1755        // a parse error.
1756        let src = "commons app.messages {\nmessages \"en\" {\n}\n}";
1757        let c = parse_str(src).unwrap();
1758        let CommonsItem::Messages(m) = &c.items[0] else {
1759            panic!("expected a messages item, got {:?}", c.items[0]);
1760        };
1761        assert_eq!(m.tag, "en");
1762        assert!(m.annotations.is_empty());
1763        assert!(m.entries.is_empty());
1764    }
1765
1766    #[test]
1767    fn messages_decl_parses_syntactically_inside_a_context_too() {
1768        // Commons-only legality is a checker concern (bynk.messages.outside_commons
1769        // in bynk-emit's project validation), not a parser rejection — mirrors
1770        // how `service`/`agent` already parse syntactically inside `adapter`
1771        // bodies for the same reason.
1772        let src = "context app.svc {\nmessages \"en\" @reference {\n\"a\" => \"b\"\n}\n}";
1773        let toks = tokenize(src).unwrap();
1774        let (unit, errors) = parse_unit_with_recovery(&toks, src);
1775        assert!(errors.is_empty(), "unexpected parse errors: {errors:?}");
1776        let Some(SourceUnit::Context(ctx)) = unit else {
1777            panic!("expected a context")
1778        };
1779        let CommonsItem::Messages(m) = &ctx.items[0] else {
1780            panic!("expected a messages item, got {:?}", ctx.items[0]);
1781        };
1782        assert_eq!(m.tag, "en");
1783    }
1784
1785    #[test]
1786    fn comment_before_let_statement_attaches() {
1787        let src = "commons x {\nfn f(n: Int) -> Int {\n-- pick a value\nlet y = n + 1\ny\n}\n}";
1788        let c = parse_str(src).unwrap();
1789        let CommonsItem::Fn(f) = &c.items[0] else {
1790            panic!()
1791        };
1792        let Statement::Let(l) = &f.body.statements[0] else {
1793            panic!()
1794        };
1795        assert_eq!(l.trivia.leading, vec![" pick a value".to_string()]);
1796    }
1797
1798    #[test]
1799    fn comment_before_tail_attaches_to_block_tail() {
1800        let src = "commons x {\nfn f(n: Int) -> Int {\nlet y = n + 1\n-- result\ny\n}\n}";
1801        let c = parse_str(src).unwrap();
1802        let CommonsItem::Fn(f) = &c.items[0] else {
1803            panic!()
1804        };
1805        assert_eq!(f.body.tail_leading_comments, vec![" result".to_string()],);
1806    }
1807
1808    /// #637 Gap A: the contextual keywords `on` / `suite` / `case` are lexer
1809    /// tokens but `expect_ident` admits them as identifiers outside their one
1810    /// keyword position, so they are valid record-field and parameter names.
1811    /// The keyword reference now renders them as a distinct "contextual" tier
1812    /// rather than claiming (falsely) that they cannot be used as identifiers.
1813    #[test]
1814    fn contextual_keywords_are_valid_identifiers() {
1815        // Record field names.
1816        let c = parse_str("commons demo {\n  type R = { on: Int, suite: String, case: Bool }\n}")
1817            .expect("`on`/`suite`/`case` are valid field names");
1818        let CommonsItem::Type(_) = &c.items[0] else {
1819            panic!("expected a type decl")
1820        };
1821
1822        // Function parameter names (the other `expect_ident` position).
1823        parse_str("commons demo {\n  fn f(on: Int, case: Int) -> Int { 0 }\n}")
1824            .expect("`on`/`case` are valid parameter names");
1825
1826        // `suite` too, as a field name.
1827        parse_str("commons demo {\n  type R = { suite: Int }\n}")
1828            .expect("`suite` is a valid field name");
1829    }
1830
1831    /// Drift guard: every alphabetic keyword the lexer declares must be
1832    /// classified by `is_reserved_keyword`, or be one of the *contextual*
1833    /// keywords `expect_ident` deliberately admits as identifiers
1834    /// (`on`/`suite`/`case`). Everything else in this codebase that can
1835    /// drift has a guard; this predicate had silently fallen 17 keywords
1836    /// behind, degrading the reserved-keyword diagnostic to the generic
1837    /// expected-token one.
1838    #[test]
1839    fn is_reserved_keyword_covers_every_lexer_keyword() {
1840        let lexer_src = include_str!("lexer.rs");
1841        let mut words = Vec::new();
1842        for line in lexer_src.lines() {
1843            let t = line.trim();
1844            if let Some(rest) = t.strip_prefix("#[token(\"")
1845                && let Some(word) = rest.split('"').next()
1846                && word.chars().next().is_some_and(|c| c.is_ascii_alphabetic())
1847                && word.chars().all(|c| c.is_ascii_alphanumeric() || c == '_')
1848            {
1849                words.push(word.to_string());
1850            }
1851        }
1852        assert!(
1853            words.len() > 30,
1854            "keyword extraction looks broken: only {} words",
1855            words.len()
1856        );
1857        // Contextual keywords double as identifiers (see `expect_ident`); the
1858        // tier is single-sourced in `keywords::RESERVED_CONTEXTUAL`.
1859        use crate::keywords::RESERVED_CONTEXTUAL;
1860        let mut unclassified = Vec::new();
1861        for word in &words {
1862            let tokens = crate::lexer::tokenize(word).expect("keyword lexes");
1863            let kind = tokens.first().expect("keyword yields a token").kind;
1864            if !is_reserved_keyword(kind) && !RESERVED_CONTEXTUAL.contains(&word.as_str()) {
1865                unclassified.push(word.clone());
1866            }
1867        }
1868        assert!(
1869            unclassified.is_empty(),
1870            "keywords missing from is_reserved_keyword (add them, or document \
1871             them as contextual): {unclassified:?}"
1872        );
1873    }
1874
1875    /// Finding #27/#30: pins `is_item_start` to exactly the keyword set
1876    /// `declarations.rs`'s six item loops (`parse_commons_brace`/`_fragment`,
1877    /// `parse_context_brace`/`_fragment`, `parse_test_brace`/`_fragment`) plus
1878    /// `parse_adapter_body` dispatch on, as of this writing — the drift this
1879    /// finding fixed (`Property`, `Actor`, `Event`, `Binding`, and the
1880    /// `adapter` unit keyword itself were all missing from
1881    /// `recover_to_top_item`'s old hand-written sync list, even though every
1882    /// one of them is a real item/unit start). Adding a new item keyword to
1883    /// any of those loops should mean deliberately updating this list too,
1884    /// not silently leaving recovery unable to resync at it.
1885    #[test]
1886    fn is_item_start_matches_the_pinned_keyword_set() {
1887        use TokenKind::*;
1888        let expected_true = [
1889            Commons, Context, Adapter, Suite, Type, Fn, Messages, Event, Uses, Consumes, Exports,
1890            Capability, Provides, Service, Agent, Actor, Binding, Stub, Case, Property,
1891        ];
1892        for kind in expected_true {
1893            assert!(is_item_start(kind), "{kind:?} must be an item start");
1894        }
1895        let expected_false = [
1896            Ident, Plus, Minus, Colon, Dot, Eq, LBrace, RBrace, LParen, RParen, If, Else, Let,
1897            Where, True, False, Match, Is, On, Given,
1898        ];
1899        for kind in expected_false {
1900            assert!(!is_item_start(kind), "{kind:?} must not be an item start");
1901        }
1902    }
1903
1904    /// Fuzz-found (#516): a context-only keyword at item position in a
1905    /// commons errors without consuming the token, and the recovery sync
1906    /// stops at exactly that keyword — without a progress guard the item
1907    /// loop re-reported the same error until memory ran out.
1908    #[test]
1909    fn recovery_makes_progress_on_context_only_keyword_in_commons() {
1910        let src = "commons demo\n\ncapability Logger {\n  fn log(m: String) -> Effect[()]\n}\n";
1911        let tokens = crate::lexer::tokenize(src).unwrap();
1912        let (unit, errors) = parse_unit_with_recovery(&tokens, src);
1913        assert!(unit.is_some(), "the commons header still parses");
1914        assert!(
1915            errors
1916                .iter()
1917                .any(|e| e.category == "bynk.capability.outside_context"),
1918            "the misplaced capability is reported: {errors:?}"
1919        );
1920        // Termination is the real assertion (this used to OOM); a bounded,
1921        // non-repeating error list is the observable proxy.
1922        assert!(errors.len() < 10, "recovery repeated itself: {errors:?}");
1923    }
1924
1925    #[test]
1926    fn trailing_file_comment_becomes_unit_trailing() {
1927        // A comment after the last item but before EOF (fragment form)
1928        // becomes the commons body's trailing comments so the formatter
1929        // can preserve it.
1930        let src = "commons x\n\ntype T = Int where Positive\n-- afterword\n";
1931        let c = parse_str(src).unwrap();
1932        assert_eq!(c.trailing_comments, vec![" afterword".to_string()]);
1933    }
1934
1935    #[test]
1936    fn trailing_file_comment_after_a_brace_form_commons_is_not_dropped() {
1937        // Regression: the brace form's item loop exits on `RBrace`, never
1938        // reaching the fragment form's end-of-input case that drains the
1939        // trivia table's epilogue — so a comment after the closing `}` was
1940        // silently discarded (and, per `epilogue_is_empty`'s debug_assert,
1941        // would panic a debug build instead of round-tripping through
1942        // `bynk-fmt`).
1943        let src = "commons x {\n  type T = Int where Positive\n}\n-- afterword\n";
1944        let c = parse_str(src).unwrap();
1945        assert_eq!(c.trailing_comments, vec![" afterword".to_string()]);
1946    }
1947
1948    #[test]
1949    fn trailing_file_comment_after_a_brace_form_context_is_not_dropped() {
1950        // Same regression as the commons case, for `parse_context_brace`.
1951        let src = "context x {\n  type T = Int where Positive\n}\n-- afterword\n";
1952        let SourceUnit::Context(c) = parse_unit_str(src).unwrap() else {
1953            panic!("expected context");
1954        };
1955        assert_eq!(c.trailing_comments, vec![" afterword".to_string()]);
1956    }
1957
1958    #[test]
1959    fn trailing_file_comment_after_a_brace_form_suite_is_not_dropped() {
1960        // Same regression as the commons case, for `parse_test_brace`.
1961        let src = "suite x {\n  case \"c\" {\n    expect 1 == 1\n  }\n}\n-- afterword\n";
1962        let SourceUnit::Suite(s) = parse_unit_str(src).unwrap() else {
1963            panic!("expected suite");
1964        };
1965        assert_eq!(s.trailing_comments, vec![" afterword".to_string()]);
1966    }
1967
1968    /// Finding #30: unlike the three regressions just above,
1969    /// `parse_adapter_body`'s brace-closing path never called
1970    /// `take_epilogue` at all (not a regression from a shared pattern — it
1971    /// simply never had the call), so a comment after a brace-form adapter's
1972    /// closing `}` was silently dropped. The fragment form (no braces) was
1973    /// already correct.
1974    #[test]
1975    fn trailing_file_comment_after_a_brace_form_adapter_is_not_dropped() {
1976        let src = "adapter x {\n  binding \"./x.ts\"\n}\n-- afterword\n";
1977        let SourceUnit::Adapter(a) = parse_unit_str(src).unwrap() else {
1978            panic!("expected adapter");
1979        };
1980        assert_eq!(a.trailing_comments, vec![" afterword".to_string()]);
1981    }
1982
1983    // -- Six-fold unification (review Part 3): the fragment-only ordering
1984    // restrictions declarations.rs's brace/fragment pairs preserve, now that
1985    // they share one function each behind `brace: bool`. None of these had
1986    // any prior test coverage at all. --
1987
1988    /// Fragment-form commons: `uses` must precede every `type`/`fn`.
1989    #[test]
1990    fn commons_fragment_rejects_uses_after_a_decl() {
1991        let src = "commons x\n\ntype T = Int where Positive\nuses bynk.list\n";
1992        let errs = parse_str(src).unwrap_err();
1993        assert!(
1994            errs.iter()
1995                .any(|e| e.category == "bynk.parse.uses_after_decls"),
1996            "{errs:?}"
1997        );
1998    }
1999
2000    /// The same ordering is NOT enforced in brace form — `uses` may appear
2001    /// anywhere in the body.
2002    #[test]
2003    fn commons_brace_allows_uses_after_a_decl() {
2004        let src = "commons x {\n  type T = Int where Positive\n  uses bynk.list\n}\n";
2005        parse_str(src).expect("brace form must not enforce fragment's uses-ordering rule");
2006    }
2007
2008    /// Fragment-form context: `consumes` must precede every `type`/`fn`/etc.
2009    #[test]
2010    fn context_fragment_rejects_consumes_after_a_decl() {
2011        let src = "context x\n\ntype T = Int where Positive\nconsumes bynk\n";
2012        let errs = parse_unit_str(src).unwrap_err();
2013        assert!(
2014            errs.iter()
2015                .any(|e| e.category == "bynk.parse.consumes_after_decls"),
2016            "{errs:?}"
2017        );
2018    }
2019
2020    /// Fragment-form context: `exports` must precede every `type`/`fn`/etc.
2021    #[test]
2022    fn context_fragment_rejects_exports_after_a_decl() {
2023        let src = "context x\n\ntype T = Int where Positive\nexports opaque { T }\n";
2024        let errs = parse_unit_str(src).unwrap_err();
2025        assert!(
2026            errs.iter()
2027                .any(|e| e.category == "bynk.parse.exports_after_decls"),
2028            "{errs:?}"
2029        );
2030    }
2031
2032    /// Brace-form context enforces none of the three orderings.
2033    #[test]
2034    fn context_brace_allows_consumes_and_exports_after_a_decl() {
2035        let src = "context x {\n  type T = Int where Positive\n  consumes bynk\n  exports opaque { T }\n}\n";
2036        parse_unit_str(src)
2037            .expect("brace form must not enforce fragment's consumes/exports-ordering rules");
2038    }
2039
2040    /// Fragment-form suite/test: `uses` must precede every `stub`/`case`/`property`.
2041    #[test]
2042    fn test_fragment_rejects_uses_after_a_decl() {
2043        let src = "suite m\n\ncase \"c\" {\n  expect true\n}\nuses bynk.list\n";
2044        let errs = parse_unit_str(src).unwrap_err();
2045        assert!(
2046            errs.iter()
2047                .any(|e| e.category == "bynk.parse.uses_after_decls"),
2048            "{errs:?}"
2049        );
2050    }
2051
2052    /// Brace-form suite/test allows `uses` anywhere.
2053    #[test]
2054    fn test_brace_allows_uses_after_a_decl() {
2055        let src = "suite m {\n  case \"c\" {\n    expect true\n  }\n  uses bynk.list\n}\n";
2056        parse_unit_str(src).expect("brace form must not enforce fragment's uses-ordering rule");
2057    }
2058
2059    // ---- #636: `if`/`match` condition vs record construction ----
2060
2061    /// Parse `body` as the tail expression of a fn and return its kind.
2062    fn body_tail(body: &str) -> ExprKind {
2063        let src = format!("commons x\n\nfn f() -> Int {{\n  {body}\n}}\n");
2064        let c = parse_str(&src).unwrap_or_else(|e| panic!("parse failed for {body:?}: {e:?}"));
2065        let CommonsItem::Fn(f) = &c.items[0] else {
2066            panic!("expected fn, got {:?}", c.items[0]);
2067        };
2068        f.body.tail.kind.clone()
2069    }
2070
2071    fn body_err(body: &str) -> Vec<CompileError> {
2072        let src = format!("commons x\n\nfn f() -> Int {{\n  {body}\n}}\n");
2073        parse_str(&src).expect_err(&format!("expected a parse error for {body:?}"))
2074    }
2075
2076    #[test]
2077    fn if_condition_ending_in_ident_does_not_swallow_a_single_ident_branch() {
2078        // #636: `ready { result }` shares its shape with a shorthand-field
2079        // record construction. In condition position the branch must win.
2080        for src in [
2081            "if ready { result } else { fallback }",
2082            "if ready { fallback } else { result }",
2083            "if !ready { result } else { fallback }",
2084            "if a == b { result } else { fallback }",
2085            "if a && b { result } else { fallback }",
2086        ] {
2087            let ExprKind::If {
2088                then_block,
2089                else_block,
2090                ..
2091            } = body_tail(src)
2092            else {
2093                panic!("expected If for {src:?}, got {:?}", body_tail(src));
2094            };
2095            // Both branches carry a bare-identifier tail — proof the `{ … }`
2096            // was read as a block, not consumed as a record by the condition.
2097            assert!(
2098                matches!(&then_block.tail.kind, ExprKind::Ident(_)),
2099                "then-branch tail not an ident for {src:?}: {:?}",
2100                then_block.tail.kind,
2101            );
2102            assert!(
2103                matches!(&else_block.tail.kind, ExprKind::Ident(_)),
2104                "else-branch tail not an ident for {src:?}: {:?}",
2105                else_block.tail.kind,
2106            );
2107        }
2108    }
2109
2110    #[test]
2111    fn else_less_if_with_single_ident_branch_parses() {
2112        // The no-`else` reproduction: previously errored `found `}``.
2113        let ExprKind::If { then_block, .. } = body_tail("if ready { result }") else {
2114            panic!("expected If");
2115        };
2116        assert!(matches!(&then_block.tail.kind, ExprKind::Ident(_)));
2117    }
2118
2119    #[test]
2120    fn record_construction_still_parses_in_value_position() {
2121        // The restriction is confined to condition spines — an ordinary value
2122        // position still constructs records, including the shorthand tail form.
2123        assert!(matches!(
2124            body_tail("Point { x }"),
2125            ExprKind::RecordConstruction { .. }
2126        ));
2127        assert!(matches!(
2128            body_tail("Point { x: 1, y: 2 }"),
2129            ExprKind::RecordConstruction { .. }
2130        ));
2131        assert!(matches!(
2132            body_tail("Empty {}"),
2133            ExprKind::RecordConstruction { .. }
2134        ));
2135    }
2136
2137    #[test]
2138    fn parenthesised_record_is_allowed_in_condition_head() {
2139        // A delimiter lifts the restriction: `(ready { result })` constructs a
2140        // record even in condition position (mirrors Rust's paren escape).
2141        let ExprKind::If { cond, .. } =
2142            body_tail("if (ready { result }) { branch } else { other }")
2143        else {
2144            panic!("expected If");
2145        };
2146        let ExprKind::Paren(inner) = &cond.kind else {
2147            panic!("expected a parenthesised condition, got {:?}", cond.kind);
2148        };
2149        assert!(
2150            matches!(&inner.kind, ExprKind::RecordConstruction { .. }),
2151            "parenthesised record in condition head should still construct: {:?}",
2152            inner.kind,
2153        );
2154    }
2155
2156    #[test]
2157    fn record_in_call_arg_within_condition_still_constructs() {
2158        // The restriction is lifted through a call-argument delimiter, so a
2159        // record literal passed to a predicate in the condition still parses.
2160        let ExprKind::If { cond, .. } = body_tail("if check(Point { x: 1 }) { a } else { b }")
2161        else {
2162            panic!("expected If");
2163        };
2164        let ExprKind::Call { args, .. } = &cond.kind else {
2165            panic!("expected Call in condition, got {:?}", cond.kind);
2166        };
2167        assert!(matches!(&args[0].kind, ExprKind::RecordConstruction { .. }));
2168    }
2169
2170    #[test]
2171    fn safe_condition_shapes_are_unaffected() {
2172        // Cases the issue lists as already-safe must stay safe.
2173        assert!(matches!(
2174            body_tail("if ready == true { result } else { fallback }"),
2175            ExprKind::If { .. }
2176        ));
2177        assert!(matches!(
2178            body_tail("if (ready) { result } else { fallback }"),
2179            ExprKind::If { .. }
2180        ));
2181        assert!(matches!(
2182            body_tail("if ready { \"a\" } else { \"b\" }"),
2183            ExprKind::If { .. }
2184        ));
2185    }
2186
2187    #[test]
2188    fn empty_match_reports_its_own_diagnostic() {
2189        // #636: `match result {}` once parsed `result {}` as an empty record,
2190        // masking `bynk.parse.empty_match`. The intended diagnostic is now
2191        // reachable.
2192        let errs = body_err("match result {}");
2193        assert!(
2194            errs.iter().any(|e| e.category == "bynk.parse.empty_match"),
2195            "expected empty_match; got {errs:?}",
2196        );
2197    }
2198
2199    #[test]
2200    fn match_discriminant_ending_in_ident_parses() {
2201        // A `match` over a bare-identifier discriminant reaches its arm list.
2202        assert!(matches!(
2203            body_tail("match ready { x => x }"),
2204            ExprKind::Match { .. }
2205        ));
2206    }
2207
2208    #[test]
2209    fn unparenthesised_record_in_condition_head_now_errors() {
2210        // #636 narrowing (matches Rust): a record literal in condition *head*
2211        // position must be parenthesised. Unparenthesised, `Point` reads as the
2212        // discriminant and `{ x: 1 }` as the arm list, whose first "arm" `x: 1`
2213        // is not an arm — so the parse fails. Pinned so the divergence from the
2214        // (still-accepting) tree-sitter grammar is deliberate, not a bug.
2215        assert!(
2216            !body_err("match Point { x: 1 } { p => p }").is_empty(),
2217            "unparenthesised record discriminant should not parse",
2218        );
2219        // Parenthesised, the record is the discriminant and the match parses.
2220        let ExprKind::Match { discriminant, .. } = body_tail("match (Point { x: 1 }) { p => p }")
2221        else {
2222            panic!("expected Match for the parenthesised form");
2223        };
2224        let ExprKind::Paren(inner) = &discriminant.kind else {
2225            panic!(
2226                "expected a parenthesised discriminant, got {:?}",
2227                discriminant.kind
2228            );
2229        };
2230        assert!(matches!(&inner.kind, ExprKind::RecordConstruction { .. }));
2231    }
2232}