Skip to main content

bynk_syntax/
parser.rs

1//! Hand-written recursive-descent parser for Bynk v0.
2//!
3//! Token grammar in spec §4. The expression parser uses one function per
4//! precedence level (§4.4). Errors carry spans and short fix-oriented
5//! messages; the parser does not currently attempt synchronisation, which
6//! means at most one parse error is reported per compilation.
7
8use crate::ast::*;
9use crate::error::CompileError;
10use crate::lexer::{Token, TokenKind, comment_body, doc_block_content, has_blank_line_between};
11use crate::span::Span;
12mod declarations;
13mod expressions;
14mod statements;
15mod types;
16
17/// Side-channel store for line-comment trivia (v1.1 LSP spec §3.5).
18///
19/// Built once up-front by [`split_trivia`] from the raw lexer token stream.
20/// Comments are removed from the token stream the parser walks; their text
21/// is filed into `leading` (comments on lines preceding a content token)
22/// and `trailing` (a single comment on the same line as a content token).
23/// The parser consumes entries through [`TriviaTable::take_leading`] and
24/// [`TriviaTable::take_trailing`] as it recognises declarations.
25#[derive(Debug, Default)]
26struct TriviaTable {
27    /// `leading[i]` holds the comment-body texts that appear immediately
28    /// before content token `i` (zero or more `--` lines, in source order,
29    /// not separated from the token by another content token).
30    leading: Vec<Vec<String>>,
31    /// `trailing[i]` holds an optional comment on the same source line as
32    /// content token `i`. Only one trailing comment is recorded per token
33    /// because a single `--` consumes the rest of the line.
34    trailing: Vec<Option<String>>,
35    /// Any pending leading comments at end-of-file (no content token
36    /// followed). Used to preserve file-trailing comments.
37    epilogue: Vec<String>,
38}
39
40impl TriviaTable {
41    fn take_leading(&mut self, index: usize) -> Vec<String> {
42        match self.leading.get_mut(index) {
43            Some(v) => std::mem::take(v),
44            None => Vec::new(),
45        }
46    }
47
48    fn take_trailing(&mut self, index: usize) -> Option<String> {
49        self.trailing.get_mut(index).and_then(|s| s.take())
50    }
51
52    fn take_epilogue(&mut self) -> Vec<String> {
53        std::mem::take(&mut self.epilogue)
54    }
55
56    /// True when every entry has been drained via `take_leading`/
57    /// `take_trailing`/`take_epilogue` — i.e. no comment was silently
58    /// dropped. Each harvest is already a `mem::take`, so anything still
59    /// present here is exactly the set of comments that never reached an
60    /// AST `Trivia` field.
61    ///
62    /// Deliberately **not** wired into a `debug_assert!` in the general parse
63    /// path: expressions carry no per-node trivia (§ "Comment trivia" in the
64    /// 2026-07-27 pipeline review), so an ordinary, valid program with a
65    /// comment inside a `match`/list/record/binop — a common, accepted
66    /// pattern `bynk-fmt`'s own comment-loss guard already handles
67    /// gracefully — would leave `leading`/`trailing` non-empty and trip it on
68    /// every compile, not just on formatting. Instead surfaced through
69    /// [`parse_units_with_drain_check`] (finding #66), whose one caller
70    /// (`bynk-fmt`) *does* care about exactly this signal.
71    /// [`Self::epilogue_is_empty`] is the narrower, safe-to-assert check.
72    fn is_fully_drained(&self) -> bool {
73        self.leading.iter().all(Vec::is_empty)
74            && self.trailing.iter().all(Option::is_none)
75            && self.epilogue.is_empty()
76    }
77
78    /// True when no file-trailing comment was left stranded. Unlike
79    /// [`Self::is_fully_drained`], this is safe to assert unconditionally: a
80    /// clean file's epilogue is empty by construction (nothing pending at
81    /// EOF), and the one shape that legitimately populates it — a top-level
82    /// trailing comment — is drained by every parse path that calls
83    /// `take_epilogue`. A brace-form declaration that forgets to is exactly
84    /// the bug this catches.
85    fn epilogue_is_empty(&self) -> bool {
86        self.epilogue.is_empty()
87    }
88}
89
90/// Remove `Comment` trivia tokens from `tokens` and bin them into a
91/// [`TriviaTable`] keyed against the surviving content tokens. A comment
92/// on the same source line as the preceding content token is recorded as
93/// that token's *trailing* trivia; everything else is *leading* for the
94/// next content token.
95fn split_trivia(tokens: &[Token], source: &str) -> (Vec<Token>, TriviaTable) {
96    let mut filtered: Vec<Token> = Vec::with_capacity(tokens.len());
97    let mut table = TriviaTable::default();
98    let mut pending_leading: Vec<String> = Vec::new();
99    let mut last_content_end: Option<usize> = None;
100    for tok in tokens {
101        if tok.kind == TokenKind::Comment {
102            let body = comment_body(source, tok.span).to_string();
103            // If nothing has been buffered as leading for the next token and
104            // there is no newline between the previous content token and
105            // this comment, it trails that token.
106            if pending_leading.is_empty()
107                && let Some(prev_end) = last_content_end
108                && !source[prev_end..tok.span.start].contains('\n')
109            {
110                let last_idx = filtered.len() - 1;
111                // Only attach if no trailing already recorded (shouldn't
112                // happen because `--` consumes through end-of-line).
113                if table.trailing[last_idx].is_none() {
114                    table.trailing[last_idx] = Some(body);
115                    continue;
116                }
117            }
118            pending_leading.push(body);
119            continue;
120        }
121        filtered.push(*tok);
122        table.leading.push(std::mem::take(&mut pending_leading));
123        table.trailing.push(None);
124        last_content_end = Some(tok.span.end);
125    }
126    table.epilogue = pending_leading;
127    (filtered, table)
128}
129
130/// Parse a token slice into a [`Commons`] AST.
131///
132/// Accepts either form of v0.3 commons file:
133/// - Brace form: `commons name { items... }` (v0–v0.2 compatible).
134/// - Fragment form: `commons name uses... items...` to EOF (v0.3).
135pub fn parse(tokens: &[Token], source: &str) -> Result<Commons, Vec<CompileError>> {
136    parse_with_warnings(tokens, source).map(|(c, _warnings)| c)
137}
138
139/// [`parse`] with the non-fatal diagnostics threaded out alongside the AST
140/// (ADR 0117) — see [`parse_units_with_warnings`].
141pub fn parse_with_warnings(
142    tokens: &[Token],
143    source: &str,
144) -> Result<(Commons, Vec<CompileError>), Vec<CompileError>> {
145    let (unit, warnings) = parse_unit_with_warnings(tokens, source)?;
146    match unit {
147        SourceUnit::Commons(c) => Ok((c, warnings)),
148        SourceUnit::Context(ctx) => Err(vec![
149            CompileError::new(
150                "bynk.parse.unexpected_context",
151                ctx.span,
152                "expected a `commons` declaration but found a `context` declaration",
153            )
154            .with_note(
155                "contexts must be compiled as part of a project — pass the source directory, e.g. `bynkc compile --target bundle --output out src`",
156            ),
157        ]),
158        SourceUnit::Suite(t) => Err(vec![
159            CompileError::new(
160                "bynk.parse.unexpected_suite",
161                t.span,
162                "expected a `commons` declaration but found a `suite` declaration",
163            )
164            .with_note(
165                "tests must be compiled as part of a project — pass the source directory, e.g. `bynkc compile --target bundle --output out src`",
166            ),
167        ]),
168        SourceUnit::Adapter(a) => Err(vec![
169            CompileError::new(
170                "bynk.parse.unexpected_adapter",
171                a.span,
172                "expected a `commons` declaration but found an `adapter` declaration",
173            )
174            .with_note(
175                "adapters must be compiled as part of a project — pass the source directory, e.g. `bynkc compile --target bundle --output out src`",
176            ),
177        ]),
178    }
179}
180
181/// Parse a token slice into a [`SourceUnit`] with error recovery, returning a
182/// best-effort partial AST plus the full list of parse errors and warnings.
183///
184/// Used by the LSP: item-level recovery skips past a malformed declaration to
185/// the next top-level item, so multiple errors are reported per compilation
186/// rather than just the first. Compared to [`parse_unit`], this never bails;
187/// if no SourceUnit could be parsed at all (e.g. the file is empty or the
188/// header itself fails) the returned `Option` is `None`.
189///
190/// Keeps only the *first* unit — v0.113 allows more than one top-level unit
191/// per file (an atomic `commons` + `suite`, DECISION S), and every existing
192/// caller here is keyed on the primary declaration. [`parse_units_with_recovery`]
193/// is the same recovery parse without that narrowing, for the one caller
194/// (finding #29/#30) that needs every unit a file declares.
195pub fn parse_unit_with_recovery(
196    tokens: &[Token],
197    source: &str,
198) -> (Option<SourceUnit>, Vec<CompileError>) {
199    let (units, errors) = parse_units_with_recovery(tokens, source);
200    (units.into_iter().next(), errors)
201}
202
203/// [`parse_unit_with_recovery`], keeping **every** top-level unit instead of
204/// discarding all but the first (finding #29/#30). Used by the IDE's own parse
205/// entry point, which needs to see a trailing `suite` in an atomic
206/// `commons`+`suite` file, not just the primary declaration.
207pub fn parse_units_with_recovery(
208    tokens: &[Token],
209    source: &str,
210) -> (Vec<SourceUnit>, Vec<CompileError>) {
211    let recovered = parse_units_recovering(tokens, source);
212    (recovered.units, recovered.errors)
213}
214
215/// #1663: everything a recovering parse learns — the units it built, every
216/// syntax error, and the names of the top-level declarations it had to skip.
217#[derive(Debug)]
218pub struct Recovered {
219    pub units: Vec<SourceUnit>,
220    pub errors: Vec<CompileError>,
221    /// Declarations recovery dropped, by name (see `Parser::broken_decl_names`).
222    /// A caller keeps references to them from echoing as unknown names.
223    pub broken_decl_names: Vec<String>,
224}
225
226/// #1663: a strict parse's error(s) together with a recovering parse's, each
227/// reported once. The strict errors come first and always survive: some rules
228/// hold only for the strict single-unit parse (`extra_tokens`, a `commons`
229/// expected but a `context` found), which the multi-unit recovering parse
230/// accepts. A recovered error at the same position as one already listed is
231/// the same fault (the two parsers may name it differently) and is dropped.
232pub fn merge_syntax_errors(
233    strict: Vec<CompileError>,
234    recovered: Vec<CompileError>,
235) -> Vec<CompileError> {
236    let mut out = strict;
237    for e in recovered {
238        if !out.iter().any(|o| o.span.start == e.span.start) {
239            out.push(e);
240        }
241    }
242    out
243}
244
245/// [`parse_units_with_recovery`], also returning the names of the declarations
246/// recovery skipped (#1663).
247pub fn parse_units_recovering(tokens: &[Token], source: &str) -> Recovered {
248    parse_units_recovering_from(tokens, source, &mut 0)
249}
250
251/// [`parse_units_recovering`], continuing [`ExprId`] allocation from `next_id`
252/// rather than starting at 0 — see [`parse_unit_with_warnings_from`]. #1710:
253/// the project path checks a recovered file's surviving declarations
254/// alongside every other file's, so its ids must come from the same durable
255/// counter (`bynk_project::parse_cache`), or they would collide.
256pub fn parse_units_recovering_from(tokens: &[Token], source: &str, next_id: &mut u32) -> Recovered {
257    let (filtered, trivia) = split_trivia(tokens, source);
258    let mut warnings = Vec::new();
259    let mut p = Parser::new(&filtered, source, trivia, &mut warnings);
260    p.recover_mode = true;
261    p.next_expr_id = *next_id;
262    let mut units = Vec::new();
263    loop {
264        match p.parse_unit() {
265            Ok(u) => units.push(u),
266            Err(e) => {
267                p.recovered_errors.push(e);
268                break;
269            }
270        }
271        // A genuinely malformed trailing declaration is still surfaced via
272        // recovery — checked *after* each successful parse, matching
273        // `parse_unit`'s own "at least once" attempt on the first unit (an
274        // empty file must still produce its usual unexpected-EOF diagnostic,
275        // not silently yield an empty `units` with no error at all).
276        if p.peek().is_none() {
277            break;
278        }
279    }
280    let broken_decl_names = std::mem::take(&mut p.broken_decl_names);
281    *next_id = p.next_expr_id;
282    let mut all_errors = p.recovered_errors;
283    all_errors.append(&mut warnings);
284    Recovered {
285        units,
286        errors: all_errors,
287        broken_decl_names,
288    }
289}
290
291/// Parse a token slice into a [`SourceUnit`] — either a commons or a context.
292///
293/// Each `.bynk` file is exactly one declaration of one kind.
294pub fn parse_unit(tokens: &[Token], source: &str) -> Result<SourceUnit, Vec<CompileError>> {
295    parse_unit_with_warnings(tokens, source).map(|(unit, _warnings)| unit)
296}
297
298/// [`parse_unit`] with the non-fatal diagnostics threaded out alongside the
299/// AST (ADR 0117) — see [`parse_units_with_warnings`].
300pub fn parse_unit_with_warnings(
301    tokens: &[Token],
302    source: &str,
303) -> Result<(SourceUnit, Vec<CompileError>), Vec<CompileError>> {
304    parse_unit_with_warnings_from(tokens, source, &mut 0)
305}
306
307/// [`parse_unit_with_warnings`], continuing [`ExprId`] allocation from
308/// `next_id` instead of starting at 0, and writing the id one past the last
309/// one this parse handed out back into it. T3.4 (R2.4): every top-level parse
310/// entry point in this file constructs its own `Parser` and therefore its own
311/// zero-based id space; a caller that will check two files' output together
312/// in one pass (a multi-file commons — `bynk-emit`'s `collect_unit_methods`
313/// merges a type's methods from sibling files into the file that declares the
314/// type, before one `check_record` call) must thread one counter across every
315/// file it parses, or two independently-numbered files collide on the same
316/// id in the same `expr_types` map. Every *other* caller (a single buffer, an
317/// LSP hover/completion query, a fixture test) never merges its output with
318/// another file's before checking, so starting at 0 every time is correct —
319/// [`parse_unit_with_warnings`] above is that default, unchanged.
320pub fn parse_unit_with_warnings_from(
321    tokens: &[Token],
322    source: &str,
323    next_id: &mut u32,
324) -> Result<(SourceUnit, Vec<CompileError>), Vec<CompileError>> {
325    let (filtered, trivia) = split_trivia(tokens, source);
326    let mut warnings = Vec::new();
327    let mut p = Parser::new(&filtered, source, trivia, &mut warnings);
328    p.next_expr_id = *next_id;
329    let result = match p.parse_unit() {
330        Ok(u) => {
331            if let Some(extra) = p.peek() {
332                Err(vec![
333                    CompileError::new(
334                        "bynk.parse.extra_tokens",
335                        extra.span,
336                        "unexpected token after top-level declaration",
337                    )
338                    .with_note(
339                        "a `.bynk` file contains exactly one `commons` or `context` declaration",
340                    ),
341                ])
342            } else {
343                Ok(u)
344            }
345        }
346        Err(e) => Err(vec![e]),
347    };
348    *next_id = p.next_expr_id;
349    // ADR 0117: warnings (e.g. orphan doc blocks) ride alongside a successful
350    // parse — severity governs gating at the caller, not here.
351    match result {
352        Ok(u) => {
353            // See `parse_units_with_warnings`: a file-trailing comment must
354            // have been drained by `take_epilogue`.
355            debug_assert!(
356                p.trivia.epilogue_is_empty(),
357                "a file-trailing comment was left undrained after a successful parse"
358            );
359            Ok((u, warnings))
360        }
361        Err(mut errs) => {
362            errs.append(&mut warnings);
363            Err(errs)
364        }
365    }
366}
367
368/// Parse a token slice into **all** the top-level [`SourceUnit`]s in one file
369/// (v0.113, testing track slice 1b). A `.bynk` file may hold more than one
370/// top-level declaration — an *atomic* file with `commons`/`context` **and** a
371/// `suite` together (DECISION S) — so the compiler parses a `Vec`, not a single
372/// unit. Test-ness is a property of each declaration, not of the file.
373///
374/// Bails on the first malformed declaration (like [`parse_unit`], not the
375/// recovering LSP path). An empty file is an error.
376pub fn parse_units(tokens: &[Token], source: &str) -> Result<Vec<SourceUnit>, Vec<CompileError>> {
377    parse_units_with_warnings(tokens, source).map(|(units, _warnings)| units)
378}
379
380/// [`parse_units`] with the non-fatal diagnostics threaded out alongside the
381/// AST (ADR 0117): a successful parse returns `Ok((units, warnings))` instead
382/// of hard-failing on a warning-severity diagnostic (an orphan doc block used
383/// to abort file discovery and throw the good AST away). A failed parse still
384/// returns every diagnostic — errors then warnings — in the `Err`.
385pub fn parse_units_with_warnings(
386    tokens: &[Token],
387    source: &str,
388) -> Result<(Vec<SourceUnit>, Vec<CompileError>), Vec<CompileError>> {
389    parse_units_with_drain_check(tokens, source)
390        .map(|(units, warnings, _drained)| (units, warnings))
391}
392
393/// [`parse_units_with_warnings`], continuing [`ExprId`] allocation from
394/// `next_id` rather than starting at 0 — see [`parse_unit_with_warnings_from`]
395/// for why this exists. The one production caller is `bynk-emit`'s per-file
396/// parse loop (`phase_parse`), which owns one counter across every file in a
397/// single project parse so two files whose methods later get merged into one
398/// `check_record` call (a multi-file commons) never collide.
399pub fn parse_units_with_warnings_from(
400    tokens: &[Token],
401    source: &str,
402    next_id: &mut u32,
403) -> Result<(Vec<SourceUnit>, Vec<CompileError>), Vec<CompileError>> {
404    parse_units_with_drain_check_from(tokens, source, next_id)
405        .map(|(units, warnings, _drained)| (units, warnings))
406}
407
408/// [`parse_units_with_warnings`] plus whether every comment's trivia was
409/// drained into the AST (`TriviaTable::is_fully_drained`). Finding #66:
410/// `bynk-fmt`'s comment-preservation guard re-tokenized its own rendered
411/// output just to diff comment bodies against the input — wasted work in the
412/// overwhelming common case where nothing was left behind. That guard uses
413/// this drain signal, computed from the same parse it already needs for
414/// rendering, as a fast-path: `true` means every comment landed in the AST,
415/// so re-checking the output can be skipped outright. No other caller needs
416/// the signal, so it rides its own entry point rather than widening
417/// [`parse_units_with_warnings`].
418pub fn parse_units_with_drain_check(
419    tokens: &[Token],
420    source: &str,
421) -> Result<(Vec<SourceUnit>, Vec<CompileError>, bool), Vec<CompileError>> {
422    parse_units_with_drain_check_from(tokens, source, &mut 0)
423}
424
425/// [`parse_units_with_drain_check`], continuing [`ExprId`] allocation from
426/// `next_id` — see [`parse_unit_with_warnings_from`].
427pub fn parse_units_with_drain_check_from(
428    tokens: &[Token],
429    source: &str,
430    next_id: &mut u32,
431) -> Result<(Vec<SourceUnit>, Vec<CompileError>, bool), Vec<CompileError>> {
432    let (filtered, trivia) = split_trivia(tokens, source);
433    let mut warnings = Vec::new();
434    let mut p = Parser::new(&filtered, source, trivia, &mut warnings);
435    p.next_expr_id = *next_id;
436    let mut units = Vec::new();
437    let mut errors: Vec<CompileError> = Vec::new();
438    while p.peek().is_some() {
439        match p.parse_unit() {
440            Ok(u) => units.push(u),
441            Err(e) => {
442                errors.push(e);
443                break;
444            }
445        }
446    }
447    *next_id = p.next_expr_id;
448    let eof = p.eof_span();
449    let fully_drained = p.trivia.is_fully_drained();
450    // `p` (and thus its `&mut warnings` borrow) is no longer used past here, so
451    // the local `warnings` are readable again.
452    if !errors.is_empty() {
453        errors.append(&mut warnings);
454        return Err(errors);
455    }
456    if units.is_empty() {
457        return Err(vec![CompileError::new(
458            "bynk.parse.unexpected_eof",
459            eof,
460            "expected `commons`, `context`, or `suite` to start the file, found end of file",
461        )]);
462    }
463    // A file-trailing comment must have been drained by `take_epilogue` — the
464    // one brace-form declarations forgot to call (a live comment-loss bug,
465    // not the fundamentally-unfixed expression-interior case: expressions
466    // carry no trivia at all, so asserting full drainage here would fire on
467    // any ordinary program with a comment inside a `match`/list/record, which
468    // `bynk-fmt`'s own comment-loss guard already handles gracefully rather
469    // than as a hard failure).
470    debug_assert!(
471        p.trivia.epilogue_is_empty(),
472        "a file-trailing comment was left undrained after a successful parse"
473    );
474    Ok((units, warnings, fully_drained))
475}
476
477/// A signed numeric literal in refinement-bound position (v0.21): `InRange`
478/// bounds are either both `Int` or both `Float`.
479enum SignedNumLit {
480    Int(IntBound),
481    Float(FloatBound),
482}
483
484struct Parser<'a> {
485    tokens: &'a [Token],
486    source: &'a str,
487    pos: usize,
488    /// Accumulated non-fatal diagnostics. v0.3 uses this for orphan-doc
489    /// warnings, which are emitted as errors with a distinguishable category.
490    warnings: &'a mut Vec<CompileError>,
491    /// When true, the item-level loops catch errors from individual item
492    /// parses, push them into `recovered_errors`, and skip forward to the
493    /// next top-level item boundary instead of bailing. Used by the LSP via
494    /// [`parse_unit_with_recovery`]; disabled in the normal `parse` path so
495    /// existing single-error behaviour is preserved.
496    recover_mode: bool,
497    /// Errors collected during recovery-mode parsing. Only populated when
498    /// `recover_mode` is true.
499    recovered_errors: Vec<CompileError>,
500    /// #1663: the token index where the current top-level item starts, set by
501    /// each top-level item loop. Read once by [`Self::handle_item_err`] to name
502    /// the declaration a recovery skips.
503    pub(crate) item_start: Option<usize>,
504    /// #1663: the names of top-level declarations recovery skipped (a `type`,
505    /// `fn`, method, `event`, `capability`, `service`, `agent` or `actor`).
506    /// They are known names whose declaration is broken, so a caller can keep
507    /// references to them from echoing as unknown. Recovery mode only.
508    broken_decl_names: Vec<String>,
509    /// Line-comment trivia separated from the token stream. See
510    /// [`TriviaTable`].
511    trivia: TriviaTable,
512    /// Live recursion depth of the three self-recursive parse entry points
513    /// (`parse_expr`, `parse_type_ref`, `parse_pattern`). Incremented on entry
514    /// and decremented on exit by [`Parser::enter_recursion`] so it tracks the
515    /// current stack depth; when it exceeds [`crate::MAX_NESTING_DEPTH`] the
516    /// parser reports a bounded-depth diagnostic instead of overflowing its
517    /// stack (#713).
518    depth: usize,
519    /// When true, a bare `ident {` on the *spine* of the current expression is
520    /// an identifier followed by an unrelated block, never a record
521    /// construction — so an `if`/`match` condition that ends in a bare
522    /// identifier does not swallow the branch/arm block as `Ident { field }`
523    /// (#636). Set only around the condition parse (see [`parse_cond_expr`]);
524    /// `parse_expr` clears it, so the restriction is lifted inside any
525    /// delimited sub-expression (parentheses, call arguments, list, record
526    /// field). Mirrors Rust's `NO_STRUCT_LITERAL` restriction.
527    no_record_literal: bool,
528    /// Running count of unclosed `{` seen so far — maintained solely by
529    /// [`Self::bump`] (the one primitive that advances `self.pos`), so it
530    /// always reflects the true nesting depth no matter which parse function
531    /// is on the call stack. Finding #27/#30: `recover_to_top_item` reads it
532    /// against [`Self::item_loop_baseline`] to tell "the enclosing item
533    /// loop's own closing brace" apart from a still-unclosed nested
534    /// construct's — without it, a sync scan that started partway through
535    /// such a construct (an error deep inside a function body) stopped at the
536    /// first `}` it saw, however deeply nested, and the enclosing item loop
537    /// mistook that for its own body's end.
538    brace_depth: usize,
539    /// Stack of `brace_depth` snapshots, one per active item-loop body
540    /// (commons/context/adapter/suite) — pushed right after that body's own
541    /// `{` is consumed (or at loop entry, for a brace-free fragment form),
542    /// popped at the loop's normal exit. `recover_to_top_item` treats its top
543    /// entry as the depth an `}` must return to before it counts as the
544    /// enclosing body's own closing brace rather than a nested construct's.
545    item_loop_baseline: Vec<usize>,
546    /// T3.4 (R2.4): next [`ExprId`] to hand out — monotonic, incremented by
547    /// [`Self::alloc_expr_id`], the sole allocation point every `Expr`
548    /// construction site in this parser calls.
549    next_expr_id: u32,
550}
551
552impl<'a> Parser<'a> {
553    fn new(
554        tokens: &'a [Token],
555        source: &'a str,
556        trivia: TriviaTable,
557        warnings: &'a mut Vec<CompileError>,
558    ) -> Self {
559        Self {
560            tokens,
561            source,
562            pos: 0,
563            warnings,
564            recover_mode: false,
565            recovered_errors: Vec::new(),
566            item_start: None,
567            broken_decl_names: Vec::new(),
568            trivia,
569            depth: 0,
570            no_record_literal: false,
571            brace_depth: 0,
572            item_loop_baseline: Vec::new(),
573            next_expr_id: 0,
574        }
575    }
576
577    /// T3.4 (R2.4): allocate the next [`ExprId`]. The sole allocation point —
578    /// every `Expr { id: self.alloc_expr_id(), .. }` construction in this
579    /// parser calls it exactly once, so two nodes never share an id and every
580    /// id a caller holds was actually handed out here.
581    fn alloc_expr_id(&mut self) -> ExprId {
582        let id = ExprId(self.next_expr_id);
583        self.next_expr_id += 1;
584        id
585    }
586
587    /// Enter a self-recursive parse step, bumping the live recursion depth and
588    /// failing with a bounded-depth diagnostic if it would exceed
589    /// [`crate::MAX_NESTING_DEPTH`]. The caller pairs a successful entry with a
590    /// matching `self.depth -= 1` on the way out (see `parse_expr` /
591    /// `parse_type_ref`); on the error path the depth is restored here so a
592    /// recovering caller is not left mis-counted. `what` names the construct
593    /// for the message (e.g. "this expression", "this type"). See #713.
594    fn enter_recursion(&mut self, what: &str) -> Result<(), CompileError> {
595        self.depth += 1;
596        if self.depth > crate::MAX_NESTING_DEPTH {
597            self.depth -= 1;
598            let span = self
599                .peek()
600                .map(|t| t.span)
601                .unwrap_or_else(|| self.eof_span());
602            return Err(self.nesting_too_deep(span, what));
603        }
604        Ok(())
605    }
606
607    /// The bounded-depth diagnostic shared by [`enter_recursion`] and
608    /// [`enter_chain_fold`].
609    fn nesting_too_deep(&self, span: Span, what: &str) -> CompileError {
610        CompileError::new(
611            "bynk.parse.nesting_too_deep",
612            span,
613            format!(
614                "{what} nests more than {} levels deep",
615                crate::MAX_NESTING_DEPTH
616            ),
617        )
618        .with_note(
619            "deeply nested source is rejected to keep the parser from overflowing its \
620             stack and aborting; flatten or split the construct",
621        )
622    }
623
624    /// The bounded-depth diagnostic for the *iteratively*-built spines —
625    /// associative operator chains ([`enter_chain_fold`]) and postfix receiver
626    /// chains ([`deepen_spine`]). Same code as [`nesting_too_deep`] (one budget,
627    /// one diagnostic) but phrased for a flat chain, which is long rather than
628    /// *nested*, and points at the idiomatic fix.
629    fn expression_too_long(&self, span: Span) -> CompileError {
630        CompileError::new(
631            "bynk.parse.nesting_too_deep",
632            span,
633            format!(
634                "this expression is more than {} levels deep",
635                crate::MAX_NESTING_DEPTH
636            ),
637        )
638        .with_note(
639            "a long operator or member chain is rejected to keep the compiler from overflowing \
640             its stack; split it across `let` bindings, or reduce a sequence with \
641             `.sum()`/`.fold(...)`",
642        )
643    }
644
645    /// Count one more operand folded onto an associative operator chain against
646    /// the same recursion budget as [`enter_recursion`] (#714).
647    ///
648    /// Associative chains (`+`, `*`, `&&`, `||`) are built *iteratively* in the
649    /// precedence ladder, so — unlike parentheses, calls, or `implies` — they
650    /// never re-enter `parse_expr` and thus slip past the `enter_recursion`
651    /// guard. Yet each fold deepens the left-nested `Expr` tree by one level,
652    /// and a long flat chain (`1 + 1 + … + 1`) overflows every *recursive*
653    /// consumer of that tree downstream — the checker's `type_of`, the
654    /// formatter, the emitter, and the AST's own recursive `Drop` — exactly as
655    /// deeply nested source overflows the parser. Counting each fold on the
656    /// shared `depth` budget bounds the whole expression's height, and because
657    /// it is the *same* budget it composes with the ambient nesting depth, so a
658    /// chain buried inside deeply nested source cannot exceed the bound either.
659    ///
660    /// The caller accumulates `folds` and subtracts them from `depth` before it
661    /// returns, so the live count unwinds as a recursive descent would; on the
662    /// overflow path the whole chain's contribution is restored here so a
663    /// recovering caller is not left mis-counted.
664    fn enter_chain_fold(&mut self, folds: &mut usize, span: Span) -> Result<(), CompileError> {
665        self.depth += 1;
666        *folds += 1;
667        if self.depth > crate::MAX_NESTING_DEPTH {
668            self.depth -= *folds;
669            *folds = 0;
670            return Err(self.expression_too_long(span));
671        }
672        Ok(())
673    }
674
675    /// Count one more level of an iteratively-built postfix receiver spine
676    /// (`a.b.c…`, `f()?.g()…`) against the shared budget (#714). Like
677    /// [`enter_chain_fold`], postfix loops rather than recurses, so a long spine
678    /// escapes [`enter_recursion`] yet grows an arbitrarily deep receiver tree
679    /// that the downstream walks recurse through. `parse_postfix` restores
680    /// `depth` wholesale on the way out (its many error paths make a
681    /// save/restore wrapper cleaner than per-fold unwinding), so this only bumps
682    /// and checks.
683    fn deepen_spine(&mut self, span: Span) -> Result<(), CompileError> {
684        self.depth += 1;
685        if self.depth > crate::MAX_NESTING_DEPTH {
686            return Err(self.expression_too_long(span));
687        }
688        Ok(())
689    }
690
691    /// Comments immediately preceding the current peek position. Consumed
692    /// (the table entry is cleared) so the same comments are not attached
693    /// to two nodes.
694    fn take_leading_trivia(&mut self) -> Vec<String> {
695        self.trivia.take_leading(self.pos)
696    }
697
698    /// Trailing comment, if any, on the same source line as the most
699    /// recently consumed content token. Call AFTER finishing a declaration
700    /// or statement, while `self.pos` points one past its last token.
701    fn take_trailing_trivia(&mut self) -> Option<String> {
702        if self.pos == 0 {
703            return None;
704        }
705        self.trivia.take_trailing(self.pos - 1)
706    }
707
708    /// Handle a per-item parse error. In recovery mode, record the error and
709    /// advance to the next sync point so the item loop can continue; otherwise
710    /// propagate as a hard failure.
711    fn handle_item_err(&mut self, e: CompileError) -> Result<(), CompileError> {
712        if self.recover_mode {
713            self.recovered_errors.push(e);
714            if let Some(start) = self.item_start.take() {
715                self.broken_decl_names.extend(self.decl_names_at(start));
716            }
717            let before = self.pos;
718            self.recover_to_top_item();
719            // The sync target may be the very token that produced the error —
720            // a context-only keyword (`capability`, `service`, …) at item
721            // position in a commons errors *without consuming it*, and it is
722            // itself a sync point. Recovery must always make progress, or the
723            // item loop re-reports the same error until memory runs out
724            // (found by the `parse` fuzz target on a seed input).
725            // #1663: unless the cursor is on the `}` that closes this item
726            // loop's own body (an error raised at the end of the last item):
727            // consuming it would make the body look unclosed, a follow-on
728            // `unexpected_eof`. The loop ends on that `}` itself. A fragment-
729            // form loop has no closing brace (baseline 0), so a stray `}`
730            // there is still consumed.
731            let baseline = self.item_loop_baseline.last().copied().unwrap_or(0);
732            let closes_body = baseline > 0
733                && self.brace_depth == baseline
734                && self.peek_kind() == Some(TokenKind::RBrace);
735            if self.pos == before && !closes_body {
736                self.bump();
737                // #1663: and skip the rest of the rejected item too (`agent`
738                // in a commons: its name and `{ … }` body). Otherwise the next
739                // loop iteration misreads its name as a malformed item and
740                // reports a second, follow-on syntax error.
741                self.recover_to_top_item();
742            }
743            Ok(())
744        } else {
745            Err(e)
746        }
747    }
748
749    /// #1663: the name(s) a declaration starting at token `start` declares:
750    /// `type T`, `fn f`, `fn T.m` (recorded as `T.m`), `event E`,
751    /// `capability C`, `service S`, `agent A`, `actor A`. Empty when the item
752    /// is not one of those or its name never parsed (`fn ( -> Int`).
753    fn decl_names_at(&self, start: usize) -> Vec<String> {
754        let kind_at = |i: usize| self.tokens.get(i).map(|t| t.kind);
755        let ident_at = |i: usize| {
756            self.tokens
757                .get(i)
758                .filter(|t| t.kind == TokenKind::Ident)
759                .map(|t| self.slice(t.span).to_string())
760        };
761        match kind_at(start) {
762            Some(
763                TokenKind::Type
764                | TokenKind::Event
765                | TokenKind::Capability
766                | TokenKind::Service
767                | TokenKind::Agent
768                | TokenKind::Actor,
769            ) => ident_at(start + 1).into_iter().collect(),
770            Some(TokenKind::Fn) => match (ident_at(start + 1), kind_at(start + 2)) {
771                // A method is recorded qualified, `T.m`, so a broken method
772                // is never mistaken for a free `fn m` of the same name.
773                (Some(owner), Some(TokenKind::Dot)) => match ident_at(start + 3) {
774                    Some(method) => vec![format!("{owner}.{method}")],
775                    None => Vec::new(),
776                },
777                (Some(name), _) => vec![name],
778                (None, _) => Vec::new(),
779            },
780            _ => Vec::new(),
781        }
782    }
783
784    /// Skip forward to the next top-level item boundary: either an
785    /// [`is_item_start`] keyword at the enclosing item loop's own nesting
786    /// depth, a closing brace that returns to that depth, or end-of-input.
787    /// Used only in recovery mode.
788    ///
789    /// Finding #27/#30: brace-depth-gated against
790    /// [`Self::item_loop_baseline`], so a `}` deep inside a still-unclosed
791    /// nested construct (an error partway through a function body, itself
792    /// inside a `match` arm) is skipped over rather than mistaken for the
793    /// enclosing body's own closing brace — the old flat scan stopped at
794    /// literally the first `}` it saw, however deep, handing the item loop a
795    /// brace that did not belong to it and making it return with zero items.
796    fn recover_to_top_item(&mut self) {
797        let baseline = self.item_loop_baseline.last().copied().unwrap_or(0);
798        while let Some(t) = self.peek() {
799            match t.kind {
800                TokenKind::RBrace if self.brace_depth == baseline => return,
801                _ if self.brace_depth == baseline && is_item_start(t.kind) => return,
802                _ => {
803                    self.bump();
804                }
805            }
806        }
807    }
808
809    /// Mark the start of a top-level item loop (`declarations.rs`'s
810    /// `parse_commons_brace`/`_fragment`, `parse_context_brace`/`_fragment`,
811    /// `parse_test_brace`/`_fragment`, `parse_adapter_body`) — called right
812    /// after that body's own `{` is consumed (brace form) or at the loop's
813    /// own entry (fragment form, which has no enclosing brace of its own).
814    /// Paired with [`Self::exit_item_loop`] at the loop's normal exit.
815    fn enter_item_loop(&mut self) {
816        self.item_loop_baseline.push(self.brace_depth);
817    }
818
819    /// Pair of [`Self::enter_item_loop`].
820    fn exit_item_loop(&mut self) {
821        self.item_loop_baseline.pop();
822    }
823
824    fn peek(&self) -> Option<Token> {
825        self.tokens.get(self.pos).copied()
826    }
827
828    fn peek_kind(&self) -> Option<TokenKind> {
829        self.peek().map(|t| t.kind)
830    }
831
832    /// The token `n` positions ahead of the cursor (`nth(0)` == `peek()`).
833    fn nth(&self, n: usize) -> Option<Token> {
834        self.tokens.get(self.pos + n).copied()
835    }
836
837    fn nth_kind(&self, n: usize) -> Option<TokenKind> {
838        self.nth(n).map(|t| t.kind)
839    }
840
841    /// The source text of the token `n` positions ahead, or `""` if none.
842    fn nth_text(&self, n: usize) -> &'a str {
843        self.nth(n).map(|t| self.slice(t.span)).unwrap_or("")
844    }
845
846    /// The span of the most recently consumed token (`self.pos - 1`). Falls back
847    /// to the current token's span when nothing has been consumed yet.
848    fn prev_span(&self) -> Span {
849        self.tokens
850            .get(self.pos.wrapping_sub(1))
851            .or_else(|| self.peek_ref())
852            .map(|t| t.span)
853            .unwrap_or_default()
854    }
855
856    fn peek_ref(&self) -> Option<&Token> {
857        self.tokens.get(self.pos)
858    }
859
860    fn bump(&mut self) -> Option<Token> {
861        let t = self.peek();
862        if let Some(t) = t {
863            match t.kind {
864                TokenKind::LBrace => self.brace_depth += 1,
865                TokenKind::RBrace => self.brace_depth = self.brace_depth.saturating_sub(1),
866                _ => {}
867            }
868            self.pos += 1;
869        }
870        t
871    }
872
873    fn eat(&mut self, kind: TokenKind) -> Option<Token> {
874        if self.peek_kind() == Some(kind) {
875            self.bump()
876        } else {
877            None
878        }
879    }
880
881    fn slice(&self, span: Span) -> &'a str {
882        &self.source[span.range()]
883    }
884
885    /// True when the next token sits on a later line than `prev`. Used to
886    /// keep a `[` that opens a new line out of the postfix type-application
887    /// form: `f` followed by `[1, 2]` on the next line is an identifier and
888    /// a list literal, not `f[…]` (v0.20b).
889    fn next_token_on_new_line(&self, prev: Span) -> bool {
890        match self.peek() {
891            Some(t) if prev.end <= t.span.start => {
892                self.source[prev.end..t.span.start].contains('\n')
893            }
894            _ => false,
895        }
896    }
897
898    /// Span pointing at the end of input — used for "unexpected EOF" reports.
899    /// The start backs up to the **start of the final char**, not `len - 1`, so
900    /// the span never splits a multibyte codepoint (an unterminated construct
901    /// whose last line ends in non-ASCII — e.g. a `--` comment ending in `→`).
902    fn eof_span(&self) -> Span {
903        let end = self.source.len();
904        let start = (0..end)
905            .rev()
906            .find(|&i| self.source.is_char_boundary(i))
907            .unwrap_or(0);
908        Span::new(start, end)
909    }
910
911    fn expect(&mut self, kind: TokenKind, ctx: &str) -> Result<Token, CompileError> {
912        match self.peek() {
913            Some(t) if t.kind == kind => {
914                self.bump();
915                Ok(t)
916            }
917            Some(t) => Err(CompileError::new(
918                "bynk.parse.expected_token",
919                t.span,
920                format!(
921                    "expected {} {ctx}, found {}",
922                    kind.describe(),
923                    t.kind.describe()
924                ),
925            )),
926            None => Err(CompileError::new(
927                "bynk.parse.unexpected_eof",
928                self.eof_span(),
929                format!("expected {} {ctx}, found end of file", kind.describe()),
930            )),
931        }
932    }
933
934    fn expect_ident(&mut self, ctx: &str) -> Result<Ident, CompileError> {
935        match self.peek() {
936            Some(t) if t.kind == TokenKind::Ident => {
937                self.bump();
938                Ok(Ident {
939                    name: self.slice(t.span).to_string(),
940                    span: t.span,
941                })
942            }
943            // v0.5 contextual keyword `on` doubles as an identifier in
944            // expression / field-access positions so users can name fields and
945            // parameters using it. It retains its keyword meaning only at
946            // handler-decl-level (`on call(...)`).
947            //
948            // v0.7 / v0.112: `suite` and `case` are contextual too — they
949            // introduce the suite declaration and its cases, but are perfectly
950            // valid commons/context/field names otherwise.
951            //
952            // The tier is single-sourced in `keywords::RESERVED_CONTEXTUAL`:
953            // this arm defers to it rather than hardcoding the token kinds, so
954            // extending that list is enough to admit a new contextual keyword
955            // here. Each of these words lexes only to its own token, so matching
956            // the source text is equivalent to matching the kind.
957            Some(t) if crate::keywords::is_reserved_contextual(self.slice(t.span)) => {
958                self.bump();
959                Ok(Ident {
960                    name: self.slice(t.span).to_string(),
961                    span: t.span,
962                })
963            }
964            Some(t) if is_reserved_keyword(t.kind) => Err(CompileError::new(
965                "bynk.parse.reserved_keyword",
966                t.span,
967                format!(
968                    "expected identifier {ctx}, but `{}` is a reserved keyword",
969                    self.slice(t.span)
970                ),
971            )
972            .with_note("rename the identifier to something that is not a keyword")),
973            Some(t) => Err(CompileError::new(
974                "bynk.parse.expected_token",
975                t.span,
976                format!("expected identifier {ctx}, found {}", t.kind.describe()),
977            )),
978            None => Err(CompileError::new(
979                "bynk.parse.unexpected_eof",
980                self.eof_span(),
981                format!("expected identifier {ctx}, found end of file"),
982            )),
983        }
984    }
985
986    // -- top level --
987
988    /// Consume an optional doc block at the current position, returning the
989    /// (content, end-of-doc span) pair. Returns None if the next token is not
990    /// a doc block.
991    fn take_doc_block(&mut self) -> Option<(String, Span)> {
992        if self.peek_kind() == Some(TokenKind::DocBlock) {
993            let t = self.bump().unwrap();
994            let body = doc_block_content(self.source, t.span);
995            return Some((body, t.span));
996        }
997        None
998    }
999
1000    /// Collect all line-comment trivia leading the next declaration plus
1001    /// the optional doc block. Comments may appear both *before* and
1002    /// *between* the doc and the declaration; the spec canonicalises both
1003    /// groups above the doc, so we concatenate them.
1004    fn collect_item_lead(&mut self) -> (Vec<String>, Option<(String, Span)>) {
1005        let mut leading = self.take_leading_trivia();
1006        let doc = self.take_doc_block();
1007        if doc.is_some() {
1008            leading.extend(self.take_leading_trivia());
1009        }
1010        (leading, doc)
1011    }
1012
1013    /// Attach a parsed doc block to a following declaration unless a blank
1014    /// line separates them, in which case the doc is orphaned (warning).
1015    fn finalize_doc(&mut self, doc: Option<(String, Span)>, next_span: Span) -> Option<String> {
1016        let (content, doc_span) = doc?;
1017        // A blank line between the doc and the next decl orphans the doc.
1018        if has_blank_line_between(self.source, doc_span.end, next_span.start) {
1019            self.warnings.push(
1020                CompileError::new(
1021                    "bynk.parse.orphan_doc_block",
1022                    doc_span,
1023                    "documentation block is separated from the following declaration by a blank line; it will not be attached",
1024                )
1025                .with_note(
1026                    "remove the blank line to attach the doc to the next declaration, \
1027                     or remove the doc block if it is not meant to document anything",
1028                ),
1029            );
1030            return None;
1031        }
1032        Some(content)
1033    }
1034}
1035
1036/// Parse the body of a lexed double-quoted string literal (the lexeme,
1037/// including surrounding quotes), applying the v0 escape rules.
1038fn parse_string_literal(lexeme: &str, span: Span) -> Result<String, CompileError> {
1039    let bytes = lexeme.as_bytes();
1040    debug_assert!(bytes.first() == Some(&b'"') && bytes.last() == Some(&b'"'));
1041    let inner = &lexeme[1..lexeme.len() - 1];
1042    let mut out = String::with_capacity(inner.len());
1043    let mut chars = inner.chars();
1044    while let Some(c) = chars.next() {
1045        if c == '\\' {
1046            match chars.next() {
1047                Some('n') => out.push('\n'),
1048                Some('t') => out.push('\t'),
1049                Some('"') => out.push('"'),
1050                Some('\\') => out.push('\\'),
1051                other => {
1052                    return Err(CompileError::new(
1053                        "bynk.lex.bad_escape",
1054                        span,
1055                        format!(
1056                            "invalid escape sequence `\\{}` in string literal",
1057                            other.map(|c| c.to_string()).unwrap_or_default()
1058                        ),
1059                    )
1060                    .with_note("supported escapes: \\n \\t \\\" \\\\"));
1061                }
1062            }
1063        } else {
1064            out.push(c);
1065        }
1066    }
1067    Ok(out)
1068}
1069
1070fn is_reserved_keyword(kind: TokenKind) -> bool {
1071    use TokenKind::*;
1072    matches!(
1073        kind,
1074        Commons
1075            | Type
1076            | Fn
1077            | Where
1078            | True
1079            | False
1080            | Int
1081            | String
1082            | Bool
1083            | Let
1084            | If
1085            | Else
1086            | Ok
1087            | Err
1088            | Result
1089            | ValidationError
1090            | Enum
1091            | Match
1092            | Option
1093            | Record
1094            | Self_
1095            | Some
1096            | None
1097            | Is
1098            | Opaque
1099            | Uses
1100            | Context
1101            | Consumes
1102            | Exports
1103            | Transparent
1104            | Agent
1105            | As
1106            | Capability
1107            | Effect
1108            | Do
1109            | Given
1110            | On
1111            | Http
1112            | Provides
1113            | Stub
1114            | Service
1115            | Actor
1116            | By
1117            | Expect
1118            | Suite
1119            | Case
1120            | Float
1121            | Duration
1122            | Instant
1123            | Bytes
1124            | JsonError
1125            | Property
1126            | Adapter
1127            | Binding
1128            | Cron
1129            | Queue
1130            | From
1131            | Protocol
1132            | Invariant
1133            | Implies
1134            | Requires
1135            | Ensures
1136            | Transition
1137    )
1138}
1139
1140/// True when `kind` starts a top-level unit (`commons`/`context`/`adapter`/
1141/// `suite`) or an item within one of their bodies — every keyword any of
1142/// `parse_commons_brace`/`_fragment`, `parse_context_brace`/`_fragment`,
1143/// `parse_test_brace`/`_fragment`, or `parse_adapter_body` dispatches on
1144/// (`declarations.rs`). The single set [`Parser::recover_to_top_item`]'s sync
1145/// scan checks against — finding #27/#30: that scan had drifted from what the
1146/// item loops actually recognise (`Property`, `Actor`, `Event`, `Binding`,
1147/// and the `adapter` unit keyword itself were all missing), so an error
1148/// recovery sync could walk past a real item/unit boundary instead of
1149/// stopping there.
1150fn is_item_start(kind: TokenKind) -> bool {
1151    use TokenKind::*;
1152    matches!(
1153        kind,
1154        // Top-level unit keywords.
1155        Commons | Context | Adapter | Suite
1156        // Body items shared across commons/context/adapter.
1157        | Type | Fn | Messages | Event | Uses
1158        // Context/adapter-only body items.
1159        | Consumes | Exports | Capability | Provides | Service | Agent | Actor
1160        // Adapter-only.
1161        | Binding
1162        // Suite/test-only body items.
1163        | Stub | Case | Property
1164    )
1165}
1166
1167#[cfg(test)]
1168mod tests {
1169    use super::*;
1170    use crate::lexer::tokenize;
1171
1172    fn parse_str(src: &str) -> Result<Commons, Vec<CompileError>> {
1173        let toks = tokenize(src).map_err(|e| vec![e])?;
1174        parse(&toks, src)
1175    }
1176
1177    fn parse_recover_str(src: &str) -> (Option<SourceUnit>, Vec<CompileError>) {
1178        let toks = match tokenize(src) {
1179            Ok(t) => t,
1180            Err(e) => return (None, vec![e]),
1181        };
1182        parse_unit_with_recovery(&toks, src)
1183    }
1184
1185    /// Finding #29/#30: `parse_units_with_recovery` keeps every top-level unit
1186    /// an atomic `commons`+`suite` file declares (v0.113, DECISION S), where
1187    /// `parse_unit_with_recovery` keeps only the first — the defect the IDE's
1188    /// old single-unit parse entry point had (it silently discarded the
1189    /// trailing `suite`).
1190    #[test]
1191    fn parse_units_with_recovery_keeps_every_top_level_unit() {
1192        let src =
1193            "commons m {\n  fn f() -> Int { 1 }\n}\n\nsuite m\n\ncase \"c\" {\n  expect true\n}\n";
1194        let toks = tokenize(src).unwrap();
1195        let (units, errors) = parse_units_with_recovery(&toks, src);
1196        assert!(errors.is_empty(), "{errors:?}");
1197        assert_eq!(units.len(), 2, "expected both units, got {units:?}");
1198        assert!(matches!(units[0], SourceUnit::Commons(_)));
1199        assert!(matches!(units[1], SourceUnit::Suite(_)));
1200
1201        // The singular wrapper still narrows to just the first, unchanged.
1202        let (unit, errors) = parse_unit_with_recovery(&toks, src);
1203        assert!(errors.is_empty(), "{errors:?}");
1204        assert!(matches!(unit, Some(SourceUnit::Commons(_))));
1205    }
1206
1207    /// `parse_unit_with_recovery` always attempts at least one parse, even on
1208    /// empty input — `parse_units_with_recovery`'s loop must preserve that
1209    /// (its `while`-style peek check alone would skip the body entirely and
1210    /// silently return no error), since 16+ existing callers rely on an empty
1211    /// file still producing its usual diagnostic rather than a silent `None`
1212    /// with no error at all.
1213    #[test]
1214    fn empty_input_still_reports_an_error_through_the_plural_entry_point() {
1215        let toks = tokenize("").unwrap();
1216        let (units, errors) = parse_units_with_recovery(&toks, "");
1217        assert!(units.is_empty());
1218        assert!(
1219            !errors.is_empty(),
1220            "an empty file must still produce a diagnostic, not silently no units and no error"
1221        );
1222
1223        let (unit, unit_errors) = parse_unit_with_recovery(&toks, "");
1224        assert!(unit.is_none());
1225        assert_eq!(
1226            errors.len(),
1227            unit_errors.len(),
1228            "the singular wrapper must see the same error(s) as the plural entry point"
1229        );
1230    }
1231
1232    /// Finding #66: `parse_units_with_drain_check`'s `fully_drained` flag is
1233    /// `bynk-fmt`'s signal for whether its comment-loss guard can skip a
1234    /// re-tokenize-and-diff of its own output. It must be `true` for an
1235    /// ordinary file (every comment sits before a declaration/statement or
1236    /// trails one) and `false` the moment a comment sits inside an expression
1237    /// subtree, where `TriviaTable` has no field to attach it to.
1238    #[test]
1239    fn drain_check_reports_expression_interior_comments_as_undrained() {
1240        let ordinary = "commons x {\n-- note\ntype T = Int where Positive\n}\n";
1241        let toks = tokenize(ordinary).unwrap();
1242        let (_, _, drained) = parse_units_with_drain_check(&toks, ordinary).unwrap();
1243        assert!(
1244            drained,
1245            "a declaration-leading comment must be fully drained"
1246        );
1247
1248        let lossy = "commons x {\n  fn f() -> Int {\n    1 + -- note\n    2\n  }\n}\n";
1249        let toks = tokenize(lossy).unwrap();
1250        let (_, _, drained) = parse_units_with_drain_check(&toks, lossy).unwrap();
1251        assert!(
1252            !drained,
1253            "a comment inside a binop expression must be reported as undrained"
1254        );
1255    }
1256
1257    #[test]
1258    fn eof_span_never_splits_a_multibyte_codepoint() {
1259        // An unterminated construct whose final line ends in a non-ASCII char
1260        // (here a `--` comment ending in `→`) once produced an `unexpected_eof`
1261        // span of `len - 1 .. len`, landing on the arrow's last continuation
1262        // byte. Every reported span must sit on char boundaries.
1263        for src in [
1264            "commons x {\n  -- ends with an arrow →",
1265            "agent A {\n  key k: String\n  -- note 🦀",
1266            "commons y {\n  type T = é",
1267        ] {
1268            let (_unit, errors) = parse_recover_str(src);
1269            for e in &errors {
1270                assert!(
1271                    src.is_char_boundary(e.span.start) && src.is_char_boundary(e.span.end),
1272                    "span {:?} splits a codepoint in {src:?}",
1273                    e.span,
1274                );
1275            }
1276        }
1277    }
1278
1279    #[test]
1280    fn reserved_contextual_keywords_readable_in_expression_position() {
1281        // Events track, slice 0 (#939): `expect_ident`'s `RESERVED_CONTEXTUAL`
1282        // exemption (`keywords::RESERVED_CONTEXTUAL`: case/event/messages/on/
1283        // suite) covers *declaring* a binding with one of these names — a
1284        // parameter, a `let` — but the primary-expression parser previously
1285        // only admitted plain `TokenKind::Ident` when *reading one back*.
1286        // Latent since messages/on/case/suite shipped (no fixture happened to
1287        // name a binding after one of them and read it back inside the body);
1288        // surfaced concretely when `event` joined the tier and collided with
1289        // `examples/event-log`'s pre-existing `add(event: Event)` handler.
1290        for kw in ["case", "event", "messages", "on", "suite"] {
1291            let src = format!("commons x\n\nfn f({kw}: Int) -> Int {{\n  {kw}\n}}\n");
1292            let result = parse_str(&src);
1293            assert!(
1294                result.is_ok(),
1295                "a parameter named `{kw}` must be readable in expression position: {:?}",
1296                result.err()
1297            );
1298        }
1299    }
1300
1301    #[test]
1302    fn recovery_skips_garbage_between_decls() {
1303        // Two `type` declarations separated by garbage. Recovery should
1304        // accept both and report one error for the garbage between them.
1305        let src = "commons x {\n\
1306                   type A = Int where NonNegative\n\
1307                   ??? !!!\n\
1308                   type B = String where NonEmpty\n\
1309                   }";
1310        let (unit, errors) = parse_recover_str(src);
1311        let unit = unit.expect("recovery should produce a partial AST");
1312        let SourceUnit::Commons(c) = unit else {
1313            panic!("expected commons")
1314        };
1315        // Both type decls should have been collected despite the garbage.
1316        let names: Vec<_> = c
1317            .items
1318            .iter()
1319            .map(|i| match i {
1320                CommonsItem::Type(t) => t.name.name.clone(),
1321                _ => panic!("expected only types"),
1322            })
1323            .collect();
1324        assert!(
1325            names.contains(&"A".to_string()) && names.contains(&"B".to_string()),
1326            "expected both A and B; got {names:?}",
1327        );
1328        assert!(!errors.is_empty(), "expected at least one parse error");
1329    }
1330
1331    #[test]
1332    fn recovery_handles_bad_first_decl_then_good_second() {
1333        // First decl is malformed (missing `=`); second is well-formed.
1334        let src = "commons x {\n\
1335                   type A Int where NonNegative\n\
1336                   type B = String where NonEmpty\n\
1337                   }";
1338        let (unit, errors) = parse_recover_str(src);
1339        let unit = unit.expect("recovery should produce a partial AST");
1340        let SourceUnit::Commons(c) = unit else {
1341            panic!("expected commons")
1342        };
1343        let names: Vec<_> = c
1344            .items
1345            .iter()
1346            .filter_map(|i| match i {
1347                CommonsItem::Type(t) => Some(t.name.name.clone()),
1348                _ => None,
1349            })
1350            .collect();
1351        assert!(
1352            names.contains(&"B".to_string()),
1353            "B should be parsed after A's failure; got {names:?}"
1354        );
1355        assert!(!errors.is_empty(), "expected at least one parse error");
1356    }
1357
1358    /// Finding #27/#30: an error two levels deep inside `f`'s body (a
1359    /// `match` arm's own block) used to make `recover_to_top_item`'s flat,
1360    /// depth-blind scan stop at the *first* `}` it saw — the arm block's own,
1361    /// not `f`'s. Two more `}` (the match's, then `f`'s) then got consumed one
1362    /// at a time across repeated recovery re-entries, and the outer item loop
1363    /// eventually mistook the commons's *own* closing `}` for having arrived
1364    /// early, returning zero items and a spurious second
1365    /// `bynk.parse.expected_unit_header` error. With brace-depth tracking, `g`
1366    /// is recovered as the sole item and only `f`'s own error is reported.
1367    /// #1663: an item that ends without its body (`fn f() -> Int` then the
1368    /// commons' own `}`) raises its error *on* that `}`, so recovery makes no
1369    /// progress. The no-progress step must not consume the brace that closes
1370    /// the body (brace form): doing so made the body look unclosed, a
1371    /// follow-on `unexpected_eof`.
1372    #[test]
1373    fn recovery_keeps_the_bodys_closing_brace_after_a_bodiless_item() {
1374        let src = "commons m {\n  fn f() -> Int\n}\n";
1375        let (unit, errors) = parse_recover_str(src);
1376        assert!(unit.is_some(), "recovery should produce a partial AST");
1377        let categories: Vec<_> = errors.iter().map(|e| e.category).collect();
1378        assert_eq!(
1379            categories.len(),
1380            1,
1381            "one syntax error, no follow-on: {categories:?}"
1382        );
1383        assert!(
1384            !categories.contains(&"bynk.parse.unexpected_eof"),
1385            "the commons' closing brace must survive: {categories:?}"
1386        );
1387    }
1388
1389    /// #1663: an item keyword illegal at this position (`agent` in a commons)
1390    /// is skipped *with* its name and `{ … }` body, not one token at a time —
1391    /// otherwise the item loop misreads the agent's name as a malformed item,
1392    /// a second, follow-on `expected_item`. The next item still parses.
1393    #[test]
1394    fn recovery_skips_a_rejected_item_whole() {
1395        let src = "commons m\n\nagent Counter {\n  key id: String\n}\n\nfn g() -> Int { 2 }\n";
1396        let recovered = parse_units_recovering(&crate::lexer::tokenize(src).unwrap(), src);
1397        assert_eq!(
1398            recovered.errors.len(),
1399            1,
1400            "one syntax error, no follow-on: {:?}",
1401            recovered
1402                .errors
1403                .iter()
1404                .map(|e| e.category)
1405                .collect::<Vec<_>>()
1406        );
1407        let Some(SourceUnit::Commons(c)) = recovered.units.first() else {
1408            panic!("expected commons")
1409        };
1410        assert!(
1411            c.items.iter().any(|i| matches!(i, CommonsItem::Fn(_))),
1412            "the `fn g` after the skipped agent must still parse"
1413        );
1414        assert_eq!(recovered.broken_decl_names, ["Counter"]);
1415    }
1416
1417    #[test]
1418    fn recovery_skips_a_nested_blocks_own_closing_brace() {
1419        let src = "commons m {\n  \
1420                   fn f() -> Int {\n    \
1421                   match 1 {\n      \
1422                   is 1 -> { let z = }\n      \
1423                   is _ -> 2\n    \
1424                   }\n  \
1425                   }\n  \
1426                   fn g() -> Int { 2 }\n\
1427                   }\n";
1428        let (unit, errors) = parse_recover_str(src);
1429        let unit = unit.expect("recovery should produce a partial AST");
1430        let SourceUnit::Commons(c) = unit else {
1431            panic!("expected commons")
1432        };
1433        let names: Vec<_> = c
1434            .items
1435            .iter()
1436            .filter_map(|i| match i {
1437                CommonsItem::Fn(f) => match &f.name {
1438                    FnName::Free(id) => Some(id.name.clone()),
1439                    _ => None,
1440                },
1441                _ => None,
1442            })
1443            .collect();
1444        assert_eq!(
1445            names,
1446            vec!["g".to_string()],
1447            "g must still be recovered as an item; got {names:?}"
1448        );
1449        assert!(
1450            !errors
1451                .iter()
1452                .any(|e| e.category == "bynk.parse.expected_unit_header"),
1453            "the outer body's own closing brace must not be mistaken for \
1454             end-of-file: {errors:?}"
1455        );
1456    }
1457
1458    #[test]
1459    fn doc_block_attaches_to_type() {
1460        let c =
1461            parse_str("commons x {\n---\nA descriptive doc.\n---\ntype T = Int where Positive\n}")
1462                .unwrap();
1463        let CommonsItem::Type(t) = &c.items[0] else {
1464            panic!()
1465        };
1466        assert!(t.documentation.is_some());
1467        assert!(
1468            t.documentation
1469                .as_ref()
1470                .unwrap()
1471                .contains("A descriptive doc.")
1472        );
1473    }
1474
1475    #[test]
1476    fn interpolated_string_parses_into_parts() {
1477        // v0.43: `"Hi, \(name)!"` splits into chunk / hole / chunk.
1478        let c = parse_str("commons x\n\nfn f(name: String) -> String {\n  \"Hi, \\(name)!\"\n}\n")
1479            .unwrap();
1480        let CommonsItem::Fn(f) = &c.items[0] else {
1481            panic!("expected fn")
1482        };
1483        let ExprKind::InterpStr(parts) = &f.body.tail.kind else {
1484            panic!("expected InterpStr, got {:?}", f.body.tail.kind)
1485        };
1486        assert_eq!(parts.len(), 3);
1487        assert!(matches!(&parts[0], InterpPart::Chunk(s) if s == "Hi, "));
1488        assert!(
1489            matches!(&parts[1], InterpPart::Hole(h) if matches!(&h.kind, ExprKind::Ident(id) if id.name == "name"))
1490        );
1491        assert!(matches!(&parts[2], InterpPart::Chunk(s) if s == "!"));
1492    }
1493
1494    #[test]
1495    fn interpolated_hole_parses_a_full_expression() {
1496        // A hole holds an arbitrary expression, not just an identifier.
1497        let c =
1498            parse_str("commons x\n\nfn f(a: Int, b: Int) -> String {\n  \"sum = \\(a + b)\"\n}\n")
1499                .unwrap();
1500        let CommonsItem::Fn(f) = &c.items[0] else {
1501            panic!("expected fn")
1502        };
1503        let ExprKind::InterpStr(parts) = &f.body.tail.kind else {
1504            panic!("expected InterpStr")
1505        };
1506        assert!(matches!(&parts[1], InterpPart::Hole(h) if matches!(&h.kind, ExprKind::BinOp(..))));
1507    }
1508
1509    #[test]
1510    fn empty_interpolation_hole_is_rejected() {
1511        let errs = parse_str("commons x\n\nfn f() -> String {\n  \"\\()\"\n}\n").unwrap_err();
1512        assert!(
1513            errs.iter()
1514                .any(|e| e.category == "bynk.parse.empty_interpolation"),
1515            "expected empty_interpolation; got {errs:?}"
1516        );
1517    }
1518
1519    #[test]
1520    fn interpolation_hole_lex_error_span_is_rebased() {
1521        // #716: a lex error inside a `\(…)` hole once carried a span relative to
1522        // the hole substring — never rebased by `hole.start` — so it pointed at
1523        // the file's opening bytes and could split a multibyte char, tripping
1524        // the char-boundary invariant. The error must land on the offending
1525        // bytes within the hole and stay on char boundaries.
1526        let cases = [
1527            // `$` is not a valid token; the error should point at it, not byte 0.
1528            "commons x\n\nfn f() -> String {\n  \"a \\($)\"\n}\n",
1529            // Integer overflow — the reported span must cover the literal itself.
1530            "commons x\n\nfn f() -> String {\n  \"n = \\(99999999999999999999)\"\n}\n",
1531            // A multibyte char before the hole means an un-rebased span could
1532            // land inside the `é`; the rebased span must not.
1533            "commons x\n\nfn f() -> String {\n  \"é \\($)\"\n}\n",
1534        ];
1535        for src in cases {
1536            let errs = parse_str(src).unwrap_err();
1537            assert!(!errs.is_empty(), "expected a lex error for {src:?}");
1538            for e in &errs {
1539                assert!(
1540                    src.is_char_boundary(e.span.start) && src.is_char_boundary(e.span.end),
1541                    "span {:?} splits a codepoint in {src:?}",
1542                    e.span,
1543                );
1544                // The error must point inside the interpolation hole, not at the
1545                // header text that precedes it.
1546                let hole_start = src.find("\\(").expect("case has a hole") + 2;
1547                assert!(
1548                    e.span.start >= hole_start,
1549                    "span {:?} precedes the hole (starts at {hole_start}) in {src:?}",
1550                    e.span,
1551                );
1552            }
1553        }
1554    }
1555
1556    #[test]
1557    fn fragment_form_parses() {
1558        let c = parse_str("commons x.y\n\ntype T = Int where NonNegative\n").unwrap();
1559        assert_eq!(c.form, CommonsForm::Fragment);
1560        assert_eq!(c.items.len(), 1);
1561    }
1562
1563    #[test]
1564    fn uses_parses() {
1565        let c = parse_str("commons x\n\nuses other.lib\n").unwrap();
1566        assert_eq!(c.uses.len(), 1);
1567        assert_eq!(c.uses[0].target.joined(), "other.lib");
1568    }
1569
1570    fn parse_unit_str(src: &str) -> Result<SourceUnit, Vec<CompileError>> {
1571        let toks = tokenize(src).map_err(|e| vec![e])?;
1572        parse_unit(&toks, src)
1573    }
1574
1575    #[test]
1576    fn minimal_context_parses() {
1577        let u = parse_unit_str("context commerce.orders {}").unwrap();
1578        let SourceUnit::Context(c) = u else {
1579            panic!("expected context");
1580        };
1581        assert_eq!(c.name.joined(), "commerce.orders");
1582        assert!(c.items.is_empty());
1583    }
1584
1585    #[test]
1586    fn context_consumes_and_exports_parse() {
1587        let src = "context commerce.orders {\n  uses commerce.money\n  consumes commerce.payment\n  exports opaque { OrderId }\n  exports transparent { OrderError }\n  type OrderId = String where Matches(\"ORD-[0-9]+\")\n  type OrderError = enum { CartEmpty, BadInput }\n}";
1588        let u = parse_unit_str(src).unwrap();
1589        let SourceUnit::Context(c) = u else { panic!() };
1590        assert_eq!(c.uses.len(), 1);
1591        assert_eq!(c.consumes.len(), 1);
1592        assert_eq!(c.exports.len(), 2);
1593        assert_eq!(c.exports[0].kind, ExportKind::Type(Visibility::Opaque));
1594        assert_eq!(c.exports[1].kind, ExportKind::Type(Visibility::Transparent));
1595    }
1596
1597    #[test]
1598    fn context_fragment_form_parses() {
1599        let src = "context x.y\n\nuses other.lib\nconsumes other.ctx\nexports opaque { T }\n\ntype T = Int where NonNegative\n";
1600        let u = parse_unit_str(src).unwrap();
1601        let SourceUnit::Context(c) = u else { panic!() };
1602        assert_eq!(c.form, CommonsForm::Fragment);
1603        assert_eq!(c.uses.len(), 1);
1604        assert_eq!(c.consumes.len(), 1);
1605        assert_eq!(c.exports.len(), 1);
1606    }
1607
1608    #[test]
1609    fn opaque_type_parses() {
1610        let c = parse_str("commons x { type T = opaque Int where NonNegative }").unwrap();
1611        let CommonsItem::Type(t) = &c.items[0] else {
1612            panic!()
1613        };
1614        assert!(matches!(t.body, TypeBody::Opaque { .. }));
1615    }
1616
1617    #[test]
1618    fn empty_commons() {
1619        let c = parse_str("commons fitness.units {}").unwrap();
1620        assert_eq!(c.name.joined(), "fitness.units");
1621        assert!(c.items.is_empty());
1622    }
1623
1624    #[test]
1625    fn one_type_decl() {
1626        let c = parse_str("commons x { type Metres = Int where NonNegative }").unwrap();
1627        assert_eq!(c.items.len(), 1);
1628        let CommonsItem::Type(t) = &c.items[0] else {
1629            panic!()
1630        };
1631        assert_eq!(t.name.name, "Metres");
1632        match &t.body {
1633            TypeBody::Refined {
1634                base, refinement, ..
1635            } => {
1636                assert_eq!(*base, BaseType::Int);
1637                assert!(refinement.is_some());
1638            }
1639            _ => panic!("expected refined body"),
1640        }
1641    }
1642
1643    #[test]
1644    fn function_decl() {
1645        let c = parse_str("commons x { fn add(a: Int, b: Int) -> Int { a + b } }").unwrap();
1646        let CommonsItem::Fn(f) = &c.items[0] else {
1647            panic!()
1648        };
1649        assert_eq!(f.name.ident().name, "add");
1650        assert_eq!(f.params.len(), 2);
1651    }
1652
1653    #[test]
1654    fn chained_comparison_is_error() {
1655        let errs = parse_str("commons x { fn f(a: Int, b: Int, c: Int) -> Bool { a < b < c } }")
1656            .unwrap_err();
1657        assert_eq!(errs[0].category, "bynk.parse.non_associative");
1658    }
1659
1660    #[test]
1661    fn chained_equality_is_error() {
1662        let errs = parse_str("commons x { fn f(a: Int, b: Int, c: Int) -> Bool { a == b == c } }")
1663            .unwrap_err();
1664        assert_eq!(errs[0].category, "bynk.parse.non_associative");
1665    }
1666
1667    /// Run `f` on a thread with a generous stack. The depth-guard tests build
1668    /// source that, *without* the guard, overflows — so if the guard ever
1669    /// regressed we want a clean assertion failure, not a `SIGABRT` that takes
1670    /// the whole test binary down. A large stack also absorbs the fat frames a
1671    /// debug build spends per recursion level (production release frames are
1672    /// ~9 KB/level, so `MAX_NESTING_DEPTH = 64` sits well inside a 1 MB stack;
1673    /// a debug frame is several times larger and would overflow libtest's
1674    /// default 2 MB test thread near the limit even though the guard fires).
1675    fn on_big_stack<T: Send + 'static>(f: impl FnOnce() -> T + Send + 'static) -> T {
1676        std::thread::Builder::new()
1677            .stack_size(64 * 1024 * 1024)
1678            .spawn(f)
1679            .unwrap()
1680            .join()
1681            .unwrap()
1682    }
1683
1684    #[test]
1685    fn deeply_nested_parens_are_bounded_not_overflowed() {
1686        // Without a depth guard the parenthesised-expression recursion
1687        // (`parse_primary` -> `parse_expr` -> …) overflows the stack and aborts
1688        // the process (#713). Well past the limit it must instead report a
1689        // bounded-depth diagnostic. The nesting is left open so the guard, not
1690        // a later `)`, is what stops the descent.
1691        let errs = on_big_stack(|| {
1692            let depth = crate::MAX_NESTING_DEPTH + 8;
1693            let src = format!(
1694                "commons x {{ fn f() -> Int {{ {}0{} }} }}",
1695                "(".repeat(depth),
1696                ")".repeat(depth),
1697            );
1698            parse_str(&src).unwrap_err()
1699        });
1700        assert_eq!(errs[0].category, "bynk.parse.nesting_too_deep");
1701    }
1702
1703    #[test]
1704    fn deeply_nested_types_are_bounded_not_overflowed() {
1705        // The type parser self-recurses through generic type arguments
1706        // (`parse_type_ref` -> `parse_type_atom` -> `parse_type_ref`); the same
1707        // guard bounds it (#713). A right-nested `Result[Int, …]` in parameter
1708        // position drives that recursion.
1709        let errs = on_big_stack(|| {
1710            let depth = crate::MAX_NESTING_DEPTH + 8;
1711            let src = format!(
1712                "commons x {{ fn f(x: {}Int{}) -> Int {{ 0 }} }}",
1713                "Result[Int, ".repeat(depth),
1714                "]".repeat(depth),
1715            );
1716            parse_str(&src).unwrap_err()
1717        });
1718        assert_eq!(errs[0].category, "bynk.parse.nesting_too_deep");
1719    }
1720
1721    #[test]
1722    fn deeply_nested_patterns_are_bounded_not_overflowed() {
1723        // Variant patterns are a third self-recursive descent (`parse_pattern`
1724        // -> `parse_pattern_binding` -> `parse_pattern`) that routes through
1725        // neither `parse_expr` nor `parse_type_ref`; without its own guard a
1726        // nested `Ok(Ok(…))` match arm reproduces the #713 crash.
1727        let errs = on_big_stack(|| {
1728            let depth = crate::MAX_NESTING_DEPTH + 8;
1729            let src = format!(
1730                "commons x {{ fn f(n: Int) -> Int {{ match n {{ {}n{} => 0 }} }} }}",
1731                "Ok(".repeat(depth),
1732                ")".repeat(depth),
1733            );
1734            parse_str(&src).unwrap_err()
1735        });
1736        assert_eq!(errs[0].category, "bynk.parse.nesting_too_deep");
1737    }
1738
1739    #[test]
1740    fn nesting_below_the_limit_still_parses() {
1741        // The guard must not reject ordinary well-nested source: a paren-nested
1742        // expression comfortably under the limit still parses cleanly.
1743        let ok = on_big_stack(|| {
1744            let depth = crate::MAX_NESTING_DEPTH - 8;
1745            let src = format!(
1746                "commons x {{ fn f() -> Int {{ {}0{} }} }}",
1747                "(".repeat(depth),
1748                ")".repeat(depth),
1749            );
1750            parse_str(&src).is_ok()
1751        });
1752        assert!(ok, "well-nested source under the limit should parse");
1753    }
1754
1755    #[test]
1756    fn let_statement_parses() {
1757        let c = parse_str("commons x { fn f(n: Int) -> Int { let y = n + 1\n y } }").unwrap();
1758        let CommonsItem::Fn(f) = &c.items[0] else {
1759            panic!()
1760        };
1761        assert_eq!(f.body.statements.len(), 1);
1762        match &f.body.statements[0] {
1763            Statement::Let(l) => {
1764                assert_eq!(l.name.name, "y");
1765                assert!(l.type_annot.is_none());
1766            }
1767            _ => panic!("expected a pure `let` statement"),
1768        }
1769    }
1770
1771    #[test]
1772    fn let_with_annotation() {
1773        let c = parse_str("commons x { fn f(n: Int) -> Int { let y: Int = n\n y } }").unwrap();
1774        let CommonsItem::Fn(f) = &c.items[0] else {
1775            panic!()
1776        };
1777        match &f.body.statements[0] {
1778            Statement::Let(l) => assert!(l.type_annot.is_some()),
1779            _ => panic!("expected a pure `let` statement"),
1780        }
1781    }
1782
1783    #[test]
1784    fn if_else_parses_as_expression() {
1785        let c = parse_str("commons x { fn f(b: Bool) -> Int { if b { 1 } else { 0 } } }").unwrap();
1786        let CommonsItem::Fn(f) = &c.items[0] else {
1787            panic!()
1788        };
1789        assert!(matches!(f.body.tail.kind, ExprKind::If { .. }));
1790    }
1791
1792    #[test]
1793    fn else_if_chain_parses() {
1794        let c = parse_str(
1795            "commons x { fn f(n: Int) -> Int { if n < 0 { -1 } else if n == 0 { 0 } else { 1 } } }",
1796        )
1797        .unwrap();
1798        let CommonsItem::Fn(f) = &c.items[0] else {
1799            panic!()
1800        };
1801        let ExprKind::If { else_block, .. } = &f.body.tail.kind else {
1802            panic!()
1803        };
1804        // The else-branch is a block whose tail is another `If`.
1805        assert!(else_block.statements.is_empty());
1806        assert!(matches!(else_block.tail.kind, ExprKind::If { .. }));
1807    }
1808
1809    #[test]
1810    fn ok_and_err_parse_as_expressions() {
1811        let c = parse_str("commons x { fn f(n: Int) -> Result[Int, String] { Ok(n) } }").unwrap();
1812        let CommonsItem::Fn(f) = &c.items[0] else {
1813            panic!()
1814        };
1815        assert!(matches!(f.body.tail.kind, ExprKind::Ok(_)));
1816
1817        let c =
1818            parse_str("commons x { fn f(n: Int) -> Result[Int, String] { Err(\"x\") } }").unwrap();
1819        let CommonsItem::Fn(f) = &c.items[0] else {
1820            panic!()
1821        };
1822        assert!(matches!(f.body.tail.kind, ExprKind::Err(_)));
1823    }
1824
1825    #[test]
1826    fn question_postfix_parses() {
1827        let c = parse_str(
1828            "commons x { type T = Int where Positive\n fn f(n: Int) -> Result[T, ValidationError] { let x = T.of(n)?\n Ok(x) } }",
1829        )
1830        .unwrap();
1831        let CommonsItem::Fn(f) = &c.items[1] else {
1832            panic!()
1833        };
1834        let Statement::Let(l) = &f.body.statements[0] else {
1835            panic!("expected a pure `let` statement");
1836        };
1837        assert!(matches!(l.value.kind, ExprKind::Question(_)));
1838    }
1839
1840    #[test]
1841    fn constructor_call_parses() {
1842        let c = parse_str(
1843            "commons x { type T = Int where Positive\n fn f(n: Int) -> Result[T, ValidationError] { T.of(n) } }",
1844        )
1845        .unwrap();
1846        let CommonsItem::Fn(f) = &c.items[1] else {
1847            panic!()
1848        };
1849        // v0.2: T.of(n) parses as a MethodCall with receiver Ident("T"); the
1850        // checker reinterprets it as a static call by noticing T is a type.
1851        let ExprKind::MethodCall {
1852            receiver, method, ..
1853        } = &f.body.tail.kind
1854        else {
1855            panic!("expected MethodCall, got {:?}", f.body.tail.kind)
1856        };
1857        let ExprKind::Ident(id) = &receiver.kind else {
1858            panic!("expected receiver Ident");
1859        };
1860        assert_eq!(id.name, "T");
1861        assert_eq!(method.name, "of");
1862    }
1863
1864    #[test]
1865    fn result_type_ref_parses() {
1866        let c = parse_str("commons x { fn f(n: Int) -> Result[Int, String] { Ok(n) } }").unwrap();
1867        let CommonsItem::Fn(f) = &c.items[0] else {
1868            panic!()
1869        };
1870        assert!(matches!(f.return_type, TypeRef::Result(_, _, _)));
1871    }
1872
1873    #[test]
1874    fn result_missing_arg_count_errors() {
1875        let errs = parse_str("commons x { fn f(n: Int) -> Result[Int] { Ok(n) } }").unwrap_err();
1876        assert_eq!(errs[0].category, "bynk.parse.generic_arg_count");
1877    }
1878
1879    #[test]
1880    fn field_access_parses_in_v0_2() {
1881        // v0.2: field access is supported (the type checker validates the
1882        // field exists on the receiver's type). Parser-level acceptance:
1883        let c =
1884            parse_str("commons x { type R = { foo: Int }\n fn f(r: R) -> Int { r.foo } }").unwrap();
1885        let CommonsItem::Fn(f) = &c.items[1] else {
1886            panic!()
1887        };
1888        assert!(matches!(f.body.tail.kind, ExprKind::FieldAccess { .. }));
1889    }
1890
1891    // -- v1.1 trivia attachment --
1892
1893    #[test]
1894    fn leading_line_comment_attaches_to_next_decl() {
1895        let src = "commons x {\n-- explain the type\ntype T = Int where NonNegative\n}";
1896        let c = parse_str(src).unwrap();
1897        let CommonsItem::Type(t) = &c.items[0] else {
1898            panic!()
1899        };
1900        assert_eq!(t.trivia.leading, vec![" explain the type".to_string()]);
1901        assert!(t.trivia.trailing.is_none());
1902    }
1903
1904    #[test]
1905    fn trailing_line_comment_attaches_to_prev_decl() {
1906        let src = "commons x {\ntype T = Int where NonNegative  -- trailing note\n}";
1907        let c = parse_str(src).unwrap();
1908        let CommonsItem::Type(t) = &c.items[0] else {
1909            panic!()
1910        };
1911        assert!(t.trivia.leading.is_empty());
1912        assert_eq!(t.trivia.trailing.as_deref(), Some(" trailing note"));
1913    }
1914
1915    #[test]
1916    fn grouped_leading_comments_attach_together() {
1917        let src = "commons x {\n-- one\n-- two\n-- three\ntype T = Int where Positive\n}";
1918        let c = parse_str(src).unwrap();
1919        let CommonsItem::Type(t) = &c.items[0] else {
1920            panic!()
1921        };
1922        assert_eq!(
1923            t.trivia.leading,
1924            vec![" one".to_string(), " two".to_string(), " three".to_string()],
1925        );
1926    }
1927
1928    #[test]
1929    fn comment_with_doc_block_keeps_both() {
1930        // Both `-- intro` and the doc block should attach to the type decl.
1931        let src = "commons x {\n-- intro\n---\ndocs\n---\ntype T = Int where Positive\n}";
1932        let c = parse_str(src).unwrap();
1933        let CommonsItem::Type(t) = &c.items[0] else {
1934            panic!()
1935        };
1936        assert_eq!(t.trivia.leading, vec![" intro".to_string()]);
1937        assert_eq!(t.documentation.as_deref(), Some("docs"));
1938    }
1939
1940    #[test]
1941    fn messages_keyword_does_not_collide_with_a_commons_name_segment() {
1942        // `messages` is RESERVED_CONTEXTUAL (like `case`/`on`/`suite`), not a
1943        // hard keyword: `commons app.messages { ... }` — the design's own
1944        // natural naming choice for a bundle commons — must still parse.
1945        // (Caught during slice-1 implementation: a first pass made `messages`
1946        // a plain hard keyword and this exact name broke.)
1947        let src = "commons app.messages {\ntype T = Int where Positive\n}";
1948        let c = parse_str(src).unwrap();
1949        assert_eq!(c.name.joined(), "app.messages");
1950    }
1951
1952    #[test]
1953    fn messages_decl_parses_tag_annotation_and_entries() {
1954        // message-bundles slice 1 (#859): the construct + doc/trivia wiring.
1955        let src = "commons app.messages {\n\
1956                   -- intro\n\
1957                   ---\n\
1958                   docs\n\
1959                   ---\n\
1960                   messages \"en\" @reference {\n\
1961                   \"greeting\" => \"Hello, {name}!\"\n\
1962                   \"farewell\" => \"Bye\"\n\
1963                   } -- trailing\n\
1964                   }";
1965        let c = parse_str(src).unwrap();
1966        let CommonsItem::Messages(m) = &c.items[0] else {
1967            panic!("expected a messages item, got {:?}", c.items[0]);
1968        };
1969        assert_eq!(m.tag, "en");
1970        assert_eq!(m.annotations.len(), 1);
1971        assert_eq!(m.annotations[0].name.name, "reference");
1972        assert!(m.annotations[0].args.is_empty());
1973        assert_eq!(m.entries.len(), 2);
1974        assert_eq!(m.entries[0].code, "greeting");
1975        assert_eq!(m.entries[0].template, "Hello, {name}!");
1976        assert_eq!(m.entries[1].code, "farewell");
1977        assert_eq!(m.entries[1].template, "Bye");
1978        assert_eq!(m.trivia.leading, vec![" intro".to_string()]);
1979        assert_eq!(m.documentation.as_deref(), Some("docs"));
1980        assert_eq!(m.trivia.trailing.as_deref(), Some(" trailing"));
1981    }
1982
1983    #[test]
1984    fn messages_decl_parses_with_no_annotation_and_no_entries() {
1985        // The parser stays permissive on annotation cardinality (zero-or-more)
1986        // — "exactly one `@reference`" is a checker concern (validate.rs), not
1987        // a parse error.
1988        let src = "commons app.messages {\nmessages \"en\" {\n}\n}";
1989        let c = parse_str(src).unwrap();
1990        let CommonsItem::Messages(m) = &c.items[0] else {
1991            panic!("expected a messages item, got {:?}", c.items[0]);
1992        };
1993        assert_eq!(m.tag, "en");
1994        assert!(m.annotations.is_empty());
1995        assert!(m.entries.is_empty());
1996    }
1997
1998    #[test]
1999    fn messages_decl_parses_syntactically_inside_a_context_too() {
2000        // Commons-only legality is a checker concern (bynk.messages.outside_commons
2001        // in bynk-emit's project validation), not a parser rejection — mirrors
2002        // how `service`/`agent` already parse syntactically inside `adapter`
2003        // bodies for the same reason.
2004        let src = "context app.svc {\nmessages \"en\" @reference {\n\"a\" => \"b\"\n}\n}";
2005        let toks = tokenize(src).unwrap();
2006        let (unit, errors) = parse_unit_with_recovery(&toks, src);
2007        assert!(errors.is_empty(), "unexpected parse errors: {errors:?}");
2008        let Some(SourceUnit::Context(ctx)) = unit else {
2009            panic!("expected a context")
2010        };
2011        let CommonsItem::Messages(m) = &ctx.items[0] else {
2012            panic!("expected a messages item, got {:?}", ctx.items[0]);
2013        };
2014        assert_eq!(m.tag, "en");
2015    }
2016
2017    #[test]
2018    fn comment_before_let_statement_attaches() {
2019        let src = "commons x {\nfn f(n: Int) -> Int {\n-- pick a value\nlet y = n + 1\ny\n}\n}";
2020        let c = parse_str(src).unwrap();
2021        let CommonsItem::Fn(f) = &c.items[0] else {
2022            panic!()
2023        };
2024        let Statement::Let(l) = &f.body.statements[0] else {
2025            panic!()
2026        };
2027        assert_eq!(l.trivia.leading, vec![" pick a value".to_string()]);
2028    }
2029
2030    #[test]
2031    fn comment_before_tail_attaches_to_block_tail() {
2032        let src = "commons x {\nfn f(n: Int) -> Int {\nlet y = n + 1\n-- result\ny\n}\n}";
2033        let c = parse_str(src).unwrap();
2034        let CommonsItem::Fn(f) = &c.items[0] else {
2035            panic!()
2036        };
2037        assert_eq!(f.body.tail_leading_comments, vec![" result".to_string()],);
2038    }
2039
2040    /// #637 Gap A: the contextual keywords `on` / `suite` / `case` are lexer
2041    /// tokens but `expect_ident` admits them as identifiers outside their one
2042    /// keyword position, so they are valid record-field and parameter names.
2043    /// The keyword reference now renders them as a distinct "contextual" tier
2044    /// rather than claiming (falsely) that they cannot be used as identifiers.
2045    #[test]
2046    fn contextual_keywords_are_valid_identifiers() {
2047        // Record field names.
2048        let c = parse_str("commons demo {\n  type R = { on: Int, suite: String, case: Bool }\n}")
2049            .expect("`on`/`suite`/`case` are valid field names");
2050        let CommonsItem::Type(_) = &c.items[0] else {
2051            panic!("expected a type decl")
2052        };
2053
2054        // Function parameter names (the other `expect_ident` position).
2055        parse_str("commons demo {\n  fn f(on: Int, case: Int) -> Int { 0 }\n}")
2056            .expect("`on`/`case` are valid parameter names");
2057
2058        // `suite` too, as a field name.
2059        parse_str("commons demo {\n  type R = { suite: Int }\n}")
2060            .expect("`suite` is a valid field name");
2061    }
2062
2063    /// Drift guard: every alphabetic keyword the lexer declares must be
2064    /// classified by `is_reserved_keyword`, or be one of the *contextual*
2065    /// keywords `expect_ident` deliberately admits as identifiers
2066    /// (`on`/`suite`/`case`). Everything else in this codebase that can
2067    /// drift has a guard; this predicate had silently fallen 17 keywords
2068    /// behind, degrading the reserved-keyword diagnostic to the generic
2069    /// expected-token one.
2070    #[test]
2071    fn is_reserved_keyword_covers_every_lexer_keyword() {
2072        let lexer_src = include_str!("lexer.rs");
2073        let mut words = Vec::new();
2074        for line in lexer_src.lines() {
2075            let t = line.trim();
2076            if let Some(rest) = t.strip_prefix("#[token(\"")
2077                && let Some(word) = rest.split('"').next()
2078                && word.chars().next().is_some_and(|c| c.is_ascii_alphabetic())
2079                && word.chars().all(|c| c.is_ascii_alphanumeric() || c == '_')
2080            {
2081                words.push(word.to_string());
2082            }
2083        }
2084        assert!(
2085            words.len() > 30,
2086            "keyword extraction looks broken: only {} words",
2087            words.len()
2088        );
2089        // Contextual keywords double as identifiers (see `expect_ident`); the
2090        // tier is single-sourced in `keywords::RESERVED_CONTEXTUAL`.
2091        use crate::keywords::RESERVED_CONTEXTUAL;
2092        let mut unclassified = Vec::new();
2093        for word in &words {
2094            let tokens = crate::lexer::tokenize(word).expect("keyword lexes");
2095            let kind = tokens.first().expect("keyword yields a token").kind;
2096            if !is_reserved_keyword(kind) && !RESERVED_CONTEXTUAL.contains(&word.as_str()) {
2097                unclassified.push(word.clone());
2098            }
2099        }
2100        assert!(
2101            unclassified.is_empty(),
2102            "keywords missing from is_reserved_keyword (add them, or document \
2103             them as contextual): {unclassified:?}"
2104        );
2105    }
2106
2107    /// Finding #27/#30: pins `is_item_start` to exactly the keyword set
2108    /// `declarations.rs`'s six item loops (`parse_commons_brace`/`_fragment`,
2109    /// `parse_context_brace`/`_fragment`, `parse_test_brace`/`_fragment`) plus
2110    /// `parse_adapter_body` dispatch on, as of this writing — the drift this
2111    /// finding fixed (`Property`, `Actor`, `Event`, `Binding`, and the
2112    /// `adapter` unit keyword itself were all missing from
2113    /// `recover_to_top_item`'s old hand-written sync list, even though every
2114    /// one of them is a real item/unit start). Adding a new item keyword to
2115    /// any of those loops should mean deliberately updating this list too,
2116    /// not silently leaving recovery unable to resync at it.
2117    #[test]
2118    fn is_item_start_matches_the_pinned_keyword_set() {
2119        use TokenKind::*;
2120        let expected_true = [
2121            Commons, Context, Adapter, Suite, Type, Fn, Messages, Event, Uses, Consumes, Exports,
2122            Capability, Provides, Service, Agent, Actor, Binding, Stub, Case, Property,
2123        ];
2124        for kind in expected_true {
2125            assert!(is_item_start(kind), "{kind:?} must be an item start");
2126        }
2127        let expected_false = [
2128            Ident, Plus, Minus, Colon, Dot, Eq, LBrace, RBrace, LParen, RParen, If, Else, Let,
2129            Where, True, False, Match, Is, On, Given,
2130        ];
2131        for kind in expected_false {
2132            assert!(!is_item_start(kind), "{kind:?} must not be an item start");
2133        }
2134    }
2135
2136    /// Fuzz-found (#516): a context-only keyword at item position in a
2137    /// commons errors without consuming the token, and the recovery sync
2138    /// stops at exactly that keyword — without a progress guard the item
2139    /// loop re-reported the same error until memory ran out.
2140    #[test]
2141    fn recovery_makes_progress_on_context_only_keyword_in_commons() {
2142        let src = "commons demo\n\ncapability Logger {\n  fn log(m: String) -> Effect[()]\n}\n";
2143        let tokens = crate::lexer::tokenize(src).unwrap();
2144        let (unit, errors) = parse_unit_with_recovery(&tokens, src);
2145        assert!(unit.is_some(), "the commons header still parses");
2146        assert!(
2147            errors
2148                .iter()
2149                .any(|e| e.category == "bynk.capability.outside_context"),
2150            "the misplaced capability is reported: {errors:?}"
2151        );
2152        // Termination is the real assertion (this used to OOM); a bounded,
2153        // non-repeating error list is the observable proxy.
2154        assert!(errors.len() < 10, "recovery repeated itself: {errors:?}");
2155    }
2156
2157    #[test]
2158    fn trailing_file_comment_becomes_unit_trailing() {
2159        // A comment after the last item but before EOF (fragment form)
2160        // becomes the commons body's trailing comments so the formatter
2161        // can preserve it.
2162        let src = "commons x\n\ntype T = Int where Positive\n-- afterword\n";
2163        let c = parse_str(src).unwrap();
2164        assert_eq!(c.trailing_comments, vec![" afterword".to_string()]);
2165    }
2166
2167    #[test]
2168    fn trailing_file_comment_after_a_brace_form_commons_is_not_dropped() {
2169        // Regression: the brace form's item loop exits on `RBrace`, never
2170        // reaching the fragment form's end-of-input case that drains the
2171        // trivia table's epilogue — so a comment after the closing `}` was
2172        // silently discarded (and, per `epilogue_is_empty`'s debug_assert,
2173        // would panic a debug build instead of round-tripping through
2174        // `bynk-fmt`).
2175        let src = "commons x {\n  type T = Int where Positive\n}\n-- afterword\n";
2176        let c = parse_str(src).unwrap();
2177        assert_eq!(c.trailing_comments, vec![" afterword".to_string()]);
2178    }
2179
2180    #[test]
2181    fn trailing_file_comment_after_a_brace_form_context_is_not_dropped() {
2182        // Same regression as the commons case, for `parse_context_brace`.
2183        let src = "context x {\n  type T = Int where Positive\n}\n-- afterword\n";
2184        let SourceUnit::Context(c) = parse_unit_str(src).unwrap() else {
2185            panic!("expected context");
2186        };
2187        assert_eq!(c.trailing_comments, vec![" afterword".to_string()]);
2188    }
2189
2190    #[test]
2191    fn trailing_file_comment_after_a_brace_form_suite_is_not_dropped() {
2192        // Same regression as the commons case, for `parse_test_brace`.
2193        let src = "suite x {\n  case \"c\" {\n    expect 1 == 1\n  }\n}\n-- afterword\n";
2194        let SourceUnit::Suite(s) = parse_unit_str(src).unwrap() else {
2195            panic!("expected suite");
2196        };
2197        assert_eq!(s.trailing_comments, vec![" afterword".to_string()]);
2198    }
2199
2200    /// Finding #30: unlike the three regressions just above,
2201    /// `parse_adapter_body`'s brace-closing path never called
2202    /// `take_epilogue` at all (not a regression from a shared pattern — it
2203    /// simply never had the call), so a comment after a brace-form adapter's
2204    /// closing `}` was silently dropped. The fragment form (no braces) was
2205    /// already correct.
2206    #[test]
2207    fn trailing_file_comment_after_a_brace_form_adapter_is_not_dropped() {
2208        let src = "adapter x {\n  binding \"./x.ts\"\n}\n-- afterword\n";
2209        let SourceUnit::Adapter(a) = parse_unit_str(src).unwrap() else {
2210            panic!("expected adapter");
2211        };
2212        assert_eq!(a.trailing_comments, vec![" afterword".to_string()]);
2213    }
2214
2215    // -- Six-fold unification (review Part 3): the fragment-only ordering
2216    // restrictions declarations.rs's brace/fragment pairs preserve, now that
2217    // they share one function each behind `brace: bool`. None of these had
2218    // any prior test coverage at all. --
2219
2220    /// Fragment-form commons: `uses` must precede every `type`/`fn`.
2221    #[test]
2222    fn commons_fragment_rejects_uses_after_a_decl() {
2223        let src = "commons x\n\ntype T = Int where Positive\nuses bynk.list\n";
2224        let errs = parse_str(src).unwrap_err();
2225        assert!(
2226            errs.iter()
2227                .any(|e| e.category == "bynk.parse.uses_after_decls"),
2228            "{errs:?}"
2229        );
2230    }
2231
2232    /// The same ordering is NOT enforced in brace form — `uses` may appear
2233    /// anywhere in the body.
2234    #[test]
2235    fn commons_brace_allows_uses_after_a_decl() {
2236        let src = "commons x {\n  type T = Int where Positive\n  uses bynk.list\n}\n";
2237        parse_str(src).expect("brace form must not enforce fragment's uses-ordering rule");
2238    }
2239
2240    /// Fragment-form context: `consumes` must precede every `type`/`fn`/etc.
2241    #[test]
2242    fn context_fragment_rejects_consumes_after_a_decl() {
2243        let src = "context x\n\ntype T = Int where Positive\nconsumes bynk\n";
2244        let errs = parse_unit_str(src).unwrap_err();
2245        assert!(
2246            errs.iter()
2247                .any(|e| e.category == "bynk.parse.consumes_after_decls"),
2248            "{errs:?}"
2249        );
2250    }
2251
2252    /// Fragment-form context: `exports` must precede every `type`/`fn`/etc.
2253    #[test]
2254    fn context_fragment_rejects_exports_after_a_decl() {
2255        let src = "context x\n\ntype T = Int where Positive\nexports opaque { T }\n";
2256        let errs = parse_unit_str(src).unwrap_err();
2257        assert!(
2258            errs.iter()
2259                .any(|e| e.category == "bynk.parse.exports_after_decls"),
2260            "{errs:?}"
2261        );
2262    }
2263
2264    /// Brace-form context enforces none of the three orderings.
2265    #[test]
2266    fn context_brace_allows_consumes_and_exports_after_a_decl() {
2267        let src = "context x {\n  type T = Int where Positive\n  consumes bynk\n  exports opaque { T }\n}\n";
2268        parse_unit_str(src)
2269            .expect("brace form must not enforce fragment's consumes/exports-ordering rules");
2270    }
2271
2272    /// Fragment-form suite/test: `uses` must precede every `stub`/`case`/`property`.
2273    #[test]
2274    fn test_fragment_rejects_uses_after_a_decl() {
2275        let src = "suite m\n\ncase \"c\" {\n  expect true\n}\nuses bynk.list\n";
2276        let errs = parse_unit_str(src).unwrap_err();
2277        assert!(
2278            errs.iter()
2279                .any(|e| e.category == "bynk.parse.uses_after_decls"),
2280            "{errs:?}"
2281        );
2282    }
2283
2284    /// Brace-form suite/test allows `uses` anywhere.
2285    #[test]
2286    fn test_brace_allows_uses_after_a_decl() {
2287        let src = "suite m {\n  case \"c\" {\n    expect true\n  }\n  uses bynk.list\n}\n";
2288        parse_unit_str(src).expect("brace form must not enforce fragment's uses-ordering rule");
2289    }
2290
2291    // ---- #636: `if`/`match` condition vs record construction ----
2292
2293    /// Parse `body` as the tail expression of a fn and return its kind.
2294    fn body_tail(body: &str) -> ExprKind {
2295        let src = format!("commons x\n\nfn f() -> Int {{\n  {body}\n}}\n");
2296        let c = parse_str(&src).unwrap_or_else(|e| panic!("parse failed for {body:?}: {e:?}"));
2297        let CommonsItem::Fn(f) = &c.items[0] else {
2298            panic!("expected fn, got {:?}", c.items[0]);
2299        };
2300        f.body.tail.kind.clone()
2301    }
2302
2303    fn body_err(body: &str) -> Vec<CompileError> {
2304        let src = format!("commons x\n\nfn f() -> Int {{\n  {body}\n}}\n");
2305        parse_str(&src).expect_err(&format!("expected a parse error for {body:?}"))
2306    }
2307
2308    #[test]
2309    fn if_condition_ending_in_ident_does_not_swallow_a_single_ident_branch() {
2310        // #636: `ready { result }` shares its shape with a shorthand-field
2311        // record construction. In condition position the branch must win.
2312        for src in [
2313            "if ready { result } else { fallback }",
2314            "if ready { fallback } else { result }",
2315            "if !ready { result } else { fallback }",
2316            "if a == b { result } else { fallback }",
2317            "if a && b { result } else { fallback }",
2318        ] {
2319            let ExprKind::If {
2320                then_block,
2321                else_block,
2322                ..
2323            } = body_tail(src)
2324            else {
2325                panic!("expected If for {src:?}, got {:?}", body_tail(src));
2326            };
2327            // Both branches carry a bare-identifier tail — proof the `{ … }`
2328            // was read as a block, not consumed as a record by the condition.
2329            assert!(
2330                matches!(&then_block.tail.kind, ExprKind::Ident(_)),
2331                "then-branch tail not an ident for {src:?}: {:?}",
2332                then_block.tail.kind,
2333            );
2334            assert!(
2335                matches!(&else_block.tail.kind, ExprKind::Ident(_)),
2336                "else-branch tail not an ident for {src:?}: {:?}",
2337                else_block.tail.kind,
2338            );
2339        }
2340    }
2341
2342    #[test]
2343    fn else_less_if_with_single_ident_branch_parses() {
2344        // The no-`else` reproduction: previously errored `found `}``.
2345        let ExprKind::If { then_block, .. } = body_tail("if ready { result }") else {
2346            panic!("expected If");
2347        };
2348        assert!(matches!(&then_block.tail.kind, ExprKind::Ident(_)));
2349    }
2350
2351    #[test]
2352    fn record_construction_still_parses_in_value_position() {
2353        // The restriction is confined to condition spines — an ordinary value
2354        // position still constructs records, including the shorthand tail form.
2355        assert!(matches!(
2356            body_tail("Point { x }"),
2357            ExprKind::RecordConstruction { .. }
2358        ));
2359        assert!(matches!(
2360            body_tail("Point { x: 1, y: 2 }"),
2361            ExprKind::RecordConstruction { .. }
2362        ));
2363        assert!(matches!(
2364            body_tail("Empty {}"),
2365            ExprKind::RecordConstruction { .. }
2366        ));
2367    }
2368
2369    #[test]
2370    fn parenthesised_record_is_allowed_in_condition_head() {
2371        // A delimiter lifts the restriction: `(ready { result })` constructs a
2372        // record even in condition position (mirrors Rust's paren escape).
2373        let ExprKind::If { cond, .. } =
2374            body_tail("if (ready { result }) { branch } else { other }")
2375        else {
2376            panic!("expected If");
2377        };
2378        let ExprKind::Paren(inner) = &cond.kind else {
2379            panic!("expected a parenthesised condition, got {:?}", cond.kind);
2380        };
2381        assert!(
2382            matches!(&inner.kind, ExprKind::RecordConstruction { .. }),
2383            "parenthesised record in condition head should still construct: {:?}",
2384            inner.kind,
2385        );
2386    }
2387
2388    #[test]
2389    fn record_in_call_arg_within_condition_still_constructs() {
2390        // The restriction is lifted through a call-argument delimiter, so a
2391        // record literal passed to a predicate in the condition still parses.
2392        let ExprKind::If { cond, .. } = body_tail("if check(Point { x: 1 }) { a } else { b }")
2393        else {
2394            panic!("expected If");
2395        };
2396        let ExprKind::Call { args, .. } = &cond.kind else {
2397            panic!("expected Call in condition, got {:?}", cond.kind);
2398        };
2399        assert!(matches!(&args[0].kind, ExprKind::RecordConstruction { .. }));
2400    }
2401
2402    #[test]
2403    fn safe_condition_shapes_are_unaffected() {
2404        // Cases the issue lists as already-safe must stay safe.
2405        assert!(matches!(
2406            body_tail("if ready == true { result } else { fallback }"),
2407            ExprKind::If { .. }
2408        ));
2409        assert!(matches!(
2410            body_tail("if (ready) { result } else { fallback }"),
2411            ExprKind::If { .. }
2412        ));
2413        assert!(matches!(
2414            body_tail("if ready { \"a\" } else { \"b\" }"),
2415            ExprKind::If { .. }
2416        ));
2417    }
2418
2419    #[test]
2420    fn empty_match_reports_its_own_diagnostic() {
2421        // #636: `match result {}` once parsed `result {}` as an empty record,
2422        // masking `bynk.parse.empty_match`. The intended diagnostic is now
2423        // reachable.
2424        let errs = body_err("match result {}");
2425        assert!(
2426            errs.iter().any(|e| e.category == "bynk.parse.empty_match"),
2427            "expected empty_match; got {errs:?}",
2428        );
2429    }
2430
2431    #[test]
2432    fn match_discriminant_ending_in_ident_parses() {
2433        // A `match` over a bare-identifier discriminant reaches its arm list.
2434        assert!(matches!(
2435            body_tail("match ready { x => x }"),
2436            ExprKind::Match { .. }
2437        ));
2438    }
2439
2440    /// #981: an identifier statement immediately followed, on its own line, by
2441    /// a standalone `()` must stay two separate constructs — not merge into a
2442    /// zero-arg call. `status := Paid` / `()` is an Assign statement whose
2443    /// value is the bare identifier `Paid`, then a unit tail; it must never
2444    /// parse as a single `status := Paid()` (a call). The call-parens rule
2445    /// mirrors the v0.20b same-line `[` rule already applied to type
2446    /// arguments: a postfix opener that begins a new line does not continue
2447    /// the previous token.
2448    #[test]
2449    fn identifier_statement_followed_by_unit_tail_does_not_merge_into_a_call() {
2450        let src = "commons c\n\nfn f() -> Int {\n  status := Paid\n  ()\n}\n";
2451        let c = parse_str(src).unwrap_or_else(|e| panic!("parse failed: {e:?}"));
2452        let CommonsItem::Fn(f) = &c.items[0] else {
2453            panic!("expected fn, got {:?}", c.items[0]);
2454        };
2455        assert_eq!(
2456            f.body.statements.len(),
2457            1,
2458            "expected exactly one Assign statement, got {:?}",
2459            f.body.statements
2460        );
2461        let Statement::Assign(a) = &f.body.statements[0] else {
2462            panic!(
2463                "expected an Assign statement, got {:?}",
2464                f.body.statements[0]
2465            );
2466        };
2467        assert!(
2468            matches!(a.value.kind, ExprKind::Ident(_)),
2469            "assign value must stay the bare identifier `Paid`, got {:?}",
2470            a.value.kind
2471        );
2472        assert!(
2473            matches!(f.body.tail.kind, ExprKind::UnitLit),
2474            "the `()` must remain the block's own tail, got {:?}",
2475            f.body.tail.kind
2476        );
2477    }
2478
2479    /// #981: the same same-line rule extends to a method call's parens — a
2480    /// `.method` immediately followed, on its own line, by a standalone `()`
2481    /// must not merge into `.method()`.
2482    #[test]
2483    fn method_reference_followed_by_unit_tail_does_not_merge_into_a_call() {
2484        let src = "commons c\n\nfn f() -> Int {\n  let y = x.field\n  ()\n}\n";
2485        let c = parse_str(src).unwrap_or_else(|e| panic!("parse failed: {e:?}"));
2486        let CommonsItem::Fn(f) = &c.items[0] else {
2487            panic!("expected fn, got {:?}", c.items[0]);
2488        };
2489        let Statement::Let(l) = &f.body.statements[0] else {
2490            panic!("expected a Let statement, got {:?}", f.body.statements[0]);
2491        };
2492        assert!(
2493            matches!(l.value.kind, ExprKind::FieldAccess { .. }),
2494            "let value must stay a field access, got {:?}",
2495            l.value.kind
2496        );
2497        assert!(
2498            matches!(f.body.tail.kind, ExprKind::UnitLit),
2499            "the `()` must remain the block's own tail, got {:?}",
2500            f.body.tail.kind
2501        );
2502    }
2503
2504    #[test]
2505    fn unparenthesised_record_in_condition_head_now_errors() {
2506        // #636 narrowing (matches Rust): a record literal in condition *head*
2507        // position must be parenthesised. Unparenthesised, `Point` reads as the
2508        // discriminant and `{ x: 1 }` as the arm list, whose first "arm" `x: 1`
2509        // is not an arm — so the parse fails. Pinned so the divergence from the
2510        // (still-accepting) tree-sitter grammar is deliberate, not a bug.
2511        assert!(
2512            !body_err("match Point { x: 1 } { p => p }").is_empty(),
2513            "unparenthesised record discriminant should not parse",
2514        );
2515        // Parenthesised, the record is the discriminant and the match parses.
2516        let ExprKind::Match { discriminant, .. } = body_tail("match (Point { x: 1 }) { p => p }")
2517        else {
2518            panic!("expected Match for the parenthesised form");
2519        };
2520        let ExprKind::Paren(inner) = &discriminant.kind else {
2521            panic!(
2522                "expected a parenthesised discriminant, got {:?}",
2523                discriminant.kind
2524            );
2525        };
2526        assert!(matches!(&inner.kind, ExprKind::RecordConstruction { .. }));
2527    }
2528}