Skip to main content

bynk_syntax/
parser.rs

1//! Hand-written recursive-descent parser for Bynk v0.
2//!
3//! Token grammar in spec §4. The expression parser uses one function per
4//! precedence level (§4.4). Errors carry spans and short fix-oriented
5//! messages; the parser does not currently attempt synchronisation, which
6//! means at most one parse error is reported per compilation.
7
8use crate::ast::*;
9use crate::error::CompileError;
10use crate::lexer::{Token, TokenKind, comment_body, doc_block_content, has_blank_line_between};
11use crate::span::Span;
12mod declarations;
13mod expressions;
14mod statements;
15mod types;
16
17/// Side-channel store for line-comment trivia (v1.1 LSP spec §3.5).
18///
19/// Built once up-front by [`split_trivia`] from the raw lexer token stream.
20/// Comments are removed from the token stream the parser walks; their text
21/// is filed into `leading` (comments on lines preceding a content token)
22/// and `trailing` (a single comment on the same line as a content token).
23/// The parser consumes entries through [`TriviaTable::take_leading`] and
24/// [`TriviaTable::take_trailing`] as it recognises declarations.
25#[derive(Debug, Default)]
26struct TriviaTable {
27    /// `leading[i]` holds the comment-body texts that appear immediately
28    /// before content token `i` (zero or more `--` lines, in source order,
29    /// not separated from the token by another content token).
30    leading: Vec<Vec<String>>,
31    /// `trailing[i]` holds an optional comment on the same source line as
32    /// content token `i`. Only one trailing comment is recorded per token
33    /// because a single `--` consumes the rest of the line. A doc block or an
34    /// opening `{` never has one: see [`split_trivia`].
35    trailing: Vec<Option<String>>,
36    /// Any pending leading comments at end-of-file (no content token
37    /// followed). Used to preserve file-trailing comments.
38    epilogue: Vec<String>,
39}
40
41impl TriviaTable {
42    fn take_leading(&mut self, index: usize) -> Vec<String> {
43        match self.leading.get_mut(index) {
44            Some(v) => std::mem::take(v),
45            None => Vec::new(),
46        }
47    }
48
49    fn take_trailing(&mut self, index: usize) -> Option<String> {
50        self.trailing.get_mut(index).and_then(|s| s.take())
51    }
52
53    fn take_epilogue(&mut self) -> Vec<String> {
54        std::mem::take(&mut self.epilogue)
55    }
56
57    /// True when every entry has been drained via `take_leading`/
58    /// `take_trailing`/`take_epilogue` — i.e. no comment was silently
59    /// dropped. Each harvest is already a `mem::take`, so anything still
60    /// present here is exactly the set of comments that never reached an
61    /// AST `Trivia` field.
62    ///
63    /// Deliberately **not** wired into a `debug_assert!` in the general parse
64    /// path: expressions carry no per-node trivia (§ "Comment trivia" in the
65    /// 2026-07-27 pipeline review), so an ordinary, valid program with a
66    /// comment inside a `match`/list/record/binop — a common, accepted
67    /// pattern `bynk-fmt`'s own comment-loss guard already handles
68    /// gracefully — would leave `leading`/`trailing` non-empty and trip it on
69    /// every compile, not just on formatting. Instead surfaced through
70    /// [`parse_units_with_drain_check`] (finding #66), whose one caller
71    /// (`bynk-fmt`) *does* care about exactly this signal.
72    /// [`Self::epilogue_is_empty`] is the narrower, safe-to-assert check.
73    fn is_fully_drained(&self) -> bool {
74        self.leading.iter().all(Vec::is_empty)
75            && self.trailing.iter().all(Option::is_none)
76            && self.epilogue.is_empty()
77    }
78
79    /// True when no file-trailing comment was left stranded. Unlike
80    /// [`Self::is_fully_drained`], this is safe to assert unconditionally: a
81    /// clean file's epilogue is empty by construction (nothing pending at
82    /// EOF), and the one shape that legitimately populates it — a top-level
83    /// trailing comment — is drained by every parse path that calls
84    /// `take_epilogue`. A brace-form declaration that forgets to is exactly
85    /// the bug this catches.
86    fn epilogue_is_empty(&self) -> bool {
87        self.epilogue.is_empty()
88    }
89}
90
91/// Remove `Comment` trivia tokens from `tokens` and bin them into a
92/// [`TriviaTable`] keyed against the surviving content tokens. A comment
93/// on the same source line as the preceding content token is recorded as
94/// that token's *trailing* trivia, unless that token is a doc block or an
95/// opening `{`; everything else is *leading* for the next content token.
96fn split_trivia(tokens: &[Token], source: &str) -> (Vec<Token>, TriviaTable) {
97    let mut filtered: Vec<Token> = Vec::with_capacity(tokens.len());
98    let mut table = TriviaTable::default();
99    let mut pending_leading: Vec<String> = Vec::new();
100    let mut last_content_end: Option<usize> = None;
101    for tok in tokens {
102        if tok.kind == TokenKind::Comment {
103            let body = comment_body(source, tok.span).to_string();
104            // If nothing has been buffered as leading for the next token and
105            // there is no newline between the previous content token and
106            // this comment, it trails that token. A doc block never has a
107            // trailing comment: its token ends past the newline after the
108            // closing `---`, so a `--` line directly under it would otherwise
109            // look same-line and be binned where nothing collects it (#1756).
110            // Nor does an opening `{`: no parser collects trivia right after
111            // one, so its comment leads whatever follows instead, the first
112            // item or the `}` of an empty body (#1788).
113            if pending_leading.is_empty()
114                && filtered
115                    .last()
116                    .is_some_and(|t| !matches!(t.kind, TokenKind::DocBlock | TokenKind::LBrace))
117                && let Some(prev_end) = last_content_end
118                && !source[prev_end..tok.span.start].contains('\n')
119            {
120                let last_idx = filtered.len() - 1;
121                // Only attach if no trailing already recorded (shouldn't
122                // happen because `--` consumes through end-of-line).
123                if table.trailing[last_idx].is_none() {
124                    table.trailing[last_idx] = Some(body);
125                    continue;
126                }
127            }
128            pending_leading.push(body);
129            continue;
130        }
131        filtered.push(*tok);
132        table.leading.push(std::mem::take(&mut pending_leading));
133        table.trailing.push(None);
134        last_content_end = Some(tok.span.end);
135    }
136    table.epilogue = pending_leading;
137    (filtered, table)
138}
139
140/// Parse a token slice into a [`Commons`] AST.
141///
142/// Accepts either form of v0.3 commons file:
143/// - Brace form: `commons name { items... }` (v0–v0.2 compatible).
144/// - Fragment form: `commons name uses... items...` to EOF (v0.3).
145pub fn parse(tokens: &[Token], source: &str) -> Result<Commons, Vec<CompileError>> {
146    parse_with_warnings(tokens, source).map(|(c, _warnings)| c)
147}
148
149/// [`parse`] with the non-fatal diagnostics threaded out alongside the AST
150/// (ADR 0117) — see [`parse_units_with_warnings`].
151pub fn parse_with_warnings(
152    tokens: &[Token],
153    source: &str,
154) -> Result<(Commons, Vec<CompileError>), Vec<CompileError>> {
155    let (unit, warnings) = parse_unit_with_warnings(tokens, source)?;
156    match unit {
157        SourceUnit::Commons(c) => Ok((c, warnings)),
158        SourceUnit::Context(ctx) => Err(vec![
159            CompileError::new(
160                "bynk.parse.unexpected_context",
161                ctx.span,
162                "expected a `commons` declaration but found a `context` declaration",
163            )
164            .with_note(
165                "contexts must be compiled as part of a project — pass the source directory, e.g. `bynkc compile --target bundle --output out src`",
166            ),
167        ]),
168        SourceUnit::Suite(t) => Err(vec![
169            CompileError::new(
170                "bynk.parse.unexpected_suite",
171                t.span,
172                "expected a `commons` declaration but found a `suite` declaration",
173            )
174            .with_note(
175                "tests must be compiled as part of a project — pass the source directory, e.g. `bynkc compile --target bundle --output out src`",
176            ),
177        ]),
178        SourceUnit::Adapter(a) => Err(vec![
179            CompileError::new(
180                "bynk.parse.unexpected_adapter",
181                a.span,
182                "expected a `commons` declaration but found an `adapter` declaration",
183            )
184            .with_note(
185                "adapters must be compiled as part of a project — pass the source directory, e.g. `bynkc compile --target bundle --output out src`",
186            ),
187        ]),
188    }
189}
190
191/// Parse a token slice into a [`SourceUnit`] with error recovery, returning a
192/// best-effort partial AST plus the full list of parse errors and warnings.
193///
194/// Used by the LSP: item-level recovery skips past a malformed declaration to
195/// the next top-level item, so multiple errors are reported per compilation
196/// rather than just the first. Compared to [`parse_unit`], this never bails;
197/// if no SourceUnit could be parsed at all (e.g. the file is empty or the
198/// header itself fails) the returned `Option` is `None`.
199///
200/// Keeps only the *first* unit — v0.113 allows more than one top-level unit
201/// per file (an atomic `commons` + `suite`, DECISION S), and every existing
202/// caller here is keyed on the primary declaration. [`parse_units_with_recovery`]
203/// is the same recovery parse without that narrowing, for the one caller
204/// (finding #29/#30) that needs every unit a file declares.
205pub fn parse_unit_with_recovery(
206    tokens: &[Token],
207    source: &str,
208) -> (Option<SourceUnit>, Vec<CompileError>) {
209    let (units, errors) = parse_units_with_recovery(tokens, source);
210    (units.into_iter().next(), errors)
211}
212
213/// [`parse_unit_with_recovery`], keeping **every** top-level unit instead of
214/// discarding all but the first (finding #29/#30). Used by the IDE's own parse
215/// entry point, which needs to see a trailing `suite` in an atomic
216/// `commons`+`suite` file, not just the primary declaration.
217pub fn parse_units_with_recovery(
218    tokens: &[Token],
219    source: &str,
220) -> (Vec<SourceUnit>, Vec<CompileError>) {
221    let recovered = parse_units_recovering(tokens, source);
222    (recovered.units, recovered.errors)
223}
224
225/// #1663: everything a recovering parse learns — the units it built, every
226/// syntax error, and the names of the top-level declarations it had to skip.
227#[derive(Debug)]
228pub struct Recovered {
229    pub units: Vec<SourceUnit>,
230    pub errors: Vec<CompileError>,
231    /// Declarations recovery dropped, by name (see `Parser::broken_decl_names`).
232    /// A caller keeps references to them from echoing as unknown names.
233    pub broken_decl_names: Vec<String>,
234}
235
236/// #1663: a strict parse's error(s) together with a recovering parse's, each
237/// reported once. The strict errors come first and always survive: some rules
238/// hold only for the strict single-unit parse (`extra_tokens`, a `commons`
239/// expected but a `context` found), which the multi-unit recovering parse
240/// accepts. A recovered error at the same position as one already listed is
241/// the same fault (the two parsers may name it differently) and is dropped.
242pub fn merge_syntax_errors(
243    strict: Vec<CompileError>,
244    recovered: Vec<CompileError>,
245) -> Vec<CompileError> {
246    let mut out = strict;
247    for e in recovered {
248        if !out.iter().any(|o| o.span.start == e.span.start) {
249            out.push(e);
250        }
251    }
252    out
253}
254
255/// [`parse_units_with_recovery`], also returning the names of the declarations
256/// recovery skipped (#1663).
257pub fn parse_units_recovering(tokens: &[Token], source: &str) -> Recovered {
258    parse_units_recovering_from(tokens, source, &mut 0)
259}
260
261/// [`parse_units_recovering`], continuing [`ExprId`] allocation from `next_id`
262/// rather than starting at 0 — see [`parse_unit_with_warnings_from`]. #1710:
263/// the project path checks a recovered file's surviving declarations
264/// alongside every other file's, so its ids must come from the same durable
265/// counter (`bynk_project::parse_cache`), or they would collide.
266pub fn parse_units_recovering_from(tokens: &[Token], source: &str, next_id: &mut u32) -> Recovered {
267    let (filtered, trivia) = split_trivia(tokens, source);
268    let mut warnings = Vec::new();
269    let mut p = Parser::new(&filtered, source, trivia, &mut warnings);
270    p.recover_mode = true;
271    p.next_expr_id = *next_id;
272    let mut units = Vec::new();
273    loop {
274        match p.parse_unit() {
275            Ok(u) => units.push(u),
276            Err(e) => {
277                p.recovered_errors.push(e);
278                break;
279            }
280        }
281        // A genuinely malformed trailing declaration is still surfaced via
282        // recovery — checked *after* each successful parse, matching
283        // `parse_unit`'s own "at least once" attempt on the first unit (an
284        // empty file must still produce its usual unexpected-EOF diagnostic,
285        // not silently yield an empty `units` with no error at all).
286        if p.peek().is_none() {
287            break;
288        }
289    }
290    let broken_decl_names = std::mem::take(&mut p.broken_decl_names);
291    *next_id = p.next_expr_id;
292    let mut all_errors = p.recovered_errors;
293    all_errors.append(&mut warnings);
294    Recovered {
295        units,
296        errors: all_errors,
297        broken_decl_names,
298    }
299}
300
301/// Parse a token slice into a [`SourceUnit`] — either a commons or a context.
302///
303/// Each `.bynk` file is exactly one declaration of one kind.
304pub fn parse_unit(tokens: &[Token], source: &str) -> Result<SourceUnit, Vec<CompileError>> {
305    parse_unit_with_warnings(tokens, source).map(|(unit, _warnings)| unit)
306}
307
308/// [`parse_unit`] with the non-fatal diagnostics threaded out alongside the
309/// AST (ADR 0117) — see [`parse_units_with_warnings`].
310pub fn parse_unit_with_warnings(
311    tokens: &[Token],
312    source: &str,
313) -> Result<(SourceUnit, Vec<CompileError>), Vec<CompileError>> {
314    parse_unit_with_warnings_from(tokens, source, &mut 0)
315}
316
317/// [`parse_unit_with_warnings`], continuing [`ExprId`] allocation from
318/// `next_id` instead of starting at 0, and writing the id one past the last
319/// one this parse handed out back into it. T3.4 (R2.4): every top-level parse
320/// entry point in this file constructs its own `Parser` and therefore its own
321/// zero-based id space; a caller that will check two files' output together
322/// in one pass (a multi-file commons — `bynk-emit`'s `collect_unit_methods`
323/// merges a type's methods from sibling files into the file that declares the
324/// type, before one `check_record` call) must thread one counter across every
325/// file it parses, or two independently-numbered files collide on the same
326/// id in the same `expr_types` map. Every *other* caller (a single buffer, an
327/// LSP hover/completion query, a fixture test) never merges its output with
328/// another file's before checking, so starting at 0 every time is correct —
329/// [`parse_unit_with_warnings`] above is that default, unchanged.
330pub fn parse_unit_with_warnings_from(
331    tokens: &[Token],
332    source: &str,
333    next_id: &mut u32,
334) -> Result<(SourceUnit, Vec<CompileError>), Vec<CompileError>> {
335    let (filtered, trivia) = split_trivia(tokens, source);
336    let mut warnings = Vec::new();
337    let mut p = Parser::new(&filtered, source, trivia, &mut warnings);
338    p.next_expr_id = *next_id;
339    let result = match p.parse_unit() {
340        Ok(u) => {
341            if let Some(extra) = p.peek() {
342                Err(vec![
343                    CompileError::new(
344                        "bynk.parse.extra_tokens",
345                        extra.span,
346                        "unexpected token after top-level declaration",
347                    )
348                    .with_note(
349                        "a `.bynk` file contains exactly one `commons` or `context` declaration",
350                    ),
351                ])
352            } else {
353                Ok(u)
354            }
355        }
356        Err(e) => Err(vec![e]),
357    };
358    *next_id = p.next_expr_id;
359    // ADR 0117: warnings (e.g. orphan doc blocks) ride alongside a successful
360    // parse — severity governs gating at the caller, not here.
361    match result {
362        Ok(u) => {
363            // See `parse_units_with_warnings`: a file-trailing comment must
364            // have been drained by `take_epilogue`.
365            debug_assert!(
366                p.trivia.epilogue_is_empty(),
367                "a file-trailing comment was left undrained after a successful parse"
368            );
369            Ok((u, warnings))
370        }
371        Err(mut errs) => {
372            errs.append(&mut warnings);
373            Err(errs)
374        }
375    }
376}
377
378/// Parse a token slice into **all** the top-level [`SourceUnit`]s in one file
379/// (v0.113, testing track slice 1b). A `.bynk` file may hold more than one
380/// top-level declaration — an *atomic* file with `commons`/`context` **and** a
381/// `suite` together (DECISION S) — so the compiler parses a `Vec`, not a single
382/// unit. Test-ness is a property of each declaration, not of the file.
383///
384/// Bails on the first malformed declaration (like [`parse_unit`], not the
385/// recovering LSP path). An empty file is an error.
386pub fn parse_units(tokens: &[Token], source: &str) -> Result<Vec<SourceUnit>, Vec<CompileError>> {
387    parse_units_with_warnings(tokens, source).map(|(units, _warnings)| units)
388}
389
390/// [`parse_units`] with the non-fatal diagnostics threaded out alongside the
391/// AST (ADR 0117): a successful parse returns `Ok((units, warnings))` instead
392/// of hard-failing on a warning-severity diagnostic (an orphan doc block used
393/// to abort file discovery and throw the good AST away). A failed parse still
394/// returns every diagnostic — errors then warnings — in the `Err`.
395pub fn parse_units_with_warnings(
396    tokens: &[Token],
397    source: &str,
398) -> Result<(Vec<SourceUnit>, Vec<CompileError>), Vec<CompileError>> {
399    parse_units_with_drain_check(tokens, source)
400        .map(|(units, warnings, _drained)| (units, warnings))
401}
402
403/// [`parse_units_with_warnings`], continuing [`ExprId`] allocation from
404/// `next_id` rather than starting at 0 — see [`parse_unit_with_warnings_from`]
405/// for why this exists. The one production caller is `bynk-emit`'s per-file
406/// parse loop (`phase_parse`), which owns one counter across every file in a
407/// single project parse so two files whose methods later get merged into one
408/// `check_record` call (a multi-file commons) never collide.
409pub fn parse_units_with_warnings_from(
410    tokens: &[Token],
411    source: &str,
412    next_id: &mut u32,
413) -> Result<(Vec<SourceUnit>, Vec<CompileError>), Vec<CompileError>> {
414    parse_units_with_drain_check_from(tokens, source, next_id)
415        .map(|(units, warnings, _drained)| (units, warnings))
416}
417
418/// [`parse_units_with_warnings`] plus whether every comment's trivia was
419/// drained into the AST (`TriviaTable::is_fully_drained`). Finding #66:
420/// `bynk-fmt`'s comment-preservation guard re-tokenized its own rendered
421/// output just to diff comment bodies against the input — wasted work in the
422/// overwhelming common case where nothing was left behind. That guard uses
423/// this drain signal, computed from the same parse it already needs for
424/// rendering, as a fast-path: `true` means every comment landed in the AST,
425/// so re-checking the output can be skipped outright. No other caller needs
426/// the signal, so it rides its own entry point rather than widening
427/// [`parse_units_with_warnings`].
428pub fn parse_units_with_drain_check(
429    tokens: &[Token],
430    source: &str,
431) -> Result<(Vec<SourceUnit>, Vec<CompileError>, bool), Vec<CompileError>> {
432    parse_units_with_drain_check_from(tokens, source, &mut 0)
433}
434
435/// [`parse_units_with_drain_check`], continuing [`ExprId`] allocation from
436/// `next_id` — see [`parse_unit_with_warnings_from`].
437pub fn parse_units_with_drain_check_from(
438    tokens: &[Token],
439    source: &str,
440    next_id: &mut u32,
441) -> Result<(Vec<SourceUnit>, Vec<CompileError>, bool), Vec<CompileError>> {
442    let (filtered, trivia) = split_trivia(tokens, source);
443    let mut warnings = Vec::new();
444    let mut p = Parser::new(&filtered, source, trivia, &mut warnings);
445    p.next_expr_id = *next_id;
446    let mut units = Vec::new();
447    let mut errors: Vec<CompileError> = Vec::new();
448    while p.peek().is_some() {
449        match p.parse_unit() {
450            Ok(u) => units.push(u),
451            Err(e) => {
452                errors.push(e);
453                break;
454            }
455        }
456    }
457    *next_id = p.next_expr_id;
458    let eof = p.eof_span();
459    let fully_drained = p.trivia.is_fully_drained();
460    // `p` (and thus its `&mut warnings` borrow) is no longer used past here, so
461    // the local `warnings` are readable again.
462    if !errors.is_empty() {
463        errors.append(&mut warnings);
464        return Err(errors);
465    }
466    if units.is_empty() {
467        return Err(vec![CompileError::new(
468            "bynk.parse.unexpected_eof",
469            eof,
470            "expected `commons`, `context`, or `suite` to start the file, found end of file",
471        )]);
472    }
473    // A file-trailing comment must have been drained by `take_epilogue` — the
474    // one brace-form declarations forgot to call (a live comment-loss bug,
475    // not the fundamentally-unfixed expression-interior case: expressions
476    // carry no trivia at all, so asserting full drainage here would fire on
477    // any ordinary program with a comment inside a `match`/list/record, which
478    // `bynk-fmt`'s own comment-loss guard already handles gracefully rather
479    // than as a hard failure).
480    debug_assert!(
481        p.trivia.epilogue_is_empty(),
482        "a file-trailing comment was left undrained after a successful parse"
483    );
484    Ok((units, warnings, fully_drained))
485}
486
487/// A signed numeric literal in refinement-bound position (v0.21): `InRange`
488/// bounds are either both `Int` or both `Float`.
489enum SignedNumLit {
490    Int(IntBound),
491    Float(FloatBound),
492}
493
494struct Parser<'a> {
495    tokens: &'a [Token],
496    source: &'a str,
497    pos: usize,
498    /// Accumulated non-fatal diagnostics. v0.3 uses this for orphan-doc
499    /// warnings, which are emitted as errors with a distinguishable category.
500    warnings: &'a mut Vec<CompileError>,
501    /// When true, the item-level loops catch errors from individual item
502    /// parses, push them into `recovered_errors`, and skip forward to the
503    /// next top-level item boundary instead of bailing. Used by the LSP via
504    /// [`parse_unit_with_recovery`]; disabled in the normal `parse` path so
505    /// existing single-error behaviour is preserved.
506    recover_mode: bool,
507    /// Errors collected during recovery-mode parsing. Only populated when
508    /// `recover_mode` is true.
509    recovered_errors: Vec<CompileError>,
510    /// #1663: the token index where the current top-level item starts, set by
511    /// each top-level item loop. Read once by [`Self::handle_item_err`] to name
512    /// the declaration a recovery skips.
513    pub(crate) item_start: Option<usize>,
514    /// #1663: the names of top-level declarations recovery skipped (a `type`,
515    /// `fn`, method, `event`, `capability`, `service`, `agent` or `actor`).
516    /// They are known names whose declaration is broken, so a caller can keep
517    /// references to them from echoing as unknown. Recovery mode only.
518    broken_decl_names: Vec<String>,
519    /// Line-comment trivia separated from the token stream. See
520    /// [`TriviaTable`].
521    trivia: TriviaTable,
522    /// Live recursion depth of the three self-recursive parse entry points
523    /// (`parse_expr`, `parse_type_ref`, `parse_pattern`). Incremented on entry
524    /// and decremented on exit by [`Parser::enter_recursion`] so it tracks the
525    /// current stack depth; when it exceeds [`crate::MAX_NESTING_DEPTH`] the
526    /// parser reports a bounded-depth diagnostic instead of overflowing its
527    /// stack (#713).
528    depth: usize,
529    /// When true, a bare `ident {` on the *spine* of the current expression is
530    /// an identifier followed by an unrelated block, never a record
531    /// construction — so an `if`/`match` condition that ends in a bare
532    /// identifier does not swallow the branch/arm block as `Ident { field }`
533    /// (#636). Set only around the condition parse (see [`parse_cond_expr`]);
534    /// `parse_expr` clears it, so the restriction is lifted inside any
535    /// delimited sub-expression (parentheses, call arguments, list, record
536    /// field). Mirrors Rust's `NO_STRUCT_LITERAL` restriction.
537    no_record_literal: bool,
538    /// Running count of unclosed `{` seen so far — maintained solely by
539    /// [`Self::bump`] (the one primitive that advances `self.pos`), so it
540    /// always reflects the true nesting depth no matter which parse function
541    /// is on the call stack. Finding #27/#30: `recover_to_top_item` reads it
542    /// against [`Self::item_loop_baseline`] to tell "the enclosing item
543    /// loop's own closing brace" apart from a still-unclosed nested
544    /// construct's — without it, a sync scan that started partway through
545    /// such a construct (an error deep inside a function body) stopped at the
546    /// first `}` it saw, however deeply nested, and the enclosing item loop
547    /// mistook that for its own body's end.
548    brace_depth: usize,
549    /// Stack of `brace_depth` snapshots, one per active item-loop body
550    /// (commons/context/adapter/suite) — pushed right after that body's own
551    /// `{` is consumed (or at loop entry, for a brace-free fragment form),
552    /// popped at the loop's normal exit. `recover_to_top_item` treats its top
553    /// entry as the depth an `}` must return to before it counts as the
554    /// enclosing body's own closing brace rather than a nested construct's.
555    item_loop_baseline: Vec<usize>,
556    /// T3.4 (R2.4): next [`ExprId`] to hand out — monotonic, incremented by
557    /// [`Self::alloc_expr_id`], the sole allocation point every `Expr`
558    /// construction site in this parser calls.
559    next_expr_id: u32,
560}
561
562impl<'a> Parser<'a> {
563    fn new(
564        tokens: &'a [Token],
565        source: &'a str,
566        trivia: TriviaTable,
567        warnings: &'a mut Vec<CompileError>,
568    ) -> Self {
569        Self {
570            tokens,
571            source,
572            pos: 0,
573            warnings,
574            recover_mode: false,
575            recovered_errors: Vec::new(),
576            item_start: None,
577            broken_decl_names: Vec::new(),
578            trivia,
579            depth: 0,
580            no_record_literal: false,
581            brace_depth: 0,
582            item_loop_baseline: Vec::new(),
583            next_expr_id: 0,
584        }
585    }
586
587    /// T3.4 (R2.4): allocate the next [`ExprId`]. The sole allocation point —
588    /// every `Expr { id: self.alloc_expr_id(), .. }` construction in this
589    /// parser calls it exactly once, so two nodes never share an id and every
590    /// id a caller holds was actually handed out here.
591    fn alloc_expr_id(&mut self) -> ExprId {
592        let id = ExprId(self.next_expr_id);
593        self.next_expr_id += 1;
594        id
595    }
596
597    /// Enter a self-recursive parse step, bumping the live recursion depth and
598    /// failing with a bounded-depth diagnostic if it would exceed
599    /// [`crate::MAX_NESTING_DEPTH`]. The caller pairs a successful entry with a
600    /// matching `self.depth -= 1` on the way out (see `parse_expr` /
601    /// `parse_type_ref`); on the error path the depth is restored here so a
602    /// recovering caller is not left mis-counted. `what` names the construct
603    /// for the message (e.g. "this expression", "this type"). See #713.
604    fn enter_recursion(&mut self, what: &str) -> Result<(), CompileError> {
605        self.depth += 1;
606        if self.depth > crate::MAX_NESTING_DEPTH {
607            self.depth -= 1;
608            let span = self
609                .peek()
610                .map(|t| t.span)
611                .unwrap_or_else(|| self.eof_span());
612            return Err(self.nesting_too_deep(span, what));
613        }
614        Ok(())
615    }
616
617    /// The bounded-depth diagnostic shared by [`enter_recursion`] and
618    /// [`enter_chain_fold`].
619    fn nesting_too_deep(&self, span: Span, what: &str) -> CompileError {
620        CompileError::new(
621            "bynk.parse.nesting_too_deep",
622            span,
623            format!(
624                "{what} nests more than {} levels deep",
625                crate::MAX_NESTING_DEPTH
626            ),
627        )
628        .with_note(
629            "deeply nested source is rejected to keep the parser from overflowing its \
630             stack and aborting; flatten or split the construct",
631        )
632    }
633
634    /// The bounded-depth diagnostic for the *iteratively*-built spines —
635    /// associative operator chains ([`enter_chain_fold`]) and postfix receiver
636    /// chains ([`deepen_spine`]). Same code as [`nesting_too_deep`] (one budget,
637    /// one diagnostic) but phrased for a flat chain, which is long rather than
638    /// *nested*, and points at the idiomatic fix.
639    fn expression_too_long(&self, span: Span) -> CompileError {
640        CompileError::new(
641            "bynk.parse.nesting_too_deep",
642            span,
643            format!(
644                "this expression is more than {} levels deep",
645                crate::MAX_NESTING_DEPTH
646            ),
647        )
648        .with_note(
649            "a long operator or member chain is rejected to keep the compiler from overflowing \
650             its stack; split it across `let` bindings, or reduce a sequence with \
651             `.sum()`/`.fold(...)`",
652        )
653    }
654
655    /// Count one more operand folded onto an associative operator chain against
656    /// the same recursion budget as [`enter_recursion`] (#714).
657    ///
658    /// Associative chains (`+`, `*`, `&&`, `||`) are built *iteratively* in the
659    /// precedence ladder, so — unlike parentheses, calls, or `implies` — they
660    /// never re-enter `parse_expr` and thus slip past the `enter_recursion`
661    /// guard. Yet each fold deepens the left-nested `Expr` tree by one level,
662    /// and a long flat chain (`1 + 1 + … + 1`) overflows every *recursive*
663    /// consumer of that tree downstream — the checker's `type_of`, the
664    /// formatter, the emitter, and the AST's own recursive `Drop` — exactly as
665    /// deeply nested source overflows the parser. Counting each fold on the
666    /// shared `depth` budget bounds the whole expression's height, and because
667    /// it is the *same* budget it composes with the ambient nesting depth, so a
668    /// chain buried inside deeply nested source cannot exceed the bound either.
669    ///
670    /// The caller accumulates `folds` and subtracts them from `depth` before it
671    /// returns, so the live count unwinds as a recursive descent would; on the
672    /// overflow path the whole chain's contribution is restored here so a
673    /// recovering caller is not left mis-counted.
674    fn enter_chain_fold(&mut self, folds: &mut usize, span: Span) -> Result<(), CompileError> {
675        self.depth += 1;
676        *folds += 1;
677        if self.depth > crate::MAX_NESTING_DEPTH {
678            self.depth -= *folds;
679            *folds = 0;
680            return Err(self.expression_too_long(span));
681        }
682        Ok(())
683    }
684
685    /// Count one more level of an iteratively-built postfix receiver spine
686    /// (`a.b.c…`, `f()?.g()…`) against the shared budget (#714). Like
687    /// [`enter_chain_fold`], postfix loops rather than recurses, so a long spine
688    /// escapes [`enter_recursion`] yet grows an arbitrarily deep receiver tree
689    /// that the downstream walks recurse through. `parse_postfix` restores
690    /// `depth` wholesale on the way out (its many error paths make a
691    /// save/restore wrapper cleaner than per-fold unwinding), so this only bumps
692    /// and checks.
693    fn deepen_spine(&mut self, span: Span) -> Result<(), CompileError> {
694        self.depth += 1;
695        if self.depth > crate::MAX_NESTING_DEPTH {
696            return Err(self.expression_too_long(span));
697        }
698        Ok(())
699    }
700
701    /// Comments immediately preceding the current peek position. Consumed
702    /// (the table entry is cleared) so the same comments are not attached
703    /// to two nodes.
704    fn take_leading_trivia(&mut self) -> Vec<Comment> {
705        self.trivia
706            .take_leading(self.pos)
707            .into_iter()
708            .map(Comment::Line)
709            .collect()
710    }
711
712    /// The comments after the last token, as [`Comment::Line`]s.
713    fn take_epilogue_trivia(&mut self) -> Vec<Comment> {
714        self.trivia
715            .take_epilogue()
716            .into_iter()
717            .map(Comment::Line)
718            .collect()
719    }
720
721    /// Trailing comment, if any, on the same source line as the most
722    /// recently consumed content token. Call AFTER finishing a declaration
723    /// or statement, while `self.pos` points one past its last token.
724    fn take_trailing_trivia(&mut self) -> Option<String> {
725        if self.pos == 0 {
726            return None;
727        }
728        self.trivia.take_trailing(self.pos - 1)
729    }
730
731    /// Handle a per-item parse error. In recovery mode, record the error and
732    /// advance to the next sync point so the item loop can continue; otherwise
733    /// propagate as a hard failure.
734    fn handle_item_err(&mut self, e: CompileError) -> Result<(), CompileError> {
735        if self.recover_mode {
736            self.recovered_errors.push(e);
737            if let Some(start) = self.item_start.take() {
738                self.broken_decl_names.extend(self.decl_names_at(start));
739            }
740            let before = self.pos;
741            self.recover_to_top_item();
742            // The sync target may be the very token that produced the error —
743            // a context-only keyword (`capability`, `service`, …) at item
744            // position in a commons errors *without consuming it*, and it is
745            // itself a sync point. Recovery must always make progress, or the
746            // item loop re-reports the same error until memory runs out
747            // (found by the `parse` fuzz target on a seed input).
748            // #1663: unless the cursor is on the `}` that closes this item
749            // loop's own body (an error raised at the end of the last item):
750            // consuming it would make the body look unclosed, a follow-on
751            // `unexpected_eof`. The loop ends on that `}` itself. A fragment-
752            // form loop has no closing brace (baseline 0), so a stray `}`
753            // there is still consumed.
754            let baseline = self.item_loop_baseline.last().copied().unwrap_or(0);
755            let closes_body = baseline > 0
756                && self.brace_depth == baseline
757                && self.peek_kind() == Some(TokenKind::RBrace);
758            if self.pos == before && !closes_body {
759                self.bump();
760                // #1663: and skip the rest of the rejected item too (`agent`
761                // in a commons: its name and `{ … }` body). Otherwise the next
762                // loop iteration misreads its name as a malformed item and
763                // reports a second, follow-on syntax error.
764                self.recover_to_top_item();
765            }
766            Ok(())
767        } else {
768            Err(e)
769        }
770    }
771
772    /// #1663: the name(s) a declaration starting at token `start` declares:
773    /// `type T`, `fn f`, `fn T.m` (recorded as `T.m`), `event E`,
774    /// `capability C`, `service S`, `agent A`, `actor A`. Empty when the item
775    /// is not one of those or its name never parsed (`fn ( -> Int`).
776    fn decl_names_at(&self, start: usize) -> Vec<String> {
777        let kind_at = |i: usize| self.tokens.get(i).map(|t| t.kind);
778        let ident_at = |i: usize| {
779            self.tokens
780                .get(i)
781                .filter(|t| t.kind == TokenKind::Ident)
782                .map(|t| self.slice(t.span).to_string())
783        };
784        match kind_at(start) {
785            Some(
786                TokenKind::Type
787                | TokenKind::Event
788                | TokenKind::Capability
789                | TokenKind::Service
790                | TokenKind::Agent
791                | TokenKind::Actor,
792            ) => ident_at(start + 1).into_iter().collect(),
793            Some(TokenKind::Fn) => match (ident_at(start + 1), kind_at(start + 2)) {
794                // A method is recorded qualified, `T.m`, so a broken method
795                // is never mistaken for a free `fn m` of the same name.
796                (Some(owner), Some(TokenKind::Dot)) => match ident_at(start + 3) {
797                    Some(method) => vec![format!("{owner}.{method}")],
798                    None => Vec::new(),
799                },
800                (Some(name), _) => vec![name],
801                (None, _) => Vec::new(),
802            },
803            _ => Vec::new(),
804        }
805    }
806
807    /// Skip forward to the next top-level item boundary: either an
808    /// [`is_item_start`] keyword at the enclosing item loop's own nesting
809    /// depth, a closing brace that returns to that depth, or end-of-input.
810    /// Used only in recovery mode.
811    ///
812    /// Finding #27/#30: brace-depth-gated against
813    /// [`Self::item_loop_baseline`], so a `}` deep inside a still-unclosed
814    /// nested construct (an error partway through a function body, itself
815    /// inside a `match` arm) is skipped over rather than mistaken for the
816    /// enclosing body's own closing brace — the old flat scan stopped at
817    /// literally the first `}` it saw, however deep, handing the item loop a
818    /// brace that did not belong to it and making it return with zero items.
819    fn recover_to_top_item(&mut self) {
820        let baseline = self.item_loop_baseline.last().copied().unwrap_or(0);
821        while let Some(t) = self.peek() {
822            match t.kind {
823                TokenKind::RBrace if self.brace_depth == baseline => return,
824                _ if self.brace_depth == baseline && is_item_start(t.kind) => return,
825                _ => {
826                    self.bump();
827                }
828            }
829        }
830    }
831
832    /// Mark the start of a top-level item loop (`declarations.rs`'s
833    /// `parse_commons_brace`/`_fragment`, `parse_context_brace`/`_fragment`,
834    /// `parse_test_brace`/`_fragment`, `parse_adapter_body`) — called right
835    /// after that body's own `{` is consumed (brace form) or at the loop's
836    /// own entry (fragment form, which has no enclosing brace of its own).
837    /// Paired with [`Self::exit_item_loop`] at the loop's normal exit.
838    fn enter_item_loop(&mut self) {
839        self.item_loop_baseline.push(self.brace_depth);
840    }
841
842    /// Pair of [`Self::enter_item_loop`].
843    fn exit_item_loop(&mut self) {
844        self.item_loop_baseline.pop();
845    }
846
847    fn peek(&self) -> Option<Token> {
848        self.tokens.get(self.pos).copied()
849    }
850
851    fn peek_kind(&self) -> Option<TokenKind> {
852        self.peek().map(|t| t.kind)
853    }
854
855    /// The token `n` positions ahead of the cursor (`nth(0)` == `peek()`).
856    fn nth(&self, n: usize) -> Option<Token> {
857        self.tokens.get(self.pos + n).copied()
858    }
859
860    fn nth_kind(&self, n: usize) -> Option<TokenKind> {
861        self.nth(n).map(|t| t.kind)
862    }
863
864    /// The source text of the token `n` positions ahead, or `""` if none.
865    fn nth_text(&self, n: usize) -> &'a str {
866        self.nth(n).map(|t| self.slice(t.span)).unwrap_or("")
867    }
868
869    /// The span of the most recently consumed token (`self.pos - 1`). Falls back
870    /// to the current token's span when nothing has been consumed yet.
871    fn prev_span(&self) -> Span {
872        self.tokens
873            .get(self.pos.wrapping_sub(1))
874            .or_else(|| self.peek_ref())
875            .map(|t| t.span)
876            .unwrap_or_default()
877    }
878
879    fn peek_ref(&self) -> Option<&Token> {
880        self.tokens.get(self.pos)
881    }
882
883    fn bump(&mut self) -> Option<Token> {
884        let t = self.peek();
885        if let Some(t) = t {
886            match t.kind {
887                TokenKind::LBrace => self.brace_depth += 1,
888                TokenKind::RBrace => self.brace_depth = self.brace_depth.saturating_sub(1),
889                _ => {}
890            }
891            self.pos += 1;
892        }
893        t
894    }
895
896    fn eat(&mut self, kind: TokenKind) -> Option<Token> {
897        if self.peek_kind() == Some(kind) {
898            self.bump()
899        } else {
900            None
901        }
902    }
903
904    fn slice(&self, span: Span) -> &'a str {
905        &self.source[span.range()]
906    }
907
908    /// True when the next token sits on a later line than `prev`. Used to
909    /// keep a `[` that opens a new line out of the postfix type-application
910    /// form: `f` followed by `[1, 2]` on the next line is an identifier and
911    /// a list literal, not `f[…]` (v0.20b).
912    fn next_token_on_new_line(&self, prev: Span) -> bool {
913        match self.peek() {
914            Some(t) if prev.end <= t.span.start => {
915                self.source[prev.end..t.span.start].contains('\n')
916            }
917            _ => false,
918        }
919    }
920
921    /// Span pointing at the end of input — used for "unexpected EOF" reports.
922    /// The start backs up to the **start of the final char**, not `len - 1`, so
923    /// the span never splits a multibyte codepoint (an unterminated construct
924    /// whose last line ends in non-ASCII — e.g. a `--` comment ending in `→`).
925    fn eof_span(&self) -> Span {
926        let end = self.source.len();
927        let start = (0..end)
928            .rev()
929            .find(|&i| self.source.is_char_boundary(i))
930            .unwrap_or(0);
931        Span::new(start, end)
932    }
933
934    fn expect(&mut self, kind: TokenKind, ctx: &str) -> Result<Token, CompileError> {
935        match self.peek() {
936            Some(t) if t.kind == kind => {
937                self.bump();
938                Ok(t)
939            }
940            Some(t) => Err(CompileError::new(
941                "bynk.parse.expected_token",
942                t.span,
943                format!(
944                    "expected {} {ctx}, found {}",
945                    kind.describe(),
946                    t.kind.describe()
947                ),
948            )),
949            None => Err(CompileError::new(
950                "bynk.parse.unexpected_eof",
951                self.eof_span(),
952                format!("expected {} {ctx}, found end of file", kind.describe()),
953            )),
954        }
955    }
956
957    fn expect_ident(&mut self, ctx: &str) -> Result<Ident, CompileError> {
958        match self.peek() {
959            Some(t) if t.kind == TokenKind::Ident => {
960                self.bump();
961                Ok(Ident {
962                    name: self.slice(t.span).to_string(),
963                    span: t.span,
964                })
965            }
966            // v0.5 contextual keyword `on` doubles as an identifier in
967            // expression / field-access positions so users can name fields and
968            // parameters using it. It retains its keyword meaning only at
969            // handler-decl-level (`on call(...)`).
970            //
971            // v0.7 / v0.112: `suite` and `case` are contextual too — they
972            // introduce the suite declaration and its cases, but are perfectly
973            // valid commons/context/field names otherwise.
974            //
975            // The tier is single-sourced in `keywords::RESERVED_CONTEXTUAL`:
976            // this arm defers to it rather than hardcoding the token kinds, so
977            // extending that list is enough to admit a new contextual keyword
978            // here. Each of these words lexes only to its own token, so matching
979            // the source text is equivalent to matching the kind.
980            Some(t) if crate::keywords::is_reserved_contextual(self.slice(t.span)) => {
981                self.bump();
982                Ok(Ident {
983                    name: self.slice(t.span).to_string(),
984                    span: t.span,
985                })
986            }
987            Some(t) if is_reserved_keyword(t.kind) => Err(CompileError::new(
988                "bynk.parse.reserved_keyword",
989                t.span,
990                format!(
991                    "expected identifier {ctx}, but `{}` is a reserved keyword",
992                    self.slice(t.span)
993                ),
994            )
995            .with_note("rename the identifier to something that is not a keyword")),
996            Some(t) => Err(CompileError::new(
997                "bynk.parse.expected_token",
998                t.span,
999                format!("expected identifier {ctx}, found {}", t.kind.describe()),
1000            )),
1001            None => Err(CompileError::new(
1002                "bynk.parse.unexpected_eof",
1003                self.eof_span(),
1004                format!("expected identifier {ctx}, found end of file"),
1005            )),
1006        }
1007    }
1008
1009    // -- top level --
1010
1011    /// Consume an optional doc block at the current position, returning the
1012    /// (content, end-of-doc span) pair. Returns None if the next token is not
1013    /// a doc block.
1014    fn take_doc_block(&mut self) -> Option<(String, Span)> {
1015        if self.peek_kind() == Some(TokenKind::DocBlock) {
1016            let t = self.bump().unwrap();
1017            let body = doc_block_content(self.source, t.span);
1018            return Some((body, t.span));
1019        }
1020        None
1021    }
1022
1023    /// Collect all line-comment trivia leading the next declaration plus
1024    /// the optional doc block. Comments may appear both *before* and
1025    /// *between* the doc and the declaration; the spec canonicalises both
1026    /// groups above the doc, so we concatenate them.
1027    ///
1028    /// The doc comes back as a [`DocLead`], which records where it sat among
1029    /// the comments, so an orphan can be kept in place ([`keep_orphan`]).
1030    fn collect_item_lead(&mut self) -> (Vec<Comment>, Option<DocLead>) {
1031        let mut leading = self.take_leading_trivia();
1032        let doc = self.take_doc_block().map(|(content, span)| DocLead {
1033            content,
1034            span,
1035            at: leading.len(),
1036        });
1037        if doc.is_some() {
1038            leading.extend(self.take_leading_trivia());
1039        }
1040        (leading, doc)
1041    }
1042
1043    /// Attach a parsed doc block to a following declaration unless a blank
1044    /// line separates them, in which case the doc is orphaned: a warning, and
1045    /// the block is kept in `leading` where it sat (#1756).
1046    fn finalize_doc(
1047        &mut self,
1048        doc: Option<DocLead>,
1049        next_span: Span,
1050        leading: &mut Vec<Comment>,
1051    ) -> Option<String> {
1052        let doc = doc?;
1053        let doc_span = doc.span;
1054        // A blank line between the doc and the next decl orphans the doc.
1055        if has_blank_line_between(self.source, doc_span.end, next_span.start) {
1056            self.warnings.push(
1057                CompileError::new(
1058                    "bynk.parse.orphan_doc_block",
1059                    doc_span,
1060                    "documentation block is separated from the following declaration by a blank line; it will not be attached",
1061                )
1062                .with_note(
1063                    "remove the blank line to attach the doc to the next declaration, \
1064                     or remove the doc block if it is not meant to document anything",
1065                ),
1066            );
1067            keep_orphan(leading, doc);
1068            return None;
1069        }
1070        Some(doc.content)
1071    }
1072}
1073
1074/// A doc block read ahead of a declaration, before it is known to attach: its
1075/// normalised content, its span, and its index among the leading comments
1076/// collected with it.
1077pub(crate) struct DocLead {
1078    pub(crate) content: String,
1079    pub(crate) span: Span,
1080    pub(crate) at: usize,
1081}
1082
1083/// #1756: keep an orphaned doc block in `leading`, at the place it was written
1084/// among those comments, so the formatter prints it where it was rather than
1085/// dropping it.
1086fn keep_orphan(leading: &mut Vec<Comment>, doc: DocLead) {
1087    let at = doc.at.min(leading.len());
1088    leading.insert(at, Comment::OrphanDoc(doc.content));
1089}
1090
1091/// Parse the body of a lexed double-quoted string literal (the lexeme,
1092/// including surrounding quotes), applying the v0 escape rules.
1093fn parse_string_literal(lexeme: &str, span: Span) -> Result<String, CompileError> {
1094    let bytes = lexeme.as_bytes();
1095    debug_assert!(bytes.first() == Some(&b'"') && bytes.last() == Some(&b'"'));
1096    let inner = &lexeme[1..lexeme.len() - 1];
1097    let mut out = String::with_capacity(inner.len());
1098    let mut chars = inner.chars();
1099    while let Some(c) = chars.next() {
1100        if c == '\\' {
1101            match chars.next() {
1102                Some('n') => out.push('\n'),
1103                Some('t') => out.push('\t'),
1104                Some('"') => out.push('"'),
1105                Some('\\') => out.push('\\'),
1106                other => {
1107                    return Err(CompileError::new(
1108                        "bynk.lex.bad_escape",
1109                        span,
1110                        format!(
1111                            "invalid escape sequence `\\{}` in string literal",
1112                            other.map(|c| c.to_string()).unwrap_or_default()
1113                        ),
1114                    )
1115                    .with_note("supported escapes: \\n \\t \\\" \\\\"));
1116                }
1117            }
1118        } else {
1119            out.push(c);
1120        }
1121    }
1122    Ok(out)
1123}
1124
1125fn is_reserved_keyword(kind: TokenKind) -> bool {
1126    use TokenKind::*;
1127    matches!(
1128        kind,
1129        Commons
1130            | Type
1131            | Fn
1132            | Where
1133            | True
1134            | False
1135            | Int
1136            | String
1137            | Bool
1138            | Let
1139            | If
1140            | Else
1141            | Ok
1142            | Err
1143            | Result
1144            | ValidationError
1145            | Enum
1146            | Match
1147            | Option
1148            | Record
1149            | Self_
1150            | Some
1151            | None
1152            | Is
1153            | Opaque
1154            | Uses
1155            | Context
1156            | Consumes
1157            | Exports
1158            | Transparent
1159            | Agent
1160            | As
1161            | Capability
1162            | Effect
1163            | Do
1164            | Given
1165            | On
1166            | Http
1167            | Provides
1168            | Stub
1169            | Service
1170            | Actor
1171            | By
1172            | Expect
1173            | Suite
1174            | Case
1175            | Float
1176            | Duration
1177            | Instant
1178            | Bytes
1179            | JsonError
1180            | Property
1181            | Adapter
1182            | Binding
1183            | Cron
1184            | Queue
1185            | From
1186            | Protocol
1187            | Invariant
1188            | Implies
1189            | Requires
1190            | Ensures
1191            | Transition
1192    )
1193}
1194
1195/// True when `kind` starts a top-level unit (`commons`/`context`/`adapter`/
1196/// `suite`) or an item within one of their bodies — every keyword any of
1197/// `parse_commons_brace`/`_fragment`, `parse_context_brace`/`_fragment`,
1198/// `parse_test_brace`/`_fragment`, or `parse_adapter_body` dispatches on
1199/// (`declarations.rs`). The single set [`Parser::recover_to_top_item`]'s sync
1200/// scan checks against — finding #27/#30: that scan had drifted from what the
1201/// item loops actually recognise (`Property`, `Actor`, `Event`, `Binding`,
1202/// and the `adapter` unit keyword itself were all missing), so an error
1203/// recovery sync could walk past a real item/unit boundary instead of
1204/// stopping there.
1205fn is_item_start(kind: TokenKind) -> bool {
1206    use TokenKind::*;
1207    matches!(
1208        kind,
1209        // Top-level unit keywords.
1210        Commons | Context | Adapter | Suite
1211        // Body items shared across commons/context/adapter.
1212        | Type | Fn | Messages | Event | Uses
1213        // Context/adapter-only body items.
1214        | Consumes | Exports | Capability | Provides | Service | Agent | Actor
1215        // Adapter-only.
1216        | Binding
1217        // Suite/test-only body items.
1218        | Stub | Case | Property
1219    )
1220}
1221
1222#[cfg(test)]
1223mod tests {
1224    use super::*;
1225    use crate::lexer::tokenize;
1226
1227    fn parse_str(src: &str) -> Result<Commons, Vec<CompileError>> {
1228        let toks = tokenize(src).map_err(|e| vec![e])?;
1229        parse(&toks, src)
1230    }
1231
1232    fn parse_recover_str(src: &str) -> (Option<SourceUnit>, Vec<CompileError>) {
1233        let toks = match tokenize(src) {
1234            Ok(t) => t,
1235            Err(e) => return (None, vec![e]),
1236        };
1237        parse_unit_with_recovery(&toks, src)
1238    }
1239
1240    /// Finding #29/#30: `parse_units_with_recovery` keeps every top-level unit
1241    /// an atomic `commons`+`suite` file declares (v0.113, DECISION S), where
1242    /// `parse_unit_with_recovery` keeps only the first — the defect the IDE's
1243    /// old single-unit parse entry point had (it silently discarded the
1244    /// trailing `suite`).
1245    #[test]
1246    fn parse_units_with_recovery_keeps_every_top_level_unit() {
1247        let src =
1248            "commons m {\n  fn f() -> Int { 1 }\n}\n\nsuite m\n\ncase \"c\" {\n  expect true\n}\n";
1249        let toks = tokenize(src).unwrap();
1250        let (units, errors) = parse_units_with_recovery(&toks, src);
1251        assert!(errors.is_empty(), "{errors:?}");
1252        assert_eq!(units.len(), 2, "expected both units, got {units:?}");
1253        assert!(matches!(units[0], SourceUnit::Commons(_)));
1254        assert!(matches!(units[1], SourceUnit::Suite(_)));
1255
1256        // The singular wrapper still narrows to just the first, unchanged.
1257        let (unit, errors) = parse_unit_with_recovery(&toks, src);
1258        assert!(errors.is_empty(), "{errors:?}");
1259        assert!(matches!(unit, Some(SourceUnit::Commons(_))));
1260    }
1261
1262    /// `parse_unit_with_recovery` always attempts at least one parse, even on
1263    /// empty input — `parse_units_with_recovery`'s loop must preserve that
1264    /// (its `while`-style peek check alone would skip the body entirely and
1265    /// silently return no error), since 16+ existing callers rely on an empty
1266    /// file still producing its usual diagnostic rather than a silent `None`
1267    /// with no error at all.
1268    #[test]
1269    fn empty_input_still_reports_an_error_through_the_plural_entry_point() {
1270        let toks = tokenize("").unwrap();
1271        let (units, errors) = parse_units_with_recovery(&toks, "");
1272        assert!(units.is_empty());
1273        assert!(
1274            !errors.is_empty(),
1275            "an empty file must still produce a diagnostic, not silently no units and no error"
1276        );
1277
1278        let (unit, unit_errors) = parse_unit_with_recovery(&toks, "");
1279        assert!(unit.is_none());
1280        assert_eq!(
1281            errors.len(),
1282            unit_errors.len(),
1283            "the singular wrapper must see the same error(s) as the plural entry point"
1284        );
1285    }
1286
1287    /// Finding #66: `parse_units_with_drain_check`'s `fully_drained` flag is
1288    /// `bynk-fmt`'s signal for whether its comment-loss guard can skip a
1289    /// re-tokenize-and-diff of its own output. It must be `true` for an
1290    /// ordinary file (every comment sits before a declaration/statement or
1291    /// trails one) and `false` the moment a comment sits inside an expression
1292    /// subtree, where `TriviaTable` has no field to attach it to.
1293    #[test]
1294    fn drain_check_reports_expression_interior_comments_as_undrained() {
1295        let ordinary = "commons x {\n-- note\ntype T = Int where Positive\n}\n";
1296        let toks = tokenize(ordinary).unwrap();
1297        let (_, _, drained) = parse_units_with_drain_check(&toks, ordinary).unwrap();
1298        assert!(
1299            drained,
1300            "a declaration-leading comment must be fully drained"
1301        );
1302
1303        let lossy = "commons x {\n  fn f() -> Int {\n    1 + -- note\n    2\n  }\n}\n";
1304        let toks = tokenize(lossy).unwrap();
1305        let (_, _, drained) = parse_units_with_drain_check(&toks, lossy).unwrap();
1306        assert!(
1307            !drained,
1308            "a comment inside a binop expression must be reported as undrained"
1309        );
1310    }
1311
1312    #[test]
1313    fn eof_span_never_splits_a_multibyte_codepoint() {
1314        // An unterminated construct whose final line ends in a non-ASCII char
1315        // (here a `--` comment ending in `→`) once produced an `unexpected_eof`
1316        // span of `len - 1 .. len`, landing on the arrow's last continuation
1317        // byte. Every reported span must sit on char boundaries.
1318        for src in [
1319            "commons x {\n  -- ends with an arrow →",
1320            "agent A {\n  key k: String\n  -- note 🦀",
1321            "commons y {\n  type T = é",
1322        ] {
1323            let (_unit, errors) = parse_recover_str(src);
1324            for e in &errors {
1325                assert!(
1326                    src.is_char_boundary(e.span.start) && src.is_char_boundary(e.span.end),
1327                    "span {:?} splits a codepoint in {src:?}",
1328                    e.span,
1329                );
1330            }
1331        }
1332    }
1333
1334    #[test]
1335    fn reserved_contextual_keywords_readable_in_expression_position() {
1336        // Events track, slice 0 (#939): `expect_ident`'s `RESERVED_CONTEXTUAL`
1337        // exemption (`keywords::RESERVED_CONTEXTUAL`: case/event/messages/on/
1338        // suite) covers *declaring* a binding with one of these names — a
1339        // parameter, a `let` — but the primary-expression parser previously
1340        // only admitted plain `TokenKind::Ident` when *reading one back*.
1341        // Latent since messages/on/case/suite shipped (no fixture happened to
1342        // name a binding after one of them and read it back inside the body);
1343        // surfaced concretely when `event` joined the tier and collided with
1344        // `examples/event-log`'s pre-existing `add(event: Event)` handler.
1345        for kw in ["case", "event", "messages", "on", "suite"] {
1346            let src = format!("commons x\n\nfn f({kw}: Int) -> Int {{\n  {kw}\n}}\n");
1347            let result = parse_str(&src);
1348            assert!(
1349                result.is_ok(),
1350                "a parameter named `{kw}` must be readable in expression position: {:?}",
1351                result.err()
1352            );
1353        }
1354    }
1355
1356    #[test]
1357    fn recovery_skips_garbage_between_decls() {
1358        // Two `type` declarations separated by garbage. Recovery should
1359        // accept both and report one error for the garbage between them.
1360        let src = "commons x {\n\
1361                   type A = Int where NonNegative\n\
1362                   ??? !!!\n\
1363                   type B = String where NonEmpty\n\
1364                   }";
1365        let (unit, errors) = parse_recover_str(src);
1366        let unit = unit.expect("recovery should produce a partial AST");
1367        let SourceUnit::Commons(c) = unit else {
1368            panic!("expected commons")
1369        };
1370        // Both type decls should have been collected despite the garbage.
1371        let names: Vec<_> = c
1372            .items
1373            .iter()
1374            .map(|i| match i {
1375                CommonsItem::Type(t) => t.name.name.clone(),
1376                _ => panic!("expected only types"),
1377            })
1378            .collect();
1379        assert!(
1380            names.contains(&"A".to_string()) && names.contains(&"B".to_string()),
1381            "expected both A and B; got {names:?}",
1382        );
1383        assert!(!errors.is_empty(), "expected at least one parse error");
1384    }
1385
1386    #[test]
1387    fn recovery_handles_bad_first_decl_then_good_second() {
1388        // First decl is malformed (missing `=`); second is well-formed.
1389        let src = "commons x {\n\
1390                   type A Int where NonNegative\n\
1391                   type B = String where NonEmpty\n\
1392                   }";
1393        let (unit, errors) = parse_recover_str(src);
1394        let unit = unit.expect("recovery should produce a partial AST");
1395        let SourceUnit::Commons(c) = unit else {
1396            panic!("expected commons")
1397        };
1398        let names: Vec<_> = c
1399            .items
1400            .iter()
1401            .filter_map(|i| match i {
1402                CommonsItem::Type(t) => Some(t.name.name.clone()),
1403                _ => None,
1404            })
1405            .collect();
1406        assert!(
1407            names.contains(&"B".to_string()),
1408            "B should be parsed after A's failure; got {names:?}"
1409        );
1410        assert!(!errors.is_empty(), "expected at least one parse error");
1411    }
1412
1413    /// Finding #27/#30: an error two levels deep inside `f`'s body (a
1414    /// `match` arm's own block) used to make `recover_to_top_item`'s flat,
1415    /// depth-blind scan stop at the *first* `}` it saw — the arm block's own,
1416    /// not `f`'s. Two more `}` (the match's, then `f`'s) then got consumed one
1417    /// at a time across repeated recovery re-entries, and the outer item loop
1418    /// eventually mistook the commons's *own* closing `}` for having arrived
1419    /// early, returning zero items and a spurious second
1420    /// `bynk.parse.expected_unit_header` error. With brace-depth tracking, `g`
1421    /// is recovered as the sole item and only `f`'s own error is reported.
1422    /// #1663: an item that ends without its body (`fn f() -> Int` then the
1423    /// commons' own `}`) raises its error *on* that `}`, so recovery makes no
1424    /// progress. The no-progress step must not consume the brace that closes
1425    /// the body (brace form): doing so made the body look unclosed, a
1426    /// follow-on `unexpected_eof`.
1427    #[test]
1428    fn recovery_keeps_the_bodys_closing_brace_after_a_bodiless_item() {
1429        let src = "commons m {\n  fn f() -> Int\n}\n";
1430        let (unit, errors) = parse_recover_str(src);
1431        assert!(unit.is_some(), "recovery should produce a partial AST");
1432        let categories: Vec<_> = errors.iter().map(|e| e.category).collect();
1433        assert_eq!(
1434            categories.len(),
1435            1,
1436            "one syntax error, no follow-on: {categories:?}"
1437        );
1438        assert!(
1439            !categories.contains(&"bynk.parse.unexpected_eof"),
1440            "the commons' closing brace must survive: {categories:?}"
1441        );
1442    }
1443
1444    /// #1663: an item keyword illegal at this position (`agent` in a commons)
1445    /// is skipped *with* its name and `{ … }` body, not one token at a time —
1446    /// otherwise the item loop misreads the agent's name as a malformed item,
1447    /// a second, follow-on `expected_item`. The next item still parses.
1448    #[test]
1449    fn recovery_skips_a_rejected_item_whole() {
1450        let src = "commons m\n\nagent Counter {\n  key id: String\n}\n\nfn g() -> Int { 2 }\n";
1451        let recovered = parse_units_recovering(&crate::lexer::tokenize(src).unwrap(), src);
1452        assert_eq!(
1453            recovered.errors.len(),
1454            1,
1455            "one syntax error, no follow-on: {:?}",
1456            recovered
1457                .errors
1458                .iter()
1459                .map(|e| e.category)
1460                .collect::<Vec<_>>()
1461        );
1462        let Some(SourceUnit::Commons(c)) = recovered.units.first() else {
1463            panic!("expected commons")
1464        };
1465        assert!(
1466            c.items.iter().any(|i| matches!(i, CommonsItem::Fn(_))),
1467            "the `fn g` after the skipped agent must still parse"
1468        );
1469        assert_eq!(recovered.broken_decl_names, ["Counter"]);
1470    }
1471
1472    #[test]
1473    fn recovery_skips_a_nested_blocks_own_closing_brace() {
1474        let src = "commons m {\n  \
1475                   fn f() -> Int {\n    \
1476                   match 1 {\n      \
1477                   is 1 -> { let z = }\n      \
1478                   is _ -> 2\n    \
1479                   }\n  \
1480                   }\n  \
1481                   fn g() -> Int { 2 }\n\
1482                   }\n";
1483        let (unit, errors) = parse_recover_str(src);
1484        let unit = unit.expect("recovery should produce a partial AST");
1485        let SourceUnit::Commons(c) = unit else {
1486            panic!("expected commons")
1487        };
1488        let names: Vec<_> = c
1489            .items
1490            .iter()
1491            .filter_map(|i| match i {
1492                CommonsItem::Fn(f) => match &f.name {
1493                    FnName::Free(id) => Some(id.name.clone()),
1494                    _ => None,
1495                },
1496                _ => None,
1497            })
1498            .collect();
1499        assert_eq!(
1500            names,
1501            vec!["g".to_string()],
1502            "g must still be recovered as an item; got {names:?}"
1503        );
1504        assert!(
1505            !errors
1506                .iter()
1507                .any(|e| e.category == "bynk.parse.expected_unit_header"),
1508            "the outer body's own closing brace must not be mistaken for \
1509             end-of-file: {errors:?}"
1510        );
1511    }
1512
1513    #[test]
1514    fn doc_block_attaches_to_type() {
1515        let c =
1516            parse_str("commons x {\n---\nA descriptive doc.\n---\ntype T = Int where Positive\n}")
1517                .unwrap();
1518        let CommonsItem::Type(t) = &c.items[0] else {
1519            panic!()
1520        };
1521        assert!(t.documentation.is_some());
1522        assert!(
1523            t.documentation
1524                .as_ref()
1525                .unwrap()
1526                .contains("A descriptive doc.")
1527        );
1528    }
1529
1530    #[test]
1531    fn interpolated_string_parses_into_parts() {
1532        // v0.43: `"Hi, \(name)!"` splits into chunk / hole / chunk.
1533        let c = parse_str("commons x\n\nfn f(name: String) -> String {\n  \"Hi, \\(name)!\"\n}\n")
1534            .unwrap();
1535        let CommonsItem::Fn(f) = &c.items[0] else {
1536            panic!("expected fn")
1537        };
1538        let ExprKind::InterpStr(parts) = &f.body.tail.kind else {
1539            panic!("expected InterpStr, got {:?}", f.body.tail.kind)
1540        };
1541        assert_eq!(parts.len(), 3);
1542        assert!(matches!(&parts[0], InterpPart::Chunk(s) if s == "Hi, "));
1543        assert!(
1544            matches!(&parts[1], InterpPart::Hole(h) if matches!(&h.kind, ExprKind::Ident(id) if id.name == "name"))
1545        );
1546        assert!(matches!(&parts[2], InterpPart::Chunk(s) if s == "!"));
1547    }
1548
1549    #[test]
1550    fn interpolated_hole_parses_a_full_expression() {
1551        // A hole holds an arbitrary expression, not just an identifier.
1552        let c =
1553            parse_str("commons x\n\nfn f(a: Int, b: Int) -> String {\n  \"sum = \\(a + b)\"\n}\n")
1554                .unwrap();
1555        let CommonsItem::Fn(f) = &c.items[0] else {
1556            panic!("expected fn")
1557        };
1558        let ExprKind::InterpStr(parts) = &f.body.tail.kind else {
1559            panic!("expected InterpStr")
1560        };
1561        assert!(matches!(&parts[1], InterpPart::Hole(h) if matches!(&h.kind, ExprKind::BinOp(..))));
1562    }
1563
1564    #[test]
1565    fn empty_interpolation_hole_is_rejected() {
1566        let errs = parse_str("commons x\n\nfn f() -> String {\n  \"\\()\"\n}\n").unwrap_err();
1567        assert!(
1568            errs.iter()
1569                .any(|e| e.category == "bynk.parse.empty_interpolation"),
1570            "expected empty_interpolation; got {errs:?}"
1571        );
1572    }
1573
1574    #[test]
1575    fn interpolation_hole_lex_error_span_is_rebased() {
1576        // #716: a lex error inside a `\(…)` hole once carried a span relative to
1577        // the hole substring — never rebased by `hole.start` — so it pointed at
1578        // the file's opening bytes and could split a multibyte char, tripping
1579        // the char-boundary invariant. The error must land on the offending
1580        // bytes within the hole and stay on char boundaries.
1581        let cases = [
1582            // `$` is not a valid token; the error should point at it, not byte 0.
1583            "commons x\n\nfn f() -> String {\n  \"a \\($)\"\n}\n",
1584            // Integer overflow — the reported span must cover the literal itself.
1585            "commons x\n\nfn f() -> String {\n  \"n = \\(99999999999999999999)\"\n}\n",
1586            // A multibyte char before the hole means an un-rebased span could
1587            // land inside the `é`; the rebased span must not.
1588            "commons x\n\nfn f() -> String {\n  \"é \\($)\"\n}\n",
1589        ];
1590        for src in cases {
1591            let errs = parse_str(src).unwrap_err();
1592            assert!(!errs.is_empty(), "expected a lex error for {src:?}");
1593            for e in &errs {
1594                assert!(
1595                    src.is_char_boundary(e.span.start) && src.is_char_boundary(e.span.end),
1596                    "span {:?} splits a codepoint in {src:?}",
1597                    e.span,
1598                );
1599                // The error must point inside the interpolation hole, not at the
1600                // header text that precedes it.
1601                let hole_start = src.find("\\(").expect("case has a hole") + 2;
1602                assert!(
1603                    e.span.start >= hole_start,
1604                    "span {:?} precedes the hole (starts at {hole_start}) in {src:?}",
1605                    e.span,
1606                );
1607            }
1608        }
1609    }
1610
1611    #[test]
1612    fn fragment_form_parses() {
1613        let c = parse_str("commons x.y\n\ntype T = Int where NonNegative\n").unwrap();
1614        assert_eq!(c.form, CommonsForm::Fragment);
1615        assert_eq!(c.items.len(), 1);
1616    }
1617
1618    #[test]
1619    fn uses_parses() {
1620        let c = parse_str("commons x\n\nuses other.lib\n").unwrap();
1621        assert_eq!(c.uses.len(), 1);
1622        assert_eq!(c.uses[0].target.joined(), "other.lib");
1623    }
1624
1625    fn parse_unit_str(src: &str) -> Result<SourceUnit, Vec<CompileError>> {
1626        let toks = tokenize(src).map_err(|e| vec![e])?;
1627        parse_unit(&toks, src)
1628    }
1629
1630    #[test]
1631    fn minimal_context_parses() {
1632        let u = parse_unit_str("context commerce.orders {}").unwrap();
1633        let SourceUnit::Context(c) = u else {
1634            panic!("expected context");
1635        };
1636        assert_eq!(c.name.joined(), "commerce.orders");
1637        assert!(c.items.is_empty());
1638    }
1639
1640    #[test]
1641    fn context_consumes_and_exports_parse() {
1642        let src = "context commerce.orders {\n  uses commerce.money\n  consumes commerce.payment\n  exports opaque { OrderId }\n  exports transparent { OrderError }\n  type OrderId = String where Matches(\"ORD-[0-9]+\")\n  type OrderError = enum { CartEmpty, BadInput }\n}";
1643        let u = parse_unit_str(src).unwrap();
1644        let SourceUnit::Context(c) = u else { panic!() };
1645        assert_eq!(c.uses.len(), 1);
1646        assert_eq!(c.consumes.len(), 1);
1647        assert_eq!(c.exports.len(), 2);
1648        assert_eq!(c.exports[0].kind, ExportKind::Type(Visibility::Opaque));
1649        assert_eq!(c.exports[1].kind, ExportKind::Type(Visibility::Transparent));
1650    }
1651
1652    #[test]
1653    fn context_fragment_form_parses() {
1654        let src = "context x.y\n\nuses other.lib\nconsumes other.ctx\nexports opaque { T }\n\ntype T = Int where NonNegative\n";
1655        let u = parse_unit_str(src).unwrap();
1656        let SourceUnit::Context(c) = u else { panic!() };
1657        assert_eq!(c.form, CommonsForm::Fragment);
1658        assert_eq!(c.uses.len(), 1);
1659        assert_eq!(c.consumes.len(), 1);
1660        assert_eq!(c.exports.len(), 1);
1661    }
1662
1663    #[test]
1664    fn opaque_type_parses() {
1665        let c = parse_str("commons x { type T = opaque Int where NonNegative }").unwrap();
1666        let CommonsItem::Type(t) = &c.items[0] else {
1667            panic!()
1668        };
1669        assert!(matches!(t.body, TypeBody::Opaque { .. }));
1670    }
1671
1672    #[test]
1673    fn empty_commons() {
1674        let c = parse_str("commons fitness.units {}").unwrap();
1675        assert_eq!(c.name.joined(), "fitness.units");
1676        assert!(c.items.is_empty());
1677    }
1678
1679    #[test]
1680    fn one_type_decl() {
1681        let c = parse_str("commons x { type Metres = Int where NonNegative }").unwrap();
1682        assert_eq!(c.items.len(), 1);
1683        let CommonsItem::Type(t) = &c.items[0] else {
1684            panic!()
1685        };
1686        assert_eq!(t.name.name, "Metres");
1687        match &t.body {
1688            TypeBody::Refined {
1689                base, refinement, ..
1690            } => {
1691                assert_eq!(*base, BaseType::Int);
1692                assert!(refinement.is_some());
1693            }
1694            _ => panic!("expected refined body"),
1695        }
1696    }
1697
1698    #[test]
1699    fn function_decl() {
1700        let c = parse_str("commons x { fn add(a: Int, b: Int) -> Int { a + b } }").unwrap();
1701        let CommonsItem::Fn(f) = &c.items[0] else {
1702            panic!()
1703        };
1704        assert_eq!(f.name.ident().name, "add");
1705        assert_eq!(f.params.len(), 2);
1706    }
1707
1708    #[test]
1709    fn chained_comparison_is_error() {
1710        let errs = parse_str("commons x { fn f(a: Int, b: Int, c: Int) -> Bool { a < b < c } }")
1711            .unwrap_err();
1712        assert_eq!(errs[0].category, "bynk.parse.non_associative");
1713    }
1714
1715    #[test]
1716    fn chained_equality_is_error() {
1717        let errs = parse_str("commons x { fn f(a: Int, b: Int, c: Int) -> Bool { a == b == c } }")
1718            .unwrap_err();
1719        assert_eq!(errs[0].category, "bynk.parse.non_associative");
1720    }
1721
1722    /// Run `f` on a thread with a generous stack. The depth-guard tests build
1723    /// source that, *without* the guard, overflows — so if the guard ever
1724    /// regressed we want a clean assertion failure, not a `SIGABRT` that takes
1725    /// the whole test binary down. A large stack also absorbs the fat frames a
1726    /// debug build spends per recursion level (production release frames are
1727    /// ~9 KB/level, so `MAX_NESTING_DEPTH = 64` sits well inside a 1 MB stack;
1728    /// a debug frame is several times larger and would overflow libtest's
1729    /// default 2 MB test thread near the limit even though the guard fires).
1730    fn on_big_stack<T: Send + 'static>(f: impl FnOnce() -> T + Send + 'static) -> T {
1731        std::thread::Builder::new()
1732            .stack_size(64 * 1024 * 1024)
1733            .spawn(f)
1734            .unwrap()
1735            .join()
1736            .unwrap()
1737    }
1738
1739    #[test]
1740    fn deeply_nested_parens_are_bounded_not_overflowed() {
1741        // Without a depth guard the parenthesised-expression recursion
1742        // (`parse_primary` -> `parse_expr` -> …) overflows the stack and aborts
1743        // the process (#713). Well past the limit it must instead report a
1744        // bounded-depth diagnostic. The nesting is left open so the guard, not
1745        // a later `)`, is what stops the descent.
1746        let errs = on_big_stack(|| {
1747            let depth = crate::MAX_NESTING_DEPTH + 8;
1748            let src = format!(
1749                "commons x {{ fn f() -> Int {{ {}0{} }} }}",
1750                "(".repeat(depth),
1751                ")".repeat(depth),
1752            );
1753            parse_str(&src).unwrap_err()
1754        });
1755        assert_eq!(errs[0].category, "bynk.parse.nesting_too_deep");
1756    }
1757
1758    #[test]
1759    fn deeply_nested_types_are_bounded_not_overflowed() {
1760        // The type parser self-recurses through generic type arguments
1761        // (`parse_type_ref` -> `parse_type_atom` -> `parse_type_ref`); the same
1762        // guard bounds it (#713). A right-nested `Result[Int, …]` in parameter
1763        // position drives that recursion.
1764        let errs = on_big_stack(|| {
1765            let depth = crate::MAX_NESTING_DEPTH + 8;
1766            let src = format!(
1767                "commons x {{ fn f(x: {}Int{}) -> Int {{ 0 }} }}",
1768                "Result[Int, ".repeat(depth),
1769                "]".repeat(depth),
1770            );
1771            parse_str(&src).unwrap_err()
1772        });
1773        assert_eq!(errs[0].category, "bynk.parse.nesting_too_deep");
1774    }
1775
1776    #[test]
1777    fn deeply_nested_patterns_are_bounded_not_overflowed() {
1778        // Variant patterns are a third self-recursive descent (`parse_pattern`
1779        // -> `parse_pattern_binding` -> `parse_pattern`) that routes through
1780        // neither `parse_expr` nor `parse_type_ref`; without its own guard a
1781        // nested `Ok(Ok(…))` match arm reproduces the #713 crash.
1782        let errs = on_big_stack(|| {
1783            let depth = crate::MAX_NESTING_DEPTH + 8;
1784            let src = format!(
1785                "commons x {{ fn f(n: Int) -> Int {{ match n {{ {}n{} => 0 }} }} }}",
1786                "Ok(".repeat(depth),
1787                ")".repeat(depth),
1788            );
1789            parse_str(&src).unwrap_err()
1790        });
1791        assert_eq!(errs[0].category, "bynk.parse.nesting_too_deep");
1792    }
1793
1794    #[test]
1795    fn nesting_below_the_limit_still_parses() {
1796        // The guard must not reject ordinary well-nested source: a paren-nested
1797        // expression comfortably under the limit still parses cleanly.
1798        let ok = on_big_stack(|| {
1799            let depth = crate::MAX_NESTING_DEPTH - 8;
1800            let src = format!(
1801                "commons x {{ fn f() -> Int {{ {}0{} }} }}",
1802                "(".repeat(depth),
1803                ")".repeat(depth),
1804            );
1805            parse_str(&src).is_ok()
1806        });
1807        assert!(ok, "well-nested source under the limit should parse");
1808    }
1809
1810    #[test]
1811    fn let_statement_parses() {
1812        let c = parse_str("commons x { fn f(n: Int) -> Int { let y = n + 1\n y } }").unwrap();
1813        let CommonsItem::Fn(f) = &c.items[0] else {
1814            panic!()
1815        };
1816        assert_eq!(f.body.statements.len(), 1);
1817        match &f.body.statements[0] {
1818            Statement::Let(l) => {
1819                assert_eq!(l.name.name, "y");
1820                assert!(l.type_annot.is_none());
1821            }
1822            _ => panic!("expected a pure `let` statement"),
1823        }
1824    }
1825
1826    #[test]
1827    fn let_with_annotation() {
1828        let c = parse_str("commons x { fn f(n: Int) -> Int { let y: Int = n\n y } }").unwrap();
1829        let CommonsItem::Fn(f) = &c.items[0] else {
1830            panic!()
1831        };
1832        match &f.body.statements[0] {
1833            Statement::Let(l) => assert!(l.type_annot.is_some()),
1834            _ => panic!("expected a pure `let` statement"),
1835        }
1836    }
1837
1838    #[test]
1839    fn if_else_parses_as_expression() {
1840        let c = parse_str("commons x { fn f(b: Bool) -> Int { if b { 1 } else { 0 } } }").unwrap();
1841        let CommonsItem::Fn(f) = &c.items[0] else {
1842            panic!()
1843        };
1844        assert!(matches!(f.body.tail.kind, ExprKind::If { .. }));
1845    }
1846
1847    #[test]
1848    fn else_if_chain_parses() {
1849        let c = parse_str(
1850            "commons x { fn f(n: Int) -> Int { if n < 0 { -1 } else if n == 0 { 0 } else { 1 } } }",
1851        )
1852        .unwrap();
1853        let CommonsItem::Fn(f) = &c.items[0] else {
1854            panic!()
1855        };
1856        let ExprKind::If { else_block, .. } = &f.body.tail.kind else {
1857            panic!()
1858        };
1859        // The else-branch is a block whose tail is another `If`.
1860        assert!(else_block.statements.is_empty());
1861        assert!(matches!(else_block.tail.kind, ExprKind::If { .. }));
1862    }
1863
1864    #[test]
1865    fn ok_and_err_parse_as_expressions() {
1866        let c = parse_str("commons x { fn f(n: Int) -> Result[Int, String] { Ok(n) } }").unwrap();
1867        let CommonsItem::Fn(f) = &c.items[0] else {
1868            panic!()
1869        };
1870        assert!(matches!(f.body.tail.kind, ExprKind::Ok(_)));
1871
1872        let c =
1873            parse_str("commons x { fn f(n: Int) -> Result[Int, String] { Err(\"x\") } }").unwrap();
1874        let CommonsItem::Fn(f) = &c.items[0] else {
1875            panic!()
1876        };
1877        assert!(matches!(f.body.tail.kind, ExprKind::Err(_)));
1878    }
1879
1880    #[test]
1881    fn question_postfix_parses() {
1882        let c = parse_str(
1883            "commons x { type T = Int where Positive\n fn f(n: Int) -> Result[T, ValidationError] { let x = T.of(n)?\n Ok(x) } }",
1884        )
1885        .unwrap();
1886        let CommonsItem::Fn(f) = &c.items[1] else {
1887            panic!()
1888        };
1889        let Statement::Let(l) = &f.body.statements[0] else {
1890            panic!("expected a pure `let` statement");
1891        };
1892        assert!(matches!(l.value.kind, ExprKind::Question(_)));
1893    }
1894
1895    #[test]
1896    fn constructor_call_parses() {
1897        let c = parse_str(
1898            "commons x { type T = Int where Positive\n fn f(n: Int) -> Result[T, ValidationError] { T.of(n) } }",
1899        )
1900        .unwrap();
1901        let CommonsItem::Fn(f) = &c.items[1] else {
1902            panic!()
1903        };
1904        // v0.2: T.of(n) parses as a MethodCall with receiver Ident("T"); the
1905        // checker reinterprets it as a static call by noticing T is a type.
1906        let ExprKind::MethodCall {
1907            receiver, method, ..
1908        } = &f.body.tail.kind
1909        else {
1910            panic!("expected MethodCall, got {:?}", f.body.tail.kind)
1911        };
1912        let ExprKind::Ident(id) = &receiver.kind else {
1913            panic!("expected receiver Ident");
1914        };
1915        assert_eq!(id.name, "T");
1916        assert_eq!(method.name, "of");
1917    }
1918
1919    #[test]
1920    fn result_type_ref_parses() {
1921        let c = parse_str("commons x { fn f(n: Int) -> Result[Int, String] { Ok(n) } }").unwrap();
1922        let CommonsItem::Fn(f) = &c.items[0] else {
1923            panic!()
1924        };
1925        assert!(matches!(f.return_type, TypeRef::Result(_, _, _)));
1926    }
1927
1928    #[test]
1929    fn result_missing_arg_count_errors() {
1930        let errs = parse_str("commons x { fn f(n: Int) -> Result[Int] { Ok(n) } }").unwrap_err();
1931        assert_eq!(errs[0].category, "bynk.parse.generic_arg_count");
1932    }
1933
1934    #[test]
1935    fn field_access_parses_in_v0_2() {
1936        // v0.2: field access is supported (the type checker validates the
1937        // field exists on the receiver's type). Parser-level acceptance:
1938        let c =
1939            parse_str("commons x { type R = { foo: Int }\n fn f(r: R) -> Int { r.foo } }").unwrap();
1940        let CommonsItem::Fn(f) = &c.items[1] else {
1941            panic!()
1942        };
1943        assert!(matches!(f.body.tail.kind, ExprKind::FieldAccess { .. }));
1944    }
1945
1946    // -- v1.1 trivia attachment --
1947
1948    #[test]
1949    fn leading_line_comment_attaches_to_next_decl() {
1950        let src = "commons x {\n-- explain the type\ntype T = Int where NonNegative\n}";
1951        let c = parse_str(src).unwrap();
1952        let CommonsItem::Type(t) = &c.items[0] else {
1953            panic!()
1954        };
1955        assert_eq!(
1956            t.trivia.leading,
1957            vec![Comment::Line(" explain the type".to_string())]
1958        );
1959        assert!(t.trivia.trailing.is_none());
1960    }
1961
1962    #[test]
1963    fn trailing_line_comment_attaches_to_prev_decl() {
1964        let src = "commons x {\ntype T = Int where NonNegative  -- trailing note\n}";
1965        let c = parse_str(src).unwrap();
1966        let CommonsItem::Type(t) = &c.items[0] else {
1967            panic!()
1968        };
1969        assert!(t.trivia.leading.is_empty());
1970        assert_eq!(t.trivia.trailing.as_deref(), Some(" trailing note"));
1971    }
1972
1973    #[test]
1974    fn grouped_leading_comments_attach_together() {
1975        let src = "commons x {\n-- one\n-- two\n-- three\ntype T = Int where Positive\n}";
1976        let c = parse_str(src).unwrap();
1977        let CommonsItem::Type(t) = &c.items[0] else {
1978            panic!()
1979        };
1980        assert_eq!(
1981            t.trivia.leading,
1982            vec![
1983                Comment::Line(" one".to_string()),
1984                Comment::Line(" two".to_string()),
1985                Comment::Line(" three".to_string())
1986            ],
1987        );
1988    }
1989
1990    #[test]
1991    fn comment_with_doc_block_keeps_both() {
1992        // Both `-- intro` and the doc block should attach to the type decl.
1993        let src = "commons x {\n-- intro\n---\ndocs\n---\ntype T = Int where Positive\n}";
1994        let c = parse_str(src).unwrap();
1995        let CommonsItem::Type(t) = &c.items[0] else {
1996            panic!()
1997        };
1998        assert_eq!(t.trivia.leading, vec![Comment::Line(" intro".to_string())]);
1999        assert_eq!(t.documentation.as_deref(), Some("docs"));
2000    }
2001
2002    #[test]
2003    fn messages_keyword_does_not_collide_with_a_commons_name_segment() {
2004        // `messages` is RESERVED_CONTEXTUAL (like `case`/`on`/`suite`), not a
2005        // hard keyword: `commons app.messages { ... }` — the design's own
2006        // natural naming choice for a bundle commons — must still parse.
2007        // (Caught during slice-1 implementation: a first pass made `messages`
2008        // a plain hard keyword and this exact name broke.)
2009        let src = "commons app.messages {\ntype T = Int where Positive\n}";
2010        let c = parse_str(src).unwrap();
2011        assert_eq!(c.name.joined(), "app.messages");
2012    }
2013
2014    #[test]
2015    fn messages_decl_parses_tag_annotation_and_entries() {
2016        // message-bundles slice 1 (#859): the construct + doc/trivia wiring.
2017        let src = "commons app.messages {\n\
2018                   -- intro\n\
2019                   ---\n\
2020                   docs\n\
2021                   ---\n\
2022                   messages \"en\" @reference {\n\
2023                   \"greeting\" => \"Hello, {name}!\"\n\
2024                   \"farewell\" => \"Bye\"\n\
2025                   } -- trailing\n\
2026                   }";
2027        let c = parse_str(src).unwrap();
2028        let CommonsItem::Messages(m) = &c.items[0] else {
2029            panic!("expected a messages item, got {:?}", c.items[0]);
2030        };
2031        assert_eq!(m.tag, "en");
2032        assert_eq!(m.annotations.len(), 1);
2033        assert_eq!(m.annotations[0].name.name, "reference");
2034        assert!(m.annotations[0].args.is_empty());
2035        assert_eq!(m.entries.len(), 2);
2036        assert_eq!(m.entries[0].code, "greeting");
2037        assert_eq!(m.entries[0].template, "Hello, {name}!");
2038        assert_eq!(m.entries[1].code, "farewell");
2039        assert_eq!(m.entries[1].template, "Bye");
2040        assert_eq!(m.trivia.leading, vec![Comment::Line(" intro".to_string())]);
2041        assert_eq!(m.documentation.as_deref(), Some("docs"));
2042        assert_eq!(m.trivia.trailing.as_deref(), Some(" trailing"));
2043    }
2044
2045    #[test]
2046    fn messages_decl_parses_with_no_annotation_and_no_entries() {
2047        // The parser stays permissive on annotation cardinality (zero-or-more)
2048        // — "exactly one `@reference`" is a checker concern (validate.rs), not
2049        // a parse error.
2050        let src = "commons app.messages {\nmessages \"en\" {\n}\n}";
2051        let c = parse_str(src).unwrap();
2052        let CommonsItem::Messages(m) = &c.items[0] else {
2053            panic!("expected a messages item, got {:?}", c.items[0]);
2054        };
2055        assert_eq!(m.tag, "en");
2056        assert!(m.annotations.is_empty());
2057        assert!(m.entries.is_empty());
2058    }
2059
2060    #[test]
2061    fn messages_decl_parses_syntactically_inside_a_context_too() {
2062        // Commons-only legality is a checker concern (bynk.messages.outside_commons
2063        // in bynk-emit's project validation), not a parser rejection — mirrors
2064        // how `service`/`agent` already parse syntactically inside `adapter`
2065        // bodies for the same reason.
2066        let src = "context app.svc {\nmessages \"en\" @reference {\n\"a\" => \"b\"\n}\n}";
2067        let toks = tokenize(src).unwrap();
2068        let (unit, errors) = parse_unit_with_recovery(&toks, src);
2069        assert!(errors.is_empty(), "unexpected parse errors: {errors:?}");
2070        let Some(SourceUnit::Context(ctx)) = unit else {
2071            panic!("expected a context")
2072        };
2073        let CommonsItem::Messages(m) = &ctx.items[0] else {
2074            panic!("expected a messages item, got {:?}", ctx.items[0]);
2075        };
2076        assert_eq!(m.tag, "en");
2077    }
2078
2079    #[test]
2080    fn comment_before_let_statement_attaches() {
2081        let src = "commons x {\nfn f(n: Int) -> Int {\n-- pick a value\nlet y = n + 1\ny\n}\n}";
2082        let c = parse_str(src).unwrap();
2083        let CommonsItem::Fn(f) = &c.items[0] else {
2084            panic!()
2085        };
2086        let Statement::Let(l) = &f.body.statements[0] else {
2087            panic!()
2088        };
2089        assert_eq!(
2090            l.trivia.leading,
2091            vec![Comment::Line(" pick a value".to_string())]
2092        );
2093    }
2094
2095    #[test]
2096    fn comment_before_tail_attaches_to_block_tail() {
2097        let src = "commons x {\nfn f(n: Int) -> Int {\nlet y = n + 1\n-- result\ny\n}\n}";
2098        let c = parse_str(src).unwrap();
2099        let CommonsItem::Fn(f) = &c.items[0] else {
2100            panic!()
2101        };
2102        assert_eq!(
2103            f.body.tail_leading_comments,
2104            vec![Comment::Line(" result".to_string())],
2105        );
2106    }
2107
2108    /// #637 Gap A: the contextual keywords `on` / `suite` / `case` are lexer
2109    /// tokens but `expect_ident` admits them as identifiers outside their one
2110    /// keyword position, so they are valid record-field and parameter names.
2111    /// The keyword reference now renders them as a distinct "contextual" tier
2112    /// rather than claiming (falsely) that they cannot be used as identifiers.
2113    #[test]
2114    fn contextual_keywords_are_valid_identifiers() {
2115        // Record field names.
2116        let c = parse_str("commons demo {\n  type R = { on: Int, suite: String, case: Bool }\n}")
2117            .expect("`on`/`suite`/`case` are valid field names");
2118        let CommonsItem::Type(_) = &c.items[0] else {
2119            panic!("expected a type decl")
2120        };
2121
2122        // Function parameter names (the other `expect_ident` position).
2123        parse_str("commons demo {\n  fn f(on: Int, case: Int) -> Int { 0 }\n}")
2124            .expect("`on`/`case` are valid parameter names");
2125
2126        // `suite` too, as a field name.
2127        parse_str("commons demo {\n  type R = { suite: Int }\n}")
2128            .expect("`suite` is a valid field name");
2129    }
2130
2131    /// Drift guard: every alphabetic keyword the lexer declares must be
2132    /// classified by `is_reserved_keyword`, or be one of the *contextual*
2133    /// keywords `expect_ident` deliberately admits as identifiers
2134    /// (`on`/`suite`/`case`). Everything else in this codebase that can
2135    /// drift has a guard; this predicate had silently fallen 17 keywords
2136    /// behind, degrading the reserved-keyword diagnostic to the generic
2137    /// expected-token one.
2138    #[test]
2139    fn is_reserved_keyword_covers_every_lexer_keyword() {
2140        let lexer_src = include_str!("lexer.rs");
2141        let mut words = Vec::new();
2142        for line in lexer_src.lines() {
2143            let t = line.trim();
2144            if let Some(rest) = t.strip_prefix("#[token(\"")
2145                && let Some(word) = rest.split('"').next()
2146                && word.chars().next().is_some_and(|c| c.is_ascii_alphabetic())
2147                && word.chars().all(|c| c.is_ascii_alphanumeric() || c == '_')
2148            {
2149                words.push(word.to_string());
2150            }
2151        }
2152        assert!(
2153            words.len() > 30,
2154            "keyword extraction looks broken: only {} words",
2155            words.len()
2156        );
2157        // Contextual keywords double as identifiers (see `expect_ident`); the
2158        // tier is single-sourced in `keywords::RESERVED_CONTEXTUAL`.
2159        use crate::keywords::RESERVED_CONTEXTUAL;
2160        let mut unclassified = Vec::new();
2161        for word in &words {
2162            let tokens = crate::lexer::tokenize(word).expect("keyword lexes");
2163            let kind = tokens.first().expect("keyword yields a token").kind;
2164            if !is_reserved_keyword(kind) && !RESERVED_CONTEXTUAL.contains(&word.as_str()) {
2165                unclassified.push(word.clone());
2166            }
2167        }
2168        assert!(
2169            unclassified.is_empty(),
2170            "keywords missing from is_reserved_keyword (add them, or document \
2171             them as contextual): {unclassified:?}"
2172        );
2173    }
2174
2175    /// Finding #27/#30: pins `is_item_start` to exactly the keyword set
2176    /// `declarations.rs`'s six item loops (`parse_commons_brace`/`_fragment`,
2177    /// `parse_context_brace`/`_fragment`, `parse_test_brace`/`_fragment`) plus
2178    /// `parse_adapter_body` dispatch on, as of this writing — the drift this
2179    /// finding fixed (`Property`, `Actor`, `Event`, `Binding`, and the
2180    /// `adapter` unit keyword itself were all missing from
2181    /// `recover_to_top_item`'s old hand-written sync list, even though every
2182    /// one of them is a real item/unit start). Adding a new item keyword to
2183    /// any of those loops should mean deliberately updating this list too,
2184    /// not silently leaving recovery unable to resync at it.
2185    #[test]
2186    fn is_item_start_matches_the_pinned_keyword_set() {
2187        use TokenKind::*;
2188        let expected_true = [
2189            Commons, Context, Adapter, Suite, Type, Fn, Messages, Event, Uses, Consumes, Exports,
2190            Capability, Provides, Service, Agent, Actor, Binding, Stub, Case, Property,
2191        ];
2192        for kind in expected_true {
2193            assert!(is_item_start(kind), "{kind:?} must be an item start");
2194        }
2195        let expected_false = [
2196            Ident, Plus, Minus, Colon, Dot, Eq, LBrace, RBrace, LParen, RParen, If, Else, Let,
2197            Where, True, False, Match, Is, On, Given,
2198        ];
2199        for kind in expected_false {
2200            assert!(!is_item_start(kind), "{kind:?} must not be an item start");
2201        }
2202    }
2203
2204    /// Fuzz-found (#516): a context-only keyword at item position in a
2205    /// commons errors without consuming the token, and the recovery sync
2206    /// stops at exactly that keyword — without a progress guard the item
2207    /// loop re-reported the same error until memory ran out.
2208    #[test]
2209    fn recovery_makes_progress_on_context_only_keyword_in_commons() {
2210        let src = "commons demo\n\ncapability Logger {\n  fn log(m: String) -> Effect[()]\n}\n";
2211        let tokens = crate::lexer::tokenize(src).unwrap();
2212        let (unit, errors) = parse_unit_with_recovery(&tokens, src);
2213        assert!(unit.is_some(), "the commons header still parses");
2214        assert!(
2215            errors
2216                .iter()
2217                .any(|e| e.category == "bynk.capability.outside_context"),
2218            "the misplaced capability is reported: {errors:?}"
2219        );
2220        // Termination is the real assertion (this used to OOM); a bounded,
2221        // non-repeating error list is the observable proxy.
2222        assert!(errors.len() < 10, "recovery repeated itself: {errors:?}");
2223    }
2224
2225    #[test]
2226    fn trailing_file_comment_becomes_unit_trailing() {
2227        // A comment after the last item but before EOF (fragment form)
2228        // becomes the commons body's trailing comments so the formatter
2229        // can preserve it.
2230        let src = "commons x\n\ntype T = Int where Positive\n-- afterword\n";
2231        let c = parse_str(src).unwrap();
2232        assert_eq!(
2233            c.trailing_comments,
2234            vec![Comment::Line(" afterword".to_string())]
2235        );
2236    }
2237
2238    #[test]
2239    fn trailing_file_comment_after_a_brace_form_commons_is_not_dropped() {
2240        // Regression: the brace form's item loop exits on `RBrace`, never
2241        // reaching the fragment form's end-of-input case that drains the
2242        // trivia table's epilogue — so a comment after the closing `}` was
2243        // silently discarded (and, per `epilogue_is_empty`'s debug_assert,
2244        // would panic a debug build instead of round-tripping through
2245        // `bynk-fmt`).
2246        let src = "commons x {\n  type T = Int where Positive\n}\n-- afterword\n";
2247        let c = parse_str(src).unwrap();
2248        assert_eq!(
2249            c.trailing_comments,
2250            vec![Comment::Line(" afterword".to_string())]
2251        );
2252    }
2253
2254    #[test]
2255    fn trailing_file_comment_after_a_brace_form_context_is_not_dropped() {
2256        // Same regression as the commons case, for `parse_context_brace`.
2257        let src = "context x {\n  type T = Int where Positive\n}\n-- afterword\n";
2258        let SourceUnit::Context(c) = parse_unit_str(src).unwrap() else {
2259            panic!("expected context");
2260        };
2261        assert_eq!(
2262            c.trailing_comments,
2263            vec![Comment::Line(" afterword".to_string())]
2264        );
2265    }
2266
2267    #[test]
2268    fn trailing_file_comment_after_a_brace_form_suite_is_not_dropped() {
2269        // Same regression as the commons case, for `parse_test_brace`.
2270        let src = "suite x {\n  case \"c\" {\n    expect 1 == 1\n  }\n}\n-- afterword\n";
2271        let SourceUnit::Suite(s) = parse_unit_str(src).unwrap() else {
2272            panic!("expected suite");
2273        };
2274        assert_eq!(
2275            s.trailing_comments,
2276            vec![Comment::Line(" afterword".to_string())]
2277        );
2278    }
2279
2280    /// Finding #30: unlike the three regressions just above,
2281    /// `parse_adapter_body`'s brace-closing path never called
2282    /// `take_epilogue` at all (not a regression from a shared pattern — it
2283    /// simply never had the call), so a comment after a brace-form adapter's
2284    /// closing `}` was silently dropped. The fragment form (no braces) was
2285    /// already correct.
2286    #[test]
2287    fn trailing_file_comment_after_a_brace_form_adapter_is_not_dropped() {
2288        let src = "adapter x {\n  binding \"./x.ts\"\n}\n-- afterword\n";
2289        let SourceUnit::Adapter(a) = parse_unit_str(src).unwrap() else {
2290            panic!("expected adapter");
2291        };
2292        assert_eq!(
2293            a.trailing_comments,
2294            vec![Comment::Line(" afterword".to_string())]
2295        );
2296    }
2297
2298    // -- Six-fold unification (review Part 3): the fragment-only ordering
2299    // restrictions declarations.rs's brace/fragment pairs preserve, now that
2300    // they share one function each behind `brace: bool`. None of these had
2301    // any prior test coverage at all. --
2302
2303    /// Fragment-form commons: `uses` must precede every `type`/`fn`.
2304    #[test]
2305    fn commons_fragment_rejects_uses_after_a_decl() {
2306        let src = "commons x\n\ntype T = Int where Positive\nuses bynk.list\n";
2307        let errs = parse_str(src).unwrap_err();
2308        assert!(
2309            errs.iter()
2310                .any(|e| e.category == "bynk.parse.uses_after_decls"),
2311            "{errs:?}"
2312        );
2313    }
2314
2315    /// The same ordering is NOT enforced in brace form — `uses` may appear
2316    /// anywhere in the body.
2317    #[test]
2318    fn commons_brace_allows_uses_after_a_decl() {
2319        let src = "commons x {\n  type T = Int where Positive\n  uses bynk.list\n}\n";
2320        parse_str(src).expect("brace form must not enforce fragment's uses-ordering rule");
2321    }
2322
2323    /// Fragment-form context: `consumes` must precede every `type`/`fn`/etc.
2324    #[test]
2325    fn context_fragment_rejects_consumes_after_a_decl() {
2326        let src = "context x\n\ntype T = Int where Positive\nconsumes bynk\n";
2327        let errs = parse_unit_str(src).unwrap_err();
2328        assert!(
2329            errs.iter()
2330                .any(|e| e.category == "bynk.parse.consumes_after_decls"),
2331            "{errs:?}"
2332        );
2333    }
2334
2335    /// Fragment-form context: `exports` must precede every `type`/`fn`/etc.
2336    #[test]
2337    fn context_fragment_rejects_exports_after_a_decl() {
2338        let src = "context x\n\ntype T = Int where Positive\nexports opaque { T }\n";
2339        let errs = parse_unit_str(src).unwrap_err();
2340        assert!(
2341            errs.iter()
2342                .any(|e| e.category == "bynk.parse.exports_after_decls"),
2343            "{errs:?}"
2344        );
2345    }
2346
2347    /// Brace-form context enforces none of the three orderings.
2348    #[test]
2349    fn context_brace_allows_consumes_and_exports_after_a_decl() {
2350        let src = "context x {\n  type T = Int where Positive\n  consumes bynk\n  exports opaque { T }\n}\n";
2351        parse_unit_str(src)
2352            .expect("brace form must not enforce fragment's consumes/exports-ordering rules");
2353    }
2354
2355    /// Fragment-form suite/test: `uses` must precede every `stub`/`case`/`property`.
2356    #[test]
2357    fn test_fragment_rejects_uses_after_a_decl() {
2358        let src = "suite m\n\ncase \"c\" {\n  expect true\n}\nuses bynk.list\n";
2359        let errs = parse_unit_str(src).unwrap_err();
2360        assert!(
2361            errs.iter()
2362                .any(|e| e.category == "bynk.parse.uses_after_decls"),
2363            "{errs:?}"
2364        );
2365    }
2366
2367    /// Brace-form suite/test allows `uses` anywhere.
2368    #[test]
2369    fn test_brace_allows_uses_after_a_decl() {
2370        let src = "suite m {\n  case \"c\" {\n    expect true\n  }\n  uses bynk.list\n}\n";
2371        parse_unit_str(src).expect("brace form must not enforce fragment's uses-ordering rule");
2372    }
2373
2374    // ---- #636: `if`/`match` condition vs record construction ----
2375
2376    /// Parse `body` as the tail expression of a fn and return its kind.
2377    fn body_tail(body: &str) -> ExprKind {
2378        let src = format!("commons x\n\nfn f() -> Int {{\n  {body}\n}}\n");
2379        let c = parse_str(&src).unwrap_or_else(|e| panic!("parse failed for {body:?}: {e:?}"));
2380        let CommonsItem::Fn(f) = &c.items[0] else {
2381            panic!("expected fn, got {:?}", c.items[0]);
2382        };
2383        f.body.tail.kind.clone()
2384    }
2385
2386    fn body_err(body: &str) -> Vec<CompileError> {
2387        let src = format!("commons x\n\nfn f() -> Int {{\n  {body}\n}}\n");
2388        parse_str(&src).expect_err(&format!("expected a parse error for {body:?}"))
2389    }
2390
2391    #[test]
2392    fn if_condition_ending_in_ident_does_not_swallow_a_single_ident_branch() {
2393        // #636: `ready { result }` shares its shape with a shorthand-field
2394        // record construction. In condition position the branch must win.
2395        for src in [
2396            "if ready { result } else { fallback }",
2397            "if ready { fallback } else { result }",
2398            "if !ready { result } else { fallback }",
2399            "if a == b { result } else { fallback }",
2400            "if a && b { result } else { fallback }",
2401        ] {
2402            let ExprKind::If {
2403                then_block,
2404                else_block,
2405                ..
2406            } = body_tail(src)
2407            else {
2408                panic!("expected If for {src:?}, got {:?}", body_tail(src));
2409            };
2410            // Both branches carry a bare-identifier tail — proof the `{ … }`
2411            // was read as a block, not consumed as a record by the condition.
2412            assert!(
2413                matches!(&then_block.tail.kind, ExprKind::Ident(_)),
2414                "then-branch tail not an ident for {src:?}: {:?}",
2415                then_block.tail.kind,
2416            );
2417            assert!(
2418                matches!(&else_block.tail.kind, ExprKind::Ident(_)),
2419                "else-branch tail not an ident for {src:?}: {:?}",
2420                else_block.tail.kind,
2421            );
2422        }
2423    }
2424
2425    #[test]
2426    fn else_less_if_with_single_ident_branch_parses() {
2427        // The no-`else` reproduction: previously errored `found `}``.
2428        let ExprKind::If { then_block, .. } = body_tail("if ready { result }") else {
2429            panic!("expected If");
2430        };
2431        assert!(matches!(&then_block.tail.kind, ExprKind::Ident(_)));
2432    }
2433
2434    #[test]
2435    fn record_construction_still_parses_in_value_position() {
2436        // The restriction is confined to condition spines — an ordinary value
2437        // position still constructs records, including the shorthand tail form.
2438        assert!(matches!(
2439            body_tail("Point { x }"),
2440            ExprKind::RecordConstruction { .. }
2441        ));
2442        assert!(matches!(
2443            body_tail("Point { x: 1, y: 2 }"),
2444            ExprKind::RecordConstruction { .. }
2445        ));
2446        assert!(matches!(
2447            body_tail("Empty {}"),
2448            ExprKind::RecordConstruction { .. }
2449        ));
2450    }
2451
2452    #[test]
2453    fn parenthesised_record_is_allowed_in_condition_head() {
2454        // A delimiter lifts the restriction: `(ready { result })` constructs a
2455        // record even in condition position (mirrors Rust's paren escape).
2456        let ExprKind::If { cond, .. } =
2457            body_tail("if (ready { result }) { branch } else { other }")
2458        else {
2459            panic!("expected If");
2460        };
2461        let ExprKind::Paren(inner) = &cond.kind else {
2462            panic!("expected a parenthesised condition, got {:?}", cond.kind);
2463        };
2464        assert!(
2465            matches!(&inner.kind, ExprKind::RecordConstruction { .. }),
2466            "parenthesised record in condition head should still construct: {:?}",
2467            inner.kind,
2468        );
2469    }
2470
2471    #[test]
2472    fn record_in_call_arg_within_condition_still_constructs() {
2473        // The restriction is lifted through a call-argument delimiter, so a
2474        // record literal passed to a predicate in the condition still parses.
2475        let ExprKind::If { cond, .. } = body_tail("if check(Point { x: 1 }) { a } else { b }")
2476        else {
2477            panic!("expected If");
2478        };
2479        let ExprKind::Call { args, .. } = &cond.kind else {
2480            panic!("expected Call in condition, got {:?}", cond.kind);
2481        };
2482        assert!(matches!(&args[0].kind, ExprKind::RecordConstruction { .. }));
2483    }
2484
2485    #[test]
2486    fn safe_condition_shapes_are_unaffected() {
2487        // Cases the issue lists as already-safe must stay safe.
2488        assert!(matches!(
2489            body_tail("if ready == true { result } else { fallback }"),
2490            ExprKind::If { .. }
2491        ));
2492        assert!(matches!(
2493            body_tail("if (ready) { result } else { fallback }"),
2494            ExprKind::If { .. }
2495        ));
2496        assert!(matches!(
2497            body_tail("if ready { \"a\" } else { \"b\" }"),
2498            ExprKind::If { .. }
2499        ));
2500    }
2501
2502    #[test]
2503    fn empty_match_reports_its_own_diagnostic() {
2504        // #636: `match result {}` once parsed `result {}` as an empty record,
2505        // masking `bynk.parse.empty_match`. The intended diagnostic is now
2506        // reachable.
2507        let errs = body_err("match result {}");
2508        assert!(
2509            errs.iter().any(|e| e.category == "bynk.parse.empty_match"),
2510            "expected empty_match; got {errs:?}",
2511        );
2512    }
2513
2514    #[test]
2515    fn match_discriminant_ending_in_ident_parses() {
2516        // A `match` over a bare-identifier discriminant reaches its arm list.
2517        assert!(matches!(
2518            body_tail("match ready { x => x }"),
2519            ExprKind::Match { .. }
2520        ));
2521    }
2522
2523    /// #981: an identifier statement immediately followed, on its own line, by
2524    /// a standalone `()` must stay two separate constructs — not merge into a
2525    /// zero-arg call. `status := Paid` / `()` is an Assign statement whose
2526    /// value is the bare identifier `Paid`, then a unit tail; it must never
2527    /// parse as a single `status := Paid()` (a call). The call-parens rule
2528    /// mirrors the v0.20b same-line `[` rule already applied to type
2529    /// arguments: a postfix opener that begins a new line does not continue
2530    /// the previous token.
2531    #[test]
2532    fn identifier_statement_followed_by_unit_tail_does_not_merge_into_a_call() {
2533        let src = "commons c\n\nfn f() -> Int {\n  status := Paid\n  ()\n}\n";
2534        let c = parse_str(src).unwrap_or_else(|e| panic!("parse failed: {e:?}"));
2535        let CommonsItem::Fn(f) = &c.items[0] else {
2536            panic!("expected fn, got {:?}", c.items[0]);
2537        };
2538        assert_eq!(
2539            f.body.statements.len(),
2540            1,
2541            "expected exactly one Assign statement, got {:?}",
2542            f.body.statements
2543        );
2544        let Statement::Assign(a) = &f.body.statements[0] else {
2545            panic!(
2546                "expected an Assign statement, got {:?}",
2547                f.body.statements[0]
2548            );
2549        };
2550        assert!(
2551            matches!(a.value.kind, ExprKind::Ident(_)),
2552            "assign value must stay the bare identifier `Paid`, got {:?}",
2553            a.value.kind
2554        );
2555        assert!(
2556            matches!(f.body.tail.kind, ExprKind::UnitLit),
2557            "the `()` must remain the block's own tail, got {:?}",
2558            f.body.tail.kind
2559        );
2560    }
2561
2562    /// #981: the same same-line rule extends to a method call's parens — a
2563    /// `.method` immediately followed, on its own line, by a standalone `()`
2564    /// must not merge into `.method()`.
2565    #[test]
2566    fn method_reference_followed_by_unit_tail_does_not_merge_into_a_call() {
2567        let src = "commons c\n\nfn f() -> Int {\n  let y = x.field\n  ()\n}\n";
2568        let c = parse_str(src).unwrap_or_else(|e| panic!("parse failed: {e:?}"));
2569        let CommonsItem::Fn(f) = &c.items[0] else {
2570            panic!("expected fn, got {:?}", c.items[0]);
2571        };
2572        let Statement::Let(l) = &f.body.statements[0] else {
2573            panic!("expected a Let statement, got {:?}", f.body.statements[0]);
2574        };
2575        assert!(
2576            matches!(l.value.kind, ExprKind::FieldAccess { .. }),
2577            "let value must stay a field access, got {:?}",
2578            l.value.kind
2579        );
2580        assert!(
2581            matches!(f.body.tail.kind, ExprKind::UnitLit),
2582            "the `()` must remain the block's own tail, got {:?}",
2583            f.body.tail.kind
2584        );
2585    }
2586
2587    #[test]
2588    fn unparenthesised_record_in_condition_head_now_errors() {
2589        // #636 narrowing (matches Rust): a record literal in condition *head*
2590        // position must be parenthesised. Unparenthesised, `Point` reads as the
2591        // discriminant and `{ x: 1 }` as the arm list, whose first "arm" `x: 1`
2592        // is not an arm — so the parse fails. Pinned so the divergence from the
2593        // (still-accepting) tree-sitter grammar is deliberate, not a bug.
2594        assert!(
2595            !body_err("match Point { x: 1 } { p => p }").is_empty(),
2596            "unparenthesised record discriminant should not parse",
2597        );
2598        // Parenthesised, the record is the discriminant and the match parses.
2599        let ExprKind::Match { discriminant, .. } = body_tail("match (Point { x: 1 }) { p => p }")
2600        else {
2601            panic!("expected Match for the parenthesised form");
2602        };
2603        let ExprKind::Paren(inner) = &discriminant.kind else {
2604            panic!(
2605                "expected a parenthesised discriminant, got {:?}",
2606                discriminant.kind
2607            );
2608        };
2609        assert!(matches!(&inner.kind, ExprKind::RecordConstruction { .. }));
2610    }
2611}