Skip to main content

bynk_project/
parse_cache.rs

1//! P8.4 (#1515): a durable path↔`FileId` interning table plus one shared,
2//! content-keyed parse cache — settled by ADR 0413 as the fix for R3.13's
3//! own file level. Replaces `bynk-ide`'s `PROJECT_UNIT_CACHE`
4//! (`bynk-check` cannot depend on `bynk-ide`, so the cache the diagnostics
5//! path (`bynk_check::analysis::analyse_project` → `phase_parse` →
6//! [`crate::discovery::parse_sources`]) actually needs has to live here, one
7//! layer down) and gives `bynk_check::project_model::phase_parse` a `FileId`
8//! that survives across separate analysis calls instead of resetting to
9//! zero on every one ([DECISION B]).
10//!
11//! [DECISION A] (ADR 0413): one cache, `Ast(FileId)`, not a separate
12//! `Tokens(FileId)` cache — neither consumer this slice migrates reads a
13//! token stream independently of the parse it feeds.
14//!
15//! [DECISION B]: the interning table is `path ↔ FileId`
16//! (`HashMap<PathBuf, FileId>`, no `index_vec` crate — matching P8.1's own
17//! "no new indexing infrastructure" posture); the parse cache keys on the
18//! interned `FileId` and invalidates by content equality, mirroring
19//! `PROJECT_UNIT_CACHE`'s own proven scheme exactly.
20//!
21//! [DECISION C]: one slice, one cache. `PROJECT_UNIT_CACHE` is deleted
22//! (`bynk-ide/src/completion.rs`), not left running alongside this one —
23//! two independently-invalidated caches of the same fact is the exact "no
24//! fact in two hand-synced copies" defect this trajectory's phase 1 already
25//! named as a standing invariant.
26//!
27//! [DECISION D] (new — neither the issue nor ADR 0413 examined this):
28//! **`ExprId` must be durably allocated too, from a counter that never
29//! resets, for the same reason `FileId` must be.** `parser::
30//! parse_units_with_warnings_from`'s own doc comment names the hazard this
31//! closes one level up: a multi-file commons merges sibling files' methods
32//! into one `check_record` call, and two independently zero-based files
33//! would collide on the same `ExprId` in the same `expr_types` map
34//! (`collect_unit_methods`, caught live by finding #28). That hazard is
35//! usually avoided by threading one counter across every file **within a
36//! single `phase_parse` call**. Caching a file's parsed `SourceUnit`
37//! *across calls* reopens it one level up: if call 2 serves file A from
38//! cache (keeping its `ExprId`s from call 1's counter position) while
39//! freshly parsing changed file B from a counter that started over at 0,
40//! A's and B's `ExprId`s collide in call 2's own `expr_types` map — the
41//! identical defect class, now triggered by caching rather than by two
42//! files in one call. Fixed the same way `FileId` is: `next_expr_id` lives
43//! in this module's own durable state, advanced only on an actual parse
44//! (a cache hit consumes no new ids, since the cached `SourceUnit`'s own
45//! ids are already fixed), never reset. Global uniqueness across the whole
46//! process trivially implies uniqueness within any one call, so this is a
47//! strict strengthening of the existing guarantee, not a new one.
48//!
49//! [DECISION E] (new): this cache stores the **strict** parse
50//! (`parser::parse_units_with_warnings_from`, `recover_mode: false`) — the
51//! one the build/diagnostics path needs, since a build must never silently
52//! succeed on broken syntax by reading a best-effort recovered AST.
53//! #1710 amends this without weakening it: when the strict parse fails in the
54//! parser, the entry *also* keeps the recovering parse of the same tokens
55//! (`parser::parse_units_recovering_from`, ids from the same durable counter),
56//! read through [`cached_recovery`] by the project path so a file's surviving
57//! declarations are still checked. The strict result is unchanged and still
58//! the `Err` the build reads; the recovered units are for diagnostics only. `bynk
59//! -ide::completion`'s own recovery-tolerant parsing
60//! (`parser::parse_unit_with_recovery`) is a genuinely different parser
61//! configuration, not just a different entry point over the same result —
62//! for syntactically **clean** source the two produce the identical AST
63//! (there is nothing to recover from), so completion reads this cache
64//! directly for the common case; only when the cached/fresh strict result
65//! actually carries errors does completion fall back to its own local,
66//! uncached recovery-parse for that one file (`bynk_ide::completion`'s own
67//! `parse_source_unit`, calling `parser::parse_unit_with_recovery` directly
68//! — not part of this crate) — the rare case (another project file mid-edit
69//! elsewhere, not the buffer under the cursor, which this cache was never
70//! in the path for to begin with) traded for never caching two different
71//! parser configurations' output under one key.
72
73use std::collections::HashMap;
74use std::path::{Path, PathBuf};
75use std::sync::{Arc, LazyLock, Mutex};
76
77use bynk_syntax::ast::SourceUnit;
78use bynk_syntax::error::CompileError;
79use bynk_syntax::span::FileId;
80use bynk_syntax::{lexer, parser};
81
82/// A strict parse's own `Result` shape — the file's units plus any
83/// warnings on success, or the hard errors on failure — each side
84/// `Arc`-wrapped so a cache hit clones cheaply.
85type StrictParseResult =
86    Result<(Arc<Vec<SourceUnit>>, Arc<Vec<CompileError>>), Arc<Vec<CompileError>>>;
87
88/// One file's cached strict-parse result, tagged with the exact content
89/// string it was parsed from — the same "compare the whole content, not a
90/// timestamp" invalidation `PROJECT_UNIT_CACHE` already trusted.
91struct CachedParse {
92    content: Arc<str>,
93    result: StrictParseResult,
94    /// #1710: when the strict parse failed in the *parser* (the source
95    /// lexed), the recovering parse of the same tokens, its ids drawn from the
96    /// same durable counter — for diagnostics only ([DECISION E]).
97    recovered: Option<Arc<parser::Recovered>>,
98}
99
100#[derive(Default)]
101struct ParseCacheState {
102    next_file_id: u32,
103    file_ids: HashMap<PathBuf, FileId>,
104    /// [DECISION D]: durable, never reset — see this module's own doc
105    /// comment. Shares its `u32` space with `bynk_check::project_model`'s
106    /// own first-party `ExprId` reservation
107    /// (`FIRSTPARTY_ID_BASE = 1_000_000_000`, spaced `FIRSTPARTY_ID_BLOCK =
108    /// 1_000_000` apart per unit) — that scheme was sized for a counter that
109    /// reset to 0 every call, so this one growing forever (never resetting)
110    /// could in principle reach it. In practice this needs on the order of a
111    /// billion `ExprId`s consumed over one process's lifetime — many orders
112    /// of magnitude past any real editing session — to become a real
113    /// concern; noted here, not defended against, the same "generous
114    /// headroom, revisit if it ever measurably matters" posture this
115    /// codebase already applies to `PROJECT_UNIT_CACHE_CAP` and to
116    /// `FIRSTPARTY_ID_BLOCK`'s own spacing.
117    next_expr_id: u32,
118    entries: HashMap<FileId, CachedParse>,
119}
120
121/// Cap on distinct cached files, mirroring `PROJECT_UNIT_CACHE_CAP`'s own
122/// reasoning exactly: without one, a long-lived server hopping across many
123/// workspaces accumulates one entry per path ever parsed, and a
124/// renamed/deleted file leaves a dangling entry behind. Past the cap the
125/// parse-result cache clears wholesale (entries repopulate lazily); the
126/// interning table (`file_ids`) is left alone — a stable `FileId` for a
127/// path already seen is cheap to keep and is exactly the durability this
128/// slice exists to provide, so clearing it would defeat the point every
129/// time the cap is hit.
130const PARSE_CACHE_CAP: usize = 4096;
131
132/// Must match `bynk_check::project_model::FIRSTPARTY_ID_BASE` exactly.
133/// `bynk-project` cannot depend on `bynk-check` (the crate graph runs the
134/// other way), so this is a duplicated, keep-in-sync value — guarded by the
135/// `debug_assert!` in `cached_parse_in`, not by the type system. PR
136/// #1520's own bot review (finding #3): a durable counter that quietly
137/// crossed this boundary would silently alias a project file's `ExprId`s
138/// onto a first-party unit's own reserved block — wrong types, wrong
139/// diagnostics, sharing one `expr_types` map, with nothing to point at. In
140/// practice this needs on the order of a billion `ExprId`s consumed over one
141/// process's lifetime to become reachable; the assertion exists so crossing
142/// it is loud, not because it is expected to fire.
143const FIRSTPARTY_ID_BASE: u32 = 1_000_000_000;
144
145static CACHE: LazyLock<Mutex<ParseCacheState>> =
146    LazyLock::new(|| Mutex::new(ParseCacheState::default()));
147
148/// The durable `FileId` for `path` — assigned once, on first use, and
149/// stable for the life of the process from then on, even across content
150/// edits to that same path (a `FileId` is a path identity, not a content
151/// one; [`cached_parse`]'s own content-keyed cache is what tracks edits).
152///
153/// PR #1520's own bot review (finding #1): recovers from a poisoned lock
154/// rather than propagating the poison — a parser panic on one input (a real,
155/// reachable failure mode this codebase already handles at the wasm
156/// boundary, `bynk-wasm/src/lib.rs`'s own `catch_unwind`) must not brick
157/// every later, perfectly valid parse for the life of the process. Safe
158/// specifically because this cache's own invariants — `file_ids` maps a
159/// path to a stable id; `entries[id].content` is the content
160/// `entries[id].result` was parsed from — both hold even mid-panic: a panic
161/// during the parse call in `cached_parse_in` happens strictly before that
162/// entry would have been inserted, so the state a panicked lock holder left
163/// behind is never a half-written entry, only a possibly-stale (never
164/// wrong) one.
165pub fn file_id_for(path: &Path) -> FileId {
166    let mut state = CACHE
167        .lock()
168        .unwrap_or_else(std::sync::PoisonError::into_inner);
169    file_id_for_locked(&mut state, path)
170}
171
172fn file_id_for_locked(state: &mut ParseCacheState, path: &Path) -> FileId {
173    if let Some(&id) = state.file_ids.get(path) {
174        return id;
175    }
176    let id = FileId(state.next_file_id);
177    state.next_file_id += 1;
178    state.file_ids.insert(path.to_path_buf(), id);
179    id
180}
181
182/// The strict parse of `path`'s `content` — [`parser::parse_units_with_warnings_from`],
183/// cached by content equality and keyed on the durable `FileId`
184/// [`file_id_for`] assigns `path`. On a cache hit, no new `ExprId`s are
185/// consumed (the cached units already carry their own, fixed at whenever
186/// they were last actually parsed); on a miss, [DECISION D]'s durable
187/// counter advances by however many the fresh parse used.
188///
189/// The lock is held across the parse itself (not released and re-acquired
190/// the way `PROJECT_UNIT_CACHE` did around its own parse) — [DECISION D]'s
191/// counter must be threaded through the parse call, and releasing the lock
192/// mid-parse would need a pre-reserved `ExprId` block of unknown size.
193/// Serialises concurrent parses process-wide; each individual file's parse
194/// is fast enough (sub-millisecond, ordinarily) that this is a deliberate,
195/// documented trade of a little concurrency for not having to invent a
196/// block-reservation scheme — revisit only if profiling ever shows real
197/// contention. See [`file_id_for`]'s own doc comment for why the lock's
198/// `unwrap_or_else(PoisonError::into_inner)` here is safe.
199pub fn cached_parse(path: &Path, content: &str) -> (FileId, StrictParseResult) {
200    let mut state = CACHE
201        .lock()
202        .unwrap_or_else(std::sync::PoisonError::into_inner);
203    cached_parse_in(&mut state, path, content, PARSE_CACHE_CAP)
204}
205
206/// [`cached_parse`]'s own core, parameterised over the cache state and its
207/// eviction cap so a test can drive a small, locally-owned
208/// `ParseCacheState` to exercise the eviction boundary precisely — PR
209/// #1520's own bot review (finding #2): a test that instead forced the
210/// real, global `CACHE` past its real, 4096-entry cap raced every other
211/// test in this module asserting on that same process-wide `static`
212/// (`cargo test` runs one binary's tests concurrently by default), and cost
213/// thousands of real parses to do it.
214fn cached_parse_in(
215    state: &mut ParseCacheState,
216    path: &Path,
217    content: &str,
218    cap: usize,
219) -> (FileId, StrictParseResult) {
220    let id = file_id_for_locked(state, path);
221    if let Some(entry) = state.entries.get(&id)
222        && &*entry.content == content
223    {
224        return (id, entry.result.clone());
225    }
226
227    let mut recovered = None;
228    let result = match lexer::tokenize_in(content, id) {
229        Ok(tokens) => {
230            match parser::parse_units_with_warnings_from(&tokens, content, &mut state.next_expr_id)
231            {
232                Ok((units, warnings)) => Ok((Arc::new(units), Arc::new(warnings))),
233                Err(errors) => {
234                    recovered = Some(Arc::new(parser::parse_units_recovering_from(
235                        &tokens,
236                        content,
237                        &mut state.next_expr_id,
238                    )));
239                    Err(Arc::new(errors))
240                }
241            }
242        }
243        Err(e) => Err(Arc::new(vec![e])),
244    };
245    debug_assert!(
246        state.next_expr_id < FIRSTPARTY_ID_BASE,
247        "durable ExprId counter ({}) reached the first-party reservation ({FIRSTPARTY_ID_BASE}) \
248         — see this module's own doc comment",
249        state.next_expr_id,
250    );
251
252    if state.entries.len() >= cap && !state.entries.contains_key(&id) {
253        state.entries.clear();
254    }
255    state.entries.insert(
256        id,
257        CachedParse {
258            content: Arc::from(content),
259            result: result.clone(),
260            recovered,
261        },
262    );
263    (id, result)
264}
265
266/// #1710: the recovering parse of `path`'s `content`, when its strict parse
267/// fails in the parser — `None` for clean source, and for source that doesn't
268/// lex. Cached with the strict parse (one entry, one content check), so a file
269/// that stays broken keeps the same `ExprId`s across calls, and the strict
270/// result is still the `Err` the build reads ([DECISION E]).
271pub fn cached_recovery(path: &Path, content: &str) -> Option<Arc<parser::Recovered>> {
272    let mut state = CACHE
273        .lock()
274        .unwrap_or_else(std::sync::PoisonError::into_inner);
275    let (id, _) = cached_parse_in(&mut state, path, content, PARSE_CACHE_CAP);
276    state.entries.get(&id).and_then(|e| e.recovered.clone())
277}
278
279#[cfg(test)]
280mod tests {
281    use super::*;
282
283    /// Each test uses its own unique path (a fresh `FileId`, never reused by
284    /// another test) — `CACHE` is a process-global `static`, so tests running
285    /// in the same binary share it; colliding on a real path would make one
286    /// test's cache entry leak into another's assertions.
287    fn unique_path(name: &str) -> PathBuf {
288        static COUNTER: std::sync::atomic::AtomicU32 = std::sync::atomic::AtomicU32::new(0);
289        let n = COUNTER.fetch_add(1, std::sync::atomic::Ordering::Relaxed);
290        PathBuf::from(format!("/parse-cache-test/{n}-{name}.bynk"))
291    }
292
293    const CLEAN: &str = "commons demo\n\ntype T = { n: Int }\n";
294    const BROKEN: &str = "commons demo\n\ntype T = {\n";
295
296    #[test]
297    fn file_id_is_stable_across_calls_for_the_same_path() {
298        let p = unique_path("stable");
299        let a = file_id_for(&p);
300        let b = file_id_for(&p);
301        assert_eq!(a, b);
302    }
303
304    #[test]
305    fn file_id_is_stable_even_after_the_files_content_changes() {
306        let p = unique_path("edited");
307        let (id_before, _) = cached_parse(&p, CLEAN);
308        let (id_after, _) = cached_parse(&p, "commons demo\n\ntype T = { n: String }\n");
309        assert_eq!(
310            id_before, id_after,
311            "a FileId is a path identity, not a content one"
312        );
313    }
314
315    #[test]
316    fn distinct_paths_get_distinct_file_ids() {
317        let a = file_id_for(&unique_path("a"));
318        let b = file_id_for(&unique_path("b"));
319        assert_ne!(a, b);
320    }
321
322    /// The R2.2 collision #1540 was about: `next_file_id` starts at `0`
323    /// (`ParseCacheState`'s `Default`), so the very first file this counter
324    /// ever interns gets `FileId(0)` — indistinguishable from
325    /// `FileId::default()` unless `FileId`'s own `Default` is `UNKNOWN`. A
326    /// fresh, locally-owned state (not the process-global `CACHE`, which
327    /// other tests may have already advanced) is what makes "first file
328    /// interned" reproducible here.
329    #[test]
330    fn the_first_file_a_fresh_counter_interns_is_not_the_default_file_id() {
331        let mut state = ParseCacheState::default();
332        let id = file_id_for_locked(&mut state, &unique_path("first"));
333        assert_eq!(id, FileId(0));
334        assert_ne!(id, FileId::default());
335    }
336
337    #[test]
338    fn cached_parse_returns_the_same_units_on_a_repeat_call_with_identical_content() {
339        let p = unique_path("repeat");
340        let (id1, r1) = cached_parse(&p, CLEAN);
341        let (id2, r2) = cached_parse(&p, CLEAN);
342        assert_eq!(id1, id2);
343        let (u1, _) = r1.expect("clean source parses");
344        let (u2, _) = r2.expect("clean source parses");
345        assert!(
346            Arc::ptr_eq(&u1, &u2),
347            "a repeat call must be served from cache, not re-parsed"
348        );
349    }
350
351    #[test]
352    fn cached_parse_reparses_when_content_changes() {
353        let p = unique_path("dirty");
354        let (_, r1) = cached_parse(&p, CLEAN);
355        let (_, r2) = cached_parse(&p, "commons demo\n\ntype T = { n: String }\n");
356        let (u1, _) = r1.expect("clean source parses");
357        let (u2, _) = r2.expect("clean source parses");
358        assert!(
359            !Arc::ptr_eq(&u1, &u2),
360            "a content change must trigger a fresh parse"
361        );
362    }
363
364    /// #1710: broken syntax also carries its recovering parse, cached in the
365    /// same entry, so a repeat call returns the very same recovery (the same
366    /// `ExprId`s, stable across project analyses) and the strict result stays
367    /// the `Err`. Clean source has none.
368    #[test]
369    fn cached_recovery_is_cached_beside_the_strict_error() {
370        let p = unique_path("recovery");
371        let r1 = cached_recovery(&p, BROKEN).expect("broken source recovers");
372        let r2 = cached_recovery(&p, BROKEN).expect("broken source recovers");
373        assert!(
374            Arc::ptr_eq(&r1, &r2),
375            "the recovery must be cached, not re-parsed"
376        );
377        assert!(
378            cached_parse(&p, BROKEN).1.is_err(),
379            "the strict result stays the error"
380        );
381
382        let clean = unique_path("recovery-clean");
383        let (_, strict) = cached_parse(&clean, "commons demo\n\nfn one() -> Int { 1 }\n");
384        assert!(strict.is_ok());
385        assert!(cached_recovery(&clean, "commons demo\n\nfn one() -> Int { 1 }\n").is_none());
386    }
387
388    /// #1710: source that doesn't lex has no recovery (there are no tokens to
389    /// recover from); the strict result carries the lexer's error.
390    #[test]
391    fn source_that_does_not_lex_has_no_recovery() {
392        let p = unique_path("unlexable");
393        let src = "commons demo\n\nfn s() -> String { \"unterminated }\n";
394        assert!(cached_parse(&p, src).1.is_err());
395        assert!(cached_recovery(&p, src).is_none());
396    }
397
398    /// [DECISION E]: broken syntax is cached too — as an `Err`, not silently
399    /// dropped — so a caller that needs the real errors (the build path) gets
400    /// them from the cache exactly like a clean parse.
401    #[test]
402    fn cached_parse_caches_the_error_for_broken_syntax() {
403        let p = unique_path("broken");
404        let (_, r1) = cached_parse(&p, BROKEN);
405        let (_, r2) = cached_parse(&p, BROKEN);
406        assert!(r1.is_err());
407        let e1 = r1.unwrap_err();
408        let e2 = r2.unwrap_err();
409        assert!(
410            Arc::ptr_eq(&e1, &e2),
411            "a repeat call on the same broken content must be cached too"
412        );
413    }
414
415    /// [DECISION D]'s own proof: two files, parsed in two separate
416    /// `cached_parse` calls (never a single `phase_parse`-style batch), must
417    /// still receive disjoint `ExprId` ranges — the hazard this decision
418    /// closes is specifically the *cross-call* case a single shared counter
419    /// within one call never had to worry about.
420    #[test]
421    fn expr_ids_never_collide_across_separate_calls() {
422        let p1 = unique_path("exprs-a");
423        let p2 = unique_path("exprs-b");
424        // Parse two identical, nontrivial files (each with a function body
425        // tail expression, so each consumes at least one real ExprId) in two
426        // separate `cached_parse` calls — never batched through one shared
427        // counter the way a single `phase_parse` call already guarantees.
428        let src = "commons demo\n\nfn f(x: Int) -> Int {\n  x + 1\n}\n";
429        let (_, r1) = cached_parse(&p1, src);
430        let (_, r2) = cached_parse(&p2, src);
431        let (u1, _) = r1.expect("parses");
432        let (u2, _) = r2.expect("parses");
433
434        fn fn_tail_expr_id(u: &SourceUnit) -> bynk_syntax::ast::ExprId {
435            let SourceUnit::Commons(c) = u else {
436                panic!("expected commons")
437            };
438            let bynk_syntax::ast::CommonsItem::Fn(f) = &c.items[0] else {
439                panic!("expected fn")
440            };
441            f.body.tail.id
442        }
443        let id1 = fn_tail_expr_id(&u1[0]);
444        let id2 = fn_tail_expr_id(&u2[0]);
445        assert_ne!(
446            id1, id2,
447            "two files parsed in separate cached_parse calls must never share an ExprId"
448        );
449    }
450
451    // -- Staleness fixture (issue #1515's own "Done when": rename, deletion,
452    //    concurrent-edit interleaving — not just a byte-golden pass). --
453
454    /// A rename is, to this cache, an old path that stops being queried and a
455    /// new path that starts being — there is no "move" operation to get
456    /// wrong, but the new path must get its own independent `FileId` and
457    /// parse result, uncontaminated by whatever the old path held.
458    #[test]
459    fn a_renamed_file_gets_its_own_independent_entry() {
460        let old_path = unique_path("renamed-old");
461        let new_path = unique_path("renamed-new");
462        let (old_id, old_result) = cached_parse(&old_path, CLEAN);
463        // The "rename": the same content now arrives under a new path, as if
464        // the old path had moved. Nothing ever queries `old_path` again.
465        let (new_id, new_result) = cached_parse(&new_path, CLEAN);
466
467        assert_ne!(old_id, new_id, "a renamed file is a new path identity");
468        let (old_units, _) = old_result.expect("clean source parses");
469        let (new_units, _) = new_result.expect("clean source parses");
470        assert!(
471            !Arc::ptr_eq(&old_units, &new_units),
472            "the new path must not silently reuse the old path's cache entry"
473        );
474        // The old path's entry is still independently queryable and correct
475        // — a rename doesn't corrupt what's left behind, it just stops being
476        // read.
477        let (old_id_again, _) = cached_parse(&old_path, CLEAN);
478        assert_eq!(old_id, old_id_again);
479    }
480
481    /// A deleted file simply stops being queried — nothing in this cache's
482    /// own API models deletion explicitly (discovery, one layer up, just
483    /// stops listing the path). The property worth proving is that an
484    /// orphaned entry cannot corrupt a *different* path's own result, and
485    /// that the cap-driven eviction `cached_parse` already performs (mirrors
486    /// `PROJECT_UNIT_CACHE_CAP`'s own precedent) recovers cleanly — a path
487    /// evicted and later re-queried reparses correctly rather than serving
488    /// stale or corrupted state.
489    ///
490    /// Drives a small, locally-owned `ParseCacheState` through
491    /// `cached_parse_in` directly (PR #1520's own bot review, finding #2)
492    /// rather than forcing the real, global `CACHE` past its real,
493    /// 4096-entry cap — that would race every other test in this module
494    /// asserting on the same process-wide `static` (`cargo test` runs one
495    /// binary's tests concurrently by default) and cost thousands of real
496    /// parses to do it. A local state with a cap of 2 exercises the exact
497    /// same eviction branch precisely, cheaply, and in isolation.
498    #[test]
499    fn an_evicted_entry_reparses_correctly_rather_than_serving_stale_state() {
500        let mut state = ParseCacheState::default();
501        let cap = 2;
502        let survivor = PathBuf::from("/parse-cache-test/evict-survivor.bynk");
503        let (_, before) = cached_parse_in(&mut state, &survivor, CLEAN, cap);
504        let (survivor_units_before, _) = before.expect("clean source parses");
505
506        // Push past the cap with filler paths — the last one triggers the
507        // "past the cap, clear and let entries repopulate lazily" eviction,
508        // clearing `survivor`'s own entry along with everything else.
509        for i in 0..cap {
510            let p = PathBuf::from(format!("/parse-cache-test/evict-filler-{i}.bynk"));
511            let _ = cached_parse_in(&mut state, &p, CLEAN, cap);
512        }
513
514        let (_, after) = cached_parse_in(&mut state, &survivor, CLEAN, cap);
515        let (survivor_units_after, _) = after.expect("clean source parses");
516        // Correctness, not identity: post-eviction the entry is necessarily
517        // re-parsed (a fresh `Arc`), but it must still be the *same* parse of
518        // the *same* content — not stale, not corrupted by whichever filler
519        // entry happened to occupy the cap-cleared map.
520        assert_eq!(
521            format!("{survivor_units_before:?}"),
522            format!("{survivor_units_after:?}"),
523            "a re-parsed-after-eviction file must produce the same result as before eviction"
524        );
525    }
526
527    /// Concurrent edits: many threads racing `cached_parse` calls against the
528    /// *same* path with *different* content must never panic or corrupt the
529    /// cache (`PROJECT_UNIT_CACHE`'s own `Mutex`-protected precedent, kept
530    /// here) — and once every writer has finished, the cache must correctly
531    /// reflect whichever content a query actually asks for next, not some
532    /// torn mix of two threads' writes.
533    #[test]
534    fn concurrent_edits_to_the_same_path_never_corrupt_the_cache() {
535        let path = Arc::new(unique_path("concurrent"));
536        let variants: Vec<String> = (0..8)
537            .map(|i| format!("commons demo\n\ntype T{i} = {{ n: Int }}\n"))
538            .collect();
539
540        std::thread::scope(|scope| {
541            for v in &variants {
542                let path = Arc::clone(&path);
543                scope.spawn(move || {
544                    for _ in 0..20 {
545                        let (_, result) = cached_parse(&path, v);
546                        assert!(result.is_ok(), "every variant here is syntactically valid");
547                    }
548                });
549            }
550        });
551
552        // After the race, a fresh query with a known, distinct piece of
553        // content must reparse and return exactly that content's own result
554        // — proving the cache settled into a consistent state, not a torn one
555        // that panics or returns nonsense on the next access.
556        let (_, tail) = cached_parse(&path, CLEAN);
557        let (units, _) = tail.expect("clean source parses");
558        assert_eq!(units.len(), 1);
559    }
560}