bynk_project/parse_cache.rs
1//! P8.4 (#1515): a durable path↔`FileId` interning table plus one shared,
2//! content-keyed parse cache — settled by ADR 0413 as the fix for R3.13's
3//! own file level. Replaces `bynk-ide`'s `PROJECT_UNIT_CACHE`
4//! (`bynk-check` cannot depend on `bynk-ide`, so the cache the diagnostics
5//! path (`bynk_check::analysis::analyse_project` → `phase_parse` →
6//! [`crate::discovery::parse_sources`]) actually needs has to live here, one
7//! layer down) and gives `bynk_check::project_model::phase_parse` a `FileId`
8//! that survives across separate analysis calls instead of resetting to
9//! zero on every one ([DECISION B]).
10//!
11//! [DECISION A] (ADR 0413): one cache, `Ast(FileId)`, not a separate
12//! `Tokens(FileId)` cache — neither consumer this slice migrates reads a
13//! token stream independently of the parse it feeds.
14//!
15//! [DECISION B]: the interning table is `path ↔ FileId`
16//! (`HashMap<PathBuf, FileId>`, no `index_vec` crate — matching P8.1's own
17//! "no new indexing infrastructure" posture); the parse cache keys on the
18//! interned `FileId` and invalidates by content equality, mirroring
19//! `PROJECT_UNIT_CACHE`'s own proven scheme exactly.
20//!
21//! [DECISION C]: one slice, one cache. `PROJECT_UNIT_CACHE` is deleted
22//! (`bynk-ide/src/completion.rs`), not left running alongside this one —
23//! two independently-invalidated caches of the same fact is the exact "no
24//! fact in two hand-synced copies" defect this trajectory's phase 1 already
25//! named as a standing invariant.
26//!
27//! [DECISION D] (new — neither the issue nor ADR 0413 examined this):
28//! **`ExprId` must be durably allocated too, from a counter that never
29//! resets, for the same reason `FileId` must be.** `parser::
30//! parse_units_with_warnings_from`'s own doc comment names the hazard this
31//! closes one level up: a multi-file commons merges sibling files' methods
32//! into one `check_record` call, and two independently zero-based files
33//! would collide on the same `ExprId` in the same `expr_types` map
34//! (`collect_unit_methods`, caught live by finding #28). That hazard is
35//! usually avoided by threading one counter across every file **within a
36//! single `phase_parse` call**. Caching a file's parsed `SourceUnit`
37//! *across calls* reopens it one level up: if call 2 serves file A from
38//! cache (keeping its `ExprId`s from call 1's counter position) while
39//! freshly parsing changed file B from a counter that started over at 0,
40//! A's and B's `ExprId`s collide in call 2's own `expr_types` map — the
41//! identical defect class, now triggered by caching rather than by two
42//! files in one call. Fixed the same way `FileId` is: `next_expr_id` lives
43//! in this module's own durable state, advanced only on an actual parse
44//! (a cache hit consumes no new ids, since the cached `SourceUnit`'s own
45//! ids are already fixed), never reset. Global uniqueness across the whole
46//! process trivially implies uniqueness within any one call, so this is a
47//! strict strengthening of the existing guarantee, not a new one.
48//!
49//! [DECISION E] (new): this cache stores the **strict** parse
50//! (`parser::parse_units_with_warnings_from`, `recover_mode: false`) — the
51//! one the build/diagnostics path needs, since a build must never silently
52//! succeed on broken syntax by reading a best-effort recovered AST.
53//! #1710 amends this without weakening it: when the strict parse fails in the
54//! parser, the entry *also* keeps the recovering parse of the same tokens
55//! (`parser::parse_units_recovering_from`, ids from the same durable counter),
56//! read through [`cached_recovery`] by the project path so a file's surviving
57//! declarations are still checked. The strict result is unchanged and still
58//! the `Err` the build reads; the recovered units are for diagnostics only. `bynk
59//! -ide::completion`'s own recovery-tolerant parsing
60//! (`parser::parse_unit_with_recovery`) is a genuinely different parser
61//! configuration, not just a different entry point over the same result —
62//! for syntactically **clean** source the two produce the identical AST
63//! (there is nothing to recover from), so completion reads this cache
64//! directly for the common case; only when the cached/fresh strict result
65//! actually carries errors does completion fall back to its own local,
66//! uncached recovery-parse for that one file (`bynk_ide::completion`'s own
67//! `parse_source_unit`, calling `parser::parse_unit_with_recovery` directly
68//! — not part of this crate) — the rare case (another project file mid-edit
69//! elsewhere, not the buffer under the cursor, which this cache was never
70//! in the path for to begin with) traded for never caching two different
71//! parser configurations' output under one key.
72
73use std::collections::HashMap;
74use std::path::{Path, PathBuf};
75use std::sync::{Arc, LazyLock, Mutex};
76
77use bynk_syntax::ast::SourceUnit;
78use bynk_syntax::error::CompileError;
79use bynk_syntax::span::FileId;
80use bynk_syntax::{lexer, parser};
81
82/// A strict parse's own `Result` shape — the file's units plus any
83/// warnings on success, or the hard errors on failure — each side
84/// `Arc`-wrapped so a cache hit clones cheaply.
85type StrictParseResult =
86 Result<(Arc<Vec<SourceUnit>>, Arc<Vec<CompileError>>), Arc<Vec<CompileError>>>;
87
88/// One file's cached strict-parse result, tagged with the exact content
89/// string it was parsed from — the same "compare the whole content, not a
90/// timestamp" invalidation `PROJECT_UNIT_CACHE` already trusted.
91struct CachedParse {
92 content: Arc<str>,
93 result: StrictParseResult,
94 /// #1710: when the strict parse failed in the *parser* (the source
95 /// lexed), the recovering parse of the same tokens, its ids drawn from the
96 /// same durable counter — for diagnostics only ([DECISION E]).
97 recovered: Option<Arc<parser::Recovered>>,
98}
99
100#[derive(Default)]
101struct ParseCacheState {
102 next_file_id: u32,
103 file_ids: HashMap<PathBuf, FileId>,
104 /// [DECISION D]: durable, never reset — see this module's own doc
105 /// comment. Shares its `u32` space with `bynk_check::project_model`'s
106 /// own first-party `ExprId` reservation
107 /// (`FIRSTPARTY_ID_BASE = 1_000_000_000`, spaced `FIRSTPARTY_ID_BLOCK =
108 /// 1_000_000` apart per unit) — that scheme was sized for a counter that
109 /// reset to 0 every call, so this one growing forever (never resetting)
110 /// could in principle reach it. In practice this needs on the order of a
111 /// billion `ExprId`s consumed over one process's lifetime — many orders
112 /// of magnitude past any real editing session — to become a real
113 /// concern; noted here, not defended against, the same "generous
114 /// headroom, revisit if it ever measurably matters" posture this
115 /// codebase already applies to `PROJECT_UNIT_CACHE_CAP` and to
116 /// `FIRSTPARTY_ID_BLOCK`'s own spacing.
117 next_expr_id: u32,
118 entries: HashMap<FileId, CachedParse>,
119}
120
121/// Cap on distinct cached files, mirroring `PROJECT_UNIT_CACHE_CAP`'s own
122/// reasoning exactly: without one, a long-lived server hopping across many
123/// workspaces accumulates one entry per path ever parsed, and a
124/// renamed/deleted file leaves a dangling entry behind. Past the cap the
125/// parse-result cache clears wholesale (entries repopulate lazily); the
126/// interning table (`file_ids`) is left alone — a stable `FileId` for a
127/// path already seen is cheap to keep and is exactly the durability this
128/// slice exists to provide, so clearing it would defeat the point every
129/// time the cap is hit.
130const PARSE_CACHE_CAP: usize = 4096;
131
132/// Must match `bynk_check::project_model::FIRSTPARTY_ID_BASE` exactly.
133/// `bynk-project` cannot depend on `bynk-check` (the crate graph runs the
134/// other way), so this is a duplicated, keep-in-sync value — guarded by the
135/// `debug_assert!` in `cached_parse_in`, not by the type system. PR
136/// #1520's own bot review (finding #3): a durable counter that quietly
137/// crossed this boundary would silently alias a project file's `ExprId`s
138/// onto a first-party unit's own reserved block — wrong types, wrong
139/// diagnostics, sharing one `expr_types` map, with nothing to point at. In
140/// practice this needs on the order of a billion `ExprId`s consumed over one
141/// process's lifetime to become reachable; the assertion exists so crossing
142/// it is loud, not because it is expected to fire.
143const FIRSTPARTY_ID_BASE: u32 = 1_000_000_000;
144
145static CACHE: LazyLock<Mutex<ParseCacheState>> =
146 LazyLock::new(|| Mutex::new(ParseCacheState::default()));
147
148/// The durable `FileId` for `path` — assigned once, on first use, and
149/// stable for the life of the process from then on, even across content
150/// edits to that same path (a `FileId` is a path identity, not a content
151/// one; [`cached_parse`]'s own content-keyed cache is what tracks edits).
152///
153/// PR #1520's own bot review (finding #1): recovers from a poisoned lock
154/// rather than propagating the poison — a parser panic on one input (a real,
155/// reachable failure mode this codebase already handles at the wasm
156/// boundary, `bynk-wasm/src/lib.rs`'s own `catch_unwind`) must not brick
157/// every later, perfectly valid parse for the life of the process. Safe
158/// specifically because this cache's own invariants — `file_ids` maps a
159/// path to a stable id; `entries[id].content` is the content
160/// `entries[id].result` was parsed from — both hold even mid-panic: a panic
161/// during the parse call in `cached_parse_in` happens strictly before that
162/// entry would have been inserted, so the state a panicked lock holder left
163/// behind is never a half-written entry, only a possibly-stale (never
164/// wrong) one.
165pub fn file_id_for(path: &Path) -> FileId {
166 let mut state = CACHE
167 .lock()
168 .unwrap_or_else(std::sync::PoisonError::into_inner);
169 file_id_for_locked(&mut state, path)
170}
171
172fn file_id_for_locked(state: &mut ParseCacheState, path: &Path) -> FileId {
173 if let Some(&id) = state.file_ids.get(path) {
174 return id;
175 }
176 let id = FileId(state.next_file_id);
177 state.next_file_id += 1;
178 state.file_ids.insert(path.to_path_buf(), id);
179 id
180}
181
182/// The strict parse of `path`'s `content` — [`parser::parse_units_with_warnings_from`],
183/// cached by content equality and keyed on the durable `FileId`
184/// [`file_id_for`] assigns `path`. On a cache hit, no new `ExprId`s are
185/// consumed (the cached units already carry their own, fixed at whenever
186/// they were last actually parsed); on a miss, [DECISION D]'s durable
187/// counter advances by however many the fresh parse used.
188///
189/// The lock is held across the parse itself (not released and re-acquired
190/// the way `PROJECT_UNIT_CACHE` did around its own parse) — [DECISION D]'s
191/// counter must be threaded through the parse call, and releasing the lock
192/// mid-parse would need a pre-reserved `ExprId` block of unknown size.
193/// Serialises concurrent parses process-wide; each individual file's parse
194/// is fast enough (sub-millisecond, ordinarily) that this is a deliberate,
195/// documented trade of a little concurrency for not having to invent a
196/// block-reservation scheme — revisit only if profiling ever shows real
197/// contention. See [`file_id_for`]'s own doc comment for why the lock's
198/// `unwrap_or_else(PoisonError::into_inner)` here is safe.
199pub fn cached_parse(path: &Path, content: &str) -> (FileId, StrictParseResult) {
200 let mut state = CACHE
201 .lock()
202 .unwrap_or_else(std::sync::PoisonError::into_inner);
203 cached_parse_in(&mut state, path, content, PARSE_CACHE_CAP)
204}
205
206/// [`cached_parse`]'s own core, parameterised over the cache state and its
207/// eviction cap so a test can drive a small, locally-owned
208/// `ParseCacheState` to exercise the eviction boundary precisely — PR
209/// #1520's own bot review (finding #2): a test that instead forced the
210/// real, global `CACHE` past its real, 4096-entry cap raced every other
211/// test in this module asserting on that same process-wide `static`
212/// (`cargo test` runs one binary's tests concurrently by default), and cost
213/// thousands of real parses to do it.
214fn cached_parse_in(
215 state: &mut ParseCacheState,
216 path: &Path,
217 content: &str,
218 cap: usize,
219) -> (FileId, StrictParseResult) {
220 let id = file_id_for_locked(state, path);
221 if let Some(entry) = state.entries.get(&id)
222 && &*entry.content == content
223 {
224 return (id, entry.result.clone());
225 }
226
227 let mut recovered = None;
228 let result = match lexer::tokenize_in(content, id) {
229 Ok(tokens) => {
230 match parser::parse_units_with_warnings_from(&tokens, content, &mut state.next_expr_id)
231 {
232 Ok((units, warnings)) => Ok((Arc::new(units), Arc::new(warnings))),
233 Err(errors) => {
234 recovered = Some(Arc::new(parser::parse_units_recovering_from(
235 &tokens,
236 content,
237 &mut state.next_expr_id,
238 )));
239 Err(Arc::new(errors))
240 }
241 }
242 }
243 Err(e) => Err(Arc::new(vec![e])),
244 };
245 debug_assert!(
246 state.next_expr_id < FIRSTPARTY_ID_BASE,
247 "durable ExprId counter ({}) reached the first-party reservation ({FIRSTPARTY_ID_BASE}) \
248 — see this module's own doc comment",
249 state.next_expr_id,
250 );
251
252 if state.entries.len() >= cap && !state.entries.contains_key(&id) {
253 state.entries.clear();
254 }
255 state.entries.insert(
256 id,
257 CachedParse {
258 content: Arc::from(content),
259 result: result.clone(),
260 recovered,
261 },
262 );
263 (id, result)
264}
265
266/// #1710: the recovering parse of `path`'s `content`, when its strict parse
267/// fails in the parser — `None` for clean source, and for source that doesn't
268/// lex. Cached with the strict parse (one entry, one content check), so a file
269/// that stays broken keeps the same `ExprId`s across calls, and the strict
270/// result is still the `Err` the build reads ([DECISION E]).
271pub fn cached_recovery(path: &Path, content: &str) -> Option<Arc<parser::Recovered>> {
272 let mut state = CACHE
273 .lock()
274 .unwrap_or_else(std::sync::PoisonError::into_inner);
275 let (id, _) = cached_parse_in(&mut state, path, content, PARSE_CACHE_CAP);
276 state.entries.get(&id).and_then(|e| e.recovered.clone())
277}
278
279#[cfg(test)]
280mod tests {
281 use super::*;
282
283 /// Each test uses its own unique path (a fresh `FileId`, never reused by
284 /// another test) — `CACHE` is a process-global `static`, so tests running
285 /// in the same binary share it; colliding on a real path would make one
286 /// test's cache entry leak into another's assertions.
287 fn unique_path(name: &str) -> PathBuf {
288 static COUNTER: std::sync::atomic::AtomicU32 = std::sync::atomic::AtomicU32::new(0);
289 let n = COUNTER.fetch_add(1, std::sync::atomic::Ordering::Relaxed);
290 PathBuf::from(format!("/parse-cache-test/{n}-{name}.bynk"))
291 }
292
293 const CLEAN: &str = "commons demo\n\ntype T = { n: Int }\n";
294 const BROKEN: &str = "commons demo\n\ntype T = {\n";
295
296 #[test]
297 fn file_id_is_stable_across_calls_for_the_same_path() {
298 let p = unique_path("stable");
299 let a = file_id_for(&p);
300 let b = file_id_for(&p);
301 assert_eq!(a, b);
302 }
303
304 #[test]
305 fn file_id_is_stable_even_after_the_files_content_changes() {
306 let p = unique_path("edited");
307 let (id_before, _) = cached_parse(&p, CLEAN);
308 let (id_after, _) = cached_parse(&p, "commons demo\n\ntype T = { n: String }\n");
309 assert_eq!(
310 id_before, id_after,
311 "a FileId is a path identity, not a content one"
312 );
313 }
314
315 #[test]
316 fn distinct_paths_get_distinct_file_ids() {
317 let a = file_id_for(&unique_path("a"));
318 let b = file_id_for(&unique_path("b"));
319 assert_ne!(a, b);
320 }
321
322 /// The R2.2 collision #1540 was about: `next_file_id` starts at `0`
323 /// (`ParseCacheState`'s `Default`), so the very first file this counter
324 /// ever interns gets `FileId(0)` — indistinguishable from
325 /// `FileId::default()` unless `FileId`'s own `Default` is `UNKNOWN`. A
326 /// fresh, locally-owned state (not the process-global `CACHE`, which
327 /// other tests may have already advanced) is what makes "first file
328 /// interned" reproducible here.
329 #[test]
330 fn the_first_file_a_fresh_counter_interns_is_not_the_default_file_id() {
331 let mut state = ParseCacheState::default();
332 let id = file_id_for_locked(&mut state, &unique_path("first"));
333 assert_eq!(id, FileId(0));
334 assert_ne!(id, FileId::default());
335 }
336
337 #[test]
338 fn cached_parse_returns_the_same_units_on_a_repeat_call_with_identical_content() {
339 let p = unique_path("repeat");
340 let (id1, r1) = cached_parse(&p, CLEAN);
341 let (id2, r2) = cached_parse(&p, CLEAN);
342 assert_eq!(id1, id2);
343 let (u1, _) = r1.expect("clean source parses");
344 let (u2, _) = r2.expect("clean source parses");
345 assert!(
346 Arc::ptr_eq(&u1, &u2),
347 "a repeat call must be served from cache, not re-parsed"
348 );
349 }
350
351 #[test]
352 fn cached_parse_reparses_when_content_changes() {
353 let p = unique_path("dirty");
354 let (_, r1) = cached_parse(&p, CLEAN);
355 let (_, r2) = cached_parse(&p, "commons demo\n\ntype T = { n: String }\n");
356 let (u1, _) = r1.expect("clean source parses");
357 let (u2, _) = r2.expect("clean source parses");
358 assert!(
359 !Arc::ptr_eq(&u1, &u2),
360 "a content change must trigger a fresh parse"
361 );
362 }
363
364 /// #1710: broken syntax also carries its recovering parse, cached in the
365 /// same entry, so a repeat call returns the very same recovery (the same
366 /// `ExprId`s, stable across project analyses) and the strict result stays
367 /// the `Err`. Clean source has none.
368 #[test]
369 fn cached_recovery_is_cached_beside_the_strict_error() {
370 let p = unique_path("recovery");
371 let r1 = cached_recovery(&p, BROKEN).expect("broken source recovers");
372 let r2 = cached_recovery(&p, BROKEN).expect("broken source recovers");
373 assert!(
374 Arc::ptr_eq(&r1, &r2),
375 "the recovery must be cached, not re-parsed"
376 );
377 assert!(
378 cached_parse(&p, BROKEN).1.is_err(),
379 "the strict result stays the error"
380 );
381
382 let clean = unique_path("recovery-clean");
383 let (_, strict) = cached_parse(&clean, "commons demo\n\nfn one() -> Int { 1 }\n");
384 assert!(strict.is_ok());
385 assert!(cached_recovery(&clean, "commons demo\n\nfn one() -> Int { 1 }\n").is_none());
386 }
387
388 /// #1710: source that doesn't lex has no recovery (there are no tokens to
389 /// recover from); the strict result carries the lexer's error.
390 #[test]
391 fn source_that_does_not_lex_has_no_recovery() {
392 let p = unique_path("unlexable");
393 let src = "commons demo\n\nfn s() -> String { \"unterminated }\n";
394 assert!(cached_parse(&p, src).1.is_err());
395 assert!(cached_recovery(&p, src).is_none());
396 }
397
398 /// [DECISION E]: broken syntax is cached too — as an `Err`, not silently
399 /// dropped — so a caller that needs the real errors (the build path) gets
400 /// them from the cache exactly like a clean parse.
401 #[test]
402 fn cached_parse_caches_the_error_for_broken_syntax() {
403 let p = unique_path("broken");
404 let (_, r1) = cached_parse(&p, BROKEN);
405 let (_, r2) = cached_parse(&p, BROKEN);
406 assert!(r1.is_err());
407 let e1 = r1.unwrap_err();
408 let e2 = r2.unwrap_err();
409 assert!(
410 Arc::ptr_eq(&e1, &e2),
411 "a repeat call on the same broken content must be cached too"
412 );
413 }
414
415 /// [DECISION D]'s own proof: two files, parsed in two separate
416 /// `cached_parse` calls (never a single `phase_parse`-style batch), must
417 /// still receive disjoint `ExprId` ranges — the hazard this decision
418 /// closes is specifically the *cross-call* case a single shared counter
419 /// within one call never had to worry about.
420 #[test]
421 fn expr_ids_never_collide_across_separate_calls() {
422 let p1 = unique_path("exprs-a");
423 let p2 = unique_path("exprs-b");
424 // Parse two identical, nontrivial files (each with a function body
425 // tail expression, so each consumes at least one real ExprId) in two
426 // separate `cached_parse` calls — never batched through one shared
427 // counter the way a single `phase_parse` call already guarantees.
428 let src = "commons demo\n\nfn f(x: Int) -> Int {\n x + 1\n}\n";
429 let (_, r1) = cached_parse(&p1, src);
430 let (_, r2) = cached_parse(&p2, src);
431 let (u1, _) = r1.expect("parses");
432 let (u2, _) = r2.expect("parses");
433
434 fn fn_tail_expr_id(u: &SourceUnit) -> bynk_syntax::ast::ExprId {
435 let SourceUnit::Commons(c) = u else {
436 panic!("expected commons")
437 };
438 let bynk_syntax::ast::CommonsItem::Fn(f) = &c.items[0] else {
439 panic!("expected fn")
440 };
441 f.body.tail.id
442 }
443 let id1 = fn_tail_expr_id(&u1[0]);
444 let id2 = fn_tail_expr_id(&u2[0]);
445 assert_ne!(
446 id1, id2,
447 "two files parsed in separate cached_parse calls must never share an ExprId"
448 );
449 }
450
451 // -- Staleness fixture (issue #1515's own "Done when": rename, deletion,
452 // concurrent-edit interleaving — not just a byte-golden pass). --
453
454 /// A rename is, to this cache, an old path that stops being queried and a
455 /// new path that starts being — there is no "move" operation to get
456 /// wrong, but the new path must get its own independent `FileId` and
457 /// parse result, uncontaminated by whatever the old path held.
458 #[test]
459 fn a_renamed_file_gets_its_own_independent_entry() {
460 let old_path = unique_path("renamed-old");
461 let new_path = unique_path("renamed-new");
462 let (old_id, old_result) = cached_parse(&old_path, CLEAN);
463 // The "rename": the same content now arrives under a new path, as if
464 // the old path had moved. Nothing ever queries `old_path` again.
465 let (new_id, new_result) = cached_parse(&new_path, CLEAN);
466
467 assert_ne!(old_id, new_id, "a renamed file is a new path identity");
468 let (old_units, _) = old_result.expect("clean source parses");
469 let (new_units, _) = new_result.expect("clean source parses");
470 assert!(
471 !Arc::ptr_eq(&old_units, &new_units),
472 "the new path must not silently reuse the old path's cache entry"
473 );
474 // The old path's entry is still independently queryable and correct
475 // — a rename doesn't corrupt what's left behind, it just stops being
476 // read.
477 let (old_id_again, _) = cached_parse(&old_path, CLEAN);
478 assert_eq!(old_id, old_id_again);
479 }
480
481 /// A deleted file simply stops being queried — nothing in this cache's
482 /// own API models deletion explicitly (discovery, one layer up, just
483 /// stops listing the path). The property worth proving is that an
484 /// orphaned entry cannot corrupt a *different* path's own result, and
485 /// that the cap-driven eviction `cached_parse` already performs (mirrors
486 /// `PROJECT_UNIT_CACHE_CAP`'s own precedent) recovers cleanly — a path
487 /// evicted and later re-queried reparses correctly rather than serving
488 /// stale or corrupted state.
489 ///
490 /// Drives a small, locally-owned `ParseCacheState` through
491 /// `cached_parse_in` directly (PR #1520's own bot review, finding #2)
492 /// rather than forcing the real, global `CACHE` past its real,
493 /// 4096-entry cap — that would race every other test in this module
494 /// asserting on the same process-wide `static` (`cargo test` runs one
495 /// binary's tests concurrently by default) and cost thousands of real
496 /// parses to do it. A local state with a cap of 2 exercises the exact
497 /// same eviction branch precisely, cheaply, and in isolation.
498 #[test]
499 fn an_evicted_entry_reparses_correctly_rather_than_serving_stale_state() {
500 let mut state = ParseCacheState::default();
501 let cap = 2;
502 let survivor = PathBuf::from("/parse-cache-test/evict-survivor.bynk");
503 let (_, before) = cached_parse_in(&mut state, &survivor, CLEAN, cap);
504 let (survivor_units_before, _) = before.expect("clean source parses");
505
506 // Push past the cap with filler paths — the last one triggers the
507 // "past the cap, clear and let entries repopulate lazily" eviction,
508 // clearing `survivor`'s own entry along with everything else.
509 for i in 0..cap {
510 let p = PathBuf::from(format!("/parse-cache-test/evict-filler-{i}.bynk"));
511 let _ = cached_parse_in(&mut state, &p, CLEAN, cap);
512 }
513
514 let (_, after) = cached_parse_in(&mut state, &survivor, CLEAN, cap);
515 let (survivor_units_after, _) = after.expect("clean source parses");
516 // Correctness, not identity: post-eviction the entry is necessarily
517 // re-parsed (a fresh `Arc`), but it must still be the *same* parse of
518 // the *same* content — not stale, not corrupted by whichever filler
519 // entry happened to occupy the cap-cleared map.
520 assert_eq!(
521 format!("{survivor_units_before:?}"),
522 format!("{survivor_units_after:?}"),
523 "a re-parsed-after-eviction file must produce the same result as before eviction"
524 );
525 }
526
527 /// Concurrent edits: many threads racing `cached_parse` calls against the
528 /// *same* path with *different* content must never panic or corrupt the
529 /// cache (`PROJECT_UNIT_CACHE`'s own `Mutex`-protected precedent, kept
530 /// here) — and once every writer has finished, the cache must correctly
531 /// reflect whichever content a query actually asks for next, not some
532 /// torn mix of two threads' writes.
533 #[test]
534 fn concurrent_edits_to_the_same_path_never_corrupt_the_cache() {
535 let path = Arc::new(unique_path("concurrent"));
536 let variants: Vec<String> = (0..8)
537 .map(|i| format!("commons demo\n\ntype T{i} = {{ n: Int }}\n"))
538 .collect();
539
540 std::thread::scope(|scope| {
541 for v in &variants {
542 let path = Arc::clone(&path);
543 scope.spawn(move || {
544 for _ in 0..20 {
545 let (_, result) = cached_parse(&path, v);
546 assert!(result.is_ok(), "every variant here is syntactically valid");
547 }
548 });
549 }
550 });
551
552 // After the race, a fresh query with a known, distinct piece of
553 // content must reparse and return exactly that content's own result
554 // — proving the cache settled into a consistent state, not a torn one
555 // that panics or returns nonsense on the next access.
556 let (_, tail) = cached_parse(&path, CLEAN);
557 let (units, _) = tail.expect("clean source parses");
558 assert_eq!(units.len(), 1);
559 }
560}