@tishlang/tish-format 3.10.4 → 3.10.11

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. package/bin/tish-format +0 -0
  2. package/crates/tish/src/cargo_native_registry.rs +4 -0
  3. package/crates/tish/src/main.rs +65 -25
  4. package/crates/tish/tests/integration_test.rs +3 -0
  5. package/crates/tish_builtins/src/globals.rs +3 -3
  6. package/crates/tish_builtins/src/symbol.rs +86 -4
  7. package/crates/tish_bytecode/src/compiler.rs +65 -25
  8. package/crates/tish_bytecode/tests/captured_param_slot_binding.rs +131 -0
  9. package/crates/tish_bytecode/tests/fundecl_slot_binding.rs +115 -0
  10. package/crates/tish_bytecode/tests/param_defaults_slot_binding.rs +86 -0
  11. package/crates/tish_core/src/shape.rs +55 -2
  12. package/crates/tish_core/src/value.rs +229 -13
  13. package/crates/tish_eval/src/timers.rs +66 -2
  14. package/crates/tish_ffi/src/lib.rs +74 -12
  15. package/crates/tish_ffi/tests/loader.rs +35 -0
  16. package/crates/tish_fmt/Cargo.toml +1 -1
  17. package/crates/tish_lsp/src/main.rs +91 -2
  18. package/crates/tish_pg/src/lib.rs +59 -8
  19. package/crates/tish_pg/src/statement_registry.rs +250 -0
  20. package/crates/tish_runtime/src/http.rs +19 -0
  21. package/crates/tish_runtime/src/http_fetch.rs +159 -65
  22. package/crates/tish_runtime/src/http_hyper.rs +18 -2
  23. package/crates/tish_runtime/src/lib.rs +4 -0
  24. package/crates/tish_runtime/src/net.rs +125 -97
  25. package/crates/tish_runtime/src/process_spawn.rs +326 -113
  26. package/crates/tish_runtime/src/promise.rs +521 -90
  27. package/crates/tish_runtime/src/promise_io.rs +29 -6
  28. package/crates/tish_runtime/src/pty.rs +161 -107
  29. package/crates/tish_runtime/src/stream_buf.rs +419 -0
  30. package/crates/tish_runtime/src/ws.rs +431 -52
  31. package/crates/tish_vm/src/jit.rs +473 -67
  32. package/crates/tish_vm/src/lib.rs +6 -0
  33. package/crates/tish_vm/src/vm.rs +8 -5
  34. package/crates/tish_vm/tests/captured_param_slot_mode.rs +210 -0
  35. package/crates/tish_vm/tests/captured_param_vm_slots_off.rs +64 -0
  36. package/crates/tish_vm/tests/fundecl_slot_no_frame_cycle.rs +152 -0
  37. package/crates/tish_vm/tests/param_defaults_slot_mode.rs +126 -0
  38. package/crates/tish_vm/tests/param_defaults_vm_slots_off.rs +44 -0
  39. package/package.json +1 -1
  40. package/platform/darwin-arm64/tish-fmt +0 -0
  41. package/platform/darwin-x64/tish-fmt +0 -0
  42. package/platform/linux-arm64/tish-fmt +0 -0
  43. package/platform/linux-x64/tish-fmt +0 -0
  44. package/platform/win32-x64/tish-fmt.exe +0 -0
@@ -129,10 +129,11 @@ pub struct NumericFn {
129
129
  /// [`NumericFn::call`] ABI; an out-of-bounds array access (or a non-numeric return) sets a
130
130
  /// per-thread deopt flag ([`jv_take_deopt`]) and the caller discards the result + re-interprets.
131
131
  jv: bool,
132
- /// #187: true when this function embeds a native call to a registered callee. Such a function is
133
- /// NOT cached by [`try_compile_numeric`] (its embedded callee address could go stale if a
134
- /// long-lived process reuses chunk addresses across programs) — it is recompiled per closure
135
- /// creation, which resolves against the live callee registry. `false` (cacheable) for all others.
132
+ /// #187: true when this function embeds a native call to a registered callee. Its cache entry is
133
+ /// scoped to the callee-registry GENERATION it compiled under (#703, [`JitGlobal::callees_gen`]):
134
+ /// within one program run the callee binding is proven stable, so the entry is reused; after a
135
+ /// program boundary ([`reset_callees`]) the entry is a miss and the function recompiles once
136
+ /// against the live registry — never against a stale callee.
136
137
  uses_xcall: bool,
137
138
  /// #187: true when this is a VOID array-mode function (only returns the implicit `null`). Its
138
139
  /// `f64` result is a dummy, so [`try_call_array_jit`] returns `Value::Null` instead of a number.
@@ -520,19 +521,24 @@ impl NumericFn {
520
521
 
521
522
  struct JitGlobal {
522
523
  module: JITModule,
523
- /// Keyed by the address of the nested `Chunk`, with a content **fingerprint** alongside the
524
- /// result. Within one program run a chunk lives for the whole run, so the address is stable and
525
- /// unique. But this cache is a process-global that is never cleared, and a `Chunk` is dropped when
526
- /// its program is — so a long-lived process that compiles/drops/recompiles programs (the REPL;
527
- /// embedders running multiple scripts) can allocate a *different* chunk at a freed address that is
528
- /// still cached. We therefore verify the fingerprint on every hit: a mismatch means the address was
529
- /// reused by a different chunk, so we recompile (and overwrite) instead of returning stale native
530
- /// code. `None` still caches "not JIT-eligible". See [`chunk_fingerprint`].
531
- cache: HashMap<usize, (u64, Option<NumericFn>)>,
532
- /// OSR loop-region cache (#190), keyed by `(chunk address, loop header ip)` with the same
533
- /// fingerprint guard as `cache`. `None` caches "region not compilable" so a loop that fails the
534
- /// whitelist is scanned once, not on every back-edge past the trigger threshold.
535
- osr_cache: HashMap<(usize, usize), (u64, Option<LoopFn>)>,
524
+ /// Keyed by **content identity** — the primary [`chunk_fingerprints`] hash — NOT by chunk
525
+ /// address (#703). The VM deep-clones a `Chunk` for every closure instance (`vm.rs`
526
+ /// `LoadConst(Closure)`), so an address key is per closure *instance*, not per program function:
527
+ /// every qualifying closure creation minted a fresh cranelift compile whose sealed executable
528
+ /// page (16 KB on macOS arm64) is never freed — ~18 KB leaked per closure creation, driven by
529
+ /// runtime event volume rather than code size. Content keys make every instance of the same
530
+ /// source function share ONE compile, so the map grows with distinct program text only. A hit is
531
+ /// honored only when the entry's [`CacheTail`] also matches (second hash + exact structural
532
+ /// dims), so a 64-bit key collision degrades to a recompile, never a miscompile. `None` still
533
+ /// caches "not JIT-eligible"; registry-sensitive entries are re-validated against
534
+ /// `callees_gen`/`callees` — see [`NumEntry`].
535
+ cache: HashMap<u64, NumEntry>,
536
+ /// OSR loop-region cache (#190), keyed by `(content fingerprint, loop header ip)` with the same
537
+ /// [`CacheTail`] collision guard as `cache` (#703 — same per-instance-address leak otherwise).
538
+ /// `None` caches "region not compilable" so a loop that fails the whitelist is scanned once, not
539
+ /// on every back-edge past the trigger threshold. Loop regions are always registry-independent
540
+ /// (the region whitelist rejects `LoadVar`), so entries never need generation re-validation.
541
+ osr_cache: HashMap<(u64, usize), OsrEntry>,
536
542
  counter: usize,
537
543
  /// `FuncId` of the imported `tish_math_call` host fn (#186), declared once at module init and
538
544
  /// re-imported into each compiled function via `declare_func_in_func`.
@@ -548,17 +554,142 @@ struct JitGlobal {
548
554
  /// the binding can never change under a cached caller. A caller that references a name NOT yet here
549
555
  /// (a forward reference) simply bails to the interpreter.
550
556
  callees: HashMap<Arc<str>, CalleeEntry>,
557
+ /// #703: generation counter for `callees`, bumped by [`reset_callees`] at every top-level program
558
+ /// boundary. A cached cross-calling function ([`NumericFn::uses_xcall`]) embeds native calls to
559
+ /// callee ids it resolved at compile time; that binding is proven stable only WITHIN one program
560
+ /// run (`global_name` is per-program-stable), so such an entry is honored only while the
561
+ /// generation it was compiled under is still current. This is what lets cross-callers be cached
562
+ /// at all (pre-#703 they recompiled — and burned a fresh executable page — on every closure
563
+ /// creation) while preserving the original soundness argument: a stale callee is never invoked.
564
+ callees_gen: u64,
551
565
  }
552
566
 
553
567
  /// #187: a registered directly-callable numeric callee (register-`f64` ABI). Callers resolve against
554
- /// the LIVE registry at compile time and are never cached ([`NumericFn::uses_xcall`]), so a name
555
- /// re-registered by a later program simply overwrites this — a stale callee is never invoked.
568
+ /// the LIVE registry at compile time; their cache entries are scoped to the registry generation they
569
+ /// compiled under (#703, see [`JitGlobal::callees_gen`]), so a name re-registered by a later program
570
+ /// simply overwrites this — a stale callee is never invoked.
556
571
  #[derive(Clone, Copy)]
557
572
  struct CalleeEntry {
558
573
  id: cranelift_module::FuncId,
559
574
  arity: u8,
560
575
  }
561
576
 
577
+ /// Collision-verification tail stored with every content-keyed cache entry (#703). The map key is a
578
+ /// single 64-bit fingerprint; unlike the old address-keyed scheme (where a wrong hit only meant a
579
+ /// freed-and-reused address), a false content hit would hand one chunk another chunk's native code —
580
+ /// a miscompile. A hit is therefore honored only when a SECOND, independently-mixed 64-bit
581
+ /// fingerprint over the same input AND the exact structural dimensions all match. Two chunks that
582
+ /// agree on both hashes (independent multipliers/avalanches over identical input streams) and every
583
+ /// length/shape field below are identical for compilation purposes to ~2^-128 confidence — stronger
584
+ /// in practice, since the inputs are compiler-generated bytecode, not adversarial hash-seeking data.
585
+ /// A mismatch is treated as a miss: recompile and overwrite (the superseded entry's code page stays
586
+ /// mapped, as all JIT pages do — see the finalize note in [`compile_chunk`]).
587
+ #[derive(Clone, Copy, PartialEq, Eq)]
588
+ struct CacheTail {
589
+ fp2: u64,
590
+ code_len: u32,
591
+ const_len: u32,
592
+ param_count: u16,
593
+ num_slots: u16,
594
+ }
595
+
596
+ impl CacheTail {
597
+ fn of(chunk: &Chunk, fp2: u64) -> Self {
598
+ Self {
599
+ fp2,
600
+ code_len: chunk.code.len() as u32,
601
+ const_len: chunk.constants.len() as u32,
602
+ param_count: chunk.param_count,
603
+ num_slots: chunk.num_slots,
604
+ }
605
+ }
606
+
607
+ fn matches(&self, chunk: &Chunk, fp2: u64) -> bool {
608
+ *self == Self::of(chunk, fp2)
609
+ }
610
+ }
611
+
612
+ /// A `cache` entry (#703): the compile result plus everything needed to re-validate and to replay
613
+ /// the compile's side effects on a content-keyed hit.
614
+ struct NumEntry {
615
+ tail: CacheTail,
616
+ result: Option<NumericFn>,
617
+ /// `Some` when compiling this chunk registered it as a directly-callable callee (#187: plain
618
+ /// register-`f64`, non-jv/non-guarded/non-bool, with a stable `global_name`). A cache hit must
619
+ /// REPLAY that registration — re-inserting the id under the current chunk's `global_name` — or a
620
+ /// program re-run in a long-lived process (its registry cleared by [`reset_callees`], its chunks
621
+ /// all cache hits) would never re-populate the registry and cross-function JIT calls would
622
+ /// silently stop resolving. The finalized id's code pointer is process-permanent, so replaying it
623
+ /// into any later generation is sound.
624
+ callee_id: Option<cranelift_module::FuncId>,
625
+ /// `callees` generation this entry was compiled under (see [`JitGlobal::callees_gen`]).
626
+ callees_gen: u64,
627
+ /// `callees.len()` right after this compile. Consulted only for registry-sensitive `None`
628
+ /// entries: a chunk that references globals may have failed to compile *because* a callee wasn't
629
+ /// registered yet, so registry growth within the generation retries it (once per growth step —
630
+ /// the registry only grows within a generation, so steady state re-hits the cached `None`).
631
+ callees_len: u32,
632
+ /// Whether the chunk references any global (`LoadVar`), i.e. whether its compile RESULT could
633
+ /// depend on the callee registry at all. `false` ⇒ the entry is valid regardless of registry
634
+ /// state or generation (a compiled non-xcall body provably contains no `LoadVar` — an unresolved
635
+ /// one bails compilation and a resolved one makes it xcall).
636
+ registry_sensitive: bool,
637
+ }
638
+
639
+ /// An `osr_cache` entry (#703). Loop regions never consult the callee registry (`LoadVar` is outside
640
+ /// the region whitelist), so only the collision tail — plus the region bounds — needs re-validation.
641
+ struct OsrEntry {
642
+ tail: CacheTail,
643
+ /// End of the compiled region. The key carries only `(fingerprint, header_ip)` (matching the old
644
+ /// address-based key's assumption that one header identifies one region); storing the end and
645
+ /// checking it on hit turns any violation of that assumption into a recompile, not a wrong region.
646
+ region_end: usize,
647
+ result: Option<LoopFn>,
648
+ }
649
+
650
+ /// Defensive bound on the JIT cache maps (`cache`, `osr_cache`, and the per-thread
651
+ /// [`OSR_EXPAND_CACHE`] memo), overridable via `TISH_JIT_CACHE_CAP` (`0` ⇒ unbounded). With content
652
+ /// keys the maps grow with DISTINCT program text, so a normal workload never approaches the cap; it
653
+ /// exists so a degenerate embedder (eval-ing freshly generated program text per event, say) cannot
654
+ /// grow the maps without bound. On overflow the code caches STOP CACHING new entries — they never
655
+ /// evict: an evicted entry's executable page may be running on another thread and could not be
656
+ /// unmapped anyway (pages are process-permanent), so eviction would only trade a map entry for a
657
+ /// fresh page-burning recompile on the next lookup. The expand memo (pure re-computable analysis,
658
+ /// no pages) clears instead.
659
+ fn cache_cap() -> usize {
660
+ static CAP: OnceLock<usize> = OnceLock::new();
661
+ *CAP.get_or_init(|| {
662
+ std::env::var("TISH_JIT_CACHE_CAP")
663
+ .ok()
664
+ .and_then(|v| v.parse::<usize>().ok())
665
+ .unwrap_or(65_536)
666
+ })
667
+ }
668
+
669
+ /// Does `chunk` reference any global name (a `LoadVar` op)? Such a chunk's compile result depends on
670
+ /// the live callee registry — `LoadVar` of a registered name lowers to a native call, of an
671
+ /// unregistered one bails compilation — so its cache entry needs registry re-validation (#703).
672
+ /// Walks with `instruction_size` so an operand byte can't be mistaken for an opcode; malformed code
673
+ /// is conservatively "sensitive".
674
+ fn chunk_references_globals(chunk: &Chunk) -> bool {
675
+ let code = &chunk.code;
676
+ let mut ip = 0usize;
677
+ while ip < code.len() {
678
+ let op = match Opcode::from_u8(code[ip]) {
679
+ Some(o) => o,
680
+ None => return true,
681
+ };
682
+ if op == Opcode::LoadVar {
683
+ return true;
684
+ }
685
+ ip += match op.instruction_size(code, ip) {
686
+ Some(s) => s,
687
+ None => return true,
688
+ };
689
+ }
690
+ false
691
+ }
692
+
562
693
  // SAFETY: `JITModule` is `!Send`, but the single instance lives behind the
563
694
  // `Mutex` in the process-global `JIT` and is never moved out or dropped; all
564
695
  // access is serialized by the mutex.
@@ -594,6 +725,13 @@ thread_local! {
594
725
  /// times (e.g. nbody's `advance`), where recomputing per call is a real regression. A stale entry
595
726
  /// (a freed chunk's address reused) can only mis-route a perf hint: `run_osr`/`compile_loop_region`
596
727
  /// re-validate the region structurally + by fingerprint, so a wrong hint never miscompiles.
728
+ ///
729
+ /// #703: this memo stays ADDRESS-keyed — it sits on the per-back-edge hot path of every hot loop,
730
+ /// where an O(chunk) content hash per lookup would be a real regression — but with per-closure-
731
+ /// instance chunk clones an address key grows one ~64 B entry per instance, so [`osr_expand_cached`]
732
+ /// bounds it: at [`cache_cap`] entries the map is CLEARED and rebuilt. Clearing is safe precisely
733
+ /// because entries hold no code pages — they are pure loop-structure analysis, recomputed on the
734
+ /// next back-edge past the threshold.
597
735
  static OSR_EXPAND_CACHE: std::cell::RefCell<HashMap<(usize, usize), (usize, usize, bool, bool)>> =
598
736
  std::cell::RefCell::new(HashMap::new());
599
737
  }
@@ -852,6 +990,7 @@ fn jit() -> Option<&'static Mutex<JitGlobal>> {
852
990
  math_binary_call_id,
853
991
  jv_fns,
854
992
  callees: HashMap::new(),
993
+ callees_gen: 0,
855
994
  })
856
995
  })
857
996
  })
@@ -867,25 +1006,35 @@ fn read_u16(code: &[u8], ip: &mut usize) -> Option<u16> {
867
1006
  Some((a << 8) | b)
868
1007
  }
869
1008
 
870
- /// Content fingerprint of everything `compile_chunk` reads, so a cache entry can be validated against
871
- /// the chunk currently at a (possibly reused) address. FNV-1a over the compile-relevant fields:
872
- /// shape (`param_count`, `num_slots`, `rest_param_index`, `slot_based`), the full `code` bytes, and
873
- /// the `constants` (the JIT emits `f64const`/bool from `LoadConst`, so their values matter). The JIT
874
- /// makes no cross-chunk calls (`op_size` allows only `SelfCall`, which recurses into *this* function),
875
- /// so nothing outside the chunk affects the result — this fingerprint is complete. Deterministic
876
- /// within a process (fixed FNV constants, not a randomized hasher), which is all the cache needs.
877
- fn chunk_fingerprint(chunk: &Chunk) -> u64 {
878
- // Mixes a u64 at a time (FNV-prime multiply + an avalanche shift). Eight bytes per round keeps
879
- // this cheap on the hot closure-creation path; correctness only needs determinism + good
880
- // distinction, not cryptographic strength.
1009
+ /// Content fingerprints of everything `compile_chunk` reads — `(primary, secondary)`. The primary
1010
+ /// hash is the cache KEY (#703: the caches are keyed on content identity, not chunk address, because
1011
+ /// the VM deep-clones a `Chunk` per closure instance); the secondary goes into the entry's
1012
+ /// [`CacheTail`] so a primary-key collision is detected as a miss rather than becoming a miscompile.
1013
+ /// Covers the compile-relevant fields: shape (`param_count`, `num_slots`, `rest_param_index`,
1014
+ /// `slot_based`), the full `code` bytes, the `constants` (the JIT emits `f64const`/bool from
1015
+ /// `LoadConst`, so their values matter), the `names` table (JV member lowering dispatches on
1016
+ /// `"length"`/`"push"` by name, and #187 `LoadVar` callee resolution is by name), and `global_name`
1017
+ /// (it drives the callee-registration side effect a hit replays). Together with the callee-registry
1018
+ /// state re-validated per entry (see [`NumEntry`]) and the process-stable `OnceLock` env flags, two
1019
+ /// chunks with equal fingerprints are interchangeable compile inputs. Deterministic within a process
1020
+ /// (fixed constants, not a randomized hasher), which is all the caches need.
1021
+ fn chunk_fingerprints(chunk: &Chunk) -> (u64, u64) {
1022
+ // Mixes a u64 at a time into TWO accumulators in one pass. h1 is the original FNV-1a-style mix
1023
+ // (FNV prime + avalanche shift); h2 uses a different offset basis, a golden-ratio pre-add, the
1024
+ // murmur3-finalizer multiplier and a different shift — an independently-mixed hash family, so a
1025
+ // simultaneous collision of both over identical-length inputs is ~2^-128. Eight bytes per round
1026
+ // keeps this cheap on the hot closure-creation path.
881
1027
  #[inline]
882
- fn mix(h: &mut u64, v: u64) {
883
- *h ^= v;
884
- *h = h.wrapping_mul(0x0000_0100_0000_01b3);
885
- *h ^= *h >> 29;
1028
+ fn mix(h: &mut (u64, u64), v: u64) {
1029
+ h.0 ^= v;
1030
+ h.0 = h.0.wrapping_mul(0x0000_0100_0000_01b3);
1031
+ h.0 ^= h.0 >> 29;
1032
+ h.1 ^= v.wrapping_add(0x9e37_79b9_7f4a_7c15);
1033
+ h.1 = h.1.wrapping_mul(0xff51_afd7_ed55_8ccd);
1034
+ h.1 ^= h.1 >> 33;
886
1035
  }
887
1036
  #[inline]
888
- fn mix_bytes(h: &mut u64, bytes: &[u8]) {
1037
+ fn mix_bytes(h: &mut (u64, u64), bytes: &[u8]) {
889
1038
  let mut it = bytes.chunks_exact(8);
890
1039
  for w in &mut it {
891
1040
  mix(h, u64::from_le_bytes(w.try_into().unwrap()));
@@ -898,7 +1047,7 @@ fn chunk_fingerprint(chunk: &Chunk) -> u64 {
898
1047
  }
899
1048
  mix(h, bytes.len() as u64);
900
1049
  }
901
- let mut h: u64 = 0xcbf2_9ce4_8422_2325;
1050
+ let mut h: (u64, u64) = (0xcbf2_9ce4_8422_2325, 0x6a09_e667_f3bc_c908);
902
1051
  mix(&mut h, chunk.param_count as u64);
903
1052
  mix(&mut h, chunk.num_slots as u64);
904
1053
  mix(&mut h, chunk.rest_param_index as u64);
@@ -923,6 +1072,17 @@ fn chunk_fingerprint(chunk: &Chunk) -> u64 {
923
1072
  }
924
1073
  }
925
1074
  }
1075
+ mix(&mut h, chunk.names.len() as u64);
1076
+ for n in &chunk.names {
1077
+ mix_bytes(&mut h, n.as_bytes());
1078
+ }
1079
+ match &chunk.global_name {
1080
+ Some(n) => {
1081
+ mix(&mut h, 7);
1082
+ mix_bytes(&mut h, n.as_bytes());
1083
+ }
1084
+ None => mix(&mut h, 8),
1085
+ }
926
1086
  h
927
1087
  }
928
1088
 
@@ -930,17 +1090,31 @@ fn chunk_fingerprint(chunk: &Chunk) -> u64 {
930
1090
  /// Returns `None` if the chunk isn't a straight-line numeric function.
931
1091
  /// #187: clear the directly-callable-callee registry at the start of each top-level program run, so a
932
1092
  /// long-lived process (REPL / embedder) never resolves a callee registered by a PRIOR program (a name
933
- /// re-registered non-numerically would otherwise leave a stale native entry). Cross-callers aren't
934
- /// cached, so they always re-resolve against the freshly-populated registry.
1093
+ /// re-registered non-numerically would otherwise leave a stale native entry). #703: also bump the
1094
+ /// registry GENERATION — cached cross-callers ([`NumericFn::uses_xcall`]) are valid only within the
1095
+ /// generation they resolved their callees under, so after this every cross-caller re-resolves against
1096
+ /// the freshly-populated registry (by recompiling once, not once per closure creation).
935
1097
  #[cfg(not(target_arch = "wasm32"))]
936
1098
  pub fn reset_callees() {
937
1099
  if let Some(lock) = jit() {
938
1100
  if let Ok(mut g) = lock.lock() {
939
1101
  g.callees.clear();
1102
+ g.callees_gen = g.callees_gen.wrapping_add(1);
940
1103
  }
941
1104
  }
942
1105
  }
943
1106
 
1107
+ /// Test/diagnostic introspection (#703): entry counts of the process-global JIT caches,
1108
+ /// `(cache, osr_cache)`. The regression tests assert these stay flat while the same closure body is
1109
+ /// instantiated (and its per-instance chunk clone compiled) many times over.
1110
+ #[doc(hidden)]
1111
+ pub fn jit_cache_lens() -> (usize, usize) {
1112
+ match jit().map(|lock| lock.lock()) {
1113
+ Some(Ok(g)) => (g.cache.len(), g.osr_cache.len()),
1114
+ _ => (0, 0),
1115
+ }
1116
+ }
1117
+
944
1118
  pub fn try_compile_numeric(chunk: &Chunk) -> Option<NumericFn> {
945
1119
  if !chunk.slot_based
946
1120
  || chunk.rest_param_index != NO_REST_PARAM
@@ -949,23 +1123,77 @@ pub fn try_compile_numeric(chunk: &Chunk) -> Option<NumericFn> {
949
1123
  {
950
1124
  return None;
951
1125
  }
952
- let key = chunk as *const Chunk as usize;
953
- let fp = chunk_fingerprint(chunk);
1126
+ let (fp, fp2) = chunk_fingerprints(chunk);
954
1127
  let lock = jit()?;
955
1128
  let mut g = lock.lock().ok()?;
956
- // Hit only counts if the fingerprint matches: otherwise this address was freed and reused by a
957
- // *different* chunk, and the cached `NumericFn` is native code for the old one (a miscompile).
958
- if let Some(&(cached_fp, cached)) = g.cache.get(&key) {
959
- if cached_fp == fp {
960
- return cached;
1129
+ let gen = g.callees_gen;
1130
+ let live_callees = g.callees.len() as u32;
1131
+ if let Some(entry) = g.cache.get(&fp) {
1132
+ // The tail must match — a bare 64-bit key collision would otherwise hand this chunk native
1133
+ // code compiled from a DIFFERENT chunk (see [`CacheTail`]). Tail mismatch ⇒ recompile below
1134
+ // (overwriting the colliding entry).
1135
+ if entry.tail.matches(chunk, fp2) {
1136
+ // Registry re-validation (#703):
1137
+ // * a compiled non-xcall body provably contains no `LoadVar` — valid forever;
1138
+ // * a cross-caller (`uses_xcall`) embeds callee ids proven stable only within its
1139
+ // compile generation — valid while the generation matches;
1140
+ // * a `None` for a globals-referencing chunk may exist only because a callee wasn't
1141
+ // registered yet — valid while the generation AND registry size are unchanged.
1142
+ let valid = match entry.result {
1143
+ Some(nf) if nf.uses_xcall => entry.callees_gen == gen,
1144
+ Some(_) => true,
1145
+ None => {
1146
+ !entry.registry_sensitive
1147
+ || (entry.callees_gen == gen && entry.callees_len == live_callees)
1148
+ }
1149
+ };
1150
+ if valid {
1151
+ let result = entry.result;
1152
+ let callee_id = entry.callee_id;
1153
+ // Replay the compile's registration side effect: without this, a program re-run in a
1154
+ // long-lived process (registry cleared, every chunk a cache hit) would leave the
1155
+ // registry empty and cross-function JIT calls would silently stop resolving. The id's
1156
+ // finalized code is process-permanent, so re-registering it is always sound;
1157
+ // `global_name` is part of the fingerprint, so it names the same source function.
1158
+ if let (Some(id), Some(name), Some(nf)) =
1159
+ (callee_id, chunk.global_name.as_ref(), result)
1160
+ {
1161
+ g.callees.insert(
1162
+ Arc::clone(name),
1163
+ CalleeEntry {
1164
+ id,
1165
+ arity: nf.arity,
1166
+ },
1167
+ );
1168
+ }
1169
+ return result;
1170
+ }
961
1171
  }
962
1172
  }
963
1173
  let result = compile_chunk(&mut g, chunk);
964
- // #187: a function that embeds a native call to a registered callee is NOT cached — its callee
965
- // address could go stale across programs in a long-lived process. It recompiles per closure
966
- // creation (once, in practice), resolving against the live registry. Everything else caches.
967
- if !result.is_some_and(|nf| nf.uses_xcall) {
968
- g.cache.insert(key, (fp, result));
1174
+ // Recover the callee id `compile_chunk` just registered (if it did — plain register-`f64`
1175
+ // functions with a stable `global_name` only), so a later hit can replay the registration.
1176
+ let callee_id = match (result, chunk.global_name.as_ref()) {
1177
+ (Some(nf), Some(name))
1178
+ if nf.array_param_mask == 0 && !nf.jv && !nf.recur_guarded && !nf.result_bool =>
1179
+ {
1180
+ g.callees.get(name).map(|e| e.id)
1181
+ }
1182
+ _ => None,
1183
+ };
1184
+ let entry = NumEntry {
1185
+ tail: CacheTail::of(chunk, fp2),
1186
+ result,
1187
+ callee_id,
1188
+ callees_gen: gen,
1189
+ callees_len: g.callees.len() as u32,
1190
+ registry_sensitive: chunk_references_globals(chunk),
1191
+ };
1192
+ // #703 defensive bound: at the cap, only overwrites of an existing key land — new entries are
1193
+ // simply not cached (never evict; see [`cache_cap`]).
1194
+ let cap = cache_cap();
1195
+ if cap == 0 || g.cache.len() < cap || g.cache.contains_key(&fp) {
1196
+ g.cache.insert(fp, entry);
969
1197
  }
970
1198
  result
971
1199
  }
@@ -1080,27 +1308,46 @@ pub fn osr_expand_cached(
1080
1308
  let has_arrays = osr_region_has_arrays(chunk, th, te);
1081
1309
  let array_worthy = has_arrays && !osr_region_enclosed(chunk, th, te);
1082
1310
  let v = (th, te, has_arrays, array_worthy);
1083
- c.borrow_mut().insert(key, v);
1311
+ let mut m = c.borrow_mut();
1312
+ // #703: address-keyed per closure instance ⇒ unbounded growth under closure-minting
1313
+ // workloads. Clear-on-full (NOT stop-caching, unlike the code caches): entries are pure
1314
+ // recomputable analysis, so clearing costs one rescan per live hot loop and bounds the map.
1315
+ let cap = cache_cap();
1316
+ if cap != 0 && m.len() >= cap {
1317
+ m.clear();
1318
+ }
1319
+ m.insert(key, v);
1084
1320
  v
1085
1321
  })
1086
1322
  }
1087
1323
 
1088
1324
  /// Compile the hot loop region `[header_ip, region_end)` of `chunk` to native code (#190 OSR), or
1089
- /// `None` if it is not a pure-numeric slot loop. Cached per `(chunk, header_ip)` with a fingerprint
1090
- /// guard (negative results included, so a non-compilable loop is scanned once). Called from the frame
1091
- /// VM's `JumpBack` handler once a loop's back-edge counter crosses the trigger threshold.
1325
+ /// `None` if it is not a pure-numeric slot loop. Cached per `(content fingerprint, header_ip)` with
1326
+ /// the [`CacheTail`] collision guard (#703 — every closure instance of the same source function
1327
+ /// shares one compile; negative results included, so a non-compilable loop is scanned once, not once
1328
+ /// per closure instance). Called from the frame VM's `JumpBack` handler once a loop's back-edge
1329
+ /// counter crosses the trigger threshold.
1092
1330
  pub fn try_compile_loop(chunk: &Chunk, header_ip: usize, region_end: usize) -> Option<LoopFn> {
1093
- let key = (chunk as *const Chunk as usize, header_ip);
1094
- let fp = chunk_fingerprint(chunk);
1331
+ let (fp, fp2) = chunk_fingerprints(chunk);
1332
+ let key = (fp, header_ip);
1095
1333
  let lock = jit()?;
1096
1334
  let mut g = lock.lock().ok()?;
1097
- if let Some((cached_fp, cached)) = g.osr_cache.get(&key) {
1098
- if *cached_fp == fp {
1099
- return cached.clone();
1335
+ if let Some(entry) = g.osr_cache.get(&key) {
1336
+ if entry.tail.matches(chunk, fp2) && entry.region_end == region_end {
1337
+ return entry.result.clone();
1100
1338
  }
1101
1339
  }
1102
1340
  let result = compile_loop_region(&mut g, chunk, header_ip, region_end);
1103
- g.osr_cache.insert(key, (fp, result.clone()));
1341
+ let entry = OsrEntry {
1342
+ tail: CacheTail::of(chunk, fp2),
1343
+ region_end,
1344
+ result: result.clone(),
1345
+ };
1346
+ // #703 defensive bound — same policy as the numeric cache: at the cap, overwrite-only.
1347
+ let cap = cache_cap();
1348
+ if cap == 0 || g.osr_cache.len() < cap || g.osr_cache.contains_key(&key) {
1349
+ g.osr_cache.insert(key, entry);
1350
+ }
1104
1351
  result
1105
1352
  }
1106
1353
 
@@ -2142,6 +2389,16 @@ fn compile_chunk(g: &mut JitGlobal, chunk: &Chunk) -> Option<NumericFn> {
2142
2389
  return None;
2143
2390
  }
2144
2391
  g.module.clear_context(&mut ctx);
2392
+ // NOTE(#703 follow-up): one `finalize_definitions` per compiled function seals the memory
2393
+ // provider's current allocation, so every function occupies its own page-aligned executable
2394
+ // allocation (16 KB on macOS arm64, 4 KB on x86-64) that is never unmapped. With the caches
2395
+ // content-keyed, compiles are bounded by distinct program text, so this is a bounded per-function
2396
+ // overhead rather than a leak — but packing functions tighter would need `finalize_definitions`
2397
+ // batched across compiles (the function pointer is needed immediately here, so that is a
2398
+ // call-site restructure), and cranelift's `ArenaMemoryProvider` alone does not help: its
2399
+ // `finalize` marks segments finalized too, so the next allocation opens a fresh page-aligned
2400
+ // segment. A per-program `JITModule` dropped with `free_memory` at teardown is the only way to
2401
+ // actually return code pages to a REPL/multi-program embedder.
2145
2402
  if g.module.finalize_definitions().is_err() {
2146
2403
  return None;
2147
2404
  }
@@ -4274,9 +4531,9 @@ mod tests {
4274
4531
  /// none compiles — so a change that makes the JIT silently *stop* compiling the target (the exact
4275
4532
  /// "vacuous fixture" miss that motivated this guard) fails loudly instead of passing emptily.
4276
4533
  ///
4277
- /// Bypasses [`try_compile_numeric`]'s cache and calls [`compile_chunk`] directly: the cache is
4278
- /// keyed by chunk address, unique-and-stable in a real run but reused across this test's transient
4279
- /// chunks. Compiling fresh is correct here and still exercises the real lowering path.
4534
+ /// Bypasses [`try_compile_numeric`]'s (content-keyed, #703) cache and calls [`compile_chunk`]
4535
+ /// directly, so every fixture exercises the real lowering path fresh instead of possibly
4536
+ /// returning another test's cached compile.
4280
4537
  fn jit_arity2(src: &str) -> NumericFn {
4281
4538
  let prog = tishlang_parser::parse(src).expect("parse");
4282
4539
  let opt = tishlang_opt::optimize(&prog);
@@ -4310,7 +4567,7 @@ mod tests {
4310
4567
  }
4311
4568
 
4312
4569
  /// #189: compile the first JV (function-local `f64` array) nested fn in `src` via `compile_chunk`,
4313
- /// bypassing the address-keyed cache (see [`jit_arity2`]). Panics if none compiles JV — so a change
4570
+ /// bypassing the content-keyed cache (see [`jit_arity2`]). Panics if none compiles JV — so a change
4314
4571
  /// that makes the classifier or lowering silently stop accepting the target fails loudly.
4315
4572
  fn jit_jv(src: &str) -> NumericFn {
4316
4573
  let prog = tishlang_parser::parse(src).expect("parse");
@@ -4775,9 +5032,10 @@ mod tests {
4775
5032
  }
4776
5033
 
4777
5034
  /// Regression for the address-reuse stale hit (#247): compile one function, then overwrite the SAME
4778
- /// heap `Chunk` (same address = the cache key) with a *different* function — what a long-lived
4779
- /// process (REPL / multi-script embedder) does when a freed chunk address is reused. Before the
4780
- /// fingerprint check the cache returned the first function's native code for the second.
5035
+ /// heap `Chunk` (same address) with a *different* function — what a long-lived process (REPL /
5036
+ /// multi-script embedder) does when a freed chunk address is reused. Originally this exercised the
5037
+ /// fingerprint-on-hit guard over the address-keyed cache; since #703 the cache key IS the content
5038
+ /// fingerprint, so distinct content can never share an entry — kept as a permanent behavior guard.
4781
5039
  #[test]
4782
5040
  fn jit_cache_detects_address_reuse() {
4783
5041
  let mut boxed: Box<Chunk> = Box::new(fn_chunk("const f = (a, b) => a - b\nf(0, 0)\n"));
@@ -4928,4 +5186,152 @@ mod tests {
4928
5186
  got.sort_by(|a, b| a.partial_cmp(b).unwrap());
4929
5187
  assert_eq!(got, vec![20.0, 90.0], "s=90 (0+2+…+18), i=20");
4930
5188
  }
5189
+
5190
+ /// #703 REGRESSION — the leak class: the VM deep-clones a `Chunk` per closure instance, so the
5191
+ /// old address-keyed cache treated every instance as a novel compile — one sealed, never-freed
5192
+ /// executable page each (measured ~18 KB per closure creation with instances retained). Content
5193
+ /// keys must dedupe: N live clones of one chunk (distinct heap addresses, exactly like N live
5194
+ /// closure instances) share ONE compile and ONE cache entry.
5195
+ #[test]
5196
+ fn jit_cache_content_identity_dedupes_cloned_chunks() {
5197
+ // Unique constants ⇒ a fingerprint no other test's chunk shares, so the deltas observed here
5198
+ // are our own even though the whole test binary shares the process-global JIT.
5199
+ let chunk = fn_chunk("function f(a, b) { return a * 703.0625 + b * 1219.5 }\nf(0, 0)\n");
5200
+ let (cache_before, _) = jit_cache_lens();
5201
+ let first = try_compile_numeric(&chunk).expect("numeric fn must compile");
5202
+ let mut keep: Vec<Box<Chunk>> = Vec::new(); // live clones ⇒ malloc can't recycle addresses
5203
+ for _ in 0..200 {
5204
+ let clone = Box::new(chunk.clone());
5205
+ let f = try_compile_numeric(&clone).expect("clone must hit the cache");
5206
+ // Compiled code is finalized at a process-unique, never-freed address, so pointer
5207
+ // equality holds iff the lookup HIT — a recompile would finalize new code elsewhere.
5208
+ assert_eq!(
5209
+ f.ptr, first.ptr,
5210
+ "cloned chunk must reuse the cached compile"
5211
+ );
5212
+ keep.push(clone);
5213
+ }
5214
+ let (cache_after, _) = jit_cache_lens();
5215
+ // Ours is exactly one entry; the slack absorbs unrelated entries from tests running in
5216
+ // parallel in this process. Pre-#703 this loop grew the cache by ~200 (one per clone).
5217
+ assert!(
5218
+ cache_after.saturating_sub(cache_before) < 50,
5219
+ "content-keyed cache must not grow per closure instance ({cache_before} -> {cache_after})"
5220
+ );
5221
+ assert_eq!(first.call(&[2.0, 4.0]), 2.0 * 703.0625 + 4.0 * 1219.5);
5222
+ }
5223
+
5224
+ /// #703 REGRESSION — negative results ("not JIT-eligible") are content-keyed too: N clones of a
5225
+ /// non-numeric body leave one cache entry, not one permanent ~56 B entry per closure instance
5226
+ /// (the issue's fully-dropped variant still leaked partly through these).
5227
+ #[test]
5228
+ fn jit_cache_negative_entries_dedupe() {
5229
+ let chunk = fn_chunk("function g(a, b) { return \"x703\" + a + b }\ng(0, 0)\n");
5230
+ let (before, _) = jit_cache_lens();
5231
+ assert!(
5232
+ try_compile_numeric(&chunk).is_none(),
5233
+ "string body must not JIT"
5234
+ );
5235
+ let mut keep: Vec<Box<Chunk>> = Vec::new();
5236
+ for _ in 0..200 {
5237
+ let clone = Box::new(chunk.clone());
5238
+ assert!(try_compile_numeric(&clone).is_none());
5239
+ keep.push(clone);
5240
+ }
5241
+ let (after, _) = jit_cache_lens();
5242
+ assert!(
5243
+ after.saturating_sub(before) < 50,
5244
+ "negative entries must dedupe by content ({before} -> {after})"
5245
+ );
5246
+ }
5247
+
5248
+ /// #703 REGRESSION — the same dedupe for the OSR loop-region cache: per-instance chunk clones of
5249
+ /// one hot-loop function must share one region compile and one `osr_cache` entry.
5250
+ #[test]
5251
+ fn osr_cache_content_identity_dedupes_cloned_chunks() {
5252
+ let chunk = top_chunk(
5253
+ "let s = 0.0\nlet i = 0.0\nwhile (i < 703.25) { s = s + i * 1.0009765625; i = i + 1.0 }\n",
5254
+ );
5255
+ let (_, osr_before) = jit_cache_lens();
5256
+ let (header, end) = first_region(&chunk);
5257
+ let first = try_compile_loop(&chunk, header, end).expect("numeric loop must OSR-compile");
5258
+ let mut keep: Vec<Box<Chunk>> = Vec::new();
5259
+ for _ in 0..100 {
5260
+ let clone = Box::new(chunk.clone());
5261
+ let lf = try_compile_loop(&clone, header, end).expect("clone must hit the osr cache");
5262
+ assert_eq!(
5263
+ lf.ptr, first.ptr,
5264
+ "cloned chunk must reuse the cached region compile"
5265
+ );
5266
+ keep.push(clone);
5267
+ }
5268
+ let (_, osr_after) = jit_cache_lens();
5269
+ assert!(
5270
+ osr_after.saturating_sub(osr_before) < 50,
5271
+ "content-keyed osr_cache must not grow per closure instance ({osr_before} -> {osr_after})"
5272
+ );
5273
+ }
5274
+
5275
+ /// #703 — cross-callers (`uses_xcall`) are cached WITHIN a callee-registry generation (pre-fix
5276
+ /// they recompiled — and burned a fresh executable page — on EVERY closure creation), are
5277
+ /// invalidated at a program boundary (`reset_callees`), and a cache hit on the callee replays
5278
+ /// its registration so a re-run of the same program resolves cross-calls again.
5279
+ /// NOTE: like [`jit_cross_function_call_matches_closed_form`], this touches the process-global
5280
+ /// callee registry and assumes no concurrent `reset_callees` (nextest isolates per process).
5281
+ #[test]
5282
+ fn jit_xcall_cached_per_generation_and_replays_registration() {
5283
+ // Unique global names — the registry is shared with any parallel test in this process.
5284
+ // Caller shape mirrors [`jit_cross_function_call_matches_closed_form`] (the proven
5285
+ // xcall-compilable shape): sum_{i<n} sq703g(i) with sq703g(x) = 31.5x ⇒ 31.5·n(n-1)/2.
5286
+ let src = "function sq703g(x) { return x * 31.5 }\n\
5287
+ function call703g(n) {\n\
5288
+ let s = 0\n\
5289
+ let i = 0\n\
5290
+ while (i < n) { s = s + sq703g(i); i = i + 1 }\n\
5291
+ return s\n\
5292
+ }\n\
5293
+ call703g(0)\n";
5294
+ let prog = tishlang_parser::parse(src).expect("parse");
5295
+ let opt = tishlang_opt::optimize(&prog);
5296
+ let top = tishlang_bytecode::compile(&opt).expect("compile");
5297
+ let find = |name: &str| {
5298
+ top.nested
5299
+ .iter()
5300
+ .find(|n| n.global_name.as_deref() == Some(name))
5301
+ .unwrap_or_else(|| panic!("no top-level chunk named {name}"))
5302
+ .clone()
5303
+ };
5304
+ let sq = find("sq703g");
5305
+ let caller = find("call703g");
5306
+
5307
+ try_compile_numeric(&sq).expect("callee must compile (and register)");
5308
+ let c1 = caller.clone();
5309
+ let f1 = try_compile_numeric(&c1).expect("cross-caller must compile");
5310
+ assert!(f1.uses_xcall, "caller embeds a native call to sq703g");
5311
+ let c2 = caller.clone();
5312
+ let f2 = try_compile_numeric(&c2).expect("cross-caller must hit the cache");
5313
+ assert_eq!(
5314
+ f2.ptr, f1.ptr,
5315
+ "xcall entry must be cached within one generation"
5316
+ );
5317
+ assert_eq!(f1.call(&[4.0]), 31.5 * (4.0 * 3.0) / 2.0); // 31.5·n(n-1)/2 for n=4
5318
+
5319
+ // Program boundary: the generation bump must invalidate the cached cross-caller. With no
5320
+ // callee registered in the new generation the chunk can't compile at all (LoadVar bails) —
5321
+ // proving the stale-generation entry was NOT returned.
5322
+ reset_callees();
5323
+ let c3 = caller.clone();
5324
+ assert!(
5325
+ try_compile_numeric(&c3).is_none(),
5326
+ "stale-generation xcall entry must miss after reset_callees"
5327
+ );
5328
+ // A cache HIT on the callee must replay its registration into the new generation; the
5329
+ // registry growth then retries the caller's negative entry, which recompiles as xcall.
5330
+ try_compile_numeric(&sq).expect("callee hit must still return its cached compile");
5331
+ let c4 = caller.clone();
5332
+ let f4 =
5333
+ try_compile_numeric(&c4).expect("caller must recompile once the callee re-registers");
5334
+ assert!(f4.uses_xcall);
5335
+ assert_eq!(f4.call(&[4.0]), 31.5 * (4.0 * 3.0) / 2.0);
5336
+ }
4931
5337
  }