@tishlang/tish-format 3.10.4 → 3.10.11
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/tish-format +0 -0
- package/crates/tish/src/cargo_native_registry.rs +4 -0
- package/crates/tish/src/main.rs +65 -25
- package/crates/tish/tests/integration_test.rs +3 -0
- package/crates/tish_builtins/src/globals.rs +3 -3
- package/crates/tish_builtins/src/symbol.rs +86 -4
- package/crates/tish_bytecode/src/compiler.rs +65 -25
- package/crates/tish_bytecode/tests/captured_param_slot_binding.rs +131 -0
- package/crates/tish_bytecode/tests/fundecl_slot_binding.rs +115 -0
- package/crates/tish_bytecode/tests/param_defaults_slot_binding.rs +86 -0
- package/crates/tish_core/src/shape.rs +55 -2
- package/crates/tish_core/src/value.rs +229 -13
- package/crates/tish_eval/src/timers.rs +66 -2
- package/crates/tish_ffi/src/lib.rs +74 -12
- package/crates/tish_ffi/tests/loader.rs +35 -0
- package/crates/tish_fmt/Cargo.toml +1 -1
- package/crates/tish_lsp/src/main.rs +91 -2
- package/crates/tish_pg/src/lib.rs +59 -8
- package/crates/tish_pg/src/statement_registry.rs +250 -0
- package/crates/tish_runtime/src/http.rs +19 -0
- package/crates/tish_runtime/src/http_fetch.rs +159 -65
- package/crates/tish_runtime/src/http_hyper.rs +18 -2
- package/crates/tish_runtime/src/lib.rs +4 -0
- package/crates/tish_runtime/src/net.rs +125 -97
- package/crates/tish_runtime/src/process_spawn.rs +326 -113
- package/crates/tish_runtime/src/promise.rs +521 -90
- package/crates/tish_runtime/src/promise_io.rs +29 -6
- package/crates/tish_runtime/src/pty.rs +161 -107
- package/crates/tish_runtime/src/stream_buf.rs +419 -0
- package/crates/tish_runtime/src/ws.rs +431 -52
- package/crates/tish_vm/src/jit.rs +473 -67
- package/crates/tish_vm/src/lib.rs +6 -0
- package/crates/tish_vm/src/vm.rs +8 -5
- package/crates/tish_vm/tests/captured_param_slot_mode.rs +210 -0
- package/crates/tish_vm/tests/captured_param_vm_slots_off.rs +64 -0
- package/crates/tish_vm/tests/fundecl_slot_no_frame_cycle.rs +152 -0
- package/crates/tish_vm/tests/param_defaults_slot_mode.rs +126 -0
- package/crates/tish_vm/tests/param_defaults_vm_slots_off.rs +44 -0
- package/package.json +1 -1
- package/platform/darwin-arm64/tish-fmt +0 -0
- package/platform/darwin-x64/tish-fmt +0 -0
- package/platform/linux-arm64/tish-fmt +0 -0
- package/platform/linux-x64/tish-fmt +0 -0
- package/platform/win32-x64/tish-fmt.exe +0 -0
|
@@ -129,10 +129,11 @@ pub struct NumericFn {
|
|
|
129
129
|
/// [`NumericFn::call`] ABI; an out-of-bounds array access (or a non-numeric return) sets a
|
|
130
130
|
/// per-thread deopt flag ([`jv_take_deopt`]) and the caller discards the result + re-interprets.
|
|
131
131
|
jv: bool,
|
|
132
|
-
/// #187: true when this function embeds a native call to a registered callee.
|
|
133
|
-
///
|
|
134
|
-
///
|
|
135
|
-
///
|
|
132
|
+
/// #187: true when this function embeds a native call to a registered callee. Its cache entry is
|
|
133
|
+
/// scoped to the callee-registry GENERATION it compiled under (#703, [`JitGlobal::callees_gen`]):
|
|
134
|
+
/// within one program run the callee binding is proven stable, so the entry is reused; after a
|
|
135
|
+
/// program boundary ([`reset_callees`]) the entry is a miss and the function recompiles once
|
|
136
|
+
/// against the live registry — never against a stale callee.
|
|
136
137
|
uses_xcall: bool,
|
|
137
138
|
/// #187: true when this is a VOID array-mode function (only returns the implicit `null`). Its
|
|
138
139
|
/// `f64` result is a dummy, so [`try_call_array_jit`] returns `Value::Null` instead of a number.
|
|
@@ -520,19 +521,24 @@ impl NumericFn {
|
|
|
520
521
|
|
|
521
522
|
struct JitGlobal {
|
|
522
523
|
module: JITModule,
|
|
523
|
-
/// Keyed by
|
|
524
|
-
///
|
|
525
|
-
///
|
|
526
|
-
///
|
|
527
|
-
///
|
|
528
|
-
///
|
|
529
|
-
///
|
|
530
|
-
///
|
|
531
|
-
|
|
532
|
-
///
|
|
533
|
-
///
|
|
534
|
-
|
|
535
|
-
|
|
524
|
+
/// Keyed by **content identity** — the primary [`chunk_fingerprints`] hash — NOT by chunk
|
|
525
|
+
/// address (#703). The VM deep-clones a `Chunk` for every closure instance (`vm.rs`
|
|
526
|
+
/// `LoadConst(Closure)`), so an address key is per closure *instance*, not per program function:
|
|
527
|
+
/// every qualifying closure creation minted a fresh cranelift compile whose sealed executable
|
|
528
|
+
/// page (16 KB on macOS arm64) is never freed — ~18 KB leaked per closure creation, driven by
|
|
529
|
+
/// runtime event volume rather than code size. Content keys make every instance of the same
|
|
530
|
+
/// source function share ONE compile, so the map grows with distinct program text only. A hit is
|
|
531
|
+
/// honored only when the entry's [`CacheTail`] also matches (second hash + exact structural
|
|
532
|
+
/// dims), so a 64-bit key collision degrades to a recompile, never a miscompile. `None` still
|
|
533
|
+
/// caches "not JIT-eligible"; registry-sensitive entries are re-validated against
|
|
534
|
+
/// `callees_gen`/`callees` — see [`NumEntry`].
|
|
535
|
+
cache: HashMap<u64, NumEntry>,
|
|
536
|
+
/// OSR loop-region cache (#190), keyed by `(content fingerprint, loop header ip)` with the same
|
|
537
|
+
/// [`CacheTail`] collision guard as `cache` (#703 — same per-instance-address leak otherwise).
|
|
538
|
+
/// `None` caches "region not compilable" so a loop that fails the whitelist is scanned once, not
|
|
539
|
+
/// on every back-edge past the trigger threshold. Loop regions are always registry-independent
|
|
540
|
+
/// (the region whitelist rejects `LoadVar`), so entries never need generation re-validation.
|
|
541
|
+
osr_cache: HashMap<(u64, usize), OsrEntry>,
|
|
536
542
|
counter: usize,
|
|
537
543
|
/// `FuncId` of the imported `tish_math_call` host fn (#186), declared once at module init and
|
|
538
544
|
/// re-imported into each compiled function via `declare_func_in_func`.
|
|
@@ -548,17 +554,142 @@ struct JitGlobal {
|
|
|
548
554
|
/// the binding can never change under a cached caller. A caller that references a name NOT yet here
|
|
549
555
|
/// (a forward reference) simply bails to the interpreter.
|
|
550
556
|
callees: HashMap<Arc<str>, CalleeEntry>,
|
|
557
|
+
/// #703: generation counter for `callees`, bumped by [`reset_callees`] at every top-level program
|
|
558
|
+
/// boundary. A cached cross-calling function ([`NumericFn::uses_xcall`]) embeds native calls to
|
|
559
|
+
/// callee ids it resolved at compile time; that binding is proven stable only WITHIN one program
|
|
560
|
+
/// run (`global_name` is per-program-stable), so such an entry is honored only while the
|
|
561
|
+
/// generation it was compiled under is still current. This is what lets cross-callers be cached
|
|
562
|
+
/// at all (pre-#703 they recompiled — and burned a fresh executable page — on every closure
|
|
563
|
+
/// creation) while preserving the original soundness argument: a stale callee is never invoked.
|
|
564
|
+
callees_gen: u64,
|
|
551
565
|
}
|
|
552
566
|
|
|
553
567
|
/// #187: a registered directly-callable numeric callee (register-`f64` ABI). Callers resolve against
|
|
554
|
-
/// the LIVE registry at compile time
|
|
555
|
-
///
|
|
568
|
+
/// the LIVE registry at compile time; their cache entries are scoped to the registry generation they
|
|
569
|
+
/// compiled under (#703, see [`JitGlobal::callees_gen`]), so a name re-registered by a later program
|
|
570
|
+
/// simply overwrites this — a stale callee is never invoked.
|
|
556
571
|
#[derive(Clone, Copy)]
|
|
557
572
|
struct CalleeEntry {
|
|
558
573
|
id: cranelift_module::FuncId,
|
|
559
574
|
arity: u8,
|
|
560
575
|
}
|
|
561
576
|
|
|
577
|
+
/// Collision-verification tail stored with every content-keyed cache entry (#703). The map key is a
|
|
578
|
+
/// single 64-bit fingerprint; unlike the old address-keyed scheme (where a wrong hit only meant a
|
|
579
|
+
/// freed-and-reused address), a false content hit would hand one chunk another chunk's native code —
|
|
580
|
+
/// a miscompile. A hit is therefore honored only when a SECOND, independently-mixed 64-bit
|
|
581
|
+
/// fingerprint over the same input AND the exact structural dimensions all match. Two chunks that
|
|
582
|
+
/// agree on both hashes (independent multipliers/avalanches over identical input streams) and every
|
|
583
|
+
/// length/shape field below are identical for compilation purposes to ~2^-128 confidence — stronger
|
|
584
|
+
/// in practice, since the inputs are compiler-generated bytecode, not adversarial hash-seeking data.
|
|
585
|
+
/// A mismatch is treated as a miss: recompile and overwrite (the superseded entry's code page stays
|
|
586
|
+
/// mapped, as all JIT pages do — see the finalize note in [`compile_chunk`]).
|
|
587
|
+
#[derive(Clone, Copy, PartialEq, Eq)]
|
|
588
|
+
struct CacheTail {
|
|
589
|
+
fp2: u64,
|
|
590
|
+
code_len: u32,
|
|
591
|
+
const_len: u32,
|
|
592
|
+
param_count: u16,
|
|
593
|
+
num_slots: u16,
|
|
594
|
+
}
|
|
595
|
+
|
|
596
|
+
impl CacheTail {
|
|
597
|
+
fn of(chunk: &Chunk, fp2: u64) -> Self {
|
|
598
|
+
Self {
|
|
599
|
+
fp2,
|
|
600
|
+
code_len: chunk.code.len() as u32,
|
|
601
|
+
const_len: chunk.constants.len() as u32,
|
|
602
|
+
param_count: chunk.param_count,
|
|
603
|
+
num_slots: chunk.num_slots,
|
|
604
|
+
}
|
|
605
|
+
}
|
|
606
|
+
|
|
607
|
+
fn matches(&self, chunk: &Chunk, fp2: u64) -> bool {
|
|
608
|
+
*self == Self::of(chunk, fp2)
|
|
609
|
+
}
|
|
610
|
+
}
|
|
611
|
+
|
|
612
|
+
/// A `cache` entry (#703): the compile result plus everything needed to re-validate and to replay
|
|
613
|
+
/// the compile's side effects on a content-keyed hit.
|
|
614
|
+
struct NumEntry {
|
|
615
|
+
tail: CacheTail,
|
|
616
|
+
result: Option<NumericFn>,
|
|
617
|
+
/// `Some` when compiling this chunk registered it as a directly-callable callee (#187: plain
|
|
618
|
+
/// register-`f64`, non-jv/non-guarded/non-bool, with a stable `global_name`). A cache hit must
|
|
619
|
+
/// REPLAY that registration — re-inserting the id under the current chunk's `global_name` — or a
|
|
620
|
+
/// program re-run in a long-lived process (its registry cleared by [`reset_callees`], its chunks
|
|
621
|
+
/// all cache hits) would never re-populate the registry and cross-function JIT calls would
|
|
622
|
+
/// silently stop resolving. The finalized id's code pointer is process-permanent, so replaying it
|
|
623
|
+
/// into any later generation is sound.
|
|
624
|
+
callee_id: Option<cranelift_module::FuncId>,
|
|
625
|
+
/// `callees` generation this entry was compiled under (see [`JitGlobal::callees_gen`]).
|
|
626
|
+
callees_gen: u64,
|
|
627
|
+
/// `callees.len()` right after this compile. Consulted only for registry-sensitive `None`
|
|
628
|
+
/// entries: a chunk that references globals may have failed to compile *because* a callee wasn't
|
|
629
|
+
/// registered yet, so registry growth within the generation retries it (once per growth step —
|
|
630
|
+
/// the registry only grows within a generation, so steady state re-hits the cached `None`).
|
|
631
|
+
callees_len: u32,
|
|
632
|
+
/// Whether the chunk references any global (`LoadVar`), i.e. whether its compile RESULT could
|
|
633
|
+
/// depend on the callee registry at all. `false` ⇒ the entry is valid regardless of registry
|
|
634
|
+
/// state or generation (a compiled non-xcall body provably contains no `LoadVar` — an unresolved
|
|
635
|
+
/// one bails compilation and a resolved one makes it xcall).
|
|
636
|
+
registry_sensitive: bool,
|
|
637
|
+
}
|
|
638
|
+
|
|
639
|
+
/// An `osr_cache` entry (#703). Loop regions never consult the callee registry (`LoadVar` is outside
|
|
640
|
+
/// the region whitelist), so only the collision tail — plus the region bounds — needs re-validation.
|
|
641
|
+
struct OsrEntry {
|
|
642
|
+
tail: CacheTail,
|
|
643
|
+
/// End of the compiled region. The key carries only `(fingerprint, header_ip)` (matching the old
|
|
644
|
+
/// address-based key's assumption that one header identifies one region); storing the end and
|
|
645
|
+
/// checking it on hit turns any violation of that assumption into a recompile, not a wrong region.
|
|
646
|
+
region_end: usize,
|
|
647
|
+
result: Option<LoopFn>,
|
|
648
|
+
}
|
|
649
|
+
|
|
650
|
+
/// Defensive bound on the JIT cache maps (`cache`, `osr_cache`, and the per-thread
|
|
651
|
+
/// [`OSR_EXPAND_CACHE`] memo), overridable via `TISH_JIT_CACHE_CAP` (`0` ⇒ unbounded). With content
|
|
652
|
+
/// keys the maps grow with DISTINCT program text, so a normal workload never approaches the cap; it
|
|
653
|
+
/// exists so a degenerate embedder (eval-ing freshly generated program text per event, say) cannot
|
|
654
|
+
/// grow the maps without bound. On overflow the code caches STOP CACHING new entries — they never
|
|
655
|
+
/// evict: an evicted entry's executable page may be running on another thread and could not be
|
|
656
|
+
/// unmapped anyway (pages are process-permanent), so eviction would only trade a map entry for a
|
|
657
|
+
/// fresh page-burning recompile on the next lookup. The expand memo (pure re-computable analysis,
|
|
658
|
+
/// no pages) clears instead.
|
|
659
|
+
fn cache_cap() -> usize {
|
|
660
|
+
static CAP: OnceLock<usize> = OnceLock::new();
|
|
661
|
+
*CAP.get_or_init(|| {
|
|
662
|
+
std::env::var("TISH_JIT_CACHE_CAP")
|
|
663
|
+
.ok()
|
|
664
|
+
.and_then(|v| v.parse::<usize>().ok())
|
|
665
|
+
.unwrap_or(65_536)
|
|
666
|
+
})
|
|
667
|
+
}
|
|
668
|
+
|
|
669
|
+
/// Does `chunk` reference any global name (a `LoadVar` op)? Such a chunk's compile result depends on
|
|
670
|
+
/// the live callee registry — `LoadVar` of a registered name lowers to a native call, of an
|
|
671
|
+
/// unregistered one bails compilation — so its cache entry needs registry re-validation (#703).
|
|
672
|
+
/// Walks with `instruction_size` so an operand byte can't be mistaken for an opcode; malformed code
|
|
673
|
+
/// is conservatively "sensitive".
|
|
674
|
+
fn chunk_references_globals(chunk: &Chunk) -> bool {
|
|
675
|
+
let code = &chunk.code;
|
|
676
|
+
let mut ip = 0usize;
|
|
677
|
+
while ip < code.len() {
|
|
678
|
+
let op = match Opcode::from_u8(code[ip]) {
|
|
679
|
+
Some(o) => o,
|
|
680
|
+
None => return true,
|
|
681
|
+
};
|
|
682
|
+
if op == Opcode::LoadVar {
|
|
683
|
+
return true;
|
|
684
|
+
}
|
|
685
|
+
ip += match op.instruction_size(code, ip) {
|
|
686
|
+
Some(s) => s,
|
|
687
|
+
None => return true,
|
|
688
|
+
};
|
|
689
|
+
}
|
|
690
|
+
false
|
|
691
|
+
}
|
|
692
|
+
|
|
562
693
|
// SAFETY: `JITModule` is `!Send`, but the single instance lives behind the
|
|
563
694
|
// `Mutex` in the process-global `JIT` and is never moved out or dropped; all
|
|
564
695
|
// access is serialized by the mutex.
|
|
@@ -594,6 +725,13 @@ thread_local! {
|
|
|
594
725
|
/// times (e.g. nbody's `advance`), where recomputing per call is a real regression. A stale entry
|
|
595
726
|
/// (a freed chunk's address reused) can only mis-route a perf hint: `run_osr`/`compile_loop_region`
|
|
596
727
|
/// re-validate the region structurally + by fingerprint, so a wrong hint never miscompiles.
|
|
728
|
+
///
|
|
729
|
+
/// #703: this memo stays ADDRESS-keyed — it sits on the per-back-edge hot path of every hot loop,
|
|
730
|
+
/// where an O(chunk) content hash per lookup would be a real regression — but with per-closure-
|
|
731
|
+
/// instance chunk clones an address key grows one ~64 B entry per instance, so [`osr_expand_cached`]
|
|
732
|
+
/// bounds it: at [`cache_cap`] entries the map is CLEARED and rebuilt. Clearing is safe precisely
|
|
733
|
+
/// because entries hold no code pages — they are pure loop-structure analysis, recomputed on the
|
|
734
|
+
/// next back-edge past the threshold.
|
|
597
735
|
static OSR_EXPAND_CACHE: std::cell::RefCell<HashMap<(usize, usize), (usize, usize, bool, bool)>> =
|
|
598
736
|
std::cell::RefCell::new(HashMap::new());
|
|
599
737
|
}
|
|
@@ -852,6 +990,7 @@ fn jit() -> Option<&'static Mutex<JitGlobal>> {
|
|
|
852
990
|
math_binary_call_id,
|
|
853
991
|
jv_fns,
|
|
854
992
|
callees: HashMap::new(),
|
|
993
|
+
callees_gen: 0,
|
|
855
994
|
})
|
|
856
995
|
})
|
|
857
996
|
})
|
|
@@ -867,25 +1006,35 @@ fn read_u16(code: &[u8], ip: &mut usize) -> Option<u16> {
|
|
|
867
1006
|
Some((a << 8) | b)
|
|
868
1007
|
}
|
|
869
1008
|
|
|
870
|
-
/// Content
|
|
871
|
-
///
|
|
872
|
-
///
|
|
873
|
-
///
|
|
874
|
-
///
|
|
875
|
-
///
|
|
876
|
-
///
|
|
877
|
-
|
|
878
|
-
|
|
879
|
-
|
|
880
|
-
|
|
1009
|
+
/// Content fingerprints of everything `compile_chunk` reads — `(primary, secondary)`. The primary
|
|
1010
|
+
/// hash is the cache KEY (#703: the caches are keyed on content identity, not chunk address, because
|
|
1011
|
+
/// the VM deep-clones a `Chunk` per closure instance); the secondary goes into the entry's
|
|
1012
|
+
/// [`CacheTail`] so a primary-key collision is detected as a miss rather than becoming a miscompile.
|
|
1013
|
+
/// Covers the compile-relevant fields: shape (`param_count`, `num_slots`, `rest_param_index`,
|
|
1014
|
+
/// `slot_based`), the full `code` bytes, the `constants` (the JIT emits `f64const`/bool from
|
|
1015
|
+
/// `LoadConst`, so their values matter), the `names` table (JV member lowering dispatches on
|
|
1016
|
+
/// `"length"`/`"push"` by name, and #187 `LoadVar` callee resolution is by name), and `global_name`
|
|
1017
|
+
/// (it drives the callee-registration side effect a hit replays). Together with the callee-registry
|
|
1018
|
+
/// state re-validated per entry (see [`NumEntry`]) and the process-stable `OnceLock` env flags, two
|
|
1019
|
+
/// chunks with equal fingerprints are interchangeable compile inputs. Deterministic within a process
|
|
1020
|
+
/// (fixed constants, not a randomized hasher), which is all the caches need.
|
|
1021
|
+
fn chunk_fingerprints(chunk: &Chunk) -> (u64, u64) {
|
|
1022
|
+
// Mixes a u64 at a time into TWO accumulators in one pass. h1 is the original FNV-1a-style mix
|
|
1023
|
+
// (FNV prime + avalanche shift); h2 uses a different offset basis, a golden-ratio pre-add, the
|
|
1024
|
+
// murmur3-finalizer multiplier and a different shift — an independently-mixed hash family, so a
|
|
1025
|
+
// simultaneous collision of both over identical-length inputs is ~2^-128. Eight bytes per round
|
|
1026
|
+
// keeps this cheap on the hot closure-creation path.
|
|
881
1027
|
#[inline]
|
|
882
|
-
fn mix(h: &mut u64, v: u64) {
|
|
883
|
-
|
|
884
|
-
|
|
885
|
-
|
|
1028
|
+
fn mix(h: &mut (u64, u64), v: u64) {
|
|
1029
|
+
h.0 ^= v;
|
|
1030
|
+
h.0 = h.0.wrapping_mul(0x0000_0100_0000_01b3);
|
|
1031
|
+
h.0 ^= h.0 >> 29;
|
|
1032
|
+
h.1 ^= v.wrapping_add(0x9e37_79b9_7f4a_7c15);
|
|
1033
|
+
h.1 = h.1.wrapping_mul(0xff51_afd7_ed55_8ccd);
|
|
1034
|
+
h.1 ^= h.1 >> 33;
|
|
886
1035
|
}
|
|
887
1036
|
#[inline]
|
|
888
|
-
fn mix_bytes(h: &mut u64, bytes: &[u8]) {
|
|
1037
|
+
fn mix_bytes(h: &mut (u64, u64), bytes: &[u8]) {
|
|
889
1038
|
let mut it = bytes.chunks_exact(8);
|
|
890
1039
|
for w in &mut it {
|
|
891
1040
|
mix(h, u64::from_le_bytes(w.try_into().unwrap()));
|
|
@@ -898,7 +1047,7 @@ fn chunk_fingerprint(chunk: &Chunk) -> u64 {
|
|
|
898
1047
|
}
|
|
899
1048
|
mix(h, bytes.len() as u64);
|
|
900
1049
|
}
|
|
901
|
-
let mut h: u64 = 0xcbf2_9ce4_8422_2325;
|
|
1050
|
+
let mut h: (u64, u64) = (0xcbf2_9ce4_8422_2325, 0x6a09_e667_f3bc_c908);
|
|
902
1051
|
mix(&mut h, chunk.param_count as u64);
|
|
903
1052
|
mix(&mut h, chunk.num_slots as u64);
|
|
904
1053
|
mix(&mut h, chunk.rest_param_index as u64);
|
|
@@ -923,6 +1072,17 @@ fn chunk_fingerprint(chunk: &Chunk) -> u64 {
|
|
|
923
1072
|
}
|
|
924
1073
|
}
|
|
925
1074
|
}
|
|
1075
|
+
mix(&mut h, chunk.names.len() as u64);
|
|
1076
|
+
for n in &chunk.names {
|
|
1077
|
+
mix_bytes(&mut h, n.as_bytes());
|
|
1078
|
+
}
|
|
1079
|
+
match &chunk.global_name {
|
|
1080
|
+
Some(n) => {
|
|
1081
|
+
mix(&mut h, 7);
|
|
1082
|
+
mix_bytes(&mut h, n.as_bytes());
|
|
1083
|
+
}
|
|
1084
|
+
None => mix(&mut h, 8),
|
|
1085
|
+
}
|
|
926
1086
|
h
|
|
927
1087
|
}
|
|
928
1088
|
|
|
@@ -930,17 +1090,31 @@ fn chunk_fingerprint(chunk: &Chunk) -> u64 {
|
|
|
930
1090
|
/// Returns `None` if the chunk isn't a straight-line numeric function.
|
|
931
1091
|
/// #187: clear the directly-callable-callee registry at the start of each top-level program run, so a
|
|
932
1092
|
/// long-lived process (REPL / embedder) never resolves a callee registered by a PRIOR program (a name
|
|
933
|
-
/// re-registered non-numerically would otherwise leave a stale native entry).
|
|
934
|
-
///
|
|
1093
|
+
/// re-registered non-numerically would otherwise leave a stale native entry). #703: also bump the
|
|
1094
|
+
/// registry GENERATION — cached cross-callers ([`NumericFn::uses_xcall`]) are valid only within the
|
|
1095
|
+
/// generation they resolved their callees under, so after this every cross-caller re-resolves against
|
|
1096
|
+
/// the freshly-populated registry (by recompiling once, not once per closure creation).
|
|
935
1097
|
#[cfg(not(target_arch = "wasm32"))]
|
|
936
1098
|
pub fn reset_callees() {
|
|
937
1099
|
if let Some(lock) = jit() {
|
|
938
1100
|
if let Ok(mut g) = lock.lock() {
|
|
939
1101
|
g.callees.clear();
|
|
1102
|
+
g.callees_gen = g.callees_gen.wrapping_add(1);
|
|
940
1103
|
}
|
|
941
1104
|
}
|
|
942
1105
|
}
|
|
943
1106
|
|
|
1107
|
+
/// Test/diagnostic introspection (#703): entry counts of the process-global JIT caches,
|
|
1108
|
+
/// `(cache, osr_cache)`. The regression tests assert these stay flat while the same closure body is
|
|
1109
|
+
/// instantiated (and its per-instance chunk clone compiled) many times over.
|
|
1110
|
+
#[doc(hidden)]
|
|
1111
|
+
pub fn jit_cache_lens() -> (usize, usize) {
|
|
1112
|
+
match jit().map(|lock| lock.lock()) {
|
|
1113
|
+
Some(Ok(g)) => (g.cache.len(), g.osr_cache.len()),
|
|
1114
|
+
_ => (0, 0),
|
|
1115
|
+
}
|
|
1116
|
+
}
|
|
1117
|
+
|
|
944
1118
|
pub fn try_compile_numeric(chunk: &Chunk) -> Option<NumericFn> {
|
|
945
1119
|
if !chunk.slot_based
|
|
946
1120
|
|| chunk.rest_param_index != NO_REST_PARAM
|
|
@@ -949,23 +1123,77 @@ pub fn try_compile_numeric(chunk: &Chunk) -> Option<NumericFn> {
|
|
|
949
1123
|
{
|
|
950
1124
|
return None;
|
|
951
1125
|
}
|
|
952
|
-
let
|
|
953
|
-
let fp = chunk_fingerprint(chunk);
|
|
1126
|
+
let (fp, fp2) = chunk_fingerprints(chunk);
|
|
954
1127
|
let lock = jit()?;
|
|
955
1128
|
let mut g = lock.lock().ok()?;
|
|
956
|
-
|
|
957
|
-
|
|
958
|
-
if let Some(
|
|
959
|
-
|
|
960
|
-
|
|
1129
|
+
let gen = g.callees_gen;
|
|
1130
|
+
let live_callees = g.callees.len() as u32;
|
|
1131
|
+
if let Some(entry) = g.cache.get(&fp) {
|
|
1132
|
+
// The tail must match — a bare 64-bit key collision would otherwise hand this chunk native
|
|
1133
|
+
// code compiled from a DIFFERENT chunk (see [`CacheTail`]). Tail mismatch ⇒ recompile below
|
|
1134
|
+
// (overwriting the colliding entry).
|
|
1135
|
+
if entry.tail.matches(chunk, fp2) {
|
|
1136
|
+
// Registry re-validation (#703):
|
|
1137
|
+
// * a compiled non-xcall body provably contains no `LoadVar` — valid forever;
|
|
1138
|
+
// * a cross-caller (`uses_xcall`) embeds callee ids proven stable only within its
|
|
1139
|
+
// compile generation — valid while the generation matches;
|
|
1140
|
+
// * a `None` for a globals-referencing chunk may exist only because a callee wasn't
|
|
1141
|
+
// registered yet — valid while the generation AND registry size are unchanged.
|
|
1142
|
+
let valid = match entry.result {
|
|
1143
|
+
Some(nf) if nf.uses_xcall => entry.callees_gen == gen,
|
|
1144
|
+
Some(_) => true,
|
|
1145
|
+
None => {
|
|
1146
|
+
!entry.registry_sensitive
|
|
1147
|
+
|| (entry.callees_gen == gen && entry.callees_len == live_callees)
|
|
1148
|
+
}
|
|
1149
|
+
};
|
|
1150
|
+
if valid {
|
|
1151
|
+
let result = entry.result;
|
|
1152
|
+
let callee_id = entry.callee_id;
|
|
1153
|
+
// Replay the compile's registration side effect: without this, a program re-run in a
|
|
1154
|
+
// long-lived process (registry cleared, every chunk a cache hit) would leave the
|
|
1155
|
+
// registry empty and cross-function JIT calls would silently stop resolving. The id's
|
|
1156
|
+
// finalized code is process-permanent, so re-registering it is always sound;
|
|
1157
|
+
// `global_name` is part of the fingerprint, so it names the same source function.
|
|
1158
|
+
if let (Some(id), Some(name), Some(nf)) =
|
|
1159
|
+
(callee_id, chunk.global_name.as_ref(), result)
|
|
1160
|
+
{
|
|
1161
|
+
g.callees.insert(
|
|
1162
|
+
Arc::clone(name),
|
|
1163
|
+
CalleeEntry {
|
|
1164
|
+
id,
|
|
1165
|
+
arity: nf.arity,
|
|
1166
|
+
},
|
|
1167
|
+
);
|
|
1168
|
+
}
|
|
1169
|
+
return result;
|
|
1170
|
+
}
|
|
961
1171
|
}
|
|
962
1172
|
}
|
|
963
1173
|
let result = compile_chunk(&mut g, chunk);
|
|
964
|
-
//
|
|
965
|
-
//
|
|
966
|
-
|
|
967
|
-
|
|
968
|
-
|
|
1174
|
+
// Recover the callee id `compile_chunk` just registered (if it did — plain register-`f64`
|
|
1175
|
+
// functions with a stable `global_name` only), so a later hit can replay the registration.
|
|
1176
|
+
let callee_id = match (result, chunk.global_name.as_ref()) {
|
|
1177
|
+
(Some(nf), Some(name))
|
|
1178
|
+
if nf.array_param_mask == 0 && !nf.jv && !nf.recur_guarded && !nf.result_bool =>
|
|
1179
|
+
{
|
|
1180
|
+
g.callees.get(name).map(|e| e.id)
|
|
1181
|
+
}
|
|
1182
|
+
_ => None,
|
|
1183
|
+
};
|
|
1184
|
+
let entry = NumEntry {
|
|
1185
|
+
tail: CacheTail::of(chunk, fp2),
|
|
1186
|
+
result,
|
|
1187
|
+
callee_id,
|
|
1188
|
+
callees_gen: gen,
|
|
1189
|
+
callees_len: g.callees.len() as u32,
|
|
1190
|
+
registry_sensitive: chunk_references_globals(chunk),
|
|
1191
|
+
};
|
|
1192
|
+
// #703 defensive bound: at the cap, only overwrites of an existing key land — new entries are
|
|
1193
|
+
// simply not cached (never evict; see [`cache_cap`]).
|
|
1194
|
+
let cap = cache_cap();
|
|
1195
|
+
if cap == 0 || g.cache.len() < cap || g.cache.contains_key(&fp) {
|
|
1196
|
+
g.cache.insert(fp, entry);
|
|
969
1197
|
}
|
|
970
1198
|
result
|
|
971
1199
|
}
|
|
@@ -1080,27 +1308,46 @@ pub fn osr_expand_cached(
|
|
|
1080
1308
|
let has_arrays = osr_region_has_arrays(chunk, th, te);
|
|
1081
1309
|
let array_worthy = has_arrays && !osr_region_enclosed(chunk, th, te);
|
|
1082
1310
|
let v = (th, te, has_arrays, array_worthy);
|
|
1083
|
-
c.borrow_mut()
|
|
1311
|
+
let mut m = c.borrow_mut();
|
|
1312
|
+
// #703: address-keyed per closure instance ⇒ unbounded growth under closure-minting
|
|
1313
|
+
// workloads. Clear-on-full (NOT stop-caching, unlike the code caches): entries are pure
|
|
1314
|
+
// recomputable analysis, so clearing costs one rescan per live hot loop and bounds the map.
|
|
1315
|
+
let cap = cache_cap();
|
|
1316
|
+
if cap != 0 && m.len() >= cap {
|
|
1317
|
+
m.clear();
|
|
1318
|
+
}
|
|
1319
|
+
m.insert(key, v);
|
|
1084
1320
|
v
|
|
1085
1321
|
})
|
|
1086
1322
|
}
|
|
1087
1323
|
|
|
1088
1324
|
/// Compile the hot loop region `[header_ip, region_end)` of `chunk` to native code (#190 OSR), or
|
|
1089
|
-
/// `None` if it is not a pure-numeric slot loop. Cached per `(
|
|
1090
|
-
///
|
|
1091
|
-
///
|
|
1325
|
+
/// `None` if it is not a pure-numeric slot loop. Cached per `(content fingerprint, header_ip)` with
|
|
1326
|
+
/// the [`CacheTail`] collision guard (#703 — every closure instance of the same source function
|
|
1327
|
+
/// shares one compile; negative results included, so a non-compilable loop is scanned once, not once
|
|
1328
|
+
/// per closure instance). Called from the frame VM's `JumpBack` handler once a loop's back-edge
|
|
1329
|
+
/// counter crosses the trigger threshold.
|
|
1092
1330
|
pub fn try_compile_loop(chunk: &Chunk, header_ip: usize, region_end: usize) -> Option<LoopFn> {
|
|
1093
|
-
let
|
|
1094
|
-
let
|
|
1331
|
+
let (fp, fp2) = chunk_fingerprints(chunk);
|
|
1332
|
+
let key = (fp, header_ip);
|
|
1095
1333
|
let lock = jit()?;
|
|
1096
1334
|
let mut g = lock.lock().ok()?;
|
|
1097
|
-
if let Some(
|
|
1098
|
-
if
|
|
1099
|
-
return
|
|
1335
|
+
if let Some(entry) = g.osr_cache.get(&key) {
|
|
1336
|
+
if entry.tail.matches(chunk, fp2) && entry.region_end == region_end {
|
|
1337
|
+
return entry.result.clone();
|
|
1100
1338
|
}
|
|
1101
1339
|
}
|
|
1102
1340
|
let result = compile_loop_region(&mut g, chunk, header_ip, region_end);
|
|
1103
|
-
|
|
1341
|
+
let entry = OsrEntry {
|
|
1342
|
+
tail: CacheTail::of(chunk, fp2),
|
|
1343
|
+
region_end,
|
|
1344
|
+
result: result.clone(),
|
|
1345
|
+
};
|
|
1346
|
+
// #703 defensive bound — same policy as the numeric cache: at the cap, overwrite-only.
|
|
1347
|
+
let cap = cache_cap();
|
|
1348
|
+
if cap == 0 || g.osr_cache.len() < cap || g.osr_cache.contains_key(&key) {
|
|
1349
|
+
g.osr_cache.insert(key, entry);
|
|
1350
|
+
}
|
|
1104
1351
|
result
|
|
1105
1352
|
}
|
|
1106
1353
|
|
|
@@ -2142,6 +2389,16 @@ fn compile_chunk(g: &mut JitGlobal, chunk: &Chunk) -> Option<NumericFn> {
|
|
|
2142
2389
|
return None;
|
|
2143
2390
|
}
|
|
2144
2391
|
g.module.clear_context(&mut ctx);
|
|
2392
|
+
// NOTE(#703 follow-up): one `finalize_definitions` per compiled function seals the memory
|
|
2393
|
+
// provider's current allocation, so every function occupies its own page-aligned executable
|
|
2394
|
+
// allocation (16 KB on macOS arm64, 4 KB on x86-64) that is never unmapped. With the caches
|
|
2395
|
+
// content-keyed, compiles are bounded by distinct program text, so this is a bounded per-function
|
|
2396
|
+
// overhead rather than a leak — but packing functions tighter would need `finalize_definitions`
|
|
2397
|
+
// batched across compiles (the function pointer is needed immediately here, so that is a
|
|
2398
|
+
// call-site restructure), and cranelift's `ArenaMemoryProvider` alone does not help: its
|
|
2399
|
+
// `finalize` marks segments finalized too, so the next allocation opens a fresh page-aligned
|
|
2400
|
+
// segment. A per-program `JITModule` dropped with `free_memory` at teardown is the only way to
|
|
2401
|
+
// actually return code pages to a REPL/multi-program embedder.
|
|
2145
2402
|
if g.module.finalize_definitions().is_err() {
|
|
2146
2403
|
return None;
|
|
2147
2404
|
}
|
|
@@ -4274,9 +4531,9 @@ mod tests {
|
|
|
4274
4531
|
/// none compiles — so a change that makes the JIT silently *stop* compiling the target (the exact
|
|
4275
4532
|
/// "vacuous fixture" miss that motivated this guard) fails loudly instead of passing emptily.
|
|
4276
4533
|
///
|
|
4277
|
-
/// Bypasses [`try_compile_numeric`]'s cache and calls [`compile_chunk`]
|
|
4278
|
-
///
|
|
4279
|
-
///
|
|
4534
|
+
/// Bypasses [`try_compile_numeric`]'s (content-keyed, #703) cache and calls [`compile_chunk`]
|
|
4535
|
+
/// directly, so every fixture exercises the real lowering path fresh instead of possibly
|
|
4536
|
+
/// returning another test's cached compile.
|
|
4280
4537
|
fn jit_arity2(src: &str) -> NumericFn {
|
|
4281
4538
|
let prog = tishlang_parser::parse(src).expect("parse");
|
|
4282
4539
|
let opt = tishlang_opt::optimize(&prog);
|
|
@@ -4310,7 +4567,7 @@ mod tests {
|
|
|
4310
4567
|
}
|
|
4311
4568
|
|
|
4312
4569
|
/// #189: compile the first JV (function-local `f64` array) nested fn in `src` via `compile_chunk`,
|
|
4313
|
-
/// bypassing the
|
|
4570
|
+
/// bypassing the content-keyed cache (see [`jit_arity2`]). Panics if none compiles JV — so a change
|
|
4314
4571
|
/// that makes the classifier or lowering silently stop accepting the target fails loudly.
|
|
4315
4572
|
fn jit_jv(src: &str) -> NumericFn {
|
|
4316
4573
|
let prog = tishlang_parser::parse(src).expect("parse");
|
|
@@ -4775,9 +5032,10 @@ mod tests {
|
|
|
4775
5032
|
}
|
|
4776
5033
|
|
|
4777
5034
|
/// Regression for the address-reuse stale hit (#247): compile one function, then overwrite the SAME
|
|
4778
|
-
/// heap `Chunk` (same address
|
|
4779
|
-
///
|
|
4780
|
-
/// fingerprint
|
|
5035
|
+
/// heap `Chunk` (same address) with a *different* function — what a long-lived process (REPL /
|
|
5036
|
+
/// multi-script embedder) does when a freed chunk address is reused. Originally this exercised the
|
|
5037
|
+
/// fingerprint-on-hit guard over the address-keyed cache; since #703 the cache key IS the content
|
|
5038
|
+
/// fingerprint, so distinct content can never share an entry — kept as a permanent behavior guard.
|
|
4781
5039
|
#[test]
|
|
4782
5040
|
fn jit_cache_detects_address_reuse() {
|
|
4783
5041
|
let mut boxed: Box<Chunk> = Box::new(fn_chunk("const f = (a, b) => a - b\nf(0, 0)\n"));
|
|
@@ -4928,4 +5186,152 @@ mod tests {
|
|
|
4928
5186
|
got.sort_by(|a, b| a.partial_cmp(b).unwrap());
|
|
4929
5187
|
assert_eq!(got, vec![20.0, 90.0], "s=90 (0+2+…+18), i=20");
|
|
4930
5188
|
}
|
|
5189
|
+
|
|
5190
|
+
/// #703 REGRESSION — the leak class: the VM deep-clones a `Chunk` per closure instance, so the
|
|
5191
|
+
/// old address-keyed cache treated every instance as a novel compile — one sealed, never-freed
|
|
5192
|
+
/// executable page each (measured ~18 KB per closure creation with instances retained). Content
|
|
5193
|
+
/// keys must dedupe: N live clones of one chunk (distinct heap addresses, exactly like N live
|
|
5194
|
+
/// closure instances) share ONE compile and ONE cache entry.
|
|
5195
|
+
#[test]
|
|
5196
|
+
fn jit_cache_content_identity_dedupes_cloned_chunks() {
|
|
5197
|
+
// Unique constants ⇒ a fingerprint no other test's chunk shares, so the deltas observed here
|
|
5198
|
+
// are our own even though the whole test binary shares the process-global JIT.
|
|
5199
|
+
let chunk = fn_chunk("function f(a, b) { return a * 703.0625 + b * 1219.5 }\nf(0, 0)\n");
|
|
5200
|
+
let (cache_before, _) = jit_cache_lens();
|
|
5201
|
+
let first = try_compile_numeric(&chunk).expect("numeric fn must compile");
|
|
5202
|
+
let mut keep: Vec<Box<Chunk>> = Vec::new(); // live clones ⇒ malloc can't recycle addresses
|
|
5203
|
+
for _ in 0..200 {
|
|
5204
|
+
let clone = Box::new(chunk.clone());
|
|
5205
|
+
let f = try_compile_numeric(&clone).expect("clone must hit the cache");
|
|
5206
|
+
// Compiled code is finalized at a process-unique, never-freed address, so pointer
|
|
5207
|
+
// equality holds iff the lookup HIT — a recompile would finalize new code elsewhere.
|
|
5208
|
+
assert_eq!(
|
|
5209
|
+
f.ptr, first.ptr,
|
|
5210
|
+
"cloned chunk must reuse the cached compile"
|
|
5211
|
+
);
|
|
5212
|
+
keep.push(clone);
|
|
5213
|
+
}
|
|
5214
|
+
let (cache_after, _) = jit_cache_lens();
|
|
5215
|
+
// Ours is exactly one entry; the slack absorbs unrelated entries from tests running in
|
|
5216
|
+
// parallel in this process. Pre-#703 this loop grew the cache by ~200 (one per clone).
|
|
5217
|
+
assert!(
|
|
5218
|
+
cache_after.saturating_sub(cache_before) < 50,
|
|
5219
|
+
"content-keyed cache must not grow per closure instance ({cache_before} -> {cache_after})"
|
|
5220
|
+
);
|
|
5221
|
+
assert_eq!(first.call(&[2.0, 4.0]), 2.0 * 703.0625 + 4.0 * 1219.5);
|
|
5222
|
+
}
|
|
5223
|
+
|
|
5224
|
+
/// #703 REGRESSION — negative results ("not JIT-eligible") are content-keyed too: N clones of a
|
|
5225
|
+
/// non-numeric body leave one cache entry, not one permanent ~56 B entry per closure instance
|
|
5226
|
+
/// (the issue's fully-dropped variant still leaked partly through these).
|
|
5227
|
+
#[test]
|
|
5228
|
+
fn jit_cache_negative_entries_dedupe() {
|
|
5229
|
+
let chunk = fn_chunk("function g(a, b) { return \"x703\" + a + b }\ng(0, 0)\n");
|
|
5230
|
+
let (before, _) = jit_cache_lens();
|
|
5231
|
+
assert!(
|
|
5232
|
+
try_compile_numeric(&chunk).is_none(),
|
|
5233
|
+
"string body must not JIT"
|
|
5234
|
+
);
|
|
5235
|
+
let mut keep: Vec<Box<Chunk>> = Vec::new();
|
|
5236
|
+
for _ in 0..200 {
|
|
5237
|
+
let clone = Box::new(chunk.clone());
|
|
5238
|
+
assert!(try_compile_numeric(&clone).is_none());
|
|
5239
|
+
keep.push(clone);
|
|
5240
|
+
}
|
|
5241
|
+
let (after, _) = jit_cache_lens();
|
|
5242
|
+
assert!(
|
|
5243
|
+
after.saturating_sub(before) < 50,
|
|
5244
|
+
"negative entries must dedupe by content ({before} -> {after})"
|
|
5245
|
+
);
|
|
5246
|
+
}
|
|
5247
|
+
|
|
5248
|
+
/// #703 REGRESSION — the same dedupe for the OSR loop-region cache: per-instance chunk clones of
|
|
5249
|
+
/// one hot-loop function must share one region compile and one `osr_cache` entry.
|
|
5250
|
+
#[test]
|
|
5251
|
+
fn osr_cache_content_identity_dedupes_cloned_chunks() {
|
|
5252
|
+
let chunk = top_chunk(
|
|
5253
|
+
"let s = 0.0\nlet i = 0.0\nwhile (i < 703.25) { s = s + i * 1.0009765625; i = i + 1.0 }\n",
|
|
5254
|
+
);
|
|
5255
|
+
let (_, osr_before) = jit_cache_lens();
|
|
5256
|
+
let (header, end) = first_region(&chunk);
|
|
5257
|
+
let first = try_compile_loop(&chunk, header, end).expect("numeric loop must OSR-compile");
|
|
5258
|
+
let mut keep: Vec<Box<Chunk>> = Vec::new();
|
|
5259
|
+
for _ in 0..100 {
|
|
5260
|
+
let clone = Box::new(chunk.clone());
|
|
5261
|
+
let lf = try_compile_loop(&clone, header, end).expect("clone must hit the osr cache");
|
|
5262
|
+
assert_eq!(
|
|
5263
|
+
lf.ptr, first.ptr,
|
|
5264
|
+
"cloned chunk must reuse the cached region compile"
|
|
5265
|
+
);
|
|
5266
|
+
keep.push(clone);
|
|
5267
|
+
}
|
|
5268
|
+
let (_, osr_after) = jit_cache_lens();
|
|
5269
|
+
assert!(
|
|
5270
|
+
osr_after.saturating_sub(osr_before) < 50,
|
|
5271
|
+
"content-keyed osr_cache must not grow per closure instance ({osr_before} -> {osr_after})"
|
|
5272
|
+
);
|
|
5273
|
+
}
|
|
5274
|
+
|
|
5275
|
+
/// #703 — cross-callers (`uses_xcall`) are cached WITHIN a callee-registry generation (pre-fix
|
|
5276
|
+
/// they recompiled — and burned a fresh executable page — on EVERY closure creation), are
|
|
5277
|
+
/// invalidated at a program boundary (`reset_callees`), and a cache hit on the callee replays
|
|
5278
|
+
/// its registration so a re-run of the same program resolves cross-calls again.
|
|
5279
|
+
/// NOTE: like [`jit_cross_function_call_matches_closed_form`], this touches the process-global
|
|
5280
|
+
/// callee registry and assumes no concurrent `reset_callees` (nextest isolates per process).
|
|
5281
|
+
#[test]
|
|
5282
|
+
fn jit_xcall_cached_per_generation_and_replays_registration() {
|
|
5283
|
+
// Unique global names — the registry is shared with any parallel test in this process.
|
|
5284
|
+
// Caller shape mirrors [`jit_cross_function_call_matches_closed_form`] (the proven
|
|
5285
|
+
// xcall-compilable shape): sum_{i<n} sq703g(i) with sq703g(x) = 31.5x ⇒ 31.5·n(n-1)/2.
|
|
5286
|
+
let src = "function sq703g(x) { return x * 31.5 }\n\
|
|
5287
|
+
function call703g(n) {\n\
|
|
5288
|
+
let s = 0\n\
|
|
5289
|
+
let i = 0\n\
|
|
5290
|
+
while (i < n) { s = s + sq703g(i); i = i + 1 }\n\
|
|
5291
|
+
return s\n\
|
|
5292
|
+
}\n\
|
|
5293
|
+
call703g(0)\n";
|
|
5294
|
+
let prog = tishlang_parser::parse(src).expect("parse");
|
|
5295
|
+
let opt = tishlang_opt::optimize(&prog);
|
|
5296
|
+
let top = tishlang_bytecode::compile(&opt).expect("compile");
|
|
5297
|
+
let find = |name: &str| {
|
|
5298
|
+
top.nested
|
|
5299
|
+
.iter()
|
|
5300
|
+
.find(|n| n.global_name.as_deref() == Some(name))
|
|
5301
|
+
.unwrap_or_else(|| panic!("no top-level chunk named {name}"))
|
|
5302
|
+
.clone()
|
|
5303
|
+
};
|
|
5304
|
+
let sq = find("sq703g");
|
|
5305
|
+
let caller = find("call703g");
|
|
5306
|
+
|
|
5307
|
+
try_compile_numeric(&sq).expect("callee must compile (and register)");
|
|
5308
|
+
let c1 = caller.clone();
|
|
5309
|
+
let f1 = try_compile_numeric(&c1).expect("cross-caller must compile");
|
|
5310
|
+
assert!(f1.uses_xcall, "caller embeds a native call to sq703g");
|
|
5311
|
+
let c2 = caller.clone();
|
|
5312
|
+
let f2 = try_compile_numeric(&c2).expect("cross-caller must hit the cache");
|
|
5313
|
+
assert_eq!(
|
|
5314
|
+
f2.ptr, f1.ptr,
|
|
5315
|
+
"xcall entry must be cached within one generation"
|
|
5316
|
+
);
|
|
5317
|
+
assert_eq!(f1.call(&[4.0]), 31.5 * (4.0 * 3.0) / 2.0); // 31.5·n(n-1)/2 for n=4
|
|
5318
|
+
|
|
5319
|
+
// Program boundary: the generation bump must invalidate the cached cross-caller. With no
|
|
5320
|
+
// callee registered in the new generation the chunk can't compile at all (LoadVar bails) —
|
|
5321
|
+
// proving the stale-generation entry was NOT returned.
|
|
5322
|
+
reset_callees();
|
|
5323
|
+
let c3 = caller.clone();
|
|
5324
|
+
assert!(
|
|
5325
|
+
try_compile_numeric(&c3).is_none(),
|
|
5326
|
+
"stale-generation xcall entry must miss after reset_callees"
|
|
5327
|
+
);
|
|
5328
|
+
// A cache HIT on the callee must replay its registration into the new generation; the
|
|
5329
|
+
// registry growth then retries the caller's negative entry, which recompiles as xcall.
|
|
5330
|
+
try_compile_numeric(&sq).expect("callee hit must still return its cached compile");
|
|
5331
|
+
let c4 = caller.clone();
|
|
5332
|
+
let f4 =
|
|
5333
|
+
try_compile_numeric(&c4).expect("caller must recompile once the callee re-registers");
|
|
5334
|
+
assert!(f4.uses_xcall);
|
|
5335
|
+
assert_eq!(f4.call(&[4.0]), 31.5 * (4.0 * 3.0) / 2.0);
|
|
5336
|
+
}
|
|
4931
5337
|
}
|