@tishlang/tish-lsp 2.39.0 → 3.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (89) hide show
  1. package/Cargo.toml +4 -1
  2. package/bin/tish-lsp +0 -0
  3. package/crates/tish/src/cli_help.rs +28 -0
  4. package/crates/tish/src/main.rs +41 -5
  5. package/crates/tish/tests/fixtures/cargo_example_project/crates/demo-shim/src/lib.rs +21 -0
  6. package/crates/tish/tests/fixtures/cargo_example_project/crates/demo-shim/tish.d.tish +4 -0
  7. package/crates/tish/tests/fixtures/cargo_example_project/src/main.tish +5 -1
  8. package/crates/tish/tests/regr_typed_externs.rs +59 -0
  9. package/crates/tish/tests/scope_merge.rs +79 -0
  10. package/crates/tish_build_utils/src/lib.rs +27 -1
  11. package/crates/tish_builtins/Cargo.toml +14 -6
  12. package/crates/tish_builtins/src/array.rs +130 -54
  13. package/crates/tish_builtins/src/collections.rs +9 -5
  14. package/crates/tish_builtins/src/construct.rs +7 -1
  15. package/crates/tish_builtins/src/date.rs +15 -1
  16. package/crates/tish_builtins/src/globals.rs +65 -3
  17. package/crates/tish_builtins/src/helpers.rs +7 -3
  18. package/crates/tish_builtins/src/iterator.rs +5 -1
  19. package/crates/tish_builtins/src/lib.rs +3 -0
  20. package/crates/tish_builtins/src/math.rs +14 -0
  21. package/crates/tish_builtins/src/number.rs +72 -6
  22. package/crates/tish_builtins/src/object.rs +6 -1
  23. package/crates/tish_builtins/src/string.rs +13 -2
  24. package/crates/tish_builtins/src/symbol.rs +17 -17
  25. package/crates/tish_builtins/src/typedarrays.rs +9 -2
  26. package/crates/tish_bytecode/src/compiler.rs +76 -2
  27. package/crates/tish_bytecode/src/lib.rs +1 -1
  28. package/crates/tish_bytecode/src/opcode.rs +77 -2
  29. package/crates/tish_compile/src/codegen.rs +2154 -225
  30. package/crates/tish_compile/src/infer.rs +37 -41
  31. package/crates/tish_compile/src/lib.rs +13 -1
  32. package/crates/tish_compile/src/platform_resolve.rs +361 -0
  33. package/crates/tish_compile/src/resolve.rs +195 -11
  34. package/crates/tish_compile/src/schemes.rs +333 -0
  35. package/crates/tish_compile/src/types.rs +205 -34
  36. package/crates/tish_compile/tests/perf_codegen_320.rs +2 -2
  37. package/crates/tish_compile/tests/perf_codegen_module_const_forof.rs +3 -1
  38. package/crates/tish_compile/tests/platform_resolve_cli.rs +164 -0
  39. package/crates/tish_compile/tests/regr_556_param_name_collision.rs +40 -0
  40. package/crates/tish_compile/tests/regr_558_struct_array_write.rs +65 -0
  41. package/crates/tish_compile/tests/regr_558b_struct_field_read.rs +21 -0
  42. package/crates/tish_compile/tests/regr_558c_wrapped_struct_array.rs +64 -0
  43. package/crates/tish_compile/tests/regr_558d_native_vec_mutators.rs +156 -0
  44. package/crates/tish_compile/tests/regr_562_wrapped_struct_array_push.rs +50 -0
  45. package/crates/tish_compile/tests/regr_fixed_native_coercions.rs +69 -0
  46. package/crates/tish_compile/tests/regr_gba_numerics_gated.rs +58 -0
  47. package/crates/tish_compile/tests/regr_gba_struct_reflection.rs +106 -0
  48. package/crates/tish_compile/tests/regression/name_collision_lib.tish +6 -0
  49. package/crates/tish_compile/tests/regression/name_collision_main.tish +13 -0
  50. package/crates/tish_compile/tests/regression/struct_array_write.tish +22 -0
  51. package/crates/tish_compile/tests/regression/struct_field_return_read.tish +10 -0
  52. package/crates/tish_compiler_wasm/Cargo.toml +1 -0
  53. package/crates/tish_compiler_wasm/src/resolve_virtual.rs +56 -6
  54. package/crates/tish_core/Cargo.toml +39 -4
  55. package/crates/tish_core/src/compat.rs +506 -0
  56. package/crates/tish_core/src/console_style.rs +23 -1
  57. package/crates/tish_core/src/json.rs +42 -10
  58. package/crates/tish_core/src/lib.rs +121 -25
  59. package/crates/tish_core/src/macros.rs +1 -1
  60. package/crates/tish_core/src/shape.rs +4 -2
  61. package/crates/tish_core/src/uri.rs +8 -1
  62. package/crates/tish_core/src/value.rs +305 -54
  63. package/crates/tish_core/src/vmref.rs +17 -5
  64. package/crates/tish_eval/Cargo.toml +2 -0
  65. package/crates/tish_eval/src/eval.rs +90 -3
  66. package/crates/tish_eval/src/value.rs +8 -0
  67. package/crates/tish_eval/src/value_convert.rs +316 -21
  68. package/crates/tish_lsp/Cargo.toml +3 -1
  69. package/crates/tish_lsp/src/import_goto.rs +41 -14
  70. package/crates/tish_lsp/src/main.rs +18 -0
  71. package/crates/tish_native/src/build.rs +246 -0
  72. package/crates/tish_native/src/config.rs +20 -0
  73. package/crates/tish_native/src/lib.rs +16 -6
  74. package/crates/tish_parser/src/lib.rs +28 -0
  75. package/crates/tish_parser/src/parser.rs +25 -4
  76. package/crates/tish_runtime/src/lib.rs +55 -9
  77. package/crates/tish_runtime_gba/Cargo.lock +837 -0
  78. package/crates/tish_runtime_gba/Cargo.toml +16 -0
  79. package/crates/tish_runtime_gba/src/gba.rs +211 -0
  80. package/crates/tish_runtime_gba/src/lib.rs +804 -0
  81. package/crates/tish_vm/src/jit.rs +1446 -164
  82. package/crates/tish_vm/src/vm.rs +583 -173
  83. package/justfile +9 -0
  84. package/package.json +1 -1
  85. package/platform/darwin-arm64/tish-lsp +0 -0
  86. package/platform/darwin-x64/tish-lsp +0 -0
  87. package/platform/linux-arm64/tish-lsp +0 -0
  88. package/platform/linux-x64/tish-lsp +0 -0
  89. package/platform/win32-x64/tish-lsp.exe +0 -0
@@ -55,6 +55,12 @@ pub struct LoopFn {
55
55
  /// region). The region always has ≥1 exit (an exit-less region would be an uninterruptible native
56
56
  /// loop, so compilation bails).
57
57
  pub exits: Vec<usize>,
58
+ /// #203 — array live-in slots `(chunk slot, writable)` in marshalling order (ascending slot). EMPTY
59
+ /// ⇒ the pure-numeric 2-pointer ABI `(slots, deopt)` (unchanged). NON-EMPTY ⇒ the 3-pointer array
60
+ /// ABI `(slots, handles, deopt)`: `run_osr` extracts each array live-in to a scratch `Vec<f64>`,
61
+ /// passes it as an [`ArrayHandle`], and (for a `writable` slot) copies the scratch back into the
62
+ /// caller's `Value::Array` — ONLY on a clean, non-deopt exit.
63
+ pub array_slots: Vec<(u16, bool)>,
58
64
  }
59
65
 
60
66
  // SAFETY: identical to `NumericFn` — `ptr` is immutable executable code in the process-global,
@@ -64,17 +70,27 @@ unsafe impl Send for LoopFn {}
64
70
  unsafe impl Sync for LoopFn {}
65
71
 
66
72
  impl LoopFn {
67
- /// Run the region. `buf` holds the live-ins (`used_slots.len()` `f64`s), updated in place with the
68
- /// live-outs on return. `deopt` is a 1-byte flag (unused in v1). Returns the exit id. Safe wrapper
69
- /// — the raw-pointer transmute (same soundness as [`NumericFn::call`]: immutable native code with a
70
- /// fixed C ABI) is confined here, so call sites need no `unsafe`.
73
+ /// Run the region. `buf` holds the numeric live-ins (`used_slots.len()` `f64`s), updated in place
74
+ /// with the live-outs on return. `handles` holds the array live-ins (`array_slots.len()`
75
+ /// [`ArrayHandle`]s, in `array_slots` order) — empty for a pure-numeric region. `deopt` is a
76
+ /// 1-byte flag the region sets on an int-slot live-in miss (#514) or an out-of-bounds array index
77
+ /// (#203); on `deopt != 0` the caller discards `buf` AND the scratch behind `handles` and
78
+ /// re-interprets. Returns the exit id. Safe wrapper — the raw-pointer transmute (same soundness as
79
+ /// [`NumericFn::call`]: immutable native code with a fixed C ABI) is confined here.
71
80
  #[inline]
72
- pub fn call(&self, buf: &mut [f64], deopt: &mut u8) -> i32 {
73
- // SAFETY: `ptr` is immutable executable code compiled for exactly this `(*mut f64, *mut u8)`
74
- // ABI; `buf`/`deopt` are valid for the call and the region only touches `buf[0..used_slots]`.
81
+ pub fn call(&self, buf: &mut [f64], handles: &mut [ArrayHandle], deopt: &mut u8) -> i32 {
82
+ // SAFETY: `ptr` is immutable executable code compiled for exactly this ABI (2-pointer when
83
+ // `array_slots` is empty, else 3-pointer); `buf`/`handles`/`deopt` are valid for the call and
84
+ // the region only touches `buf[0..used_slots]` + the handles' backing scratch.
75
85
  unsafe {
76
- let f: extern "C" fn(*mut f64, *mut u8) -> i32 = std::mem::transmute(self.ptr);
77
- f(buf.as_mut_ptr(), deopt as *mut u8)
86
+ if self.array_slots.is_empty() {
87
+ let f: extern "C" fn(*mut f64, *mut u8) -> i32 = std::mem::transmute(self.ptr);
88
+ f(buf.as_mut_ptr(), deopt as *mut u8)
89
+ } else {
90
+ let f: extern "C" fn(*mut f64, *mut ArrayHandle, *mut u8) -> i32 =
91
+ std::mem::transmute(self.ptr);
92
+ f(buf.as_mut_ptr(), handles.as_mut_ptr(), deopt as *mut u8)
93
+ }
78
94
  }
79
95
  }
80
96
  }
@@ -180,6 +196,30 @@ pub fn jit_jv_enabled() -> bool {
180
196
  })
181
197
  }
182
198
 
199
+ /// #203 — bounded array-index (`arr[i]` read / `arr[i] = v` write) inside an OSR loop **region**
200
+ /// (the matmul lever). A hot loop that indexes a numeric `Value::Array` live-in (created BEFORE the
201
+ /// region, e.g. matmul's `a`/`b`/`c`) is marshalled through the [`ArrayHandle`] inline ABI: each
202
+ /// array live-in is extracted to a scratch `Vec<f64>` and passed as `(ptr,len)`; `GetIndex`/`SetIndex`
203
+ /// lower to a bounds-checked native load/store over a COMPUTED index (`a[i*N+k]`), and writable arrays
204
+ /// are copied back only on a clean (non-deopt) exit. Any out-of-bounds index sets the region deopt
205
+ /// flag and the VM re-interprets from the pristine pre-region state (scratch discarded, real array
206
+ /// untouched) — identical to the entry int-slot guard, so a mid-region bail never commits a partial
207
+ /// result. Additive + bail-safe: a misclassified slot fails the runtime `Value::Array` marshalling
208
+ /// check (→ interpret), and an unhandleable region shape is simply not compiled.
209
+ ///
210
+ /// **Default ON** (validated: matmul 87×, zero perf regressions across the 29-fixture gauntlet, parity
211
+ /// interp==vm==node, all six backends agree, adversarial review found + fixed the one bool-store bug);
212
+ /// `TISH_JIT_OSR_ARRAY=0` disables it (escape hatch).
213
+ #[cfg(not(target_arch = "wasm32"))]
214
+ pub fn osr_array_enabled() -> bool {
215
+ static ENABLED: OnceLock<bool> = OnceLock::new();
216
+ *ENABLED.get_or_init(|| {
217
+ std::env::var("TISH_JIT_OSR_ARRAY")
218
+ .map(|v| v != "0")
219
+ .unwrap_or(true)
220
+ })
221
+ }
222
+
183
223
  /// Boolean scalar local slots in the numeric CFG JIT (#187). **Default ON**; `TISH_JIT_BOOL_SLOTS=0`
184
224
  /// disables it. A `let flag = false` / `flag = true` / `if (flag)` local is represented as an `f64`
185
225
  /// `0.0`/`1.0`; a syntactic pre-pass ([`classify_bool_slots`]) tags the slots, and the equality
@@ -194,6 +234,26 @@ pub fn jit_bool_slots_enabled() -> bool {
194
234
  })
195
235
  }
196
236
 
237
+ /// #168 int-typed slots. **Default ON**; `TISH_JIT_INT_SLOTS=0` disables (falls back to f64
238
+ /// slots — the pre-#168 behavior, byte-identical results).
239
+ ///
240
+ /// A local whose every store is a bitwise/shift result (or an integral constant) lives in an
241
+ /// `i64` Variable holding the EXACT JS number ([`Repr::I64Num`]), so an integer hash/PRNG
242
+ /// accumulator (`h = ((h << 13) | (h >>> 19)) >>> 0; h = h ^ …`) stays in integer registers
243
+ /// ACROSS statements instead of paying an f64 materialize at every store plus a `ToInt32`
244
+ /// re-derivation at every load. A syntactic pre-pass ([`classify_int_slots`]) tags the slots;
245
+ /// the StoreLocal translator bails compilation on any store the pre-pass mis-tagged (a
246
+ /// non-integer value reaching an int slot), so a wrong tag can never miscompile — it just
247
+ /// runs the VM.
248
+ pub fn jit_int_slots_enabled() -> bool {
249
+ static ENABLED: OnceLock<bool> = OnceLock::new();
250
+ *ENABLED.get_or_init(|| {
251
+ std::env::var("TISH_JIT_INT_SLOTS")
252
+ .map(|v| v != "0")
253
+ .unwrap_or(true)
254
+ })
255
+ }
256
+
197
257
  /// The self-recursion stack guard (#381). **Default ON**; `TISH_JIT_RECUR_GUARD=0` disables it.
198
258
  ///
199
259
  /// A JIT'd self-recursive numeric function recurses on the native stack (SelfCall lowers to a native
@@ -477,6 +537,8 @@ struct JitGlobal {
477
537
  /// `FuncId` of the imported `tish_math_call` host fn (#186), declared once at module init and
478
538
  /// re-imported into each compiled function via `declare_func_in_func`.
479
539
  math_call_id: cranelift_module::FuncId,
540
+ /// `FuncId` of the imported `tish_math_binary_call` host fn (#203), for the `MathBinary` intrinsic.
541
+ math_binary_call_id: cranelift_module::FuncId,
480
542
  /// `FuncId`s of the imported `tish_jv_*` vector runtime (#189).
481
543
  jv_fns: JvFns,
482
544
  /// #187: directly-callable numeric callees, keyed by the stable global name a top-level function is
@@ -525,6 +587,15 @@ thread_local! {
525
587
  static JV_ARENA: std::cell::RefCell<JvArena> = const {
526
588
  std::cell::RefCell::new(JvArena { vecs: Vec::new(), free: Vec::new(), deopt: false })
527
589
  };
590
+ /// #203 — cache of `(chunk ptr, inner loop header) → (trig header, trig end, trig indexes arrays)`
591
+ /// for [`osr_expand_cached`]. The expansion analysis (a whole-chunk scan + per-candidate classify)
592
+ /// is loop-structure-only, so it is stable for a given chunk+header and computed ONCE here rather
593
+ /// than on every back-edge — critical for a small hot loop inside a function called millions of
594
+ /// times (e.g. nbody's `advance`), where recomputing per call is a real regression. A stale entry
595
+ /// (a freed chunk's address reused) can only mis-route a perf hint: `run_osr`/`compile_loop_region`
596
+ /// re-validate the region structurally + by fingerprint, so a wrong hint never miscompiles.
597
+ static OSR_EXPAND_CACHE: std::cell::RefCell<HashMap<(usize, usize), (usize, usize, bool, bool)>> =
598
+ std::cell::RefCell::new(HashMap::new());
528
599
  }
529
600
  /// `handle` (1-based, `0` = null) → arena index, or `None` for the null handle.
530
601
  #[cfg(not(target_arch = "wasm32"))]
@@ -667,9 +738,23 @@ extern "C" fn tish_math_call(id: i32, x: f64) -> f64 {
667
738
  }
668
739
  }
669
740
 
741
+ /// #203 — host call for the `MathBinary` intrinsic (max/min/pow/atan2). Routes through
742
+ /// `MathBinaryFn::apply`, the single source of truth, so JIT ≡ VM ≡ interp.
743
+ extern "C" fn tish_math_binary_call(id: i32, a: f64, b: f64) -> f64 {
744
+ match tishlang_bytecode::MathBinaryFn::from_u16(id as u16) {
745
+ Some(m) => m.apply(a, b),
746
+ None => f64::NAN,
747
+ }
748
+ }
749
+
670
750
  /// Build the JIT module and declare the imported host functions (`tish_math_call` #186, the
671
751
  /// `tish_jv_*` vector runtime #189), returning the module + their `FuncId`s.
672
- fn new_module() -> Option<(JITModule, cranelift_module::FuncId, JvFns)> {
752
+ fn new_module() -> Option<(
753
+ JITModule,
754
+ cranelift_module::FuncId,
755
+ cranelift_module::FuncId,
756
+ JvFns,
757
+ )> {
673
758
  let mut flag_builder = settings::builder();
674
759
  // JIT code is loaded at a fixed address; no PIC / colocated libcalls needed.
675
760
  flag_builder.set("use_colocated_libcalls", "false").ok()?;
@@ -681,6 +766,7 @@ fn new_module() -> Option<(JITModule, cranelift_module::FuncId, JvFns)> {
681
766
  .ok()?;
682
767
  let mut builder = JITBuilder::with_isa(isa, cranelift_module::default_libcall_names());
683
768
  builder.symbol("tish_math_call", tish_math_call as *const u8);
769
+ builder.symbol("tish_math_binary_call", tish_math_binary_call as *const u8);
684
770
  builder.symbol("tish_jv_new", tish_jv_new as *const u8);
685
771
  builder.symbol("tish_jv_push", tish_jv_push as *const u8);
686
772
  builder.symbol("tish_jv_get", tish_jv_get as *const u8);
@@ -699,6 +785,16 @@ fn new_module() -> Option<(JITModule, cranelift_module::FuncId, JvFns)> {
699
785
  .declare_function("tish_math_call", Linkage::Import, &msig)
700
786
  .ok()?;
701
787
 
788
+ // `tish_math_binary_call(i32 fn-id, f64 a, f64 b) -> f64` (#203).
789
+ let mut mbsig = module.make_signature();
790
+ mbsig.params.push(AbiParam::new(types::I32));
791
+ mbsig.params.push(AbiParam::new(types::F64));
792
+ mbsig.params.push(AbiParam::new(types::F64));
793
+ mbsig.returns.push(AbiParam::new(types::F64));
794
+ let math_binary_id = module
795
+ .declare_function("tish_math_binary_call", Linkage::Import, &mbsig)
796
+ .ok()?;
797
+
702
798
  // Helper to declare a `tish_jv_*` import from param/return abi lists.
703
799
  let mut declare = |name: &str, params: &[AbiParam], rets: &[AbiParam]| {
704
800
  let mut s = module.make_signature();
@@ -741,18 +837,19 @@ fn new_module() -> Option<(JITModule, cranelift_module::FuncId, JvFns)> {
741
837
  free: declare("tish_jv_free", &[AbiParam::new(types::I64)], &[])?,
742
838
  deopt: declare("tish_jv_deopt", &[], &[])?,
743
839
  };
744
- Some((module, math_id, jv))
840
+ Some((module, math_id, math_binary_id, jv))
745
841
  }
746
842
 
747
843
  fn jit() -> Option<&'static Mutex<JitGlobal>> {
748
844
  JIT.get_or_init(|| {
749
- new_module().map(|(module, math_call_id, jv_fns)| {
845
+ new_module().map(|(module, math_call_id, math_binary_call_id, jv_fns)| {
750
846
  Mutex::new(JitGlobal {
751
847
  module,
752
848
  cache: HashMap::new(),
753
849
  osr_cache: HashMap::new(),
754
850
  counter: 0,
755
851
  math_call_id,
852
+ math_binary_call_id,
756
853
  jv_fns,
757
854
  callees: HashMap::new(),
758
855
  })
@@ -873,6 +970,121 @@ pub fn try_compile_numeric(chunk: &Chunk) -> Option<NumericFn> {
873
970
  result
874
971
  }
875
972
 
973
+ /// #203 — expand a hot inner-loop region to the OUTERMOST enclosing loop that is itself array-OSR-
974
+ /// compilable, so an array-indexing loop nest (matmul) marshals its arrays ONCE and runs the whole
975
+ /// nest natively, instead of re-marshalling O(array) per inner-loop entry (a net wash). Returns the
976
+ /// widest enclosing loop `(header, end)` whose region `classify_osr_arrays` accepts, or the input
977
+ /// region if none encloses it. Only meaningful under `TISH_JIT_OSR_ARRAY`; the caller gates on the
978
+ /// flag. SOUND to trigger at the returned loop's own back-edge: a `for` increment sits inside the
979
+ /// region before the back-edge (so the live-in captures the *next* index) and inner loop vars are
980
+ /// re-initialized by the body, so a native re-entry continues, never re-runs, completed iterations.
981
+ #[cfg(not(target_arch = "wasm32"))]
982
+ pub fn osr_expand_region(chunk: &Chunk, header_ip: usize, region_end: usize) -> (usize, usize) {
983
+ let code = &chunk.code;
984
+ let mut best = (header_ip, region_end);
985
+ let mut ip = 0usize;
986
+ while ip < code.len() {
987
+ let op = match Opcode::from_u8(code[ip]) {
988
+ Some(o) => o,
989
+ None => break,
990
+ };
991
+ let size = match op.instruction_size(code, ip) {
992
+ Some(s) => s,
993
+ None => break,
994
+ };
995
+ if op == Opcode::JumpBack {
996
+ if let Some(dist) = peek_u16(code, ip + 1) {
997
+ let pos = ip + 3; // region_end convention: byte AFTER the JumpBack instruction
998
+ let tgt = pos.saturating_sub(dist as usize); // that loop's header
999
+ // A loop that STRICTLY encloses the input region (`tgt < header_ip` — starts before it —
1000
+ // and `pos >= region_end`), and that ITSELF INDEXES ARRAYS, is worth expanding to: the
1001
+ // whole point is to marshal its arrays once. Pick the WIDEST (smallest header). Keeping
1002
+ // the header strictly smaller guarantees `best.0 == header_ip ⇒ best.1 == region_end`
1003
+ // (no expansion), so the trigger's "sound entry" test stays exact.
1004
+ let encloses = tgt < header_ip && pos >= region_end && tgt < best.0;
1005
+ if encloses
1006
+ && classify_osr_arrays(chunk, tgt, pos).is_some_and(|a| !a.slots.is_empty())
1007
+ {
1008
+ best = (tgt, pos);
1009
+ }
1010
+ }
1011
+ }
1012
+ ip += size;
1013
+ }
1014
+ best
1015
+ }
1016
+
1017
+ /// #203 — does the OSR region `[header_ip, region_end)` index at least one array? Drives the OSR
1018
+ /// trigger cadence: an array-indexing (expandable) loop fires on the first sound entry past the
1019
+ /// threshold; a pure-numeric loop keeps the original threshold+retry-modulo sampling.
1020
+ #[cfg(not(target_arch = "wasm32"))]
1021
+ pub fn osr_region_has_arrays(chunk: &Chunk, header_ip: usize, region_end: usize) -> bool {
1022
+ classify_osr_arrays(chunk, header_ip, region_end).is_some_and(|a| !a.slots.is_empty())
1023
+ }
1024
+
1025
+ /// #203 — is the region `[header_ip, region_end)` strictly enclosed by ANOTHER loop? An array region
1026
+ /// nested inside a non-array outer loop (e.g. k_nucleotide's `seq[i+j]` inner loop inside the Map-op
1027
+ /// outer loop) is re-entered per outer iteration, so array-OSR'ing it would re-marshal the whole array
1028
+ /// each entry — a net wash. Such regions are left to the interpreter (matching the flag-off path, where
1029
+ /// their `GetIndex` bails the numeric OSR anyway). Only an OUTERMOST array loop (matmul's `i` loop)
1030
+ /// marshals once and wins.
1031
+ #[cfg(not(target_arch = "wasm32"))]
1032
+ fn osr_region_enclosed(chunk: &Chunk, header_ip: usize, region_end: usize) -> bool {
1033
+ let code = &chunk.code;
1034
+ let mut ip = 0usize;
1035
+ while ip < code.len() {
1036
+ let op = match Opcode::from_u8(code[ip]) {
1037
+ Some(o) => o,
1038
+ None => break,
1039
+ };
1040
+ let size = match op.instruction_size(code, ip) {
1041
+ Some(s) => s,
1042
+ None => break,
1043
+ };
1044
+ if op == Opcode::JumpBack {
1045
+ if let Some(dist) = peek_u16(code, ip + 1) {
1046
+ let pos = ip + 3;
1047
+ let tgt = pos.saturating_sub(dist as usize);
1048
+ // strictly encloses on at least one side, contains on the other
1049
+ if tgt <= header_ip && pos >= region_end && (tgt < header_ip || pos > region_end) {
1050
+ return true;
1051
+ }
1052
+ }
1053
+ }
1054
+ ip += size;
1055
+ }
1056
+ false
1057
+ }
1058
+
1059
+ /// #203 — the OSR trigger decision for the loop at `header_ip`, memoized per `(chunk, header)` in a
1060
+ /// thread-local so the whole-chunk scans run at most once per loop header for the life of the thread
1061
+ /// (not once per enclosing-function call). Returns `(trig header, trig end, has_arrays, array_worthy)`:
1062
+ /// * `trig` = the outermost enclosing array loop to run instead (`== header_ip` if none);
1063
+ /// * `has_arrays` = the trig region indexes an array;
1064
+ /// * `array_worthy` = `has_arrays` AND the trig region is NOT itself nested in another loop — i.e.
1065
+ /// array-OSR'ing it marshals once and wins, rather than re-marshalling per outer iteration (a wash).
1066
+ /// The VM's `JumpBack` handler runs the array path only when `array_worthy`, gives up on a `has_arrays
1067
+ /// && !array_worthy` nested wash (interpret, matching flag-off), and takes the numeric path otherwise.
1068
+ #[cfg(not(target_arch = "wasm32"))]
1069
+ pub fn osr_expand_cached(
1070
+ chunk: &Chunk,
1071
+ header_ip: usize,
1072
+ region_end: usize,
1073
+ ) -> (usize, usize, bool, bool) {
1074
+ let key = (chunk as *const Chunk as usize, header_ip);
1075
+ OSR_EXPAND_CACHE.with(|c| {
1076
+ if let Some(&v) = c.borrow().get(&key) {
1077
+ return v;
1078
+ }
1079
+ let (th, te) = osr_expand_region(chunk, header_ip, region_end);
1080
+ let has_arrays = osr_region_has_arrays(chunk, th, te);
1081
+ let array_worthy = has_arrays && !osr_region_enclosed(chunk, th, te);
1082
+ let v = (th, te, has_arrays, array_worthy);
1083
+ c.borrow_mut().insert(key, v);
1084
+ v
1085
+ })
1086
+ }
1087
+
876
1088
  /// Compile the hot loop region `[header_ip, region_end)` of `chunk` to native code (#190 OSR), or
877
1089
  /// `None` if it is not a pure-numeric slot loop. Cached per `(chunk, header_ip)` with a fingerprint
878
1090
  /// guard (negative results included, so a non-compilable loop is scanned once). Called from the frame
@@ -909,6 +1121,20 @@ fn compile_loop_region(
909
1121
  return None;
910
1122
  }
911
1123
 
1124
+ // #203 array-index (the matmul lever): behind `TISH_JIT_OSR_ARRAY`, classify the array live-in
1125
+ // slots (`arr[i]` read / `arr[i]=v` write over a computed index). `None` ⇒ the region has array
1126
+ // ops this pass can't lower ⇒ not compilable (its index ops keep bailing). `Some(empty)` ⇒ a
1127
+ // pure-numeric region on the unchanged 2-pointer ABI. `Some(non-empty)` ⇒ the 3-pointer array ABI.
1128
+ let arrays = if osr_array_enabled() {
1129
+ classify_osr_arrays(chunk, header_ip, region_end)?
1130
+ } else {
1131
+ OsrArrays {
1132
+ slots: Vec::new(),
1133
+ writable: std::collections::BTreeSet::new(),
1134
+ }
1135
+ };
1136
+ let array_set: std::collections::BTreeSet<u16> = arrays.slots.iter().copied().collect();
1137
+
912
1138
  // 1. Scan the region: validate the whitelist (op_size = None ⇒ bail), collect in-region block
913
1139
  // leaders, the live slot set, and the EXIT targets (jump targets outside the region). A
914
1140
  // JumpBack must stay inside the region (its own loop, possibly nested); one leaving the region
@@ -921,6 +1147,10 @@ fn compile_loop_region(
921
1147
  while ip < region_end {
922
1148
  let op = Opcode::from_u8(*code.get(ip)?)?;
923
1149
  match op {
1150
+ // #203: an `arr[i]` read / `arr[i]=v` write over a classified array slot — validated by
1151
+ // `classify_osr_arrays`, lowered by the region body. `array_set` empty ⇒ these never appear
1152
+ // (their base slot would have been classified an array), so the arm below can't fire.
1153
+ Opcode::GetIndex | Opcode::SetIndex if !array_set.is_empty() => {}
924
1154
  // Pure slot / stack / arithmetic / structured control flow — the region vocabulary.
925
1155
  Opcode::Nop
926
1156
  | Opcode::Pop
@@ -930,9 +1160,15 @@ fn compile_loop_region(
930
1160
  | Opcode::LoopVarsEnd
931
1161
  | Opcode::LoopVarsBegin
932
1162
  | Opcode::UnaryOp
933
- | Opcode::MathUnary => {} // #186 — `Math.<fn>(x)`: 1 f64 in, 1 f64 out (like UnaryOp)
1163
+ | Opcode::MathUnary // #186 — `Math.<fn>(x)`: 1 f64 in, 1 f64 out (like UnaryOp)
1164
+ | Opcode::MathBinary => {} // #203 — `Math.<fn>(a,b)`: 2 f64 in, 1 f64 out
934
1165
  Opcode::LoadLocal | Opcode::StoreLocal => {
935
- used.insert(peek_u16(code, ip + 1)?);
1166
+ let s = peek_u16(code, ip + 1)?;
1167
+ // #203: an array slot's live-in is a `Value::Array` marshalled through the handles
1168
+ // buffer, NOT an f64 in the slots buffer — keep it out of the numeric live set.
1169
+ if !array_set.contains(&s) {
1170
+ used.insert(s);
1171
+ }
936
1172
  }
937
1173
  Opcode::LoadConst => match chunk.constants.get(peek_u16(code, ip + 1)? as usize) {
938
1174
  Some(Constant::Number(_)) | Some(Constant::Bool(_)) => {}
@@ -944,7 +1180,9 @@ fn compile_loop_region(
944
1180
  .map(|r| r as u8)
945
1181
  .and_then(u8_to_binop)?
946
1182
  {
947
- BinOp::And | BinOp::Or | BinOp::Pow | BinOp::In => return None,
1183
+ // #203: `**` (BinOp::Pow) IS lowerable now (a host call to `tish_math_binary_call`
1184
+ // in the region body), so it stays in the OSR whitelist. Logical/`in` still bail.
1185
+ BinOp::And | BinOp::Or | BinOp::In => return None,
948
1186
  _ => {}
949
1187
  }
950
1188
  }
@@ -996,11 +1234,26 @@ fn compile_loop_region(
996
1234
  let exits: Vec<usize> = exit_targets.iter().copied().collect();
997
1235
  let exit_id: HashMap<usize, usize> = exits.iter().enumerate().map(|(i, &t)| (t, i)).collect();
998
1236
 
999
- // 2. Build the region function. Signature: (slots: i64 ptr, deopt: i64 ptr) -> i32 exit id.
1237
+ // #203: array slot → its index in the handles buffer (marshalling order = ascending slot). Empty
1238
+ // ⇒ the pure-numeric 2-pointer ABI (unchanged). The region body loads `(ptr,len)` from the handles
1239
+ // buffer at `16 * array_pos[slot]` for each `arr[i]` access.
1240
+ let has_arrays = !arrays.slots.is_empty();
1241
+ let array_pos: HashMap<u16, usize> = arrays
1242
+ .slots
1243
+ .iter()
1244
+ .enumerate()
1245
+ .map(|(p, &s)| (s, p))
1246
+ .collect();
1247
+
1248
+ // 2. Build the region function. Signature: `(slots, deopt) -> i32` (pure numeric) or
1249
+ // `(slots, handles, deopt) -> i32` (#203 array mode) — all pointers; returns the exit id.
1000
1250
  let ptr_ty = g.module.target_config().pointer_type();
1001
1251
  let mut sig = g.module.make_signature();
1002
1252
  sig.params.push(AbiParam::new(ptr_ty)); // slots buffer
1003
- sig.params.push(AbiParam::new(ptr_ty)); // deopt flag (reserved)
1253
+ if has_arrays {
1254
+ sig.params.push(AbiParam::new(ptr_ty)); // #203 handles buffer (ArrayHandle[])
1255
+ }
1256
+ sig.params.push(AbiParam::new(ptr_ty)); // deopt flag (int-slot miss #514 / OOB #203)
1004
1257
  sig.returns.push(AbiParam::new(types::I32));
1005
1258
 
1006
1259
  let name = format!("tish_osr_{}", g.counter);
@@ -1013,6 +1266,7 @@ fn compile_loop_region(
1013
1266
  let mut ctx = g.module.make_context();
1014
1267
  ctx.func.signature = sig.clone();
1015
1268
  let math_fref = g.module.declare_func_in_func(g.math_call_id, &mut ctx.func);
1269
+ let math_binary_fref = g.module.declare_func_in_func(g.math_binary_call_id, &mut ctx.func);
1016
1270
  let mut fbctx = FunctionBuilderContext::new();
1017
1271
  let built = build_loop_region_body(
1018
1272
  &mut ctx.func,
@@ -1025,6 +1279,8 @@ fn compile_loop_region(
1025
1279
  &buf_pos,
1026
1280
  &exit_id,
1027
1281
  math_fref,
1282
+ math_binary_fref,
1283
+ &array_pos,
1028
1284
  );
1029
1285
  if !built {
1030
1286
  g.module.clear_context(&mut ctx);
@@ -1039,10 +1295,16 @@ fn compile_loop_region(
1039
1295
  return None;
1040
1296
  }
1041
1297
  let fptr = g.module.get_finalized_function(id);
1298
+ let array_slots: Vec<(u16, bool)> = arrays
1299
+ .slots
1300
+ .iter()
1301
+ .map(|&s| (s, arrays.writable.contains(&s)))
1302
+ .collect();
1042
1303
  Some(LoopFn {
1043
1304
  ptr: fptr as usize,
1044
1305
  used_slots,
1045
1306
  exits,
1307
+ array_slots,
1046
1308
  })
1047
1309
  }
1048
1310
 
@@ -1063,6 +1325,10 @@ fn build_loop_region_body(
1063
1325
  buf_pos: &HashMap<u16, usize>,
1064
1326
  exit_id: &HashMap<usize, usize>,
1065
1327
  math_fref: cranelift::codegen::ir::FuncRef,
1328
+ math_binary_fref: cranelift::codegen::ir::FuncRef,
1329
+ // #203: array slot → its index in the handles buffer (marshalling order). Empty ⇒ the pure-numeric
1330
+ // 2-pointer ABI; non-empty ⇒ the 3-pointer array ABI `(slots, handles, deopt)`.
1331
+ array_pos: &HashMap<u16, usize>,
1066
1332
  ) -> bool {
1067
1333
  let code = &chunk.code;
1068
1334
  let num_slots = chunk.num_slots as usize;
@@ -1081,28 +1347,137 @@ fn build_loop_region_body(
1081
1347
  bcx.append_block_params_for_function_params(entry);
1082
1348
  bcx.switch_to_block(entry);
1083
1349
  let params: Vec<ClifValue> = bcx.block_params(entry).to_vec();
1084
- let slots_ptr = params[0]; // params[1] = deopt flag, reserved (v1 never writes it)
1350
+ let slots_ptr = params[0];
1351
+ // #203: the array ABI inserts a `handles` pointer between `slots` and `deopt`. `deopt` is set on an
1352
+ // int-slot live-in guard fail (#514) or an out-of-bounds array index (#203) → the VM re-interprets.
1353
+ let (handles_ptr, deopt_ptr) = if array_pos.is_empty() {
1354
+ (None, params[1])
1355
+ } else {
1356
+ (Some(params[1]), params[2])
1357
+ };
1358
+ // #203: load each array live-in's `(ptr, len)` ONCE at entry (loop-invariant; the entry block
1359
+ // dominates the whole loop). `ArrayHandle` is `#[repr(C)] { ptr: *mut f64 @0, len: usize @8 }`,
1360
+ // 16 bytes. `array_handles[slot] = (data_ptr, len_i64)`.
1361
+ let ptr_ty = bcx.func.dfg.value_type(slots_ptr);
1362
+ let mut array_handles: HashMap<u16, (ClifValue, ClifValue)> = HashMap::new();
1363
+ if let Some(hptr) = handles_ptr {
1364
+ for (&slot, &pos) in array_pos.iter() {
1365
+ let base = (pos * 16) as i32;
1366
+ let data = bcx.ins().load(ptr_ty, MemFlags::new(), hptr, base);
1367
+ let len = bcx.ins().load(types::I64, MemFlags::new(), hptr, base + 8);
1368
+ array_handles.insert(slot, (data, len));
1369
+ }
1370
+ }
1371
+
1372
+ // #514: port the fn-body int-typed slots (#511) into the OSR loop region. A slot whose every store
1373
+ // is integer-typed lives in an `i64` Variable (`Repr::I64Num`), so an integer hash/PRNG accumulator
1374
+ // stays in integer registers across the loop instead of an f64 materialize per store + ToInt32 per
1375
+ // load. `arity` is 0 at top level (no params). `classify_int_slots` is chunk-level, so a slot that
1376
+ // is int-stored in the region but f64-stored elsewhere in the chunk is (safely) NOT tagged.
1377
+ let int_slots = classify_int_slots(chunk, 0);
1085
1378
  let vars: Vec<Variable> = (0..num_slots)
1086
- .map(|_| bcx.declare_var(types::F64))
1379
+ .map(|slot| {
1380
+ let ty = if int_slots.contains(&slot) {
1381
+ types::I64
1382
+ } else {
1383
+ types::F64
1384
+ };
1385
+ bcx.declare_var(ty)
1386
+ })
1087
1387
  .collect();
1388
+
1389
+ // Entry marshalling. A non-int slot loads its f64 live-in straight. An int slot's live-in arrives
1390
+ // as f64 from the VM frame, so it must be an EXACT integer that round-trips through i64 — this
1391
+ // rejects non-integers, NaN / ±Inf, out-of-i64-range, and -0 — or the `I64Num` invariant breaks.
1392
+ // On any failure we set the deopt flag and return; the VM discards the (still-pristine) slots and
1393
+ // re-interprets the loop. The guard runs once at region entry, off the per-iteration path.
1394
+ let mut all_ok: Option<ClifValue> = None;
1088
1395
  for (slot, &var) in vars.iter().enumerate() {
1089
- // Live slots load from their buffer position; slots the region never touches init to 0 (dead).
1090
- let init = if let Some(&p) = buf_pos.get(&(slot as u16)) {
1091
- bcx.ins()
1092
- .load(types::F64, MemFlags::new(), slots_ptr, (p * 8) as i32)
1396
+ let live: Option<ClifValue> = if let Some(&p) = buf_pos.get(&(slot as u16)) {
1397
+ Some(
1398
+ bcx.ins()
1399
+ .load(types::F64, MemFlags::new(), slots_ptr, (p * 8) as i32),
1400
+ )
1093
1401
  } else {
1094
- bcx.ins().f64const(0.0)
1402
+ None
1095
1403
  };
1096
- bcx.def_var(var, init);
1404
+ if int_slots.contains(&slot) {
1405
+ match live {
1406
+ Some(x) => {
1407
+ let i64c = bcx.ins().fcvt_to_sint_sat(types::I64, x);
1408
+ let back = bcx.ins().fcvt_from_sint(types::F64, i64c);
1409
+ let exact = bcx.ins().fcmp(FloatCC::Equal, back, x);
1410
+ // -0 round-trips to +0 (IEEE `==` is true) but differs as an f64 read, so reject
1411
+ // it explicitly: 1/(-0) is -∞, 1/(+0) is +∞.
1412
+ let zero = bcx.ins().f64const(0.0);
1413
+ let is_zero = bcx.ins().fcmp(FloatCC::Equal, x, zero);
1414
+ let one = bcx.ins().f64const(1.0);
1415
+ let recip = bcx.ins().fdiv(one, x);
1416
+ let recip_neg = bcx.ins().fcmp(FloatCC::LessThan, recip, zero);
1417
+ let is_neg0 = bcx.ins().band(is_zero, recip_neg);
1418
+ let not_neg0 = bcx.ins().bxor_imm(is_neg0, 1);
1419
+ let ok = bcx.ins().band(exact, not_neg0);
1420
+ all_ok = Some(match all_ok {
1421
+ Some(a) => bcx.ins().band(a, ok),
1422
+ None => ok,
1423
+ });
1424
+ bcx.def_var(var, i64c);
1425
+ }
1426
+ None => {
1427
+ let z = bcx.ins().iconst(types::I64, 0);
1428
+ bcx.def_var(var, z);
1429
+ }
1430
+ }
1431
+ } else {
1432
+ let init = match live {
1433
+ Some(x) => x,
1434
+ None => bcx.ins().f64const(0.0),
1435
+ };
1436
+ bcx.def_var(var, init);
1437
+ }
1097
1438
  }
1098
1439
  let header_block = match blocks.get(&header_ip) {
1099
1440
  Some(&b) => b,
1100
1441
  None => return false,
1101
1442
  };
1102
- bcx.ins().jump(header_block, &[]);
1443
+ match all_ok {
1444
+ // At least one int slot has a live-in: enter the loop only if every guard passed, else deopt.
1445
+ Some(ok) => {
1446
+ let deopt_block = bcx.create_block();
1447
+ bcx.ins().brif(ok, header_block, &[], deopt_block, &[]);
1448
+ bcx.switch_to_block(deopt_block);
1449
+ let flag = bcx.ins().iconst(types::I8, 1);
1450
+ bcx.ins().store(MemFlags::new(), flag, deopt_ptr, 0);
1451
+ let zero_id = bcx.ins().iconst(types::I32, 0);
1452
+ bcx.ins().return_(&[zero_id]);
1453
+ }
1454
+ None => {
1455
+ bcx.ins().jump(header_block, &[]);
1456
+ }
1457
+ }
1458
+
1459
+ // #203: shared out-of-bounds pad — every array bounds-check that fails branches here, sets the
1460
+ // deopt flag, and returns (exit id 0). The VM sees `deopt != 0` and re-interprets from the pristine
1461
+ // pre-region state (the numeric buffer AND the array scratch are discarded, the real arrays are
1462
+ // untouched), so a mid-region OOB never commits a partial result. Filled now; sealed at the end.
1463
+ let array_deopt_block = if array_pos.is_empty() {
1464
+ None
1465
+ } else {
1466
+ let b = bcx.create_block();
1467
+ bcx.switch_to_block(b);
1468
+ let flag = bcx.ins().iconst(types::I8, 1);
1469
+ bcx.ins().store(MemFlags::new(), flag, deopt_ptr, 0);
1470
+ let zero = bcx.ins().iconst(types::I32, 0);
1471
+ bcx.ins().return_(&[zero]);
1472
+ Some(b)
1473
+ };
1103
1474
 
1104
1475
  // Translate the region. Operand stack is empty at every block boundary (statement-level flow).
1105
- let mut stack: Vec<(ClifValue, bool)> = Vec::new();
1476
+ let mut stack: Vec<JV> = Vec::new();
1477
+ // #203: array handles "in flight" — pushed by `LoadLocal(array slot)` as `(data_ptr, len)`, consumed
1478
+ // by the very next `GetIndex`/`SetIndex`. Like `stack`, must be empty at every block boundary (an
1479
+ // array access never spans a branch).
1480
+ let mut arr_pending: Vec<(ClifValue, ClifValue)> = Vec::new();
1106
1481
  bcx.switch_to_block(header_block);
1107
1482
  let mut cur = header_block;
1108
1483
  let mut terminated = false;
@@ -1111,7 +1486,9 @@ fn build_loop_region_body(
1111
1486
  if let Some(&blk) = blocks.get(&ip) {
1112
1487
  if blk != cur {
1113
1488
  if !terminated {
1114
- if !stack.is_empty() {
1489
+ // #203: an operand OR a pending array handle spanning a block boundary is a shape
1490
+ // this straight-line-per-block emitter can't carry → bail (region interpreted).
1491
+ if !stack.is_empty() || !arr_pending.is_empty() {
1115
1492
  return false;
1116
1493
  }
1117
1494
  bcx.ins().jump(blk, &[]);
@@ -1120,6 +1497,7 @@ fn build_loop_region_body(
1120
1497
  cur = blk;
1121
1498
  terminated = false;
1122
1499
  stack.clear();
1500
+ arr_pending.clear();
1123
1501
  }
1124
1502
  }
1125
1503
  let op = match Opcode::from_u8(code[ip]).zip(op_size_at(code, ip)) {
@@ -1139,11 +1517,25 @@ fn build_loop_region_body(
1139
1517
  Some(s) => s as usize,
1140
1518
  None => return false,
1141
1519
  };
1520
+ // #203: an array slot's "value" is its `(data_ptr, len)` handle — stage it OFF the
1521
+ // numeric stack for the next `GetIndex`/`SetIndex` (matches build_body_cfg's jv_pending).
1522
+ if let Some(&handle) = array_handles.get(&(slot as u16)) {
1523
+ arr_pending.push(handle);
1524
+ ip += 3;
1525
+ continue;
1526
+ }
1142
1527
  let v = match vars.get(slot) {
1143
1528
  Some(v) => *v,
1144
1529
  None => return false,
1145
1530
  };
1146
- stack.push((bcx.use_var(v), false));
1531
+ // #514: an int slot's `use_var` is an i64 carrying the exact number (`I64Num`);
1532
+ // downstream int ops take it as identity, one exact convert at an f64 boundary.
1533
+ let lv = bcx.use_var(v);
1534
+ stack.push(if int_slots.contains(&slot) {
1535
+ JV::i64num(lv)
1536
+ } else {
1537
+ JV::f64(lv)
1538
+ });
1147
1539
  ip += 3;
1148
1540
  }
1149
1541
  Opcode::StoreLocal => {
@@ -1151,18 +1543,38 @@ fn build_loop_region_body(
1151
1543
  Some(s) => s as usize,
1152
1544
  None => return false,
1153
1545
  };
1154
- let (val, is_bool) = match stack.pop() {
1546
+ let jval = match stack.pop() {
1155
1547
  Some(x) => x,
1156
1548
  None => return false,
1157
1549
  };
1158
- if is_bool {
1550
+ if jval.is_bool() {
1159
1551
  return false; // no boolean slots (keeps the number/bool distinction clean)
1160
1552
  }
1161
1553
  let v = match vars.get(slot) {
1162
1554
  Some(v) => *v,
1163
1555
  None => return false,
1164
1556
  };
1165
- bcx.def_var(v, val);
1557
+ if int_slots.contains(&slot) {
1558
+ // #514/#168: store the EXACT number as i64 — I32 stores sign-extend, U32
1559
+ // zero-extend, I64Num is identity, an integral constant re-derives its exact
1560
+ // i64. Anything else means the optimistic pre-pass mis-tagged this slot (e.g. a
1561
+ // merge point whose linear predecessor differed) → bail; the VM keeps semantics.
1562
+ let iv = match jval.repr {
1563
+ Repr::I32 => bcx.ins().sextend(types::I64, jval.v),
1564
+ Repr::U32 => bcx.ins().uextend(types::I64, jval.v),
1565
+ Repr::I64Num => jval.v,
1566
+ Repr::F64 => match int_const_i64(&mut bcx, chunk, code, ip) {
1567
+ Some(iv) => iv,
1568
+ None => return false,
1569
+ },
1570
+ Repr::Bool => return false,
1571
+ };
1572
+ bcx.def_var(v, iv);
1573
+ } else {
1574
+ // f64 Variable — one materialize at the store boundary.
1575
+ let val = jv_f64(&mut bcx, jval);
1576
+ bcx.def_var(v, val);
1577
+ }
1166
1578
  ip += 3;
1167
1579
  }
1168
1580
  Opcode::Pop => {
@@ -1223,13 +1635,14 @@ fn build_loop_region_body(
1223
1635
  Some(o) => o as i16 as isize,
1224
1636
  None => return false,
1225
1637
  };
1226
- let (cond, _) = match stack.pop() {
1638
+ let cond = match stack.pop() {
1227
1639
  Some(x) => x,
1228
1640
  None => return false,
1229
1641
  };
1230
1642
  if !stack.is_empty() {
1231
1643
  return false;
1232
1644
  }
1645
+ let cond = jv_f64(&mut bcx, cond);
1233
1646
  let falsy = falsy_flag(&mut bcx, cond);
1234
1647
  let t = ((ip + 3) as isize + off).max(0) as usize;
1235
1648
  let target = match target_block(t) {
@@ -1254,12 +1667,133 @@ fn build_loop_region_body(
1254
1667
  Some(m) => m,
1255
1668
  None => return false,
1256
1669
  };
1257
- let (x, _) = match stack.pop() {
1670
+ let x = match stack.pop() {
1258
1671
  Some(v) => v,
1259
1672
  None => return false,
1260
1673
  };
1674
+ let x = jv_f64(&mut bcx, x);
1261
1675
  let r = emit_math_unary(&mut bcx, math_fref, mfn, x);
1262
- stack.push((r, false));
1676
+ stack.push(JV::f64(r));
1677
+ ip += 3;
1678
+ }
1679
+ // #203 — `arr[idx]` read. `arr`'s `(data,len)` handle is in `arr_pending`; the computed
1680
+ // index is the f64 top of stack. `idx as usize` (saturating: NaN/neg → 0) matches the VM's
1681
+ // coercion; an out-of-bounds index branches to the shared deopt pad (the VM re-interprets).
1682
+ Opcode::GetIndex if !arr_pending.is_empty() => {
1683
+ let (data, len) = match arr_pending.pop() {
1684
+ Some(h) => h,
1685
+ None => return false,
1686
+ };
1687
+ let idx = match stack.pop() {
1688
+ Some(v) => v,
1689
+ None => return false,
1690
+ };
1691
+ let db = match array_deopt_block {
1692
+ Some(b) => b,
1693
+ None => return false,
1694
+ };
1695
+ let idx = jv_f64(&mut bcx, idx);
1696
+ let i = bcx.ins().fcvt_to_uint_sat(types::I64, idx);
1697
+ let inb = bcx.ins().icmp(IntCC::UnsignedLessThan, i, len);
1698
+ let cont = bcx.create_block();
1699
+ bcx.ins().brif(inb, cont, &[], db, &[]);
1700
+ bcx.switch_to_block(cont);
1701
+ cur = cont; // keep block-boundary tracking accurate after the mid-stream split
1702
+ let off = bcx.ins().imul_imm(i, 8);
1703
+ let addr = bcx.ins().iadd(data, off);
1704
+ let val = bcx.ins().load(types::F64, MemFlags::new(), addr, 0);
1705
+ stack.push(JV::f64(val));
1706
+ ip += 1;
1707
+ }
1708
+ // #203 — `arr[idx] = v`. Stack: `[idx, val, dup_val]` (a `Dup` of `val` for the expression
1709
+ // result); `arr`'s handle is in `arr_pending`. Store `dup_val` (== `val`) at the bounds-
1710
+ // checked address; OOB → the deopt pad (scratch discarded, real array untouched).
1711
+ Opcode::SetIndex if !arr_pending.is_empty() => {
1712
+ let (data, len) = match arr_pending.pop() {
1713
+ Some(h) => h,
1714
+ None => return false,
1715
+ };
1716
+ let dup = match stack.pop() {
1717
+ Some(v) => v,
1718
+ None => return false,
1719
+ };
1720
+ let _val = match stack.pop() {
1721
+ Some(v) => v,
1722
+ None => return false,
1723
+ };
1724
+ let idx = match stack.pop() {
1725
+ Some(v) => v,
1726
+ None => return false,
1727
+ };
1728
+ // #203 SOUNDNESS: the region flattens every element to f64 and `run_osr` re-boxes the
1729
+ // writeback as `Value::Number`. Storing a BOOLEAN (a comparison result `arr[i] = x>y`,
1730
+ // a `LoadConst(Bool)`, or `!x`) would therefore land as `Number 1`/`0` where the
1731
+ // interpreter stores `Bool true`/`false` — a divergence. There are no bool SLOTS in a
1732
+ // region (StoreLocal bails on a bool), so a bool value is always this transient Repr::Bool
1733
+ // — bail the whole region compile, let the interpreter handle the write.
1734
+ if dup.is_bool() {
1735
+ return false;
1736
+ }
1737
+ let db = match array_deopt_block {
1738
+ Some(b) => b,
1739
+ None => return false,
1740
+ };
1741
+ let store_val = jv_f64(&mut bcx, dup);
1742
+ let idx = jv_f64(&mut bcx, idx);
1743
+ let i = bcx.ins().fcvt_to_uint_sat(types::I64, idx);
1744
+ let inb = bcx.ins().icmp(IntCC::UnsignedLessThan, i, len);
1745
+ let cont = bcx.create_block();
1746
+ bcx.ins().brif(inb, cont, &[], db, &[]);
1747
+ bcx.switch_to_block(cont);
1748
+ cur = cont;
1749
+ let off = bcx.ins().imul_imm(i, 8);
1750
+ let addr = bcx.ins().iadd(data, off);
1751
+ bcx.ins().store(MemFlags::new(), store_val, addr, 0);
1752
+ stack.push(JV::f64(store_val)); // assignment yields the value
1753
+ ip += 1;
1754
+ }
1755
+ Opcode::MathBinary => {
1756
+ // #203 — `Math.<fn>(a, b)`: pop b, pop a, host-call, push. Every 2-arg Math fn routes
1757
+ // through the host call (identical to the VM), so there are no native-op semantics to
1758
+ // match — the win is skipping the boxed GetMember+value_call the generic path pays.
1759
+ let id = match peek_u16(code, ip + 1) {
1760
+ Some(v) => v,
1761
+ None => return false,
1762
+ };
1763
+ let b = match stack.pop() {
1764
+ Some(v) => v,
1765
+ None => return false,
1766
+ };
1767
+ let a = match stack.pop() {
1768
+ Some(v) => v,
1769
+ None => return false,
1770
+ };
1771
+ let b = jv_f64(&mut bcx, b);
1772
+ let a = jv_f64(&mut bcx, a);
1773
+ let idc = bcx.ins().iconst(types::I32, id as i64);
1774
+ let call = bcx.ins().call(math_binary_fref, &[idc, a, b]);
1775
+ let r = bcx.inst_results(call)[0];
1776
+ stack.push(JV::f64(r));
1777
+ ip += 3;
1778
+ }
1779
+ _ if is_binop_pow(op, code, ip) => {
1780
+ // #203: `a ** b` → host call to `tish_math_binary_call(Pow, a, b)` (== VM's `powf`).
1781
+ let r = match stack.pop() {
1782
+ Some(v) => v,
1783
+ None => return false,
1784
+ };
1785
+ let l = match stack.pop() {
1786
+ Some(v) => v,
1787
+ None => return false,
1788
+ };
1789
+ let r = jv_f64(&mut bcx, r);
1790
+ let l = jv_f64(&mut bcx, l);
1791
+ let idc = bcx
1792
+ .ins()
1793
+ .iconst(types::I32, tishlang_bytecode::MathBinaryFn::Pow as i64);
1794
+ let call = bcx.ins().call(math_binary_fref, &[idc, l, r]);
1795
+ let res = bcx.inst_results(call)[0];
1796
+ stack.push(JV::f64(res));
1263
1797
  ip += 3;
1264
1798
  }
1265
1799
  _ => match emit_simple_op(&mut bcx, chunk, code, &mut ip, &mut stack, &[], 0) {
@@ -1276,7 +1810,13 @@ fn build_loop_region_body(
1276
1810
  for (&t, &blk) in &exit_blocks {
1277
1811
  bcx.switch_to_block(blk);
1278
1812
  for (p, &slot) in used_slots.iter().enumerate() {
1279
- let v = bcx.use_var(vars[slot as usize]);
1813
+ let raw = bcx.use_var(vars[slot as usize]);
1814
+ // #514: an int slot holds an i64 — flush its exact f64 value back to the buffer.
1815
+ let v = if int_slots.contains(&(slot as usize)) {
1816
+ bcx.ins().fcvt_from_sint(types::F64, raw)
1817
+ } else {
1818
+ raw
1819
+ };
1280
1820
  bcx.ins()
1281
1821
  .store(MemFlags::new(), v, slots_ptr, (p * 8) as i32);
1282
1822
  }
@@ -1327,6 +1867,90 @@ fn fcmp_f64(bcx: &mut FunctionBuilder, cc: FloatCC, a: ClifValue, b: ClifValue)
1327
1867
  bcx.ins().select(cond, one, zero)
1328
1868
  }
1329
1869
 
1870
+ /// #168 — which Cranelift representation a JIT stack slot currently holds.
1871
+ ///
1872
+ /// `F64`: an f64 number. `Bool`: an f64 constrained to 0.0/1.0 with JS-boolean semantics (the
1873
+ /// old `is_bool` flag — comparisons/`!` produce it, `LoadConst Bool` pushes it). `I32`: an i32
1874
+ /// holding `ToInt32` bits (signed→f64 on materialize). `U32`: an i32 holding `ToUint32` bits
1875
+ /// (UNSIGNED→f64 on materialize — `>>>` results past 2³¹ stay positive numbers).
1876
+ ///
1877
+ /// The integer reprs are the point: a bitwise/shift chain (`h = ((h<<13)|(h>>>19)) >>> 0`) used
1878
+ /// to pay `int→f64→int` conversion ROUND-TRIPS between every op, because the stack could only
1879
+ /// say "f64 or bool". Now each such op consumes raw int bits via [`jv_i32_bits`] (an identity
1880
+ /// for `I32`/`U32`) and pushes an int repr; f64 materialization happens once, at a genuine
1881
+ /// boundary (store/return/float-arith/compare/call), via [`jv_f64`]. `ToInt32` and `ToUint32`
1882
+ /// share bit patterns, so `I32` vs `U32` only matters at the f64 boundary (signed vs unsigned
1883
+ /// convert) — the bits themselves are interchangeable as shift/bitwise inputs.
1884
+ #[derive(Clone, Copy, PartialEq, Eq)]
1885
+ enum Repr {
1886
+ F64,
1887
+ Bool,
1888
+ I32,
1889
+ U32,
1890
+ /// #168 int-typed slots: an i64 holding an EXACT integral JS number in [-2^31, 2^32).
1891
+ /// Signedness is encoded in the value itself (an I32 store sign-extends, a U32 store
1892
+ /// zero-extends, an integral constant loads exactly), so a slot whose stores mix `^`
1893
+ /// (ToInt32 domain) and `>>> 0` (ToUint32 domain) needs no dynamic tag: materializing
1894
+ /// is one exact `fcvt_from_sint`, and re-entering the 32-bit domain is one `ireduce`
1895
+ /// (ToInt32 of an integral value in this range IS its low 32 bits).
1896
+ I64Num,
1897
+ }
1898
+
1899
+ /// A typed JIT stack value: the Cranelift SSA value plus its current [`Repr`].
1900
+ #[derive(Clone, Copy)]
1901
+ struct JV {
1902
+ v: ClifValue,
1903
+ repr: Repr,
1904
+ }
1905
+
1906
+ impl JV {
1907
+ fn f64(v: ClifValue) -> Self {
1908
+ Self { v, repr: Repr::F64 }
1909
+ }
1910
+ fn boolean(v: ClifValue) -> Self {
1911
+ Self { v, repr: Repr::Bool }
1912
+ }
1913
+ fn int32(v: ClifValue) -> Self {
1914
+ Self { v, repr: Repr::I32 }
1915
+ }
1916
+ fn uint32(v: ClifValue) -> Self {
1917
+ Self { v, repr: Repr::U32 }
1918
+ }
1919
+ fn is_bool(&self) -> bool {
1920
+ self.repr == Repr::Bool
1921
+ }
1922
+ }
1923
+
1924
+ /// Materialize a [`JV`] as an f64 — identity for `F64`/`Bool` (a Bool already IS an f64 0/1),
1925
+ /// one signed/unsigned convert for the integer reprs.
1926
+ fn jv_f64(bcx: &mut FunctionBuilder, jv: JV) -> ClifValue {
1927
+ match jv.repr {
1928
+ Repr::F64 | Repr::Bool => jv.v,
1929
+ Repr::I32 => bcx.ins().fcvt_from_sint(types::F64, jv.v),
1930
+ Repr::U32 => bcx.ins().fcvt_from_uint(types::F64, jv.v),
1931
+ // Exact by construction: an I64Num is integral and |v| < 2^32 << 2^53.
1932
+ Repr::I64Num => bcx.ins().fcvt_from_sint(types::F64, jv.v),
1933
+ }
1934
+ }
1935
+
1936
+ /// Raw `ToInt32` bit pattern of a [`JV`] as an i32 — an identity (zero instructions) for
1937
+ /// `I32`/`U32` (they share bit patterns), [`js_to_int32`] for the f64 reprs (a Bool's 0.0/1.0
1938
+ /// converts to 0/1 exactly).
1939
+ fn jv_i32_bits(bcx: &mut FunctionBuilder, jv: JV) -> ClifValue {
1940
+ match jv.repr {
1941
+ Repr::I32 | Repr::U32 => jv.v,
1942
+ Repr::F64 | Repr::Bool => js_to_int32(bcx, jv.v),
1943
+ // ToInt32 of an exact integral in [-2^31, 2^32) is its low 32 bits.
1944
+ Repr::I64Num => bcx.ins().ireduce(types::I32, jv.v),
1945
+ }
1946
+ }
1947
+
1948
+ impl JV {
1949
+ fn i64num(v: ClifValue) -> Self {
1950
+ Self { v, repr: Repr::I64Num }
1951
+ }
1952
+ }
1953
+
1330
1954
  /// f64 → JS `ToInt32` as an `I32` clif value, matching `tishlang_core::to_int32` (so a JIT-compiled
1331
1955
  /// `& | ^ ~` agrees with the VM fallback). Saturating-cast→`ireduce` is the modulo-2³² for finite
1332
1956
  /// values and already gives 0 for NaN / `-∞`; the branchless `select` on `|x| < ∞` maps `+∞` (which
@@ -1457,6 +2081,7 @@ fn compile_chunk(g: &mut JitGlobal, chunk: &Chunk) -> Option<NumericFn> {
1457
2081
  ctx.func.signature = sig.clone();
1458
2082
  let self_ref = g.module.declare_func_in_func(id, &mut ctx.func);
1459
2083
  let math_fref = g.module.declare_func_in_func(g.math_call_id, &mut ctx.func);
2084
+ let math_binary_fref = g.module.declare_func_in_func(g.math_binary_call_id, &mut ctx.func);
1460
2085
  // #189: import the `tish_jv_*` FuncRefs into this function when it has local arrays.
1461
2086
  let jv_ctx = if is_jv {
1462
2087
  Some(JvCtx {
@@ -1485,6 +2110,7 @@ fn compile_chunk(g: &mut JitGlobal, chunk: &Chunk) -> Option<NumericFn> {
1485
2110
  0,
1486
2111
  recur_guard,
1487
2112
  math_fref,
2113
+ math_binary_fref,
1488
2114
  jv_ctx.as_ref(),
1489
2115
  &resolved,
1490
2116
  ) {
@@ -1736,6 +2362,7 @@ fn compile_chunk_arrays(
1736
2362
  // `handles_ptr`/`deopt_ptr` — only the numeric args are re-marshalled per level.
1737
2363
  let self_ref = g.module.declare_func_in_func(id, &mut ctx.func);
1738
2364
  let math_fref = g.module.declare_func_in_func(g.math_call_id, &mut ctx.func);
2365
+ let math_binary_fref = g.module.declare_func_in_func(g.math_binary_call_id, &mut ctx.func);
1739
2366
  // #187: array-mode functions (e.g. spectral_norm's multiplyAv) may call a register-f64 callee.
1740
2367
  let resolved = build_resolved_callees(g, chunk, &mut ctx.func);
1741
2368
  let mut fbctx = FunctionBuilderContext::new();
@@ -1749,6 +2376,7 @@ fn compile_chunk_arrays(
1749
2376
  mask,
1750
2377
  recursive, // #187: array-mode recursion guard (the entry SP-bail keys off this)
1751
2378
  math_fref,
2379
+ math_binary_fref,
1752
2380
  None,
1753
2381
  &resolved,
1754
2382
  )
@@ -1800,7 +2428,7 @@ fn emit_simple_op(
1800
2428
  chunk: &Chunk,
1801
2429
  code: &[u8],
1802
2430
  ip: &mut usize,
1803
- stack: &mut Vec<(ClifValue, bool)>,
2431
+ stack: &mut Vec<JV>,
1804
2432
  params: &[ClifValue],
1805
2433
  arity: usize,
1806
2434
  ) -> SimpleOp {
@@ -1825,7 +2453,7 @@ fn emit_simple_op(
1825
2453
  if slot >= arity {
1826
2454
  return SimpleOp::Unsupported;
1827
2455
  }
1828
- stack.push((params[slot], false));
2456
+ stack.push(JV::f64(params[slot]));
1829
2457
  }
1830
2458
  Opcode::LoadConst => {
1831
2459
  let idx = match read_u16(code, ip) {
@@ -1835,11 +2463,11 @@ fn emit_simple_op(
1835
2463
  match chunk.constants.get(idx) {
1836
2464
  Some(Constant::Number(n)) => {
1837
2465
  let v = bcx.ins().f64const(*n);
1838
- stack.push((v, false));
2466
+ stack.push(JV::f64(v));
1839
2467
  }
1840
2468
  Some(Constant::Bool(b)) => {
1841
2469
  let v = bcx.ins().f64const(if *b { 1.0 } else { 0.0 });
1842
- stack.push((v, true));
2470
+ stack.push(JV::boolean(v));
1843
2471
  }
1844
2472
  _ => return SimpleOp::Unsupported,
1845
2473
  }
@@ -1852,121 +2480,143 @@ fn emit_simple_op(
1852
2480
  if stack.len() < 2 {
1853
2481
  return SimpleOp::Unsupported;
1854
2482
  }
1855
- let (r, r_bool) = stack.pop().unwrap();
1856
- let (l, l_bool) = stack.pop().unwrap();
2483
+ let r = stack.pop().unwrap();
2484
+ let l = stack.pop().unwrap();
1857
2485
  // #187: a bool value (from a bool slot or a `LoadConst Bool`) can't take part in an
1858
2486
  // EQUALITY compare here — the JIT compares the `f64` 0/1 bits, but JS `===`/`!==` (and
1859
2487
  // strict `==`/`!=`) treat `0 === false` as FALSE across types. Bail so the interpreter
1860
2488
  // decides. Relational (`<`/`>`/…) coerces bool→0/1 in both, so those stay JIT'd.
2489
+ // Integer reprs are plain NUMBERS, so they take part in every compare.
1861
2490
  let is_eq = matches!(
1862
2491
  bop,
1863
2492
  BinOp::Eq | BinOp::Ne | BinOp::StrictEq | BinOp::StrictNe
1864
2493
  );
1865
- if is_eq && (l_bool || r_bool) {
2494
+ if is_eq && (l.is_bool() || r.is_bool()) {
1866
2495
  return SimpleOp::Unsupported;
1867
2496
  }
1868
- let is_cmp = matches!(
1869
- bop,
1870
- BinOp::Eq
1871
- | BinOp::Ne
1872
- | BinOp::StrictEq
1873
- | BinOp::StrictNe
1874
- | BinOp::Lt
1875
- | BinOp::Le
1876
- | BinOp::Gt
1877
- | BinOp::Ge
1878
- );
1879
- let v = match bop {
1880
- BinOp::Add => bcx.ins().fadd(l, r),
1881
- BinOp::Sub => bcx.ins().fsub(l, r),
1882
- BinOp::Mul => bcx.ins().fmul(l, r),
1883
- BinOp::Div => bcx.ins().fdiv(l, r),
1884
- BinOp::Eq | BinOp::StrictEq => fcmp_f64(bcx, FloatCC::Equal, l, r),
1885
- BinOp::Ne | BinOp::StrictNe => fcmp_f64(bcx, FloatCC::NotEqual, l, r),
1886
- BinOp::Lt => fcmp_f64(bcx, FloatCC::LessThan, l, r),
1887
- BinOp::Le => fcmp_f64(bcx, FloatCC::LessThanOrEqual, l, r),
1888
- BinOp::Gt => fcmp_f64(bcx, FloatCC::GreaterThan, l, r),
1889
- BinOp::Ge => fcmp_f64(bcx, FloatCC::GreaterThanOrEqual, l, r),
2497
+ // #168: float arithmetic / comparisons materialize both operands as f64 at this
2498
+ // boundary; bitwise/shift ops consume raw int bits (identity for I32/U32 operands)
2499
+ // and PUSH an integer repr — a chained `((h<<13)|(h>>>19))>>>0` stays in i32
2500
+ // registers with zero intermediate converts. `Mul` stays f64 ON PURPOSE: V8 rounds
2501
+ // `h * K` past 2^53 the same way, and the gauntlet checksum pins that agreement.
2502
+ let v: JV = match bop {
2503
+ BinOp::Add => {
2504
+ let (lf, rf) = (jv_f64(bcx, l), jv_f64(bcx, r));
2505
+ JV::f64(bcx.ins().fadd(lf, rf))
2506
+ }
2507
+ BinOp::Sub => {
2508
+ let (lf, rf) = (jv_f64(bcx, l), jv_f64(bcx, r));
2509
+ JV::f64(bcx.ins().fsub(lf, rf))
2510
+ }
2511
+ BinOp::Mul => {
2512
+ let (lf, rf) = (jv_f64(bcx, l), jv_f64(bcx, r));
2513
+ JV::f64(bcx.ins().fmul(lf, rf))
2514
+ }
2515
+ BinOp::Div => {
2516
+ let (lf, rf) = (jv_f64(bcx, l), jv_f64(bcx, r));
2517
+ JV::f64(bcx.ins().fdiv(lf, rf))
2518
+ }
2519
+ BinOp::Eq | BinOp::StrictEq => {
2520
+ let (lf, rf) = (jv_f64(bcx, l), jv_f64(bcx, r));
2521
+ JV::boolean(fcmp_f64(bcx, FloatCC::Equal, lf, rf))
2522
+ }
2523
+ BinOp::Ne | BinOp::StrictNe => {
2524
+ let (lf, rf) = (jv_f64(bcx, l), jv_f64(bcx, r));
2525
+ JV::boolean(fcmp_f64(bcx, FloatCC::NotEqual, lf, rf))
2526
+ }
2527
+ BinOp::Lt => {
2528
+ let (lf, rf) = (jv_f64(bcx, l), jv_f64(bcx, r));
2529
+ JV::boolean(fcmp_f64(bcx, FloatCC::LessThan, lf, rf))
2530
+ }
2531
+ BinOp::Le => {
2532
+ let (lf, rf) = (jv_f64(bcx, l), jv_f64(bcx, r));
2533
+ JV::boolean(fcmp_f64(bcx, FloatCC::LessThanOrEqual, lf, rf))
2534
+ }
2535
+ BinOp::Gt => {
2536
+ let (lf, rf) = (jv_f64(bcx, l), jv_f64(bcx, r));
2537
+ JV::boolean(fcmp_f64(bcx, FloatCC::GreaterThan, lf, rf))
2538
+ }
2539
+ BinOp::Ge => {
2540
+ let (lf, rf) = (jv_f64(bcx, l), jv_f64(bcx, r));
2541
+ JV::boolean(fcmp_f64(bcx, FloatCC::GreaterThanOrEqual, lf, rf))
2542
+ }
1890
2543
  BinOp::Mod => {
1891
2544
  // f64 remainder a - trunc(a/b)*b — exactly Rust's `%`, which the
1892
2545
  // VM's eval_binop uses, so JIT and VM-fallback agree bit-for-bit.
1893
- let q = bcx.ins().fdiv(l, r);
2546
+ let (lf, rf) = (jv_f64(bcx, l), jv_f64(bcx, r));
2547
+ let q = bcx.ins().fdiv(lf, rf);
1894
2548
  let t = bcx.ins().trunc(q);
1895
- let p = bcx.ins().fmul(t, r);
1896
- bcx.ins().fsub(l, p)
2549
+ let p = bcx.ins().fmul(t, rf);
2550
+ JV::f64(bcx.ins().fsub(lf, p))
1897
2551
  }
1898
- // Bitwise AND/OR/XOR via JS ToInt32 (modulo 2³², NaN/±∞ → 0) — see [`js_to_int32`].
2552
+ // Bitwise AND/OR/XOR via JS ToInt32 (modulo 2³², NaN/±∞ → 0) — [`jv_i32_bits`]
2553
+ // is [`js_to_int32`] for f64 operands and an IDENTITY for int-repr operands.
1899
2554
  BinOp::BitAnd | BinOp::BitOr | BinOp::BitXor => {
1900
- let li = js_to_int32(bcx, l);
1901
- let ri = js_to_int32(bcx, r);
2555
+ let li = jv_i32_bits(bcx, l);
2556
+ let ri = jv_i32_bits(bcx, r);
1902
2557
  let res = match bop {
1903
2558
  BinOp::BitAnd => bcx.ins().band(li, ri),
1904
2559
  BinOp::BitOr => bcx.ins().bor(li, ri),
1905
2560
  BinOp::BitXor => bcx.ins().bxor(li, ri),
1906
2561
  _ => unreachable!(),
1907
2562
  };
1908
- bcx.ins().fcvt_from_sint(types::F64, res)
1909
- }
1910
- // Shifts. JS masks the count to the low 5 bits (`& 31`); the low 5 bits of `ToInt32(r)`
1911
- // equal `ToUint32(r)`, so `js_to_int32(r)` carries the right amount. We mask explicitly
1912
- // (`& 31`) so correctness never depends on Cranelift's own amount-masking convention.
1913
- // `<<`/`>>` are signed-domain (ToInt32 → i32 → signed→f64); `>>>` is logical on the
1914
- // unsigned bits with an UNSIGNED→f64 convert (result may exceed 2³¹). Bit-for-bit with
1915
- // vm.rs `eval_binop`: Shl/Shr = `to_int32(l).wrapping_sh*(to_uint32(r))`,
1916
- // UShr = `to_uint32(l).wrapping_shr(to_uint32(r))`.
2563
+ JV::int32(res)
2564
+ }
2565
+ // Shifts. JS masks the count to the low 5 bits (`& 31`); the low 5 bits of
2566
+ // `ToInt32(r)` equal `ToUint32(r)`, so `jv_i32_bits(r)` carries the right amount.
2567
+ // We mask explicitly (`& 31`) so correctness never depends on Cranelift's own
2568
+ // amount-masking convention. `<<`/`>>` are signed-domain (I32 repr — signed→f64
2569
+ // when materialized); `>>>` is logical on the unsigned bits (U32 repr — an
2570
+ // UNSIGNED→f64 materialize keeps a bit-31 result a positive number, JS `>>>`).
2571
+ // Bit-for-bit with vm.rs `eval_binop`: Shl/Shr = `to_int32(l).wrapping_sh*
2572
+ // (to_uint32(r))`, UShr = `to_uint32(l).wrapping_shr(to_uint32(r))`.
1917
2573
  BinOp::Shl | BinOp::Shr | BinOp::UShr => {
1918
- let li = js_to_int32(bcx, l);
1919
- let amt = js_to_int32(bcx, r);
2574
+ let li = jv_i32_bits(bcx, l);
2575
+ let amt = jv_i32_bits(bcx, r);
1920
2576
  let mask = bcx.ins().iconst(types::I32, 31);
1921
2577
  let amt = bcx.ins().band(amt, mask);
1922
2578
  match bop {
1923
- BinOp::Shl => {
1924
- let res = bcx.ins().ishl(li, amt);
1925
- bcx.ins().fcvt_from_sint(types::F64, res)
1926
- }
1927
- BinOp::Shr => {
1928
- let res = bcx.ins().sshr(li, amt);
1929
- bcx.ins().fcvt_from_sint(types::F64, res)
1930
- }
1931
- // UShr: logical shift on the same 32-bit value bits as ToUint32(l), then
1932
- // unsigned→f64 so a result with bit 31 set stays a positive number (JS `>>>`).
1933
- _ => {
1934
- let res = bcx.ins().ushr(li, amt);
1935
- bcx.ins().fcvt_from_uint(types::F64, res)
1936
- }
2579
+ BinOp::Shl => JV::int32(bcx.ins().ishl(li, amt)),
2580
+ BinOp::Shr => JV::int32(bcx.ins().sshr(li, amt)),
2581
+ _ => JV::uint32(bcx.ins().ushr(li, amt)),
1937
2582
  }
1938
2583
  }
1939
2584
  // Pow/In/And/Or: fall back to the VM.
1940
2585
  _ => return SimpleOp::Unsupported,
1941
2586
  };
1942
- stack.push((v, is_cmp));
2587
+ stack.push(v);
1943
2588
  }
1944
2589
  Opcode::UnaryOp => {
1945
2590
  let uop = match read_u16(code, ip).map(|r| r as u8).and_then(u8_to_unaryop) {
1946
2591
  Some(u) => u,
1947
2592
  None => return SimpleOp::Unsupported,
1948
2593
  };
1949
- let (o, _) = match stack.pop() {
2594
+ let o = match stack.pop() {
1950
2595
  Some(x) => x,
1951
2596
  None => return SimpleOp::Unsupported,
1952
2597
  };
1953
- let (v, is_bool) = match uop {
1954
- UnaryOp::Neg => (bcx.ins().fneg(o), false),
1955
- UnaryOp::Pos => (o, false),
2598
+ let v: JV = match uop {
2599
+ UnaryOp::Neg => {
2600
+ let of = jv_f64(bcx, o);
2601
+ JV::f64(bcx.ins().fneg(of))
2602
+ }
2603
+ UnaryOp::Pos => JV::f64(jv_f64(bcx, o)),
1956
2604
  UnaryOp::Not => {
2605
+ let of = jv_f64(bcx, o);
1957
2606
  let zero = bcx.ins().f64const(0.0);
1958
- (fcmp_f64(bcx, FloatCC::Equal, o, zero), true)
2607
+ JV::boolean(fcmp_f64(bcx, FloatCC::Equal, of, zero))
1959
2608
  }
1960
- // `~x` = `!ToInt32(x) as f64` — JS ToInt32 (modulo, NaN/±∞ → 0) via [`js_to_int32`],
1961
- // matching the VM so a JIT-compiled `~` can't diverge on large/non-finite values.
2609
+ // `~x` = `!ToInt32(x)` — JS ToInt32 (modulo, NaN/±∞ → 0) via [`jv_i32_bits`]
2610
+ // (identity for an int-repr operand), matching the VM so a JIT-compiled `~`
2611
+ // can't diverge on large/non-finite values. Pushes I32 — a `~` chain stays
2612
+ // in integer registers.
1962
2613
  UnaryOp::BitNot => {
1963
- let oi = js_to_int32(bcx, o);
1964
- let res = bcx.ins().bnot(oi);
1965
- (bcx.ins().fcvt_from_sint(types::F64, res), false)
2614
+ let oi = jv_i32_bits(bcx, o);
2615
+ JV::int32(bcx.ins().bnot(oi))
1966
2616
  }
1967
2617
  _ => return SimpleOp::Unsupported,
1968
2618
  };
1969
- stack.push((v, is_bool));
2619
+ stack.push(v);
1970
2620
  }
1971
2621
  _ => unreachable!("guarded above"),
1972
2622
  }
@@ -1989,12 +2639,209 @@ fn falsy_flag(bcx: &mut FunctionBuilder, cond: ClifValue) -> ClifValue {
1989
2639
  /// branch, nested branches, calls, member/index, or mismatched `is_bool` all return `None` so the
1990
2640
  /// VM runs the chunk instead — purely additive. Returns `Some(result_is_bool)`.
1991
2641
  /// Byte size of an opcode the loop-JIT understands; `None` ⇒ unsupported (bail → VM).
2642
+ /// #203: is the op at `ip` a `BinOp` whose operator is `**` (Pow)? Lets the loop builders intercept
2643
+ /// `a ** b` and lower it to a `tish_math_binary_call(Pow, …)` host call before the generic
2644
+ /// `emit_simple_op` (which rejects Pow) would bail the function.
2645
+ fn is_binop_pow(op: Opcode, code: &[u8], ip: usize) -> bool {
2646
+ op == Opcode::BinOp
2647
+ && peek_u16(code, ip + 1).map(|r| r as u8).and_then(u8_to_binop) == Some(BinOp::Pow)
2648
+ }
2649
+
2650
+ /// #203 branch-free ternary lowering in the LOOP builder. **Default ON**; `TISH_JIT_TERNARY=0` falls
2651
+ /// back to bailing the function (byte-identical, just interpreted). Additive: a non-matching shape or
2652
+ /// disabled flag simply doesn't take the fast path.
2653
+ fn jit_ternary_enabled() -> bool {
2654
+ static ENABLED: OnceLock<bool> = OnceLock::new();
2655
+ *ENABLED.get_or_init(|| {
2656
+ std::env::var("TISH_JIT_TERNARY")
2657
+ .map(|v| v != "0")
2658
+ .unwrap_or(true)
2659
+ })
2660
+ }
2661
+
2662
+ /// #203: ops allowed inside a ternary ARM — pure, side-effect-free value producers the inline
2663
+ /// `select` can emit UNCONDITIONALLY (both arms run, then `select` picks). No control flow / calls /
2664
+ /// stores / member / index / array ops (which could have side effects or need a block).
2665
+ fn is_ternary_arm_op(op: Opcode) -> bool {
2666
+ matches!(
2667
+ op,
2668
+ Opcode::LoadLocal
2669
+ | Opcode::LoadConst
2670
+ | Opcode::BinOp
2671
+ | Opcode::UnaryOp
2672
+ | Opcode::MathUnary
2673
+ | Opcode::MathBinary
2674
+ )
2675
+ }
2676
+
2677
+ /// #203: if the `JumpIfFalse` at `jif_ip` begins a clean ternary `cond ? A : B` — a forward branch
2678
+ /// whose THEN arm is straight-line pure-value ops ending in a `Jump`, whose ELSE arm begins exactly
2679
+ /// where that `Jump` lands past and runs pure-value ops to a single merge — return the merge ip.
2680
+ /// Conservative: a nested branch in an arm, a back-edge, or any non-value op → `None`, and the caller
2681
+ /// leaves it to the block CFG (which bails) / VM. Sound for compiler-generated bytecode: nothing jumps
2682
+ /// INTO a ternary's arms, so treating the whole span as one inline unit never drops a CFG edge.
2683
+ fn ternary_span(code: &[u8], jif_ip: usize) -> Option<(usize, usize, usize)> {
2684
+ if Opcode::from_u8(*code.get(jif_ip)?)? != Opcode::JumpIfFalse {
2685
+ return None;
2686
+ }
2687
+ let off = peek_u16(code, jif_ip + 1)? as i16 as isize;
2688
+ let else_target = ((jif_ip + 3) as isize + off).max(0) as usize;
2689
+ // THEN arm: pure-value ops from jif+3 up to the trailing `Jump`.
2690
+ let then_start = jif_ip + 3;
2691
+ let mut tip = then_start;
2692
+ loop {
2693
+ let op = Opcode::from_u8(*code.get(tip)?)?;
2694
+ if op == Opcode::Jump {
2695
+ break;
2696
+ }
2697
+ if !is_ternary_arm_op(op) {
2698
+ return None;
2699
+ }
2700
+ tip += op_size(op)?;
2701
+ }
2702
+ let then_end = tip; // at the trailing Jump
2703
+ let joff = peek_u16(code, tip + 1)? as i16 as isize;
2704
+ let else_start = tip + 3;
2705
+ let merge = (else_start as isize + joff).max(0) as usize;
2706
+ if else_target != else_start || merge <= else_start {
2707
+ return None;
2708
+ }
2709
+ // ELSE arm: pure-value ops from else_start up to merge.
2710
+ let mut eip = else_start;
2711
+ while eip < merge {
2712
+ let op = Opcode::from_u8(*code.get(eip)?)?;
2713
+ if !is_ternary_arm_op(op) {
2714
+ return None;
2715
+ }
2716
+ eip += op_size(op)?;
2717
+ }
2718
+ if eip != merge || then_end <= then_start {
2719
+ return None; // else arm must land exactly on merge; then arm must be non-empty
2720
+ }
2721
+ Some((then_end, else_start, merge))
2722
+ }
2723
+
2724
+ /// #203: emit one ternary ARM — the pure-value ops in `[start, end)` — in the `build_body_cfg` context
2725
+ /// (`LoadLocal` reads the slot's CURRENT SSA `Variable`, unlike `emit_simple_op`'s params-only path).
2726
+ /// Returns `false` (bail the whole compile) on any op not in the arm whitelist or a decode failure.
2727
+ /// Both arms are emitted unconditionally, so every op here is side-effect-free (`is_ternary_arm_op`).
2728
+ #[allow(clippy::too_many_arguments)]
2729
+ fn emit_ternary_arm(
2730
+ bcx: &mut FunctionBuilder,
2731
+ chunk: &Chunk,
2732
+ code: &[u8],
2733
+ start: usize,
2734
+ end: usize,
2735
+ stack: &mut Vec<JV>,
2736
+ vars: &[Variable],
2737
+ int_slots: &std::collections::HashSet<usize>,
2738
+ bool_slots: &std::collections::HashSet<usize>,
2739
+ math_fref: cranelift::codegen::ir::FuncRef,
2740
+ math_binary_fref: cranelift::codegen::ir::FuncRef,
2741
+ ) -> bool {
2742
+ let mut ip = start;
2743
+ while ip < end {
2744
+ let op = match Opcode::from_u8(code[ip]) {
2745
+ Some(o) => o,
2746
+ None => return false,
2747
+ };
2748
+ match op {
2749
+ Opcode::LoadLocal => {
2750
+ let slot = match peek_u16(code, ip + 1) {
2751
+ Some(s) => s as usize,
2752
+ None => return false,
2753
+ };
2754
+ let v = match vars.get(slot) {
2755
+ Some(v) => *v,
2756
+ None => return false,
2757
+ };
2758
+ let lv = bcx.use_var(v);
2759
+ stack.push(if bool_slots.contains(&slot) {
2760
+ JV::boolean(lv)
2761
+ } else if int_slots.contains(&slot) {
2762
+ JV::i64num(lv)
2763
+ } else {
2764
+ JV::f64(lv)
2765
+ });
2766
+ ip += 3;
2767
+ }
2768
+ Opcode::MathUnary => {
2769
+ let id = match peek_u16(code, ip + 1) {
2770
+ Some(v) => v,
2771
+ None => return false,
2772
+ };
2773
+ let mfn = match MathUnaryFn::from_u16(id) {
2774
+ Some(m) => m,
2775
+ None => return false,
2776
+ };
2777
+ let x = match stack.pop() {
2778
+ Some(v) => v,
2779
+ None => return false,
2780
+ };
2781
+ let x = jv_f64(bcx, x);
2782
+ let r = emit_math_unary(bcx, math_fref, mfn, x);
2783
+ stack.push(JV::f64(r));
2784
+ ip += 3;
2785
+ }
2786
+ Opcode::MathBinary => {
2787
+ let id = match peek_u16(code, ip + 1) {
2788
+ Some(v) => v,
2789
+ None => return false,
2790
+ };
2791
+ let b = match stack.pop() {
2792
+ Some(v) => v,
2793
+ None => return false,
2794
+ };
2795
+ let a = match stack.pop() {
2796
+ Some(v) => v,
2797
+ None => return false,
2798
+ };
2799
+ let b = jv_f64(bcx, b);
2800
+ let a = jv_f64(bcx, a);
2801
+ let idc = bcx.ins().iconst(types::I32, id as i64);
2802
+ let call = bcx.ins().call(math_binary_fref, &[idc, a, b]);
2803
+ stack.push(JV::f64(bcx.inst_results(call)[0]));
2804
+ ip += 3;
2805
+ }
2806
+ _ if is_binop_pow(op, code, ip) => {
2807
+ let r = match stack.pop() {
2808
+ Some(v) => v,
2809
+ None => return false,
2810
+ };
2811
+ let l = match stack.pop() {
2812
+ Some(v) => v,
2813
+ None => return false,
2814
+ };
2815
+ let r = jv_f64(bcx, r);
2816
+ let l = jv_f64(bcx, l);
2817
+ let idc = bcx
2818
+ .ins()
2819
+ .iconst(types::I32, tishlang_bytecode::MathBinaryFn::Pow as i64);
2820
+ let call = bcx.ins().call(math_binary_fref, &[idc, l, r]);
2821
+ stack.push(JV::f64(bcx.inst_results(call)[0]));
2822
+ ip += 3;
2823
+ }
2824
+ // LoadConst / BinOp / UnaryOp are param-free — `emit_simple_op` handles them (advancing a
2825
+ // local ip copy). It only ever touches `stack`, not slots, for these.
2826
+ Opcode::LoadConst | Opcode::BinOp | Opcode::UnaryOp => {
2827
+ let mut sip = ip;
2828
+ match emit_simple_op(bcx, chunk, code, &mut sip, stack, &[], 0) {
2829
+ SimpleOp::Handled(_) => ip = sip,
2830
+ _ => return false,
2831
+ }
2832
+ }
2833
+ _ => return false,
2834
+ }
2835
+ }
2836
+ true
2837
+ }
2838
+
1992
2839
  fn op_size(op: Opcode) -> Option<usize> {
1993
2840
  use Opcode::*;
1994
2841
  Some(match op {
1995
2842
  Nop | Pop | Dup | Return | LoopVarsEnd | EnterBlock | ExitBlock | GetIndex | SetIndex => 1,
1996
2843
  LoadLocal | StoreLocal | LoadConst | BinOp | UnaryOp | Jump | JumpIfFalse | JumpBack
1997
- | LoopVarsBegin | SelfCall | MathUnary => 3,
2844
+ | LoopVarsBegin | SelfCall | MathUnary | MathBinary => 3,
1998
2845
  _ => return None,
1999
2846
  })
2000
2847
  }
@@ -2123,6 +2970,217 @@ fn classify_bool_slots(chunk: &Chunk) -> std::collections::HashSet<usize> {
2123
2970
  set
2124
2971
  }
2125
2972
 
2973
+ /// #168: slots eligible for i64 int-typed storage ([`Repr::I64Num`]): every `StoreLocal` to the
2974
+ /// slot is IMMEDIATELY preceded (linearly — statement boundaries have empty stacks, and the
2975
+ /// ternary shape bails out of `build_body_cfg`, so the linear predecessor IS the value producer;
2976
+ /// same assumption [`classify_bool_slots`] rests on) by an op that pushes integer bits — a
2977
+ /// bitwise/shift `BinOp`, a `~` `UnaryOp`, or a `LoadConst` of an integral Number in
2978
+ /// [-2^31, 2^32) (excluding `-0`, whose sign an integer store would erase). Param slots are
2979
+ /// excluded (they arrive as arbitrary f64s through the ABI). Like `classify_bool_slots` this is
2980
+ /// an OPTIMISTIC tag: a merge point could make the linear predecessor differ from the dynamic
2981
+ /// one, so the StoreLocal translator re-checks the actual stored repr and BAILS compilation on
2982
+ /// any non-integer store to a tagged slot — misclassification runs the VM, never miscompiles.
2983
+ fn classify_int_slots(chunk: &Chunk, arity: usize) -> std::collections::HashSet<usize> {
2984
+ if !jit_int_slots_enabled() {
2985
+ return std::collections::HashSet::new();
2986
+ }
2987
+ let code = &chunk.code;
2988
+ let mut int_stores: std::collections::HashSet<usize> = std::collections::HashSet::new();
2989
+ let mut other_stores: std::collections::HashSet<usize> = std::collections::HashSet::new();
2990
+ let mut ip = 0usize;
2991
+ let mut prev_pushes_int = false;
2992
+ while ip < code.len() {
2993
+ let op = match Opcode::from_u8(code[ip]) {
2994
+ Some(o) => o,
2995
+ None => break,
2996
+ };
2997
+ let size = match op.instruction_size(code, ip) {
2998
+ Some(s) => s,
2999
+ None => break,
3000
+ };
3001
+ if op == Opcode::StoreLocal {
3002
+ if let Some(s) = peek_u16(code, ip + 1) {
3003
+ let slot = s as usize;
3004
+ if prev_pushes_int && slot >= arity {
3005
+ int_stores.insert(slot);
3006
+ } else {
3007
+ other_stores.insert(slot);
3008
+ }
3009
+ }
3010
+ }
3011
+ prev_pushes_int = match op {
3012
+ Opcode::BinOp => matches!(
3013
+ peek_u16(code, ip + 1)
3014
+ .map(|r| r as u8)
3015
+ .and_then(u8_to_binop),
3016
+ Some(
3017
+ BinOp::BitAnd
3018
+ | BinOp::BitOr
3019
+ | BinOp::BitXor
3020
+ | BinOp::Shl
3021
+ | BinOp::Shr
3022
+ | BinOp::UShr
3023
+ )
3024
+ ),
3025
+ Opcode::UnaryOp => matches!(
3026
+ peek_u16(code, ip + 1)
3027
+ .map(|r| r as u8)
3028
+ .and_then(u8_to_unaryop),
3029
+ Some(UnaryOp::BitNot)
3030
+ ),
3031
+ Opcode::LoadConst => matches!(
3032
+ peek_u16(code, ip + 1).and_then(|i| chunk.constants.get(i as usize)),
3033
+ Some(Constant::Number(n))
3034
+ if n.fract() == 0.0
3035
+ && *n >= -(2f64.powi(31))
3036
+ && *n < 2f64.powi(32)
3037
+ && n.to_bits() != (-0f64).to_bits()
3038
+ ),
3039
+ _ => false,
3040
+ };
3041
+ ip += size;
3042
+ }
3043
+ int_stores
3044
+ .difference(&other_stores)
3045
+ .copied()
3046
+ .collect()
3047
+ }
3048
+
3049
+ /// #203 — the array live-in slots of an OSR loop region (the matmul lever), with the written subset.
3050
+ #[cfg(not(target_arch = "wasm32"))]
3051
+ struct OsrArrays {
3052
+ /// Array-holding slots in ascending slot order — the marshalling order of the handles buffer.
3053
+ slots: Vec<u16>,
3054
+ /// Subset of `slots` written via `SetIndex` (copied back only after a clean, non-deopt exit).
3055
+ writable: std::collections::BTreeSet<u16>,
3056
+ }
3057
+
3058
+ /// #203 — classify the array live-in slots of an OSR loop region `[header_ip, region_end)`. Abstract-
3059
+ /// interprets the region's operand stack to find every slot used PURELY as the base of an `arr[i]`
3060
+ /// read (`GetIndex`) or `arr[i] = v` write (`SetIndex`) with an arbitrary COMPUTED index (`a[i*N+k]`).
3061
+ ///
3062
+ /// Returns `Some(arrays)` — possibly with an empty `slots` (a pure-numeric region: caller keeps the
3063
+ /// unchanged 2-pointer ABI) — when the region's array usage is fully lowerable; `None` when it holds a
3064
+ /// `GetIndex`/`SetIndex` the emitter can't handle: a non-slot / computed array BASE (`a[i][j]`), an
3065
+ /// array slot also used as a scalar or reassigned (`arr = …`), or an operand stack the linear abstract
3066
+ /// interp can't balance. `None` leaves the region non-OSR-compilable (its index ops still bail).
3067
+ ///
3068
+ /// Imprecision is always SAFE — never a miscompile: a slot wrongly called an array fails the runtime
3069
+ /// `Value::Array` marshalling guard in `run_osr` (→ interpret); a slot wrongly NOT called an array
3070
+ /// leaves a `GetIndex`/`SetIndex` the region emitter rejects (→ region not compiled). Only correctly-
3071
+ /// classified, all-`Number`-element arrays are ever compiled.
3072
+ #[cfg(not(target_arch = "wasm32"))]
3073
+ fn classify_osr_arrays(chunk: &Chunk, header_ip: usize, region_end: usize) -> Option<OsrArrays> {
3074
+ let code = &chunk.code;
3075
+ // An abstract stack entry: `Ref(slot)` = a bare `LoadLocal(slot)` result untouched since; any other
3076
+ // producer (const, arithmetic, a loaded element, …) is `Scalar`. An array base must be a `Ref`.
3077
+ #[derive(Clone, Copy, PartialEq)]
3078
+ enum Av {
3079
+ Scalar,
3080
+ Ref(u16),
3081
+ }
3082
+ let mut astack: Vec<Av> = Vec::new();
3083
+ let mut array_slots: std::collections::BTreeSet<u16> = Default::default();
3084
+ let mut writable: std::collections::BTreeSet<u16> = Default::default();
3085
+ // Slots ever consumed as a scalar (arithmetic operand, index, store source, cond, …) or reassigned
3086
+ // — an array slot must appear in NEITHER, else its live-in isn't a stable, index-only array handle.
3087
+ let mut scalar_use: std::collections::BTreeSet<u16> = Default::default();
3088
+ let mut stored: std::collections::BTreeSet<u16> = Default::default();
3089
+ // Consume one operand; a bare `Ref(slot)` reaching a scalar position taints that slot.
3090
+ let taint = |astack: &mut Vec<Av>, scalar_use: &mut std::collections::BTreeSet<u16>| -> Option<()> {
3091
+ match astack.pop()? {
3092
+ Av::Ref(s) => {
3093
+ scalar_use.insert(s);
3094
+ }
3095
+ Av::Scalar => {}
3096
+ }
3097
+ Some(())
3098
+ };
3099
+ let mut ip = header_ip;
3100
+ while ip < region_end {
3101
+ let op = Opcode::from_u8(*code.get(ip)?)?;
3102
+ let size = op_size(op)?; // whitelist-sized ops only; anything else ⇒ region not compilable
3103
+ match op {
3104
+ Opcode::LoadLocal => astack.push(Av::Ref(peek_u16(code, ip + 1)?)),
3105
+ Opcode::LoadConst => astack.push(Av::Scalar),
3106
+ Opcode::StoreLocal => {
3107
+ stored.insert(peek_u16(code, ip + 1)?);
3108
+ taint(&mut astack, &mut scalar_use)?; // the stored value
3109
+ }
3110
+ Opcode::Pop => {
3111
+ astack.pop()?; // discarded — no taint (a bare pop doesn't "use" the slot)
3112
+ }
3113
+ Opcode::Dup => {
3114
+ let top = *astack.last()?;
3115
+ astack.push(top);
3116
+ }
3117
+ Opcode::Nop
3118
+ | Opcode::EnterBlock
3119
+ | Opcode::ExitBlock
3120
+ | Opcode::LoopVarsEnd
3121
+ | Opcode::LoopVarsBegin => {}
3122
+ Opcode::UnaryOp | Opcode::MathUnary => {
3123
+ taint(&mut astack, &mut scalar_use)?;
3124
+ astack.push(Av::Scalar);
3125
+ }
3126
+ Opcode::BinOp | Opcode::MathBinary => {
3127
+ taint(&mut astack, &mut scalar_use)?;
3128
+ taint(&mut astack, &mut scalar_use)?;
3129
+ astack.push(Av::Scalar);
3130
+ }
3131
+ Opcode::GetIndex => {
3132
+ taint(&mut astack, &mut scalar_use)?; // index (a scalar use of whatever produced it)
3133
+ match astack.pop()? {
3134
+ Av::Ref(s) => {
3135
+ array_slots.insert(s);
3136
+ }
3137
+ Av::Scalar => return None, // computed / nested array base (`a[i][j]`) — bail
3138
+ }
3139
+ astack.push(Av::Scalar); // the loaded element
3140
+ }
3141
+ Opcode::SetIndex => {
3142
+ taint(&mut astack, &mut scalar_use)?; // dup_val
3143
+ taint(&mut astack, &mut scalar_use)?; // val
3144
+ taint(&mut astack, &mut scalar_use)?; // idx
3145
+ match astack.pop()? {
3146
+ Av::Ref(s) => {
3147
+ array_slots.insert(s);
3148
+ writable.insert(s);
3149
+ }
3150
+ Av::Scalar => return None,
3151
+ }
3152
+ astack.push(Av::Scalar); // SetIndex leaves the assigned value
3153
+ }
3154
+ // A terminator ends the basic block: the region invariant makes the operand stack empty
3155
+ // here (JumpIfFalse pops its cond first). Clearing keeps the abstract stack self-contained
3156
+ // per block — array access never spans a branch, so this loses nothing we can lower.
3157
+ Opcode::Jump | Opcode::JumpBack => astack.clear(),
3158
+ Opcode::JumpIfFalse => {
3159
+ taint(&mut astack, &mut scalar_use)?; // cond
3160
+ astack.clear();
3161
+ }
3162
+ _ => return None, // an opcode outside the region vocabulary ⇒ not compilable
3163
+ }
3164
+ ip += size;
3165
+ }
3166
+ // An array slot used anywhere as a scalar, or reassigned, isn't a stable index-only handle ⇒ the
3167
+ // whole region is not array-compilable (its index ops keep bailing).
3168
+ if array_slots
3169
+ .iter()
3170
+ .any(|s| scalar_use.contains(s) || stored.contains(s))
3171
+ {
3172
+ return None;
3173
+ }
3174
+ // Cap the handle count (matmul uses 3); keep it small so the marshalling stays cheap.
3175
+ if array_slots.len() > 8 {
3176
+ return None;
3177
+ }
3178
+ Some(OsrArrays {
3179
+ slots: array_slots.into_iter().collect(),
3180
+ writable,
3181
+ })
3182
+ }
3183
+
2126
3184
  /// on any misuse of a JV ref (it reaching a numeric op, a return, a call arg, a block boundary, …),
2127
3185
  /// so a mis-shaped use never miscompiles — it just falls back to the interpreter.
2128
3186
  fn classify_jv_slots(chunk: &Chunk) -> Option<std::collections::HashSet<usize>> {
@@ -2181,17 +3239,53 @@ fn peek_u16(code: &[u8], off: usize) -> Option<u16> {
2181
3239
  /// `LoadConst(Number)` (→ an f64 const) — for the array read/write peepholes. `None` for any other
2182
3240
  /// opcode / a non-numeric const, which makes the caller bail. #187
2183
3241
  #[cfg(not(target_arch = "wasm32"))]
3242
+ /// #168: the exact i64 for a `StoreLocal` into an int slot whose value arrived as a plain F64
3243
+ /// push — the classifier-approved case is a linearly-preceding `LoadConst` of an integral
3244
+ /// Number in [-2^31, 2^32) (excluding `-0`). `store_ip` points AT the StoreLocal opcode; the
3245
+ /// cfg builder's empty-stack-at-block-boundary discipline guarantees the linear predecessor is
3246
+ /// the dynamic value producer. Any other shape → `None` → the caller bails the whole compile.
3247
+ fn int_const_i64(
3248
+ bcx: &mut FunctionBuilder,
3249
+ chunk: &Chunk,
3250
+ code: &[u8],
3251
+ store_ip: usize,
3252
+ ) -> Option<ClifValue> {
3253
+ let lc = store_ip.checked_sub(3)?;
3254
+ if Opcode::from_u8(*code.get(lc)?)? != Opcode::LoadConst {
3255
+ return None;
3256
+ }
3257
+ match chunk.constants.get(peek_u16(code, lc + 1)? as usize)? {
3258
+ Constant::Number(n)
3259
+ if n.fract() == 0.0
3260
+ && *n >= -(2f64.powi(31))
3261
+ && *n < 2f64.powi(32)
3262
+ && n.to_bits() != (-0f64).to_bits() =>
3263
+ {
3264
+ Some(bcx.ins().iconst(types::I64, *n as i64))
3265
+ }
3266
+ _ => None,
3267
+ }
3268
+ }
3269
+
2184
3270
  fn read_simple_operand(
2185
3271
  bcx: &mut FunctionBuilder,
2186
3272
  code: &[u8],
2187
3273
  at: usize,
2188
3274
  chunk: &Chunk,
2189
3275
  vars: &[Variable],
3276
+ int_slots: &std::collections::HashSet<usize>,
2190
3277
  ) -> Option<ClifValue> {
2191
3278
  match Opcode::from_u8(*code.get(at)?)? {
2192
3279
  Opcode::LoadLocal => {
2193
3280
  let s = peek_u16(code, at + 1)? as usize;
2194
- Some(bcx.use_var(*vars.get(s)?))
3281
+ let v = bcx.use_var(*vars.get(s)?);
3282
+ // #168: an int slot's Variable is i64 (exact number) — materialize to the f64 this
3283
+ // fast path assumes. Without this, the i64 value would flow into f64 instructions
3284
+ // and trip the cranelift verifier (a crash, not a bail).
3285
+ if int_slots.contains(&s) {
3286
+ return Some(bcx.ins().fcvt_from_sint(types::F64, v));
3287
+ }
3288
+ Some(v)
2195
3289
  }
2196
3290
  Opcode::LoadConst => {
2197
3291
  let ci = peek_u16(code, at + 1)? as usize;
@@ -2233,6 +3327,8 @@ fn build_body_cfg(
2233
3327
  recur_guard: bool,
2234
3328
  // #186: the imported `tish_math_call` host fn, for lowering `Math.<fn>` (`MathUnary`) intrinsics.
2235
3329
  math_fref: cranelift::codegen::ir::FuncRef,
3330
+ // #203: the imported `tish_math_binary_call` host fn, for lowering `MathBinary` (max/min/pow/atan2).
3331
+ math_binary_fref: cranelift::codegen::ir::FuncRef,
2236
3332
  // #189: `Some` when the function has local `f64` arrays — carries the `tish_jv_*` `FuncRef`s + the
2237
3333
  // JV slot set. Those slots become `i64` handle Variables and their array ops lower to `tish_jv_*`
2238
3334
  // calls; an out-of-bounds index sets the per-thread deopt flag the wrapper re-interprets on.
@@ -2248,6 +3344,7 @@ fn build_body_cfg(
2248
3344
  }
2249
3345
  // #187: slots that hold a boolean (represented as `f64` 0/1). A `LoadLocal` of one carries
2250
3346
  // `is_bool` so a diverging `bool === number` compare or a `return bool` bails to the interpreter.
3347
+ let int_slots = classify_int_slots(chunk, arity);
2251
3348
  let bool_slots = if jit_bool_slots_enabled() {
2252
3349
  classify_bool_slots(chunk)
2253
3350
  } else {
@@ -2263,9 +3360,22 @@ fn build_body_cfg(
2263
3360
  leaders.insert(0);
2264
3361
  let mut has_loop = false;
2265
3362
  let mut has_self_call = false;
3363
+ // #203: `JumpIfFalse` ip → (then_end, else_start, merge) for each clean ternary lowered inline.
3364
+ let mut ternaries: HashMap<usize, (usize, usize, usize)> = HashMap::new();
2266
3365
  let mut ip = 0;
2267
3366
  while ip < code.len() {
2268
3367
  let op = Opcode::from_u8(code[ip])?;
3368
+ // #203: a clean ternary `cond ? A : B` is lowered branch-free (an inline `select`), so record
3369
+ // it and SKIP the whole span — none of its internal targets become block leaders (no orphan
3370
+ // blocks). Plain-numeric functions only (no JV arrays / array-mode); `TISH_JIT_TERNARY=0` off.
3371
+ if op == Opcode::JumpIfFalse && jit_ternary_enabled() && jv.is_none() && array_mask == 0 {
3372
+ if let Some(span) = ternary_span(code, ip) {
3373
+ let (_then_end, _else_start, merge) = span;
3374
+ ternaries.insert(ip, span);
3375
+ ip = merge;
3376
+ continue;
3377
+ }
3378
+ }
2269
3379
  // For a JV function the array opcodes (NewArray/GetMember/GetIndex/SetIndex/Call) are valid
2270
3380
  // and must be sized via the full instruction table; the translate loop then validates each is
2271
3381
  // a real JV op (else it bails). Non-JV functions keep the strict `op_size` whitelist.
@@ -2358,11 +3468,12 @@ fn build_body_cfg(
2358
3468
  };
2359
3469
  let body_start = *blocks.get(&0)?; // `entry` normally; `body_block` when guarded
2360
3470
 
2361
- // 2. A Variable per slot, all defined at entry so every path defines them. A JV slot (#189) holds
2362
- // an arena HANDLE (a `u64`, 0 = null) so its Variable is `i64`, not `f64`.
3471
+ // 2. A Variable per slot, all defined at entry so every path defines them. A JV slot (#189)
3472
+ // holds an arena HANDLE (a `u64`, 0 = null) so its Variable is `i64`, not `f64`; an int
3473
+ // slot (#168) holds an exact integral JS number as `i64` ([`Repr::I64Num`]).
2363
3474
  let vars: Vec<Variable> = (0..num_slots)
2364
3475
  .map(|i| {
2365
- let ty = if jv.is_some_and(|j| j.slots.contains(&i)) {
3476
+ let ty = if jv.is_some_and(|j| j.slots.contains(&i)) || int_slots.contains(&i) {
2366
3477
  types::I64
2367
3478
  } else {
2368
3479
  types::F64
@@ -2398,6 +3509,8 @@ fn build_body_cfg(
2398
3509
  .load(types::F64, MemFlags::new(), numeric_ptr, numeric_i * 8);
2399
3510
  numeric_i += 1;
2400
3511
  v
3512
+ } else if int_slots.contains(&slot) {
3513
+ bcx.ins().iconst(types::I64, 0)
2401
3514
  } else {
2402
3515
  bcx.ins().f64const(0.0)
2403
3516
  };
@@ -2415,6 +3528,8 @@ fn build_body_cfg(
2415
3528
  } else {
2416
3529
  let init = if i < arity {
2417
3530
  params[i]
3531
+ } else if int_slots.contains(&i) {
3532
+ bcx.ins().iconst(types::I64, 0)
2418
3533
  } else {
2419
3534
  bcx.ins().f64const(0.0)
2420
3535
  };
@@ -2424,7 +3539,7 @@ fn build_body_cfg(
2424
3539
  }
2425
3540
 
2426
3541
  // 3. Translate. The operand stack is empty at every block boundary (statement-level control flow).
2427
- let mut stack: Vec<(ClifValue, bool)> = Vec::new();
3542
+ let mut stack: Vec<JV> = Vec::new();
2428
3543
  // #189: JV array handles "in flight" (pushed by `LoadLocal`/`NewArray` of a JV slot, consumed by
2429
3544
  // the very next `GetIndex`/`SetIndex`/`GetMember`), kept off the f64 `stack`; and a pending
2430
3545
  // `arr.push` awaiting its arg + `Call`. Both must be empty at every block boundary.
@@ -2509,13 +3624,16 @@ fn build_body_cfg(
2509
3624
  ip += 3;
2510
3625
  continue;
2511
3626
  }
2512
- let idx_f64 = read_simple_operand(&mut bcx, code, ip + 3, chunk, &vars)?;
3627
+ let idx_f64 =
3628
+ read_simple_operand(&mut bcx, code, ip + 3, chunk, &vars, &int_slots)?;
2513
3629
  // i = idx as usize (saturating: NaN→0, neg→0 — matches the VM's `n as usize`).
2514
3630
  let i = bcx.ins().fcvt_to_uint_sat(types::I64, idx_f64);
2515
3631
  // Read `val` (write only) BEFORE the bounds-check split so its LoadLocal reads the
2516
3632
  // slot's current SSA value in `cur`, not the fresh `cont` block.
2517
3633
  let store_val = if is_write {
2518
- Some(read_simple_operand(&mut bcx, code, ip + 6, chunk, &vars)?)
3634
+ Some(read_simple_operand(
3635
+ &mut bcx, code, ip + 6, chunk, &vars, &int_slots,
3636
+ )?)
2519
3637
  } else {
2520
3638
  None
2521
3639
  };
@@ -2529,19 +3647,28 @@ fn build_body_cfg(
2529
3647
  let addr = bcx.ins().iadd(aptr, off);
2530
3648
  if let Some(val) = store_val {
2531
3649
  bcx.ins().store(MemFlags::new(), val, addr, 0);
2532
- stack.push((val, false)); // assignment yields the value (a Pop usually discards it)
3650
+ stack.push(JV::f64(val)); // assignment yields the value (a Pop usually discards it)
2533
3651
  ip += 11; // LoadLocal(arr) + idx + val + Dup + SetIndex
2534
3652
  } else {
2535
3653
  let val = bcx.ins().load(types::F64, MemFlags::new(), addr, 0);
2536
- stack.push((val, false));
3654
+ stack.push(JV::f64(val));
2537
3655
  ip += 7; // LoadLocal(arr) + idx + GetIndex
2538
3656
  }
2539
3657
  continue;
2540
3658
  }
2541
3659
  let v = *vars.get(slot)?;
2542
- // #187: a bool slot's `f64` 0/1 value carries `is_bool` so downstream equality/return
2543
- // guards fire; a plain numeric slot pushes `false`.
2544
- stack.push((bcx.use_var(v), bool_slots.contains(&slot)));
3660
+ // #187: a bool slot's `f64` 0/1 value carries the Bool repr so downstream
3661
+ // equality/return guards fire; an int slot (#168) pushes its exact-i64 repr
3662
+ // (identity into further int ops, one exact convert at an f64 boundary); a
3663
+ // plain numeric slot pushes F64.
3664
+ let lv = bcx.use_var(v);
3665
+ stack.push(if bool_slots.contains(&slot) {
3666
+ JV::boolean(lv)
3667
+ } else if int_slots.contains(&slot) {
3668
+ JV::i64num(lv)
3669
+ } else {
3670
+ JV::f64(lv)
3671
+ });
2545
3672
  ip += 3;
2546
3673
  }
2547
3674
  Opcode::StoreLocal => {
@@ -2554,15 +3681,43 @@ fn build_body_cfg(
2554
3681
  ip += 3;
2555
3682
  continue;
2556
3683
  }
2557
- let (val, is_bool) = stack.pop()?;
3684
+ let jval = stack.pop()?;
2558
3685
  // #187: a boolean value may be stored only into a slot the pre-pass tagged as a bool
2559
3686
  // slot (represented as `f64` 0/1). Any other bool store → bail (keeps unknown shapes
2560
3687
  // on the interpreter). A numeric store into a bool-tagged slot is fine — the slot is
2561
- // still an `f64`; its `LoadLocal`s just carry `is_bool` (conservatively).
2562
- if is_bool && !bool_slots.contains(&slot) {
3688
+ // still an `f64`; its `LoadLocal`s just carry the Bool repr (conservatively).
3689
+ if jval.is_bool() && !bool_slots.contains(&slot) {
2563
3690
  return None;
2564
3691
  }
2565
3692
  let v = *vars.get(slot)?;
3693
+ if int_slots.contains(&slot) {
3694
+ // #168: an int slot stores the EXACT number as i64 — sign-extend ToInt32
3695
+ // results, zero-extend ToUint32 results, integral constants load exactly.
3696
+ // Anything else reaching here means the optimistic pre-pass mis-tagged the
3697
+ // slot (e.g. a merge point whose linear predecessor differed) → bail the
3698
+ // whole compile; the VM keeps semantics.
3699
+ let iv = match jval.repr {
3700
+ Repr::I32 => bcx.ins().sextend(types::I64, jval.v),
3701
+ Repr::U32 => bcx.ins().uextend(types::I64, jval.v),
3702
+ Repr::I64Num => jval.v,
3703
+ Repr::F64 => {
3704
+ // The classifier only tags int-op/const-preceded stores, but a
3705
+ // constant reaches here as an F64 push — re-derive its exact i64
3706
+ // when it is one of the classifier-approved integral constants.
3707
+ match int_const_i64(&mut bcx, chunk, code, ip) {
3708
+ Some(iv) => iv,
3709
+ None => return None,
3710
+ }
3711
+ }
3712
+ Repr::Bool => return None,
3713
+ };
3714
+ bcx.def_var(v, iv);
3715
+ ip += 3;
3716
+ continue;
3717
+ }
3718
+ // Slots are f64 Variables — materialize an int repr once, at the store (an
3719
+ // `h = ((h<<13)|(h>>>19))>>>0` chain pays exactly ONE convert here, not per op).
3720
+ let val = jv_f64(&mut bcx, jval);
2566
3721
  bcx.def_var(v, val);
2567
3722
  ip += 3;
2568
3723
  }
@@ -2580,10 +3735,11 @@ fn build_body_cfg(
2580
3735
  Opcode::Nop | Opcode::EnterBlock | Opcode::ExitBlock | Opcode::LoopVarsEnd => ip += 1,
2581
3736
  Opcode::LoopVarsBegin => ip += 3,
2582
3737
  Opcode::Return => {
2583
- let (v, is_bool) = stack.pop()?;
2584
- if is_bool {
3738
+ let ret = stack.pop()?;
3739
+ if ret.is_bool() {
2585
3740
  return None;
2586
3741
  }
3742
+ let v = jv_f64(&mut bcx, ret);
2587
3743
  // #189: free every JV array (return its arena slot to the free list) before leaving
2588
3744
  // the frame — handle 0 (a slot not yet allocated on this path) is a no-op. This is why
2589
3745
  // JV arrays must never escape.
@@ -2617,12 +3773,73 @@ fn build_body_cfg(
2617
3773
  terminated = true;
2618
3774
  ip += 3;
2619
3775
  }
3776
+ // #203: a clean ternary detected in the leader scan → branch-free `select`, INLINE (no
3777
+ // blocks). The stack BELOW the cond stays untouched (both arms push/pop their one result
3778
+ // on top of it), so no operand crosses a block boundary — this is the only reason
3779
+ // build_body_cfg's empty-stack-at-boundary invariant is not violated.
3780
+ Opcode::JumpIfFalse if ternaries.contains_key(&ip) => {
3781
+ let (then_end, else_start, merge) = *ternaries.get(&ip).unwrap();
3782
+ let cond = stack.pop()?;
3783
+ let cond = jv_f64(&mut bcx, cond);
3784
+ let base = stack.len();
3785
+ // THEN arm: [ip+3, then_end) → exactly one value.
3786
+ if !emit_ternary_arm(
3787
+ &mut bcx,
3788
+ chunk,
3789
+ code,
3790
+ ip + 3,
3791
+ then_end,
3792
+ &mut stack,
3793
+ &vars,
3794
+ &int_slots,
3795
+ &bool_slots,
3796
+ math_fref,
3797
+ math_binary_fref,
3798
+ ) || stack.len() != base + 1
3799
+ {
3800
+ return None;
3801
+ }
3802
+ let then_jv = stack.pop()?;
3803
+ let then_v = jv_f64(&mut bcx, then_jv);
3804
+ // ELSE arm: [else_start, merge) → exactly one value.
3805
+ if !emit_ternary_arm(
3806
+ &mut bcx,
3807
+ chunk,
3808
+ code,
3809
+ else_start,
3810
+ merge,
3811
+ &mut stack,
3812
+ &vars,
3813
+ &int_slots,
3814
+ &bool_slots,
3815
+ math_fref,
3816
+ math_binary_fref,
3817
+ ) || stack.len() != base + 1
3818
+ {
3819
+ return None;
3820
+ }
3821
+ let else_jv = stack.pop()?;
3822
+ let else_v = jv_f64(&mut bcx, else_jv);
3823
+ // Both arms must agree Bool-vs-Number (one `result_bool` shape, like build_body).
3824
+ if then_jv.is_bool() != else_jv.is_bool() {
3825
+ return None;
3826
+ }
3827
+ let falsy = falsy_flag(&mut bcx, cond);
3828
+ let sel = bcx.ins().select(falsy, else_v, then_v);
3829
+ stack.push(if then_jv.is_bool() {
3830
+ JV::boolean(sel)
3831
+ } else {
3832
+ JV::f64(sel)
3833
+ });
3834
+ ip = merge;
3835
+ }
2620
3836
  Opcode::JumpIfFalse => {
2621
3837
  let off = peek_u16(code, ip + 1)? as i16 as isize;
2622
- let (cond, _) = stack.pop()?;
3838
+ let cond = stack.pop()?;
2623
3839
  if !stack.is_empty() {
2624
3840
  return None; // non-empty stack ⇒ ternary shape ⇒ leave to build_body / VM
2625
3841
  }
3842
+ let cond = jv_f64(&mut bcx, cond);
2626
3843
  let falsy = falsy_flag(&mut bcx, cond);
2627
3844
  let target = *blocks.get(&(((ip + 3) as isize + off).max(0) as usize))?;
2628
3845
  let fallthrough = *blocks.get(&(ip + 3))?;
@@ -2651,10 +3868,12 @@ fn build_body_cfg(
2651
3868
  3, // 2^3 = 8-byte alignment for f64
2652
3869
  ));
2653
3870
  let arg_start = stack.len() - num_numeric;
2654
- for (j, (v, is_bool)) in stack.drain(arg_start..).enumerate() {
2655
- if is_bool {
3871
+ let args: Vec<JV> = stack.drain(arg_start..).collect();
3872
+ for (j, jv) in args.into_iter().enumerate() {
3873
+ if jv.is_bool() {
2656
3874
  return None; // a bool numeric arg doesn't match the f64 ABI
2657
3875
  }
3876
+ let v = jv_f64(&mut bcx, jv);
2658
3877
  bcx.ins().stack_store(v, slot, (j * 8) as i32);
2659
3878
  }
2660
3879
  let num_ptr = bcx.ins().stack_addr(types::I64, slot, 0);
@@ -2667,7 +3886,8 @@ fn build_body_cfg(
2667
3886
  call_args.push(gp);
2668
3887
  }
2669
3888
  let call = bcx.ins().call(sref, &call_args);
2670
- stack.push((bcx.inst_results(call)[0], false));
3889
+ let res = bcx.inst_results(call)[0];
3890
+ stack.push(JV::f64(res));
2671
3891
  ip += 3;
2672
3892
  }
2673
3893
  Opcode::SelfCall => {
@@ -2679,12 +3899,13 @@ fn build_body_cfg(
2679
3899
  return None;
2680
3900
  }
2681
3901
  let arg_start = stack.len() - arity;
3902
+ let args: Vec<JV> = stack.drain(arg_start..).collect();
2682
3903
  let mut call_args = Vec::with_capacity(arity);
2683
- for (v, is_bool) in stack.drain(arg_start..) {
2684
- if is_bool {
3904
+ for jv in args {
3905
+ if jv.is_bool() {
2685
3906
  return None; // boolean args don't match the f64 ABI
2686
3907
  }
2687
- call_args.push(v);
3908
+ call_args.push(jv_f64(&mut bcx, jv));
2688
3909
  }
2689
3910
  // #381: thread the RecurGuard pointer through the recursive call so every level
2690
3911
  // re-checks the stack at its entry. Present iff this function was compiled guarded.
@@ -2693,16 +3914,30 @@ fn build_body_cfg(
2693
3914
  }
2694
3915
  let call = bcx.ins().call(sref, &call_args);
2695
3916
  let result = bcx.inst_results(call)[0];
2696
- stack.push((result, false));
3917
+ stack.push(JV::f64(result));
2697
3918
  ip += 3;
2698
3919
  }
2699
3920
  Opcode::MathUnary => {
2700
3921
  // #186 — `Math.<fn>(x)`: pop the arg, emit native op / host call, push the result.
2701
3922
  let id = peek_u16(code, ip + 1)?;
2702
3923
  let mfn = MathUnaryFn::from_u16(id)?;
2703
- let (x, _) = stack.pop()?;
3924
+ let x = stack.pop()?;
3925
+ let x = jv_f64(&mut bcx, x);
2704
3926
  let r = emit_math_unary(&mut bcx, math_fref, mfn, x);
2705
- stack.push((r, false));
3927
+ stack.push(JV::f64(r));
3928
+ ip += 3;
3929
+ }
3930
+ Opcode::MathBinary => {
3931
+ // #203 — `Math.<fn>(a, b)`: pop b, pop a, host-call, push.
3932
+ let id = peek_u16(code, ip + 1)?;
3933
+ let b = stack.pop()?;
3934
+ let a = stack.pop()?;
3935
+ let b = jv_f64(&mut bcx, b);
3936
+ let a = jv_f64(&mut bcx, a);
3937
+ let idc = bcx.ins().iconst(types::I32, id as i64);
3938
+ let call = bcx.ins().call(math_binary_fref, &[idc, a, b]);
3939
+ let r = bcx.inst_results(call)[0];
3940
+ stack.push(JV::f64(r));
2706
3941
  ip += 3;
2707
3942
  }
2708
3943
  // #189 — local-array ops, only for JV functions (`jv` = `Some`). Each consumes the array
@@ -2719,25 +3954,29 @@ fn build_body_cfg(
2719
3954
  }
2720
3955
  Opcode::GetIndex if !jv_pending.is_empty() => {
2721
3956
  let jvc = jv?;
2722
- let idx = stack.pop()?.0;
3957
+ let idx = stack.pop()?;
3958
+ let idx = jv_f64(&mut bcx, idx);
2723
3959
  let handle = jv_pending.pop()?;
2724
3960
  // `idx as usize` (saturating: NaN/neg → 0), matching the VM's index coercion. OOB sets
2725
3961
  // the per-thread deopt flag inside `tish_jv_get` and returns NaN.
2726
3962
  let i = bcx.ins().fcvt_to_uint_sat(types::I64, idx);
2727
3963
  let call = bcx.ins().call(jvc.get, &[handle, i]);
2728
- stack.push((bcx.inst_results(call)[0], false));
3964
+ let res = bcx.inst_results(call)[0];
3965
+ stack.push(JV::f64(res));
2729
3966
  ip += 1;
2730
3967
  }
2731
3968
  Opcode::SetIndex if !jv_pending.is_empty() => {
2732
3969
  let jvc = jv?;
2733
3970
  // Stack: [ (array→jv_pending), idx, val, dup_val ]. `Dup` left `dup_val` == `val`.
2734
- let dup_val = stack.pop()?.0;
2735
- let _val = stack.pop()?.0;
2736
- let idx = stack.pop()?.0;
3971
+ let dup_jv = stack.pop()?;
3972
+ let _val = stack.pop()?;
3973
+ let idx = stack.pop()?;
3974
+ let dup_val = jv_f64(&mut bcx, dup_jv);
3975
+ let idx = jv_f64(&mut bcx, idx);
2737
3976
  let handle = jv_pending.pop()?;
2738
3977
  let i = bcx.ins().fcvt_to_uint_sat(types::I64, idx);
2739
3978
  bcx.ins().call(jvc.set, &[handle, i, dup_val]); // OOB → deopt flag inside tish_jv_set
2740
- stack.push((dup_val, false)); // assignment yields the value
3979
+ stack.push(JV::f64(dup_val)); // assignment yields the value
2741
3980
  ip += 1;
2742
3981
  }
2743
3982
  Opcode::GetMember if !jv_pending.is_empty() => {
@@ -2748,7 +3987,8 @@ fn build_body_cfg(
2748
3987
  "length" => {
2749
3988
  let call = bcx.ins().call(jvc.len, &[ptr]);
2750
3989
  let len_i = bcx.inst_results(call)[0];
2751
- stack.push((bcx.ins().fcvt_from_uint(types::F64, len_i), false));
3990
+ let len_f = bcx.ins().fcvt_from_uint(types::F64, len_i);
3991
+ stack.push(JV::f64(len_f));
2752
3992
  }
2753
3993
  "push" => pending_push = Some(ptr),
2754
3994
  _ => return None, // any other member of a JV array → bail
@@ -2761,12 +4001,14 @@ fn build_body_cfg(
2761
4001
  return None; // `push` takes exactly one arg in the JV fast path
2762
4002
  }
2763
4003
  let ptr = pending_push.take()?;
2764
- let arg = stack.pop()?.0;
4004
+ let arg = stack.pop()?;
4005
+ let arg = jv_f64(&mut bcx, arg);
2765
4006
  bcx.ins().call(jvc.push, &[ptr, arg]);
2766
4007
  // `Array.push` returns the new length.
2767
4008
  let call = bcx.ins().call(jvc.len, &[ptr]);
2768
4009
  let len_i = bcx.inst_results(call)[0];
2769
- stack.push((bcx.ins().fcvt_from_uint(types::F64, len_i), false));
4010
+ let len_f = bcx.ins().fcvt_from_uint(types::F64, len_i);
4011
+ stack.push(JV::f64(len_f));
2770
4012
  ip += 3;
2771
4013
  }
2772
4014
  // #187: `LoadVar name` where `name` is a resolved directly-callable callee — stage it for the
@@ -2792,15 +4034,17 @@ fn build_body_cfg(
2792
4034
  return None;
2793
4035
  }
2794
4036
  let arg_start = stack.len() - callee_arity as usize;
4037
+ let args: Vec<JV> = stack.drain(arg_start..).collect();
2795
4038
  let mut call_args = Vec::with_capacity(callee_arity as usize);
2796
- for (v, is_bool) in stack.drain(arg_start..) {
2797
- if is_bool {
4039
+ for jv in args {
4040
+ if jv.is_bool() {
2798
4041
  return None; // a bool arg doesn't match the callee's f64 ABI
2799
4042
  }
2800
- call_args.push(v);
4043
+ call_args.push(jv_f64(&mut bcx, jv));
2801
4044
  }
2802
4045
  let call = bcx.ins().call(fref, &call_args);
2803
- stack.push((bcx.inst_results(call)[0], false));
4046
+ let res = bcx.inst_results(call)[0];
4047
+ stack.push(JV::f64(res));
2804
4048
  ip += 3;
2805
4049
  }
2806
4050
  // #187: a VOID array-mode function's implicit `return null` (the fall-through of a
@@ -2850,6 +4094,20 @@ fn build_body_cfg(
2850
4094
  terminated = true; // the following `Return` (and any dead tail) is now skipped
2851
4095
  ip += 3;
2852
4096
  }
4097
+ _ if is_binop_pow(op, code, ip) => {
4098
+ // #203: `a ** b` → host call to `tish_math_binary_call(Pow, a, b)` (== VM's `powf`).
4099
+ let r = stack.pop()?;
4100
+ let l = stack.pop()?;
4101
+ let r = jv_f64(&mut bcx, r);
4102
+ let l = jv_f64(&mut bcx, l);
4103
+ let idc = bcx
4104
+ .ins()
4105
+ .iconst(types::I32, tishlang_bytecode::MathBinaryFn::Pow as i64);
4106
+ let call = bcx.ins().call(math_binary_fref, &[idc, l, r]);
4107
+ let res = bcx.inst_results(call)[0];
4108
+ stack.push(JV::f64(res));
4109
+ ip += 3;
4110
+ }
2853
4111
  _ => match emit_simple_op(&mut bcx, chunk, code, &mut ip, &mut stack, &params, arity) {
2854
4112
  SimpleOp::Handled(_) => {}
2855
4113
  _ => return None, // LoadConst/BinOp/UnaryOp handled; anything else → VM
@@ -2888,9 +4146,10 @@ fn build_body(
2888
4146
  let params: Vec<ClifValue> = bcx.block_params(entry).to_vec();
2889
4147
 
2890
4148
  let code = &chunk.code;
2891
- // Each entry is (clif f64 value, is_bool). `is_bool` marks comparison/`!`
2892
- // results (logical 0.0/1.0) so the final value boxes as Bool, not Number.
2893
- let mut stack: Vec<(ClifValue, bool)> = Vec::new();
4149
+ // Each entry is a typed [`JV`]. The Bool repr marks comparison/`!` results
4150
+ // (logical 0.0/1.0) so the final value boxes as Bool, not Number; integer
4151
+ // reprs materialize to f64 at the Return / select boundaries below.
4152
+ let mut stack: Vec<JV> = Vec::new();
2894
4153
  let mut ip = 0usize;
2895
4154
  let mut result: Option<bool> = None;
2896
4155
 
@@ -2903,15 +4162,17 @@ fn build_body(
2903
4162
  let op = Opcode::from_u8(code[ip])?;
2904
4163
  match op {
2905
4164
  Opcode::Return => {
2906
- let (v, is_bool) = stack.pop()?;
4165
+ let jv = stack.pop()?;
4166
+ let v = jv_f64(&mut bcx, jv);
2907
4167
  bcx.ins().return_(&[v]);
2908
- result = Some(is_bool); // first Return ends a (sub)path
4168
+ result = Some(jv.is_bool()); // first Return ends a (sub)path
2909
4169
  break;
2910
4170
  }
2911
4171
  // Ternary `cond ? A : B` → `select`. Both arms must be branch-free numeric
2912
4172
  // sub-sequences, each pushing exactly one value, with matching is_bool.
2913
4173
  Opcode::JumpIfFalse => {
2914
- let (cond, _) = stack.pop()?;
4174
+ let cond = stack.pop()?;
4175
+ let cond = jv_f64(&mut bcx, cond);
2915
4176
  let mut p = ip + 1;
2916
4177
  let off = read_u16(code, &mut p)? as i16 as isize; // p now past the operand
2917
4178
  let else_target = (p as isize + off).max(0) as usize;
@@ -2938,7 +4199,8 @@ fn build_body(
2938
4199
  if else_target != jp || stack.len() != base + 1 {
2939
4200
  return None;
2940
4201
  }
2941
- let (then_v, then_b) = stack.pop()?;
4202
+ let then_jv = stack.pop()?;
4203
+ let then_v = jv_f64(&mut bcx, then_jv);
2942
4204
 
2943
4205
  // ELSE arm: straight-line ops from `jp` up to the merge point.
2944
4206
  let mut eip = jp;
@@ -2953,15 +4215,20 @@ fn build_body(
2953
4215
  if eip != merge_target || stack.len() != base + 1 {
2954
4216
  return None;
2955
4217
  }
2956
- let (else_v, else_b) = stack.pop()?;
4218
+ let else_jv = stack.pop()?;
4219
+ let else_v = jv_f64(&mut bcx, else_jv);
2957
4220
  // One result_bool per function: arms must agree on Bool-vs-Number.
2958
- if then_b != else_b {
4221
+ if then_jv.is_bool() != else_jv.is_bool() {
2959
4222
  return None;
2960
4223
  }
2961
4224
 
2962
4225
  let falsy = falsy_flag(&mut bcx, cond);
2963
4226
  let sel = bcx.ins().select(falsy, else_v, then_v);
2964
- stack.push((sel, then_b));
4227
+ stack.push(if then_jv.is_bool() {
4228
+ JV::boolean(sel)
4229
+ } else {
4230
+ JV::f64(sel)
4231
+ });
2965
4232
  ip = merge_target;
2966
4233
  }
2967
4234
  // #187: a `function name(x) { … }` block body wraps its statements in EnterBlock/ExitBlock
@@ -3602,7 +4869,7 @@ mod tests {
3602
4869
  assert!(!lf.used_slots.is_empty() && !lf.exits.is_empty());
3603
4870
  let mut buf = vec![0.0f64; lf.used_slots.len()];
3604
4871
  let mut deopt = 0u8;
3605
- let exit = lf.call(&mut buf, &mut deopt);
4872
+ let exit = lf.call(&mut buf, &mut [], &mut deopt);
3606
4873
  assert!((exit as usize) < lf.exits.len(), "exit id in range");
3607
4874
  assert_eq!(deopt, 0, "v1 region never sets the deopt flag");
3608
4875
  let mut got = buf.clone();
@@ -3615,12 +4882,13 @@ mod tests {
3615
4882
  }
3616
4883
 
3617
4884
  /// #190 — a loop that touches a non-slot value (a general call) is not pure-numeric slot math, so
3618
- /// the region must be rejected (negative-cached) and the VM keeps interpreting. `Math.max` is a
3619
- /// 2-arg call (NOT the #186 unary intrinsic), so it stays a `Call` the region must reject.
4885
+ /// the region must be rejected (negative-cached) and the VM keeps interpreting. A user-function
4886
+ /// call is such a value. (`Math.max`/`min`/`pow` used to serve as the example here, but they are
4887
+ /// now the #203 `MathBinary` intrinsic and DO OSR-compile — see `osr_region_handles_math_binary`.)
3620
4888
  #[test]
3621
4889
  fn osr_region_rejects_calls() {
3622
4890
  let chunk = top_chunk(
3623
- "let a = 0.0\nfor (let i = 0; i < 100; i = i + 1) { a = a + Math.max(i, 2.0) }\n",
4891
+ "function g(x) { return x + 1.0 }\nlet a = 0.0\nfor (let i = 0; i < 100; i = i + 1) { a = a + g(i) }\n",
3624
4892
  );
3625
4893
  let (header, end) = first_region(&chunk);
3626
4894
  assert!(
@@ -3629,6 +4897,20 @@ mod tests {
3629
4897
  );
3630
4898
  }
3631
4899
 
4900
+ /// #203 — a loop whose only "call" is a 2-arg `Math.<fn>` (the `MathBinary` intrinsic) IS
4901
+ /// pure-numeric slot math, so the OSR region compiles it (the win for clamp/pow kernels).
4902
+ #[test]
4903
+ fn osr_region_handles_math_binary() {
4904
+ let chunk = top_chunk(
4905
+ "let a = 0.0\nfor (let i = 0; i < 100; i = i + 1) { a = a + Math.max(i, 2.0) }\n",
4906
+ );
4907
+ let (header, end) = first_region(&chunk);
4908
+ assert!(
4909
+ try_compile_loop(&chunk, header, end).is_some(),
4910
+ "a loop with only a 2-arg Math intrinsic must OSR-compile"
4911
+ );
4912
+ }
4913
+
3632
4914
  /// #190 — a loop with nested branches compiles and computes correctly through multiple blocks:
3633
4915
  /// `while (i < 20) { if (i % 2 == 0) s = s + i; i = i + 1 }` → s = sum of evens in 0..19 = 90.
3634
4916
  #[test]
@@ -3641,7 +4923,7 @@ mod tests {
3641
4923
  try_compile_loop(&chunk, header, end).expect("branchy numeric loop must OSR-compile");
3642
4924
  let mut buf = vec![0.0f64; lf.used_slots.len()];
3643
4925
  let mut deopt = 0u8;
3644
- lf.call(&mut buf, &mut deopt);
4926
+ lf.call(&mut buf, &mut [], &mut deopt);
3645
4927
  let mut got = buf.clone();
3646
4928
  got.sort_by(|a, b| a.partial_cmp(b).unwrap());
3647
4929
  assert_eq!(got, vec![20.0, 90.0], "s=90 (0+2+…+18), i=20");