@tishlang/tish-lsp 2.38.0 → 2.43.8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/Cargo.toml +4 -1
- package/bin/tish-lsp +0 -0
- package/crates/tish/src/cli_help.rs +28 -0
- package/crates/tish/src/main.rs +41 -5
- package/crates/tish/tests/scope_merge.rs +79 -0
- package/crates/tish_build_utils/src/lib.rs +27 -1
- package/crates/tish_builtins/Cargo.toml +14 -6
- package/crates/tish_builtins/src/array.rs +130 -54
- package/crates/tish_builtins/src/collections.rs +9 -5
- package/crates/tish_builtins/src/construct.rs +7 -1
- package/crates/tish_builtins/src/date.rs +15 -1
- package/crates/tish_builtins/src/globals.rs +38 -3
- package/crates/tish_builtins/src/helpers.rs +7 -3
- package/crates/tish_builtins/src/iterator.rs +5 -1
- package/crates/tish_builtins/src/lib.rs +3 -0
- package/crates/tish_builtins/src/math.rs +14 -0
- package/crates/tish_builtins/src/number.rs +72 -6
- package/crates/tish_builtins/src/object.rs +6 -1
- package/crates/tish_builtins/src/string.rs +13 -2
- package/crates/tish_builtins/src/symbol.rs +17 -17
- package/crates/tish_builtins/src/typedarrays.rs +9 -2
- package/crates/tish_bytecode/src/compiler.rs +76 -2
- package/crates/tish_bytecode/src/lib.rs +1 -1
- package/crates/tish_bytecode/src/opcode.rs +77 -2
- package/crates/tish_compile/src/codegen.rs +812 -47
- package/crates/tish_compile/src/infer.rs +37 -41
- package/crates/tish_compile/src/lib.rs +13 -1
- package/crates/tish_compile/src/platform_resolve.rs +361 -0
- package/crates/tish_compile/src/resolve.rs +148 -10
- package/crates/tish_compile/src/schemes.rs +278 -0
- package/crates/tish_compile/src/types.rs +141 -2
- package/crates/tish_compile/tests/perf_codegen_320.rs +2 -2
- package/crates/tish_compile/tests/perf_codegen_module_const_forof.rs +3 -1
- package/crates/tish_compile/tests/platform_resolve_cli.rs +164 -0
- package/crates/tish_compile/tests/regr_gba_numerics_gated.rs +58 -0
- package/crates/tish_compiler_wasm/Cargo.toml +1 -0
- package/crates/tish_compiler_wasm/src/resolve_virtual.rs +56 -6
- package/crates/tish_core/Cargo.toml +39 -4
- package/crates/tish_core/src/compat.rs +506 -0
- package/crates/tish_core/src/console_style.rs +19 -1
- package/crates/tish_core/src/json.rs +34 -10
- package/crates/tish_core/src/lib.rs +121 -25
- package/crates/tish_core/src/macros.rs +1 -1
- package/crates/tish_core/src/shape.rs +4 -2
- package/crates/tish_core/src/uri.rs +8 -1
- package/crates/tish_core/src/value.rs +182 -54
- package/crates/tish_core/src/vmref.rs +17 -5
- package/crates/tish_eval/Cargo.toml +2 -0
- package/crates/tish_eval/src/eval.rs +90 -3
- package/crates/tish_eval/src/value.rs +8 -0
- package/crates/tish_eval/src/value_convert.rs +316 -21
- package/crates/tish_lsp/Cargo.toml +3 -1
- package/crates/tish_lsp/src/import_goto.rs +41 -14
- package/crates/tish_lsp/src/main.rs +18 -0
- package/crates/tish_native/src/build.rs +246 -0
- package/crates/tish_native/src/config.rs +20 -0
- package/crates/tish_native/src/lib.rs +16 -6
- package/crates/tish_parser/src/lib.rs +28 -0
- package/crates/tish_parser/src/parser.rs +25 -4
- package/crates/tish_runtime/src/http_hyper.rs +70 -0
- package/crates/tish_runtime/src/lib.rs +55 -10
- package/crates/tish_runtime/src/ws.rs +94 -21
- package/crates/tish_runtime_gba/Cargo.lock +837 -0
- package/crates/tish_runtime_gba/Cargo.toml +16 -0
- package/crates/tish_runtime_gba/src/gba.rs +211 -0
- package/crates/tish_runtime_gba/src/lib.rs +620 -0
- package/crates/tish_vm/src/jit.rs +1446 -164
- package/crates/tish_vm/src/vm.rs +535 -173
- package/justfile +9 -0
- package/package.json +1 -1
- package/platform/darwin-arm64/tish-lsp +0 -0
- package/platform/darwin-x64/tish-lsp +0 -0
- package/platform/linux-arm64/tish-lsp +0 -0
- package/platform/linux-x64/tish-lsp +0 -0
- package/platform/win32-x64/tish-lsp.exe +0 -0
|
@@ -55,6 +55,12 @@ pub struct LoopFn {
|
|
|
55
55
|
/// region). The region always has ≥1 exit (an exit-less region would be an uninterruptible native
|
|
56
56
|
/// loop, so compilation bails).
|
|
57
57
|
pub exits: Vec<usize>,
|
|
58
|
+
/// #203 — array live-in slots `(chunk slot, writable)` in marshalling order (ascending slot). EMPTY
|
|
59
|
+
/// ⇒ the pure-numeric 2-pointer ABI `(slots, deopt)` (unchanged). NON-EMPTY ⇒ the 3-pointer array
|
|
60
|
+
/// ABI `(slots, handles, deopt)`: `run_osr` extracts each array live-in to a scratch `Vec<f64>`,
|
|
61
|
+
/// passes it as an [`ArrayHandle`], and (for a `writable` slot) copies the scratch back into the
|
|
62
|
+
/// caller's `Value::Array` — ONLY on a clean, non-deopt exit.
|
|
63
|
+
pub array_slots: Vec<(u16, bool)>,
|
|
58
64
|
}
|
|
59
65
|
|
|
60
66
|
// SAFETY: identical to `NumericFn` — `ptr` is immutable executable code in the process-global,
|
|
@@ -64,17 +70,27 @@ unsafe impl Send for LoopFn {}
|
|
|
64
70
|
unsafe impl Sync for LoopFn {}
|
|
65
71
|
|
|
66
72
|
impl LoopFn {
|
|
67
|
-
/// Run the region. `buf` holds the live-ins (`used_slots.len()` `f64`s), updated in place
|
|
68
|
-
/// live-outs on return. `
|
|
69
|
-
///
|
|
70
|
-
///
|
|
73
|
+
/// Run the region. `buf` holds the numeric live-ins (`used_slots.len()` `f64`s), updated in place
|
|
74
|
+
/// with the live-outs on return. `handles` holds the array live-ins (`array_slots.len()`
|
|
75
|
+
/// [`ArrayHandle`]s, in `array_slots` order) — empty for a pure-numeric region. `deopt` is a
|
|
76
|
+
/// 1-byte flag the region sets on an int-slot live-in miss (#514) or an out-of-bounds array index
|
|
77
|
+
/// (#203); on `deopt != 0` the caller discards `buf` AND the scratch behind `handles` and
|
|
78
|
+
/// re-interprets. Returns the exit id. Safe wrapper — the raw-pointer transmute (same soundness as
|
|
79
|
+
/// [`NumericFn::call`]: immutable native code with a fixed C ABI) is confined here.
|
|
71
80
|
#[inline]
|
|
72
|
-
pub fn call(&self, buf: &mut [f64], deopt: &mut u8) -> i32 {
|
|
73
|
-
// SAFETY: `ptr` is immutable executable code compiled for exactly this
|
|
74
|
-
//
|
|
81
|
+
pub fn call(&self, buf: &mut [f64], handles: &mut [ArrayHandle], deopt: &mut u8) -> i32 {
|
|
82
|
+
// SAFETY: `ptr` is immutable executable code compiled for exactly this ABI (2-pointer when
|
|
83
|
+
// `array_slots` is empty, else 3-pointer); `buf`/`handles`/`deopt` are valid for the call and
|
|
84
|
+
// the region only touches `buf[0..used_slots]` + the handles' backing scratch.
|
|
75
85
|
unsafe {
|
|
76
|
-
|
|
77
|
-
|
|
86
|
+
if self.array_slots.is_empty() {
|
|
87
|
+
let f: extern "C" fn(*mut f64, *mut u8) -> i32 = std::mem::transmute(self.ptr);
|
|
88
|
+
f(buf.as_mut_ptr(), deopt as *mut u8)
|
|
89
|
+
} else {
|
|
90
|
+
let f: extern "C" fn(*mut f64, *mut ArrayHandle, *mut u8) -> i32 =
|
|
91
|
+
std::mem::transmute(self.ptr);
|
|
92
|
+
f(buf.as_mut_ptr(), handles.as_mut_ptr(), deopt as *mut u8)
|
|
93
|
+
}
|
|
78
94
|
}
|
|
79
95
|
}
|
|
80
96
|
}
|
|
@@ -180,6 +196,30 @@ pub fn jit_jv_enabled() -> bool {
|
|
|
180
196
|
})
|
|
181
197
|
}
|
|
182
198
|
|
|
199
|
+
/// #203 — bounded array-index (`arr[i]` read / `arr[i] = v` write) inside an OSR loop **region**
|
|
200
|
+
/// (the matmul lever). A hot loop that indexes a numeric `Value::Array` live-in (created BEFORE the
|
|
201
|
+
/// region, e.g. matmul's `a`/`b`/`c`) is marshalled through the [`ArrayHandle`] inline ABI: each
|
|
202
|
+
/// array live-in is extracted to a scratch `Vec<f64>` and passed as `(ptr,len)`; `GetIndex`/`SetIndex`
|
|
203
|
+
/// lower to a bounds-checked native load/store over a COMPUTED index (`a[i*N+k]`), and writable arrays
|
|
204
|
+
/// are copied back only on a clean (non-deopt) exit. Any out-of-bounds index sets the region deopt
|
|
205
|
+
/// flag and the VM re-interprets from the pristine pre-region state (scratch discarded, real array
|
|
206
|
+
/// untouched) — identical to the entry int-slot guard, so a mid-region bail never commits a partial
|
|
207
|
+
/// result. Additive + bail-safe: a misclassified slot fails the runtime `Value::Array` marshalling
|
|
208
|
+
/// check (→ interpret), and an unhandleable region shape is simply not compiled.
|
|
209
|
+
///
|
|
210
|
+
/// **Default ON** (validated: matmul 87×, zero perf regressions across the 29-fixture gauntlet, parity
|
|
211
|
+
/// interp==vm==node, all six backends agree, adversarial review found + fixed the one bool-store bug);
|
|
212
|
+
/// `TISH_JIT_OSR_ARRAY=0` disables it (escape hatch).
|
|
213
|
+
#[cfg(not(target_arch = "wasm32"))]
|
|
214
|
+
pub fn osr_array_enabled() -> bool {
|
|
215
|
+
static ENABLED: OnceLock<bool> = OnceLock::new();
|
|
216
|
+
*ENABLED.get_or_init(|| {
|
|
217
|
+
std::env::var("TISH_JIT_OSR_ARRAY")
|
|
218
|
+
.map(|v| v != "0")
|
|
219
|
+
.unwrap_or(true)
|
|
220
|
+
})
|
|
221
|
+
}
|
|
222
|
+
|
|
183
223
|
/// Boolean scalar local slots in the numeric CFG JIT (#187). **Default ON**; `TISH_JIT_BOOL_SLOTS=0`
|
|
184
224
|
/// disables it. A `let flag = false` / `flag = true` / `if (flag)` local is represented as an `f64`
|
|
185
225
|
/// `0.0`/`1.0`; a syntactic pre-pass ([`classify_bool_slots`]) tags the slots, and the equality
|
|
@@ -194,6 +234,26 @@ pub fn jit_bool_slots_enabled() -> bool {
|
|
|
194
234
|
})
|
|
195
235
|
}
|
|
196
236
|
|
|
237
|
+
/// #168 int-typed slots. **Default ON**; `TISH_JIT_INT_SLOTS=0` disables (falls back to f64
|
|
238
|
+
/// slots — the pre-#168 behavior, byte-identical results).
|
|
239
|
+
///
|
|
240
|
+
/// A local whose every store is a bitwise/shift result (or an integral constant) lives in an
|
|
241
|
+
/// `i64` Variable holding the EXACT JS number ([`Repr::I64Num`]), so an integer hash/PRNG
|
|
242
|
+
/// accumulator (`h = ((h << 13) | (h >>> 19)) >>> 0; h = h ^ …`) stays in integer registers
|
|
243
|
+
/// ACROSS statements instead of paying an f64 materialize at every store plus a `ToInt32`
|
|
244
|
+
/// re-derivation at every load. A syntactic pre-pass ([`classify_int_slots`]) tags the slots;
|
|
245
|
+
/// the StoreLocal translator bails compilation on any store the pre-pass mis-tagged (a
|
|
246
|
+
/// non-integer value reaching an int slot), so a wrong tag can never miscompile — it just
|
|
247
|
+
/// runs the VM.
|
|
248
|
+
pub fn jit_int_slots_enabled() -> bool {
|
|
249
|
+
static ENABLED: OnceLock<bool> = OnceLock::new();
|
|
250
|
+
*ENABLED.get_or_init(|| {
|
|
251
|
+
std::env::var("TISH_JIT_INT_SLOTS")
|
|
252
|
+
.map(|v| v != "0")
|
|
253
|
+
.unwrap_or(true)
|
|
254
|
+
})
|
|
255
|
+
}
|
|
256
|
+
|
|
197
257
|
/// The self-recursion stack guard (#381). **Default ON**; `TISH_JIT_RECUR_GUARD=0` disables it.
|
|
198
258
|
///
|
|
199
259
|
/// A JIT'd self-recursive numeric function recurses on the native stack (SelfCall lowers to a native
|
|
@@ -477,6 +537,8 @@ struct JitGlobal {
|
|
|
477
537
|
/// `FuncId` of the imported `tish_math_call` host fn (#186), declared once at module init and
|
|
478
538
|
/// re-imported into each compiled function via `declare_func_in_func`.
|
|
479
539
|
math_call_id: cranelift_module::FuncId,
|
|
540
|
+
/// `FuncId` of the imported `tish_math_binary_call` host fn (#203), for the `MathBinary` intrinsic.
|
|
541
|
+
math_binary_call_id: cranelift_module::FuncId,
|
|
480
542
|
/// `FuncId`s of the imported `tish_jv_*` vector runtime (#189).
|
|
481
543
|
jv_fns: JvFns,
|
|
482
544
|
/// #187: directly-callable numeric callees, keyed by the stable global name a top-level function is
|
|
@@ -525,6 +587,15 @@ thread_local! {
|
|
|
525
587
|
static JV_ARENA: std::cell::RefCell<JvArena> = const {
|
|
526
588
|
std::cell::RefCell::new(JvArena { vecs: Vec::new(), free: Vec::new(), deopt: false })
|
|
527
589
|
};
|
|
590
|
+
/// #203 — cache of `(chunk ptr, inner loop header) → (trig header, trig end, trig indexes arrays)`
|
|
591
|
+
/// for [`osr_expand_cached`]. The expansion analysis (a whole-chunk scan + per-candidate classify)
|
|
592
|
+
/// is loop-structure-only, so it is stable for a given chunk+header and computed ONCE here rather
|
|
593
|
+
/// than on every back-edge — critical for a small hot loop inside a function called millions of
|
|
594
|
+
/// times (e.g. nbody's `advance`), where recomputing per call is a real regression. A stale entry
|
|
595
|
+
/// (a freed chunk's address reused) can only mis-route a perf hint: `run_osr`/`compile_loop_region`
|
|
596
|
+
/// re-validate the region structurally + by fingerprint, so a wrong hint never miscompiles.
|
|
597
|
+
static OSR_EXPAND_CACHE: std::cell::RefCell<HashMap<(usize, usize), (usize, usize, bool, bool)>> =
|
|
598
|
+
std::cell::RefCell::new(HashMap::new());
|
|
528
599
|
}
|
|
529
600
|
/// `handle` (1-based, `0` = null) → arena index, or `None` for the null handle.
|
|
530
601
|
#[cfg(not(target_arch = "wasm32"))]
|
|
@@ -667,9 +738,23 @@ extern "C" fn tish_math_call(id: i32, x: f64) -> f64 {
|
|
|
667
738
|
}
|
|
668
739
|
}
|
|
669
740
|
|
|
741
|
+
/// #203 — host call for the `MathBinary` intrinsic (max/min/pow/atan2). Routes through
|
|
742
|
+
/// `MathBinaryFn::apply`, the single source of truth, so JIT ≡ VM ≡ interp.
|
|
743
|
+
extern "C" fn tish_math_binary_call(id: i32, a: f64, b: f64) -> f64 {
|
|
744
|
+
match tishlang_bytecode::MathBinaryFn::from_u16(id as u16) {
|
|
745
|
+
Some(m) => m.apply(a, b),
|
|
746
|
+
None => f64::NAN,
|
|
747
|
+
}
|
|
748
|
+
}
|
|
749
|
+
|
|
670
750
|
/// Build the JIT module and declare the imported host functions (`tish_math_call` #186, the
|
|
671
751
|
/// `tish_jv_*` vector runtime #189), returning the module + their `FuncId`s.
|
|
672
|
-
fn new_module() -> Option<(
|
|
752
|
+
fn new_module() -> Option<(
|
|
753
|
+
JITModule,
|
|
754
|
+
cranelift_module::FuncId,
|
|
755
|
+
cranelift_module::FuncId,
|
|
756
|
+
JvFns,
|
|
757
|
+
)> {
|
|
673
758
|
let mut flag_builder = settings::builder();
|
|
674
759
|
// JIT code is loaded at a fixed address; no PIC / colocated libcalls needed.
|
|
675
760
|
flag_builder.set("use_colocated_libcalls", "false").ok()?;
|
|
@@ -681,6 +766,7 @@ fn new_module() -> Option<(JITModule, cranelift_module::FuncId, JvFns)> {
|
|
|
681
766
|
.ok()?;
|
|
682
767
|
let mut builder = JITBuilder::with_isa(isa, cranelift_module::default_libcall_names());
|
|
683
768
|
builder.symbol("tish_math_call", tish_math_call as *const u8);
|
|
769
|
+
builder.symbol("tish_math_binary_call", tish_math_binary_call as *const u8);
|
|
684
770
|
builder.symbol("tish_jv_new", tish_jv_new as *const u8);
|
|
685
771
|
builder.symbol("tish_jv_push", tish_jv_push as *const u8);
|
|
686
772
|
builder.symbol("tish_jv_get", tish_jv_get as *const u8);
|
|
@@ -699,6 +785,16 @@ fn new_module() -> Option<(JITModule, cranelift_module::FuncId, JvFns)> {
|
|
|
699
785
|
.declare_function("tish_math_call", Linkage::Import, &msig)
|
|
700
786
|
.ok()?;
|
|
701
787
|
|
|
788
|
+
// `tish_math_binary_call(i32 fn-id, f64 a, f64 b) -> f64` (#203).
|
|
789
|
+
let mut mbsig = module.make_signature();
|
|
790
|
+
mbsig.params.push(AbiParam::new(types::I32));
|
|
791
|
+
mbsig.params.push(AbiParam::new(types::F64));
|
|
792
|
+
mbsig.params.push(AbiParam::new(types::F64));
|
|
793
|
+
mbsig.returns.push(AbiParam::new(types::F64));
|
|
794
|
+
let math_binary_id = module
|
|
795
|
+
.declare_function("tish_math_binary_call", Linkage::Import, &mbsig)
|
|
796
|
+
.ok()?;
|
|
797
|
+
|
|
702
798
|
// Helper to declare a `tish_jv_*` import from param/return abi lists.
|
|
703
799
|
let mut declare = |name: &str, params: &[AbiParam], rets: &[AbiParam]| {
|
|
704
800
|
let mut s = module.make_signature();
|
|
@@ -741,18 +837,19 @@ fn new_module() -> Option<(JITModule, cranelift_module::FuncId, JvFns)> {
|
|
|
741
837
|
free: declare("tish_jv_free", &[AbiParam::new(types::I64)], &[])?,
|
|
742
838
|
deopt: declare("tish_jv_deopt", &[], &[])?,
|
|
743
839
|
};
|
|
744
|
-
Some((module, math_id, jv))
|
|
840
|
+
Some((module, math_id, math_binary_id, jv))
|
|
745
841
|
}
|
|
746
842
|
|
|
747
843
|
fn jit() -> Option<&'static Mutex<JitGlobal>> {
|
|
748
844
|
JIT.get_or_init(|| {
|
|
749
|
-
new_module().map(|(module, math_call_id, jv_fns)| {
|
|
845
|
+
new_module().map(|(module, math_call_id, math_binary_call_id, jv_fns)| {
|
|
750
846
|
Mutex::new(JitGlobal {
|
|
751
847
|
module,
|
|
752
848
|
cache: HashMap::new(),
|
|
753
849
|
osr_cache: HashMap::new(),
|
|
754
850
|
counter: 0,
|
|
755
851
|
math_call_id,
|
|
852
|
+
math_binary_call_id,
|
|
756
853
|
jv_fns,
|
|
757
854
|
callees: HashMap::new(),
|
|
758
855
|
})
|
|
@@ -873,6 +970,121 @@ pub fn try_compile_numeric(chunk: &Chunk) -> Option<NumericFn> {
|
|
|
873
970
|
result
|
|
874
971
|
}
|
|
875
972
|
|
|
973
|
+
/// #203 — expand a hot inner-loop region to the OUTERMOST enclosing loop that is itself array-OSR-
|
|
974
|
+
/// compilable, so an array-indexing loop nest (matmul) marshals its arrays ONCE and runs the whole
|
|
975
|
+
/// nest natively, instead of re-marshalling O(array) per inner-loop entry (a net wash). Returns the
|
|
976
|
+
/// widest enclosing loop `(header, end)` whose region `classify_osr_arrays` accepts, or the input
|
|
977
|
+
/// region if none encloses it. Only meaningful under `TISH_JIT_OSR_ARRAY`; the caller gates on the
|
|
978
|
+
/// flag. SOUND to trigger at the returned loop's own back-edge: a `for` increment sits inside the
|
|
979
|
+
/// region before the back-edge (so the live-in captures the *next* index) and inner loop vars are
|
|
980
|
+
/// re-initialized by the body, so a native re-entry continues, never re-runs, completed iterations.
|
|
981
|
+
#[cfg(not(target_arch = "wasm32"))]
|
|
982
|
+
pub fn osr_expand_region(chunk: &Chunk, header_ip: usize, region_end: usize) -> (usize, usize) {
|
|
983
|
+
let code = &chunk.code;
|
|
984
|
+
let mut best = (header_ip, region_end);
|
|
985
|
+
let mut ip = 0usize;
|
|
986
|
+
while ip < code.len() {
|
|
987
|
+
let op = match Opcode::from_u8(code[ip]) {
|
|
988
|
+
Some(o) => o,
|
|
989
|
+
None => break,
|
|
990
|
+
};
|
|
991
|
+
let size = match op.instruction_size(code, ip) {
|
|
992
|
+
Some(s) => s,
|
|
993
|
+
None => break,
|
|
994
|
+
};
|
|
995
|
+
if op == Opcode::JumpBack {
|
|
996
|
+
if let Some(dist) = peek_u16(code, ip + 1) {
|
|
997
|
+
let pos = ip + 3; // region_end convention: byte AFTER the JumpBack instruction
|
|
998
|
+
let tgt = pos.saturating_sub(dist as usize); // that loop's header
|
|
999
|
+
// A loop that STRICTLY encloses the input region (`tgt < header_ip` — starts before it —
|
|
1000
|
+
// and `pos >= region_end`), and that ITSELF INDEXES ARRAYS, is worth expanding to: the
|
|
1001
|
+
// whole point is to marshal its arrays once. Pick the WIDEST (smallest header). Keeping
|
|
1002
|
+
// the header strictly smaller guarantees `best.0 == header_ip ⇒ best.1 == region_end`
|
|
1003
|
+
// (no expansion), so the trigger's "sound entry" test stays exact.
|
|
1004
|
+
let encloses = tgt < header_ip && pos >= region_end && tgt < best.0;
|
|
1005
|
+
if encloses
|
|
1006
|
+
&& classify_osr_arrays(chunk, tgt, pos).is_some_and(|a| !a.slots.is_empty())
|
|
1007
|
+
{
|
|
1008
|
+
best = (tgt, pos);
|
|
1009
|
+
}
|
|
1010
|
+
}
|
|
1011
|
+
}
|
|
1012
|
+
ip += size;
|
|
1013
|
+
}
|
|
1014
|
+
best
|
|
1015
|
+
}
|
|
1016
|
+
|
|
1017
|
+
/// #203 — does the OSR region `[header_ip, region_end)` index at least one array? Drives the OSR
|
|
1018
|
+
/// trigger cadence: an array-indexing (expandable) loop fires on the first sound entry past the
|
|
1019
|
+
/// threshold; a pure-numeric loop keeps the original threshold+retry-modulo sampling.
|
|
1020
|
+
#[cfg(not(target_arch = "wasm32"))]
|
|
1021
|
+
pub fn osr_region_has_arrays(chunk: &Chunk, header_ip: usize, region_end: usize) -> bool {
|
|
1022
|
+
classify_osr_arrays(chunk, header_ip, region_end).is_some_and(|a| !a.slots.is_empty())
|
|
1023
|
+
}
|
|
1024
|
+
|
|
1025
|
+
/// #203 — is the region `[header_ip, region_end)` strictly enclosed by ANOTHER loop? An array region
|
|
1026
|
+
/// nested inside a non-array outer loop (e.g. k_nucleotide's `seq[i+j]` inner loop inside the Map-op
|
|
1027
|
+
/// outer loop) is re-entered per outer iteration, so array-OSR'ing it would re-marshal the whole array
|
|
1028
|
+
/// each entry — a net wash. Such regions are left to the interpreter (matching the flag-off path, where
|
|
1029
|
+
/// their `GetIndex` bails the numeric OSR anyway). Only an OUTERMOST array loop (matmul's `i` loop)
|
|
1030
|
+
/// marshals once and wins.
|
|
1031
|
+
#[cfg(not(target_arch = "wasm32"))]
|
|
1032
|
+
fn osr_region_enclosed(chunk: &Chunk, header_ip: usize, region_end: usize) -> bool {
|
|
1033
|
+
let code = &chunk.code;
|
|
1034
|
+
let mut ip = 0usize;
|
|
1035
|
+
while ip < code.len() {
|
|
1036
|
+
let op = match Opcode::from_u8(code[ip]) {
|
|
1037
|
+
Some(o) => o,
|
|
1038
|
+
None => break,
|
|
1039
|
+
};
|
|
1040
|
+
let size = match op.instruction_size(code, ip) {
|
|
1041
|
+
Some(s) => s,
|
|
1042
|
+
None => break,
|
|
1043
|
+
};
|
|
1044
|
+
if op == Opcode::JumpBack {
|
|
1045
|
+
if let Some(dist) = peek_u16(code, ip + 1) {
|
|
1046
|
+
let pos = ip + 3;
|
|
1047
|
+
let tgt = pos.saturating_sub(dist as usize);
|
|
1048
|
+
// strictly encloses on at least one side, contains on the other
|
|
1049
|
+
if tgt <= header_ip && pos >= region_end && (tgt < header_ip || pos > region_end) {
|
|
1050
|
+
return true;
|
|
1051
|
+
}
|
|
1052
|
+
}
|
|
1053
|
+
}
|
|
1054
|
+
ip += size;
|
|
1055
|
+
}
|
|
1056
|
+
false
|
|
1057
|
+
}
|
|
1058
|
+
|
|
1059
|
+
/// #203 — the OSR trigger decision for the loop at `header_ip`, memoized per `(chunk, header)` in a
|
|
1060
|
+
/// thread-local so the whole-chunk scans run at most once per loop header for the life of the thread
|
|
1061
|
+
/// (not once per enclosing-function call). Returns `(trig header, trig end, has_arrays, array_worthy)`:
|
|
1062
|
+
/// * `trig` = the outermost enclosing array loop to run instead (`== header_ip` if none);
|
|
1063
|
+
/// * `has_arrays` = the trig region indexes an array;
|
|
1064
|
+
/// * `array_worthy` = `has_arrays` AND the trig region is NOT itself nested in another loop — i.e.
|
|
1065
|
+
/// array-OSR'ing it marshals once and wins, rather than re-marshalling per outer iteration (a wash).
|
|
1066
|
+
/// The VM's `JumpBack` handler runs the array path only when `array_worthy`, gives up on a `has_arrays
|
|
1067
|
+
/// && !array_worthy` nested wash (interpret, matching flag-off), and takes the numeric path otherwise.
|
|
1068
|
+
#[cfg(not(target_arch = "wasm32"))]
|
|
1069
|
+
pub fn osr_expand_cached(
|
|
1070
|
+
chunk: &Chunk,
|
|
1071
|
+
header_ip: usize,
|
|
1072
|
+
region_end: usize,
|
|
1073
|
+
) -> (usize, usize, bool, bool) {
|
|
1074
|
+
let key = (chunk as *const Chunk as usize, header_ip);
|
|
1075
|
+
OSR_EXPAND_CACHE.with(|c| {
|
|
1076
|
+
if let Some(&v) = c.borrow().get(&key) {
|
|
1077
|
+
return v;
|
|
1078
|
+
}
|
|
1079
|
+
let (th, te) = osr_expand_region(chunk, header_ip, region_end);
|
|
1080
|
+
let has_arrays = osr_region_has_arrays(chunk, th, te);
|
|
1081
|
+
let array_worthy = has_arrays && !osr_region_enclosed(chunk, th, te);
|
|
1082
|
+
let v = (th, te, has_arrays, array_worthy);
|
|
1083
|
+
c.borrow_mut().insert(key, v);
|
|
1084
|
+
v
|
|
1085
|
+
})
|
|
1086
|
+
}
|
|
1087
|
+
|
|
876
1088
|
/// Compile the hot loop region `[header_ip, region_end)` of `chunk` to native code (#190 OSR), or
|
|
877
1089
|
/// `None` if it is not a pure-numeric slot loop. Cached per `(chunk, header_ip)` with a fingerprint
|
|
878
1090
|
/// guard (negative results included, so a non-compilable loop is scanned once). Called from the frame
|
|
@@ -909,6 +1121,20 @@ fn compile_loop_region(
|
|
|
909
1121
|
return None;
|
|
910
1122
|
}
|
|
911
1123
|
|
|
1124
|
+
// #203 array-index (the matmul lever): behind `TISH_JIT_OSR_ARRAY`, classify the array live-in
|
|
1125
|
+
// slots (`arr[i]` read / `arr[i]=v` write over a computed index). `None` ⇒ the region has array
|
|
1126
|
+
// ops this pass can't lower ⇒ not compilable (its index ops keep bailing). `Some(empty)` ⇒ a
|
|
1127
|
+
// pure-numeric region on the unchanged 2-pointer ABI. `Some(non-empty)` ⇒ the 3-pointer array ABI.
|
|
1128
|
+
let arrays = if osr_array_enabled() {
|
|
1129
|
+
classify_osr_arrays(chunk, header_ip, region_end)?
|
|
1130
|
+
} else {
|
|
1131
|
+
OsrArrays {
|
|
1132
|
+
slots: Vec::new(),
|
|
1133
|
+
writable: std::collections::BTreeSet::new(),
|
|
1134
|
+
}
|
|
1135
|
+
};
|
|
1136
|
+
let array_set: std::collections::BTreeSet<u16> = arrays.slots.iter().copied().collect();
|
|
1137
|
+
|
|
912
1138
|
// 1. Scan the region: validate the whitelist (op_size = None ⇒ bail), collect in-region block
|
|
913
1139
|
// leaders, the live slot set, and the EXIT targets (jump targets outside the region). A
|
|
914
1140
|
// JumpBack must stay inside the region (its own loop, possibly nested); one leaving the region
|
|
@@ -921,6 +1147,10 @@ fn compile_loop_region(
|
|
|
921
1147
|
while ip < region_end {
|
|
922
1148
|
let op = Opcode::from_u8(*code.get(ip)?)?;
|
|
923
1149
|
match op {
|
|
1150
|
+
// #203: an `arr[i]` read / `arr[i]=v` write over a classified array slot — validated by
|
|
1151
|
+
// `classify_osr_arrays`, lowered by the region body. `array_set` empty ⇒ these never appear
|
|
1152
|
+
// (their base slot would have been classified an array), so the arm below can't fire.
|
|
1153
|
+
Opcode::GetIndex | Opcode::SetIndex if !array_set.is_empty() => {}
|
|
924
1154
|
// Pure slot / stack / arithmetic / structured control flow — the region vocabulary.
|
|
925
1155
|
Opcode::Nop
|
|
926
1156
|
| Opcode::Pop
|
|
@@ -930,9 +1160,15 @@ fn compile_loop_region(
|
|
|
930
1160
|
| Opcode::LoopVarsEnd
|
|
931
1161
|
| Opcode::LoopVarsBegin
|
|
932
1162
|
| Opcode::UnaryOp
|
|
933
|
-
| Opcode::MathUnary
|
|
1163
|
+
| Opcode::MathUnary // #186 — `Math.<fn>(x)`: 1 f64 in, 1 f64 out (like UnaryOp)
|
|
1164
|
+
| Opcode::MathBinary => {} // #203 — `Math.<fn>(a,b)`: 2 f64 in, 1 f64 out
|
|
934
1165
|
Opcode::LoadLocal | Opcode::StoreLocal => {
|
|
935
|
-
|
|
1166
|
+
let s = peek_u16(code, ip + 1)?;
|
|
1167
|
+
// #203: an array slot's live-in is a `Value::Array` marshalled through the handles
|
|
1168
|
+
// buffer, NOT an f64 in the slots buffer — keep it out of the numeric live set.
|
|
1169
|
+
if !array_set.contains(&s) {
|
|
1170
|
+
used.insert(s);
|
|
1171
|
+
}
|
|
936
1172
|
}
|
|
937
1173
|
Opcode::LoadConst => match chunk.constants.get(peek_u16(code, ip + 1)? as usize) {
|
|
938
1174
|
Some(Constant::Number(_)) | Some(Constant::Bool(_)) => {}
|
|
@@ -944,7 +1180,9 @@ fn compile_loop_region(
|
|
|
944
1180
|
.map(|r| r as u8)
|
|
945
1181
|
.and_then(u8_to_binop)?
|
|
946
1182
|
{
|
|
947
|
-
|
|
1183
|
+
// #203: `**` (BinOp::Pow) IS lowerable now (a host call to `tish_math_binary_call`
|
|
1184
|
+
// in the region body), so it stays in the OSR whitelist. Logical/`in` still bail.
|
|
1185
|
+
BinOp::And | BinOp::Or | BinOp::In => return None,
|
|
948
1186
|
_ => {}
|
|
949
1187
|
}
|
|
950
1188
|
}
|
|
@@ -996,11 +1234,26 @@ fn compile_loop_region(
|
|
|
996
1234
|
let exits: Vec<usize> = exit_targets.iter().copied().collect();
|
|
997
1235
|
let exit_id: HashMap<usize, usize> = exits.iter().enumerate().map(|(i, &t)| (t, i)).collect();
|
|
998
1236
|
|
|
999
|
-
//
|
|
1237
|
+
// #203: array slot → its index in the handles buffer (marshalling order = ascending slot). Empty
|
|
1238
|
+
// ⇒ the pure-numeric 2-pointer ABI (unchanged). The region body loads `(ptr,len)` from the handles
|
|
1239
|
+
// buffer at `16 * array_pos[slot]` for each `arr[i]` access.
|
|
1240
|
+
let has_arrays = !arrays.slots.is_empty();
|
|
1241
|
+
let array_pos: HashMap<u16, usize> = arrays
|
|
1242
|
+
.slots
|
|
1243
|
+
.iter()
|
|
1244
|
+
.enumerate()
|
|
1245
|
+
.map(|(p, &s)| (s, p))
|
|
1246
|
+
.collect();
|
|
1247
|
+
|
|
1248
|
+
// 2. Build the region function. Signature: `(slots, deopt) -> i32` (pure numeric) or
|
|
1249
|
+
// `(slots, handles, deopt) -> i32` (#203 array mode) — all pointers; returns the exit id.
|
|
1000
1250
|
let ptr_ty = g.module.target_config().pointer_type();
|
|
1001
1251
|
let mut sig = g.module.make_signature();
|
|
1002
1252
|
sig.params.push(AbiParam::new(ptr_ty)); // slots buffer
|
|
1003
|
-
|
|
1253
|
+
if has_arrays {
|
|
1254
|
+
sig.params.push(AbiParam::new(ptr_ty)); // #203 handles buffer (ArrayHandle[])
|
|
1255
|
+
}
|
|
1256
|
+
sig.params.push(AbiParam::new(ptr_ty)); // deopt flag (int-slot miss #514 / OOB #203)
|
|
1004
1257
|
sig.returns.push(AbiParam::new(types::I32));
|
|
1005
1258
|
|
|
1006
1259
|
let name = format!("tish_osr_{}", g.counter);
|
|
@@ -1013,6 +1266,7 @@ fn compile_loop_region(
|
|
|
1013
1266
|
let mut ctx = g.module.make_context();
|
|
1014
1267
|
ctx.func.signature = sig.clone();
|
|
1015
1268
|
let math_fref = g.module.declare_func_in_func(g.math_call_id, &mut ctx.func);
|
|
1269
|
+
let math_binary_fref = g.module.declare_func_in_func(g.math_binary_call_id, &mut ctx.func);
|
|
1016
1270
|
let mut fbctx = FunctionBuilderContext::new();
|
|
1017
1271
|
let built = build_loop_region_body(
|
|
1018
1272
|
&mut ctx.func,
|
|
@@ -1025,6 +1279,8 @@ fn compile_loop_region(
|
|
|
1025
1279
|
&buf_pos,
|
|
1026
1280
|
&exit_id,
|
|
1027
1281
|
math_fref,
|
|
1282
|
+
math_binary_fref,
|
|
1283
|
+
&array_pos,
|
|
1028
1284
|
);
|
|
1029
1285
|
if !built {
|
|
1030
1286
|
g.module.clear_context(&mut ctx);
|
|
@@ -1039,10 +1295,16 @@ fn compile_loop_region(
|
|
|
1039
1295
|
return None;
|
|
1040
1296
|
}
|
|
1041
1297
|
let fptr = g.module.get_finalized_function(id);
|
|
1298
|
+
let array_slots: Vec<(u16, bool)> = arrays
|
|
1299
|
+
.slots
|
|
1300
|
+
.iter()
|
|
1301
|
+
.map(|&s| (s, arrays.writable.contains(&s)))
|
|
1302
|
+
.collect();
|
|
1042
1303
|
Some(LoopFn {
|
|
1043
1304
|
ptr: fptr as usize,
|
|
1044
1305
|
used_slots,
|
|
1045
1306
|
exits,
|
|
1307
|
+
array_slots,
|
|
1046
1308
|
})
|
|
1047
1309
|
}
|
|
1048
1310
|
|
|
@@ -1063,6 +1325,10 @@ fn build_loop_region_body(
|
|
|
1063
1325
|
buf_pos: &HashMap<u16, usize>,
|
|
1064
1326
|
exit_id: &HashMap<usize, usize>,
|
|
1065
1327
|
math_fref: cranelift::codegen::ir::FuncRef,
|
|
1328
|
+
math_binary_fref: cranelift::codegen::ir::FuncRef,
|
|
1329
|
+
// #203: array slot → its index in the handles buffer (marshalling order). Empty ⇒ the pure-numeric
|
|
1330
|
+
// 2-pointer ABI; non-empty ⇒ the 3-pointer array ABI `(slots, handles, deopt)`.
|
|
1331
|
+
array_pos: &HashMap<u16, usize>,
|
|
1066
1332
|
) -> bool {
|
|
1067
1333
|
let code = &chunk.code;
|
|
1068
1334
|
let num_slots = chunk.num_slots as usize;
|
|
@@ -1081,28 +1347,137 @@ fn build_loop_region_body(
|
|
|
1081
1347
|
bcx.append_block_params_for_function_params(entry);
|
|
1082
1348
|
bcx.switch_to_block(entry);
|
|
1083
1349
|
let params: Vec<ClifValue> = bcx.block_params(entry).to_vec();
|
|
1084
|
-
let slots_ptr = params[0];
|
|
1350
|
+
let slots_ptr = params[0];
|
|
1351
|
+
// #203: the array ABI inserts a `handles` pointer between `slots` and `deopt`. `deopt` is set on an
|
|
1352
|
+
// int-slot live-in guard fail (#514) or an out-of-bounds array index (#203) → the VM re-interprets.
|
|
1353
|
+
let (handles_ptr, deopt_ptr) = if array_pos.is_empty() {
|
|
1354
|
+
(None, params[1])
|
|
1355
|
+
} else {
|
|
1356
|
+
(Some(params[1]), params[2])
|
|
1357
|
+
};
|
|
1358
|
+
// #203: load each array live-in's `(ptr, len)` ONCE at entry (loop-invariant; the entry block
|
|
1359
|
+
// dominates the whole loop). `ArrayHandle` is `#[repr(C)] { ptr: *mut f64 @0, len: usize @8 }`,
|
|
1360
|
+
// 16 bytes. `array_handles[slot] = (data_ptr, len_i64)`.
|
|
1361
|
+
let ptr_ty = bcx.func.dfg.value_type(slots_ptr);
|
|
1362
|
+
let mut array_handles: HashMap<u16, (ClifValue, ClifValue)> = HashMap::new();
|
|
1363
|
+
if let Some(hptr) = handles_ptr {
|
|
1364
|
+
for (&slot, &pos) in array_pos.iter() {
|
|
1365
|
+
let base = (pos * 16) as i32;
|
|
1366
|
+
let data = bcx.ins().load(ptr_ty, MemFlags::new(), hptr, base);
|
|
1367
|
+
let len = bcx.ins().load(types::I64, MemFlags::new(), hptr, base + 8);
|
|
1368
|
+
array_handles.insert(slot, (data, len));
|
|
1369
|
+
}
|
|
1370
|
+
}
|
|
1371
|
+
|
|
1372
|
+
// #514: port the fn-body int-typed slots (#511) into the OSR loop region. A slot whose every store
|
|
1373
|
+
// is integer-typed lives in an `i64` Variable (`Repr::I64Num`), so an integer hash/PRNG accumulator
|
|
1374
|
+
// stays in integer registers across the loop instead of an f64 materialize per store + ToInt32 per
|
|
1375
|
+
// load. `arity` is 0 at top level (no params). `classify_int_slots` is chunk-level, so a slot that
|
|
1376
|
+
// is int-stored in the region but f64-stored elsewhere in the chunk is (safely) NOT tagged.
|
|
1377
|
+
let int_slots = classify_int_slots(chunk, 0);
|
|
1085
1378
|
let vars: Vec<Variable> = (0..num_slots)
|
|
1086
|
-
.map(|
|
|
1379
|
+
.map(|slot| {
|
|
1380
|
+
let ty = if int_slots.contains(&slot) {
|
|
1381
|
+
types::I64
|
|
1382
|
+
} else {
|
|
1383
|
+
types::F64
|
|
1384
|
+
};
|
|
1385
|
+
bcx.declare_var(ty)
|
|
1386
|
+
})
|
|
1087
1387
|
.collect();
|
|
1388
|
+
|
|
1389
|
+
// Entry marshalling. A non-int slot loads its f64 live-in straight. An int slot's live-in arrives
|
|
1390
|
+
// as f64 from the VM frame, so it must be an EXACT integer that round-trips through i64 — this
|
|
1391
|
+
// rejects non-integers, NaN / ±Inf, out-of-i64-range, and -0 — or the `I64Num` invariant breaks.
|
|
1392
|
+
// On any failure we set the deopt flag and return; the VM discards the (still-pristine) slots and
|
|
1393
|
+
// re-interprets the loop. The guard runs once at region entry, off the per-iteration path.
|
|
1394
|
+
let mut all_ok: Option<ClifValue> = None;
|
|
1088
1395
|
for (slot, &var) in vars.iter().enumerate() {
|
|
1089
|
-
|
|
1090
|
-
|
|
1091
|
-
|
|
1092
|
-
|
|
1396
|
+
let live: Option<ClifValue> = if let Some(&p) = buf_pos.get(&(slot as u16)) {
|
|
1397
|
+
Some(
|
|
1398
|
+
bcx.ins()
|
|
1399
|
+
.load(types::F64, MemFlags::new(), slots_ptr, (p * 8) as i32),
|
|
1400
|
+
)
|
|
1093
1401
|
} else {
|
|
1094
|
-
|
|
1402
|
+
None
|
|
1095
1403
|
};
|
|
1096
|
-
|
|
1404
|
+
if int_slots.contains(&slot) {
|
|
1405
|
+
match live {
|
|
1406
|
+
Some(x) => {
|
|
1407
|
+
let i64c = bcx.ins().fcvt_to_sint_sat(types::I64, x);
|
|
1408
|
+
let back = bcx.ins().fcvt_from_sint(types::F64, i64c);
|
|
1409
|
+
let exact = bcx.ins().fcmp(FloatCC::Equal, back, x);
|
|
1410
|
+
// -0 round-trips to +0 (IEEE `==` is true) but differs as an f64 read, so reject
|
|
1411
|
+
// it explicitly: 1/(-0) is -∞, 1/(+0) is +∞.
|
|
1412
|
+
let zero = bcx.ins().f64const(0.0);
|
|
1413
|
+
let is_zero = bcx.ins().fcmp(FloatCC::Equal, x, zero);
|
|
1414
|
+
let one = bcx.ins().f64const(1.0);
|
|
1415
|
+
let recip = bcx.ins().fdiv(one, x);
|
|
1416
|
+
let recip_neg = bcx.ins().fcmp(FloatCC::LessThan, recip, zero);
|
|
1417
|
+
let is_neg0 = bcx.ins().band(is_zero, recip_neg);
|
|
1418
|
+
let not_neg0 = bcx.ins().bxor_imm(is_neg0, 1);
|
|
1419
|
+
let ok = bcx.ins().band(exact, not_neg0);
|
|
1420
|
+
all_ok = Some(match all_ok {
|
|
1421
|
+
Some(a) => bcx.ins().band(a, ok),
|
|
1422
|
+
None => ok,
|
|
1423
|
+
});
|
|
1424
|
+
bcx.def_var(var, i64c);
|
|
1425
|
+
}
|
|
1426
|
+
None => {
|
|
1427
|
+
let z = bcx.ins().iconst(types::I64, 0);
|
|
1428
|
+
bcx.def_var(var, z);
|
|
1429
|
+
}
|
|
1430
|
+
}
|
|
1431
|
+
} else {
|
|
1432
|
+
let init = match live {
|
|
1433
|
+
Some(x) => x,
|
|
1434
|
+
None => bcx.ins().f64const(0.0),
|
|
1435
|
+
};
|
|
1436
|
+
bcx.def_var(var, init);
|
|
1437
|
+
}
|
|
1097
1438
|
}
|
|
1098
1439
|
let header_block = match blocks.get(&header_ip) {
|
|
1099
1440
|
Some(&b) => b,
|
|
1100
1441
|
None => return false,
|
|
1101
1442
|
};
|
|
1102
|
-
|
|
1443
|
+
match all_ok {
|
|
1444
|
+
// At least one int slot has a live-in: enter the loop only if every guard passed, else deopt.
|
|
1445
|
+
Some(ok) => {
|
|
1446
|
+
let deopt_block = bcx.create_block();
|
|
1447
|
+
bcx.ins().brif(ok, header_block, &[], deopt_block, &[]);
|
|
1448
|
+
bcx.switch_to_block(deopt_block);
|
|
1449
|
+
let flag = bcx.ins().iconst(types::I8, 1);
|
|
1450
|
+
bcx.ins().store(MemFlags::new(), flag, deopt_ptr, 0);
|
|
1451
|
+
let zero_id = bcx.ins().iconst(types::I32, 0);
|
|
1452
|
+
bcx.ins().return_(&[zero_id]);
|
|
1453
|
+
}
|
|
1454
|
+
None => {
|
|
1455
|
+
bcx.ins().jump(header_block, &[]);
|
|
1456
|
+
}
|
|
1457
|
+
}
|
|
1458
|
+
|
|
1459
|
+
// #203: shared out-of-bounds pad — every array bounds-check that fails branches here, sets the
|
|
1460
|
+
// deopt flag, and returns (exit id 0). The VM sees `deopt != 0` and re-interprets from the pristine
|
|
1461
|
+
// pre-region state (the numeric buffer AND the array scratch are discarded, the real arrays are
|
|
1462
|
+
// untouched), so a mid-region OOB never commits a partial result. Filled now; sealed at the end.
|
|
1463
|
+
let array_deopt_block = if array_pos.is_empty() {
|
|
1464
|
+
None
|
|
1465
|
+
} else {
|
|
1466
|
+
let b = bcx.create_block();
|
|
1467
|
+
bcx.switch_to_block(b);
|
|
1468
|
+
let flag = bcx.ins().iconst(types::I8, 1);
|
|
1469
|
+
bcx.ins().store(MemFlags::new(), flag, deopt_ptr, 0);
|
|
1470
|
+
let zero = bcx.ins().iconst(types::I32, 0);
|
|
1471
|
+
bcx.ins().return_(&[zero]);
|
|
1472
|
+
Some(b)
|
|
1473
|
+
};
|
|
1103
1474
|
|
|
1104
1475
|
// Translate the region. Operand stack is empty at every block boundary (statement-level flow).
|
|
1105
|
-
let mut stack: Vec<
|
|
1476
|
+
let mut stack: Vec<JV> = Vec::new();
|
|
1477
|
+
// #203: array handles "in flight" — pushed by `LoadLocal(array slot)` as `(data_ptr, len)`, consumed
|
|
1478
|
+
// by the very next `GetIndex`/`SetIndex`. Like `stack`, must be empty at every block boundary (an
|
|
1479
|
+
// array access never spans a branch).
|
|
1480
|
+
let mut arr_pending: Vec<(ClifValue, ClifValue)> = Vec::new();
|
|
1106
1481
|
bcx.switch_to_block(header_block);
|
|
1107
1482
|
let mut cur = header_block;
|
|
1108
1483
|
let mut terminated = false;
|
|
@@ -1111,7 +1486,9 @@ fn build_loop_region_body(
|
|
|
1111
1486
|
if let Some(&blk) = blocks.get(&ip) {
|
|
1112
1487
|
if blk != cur {
|
|
1113
1488
|
if !terminated {
|
|
1114
|
-
|
|
1489
|
+
// #203: an operand OR a pending array handle spanning a block boundary is a shape
|
|
1490
|
+
// this straight-line-per-block emitter can't carry → bail (region interpreted).
|
|
1491
|
+
if !stack.is_empty() || !arr_pending.is_empty() {
|
|
1115
1492
|
return false;
|
|
1116
1493
|
}
|
|
1117
1494
|
bcx.ins().jump(blk, &[]);
|
|
@@ -1120,6 +1497,7 @@ fn build_loop_region_body(
|
|
|
1120
1497
|
cur = blk;
|
|
1121
1498
|
terminated = false;
|
|
1122
1499
|
stack.clear();
|
|
1500
|
+
arr_pending.clear();
|
|
1123
1501
|
}
|
|
1124
1502
|
}
|
|
1125
1503
|
let op = match Opcode::from_u8(code[ip]).zip(op_size_at(code, ip)) {
|
|
@@ -1139,11 +1517,25 @@ fn build_loop_region_body(
|
|
|
1139
1517
|
Some(s) => s as usize,
|
|
1140
1518
|
None => return false,
|
|
1141
1519
|
};
|
|
1520
|
+
// #203: an array slot's "value" is its `(data_ptr, len)` handle — stage it OFF the
|
|
1521
|
+
// numeric stack for the next `GetIndex`/`SetIndex` (matches build_body_cfg's jv_pending).
|
|
1522
|
+
if let Some(&handle) = array_handles.get(&(slot as u16)) {
|
|
1523
|
+
arr_pending.push(handle);
|
|
1524
|
+
ip += 3;
|
|
1525
|
+
continue;
|
|
1526
|
+
}
|
|
1142
1527
|
let v = match vars.get(slot) {
|
|
1143
1528
|
Some(v) => *v,
|
|
1144
1529
|
None => return false,
|
|
1145
1530
|
};
|
|
1146
|
-
|
|
1531
|
+
// #514: an int slot's `use_var` is an i64 carrying the exact number (`I64Num`);
|
|
1532
|
+
// downstream int ops take it as identity, one exact convert at an f64 boundary.
|
|
1533
|
+
let lv = bcx.use_var(v);
|
|
1534
|
+
stack.push(if int_slots.contains(&slot) {
|
|
1535
|
+
JV::i64num(lv)
|
|
1536
|
+
} else {
|
|
1537
|
+
JV::f64(lv)
|
|
1538
|
+
});
|
|
1147
1539
|
ip += 3;
|
|
1148
1540
|
}
|
|
1149
1541
|
Opcode::StoreLocal => {
|
|
@@ -1151,18 +1543,38 @@ fn build_loop_region_body(
|
|
|
1151
1543
|
Some(s) => s as usize,
|
|
1152
1544
|
None => return false,
|
|
1153
1545
|
};
|
|
1154
|
-
let
|
|
1546
|
+
let jval = match stack.pop() {
|
|
1155
1547
|
Some(x) => x,
|
|
1156
1548
|
None => return false,
|
|
1157
1549
|
};
|
|
1158
|
-
if is_bool {
|
|
1550
|
+
if jval.is_bool() {
|
|
1159
1551
|
return false; // no boolean slots (keeps the number/bool distinction clean)
|
|
1160
1552
|
}
|
|
1161
1553
|
let v = match vars.get(slot) {
|
|
1162
1554
|
Some(v) => *v,
|
|
1163
1555
|
None => return false,
|
|
1164
1556
|
};
|
|
1165
|
-
|
|
1557
|
+
if int_slots.contains(&slot) {
|
|
1558
|
+
// #514/#168: store the EXACT number as i64 — I32 stores sign-extend, U32
|
|
1559
|
+
// zero-extend, I64Num is identity, an integral constant re-derives its exact
|
|
1560
|
+
// i64. Anything else means the optimistic pre-pass mis-tagged this slot (e.g. a
|
|
1561
|
+
// merge point whose linear predecessor differed) → bail; the VM keeps semantics.
|
|
1562
|
+
let iv = match jval.repr {
|
|
1563
|
+
Repr::I32 => bcx.ins().sextend(types::I64, jval.v),
|
|
1564
|
+
Repr::U32 => bcx.ins().uextend(types::I64, jval.v),
|
|
1565
|
+
Repr::I64Num => jval.v,
|
|
1566
|
+
Repr::F64 => match int_const_i64(&mut bcx, chunk, code, ip) {
|
|
1567
|
+
Some(iv) => iv,
|
|
1568
|
+
None => return false,
|
|
1569
|
+
},
|
|
1570
|
+
Repr::Bool => return false,
|
|
1571
|
+
};
|
|
1572
|
+
bcx.def_var(v, iv);
|
|
1573
|
+
} else {
|
|
1574
|
+
// f64 Variable — one materialize at the store boundary.
|
|
1575
|
+
let val = jv_f64(&mut bcx, jval);
|
|
1576
|
+
bcx.def_var(v, val);
|
|
1577
|
+
}
|
|
1166
1578
|
ip += 3;
|
|
1167
1579
|
}
|
|
1168
1580
|
Opcode::Pop => {
|
|
@@ -1223,13 +1635,14 @@ fn build_loop_region_body(
|
|
|
1223
1635
|
Some(o) => o as i16 as isize,
|
|
1224
1636
|
None => return false,
|
|
1225
1637
|
};
|
|
1226
|
-
let
|
|
1638
|
+
let cond = match stack.pop() {
|
|
1227
1639
|
Some(x) => x,
|
|
1228
1640
|
None => return false,
|
|
1229
1641
|
};
|
|
1230
1642
|
if !stack.is_empty() {
|
|
1231
1643
|
return false;
|
|
1232
1644
|
}
|
|
1645
|
+
let cond = jv_f64(&mut bcx, cond);
|
|
1233
1646
|
let falsy = falsy_flag(&mut bcx, cond);
|
|
1234
1647
|
let t = ((ip + 3) as isize + off).max(0) as usize;
|
|
1235
1648
|
let target = match target_block(t) {
|
|
@@ -1254,12 +1667,133 @@ fn build_loop_region_body(
|
|
|
1254
1667
|
Some(m) => m,
|
|
1255
1668
|
None => return false,
|
|
1256
1669
|
};
|
|
1257
|
-
let
|
|
1670
|
+
let x = match stack.pop() {
|
|
1258
1671
|
Some(v) => v,
|
|
1259
1672
|
None => return false,
|
|
1260
1673
|
};
|
|
1674
|
+
let x = jv_f64(&mut bcx, x);
|
|
1261
1675
|
let r = emit_math_unary(&mut bcx, math_fref, mfn, x);
|
|
1262
|
-
stack.push((r
|
|
1676
|
+
stack.push(JV::f64(r));
|
|
1677
|
+
ip += 3;
|
|
1678
|
+
}
|
|
1679
|
+
// #203 — `arr[idx]` read. `arr`'s `(data,len)` handle is in `arr_pending`; the computed
|
|
1680
|
+
// index is the f64 top of stack. `idx as usize` (saturating: NaN/neg → 0) matches the VM's
|
|
1681
|
+
// coercion; an out-of-bounds index branches to the shared deopt pad (the VM re-interprets).
|
|
1682
|
+
Opcode::GetIndex if !arr_pending.is_empty() => {
|
|
1683
|
+
let (data, len) = match arr_pending.pop() {
|
|
1684
|
+
Some(h) => h,
|
|
1685
|
+
None => return false,
|
|
1686
|
+
};
|
|
1687
|
+
let idx = match stack.pop() {
|
|
1688
|
+
Some(v) => v,
|
|
1689
|
+
None => return false,
|
|
1690
|
+
};
|
|
1691
|
+
let db = match array_deopt_block {
|
|
1692
|
+
Some(b) => b,
|
|
1693
|
+
None => return false,
|
|
1694
|
+
};
|
|
1695
|
+
let idx = jv_f64(&mut bcx, idx);
|
|
1696
|
+
let i = bcx.ins().fcvt_to_uint_sat(types::I64, idx);
|
|
1697
|
+
let inb = bcx.ins().icmp(IntCC::UnsignedLessThan, i, len);
|
|
1698
|
+
let cont = bcx.create_block();
|
|
1699
|
+
bcx.ins().brif(inb, cont, &[], db, &[]);
|
|
1700
|
+
bcx.switch_to_block(cont);
|
|
1701
|
+
cur = cont; // keep block-boundary tracking accurate after the mid-stream split
|
|
1702
|
+
let off = bcx.ins().imul_imm(i, 8);
|
|
1703
|
+
let addr = bcx.ins().iadd(data, off);
|
|
1704
|
+
let val = bcx.ins().load(types::F64, MemFlags::new(), addr, 0);
|
|
1705
|
+
stack.push(JV::f64(val));
|
|
1706
|
+
ip += 1;
|
|
1707
|
+
}
|
|
1708
|
+
// #203 — `arr[idx] = v`. Stack: `[idx, val, dup_val]` (a `Dup` of `val` for the expression
|
|
1709
|
+
// result); `arr`'s handle is in `arr_pending`. Store `dup_val` (== `val`) at the bounds-
|
|
1710
|
+
// checked address; OOB → the deopt pad (scratch discarded, real array untouched).
|
|
1711
|
+
Opcode::SetIndex if !arr_pending.is_empty() => {
|
|
1712
|
+
let (data, len) = match arr_pending.pop() {
|
|
1713
|
+
Some(h) => h,
|
|
1714
|
+
None => return false,
|
|
1715
|
+
};
|
|
1716
|
+
let dup = match stack.pop() {
|
|
1717
|
+
Some(v) => v,
|
|
1718
|
+
None => return false,
|
|
1719
|
+
};
|
|
1720
|
+
let _val = match stack.pop() {
|
|
1721
|
+
Some(v) => v,
|
|
1722
|
+
None => return false,
|
|
1723
|
+
};
|
|
1724
|
+
let idx = match stack.pop() {
|
|
1725
|
+
Some(v) => v,
|
|
1726
|
+
None => return false,
|
|
1727
|
+
};
|
|
1728
|
+
// #203 SOUNDNESS: the region flattens every element to f64 and `run_osr` re-boxes the
|
|
1729
|
+
// writeback as `Value::Number`. Storing a BOOLEAN (a comparison result `arr[i] = x>y`,
|
|
1730
|
+
// a `LoadConst(Bool)`, or `!x`) would therefore land as `Number 1`/`0` where the
|
|
1731
|
+
// interpreter stores `Bool true`/`false` — a divergence. There are no bool SLOTS in a
|
|
1732
|
+
// region (StoreLocal bails on a bool), so a bool value is always this transient Repr::Bool
|
|
1733
|
+
// — bail the whole region compile, let the interpreter handle the write.
|
|
1734
|
+
if dup.is_bool() {
|
|
1735
|
+
return false;
|
|
1736
|
+
}
|
|
1737
|
+
let db = match array_deopt_block {
|
|
1738
|
+
Some(b) => b,
|
|
1739
|
+
None => return false,
|
|
1740
|
+
};
|
|
1741
|
+
let store_val = jv_f64(&mut bcx, dup);
|
|
1742
|
+
let idx = jv_f64(&mut bcx, idx);
|
|
1743
|
+
let i = bcx.ins().fcvt_to_uint_sat(types::I64, idx);
|
|
1744
|
+
let inb = bcx.ins().icmp(IntCC::UnsignedLessThan, i, len);
|
|
1745
|
+
let cont = bcx.create_block();
|
|
1746
|
+
bcx.ins().brif(inb, cont, &[], db, &[]);
|
|
1747
|
+
bcx.switch_to_block(cont);
|
|
1748
|
+
cur = cont;
|
|
1749
|
+
let off = bcx.ins().imul_imm(i, 8);
|
|
1750
|
+
let addr = bcx.ins().iadd(data, off);
|
|
1751
|
+
bcx.ins().store(MemFlags::new(), store_val, addr, 0);
|
|
1752
|
+
stack.push(JV::f64(store_val)); // assignment yields the value
|
|
1753
|
+
ip += 1;
|
|
1754
|
+
}
|
|
1755
|
+
Opcode::MathBinary => {
|
|
1756
|
+
// #203 — `Math.<fn>(a, b)`: pop b, pop a, host-call, push. Every 2-arg Math fn routes
|
|
1757
|
+
// through the host call (identical to the VM), so there are no native-op semantics to
|
|
1758
|
+
// match — the win is skipping the boxed GetMember+value_call the generic path pays.
|
|
1759
|
+
let id = match peek_u16(code, ip + 1) {
|
|
1760
|
+
Some(v) => v,
|
|
1761
|
+
None => return false,
|
|
1762
|
+
};
|
|
1763
|
+
let b = match stack.pop() {
|
|
1764
|
+
Some(v) => v,
|
|
1765
|
+
None => return false,
|
|
1766
|
+
};
|
|
1767
|
+
let a = match stack.pop() {
|
|
1768
|
+
Some(v) => v,
|
|
1769
|
+
None => return false,
|
|
1770
|
+
};
|
|
1771
|
+
let b = jv_f64(&mut bcx, b);
|
|
1772
|
+
let a = jv_f64(&mut bcx, a);
|
|
1773
|
+
let idc = bcx.ins().iconst(types::I32, id as i64);
|
|
1774
|
+
let call = bcx.ins().call(math_binary_fref, &[idc, a, b]);
|
|
1775
|
+
let r = bcx.inst_results(call)[0];
|
|
1776
|
+
stack.push(JV::f64(r));
|
|
1777
|
+
ip += 3;
|
|
1778
|
+
}
|
|
1779
|
+
_ if is_binop_pow(op, code, ip) => {
|
|
1780
|
+
// #203: `a ** b` → host call to `tish_math_binary_call(Pow, a, b)` (== VM's `powf`).
|
|
1781
|
+
let r = match stack.pop() {
|
|
1782
|
+
Some(v) => v,
|
|
1783
|
+
None => return false,
|
|
1784
|
+
};
|
|
1785
|
+
let l = match stack.pop() {
|
|
1786
|
+
Some(v) => v,
|
|
1787
|
+
None => return false,
|
|
1788
|
+
};
|
|
1789
|
+
let r = jv_f64(&mut bcx, r);
|
|
1790
|
+
let l = jv_f64(&mut bcx, l);
|
|
1791
|
+
let idc = bcx
|
|
1792
|
+
.ins()
|
|
1793
|
+
.iconst(types::I32, tishlang_bytecode::MathBinaryFn::Pow as i64);
|
|
1794
|
+
let call = bcx.ins().call(math_binary_fref, &[idc, l, r]);
|
|
1795
|
+
let res = bcx.inst_results(call)[0];
|
|
1796
|
+
stack.push(JV::f64(res));
|
|
1263
1797
|
ip += 3;
|
|
1264
1798
|
}
|
|
1265
1799
|
_ => match emit_simple_op(&mut bcx, chunk, code, &mut ip, &mut stack, &[], 0) {
|
|
@@ -1276,7 +1810,13 @@ fn build_loop_region_body(
|
|
|
1276
1810
|
for (&t, &blk) in &exit_blocks {
|
|
1277
1811
|
bcx.switch_to_block(blk);
|
|
1278
1812
|
for (p, &slot) in used_slots.iter().enumerate() {
|
|
1279
|
-
let
|
|
1813
|
+
let raw = bcx.use_var(vars[slot as usize]);
|
|
1814
|
+
// #514: an int slot holds an i64 — flush its exact f64 value back to the buffer.
|
|
1815
|
+
let v = if int_slots.contains(&(slot as usize)) {
|
|
1816
|
+
bcx.ins().fcvt_from_sint(types::F64, raw)
|
|
1817
|
+
} else {
|
|
1818
|
+
raw
|
|
1819
|
+
};
|
|
1280
1820
|
bcx.ins()
|
|
1281
1821
|
.store(MemFlags::new(), v, slots_ptr, (p * 8) as i32);
|
|
1282
1822
|
}
|
|
@@ -1327,6 +1867,90 @@ fn fcmp_f64(bcx: &mut FunctionBuilder, cc: FloatCC, a: ClifValue, b: ClifValue)
|
|
|
1327
1867
|
bcx.ins().select(cond, one, zero)
|
|
1328
1868
|
}
|
|
1329
1869
|
|
|
1870
|
+
/// #168 — which Cranelift representation a JIT stack slot currently holds.
|
|
1871
|
+
///
|
|
1872
|
+
/// `F64`: an f64 number. `Bool`: an f64 constrained to 0.0/1.0 with JS-boolean semantics (the
|
|
1873
|
+
/// old `is_bool` flag — comparisons/`!` produce it, `LoadConst Bool` pushes it). `I32`: an i32
|
|
1874
|
+
/// holding `ToInt32` bits (signed→f64 on materialize). `U32`: an i32 holding `ToUint32` bits
|
|
1875
|
+
/// (UNSIGNED→f64 on materialize — `>>>` results past 2³¹ stay positive numbers).
|
|
1876
|
+
///
|
|
1877
|
+
/// The integer reprs are the point: a bitwise/shift chain (`h = ((h<<13)|(h>>>19)) >>> 0`) used
|
|
1878
|
+
/// to pay `int→f64→int` conversion ROUND-TRIPS between every op, because the stack could only
|
|
1879
|
+
/// say "f64 or bool". Now each such op consumes raw int bits via [`jv_i32_bits`] (an identity
|
|
1880
|
+
/// for `I32`/`U32`) and pushes an int repr; f64 materialization happens once, at a genuine
|
|
1881
|
+
/// boundary (store/return/float-arith/compare/call), via [`jv_f64`]. `ToInt32` and `ToUint32`
|
|
1882
|
+
/// share bit patterns, so `I32` vs `U32` only matters at the f64 boundary (signed vs unsigned
|
|
1883
|
+
/// convert) — the bits themselves are interchangeable as shift/bitwise inputs.
|
|
1884
|
+
#[derive(Clone, Copy, PartialEq, Eq)]
|
|
1885
|
+
enum Repr {
|
|
1886
|
+
F64,
|
|
1887
|
+
Bool,
|
|
1888
|
+
I32,
|
|
1889
|
+
U32,
|
|
1890
|
+
/// #168 int-typed slots: an i64 holding an EXACT integral JS number in [-2^31, 2^32).
|
|
1891
|
+
/// Signedness is encoded in the value itself (an I32 store sign-extends, a U32 store
|
|
1892
|
+
/// zero-extends, an integral constant loads exactly), so a slot whose stores mix `^`
|
|
1893
|
+
/// (ToInt32 domain) and `>>> 0` (ToUint32 domain) needs no dynamic tag: materializing
|
|
1894
|
+
/// is one exact `fcvt_from_sint`, and re-entering the 32-bit domain is one `ireduce`
|
|
1895
|
+
/// (ToInt32 of an integral value in this range IS its low 32 bits).
|
|
1896
|
+
I64Num,
|
|
1897
|
+
}
|
|
1898
|
+
|
|
1899
|
+
/// A typed JIT stack value: the Cranelift SSA value plus its current [`Repr`].
|
|
1900
|
+
#[derive(Clone, Copy)]
|
|
1901
|
+
struct JV {
|
|
1902
|
+
v: ClifValue,
|
|
1903
|
+
repr: Repr,
|
|
1904
|
+
}
|
|
1905
|
+
|
|
1906
|
+
impl JV {
|
|
1907
|
+
fn f64(v: ClifValue) -> Self {
|
|
1908
|
+
Self { v, repr: Repr::F64 }
|
|
1909
|
+
}
|
|
1910
|
+
fn boolean(v: ClifValue) -> Self {
|
|
1911
|
+
Self { v, repr: Repr::Bool }
|
|
1912
|
+
}
|
|
1913
|
+
fn int32(v: ClifValue) -> Self {
|
|
1914
|
+
Self { v, repr: Repr::I32 }
|
|
1915
|
+
}
|
|
1916
|
+
fn uint32(v: ClifValue) -> Self {
|
|
1917
|
+
Self { v, repr: Repr::U32 }
|
|
1918
|
+
}
|
|
1919
|
+
fn is_bool(&self) -> bool {
|
|
1920
|
+
self.repr == Repr::Bool
|
|
1921
|
+
}
|
|
1922
|
+
}
|
|
1923
|
+
|
|
1924
|
+
/// Materialize a [`JV`] as an f64 — identity for `F64`/`Bool` (a Bool already IS an f64 0/1),
|
|
1925
|
+
/// one signed/unsigned convert for the integer reprs.
|
|
1926
|
+
fn jv_f64(bcx: &mut FunctionBuilder, jv: JV) -> ClifValue {
|
|
1927
|
+
match jv.repr {
|
|
1928
|
+
Repr::F64 | Repr::Bool => jv.v,
|
|
1929
|
+
Repr::I32 => bcx.ins().fcvt_from_sint(types::F64, jv.v),
|
|
1930
|
+
Repr::U32 => bcx.ins().fcvt_from_uint(types::F64, jv.v),
|
|
1931
|
+
// Exact by construction: an I64Num is integral and |v| < 2^32 << 2^53.
|
|
1932
|
+
Repr::I64Num => bcx.ins().fcvt_from_sint(types::F64, jv.v),
|
|
1933
|
+
}
|
|
1934
|
+
}
|
|
1935
|
+
|
|
1936
|
+
/// Raw `ToInt32` bit pattern of a [`JV`] as an i32 — an identity (zero instructions) for
|
|
1937
|
+
/// `I32`/`U32` (they share bit patterns), [`js_to_int32`] for the f64 reprs (a Bool's 0.0/1.0
|
|
1938
|
+
/// converts to 0/1 exactly).
|
|
1939
|
+
fn jv_i32_bits(bcx: &mut FunctionBuilder, jv: JV) -> ClifValue {
|
|
1940
|
+
match jv.repr {
|
|
1941
|
+
Repr::I32 | Repr::U32 => jv.v,
|
|
1942
|
+
Repr::F64 | Repr::Bool => js_to_int32(bcx, jv.v),
|
|
1943
|
+
// ToInt32 of an exact integral in [-2^31, 2^32) is its low 32 bits.
|
|
1944
|
+
Repr::I64Num => bcx.ins().ireduce(types::I32, jv.v),
|
|
1945
|
+
}
|
|
1946
|
+
}
|
|
1947
|
+
|
|
1948
|
+
impl JV {
|
|
1949
|
+
fn i64num(v: ClifValue) -> Self {
|
|
1950
|
+
Self { v, repr: Repr::I64Num }
|
|
1951
|
+
}
|
|
1952
|
+
}
|
|
1953
|
+
|
|
1330
1954
|
/// f64 → JS `ToInt32` as an `I32` clif value, matching `tishlang_core::to_int32` (so a JIT-compiled
|
|
1331
1955
|
/// `& | ^ ~` agrees with the VM fallback). Saturating-cast→`ireduce` is the modulo-2³² for finite
|
|
1332
1956
|
/// values and already gives 0 for NaN / `-∞`; the branchless `select` on `|x| < ∞` maps `+∞` (which
|
|
@@ -1457,6 +2081,7 @@ fn compile_chunk(g: &mut JitGlobal, chunk: &Chunk) -> Option<NumericFn> {
|
|
|
1457
2081
|
ctx.func.signature = sig.clone();
|
|
1458
2082
|
let self_ref = g.module.declare_func_in_func(id, &mut ctx.func);
|
|
1459
2083
|
let math_fref = g.module.declare_func_in_func(g.math_call_id, &mut ctx.func);
|
|
2084
|
+
let math_binary_fref = g.module.declare_func_in_func(g.math_binary_call_id, &mut ctx.func);
|
|
1460
2085
|
// #189: import the `tish_jv_*` FuncRefs into this function when it has local arrays.
|
|
1461
2086
|
let jv_ctx = if is_jv {
|
|
1462
2087
|
Some(JvCtx {
|
|
@@ -1485,6 +2110,7 @@ fn compile_chunk(g: &mut JitGlobal, chunk: &Chunk) -> Option<NumericFn> {
|
|
|
1485
2110
|
0,
|
|
1486
2111
|
recur_guard,
|
|
1487
2112
|
math_fref,
|
|
2113
|
+
math_binary_fref,
|
|
1488
2114
|
jv_ctx.as_ref(),
|
|
1489
2115
|
&resolved,
|
|
1490
2116
|
) {
|
|
@@ -1736,6 +2362,7 @@ fn compile_chunk_arrays(
|
|
|
1736
2362
|
// `handles_ptr`/`deopt_ptr` — only the numeric args are re-marshalled per level.
|
|
1737
2363
|
let self_ref = g.module.declare_func_in_func(id, &mut ctx.func);
|
|
1738
2364
|
let math_fref = g.module.declare_func_in_func(g.math_call_id, &mut ctx.func);
|
|
2365
|
+
let math_binary_fref = g.module.declare_func_in_func(g.math_binary_call_id, &mut ctx.func);
|
|
1739
2366
|
// #187: array-mode functions (e.g. spectral_norm's multiplyAv) may call a register-f64 callee.
|
|
1740
2367
|
let resolved = build_resolved_callees(g, chunk, &mut ctx.func);
|
|
1741
2368
|
let mut fbctx = FunctionBuilderContext::new();
|
|
@@ -1749,6 +2376,7 @@ fn compile_chunk_arrays(
|
|
|
1749
2376
|
mask,
|
|
1750
2377
|
recursive, // #187: array-mode recursion guard (the entry SP-bail keys off this)
|
|
1751
2378
|
math_fref,
|
|
2379
|
+
math_binary_fref,
|
|
1752
2380
|
None,
|
|
1753
2381
|
&resolved,
|
|
1754
2382
|
)
|
|
@@ -1800,7 +2428,7 @@ fn emit_simple_op(
|
|
|
1800
2428
|
chunk: &Chunk,
|
|
1801
2429
|
code: &[u8],
|
|
1802
2430
|
ip: &mut usize,
|
|
1803
|
-
stack: &mut Vec<
|
|
2431
|
+
stack: &mut Vec<JV>,
|
|
1804
2432
|
params: &[ClifValue],
|
|
1805
2433
|
arity: usize,
|
|
1806
2434
|
) -> SimpleOp {
|
|
@@ -1825,7 +2453,7 @@ fn emit_simple_op(
|
|
|
1825
2453
|
if slot >= arity {
|
|
1826
2454
|
return SimpleOp::Unsupported;
|
|
1827
2455
|
}
|
|
1828
|
-
stack.push((params[slot]
|
|
2456
|
+
stack.push(JV::f64(params[slot]));
|
|
1829
2457
|
}
|
|
1830
2458
|
Opcode::LoadConst => {
|
|
1831
2459
|
let idx = match read_u16(code, ip) {
|
|
@@ -1835,11 +2463,11 @@ fn emit_simple_op(
|
|
|
1835
2463
|
match chunk.constants.get(idx) {
|
|
1836
2464
|
Some(Constant::Number(n)) => {
|
|
1837
2465
|
let v = bcx.ins().f64const(*n);
|
|
1838
|
-
stack.push((v
|
|
2466
|
+
stack.push(JV::f64(v));
|
|
1839
2467
|
}
|
|
1840
2468
|
Some(Constant::Bool(b)) => {
|
|
1841
2469
|
let v = bcx.ins().f64const(if *b { 1.0 } else { 0.0 });
|
|
1842
|
-
stack.push((v
|
|
2470
|
+
stack.push(JV::boolean(v));
|
|
1843
2471
|
}
|
|
1844
2472
|
_ => return SimpleOp::Unsupported,
|
|
1845
2473
|
}
|
|
@@ -1852,121 +2480,143 @@ fn emit_simple_op(
|
|
|
1852
2480
|
if stack.len() < 2 {
|
|
1853
2481
|
return SimpleOp::Unsupported;
|
|
1854
2482
|
}
|
|
1855
|
-
let
|
|
1856
|
-
let
|
|
2483
|
+
let r = stack.pop().unwrap();
|
|
2484
|
+
let l = stack.pop().unwrap();
|
|
1857
2485
|
// #187: a bool value (from a bool slot or a `LoadConst Bool`) can't take part in an
|
|
1858
2486
|
// EQUALITY compare here — the JIT compares the `f64` 0/1 bits, but JS `===`/`!==` (and
|
|
1859
2487
|
// strict `==`/`!=`) treat `0 === false` as FALSE across types. Bail so the interpreter
|
|
1860
2488
|
// decides. Relational (`<`/`>`/…) coerces bool→0/1 in both, so those stay JIT'd.
|
|
2489
|
+
// Integer reprs are plain NUMBERS, so they take part in every compare.
|
|
1861
2490
|
let is_eq = matches!(
|
|
1862
2491
|
bop,
|
|
1863
2492
|
BinOp::Eq | BinOp::Ne | BinOp::StrictEq | BinOp::StrictNe
|
|
1864
2493
|
);
|
|
1865
|
-
if is_eq && (
|
|
2494
|
+
if is_eq && (l.is_bool() || r.is_bool()) {
|
|
1866
2495
|
return SimpleOp::Unsupported;
|
|
1867
2496
|
}
|
|
1868
|
-
|
|
1869
|
-
|
|
1870
|
-
|
|
1871
|
-
|
|
1872
|
-
|
|
1873
|
-
|
|
1874
|
-
|
|
1875
|
-
|
|
1876
|
-
|
|
1877
|
-
|
|
1878
|
-
|
|
1879
|
-
|
|
1880
|
-
|
|
1881
|
-
|
|
1882
|
-
BinOp::Mul =>
|
|
1883
|
-
|
|
1884
|
-
|
|
1885
|
-
|
|
1886
|
-
BinOp::
|
|
1887
|
-
|
|
1888
|
-
|
|
1889
|
-
|
|
2497
|
+
// #168: float arithmetic / comparisons materialize both operands as f64 at this
|
|
2498
|
+
// boundary; bitwise/shift ops consume raw int bits (identity for I32/U32 operands)
|
|
2499
|
+
// and PUSH an integer repr — a chained `((h<<13)|(h>>>19))>>>0` stays in i32
|
|
2500
|
+
// registers with zero intermediate converts. `Mul` stays f64 ON PURPOSE: V8 rounds
|
|
2501
|
+
// `h * K` past 2^53 the same way, and the gauntlet checksum pins that agreement.
|
|
2502
|
+
let v: JV = match bop {
|
|
2503
|
+
BinOp::Add => {
|
|
2504
|
+
let (lf, rf) = (jv_f64(bcx, l), jv_f64(bcx, r));
|
|
2505
|
+
JV::f64(bcx.ins().fadd(lf, rf))
|
|
2506
|
+
}
|
|
2507
|
+
BinOp::Sub => {
|
|
2508
|
+
let (lf, rf) = (jv_f64(bcx, l), jv_f64(bcx, r));
|
|
2509
|
+
JV::f64(bcx.ins().fsub(lf, rf))
|
|
2510
|
+
}
|
|
2511
|
+
BinOp::Mul => {
|
|
2512
|
+
let (lf, rf) = (jv_f64(bcx, l), jv_f64(bcx, r));
|
|
2513
|
+
JV::f64(bcx.ins().fmul(lf, rf))
|
|
2514
|
+
}
|
|
2515
|
+
BinOp::Div => {
|
|
2516
|
+
let (lf, rf) = (jv_f64(bcx, l), jv_f64(bcx, r));
|
|
2517
|
+
JV::f64(bcx.ins().fdiv(lf, rf))
|
|
2518
|
+
}
|
|
2519
|
+
BinOp::Eq | BinOp::StrictEq => {
|
|
2520
|
+
let (lf, rf) = (jv_f64(bcx, l), jv_f64(bcx, r));
|
|
2521
|
+
JV::boolean(fcmp_f64(bcx, FloatCC::Equal, lf, rf))
|
|
2522
|
+
}
|
|
2523
|
+
BinOp::Ne | BinOp::StrictNe => {
|
|
2524
|
+
let (lf, rf) = (jv_f64(bcx, l), jv_f64(bcx, r));
|
|
2525
|
+
JV::boolean(fcmp_f64(bcx, FloatCC::NotEqual, lf, rf))
|
|
2526
|
+
}
|
|
2527
|
+
BinOp::Lt => {
|
|
2528
|
+
let (lf, rf) = (jv_f64(bcx, l), jv_f64(bcx, r));
|
|
2529
|
+
JV::boolean(fcmp_f64(bcx, FloatCC::LessThan, lf, rf))
|
|
2530
|
+
}
|
|
2531
|
+
BinOp::Le => {
|
|
2532
|
+
let (lf, rf) = (jv_f64(bcx, l), jv_f64(bcx, r));
|
|
2533
|
+
JV::boolean(fcmp_f64(bcx, FloatCC::LessThanOrEqual, lf, rf))
|
|
2534
|
+
}
|
|
2535
|
+
BinOp::Gt => {
|
|
2536
|
+
let (lf, rf) = (jv_f64(bcx, l), jv_f64(bcx, r));
|
|
2537
|
+
JV::boolean(fcmp_f64(bcx, FloatCC::GreaterThan, lf, rf))
|
|
2538
|
+
}
|
|
2539
|
+
BinOp::Ge => {
|
|
2540
|
+
let (lf, rf) = (jv_f64(bcx, l), jv_f64(bcx, r));
|
|
2541
|
+
JV::boolean(fcmp_f64(bcx, FloatCC::GreaterThanOrEqual, lf, rf))
|
|
2542
|
+
}
|
|
1890
2543
|
BinOp::Mod => {
|
|
1891
2544
|
// f64 remainder a - trunc(a/b)*b — exactly Rust's `%`, which the
|
|
1892
2545
|
// VM's eval_binop uses, so JIT and VM-fallback agree bit-for-bit.
|
|
1893
|
-
let
|
|
2546
|
+
let (lf, rf) = (jv_f64(bcx, l), jv_f64(bcx, r));
|
|
2547
|
+
let q = bcx.ins().fdiv(lf, rf);
|
|
1894
2548
|
let t = bcx.ins().trunc(q);
|
|
1895
|
-
let p = bcx.ins().fmul(t,
|
|
1896
|
-
bcx.ins().fsub(
|
|
2549
|
+
let p = bcx.ins().fmul(t, rf);
|
|
2550
|
+
JV::f64(bcx.ins().fsub(lf, p))
|
|
1897
2551
|
}
|
|
1898
|
-
// Bitwise AND/OR/XOR via JS ToInt32 (modulo 2³², NaN/±∞ → 0) —
|
|
2552
|
+
// Bitwise AND/OR/XOR via JS ToInt32 (modulo 2³², NaN/±∞ → 0) — [`jv_i32_bits`]
|
|
2553
|
+
// is [`js_to_int32`] for f64 operands and an IDENTITY for int-repr operands.
|
|
1899
2554
|
BinOp::BitAnd | BinOp::BitOr | BinOp::BitXor => {
|
|
1900
|
-
let li =
|
|
1901
|
-
let ri =
|
|
2555
|
+
let li = jv_i32_bits(bcx, l);
|
|
2556
|
+
let ri = jv_i32_bits(bcx, r);
|
|
1902
2557
|
let res = match bop {
|
|
1903
2558
|
BinOp::BitAnd => bcx.ins().band(li, ri),
|
|
1904
2559
|
BinOp::BitOr => bcx.ins().bor(li, ri),
|
|
1905
2560
|
BinOp::BitXor => bcx.ins().bxor(li, ri),
|
|
1906
2561
|
_ => unreachable!(),
|
|
1907
2562
|
};
|
|
1908
|
-
|
|
1909
|
-
}
|
|
1910
|
-
// Shifts. JS masks the count to the low 5 bits (`& 31`); the low 5 bits of
|
|
1911
|
-
// equal `ToUint32(r)`, so `
|
|
1912
|
-
// (`& 31`) so correctness never depends on Cranelift's own
|
|
1913
|
-
// `<<`/`>>` are signed-domain (
|
|
1914
|
-
//
|
|
1915
|
-
//
|
|
1916
|
-
//
|
|
2563
|
+
JV::int32(res)
|
|
2564
|
+
}
|
|
2565
|
+
// Shifts. JS masks the count to the low 5 bits (`& 31`); the low 5 bits of
|
|
2566
|
+
// `ToInt32(r)` equal `ToUint32(r)`, so `jv_i32_bits(r)` carries the right amount.
|
|
2567
|
+
// We mask explicitly (`& 31`) so correctness never depends on Cranelift's own
|
|
2568
|
+
// amount-masking convention. `<<`/`>>` are signed-domain (I32 repr — signed→f64
|
|
2569
|
+
// when materialized); `>>>` is logical on the unsigned bits (U32 repr — an
|
|
2570
|
+
// UNSIGNED→f64 materialize keeps a bit-31 result a positive number, JS `>>>`).
|
|
2571
|
+
// Bit-for-bit with vm.rs `eval_binop`: Shl/Shr = `to_int32(l).wrapping_sh*
|
|
2572
|
+
// (to_uint32(r))`, UShr = `to_uint32(l).wrapping_shr(to_uint32(r))`.
|
|
1917
2573
|
BinOp::Shl | BinOp::Shr | BinOp::UShr => {
|
|
1918
|
-
let li =
|
|
1919
|
-
let amt =
|
|
2574
|
+
let li = jv_i32_bits(bcx, l);
|
|
2575
|
+
let amt = jv_i32_bits(bcx, r);
|
|
1920
2576
|
let mask = bcx.ins().iconst(types::I32, 31);
|
|
1921
2577
|
let amt = bcx.ins().band(amt, mask);
|
|
1922
2578
|
match bop {
|
|
1923
|
-
BinOp::Shl =>
|
|
1924
|
-
|
|
1925
|
-
|
|
1926
|
-
}
|
|
1927
|
-
BinOp::Shr => {
|
|
1928
|
-
let res = bcx.ins().sshr(li, amt);
|
|
1929
|
-
bcx.ins().fcvt_from_sint(types::F64, res)
|
|
1930
|
-
}
|
|
1931
|
-
// UShr: logical shift on the same 32-bit value bits as ToUint32(l), then
|
|
1932
|
-
// unsigned→f64 so a result with bit 31 set stays a positive number (JS `>>>`).
|
|
1933
|
-
_ => {
|
|
1934
|
-
let res = bcx.ins().ushr(li, amt);
|
|
1935
|
-
bcx.ins().fcvt_from_uint(types::F64, res)
|
|
1936
|
-
}
|
|
2579
|
+
BinOp::Shl => JV::int32(bcx.ins().ishl(li, amt)),
|
|
2580
|
+
BinOp::Shr => JV::int32(bcx.ins().sshr(li, amt)),
|
|
2581
|
+
_ => JV::uint32(bcx.ins().ushr(li, amt)),
|
|
1937
2582
|
}
|
|
1938
2583
|
}
|
|
1939
2584
|
// Pow/In/And/Or: fall back to the VM.
|
|
1940
2585
|
_ => return SimpleOp::Unsupported,
|
|
1941
2586
|
};
|
|
1942
|
-
stack.push(
|
|
2587
|
+
stack.push(v);
|
|
1943
2588
|
}
|
|
1944
2589
|
Opcode::UnaryOp => {
|
|
1945
2590
|
let uop = match read_u16(code, ip).map(|r| r as u8).and_then(u8_to_unaryop) {
|
|
1946
2591
|
Some(u) => u,
|
|
1947
2592
|
None => return SimpleOp::Unsupported,
|
|
1948
2593
|
};
|
|
1949
|
-
let
|
|
2594
|
+
let o = match stack.pop() {
|
|
1950
2595
|
Some(x) => x,
|
|
1951
2596
|
None => return SimpleOp::Unsupported,
|
|
1952
2597
|
};
|
|
1953
|
-
let
|
|
1954
|
-
UnaryOp::Neg =>
|
|
1955
|
-
|
|
2598
|
+
let v: JV = match uop {
|
|
2599
|
+
UnaryOp::Neg => {
|
|
2600
|
+
let of = jv_f64(bcx, o);
|
|
2601
|
+
JV::f64(bcx.ins().fneg(of))
|
|
2602
|
+
}
|
|
2603
|
+
UnaryOp::Pos => JV::f64(jv_f64(bcx, o)),
|
|
1956
2604
|
UnaryOp::Not => {
|
|
2605
|
+
let of = jv_f64(bcx, o);
|
|
1957
2606
|
let zero = bcx.ins().f64const(0.0);
|
|
1958
|
-
(fcmp_f64(bcx, FloatCC::Equal,
|
|
2607
|
+
JV::boolean(fcmp_f64(bcx, FloatCC::Equal, of, zero))
|
|
1959
2608
|
}
|
|
1960
|
-
// `~x` = `!ToInt32(x)
|
|
1961
|
-
// matching the VM so a JIT-compiled `~`
|
|
2609
|
+
// `~x` = `!ToInt32(x)` — JS ToInt32 (modulo, NaN/±∞ → 0) via [`jv_i32_bits`]
|
|
2610
|
+
// (identity for an int-repr operand), matching the VM so a JIT-compiled `~`
|
|
2611
|
+
// can't diverge on large/non-finite values. Pushes I32 — a `~` chain stays
|
|
2612
|
+
// in integer registers.
|
|
1962
2613
|
UnaryOp::BitNot => {
|
|
1963
|
-
let oi =
|
|
1964
|
-
|
|
1965
|
-
(bcx.ins().fcvt_from_sint(types::F64, res), false)
|
|
2614
|
+
let oi = jv_i32_bits(bcx, o);
|
|
2615
|
+
JV::int32(bcx.ins().bnot(oi))
|
|
1966
2616
|
}
|
|
1967
2617
|
_ => return SimpleOp::Unsupported,
|
|
1968
2618
|
};
|
|
1969
|
-
stack.push(
|
|
2619
|
+
stack.push(v);
|
|
1970
2620
|
}
|
|
1971
2621
|
_ => unreachable!("guarded above"),
|
|
1972
2622
|
}
|
|
@@ -1989,12 +2639,209 @@ fn falsy_flag(bcx: &mut FunctionBuilder, cond: ClifValue) -> ClifValue {
|
|
|
1989
2639
|
/// branch, nested branches, calls, member/index, or mismatched `is_bool` all return `None` so the
|
|
1990
2640
|
/// VM runs the chunk instead — purely additive. Returns `Some(result_is_bool)`.
|
|
1991
2641
|
/// Byte size of an opcode the loop-JIT understands; `None` ⇒ unsupported (bail → VM).
|
|
2642
|
+
/// #203: is the op at `ip` a `BinOp` whose operator is `**` (Pow)? Lets the loop builders intercept
|
|
2643
|
+
/// `a ** b` and lower it to a `tish_math_binary_call(Pow, …)` host call before the generic
|
|
2644
|
+
/// `emit_simple_op` (which rejects Pow) would bail the function.
|
|
2645
|
+
fn is_binop_pow(op: Opcode, code: &[u8], ip: usize) -> bool {
|
|
2646
|
+
op == Opcode::BinOp
|
|
2647
|
+
&& peek_u16(code, ip + 1).map(|r| r as u8).and_then(u8_to_binop) == Some(BinOp::Pow)
|
|
2648
|
+
}
|
|
2649
|
+
|
|
2650
|
+
/// #203 branch-free ternary lowering in the LOOP builder. **Default ON**; `TISH_JIT_TERNARY=0` falls
|
|
2651
|
+
/// back to bailing the function (byte-identical, just interpreted). Additive: a non-matching shape or
|
|
2652
|
+
/// disabled flag simply doesn't take the fast path.
|
|
2653
|
+
fn jit_ternary_enabled() -> bool {
|
|
2654
|
+
static ENABLED: OnceLock<bool> = OnceLock::new();
|
|
2655
|
+
*ENABLED.get_or_init(|| {
|
|
2656
|
+
std::env::var("TISH_JIT_TERNARY")
|
|
2657
|
+
.map(|v| v != "0")
|
|
2658
|
+
.unwrap_or(true)
|
|
2659
|
+
})
|
|
2660
|
+
}
|
|
2661
|
+
|
|
2662
|
+
/// #203: ops allowed inside a ternary ARM — pure, side-effect-free value producers the inline
|
|
2663
|
+
/// `select` can emit UNCONDITIONALLY (both arms run, then `select` picks). No control flow / calls /
|
|
2664
|
+
/// stores / member / index / array ops (which could have side effects or need a block).
|
|
2665
|
+
fn is_ternary_arm_op(op: Opcode) -> bool {
|
|
2666
|
+
matches!(
|
|
2667
|
+
op,
|
|
2668
|
+
Opcode::LoadLocal
|
|
2669
|
+
| Opcode::LoadConst
|
|
2670
|
+
| Opcode::BinOp
|
|
2671
|
+
| Opcode::UnaryOp
|
|
2672
|
+
| Opcode::MathUnary
|
|
2673
|
+
| Opcode::MathBinary
|
|
2674
|
+
)
|
|
2675
|
+
}
|
|
2676
|
+
|
|
2677
|
+
/// #203: if the `JumpIfFalse` at `jif_ip` begins a clean ternary `cond ? A : B` — a forward branch
|
|
2678
|
+
/// whose THEN arm is straight-line pure-value ops ending in a `Jump`, whose ELSE arm begins exactly
|
|
2679
|
+
/// where that `Jump` lands past and runs pure-value ops to a single merge — return the merge ip.
|
|
2680
|
+
/// Conservative: a nested branch in an arm, a back-edge, or any non-value op → `None`, and the caller
|
|
2681
|
+
/// leaves it to the block CFG (which bails) / VM. Sound for compiler-generated bytecode: nothing jumps
|
|
2682
|
+
/// INTO a ternary's arms, so treating the whole span as one inline unit never drops a CFG edge.
|
|
2683
|
+
fn ternary_span(code: &[u8], jif_ip: usize) -> Option<(usize, usize, usize)> {
|
|
2684
|
+
if Opcode::from_u8(*code.get(jif_ip)?)? != Opcode::JumpIfFalse {
|
|
2685
|
+
return None;
|
|
2686
|
+
}
|
|
2687
|
+
let off = peek_u16(code, jif_ip + 1)? as i16 as isize;
|
|
2688
|
+
let else_target = ((jif_ip + 3) as isize + off).max(0) as usize;
|
|
2689
|
+
// THEN arm: pure-value ops from jif+3 up to the trailing `Jump`.
|
|
2690
|
+
let then_start = jif_ip + 3;
|
|
2691
|
+
let mut tip = then_start;
|
|
2692
|
+
loop {
|
|
2693
|
+
let op = Opcode::from_u8(*code.get(tip)?)?;
|
|
2694
|
+
if op == Opcode::Jump {
|
|
2695
|
+
break;
|
|
2696
|
+
}
|
|
2697
|
+
if !is_ternary_arm_op(op) {
|
|
2698
|
+
return None;
|
|
2699
|
+
}
|
|
2700
|
+
tip += op_size(op)?;
|
|
2701
|
+
}
|
|
2702
|
+
let then_end = tip; // at the trailing Jump
|
|
2703
|
+
let joff = peek_u16(code, tip + 1)? as i16 as isize;
|
|
2704
|
+
let else_start = tip + 3;
|
|
2705
|
+
let merge = (else_start as isize + joff).max(0) as usize;
|
|
2706
|
+
if else_target != else_start || merge <= else_start {
|
|
2707
|
+
return None;
|
|
2708
|
+
}
|
|
2709
|
+
// ELSE arm: pure-value ops from else_start up to merge.
|
|
2710
|
+
let mut eip = else_start;
|
|
2711
|
+
while eip < merge {
|
|
2712
|
+
let op = Opcode::from_u8(*code.get(eip)?)?;
|
|
2713
|
+
if !is_ternary_arm_op(op) {
|
|
2714
|
+
return None;
|
|
2715
|
+
}
|
|
2716
|
+
eip += op_size(op)?;
|
|
2717
|
+
}
|
|
2718
|
+
if eip != merge || then_end <= then_start {
|
|
2719
|
+
return None; // else arm must land exactly on merge; then arm must be non-empty
|
|
2720
|
+
}
|
|
2721
|
+
Some((then_end, else_start, merge))
|
|
2722
|
+
}
|
|
2723
|
+
|
|
2724
|
+
/// #203: emit one ternary ARM — the pure-value ops in `[start, end)` — in the `build_body_cfg` context
|
|
2725
|
+
/// (`LoadLocal` reads the slot's CURRENT SSA `Variable`, unlike `emit_simple_op`'s params-only path).
|
|
2726
|
+
/// Returns `false` (bail the whole compile) on any op not in the arm whitelist or a decode failure.
|
|
2727
|
+
/// Both arms are emitted unconditionally, so every op here is side-effect-free (`is_ternary_arm_op`).
|
|
2728
|
+
#[allow(clippy::too_many_arguments)]
|
|
2729
|
+
fn emit_ternary_arm(
|
|
2730
|
+
bcx: &mut FunctionBuilder,
|
|
2731
|
+
chunk: &Chunk,
|
|
2732
|
+
code: &[u8],
|
|
2733
|
+
start: usize,
|
|
2734
|
+
end: usize,
|
|
2735
|
+
stack: &mut Vec<JV>,
|
|
2736
|
+
vars: &[Variable],
|
|
2737
|
+
int_slots: &std::collections::HashSet<usize>,
|
|
2738
|
+
bool_slots: &std::collections::HashSet<usize>,
|
|
2739
|
+
math_fref: cranelift::codegen::ir::FuncRef,
|
|
2740
|
+
math_binary_fref: cranelift::codegen::ir::FuncRef,
|
|
2741
|
+
) -> bool {
|
|
2742
|
+
let mut ip = start;
|
|
2743
|
+
while ip < end {
|
|
2744
|
+
let op = match Opcode::from_u8(code[ip]) {
|
|
2745
|
+
Some(o) => o,
|
|
2746
|
+
None => return false,
|
|
2747
|
+
};
|
|
2748
|
+
match op {
|
|
2749
|
+
Opcode::LoadLocal => {
|
|
2750
|
+
let slot = match peek_u16(code, ip + 1) {
|
|
2751
|
+
Some(s) => s as usize,
|
|
2752
|
+
None => return false,
|
|
2753
|
+
};
|
|
2754
|
+
let v = match vars.get(slot) {
|
|
2755
|
+
Some(v) => *v,
|
|
2756
|
+
None => return false,
|
|
2757
|
+
};
|
|
2758
|
+
let lv = bcx.use_var(v);
|
|
2759
|
+
stack.push(if bool_slots.contains(&slot) {
|
|
2760
|
+
JV::boolean(lv)
|
|
2761
|
+
} else if int_slots.contains(&slot) {
|
|
2762
|
+
JV::i64num(lv)
|
|
2763
|
+
} else {
|
|
2764
|
+
JV::f64(lv)
|
|
2765
|
+
});
|
|
2766
|
+
ip += 3;
|
|
2767
|
+
}
|
|
2768
|
+
Opcode::MathUnary => {
|
|
2769
|
+
let id = match peek_u16(code, ip + 1) {
|
|
2770
|
+
Some(v) => v,
|
|
2771
|
+
None => return false,
|
|
2772
|
+
};
|
|
2773
|
+
let mfn = match MathUnaryFn::from_u16(id) {
|
|
2774
|
+
Some(m) => m,
|
|
2775
|
+
None => return false,
|
|
2776
|
+
};
|
|
2777
|
+
let x = match stack.pop() {
|
|
2778
|
+
Some(v) => v,
|
|
2779
|
+
None => return false,
|
|
2780
|
+
};
|
|
2781
|
+
let x = jv_f64(bcx, x);
|
|
2782
|
+
let r = emit_math_unary(bcx, math_fref, mfn, x);
|
|
2783
|
+
stack.push(JV::f64(r));
|
|
2784
|
+
ip += 3;
|
|
2785
|
+
}
|
|
2786
|
+
Opcode::MathBinary => {
|
|
2787
|
+
let id = match peek_u16(code, ip + 1) {
|
|
2788
|
+
Some(v) => v,
|
|
2789
|
+
None => return false,
|
|
2790
|
+
};
|
|
2791
|
+
let b = match stack.pop() {
|
|
2792
|
+
Some(v) => v,
|
|
2793
|
+
None => return false,
|
|
2794
|
+
};
|
|
2795
|
+
let a = match stack.pop() {
|
|
2796
|
+
Some(v) => v,
|
|
2797
|
+
None => return false,
|
|
2798
|
+
};
|
|
2799
|
+
let b = jv_f64(bcx, b);
|
|
2800
|
+
let a = jv_f64(bcx, a);
|
|
2801
|
+
let idc = bcx.ins().iconst(types::I32, id as i64);
|
|
2802
|
+
let call = bcx.ins().call(math_binary_fref, &[idc, a, b]);
|
|
2803
|
+
stack.push(JV::f64(bcx.inst_results(call)[0]));
|
|
2804
|
+
ip += 3;
|
|
2805
|
+
}
|
|
2806
|
+
_ if is_binop_pow(op, code, ip) => {
|
|
2807
|
+
let r = match stack.pop() {
|
|
2808
|
+
Some(v) => v,
|
|
2809
|
+
None => return false,
|
|
2810
|
+
};
|
|
2811
|
+
let l = match stack.pop() {
|
|
2812
|
+
Some(v) => v,
|
|
2813
|
+
None => return false,
|
|
2814
|
+
};
|
|
2815
|
+
let r = jv_f64(bcx, r);
|
|
2816
|
+
let l = jv_f64(bcx, l);
|
|
2817
|
+
let idc = bcx
|
|
2818
|
+
.ins()
|
|
2819
|
+
.iconst(types::I32, tishlang_bytecode::MathBinaryFn::Pow as i64);
|
|
2820
|
+
let call = bcx.ins().call(math_binary_fref, &[idc, l, r]);
|
|
2821
|
+
stack.push(JV::f64(bcx.inst_results(call)[0]));
|
|
2822
|
+
ip += 3;
|
|
2823
|
+
}
|
|
2824
|
+
// LoadConst / BinOp / UnaryOp are param-free — `emit_simple_op` handles them (advancing a
|
|
2825
|
+
// local ip copy). It only ever touches `stack`, not slots, for these.
|
|
2826
|
+
Opcode::LoadConst | Opcode::BinOp | Opcode::UnaryOp => {
|
|
2827
|
+
let mut sip = ip;
|
|
2828
|
+
match emit_simple_op(bcx, chunk, code, &mut sip, stack, &[], 0) {
|
|
2829
|
+
SimpleOp::Handled(_) => ip = sip,
|
|
2830
|
+
_ => return false,
|
|
2831
|
+
}
|
|
2832
|
+
}
|
|
2833
|
+
_ => return false,
|
|
2834
|
+
}
|
|
2835
|
+
}
|
|
2836
|
+
true
|
|
2837
|
+
}
|
|
2838
|
+
|
|
1992
2839
|
fn op_size(op: Opcode) -> Option<usize> {
|
|
1993
2840
|
use Opcode::*;
|
|
1994
2841
|
Some(match op {
|
|
1995
2842
|
Nop | Pop | Dup | Return | LoopVarsEnd | EnterBlock | ExitBlock | GetIndex | SetIndex => 1,
|
|
1996
2843
|
LoadLocal | StoreLocal | LoadConst | BinOp | UnaryOp | Jump | JumpIfFalse | JumpBack
|
|
1997
|
-
| LoopVarsBegin | SelfCall | MathUnary => 3,
|
|
2844
|
+
| LoopVarsBegin | SelfCall | MathUnary | MathBinary => 3,
|
|
1998
2845
|
_ => return None,
|
|
1999
2846
|
})
|
|
2000
2847
|
}
|
|
@@ -2123,6 +2970,217 @@ fn classify_bool_slots(chunk: &Chunk) -> std::collections::HashSet<usize> {
|
|
|
2123
2970
|
set
|
|
2124
2971
|
}
|
|
2125
2972
|
|
|
2973
|
+
/// #168: slots eligible for i64 int-typed storage ([`Repr::I64Num`]): every `StoreLocal` to the
|
|
2974
|
+
/// slot is IMMEDIATELY preceded (linearly — statement boundaries have empty stacks, and the
|
|
2975
|
+
/// ternary shape bails out of `build_body_cfg`, so the linear predecessor IS the value producer;
|
|
2976
|
+
/// same assumption [`classify_bool_slots`] rests on) by an op that pushes integer bits — a
|
|
2977
|
+
/// bitwise/shift `BinOp`, a `~` `UnaryOp`, or a `LoadConst` of an integral Number in
|
|
2978
|
+
/// [-2^31, 2^32) (excluding `-0`, whose sign an integer store would erase). Param slots are
|
|
2979
|
+
/// excluded (they arrive as arbitrary f64s through the ABI). Like `classify_bool_slots` this is
|
|
2980
|
+
/// an OPTIMISTIC tag: a merge point could make the linear predecessor differ from the dynamic
|
|
2981
|
+
/// one, so the StoreLocal translator re-checks the actual stored repr and BAILS compilation on
|
|
2982
|
+
/// any non-integer store to a tagged slot — misclassification runs the VM, never miscompiles.
|
|
2983
|
+
fn classify_int_slots(chunk: &Chunk, arity: usize) -> std::collections::HashSet<usize> {
|
|
2984
|
+
if !jit_int_slots_enabled() {
|
|
2985
|
+
return std::collections::HashSet::new();
|
|
2986
|
+
}
|
|
2987
|
+
let code = &chunk.code;
|
|
2988
|
+
let mut int_stores: std::collections::HashSet<usize> = std::collections::HashSet::new();
|
|
2989
|
+
let mut other_stores: std::collections::HashSet<usize> = std::collections::HashSet::new();
|
|
2990
|
+
let mut ip = 0usize;
|
|
2991
|
+
let mut prev_pushes_int = false;
|
|
2992
|
+
while ip < code.len() {
|
|
2993
|
+
let op = match Opcode::from_u8(code[ip]) {
|
|
2994
|
+
Some(o) => o,
|
|
2995
|
+
None => break,
|
|
2996
|
+
};
|
|
2997
|
+
let size = match op.instruction_size(code, ip) {
|
|
2998
|
+
Some(s) => s,
|
|
2999
|
+
None => break,
|
|
3000
|
+
};
|
|
3001
|
+
if op == Opcode::StoreLocal {
|
|
3002
|
+
if let Some(s) = peek_u16(code, ip + 1) {
|
|
3003
|
+
let slot = s as usize;
|
|
3004
|
+
if prev_pushes_int && slot >= arity {
|
|
3005
|
+
int_stores.insert(slot);
|
|
3006
|
+
} else {
|
|
3007
|
+
other_stores.insert(slot);
|
|
3008
|
+
}
|
|
3009
|
+
}
|
|
3010
|
+
}
|
|
3011
|
+
prev_pushes_int = match op {
|
|
3012
|
+
Opcode::BinOp => matches!(
|
|
3013
|
+
peek_u16(code, ip + 1)
|
|
3014
|
+
.map(|r| r as u8)
|
|
3015
|
+
.and_then(u8_to_binop),
|
|
3016
|
+
Some(
|
|
3017
|
+
BinOp::BitAnd
|
|
3018
|
+
| BinOp::BitOr
|
|
3019
|
+
| BinOp::BitXor
|
|
3020
|
+
| BinOp::Shl
|
|
3021
|
+
| BinOp::Shr
|
|
3022
|
+
| BinOp::UShr
|
|
3023
|
+
)
|
|
3024
|
+
),
|
|
3025
|
+
Opcode::UnaryOp => matches!(
|
|
3026
|
+
peek_u16(code, ip + 1)
|
|
3027
|
+
.map(|r| r as u8)
|
|
3028
|
+
.and_then(u8_to_unaryop),
|
|
3029
|
+
Some(UnaryOp::BitNot)
|
|
3030
|
+
),
|
|
3031
|
+
Opcode::LoadConst => matches!(
|
|
3032
|
+
peek_u16(code, ip + 1).and_then(|i| chunk.constants.get(i as usize)),
|
|
3033
|
+
Some(Constant::Number(n))
|
|
3034
|
+
if n.fract() == 0.0
|
|
3035
|
+
&& *n >= -(2f64.powi(31))
|
|
3036
|
+
&& *n < 2f64.powi(32)
|
|
3037
|
+
&& n.to_bits() != (-0f64).to_bits()
|
|
3038
|
+
),
|
|
3039
|
+
_ => false,
|
|
3040
|
+
};
|
|
3041
|
+
ip += size;
|
|
3042
|
+
}
|
|
3043
|
+
int_stores
|
|
3044
|
+
.difference(&other_stores)
|
|
3045
|
+
.copied()
|
|
3046
|
+
.collect()
|
|
3047
|
+
}
|
|
3048
|
+
|
|
3049
|
+
/// #203 — the array live-in slots of an OSR loop region (the matmul lever), with the written subset.
|
|
3050
|
+
#[cfg(not(target_arch = "wasm32"))]
|
|
3051
|
+
struct OsrArrays {
|
|
3052
|
+
/// Array-holding slots in ascending slot order — the marshalling order of the handles buffer.
|
|
3053
|
+
slots: Vec<u16>,
|
|
3054
|
+
/// Subset of `slots` written via `SetIndex` (copied back only after a clean, non-deopt exit).
|
|
3055
|
+
writable: std::collections::BTreeSet<u16>,
|
|
3056
|
+
}
|
|
3057
|
+
|
|
3058
|
+
/// #203 — classify the array live-in slots of an OSR loop region `[header_ip, region_end)`. Abstract-
|
|
3059
|
+
/// interprets the region's operand stack to find every slot used PURELY as the base of an `arr[i]`
|
|
3060
|
+
/// read (`GetIndex`) or `arr[i] = v` write (`SetIndex`) with an arbitrary COMPUTED index (`a[i*N+k]`).
|
|
3061
|
+
///
|
|
3062
|
+
/// Returns `Some(arrays)` — possibly with an empty `slots` (a pure-numeric region: caller keeps the
|
|
3063
|
+
/// unchanged 2-pointer ABI) — when the region's array usage is fully lowerable; `None` when it holds a
|
|
3064
|
+
/// `GetIndex`/`SetIndex` the emitter can't handle: a non-slot / computed array BASE (`a[i][j]`), an
|
|
3065
|
+
/// array slot also used as a scalar or reassigned (`arr = …`), or an operand stack the linear abstract
|
|
3066
|
+
/// interp can't balance. `None` leaves the region non-OSR-compilable (its index ops still bail).
|
|
3067
|
+
///
|
|
3068
|
+
/// Imprecision is always SAFE — never a miscompile: a slot wrongly called an array fails the runtime
|
|
3069
|
+
/// `Value::Array` marshalling guard in `run_osr` (→ interpret); a slot wrongly NOT called an array
|
|
3070
|
+
/// leaves a `GetIndex`/`SetIndex` the region emitter rejects (→ region not compiled). Only correctly-
|
|
3071
|
+
/// classified, all-`Number`-element arrays are ever compiled.
|
|
3072
|
+
#[cfg(not(target_arch = "wasm32"))]
|
|
3073
|
+
fn classify_osr_arrays(chunk: &Chunk, header_ip: usize, region_end: usize) -> Option<OsrArrays> {
|
|
3074
|
+
let code = &chunk.code;
|
|
3075
|
+
// An abstract stack entry: `Ref(slot)` = a bare `LoadLocal(slot)` result untouched since; any other
|
|
3076
|
+
// producer (const, arithmetic, a loaded element, …) is `Scalar`. An array base must be a `Ref`.
|
|
3077
|
+
#[derive(Clone, Copy, PartialEq)]
|
|
3078
|
+
enum Av {
|
|
3079
|
+
Scalar,
|
|
3080
|
+
Ref(u16),
|
|
3081
|
+
}
|
|
3082
|
+
let mut astack: Vec<Av> = Vec::new();
|
|
3083
|
+
let mut array_slots: std::collections::BTreeSet<u16> = Default::default();
|
|
3084
|
+
let mut writable: std::collections::BTreeSet<u16> = Default::default();
|
|
3085
|
+
// Slots ever consumed as a scalar (arithmetic operand, index, store source, cond, …) or reassigned
|
|
3086
|
+
// — an array slot must appear in NEITHER, else its live-in isn't a stable, index-only array handle.
|
|
3087
|
+
let mut scalar_use: std::collections::BTreeSet<u16> = Default::default();
|
|
3088
|
+
let mut stored: std::collections::BTreeSet<u16> = Default::default();
|
|
3089
|
+
// Consume one operand; a bare `Ref(slot)` reaching a scalar position taints that slot.
|
|
3090
|
+
let taint = |astack: &mut Vec<Av>, scalar_use: &mut std::collections::BTreeSet<u16>| -> Option<()> {
|
|
3091
|
+
match astack.pop()? {
|
|
3092
|
+
Av::Ref(s) => {
|
|
3093
|
+
scalar_use.insert(s);
|
|
3094
|
+
}
|
|
3095
|
+
Av::Scalar => {}
|
|
3096
|
+
}
|
|
3097
|
+
Some(())
|
|
3098
|
+
};
|
|
3099
|
+
let mut ip = header_ip;
|
|
3100
|
+
while ip < region_end {
|
|
3101
|
+
let op = Opcode::from_u8(*code.get(ip)?)?;
|
|
3102
|
+
let size = op_size(op)?; // whitelist-sized ops only; anything else ⇒ region not compilable
|
|
3103
|
+
match op {
|
|
3104
|
+
Opcode::LoadLocal => astack.push(Av::Ref(peek_u16(code, ip + 1)?)),
|
|
3105
|
+
Opcode::LoadConst => astack.push(Av::Scalar),
|
|
3106
|
+
Opcode::StoreLocal => {
|
|
3107
|
+
stored.insert(peek_u16(code, ip + 1)?);
|
|
3108
|
+
taint(&mut astack, &mut scalar_use)?; // the stored value
|
|
3109
|
+
}
|
|
3110
|
+
Opcode::Pop => {
|
|
3111
|
+
astack.pop()?; // discarded — no taint (a bare pop doesn't "use" the slot)
|
|
3112
|
+
}
|
|
3113
|
+
Opcode::Dup => {
|
|
3114
|
+
let top = *astack.last()?;
|
|
3115
|
+
astack.push(top);
|
|
3116
|
+
}
|
|
3117
|
+
Opcode::Nop
|
|
3118
|
+
| Opcode::EnterBlock
|
|
3119
|
+
| Opcode::ExitBlock
|
|
3120
|
+
| Opcode::LoopVarsEnd
|
|
3121
|
+
| Opcode::LoopVarsBegin => {}
|
|
3122
|
+
Opcode::UnaryOp | Opcode::MathUnary => {
|
|
3123
|
+
taint(&mut astack, &mut scalar_use)?;
|
|
3124
|
+
astack.push(Av::Scalar);
|
|
3125
|
+
}
|
|
3126
|
+
Opcode::BinOp | Opcode::MathBinary => {
|
|
3127
|
+
taint(&mut astack, &mut scalar_use)?;
|
|
3128
|
+
taint(&mut astack, &mut scalar_use)?;
|
|
3129
|
+
astack.push(Av::Scalar);
|
|
3130
|
+
}
|
|
3131
|
+
Opcode::GetIndex => {
|
|
3132
|
+
taint(&mut astack, &mut scalar_use)?; // index (a scalar use of whatever produced it)
|
|
3133
|
+
match astack.pop()? {
|
|
3134
|
+
Av::Ref(s) => {
|
|
3135
|
+
array_slots.insert(s);
|
|
3136
|
+
}
|
|
3137
|
+
Av::Scalar => return None, // computed / nested array base (`a[i][j]`) — bail
|
|
3138
|
+
}
|
|
3139
|
+
astack.push(Av::Scalar); // the loaded element
|
|
3140
|
+
}
|
|
3141
|
+
Opcode::SetIndex => {
|
|
3142
|
+
taint(&mut astack, &mut scalar_use)?; // dup_val
|
|
3143
|
+
taint(&mut astack, &mut scalar_use)?; // val
|
|
3144
|
+
taint(&mut astack, &mut scalar_use)?; // idx
|
|
3145
|
+
match astack.pop()? {
|
|
3146
|
+
Av::Ref(s) => {
|
|
3147
|
+
array_slots.insert(s);
|
|
3148
|
+
writable.insert(s);
|
|
3149
|
+
}
|
|
3150
|
+
Av::Scalar => return None,
|
|
3151
|
+
}
|
|
3152
|
+
astack.push(Av::Scalar); // SetIndex leaves the assigned value
|
|
3153
|
+
}
|
|
3154
|
+
// A terminator ends the basic block: the region invariant makes the operand stack empty
|
|
3155
|
+
// here (JumpIfFalse pops its cond first). Clearing keeps the abstract stack self-contained
|
|
3156
|
+
// per block — array access never spans a branch, so this loses nothing we can lower.
|
|
3157
|
+
Opcode::Jump | Opcode::JumpBack => astack.clear(),
|
|
3158
|
+
Opcode::JumpIfFalse => {
|
|
3159
|
+
taint(&mut astack, &mut scalar_use)?; // cond
|
|
3160
|
+
astack.clear();
|
|
3161
|
+
}
|
|
3162
|
+
_ => return None, // an opcode outside the region vocabulary ⇒ not compilable
|
|
3163
|
+
}
|
|
3164
|
+
ip += size;
|
|
3165
|
+
}
|
|
3166
|
+
// An array slot used anywhere as a scalar, or reassigned, isn't a stable index-only handle ⇒ the
|
|
3167
|
+
// whole region is not array-compilable (its index ops keep bailing).
|
|
3168
|
+
if array_slots
|
|
3169
|
+
.iter()
|
|
3170
|
+
.any(|s| scalar_use.contains(s) || stored.contains(s))
|
|
3171
|
+
{
|
|
3172
|
+
return None;
|
|
3173
|
+
}
|
|
3174
|
+
// Cap the handle count (matmul uses 3); keep it small so the marshalling stays cheap.
|
|
3175
|
+
if array_slots.len() > 8 {
|
|
3176
|
+
return None;
|
|
3177
|
+
}
|
|
3178
|
+
Some(OsrArrays {
|
|
3179
|
+
slots: array_slots.into_iter().collect(),
|
|
3180
|
+
writable,
|
|
3181
|
+
})
|
|
3182
|
+
}
|
|
3183
|
+
|
|
2126
3184
|
/// on any misuse of a JV ref (it reaching a numeric op, a return, a call arg, a block boundary, …),
|
|
2127
3185
|
/// so a mis-shaped use never miscompiles — it just falls back to the interpreter.
|
|
2128
3186
|
fn classify_jv_slots(chunk: &Chunk) -> Option<std::collections::HashSet<usize>> {
|
|
@@ -2181,17 +3239,53 @@ fn peek_u16(code: &[u8], off: usize) -> Option<u16> {
|
|
|
2181
3239
|
/// `LoadConst(Number)` (→ an f64 const) — for the array read/write peepholes. `None` for any other
|
|
2182
3240
|
/// opcode / a non-numeric const, which makes the caller bail. #187
|
|
2183
3241
|
#[cfg(not(target_arch = "wasm32"))]
|
|
3242
|
+
/// #168: the exact i64 for a `StoreLocal` into an int slot whose value arrived as a plain F64
|
|
3243
|
+
/// push — the classifier-approved case is a linearly-preceding `LoadConst` of an integral
|
|
3244
|
+
/// Number in [-2^31, 2^32) (excluding `-0`). `store_ip` points AT the StoreLocal opcode; the
|
|
3245
|
+
/// cfg builder's empty-stack-at-block-boundary discipline guarantees the linear predecessor is
|
|
3246
|
+
/// the dynamic value producer. Any other shape → `None` → the caller bails the whole compile.
|
|
3247
|
+
fn int_const_i64(
|
|
3248
|
+
bcx: &mut FunctionBuilder,
|
|
3249
|
+
chunk: &Chunk,
|
|
3250
|
+
code: &[u8],
|
|
3251
|
+
store_ip: usize,
|
|
3252
|
+
) -> Option<ClifValue> {
|
|
3253
|
+
let lc = store_ip.checked_sub(3)?;
|
|
3254
|
+
if Opcode::from_u8(*code.get(lc)?)? != Opcode::LoadConst {
|
|
3255
|
+
return None;
|
|
3256
|
+
}
|
|
3257
|
+
match chunk.constants.get(peek_u16(code, lc + 1)? as usize)? {
|
|
3258
|
+
Constant::Number(n)
|
|
3259
|
+
if n.fract() == 0.0
|
|
3260
|
+
&& *n >= -(2f64.powi(31))
|
|
3261
|
+
&& *n < 2f64.powi(32)
|
|
3262
|
+
&& n.to_bits() != (-0f64).to_bits() =>
|
|
3263
|
+
{
|
|
3264
|
+
Some(bcx.ins().iconst(types::I64, *n as i64))
|
|
3265
|
+
}
|
|
3266
|
+
_ => None,
|
|
3267
|
+
}
|
|
3268
|
+
}
|
|
3269
|
+
|
|
2184
3270
|
fn read_simple_operand(
|
|
2185
3271
|
bcx: &mut FunctionBuilder,
|
|
2186
3272
|
code: &[u8],
|
|
2187
3273
|
at: usize,
|
|
2188
3274
|
chunk: &Chunk,
|
|
2189
3275
|
vars: &[Variable],
|
|
3276
|
+
int_slots: &std::collections::HashSet<usize>,
|
|
2190
3277
|
) -> Option<ClifValue> {
|
|
2191
3278
|
match Opcode::from_u8(*code.get(at)?)? {
|
|
2192
3279
|
Opcode::LoadLocal => {
|
|
2193
3280
|
let s = peek_u16(code, at + 1)? as usize;
|
|
2194
|
-
|
|
3281
|
+
let v = bcx.use_var(*vars.get(s)?);
|
|
3282
|
+
// #168: an int slot's Variable is i64 (exact number) — materialize to the f64 this
|
|
3283
|
+
// fast path assumes. Without this, the i64 value would flow into f64 instructions
|
|
3284
|
+
// and trip the cranelift verifier (a crash, not a bail).
|
|
3285
|
+
if int_slots.contains(&s) {
|
|
3286
|
+
return Some(bcx.ins().fcvt_from_sint(types::F64, v));
|
|
3287
|
+
}
|
|
3288
|
+
Some(v)
|
|
2195
3289
|
}
|
|
2196
3290
|
Opcode::LoadConst => {
|
|
2197
3291
|
let ci = peek_u16(code, at + 1)? as usize;
|
|
@@ -2233,6 +3327,8 @@ fn build_body_cfg(
|
|
|
2233
3327
|
recur_guard: bool,
|
|
2234
3328
|
// #186: the imported `tish_math_call` host fn, for lowering `Math.<fn>` (`MathUnary`) intrinsics.
|
|
2235
3329
|
math_fref: cranelift::codegen::ir::FuncRef,
|
|
3330
|
+
// #203: the imported `tish_math_binary_call` host fn, for lowering `MathBinary` (max/min/pow/atan2).
|
|
3331
|
+
math_binary_fref: cranelift::codegen::ir::FuncRef,
|
|
2236
3332
|
// #189: `Some` when the function has local `f64` arrays — carries the `tish_jv_*` `FuncRef`s + the
|
|
2237
3333
|
// JV slot set. Those slots become `i64` handle Variables and their array ops lower to `tish_jv_*`
|
|
2238
3334
|
// calls; an out-of-bounds index sets the per-thread deopt flag the wrapper re-interprets on.
|
|
@@ -2248,6 +3344,7 @@ fn build_body_cfg(
|
|
|
2248
3344
|
}
|
|
2249
3345
|
// #187: slots that hold a boolean (represented as `f64` 0/1). A `LoadLocal` of one carries
|
|
2250
3346
|
// `is_bool` so a diverging `bool === number` compare or a `return bool` bails to the interpreter.
|
|
3347
|
+
let int_slots = classify_int_slots(chunk, arity);
|
|
2251
3348
|
let bool_slots = if jit_bool_slots_enabled() {
|
|
2252
3349
|
classify_bool_slots(chunk)
|
|
2253
3350
|
} else {
|
|
@@ -2263,9 +3360,22 @@ fn build_body_cfg(
|
|
|
2263
3360
|
leaders.insert(0);
|
|
2264
3361
|
let mut has_loop = false;
|
|
2265
3362
|
let mut has_self_call = false;
|
|
3363
|
+
// #203: `JumpIfFalse` ip → (then_end, else_start, merge) for each clean ternary lowered inline.
|
|
3364
|
+
let mut ternaries: HashMap<usize, (usize, usize, usize)> = HashMap::new();
|
|
2266
3365
|
let mut ip = 0;
|
|
2267
3366
|
while ip < code.len() {
|
|
2268
3367
|
let op = Opcode::from_u8(code[ip])?;
|
|
3368
|
+
// #203: a clean ternary `cond ? A : B` is lowered branch-free (an inline `select`), so record
|
|
3369
|
+
// it and SKIP the whole span — none of its internal targets become block leaders (no orphan
|
|
3370
|
+
// blocks). Plain-numeric functions only (no JV arrays / array-mode); `TISH_JIT_TERNARY=0` off.
|
|
3371
|
+
if op == Opcode::JumpIfFalse && jit_ternary_enabled() && jv.is_none() && array_mask == 0 {
|
|
3372
|
+
if let Some(span) = ternary_span(code, ip) {
|
|
3373
|
+
let (_then_end, _else_start, merge) = span;
|
|
3374
|
+
ternaries.insert(ip, span);
|
|
3375
|
+
ip = merge;
|
|
3376
|
+
continue;
|
|
3377
|
+
}
|
|
3378
|
+
}
|
|
2269
3379
|
// For a JV function the array opcodes (NewArray/GetMember/GetIndex/SetIndex/Call) are valid
|
|
2270
3380
|
// and must be sized via the full instruction table; the translate loop then validates each is
|
|
2271
3381
|
// a real JV op (else it bails). Non-JV functions keep the strict `op_size` whitelist.
|
|
@@ -2358,11 +3468,12 @@ fn build_body_cfg(
|
|
|
2358
3468
|
};
|
|
2359
3469
|
let body_start = *blocks.get(&0)?; // `entry` normally; `body_block` when guarded
|
|
2360
3470
|
|
|
2361
|
-
// 2. A Variable per slot, all defined at entry so every path defines them. A JV slot (#189)
|
|
2362
|
-
// an arena HANDLE (a `u64`, 0 = null) so its Variable is `i64`, not `f64
|
|
3471
|
+
// 2. A Variable per slot, all defined at entry so every path defines them. A JV slot (#189)
|
|
3472
|
+
// holds an arena HANDLE (a `u64`, 0 = null) so its Variable is `i64`, not `f64`; an int
|
|
3473
|
+
// slot (#168) holds an exact integral JS number as `i64` ([`Repr::I64Num`]).
|
|
2363
3474
|
let vars: Vec<Variable> = (0..num_slots)
|
|
2364
3475
|
.map(|i| {
|
|
2365
|
-
let ty = if jv.is_some_and(|j| j.slots.contains(&i)) {
|
|
3476
|
+
let ty = if jv.is_some_and(|j| j.slots.contains(&i)) || int_slots.contains(&i) {
|
|
2366
3477
|
types::I64
|
|
2367
3478
|
} else {
|
|
2368
3479
|
types::F64
|
|
@@ -2398,6 +3509,8 @@ fn build_body_cfg(
|
|
|
2398
3509
|
.load(types::F64, MemFlags::new(), numeric_ptr, numeric_i * 8);
|
|
2399
3510
|
numeric_i += 1;
|
|
2400
3511
|
v
|
|
3512
|
+
} else if int_slots.contains(&slot) {
|
|
3513
|
+
bcx.ins().iconst(types::I64, 0)
|
|
2401
3514
|
} else {
|
|
2402
3515
|
bcx.ins().f64const(0.0)
|
|
2403
3516
|
};
|
|
@@ -2415,6 +3528,8 @@ fn build_body_cfg(
|
|
|
2415
3528
|
} else {
|
|
2416
3529
|
let init = if i < arity {
|
|
2417
3530
|
params[i]
|
|
3531
|
+
} else if int_slots.contains(&i) {
|
|
3532
|
+
bcx.ins().iconst(types::I64, 0)
|
|
2418
3533
|
} else {
|
|
2419
3534
|
bcx.ins().f64const(0.0)
|
|
2420
3535
|
};
|
|
@@ -2424,7 +3539,7 @@ fn build_body_cfg(
|
|
|
2424
3539
|
}
|
|
2425
3540
|
|
|
2426
3541
|
// 3. Translate. The operand stack is empty at every block boundary (statement-level control flow).
|
|
2427
|
-
let mut stack: Vec<
|
|
3542
|
+
let mut stack: Vec<JV> = Vec::new();
|
|
2428
3543
|
// #189: JV array handles "in flight" (pushed by `LoadLocal`/`NewArray` of a JV slot, consumed by
|
|
2429
3544
|
// the very next `GetIndex`/`SetIndex`/`GetMember`), kept off the f64 `stack`; and a pending
|
|
2430
3545
|
// `arr.push` awaiting its arg + `Call`. Both must be empty at every block boundary.
|
|
@@ -2509,13 +3624,16 @@ fn build_body_cfg(
|
|
|
2509
3624
|
ip += 3;
|
|
2510
3625
|
continue;
|
|
2511
3626
|
}
|
|
2512
|
-
let idx_f64 =
|
|
3627
|
+
let idx_f64 =
|
|
3628
|
+
read_simple_operand(&mut bcx, code, ip + 3, chunk, &vars, &int_slots)?;
|
|
2513
3629
|
// i = idx as usize (saturating: NaN→0, neg→0 — matches the VM's `n as usize`).
|
|
2514
3630
|
let i = bcx.ins().fcvt_to_uint_sat(types::I64, idx_f64);
|
|
2515
3631
|
// Read `val` (write only) BEFORE the bounds-check split so its LoadLocal reads the
|
|
2516
3632
|
// slot's current SSA value in `cur`, not the fresh `cont` block.
|
|
2517
3633
|
let store_val = if is_write {
|
|
2518
|
-
Some(read_simple_operand(
|
|
3634
|
+
Some(read_simple_operand(
|
|
3635
|
+
&mut bcx, code, ip + 6, chunk, &vars, &int_slots,
|
|
3636
|
+
)?)
|
|
2519
3637
|
} else {
|
|
2520
3638
|
None
|
|
2521
3639
|
};
|
|
@@ -2529,19 +3647,28 @@ fn build_body_cfg(
|
|
|
2529
3647
|
let addr = bcx.ins().iadd(aptr, off);
|
|
2530
3648
|
if let Some(val) = store_val {
|
|
2531
3649
|
bcx.ins().store(MemFlags::new(), val, addr, 0);
|
|
2532
|
-
stack.push((val
|
|
3650
|
+
stack.push(JV::f64(val)); // assignment yields the value (a Pop usually discards it)
|
|
2533
3651
|
ip += 11; // LoadLocal(arr) + idx + val + Dup + SetIndex
|
|
2534
3652
|
} else {
|
|
2535
3653
|
let val = bcx.ins().load(types::F64, MemFlags::new(), addr, 0);
|
|
2536
|
-
stack.push((val
|
|
3654
|
+
stack.push(JV::f64(val));
|
|
2537
3655
|
ip += 7; // LoadLocal(arr) + idx + GetIndex
|
|
2538
3656
|
}
|
|
2539
3657
|
continue;
|
|
2540
3658
|
}
|
|
2541
3659
|
let v = *vars.get(slot)?;
|
|
2542
|
-
// #187: a bool slot's `f64` 0/1 value carries
|
|
2543
|
-
// guards fire;
|
|
2544
|
-
|
|
3660
|
+
// #187: a bool slot's `f64` 0/1 value carries the Bool repr so downstream
|
|
3661
|
+
// equality/return guards fire; an int slot (#168) pushes its exact-i64 repr
|
|
3662
|
+
// (identity into further int ops, one exact convert at an f64 boundary); a
|
|
3663
|
+
// plain numeric slot pushes F64.
|
|
3664
|
+
let lv = bcx.use_var(v);
|
|
3665
|
+
stack.push(if bool_slots.contains(&slot) {
|
|
3666
|
+
JV::boolean(lv)
|
|
3667
|
+
} else if int_slots.contains(&slot) {
|
|
3668
|
+
JV::i64num(lv)
|
|
3669
|
+
} else {
|
|
3670
|
+
JV::f64(lv)
|
|
3671
|
+
});
|
|
2545
3672
|
ip += 3;
|
|
2546
3673
|
}
|
|
2547
3674
|
Opcode::StoreLocal => {
|
|
@@ -2554,15 +3681,43 @@ fn build_body_cfg(
|
|
|
2554
3681
|
ip += 3;
|
|
2555
3682
|
continue;
|
|
2556
3683
|
}
|
|
2557
|
-
let
|
|
3684
|
+
let jval = stack.pop()?;
|
|
2558
3685
|
// #187: a boolean value may be stored only into a slot the pre-pass tagged as a bool
|
|
2559
3686
|
// slot (represented as `f64` 0/1). Any other bool store → bail (keeps unknown shapes
|
|
2560
3687
|
// on the interpreter). A numeric store into a bool-tagged slot is fine — the slot is
|
|
2561
|
-
// still an `f64`; its `LoadLocal`s just carry
|
|
2562
|
-
if is_bool && !bool_slots.contains(&slot) {
|
|
3688
|
+
// still an `f64`; its `LoadLocal`s just carry the Bool repr (conservatively).
|
|
3689
|
+
if jval.is_bool() && !bool_slots.contains(&slot) {
|
|
2563
3690
|
return None;
|
|
2564
3691
|
}
|
|
2565
3692
|
let v = *vars.get(slot)?;
|
|
3693
|
+
if int_slots.contains(&slot) {
|
|
3694
|
+
// #168: an int slot stores the EXACT number as i64 — sign-extend ToInt32
|
|
3695
|
+
// results, zero-extend ToUint32 results, integral constants load exactly.
|
|
3696
|
+
// Anything else reaching here means the optimistic pre-pass mis-tagged the
|
|
3697
|
+
// slot (e.g. a merge point whose linear predecessor differed) → bail the
|
|
3698
|
+
// whole compile; the VM keeps semantics.
|
|
3699
|
+
let iv = match jval.repr {
|
|
3700
|
+
Repr::I32 => bcx.ins().sextend(types::I64, jval.v),
|
|
3701
|
+
Repr::U32 => bcx.ins().uextend(types::I64, jval.v),
|
|
3702
|
+
Repr::I64Num => jval.v,
|
|
3703
|
+
Repr::F64 => {
|
|
3704
|
+
// The classifier only tags int-op/const-preceded stores, but a
|
|
3705
|
+
// constant reaches here as an F64 push — re-derive its exact i64
|
|
3706
|
+
// when it is one of the classifier-approved integral constants.
|
|
3707
|
+
match int_const_i64(&mut bcx, chunk, code, ip) {
|
|
3708
|
+
Some(iv) => iv,
|
|
3709
|
+
None => return None,
|
|
3710
|
+
}
|
|
3711
|
+
}
|
|
3712
|
+
Repr::Bool => return None,
|
|
3713
|
+
};
|
|
3714
|
+
bcx.def_var(v, iv);
|
|
3715
|
+
ip += 3;
|
|
3716
|
+
continue;
|
|
3717
|
+
}
|
|
3718
|
+
// Slots are f64 Variables — materialize an int repr once, at the store (an
|
|
3719
|
+
// `h = ((h<<13)|(h>>>19))>>>0` chain pays exactly ONE convert here, not per op).
|
|
3720
|
+
let val = jv_f64(&mut bcx, jval);
|
|
2566
3721
|
bcx.def_var(v, val);
|
|
2567
3722
|
ip += 3;
|
|
2568
3723
|
}
|
|
@@ -2580,10 +3735,11 @@ fn build_body_cfg(
|
|
|
2580
3735
|
Opcode::Nop | Opcode::EnterBlock | Opcode::ExitBlock | Opcode::LoopVarsEnd => ip += 1,
|
|
2581
3736
|
Opcode::LoopVarsBegin => ip += 3,
|
|
2582
3737
|
Opcode::Return => {
|
|
2583
|
-
let
|
|
2584
|
-
if is_bool {
|
|
3738
|
+
let ret = stack.pop()?;
|
|
3739
|
+
if ret.is_bool() {
|
|
2585
3740
|
return None;
|
|
2586
3741
|
}
|
|
3742
|
+
let v = jv_f64(&mut bcx, ret);
|
|
2587
3743
|
// #189: free every JV array (return its arena slot to the free list) before leaving
|
|
2588
3744
|
// the frame — handle 0 (a slot not yet allocated on this path) is a no-op. This is why
|
|
2589
3745
|
// JV arrays must never escape.
|
|
@@ -2617,12 +3773,73 @@ fn build_body_cfg(
|
|
|
2617
3773
|
terminated = true;
|
|
2618
3774
|
ip += 3;
|
|
2619
3775
|
}
|
|
3776
|
+
// #203: a clean ternary detected in the leader scan → branch-free `select`, INLINE (no
|
|
3777
|
+
// blocks). The stack BELOW the cond stays untouched (both arms push/pop their one result
|
|
3778
|
+
// on top of it), so no operand crosses a block boundary — this is the only reason
|
|
3779
|
+
// build_body_cfg's empty-stack-at-boundary invariant is not violated.
|
|
3780
|
+
Opcode::JumpIfFalse if ternaries.contains_key(&ip) => {
|
|
3781
|
+
let (then_end, else_start, merge) = *ternaries.get(&ip).unwrap();
|
|
3782
|
+
let cond = stack.pop()?;
|
|
3783
|
+
let cond = jv_f64(&mut bcx, cond);
|
|
3784
|
+
let base = stack.len();
|
|
3785
|
+
// THEN arm: [ip+3, then_end) → exactly one value.
|
|
3786
|
+
if !emit_ternary_arm(
|
|
3787
|
+
&mut bcx,
|
|
3788
|
+
chunk,
|
|
3789
|
+
code,
|
|
3790
|
+
ip + 3,
|
|
3791
|
+
then_end,
|
|
3792
|
+
&mut stack,
|
|
3793
|
+
&vars,
|
|
3794
|
+
&int_slots,
|
|
3795
|
+
&bool_slots,
|
|
3796
|
+
math_fref,
|
|
3797
|
+
math_binary_fref,
|
|
3798
|
+
) || stack.len() != base + 1
|
|
3799
|
+
{
|
|
3800
|
+
return None;
|
|
3801
|
+
}
|
|
3802
|
+
let then_jv = stack.pop()?;
|
|
3803
|
+
let then_v = jv_f64(&mut bcx, then_jv);
|
|
3804
|
+
// ELSE arm: [else_start, merge) → exactly one value.
|
|
3805
|
+
if !emit_ternary_arm(
|
|
3806
|
+
&mut bcx,
|
|
3807
|
+
chunk,
|
|
3808
|
+
code,
|
|
3809
|
+
else_start,
|
|
3810
|
+
merge,
|
|
3811
|
+
&mut stack,
|
|
3812
|
+
&vars,
|
|
3813
|
+
&int_slots,
|
|
3814
|
+
&bool_slots,
|
|
3815
|
+
math_fref,
|
|
3816
|
+
math_binary_fref,
|
|
3817
|
+
) || stack.len() != base + 1
|
|
3818
|
+
{
|
|
3819
|
+
return None;
|
|
3820
|
+
}
|
|
3821
|
+
let else_jv = stack.pop()?;
|
|
3822
|
+
let else_v = jv_f64(&mut bcx, else_jv);
|
|
3823
|
+
// Both arms must agree Bool-vs-Number (one `result_bool` shape, like build_body).
|
|
3824
|
+
if then_jv.is_bool() != else_jv.is_bool() {
|
|
3825
|
+
return None;
|
|
3826
|
+
}
|
|
3827
|
+
let falsy = falsy_flag(&mut bcx, cond);
|
|
3828
|
+
let sel = bcx.ins().select(falsy, else_v, then_v);
|
|
3829
|
+
stack.push(if then_jv.is_bool() {
|
|
3830
|
+
JV::boolean(sel)
|
|
3831
|
+
} else {
|
|
3832
|
+
JV::f64(sel)
|
|
3833
|
+
});
|
|
3834
|
+
ip = merge;
|
|
3835
|
+
}
|
|
2620
3836
|
Opcode::JumpIfFalse => {
|
|
2621
3837
|
let off = peek_u16(code, ip + 1)? as i16 as isize;
|
|
2622
|
-
let
|
|
3838
|
+
let cond = stack.pop()?;
|
|
2623
3839
|
if !stack.is_empty() {
|
|
2624
3840
|
return None; // non-empty stack ⇒ ternary shape ⇒ leave to build_body / VM
|
|
2625
3841
|
}
|
|
3842
|
+
let cond = jv_f64(&mut bcx, cond);
|
|
2626
3843
|
let falsy = falsy_flag(&mut bcx, cond);
|
|
2627
3844
|
let target = *blocks.get(&(((ip + 3) as isize + off).max(0) as usize))?;
|
|
2628
3845
|
let fallthrough = *blocks.get(&(ip + 3))?;
|
|
@@ -2651,10 +3868,12 @@ fn build_body_cfg(
|
|
|
2651
3868
|
3, // 2^3 = 8-byte alignment for f64
|
|
2652
3869
|
));
|
|
2653
3870
|
let arg_start = stack.len() - num_numeric;
|
|
2654
|
-
|
|
2655
|
-
|
|
3871
|
+
let args: Vec<JV> = stack.drain(arg_start..).collect();
|
|
3872
|
+
for (j, jv) in args.into_iter().enumerate() {
|
|
3873
|
+
if jv.is_bool() {
|
|
2656
3874
|
return None; // a bool numeric arg doesn't match the f64 ABI
|
|
2657
3875
|
}
|
|
3876
|
+
let v = jv_f64(&mut bcx, jv);
|
|
2658
3877
|
bcx.ins().stack_store(v, slot, (j * 8) as i32);
|
|
2659
3878
|
}
|
|
2660
3879
|
let num_ptr = bcx.ins().stack_addr(types::I64, slot, 0);
|
|
@@ -2667,7 +3886,8 @@ fn build_body_cfg(
|
|
|
2667
3886
|
call_args.push(gp);
|
|
2668
3887
|
}
|
|
2669
3888
|
let call = bcx.ins().call(sref, &call_args);
|
|
2670
|
-
|
|
3889
|
+
let res = bcx.inst_results(call)[0];
|
|
3890
|
+
stack.push(JV::f64(res));
|
|
2671
3891
|
ip += 3;
|
|
2672
3892
|
}
|
|
2673
3893
|
Opcode::SelfCall => {
|
|
@@ -2679,12 +3899,13 @@ fn build_body_cfg(
|
|
|
2679
3899
|
return None;
|
|
2680
3900
|
}
|
|
2681
3901
|
let arg_start = stack.len() - arity;
|
|
3902
|
+
let args: Vec<JV> = stack.drain(arg_start..).collect();
|
|
2682
3903
|
let mut call_args = Vec::with_capacity(arity);
|
|
2683
|
-
for
|
|
2684
|
-
if is_bool {
|
|
3904
|
+
for jv in args {
|
|
3905
|
+
if jv.is_bool() {
|
|
2685
3906
|
return None; // boolean args don't match the f64 ABI
|
|
2686
3907
|
}
|
|
2687
|
-
call_args.push(
|
|
3908
|
+
call_args.push(jv_f64(&mut bcx, jv));
|
|
2688
3909
|
}
|
|
2689
3910
|
// #381: thread the RecurGuard pointer through the recursive call so every level
|
|
2690
3911
|
// re-checks the stack at its entry. Present iff this function was compiled guarded.
|
|
@@ -2693,16 +3914,30 @@ fn build_body_cfg(
|
|
|
2693
3914
|
}
|
|
2694
3915
|
let call = bcx.ins().call(sref, &call_args);
|
|
2695
3916
|
let result = bcx.inst_results(call)[0];
|
|
2696
|
-
stack.push((result
|
|
3917
|
+
stack.push(JV::f64(result));
|
|
2697
3918
|
ip += 3;
|
|
2698
3919
|
}
|
|
2699
3920
|
Opcode::MathUnary => {
|
|
2700
3921
|
// #186 — `Math.<fn>(x)`: pop the arg, emit native op / host call, push the result.
|
|
2701
3922
|
let id = peek_u16(code, ip + 1)?;
|
|
2702
3923
|
let mfn = MathUnaryFn::from_u16(id)?;
|
|
2703
|
-
let
|
|
3924
|
+
let x = stack.pop()?;
|
|
3925
|
+
let x = jv_f64(&mut bcx, x);
|
|
2704
3926
|
let r = emit_math_unary(&mut bcx, math_fref, mfn, x);
|
|
2705
|
-
stack.push((r
|
|
3927
|
+
stack.push(JV::f64(r));
|
|
3928
|
+
ip += 3;
|
|
3929
|
+
}
|
|
3930
|
+
Opcode::MathBinary => {
|
|
3931
|
+
// #203 — `Math.<fn>(a, b)`: pop b, pop a, host-call, push.
|
|
3932
|
+
let id = peek_u16(code, ip + 1)?;
|
|
3933
|
+
let b = stack.pop()?;
|
|
3934
|
+
let a = stack.pop()?;
|
|
3935
|
+
let b = jv_f64(&mut bcx, b);
|
|
3936
|
+
let a = jv_f64(&mut bcx, a);
|
|
3937
|
+
let idc = bcx.ins().iconst(types::I32, id as i64);
|
|
3938
|
+
let call = bcx.ins().call(math_binary_fref, &[idc, a, b]);
|
|
3939
|
+
let r = bcx.inst_results(call)[0];
|
|
3940
|
+
stack.push(JV::f64(r));
|
|
2706
3941
|
ip += 3;
|
|
2707
3942
|
}
|
|
2708
3943
|
// #189 — local-array ops, only for JV functions (`jv` = `Some`). Each consumes the array
|
|
@@ -2719,25 +3954,29 @@ fn build_body_cfg(
|
|
|
2719
3954
|
}
|
|
2720
3955
|
Opcode::GetIndex if !jv_pending.is_empty() => {
|
|
2721
3956
|
let jvc = jv?;
|
|
2722
|
-
let idx = stack.pop()
|
|
3957
|
+
let idx = stack.pop()?;
|
|
3958
|
+
let idx = jv_f64(&mut bcx, idx);
|
|
2723
3959
|
let handle = jv_pending.pop()?;
|
|
2724
3960
|
// `idx as usize` (saturating: NaN/neg → 0), matching the VM's index coercion. OOB sets
|
|
2725
3961
|
// the per-thread deopt flag inside `tish_jv_get` and returns NaN.
|
|
2726
3962
|
let i = bcx.ins().fcvt_to_uint_sat(types::I64, idx);
|
|
2727
3963
|
let call = bcx.ins().call(jvc.get, &[handle, i]);
|
|
2728
|
-
|
|
3964
|
+
let res = bcx.inst_results(call)[0];
|
|
3965
|
+
stack.push(JV::f64(res));
|
|
2729
3966
|
ip += 1;
|
|
2730
3967
|
}
|
|
2731
3968
|
Opcode::SetIndex if !jv_pending.is_empty() => {
|
|
2732
3969
|
let jvc = jv?;
|
|
2733
3970
|
// Stack: [ (array→jv_pending), idx, val, dup_val ]. `Dup` left `dup_val` == `val`.
|
|
2734
|
-
let
|
|
2735
|
-
let _val = stack.pop()
|
|
2736
|
-
let idx = stack.pop()
|
|
3971
|
+
let dup_jv = stack.pop()?;
|
|
3972
|
+
let _val = stack.pop()?;
|
|
3973
|
+
let idx = stack.pop()?;
|
|
3974
|
+
let dup_val = jv_f64(&mut bcx, dup_jv);
|
|
3975
|
+
let idx = jv_f64(&mut bcx, idx);
|
|
2737
3976
|
let handle = jv_pending.pop()?;
|
|
2738
3977
|
let i = bcx.ins().fcvt_to_uint_sat(types::I64, idx);
|
|
2739
3978
|
bcx.ins().call(jvc.set, &[handle, i, dup_val]); // OOB → deopt flag inside tish_jv_set
|
|
2740
|
-
stack.push((dup_val
|
|
3979
|
+
stack.push(JV::f64(dup_val)); // assignment yields the value
|
|
2741
3980
|
ip += 1;
|
|
2742
3981
|
}
|
|
2743
3982
|
Opcode::GetMember if !jv_pending.is_empty() => {
|
|
@@ -2748,7 +3987,8 @@ fn build_body_cfg(
|
|
|
2748
3987
|
"length" => {
|
|
2749
3988
|
let call = bcx.ins().call(jvc.len, &[ptr]);
|
|
2750
3989
|
let len_i = bcx.inst_results(call)[0];
|
|
2751
|
-
|
|
3990
|
+
let len_f = bcx.ins().fcvt_from_uint(types::F64, len_i);
|
|
3991
|
+
stack.push(JV::f64(len_f));
|
|
2752
3992
|
}
|
|
2753
3993
|
"push" => pending_push = Some(ptr),
|
|
2754
3994
|
_ => return None, // any other member of a JV array → bail
|
|
@@ -2761,12 +4001,14 @@ fn build_body_cfg(
|
|
|
2761
4001
|
return None; // `push` takes exactly one arg in the JV fast path
|
|
2762
4002
|
}
|
|
2763
4003
|
let ptr = pending_push.take()?;
|
|
2764
|
-
let arg = stack.pop()
|
|
4004
|
+
let arg = stack.pop()?;
|
|
4005
|
+
let arg = jv_f64(&mut bcx, arg);
|
|
2765
4006
|
bcx.ins().call(jvc.push, &[ptr, arg]);
|
|
2766
4007
|
// `Array.push` returns the new length.
|
|
2767
4008
|
let call = bcx.ins().call(jvc.len, &[ptr]);
|
|
2768
4009
|
let len_i = bcx.inst_results(call)[0];
|
|
2769
|
-
|
|
4010
|
+
let len_f = bcx.ins().fcvt_from_uint(types::F64, len_i);
|
|
4011
|
+
stack.push(JV::f64(len_f));
|
|
2770
4012
|
ip += 3;
|
|
2771
4013
|
}
|
|
2772
4014
|
// #187: `LoadVar name` where `name` is a resolved directly-callable callee — stage it for the
|
|
@@ -2792,15 +4034,17 @@ fn build_body_cfg(
|
|
|
2792
4034
|
return None;
|
|
2793
4035
|
}
|
|
2794
4036
|
let arg_start = stack.len() - callee_arity as usize;
|
|
4037
|
+
let args: Vec<JV> = stack.drain(arg_start..).collect();
|
|
2795
4038
|
let mut call_args = Vec::with_capacity(callee_arity as usize);
|
|
2796
|
-
for
|
|
2797
|
-
if is_bool {
|
|
4039
|
+
for jv in args {
|
|
4040
|
+
if jv.is_bool() {
|
|
2798
4041
|
return None; // a bool arg doesn't match the callee's f64 ABI
|
|
2799
4042
|
}
|
|
2800
|
-
call_args.push(
|
|
4043
|
+
call_args.push(jv_f64(&mut bcx, jv));
|
|
2801
4044
|
}
|
|
2802
4045
|
let call = bcx.ins().call(fref, &call_args);
|
|
2803
|
-
|
|
4046
|
+
let res = bcx.inst_results(call)[0];
|
|
4047
|
+
stack.push(JV::f64(res));
|
|
2804
4048
|
ip += 3;
|
|
2805
4049
|
}
|
|
2806
4050
|
// #187: a VOID array-mode function's implicit `return null` (the fall-through of a
|
|
@@ -2850,6 +4094,20 @@ fn build_body_cfg(
|
|
|
2850
4094
|
terminated = true; // the following `Return` (and any dead tail) is now skipped
|
|
2851
4095
|
ip += 3;
|
|
2852
4096
|
}
|
|
4097
|
+
_ if is_binop_pow(op, code, ip) => {
|
|
4098
|
+
// #203: `a ** b` → host call to `tish_math_binary_call(Pow, a, b)` (== VM's `powf`).
|
|
4099
|
+
let r = stack.pop()?;
|
|
4100
|
+
let l = stack.pop()?;
|
|
4101
|
+
let r = jv_f64(&mut bcx, r);
|
|
4102
|
+
let l = jv_f64(&mut bcx, l);
|
|
4103
|
+
let idc = bcx
|
|
4104
|
+
.ins()
|
|
4105
|
+
.iconst(types::I32, tishlang_bytecode::MathBinaryFn::Pow as i64);
|
|
4106
|
+
let call = bcx.ins().call(math_binary_fref, &[idc, l, r]);
|
|
4107
|
+
let res = bcx.inst_results(call)[0];
|
|
4108
|
+
stack.push(JV::f64(res));
|
|
4109
|
+
ip += 3;
|
|
4110
|
+
}
|
|
2853
4111
|
_ => match emit_simple_op(&mut bcx, chunk, code, &mut ip, &mut stack, ¶ms, arity) {
|
|
2854
4112
|
SimpleOp::Handled(_) => {}
|
|
2855
4113
|
_ => return None, // LoadConst/BinOp/UnaryOp handled; anything else → VM
|
|
@@ -2888,9 +4146,10 @@ fn build_body(
|
|
|
2888
4146
|
let params: Vec<ClifValue> = bcx.block_params(entry).to_vec();
|
|
2889
4147
|
|
|
2890
4148
|
let code = &chunk.code;
|
|
2891
|
-
// Each entry is
|
|
2892
|
-
//
|
|
2893
|
-
|
|
4149
|
+
// Each entry is a typed [`JV`]. The Bool repr marks comparison/`!` results
|
|
4150
|
+
// (logical 0.0/1.0) so the final value boxes as Bool, not Number; integer
|
|
4151
|
+
// reprs materialize to f64 at the Return / select boundaries below.
|
|
4152
|
+
let mut stack: Vec<JV> = Vec::new();
|
|
2894
4153
|
let mut ip = 0usize;
|
|
2895
4154
|
let mut result: Option<bool> = None;
|
|
2896
4155
|
|
|
@@ -2903,15 +4162,17 @@ fn build_body(
|
|
|
2903
4162
|
let op = Opcode::from_u8(code[ip])?;
|
|
2904
4163
|
match op {
|
|
2905
4164
|
Opcode::Return => {
|
|
2906
|
-
let
|
|
4165
|
+
let jv = stack.pop()?;
|
|
4166
|
+
let v = jv_f64(&mut bcx, jv);
|
|
2907
4167
|
bcx.ins().return_(&[v]);
|
|
2908
|
-
result = Some(is_bool); // first Return ends a (sub)path
|
|
4168
|
+
result = Some(jv.is_bool()); // first Return ends a (sub)path
|
|
2909
4169
|
break;
|
|
2910
4170
|
}
|
|
2911
4171
|
// Ternary `cond ? A : B` → `select`. Both arms must be branch-free numeric
|
|
2912
4172
|
// sub-sequences, each pushing exactly one value, with matching is_bool.
|
|
2913
4173
|
Opcode::JumpIfFalse => {
|
|
2914
|
-
let
|
|
4174
|
+
let cond = stack.pop()?;
|
|
4175
|
+
let cond = jv_f64(&mut bcx, cond);
|
|
2915
4176
|
let mut p = ip + 1;
|
|
2916
4177
|
let off = read_u16(code, &mut p)? as i16 as isize; // p now past the operand
|
|
2917
4178
|
let else_target = (p as isize + off).max(0) as usize;
|
|
@@ -2938,7 +4199,8 @@ fn build_body(
|
|
|
2938
4199
|
if else_target != jp || stack.len() != base + 1 {
|
|
2939
4200
|
return None;
|
|
2940
4201
|
}
|
|
2941
|
-
let
|
|
4202
|
+
let then_jv = stack.pop()?;
|
|
4203
|
+
let then_v = jv_f64(&mut bcx, then_jv);
|
|
2942
4204
|
|
|
2943
4205
|
// ELSE arm: straight-line ops from `jp` up to the merge point.
|
|
2944
4206
|
let mut eip = jp;
|
|
@@ -2953,15 +4215,20 @@ fn build_body(
|
|
|
2953
4215
|
if eip != merge_target || stack.len() != base + 1 {
|
|
2954
4216
|
return None;
|
|
2955
4217
|
}
|
|
2956
|
-
let
|
|
4218
|
+
let else_jv = stack.pop()?;
|
|
4219
|
+
let else_v = jv_f64(&mut bcx, else_jv);
|
|
2957
4220
|
// One result_bool per function: arms must agree on Bool-vs-Number.
|
|
2958
|
-
if
|
|
4221
|
+
if then_jv.is_bool() != else_jv.is_bool() {
|
|
2959
4222
|
return None;
|
|
2960
4223
|
}
|
|
2961
4224
|
|
|
2962
4225
|
let falsy = falsy_flag(&mut bcx, cond);
|
|
2963
4226
|
let sel = bcx.ins().select(falsy, else_v, then_v);
|
|
2964
|
-
stack.push((
|
|
4227
|
+
stack.push(if then_jv.is_bool() {
|
|
4228
|
+
JV::boolean(sel)
|
|
4229
|
+
} else {
|
|
4230
|
+
JV::f64(sel)
|
|
4231
|
+
});
|
|
2965
4232
|
ip = merge_target;
|
|
2966
4233
|
}
|
|
2967
4234
|
// #187: a `function name(x) { … }` block body wraps its statements in EnterBlock/ExitBlock
|
|
@@ -3602,7 +4869,7 @@ mod tests {
|
|
|
3602
4869
|
assert!(!lf.used_slots.is_empty() && !lf.exits.is_empty());
|
|
3603
4870
|
let mut buf = vec![0.0f64; lf.used_slots.len()];
|
|
3604
4871
|
let mut deopt = 0u8;
|
|
3605
|
-
let exit = lf.call(&mut buf, &mut deopt);
|
|
4872
|
+
let exit = lf.call(&mut buf, &mut [], &mut deopt);
|
|
3606
4873
|
assert!((exit as usize) < lf.exits.len(), "exit id in range");
|
|
3607
4874
|
assert_eq!(deopt, 0, "v1 region never sets the deopt flag");
|
|
3608
4875
|
let mut got = buf.clone();
|
|
@@ -3615,12 +4882,13 @@ mod tests {
|
|
|
3615
4882
|
}
|
|
3616
4883
|
|
|
3617
4884
|
/// #190 — a loop that touches a non-slot value (a general call) is not pure-numeric slot math, so
|
|
3618
|
-
/// the region must be rejected (negative-cached) and the VM keeps interpreting.
|
|
3619
|
-
///
|
|
4885
|
+
/// the region must be rejected (negative-cached) and the VM keeps interpreting. A user-function
|
|
4886
|
+
/// call is such a value. (`Math.max`/`min`/`pow` used to serve as the example here, but they are
|
|
4887
|
+
/// now the #203 `MathBinary` intrinsic and DO OSR-compile — see `osr_region_handles_math_binary`.)
|
|
3620
4888
|
#[test]
|
|
3621
4889
|
fn osr_region_rejects_calls() {
|
|
3622
4890
|
let chunk = top_chunk(
|
|
3623
|
-
"
|
|
4891
|
+
"function g(x) { return x + 1.0 }\nlet a = 0.0\nfor (let i = 0; i < 100; i = i + 1) { a = a + g(i) }\n",
|
|
3624
4892
|
);
|
|
3625
4893
|
let (header, end) = first_region(&chunk);
|
|
3626
4894
|
assert!(
|
|
@@ -3629,6 +4897,20 @@ mod tests {
|
|
|
3629
4897
|
);
|
|
3630
4898
|
}
|
|
3631
4899
|
|
|
4900
|
+
/// #203 — a loop whose only "call" is a 2-arg `Math.<fn>` (the `MathBinary` intrinsic) IS
|
|
4901
|
+
/// pure-numeric slot math, so the OSR region compiles it (the win for clamp/pow kernels).
|
|
4902
|
+
#[test]
|
|
4903
|
+
fn osr_region_handles_math_binary() {
|
|
4904
|
+
let chunk = top_chunk(
|
|
4905
|
+
"let a = 0.0\nfor (let i = 0; i < 100; i = i + 1) { a = a + Math.max(i, 2.0) }\n",
|
|
4906
|
+
);
|
|
4907
|
+
let (header, end) = first_region(&chunk);
|
|
4908
|
+
assert!(
|
|
4909
|
+
try_compile_loop(&chunk, header, end).is_some(),
|
|
4910
|
+
"a loop with only a 2-arg Math intrinsic must OSR-compile"
|
|
4911
|
+
);
|
|
4912
|
+
}
|
|
4913
|
+
|
|
3632
4914
|
/// #190 — a loop with nested branches compiles and computes correctly through multiple blocks:
|
|
3633
4915
|
/// `while (i < 20) { if (i % 2 == 0) s = s + i; i = i + 1 }` → s = sum of evens in 0..19 = 90.
|
|
3634
4916
|
#[test]
|
|
@@ -3641,7 +4923,7 @@ mod tests {
|
|
|
3641
4923
|
try_compile_loop(&chunk, header, end).expect("branchy numeric loop must OSR-compile");
|
|
3642
4924
|
let mut buf = vec![0.0f64; lf.used_slots.len()];
|
|
3643
4925
|
let mut deopt = 0u8;
|
|
3644
|
-
lf.call(&mut buf, &mut deopt);
|
|
4926
|
+
lf.call(&mut buf, &mut [], &mut deopt);
|
|
3645
4927
|
let mut got = buf.clone();
|
|
3646
4928
|
got.sort_by(|a, b| a.partial_cmp(b).unwrap());
|
|
3647
4929
|
assert_eq!(got, vec![20.0, 90.0], "s=90 (0+2+…+18), i=20");
|