@tishlang/tish-lsp 2.12.0 → 2.35.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/Cargo.toml +3 -0
- package/bin/tish-lsp +0 -0
- package/crates/js_to_tish/src/transform/expr.rs +12 -2
- package/crates/tish/tests/fixtures/fs_parity_callbacks.tish +22 -0
- package/crates/tish/tests/fixtures/fs_parity_promises.tish +21 -0
- package/crates/tish/tests/fixtures/fs_parity_sync.tish +20 -0
- package/crates/tish/tests/fs_parity.rs +103 -0
- package/crates/tish/tests/integration_test.rs +80 -0
- package/crates/tish/tests/shortcircuit.rs +3 -3
- package/crates/tish_ast/src/ast.rs +84 -0
- package/crates/tish_builtins/Cargo.toml +1 -0
- package/crates/tish_builtins/src/array.rs +724 -72
- package/crates/tish_builtins/src/collections.rs +149 -40
- package/crates/tish_builtins/src/globals.rs +215 -25
- package/crates/tish_builtins/src/math.rs +65 -7
- package/crates/tish_builtins/src/number.rs +173 -0
- package/crates/tish_builtins/src/string.rs +263 -23
- package/crates/tish_bytecode/src/chunk.rs +7 -0
- package/crates/tish_bytecode/src/compiler.rs +611 -63
- package/crates/tish_bytecode/src/lib.rs +1 -1
- package/crates/tish_bytecode/src/opcode.rs +156 -2
- package/crates/tish_bytecode/src/serialize.rs +2 -0
- package/crates/tish_bytecode/tests/append_local_string_builder.rs +63 -0
- package/crates/tish_bytecode/tests/math_unary_intrinsic.rs +47 -0
- package/crates/tish_compile/src/codegen.rs +18219 -5283
- package/crates/tish_compile/src/infer.rs +1562 -30
- package/crates/tish_compile/src/lib.rs +29 -4
- package/crates/tish_compile/src/resolve.rs +151 -13
- package/crates/tish_compile/src/types.rs +224 -28
- package/crates/tish_compile/tests/dump_codegen.rs +21 -0
- package/crates/tish_compile/tests/perf_codegen_169_173.rs +8 -17
- package/crates/tish_compile/tests/perf_codegen_173_part3.rs +61 -0
- package/crates/tish_compile/tests/perf_codegen_174.rs +160 -0
- package/crates/tish_compile/tests/perf_codegen_175.rs +190 -0
- package/crates/tish_compile/tests/perf_codegen_176.rs +60 -0
- package/crates/tish_compile/tests/perf_codegen_177.rs +167 -0
- package/crates/tish_compile/tests/perf_codegen_178.rs +114 -0
- package/crates/tish_compile/tests/perf_codegen_178_rec.rs +124 -0
- package/crates/tish_compile/tests/perf_codegen_181.rs +36 -0
- package/crates/tish_compile/tests/perf_codegen_320.rs +211 -0
- package/crates/tish_compile/tests/perf_codegen_module_const_forof.rs +40 -0
- package/crates/tish_compile_js/src/codegen.rs +94 -3
- package/crates/tish_compile_js/src/tests_jsx.rs +27 -0
- package/crates/tish_compiler_wasm/src/resolve_virtual.rs +115 -2
- package/crates/tish_core/Cargo.toml +4 -0
- package/crates/tish_core/src/json.rs +109 -6
- package/crates/tish_core/src/lib.rs +222 -2
- package/crates/tish_core/src/shape.rs +4 -2
- package/crates/tish_core/src/uri.rs +45 -0
- package/crates/tish_core/src/value.rs +571 -35
- package/crates/tish_core/src/vmref.rs +14 -0
- package/crates/tish_eval/Cargo.toml +2 -1
- package/crates/tish_eval/src/eval.rs +1328 -101
- package/crates/tish_eval/src/natives.rs +283 -18
- package/crates/tish_eval/src/regex.rs +47 -1
- package/crates/tish_eval/src/value.rs +76 -21
- package/crates/tish_ffi/src/lib.rs +11 -1
- package/crates/tish_ffi/tests/double_free.rs +35 -0
- package/crates/tish_fmt/src/lib.rs +94 -1
- package/crates/tish_lexer/src/lib.rs +76 -0
- package/crates/tish_lexer/src/token.rs +4 -0
- package/crates/tish_lint/src/lib.rs +126 -0
- package/crates/tish_lsp/Cargo.toml +1 -1
- package/crates/tish_lsp/README.md +2 -1
- package/crates/tish_lsp/src/main.rs +378 -28
- package/crates/tish_native/src/build.rs +41 -0
- package/crates/tish_opt/src/lib.rs +20 -1
- package/crates/tish_parser/Cargo.toml +4 -0
- package/crates/tish_parser/src/lib.rs +68 -0
- package/crates/tish_parser/src/parser.rs +479 -28
- package/crates/tish_resolve/src/lib.rs +75 -5
- package/crates/tish_runtime/Cargo.toml +10 -1
- package/crates/tish_runtime/src/fs_ext.rs +359 -0
- package/crates/tish_runtime/src/http.rs +28 -10
- package/crates/tish_runtime/src/http_fetch.rs +150 -1
- package/crates/tish_runtime/src/http_hyper.rs +41 -18
- package/crates/tish_runtime/src/http_prefork.rs +72 -11
- package/crates/tish_runtime/src/lib.rs +651 -51
- package/crates/tish_runtime/src/timers.rs +49 -2
- package/crates/tish_ui/src/jsx.rs +21 -3
- package/crates/tish_vm/src/jit.rs +2514 -117
- package/crates/tish_vm/src/vm.rs +1479 -224
- package/package.json +1 -1
- package/platform/darwin-arm64/tish-lsp +0 -0
- package/platform/darwin-x64/tish-lsp +0 -0
- package/platform/linux-arm64/tish-lsp +0 -0
- package/platform/linux-x64/tish-lsp +0 -0
- package/platform/win32-x64/tish-lsp.exe +0 -0
|
@@ -43,6 +43,179 @@ pub fn to_fixed_str(num: f64, digits: usize) -> String {
|
|
|
43
43
|
format!("{:.*}", digits, rounded)
|
|
44
44
|
}
|
|
45
45
|
|
|
46
|
+
/// `Number.prototype.toExponential(fractionDigits?)` — ECMA-262 §21.1.3.2. Exponential notation with
|
|
47
|
+
/// `fractionDigits` mantissa fraction digits (0–100), or the minimal digits needed when omitted. The
|
|
48
|
+
/// exponent always carries an explicit sign (`1.23e+4`, `1e-7`) to match V8.
|
|
49
|
+
pub fn to_exponential(n: &Value, digits: &Value) -> Value {
|
|
50
|
+
let num = match n {
|
|
51
|
+
Value::Number(x) => *x,
|
|
52
|
+
_ => f64::NAN,
|
|
53
|
+
};
|
|
54
|
+
let d = match digits {
|
|
55
|
+
Value::Number(x) => Some((*x as i32).clamp(0, 100) as usize),
|
|
56
|
+
_ => None,
|
|
57
|
+
};
|
|
58
|
+
Value::String(to_exponential_str(num, d).into())
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
/// f64-domain core of `toExponential`, shared with the tree-walk interpreter.
|
|
62
|
+
pub fn to_exponential_str(num: f64, digits: Option<usize>) -> String {
|
|
63
|
+
if num.is_nan() {
|
|
64
|
+
return "NaN".to_string();
|
|
65
|
+
}
|
|
66
|
+
if num.is_infinite() {
|
|
67
|
+
return if num < 0.0 { "-Infinity" } else { "Infinity" }.to_string();
|
|
68
|
+
}
|
|
69
|
+
let d = match digits {
|
|
70
|
+
// No fractionDigits: shortest round-trip mantissa (Rust's `{:e}` matches V8 here).
|
|
71
|
+
None => return fix_exponent_sign(&format!("{:e}", num)),
|
|
72
|
+
Some(d) => d,
|
|
73
|
+
};
|
|
74
|
+
if num == 0.0 {
|
|
75
|
+
let mant = if d == 0 {
|
|
76
|
+
"0".to_string()
|
|
77
|
+
} else {
|
|
78
|
+
format!("0.{}", "0".repeat(d))
|
|
79
|
+
};
|
|
80
|
+
return format!("{}e+0", mant);
|
|
81
|
+
}
|
|
82
|
+
let sign = if num < 0.0 { "-" } else { "" };
|
|
83
|
+
let (ds, e) = sig_digits(num.abs(), d + 1);
|
|
84
|
+
let mant = if d == 0 {
|
|
85
|
+
ds
|
|
86
|
+
} else {
|
|
87
|
+
format!("{}.{}", &ds[..1], &ds[1..])
|
|
88
|
+
};
|
|
89
|
+
let esign = if e >= 0 { "+" } else { "-" };
|
|
90
|
+
format!("{}{}e{}{}", sign, mant, esign, e.abs())
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
/// Round `x` (finite, `x > 0`) to `sig` significant decimal digits with **half-away-from-zero**
|
|
94
|
+
/// rounding — ECMA's "pick the larger n" tie rule (`2.5.toPrecision(1) === "3"`, not "2"). Returns the
|
|
95
|
+
/// `sig`-digit string and the base-10 exponent of the leading digit.
|
|
96
|
+
///
|
|
97
|
+
/// The exact value of an f64 has a finite decimal expansion; scaling it through `f64` arithmetic
|
|
98
|
+
/// re-rounds and can land a genuinely-below-half value (e.g. `2.675` ≈ `2.67499…`) exactly on `.5`,
|
|
99
|
+
/// corrupting the tie test. So we format with many guard digits — `{:.Ne}` gives the correctly-rounded
|
|
100
|
+
/// exact-value expansion — capturing the TRUE digit at position `sig`, then round the digit string:
|
|
101
|
+
/// `digit[sig] >= 5` rounds up (half-away; `== 5` is a true tie only because the guard digits show no
|
|
102
|
+
/// nonzero remainder). Rust's `{:e}` alone rounds half-to-even, which is exactly what we must avoid.
|
|
103
|
+
fn sig_digits(x: f64, sig: usize) -> (String, i32) {
|
|
104
|
+
let sig = sig.max(1);
|
|
105
|
+
// Guard digits push the format's own (half-to-even) rounding far to the right of digit[sig], so
|
|
106
|
+
// digit[sig] is exact. A cascade into digit[sig] would need 16 consecutive 9s — not a real input.
|
|
107
|
+
let guard = sig + 16;
|
|
108
|
+
let raw = format!("{:.*e}", guard, x);
|
|
109
|
+
let (mant, exp_str) = raw.split_once('e').unwrap();
|
|
110
|
+
let mut e: i32 = exp_str.parse().unwrap_or(0);
|
|
111
|
+
let mut digits: Vec<u8> = mant
|
|
112
|
+
.bytes()
|
|
113
|
+
.filter(u8::is_ascii_digit)
|
|
114
|
+
.map(|b| b - b'0')
|
|
115
|
+
.collect();
|
|
116
|
+
if digits.len() > sig {
|
|
117
|
+
let round_up = digits[sig] >= 5;
|
|
118
|
+
digits.truncate(sig);
|
|
119
|
+
if round_up {
|
|
120
|
+
let mut i = sig;
|
|
121
|
+
loop {
|
|
122
|
+
if i == 0 {
|
|
123
|
+
// carry out of the leading digit: 9…9 → 1 0…0, one more place
|
|
124
|
+
digits.insert(0, 1);
|
|
125
|
+
digits.truncate(sig);
|
|
126
|
+
e += 1;
|
|
127
|
+
break;
|
|
128
|
+
}
|
|
129
|
+
i -= 1;
|
|
130
|
+
if digits[i] == 9 {
|
|
131
|
+
digits[i] = 0;
|
|
132
|
+
} else {
|
|
133
|
+
digits[i] += 1;
|
|
134
|
+
break;
|
|
135
|
+
}
|
|
136
|
+
}
|
|
137
|
+
}
|
|
138
|
+
}
|
|
139
|
+
let s: String = digits.iter().map(|d| (d + b'0') as char).collect();
|
|
140
|
+
(s, e)
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
/// Rust's `{:e}` writes a bare positive exponent (`1.23e4`); JS/V8 always signs it (`1.23e+4`). Negative
|
|
144
|
+
/// exponents already carry `-`.
|
|
145
|
+
fn fix_exponent_sign(s: &str) -> String {
|
|
146
|
+
match s.split_once('e') {
|
|
147
|
+
Some((mantissa, exp)) if !exp.starts_with('-') && !exp.starts_with('+') => {
|
|
148
|
+
format!("{}e+{}", mantissa, exp)
|
|
149
|
+
}
|
|
150
|
+
_ => s.to_string(),
|
|
151
|
+
}
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
/// `Number.prototype.toPrecision(precision?)` — ECMA-262 §21.1.3.5. With `precision` significant digits,
|
|
155
|
+
/// choosing fixed or exponential per the spec; omitted precision falls back to `ToString`. An
|
|
156
|
+
/// out-of-range precision (`<1` or `>100`) parks a catchable `RangeError`.
|
|
157
|
+
pub fn to_precision(n: &Value, precision: &Value) -> Value {
|
|
158
|
+
let num = match n {
|
|
159
|
+
Value::Number(x) => *x,
|
|
160
|
+
_ => f64::NAN,
|
|
161
|
+
};
|
|
162
|
+
match precision {
|
|
163
|
+
Value::Number(p) => {
|
|
164
|
+
let p = *p as i32;
|
|
165
|
+
if !(1..=100).contains(&p) && num.is_finite() {
|
|
166
|
+
tishlang_core::set_pending_throw(tishlang_core::range_error(
|
|
167
|
+
"toPrecision() argument must be between 1 and 100",
|
|
168
|
+
));
|
|
169
|
+
return Value::Null;
|
|
170
|
+
}
|
|
171
|
+
Value::String(to_precision_str(num, p).into())
|
|
172
|
+
}
|
|
173
|
+
// precision undefined → behave like ToString(number)
|
|
174
|
+
_ => Value::String(tishlang_core::js_number_to_string(num).into()),
|
|
175
|
+
}
|
|
176
|
+
}
|
|
177
|
+
|
|
178
|
+
/// f64-domain core of `toPrecision`, shared with the tree-walk interpreter. Assumes `precision` is in
|
|
179
|
+
/// range (the `&Value` entry point validates and parks a RangeError otherwise).
|
|
180
|
+
pub fn to_precision_str(num: f64, precision: i32) -> String {
|
|
181
|
+
if num.is_nan() {
|
|
182
|
+
return "NaN".to_string();
|
|
183
|
+
}
|
|
184
|
+
if num.is_infinite() {
|
|
185
|
+
return if num < 0.0 { "-Infinity" } else { "Infinity" }.to_string();
|
|
186
|
+
}
|
|
187
|
+
let p = precision.clamp(1, 100) as usize;
|
|
188
|
+
if num == 0.0 {
|
|
189
|
+
if p == 1 {
|
|
190
|
+
return "0".to_string();
|
|
191
|
+
}
|
|
192
|
+
return format!("0.{}", "0".repeat(p - 1));
|
|
193
|
+
}
|
|
194
|
+
let sign = if num < 0.0 { "-" } else { "" };
|
|
195
|
+
// `p` significant digits + base-10 exponent, half-away rounded (shared with toExponential).
|
|
196
|
+
let (digits, e) = sig_digits(num.abs(), p);
|
|
197
|
+
let body = if e < -6 || e >= p as i32 {
|
|
198
|
+
// exponential form
|
|
199
|
+
let mut m = String::new();
|
|
200
|
+
m.push(digits.as_bytes()[0] as char);
|
|
201
|
+
if p > 1 {
|
|
202
|
+
m.push('.');
|
|
203
|
+
m.push_str(&digits[1..]);
|
|
204
|
+
}
|
|
205
|
+
let esign = if e >= 0 { "+" } else { "-" };
|
|
206
|
+
format!("{}e{}{}", m, esign, e.abs())
|
|
207
|
+
} else if e == p as i32 - 1 {
|
|
208
|
+
digits
|
|
209
|
+
} else if e >= 0 {
|
|
210
|
+
let split = (e + 1) as usize;
|
|
211
|
+
format!("{}.{}", &digits[..split], &digits[split..])
|
|
212
|
+
} else {
|
|
213
|
+
let zeros = (-e - 1) as usize;
|
|
214
|
+
format!("0.{}{}", "0".repeat(zeros), digits)
|
|
215
|
+
};
|
|
216
|
+
format!("{}{}", sign, body)
|
|
217
|
+
}
|
|
218
|
+
|
|
46
219
|
/// `Number.prototype.toString([radix])` — ECMA-262 §21.1.3.6.
|
|
47
220
|
///
|
|
48
221
|
/// Radix defaults to 10 (canonical JS number formatting). For radix 2–36 the value is
|
|
@@ -4,9 +4,64 @@
|
|
|
4
4
|
//! JavaScript, matching .length and .charAt(). Byte offsets are never exposed.
|
|
5
5
|
|
|
6
6
|
use crate::helpers::normalize_index;
|
|
7
|
+
use std::cell::RefCell;
|
|
8
|
+
use tishlang_core::ArcStr;
|
|
7
9
|
use tishlang_core::Value;
|
|
8
10
|
use tishlang_core::VmRef;
|
|
9
11
|
|
|
12
|
+
// #203: a per-thread cursor cache that makes repeated character indexing (`charCodeAt(i)`, `s[i]`,
|
|
13
|
+
// `charAt(i)`) O(1)/near-O(1) instead of O(i). tish strings are UTF-8, so `chars().nth(i)` scans from
|
|
14
|
+
// the start — turning indexed/strided scans into O(n^2) (a strided checksum over a 1.3 MB string was
|
|
15
|
+
// 4939ms vs node 1ms). For each recently-indexed string we cache whether it is all-ASCII (then a
|
|
16
|
+
// character index equals a byte index → O(1) byte lookup) plus a forward cursor (so non-ASCII
|
|
17
|
+
// sequential/strided scans advance from the last position, not from 0). Safety: the entry holds an
|
|
18
|
+
// `ArcStr` CLONE, which keeps the backing allocation alive — so its data pointer can't be freed and
|
|
19
|
+
// reused by another string while cached (no ABA), and since strings are immutable the cached ASCII
|
|
20
|
+
// flag stays valid. Backends share this via `char_at_idx` (native + VM route through the builtin).
|
|
21
|
+
// Semantics are unchanged: still character (Unicode scalar) indexing, identical to `chars().nth(i)`.
|
|
22
|
+
struct CharCursor {
|
|
23
|
+
s: ArcStr,
|
|
24
|
+
ascii: bool,
|
|
25
|
+
/// Character (Unicode scalar) count, cached so `.length` is O(1) too — `for (i=0;i<s.length;i++)`
|
|
26
|
+
/// re-evaluates the bound every iteration, so an O(n) `chars().count()` there is itself O(n^2).
|
|
27
|
+
len_chars: usize,
|
|
28
|
+
char_idx: usize,
|
|
29
|
+
byte_off: usize,
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
thread_local! {
|
|
33
|
+
static INDEX_CURSOR: RefCell<Option<CharCursor>> = const { RefCell::new(None) };
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
/// Borrow the cursor entry for `s`, reseeding (compute ASCII flag + character count) if it currently
|
|
37
|
+
/// caches a different backing allocation, then run `f` against it. Centralises the pointer-keyed
|
|
38
|
+
/// reseed shared by [`char_at_idx`] and [`char_count`].
|
|
39
|
+
fn with_cursor<R>(s: &ArcStr, f: impl FnOnce(&mut CharCursor, &ArcStr) -> R) -> R {
|
|
40
|
+
INDEX_CURSOR.with(|cell| {
|
|
41
|
+
let mut slot = cell.borrow_mut();
|
|
42
|
+
let hit = matches!(slot.as_ref(), Some(c)
|
|
43
|
+
if std::ptr::eq(c.s.as_bytes().as_ptr(), s.as_bytes().as_ptr()) && c.s.len() == s.len());
|
|
44
|
+
if !hit {
|
|
45
|
+
let ascii = s.as_bytes().is_ascii();
|
|
46
|
+
// ASCII → character count equals byte length (free); otherwise count once.
|
|
47
|
+
let len_chars = if ascii { s.len() } else { s.chars().count() };
|
|
48
|
+
*slot = Some(CharCursor {
|
|
49
|
+
s: s.clone(),
|
|
50
|
+
ascii,
|
|
51
|
+
len_chars,
|
|
52
|
+
char_idx: 0,
|
|
53
|
+
byte_off: 0,
|
|
54
|
+
});
|
|
55
|
+
}
|
|
56
|
+
f(slot.as_mut().unwrap(), s)
|
|
57
|
+
})
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
/// Character (Unicode scalar) count of `s`, O(1) after the first call on a given string.
|
|
61
|
+
pub fn char_count(s: &ArcStr) -> usize {
|
|
62
|
+
with_cursor(s, |c, _| c.len_chars)
|
|
63
|
+
}
|
|
64
|
+
|
|
10
65
|
/// Byte offset -> character index.
|
|
11
66
|
fn byte_to_char_index(s: &str, byte_offset: usize) -> usize {
|
|
12
67
|
s.char_indices()
|
|
@@ -30,7 +85,7 @@ pub fn from_str(s: &str) -> Value {
|
|
|
30
85
|
/// Get the length of a string (character count).
|
|
31
86
|
pub fn len(s: &Value) -> Option<usize> {
|
|
32
87
|
match s {
|
|
33
|
-
Value::String(str) => Some(str
|
|
88
|
+
Value::String(str) => Some(char_count(str)),
|
|
34
89
|
_ => None,
|
|
35
90
|
}
|
|
36
91
|
}
|
|
@@ -262,6 +317,59 @@ pub fn trim(s: &Value) -> Value {
|
|
|
262
317
|
}
|
|
263
318
|
}
|
|
264
319
|
|
|
320
|
+
pub fn trim_start(s: &Value) -> Value {
|
|
321
|
+
if let Value::String(s) = s {
|
|
322
|
+
Value::String(s.trim_start().into())
|
|
323
|
+
} else {
|
|
324
|
+
Value::Null
|
|
325
|
+
}
|
|
326
|
+
}
|
|
327
|
+
|
|
328
|
+
/// Pure `String.prototype.normalize` core (ECMA-262 §22.1.3.13). `form` is one of NFC/NFD/NFKC/NFKD;
|
|
329
|
+
/// returns `None` for an unrecognized form so the caller can surface a `RangeError`. Shared by every
|
|
330
|
+
/// backend so interp/vm/native stay byte-identical.
|
|
331
|
+
pub fn normalize_form(s: &str, form: &str) -> Option<String> {
|
|
332
|
+
use unicode_normalization::UnicodeNormalization;
|
|
333
|
+
match form {
|
|
334
|
+
"NFC" => Some(s.nfc().collect()),
|
|
335
|
+
"NFD" => Some(s.nfd().collect()),
|
|
336
|
+
"NFKC" => Some(s.nfkc().collect()),
|
|
337
|
+
"NFKD" => Some(s.nfkd().collect()),
|
|
338
|
+
_ => None,
|
|
339
|
+
}
|
|
340
|
+
}
|
|
341
|
+
|
|
342
|
+
/// `String.prototype.normalize(form?)` — core-`Value` entry (vm/native). An omitted form defaults to
|
|
343
|
+
/// NFC; an invalid form parks a catchable `RangeError`.
|
|
344
|
+
pub fn normalize(s: &Value, form: &Value) -> Value {
|
|
345
|
+
let input = match s {
|
|
346
|
+
Value::String(s) => s.as_ref(),
|
|
347
|
+
_ => return Value::Null,
|
|
348
|
+
};
|
|
349
|
+
let f: String = match form {
|
|
350
|
+
Value::Null => "NFC".to_string(),
|
|
351
|
+
Value::String(f) => f.to_string(),
|
|
352
|
+
v => v.to_display_string(),
|
|
353
|
+
};
|
|
354
|
+
match normalize_form(input, &f) {
|
|
355
|
+
Some(out) => Value::String(out.into()),
|
|
356
|
+
None => {
|
|
357
|
+
tishlang_core::set_pending_throw(tishlang_core::range_error(
|
|
358
|
+
"The normalization form should be one of NFC, NFD, NFKC, NFKD.",
|
|
359
|
+
));
|
|
360
|
+
Value::Null
|
|
361
|
+
}
|
|
362
|
+
}
|
|
363
|
+
}
|
|
364
|
+
|
|
365
|
+
pub fn trim_end(s: &Value) -> Value {
|
|
366
|
+
if let Value::String(s) = s {
|
|
367
|
+
Value::String(s.trim_end().into())
|
|
368
|
+
} else {
|
|
369
|
+
Value::Null
|
|
370
|
+
}
|
|
371
|
+
}
|
|
372
|
+
|
|
265
373
|
pub fn to_upper_case(s: &Value) -> Value {
|
|
266
374
|
if let Value::String(s) = s {
|
|
267
375
|
Value::String(s.to_uppercase().into())
|
|
@@ -278,22 +386,74 @@ pub fn to_lower_case(s: &Value) -> Value {
|
|
|
278
386
|
}
|
|
279
387
|
}
|
|
280
388
|
|
|
281
|
-
pub fn starts_with(s: &Value, search: &Value) -> Value {
|
|
389
|
+
pub fn starts_with(s: &Value, search: &Value, position: Option<&Value>) -> Value {
|
|
282
390
|
if let (Value::String(s), Value::String(search)) = (s, search) {
|
|
283
|
-
|
|
391
|
+
// `position`: test as if the string began at char `position` (clamped to [0, len]). Absent → 0.
|
|
392
|
+
let pos = match position {
|
|
393
|
+
Some(Value::Number(n)) if *n > 0.0 => *n as usize,
|
|
394
|
+
_ => 0,
|
|
395
|
+
};
|
|
396
|
+
let byte_start = s.char_indices().nth(pos).map(|(b, _)| b).unwrap_or(s.len());
|
|
397
|
+
Value::Bool(s[byte_start..].starts_with(search.as_str()))
|
|
284
398
|
} else {
|
|
285
399
|
Value::Bool(false)
|
|
286
400
|
}
|
|
287
401
|
}
|
|
288
402
|
|
|
289
|
-
pub fn ends_with(s: &Value, search: &Value) -> Value {
|
|
403
|
+
pub fn ends_with(s: &Value, search: &Value, end_position: Option<&Value>) -> Value {
|
|
290
404
|
if let (Value::String(s), Value::String(search)) = (s, search) {
|
|
291
|
-
|
|
405
|
+
// `endPosition`: test as if the string ended at char `endPosition` (clamped to [0, len]).
|
|
406
|
+
// Absent → the full length.
|
|
407
|
+
let char_count = s.chars().count();
|
|
408
|
+
let end = match end_position {
|
|
409
|
+
Some(Value::Number(n)) if *n >= 0.0 => (*n as usize).min(char_count),
|
|
410
|
+
Some(Value::Number(_)) => 0,
|
|
411
|
+
_ => char_count,
|
|
412
|
+
};
|
|
413
|
+
let byte_end = s.char_indices().nth(end).map(|(b, _)| b).unwrap_or(s.len());
|
|
414
|
+
Value::Bool(s[..byte_end].ends_with(search.as_str()))
|
|
292
415
|
} else {
|
|
293
416
|
Value::Bool(false)
|
|
294
417
|
}
|
|
295
418
|
}
|
|
296
419
|
|
|
420
|
+
/// Expand a string replacement's `$` patterns for one match: `$$`→`$`, `$&`→the match, `` $` ``→the
|
|
421
|
+
/// text before it, `$'`→the text after it. Any other `$x` is literal. (String-search `replace`; the
|
|
422
|
+
/// numbered `$1` captures only apply to regex replacements.)
|
|
423
|
+
pub fn expand_replacement(repl: &str, matched: &str, before: &str, after: &str) -> String {
|
|
424
|
+
if !repl.contains('$') {
|
|
425
|
+
return repl.to_string();
|
|
426
|
+
}
|
|
427
|
+
let mut out = String::with_capacity(repl.len());
|
|
428
|
+
let mut chars = repl.chars().peekable();
|
|
429
|
+
while let Some(c) = chars.next() {
|
|
430
|
+
if c == '$' {
|
|
431
|
+
match chars.peek() {
|
|
432
|
+
Some('$') => {
|
|
433
|
+
out.push('$');
|
|
434
|
+
chars.next();
|
|
435
|
+
}
|
|
436
|
+
Some('&') => {
|
|
437
|
+
out.push_str(matched);
|
|
438
|
+
chars.next();
|
|
439
|
+
}
|
|
440
|
+
Some('`') => {
|
|
441
|
+
out.push_str(before);
|
|
442
|
+
chars.next();
|
|
443
|
+
}
|
|
444
|
+
Some('\'') => {
|
|
445
|
+
out.push_str(after);
|
|
446
|
+
chars.next();
|
|
447
|
+
}
|
|
448
|
+
_ => out.push('$'),
|
|
449
|
+
}
|
|
450
|
+
} else {
|
|
451
|
+
out.push(c);
|
|
452
|
+
}
|
|
453
|
+
}
|
|
454
|
+
out
|
|
455
|
+
}
|
|
456
|
+
|
|
297
457
|
fn replace_impl(s: &Value, search: &Value, replacement: &Value, all: bool) -> Value {
|
|
298
458
|
if let Value::String(s) = s {
|
|
299
459
|
let search_str = match search {
|
|
@@ -304,12 +464,37 @@ fn replace_impl(s: &Value, search: &Value, replacement: &Value, all: bool) -> Va
|
|
|
304
464
|
Value::String(ss) => ss.as_str(),
|
|
305
465
|
_ => "",
|
|
306
466
|
};
|
|
307
|
-
|
|
308
|
-
|
|
309
|
-
|
|
310
|
-
|
|
311
|
-
|
|
312
|
-
|
|
467
|
+
// Fast path: no `$` in the replacement (and non-empty search) → the library replace.
|
|
468
|
+
if !repl_str.contains('$') || search_str.is_empty() {
|
|
469
|
+
let result = if all {
|
|
470
|
+
s.replace(search_str, repl_str)
|
|
471
|
+
} else {
|
|
472
|
+
s.replacen(search_str, repl_str, 1)
|
|
473
|
+
};
|
|
474
|
+
return Value::String(result.into());
|
|
475
|
+
}
|
|
476
|
+
// `$`-expansion path: expand per match against the original string's before/after context.
|
|
477
|
+
let sref = s.as_str();
|
|
478
|
+
let mut out = String::with_capacity(sref.len());
|
|
479
|
+
let mut last = 0usize;
|
|
480
|
+
let mut start = 0usize;
|
|
481
|
+
while let Some(pos) = sref[start..].find(search_str) {
|
|
482
|
+
let m = start + pos;
|
|
483
|
+
out.push_str(&sref[last..m]);
|
|
484
|
+
out.push_str(&expand_replacement(
|
|
485
|
+
repl_str,
|
|
486
|
+
search_str,
|
|
487
|
+
&sref[..m],
|
|
488
|
+
&sref[m + search_str.len()..],
|
|
489
|
+
));
|
|
490
|
+
last = m + search_str.len();
|
|
491
|
+
start = last;
|
|
492
|
+
if !all {
|
|
493
|
+
break;
|
|
494
|
+
}
|
|
495
|
+
}
|
|
496
|
+
out.push_str(&sref[last..]);
|
|
497
|
+
Value::String(out.into())
|
|
313
498
|
} else {
|
|
314
499
|
Value::Null
|
|
315
500
|
}
|
|
@@ -370,8 +555,38 @@ pub fn escape_html(s: &Value) -> Value {
|
|
|
370
555
|
Value::String(tishlang_core::ArcStr::from(out))
|
|
371
556
|
}
|
|
372
557
|
|
|
373
|
-
|
|
374
|
-
|
|
558
|
+
/// Character (Unicode scalar) at index `idx`, using the cursor cache (see [`CharCursor`]).
|
|
559
|
+
/// Equivalent to `s.chars().nth(idx)` but O(1) for ASCII strings and near-O(1) for forward/strided
|
|
560
|
+
/// scans of non-ASCII strings, instead of O(idx) every call.
|
|
561
|
+
fn char_at_idx(s: &ArcStr, idx: usize) -> Option<char> {
|
|
562
|
+
with_cursor(s, |c, s| {
|
|
563
|
+
if c.ascii {
|
|
564
|
+
// ASCII: character index == byte index, and every byte is its own scalar.
|
|
565
|
+
return s.as_bytes().get(idx).map(|&b| b as char);
|
|
566
|
+
}
|
|
567
|
+
// Non-ASCII: advance from the nearest known position (forward fast path); restart from 0 only
|
|
568
|
+
// when indexing backwards relative to the cursor.
|
|
569
|
+
let (base_idx, base_off) = if idx >= c.char_idx {
|
|
570
|
+
(c.char_idx, c.byte_off)
|
|
571
|
+
} else {
|
|
572
|
+
(0, 0)
|
|
573
|
+
};
|
|
574
|
+
match s[base_off..].char_indices().nth(idx - base_idx) {
|
|
575
|
+
Some((rel_off, ch)) => {
|
|
576
|
+
c.char_idx = idx;
|
|
577
|
+
c.byte_off = base_off + rel_off;
|
|
578
|
+
Some(ch)
|
|
579
|
+
}
|
|
580
|
+
None => None,
|
|
581
|
+
}
|
|
582
|
+
})
|
|
583
|
+
}
|
|
584
|
+
|
|
585
|
+
/// Character (Unicode scalar) at index `idx` via the cursor cache — the O(1)/near-O(1) primitive
|
|
586
|
+
/// behind `s[i]`. Returns `None` for an out-of-range index; each backend maps that to its own
|
|
587
|
+
/// out-of-bounds behaviour (interpreter/native → null, VM → error).
|
|
588
|
+
pub fn nth_char(s: &ArcStr, idx: usize) -> Option<char> {
|
|
589
|
+
char_at_idx(s, idx)
|
|
375
590
|
}
|
|
376
591
|
|
|
377
592
|
pub fn char_at(s: &Value, idx: &Value) -> Value {
|
|
@@ -396,11 +611,13 @@ pub fn at(s: &Value, index: &Value) -> Value {
|
|
|
396
611
|
Value::Number(n) => *n as i64,
|
|
397
612
|
_ => 0,
|
|
398
613
|
};
|
|
399
|
-
|
|
400
|
-
|
|
401
|
-
let idx = if i < 0 {
|
|
402
|
-
if idx >= 0
|
|
403
|
-
|
|
614
|
+
// Non-negative indices use the cursor cache directly; a negative index counts from the end,
|
|
615
|
+
// which needs the character length first (inherently O(n)).
|
|
616
|
+
let idx = if i < 0 { s.chars().count() as i64 + i } else { i };
|
|
617
|
+
if idx >= 0 {
|
|
618
|
+
if let Some(c) = char_at_idx(s, idx as usize) {
|
|
619
|
+
return Value::String(c.to_string().into());
|
|
620
|
+
}
|
|
404
621
|
}
|
|
405
622
|
}
|
|
406
623
|
Value::Null
|
|
@@ -420,6 +637,23 @@ pub fn char_code_at(s: &Value, idx: &Value) -> Value {
|
|
|
420
637
|
}
|
|
421
638
|
}
|
|
422
639
|
|
|
640
|
+
/// `codePointAt(i)` — the Unicode code point at char index `i`, or `null` (JS `undefined`) if out of
|
|
641
|
+
/// range. tish strings are code-point sequences, so this returns the same value as `charCodeAt` for
|
|
642
|
+
/// an in-range index, but reads `null` (not `NaN`) past the end.
|
|
643
|
+
pub fn code_point_at(s: &Value, idx: &Value) -> Value {
|
|
644
|
+
if let Value::String(s) = s {
|
|
645
|
+
let idx = match idx {
|
|
646
|
+
Value::Number(n) if *n >= 0.0 => *n as usize,
|
|
647
|
+
_ => return Value::Null,
|
|
648
|
+
};
|
|
649
|
+
char_at_idx(s, idx)
|
|
650
|
+
.map(|c| Value::Number(c as u32 as f64))
|
|
651
|
+
.unwrap_or(Value::Null)
|
|
652
|
+
} else {
|
|
653
|
+
Value::Null
|
|
654
|
+
}
|
|
655
|
+
}
|
|
656
|
+
|
|
423
657
|
pub fn repeat(s: &Value, count: &Value) -> Value {
|
|
424
658
|
if let Value::String(s) = s {
|
|
425
659
|
let count = match count {
|
|
@@ -438,12 +672,15 @@ fn pad_impl(s: &Value, target_len: &Value, pad: &Value, at_start: bool) -> Value
|
|
|
438
672
|
Value::Number(n) => *n as usize,
|
|
439
673
|
_ => return Value::String(s.clone()),
|
|
440
674
|
};
|
|
675
|
+
// An *explicit* empty fill string means "no padding" (spec: `padStart(n, "")` → original,
|
|
676
|
+
// unchanged); only an ABSENT pad arg (`Value::Null`) defaults to a space. The old
|
|
677
|
+
// `if !p.is_empty()` guard conflated the two, space-padding on an explicit `""`.
|
|
441
678
|
let pad_str = match pad {
|
|
442
|
-
Value::String(p)
|
|
679
|
+
Value::String(p) => p.as_str(),
|
|
443
680
|
_ => " ",
|
|
444
681
|
};
|
|
445
682
|
let char_count = s.chars().count();
|
|
446
|
-
if char_count >= target_len {
|
|
683
|
+
if char_count >= target_len || pad_str.is_empty() {
|
|
447
684
|
return Value::String(s.clone());
|
|
448
685
|
}
|
|
449
686
|
let needed = target_len - char_count;
|
|
@@ -609,9 +846,12 @@ mod tests {
|
|
|
609
846
|
fn case_and_prefix_suffix() {
|
|
610
847
|
assert_same!(to_upper_case(&s("aB")), s("AB"));
|
|
611
848
|
assert_same!(to_lower_case(&s("aB")), s("ab"));
|
|
612
|
-
assert_same!(starts_with(&s("/api"), &s("/api")), Value::Bool(true));
|
|
613
|
-
assert_same!(ends_with(&s("x.js"), &s(".js")), Value::Bool(true));
|
|
614
|
-
assert_same!(starts_with(&n(1.0), &s("")), Value::Bool(false));
|
|
849
|
+
assert_same!(starts_with(&s("/api"), &s("/api"), None), Value::Bool(true));
|
|
850
|
+
assert_same!(ends_with(&s("x.js"), &s(".js"), None), Value::Bool(true));
|
|
851
|
+
assert_same!(starts_with(&n(1.0), &s(""), None), Value::Bool(false));
|
|
852
|
+
// 2nd-arg: position / endPosition.
|
|
853
|
+
assert_same!(starts_with(&s("abc"), &s("bc"), Some(&n(1.0))), Value::Bool(true));
|
|
854
|
+
assert_same!(ends_with(&s("abc"), &s("ab"), Some(&n(2.0))), Value::Bool(true));
|
|
615
855
|
}
|
|
616
856
|
|
|
617
857
|
#[test]
|
|
@@ -79,6 +79,12 @@ pub struct Chunk {
|
|
|
79
79
|
pub lines: Vec<(u32, u32)>,
|
|
80
80
|
/// Source file path for error messages (`file:line`); propagated to nested chunks. Runtime-only.
|
|
81
81
|
pub source: Option<Arc<str>>,
|
|
82
|
+
/// #187: when `Some(name)`, this chunk is a top-level `function name` whose binding is provably
|
|
83
|
+
/// stable across the whole program (never reassigned/shadowed/redeclared). The numeric JIT
|
|
84
|
+
/// registers such a chunk under `name` so a caller's `name(args)` can lower to a direct native
|
|
85
|
+
/// call. `None` for anonymous/nested/unstable functions. Runtime-only; not serialized (a reloaded
|
|
86
|
+
/// program just forgoes the cross-function-call optimization).
|
|
87
|
+
pub global_name: Option<Arc<str>>,
|
|
82
88
|
}
|
|
83
89
|
|
|
84
90
|
impl Chunk {
|
|
@@ -95,6 +101,7 @@ impl Chunk {
|
|
|
95
101
|
inline_caches: InlineCaches::default(),
|
|
96
102
|
lines: Vec::new(),
|
|
97
103
|
source: None,
|
|
104
|
+
global_name: None,
|
|
98
105
|
}
|
|
99
106
|
}
|
|
100
107
|
|