@lucent-lang/lucent 0.0.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/app.plugin.js +92 -0
- package/bin/lucent.cjs +16 -0
- package/core/index.cjs +11 -0
- package/dist/bench-CR3E7e2e.js +304 -0
- package/dist/build-CAIrRSYA.js +7 -0
- package/dist/build-check-QozuNJm5.js +92 -0
- package/dist/check-1SCmYlqA.js +7 -0
- package/dist/clean-BVwpbJ5K.js +35 -0
- package/dist/cli.js +406 -0
- package/dist/codes-CIuwXt1S.js +260 -0
- package/dist/compiler.js +4 -0
- package/dist/confirm-BYF5eY3q.js +34 -0
- package/dist/dashboard-B-NzjjaN.js +91 -0
- package/dist/dev-BYWMP7oC.js +221 -0
- package/dist/diagnostic-BM9sXmwl.js +30 -0
- package/dist/diff-BTDo3Sro.js +44 -0
- package/dist/doctor-C0PrfWT6.js +230 -0
- package/dist/doctor-DdR4JnVc.js +30 -0
- package/dist/explain--vV1oaRl.js +35 -0
- package/dist/format-D9fnIMEx.js +53 -0
- package/dist/init-gIabu1S_.js +207 -0
- package/dist/live-steps-BGcWFM9M.js +79 -0
- package/dist/new-module-C2mtb0op.js +66 -0
- package/dist/package-manager-BnxwCySi.js +30 -0
- package/dist/package.json +3 -0
- package/dist/packages-HyskKIsY.js +150 -0
- package/dist/pipeline-CH7-2cxu.js +232 -0
- package/dist/project-D8tEet0k.js +150 -0
- package/dist/sdk-coverage-BoW_3Efd.js +64 -0
- package/dist/sdk-prefetch-mo04JDZV.js +89 -0
- package/dist/sdk-search-Iv4K3-8a.js +124 -0
- package/dist/sdk-show-CkrdlqlP.js +74 -0
- package/dist/sdks-Xoyiy2E4.js +28 -0
- package/dist/src-BD7f1_aZ.js +11493 -0
- package/dist/steps--_h-FFS7.js +24 -0
- package/dist/theme-uv5bc0aK.js +134 -0
- package/dist/tsconfig-S4uyLNzf.js +59 -0
- package/dist/version-bqpIO5sP.js +26 -0
- package/gradle/lucent-classpath.gradle +38 -0
- package/gradle/lucent-classpath.init.gradle +6 -0
- package/gradle/lucent.gradle +24 -0
- package/lib/core.cjs +106 -0
- package/lib/globals.d.ts +36 -0
- package/lib/sdk/android.d.ts +8 -0
- package/lib/sdk/core.d.ts +23 -0
- package/lib/sdk/ios.d.ts +33 -0
- package/lib/sdk/platform.d.ts +8 -0
- package/lib/sdk/thread.d.ts +8 -0
- package/metro/index.cjs +72 -0
- package/metro/index.d.ts +6 -0
- package/metro/transformer.cjs +70 -0
- package/package.json +61 -0
- package/runtime/cpp/lucent/abort.cpp +61 -0
- package/runtime/cpp/lucent/abort.h +54 -0
- package/runtime/cpp/lucent/array.h +442 -0
- package/runtime/cpp/lucent/async.h +236 -0
- package/runtime/cpp/lucent/builtins.cpp +163 -0
- package/runtime/cpp/lucent/bytes.h +129 -0
- package/runtime/cpp/lucent/console.h +17 -0
- package/runtime/cpp/lucent/core.h +129 -0
- package/runtime/cpp/lucent/date.cpp +441 -0
- package/runtime/cpp/lucent/date.h +82 -0
- package/runtime/cpp/lucent/equality.h +87 -0
- package/runtime/cpp/lucent/function.h +61 -0
- package/runtime/cpp/lucent/generator.h +244 -0
- package/runtime/cpp/lucent/helpers.h +226 -0
- package/runtime/cpp/lucent/jserror.cpp +51 -0
- package/runtime/cpp/lucent/jserror.h +66 -0
- package/runtime/cpp/lucent/jsi/convert.cpp +202 -0
- package/runtime/cpp/lucent/jsi/convert.h +612 -0
- package/runtime/cpp/lucent/jsi/host.cpp +322 -0
- package/runtime/cpp/lucent/jsi/host.h +166 -0
- package/runtime/cpp/lucent/json.cpp +243 -0
- package/runtime/cpp/lucent/json.h +164 -0
- package/runtime/cpp/lucent/json_parse.h +138 -0
- package/runtime/cpp/lucent/jsstring.cpp +874 -0
- package/runtime/cpp/lucent/jsstring.h +152 -0
- package/runtime/cpp/lucent/lucent.h +24 -0
- package/runtime/cpp/lucent/map.h +344 -0
- package/runtime/cpp/lucent/native.cpp +76 -0
- package/runtime/cpp/lucent/native.h +116 -0
- package/runtime/cpp/lucent/number.cpp +545 -0
- package/runtime/cpp/lucent/number.h +190 -0
- package/runtime/cpp/lucent/ops.h +165 -0
- package/runtime/cpp/lucent/platform/android.cpp +437 -0
- package/runtime/cpp/lucent/platform/android.h +117 -0
- package/runtime/cpp/lucent/platform/ios.h +297 -0
- package/runtime/cpp/lucent/platform/ios.mm +31 -0
- package/runtime/cpp/lucent/regexp.cpp +440 -0
- package/runtime/cpp/lucent/regexp.h +107 -0
- package/runtime/cpp/lucent/scheduler.cpp +153 -0
- package/runtime/cpp/lucent/scheduler.h +160 -0
- package/runtime/cpp/lucent/unicode_data.inc +4051 -0
- package/runtime/cpp/rn/LucentModule.cpp +35 -0
- package/runtime/cpp/rn/LucentModule.h +37 -0
- package/runtime/cpp/third_party/quickjs/LICENSE +22 -0
- package/runtime/cpp/third_party/quickjs/README.md +11 -0
- package/runtime/cpp/third_party/quickjs/cutils.c +641 -0
- package/runtime/cpp/third_party/quickjs/cutils.h +457 -0
- package/runtime/cpp/third_party/quickjs/libregexp-opcode.h +73 -0
- package/runtime/cpp/third_party/quickjs/libregexp.c +3448 -0
- package/runtime/cpp/third_party/quickjs/libregexp.h +65 -0
- package/runtime/cpp/third_party/quickjs/libunicode-table.h +5206 -0
- package/runtime/cpp/third_party/quickjs/libunicode.c +2124 -0
- package/runtime/cpp/third_party/quickjs/libunicode.h +196 -0
- package/runtime/js/index.d.ts +7 -0
- package/runtime/js/index.js +60 -0
- package/runtime/native/LucentNative.podspec +28 -0
- package/runtime/native/android/CMakeLists.txt +51 -0
- package/runtime/native/android/build.gradle +22 -0
- package/runtime/native/android/include/lucentnative.h +16 -0
- package/runtime/native/android/src/main/java/dev/lucent/LucentPackage.java +25 -0
- package/runtime/native/android/src/main/java/dev/lucent/NativeProxy.java +132 -0
- package/runtime/native/ios/LucentRegistration.mm +23 -0
- package/runtime/native/package.json +6 -0
- package/runtime/native/react-native.config.js +19 -0
- package/runtime/test/jsi/harness.cpp +181 -0
- package/schemas/build.schema.json +152 -0
- package/schemas/check.schema.json +74 -0
- package/schemas/doctor.schema.json +40 -0
- package/schemas/lucent.schema.json +43 -0
- package/schemas/sdk-prefetch.schema.json +37 -0
- package/schemas/sdk-search.schema.json +54 -0
- package/ts-plugin/index.d.ts +4 -0
- package/ts-plugin/index.js +99 -0
|
@@ -0,0 +1,874 @@
|
|
|
1
|
+
#include "jsstring.h"
|
|
2
|
+
|
|
3
|
+
#if defined(__APPLE__) && !defined(LUCENT_PORTABLE_COLLATION)
|
|
4
|
+
#include <CoreFoundation/CoreFoundation.h>
|
|
5
|
+
#elif defined(__ANDROID__) && !defined(LUCENT_PORTABLE_COLLATION)
|
|
6
|
+
#include <fbjni/fbjni.h>
|
|
7
|
+
#endif
|
|
8
|
+
|
|
9
|
+
#include <algorithm>
|
|
10
|
+
#include <cctype>
|
|
11
|
+
#include <array>
|
|
12
|
+
#include <cmath>
|
|
13
|
+
#include <cstring>
|
|
14
|
+
#include <vector>
|
|
15
|
+
|
|
16
|
+
#include "number.h"
|
|
17
|
+
|
|
18
|
+
namespace lucent {
|
|
19
|
+
|
|
20
|
+
namespace {
|
|
21
|
+
|
|
22
|
+
// Clamp a JS relative index (ToIntegerOrInfinity + clamping) into [0, len].
|
|
23
|
+
size_t clampIndex(double v, size_t len) {
|
|
24
|
+
if (std::isnan(v)) return 0;
|
|
25
|
+
v = std::trunc(v);
|
|
26
|
+
if (v < 0) {
|
|
27
|
+
v += static_cast<double>(len);
|
|
28
|
+
return v < 0 ? 0 : static_cast<size_t>(v);
|
|
29
|
+
}
|
|
30
|
+
return v > static_cast<double>(len) ? len : static_cast<size_t>(v);
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
// ToIntegerOrInfinity clamped to [0, len] without negative wrap-around.
|
|
34
|
+
size_t clampPositive(double v, size_t len) {
|
|
35
|
+
if (std::isnan(v) || v <= 0) return 0;
|
|
36
|
+
v = std::trunc(v);
|
|
37
|
+
return v > static_cast<double>(len) ? len : static_cast<size_t>(v);
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
void appendUtf8(std::string& out, uint32_t cp) {
|
|
41
|
+
if (cp < 0x80) {
|
|
42
|
+
out.push_back(static_cast<char>(cp));
|
|
43
|
+
} else if (cp < 0x800) {
|
|
44
|
+
out.push_back(static_cast<char>(0xC0 | (cp >> 6)));
|
|
45
|
+
out.push_back(static_cast<char>(0x80 | (cp & 0x3F)));
|
|
46
|
+
} else if (cp < 0x10000) {
|
|
47
|
+
out.push_back(static_cast<char>(0xE0 | (cp >> 12)));
|
|
48
|
+
out.push_back(static_cast<char>(0x80 | ((cp >> 6) & 0x3F)));
|
|
49
|
+
out.push_back(static_cast<char>(0x80 | (cp & 0x3F)));
|
|
50
|
+
} else {
|
|
51
|
+
out.push_back(static_cast<char>(0xF0 | (cp >> 18)));
|
|
52
|
+
out.push_back(static_cast<char>(0x80 | ((cp >> 12) & 0x3F)));
|
|
53
|
+
out.push_back(static_cast<char>(0x80 | ((cp >> 6) & 0x3F)));
|
|
54
|
+
out.push_back(static_cast<char>(0x80 | (cp & 0x3F)));
|
|
55
|
+
}
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
// Simple (1:1) case mappings for the scripts most text uses. Code points
|
|
59
|
+
// outside these ranges map to themselves.
|
|
60
|
+
char16_t upperOf(char16_t c) {
|
|
61
|
+
if (c < 0x80) return (c >= 'a' && c <= 'z') ? c - 32 : c;
|
|
62
|
+
if (c == 0xB5) return 0x39C;
|
|
63
|
+
if (c >= 0xE0 && c <= 0xFE && c != 0xF7) return c - 32;
|
|
64
|
+
if (c == 0xFF) return 0x178;
|
|
65
|
+
if (c >= 0x100 && c <= 0x17F) {
|
|
66
|
+
if (c == 0x131) return 'I';
|
|
67
|
+
if (c == 0x17F) return 'S';
|
|
68
|
+
if ((c >= 0x139 && c <= 0x148) || (c >= 0x179 && c <= 0x17E)) return (c % 2 == 0) ? c - 1 : c;
|
|
69
|
+
if (c == 0x130 || c == 0x138 || c == 0x149 || c == 0x178) return c;
|
|
70
|
+
return (c % 2 == 1) ? c - 1 : c;
|
|
71
|
+
}
|
|
72
|
+
if (c >= 0x3B1 && c <= 0x3C9) return c == 0x3C2 ? 0x3A3 : c - 32;
|
|
73
|
+
if (c >= 0x3AC && c <= 0x3AF) {
|
|
74
|
+
static const char16_t map[] = {0x386, 0x388, 0x389, 0x38A};
|
|
75
|
+
return map[c - 0x3AC];
|
|
76
|
+
}
|
|
77
|
+
if (c >= 0x430 && c <= 0x44F) return c - 32;
|
|
78
|
+
if (c >= 0x450 && c <= 0x45F) return c - 80;
|
|
79
|
+
return c;
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
char16_t lowerOf(char16_t c) {
|
|
83
|
+
if (c < 0x80) return (c >= 'A' && c <= 'Z') ? c + 32 : c;
|
|
84
|
+
if (c >= 0xC0 && c <= 0xDE && c != 0xD7) return c + 32;
|
|
85
|
+
if (c == 0x178) return 0xFF;
|
|
86
|
+
if (c >= 0x100 && c <= 0x17F) {
|
|
87
|
+
if (c == 0x130) return 'i';
|
|
88
|
+
if ((c >= 0x139 && c <= 0x148) || (c >= 0x179 && c <= 0x17E)) return (c % 2 == 1) ? c + 1 : c;
|
|
89
|
+
if (c == 0x131 || c == 0x138 || c == 0x149 || c == 0x17F) return c;
|
|
90
|
+
return (c % 2 == 0) ? c + 1 : c;
|
|
91
|
+
}
|
|
92
|
+
if (c >= 0x391 && c <= 0x3A9 && c != 0x3A2) return c + 32;
|
|
93
|
+
if (c >= 0x386 && c <= 0x38A) {
|
|
94
|
+
switch (c) {
|
|
95
|
+
case 0x386: return 0x3AC;
|
|
96
|
+
case 0x388: return 0x3AD;
|
|
97
|
+
case 0x389: return 0x3AE;
|
|
98
|
+
case 0x38A: return 0x3AF;
|
|
99
|
+
default: return c;
|
|
100
|
+
}
|
|
101
|
+
}
|
|
102
|
+
if (c >= 0x410 && c <= 0x42F) return c + 32;
|
|
103
|
+
if (c >= 0x400 && c <= 0x40F) return c + 80;
|
|
104
|
+
return c;
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
} // namespace
|
|
108
|
+
|
|
109
|
+
bool isJsWhitespace(char16_t c) {
|
|
110
|
+
switch (c) {
|
|
111
|
+
case 0x09: case 0x0A: case 0x0B: case 0x0C: case 0x0D: case 0x20: case 0xA0:
|
|
112
|
+
case 0x1680: case 0x2028: case 0x2029: case 0x202F: case 0x205F: case 0x3000: case 0xFEFF:
|
|
113
|
+
return true;
|
|
114
|
+
default:
|
|
115
|
+
return c >= 0x2000 && c <= 0x200A;
|
|
116
|
+
}
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
String String::make(std::string&& latin1) {
|
|
120
|
+
if (latin1.empty()) return String();
|
|
121
|
+
if (latin1.size() <= kInline) return inlined(latin1, {});
|
|
122
|
+
auto d = std::make_shared<Data>();
|
|
123
|
+
d->oneByte = true;
|
|
124
|
+
d->bytes = std::move(latin1);
|
|
125
|
+
return String(std::move(d));
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
String String::make(std::u16string&& wide) {
|
|
129
|
+
if (wide.empty()) return String();
|
|
130
|
+
bool narrow = true;
|
|
131
|
+
for (char16_t c : wide) {
|
|
132
|
+
if (c > 0xFF) {
|
|
133
|
+
narrow = false;
|
|
134
|
+
break;
|
|
135
|
+
}
|
|
136
|
+
}
|
|
137
|
+
if (narrow) {
|
|
138
|
+
std::string bytes(wide.size(), '\0');
|
|
139
|
+
for (size_t i = 0; i < wide.size(); i++) bytes[i] = static_cast<char>(wide[i]);
|
|
140
|
+
return make(std::move(bytes));
|
|
141
|
+
}
|
|
142
|
+
auto d = std::make_shared<Data>();
|
|
143
|
+
d->oneByte = false;
|
|
144
|
+
d->wide = std::move(wide);
|
|
145
|
+
return String(std::move(d));
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
String String::fromLatin1(std::string_view bytes) {
|
|
149
|
+
if (bytes.empty()) return String();
|
|
150
|
+
if (bytes.size() <= kInline) return inlined(bytes, {});
|
|
151
|
+
return make(std::string(bytes));
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
namespace {
|
|
155
|
+
// Copies n <= 16 bytes as two overlapping fixed-size copies, which compile
|
|
156
|
+
// to loads and stores: memcpy with a variable size is a library call, and
|
|
157
|
+
// short strings cross the boundary on every call. Reads no byte past n.
|
|
158
|
+
inline void copySmall(char* dst, const char* src, size_t n) {
|
|
159
|
+
if (n >= 8) {
|
|
160
|
+
uint64_t x, y;
|
|
161
|
+
std::memcpy(&x, src, 8);
|
|
162
|
+
std::memcpy(&y, src + n - 8, 8);
|
|
163
|
+
std::memcpy(dst, &x, 8);
|
|
164
|
+
std::memcpy(dst + n - 8, &y, 8);
|
|
165
|
+
} else if (n >= 4) {
|
|
166
|
+
uint32_t x, y;
|
|
167
|
+
std::memcpy(&x, src, 4);
|
|
168
|
+
std::memcpy(&y, src + n - 4, 4);
|
|
169
|
+
std::memcpy(dst, &x, 4);
|
|
170
|
+
std::memcpy(dst + n - 4, &y, 4);
|
|
171
|
+
} else {
|
|
172
|
+
for (size_t i = 0; i < n; i++) dst[i] = src[i];
|
|
173
|
+
}
|
|
174
|
+
}
|
|
175
|
+
} // namespace
|
|
176
|
+
|
|
177
|
+
String String::inlined(std::string_view a, std::string_view b) {
|
|
178
|
+
String out;
|
|
179
|
+
out.n_ = static_cast<uint8_t>(a.size() + b.size());
|
|
180
|
+
copySmall(out.s_, a.data(), a.size());
|
|
181
|
+
copySmall(out.s_ + a.size(), b.data(), b.size());
|
|
182
|
+
return out;
|
|
183
|
+
}
|
|
184
|
+
|
|
185
|
+
String String::fromUtf8(std::string_view s) {
|
|
186
|
+
bool ascii = true;
|
|
187
|
+
for (unsigned char c : s) {
|
|
188
|
+
if (c >= 0x80) {
|
|
189
|
+
ascii = false;
|
|
190
|
+
break;
|
|
191
|
+
}
|
|
192
|
+
}
|
|
193
|
+
if (ascii) return make(std::string(s));
|
|
194
|
+
std::u16string out;
|
|
195
|
+
out.reserve(s.size());
|
|
196
|
+
size_t i = 0;
|
|
197
|
+
const size_t n = s.size();
|
|
198
|
+
auto cont = [&](size_t k) -> int {
|
|
199
|
+
if (i + k >= n) return -1;
|
|
200
|
+
unsigned char c = static_cast<unsigned char>(s[i + k]);
|
|
201
|
+
return (c & 0xC0) == 0x80 ? (c & 0x3F) : -1;
|
|
202
|
+
};
|
|
203
|
+
while (i < n) {
|
|
204
|
+
unsigned char c = static_cast<unsigned char>(s[i]);
|
|
205
|
+
uint32_t cp = 0xFFFD;
|
|
206
|
+
size_t len = 1;
|
|
207
|
+
if (c < 0x80) {
|
|
208
|
+
cp = c;
|
|
209
|
+
} else if ((c & 0xE0) == 0xC0) {
|
|
210
|
+
int a = cont(1);
|
|
211
|
+
if (a >= 0 && c >= 0xC2) {
|
|
212
|
+
cp = ((c & 0x1F) << 6) | a;
|
|
213
|
+
len = 2;
|
|
214
|
+
}
|
|
215
|
+
} else if ((c & 0xF0) == 0xE0) {
|
|
216
|
+
int a = cont(1), b = cont(2);
|
|
217
|
+
if (a >= 0 && b >= 0) {
|
|
218
|
+
uint32_t v = ((c & 0x0F) << 12) | (a << 6) | b;
|
|
219
|
+
if (v >= 0x800) {
|
|
220
|
+
cp = v;
|
|
221
|
+
len = 3;
|
|
222
|
+
}
|
|
223
|
+
}
|
|
224
|
+
} else if ((c & 0xF8) == 0xF0) {
|
|
225
|
+
int a = cont(1), b = cont(2), d = cont(3);
|
|
226
|
+
if (a >= 0 && b >= 0 && d >= 0) {
|
|
227
|
+
uint32_t v = ((c & 0x07) << 18) | (a << 12) | (b << 6) | d;
|
|
228
|
+
if (v >= 0x10000 && v <= 0x10FFFF) {
|
|
229
|
+
cp = v;
|
|
230
|
+
len = 4;
|
|
231
|
+
}
|
|
232
|
+
}
|
|
233
|
+
}
|
|
234
|
+
if (cp >= 0x10000) {
|
|
235
|
+
cp -= 0x10000;
|
|
236
|
+
out.push_back(static_cast<char16_t>(0xD800 + (cp >> 10)));
|
|
237
|
+
out.push_back(static_cast<char16_t>(0xDC00 + (cp & 0x3FF)));
|
|
238
|
+
} else {
|
|
239
|
+
out.push_back(static_cast<char16_t>(cp));
|
|
240
|
+
}
|
|
241
|
+
i += len;
|
|
242
|
+
}
|
|
243
|
+
return make(std::move(out));
|
|
244
|
+
}
|
|
245
|
+
|
|
246
|
+
String String::fromUtf16(const char16_t* units, size_t length) { return make(std::u16string(units, length)); }
|
|
247
|
+
|
|
248
|
+
String String::fromCodeUnit(char16_t unit) {
|
|
249
|
+
if (unit <= 0xFF) {
|
|
250
|
+
// Shared, like JavaScript engines' single-character strings. `+=` never
|
|
251
|
+
// grows a shared string in place, so the cache cannot change.
|
|
252
|
+
static const auto* cache = [] {
|
|
253
|
+
auto* c = new std::array<String, 256>();
|
|
254
|
+
for (int i = 0; i < 256; i++) (*c)[i] = make(std::string(1, static_cast<char>(i)));
|
|
255
|
+
return c;
|
|
256
|
+
}();
|
|
257
|
+
return (*cache)[unit];
|
|
258
|
+
}
|
|
259
|
+
return make(std::u16string(1, unit));
|
|
260
|
+
}
|
|
261
|
+
|
|
262
|
+
String String::fromCodePoint(double v) {
|
|
263
|
+
if (!(v >= 0 && v <= 0x10FFFF) || std::trunc(v) != v) throwRangeError("Invalid code point");
|
|
264
|
+
uint32_t cp = static_cast<uint32_t>(v);
|
|
265
|
+
if (cp < 0x10000) return fromCodeUnit(static_cast<char16_t>(cp));
|
|
266
|
+
cp -= 0x10000;
|
|
267
|
+
char16_t pair[2] = {static_cast<char16_t>(0xD800 + (cp >> 10)), static_cast<char16_t>(0xDC00 + (cp & 0x3FF))};
|
|
268
|
+
return fromUtf16(pair, 2);
|
|
269
|
+
}
|
|
270
|
+
|
|
271
|
+
std::string String::toUtf8() const {
|
|
272
|
+
std::string out;
|
|
273
|
+
if (isOneByte()) {
|
|
274
|
+
std::string_view b = latin1();
|
|
275
|
+
out.reserve(b.size());
|
|
276
|
+
for (unsigned char c : b) appendUtf8(out, c);
|
|
277
|
+
return out;
|
|
278
|
+
}
|
|
279
|
+
const auto& w = d_->wide;
|
|
280
|
+
out.reserve(w.size() * 2);
|
|
281
|
+
for (size_t i = 0; i < w.size(); i++) {
|
|
282
|
+
char16_t c = w[i];
|
|
283
|
+
if (c >= 0xD800 && c <= 0xDBFF && i + 1 < w.size() && w[i + 1] >= 0xDC00 && w[i + 1] <= 0xDFFF) {
|
|
284
|
+
uint32_t cp = 0x10000 + ((c - 0xD800) << 10) + (w[i + 1] - 0xDC00);
|
|
285
|
+
appendUtf8(out, cp);
|
|
286
|
+
i++;
|
|
287
|
+
} else if (c >= 0xD800 && c <= 0xDFFF) {
|
|
288
|
+
appendUtf8(out, 0xFFFD);
|
|
289
|
+
} else {
|
|
290
|
+
appendUtf8(out, c);
|
|
291
|
+
}
|
|
292
|
+
}
|
|
293
|
+
return out;
|
|
294
|
+
}
|
|
295
|
+
|
|
296
|
+
std::u16string String::toUtf16() const {
|
|
297
|
+
std::u16string out;
|
|
298
|
+
appendUnitsTo(out);
|
|
299
|
+
return out;
|
|
300
|
+
}
|
|
301
|
+
|
|
302
|
+
void String::appendUnitsTo(std::u16string& out) const {
|
|
303
|
+
if (isOneByte()) {
|
|
304
|
+
std::string_view b = latin1();
|
|
305
|
+
out.reserve(out.size() + b.size());
|
|
306
|
+
for (unsigned char c : b) out.push_back(c);
|
|
307
|
+
} else {
|
|
308
|
+
out.append(d_->wide);
|
|
309
|
+
}
|
|
310
|
+
}
|
|
311
|
+
|
|
312
|
+
String operator+(const String& a, const String& b) {
|
|
313
|
+
if (a.empty()) return b;
|
|
314
|
+
if (b.empty()) return a;
|
|
315
|
+
if (a.isOneByte() && b.isOneByte()) {
|
|
316
|
+
std::string_view x = a.latin1(), y = b.latin1();
|
|
317
|
+
if (x.size() + y.size() <= String::kInline) return String::inlined(x, y);
|
|
318
|
+
std::string s;
|
|
319
|
+
s.reserve(x.size() + y.size());
|
|
320
|
+
s.append(x).append(y);
|
|
321
|
+
return String::make(std::move(s));
|
|
322
|
+
}
|
|
323
|
+
std::u16string w;
|
|
324
|
+
w.reserve(a.length() + b.length());
|
|
325
|
+
a.appendUnitsTo(w);
|
|
326
|
+
b.appendUnitsTo(w);
|
|
327
|
+
auto d = std::make_shared<String::Data>();
|
|
328
|
+
d->oneByte = false;
|
|
329
|
+
d->wide = std::move(w);
|
|
330
|
+
return String(std::move(d));
|
|
331
|
+
}
|
|
332
|
+
|
|
333
|
+
StringBuilder::StringBuilder(size_t capacity, bool oneByte) : oneByte_(oneByte) {
|
|
334
|
+
if (oneByte) bytes_.reserve(capacity);
|
|
335
|
+
else wide_.reserve(capacity);
|
|
336
|
+
}
|
|
337
|
+
|
|
338
|
+
void StringBuilder::append(const String& s) {
|
|
339
|
+
if (s.empty()) return;
|
|
340
|
+
if (oneByte_ && s.isOneByte()) {
|
|
341
|
+
bytes_.append(s.latin1());
|
|
342
|
+
return;
|
|
343
|
+
}
|
|
344
|
+
if (oneByte_) {
|
|
345
|
+
// Parts were promised Latin-1; widen what was built so far.
|
|
346
|
+
wide_.reserve(bytes_.size() + s.length());
|
|
347
|
+
for (unsigned char b : bytes_) wide_.push_back(b);
|
|
348
|
+
bytes_.clear();
|
|
349
|
+
oneByte_ = false;
|
|
350
|
+
}
|
|
351
|
+
s.appendUnitsTo(wide_);
|
|
352
|
+
}
|
|
353
|
+
|
|
354
|
+
void StringBuilder::appendAscii(std::string_view ascii) {
|
|
355
|
+
if (oneByte_) bytes_.append(ascii);
|
|
356
|
+
else wide_.append(ascii.begin(), ascii.end());
|
|
357
|
+
}
|
|
358
|
+
|
|
359
|
+
String StringBuilder::build() && {
|
|
360
|
+
if (oneByte_) return String::make(std::move(bytes_));
|
|
361
|
+
return String::make(std::move(wide_));
|
|
362
|
+
}
|
|
363
|
+
|
|
364
|
+
String& String::operator+=(const String& other) {
|
|
365
|
+
if (other.empty()) return *this;
|
|
366
|
+
if (empty()) {
|
|
367
|
+
*this = other;
|
|
368
|
+
return *this;
|
|
369
|
+
}
|
|
370
|
+
if (d_ && d_.use_count() == 1 && &other != this) {
|
|
371
|
+
// Sole owner: grow in place so a `+=` loop is amortized linear. (Inline
|
|
372
|
+
// strings are short: they grow by concatenation until they move out.)
|
|
373
|
+
d_->hash.store(0, std::memory_order_relaxed);
|
|
374
|
+
if (d_->oneByte && other.isOneByte()) {
|
|
375
|
+
d_->bytes.append(other.latin1());
|
|
376
|
+
return *this;
|
|
377
|
+
}
|
|
378
|
+
if (d_->oneByte) {
|
|
379
|
+
std::u16string w;
|
|
380
|
+
w.reserve((d_->bytes.size() + other.length()) * 2);
|
|
381
|
+
for (unsigned char c : d_->bytes) w.push_back(c);
|
|
382
|
+
d_->bytes.clear();
|
|
383
|
+
d_->bytes.shrink_to_fit();
|
|
384
|
+
d_->oneByte = false;
|
|
385
|
+
d_->wide = std::move(w);
|
|
386
|
+
}
|
|
387
|
+
other.appendUnitsTo(d_->wide);
|
|
388
|
+
return *this;
|
|
389
|
+
}
|
|
390
|
+
*this = *this + other;
|
|
391
|
+
return *this;
|
|
392
|
+
}
|
|
393
|
+
|
|
394
|
+
bool operator==(const String& a, const String& b) {
|
|
395
|
+
if (a.d_ && a.d_ == b.d_) return true;
|
|
396
|
+
size_t n = a.length();
|
|
397
|
+
if (n != b.length()) return false;
|
|
398
|
+
if (n == 0) return true;
|
|
399
|
+
if (a.isOneByte() && b.isOneByte()) return a.latin1() == b.latin1();
|
|
400
|
+
if (!a.d_->oneByte && !b.d_->oneByte) return a.d_->wide == b.d_->wide;
|
|
401
|
+
// Mixed representations never compare equal: a wide string always holds a
|
|
402
|
+
// unit > 0xFF (make() narrows otherwise).
|
|
403
|
+
return false;
|
|
404
|
+
}
|
|
405
|
+
|
|
406
|
+
int String::compare(const String& a, const String& b) {
|
|
407
|
+
size_t n = std::min(a.length(), b.length());
|
|
408
|
+
for (size_t i = 0; i < n; i++) {
|
|
409
|
+
char16_t x = a.unit(i), y = b.unit(i);
|
|
410
|
+
if (x != y) return x < y ? -1 : 1;
|
|
411
|
+
}
|
|
412
|
+
if (a.length() == b.length()) return 0;
|
|
413
|
+
return a.length() < b.length() ? -1 : 1;
|
|
414
|
+
}
|
|
415
|
+
|
|
416
|
+
size_t String::hash() const {
|
|
417
|
+
if (empty()) return 0x9e3779b9;
|
|
418
|
+
size_t h = d_ ? d_->hash.load(std::memory_order_relaxed) : 0;
|
|
419
|
+
if (h != 0) return h;
|
|
420
|
+
// FNV-1a over code units, so both representations hash alike.
|
|
421
|
+
uint64_t x = 1469598103934665603ULL;
|
|
422
|
+
size_t n = length();
|
|
423
|
+
for (size_t i = 0; i < n; i++) {
|
|
424
|
+
x ^= unit(i);
|
|
425
|
+
x *= 1099511628211ULL;
|
|
426
|
+
}
|
|
427
|
+
h = static_cast<size_t>(x) | 1;
|
|
428
|
+
if (d_) d_->hash.store(h, std::memory_order_relaxed);
|
|
429
|
+
return h;
|
|
430
|
+
}
|
|
431
|
+
|
|
432
|
+
String String::sub(size_t begin, size_t end) const {
|
|
433
|
+
size_t n = length();
|
|
434
|
+
if (end > n) end = n;
|
|
435
|
+
if (begin >= end) return String();
|
|
436
|
+
if (begin == 0 && end == n) return *this;
|
|
437
|
+
if (isOneByte()) return make(std::string(latin1().substr(begin, end - begin)));
|
|
438
|
+
return make(d_->wide.substr(begin, end - begin));
|
|
439
|
+
}
|
|
440
|
+
|
|
441
|
+
size_t String::find(const String& needle, size_t from) const {
|
|
442
|
+
size_t n = length(), m = needle.length();
|
|
443
|
+
if (m == 0) return from <= n ? from : std::string::npos;
|
|
444
|
+
if (m > n) return std::string::npos;
|
|
445
|
+
if (isOneByte() && needle.isOneByte()) {
|
|
446
|
+
return latin1().find(needle.latin1(), from);
|
|
447
|
+
}
|
|
448
|
+
for (size_t i = from; i + m <= n; i++) {
|
|
449
|
+
size_t j = 0;
|
|
450
|
+
while (j < m && unit(i + j) == needle.unit(j)) j++;
|
|
451
|
+
if (j == m) return i;
|
|
452
|
+
}
|
|
453
|
+
return std::string::npos;
|
|
454
|
+
}
|
|
455
|
+
|
|
456
|
+
double String::charCodeAt(double index) const {
|
|
457
|
+
if (std::isnan(index)) index = 0;
|
|
458
|
+
index = std::trunc(index);
|
|
459
|
+
if (index < 0 || index >= static_cast<double>(length())) return std::nan("");
|
|
460
|
+
return unit(static_cast<size_t>(index));
|
|
461
|
+
}
|
|
462
|
+
|
|
463
|
+
String String::charAt(double index) const {
|
|
464
|
+
if (std::isnan(index)) index = 0;
|
|
465
|
+
index = std::trunc(index);
|
|
466
|
+
if (index < 0 || index >= static_cast<double>(length())) return String();
|
|
467
|
+
return fromCodeUnit(unit(static_cast<size_t>(index)));
|
|
468
|
+
}
|
|
469
|
+
|
|
470
|
+
Opt<String> String::at(double index) const {
|
|
471
|
+
if (std::isnan(index)) index = 0;
|
|
472
|
+
index = std::trunc(index);
|
|
473
|
+
double n = static_cast<double>(length());
|
|
474
|
+
if (index < 0) index += n;
|
|
475
|
+
if (index < 0 || index >= n) return undefined;
|
|
476
|
+
return fromCodeUnit(unit(static_cast<size_t>(index)));
|
|
477
|
+
}
|
|
478
|
+
|
|
479
|
+
Opt<double> String::codePointAt(double index) const {
|
|
480
|
+
if (std::isnan(index)) index = 0;
|
|
481
|
+
index = std::trunc(index);
|
|
482
|
+
if (index < 0 || index >= static_cast<double>(length())) return undefined;
|
|
483
|
+
size_t i = static_cast<size_t>(index);
|
|
484
|
+
char16_t c = unit(i);
|
|
485
|
+
if (c >= 0xD800 && c <= 0xDBFF && i + 1 < length()) {
|
|
486
|
+
char16_t d = unit(i + 1);
|
|
487
|
+
if (d >= 0xDC00 && d <= 0xDFFF) return static_cast<double>(0x10000 + ((c - 0xD800) << 10) + (d - 0xDC00));
|
|
488
|
+
}
|
|
489
|
+
return static_cast<double>(c);
|
|
490
|
+
}
|
|
491
|
+
|
|
492
|
+
double String::indexOf(const String& search, double from) const {
|
|
493
|
+
size_t start = clampPositive(from, length());
|
|
494
|
+
size_t r = find(search, start);
|
|
495
|
+
return r == std::string::npos ? -1 : static_cast<double>(r);
|
|
496
|
+
}
|
|
497
|
+
|
|
498
|
+
double String::lastIndexOf(const String& search) const {
|
|
499
|
+
return lastIndexOf(search, static_cast<double>(length()));
|
|
500
|
+
}
|
|
501
|
+
|
|
502
|
+
double String::lastIndexOf(const String& search, double from) const {
|
|
503
|
+
size_t n = length(), m = search.length();
|
|
504
|
+
if (m > n) return -1;
|
|
505
|
+
size_t start = std::isnan(from) ? n - m : std::min(clampPositive(from, n), n - m);
|
|
506
|
+
for (size_t i = start + 1; i-- > 0;) {
|
|
507
|
+
size_t j = 0;
|
|
508
|
+
while (j < m && unit(i + j) == search.unit(j)) j++;
|
|
509
|
+
if (j == m) return static_cast<double>(i);
|
|
510
|
+
}
|
|
511
|
+
return -1;
|
|
512
|
+
}
|
|
513
|
+
|
|
514
|
+
bool String::startsWith(const String& search, double from) const {
|
|
515
|
+
size_t start = clampPositive(from, length());
|
|
516
|
+
size_t m = search.length();
|
|
517
|
+
if (start + m > length()) return false;
|
|
518
|
+
for (size_t j = 0; j < m; j++) {
|
|
519
|
+
if (unit(start + j) != search.unit(j)) return false;
|
|
520
|
+
}
|
|
521
|
+
return true;
|
|
522
|
+
}
|
|
523
|
+
|
|
524
|
+
bool String::endsWith(const String& search) const { return endsWith(search, static_cast<double>(length())); }
|
|
525
|
+
|
|
526
|
+
bool String::endsWith(const String& search, double endPosition) const {
|
|
527
|
+
size_t end = clampPositive(endPosition, length());
|
|
528
|
+
size_t m = search.length();
|
|
529
|
+
if (m > end) return false;
|
|
530
|
+
size_t start = end - m;
|
|
531
|
+
for (size_t j = 0; j < m; j++) {
|
|
532
|
+
if (unit(start + j) != search.unit(j)) return false;
|
|
533
|
+
}
|
|
534
|
+
return true;
|
|
535
|
+
}
|
|
536
|
+
|
|
537
|
+
String String::slice(double start) const { return sub(clampIndex(start, length()), length()); }
|
|
538
|
+
|
|
539
|
+
String String::slice(double start, double end) const {
|
|
540
|
+
return sub(clampIndex(start, length()), clampIndex(end, length()));
|
|
541
|
+
}
|
|
542
|
+
|
|
543
|
+
String String::substring(double start) const { return sub(clampPositive(start, length()), length()); }
|
|
544
|
+
|
|
545
|
+
String String::substring(double start, double end) const {
|
|
546
|
+
size_t a = clampPositive(start, length()), b = clampPositive(end, length());
|
|
547
|
+
return a <= b ? sub(a, b) : sub(b, a);
|
|
548
|
+
}
|
|
549
|
+
|
|
550
|
+
namespace {
|
|
551
|
+
|
|
552
|
+
#include "unicode_data.inc"
|
|
553
|
+
|
|
554
|
+
template <class Table, size_t N>
|
|
555
|
+
const Table* findByCodePoint(const Table (&table)[N], uint32_t cp) {
|
|
556
|
+
size_t lo = 0, hi = N;
|
|
557
|
+
while (lo < hi) {
|
|
558
|
+
size_t mid = (lo + hi) / 2;
|
|
559
|
+
if (table[mid].cp < cp) lo = mid + 1;
|
|
560
|
+
else hi = mid;
|
|
561
|
+
}
|
|
562
|
+
return lo < N && table[lo].cp == cp ? &table[lo] : nullptr;
|
|
563
|
+
}
|
|
564
|
+
|
|
565
|
+
template <size_t N>
|
|
566
|
+
bool inRanges(const CodePointRange (&table)[N], uint32_t cp) {
|
|
567
|
+
size_t lo = 0, hi = N;
|
|
568
|
+
while (lo < hi) {
|
|
569
|
+
size_t mid = (lo + hi) / 2;
|
|
570
|
+
if (table[mid].last < cp) lo = mid + 1;
|
|
571
|
+
else hi = mid;
|
|
572
|
+
}
|
|
573
|
+
return lo < N && table[lo].first <= cp;
|
|
574
|
+
}
|
|
575
|
+
|
|
576
|
+
/// The code point at `i` (lone surrogates are code points of their own).
|
|
577
|
+
uint32_t codePointAt(const String& s, size_t i, size_t& width) {
|
|
578
|
+
char16_t c = s.unit(i);
|
|
579
|
+
width = 1;
|
|
580
|
+
if (c >= 0xD800 && c <= 0xDBFF && i + 1 < s.length()) {
|
|
581
|
+
char16_t d = s.unit(i + 1);
|
|
582
|
+
if (d >= 0xDC00 && d <= 0xDFFF) {
|
|
583
|
+
width = 2;
|
|
584
|
+
return 0x10000 + ((c - 0xD800) << 10) + (d - 0xDC00);
|
|
585
|
+
}
|
|
586
|
+
}
|
|
587
|
+
return c;
|
|
588
|
+
}
|
|
589
|
+
|
|
590
|
+
/// The code point that ends before `i`.
|
|
591
|
+
uint32_t codePointBefore(const String& s, size_t i, size_t& width) {
|
|
592
|
+
char16_t c = s.unit(i - 1);
|
|
593
|
+
width = 1;
|
|
594
|
+
if (c >= 0xDC00 && c <= 0xDFFF && i >= 2) {
|
|
595
|
+
char16_t h = s.unit(i - 2);
|
|
596
|
+
if (h >= 0xD800 && h <= 0xDBFF) {
|
|
597
|
+
width = 2;
|
|
598
|
+
return 0x10000 + ((h - 0xD800) << 10) + (c - 0xDC00);
|
|
599
|
+
}
|
|
600
|
+
}
|
|
601
|
+
return c;
|
|
602
|
+
}
|
|
603
|
+
|
|
604
|
+
void appendCodePoint(std::u16string& out, uint32_t cp) {
|
|
605
|
+
if (cp < 0x10000) {
|
|
606
|
+
out.push_back(static_cast<char16_t>(cp));
|
|
607
|
+
} else {
|
|
608
|
+
cp -= 0x10000;
|
|
609
|
+
out.push_back(static_cast<char16_t>(0xD800 + (cp >> 10)));
|
|
610
|
+
out.push_back(static_cast<char16_t>(0xDC00 + (cp & 0x3FF)));
|
|
611
|
+
}
|
|
612
|
+
}
|
|
613
|
+
|
|
614
|
+
/// Unicode's Final_Sigma: a cased letter before (past case-ignorables), none after.
|
|
615
|
+
bool isFinalSigma(const String& s, size_t i) {
|
|
616
|
+
bool casedBefore = false;
|
|
617
|
+
for (size_t j = i; j > 0;) {
|
|
618
|
+
size_t w;
|
|
619
|
+
uint32_t cp = codePointBefore(s, j, w);
|
|
620
|
+
j -= w;
|
|
621
|
+
if (inRanges(kCaseIgnorable, cp)) continue;
|
|
622
|
+
casedBefore = inRanges(kCased, cp);
|
|
623
|
+
break;
|
|
624
|
+
}
|
|
625
|
+
if (!casedBefore) return false;
|
|
626
|
+
for (size_t k = i + 1; k < s.length();) {
|
|
627
|
+
size_t w;
|
|
628
|
+
uint32_t cp = codePointAt(s, k, w);
|
|
629
|
+
k += w;
|
|
630
|
+
if (inRanges(kCaseIgnorable, cp)) continue;
|
|
631
|
+
return !inRanges(kCased, cp);
|
|
632
|
+
}
|
|
633
|
+
return true;
|
|
634
|
+
}
|
|
635
|
+
|
|
636
|
+
template <size_t N>
|
|
637
|
+
String mapCase(const String& s, const CaseMapping (&table)[N], bool lower) {
|
|
638
|
+
std::u16string out;
|
|
639
|
+
out.reserve(s.length());
|
|
640
|
+
for (size_t i = 0; i < s.length();) {
|
|
641
|
+
size_t w;
|
|
642
|
+
uint32_t cp = codePointAt(s, i, w);
|
|
643
|
+
if (lower && cp == 0x03A3) {
|
|
644
|
+
out.push_back(isFinalSigma(s, i) ? u'\u03C2' : u'\u03C3');
|
|
645
|
+
} else if (const CaseMapping* m = findByCodePoint(table, cp)) {
|
|
646
|
+
out.append(m->units, m->length);
|
|
647
|
+
} else {
|
|
648
|
+
appendCodePoint(out, cp);
|
|
649
|
+
}
|
|
650
|
+
i += w;
|
|
651
|
+
}
|
|
652
|
+
return String::fromUtf16(out);
|
|
653
|
+
}
|
|
654
|
+
|
|
655
|
+
} // namespace
|
|
656
|
+
|
|
657
|
+
String String::toUpperCase() const {
|
|
658
|
+
size_t n = length();
|
|
659
|
+
if (n == 0) return *this;
|
|
660
|
+
if (isOneByte()) {
|
|
661
|
+
// Latin-1 maps within Latin-1 except ß (SS), µ (Μ) and ÿ (Ÿ).
|
|
662
|
+
bool special = false;
|
|
663
|
+
std::string out(latin1());
|
|
664
|
+
for (auto& ch : out) {
|
|
665
|
+
unsigned char c = static_cast<unsigned char>(ch);
|
|
666
|
+
if (c == 0xDF || c == 0xB5 || c == 0xFF) {
|
|
667
|
+
special = true;
|
|
668
|
+
break;
|
|
669
|
+
}
|
|
670
|
+
ch = static_cast<char>(upperOf(c));
|
|
671
|
+
}
|
|
672
|
+
if (!special) return make(std::move(out));
|
|
673
|
+
}
|
|
674
|
+
return mapCase(*this, kUpperCase, false);
|
|
675
|
+
}
|
|
676
|
+
|
|
677
|
+
String String::toLowerCase() const {
|
|
678
|
+
size_t n = length();
|
|
679
|
+
if (n == 0) return *this;
|
|
680
|
+
if (isOneByte()) {
|
|
681
|
+
std::string out(latin1());
|
|
682
|
+
for (auto& ch : out) ch = static_cast<char>(lowerOf(static_cast<unsigned char>(ch)));
|
|
683
|
+
return make(std::move(out));
|
|
684
|
+
}
|
|
685
|
+
return mapCase(*this, kLowerCase, true);
|
|
686
|
+
}
|
|
687
|
+
|
|
688
|
+
String String::trim() const {
|
|
689
|
+
size_t b = 0, e = length();
|
|
690
|
+
while (b < e && isJsWhitespace(unit(b))) b++;
|
|
691
|
+
while (e > b && isJsWhitespace(unit(e - 1))) e--;
|
|
692
|
+
return sub(b, e);
|
|
693
|
+
}
|
|
694
|
+
|
|
695
|
+
String String::trimStart() const {
|
|
696
|
+
size_t b = 0, e = length();
|
|
697
|
+
while (b < e && isJsWhitespace(unit(b))) b++;
|
|
698
|
+
return sub(b, e);
|
|
699
|
+
}
|
|
700
|
+
|
|
701
|
+
String String::trimEnd() const {
|
|
702
|
+
size_t e = length();
|
|
703
|
+
while (e > 0 && isJsWhitespace(unit(e - 1))) e--;
|
|
704
|
+
return sub(0, e);
|
|
705
|
+
}
|
|
706
|
+
|
|
707
|
+
String String::repeat(double count) const {
|
|
708
|
+
if (std::isnan(count)) count = 0;
|
|
709
|
+
count = std::trunc(count);
|
|
710
|
+
if (count < 0 || std::isinf(count)) throwRangeError("Invalid count value");
|
|
711
|
+
if (count == 0 || empty()) return String();
|
|
712
|
+
if (count * static_cast<double>(length()) > (1u << 29)) throwRangeError("Invalid string length");
|
|
713
|
+
size_t c = static_cast<size_t>(count);
|
|
714
|
+
if (isOneByte()) {
|
|
715
|
+
std::string out;
|
|
716
|
+
out.reserve(length() * c);
|
|
717
|
+
for (size_t i = 0; i < c; i++) out.append(latin1());
|
|
718
|
+
return make(std::move(out));
|
|
719
|
+
}
|
|
720
|
+
std::u16string out;
|
|
721
|
+
out.reserve(length() * c);
|
|
722
|
+
for (size_t i = 0; i < c; i++) out.append(d_->wide);
|
|
723
|
+
return make(std::move(out));
|
|
724
|
+
}
|
|
725
|
+
|
|
726
|
+
static String pad(const String& s, double target, const String& fill, bool atStart) {
|
|
727
|
+
if (std::isnan(target)) return s;
|
|
728
|
+
target = std::trunc(target);
|
|
729
|
+
if (target <= static_cast<double>(s.length()) || fill.empty()) return s;
|
|
730
|
+
if (target > (1u << 29)) throwRangeError("Invalid string length");
|
|
731
|
+
size_t need = static_cast<size_t>(target) - s.length();
|
|
732
|
+
std::u16string p;
|
|
733
|
+
p.reserve(need);
|
|
734
|
+
while (p.size() < need) {
|
|
735
|
+
for (size_t i = 0; i < fill.length() && p.size() < need; i++) p.push_back(fill.unit(i));
|
|
736
|
+
}
|
|
737
|
+
String padding = String::fromUtf16(p);
|
|
738
|
+
return atStart ? padding + s : s + padding;
|
|
739
|
+
}
|
|
740
|
+
|
|
741
|
+
String String::padStart(double targetLength, const String& fill) const { return pad(*this, targetLength, fill, true); }
|
|
742
|
+
String String::padEnd(double targetLength, const String& fill) const { return pad(*this, targetLength, fill, false); }
|
|
743
|
+
|
|
744
|
+
String String::replace(const String& search, const String& replacement) const {
|
|
745
|
+
size_t at = find(search, 0);
|
|
746
|
+
if (at == std::string::npos) return *this;
|
|
747
|
+
return sub(0, at) + replacement + sub(at + search.length(), length());
|
|
748
|
+
}
|
|
749
|
+
|
|
750
|
+
String String::replaceAll(const String& search, const String& replacement) const {
|
|
751
|
+
std::u16string out;
|
|
752
|
+
size_t n = length(), m = search.length();
|
|
753
|
+
if (m == 0) {
|
|
754
|
+
// "ab".replaceAll("", "-") === "-a-b-"
|
|
755
|
+
for (size_t i = 0; i < n; i++) {
|
|
756
|
+
replacement.appendUnitsTo(out);
|
|
757
|
+
out.push_back(unit(i));
|
|
758
|
+
}
|
|
759
|
+
replacement.appendUnitsTo(out);
|
|
760
|
+
return make(std::move(out));
|
|
761
|
+
}
|
|
762
|
+
size_t pos = 0;
|
|
763
|
+
for (;;) {
|
|
764
|
+
size_t at = find(search, pos);
|
|
765
|
+
if (at == std::string::npos) break;
|
|
766
|
+
for (size_t i = pos; i < at; i++) out.push_back(unit(i));
|
|
767
|
+
replacement.appendUnitsTo(out);
|
|
768
|
+
pos = at + m;
|
|
769
|
+
}
|
|
770
|
+
if (pos == 0) return *this;
|
|
771
|
+
for (size_t i = pos; i < n; i++) out.push_back(unit(i));
|
|
772
|
+
return make(std::move(out));
|
|
773
|
+
}
|
|
774
|
+
|
|
775
|
+
#if defined(__APPLE__) && !defined(LUCENT_PORTABLE_COLLATION)
|
|
776
|
+
|
|
777
|
+
// Like Hermes on Apple platforms: CoreFoundation with the current locale,
|
|
778
|
+
// comparing canonically equivalent strings as equal.
|
|
779
|
+
double String::localeCompare(const String& other) const {
|
|
780
|
+
std::u16string a = toUtf16(), b = other.toUtf16();
|
|
781
|
+
CFStringRef s1 = CFStringCreateWithCharacters(nullptr, reinterpret_cast<const UniChar*>(a.data()), static_cast<CFIndex>(a.size()));
|
|
782
|
+
CFStringRef s2 = CFStringCreateWithCharacters(nullptr, reinterpret_cast<const UniChar*>(b.data()), static_cast<CFIndex>(b.size()));
|
|
783
|
+
CFLocaleRef locale = CFLocaleCopyCurrent();
|
|
784
|
+
CFComparisonResult r = CFStringCompareWithOptionsAndLocale(s1, s2, CFRangeMake(0, CFStringGetLength(s1)), kCFCompareLocalized | kCFCompareNonliteral, locale);
|
|
785
|
+
CFRelease(s1);
|
|
786
|
+
CFRelease(s2);
|
|
787
|
+
CFRelease(locale);
|
|
788
|
+
return r == kCFCompareLessThan ? -1 : r == kCFCompareGreaterThan ? 1 : 0;
|
|
789
|
+
}
|
|
790
|
+
|
|
791
|
+
#elif defined(__ANDROID__) && !defined(LUCENT_PORTABLE_COLLATION)
|
|
792
|
+
|
|
793
|
+
// Like Hermes on Android: java.text.Collator for the default locale.
|
|
794
|
+
double String::localeCompare(const String& other) const {
|
|
795
|
+
JNIEnv* env = facebook::jni::Environment::ensureCurrentThreadIsAttached();
|
|
796
|
+
static jclass collatorClass = static_cast<jclass>(env->NewGlobalRef(env->FindClass("java/text/Collator")));
|
|
797
|
+
static jmethodID getInstance = env->GetStaticMethodID(collatorClass, "getInstance", "()Ljava/text/Collator;");
|
|
798
|
+
static jmethodID compare = env->GetMethodID(collatorClass, "compare", "(Ljava/lang/String;Ljava/lang/String;)I");
|
|
799
|
+
std::u16string a = toUtf16(), b = other.toUtf16();
|
|
800
|
+
jstring ja = env->NewString(reinterpret_cast<const jchar*>(a.data()), static_cast<jsize>(a.size()));
|
|
801
|
+
jstring jb = env->NewString(reinterpret_cast<const jchar*>(b.data()), static_cast<jsize>(b.size()));
|
|
802
|
+
jobject collator = env->CallStaticObjectMethod(collatorClass, getInstance);
|
|
803
|
+
jint r = env->CallIntMethod(collator, compare, ja, jb);
|
|
804
|
+
env->DeleteLocalRef(collator);
|
|
805
|
+
env->DeleteLocalRef(ja);
|
|
806
|
+
env->DeleteLocalRef(jb);
|
|
807
|
+
return r < 0 ? -1 : r > 0 ? 1 : 0;
|
|
808
|
+
}
|
|
809
|
+
|
|
810
|
+
#else
|
|
811
|
+
|
|
812
|
+
namespace {
|
|
813
|
+
|
|
814
|
+
// A small approximation of root collation for hosts without a platform
|
|
815
|
+
// collator (tests on Linux): base letters first (whitespace, then
|
|
816
|
+
// punctuation and symbols, then digits, then letters), then accents, then
|
|
817
|
+
// case with lowercase first.
|
|
818
|
+
struct CollationElement {
|
|
819
|
+
int group;
|
|
820
|
+
uint32_t base;
|
|
821
|
+
uint32_t accent;
|
|
822
|
+
bool upper;
|
|
823
|
+
};
|
|
824
|
+
|
|
825
|
+
std::vector<CollationElement> collationElements(const String& s) {
|
|
826
|
+
std::vector<CollationElement> out;
|
|
827
|
+
for (size_t i = 0; i < s.length();) {
|
|
828
|
+
size_t w;
|
|
829
|
+
uint32_t cp = codePointAt(s, i, w);
|
|
830
|
+
i += w;
|
|
831
|
+
uint32_t base = cp, accent = 0;
|
|
832
|
+
if (cp < 0x10000) {
|
|
833
|
+
if (const Decomposition* d = findByCodePoint(kDecompositions, cp)) {
|
|
834
|
+
base = d->base;
|
|
835
|
+
accent = d->accent;
|
|
836
|
+
}
|
|
837
|
+
}
|
|
838
|
+
bool upper = false;
|
|
839
|
+
if (const CaseMapping* m = findByCodePoint(kLowerCase, base); m && m->length == 1) {
|
|
840
|
+
base = m->units[0];
|
|
841
|
+
upper = true;
|
|
842
|
+
}
|
|
843
|
+
int group = 3;
|
|
844
|
+
if (base == ' ' || base == '\t' || base == '\n' || base == '\r') group = 0;
|
|
845
|
+
else if (base < 0x80 && std::ispunct(static_cast<int>(base))) group = 1;
|
|
846
|
+
else if (base >= '0' && base <= '9') group = 2;
|
|
847
|
+
out.push_back({group, base, accent, upper});
|
|
848
|
+
}
|
|
849
|
+
return out;
|
|
850
|
+
}
|
|
851
|
+
|
|
852
|
+
template <class Key>
|
|
853
|
+
int compareLevel(const std::vector<CollationElement>& a, const std::vector<CollationElement>& b, Key key) {
|
|
854
|
+
size_t n = std::min(a.size(), b.size());
|
|
855
|
+
for (size_t i = 0; i < n; i++) {
|
|
856
|
+
auto x = key(a[i]), y = key(b[i]);
|
|
857
|
+
if (x != y) return x < y ? -1 : 1;
|
|
858
|
+
}
|
|
859
|
+
return a.size() == b.size() ? 0 : a.size() < b.size() ? -1 : 1;
|
|
860
|
+
}
|
|
861
|
+
|
|
862
|
+
} // namespace
|
|
863
|
+
|
|
864
|
+
double String::localeCompare(const String& other) const {
|
|
865
|
+
auto a = collationElements(*this), b = collationElements(other);
|
|
866
|
+
if (int r = compareLevel(a, b, [](const CollationElement& e) { return std::make_pair(e.group, e.base); })) return r;
|
|
867
|
+
if (int r = compareLevel(a, b, [](const CollationElement& e) { return e.accent; })) return r;
|
|
868
|
+
if (int r = compareLevel(a, b, [](const CollationElement& e) { return e.upper; })) return r;
|
|
869
|
+
return static_cast<double>(compare(*this, other));
|
|
870
|
+
}
|
|
871
|
+
|
|
872
|
+
#endif
|
|
873
|
+
|
|
874
|
+
} // namespace lucent
|