ranger-compiler 3.2.0 → 3.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +2482 -0
- package/LICENSE +28 -0
- package/LICENSE-MIT +21 -0
- package/README.md +1650 -1895
- package/dist/Lang.rgr +10946 -5852
- package/dist/README.md +3 -2
- package/dist/api.d.ts +1157 -29
- package/dist/api.js +66623 -38073
- package/dist/lib/ACEEditor.rgr +2 -0
- package/dist/lib/Ajax.rgr +2 -0
- package/dist/lib/CmdParams.rgr +2 -0
- package/dist/lib/Crypto.rgr +2 -0
- package/dist/lib/DOMLib.rgr +2 -0
- package/dist/lib/Engine3D.rgr +2 -0
- package/dist/lib/ImmutableVector.rgr +2 -0
- package/dist/lib/IndexedDB.rgr +2 -0
- package/dist/lib/IsoDate/DateMath.rgr +2 -0
- package/dist/lib/IsoDate/IsoCalendar.rgr +2 -0
- package/dist/lib/IsoDate/IsoDateParse.rgr +2 -0
- package/dist/lib/IsoDateLib.rgr +2 -0
- package/dist/lib/JSON.rgr +4906 -48
- package/dist/lib/JinxProcess.rgr +2 -0
- package/dist/lib/RangerProcess.rgr +2 -0
- package/dist/lib/Regex/RegexMatch.rgr +2 -0
- package/dist/lib/RegexLib.rgr +2 -0
- package/dist/lib/SQL.rgr +2 -0
- package/dist/lib/ServiceLib.rgr +2 -0
- package/dist/lib/Shell.rgr +326 -0
- package/dist/lib/Storage.rgr +2 -0
- package/dist/lib/Time.rgr +2 -0
- package/dist/lib/Timers.rgr +2 -0
- package/dist/lib/TypedArrays.rgr +2 -0
- package/dist/lib/ViewLib.rgr +2 -0
- package/dist/lib/WebLib.rgr +2 -0
- package/dist/lib/WebServerLib.rgr +2 -0
- package/dist/lib/apple/AppleAppBuilder.rgr +572 -0
- package/dist/lib/apple/AppleAppSpec.rgr +201 -0
- package/dist/lib/apple/AppleDevice.rgr +295 -0
- package/dist/lib/apple/AppleDeviceDoctor.rgr +453 -0
- package/dist/lib/apple/AppleSigning.rgr +379 -0
- package/dist/lib/apple/AppleSimulator.rgr +248 -0
- package/dist/lib/apple/AppleTarget.rgr +185 -0
- package/dist/lib/apple/AppleToolchain.rgr +458 -0
- package/dist/lib/apple/README.md +311 -0
- package/dist/lib/apple/apple_test.rgr +669 -0
- package/dist/lib/core/README.md +251 -0
- package/dist/lib/core/RgBase.rgr +313 -0
- package/dist/lib/core/RgNum.rgr +653 -0
- package/dist/lib/core/RgText.rgr +680 -0
- package/dist/lib/core/RgU32.rgr +309 -0
- package/dist/lib/ranger-dir.rgr +2 -0
- package/dist/lib/shell_test.rgr +193 -0
- package/dist/lib/stdlib.rgr +1176 -665
- package/dist/lib/stdops.rgr +2 -0
- package/dist/package.json +1 -1
- package/dist/rgrc.js +72955 -40753
- package/dist/stdops.rgr +2 -0
- package/package.json +1121 -339
|
@@ -0,0 +1,680 @@
|
|
|
1
|
+
; SPDX-License-Identifier: MIT
|
|
2
|
+
|
|
3
|
+
; =============================================================================
|
|
4
|
+
; RgText — one string model, whichever one the target actually has
|
|
5
|
+
; =============================================================================
|
|
6
|
+
; A Ranger `string` is not one thing. Measured, by compiling the same source:
|
|
7
|
+
;
|
|
8
|
+
; strlen es6 python go rust cpp
|
|
9
|
+
; "é" 1 1 1 1 2
|
|
10
|
+
; "😀" 2 1 1 1 4
|
|
11
|
+
;
|
|
12
|
+
; es6 counts UTF-16 CODE UNITS, python/go/rust count CODE POINTS, and C++ counts
|
|
13
|
+
; UTF-8 BYTES. `charAt` follows suit: on "é" it answers 233 on the first four and
|
|
14
|
+
; 195 (0xC3, the leading UTF-8 byte) on C++.
|
|
15
|
+
;
|
|
16
|
+
; So any function that indexes a string is three functions. Every higher layer —
|
|
17
|
+
; Intl, base64, digests, JSON quoting, URI escaping, regex — indexes strings, and
|
|
18
|
+
; each one would otherwise carry its own version of this. It is written once here.
|
|
19
|
+
;
|
|
20
|
+
; The addressing this file exposes is UTF-16 CODE UNITS, because that is what
|
|
21
|
+
; JavaScript's String is defined over and therefore what the engine's semantics
|
|
22
|
+
; are specified against. `RgText.len("😀")` is 2 on every target, including the
|
|
23
|
+
; ones whose own strlen says 1 or 4. A slice that cuts a surrogate pair yields
|
|
24
|
+
; the lone surrogate on a unit-model target and drops the half on a point-model
|
|
25
|
+
; one, where half a character has no representation — that difference is
|
|
26
|
+
; documented at `substr` rather than papered over.
|
|
27
|
+
;
|
|
28
|
+
; Lifted from ComponentEngine.rgr:39365-39744, which is where these algorithms
|
|
29
|
+
; were proven; see PLAN_JS_STDLIB_TEXT.md §2.
|
|
30
|
+
;
|
|
31
|
+
; TWO TARGET CONSTRAINTS THIS FILE WORKS AROUND
|
|
32
|
+
;
|
|
33
|
+
; 1. `strfromcode` used directly as an ARGUMENT does not compile on Rust: the
|
|
34
|
+
; writer emits `char::from_u32(n)`, a `char`, and then calls a String method
|
|
35
|
+
; on it (`no method named 'chars' found for type 'char'`). Concatenation
|
|
36
|
+
; coerces fine. Every `strfromcode` here is therefore BOUND to a `string`
|
|
37
|
+
; before it is passed anywhere.
|
|
38
|
+
;
|
|
39
|
+
; 2. On es6, `strfromcode` is `String.fromCharCode`, which truncates to 16
|
|
40
|
+
; bits — `strfromcode 128512` gives one BMP character, not an emoji. So
|
|
41
|
+
; `fromCodePoint` builds a surrogate PAIR there rather than trusting it.
|
|
42
|
+
;
|
|
43
|
+
; NAMING: `kind`, `len`, `substr` and the rest avoid the global operator namespace,
|
|
44
|
+
; which a static method may not enter — see lib/core/README.md.
|
|
45
|
+
; =============================================================================
|
|
46
|
+
|
|
47
|
+
class RgText {
|
|
48
|
+
|
|
49
|
+
; ---- which model is this ------------------------------------------------
|
|
50
|
+
|
|
51
|
+
; 0 = UTF-16 code units, 1 = UTF-8 bytes, 2 = code points.
|
|
52
|
+
;
|
|
53
|
+
; Recomputed rather than cached: the core-layer rules forbid mutable
|
|
54
|
+
; file-level state, and this is two strlen calls on one-character literals.
|
|
55
|
+
; If a profile ever says it matters, the fix is a value threaded by the
|
|
56
|
+
; caller, not a static field.
|
|
57
|
+
static sfn kind:int () {
|
|
58
|
+
if ((strlen "é") > 1) {
|
|
59
|
+
return 1
|
|
60
|
+
}
|
|
61
|
+
if ((strlen "😀") == 1) {
|
|
62
|
+
return 2
|
|
63
|
+
}
|
|
64
|
+
return 0
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
static sfn unitsAreBytes:boolean () {
|
|
68
|
+
return ((RgText.kind()) == 1)
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
; True where strlen/charAt/substring index whole CHARACTERS, so the only gap
|
|
72
|
+
; to UTF-16 is that a supplementary character is one unit here and two there.
|
|
73
|
+
static sfn unitsArePoints:boolean () {
|
|
74
|
+
return ((RgText.kind()) == 2)
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
; ---- UTF-8 geometry ----------------------------------------------------
|
|
78
|
+
|
|
79
|
+
static sfn utf8WidthOf:int (cp:int) {
|
|
80
|
+
if (cp < 128) {
|
|
81
|
+
return 1
|
|
82
|
+
}
|
|
83
|
+
if (cp < 2048) {
|
|
84
|
+
return 2
|
|
85
|
+
}
|
|
86
|
+
if (cp < 65536) {
|
|
87
|
+
return 3
|
|
88
|
+
}
|
|
89
|
+
return 4
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
; Bytes [i, j) of a string on a byte-model target.
|
|
93
|
+
;
|
|
94
|
+
; `substring` IS the byte slice there — on a byte-model target strlen,
|
|
95
|
+
; charAt and substring all index bytes. The obvious alternative, rebuilding
|
|
96
|
+
; the run byte by byte through `strfromcode`, is WRONG: on C++
|
|
97
|
+
; `strfromcode 195` answers the two-byte UTF-8 encoding of U+00C3, not the
|
|
98
|
+
; single byte 0xC3, so slicing "héllo" that way double-encoded it to "él".
|
|
99
|
+
; Measured; the cross-target suite caught it.
|
|
100
|
+
static sfn byteSlice:string (s:string i:int j:int) {
|
|
101
|
+
def n:int (strlen s)
|
|
102
|
+
def a:int i
|
|
103
|
+
def b:int j
|
|
104
|
+
if (a < 0) {
|
|
105
|
+
a = 0
|
|
106
|
+
}
|
|
107
|
+
if (b > n) {
|
|
108
|
+
b = n
|
|
109
|
+
}
|
|
110
|
+
if (b <= a) {
|
|
111
|
+
return ""
|
|
112
|
+
}
|
|
113
|
+
return (substring s a b)
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
; The code POINT whose encoding starts at byte `i`, and its width in bytes,
|
|
117
|
+
; as a two-element list. Byte-model targets only; [-1, 0] past the end.
|
|
118
|
+
static sfn decodeAt:[int] (s:string i:int) {
|
|
119
|
+
def out:[int]
|
|
120
|
+
def n:int (strlen s)
|
|
121
|
+
if (i >= n) {
|
|
122
|
+
push out (0 - 1)
|
|
123
|
+
push out 0
|
|
124
|
+
return out
|
|
125
|
+
}
|
|
126
|
+
def b0:int (bit_and (charAt s i) 255)
|
|
127
|
+
if (b0 < 128) {
|
|
128
|
+
push out b0
|
|
129
|
+
push out 1
|
|
130
|
+
return out
|
|
131
|
+
}
|
|
132
|
+
if (b0 < 224) {
|
|
133
|
+
def c1:int 0
|
|
134
|
+
if ((i + 1) < n) {
|
|
135
|
+
c1 = (bit_and (charAt s (i + 1)) 63)
|
|
136
|
+
}
|
|
137
|
+
push out (((bit_and b0 31) * 64) + c1)
|
|
138
|
+
push out 2
|
|
139
|
+
return out
|
|
140
|
+
}
|
|
141
|
+
if (b0 < 240) {
|
|
142
|
+
def d1:int 0
|
|
143
|
+
def d2:int 0
|
|
144
|
+
if ((i + 1) < n) {
|
|
145
|
+
d1 = (bit_and (charAt s (i + 1)) 63)
|
|
146
|
+
}
|
|
147
|
+
if ((i + 2) < n) {
|
|
148
|
+
d2 = (bit_and (charAt s (i + 2)) 63)
|
|
149
|
+
}
|
|
150
|
+
push out ((((bit_and b0 15) * 4096) + (d1 * 64)) + d2)
|
|
151
|
+
push out 3
|
|
152
|
+
return out
|
|
153
|
+
}
|
|
154
|
+
def e1:int 0
|
|
155
|
+
def e2:int 0
|
|
156
|
+
def e3:int 0
|
|
157
|
+
if ((i + 1) < n) {
|
|
158
|
+
e1 = (bit_and (charAt s (i + 1)) 63)
|
|
159
|
+
}
|
|
160
|
+
if ((i + 2) < n) {
|
|
161
|
+
e2 = (bit_and (charAt s (i + 2)) 63)
|
|
162
|
+
}
|
|
163
|
+
if ((i + 3) < n) {
|
|
164
|
+
e3 = (bit_and (charAt s (i + 3)) 63)
|
|
165
|
+
}
|
|
166
|
+
push out (((((bit_and b0 7) * 262144) + (e1 * 4096)) + (e2 * 64)) + e3)
|
|
167
|
+
push out 4
|
|
168
|
+
return out
|
|
169
|
+
}
|
|
170
|
+
|
|
171
|
+
; ---- construction ------------------------------------------------------
|
|
172
|
+
|
|
173
|
+
; One code point as a string. On a byte target that is its UTF-8 encoding;
|
|
174
|
+
; on a unit target an astral point becomes a SURROGATE PAIR, because
|
|
175
|
+
; `strfromcode` there is String.fromCharCode and truncates to 16 bits.
|
|
176
|
+
static sfn fromCodePoint:string (cp:int) {
|
|
177
|
+
if (RgText.unitsAreBytes()) {
|
|
178
|
+
return (RgText.encodeUtf8(cp))
|
|
179
|
+
}
|
|
180
|
+
if (RgText.unitsArePoints()) {
|
|
181
|
+
def p:string (strfromcode cp)
|
|
182
|
+
return p
|
|
183
|
+
}
|
|
184
|
+
if (cp < 65536) {
|
|
185
|
+
def q:string (strfromcode cp)
|
|
186
|
+
return q
|
|
187
|
+
}
|
|
188
|
+
def adj:int (cp - 65536)
|
|
189
|
+
def hi:string (strfromcode (55296 + (bit_shr adj 10)))
|
|
190
|
+
def lo:string (strfromcode (56320 + (bit_and adj 1023)))
|
|
191
|
+
return (hi + lo)
|
|
192
|
+
}
|
|
193
|
+
|
|
194
|
+
; One code point as a string on a byte-model target — that is, as its UTF-8
|
|
195
|
+
; encoding, since the string IS bytes there.
|
|
196
|
+
;
|
|
197
|
+
; `strfromcode` already does exactly this: on C++ it answers the two-byte
|
|
198
|
+
; encoding of U+00C3 for 195, not the raw byte. So the encoding must NOT be
|
|
199
|
+
; assembled by hand out of per-byte `strfromcode` calls — that produced
|
|
200
|
+
; double-encoded UTF-8 ("é" for "é"). Handing the whole code point over is
|
|
201
|
+
; both correct and shorter.
|
|
202
|
+
;
|
|
203
|
+
; A lone surrogate arrives here from `substr` when a cut splits a pair. The
|
|
204
|
+
; result is CESU-8, which is the only way a byte string can carry half a
|
|
205
|
+
; character; that is documented at `substr`.
|
|
206
|
+
static sfn encodeUtf8:string (cp:int) {
|
|
207
|
+
def p:string (strfromcode cp)
|
|
208
|
+
return p
|
|
209
|
+
}
|
|
210
|
+
|
|
211
|
+
; ---- code-unit addressing ----------------------------------------------
|
|
212
|
+
|
|
213
|
+
; Code units in `s`. This is JavaScript's `.length`.
|
|
214
|
+
static sfn len:int (s:string) {
|
|
215
|
+
def n:int (strlen s)
|
|
216
|
+
if (RgText.unitsArePoints()) {
|
|
217
|
+
def p:int 0
|
|
218
|
+
def cnt:int 0
|
|
219
|
+
while (p < n) {
|
|
220
|
+
if ((charAt s p) > 65535) {
|
|
221
|
+
cnt = (cnt + 2)
|
|
222
|
+
} {
|
|
223
|
+
cnt = (cnt + 1)
|
|
224
|
+
}
|
|
225
|
+
p = (p + 1)
|
|
226
|
+
}
|
|
227
|
+
return cnt
|
|
228
|
+
}
|
|
229
|
+
if (false == (RgText.unitsAreBytes())) {
|
|
230
|
+
return n
|
|
231
|
+
}
|
|
232
|
+
def i:int 0
|
|
233
|
+
def units:int 0
|
|
234
|
+
while (i < n) {
|
|
235
|
+
def b:int (bit_and (charAt s i) 255)
|
|
236
|
+
if (b < 128) {
|
|
237
|
+
units = (units + 1)
|
|
238
|
+
i = (i + 1)
|
|
239
|
+
} {
|
|
240
|
+
; Anything past U+FFFF needs TWO code units: JavaScript spells
|
|
241
|
+
; it as a surrogate pair.
|
|
242
|
+
if (b < 224) {
|
|
243
|
+
units = (units + 1)
|
|
244
|
+
i = (i + 2)
|
|
245
|
+
} {
|
|
246
|
+
if (b < 240) {
|
|
247
|
+
units = (units + 1)
|
|
248
|
+
i = (i + 3)
|
|
249
|
+
} {
|
|
250
|
+
units = (units + 2)
|
|
251
|
+
i = (i + 4)
|
|
252
|
+
}
|
|
253
|
+
}
|
|
254
|
+
}
|
|
255
|
+
}
|
|
256
|
+
return units
|
|
257
|
+
}
|
|
258
|
+
|
|
259
|
+
; The code UNIT at index `u`: the character itself inside the BMP, and the
|
|
260
|
+
; high or low surrogate for an astral one. -1 past the end.
|
|
261
|
+
static sfn unitAt:int (s:string u:int) {
|
|
262
|
+
def n:int (strlen s)
|
|
263
|
+
if (u < 0) {
|
|
264
|
+
return (0 - 1)
|
|
265
|
+
}
|
|
266
|
+
if (RgText.unitsArePoints()) {
|
|
267
|
+
def p:int 0
|
|
268
|
+
def units:int 0
|
|
269
|
+
while (p < n) {
|
|
270
|
+
def cp:int (charAt s p)
|
|
271
|
+
if (cp < 65536) {
|
|
272
|
+
if (units == u) {
|
|
273
|
+
return cp
|
|
274
|
+
}
|
|
275
|
+
units = (units + 1)
|
|
276
|
+
} {
|
|
277
|
+
def adj:int (cp - 65536)
|
|
278
|
+
if (units == u) {
|
|
279
|
+
return (55296 + (bit_shr adj 10))
|
|
280
|
+
}
|
|
281
|
+
if ((units + 1) == u) {
|
|
282
|
+
return (56320 + (bit_and adj 1023))
|
|
283
|
+
}
|
|
284
|
+
units = (units + 2)
|
|
285
|
+
}
|
|
286
|
+
p = (p + 1)
|
|
287
|
+
}
|
|
288
|
+
return (0 - 1)
|
|
289
|
+
}
|
|
290
|
+
if (false == (RgText.unitsAreBytes())) {
|
|
291
|
+
if (u >= n) {
|
|
292
|
+
return (0 - 1)
|
|
293
|
+
}
|
|
294
|
+
return (bit_and (charAt s u) 65535)
|
|
295
|
+
}
|
|
296
|
+
def i:int 0
|
|
297
|
+
def units:int 0
|
|
298
|
+
while (i < n) {
|
|
299
|
+
def dec:[int] (RgText.decodeAt(s i))
|
|
300
|
+
def cp:int (itemAt dec 0)
|
|
301
|
+
def w:int (itemAt dec 1)
|
|
302
|
+
if (w == 0) {
|
|
303
|
+
return (0 - 1)
|
|
304
|
+
}
|
|
305
|
+
if (cp < 65536) {
|
|
306
|
+
if (units == u) {
|
|
307
|
+
return cp
|
|
308
|
+
}
|
|
309
|
+
units = (units + 1)
|
|
310
|
+
} {
|
|
311
|
+
def adj:int (cp - 65536)
|
|
312
|
+
if (units == u) {
|
|
313
|
+
return (55296 + (bit_shr adj 10))
|
|
314
|
+
}
|
|
315
|
+
if ((units + 1) == u) {
|
|
316
|
+
return (56320 + (bit_and adj 1023))
|
|
317
|
+
}
|
|
318
|
+
units = (units + 2)
|
|
319
|
+
}
|
|
320
|
+
i = (i + w)
|
|
321
|
+
}
|
|
322
|
+
return (0 - 1)
|
|
323
|
+
}
|
|
324
|
+
|
|
325
|
+
; The substring between two code-unit indexes.
|
|
326
|
+
;
|
|
327
|
+
; A cut that lands INSIDE a surrogate pair yields the corresponding lone
|
|
328
|
+
; surrogate on a unit-model target, which is what keeps `s.slice(0, 1)` on an
|
|
329
|
+
; astral character one unit long instead of silently the whole character. On a
|
|
330
|
+
; POINT-model target half a pair has no representation — that string type
|
|
331
|
+
; holds characters — so the half is dropped rather than faked, which keeps the
|
|
332
|
+
; result well-formed. Slicing that does not split a pair, which is everything
|
|
333
|
+
; a program normally does, is exact on every target.
|
|
334
|
+
static sfn substr:string (s:string a:int b:int) {
|
|
335
|
+
def lo:int a
|
|
336
|
+
def hi:int b
|
|
337
|
+
if (lo < 0) {
|
|
338
|
+
lo = 0
|
|
339
|
+
}
|
|
340
|
+
if (hi < lo) {
|
|
341
|
+
hi = lo
|
|
342
|
+
}
|
|
343
|
+
def n:int (strlen s)
|
|
344
|
+
if (RgText.unitsArePoints()) {
|
|
345
|
+
def out:string ""
|
|
346
|
+
def p:int 0
|
|
347
|
+
def units:int 0
|
|
348
|
+
while (p < n) {
|
|
349
|
+
def cp:int (charAt s p)
|
|
350
|
+
def w:int 1
|
|
351
|
+
if (cp > 65535) {
|
|
352
|
+
w = 2
|
|
353
|
+
}
|
|
354
|
+
if ((units >= lo) && ((units + w) <= hi)) {
|
|
355
|
+
out = (out + (at s p))
|
|
356
|
+
}
|
|
357
|
+
units = (units + w)
|
|
358
|
+
p = (p + 1)
|
|
359
|
+
}
|
|
360
|
+
return out
|
|
361
|
+
}
|
|
362
|
+
if (false == (RgText.unitsAreBytes())) {
|
|
363
|
+
if (lo >= n) {
|
|
364
|
+
return ""
|
|
365
|
+
}
|
|
366
|
+
def hi2:int hi
|
|
367
|
+
if (hi2 > n) {
|
|
368
|
+
hi2 = n
|
|
369
|
+
}
|
|
370
|
+
return (substring s lo hi2)
|
|
371
|
+
}
|
|
372
|
+
def out2:string ""
|
|
373
|
+
def i:int 0
|
|
374
|
+
def units2:int 0
|
|
375
|
+
while (i < n) {
|
|
376
|
+
def dec:[int] (RgText.decodeAt(s i))
|
|
377
|
+
def cp:int (itemAt dec 0)
|
|
378
|
+
def w:int (itemAt dec 1)
|
|
379
|
+
if (w == 0) {
|
|
380
|
+
i = n
|
|
381
|
+
} {
|
|
382
|
+
if (cp < 65536) {
|
|
383
|
+
if ((units2 >= lo) && (units2 < hi)) {
|
|
384
|
+
out2 = (out2 + (RgText.byteSlice(s i (i + w))))
|
|
385
|
+
}
|
|
386
|
+
units2 = (units2 + 1)
|
|
387
|
+
} {
|
|
388
|
+
def adj:int (cp - 65536)
|
|
389
|
+
def takeHi:boolean ((units2 >= lo) && (units2 < hi))
|
|
390
|
+
def takeLo:boolean (((units2 + 1) >= lo) && ((units2 + 1) < hi))
|
|
391
|
+
if (takeHi && takeLo) {
|
|
392
|
+
out2 = (out2 + (RgText.byteSlice(s i (i + w))))
|
|
393
|
+
} {
|
|
394
|
+
; One half of a pair, encoded as the surrogate it is.
|
|
395
|
+
; CESU-8 rather than UTF-8, which is the only way a byte
|
|
396
|
+
; string can carry half a character at all.
|
|
397
|
+
if takeHi {
|
|
398
|
+
out2 = (out2 + (RgText.encodeUtf8((55296 + (bit_shr adj 10)))))
|
|
399
|
+
}
|
|
400
|
+
if takeLo {
|
|
401
|
+
out2 = (out2 + (RgText.encodeUtf8((56320 + (bit_and adj 1023)))))
|
|
402
|
+
}
|
|
403
|
+
}
|
|
404
|
+
units2 = (units2 + 2)
|
|
405
|
+
}
|
|
406
|
+
i = (i + w)
|
|
407
|
+
}
|
|
408
|
+
}
|
|
409
|
+
return out2
|
|
410
|
+
}
|
|
411
|
+
|
|
412
|
+
; ---- UTF-8 offsets -----------------------------------------------------
|
|
413
|
+
;
|
|
414
|
+
; These convert between a UTF-16 code-unit index and a UTF-8 BYTE offset.
|
|
415
|
+
; Both meanings are target-independent: the answer for "a😀b" is the same
|
|
416
|
+
; number on es6 as on C++, even though the two disagree about what their own
|
|
417
|
+
; strlen counts.
|
|
418
|
+
;
|
|
419
|
+
; The engine's `cuByteOf` that these grew out of meant something subtly
|
|
420
|
+
; different — "the offset the TARGET'S OWN string search reports" — which is
|
|
421
|
+
; code units on es6 and bytes elsewhere. That contract is target-relative by
|
|
422
|
+
; construction, and it is why the first version of this file answered 3 on
|
|
423
|
+
; es6 and 5 everywhere else for the same call. A bridge to native search
|
|
424
|
+
; offsets is a real need, but it belongs to whoever is calling the native
|
|
425
|
+
; search, not to a portable library.
|
|
426
|
+
|
|
427
|
+
; The UTF-8 byte offset at which code unit `u` begins; the encoded length
|
|
428
|
+
; when `u` is at or past the end. A `u` that lands inside a surrogate pair
|
|
429
|
+
; reports the offset just past that pair — half a character has no offset.
|
|
430
|
+
static sfn utf8ByteOfUnit:int (s:string u:int) {
|
|
431
|
+
if (u <= 0) {
|
|
432
|
+
return 0
|
|
433
|
+
}
|
|
434
|
+
def n:int (RgText.len(s))
|
|
435
|
+
def bytes:int 0
|
|
436
|
+
def i:int 0
|
|
437
|
+
while (i < n) {
|
|
438
|
+
if (i >= u) {
|
|
439
|
+
return bytes
|
|
440
|
+
}
|
|
441
|
+
def cp:int (RgText.codePointAt(s i))
|
|
442
|
+
if (cp < 0) {
|
|
443
|
+
return bytes
|
|
444
|
+
}
|
|
445
|
+
def w:int 1
|
|
446
|
+
if (cp > 65535) {
|
|
447
|
+
w = 2
|
|
448
|
+
}
|
|
449
|
+
bytes = (bytes + (RgText.utf8WidthOf(cp)))
|
|
450
|
+
i = (i + w)
|
|
451
|
+
}
|
|
452
|
+
return bytes
|
|
453
|
+
}
|
|
454
|
+
|
|
455
|
+
; The inverse: the code-unit index that a UTF-8 byte offset corresponds to.
|
|
456
|
+
; What a search that ran over the encoded bytes has to report back.
|
|
457
|
+
static sfn unitOfUtf8Byte:int (s:string byteIdx:int) {
|
|
458
|
+
if (byteIdx <= 0) {
|
|
459
|
+
return 0
|
|
460
|
+
}
|
|
461
|
+
def n:int (RgText.len(s))
|
|
462
|
+
def bytes:int 0
|
|
463
|
+
def units:int 0
|
|
464
|
+
def i:int 0
|
|
465
|
+
while (i < n) {
|
|
466
|
+
if (bytes >= byteIdx) {
|
|
467
|
+
return units
|
|
468
|
+
}
|
|
469
|
+
def cp:int (RgText.codePointAt(s i))
|
|
470
|
+
if (cp < 0) {
|
|
471
|
+
return units
|
|
472
|
+
}
|
|
473
|
+
def w:int 1
|
|
474
|
+
if (cp > 65535) {
|
|
475
|
+
w = 2
|
|
476
|
+
}
|
|
477
|
+
bytes = (bytes + (RgText.utf8WidthOf(cp)))
|
|
478
|
+
units = (units + w)
|
|
479
|
+
i = (i + w)
|
|
480
|
+
}
|
|
481
|
+
return units
|
|
482
|
+
}
|
|
483
|
+
|
|
484
|
+
; ---- code points -------------------------------------------------------
|
|
485
|
+
|
|
486
|
+
; The full code point beginning at code-unit index `u`, combining a surrogate
|
|
487
|
+
; pair when one starts there. This is JavaScript's codePointAt.
|
|
488
|
+
static sfn codePointAt:int (s:string u:int) {
|
|
489
|
+
def first:int (RgText.unitAt(s u))
|
|
490
|
+
if (first < 0) {
|
|
491
|
+
return (0 - 1)
|
|
492
|
+
}
|
|
493
|
+
if (first < 55296) {
|
|
494
|
+
return first
|
|
495
|
+
}
|
|
496
|
+
if (first > 56319) {
|
|
497
|
+
return first
|
|
498
|
+
}
|
|
499
|
+
def second:int (RgText.unitAt(s (u + 1)))
|
|
500
|
+
if (second < 56320) {
|
|
501
|
+
return first
|
|
502
|
+
}
|
|
503
|
+
if (second > 57343) {
|
|
504
|
+
return first
|
|
505
|
+
}
|
|
506
|
+
return ((((first - 55296) * 1024) + (second - 56320)) + 65536)
|
|
507
|
+
}
|
|
508
|
+
|
|
509
|
+
static sfn toCodePoints:[int] (s:string) {
|
|
510
|
+
def out:[int]
|
|
511
|
+
def n:int (RgText.len(s))
|
|
512
|
+
def u:int 0
|
|
513
|
+
while (u < n) {
|
|
514
|
+
def cp:int (RgText.codePointAt(s u))
|
|
515
|
+
if (cp < 0) {
|
|
516
|
+
u = n
|
|
517
|
+
} {
|
|
518
|
+
push out cp
|
|
519
|
+
if (cp > 65535) {
|
|
520
|
+
u = (u + 2)
|
|
521
|
+
} {
|
|
522
|
+
u = (u + 1)
|
|
523
|
+
}
|
|
524
|
+
}
|
|
525
|
+
}
|
|
526
|
+
return out
|
|
527
|
+
}
|
|
528
|
+
|
|
529
|
+
static sfn codePointCount:int (s:string) {
|
|
530
|
+
def cps:[int] (RgText.toCodePoints(s))
|
|
531
|
+
return (array_length cps)
|
|
532
|
+
}
|
|
533
|
+
|
|
534
|
+
static sfn fromCodePoints:string (cps:[int]) {
|
|
535
|
+
def out:string ""
|
|
536
|
+
def i:int 0
|
|
537
|
+
while (i < (array_length cps)) {
|
|
538
|
+
out = (out + (RgText.fromCodePoint((itemAt cps i))))
|
|
539
|
+
i = (i + 1)
|
|
540
|
+
}
|
|
541
|
+
return out
|
|
542
|
+
}
|
|
543
|
+
|
|
544
|
+
; ---- code units as data ------------------------------------------------
|
|
545
|
+
|
|
546
|
+
static sfn toCodeUnits:[int] (s:string) {
|
|
547
|
+
def out:[int]
|
|
548
|
+
def n:int (RgText.len(s))
|
|
549
|
+
def u:int 0
|
|
550
|
+
while (u < n) {
|
|
551
|
+
push out (RgText.unitAt(s u))
|
|
552
|
+
u = (u + 1)
|
|
553
|
+
}
|
|
554
|
+
return out
|
|
555
|
+
}
|
|
556
|
+
|
|
557
|
+
static sfn fromCodeUnits:string (units:[int]) {
|
|
558
|
+
def out:string ""
|
|
559
|
+
def i:int 0
|
|
560
|
+
def n:int (array_length units)
|
|
561
|
+
while (i < n) {
|
|
562
|
+
def hi:int (itemAt units i)
|
|
563
|
+
; Recombine a well-formed pair so a byte-model target writes UTF-8
|
|
564
|
+
; rather than CESU-8 for the astral characters.
|
|
565
|
+
def paired:boolean false
|
|
566
|
+
if ((hi >= 55296) && (hi <= 56319)) {
|
|
567
|
+
if ((i + 1) < n) {
|
|
568
|
+
def lo:int (itemAt units (i + 1))
|
|
569
|
+
if ((lo >= 56320) && (lo <= 57343)) {
|
|
570
|
+
def cp:int ((((hi - 55296) * 1024) + (lo - 56320)) + 65536)
|
|
571
|
+
out = (out + (RgText.fromCodePoint(cp)))
|
|
572
|
+
i = (i + 2)
|
|
573
|
+
paired = true
|
|
574
|
+
}
|
|
575
|
+
}
|
|
576
|
+
}
|
|
577
|
+
if (false == paired) {
|
|
578
|
+
out = (out + (RgText.fromCodePoint(hi)))
|
|
579
|
+
i = (i + 1)
|
|
580
|
+
}
|
|
581
|
+
}
|
|
582
|
+
return out
|
|
583
|
+
}
|
|
584
|
+
|
|
585
|
+
; ---- UTF-8 bytes -------------------------------------------------------
|
|
586
|
+
; The one bridge between text and bytes in lib/core. RgBase and RgCrypto go
|
|
587
|
+
; through this and nothing else, so the string model is dealt with in exactly
|
|
588
|
+
; one place rather than once per digest and once per encoder.
|
|
589
|
+
|
|
590
|
+
static sfn toUtf8Bytes:[int] (s:string) {
|
|
591
|
+
def out:[int]
|
|
592
|
+
; A byte-model string already IS the encoding.
|
|
593
|
+
if (RgText.unitsAreBytes()) {
|
|
594
|
+
def n:int (strlen s)
|
|
595
|
+
def i:int 0
|
|
596
|
+
while (i < n) {
|
|
597
|
+
push out (bit_and (charAt s i) 255)
|
|
598
|
+
i = (i + 1)
|
|
599
|
+
}
|
|
600
|
+
return out
|
|
601
|
+
}
|
|
602
|
+
def cps:[int] (RgText.toCodePoints(s))
|
|
603
|
+
def k:int 0
|
|
604
|
+
while (k < (array_length cps)) {
|
|
605
|
+
def cp:int (itemAt cps k)
|
|
606
|
+
if (cp < 128) {
|
|
607
|
+
push out cp
|
|
608
|
+
} {
|
|
609
|
+
if (cp < 2048) {
|
|
610
|
+
push out (bit_or 192 (bit_shr cp 6))
|
|
611
|
+
push out (bit_or 128 (bit_and cp 63))
|
|
612
|
+
} {
|
|
613
|
+
if (cp < 65536) {
|
|
614
|
+
push out (bit_or 224 (bit_shr cp 12))
|
|
615
|
+
push out (bit_or 128 (bit_and (bit_shr cp 6) 63))
|
|
616
|
+
push out (bit_or 128 (bit_and cp 63))
|
|
617
|
+
} {
|
|
618
|
+
push out (bit_or 240 (bit_shr cp 18))
|
|
619
|
+
push out (bit_or 128 (bit_and (bit_shr cp 12) 63))
|
|
620
|
+
push out (bit_or 128 (bit_and (bit_shr cp 6) 63))
|
|
621
|
+
push out (bit_or 128 (bit_and cp 63))
|
|
622
|
+
}
|
|
623
|
+
}
|
|
624
|
+
}
|
|
625
|
+
k = (k + 1)
|
|
626
|
+
}
|
|
627
|
+
return out
|
|
628
|
+
}
|
|
629
|
+
|
|
630
|
+
static sfn fromUtf8Bytes:string (bytes:[int]) {
|
|
631
|
+
def cps:[int]
|
|
632
|
+
def n:int (array_length bytes)
|
|
633
|
+
def i:int 0
|
|
634
|
+
while (i < n) {
|
|
635
|
+
def b0:int (bit_and (itemAt bytes i) 255)
|
|
636
|
+
if (b0 < 128) {
|
|
637
|
+
push cps b0
|
|
638
|
+
i = (i + 1)
|
|
639
|
+
} {
|
|
640
|
+
if (b0 < 224) {
|
|
641
|
+
def c1:int 0
|
|
642
|
+
if ((i + 1) < n) {
|
|
643
|
+
c1 = (bit_and (itemAt bytes (i + 1)) 63)
|
|
644
|
+
}
|
|
645
|
+
push cps (((bit_and b0 31) * 64) + c1)
|
|
646
|
+
i = (i + 2)
|
|
647
|
+
} {
|
|
648
|
+
if (b0 < 240) {
|
|
649
|
+
def d1:int 0
|
|
650
|
+
def d2:int 0
|
|
651
|
+
if ((i + 1) < n) {
|
|
652
|
+
d1 = (bit_and (itemAt bytes (i + 1)) 63)
|
|
653
|
+
}
|
|
654
|
+
if ((i + 2) < n) {
|
|
655
|
+
d2 = (bit_and (itemAt bytes (i + 2)) 63)
|
|
656
|
+
}
|
|
657
|
+
push cps ((((bit_and b0 15) * 4096) + (d1 * 64)) + d2)
|
|
658
|
+
i = (i + 3)
|
|
659
|
+
} {
|
|
660
|
+
def e1:int 0
|
|
661
|
+
def e2:int 0
|
|
662
|
+
def e3:int 0
|
|
663
|
+
if ((i + 1) < n) {
|
|
664
|
+
e1 = (bit_and (itemAt bytes (i + 1)) 63)
|
|
665
|
+
}
|
|
666
|
+
if ((i + 2) < n) {
|
|
667
|
+
e2 = (bit_and (itemAt bytes (i + 2)) 63)
|
|
668
|
+
}
|
|
669
|
+
if ((i + 3) < n) {
|
|
670
|
+
e3 = (bit_and (itemAt bytes (i + 3)) 63)
|
|
671
|
+
}
|
|
672
|
+
push cps (((((bit_and b0 7) * 262144) + (e1 * 4096)) + (e2 * 64)) + e3)
|
|
673
|
+
i = (i + 4)
|
|
674
|
+
}
|
|
675
|
+
}
|
|
676
|
+
}
|
|
677
|
+
}
|
|
678
|
+
return (RgText.fromCodePoints(cps))
|
|
679
|
+
}
|
|
680
|
+
}
|