ranger-compiler 3.1.1 → 3.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (58) hide show
  1. package/CHANGELOG.md +2527 -0
  2. package/LICENSE +28 -0
  3. package/LICENSE-MIT +21 -0
  4. package/README.md +1650 -1875
  5. package/dist/Lang.rgr +10946 -5662
  6. package/dist/README.md +3 -2
  7. package/dist/api.d.ts +2023 -61
  8. package/dist/api.js +68930 -27355
  9. package/dist/lib/ACEEditor.rgr +2 -0
  10. package/dist/lib/Ajax.rgr +2 -0
  11. package/dist/lib/CmdParams.rgr +2 -0
  12. package/dist/lib/Crypto.rgr +2 -0
  13. package/dist/lib/DOMLib.rgr +2 -0
  14. package/dist/lib/Engine3D.rgr +2 -0
  15. package/dist/lib/ImmutableVector.rgr +2 -0
  16. package/dist/lib/IndexedDB.rgr +2 -0
  17. package/dist/lib/IsoDate/DateMath.rgr +2 -0
  18. package/dist/lib/IsoDate/IsoCalendar.rgr +2 -0
  19. package/dist/lib/IsoDate/IsoDateParse.rgr +2 -0
  20. package/dist/lib/IsoDateLib.rgr +2 -0
  21. package/dist/lib/JSON.rgr +4906 -48
  22. package/dist/lib/JinxProcess.rgr +2 -0
  23. package/dist/lib/RangerProcess.rgr +2 -0
  24. package/dist/lib/Regex/RegexMatch.rgr +2 -0
  25. package/dist/lib/RegexLib.rgr +2 -0
  26. package/dist/lib/SQL.rgr +2 -0
  27. package/dist/lib/ServiceLib.rgr +2 -0
  28. package/dist/lib/Shell.rgr +326 -0
  29. package/dist/lib/Storage.rgr +2 -0
  30. package/dist/lib/Time.rgr +2 -0
  31. package/dist/lib/Timers.rgr +2 -0
  32. package/dist/lib/TypedArrays.rgr +2 -0
  33. package/dist/lib/ViewLib.rgr +2 -0
  34. package/dist/lib/WebLib.rgr +2 -0
  35. package/dist/lib/WebServerLib.rgr +2 -0
  36. package/dist/lib/apple/AppleAppBuilder.rgr +572 -0
  37. package/dist/lib/apple/AppleAppSpec.rgr +201 -0
  38. package/dist/lib/apple/AppleDevice.rgr +295 -0
  39. package/dist/lib/apple/AppleDeviceDoctor.rgr +453 -0
  40. package/dist/lib/apple/AppleSigning.rgr +379 -0
  41. package/dist/lib/apple/AppleSimulator.rgr +248 -0
  42. package/dist/lib/apple/AppleTarget.rgr +185 -0
  43. package/dist/lib/apple/AppleToolchain.rgr +458 -0
  44. package/dist/lib/apple/README.md +311 -0
  45. package/dist/lib/apple/apple_test.rgr +669 -0
  46. package/dist/lib/core/README.md +251 -0
  47. package/dist/lib/core/RgBase.rgr +313 -0
  48. package/dist/lib/core/RgNum.rgr +653 -0
  49. package/dist/lib/core/RgText.rgr +680 -0
  50. package/dist/lib/core/RgU32.rgr +309 -0
  51. package/dist/lib/ranger-dir.rgr +27 -5
  52. package/dist/lib/shell_test.rgr +193 -0
  53. package/dist/lib/stdlib.rgr +1176 -665
  54. package/dist/lib/stdops.rgr +2 -0
  55. package/dist/package.json +1 -1
  56. package/dist/rgrc.js +73013 -37819
  57. package/dist/stdops.rgr +2 -0
  58. package/package.json +1121 -205
@@ -0,0 +1,680 @@
1
+ ; SPDX-License-Identifier: MIT
2
+
3
+ ; =============================================================================
4
+ ; RgText — one string model, whichever one the target actually has
5
+ ; =============================================================================
6
+ ; A Ranger `string` is not one thing. Measured, by compiling the same source:
7
+ ;
8
+ ; strlen es6 python go rust cpp
9
+ ; "é" 1 1 1 1 2
10
+ ; "😀" 2 1 1 1 4
11
+ ;
12
+ ; es6 counts UTF-16 CODE UNITS, python/go/rust count CODE POINTS, and C++ counts
13
+ ; UTF-8 BYTES. `charAt` follows suit: on "é" it answers 233 on the first four and
14
+ ; 195 (0xC3, the leading UTF-8 byte) on C++.
15
+ ;
16
+ ; So any function that indexes a string is three functions. Every higher layer —
17
+ ; Intl, base64, digests, JSON quoting, URI escaping, regex — indexes strings, and
18
+ ; each one would otherwise carry its own version of this. It is written once here.
19
+ ;
20
+ ; The addressing this file exposes is UTF-16 CODE UNITS, because that is what
21
+ ; JavaScript's String is defined over and therefore what the engine's semantics
22
+ ; are specified against. `RgText.len("😀")` is 2 on every target, including the
23
+ ; ones whose own strlen says 1 or 4. A slice that cuts a surrogate pair yields
24
+ ; the lone surrogate on a unit-model target and drops the half on a point-model
25
+ ; one, where half a character has no representation — that difference is
26
+ ; documented at `substr` rather than papered over.
27
+ ;
28
+ ; Lifted from ComponentEngine.rgr:39365-39744, which is where these algorithms
29
+ ; were proven; see PLAN_JS_STDLIB_TEXT.md §2.
30
+ ;
31
+ ; TWO TARGET CONSTRAINTS THIS FILE WORKS AROUND
32
+ ;
33
+ ; 1. `strfromcode` used directly as an ARGUMENT does not compile on Rust: the
34
+ ; writer emits `char::from_u32(n)`, a `char`, and then calls a String method
35
+ ; on it (`no method named 'chars' found for type 'char'`). Concatenation
36
+ ; coerces fine. Every `strfromcode` here is therefore BOUND to a `string`
37
+ ; before it is passed anywhere.
38
+ ;
39
+ ; 2. On es6, `strfromcode` is `String.fromCharCode`, which truncates to 16
40
+ ; bits — `strfromcode 128512` gives one BMP character, not an emoji. So
41
+ ; `fromCodePoint` builds a surrogate PAIR there rather than trusting it.
42
+ ;
43
+ ; NAMING: `kind`, `len`, `substr` and the rest avoid the global operator namespace,
44
+ ; which a static method may not enter — see lib/core/README.md.
45
+ ; =============================================================================
46
+
47
+ class RgText {
48
+
49
+ ; ---- which model is this ------------------------------------------------
50
+
51
+ ; 0 = UTF-16 code units, 1 = UTF-8 bytes, 2 = code points.
52
+ ;
53
+ ; Recomputed rather than cached: the core-layer rules forbid mutable
54
+ ; file-level state, and this is two strlen calls on one-character literals.
55
+ ; If a profile ever says it matters, the fix is a value threaded by the
56
+ ; caller, not a static field.
57
+ static sfn kind:int () {
58
+ if ((strlen "é") > 1) {
59
+ return 1
60
+ }
61
+ if ((strlen "😀") == 1) {
62
+ return 2
63
+ }
64
+ return 0
65
+ }
66
+
67
+ static sfn unitsAreBytes:boolean () {
68
+ return ((RgText.kind()) == 1)
69
+ }
70
+
71
+ ; True where strlen/charAt/substring index whole CHARACTERS, so the only gap
72
+ ; to UTF-16 is that a supplementary character is one unit here and two there.
73
+ static sfn unitsArePoints:boolean () {
74
+ return ((RgText.kind()) == 2)
75
+ }
76
+
77
+ ; ---- UTF-8 geometry ----------------------------------------------------
78
+
79
+ static sfn utf8WidthOf:int (cp:int) {
80
+ if (cp < 128) {
81
+ return 1
82
+ }
83
+ if (cp < 2048) {
84
+ return 2
85
+ }
86
+ if (cp < 65536) {
87
+ return 3
88
+ }
89
+ return 4
90
+ }
91
+
92
+ ; Bytes [i, j) of a string on a byte-model target.
93
+ ;
94
+ ; `substring` IS the byte slice there — on a byte-model target strlen,
95
+ ; charAt and substring all index bytes. The obvious alternative, rebuilding
96
+ ; the run byte by byte through `strfromcode`, is WRONG: on C++
97
+ ; `strfromcode 195` answers the two-byte UTF-8 encoding of U+00C3, not the
98
+ ; single byte 0xC3, so slicing "héllo" that way double-encoded it to "él".
99
+ ; Measured; the cross-target suite caught it.
100
+ static sfn byteSlice:string (s:string i:int j:int) {
101
+ def n:int (strlen s)
102
+ def a:int i
103
+ def b:int j
104
+ if (a < 0) {
105
+ a = 0
106
+ }
107
+ if (b > n) {
108
+ b = n
109
+ }
110
+ if (b <= a) {
111
+ return ""
112
+ }
113
+ return (substring s a b)
114
+ }
115
+
116
+ ; The code POINT whose encoding starts at byte `i`, and its width in bytes,
117
+ ; as a two-element list. Byte-model targets only; [-1, 0] past the end.
118
+ static sfn decodeAt:[int] (s:string i:int) {
119
+ def out:[int]
120
+ def n:int (strlen s)
121
+ if (i >= n) {
122
+ push out (0 - 1)
123
+ push out 0
124
+ return out
125
+ }
126
+ def b0:int (bit_and (charAt s i) 255)
127
+ if (b0 < 128) {
128
+ push out b0
129
+ push out 1
130
+ return out
131
+ }
132
+ if (b0 < 224) {
133
+ def c1:int 0
134
+ if ((i + 1) < n) {
135
+ c1 = (bit_and (charAt s (i + 1)) 63)
136
+ }
137
+ push out (((bit_and b0 31) * 64) + c1)
138
+ push out 2
139
+ return out
140
+ }
141
+ if (b0 < 240) {
142
+ def d1:int 0
143
+ def d2:int 0
144
+ if ((i + 1) < n) {
145
+ d1 = (bit_and (charAt s (i + 1)) 63)
146
+ }
147
+ if ((i + 2) < n) {
148
+ d2 = (bit_and (charAt s (i + 2)) 63)
149
+ }
150
+ push out ((((bit_and b0 15) * 4096) + (d1 * 64)) + d2)
151
+ push out 3
152
+ return out
153
+ }
154
+ def e1:int 0
155
+ def e2:int 0
156
+ def e3:int 0
157
+ if ((i + 1) < n) {
158
+ e1 = (bit_and (charAt s (i + 1)) 63)
159
+ }
160
+ if ((i + 2) < n) {
161
+ e2 = (bit_and (charAt s (i + 2)) 63)
162
+ }
163
+ if ((i + 3) < n) {
164
+ e3 = (bit_and (charAt s (i + 3)) 63)
165
+ }
166
+ push out (((((bit_and b0 7) * 262144) + (e1 * 4096)) + (e2 * 64)) + e3)
167
+ push out 4
168
+ return out
169
+ }
170
+
171
+ ; ---- construction ------------------------------------------------------
172
+
173
+ ; One code point as a string. On a byte target that is its UTF-8 encoding;
174
+ ; on a unit target an astral point becomes a SURROGATE PAIR, because
175
+ ; `strfromcode` there is String.fromCharCode and truncates to 16 bits.
176
+ static sfn fromCodePoint:string (cp:int) {
177
+ if (RgText.unitsAreBytes()) {
178
+ return (RgText.encodeUtf8(cp))
179
+ }
180
+ if (RgText.unitsArePoints()) {
181
+ def p:string (strfromcode cp)
182
+ return p
183
+ }
184
+ if (cp < 65536) {
185
+ def q:string (strfromcode cp)
186
+ return q
187
+ }
188
+ def adj:int (cp - 65536)
189
+ def hi:string (strfromcode (55296 + (bit_shr adj 10)))
190
+ def lo:string (strfromcode (56320 + (bit_and adj 1023)))
191
+ return (hi + lo)
192
+ }
193
+
194
+ ; One code point as a string on a byte-model target — that is, as its UTF-8
195
+ ; encoding, since the string IS bytes there.
196
+ ;
197
+ ; `strfromcode` already does exactly this: on C++ it answers the two-byte
198
+ ; encoding of U+00C3 for 195, not the raw byte. So the encoding must NOT be
199
+ ; assembled by hand out of per-byte `strfromcode` calls — that produced
200
+ ; double-encoded UTF-8 ("é" for "é"). Handing the whole code point over is
201
+ ; both correct and shorter.
202
+ ;
203
+ ; A lone surrogate arrives here from `substr` when a cut splits a pair. The
204
+ ; result is CESU-8, which is the only way a byte string can carry half a
205
+ ; character; that is documented at `substr`.
206
+ static sfn encodeUtf8:string (cp:int) {
207
+ def p:string (strfromcode cp)
208
+ return p
209
+ }
210
+
211
+ ; ---- code-unit addressing ----------------------------------------------
212
+
213
+ ; Code units in `s`. This is JavaScript's `.length`.
214
+ static sfn len:int (s:string) {
215
+ def n:int (strlen s)
216
+ if (RgText.unitsArePoints()) {
217
+ def p:int 0
218
+ def cnt:int 0
219
+ while (p < n) {
220
+ if ((charAt s p) > 65535) {
221
+ cnt = (cnt + 2)
222
+ } {
223
+ cnt = (cnt + 1)
224
+ }
225
+ p = (p + 1)
226
+ }
227
+ return cnt
228
+ }
229
+ if (false == (RgText.unitsAreBytes())) {
230
+ return n
231
+ }
232
+ def i:int 0
233
+ def units:int 0
234
+ while (i < n) {
235
+ def b:int (bit_and (charAt s i) 255)
236
+ if (b < 128) {
237
+ units = (units + 1)
238
+ i = (i + 1)
239
+ } {
240
+ ; Anything past U+FFFF needs TWO code units: JavaScript spells
241
+ ; it as a surrogate pair.
242
+ if (b < 224) {
243
+ units = (units + 1)
244
+ i = (i + 2)
245
+ } {
246
+ if (b < 240) {
247
+ units = (units + 1)
248
+ i = (i + 3)
249
+ } {
250
+ units = (units + 2)
251
+ i = (i + 4)
252
+ }
253
+ }
254
+ }
255
+ }
256
+ return units
257
+ }
258
+
259
+ ; The code UNIT at index `u`: the character itself inside the BMP, and the
260
+ ; high or low surrogate for an astral one. -1 past the end.
261
+ static sfn unitAt:int (s:string u:int) {
262
+ def n:int (strlen s)
263
+ if (u < 0) {
264
+ return (0 - 1)
265
+ }
266
+ if (RgText.unitsArePoints()) {
267
+ def p:int 0
268
+ def units:int 0
269
+ while (p < n) {
270
+ def cp:int (charAt s p)
271
+ if (cp < 65536) {
272
+ if (units == u) {
273
+ return cp
274
+ }
275
+ units = (units + 1)
276
+ } {
277
+ def adj:int (cp - 65536)
278
+ if (units == u) {
279
+ return (55296 + (bit_shr adj 10))
280
+ }
281
+ if ((units + 1) == u) {
282
+ return (56320 + (bit_and adj 1023))
283
+ }
284
+ units = (units + 2)
285
+ }
286
+ p = (p + 1)
287
+ }
288
+ return (0 - 1)
289
+ }
290
+ if (false == (RgText.unitsAreBytes())) {
291
+ if (u >= n) {
292
+ return (0 - 1)
293
+ }
294
+ return (bit_and (charAt s u) 65535)
295
+ }
296
+ def i:int 0
297
+ def units:int 0
298
+ while (i < n) {
299
+ def dec:[int] (RgText.decodeAt(s i))
300
+ def cp:int (itemAt dec 0)
301
+ def w:int (itemAt dec 1)
302
+ if (w == 0) {
303
+ return (0 - 1)
304
+ }
305
+ if (cp < 65536) {
306
+ if (units == u) {
307
+ return cp
308
+ }
309
+ units = (units + 1)
310
+ } {
311
+ def adj:int (cp - 65536)
312
+ if (units == u) {
313
+ return (55296 + (bit_shr adj 10))
314
+ }
315
+ if ((units + 1) == u) {
316
+ return (56320 + (bit_and adj 1023))
317
+ }
318
+ units = (units + 2)
319
+ }
320
+ i = (i + w)
321
+ }
322
+ return (0 - 1)
323
+ }
324
+
325
+ ; The substring between two code-unit indexes.
326
+ ;
327
+ ; A cut that lands INSIDE a surrogate pair yields the corresponding lone
328
+ ; surrogate on a unit-model target, which is what keeps `s.slice(0, 1)` on an
329
+ ; astral character one unit long instead of silently the whole character. On a
330
+ ; POINT-model target half a pair has no representation — that string type
331
+ ; holds characters — so the half is dropped rather than faked, which keeps the
332
+ ; result well-formed. Slicing that does not split a pair, which is everything
333
+ ; a program normally does, is exact on every target.
334
+ static sfn substr:string (s:string a:int b:int) {
335
+ def lo:int a
336
+ def hi:int b
337
+ if (lo < 0) {
338
+ lo = 0
339
+ }
340
+ if (hi < lo) {
341
+ hi = lo
342
+ }
343
+ def n:int (strlen s)
344
+ if (RgText.unitsArePoints()) {
345
+ def out:string ""
346
+ def p:int 0
347
+ def units:int 0
348
+ while (p < n) {
349
+ def cp:int (charAt s p)
350
+ def w:int 1
351
+ if (cp > 65535) {
352
+ w = 2
353
+ }
354
+ if ((units >= lo) && ((units + w) <= hi)) {
355
+ out = (out + (at s p))
356
+ }
357
+ units = (units + w)
358
+ p = (p + 1)
359
+ }
360
+ return out
361
+ }
362
+ if (false == (RgText.unitsAreBytes())) {
363
+ if (lo >= n) {
364
+ return ""
365
+ }
366
+ def hi2:int hi
367
+ if (hi2 > n) {
368
+ hi2 = n
369
+ }
370
+ return (substring s lo hi2)
371
+ }
372
+ def out2:string ""
373
+ def i:int 0
374
+ def units2:int 0
375
+ while (i < n) {
376
+ def dec:[int] (RgText.decodeAt(s i))
377
+ def cp:int (itemAt dec 0)
378
+ def w:int (itemAt dec 1)
379
+ if (w == 0) {
380
+ i = n
381
+ } {
382
+ if (cp < 65536) {
383
+ if ((units2 >= lo) && (units2 < hi)) {
384
+ out2 = (out2 + (RgText.byteSlice(s i (i + w))))
385
+ }
386
+ units2 = (units2 + 1)
387
+ } {
388
+ def adj:int (cp - 65536)
389
+ def takeHi:boolean ((units2 >= lo) && (units2 < hi))
390
+ def takeLo:boolean (((units2 + 1) >= lo) && ((units2 + 1) < hi))
391
+ if (takeHi && takeLo) {
392
+ out2 = (out2 + (RgText.byteSlice(s i (i + w))))
393
+ } {
394
+ ; One half of a pair, encoded as the surrogate it is.
395
+ ; CESU-8 rather than UTF-8, which is the only way a byte
396
+ ; string can carry half a character at all.
397
+ if takeHi {
398
+ out2 = (out2 + (RgText.encodeUtf8((55296 + (bit_shr adj 10)))))
399
+ }
400
+ if takeLo {
401
+ out2 = (out2 + (RgText.encodeUtf8((56320 + (bit_and adj 1023)))))
402
+ }
403
+ }
404
+ units2 = (units2 + 2)
405
+ }
406
+ i = (i + w)
407
+ }
408
+ }
409
+ return out2
410
+ }
411
+
412
+ ; ---- UTF-8 offsets -----------------------------------------------------
413
+ ;
414
+ ; These convert between a UTF-16 code-unit index and a UTF-8 BYTE offset.
415
+ ; Both meanings are target-independent: the answer for "a😀b" is the same
416
+ ; number on es6 as on C++, even though the two disagree about what their own
417
+ ; strlen counts.
418
+ ;
419
+ ; The engine's `cuByteOf` that these grew out of meant something subtly
420
+ ; different — "the offset the TARGET'S OWN string search reports" — which is
421
+ ; code units on es6 and bytes elsewhere. That contract is target-relative by
422
+ ; construction, and it is why the first version of this file answered 3 on
423
+ ; es6 and 5 everywhere else for the same call. A bridge to native search
424
+ ; offsets is a real need, but it belongs to whoever is calling the native
425
+ ; search, not to a portable library.
426
+
427
+ ; The UTF-8 byte offset at which code unit `u` begins; the encoded length
428
+ ; when `u` is at or past the end. A `u` that lands inside a surrogate pair
429
+ ; reports the offset just past that pair — half a character has no offset.
430
+ static sfn utf8ByteOfUnit:int (s:string u:int) {
431
+ if (u <= 0) {
432
+ return 0
433
+ }
434
+ def n:int (RgText.len(s))
435
+ def bytes:int 0
436
+ def i:int 0
437
+ while (i < n) {
438
+ if (i >= u) {
439
+ return bytes
440
+ }
441
+ def cp:int (RgText.codePointAt(s i))
442
+ if (cp < 0) {
443
+ return bytes
444
+ }
445
+ def w:int 1
446
+ if (cp > 65535) {
447
+ w = 2
448
+ }
449
+ bytes = (bytes + (RgText.utf8WidthOf(cp)))
450
+ i = (i + w)
451
+ }
452
+ return bytes
453
+ }
454
+
455
+ ; The inverse: the code-unit index that a UTF-8 byte offset corresponds to.
456
+ ; What a search that ran over the encoded bytes has to report back.
457
+ static sfn unitOfUtf8Byte:int (s:string byteIdx:int) {
458
+ if (byteIdx <= 0) {
459
+ return 0
460
+ }
461
+ def n:int (RgText.len(s))
462
+ def bytes:int 0
463
+ def units:int 0
464
+ def i:int 0
465
+ while (i < n) {
466
+ if (bytes >= byteIdx) {
467
+ return units
468
+ }
469
+ def cp:int (RgText.codePointAt(s i))
470
+ if (cp < 0) {
471
+ return units
472
+ }
473
+ def w:int 1
474
+ if (cp > 65535) {
475
+ w = 2
476
+ }
477
+ bytes = (bytes + (RgText.utf8WidthOf(cp)))
478
+ units = (units + w)
479
+ i = (i + w)
480
+ }
481
+ return units
482
+ }
483
+
484
+ ; ---- code points -------------------------------------------------------
485
+
486
+ ; The full code point beginning at code-unit index `u`, combining a surrogate
487
+ ; pair when one starts there. This is JavaScript's codePointAt.
488
+ static sfn codePointAt:int (s:string u:int) {
489
+ def first:int (RgText.unitAt(s u))
490
+ if (first < 0) {
491
+ return (0 - 1)
492
+ }
493
+ if (first < 55296) {
494
+ return first
495
+ }
496
+ if (first > 56319) {
497
+ return first
498
+ }
499
+ def second:int (RgText.unitAt(s (u + 1)))
500
+ if (second < 56320) {
501
+ return first
502
+ }
503
+ if (second > 57343) {
504
+ return first
505
+ }
506
+ return ((((first - 55296) * 1024) + (second - 56320)) + 65536)
507
+ }
508
+
509
+ static sfn toCodePoints:[int] (s:string) {
510
+ def out:[int]
511
+ def n:int (RgText.len(s))
512
+ def u:int 0
513
+ while (u < n) {
514
+ def cp:int (RgText.codePointAt(s u))
515
+ if (cp < 0) {
516
+ u = n
517
+ } {
518
+ push out cp
519
+ if (cp > 65535) {
520
+ u = (u + 2)
521
+ } {
522
+ u = (u + 1)
523
+ }
524
+ }
525
+ }
526
+ return out
527
+ }
528
+
529
+ static sfn codePointCount:int (s:string) {
530
+ def cps:[int] (RgText.toCodePoints(s))
531
+ return (array_length cps)
532
+ }
533
+
534
+ static sfn fromCodePoints:string (cps:[int]) {
535
+ def out:string ""
536
+ def i:int 0
537
+ while (i < (array_length cps)) {
538
+ out = (out + (RgText.fromCodePoint((itemAt cps i))))
539
+ i = (i + 1)
540
+ }
541
+ return out
542
+ }
543
+
544
+ ; ---- code units as data ------------------------------------------------
545
+
546
+ static sfn toCodeUnits:[int] (s:string) {
547
+ def out:[int]
548
+ def n:int (RgText.len(s))
549
+ def u:int 0
550
+ while (u < n) {
551
+ push out (RgText.unitAt(s u))
552
+ u = (u + 1)
553
+ }
554
+ return out
555
+ }
556
+
557
+ static sfn fromCodeUnits:string (units:[int]) {
558
+ def out:string ""
559
+ def i:int 0
560
+ def n:int (array_length units)
561
+ while (i < n) {
562
+ def hi:int (itemAt units i)
563
+ ; Recombine a well-formed pair so a byte-model target writes UTF-8
564
+ ; rather than CESU-8 for the astral characters.
565
+ def paired:boolean false
566
+ if ((hi >= 55296) && (hi <= 56319)) {
567
+ if ((i + 1) < n) {
568
+ def lo:int (itemAt units (i + 1))
569
+ if ((lo >= 56320) && (lo <= 57343)) {
570
+ def cp:int ((((hi - 55296) * 1024) + (lo - 56320)) + 65536)
571
+ out = (out + (RgText.fromCodePoint(cp)))
572
+ i = (i + 2)
573
+ paired = true
574
+ }
575
+ }
576
+ }
577
+ if (false == paired) {
578
+ out = (out + (RgText.fromCodePoint(hi)))
579
+ i = (i + 1)
580
+ }
581
+ }
582
+ return out
583
+ }
584
+
585
+ ; ---- UTF-8 bytes -------------------------------------------------------
586
+ ; The one bridge between text and bytes in lib/core. RgBase and RgCrypto go
587
+ ; through this and nothing else, so the string model is dealt with in exactly
588
+ ; one place rather than once per digest and once per encoder.
589
+
590
+ static sfn toUtf8Bytes:[int] (s:string) {
591
+ def out:[int]
592
+ ; A byte-model string already IS the encoding.
593
+ if (RgText.unitsAreBytes()) {
594
+ def n:int (strlen s)
595
+ def i:int 0
596
+ while (i < n) {
597
+ push out (bit_and (charAt s i) 255)
598
+ i = (i + 1)
599
+ }
600
+ return out
601
+ }
602
+ def cps:[int] (RgText.toCodePoints(s))
603
+ def k:int 0
604
+ while (k < (array_length cps)) {
605
+ def cp:int (itemAt cps k)
606
+ if (cp < 128) {
607
+ push out cp
608
+ } {
609
+ if (cp < 2048) {
610
+ push out (bit_or 192 (bit_shr cp 6))
611
+ push out (bit_or 128 (bit_and cp 63))
612
+ } {
613
+ if (cp < 65536) {
614
+ push out (bit_or 224 (bit_shr cp 12))
615
+ push out (bit_or 128 (bit_and (bit_shr cp 6) 63))
616
+ push out (bit_or 128 (bit_and cp 63))
617
+ } {
618
+ push out (bit_or 240 (bit_shr cp 18))
619
+ push out (bit_or 128 (bit_and (bit_shr cp 12) 63))
620
+ push out (bit_or 128 (bit_and (bit_shr cp 6) 63))
621
+ push out (bit_or 128 (bit_and cp 63))
622
+ }
623
+ }
624
+ }
625
+ k = (k + 1)
626
+ }
627
+ return out
628
+ }
629
+
630
+ static sfn fromUtf8Bytes:string (bytes:[int]) {
631
+ def cps:[int]
632
+ def n:int (array_length bytes)
633
+ def i:int 0
634
+ while (i < n) {
635
+ def b0:int (bit_and (itemAt bytes i) 255)
636
+ if (b0 < 128) {
637
+ push cps b0
638
+ i = (i + 1)
639
+ } {
640
+ if (b0 < 224) {
641
+ def c1:int 0
642
+ if ((i + 1) < n) {
643
+ c1 = (bit_and (itemAt bytes (i + 1)) 63)
644
+ }
645
+ push cps (((bit_and b0 31) * 64) + c1)
646
+ i = (i + 2)
647
+ } {
648
+ if (b0 < 240) {
649
+ def d1:int 0
650
+ def d2:int 0
651
+ if ((i + 1) < n) {
652
+ d1 = (bit_and (itemAt bytes (i + 1)) 63)
653
+ }
654
+ if ((i + 2) < n) {
655
+ d2 = (bit_and (itemAt bytes (i + 2)) 63)
656
+ }
657
+ push cps ((((bit_and b0 15) * 4096) + (d1 * 64)) + d2)
658
+ i = (i + 3)
659
+ } {
660
+ def e1:int 0
661
+ def e2:int 0
662
+ def e3:int 0
663
+ if ((i + 1) < n) {
664
+ e1 = (bit_and (itemAt bytes (i + 1)) 63)
665
+ }
666
+ if ((i + 2) < n) {
667
+ e2 = (bit_and (itemAt bytes (i + 2)) 63)
668
+ }
669
+ if ((i + 3) < n) {
670
+ e3 = (bit_and (itemAt bytes (i + 3)) 63)
671
+ }
672
+ push cps (((((bit_and b0 7) * 262144) + (e1 * 4096)) + (e2 * 64)) + e3)
673
+ i = (i + 4)
674
+ }
675
+ }
676
+ }
677
+ }
678
+ return (RgText.fromCodePoints(cps))
679
+ }
680
+ }