ranger-compiler 3.2.0 → 3.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (61) hide show
  1. package/CHANGELOG.md +2463 -0
  2. package/LICENSE +28 -0
  3. package/LICENSE-MIT +21 -0
  4. package/README.md +1651 -1895
  5. package/dist/Lang.rgr +10946 -5852
  6. package/dist/README.md +3 -2
  7. package/dist/api.d.ts +1548 -27
  8. package/dist/api.js +70572 -38295
  9. package/dist/git-http.mjs +126 -0
  10. package/dist/lib/ACEEditor.rgr +2 -0
  11. package/dist/lib/Ajax.rgr +2 -0
  12. package/dist/lib/CmdParams.rgr +2 -0
  13. package/dist/lib/Crypto.rgr +2 -0
  14. package/dist/lib/DOMLib.rgr +2 -0
  15. package/dist/lib/Engine3D.rgr +2 -0
  16. package/dist/lib/ImmutableVector.rgr +2 -0
  17. package/dist/lib/IndexedDB.rgr +2 -0
  18. package/dist/lib/IsoDate/DateMath.rgr +2 -0
  19. package/dist/lib/IsoDate/IsoCalendar.rgr +2 -0
  20. package/dist/lib/IsoDate/IsoDateParse.rgr +2 -0
  21. package/dist/lib/IsoDateLib.rgr +2 -0
  22. package/dist/lib/JSON.rgr +4906 -48
  23. package/dist/lib/JinxProcess.rgr +2 -0
  24. package/dist/lib/RangerProcess.rgr +2 -0
  25. package/dist/lib/Regex/RegexMatch.rgr +2 -0
  26. package/dist/lib/RegexLib.rgr +2 -0
  27. package/dist/lib/SQL.rgr +2 -0
  28. package/dist/lib/ServiceLib.rgr +2 -0
  29. package/dist/lib/Shell.rgr +326 -0
  30. package/dist/lib/Storage.rgr +2 -0
  31. package/dist/lib/Time.rgr +2 -0
  32. package/dist/lib/Timers.rgr +2 -0
  33. package/dist/lib/TypedArrays.rgr +2 -0
  34. package/dist/lib/ViewLib.rgr +2 -0
  35. package/dist/lib/WebLib.rgr +2 -0
  36. package/dist/lib/WebServerLib.rgr +2 -0
  37. package/dist/lib/apple/AppleAppBuilder.rgr +572 -0
  38. package/dist/lib/apple/AppleAppSpec.rgr +201 -0
  39. package/dist/lib/apple/AppleDevice.rgr +295 -0
  40. package/dist/lib/apple/AppleDeviceDoctor.rgr +453 -0
  41. package/dist/lib/apple/AppleSigning.rgr +379 -0
  42. package/dist/lib/apple/AppleSimulator.rgr +248 -0
  43. package/dist/lib/apple/AppleTarget.rgr +185 -0
  44. package/dist/lib/apple/AppleToolchain.rgr +458 -0
  45. package/dist/lib/apple/README.md +311 -0
  46. package/dist/lib/apple/apple_test.rgr +669 -0
  47. package/dist/lib/core/README.md +251 -0
  48. package/dist/lib/core/RgBase.rgr +313 -0
  49. package/dist/lib/core/RgNum.rgr +653 -0
  50. package/dist/lib/core/RgText.rgr +680 -0
  51. package/dist/lib/core/RgU32.rgr +309 -0
  52. package/dist/lib/ranger-dir.rgr +2 -0
  53. package/dist/lib/shell_test.rgr +193 -0
  54. package/dist/lib/stdlib.rgr +1176 -665
  55. package/dist/lib/stdops.rgr +2 -0
  56. package/dist/lib/zip/Inflate.rgr +675 -0
  57. package/dist/lib/zip/ZipBuffer.rgr +358 -0
  58. package/dist/package.json +1 -1
  59. package/dist/rgrc.js +76333 -40647
  60. package/dist/stdops.rgr +2 -0
  61. package/package.json +1121 -339
@@ -0,0 +1,251 @@
1
+ # `lib/core` — the Ranger Core API
2
+
3
+ Portable, dependency-free Ranger classes for the things every language's standard
4
+ library has and Ranger does not: IEEE-754 arithmetic, 32-bit integer algebra,
5
+ string/Unicode handling, dates, formatting, hashing.
6
+
7
+ One implementation, compiled to every target. No `systemclass`, no operator
8
+ templates, no per-target branches.
9
+
10
+ **Status: first slice, wired into the JS engine.** `RgNum`, `RgU32`, `RgText`
11
+ and `RgBase` are landed and gated, and `ComponentEngine.rgr` now delegates 16 of
12
+ its own functions here instead of carrying copies (268 lines deleted). The rest
13
+ is planned in [PLAN_JS_STDLIB.md](../../PLAN_JS_STDLIB.md).
14
+
15
+ | File | What | State |
16
+ | --- | --- | --- |
17
+ | `RgNum.rgr` | IEEE-754 doubles: rounding, `exp`, `log`, `pow`, signed zero, ToInt32/ToUint32, float32 | landed |
18
+ | `RgU32.rgr` | 32-bit unsigned algebra: and/or/xor/not, shifts, rotations, add/sub, clz, popcount, big-endian bytes | landed |
19
+ | `RgText.rgr` | UTF-16 code-unit addressing over all three string models; code points; UTF-8 bytes | landed |
20
+ | `RgBase.rgr` | hex, base64, base64url, and `RgBytesResult` | landed |
21
+ | `RgDate.rgr` | ES5 time algebra, ISO parsing and formatting | planned |
22
+ | `RgIntl.rgr` | collation, number and date formatting over CLDR | planned |
23
+ | `RgCrypto.rgr` | SHA-2, HMAC, PBKDF2, secure random, UUID | planned |
24
+
25
+ ## Why this is not `lib/stdlib.rgr`
26
+
27
+ `lib/stdlib.rgr` and `lib/stdops.rgr` contain **only operator templates** — no
28
+ classes at all. "stdlib" in this repository already means "language-level
29
+ operator definitions with a per-target template each". This directory is the
30
+ opposite: plain Ranger classes with one body that every target compiles.
31
+
32
+ The rest of `lib/` is mostly host shims — `lib/Time.rgr` is `es6`-only,
33
+ `lib/Crypto.rgr` has a `java7` template and nothing else. Those work on the
34
+ target they were written for. `lib/core` is for code that has to work on all of
35
+ them.
36
+
37
+ ## Rules for a file in here
38
+
39
+ 1. **No host operators with partial coverage.** If an operator has no `*`
40
+ fallback and is missing on a target, it does not belong in a signature here.
41
+ 2. **No `EvHandle`, `EvalValue`, `TSNode` or any engine type.** These are
42
+ callable from a native Ranger program with no interpreter present.
43
+ 3. **Primitive types only** in signatures: `int`, `double`, `boolean`, `string`,
44
+ `char`, `[T]`, and classes declared in `lib/core/`.
45
+ 4. **No `throw`.** Where JavaScript throws, return a result carrier — the value
46
+ plus an error kind — so the same function can become a JS exception on one
47
+ side and an ordinary check on the other.
48
+ 5. **No mutable file-level state**, so two callers cannot interfere.
49
+ 6. **Every claim gets a vector.** Anything asserted about behaviour goes in
50
+ `tests/native/core_vectors.rgr` and is compared against Node and across
51
+ targets. See [Testing](#testing).
52
+
53
+ ### Naming: the `D` suffix, and why
54
+
55
+ **A static method may not take the name of a global operator.** The resolver
56
+ rewrites `RgNum.floor(x)` into the operator call and then reports
57
+ `Class RgNum does not have method floor`. `EvValueBridge.rgr` documents the same
58
+ trap for `isArray`.
59
+
60
+ Nine names are affected: `floor`, `ceil`, `sqrt`, `sin`, `cos`, `tan`, `asin`,
61
+ `acos`, `atan2`. Each carries a `D` suffix here — `RgNum.floorD`, `RgNum.ceilD`,
62
+ `RgNum.sqrtD` — which is the convention `DateTime.rgr` already uses for `floorD`,
63
+ `truncD` and `modD`. For `floor` and `ceil` the suffix marks a real distinction:
64
+ **the operators return `int`**, and these return `double`.
65
+
66
+ Two more names collide for the same reason and are spelled around it: `wrap` (the
67
+ optional-wrapping operator) is `RgU32.wrap32`, and `add` is `RgU32.addU`.
68
+ Instance methods are unaffected — only statics enter the operator namespace.
69
+
70
+ ### Class names the compiler has already taken
71
+
72
+ Separately from the operator namespace, `compiler/Lang.rgr` **emits helper
73
+ classes into target output**, and a Ranger class of the same name lands beside
74
+ them in the same file. These are reserved:
75
+
76
+ ```
77
+ RgFiles RgHash RgList RgParse RgPath RgSort RgStr RgUtf8
78
+ ```
79
+
80
+ `RgStr` is why this file's string class is called **`RgText`** — the C# template
81
+ for `strsplit` emits `static class RgStr { … Split … }` (`Lang.rgr:4645`), so any
82
+ C# program using both would have declared the class twice. Nothing caught it,
83
+ because C# has no toolchain on the machine the suite runs on; it turned up on a
84
+ grep. Check a new class name against this list, not only against the operators:
85
+
86
+ ```
87
+ grep -oE "(static )?class (Rg[A-Za-z0-9_]*)" compiler/Lang.rgr | sort -u
88
+ ```
89
+
90
+ Before adding a method, check the name:
91
+
92
+ ```
93
+ node -e 'const fs=require("fs");const n=process.argv[1];
94
+ const re=/^\s{4,20}([a-zA-Z_][\w]*)\s+(?:_|[a-zA-Z_][\w]*)(?:@\([^)]*\))?\s*:/;
95
+ for(const f of ["compiler/Lang.rgr","lib/stdops.rgr","lib/stdlib.rgr"])
96
+ for(const l of fs.readFileSync(f,"utf8").split("\n")){const m=l.match(re);
97
+ if(m&&m[1]===n)console.log("COLLIDES with an operator in",f);}' floor
98
+ ```
99
+
100
+ ## What was measured
101
+
102
+ Everything below came out of running the same Ranger source on five targets. It
103
+ is written down because each item silently rules out an implementation that looks
104
+ obviously correct.
105
+
106
+ ### `int` is not one type
107
+
108
+ | | es6 | python | go | cpp | rust |
109
+ | --- | --- | --- | --- | --- | --- |
110
+ | `2147483647 + 1` | 2147483648 | 2147483648 | 2147483648 | **-2147483648** | 64-bit |
111
+ | `bit_shl 1 31` | **-2147483648** | 2147483648 | 2147483648 | -2147483648 | 64-bit |
112
+ | `bit_shl 1 40` | **256** | 1099511627776 | 1099511627776 | **0** | 64-bit |
113
+ | `bit_ushr (0-1) 4` | 268435455 | 268435455 | **1152921504606846975** | 64-bit | 64-bit |
114
+
115
+ - **`int` is 32-bit signed on C++.** Plain *arithmetic* overflows there. A u32
116
+ cannot be held as an `int` in `[0, 2^32)`, because that range does not exist.
117
+ - **`bit_shl` is not portable.** On es6 it is JavaScript's `<<`: the result wraps
118
+ to 32 bits and the **count is taken mod 32**, so `1 << 40` is 256. On C++ it is
119
+ undefined past the width and gives 0. On go/python/rust it grows unbounded.
120
+ - **`bit_ushr` is three different operators** — 32-bit on es6 and python, 64-bit
121
+ on go, cpp, rust and swift. `RgU32` does not use it at all.
122
+
123
+ `RgU32` therefore carries a u32 as a **signed 32-bit bit pattern**, keeps every
124
+ intermediate under 2^31 by splitting into 16-bit halves, and never writes a
125
+ literal above 2147483647.
126
+
127
+ ### Double literals may be emitted as int literals
128
+
129
+ `def z:double 0.0` compiles to `z = 0` on the **Python** target — an int. So
130
+
131
+ ```ranger
132
+ def z:double 0.0
133
+ return (z * (0.0 - 1.0)) ; +0 on Python. The sign was gone a line earlier.
134
+ ```
135
+
136
+ The C++ writer has the same habit for bare literals, which is why the JS engine's
137
+ `negativeZero()` bound the value first. That fixes C++ and not Python.
138
+
139
+ **A division cannot be folded into an int on any target**, because every one of
140
+ them routes double division through a helper (Python's `/` raises
141
+ `ZeroDivisionError`, so `r_div_f64` is already generated). `RgNum.negZero()` is
142
+ therefore `-1 / Infinity`, and the cross-target suite is what found this.
143
+
144
+ ### Strings have three models, and all three are live
145
+
146
+ | `strlen` | es6 | python | go | rust | cpp |
147
+ | --- | --- | --- | --- | --- | --- |
148
+ | `"é"` | 1 | 1 | 1 | 1 | **2** |
149
+ | `"😀"` | **2** | 1 | 1 | 1 | **4** |
150
+
151
+ es6 counts UTF-16 code units, python/go/rust count code points, C++ counts UTF-8
152
+ bytes. `charAt` follows suit: on `"é"` it answers 233 on the first four and 195
153
+ on C++. `RgText` encapsulates all three and exposes UTF-16 code units, so
154
+ `RgText.len("😀")` is 2 everywhere.
155
+
156
+ ### `strfromcode` encodes a code point; it cannot emit a byte
157
+
158
+ On C++, `strfromcode 195` answers the **two-byte** UTF-8 encoding of U+00C3, not
159
+ the single byte 0xC3. So a UTF-8 encoder assembled from per-byte `strfromcode`
160
+ calls — which is the obvious way to write one, and the way the JS engine writes
161
+ it — double-encodes: `"héllo"` came back as `"éllo"`. Two consequences, both in
162
+ `RgText`:
163
+
164
+ - `encodeUtf8(cp)` is just `strfromcode cp`, because that already IS the UTF-8
165
+ encoding on a byte-model target.
166
+ - `byteSlice` uses `substring`, which is byte-indexed on a byte-model target,
167
+ rather than rebuilding the run byte by byte.
168
+
169
+ Separately, `strfromcode` used **directly as an argument** does not compile on
170
+ Rust (the writer emits a `char` and then calls a `String` method on it), so every
171
+ call here is bound to a `string` first. And on es6 it is `String.fromCharCode`,
172
+ which truncates to 16 bits — `fromCodePoint` builds a surrogate pair there.
173
+
174
+ ### Mutating an array parameter does not reach the caller on Go
175
+
176
+ ```ranger
177
+ sfn appendAll:void (into:[string] more:[string]) {
178
+ push into (itemAt more 0) ; the caller never sees this on Go
179
+ }
180
+ ```
181
+
182
+ A Go slice is passed by value and `append` rebinds the local. The symptom is
183
+ silent: every individual vector set was correct while the combined `all` set came
184
+ out **empty**, on Go only. Functions here return new arrays instead.
185
+
186
+ ### "Byte offset" has to mean UTF-8, not "whatever the target indexes by"
187
+
188
+ The engine's `cuByteOf` means "the offset the target's own string search
189
+ reports" — code units on es6, bytes elsewhere. Lifted unchanged, it answered 3 on
190
+ es6 and 5 everywhere else for the same call, because that contract is
191
+ target-relative by construction. `RgText.utf8ByteOfUnit` and `unitOfUtf8Byte`
192
+ are defined over UTF-8 on every target instead. A bridge to native search offsets
193
+ is a real need, but it belongs to whoever is calling the native search.
194
+
195
+ ### There is no `exp`, `log`, `pow`, `round`, `min` or `atan` operator
196
+
197
+ `sqrt`, `sin`, `cos`, `tan`, `asin`, `acos`, `atan2` and `fabs` exist as host
198
+ operators. `exp`, `log` and `pow` **do not exist on any target**, and
199
+ `floor`/`ceil` return `int`. So `RgNum` implements them: `exp` by range reduction
200
+ plus a Taylor series, `log` by reduction plus the atanh series, `pow` by squaring
201
+ for integer exponents and the identity otherwise.
202
+
203
+ The trade is explicit. These land **1 ULP** from each platform's libm — so they
204
+ are *not* bit-identical to `Math.exp` in Node. In exchange they are **identical
205
+ across all five targets**, which delegating to each platform's libm would not
206
+ have been. For a program that has to agree with itself on Go and on Node, that is
207
+ the direction worth being wrong in. The suite asserts the ULP bound rather than
208
+ equality, so a regression in the series cannot hide behind "it was never exact".
209
+
210
+ ## Testing
211
+
212
+ ```bash
213
+ npm run test:core # oracle + every target
214
+ ```
215
+
216
+ Five targets are **built and run**: es6, python, go, cpp, rust. The other five —
217
+ csharp, kotlin, dart, swift6, java7 — have no toolchain on the machine this was
218
+ developed on, so they are only checked as far as *the Ranger compiler writing
219
+ them without error*, which is still worth doing: it is how the `RgStr` clash
220
+ above would have been caught. Scala does not write, because `shell_arg` and
221
+ `shell_arg_cnt` have no Scala template — a gap in the harness's `main`, not in
222
+ `lib/core`, and Scala is outside the large-program CI path anyway.
223
+
224
+ Three legs, in `tests/core-targets.test.ts` over
225
+ `tests/native/core_vectors.rgr`:
226
+
227
+ 1. **Node as oracle.** Every expectation is computed by the JavaScript built-in
228
+ the function mirrors, never hand-written. `CONFORMANCE.md` records
229
+ hand-written expectations encoding a misunderstanding three separate times.
230
+ 2. **Byte-for-byte across targets.** Compiled to es6, python, go, cpp and rust;
231
+ built and run where the toolchain is on `PATH`; output compared character for
232
+ character against the es6 run.
233
+ 3. **The `D` bound** on `exp`/`log`, asserted as a bound.
234
+
235
+ Doubles are never printed directly — each target's own number formatter
236
+ disagrees about trailing `.0`, exponent spelling and digit count. `showD`
237
+ decomposes a double into sign, a 52-bit mantissa in two halves, and a binary
238
+ exponent, and the test does the same decomposition in JavaScript.
239
+
240
+ Current reading: **173 vectors, identical on es6 / python / go / cpp / rust;
241
+ 170 of them exact against Node**, the other three being the `exp`/`log` series.
242
+
243
+ The second gate is the JS engine itself. `ComponentEngine.rgr` delegates to this
244
+ directory, so `npm run test:tsengine` — 45,221 lines of interpreter compiled to
245
+ six targets and checked against Node's answers for seven workloads — is also a
246
+ test of `lib/core`. Verified by hand here (vitest is not installed on this
247
+ machine): all seven answers correct on es6 and on Go, and the engine still writes
248
+ for kotlin, csharp, dart and swift6.
249
+
250
+ `strmodel` is the one set that is reported rather than compared — `RgText.kind()`
251
+ is 0 on es6, 2 on python/go/rust and 1 on cpp, and that is the point.
@@ -0,0 +1,313 @@
1
+ ; SPDX-License-Identifier: MIT
2
+
3
+ ; =============================================================================
4
+ ; RgBase — hex, base64 and base64url
5
+ ; =============================================================================
6
+ ; There is no portable base64 in Ranger. There is no portable hex either. Both
7
+ ; are wanted by anything that moves bytes: data URIs, JWTs, HTTP basic auth,
8
+ ; digests printed for a human, binary embedded in JSON. `RgCrypto` needs all
9
+ ; three, and `btoa`/`atob` on the JS side map straight onto them.
10
+ ;
11
+ ; Bytes are `[int]` with each element in [0, 255]. Not `string`, because a Ranger
12
+ ; string is bytes on C++, code points on python/go/rust and UTF-16 units on es6
13
+ ; (see RgText) — "base64 of a string" would encode different input per target,
14
+ ; which is the exact bug this library exists to prevent. Text goes through
15
+ ; `RgText.toUtf8Bytes` first, which is the single place the string model is
16
+ ; handled.
17
+ ;
18
+ ; Decoding returns `RgBytesResult` rather than throwing: the core layer has no
19
+ ; `throw` (see lib/core/README.md), so a caller checks `ok` and the JS binding
20
+ ; turns `errorKind` into the exception the spec names.
21
+ ;
22
+ ; RFC 4648 throughout, including the parts people skip: padding is REQUIRED and
23
+ ; validated for base64, forbidden for base64url, and a non-zero tail bit in the
24
+ ; final quantum is rejected rather than silently discarded.
25
+ ; =============================================================================
26
+
27
+ Import "RgText.rgr"
28
+
29
+ ; A decode either produced bytes or says why it did not.
30
+ class RgBytesResult {
31
+ def ok:boolean false
32
+ def value:[int]
33
+ ; "" when ok. Otherwise the DOM/JS error name the binding should throw:
34
+ ; InvalidCharacterError for a bad character or a bad length.
35
+ def errorKind:string ""
36
+ def errorMessage:string ""
37
+
38
+ static sfn good:RgBytesResult (bytes:[int]) {
39
+ def r (new RgBytesResult)
40
+ r.ok = true
41
+ r.value = bytes
42
+ return r
43
+ }
44
+
45
+ static sfn bad:RgBytesResult (kind:string message:string) {
46
+ def r (new RgBytesResult)
47
+ r.ok = false
48
+ r.errorKind = kind
49
+ r.errorMessage = message
50
+ return r
51
+ }
52
+ }
53
+
54
+ class RgBase {
55
+
56
+ ; ---- alphabets ---------------------------------------------------------
57
+ ; As strings rather than arrays so the tables cost one literal each. Indexed
58
+ ; with RgText.unitAt, which is ASCII-safe on every model.
59
+
60
+ static sfn hexDigits:string () {
61
+ return "0123456789abcdef"
62
+ }
63
+
64
+ static sfn b64Alphabet:string () {
65
+ return "ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789+/"
66
+ }
67
+
68
+ static sfn b64UrlAlphabet:string () {
69
+ return "ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789-_"
70
+ }
71
+
72
+ ; The value of one base64 character, or -1. A table scan rather than
73
+ ; arithmetic on ranges: the two alphabets differ only in the last two
74
+ ; entries, so one function serves both and there is no second place for the
75
+ ; `+/` versus `-_` distinction to be got wrong.
76
+ static sfn b64ValueOf:int (alphabet:string ch:int) {
77
+ def i:int 0
78
+ while (i < 64) {
79
+ if ((RgText.unitAt(alphabet i)) == ch) {
80
+ return i
81
+ }
82
+ i = (i + 1)
83
+ }
84
+ return (0 - 1)
85
+ }
86
+
87
+ static sfn hexValueOf:int (ch:int) {
88
+ if ((ch >= 48) && (ch <= 57)) {
89
+ return (ch - 48)
90
+ }
91
+ if ((ch >= 97) && (ch <= 102)) {
92
+ return ((ch - 97) + 10)
93
+ }
94
+ if ((ch >= 65) && (ch <= 70)) {
95
+ return ((ch - 65) + 10)
96
+ }
97
+ return (0 - 1)
98
+ }
99
+
100
+ ; ---- hex ----------------------------------------------------------------
101
+
102
+ ; Lowercase, two characters per byte, no separator. This is what every
103
+ ; digest in the wild is printed as.
104
+ static sfn hexEncode:string (bytes:[int]) {
105
+ def digits:string (RgBase.hexDigits())
106
+ def out:string ""
107
+ def i:int 0
108
+ def n:int (array_length bytes)
109
+ while (i < n) {
110
+ def b:int (bit_and (itemAt bytes i) 255)
111
+ def hi:int (bit_shr b 4)
112
+ def lo:int (bit_and b 15)
113
+ out = (out + (RgText.substr(digits hi (hi + 1))))
114
+ out = (out + (RgText.substr(digits lo (lo + 1))))
115
+ i = (i + 1)
116
+ }
117
+ return out
118
+ }
119
+
120
+ ; Accepts either case. An odd length is an error rather than a silent
121
+ ; zero-pad, because "0f" and "f0" are different bytes and guessing which one
122
+ ; a caller meant is not this function's business.
123
+ static sfn hexDecode:RgBytesResult (s:string) {
124
+ def n:int (RgText.len(s))
125
+ if ((bit_and n 1) != 0) {
126
+ return (RgBytesResult.bad("InvalidCharacterError" "hex string has an odd length"))
127
+ }
128
+ def out:[int]
129
+ def i:int 0
130
+ while (i < n) {
131
+ def hi:int (RgBase.hexValueOf((RgText.unitAt(s i))))
132
+ def lo:int (RgBase.hexValueOf((RgText.unitAt(s (i + 1)))))
133
+ if ((hi < 0) || (lo < 0)) {
134
+ return (RgBytesResult.bad("InvalidCharacterError" "hex string has a non-hex character"))
135
+ }
136
+ push out ((hi * 16) + lo)
137
+ i = (i + 2)
138
+ }
139
+ return (RgBytesResult.good(out))
140
+ }
141
+
142
+ ; ---- base64 -------------------------------------------------------------
143
+
144
+ ; Shared by both alphabets. `pad` is what separates base64 (RFC 4648 §4,
145
+ ; padded) from base64url (§5, unpadded by convention and by every consumer
146
+ ; that matters — JWT forbids the padding outright).
147
+ static sfn encodeWith:string (alphabet:string bytes:[int] pad:boolean) {
148
+ def out:string ""
149
+ def n:int (array_length bytes)
150
+ def i:int 0
151
+ while ((i + 2) < n) {
152
+ def b0:int (bit_and (itemAt bytes i) 255)
153
+ def b1:int (bit_and (itemAt bytes (i + 1)) 255)
154
+ def b2:int (bit_and (itemAt bytes (i + 2)) 255)
155
+ def t:int (((b0 * 65536) + (b1 * 256)) + b2)
156
+ def c0:int (bit_and (bit_shr t 18) 63)
157
+ def c1:int (bit_and (bit_shr t 12) 63)
158
+ def c2:int (bit_and (bit_shr t 6) 63)
159
+ def c3:int (bit_and t 63)
160
+ out = (out + (RgText.substr(alphabet c0 (c0 + 1))))
161
+ out = (out + (RgText.substr(alphabet c1 (c1 + 1))))
162
+ out = (out + (RgText.substr(alphabet c2 (c2 + 1))))
163
+ out = (out + (RgText.substr(alphabet c3 (c3 + 1))))
164
+ i = (i + 3)
165
+ }
166
+ def left:int (n - i)
167
+ if (left == 1) {
168
+ def a0:int (bit_and (itemAt bytes i) 255)
169
+ def d0:int (bit_shr a0 2)
170
+ def d1:int (bit_and (a0 * 16) 63)
171
+ out = (out + (RgText.substr(alphabet d0 (d0 + 1))))
172
+ out = (out + (RgText.substr(alphabet d1 (d1 + 1))))
173
+ if pad {
174
+ out = (out + "==")
175
+ }
176
+ }
177
+ if (left == 2) {
178
+ def e0:int (bit_and (itemAt bytes i) 255)
179
+ def e1:int (bit_and (itemAt bytes (i + 1)) 255)
180
+ def f0:int (bit_shr e0 2)
181
+ def f1:int (bit_and ((e0 * 16) + (bit_shr e1 4)) 63)
182
+ def f2:int (bit_and (e1 * 4) 63)
183
+ out = (out + (RgText.substr(alphabet f0 (f0 + 1))))
184
+ out = (out + (RgText.substr(alphabet f1 (f1 + 1))))
185
+ out = (out + (RgText.substr(alphabet f2 (f2 + 1))))
186
+ if pad {
187
+ out = (out + "=")
188
+ }
189
+ }
190
+ return out
191
+ }
192
+
193
+ ; `strict` demands correct padding and rejects a final quantum whose unused
194
+ ; low bits are non-zero. A decoder that ignores those accepts several
195
+ ; different strings for the same bytes, which is how base64 malleability
196
+ ; bugs get in.
197
+ static sfn decodeWith:RgBytesResult (alphabet:string s:string requirePad:boolean) {
198
+ def units:[int] (RgText.toCodeUnits(s))
199
+ def clean:[int]
200
+ def padCount:int 0
201
+ def i:int 0
202
+ def n:int (array_length units)
203
+ while (i < n) {
204
+ def ch:int (itemAt units i)
205
+ if (ch == 61) {
206
+ padCount = (padCount + 1)
207
+ } {
208
+ if (padCount > 0) {
209
+ return (RgBytesResult.bad("InvalidCharacterError" "base64 has a character after the padding"))
210
+ }
211
+ def v:int (RgBase.b64ValueOf(alphabet ch))
212
+ if (v < 0) {
213
+ return (RgBytesResult.bad("InvalidCharacterError" "base64 has a character outside the alphabet"))
214
+ }
215
+ push clean v
216
+ }
217
+ i = (i + 1)
218
+ }
219
+ if (padCount > 2) {
220
+ return (RgBytesResult.bad("InvalidCharacterError" "base64 has more than two padding characters"))
221
+ }
222
+ def m:int (array_length clean)
223
+ def rem:int (m - ((idiv m 4) * 4))
224
+ if (rem == 1) {
225
+ return (RgBytesResult.bad("InvalidCharacterError" "base64 has a one-character final quantum"))
226
+ }
227
+ if requirePad {
228
+ if (rem != 0) {
229
+ def want:int 0
230
+ if (rem == 2) {
231
+ want = 2
232
+ }
233
+ if (rem == 3) {
234
+ want = 1
235
+ }
236
+ if (padCount != want) {
237
+ return (RgBytesResult.bad("InvalidCharacterError" "base64 is not padded to a multiple of four"))
238
+ }
239
+ }
240
+ }
241
+ def out:[int]
242
+ def k:int 0
243
+ while ((k + 3) < m) {
244
+ def q0:int (itemAt clean k)
245
+ def q1:int (itemAt clean (k + 1))
246
+ def q2:int (itemAt clean (k + 2))
247
+ def q3:int (itemAt clean (k + 3))
248
+ def t:int ((((q0 * 262144) + (q1 * 4096)) + (q2 * 64)) + q3)
249
+ push out (bit_and (bit_shr t 16) 255)
250
+ push out (bit_and (bit_shr t 8) 255)
251
+ push out (bit_and t 255)
252
+ k = (k + 4)
253
+ }
254
+ def tail:int (m - k)
255
+ if (tail == 2) {
256
+ def r0:int (itemAt clean k)
257
+ def r1:int (itemAt clean (k + 1))
258
+ ; The low 4 bits of the second character are not part of any byte.
259
+ if ((bit_and r1 15) != 0) {
260
+ return (RgBytesResult.bad("InvalidCharacterError" "base64 final quantum has non-zero unused bits"))
261
+ }
262
+ push out (bit_and ((r0 * 4) + (bit_shr r1 4)) 255)
263
+ }
264
+ if (tail == 3) {
265
+ def u0:int (itemAt clean k)
266
+ def u1:int (itemAt clean (k + 1))
267
+ def u2:int (itemAt clean (k + 2))
268
+ ; The low 2 bits of the third character are likewise unused.
269
+ if ((bit_and u2 3) != 0) {
270
+ return (RgBytesResult.bad("InvalidCharacterError" "base64 final quantum has non-zero unused bits"))
271
+ }
272
+ push out (bit_and ((u0 * 4) + (bit_shr u1 4)) 255)
273
+ push out (bit_and ((u1 * 16) + (bit_shr u2 2)) 255)
274
+ }
275
+ return (RgBytesResult.good(out))
276
+ }
277
+
278
+ static sfn base64Encode:string (bytes:[int]) {
279
+ return (RgBase.encodeWith((RgBase.b64Alphabet()) bytes true))
280
+ }
281
+
282
+ static sfn base64Decode:RgBytesResult (s:string) {
283
+ return (RgBase.decodeWith((RgBase.b64Alphabet()) s true))
284
+ }
285
+
286
+ static sfn base64UrlEncode:string (bytes:[int]) {
287
+ return (RgBase.encodeWith((RgBase.b64UrlAlphabet()) bytes false))
288
+ }
289
+
290
+ static sfn base64UrlDecode:RgBytesResult (s:string) {
291
+ return (RgBase.decodeWith((RgBase.b64UrlAlphabet()) s false))
292
+ }
293
+
294
+ ; ---- text convenience ---------------------------------------------------
295
+ ; UTF-8 in, UTF-8 out. These are what a caller actually reaches for, and
296
+ ; routing them through RgText keeps the string model in one place.
297
+
298
+ static sfn base64EncodeText:string (text:string) {
299
+ return (RgBase.base64Encode((RgText.toUtf8Bytes(text))))
300
+ }
301
+
302
+ static sfn base64DecodeText:string (s:string) {
303
+ def r:RgBytesResult (RgBase.base64Decode(s))
304
+ if (false == r.ok) {
305
+ return ""
306
+ }
307
+ return (RgText.fromUtf8Bytes(r.value))
308
+ }
309
+
310
+ static sfn hexEncodeText:string (text:string) {
311
+ return (RgBase.hexEncode((RgText.toUtf8Bytes(text))))
312
+ }
313
+ }