ranger-compiler 3.2.0 → 3.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +2482 -0
- package/LICENSE +28 -0
- package/LICENSE-MIT +21 -0
- package/README.md +1650 -1895
- package/dist/Lang.rgr +10946 -5852
- package/dist/README.md +3 -2
- package/dist/api.d.ts +1157 -29
- package/dist/api.js +66623 -38073
- package/dist/lib/ACEEditor.rgr +2 -0
- package/dist/lib/Ajax.rgr +2 -0
- package/dist/lib/CmdParams.rgr +2 -0
- package/dist/lib/Crypto.rgr +2 -0
- package/dist/lib/DOMLib.rgr +2 -0
- package/dist/lib/Engine3D.rgr +2 -0
- package/dist/lib/ImmutableVector.rgr +2 -0
- package/dist/lib/IndexedDB.rgr +2 -0
- package/dist/lib/IsoDate/DateMath.rgr +2 -0
- package/dist/lib/IsoDate/IsoCalendar.rgr +2 -0
- package/dist/lib/IsoDate/IsoDateParse.rgr +2 -0
- package/dist/lib/IsoDateLib.rgr +2 -0
- package/dist/lib/JSON.rgr +4906 -48
- package/dist/lib/JinxProcess.rgr +2 -0
- package/dist/lib/RangerProcess.rgr +2 -0
- package/dist/lib/Regex/RegexMatch.rgr +2 -0
- package/dist/lib/RegexLib.rgr +2 -0
- package/dist/lib/SQL.rgr +2 -0
- package/dist/lib/ServiceLib.rgr +2 -0
- package/dist/lib/Shell.rgr +326 -0
- package/dist/lib/Storage.rgr +2 -0
- package/dist/lib/Time.rgr +2 -0
- package/dist/lib/Timers.rgr +2 -0
- package/dist/lib/TypedArrays.rgr +2 -0
- package/dist/lib/ViewLib.rgr +2 -0
- package/dist/lib/WebLib.rgr +2 -0
- package/dist/lib/WebServerLib.rgr +2 -0
- package/dist/lib/apple/AppleAppBuilder.rgr +572 -0
- package/dist/lib/apple/AppleAppSpec.rgr +201 -0
- package/dist/lib/apple/AppleDevice.rgr +295 -0
- package/dist/lib/apple/AppleDeviceDoctor.rgr +453 -0
- package/dist/lib/apple/AppleSigning.rgr +379 -0
- package/dist/lib/apple/AppleSimulator.rgr +248 -0
- package/dist/lib/apple/AppleTarget.rgr +185 -0
- package/dist/lib/apple/AppleToolchain.rgr +458 -0
- package/dist/lib/apple/README.md +311 -0
- package/dist/lib/apple/apple_test.rgr +669 -0
- package/dist/lib/core/README.md +251 -0
- package/dist/lib/core/RgBase.rgr +313 -0
- package/dist/lib/core/RgNum.rgr +653 -0
- package/dist/lib/core/RgText.rgr +680 -0
- package/dist/lib/core/RgU32.rgr +309 -0
- package/dist/lib/ranger-dir.rgr +2 -0
- package/dist/lib/shell_test.rgr +193 -0
- package/dist/lib/stdlib.rgr +1176 -665
- package/dist/lib/stdops.rgr +2 -0
- package/dist/package.json +1 -1
- package/dist/rgrc.js +72955 -40753
- package/dist/stdops.rgr +2 -0
- package/package.json +1121 -339
|
@@ -0,0 +1,251 @@
|
|
|
1
|
+
# `lib/core` — the Ranger Core API
|
|
2
|
+
|
|
3
|
+
Portable, dependency-free Ranger classes for the things every language's standard
|
|
4
|
+
library has and Ranger does not: IEEE-754 arithmetic, 32-bit integer algebra,
|
|
5
|
+
string/Unicode handling, dates, formatting, hashing.
|
|
6
|
+
|
|
7
|
+
One implementation, compiled to every target. No `systemclass`, no operator
|
|
8
|
+
templates, no per-target branches.
|
|
9
|
+
|
|
10
|
+
**Status: first slice, wired into the JS engine.** `RgNum`, `RgU32`, `RgText`
|
|
11
|
+
and `RgBase` are landed and gated, and `ComponentEngine.rgr` now delegates 16 of
|
|
12
|
+
its own functions here instead of carrying copies (268 lines deleted). The rest
|
|
13
|
+
is planned in [PLAN_JS_STDLIB.md](../../PLAN_JS_STDLIB.md).
|
|
14
|
+
|
|
15
|
+
| File | What | State |
|
|
16
|
+
| --- | --- | --- |
|
|
17
|
+
| `RgNum.rgr` | IEEE-754 doubles: rounding, `exp`, `log`, `pow`, signed zero, ToInt32/ToUint32, float32 | landed |
|
|
18
|
+
| `RgU32.rgr` | 32-bit unsigned algebra: and/or/xor/not, shifts, rotations, add/sub, clz, popcount, big-endian bytes | landed |
|
|
19
|
+
| `RgText.rgr` | UTF-16 code-unit addressing over all three string models; code points; UTF-8 bytes | landed |
|
|
20
|
+
| `RgBase.rgr` | hex, base64, base64url, and `RgBytesResult` | landed |
|
|
21
|
+
| `RgDate.rgr` | ES5 time algebra, ISO parsing and formatting | planned |
|
|
22
|
+
| `RgIntl.rgr` | collation, number and date formatting over CLDR | planned |
|
|
23
|
+
| `RgCrypto.rgr` | SHA-2, HMAC, PBKDF2, secure random, UUID | planned |
|
|
24
|
+
|
|
25
|
+
## Why this is not `lib/stdlib.rgr`
|
|
26
|
+
|
|
27
|
+
`lib/stdlib.rgr` and `lib/stdops.rgr` contain **only operator templates** — no
|
|
28
|
+
classes at all. "stdlib" in this repository already means "language-level
|
|
29
|
+
operator definitions with a per-target template each". This directory is the
|
|
30
|
+
opposite: plain Ranger classes with one body that every target compiles.
|
|
31
|
+
|
|
32
|
+
The rest of `lib/` is mostly host shims — `lib/Time.rgr` is `es6`-only,
|
|
33
|
+
`lib/Crypto.rgr` has a `java7` template and nothing else. Those work on the
|
|
34
|
+
target they were written for. `lib/core` is for code that has to work on all of
|
|
35
|
+
them.
|
|
36
|
+
|
|
37
|
+
## Rules for a file in here
|
|
38
|
+
|
|
39
|
+
1. **No host operators with partial coverage.** If an operator has no `*`
|
|
40
|
+
fallback and is missing on a target, it does not belong in a signature here.
|
|
41
|
+
2. **No `EvHandle`, `EvalValue`, `TSNode` or any engine type.** These are
|
|
42
|
+
callable from a native Ranger program with no interpreter present.
|
|
43
|
+
3. **Primitive types only** in signatures: `int`, `double`, `boolean`, `string`,
|
|
44
|
+
`char`, `[T]`, and classes declared in `lib/core/`.
|
|
45
|
+
4. **No `throw`.** Where JavaScript throws, return a result carrier — the value
|
|
46
|
+
plus an error kind — so the same function can become a JS exception on one
|
|
47
|
+
side and an ordinary check on the other.
|
|
48
|
+
5. **No mutable file-level state**, so two callers cannot interfere.
|
|
49
|
+
6. **Every claim gets a vector.** Anything asserted about behaviour goes in
|
|
50
|
+
`tests/native/core_vectors.rgr` and is compared against Node and across
|
|
51
|
+
targets. See [Testing](#testing).
|
|
52
|
+
|
|
53
|
+
### Naming: the `D` suffix, and why
|
|
54
|
+
|
|
55
|
+
**A static method may not take the name of a global operator.** The resolver
|
|
56
|
+
rewrites `RgNum.floor(x)` into the operator call and then reports
|
|
57
|
+
`Class RgNum does not have method floor`. `EvValueBridge.rgr` documents the same
|
|
58
|
+
trap for `isArray`.
|
|
59
|
+
|
|
60
|
+
Nine names are affected: `floor`, `ceil`, `sqrt`, `sin`, `cos`, `tan`, `asin`,
|
|
61
|
+
`acos`, `atan2`. Each carries a `D` suffix here — `RgNum.floorD`, `RgNum.ceilD`,
|
|
62
|
+
`RgNum.sqrtD` — which is the convention `DateTime.rgr` already uses for `floorD`,
|
|
63
|
+
`truncD` and `modD`. For `floor` and `ceil` the suffix marks a real distinction:
|
|
64
|
+
**the operators return `int`**, and these return `double`.
|
|
65
|
+
|
|
66
|
+
Two more names collide for the same reason and are spelled around it: `wrap` (the
|
|
67
|
+
optional-wrapping operator) is `RgU32.wrap32`, and `add` is `RgU32.addU`.
|
|
68
|
+
Instance methods are unaffected — only statics enter the operator namespace.
|
|
69
|
+
|
|
70
|
+
### Class names the compiler has already taken
|
|
71
|
+
|
|
72
|
+
Separately from the operator namespace, `compiler/Lang.rgr` **emits helper
|
|
73
|
+
classes into target output**, and a Ranger class of the same name lands beside
|
|
74
|
+
them in the same file. These are reserved:
|
|
75
|
+
|
|
76
|
+
```
|
|
77
|
+
RgFiles RgHash RgList RgParse RgPath RgSort RgStr RgUtf8
|
|
78
|
+
```
|
|
79
|
+
|
|
80
|
+
`RgStr` is why this file's string class is called **`RgText`** — the C# template
|
|
81
|
+
for `strsplit` emits `static class RgStr { … Split … }` (`Lang.rgr:4645`), so any
|
|
82
|
+
C# program using both would have declared the class twice. Nothing caught it,
|
|
83
|
+
because C# has no toolchain on the machine the suite runs on; it turned up on a
|
|
84
|
+
grep. Check a new class name against this list, not only against the operators:
|
|
85
|
+
|
|
86
|
+
```
|
|
87
|
+
grep -oE "(static )?class (Rg[A-Za-z0-9_]*)" compiler/Lang.rgr | sort -u
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
Before adding a method, check the name:
|
|
91
|
+
|
|
92
|
+
```
|
|
93
|
+
node -e 'const fs=require("fs");const n=process.argv[1];
|
|
94
|
+
const re=/^\s{4,20}([a-zA-Z_][\w]*)\s+(?:_|[a-zA-Z_][\w]*)(?:@\([^)]*\))?\s*:/;
|
|
95
|
+
for(const f of ["compiler/Lang.rgr","lib/stdops.rgr","lib/stdlib.rgr"])
|
|
96
|
+
for(const l of fs.readFileSync(f,"utf8").split("\n")){const m=l.match(re);
|
|
97
|
+
if(m&&m[1]===n)console.log("COLLIDES with an operator in",f);}' floor
|
|
98
|
+
```
|
|
99
|
+
|
|
100
|
+
## What was measured
|
|
101
|
+
|
|
102
|
+
Everything below came out of running the same Ranger source on five targets. It
|
|
103
|
+
is written down because each item silently rules out an implementation that looks
|
|
104
|
+
obviously correct.
|
|
105
|
+
|
|
106
|
+
### `int` is not one type
|
|
107
|
+
|
|
108
|
+
| | es6 | python | go | cpp | rust |
|
|
109
|
+
| --- | --- | --- | --- | --- | --- |
|
|
110
|
+
| `2147483647 + 1` | 2147483648 | 2147483648 | 2147483648 | **-2147483648** | 64-bit |
|
|
111
|
+
| `bit_shl 1 31` | **-2147483648** | 2147483648 | 2147483648 | -2147483648 | 64-bit |
|
|
112
|
+
| `bit_shl 1 40` | **256** | 1099511627776 | 1099511627776 | **0** | 64-bit |
|
|
113
|
+
| `bit_ushr (0-1) 4` | 268435455 | 268435455 | **1152921504606846975** | 64-bit | 64-bit |
|
|
114
|
+
|
|
115
|
+
- **`int` is 32-bit signed on C++.** Plain *arithmetic* overflows there. A u32
|
|
116
|
+
cannot be held as an `int` in `[0, 2^32)`, because that range does not exist.
|
|
117
|
+
- **`bit_shl` is not portable.** On es6 it is JavaScript's `<<`: the result wraps
|
|
118
|
+
to 32 bits and the **count is taken mod 32**, so `1 << 40` is 256. On C++ it is
|
|
119
|
+
undefined past the width and gives 0. On go/python/rust it grows unbounded.
|
|
120
|
+
- **`bit_ushr` is three different operators** — 32-bit on es6 and python, 64-bit
|
|
121
|
+
on go, cpp, rust and swift. `RgU32` does not use it at all.
|
|
122
|
+
|
|
123
|
+
`RgU32` therefore carries a u32 as a **signed 32-bit bit pattern**, keeps every
|
|
124
|
+
intermediate under 2^31 by splitting into 16-bit halves, and never writes a
|
|
125
|
+
literal above 2147483647.
|
|
126
|
+
|
|
127
|
+
### Double literals may be emitted as int literals
|
|
128
|
+
|
|
129
|
+
`def z:double 0.0` compiles to `z = 0` on the **Python** target — an int. So
|
|
130
|
+
|
|
131
|
+
```ranger
|
|
132
|
+
def z:double 0.0
|
|
133
|
+
return (z * (0.0 - 1.0)) ; +0 on Python. The sign was gone a line earlier.
|
|
134
|
+
```
|
|
135
|
+
|
|
136
|
+
The C++ writer has the same habit for bare literals, which is why the JS engine's
|
|
137
|
+
`negativeZero()` bound the value first. That fixes C++ and not Python.
|
|
138
|
+
|
|
139
|
+
**A division cannot be folded into an int on any target**, because every one of
|
|
140
|
+
them routes double division through a helper (Python's `/` raises
|
|
141
|
+
`ZeroDivisionError`, so `r_div_f64` is already generated). `RgNum.negZero()` is
|
|
142
|
+
therefore `-1 / Infinity`, and the cross-target suite is what found this.
|
|
143
|
+
|
|
144
|
+
### Strings have three models, and all three are live
|
|
145
|
+
|
|
146
|
+
| `strlen` | es6 | python | go | rust | cpp |
|
|
147
|
+
| --- | --- | --- | --- | --- | --- |
|
|
148
|
+
| `"é"` | 1 | 1 | 1 | 1 | **2** |
|
|
149
|
+
| `"😀"` | **2** | 1 | 1 | 1 | **4** |
|
|
150
|
+
|
|
151
|
+
es6 counts UTF-16 code units, python/go/rust count code points, C++ counts UTF-8
|
|
152
|
+
bytes. `charAt` follows suit: on `"é"` it answers 233 on the first four and 195
|
|
153
|
+
on C++. `RgText` encapsulates all three and exposes UTF-16 code units, so
|
|
154
|
+
`RgText.len("😀")` is 2 everywhere.
|
|
155
|
+
|
|
156
|
+
### `strfromcode` encodes a code point; it cannot emit a byte
|
|
157
|
+
|
|
158
|
+
On C++, `strfromcode 195` answers the **two-byte** UTF-8 encoding of U+00C3, not
|
|
159
|
+
the single byte 0xC3. So a UTF-8 encoder assembled from per-byte `strfromcode`
|
|
160
|
+
calls — which is the obvious way to write one, and the way the JS engine writes
|
|
161
|
+
it — double-encodes: `"héllo"` came back as `"éllo"`. Two consequences, both in
|
|
162
|
+
`RgText`:
|
|
163
|
+
|
|
164
|
+
- `encodeUtf8(cp)` is just `strfromcode cp`, because that already IS the UTF-8
|
|
165
|
+
encoding on a byte-model target.
|
|
166
|
+
- `byteSlice` uses `substring`, which is byte-indexed on a byte-model target,
|
|
167
|
+
rather than rebuilding the run byte by byte.
|
|
168
|
+
|
|
169
|
+
Separately, `strfromcode` used **directly as an argument** does not compile on
|
|
170
|
+
Rust (the writer emits a `char` and then calls a `String` method on it), so every
|
|
171
|
+
call here is bound to a `string` first. And on es6 it is `String.fromCharCode`,
|
|
172
|
+
which truncates to 16 bits — `fromCodePoint` builds a surrogate pair there.
|
|
173
|
+
|
|
174
|
+
### Mutating an array parameter does not reach the caller on Go
|
|
175
|
+
|
|
176
|
+
```ranger
|
|
177
|
+
sfn appendAll:void (into:[string] more:[string]) {
|
|
178
|
+
push into (itemAt more 0) ; the caller never sees this on Go
|
|
179
|
+
}
|
|
180
|
+
```
|
|
181
|
+
|
|
182
|
+
A Go slice is passed by value and `append` rebinds the local. The symptom is
|
|
183
|
+
silent: every individual vector set was correct while the combined `all` set came
|
|
184
|
+
out **empty**, on Go only. Functions here return new arrays instead.
|
|
185
|
+
|
|
186
|
+
### "Byte offset" has to mean UTF-8, not "whatever the target indexes by"
|
|
187
|
+
|
|
188
|
+
The engine's `cuByteOf` means "the offset the target's own string search
|
|
189
|
+
reports" — code units on es6, bytes elsewhere. Lifted unchanged, it answered 3 on
|
|
190
|
+
es6 and 5 everywhere else for the same call, because that contract is
|
|
191
|
+
target-relative by construction. `RgText.utf8ByteOfUnit` and `unitOfUtf8Byte`
|
|
192
|
+
are defined over UTF-8 on every target instead. A bridge to native search offsets
|
|
193
|
+
is a real need, but it belongs to whoever is calling the native search.
|
|
194
|
+
|
|
195
|
+
### There is no `exp`, `log`, `pow`, `round`, `min` or `atan` operator
|
|
196
|
+
|
|
197
|
+
`sqrt`, `sin`, `cos`, `tan`, `asin`, `acos`, `atan2` and `fabs` exist as host
|
|
198
|
+
operators. `exp`, `log` and `pow` **do not exist on any target**, and
|
|
199
|
+
`floor`/`ceil` return `int`. So `RgNum` implements them: `exp` by range reduction
|
|
200
|
+
plus a Taylor series, `log` by reduction plus the atanh series, `pow` by squaring
|
|
201
|
+
for integer exponents and the identity otherwise.
|
|
202
|
+
|
|
203
|
+
The trade is explicit. These land **1 ULP** from each platform's libm — so they
|
|
204
|
+
are *not* bit-identical to `Math.exp` in Node. In exchange they are **identical
|
|
205
|
+
across all five targets**, which delegating to each platform's libm would not
|
|
206
|
+
have been. For a program that has to agree with itself on Go and on Node, that is
|
|
207
|
+
the direction worth being wrong in. The suite asserts the ULP bound rather than
|
|
208
|
+
equality, so a regression in the series cannot hide behind "it was never exact".
|
|
209
|
+
|
|
210
|
+
## Testing
|
|
211
|
+
|
|
212
|
+
```bash
|
|
213
|
+
npm run test:core # oracle + every target
|
|
214
|
+
```
|
|
215
|
+
|
|
216
|
+
Five targets are **built and run**: es6, python, go, cpp, rust. The other five —
|
|
217
|
+
csharp, kotlin, dart, swift6, java7 — have no toolchain on the machine this was
|
|
218
|
+
developed on, so they are only checked as far as *the Ranger compiler writing
|
|
219
|
+
them without error*, which is still worth doing: it is how the `RgStr` clash
|
|
220
|
+
above would have been caught. Scala does not write, because `shell_arg` and
|
|
221
|
+
`shell_arg_cnt` have no Scala template — a gap in the harness's `main`, not in
|
|
222
|
+
`lib/core`, and Scala is outside the large-program CI path anyway.
|
|
223
|
+
|
|
224
|
+
Three legs, in `tests/core-targets.test.ts` over
|
|
225
|
+
`tests/native/core_vectors.rgr`:
|
|
226
|
+
|
|
227
|
+
1. **Node as oracle.** Every expectation is computed by the JavaScript built-in
|
|
228
|
+
the function mirrors, never hand-written. `CONFORMANCE.md` records
|
|
229
|
+
hand-written expectations encoding a misunderstanding three separate times.
|
|
230
|
+
2. **Byte-for-byte across targets.** Compiled to es6, python, go, cpp and rust;
|
|
231
|
+
built and run where the toolchain is on `PATH`; output compared character for
|
|
232
|
+
character against the es6 run.
|
|
233
|
+
3. **The `D` bound** on `exp`/`log`, asserted as a bound.
|
|
234
|
+
|
|
235
|
+
Doubles are never printed directly — each target's own number formatter
|
|
236
|
+
disagrees about trailing `.0`, exponent spelling and digit count. `showD`
|
|
237
|
+
decomposes a double into sign, a 52-bit mantissa in two halves, and a binary
|
|
238
|
+
exponent, and the test does the same decomposition in JavaScript.
|
|
239
|
+
|
|
240
|
+
Current reading: **173 vectors, identical on es6 / python / go / cpp / rust;
|
|
241
|
+
170 of them exact against Node**, the other three being the `exp`/`log` series.
|
|
242
|
+
|
|
243
|
+
The second gate is the JS engine itself. `ComponentEngine.rgr` delegates to this
|
|
244
|
+
directory, so `npm run test:tsengine` — 45,221 lines of interpreter compiled to
|
|
245
|
+
six targets and checked against Node's answers for seven workloads — is also a
|
|
246
|
+
test of `lib/core`. Verified by hand here (vitest is not installed on this
|
|
247
|
+
machine): all seven answers correct on es6 and on Go, and the engine still writes
|
|
248
|
+
for kotlin, csharp, dart and swift6.
|
|
249
|
+
|
|
250
|
+
`strmodel` is the one set that is reported rather than compared — `RgText.kind()`
|
|
251
|
+
is 0 on es6, 2 on python/go/rust and 1 on cpp, and that is the point.
|
|
@@ -0,0 +1,313 @@
|
|
|
1
|
+
; SPDX-License-Identifier: MIT
|
|
2
|
+
|
|
3
|
+
; =============================================================================
|
|
4
|
+
; RgBase — hex, base64 and base64url
|
|
5
|
+
; =============================================================================
|
|
6
|
+
; There is no portable base64 in Ranger. There is no portable hex either. Both
|
|
7
|
+
; are wanted by anything that moves bytes: data URIs, JWTs, HTTP basic auth,
|
|
8
|
+
; digests printed for a human, binary embedded in JSON. `RgCrypto` needs all
|
|
9
|
+
; three, and `btoa`/`atob` on the JS side map straight onto them.
|
|
10
|
+
;
|
|
11
|
+
; Bytes are `[int]` with each element in [0, 255]. Not `string`, because a Ranger
|
|
12
|
+
; string is bytes on C++, code points on python/go/rust and UTF-16 units on es6
|
|
13
|
+
; (see RgText) — "base64 of a string" would encode different input per target,
|
|
14
|
+
; which is the exact bug this library exists to prevent. Text goes through
|
|
15
|
+
; `RgText.toUtf8Bytes` first, which is the single place the string model is
|
|
16
|
+
; handled.
|
|
17
|
+
;
|
|
18
|
+
; Decoding returns `RgBytesResult` rather than throwing: the core layer has no
|
|
19
|
+
; `throw` (see lib/core/README.md), so a caller checks `ok` and the JS binding
|
|
20
|
+
; turns `errorKind` into the exception the spec names.
|
|
21
|
+
;
|
|
22
|
+
; RFC 4648 throughout, including the parts people skip: padding is REQUIRED and
|
|
23
|
+
; validated for base64, forbidden for base64url, and a non-zero tail bit in the
|
|
24
|
+
; final quantum is rejected rather than silently discarded.
|
|
25
|
+
; =============================================================================
|
|
26
|
+
|
|
27
|
+
Import "RgText.rgr"
|
|
28
|
+
|
|
29
|
+
; A decode either produced bytes or says why it did not.
|
|
30
|
+
class RgBytesResult {
|
|
31
|
+
def ok:boolean false
|
|
32
|
+
def value:[int]
|
|
33
|
+
; "" when ok. Otherwise the DOM/JS error name the binding should throw:
|
|
34
|
+
; InvalidCharacterError for a bad character or a bad length.
|
|
35
|
+
def errorKind:string ""
|
|
36
|
+
def errorMessage:string ""
|
|
37
|
+
|
|
38
|
+
static sfn good:RgBytesResult (bytes:[int]) {
|
|
39
|
+
def r (new RgBytesResult)
|
|
40
|
+
r.ok = true
|
|
41
|
+
r.value = bytes
|
|
42
|
+
return r
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
static sfn bad:RgBytesResult (kind:string message:string) {
|
|
46
|
+
def r (new RgBytesResult)
|
|
47
|
+
r.ok = false
|
|
48
|
+
r.errorKind = kind
|
|
49
|
+
r.errorMessage = message
|
|
50
|
+
return r
|
|
51
|
+
}
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
class RgBase {
|
|
55
|
+
|
|
56
|
+
; ---- alphabets ---------------------------------------------------------
|
|
57
|
+
; As strings rather than arrays so the tables cost one literal each. Indexed
|
|
58
|
+
; with RgText.unitAt, which is ASCII-safe on every model.
|
|
59
|
+
|
|
60
|
+
static sfn hexDigits:string () {
|
|
61
|
+
return "0123456789abcdef"
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
static sfn b64Alphabet:string () {
|
|
65
|
+
return "ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789+/"
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
static sfn b64UrlAlphabet:string () {
|
|
69
|
+
return "ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789-_"
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
; The value of one base64 character, or -1. A table scan rather than
|
|
73
|
+
; arithmetic on ranges: the two alphabets differ only in the last two
|
|
74
|
+
; entries, so one function serves both and there is no second place for the
|
|
75
|
+
; `+/` versus `-_` distinction to be got wrong.
|
|
76
|
+
static sfn b64ValueOf:int (alphabet:string ch:int) {
|
|
77
|
+
def i:int 0
|
|
78
|
+
while (i < 64) {
|
|
79
|
+
if ((RgText.unitAt(alphabet i)) == ch) {
|
|
80
|
+
return i
|
|
81
|
+
}
|
|
82
|
+
i = (i + 1)
|
|
83
|
+
}
|
|
84
|
+
return (0 - 1)
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
static sfn hexValueOf:int (ch:int) {
|
|
88
|
+
if ((ch >= 48) && (ch <= 57)) {
|
|
89
|
+
return (ch - 48)
|
|
90
|
+
}
|
|
91
|
+
if ((ch >= 97) && (ch <= 102)) {
|
|
92
|
+
return ((ch - 97) + 10)
|
|
93
|
+
}
|
|
94
|
+
if ((ch >= 65) && (ch <= 70)) {
|
|
95
|
+
return ((ch - 65) + 10)
|
|
96
|
+
}
|
|
97
|
+
return (0 - 1)
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
; ---- hex ----------------------------------------------------------------
|
|
101
|
+
|
|
102
|
+
; Lowercase, two characters per byte, no separator. This is what every
|
|
103
|
+
; digest in the wild is printed as.
|
|
104
|
+
static sfn hexEncode:string (bytes:[int]) {
|
|
105
|
+
def digits:string (RgBase.hexDigits())
|
|
106
|
+
def out:string ""
|
|
107
|
+
def i:int 0
|
|
108
|
+
def n:int (array_length bytes)
|
|
109
|
+
while (i < n) {
|
|
110
|
+
def b:int (bit_and (itemAt bytes i) 255)
|
|
111
|
+
def hi:int (bit_shr b 4)
|
|
112
|
+
def lo:int (bit_and b 15)
|
|
113
|
+
out = (out + (RgText.substr(digits hi (hi + 1))))
|
|
114
|
+
out = (out + (RgText.substr(digits lo (lo + 1))))
|
|
115
|
+
i = (i + 1)
|
|
116
|
+
}
|
|
117
|
+
return out
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
; Accepts either case. An odd length is an error rather than a silent
|
|
121
|
+
; zero-pad, because "0f" and "f0" are different bytes and guessing which one
|
|
122
|
+
; a caller meant is not this function's business.
|
|
123
|
+
static sfn hexDecode:RgBytesResult (s:string) {
|
|
124
|
+
def n:int (RgText.len(s))
|
|
125
|
+
if ((bit_and n 1) != 0) {
|
|
126
|
+
return (RgBytesResult.bad("InvalidCharacterError" "hex string has an odd length"))
|
|
127
|
+
}
|
|
128
|
+
def out:[int]
|
|
129
|
+
def i:int 0
|
|
130
|
+
while (i < n) {
|
|
131
|
+
def hi:int (RgBase.hexValueOf((RgText.unitAt(s i))))
|
|
132
|
+
def lo:int (RgBase.hexValueOf((RgText.unitAt(s (i + 1)))))
|
|
133
|
+
if ((hi < 0) || (lo < 0)) {
|
|
134
|
+
return (RgBytesResult.bad("InvalidCharacterError" "hex string has a non-hex character"))
|
|
135
|
+
}
|
|
136
|
+
push out ((hi * 16) + lo)
|
|
137
|
+
i = (i + 2)
|
|
138
|
+
}
|
|
139
|
+
return (RgBytesResult.good(out))
|
|
140
|
+
}
|
|
141
|
+
|
|
142
|
+
; ---- base64 -------------------------------------------------------------
|
|
143
|
+
|
|
144
|
+
; Shared by both alphabets. `pad` is what separates base64 (RFC 4648 §4,
|
|
145
|
+
; padded) from base64url (§5, unpadded by convention and by every consumer
|
|
146
|
+
; that matters — JWT forbids the padding outright).
|
|
147
|
+
static sfn encodeWith:string (alphabet:string bytes:[int] pad:boolean) {
|
|
148
|
+
def out:string ""
|
|
149
|
+
def n:int (array_length bytes)
|
|
150
|
+
def i:int 0
|
|
151
|
+
while ((i + 2) < n) {
|
|
152
|
+
def b0:int (bit_and (itemAt bytes i) 255)
|
|
153
|
+
def b1:int (bit_and (itemAt bytes (i + 1)) 255)
|
|
154
|
+
def b2:int (bit_and (itemAt bytes (i + 2)) 255)
|
|
155
|
+
def t:int (((b0 * 65536) + (b1 * 256)) + b2)
|
|
156
|
+
def c0:int (bit_and (bit_shr t 18) 63)
|
|
157
|
+
def c1:int (bit_and (bit_shr t 12) 63)
|
|
158
|
+
def c2:int (bit_and (bit_shr t 6) 63)
|
|
159
|
+
def c3:int (bit_and t 63)
|
|
160
|
+
out = (out + (RgText.substr(alphabet c0 (c0 + 1))))
|
|
161
|
+
out = (out + (RgText.substr(alphabet c1 (c1 + 1))))
|
|
162
|
+
out = (out + (RgText.substr(alphabet c2 (c2 + 1))))
|
|
163
|
+
out = (out + (RgText.substr(alphabet c3 (c3 + 1))))
|
|
164
|
+
i = (i + 3)
|
|
165
|
+
}
|
|
166
|
+
def left:int (n - i)
|
|
167
|
+
if (left == 1) {
|
|
168
|
+
def a0:int (bit_and (itemAt bytes i) 255)
|
|
169
|
+
def d0:int (bit_shr a0 2)
|
|
170
|
+
def d1:int (bit_and (a0 * 16) 63)
|
|
171
|
+
out = (out + (RgText.substr(alphabet d0 (d0 + 1))))
|
|
172
|
+
out = (out + (RgText.substr(alphabet d1 (d1 + 1))))
|
|
173
|
+
if pad {
|
|
174
|
+
out = (out + "==")
|
|
175
|
+
}
|
|
176
|
+
}
|
|
177
|
+
if (left == 2) {
|
|
178
|
+
def e0:int (bit_and (itemAt bytes i) 255)
|
|
179
|
+
def e1:int (bit_and (itemAt bytes (i + 1)) 255)
|
|
180
|
+
def f0:int (bit_shr e0 2)
|
|
181
|
+
def f1:int (bit_and ((e0 * 16) + (bit_shr e1 4)) 63)
|
|
182
|
+
def f2:int (bit_and (e1 * 4) 63)
|
|
183
|
+
out = (out + (RgText.substr(alphabet f0 (f0 + 1))))
|
|
184
|
+
out = (out + (RgText.substr(alphabet f1 (f1 + 1))))
|
|
185
|
+
out = (out + (RgText.substr(alphabet f2 (f2 + 1))))
|
|
186
|
+
if pad {
|
|
187
|
+
out = (out + "=")
|
|
188
|
+
}
|
|
189
|
+
}
|
|
190
|
+
return out
|
|
191
|
+
}
|
|
192
|
+
|
|
193
|
+
; `strict` demands correct padding and rejects a final quantum whose unused
|
|
194
|
+
; low bits are non-zero. A decoder that ignores those accepts several
|
|
195
|
+
; different strings for the same bytes, which is how base64 malleability
|
|
196
|
+
; bugs get in.
|
|
197
|
+
static sfn decodeWith:RgBytesResult (alphabet:string s:string requirePad:boolean) {
|
|
198
|
+
def units:[int] (RgText.toCodeUnits(s))
|
|
199
|
+
def clean:[int]
|
|
200
|
+
def padCount:int 0
|
|
201
|
+
def i:int 0
|
|
202
|
+
def n:int (array_length units)
|
|
203
|
+
while (i < n) {
|
|
204
|
+
def ch:int (itemAt units i)
|
|
205
|
+
if (ch == 61) {
|
|
206
|
+
padCount = (padCount + 1)
|
|
207
|
+
} {
|
|
208
|
+
if (padCount > 0) {
|
|
209
|
+
return (RgBytesResult.bad("InvalidCharacterError" "base64 has a character after the padding"))
|
|
210
|
+
}
|
|
211
|
+
def v:int (RgBase.b64ValueOf(alphabet ch))
|
|
212
|
+
if (v < 0) {
|
|
213
|
+
return (RgBytesResult.bad("InvalidCharacterError" "base64 has a character outside the alphabet"))
|
|
214
|
+
}
|
|
215
|
+
push clean v
|
|
216
|
+
}
|
|
217
|
+
i = (i + 1)
|
|
218
|
+
}
|
|
219
|
+
if (padCount > 2) {
|
|
220
|
+
return (RgBytesResult.bad("InvalidCharacterError" "base64 has more than two padding characters"))
|
|
221
|
+
}
|
|
222
|
+
def m:int (array_length clean)
|
|
223
|
+
def rem:int (m - ((idiv m 4) * 4))
|
|
224
|
+
if (rem == 1) {
|
|
225
|
+
return (RgBytesResult.bad("InvalidCharacterError" "base64 has a one-character final quantum"))
|
|
226
|
+
}
|
|
227
|
+
if requirePad {
|
|
228
|
+
if (rem != 0) {
|
|
229
|
+
def want:int 0
|
|
230
|
+
if (rem == 2) {
|
|
231
|
+
want = 2
|
|
232
|
+
}
|
|
233
|
+
if (rem == 3) {
|
|
234
|
+
want = 1
|
|
235
|
+
}
|
|
236
|
+
if (padCount != want) {
|
|
237
|
+
return (RgBytesResult.bad("InvalidCharacterError" "base64 is not padded to a multiple of four"))
|
|
238
|
+
}
|
|
239
|
+
}
|
|
240
|
+
}
|
|
241
|
+
def out:[int]
|
|
242
|
+
def k:int 0
|
|
243
|
+
while ((k + 3) < m) {
|
|
244
|
+
def q0:int (itemAt clean k)
|
|
245
|
+
def q1:int (itemAt clean (k + 1))
|
|
246
|
+
def q2:int (itemAt clean (k + 2))
|
|
247
|
+
def q3:int (itemAt clean (k + 3))
|
|
248
|
+
def t:int ((((q0 * 262144) + (q1 * 4096)) + (q2 * 64)) + q3)
|
|
249
|
+
push out (bit_and (bit_shr t 16) 255)
|
|
250
|
+
push out (bit_and (bit_shr t 8) 255)
|
|
251
|
+
push out (bit_and t 255)
|
|
252
|
+
k = (k + 4)
|
|
253
|
+
}
|
|
254
|
+
def tail:int (m - k)
|
|
255
|
+
if (tail == 2) {
|
|
256
|
+
def r0:int (itemAt clean k)
|
|
257
|
+
def r1:int (itemAt clean (k + 1))
|
|
258
|
+
; The low 4 bits of the second character are not part of any byte.
|
|
259
|
+
if ((bit_and r1 15) != 0) {
|
|
260
|
+
return (RgBytesResult.bad("InvalidCharacterError" "base64 final quantum has non-zero unused bits"))
|
|
261
|
+
}
|
|
262
|
+
push out (bit_and ((r0 * 4) + (bit_shr r1 4)) 255)
|
|
263
|
+
}
|
|
264
|
+
if (tail == 3) {
|
|
265
|
+
def u0:int (itemAt clean k)
|
|
266
|
+
def u1:int (itemAt clean (k + 1))
|
|
267
|
+
def u2:int (itemAt clean (k + 2))
|
|
268
|
+
; The low 2 bits of the third character are likewise unused.
|
|
269
|
+
if ((bit_and u2 3) != 0) {
|
|
270
|
+
return (RgBytesResult.bad("InvalidCharacterError" "base64 final quantum has non-zero unused bits"))
|
|
271
|
+
}
|
|
272
|
+
push out (bit_and ((u0 * 4) + (bit_shr u1 4)) 255)
|
|
273
|
+
push out (bit_and ((u1 * 16) + (bit_shr u2 2)) 255)
|
|
274
|
+
}
|
|
275
|
+
return (RgBytesResult.good(out))
|
|
276
|
+
}
|
|
277
|
+
|
|
278
|
+
static sfn base64Encode:string (bytes:[int]) {
|
|
279
|
+
return (RgBase.encodeWith((RgBase.b64Alphabet()) bytes true))
|
|
280
|
+
}
|
|
281
|
+
|
|
282
|
+
static sfn base64Decode:RgBytesResult (s:string) {
|
|
283
|
+
return (RgBase.decodeWith((RgBase.b64Alphabet()) s true))
|
|
284
|
+
}
|
|
285
|
+
|
|
286
|
+
static sfn base64UrlEncode:string (bytes:[int]) {
|
|
287
|
+
return (RgBase.encodeWith((RgBase.b64UrlAlphabet()) bytes false))
|
|
288
|
+
}
|
|
289
|
+
|
|
290
|
+
static sfn base64UrlDecode:RgBytesResult (s:string) {
|
|
291
|
+
return (RgBase.decodeWith((RgBase.b64UrlAlphabet()) s false))
|
|
292
|
+
}
|
|
293
|
+
|
|
294
|
+
; ---- text convenience ---------------------------------------------------
|
|
295
|
+
; UTF-8 in, UTF-8 out. These are what a caller actually reaches for, and
|
|
296
|
+
; routing them through RgText keeps the string model in one place.
|
|
297
|
+
|
|
298
|
+
static sfn base64EncodeText:string (text:string) {
|
|
299
|
+
return (RgBase.base64Encode((RgText.toUtf8Bytes(text))))
|
|
300
|
+
}
|
|
301
|
+
|
|
302
|
+
static sfn base64DecodeText:string (s:string) {
|
|
303
|
+
def r:RgBytesResult (RgBase.base64Decode(s))
|
|
304
|
+
if (false == r.ok) {
|
|
305
|
+
return ""
|
|
306
|
+
}
|
|
307
|
+
return (RgText.fromUtf8Bytes(r.value))
|
|
308
|
+
}
|
|
309
|
+
|
|
310
|
+
static sfn hexEncodeText:string (text:string) {
|
|
311
|
+
return (RgBase.hexEncode((RgText.toUtf8Bytes(text))))
|
|
312
|
+
}
|
|
313
|
+
}
|