@amritk/nish 0.13.0 → 0.14.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -0
- package/docs/AI.md +63 -26
- package/llms.txt +1 -1
- package/package.json +5 -5
- package/runtime/nish.d.ts +65 -0
- package/runtime/nish.h +31 -0
- package/runtime/nish.mjs +35 -0
- package/runtime/runtime-host.c +178 -0
- package/runtime/runtime-os.c +15 -0
- package/runtime/shim.mjs +122 -0
- package/scripts/build.sh +8 -7
- package/scripts/gen-diagnostic-codes.mjs +47 -25
- package/std/README.md +40 -0
- package/std/crypto/base64url.ts +145 -0
- package/std/crypto/ct.ts +64 -0
- package/std/crypto/hkdf.ts +118 -0
- package/std/crypto/hmac.ts +155 -0
- package/std/crypto/sha256.ts +444 -0
- package/std/crypto/sha512.ts +510 -0
- package/std/crypto/x25519.ts +494 -0
package/runtime/shim.mjs
CHANGED
|
@@ -42,6 +42,7 @@
|
|
|
42
42
|
* The rewrite rules that call these helpers are listed in docs/wp13-differential.md.
|
|
43
43
|
*/
|
|
44
44
|
import child_process from "node:child_process";
|
|
45
|
+
import { webcrypto } from "node:crypto";
|
|
45
46
|
import fs from "node:fs";
|
|
46
47
|
import os from "node:os";
|
|
47
48
|
|
|
@@ -91,6 +92,37 @@ export function bitsToF64(b) {
|
|
|
91
92
|
return BITS.getFloat64(0);
|
|
92
93
|
}
|
|
93
94
|
|
|
95
|
+
/**
|
|
96
|
+
* `ctSelect` / `ctEq` (WP34 N6). A `u32` is a `number` here and a `u64` a
|
|
97
|
+
* BigInt, so the operands' kind picks the width, and a mix of the two — which
|
|
98
|
+
* the native checker refuses, and which a `u64` written as a bare literal is
|
|
99
|
+
* under an unrewritten run — throws the `TypeError` BigInt arithmetic throws
|
|
100
|
+
* rather than comparing a number with a BigInt and answering zero. JavaScript's
|
|
101
|
+
* `&` reads a `number` as a signed 32-bit integer, so each answer is put back in
|
|
102
|
+
* range with `>>> 0` or `asUintN(64, ...)`. These branch: only the native
|
|
103
|
+
* lowering promises constant time.
|
|
104
|
+
*/
|
|
105
|
+
function ctWide(name, first, second, third) {
|
|
106
|
+
const wide = typeof first === "bigint";
|
|
107
|
+
if ((typeof second === "bigint") !== wide || (typeof third === "bigint") !== wide) {
|
|
108
|
+
throw new TypeError(`${name}: cannot mix a u64 (BigInt) with a u32 (number)`);
|
|
109
|
+
}
|
|
110
|
+
return wide;
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
const U64_ONES = (1n << 64n) - 1n;
|
|
114
|
+
|
|
115
|
+
export function ctSelect(mask, a, b) {
|
|
116
|
+
if (ctWide("ctSelect", mask, a, b)) return wrapU64((a & mask) | (b & ~mask));
|
|
117
|
+
return ((a & mask) | (b & ~mask)) >>> 0;
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
export function ctEq(a, b) {
|
|
121
|
+
// `ctEq` has two operands, so the second one stands in for the third.
|
|
122
|
+
if (ctWide("ctEq", a, b, b)) return wrapU64(a ^ b) === 0n ? U64_ONES : 0n;
|
|
123
|
+
return (a ^ b) === 0 ? 0xffffffff : 0;
|
|
124
|
+
}
|
|
125
|
+
|
|
94
126
|
/** Wrap a BigInt to the i64 range: every i64 `+ - * /` and unary minus goes through here. */
|
|
95
127
|
export function wrapI64(x) {
|
|
96
128
|
return BigInt.asIntN(64, x);
|
|
@@ -401,6 +433,28 @@ export function updIdx(a, i, f) {
|
|
|
401
433
|
return v;
|
|
402
434
|
}
|
|
403
435
|
|
|
436
|
+
/**
|
|
437
|
+
* `dst.set(src, offset)` (WP34 N2): `TypedArray.prototype.set`'s copy on the
|
|
438
|
+
* plain array a `u8[]` is here. The source is copied first, so a self-copy or
|
|
439
|
+
* an overlapping one reads what was there before, as `memmove` does natively;
|
|
440
|
+
* a range past the end fails with the native panic and its words, where a
|
|
441
|
+
* typed array would throw a `RangeError` for the same offsets.
|
|
442
|
+
*/
|
|
443
|
+
export function arraySet(dst, src, offset) {
|
|
444
|
+
// `ToIntegerOrInfinity`: NaN is 0, as `llvm.fptosi.sat` makes it natively.
|
|
445
|
+
// `Math.trunc` rather than `toIndex`, which converts a bigint: an `i64` or
|
|
446
|
+
// `u64` offset is a bigint here, and it throws the `TypeError` the typed
|
|
447
|
+
// array and `Array.prototype.fill` throw for one, instead of being rounded
|
|
448
|
+
// to the nearest double past 2^53 (docs/RUN_UNDER_NODE.md).
|
|
449
|
+
const at = offset === undefined ? 0 : Math.trunc(offset) || 0;
|
|
450
|
+
const end = at + src.length;
|
|
451
|
+
if (!(at >= 0 && end <= dst.length)) panicSlice(at, end, dst.length);
|
|
452
|
+
// Two plain arrays overlap only when they are one array, which is the one
|
|
453
|
+
// case that must read the source before writing it.
|
|
454
|
+
const from = src === dst ? src.slice() : src;
|
|
455
|
+
for (let i = 0; i < from.length; i++) dst[at + i] = from[i];
|
|
456
|
+
}
|
|
457
|
+
|
|
404
458
|
/** `new Array<T>(n)`: `n` zero-filled elements (`0`, `0n`, or `false`). */
|
|
405
459
|
export function newArray(n, zero) {
|
|
406
460
|
return new Array(toIndex(n)).fill(zero);
|
|
@@ -434,6 +488,15 @@ export function readFileSyncOrNull(path) {
|
|
|
434
488
|
}
|
|
435
489
|
}
|
|
436
490
|
|
|
491
|
+
/** `readFileBytesSync(path)` (WP34 N2): the bytes as a plain array of numbers, or null. */
|
|
492
|
+
export function readFileBytesSync(path) {
|
|
493
|
+
try {
|
|
494
|
+
return Array.from(fs.readFileSync(path));
|
|
495
|
+
} catch {
|
|
496
|
+
return null;
|
|
497
|
+
}
|
|
498
|
+
}
|
|
499
|
+
|
|
437
500
|
export function writeFileSync(path, data) {
|
|
438
501
|
try {
|
|
439
502
|
fs.writeFileSync(path, data, "utf8");
|
|
@@ -618,6 +681,65 @@ export function monotonicNanos() {
|
|
|
618
681
|
return process.hrtime.bigint();
|
|
619
682
|
}
|
|
620
683
|
|
|
684
|
+
// ---- The host (WP34 N3) ------------------------------------------------------
|
|
685
|
+
|
|
686
|
+
/**
|
|
687
|
+
* `statMtimeSync(path)`: Node's `mtimeMs` for the path, or NaN when it cannot
|
|
688
|
+
* be stat'd, which is the native answer too. `mtimeMs` is the same arithmetic
|
|
689
|
+
* `runtime-host.c` does, so the two print the same digits, fraction and all.
|
|
690
|
+
*/
|
|
691
|
+
export function statMtimeSync(path) {
|
|
692
|
+
const st = fs.statSync(path, { throwIfNoEntry: false });
|
|
693
|
+
if (st === undefined) {
|
|
694
|
+
return Number.NaN;
|
|
695
|
+
}
|
|
696
|
+
return st.mtimeMs;
|
|
697
|
+
}
|
|
698
|
+
|
|
699
|
+
/**
|
|
700
|
+
* Node's own fill, taken before `runtime/nish.mjs` puts the one below in its
|
|
701
|
+
* place on the same object: `webcrypto` is the global `crypto`.
|
|
702
|
+
*/
|
|
703
|
+
const webRandom = webcrypto.getRandomValues.bind(webcrypto);
|
|
704
|
+
|
|
705
|
+
/**
|
|
706
|
+
* `crypto.getRandomValues(bytes)` for the plain array a `u8[]` is here. Node's
|
|
707
|
+
* own takes only a typed array, so the bytes are drawn into one and copied
|
|
708
|
+
* across. More than 65,536 fails with the native panic and its words, where
|
|
709
|
+
* Node would throw a `QuotaExceededError`: the exit status is 1 either way. A
|
|
710
|
+
* typed array goes straight through.
|
|
711
|
+
*/
|
|
712
|
+
export function getRandomValues(bytes) {
|
|
713
|
+
if (!Array.isArray(bytes)) {
|
|
714
|
+
return webRandom(bytes);
|
|
715
|
+
}
|
|
716
|
+
if (bytes.length > 65536) {
|
|
717
|
+
panic(`crypto.getRandomValues: ${bytes.length} bytes asked for, and one call fills at most 65536`);
|
|
718
|
+
}
|
|
719
|
+
const drawn = webRandom(new Uint8Array(bytes.length));
|
|
720
|
+
for (let i = 0; i < drawn.length; i++) {
|
|
721
|
+
bytes[i] = drawn[i];
|
|
722
|
+
}
|
|
723
|
+
return bytes;
|
|
724
|
+
}
|
|
725
|
+
|
|
726
|
+
/**
|
|
727
|
+
* `signalFd()` and `readSignal(fd)` have no faithful reading under Node, and
|
|
728
|
+
* these say so rather than answer something else. Node delivers a signal to
|
|
729
|
+
* its event loop (`process.on("SIGTERM")`), and a blocking read keeps the loop
|
|
730
|
+
* from ever running, so no synchronous function here can learn that one
|
|
731
|
+
* arrived. docs/wp33-round-trip.md §3.5 has the row and the translation.
|
|
732
|
+
*/
|
|
733
|
+
export function signalFd() {
|
|
734
|
+
throw new Error(
|
|
735
|
+
"signalFd has no synchronous reading under Node: a signal reaches the event loop, which a blocking readSignal never returns to (docs/wp33-round-trip.md)"
|
|
736
|
+
);
|
|
737
|
+
}
|
|
738
|
+
|
|
739
|
+
export function readSignal() {
|
|
740
|
+
return signalFd();
|
|
741
|
+
}
|
|
742
|
+
|
|
621
743
|
/** `process.argv`: index 0 is the program (the script here, the executable natively), then the arguments. */
|
|
622
744
|
export function argv() {
|
|
623
745
|
return process.argv.slice(1);
|
package/scripts/build.sh
CHANGED
|
@@ -3,11 +3,12 @@
|
|
|
3
3
|
#
|
|
4
4
|
# scripts/build.sh <module.ll> [more .ll/.c files...] -o <out> [--profile debug|speed|size|wasm]
|
|
5
5
|
#
|
|
6
|
-
# The C runtime is
|
|
6
|
+
# The C runtime is four translation units and is named as one: an input
|
|
7
7
|
# <dir>/runtime.c also compiles <dir>/runtime-os.c, the half that wraps the
|
|
8
8
|
# system calls (files, directories, subprocesses, the environment, the clock),
|
|
9
|
-
#
|
|
10
|
-
# threads
|
|
9
|
+
# <dir>/runtime-parallel.c, the half that divides a range of work across
|
|
10
|
+
# threads, and <dir>/runtime-host.c, the wall clock, entropy, file times and
|
|
11
|
+
# signals. Each of those files says why they are compiled and measured apart.
|
|
11
12
|
#
|
|
12
13
|
# Profiles:
|
|
13
14
|
# debug clang defaults: no optimisation, symbols kept. The "before" number.
|
|
@@ -83,9 +84,9 @@ done
|
|
|
83
84
|
[ ${#inputs[@]} -gt 0 ] || { echo "error: no input files" >&2; exit 2; }
|
|
84
85
|
[ -n "$out" ] || { echo "error: -o <out> is required" >&2; exit 2; }
|
|
85
86
|
|
|
86
|
-
# The runtime is
|
|
87
|
-
# <dir>/runtime.c gets <dir>/runtime-os.c
|
|
88
|
-
# beside it. They were one file until the operating-system half was split out
|
|
87
|
+
# The runtime is four translation units, and a caller names one: whoever passes
|
|
88
|
+
# <dir>/runtime.c gets <dir>/runtime-os.c, <dir>/runtime-parallel.c and
|
|
89
|
+
# <dir>/runtime-host.c compiled beside it. They were one file until the operating-system half was split out
|
|
89
90
|
# for its own size budget, and the parallel half followed for the same reason
|
|
90
91
|
# (each file's header comment says why), and a link line is where those splits
|
|
91
92
|
# would otherwise leak: `nish --link` builds its command line in
|
|
@@ -97,7 +98,7 @@ done
|
|
|
97
98
|
for i in ${inputs[@]+"${inputs[@]}"}; do
|
|
98
99
|
case "$i" in
|
|
99
100
|
*/runtime.c|runtime.c)
|
|
100
|
-
for half in runtime-os.c runtime-parallel.c; do
|
|
101
|
+
for half in runtime-os.c runtime-parallel.c runtime-host.c; do
|
|
101
102
|
side="${i%runtime.c}$half"
|
|
102
103
|
have=0
|
|
103
104
|
for j in "${inputs[@]}"; do
|
|
@@ -25,10 +25,12 @@
|
|
|
25
25
|
*
|
|
26
26
|
* - **Every number is well-formed and in its band.** `NL1xxx` Phase 0,
|
|
27
27
|
* `NL2xxx` the checker, `NL3xxx` the driver, `NL4xxx` the interop
|
|
28
|
-
* sidecars, `
|
|
29
|
-
* in the tables: `NL0000` (no rule
|
|
30
|
-
* `
|
|
31
|
-
* A performance fragment is in
|
|
28
|
+
* sidecars, `NL8xxx` a WP33 portability warning, `NL9xxx` a WP15 section
|
|
29
|
+
* 8 performance warning. Band 0 is not in the tables: `NL0000` (no rule
|
|
30
|
+
* matched), `NL0001` (a syntax error), `NL0002` (the toolchain) and
|
|
31
|
+
* `NL0003` (an internal error) are constants. A performance fragment is in
|
|
32
|
+
* `performanceRules` and nowhere else, and a portability fragment is in
|
|
33
|
+
* `portabilityRules` and nowhere else.
|
|
32
34
|
* - **Nothing is used twice.** A number handed out once is never handed to a
|
|
33
35
|
* different rule, and a retired rule keeps its entry -- it matches nothing,
|
|
34
36
|
* so carrying it costs a string -- precisely so that its number stays
|
|
@@ -48,7 +50,8 @@
|
|
|
48
50
|
* simply not a match. So each table's strings are counted on their own
|
|
49
51
|
* and have to come to twice its pairs ([#107](https://github.com/amritk/nish/issues/107)).
|
|
50
52
|
* - **The `NL9xxx` codes run from `NL9001` with no gap**, one per WP15
|
|
51
|
-
* section 8 rule, so a missing number is a rule that lost its code
|
|
53
|
+
* section 8 rule, so a missing number is a rule that lost its code; and
|
|
54
|
+
* the `NL8xxx` codes run from `NL8001` the same way, one per WP33 row.
|
|
52
55
|
* - **`RULE_COUNT` is the number of entries.**
|
|
53
56
|
*
|
|
54
57
|
* A fragment is the longest literal run of its message's template -- the rule
|
|
@@ -68,11 +71,24 @@ const REGISTRY =
|
|
|
68
71
|
process.argv.slice(2).find((arg) => !arg.startsWith("--")) ?? path.join(ROOT, "src", "codes.ts")
|
|
69
72
|
|
|
70
73
|
/** The bands a table entry may use. Band 0 is constants, never a table row. */
|
|
71
|
-
const BANDS = new Set(["1", "2", "3", "4", "9"])
|
|
74
|
+
const BANDS = new Set(["1", "2", "3", "4", "8", "9"])
|
|
72
75
|
|
|
73
76
|
/** Shorter than this, a fragment would match half the suite. */
|
|
74
77
|
const MIN_FRAGMENT = 10
|
|
75
78
|
|
|
79
|
+
/**
|
|
80
|
+
* The three tables, the band each warning table owns alone, and that warning
|
|
81
|
+
* class's name: a warning table holds only its band's codes and its band's
|
|
82
|
+
* codes live only there, because `codeFor` matches a warning's message against
|
|
83
|
+
* its own table and nothing else. `null` is the table of errors, which takes
|
|
84
|
+
* every other band.
|
|
85
|
+
*/
|
|
86
|
+
const TABLES = [
|
|
87
|
+
["diagnosticRules", null, null],
|
|
88
|
+
["portabilityRules", "8", "portability"],
|
|
89
|
+
["performanceRules", "9", "performance"],
|
|
90
|
+
]
|
|
91
|
+
|
|
76
92
|
/**
|
|
77
93
|
* The text of one table in `src/codes.ts`: from its `name = (): string[] => [`
|
|
78
94
|
* to the `];` that closes it. Parsed per table, because the table a fragment
|
|
@@ -104,10 +120,7 @@ const inOrder = (a, b) => b.fragment.length - a.fragment.length || a.fragment.lo
|
|
|
104
120
|
const problems = (text) => {
|
|
105
121
|
const found = []
|
|
106
122
|
const tables = []
|
|
107
|
-
for (const [name,
|
|
108
|
-
["diagnosticRules", false],
|
|
109
|
-
["performanceRules", true],
|
|
110
|
-
]) {
|
|
123
|
+
for (const [name, owns, kind] of TABLES) {
|
|
111
124
|
const body = tableText(text, name)
|
|
112
125
|
if (body === null) {
|
|
113
126
|
found.push(`src/codes.ts has no \`${name}\` table`)
|
|
@@ -122,7 +135,7 @@ const problems = (text) => {
|
|
|
122
135
|
"without its other half shifts every later pairing `codeFor` makes"
|
|
123
136
|
)
|
|
124
137
|
}
|
|
125
|
-
tables.push({ name,
|
|
138
|
+
tables.push({ name, owns, kind, pairs })
|
|
126
139
|
} catch (err) {
|
|
127
140
|
found.push(err.message)
|
|
128
141
|
}
|
|
@@ -139,18 +152,22 @@ const problems = (text) => {
|
|
|
139
152
|
found.push(err.message)
|
|
140
153
|
}
|
|
141
154
|
if (whole.length !== all.length) {
|
|
142
|
-
found.push(`src/codes.ts holds ${whole.length} pairs, ${all.length} of them inside the
|
|
155
|
+
found.push(`src/codes.ts holds ${whole.length} pairs, ${all.length} of them inside the three tables`)
|
|
143
156
|
}
|
|
144
157
|
|
|
145
|
-
|
|
158
|
+
const warningBands = TABLES.map(([, owns]) => owns).filter((band) => band !== null)
|
|
159
|
+
for (const { name, owns, pairs } of tables) {
|
|
146
160
|
for (let i = 0; i < pairs.length; i++) {
|
|
147
161
|
const { fragment, code } = pairs[i]
|
|
148
162
|
const band = code[2]
|
|
149
163
|
if (!BANDS.has(band)) {
|
|
150
|
-
found.push(`${code} is not in a table band (1, 2, 3, 4 or 9): ${JSON.stringify(fragment)}`)
|
|
164
|
+
found.push(`${code} is not in a table band (1, 2, 3, 4, 8 or 9): ${JSON.stringify(fragment)}`)
|
|
165
|
+
}
|
|
166
|
+
if (owns !== null && band !== owns) {
|
|
167
|
+
found.push(`${code} is in \`${name}\`, which holds only NL${owns}xxx codes`)
|
|
151
168
|
}
|
|
152
|
-
if (
|
|
153
|
-
found.push(`${code} is in \`${name}\`, which holds ${
|
|
169
|
+
if (owns === null && warningBands.includes(band)) {
|
|
170
|
+
found.push(`${code} is in \`${name}\`, which holds no NL${band}xxx codes`)
|
|
154
171
|
}
|
|
155
172
|
if (fragment.trim().length < MIN_FRAGMENT) {
|
|
156
173
|
found.push(`${code}'s fragment ${JSON.stringify(fragment)} is under ${MIN_FRAGMENT} characters`)
|
|
@@ -161,15 +178,20 @@ const problems = (text) => {
|
|
|
161
178
|
}
|
|
162
179
|
}
|
|
163
180
|
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
)
|
|
181
|
+
// A warning band is gap-free from its first number, so a missing one is a
|
|
182
|
+
// rule that lost its code rather than a number nobody took yet.
|
|
183
|
+
for (const { name, owns, kind, pairs } of tables) {
|
|
184
|
+
if (owns === null) {
|
|
185
|
+
continue
|
|
186
|
+
}
|
|
187
|
+
const numbers = pairs.map((p) => Number(p.code.slice(3))).sort((a, b) => a - b)
|
|
188
|
+
const gap = numbers.findIndex((n, i) => n !== i + 1)
|
|
189
|
+
if (gap >= 0) {
|
|
190
|
+
const base = Number(owns) * 1000
|
|
191
|
+
found.push(
|
|
192
|
+
`\`${name}\` has no NL${base + gap + 1} in its place: its ${kind} codes run from NL${base + 1} with no gap`
|
|
193
|
+
)
|
|
194
|
+
}
|
|
173
195
|
}
|
|
174
196
|
|
|
175
197
|
const seen = (key) => {
|
package/std/README.md
CHANGED
|
@@ -16,6 +16,46 @@ whatever program imports it, and subject to the same rules as `examples/` or
|
|
|
16
16
|
| [`map.ts`](./map.ts) | `reserve(m, n)` and `getOrInsert(m, k, v)` for the global `Map`. Their bodies are the meaning, and what runs under Node: `reserve` does nothing, and `getOrInsert` is a `get`, and a `set` of `v` when the key was missing. Natively the compiler lowers every call in place — `reserve` to the table's `reserveSlots`, which grows the buckets once so that `n` entries fit without a rebuild, and `getOrInsert` to one `probe` and a `valueAt` or an `insertAt` through its answer — so, like `collections.ts`, it writes no `.ll` of its own ([`docs/wp32-map.md`](../docs/wp32-map.md) §9.2, [`docs/LANGUAGE.md`](../docs/LANGUAGE.md#map-and-set)) |
|
|
17
17
|
| [`threads.ts`](./threads.ts) | `parallelMapInto(src, dst, f)` and `parallelReduce(src, f, identity)`: a function over every element of an array, on as many threads as the length is worth. Its bodies are the sequential meaning, which is what runs under Node; the compiler recognises the two templates by module and name, lowers the one loop in each onto `nish_parallel_range`, holds the function to the rules that make that safe, and compiles an importing program with `--threads` ([`docs/LANGUAGE.md`](../docs/LANGUAGE.md#data-parallelism-nishthreads)). `tests/link/par_*` are its programs |
|
|
18
18
|
|
|
19
|
+
## `nish/crypto` — the primitives under TLS 1.3
|
|
20
|
+
|
|
21
|
+
The first lanes of [WP34](../docs/wp34-hosting-cs.md) §5: K1's hashes, MACs and
|
|
22
|
+
key derivation, and K4's key exchange, in pure Nish (decision S1), each module
|
|
23
|
+
imported by its own specifier. Every one is written from its specification
|
|
24
|
+
rather than ported, and reproduces that specification's published vectors in
|
|
25
|
+
its `tests/link/crypto_*` programs. The performance gate compiles every module
|
|
26
|
+
with no diagnostics under both `--number-mode i32` and `f64`, and the hashes,
|
|
27
|
+
HMAC, HKDF and X25519 also run their vectors in `f64` (`crypto_*_f64`).
|
|
28
|
+
|
|
29
|
+
| Module | What it is | Reproduces |
|
|
30
|
+
| --- | --- | --- |
|
|
31
|
+
| [`crypto/sha256.ts`](./crypto/sha256.ts) | `sha256(data)`, and `Sha256`, a streaming hasher: `update(buf, off, len)` over a window of a `u8[]`, `copy()` for the hash of a prefix while the original keeps going, and `digest()`, a fresh 32-byte array. `SHA256_SIZE` and `SHA256_BLOCK` | FIPS 180-4 §6.2 |
|
|
32
|
+
| [`crypto/sha512.ts`](./crypto/sha512.ts) | SHA-512 and SHA-384 on one compression function: `sha512` and `sha384`, and the streaming `Sha512` and `Sha384` with `Sha256`'s three methods; digests of 64 and 48 bytes. `SHA512_SIZE`, `SHA384_SIZE` and `SHA512_BLOCK` | FIPS 180-4 §6.4, §6.5 |
|
|
33
|
+
| [`crypto/hmac.ts`](./crypto/hmac.ts) | `hmacSha256` and `hmacSha384`, the streaming `HmacSha256` and `HmacSha384` (keyed in the constructor, then `update` and `digest`), and `hmacSha256Verify` / `hmacSha384Verify`, which compare a received tag with `timingSafeEqual` | RFC 2104, RFC 4231 §4 |
|
|
34
|
+
| [`crypto/hkdf.ts`](./crypto/hkdf.ts) | `hkdfExtractSha256` / `hkdfExtractSha384` (an empty salt is HashLen zeros) and `hkdfExpandSha256` / `hkdfExpandSha384`, which answer `null` for a length below zero or above 255 × HashLen. TLS 1.3's HKDF-Expand-Label is not here; it belongs with TLS | RFC 5869 §2, Appendix A |
|
|
35
|
+
| [`crypto/ct.ts`](./crypto/ct.ts) | `timingSafeEqual(a, b)`, which reads every byte whatever it holds, and `timingSafeEqualAt(a, aOff, b, bOff, len)` over two windows, which answers `false` for a window outside its array. Two lengths that differ answer `false` at once, because a length is public | — |
|
|
36
|
+
| [`crypto/base64url.ts`](./crypto/base64url.ts) | `base64urlEncode(data)` and `base64urlDecode(text)`, unpadded. Decoding is strict, so every byte string has one spelling: a `=`, a character outside the alphabet, a length of 1 mod 4 or nonzero unused low bits answer `null` | RFC 4648 §5, §10 |
|
|
37
|
+
| [`crypto/x25519.ts`](./crypto/x25519.ts) | `x25519(scalar, u)` and `x25519Base(scalar)`, on ten 25.5-bit limbs in `i64`. Either answers `null` unless its arguments are `X25519_SIZE` (32) bytes; the scalar is clamped on a copy | RFC 7748 §5.2, §6.1 |
|
|
38
|
+
|
|
39
|
+
Three rules hold across the modules:
|
|
40
|
+
|
|
41
|
+
- **A digest ends the computation.** After `digest()` on a hasher or an HMAC, a
|
|
42
|
+
further `update` or `digest` panics rather than answering a hash over the
|
|
43
|
+
padding, and so does a window outside its buffer. `Sha256.copy()` on a
|
|
44
|
+
digested hasher panics too; `Sha512.copy()` and `Sha384.copy()` answer a copy
|
|
45
|
+
that is itself spent, so any `update` or `digest` on it panics. Either way,
|
|
46
|
+
copy *before* `digest` when the computation has to go on.
|
|
47
|
+
- **An all-zero X25519 result is returned, not refused.** It is what a
|
|
48
|
+
low-order `u` gives, and RFC 7748 §6.1 leaves the check to the protocol; TLS
|
|
49
|
+
1.3 (WP34 T1) makes it. A key exchange outside TLS has to make it itself.
|
|
50
|
+
- **Constant time by construction, not yet by proof.** No module branches on,
|
|
51
|
+
or indexes by, a secret: comparisons OR the differences into one word and
|
|
52
|
+
test it once, the ladder swaps with a mask and always runs 255 steps, and
|
|
53
|
+
base64url maps characters by arithmetic on range masks rather than a table.
|
|
54
|
+
Every branch is on a length, a loop counter or a bit position. What checks
|
|
55
|
+
that the machine code kept that shape is WP34 N6 — the `ctSelect` / `ctEq`
|
|
56
|
+
builtins behind an optimisation barrier, and a disassembly check — and it is
|
|
57
|
+
not built yet, so this is the discipline and not a verified property.
|
|
58
|
+
|
|
19
59
|
## How a program imports it
|
|
20
60
|
|
|
21
61
|
By its package specifier:
|
|
@@ -0,0 +1,145 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `nish/crypto/base64url` — RFC 4648 §5 base64url, unpadded, as JWS and the
|
|
3
|
+
* relay's grants spell a key or a tag inside a URL or a header.
|
|
4
|
+
*
|
|
5
|
+
* The alphabet is `A-Z a-z 0-9 - _`, and there is no `=`: the length of the
|
|
6
|
+
* text already says how many bytes the last group holds (RFC 4648 §3.2 lets a
|
|
7
|
+
* specification that knows its lengths drop the padding, and RFC 7515 §2 does).
|
|
8
|
+
*
|
|
9
|
+
* **Decoding is strict, so that each byte string has exactly one spelling.**
|
|
10
|
+
* `base64urlDecode` answers `null` for a `=` anywhere, for any character outside
|
|
11
|
+
* the alphabet (whitespace included), for a length of 1 mod 4 — six bits, not a
|
|
12
|
+
* byte — and for a final character whose unused low bits are not zero (RFC 4648
|
|
13
|
+
* §3.5). A lenient decoder would let `AB` and `AA` both mean `00`, and a token
|
|
14
|
+
* compared or cached by its text would then have two identities.
|
|
15
|
+
*
|
|
16
|
+
* **Neither direction indexes or branches on the data.** What is encoded here
|
|
17
|
+
* is usually a key, a nonce or a MAC, and a lookup table indexed by a secret
|
|
18
|
+
* sextet leaves its trace in the cache. So a sextet becomes a character, and a
|
|
19
|
+
* character a sextet, by arithmetic on range masks: `(lo - 1 - c) & (c - hi - 1)`
|
|
20
|
+
* is negative exactly when `lo <= c <= hi`, and an arithmetic shift by 31 turns
|
|
21
|
+
* that sign into an all-ones or all-zeros mask. A malformed character is not
|
|
22
|
+
* refused where it is found either: the verdict is ORed into one word and read
|
|
23
|
+
* once, after the whole text. The text's length is public and is tested first.
|
|
24
|
+
*
|
|
25
|
+
* Private helpers share the importing program's flat symbol namespace
|
|
26
|
+
* (`docs/wp26-stdlib.md` §3e), which is why each one carries the module's name.
|
|
27
|
+
*/
|
|
28
|
+
|
|
29
|
+
/**
|
|
30
|
+
* All ones when `lo <= c <= hi`, else zero, without a branch. `c` is a byte or
|
|
31
|
+
* a sextet, so neither subtraction can overflow.
|
|
32
|
+
*/
|
|
33
|
+
const base64urlRangeMask = (c: i32, lo: i32, hi: i32): i32 => ((lo - 1 - c) & (c - hi - 1)) >> 31
|
|
34
|
+
|
|
35
|
+
/**
|
|
36
|
+
* The URL-alphabet character for the sextet `v` (`0 <= v <= 63`).
|
|
37
|
+
*
|
|
38
|
+
* It starts from `'A' + v` and adds, for each range `v` has passed, the step
|
|
39
|
+
* from the previous range's first character to this one's: 26 lands on `a`, 52
|
|
40
|
+
* on `0`, 62 on `-` and 63 on `_`.
|
|
41
|
+
*/
|
|
42
|
+
const base64urlCharOf = (v: i32): i32 =>
|
|
43
|
+
65 +
|
|
44
|
+
v +
|
|
45
|
+
(base64urlRangeMask(v, 26, 63) & 6) -
|
|
46
|
+
(base64urlRangeMask(v, 52, 63) & 75) -
|
|
47
|
+
(base64urlRangeMask(v, 62, 63) & 13) +
|
|
48
|
+
(base64urlRangeMask(v, 63, 63) & 49)
|
|
49
|
+
|
|
50
|
+
/**
|
|
51
|
+
* The sextet the byte `c` stands for, or `-1` when `c` is not in the alphabet.
|
|
52
|
+
* Each range contributes its value under its own mask, and a byte in none of
|
|
53
|
+
* them has every bit set by the final OR.
|
|
54
|
+
*/
|
|
55
|
+
const base64urlSextetOf = (c: i32): i32 => {
|
|
56
|
+
const upper: i32 = base64urlRangeMask(c, 65, 90)
|
|
57
|
+
const lower: i32 = base64urlRangeMask(c, 97, 122)
|
|
58
|
+
const digit: i32 = base64urlRangeMask(c, 48, 57)
|
|
59
|
+
const dash: i32 = base64urlRangeMask(c, 45, 45)
|
|
60
|
+
const underscore: i32 = base64urlRangeMask(c, 95, 95)
|
|
61
|
+
const value: i32 =
|
|
62
|
+
(upper & (c - 65)) | (lower & (c - 71)) | (digit & (c + 4)) | (dash & 62) | (underscore & 63)
|
|
63
|
+
return value | ~(upper | lower | digit | dash | underscore)
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
/**
|
|
67
|
+
* `data` as unpadded base64url: four characters per three bytes, and two or
|
|
68
|
+
* three for a short final group.
|
|
69
|
+
*
|
|
70
|
+
* Bytes go into a bit accumulator eight at a time and come out six at a time,
|
|
71
|
+
* so one index walks the input and the only branches are on how many bits are
|
|
72
|
+
* waiting — which depends on the position, never on the bytes.
|
|
73
|
+
*/
|
|
74
|
+
export const base64urlEncode = (data: u8[]): string => {
|
|
75
|
+
const parts: string[] = []
|
|
76
|
+
let acc: i32 = 0
|
|
77
|
+
let bits: i32 = 0
|
|
78
|
+
for (let k: i32 = 0; k < toI32(data.length); k++) {
|
|
79
|
+
// At most four bits wait from the bytes before, so twelve are live here;
|
|
80
|
+
// the mask keeps the accumulator from growing without bound.
|
|
81
|
+
acc = ((acc << 8) | toI32(data[k])) & 0xfff
|
|
82
|
+
bits += 8
|
|
83
|
+
while (bits >= 6) {
|
|
84
|
+
bits -= 6
|
|
85
|
+
parts.push(String.fromCharCode(base64urlCharOf((acc >> bits) & 63)))
|
|
86
|
+
}
|
|
87
|
+
}
|
|
88
|
+
// Two or four bits of a final byte are left: they go out as the high bits of
|
|
89
|
+
// one more sextet, filled out with zeros.
|
|
90
|
+
if (bits > 0) {
|
|
91
|
+
parts.push(String.fromCharCode(base64urlCharOf((acc << (6 - bits)) & 63)))
|
|
92
|
+
}
|
|
93
|
+
return parts.join("")
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
/**
|
|
97
|
+
* The bytes `text` spells in unpadded base64url, or `null` when it is not the
|
|
98
|
+
* one canonical spelling of any byte string — see the module comment for the
|
|
99
|
+
* four refusals.
|
|
100
|
+
*
|
|
101
|
+
* The same accumulator as `base64urlEncode`, run the other way: six bits in per
|
|
102
|
+
* character, eight out per byte.
|
|
103
|
+
*/
|
|
104
|
+
export const base64urlDecode = (text: string): u8[] | null => {
|
|
105
|
+
const n: i32 = toI32(text.length)
|
|
106
|
+
const tail: i32 = n & 3
|
|
107
|
+
// Six bits cannot finish a byte, so a single character after the last full
|
|
108
|
+
// group spells nothing.
|
|
109
|
+
if (tail === 1) {
|
|
110
|
+
return null
|
|
111
|
+
}
|
|
112
|
+
// Three bytes per full group of four, and one fewer than the characters in a
|
|
113
|
+
// short final group; `n * 3 / 4` would overflow for a text over 700 MB.
|
|
114
|
+
const out: u8[] = new Array<u8>((n >> 2) * 3 + (tail === 0 ? 0 : tail - 1))
|
|
115
|
+
const outLen: i32 = toI32(out.length)
|
|
116
|
+
// All ones once any character is outside the alphabet (its sextet of `-1`
|
|
117
|
+
// shifted right by 31), and non-zero once the text ends on a bit that
|
|
118
|
+
// encodes no byte.
|
|
119
|
+
let bad: i32 = 0
|
|
120
|
+
let acc: i32 = 0
|
|
121
|
+
let bits: i32 = 0
|
|
122
|
+
let j: i32 = 0
|
|
123
|
+
for (let k: i32 = 0; k < n; k++) {
|
|
124
|
+
const v: i32 = base64urlSextetOf(toI32(text.charCodeAt(k)))
|
|
125
|
+
bad = bad | (v >> 31)
|
|
126
|
+
acc = ((acc << 6) | (v & 63)) & 0xfff
|
|
127
|
+
bits += 6
|
|
128
|
+
if (bits >= 8) {
|
|
129
|
+
bits -= 8
|
|
130
|
+
// Always true — `out` was sized from the same length — but written out
|
|
131
|
+
// so that the store keeps no bounds check.
|
|
132
|
+
if (j >= 0 && j < outLen) {
|
|
133
|
+
out[j] = toU8((acc >> bits) & 255)
|
|
134
|
+
}
|
|
135
|
+
j++
|
|
136
|
+
}
|
|
137
|
+
}
|
|
138
|
+
// The two or four bits left over belong to no byte, and must be zero so that
|
|
139
|
+
// `AA` is the only spelling of a zero byte and `AB` is refused (RFC 4648 §3.5).
|
|
140
|
+
bad = bad | (acc & ((1 << bits) - 1))
|
|
141
|
+
if (bad !== 0) {
|
|
142
|
+
return null
|
|
143
|
+
}
|
|
144
|
+
return out
|
|
145
|
+
}
|
package/std/crypto/ct.ts
ADDED
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `nish/crypto/ct` — comparing secrets without telling the clock where they differ.
|
|
3
|
+
*
|
|
4
|
+
* An `===` loop over two MACs stops at the first differing byte, so the time it
|
|
5
|
+
* takes says how many leading bytes an attacker already guessed right, and a
|
|
6
|
+
* tag can be recovered one byte at a time. These functions read every byte of
|
|
7
|
+
* the window whatever it holds: each pair is XORed, the differences are ORed
|
|
8
|
+
* into one accumulator, and that accumulator is tested once, after the loop.
|
|
9
|
+
*
|
|
10
|
+
* What is *not* secret is checked first and may branch: the two lengths, and
|
|
11
|
+
* whether a window lies inside its array. A caller comparing a received tag
|
|
12
|
+
* against an expected one already knows both lengths, so refusing a mismatch
|
|
13
|
+
* early leaks nothing the attacker did not send.
|
|
14
|
+
*
|
|
15
|
+
* The names are `timingSafeEqual*`, after Node's `crypto.timingSafeEqual`,
|
|
16
|
+
* rather than `ctEq`: that one is reserved for the builtin WP34 N6 will add,
|
|
17
|
+
* together with the disassembly check that proves the loop stays branch-free
|
|
18
|
+
* after LLVM has seen it. Until then this is the discipline and not a proof.
|
|
19
|
+
*/
|
|
20
|
+
|
|
21
|
+
/**
|
|
22
|
+
* Whether `a` and `b` hold the same bytes, reading all of them.
|
|
23
|
+
*
|
|
24
|
+
* Arrays of different lengths answer `false` at once, because a length is
|
|
25
|
+
* public; arrays of one length are compared in full, with no early exit.
|
|
26
|
+
*/
|
|
27
|
+
export const timingSafeEqual = (a: u8[], b: u8[]): boolean => {
|
|
28
|
+
const n: i32 = toI32(a.length)
|
|
29
|
+
if (n !== toI32(b.length)) {
|
|
30
|
+
return false
|
|
31
|
+
}
|
|
32
|
+
let diff: i32 = 0
|
|
33
|
+
// Bounded by both lengths, which are equal here, so the prover drops both
|
|
34
|
+
// bounds checks rather than trusting the comparison above.
|
|
35
|
+
for (let i: i32 = 0; i < n && i < toI32(b.length); i++) {
|
|
36
|
+
diff = diff | toI32(a[i] ^ b[i])
|
|
37
|
+
}
|
|
38
|
+
return diff === 0
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
/**
|
|
42
|
+
* Whether the `len` bytes of `a` from `aOff` equal the `len` bytes of `b` from
|
|
43
|
+
* `bOff`, reading all of them.
|
|
44
|
+
*
|
|
45
|
+
* A window that does not lie inside its array — a negative offset or length,
|
|
46
|
+
* or `off + len` past the end — answers `false` instead of panicking on the
|
|
47
|
+
* first out-of-range read. That is a decision about the caller's position: a
|
|
48
|
+
* verifier handed a truncated record should say "not equal", not take the
|
|
49
|
+
* process down, and the bounds are public, so testing them first is safe.
|
|
50
|
+
*/
|
|
51
|
+
export const timingSafeEqualAt = (a: u8[], aOff: i32, b: u8[], bOff: i32, len: i32): boolean => {
|
|
52
|
+
const aLen: i32 = toI32(a.length)
|
|
53
|
+
const bLen: i32 = toI32(b.length)
|
|
54
|
+
// Written as `off > length - len` rather than `off + len > length`, so that
|
|
55
|
+
// a huge offset cannot wrap round to a small sum and pass.
|
|
56
|
+
if (aOff < 0 || bOff < 0 || len < 0 || aOff > aLen - len || bOff > bLen - len) {
|
|
57
|
+
return false
|
|
58
|
+
}
|
|
59
|
+
let diff: i32 = 0
|
|
60
|
+
for (let k: i32 = 0; k < len; k++) {
|
|
61
|
+
diff = diff | toI32(a[aOff + k] ^ b[bOff + k])
|
|
62
|
+
}
|
|
63
|
+
return diff === 0
|
|
64
|
+
}
|