@amritk/nish-aarch64-linux 0.14.0 → 0.15.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/nish +0 -0
- package/package.json +1 -1
- package/runtime/nish.d.ts +91 -0
- package/runtime/nish.h +58 -0
- package/runtime/nish.mjs +23 -0
- package/runtime/runtime-net.c +528 -0
- package/runtime/shim.mjs +15 -0
- package/scripts/build.sh +8 -7
- package/std/README.md +54 -14
- package/std/crypto/LICENSE-bearssl +21 -0
- package/std/crypto/LICENSE-fiat-crypto +21 -0
- package/std/crypto/aes.ts +899 -0
- package/std/crypto/chacha20poly1305.ts +698 -0
- package/std/crypto/p256.ts +4337 -0
- package/std/crypto/x25519.ts +22 -7
- package/std/crypto/x509.ts +1214 -0
|
@@ -0,0 +1,899 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `nish/crypto/aes` — AES-128 and AES-256 (FIPS 197), GCM on top of them
|
|
3
|
+
* (NIST SP 800-38D), and QUIC's AES header-protection mask (RFC 9001 §5.4.3).
|
|
4
|
+
*
|
|
5
|
+
* Written from the specifications and the papers below, in this module's own
|
|
6
|
+
* layout, with one exception: `ghashMul32` is adapted from BearSSL's
|
|
7
|
+
* `bmul64`, and carries BearSSL's notice. Nothing else here is ported from
|
|
8
|
+
* another implementation.
|
|
9
|
+
*
|
|
10
|
+
* - **Bitslicing** is the idea of E. Käsper and P. Schwabe, "Faster and
|
|
11
|
+
* Timing-Attack Resistant AES-GCM" (CHES 2009): hold bit `b` of many bytes
|
|
12
|
+
* in one word, so that SubBytes is a Boolean circuit run on whole words and
|
|
13
|
+
* no byte ever becomes a table index.
|
|
14
|
+
* - **The S-box circuit** is J. Boyar and R. Peralta's, "A depth-16 circuit
|
|
15
|
+
* for the AES S-box" (SEC 2012) and the 113-gate listing published with it:
|
|
16
|
+
* a linear layer in, 32 ANDs over GF(2^4) and GF(2^2) arithmetic, a linear
|
|
17
|
+
* layer out. `aesSbox` keeps their variable names so it can be read against
|
|
18
|
+
* the paper, gate by gate.
|
|
19
|
+
* - **GHASH's multiply** is carry-less multiplication done with the integer
|
|
20
|
+
* multiplier, by leaving three zero bits of "holes" between the bits of each
|
|
21
|
+
* operand so that the carries of an integer product land where nothing is
|
|
22
|
+
* read — T. Pornin's technique, and `ghashMul32` is his `bmul64` from
|
|
23
|
+
* BearSSL narrowed to 32-bit operands — with Karatsuba's three-for-four
|
|
24
|
+
* split on top (this module's own), and the
|
|
25
|
+
* reduction of S. Gueron and M. Kounavis, "Intel Carry-Less Multiplication
|
|
26
|
+
* Instruction and its Usage for Computing the GCM Mode" (2010), for GCM's
|
|
27
|
+
* bit-reflected field.
|
|
28
|
+
*
|
|
29
|
+
* **The bitsliced state.** Four blocks at a time sit in eight `u64` words,
|
|
30
|
+
* `q[0..7]`, word `b` holding bit `b` of all 64 bytes. Inside a word, block `k`
|
|
31
|
+
* owns the 16 bits from `16k`, and in those the byte in row `r` and column `c`
|
|
32
|
+
* of the AES state is bit `4r + c`: each row is one nibble. That makes the two
|
|
33
|
+
* permutations cheap. ShiftRows turns row `r` by `r` columns, which is a
|
|
34
|
+
* rotation of nibble `r` by `r` bits; MixColumns needs each byte's neighbour
|
|
35
|
+
* in the next row of its column, which is the word rotated by four bits inside
|
|
36
|
+
* each 16-bit lane. Getting 64 bytes in and out of that layout is an 8 × 8
|
|
37
|
+
* bit-matrix transpose per byte lane (`aesTranspose`, three rounds of masked
|
|
38
|
+
* swaps between words) after placing each byte in the word and slot the
|
|
39
|
+
* transpose will carry to its bit (`aesPack`, `aesUnpack`).
|
|
40
|
+
*
|
|
41
|
+
* **Constant time.** No table is read anywhere, key schedule included: the
|
|
42
|
+
* S-box is the circuit, and `SubWord` in the key schedule runs the same
|
|
43
|
+
* circuit on a four-byte state. No branch or index depends on a key, a
|
|
44
|
+
* plaintext, a tag or H: every `if` and every loop bound is on a length, a
|
|
45
|
+
* block number or the round count. GHASH multiplies with no table and no
|
|
46
|
+
* data-dependent shift, and `aesGcmOpen` computes the whole tag and compares
|
|
47
|
+
* it with `aesGcmTagMask` — a difference ORed into one word and handed to
|
|
48
|
+
* `ctEq` — before it decrypts or returns anything. What `tests/ct-asm.js`
|
|
49
|
+
* verifies by disassembly on x86-64 and aarch64 is one full round
|
|
50
|
+
* (`aesBitslicedRound`: S-box circuit, ShiftRows, MixColumns, AddRoundKey),
|
|
51
|
+
* one GHASH multiply (`ghashMultiply`) and the tag compare
|
|
52
|
+
* (`aesGcmTagMask`). A golden case compiles to one module and cannot import
|
|
53
|
+
* this one, so `tests/cases/ct_asm_aes` holds a verbatim copy of those three
|
|
54
|
+
* and the helpers they call, and `tests/link/crypto_aes` checks that the copy
|
|
55
|
+
* and this module agree on generated inputs, so neither can drift from the
|
|
56
|
+
* other unnoticed. Everything around them — packing bytes, the key schedule,
|
|
57
|
+
* the loops over blocks — is the discipline above and not a verified
|
|
58
|
+
* property.
|
|
59
|
+
*
|
|
60
|
+
* **What is refused.** A key of any length but 16 or 32 bytes (AES-192 is out
|
|
61
|
+
* of scope), a block or a sample that is not 16 bytes, an empty IV (SP 800-38D
|
|
62
|
+
* §5.2.1.1 asks for at least one bit), a sealed input shorter than its tag and
|
|
63
|
+
* a plaintext too long for its sealed form to be an array answer `null`; so
|
|
64
|
+
* does a tag that does not verify. An IV of any other non-zero
|
|
65
|
+
* length is hashed into the first counter block as §7.1 says.
|
|
66
|
+
*
|
|
67
|
+
* Private names carry the `aes` / `ghash` prefix because a `std/` module's
|
|
68
|
+
* private functions share the importing program's flat symbol namespace
|
|
69
|
+
* (`docs/wp26-stdlib.md` §3e).
|
|
70
|
+
*/
|
|
71
|
+
|
|
72
|
+
/** The AES block, in bytes. */
|
|
73
|
+
export const AES_BLOCK: i32 = 16
|
|
74
|
+
/** GCM's tag, in bytes: this module makes and accepts only the full 128 bits. */
|
|
75
|
+
export const AES_GCM_TAG_SIZE: i32 = 16
|
|
76
|
+
|
|
77
|
+
/** The 64 bytes one pass of the bitsliced cipher encrypts: four blocks. */
|
|
78
|
+
const AES_BATCH: i32 = 64
|
|
79
|
+
|
|
80
|
+
/**
|
|
81
|
+
* An AES key, expanded: the round keys already bitsliced, and GCM's hash key
|
|
82
|
+
* H = E(K, 0^128), so a key used for many messages is expanded once.
|
|
83
|
+
*/
|
|
84
|
+
export class AesKey {
|
|
85
|
+
/** Nr: 10 for AES-128, 14 for AES-256. */
|
|
86
|
+
rounds: i32 = 0
|
|
87
|
+
/**
|
|
88
|
+
* Round key `i` in the bitsliced layout, replicated into all four block
|
|
89
|
+
* lanes, as eight words from `8 * i`: `8 * (rounds + 1)` words in all.
|
|
90
|
+
*/
|
|
91
|
+
roundKeys: u64[]
|
|
92
|
+
/** H as a big-endian 128-bit number, high half. */
|
|
93
|
+
hHi: u64 = 0
|
|
94
|
+
/** H, low half. */
|
|
95
|
+
hLo: u64 = 0
|
|
96
|
+
|
|
97
|
+
constructor(rounds: i32, roundKeys: u64[]) {
|
|
98
|
+
this.rounds = rounds
|
|
99
|
+
this.roundKeys = roundKeys
|
|
100
|
+
}
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
/** A 16-bit pattern repeated into each of a word's four 16-bit lanes. */
|
|
104
|
+
const aesLanes = (pattern: u64): u64 => {
|
|
105
|
+
const two: u64 = pattern | (pattern << toU64(16))
|
|
106
|
+
return two | (two << toU64(32))
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
/** An 8-bit pattern repeated into each of a word's eight bytes. */
|
|
110
|
+
const aesBytes = (pattern: u64): u64 => aesLanes(pattern | (pattern << toU64(8)))
|
|
111
|
+
|
|
112
|
+
/**
|
|
113
|
+
* Swaps, in every byte lane, the bits of `q[i]` selected by `mask << shift`
|
|
114
|
+
* with the bits of `q[j]` selected by `mask`: one round of the transpose.
|
|
115
|
+
*/
|
|
116
|
+
const aesSwapBits = (q: u64[], i: i32, j: i32, mask: u64, shift: u64): void => {
|
|
117
|
+
const t: u64 = ((q[i] >> shift) ^ q[j]) & mask
|
|
118
|
+
q[j] = q[j] ^ t
|
|
119
|
+
q[i] = q[i] ^ (t << shift)
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
/**
|
|
123
|
+
* Transposes, in each of the eight byte lanes, the 8 × 8 bit matrix whose
|
|
124
|
+
* row `i` is that byte of `q[i]`: afterwards bit `i` of the byte in `q[b]` is
|
|
125
|
+
* what bit `b` of the byte in `q[i]` was. Its own inverse, so it both enters
|
|
126
|
+
* and leaves the bitsliced layout. Swaps 1 × 1, then 2 × 2, then 4 × 4
|
|
127
|
+
* blocks across the diagonal.
|
|
128
|
+
*/
|
|
129
|
+
const aesTranspose = (q: u64[]): void => {
|
|
130
|
+
const m1: u64 = aesBytes(toU64(0x55))
|
|
131
|
+
const m2: u64 = aesBytes(toU64(0x33))
|
|
132
|
+
const m4: u64 = aesBytes(toU64(0x0f))
|
|
133
|
+
aesSwapBits(q, 0, 1, m1, toU64(1))
|
|
134
|
+
aesSwapBits(q, 2, 3, m1, toU64(1))
|
|
135
|
+
aesSwapBits(q, 4, 5, m1, toU64(1))
|
|
136
|
+
aesSwapBits(q, 6, 7, m1, toU64(1))
|
|
137
|
+
aesSwapBits(q, 0, 2, m2, toU64(2))
|
|
138
|
+
aesSwapBits(q, 1, 3, m2, toU64(2))
|
|
139
|
+
aesSwapBits(q, 4, 6, m2, toU64(2))
|
|
140
|
+
aesSwapBits(q, 5, 7, m2, toU64(2))
|
|
141
|
+
aesSwapBits(q, 0, 4, m4, toU64(4))
|
|
142
|
+
aesSwapBits(q, 1, 5, m4, toU64(4))
|
|
143
|
+
aesSwapBits(q, 2, 6, m4, toU64(4))
|
|
144
|
+
aesSwapBits(q, 3, 7, m4, toU64(4))
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
/**
|
|
148
|
+
* Loads the four blocks of `buf[0 .. 64)` into the bitsliced state. After the
|
|
149
|
+
* transpose, bit `8p + i` of every word comes from byte slot `p` of `q[i]`,
|
|
150
|
+
* and that bit must be `16k + 4r + c` for block `k`'s byte in row `r` and
|
|
151
|
+
* column `c` (`buf[16k + 4c + r]`): so `i = 4 * (r & 1) + c` and
|
|
152
|
+
* `p = 2k + (r >> 1)`, which is the four lines in the loop.
|
|
153
|
+
*/
|
|
154
|
+
const aesPack = (q: u64[], buf: u8[]): void => {
|
|
155
|
+
if (toI32(q.length) < 8 || toI32(buf.length) < 64) {
|
|
156
|
+
return
|
|
157
|
+
}
|
|
158
|
+
for (let i: i32 = 0; i < 8; i++) {
|
|
159
|
+
q[i] = 0
|
|
160
|
+
}
|
|
161
|
+
for (let k: i32 = 0; k < 4; k++) {
|
|
162
|
+
const low: u64 = toU64(16 * k)
|
|
163
|
+
const high: u64 = toU64(16 * k + 8)
|
|
164
|
+
for (let c: i32 = 0; c < 4; c++) {
|
|
165
|
+
q[c] = q[c] | (toU64(buf[16 * k + 4 * c]) << low) | (toU64(buf[16 * k + 4 * c + 2]) << high)
|
|
166
|
+
q[c + 4] = q[c + 4] | (toU64(buf[16 * k + 4 * c + 1]) << low) | (toU64(buf[16 * k + 4 * c + 3]) << high)
|
|
167
|
+
}
|
|
168
|
+
}
|
|
169
|
+
aesTranspose(q)
|
|
170
|
+
}
|
|
171
|
+
|
|
172
|
+
/** The inverse of `aesPack`: the four blocks of the state into `buf[0 .. 64)`. Spends `q`. */
|
|
173
|
+
const aesUnpack = (buf: u8[], q: u64[]): void => {
|
|
174
|
+
if (toI32(q.length) < 8 || toI32(buf.length) < 64) {
|
|
175
|
+
return
|
|
176
|
+
}
|
|
177
|
+
aesTranspose(q)
|
|
178
|
+
// `toU8` keeps the low eight bits, so each byte needs no mask.
|
|
179
|
+
for (let k: i32 = 0; k < 4; k++) {
|
|
180
|
+
const low: u64 = toU64(16 * k)
|
|
181
|
+
const high: u64 = toU64(16 * k + 8)
|
|
182
|
+
for (let c: i32 = 0; c < 4; c++) {
|
|
183
|
+
buf[16 * k + 4 * c] = toU8(q[c] >> low)
|
|
184
|
+
buf[16 * k + 4 * c + 2] = toU8(q[c] >> high)
|
|
185
|
+
buf[16 * k + 4 * c + 1] = toU8(q[c + 4] >> low)
|
|
186
|
+
buf[16 * k + 4 * c + 3] = toU8(q[c + 4] >> high)
|
|
187
|
+
}
|
|
188
|
+
}
|
|
189
|
+
}
|
|
190
|
+
|
|
191
|
+
/**
|
|
192
|
+
* SubBytes on every byte of the state at once: Boyar and Peralta's circuit,
|
|
193
|
+
* with their names. `x0` is the most significant bit of a byte, so it is
|
|
194
|
+
* `q[7]`, and the answer `s0` goes back to `q[7]` too. The three `^ ~` terms
|
|
195
|
+
* are the circuit's XNOR gates, the affine constant 0x63.
|
|
196
|
+
*/
|
|
197
|
+
const aesSbox = (q: u64[]): void => {
|
|
198
|
+
const x0: u64 = q[7]
|
|
199
|
+
const x1: u64 = q[6]
|
|
200
|
+
const x2: u64 = q[5]
|
|
201
|
+
const x3: u64 = q[4]
|
|
202
|
+
const x4: u64 = q[3]
|
|
203
|
+
const x5: u64 = q[2]
|
|
204
|
+
const x6: u64 = q[1]
|
|
205
|
+
const x7: u64 = q[0]
|
|
206
|
+
|
|
207
|
+
// The top linear layer.
|
|
208
|
+
const y14: u64 = x3 ^ x5
|
|
209
|
+
const y13: u64 = x0 ^ x6
|
|
210
|
+
const y9: u64 = x0 ^ x3
|
|
211
|
+
const y8: u64 = x0 ^ x5
|
|
212
|
+
const t0: u64 = x1 ^ x2
|
|
213
|
+
const y1: u64 = t0 ^ x7
|
|
214
|
+
const y4: u64 = y1 ^ x3
|
|
215
|
+
const y12: u64 = y13 ^ y14
|
|
216
|
+
const y2: u64 = y1 ^ x0
|
|
217
|
+
const y5: u64 = y1 ^ x6
|
|
218
|
+
const y3: u64 = y5 ^ y8
|
|
219
|
+
const t1: u64 = x4 ^ y12
|
|
220
|
+
const y15: u64 = t1 ^ x5
|
|
221
|
+
const y20: u64 = t1 ^ x1
|
|
222
|
+
const y6: u64 = y15 ^ x7
|
|
223
|
+
const y10: u64 = y15 ^ t0
|
|
224
|
+
const y11: u64 = y20 ^ y9
|
|
225
|
+
const y7: u64 = x7 ^ y11
|
|
226
|
+
const y17: u64 = y10 ^ y11
|
|
227
|
+
const y19: u64 = y10 ^ y8
|
|
228
|
+
const y16: u64 = t0 ^ y11
|
|
229
|
+
const y21: u64 = y13 ^ y16
|
|
230
|
+
const y18: u64 = x0 ^ y16
|
|
231
|
+
|
|
232
|
+
// The non-linear middle: inversion in GF(2^8) through GF(2^4).
|
|
233
|
+
const t2: u64 = y12 & y15
|
|
234
|
+
const t3: u64 = y3 & y6
|
|
235
|
+
const t4: u64 = t3 ^ t2
|
|
236
|
+
const t5: u64 = y4 & x7
|
|
237
|
+
const t6: u64 = t5 ^ t2
|
|
238
|
+
const t7: u64 = y13 & y16
|
|
239
|
+
const t8: u64 = y5 & y1
|
|
240
|
+
const t9: u64 = t8 ^ t7
|
|
241
|
+
const t10: u64 = y2 & y7
|
|
242
|
+
const t11: u64 = t10 ^ t7
|
|
243
|
+
const t12: u64 = y9 & y11
|
|
244
|
+
const t13: u64 = y14 & y17
|
|
245
|
+
const t14: u64 = t13 ^ t12
|
|
246
|
+
const t15: u64 = y8 & y10
|
|
247
|
+
const t16: u64 = t15 ^ t12
|
|
248
|
+
const t17: u64 = t4 ^ t14
|
|
249
|
+
const t18: u64 = t6 ^ t16
|
|
250
|
+
const t19: u64 = t9 ^ t14
|
|
251
|
+
const t20: u64 = t11 ^ t16
|
|
252
|
+
const t21: u64 = t17 ^ y20
|
|
253
|
+
const t22: u64 = t18 ^ y19
|
|
254
|
+
const t23: u64 = t19 ^ y21
|
|
255
|
+
const t24: u64 = t20 ^ y18
|
|
256
|
+
|
|
257
|
+
const t25: u64 = t21 ^ t22
|
|
258
|
+
const t26: u64 = t21 & t23
|
|
259
|
+
const t27: u64 = t24 ^ t26
|
|
260
|
+
const t28: u64 = t25 & t27
|
|
261
|
+
const t29: u64 = t28 ^ t22
|
|
262
|
+
const t30: u64 = t23 ^ t24
|
|
263
|
+
const t31: u64 = t22 ^ t26
|
|
264
|
+
const t32: u64 = t31 & t30
|
|
265
|
+
const t33: u64 = t32 ^ t24
|
|
266
|
+
const t34: u64 = t23 ^ t33
|
|
267
|
+
const t35: u64 = t27 ^ t33
|
|
268
|
+
const t36: u64 = t24 & t35
|
|
269
|
+
const t37: u64 = t36 ^ t34
|
|
270
|
+
const t38: u64 = t27 ^ t36
|
|
271
|
+
const t39: u64 = t29 & t38
|
|
272
|
+
const t40: u64 = t25 ^ t39
|
|
273
|
+
|
|
274
|
+
const t41: u64 = t40 ^ t37
|
|
275
|
+
const t42: u64 = t29 ^ t33
|
|
276
|
+
const t43: u64 = t29 ^ t40
|
|
277
|
+
const t44: u64 = t33 ^ t37
|
|
278
|
+
const t45: u64 = t42 ^ t41
|
|
279
|
+
const z0: u64 = t44 & y15
|
|
280
|
+
const z1: u64 = t37 & y6
|
|
281
|
+
const z2: u64 = t33 & x7
|
|
282
|
+
const z3: u64 = t43 & y16
|
|
283
|
+
const z4: u64 = t40 & y1
|
|
284
|
+
const z5: u64 = t29 & y7
|
|
285
|
+
const z6: u64 = t42 & y11
|
|
286
|
+
const z7: u64 = t45 & y17
|
|
287
|
+
const z8: u64 = t41 & y10
|
|
288
|
+
const z9: u64 = t44 & y12
|
|
289
|
+
const z10: u64 = t37 & y3
|
|
290
|
+
const z11: u64 = t33 & y4
|
|
291
|
+
const z12: u64 = t43 & y13
|
|
292
|
+
const z13: u64 = t40 & y5
|
|
293
|
+
const z14: u64 = t29 & y2
|
|
294
|
+
const z15: u64 = t42 & y9
|
|
295
|
+
const z16: u64 = t45 & y14
|
|
296
|
+
const z17: u64 = t41 & y8
|
|
297
|
+
|
|
298
|
+
// The bottom linear layer.
|
|
299
|
+
const t46: u64 = z15 ^ z16
|
|
300
|
+
const t47: u64 = z10 ^ z11
|
|
301
|
+
const t48: u64 = z5 ^ z13
|
|
302
|
+
const t49: u64 = z9 ^ z10
|
|
303
|
+
const t50: u64 = z2 ^ z12
|
|
304
|
+
const t51: u64 = z2 ^ z5
|
|
305
|
+
const t52: u64 = z7 ^ z8
|
|
306
|
+
const t53: u64 = z0 ^ z3
|
|
307
|
+
const t54: u64 = z6 ^ z7
|
|
308
|
+
const t55: u64 = z16 ^ z17
|
|
309
|
+
const t56: u64 = z12 ^ t48
|
|
310
|
+
const t57: u64 = t50 ^ t53
|
|
311
|
+
const t58: u64 = z4 ^ t46
|
|
312
|
+
const t59: u64 = z3 ^ t54
|
|
313
|
+
const t60: u64 = t46 ^ t57
|
|
314
|
+
const t61: u64 = z14 ^ t57
|
|
315
|
+
const t62: u64 = t52 ^ t58
|
|
316
|
+
const t63: u64 = t49 ^ t58
|
|
317
|
+
const t64: u64 = z4 ^ t59
|
|
318
|
+
const t65: u64 = t61 ^ t62
|
|
319
|
+
const t66: u64 = z1 ^ t63
|
|
320
|
+
const s0: u64 = t59 ^ t63
|
|
321
|
+
const s6: u64 = t56 ^ ~t62
|
|
322
|
+
const s7: u64 = t48 ^ ~t60
|
|
323
|
+
const t67: u64 = t64 ^ t65
|
|
324
|
+
const s3: u64 = t53 ^ t66
|
|
325
|
+
const s4: u64 = t51 ^ t66
|
|
326
|
+
const s5: u64 = t47 ^ t65
|
|
327
|
+
const s1: u64 = t64 ^ ~s3
|
|
328
|
+
const s2: u64 = t55 ^ ~t67
|
|
329
|
+
|
|
330
|
+
q[7] = s0
|
|
331
|
+
q[6] = s1
|
|
332
|
+
q[5] = s2
|
|
333
|
+
q[4] = s3
|
|
334
|
+
q[3] = s4
|
|
335
|
+
q[2] = s5
|
|
336
|
+
q[1] = s6
|
|
337
|
+
q[0] = s7
|
|
338
|
+
}
|
|
339
|
+
|
|
340
|
+
/**
|
|
341
|
+
* ShiftRows on one bit plane: row `r` is nibble `r` of each 16-bit lane, and
|
|
342
|
+
* its column `c` takes column `c + r` (mod 4), so the nibble turns right by `r`.
|
|
343
|
+
*/
|
|
344
|
+
const aesShiftPlane = (x: u64): u64 =>
|
|
345
|
+
(x & aesLanes(toU64(0x000f))) |
|
|
346
|
+
((x >> toU64(1)) & aesLanes(toU64(0x0070))) |
|
|
347
|
+
((x << toU64(3)) & aesLanes(toU64(0x0080))) |
|
|
348
|
+
((x >> toU64(2)) & aesLanes(toU64(0x0300))) |
|
|
349
|
+
((x << toU64(2)) & aesLanes(toU64(0x0c00))) |
|
|
350
|
+
((x >> toU64(3)) & aesLanes(toU64(0x1000))) |
|
|
351
|
+
((x << toU64(1)) & aesLanes(toU64(0xe000)))
|
|
352
|
+
|
|
353
|
+
/** ShiftRows (FIPS 197 §5.1.2) on the whole state. */
|
|
354
|
+
const aesShiftRows = (q: u64[]): void => {
|
|
355
|
+
q[0] = aesShiftPlane(q[0])
|
|
356
|
+
q[1] = aesShiftPlane(q[1])
|
|
357
|
+
q[2] = aesShiftPlane(q[2])
|
|
358
|
+
q[3] = aesShiftPlane(q[3])
|
|
359
|
+
q[4] = aesShiftPlane(q[4])
|
|
360
|
+
q[5] = aesShiftPlane(q[5])
|
|
361
|
+
q[6] = aesShiftPlane(q[6])
|
|
362
|
+
q[7] = aesShiftPlane(q[7])
|
|
363
|
+
}
|
|
364
|
+
|
|
365
|
+
/** Each byte replaced by the byte one row down its column (row 3 by row 0): nibbles turned by one. */
|
|
366
|
+
const aesNextRow = (x: u64): u64 =>
|
|
367
|
+
((x >> toU64(4)) & aesLanes(toU64(0x0fff))) | ((x << toU64(12)) & aesLanes(toU64(0xf000)))
|
|
368
|
+
|
|
369
|
+
/** Each byte replaced by the byte two rows down its column. */
|
|
370
|
+
const aesRowAfterNext = (x: u64): u64 =>
|
|
371
|
+
((x >> toU64(8)) & aesLanes(toU64(0x00ff))) | ((x << toU64(8)) & aesLanes(toU64(0xff00)))
|
|
372
|
+
|
|
373
|
+
/**
|
|
374
|
+
* MixColumns (FIPS 197 §5.1.3): s'_r = 2·s_r ⊕ 3·s_{r+1} ⊕ s_{r+2} ⊕ s_{r+3},
|
|
375
|
+
* rewritten as 2·t ⊕ s_{r+1} ⊕ (t two rows down) with t = s_r ⊕ s_{r+1}, so
|
|
376
|
+
* the one multiplication by 2 is a shift of the planes: plane `b` of 2·t is
|
|
377
|
+
* plane `b - 1` of t, and t's top plane folds back into planes 0, 1, 3 and 4
|
|
378
|
+
* for the reduction by x^8 + x^4 + x^3 + x + 1.
|
|
379
|
+
*/
|
|
380
|
+
const aesMixColumns = (q: u64[]): void => {
|
|
381
|
+
const n0: u64 = aesNextRow(q[0])
|
|
382
|
+
const n1: u64 = aesNextRow(q[1])
|
|
383
|
+
const n2: u64 = aesNextRow(q[2])
|
|
384
|
+
const n3: u64 = aesNextRow(q[3])
|
|
385
|
+
const n4: u64 = aesNextRow(q[4])
|
|
386
|
+
const n5: u64 = aesNextRow(q[5])
|
|
387
|
+
const n6: u64 = aesNextRow(q[6])
|
|
388
|
+
const n7: u64 = aesNextRow(q[7])
|
|
389
|
+
const t0: u64 = q[0] ^ n0
|
|
390
|
+
const t1: u64 = q[1] ^ n1
|
|
391
|
+
const t2: u64 = q[2] ^ n2
|
|
392
|
+
const t3: u64 = q[3] ^ n3
|
|
393
|
+
const t4: u64 = q[4] ^ n4
|
|
394
|
+
const t5: u64 = q[5] ^ n5
|
|
395
|
+
const t6: u64 = q[6] ^ n6
|
|
396
|
+
const t7: u64 = q[7] ^ n7
|
|
397
|
+
q[0] = t7 ^ n0 ^ aesRowAfterNext(t0)
|
|
398
|
+
q[1] = t0 ^ t7 ^ n1 ^ aesRowAfterNext(t1)
|
|
399
|
+
q[2] = t1 ^ n2 ^ aesRowAfterNext(t2)
|
|
400
|
+
q[3] = t2 ^ t7 ^ n3 ^ aesRowAfterNext(t3)
|
|
401
|
+
q[4] = t3 ^ t7 ^ n4 ^ aesRowAfterNext(t4)
|
|
402
|
+
q[5] = t4 ^ n5 ^ aesRowAfterNext(t5)
|
|
403
|
+
q[6] = t5 ^ n6 ^ aesRowAfterNext(t6)
|
|
404
|
+
q[7] = t6 ^ n7 ^ aesRowAfterNext(t7)
|
|
405
|
+
}
|
|
406
|
+
|
|
407
|
+
/** AddRoundKey (FIPS 197 §5.1.4): the eight words of `rk` from `at` into the state. */
|
|
408
|
+
const aesAddRoundKey = (q: u64[], rk: u64[], at: i32): void => {
|
|
409
|
+
q[0] = q[0] ^ rk[at]
|
|
410
|
+
q[1] = q[1] ^ rk[at + 1]
|
|
411
|
+
q[2] = q[2] ^ rk[at + 2]
|
|
412
|
+
q[3] = q[3] ^ rk[at + 3]
|
|
413
|
+
q[4] = q[4] ^ rk[at + 4]
|
|
414
|
+
q[5] = q[5] ^ rk[at + 5]
|
|
415
|
+
q[6] = q[6] ^ rk[at + 6]
|
|
416
|
+
q[7] = q[7] ^ rk[at + 7]
|
|
417
|
+
}
|
|
418
|
+
|
|
419
|
+
/**
|
|
420
|
+
* One full middle round of the cipher on bitsliced state — SubBytes,
|
|
421
|
+
* ShiftRows, MixColumns, then the round key of eight words from `at` — for
|
|
422
|
+
* four blocks at once. Straight-line code with no branch and no index but
|
|
423
|
+
* `at`, which is a round number times eight. Exported so that
|
|
424
|
+
* `tests/link/crypto_aes` can hold the copy the disassembly check reads to
|
|
425
|
+
* this one; a caller has no other use for it.
|
|
426
|
+
*/
|
|
427
|
+
export const aesBitslicedRound = (q: u64[], rk: u64[], at: i32): void => {
|
|
428
|
+
aesSbox(q)
|
|
429
|
+
aesShiftRows(q)
|
|
430
|
+
aesMixColumns(q)
|
|
431
|
+
aesAddRoundKey(q, rk, at)
|
|
432
|
+
}
|
|
433
|
+
|
|
434
|
+
/** The cipher (FIPS 197 §5.1) on the four blocks of `buf[0 .. 64)`, in place. `q` is scratch. */
|
|
435
|
+
const aesEncryptBatch = (key: AesKey, buf: u8[], q: u64[]): void => {
|
|
436
|
+
const rk: u64[] = key.roundKeys
|
|
437
|
+
aesPack(q, buf)
|
|
438
|
+
aesAddRoundKey(q, rk, 0)
|
|
439
|
+
for (let round: i32 = 1; round < key.rounds; round++) {
|
|
440
|
+
aesBitslicedRound(q, rk, 8 * round)
|
|
441
|
+
}
|
|
442
|
+
aesSbox(q)
|
|
443
|
+
aesShiftRows(q)
|
|
444
|
+
aesAddRoundKey(q, rk, 8 * key.rounds)
|
|
445
|
+
aesUnpack(buf, q)
|
|
446
|
+
}
|
|
447
|
+
|
|
448
|
+
/**
|
|
449
|
+
* SubWord (FIPS 197 §5.2) through the same circuit: the four bytes of `w`
|
|
450
|
+
* become bits 0 to 3 of eight planes, the circuit runs, and the bits come
|
|
451
|
+
* back. The unused bits of the planes are zero going in and are ignored
|
|
452
|
+
* coming out. No table, so the key schedule is as constant-time as a round.
|
|
453
|
+
*/
|
|
454
|
+
const aesSubWord = (w: u32, q: u64[]): u32 => {
|
|
455
|
+
if (toI32(q.length) < 8) {
|
|
456
|
+
return 0
|
|
457
|
+
}
|
|
458
|
+
for (let b: i32 = 0; b < 8; b++) {
|
|
459
|
+
let plane: u64 = 0
|
|
460
|
+
for (let m: i32 = 0; m < 4; m++) {
|
|
461
|
+
plane = plane | (toU64((w >> toU32(8 * m + b)) & toU32(1)) << toU64(m))
|
|
462
|
+
}
|
|
463
|
+
q[b] = plane
|
|
464
|
+
}
|
|
465
|
+
aesSbox(q)
|
|
466
|
+
let out: u32 = 0
|
|
467
|
+
for (let b: i32 = 0; b < 8; b++) {
|
|
468
|
+
for (let m: i32 = 0; m < 4; m++) {
|
|
469
|
+
out = out | (toU32((q[b] >> toU64(m)) & toU64(1)) << toU32(8 * m + b))
|
|
470
|
+
}
|
|
471
|
+
}
|
|
472
|
+
return out
|
|
473
|
+
}
|
|
474
|
+
|
|
475
|
+
/** Four bytes of `data` from `at` as a big-endian word; the caller has checked the window. */
|
|
476
|
+
const aesLoad32 = (data: u8[], at: i32): u32 =>
|
|
477
|
+
(toU32(data[at]) << toU32(24)) |
|
|
478
|
+
(toU32(data[at + 1]) << toU32(16)) |
|
|
479
|
+
(toU32(data[at + 2]) << toU32(8)) |
|
|
480
|
+
toU32(data[at + 3])
|
|
481
|
+
|
|
482
|
+
/** `w` big-endian into `out[at .. at + 4)`; the caller has checked the window. */
|
|
483
|
+
const aesStore32 = (out: u8[], at: i32, w: u32): void => {
|
|
484
|
+
out[at] = toU8(w >> toU32(24))
|
|
485
|
+
out[at + 1] = toU8(w >> toU32(16))
|
|
486
|
+
out[at + 2] = toU8(w >> toU32(8))
|
|
487
|
+
out[at + 3] = toU8(w)
|
|
488
|
+
}
|
|
489
|
+
|
|
490
|
+
/**
|
|
491
|
+
* KeyExpansion (FIPS 197 §5.2) for Nk = 4 or 8, then every round key
|
|
492
|
+
* bitsliced and replicated into the four block lanes. `w` holds the words
|
|
493
|
+
* big-endian, byte 0 of the key in the high byte of `w[0]`, as the standard
|
|
494
|
+
* writes them. Branches only on the word number.
|
|
495
|
+
*/
|
|
496
|
+
const aesExpand = (key: u8[], rounds: i32): u64[] => {
|
|
497
|
+
const nk: i32 = rounds - 6
|
|
498
|
+
const total: i32 = (rounds + 1) * 4
|
|
499
|
+
const w: u32[] = new Array<u32>(total)
|
|
500
|
+
const q: u64[] = new Array<u64>(8)
|
|
501
|
+
const keyLength: i32 = toI32(key.length)
|
|
502
|
+
const wLength: i32 = toI32(w.length)
|
|
503
|
+
// Rcon[i/Nk] = x^(i/Nk - 1) in GF(2^8), doubled once per Nk words.
|
|
504
|
+
let rcon: u32 = 1
|
|
505
|
+
for (let i: i32 = 0; i < wLength; i++) {
|
|
506
|
+
if (i < nk) {
|
|
507
|
+
const at: i32 = 4 * i
|
|
508
|
+
if (at >= 0 && at + 3 < keyLength) {
|
|
509
|
+
w[i] = aesLoad32(key, at)
|
|
510
|
+
}
|
|
511
|
+
} else {
|
|
512
|
+
let temp: u32 = w[i - 1]
|
|
513
|
+
if (i % nk === 0) {
|
|
514
|
+
temp = aesSubWord((temp << toU32(8)) | (temp >> toU32(24)), q) ^ (rcon << toU32(24))
|
|
515
|
+
rcon = ((rcon << toU32(1)) ^ (toU32(0x1b) & (toU32(0) - (rcon >> toU32(7))))) & toU32(0xff)
|
|
516
|
+
} else if (nk > 6 && i % nk === 4) {
|
|
517
|
+
temp = aesSubWord(temp, q)
|
|
518
|
+
}
|
|
519
|
+
w[i] = w[i - nk] ^ temp
|
|
520
|
+
}
|
|
521
|
+
}
|
|
522
|
+
// Each round key, written into all four blocks of a batch and packed.
|
|
523
|
+
const rk: u64[] = new Array<u64>((rounds + 1) * 8)
|
|
524
|
+
const buf: u8[] = new Array<u8>(AES_BATCH)
|
|
525
|
+
for (let round: i32 = 0; round <= rounds; round++) {
|
|
526
|
+
for (let k: i32 = 0; k < 4; k++) {
|
|
527
|
+
for (let j: i32 = 0; j < 4; j++) {
|
|
528
|
+
aesStore32(buf, 16 * k + 4 * j, w[4 * round + j])
|
|
529
|
+
}
|
|
530
|
+
}
|
|
531
|
+
aesPack(q, buf)
|
|
532
|
+
for (let b: i32 = 0; b < 8; b++) {
|
|
533
|
+
rk[8 * round + b] = q[b]
|
|
534
|
+
}
|
|
535
|
+
}
|
|
536
|
+
return rk
|
|
537
|
+
}
|
|
538
|
+
|
|
539
|
+
/**
|
|
540
|
+
* Eight bytes of `data` from `at`, big-endian, reading only below `end` and
|
|
541
|
+
* zero past it: how GHASH pads a partial last block (SP 800-38D §6.4).
|
|
542
|
+
*/
|
|
543
|
+
const aesLoad64 = (data: u8[], at: i32, end: i32): u64 => {
|
|
544
|
+
const length: i32 = toI32(data.length)
|
|
545
|
+
let v: u64 = 0
|
|
546
|
+
for (let i: i32 = 0; i < 8; i++) {
|
|
547
|
+
const from: i32 = at + i
|
|
548
|
+
let byte: u64 = 0
|
|
549
|
+
if (from >= 0 && from < end && from < length) {
|
|
550
|
+
byte = toU64(data[from])
|
|
551
|
+
}
|
|
552
|
+
v = v | (byte << toU64((7 - i) * 8))
|
|
553
|
+
}
|
|
554
|
+
return v
|
|
555
|
+
}
|
|
556
|
+
|
|
557
|
+
/** `v` big-endian into `out[at .. at + 8)`; a byte outside `out` is dropped. */
|
|
558
|
+
const aesStore64 = (out: u8[], at: i32, v: u64): void => {
|
|
559
|
+
const length: i32 = toI32(out.length)
|
|
560
|
+
for (let i: i32 = 0; i < 8; i++) {
|
|
561
|
+
const to: i32 = at + i
|
|
562
|
+
if (to >= 0 && to < length) {
|
|
563
|
+
out[to] = toU8(v >> toU64((7 - i) * 8))
|
|
564
|
+
}
|
|
565
|
+
}
|
|
566
|
+
}
|
|
567
|
+
|
|
568
|
+
/** `dst[i] = src[i]` for `i` below `count` and inside both arrays. */
|
|
569
|
+
const aesCopyPrefix = (dst: u8[], src: u8[], count: i32): void => {
|
|
570
|
+
const dstLength: i32 = toI32(dst.length)
|
|
571
|
+
const srcLength: i32 = toI32(src.length)
|
|
572
|
+
for (let i: i32 = 0; i < count && i < dstLength && i < srcLength; i++) {
|
|
573
|
+
dst[i] = src[i]
|
|
574
|
+
}
|
|
575
|
+
}
|
|
576
|
+
|
|
577
|
+
/**
|
|
578
|
+
* The first 16 bytes of `block` encrypted, in the first 16 bytes of a fresh
|
|
579
|
+
* batch; the rest of the batch is the cipher of zeros and is not read.
|
|
580
|
+
*/
|
|
581
|
+
const aesEncryptOne = (key: AesKey, block: u8[]): u8[] => {
|
|
582
|
+
const buf: u8[] = new Array<u8>(AES_BATCH)
|
|
583
|
+
aesCopyPrefix(buf, block, AES_BLOCK)
|
|
584
|
+
aesEncryptBatch(key, buf, new Array<u64>(8))
|
|
585
|
+
return buf
|
|
586
|
+
}
|
|
587
|
+
|
|
588
|
+
/**
|
|
589
|
+
* Makes an `AesKey` from 16 bytes (AES-128) or 32 (AES-256): the key
|
|
590
|
+
* schedule, bitsliced, and H = E(K, 0^128) for GCM. Any other length — 24
|
|
591
|
+
* bytes included, since AES-192 is not offered — answers `null`.
|
|
592
|
+
*/
|
|
593
|
+
export const aesKey = (key: u8[]): AesKey | null => {
|
|
594
|
+
const length: i32 = toI32(key.length)
|
|
595
|
+
if (length !== 16 && length !== 32) {
|
|
596
|
+
return null
|
|
597
|
+
}
|
|
598
|
+
const rounds: i32 = length === 16 ? 10 : 14
|
|
599
|
+
const expanded: AesKey = new AesKey(rounds, aesExpand(key, rounds))
|
|
600
|
+
const buf: u8[] = aesEncryptOne(expanded, [])
|
|
601
|
+
expanded.hHi = aesLoad64(buf, 0, 8)
|
|
602
|
+
expanded.hLo = aesLoad64(buf, 8, 16)
|
|
603
|
+
return expanded
|
|
604
|
+
}
|
|
605
|
+
|
|
606
|
+
/** One block encrypted (FIPS 197 §5.1, ECB): 16 bytes in, 16 fresh bytes out, or `null` for another length. */
|
|
607
|
+
export const aesEncryptBlock = (key: AesKey, block: u8[]): u8[] | null => {
|
|
608
|
+
if (toI32(block.length) !== AES_BLOCK) {
|
|
609
|
+
return null
|
|
610
|
+
}
|
|
611
|
+
const out: u8[] = new Array<u8>(AES_BLOCK)
|
|
612
|
+
aesCopyPrefix(out, aesEncryptOne(key, block), AES_BLOCK)
|
|
613
|
+
return out
|
|
614
|
+
}
|
|
615
|
+
|
|
616
|
+
/**
|
|
617
|
+
* The AES header-protection mask of RFC 9001 §5.4.3: the first five bytes of
|
|
618
|
+
* the 16-byte `sample` encrypted under the header-protection key. `null` for
|
|
619
|
+
* a sample of another length.
|
|
620
|
+
*/
|
|
621
|
+
export const aesHeaderMask = (key: AesKey, sample: u8[]): u8[] | null => {
|
|
622
|
+
if (toI32(sample.length) !== AES_BLOCK) {
|
|
623
|
+
return null
|
|
624
|
+
}
|
|
625
|
+
const mask: u8[] = new Array<u8>(5)
|
|
626
|
+
aesCopyPrefix(mask, aesEncryptOne(key, sample), 5)
|
|
627
|
+
return mask
|
|
628
|
+
}
|
|
629
|
+
|
|
630
|
+
/*
|
|
631
|
+
* Adapted from BearSSL (https://www.bearssl.org/), src/hash/ghash_ctmul64.c
|
|
632
|
+
* bmul64(). Copyright (c) 2016 Thomas Pornin. Used under the MIT licence;
|
|
633
|
+
* see std/crypto/LICENSE-bearssl.
|
|
634
|
+
*/
|
|
635
|
+
/**
|
|
636
|
+
* The carry-less product of two 32-bit polynomials, as a 64-bit one. Each
|
|
637
|
+
* operand is split into four with one bit in every four kept, so that an
|
|
638
|
+
* integer product of two parts has at most eight terms at any bit — below 16,
|
|
639
|
+
* so its carries stay inside the three zero bits above it — and the bits of
|
|
640
|
+
* the true product are the low bit of each four, read back through the same
|
|
641
|
+
* masks. Sixteen integer multiplications and no branch.
|
|
642
|
+
*/
|
|
643
|
+
const ghashMul32 = (x: u64, y: u64): u64 => {
|
|
644
|
+
const m0: u64 = toU64(0x11111111)
|
|
645
|
+
const m1: u64 = toU64(0x22222222)
|
|
646
|
+
const m2: u64 = toU64(0x44444444)
|
|
647
|
+
const m3: u64 = m2 << toU64(1)
|
|
648
|
+
const x0: u64 = x & m0
|
|
649
|
+
const x1: u64 = x & m1
|
|
650
|
+
const x2: u64 = x & m2
|
|
651
|
+
const x3: u64 = x & m3
|
|
652
|
+
const y0: u64 = y & m0
|
|
653
|
+
const y1: u64 = y & m1
|
|
654
|
+
const y2: u64 = y & m2
|
|
655
|
+
const y3: u64 = y & m3
|
|
656
|
+
const z0: u64 = (x0 * y0) ^ (x1 * y3) ^ (x2 * y2) ^ (x3 * y1)
|
|
657
|
+
const z1: u64 = (x0 * y1) ^ (x1 * y0) ^ (x2 * y3) ^ (x3 * y2)
|
|
658
|
+
const z2: u64 = (x0 * y2) ^ (x1 * y1) ^ (x2 * y0) ^ (x3 * y3)
|
|
659
|
+
const z3: u64 = (x0 * y3) ^ (x1 * y2) ^ (x2 * y1) ^ (x3 * y0)
|
|
660
|
+
const w0: u64 = m0 | (m0 << toU64(32))
|
|
661
|
+
return (z0 & w0) | (z1 & (w0 << toU64(1))) | (z2 & (w0 << toU64(2))) | (z3 & (w0 << toU64(3)))
|
|
662
|
+
}
|
|
663
|
+
|
|
664
|
+
/**
|
|
665
|
+
* GHASH's multiply (SP 800-38D §6.3), `y = y · H` in GF(2^128), with `y` as
|
|
666
|
+
* two words `y[0]` (high) and `y[1]` (low) of the block read big-endian and
|
|
667
|
+
* H likewise in `hHi` and `hLo`.
|
|
668
|
+
*
|
|
669
|
+
* GCM numbers a block's bits from the first, so the block read big-endian is
|
|
670
|
+
* the polynomial with its coefficients reversed. The carry-less product of
|
|
671
|
+
* two reversed polynomials is the reversed product one bit short, so the
|
|
672
|
+
* 256-bit product is shifted left once, and then its low half — the
|
|
673
|
+
* coefficients of x^128 and up — is folded into the high half by
|
|
674
|
+
* x^128 = x^7 + x^2 + x + 1, in two steps because the fold of the lowest
|
|
675
|
+
* seven bits lands back in the low half (Gueron and Kounavis, §4 of the
|
|
676
|
+
* paper cited in the header).
|
|
677
|
+
*
|
|
678
|
+
* The 128 × 128 product is Karatsuba twice over: three 64 × 64 products, each
|
|
679
|
+
* of three 32 × 32 ones from `ghashMul32`. Straight-line code; exported, like
|
|
680
|
+
* `aesBitslicedRound`, so the disassembly check's copy can be held to it.
|
|
681
|
+
*/
|
|
682
|
+
export const ghashMultiply = (y: u64[], hHi: u64, hLo: u64): void => {
|
|
683
|
+
const low32: u64 = (toU64(1) << toU64(32)) - toU64(1)
|
|
684
|
+
const thirtyTwo: u64 = toU64(32)
|
|
685
|
+
const a1: u64 = y[0]
|
|
686
|
+
const a0: u64 = y[1]
|
|
687
|
+
const a2: u64 = a1 ^ a0
|
|
688
|
+
const b2: u64 = hHi ^ hLo
|
|
689
|
+
|
|
690
|
+
// A0 · B0, as `lHi:lLo`.
|
|
691
|
+
const l0: u64 = ghashMul32(a0 & low32, hLo & low32)
|
|
692
|
+
const l1: u64 = ghashMul32(a0 >> thirtyTwo, hLo >> thirtyTwo)
|
|
693
|
+
const l2: u64 = ghashMul32((a0 ^ (a0 >> thirtyTwo)) & low32, (hLo ^ (hLo >> thirtyTwo)) & low32) ^ l0 ^ l1
|
|
694
|
+
const lHi: u64 = l1 ^ (l2 >> thirtyTwo)
|
|
695
|
+
const lLo: u64 = l0 ^ (l2 << thirtyTwo)
|
|
696
|
+
|
|
697
|
+
// A1 · B1, as `hHi2:hLo2`.
|
|
698
|
+
const h0: u64 = ghashMul32(a1 & low32, hHi & low32)
|
|
699
|
+
const h1: u64 = ghashMul32(a1 >> thirtyTwo, hHi >> thirtyTwo)
|
|
700
|
+
const h2: u64 = ghashMul32((a1 ^ (a1 >> thirtyTwo)) & low32, (hHi ^ (hHi >> thirtyTwo)) & low32) ^ h0 ^ h1
|
|
701
|
+
const hiHi: u64 = h1 ^ (h2 >> thirtyTwo)
|
|
702
|
+
const hiLo: u64 = h0 ^ (h2 << thirtyTwo)
|
|
703
|
+
|
|
704
|
+
// (A0 ⊕ A1) · (B0 ⊕ B1), less the other two, as `mHi:mLo`.
|
|
705
|
+
const m0: u64 = ghashMul32(a2 & low32, b2 & low32)
|
|
706
|
+
const m1: u64 = ghashMul32(a2 >> thirtyTwo, b2 >> thirtyTwo)
|
|
707
|
+
const m2: u64 = ghashMul32((a2 ^ (a2 >> thirtyTwo)) & low32, (b2 ^ (b2 >> thirtyTwo)) & low32) ^ m0 ^ m1
|
|
708
|
+
const mHi: u64 = m1 ^ (m2 >> thirtyTwo) ^ lHi ^ hiHi
|
|
709
|
+
const mLo: u64 = m0 ^ (m2 << thirtyTwo) ^ lLo ^ hiLo
|
|
710
|
+
|
|
711
|
+
// The 256-bit product p3:p2:p1:p0, shifted left once.
|
|
712
|
+
const r3: u64 = hiHi
|
|
713
|
+
const r2: u64 = hiLo ^ mHi
|
|
714
|
+
const r1: u64 = lHi ^ mLo
|
|
715
|
+
const r0: u64 = lLo
|
|
716
|
+
const one: u64 = toU64(1)
|
|
717
|
+
const top: u64 = toU64(63)
|
|
718
|
+
const p3: u64 = (r3 << one) | (r2 >> top)
|
|
719
|
+
const p2: u64 = (r2 << one) | (r1 >> top)
|
|
720
|
+
const p1: u64 = (r1 << one) | (r0 >> top)
|
|
721
|
+
const p0: u64 = r0 << one
|
|
722
|
+
|
|
723
|
+
// Reduce: fold the lowest bits of p1:p0 into p1 first, then all of it up.
|
|
724
|
+
const d1: u64 = p1 ^ (p0 << toU64(63)) ^ (p0 << toU64(62)) ^ (p0 << toU64(57))
|
|
725
|
+
y[0] = p3 ^ d1 ^ (d1 >> one) ^ (d1 >> toU64(2)) ^ (d1 >> toU64(7))
|
|
726
|
+
y[1] =
|
|
727
|
+
p2 ^
|
|
728
|
+
p0 ^
|
|
729
|
+
(p0 >> one) ^
|
|
730
|
+
(d1 << top) ^
|
|
731
|
+
(p0 >> toU64(2)) ^
|
|
732
|
+
(d1 << toU64(62)) ^
|
|
733
|
+
(p0 >> toU64(7)) ^
|
|
734
|
+
(d1 << toU64(57))
|
|
735
|
+
}
|
|
736
|
+
|
|
737
|
+
/**
|
|
738
|
+
* Absorbs `data[0 .. length)` into the GHASH state `y`, 16 bytes at a time,
|
|
739
|
+
* the last block padded with zeros (SP 800-38D §6.4, §7.1 step 5).
|
|
740
|
+
*/
|
|
741
|
+
const ghashUpdate = (y: u64[], key: AesKey, data: u8[], length: i32): void => {
|
|
742
|
+
if (toI32(y.length) < 2) {
|
|
743
|
+
return
|
|
744
|
+
}
|
|
745
|
+
// Counted in blocks rather than stepped by 16, so no offset is ever formed
|
|
746
|
+
// past `length`: an AAD or IV within 16 bytes of 2^31 would overflow `at + 16`.
|
|
747
|
+
const blocks: i32 = (length >> 4) + ((length & 15) !== 0 ? 1 : 0)
|
|
748
|
+
for (let block: i32 = 0; block < blocks; block++) {
|
|
749
|
+
const at: i32 = block * 16
|
|
750
|
+
y[0] = y[0] ^ aesLoad64(data, at, length)
|
|
751
|
+
y[1] = y[1] ^ aesLoad64(data, at + 8, length)
|
|
752
|
+
ghashMultiply(y, key.hHi, key.hLo)
|
|
753
|
+
}
|
|
754
|
+
}
|
|
755
|
+
|
|
756
|
+
/** The last GHASH block: the AAD's and the ciphertext's lengths in bits, 64 bits each. */
|
|
757
|
+
const ghashLengths = (y: u64[], key: AesKey, aadLength: i32, textLength: i32): void => {
|
|
758
|
+
y[0] = y[0] ^ (toU64(aadLength) << toU64(3))
|
|
759
|
+
y[1] = y[1] ^ (toU64(textLength) << toU64(3))
|
|
760
|
+
ghashMultiply(y, key.hHi, key.hLo)
|
|
761
|
+
}
|
|
762
|
+
|
|
763
|
+
/**
|
|
764
|
+
* The pre-counter block J0 (SP 800-38D §7.1 step 2): a 12-byte IV followed by
|
|
765
|
+
* the 32-bit counter 1, or for any other length GHASH of the IV padded to a
|
|
766
|
+
* block and followed by its length in bits.
|
|
767
|
+
*/
|
|
768
|
+
const aesGcmJ0 = (key: AesKey, iv: u8[]): u8[] => {
|
|
769
|
+
const length: i32 = toI32(iv.length)
|
|
770
|
+
const j0: u8[] = new Array<u8>(AES_BLOCK)
|
|
771
|
+
if (length === 12) {
|
|
772
|
+
aesCopyPrefix(j0, iv, 12)
|
|
773
|
+
j0[15] = 1
|
|
774
|
+
return j0
|
|
775
|
+
}
|
|
776
|
+
const y: u64[] = new Array<u64>(2)
|
|
777
|
+
ghashUpdate(y, key, iv, length)
|
|
778
|
+
ghashLengths(y, key, 0, length)
|
|
779
|
+
aesStore64(j0, 0, y[0])
|
|
780
|
+
aesStore64(j0, 8, y[1])
|
|
781
|
+
return j0
|
|
782
|
+
}
|
|
783
|
+
|
|
784
|
+
/**
|
|
785
|
+
* GCTR (SP 800-38D §6.5) from inc32(J0): `dst[i] = src[i] ⊕ keystream[i]` for
|
|
786
|
+
* `i` below `length`, four blocks per pass of the cipher. The counter is the
|
|
787
|
+
* last four bytes of J0 as a big-endian `u32` and wraps modulo 2^32, as
|
|
788
|
+
* inc32 does, while the first twelve bytes stay as they are.
|
|
789
|
+
*/
|
|
790
|
+
const aesGcmCtr = (key: AesKey, j0: u8[], src: u8[], dst: u8[], length: i32): void => {
|
|
791
|
+
if (toI32(j0.length) < 16) {
|
|
792
|
+
return
|
|
793
|
+
}
|
|
794
|
+
const counter: u32 = aesLoad32(j0, 12)
|
|
795
|
+
const buf: u8[] = new Array<u8>(AES_BATCH)
|
|
796
|
+
const q: u64[] = new Array<u64>(8)
|
|
797
|
+
const srcLength: i32 = toI32(src.length)
|
|
798
|
+
const dstLength: i32 = toI32(dst.length)
|
|
799
|
+
// Counted in batches, like `ghashUpdate`'s blocks, so `base` never passes `length`.
|
|
800
|
+
const batches: i32 = (length >> 6) + ((length & 63) !== 0 ? 1 : 0)
|
|
801
|
+
for (let batch: i32 = 0; batch < batches; batch++) {
|
|
802
|
+
const base: i32 = batch * AES_BATCH
|
|
803
|
+
for (let k: i32 = 0; k < 4; k++) {
|
|
804
|
+
for (let i: i32 = 0; i < 12; i++) {
|
|
805
|
+
buf[16 * k + i] = j0[i]
|
|
806
|
+
}
|
|
807
|
+
aesStore32(buf, 16 * k + 12, counter + toU32((base >> 4) + k + 1))
|
|
808
|
+
}
|
|
809
|
+
aesEncryptBatch(key, buf, q)
|
|
810
|
+
for (let j: i32 = 0; j < AES_BATCH && j < toI32(buf.length); j++) {
|
|
811
|
+
const at: i32 = base + j
|
|
812
|
+
if (at >= 0 && at < length && at < srcLength && at < dstLength) {
|
|
813
|
+
dst[at] = src[at] ^ buf[j]
|
|
814
|
+
}
|
|
815
|
+
}
|
|
816
|
+
}
|
|
817
|
+
}
|
|
818
|
+
|
|
819
|
+
/**
|
|
820
|
+
* E(K, J0) ⊕ S, GCM's tag (SP 800-38D §7.1 step 6), into `y` (zero, two
|
|
821
|
+
* words) as two big-endian words: S is the GHASH of the AAD and the
|
|
822
|
+
* ciphertext `text[0 .. textLength)`.
|
|
823
|
+
*/
|
|
824
|
+
const aesGcmTag = (y: u64[], key: AesKey, j0: u8[], aad: u8[], text: u8[], textLength: i32): void => {
|
|
825
|
+
const aadLength: i32 = toI32(aad.length)
|
|
826
|
+
ghashUpdate(y, key, aad, aadLength)
|
|
827
|
+
ghashUpdate(y, key, text, textLength)
|
|
828
|
+
ghashLengths(y, key, aadLength, textLength)
|
|
829
|
+
const mask: u8[] = aesEncryptOne(key, j0)
|
|
830
|
+
y[0] = y[0] ^ aesLoad64(mask, 0, 16)
|
|
831
|
+
y[1] = y[1] ^ aesLoad64(mask, 8, 16)
|
|
832
|
+
}
|
|
833
|
+
|
|
834
|
+
/**
|
|
835
|
+
* All-ones when the tag `tagHi:tagLo` equals `gotHi:gotLo`, zero otherwise:
|
|
836
|
+
* the difference of both halves ORed into one word and handed to `ctEq`, so
|
|
837
|
+
* every bit of both tags is read and nothing is branched on until the caller
|
|
838
|
+
* tests the one answer. Exported, like `aesBitslicedRound`, so the
|
|
839
|
+
* disassembly check's copy can be held to it.
|
|
840
|
+
*/
|
|
841
|
+
export const aesGcmTagMask = (tagHi: u64, tagLo: u64, gotHi: u64, gotLo: u64): u64 =>
|
|
842
|
+
ctEq((tagHi ^ gotHi) | (tagLo ^ gotLo), toU64(0))
|
|
843
|
+
|
|
844
|
+
/**
|
|
845
|
+
* GCM authenticated encryption (SP 800-38D §7.1): the ciphertext followed by
|
|
846
|
+
* the 16-byte tag, in one fresh array. The IV may be any non-zero length; 12
|
|
847
|
+
* bytes is the fast and usual one. `null` for an empty IV, and for a
|
|
848
|
+
* plaintext longer than 2^31 - 17 bytes, whose sealed form would not fit in an
|
|
849
|
+
* array.
|
|
850
|
+
*/
|
|
851
|
+
export const aesGcmSeal = (key: AesKey, iv: u8[], aad: u8[], plaintext: u8[]): u8[] | null => {
|
|
852
|
+
const length: i32 = toI32(plaintext.length)
|
|
853
|
+
// 2^31 - 17: the longest plaintext whose sealed form, sixteen bytes longer,
|
|
854
|
+
// is still an array length.
|
|
855
|
+
const longest: i32 = 0x7fffffef
|
|
856
|
+
if (toI32(iv.length) === 0 || length > longest) {
|
|
857
|
+
return null
|
|
858
|
+
}
|
|
859
|
+
const j0: u8[] = aesGcmJ0(key, iv)
|
|
860
|
+
const out: u8[] = new Array<u8>(length + AES_GCM_TAG_SIZE)
|
|
861
|
+
aesGcmCtr(key, j0, plaintext, out, length)
|
|
862
|
+
const tag: u64[] = new Array<u64>(2)
|
|
863
|
+
aesGcmTag(tag, key, j0, aad, out, length)
|
|
864
|
+
aesStore64(out, length, tag[0])
|
|
865
|
+
aesStore64(out, length + 8, tag[1])
|
|
866
|
+
return out
|
|
867
|
+
}
|
|
868
|
+
|
|
869
|
+
/**
|
|
870
|
+
* GCM authenticated decryption (SP 800-38D §7.2): `sealed` is the ciphertext
|
|
871
|
+
* followed by its 16-byte tag, and the answer is the plaintext, or `null` when
|
|
872
|
+
* the tag does not verify, the IV is empty or `sealed` is shorter than a tag.
|
|
873
|
+
* The tag is computed over the received ciphertext and compared in constant
|
|
874
|
+
* time before anything is decrypted, so a forgery yields no plaintext at all.
|
|
875
|
+
*/
|
|
876
|
+
export const aesGcmOpen = (key: AesKey, iv: u8[], aad: u8[], sealed: u8[]): u8[] | null => {
|
|
877
|
+
const total: i32 = toI32(sealed.length)
|
|
878
|
+
if (toI32(iv.length) === 0 || total < AES_GCM_TAG_SIZE) {
|
|
879
|
+
return null
|
|
880
|
+
}
|
|
881
|
+
// `total` is an array length and at least a tag, so neither this nor the
|
|
882
|
+
// tag's second word at `length + 8` can leave the range of an `i32`.
|
|
883
|
+
const length: i32 = total - AES_GCM_TAG_SIZE
|
|
884
|
+
const j0: u8[] = aesGcmJ0(key, iv)
|
|
885
|
+
const tag: u64[] = new Array<u64>(2)
|
|
886
|
+
aesGcmTag(tag, key, j0, aad, sealed, length)
|
|
887
|
+
const same: u64 = aesGcmTagMask(
|
|
888
|
+
tag[0],
|
|
889
|
+
tag[1],
|
|
890
|
+
aesLoad64(sealed, length, total),
|
|
891
|
+
aesLoad64(sealed, length + 8, total)
|
|
892
|
+
)
|
|
893
|
+
if (same === toU64(0)) {
|
|
894
|
+
return null
|
|
895
|
+
}
|
|
896
|
+
const out: u8[] = new Array<u8>(length)
|
|
897
|
+
aesGcmCtr(key, j0, sealed, out, length)
|
|
898
|
+
return out
|
|
899
|
+
}
|