@amritk/nish-aarch64-linux 0.12.0 → 0.14.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,494 @@
1
+ /**
2
+ * `nish/crypto/x25519` — the X25519 Diffie-Hellman function of RFC 7748.
3
+ *
4
+ * Written from the specification (RFC 7748 §4.1, §5 and §6.1), in this
5
+ * module's own structure: nothing here is ported from another implementation.
6
+ * The radix-2^25.5 limbs and the shape of the inversion chain are the ideas
7
+ * D. J. Bernstein's "Curve25519: new Diffie-Hellman speed records" (PKC 2006)
8
+ * describes; the code is this module's.
9
+ *
10
+ * **The field.** An element of GF(p), p = 2^255 - 19, is ten signed limbs in
11
+ * `i64`, alternately 26 and 25 bits wide, so limb `i` sits at bit
12
+ * `ceil(25.5 * i)` (0, 26, 51, 77, …, 230) and the ten of them span exactly
13
+ * 255 bits. Two things make that layout pay:
14
+ *
15
+ * - The product of the limbs at `i` and `j` lands exactly on the limb at
16
+ * `i + j`, except when both are odd, where it lands one bit above and is
17
+ * doubled to make up for it. A product that runs past limb 9 wraps to limb
18
+ * `i + j - 10` times 19, because 2^255 = 19 (mod p). So a multiplication is
19
+ * a hundred `i64` products and no wide arithmetic.
20
+ * - The limbs are signed, and `>>` on an `i64` is an arithmetic shift, so a
21
+ * subtraction needs no bias and a carry is a floor division that works the
22
+ * same on a negative limb.
23
+ *
24
+ * Signed `i64` overflow is undefined behaviour in this language unless a
25
+ * program is compiled with `--wrapping`, so the limb bounds below are not a
26
+ * nicety: each function states what it accepts and what it answers, and
27
+ * `f25519Mul` says why its sums fit.
28
+ *
29
+ * **Constant time, by construction and not yet by proof.** Nothing branches
30
+ * on, or indexes by, a secret: the ladder swaps with a mask and runs all 255
31
+ * steps whatever the scalar, and the final reduction subtracts p by adding a
32
+ * carry bit times 19 rather than by comparing. Every `if` and every loop bound
33
+ * here is on a public quantity — a limb number, a bit position or an input
34
+ * length. WP34 N6's `ctSelect` and its disassembly check are what will verify
35
+ * that; they do not exist yet.
36
+ *
37
+ * **Loop bounds are literals.** A field element is always ten limbs and an
38
+ * encoding always 32 bytes, and each function that indexes one opens with an
39
+ * early `return` for a shorter array. That cannot happen — every element is
40
+ * made here, and `x25519` checks its arguments' lengths before anything else
41
+ * — but it is what lets the compiler prove the loops after it, which count to
42
+ * the literal `10` or `32`, in range and drop their bounds checks. (A named
43
+ * constant as the bound, or a `panic` as the guard, does not prove them.)
44
+ *
45
+ * Private names carry the `f25519` / `x25519` prefix because a `std/` module's
46
+ * private functions share the importing program's flat symbol namespace
47
+ * (`docs/wp26-stdlib.md` §3e).
48
+ */
49
+
50
+ /** The length in bytes of a scalar, a u-coordinate and an X25519 output. */
51
+ export const X25519_SIZE: i32 = 32
52
+
53
+ /** (A - 2) / 4 for curve25519's A = 486662 (RFC 7748 §5). */
54
+ const X25519_A24: i64 = 121665
55
+
56
+ /** The width of limb `i`: 26 bits for an even limb, 25 for an odd one. */
57
+ const f25519Width = (i: i32): i64 => {
58
+ const odd: i64 = toI64(i & 1)
59
+ return 26 - odd
60
+ }
61
+
62
+ /** The mask of limb `i`'s bits, `2^width - 1`. */
63
+ const f25519LimbMask = (i: i32): i64 => (toI64(1) << f25519Width(i)) - toI64(1)
64
+
65
+ /** A fresh field element, zero. */
66
+ const f25519Zero = (): i64[] => {
67
+ const f: i64[] = new Array<i64>(10)
68
+ return f
69
+ }
70
+
71
+ /** A fresh field element holding the small constant `n`. */
72
+ const f25519Small = (n: i64): i64[] => {
73
+ const f: i64[] = f25519Zero()
74
+ f[0] = n
75
+ return f
76
+ }
77
+
78
+ /** `out = f`, limb by limb. */
79
+ const f25519Copy = (out: i64[], f: i64[]): void => {
80
+ if (toI32(out.length) < 10 || toI32(f.length) < 10) {
81
+ return
82
+ }
83
+ for (let i: i32 = 0; i < 10; i++) {
84
+ out[i] = f[i]
85
+ }
86
+ }
87
+
88
+ /**
89
+ * Carries limbs 0 to 8 each into the next, leaving limb 9 as it is: the part
90
+ * of a carry that `f25519Carry` and `f25519Encode` share, and the two differ
91
+ * only in what they do with limb 9's overflow.
92
+ */
93
+ const f25519CarryChain = (h: i64[]): void => {
94
+ if (toI32(h.length) < 10) {
95
+ return
96
+ }
97
+ for (let i: i32 = 0; i < 9; i++) {
98
+ const c: i64 = h[i] >> f25519Width(i)
99
+ h[i] = h[i] & f25519LimbMask(i)
100
+ h[i + 1] = h[i + 1] + c
101
+ }
102
+ }
103
+
104
+ /**
105
+ * Carries every limb into the next, folding the carry out of limb 9 back into
106
+ * limb 0 times 19, and then carries limb 0 once more.
107
+ *
108
+ * Accepts limbs of magnitude below 2^63 minus the carries they receive — in
109
+ * practice below 2^62.6, which is what `f25519Mul` produces. The carry out of
110
+ * a limb that large is below 2^37.6, the fold is below 2^42, so no sum along
111
+ * the chain overflows. Answers a **reduced** element: limbs 0 and 2–9 in
112
+ * `[0, 2^width)`, and limb 1 in that range plus the final carry out of limb 0,
113
+ * which is at most 2^17 either way. So every limb of a reduced element has
114
+ * magnitude below 2^26.
115
+ *
116
+ * Each carry is `h >> width` (arithmetic, so a floor) and each remainder is
117
+ * `h & (2^width - 1)`, which is the matching non-negative remainder for a
118
+ * negative limb too, without shifting a negative number left.
119
+ */
120
+ const f25519Carry = (h: i64[]): void => {
121
+ if (toI32(h.length) < 10) {
122
+ return
123
+ }
124
+ f25519CarryChain(h)
125
+ const top: i64 = h[9] >> toI64(25)
126
+ h[9] = h[9] & f25519LimbMask(9)
127
+ h[0] = h[0] + top * toI64(19)
128
+ const c0: i64 = h[0] >> toI64(26)
129
+ h[0] = h[0] & f25519LimbMask(0)
130
+ h[1] = h[1] + c0
131
+ }
132
+
133
+ /**
134
+ * `out = f + g`, not carried. With both reduced (limbs below 2^26 in
135
+ * magnitude) the sum's limbs are below 2^27, which `f25519Mul` accepts.
136
+ */
137
+ const f25519Add = (out: i64[], f: i64[], g: i64[]): void => {
138
+ if (toI32(out.length) < 10 || toI32(f.length) < 10 || toI32(g.length) < 10) {
139
+ return
140
+ }
141
+ for (let i: i32 = 0; i < 10; i++) {
142
+ out[i] = f[i] + g[i]
143
+ }
144
+ }
145
+
146
+ /**
147
+ * `out = f - g`, not carried. Signed limbs need no bias to stay positive; with
148
+ * both reduced the difference's limbs are below 2^27 in magnitude, which
149
+ * `f25519Mul` accepts.
150
+ */
151
+ const f25519Sub = (out: i64[], f: i64[], g: i64[]): void => {
152
+ if (toI32(out.length) < 10 || toI32(f.length) < 10 || toI32(g.length) < 10) {
153
+ return
154
+ }
155
+ for (let i: i32 = 0; i < 10; i++) {
156
+ out[i] = f[i] - g[i]
157
+ }
158
+ }
159
+
160
+ /**
161
+ * `out = f * g`, reduced. `out` may be `f` or `g`: the product is built in
162
+ * scratch arrays and only copied out at the end.
163
+ *
164
+ * **Why the sums fit in `i64`.** Accepts limbs below 2^27 in magnitude — a
165
+ * reduced element, or the sum or difference of two reduced elements. Output
166
+ * limb `k` collects exactly ten products `f[i] * g[j]` with
167
+ * `i + j = k (mod 10)`, each scaled by at most 2 (both indices odd) times 19
168
+ * (the product wrapped past limb 9). So each term is below 38 * 2^54 and each
169
+ * sum below 380 * 2^54 < 2^62.6, under 2^63 with room for the carries
170
+ * `f25519Carry` adds on top; the partial sums in `wide` are parts of those
171
+ * same sums, and `wide[k + 10] * 19` is below 18 * 19 * 2^54 < 2^58.5 on its
172
+ * own, so no intermediate is larger. Giving `f25519Mul` anything wider — two unreduced
173
+ * sums added again, say — breaks this bound, which is why every caller carries
174
+ * or multiplies before it adds a second time.
175
+ */
176
+ const f25519Mul = (out: i64[], f: i64[], g: i64[]): void => {
177
+ if (toI32(f.length) < 10 || toI32(g.length) < 10) {
178
+ return
179
+ }
180
+ // Two odd limbs meet one bit above their target limb, so an odd-by-odd
181
+ // product takes `f`'s limb doubled.
182
+ const doubled: i64[] = new Array<i64>(10)
183
+ for (let i: i32 = 0; i < 10; i++) {
184
+ doubled[i] = f[i] * toI64((i & 1) + 1)
185
+ }
186
+ // The plain product, limb `i + j` of nineteen.
187
+ const wide: i64[] = new Array<i64>(19)
188
+ for (let i: i32 = 0; i < 10; i++) {
189
+ for (let j: i32 = 0; j < 10; j++) {
190
+ const k: i32 = i + j
191
+ const fi: i64 = (i & j & 1) === 1 ? doubled[i] : f[i]
192
+ if (k >= 0 && k < 19) {
193
+ wide[k] = wide[k] + fi * g[j]
194
+ }
195
+ }
196
+ }
197
+ // Limbs 10 to 18 are past bit 255, and 2^255 = 19 (mod p): fold them down.
198
+ const h: i64[] = new Array<i64>(10)
199
+ for (let k: i32 = 0; k < 9; k++) {
200
+ h[k] = wide[k] + wide[k + 10] * toI64(19)
201
+ }
202
+ h[9] = wide[9]
203
+ f25519Carry(h)
204
+ f25519Copy(out, h)
205
+ }
206
+
207
+ /** `out = f * f`, reduced; the same bounds as `f25519Mul`. */
208
+ const f25519Square = (out: i64[], f: i64[]): void => {
209
+ f25519Mul(out, f, f)
210
+ }
211
+
212
+ /** `out = f` squared `n` times in a row (`n` at least 1), reduced. */
213
+ const f25519SquareTimes = (out: i64[], f: i64[], n: i32): void => {
214
+ f25519Square(out, f)
215
+ for (let i: i32 = 1; i < n; i++) {
216
+ f25519Square(out, out)
217
+ }
218
+ }
219
+
220
+ /**
221
+ * `out = f * a24`, reduced. Accepts limbs below 2^27 in magnitude; times
222
+ * 121665 (under 2^17) each is below 2^44, far inside what `f25519Carry`
223
+ * accepts.
224
+ */
225
+ const f25519MulA24 = (out: i64[], f: i64[]): void => {
226
+ if (toI32(out.length) < 10 || toI32(f.length) < 10) {
227
+ return
228
+ }
229
+ for (let i: i32 = 0; i < 10; i++) {
230
+ out[i] = f[i] * X25519_A24
231
+ }
232
+ f25519Carry(out)
233
+ }
234
+
235
+ /**
236
+ * `out = z^(p - 2)`, which is `1 / z` by Fermat's little theorem, and zero for
237
+ * a zero `z` — what RFC 7748 §5 asks of the ladder's last step.
238
+ *
239
+ * p - 2 = 2^255 - 21 = (2^250 - 1) * 2^5 + 11, so the exponent is 250 ones
240
+ * followed by `01011`. The chain builds `z^(2^n - 1)` for n = 2, 4, 5, 10, 20,
241
+ * 40, 50, 100, 200 and 250 from the rule
242
+ * `z^(2^(a+b) - 1) = (z^(2^a - 1))^(2^b) * z^(2^b - 1)`, then squares five
243
+ * times and multiplies by `z^11`: 254 squarings and 11 multiplications, the
244
+ * same for every `z`.
245
+ */
246
+ const f25519Invert = (out: i64[], z: i64[]): void => {
247
+ const z2: i64[] = f25519Zero()
248
+ const z11: i64[] = f25519Zero()
249
+ const e2: i64[] = f25519Zero()
250
+ const e5: i64[] = f25519Zero()
251
+ const e10: i64[] = f25519Zero()
252
+ const e50: i64[] = f25519Zero()
253
+ const t: i64[] = f25519Zero()
254
+ const acc: i64[] = f25519Zero()
255
+
256
+ f25519Square(z2, z)
257
+ f25519Mul(e2, z2, z) // z^3 = z^(2^2 - 1)
258
+ f25519SquareTimes(t, z2, 2) // z^8
259
+ f25519Mul(z11, t, e2)
260
+
261
+ f25519SquareTimes(t, e2, 2)
262
+ f25519Mul(t, t, e2) // 2^4 - 1
263
+ f25519Square(t, t)
264
+ f25519Mul(e5, t, z) // 2^5 - 1
265
+ f25519SquareTimes(t, e5, 5)
266
+ f25519Mul(e10, t, e5) // 2^10 - 1
267
+ f25519SquareTimes(t, e10, 10)
268
+ f25519Mul(acc, t, e10) // 2^20 - 1
269
+ f25519SquareTimes(t, acc, 20)
270
+ f25519Mul(acc, t, acc) // 2^40 - 1
271
+ f25519SquareTimes(t, acc, 10)
272
+ f25519Mul(e50, t, e10) // 2^50 - 1
273
+ f25519SquareTimes(t, e50, 50)
274
+ f25519Mul(acc, t, e50) // 2^100 - 1
275
+ f25519SquareTimes(t, acc, 100)
276
+ f25519Mul(acc, t, acc) // 2^200 - 1
277
+ f25519SquareTimes(t, acc, 50)
278
+ f25519Mul(acc, t, e50) // 2^250 - 1
279
+ f25519SquareTimes(t, acc, 5)
280
+ f25519Mul(out, t, z11) // 2^255 - 21
281
+ }
282
+
283
+ /**
284
+ * Swaps `f` and `g` when `bit` is 1 and leaves them when it is 0, with the
285
+ * same instructions either way: the mask is all ones or all zeros, and the
286
+ * swap is the masked xor of the two.
287
+ */
288
+ const f25519Swap = (f: i64[], g: i64[], bit: i64): void => {
289
+ if (toI32(f.length) < 10 || toI32(g.length) < 10) {
290
+ return
291
+ }
292
+ const mask: i64 = toI64(0) - bit
293
+ for (let i: i32 = 0; i < 10; i++) {
294
+ const t: i64 = mask & (f[i] ^ g[i])
295
+ f[i] = f[i] ^ t
296
+ g[i] = g[i] ^ t
297
+ }
298
+ }
299
+
300
+ /**
301
+ * The field element that 32 little-endian bytes encode, with the top bit of
302
+ * the last byte masked off as RFC 7748 §5 requires of a u-coordinate. The
303
+ * bytes stream into an accumulator and each limb is taken off its low end once
304
+ * it holds enough bits (one byte never completes two limbs), so every limb
305
+ * lands in `[0, 2^width)` and the element is reduced. A value in
306
+ * `[p, 2^255)` is neither refused nor adjusted: §5 has it processed as its
307
+ * residue, and the arithmetic does that by itself.
308
+ */
309
+ const f25519Decode = (bytes: u8[]): i64[] => {
310
+ const f: i64[] = new Array<i64>(10)
311
+ if (toI32(bytes.length) < 32) {
312
+ return f
313
+ }
314
+ let acc: i64 = 0
315
+ let bits: i64 = 0
316
+ let limb: i32 = 0
317
+ for (let i: i32 = 0; i < 32; i++) {
318
+ let b: i64 = toI64(bytes[i])
319
+ if (i === 31) {
320
+ b = b & toI64(0x7f)
321
+ }
322
+ acc = acc | (b << bits)
323
+ bits = bits + toI64(8)
324
+ if (limb >= 0 && limb < 10) {
325
+ const width: i64 = f25519Width(limb)
326
+ if (bits >= width) {
327
+ f[limb] = acc & f25519LimbMask(limb)
328
+ acc = acc >> width
329
+ bits = bits - width
330
+ limb = limb + 1
331
+ }
332
+ }
333
+ }
334
+ return f
335
+ }
336
+
337
+ /**
338
+ * The canonical 32-byte encoding of `f`, fully reduced into `[0, p)`.
339
+ *
340
+ * Accepts a reduced element. Two more `f25519Carry` passes bring every limb
341
+ * into `[0, 2^width)`. The first leaves every limb in range except limb 1,
342
+ * which may be one over or one under. The second carries that through, and its
343
+ * fold is ±19 at most on a value that is then within 2^26 of 0 or of 2^255, so
344
+ * its last carry out of limb 0 is at most one and leaves limb 1 in range. The
345
+ * value is then in `[0, 2^255)`, which is `[0, p)` or `[p, p + 19)`. Adding 19 and reading the
346
+ * carry out of bit 255 tells the two apart without a branch — it is 1 exactly
347
+ * when the value is at least p — so the element adds `19 * q` and drops bit
348
+ * 255, which is subtracting `p * q`.
349
+ */
350
+ const f25519Encode = (f: i64[]): u8[] => {
351
+ const h: i64[] = new Array<i64>(10)
352
+ f25519Copy(h, f)
353
+ f25519Carry(h)
354
+ f25519Carry(h)
355
+
356
+ let q: i64 = (h[0] + toI64(19)) >> toI64(26)
357
+ for (let i: i32 = 1; i < 10; i++) {
358
+ q = (h[i] + q) >> f25519Width(i)
359
+ }
360
+ h[0] = h[0] + q * toI64(19)
361
+ f25519CarryChain(h)
362
+ h[9] = h[9] & f25519LimbMask(9)
363
+
364
+ const out: u8[] = new Array<u8>(32)
365
+ let acc: i64 = 0
366
+ let bits: i64 = 0
367
+ let at: i32 = 0
368
+ for (let i: i32 = 0; i < 10; i++) {
369
+ acc = acc | (h[i] << bits)
370
+ bits = bits + f25519Width(i)
371
+ while (bits >= toI64(8) && at >= 0 && at < 32) {
372
+ out[at] = toU8(acc & toI64(0xff))
373
+ acc = acc >> toI64(8)
374
+ bits = bits - toI64(8)
375
+ at = at + 1
376
+ }
377
+ }
378
+ // 255 bits leave seven over: the last byte, with its top bit clear.
379
+ if (at >= 0 && at < 32) {
380
+ out[at] = toU8(acc & toI64(0xff))
381
+ }
382
+ return out
383
+ }
384
+
385
+ /**
386
+ * The Montgomery ladder of RFC 7748 §5 on a clamped scalar `k` and a decoded
387
+ * `u`, answering the encoded u-coordinate of `k * u`.
388
+ *
389
+ * The step formulas are the RFC's, in its order. Every multiplication is given
390
+ * a reduced element or one sum or difference of two, which is what
391
+ * `f25519Mul` accepts: `A`, `B`, `C`, `D`, `E`, `DA + CB`, `DA - CB` and
392
+ * `AA + a24 * E` are each one addition away from reduced values.
393
+ */
394
+ const x25519Ladder = (k: u8[], u: i64[]): u8[] => {
395
+ if (toI32(k.length) < 32) {
396
+ return new Array<u8>(32)
397
+ }
398
+ const x2: i64[] = f25519Small(toI64(1))
399
+ const z2: i64[] = f25519Zero()
400
+ const x3: i64[] = f25519Zero()
401
+ const z3: i64[] = f25519Small(toI64(1))
402
+ f25519Copy(x3, u)
403
+
404
+ const a: i64[] = f25519Zero()
405
+ const aa: i64[] = f25519Zero()
406
+ const b: i64[] = f25519Zero()
407
+ const bb: i64[] = f25519Zero()
408
+ const e: i64[] = f25519Zero()
409
+ const c: i64[] = f25519Zero()
410
+ const d: i64[] = f25519Zero()
411
+ const da: i64[] = f25519Zero()
412
+ const cb: i64[] = f25519Zero()
413
+ const t: i64[] = f25519Zero()
414
+
415
+ let swap: i64 = 0
416
+ // Bit 254 down to bit 0: the position is public and picks the byte and the
417
+ // shift; the bit itself is secret and only ever feeds a mask.
418
+ for (let byte: i32 = 31; byte >= 0; byte--) {
419
+ for (let shift: i32 = 7; shift >= 0; shift--) {
420
+ if (byte === 31 && shift === 7) {
421
+ continue
422
+ }
423
+ const bit: i64 = toI64(k[byte] >> toU8(shift)) & toI64(1)
424
+ swap = swap ^ bit
425
+ f25519Swap(x2, x3, swap)
426
+ f25519Swap(z2, z3, swap)
427
+ swap = bit
428
+
429
+ f25519Add(a, x2, z2)
430
+ f25519Square(aa, a)
431
+ f25519Sub(b, x2, z2)
432
+ f25519Square(bb, b)
433
+ f25519Sub(e, aa, bb)
434
+ f25519Add(c, x3, z3)
435
+ f25519Sub(d, x3, z3)
436
+ f25519Mul(da, d, a)
437
+ f25519Mul(cb, c, b)
438
+ f25519Add(t, da, cb)
439
+ f25519Square(x3, t)
440
+ f25519Sub(t, da, cb)
441
+ f25519Square(t, t)
442
+ f25519Mul(z3, u, t)
443
+ f25519Mul(x2, aa, bb)
444
+ f25519MulA24(t, e)
445
+ f25519Add(t, aa, t)
446
+ f25519Mul(z2, e, t)
447
+ }
448
+ }
449
+ f25519Swap(x2, x3, swap)
450
+ f25519Swap(z2, z3, swap)
451
+
452
+ f25519Invert(t, z2)
453
+ f25519Mul(x2, x2, t)
454
+ return f25519Encode(x2)
455
+ }
456
+
457
+ /**
458
+ * X25519 (RFC 7748 §5): the u-coordinate of `scalar * u` on curve25519, as 32
459
+ * little-endian bytes.
460
+ *
461
+ * `scalar` is clamped as §5 says — bits 0, 1, 2 and 255 cleared, bit 254 set —
462
+ * on a copy, so the caller's array is not changed. The top bit of `u` is
463
+ * ignored, and a `u` of p or more is taken modulo p. The answer is always
464
+ * canonical (below p).
465
+ *
466
+ * Answers `null` unless both arguments are exactly `X25519_SIZE` bytes. An
467
+ * all-zero answer is **returned, not refused**: it is what a low-order `u`
468
+ * gives, and RFC 7748 §6.1 leaves checking for it to the protocol, which is
469
+ * where TLS 1.3 (WP34 T1) makes that check. A caller doing a key exchange
470
+ * outside TLS should make it too.
471
+ */
472
+ export const x25519 = (scalar: u8[], u: u8[]): u8[] | null => {
473
+ if (toI32(scalar.length) !== 32 || toI32(u.length) !== 32) {
474
+ return null
475
+ }
476
+ const k: u8[] = new Array<u8>(32)
477
+ for (let i: i32 = 0; i < 32; i++) {
478
+ k[i] = scalar[i]
479
+ }
480
+ k[0] = k[0] & toU8(248)
481
+ k[31] = (k[31] & toU8(127)) | toU8(64)
482
+ return x25519Ladder(k, f25519Decode(u))
483
+ }
484
+
485
+ /**
486
+ * `x25519(scalar, 9)`: the public key for the private key `scalar`, since 9 is
487
+ * curve25519's base point (RFC 7748 §4.1, §6.1). Answers `null` unless
488
+ * `scalar` is `X25519_SIZE` bytes.
489
+ */
490
+ export const x25519Base = (scalar: u8[]): u8[] | null => {
491
+ const base: u8[] = new Array<u8>(32)
492
+ base[0] = toU8(9)
493
+ return x25519(scalar, base)
494
+ }