@pulse-compute/crypto 1.0.0-beta.4 → 1.0.0-beta.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (95) hide show
  1. package/README.md +89 -8
  2. package/as/digest-text.as.ts +34 -0
  3. package/as/pulse-hmac-as.ts +7 -1
  4. package/conformance/digest-text.json +38 -0
  5. package/conformance/hs256.json +85 -0
  6. package/dist/contracts.d.ts +34 -12
  7. package/dist/contracts.d.ts.map +1 -1
  8. package/dist/contracts.js +9 -3
  9. package/dist/contracts.js.map +1 -1
  10. package/dist/digest.d.ts +14 -0
  11. package/dist/digest.d.ts.map +1 -0
  12. package/dist/digest.js +17 -0
  13. package/dist/digest.js.map +1 -0
  14. package/dist/index.d.ts +6 -0
  15. package/dist/index.d.ts.map +1 -1
  16. package/dist/index.js +20 -0
  17. package/dist/index.js.map +1 -1
  18. package/dist/internal/rsa.d.ts +58 -0
  19. package/dist/internal/rsa.d.ts.map +1 -0
  20. package/dist/internal/rsa.js +185 -0
  21. package/dist/internal/rsa.js.map +1 -0
  22. package/dist/internal/verification.d.ts +3 -3
  23. package/dist/internal/verification.d.ts.map +1 -1
  24. package/dist/internal/verification.js +2 -2
  25. package/dist/internal/verification.js.map +1 -1
  26. package/guests/es256-rustcrypto/README.md +43 -0
  27. package/guests/es256-rustcrypto/build.cjs +297 -0
  28. package/guests/es256-rustcrypto/prebuilt/es256-verifier.unoptimized.wasm +0 -0
  29. package/guests/es256-rustcrypto/prebuilt/es256-verifier.wasm +0 -0
  30. package/guests/es256-rustcrypto/probe/.cargo/config.toml +18 -0
  31. package/guests/es256-rustcrypto/probe/Cargo.lock +238 -0
  32. package/guests/es256-rustcrypto/probe/Cargo.toml +26 -0
  33. package/guests/es256-rustcrypto/probe/README.md +35 -0
  34. package/guests/es256-rustcrypto/probe/rust-toolchain.toml +5 -0
  35. package/guests/es256-rustcrypto/probe/src/lib.rs +108 -0
  36. package/guests/es256-rustcrypto/pulse.guest-unit.json +119 -0
  37. package/guests/es256-rustcrypto/source/.cargo/config.toml +18 -0
  38. package/guests/es256-rustcrypto/source/Cargo.lock +265 -0
  39. package/guests/es256-rustcrypto/source/Cargo.toml +23 -0
  40. package/guests/es256-rustcrypto/source/README.md +26 -0
  41. package/guests/es256-rustcrypto/source/bearssl/LICENSE.txt +21 -0
  42. package/guests/es256-rustcrypto/source/bearssl/UPSTREAM.json +49 -0
  43. package/guests/es256-rustcrypto/source/bearssl/inc/bearssl.h +170 -0
  44. package/guests/es256-rustcrypto/source/bearssl/inc/bearssl_aead.h +1059 -0
  45. package/guests/es256-rustcrypto/source/bearssl/inc/bearssl_block.h +2618 -0
  46. package/guests/es256-rustcrypto/source/bearssl/inc/bearssl_ec.h +883 -0
  47. package/guests/es256-rustcrypto/source/bearssl/inc/bearssl_hash.h +1346 -0
  48. package/guests/es256-rustcrypto/source/bearssl/inc/bearssl_hmac.h +241 -0
  49. package/guests/es256-rustcrypto/source/bearssl/inc/bearssl_kdf.h +185 -0
  50. package/guests/es256-rustcrypto/source/bearssl/inc/bearssl_pem.h +294 -0
  51. package/guests/es256-rustcrypto/source/bearssl/inc/bearssl_prf.h +150 -0
  52. package/guests/es256-rustcrypto/source/bearssl/inc/bearssl_rand.h +397 -0
  53. package/guests/es256-rustcrypto/source/bearssl/inc/bearssl_rsa.h +1366 -0
  54. package/guests/es256-rustcrypto/source/bearssl/inc/bearssl_ssl.h +4296 -0
  55. package/guests/es256-rustcrypto/source/bearssl/inc/bearssl_x509.h +1397 -0
  56. package/guests/es256-rustcrypto/source/bearssl/src/codec/ccopy.c +44 -0
  57. package/guests/es256-rustcrypto/source/bearssl/src/codec/dec32be.c +38 -0
  58. package/guests/es256-rustcrypto/source/bearssl/src/codec/enc32be.c +38 -0
  59. package/guests/es256-rustcrypto/source/bearssl/src/config.h +229 -0
  60. package/guests/es256-rustcrypto/source/bearssl/src/inner.h +2532 -0
  61. package/guests/es256-rustcrypto/source/bearssl/src/int/i31_add.c +46 -0
  62. package/guests/es256-rustcrypto/source/bearssl/src/int/i31_bitlen.c +44 -0
  63. package/guests/es256-rustcrypto/source/bearssl/src/int/i31_decmod.c +124 -0
  64. package/guests/es256-rustcrypto/source/bearssl/src/int/i31_decode.c +57 -0
  65. package/guests/es256-rustcrypto/source/bearssl/src/int/i31_decred.c +103 -0
  66. package/guests/es256-rustcrypto/source/bearssl/src/int/i31_encode.c +79 -0
  67. package/guests/es256-rustcrypto/source/bearssl/src/int/i31_fmont.c +60 -0
  68. package/guests/es256-rustcrypto/source/bearssl/src/int/i31_iszero.c +39 -0
  69. package/guests/es256-rustcrypto/source/bearssl/src/int/i31_moddiv.c +488 -0
  70. package/guests/es256-rustcrypto/source/bearssl/src/int/i31_modpow.c +65 -0
  71. package/guests/es256-rustcrypto/source/bearssl/src/int/i31_modpow2.c +160 -0
  72. package/guests/es256-rustcrypto/source/bearssl/src/int/i31_montmul.c +93 -0
  73. package/guests/es256-rustcrypto/source/bearssl/src/int/i31_mulacc.c +61 -0
  74. package/guests/es256-rustcrypto/source/bearssl/src/int/i31_muladd.c +157 -0
  75. package/guests/es256-rustcrypto/source/bearssl/src/int/i31_ninv31.c +39 -0
  76. package/guests/es256-rustcrypto/source/bearssl/src/int/i31_reduce.c +66 -0
  77. package/guests/es256-rustcrypto/source/bearssl/src/int/i31_rshift.c +47 -0
  78. package/guests/es256-rustcrypto/source/bearssl/src/int/i31_sub.c +46 -0
  79. package/guests/es256-rustcrypto/source/bearssl/src/int/i31_tmont.c +36 -0
  80. package/guests/es256-rustcrypto/source/bearssl/src/int/i32_div32.c +56 -0
  81. package/guests/es256-rustcrypto/source/bearssl/src/rsa/rsa_i31_priv.c +203 -0
  82. package/guests/es256-rustcrypto/source/bearssl/src/rsa/rsa_i31_pub.c +106 -0
  83. package/guests/es256-rustcrypto/source/bearssl/src/rsa/rsa_pkcs1_sig_pad.c +100 -0
  84. package/guests/es256-rustcrypto/source/build.rs +38 -0
  85. package/guests/es256-rustcrypto/source/freestanding/string.h +8 -0
  86. package/guests/es256-rustcrypto/source/rs256.c +103 -0
  87. package/guests/es256-rustcrypto/source/rust-toolchain.toml +5 -0
  88. package/guests/es256-rustcrypto/source/src/lib.rs +308 -0
  89. package/package.json +22 -4
  90. package/pulse.package.json +64 -0
  91. package/pulsewasm.compiler.cjs +89 -0
  92. package/pulsewasm.manifest.cjs +15 -0
  93. package/pulsewasm.native.cjs +79 -7
  94. package/src/provider.cjs +141 -2
  95. package/src/provider.d.cts +6 -0
@@ -0,0 +1,488 @@
1
+ /*
2
+ * Copyright (c) 2018 Thomas Pornin <pornin@bolet.org>
3
+ *
4
+ * Permission is hereby granted, free of charge, to any person obtaining
5
+ * a copy of this software and associated documentation files (the
6
+ * "Software"), to deal in the Software without restriction, including
7
+ * without limitation the rights to use, copy, modify, merge, publish,
8
+ * distribute, sublicense, and/or sell copies of the Software, and to
9
+ * permit persons to whom the Software is furnished to do so, subject to
10
+ * the following conditions:
11
+ *
12
+ * The above copyright notice and this permission notice shall be
13
+ * included in all copies or substantial portions of the Software.
14
+ *
15
+ * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
16
+ * EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
17
+ * MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND
18
+ * NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS
19
+ * BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN
20
+ * ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
21
+ * CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
22
+ * SOFTWARE.
23
+ */
24
+
25
+ #include "inner.h"
26
+
27
+ /*
28
+ * In this file, we handle big integers with a custom format, i.e.
29
+ * without the usual one-word header. Value is split into 31-bit words,
30
+ * each stored in a 32-bit slot (top bit is zero) in little-endian
31
+ * order. The length (in words) is provided explicitly. In some cases,
32
+ * the value can be negative (using two's complement representation). In
33
+ * some cases, the top word is allowed to have a 32th bit.
34
+ */
35
+
36
+ /*
37
+ * Negate big integer conditionally. The value consists of 'len' words,
38
+ * with 31 bits in each word (the top bit of each word should be 0,
39
+ * except possibly for the last word). If 'ctl' is 1, the negation is
40
+ * computed; otherwise, if 'ctl' is 0, then the value is unchanged.
41
+ */
42
+ static void
43
+ cond_negate(uint32_t *a, size_t len, uint32_t ctl)
44
+ {
45
+ size_t k;
46
+ uint32_t cc, xm;
47
+
48
+ cc = ctl;
49
+ xm = -ctl >> 1;
50
+ for (k = 0; k < len; k ++) {
51
+ uint32_t aw;
52
+
53
+ aw = a[k];
54
+ aw = (aw ^ xm) + cc;
55
+ a[k] = aw & 0x7FFFFFFF;
56
+ cc = aw >> 31;
57
+ }
58
+ }
59
+
60
+ /*
61
+ * Finish modular reduction. Rules on input parameters:
62
+ *
63
+ * if neg = 1, then -m <= a < 0
64
+ * if neg = 0, then 0 <= a < 2*m
65
+ *
66
+ * If neg = 0, then the top word of a[] may use 32 bits.
67
+ *
68
+ * Also, modulus m must be odd.
69
+ */
70
+ static void
71
+ finish_mod(uint32_t *a, size_t len, const uint32_t *m, uint32_t neg)
72
+ {
73
+ size_t k;
74
+ uint32_t cc, xm, ym;
75
+
76
+ /*
77
+ * First pass: compare a (assumed nonnegative) with m.
78
+ * Note that if the final word uses the top extra bit, then
79
+ * subtracting m must yield a value less than 2^31, since we
80
+ * assumed that a < 2*m.
81
+ */
82
+ cc = 0;
83
+ for (k = 0; k < len; k ++) {
84
+ uint32_t aw, mw;
85
+
86
+ aw = a[k];
87
+ mw = m[k];
88
+ cc = (aw - mw - cc) >> 31;
89
+ }
90
+
91
+ /*
92
+ * At this point:
93
+ * if neg = 1, then we must add m (regardless of cc)
94
+ * if neg = 0 and cc = 0, then we must subtract m
95
+ * if neg = 0 and cc = 1, then we must do nothing
96
+ */
97
+ xm = -neg >> 1;
98
+ ym = -(neg | (1 - cc));
99
+ cc = neg;
100
+ for (k = 0; k < len; k ++) {
101
+ uint32_t aw, mw;
102
+
103
+ aw = a[k];
104
+ mw = (m[k] ^ xm) & ym;
105
+ aw = aw - mw - cc;
106
+ a[k] = aw & 0x7FFFFFFF;
107
+ cc = aw >> 31;
108
+ }
109
+ }
110
+
111
+ /*
112
+ * Compute:
113
+ * a <- (a*pa+b*pb)/(2^31)
114
+ * b <- (a*qa+b*qb)/(2^31)
115
+ * The division is assumed to be exact (i.e. the low word is dropped).
116
+ * If the final a is negative, then it is negated. Similarly for b.
117
+ * Returned value is the combination of two bits:
118
+ * bit 0: 1 if a had to be negated, 0 otherwise
119
+ * bit 1: 1 if b had to be negated, 0 otherwise
120
+ *
121
+ * Factors pa, pb, qa and qb must be at most 2^31 in absolute value.
122
+ * Source integers a and b must be nonnegative; top word is not allowed
123
+ * to contain an extra 32th bit.
124
+ */
125
+ static uint32_t
126
+ co_reduce(uint32_t *a, uint32_t *b, size_t len,
127
+ int64_t pa, int64_t pb, int64_t qa, int64_t qb)
128
+ {
129
+ size_t k;
130
+ int64_t cca, ccb;
131
+ uint32_t nega, negb;
132
+
133
+ cca = 0;
134
+ ccb = 0;
135
+ for (k = 0; k < len; k ++) {
136
+ uint32_t wa, wb;
137
+ uint64_t za, zb;
138
+ uint64_t tta, ttb;
139
+
140
+ /*
141
+ * Since:
142
+ * |pa| <= 2^31
143
+ * |pb| <= 2^31
144
+ * 0 <= wa <= 2^31 - 1
145
+ * 0 <= wb <= 2^31 - 1
146
+ * |cca| <= 2^32 - 1
147
+ * Then:
148
+ * |za| <= (2^31-1)*(2^32) + (2^32-1) = 2^63 - 1
149
+ *
150
+ * Thus, the new value of cca is such that |cca| <= 2^32 - 1.
151
+ * The same applies to ccb.
152
+ */
153
+ wa = a[k];
154
+ wb = b[k];
155
+ za = wa * (uint64_t)pa + wb * (uint64_t)pb + (uint64_t)cca;
156
+ zb = wa * (uint64_t)qa + wb * (uint64_t)qb + (uint64_t)ccb;
157
+ if (k > 0) {
158
+ a[k - 1] = za & 0x7FFFFFFF;
159
+ b[k - 1] = zb & 0x7FFFFFFF;
160
+ }
161
+
162
+ /*
163
+ * For the new values of cca and ccb, we need a signed
164
+ * right-shift; since, in C, right-shifting a signed
165
+ * negative value is implementation-defined, we use a
166
+ * custom portable sign extension expression.
167
+ */
168
+ #define M ((uint64_t)1 << 32)
169
+ tta = za >> 31;
170
+ ttb = zb >> 31;
171
+ tta = (tta ^ M) - M;
172
+ ttb = (ttb ^ M) - M;
173
+ cca = *(int64_t *)&tta;
174
+ ccb = *(int64_t *)&ttb;
175
+ #undef M
176
+ }
177
+ a[len - 1] = (uint32_t)cca;
178
+ b[len - 1] = (uint32_t)ccb;
179
+
180
+ nega = (uint32_t)((uint64_t)cca >> 63);
181
+ negb = (uint32_t)((uint64_t)ccb >> 63);
182
+ cond_negate(a, len, nega);
183
+ cond_negate(b, len, negb);
184
+ return nega | (negb << 1);
185
+ }
186
+
187
+ /*
188
+ * Compute:
189
+ * a <- (a*pa+b*pb)/(2^31) mod m
190
+ * b <- (a*qa+b*qb)/(2^31) mod m
191
+ *
192
+ * m0i is equal to -1/m[0] mod 2^31.
193
+ *
194
+ * Factors pa, pb, qa and qb must be at most 2^31 in absolute value.
195
+ * Source integers a and b must be nonnegative; top word is not allowed
196
+ * to contain an extra 32th bit.
197
+ */
198
+ static void
199
+ co_reduce_mod(uint32_t *a, uint32_t *b, size_t len,
200
+ int64_t pa, int64_t pb, int64_t qa, int64_t qb,
201
+ const uint32_t *m, uint32_t m0i)
202
+ {
203
+ size_t k;
204
+ int64_t cca, ccb;
205
+ uint32_t fa, fb;
206
+
207
+ cca = 0;
208
+ ccb = 0;
209
+ fa = ((a[0] * (uint32_t)pa + b[0] * (uint32_t)pb) * m0i) & 0x7FFFFFFF;
210
+ fb = ((a[0] * (uint32_t)qa + b[0] * (uint32_t)qb) * m0i) & 0x7FFFFFFF;
211
+ for (k = 0; k < len; k ++) {
212
+ uint32_t wa, wb;
213
+ uint64_t za, zb;
214
+ uint64_t tta, ttb;
215
+
216
+ /*
217
+ * In this loop, carries 'cca' and 'ccb' always fit on
218
+ * 33 bits (in absolute value).
219
+ */
220
+ wa = a[k];
221
+ wb = b[k];
222
+ za = wa * (uint64_t)pa + wb * (uint64_t)pb
223
+ + m[k] * (uint64_t)fa + (uint64_t)cca;
224
+ zb = wa * (uint64_t)qa + wb * (uint64_t)qb
225
+ + m[k] * (uint64_t)fb + (uint64_t)ccb;
226
+ if (k > 0) {
227
+ a[k - 1] = (uint32_t)za & 0x7FFFFFFF;
228
+ b[k - 1] = (uint32_t)zb & 0x7FFFFFFF;
229
+ }
230
+
231
+ #define M ((uint64_t)1 << 32)
232
+ tta = za >> 31;
233
+ ttb = zb >> 31;
234
+ tta = (tta ^ M) - M;
235
+ ttb = (ttb ^ M) - M;
236
+ cca = *(int64_t *)&tta;
237
+ ccb = *(int64_t *)&ttb;
238
+ #undef M
239
+ }
240
+ a[len - 1] = (uint32_t)cca;
241
+ b[len - 1] = (uint32_t)ccb;
242
+
243
+ /*
244
+ * At this point:
245
+ * -m <= a < 2*m
246
+ * -m <= b < 2*m
247
+ * (this is a case of Montgomery reduction)
248
+ * The top word of 'a' and 'b' may have a 32-th bit set.
249
+ * We may have to add or subtract the modulus.
250
+ */
251
+ finish_mod(a, len, m, (uint32_t)((uint64_t)cca >> 63));
252
+ finish_mod(b, len, m, (uint32_t)((uint64_t)ccb >> 63));
253
+ }
254
+
255
+ /* see inner.h */
256
+ uint32_t
257
+ br_i31_moddiv(uint32_t *x, const uint32_t *y, const uint32_t *m, uint32_t m0i,
258
+ uint32_t *t)
259
+ {
260
+ /*
261
+ * Algorithm is an extended binary GCD. We maintain four values
262
+ * a, b, u and v, with the following invariants:
263
+ *
264
+ * a * x = y * u mod m
265
+ * b * x = y * v mod m
266
+ *
267
+ * Starting values are:
268
+ *
269
+ * a = y
270
+ * b = m
271
+ * u = x
272
+ * v = 0
273
+ *
274
+ * The formal definition of the algorithm is a sequence of steps:
275
+ *
276
+ * - If a is even, then a <- a/2 and u <- u/2 mod m.
277
+ * - Otherwise, if b is even, then b <- b/2 and v <- v/2 mod m.
278
+ * - Otherwise, if a > b, then a <- (a-b)/2 and u <- (u-v)/2 mod m.
279
+ * - Otherwise, b <- (b-a)/2 and v <- (v-u)/2 mod m.
280
+ *
281
+ * Algorithm stops when a = b. At that point, they both are equal
282
+ * to GCD(y,m); the modular division succeeds if that value is 1.
283
+ * The result of the modular division is then u (or v: both are
284
+ * equal at that point).
285
+ *
286
+ * Each step makes either a or b shrink by at least one bit; hence,
287
+ * if m has bit length k bits, then 2k-2 steps are sufficient.
288
+ *
289
+ *
290
+ * Though complexity is quadratic in the size of m, the bit-by-bit
291
+ * processing is not very efficient. We can speed up processing by
292
+ * remarking that the decisions are taken based only on observation
293
+ * of the top and low bits of a and b.
294
+ *
295
+ * In the loop below, at each iteration, we use the two top words
296
+ * of a and b, and the low words of a and b, to compute reduction
297
+ * parameters pa, pb, qa and qb such that the new values for a
298
+ * and b are:
299
+ *
300
+ * a' = (a*pa + b*pb) / (2^31)
301
+ * b' = (a*qa + b*qb) / (2^31)
302
+ *
303
+ * the division being exact.
304
+ *
305
+ * Since the choices are based on the top words, they may be slightly
306
+ * off, requiring an optional correction: if a' < 0, then we replace
307
+ * pa with -pa, and pb with -pb. The total length of a and b is
308
+ * thus reduced by at least 30 bits at each iteration.
309
+ *
310
+ * The stopping conditions are still the same, though: when a
311
+ * and b become equal, they must be both odd (since m is odd,
312
+ * the GCD cannot be even), therefore the next operation is a
313
+ * subtraction, and one of the values becomes 0. At that point,
314
+ * nothing else happens, i.e. one value is stuck at 0, and the
315
+ * other one is the GCD.
316
+ */
317
+ size_t len, k;
318
+ uint32_t *a, *b, *u, *v;
319
+ uint32_t num, r;
320
+
321
+ len = (m[0] + 31) >> 5;
322
+ a = t;
323
+ b = a + len;
324
+ u = x + 1;
325
+ v = b + len;
326
+ memcpy(a, y + 1, len * sizeof *y);
327
+ memcpy(b, m + 1, len * sizeof *m);
328
+ memset(v, 0, len * sizeof *v);
329
+
330
+ /*
331
+ * Loop below ensures that a and b are reduced by some bits each,
332
+ * for a total of at least 30 bits.
333
+ */
334
+ for (num = ((m[0] - (m[0] >> 5)) << 1) + 30; num >= 30; num -= 30) {
335
+ size_t j;
336
+ uint32_t c0, c1;
337
+ uint32_t a0, a1, b0, b1;
338
+ uint64_t a_hi, b_hi;
339
+ uint32_t a_lo, b_lo;
340
+ int64_t pa, pb, qa, qb;
341
+ int i;
342
+
343
+ /*
344
+ * Extract top words of a and b. If j is the highest
345
+ * index >= 1 such that a[j] != 0 or b[j] != 0, then we want
346
+ * (a[j] << 31) + a[j - 1], and (b[j] << 31) + b[j - 1].
347
+ * If a and b are down to one word each, then we use a[0]
348
+ * and b[0].
349
+ */
350
+ c0 = (uint32_t)-1;
351
+ c1 = (uint32_t)-1;
352
+ a0 = 0;
353
+ a1 = 0;
354
+ b0 = 0;
355
+ b1 = 0;
356
+ j = len;
357
+ while (j -- > 0) {
358
+ uint32_t aw, bw;
359
+
360
+ aw = a[j];
361
+ bw = b[j];
362
+ a0 ^= (a0 ^ aw) & c0;
363
+ a1 ^= (a1 ^ aw) & c1;
364
+ b0 ^= (b0 ^ bw) & c0;
365
+ b1 ^= (b1 ^ bw) & c1;
366
+ c1 = c0;
367
+ c0 &= (((aw | bw) + 0x7FFFFFFF) >> 31) - (uint32_t)1;
368
+ }
369
+
370
+ /*
371
+ * If c1 = 0, then we grabbed two words for a and b.
372
+ * If c1 != 0 but c0 = 0, then we grabbed one word. It
373
+ * is not possible that c1 != 0 and c0 != 0, because that
374
+ * would mean that both integers are zero.
375
+ */
376
+ a1 |= a0 & c1;
377
+ a0 &= ~c1;
378
+ b1 |= b0 & c1;
379
+ b0 &= ~c1;
380
+ a_hi = ((uint64_t)a0 << 31) + a1;
381
+ b_hi = ((uint64_t)b0 << 31) + b1;
382
+ a_lo = a[0];
383
+ b_lo = b[0];
384
+
385
+ /*
386
+ * Compute reduction factors:
387
+ *
388
+ * a' = a*pa + b*pb
389
+ * b' = a*qa + b*qb
390
+ *
391
+ * such that a' and b' are both multiple of 2^31, but are
392
+ * only marginally larger than a and b.
393
+ */
394
+ pa = 1;
395
+ pb = 0;
396
+ qa = 0;
397
+ qb = 1;
398
+ for (i = 0; i < 31; i ++) {
399
+ /*
400
+ * At each iteration:
401
+ *
402
+ * a <- (a-b)/2 if: a is odd, b is odd, a_hi > b_hi
403
+ * b <- (b-a)/2 if: a is odd, b is odd, a_hi <= b_hi
404
+ * a <- a/2 if: a is even
405
+ * b <- b/2 if: a is odd, b is even
406
+ *
407
+ * We multiply a_lo and b_lo by 2 at each
408
+ * iteration, thus a division by 2 really is a
409
+ * non-multiplication by 2.
410
+ */
411
+ uint32_t r, oa, ob, cAB, cBA, cA;
412
+ uint64_t rz;
413
+
414
+ /*
415
+ * r = GT(a_hi, b_hi)
416
+ * But the GT() function works on uint32_t operands,
417
+ * so we inline a 64-bit version here.
418
+ */
419
+ rz = b_hi - a_hi;
420
+ r = (uint32_t)((rz ^ ((a_hi ^ b_hi)
421
+ & (a_hi ^ rz))) >> 63);
422
+
423
+ /*
424
+ * cAB = 1 if b must be subtracted from a
425
+ * cBA = 1 if a must be subtracted from b
426
+ * cA = 1 if a is divided by 2, 0 otherwise
427
+ *
428
+ * Rules:
429
+ *
430
+ * cAB and cBA cannot be both 1.
431
+ * if a is not divided by 2, b is.
432
+ */
433
+ oa = (a_lo >> i) & 1;
434
+ ob = (b_lo >> i) & 1;
435
+ cAB = oa & ob & r;
436
+ cBA = oa & ob & NOT(r);
437
+ cA = cAB | NOT(oa);
438
+
439
+ /*
440
+ * Conditional subtractions.
441
+ */
442
+ a_lo -= b_lo & -cAB;
443
+ a_hi -= b_hi & -(uint64_t)cAB;
444
+ pa -= qa & -(int64_t)cAB;
445
+ pb -= qb & -(int64_t)cAB;
446
+ b_lo -= a_lo & -cBA;
447
+ b_hi -= a_hi & -(uint64_t)cBA;
448
+ qa -= pa & -(int64_t)cBA;
449
+ qb -= pb & -(int64_t)cBA;
450
+
451
+ /*
452
+ * Shifting.
453
+ */
454
+ a_lo += a_lo & (cA - 1);
455
+ pa += pa & ((int64_t)cA - 1);
456
+ pb += pb & ((int64_t)cA - 1);
457
+ a_hi ^= (a_hi ^ (a_hi >> 1)) & -(uint64_t)cA;
458
+ b_lo += b_lo & -cA;
459
+ qa += qa & -(int64_t)cA;
460
+ qb += qb & -(int64_t)cA;
461
+ b_hi ^= (b_hi ^ (b_hi >> 1)) & ((uint64_t)cA - 1);
462
+ }
463
+
464
+ /*
465
+ * Replace a and b with new values a' and b'.
466
+ */
467
+ r = co_reduce(a, b, len, pa, pb, qa, qb);
468
+ pa -= pa * ((r & 1) << 1);
469
+ pb -= pb * ((r & 1) << 1);
470
+ qa -= qa * (r & 2);
471
+ qb -= qb * (r & 2);
472
+ co_reduce_mod(u, v, len, pa, pb, qa, qb, m + 1, m0i);
473
+ }
474
+
475
+ /*
476
+ * Now one of the arrays should be 0, and the other contains
477
+ * the GCD. If a is 0, then u is 0 as well, and v contains
478
+ * the division result.
479
+ * Result is correct if and only if GCD is 1.
480
+ */
481
+ r = (a[0] | b[0]) ^ 1;
482
+ u[0] |= v[0];
483
+ for (k = 1; k < len; k ++) {
484
+ r |= a[k] | b[k];
485
+ u[k] |= v[k];
486
+ }
487
+ return EQ0(r);
488
+ }
@@ -0,0 +1,65 @@
1
+ /*
2
+ * Copyright (c) 2016 Thomas Pornin <pornin@bolet.org>
3
+ *
4
+ * Permission is hereby granted, free of charge, to any person obtaining
5
+ * a copy of this software and associated documentation files (the
6
+ * "Software"), to deal in the Software without restriction, including
7
+ * without limitation the rights to use, copy, modify, merge, publish,
8
+ * distribute, sublicense, and/or sell copies of the Software, and to
9
+ * permit persons to whom the Software is furnished to do so, subject to
10
+ * the following conditions:
11
+ *
12
+ * The above copyright notice and this permission notice shall be
13
+ * included in all copies or substantial portions of the Software.
14
+ *
15
+ * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
16
+ * EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
17
+ * MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND
18
+ * NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS
19
+ * BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN
20
+ * ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
21
+ * CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
22
+ * SOFTWARE.
23
+ */
24
+
25
+ #include "inner.h"
26
+
27
+ /* see inner.h */
28
+ void
29
+ br_i31_modpow(uint32_t *x,
30
+ const unsigned char *e, size_t elen,
31
+ const uint32_t *m, uint32_t m0i, uint32_t *t1, uint32_t *t2)
32
+ {
33
+ size_t mlen;
34
+ uint32_t k;
35
+
36
+ /*
37
+ * 'mlen' is the length of m[] expressed in bytes (including
38
+ * the "bit length" first field).
39
+ */
40
+ mlen = ((m[0] + 63) >> 5) * sizeof m[0];
41
+
42
+ /*
43
+ * Throughout the algorithm:
44
+ * -- t1[] is in Montgomery representation; it contains x, x^2,
45
+ * x^4, x^8...
46
+ * -- The result is accumulated, in normal representation, in
47
+ * the x[] array.
48
+ * -- t2[] is used as destination buffer for each multiplication.
49
+ *
50
+ * Note that there is no need to call br_i32_from_monty().
51
+ */
52
+ memcpy(t1, x, mlen);
53
+ br_i31_to_monty(t1, m);
54
+ br_i31_zero(x, m[0]);
55
+ x[1] = 1;
56
+ for (k = 0; k < ((uint32_t)elen << 3); k ++) {
57
+ uint32_t ctl;
58
+
59
+ ctl = (e[elen - 1 - (k >> 3)] >> (k & 7)) & 1;
60
+ br_i31_montymul(t2, x, t1, m, m0i);
61
+ CCOPY(ctl, x, t2, mlen);
62
+ br_i31_montymul(t2, t1, t1, m, m0i);
63
+ memcpy(t1, t2, mlen);
64
+ }
65
+ }
@@ -0,0 +1,160 @@
1
+ /*
2
+ * Copyright (c) 2017 Thomas Pornin <pornin@bolet.org>
3
+ *
4
+ * Permission is hereby granted, free of charge, to any person obtaining
5
+ * a copy of this software and associated documentation files (the
6
+ * "Software"), to deal in the Software without restriction, including
7
+ * without limitation the rights to use, copy, modify, merge, publish,
8
+ * distribute, sublicense, and/or sell copies of the Software, and to
9
+ * permit persons to whom the Software is furnished to do so, subject to
10
+ * the following conditions:
11
+ *
12
+ * The above copyright notice and this permission notice shall be
13
+ * included in all copies or substantial portions of the Software.
14
+ *
15
+ * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
16
+ * EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
17
+ * MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND
18
+ * NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS
19
+ * BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN
20
+ * ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
21
+ * CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
22
+ * SOFTWARE.
23
+ */
24
+
25
+ #include "inner.h"
26
+
27
+ /* see inner.h */
28
+ uint32_t
29
+ br_i31_modpow_opt(uint32_t *x,
30
+ const unsigned char *e, size_t elen,
31
+ const uint32_t *m, uint32_t m0i, uint32_t *tmp, size_t twlen)
32
+ {
33
+ size_t mlen, mwlen;
34
+ uint32_t *t1, *t2, *base;
35
+ size_t u, v;
36
+ uint32_t acc;
37
+ int acc_len, win_len;
38
+
39
+ /*
40
+ * Get modulus size.
41
+ */
42
+ mwlen = (m[0] + 63) >> 5;
43
+ mlen = mwlen * sizeof m[0];
44
+ mwlen += (mwlen & 1);
45
+ t1 = tmp;
46
+ t2 = tmp + mwlen;
47
+
48
+ /*
49
+ * Compute possible window size, with a maximum of 5 bits.
50
+ * When the window has size 1 bit, we use a specific code
51
+ * that requires only two temporaries. Otherwise, for a
52
+ * window of k bits, we need 2^k+1 temporaries.
53
+ */
54
+ if (twlen < (mwlen << 1)) {
55
+ return 0;
56
+ }
57
+ for (win_len = 5; win_len > 1; win_len --) {
58
+ if ((((uint32_t)1 << win_len) + 1) * mwlen <= twlen) {
59
+ break;
60
+ }
61
+ }
62
+
63
+ /*
64
+ * Everything is done in Montgomery representation.
65
+ */
66
+ br_i31_to_monty(x, m);
67
+
68
+ /*
69
+ * Compute window contents. If the window has size one bit only,
70
+ * then t2 is set to x; otherwise, t2[0] is left untouched, and
71
+ * t2[k] is set to x^k (for k >= 1).
72
+ */
73
+ if (win_len == 1) {
74
+ memcpy(t2, x, mlen);
75
+ } else {
76
+ memcpy(t2 + mwlen, x, mlen);
77
+ base = t2 + mwlen;
78
+ for (u = 2; u < ((unsigned)1 << win_len); u ++) {
79
+ br_i31_montymul(base + mwlen, base, x, m, m0i);
80
+ base += mwlen;
81
+ }
82
+ }
83
+
84
+ /*
85
+ * We need to set x to 1, in Montgomery representation. This can
86
+ * be done efficiently by setting the high word to 1, then doing
87
+ * one word-sized shift.
88
+ */
89
+ br_i31_zero(x, m[0]);
90
+ x[(m[0] + 31) >> 5] = 1;
91
+ br_i31_muladd_small(x, 0, m);
92
+
93
+ /*
94
+ * We process bits from most to least significant. At each
95
+ * loop iteration, we have acc_len bits in acc.
96
+ */
97
+ acc = 0;
98
+ acc_len = 0;
99
+ while (acc_len > 0 || elen > 0) {
100
+ int i, k;
101
+ uint32_t bits;
102
+
103
+ /*
104
+ * Get the next bits.
105
+ */
106
+ k = win_len;
107
+ if (acc_len < win_len) {
108
+ if (elen > 0) {
109
+ acc = (acc << 8) | *e ++;
110
+ elen --;
111
+ acc_len += 8;
112
+ } else {
113
+ k = acc_len;
114
+ }
115
+ }
116
+ bits = (acc >> (acc_len - k)) & (((uint32_t)1 << k) - 1);
117
+ acc_len -= k;
118
+
119
+ /*
120
+ * We could get exactly k bits. Compute k squarings.
121
+ */
122
+ for (i = 0; i < k; i ++) {
123
+ br_i31_montymul(t1, x, x, m, m0i);
124
+ memcpy(x, t1, mlen);
125
+ }
126
+
127
+ /*
128
+ * Window lookup: we want to set t2 to the window
129
+ * lookup value, assuming the bits are non-zero. If
130
+ * the window length is 1 bit only, then t2 is
131
+ * already set; otherwise, we do a constant-time lookup.
132
+ */
133
+ if (win_len > 1) {
134
+ br_i31_zero(t2, m[0]);
135
+ base = t2 + mwlen;
136
+ for (u = 1; u < ((uint32_t)1 << k); u ++) {
137
+ uint32_t mask;
138
+
139
+ mask = -EQ(u, bits);
140
+ for (v = 1; v < mwlen; v ++) {
141
+ t2[v] |= mask & base[v];
142
+ }
143
+ base += mwlen;
144
+ }
145
+ }
146
+
147
+ /*
148
+ * Multiply with the looked-up value. We keep the
149
+ * product only if the exponent bits are not all-zero.
150
+ */
151
+ br_i31_montymul(t1, x, t2, m, m0i);
152
+ CCOPY(NEQ(bits, 0), x, t1, mlen);
153
+ }
154
+
155
+ /*
156
+ * Convert back from Montgomery representation, and exit.
157
+ */
158
+ br_i31_from_monty(x, m, m0i);
159
+ return 1;
160
+ }