pq_crypto 0.6.3 → 0.6.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/.github/workflows/ci.yml +149 -17
- data/CHANGELOG.md +73 -0
- data/GET_STARTED.md +1 -1
- data/README.md +7 -2
- data/ext/pqcrypto/extconf.rb +264 -24
- data/ext/pqcrypto/pqcrypto_version.h +1 -1
- data/ext/pqcrypto/vendor/.vendored +4 -4
- data/ext/pqcrypto/vendor/mlkem-native/README.md +2 -4
- data/ext/pqcrypto/vendor/mlkem-native/RELEASE.md +98 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/mlkem_native.c +8 -7
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/mlkem_native.h +21 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/mlkem_native_asm.S +45 -44
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/mlkem_native_config.h +36 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/common.h +13 -30
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/compress.c +25 -5
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/compress.h +14 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/context.h +42 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/auto.h +20 -12
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/src/keccak_f1600_x1_scalar_aarch64_asm.S +5 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/src/keccak_f1600_x1_v84a_aarch64_asm.S +5 -2
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/src/keccak_f1600_x2_v84a_aarch64_asm.S +5 -2
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/src/keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm.S +10 -7
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/src/keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm.S +10 -7
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x1_v84a.h +2 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x2_v84a.h +2 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x4_v8a_scalar.h +5 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x4_v8a_v84a_scalar.h +2 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/armv81m/mve.h +5 -19
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/armv81m/src/keccak_f1600_x4_mve.S +8 -5
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/armv81m/src/{state_extract_bytes_x4_mve.S → keccak_f1600_x4_state_extract_bytes_mve.S} +14 -14
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/armv81m/src/{state_xor_bytes_x4_mve.S → keccak_f1600_x4_state_xor_bytes_mve.S} +12 -12
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/x86_64/src/keccak_f1600_x4_avx2_asm.S +36 -2
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/indcpa.c +22 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/indcpa.h +6 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/kem.c +11 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/kem.h +25 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/meta.h +44 -2
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/aarch64_zetas.c +2 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/arith_native_aarch64.h +11 -10
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{intt_aarch64_asm.S → mlkem_intt_aarch64_asm.S} +16 -9
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{ntt_aarch64_asm.S → mlkem_ntt_aarch64_asm.S} +10 -7
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{poly_mulcache_compute_aarch64_asm.S → mlkem_poly_mulcache_compute_aarch64_asm.S} +6 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{poly_reduce_aarch64_asm.S → mlkem_poly_reduce_aarch64_asm.S} +6 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{poly_tobytes_aarch64_asm.S → mlkem_poly_tobytes_aarch64_asm.S} +12 -5
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{poly_tomont_aarch64_asm.S → mlkem_poly_tomont_aarch64_asm.S} +11 -5
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{polyvec_basemul_acc_montgomery_cached_k2_aarch64_asm.S → mlkem_polyvec_basemul_acc_montgomery_cached_k2_aarch64_asm.S} +6 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{polyvec_basemul_acc_montgomery_cached_k3_aarch64_asm.S → mlkem_polyvec_basemul_acc_montgomery_cached_k3_aarch64_asm.S} +6 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{polyvec_basemul_acc_montgomery_cached_k4_aarch64_asm.S → mlkem_polyvec_basemul_acc_montgomery_cached_k4_aarch64_asm.S} +6 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{rej_uniform_aarch64_asm.S → mlkem_rej_uniform_aarch64_asm.S} +17 -17
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/api.h +21 -8
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/meta.h +1 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/ppc64le/meta.h +4 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/ppc64le/src/{intt_ppc_asm.S → mlkem_intt_ppc_asm.S} +29 -5
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/ppc64le/src/{ntt_ppc_asm.S → mlkem_ntt_ppc_asm.S} +22 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/ppc64le/src/{poly_tomont_ppc_asm.S → mlkem_poly_tomont_ppc_asm.S} +25 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/ppc64le/src/{reduce_ppc_asm.S → mlkem_reduce_ppc_asm.S} +22 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/riscv64/meta.h +4 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/riscv64/src/arith_native_riscv64.h +4 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/riscv64/src/rv64v_poly.c +6 -2
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/meta.h +25 -5
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/arith_native_x86_64.h +20 -20
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/compress_consts.c +14 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/compress_consts.h +8 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{intt_avx2_asm.S → mlkem_intt_avx2_asm.S} +27 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{ntt_avx2_asm.S → mlkem_ntt_avx2_asm.S} +23 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{nttfrombytes_avx2_asm.S → mlkem_nttfrombytes_avx2_asm.S} +27 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{ntttobytes_avx2_asm.S → mlkem_ntttobytes_avx2_asm.S} +27 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{nttunpack_avx2_asm.S → mlkem_nttunpack_avx2_asm.S} +17 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{poly_compress_d10_avx2_asm.S → mlkem_poly_compress_d10_avx2_asm.S} +34 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{poly_compress_d11_avx2_asm.S → mlkem_poly_compress_d11_avx2_asm.S} +33 -2
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{poly_compress_d4_avx2_asm.S → mlkem_poly_compress_d4_avx2_asm.S} +34 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{poly_compress_d5_avx2_asm.S → mlkem_poly_compress_d5_avx2_asm.S} +33 -2
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{poly_decompress_d10_avx2_asm.S → mlkem_poly_decompress_d10_avx2_asm.S} +32 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{poly_decompress_d11_avx2_asm.S → mlkem_poly_decompress_d11_avx2_asm.S} +32 -2
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{poly_decompress_d4_avx2_asm.S → mlkem_poly_decompress_d4_avx2_asm.S} +32 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{poly_decompress_d5_avx2_asm.S → mlkem_poly_decompress_d5_avx2_asm.S} +32 -2
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{poly_mulcache_compute_avx2_asm.S → mlkem_poly_mulcache_compute_avx2_asm.S} +29 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{polyvec_basemul_acc_montgomery_cached_k2_avx2_asm.S → mlkem_polyvec_basemul_acc_montgomery_cached_k2_avx2_asm.S} +35 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{polyvec_basemul_acc_montgomery_cached_k3_avx2_asm.S → mlkem_polyvec_basemul_acc_montgomery_cached_k3_avx2_asm.S} +35 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{polyvec_basemul_acc_montgomery_cached_k4_avx2_asm.S → mlkem_polyvec_basemul_acc_montgomery_cached_k4_avx2_asm.S} +35 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{reduce_avx2_asm.S → mlkem_reduce_avx2_asm.S} +17 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{rej_uniform_avx2_asm.S → mlkem_rej_uniform_avx2_asm.S} +40 -7
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{tomont_avx2_asm.S → mlkem_tomont_avx2_asm.S} +20 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/poly.c +7 -2
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/poly.h +6 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/poly_k.c +25 -4
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/poly_k.h +33 -4
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/sampling.c +4 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/sampling.h +4 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/sys.h +19 -1
- data/lib/pq_crypto/version.rb +1 -1
- data/script/vendor_libs.rb +3 -3
- metadata +41 -43
|
@@ -8,6 +8,9 @@
|
|
|
8
8
|
Description: Convert polynomial to Montgomery domain
|
|
9
9
|
Signature: void mlk_poly_tomont_aarch64_asm(int16_t p[256])
|
|
10
10
|
ABI:
|
|
11
|
+
Architecture: aarch64
|
|
12
|
+
CallingConvention: AAPCS64
|
|
13
|
+
Features: [NEON]
|
|
11
14
|
x0:
|
|
12
15
|
type: buffer
|
|
13
16
|
size_bytes: 512
|
|
@@ -19,11 +22,13 @@
|
|
|
19
22
|
*/
|
|
20
23
|
|
|
21
24
|
#include "../../../common.h"
|
|
22
|
-
#if defined(MLK_ARITH_BACKEND_AARCH64) &&
|
|
25
|
+
#if defined(MLK_ARITH_BACKEND_AARCH64) && \
|
|
26
|
+
!defined(MLK_CONFIG_MULTILEVEL_NO_SHARED) && \
|
|
27
|
+
!defined(MLK_CONFIG_NO_KEYPAIR_API)
|
|
23
28
|
|
|
24
29
|
/*
|
|
25
30
|
* WARNING: This file is auto-derived from the mlkem-native source file
|
|
26
|
-
* dev/aarch64_opt/src/
|
|
31
|
+
* dev/aarch64_opt/src/mlkem_poly_tomont_aarch64_asm.S using scripts/simpasm. Do not modify it directly.
|
|
27
32
|
*/
|
|
28
33
|
|
|
29
34
|
.text
|
|
@@ -57,7 +62,7 @@ MLK_ASM_FN_SYMBOL(poly_tomont_aarch64_asm)
|
|
|
57
62
|
mls v18.8h, v26.8h, v4.h[0]
|
|
58
63
|
sub x1, x1, #0x1
|
|
59
64
|
|
|
60
|
-
|
|
65
|
+
Lmlk_poly_tomont_loop:
|
|
61
66
|
ldr q19, [x0, #0x10]
|
|
62
67
|
mul v26.8h, v16.8h, v2.8h
|
|
63
68
|
ldr q23, [x0, #0x20]
|
|
@@ -79,7 +84,7 @@ Lpoly_tomont_loop:
|
|
|
79
84
|
mls v18.8h, v24.8h, v4.h[0]
|
|
80
85
|
stur q26, [x0, #-0x40]
|
|
81
86
|
sub x1, x1, #0x1
|
|
82
|
-
cbnz x1,
|
|
87
|
+
cbnz x1, Lmlk_poly_tomont_loop
|
|
83
88
|
mul v16.8h, v16.8h, v2.8h
|
|
84
89
|
stur q18, [x0, #-0x20]
|
|
85
90
|
mls v16.8h, v29.8h, v4.h[0]
|
|
@@ -89,7 +94,8 @@ Lpoly_tomont_loop:
|
|
|
89
94
|
|
|
90
95
|
MLK_ASM_FN_SIZE(poly_tomont_aarch64_asm)
|
|
91
96
|
|
|
92
|
-
#endif /* MLK_ARITH_BACKEND_AARCH64 && !MLK_CONFIG_MULTILEVEL_NO_SHARED
|
|
97
|
+
#endif /* MLK_ARITH_BACKEND_AARCH64 && !MLK_CONFIG_MULTILEVEL_NO_SHARED && \
|
|
98
|
+
!MLK_CONFIG_NO_KEYPAIR_API */
|
|
93
99
|
|
|
94
100
|
#if defined(__ELF__)
|
|
95
101
|
.section .note.GNU-stack,"",%progbits
|
|
@@ -17,6 +17,9 @@
|
|
|
17
17
|
Description: Re-implementation of asymmetric base multiplication following @[NeonNTT] for k=2
|
|
18
18
|
Signature: void mlk_polyvec_basemul_acc_montgomery_cached_k2_aarch64_asm(int16_t r[256], const int16_t a[512], const int16_t b[512], const int16_t b_cache[256])
|
|
19
19
|
ABI:
|
|
20
|
+
Architecture: aarch64
|
|
21
|
+
CallingConvention: AAPCS64
|
|
22
|
+
Features: [NEON]
|
|
20
23
|
x0:
|
|
21
24
|
type: buffer
|
|
22
25
|
size_bytes: 512
|
|
@@ -53,7 +56,7 @@
|
|
|
53
56
|
|
|
54
57
|
/*
|
|
55
58
|
* WARNING: This file is auto-derived from the mlkem-native source file
|
|
56
|
-
* dev/aarch64_opt/src/
|
|
59
|
+
* dev/aarch64_opt/src/mlkem_polyvec_basemul_acc_montgomery_cached_k2_aarch64_asm.S using scripts/simpasm. Do not modify it directly.
|
|
57
60
|
*/
|
|
58
61
|
|
|
59
62
|
.text
|
|
@@ -156,7 +159,7 @@ MLK_ASM_FN_SYMBOL(polyvec_basemul_acc_montgomery_cached_k2_aarch64_asm)
|
|
|
156
159
|
uzp2 v19.8h, v11.8h, v4.8h
|
|
157
160
|
sub x13, x13, #0x2
|
|
158
161
|
|
|
159
|
-
|
|
162
|
+
Lmlk_polyvec_basemul_acc_montgomery_cached_k2_loop_start:
|
|
160
163
|
smlal v18.4s, v16.4h, v17.4h
|
|
161
164
|
ldr q7, [x4], #0x20
|
|
162
165
|
ldr q10, [x2, #0x10]
|
|
@@ -206,7 +209,7 @@ Lpolyvec_basemul_acc_montgomery_cached_k2_loop_start:
|
|
|
206
209
|
smull2 v23.4s, v1.8h, v24.8h
|
|
207
210
|
smull v26.4s, v1.4h, v24.4h
|
|
208
211
|
subs x13, x13, #0x1
|
|
209
|
-
cbnz x13,
|
|
212
|
+
cbnz x13, Lmlk_polyvec_basemul_acc_montgomery_cached_k2_loop_start
|
|
210
213
|
smlal v26.4s, v6.4h, v20.4h
|
|
211
214
|
smlal2 v23.4s, v6.8h, v20.8h
|
|
212
215
|
smlal v26.4s, v16.4h, v30.4h
|
|
@@ -17,6 +17,9 @@
|
|
|
17
17
|
Description: Re-implementation of asymmetric base multiplication following @[NeonNTT] for k=3
|
|
18
18
|
Signature: void mlk_polyvec_basemul_acc_montgomery_cached_k3_aarch64_asm(int16_t r[256], const int16_t a[768], const int16_t b[768], const int16_t b_cache[384])
|
|
19
19
|
ABI:
|
|
20
|
+
Architecture: aarch64
|
|
21
|
+
CallingConvention: AAPCS64
|
|
22
|
+
Features: [NEON]
|
|
20
23
|
x0:
|
|
21
24
|
type: buffer
|
|
22
25
|
size_bytes: 512
|
|
@@ -53,7 +56,7 @@
|
|
|
53
56
|
|
|
54
57
|
/*
|
|
55
58
|
* WARNING: This file is auto-derived from the mlkem-native source file
|
|
56
|
-
* dev/aarch64_opt/src/
|
|
59
|
+
* dev/aarch64_opt/src/mlkem_polyvec_basemul_acc_montgomery_cached_k3_aarch64_asm.S using scripts/simpasm. Do not modify it directly.
|
|
57
60
|
*/
|
|
58
61
|
|
|
59
62
|
.text
|
|
@@ -178,7 +181,7 @@ MLK_ASM_FN_SYMBOL(polyvec_basemul_acc_montgomery_cached_k3_aarch64_asm)
|
|
|
178
181
|
mul v14.8h, v14.8h, v2.8h
|
|
179
182
|
sub x13, x13, #0x2
|
|
180
183
|
|
|
181
|
-
|
|
184
|
+
Lmlk_polyvec_basemul_acc_montgomery_cached_k3_loop_start:
|
|
182
185
|
uzp1 v6.8h, v27.8h, v11.8h
|
|
183
186
|
smlal v26.4s, v29.4h, v24.4h
|
|
184
187
|
uzp2 v16.8h, v25.8h, v3.8h
|
|
@@ -245,7 +248,7 @@ Lpolyvec_basemul_acc_montgomery_cached_k3_loop_start:
|
|
|
245
248
|
stur q10, [x0, #-0x10]
|
|
246
249
|
uzp2 v17.8h, v27.8h, v11.8h
|
|
247
250
|
subs x13, x13, #0x1
|
|
248
|
-
cbnz x13,
|
|
251
|
+
cbnz x13, Lmlk_polyvec_basemul_acc_montgomery_cached_k3_loop_start
|
|
249
252
|
uzp2 v3.8h, v25.8h, v3.8h
|
|
250
253
|
smull2 v16.4s, v1.8h, v20.8h
|
|
251
254
|
smull v25.4s, v1.4h, v20.4h
|
|
@@ -17,6 +17,9 @@
|
|
|
17
17
|
Description: Re-implementation of asymmetric base multiplication following @[NeonNTT] for k=4
|
|
18
18
|
Signature: void mlk_polyvec_basemul_acc_montgomery_cached_k4_aarch64_asm(int16_t r[256], const int16_t a[1024], const int16_t b[1024], const int16_t b_cache[512])
|
|
19
19
|
ABI:
|
|
20
|
+
Architecture: aarch64
|
|
21
|
+
CallingConvention: AAPCS64
|
|
22
|
+
Features: [NEON]
|
|
20
23
|
x0:
|
|
21
24
|
type: buffer
|
|
22
25
|
size_bytes: 512
|
|
@@ -53,7 +56,7 @@
|
|
|
53
56
|
|
|
54
57
|
/*
|
|
55
58
|
* WARNING: This file is auto-derived from the mlkem-native source file
|
|
56
|
-
* dev/aarch64_opt/src/
|
|
59
|
+
* dev/aarch64_opt/src/mlkem_polyvec_basemul_acc_montgomery_cached_k4_aarch64_asm.S using scripts/simpasm. Do not modify it directly.
|
|
57
60
|
*/
|
|
58
61
|
|
|
59
62
|
.text
|
|
@@ -173,7 +176,7 @@ MLK_ASM_FN_SYMBOL(polyvec_basemul_acc_montgomery_cached_k4_aarch64_asm)
|
|
|
173
176
|
mul v29.8h, v28.8h, v2.8h
|
|
174
177
|
sub x13, x13, #0x2
|
|
175
178
|
|
|
176
|
-
|
|
179
|
+
Lmlk_polyvec_basemul_acc_montgomery_cached_k4_loop_start:
|
|
177
180
|
smlal2 v23.4s, v30.8h, v21.8h
|
|
178
181
|
ldr q11, [x1], #0x20
|
|
179
182
|
uzp2 v15.8h, v5.8h, v22.8h
|
|
@@ -257,7 +260,7 @@ Lpolyvec_basemul_acc_montgomery_cached_k4_loop_start:
|
|
|
257
260
|
smlal2 v23.4s, v25.8h, v10.8h
|
|
258
261
|
uzp1 v14.8h, v5.8h, v22.8h
|
|
259
262
|
subs x13, x13, #0x1
|
|
260
|
-
cbnz x13,
|
|
263
|
+
cbnz x13, Lmlk_polyvec_basemul_acc_montgomery_cached_k4_loop_start
|
|
261
264
|
smlal v18.4s, v29.4h, v0.4h
|
|
262
265
|
ldr q11, [x1], #0x20
|
|
263
266
|
uzp2 v28.8h, v5.8h, v22.8h
|
|
@@ -8,6 +8,9 @@
|
|
|
8
8
|
Description: Run rejection sampling on uniform random bytes to generate uniform random integers mod q
|
|
9
9
|
Signature: uint64_t mlk_rej_uniform_aarch64_asm(int16_t r[256], const uint8_t *buf, unsigned buflen, const uint8_t table[4096])
|
|
10
10
|
ABI:
|
|
11
|
+
Architecture: aarch64
|
|
12
|
+
CallingConvention: AAPCS64
|
|
13
|
+
Features: [NEON]
|
|
11
14
|
x0:
|
|
12
15
|
type: buffer
|
|
13
16
|
size_bytes: 512
|
|
@@ -42,7 +45,7 @@
|
|
|
42
45
|
|
|
43
46
|
/*
|
|
44
47
|
* WARNING: This file is auto-derived from the mlkem-native source file
|
|
45
|
-
* dev/aarch64_opt/src/
|
|
48
|
+
* dev/aarch64_opt/src/mlkem_rej_uniform_aarch64_asm.S using scripts/simpasm. Do not modify it directly.
|
|
46
49
|
*/
|
|
47
50
|
|
|
48
51
|
.text
|
|
@@ -70,23 +73,23 @@ MLK_ASM_FN_SYMBOL(rej_uniform_aarch64_asm)
|
|
|
70
73
|
mov x11, #0x0 // =0
|
|
71
74
|
eor v16.16b, v16.16b, v16.16b
|
|
72
75
|
|
|
73
|
-
|
|
76
|
+
Lmlk_rej_uniform_initial_zero:
|
|
74
77
|
str q16, [x7], #0x40
|
|
75
78
|
stur q16, [x7, #-0x30]
|
|
76
79
|
stur q16, [x7, #-0x20]
|
|
77
80
|
stur q16, [x7, #-0x10]
|
|
78
81
|
add x11, x11, #0x20
|
|
79
82
|
cmp x11, #0x100
|
|
80
|
-
b.lt
|
|
83
|
+
b.lt Lmlk_rej_uniform_initial_zero
|
|
81
84
|
mov x7, x8
|
|
82
85
|
mov x9, #0x0 // =0
|
|
83
86
|
mov x4, #0x100 // =256
|
|
84
87
|
cmp x2, #0x30
|
|
85
|
-
b.lo
|
|
88
|
+
b.lo Lmlk_rej_uniform_loop48_end
|
|
86
89
|
|
|
87
|
-
|
|
90
|
+
Lmlk_rej_uniform_loop48:
|
|
88
91
|
cmp x9, x4
|
|
89
|
-
b.hs
|
|
92
|
+
b.hs Lmlk_rej_uniform_memory_copy
|
|
90
93
|
sub x2, x2, #0x30
|
|
91
94
|
ld3 { v0.16b, v1.16b, v2.16b }, [x1], #48
|
|
92
95
|
zip1 v4.16b, v0.16b, v1.16b
|
|
@@ -150,14 +153,13 @@ Lrej_uniform_loop48:
|
|
|
150
153
|
add x9, x9, x12
|
|
151
154
|
add x9, x9, x14
|
|
152
155
|
cmp x2, #0x30
|
|
153
|
-
b.hs
|
|
156
|
+
b.hs Lmlk_rej_uniform_loop48
|
|
154
157
|
|
|
155
|
-
|
|
158
|
+
Lmlk_rej_uniform_loop48_end:
|
|
156
159
|
cmp x9, x4
|
|
157
|
-
b.hs
|
|
160
|
+
b.hs Lmlk_rej_uniform_memory_copy
|
|
158
161
|
cmp x2, #0x18
|
|
159
|
-
b.lo
|
|
160
|
-
sub x2, x2, #0x18
|
|
162
|
+
b.lo Lmlk_rej_uniform_memory_copy
|
|
161
163
|
ld3 { v0.8b, v1.8b, v2.8b }, [x1], #24
|
|
162
164
|
zip1 v4.16b, v0.16b, v1.16b
|
|
163
165
|
zip1 v5.16b, v1.16b, v2.16b
|
|
@@ -186,17 +188,16 @@ Lrej_uniform_loop48_end:
|
|
|
186
188
|
st1 { v16.8h }, [x7]
|
|
187
189
|
add x7, x7, x12, lsl #1
|
|
188
190
|
st1 { v17.8h }, [x7]
|
|
189
|
-
add x7, x7, x13, lsl #1
|
|
190
191
|
add x9, x9, x12
|
|
191
192
|
add x9, x9, x13
|
|
192
193
|
|
|
193
|
-
|
|
194
|
+
Lmlk_rej_uniform_memory_copy:
|
|
194
195
|
cmp x9, x4
|
|
195
196
|
csel x9, x9, x4, lo
|
|
196
197
|
mov x11, #0x0 // =0
|
|
197
198
|
mov x7, x8
|
|
198
199
|
|
|
199
|
-
|
|
200
|
+
Lmlk_rej_uniform_final_copy:
|
|
200
201
|
ldr q16, [x7], #0x40
|
|
201
202
|
ldur q17, [x7, #-0x30]
|
|
202
203
|
ldur q18, [x7, #-0x20]
|
|
@@ -207,11 +208,10 @@ Lrej_uniform_final_copy:
|
|
|
207
208
|
stur q19, [x0, #-0x10]
|
|
208
209
|
add x11, x11, #0x20
|
|
209
210
|
cmp x11, #0x100
|
|
210
|
-
b.lt
|
|
211
|
+
b.lt Lmlk_rej_uniform_final_copy
|
|
211
212
|
mov x0, x9
|
|
212
|
-
b Lrej_uniform_return
|
|
213
213
|
|
|
214
|
-
|
|
214
|
+
Lmlk_rej_uniform_return:
|
|
215
215
|
add sp, sp, #0x240
|
|
216
216
|
.cfi_adjust_cfa_offset -0x240
|
|
217
217
|
ret
|
|
@@ -144,7 +144,8 @@ __contract__(
|
|
|
144
144
|
ensures(array_bound(p, 0, MLKEM_N, 0, MLKEM_Q)));
|
|
145
145
|
#endif /* MLK_USE_NATIVE_NTT_CUSTOM_ORDER */
|
|
146
146
|
|
|
147
|
-
#if defined(MLK_USE_NATIVE_INTT)
|
|
147
|
+
#if defined(MLK_USE_NATIVE_INTT) && \
|
|
148
|
+
(!defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API))
|
|
148
149
|
/**
|
|
149
150
|
* Compute the inverse negacyclic number-theoretic transform (NTT) of a
|
|
150
151
|
* polynomial in place.
|
|
@@ -168,7 +169,8 @@ __contract__(
|
|
|
168
169
|
ensures((return_value == MLK_NATIVE_FUNC_SUCCESS) ==> array_abs_bound(p, 0, MLKEM_N, MLK_INVNTT_BOUND))
|
|
169
170
|
ensures((return_value == MLK_NATIVE_FUNC_FALLBACK) ==> array_unchanged(p, MLKEM_N))
|
|
170
171
|
);
|
|
171
|
-
#endif /* MLK_USE_NATIVE_INTT
|
|
172
|
+
#endif /* MLK_USE_NATIVE_INTT && (!MLK_CONFIG_NO_ENCAPS_API || \
|
|
173
|
+
!MLK_CONFIG_NO_DECAPS_API) */
|
|
172
174
|
|
|
173
175
|
#if defined(MLK_USE_NATIVE_POLY_REDUCE)
|
|
174
176
|
/**
|
|
@@ -191,7 +193,7 @@ __contract__(
|
|
|
191
193
|
);
|
|
192
194
|
#endif /* MLK_USE_NATIVE_POLY_REDUCE */
|
|
193
195
|
|
|
194
|
-
#if defined(MLK_USE_NATIVE_POLY_TOMONT)
|
|
196
|
+
#if defined(MLK_USE_NATIVE_POLY_TOMONT) && !defined(MLK_CONFIG_NO_KEYPAIR_API)
|
|
195
197
|
/**
|
|
196
198
|
* In-place conversion of all coefficients of a polynomial from the normal
|
|
197
199
|
* domain to the Montgomery domain.
|
|
@@ -210,7 +212,7 @@ __contract__(
|
|
|
210
212
|
ensures((return_value == MLK_NATIVE_FUNC_SUCCESS) ==> array_abs_bound(p, 0, MLKEM_N, MLKEM_Q))
|
|
211
213
|
ensures((return_value == MLK_NATIVE_FUNC_FALLBACK) ==> array_unchanged(p, MLKEM_N))
|
|
212
214
|
);
|
|
213
|
-
#endif /* MLK_USE_NATIVE_POLY_TOMONT */
|
|
215
|
+
#endif /* MLK_USE_NATIVE_POLY_TOMONT && !MLK_CONFIG_NO_KEYPAIR_API */
|
|
214
216
|
|
|
215
217
|
#if defined(MLK_USE_NATIVE_POLY_MULCACHE_COMPUTE)
|
|
216
218
|
/**
|
|
@@ -340,7 +342,9 @@ __contract__(
|
|
|
340
342
|
#endif /* MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 4 */
|
|
341
343
|
#endif /* MLK_USE_NATIVE_POLYVEC_BASEMUL_ACC_MONTGOMERY_CACHED */
|
|
342
344
|
|
|
343
|
-
#if defined(MLK_USE_NATIVE_POLY_TOBYTES)
|
|
345
|
+
#if defined(MLK_USE_NATIVE_POLY_TOBYTES) && \
|
|
346
|
+
(!defined(MLK_CONFIG_NO_KEYPAIR_API) || \
|
|
347
|
+
!defined(MLK_CONFIG_NO_ENCAPS_API))
|
|
344
348
|
/**
|
|
345
349
|
* Serialization of a polynomial with unsigned canonical coefficients.
|
|
346
350
|
*
|
|
@@ -362,9 +366,11 @@ __contract__(
|
|
|
362
366
|
assigns(memory_slice(r, MLKEM_POLYBYTES))
|
|
363
367
|
ensures(return_value == MLK_NATIVE_FUNC_SUCCESS || return_value == MLK_NATIVE_FUNC_FALLBACK)
|
|
364
368
|
);
|
|
365
|
-
#endif /* MLK_USE_NATIVE_POLY_TOBYTES
|
|
369
|
+
#endif /* MLK_USE_NATIVE_POLY_TOBYTES && (!MLK_CONFIG_NO_KEYPAIR_API || \
|
|
370
|
+
!MLK_CONFIG_NO_ENCAPS_API) */
|
|
366
371
|
|
|
367
|
-
#if defined(MLK_USE_NATIVE_POLY_FROMBYTES)
|
|
372
|
+
#if defined(MLK_USE_NATIVE_POLY_FROMBYTES) && \
|
|
373
|
+
(!defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API))
|
|
368
374
|
/**
|
|
369
375
|
* Deserialization of a polynomial.
|
|
370
376
|
*
|
|
@@ -386,7 +392,8 @@ __contract__(
|
|
|
386
392
|
ensures(return_value == MLK_NATIVE_FUNC_SUCCESS || return_value == MLK_NATIVE_FUNC_FALLBACK)
|
|
387
393
|
ensures((return_value == MLK_NATIVE_FUNC_SUCCESS) ==> array_bound(a, 0, MLKEM_N, 0, MLKEM_UINT12_LIMIT))
|
|
388
394
|
);
|
|
389
|
-
#endif /* MLK_USE_NATIVE_POLY_FROMBYTES
|
|
395
|
+
#endif /* MLK_USE_NATIVE_POLY_FROMBYTES && (!MLK_CONFIG_NO_ENCAPS_API || \
|
|
396
|
+
!MLK_CONFIG_NO_DECAPS_API) */
|
|
390
397
|
|
|
391
398
|
#if defined(MLK_USE_NATIVE_REJ_UNIFORM)
|
|
392
399
|
/**
|
|
@@ -419,6 +426,7 @@ __contract__(
|
|
|
419
426
|
);
|
|
420
427
|
#endif /* MLK_USE_NATIVE_REJ_UNIFORM */
|
|
421
428
|
|
|
429
|
+
#if !defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API)
|
|
422
430
|
#if defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || (MLKEM_K == 2 || MLKEM_K == 3)
|
|
423
431
|
#if defined(MLK_USE_NATIVE_POLY_COMPRESS_D4)
|
|
424
432
|
/**
|
|
@@ -470,6 +478,7 @@ __contract__(
|
|
|
470
478
|
ensures(return_value == MLK_NATIVE_FUNC_SUCCESS || return_value == MLK_NATIVE_FUNC_FALLBACK));
|
|
471
479
|
#endif /* MLK_USE_NATIVE_POLY_COMPRESS_D10 */
|
|
472
480
|
|
|
481
|
+
#if !defined(MLK_CONFIG_NO_DECAPS_API)
|
|
473
482
|
#if defined(MLK_USE_NATIVE_POLY_DECOMPRESS_D4)
|
|
474
483
|
/**
|
|
475
484
|
* De-serialization and subsequent decompression (4 bits) of a polynomial;
|
|
@@ -525,6 +534,7 @@ __contract__(
|
|
|
525
534
|
ensures(return_value == MLK_NATIVE_FUNC_SUCCESS || return_value == MLK_NATIVE_FUNC_FALLBACK)
|
|
526
535
|
ensures((return_value == MLK_NATIVE_FUNC_SUCCESS) ==> array_bound(r, 0, MLKEM_N, 0, MLKEM_Q)));
|
|
527
536
|
#endif /* MLK_USE_NATIVE_POLY_DECOMPRESS_D10 */
|
|
537
|
+
#endif /* !MLK_CONFIG_NO_DECAPS_API */
|
|
528
538
|
#endif /* MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 2 || MLKEM_K == 3 */
|
|
529
539
|
|
|
530
540
|
#if defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_K == 4
|
|
@@ -578,6 +588,7 @@ __contract__(
|
|
|
578
588
|
ensures(return_value == MLK_NATIVE_FUNC_SUCCESS || return_value == MLK_NATIVE_FUNC_FALLBACK));
|
|
579
589
|
#endif /* MLK_USE_NATIVE_POLY_COMPRESS_D11 */
|
|
580
590
|
|
|
591
|
+
#if !defined(MLK_CONFIG_NO_DECAPS_API)
|
|
581
592
|
#if defined(MLK_USE_NATIVE_POLY_DECOMPRESS_D5)
|
|
582
593
|
/**
|
|
583
594
|
* De-serialization and subsequent decompression (5 bits) of a polynomial;
|
|
@@ -633,6 +644,8 @@ __contract__(
|
|
|
633
644
|
ensures(return_value == MLK_NATIVE_FUNC_SUCCESS || return_value == MLK_NATIVE_FUNC_FALLBACK)
|
|
634
645
|
ensures((return_value == MLK_NATIVE_FUNC_SUCCESS) ==> array_bound(r, 0, MLKEM_N, 0, MLKEM_Q)));
|
|
635
646
|
#endif /* MLK_USE_NATIVE_POLY_DECOMPRESS_D11 */
|
|
647
|
+
#endif /* !MLK_CONFIG_NO_DECAPS_API */
|
|
636
648
|
#endif /* MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 4 */
|
|
649
|
+
#endif /* !MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API */
|
|
637
650
|
|
|
638
651
|
#endif /* !MLK_NATIVE_API_H */
|
|
@@ -37,6 +37,7 @@ static MLK_INLINE int mlk_ntt_native(int16_t data[MLKEM_N])
|
|
|
37
37
|
#endif
|
|
38
38
|
}
|
|
39
39
|
|
|
40
|
+
#if !defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API)
|
|
40
41
|
MLK_MUST_CHECK_RETURN_VALUE
|
|
41
42
|
static MLK_INLINE int mlk_intt_native(int16_t data[MLKEM_N])
|
|
42
43
|
{
|
|
@@ -48,6 +49,7 @@ static MLK_INLINE int mlk_intt_native(int16_t data[MLKEM_N])
|
|
|
48
49
|
return MLK_NATIVE_FUNC_FALLBACK;
|
|
49
50
|
#endif
|
|
50
51
|
}
|
|
52
|
+
#endif /* !MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API */
|
|
51
53
|
|
|
52
54
|
MLK_MUST_CHECK_RETURN_VALUE
|
|
53
55
|
static MLK_INLINE int mlk_poly_reduce_native(int16_t data[MLKEM_N])
|
|
@@ -61,6 +63,7 @@ static MLK_INLINE int mlk_poly_reduce_native(int16_t data[MLKEM_N])
|
|
|
61
63
|
#endif
|
|
62
64
|
}
|
|
63
65
|
|
|
66
|
+
#if !defined(MLK_CONFIG_NO_KEYPAIR_API)
|
|
64
67
|
MLK_MUST_CHECK_RETURN_VALUE
|
|
65
68
|
static MLK_INLINE int mlk_poly_tomont_native(int16_t data[MLKEM_N])
|
|
66
69
|
{
|
|
@@ -72,6 +75,7 @@ static MLK_INLINE int mlk_poly_tomont_native(int16_t data[MLKEM_N])
|
|
|
72
75
|
return MLK_NATIVE_FUNC_FALLBACK;
|
|
73
76
|
#endif
|
|
74
77
|
}
|
|
78
|
+
#endif /* !MLK_CONFIG_NO_KEYPAIR_API */
|
|
75
79
|
#endif /* !__ASSEMBLER__ */
|
|
76
80
|
|
|
77
81
|
#endif /* !MLK_NATIVE_PPC64LE_META_H */
|
|
@@ -8,11 +8,34 @@
|
|
|
8
8
|
|
|
9
9
|
#include "../../../common.h"
|
|
10
10
|
#if defined(MLK_ARITH_BACKEND_PPC64LE_DEFAULT) && \
|
|
11
|
-
!defined(MLK_CONFIG_MULTILEVEL_NO_SHARED) && defined(__POWER8_VECTOR__)
|
|
11
|
+
!defined(MLK_CONFIG_MULTILEVEL_NO_SHARED) && defined(__POWER8_VECTOR__) && \
|
|
12
|
+
(!defined(MLK_CONFIG_NO_ENCAPS_API) || \
|
|
13
|
+
!defined(MLK_CONFIG_NO_DECAPS_API))
|
|
14
|
+
/*yaml
|
|
15
|
+
Name: intt_ppc_asm
|
|
16
|
+
Description: PowerPC64 VSX inverse NTT
|
|
17
|
+
Signature: void mlk_intt_ppc_asm(int16_t *r, const int16_t *qdata)
|
|
18
|
+
ABI:
|
|
19
|
+
Architecture: powerpc64le
|
|
20
|
+
CallingConvention: ELFv2
|
|
21
|
+
Features: [VSX]
|
|
22
|
+
r3:
|
|
23
|
+
type: buffer
|
|
24
|
+
size_bytes: 512
|
|
25
|
+
permissions: read/write
|
|
26
|
+
c_parameter: int16_t *r
|
|
27
|
+
description: Input/output polynomial (256 x int16_t)
|
|
28
|
+
r4:
|
|
29
|
+
type: buffer
|
|
30
|
+
size_bytes: 4144
|
|
31
|
+
permissions: read-only
|
|
32
|
+
c_parameter: const int16_t *qdata
|
|
33
|
+
description: Precomputed constants (2072 x int16_t)
|
|
34
|
+
*/
|
|
12
35
|
|
|
13
36
|
/*
|
|
14
37
|
* WARNING: This file is auto-derived from the mlkem-native source file
|
|
15
|
-
* dev/ppc64le/src/
|
|
38
|
+
* dev/ppc64le/src/mlkem_intt_ppc_asm.S using scripts/simpasm. Do not modify it directly.
|
|
16
39
|
*/
|
|
17
40
|
|
|
18
41
|
.text
|
|
@@ -84,7 +107,7 @@ MLK_ASM_FN_SYMBOL(intt_ppc_asm)
|
|
|
84
107
|
mtctr 8
|
|
85
108
|
xxlor 37, 0, 0
|
|
86
109
|
|
|
87
|
-
|
|
110
|
+
mlk_intt_ppc_asm_Loopf:
|
|
88
111
|
lxvd2x 57, 0, 3
|
|
89
112
|
lxvd2x 58, 10, 3
|
|
90
113
|
lxvd2x 62, 11, 3
|
|
@@ -129,7 +152,7 @@ intt_ppc_asm_Loopf:
|
|
|
129
152
|
stxvd2x 55, 17, 3
|
|
130
153
|
stxvd2x 60, 18, 3
|
|
131
154
|
addi 3, 3, 128
|
|
132
|
-
bdnz
|
|
155
|
+
bdnz mlk_intt_ppc_asm_Loopf
|
|
133
156
|
addi 3, 3, -512
|
|
134
157
|
nop
|
|
135
158
|
nop
|
|
@@ -3215,7 +3238,8 @@ intt_ppc_asm_Loopf:
|
|
|
3215
3238
|
MLK_ASM_FN_SIZE(intt_ppc_asm)
|
|
3216
3239
|
|
|
3217
3240
|
#endif /* MLK_ARITH_BACKEND_PPC64LE_DEFAULT && !MLK_CONFIG_MULTILEVEL_NO_SHARED \
|
|
3218
|
-
&& __POWER8_VECTOR__
|
|
3241
|
+
&& __POWER8_VECTOR__ && (!MLK_CONFIG_NO_ENCAPS_API || \
|
|
3242
|
+
!MLK_CONFIG_NO_DECAPS_API) */
|
|
3219
3243
|
|
|
3220
3244
|
#if defined(__ELF__)
|
|
3221
3245
|
.section .note.GNU-stack,"",%progbits
|
|
@@ -9,10 +9,31 @@
|
|
|
9
9
|
#include "../../../common.h"
|
|
10
10
|
#if defined(MLK_ARITH_BACKEND_PPC64LE_DEFAULT) && \
|
|
11
11
|
!defined(MLK_CONFIG_MULTILEVEL_NO_SHARED) && defined(__POWER8_VECTOR__)
|
|
12
|
+
/*yaml
|
|
13
|
+
Name: ntt_ppc_asm
|
|
14
|
+
Description: PowerPC64 VSX forward NTT
|
|
15
|
+
Signature: void mlk_ntt_ppc_asm(int16_t *r, const int16_t *qdata)
|
|
16
|
+
ABI:
|
|
17
|
+
Architecture: powerpc64le
|
|
18
|
+
CallingConvention: ELFv2
|
|
19
|
+
Features: [VSX]
|
|
20
|
+
r3:
|
|
21
|
+
type: buffer
|
|
22
|
+
size_bytes: 512
|
|
23
|
+
permissions: read/write
|
|
24
|
+
c_parameter: int16_t *r
|
|
25
|
+
description: Input/output polynomial (256 x int16_t)
|
|
26
|
+
r4:
|
|
27
|
+
type: buffer
|
|
28
|
+
size_bytes: 4144
|
|
29
|
+
permissions: read-only
|
|
30
|
+
c_parameter: const int16_t *qdata
|
|
31
|
+
description: Precomputed constants (2072 x int16_t)
|
|
32
|
+
*/
|
|
12
33
|
|
|
13
34
|
/*
|
|
14
35
|
* WARNING: This file is auto-derived from the mlkem-native source file
|
|
15
|
-
* dev/ppc64le/src/
|
|
36
|
+
* dev/ppc64le/src/mlkem_ntt_ppc_asm.S using scripts/simpasm. Do not modify it directly.
|
|
16
37
|
*/
|
|
17
38
|
|
|
18
39
|
.text
|
|
@@ -15,11 +15,33 @@
|
|
|
15
15
|
|
|
16
16
|
#include "../../../common.h"
|
|
17
17
|
#if defined(MLK_ARITH_BACKEND_PPC64LE_DEFAULT) && \
|
|
18
|
-
!defined(MLK_CONFIG_MULTILEVEL_NO_SHARED) && defined(__POWER8_VECTOR__)
|
|
18
|
+
!defined(MLK_CONFIG_MULTILEVEL_NO_SHARED) && defined(__POWER8_VECTOR__) && \
|
|
19
|
+
!defined(MLK_CONFIG_NO_KEYPAIR_API)
|
|
20
|
+
/*yaml
|
|
21
|
+
Name: poly_tomont_ppc_asm
|
|
22
|
+
Description: PowerPC64 VSX conversion to Montgomery form
|
|
23
|
+
Signature: void mlk_poly_tomont_ppc_asm(int16_t *r, const int16_t *qdata)
|
|
24
|
+
ABI:
|
|
25
|
+
Architecture: powerpc64le
|
|
26
|
+
CallingConvention: ELFv2
|
|
27
|
+
Features: [VSX]
|
|
28
|
+
r3:
|
|
29
|
+
type: buffer
|
|
30
|
+
size_bytes: 512
|
|
31
|
+
permissions: read/write
|
|
32
|
+
c_parameter: int16_t *r
|
|
33
|
+
description: Input/output polynomial (256 x int16_t)
|
|
34
|
+
r4:
|
|
35
|
+
type: buffer
|
|
36
|
+
size_bytes: 4144
|
|
37
|
+
permissions: read-only
|
|
38
|
+
c_parameter: const int16_t *qdata
|
|
39
|
+
description: Precomputed constants (2072 x int16_t)
|
|
40
|
+
*/
|
|
19
41
|
|
|
20
42
|
/*
|
|
21
43
|
* WARNING: This file is auto-derived from the mlkem-native source file
|
|
22
|
-
* dev/ppc64le/src/
|
|
44
|
+
* dev/ppc64le/src/mlkem_poly_tomont_ppc_asm.S using scripts/simpasm. Do not modify it directly.
|
|
23
45
|
*/
|
|
24
46
|
|
|
25
47
|
.text
|
|
@@ -287,7 +309,7 @@ MLK_ASM_FN_SYMBOL(poly_tomont_ppc_asm)
|
|
|
287
309
|
MLK_ASM_FN_SIZE(poly_tomont_ppc_asm)
|
|
288
310
|
|
|
289
311
|
#endif /* MLK_ARITH_BACKEND_PPC64LE_DEFAULT && !MLK_CONFIG_MULTILEVEL_NO_SHARED \
|
|
290
|
-
&& __POWER8_VECTOR__ */
|
|
312
|
+
&& __POWER8_VECTOR__ && !MLK_CONFIG_NO_KEYPAIR_API */
|
|
291
313
|
|
|
292
314
|
#if defined(__ELF__)
|
|
293
315
|
.section .note.GNU-stack,"",%progbits
|
|
@@ -8,10 +8,31 @@
|
|
|
8
8
|
#include "../../../common.h"
|
|
9
9
|
#if defined(MLK_ARITH_BACKEND_PPC64LE_DEFAULT) && \
|
|
10
10
|
!defined(MLK_CONFIG_MULTILEVEL_NO_SHARED) && defined(__POWER8_VECTOR__)
|
|
11
|
+
/*yaml
|
|
12
|
+
Name: reduce_ppc_asm
|
|
13
|
+
Description: PowerPC64 VSX modular reduction
|
|
14
|
+
Signature: void mlk_reduce_ppc_asm(int16_t *r, const int16_t *qdata)
|
|
15
|
+
ABI:
|
|
16
|
+
Architecture: powerpc64le
|
|
17
|
+
CallingConvention: ELFv2
|
|
18
|
+
Features: [VSX]
|
|
19
|
+
r3:
|
|
20
|
+
type: buffer
|
|
21
|
+
size_bytes: 512
|
|
22
|
+
permissions: read/write
|
|
23
|
+
c_parameter: int16_t *r
|
|
24
|
+
description: Input/output polynomial (256 x int16_t)
|
|
25
|
+
r4:
|
|
26
|
+
type: buffer
|
|
27
|
+
size_bytes: 4144
|
|
28
|
+
permissions: read-only
|
|
29
|
+
c_parameter: const int16_t *qdata
|
|
30
|
+
description: Precomputed constants (2072 x int16_t)
|
|
31
|
+
*/
|
|
11
32
|
|
|
12
33
|
/*
|
|
13
34
|
* WARNING: This file is auto-derived from the mlkem-native source file
|
|
14
|
-
* dev/ppc64le/src/
|
|
35
|
+
* dev/ppc64le/src/mlkem_reduce_ppc_asm.S using scripts/simpasm. Do not modify it directly.
|
|
15
36
|
*/
|
|
16
37
|
|
|
17
38
|
.text
|
|
@@ -40,6 +40,7 @@ static MLK_INLINE int mlk_ntt_native(int16_t data[MLKEM_N])
|
|
|
40
40
|
return MLK_NATIVE_FUNC_SUCCESS;
|
|
41
41
|
}
|
|
42
42
|
|
|
43
|
+
#if !defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API)
|
|
43
44
|
MLK_MUST_CHECK_RETURN_VALUE
|
|
44
45
|
static MLK_INLINE int mlk_intt_native(int16_t data[MLKEM_N])
|
|
45
46
|
{
|
|
@@ -52,13 +53,16 @@ static MLK_INLINE int mlk_intt_native(int16_t data[MLKEM_N])
|
|
|
52
53
|
mlk_rv64v_poly_invntt_tomont(data);
|
|
53
54
|
return MLK_NATIVE_FUNC_SUCCESS;
|
|
54
55
|
}
|
|
56
|
+
#endif /* !MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API */
|
|
55
57
|
|
|
58
|
+
#if !defined(MLK_CONFIG_NO_KEYPAIR_API)
|
|
56
59
|
MLK_MUST_CHECK_RETURN_VALUE
|
|
57
60
|
static MLK_INLINE int mlk_poly_tomont_native(int16_t data[MLKEM_N])
|
|
58
61
|
{
|
|
59
62
|
mlk_rv64v_poly_tomont(data);
|
|
60
63
|
return MLK_NATIVE_FUNC_SUCCESS;
|
|
61
64
|
}
|
|
65
|
+
#endif /* !MLK_CONFIG_NO_KEYPAIR_API */
|
|
62
66
|
|
|
63
67
|
MLK_MUST_CHECK_RETURN_VALUE
|
|
64
68
|
static MLK_INLINE int mlk_rej_uniform_native(int16_t *r, unsigned len,
|
|
@@ -10,8 +10,10 @@
|
|
|
10
10
|
#define mlk_rv64v_poly_ntt MLK_NAMESPACE(ntt_riscv64)
|
|
11
11
|
void mlk_rv64v_poly_ntt(int16_t *);
|
|
12
12
|
|
|
13
|
+
#if !defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API)
|
|
13
14
|
#define mlk_rv64v_poly_invntt_tomont MLK_NAMESPACE(intt_riscv64)
|
|
14
15
|
void mlk_rv64v_poly_invntt_tomont(int16_t *r);
|
|
16
|
+
#endif
|
|
15
17
|
|
|
16
18
|
#define mlk_rv64v_poly_basemul_mont_add_k2 MLK_NAMESPACE(basemul_add_k2_riscv64)
|
|
17
19
|
void mlk_rv64v_poly_basemul_mont_add_k2(int16_t *r, const int16_t *a,
|
|
@@ -25,8 +27,10 @@ void mlk_rv64v_poly_basemul_mont_add_k3(int16_t *r, const int16_t *a,
|
|
|
25
27
|
void mlk_rv64v_poly_basemul_mont_add_k4(int16_t *r, const int16_t *a,
|
|
26
28
|
const int16_t *b);
|
|
27
29
|
|
|
30
|
+
#if !defined(MLK_CONFIG_NO_KEYPAIR_API)
|
|
28
31
|
#define mlk_rv64v_poly_tomont MLK_NAMESPACE(tomont_riscv64)
|
|
29
32
|
void mlk_rv64v_poly_tomont(int16_t *r);
|
|
33
|
+
#endif
|
|
30
34
|
|
|
31
35
|
#define mlk_rv64v_poly_reduce MLK_NAMESPACE(reduce_riscv64)
|
|
32
36
|
void mlk_rv64v_poly_reduce(int16_t *r);
|