pq_crypto 0.6.4 → 0.6.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +46 -0
- data/README.md +5 -0
- data/ext/pqcrypto/pqcrypto_version.h +1 -1
- data/ext/pqcrypto/vendor/.vendored +4 -4
- data/ext/pqcrypto/vendor/mlkem-native/README.md +2 -4
- data/ext/pqcrypto/vendor/mlkem-native/RELEASE.md +98 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/mlkem_native.c +8 -7
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/mlkem_native.h +21 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/mlkem_native_asm.S +45 -44
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/mlkem_native_config.h +36 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/common.h +13 -30
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/compress.c +25 -5
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/compress.h +14 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/context.h +42 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/auto.h +20 -12
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/src/keccak_f1600_x1_scalar_aarch64_asm.S +5 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/src/keccak_f1600_x1_v84a_aarch64_asm.S +5 -2
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/src/keccak_f1600_x2_v84a_aarch64_asm.S +5 -2
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/src/keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm.S +10 -7
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/src/keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm.S +10 -7
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x1_v84a.h +2 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x2_v84a.h +2 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x4_v8a_scalar.h +5 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x4_v8a_v84a_scalar.h +2 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/armv81m/mve.h +5 -19
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/armv81m/src/keccak_f1600_x4_mve.S +8 -5
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/armv81m/src/{state_extract_bytes_x4_mve.S → keccak_f1600_x4_state_extract_bytes_mve.S} +14 -14
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/armv81m/src/{state_xor_bytes_x4_mve.S → keccak_f1600_x4_state_xor_bytes_mve.S} +12 -12
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/x86_64/src/keccak_f1600_x4_avx2_asm.S +36 -2
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/indcpa.c +22 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/indcpa.h +6 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/kem.c +11 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/kem.h +25 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/meta.h +44 -2
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/aarch64_zetas.c +2 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/arith_native_aarch64.h +11 -10
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{intt_aarch64_asm.S → mlkem_intt_aarch64_asm.S} +16 -9
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{ntt_aarch64_asm.S → mlkem_ntt_aarch64_asm.S} +10 -7
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{poly_mulcache_compute_aarch64_asm.S → mlkem_poly_mulcache_compute_aarch64_asm.S} +6 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{poly_reduce_aarch64_asm.S → mlkem_poly_reduce_aarch64_asm.S} +6 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{poly_tobytes_aarch64_asm.S → mlkem_poly_tobytes_aarch64_asm.S} +12 -5
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{poly_tomont_aarch64_asm.S → mlkem_poly_tomont_aarch64_asm.S} +11 -5
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{polyvec_basemul_acc_montgomery_cached_k2_aarch64_asm.S → mlkem_polyvec_basemul_acc_montgomery_cached_k2_aarch64_asm.S} +6 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{polyvec_basemul_acc_montgomery_cached_k3_aarch64_asm.S → mlkem_polyvec_basemul_acc_montgomery_cached_k3_aarch64_asm.S} +6 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{polyvec_basemul_acc_montgomery_cached_k4_aarch64_asm.S → mlkem_polyvec_basemul_acc_montgomery_cached_k4_aarch64_asm.S} +6 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{rej_uniform_aarch64_asm.S → mlkem_rej_uniform_aarch64_asm.S} +17 -17
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/api.h +21 -8
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/meta.h +1 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/ppc64le/meta.h +4 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/ppc64le/src/{intt_ppc_asm.S → mlkem_intt_ppc_asm.S} +29 -5
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/ppc64le/src/{ntt_ppc_asm.S → mlkem_ntt_ppc_asm.S} +22 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/ppc64le/src/{poly_tomont_ppc_asm.S → mlkem_poly_tomont_ppc_asm.S} +25 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/ppc64le/src/{reduce_ppc_asm.S → mlkem_reduce_ppc_asm.S} +22 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/riscv64/meta.h +4 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/riscv64/src/arith_native_riscv64.h +4 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/riscv64/src/rv64v_poly.c +6 -2
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/meta.h +25 -5
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/arith_native_x86_64.h +20 -20
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/compress_consts.c +14 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/compress_consts.h +8 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{intt_avx2_asm.S → mlkem_intt_avx2_asm.S} +27 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{ntt_avx2_asm.S → mlkem_ntt_avx2_asm.S} +23 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{nttfrombytes_avx2_asm.S → mlkem_nttfrombytes_avx2_asm.S} +27 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{ntttobytes_avx2_asm.S → mlkem_ntttobytes_avx2_asm.S} +27 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{nttunpack_avx2_asm.S → mlkem_nttunpack_avx2_asm.S} +17 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{poly_compress_d10_avx2_asm.S → mlkem_poly_compress_d10_avx2_asm.S} +34 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{poly_compress_d11_avx2_asm.S → mlkem_poly_compress_d11_avx2_asm.S} +33 -2
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{poly_compress_d4_avx2_asm.S → mlkem_poly_compress_d4_avx2_asm.S} +34 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{poly_compress_d5_avx2_asm.S → mlkem_poly_compress_d5_avx2_asm.S} +33 -2
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{poly_decompress_d10_avx2_asm.S → mlkem_poly_decompress_d10_avx2_asm.S} +32 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{poly_decompress_d11_avx2_asm.S → mlkem_poly_decompress_d11_avx2_asm.S} +32 -2
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{poly_decompress_d4_avx2_asm.S → mlkem_poly_decompress_d4_avx2_asm.S} +32 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{poly_decompress_d5_avx2_asm.S → mlkem_poly_decompress_d5_avx2_asm.S} +32 -2
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{poly_mulcache_compute_avx2_asm.S → mlkem_poly_mulcache_compute_avx2_asm.S} +29 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{polyvec_basemul_acc_montgomery_cached_k2_avx2_asm.S → mlkem_polyvec_basemul_acc_montgomery_cached_k2_avx2_asm.S} +35 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{polyvec_basemul_acc_montgomery_cached_k3_avx2_asm.S → mlkem_polyvec_basemul_acc_montgomery_cached_k3_avx2_asm.S} +35 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{polyvec_basemul_acc_montgomery_cached_k4_avx2_asm.S → mlkem_polyvec_basemul_acc_montgomery_cached_k4_avx2_asm.S} +35 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{reduce_avx2_asm.S → mlkem_reduce_avx2_asm.S} +17 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{rej_uniform_avx2_asm.S → mlkem_rej_uniform_avx2_asm.S} +40 -7
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{tomont_avx2_asm.S → mlkem_tomont_avx2_asm.S} +20 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/poly.c +7 -2
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/poly.h +6 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/poly_k.c +25 -4
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/poly_k.h +33 -4
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/sampling.c +4 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/sampling.h +4 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/sys.h +19 -1
- data/lib/pq_crypto/version.rb +1 -1
- data/script/vendor_libs.rb +3 -3
- metadata +40 -42
|
@@ -28,37 +28,61 @@
|
|
|
28
28
|
MLK_MUST_CHECK_RETURN_VALUE
|
|
29
29
|
static MLK_INLINE int mlk_ntt_native(int16_t data[MLKEM_N])
|
|
30
30
|
{
|
|
31
|
+
if (!mlk_sys_check_capability(MLK_SYS_CAP_NEON))
|
|
32
|
+
{
|
|
33
|
+
return MLK_NATIVE_FUNC_FALLBACK;
|
|
34
|
+
}
|
|
31
35
|
mlk_ntt_aarch64_asm(data, mlk_aarch64_ntt_zetas_layer12345,
|
|
32
36
|
mlk_aarch64_ntt_zetas_layer67);
|
|
33
37
|
return MLK_NATIVE_FUNC_SUCCESS;
|
|
34
38
|
}
|
|
35
39
|
|
|
40
|
+
#if !defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API)
|
|
36
41
|
MLK_MUST_CHECK_RETURN_VALUE
|
|
37
42
|
static MLK_INLINE int mlk_intt_native(int16_t data[MLKEM_N])
|
|
38
43
|
{
|
|
44
|
+
if (!mlk_sys_check_capability(MLK_SYS_CAP_NEON))
|
|
45
|
+
{
|
|
46
|
+
return MLK_NATIVE_FUNC_FALLBACK;
|
|
47
|
+
}
|
|
39
48
|
mlk_intt_aarch64_asm(data, mlk_aarch64_invntt_zetas_layer12345,
|
|
40
49
|
mlk_aarch64_invntt_zetas_layer67);
|
|
41
50
|
return MLK_NATIVE_FUNC_SUCCESS;
|
|
42
51
|
}
|
|
52
|
+
#endif /* !MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API */
|
|
43
53
|
|
|
44
54
|
MLK_MUST_CHECK_RETURN_VALUE
|
|
45
55
|
static MLK_INLINE int mlk_poly_reduce_native(int16_t data[MLKEM_N])
|
|
46
56
|
{
|
|
57
|
+
if (!mlk_sys_check_capability(MLK_SYS_CAP_NEON))
|
|
58
|
+
{
|
|
59
|
+
return MLK_NATIVE_FUNC_FALLBACK;
|
|
60
|
+
}
|
|
47
61
|
mlk_poly_reduce_aarch64_asm(data);
|
|
48
62
|
return MLK_NATIVE_FUNC_SUCCESS;
|
|
49
63
|
}
|
|
50
64
|
|
|
65
|
+
#if !defined(MLK_CONFIG_NO_KEYPAIR_API)
|
|
51
66
|
MLK_MUST_CHECK_RETURN_VALUE
|
|
52
67
|
static MLK_INLINE int mlk_poly_tomont_native(int16_t data[MLKEM_N])
|
|
53
68
|
{
|
|
69
|
+
if (!mlk_sys_check_capability(MLK_SYS_CAP_NEON))
|
|
70
|
+
{
|
|
71
|
+
return MLK_NATIVE_FUNC_FALLBACK;
|
|
72
|
+
}
|
|
54
73
|
mlk_poly_tomont_aarch64_asm(data);
|
|
55
74
|
return MLK_NATIVE_FUNC_SUCCESS;
|
|
56
75
|
}
|
|
76
|
+
#endif /* !MLK_CONFIG_NO_KEYPAIR_API */
|
|
57
77
|
|
|
58
78
|
MLK_MUST_CHECK_RETURN_VALUE
|
|
59
79
|
static MLK_INLINE int mlk_poly_mulcache_compute_native(int16_t x[MLKEM_N / 2],
|
|
60
80
|
const int16_t y[MLKEM_N])
|
|
61
81
|
{
|
|
82
|
+
if (!mlk_sys_check_capability(MLK_SYS_CAP_NEON))
|
|
83
|
+
{
|
|
84
|
+
return MLK_NATIVE_FUNC_FALLBACK;
|
|
85
|
+
}
|
|
62
86
|
mlk_poly_mulcache_compute_aarch64_asm(
|
|
63
87
|
x, y, mlk_aarch64_zetas_mulcache_native,
|
|
64
88
|
mlk_aarch64_zetas_mulcache_twisted_native);
|
|
@@ -71,6 +95,10 @@ static MLK_INLINE int mlk_polyvec_basemul_acc_montgomery_cached_k2_native(
|
|
|
71
95
|
int16_t r[MLKEM_N], const int16_t a[2 * MLKEM_N],
|
|
72
96
|
const int16_t b[2 * MLKEM_N], const int16_t b_cache[2 * (MLKEM_N / 2)])
|
|
73
97
|
{
|
|
98
|
+
if (!mlk_sys_check_capability(MLK_SYS_CAP_NEON))
|
|
99
|
+
{
|
|
100
|
+
return MLK_NATIVE_FUNC_FALLBACK;
|
|
101
|
+
}
|
|
74
102
|
mlk_polyvec_basemul_acc_montgomery_cached_k2_aarch64_asm(r, a, b, b_cache);
|
|
75
103
|
return MLK_NATIVE_FUNC_SUCCESS;
|
|
76
104
|
}
|
|
@@ -82,6 +110,10 @@ static MLK_INLINE int mlk_polyvec_basemul_acc_montgomery_cached_k3_native(
|
|
|
82
110
|
int16_t r[MLKEM_N], const int16_t a[3 * MLKEM_N],
|
|
83
111
|
const int16_t b[3 * MLKEM_N], const int16_t b_cache[3 * (MLKEM_N / 2)])
|
|
84
112
|
{
|
|
113
|
+
if (!mlk_sys_check_capability(MLK_SYS_CAP_NEON))
|
|
114
|
+
{
|
|
115
|
+
return MLK_NATIVE_FUNC_FALLBACK;
|
|
116
|
+
}
|
|
85
117
|
mlk_polyvec_basemul_acc_montgomery_cached_k3_aarch64_asm(r, a, b, b_cache);
|
|
86
118
|
return MLK_NATIVE_FUNC_SUCCESS;
|
|
87
119
|
}
|
|
@@ -93,26 +125,36 @@ static MLK_INLINE int mlk_polyvec_basemul_acc_montgomery_cached_k4_native(
|
|
|
93
125
|
int16_t r[MLKEM_N], const int16_t a[4 * MLKEM_N],
|
|
94
126
|
const int16_t b[4 * MLKEM_N], const int16_t b_cache[4 * (MLKEM_N / 2)])
|
|
95
127
|
{
|
|
128
|
+
if (!mlk_sys_check_capability(MLK_SYS_CAP_NEON))
|
|
129
|
+
{
|
|
130
|
+
return MLK_NATIVE_FUNC_FALLBACK;
|
|
131
|
+
}
|
|
96
132
|
mlk_polyvec_basemul_acc_montgomery_cached_k4_aarch64_asm(r, a, b, b_cache);
|
|
97
133
|
return MLK_NATIVE_FUNC_SUCCESS;
|
|
98
134
|
}
|
|
99
135
|
#endif /* MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 4 */
|
|
100
136
|
|
|
137
|
+
#if !defined(MLK_CONFIG_NO_KEYPAIR_API) || !defined(MLK_CONFIG_NO_ENCAPS_API)
|
|
101
138
|
MLK_MUST_CHECK_RETURN_VALUE
|
|
102
139
|
static MLK_INLINE int mlk_poly_tobytes_native(uint8_t r[MLKEM_POLYBYTES],
|
|
103
140
|
const int16_t a[MLKEM_N])
|
|
104
141
|
{
|
|
142
|
+
if (!mlk_sys_check_capability(MLK_SYS_CAP_NEON))
|
|
143
|
+
{
|
|
144
|
+
return MLK_NATIVE_FUNC_FALLBACK;
|
|
145
|
+
}
|
|
105
146
|
mlk_poly_tobytes_aarch64_asm(r, a);
|
|
106
147
|
return MLK_NATIVE_FUNC_SUCCESS;
|
|
107
148
|
}
|
|
149
|
+
#endif /* !MLK_CONFIG_NO_KEYPAIR_API || !MLK_CONFIG_NO_ENCAPS_API */
|
|
108
150
|
|
|
109
151
|
MLK_MUST_CHECK_RETURN_VALUE
|
|
110
152
|
static MLK_INLINE int mlk_rej_uniform_native(int16_t *r, unsigned len,
|
|
111
153
|
const uint8_t *buf,
|
|
112
154
|
unsigned buflen)
|
|
113
155
|
{
|
|
114
|
-
if (len != MLKEM_N ||
|
|
115
|
-
buflen % 24 != 0)
|
|
156
|
+
if (!mlk_sys_check_capability(MLK_SYS_CAP_NEON) || len != MLKEM_N ||
|
|
157
|
+
buflen % 24 != 0)
|
|
116
158
|
{
|
|
117
159
|
return MLK_NATIVE_FUNC_FALLBACK;
|
|
118
160
|
}
|
|
@@ -79,6 +79,7 @@ MLK_ALIGN MLK_INTERNAL_DATA_DEFINITION const int16_t
|
|
|
79
79
|
10129, 10129, -3878, -3878, -11566, -11566,
|
|
80
80
|
};
|
|
81
81
|
|
|
82
|
+
#if !defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API)
|
|
82
83
|
MLK_ALIGN MLK_INTERNAL_DATA_DEFINITION const int16_t
|
|
83
84
|
mlk_aarch64_invntt_zetas_layer12345[80] = {
|
|
84
85
|
1583, 15582, -821, -8081, 1355, 13338, 0, 0, -569,
|
|
@@ -138,6 +139,7 @@ MLK_ALIGN MLK_INTERNAL_DATA_DEFINITION const int16_t
|
|
|
138
139
|
1041, -1637, -1637, -583, -583, -17, -17, 10247, 10247,
|
|
139
140
|
-16113, -16113, -5739, -5739, -167, -167,
|
|
140
141
|
};
|
|
142
|
+
#endif /* !MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API */
|
|
141
143
|
|
|
142
144
|
MLK_ALIGN MLK_INTERNAL_DATA_DEFINITION const int16_t
|
|
143
145
|
mlk_aarch64_zetas_mulcache_native[128] = {
|
|
@@ -38,7 +38,7 @@ MLK_INTERNAL_DATA_DECLARATION const uint8_t mlk_rej_uniform_table[4096];
|
|
|
38
38
|
void mlk_ntt_aarch64_asm(int16_t p[256], const int16_t twiddles12345[80],
|
|
39
39
|
const int16_t twiddles56[384])
|
|
40
40
|
/* This must be kept in sync with the HOL-Light specification
|
|
41
|
-
* in proofs/hol_light/aarch64/proofs/
|
|
41
|
+
* in proofs/hol_light/aarch64/proofs/mlkem_ntt_aarch64_asm.ml */
|
|
42
42
|
__contract__(
|
|
43
43
|
requires(memory_no_alias(p, sizeof(int16_t) * MLKEM_N))
|
|
44
44
|
requires(array_abs_bound(p, 0, MLKEM_N, 8192))
|
|
@@ -54,7 +54,7 @@ __contract__(
|
|
|
54
54
|
void mlk_intt_aarch64_asm(int16_t p[256], const int16_t twiddles12345[80],
|
|
55
55
|
const int16_t twiddles56[384])
|
|
56
56
|
/* This must be kept in sync with the HOL-Light specification
|
|
57
|
-
* in proofs/hol_light/aarch64/proofs/
|
|
57
|
+
* in proofs/hol_light/aarch64/proofs/mlkem_intt_aarch64_asm.ml */
|
|
58
58
|
__contract__(
|
|
59
59
|
requires(memory_no_alias(p, sizeof(int16_t) * MLKEM_N))
|
|
60
60
|
requires(twiddles12345 == mlk_aarch64_invntt_zetas_layer12345)
|
|
@@ -68,7 +68,7 @@ __contract__(
|
|
|
68
68
|
#define mlk_poly_reduce_aarch64_asm MLK_NAMESPACE(poly_reduce_aarch64_asm)
|
|
69
69
|
void mlk_poly_reduce_aarch64_asm(int16_t p[256])
|
|
70
70
|
/* This must be kept in sync with the HOL-Light specification
|
|
71
|
-
* in proofs/hol_light/aarch64/proofs/
|
|
71
|
+
* in proofs/hol_light/aarch64/proofs/mlkem_poly_reduce_aarch64_asm.ml */
|
|
72
72
|
__contract__(
|
|
73
73
|
requires(memory_no_alias(p, sizeof(int16_t) * MLKEM_N))
|
|
74
74
|
assigns(memory_slice(p, sizeof(int16_t) * MLKEM_N))
|
|
@@ -78,7 +78,7 @@ __contract__(
|
|
|
78
78
|
#define mlk_poly_tomont_aarch64_asm MLK_NAMESPACE(poly_tomont_aarch64_asm)
|
|
79
79
|
void mlk_poly_tomont_aarch64_asm(int16_t p[256])
|
|
80
80
|
/* This must be kept in sync with the HOL-Light specification
|
|
81
|
-
* in proofs/hol_light/aarch64/proofs/
|
|
81
|
+
* in proofs/hol_light/aarch64/proofs/mlkem_poly_tomont_aarch64_asm.ml */
|
|
82
82
|
__contract__(
|
|
83
83
|
requires(memory_no_alias(p, sizeof(int16_t) * MLKEM_N))
|
|
84
84
|
assigns(memory_slice(p, sizeof(int16_t) * MLKEM_N))
|
|
@@ -92,7 +92,8 @@ void mlk_poly_mulcache_compute_aarch64_asm(int16_t cache[128],
|
|
|
92
92
|
const int16_t zetas[128],
|
|
93
93
|
const int16_t zetas_twisted[128])
|
|
94
94
|
/* This must be kept in sync with the HOL-Light specification
|
|
95
|
-
* in proofs/hol_light/aarch64/proofs/
|
|
95
|
+
* in proofs/hol_light/aarch64/proofs/mlkem_poly_mulcache_compute_aarch64_asm.ml
|
|
96
|
+
*/
|
|
96
97
|
__contract__(
|
|
97
98
|
requires(memory_no_alias(cache, sizeof(int16_t) * (MLKEM_N / 2)))
|
|
98
99
|
requires(memory_no_alias(mlk_poly, sizeof(int16_t) * MLKEM_N))
|
|
@@ -105,7 +106,7 @@ __contract__(
|
|
|
105
106
|
#define mlk_poly_tobytes_aarch64_asm MLK_NAMESPACE(poly_tobytes_aarch64_asm)
|
|
106
107
|
void mlk_poly_tobytes_aarch64_asm(uint8_t r[384], const int16_t a[256])
|
|
107
108
|
/* This must be kept in sync with the HOL-Light specification
|
|
108
|
-
* in proofs/hol_light/aarch64/proofs/
|
|
109
|
+
* in proofs/hol_light/aarch64/proofs/mlkem_poly_tobytes_aarch64_asm.ml */
|
|
109
110
|
__contract__(
|
|
110
111
|
requires(memory_no_alias(r, MLKEM_POLYBYTES))
|
|
111
112
|
requires(memory_no_alias(a, sizeof(int16_t) * MLKEM_N))
|
|
@@ -119,7 +120,7 @@ void mlk_polyvec_basemul_acc_montgomery_cached_k2_aarch64_asm(
|
|
|
119
120
|
int16_t r[256], const int16_t a[512], const int16_t b[512],
|
|
120
121
|
const int16_t b_cache[256])
|
|
121
122
|
/* This must be kept in sync with the HOL-Light specification in
|
|
122
|
-
* proofs/hol_light/aarch64/proofs/
|
|
123
|
+
* proofs/hol_light/aarch64/proofs/mlkem_polyvec_basemul_acc_montgomery_cached_k2_aarch64_asm.ml.
|
|
123
124
|
*/
|
|
124
125
|
__contract__(
|
|
125
126
|
requires(memory_no_alias(r, sizeof(int16_t) * MLKEM_N))
|
|
@@ -136,7 +137,7 @@ void mlk_polyvec_basemul_acc_montgomery_cached_k3_aarch64_asm(
|
|
|
136
137
|
int16_t r[256], const int16_t a[768], const int16_t b[768],
|
|
137
138
|
const int16_t b_cache[384])
|
|
138
139
|
/* This must be kept in sync with the HOL-Light specification in
|
|
139
|
-
* proofs/hol_light/aarch64/proofs/
|
|
140
|
+
* proofs/hol_light/aarch64/proofs/mlkem_polyvec_basemul_acc_montgomery_cached_k3_aarch64_asm.ml.
|
|
140
141
|
*/
|
|
141
142
|
__contract__(
|
|
142
143
|
requires(memory_no_alias(r, sizeof(int16_t) * MLKEM_N))
|
|
@@ -153,7 +154,7 @@ void mlk_polyvec_basemul_acc_montgomery_cached_k4_aarch64_asm(
|
|
|
153
154
|
int16_t r[256], const int16_t a[1024], const int16_t b[1024],
|
|
154
155
|
const int16_t b_cache[512])
|
|
155
156
|
/* This must be kept in sync with the HOL-Light specification in
|
|
156
|
-
* proofs/hol_light/aarch64/proofs/
|
|
157
|
+
* proofs/hol_light/aarch64/proofs/mlkem_polyvec_basemul_acc_montgomery_cached_k4_aarch64_asm.ml.
|
|
157
158
|
*/
|
|
158
159
|
__contract__(
|
|
159
160
|
requires(memory_no_alias(r, sizeof(int16_t) * MLKEM_N))
|
|
@@ -169,7 +170,7 @@ MLK_MUST_CHECK_RETURN_VALUE
|
|
|
169
170
|
uint64_t mlk_rej_uniform_aarch64_asm(int16_t r[256], const uint8_t *buf,
|
|
170
171
|
unsigned buflen, const uint8_t table[4096])
|
|
171
172
|
/* This must be kept in sync with the HOL-Light specification
|
|
172
|
-
* in proofs/hol_light/aarch64/proofs/
|
|
173
|
+
* in proofs/hol_light/aarch64/proofs/mlkem_rej_uniform_aarch64_asm.ml. */
|
|
173
174
|
__contract__(
|
|
174
175
|
requires(buflen % 24 == 0)
|
|
175
176
|
requires(memory_no_alias(buf, buflen))
|
|
@@ -24,6 +24,9 @@
|
|
|
24
24
|
Description: AArch64 ML-KEM inverse NTT following @[NeonNTT] and @[SLOTHY_Paper]
|
|
25
25
|
Signature: void mlk_intt_aarch64_asm(int16_t p[256], const int16_t twiddles12345[80], const int16_t twiddles56[384])
|
|
26
26
|
ABI:
|
|
27
|
+
Architecture: aarch64
|
|
28
|
+
CallingConvention: AAPCS64
|
|
29
|
+
Features: [NEON]
|
|
27
30
|
x0:
|
|
28
31
|
type: buffer
|
|
29
32
|
size_bytes: 512
|
|
@@ -48,11 +51,14 @@
|
|
|
48
51
|
*/
|
|
49
52
|
|
|
50
53
|
#include "../../../common.h"
|
|
51
|
-
#if defined(MLK_ARITH_BACKEND_AARCH64) &&
|
|
54
|
+
#if defined(MLK_ARITH_BACKEND_AARCH64) && \
|
|
55
|
+
!defined(MLK_CONFIG_MULTILEVEL_NO_SHARED) && \
|
|
56
|
+
(!defined(MLK_CONFIG_NO_ENCAPS_API) || \
|
|
57
|
+
!defined(MLK_CONFIG_NO_DECAPS_API))
|
|
52
58
|
|
|
53
59
|
/*
|
|
54
60
|
* WARNING: This file is auto-derived from the mlkem-native source file
|
|
55
|
-
* dev/aarch64_opt/src/
|
|
61
|
+
* dev/aarch64_opt/src/mlkem_intt_aarch64_asm.S using scripts/simpasm. Do not modify it directly.
|
|
56
62
|
*/
|
|
57
63
|
|
|
58
64
|
.text
|
|
@@ -200,7 +206,7 @@ MLK_ASM_FN_SYMBOL(intt_aarch64_asm)
|
|
|
200
206
|
mls v24.8h, v19.8h, v7.h[0]
|
|
201
207
|
sub x4, x4, #0x2
|
|
202
208
|
|
|
203
|
-
|
|
209
|
+
Lmlk_intt_layer4567_start:
|
|
204
210
|
add v16.8h, v21.8h, v10.8h
|
|
205
211
|
mul v18.8h, v6.8h, v15.8h
|
|
206
212
|
sub v19.8h, v20.8h, v25.8h
|
|
@@ -293,8 +299,8 @@ Lintt_layer4567_start:
|
|
|
293
299
|
sqrdmulh v31.8h, v8.8h, v11.h[5]
|
|
294
300
|
sub v14.8h, v21.8h, v10.8h
|
|
295
301
|
stur q0, [x3, #-0x30]
|
|
296
|
-
|
|
297
|
-
cbnz x4,
|
|
302
|
+
sub x4, x4, #0x1
|
|
303
|
+
cbnz x4, Lmlk_intt_layer4567_start
|
|
298
304
|
mul v15.8h, v6.8h, v15.8h
|
|
299
305
|
sub v22.8h, v20.8h, v25.8h
|
|
300
306
|
add v4.8h, v21.8h, v10.8h
|
|
@@ -442,7 +448,7 @@ Lintt_layer4567_start:
|
|
|
442
448
|
sqrdmulh v10.8h, v10.8h, v0.h[1]
|
|
443
449
|
sub x4, x4, #0x2
|
|
444
450
|
|
|
445
|
-
|
|
451
|
+
Lmlk_intt_layer123_start:
|
|
446
452
|
sub v12.8h, v3.8h, v30.8h
|
|
447
453
|
mul v11.8h, v21.8h, v1.h[0]
|
|
448
454
|
add v28.8h, v4.8h, v8.8h
|
|
@@ -519,8 +525,8 @@ Lintt_layer123_start:
|
|
|
519
525
|
sqrdmulh v9.8h, v13.8h, v0.h[1]
|
|
520
526
|
str q17, [x0, #0x130]
|
|
521
527
|
mul v17.8h, v19.8h, v0.h[0]
|
|
522
|
-
|
|
523
|
-
cbnz x4,
|
|
528
|
+
sub x4, x4, #0x1
|
|
529
|
+
cbnz x4, Lmlk_intt_layer123_start
|
|
524
530
|
mls v23.8h, v10.8h, v7.h[0]
|
|
525
531
|
ldr q11, [x0, #0x190]
|
|
526
532
|
str q18, [x0, #0x100]
|
|
@@ -621,7 +627,8 @@ Lintt_layer123_start:
|
|
|
621
627
|
|
|
622
628
|
MLK_ASM_FN_SIZE(intt_aarch64_asm)
|
|
623
629
|
|
|
624
|
-
#endif /* MLK_ARITH_BACKEND_AARCH64 && !MLK_CONFIG_MULTILEVEL_NO_SHARED
|
|
630
|
+
#endif /* MLK_ARITH_BACKEND_AARCH64 && !MLK_CONFIG_MULTILEVEL_NO_SHARED && \
|
|
631
|
+
(!MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API) */
|
|
625
632
|
|
|
626
633
|
#if defined(__ELF__)
|
|
627
634
|
.section .note.GNU-stack,"",%progbits
|
|
@@ -24,6 +24,9 @@
|
|
|
24
24
|
Description: AArch64 ML-KEM forward NTT following @[NeonNTT] and @[SLOTHY_Paper]
|
|
25
25
|
Signature: void mlk_ntt_aarch64_asm(int16_t p[256], const int16_t twiddles12345[80], const int16_t twiddles56[384])
|
|
26
26
|
ABI:
|
|
27
|
+
Architecture: aarch64
|
|
28
|
+
CallingConvention: AAPCS64
|
|
29
|
+
Features: [NEON]
|
|
27
30
|
x0:
|
|
28
31
|
type: buffer
|
|
29
32
|
size_bytes: 512
|
|
@@ -52,7 +55,7 @@
|
|
|
52
55
|
|
|
53
56
|
/*
|
|
54
57
|
* WARNING: This file is auto-derived from the mlkem-native source file
|
|
55
|
-
* dev/aarch64_opt/src/
|
|
58
|
+
* dev/aarch64_opt/src/mlkem_ntt_aarch64_asm.S using scripts/simpasm. Do not modify it directly.
|
|
56
59
|
*/
|
|
57
60
|
|
|
58
61
|
.text
|
|
@@ -144,7 +147,7 @@ MLK_ASM_FN_SYMBOL(ntt_aarch64_asm)
|
|
|
144
147
|
mul v4.8h, v5.8h, v1.h[0]
|
|
145
148
|
sub x4, x4, #0x2
|
|
146
149
|
|
|
147
|
-
|
|
150
|
+
Lmlk_ntt_layer123_start:
|
|
148
151
|
mls v23.8h, v15.8h, v7.h[0]
|
|
149
152
|
ldr q6, [x0, #0x190]
|
|
150
153
|
ldr q15, [x0, #0x90]
|
|
@@ -221,8 +224,8 @@ Lntt_layer123_start:
|
|
|
221
224
|
add v28.8h, v13.8h, v23.8h
|
|
222
225
|
sub v13.8h, v13.8h, v23.8h
|
|
223
226
|
mul v23.8h, v20.8h, v0.h[2]
|
|
224
|
-
|
|
225
|
-
cbnz x4,
|
|
227
|
+
sub x4, x4, #0x1
|
|
228
|
+
cbnz x4, Lmlk_ntt_layer123_start
|
|
226
229
|
sqrdmulh v3.8h, v5.8h, v1.h[1]
|
|
227
230
|
mls v23.8h, v15.8h, v7.h[0]
|
|
228
231
|
ldr q5, [x0, #0x190]
|
|
@@ -396,7 +399,7 @@ Lntt_layer123_start:
|
|
|
396
399
|
mls v23.8h, v16.8h, v7.h[0]
|
|
397
400
|
sub x4, x4, #0x2
|
|
398
401
|
|
|
399
|
-
|
|
402
|
+
Lmlk_ntt_layer4567_start:
|
|
400
403
|
ldr q19, [x2, #0x50]
|
|
401
404
|
sub v31.8h, v30.8h, v23.8h
|
|
402
405
|
mls v0.8h, v29.8h, v7.h[0]
|
|
@@ -468,8 +471,8 @@ Lntt_layer4567_start:
|
|
|
468
471
|
sub v30.8h, v21.8h, v17.8h
|
|
469
472
|
mul v0.8h, v16.8h, v1.8h
|
|
470
473
|
trn1 v28.4s, v6.4s, v12.4s
|
|
471
|
-
|
|
472
|
-
cbnz x4,
|
|
474
|
+
sub x4, x4, #0x1
|
|
475
|
+
cbnz x4, Lmlk_ntt_layer4567_start
|
|
473
476
|
add v22.8h, v11.8h, v8.8h
|
|
474
477
|
mul v27.8h, v27.8h, v4.h[2]
|
|
475
478
|
trn2 v17.4s, v6.4s, v12.4s
|
|
@@ -8,6 +8,9 @@
|
|
|
8
8
|
Description: Compute multiplication cache for polynomial
|
|
9
9
|
Signature: void mlk_poly_mulcache_compute_aarch64_asm(int16_t cache[128], const int16_t mlk_poly[256], const int16_t zetas[128], const int16_t zetas_twisted[128])
|
|
10
10
|
ABI:
|
|
11
|
+
Architecture: aarch64
|
|
12
|
+
CallingConvention: AAPCS64
|
|
13
|
+
Features: [NEON]
|
|
11
14
|
x0:
|
|
12
15
|
type: buffer
|
|
13
16
|
size_bytes: 256
|
|
@@ -41,7 +44,7 @@
|
|
|
41
44
|
|
|
42
45
|
/*
|
|
43
46
|
* WARNING: This file is auto-derived from the mlkem-native source file
|
|
44
|
-
* dev/aarch64_opt/src/
|
|
47
|
+
* dev/aarch64_opt/src/mlkem_poly_mulcache_compute_aarch64_asm.S using scripts/simpasm. Do not modify it directly.
|
|
45
48
|
*/
|
|
46
49
|
|
|
47
50
|
.text
|
|
@@ -79,7 +82,7 @@ MLK_ASM_FN_SYMBOL(poly_mulcache_compute_aarch64_asm)
|
|
|
79
82
|
lsr x4, x4, #1
|
|
80
83
|
sub x4, x4, #0x2
|
|
81
84
|
|
|
82
|
-
|
|
85
|
+
Lmlk_poly_mulcache_compute_loop_start:
|
|
83
86
|
str q5, [x0], #0x10
|
|
84
87
|
sqrdmulh v22.8h, v29.8h, v17.8h
|
|
85
88
|
ldr q28, [x2], #0x10
|
|
@@ -99,7 +102,7 @@ Lpoly_mulcache_compute_loop_start:
|
|
|
99
102
|
str q26, [x0], #0x10
|
|
100
103
|
mul v26.8h, v23.8h, v18.8h
|
|
101
104
|
subs x4, x4, #0x1
|
|
102
|
-
cbnz x4,
|
|
105
|
+
cbnz x4, Lmlk_poly_mulcache_compute_loop_start
|
|
103
106
|
mls v26.8h, v4.8h, v6.h[0]
|
|
104
107
|
str q5, [x0], #0x10
|
|
105
108
|
ldr q5, [x2], #0x10
|
|
@@ -8,6 +8,9 @@
|
|
|
8
8
|
Description: Barrett reduction of polynomial coefficients
|
|
9
9
|
Signature: void mlk_poly_reduce_aarch64_asm(int16_t p[256])
|
|
10
10
|
ABI:
|
|
11
|
+
Architecture: aarch64
|
|
12
|
+
CallingConvention: AAPCS64
|
|
13
|
+
Features: [NEON]
|
|
11
14
|
x0:
|
|
12
15
|
type: buffer
|
|
13
16
|
size_bytes: 512
|
|
@@ -23,7 +26,7 @@
|
|
|
23
26
|
|
|
24
27
|
/*
|
|
25
28
|
* WARNING: This file is auto-derived from the mlkem-native source file
|
|
26
|
-
* dev/aarch64_opt/src/
|
|
29
|
+
* dev/aarch64_opt/src/mlkem_poly_reduce_aarch64_asm.S using scripts/simpasm. Do not modify it directly.
|
|
27
30
|
*/
|
|
28
31
|
|
|
29
32
|
.text
|
|
@@ -64,7 +67,7 @@ MLK_ASM_FN_SYMBOL(poly_reduce_aarch64_asm)
|
|
|
64
67
|
and v31.16b, v3.16b, v31.16b
|
|
65
68
|
sub x1, x1, #0x2
|
|
66
69
|
|
|
67
|
-
|
|
70
|
+
Lmlk_poly_reduce_loop_start:
|
|
68
71
|
add v21.8h, v18.8h, v21.8h
|
|
69
72
|
ldur q18, [x0, #-0x20]
|
|
70
73
|
add v25.8h, v0.8h, v31.8h
|
|
@@ -98,7 +101,7 @@ Lpoly_reduce_loop_start:
|
|
|
98
101
|
and v31.16b, v3.16b, v20.16b
|
|
99
102
|
sqdmulh v2.8h, v26.8h, v4.h[0]
|
|
100
103
|
subs x1, x1, #0x1
|
|
101
|
-
cbnz x1,
|
|
104
|
+
cbnz x1, Lmlk_poly_reduce_loop_start
|
|
102
105
|
add v28.8h, v0.8h, v31.8h
|
|
103
106
|
ldur q29, [x0, #-0x10]
|
|
104
107
|
add v21.8h, v18.8h, v21.8h
|
|
@@ -8,6 +8,9 @@
|
|
|
8
8
|
Description: Convert polynomial to byte representation
|
|
9
9
|
Signature: void mlk_poly_tobytes_aarch64_asm(uint8_t r[384], const int16_t a[256])
|
|
10
10
|
ABI:
|
|
11
|
+
Architecture: aarch64
|
|
12
|
+
CallingConvention: AAPCS64
|
|
13
|
+
Features: [NEON]
|
|
11
14
|
x0:
|
|
12
15
|
type: buffer
|
|
13
16
|
size_bytes: 384
|
|
@@ -25,11 +28,14 @@
|
|
|
25
28
|
*/
|
|
26
29
|
|
|
27
30
|
#include "../../../common.h"
|
|
28
|
-
#if defined(MLK_ARITH_BACKEND_AARCH64) &&
|
|
31
|
+
#if defined(MLK_ARITH_BACKEND_AARCH64) && \
|
|
32
|
+
!defined(MLK_CONFIG_MULTILEVEL_NO_SHARED) && \
|
|
33
|
+
(!defined(MLK_CONFIG_NO_KEYPAIR_API) || \
|
|
34
|
+
!defined(MLK_CONFIG_NO_ENCAPS_API))
|
|
29
35
|
|
|
30
36
|
/*
|
|
31
37
|
* WARNING: This file is auto-derived from the mlkem-native source file
|
|
32
|
-
* dev/aarch64_opt/src/
|
|
38
|
+
* dev/aarch64_opt/src/mlkem_poly_tobytes_aarch64_asm.S using scripts/simpasm. Do not modify it directly.
|
|
33
39
|
*/
|
|
34
40
|
|
|
35
41
|
.text
|
|
@@ -64,7 +70,7 @@ MLK_ASM_FN_SYMBOL(poly_tobytes_aarch64_asm)
|
|
|
64
70
|
lsr x2, x2, #1
|
|
65
71
|
sub x2, x2, #0x2
|
|
66
72
|
|
|
67
|
-
|
|
73
|
+
Lmlk_poly_tobytes_loop_start:
|
|
68
74
|
uzp1 v25.8h, v17.8h, v27.8h
|
|
69
75
|
uzp2 v31.8h, v17.8h, v27.8h
|
|
70
76
|
uzp1 v24.8h, v16.8h, v23.8h
|
|
@@ -86,7 +92,7 @@ Lpoly_tobytes_loop_start:
|
|
|
86
92
|
xtn v28.8b, v24.8h
|
|
87
93
|
shrn v30.8b, v6.8h, #0x4
|
|
88
94
|
subs x2, x2, #0x1
|
|
89
|
-
cbnz x2,
|
|
95
|
+
cbnz x2, Lmlk_poly_tobytes_loop_start
|
|
90
96
|
uzp2 v7.8h, v17.8h, v27.8h
|
|
91
97
|
uzp1 v25.8h, v17.8h, v27.8h
|
|
92
98
|
uzp2 v0.8h, v16.8h, v23.8h
|
|
@@ -110,7 +116,8 @@ Lpoly_tobytes_loop_start:
|
|
|
110
116
|
|
|
111
117
|
MLK_ASM_FN_SIZE(poly_tobytes_aarch64_asm)
|
|
112
118
|
|
|
113
|
-
#endif /* MLK_ARITH_BACKEND_AARCH64 && !MLK_CONFIG_MULTILEVEL_NO_SHARED
|
|
119
|
+
#endif /* MLK_ARITH_BACKEND_AARCH64 && !MLK_CONFIG_MULTILEVEL_NO_SHARED && \
|
|
120
|
+
(!MLK_CONFIG_NO_KEYPAIR_API || !MLK_CONFIG_NO_ENCAPS_API) */
|
|
114
121
|
|
|
115
122
|
#if defined(__ELF__)
|
|
116
123
|
.section .note.GNU-stack,"",%progbits
|
|
@@ -8,6 +8,9 @@
|
|
|
8
8
|
Description: Convert polynomial to Montgomery domain
|
|
9
9
|
Signature: void mlk_poly_tomont_aarch64_asm(int16_t p[256])
|
|
10
10
|
ABI:
|
|
11
|
+
Architecture: aarch64
|
|
12
|
+
CallingConvention: AAPCS64
|
|
13
|
+
Features: [NEON]
|
|
11
14
|
x0:
|
|
12
15
|
type: buffer
|
|
13
16
|
size_bytes: 512
|
|
@@ -19,11 +22,13 @@
|
|
|
19
22
|
*/
|
|
20
23
|
|
|
21
24
|
#include "../../../common.h"
|
|
22
|
-
#if defined(MLK_ARITH_BACKEND_AARCH64) &&
|
|
25
|
+
#if defined(MLK_ARITH_BACKEND_AARCH64) && \
|
|
26
|
+
!defined(MLK_CONFIG_MULTILEVEL_NO_SHARED) && \
|
|
27
|
+
!defined(MLK_CONFIG_NO_KEYPAIR_API)
|
|
23
28
|
|
|
24
29
|
/*
|
|
25
30
|
* WARNING: This file is auto-derived from the mlkem-native source file
|
|
26
|
-
* dev/aarch64_opt/src/
|
|
31
|
+
* dev/aarch64_opt/src/mlkem_poly_tomont_aarch64_asm.S using scripts/simpasm. Do not modify it directly.
|
|
27
32
|
*/
|
|
28
33
|
|
|
29
34
|
.text
|
|
@@ -57,7 +62,7 @@ MLK_ASM_FN_SYMBOL(poly_tomont_aarch64_asm)
|
|
|
57
62
|
mls v18.8h, v26.8h, v4.h[0]
|
|
58
63
|
sub x1, x1, #0x1
|
|
59
64
|
|
|
60
|
-
|
|
65
|
+
Lmlk_poly_tomont_loop:
|
|
61
66
|
ldr q19, [x0, #0x10]
|
|
62
67
|
mul v26.8h, v16.8h, v2.8h
|
|
63
68
|
ldr q23, [x0, #0x20]
|
|
@@ -79,7 +84,7 @@ Lpoly_tomont_loop:
|
|
|
79
84
|
mls v18.8h, v24.8h, v4.h[0]
|
|
80
85
|
stur q26, [x0, #-0x40]
|
|
81
86
|
sub x1, x1, #0x1
|
|
82
|
-
cbnz x1,
|
|
87
|
+
cbnz x1, Lmlk_poly_tomont_loop
|
|
83
88
|
mul v16.8h, v16.8h, v2.8h
|
|
84
89
|
stur q18, [x0, #-0x20]
|
|
85
90
|
mls v16.8h, v29.8h, v4.h[0]
|
|
@@ -89,7 +94,8 @@ Lpoly_tomont_loop:
|
|
|
89
94
|
|
|
90
95
|
MLK_ASM_FN_SIZE(poly_tomont_aarch64_asm)
|
|
91
96
|
|
|
92
|
-
#endif /* MLK_ARITH_BACKEND_AARCH64 && !MLK_CONFIG_MULTILEVEL_NO_SHARED
|
|
97
|
+
#endif /* MLK_ARITH_BACKEND_AARCH64 && !MLK_CONFIG_MULTILEVEL_NO_SHARED && \
|
|
98
|
+
!MLK_CONFIG_NO_KEYPAIR_API */
|
|
93
99
|
|
|
94
100
|
#if defined(__ELF__)
|
|
95
101
|
.section .note.GNU-stack,"",%progbits
|
|
@@ -17,6 +17,9 @@
|
|
|
17
17
|
Description: Re-implementation of asymmetric base multiplication following @[NeonNTT] for k=2
|
|
18
18
|
Signature: void mlk_polyvec_basemul_acc_montgomery_cached_k2_aarch64_asm(int16_t r[256], const int16_t a[512], const int16_t b[512], const int16_t b_cache[256])
|
|
19
19
|
ABI:
|
|
20
|
+
Architecture: aarch64
|
|
21
|
+
CallingConvention: AAPCS64
|
|
22
|
+
Features: [NEON]
|
|
20
23
|
x0:
|
|
21
24
|
type: buffer
|
|
22
25
|
size_bytes: 512
|
|
@@ -53,7 +56,7 @@
|
|
|
53
56
|
|
|
54
57
|
/*
|
|
55
58
|
* WARNING: This file is auto-derived from the mlkem-native source file
|
|
56
|
-
* dev/aarch64_opt/src/
|
|
59
|
+
* dev/aarch64_opt/src/mlkem_polyvec_basemul_acc_montgomery_cached_k2_aarch64_asm.S using scripts/simpasm. Do not modify it directly.
|
|
57
60
|
*/
|
|
58
61
|
|
|
59
62
|
.text
|
|
@@ -156,7 +159,7 @@ MLK_ASM_FN_SYMBOL(polyvec_basemul_acc_montgomery_cached_k2_aarch64_asm)
|
|
|
156
159
|
uzp2 v19.8h, v11.8h, v4.8h
|
|
157
160
|
sub x13, x13, #0x2
|
|
158
161
|
|
|
159
|
-
|
|
162
|
+
Lmlk_polyvec_basemul_acc_montgomery_cached_k2_loop_start:
|
|
160
163
|
smlal v18.4s, v16.4h, v17.4h
|
|
161
164
|
ldr q7, [x4], #0x20
|
|
162
165
|
ldr q10, [x2, #0x10]
|
|
@@ -206,7 +209,7 @@ Lpolyvec_basemul_acc_montgomery_cached_k2_loop_start:
|
|
|
206
209
|
smull2 v23.4s, v1.8h, v24.8h
|
|
207
210
|
smull v26.4s, v1.4h, v24.4h
|
|
208
211
|
subs x13, x13, #0x1
|
|
209
|
-
cbnz x13,
|
|
212
|
+
cbnz x13, Lmlk_polyvec_basemul_acc_montgomery_cached_k2_loop_start
|
|
210
213
|
smlal v26.4s, v6.4h, v20.4h
|
|
211
214
|
smlal2 v23.4s, v6.8h, v20.8h
|
|
212
215
|
smlal v26.4s, v16.4h, v30.4h
|
|
@@ -17,6 +17,9 @@
|
|
|
17
17
|
Description: Re-implementation of asymmetric base multiplication following @[NeonNTT] for k=3
|
|
18
18
|
Signature: void mlk_polyvec_basemul_acc_montgomery_cached_k3_aarch64_asm(int16_t r[256], const int16_t a[768], const int16_t b[768], const int16_t b_cache[384])
|
|
19
19
|
ABI:
|
|
20
|
+
Architecture: aarch64
|
|
21
|
+
CallingConvention: AAPCS64
|
|
22
|
+
Features: [NEON]
|
|
20
23
|
x0:
|
|
21
24
|
type: buffer
|
|
22
25
|
size_bytes: 512
|
|
@@ -53,7 +56,7 @@
|
|
|
53
56
|
|
|
54
57
|
/*
|
|
55
58
|
* WARNING: This file is auto-derived from the mlkem-native source file
|
|
56
|
-
* dev/aarch64_opt/src/
|
|
59
|
+
* dev/aarch64_opt/src/mlkem_polyvec_basemul_acc_montgomery_cached_k3_aarch64_asm.S using scripts/simpasm. Do not modify it directly.
|
|
57
60
|
*/
|
|
58
61
|
|
|
59
62
|
.text
|
|
@@ -178,7 +181,7 @@ MLK_ASM_FN_SYMBOL(polyvec_basemul_acc_montgomery_cached_k3_aarch64_asm)
|
|
|
178
181
|
mul v14.8h, v14.8h, v2.8h
|
|
179
182
|
sub x13, x13, #0x2
|
|
180
183
|
|
|
181
|
-
|
|
184
|
+
Lmlk_polyvec_basemul_acc_montgomery_cached_k3_loop_start:
|
|
182
185
|
uzp1 v6.8h, v27.8h, v11.8h
|
|
183
186
|
smlal v26.4s, v29.4h, v24.4h
|
|
184
187
|
uzp2 v16.8h, v25.8h, v3.8h
|
|
@@ -245,7 +248,7 @@ Lpolyvec_basemul_acc_montgomery_cached_k3_loop_start:
|
|
|
245
248
|
stur q10, [x0, #-0x10]
|
|
246
249
|
uzp2 v17.8h, v27.8h, v11.8h
|
|
247
250
|
subs x13, x13, #0x1
|
|
248
|
-
cbnz x13,
|
|
251
|
+
cbnz x13, Lmlk_polyvec_basemul_acc_montgomery_cached_k3_loop_start
|
|
249
252
|
uzp2 v3.8h, v25.8h, v3.8h
|
|
250
253
|
smull2 v16.4s, v1.8h, v20.8h
|
|
251
254
|
smull v25.4s, v1.4h, v20.4h
|
|
@@ -17,6 +17,9 @@
|
|
|
17
17
|
Description: Re-implementation of asymmetric base multiplication following @[NeonNTT] for k=4
|
|
18
18
|
Signature: void mlk_polyvec_basemul_acc_montgomery_cached_k4_aarch64_asm(int16_t r[256], const int16_t a[1024], const int16_t b[1024], const int16_t b_cache[512])
|
|
19
19
|
ABI:
|
|
20
|
+
Architecture: aarch64
|
|
21
|
+
CallingConvention: AAPCS64
|
|
22
|
+
Features: [NEON]
|
|
20
23
|
x0:
|
|
21
24
|
type: buffer
|
|
22
25
|
size_bytes: 512
|
|
@@ -53,7 +56,7 @@
|
|
|
53
56
|
|
|
54
57
|
/*
|
|
55
58
|
* WARNING: This file is auto-derived from the mlkem-native source file
|
|
56
|
-
* dev/aarch64_opt/src/
|
|
59
|
+
* dev/aarch64_opt/src/mlkem_polyvec_basemul_acc_montgomery_cached_k4_aarch64_asm.S using scripts/simpasm. Do not modify it directly.
|
|
57
60
|
*/
|
|
58
61
|
|
|
59
62
|
.text
|
|
@@ -173,7 +176,7 @@ MLK_ASM_FN_SYMBOL(polyvec_basemul_acc_montgomery_cached_k4_aarch64_asm)
|
|
|
173
176
|
mul v29.8h, v28.8h, v2.8h
|
|
174
177
|
sub x13, x13, #0x2
|
|
175
178
|
|
|
176
|
-
|
|
179
|
+
Lmlk_polyvec_basemul_acc_montgomery_cached_k4_loop_start:
|
|
177
180
|
smlal2 v23.4s, v30.8h, v21.8h
|
|
178
181
|
ldr q11, [x1], #0x20
|
|
179
182
|
uzp2 v15.8h, v5.8h, v22.8h
|
|
@@ -257,7 +260,7 @@ Lpolyvec_basemul_acc_montgomery_cached_k4_loop_start:
|
|
|
257
260
|
smlal2 v23.4s, v25.8h, v10.8h
|
|
258
261
|
uzp1 v14.8h, v5.8h, v22.8h
|
|
259
262
|
subs x13, x13, #0x1
|
|
260
|
-
cbnz x13,
|
|
263
|
+
cbnz x13, Lmlk_polyvec_basemul_acc_montgomery_cached_k4_loop_start
|
|
261
264
|
smlal v18.4s, v29.4h, v0.4h
|
|
262
265
|
ldr q11, [x1], #0x20
|
|
263
266
|
uzp2 v28.8h, v5.8h, v22.8h
|