pq_crypto 0.6.5 → 0.6.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +134 -0
- data/README.md +5 -3
- data/ext/pqcrypto/extconf.rb +4 -1
- data/ext/pqcrypto/pq_externalmu.c +35 -0
- data/ext/pqcrypto/pqcrypto_native_api.h +85 -75
- data/ext/pqcrypto/pqcrypto_ruby_secure.c +16 -9
- data/ext/pqcrypto/pqcrypto_secure.c +43 -33
- data/ext/pqcrypto/pqcrypto_secure.h +52 -47
- data/ext/pqcrypto/pqcrypto_version.h +1 -1
- data/ext/pqcrypto/vendor/.vendored +7 -7
- data/ext/pqcrypto/vendor/mldsa-native/BUILDING.md +5 -2
- data/ext/pqcrypto/vendor/mldsa-native/LICENSE +21 -2
- data/ext/pqcrypto/vendor/mldsa-native/README.md +20 -7
- data/ext/pqcrypto/vendor/mldsa-native/RELEASE.md +160 -0
- data/ext/pqcrypto/vendor/mldsa-native/SECURITY.md +1 -1
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/README.md +2 -2
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/mldsa_native.c +85 -59
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/mldsa_native.h +292 -348
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/mldsa_native_asm.S +122 -76
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/mldsa_native_config.h +184 -86
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/cbmc.h +49 -4
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/common.h +49 -81
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/context.h +152 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/ct.h +25 -12
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/debug.c +2 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/debug.h +2 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/fips202x4.c +2 -2
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/keccakf1600.c +9 -11
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/auto.h +19 -11
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x1_scalar_aarch64_asm.S +6 -4
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x1_v84a_aarch64_asm.S +7 -4
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x2_v84a_aarch64_asm.S +7 -4
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm.S +12 -9
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm.S +12 -9
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/x1_scalar.h +1 -1
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/x1_v84a.h +3 -2
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/x2_v84a.h +3 -2
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/x4_v8a_scalar.h +6 -1
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/x4_v8a_v84a_scalar.h +3 -2
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/api.h +11 -11
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/armv81m/mve.h +9 -22
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/armv81m/src/keccak_f1600_x4_mve.S +8 -5
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/armv81m/src/keccak_f1600_x4_mve.c +1 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/armv81m/src/{state_extract_bytes_x4_mve.S → keccak_f1600_x4_state_extract_bytes_mve.S} +14 -14
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/armv81m/src/{state_xor_bytes_x4_mve.S → keccak_f1600_x4_state_xor_bytes_mve.S} +12 -12
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/auto.h +5 -4
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/x86_64/keccak_f1600_x4_avx2.h +2 -2
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/x86_64/src/fips202_native_x86_64.h +1 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/x86_64/src/keccak_f1600_x4_avx2_asm.S +36 -2
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/meta.h +62 -4
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/arith_native_aarch64.h +87 -54
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{intt_aarch64_asm.S → mldsa_intt_aarch64_asm.S} +39 -6
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{ntt_aarch64_asm.S → mldsa_ntt_aarch64_asm.S} +39 -6
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{pointwise_montgomery_aarch64_asm.S → mldsa_pointwise_montgomery_aarch64_asm.S} +25 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{poly_caddq_aarch64_asm.S → mldsa_poly_caddq_aarch64_asm.S} +19 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{poly_chknorm_aarch64_asm.S → mldsa_poly_chknorm_aarch64_asm.S} +24 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{poly_decompose_32_aarch64_asm.S → mldsa_poly_decompose_32_aarch64_asm.S} +25 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{poly_decompose_88_aarch64_asm.S → mldsa_poly_decompose_88_aarch64_asm.S} +25 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{poly_use_hint_32_aarch64_asm.S → mldsa_poly_use_hint_32_aarch64_asm.S} +25 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{poly_use_hint_88_aarch64_asm.S → mldsa_poly_use_hint_88_aarch64_asm.S} +25 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{mld_polyvecl_pointwise_acc_montgomery_l4_aarch64_asm.S → mldsa_polyvecl_pointwise_acc_montgomery_l4_aarch64_asm.S} +31 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{mld_polyvecl_pointwise_acc_montgomery_l5_aarch64_asm.S → mldsa_polyvecl_pointwise_acc_montgomery_l5_aarch64_asm.S} +31 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{mld_polyvecl_pointwise_acc_montgomery_l7_aarch64_asm.S → mldsa_polyvecl_pointwise_acc_montgomery_l7_aarch64_asm.S} +31 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{polyz_unpack_17_aarch64_asm.S → mldsa_polyz_unpack_17_aarch64_asm.S} +31 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{polyz_unpack_19_aarch64_asm.S → mldsa_polyz_unpack_19_aarch64_asm.S} +31 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{rej_uniform_aarch64_asm.S → mldsa_rej_uniform_aarch64_asm.S} +48 -15
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{rej_uniform_eta2_aarch64_asm.S → mldsa_rej_uniform_eta2_aarch64_asm.S} +42 -9
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{rej_uniform_eta4_aarch64_asm.S → mldsa_rej_uniform_eta4_aarch64_asm.S} +42 -9
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/api.h +11 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/meta.h +3 -2
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/meta.h +28 -28
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/arith_native_x86_64.h +171 -49
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{intt_avx2_asm.S → mldsa_intt_avx2_asm.S} +23 -1
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{ntt_avx2_asm.S → mldsa_ntt_avx2_asm.S} +23 -1
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{nttunpack_avx2_asm.S → mldsa_nttunpack_avx2_asm.S} +17 -1
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{pointwise_acc_l4_avx2_asm.S → mldsa_pointwise_acc_l4_avx2_asm.S} +37 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{pointwise_acc_l5_avx2_asm.S → mldsa_pointwise_acc_l5_avx2_asm.S} +37 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{pointwise_acc_l7_avx2_asm.S → mldsa_pointwise_acc_l7_avx2_asm.S} +37 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{pointwise_avx2_asm.S → mldsa_pointwise_avx2_asm.S} +31 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{poly_caddq_avx2_asm.S → mldsa_poly_caddq_avx2_asm.S} +18 -9
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_poly_chknorm_avx2_asm.S +176 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_poly_decompose_32_avx2_asm.S +490 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_poly_decompose_88_avx2_asm.S +489 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_poly_use_hint_32_avx2_asm.S +123 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_poly_use_hint_88_avx2_asm.S +125 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_polyz_unpack_17_avx2_asm.S +355 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_polyz_unpack_19_avx2_asm.S +355 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_rej_uniform_avx2_asm.S +132 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_rej_uniform_eta2_avx2_asm.S +205 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_rej_uniform_eta4_avx2_asm.S +176 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/packing.c +27 -36
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/packing.h +42 -8
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/params.h +93 -17
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/poly.c +74 -15
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/poly.h +97 -11
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/poly_kl.c +7 -38
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/poly_kl.h +49 -7
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/polyvec.c +16 -17
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/polyvec.h +26 -9
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/polyvec_lazy.c +3 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/polyvec_lazy.h +18 -19
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/reduce.h +15 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/rounding.h +28 -6
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/sign.c +311 -246
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/sign.h +245 -240
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/sys.h +64 -5
- data/ext/pqcrypto/vendor/mlkem-native/BUILDING.md +5 -2
- data/ext/pqcrypto/vendor/mlkem-native/LICENSE +21 -3
- data/ext/pqcrypto/vendor/mlkem-native/README.md +2 -2
- data/ext/pqcrypto/vendor/mlkem-native/RELEASE.md +113 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/README.md +2 -2
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/mlkem_native.c +17 -27
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/mlkem_native.h +68 -151
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/mlkem_native_asm.S +17 -27
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/mlkem_native_config.h +46 -44
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/cbmc.h +25 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/common.h +37 -6
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/context.h +9 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/fips202.h +2 -2
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/keccakf1600.c +8 -8
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x1_scalar.h +1 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x1_v84a.h +3 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x2_v84a.h +3 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x4_v8a_scalar.h +2 -2
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x4_v8a_v84a_scalar.h +3 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/api.h +11 -11
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/armv81m/mve.h +3 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/x86_64/keccak_f1600_x4_avx2.h +2 -2
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/indcpa.h +14 -11
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/kem.c +28 -11
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/kem.h +39 -14
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/meta.h +10 -10
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/meta.h +20 -20
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/sys.h +5 -5
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/verify.h +11 -10
- data/lib/pq_crypto/version.rb +1 -1
- data/script/vendor_libs.rb +6 -6
- metadata +40 -38
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/poly_chknorm_avx2.c +0 -52
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/poly_decompose_32_avx2.c +0 -157
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/poly_decompose_88_avx2.c +0 -157
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/poly_use_hint_32_avx2.c +0 -103
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/poly_use_hint_88_avx2.c +0 -105
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/polyz_unpack_17_avx2.c +0 -94
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/polyz_unpack_19_avx2.c +0 -96
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/rej_uniform_avx2.c +0 -126
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/rej_uniform_eta2_avx2.c +0 -157
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/rej_uniform_eta4_avx2.c +0 -141
|
@@ -1,105 +0,0 @@
|
|
|
1
|
-
/*
|
|
2
|
-
* Copyright (c) The mldsa-native project authors
|
|
3
|
-
* SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT
|
|
4
|
-
*/
|
|
5
|
-
|
|
6
|
-
/* References
|
|
7
|
-
* ==========
|
|
8
|
-
*
|
|
9
|
-
* - [REF_AVX2]
|
|
10
|
-
* CRYSTALS-Dilithium optimized AVX2 implementation
|
|
11
|
-
* Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé
|
|
12
|
-
* https://github.com/pq-crystals/dilithium/tree/master/avx2
|
|
13
|
-
*/
|
|
14
|
-
|
|
15
|
-
/*
|
|
16
|
-
* This file is derived from the public domain
|
|
17
|
-
* AVX2 Dilithium implementation @[REF_AVX2].
|
|
18
|
-
*/
|
|
19
|
-
|
|
20
|
-
#include "../../../common.h"
|
|
21
|
-
|
|
22
|
-
#if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \
|
|
23
|
-
!defined(MLD_CONFIG_NO_VERIFY_API) && \
|
|
24
|
-
!defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \
|
|
25
|
-
(defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \
|
|
26
|
-
MLD_CONFIG_PARAMETER_SET == 44)
|
|
27
|
-
|
|
28
|
-
#include <immintrin.h>
|
|
29
|
-
#include "arith_native_x86_64.h"
|
|
30
|
-
#include "consts.h"
|
|
31
|
-
|
|
32
|
-
#define MLD_MM256_BLENDV_EPI32(a, b, mask) \
|
|
33
|
-
_mm256_castps_si256(_mm256_blendv_ps(_mm256_castsi256_ps(a), \
|
|
34
|
-
_mm256_castsi256_ps(b), \
|
|
35
|
-
_mm256_castsi256_ps(mask)))
|
|
36
|
-
|
|
37
|
-
void mld_poly_use_hint_88_avx2(int32_t *a, const int32_t *hint)
|
|
38
|
-
{
|
|
39
|
-
unsigned int i;
|
|
40
|
-
__m256i f, f0, f1, h, t;
|
|
41
|
-
const __m256i q_bound = _mm256_set1_epi32(87 * ((MLDSA_Q - 1) / 88));
|
|
42
|
-
/* check-magic: 11275 == floor(2**24 / 1488) */
|
|
43
|
-
const __m256i v = _mm256_set1_epi32(11275);
|
|
44
|
-
const __m256i alpha = _mm256_set1_epi32(2 * ((MLDSA_Q - 1) / 88));
|
|
45
|
-
const __m256i off = _mm256_set1_epi32(127);
|
|
46
|
-
const __m256i shift = _mm256_set1_epi32(128);
|
|
47
|
-
const __m256i max = _mm256_set1_epi32(43);
|
|
48
|
-
const __m256i zero = _mm256_setzero_si256();
|
|
49
|
-
|
|
50
|
-
for (i = 0; i < MLDSA_N / 8; i++)
|
|
51
|
-
{
|
|
52
|
-
f = _mm256_load_si256((const __m256i *)&a[8 * i]);
|
|
53
|
-
h = _mm256_load_si256((const __m256i *)&hint[8 * i]);
|
|
54
|
-
|
|
55
|
-
/* Reference:
|
|
56
|
-
* - @[REF_AVX2] calls poly_decompose to compute all a1, a0 before the loop.
|
|
57
|
-
* - Our implementation of decompose() is slightly different from that in
|
|
58
|
-
* @[REF_AVX2]. See poly_decompose_88_avx2.c for more information.
|
|
59
|
-
*/
|
|
60
|
-
/* f1, f2 = decompose(f) */
|
|
61
|
-
f1 = _mm256_add_epi32(f, off);
|
|
62
|
-
f1 = _mm256_srli_epi32(f1, 7);
|
|
63
|
-
f1 = _mm256_mulhi_epu16(f1, v);
|
|
64
|
-
f1 = _mm256_mulhrs_epi16(f1, shift);
|
|
65
|
-
t = _mm256_cmpgt_epi32(f, q_bound);
|
|
66
|
-
f0 = _mm256_mullo_epi32(f1, alpha);
|
|
67
|
-
f0 = _mm256_sub_epi32(f, f0);
|
|
68
|
-
f1 = _mm256_andnot_si256(t, f1);
|
|
69
|
-
f0 = _mm256_add_epi32(f0, t);
|
|
70
|
-
|
|
71
|
-
/* Reference: The reference avx2 implementation checks a0 >= 0, which is
|
|
72
|
-
* different from the specification and the reference C implementation. We
|
|
73
|
-
* follow the specification and check a0 > 0.
|
|
74
|
-
*/
|
|
75
|
-
/* t = (f0 > 0) ? h : -h */
|
|
76
|
-
f0 = _mm256_cmpgt_epi32(f0, zero);
|
|
77
|
-
t = MLD_MM256_BLENDV_EPI32(h, zero, f0);
|
|
78
|
-
t = _mm256_slli_epi32(t, 1);
|
|
79
|
-
h = _mm256_sub_epi32(h, t);
|
|
80
|
-
|
|
81
|
-
/* f1 = (f1 + t) % 44 */
|
|
82
|
-
f1 = _mm256_add_epi32(f1, h);
|
|
83
|
-
f1 = MLD_MM256_BLENDV_EPI32(f1, max, f1);
|
|
84
|
-
f = _mm256_cmpgt_epi32(f1, max);
|
|
85
|
-
f1 = MLD_MM256_BLENDV_EPI32(f1, zero, f);
|
|
86
|
-
|
|
87
|
-
_mm256_store_si256((__m256i *)&a[8 * i], f1);
|
|
88
|
-
}
|
|
89
|
-
}
|
|
90
|
-
|
|
91
|
-
#else /* MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_NO_VERIFY_API && \
|
|
92
|
-
!MLD_CONFIG_MULTILEVEL_NO_SHARED && \
|
|
93
|
-
(MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 44) \
|
|
94
|
-
*/
|
|
95
|
-
|
|
96
|
-
MLD_EMPTY_CU(avx2_poly_use_hint_88)
|
|
97
|
-
|
|
98
|
-
#endif /* !(MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_NO_VERIFY_API && \
|
|
99
|
-
!MLD_CONFIG_MULTILEVEL_NO_SHARED && \
|
|
100
|
-
(MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == \
|
|
101
|
-
44)) */
|
|
102
|
-
|
|
103
|
-
/* To facilitate single-compilation-unit (SCU) builds, undefine all macros.
|
|
104
|
-
* Don't modify by hand -- this is auto-generated by scripts/autogen. */
|
|
105
|
-
#undef MLD_MM256_BLENDV_EPI32
|
|
@@ -1,94 +0,0 @@
|
|
|
1
|
-
/*
|
|
2
|
-
* Copyright (c) The mldsa-native project authors
|
|
3
|
-
* SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT
|
|
4
|
-
*/
|
|
5
|
-
|
|
6
|
-
/* References
|
|
7
|
-
* ==========
|
|
8
|
-
*
|
|
9
|
-
* - [REF_AVX2]
|
|
10
|
-
* CRYSTALS-Dilithium optimized AVX2 implementation
|
|
11
|
-
* Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé
|
|
12
|
-
* https://github.com/pq-crystals/dilithium/tree/master/avx2
|
|
13
|
-
*/
|
|
14
|
-
|
|
15
|
-
/*
|
|
16
|
-
* This file is derived from the public domain
|
|
17
|
-
* AVX2 Dilithium implementation @[REF_AVX2].
|
|
18
|
-
*/
|
|
19
|
-
|
|
20
|
-
#include "../../../common.h"
|
|
21
|
-
|
|
22
|
-
#if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \
|
|
23
|
-
(!defined(MLD_CONFIG_NO_SIGN_API) || \
|
|
24
|
-
!defined(MLD_CONFIG_NO_VERIFY_API)) && \
|
|
25
|
-
!defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \
|
|
26
|
-
(defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \
|
|
27
|
-
MLD_CONFIG_PARAMETER_SET == 44)
|
|
28
|
-
|
|
29
|
-
#include <immintrin.h>
|
|
30
|
-
#include "arith_native_x86_64.h"
|
|
31
|
-
|
|
32
|
-
void mld_polyz_unpack_17_avx2(int32_t *r, const uint8_t *a)
|
|
33
|
-
{
|
|
34
|
-
unsigned int i;
|
|
35
|
-
__m256i f;
|
|
36
|
-
__m128i low, high;
|
|
37
|
-
|
|
38
|
-
const __m256i shufbidx = _mm256_set_epi8(
|
|
39
|
-
-1, 31, 30, 29, -1, 29, 28, 27, -1, 27, 26, 25, -1, 25, 24, 23, -1, 8, 7,
|
|
40
|
-
6, -1, 6, 5, 4, -1, 4, 3, 2, -1, 2, 1, 0);
|
|
41
|
-
const __m256i srlvdidx = _mm256_set_epi32(6, 4, 2, 0, 6, 4, 2, 0);
|
|
42
|
-
const __m256i mask = _mm256_set1_epi32(0x3FFFF);
|
|
43
|
-
const __m256i gamma1 = _mm256_set1_epi32((1 << 17));
|
|
44
|
-
|
|
45
|
-
for (i = 0; i < MLDSA_N / 8; i++)
|
|
46
|
-
{
|
|
47
|
-
/* Load bytes 0..15 into low 128-bit vector */
|
|
48
|
-
low = _mm_loadu_si128((__m128i *)&a[18 * i]);
|
|
49
|
-
/* Load bytes 2..17 into high 128-bit vector */
|
|
50
|
-
high = _mm_loadu_si128((__m128i *)&a[18 * i + 2]);
|
|
51
|
-
/* Combine into 256-bit vector */
|
|
52
|
-
f = _mm256_inserti128_si256(_mm256_castsi128_si256(low), high, 1);
|
|
53
|
-
|
|
54
|
-
/* Shuffling 8-bit lanes
|
|
55
|
-
*
|
|
56
|
-
* ┌─ Indices 0-8 into low 128-bit half ───────────────────────────────────┐
|
|
57
|
-
* │ Shuffle: [-1, 8, 7, 6, -1, 6, 5, 4, -1, 4, 3, 2, -1, 2, 1, 0] │
|
|
58
|
-
* │ Result: [0, byte8, byte7, byte6, ..., 0, byte2, byte1, byte0] │
|
|
59
|
-
* └───────────────────────────────────────────────────────────────────────┘
|
|
60
|
-
*
|
|
61
|
-
* ┌─ Indices 16-31 into high 128-bit half ────────────────────────────────┐
|
|
62
|
-
* │ Shuffle: [-1,31, 30, 29, -1,29, 28, 27, -1,27, 26, 25, -1,25, 24, 23] │
|
|
63
|
-
* │ Result: [0, byte17, byte16, byte15, ..., 0, byte11, byte10, byte9] │
|
|
64
|
-
* └───────────────────────────────────────────────────────────────────────┘
|
|
65
|
-
*/
|
|
66
|
-
f = _mm256_shuffle_epi8(f, shufbidx);
|
|
67
|
-
|
|
68
|
-
/* Keep only 18 out of 24 bits in each 32-bit lane */
|
|
69
|
-
/* Bits 0..23 16..39 32..55 48..71
|
|
70
|
-
* 72..95 88..111 104..127 120..143 */
|
|
71
|
-
f = _mm256_srlv_epi32(f, srlvdidx);
|
|
72
|
-
/* Bits 0..23 18..39 36..55 54..71
|
|
73
|
-
* 72..95 90..111 108..127 126..143 */
|
|
74
|
-
f = _mm256_and_si256(f, mask);
|
|
75
|
-
/* Bits 0..17 18..35 36..53 54..71
|
|
76
|
-
* 72..89 90..107 108..125 126..143 */
|
|
77
|
-
|
|
78
|
-
/* Map [0, 1, ..., 2^18-1] to [2^17, 2^17-1, ..., -2^17+1] */
|
|
79
|
-
f = _mm256_sub_epi32(gamma1, f);
|
|
80
|
-
|
|
81
|
-
_mm256_store_si256((__m256i *)&r[8 * i], f);
|
|
82
|
-
}
|
|
83
|
-
}
|
|
84
|
-
#else /* MLD_ARITH_BACKEND_X86_64_DEFAULT && (!MLD_CONFIG_NO_SIGN_API || \
|
|
85
|
-
!MLD_CONFIG_NO_VERIFY_API) && !MLD_CONFIG_MULTILEVEL_NO_SHARED && \
|
|
86
|
-
(MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 44) \
|
|
87
|
-
*/
|
|
88
|
-
|
|
89
|
-
MLD_EMPTY_CU(avx2_polyz_unpack_17)
|
|
90
|
-
|
|
91
|
-
#endif /* !(MLD_ARITH_BACKEND_X86_64_DEFAULT && (!MLD_CONFIG_NO_SIGN_API || \
|
|
92
|
-
!MLD_CONFIG_NO_VERIFY_API) && !MLD_CONFIG_MULTILEVEL_NO_SHARED && \
|
|
93
|
-
(MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == \
|
|
94
|
-
44)) */
|
|
@@ -1,96 +0,0 @@
|
|
|
1
|
-
/*
|
|
2
|
-
* Copyright (c) The mldsa-native project authors
|
|
3
|
-
* SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT
|
|
4
|
-
*/
|
|
5
|
-
|
|
6
|
-
/* References
|
|
7
|
-
* ==========
|
|
8
|
-
*
|
|
9
|
-
* - [REF_AVX2]
|
|
10
|
-
* CRYSTALS-Dilithium optimized AVX2 implementation
|
|
11
|
-
* Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé
|
|
12
|
-
* https://github.com/pq-crystals/dilithium/tree/master/avx2
|
|
13
|
-
*/
|
|
14
|
-
|
|
15
|
-
/*
|
|
16
|
-
* This file is derived from the public domain
|
|
17
|
-
* AVX2 Dilithium implementation @[REF_AVX2].
|
|
18
|
-
*/
|
|
19
|
-
|
|
20
|
-
#include "../../../common.h"
|
|
21
|
-
|
|
22
|
-
#if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \
|
|
23
|
-
(!defined(MLD_CONFIG_NO_SIGN_API) || \
|
|
24
|
-
!defined(MLD_CONFIG_NO_VERIFY_API)) && \
|
|
25
|
-
!defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \
|
|
26
|
-
(defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \
|
|
27
|
-
(MLD_CONFIG_PARAMETER_SET == 65 || MLD_CONFIG_PARAMETER_SET == 87))
|
|
28
|
-
|
|
29
|
-
#include <immintrin.h>
|
|
30
|
-
#include "arith_native_x86_64.h"
|
|
31
|
-
|
|
32
|
-
void mld_polyz_unpack_19_avx2(int32_t *r, const uint8_t *a)
|
|
33
|
-
{
|
|
34
|
-
unsigned int i;
|
|
35
|
-
__m256i f;
|
|
36
|
-
__m128i low, high;
|
|
37
|
-
|
|
38
|
-
const __m256i shufbidx = _mm256_set_epi8(
|
|
39
|
-
-1, 31, 30, 29, -1, 29, 28, 27, -1, 26, 25, 24, -1, 24, 23, 22, -1, 9, 8,
|
|
40
|
-
7, -1, 7, 6, 5, -1, 4, 3, 2, -1, 2, 1, 0);
|
|
41
|
-
/* Equivalent to _mm256_set_epi32(4, 0, 4, 0, 4, 0, 4, 0) */
|
|
42
|
-
const __m256i srlvdidx = _mm256_set1_epi64x((uint64_t)4 << 32);
|
|
43
|
-
const __m256i mask = _mm256_set1_epi32(0xFFFFF);
|
|
44
|
-
const __m256i gamma1 = _mm256_set1_epi32((1 << 19));
|
|
45
|
-
|
|
46
|
-
for (i = 0; i < MLDSA_N / 8; i++)
|
|
47
|
-
{
|
|
48
|
-
/* Load bytes 0..15 into low 128-bit vector */
|
|
49
|
-
low = _mm_loadu_si128((__m128i *)&a[20 * i]);
|
|
50
|
-
/* Load bytes 4..19 into high 128-bit vector */
|
|
51
|
-
high = _mm_loadu_si128((__m128i *)&a[20 * i + 4]);
|
|
52
|
-
/* Combine into 256-bit vector */
|
|
53
|
-
f = _mm256_inserti128_si256(_mm256_castsi128_si256(low), high, 1);
|
|
54
|
-
|
|
55
|
-
/* Shuffling 8-bit lanes
|
|
56
|
-
*
|
|
57
|
-
* ┌─ Indices 0-9 into low 128-bit half ───────────────────────────────────┐
|
|
58
|
-
* │ Shuffle: [-1, 9, 8, 7, -1, 7, 6, 5, -1, 4, 3, 2, -1, 2, 1, 0] │
|
|
59
|
-
* │ Result: [0, byte9, byte8, byte7, ..., 0, byte2, byte1, byte0] │
|
|
60
|
-
* └───────────────────────────────────────────────────────────────────────┘
|
|
61
|
-
*
|
|
62
|
-
* ┌─ Indices 16-31 into high 128-bit half ────────────────────────────────┐
|
|
63
|
-
* │ Shuffle: [-1,31, 30, 29, -1,29, 28, 27, -1,26, 25, 24, -1,24, 23, 22] │
|
|
64
|
-
* │ Result: [0, byte19, byte18, byte17, ..., 0, byte12, byte11, byte10] │
|
|
65
|
-
* └───────────────────────────────────────────────────────────────────────┘
|
|
66
|
-
*/
|
|
67
|
-
f = _mm256_shuffle_epi8(f, shufbidx);
|
|
68
|
-
|
|
69
|
-
/* Keep only 20 out of 24 bits in each 32-bit lane */
|
|
70
|
-
/* Bits 0..23 16..39 40..63 56..79
|
|
71
|
-
* 80..103 96..119 120..143 136..159 */
|
|
72
|
-
f = _mm256_srlv_epi32(f, srlvdidx);
|
|
73
|
-
/* Bits 0..23 20..39 40..63 60..79
|
|
74
|
-
* 80..103 100..119 120..143 140..159 */
|
|
75
|
-
f = _mm256_and_si256(f, mask);
|
|
76
|
-
/* Bits 0..19 20..39 40..59 60..79
|
|
77
|
-
* 80..99 100..119 120..139 140..159 */
|
|
78
|
-
|
|
79
|
-
/* Map [0, 1, ..., 2^20-1] to [2^19, 2^19-1, ..., -2^19+1] */
|
|
80
|
-
f = _mm256_sub_epi32(gamma1, f);
|
|
81
|
-
|
|
82
|
-
_mm256_store_si256((__m256i *)&r[8 * i], f);
|
|
83
|
-
}
|
|
84
|
-
}
|
|
85
|
-
|
|
86
|
-
#else /* MLD_ARITH_BACKEND_X86_64_DEFAULT && (!MLD_CONFIG_NO_SIGN_API || \
|
|
87
|
-
!MLD_CONFIG_NO_VERIFY_API) && !MLD_CONFIG_MULTILEVEL_NO_SHARED && \
|
|
88
|
-
(MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 65 \
|
|
89
|
-
|| MLD_CONFIG_PARAMETER_SET == 87) */
|
|
90
|
-
|
|
91
|
-
MLD_EMPTY_CU(avx2_polyz_unpack_19)
|
|
92
|
-
|
|
93
|
-
#endif /* !(MLD_ARITH_BACKEND_X86_64_DEFAULT && (!MLD_CONFIG_NO_SIGN_API || \
|
|
94
|
-
!MLD_CONFIG_NO_VERIFY_API) && !MLD_CONFIG_MULTILEVEL_NO_SHARED && \
|
|
95
|
-
(MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 65 \
|
|
96
|
-
|| MLD_CONFIG_PARAMETER_SET == 87)) */
|
|
@@ -1,126 +0,0 @@
|
|
|
1
|
-
/*
|
|
2
|
-
* Copyright (c) The mldsa-native project authors
|
|
3
|
-
* SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT
|
|
4
|
-
*/
|
|
5
|
-
|
|
6
|
-
/* References
|
|
7
|
-
* ==========
|
|
8
|
-
*
|
|
9
|
-
* - [REF_AVX2]
|
|
10
|
-
* CRYSTALS-Dilithium optimized AVX2 implementation
|
|
11
|
-
* Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé
|
|
12
|
-
* https://github.com/pq-crystals/dilithium/tree/master/avx2
|
|
13
|
-
*/
|
|
14
|
-
|
|
15
|
-
/*
|
|
16
|
-
* This file is derived from the public domain
|
|
17
|
-
* AVX2 Dilithium implementation @[REF_AVX2].
|
|
18
|
-
*/
|
|
19
|
-
|
|
20
|
-
#include "../../../common.h"
|
|
21
|
-
|
|
22
|
-
#if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \
|
|
23
|
-
!defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)
|
|
24
|
-
|
|
25
|
-
#include <immintrin.h>
|
|
26
|
-
#include "arith_native_x86_64.h"
|
|
27
|
-
#include "consts.h"
|
|
28
|
-
|
|
29
|
-
/*
|
|
30
|
-
* Reference: The pqcrystals implementation assumes a buffer that is 8 bytes
|
|
31
|
-
*. larger as the first loop overreads by 8 bytes that are then
|
|
32
|
-
* discarded. We instead do not pad the buffer and do not overread.
|
|
33
|
-
* The performance impact is negligible and it does not force the
|
|
34
|
-
* frontend to perform the unintuitive padding.
|
|
35
|
-
*/
|
|
36
|
-
|
|
37
|
-
unsigned int mld_rej_uniform_avx2(
|
|
38
|
-
int32_t *MLD_RESTRICT r, const uint8_t buf[MLD_AVX2_REJ_UNIFORM_BUFLEN])
|
|
39
|
-
{
|
|
40
|
-
unsigned int ctr, pos;
|
|
41
|
-
uint32_t good;
|
|
42
|
-
__m256i d, tmp;
|
|
43
|
-
const __m256i bound = _mm256_set1_epi32(MLDSA_Q);
|
|
44
|
-
const __m256i mask = _mm256_set1_epi32(0x7FFFFF);
|
|
45
|
-
const __m256i idx8 =
|
|
46
|
-
_mm256_set_epi8(-1, 15, 14, 13, -1, 12, 11, 10, -1, 9, 8, 7, -1, 6, 5, 4,
|
|
47
|
-
-1, 11, 10, 9, -1, 8, 7, 6, -1, 5, 4, 3, -1, 2, 1, 0);
|
|
48
|
-
|
|
49
|
-
ctr = pos = 0;
|
|
50
|
-
while (ctr <= MLDSA_N - 8 && pos <= MLD_AVX2_REJ_UNIFORM_BUFLEN - 32)
|
|
51
|
-
{
|
|
52
|
-
d = _mm256_loadu_si256((__m256i *)&buf[pos]);
|
|
53
|
-
|
|
54
|
-
/* Permute 64-bit lanes
|
|
55
|
-
* 0x94 = 10010100b rearranges 64-bit lanes as: [3,2,1,0] -> [2,1,1,0]
|
|
56
|
-
*
|
|
57
|
-
* ╔═══════════════════════════════════════════════════════════════════════╗
|
|
58
|
-
* ║ Original Layout ║
|
|
59
|
-
* ╚═══════════════════════════════════════════════════════════════════════╝
|
|
60
|
-
* ┌─────────────────┬─────────────────┬─────────────────┬─────────────────┐
|
|
61
|
-
* │ Lane 0 │ Lane 1 │ Lane 2 │ Lane 3 │
|
|
62
|
-
* │ bytes 0..7 │ bytes 8..15 │ bytes 16..23 │ bytes 24..31 │
|
|
63
|
-
* └─────────────────┴─────────────────┴─────────────────┴─────────────────┘
|
|
64
|
-
*
|
|
65
|
-
* ╔═══════════════════════════════════════════════════════════════════════╗
|
|
66
|
-
* ║ Layout after permute ║
|
|
67
|
-
* ║ Byte indices in high half shifted down by 8 positions ║
|
|
68
|
-
* ╚═══════════════════════════════════════════════════════════════════════╝
|
|
69
|
-
* ┌───────────────┬─────────────────┐ ┌─────────────────┬─────────────────┐
|
|
70
|
-
* │ Lane 0 │ Lane 1 │ │ Lane 2 │ Lane 3 │
|
|
71
|
-
* │ bytes 0..7 │ bytes 8..15 │ │ bytes 8..15 │ bytes 16..23 │
|
|
72
|
-
* └───────────────┴─────────────────┘ └─────────────────┴─────────────────┘
|
|
73
|
-
* Lower 128-bit lane (bytes 0-15) Upper 128-bit lane (bytes 16-31)
|
|
74
|
-
*/
|
|
75
|
-
d = _mm256_permute4x64_epi64(d, 0x94);
|
|
76
|
-
|
|
77
|
-
/* Shuffling 8-bit lanes
|
|
78
|
-
*
|
|
79
|
-
* ┌─ Indices 0-11 into low 128-bit half of permuted vector────────────────┐
|
|
80
|
-
* │ Shuffle: [-1, 11, 10, 9, -1, 8, 7, 6, -1, 5, 4, 3, -1, 2, 1, 0] │
|
|
81
|
-
* │ Result: [0, byte11, byte10, byte9, ..., 0, byte2, byte1, byte0] │
|
|
82
|
-
* └───────────────────────────────────────────────────────────────────────┘
|
|
83
|
-
*
|
|
84
|
-
* ┌─ Indices 4-15 into high 128-bit half of permuted vector ──────────────┐
|
|
85
|
-
* │ Shuffle: [-1, 15, 14, 13, -1, 12, 11, 10, -1, 9, 8, 7, -1, 6, 5, 4] │
|
|
86
|
-
* │ Result: [0, byte23, byte22, byte21, ..., 0, byte14, byte13, byte12 │
|
|
87
|
-
* └───────────────────────────────────────────────────────────────────────┘
|
|
88
|
-
*/
|
|
89
|
-
d = _mm256_shuffle_epi8(d, idx8);
|
|
90
|
-
d = _mm256_and_si256(d, mask);
|
|
91
|
-
pos += 24;
|
|
92
|
-
|
|
93
|
-
tmp = _mm256_sub_epi32(d, bound);
|
|
94
|
-
good = (uint32_t)_mm256_movemask_ps((__m256)tmp);
|
|
95
|
-
tmp = _mm256_cvtepu8_epi32(
|
|
96
|
-
_mm_loadl_epi64((__m128i *)&mld_rej_uniform_table[good]));
|
|
97
|
-
d = _mm256_permutevar8x32_epi32(d, tmp);
|
|
98
|
-
|
|
99
|
-
_mm256_storeu_si256((__m256i *)&r[ctr], d);
|
|
100
|
-
ctr += (unsigned)_mm_popcnt_u32(good);
|
|
101
|
-
}
|
|
102
|
-
|
|
103
|
-
while (ctr < MLDSA_N && pos <= MLD_AVX2_REJ_UNIFORM_BUFLEN - 3)
|
|
104
|
-
{
|
|
105
|
-
uint32_t t = buf[pos++];
|
|
106
|
-
t |= (uint32_t)buf[pos++] << 8;
|
|
107
|
-
t |= (uint32_t)buf[pos++] << 16;
|
|
108
|
-
t &= 0x7FFFFF;
|
|
109
|
-
|
|
110
|
-
if (t < MLDSA_Q)
|
|
111
|
-
{
|
|
112
|
-
/* Safe because t < MLDSA_Q. */
|
|
113
|
-
r[ctr++] = (int32_t)t;
|
|
114
|
-
}
|
|
115
|
-
}
|
|
116
|
-
|
|
117
|
-
return ctr;
|
|
118
|
-
}
|
|
119
|
-
|
|
120
|
-
#else /* MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_MULTILEVEL_NO_SHARED \
|
|
121
|
-
*/
|
|
122
|
-
|
|
123
|
-
MLD_EMPTY_CU(avx2_rej_uniform)
|
|
124
|
-
|
|
125
|
-
#endif /* !(MLD_ARITH_BACKEND_X86_64_DEFAULT && \
|
|
126
|
-
!MLD_CONFIG_MULTILEVEL_NO_SHARED) */
|
|
@@ -1,157 +0,0 @@
|
|
|
1
|
-
/*
|
|
2
|
-
* Copyright (c) The mldsa-native project authors
|
|
3
|
-
* SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT
|
|
4
|
-
*/
|
|
5
|
-
|
|
6
|
-
/* References
|
|
7
|
-
* ==========
|
|
8
|
-
*
|
|
9
|
-
* - [REF_AVX2]
|
|
10
|
-
* CRYSTALS-Dilithium optimized AVX2 implementation
|
|
11
|
-
* Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé
|
|
12
|
-
* https://github.com/pq-crystals/dilithium/tree/master/avx2
|
|
13
|
-
*/
|
|
14
|
-
|
|
15
|
-
/*
|
|
16
|
-
* This file is derived from the public domain
|
|
17
|
-
* AVX2 Dilithium implementation @[REF_AVX2].
|
|
18
|
-
*/
|
|
19
|
-
|
|
20
|
-
#include "../../../common.h"
|
|
21
|
-
|
|
22
|
-
#if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \
|
|
23
|
-
!defined(MLD_CONFIG_NO_KEYPAIR_API) && \
|
|
24
|
-
!defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \
|
|
25
|
-
(defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLDSA_ETA == 2)
|
|
26
|
-
|
|
27
|
-
#include <immintrin.h>
|
|
28
|
-
#include "arith_native_x86_64.h"
|
|
29
|
-
#include "consts.h"
|
|
30
|
-
|
|
31
|
-
#define MLD_AVX2_ETA2 2
|
|
32
|
-
|
|
33
|
-
/*
|
|
34
|
-
* Reference: In the pqcrystals implementation this function is called
|
|
35
|
-
* rej_eta_avx and supports multiple values for ETA via preprocessor
|
|
36
|
-
* conditionals. We move the conditionals to the frontend.
|
|
37
|
-
*/
|
|
38
|
-
unsigned int mld_rej_uniform_eta2_avx2(
|
|
39
|
-
int32_t *MLD_RESTRICT r,
|
|
40
|
-
const uint8_t buf[MLD_AVX2_REJ_UNIFORM_ETA2_BUFLEN])
|
|
41
|
-
{
|
|
42
|
-
unsigned int ctr, pos;
|
|
43
|
-
uint32_t good;
|
|
44
|
-
__m256i f0, f1, f2;
|
|
45
|
-
__m128i g0, g1;
|
|
46
|
-
const __m256i mask = _mm256_set1_epi8(15);
|
|
47
|
-
const __m256i eta = _mm256_set1_epi8(MLD_AVX2_ETA2);
|
|
48
|
-
const __m256i bound = mask;
|
|
49
|
-
/* check-magic: -6560 == 32*round(-2**10 / 5) */
|
|
50
|
-
const __m256i v = _mm256_set1_epi32(-6560);
|
|
51
|
-
const __m256i p = _mm256_set1_epi32(5);
|
|
52
|
-
|
|
53
|
-
ctr = pos = 0;
|
|
54
|
-
while (ctr <= MLDSA_N - 8 && pos <= MLD_AVX2_REJ_UNIFORM_ETA2_BUFLEN - 16)
|
|
55
|
-
{
|
|
56
|
-
f0 = _mm256_cvtepu8_epi16(_mm_loadu_si128((__m128i *)&buf[pos]));
|
|
57
|
-
f1 = _mm256_slli_epi16(f0, 4);
|
|
58
|
-
f0 = _mm256_or_si256(f0, f1);
|
|
59
|
-
f0 = _mm256_and_si256(f0, mask);
|
|
60
|
-
|
|
61
|
-
f1 = _mm256_sub_epi8(f0, bound);
|
|
62
|
-
f0 = _mm256_sub_epi8(eta, f0);
|
|
63
|
-
good = (uint32_t)_mm256_movemask_epi8(f1);
|
|
64
|
-
|
|
65
|
-
g0 = _mm256_castsi256_si128(f0);
|
|
66
|
-
g1 = _mm_loadl_epi64((__m128i *)&mld_rej_uniform_table[good & 0xFF]);
|
|
67
|
-
g1 = _mm_shuffle_epi8(g0, g1);
|
|
68
|
-
f1 = _mm256_cvtepi8_epi32(g1);
|
|
69
|
-
f2 = _mm256_mulhrs_epi16(f1, v);
|
|
70
|
-
f2 = _mm256_mullo_epi16(f2, p);
|
|
71
|
-
f1 = _mm256_add_epi32(f1, f2);
|
|
72
|
-
_mm256_storeu_si256((__m256i *)&r[ctr], f1);
|
|
73
|
-
ctr += (unsigned)_mm_popcnt_u32(good & 0xFF);
|
|
74
|
-
good >>= 8;
|
|
75
|
-
pos += 4;
|
|
76
|
-
|
|
77
|
-
if (ctr > MLDSA_N - 8)
|
|
78
|
-
{
|
|
79
|
-
break;
|
|
80
|
-
}
|
|
81
|
-
g0 = _mm_bsrli_si128(g0, 8);
|
|
82
|
-
g1 = _mm_loadl_epi64((__m128i *)&mld_rej_uniform_table[good & 0xFF]);
|
|
83
|
-
g1 = _mm_shuffle_epi8(g0, g1);
|
|
84
|
-
f1 = _mm256_cvtepi8_epi32(g1);
|
|
85
|
-
f2 = _mm256_mulhrs_epi16(f1, v);
|
|
86
|
-
f2 = _mm256_mullo_epi16(f2, p);
|
|
87
|
-
f1 = _mm256_add_epi32(f1, f2);
|
|
88
|
-
_mm256_storeu_si256((__m256i *)&r[ctr], f1);
|
|
89
|
-
ctr += (unsigned)_mm_popcnt_u32(good & 0xFF);
|
|
90
|
-
good >>= 8;
|
|
91
|
-
pos += 4;
|
|
92
|
-
|
|
93
|
-
if (ctr > MLDSA_N - 8)
|
|
94
|
-
{
|
|
95
|
-
break;
|
|
96
|
-
}
|
|
97
|
-
g0 = _mm256_extracti128_si256(f0, 1);
|
|
98
|
-
g1 = _mm_loadl_epi64((__m128i *)&mld_rej_uniform_table[good & 0xFF]);
|
|
99
|
-
g1 = _mm_shuffle_epi8(g0, g1);
|
|
100
|
-
f1 = _mm256_cvtepi8_epi32(g1);
|
|
101
|
-
f2 = _mm256_mulhrs_epi16(f1, v);
|
|
102
|
-
f2 = _mm256_mullo_epi16(f2, p);
|
|
103
|
-
f1 = _mm256_add_epi32(f1, f2);
|
|
104
|
-
_mm256_storeu_si256((__m256i *)&r[ctr], f1);
|
|
105
|
-
ctr += (unsigned)_mm_popcnt_u32(good & 0xFF);
|
|
106
|
-
good >>= 8;
|
|
107
|
-
pos += 4;
|
|
108
|
-
|
|
109
|
-
if (ctr > MLDSA_N - 8)
|
|
110
|
-
{
|
|
111
|
-
break;
|
|
112
|
-
}
|
|
113
|
-
g0 = _mm_bsrli_si128(g0, 8);
|
|
114
|
-
g1 = _mm_loadl_epi64((__m128i *)&mld_rej_uniform_table[good]);
|
|
115
|
-
g1 = _mm_shuffle_epi8(g0, g1);
|
|
116
|
-
f1 = _mm256_cvtepi8_epi32(g1);
|
|
117
|
-
f2 = _mm256_mulhrs_epi16(f1, v);
|
|
118
|
-
f2 = _mm256_mullo_epi16(f2, p);
|
|
119
|
-
f1 = _mm256_add_epi32(f1, f2);
|
|
120
|
-
_mm256_storeu_si256((__m256i *)&r[ctr], f1);
|
|
121
|
-
ctr += (unsigned)_mm_popcnt_u32(good);
|
|
122
|
-
pos += 4;
|
|
123
|
-
}
|
|
124
|
-
|
|
125
|
-
while (ctr < MLDSA_N && pos < MLD_AVX2_REJ_UNIFORM_ETA2_BUFLEN)
|
|
126
|
-
{
|
|
127
|
-
uint32_t t0 = buf[pos] & 0x0F;
|
|
128
|
-
uint32_t t1 = buf[pos++] >> 4;
|
|
129
|
-
|
|
130
|
-
if (t0 < 15)
|
|
131
|
-
{
|
|
132
|
-
t0 = t0 - (205 * t0 >> 10) * 5;
|
|
133
|
-
r[ctr++] = (int32_t)(2 - t0);
|
|
134
|
-
}
|
|
135
|
-
if (t1 < 15 && ctr < MLDSA_N)
|
|
136
|
-
{
|
|
137
|
-
t1 = t1 - (205 * t1 >> 10) * 5;
|
|
138
|
-
r[ctr++] = (int32_t)(2 - t1);
|
|
139
|
-
}
|
|
140
|
-
}
|
|
141
|
-
|
|
142
|
-
return ctr;
|
|
143
|
-
}
|
|
144
|
-
|
|
145
|
-
#else /* MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_NO_KEYPAIR_API && \
|
|
146
|
-
!MLD_CONFIG_MULTILEVEL_NO_SHARED && \
|
|
147
|
-
(MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLDSA_ETA == 2) */
|
|
148
|
-
|
|
149
|
-
MLD_EMPTY_CU(avx2_rej_uniform_eta2)
|
|
150
|
-
|
|
151
|
-
#endif /* !(MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_NO_KEYPAIR_API && \
|
|
152
|
-
!MLD_CONFIG_MULTILEVEL_NO_SHARED && \
|
|
153
|
-
(MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLDSA_ETA == 2)) */
|
|
154
|
-
|
|
155
|
-
/* To facilitate single-compilation-unit (SCU) builds, undefine all macros.
|
|
156
|
-
* Don't modify by hand -- this is auto-generated by scripts/autogen. */
|
|
157
|
-
#undef MLD_AVX2_ETA2
|