pq_crypto 0.6.5 → 0.6.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +233 -0
- data/README.md +16 -3
- data/SECURITY.md +46 -0
- data/ext/pqcrypto/extconf.rb +8 -1
- data/ext/pqcrypto/pq_externalmu.c +35 -0
- data/ext/pqcrypto/pqcrypto_native_api.h +91 -75
- data/ext/pqcrypto/pqcrypto_ruby_secure.c +97 -9
- data/ext/pqcrypto/pqcrypto_secure.c +95 -33
- data/ext/pqcrypto/pqcrypto_secure.h +66 -48
- data/ext/pqcrypto/pqcrypto_version.h +1 -1
- data/ext/pqcrypto/vendor/.vendored +7 -7
- data/ext/pqcrypto/vendor/mldsa-native/BUILDING.md +5 -2
- data/ext/pqcrypto/vendor/mldsa-native/LICENSE +21 -2
- data/ext/pqcrypto/vendor/mldsa-native/README.md +20 -7
- data/ext/pqcrypto/vendor/mldsa-native/RELEASE.md +160 -0
- data/ext/pqcrypto/vendor/mldsa-native/SECURITY.md +1 -1
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/README.md +2 -2
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/mldsa_native.c +85 -59
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/mldsa_native.h +292 -348
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/mldsa_native_asm.S +122 -76
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/mldsa_native_config.h +184 -86
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/cbmc.h +49 -4
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/common.h +49 -81
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/context.h +152 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/ct.h +25 -12
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/debug.c +2 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/debug.h +2 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/fips202x4.c +2 -2
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/keccakf1600.c +9 -11
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/auto.h +19 -11
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x1_scalar_aarch64_asm.S +6 -4
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x1_v84a_aarch64_asm.S +7 -4
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x2_v84a_aarch64_asm.S +7 -4
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm.S +12 -9
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm.S +12 -9
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/x1_scalar.h +1 -1
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/x1_v84a.h +3 -2
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/x2_v84a.h +3 -2
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/x4_v8a_scalar.h +6 -1
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/x4_v8a_v84a_scalar.h +3 -2
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/api.h +11 -11
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/armv81m/mve.h +9 -22
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/armv81m/src/keccak_f1600_x4_mve.S +8 -5
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/armv81m/src/keccak_f1600_x4_mve.c +1 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/armv81m/src/{state_extract_bytes_x4_mve.S → keccak_f1600_x4_state_extract_bytes_mve.S} +14 -14
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/armv81m/src/{state_xor_bytes_x4_mve.S → keccak_f1600_x4_state_xor_bytes_mve.S} +12 -12
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/auto.h +5 -4
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/x86_64/keccak_f1600_x4_avx2.h +2 -2
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/x86_64/src/fips202_native_x86_64.h +1 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/x86_64/src/keccak_f1600_x4_avx2_asm.S +36 -2
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/meta.h +62 -4
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/arith_native_aarch64.h +87 -54
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{intt_aarch64_asm.S → mldsa_intt_aarch64_asm.S} +39 -6
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{ntt_aarch64_asm.S → mldsa_ntt_aarch64_asm.S} +39 -6
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{pointwise_montgomery_aarch64_asm.S → mldsa_pointwise_montgomery_aarch64_asm.S} +25 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{poly_caddq_aarch64_asm.S → mldsa_poly_caddq_aarch64_asm.S} +19 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{poly_chknorm_aarch64_asm.S → mldsa_poly_chknorm_aarch64_asm.S} +24 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{poly_decompose_32_aarch64_asm.S → mldsa_poly_decompose_32_aarch64_asm.S} +25 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{poly_decompose_88_aarch64_asm.S → mldsa_poly_decompose_88_aarch64_asm.S} +25 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{poly_use_hint_32_aarch64_asm.S → mldsa_poly_use_hint_32_aarch64_asm.S} +25 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{poly_use_hint_88_aarch64_asm.S → mldsa_poly_use_hint_88_aarch64_asm.S} +25 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{mld_polyvecl_pointwise_acc_montgomery_l4_aarch64_asm.S → mldsa_polyvecl_pointwise_acc_montgomery_l4_aarch64_asm.S} +31 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{mld_polyvecl_pointwise_acc_montgomery_l5_aarch64_asm.S → mldsa_polyvecl_pointwise_acc_montgomery_l5_aarch64_asm.S} +31 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{mld_polyvecl_pointwise_acc_montgomery_l7_aarch64_asm.S → mldsa_polyvecl_pointwise_acc_montgomery_l7_aarch64_asm.S} +31 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{polyz_unpack_17_aarch64_asm.S → mldsa_polyz_unpack_17_aarch64_asm.S} +31 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{polyz_unpack_19_aarch64_asm.S → mldsa_polyz_unpack_19_aarch64_asm.S} +31 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{rej_uniform_aarch64_asm.S → mldsa_rej_uniform_aarch64_asm.S} +48 -15
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{rej_uniform_eta2_aarch64_asm.S → mldsa_rej_uniform_eta2_aarch64_asm.S} +42 -9
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{rej_uniform_eta4_aarch64_asm.S → mldsa_rej_uniform_eta4_aarch64_asm.S} +42 -9
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/api.h +11 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/meta.h +3 -2
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/meta.h +28 -28
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/arith_native_x86_64.h +171 -49
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{intt_avx2_asm.S → mldsa_intt_avx2_asm.S} +23 -1
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{ntt_avx2_asm.S → mldsa_ntt_avx2_asm.S} +23 -1
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{nttunpack_avx2_asm.S → mldsa_nttunpack_avx2_asm.S} +17 -1
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{pointwise_acc_l4_avx2_asm.S → mldsa_pointwise_acc_l4_avx2_asm.S} +37 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{pointwise_acc_l5_avx2_asm.S → mldsa_pointwise_acc_l5_avx2_asm.S} +37 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{pointwise_acc_l7_avx2_asm.S → mldsa_pointwise_acc_l7_avx2_asm.S} +37 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{pointwise_avx2_asm.S → mldsa_pointwise_avx2_asm.S} +31 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{poly_caddq_avx2_asm.S → mldsa_poly_caddq_avx2_asm.S} +18 -9
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_poly_chknorm_avx2_asm.S +176 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_poly_decompose_32_avx2_asm.S +490 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_poly_decompose_88_avx2_asm.S +489 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_poly_use_hint_32_avx2_asm.S +123 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_poly_use_hint_88_avx2_asm.S +125 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_polyz_unpack_17_avx2_asm.S +355 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_polyz_unpack_19_avx2_asm.S +355 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_rej_uniform_avx2_asm.S +132 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_rej_uniform_eta2_avx2_asm.S +205 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_rej_uniform_eta4_avx2_asm.S +176 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/packing.c +27 -36
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/packing.h +42 -8
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/params.h +93 -17
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/poly.c +74 -15
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/poly.h +97 -11
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/poly_kl.c +7 -38
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/poly_kl.h +49 -7
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/polyvec.c +16 -17
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/polyvec.h +26 -9
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/polyvec_lazy.c +3 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/polyvec_lazy.h +18 -19
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/reduce.h +15 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/rounding.h +28 -6
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/sign.c +311 -246
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/sign.h +245 -240
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/sys.h +64 -5
- data/ext/pqcrypto/vendor/mlkem-native/BUILDING.md +5 -2
- data/ext/pqcrypto/vendor/mlkem-native/LICENSE +21 -3
- data/ext/pqcrypto/vendor/mlkem-native/README.md +2 -2
- data/ext/pqcrypto/vendor/mlkem-native/RELEASE.md +113 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/README.md +2 -2
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/mlkem_native.c +17 -27
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/mlkem_native.h +68 -151
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/mlkem_native_asm.S +17 -27
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/mlkem_native_config.h +46 -44
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/cbmc.h +25 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/common.h +37 -6
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/context.h +9 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/fips202.h +2 -2
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/keccakf1600.c +8 -8
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x1_scalar.h +1 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x1_v84a.h +3 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x2_v84a.h +3 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x4_v8a_scalar.h +2 -2
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x4_v8a_v84a_scalar.h +3 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/api.h +11 -11
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/armv81m/mve.h +3 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/x86_64/keccak_f1600_x4_avx2.h +2 -2
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/indcpa.h +14 -11
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/kem.c +28 -11
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/kem.h +39 -14
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/meta.h +10 -10
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/meta.h +20 -20
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/sys.h +5 -5
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/verify.h +11 -10
- data/lib/pq_crypto/internal.rb +10 -0
- data/lib/pq_crypto/kem.rb +47 -5
- data/lib/pq_crypto/key.rb +14 -8
- data/lib/pq_crypto/pkcs8.rb +13 -5
- data/lib/pq_crypto/signature.rb +34 -9
- data/lib/pq_crypto/spki.rb +4 -2
- data/lib/pq_crypto/version.rb +1 -1
- data/lib/pq_crypto.rb +9 -0
- data/script/vendor_libs.rb +6 -6
- metadata +40 -38
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/poly_chknorm_avx2.c +0 -52
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/poly_decompose_32_avx2.c +0 -157
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/poly_decompose_88_avx2.c +0 -157
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/poly_use_hint_32_avx2.c +0 -103
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/poly_use_hint_88_avx2.c +0 -105
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/polyz_unpack_17_avx2.c +0 -94
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/polyz_unpack_19_avx2.c +0 -96
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/rej_uniform_avx2.c +0 -126
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/rej_uniform_eta2_avx2.c +0 -157
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/rej_uniform_eta4_avx2.c +0 -141
|
@@ -1,96 +0,0 @@
|
|
|
1
|
-
/*
|
|
2
|
-
* Copyright (c) The mldsa-native project authors
|
|
3
|
-
* SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT
|
|
4
|
-
*/
|
|
5
|
-
|
|
6
|
-
/* References
|
|
7
|
-
* ==========
|
|
8
|
-
*
|
|
9
|
-
* - [REF_AVX2]
|
|
10
|
-
* CRYSTALS-Dilithium optimized AVX2 implementation
|
|
11
|
-
* Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé
|
|
12
|
-
* https://github.com/pq-crystals/dilithium/tree/master/avx2
|
|
13
|
-
*/
|
|
14
|
-
|
|
15
|
-
/*
|
|
16
|
-
* This file is derived from the public domain
|
|
17
|
-
* AVX2 Dilithium implementation @[REF_AVX2].
|
|
18
|
-
*/
|
|
19
|
-
|
|
20
|
-
#include "../../../common.h"
|
|
21
|
-
|
|
22
|
-
#if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \
|
|
23
|
-
(!defined(MLD_CONFIG_NO_SIGN_API) || \
|
|
24
|
-
!defined(MLD_CONFIG_NO_VERIFY_API)) && \
|
|
25
|
-
!defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \
|
|
26
|
-
(defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \
|
|
27
|
-
(MLD_CONFIG_PARAMETER_SET == 65 || MLD_CONFIG_PARAMETER_SET == 87))
|
|
28
|
-
|
|
29
|
-
#include <immintrin.h>
|
|
30
|
-
#include "arith_native_x86_64.h"
|
|
31
|
-
|
|
32
|
-
void mld_polyz_unpack_19_avx2(int32_t *r, const uint8_t *a)
|
|
33
|
-
{
|
|
34
|
-
unsigned int i;
|
|
35
|
-
__m256i f;
|
|
36
|
-
__m128i low, high;
|
|
37
|
-
|
|
38
|
-
const __m256i shufbidx = _mm256_set_epi8(
|
|
39
|
-
-1, 31, 30, 29, -1, 29, 28, 27, -1, 26, 25, 24, -1, 24, 23, 22, -1, 9, 8,
|
|
40
|
-
7, -1, 7, 6, 5, -1, 4, 3, 2, -1, 2, 1, 0);
|
|
41
|
-
/* Equivalent to _mm256_set_epi32(4, 0, 4, 0, 4, 0, 4, 0) */
|
|
42
|
-
const __m256i srlvdidx = _mm256_set1_epi64x((uint64_t)4 << 32);
|
|
43
|
-
const __m256i mask = _mm256_set1_epi32(0xFFFFF);
|
|
44
|
-
const __m256i gamma1 = _mm256_set1_epi32((1 << 19));
|
|
45
|
-
|
|
46
|
-
for (i = 0; i < MLDSA_N / 8; i++)
|
|
47
|
-
{
|
|
48
|
-
/* Load bytes 0..15 into low 128-bit vector */
|
|
49
|
-
low = _mm_loadu_si128((__m128i *)&a[20 * i]);
|
|
50
|
-
/* Load bytes 4..19 into high 128-bit vector */
|
|
51
|
-
high = _mm_loadu_si128((__m128i *)&a[20 * i + 4]);
|
|
52
|
-
/* Combine into 256-bit vector */
|
|
53
|
-
f = _mm256_inserti128_si256(_mm256_castsi128_si256(low), high, 1);
|
|
54
|
-
|
|
55
|
-
/* Shuffling 8-bit lanes
|
|
56
|
-
*
|
|
57
|
-
* ┌─ Indices 0-9 into low 128-bit half ───────────────────────────────────┐
|
|
58
|
-
* │ Shuffle: [-1, 9, 8, 7, -1, 7, 6, 5, -1, 4, 3, 2, -1, 2, 1, 0] │
|
|
59
|
-
* │ Result: [0, byte9, byte8, byte7, ..., 0, byte2, byte1, byte0] │
|
|
60
|
-
* └───────────────────────────────────────────────────────────────────────┘
|
|
61
|
-
*
|
|
62
|
-
* ┌─ Indices 16-31 into high 128-bit half ────────────────────────────────┐
|
|
63
|
-
* │ Shuffle: [-1,31, 30, 29, -1,29, 28, 27, -1,26, 25, 24, -1,24, 23, 22] │
|
|
64
|
-
* │ Result: [0, byte19, byte18, byte17, ..., 0, byte12, byte11, byte10] │
|
|
65
|
-
* └───────────────────────────────────────────────────────────────────────┘
|
|
66
|
-
*/
|
|
67
|
-
f = _mm256_shuffle_epi8(f, shufbidx);
|
|
68
|
-
|
|
69
|
-
/* Keep only 20 out of 24 bits in each 32-bit lane */
|
|
70
|
-
/* Bits 0..23 16..39 40..63 56..79
|
|
71
|
-
* 80..103 96..119 120..143 136..159 */
|
|
72
|
-
f = _mm256_srlv_epi32(f, srlvdidx);
|
|
73
|
-
/* Bits 0..23 20..39 40..63 60..79
|
|
74
|
-
* 80..103 100..119 120..143 140..159 */
|
|
75
|
-
f = _mm256_and_si256(f, mask);
|
|
76
|
-
/* Bits 0..19 20..39 40..59 60..79
|
|
77
|
-
* 80..99 100..119 120..139 140..159 */
|
|
78
|
-
|
|
79
|
-
/* Map [0, 1, ..., 2^20-1] to [2^19, 2^19-1, ..., -2^19+1] */
|
|
80
|
-
f = _mm256_sub_epi32(gamma1, f);
|
|
81
|
-
|
|
82
|
-
_mm256_store_si256((__m256i *)&r[8 * i], f);
|
|
83
|
-
}
|
|
84
|
-
}
|
|
85
|
-
|
|
86
|
-
#else /* MLD_ARITH_BACKEND_X86_64_DEFAULT && (!MLD_CONFIG_NO_SIGN_API || \
|
|
87
|
-
!MLD_CONFIG_NO_VERIFY_API) && !MLD_CONFIG_MULTILEVEL_NO_SHARED && \
|
|
88
|
-
(MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 65 \
|
|
89
|
-
|| MLD_CONFIG_PARAMETER_SET == 87) */
|
|
90
|
-
|
|
91
|
-
MLD_EMPTY_CU(avx2_polyz_unpack_19)
|
|
92
|
-
|
|
93
|
-
#endif /* !(MLD_ARITH_BACKEND_X86_64_DEFAULT && (!MLD_CONFIG_NO_SIGN_API || \
|
|
94
|
-
!MLD_CONFIG_NO_VERIFY_API) && !MLD_CONFIG_MULTILEVEL_NO_SHARED && \
|
|
95
|
-
(MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 65 \
|
|
96
|
-
|| MLD_CONFIG_PARAMETER_SET == 87)) */
|
|
@@ -1,126 +0,0 @@
|
|
|
1
|
-
/*
|
|
2
|
-
* Copyright (c) The mldsa-native project authors
|
|
3
|
-
* SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT
|
|
4
|
-
*/
|
|
5
|
-
|
|
6
|
-
/* References
|
|
7
|
-
* ==========
|
|
8
|
-
*
|
|
9
|
-
* - [REF_AVX2]
|
|
10
|
-
* CRYSTALS-Dilithium optimized AVX2 implementation
|
|
11
|
-
* Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé
|
|
12
|
-
* https://github.com/pq-crystals/dilithium/tree/master/avx2
|
|
13
|
-
*/
|
|
14
|
-
|
|
15
|
-
/*
|
|
16
|
-
* This file is derived from the public domain
|
|
17
|
-
* AVX2 Dilithium implementation @[REF_AVX2].
|
|
18
|
-
*/
|
|
19
|
-
|
|
20
|
-
#include "../../../common.h"
|
|
21
|
-
|
|
22
|
-
#if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \
|
|
23
|
-
!defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)
|
|
24
|
-
|
|
25
|
-
#include <immintrin.h>
|
|
26
|
-
#include "arith_native_x86_64.h"
|
|
27
|
-
#include "consts.h"
|
|
28
|
-
|
|
29
|
-
/*
|
|
30
|
-
* Reference: The pqcrystals implementation assumes a buffer that is 8 bytes
|
|
31
|
-
*. larger as the first loop overreads by 8 bytes that are then
|
|
32
|
-
* discarded. We instead do not pad the buffer and do not overread.
|
|
33
|
-
* The performance impact is negligible and it does not force the
|
|
34
|
-
* frontend to perform the unintuitive padding.
|
|
35
|
-
*/
|
|
36
|
-
|
|
37
|
-
unsigned int mld_rej_uniform_avx2(
|
|
38
|
-
int32_t *MLD_RESTRICT r, const uint8_t buf[MLD_AVX2_REJ_UNIFORM_BUFLEN])
|
|
39
|
-
{
|
|
40
|
-
unsigned int ctr, pos;
|
|
41
|
-
uint32_t good;
|
|
42
|
-
__m256i d, tmp;
|
|
43
|
-
const __m256i bound = _mm256_set1_epi32(MLDSA_Q);
|
|
44
|
-
const __m256i mask = _mm256_set1_epi32(0x7FFFFF);
|
|
45
|
-
const __m256i idx8 =
|
|
46
|
-
_mm256_set_epi8(-1, 15, 14, 13, -1, 12, 11, 10, -1, 9, 8, 7, -1, 6, 5, 4,
|
|
47
|
-
-1, 11, 10, 9, -1, 8, 7, 6, -1, 5, 4, 3, -1, 2, 1, 0);
|
|
48
|
-
|
|
49
|
-
ctr = pos = 0;
|
|
50
|
-
while (ctr <= MLDSA_N - 8 && pos <= MLD_AVX2_REJ_UNIFORM_BUFLEN - 32)
|
|
51
|
-
{
|
|
52
|
-
d = _mm256_loadu_si256((__m256i *)&buf[pos]);
|
|
53
|
-
|
|
54
|
-
/* Permute 64-bit lanes
|
|
55
|
-
* 0x94 = 10010100b rearranges 64-bit lanes as: [3,2,1,0] -> [2,1,1,0]
|
|
56
|
-
*
|
|
57
|
-
* ╔═══════════════════════════════════════════════════════════════════════╗
|
|
58
|
-
* ║ Original Layout ║
|
|
59
|
-
* ╚═══════════════════════════════════════════════════════════════════════╝
|
|
60
|
-
* ┌─────────────────┬─────────────────┬─────────────────┬─────────────────┐
|
|
61
|
-
* │ Lane 0 │ Lane 1 │ Lane 2 │ Lane 3 │
|
|
62
|
-
* │ bytes 0..7 │ bytes 8..15 │ bytes 16..23 │ bytes 24..31 │
|
|
63
|
-
* └─────────────────┴─────────────────┴─────────────────┴─────────────────┘
|
|
64
|
-
*
|
|
65
|
-
* ╔═══════════════════════════════════════════════════════════════════════╗
|
|
66
|
-
* ║ Layout after permute ║
|
|
67
|
-
* ║ Byte indices in high half shifted down by 8 positions ║
|
|
68
|
-
* ╚═══════════════════════════════════════════════════════════════════════╝
|
|
69
|
-
* ┌───────────────┬─────────────────┐ ┌─────────────────┬─────────────────┐
|
|
70
|
-
* │ Lane 0 │ Lane 1 │ │ Lane 2 │ Lane 3 │
|
|
71
|
-
* │ bytes 0..7 │ bytes 8..15 │ │ bytes 8..15 │ bytes 16..23 │
|
|
72
|
-
* └───────────────┴─────────────────┘ └─────────────────┴─────────────────┘
|
|
73
|
-
* Lower 128-bit lane (bytes 0-15) Upper 128-bit lane (bytes 16-31)
|
|
74
|
-
*/
|
|
75
|
-
d = _mm256_permute4x64_epi64(d, 0x94);
|
|
76
|
-
|
|
77
|
-
/* Shuffling 8-bit lanes
|
|
78
|
-
*
|
|
79
|
-
* ┌─ Indices 0-11 into low 128-bit half of permuted vector────────────────┐
|
|
80
|
-
* │ Shuffle: [-1, 11, 10, 9, -1, 8, 7, 6, -1, 5, 4, 3, -1, 2, 1, 0] │
|
|
81
|
-
* │ Result: [0, byte11, byte10, byte9, ..., 0, byte2, byte1, byte0] │
|
|
82
|
-
* └───────────────────────────────────────────────────────────────────────┘
|
|
83
|
-
*
|
|
84
|
-
* ┌─ Indices 4-15 into high 128-bit half of permuted vector ──────────────┐
|
|
85
|
-
* │ Shuffle: [-1, 15, 14, 13, -1, 12, 11, 10, -1, 9, 8, 7, -1, 6, 5, 4] │
|
|
86
|
-
* │ Result: [0, byte23, byte22, byte21, ..., 0, byte14, byte13, byte12 │
|
|
87
|
-
* └───────────────────────────────────────────────────────────────────────┘
|
|
88
|
-
*/
|
|
89
|
-
d = _mm256_shuffle_epi8(d, idx8);
|
|
90
|
-
d = _mm256_and_si256(d, mask);
|
|
91
|
-
pos += 24;
|
|
92
|
-
|
|
93
|
-
tmp = _mm256_sub_epi32(d, bound);
|
|
94
|
-
good = (uint32_t)_mm256_movemask_ps((__m256)tmp);
|
|
95
|
-
tmp = _mm256_cvtepu8_epi32(
|
|
96
|
-
_mm_loadl_epi64((__m128i *)&mld_rej_uniform_table[good]));
|
|
97
|
-
d = _mm256_permutevar8x32_epi32(d, tmp);
|
|
98
|
-
|
|
99
|
-
_mm256_storeu_si256((__m256i *)&r[ctr], d);
|
|
100
|
-
ctr += (unsigned)_mm_popcnt_u32(good);
|
|
101
|
-
}
|
|
102
|
-
|
|
103
|
-
while (ctr < MLDSA_N && pos <= MLD_AVX2_REJ_UNIFORM_BUFLEN - 3)
|
|
104
|
-
{
|
|
105
|
-
uint32_t t = buf[pos++];
|
|
106
|
-
t |= (uint32_t)buf[pos++] << 8;
|
|
107
|
-
t |= (uint32_t)buf[pos++] << 16;
|
|
108
|
-
t &= 0x7FFFFF;
|
|
109
|
-
|
|
110
|
-
if (t < MLDSA_Q)
|
|
111
|
-
{
|
|
112
|
-
/* Safe because t < MLDSA_Q. */
|
|
113
|
-
r[ctr++] = (int32_t)t;
|
|
114
|
-
}
|
|
115
|
-
}
|
|
116
|
-
|
|
117
|
-
return ctr;
|
|
118
|
-
}
|
|
119
|
-
|
|
120
|
-
#else /* MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_MULTILEVEL_NO_SHARED \
|
|
121
|
-
*/
|
|
122
|
-
|
|
123
|
-
MLD_EMPTY_CU(avx2_rej_uniform)
|
|
124
|
-
|
|
125
|
-
#endif /* !(MLD_ARITH_BACKEND_X86_64_DEFAULT && \
|
|
126
|
-
!MLD_CONFIG_MULTILEVEL_NO_SHARED) */
|
|
@@ -1,157 +0,0 @@
|
|
|
1
|
-
/*
|
|
2
|
-
* Copyright (c) The mldsa-native project authors
|
|
3
|
-
* SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT
|
|
4
|
-
*/
|
|
5
|
-
|
|
6
|
-
/* References
|
|
7
|
-
* ==========
|
|
8
|
-
*
|
|
9
|
-
* - [REF_AVX2]
|
|
10
|
-
* CRYSTALS-Dilithium optimized AVX2 implementation
|
|
11
|
-
* Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé
|
|
12
|
-
* https://github.com/pq-crystals/dilithium/tree/master/avx2
|
|
13
|
-
*/
|
|
14
|
-
|
|
15
|
-
/*
|
|
16
|
-
* This file is derived from the public domain
|
|
17
|
-
* AVX2 Dilithium implementation @[REF_AVX2].
|
|
18
|
-
*/
|
|
19
|
-
|
|
20
|
-
#include "../../../common.h"
|
|
21
|
-
|
|
22
|
-
#if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \
|
|
23
|
-
!defined(MLD_CONFIG_NO_KEYPAIR_API) && \
|
|
24
|
-
!defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \
|
|
25
|
-
(defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLDSA_ETA == 2)
|
|
26
|
-
|
|
27
|
-
#include <immintrin.h>
|
|
28
|
-
#include "arith_native_x86_64.h"
|
|
29
|
-
#include "consts.h"
|
|
30
|
-
|
|
31
|
-
#define MLD_AVX2_ETA2 2
|
|
32
|
-
|
|
33
|
-
/*
|
|
34
|
-
* Reference: In the pqcrystals implementation this function is called
|
|
35
|
-
* rej_eta_avx and supports multiple values for ETA via preprocessor
|
|
36
|
-
* conditionals. We move the conditionals to the frontend.
|
|
37
|
-
*/
|
|
38
|
-
unsigned int mld_rej_uniform_eta2_avx2(
|
|
39
|
-
int32_t *MLD_RESTRICT r,
|
|
40
|
-
const uint8_t buf[MLD_AVX2_REJ_UNIFORM_ETA2_BUFLEN])
|
|
41
|
-
{
|
|
42
|
-
unsigned int ctr, pos;
|
|
43
|
-
uint32_t good;
|
|
44
|
-
__m256i f0, f1, f2;
|
|
45
|
-
__m128i g0, g1;
|
|
46
|
-
const __m256i mask = _mm256_set1_epi8(15);
|
|
47
|
-
const __m256i eta = _mm256_set1_epi8(MLD_AVX2_ETA2);
|
|
48
|
-
const __m256i bound = mask;
|
|
49
|
-
/* check-magic: -6560 == 32*round(-2**10 / 5) */
|
|
50
|
-
const __m256i v = _mm256_set1_epi32(-6560);
|
|
51
|
-
const __m256i p = _mm256_set1_epi32(5);
|
|
52
|
-
|
|
53
|
-
ctr = pos = 0;
|
|
54
|
-
while (ctr <= MLDSA_N - 8 && pos <= MLD_AVX2_REJ_UNIFORM_ETA2_BUFLEN - 16)
|
|
55
|
-
{
|
|
56
|
-
f0 = _mm256_cvtepu8_epi16(_mm_loadu_si128((__m128i *)&buf[pos]));
|
|
57
|
-
f1 = _mm256_slli_epi16(f0, 4);
|
|
58
|
-
f0 = _mm256_or_si256(f0, f1);
|
|
59
|
-
f0 = _mm256_and_si256(f0, mask);
|
|
60
|
-
|
|
61
|
-
f1 = _mm256_sub_epi8(f0, bound);
|
|
62
|
-
f0 = _mm256_sub_epi8(eta, f0);
|
|
63
|
-
good = (uint32_t)_mm256_movemask_epi8(f1);
|
|
64
|
-
|
|
65
|
-
g0 = _mm256_castsi256_si128(f0);
|
|
66
|
-
g1 = _mm_loadl_epi64((__m128i *)&mld_rej_uniform_table[good & 0xFF]);
|
|
67
|
-
g1 = _mm_shuffle_epi8(g0, g1);
|
|
68
|
-
f1 = _mm256_cvtepi8_epi32(g1);
|
|
69
|
-
f2 = _mm256_mulhrs_epi16(f1, v);
|
|
70
|
-
f2 = _mm256_mullo_epi16(f2, p);
|
|
71
|
-
f1 = _mm256_add_epi32(f1, f2);
|
|
72
|
-
_mm256_storeu_si256((__m256i *)&r[ctr], f1);
|
|
73
|
-
ctr += (unsigned)_mm_popcnt_u32(good & 0xFF);
|
|
74
|
-
good >>= 8;
|
|
75
|
-
pos += 4;
|
|
76
|
-
|
|
77
|
-
if (ctr > MLDSA_N - 8)
|
|
78
|
-
{
|
|
79
|
-
break;
|
|
80
|
-
}
|
|
81
|
-
g0 = _mm_bsrli_si128(g0, 8);
|
|
82
|
-
g1 = _mm_loadl_epi64((__m128i *)&mld_rej_uniform_table[good & 0xFF]);
|
|
83
|
-
g1 = _mm_shuffle_epi8(g0, g1);
|
|
84
|
-
f1 = _mm256_cvtepi8_epi32(g1);
|
|
85
|
-
f2 = _mm256_mulhrs_epi16(f1, v);
|
|
86
|
-
f2 = _mm256_mullo_epi16(f2, p);
|
|
87
|
-
f1 = _mm256_add_epi32(f1, f2);
|
|
88
|
-
_mm256_storeu_si256((__m256i *)&r[ctr], f1);
|
|
89
|
-
ctr += (unsigned)_mm_popcnt_u32(good & 0xFF);
|
|
90
|
-
good >>= 8;
|
|
91
|
-
pos += 4;
|
|
92
|
-
|
|
93
|
-
if (ctr > MLDSA_N - 8)
|
|
94
|
-
{
|
|
95
|
-
break;
|
|
96
|
-
}
|
|
97
|
-
g0 = _mm256_extracti128_si256(f0, 1);
|
|
98
|
-
g1 = _mm_loadl_epi64((__m128i *)&mld_rej_uniform_table[good & 0xFF]);
|
|
99
|
-
g1 = _mm_shuffle_epi8(g0, g1);
|
|
100
|
-
f1 = _mm256_cvtepi8_epi32(g1);
|
|
101
|
-
f2 = _mm256_mulhrs_epi16(f1, v);
|
|
102
|
-
f2 = _mm256_mullo_epi16(f2, p);
|
|
103
|
-
f1 = _mm256_add_epi32(f1, f2);
|
|
104
|
-
_mm256_storeu_si256((__m256i *)&r[ctr], f1);
|
|
105
|
-
ctr += (unsigned)_mm_popcnt_u32(good & 0xFF);
|
|
106
|
-
good >>= 8;
|
|
107
|
-
pos += 4;
|
|
108
|
-
|
|
109
|
-
if (ctr > MLDSA_N - 8)
|
|
110
|
-
{
|
|
111
|
-
break;
|
|
112
|
-
}
|
|
113
|
-
g0 = _mm_bsrli_si128(g0, 8);
|
|
114
|
-
g1 = _mm_loadl_epi64((__m128i *)&mld_rej_uniform_table[good]);
|
|
115
|
-
g1 = _mm_shuffle_epi8(g0, g1);
|
|
116
|
-
f1 = _mm256_cvtepi8_epi32(g1);
|
|
117
|
-
f2 = _mm256_mulhrs_epi16(f1, v);
|
|
118
|
-
f2 = _mm256_mullo_epi16(f2, p);
|
|
119
|
-
f1 = _mm256_add_epi32(f1, f2);
|
|
120
|
-
_mm256_storeu_si256((__m256i *)&r[ctr], f1);
|
|
121
|
-
ctr += (unsigned)_mm_popcnt_u32(good);
|
|
122
|
-
pos += 4;
|
|
123
|
-
}
|
|
124
|
-
|
|
125
|
-
while (ctr < MLDSA_N && pos < MLD_AVX2_REJ_UNIFORM_ETA2_BUFLEN)
|
|
126
|
-
{
|
|
127
|
-
uint32_t t0 = buf[pos] & 0x0F;
|
|
128
|
-
uint32_t t1 = buf[pos++] >> 4;
|
|
129
|
-
|
|
130
|
-
if (t0 < 15)
|
|
131
|
-
{
|
|
132
|
-
t0 = t0 - (205 * t0 >> 10) * 5;
|
|
133
|
-
r[ctr++] = (int32_t)(2 - t0);
|
|
134
|
-
}
|
|
135
|
-
if (t1 < 15 && ctr < MLDSA_N)
|
|
136
|
-
{
|
|
137
|
-
t1 = t1 - (205 * t1 >> 10) * 5;
|
|
138
|
-
r[ctr++] = (int32_t)(2 - t1);
|
|
139
|
-
}
|
|
140
|
-
}
|
|
141
|
-
|
|
142
|
-
return ctr;
|
|
143
|
-
}
|
|
144
|
-
|
|
145
|
-
#else /* MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_NO_KEYPAIR_API && \
|
|
146
|
-
!MLD_CONFIG_MULTILEVEL_NO_SHARED && \
|
|
147
|
-
(MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLDSA_ETA == 2) */
|
|
148
|
-
|
|
149
|
-
MLD_EMPTY_CU(avx2_rej_uniform_eta2)
|
|
150
|
-
|
|
151
|
-
#endif /* !(MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_NO_KEYPAIR_API && \
|
|
152
|
-
!MLD_CONFIG_MULTILEVEL_NO_SHARED && \
|
|
153
|
-
(MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLDSA_ETA == 2)) */
|
|
154
|
-
|
|
155
|
-
/* To facilitate single-compilation-unit (SCU) builds, undefine all macros.
|
|
156
|
-
* Don't modify by hand -- this is auto-generated by scripts/autogen. */
|
|
157
|
-
#undef MLD_AVX2_ETA2
|
|
@@ -1,141 +0,0 @@
|
|
|
1
|
-
/*
|
|
2
|
-
* Copyright (c) The mldsa-native project authors
|
|
3
|
-
* SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT
|
|
4
|
-
*/
|
|
5
|
-
|
|
6
|
-
/* References
|
|
7
|
-
* ==========
|
|
8
|
-
*
|
|
9
|
-
* - [REF_AVX2]
|
|
10
|
-
* CRYSTALS-Dilithium optimized AVX2 implementation
|
|
11
|
-
* Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé
|
|
12
|
-
* https://github.com/pq-crystals/dilithium/tree/master/avx2
|
|
13
|
-
*/
|
|
14
|
-
|
|
15
|
-
/*
|
|
16
|
-
* This file is derived from the public domain
|
|
17
|
-
* AVX2 Dilithium implementation @[REF_AVX2].
|
|
18
|
-
*/
|
|
19
|
-
|
|
20
|
-
#include "../../../common.h"
|
|
21
|
-
|
|
22
|
-
#if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \
|
|
23
|
-
!defined(MLD_CONFIG_NO_KEYPAIR_API) && \
|
|
24
|
-
!defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \
|
|
25
|
-
(defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLDSA_ETA == 4)
|
|
26
|
-
|
|
27
|
-
#include <immintrin.h>
|
|
28
|
-
#include "arith_native_x86_64.h"
|
|
29
|
-
#include "consts.h"
|
|
30
|
-
|
|
31
|
-
#define MLD_AVX2_ETA4 4
|
|
32
|
-
|
|
33
|
-
/*
|
|
34
|
-
* Reference: In the pqcrystals implementation this function is called
|
|
35
|
-
* rej_eta_avx and supports multiple values for ETA via preprocessor
|
|
36
|
-
* conditionals. We move the conditionals to the frontend.
|
|
37
|
-
*/
|
|
38
|
-
|
|
39
|
-
unsigned int mld_rej_uniform_eta4_avx2(
|
|
40
|
-
int32_t *MLD_RESTRICT r,
|
|
41
|
-
const uint8_t buf[MLD_AVX2_REJ_UNIFORM_ETA4_BUFLEN])
|
|
42
|
-
{
|
|
43
|
-
unsigned int ctr, pos;
|
|
44
|
-
uint32_t good;
|
|
45
|
-
__m256i f0, f1;
|
|
46
|
-
__m128i g0, g1;
|
|
47
|
-
const __m256i mask = _mm256_set1_epi8(15);
|
|
48
|
-
const __m256i eta = _mm256_set1_epi8(MLD_AVX2_ETA4);
|
|
49
|
-
const __m256i bound = _mm256_set1_epi8(9);
|
|
50
|
-
|
|
51
|
-
ctr = pos = 0;
|
|
52
|
-
while (ctr <= MLDSA_N - 8 && pos <= MLD_AVX2_REJ_UNIFORM_ETA4_BUFLEN - 16)
|
|
53
|
-
{
|
|
54
|
-
f0 = _mm256_cvtepu8_epi16(_mm_loadu_si128((__m128i *)&buf[pos]));
|
|
55
|
-
f1 = _mm256_slli_epi16(f0, 4);
|
|
56
|
-
f0 = _mm256_or_si256(f0, f1);
|
|
57
|
-
f0 = _mm256_and_si256(f0, mask);
|
|
58
|
-
|
|
59
|
-
f1 = _mm256_sub_epi8(f0, bound);
|
|
60
|
-
f0 = _mm256_sub_epi8(eta, f0);
|
|
61
|
-
good = (uint32_t)_mm256_movemask_epi8(f1);
|
|
62
|
-
|
|
63
|
-
g0 = _mm256_castsi256_si128(f0);
|
|
64
|
-
g1 = _mm_loadl_epi64((__m128i *)&mld_rej_uniform_table[good & 0xFF]);
|
|
65
|
-
g1 = _mm_shuffle_epi8(g0, g1);
|
|
66
|
-
f1 = _mm256_cvtepi8_epi32(g1);
|
|
67
|
-
_mm256_storeu_si256((__m256i *)&r[ctr], f1);
|
|
68
|
-
ctr += (unsigned)_mm_popcnt_u32(good & 0xFF);
|
|
69
|
-
good >>= 8;
|
|
70
|
-
pos += 4;
|
|
71
|
-
|
|
72
|
-
if (ctr > MLDSA_N - 8)
|
|
73
|
-
{
|
|
74
|
-
break;
|
|
75
|
-
}
|
|
76
|
-
g0 = _mm_bsrli_si128(g0, 8);
|
|
77
|
-
g1 = _mm_loadl_epi64((__m128i *)&mld_rej_uniform_table[good & 0xFF]);
|
|
78
|
-
g1 = _mm_shuffle_epi8(g0, g1);
|
|
79
|
-
f1 = _mm256_cvtepi8_epi32(g1);
|
|
80
|
-
_mm256_storeu_si256((__m256i *)&r[ctr], f1);
|
|
81
|
-
ctr += (unsigned)_mm_popcnt_u32(good & 0xFF);
|
|
82
|
-
good >>= 8;
|
|
83
|
-
pos += 4;
|
|
84
|
-
|
|
85
|
-
if (ctr > MLDSA_N - 8)
|
|
86
|
-
{
|
|
87
|
-
break;
|
|
88
|
-
}
|
|
89
|
-
g0 = _mm256_extracti128_si256(f0, 1);
|
|
90
|
-
g1 = _mm_loadl_epi64((__m128i *)&mld_rej_uniform_table[good & 0xFF]);
|
|
91
|
-
g1 = _mm_shuffle_epi8(g0, g1);
|
|
92
|
-
f1 = _mm256_cvtepi8_epi32(g1);
|
|
93
|
-
_mm256_storeu_si256((__m256i *)&r[ctr], f1);
|
|
94
|
-
ctr += (unsigned)_mm_popcnt_u32(good & 0xFF);
|
|
95
|
-
good >>= 8;
|
|
96
|
-
pos += 4;
|
|
97
|
-
|
|
98
|
-
if (ctr > MLDSA_N - 8)
|
|
99
|
-
{
|
|
100
|
-
break;
|
|
101
|
-
}
|
|
102
|
-
g0 = _mm_bsrli_si128(g0, 8);
|
|
103
|
-
g1 = _mm_loadl_epi64((__m128i *)&mld_rej_uniform_table[good]);
|
|
104
|
-
g1 = _mm_shuffle_epi8(g0, g1);
|
|
105
|
-
f1 = _mm256_cvtepi8_epi32(g1);
|
|
106
|
-
_mm256_storeu_si256((__m256i *)&r[ctr], f1);
|
|
107
|
-
ctr += (unsigned)_mm_popcnt_u32(good);
|
|
108
|
-
pos += 4;
|
|
109
|
-
}
|
|
110
|
-
|
|
111
|
-
while (ctr < MLDSA_N && pos < MLD_AVX2_REJ_UNIFORM_ETA4_BUFLEN)
|
|
112
|
-
{
|
|
113
|
-
uint32_t t0 = buf[pos] & 0x0F;
|
|
114
|
-
uint32_t t1 = buf[pos++] >> 4;
|
|
115
|
-
|
|
116
|
-
if (t0 < 9)
|
|
117
|
-
{
|
|
118
|
-
r[ctr++] = (int32_t)(4 - t0);
|
|
119
|
-
}
|
|
120
|
-
if (t1 < 9 && ctr < MLDSA_N)
|
|
121
|
-
{
|
|
122
|
-
r[ctr++] = (int32_t)(4 - t1);
|
|
123
|
-
}
|
|
124
|
-
}
|
|
125
|
-
|
|
126
|
-
return ctr;
|
|
127
|
-
}
|
|
128
|
-
|
|
129
|
-
#else /* MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_NO_KEYPAIR_API && \
|
|
130
|
-
!MLD_CONFIG_MULTILEVEL_NO_SHARED && \
|
|
131
|
-
(MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLDSA_ETA == 4) */
|
|
132
|
-
|
|
133
|
-
MLD_EMPTY_CU(avx2_rej_uniform_eta4)
|
|
134
|
-
|
|
135
|
-
#endif /* !(MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_NO_KEYPAIR_API && \
|
|
136
|
-
!MLD_CONFIG_MULTILEVEL_NO_SHARED && \
|
|
137
|
-
(MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLDSA_ETA == 4)) */
|
|
138
|
-
|
|
139
|
-
/* To facilitate single-compilation-unit (SCU) builds, undefine all macros.
|
|
140
|
-
* Don't modify by hand -- this is auto-generated by scripts/autogen. */
|
|
141
|
-
#undef MLD_AVX2_ETA4
|