pq_crypto 0.6.5 → 0.6.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +233 -0
- data/README.md +16 -3
- data/SECURITY.md +46 -0
- data/ext/pqcrypto/extconf.rb +8 -1
- data/ext/pqcrypto/pq_externalmu.c +35 -0
- data/ext/pqcrypto/pqcrypto_native_api.h +91 -75
- data/ext/pqcrypto/pqcrypto_ruby_secure.c +97 -9
- data/ext/pqcrypto/pqcrypto_secure.c +95 -33
- data/ext/pqcrypto/pqcrypto_secure.h +66 -48
- data/ext/pqcrypto/pqcrypto_version.h +1 -1
- data/ext/pqcrypto/vendor/.vendored +7 -7
- data/ext/pqcrypto/vendor/mldsa-native/BUILDING.md +5 -2
- data/ext/pqcrypto/vendor/mldsa-native/LICENSE +21 -2
- data/ext/pqcrypto/vendor/mldsa-native/README.md +20 -7
- data/ext/pqcrypto/vendor/mldsa-native/RELEASE.md +160 -0
- data/ext/pqcrypto/vendor/mldsa-native/SECURITY.md +1 -1
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/README.md +2 -2
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/mldsa_native.c +85 -59
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/mldsa_native.h +292 -348
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/mldsa_native_asm.S +122 -76
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/mldsa_native_config.h +184 -86
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/cbmc.h +49 -4
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/common.h +49 -81
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/context.h +152 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/ct.h +25 -12
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/debug.c +2 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/debug.h +2 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/fips202x4.c +2 -2
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/keccakf1600.c +9 -11
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/auto.h +19 -11
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x1_scalar_aarch64_asm.S +6 -4
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x1_v84a_aarch64_asm.S +7 -4
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x2_v84a_aarch64_asm.S +7 -4
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm.S +12 -9
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm.S +12 -9
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/x1_scalar.h +1 -1
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/x1_v84a.h +3 -2
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/x2_v84a.h +3 -2
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/x4_v8a_scalar.h +6 -1
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/x4_v8a_v84a_scalar.h +3 -2
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/api.h +11 -11
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/armv81m/mve.h +9 -22
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/armv81m/src/keccak_f1600_x4_mve.S +8 -5
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/armv81m/src/keccak_f1600_x4_mve.c +1 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/armv81m/src/{state_extract_bytes_x4_mve.S → keccak_f1600_x4_state_extract_bytes_mve.S} +14 -14
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/armv81m/src/{state_xor_bytes_x4_mve.S → keccak_f1600_x4_state_xor_bytes_mve.S} +12 -12
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/auto.h +5 -4
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/x86_64/keccak_f1600_x4_avx2.h +2 -2
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/x86_64/src/fips202_native_x86_64.h +1 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/x86_64/src/keccak_f1600_x4_avx2_asm.S +36 -2
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/meta.h +62 -4
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/arith_native_aarch64.h +87 -54
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{intt_aarch64_asm.S → mldsa_intt_aarch64_asm.S} +39 -6
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{ntt_aarch64_asm.S → mldsa_ntt_aarch64_asm.S} +39 -6
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{pointwise_montgomery_aarch64_asm.S → mldsa_pointwise_montgomery_aarch64_asm.S} +25 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{poly_caddq_aarch64_asm.S → mldsa_poly_caddq_aarch64_asm.S} +19 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{poly_chknorm_aarch64_asm.S → mldsa_poly_chknorm_aarch64_asm.S} +24 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{poly_decompose_32_aarch64_asm.S → mldsa_poly_decompose_32_aarch64_asm.S} +25 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{poly_decompose_88_aarch64_asm.S → mldsa_poly_decompose_88_aarch64_asm.S} +25 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{poly_use_hint_32_aarch64_asm.S → mldsa_poly_use_hint_32_aarch64_asm.S} +25 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{poly_use_hint_88_aarch64_asm.S → mldsa_poly_use_hint_88_aarch64_asm.S} +25 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{mld_polyvecl_pointwise_acc_montgomery_l4_aarch64_asm.S → mldsa_polyvecl_pointwise_acc_montgomery_l4_aarch64_asm.S} +31 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{mld_polyvecl_pointwise_acc_montgomery_l5_aarch64_asm.S → mldsa_polyvecl_pointwise_acc_montgomery_l5_aarch64_asm.S} +31 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{mld_polyvecl_pointwise_acc_montgomery_l7_aarch64_asm.S → mldsa_polyvecl_pointwise_acc_montgomery_l7_aarch64_asm.S} +31 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{polyz_unpack_17_aarch64_asm.S → mldsa_polyz_unpack_17_aarch64_asm.S} +31 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{polyz_unpack_19_aarch64_asm.S → mldsa_polyz_unpack_19_aarch64_asm.S} +31 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{rej_uniform_aarch64_asm.S → mldsa_rej_uniform_aarch64_asm.S} +48 -15
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{rej_uniform_eta2_aarch64_asm.S → mldsa_rej_uniform_eta2_aarch64_asm.S} +42 -9
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{rej_uniform_eta4_aarch64_asm.S → mldsa_rej_uniform_eta4_aarch64_asm.S} +42 -9
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/api.h +11 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/meta.h +3 -2
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/meta.h +28 -28
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/arith_native_x86_64.h +171 -49
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{intt_avx2_asm.S → mldsa_intt_avx2_asm.S} +23 -1
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{ntt_avx2_asm.S → mldsa_ntt_avx2_asm.S} +23 -1
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{nttunpack_avx2_asm.S → mldsa_nttunpack_avx2_asm.S} +17 -1
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{pointwise_acc_l4_avx2_asm.S → mldsa_pointwise_acc_l4_avx2_asm.S} +37 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{pointwise_acc_l5_avx2_asm.S → mldsa_pointwise_acc_l5_avx2_asm.S} +37 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{pointwise_acc_l7_avx2_asm.S → mldsa_pointwise_acc_l7_avx2_asm.S} +37 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{pointwise_avx2_asm.S → mldsa_pointwise_avx2_asm.S} +31 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{poly_caddq_avx2_asm.S → mldsa_poly_caddq_avx2_asm.S} +18 -9
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_poly_chknorm_avx2_asm.S +176 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_poly_decompose_32_avx2_asm.S +490 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_poly_decompose_88_avx2_asm.S +489 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_poly_use_hint_32_avx2_asm.S +123 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_poly_use_hint_88_avx2_asm.S +125 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_polyz_unpack_17_avx2_asm.S +355 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_polyz_unpack_19_avx2_asm.S +355 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_rej_uniform_avx2_asm.S +132 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_rej_uniform_eta2_avx2_asm.S +205 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_rej_uniform_eta4_avx2_asm.S +176 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/packing.c +27 -36
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/packing.h +42 -8
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/params.h +93 -17
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/poly.c +74 -15
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/poly.h +97 -11
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/poly_kl.c +7 -38
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/poly_kl.h +49 -7
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/polyvec.c +16 -17
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/polyvec.h +26 -9
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/polyvec_lazy.c +3 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/polyvec_lazy.h +18 -19
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/reduce.h +15 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/rounding.h +28 -6
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/sign.c +311 -246
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/sign.h +245 -240
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/sys.h +64 -5
- data/ext/pqcrypto/vendor/mlkem-native/BUILDING.md +5 -2
- data/ext/pqcrypto/vendor/mlkem-native/LICENSE +21 -3
- data/ext/pqcrypto/vendor/mlkem-native/README.md +2 -2
- data/ext/pqcrypto/vendor/mlkem-native/RELEASE.md +113 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/README.md +2 -2
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/mlkem_native.c +17 -27
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/mlkem_native.h +68 -151
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/mlkem_native_asm.S +17 -27
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/mlkem_native_config.h +46 -44
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/cbmc.h +25 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/common.h +37 -6
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/context.h +9 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/fips202.h +2 -2
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/keccakf1600.c +8 -8
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x1_scalar.h +1 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x1_v84a.h +3 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x2_v84a.h +3 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x4_v8a_scalar.h +2 -2
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x4_v8a_v84a_scalar.h +3 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/api.h +11 -11
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/armv81m/mve.h +3 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/x86_64/keccak_f1600_x4_avx2.h +2 -2
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/indcpa.h +14 -11
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/kem.c +28 -11
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/kem.h +39 -14
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/meta.h +10 -10
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/meta.h +20 -20
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/sys.h +5 -5
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/verify.h +11 -10
- data/lib/pq_crypto/internal.rb +10 -0
- data/lib/pq_crypto/kem.rb +47 -5
- data/lib/pq_crypto/key.rb +14 -8
- data/lib/pq_crypto/pkcs8.rb +13 -5
- data/lib/pq_crypto/signature.rb +34 -9
- data/lib/pq_crypto/spki.rb +4 -2
- data/lib/pq_crypto/version.rb +1 -1
- data/lib/pq_crypto.rb +9 -0
- data/script/vendor_libs.rb +6 -6
- metadata +40 -38
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/poly_chknorm_avx2.c +0 -52
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/poly_decompose_32_avx2.c +0 -157
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/poly_decompose_88_avx2.c +0 -157
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/poly_use_hint_32_avx2.c +0 -103
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/poly_use_hint_88_avx2.c +0 -105
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/polyz_unpack_17_avx2.c +0 -94
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/polyz_unpack_19_avx2.c +0 -96
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/rej_uniform_avx2.c +0 -126
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/rej_uniform_eta2_avx2.c +0 -157
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/rej_uniform_eta4_avx2.c +0 -141
|
@@ -17,6 +17,40 @@
|
|
|
17
17
|
* AVX2 Dilithium implementation @[REF_AVX2].
|
|
18
18
|
*/
|
|
19
19
|
|
|
20
|
+
/*yaml
|
|
21
|
+
Name: pointwise_acc_l4_avx2_asm
|
|
22
|
+
Description: x86_64 AVX2 pointwise multiply-accumulate of length-4 polynomial vectors
|
|
23
|
+
Signature: void mld_pointwise_acc_l4_avx2_asm(int32_t *c, const int32_t a[4][256], const int32_t b[4][256], const int32_t *qdata)
|
|
24
|
+
ABI:
|
|
25
|
+
Architecture: x86_64
|
|
26
|
+
CallingConvention: SysV
|
|
27
|
+
Features: [AVX2]
|
|
28
|
+
rdi:
|
|
29
|
+
type: buffer
|
|
30
|
+
size_bytes: 1024
|
|
31
|
+
permissions: write-only
|
|
32
|
+
c_parameter: int32_t *c
|
|
33
|
+
description: Output polynomial (256 x int32_t)
|
|
34
|
+
rsi:
|
|
35
|
+
type: buffer
|
|
36
|
+
size_bytes: 4096
|
|
37
|
+
permissions: read-only
|
|
38
|
+
c_parameter: const int32_t a[4][256]
|
|
39
|
+
description: Input polynomial vector a (4 x 256 x int32_t)
|
|
40
|
+
rdx:
|
|
41
|
+
type: buffer
|
|
42
|
+
size_bytes: 4096
|
|
43
|
+
permissions: read-only
|
|
44
|
+
c_parameter: const int32_t b[4][256]
|
|
45
|
+
description: Input polynomial vector b (4 x 256 x int32_t)
|
|
46
|
+
rcx:
|
|
47
|
+
type: buffer
|
|
48
|
+
size_bytes: 2496
|
|
49
|
+
permissions: read-only
|
|
50
|
+
c_parameter: const int32_t *qdata
|
|
51
|
+
description: Precomputed constants (624 x int32_t)
|
|
52
|
+
*/
|
|
53
|
+
|
|
20
54
|
#include "../../../common.h"
|
|
21
55
|
#if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \
|
|
22
56
|
!defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \
|
|
@@ -24,7 +58,7 @@
|
|
|
24
58
|
|
|
25
59
|
/*
|
|
26
60
|
* WARNING: This file is auto-derived from the mldsa-native source file
|
|
27
|
-
* dev/x86_64/src/
|
|
61
|
+
* dev/x86_64/src/mldsa_pointwise_acc_l4_avx2_asm.S using scripts/simpasm. Do not modify it directly.
|
|
28
62
|
*/
|
|
29
63
|
|
|
30
64
|
.text
|
|
@@ -37,7 +71,7 @@ MLD_ASM_FN_SYMBOL(pointwise_acc_l4_avx2_asm)
|
|
|
37
71
|
vmovdqa (%rcx), %ymm1
|
|
38
72
|
xorl %eax, %eax
|
|
39
73
|
|
|
40
|
-
|
|
74
|
+
Lmld_pointwise_acc_l4_avx2_looptop2:
|
|
41
75
|
vmovdqa (%rsi), %ymm6
|
|
42
76
|
vmovdqa 0x20(%rsi), %ymm8
|
|
43
77
|
vmovdqa (%rdx), %ymm10
|
|
@@ -125,7 +159,7 @@ Lpointwise_acc_l4_avx2_looptop2:
|
|
|
125
159
|
addq $0x40, %rdi
|
|
126
160
|
addl $0x1, %eax
|
|
127
161
|
cmpl $0x10, %eax
|
|
128
|
-
jb
|
|
162
|
+
jb Lmld_pointwise_acc_l4_avx2_looptop2
|
|
129
163
|
retq
|
|
130
164
|
.cfi_endproc
|
|
131
165
|
|
|
@@ -17,6 +17,40 @@
|
|
|
17
17
|
* AVX2 Dilithium implementation @[REF_AVX2].
|
|
18
18
|
*/
|
|
19
19
|
|
|
20
|
+
/*yaml
|
|
21
|
+
Name: pointwise_acc_l5_avx2_asm
|
|
22
|
+
Description: x86_64 AVX2 pointwise multiply-accumulate of length-5 polynomial vectors
|
|
23
|
+
Signature: void mld_pointwise_acc_l5_avx2_asm(int32_t *c, const int32_t a[5][256], const int32_t b[5][256], const int32_t *qdata)
|
|
24
|
+
ABI:
|
|
25
|
+
Architecture: x86_64
|
|
26
|
+
CallingConvention: SysV
|
|
27
|
+
Features: [AVX2]
|
|
28
|
+
rdi:
|
|
29
|
+
type: buffer
|
|
30
|
+
size_bytes: 1024
|
|
31
|
+
permissions: write-only
|
|
32
|
+
c_parameter: int32_t *c
|
|
33
|
+
description: Output polynomial (256 x int32_t)
|
|
34
|
+
rsi:
|
|
35
|
+
type: buffer
|
|
36
|
+
size_bytes: 5120
|
|
37
|
+
permissions: read-only
|
|
38
|
+
c_parameter: const int32_t a[5][256]
|
|
39
|
+
description: Input polynomial vector a (5 x 256 x int32_t)
|
|
40
|
+
rdx:
|
|
41
|
+
type: buffer
|
|
42
|
+
size_bytes: 5120
|
|
43
|
+
permissions: read-only
|
|
44
|
+
c_parameter: const int32_t b[5][256]
|
|
45
|
+
description: Input polynomial vector b (5 x 256 x int32_t)
|
|
46
|
+
rcx:
|
|
47
|
+
type: buffer
|
|
48
|
+
size_bytes: 2496
|
|
49
|
+
permissions: read-only
|
|
50
|
+
c_parameter: const int32_t *qdata
|
|
51
|
+
description: Precomputed constants (624 x int32_t)
|
|
52
|
+
*/
|
|
53
|
+
|
|
20
54
|
#include "../../../common.h"
|
|
21
55
|
#if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \
|
|
22
56
|
!defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \
|
|
@@ -24,7 +58,7 @@
|
|
|
24
58
|
|
|
25
59
|
/*
|
|
26
60
|
* WARNING: This file is auto-derived from the mldsa-native source file
|
|
27
|
-
* dev/x86_64/src/
|
|
61
|
+
* dev/x86_64/src/mldsa_pointwise_acc_l5_avx2_asm.S using scripts/simpasm. Do not modify it directly.
|
|
28
62
|
*/
|
|
29
63
|
|
|
30
64
|
.text
|
|
@@ -37,7 +71,7 @@ MLD_ASM_FN_SYMBOL(pointwise_acc_l5_avx2_asm)
|
|
|
37
71
|
vmovdqa (%rcx), %ymm1
|
|
38
72
|
xorl %eax, %eax
|
|
39
73
|
|
|
40
|
-
|
|
74
|
+
Lmld_pointwise_acc_l5_avx2_looptop2:
|
|
41
75
|
vmovdqa (%rsi), %ymm6
|
|
42
76
|
vmovdqa 0x20(%rsi), %ymm8
|
|
43
77
|
vmovdqa (%rdx), %ymm10
|
|
@@ -141,7 +175,7 @@ Lpointwise_acc_l5_avx2_looptop2:
|
|
|
141
175
|
addq $0x40, %rdi
|
|
142
176
|
addl $0x1, %eax
|
|
143
177
|
cmpl $0x10, %eax
|
|
144
|
-
jb
|
|
178
|
+
jb Lmld_pointwise_acc_l5_avx2_looptop2
|
|
145
179
|
retq
|
|
146
180
|
.cfi_endproc
|
|
147
181
|
|
|
@@ -17,6 +17,40 @@
|
|
|
17
17
|
* AVX2 Dilithium implementation @[REF_AVX2].
|
|
18
18
|
*/
|
|
19
19
|
|
|
20
|
+
/*yaml
|
|
21
|
+
Name: pointwise_acc_l7_avx2_asm
|
|
22
|
+
Description: x86_64 AVX2 pointwise multiply-accumulate of length-7 polynomial vectors
|
|
23
|
+
Signature: void mld_pointwise_acc_l7_avx2_asm(int32_t *c, const int32_t a[7][256], const int32_t b[7][256], const int32_t *qdata)
|
|
24
|
+
ABI:
|
|
25
|
+
Architecture: x86_64
|
|
26
|
+
CallingConvention: SysV
|
|
27
|
+
Features: [AVX2]
|
|
28
|
+
rdi:
|
|
29
|
+
type: buffer
|
|
30
|
+
size_bytes: 1024
|
|
31
|
+
permissions: write-only
|
|
32
|
+
c_parameter: int32_t *c
|
|
33
|
+
description: Output polynomial (256 x int32_t)
|
|
34
|
+
rsi:
|
|
35
|
+
type: buffer
|
|
36
|
+
size_bytes: 7168
|
|
37
|
+
permissions: read-only
|
|
38
|
+
c_parameter: const int32_t a[7][256]
|
|
39
|
+
description: Input polynomial vector a (7 x 256 x int32_t)
|
|
40
|
+
rdx:
|
|
41
|
+
type: buffer
|
|
42
|
+
size_bytes: 7168
|
|
43
|
+
permissions: read-only
|
|
44
|
+
c_parameter: const int32_t b[7][256]
|
|
45
|
+
description: Input polynomial vector b (7 x 256 x int32_t)
|
|
46
|
+
rcx:
|
|
47
|
+
type: buffer
|
|
48
|
+
size_bytes: 2496
|
|
49
|
+
permissions: read-only
|
|
50
|
+
c_parameter: const int32_t *qdata
|
|
51
|
+
description: Precomputed constants (624 x int32_t)
|
|
52
|
+
*/
|
|
53
|
+
|
|
20
54
|
#include "../../../common.h"
|
|
21
55
|
#if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \
|
|
22
56
|
!defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \
|
|
@@ -24,7 +58,7 @@
|
|
|
24
58
|
|
|
25
59
|
/*
|
|
26
60
|
* WARNING: This file is auto-derived from the mldsa-native source file
|
|
27
|
-
* dev/x86_64/src/
|
|
61
|
+
* dev/x86_64/src/mldsa_pointwise_acc_l7_avx2_asm.S using scripts/simpasm. Do not modify it directly.
|
|
28
62
|
*/
|
|
29
63
|
|
|
30
64
|
.text
|
|
@@ -37,7 +71,7 @@ MLD_ASM_FN_SYMBOL(pointwise_acc_l7_avx2_asm)
|
|
|
37
71
|
vmovdqa (%rcx), %ymm1
|
|
38
72
|
xorl %eax, %eax
|
|
39
73
|
|
|
40
|
-
|
|
74
|
+
Lmld_pointwise_acc_l7_avx2_looptop2:
|
|
41
75
|
vmovdqa (%rsi), %ymm6
|
|
42
76
|
vmovdqa 0x20(%rsi), %ymm8
|
|
43
77
|
vmovdqa (%rdx), %ymm10
|
|
@@ -173,7 +207,7 @@ Lpointwise_acc_l7_avx2_looptop2:
|
|
|
173
207
|
addq $0x40, %rdi
|
|
174
208
|
addl $0x1, %eax
|
|
175
209
|
cmpl $0x10, %eax
|
|
176
|
-
jb
|
|
210
|
+
jb Lmld_pointwise_acc_l7_avx2_looptop2
|
|
177
211
|
retq
|
|
178
212
|
.cfi_endproc
|
|
179
213
|
|
|
@@ -17,13 +17,41 @@
|
|
|
17
17
|
* AVX2 Dilithium implementation @[REF_AVX2].
|
|
18
18
|
*/
|
|
19
19
|
|
|
20
|
+
/*yaml
|
|
21
|
+
Name: pointwise_avx2_asm
|
|
22
|
+
Description: x86_64 AVX2 pointwise Montgomery multiplication
|
|
23
|
+
Signature: void mld_pointwise_avx2_asm(int32_t *a, const int32_t *b, const int32_t *qdata)
|
|
24
|
+
ABI:
|
|
25
|
+
Architecture: x86_64
|
|
26
|
+
CallingConvention: SysV
|
|
27
|
+
Features: [AVX2]
|
|
28
|
+
rdi:
|
|
29
|
+
type: buffer
|
|
30
|
+
size_bytes: 1024
|
|
31
|
+
permissions: read/write
|
|
32
|
+
c_parameter: int32_t *a
|
|
33
|
+
description: Input/output polynomial (256 x int32_t)
|
|
34
|
+
rsi:
|
|
35
|
+
type: buffer
|
|
36
|
+
size_bytes: 1024
|
|
37
|
+
permissions: read-only
|
|
38
|
+
c_parameter: const int32_t *b
|
|
39
|
+
description: Input polynomial (256 x int32_t)
|
|
40
|
+
rdx:
|
|
41
|
+
type: buffer
|
|
42
|
+
size_bytes: 2496
|
|
43
|
+
permissions: read-only
|
|
44
|
+
c_parameter: const int32_t *qdata
|
|
45
|
+
description: Precomputed constants (624 x int32_t)
|
|
46
|
+
*/
|
|
47
|
+
|
|
20
48
|
#include "../../../common.h"
|
|
21
49
|
#if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \
|
|
22
50
|
!defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)
|
|
23
51
|
|
|
24
52
|
/*
|
|
25
53
|
* WARNING: This file is auto-derived from the mldsa-native source file
|
|
26
|
-
* dev/x86_64/src/
|
|
54
|
+
* dev/x86_64/src/mldsa_pointwise_avx2_asm.S using scripts/simpasm. Do not modify it directly.
|
|
27
55
|
*/
|
|
28
56
|
|
|
29
57
|
.text
|
|
@@ -36,7 +64,7 @@ MLD_ASM_FN_SYMBOL(pointwise_avx2_asm)
|
|
|
36
64
|
vmovdqa (%rdx), %ymm1
|
|
37
65
|
xorl %eax, %eax
|
|
38
66
|
|
|
39
|
-
|
|
67
|
+
Lmld_pointwise_avx2_looptop1:
|
|
40
68
|
vmovdqa (%rdi), %ymm2
|
|
41
69
|
vmovdqa 0x20(%rdi), %ymm4
|
|
42
70
|
vmovdqa 0x40(%rdi), %ymm6
|
|
@@ -86,7 +114,7 @@ Lpointwise_avx2_looptop1:
|
|
|
86
114
|
addq $0x60, %rsi
|
|
87
115
|
addl $0x1, %eax
|
|
88
116
|
cmpl $0xa, %eax
|
|
89
|
-
jb
|
|
117
|
+
jb Lmld_pointwise_avx2_looptop1
|
|
90
118
|
vmovdqa (%rdi), %ymm2
|
|
91
119
|
vmovdqa 0x20(%rdi), %ymm4
|
|
92
120
|
vmovdqa (%rsi), %ymm10
|
|
@@ -18,14 +18,23 @@
|
|
|
18
18
|
*/
|
|
19
19
|
|
|
20
20
|
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
*
|
|
27
|
-
|
|
28
|
-
|
|
21
|
+
/*yaml
|
|
22
|
+
Name: poly_caddq_avx2_asm
|
|
23
|
+
Description: x86_64 AVX2 conditional addition of q to each coefficient.
|
|
24
|
+
For all coefficients of the in/out polynomial, add Q if the coefficient
|
|
25
|
+
is negative.
|
|
26
|
+
Signature: void mld_poly_caddq_avx2_asm(int32_t *r)
|
|
27
|
+
ABI:
|
|
28
|
+
Architecture: x86_64
|
|
29
|
+
CallingConvention: SysV
|
|
30
|
+
Features: [AVX2]
|
|
31
|
+
rdi:
|
|
32
|
+
type: buffer
|
|
33
|
+
size_bytes: 1024
|
|
34
|
+
permissions: read/write
|
|
35
|
+
c_parameter: int32_t *r
|
|
36
|
+
description: Input/output polynomial (256 x int32_t)
|
|
37
|
+
*/
|
|
29
38
|
|
|
30
39
|
#include "../../../common.h"
|
|
31
40
|
|
|
@@ -35,7 +44,7 @@
|
|
|
35
44
|
|
|
36
45
|
/*
|
|
37
46
|
* WARNING: This file is auto-derived from the mldsa-native source file
|
|
38
|
-
* dev/x86_64/src/
|
|
47
|
+
* dev/x86_64/src/mldsa_poly_caddq_avx2_asm.S using scripts/simpasm. Do not modify it directly.
|
|
39
48
|
*/
|
|
40
49
|
|
|
41
50
|
.text
|
data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_poly_chknorm_avx2_asm.S
ADDED
|
@@ -0,0 +1,176 @@
|
|
|
1
|
+
/*
|
|
2
|
+
* Copyright (c) The mldsa-native project authors
|
|
3
|
+
* SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT
|
|
4
|
+
*/
|
|
5
|
+
|
|
6
|
+
/* References
|
|
7
|
+
* ==========
|
|
8
|
+
*
|
|
9
|
+
* - [REF_AVX2]
|
|
10
|
+
* CRYSTALS-Dilithium optimized AVX2 implementation
|
|
11
|
+
* Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé
|
|
12
|
+
* https://github.com/pq-crystals/dilithium/tree/master/avx2
|
|
13
|
+
*/
|
|
14
|
+
|
|
15
|
+
/*
|
|
16
|
+
* This file is derived from the public domain
|
|
17
|
+
* AVX2 Dilithium implementation @[REF_AVX2].
|
|
18
|
+
*/
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
/*yaml
|
|
22
|
+
Name: poly_chknorm_avx2_asm
|
|
23
|
+
Description: x86_64 AVX2 infinity-norm bound check on polynomial coefficients.
|
|
24
|
+
Check the infinity norm of the polynomial against the given bound B.
|
|
25
|
+
Returns 0 if the norm is strictly smaller than B; otherwise returns 1
|
|
26
|
+
(i.e. returns 1 if any |coefficient| >= B, 0 otherwise).
|
|
27
|
+
Signature: int mld_poly_chknorm_avx2_asm(const int32_t *a, int32_t B)
|
|
28
|
+
ABI:
|
|
29
|
+
Architecture: x86_64
|
|
30
|
+
CallingConvention: SysV
|
|
31
|
+
Features: [AVX2]
|
|
32
|
+
rdi:
|
|
33
|
+
type: buffer
|
|
34
|
+
size_bytes: 1024
|
|
35
|
+
permissions: read-only
|
|
36
|
+
c_parameter: const int32_t *a
|
|
37
|
+
description: Input polynomial (256 x int32_t)
|
|
38
|
+
rsi:
|
|
39
|
+
type: scalar
|
|
40
|
+
c_parameter: int32_t B
|
|
41
|
+
description: Norm bound (must be non-negative)
|
|
42
|
+
test_with: 131072 # representative non-negative bound (1 << 17)
|
|
43
|
+
*/
|
|
44
|
+
|
|
45
|
+
#include "../../../common.h"
|
|
46
|
+
|
|
47
|
+
#if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \
|
|
48
|
+
!defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
/*
|
|
52
|
+
* WARNING: This file is auto-derived from the mldsa-native source file
|
|
53
|
+
* dev/x86_64/src/mldsa_poly_chknorm_avx2_asm.S using scripts/simpasm. Do not modify it directly.
|
|
54
|
+
*/
|
|
55
|
+
|
|
56
|
+
.text
|
|
57
|
+
.balign 4
|
|
58
|
+
.global MLD_ASM_NAMESPACE(poly_chknorm_avx2_asm)
|
|
59
|
+
MLD_ASM_FN_SYMBOL(poly_chknorm_avx2_asm)
|
|
60
|
+
|
|
61
|
+
.cfi_startproc
|
|
62
|
+
subl $0x1, %esi
|
|
63
|
+
vpxor %xmm1, %xmm1, %xmm1
|
|
64
|
+
vmovd %esi, %xmm2
|
|
65
|
+
vpbroadcastd %xmm2, %ymm2
|
|
66
|
+
vpabsd (%rdi), %ymm0
|
|
67
|
+
vpcmpgtd %ymm2, %ymm0, %ymm0
|
|
68
|
+
vpor %ymm0, %ymm1, %ymm1
|
|
69
|
+
vpabsd 0x20(%rdi), %ymm3
|
|
70
|
+
vpcmpgtd %ymm2, %ymm3, %ymm3
|
|
71
|
+
vpor %ymm3, %ymm1, %ymm1
|
|
72
|
+
vpabsd 0x40(%rdi), %ymm4
|
|
73
|
+
vpcmpgtd %ymm2, %ymm4, %ymm4
|
|
74
|
+
vpor %ymm4, %ymm1, %ymm1
|
|
75
|
+
vpabsd 0x60(%rdi), %ymm5
|
|
76
|
+
vpcmpgtd %ymm2, %ymm5, %ymm5
|
|
77
|
+
vpor %ymm5, %ymm1, %ymm1
|
|
78
|
+
vpabsd 0x80(%rdi), %ymm0
|
|
79
|
+
vpcmpgtd %ymm2, %ymm0, %ymm0
|
|
80
|
+
vpor %ymm0, %ymm1, %ymm1
|
|
81
|
+
vpabsd 0xa0(%rdi), %ymm3
|
|
82
|
+
vpcmpgtd %ymm2, %ymm3, %ymm3
|
|
83
|
+
vpor %ymm3, %ymm1, %ymm1
|
|
84
|
+
vpabsd 0xc0(%rdi), %ymm4
|
|
85
|
+
vpcmpgtd %ymm2, %ymm4, %ymm4
|
|
86
|
+
vpor %ymm4, %ymm1, %ymm1
|
|
87
|
+
vpabsd 0xe0(%rdi), %ymm5
|
|
88
|
+
vpcmpgtd %ymm2, %ymm5, %ymm5
|
|
89
|
+
vpor %ymm5, %ymm1, %ymm1
|
|
90
|
+
vpabsd 0x100(%rdi), %ymm0
|
|
91
|
+
vpcmpgtd %ymm2, %ymm0, %ymm0
|
|
92
|
+
vpor %ymm0, %ymm1, %ymm1
|
|
93
|
+
vpabsd 0x120(%rdi), %ymm3
|
|
94
|
+
vpcmpgtd %ymm2, %ymm3, %ymm3
|
|
95
|
+
vpor %ymm3, %ymm1, %ymm1
|
|
96
|
+
vpabsd 0x140(%rdi), %ymm4
|
|
97
|
+
vpcmpgtd %ymm2, %ymm4, %ymm4
|
|
98
|
+
vpor %ymm4, %ymm1, %ymm1
|
|
99
|
+
vpabsd 0x160(%rdi), %ymm5
|
|
100
|
+
vpcmpgtd %ymm2, %ymm5, %ymm5
|
|
101
|
+
vpor %ymm5, %ymm1, %ymm1
|
|
102
|
+
vpabsd 0x180(%rdi), %ymm0
|
|
103
|
+
vpcmpgtd %ymm2, %ymm0, %ymm0
|
|
104
|
+
vpor %ymm0, %ymm1, %ymm1
|
|
105
|
+
vpabsd 0x1a0(%rdi), %ymm3
|
|
106
|
+
vpcmpgtd %ymm2, %ymm3, %ymm3
|
|
107
|
+
vpor %ymm3, %ymm1, %ymm1
|
|
108
|
+
vpabsd 0x1c0(%rdi), %ymm4
|
|
109
|
+
vpcmpgtd %ymm2, %ymm4, %ymm4
|
|
110
|
+
vpor %ymm4, %ymm1, %ymm1
|
|
111
|
+
vpabsd 0x1e0(%rdi), %ymm5
|
|
112
|
+
vpcmpgtd %ymm2, %ymm5, %ymm5
|
|
113
|
+
vpor %ymm5, %ymm1, %ymm1
|
|
114
|
+
vpabsd 0x200(%rdi), %ymm0
|
|
115
|
+
vpcmpgtd %ymm2, %ymm0, %ymm0
|
|
116
|
+
vpor %ymm0, %ymm1, %ymm1
|
|
117
|
+
vpabsd 0x220(%rdi), %ymm3
|
|
118
|
+
vpcmpgtd %ymm2, %ymm3, %ymm3
|
|
119
|
+
vpor %ymm3, %ymm1, %ymm1
|
|
120
|
+
vpabsd 0x240(%rdi), %ymm4
|
|
121
|
+
vpcmpgtd %ymm2, %ymm4, %ymm4
|
|
122
|
+
vpor %ymm4, %ymm1, %ymm1
|
|
123
|
+
vpabsd 0x260(%rdi), %ymm5
|
|
124
|
+
vpcmpgtd %ymm2, %ymm5, %ymm5
|
|
125
|
+
vpor %ymm5, %ymm1, %ymm1
|
|
126
|
+
vpabsd 0x280(%rdi), %ymm0
|
|
127
|
+
vpcmpgtd %ymm2, %ymm0, %ymm0
|
|
128
|
+
vpor %ymm0, %ymm1, %ymm1
|
|
129
|
+
vpabsd 0x2a0(%rdi), %ymm3
|
|
130
|
+
vpcmpgtd %ymm2, %ymm3, %ymm3
|
|
131
|
+
vpor %ymm3, %ymm1, %ymm1
|
|
132
|
+
vpabsd 0x2c0(%rdi), %ymm4
|
|
133
|
+
vpcmpgtd %ymm2, %ymm4, %ymm4
|
|
134
|
+
vpor %ymm4, %ymm1, %ymm1
|
|
135
|
+
vpabsd 0x2e0(%rdi), %ymm5
|
|
136
|
+
vpcmpgtd %ymm2, %ymm5, %ymm5
|
|
137
|
+
vpor %ymm5, %ymm1, %ymm1
|
|
138
|
+
vpabsd 0x300(%rdi), %ymm0
|
|
139
|
+
vpcmpgtd %ymm2, %ymm0, %ymm0
|
|
140
|
+
vpor %ymm0, %ymm1, %ymm1
|
|
141
|
+
vpabsd 0x320(%rdi), %ymm3
|
|
142
|
+
vpcmpgtd %ymm2, %ymm3, %ymm3
|
|
143
|
+
vpor %ymm3, %ymm1, %ymm1
|
|
144
|
+
vpabsd 0x340(%rdi), %ymm4
|
|
145
|
+
vpcmpgtd %ymm2, %ymm4, %ymm4
|
|
146
|
+
vpor %ymm4, %ymm1, %ymm1
|
|
147
|
+
vpabsd 0x360(%rdi), %ymm5
|
|
148
|
+
vpcmpgtd %ymm2, %ymm5, %ymm5
|
|
149
|
+
vpor %ymm5, %ymm1, %ymm1
|
|
150
|
+
vpabsd 0x380(%rdi), %ymm0
|
|
151
|
+
vpcmpgtd %ymm2, %ymm0, %ymm0
|
|
152
|
+
vpor %ymm0, %ymm1, %ymm1
|
|
153
|
+
vpabsd 0x3a0(%rdi), %ymm3
|
|
154
|
+
vpcmpgtd %ymm2, %ymm3, %ymm3
|
|
155
|
+
vpor %ymm3, %ymm1, %ymm1
|
|
156
|
+
vpabsd 0x3c0(%rdi), %ymm4
|
|
157
|
+
vpcmpgtd %ymm2, %ymm4, %ymm4
|
|
158
|
+
vpor %ymm4, %ymm1, %ymm1
|
|
159
|
+
vpabsd 0x3e0(%rdi), %ymm5
|
|
160
|
+
vpcmpgtd %ymm2, %ymm5, %ymm5
|
|
161
|
+
vpor %ymm5, %ymm1, %ymm1
|
|
162
|
+
xorl %eax, %eax
|
|
163
|
+
vptest %ymm1, %ymm1
|
|
164
|
+
setne %al
|
|
165
|
+
retq
|
|
166
|
+
.cfi_endproc
|
|
167
|
+
|
|
168
|
+
MLD_ASM_FN_SIZE(poly_chknorm_avx2_asm)
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
#endif /* MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_MULTILEVEL_NO_SHARED \
|
|
172
|
+
*/
|
|
173
|
+
|
|
174
|
+
#if defined(__ELF__)
|
|
175
|
+
.section .note.GNU-stack,"",%progbits
|
|
176
|
+
#endif
|