pq_crypto 0.6.5 → 0.6.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (149) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +134 -0
  3. data/README.md +5 -3
  4. data/ext/pqcrypto/extconf.rb +4 -1
  5. data/ext/pqcrypto/pq_externalmu.c +35 -0
  6. data/ext/pqcrypto/pqcrypto_native_api.h +85 -75
  7. data/ext/pqcrypto/pqcrypto_ruby_secure.c +16 -9
  8. data/ext/pqcrypto/pqcrypto_secure.c +43 -33
  9. data/ext/pqcrypto/pqcrypto_secure.h +52 -47
  10. data/ext/pqcrypto/pqcrypto_version.h +1 -1
  11. data/ext/pqcrypto/vendor/.vendored +7 -7
  12. data/ext/pqcrypto/vendor/mldsa-native/BUILDING.md +5 -2
  13. data/ext/pqcrypto/vendor/mldsa-native/LICENSE +21 -2
  14. data/ext/pqcrypto/vendor/mldsa-native/README.md +20 -7
  15. data/ext/pqcrypto/vendor/mldsa-native/RELEASE.md +160 -0
  16. data/ext/pqcrypto/vendor/mldsa-native/SECURITY.md +1 -1
  17. data/ext/pqcrypto/vendor/mldsa-native/mldsa/README.md +2 -2
  18. data/ext/pqcrypto/vendor/mldsa-native/mldsa/mldsa_native.c +85 -59
  19. data/ext/pqcrypto/vendor/mldsa-native/mldsa/mldsa_native.h +292 -348
  20. data/ext/pqcrypto/vendor/mldsa-native/mldsa/mldsa_native_asm.S +122 -76
  21. data/ext/pqcrypto/vendor/mldsa-native/mldsa/mldsa_native_config.h +184 -86
  22. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/cbmc.h +49 -4
  23. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/common.h +49 -81
  24. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/context.h +152 -0
  25. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/ct.h +25 -12
  26. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/debug.c +2 -0
  27. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/debug.h +2 -0
  28. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/fips202x4.c +2 -2
  29. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/keccakf1600.c +9 -11
  30. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/auto.h +19 -11
  31. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x1_scalar_aarch64_asm.S +6 -4
  32. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x1_v84a_aarch64_asm.S +7 -4
  33. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x2_v84a_aarch64_asm.S +7 -4
  34. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm.S +12 -9
  35. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm.S +12 -9
  36. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/x1_scalar.h +1 -1
  37. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/x1_v84a.h +3 -2
  38. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/x2_v84a.h +3 -2
  39. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/x4_v8a_scalar.h +6 -1
  40. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/x4_v8a_v84a_scalar.h +3 -2
  41. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/api.h +11 -11
  42. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/armv81m/mve.h +9 -22
  43. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/armv81m/src/keccak_f1600_x4_mve.S +8 -5
  44. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/armv81m/src/keccak_f1600_x4_mve.c +1 -0
  45. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/armv81m/src/{state_extract_bytes_x4_mve.S → keccak_f1600_x4_state_extract_bytes_mve.S} +14 -14
  46. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/armv81m/src/{state_xor_bytes_x4_mve.S → keccak_f1600_x4_state_xor_bytes_mve.S} +12 -12
  47. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/auto.h +5 -4
  48. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/x86_64/keccak_f1600_x4_avx2.h +2 -2
  49. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/x86_64/src/fips202_native_x86_64.h +1 -0
  50. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/x86_64/src/keccak_f1600_x4_avx2_asm.S +36 -2
  51. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/meta.h +62 -4
  52. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/arith_native_aarch64.h +87 -54
  53. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{intt_aarch64_asm.S → mldsa_intt_aarch64_asm.S} +39 -6
  54. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{ntt_aarch64_asm.S → mldsa_ntt_aarch64_asm.S} +39 -6
  55. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{pointwise_montgomery_aarch64_asm.S → mldsa_pointwise_montgomery_aarch64_asm.S} +25 -3
  56. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{poly_caddq_aarch64_asm.S → mldsa_poly_caddq_aarch64_asm.S} +19 -3
  57. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{poly_chknorm_aarch64_asm.S → mldsa_poly_chknorm_aarch64_asm.S} +24 -3
  58. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{poly_decompose_32_aarch64_asm.S → mldsa_poly_decompose_32_aarch64_asm.S} +25 -3
  59. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{poly_decompose_88_aarch64_asm.S → mldsa_poly_decompose_88_aarch64_asm.S} +25 -3
  60. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{poly_use_hint_32_aarch64_asm.S → mldsa_poly_use_hint_32_aarch64_asm.S} +25 -3
  61. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{poly_use_hint_88_aarch64_asm.S → mldsa_poly_use_hint_88_aarch64_asm.S} +25 -3
  62. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{mld_polyvecl_pointwise_acc_montgomery_l4_aarch64_asm.S → mldsa_polyvecl_pointwise_acc_montgomery_l4_aarch64_asm.S} +31 -3
  63. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{mld_polyvecl_pointwise_acc_montgomery_l5_aarch64_asm.S → mldsa_polyvecl_pointwise_acc_montgomery_l5_aarch64_asm.S} +31 -3
  64. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{mld_polyvecl_pointwise_acc_montgomery_l7_aarch64_asm.S → mldsa_polyvecl_pointwise_acc_montgomery_l7_aarch64_asm.S} +31 -3
  65. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{polyz_unpack_17_aarch64_asm.S → mldsa_polyz_unpack_17_aarch64_asm.S} +31 -3
  66. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{polyz_unpack_19_aarch64_asm.S → mldsa_polyz_unpack_19_aarch64_asm.S} +31 -3
  67. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{rej_uniform_aarch64_asm.S → mldsa_rej_uniform_aarch64_asm.S} +48 -15
  68. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{rej_uniform_eta2_aarch64_asm.S → mldsa_rej_uniform_eta2_aarch64_asm.S} +42 -9
  69. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{rej_uniform_eta4_aarch64_asm.S → mldsa_rej_uniform_eta4_aarch64_asm.S} +42 -9
  70. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/api.h +11 -3
  71. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/meta.h +3 -2
  72. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/meta.h +28 -28
  73. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/arith_native_x86_64.h +171 -49
  74. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{intt_avx2_asm.S → mldsa_intt_avx2_asm.S} +23 -1
  75. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{ntt_avx2_asm.S → mldsa_ntt_avx2_asm.S} +23 -1
  76. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{nttunpack_avx2_asm.S → mldsa_nttunpack_avx2_asm.S} +17 -1
  77. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{pointwise_acc_l4_avx2_asm.S → mldsa_pointwise_acc_l4_avx2_asm.S} +37 -3
  78. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{pointwise_acc_l5_avx2_asm.S → mldsa_pointwise_acc_l5_avx2_asm.S} +37 -3
  79. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{pointwise_acc_l7_avx2_asm.S → mldsa_pointwise_acc_l7_avx2_asm.S} +37 -3
  80. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{pointwise_avx2_asm.S → mldsa_pointwise_avx2_asm.S} +31 -3
  81. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{poly_caddq_avx2_asm.S → mldsa_poly_caddq_avx2_asm.S} +18 -9
  82. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_poly_chknorm_avx2_asm.S +176 -0
  83. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_poly_decompose_32_avx2_asm.S +490 -0
  84. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_poly_decompose_88_avx2_asm.S +489 -0
  85. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_poly_use_hint_32_avx2_asm.S +123 -0
  86. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_poly_use_hint_88_avx2_asm.S +125 -0
  87. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_polyz_unpack_17_avx2_asm.S +355 -0
  88. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_polyz_unpack_19_avx2_asm.S +355 -0
  89. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_rej_uniform_avx2_asm.S +132 -0
  90. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_rej_uniform_eta2_avx2_asm.S +205 -0
  91. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_rej_uniform_eta4_avx2_asm.S +176 -0
  92. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/packing.c +27 -36
  93. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/packing.h +42 -8
  94. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/params.h +93 -17
  95. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/poly.c +74 -15
  96. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/poly.h +97 -11
  97. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/poly_kl.c +7 -38
  98. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/poly_kl.h +49 -7
  99. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/polyvec.c +16 -17
  100. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/polyvec.h +26 -9
  101. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/polyvec_lazy.c +3 -0
  102. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/polyvec_lazy.h +18 -19
  103. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/reduce.h +15 -3
  104. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/rounding.h +28 -6
  105. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/sign.c +311 -246
  106. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/sign.h +245 -240
  107. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/sys.h +64 -5
  108. data/ext/pqcrypto/vendor/mlkem-native/BUILDING.md +5 -2
  109. data/ext/pqcrypto/vendor/mlkem-native/LICENSE +21 -3
  110. data/ext/pqcrypto/vendor/mlkem-native/README.md +2 -2
  111. data/ext/pqcrypto/vendor/mlkem-native/RELEASE.md +113 -0
  112. data/ext/pqcrypto/vendor/mlkem-native/mlkem/README.md +2 -2
  113. data/ext/pqcrypto/vendor/mlkem-native/mlkem/mlkem_native.c +17 -27
  114. data/ext/pqcrypto/vendor/mlkem-native/mlkem/mlkem_native.h +68 -151
  115. data/ext/pqcrypto/vendor/mlkem-native/mlkem/mlkem_native_asm.S +17 -27
  116. data/ext/pqcrypto/vendor/mlkem-native/mlkem/mlkem_native_config.h +46 -44
  117. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/cbmc.h +25 -0
  118. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/common.h +37 -6
  119. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/context.h +9 -0
  120. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/fips202.h +2 -2
  121. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/keccakf1600.c +8 -8
  122. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x1_scalar.h +1 -1
  123. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x1_v84a.h +3 -3
  124. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x2_v84a.h +3 -3
  125. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x4_v8a_scalar.h +2 -2
  126. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x4_v8a_v84a_scalar.h +3 -3
  127. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/api.h +11 -11
  128. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/armv81m/mve.h +3 -3
  129. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/x86_64/keccak_f1600_x4_avx2.h +2 -2
  130. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/indcpa.h +14 -11
  131. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/kem.c +28 -11
  132. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/kem.h +39 -14
  133. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/meta.h +10 -10
  134. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/meta.h +20 -20
  135. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/sys.h +5 -5
  136. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/verify.h +11 -10
  137. data/lib/pq_crypto/version.rb +1 -1
  138. data/script/vendor_libs.rb +6 -6
  139. metadata +40 -38
  140. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/poly_chknorm_avx2.c +0 -52
  141. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/poly_decompose_32_avx2.c +0 -157
  142. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/poly_decompose_88_avx2.c +0 -157
  143. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/poly_use_hint_32_avx2.c +0 -103
  144. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/poly_use_hint_88_avx2.c +0 -105
  145. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/polyz_unpack_17_avx2.c +0 -94
  146. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/polyz_unpack_19_avx2.c +0 -96
  147. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/rej_uniform_avx2.c +0 -126
  148. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/rej_uniform_eta2_avx2.c +0 -157
  149. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/rej_uniform_eta4_avx2.c +0 -141
@@ -1,105 +0,0 @@
1
- /*
2
- * Copyright (c) The mldsa-native project authors
3
- * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT
4
- */
5
-
6
- /* References
7
- * ==========
8
- *
9
- * - [REF_AVX2]
10
- * CRYSTALS-Dilithium optimized AVX2 implementation
11
- * Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé
12
- * https://github.com/pq-crystals/dilithium/tree/master/avx2
13
- */
14
-
15
- /*
16
- * This file is derived from the public domain
17
- * AVX2 Dilithium implementation @[REF_AVX2].
18
- */
19
-
20
- #include "../../../common.h"
21
-
22
- #if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \
23
- !defined(MLD_CONFIG_NO_VERIFY_API) && \
24
- !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \
25
- (defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \
26
- MLD_CONFIG_PARAMETER_SET == 44)
27
-
28
- #include <immintrin.h>
29
- #include "arith_native_x86_64.h"
30
- #include "consts.h"
31
-
32
- #define MLD_MM256_BLENDV_EPI32(a, b, mask) \
33
- _mm256_castps_si256(_mm256_blendv_ps(_mm256_castsi256_ps(a), \
34
- _mm256_castsi256_ps(b), \
35
- _mm256_castsi256_ps(mask)))
36
-
37
- void mld_poly_use_hint_88_avx2(int32_t *a, const int32_t *hint)
38
- {
39
- unsigned int i;
40
- __m256i f, f0, f1, h, t;
41
- const __m256i q_bound = _mm256_set1_epi32(87 * ((MLDSA_Q - 1) / 88));
42
- /* check-magic: 11275 == floor(2**24 / 1488) */
43
- const __m256i v = _mm256_set1_epi32(11275);
44
- const __m256i alpha = _mm256_set1_epi32(2 * ((MLDSA_Q - 1) / 88));
45
- const __m256i off = _mm256_set1_epi32(127);
46
- const __m256i shift = _mm256_set1_epi32(128);
47
- const __m256i max = _mm256_set1_epi32(43);
48
- const __m256i zero = _mm256_setzero_si256();
49
-
50
- for (i = 0; i < MLDSA_N / 8; i++)
51
- {
52
- f = _mm256_load_si256((const __m256i *)&a[8 * i]);
53
- h = _mm256_load_si256((const __m256i *)&hint[8 * i]);
54
-
55
- /* Reference:
56
- * - @[REF_AVX2] calls poly_decompose to compute all a1, a0 before the loop.
57
- * - Our implementation of decompose() is slightly different from that in
58
- * @[REF_AVX2]. See poly_decompose_88_avx2.c for more information.
59
- */
60
- /* f1, f2 = decompose(f) */
61
- f1 = _mm256_add_epi32(f, off);
62
- f1 = _mm256_srli_epi32(f1, 7);
63
- f1 = _mm256_mulhi_epu16(f1, v);
64
- f1 = _mm256_mulhrs_epi16(f1, shift);
65
- t = _mm256_cmpgt_epi32(f, q_bound);
66
- f0 = _mm256_mullo_epi32(f1, alpha);
67
- f0 = _mm256_sub_epi32(f, f0);
68
- f1 = _mm256_andnot_si256(t, f1);
69
- f0 = _mm256_add_epi32(f0, t);
70
-
71
- /* Reference: The reference avx2 implementation checks a0 >= 0, which is
72
- * different from the specification and the reference C implementation. We
73
- * follow the specification and check a0 > 0.
74
- */
75
- /* t = (f0 > 0) ? h : -h */
76
- f0 = _mm256_cmpgt_epi32(f0, zero);
77
- t = MLD_MM256_BLENDV_EPI32(h, zero, f0);
78
- t = _mm256_slli_epi32(t, 1);
79
- h = _mm256_sub_epi32(h, t);
80
-
81
- /* f1 = (f1 + t) % 44 */
82
- f1 = _mm256_add_epi32(f1, h);
83
- f1 = MLD_MM256_BLENDV_EPI32(f1, max, f1);
84
- f = _mm256_cmpgt_epi32(f1, max);
85
- f1 = MLD_MM256_BLENDV_EPI32(f1, zero, f);
86
-
87
- _mm256_store_si256((__m256i *)&a[8 * i], f1);
88
- }
89
- }
90
-
91
- #else /* MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_NO_VERIFY_API && \
92
- !MLD_CONFIG_MULTILEVEL_NO_SHARED && \
93
- (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 44) \
94
- */
95
-
96
- MLD_EMPTY_CU(avx2_poly_use_hint_88)
97
-
98
- #endif /* !(MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_NO_VERIFY_API && \
99
- !MLD_CONFIG_MULTILEVEL_NO_SHARED && \
100
- (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == \
101
- 44)) */
102
-
103
- /* To facilitate single-compilation-unit (SCU) builds, undefine all macros.
104
- * Don't modify by hand -- this is auto-generated by scripts/autogen. */
105
- #undef MLD_MM256_BLENDV_EPI32
@@ -1,94 +0,0 @@
1
- /*
2
- * Copyright (c) The mldsa-native project authors
3
- * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT
4
- */
5
-
6
- /* References
7
- * ==========
8
- *
9
- * - [REF_AVX2]
10
- * CRYSTALS-Dilithium optimized AVX2 implementation
11
- * Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé
12
- * https://github.com/pq-crystals/dilithium/tree/master/avx2
13
- */
14
-
15
- /*
16
- * This file is derived from the public domain
17
- * AVX2 Dilithium implementation @[REF_AVX2].
18
- */
19
-
20
- #include "../../../common.h"
21
-
22
- #if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \
23
- (!defined(MLD_CONFIG_NO_SIGN_API) || \
24
- !defined(MLD_CONFIG_NO_VERIFY_API)) && \
25
- !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \
26
- (defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \
27
- MLD_CONFIG_PARAMETER_SET == 44)
28
-
29
- #include <immintrin.h>
30
- #include "arith_native_x86_64.h"
31
-
32
- void mld_polyz_unpack_17_avx2(int32_t *r, const uint8_t *a)
33
- {
34
- unsigned int i;
35
- __m256i f;
36
- __m128i low, high;
37
-
38
- const __m256i shufbidx = _mm256_set_epi8(
39
- -1, 31, 30, 29, -1, 29, 28, 27, -1, 27, 26, 25, -1, 25, 24, 23, -1, 8, 7,
40
- 6, -1, 6, 5, 4, -1, 4, 3, 2, -1, 2, 1, 0);
41
- const __m256i srlvdidx = _mm256_set_epi32(6, 4, 2, 0, 6, 4, 2, 0);
42
- const __m256i mask = _mm256_set1_epi32(0x3FFFF);
43
- const __m256i gamma1 = _mm256_set1_epi32((1 << 17));
44
-
45
- for (i = 0; i < MLDSA_N / 8; i++)
46
- {
47
- /* Load bytes 0..15 into low 128-bit vector */
48
- low = _mm_loadu_si128((__m128i *)&a[18 * i]);
49
- /* Load bytes 2..17 into high 128-bit vector */
50
- high = _mm_loadu_si128((__m128i *)&a[18 * i + 2]);
51
- /* Combine into 256-bit vector */
52
- f = _mm256_inserti128_si256(_mm256_castsi128_si256(low), high, 1);
53
-
54
- /* Shuffling 8-bit lanes
55
- *
56
- * ┌─ Indices 0-8 into low 128-bit half ───────────────────────────────────┐
57
- * │ Shuffle: [-1, 8, 7, 6, -1, 6, 5, 4, -1, 4, 3, 2, -1, 2, 1, 0] │
58
- * │ Result: [0, byte8, byte7, byte6, ..., 0, byte2, byte1, byte0] │
59
- * └───────────────────────────────────────────────────────────────────────┘
60
- *
61
- * ┌─ Indices 16-31 into high 128-bit half ────────────────────────────────┐
62
- * │ Shuffle: [-1,31, 30, 29, -1,29, 28, 27, -1,27, 26, 25, -1,25, 24, 23] │
63
- * │ Result: [0, byte17, byte16, byte15, ..., 0, byte11, byte10, byte9] │
64
- * └───────────────────────────────────────────────────────────────────────┘
65
- */
66
- f = _mm256_shuffle_epi8(f, shufbidx);
67
-
68
- /* Keep only 18 out of 24 bits in each 32-bit lane */
69
- /* Bits 0..23 16..39 32..55 48..71
70
- * 72..95 88..111 104..127 120..143 */
71
- f = _mm256_srlv_epi32(f, srlvdidx);
72
- /* Bits 0..23 18..39 36..55 54..71
73
- * 72..95 90..111 108..127 126..143 */
74
- f = _mm256_and_si256(f, mask);
75
- /* Bits 0..17 18..35 36..53 54..71
76
- * 72..89 90..107 108..125 126..143 */
77
-
78
- /* Map [0, 1, ..., 2^18-1] to [2^17, 2^17-1, ..., -2^17+1] */
79
- f = _mm256_sub_epi32(gamma1, f);
80
-
81
- _mm256_store_si256((__m256i *)&r[8 * i], f);
82
- }
83
- }
84
- #else /* MLD_ARITH_BACKEND_X86_64_DEFAULT && (!MLD_CONFIG_NO_SIGN_API || \
85
- !MLD_CONFIG_NO_VERIFY_API) && !MLD_CONFIG_MULTILEVEL_NO_SHARED && \
86
- (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 44) \
87
- */
88
-
89
- MLD_EMPTY_CU(avx2_polyz_unpack_17)
90
-
91
- #endif /* !(MLD_ARITH_BACKEND_X86_64_DEFAULT && (!MLD_CONFIG_NO_SIGN_API || \
92
- !MLD_CONFIG_NO_VERIFY_API) && !MLD_CONFIG_MULTILEVEL_NO_SHARED && \
93
- (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == \
94
- 44)) */
@@ -1,96 +0,0 @@
1
- /*
2
- * Copyright (c) The mldsa-native project authors
3
- * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT
4
- */
5
-
6
- /* References
7
- * ==========
8
- *
9
- * - [REF_AVX2]
10
- * CRYSTALS-Dilithium optimized AVX2 implementation
11
- * Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé
12
- * https://github.com/pq-crystals/dilithium/tree/master/avx2
13
- */
14
-
15
- /*
16
- * This file is derived from the public domain
17
- * AVX2 Dilithium implementation @[REF_AVX2].
18
- */
19
-
20
- #include "../../../common.h"
21
-
22
- #if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \
23
- (!defined(MLD_CONFIG_NO_SIGN_API) || \
24
- !defined(MLD_CONFIG_NO_VERIFY_API)) && \
25
- !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \
26
- (defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \
27
- (MLD_CONFIG_PARAMETER_SET == 65 || MLD_CONFIG_PARAMETER_SET == 87))
28
-
29
- #include <immintrin.h>
30
- #include "arith_native_x86_64.h"
31
-
32
- void mld_polyz_unpack_19_avx2(int32_t *r, const uint8_t *a)
33
- {
34
- unsigned int i;
35
- __m256i f;
36
- __m128i low, high;
37
-
38
- const __m256i shufbidx = _mm256_set_epi8(
39
- -1, 31, 30, 29, -1, 29, 28, 27, -1, 26, 25, 24, -1, 24, 23, 22, -1, 9, 8,
40
- 7, -1, 7, 6, 5, -1, 4, 3, 2, -1, 2, 1, 0);
41
- /* Equivalent to _mm256_set_epi32(4, 0, 4, 0, 4, 0, 4, 0) */
42
- const __m256i srlvdidx = _mm256_set1_epi64x((uint64_t)4 << 32);
43
- const __m256i mask = _mm256_set1_epi32(0xFFFFF);
44
- const __m256i gamma1 = _mm256_set1_epi32((1 << 19));
45
-
46
- for (i = 0; i < MLDSA_N / 8; i++)
47
- {
48
- /* Load bytes 0..15 into low 128-bit vector */
49
- low = _mm_loadu_si128((__m128i *)&a[20 * i]);
50
- /* Load bytes 4..19 into high 128-bit vector */
51
- high = _mm_loadu_si128((__m128i *)&a[20 * i + 4]);
52
- /* Combine into 256-bit vector */
53
- f = _mm256_inserti128_si256(_mm256_castsi128_si256(low), high, 1);
54
-
55
- /* Shuffling 8-bit lanes
56
- *
57
- * ┌─ Indices 0-9 into low 128-bit half ───────────────────────────────────┐
58
- * │ Shuffle: [-1, 9, 8, 7, -1, 7, 6, 5, -1, 4, 3, 2, -1, 2, 1, 0] │
59
- * │ Result: [0, byte9, byte8, byte7, ..., 0, byte2, byte1, byte0] │
60
- * └───────────────────────────────────────────────────────────────────────┘
61
- *
62
- * ┌─ Indices 16-31 into high 128-bit half ────────────────────────────────┐
63
- * │ Shuffle: [-1,31, 30, 29, -1,29, 28, 27, -1,26, 25, 24, -1,24, 23, 22] │
64
- * │ Result: [0, byte19, byte18, byte17, ..., 0, byte12, byte11, byte10] │
65
- * └───────────────────────────────────────────────────────────────────────┘
66
- */
67
- f = _mm256_shuffle_epi8(f, shufbidx);
68
-
69
- /* Keep only 20 out of 24 bits in each 32-bit lane */
70
- /* Bits 0..23 16..39 40..63 56..79
71
- * 80..103 96..119 120..143 136..159 */
72
- f = _mm256_srlv_epi32(f, srlvdidx);
73
- /* Bits 0..23 20..39 40..63 60..79
74
- * 80..103 100..119 120..143 140..159 */
75
- f = _mm256_and_si256(f, mask);
76
- /* Bits 0..19 20..39 40..59 60..79
77
- * 80..99 100..119 120..139 140..159 */
78
-
79
- /* Map [0, 1, ..., 2^20-1] to [2^19, 2^19-1, ..., -2^19+1] */
80
- f = _mm256_sub_epi32(gamma1, f);
81
-
82
- _mm256_store_si256((__m256i *)&r[8 * i], f);
83
- }
84
- }
85
-
86
- #else /* MLD_ARITH_BACKEND_X86_64_DEFAULT && (!MLD_CONFIG_NO_SIGN_API || \
87
- !MLD_CONFIG_NO_VERIFY_API) && !MLD_CONFIG_MULTILEVEL_NO_SHARED && \
88
- (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 65 \
89
- || MLD_CONFIG_PARAMETER_SET == 87) */
90
-
91
- MLD_EMPTY_CU(avx2_polyz_unpack_19)
92
-
93
- #endif /* !(MLD_ARITH_BACKEND_X86_64_DEFAULT && (!MLD_CONFIG_NO_SIGN_API || \
94
- !MLD_CONFIG_NO_VERIFY_API) && !MLD_CONFIG_MULTILEVEL_NO_SHARED && \
95
- (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 65 \
96
- || MLD_CONFIG_PARAMETER_SET == 87)) */
@@ -1,126 +0,0 @@
1
- /*
2
- * Copyright (c) The mldsa-native project authors
3
- * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT
4
- */
5
-
6
- /* References
7
- * ==========
8
- *
9
- * - [REF_AVX2]
10
- * CRYSTALS-Dilithium optimized AVX2 implementation
11
- * Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé
12
- * https://github.com/pq-crystals/dilithium/tree/master/avx2
13
- */
14
-
15
- /*
16
- * This file is derived from the public domain
17
- * AVX2 Dilithium implementation @[REF_AVX2].
18
- */
19
-
20
- #include "../../../common.h"
21
-
22
- #if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \
23
- !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)
24
-
25
- #include <immintrin.h>
26
- #include "arith_native_x86_64.h"
27
- #include "consts.h"
28
-
29
- /*
30
- * Reference: The pqcrystals implementation assumes a buffer that is 8 bytes
31
- *. larger as the first loop overreads by 8 bytes that are then
32
- * discarded. We instead do not pad the buffer and do not overread.
33
- * The performance impact is negligible and it does not force the
34
- * frontend to perform the unintuitive padding.
35
- */
36
-
37
- unsigned int mld_rej_uniform_avx2(
38
- int32_t *MLD_RESTRICT r, const uint8_t buf[MLD_AVX2_REJ_UNIFORM_BUFLEN])
39
- {
40
- unsigned int ctr, pos;
41
- uint32_t good;
42
- __m256i d, tmp;
43
- const __m256i bound = _mm256_set1_epi32(MLDSA_Q);
44
- const __m256i mask = _mm256_set1_epi32(0x7FFFFF);
45
- const __m256i idx8 =
46
- _mm256_set_epi8(-1, 15, 14, 13, -1, 12, 11, 10, -1, 9, 8, 7, -1, 6, 5, 4,
47
- -1, 11, 10, 9, -1, 8, 7, 6, -1, 5, 4, 3, -1, 2, 1, 0);
48
-
49
- ctr = pos = 0;
50
- while (ctr <= MLDSA_N - 8 && pos <= MLD_AVX2_REJ_UNIFORM_BUFLEN - 32)
51
- {
52
- d = _mm256_loadu_si256((__m256i *)&buf[pos]);
53
-
54
- /* Permute 64-bit lanes
55
- * 0x94 = 10010100b rearranges 64-bit lanes as: [3,2,1,0] -> [2,1,1,0]
56
- *
57
- * ╔═══════════════════════════════════════════════════════════════════════╗
58
- * ║ Original Layout ║
59
- * ╚═══════════════════════════════════════════════════════════════════════╝
60
- * ┌─────────────────┬─────────────────┬─────────────────┬─────────────────┐
61
- * │ Lane 0 │ Lane 1 │ Lane 2 │ Lane 3 │
62
- * │ bytes 0..7 │ bytes 8..15 │ bytes 16..23 │ bytes 24..31 │
63
- * └─────────────────┴─────────────────┴─────────────────┴─────────────────┘
64
- *
65
- * ╔═══════════════════════════════════════════════════════════════════════╗
66
- * ║ Layout after permute ║
67
- * ║ Byte indices in high half shifted down by 8 positions ║
68
- * ╚═══════════════════════════════════════════════════════════════════════╝
69
- * ┌───────────────┬─────────────────┐ ┌─────────────────┬─────────────────┐
70
- * │ Lane 0 │ Lane 1 │ │ Lane 2 │ Lane 3 │
71
- * │ bytes 0..7 │ bytes 8..15 │ │ bytes 8..15 │ bytes 16..23 │
72
- * └───────────────┴─────────────────┘ └─────────────────┴─────────────────┘
73
- * Lower 128-bit lane (bytes 0-15) Upper 128-bit lane (bytes 16-31)
74
- */
75
- d = _mm256_permute4x64_epi64(d, 0x94);
76
-
77
- /* Shuffling 8-bit lanes
78
- *
79
- * ┌─ Indices 0-11 into low 128-bit half of permuted vector────────────────┐
80
- * │ Shuffle: [-1, 11, 10, 9, -1, 8, 7, 6, -1, 5, 4, 3, -1, 2, 1, 0] │
81
- * │ Result: [0, byte11, byte10, byte9, ..., 0, byte2, byte1, byte0] │
82
- * └───────────────────────────────────────────────────────────────────────┘
83
- *
84
- * ┌─ Indices 4-15 into high 128-bit half of permuted vector ──────────────┐
85
- * │ Shuffle: [-1, 15, 14, 13, -1, 12, 11, 10, -1, 9, 8, 7, -1, 6, 5, 4] │
86
- * │ Result: [0, byte23, byte22, byte21, ..., 0, byte14, byte13, byte12 │
87
- * └───────────────────────────────────────────────────────────────────────┘
88
- */
89
- d = _mm256_shuffle_epi8(d, idx8);
90
- d = _mm256_and_si256(d, mask);
91
- pos += 24;
92
-
93
- tmp = _mm256_sub_epi32(d, bound);
94
- good = (uint32_t)_mm256_movemask_ps((__m256)tmp);
95
- tmp = _mm256_cvtepu8_epi32(
96
- _mm_loadl_epi64((__m128i *)&mld_rej_uniform_table[good]));
97
- d = _mm256_permutevar8x32_epi32(d, tmp);
98
-
99
- _mm256_storeu_si256((__m256i *)&r[ctr], d);
100
- ctr += (unsigned)_mm_popcnt_u32(good);
101
- }
102
-
103
- while (ctr < MLDSA_N && pos <= MLD_AVX2_REJ_UNIFORM_BUFLEN - 3)
104
- {
105
- uint32_t t = buf[pos++];
106
- t |= (uint32_t)buf[pos++] << 8;
107
- t |= (uint32_t)buf[pos++] << 16;
108
- t &= 0x7FFFFF;
109
-
110
- if (t < MLDSA_Q)
111
- {
112
- /* Safe because t < MLDSA_Q. */
113
- r[ctr++] = (int32_t)t;
114
- }
115
- }
116
-
117
- return ctr;
118
- }
119
-
120
- #else /* MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_MULTILEVEL_NO_SHARED \
121
- */
122
-
123
- MLD_EMPTY_CU(avx2_rej_uniform)
124
-
125
- #endif /* !(MLD_ARITH_BACKEND_X86_64_DEFAULT && \
126
- !MLD_CONFIG_MULTILEVEL_NO_SHARED) */
@@ -1,157 +0,0 @@
1
- /*
2
- * Copyright (c) The mldsa-native project authors
3
- * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT
4
- */
5
-
6
- /* References
7
- * ==========
8
- *
9
- * - [REF_AVX2]
10
- * CRYSTALS-Dilithium optimized AVX2 implementation
11
- * Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé
12
- * https://github.com/pq-crystals/dilithium/tree/master/avx2
13
- */
14
-
15
- /*
16
- * This file is derived from the public domain
17
- * AVX2 Dilithium implementation @[REF_AVX2].
18
- */
19
-
20
- #include "../../../common.h"
21
-
22
- #if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \
23
- !defined(MLD_CONFIG_NO_KEYPAIR_API) && \
24
- !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \
25
- (defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLDSA_ETA == 2)
26
-
27
- #include <immintrin.h>
28
- #include "arith_native_x86_64.h"
29
- #include "consts.h"
30
-
31
- #define MLD_AVX2_ETA2 2
32
-
33
- /*
34
- * Reference: In the pqcrystals implementation this function is called
35
- * rej_eta_avx and supports multiple values for ETA via preprocessor
36
- * conditionals. We move the conditionals to the frontend.
37
- */
38
- unsigned int mld_rej_uniform_eta2_avx2(
39
- int32_t *MLD_RESTRICT r,
40
- const uint8_t buf[MLD_AVX2_REJ_UNIFORM_ETA2_BUFLEN])
41
- {
42
- unsigned int ctr, pos;
43
- uint32_t good;
44
- __m256i f0, f1, f2;
45
- __m128i g0, g1;
46
- const __m256i mask = _mm256_set1_epi8(15);
47
- const __m256i eta = _mm256_set1_epi8(MLD_AVX2_ETA2);
48
- const __m256i bound = mask;
49
- /* check-magic: -6560 == 32*round(-2**10 / 5) */
50
- const __m256i v = _mm256_set1_epi32(-6560);
51
- const __m256i p = _mm256_set1_epi32(5);
52
-
53
- ctr = pos = 0;
54
- while (ctr <= MLDSA_N - 8 && pos <= MLD_AVX2_REJ_UNIFORM_ETA2_BUFLEN - 16)
55
- {
56
- f0 = _mm256_cvtepu8_epi16(_mm_loadu_si128((__m128i *)&buf[pos]));
57
- f1 = _mm256_slli_epi16(f0, 4);
58
- f0 = _mm256_or_si256(f0, f1);
59
- f0 = _mm256_and_si256(f0, mask);
60
-
61
- f1 = _mm256_sub_epi8(f0, bound);
62
- f0 = _mm256_sub_epi8(eta, f0);
63
- good = (uint32_t)_mm256_movemask_epi8(f1);
64
-
65
- g0 = _mm256_castsi256_si128(f0);
66
- g1 = _mm_loadl_epi64((__m128i *)&mld_rej_uniform_table[good & 0xFF]);
67
- g1 = _mm_shuffle_epi8(g0, g1);
68
- f1 = _mm256_cvtepi8_epi32(g1);
69
- f2 = _mm256_mulhrs_epi16(f1, v);
70
- f2 = _mm256_mullo_epi16(f2, p);
71
- f1 = _mm256_add_epi32(f1, f2);
72
- _mm256_storeu_si256((__m256i *)&r[ctr], f1);
73
- ctr += (unsigned)_mm_popcnt_u32(good & 0xFF);
74
- good >>= 8;
75
- pos += 4;
76
-
77
- if (ctr > MLDSA_N - 8)
78
- {
79
- break;
80
- }
81
- g0 = _mm_bsrli_si128(g0, 8);
82
- g1 = _mm_loadl_epi64((__m128i *)&mld_rej_uniform_table[good & 0xFF]);
83
- g1 = _mm_shuffle_epi8(g0, g1);
84
- f1 = _mm256_cvtepi8_epi32(g1);
85
- f2 = _mm256_mulhrs_epi16(f1, v);
86
- f2 = _mm256_mullo_epi16(f2, p);
87
- f1 = _mm256_add_epi32(f1, f2);
88
- _mm256_storeu_si256((__m256i *)&r[ctr], f1);
89
- ctr += (unsigned)_mm_popcnt_u32(good & 0xFF);
90
- good >>= 8;
91
- pos += 4;
92
-
93
- if (ctr > MLDSA_N - 8)
94
- {
95
- break;
96
- }
97
- g0 = _mm256_extracti128_si256(f0, 1);
98
- g1 = _mm_loadl_epi64((__m128i *)&mld_rej_uniform_table[good & 0xFF]);
99
- g1 = _mm_shuffle_epi8(g0, g1);
100
- f1 = _mm256_cvtepi8_epi32(g1);
101
- f2 = _mm256_mulhrs_epi16(f1, v);
102
- f2 = _mm256_mullo_epi16(f2, p);
103
- f1 = _mm256_add_epi32(f1, f2);
104
- _mm256_storeu_si256((__m256i *)&r[ctr], f1);
105
- ctr += (unsigned)_mm_popcnt_u32(good & 0xFF);
106
- good >>= 8;
107
- pos += 4;
108
-
109
- if (ctr > MLDSA_N - 8)
110
- {
111
- break;
112
- }
113
- g0 = _mm_bsrli_si128(g0, 8);
114
- g1 = _mm_loadl_epi64((__m128i *)&mld_rej_uniform_table[good]);
115
- g1 = _mm_shuffle_epi8(g0, g1);
116
- f1 = _mm256_cvtepi8_epi32(g1);
117
- f2 = _mm256_mulhrs_epi16(f1, v);
118
- f2 = _mm256_mullo_epi16(f2, p);
119
- f1 = _mm256_add_epi32(f1, f2);
120
- _mm256_storeu_si256((__m256i *)&r[ctr], f1);
121
- ctr += (unsigned)_mm_popcnt_u32(good);
122
- pos += 4;
123
- }
124
-
125
- while (ctr < MLDSA_N && pos < MLD_AVX2_REJ_UNIFORM_ETA2_BUFLEN)
126
- {
127
- uint32_t t0 = buf[pos] & 0x0F;
128
- uint32_t t1 = buf[pos++] >> 4;
129
-
130
- if (t0 < 15)
131
- {
132
- t0 = t0 - (205 * t0 >> 10) * 5;
133
- r[ctr++] = (int32_t)(2 - t0);
134
- }
135
- if (t1 < 15 && ctr < MLDSA_N)
136
- {
137
- t1 = t1 - (205 * t1 >> 10) * 5;
138
- r[ctr++] = (int32_t)(2 - t1);
139
- }
140
- }
141
-
142
- return ctr;
143
- }
144
-
145
- #else /* MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_NO_KEYPAIR_API && \
146
- !MLD_CONFIG_MULTILEVEL_NO_SHARED && \
147
- (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLDSA_ETA == 2) */
148
-
149
- MLD_EMPTY_CU(avx2_rej_uniform_eta2)
150
-
151
- #endif /* !(MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_NO_KEYPAIR_API && \
152
- !MLD_CONFIG_MULTILEVEL_NO_SHARED && \
153
- (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLDSA_ETA == 2)) */
154
-
155
- /* To facilitate single-compilation-unit (SCU) builds, undefine all macros.
156
- * Don't modify by hand -- this is auto-generated by scripts/autogen. */
157
- #undef MLD_AVX2_ETA2