pq_crypto 0.6.5 → 0.6.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (157) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +233 -0
  3. data/README.md +16 -3
  4. data/SECURITY.md +46 -0
  5. data/ext/pqcrypto/extconf.rb +8 -1
  6. data/ext/pqcrypto/pq_externalmu.c +35 -0
  7. data/ext/pqcrypto/pqcrypto_native_api.h +91 -75
  8. data/ext/pqcrypto/pqcrypto_ruby_secure.c +97 -9
  9. data/ext/pqcrypto/pqcrypto_secure.c +95 -33
  10. data/ext/pqcrypto/pqcrypto_secure.h +66 -48
  11. data/ext/pqcrypto/pqcrypto_version.h +1 -1
  12. data/ext/pqcrypto/vendor/.vendored +7 -7
  13. data/ext/pqcrypto/vendor/mldsa-native/BUILDING.md +5 -2
  14. data/ext/pqcrypto/vendor/mldsa-native/LICENSE +21 -2
  15. data/ext/pqcrypto/vendor/mldsa-native/README.md +20 -7
  16. data/ext/pqcrypto/vendor/mldsa-native/RELEASE.md +160 -0
  17. data/ext/pqcrypto/vendor/mldsa-native/SECURITY.md +1 -1
  18. data/ext/pqcrypto/vendor/mldsa-native/mldsa/README.md +2 -2
  19. data/ext/pqcrypto/vendor/mldsa-native/mldsa/mldsa_native.c +85 -59
  20. data/ext/pqcrypto/vendor/mldsa-native/mldsa/mldsa_native.h +292 -348
  21. data/ext/pqcrypto/vendor/mldsa-native/mldsa/mldsa_native_asm.S +122 -76
  22. data/ext/pqcrypto/vendor/mldsa-native/mldsa/mldsa_native_config.h +184 -86
  23. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/cbmc.h +49 -4
  24. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/common.h +49 -81
  25. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/context.h +152 -0
  26. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/ct.h +25 -12
  27. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/debug.c +2 -0
  28. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/debug.h +2 -0
  29. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/fips202x4.c +2 -2
  30. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/keccakf1600.c +9 -11
  31. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/auto.h +19 -11
  32. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x1_scalar_aarch64_asm.S +6 -4
  33. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x1_v84a_aarch64_asm.S +7 -4
  34. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x2_v84a_aarch64_asm.S +7 -4
  35. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm.S +12 -9
  36. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm.S +12 -9
  37. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/x1_scalar.h +1 -1
  38. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/x1_v84a.h +3 -2
  39. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/x2_v84a.h +3 -2
  40. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/x4_v8a_scalar.h +6 -1
  41. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/x4_v8a_v84a_scalar.h +3 -2
  42. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/api.h +11 -11
  43. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/armv81m/mve.h +9 -22
  44. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/armv81m/src/keccak_f1600_x4_mve.S +8 -5
  45. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/armv81m/src/keccak_f1600_x4_mve.c +1 -0
  46. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/armv81m/src/{state_extract_bytes_x4_mve.S → keccak_f1600_x4_state_extract_bytes_mve.S} +14 -14
  47. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/armv81m/src/{state_xor_bytes_x4_mve.S → keccak_f1600_x4_state_xor_bytes_mve.S} +12 -12
  48. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/auto.h +5 -4
  49. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/x86_64/keccak_f1600_x4_avx2.h +2 -2
  50. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/x86_64/src/fips202_native_x86_64.h +1 -0
  51. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/x86_64/src/keccak_f1600_x4_avx2_asm.S +36 -2
  52. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/meta.h +62 -4
  53. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/arith_native_aarch64.h +87 -54
  54. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{intt_aarch64_asm.S → mldsa_intt_aarch64_asm.S} +39 -6
  55. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{ntt_aarch64_asm.S → mldsa_ntt_aarch64_asm.S} +39 -6
  56. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{pointwise_montgomery_aarch64_asm.S → mldsa_pointwise_montgomery_aarch64_asm.S} +25 -3
  57. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{poly_caddq_aarch64_asm.S → mldsa_poly_caddq_aarch64_asm.S} +19 -3
  58. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{poly_chknorm_aarch64_asm.S → mldsa_poly_chknorm_aarch64_asm.S} +24 -3
  59. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{poly_decompose_32_aarch64_asm.S → mldsa_poly_decompose_32_aarch64_asm.S} +25 -3
  60. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{poly_decompose_88_aarch64_asm.S → mldsa_poly_decompose_88_aarch64_asm.S} +25 -3
  61. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{poly_use_hint_32_aarch64_asm.S → mldsa_poly_use_hint_32_aarch64_asm.S} +25 -3
  62. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{poly_use_hint_88_aarch64_asm.S → mldsa_poly_use_hint_88_aarch64_asm.S} +25 -3
  63. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{mld_polyvecl_pointwise_acc_montgomery_l4_aarch64_asm.S → mldsa_polyvecl_pointwise_acc_montgomery_l4_aarch64_asm.S} +31 -3
  64. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{mld_polyvecl_pointwise_acc_montgomery_l5_aarch64_asm.S → mldsa_polyvecl_pointwise_acc_montgomery_l5_aarch64_asm.S} +31 -3
  65. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{mld_polyvecl_pointwise_acc_montgomery_l7_aarch64_asm.S → mldsa_polyvecl_pointwise_acc_montgomery_l7_aarch64_asm.S} +31 -3
  66. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{polyz_unpack_17_aarch64_asm.S → mldsa_polyz_unpack_17_aarch64_asm.S} +31 -3
  67. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{polyz_unpack_19_aarch64_asm.S → mldsa_polyz_unpack_19_aarch64_asm.S} +31 -3
  68. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{rej_uniform_aarch64_asm.S → mldsa_rej_uniform_aarch64_asm.S} +48 -15
  69. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{rej_uniform_eta2_aarch64_asm.S → mldsa_rej_uniform_eta2_aarch64_asm.S} +42 -9
  70. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{rej_uniform_eta4_aarch64_asm.S → mldsa_rej_uniform_eta4_aarch64_asm.S} +42 -9
  71. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/api.h +11 -3
  72. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/meta.h +3 -2
  73. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/meta.h +28 -28
  74. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/arith_native_x86_64.h +171 -49
  75. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{intt_avx2_asm.S → mldsa_intt_avx2_asm.S} +23 -1
  76. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{ntt_avx2_asm.S → mldsa_ntt_avx2_asm.S} +23 -1
  77. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{nttunpack_avx2_asm.S → mldsa_nttunpack_avx2_asm.S} +17 -1
  78. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{pointwise_acc_l4_avx2_asm.S → mldsa_pointwise_acc_l4_avx2_asm.S} +37 -3
  79. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{pointwise_acc_l5_avx2_asm.S → mldsa_pointwise_acc_l5_avx2_asm.S} +37 -3
  80. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{pointwise_acc_l7_avx2_asm.S → mldsa_pointwise_acc_l7_avx2_asm.S} +37 -3
  81. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{pointwise_avx2_asm.S → mldsa_pointwise_avx2_asm.S} +31 -3
  82. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{poly_caddq_avx2_asm.S → mldsa_poly_caddq_avx2_asm.S} +18 -9
  83. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_poly_chknorm_avx2_asm.S +176 -0
  84. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_poly_decompose_32_avx2_asm.S +490 -0
  85. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_poly_decompose_88_avx2_asm.S +489 -0
  86. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_poly_use_hint_32_avx2_asm.S +123 -0
  87. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_poly_use_hint_88_avx2_asm.S +125 -0
  88. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_polyz_unpack_17_avx2_asm.S +355 -0
  89. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_polyz_unpack_19_avx2_asm.S +355 -0
  90. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_rej_uniform_avx2_asm.S +132 -0
  91. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_rej_uniform_eta2_avx2_asm.S +205 -0
  92. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_rej_uniform_eta4_avx2_asm.S +176 -0
  93. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/packing.c +27 -36
  94. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/packing.h +42 -8
  95. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/params.h +93 -17
  96. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/poly.c +74 -15
  97. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/poly.h +97 -11
  98. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/poly_kl.c +7 -38
  99. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/poly_kl.h +49 -7
  100. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/polyvec.c +16 -17
  101. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/polyvec.h +26 -9
  102. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/polyvec_lazy.c +3 -0
  103. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/polyvec_lazy.h +18 -19
  104. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/reduce.h +15 -3
  105. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/rounding.h +28 -6
  106. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/sign.c +311 -246
  107. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/sign.h +245 -240
  108. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/sys.h +64 -5
  109. data/ext/pqcrypto/vendor/mlkem-native/BUILDING.md +5 -2
  110. data/ext/pqcrypto/vendor/mlkem-native/LICENSE +21 -3
  111. data/ext/pqcrypto/vendor/mlkem-native/README.md +2 -2
  112. data/ext/pqcrypto/vendor/mlkem-native/RELEASE.md +113 -0
  113. data/ext/pqcrypto/vendor/mlkem-native/mlkem/README.md +2 -2
  114. data/ext/pqcrypto/vendor/mlkem-native/mlkem/mlkem_native.c +17 -27
  115. data/ext/pqcrypto/vendor/mlkem-native/mlkem/mlkem_native.h +68 -151
  116. data/ext/pqcrypto/vendor/mlkem-native/mlkem/mlkem_native_asm.S +17 -27
  117. data/ext/pqcrypto/vendor/mlkem-native/mlkem/mlkem_native_config.h +46 -44
  118. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/cbmc.h +25 -0
  119. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/common.h +37 -6
  120. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/context.h +9 -0
  121. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/fips202.h +2 -2
  122. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/keccakf1600.c +8 -8
  123. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x1_scalar.h +1 -1
  124. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x1_v84a.h +3 -3
  125. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x2_v84a.h +3 -3
  126. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x4_v8a_scalar.h +2 -2
  127. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x4_v8a_v84a_scalar.h +3 -3
  128. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/api.h +11 -11
  129. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/armv81m/mve.h +3 -3
  130. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/x86_64/keccak_f1600_x4_avx2.h +2 -2
  131. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/indcpa.h +14 -11
  132. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/kem.c +28 -11
  133. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/kem.h +39 -14
  134. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/meta.h +10 -10
  135. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/meta.h +20 -20
  136. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/sys.h +5 -5
  137. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/verify.h +11 -10
  138. data/lib/pq_crypto/internal.rb +10 -0
  139. data/lib/pq_crypto/kem.rb +47 -5
  140. data/lib/pq_crypto/key.rb +14 -8
  141. data/lib/pq_crypto/pkcs8.rb +13 -5
  142. data/lib/pq_crypto/signature.rb +34 -9
  143. data/lib/pq_crypto/spki.rb +4 -2
  144. data/lib/pq_crypto/version.rb +1 -1
  145. data/lib/pq_crypto.rb +9 -0
  146. data/script/vendor_libs.rb +6 -6
  147. metadata +40 -38
  148. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/poly_chknorm_avx2.c +0 -52
  149. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/poly_decompose_32_avx2.c +0 -157
  150. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/poly_decompose_88_avx2.c +0 -157
  151. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/poly_use_hint_32_avx2.c +0 -103
  152. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/poly_use_hint_88_avx2.c +0 -105
  153. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/polyz_unpack_17_avx2.c +0 -94
  154. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/polyz_unpack_19_avx2.c +0 -96
  155. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/rej_uniform_avx2.c +0 -126
  156. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/rej_uniform_eta2_avx2.c +0 -157
  157. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/rej_uniform_eta4_avx2.c +0 -141
@@ -1,157 +0,0 @@
1
- /*
2
- * Copyright (c) The mldsa-native project authors
3
- * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT
4
- */
5
-
6
- /* References
7
- * ==========
8
- *
9
- * - [REF_AVX2]
10
- * CRYSTALS-Dilithium optimized AVX2 implementation
11
- * Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé
12
- * https://github.com/pq-crystals/dilithium/tree/master/avx2
13
- */
14
-
15
- /*
16
- * This file is derived from the public domain
17
- * AVX2 Dilithium implementation @[REF_AVX2].
18
- *
19
- * The algorithm for Decompose(r) (more specifically the handling for the
20
- * wrap-around cases) are modified. See the "Reference" section in the comments
21
- * below for a more detailed comparison.
22
- */
23
-
24
- #include "../../../common.h"
25
-
26
- #if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \
27
- !defined(MLD_CONFIG_NO_SIGN_API) && \
28
- !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \
29
- (defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \
30
- (MLD_CONFIG_PARAMETER_SET == 65 || MLD_CONFIG_PARAMETER_SET == 87))
31
-
32
- #include <immintrin.h>
33
- #include "arith_native_x86_64.h"
34
- #include "consts.h"
35
-
36
- /*
37
- * Reference: The reference implementation has the input polynomial as a
38
- * separate argument that may be aliased with either of the outputs.
39
- * Removing the aliasing eases CBMC proofs.
40
- */
41
- void mld_poly_decompose_32_avx2(int32_t *a1, int32_t *a0)
42
- {
43
- unsigned int i;
44
- __m256i f, f0, f1, t;
45
- const __m256i q_bound = _mm256_set1_epi32(31 * ((MLDSA_Q - 1) / 32));
46
- /* check-magic: 1025 == floor(2**22 / 4092) */
47
- const __m256i v = _mm256_set1_epi32(1025);
48
- const __m256i alpha = _mm256_set1_epi32(2 * ((MLDSA_Q - 1) / 32));
49
- const __m256i off = _mm256_set1_epi32(127);
50
- const __m256i shift = _mm256_set1_epi32(512);
51
-
52
- for (i = 0; i < MLDSA_N / 8; i++)
53
- {
54
- f = _mm256_load_si256((__m256i *)&a0[8 * i]);
55
-
56
- /* check-magic: 4092 == intdiv(2 * intdiv(MLDSA_Q - 1, 32), 128) */
57
- /*
58
- * Compute f1 = round-(f / (2*GAMMA2)) as round-(f / (128B)) =
59
- * round-(ceil(f / 128) / B) where B = 2*GAMMA2 / 128 = 4092. See
60
- * mld_decompose() in mldsa/src/rounding.h for more details.
61
- *
62
- * range: 0 <= f <= Q-1 = 32*GAMMA2 = 16*128*B
63
- */
64
-
65
- /* Compute f1' = ceil(f / 128) as floor((f + 127) / 2^7) */
66
- f1 = _mm256_add_epi32(f, off);
67
- f1 = _mm256_srli_epi32(f1, 7);
68
- /*
69
- * range: 0 <= f1' <= (Q-1)/128 = 16B
70
- *
71
- * Also, f1' <= (Q-1)/128 = 2^16 - 2^6 < 2^16 ensures that the odd-index
72
- * 16-bit lanes are all 0, so no bits will be dropped in the input of the
73
- * _mm256_mulhi_epu16() below.
74
- */
75
-
76
- /*
77
- * Compute f1 = round-(f1' / B) ≈ round(f1' * 1025 / 2^22). This is exact
78
- * for 0 <= f1' < 2^16. See mld_decompose() in mldsa/src/rounding.h for the
79
- * proof, and proofs/isabelle/compress for a formalization of the argument.
80
- *
81
- * round(f1' * 1025 / 2^22) is in turn computed in 2 steps as
82
- * round(floor(f1' * 1025 / 2^16) / 2^6). The mulhi computes f1'' =
83
- * floor(f1' * 1025 / 2^16). As for the next step f1 = round(f1'' / 2^6),
84
- * because AVX2 doesn't have rounding right-shift (e.g. urshr in Neon), we
85
- * simulate it using mulhrs with a power of 2, in this case mulhrs(f1'',
86
- * 2^9) = round(f1'' * 2^9 / 2^15). (Note that the denominator is 2^15,
87
- * not 2^16 as in mulhi.)
88
- */
89
- f1 = _mm256_mulhi_epu16(f1, v);
90
- /*
91
- * range: 0 <= f1'' = floor(f1' * 1025 / 2^16)
92
- * <= f1' * 1025 / 2^16
93
- * < 2^16 * 1025 / 2^16 = 1025
94
- *
95
- * Because 0 <= f1'' < 2^15, the multiplication in mulhrs is unsigned, that
96
- * is, no erroneous sign-extension occurs.
97
- */
98
- f1 = _mm256_mulhrs_epi16(f1, shift);
99
- /*
100
- * range: 0 <= f1 = round-(f1' / B) <= round-(16B / B) = 16
101
- *
102
- * Note that the odd-index 16-bit lanes are still all 0 right now, so
103
- * reinterpreting f1 as 8 lanes of int32_t (as done in the following) does
104
- * not affect its value.
105
- */
106
-
107
- /*
108
- * If f1 = 16, i.e. f > 31*GAMMA2, proceed as if f' = f - Q was given
109
- * instead. (For f = 31*GAMMA2 + 1 thus f' = -GAMMA2, we still round it to 0
110
- * like other "wrapped around" cases.)
111
- *
112
- * Reference: They handle wrap-around in a somewhat convoluted way. Most
113
- * notably, they compute remainder f0 with quotient f1 that's
114
- * already wrapped around, so is off by q (instead of by 1) from
115
- * what it should be ultimately. They detect the need for
116
- * correction by checking if f0 is abnormally large.
117
- *
118
- * Our approach is closer to Algorithm 36 in the specification,
119
- * in that we compute f0 normally and correct f1, f0 in the way
120
- * they prescribed. The only real difference is that we check for
121
- * wrap-around by examining f directly, instead of some other
122
- * intermediates computed from it.
123
- */
124
-
125
- /* Check for wrap-around */
126
- t = _mm256_cmpgt_epi32(f, q_bound);
127
-
128
- /* Compute remainder f0 */
129
- f0 = _mm256_mullo_epi32(f1, alpha);
130
- f0 = _mm256_sub_epi32(f, f0);
131
- /*
132
- * range: -GAMMA2 < f0 <= GAMMA2
133
- *
134
- * This holds since f1 = round-(f / (2*GAMMA2)) was computed exactly.
135
- */
136
-
137
- /* If wrap-around is required, set f1 = 0 and f0 -= 1 */
138
- f1 = _mm256_andnot_si256(t, f1);
139
- f0 = _mm256_add_epi32(f0, t);
140
- /* range: 0 <= f1 <= 15, -GAMMA2 <= f0 <= GAMMA2 */
141
-
142
- _mm256_store_si256((__m256i *)&a1[8 * i], f1);
143
- _mm256_store_si256((__m256i *)&a0[8 * i], f0);
144
- }
145
- }
146
-
147
- #else /* MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_NO_SIGN_API && \
148
- !MLD_CONFIG_MULTILEVEL_NO_SHARED && \
149
- (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 65 \
150
- || MLD_CONFIG_PARAMETER_SET == 87) */
151
-
152
- MLD_EMPTY_CU(avx2_poly_decompose_32)
153
-
154
- #endif /* !(MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_NO_SIGN_API && \
155
- !MLD_CONFIG_MULTILEVEL_NO_SHARED && \
156
- (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 65 \
157
- || MLD_CONFIG_PARAMETER_SET == 87)) */
@@ -1,157 +0,0 @@
1
- /*
2
- * Copyright (c) The mldsa-native project authors
3
- * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT
4
- */
5
-
6
- /* References
7
- * ==========
8
- *
9
- * - [REF_AVX2]
10
- * CRYSTALS-Dilithium optimized AVX2 implementation
11
- * Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé
12
- * https://github.com/pq-crystals/dilithium/tree/master/avx2
13
- */
14
-
15
- /*
16
- * This file is derived from the public domain
17
- * AVX2 Dilithium implementation @[REF_AVX2].
18
- *
19
- * The algorithm for Decompose(r) (more specifically the handling for the
20
- * wrap-around cases) are modified. See the "Reference" section in the comments
21
- * below for a more detailed comparison.
22
- */
23
-
24
- #include "../../../common.h"
25
-
26
- #if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \
27
- !defined(MLD_CONFIG_NO_SIGN_API) && \
28
- !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \
29
- (defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \
30
- MLD_CONFIG_PARAMETER_SET == 44)
31
-
32
- #include <immintrin.h>
33
- #include "arith_native_x86_64.h"
34
- #include "consts.h"
35
-
36
- /*
37
- * Reference: The reference implementation has the input polynomial as a
38
- * separate argument that may be aliased with either of the outputs.
39
- * Removing the aliasing eases CBMC proofs.
40
- */
41
-
42
- void mld_poly_decompose_88_avx2(int32_t *a1, int32_t *a0)
43
- {
44
- unsigned int i;
45
- __m256i f, f0, f1, t;
46
- const __m256i q_bound = _mm256_set1_epi32(87 * ((MLDSA_Q - 1) / 88));
47
- /* check-magic: 11275 == floor(2**24 / 1488) */
48
- const __m256i v = _mm256_set1_epi32(11275);
49
- const __m256i alpha = _mm256_set1_epi32(2 * ((MLDSA_Q - 1) / 88));
50
- const __m256i off = _mm256_set1_epi32(127);
51
- const __m256i shift = _mm256_set1_epi32(128);
52
-
53
- for (i = 0; i < MLDSA_N / 8; i++)
54
- {
55
- f = _mm256_load_si256((__m256i *)&a0[8 * i]);
56
-
57
- /* check-magic: 1488 == intdiv(2 * intdiv(MLDSA_Q - 1, 88), 128) */
58
- /*
59
- * Compute f1 = round-(f / (2*GAMMA2)) as round-(f / (128B)) =
60
- * round-(ceil(f / 128) / B) where B = 2*GAMMA2 / 128 = 1488. See
61
- * mld_decompose() in mldsa/src/rounding.h for more details.
62
- *
63
- * range: 0 <= f <= Q-1 = 88*GAMMA2 = 44*128*B
64
- */
65
-
66
- /* Compute f1' = ceil(f / 128) as floor((f + 127) / 2^7) */
67
- f1 = _mm256_add_epi32(f, off);
68
- f1 = _mm256_srli_epi32(f1, 7);
69
- /*
70
- * range: 0 <= f1' <= (Q-1)/128 = 44B
71
- *
72
- * Also, f1' <= (Q-1)/128 = 2^16 - 2^6 < 2^16 ensures that the odd-index
73
- * 16-bit lanes are all 0, so no bits will be dropped in the input of the
74
- * _mm256_mulhi_epu16() below.
75
- */
76
-
77
- /*
78
- * Compute f1 = round-(f1' / B) ≈ round(f1' * 11275 / 2^24). This is exact
79
- * for 0 <= f1' < 2^16. See mld_decompose() in mldsa/src/rounding.h for the
80
- * proof, and proofs/isabelle/compress for a formalization of the argument.
81
- *
82
- * round(f1' * 11275 / 2^24) is in turn computed in 2 steps as
83
- * round(floor(f1' * 11275 / 2^16) / 2^8). The mulhi computes f1'' =
84
- * floor(f1' * 11275 / 2^16). As for the next step f1 = round(f1'' / 2^8),
85
- * because AVX2 doesn't have rounding right-shift (e.g. urshr in Neon), we
86
- * simulate it using mulhrs with a power of 2, in this case mulhrs(f1'',
87
- * 2^7) = round(f1'' * 2^7 / 2^15). (Note that the denominator is 2^15,
88
- * not 2^16 as in mulhi.)
89
- */
90
- f1 = _mm256_mulhi_epu16(f1, v);
91
- /*
92
- * range: 0 <= f1'' = floor(f1' * 11275 / 2^16)
93
- * <= f1' * 11275 / 2^16
94
- * < 2^16 * 11275 / 2^16 = 11275
95
- *
96
- * Because 0 <= f1'' < 2^15, the multiplication in mulhrs is unsigned, that
97
- * is, no erroneous sign-extension occurs.
98
- */
99
- f1 = _mm256_mulhrs_epi16(f1, shift);
100
- /*
101
- * range: 0 <= f1 = round-(f1' / B) <= round-(44B / B) = 44
102
- *
103
- * Note that the odd-index 16-bit lanes are still all 0 right now, so
104
- * reinterpreting f1 as 8 lanes of int32_t (as done in the following) does
105
- * not affect its value.
106
- */
107
-
108
- /*
109
- * If f1 = 44, i.e. f > 87*GAMMA2, proceed as if f' = f - Q was given
110
- * instead. (For f = 87*GAMMA2 + 1 thus f' = -GAMMA2, we still round it to 0
111
- * like other "wrapped around" cases.)
112
- *
113
- * Reference: They handle wrap-around in a somewhat convoluted way. Most
114
- * notably, they compute remainder f0 with quotient f1 that's
115
- * already wrapped around, so is off by q (instead of by 1) from
116
- * what it should be ultimately. They detect the need for
117
- * correction by checking if f0 is abnormally large.
118
- *
119
- * Our approach is closer to Algorithm 36 in the specification,
120
- * in that we compute f0 normally and correct f1, f0 in the way
121
- * they prescribed. The only real difference is that we check for
122
- * wrap-around by examining f directly, instead of some other
123
- * intermediates computed from it.
124
- */
125
-
126
- /* Check for wrap-around */
127
- t = _mm256_cmpgt_epi32(f, q_bound);
128
-
129
- /* Compute remainder f0 */
130
- f0 = _mm256_mullo_epi32(f1, alpha);
131
- f0 = _mm256_sub_epi32(f, f0);
132
- /*
133
- * range: -GAMMA2 < f0 <= GAMMA2
134
- *
135
- * This holds since f1 = round-(f / (2*GAMMA2)) was computed exactly.
136
- */
137
-
138
- /* If wrap-around is required, set f1 = 0 and f0 -= 1 */
139
- f1 = _mm256_andnot_si256(t, f1);
140
- f0 = _mm256_add_epi32(f0, t);
141
- /* range: 0 <= f1 <= 43, -GAMMA2 <= f0 <= GAMMA2 */
142
-
143
- _mm256_store_si256((__m256i *)&a1[8 * i], f1);
144
- _mm256_store_si256((__m256i *)&a0[8 * i], f0);
145
- }
146
- }
147
- #else /* MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_NO_SIGN_API && \
148
- !MLD_CONFIG_MULTILEVEL_NO_SHARED && \
149
- (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 44) \
150
- */
151
-
152
- MLD_EMPTY_CU(avx2_poly_decompose_88)
153
-
154
- #endif /* !(MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_NO_SIGN_API && \
155
- !MLD_CONFIG_MULTILEVEL_NO_SHARED && \
156
- (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == \
157
- 44)) */
@@ -1,103 +0,0 @@
1
- /*
2
- * Copyright (c) The mldsa-native project authors
3
- * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT
4
- */
5
-
6
- /* References
7
- * ==========
8
- *
9
- * - [REF_AVX2]
10
- * CRYSTALS-Dilithium optimized AVX2 implementation
11
- * Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé
12
- * https://github.com/pq-crystals/dilithium/tree/master/avx2
13
- */
14
-
15
- /*
16
- * This file is derived from the public domain
17
- * AVX2 Dilithium implementation @[REF_AVX2].
18
- */
19
-
20
- #include "../../../common.h"
21
-
22
- #if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \
23
- !defined(MLD_CONFIG_NO_VERIFY_API) && \
24
- !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \
25
- (defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \
26
- (MLD_CONFIG_PARAMETER_SET == 65 || MLD_CONFIG_PARAMETER_SET == 87))
27
-
28
- #include <immintrin.h>
29
- #include "arith_native_x86_64.h"
30
- #include "consts.h"
31
-
32
- #define MLD_MM256_BLENDV_EPI32(a, b, mask) \
33
- _mm256_castps_si256(_mm256_blendv_ps(_mm256_castsi256_ps(a), \
34
- _mm256_castsi256_ps(b), \
35
- _mm256_castsi256_ps(mask)))
36
-
37
- void mld_poly_use_hint_32_avx2(int32_t *a, const int32_t *hint)
38
- {
39
- unsigned int i;
40
- __m256i f, f0, f1, h, t;
41
- const __m256i q_bound = _mm256_set1_epi32(31 * ((MLDSA_Q - 1) / 32));
42
- /* check-magic: 1025 == floor(2**22 / 4092) */
43
- const __m256i v = _mm256_set1_epi32(1025);
44
- const __m256i alpha = _mm256_set1_epi32(2 * ((MLDSA_Q - 1) / 32));
45
- const __m256i off = _mm256_set1_epi32(127);
46
- const __m256i shift = _mm256_set1_epi32(512);
47
- const __m256i mask = _mm256_set1_epi32(15);
48
- const __m256i zero = _mm256_setzero_si256();
49
-
50
- for (i = 0; i < MLDSA_N / 8; i++)
51
- {
52
- f = _mm256_load_si256((const __m256i *)&a[8 * i]);
53
- h = _mm256_load_si256((const __m256i *)&hint[8 * i]);
54
-
55
- /* Reference:
56
- * - @[REF_AVX2] calls poly_decompose to compute all a1, a0 before the loop.
57
- * - Our implementation of decompose() is slightly different from that in
58
- * @[REF_AVX2]. See poly_decompose_32_avx2.c for more information.
59
- */
60
- /* f1, f2 = decompose(f) */
61
- f1 = _mm256_add_epi32(f, off);
62
- f1 = _mm256_srli_epi32(f1, 7);
63
- f1 = _mm256_mulhi_epu16(f1, v);
64
- f1 = _mm256_mulhrs_epi16(f1, shift);
65
- t = _mm256_cmpgt_epi32(f, q_bound);
66
- f0 = _mm256_mullo_epi32(f1, alpha);
67
- f0 = _mm256_sub_epi32(f, f0);
68
- f1 = _mm256_andnot_si256(t, f1);
69
- f0 = _mm256_add_epi32(f0, t);
70
-
71
- /* Reference: The reference avx2 implementation checks a0 >= 0, which is
72
- * different from the specification and the reference C implementation. We
73
- * follow the specification and check a0 > 0.
74
- */
75
- /* t = (f0 > 0) ? h : -h */
76
- f0 = _mm256_cmpgt_epi32(f0, zero);
77
- t = MLD_MM256_BLENDV_EPI32(h, zero, f0);
78
- t = _mm256_slli_epi32(t, 1);
79
- h = _mm256_sub_epi32(h, t);
80
-
81
- /* f1 = (f1 + t) % 16 */
82
- f1 = _mm256_add_epi32(f1, h);
83
- f1 = _mm256_and_si256(f1, mask);
84
-
85
- _mm256_store_si256((__m256i *)&a[8 * i], f1);
86
- }
87
- }
88
-
89
- #else /* MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_NO_VERIFY_API && \
90
- !MLD_CONFIG_MULTILEVEL_NO_SHARED && \
91
- (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 65 \
92
- || MLD_CONFIG_PARAMETER_SET == 87) */
93
-
94
- MLD_EMPTY_CU(avx2_poly_use_hint_32)
95
-
96
- #endif /* !(MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_NO_VERIFY_API && \
97
- !MLD_CONFIG_MULTILEVEL_NO_SHARED && \
98
- (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 65 \
99
- || MLD_CONFIG_PARAMETER_SET == 87)) */
100
-
101
- /* To facilitate single-compilation-unit (SCU) builds, undefine all macros.
102
- * Don't modify by hand -- this is auto-generated by scripts/autogen. */
103
- #undef MLD_MM256_BLENDV_EPI32
@@ -1,105 +0,0 @@
1
- /*
2
- * Copyright (c) The mldsa-native project authors
3
- * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT
4
- */
5
-
6
- /* References
7
- * ==========
8
- *
9
- * - [REF_AVX2]
10
- * CRYSTALS-Dilithium optimized AVX2 implementation
11
- * Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé
12
- * https://github.com/pq-crystals/dilithium/tree/master/avx2
13
- */
14
-
15
- /*
16
- * This file is derived from the public domain
17
- * AVX2 Dilithium implementation @[REF_AVX2].
18
- */
19
-
20
- #include "../../../common.h"
21
-
22
- #if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \
23
- !defined(MLD_CONFIG_NO_VERIFY_API) && \
24
- !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \
25
- (defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \
26
- MLD_CONFIG_PARAMETER_SET == 44)
27
-
28
- #include <immintrin.h>
29
- #include "arith_native_x86_64.h"
30
- #include "consts.h"
31
-
32
- #define MLD_MM256_BLENDV_EPI32(a, b, mask) \
33
- _mm256_castps_si256(_mm256_blendv_ps(_mm256_castsi256_ps(a), \
34
- _mm256_castsi256_ps(b), \
35
- _mm256_castsi256_ps(mask)))
36
-
37
- void mld_poly_use_hint_88_avx2(int32_t *a, const int32_t *hint)
38
- {
39
- unsigned int i;
40
- __m256i f, f0, f1, h, t;
41
- const __m256i q_bound = _mm256_set1_epi32(87 * ((MLDSA_Q - 1) / 88));
42
- /* check-magic: 11275 == floor(2**24 / 1488) */
43
- const __m256i v = _mm256_set1_epi32(11275);
44
- const __m256i alpha = _mm256_set1_epi32(2 * ((MLDSA_Q - 1) / 88));
45
- const __m256i off = _mm256_set1_epi32(127);
46
- const __m256i shift = _mm256_set1_epi32(128);
47
- const __m256i max = _mm256_set1_epi32(43);
48
- const __m256i zero = _mm256_setzero_si256();
49
-
50
- for (i = 0; i < MLDSA_N / 8; i++)
51
- {
52
- f = _mm256_load_si256((const __m256i *)&a[8 * i]);
53
- h = _mm256_load_si256((const __m256i *)&hint[8 * i]);
54
-
55
- /* Reference:
56
- * - @[REF_AVX2] calls poly_decompose to compute all a1, a0 before the loop.
57
- * - Our implementation of decompose() is slightly different from that in
58
- * @[REF_AVX2]. See poly_decompose_88_avx2.c for more information.
59
- */
60
- /* f1, f2 = decompose(f) */
61
- f1 = _mm256_add_epi32(f, off);
62
- f1 = _mm256_srli_epi32(f1, 7);
63
- f1 = _mm256_mulhi_epu16(f1, v);
64
- f1 = _mm256_mulhrs_epi16(f1, shift);
65
- t = _mm256_cmpgt_epi32(f, q_bound);
66
- f0 = _mm256_mullo_epi32(f1, alpha);
67
- f0 = _mm256_sub_epi32(f, f0);
68
- f1 = _mm256_andnot_si256(t, f1);
69
- f0 = _mm256_add_epi32(f0, t);
70
-
71
- /* Reference: The reference avx2 implementation checks a0 >= 0, which is
72
- * different from the specification and the reference C implementation. We
73
- * follow the specification and check a0 > 0.
74
- */
75
- /* t = (f0 > 0) ? h : -h */
76
- f0 = _mm256_cmpgt_epi32(f0, zero);
77
- t = MLD_MM256_BLENDV_EPI32(h, zero, f0);
78
- t = _mm256_slli_epi32(t, 1);
79
- h = _mm256_sub_epi32(h, t);
80
-
81
- /* f1 = (f1 + t) % 44 */
82
- f1 = _mm256_add_epi32(f1, h);
83
- f1 = MLD_MM256_BLENDV_EPI32(f1, max, f1);
84
- f = _mm256_cmpgt_epi32(f1, max);
85
- f1 = MLD_MM256_BLENDV_EPI32(f1, zero, f);
86
-
87
- _mm256_store_si256((__m256i *)&a[8 * i], f1);
88
- }
89
- }
90
-
91
- #else /* MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_NO_VERIFY_API && \
92
- !MLD_CONFIG_MULTILEVEL_NO_SHARED && \
93
- (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 44) \
94
- */
95
-
96
- MLD_EMPTY_CU(avx2_poly_use_hint_88)
97
-
98
- #endif /* !(MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_NO_VERIFY_API && \
99
- !MLD_CONFIG_MULTILEVEL_NO_SHARED && \
100
- (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == \
101
- 44)) */
102
-
103
- /* To facilitate single-compilation-unit (SCU) builds, undefine all macros.
104
- * Don't modify by hand -- this is auto-generated by scripts/autogen. */
105
- #undef MLD_MM256_BLENDV_EPI32
@@ -1,94 +0,0 @@
1
- /*
2
- * Copyright (c) The mldsa-native project authors
3
- * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT
4
- */
5
-
6
- /* References
7
- * ==========
8
- *
9
- * - [REF_AVX2]
10
- * CRYSTALS-Dilithium optimized AVX2 implementation
11
- * Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé
12
- * https://github.com/pq-crystals/dilithium/tree/master/avx2
13
- */
14
-
15
- /*
16
- * This file is derived from the public domain
17
- * AVX2 Dilithium implementation @[REF_AVX2].
18
- */
19
-
20
- #include "../../../common.h"
21
-
22
- #if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \
23
- (!defined(MLD_CONFIG_NO_SIGN_API) || \
24
- !defined(MLD_CONFIG_NO_VERIFY_API)) && \
25
- !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \
26
- (defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \
27
- MLD_CONFIG_PARAMETER_SET == 44)
28
-
29
- #include <immintrin.h>
30
- #include "arith_native_x86_64.h"
31
-
32
- void mld_polyz_unpack_17_avx2(int32_t *r, const uint8_t *a)
33
- {
34
- unsigned int i;
35
- __m256i f;
36
- __m128i low, high;
37
-
38
- const __m256i shufbidx = _mm256_set_epi8(
39
- -1, 31, 30, 29, -1, 29, 28, 27, -1, 27, 26, 25, -1, 25, 24, 23, -1, 8, 7,
40
- 6, -1, 6, 5, 4, -1, 4, 3, 2, -1, 2, 1, 0);
41
- const __m256i srlvdidx = _mm256_set_epi32(6, 4, 2, 0, 6, 4, 2, 0);
42
- const __m256i mask = _mm256_set1_epi32(0x3FFFF);
43
- const __m256i gamma1 = _mm256_set1_epi32((1 << 17));
44
-
45
- for (i = 0; i < MLDSA_N / 8; i++)
46
- {
47
- /* Load bytes 0..15 into low 128-bit vector */
48
- low = _mm_loadu_si128((__m128i *)&a[18 * i]);
49
- /* Load bytes 2..17 into high 128-bit vector */
50
- high = _mm_loadu_si128((__m128i *)&a[18 * i + 2]);
51
- /* Combine into 256-bit vector */
52
- f = _mm256_inserti128_si256(_mm256_castsi128_si256(low), high, 1);
53
-
54
- /* Shuffling 8-bit lanes
55
- *
56
- * ┌─ Indices 0-8 into low 128-bit half ───────────────────────────────────┐
57
- * │ Shuffle: [-1, 8, 7, 6, -1, 6, 5, 4, -1, 4, 3, 2, -1, 2, 1, 0] │
58
- * │ Result: [0, byte8, byte7, byte6, ..., 0, byte2, byte1, byte0] │
59
- * └───────────────────────────────────────────────────────────────────────┘
60
- *
61
- * ┌─ Indices 16-31 into high 128-bit half ────────────────────────────────┐
62
- * │ Shuffle: [-1,31, 30, 29, -1,29, 28, 27, -1,27, 26, 25, -1,25, 24, 23] │
63
- * │ Result: [0, byte17, byte16, byte15, ..., 0, byte11, byte10, byte9] │
64
- * └───────────────────────────────────────────────────────────────────────┘
65
- */
66
- f = _mm256_shuffle_epi8(f, shufbidx);
67
-
68
- /* Keep only 18 out of 24 bits in each 32-bit lane */
69
- /* Bits 0..23 16..39 32..55 48..71
70
- * 72..95 88..111 104..127 120..143 */
71
- f = _mm256_srlv_epi32(f, srlvdidx);
72
- /* Bits 0..23 18..39 36..55 54..71
73
- * 72..95 90..111 108..127 126..143 */
74
- f = _mm256_and_si256(f, mask);
75
- /* Bits 0..17 18..35 36..53 54..71
76
- * 72..89 90..107 108..125 126..143 */
77
-
78
- /* Map [0, 1, ..., 2^18-1] to [2^17, 2^17-1, ..., -2^17+1] */
79
- f = _mm256_sub_epi32(gamma1, f);
80
-
81
- _mm256_store_si256((__m256i *)&r[8 * i], f);
82
- }
83
- }
84
- #else /* MLD_ARITH_BACKEND_X86_64_DEFAULT && (!MLD_CONFIG_NO_SIGN_API || \
85
- !MLD_CONFIG_NO_VERIFY_API) && !MLD_CONFIG_MULTILEVEL_NO_SHARED && \
86
- (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 44) \
87
- */
88
-
89
- MLD_EMPTY_CU(avx2_polyz_unpack_17)
90
-
91
- #endif /* !(MLD_ARITH_BACKEND_X86_64_DEFAULT && (!MLD_CONFIG_NO_SIGN_API || \
92
- !MLD_CONFIG_NO_VERIFY_API) && !MLD_CONFIG_MULTILEVEL_NO_SHARED && \
93
- (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == \
94
- 44)) */