pq_crypto 0.6.5 → 0.6.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (157) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +233 -0
  3. data/README.md +16 -3
  4. data/SECURITY.md +46 -0
  5. data/ext/pqcrypto/extconf.rb +8 -1
  6. data/ext/pqcrypto/pq_externalmu.c +35 -0
  7. data/ext/pqcrypto/pqcrypto_native_api.h +91 -75
  8. data/ext/pqcrypto/pqcrypto_ruby_secure.c +97 -9
  9. data/ext/pqcrypto/pqcrypto_secure.c +95 -33
  10. data/ext/pqcrypto/pqcrypto_secure.h +66 -48
  11. data/ext/pqcrypto/pqcrypto_version.h +1 -1
  12. data/ext/pqcrypto/vendor/.vendored +7 -7
  13. data/ext/pqcrypto/vendor/mldsa-native/BUILDING.md +5 -2
  14. data/ext/pqcrypto/vendor/mldsa-native/LICENSE +21 -2
  15. data/ext/pqcrypto/vendor/mldsa-native/README.md +20 -7
  16. data/ext/pqcrypto/vendor/mldsa-native/RELEASE.md +160 -0
  17. data/ext/pqcrypto/vendor/mldsa-native/SECURITY.md +1 -1
  18. data/ext/pqcrypto/vendor/mldsa-native/mldsa/README.md +2 -2
  19. data/ext/pqcrypto/vendor/mldsa-native/mldsa/mldsa_native.c +85 -59
  20. data/ext/pqcrypto/vendor/mldsa-native/mldsa/mldsa_native.h +292 -348
  21. data/ext/pqcrypto/vendor/mldsa-native/mldsa/mldsa_native_asm.S +122 -76
  22. data/ext/pqcrypto/vendor/mldsa-native/mldsa/mldsa_native_config.h +184 -86
  23. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/cbmc.h +49 -4
  24. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/common.h +49 -81
  25. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/context.h +152 -0
  26. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/ct.h +25 -12
  27. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/debug.c +2 -0
  28. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/debug.h +2 -0
  29. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/fips202x4.c +2 -2
  30. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/keccakf1600.c +9 -11
  31. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/auto.h +19 -11
  32. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x1_scalar_aarch64_asm.S +6 -4
  33. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x1_v84a_aarch64_asm.S +7 -4
  34. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x2_v84a_aarch64_asm.S +7 -4
  35. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm.S +12 -9
  36. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm.S +12 -9
  37. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/x1_scalar.h +1 -1
  38. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/x1_v84a.h +3 -2
  39. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/x2_v84a.h +3 -2
  40. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/x4_v8a_scalar.h +6 -1
  41. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/x4_v8a_v84a_scalar.h +3 -2
  42. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/api.h +11 -11
  43. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/armv81m/mve.h +9 -22
  44. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/armv81m/src/keccak_f1600_x4_mve.S +8 -5
  45. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/armv81m/src/keccak_f1600_x4_mve.c +1 -0
  46. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/armv81m/src/{state_extract_bytes_x4_mve.S → keccak_f1600_x4_state_extract_bytes_mve.S} +14 -14
  47. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/armv81m/src/{state_xor_bytes_x4_mve.S → keccak_f1600_x4_state_xor_bytes_mve.S} +12 -12
  48. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/auto.h +5 -4
  49. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/x86_64/keccak_f1600_x4_avx2.h +2 -2
  50. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/x86_64/src/fips202_native_x86_64.h +1 -0
  51. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/x86_64/src/keccak_f1600_x4_avx2_asm.S +36 -2
  52. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/meta.h +62 -4
  53. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/arith_native_aarch64.h +87 -54
  54. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{intt_aarch64_asm.S → mldsa_intt_aarch64_asm.S} +39 -6
  55. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{ntt_aarch64_asm.S → mldsa_ntt_aarch64_asm.S} +39 -6
  56. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{pointwise_montgomery_aarch64_asm.S → mldsa_pointwise_montgomery_aarch64_asm.S} +25 -3
  57. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{poly_caddq_aarch64_asm.S → mldsa_poly_caddq_aarch64_asm.S} +19 -3
  58. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{poly_chknorm_aarch64_asm.S → mldsa_poly_chknorm_aarch64_asm.S} +24 -3
  59. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{poly_decompose_32_aarch64_asm.S → mldsa_poly_decompose_32_aarch64_asm.S} +25 -3
  60. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{poly_decompose_88_aarch64_asm.S → mldsa_poly_decompose_88_aarch64_asm.S} +25 -3
  61. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{poly_use_hint_32_aarch64_asm.S → mldsa_poly_use_hint_32_aarch64_asm.S} +25 -3
  62. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{poly_use_hint_88_aarch64_asm.S → mldsa_poly_use_hint_88_aarch64_asm.S} +25 -3
  63. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{mld_polyvecl_pointwise_acc_montgomery_l4_aarch64_asm.S → mldsa_polyvecl_pointwise_acc_montgomery_l4_aarch64_asm.S} +31 -3
  64. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{mld_polyvecl_pointwise_acc_montgomery_l5_aarch64_asm.S → mldsa_polyvecl_pointwise_acc_montgomery_l5_aarch64_asm.S} +31 -3
  65. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{mld_polyvecl_pointwise_acc_montgomery_l7_aarch64_asm.S → mldsa_polyvecl_pointwise_acc_montgomery_l7_aarch64_asm.S} +31 -3
  66. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{polyz_unpack_17_aarch64_asm.S → mldsa_polyz_unpack_17_aarch64_asm.S} +31 -3
  67. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{polyz_unpack_19_aarch64_asm.S → mldsa_polyz_unpack_19_aarch64_asm.S} +31 -3
  68. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{rej_uniform_aarch64_asm.S → mldsa_rej_uniform_aarch64_asm.S} +48 -15
  69. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{rej_uniform_eta2_aarch64_asm.S → mldsa_rej_uniform_eta2_aarch64_asm.S} +42 -9
  70. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{rej_uniform_eta4_aarch64_asm.S → mldsa_rej_uniform_eta4_aarch64_asm.S} +42 -9
  71. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/api.h +11 -3
  72. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/meta.h +3 -2
  73. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/meta.h +28 -28
  74. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/arith_native_x86_64.h +171 -49
  75. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{intt_avx2_asm.S → mldsa_intt_avx2_asm.S} +23 -1
  76. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{ntt_avx2_asm.S → mldsa_ntt_avx2_asm.S} +23 -1
  77. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{nttunpack_avx2_asm.S → mldsa_nttunpack_avx2_asm.S} +17 -1
  78. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{pointwise_acc_l4_avx2_asm.S → mldsa_pointwise_acc_l4_avx2_asm.S} +37 -3
  79. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{pointwise_acc_l5_avx2_asm.S → mldsa_pointwise_acc_l5_avx2_asm.S} +37 -3
  80. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{pointwise_acc_l7_avx2_asm.S → mldsa_pointwise_acc_l7_avx2_asm.S} +37 -3
  81. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{pointwise_avx2_asm.S → mldsa_pointwise_avx2_asm.S} +31 -3
  82. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{poly_caddq_avx2_asm.S → mldsa_poly_caddq_avx2_asm.S} +18 -9
  83. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_poly_chknorm_avx2_asm.S +176 -0
  84. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_poly_decompose_32_avx2_asm.S +490 -0
  85. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_poly_decompose_88_avx2_asm.S +489 -0
  86. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_poly_use_hint_32_avx2_asm.S +123 -0
  87. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_poly_use_hint_88_avx2_asm.S +125 -0
  88. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_polyz_unpack_17_avx2_asm.S +355 -0
  89. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_polyz_unpack_19_avx2_asm.S +355 -0
  90. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_rej_uniform_avx2_asm.S +132 -0
  91. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_rej_uniform_eta2_avx2_asm.S +205 -0
  92. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_rej_uniform_eta4_avx2_asm.S +176 -0
  93. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/packing.c +27 -36
  94. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/packing.h +42 -8
  95. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/params.h +93 -17
  96. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/poly.c +74 -15
  97. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/poly.h +97 -11
  98. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/poly_kl.c +7 -38
  99. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/poly_kl.h +49 -7
  100. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/polyvec.c +16 -17
  101. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/polyvec.h +26 -9
  102. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/polyvec_lazy.c +3 -0
  103. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/polyvec_lazy.h +18 -19
  104. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/reduce.h +15 -3
  105. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/rounding.h +28 -6
  106. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/sign.c +311 -246
  107. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/sign.h +245 -240
  108. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/sys.h +64 -5
  109. data/ext/pqcrypto/vendor/mlkem-native/BUILDING.md +5 -2
  110. data/ext/pqcrypto/vendor/mlkem-native/LICENSE +21 -3
  111. data/ext/pqcrypto/vendor/mlkem-native/README.md +2 -2
  112. data/ext/pqcrypto/vendor/mlkem-native/RELEASE.md +113 -0
  113. data/ext/pqcrypto/vendor/mlkem-native/mlkem/README.md +2 -2
  114. data/ext/pqcrypto/vendor/mlkem-native/mlkem/mlkem_native.c +17 -27
  115. data/ext/pqcrypto/vendor/mlkem-native/mlkem/mlkem_native.h +68 -151
  116. data/ext/pqcrypto/vendor/mlkem-native/mlkem/mlkem_native_asm.S +17 -27
  117. data/ext/pqcrypto/vendor/mlkem-native/mlkem/mlkem_native_config.h +46 -44
  118. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/cbmc.h +25 -0
  119. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/common.h +37 -6
  120. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/context.h +9 -0
  121. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/fips202.h +2 -2
  122. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/keccakf1600.c +8 -8
  123. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x1_scalar.h +1 -1
  124. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x1_v84a.h +3 -3
  125. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x2_v84a.h +3 -3
  126. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x4_v8a_scalar.h +2 -2
  127. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x4_v8a_v84a_scalar.h +3 -3
  128. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/api.h +11 -11
  129. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/armv81m/mve.h +3 -3
  130. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/x86_64/keccak_f1600_x4_avx2.h +2 -2
  131. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/indcpa.h +14 -11
  132. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/kem.c +28 -11
  133. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/kem.h +39 -14
  134. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/meta.h +10 -10
  135. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/meta.h +20 -20
  136. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/sys.h +5 -5
  137. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/verify.h +11 -10
  138. data/lib/pq_crypto/internal.rb +10 -0
  139. data/lib/pq_crypto/kem.rb +47 -5
  140. data/lib/pq_crypto/key.rb +14 -8
  141. data/lib/pq_crypto/pkcs8.rb +13 -5
  142. data/lib/pq_crypto/signature.rb +34 -9
  143. data/lib/pq_crypto/spki.rb +4 -2
  144. data/lib/pq_crypto/version.rb +1 -1
  145. data/lib/pq_crypto.rb +9 -0
  146. data/script/vendor_libs.rb +6 -6
  147. metadata +40 -38
  148. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/poly_chknorm_avx2.c +0 -52
  149. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/poly_decompose_32_avx2.c +0 -157
  150. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/poly_decompose_88_avx2.c +0 -157
  151. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/poly_use_hint_32_avx2.c +0 -103
  152. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/poly_use_hint_88_avx2.c +0 -105
  153. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/polyz_unpack_17_avx2.c +0 -94
  154. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/polyz_unpack_19_avx2.c +0 -96
  155. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/rej_uniform_avx2.c +0 -126
  156. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/rej_uniform_eta2_avx2.c +0 -157
  157. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/rej_uniform_eta4_avx2.c +0 -141
@@ -1,96 +0,0 @@
1
- /*
2
- * Copyright (c) The mldsa-native project authors
3
- * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT
4
- */
5
-
6
- /* References
7
- * ==========
8
- *
9
- * - [REF_AVX2]
10
- * CRYSTALS-Dilithium optimized AVX2 implementation
11
- * Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé
12
- * https://github.com/pq-crystals/dilithium/tree/master/avx2
13
- */
14
-
15
- /*
16
- * This file is derived from the public domain
17
- * AVX2 Dilithium implementation @[REF_AVX2].
18
- */
19
-
20
- #include "../../../common.h"
21
-
22
- #if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \
23
- (!defined(MLD_CONFIG_NO_SIGN_API) || \
24
- !defined(MLD_CONFIG_NO_VERIFY_API)) && \
25
- !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \
26
- (defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \
27
- (MLD_CONFIG_PARAMETER_SET == 65 || MLD_CONFIG_PARAMETER_SET == 87))
28
-
29
- #include <immintrin.h>
30
- #include "arith_native_x86_64.h"
31
-
32
- void mld_polyz_unpack_19_avx2(int32_t *r, const uint8_t *a)
33
- {
34
- unsigned int i;
35
- __m256i f;
36
- __m128i low, high;
37
-
38
- const __m256i shufbidx = _mm256_set_epi8(
39
- -1, 31, 30, 29, -1, 29, 28, 27, -1, 26, 25, 24, -1, 24, 23, 22, -1, 9, 8,
40
- 7, -1, 7, 6, 5, -1, 4, 3, 2, -1, 2, 1, 0);
41
- /* Equivalent to _mm256_set_epi32(4, 0, 4, 0, 4, 0, 4, 0) */
42
- const __m256i srlvdidx = _mm256_set1_epi64x((uint64_t)4 << 32);
43
- const __m256i mask = _mm256_set1_epi32(0xFFFFF);
44
- const __m256i gamma1 = _mm256_set1_epi32((1 << 19));
45
-
46
- for (i = 0; i < MLDSA_N / 8; i++)
47
- {
48
- /* Load bytes 0..15 into low 128-bit vector */
49
- low = _mm_loadu_si128((__m128i *)&a[20 * i]);
50
- /* Load bytes 4..19 into high 128-bit vector */
51
- high = _mm_loadu_si128((__m128i *)&a[20 * i + 4]);
52
- /* Combine into 256-bit vector */
53
- f = _mm256_inserti128_si256(_mm256_castsi128_si256(low), high, 1);
54
-
55
- /* Shuffling 8-bit lanes
56
- *
57
- * ┌─ Indices 0-9 into low 128-bit half ───────────────────────────────────┐
58
- * │ Shuffle: [-1, 9, 8, 7, -1, 7, 6, 5, -1, 4, 3, 2, -1, 2, 1, 0] │
59
- * │ Result: [0, byte9, byte8, byte7, ..., 0, byte2, byte1, byte0] │
60
- * └───────────────────────────────────────────────────────────────────────┘
61
- *
62
- * ┌─ Indices 16-31 into high 128-bit half ────────────────────────────────┐
63
- * │ Shuffle: [-1,31, 30, 29, -1,29, 28, 27, -1,26, 25, 24, -1,24, 23, 22] │
64
- * │ Result: [0, byte19, byte18, byte17, ..., 0, byte12, byte11, byte10] │
65
- * └───────────────────────────────────────────────────────────────────────┘
66
- */
67
- f = _mm256_shuffle_epi8(f, shufbidx);
68
-
69
- /* Keep only 20 out of 24 bits in each 32-bit lane */
70
- /* Bits 0..23 16..39 40..63 56..79
71
- * 80..103 96..119 120..143 136..159 */
72
- f = _mm256_srlv_epi32(f, srlvdidx);
73
- /* Bits 0..23 20..39 40..63 60..79
74
- * 80..103 100..119 120..143 140..159 */
75
- f = _mm256_and_si256(f, mask);
76
- /* Bits 0..19 20..39 40..59 60..79
77
- * 80..99 100..119 120..139 140..159 */
78
-
79
- /* Map [0, 1, ..., 2^20-1] to [2^19, 2^19-1, ..., -2^19+1] */
80
- f = _mm256_sub_epi32(gamma1, f);
81
-
82
- _mm256_store_si256((__m256i *)&r[8 * i], f);
83
- }
84
- }
85
-
86
- #else /* MLD_ARITH_BACKEND_X86_64_DEFAULT && (!MLD_CONFIG_NO_SIGN_API || \
87
- !MLD_CONFIG_NO_VERIFY_API) && !MLD_CONFIG_MULTILEVEL_NO_SHARED && \
88
- (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 65 \
89
- || MLD_CONFIG_PARAMETER_SET == 87) */
90
-
91
- MLD_EMPTY_CU(avx2_polyz_unpack_19)
92
-
93
- #endif /* !(MLD_ARITH_BACKEND_X86_64_DEFAULT && (!MLD_CONFIG_NO_SIGN_API || \
94
- !MLD_CONFIG_NO_VERIFY_API) && !MLD_CONFIG_MULTILEVEL_NO_SHARED && \
95
- (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 65 \
96
- || MLD_CONFIG_PARAMETER_SET == 87)) */
@@ -1,126 +0,0 @@
1
- /*
2
- * Copyright (c) The mldsa-native project authors
3
- * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT
4
- */
5
-
6
- /* References
7
- * ==========
8
- *
9
- * - [REF_AVX2]
10
- * CRYSTALS-Dilithium optimized AVX2 implementation
11
- * Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé
12
- * https://github.com/pq-crystals/dilithium/tree/master/avx2
13
- */
14
-
15
- /*
16
- * This file is derived from the public domain
17
- * AVX2 Dilithium implementation @[REF_AVX2].
18
- */
19
-
20
- #include "../../../common.h"
21
-
22
- #if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \
23
- !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)
24
-
25
- #include <immintrin.h>
26
- #include "arith_native_x86_64.h"
27
- #include "consts.h"
28
-
29
- /*
30
- * Reference: The pqcrystals implementation assumes a buffer that is 8 bytes
31
- *. larger as the first loop overreads by 8 bytes that are then
32
- * discarded. We instead do not pad the buffer and do not overread.
33
- * The performance impact is negligible and it does not force the
34
- * frontend to perform the unintuitive padding.
35
- */
36
-
37
- unsigned int mld_rej_uniform_avx2(
38
- int32_t *MLD_RESTRICT r, const uint8_t buf[MLD_AVX2_REJ_UNIFORM_BUFLEN])
39
- {
40
- unsigned int ctr, pos;
41
- uint32_t good;
42
- __m256i d, tmp;
43
- const __m256i bound = _mm256_set1_epi32(MLDSA_Q);
44
- const __m256i mask = _mm256_set1_epi32(0x7FFFFF);
45
- const __m256i idx8 =
46
- _mm256_set_epi8(-1, 15, 14, 13, -1, 12, 11, 10, -1, 9, 8, 7, -1, 6, 5, 4,
47
- -1, 11, 10, 9, -1, 8, 7, 6, -1, 5, 4, 3, -1, 2, 1, 0);
48
-
49
- ctr = pos = 0;
50
- while (ctr <= MLDSA_N - 8 && pos <= MLD_AVX2_REJ_UNIFORM_BUFLEN - 32)
51
- {
52
- d = _mm256_loadu_si256((__m256i *)&buf[pos]);
53
-
54
- /* Permute 64-bit lanes
55
- * 0x94 = 10010100b rearranges 64-bit lanes as: [3,2,1,0] -> [2,1,1,0]
56
- *
57
- * ╔═══════════════════════════════════════════════════════════════════════╗
58
- * ║ Original Layout ║
59
- * ╚═══════════════════════════════════════════════════════════════════════╝
60
- * ┌─────────────────┬─────────────────┬─────────────────┬─────────────────┐
61
- * │ Lane 0 │ Lane 1 │ Lane 2 │ Lane 3 │
62
- * │ bytes 0..7 │ bytes 8..15 │ bytes 16..23 │ bytes 24..31 │
63
- * └─────────────────┴─────────────────┴─────────────────┴─────────────────┘
64
- *
65
- * ╔═══════════════════════════════════════════════════════════════════════╗
66
- * ║ Layout after permute ║
67
- * ║ Byte indices in high half shifted down by 8 positions ║
68
- * ╚═══════════════════════════════════════════════════════════════════════╝
69
- * ┌───────────────┬─────────────────┐ ┌─────────────────┬─────────────────┐
70
- * │ Lane 0 │ Lane 1 │ │ Lane 2 │ Lane 3 │
71
- * │ bytes 0..7 │ bytes 8..15 │ │ bytes 8..15 │ bytes 16..23 │
72
- * └───────────────┴─────────────────┘ └─────────────────┴─────────────────┘
73
- * Lower 128-bit lane (bytes 0-15) Upper 128-bit lane (bytes 16-31)
74
- */
75
- d = _mm256_permute4x64_epi64(d, 0x94);
76
-
77
- /* Shuffling 8-bit lanes
78
- *
79
- * ┌─ Indices 0-11 into low 128-bit half of permuted vector────────────────┐
80
- * │ Shuffle: [-1, 11, 10, 9, -1, 8, 7, 6, -1, 5, 4, 3, -1, 2, 1, 0] │
81
- * │ Result: [0, byte11, byte10, byte9, ..., 0, byte2, byte1, byte0] │
82
- * └───────────────────────────────────────────────────────────────────────┘
83
- *
84
- * ┌─ Indices 4-15 into high 128-bit half of permuted vector ──────────────┐
85
- * │ Shuffle: [-1, 15, 14, 13, -1, 12, 11, 10, -1, 9, 8, 7, -1, 6, 5, 4] │
86
- * │ Result: [0, byte23, byte22, byte21, ..., 0, byte14, byte13, byte12 │
87
- * └───────────────────────────────────────────────────────────────────────┘
88
- */
89
- d = _mm256_shuffle_epi8(d, idx8);
90
- d = _mm256_and_si256(d, mask);
91
- pos += 24;
92
-
93
- tmp = _mm256_sub_epi32(d, bound);
94
- good = (uint32_t)_mm256_movemask_ps((__m256)tmp);
95
- tmp = _mm256_cvtepu8_epi32(
96
- _mm_loadl_epi64((__m128i *)&mld_rej_uniform_table[good]));
97
- d = _mm256_permutevar8x32_epi32(d, tmp);
98
-
99
- _mm256_storeu_si256((__m256i *)&r[ctr], d);
100
- ctr += (unsigned)_mm_popcnt_u32(good);
101
- }
102
-
103
- while (ctr < MLDSA_N && pos <= MLD_AVX2_REJ_UNIFORM_BUFLEN - 3)
104
- {
105
- uint32_t t = buf[pos++];
106
- t |= (uint32_t)buf[pos++] << 8;
107
- t |= (uint32_t)buf[pos++] << 16;
108
- t &= 0x7FFFFF;
109
-
110
- if (t < MLDSA_Q)
111
- {
112
- /* Safe because t < MLDSA_Q. */
113
- r[ctr++] = (int32_t)t;
114
- }
115
- }
116
-
117
- return ctr;
118
- }
119
-
120
- #else /* MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_MULTILEVEL_NO_SHARED \
121
- */
122
-
123
- MLD_EMPTY_CU(avx2_rej_uniform)
124
-
125
- #endif /* !(MLD_ARITH_BACKEND_X86_64_DEFAULT && \
126
- !MLD_CONFIG_MULTILEVEL_NO_SHARED) */
@@ -1,157 +0,0 @@
1
- /*
2
- * Copyright (c) The mldsa-native project authors
3
- * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT
4
- */
5
-
6
- /* References
7
- * ==========
8
- *
9
- * - [REF_AVX2]
10
- * CRYSTALS-Dilithium optimized AVX2 implementation
11
- * Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé
12
- * https://github.com/pq-crystals/dilithium/tree/master/avx2
13
- */
14
-
15
- /*
16
- * This file is derived from the public domain
17
- * AVX2 Dilithium implementation @[REF_AVX2].
18
- */
19
-
20
- #include "../../../common.h"
21
-
22
- #if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \
23
- !defined(MLD_CONFIG_NO_KEYPAIR_API) && \
24
- !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \
25
- (defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLDSA_ETA == 2)
26
-
27
- #include <immintrin.h>
28
- #include "arith_native_x86_64.h"
29
- #include "consts.h"
30
-
31
- #define MLD_AVX2_ETA2 2
32
-
33
- /*
34
- * Reference: In the pqcrystals implementation this function is called
35
- * rej_eta_avx and supports multiple values for ETA via preprocessor
36
- * conditionals. We move the conditionals to the frontend.
37
- */
38
- unsigned int mld_rej_uniform_eta2_avx2(
39
- int32_t *MLD_RESTRICT r,
40
- const uint8_t buf[MLD_AVX2_REJ_UNIFORM_ETA2_BUFLEN])
41
- {
42
- unsigned int ctr, pos;
43
- uint32_t good;
44
- __m256i f0, f1, f2;
45
- __m128i g0, g1;
46
- const __m256i mask = _mm256_set1_epi8(15);
47
- const __m256i eta = _mm256_set1_epi8(MLD_AVX2_ETA2);
48
- const __m256i bound = mask;
49
- /* check-magic: -6560 == 32*round(-2**10 / 5) */
50
- const __m256i v = _mm256_set1_epi32(-6560);
51
- const __m256i p = _mm256_set1_epi32(5);
52
-
53
- ctr = pos = 0;
54
- while (ctr <= MLDSA_N - 8 && pos <= MLD_AVX2_REJ_UNIFORM_ETA2_BUFLEN - 16)
55
- {
56
- f0 = _mm256_cvtepu8_epi16(_mm_loadu_si128((__m128i *)&buf[pos]));
57
- f1 = _mm256_slli_epi16(f0, 4);
58
- f0 = _mm256_or_si256(f0, f1);
59
- f0 = _mm256_and_si256(f0, mask);
60
-
61
- f1 = _mm256_sub_epi8(f0, bound);
62
- f0 = _mm256_sub_epi8(eta, f0);
63
- good = (uint32_t)_mm256_movemask_epi8(f1);
64
-
65
- g0 = _mm256_castsi256_si128(f0);
66
- g1 = _mm_loadl_epi64((__m128i *)&mld_rej_uniform_table[good & 0xFF]);
67
- g1 = _mm_shuffle_epi8(g0, g1);
68
- f1 = _mm256_cvtepi8_epi32(g1);
69
- f2 = _mm256_mulhrs_epi16(f1, v);
70
- f2 = _mm256_mullo_epi16(f2, p);
71
- f1 = _mm256_add_epi32(f1, f2);
72
- _mm256_storeu_si256((__m256i *)&r[ctr], f1);
73
- ctr += (unsigned)_mm_popcnt_u32(good & 0xFF);
74
- good >>= 8;
75
- pos += 4;
76
-
77
- if (ctr > MLDSA_N - 8)
78
- {
79
- break;
80
- }
81
- g0 = _mm_bsrli_si128(g0, 8);
82
- g1 = _mm_loadl_epi64((__m128i *)&mld_rej_uniform_table[good & 0xFF]);
83
- g1 = _mm_shuffle_epi8(g0, g1);
84
- f1 = _mm256_cvtepi8_epi32(g1);
85
- f2 = _mm256_mulhrs_epi16(f1, v);
86
- f2 = _mm256_mullo_epi16(f2, p);
87
- f1 = _mm256_add_epi32(f1, f2);
88
- _mm256_storeu_si256((__m256i *)&r[ctr], f1);
89
- ctr += (unsigned)_mm_popcnt_u32(good & 0xFF);
90
- good >>= 8;
91
- pos += 4;
92
-
93
- if (ctr > MLDSA_N - 8)
94
- {
95
- break;
96
- }
97
- g0 = _mm256_extracti128_si256(f0, 1);
98
- g1 = _mm_loadl_epi64((__m128i *)&mld_rej_uniform_table[good & 0xFF]);
99
- g1 = _mm_shuffle_epi8(g0, g1);
100
- f1 = _mm256_cvtepi8_epi32(g1);
101
- f2 = _mm256_mulhrs_epi16(f1, v);
102
- f2 = _mm256_mullo_epi16(f2, p);
103
- f1 = _mm256_add_epi32(f1, f2);
104
- _mm256_storeu_si256((__m256i *)&r[ctr], f1);
105
- ctr += (unsigned)_mm_popcnt_u32(good & 0xFF);
106
- good >>= 8;
107
- pos += 4;
108
-
109
- if (ctr > MLDSA_N - 8)
110
- {
111
- break;
112
- }
113
- g0 = _mm_bsrli_si128(g0, 8);
114
- g1 = _mm_loadl_epi64((__m128i *)&mld_rej_uniform_table[good]);
115
- g1 = _mm_shuffle_epi8(g0, g1);
116
- f1 = _mm256_cvtepi8_epi32(g1);
117
- f2 = _mm256_mulhrs_epi16(f1, v);
118
- f2 = _mm256_mullo_epi16(f2, p);
119
- f1 = _mm256_add_epi32(f1, f2);
120
- _mm256_storeu_si256((__m256i *)&r[ctr], f1);
121
- ctr += (unsigned)_mm_popcnt_u32(good);
122
- pos += 4;
123
- }
124
-
125
- while (ctr < MLDSA_N && pos < MLD_AVX2_REJ_UNIFORM_ETA2_BUFLEN)
126
- {
127
- uint32_t t0 = buf[pos] & 0x0F;
128
- uint32_t t1 = buf[pos++] >> 4;
129
-
130
- if (t0 < 15)
131
- {
132
- t0 = t0 - (205 * t0 >> 10) * 5;
133
- r[ctr++] = (int32_t)(2 - t0);
134
- }
135
- if (t1 < 15 && ctr < MLDSA_N)
136
- {
137
- t1 = t1 - (205 * t1 >> 10) * 5;
138
- r[ctr++] = (int32_t)(2 - t1);
139
- }
140
- }
141
-
142
- return ctr;
143
- }
144
-
145
- #else /* MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_NO_KEYPAIR_API && \
146
- !MLD_CONFIG_MULTILEVEL_NO_SHARED && \
147
- (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLDSA_ETA == 2) */
148
-
149
- MLD_EMPTY_CU(avx2_rej_uniform_eta2)
150
-
151
- #endif /* !(MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_NO_KEYPAIR_API && \
152
- !MLD_CONFIG_MULTILEVEL_NO_SHARED && \
153
- (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLDSA_ETA == 2)) */
154
-
155
- /* To facilitate single-compilation-unit (SCU) builds, undefine all macros.
156
- * Don't modify by hand -- this is auto-generated by scripts/autogen. */
157
- #undef MLD_AVX2_ETA2
@@ -1,141 +0,0 @@
1
- /*
2
- * Copyright (c) The mldsa-native project authors
3
- * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT
4
- */
5
-
6
- /* References
7
- * ==========
8
- *
9
- * - [REF_AVX2]
10
- * CRYSTALS-Dilithium optimized AVX2 implementation
11
- * Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé
12
- * https://github.com/pq-crystals/dilithium/tree/master/avx2
13
- */
14
-
15
- /*
16
- * This file is derived from the public domain
17
- * AVX2 Dilithium implementation @[REF_AVX2].
18
- */
19
-
20
- #include "../../../common.h"
21
-
22
- #if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \
23
- !defined(MLD_CONFIG_NO_KEYPAIR_API) && \
24
- !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \
25
- (defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLDSA_ETA == 4)
26
-
27
- #include <immintrin.h>
28
- #include "arith_native_x86_64.h"
29
- #include "consts.h"
30
-
31
- #define MLD_AVX2_ETA4 4
32
-
33
- /*
34
- * Reference: In the pqcrystals implementation this function is called
35
- * rej_eta_avx and supports multiple values for ETA via preprocessor
36
- * conditionals. We move the conditionals to the frontend.
37
- */
38
-
39
- unsigned int mld_rej_uniform_eta4_avx2(
40
- int32_t *MLD_RESTRICT r,
41
- const uint8_t buf[MLD_AVX2_REJ_UNIFORM_ETA4_BUFLEN])
42
- {
43
- unsigned int ctr, pos;
44
- uint32_t good;
45
- __m256i f0, f1;
46
- __m128i g0, g1;
47
- const __m256i mask = _mm256_set1_epi8(15);
48
- const __m256i eta = _mm256_set1_epi8(MLD_AVX2_ETA4);
49
- const __m256i bound = _mm256_set1_epi8(9);
50
-
51
- ctr = pos = 0;
52
- while (ctr <= MLDSA_N - 8 && pos <= MLD_AVX2_REJ_UNIFORM_ETA4_BUFLEN - 16)
53
- {
54
- f0 = _mm256_cvtepu8_epi16(_mm_loadu_si128((__m128i *)&buf[pos]));
55
- f1 = _mm256_slli_epi16(f0, 4);
56
- f0 = _mm256_or_si256(f0, f1);
57
- f0 = _mm256_and_si256(f0, mask);
58
-
59
- f1 = _mm256_sub_epi8(f0, bound);
60
- f0 = _mm256_sub_epi8(eta, f0);
61
- good = (uint32_t)_mm256_movemask_epi8(f1);
62
-
63
- g0 = _mm256_castsi256_si128(f0);
64
- g1 = _mm_loadl_epi64((__m128i *)&mld_rej_uniform_table[good & 0xFF]);
65
- g1 = _mm_shuffle_epi8(g0, g1);
66
- f1 = _mm256_cvtepi8_epi32(g1);
67
- _mm256_storeu_si256((__m256i *)&r[ctr], f1);
68
- ctr += (unsigned)_mm_popcnt_u32(good & 0xFF);
69
- good >>= 8;
70
- pos += 4;
71
-
72
- if (ctr > MLDSA_N - 8)
73
- {
74
- break;
75
- }
76
- g0 = _mm_bsrli_si128(g0, 8);
77
- g1 = _mm_loadl_epi64((__m128i *)&mld_rej_uniform_table[good & 0xFF]);
78
- g1 = _mm_shuffle_epi8(g0, g1);
79
- f1 = _mm256_cvtepi8_epi32(g1);
80
- _mm256_storeu_si256((__m256i *)&r[ctr], f1);
81
- ctr += (unsigned)_mm_popcnt_u32(good & 0xFF);
82
- good >>= 8;
83
- pos += 4;
84
-
85
- if (ctr > MLDSA_N - 8)
86
- {
87
- break;
88
- }
89
- g0 = _mm256_extracti128_si256(f0, 1);
90
- g1 = _mm_loadl_epi64((__m128i *)&mld_rej_uniform_table[good & 0xFF]);
91
- g1 = _mm_shuffle_epi8(g0, g1);
92
- f1 = _mm256_cvtepi8_epi32(g1);
93
- _mm256_storeu_si256((__m256i *)&r[ctr], f1);
94
- ctr += (unsigned)_mm_popcnt_u32(good & 0xFF);
95
- good >>= 8;
96
- pos += 4;
97
-
98
- if (ctr > MLDSA_N - 8)
99
- {
100
- break;
101
- }
102
- g0 = _mm_bsrli_si128(g0, 8);
103
- g1 = _mm_loadl_epi64((__m128i *)&mld_rej_uniform_table[good]);
104
- g1 = _mm_shuffle_epi8(g0, g1);
105
- f1 = _mm256_cvtepi8_epi32(g1);
106
- _mm256_storeu_si256((__m256i *)&r[ctr], f1);
107
- ctr += (unsigned)_mm_popcnt_u32(good);
108
- pos += 4;
109
- }
110
-
111
- while (ctr < MLDSA_N && pos < MLD_AVX2_REJ_UNIFORM_ETA4_BUFLEN)
112
- {
113
- uint32_t t0 = buf[pos] & 0x0F;
114
- uint32_t t1 = buf[pos++] >> 4;
115
-
116
- if (t0 < 9)
117
- {
118
- r[ctr++] = (int32_t)(4 - t0);
119
- }
120
- if (t1 < 9 && ctr < MLDSA_N)
121
- {
122
- r[ctr++] = (int32_t)(4 - t1);
123
- }
124
- }
125
-
126
- return ctr;
127
- }
128
-
129
- #else /* MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_NO_KEYPAIR_API && \
130
- !MLD_CONFIG_MULTILEVEL_NO_SHARED && \
131
- (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLDSA_ETA == 4) */
132
-
133
- MLD_EMPTY_CU(avx2_rej_uniform_eta4)
134
-
135
- #endif /* !(MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_NO_KEYPAIR_API && \
136
- !MLD_CONFIG_MULTILEVEL_NO_SHARED && \
137
- (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLDSA_ETA == 4)) */
138
-
139
- /* To facilitate single-compilation-unit (SCU) builds, undefine all macros.
140
- * Don't modify by hand -- this is auto-generated by scripts/autogen. */
141
- #undef MLD_AVX2_ETA4