pq_crypto 0.6.5 → 0.6.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (157) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +233 -0
  3. data/README.md +16 -3
  4. data/SECURITY.md +46 -0
  5. data/ext/pqcrypto/extconf.rb +8 -1
  6. data/ext/pqcrypto/pq_externalmu.c +35 -0
  7. data/ext/pqcrypto/pqcrypto_native_api.h +91 -75
  8. data/ext/pqcrypto/pqcrypto_ruby_secure.c +97 -9
  9. data/ext/pqcrypto/pqcrypto_secure.c +95 -33
  10. data/ext/pqcrypto/pqcrypto_secure.h +66 -48
  11. data/ext/pqcrypto/pqcrypto_version.h +1 -1
  12. data/ext/pqcrypto/vendor/.vendored +7 -7
  13. data/ext/pqcrypto/vendor/mldsa-native/BUILDING.md +5 -2
  14. data/ext/pqcrypto/vendor/mldsa-native/LICENSE +21 -2
  15. data/ext/pqcrypto/vendor/mldsa-native/README.md +20 -7
  16. data/ext/pqcrypto/vendor/mldsa-native/RELEASE.md +160 -0
  17. data/ext/pqcrypto/vendor/mldsa-native/SECURITY.md +1 -1
  18. data/ext/pqcrypto/vendor/mldsa-native/mldsa/README.md +2 -2
  19. data/ext/pqcrypto/vendor/mldsa-native/mldsa/mldsa_native.c +85 -59
  20. data/ext/pqcrypto/vendor/mldsa-native/mldsa/mldsa_native.h +292 -348
  21. data/ext/pqcrypto/vendor/mldsa-native/mldsa/mldsa_native_asm.S +122 -76
  22. data/ext/pqcrypto/vendor/mldsa-native/mldsa/mldsa_native_config.h +184 -86
  23. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/cbmc.h +49 -4
  24. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/common.h +49 -81
  25. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/context.h +152 -0
  26. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/ct.h +25 -12
  27. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/debug.c +2 -0
  28. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/debug.h +2 -0
  29. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/fips202x4.c +2 -2
  30. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/keccakf1600.c +9 -11
  31. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/auto.h +19 -11
  32. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x1_scalar_aarch64_asm.S +6 -4
  33. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x1_v84a_aarch64_asm.S +7 -4
  34. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x2_v84a_aarch64_asm.S +7 -4
  35. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm.S +12 -9
  36. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm.S +12 -9
  37. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/x1_scalar.h +1 -1
  38. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/x1_v84a.h +3 -2
  39. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/x2_v84a.h +3 -2
  40. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/x4_v8a_scalar.h +6 -1
  41. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/x4_v8a_v84a_scalar.h +3 -2
  42. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/api.h +11 -11
  43. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/armv81m/mve.h +9 -22
  44. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/armv81m/src/keccak_f1600_x4_mve.S +8 -5
  45. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/armv81m/src/keccak_f1600_x4_mve.c +1 -0
  46. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/armv81m/src/{state_extract_bytes_x4_mve.S → keccak_f1600_x4_state_extract_bytes_mve.S} +14 -14
  47. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/armv81m/src/{state_xor_bytes_x4_mve.S → keccak_f1600_x4_state_xor_bytes_mve.S} +12 -12
  48. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/auto.h +5 -4
  49. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/x86_64/keccak_f1600_x4_avx2.h +2 -2
  50. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/x86_64/src/fips202_native_x86_64.h +1 -0
  51. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/x86_64/src/keccak_f1600_x4_avx2_asm.S +36 -2
  52. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/meta.h +62 -4
  53. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/arith_native_aarch64.h +87 -54
  54. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{intt_aarch64_asm.S → mldsa_intt_aarch64_asm.S} +39 -6
  55. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{ntt_aarch64_asm.S → mldsa_ntt_aarch64_asm.S} +39 -6
  56. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{pointwise_montgomery_aarch64_asm.S → mldsa_pointwise_montgomery_aarch64_asm.S} +25 -3
  57. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{poly_caddq_aarch64_asm.S → mldsa_poly_caddq_aarch64_asm.S} +19 -3
  58. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{poly_chknorm_aarch64_asm.S → mldsa_poly_chknorm_aarch64_asm.S} +24 -3
  59. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{poly_decompose_32_aarch64_asm.S → mldsa_poly_decompose_32_aarch64_asm.S} +25 -3
  60. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{poly_decompose_88_aarch64_asm.S → mldsa_poly_decompose_88_aarch64_asm.S} +25 -3
  61. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{poly_use_hint_32_aarch64_asm.S → mldsa_poly_use_hint_32_aarch64_asm.S} +25 -3
  62. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{poly_use_hint_88_aarch64_asm.S → mldsa_poly_use_hint_88_aarch64_asm.S} +25 -3
  63. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{mld_polyvecl_pointwise_acc_montgomery_l4_aarch64_asm.S → mldsa_polyvecl_pointwise_acc_montgomery_l4_aarch64_asm.S} +31 -3
  64. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{mld_polyvecl_pointwise_acc_montgomery_l5_aarch64_asm.S → mldsa_polyvecl_pointwise_acc_montgomery_l5_aarch64_asm.S} +31 -3
  65. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{mld_polyvecl_pointwise_acc_montgomery_l7_aarch64_asm.S → mldsa_polyvecl_pointwise_acc_montgomery_l7_aarch64_asm.S} +31 -3
  66. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{polyz_unpack_17_aarch64_asm.S → mldsa_polyz_unpack_17_aarch64_asm.S} +31 -3
  67. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{polyz_unpack_19_aarch64_asm.S → mldsa_polyz_unpack_19_aarch64_asm.S} +31 -3
  68. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{rej_uniform_aarch64_asm.S → mldsa_rej_uniform_aarch64_asm.S} +48 -15
  69. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{rej_uniform_eta2_aarch64_asm.S → mldsa_rej_uniform_eta2_aarch64_asm.S} +42 -9
  70. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{rej_uniform_eta4_aarch64_asm.S → mldsa_rej_uniform_eta4_aarch64_asm.S} +42 -9
  71. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/api.h +11 -3
  72. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/meta.h +3 -2
  73. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/meta.h +28 -28
  74. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/arith_native_x86_64.h +171 -49
  75. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{intt_avx2_asm.S → mldsa_intt_avx2_asm.S} +23 -1
  76. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{ntt_avx2_asm.S → mldsa_ntt_avx2_asm.S} +23 -1
  77. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{nttunpack_avx2_asm.S → mldsa_nttunpack_avx2_asm.S} +17 -1
  78. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{pointwise_acc_l4_avx2_asm.S → mldsa_pointwise_acc_l4_avx2_asm.S} +37 -3
  79. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{pointwise_acc_l5_avx2_asm.S → mldsa_pointwise_acc_l5_avx2_asm.S} +37 -3
  80. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{pointwise_acc_l7_avx2_asm.S → mldsa_pointwise_acc_l7_avx2_asm.S} +37 -3
  81. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{pointwise_avx2_asm.S → mldsa_pointwise_avx2_asm.S} +31 -3
  82. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{poly_caddq_avx2_asm.S → mldsa_poly_caddq_avx2_asm.S} +18 -9
  83. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_poly_chknorm_avx2_asm.S +176 -0
  84. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_poly_decompose_32_avx2_asm.S +490 -0
  85. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_poly_decompose_88_avx2_asm.S +489 -0
  86. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_poly_use_hint_32_avx2_asm.S +123 -0
  87. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_poly_use_hint_88_avx2_asm.S +125 -0
  88. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_polyz_unpack_17_avx2_asm.S +355 -0
  89. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_polyz_unpack_19_avx2_asm.S +355 -0
  90. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_rej_uniform_avx2_asm.S +132 -0
  91. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_rej_uniform_eta2_avx2_asm.S +205 -0
  92. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_rej_uniform_eta4_avx2_asm.S +176 -0
  93. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/packing.c +27 -36
  94. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/packing.h +42 -8
  95. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/params.h +93 -17
  96. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/poly.c +74 -15
  97. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/poly.h +97 -11
  98. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/poly_kl.c +7 -38
  99. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/poly_kl.h +49 -7
  100. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/polyvec.c +16 -17
  101. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/polyvec.h +26 -9
  102. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/polyvec_lazy.c +3 -0
  103. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/polyvec_lazy.h +18 -19
  104. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/reduce.h +15 -3
  105. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/rounding.h +28 -6
  106. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/sign.c +311 -246
  107. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/sign.h +245 -240
  108. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/sys.h +64 -5
  109. data/ext/pqcrypto/vendor/mlkem-native/BUILDING.md +5 -2
  110. data/ext/pqcrypto/vendor/mlkem-native/LICENSE +21 -3
  111. data/ext/pqcrypto/vendor/mlkem-native/README.md +2 -2
  112. data/ext/pqcrypto/vendor/mlkem-native/RELEASE.md +113 -0
  113. data/ext/pqcrypto/vendor/mlkem-native/mlkem/README.md +2 -2
  114. data/ext/pqcrypto/vendor/mlkem-native/mlkem/mlkem_native.c +17 -27
  115. data/ext/pqcrypto/vendor/mlkem-native/mlkem/mlkem_native.h +68 -151
  116. data/ext/pqcrypto/vendor/mlkem-native/mlkem/mlkem_native_asm.S +17 -27
  117. data/ext/pqcrypto/vendor/mlkem-native/mlkem/mlkem_native_config.h +46 -44
  118. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/cbmc.h +25 -0
  119. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/common.h +37 -6
  120. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/context.h +9 -0
  121. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/fips202.h +2 -2
  122. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/keccakf1600.c +8 -8
  123. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x1_scalar.h +1 -1
  124. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x1_v84a.h +3 -3
  125. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x2_v84a.h +3 -3
  126. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x4_v8a_scalar.h +2 -2
  127. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x4_v8a_v84a_scalar.h +3 -3
  128. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/api.h +11 -11
  129. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/armv81m/mve.h +3 -3
  130. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/x86_64/keccak_f1600_x4_avx2.h +2 -2
  131. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/indcpa.h +14 -11
  132. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/kem.c +28 -11
  133. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/kem.h +39 -14
  134. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/meta.h +10 -10
  135. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/meta.h +20 -20
  136. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/sys.h +5 -5
  137. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/verify.h +11 -10
  138. data/lib/pq_crypto/internal.rb +10 -0
  139. data/lib/pq_crypto/kem.rb +47 -5
  140. data/lib/pq_crypto/key.rb +14 -8
  141. data/lib/pq_crypto/pkcs8.rb +13 -5
  142. data/lib/pq_crypto/signature.rb +34 -9
  143. data/lib/pq_crypto/spki.rb +4 -2
  144. data/lib/pq_crypto/version.rb +1 -1
  145. data/lib/pq_crypto.rb +9 -0
  146. data/script/vendor_libs.rb +6 -6
  147. metadata +40 -38
  148. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/poly_chknorm_avx2.c +0 -52
  149. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/poly_decompose_32_avx2.c +0 -157
  150. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/poly_decompose_88_avx2.c +0 -157
  151. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/poly_use_hint_32_avx2.c +0 -103
  152. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/poly_use_hint_88_avx2.c +0 -105
  153. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/polyz_unpack_17_avx2.c +0 -94
  154. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/polyz_unpack_19_avx2.c +0 -96
  155. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/rej_uniform_avx2.c +0 -126
  156. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/rej_uniform_eta2_avx2.c +0 -157
  157. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/rej_uniform_eta4_avx2.c +0 -141
@@ -167,16 +167,18 @@ void mld_poly_shiftl(mld_poly *a)
167
167
  static MLD_INLINE int32_t mld_fqmul(int32_t a, int32_t b)
168
168
  __contract__(
169
169
  requires(b > -MLDSA_Q_HALF && b < MLDSA_Q_HALF)
170
- ensures(return_value > -MLDSA_Q && return_value < MLDSA_Q)
170
+ ensures(return_value > -MLD_FQMUL_BOUND && return_value < MLD_FQMUL_BOUND)
171
171
  )
172
172
  {
173
- /* Bounds: We argue in mld_montgomery_reduce() that the reult
173
+ /* Bounds: We argue in mld_montgomery_reduce() that the result
174
174
  * of Montgomery reduction is < MLDSA_Q if the input is smaller
175
175
  * than 2^31 * MLDSA_Q in absolute value. Indeed, we have:
176
176
  *
177
177
  * |a * b| = |a| * |b|
178
178
  * < 2^31 * MLDSA_Q_HALF
179
179
  * < 2^31 * MLDSA_Q
180
+ *
181
+ * So the output is < MLDSA_Q < MLD_FQMUL_BOUND.
180
182
  */
181
183
  return mld_montgomery_reduce((int64_t)a * (int64_t)b);
182
184
  }
@@ -215,17 +217,17 @@ static MLD_INLINE void mld_ntt_butterfly_block(int32_t r[MLDSA_N],
215
217
  const int32_t zeta,
216
218
  const unsigned start,
217
219
  const unsigned len,
218
- const unsigned bound)
220
+ const uint32_t bound)
219
221
  __contract__(
220
222
  requires(start < MLDSA_N)
221
223
  requires(1 <= len && len <= MLDSA_N / 2 && start + 2 * len <= MLDSA_N)
222
- requires(0 <= bound && bound < INT32_MAX - MLDSA_Q)
224
+ requires(0 <= bound && bound < INT32_MAX - MLD_FQMUL_BOUND)
223
225
  requires(-MLDSA_Q_HALF < zeta && zeta < MLDSA_Q_HALF)
224
226
  requires(memory_no_alias(r, sizeof(int32_t) * MLDSA_N))
225
- requires(array_abs_bound(r, 0, start, bound + MLDSA_Q))
227
+ requires(array_abs_bound(r, 0, start, bound + MLD_FQMUL_BOUND))
226
228
  requires(array_abs_bound(r, start, MLDSA_N, bound))
227
229
  assigns(memory_slice(r, sizeof(int32_t) * MLDSA_N))
228
- ensures(array_abs_bound(r, 0, start + 2*len, bound + MLDSA_Q))
230
+ ensures(array_abs_bound(r, 0, start + 2*len, bound + MLD_FQMUL_BOUND))
229
231
  ensures(array_abs_bound(r, start + 2 * len, MLDSA_N, bound)))
230
232
  {
231
233
  /* `bound` is a ghost variable only needed in the CBMC specification */
@@ -238,9 +240,9 @@ __contract__(
238
240
  * Coefficients are updated in strided pairs, so the bounds for the
239
241
  * intermediate states alternate twice between the old and new bound
240
242
  */
241
- invariant(array_abs_bound(r, 0, j, bound + MLDSA_Q))
243
+ invariant(array_abs_bound(r, 0, j, bound + MLD_FQMUL_BOUND))
242
244
  invariant(array_abs_bound(r, j, start + len, bound))
243
- invariant(array_abs_bound(r, start + len, j + len, bound + MLDSA_Q))
245
+ invariant(array_abs_bound(r, start + len, j + len, bound + MLD_FQMUL_BOUND))
244
246
  invariant(array_abs_bound(r, j + len, MLDSA_N, bound))
245
247
  decreases(start + len - j))
246
248
  {
@@ -265,9 +267,9 @@ static MLD_INLINE void mld_ntt_layer(int32_t r[MLDSA_N], const unsigned layer)
265
267
  __contract__(
266
268
  requires(memory_no_alias(r, sizeof(int32_t) * MLDSA_N))
267
269
  requires(1 <= layer && layer <= 8)
268
- requires(array_abs_bound(r, 0, MLDSA_N, layer * MLDSA_Q))
270
+ requires(array_abs_bound(r, 0, MLDSA_N, layer * MLD_FQMUL_BOUND))
269
271
  assigns(memory_slice(r, sizeof(int32_t) * MLDSA_N))
270
- ensures(array_abs_bound(r, 0, MLDSA_N, (layer + 1) * MLDSA_Q)))
272
+ ensures(array_abs_bound(r, 0, MLDSA_N, (layer + 1) * MLD_FQMUL_BOUND)))
271
273
  {
272
274
  unsigned start, k, len;
273
275
  /* Twiddle factors for layer n are at indices 2^(n-1)..2^n-1. */
@@ -278,12 +280,12 @@ __contract__(
278
280
  invariant(start < MLDSA_N + 2 * len)
279
281
  invariant(k <= MLDSA_N)
280
282
  invariant(2 * len * k == start + MLDSA_N)
281
- invariant(array_abs_bound(r, 0, start, layer * MLDSA_Q + MLDSA_Q))
282
- invariant(array_abs_bound(r, start, MLDSA_N, layer * MLDSA_Q))
283
+ invariant(array_abs_bound(r, 0, start, layer * MLD_FQMUL_BOUND + MLD_FQMUL_BOUND))
284
+ invariant(array_abs_bound(r, start, MLDSA_N, layer * MLD_FQMUL_BOUND))
283
285
  decreases(MLDSA_N - start))
284
286
  {
285
287
  int32_t zeta = mld_zetas[k++];
286
- mld_ntt_butterfly_block(r, zeta, start, len, layer * MLDSA_Q);
288
+ mld_ntt_butterfly_block(r, zeta, start, len, layer * MLD_FQMUL_BOUND);
287
289
  }
288
290
  }
289
291
 
@@ -305,7 +307,7 @@ __contract__(
305
307
  for (layer = 1; layer < 9; layer++)
306
308
  __loop__(
307
309
  invariant(1 <= layer && layer <= 9)
308
- invariant(array_abs_bound(r, 0, MLDSA_N, layer * MLDSA_Q))
310
+ invariant(array_abs_bound(r, 0, MLDSA_N, layer * MLD_FQMUL_BOUND))
309
311
  decreases(9 - layer)
310
312
  )
311
313
  {
@@ -377,13 +379,18 @@ __contract__(
377
379
  unsigned j;
378
380
  int32_t zeta = -mld_zetas[k--];
379
381
 
382
+ /* The bound `(MLDSA_N >> (layer - 1)) * MLDSA_Q` is loose enough to
383
+ * cover both the input bound `(MLDSA_N >> layer) * MLDSA_Q`
384
+ * (for layers >= 1) and the fqmul output bound `MLD_FQMUL_BOUND`
385
+ * (which is < 2 * MLDSA_Q <= (MLDSA_N >> (layer - 1)) * MLDSA_Q). */
380
386
  for (j = start; j < start + len; j++)
381
387
  __loop__(
382
388
  invariant(start <= j && j <= start + len)
383
389
  invariant(array_abs_bound(r, 0, start, (MLDSA_N >> (layer - 1)) * MLDSA_Q))
384
390
  invariant(array_abs_bound(r, start, j, (MLDSA_N >> (layer - 1)) * MLDSA_Q))
385
391
  invariant(array_abs_bound(r, j, start + len, (MLDSA_N >> layer) * MLDSA_Q))
386
- invariant(array_abs_bound(r, start + len, MLDSA_N, (MLDSA_N >> layer) * MLDSA_Q))
392
+ invariant(array_abs_bound(r, start + len, j + len, (MLDSA_N >> (layer - 1)) * MLDSA_Q))
393
+ invariant(array_abs_bound(r, j + len, MLDSA_N, (MLDSA_N >> layer) * MLDSA_Q))
387
394
  decreases(start + len - j))
388
395
  {
389
396
  int32_t t = r[j];
@@ -998,6 +1005,58 @@ uint32_t mld_poly_chknorm(const mld_poly *a, int32_t B)
998
1005
  return mld_poly_chknorm_c(a, B);
999
1006
  }
1000
1007
 
1008
+ #if !defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API)
1009
+ #if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \
1010
+ MLD_CONFIG_PARAMETER_SET == 44
1011
+ MLD_INTERNAL_API
1012
+ void mld_polyw1_pack_88(uint8_t r[MLDSA_POLYW1_PACKEDBYTES_88],
1013
+ const mld_poly *a)
1014
+ {
1015
+ unsigned int i;
1016
+
1017
+ mld_assert_bound(a->coeffs, MLDSA_N, 0,
1018
+ (MLDSA_Q - 1) / (2 * MLDSA_GAMMA2_88));
1019
+
1020
+ for (i = 0; i < MLDSA_N / 4; ++i)
1021
+ __loop__(
1022
+ invariant(i <= MLDSA_N/4)
1023
+ decreases(MLDSA_N / 4 - i))
1024
+ {
1025
+ r[3 * i + 0] = (uint8_t)((a->coeffs[4 * i + 0]) & 0xFF);
1026
+ r[3 * i + 0] |= (uint8_t)((a->coeffs[4 * i + 1] << 6) & 0xFF);
1027
+ r[3 * i + 1] = (uint8_t)((a->coeffs[4 * i + 1] >> 2) & 0xFF);
1028
+ r[3 * i + 1] |= (uint8_t)((a->coeffs[4 * i + 2] << 4) & 0xFF);
1029
+ r[3 * i + 2] = (uint8_t)((a->coeffs[4 * i + 2] >> 4) & 0xFF);
1030
+ r[3 * i + 2] |= (uint8_t)((a->coeffs[4 * i + 3] << 2) & 0xFF);
1031
+ }
1032
+ }
1033
+ #endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 44 \
1034
+ */
1035
+
1036
+ #if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \
1037
+ (MLD_CONFIG_PARAMETER_SET == 65 || MLD_CONFIG_PARAMETER_SET == 87)
1038
+ MLD_INTERNAL_API
1039
+ void mld_polyw1_pack_32(uint8_t r[MLDSA_POLYW1_PACKEDBYTES_32],
1040
+ const mld_poly *a)
1041
+ {
1042
+ unsigned int i;
1043
+
1044
+ mld_assert_bound(a->coeffs, MLDSA_N, 0,
1045
+ (MLDSA_Q - 1) / (2 * MLDSA_GAMMA2_32));
1046
+
1047
+ for (i = 0; i < MLDSA_N / 2; ++i)
1048
+ __loop__(
1049
+ invariant(i <= MLDSA_N/2)
1050
+ decreases(MLDSA_N / 2 - i))
1051
+ {
1052
+ r[i] =
1053
+ (uint8_t)((a->coeffs[2 * i + 0] | (a->coeffs[2 * i + 1] << 4)) & 0xFF);
1054
+ }
1055
+ }
1056
+ #endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 65 \
1057
+ || MLD_CONFIG_PARAMETER_SET == 87 */
1058
+ #endif /* !MLD_CONFIG_NO_SIGN_API || !MLD_CONFIG_NO_VERIFY_API */
1059
+
1001
1060
  #else /* !MLD_CONFIG_MULTILEVEL_NO_SHARED */
1002
1061
  MLD_EMPTY_CU(mld_poly)
1003
1062
  #endif /* MLD_CONFIG_MULTILEVEL_NO_SHARED */
@@ -2,6 +2,16 @@
2
2
  * Copyright (c) The mldsa-native project authors
3
3
  * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT
4
4
  */
5
+
6
+ /* References
7
+ * ==========
8
+ *
9
+ * - [FIPS204]
10
+ * FIPS 204 Module-Lattice-Based Digital Signature Standard
11
+ * National Institute of Standards and Technology
12
+ * https://csrc.nist.gov/pubs/fips/204/final
13
+ */
14
+
5
15
  #ifndef MLD_POLY_H
6
16
  #define MLD_POLY_H
7
17
 
@@ -10,8 +20,10 @@
10
20
  #include "reduce.h"
11
21
  #include "rounding.h"
12
22
 
23
+ /* Absolute exclusive upper bound for the output of fqmul */
24
+ #define MLD_FQMUL_BOUND ((5 * MLDSA_Q + 3) / 4)
13
25
  /* Absolute exclusive upper bound for the output of the forward NTT */
14
- #define MLD_NTT_BOUND (9 * MLDSA_Q)
26
+ #define MLD_NTT_BOUND (9 * MLD_FQMUL_BOUND)
15
27
  /* Absolute exclusive upper bound for the output of the inverse NTT*/
16
28
  #define MLD_INTT_BOUND MLDSA_Q
17
29
 
@@ -62,6 +74,9 @@ __contract__(
62
74
  /**
63
75
  * Add polynomials. No modular reduction is performed.
64
76
  *
77
+ * @spec{Implements @[FIPS204, Algorithm 44, AddNTT] (coefficientwise
78
+ * polynomial addition; also used for addition in the normal domain).}
79
+ *
65
80
  * @param[in,out] r Pointer to input-output polynomial to be added to.
66
81
  * @param[in] b Pointer to input polynomial that should be added to r.
67
82
  * Must be disjoint from r.
@@ -131,7 +146,10 @@ __contract__(
131
146
 
132
147
  #define mld_poly_ntt MLD_NAMESPACE(poly_ntt)
133
148
  /**
134
- * In-place forward NTT. Coefficients can grow by 8*MLDSA_Q in absolute value.
149
+ * In-place forward NTT. Output coefficients are bounded by MLD_NTT_BOUND in
150
+ * absolute value.
151
+ *
152
+ * @spec{Implements @[FIPS204, Algorithm 41, NTT].}
135
153
  *
136
154
  * @param[in,out] a Pointer to input/output polynomial.
137
155
  */
@@ -147,11 +165,16 @@ __contract__(
147
165
 
148
166
  #define mld_poly_invntt_tomont MLD_NAMESPACE(poly_invntt_tomont)
149
167
  /**
150
- * In-place inverse NTT and multiplication by 2^{32}.
168
+ * In-place inverse NTT.
151
169
  *
152
170
  * Input coefficients need to be less than MLDSA_Q in absolute value and
153
171
  * output coefficients are bounded by MLD_INTT_BOUND.
154
172
  *
173
+ * @spec{Implements @[FIPS204, Algorithm 42, NTT^{-1}] up to scaling:
174
+ * The input is scaled by 2^{-32} as a result of the Montgomery base
175
+ * multiplication. The output is in normal domain. In other words, this
176
+ * function implements `NTT^{-1} o mult(2^32)`.}
177
+ *
155
178
  * @param[in,out] a Pointer to input/output polynomial.
156
179
  */
157
180
  MLD_INTERNAL_API
@@ -167,9 +190,12 @@ __contract__(
167
190
  defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)
168
191
  #define mld_poly_pointwise_montgomery MLD_NAMESPACE(poly_pointwise_montgomery)
169
192
  /**
170
- * Pointwise multiplication of polynomials in NTT domain representation and
171
- * multiplication of resulting polynomial by 2^{-32}. Destructive in the first
172
- * argument.
193
+ * Pointwise multiplication of polynomials. Destructive in the first argument.
194
+ *
195
+ * @spec{Implements @[FIPS204, Algorithm 45, MultiplyNTT], up to scaling: The
196
+ * input is in normal domain, the output is scaled by 2^{-32} as a result of
197
+ * the use of Montgomery multiplication. In other words, this function
198
+ * implements `mult(2^{-32}) o MultiplyNTT`.}
173
199
  *
174
200
  * @param[in,out] a Pointer to first input/output polynomial. On entry, holds
175
201
  * the first multiplicand; on exit, holds the product
@@ -222,6 +248,8 @@ __contract__(
222
248
  * Sample polynomial with uniformly random coefficients in [0, MLDSA_Q-1] by
223
249
  * performing rejection sampling on the output stream of SHAKE128(seed|nonce).
224
250
  *
251
+ * @spec{Implements @[FIPS204, Algorithm 30, RejNTTPoly].}
252
+ *
225
253
  * @param[out] a Pointer to output polynomial.
226
254
  * @param[in] seed Byte array with seed of length MLDSA_SEEDBYTES and the
227
255
  * packed 2-byte nonce.
@@ -242,6 +270,8 @@ __contract__(
242
270
  * Generate four polynomials using rejection sampling on (pseudo-)uniformly
243
271
  * random bytes sampled from a seed.
244
272
  *
273
+ * @spec{Implements @[FIPS204, Algorithm 30, RejNTTPoly] (four-way batched).}
274
+ *
245
275
  * @param[out] vec0 Pointer to first polynomial to be sampled.
246
276
  * @param[out] vec1 Pointer to second polynomial to be sampled.
247
277
  * @param[out] vec2 Pointer to third polynomial to be sampled.
@@ -277,6 +307,8 @@ __contract__(
277
307
  * Bit-pack polynomial t1 with coefficients fitting in 10 bits. Input
278
308
  * coefficients are assumed to be standard representatives.
279
309
  *
310
+ * @spec{Implements @[FIPS204, Algorithm 16, SimpleBitPack].}
311
+ *
280
312
  * @param[out] r Pointer to output byte array with at least
281
313
  * MLDSA_POLYT1_PACKEDBYTES bytes.
282
314
  * @param[in] a Pointer to input polynomial.
@@ -297,6 +329,8 @@ __contract__(
297
329
  * Unpack polynomial t1 with 10-bit coefficients. Output coefficients are
298
330
  * standard representatives.
299
331
  *
332
+ * @spec{Implements @[FIPS204, Algorithm 18, SimpleBitUnpack].}
333
+ *
300
334
  * @param[out] r Pointer to output polynomial.
301
335
  * @param[in] a Byte array with bit-packed polynomial.
302
336
  */
@@ -315,6 +349,8 @@ __contract__(
315
349
  /**
316
350
  * Bit-pack polynomial t0 with coefficients in ]-2^{MLDSA_D-1}, 2^{MLDSA_D-1}].
317
351
  *
352
+ * @spec{Implements @[FIPS204, Algorithm 17, BitPack].}
353
+ *
318
354
  * @param[out] r Pointer to output byte array with at least
319
355
  * MLDSA_POLYT0_PACKEDBYTES bytes.
320
356
  * @param[in] a Pointer to input polynomial.
@@ -334,6 +370,8 @@ __contract__(
334
370
  /**
335
371
  * Unpack polynomial t0 with coefficients in ]-2^{MLDSA_D-1}, 2^{MLDSA_D-1}].
336
372
  *
373
+ * @spec{Implements @[FIPS204, Algorithm 19, BitUnpack].}
374
+ *
337
375
  * @param[out] r Pointer to output polynomial.
338
376
  * @param[in] a Byte array with bit-packed polynomial.
339
377
  */
@@ -352,11 +390,12 @@ __contract__(
352
390
  * Check infinity norm of polynomial against given bound. Assumes input
353
391
  * coefficients were reduced by mld_reduce32().
354
392
  *
355
- * @spec{The definition in FIPS-204 requires signed canonical reduction prior
356
- * to applying the bounds check. However, `-B < (a mod± MLDSA_Q) < B` is
357
- * equivalent to `-B < a < B` under the assumption that
358
- * `B <= MLDSA_Q - MLD_REDUCE32_RANGE_MAX` (cf. the assertion in the code).
359
- * Hence, the present spec and implementation are correct without reduction.}
393
+ * @spec{@[FIPS204] defines the infinity norm via signed canonical reduction
394
+ * (mod± MLDSA_Q) prior to applying the bounds check. However,
395
+ * `-B < (a mod± MLDSA_Q) < B` is equivalent to `-B < a < B` under the
396
+ * assumption that `B <= MLDSA_Q - MLD_REDUCE32_RANGE_MAX` (cf. the assertion in
397
+ * the code). Hence, this contract and implementation are correct without
398
+ * reduction.}
360
399
  *
361
400
  * @param[in] a Pointer to polynomial.
362
401
  * @param B Norm bound.
@@ -375,4 +414,51 @@ __contract__(
375
414
  ensures((return_value == 0) == array_abs_bound(a->coeffs, 0, MLDSA_N, B))
376
415
  );
377
416
 
417
+ #if !defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API)
418
+ #if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLD_CONFIG_PARAMETER_SET == 44
419
+ #define mld_polyw1_pack_88 MLD_NAMESPACE(polyw1_pack_88)
420
+ /**
421
+ * Bit-pack polynomial w1, using 6 bits per coefficient.
422
+ * This is the variant for parameter sets with MLDSA_GAMMA2 = (MLDSA_Q-1)/88
423
+ * (ML-DSA-44), for which w1 coefficients lie in [0, 43].
424
+ *
425
+ * @param[out] r Pointer to output byte array (MLDSA_POLYW1_PACKEDBYTES_88).
426
+ * @param[in] a Pointer to input polynomial.
427
+ */
428
+ MLD_INTERNAL_API
429
+ void mld_polyw1_pack_88(uint8_t r[MLDSA_POLYW1_PACKEDBYTES_88],
430
+ const mld_poly *a)
431
+ __contract__(
432
+ requires(memory_no_alias(r, MLDSA_POLYW1_PACKEDBYTES_88))
433
+ requires(memory_no_alias(a, sizeof(mld_poly)))
434
+ requires(array_bound(a->coeffs, 0, MLDSA_N, 0, (MLDSA_Q-1)/(2*MLDSA_GAMMA2_88)))
435
+ assigns(memory_slice(r, MLDSA_POLYW1_PACKEDBYTES_88))
436
+ );
437
+ #endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 44 \
438
+ */
439
+
440
+ #if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \
441
+ (MLD_CONFIG_PARAMETER_SET == 65 || MLD_CONFIG_PARAMETER_SET == 87)
442
+ #define mld_polyw1_pack_32 MLD_NAMESPACE(polyw1_pack_32)
443
+ /**
444
+ * Bit-pack polynomial w1, using 4 bits per coefficient.
445
+ * This is the variant for parameter sets with MLDSA_GAMMA2 = (MLDSA_Q-1)/32
446
+ * (ML-DSA-65 and ML-DSA-87), for which w1 coefficients lie in [0, 15].
447
+ *
448
+ * @param[out] r Pointer to output byte array (MLDSA_POLYW1_PACKEDBYTES_32).
449
+ * @param[in] a Pointer to input polynomial.
450
+ */
451
+ MLD_INTERNAL_API
452
+ void mld_polyw1_pack_32(uint8_t r[MLDSA_POLYW1_PACKEDBYTES_32],
453
+ const mld_poly *a)
454
+ __contract__(
455
+ requires(memory_no_alias(r, MLDSA_POLYW1_PACKEDBYTES_32))
456
+ requires(memory_no_alias(a, sizeof(mld_poly)))
457
+ requires(array_bound(a->coeffs, 0, MLDSA_N, 0, (MLDSA_Q-1)/(2*MLDSA_GAMMA2_32)))
458
+ assigns(memory_slice(r, MLDSA_POLYW1_PACKEDBYTES_32))
459
+ );
460
+ #endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 65 \
461
+ || MLD_CONFIG_PARAMETER_SET == 87 */
462
+ #endif /* !MLD_CONFIG_NO_SIGN_API || !MLD_CONFIG_NO_VERIFY_API */
463
+
378
464
  #endif /* !MLD_POLY_H */
@@ -601,7 +601,8 @@ void mld_poly_challenge(mld_poly *c, const uint8_t seed[MLDSA_CTILDEBYTES])
601
601
  decreases(MLDSA_N - i)
602
602
  )
603
603
  {
604
- /* This loop teminates only probabilistically, hence no decreases clause. */
604
+ /* This loop terminates only probabilistically, hence no decreases
605
+ * clause. */
605
606
  do
606
607
  __loop__(
607
608
  assigns(j, object_whole(buf), state, pos)
@@ -618,10 +619,10 @@ void mld_poly_challenge(mld_poly *c, const uint8_t seed[MLDSA_CTILDEBYTES])
618
619
 
619
620
  c->coeffs[i] = c->coeffs[j];
620
621
 
621
- /* Reference: Compute coefficent value here in two steps to */
622
- /* mixinf unsigned and signed arithmetic with implicit */
623
- /* conversions, and so that CBMC can keep track of ranges */
624
- /* to complete type-safety proof here. */
622
+ /* Reference: Compute coefficient value here in two steps to */
623
+ /* avoid mixing unsigned and signed arithmetic with implicit */
624
+ /* conversions, and so that CBMC can keep track of ranges */
625
+ /* to complete type-safety proof here. */
625
626
 
626
627
  /* The least-significant bit of signs tells us if we want -1 or +1 */
627
628
  offset = 2 * (signs & 1);
@@ -694,6 +695,7 @@ void mld_polyeta_pack(uint8_t r[MLDSA_POLYETA_PACKEDBYTES], const mld_poly *a)
694
695
  #endif /* !MLD_CONFIG_NO_KEYPAIR_API */
695
696
 
696
697
  #if !defined(MLD_CONFIG_NO_KEYPAIR_API) || !defined(MLD_CONFIG_NO_SIGN_API)
698
+ MLD_INTERNAL_API
697
699
  void mld_polyeta_unpack(mld_poly *r, const uint8_t a[MLDSA_POLYETA_PACKEDBYTES])
698
700
  {
699
701
  unsigned int i;
@@ -893,38 +895,6 @@ void mld_polyz_unpack(mld_poly *r, const uint8_t a[MLDSA_POLYZ_PACKEDBYTES])
893
895
 
894
896
  mld_polyz_unpack_c(r, a);
895
897
  }
896
-
897
- MLD_INTERNAL_API
898
- void mld_polyw1_pack(uint8_t r[MLDSA_POLYW1_PACKEDBYTES], const mld_poly *a)
899
- {
900
- unsigned int i;
901
-
902
- mld_assert_bound(a->coeffs, MLDSA_N, 0, (MLDSA_Q - 1) / (2 * MLDSA_GAMMA2));
903
-
904
- #if MLD_CONFIG_PARAMETER_SET == 44
905
- for (i = 0; i < MLDSA_N / 4; ++i)
906
- __loop__(
907
- invariant(i <= MLDSA_N/4)
908
- decreases(MLDSA_N / 4 - i))
909
- {
910
- r[3 * i + 0] = (uint8_t)((a->coeffs[4 * i + 0]) & 0xFF);
911
- r[3 * i + 0] |= (uint8_t)((a->coeffs[4 * i + 1] << 6) & 0xFF);
912
- r[3 * i + 1] = (uint8_t)((a->coeffs[4 * i + 1] >> 2) & 0xFF);
913
- r[3 * i + 1] |= (uint8_t)((a->coeffs[4 * i + 2] << 4) & 0xFF);
914
- r[3 * i + 2] = (uint8_t)((a->coeffs[4 * i + 2] >> 4) & 0xFF);
915
- r[3 * i + 2] |= (uint8_t)((a->coeffs[4 * i + 3] << 2) & 0xFF);
916
- }
917
- #else /* MLD_CONFIG_PARAMETER_SET == 44 */
918
- for (i = 0; i < MLDSA_N / 2; ++i)
919
- __loop__(
920
- invariant(i <= MLDSA_N/2)
921
- decreases(MLDSA_N / 2 - i))
922
- {
923
- r[i] =
924
- (uint8_t)((a->coeffs[2 * i + 0] | (a->coeffs[2 * i + 1] << 4)) & 0xFF);
925
- }
926
- #endif /* MLD_CONFIG_PARAMETER_SET != 44 */
927
- }
928
898
  #endif /* !MLD_CONFIG_NO_SIGN_API || !MLD_CONFIG_NO_VERIFY_API */
929
899
 
930
900
  /* To facilitate single-compilation-unit (SCU) builds, undefine all macros. */
@@ -937,5 +907,4 @@ void mld_polyw1_pack(uint8_t r[MLDSA_POLYW1_PACKEDBYTES], const mld_poly *a)
937
907
  #undef mld_poly_use_hint_c
938
908
  #undef mld_polyz_unpack_c
939
909
  #undef MLD_POLY_UNIFORM_ETA_NBLOCKS
940
- #undef MLD_POLY_UNIFORM_ETA_NBLOCKS
941
910
  #undef MLD_POLY_UNIFORM_GAMMA1_NBLOCKS
@@ -2,6 +2,16 @@
2
2
  * Copyright (c) The mldsa-native project authors
3
3
  * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT
4
4
  */
5
+
6
+ /* References
7
+ * ==========
8
+ *
9
+ * - [FIPS204]
10
+ * FIPS 204 Module-Lattice-Based Digital Signature Standard
11
+ * National Institute of Standards and Technology
12
+ * https://csrc.nist.gov/pubs/fips/204/final
13
+ */
14
+
5
15
  #ifndef MLD_POLY_KL_H
6
16
  #define MLD_POLY_KL_H
7
17
 
@@ -68,6 +78,9 @@ __contract__(
68
78
  * [-MLDSA_ETA, MLDSA_ETA] by performing rejection sampling on the output
69
79
  * stream from SHAKE256(seed|nonce_i).
70
80
  *
81
+ * @spec{Implements @[FIPS204, Algorithm 31, RejBoundedPoly] (four-way
82
+ * batched).}
83
+ *
71
84
  * @param[out] r0 Pointer to first output polynomial.
72
85
  * @param[out] r1 Pointer to second output polynomial.
73
86
  * @param[out] r2 Pointer to third output polynomial.
@@ -107,6 +120,8 @@ __contract__(
107
120
  * [-MLDSA_ETA, MLDSA_ETA] by performing rejection sampling on the output
108
121
  * stream from SHAKE256(seed|nonce).
109
122
  *
123
+ * @spec{Implements @[FIPS204, Algorithm 31, RejBoundedPoly].}
124
+ *
110
125
  * @param[out] r Pointer to output polynomial.
111
126
  * @param[in] seed Byte array with seed of length MLDSA_CRHBYTES.
112
127
  * @param nonce Nonce.
@@ -133,6 +148,9 @@ __contract__(
133
148
  * [-(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1] by unpacking output stream of
134
149
  * SHAKE256(seed|nonce).
135
150
  *
151
+ * @spec{Partially implements @[FIPS204, Algorithm 34, ExpandMask] (one
152
+ * polynomial, i.e. the loop body of lines 3-5).}
153
+ *
136
154
  * @param[out] a Pointer to output polynomial.
137
155
  * @param[in] seed Byte array with seed of length MLDSA_CRHBYTES.
138
156
  * @param nonce 16-bit nonce.
@@ -157,6 +175,9 @@ __contract__(
157
175
  * [-(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1] by unpacking output streams of
158
176
  * SHAKE256(seed|nonce_i).
159
177
  *
178
+ * @spec{Partially implements @[FIPS204, Algorithm 34, ExpandMask] (four-way
179
+ * batched, i.e. four iterations of the loop body of lines 3-5).}
180
+ *
160
181
  * @param[out] r0 Pointer to first output polynomial.
161
182
  * @param[out] r1 Pointer to second output polynomial.
162
183
  * @param[out] r2 Pointer to third output polynomial.
@@ -195,8 +216,10 @@ __contract__(
195
216
  #if !defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API)
196
217
  #define mld_poly_challenge MLD_NAMESPACE_KL(poly_challenge)
197
218
  /**
198
- * Implementation of H. Samples polynomial with MLDSA_TAU nonzero coefficients
199
- * in {-1, 1} using the output stream of SHAKE256(seed).
219
+ * Samples polynomial with MLDSA_TAU nonzero coefficients in {-1, 1} using the
220
+ * output stream of SHAKE256(seed).
221
+ *
222
+ * @spec{Implements @[FIPS204, Algorithm 29, SampleInBall].}
200
223
  *
201
224
  * @param[out] c Pointer to output polynomial.
202
225
  * @param[in] seed Byte array containing seed of length MLDSA_CTILDEBYTES.
@@ -217,6 +240,8 @@ __contract__(
217
240
  /**
218
241
  * Bit-pack polynomial with coefficients in [-MLDSA_ETA, MLDSA_ETA].
219
242
  *
243
+ * @spec{Implements @[FIPS204, Algorithm 17, BitPack].}
244
+ *
220
245
  * @param[out] r Pointer to output byte array with at least
221
246
  * MLDSA_POLYETA_PACKEDBYTES bytes.
222
247
  * @param[in] a Pointer to input polynomial.
@@ -252,6 +277,8 @@ __contract__(
252
277
  /**
253
278
  * Unpack polynomial with coefficients in [-MLDSA_ETA, MLDSA_ETA].
254
279
  *
280
+ * @spec{Implements @[FIPS204, Algorithm 19, BitUnpack].}
281
+ *
255
282
  * @param[out] r Pointer to output polynomial.
256
283
  * @param[in] a Byte array with bit-packed polynomial.
257
284
  */
@@ -271,6 +298,8 @@ __contract__(
271
298
  * Bit-pack polynomial with coefficients in
272
299
  * [-(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1].
273
300
  *
301
+ * @spec{Implements @[FIPS204, Algorithm 17, BitPack].}
302
+ *
274
303
  * @param[out] r Pointer to output byte array with at least
275
304
  * MLDSA_POLYZ_PACKEDBYTES bytes.
276
305
  * @param[in] a Pointer to input polynomial.
@@ -291,6 +320,8 @@ __contract__(
291
320
  * Unpack polynomial z with coefficients in
292
321
  * [-(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1].
293
322
  *
323
+ * @spec{Implements @[FIPS204, Algorithm 19, BitUnpack].}
324
+ *
294
325
  * @param[out] r Pointer to output polynomial.
295
326
  * @param[in] a Byte array with bit-packed polynomial.
296
327
  */
@@ -305,21 +336,32 @@ __contract__(
305
336
 
306
337
  #define mld_polyw1_pack MLD_NAMESPACE_KL(polyw1_pack)
307
338
  /**
308
- * Bit-pack polynomial w1 with coefficients in [0, 15] or [0, 43]. Input
309
- * coefficients are assumed to be standard representatives.
339
+ * Bit-pack polynomial w1. Input coefficients must be in
340
+ * [0, (MLDSA_Q-1)/(2*MLDSA_GAMMA2)), i.e. [0, 43] for ML-DSA-44 and [0, 15]
341
+ * for ML-DSA-65/87. Dispatches to the value-specialized variant for the
342
+ * selected parameter set.
343
+ *
344
+ * @spec{Implements @[FIPS204, Algorithm 16, SimpleBitPack].}
310
345
  *
311
346
  * @param[out] r Pointer to output byte array with at least
312
347
  * MLDSA_POLYW1_PACKEDBYTES bytes.
313
348
  * @param[in] a Pointer to input polynomial.
314
349
  */
315
- MLD_INTERNAL_API
316
- void mld_polyw1_pack(uint8_t r[MLDSA_POLYW1_PACKEDBYTES], const mld_poly *a)
350
+ static MLD_INLINE void mld_polyw1_pack(uint8_t r[MLDSA_POLYW1_PACKEDBYTES],
351
+ const mld_poly *a)
317
352
  __contract__(
318
353
  requires(memory_no_alias(r, MLDSA_POLYW1_PACKEDBYTES))
319
354
  requires(memory_no_alias(a, sizeof(mld_poly)))
320
355
  requires(array_bound(a->coeffs, 0, MLDSA_N, 0, (MLDSA_Q-1)/(2*MLDSA_GAMMA2)))
321
356
  assigns(memory_slice(r, MLDSA_POLYW1_PACKEDBYTES))
322
- );
357
+ )
358
+ {
359
+ #if MLD_CONFIG_PARAMETER_SET == 44
360
+ mld_polyw1_pack_88(r, a);
361
+ #else
362
+ mld_polyw1_pack_32(r, a);
363
+ #endif
364
+ }
323
365
  #endif /* !MLD_CONFIG_NO_SIGN_API || !MLD_CONFIG_NO_VERIFY_API */
324
366
 
325
367
  #endif /* !MLD_POLY_KL_H */
@@ -30,39 +30,38 @@
30
30
  MLD_INTERNAL_API
31
31
  void mld_polyvecl_uniform_gamma1(mld_polyvecl *v,
32
32
  const uint8_t seed[MLDSA_CRHBYTES],
33
- uint16_t nonce)
33
+ uint16_t kappa)
34
34
  {
35
35
  #if defined(MLD_CONFIG_SERIAL_FIPS202_ONLY)
36
36
  int i;
37
37
  #endif
38
38
 
39
- /* Safety: nonce is at most ((UINT16_MAX - MLDSA_L) / MLDSA_L), and, hence,
40
- * this cast is safe. See MLD_NONCE_UB comment in sign.c. */
41
- nonce = (uint16_t)(MLDSA_L * nonce);
42
- /* Now, nonce <= UINT16_MAX - (MLDSA_L - 1), so the casts below are safe. */
39
+ /* The caller passes the base counter kappa; component i is sampled from
40
+ * kappa + i. Safety: kappa <= MLD_MAX_KAPPA and i < MLDSA_L, so the
41
+ * casts below are safe. See MLD_MAX_KAPPA comment in params.h. */
43
42
  #if defined(MLD_CONFIG_SERIAL_FIPS202_ONLY)
44
43
  for (i = 0; i < MLDSA_L; i++)
45
44
  {
46
- mld_poly_uniform_gamma1(&v->vec[i], seed, (uint16_t)(nonce + i));
45
+ mld_poly_uniform_gamma1(&v->vec[i], seed, (uint16_t)(kappa + i));
47
46
  }
48
47
  #else /* MLD_CONFIG_SERIAL_FIPS202_ONLY */
49
48
  #if MLDSA_L == 4
50
49
  mld_poly_uniform_gamma1_4x(&v->vec[0], &v->vec[1], &v->vec[2], &v->vec[3],
51
- seed, nonce, (uint16_t)(nonce + 1),
52
- (uint16_t)(nonce + 2), (uint16_t)(nonce + 3));
50
+ seed, kappa, (uint16_t)(kappa + 1),
51
+ (uint16_t)(kappa + 2), (uint16_t)(kappa + 3));
53
52
  #elif MLDSA_L == 5
54
53
  mld_poly_uniform_gamma1_4x(&v->vec[0], &v->vec[1], &v->vec[2], &v->vec[3],
55
- seed, nonce, (uint16_t)(nonce + 1),
56
- (uint16_t)(nonce + 2), (uint16_t)(nonce + 3));
57
- mld_poly_uniform_gamma1(&v->vec[4], seed, (uint16_t)(nonce + 4));
54
+ seed, kappa, (uint16_t)(kappa + 1),
55
+ (uint16_t)(kappa + 2), (uint16_t)(kappa + 3));
56
+ mld_poly_uniform_gamma1(&v->vec[4], seed, (uint16_t)(kappa + 4));
58
57
  #elif MLDSA_L == 7
59
58
  mld_poly_uniform_gamma1_4x(&v->vec[0], &v->vec[1], &v->vec[2],
60
- &v->vec[3 /* irrelevant */], seed, nonce,
61
- (uint16_t)(nonce + 1), (uint16_t)(nonce + 2),
59
+ &v->vec[3 /* irrelevant */], seed, kappa,
60
+ (uint16_t)(kappa + 1), (uint16_t)(kappa + 2),
62
61
  0xFF /* irrelevant */);
63
62
  mld_poly_uniform_gamma1_4x(&v->vec[3], &v->vec[4], &v->vec[5], &v->vec[6],
64
- seed, (uint16_t)(nonce + 3), (uint16_t)(nonce + 4),
65
- (uint16_t)(nonce + 5), (uint16_t)(nonce + 6));
63
+ seed, (uint16_t)(kappa + 3), (uint16_t)(kappa + 4),
64
+ (uint16_t)(kappa + 5), (uint16_t)(kappa + 6));
66
65
  #endif /* MLDSA_L == 7 */
67
66
  #endif /* !MLD_CONFIG_SERIAL_FIPS202_ONLY */
68
67
 
@@ -196,11 +195,11 @@ void mld_polyvecl_pointwise_acc_montgomery(mld_poly *w, const mld_polyvecl *u,
196
195
  MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L7 && \
197
196
  MLD_CONFIG_PARAMETER_SET == 87 */
198
197
  /* The first input is bounded by [0, MLDSA_Q-1] inclusive.
199
- * The second input is bounded by [-(9*MLDSA_Q-1), 9*MLDSA_Q-1] inclusive.
198
+ * The second input is bounded by [-(MLD_NTT_BOUND-1), MLD_NTT_BOUND-1].
200
199
  * Hence, we can safely accumulate in 64-bits without intermediate reductions
201
200
  * as MLDSA_L * (MLD_NTT_BOUND-1) * (MLDSA_Q-1) < INT64_MAX.
202
201
  *
203
- * The worst case is ML-DSA-87: 7 * (9*MLDSA_Q-1) * (MLDSA_Q-1) < 2**52
202
+ * The worst case is ML-DSA-87: 7 * (MLD_NTT_BOUND-1) * (MLDSA_Q-1) < 2**53
204
203
  * (and likewise for negative values).
205
204
  */
206
205
  mld_polyvecl_pointwise_acc_montgomery_c(w, u, v);