mldsa_gh 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (128) hide show
  1. checksums.yaml +7 -0
  2. data/LICENSE +21 -0
  3. data/README.md +88 -0
  4. data/ext/mldsa_gh_native/extconf.rb +14 -0
  5. data/ext/mldsa_gh_native/mldsa_gh_native.c +222 -0
  6. data/ext/mldsa_gh_native/mldsa_gh_native_all.c +24 -0
  7. data/ext/mldsa_gh_native/mldsa_gh_native_all.h +26 -0
  8. data/ext/mldsa_gh_native/vendor/mldsa-native/BUILDING.md +108 -0
  9. data/ext/mldsa_gh_native/vendor/mldsa-native/LICENSE +305 -0
  10. data/ext/mldsa_gh_native/vendor/mldsa-native/README.md +247 -0
  11. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/README.md +23 -0
  12. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/mldsa_native.c +803 -0
  13. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/mldsa_native.h +956 -0
  14. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/mldsa_native_asm.S +830 -0
  15. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/mldsa_native_config.h +855 -0
  16. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/cbmc.h +233 -0
  17. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/common.h +301 -0
  18. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/context.h +152 -0
  19. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/ct.c +21 -0
  20. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/ct.h +373 -0
  21. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/debug.c +75 -0
  22. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/debug.h +125 -0
  23. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/fips202/fips202.c +270 -0
  24. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/fips202/fips202.h +224 -0
  25. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/fips202/fips202x4.c +187 -0
  26. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/fips202/fips202x4.h +125 -0
  27. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/fips202/keccakf1600.c +510 -0
  28. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/fips202/keccakf1600.h +110 -0
  29. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/auto.h +85 -0
  30. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/src/fips202_native_aarch64.h +69 -0
  31. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x1_scalar_aarch64_asm.S +378 -0
  32. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x1_v84a_aarch64_asm.S +207 -0
  33. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x2_v84a_aarch64_asm.S +262 -0
  34. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm.S +1080 -0
  35. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm.S +990 -0
  36. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/src/keccakf1600_round_constants.c +47 -0
  37. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/x1_scalar.h +27 -0
  38. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/x1_v84a.h +36 -0
  39. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/x2_v84a.h +40 -0
  40. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/x4_v8a_scalar.h +32 -0
  41. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/x4_v8a_v84a_scalar.h +37 -0
  42. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/fips202/native/api.h +129 -0
  43. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/fips202/native/armv81m/README.md +10 -0
  44. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/fips202/native/armv81m/mve.h +67 -0
  45. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/fips202/native/armv81m/src/fips202_native_armv81m.h +37 -0
  46. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/fips202/native/armv81m/src/keccak_f1600_x4_mve.S +717 -0
  47. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/fips202/native/armv81m/src/keccak_f1600_x4_mve.c +42 -0
  48. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/fips202/native/armv81m/src/keccak_f1600_x4_state_extract_bytes_mve.S +334 -0
  49. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/fips202/native/armv81m/src/keccak_f1600_x4_state_xor_bytes_mve.S +355 -0
  50. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/fips202/native/armv81m/src/keccakf1600_round_constants.c +53 -0
  51. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/fips202/native/auto.h +35 -0
  52. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/fips202/native/x86_64/keccak_f1600_x4_avx2.h +34 -0
  53. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/fips202/native/x86_64/src/fips202_native_x86_64.h +45 -0
  54. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/fips202/native/x86_64/src/keccak_f1600_x4_avx2_asm.S +488 -0
  55. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/fips202/native/x86_64/src/keccakf1600_constants.c +52 -0
  56. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/native/aarch64/meta.h +314 -0
  57. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/native/aarch64/src/aarch64_zetas.c +248 -0
  58. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/native/aarch64/src/arith_native_aarch64.h +367 -0
  59. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/native/aarch64/src/mldsa_intt_aarch64_asm.S +786 -0
  60. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/native/aarch64/src/mldsa_ntt_aarch64_asm.S +686 -0
  61. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/native/aarch64/src/mldsa_pointwise_montgomery_aarch64_asm.S +106 -0
  62. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/native/aarch64/src/mldsa_poly_caddq_aarch64_asm.S +69 -0
  63. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/native/aarch64/src/mldsa_poly_chknorm_aarch64_asm.S +76 -0
  64. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/native/aarch64/src/mldsa_poly_decompose_32_aarch64_asm.S +108 -0
  65. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/native/aarch64/src/mldsa_poly_decompose_88_aarch64_asm.S +108 -0
  66. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/native/aarch64/src/mldsa_poly_use_hint_32_aarch64_asm.S +125 -0
  67. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/native/aarch64/src/mldsa_poly_use_hint_88_aarch64_asm.S +133 -0
  68. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/native/aarch64/src/mldsa_polyvecl_pointwise_acc_montgomery_l4_aarch64_asm.S +157 -0
  69. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/native/aarch64/src/mldsa_polyvecl_pointwise_acc_montgomery_l5_aarch64_asm.S +173 -0
  70. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/native/aarch64/src/mldsa_polyvecl_pointwise_acc_montgomery_l7_aarch64_asm.S +205 -0
  71. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/native/aarch64/src/mldsa_polyz_unpack_17_aarch64_asm.S +103 -0
  72. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/native/aarch64/src/mldsa_polyz_unpack_19_aarch64_asm.S +100 -0
  73. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/native/aarch64/src/mldsa_rej_uniform_aarch64_asm.S +222 -0
  74. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/native/aarch64/src/mldsa_rej_uniform_eta2_aarch64_asm.S +170 -0
  75. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/native/aarch64/src/mldsa_rej_uniform_eta4_aarch64_asm.S +163 -0
  76. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/native/aarch64/src/polyz_unpack_table.c +52 -0
  77. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/native/aarch64/src/rej_uniform_eta_table.c +547 -0
  78. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/native/aarch64/src/rej_uniform_table.c +63 -0
  79. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/native/api.h +617 -0
  80. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/native/meta.h +24 -0
  81. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/native/x86_64/meta.h +323 -0
  82. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/native/x86_64/src/arith_native_x86_64.h +330 -0
  83. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/native/x86_64/src/consts.c +157 -0
  84. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/native/x86_64/src/consts.h +27 -0
  85. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_intt_avx2_asm.S +2333 -0
  86. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_ntt_avx2_asm.S +2405 -0
  87. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_nttunpack_avx2_asm.S +254 -0
  88. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_pointwise_acc_l4_avx2_asm.S +173 -0
  89. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_pointwise_acc_l5_avx2_asm.S +189 -0
  90. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_pointwise_acc_l7_avx2_asm.S +221 -0
  91. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_pointwise_avx2_asm.S +158 -0
  92. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_poly_caddq_avx2_asm.S +199 -0
  93. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_poly_chknorm_avx2_asm.S +176 -0
  94. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_poly_decompose_32_avx2_asm.S +490 -0
  95. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_poly_decompose_88_avx2_asm.S +489 -0
  96. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_poly_use_hint_32_avx2_asm.S +123 -0
  97. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_poly_use_hint_88_avx2_asm.S +125 -0
  98. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_polyz_unpack_17_avx2_asm.S +355 -0
  99. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_polyz_unpack_19_avx2_asm.S +355 -0
  100. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_rej_uniform_avx2_asm.S +132 -0
  101. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_rej_uniform_eta2_avx2_asm.S +205 -0
  102. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_rej_uniform_eta4_avx2_asm.S +176 -0
  103. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/native/x86_64/src/rej_uniform_table.c +161 -0
  104. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/packing.c +213 -0
  105. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/packing.h +277 -0
  106. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/params.h +153 -0
  107. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/poly.c +1066 -0
  108. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/poly.h +464 -0
  109. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/poly_kl.c +910 -0
  110. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/poly_kl.h +367 -0
  111. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/polyvec.c +509 -0
  112. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/polyvec.h +435 -0
  113. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/polyvec_lazy.c +311 -0
  114. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/polyvec_lazy.h +652 -0
  115. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/randombytes.h +26 -0
  116. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/reduce.h +144 -0
  117. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/rounding.h +265 -0
  118. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/sign.c +1720 -0
  119. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/sign.h +850 -0
  120. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/symmetric.h +68 -0
  121. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/sys.h +327 -0
  122. data/ext/mldsa_gh_native/vendor/mldsa-native/mldsa/src/zetas.inc +55 -0
  123. data/lib/mldsa/parameter_set.rb +82 -0
  124. data/lib/mldsa/signing_key.rb +147 -0
  125. data/lib/mldsa/verify_key.rb +72 -0
  126. data/lib/mldsa/version.rb +7 -0
  127. data/lib/mldsa_gh.rb +93 -0
  128. metadata +225 -0
@@ -0,0 +1,1066 @@
1
+ /*
2
+ * Copyright (c) The mldsa-native project authors
3
+ * Copyright (c) The mlkem-native project authors
4
+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT
5
+ */
6
+
7
+ /* References
8
+ * ==========
9
+ *
10
+ * - [FIPS204]
11
+ * FIPS 204 Module-Lattice-Based Digital Signature Standard
12
+ * National Institute of Standards and Technology
13
+ * https://csrc.nist.gov/pubs/fips/204/final
14
+ *
15
+ * - [REF]
16
+ * CRYSTALS-Dilithium reference implementation
17
+ * Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé
18
+ * https://github.com/pq-crystals/dilithium/tree/master/ref
19
+ */
20
+
21
+ #include "poly.h"
22
+
23
+ #include "common.h"
24
+ #include "ct.h"
25
+ #include "debug.h"
26
+ #include "reduce.h"
27
+ #include "rounding.h"
28
+ #include "symmetric.h"
29
+
30
+ #if !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)
31
+ #include "zetas.inc"
32
+
33
+ MLD_INTERNAL_API
34
+ void mld_poly_reduce(mld_poly *a)
35
+ {
36
+ unsigned int i;
37
+ mld_assert_bound(a->coeffs, MLDSA_N, INT32_MIN, MLD_REDUCE32_DOMAIN_MAX);
38
+
39
+ for (i = 0; i < MLDSA_N; ++i)
40
+ __loop__(
41
+ invariant(i <= MLDSA_N)
42
+ invariant(forall(k0, i, MLDSA_N, a->coeffs[k0] == loop_entry(*a).coeffs[k0]))
43
+ invariant(array_bound(a->coeffs, 0, i, -MLD_REDUCE32_RANGE_MAX, MLD_REDUCE32_RANGE_MAX))
44
+ decreases(MLDSA_N - i))
45
+ {
46
+ a->coeffs[i] = mld_reduce32(a->coeffs[i]);
47
+ }
48
+
49
+ mld_assert_bound(a->coeffs, MLDSA_N, -MLD_REDUCE32_RANGE_MAX,
50
+ MLD_REDUCE32_RANGE_MAX);
51
+ }
52
+
53
+ MLD_STATIC_TESTABLE void mld_poly_caddq_c(mld_poly *a)
54
+ __contract__(
55
+ requires(memory_no_alias(a, sizeof(mld_poly)))
56
+ requires(array_abs_bound(a->coeffs, 0, MLDSA_N, MLDSA_Q))
57
+ assigns(memory_slice(a, sizeof(mld_poly)))
58
+ ensures(array_bound(a->coeffs, 0, MLDSA_N, 0, MLDSA_Q))
59
+ )
60
+ {
61
+ unsigned int i;
62
+ mld_assert_abs_bound(a->coeffs, MLDSA_N, MLDSA_Q);
63
+
64
+ for (i = 0; i < MLDSA_N; ++i)
65
+ __loop__(
66
+ invariant(i <= MLDSA_N)
67
+ invariant(forall(k0, i, MLDSA_N, a->coeffs[k0] == loop_entry(*a).coeffs[k0]))
68
+ invariant(array_bound(a->coeffs, 0, i, 0, MLDSA_Q))
69
+ decreases(MLDSA_N - i)
70
+ )
71
+ {
72
+ a->coeffs[i] = mld_caddq(a->coeffs[i]);
73
+ }
74
+
75
+ mld_assert_bound(a->coeffs, MLDSA_N, 0, MLDSA_Q);
76
+ }
77
+
78
+ MLD_INTERNAL_API
79
+ void mld_poly_caddq(mld_poly *a)
80
+ {
81
+ #if defined(MLD_USE_NATIVE_POLY_CADDQ)
82
+ int ret;
83
+ mld_assert_abs_bound(a->coeffs, MLDSA_N, MLDSA_Q);
84
+ ret = mld_poly_caddq_native(a->coeffs);
85
+ if (ret == MLD_NATIVE_FUNC_SUCCESS)
86
+ {
87
+ mld_assert_bound(a->coeffs, MLDSA_N, 0, MLDSA_Q);
88
+ return;
89
+ }
90
+ #endif /* MLD_USE_NATIVE_POLY_CADDQ */
91
+ mld_poly_caddq_c(a);
92
+ }
93
+
94
+ #if !defined(MLD_CONFIG_NO_KEYPAIR_API) || !defined(MLD_CONFIG_NO_SIGN_API) || \
95
+ defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)
96
+ /* Reference: We use destructive version (output=first input) to avoid
97
+ * reasoning about aliasing in the CBMC specification */
98
+ MLD_INTERNAL_API
99
+ void mld_poly_add(mld_poly *r, const mld_poly *b)
100
+ {
101
+ unsigned int i;
102
+ for (i = 0; i < MLDSA_N; ++i)
103
+ __loop__(
104
+ assigns(i, memory_slice(r, sizeof(mld_poly)))
105
+ invariant(i <= MLDSA_N)
106
+ invariant(forall(k0, i, MLDSA_N, r->coeffs[k0] == loop_entry(*r).coeffs[k0]))
107
+ invariant(forall(k1, 0, i, r->coeffs[k1] == loop_entry(*r).coeffs[k1] + b->coeffs[k1]))
108
+ invariant(forall(k2, 0, i, r->coeffs[k2] < MLD_REDUCE32_DOMAIN_MAX))
109
+ invariant(forall(k2, 0, i, r->coeffs[k2] >= INT32_MIN))
110
+ decreases(MLDSA_N - i)
111
+ )
112
+ {
113
+ r->coeffs[i] = r->coeffs[i] + b->coeffs[i];
114
+ }
115
+ }
116
+ #endif /* !MLD_CONFIG_NO_KEYPAIR_API || !MLD_CONFIG_NO_SIGN_API || \
117
+ MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST */
118
+
119
+ #if !defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API)
120
+ /* Reference: We use destructive version (output=first input) to avoid
121
+ * reasoning about aliasing in the CBMC specification */
122
+ MLD_INTERNAL_API
123
+ void mld_poly_sub(mld_poly *r, const mld_poly *b)
124
+ {
125
+ unsigned int i;
126
+ mld_assert_abs_bound(b->coeffs, MLDSA_N, MLDSA_Q);
127
+ mld_assert_abs_bound(r->coeffs, MLDSA_N, MLDSA_Q);
128
+
129
+ for (i = 0; i < MLDSA_N; ++i)
130
+ __loop__(
131
+ invariant(i <= MLDSA_N)
132
+ invariant(array_bound(r->coeffs, 0, i, INT32_MIN, MLD_REDUCE32_DOMAIN_MAX))
133
+ invariant(forall(k0, i, MLDSA_N, r->coeffs[k0] == loop_entry(*r).coeffs[k0]))
134
+ decreases(MLDSA_N - i)
135
+ )
136
+ {
137
+ r->coeffs[i] = r->coeffs[i] - b->coeffs[i];
138
+ }
139
+
140
+ mld_assert_bound(r->coeffs, MLDSA_N, INT32_MIN, MLD_REDUCE32_DOMAIN_MAX);
141
+ }
142
+ #endif /* !MLD_CONFIG_NO_SIGN_API || !MLD_CONFIG_NO_VERIFY_API */
143
+
144
+ #if !defined(MLD_CONFIG_NO_VERIFY_API)
145
+ MLD_INTERNAL_API
146
+ void mld_poly_shiftl(mld_poly *a)
147
+ {
148
+ unsigned int i;
149
+ mld_assert_bound(a->coeffs, MLDSA_N, 0, 1 << 10);
150
+
151
+ for (i = 0; i < MLDSA_N; i++)
152
+ __loop__(
153
+ invariant(i <= MLDSA_N)
154
+ invariant(array_bound(a->coeffs, 0, i, 0, MLDSA_Q))
155
+ invariant(forall(k0, i, MLDSA_N, a->coeffs[k0] == loop_entry(*a).coeffs[k0]))
156
+ decreases(MLDSA_N - i))
157
+ {
158
+ /* Reference: uses a left shift by MLDSA_D which is undefined behaviour in
159
+ * C90/C99
160
+ */
161
+ a->coeffs[i] *= (1 << MLDSA_D);
162
+ }
163
+ mld_assert_bound(a->coeffs, MLDSA_N, 0, MLDSA_Q);
164
+ }
165
+ #endif /* !MLD_CONFIG_NO_VERIFY_API */
166
+
167
+ static MLD_INLINE int32_t mld_fqmul(int32_t a, int32_t b)
168
+ __contract__(
169
+ requires(b > -MLDSA_Q_HALF && b < MLDSA_Q_HALF)
170
+ ensures(return_value > -MLD_FQMUL_BOUND && return_value < MLD_FQMUL_BOUND)
171
+ )
172
+ {
173
+ /* Bounds: We argue in mld_montgomery_reduce() that the result
174
+ * of Montgomery reduction is < MLDSA_Q if the input is smaller
175
+ * than 2^31 * MLDSA_Q in absolute value. Indeed, we have:
176
+ *
177
+ * |a * b| = |a| * |b|
178
+ * < 2^31 * MLDSA_Q_HALF
179
+ * < 2^31 * MLDSA_Q
180
+ *
181
+ * So the output is < MLDSA_Q < MLD_FQMUL_BOUND.
182
+ */
183
+ return mld_montgomery_reduce((int64_t)a * (int64_t)b);
184
+ }
185
+
186
+ /* mld_ntt_butterfly_block()
187
+ *
188
+ * Computes a block CT butterflies with a fixed twiddle factor,
189
+ * using Montgomery multiplication.
190
+ *
191
+ * Parameters:
192
+ * - r: Pointer to base of polynomial (_not_ the base of butterfly block)
193
+ * - zeta: Twiddle factor to use for the butterfly. This must be in
194
+ * Montgomery form and signed canonical.
195
+ * - start: Offset to the beginning of the butterfly block
196
+ * - len: Index difference between coefficients subject to a butterfly
197
+ * - bound: Ghost variable describing coefficient bound: Prior to `start`,
198
+ * coefficients must be bound by `bound + MLDSA_Q`. Post `start`,
199
+ * they must be bound by `bound`.
200
+ * When this function returns, output coefficients in the index range
201
+ * [start, start+2*len) have bound bumped to `bound + MLDSA_Q`.
202
+ * Example:
203
+ * - start=8, len=4
204
+ * This would compute the following four butterflies
205
+ * 8 -- 12
206
+ * 9 -- 13
207
+ * 10 -- 14
208
+ * 11 -- 15
209
+ * - start=4, len=2
210
+ * This would compute the following two butterflies
211
+ * 4 -- 6
212
+ * 5 -- 7
213
+ */
214
+
215
+ /* Reference: Embedded in `ntt()` in the reference implementation @[REF]. */
216
+ static MLD_INLINE void mld_ntt_butterfly_block(int32_t r[MLDSA_N],
217
+ const int32_t zeta,
218
+ const unsigned start,
219
+ const unsigned len,
220
+ const uint32_t bound)
221
+ __contract__(
222
+ requires(start < MLDSA_N)
223
+ requires(1 <= len && len <= MLDSA_N / 2 && start + 2 * len <= MLDSA_N)
224
+ requires(0 <= bound && bound < INT32_MAX - MLD_FQMUL_BOUND)
225
+ requires(-MLDSA_Q_HALF < zeta && zeta < MLDSA_Q_HALF)
226
+ requires(memory_no_alias(r, sizeof(int32_t) * MLDSA_N))
227
+ requires(array_abs_bound(r, 0, start, bound + MLD_FQMUL_BOUND))
228
+ requires(array_abs_bound(r, start, MLDSA_N, bound))
229
+ assigns(memory_slice(r, sizeof(int32_t) * MLDSA_N))
230
+ ensures(array_abs_bound(r, 0, start + 2*len, bound + MLD_FQMUL_BOUND))
231
+ ensures(array_abs_bound(r, start + 2 * len, MLDSA_N, bound)))
232
+ {
233
+ /* `bound` is a ghost variable only needed in the CBMC specification */
234
+ unsigned j;
235
+ ((void)bound);
236
+ for (j = start; j < start + len; j++)
237
+ __loop__(
238
+ invariant(start <= j && j <= start + len)
239
+ /*
240
+ * Coefficients are updated in strided pairs, so the bounds for the
241
+ * intermediate states alternate twice between the old and new bound
242
+ */
243
+ invariant(array_abs_bound(r, 0, j, bound + MLD_FQMUL_BOUND))
244
+ invariant(array_abs_bound(r, j, start + len, bound))
245
+ invariant(array_abs_bound(r, start + len, j + len, bound + MLD_FQMUL_BOUND))
246
+ invariant(array_abs_bound(r, j + len, MLDSA_N, bound))
247
+ decreases(start + len - j))
248
+ {
249
+ int32_t t;
250
+ t = mld_fqmul(r[j + len], zeta);
251
+ r[j + len] = r[j] - t;
252
+ r[j] = r[j] + t;
253
+ }
254
+ }
255
+
256
+ /* mld_ntt_layer()
257
+ *
258
+ * Compute one layer of forward NTT
259
+ *
260
+ * Parameters:
261
+ * - r: Pointer to base of polynomial
262
+ * - layer: Indicates which layer is being applied.
263
+ */
264
+
265
+ /* Reference: Embedded in `ntt()` in the reference implementation @[REF]. */
266
+ static MLD_INLINE void mld_ntt_layer(int32_t r[MLDSA_N], const unsigned layer)
267
+ __contract__(
268
+ requires(memory_no_alias(r, sizeof(int32_t) * MLDSA_N))
269
+ requires(1 <= layer && layer <= 8)
270
+ requires(array_abs_bound(r, 0, MLDSA_N, layer * MLD_FQMUL_BOUND))
271
+ assigns(memory_slice(r, sizeof(int32_t) * MLDSA_N))
272
+ ensures(array_abs_bound(r, 0, MLDSA_N, (layer + 1) * MLD_FQMUL_BOUND)))
273
+ {
274
+ unsigned start, k, len;
275
+ /* Twiddle factors for layer n are at indices 2^(n-1)..2^n-1. */
276
+ k = 1u << (layer - 1);
277
+ len = (unsigned)MLDSA_N >> layer;
278
+ for (start = 0; start < MLDSA_N; start += 2 * len)
279
+ __loop__(
280
+ invariant(start < MLDSA_N + 2 * len)
281
+ invariant(k <= MLDSA_N)
282
+ invariant(2 * len * k == start + MLDSA_N)
283
+ invariant(array_abs_bound(r, 0, start, layer * MLD_FQMUL_BOUND + MLD_FQMUL_BOUND))
284
+ invariant(array_abs_bound(r, start, MLDSA_N, layer * MLD_FQMUL_BOUND))
285
+ decreases(MLDSA_N - start))
286
+ {
287
+ int32_t zeta = mld_zetas[k++];
288
+ mld_ntt_butterfly_block(r, zeta, start, len, layer * MLD_FQMUL_BOUND);
289
+ }
290
+ }
291
+
292
+ MLD_STATIC_TESTABLE void mld_poly_ntt_c(mld_poly *a)
293
+ __contract__(
294
+ requires(memory_no_alias(a, sizeof(mld_poly)))
295
+ requires(array_abs_bound(a->coeffs, 0, MLDSA_N, MLDSA_Q))
296
+ assigns(memory_slice(a, sizeof(mld_poly)))
297
+ ensures(array_abs_bound(a->coeffs, 0, MLDSA_N, MLD_NTT_BOUND))
298
+ )
299
+ {
300
+ unsigned int layer;
301
+ int32_t *r;
302
+
303
+
304
+ mld_assert_abs_bound(a->coeffs, MLDSA_N, MLDSA_Q);
305
+ r = a->coeffs;
306
+
307
+ for (layer = 1; layer < 9; layer++)
308
+ __loop__(
309
+ invariant(1 <= layer && layer <= 9)
310
+ invariant(array_abs_bound(r, 0, MLDSA_N, layer * MLD_FQMUL_BOUND))
311
+ decreases(9 - layer)
312
+ )
313
+ {
314
+ mld_ntt_layer(r, layer);
315
+ }
316
+
317
+ mld_assert_abs_bound(a->coeffs, MLDSA_N, MLD_NTT_BOUND);
318
+ }
319
+
320
+ MLD_INTERNAL_API
321
+ void mld_poly_ntt(mld_poly *a)
322
+ {
323
+ #if defined(MLD_USE_NATIVE_NTT)
324
+ int ret;
325
+ mld_assert_abs_bound(a->coeffs, MLDSA_N, MLDSA_Q);
326
+ ret = mld_ntt_native(a->coeffs);
327
+ if (ret == MLD_NATIVE_FUNC_SUCCESS)
328
+ {
329
+ mld_assert_abs_bound(a->coeffs, MLDSA_N, MLD_NTT_BOUND);
330
+ return;
331
+ }
332
+ #endif /* MLD_USE_NATIVE_NTT */
333
+ mld_poly_ntt_c(a);
334
+ }
335
+
336
+ /**
337
+ * Scale a field element by mont/256, i.e., perform Montgomery multiplication
338
+ * by mont^2/256.
339
+ *
340
+ * Input is expected to have absolute value smaller than 256 * MLDSA_Q. Output
341
+ * has absolute value smaller than MLD_INTT_BOUND.
342
+ *
343
+ * @param a Field element to be scaled.
344
+ */
345
+ static MLD_INLINE int32_t mld_fqscale(int32_t a)
346
+ __contract__(
347
+ requires(a > -256*MLDSA_Q && a < 256*MLDSA_Q)
348
+ ensures(return_value > -MLD_INTT_BOUND && return_value < MLD_INTT_BOUND)
349
+ )
350
+ {
351
+ /* check-magic: 41978 == pow(2,64-8,MLDSA_Q) */
352
+ const int32_t f = 41978;
353
+ /* Bounds: MLD_INTT_BOUND is MLDSA_Q, so the bounds reasoning is just
354
+ * a special case of that in mld_fqmul(). */
355
+ return mld_montgomery_reduce((int64_t)a * f);
356
+ }
357
+
358
+ /* Reference: Embedded into `invntt_tomont()` in the reference implementation
359
+ * @[REF] */
360
+ static MLD_INLINE void mld_invntt_layer(int32_t r[MLDSA_N], unsigned layer)
361
+ __contract__(
362
+ requires(memory_no_alias(r, sizeof(int32_t) * MLDSA_N))
363
+ requires(1 <= layer && layer <= 8)
364
+ requires(array_abs_bound(r, 0, MLDSA_N, (MLDSA_N >> layer) * MLDSA_Q))
365
+ assigns(memory_slice(r, sizeof(int32_t) * MLDSA_N))
366
+ ensures(array_abs_bound(r, 0, MLDSA_N, (MLDSA_N >> (layer - 1)) * MLDSA_Q)))
367
+ {
368
+ unsigned start, k, len;
369
+ len = (unsigned)MLDSA_N >> layer;
370
+ k = (1u << layer) - 1;
371
+ for (start = 0; start < MLDSA_N; start += 2 * len)
372
+ __loop__(
373
+ invariant(start <= MLDSA_N && k <= 255)
374
+ invariant(2 * len * k + start == 2 * MLDSA_N - 2 * len)
375
+ invariant(array_abs_bound(r, 0, start, (MLDSA_N >> (layer - 1)) * MLDSA_Q))
376
+ invariant(array_abs_bound(r, start, MLDSA_N, (MLDSA_N >> layer) * MLDSA_Q))
377
+ decreases(MLDSA_N - start))
378
+ {
379
+ unsigned j;
380
+ int32_t zeta = -mld_zetas[k--];
381
+
382
+ /* The bound `(MLDSA_N >> (layer - 1)) * MLDSA_Q` is loose enough to
383
+ * cover both the input bound `(MLDSA_N >> layer) * MLDSA_Q`
384
+ * (for layers >= 1) and the fqmul output bound `MLD_FQMUL_BOUND`
385
+ * (which is < 2 * MLDSA_Q <= (MLDSA_N >> (layer - 1)) * MLDSA_Q). */
386
+ for (j = start; j < start + len; j++)
387
+ __loop__(
388
+ invariant(start <= j && j <= start + len)
389
+ invariant(array_abs_bound(r, 0, start, (MLDSA_N >> (layer - 1)) * MLDSA_Q))
390
+ invariant(array_abs_bound(r, start, j, (MLDSA_N >> (layer - 1)) * MLDSA_Q))
391
+ invariant(array_abs_bound(r, j, start + len, (MLDSA_N >> layer) * MLDSA_Q))
392
+ invariant(array_abs_bound(r, start + len, j + len, (MLDSA_N >> (layer - 1)) * MLDSA_Q))
393
+ invariant(array_abs_bound(r, j + len, MLDSA_N, (MLDSA_N >> layer) * MLDSA_Q))
394
+ decreases(start + len - j))
395
+ {
396
+ int32_t t = r[j];
397
+ r[j] = t + r[j + len];
398
+ r[j + len] = t - r[j + len];
399
+ r[j + len] = mld_fqmul(r[j + len], zeta);
400
+ }
401
+ }
402
+ }
403
+
404
+ MLD_STATIC_TESTABLE void mld_poly_invntt_tomont_c(mld_poly *a)
405
+ __contract__(
406
+ requires(memory_no_alias(a, sizeof(mld_poly)))
407
+ requires(array_abs_bound(a->coeffs, 0, MLDSA_N, MLDSA_Q))
408
+ assigns(memory_slice(a, sizeof(mld_poly)))
409
+ ensures(array_abs_bound(a->coeffs, 0, MLDSA_N, MLD_INTT_BOUND))
410
+ )
411
+ {
412
+ unsigned int layer, j;
413
+ int32_t *r;
414
+
415
+ mld_assert_abs_bound(a->coeffs, MLDSA_N, MLDSA_Q);
416
+
417
+ r = a->coeffs;
418
+ for (layer = 8; layer >= 1; layer--)
419
+ __loop__(
420
+ invariant(layer <= 8)
421
+ /* Absolute bounds increase from 1Q before layer 8 */
422
+ /* up to 256Q after layer 1 */
423
+ invariant(array_abs_bound(r, 0, MLDSA_N, (MLDSA_N >> layer) * MLDSA_Q))
424
+ decreases(layer))
425
+ {
426
+ mld_invntt_layer(r, layer);
427
+ }
428
+
429
+ /* Coefficient bounds are now at 256Q. We now scale by mont / 256,
430
+ * i.e., compute the Montgomery multiplication by mont^2 / 256.
431
+ * mont corrects the mont^-1 factor introduced in the basemul.
432
+ * 1/256 performs that scaling of the inverse NTT.
433
+ * The reduced value is bounded by MLD_INTT_BOUND in absolute
434
+ * value.*/
435
+ for (j = 0; j < MLDSA_N; ++j)
436
+ __loop__(
437
+ invariant(j <= MLDSA_N)
438
+ invariant(array_abs_bound(r, 0, j, MLD_INTT_BOUND))
439
+ invariant(array_abs_bound(r, j, MLDSA_N, MLDSA_N * MLDSA_Q))
440
+ decreases(MLDSA_N - j)
441
+ )
442
+ {
443
+ r[j] = mld_fqscale(r[j]);
444
+ }
445
+
446
+ mld_assert_abs_bound(a->coeffs, MLDSA_N, MLD_INTT_BOUND);
447
+ }
448
+
449
+
450
+ MLD_INTERNAL_API
451
+ void mld_poly_invntt_tomont(mld_poly *a)
452
+ {
453
+ #if defined(MLD_USE_NATIVE_INTT)
454
+ int ret;
455
+ mld_assert_abs_bound(a->coeffs, MLDSA_N, MLDSA_Q);
456
+ ret = mld_intt_native(a->coeffs);
457
+ if (ret == MLD_NATIVE_FUNC_SUCCESS)
458
+ {
459
+ mld_assert_abs_bound(a->coeffs, MLDSA_N, MLD_INTT_BOUND);
460
+ return;
461
+ }
462
+ #endif /* MLD_USE_NATIVE_INTT */
463
+ mld_poly_invntt_tomont_c(a);
464
+ }
465
+
466
+ #if !defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API) || \
467
+ defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)
468
+ MLD_STATIC_TESTABLE void mld_poly_pointwise_montgomery_c(mld_poly *a,
469
+ const mld_poly *b)
470
+ __contract__(
471
+ requires(memory_no_alias(a, sizeof(mld_poly)))
472
+ requires(memory_no_alias(b, sizeof(mld_poly)))
473
+ requires(array_abs_bound(a->coeffs, 0, MLDSA_N, MLD_NTT_BOUND))
474
+ requires(array_abs_bound(b->coeffs, 0, MLDSA_N, MLD_NTT_BOUND))
475
+ assigns(memory_slice(a, sizeof(mld_poly)))
476
+ ensures(array_abs_bound(a->coeffs, 0, MLDSA_N, MLDSA_Q))
477
+ )
478
+ {
479
+ unsigned int i;
480
+ mld_assert_abs_bound(a->coeffs, MLDSA_N, MLD_NTT_BOUND);
481
+ mld_assert_abs_bound(b->coeffs, MLDSA_N, MLD_NTT_BOUND);
482
+
483
+ for (i = 0; i < MLDSA_N; ++i)
484
+ __loop__(
485
+ invariant(i <= MLDSA_N)
486
+ invariant(array_abs_bound(a->coeffs, 0, i, MLDSA_Q))
487
+ invariant(array_abs_bound(a->coeffs, i, MLDSA_N, MLD_NTT_BOUND))
488
+ decreases(MLDSA_N - i)
489
+ )
490
+ {
491
+ a->coeffs[i] = mld_montgomery_reduce((int64_t)a->coeffs[i] * b->coeffs[i]);
492
+ }
493
+ mld_assert_abs_bound(a->coeffs, MLDSA_N, MLDSA_Q);
494
+ }
495
+
496
+ MLD_INTERNAL_API
497
+ void mld_poly_pointwise_montgomery(mld_poly *a, const mld_poly *b)
498
+ {
499
+ #if defined(MLD_USE_NATIVE_POINTWISE_MONTGOMERY)
500
+ int ret;
501
+ mld_assert_abs_bound(a->coeffs, MLDSA_N, MLD_NTT_BOUND);
502
+ mld_assert_abs_bound(b->coeffs, MLDSA_N, MLD_NTT_BOUND);
503
+ ret = mld_poly_pointwise_montgomery_native(a->coeffs, b->coeffs);
504
+ if (ret == MLD_NATIVE_FUNC_SUCCESS)
505
+ {
506
+ mld_assert_abs_bound(a->coeffs, MLDSA_N, MLDSA_Q);
507
+ return;
508
+ }
509
+ #endif /* MLD_USE_NATIVE_POINTWISE_MONTGOMERY */
510
+ mld_poly_pointwise_montgomery_c(a, b);
511
+ }
512
+ #endif /* !MLD_CONFIG_NO_SIGN_API || !MLD_CONFIG_NO_VERIFY_API || \
513
+ MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST */
514
+
515
+ #if !defined(MLD_CONFIG_NO_KEYPAIR_API)
516
+ MLD_INTERNAL_API
517
+ void mld_poly_power2round(mld_poly *a1, mld_poly *a0, const mld_poly *a)
518
+ {
519
+ unsigned int i;
520
+ mld_assert_bound(a->coeffs, MLDSA_N, 0, MLDSA_Q);
521
+
522
+ for (i = 0; i < MLDSA_N; ++i)
523
+ __loop__(
524
+ assigns(i, memory_slice(a0, sizeof(mld_poly)), memory_slice(a1, sizeof(mld_poly)))
525
+ invariant(i <= MLDSA_N)
526
+ invariant(forall(k0, i, MLDSA_N, a->coeffs[k0] == loop_entry(*a).coeffs[k0]))
527
+ invariant(array_bound(a0->coeffs, 0, i, -(MLD_2_POW_D/2)+1, (MLD_2_POW_D/2)+1))
528
+ invariant(array_bound(a1->coeffs, 0, i, 0, ((MLDSA_Q - 1) / MLD_2_POW_D) + 1))
529
+ decreases(MLDSA_N - i)
530
+ )
531
+ {
532
+ mld_power2round(&a0->coeffs[i], &a1->coeffs[i], a->coeffs[i]);
533
+ }
534
+
535
+ mld_assert_bound(a0->coeffs, MLDSA_N, -(MLD_2_POW_D / 2) + 1,
536
+ (MLD_2_POW_D / 2) + 1);
537
+ mld_assert_bound(a1->coeffs, MLDSA_N, 0, ((MLDSA_Q - 1) / MLD_2_POW_D) + 1);
538
+ }
539
+ #endif /* !MLD_CONFIG_NO_KEYPAIR_API */
540
+
541
+ #ifndef MLD_POLY_UNIFORM_NBLOCKS
542
+ #define MLD_POLY_UNIFORM_NBLOCKS \
543
+ ((768 + MLD_STREAM128_BLOCKBYTES - 1) / MLD_STREAM128_BLOCKBYTES)
544
+ #endif
545
+ /* Reference: `mld_rej_uniform()` in the reference implementation @[REF].
546
+ * - Our signature differs from the reference implementation
547
+ * in that it adds the offset and always expects the base of the
548
+ * target buffer. This avoids shifting the buffer base in the
549
+ * caller, which appears tricky to reason about. */
550
+ MLD_STATIC_TESTABLE unsigned int mld_rej_uniform_c(int32_t *a,
551
+ unsigned int target,
552
+ unsigned int offset,
553
+ const uint8_t *buf,
554
+ unsigned int buflen)
555
+ __contract__(
556
+ requires(offset <= target && target <= MLDSA_N)
557
+ requires(buflen <= (MLD_POLY_UNIFORM_NBLOCKS * MLD_STREAM128_BLOCKBYTES) && buflen % 3 == 0)
558
+ requires(memory_no_alias(a, sizeof(int32_t) * target))
559
+ requires(memory_no_alias(buf, buflen))
560
+ requires(array_bound(a, 0, offset, 0, MLDSA_Q))
561
+ assigns(memory_slice(a, sizeof(int32_t) * target))
562
+ ensures(offset <= return_value && return_value <= target)
563
+ ensures(array_bound(a, 0, return_value, 0, MLDSA_Q))
564
+ )
565
+ {
566
+ unsigned int ctr, pos;
567
+ uint32_t t;
568
+ mld_assert_bound(a, offset, 0, MLDSA_Q);
569
+
570
+ ctr = offset;
571
+ pos = 0;
572
+ /* pos + 3 cannot overflow due to the assumption
573
+ buflen <= (MLD_POLY_UNIFORM_NBLOCKS * MLD_STREAM128_BLOCKBYTES) */
574
+ while (ctr < target && pos + 3 <= buflen)
575
+ __loop__(
576
+ invariant(offset <= ctr && ctr <= target && pos <= buflen)
577
+ invariant(array_bound(a, 0, ctr, 0, MLDSA_Q))
578
+ decreases(buflen - pos))
579
+ {
580
+ t = buf[pos++];
581
+ t |= (uint32_t)buf[pos++] << 8;
582
+ t |= (uint32_t)buf[pos++] << 16;
583
+ t &= 0x7FFFFF;
584
+
585
+ if (t < MLDSA_Q)
586
+ {
587
+ a[ctr++] = (int32_t)t;
588
+ }
589
+ }
590
+
591
+ mld_assert_bound(a, ctr, 0, MLDSA_Q);
592
+
593
+ return ctr;
594
+ }
595
+ /**
596
+ * Sample uniformly random coefficients in [0, MLDSA_Q-1] by performing
597
+ * rejection sampling on an array of random bytes.
598
+ *
599
+ * @param[out] a Pointer to output array (allocated).
600
+ * @param target Requested number of coefficients to sample.
601
+ * @param offset Number of coefficients already sampled.
602
+ * @param[in] buf Array of random bytes to sample from.
603
+ * @param buflen Length of array of random bytes (must be multiple of 3).
604
+ *
605
+ * @return Number of sampled coefficients. Can be smaller than len if not
606
+ * enough random bytes were given.
607
+ */
608
+
609
+ /* Reference: `mld_rej_uniform()` in the reference implementation @[REF].
610
+ * - Our signature differs from the reference implementation
611
+ * in that it adds the offset and always expects the base of the
612
+ * target buffer. This avoids shifting the buffer base in the
613
+ * caller, which appears tricky to reason about. */
614
+ static unsigned int mld_rej_uniform(int32_t *a, unsigned int target,
615
+ unsigned int offset, const uint8_t *buf,
616
+ unsigned int buflen)
617
+ __contract__(
618
+ requires(offset <= target && target <= MLDSA_N)
619
+ requires(buflen <= (MLD_POLY_UNIFORM_NBLOCKS * MLD_STREAM128_BLOCKBYTES) && buflen % 3 == 0)
620
+ requires(memory_no_alias(a, sizeof(int32_t) * target))
621
+ requires(memory_no_alias(buf, buflen))
622
+ requires(array_bound(a, 0, offset, 0, MLDSA_Q))
623
+ assigns(memory_slice(a, sizeof(int32_t) * target))
624
+ ensures(offset <= return_value && return_value <= target)
625
+ ensures(array_bound(a, 0, return_value, 0, MLDSA_Q))
626
+ )
627
+ {
628
+ #if defined(MLD_USE_NATIVE_REJ_UNIFORM)
629
+ int ret;
630
+ mld_assert_bound(a, offset, 0, MLDSA_Q);
631
+ if (offset == 0)
632
+ {
633
+ ret = mld_rej_uniform_native(a, target, buf, buflen);
634
+ if (ret != MLD_NATIVE_FUNC_FALLBACK)
635
+ {
636
+ unsigned res = (unsigned)ret;
637
+ mld_assert_bound(a, res, 0, MLDSA_Q);
638
+ return res;
639
+ }
640
+ }
641
+ #endif /* MLD_USE_NATIVE_REJ_UNIFORM */
642
+
643
+ return mld_rej_uniform_c(a, target, offset, buf, buflen);
644
+ }
645
+
646
+ /* Reference: poly_uniform() in the reference implementation @[REF].
647
+ * - Simplified from reference by removing buffer tail handling
648
+ * since buflen % 3 = 0 always holds true (MLD_STREAM128_BLOCKBYTES
649
+ * = 168).
650
+ * - Modified rej_uniform interface to track offset directly.
651
+ * - Pass nonce packed in the extended seed array instead of a third
652
+ * argument.
653
+ * */
654
+ MLD_INTERNAL_API
655
+ void mld_poly_uniform(mld_poly *a, const uint8_t seed[MLDSA_SEEDBYTES + 2])
656
+ {
657
+ unsigned int ctr;
658
+ unsigned int buflen = MLD_POLY_UNIFORM_NBLOCKS * MLD_STREAM128_BLOCKBYTES;
659
+ MLD_ALIGN uint8_t buf[MLD_POLY_UNIFORM_NBLOCKS * MLD_STREAM128_BLOCKBYTES];
660
+ mld_xof128_ctx state;
661
+
662
+ mld_xof128_init(&state);
663
+ mld_xof128_absorb_once(&state, seed, MLDSA_SEEDBYTES + 2);
664
+ mld_xof128_squeezeblocks(buf, MLD_POLY_UNIFORM_NBLOCKS, &state);
665
+
666
+ ctr = mld_rej_uniform(a->coeffs, MLDSA_N, 0, buf, buflen);
667
+ buflen = MLD_STREAM128_BLOCKBYTES;
668
+ while (ctr < MLDSA_N)
669
+ __loop__(
670
+ assigns(ctr, state, memory_slice(a, sizeof(mld_poly)), object_whole(buf))
671
+ invariant(ctr <= MLDSA_N)
672
+ invariant(array_bound(a->coeffs, 0, ctr, 0, MLDSA_Q))
673
+ invariant(state.pos <= SHAKE128_RATE)
674
+ )
675
+ {
676
+ mld_xof128_squeezeblocks(buf, 1, &state);
677
+ ctr = mld_rej_uniform(a->coeffs, MLDSA_N, ctr, buf, buflen);
678
+ }
679
+ mld_xof128_release(&state);
680
+ mld_assert_bound(a->coeffs, MLDSA_N, 0, MLDSA_Q);
681
+
682
+ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */
683
+ mld_zeroize(buf, sizeof(buf));
684
+ }
685
+
686
+ #if !defined(MLD_CONFIG_SERIAL_FIPS202_ONLY) && \
687
+ (!defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST))
688
+ MLD_INTERNAL_API
689
+ void mld_poly_uniform_4x(mld_poly *vec0, mld_poly *vec1, mld_poly *vec2,
690
+ mld_poly *vec3,
691
+ uint8_t seed[4][MLD_ALIGN_UP(MLDSA_SEEDBYTES + 2)])
692
+ {
693
+ /* Temporary buffers for XOF output before rejection sampling */
694
+ MLD_ALIGN uint8_t
695
+ buf[4][MLD_ALIGN_UP(MLD_POLY_UNIFORM_NBLOCKS * MLD_STREAM128_BLOCKBYTES)];
696
+
697
+ /* Tracks the number of coefficients we have already sampled */
698
+ unsigned ctr[4];
699
+ mld_xof128_x4_ctx state;
700
+ unsigned buflen;
701
+
702
+ mld_xof128_x4_init(&state);
703
+ mld_xof128_x4_absorb(&state, seed, MLDSA_SEEDBYTES + 2);
704
+
705
+ /*
706
+ * Initially, squeeze heuristic number of MLD_POLY_UNIFORM_NBLOCKS.
707
+ * This should generate the matrix entries with high probability.
708
+ */
709
+
710
+ mld_xof128_x4_squeezeblocks(buf, MLD_POLY_UNIFORM_NBLOCKS, &state);
711
+ buflen = MLD_POLY_UNIFORM_NBLOCKS * MLD_STREAM128_BLOCKBYTES;
712
+ ctr[0] = mld_rej_uniform(vec0->coeffs, MLDSA_N, 0, buf[0], buflen);
713
+ ctr[1] = mld_rej_uniform(vec1->coeffs, MLDSA_N, 0, buf[1], buflen);
714
+ ctr[2] = mld_rej_uniform(vec2->coeffs, MLDSA_N, 0, buf[2], buflen);
715
+ ctr[3] = mld_rej_uniform(vec3->coeffs, MLDSA_N, 0, buf[3], buflen);
716
+
717
+ /*
718
+ * So long as not all matrix entries have been generated, squeeze
719
+ * one more block a time until we're done.
720
+ */
721
+ buflen = MLD_STREAM128_BLOCKBYTES;
722
+ while (ctr[0] < MLDSA_N || ctr[1] < MLDSA_N || ctr[2] < MLDSA_N ||
723
+ ctr[3] < MLDSA_N)
724
+ __loop__(
725
+ assigns(ctr, state, object_whole(buf),
726
+ memory_slice(vec0, sizeof(mld_poly)), memory_slice(vec1, sizeof(mld_poly)),
727
+ memory_slice(vec2, sizeof(mld_poly)), memory_slice(vec3, sizeof(mld_poly)))
728
+ invariant(ctr[0] <= MLDSA_N && ctr[1] <= MLDSA_N)
729
+ invariant(ctr[2] <= MLDSA_N && ctr[3] <= MLDSA_N)
730
+ invariant(array_bound(vec0->coeffs, 0, ctr[0], 0, MLDSA_Q))
731
+ invariant(array_bound(vec1->coeffs, 0, ctr[1], 0, MLDSA_Q))
732
+ invariant(array_bound(vec2->coeffs, 0, ctr[2], 0, MLDSA_Q))
733
+ invariant(array_bound(vec3->coeffs, 0, ctr[3], 0, MLDSA_Q)))
734
+ {
735
+ mld_xof128_x4_squeezeblocks(buf, 1, &state);
736
+ ctr[0] = mld_rej_uniform(vec0->coeffs, MLDSA_N, ctr[0], buf[0], buflen);
737
+ ctr[1] = mld_rej_uniform(vec1->coeffs, MLDSA_N, ctr[1], buf[1], buflen);
738
+ ctr[2] = mld_rej_uniform(vec2->coeffs, MLDSA_N, ctr[2], buf[2], buflen);
739
+ ctr[3] = mld_rej_uniform(vec3->coeffs, MLDSA_N, ctr[3], buf[3], buflen);
740
+ }
741
+ mld_xof128_x4_release(&state);
742
+
743
+ mld_assert_bound(vec0->coeffs, MLDSA_N, 0, MLDSA_Q);
744
+ mld_assert_bound(vec1->coeffs, MLDSA_N, 0, MLDSA_Q);
745
+ mld_assert_bound(vec2->coeffs, MLDSA_N, 0, MLDSA_Q);
746
+ mld_assert_bound(vec3->coeffs, MLDSA_N, 0, MLDSA_Q);
747
+
748
+ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */
749
+ mld_zeroize(buf, sizeof(buf));
750
+ }
751
+
752
+ #endif /* !MLD_CONFIG_SERIAL_FIPS202_ONLY && (!MLD_CONFIG_REDUCE_RAM || \
753
+ MLD_UNIT_TEST) */
754
+
755
+ #if !defined(MLD_CONFIG_NO_KEYPAIR_API)
756
+ MLD_INTERNAL_API
757
+ void mld_polyt1_pack(uint8_t r[MLDSA_POLYT1_PACKEDBYTES], const mld_poly *a)
758
+ {
759
+ unsigned int i;
760
+ mld_assert_bound(a->coeffs, MLDSA_N, 0, 1 << 10);
761
+
762
+ for (i = 0; i < MLDSA_N / 4; ++i)
763
+ __loop__(
764
+ invariant(i <= MLDSA_N/4)
765
+ decreases(MLDSA_N / 4 - i))
766
+ {
767
+ r[5 * i + 0] = (uint8_t)((a->coeffs[4 * i + 0] >> 0) & 0xFF);
768
+ r[5 * i + 1] =
769
+ (uint8_t)(((a->coeffs[4 * i + 0] >> 8) | (a->coeffs[4 * i + 1] << 2)) &
770
+ 0xFF);
771
+ r[5 * i + 2] =
772
+ (uint8_t)(((a->coeffs[4 * i + 1] >> 6) | (a->coeffs[4 * i + 2] << 4)) &
773
+ 0xFF);
774
+ r[5 * i + 3] =
775
+ (uint8_t)(((a->coeffs[4 * i + 2] >> 4) | (a->coeffs[4 * i + 3] << 6)) &
776
+ 0xFF);
777
+ r[5 * i + 4] = (uint8_t)((a->coeffs[4 * i + 3] >> 2) & 0xFF);
778
+ }
779
+ }
780
+ #endif /* !MLD_CONFIG_NO_KEYPAIR_API */
781
+
782
+ #if !defined(MLD_CONFIG_NO_VERIFY_API)
783
+ MLD_INTERNAL_API
784
+ void mld_polyt1_unpack(mld_poly *r, const uint8_t a[MLDSA_POLYT1_PACKEDBYTES])
785
+ {
786
+ unsigned int i;
787
+
788
+ for (i = 0; i < MLDSA_N / 4; ++i)
789
+ __loop__(
790
+ invariant(i <= MLDSA_N/4)
791
+ invariant(array_bound(r->coeffs, 0, i*4, 0, 1 << 10))
792
+ decreases(MLDSA_N / 4 - i))
793
+ {
794
+ r->coeffs[4 * i + 0] =
795
+ ((a[5 * i + 0] >> 0) | ((int32_t)a[5 * i + 1] << 8)) & 0x3FF;
796
+ r->coeffs[4 * i + 1] =
797
+ ((a[5 * i + 1] >> 2) | ((int32_t)a[5 * i + 2] << 6)) & 0x3FF;
798
+ r->coeffs[4 * i + 2] =
799
+ ((a[5 * i + 2] >> 4) | ((int32_t)a[5 * i + 3] << 4)) & 0x3FF;
800
+ r->coeffs[4 * i + 3] =
801
+ ((a[5 * i + 3] >> 6) | ((int32_t)a[5 * i + 4] << 2)) & 0x3FF;
802
+ }
803
+
804
+ mld_assert_bound(r->coeffs, MLDSA_N, 0, 1 << 10);
805
+ }
806
+ #endif /* !MLD_CONFIG_NO_VERIFY_API */
807
+
808
+ #if !defined(MLD_CONFIG_NO_KEYPAIR_API)
809
+ MLD_INTERNAL_API
810
+ void mld_polyt0_pack(uint8_t r[MLDSA_POLYT0_PACKEDBYTES], const mld_poly *a)
811
+ {
812
+ unsigned int i;
813
+ uint32_t t[8];
814
+
815
+ mld_assert_bound(a->coeffs, MLDSA_N, -(1 << (MLDSA_D - 1)) + 1,
816
+ (1 << (MLDSA_D - 1)) + 1);
817
+
818
+ for (i = 0; i < MLDSA_N / 8; ++i)
819
+ __loop__(
820
+ invariant(i <= MLDSA_N/8)
821
+ decreases(MLDSA_N / 8 - i))
822
+ {
823
+ /* Safety: a->coeffs[i] <= (1 << (MLDSA_D - 1) as they are output of
824
+ * power2round, hence, these casts are safe. */
825
+ t[0] = (uint32_t)((1 << (MLDSA_D - 1)) - a->coeffs[8 * i + 0]);
826
+ t[1] = (uint32_t)((1 << (MLDSA_D - 1)) - a->coeffs[8 * i + 1]);
827
+ t[2] = (uint32_t)((1 << (MLDSA_D - 1)) - a->coeffs[8 * i + 2]);
828
+ t[3] = (uint32_t)((1 << (MLDSA_D - 1)) - a->coeffs[8 * i + 3]);
829
+ t[4] = (uint32_t)((1 << (MLDSA_D - 1)) - a->coeffs[8 * i + 4]);
830
+ t[5] = (uint32_t)((1 << (MLDSA_D - 1)) - a->coeffs[8 * i + 5]);
831
+ t[6] = (uint32_t)((1 << (MLDSA_D - 1)) - a->coeffs[8 * i + 6]);
832
+ t[7] = (uint32_t)((1 << (MLDSA_D - 1)) - a->coeffs[8 * i + 7]);
833
+
834
+ r[13 * i + 0] = (uint8_t)((t[0]) & 0xFF);
835
+ r[13 * i + 1] = (uint8_t)((t[0] >> 8) & 0xFF);
836
+ r[13 * i + 1] |= (uint8_t)((t[1] << 5) & 0xFF);
837
+ r[13 * i + 2] = (uint8_t)((t[1] >> 3) & 0xFF);
838
+ r[13 * i + 3] = (uint8_t)((t[1] >> 11) & 0xFF);
839
+ r[13 * i + 3] |= (uint8_t)((t[2] << 2) & 0xFF);
840
+ r[13 * i + 4] = (uint8_t)((t[2] >> 6) & 0xFF);
841
+ r[13 * i + 4] |= (uint8_t)((t[3] << 7) & 0xFF);
842
+ r[13 * i + 5] = (uint8_t)((t[3] >> 1) & 0xFF);
843
+ r[13 * i + 6] = (uint8_t)((t[3] >> 9) & 0xFF);
844
+ r[13 * i + 6] |= (uint8_t)((t[4] << 4) & 0xFF);
845
+ r[13 * i + 7] = (uint8_t)((t[4] >> 4) & 0xFF);
846
+ r[13 * i + 8] = (uint8_t)((t[4] >> 12) & 0xFF);
847
+ r[13 * i + 8] |= (uint8_t)((t[5] << 1) & 0xFF);
848
+ r[13 * i + 9] = (uint8_t)((t[5] >> 7) & 0xFF);
849
+ r[13 * i + 9] |= (uint8_t)((t[6] << 6) & 0xFF);
850
+ r[13 * i + 10] = (uint8_t)((t[6] >> 2) & 0xFF);
851
+ r[13 * i + 11] = (uint8_t)((t[6] >> 10) & 0xFF);
852
+ r[13 * i + 11] |= (uint8_t)((t[7] << 3) & 0xFF);
853
+ r[13 * i + 12] = (uint8_t)((t[7] >> 5) & 0xFF);
854
+ }
855
+ }
856
+ #endif /* !MLD_CONFIG_NO_KEYPAIR_API */
857
+
858
+ #if !defined(MLD_CONFIG_NO_SIGN_API) || defined(MLD_UNIT_TEST)
859
+ MLD_INTERNAL_API
860
+ void mld_polyt0_unpack(mld_poly *r, const uint8_t a[MLDSA_POLYT0_PACKEDBYTES])
861
+ {
862
+ unsigned int i;
863
+
864
+ for (i = 0; i < MLDSA_N / 8; ++i)
865
+ __loop__(
866
+ invariant(i <= MLDSA_N/8)
867
+ invariant(array_bound(r->coeffs, 0, i*8, -(1<<(MLDSA_D-1)) + 1, (1<<(MLDSA_D-1)) + 1))
868
+ decreases(MLDSA_N / 8 - i))
869
+ {
870
+ r->coeffs[8 * i + 0] = a[13 * i + 0];
871
+ r->coeffs[8 * i + 0] |= (int32_t)a[13 * i + 1] << 8;
872
+ r->coeffs[8 * i + 0] &= 0x1FFF;
873
+
874
+ r->coeffs[8 * i + 1] = a[13 * i + 1] >> 5;
875
+ r->coeffs[8 * i + 1] |= (int32_t)a[13 * i + 2] << 3;
876
+ r->coeffs[8 * i + 1] |= (int32_t)a[13 * i + 3] << 11;
877
+ r->coeffs[8 * i + 1] &= 0x1FFF;
878
+
879
+ r->coeffs[8 * i + 2] = a[13 * i + 3] >> 2;
880
+ r->coeffs[8 * i + 2] |= (int32_t)a[13 * i + 4] << 6;
881
+ r->coeffs[8 * i + 2] &= 0x1FFF;
882
+
883
+ r->coeffs[8 * i + 3] = a[13 * i + 4] >> 7;
884
+ r->coeffs[8 * i + 3] |= (int32_t)a[13 * i + 5] << 1;
885
+ r->coeffs[8 * i + 3] |= (int32_t)a[13 * i + 6] << 9;
886
+ r->coeffs[8 * i + 3] &= 0x1FFF;
887
+
888
+ r->coeffs[8 * i + 4] = a[13 * i + 6] >> 4;
889
+ r->coeffs[8 * i + 4] |= (int32_t)a[13 * i + 7] << 4;
890
+ r->coeffs[8 * i + 4] |= (int32_t)a[13 * i + 8] << 12;
891
+ r->coeffs[8 * i + 4] &= 0x1FFF;
892
+
893
+ r->coeffs[8 * i + 5] = a[13 * i + 8] >> 1;
894
+ r->coeffs[8 * i + 5] |= (int32_t)a[13 * i + 9] << 7;
895
+ r->coeffs[8 * i + 5] &= 0x1FFF;
896
+
897
+ r->coeffs[8 * i + 6] = a[13 * i + 9] >> 6;
898
+ r->coeffs[8 * i + 6] |= (int32_t)a[13 * i + 10] << 2;
899
+ r->coeffs[8 * i + 6] |= (int32_t)a[13 * i + 11] << 10;
900
+ r->coeffs[8 * i + 6] &= 0x1FFF;
901
+
902
+ r->coeffs[8 * i + 7] = a[13 * i + 11] >> 3;
903
+ r->coeffs[8 * i + 7] |= (int32_t)a[13 * i + 12] << 5;
904
+ r->coeffs[8 * i + 7] &= 0x1FFF;
905
+
906
+ r->coeffs[8 * i + 0] = (1 << (MLDSA_D - 1)) - r->coeffs[8 * i + 0];
907
+ r->coeffs[8 * i + 1] = (1 << (MLDSA_D - 1)) - r->coeffs[8 * i + 1];
908
+ r->coeffs[8 * i + 2] = (1 << (MLDSA_D - 1)) - r->coeffs[8 * i + 2];
909
+ r->coeffs[8 * i + 3] = (1 << (MLDSA_D - 1)) - r->coeffs[8 * i + 3];
910
+ r->coeffs[8 * i + 4] = (1 << (MLDSA_D - 1)) - r->coeffs[8 * i + 4];
911
+ r->coeffs[8 * i + 5] = (1 << (MLDSA_D - 1)) - r->coeffs[8 * i + 5];
912
+ r->coeffs[8 * i + 6] = (1 << (MLDSA_D - 1)) - r->coeffs[8 * i + 6];
913
+ r->coeffs[8 * i + 7] = (1 << (MLDSA_D - 1)) - r->coeffs[8 * i + 7];
914
+ }
915
+
916
+ mld_assert_bound(r->coeffs, MLDSA_N, -(1 << (MLDSA_D - 1)) + 1,
917
+ (1 << (MLDSA_D - 1)) + 1);
918
+ }
919
+ #endif /* !MLD_CONFIG_NO_SIGN_API || MLD_UNIT_TEST */
920
+
921
+ MLD_STATIC_TESTABLE uint32_t mld_poly_chknorm_c(const mld_poly *a, int32_t B)
922
+ __contract__(
923
+ requires(memory_no_alias(a, sizeof(mld_poly)))
924
+ requires(0 <= B && B <= MLDSA_Q - MLD_REDUCE32_RANGE_MAX)
925
+ requires(array_bound(a->coeffs, 0, MLDSA_N, -MLD_REDUCE32_RANGE_MAX, MLD_REDUCE32_RANGE_MAX))
926
+ ensures(return_value == 0 || return_value == 0xFFFFFFFF)
927
+ ensures((return_value == 0) == array_abs_bound(a->coeffs, 0, MLDSA_N, B))
928
+ )
929
+ {
930
+ unsigned int i;
931
+ uint32_t t = 0;
932
+ mld_assert_bound(a->coeffs, MLDSA_N, -MLD_REDUCE32_RANGE_MAX,
933
+ MLD_REDUCE32_RANGE_MAX);
934
+ for (i = 0; i < MLDSA_N; ++i)
935
+ __loop__(
936
+ invariant(i <= MLDSA_N)
937
+ invariant(t == 0 || t == 0xFFFFFFFF)
938
+ invariant((t == 0) == array_abs_bound(a->coeffs, 0, i, B))
939
+ decreases(MLDSA_N - i)
940
+ )
941
+ {
942
+ /*
943
+ * Since we know that -MLD_REDUCE32_RANGE_MAX <= a < MLD_REDUCE32_RANGE_MAX,
944
+ * and B <= MLDSA_Q - MLD_REDUCE32_RANGE_MAX, to check if
945
+ * -B < (a mod± MLDSA_Q) < B, it suffices to check if -B < a < B.
946
+ *
947
+ * We prove this to be true using the following CBMC assertions.
948
+ * a ==> b expressed as !a || b to also allow run-time assertion.
949
+ */
950
+ mld_assert(a->coeffs[i] < B || a->coeffs[i] - MLDSA_Q <= -B);
951
+ mld_assert(a->coeffs[i] > -B || a->coeffs[i] + MLDSA_Q >= B);
952
+
953
+ /* Reference: Leaks which coefficient violates the bound via a conditional.
954
+ * We are more conservative to reduce the number of declassifications in
955
+ * constant-time testing.
956
+ */
957
+
958
+ /* if (abs(a[i]) >= B) */
959
+ t |= mld_ct_cmask_neg_i32(B - 1 - mld_ct_abs_i32(a->coeffs[i]));
960
+ }
961
+
962
+ return t;
963
+ }
964
+
965
+ /* Reference: explicitly checks the bound B to be <= (MLDSA_Q - 1) / 8).
966
+ * This is unnecessary as it's always a compile-time constant.
967
+ * We instead model it as a precondition.
968
+ * Checking the bound is performed using a conditional arguing
969
+ * that it is okay to leak which coefficient violates the bound (while the
970
+ * coefficient itself must remain secret).
971
+ * We instead perform everything in constant-time.
972
+ * Also it is sufficient to check that it is smaller than
973
+ * MLDSA_Q - MLD_REDUCE32_RANGE_MAX > (MLDSA_Q - 1) / 8).
974
+ */
975
+ MLD_INTERNAL_API
976
+ uint32_t mld_poly_chknorm(const mld_poly *a, int32_t B)
977
+ {
978
+ #if defined(MLD_USE_NATIVE_POLY_CHKNORM)
979
+ int ret;
980
+ int success;
981
+ mld_assert_bound(a->coeffs, MLDSA_N, -MLD_REDUCE32_RANGE_MAX,
982
+ MLD_REDUCE32_RANGE_MAX);
983
+ /* The native backend returns 0 if all coefficients are within the bound,
984
+ * 1 if at least one coefficient exceeds the bound, and
985
+ * -1 (MLD_NATIVE_FUNC_FALLBACK) if the platform does not have the
986
+ * required capabilities to run the native function.
987
+ */
988
+ ret = mld_poly_chknorm_native(a->coeffs, B);
989
+
990
+ success = (ret != MLD_NATIVE_FUNC_FALLBACK);
991
+ /* Constant-time: It would be fine to leak the return value of chknorm
992
+ * entirely (as it is fine to leak if any coefficient exceeded the bound or
993
+ * not). However, it is cleaner to perform declassification in sign.c.
994
+ * Hence, here we only declassify if the native function returned
995
+ * MLD_NATIVE_FUNC_FALLBACK or not (which solely depends on system
996
+ * capabilities).
997
+ */
998
+ MLD_CT_TESTING_DECLASSIFY(&success, sizeof(int));
999
+ if (success)
1000
+ {
1001
+ /* Convert 0 / 1 to 0 / 0xFFFFFFFF here */
1002
+ return mld_ct_cmask_nonzero_u32((uint32_t)ret);
1003
+ }
1004
+ #endif /* MLD_USE_NATIVE_POLY_CHKNORM */
1005
+ return mld_poly_chknorm_c(a, B);
1006
+ }
1007
+
1008
+ #if !defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API)
1009
+ #if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \
1010
+ MLD_CONFIG_PARAMETER_SET == 44
1011
+ MLD_INTERNAL_API
1012
+ void mld_polyw1_pack_88(uint8_t r[MLDSA_POLYW1_PACKEDBYTES_88],
1013
+ const mld_poly *a)
1014
+ {
1015
+ unsigned int i;
1016
+
1017
+ mld_assert_bound(a->coeffs, MLDSA_N, 0,
1018
+ (MLDSA_Q - 1) / (2 * MLDSA_GAMMA2_88));
1019
+
1020
+ for (i = 0; i < MLDSA_N / 4; ++i)
1021
+ __loop__(
1022
+ invariant(i <= MLDSA_N/4)
1023
+ decreases(MLDSA_N / 4 - i))
1024
+ {
1025
+ r[3 * i + 0] = (uint8_t)((a->coeffs[4 * i + 0]) & 0xFF);
1026
+ r[3 * i + 0] |= (uint8_t)((a->coeffs[4 * i + 1] << 6) & 0xFF);
1027
+ r[3 * i + 1] = (uint8_t)((a->coeffs[4 * i + 1] >> 2) & 0xFF);
1028
+ r[3 * i + 1] |= (uint8_t)((a->coeffs[4 * i + 2] << 4) & 0xFF);
1029
+ r[3 * i + 2] = (uint8_t)((a->coeffs[4 * i + 2] >> 4) & 0xFF);
1030
+ r[3 * i + 2] |= (uint8_t)((a->coeffs[4 * i + 3] << 2) & 0xFF);
1031
+ }
1032
+ }
1033
+ #endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 44 \
1034
+ */
1035
+
1036
+ #if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \
1037
+ (MLD_CONFIG_PARAMETER_SET == 65 || MLD_CONFIG_PARAMETER_SET == 87)
1038
+ MLD_INTERNAL_API
1039
+ void mld_polyw1_pack_32(uint8_t r[MLDSA_POLYW1_PACKEDBYTES_32],
1040
+ const mld_poly *a)
1041
+ {
1042
+ unsigned int i;
1043
+
1044
+ mld_assert_bound(a->coeffs, MLDSA_N, 0,
1045
+ (MLDSA_Q - 1) / (2 * MLDSA_GAMMA2_32));
1046
+
1047
+ for (i = 0; i < MLDSA_N / 2; ++i)
1048
+ __loop__(
1049
+ invariant(i <= MLDSA_N/2)
1050
+ decreases(MLDSA_N / 2 - i))
1051
+ {
1052
+ r[i] =
1053
+ (uint8_t)((a->coeffs[2 * i + 0] | (a->coeffs[2 * i + 1] << 4)) & 0xFF);
1054
+ }
1055
+ }
1056
+ #endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 65 \
1057
+ || MLD_CONFIG_PARAMETER_SET == 87 */
1058
+ #endif /* !MLD_CONFIG_NO_SIGN_API || !MLD_CONFIG_NO_VERIFY_API */
1059
+
1060
+ #else /* !MLD_CONFIG_MULTILEVEL_NO_SHARED */
1061
+ MLD_EMPTY_CU(mld_poly)
1062
+ #endif /* MLD_CONFIG_MULTILEVEL_NO_SHARED */
1063
+
1064
+ /* To facilitate single-compilation-unit (SCU) builds, undefine all macros.
1065
+ * Don't modify by hand -- this is auto-generated by scripts/autogen. */
1066
+ #undef MLD_POLY_UNIFORM_NBLOCKS