pq_crypto 0.6.4 → 0.6.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (213) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +180 -0
  3. data/README.md +7 -0
  4. data/ext/pqcrypto/extconf.rb +4 -1
  5. data/ext/pqcrypto/pq_externalmu.c +35 -0
  6. data/ext/pqcrypto/pqcrypto_native_api.h +85 -75
  7. data/ext/pqcrypto/pqcrypto_ruby_secure.c +16 -9
  8. data/ext/pqcrypto/pqcrypto_secure.c +43 -33
  9. data/ext/pqcrypto/pqcrypto_secure.h +52 -47
  10. data/ext/pqcrypto/pqcrypto_version.h +1 -1
  11. data/ext/pqcrypto/vendor/.vendored +7 -7
  12. data/ext/pqcrypto/vendor/mldsa-native/BUILDING.md +5 -2
  13. data/ext/pqcrypto/vendor/mldsa-native/LICENSE +21 -2
  14. data/ext/pqcrypto/vendor/mldsa-native/README.md +20 -7
  15. data/ext/pqcrypto/vendor/mldsa-native/RELEASE.md +160 -0
  16. data/ext/pqcrypto/vendor/mldsa-native/SECURITY.md +1 -1
  17. data/ext/pqcrypto/vendor/mldsa-native/mldsa/README.md +2 -2
  18. data/ext/pqcrypto/vendor/mldsa-native/mldsa/mldsa_native.c +85 -59
  19. data/ext/pqcrypto/vendor/mldsa-native/mldsa/mldsa_native.h +292 -348
  20. data/ext/pqcrypto/vendor/mldsa-native/mldsa/mldsa_native_asm.S +122 -76
  21. data/ext/pqcrypto/vendor/mldsa-native/mldsa/mldsa_native_config.h +184 -86
  22. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/cbmc.h +49 -4
  23. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/common.h +49 -81
  24. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/context.h +152 -0
  25. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/ct.h +25 -12
  26. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/debug.c +2 -0
  27. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/debug.h +2 -0
  28. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/fips202x4.c +2 -2
  29. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/keccakf1600.c +9 -11
  30. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/auto.h +19 -11
  31. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x1_scalar_aarch64_asm.S +6 -4
  32. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x1_v84a_aarch64_asm.S +7 -4
  33. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x2_v84a_aarch64_asm.S +7 -4
  34. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm.S +12 -9
  35. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm.S +12 -9
  36. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/x1_scalar.h +1 -1
  37. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/x1_v84a.h +3 -2
  38. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/x2_v84a.h +3 -2
  39. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/x4_v8a_scalar.h +6 -1
  40. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/x4_v8a_v84a_scalar.h +3 -2
  41. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/api.h +11 -11
  42. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/armv81m/mve.h +9 -22
  43. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/armv81m/src/keccak_f1600_x4_mve.S +8 -5
  44. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/armv81m/src/keccak_f1600_x4_mve.c +1 -0
  45. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/armv81m/src/{state_extract_bytes_x4_mve.S → keccak_f1600_x4_state_extract_bytes_mve.S} +14 -14
  46. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/armv81m/src/{state_xor_bytes_x4_mve.S → keccak_f1600_x4_state_xor_bytes_mve.S} +12 -12
  47. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/auto.h +5 -4
  48. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/x86_64/keccak_f1600_x4_avx2.h +2 -2
  49. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/x86_64/src/fips202_native_x86_64.h +1 -0
  50. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/x86_64/src/keccak_f1600_x4_avx2_asm.S +36 -2
  51. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/meta.h +62 -4
  52. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/arith_native_aarch64.h +87 -54
  53. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{intt_aarch64_asm.S → mldsa_intt_aarch64_asm.S} +39 -6
  54. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{ntt_aarch64_asm.S → mldsa_ntt_aarch64_asm.S} +39 -6
  55. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{pointwise_montgomery_aarch64_asm.S → mldsa_pointwise_montgomery_aarch64_asm.S} +25 -3
  56. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{poly_caddq_aarch64_asm.S → mldsa_poly_caddq_aarch64_asm.S} +19 -3
  57. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{poly_chknorm_aarch64_asm.S → mldsa_poly_chknorm_aarch64_asm.S} +24 -3
  58. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{poly_decompose_32_aarch64_asm.S → mldsa_poly_decompose_32_aarch64_asm.S} +25 -3
  59. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{poly_decompose_88_aarch64_asm.S → mldsa_poly_decompose_88_aarch64_asm.S} +25 -3
  60. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{poly_use_hint_32_aarch64_asm.S → mldsa_poly_use_hint_32_aarch64_asm.S} +25 -3
  61. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{poly_use_hint_88_aarch64_asm.S → mldsa_poly_use_hint_88_aarch64_asm.S} +25 -3
  62. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{mld_polyvecl_pointwise_acc_montgomery_l4_aarch64_asm.S → mldsa_polyvecl_pointwise_acc_montgomery_l4_aarch64_asm.S} +31 -3
  63. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{mld_polyvecl_pointwise_acc_montgomery_l5_aarch64_asm.S → mldsa_polyvecl_pointwise_acc_montgomery_l5_aarch64_asm.S} +31 -3
  64. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{mld_polyvecl_pointwise_acc_montgomery_l7_aarch64_asm.S → mldsa_polyvecl_pointwise_acc_montgomery_l7_aarch64_asm.S} +31 -3
  65. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{polyz_unpack_17_aarch64_asm.S → mldsa_polyz_unpack_17_aarch64_asm.S} +31 -3
  66. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{polyz_unpack_19_aarch64_asm.S → mldsa_polyz_unpack_19_aarch64_asm.S} +31 -3
  67. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{rej_uniform_aarch64_asm.S → mldsa_rej_uniform_aarch64_asm.S} +48 -15
  68. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{rej_uniform_eta2_aarch64_asm.S → mldsa_rej_uniform_eta2_aarch64_asm.S} +42 -9
  69. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{rej_uniform_eta4_aarch64_asm.S → mldsa_rej_uniform_eta4_aarch64_asm.S} +42 -9
  70. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/api.h +11 -3
  71. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/meta.h +3 -2
  72. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/meta.h +28 -28
  73. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/arith_native_x86_64.h +171 -49
  74. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{intt_avx2_asm.S → mldsa_intt_avx2_asm.S} +23 -1
  75. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{ntt_avx2_asm.S → mldsa_ntt_avx2_asm.S} +23 -1
  76. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{nttunpack_avx2_asm.S → mldsa_nttunpack_avx2_asm.S} +17 -1
  77. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{pointwise_acc_l4_avx2_asm.S → mldsa_pointwise_acc_l4_avx2_asm.S} +37 -3
  78. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{pointwise_acc_l5_avx2_asm.S → mldsa_pointwise_acc_l5_avx2_asm.S} +37 -3
  79. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{pointwise_acc_l7_avx2_asm.S → mldsa_pointwise_acc_l7_avx2_asm.S} +37 -3
  80. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{pointwise_avx2_asm.S → mldsa_pointwise_avx2_asm.S} +31 -3
  81. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{poly_caddq_avx2_asm.S → mldsa_poly_caddq_avx2_asm.S} +18 -9
  82. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_poly_chknorm_avx2_asm.S +176 -0
  83. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_poly_decompose_32_avx2_asm.S +490 -0
  84. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_poly_decompose_88_avx2_asm.S +489 -0
  85. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_poly_use_hint_32_avx2_asm.S +123 -0
  86. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_poly_use_hint_88_avx2_asm.S +125 -0
  87. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_polyz_unpack_17_avx2_asm.S +355 -0
  88. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_polyz_unpack_19_avx2_asm.S +355 -0
  89. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_rej_uniform_avx2_asm.S +132 -0
  90. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_rej_uniform_eta2_avx2_asm.S +205 -0
  91. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_rej_uniform_eta4_avx2_asm.S +176 -0
  92. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/packing.c +27 -36
  93. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/packing.h +42 -8
  94. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/params.h +93 -17
  95. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/poly.c +74 -15
  96. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/poly.h +97 -11
  97. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/poly_kl.c +7 -38
  98. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/poly_kl.h +49 -7
  99. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/polyvec.c +16 -17
  100. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/polyvec.h +26 -9
  101. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/polyvec_lazy.c +3 -0
  102. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/polyvec_lazy.h +18 -19
  103. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/reduce.h +15 -3
  104. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/rounding.h +28 -6
  105. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/sign.c +311 -246
  106. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/sign.h +245 -240
  107. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/sys.h +64 -5
  108. data/ext/pqcrypto/vendor/mlkem-native/BUILDING.md +5 -2
  109. data/ext/pqcrypto/vendor/mlkem-native/LICENSE +21 -3
  110. data/ext/pqcrypto/vendor/mlkem-native/README.md +4 -6
  111. data/ext/pqcrypto/vendor/mlkem-native/RELEASE.md +211 -0
  112. data/ext/pqcrypto/vendor/mlkem-native/mlkem/README.md +2 -2
  113. data/ext/pqcrypto/vendor/mlkem-native/mlkem/mlkem_native.c +25 -34
  114. data/ext/pqcrypto/vendor/mlkem-native/mlkem/mlkem_native.h +79 -142
  115. data/ext/pqcrypto/vendor/mlkem-native/mlkem/mlkem_native_asm.S +62 -71
  116. data/ext/pqcrypto/vendor/mlkem-native/mlkem/mlkem_native_config.h +77 -39
  117. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/cbmc.h +25 -0
  118. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/common.h +48 -34
  119. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/compress.c +25 -5
  120. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/compress.h +14 -0
  121. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/context.h +51 -0
  122. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/fips202.h +2 -2
  123. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/keccakf1600.c +8 -8
  124. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/auto.h +20 -12
  125. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/src/keccak_f1600_x1_scalar_aarch64_asm.S +5 -3
  126. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/src/keccak_f1600_x1_v84a_aarch64_asm.S +5 -2
  127. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/src/keccak_f1600_x2_v84a_aarch64_asm.S +5 -2
  128. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/src/keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm.S +10 -7
  129. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/src/keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm.S +10 -7
  130. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x1_scalar.h +1 -1
  131. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x1_v84a.h +3 -2
  132. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x2_v84a.h +3 -2
  133. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x4_v8a_scalar.h +6 -1
  134. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x4_v8a_v84a_scalar.h +3 -2
  135. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/api.h +11 -11
  136. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/armv81m/mve.h +8 -22
  137. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/armv81m/src/keccak_f1600_x4_mve.S +8 -5
  138. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/armv81m/src/{state_extract_bytes_x4_mve.S → keccak_f1600_x4_state_extract_bytes_mve.S} +14 -14
  139. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/armv81m/src/{state_xor_bytes_x4_mve.S → keccak_f1600_x4_state_xor_bytes_mve.S} +12 -12
  140. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/x86_64/keccak_f1600_x4_avx2.h +2 -2
  141. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/x86_64/src/keccak_f1600_x4_avx2_asm.S +36 -2
  142. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/indcpa.c +22 -0
  143. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/indcpa.h +20 -11
  144. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/kem.c +39 -11
  145. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/kem.h +64 -15
  146. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/meta.h +44 -2
  147. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/aarch64_zetas.c +2 -0
  148. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/arith_native_aarch64.h +11 -10
  149. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{intt_aarch64_asm.S → mlkem_intt_aarch64_asm.S} +16 -9
  150. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{ntt_aarch64_asm.S → mlkem_ntt_aarch64_asm.S} +10 -7
  151. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{poly_mulcache_compute_aarch64_asm.S → mlkem_poly_mulcache_compute_aarch64_asm.S} +6 -3
  152. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{poly_reduce_aarch64_asm.S → mlkem_poly_reduce_aarch64_asm.S} +6 -3
  153. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{poly_tobytes_aarch64_asm.S → mlkem_poly_tobytes_aarch64_asm.S} +12 -5
  154. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{poly_tomont_aarch64_asm.S → mlkem_poly_tomont_aarch64_asm.S} +11 -5
  155. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{polyvec_basemul_acc_montgomery_cached_k2_aarch64_asm.S → mlkem_polyvec_basemul_acc_montgomery_cached_k2_aarch64_asm.S} +6 -3
  156. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{polyvec_basemul_acc_montgomery_cached_k3_aarch64_asm.S → mlkem_polyvec_basemul_acc_montgomery_cached_k3_aarch64_asm.S} +6 -3
  157. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{polyvec_basemul_acc_montgomery_cached_k4_aarch64_asm.S → mlkem_polyvec_basemul_acc_montgomery_cached_k4_aarch64_asm.S} +6 -3
  158. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{rej_uniform_aarch64_asm.S → mlkem_rej_uniform_aarch64_asm.S} +17 -17
  159. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/api.h +21 -8
  160. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/meta.h +1 -1
  161. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/ppc64le/meta.h +4 -0
  162. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/ppc64le/src/{intt_ppc_asm.S → mlkem_intt_ppc_asm.S} +29 -5
  163. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/ppc64le/src/{ntt_ppc_asm.S → mlkem_ntt_ppc_asm.S} +22 -1
  164. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/ppc64le/src/{poly_tomont_ppc_asm.S → mlkem_poly_tomont_ppc_asm.S} +25 -3
  165. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/ppc64le/src/{reduce_ppc_asm.S → mlkem_reduce_ppc_asm.S} +22 -1
  166. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/riscv64/meta.h +4 -0
  167. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/riscv64/src/arith_native_riscv64.h +4 -0
  168. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/riscv64/src/rv64v_poly.c +6 -2
  169. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/meta.h +45 -25
  170. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/arith_native_x86_64.h +20 -20
  171. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/compress_consts.c +14 -3
  172. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/compress_consts.h +8 -0
  173. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{intt_avx2_asm.S → mlkem_intt_avx2_asm.S} +27 -3
  174. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{ntt_avx2_asm.S → mlkem_ntt_avx2_asm.S} +23 -1
  175. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{nttfrombytes_avx2_asm.S → mlkem_nttfrombytes_avx2_asm.S} +27 -3
  176. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{ntttobytes_avx2_asm.S → mlkem_ntttobytes_avx2_asm.S} +27 -3
  177. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{nttunpack_avx2_asm.S → mlkem_nttunpack_avx2_asm.S} +17 -1
  178. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{poly_compress_d10_avx2_asm.S → mlkem_poly_compress_d10_avx2_asm.S} +34 -3
  179. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{poly_compress_d11_avx2_asm.S → mlkem_poly_compress_d11_avx2_asm.S} +33 -2
  180. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{poly_compress_d4_avx2_asm.S → mlkem_poly_compress_d4_avx2_asm.S} +34 -3
  181. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{poly_compress_d5_avx2_asm.S → mlkem_poly_compress_d5_avx2_asm.S} +33 -2
  182. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{poly_decompress_d10_avx2_asm.S → mlkem_poly_decompress_d10_avx2_asm.S} +32 -3
  183. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{poly_decompress_d11_avx2_asm.S → mlkem_poly_decompress_d11_avx2_asm.S} +32 -2
  184. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{poly_decompress_d4_avx2_asm.S → mlkem_poly_decompress_d4_avx2_asm.S} +32 -3
  185. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{poly_decompress_d5_avx2_asm.S → mlkem_poly_decompress_d5_avx2_asm.S} +32 -2
  186. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{poly_mulcache_compute_avx2_asm.S → mlkem_poly_mulcache_compute_avx2_asm.S} +29 -1
  187. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{polyvec_basemul_acc_montgomery_cached_k2_avx2_asm.S → mlkem_polyvec_basemul_acc_montgomery_cached_k2_avx2_asm.S} +35 -1
  188. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{polyvec_basemul_acc_montgomery_cached_k3_avx2_asm.S → mlkem_polyvec_basemul_acc_montgomery_cached_k3_avx2_asm.S} +35 -1
  189. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{polyvec_basemul_acc_montgomery_cached_k4_avx2_asm.S → mlkem_polyvec_basemul_acc_montgomery_cached_k4_avx2_asm.S} +35 -1
  190. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{reduce_avx2_asm.S → mlkem_reduce_avx2_asm.S} +17 -1
  191. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{rej_uniform_avx2_asm.S → mlkem_rej_uniform_avx2_asm.S} +40 -7
  192. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{tomont_avx2_asm.S → mlkem_tomont_avx2_asm.S} +20 -3
  193. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/poly.c +7 -2
  194. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/poly.h +6 -0
  195. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/poly_k.c +25 -4
  196. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/poly_k.h +33 -4
  197. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/sampling.c +4 -0
  198. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/sampling.h +4 -0
  199. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/sys.h +21 -3
  200. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/verify.h +11 -10
  201. data/lib/pq_crypto/version.rb +1 -1
  202. data/script/vendor_libs.rb +6 -6
  203. metadata +79 -79
  204. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/poly_chknorm_avx2.c +0 -52
  205. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/poly_decompose_32_avx2.c +0 -157
  206. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/poly_decompose_88_avx2.c +0 -157
  207. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/poly_use_hint_32_avx2.c +0 -103
  208. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/poly_use_hint_88_avx2.c +0 -105
  209. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/polyz_unpack_17_avx2.c +0 -94
  210. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/polyz_unpack_19_avx2.c +0 -96
  211. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/rej_uniform_avx2.c +0 -126
  212. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/rej_uniform_eta2_avx2.c +0 -157
  213. data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/rej_uniform_eta4_avx2.c +0 -141
@@ -14,20 +14,53 @@
14
14
  * Becker, Hwang, Kannwischer, Yang, Yang
15
15
  * https://eprint.iacr.org/2021/986
16
16
  *
17
+ * - [NeonNTT_Autoformalised]
18
+ * Neon NTT - (Auto)formalised
19
+ * Hanno Becker
20
+ * https://eprint.iacr.org/2026/1223
21
+ *
17
22
  * - [SLOTHY_Paper]
18
23
  * Fast and Clean: Auditable high-performance assembly via constraint solving
19
24
  * Abdulrahman, Becker, Kannwischer, Klein
20
25
  * https://eprint.iacr.org/2022/1303
21
26
  */
22
27
 
23
- /* AArch64 ML-DSA forward NTT following @[NeonNTT] and @[SLOTHY_Paper] */
28
+ /* AArch64 ML-DSA forward NTT following @[NeonNTT], @[SLOTHY_Paper], and @[NeonNTT_Autoformalised] */
29
+
30
+ /*yaml
31
+ Name: ntt_aarch64_asm
32
+ Description: AArch64 ML-DSA forward NTT
33
+ Signature: void mld_ntt_aarch64_asm(int32_t r[256], const int32_t zetas_l123456[144], const int32_t zetas_l78[384])
34
+ ABI:
35
+ Architecture: aarch64
36
+ CallingConvention: AAPCS64
37
+ Features: [NEON]
38
+ x0:
39
+ type: buffer
40
+ size_bytes: 1024
41
+ permissions: read/write
42
+ c_parameter: int32_t r[256]
43
+ description: Input/output polynomial (256 x int32_t)
44
+ x1:
45
+ type: buffer
46
+ size_bytes: 576
47
+ permissions: read-only
48
+ c_parameter: const int32_t zetas_l123456[144]
49
+ description: Twiddle factors for layers 1-6 (144 x int32_t)
50
+ x2:
51
+ type: buffer
52
+ size_bytes: 1536
53
+ permissions: read-only
54
+ c_parameter: const int32_t zetas_l78[384]
55
+ description: Twiddle factors for layers 7-8 (384 x int32_t)
56
+ */
24
57
 
25
58
  #include "../../../common.h"
26
59
  #if defined(MLD_ARITH_BACKEND_AARCH64) && !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)
27
60
 
28
61
  /*
29
62
  * WARNING: This file is auto-derived from the mldsa-native source file
30
- * dev/aarch64_opt/src/ntt_aarch64_asm.S using scripts/simpasm. Do not modify it directly.
63
+ * dev/aarch64_opt/src/mldsa_ntt_aarch64_asm.S using scripts/simpasm. Do not modify it directly.
31
64
  */
32
65
 
33
66
  .text
@@ -138,7 +171,7 @@ MLD_ASM_FN_SYMBOL(ntt_aarch64_asm)
138
171
  sub v15.4s, v10.4s, v26.4s
139
172
  sub x4, x4, #0x2
140
173
 
141
- Lntt_layer123_start:
174
+ Lmld_ntt_layer123_start:
142
175
  add v31.4s, v10.4s, v26.4s
143
176
  mul v17.4s, v19.4s, v1.s[2]
144
177
  add v26.4s, v15.4s, v23.4s
@@ -216,7 +249,7 @@ Lntt_layer123_start:
216
249
  sqrdmulh v12.4s, v19.4s, v1.s[3]
217
250
  sub v15.4s, v10.4s, v26.4s
218
251
  subs x4, x4, #0x1
219
- cbnz x4, Lntt_layer123_start
252
+ cbnz x4, Lmld_ntt_layer123_start
220
253
  add v13.4s, v10.4s, v26.4s
221
254
  mls v18.4s, v5.4s, v7.s[0]
222
255
  str q22, [x0, #0x180]
@@ -378,7 +411,7 @@ Lntt_layer123_start:
378
411
  sub v31.4s, v6.4s, v9.4s
379
412
  sub x4, x4, #0x1
380
413
 
381
- Lntt_layer45678_start:
414
+ Lmld_ntt_layer45678_start:
382
415
  add v2.4s, v13.4s, v12.4s
383
416
  sqrdmulh v5.4s, v30.4s, v20.4s
384
417
  sub v25.4s, v13.4s, v12.4s
@@ -544,7 +577,7 @@ Lntt_layer45678_start:
544
577
  trn2 v10.2d, v22.2d, v1.2d
545
578
  mul v28.4s, v30.4s, v3.4s
546
579
  subs x4, x4, #0x1
547
- cbnz x4, Lntt_layer45678_start
580
+ cbnz x4, Lmld_ntt_layer45678_start
548
581
  add v9.4s, v6.4s, v9.4s
549
582
  sqrdmulh v6.4s, v30.4s, v20.4s
550
583
  ldur q24, [x2, #-0xa0]
@@ -2,6 +2,28 @@
2
2
  * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT
3
3
  */
4
4
 
5
+ /*yaml
6
+ Name: poly_pointwise_montgomery_aarch64_asm
7
+ Description: AArch64 pointwise Montgomery multiplication of two polynomials
8
+ Signature: void mld_poly_pointwise_montgomery_aarch64_asm(int32_t a[256], const int32_t b[256])
9
+ ABI:
10
+ Architecture: aarch64
11
+ CallingConvention: AAPCS64
12
+ Features: [NEON]
13
+ x0:
14
+ type: buffer
15
+ size_bytes: 1024
16
+ permissions: read/write
17
+ c_parameter: int32_t a[256]
18
+ description: Input/output polynomial (256 x int32_t)
19
+ x1:
20
+ type: buffer
21
+ size_bytes: 1024
22
+ permissions: read-only
23
+ c_parameter: const int32_t b[256]
24
+ description: Input polynomial (256 x int32_t)
25
+ */
26
+
5
27
  #include "../../../common.h"
6
28
  #if defined(MLD_ARITH_BACKEND_AARCH64) && \
7
29
  (!defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API) || \
@@ -10,7 +32,7 @@
10
32
 
11
33
  /*
12
34
  * WARNING: This file is auto-derived from the mldsa-native source file
13
- * dev/aarch64_opt/src/pointwise_montgomery_aarch64_asm.S using scripts/simpasm. Do not modify it directly.
35
+ * dev/aarch64_opt/src/mldsa_pointwise_montgomery_aarch64_asm.S using scripts/simpasm. Do not modify it directly.
14
36
  */
15
37
 
16
38
  .text
@@ -27,7 +49,7 @@ MLD_ASM_FN_SYMBOL(poly_pointwise_montgomery_aarch64_asm)
27
49
  dup v1.4s, w3
28
50
  mov x3, #0x40 // =64
29
51
 
30
- Lpoly_pointwise_montgomery_loop_start:
52
+ Lmld_poly_pointwise_montgomery_loop_start:
31
53
  ldr q16, [x0]
32
54
  ldr q17, [x0, #0x10]
33
55
  ldr q18, [x0, #0x20]
@@ -69,7 +91,7 @@ Lpoly_pointwise_montgomery_loop_start:
69
91
  str q19, [x0, #0x30]
70
92
  str q16, [x0], #0x40
71
93
  subs x3, x3, #0x4
72
- cbnz x3, Lpoly_pointwise_montgomery_loop_start
94
+ cbnz x3, Lmld_poly_pointwise_montgomery_loop_start
73
95
  ret
74
96
  .cfi_endproc
75
97
 
@@ -2,13 +2,29 @@
2
2
  * Copyright (c) The mldsa-native project authors
3
3
  * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT
4
4
  */
5
+ /*yaml
6
+ Name: poly_caddq_aarch64_asm
7
+ Description: AArch64 conditional addition of q to each coefficient
8
+ Signature: void mld_poly_caddq_aarch64_asm(int32_t a[256])
9
+ ABI:
10
+ Architecture: aarch64
11
+ CallingConvention: AAPCS64
12
+ Features: [NEON]
13
+ x0:
14
+ type: buffer
15
+ size_bytes: 1024
16
+ permissions: read/write
17
+ c_parameter: int32_t a[256]
18
+ description: Input/output polynomial (256 x int32_t)
19
+ */
20
+
5
21
  #include "../../../common.h"
6
22
 
7
23
  #if defined(MLD_ARITH_BACKEND_AARCH64) && !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)
8
24
 
9
25
  /*
10
26
  * WARNING: This file is auto-derived from the mldsa-native source file
11
- * dev/aarch64_opt/src/poly_caddq_aarch64_asm.S using scripts/simpasm. Do not modify it directly.
27
+ * dev/aarch64_opt/src/mldsa_poly_caddq_aarch64_asm.S using scripts/simpasm. Do not modify it directly.
12
28
  */
13
29
 
14
30
  .text
@@ -22,7 +38,7 @@ MLD_ASM_FN_SYMBOL(poly_caddq_aarch64_asm)
22
38
  dup v4.4s, w9
23
39
  mov x1, #0x10 // =16
24
40
 
25
- Lpoly_caddq_loop:
41
+ Lmld_poly_caddq_loop:
26
42
  ldr q0, [x0]
27
43
  ldr q1, [x0, #0x10]
28
44
  ldr q2, [x0, #0x20]
@@ -40,7 +56,7 @@ Lpoly_caddq_loop:
40
56
  str q3, [x0, #0x30]
41
57
  str q0, [x0], #0x40
42
58
  subs x1, x1, #0x1
43
- b.ne Lpoly_caddq_loop
59
+ b.ne Lmld_poly_caddq_loop
44
60
  ret
45
61
  .cfi_endproc
46
62
 
@@ -2,13 +2,34 @@
2
2
  * Copyright (c) The mldsa-native project authors
3
3
  * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT
4
4
  */
5
+ /*yaml
6
+ Name: poly_chknorm_aarch64_asm
7
+ Description: AArch64 infinity-norm bound check on polynomial coefficients
8
+ Signature: int mld_poly_chknorm_aarch64_asm(const int32_t a[256], int32_t B)
9
+ ABI:
10
+ Architecture: aarch64
11
+ CallingConvention: AAPCS64
12
+ Features: [NEON]
13
+ x0:
14
+ type: buffer
15
+ size_bytes: 1024
16
+ permissions: read-only
17
+ c_parameter: const int32_t a[256]
18
+ description: Input polynomial (256 x int32_t)
19
+ x1:
20
+ type: scalar
21
+ c_parameter: int32_t B
22
+ description: Norm bound
23
+ test_with: 131072 # representative non-negative bound (1 << 17)
24
+ */
25
+
5
26
  #include "../../../common.h"
6
27
 
7
28
  #if defined(MLD_ARITH_BACKEND_AARCH64) && !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)
8
29
 
9
30
  /*
10
31
  * WARNING: This file is auto-derived from the mldsa-native source file
11
- * dev/aarch64_opt/src/poly_chknorm_aarch64_asm.S using scripts/simpasm. Do not modify it directly.
32
+ * dev/aarch64_opt/src/mldsa_poly_chknorm_aarch64_asm.S using scripts/simpasm. Do not modify it directly.
12
33
  */
13
34
 
14
35
  .text
@@ -21,7 +42,7 @@ MLD_ASM_FN_SYMBOL(poly_chknorm_aarch64_asm)
21
42
  eor v21.16b, v21.16b, v21.16b
22
43
  mov x2, #0x10 // =16
23
44
 
24
- Lpoly_chknorm_loop:
45
+ Lmld_poly_chknorm_loop:
25
46
  ldr q1, [x0, #0x10]
26
47
  ldr q2, [x0, #0x20]
27
48
  ldr q3, [x0, #0x30]
@@ -39,7 +60,7 @@ Lpoly_chknorm_loop:
39
60
  cmge v0.4s, v0.4s, v20.4s
40
61
  orr v21.16b, v21.16b, v0.16b
41
62
  subs x2, x2, #0x1
42
- b.ne Lpoly_chknorm_loop
63
+ b.ne Lmld_poly_chknorm_loop
43
64
  umaxv s21, v21.4s
44
65
  fmov w0, s21
45
66
  and w0, w0, #0x1
@@ -2,6 +2,28 @@
2
2
  * Copyright (c) The mldsa-native project authors
3
3
  * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT
4
4
  */
5
+ /*yaml
6
+ Name: poly_decompose_32_aarch64_asm
7
+ Description: AArch64 coefficient decomposition (alpha = (Q-1)/32)
8
+ Signature: void mld_poly_decompose_32_aarch64_asm(int32_t a1[256], int32_t a0[256])
9
+ ABI:
10
+ Architecture: aarch64
11
+ CallingConvention: AAPCS64
12
+ Features: [NEON]
13
+ x0:
14
+ type: buffer
15
+ size_bytes: 1024
16
+ permissions: write-only
17
+ c_parameter: int32_t a1[256]
18
+ description: Output high-part polynomial (256 x int32_t)
19
+ x1:
20
+ type: buffer
21
+ size_bytes: 1024
22
+ permissions: read/write
23
+ c_parameter: int32_t a0[256]
24
+ description: Input polynomial / output low-part (256 x int32_t)
25
+ */
26
+
5
27
  #include "../../../common.h"
6
28
 
7
29
  #if defined(MLD_ARITH_BACKEND_AARCH64) && !defined(MLD_CONFIG_NO_SIGN_API) && !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \
@@ -9,7 +31,7 @@
9
31
 
10
32
  /*
11
33
  * WARNING: This file is auto-derived from the mldsa-native source file
12
- * dev/aarch64_opt/src/poly_decompose_32_aarch64_asm.S using scripts/simpasm. Do not modify it directly.
34
+ * dev/aarch64_opt/src/mldsa_poly_decompose_32_aarch64_asm.S using scripts/simpasm. Do not modify it directly.
13
35
  */
14
36
 
15
37
  .text
@@ -32,7 +54,7 @@ MLD_ASM_FN_SYMBOL(poly_decompose_32_aarch64_asm)
32
54
  dup v23.4s, w11
33
55
  mov x3, #0x10 // =16
34
56
 
35
- Lpoly_decompose_32_loop:
57
+ Lmld_poly_decompose_32_loop:
36
58
  ldr q0, [x1]
37
59
  ldr q1, [x1, #0x10]
38
60
  ldr q2, [x1, #0x20]
@@ -70,7 +92,7 @@ Lpoly_decompose_32_loop:
70
92
  str q3, [x1, #0x30]
71
93
  str q0, [x1], #0x40
72
94
  subs x3, x3, #0x1
73
- b.ne Lpoly_decompose_32_loop
95
+ b.ne Lmld_poly_decompose_32_loop
74
96
  ret
75
97
  .cfi_endproc
76
98
 
@@ -2,6 +2,28 @@
2
2
  * Copyright (c) The mldsa-native project authors
3
3
  * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT
4
4
  */
5
+ /*yaml
6
+ Name: poly_decompose_88_aarch64_asm
7
+ Description: AArch64 coefficient decomposition (alpha = (Q-1)/88)
8
+ Signature: void mld_poly_decompose_88_aarch64_asm(int32_t a1[256], int32_t a0[256])
9
+ ABI:
10
+ Architecture: aarch64
11
+ CallingConvention: AAPCS64
12
+ Features: [NEON]
13
+ x0:
14
+ type: buffer
15
+ size_bytes: 1024
16
+ permissions: write-only
17
+ c_parameter: int32_t a1[256]
18
+ description: Output high-part polynomial (256 x int32_t)
19
+ x1:
20
+ type: buffer
21
+ size_bytes: 1024
22
+ permissions: read/write
23
+ c_parameter: int32_t a0[256]
24
+ description: Input polynomial / output low-part (256 x int32_t)
25
+ */
26
+
5
27
  #include "../../../common.h"
6
28
 
7
29
  #if defined(MLD_ARITH_BACKEND_AARCH64) && !defined(MLD_CONFIG_NO_SIGN_API) && !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \
@@ -9,7 +31,7 @@
9
31
 
10
32
  /*
11
33
  * WARNING: This file is auto-derived from the mldsa-native source file
12
- * dev/aarch64_opt/src/poly_decompose_88_aarch64_asm.S using scripts/simpasm. Do not modify it directly.
34
+ * dev/aarch64_opt/src/mldsa_poly_decompose_88_aarch64_asm.S using scripts/simpasm. Do not modify it directly.
13
35
  */
14
36
 
15
37
  .text
@@ -32,7 +54,7 @@ MLD_ASM_FN_SYMBOL(poly_decompose_88_aarch64_asm)
32
54
  dup v23.4s, w11
33
55
  mov x3, #0x10 // =16
34
56
 
35
- Lpoly_decompose_88_loop:
57
+ Lmld_poly_decompose_88_loop:
36
58
  ldr q0, [x1]
37
59
  ldr q1, [x1, #0x10]
38
60
  ldr q2, [x1, #0x20]
@@ -70,7 +92,7 @@ Lpoly_decompose_88_loop:
70
92
  str q3, [x1, #0x30]
71
93
  str q0, [x1], #0x40
72
94
  subs x3, x3, #0x1
73
- b.ne Lpoly_decompose_88_loop
95
+ b.ne Lmld_poly_decompose_88_loop
74
96
  ret
75
97
  .cfi_endproc
76
98
 
@@ -2,6 +2,28 @@
2
2
  * Copyright (c) The mldsa-native project authors
3
3
  * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT
4
4
  */
5
+ /*yaml
6
+ Name: poly_use_hint_32_aarch64_asm
7
+ Description: AArch64 hint application (alpha = (Q-1)/32)
8
+ Signature: void mld_poly_use_hint_32_aarch64_asm(int32_t a[256], const int32_t h[256])
9
+ ABI:
10
+ Architecture: aarch64
11
+ CallingConvention: AAPCS64
12
+ Features: [NEON]
13
+ x0:
14
+ type: buffer
15
+ size_bytes: 1024
16
+ permissions: read/write
17
+ c_parameter: int32_t a[256]
18
+ description: Input/output polynomial (256 x int32_t)
19
+ x1:
20
+ type: buffer
21
+ size_bytes: 1024
22
+ permissions: read-only
23
+ c_parameter: const int32_t h[256]
24
+ description: Hint polynomial (256 x int32_t)
25
+ */
26
+
5
27
  #include "../../../common.h"
6
28
 
7
29
  #if defined(MLD_ARITH_BACKEND_AARCH64) && !defined(MLD_CONFIG_NO_VERIFY_API) && !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \
@@ -9,7 +31,7 @@
9
31
 
10
32
  /*
11
33
  * WARNING: This file is auto-derived from the mldsa-native source file
12
- * dev/aarch64_opt/src/poly_use_hint_32_aarch64_asm.S using scripts/simpasm. Do not modify it directly.
34
+ * dev/aarch64_opt/src/mldsa_poly_use_hint_32_aarch64_asm.S using scripts/simpasm. Do not modify it directly.
13
35
  */
14
36
 
15
37
  .text
@@ -33,7 +55,7 @@ MLD_ASM_FN_SYMBOL(poly_use_hint_32_aarch64_asm)
33
55
  movi v24.4s, #0xf
34
56
  mov x3, #0x10 // =16
35
57
 
36
- Lpoly_use_hint_32_loop:
58
+ Lmld_poly_use_hint_32_loop:
37
59
  ldr q1, [x0, #0x10]
38
60
  ldr q2, [x0, #0x20]
39
61
  ldr q3, [x0, #0x30]
@@ -87,7 +109,7 @@ Lpoly_use_hint_32_loop:
87
109
  str q19, [x0, #0x30]
88
110
  str q16, [x0], #0x40
89
111
  subs x3, x3, #0x1
90
- b.ne Lpoly_use_hint_32_loop
112
+ b.ne Lmld_poly_use_hint_32_loop
91
113
  ret
92
114
  .cfi_endproc
93
115
 
@@ -2,6 +2,28 @@
2
2
  * Copyright (c) The mldsa-native project authors
3
3
  * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT
4
4
  */
5
+ /*yaml
6
+ Name: poly_use_hint_88_aarch64_asm
7
+ Description: AArch64 hint application (alpha = (Q-1)/88)
8
+ Signature: void mld_poly_use_hint_88_aarch64_asm(int32_t a[256], const int32_t h[256])
9
+ ABI:
10
+ Architecture: aarch64
11
+ CallingConvention: AAPCS64
12
+ Features: [NEON]
13
+ x0:
14
+ type: buffer
15
+ size_bytes: 1024
16
+ permissions: read/write
17
+ c_parameter: int32_t a[256]
18
+ description: Input/output polynomial (256 x int32_t)
19
+ x1:
20
+ type: buffer
21
+ size_bytes: 1024
22
+ permissions: read-only
23
+ c_parameter: const int32_t h[256]
24
+ description: Hint polynomial (256 x int32_t)
25
+ */
26
+
5
27
  #include "../../../common.h"
6
28
 
7
29
  #if defined(MLD_ARITH_BACKEND_AARCH64) && !defined(MLD_CONFIG_NO_VERIFY_API) && !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \
@@ -9,7 +31,7 @@
9
31
 
10
32
  /*
11
33
  * WARNING: This file is auto-derived from the mldsa-native source file
12
- * dev/aarch64_opt/src/poly_use_hint_88_aarch64_asm.S using scripts/simpasm. Do not modify it directly.
34
+ * dev/aarch64_opt/src/mldsa_poly_use_hint_88_aarch64_asm.S using scripts/simpasm. Do not modify it directly.
13
35
  */
14
36
 
15
37
  .text
@@ -33,7 +55,7 @@ MLD_ASM_FN_SYMBOL(poly_use_hint_88_aarch64_asm)
33
55
  movi v24.4s, #0x2b
34
56
  mov x3, #0x10 // =16
35
57
 
36
- Lpoly_use_hint_88_loop:
58
+ Lmld_poly_use_hint_88_loop:
37
59
  ldr q1, [x0, #0x10]
38
60
  ldr q2, [x0, #0x20]
39
61
  ldr q3, [x0, #0x30]
@@ -95,7 +117,7 @@ Lpoly_use_hint_88_loop:
95
117
  str q19, [x0, #0x30]
96
118
  str q16, [x0], #0x40
97
119
  subs x3, x3, #0x1
98
- b.ne Lpoly_use_hint_88_loop
120
+ b.ne Lmld_poly_use_hint_88_loop
99
121
  ret
100
122
  .cfi_endproc
101
123
 
@@ -2,13 +2,41 @@
2
2
  * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT
3
3
  */
4
4
 
5
+ /*yaml
6
+ Name: polyvecl_pointwise_acc_montgomery_l4_aarch64_asm
7
+ Description: AArch64 pointwise multiply-accumulate of length-4 polynomial vectors
8
+ Signature: void mld_polyvecl_pointwise_acc_montgomery_l4_aarch64_asm(int32_t r[256], const int32_t a[4][256], const int32_t b[4][256])
9
+ ABI:
10
+ Architecture: aarch64
11
+ CallingConvention: AAPCS64
12
+ Features: [NEON]
13
+ x0:
14
+ type: buffer
15
+ size_bytes: 1024
16
+ permissions: write-only
17
+ c_parameter: int32_t r[256]
18
+ description: Output polynomial (256 x int32_t)
19
+ x1:
20
+ type: buffer
21
+ size_bytes: 4096
22
+ permissions: read-only
23
+ c_parameter: const int32_t a[4][256]
24
+ description: Input polynomial vector a (4 x 256 x int32_t)
25
+ x2:
26
+ type: buffer
27
+ size_bytes: 4096
28
+ permissions: read-only
29
+ c_parameter: const int32_t b[4][256]
30
+ description: Input polynomial vector b (4 x 256 x int32_t)
31
+ */
32
+
5
33
  #include "../../../common.h"
6
34
  #if defined(MLD_ARITH_BACKEND_AARCH64) && !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \
7
35
  (defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLDSA_L == 4)
8
36
 
9
37
  /*
10
38
  * WARNING: This file is auto-derived from the mldsa-native source file
11
- * dev/aarch64_opt/src/mld_polyvecl_pointwise_acc_montgomery_l4_aarch64_asm.S using scripts/simpasm. Do not modify it directly.
39
+ * dev/aarch64_opt/src/mldsa_polyvecl_pointwise_acc_montgomery_l4_aarch64_asm.S using scripts/simpasm. Do not modify it directly.
12
40
  */
13
41
 
14
42
  .text
@@ -25,7 +53,7 @@ MLD_ASM_FN_SYMBOL(polyvecl_pointwise_acc_montgomery_l4_aarch64_asm)
25
53
  dup v1.4s, w3
26
54
  mov x3, #0x40 // =64
27
55
 
28
- Lpolyvecl_pointwise_acc_montgomery_l4_loop_start:
56
+ Lmld_polyvecl_pointwise_acc_montgomery_l4_loop_start:
29
57
  ldr q17, [x1, #0x10]
30
58
  ldr q18, [x1, #0x20]
31
59
  ldr q19, [x1, #0x30]
@@ -115,7 +143,7 @@ Lpolyvecl_pointwise_acc_montgomery_l4_loop_start:
115
143
  str q19, [x0, #0x30]
116
144
  str q16, [x0], #0x40
117
145
  subs x3, x3, #0x4
118
- cbnz x3, Lpolyvecl_pointwise_acc_montgomery_l4_loop_start
146
+ cbnz x3, Lmld_polyvecl_pointwise_acc_montgomery_l4_loop_start
119
147
  ret
120
148
  .cfi_endproc
121
149
 
@@ -2,13 +2,41 @@
2
2
  * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT
3
3
  */
4
4
 
5
+ /*yaml
6
+ Name: polyvecl_pointwise_acc_montgomery_l5_aarch64_asm
7
+ Description: AArch64 pointwise multiply-accumulate of length-5 polynomial vectors
8
+ Signature: void mld_polyvecl_pointwise_acc_montgomery_l5_aarch64_asm(int32_t r[256], const int32_t a[5][256], const int32_t b[5][256])
9
+ ABI:
10
+ Architecture: aarch64
11
+ CallingConvention: AAPCS64
12
+ Features: [NEON]
13
+ x0:
14
+ type: buffer
15
+ size_bytes: 1024
16
+ permissions: write-only
17
+ c_parameter: int32_t r[256]
18
+ description: Output polynomial (256 x int32_t)
19
+ x1:
20
+ type: buffer
21
+ size_bytes: 5120
22
+ permissions: read-only
23
+ c_parameter: const int32_t a[5][256]
24
+ description: Input polynomial vector a (5 x 256 x int32_t)
25
+ x2:
26
+ type: buffer
27
+ size_bytes: 5120
28
+ permissions: read-only
29
+ c_parameter: const int32_t b[5][256]
30
+ description: Input polynomial vector b (5 x 256 x int32_t)
31
+ */
32
+
5
33
  #include "../../../common.h"
6
34
  #if defined(MLD_ARITH_BACKEND_AARCH64) && !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \
7
35
  (defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLDSA_L == 5)
8
36
 
9
37
  /*
10
38
  * WARNING: This file is auto-derived from the mldsa-native source file
11
- * dev/aarch64_opt/src/mld_polyvecl_pointwise_acc_montgomery_l5_aarch64_asm.S using scripts/simpasm. Do not modify it directly.
39
+ * dev/aarch64_opt/src/mldsa_polyvecl_pointwise_acc_montgomery_l5_aarch64_asm.S using scripts/simpasm. Do not modify it directly.
12
40
  */
13
41
 
14
42
  .text
@@ -25,7 +53,7 @@ MLD_ASM_FN_SYMBOL(polyvecl_pointwise_acc_montgomery_l5_aarch64_asm)
25
53
  dup v1.4s, w3
26
54
  mov x3, #0x40 // =64
27
55
 
28
- Lpolyvecl_pointwise_acc_montgomery_l5_loop_start:
56
+ Lmld_polyvecl_pointwise_acc_montgomery_l5_loop_start:
29
57
  ldr q17, [x1, #0x10]
30
58
  ldr q18, [x1, #0x20]
31
59
  ldr q19, [x1, #0x30]
@@ -131,7 +159,7 @@ Lpolyvecl_pointwise_acc_montgomery_l5_loop_start:
131
159
  str q19, [x0, #0x30]
132
160
  str q16, [x0], #0x40
133
161
  subs x3, x3, #0x4
134
- cbnz x3, Lpolyvecl_pointwise_acc_montgomery_l5_loop_start
162
+ cbnz x3, Lmld_polyvecl_pointwise_acc_montgomery_l5_loop_start
135
163
  ret
136
164
  .cfi_endproc
137
165
 
@@ -2,13 +2,41 @@
2
2
  * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT
3
3
  */
4
4
 
5
+ /*yaml
6
+ Name: polyvecl_pointwise_acc_montgomery_l7_aarch64_asm
7
+ Description: AArch64 pointwise multiply-accumulate of length-7 polynomial vectors
8
+ Signature: void mld_polyvecl_pointwise_acc_montgomery_l7_aarch64_asm(int32_t r[256], const int32_t a[7][256], const int32_t b[7][256])
9
+ ABI:
10
+ Architecture: aarch64
11
+ CallingConvention: AAPCS64
12
+ Features: [NEON]
13
+ x0:
14
+ type: buffer
15
+ size_bytes: 1024
16
+ permissions: write-only
17
+ c_parameter: int32_t r[256]
18
+ description: Output polynomial (256 x int32_t)
19
+ x1:
20
+ type: buffer
21
+ size_bytes: 7168
22
+ permissions: read-only
23
+ c_parameter: const int32_t a[7][256]
24
+ description: Input polynomial vector a (7 x 256 x int32_t)
25
+ x2:
26
+ type: buffer
27
+ size_bytes: 7168
28
+ permissions: read-only
29
+ c_parameter: const int32_t b[7][256]
30
+ description: Input polynomial vector b (7 x 256 x int32_t)
31
+ */
32
+
5
33
  #include "../../../common.h"
6
34
  #if defined(MLD_ARITH_BACKEND_AARCH64) && !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \
7
35
  (defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLDSA_L == 7)
8
36
 
9
37
  /*
10
38
  * WARNING: This file is auto-derived from the mldsa-native source file
11
- * dev/aarch64_opt/src/mld_polyvecl_pointwise_acc_montgomery_l7_aarch64_asm.S using scripts/simpasm. Do not modify it directly.
39
+ * dev/aarch64_opt/src/mldsa_polyvecl_pointwise_acc_montgomery_l7_aarch64_asm.S using scripts/simpasm. Do not modify it directly.
12
40
  */
13
41
 
14
42
  .text
@@ -25,7 +53,7 @@ MLD_ASM_FN_SYMBOL(polyvecl_pointwise_acc_montgomery_l7_aarch64_asm)
25
53
  dup v1.4s, w3
26
54
  mov x3, #0x40 // =64
27
55
 
28
- Lpolyvecl_pointwise_acc_montgomery_l7_loop_start:
56
+ Lmld_polyvecl_pointwise_acc_montgomery_l7_loop_start:
29
57
  ldr q17, [x1, #0x10]
30
58
  ldr q18, [x1, #0x20]
31
59
  ldr q19, [x1, #0x30]
@@ -163,7 +191,7 @@ Lpolyvecl_pointwise_acc_montgomery_l7_loop_start:
163
191
  str q19, [x0, #0x30]
164
192
  str q16, [x0], #0x40
165
193
  subs x3, x3, #0x4
166
- cbnz x3, Lpolyvecl_pointwise_acc_montgomery_l7_loop_start
194
+ cbnz x3, Lmld_polyvecl_pointwise_acc_montgomery_l7_loop_start
167
195
  ret
168
196
  .cfi_endproc
169
197