pq_crypto 0.6.4 → 0.6.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (91) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +46 -0
  3. data/README.md +5 -0
  4. data/ext/pqcrypto/pqcrypto_version.h +1 -1
  5. data/ext/pqcrypto/vendor/.vendored +4 -4
  6. data/ext/pqcrypto/vendor/mlkem-native/README.md +2 -4
  7. data/ext/pqcrypto/vendor/mlkem-native/RELEASE.md +98 -0
  8. data/ext/pqcrypto/vendor/mlkem-native/mlkem/mlkem_native.c +8 -7
  9. data/ext/pqcrypto/vendor/mlkem-native/mlkem/mlkem_native.h +21 -1
  10. data/ext/pqcrypto/vendor/mlkem-native/mlkem/mlkem_native_asm.S +45 -44
  11. data/ext/pqcrypto/vendor/mlkem-native/mlkem/mlkem_native_config.h +36 -0
  12. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/common.h +13 -30
  13. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/compress.c +25 -5
  14. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/compress.h +14 -0
  15. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/context.h +42 -0
  16. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/auto.h +20 -12
  17. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/src/keccak_f1600_x1_scalar_aarch64_asm.S +5 -3
  18. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/src/keccak_f1600_x1_v84a_aarch64_asm.S +5 -2
  19. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/src/keccak_f1600_x2_v84a_aarch64_asm.S +5 -2
  20. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/src/keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm.S +10 -7
  21. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/src/keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm.S +10 -7
  22. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x1_v84a.h +2 -1
  23. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x2_v84a.h +2 -1
  24. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x4_v8a_scalar.h +5 -0
  25. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x4_v8a_v84a_scalar.h +2 -1
  26. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/armv81m/mve.h +5 -19
  27. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/armv81m/src/keccak_f1600_x4_mve.S +8 -5
  28. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/armv81m/src/{state_extract_bytes_x4_mve.S → keccak_f1600_x4_state_extract_bytes_mve.S} +14 -14
  29. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/armv81m/src/{state_xor_bytes_x4_mve.S → keccak_f1600_x4_state_xor_bytes_mve.S} +12 -12
  30. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/x86_64/src/keccak_f1600_x4_avx2_asm.S +36 -2
  31. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/indcpa.c +22 -0
  32. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/indcpa.h +6 -0
  33. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/kem.c +11 -0
  34. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/kem.h +25 -1
  35. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/meta.h +44 -2
  36. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/aarch64_zetas.c +2 -0
  37. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/arith_native_aarch64.h +11 -10
  38. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{intt_aarch64_asm.S → mlkem_intt_aarch64_asm.S} +16 -9
  39. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{ntt_aarch64_asm.S → mlkem_ntt_aarch64_asm.S} +10 -7
  40. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{poly_mulcache_compute_aarch64_asm.S → mlkem_poly_mulcache_compute_aarch64_asm.S} +6 -3
  41. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{poly_reduce_aarch64_asm.S → mlkem_poly_reduce_aarch64_asm.S} +6 -3
  42. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{poly_tobytes_aarch64_asm.S → mlkem_poly_tobytes_aarch64_asm.S} +12 -5
  43. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{poly_tomont_aarch64_asm.S → mlkem_poly_tomont_aarch64_asm.S} +11 -5
  44. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{polyvec_basemul_acc_montgomery_cached_k2_aarch64_asm.S → mlkem_polyvec_basemul_acc_montgomery_cached_k2_aarch64_asm.S} +6 -3
  45. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{polyvec_basemul_acc_montgomery_cached_k3_aarch64_asm.S → mlkem_polyvec_basemul_acc_montgomery_cached_k3_aarch64_asm.S} +6 -3
  46. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{polyvec_basemul_acc_montgomery_cached_k4_aarch64_asm.S → mlkem_polyvec_basemul_acc_montgomery_cached_k4_aarch64_asm.S} +6 -3
  47. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{rej_uniform_aarch64_asm.S → mlkem_rej_uniform_aarch64_asm.S} +17 -17
  48. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/api.h +21 -8
  49. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/meta.h +1 -1
  50. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/ppc64le/meta.h +4 -0
  51. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/ppc64le/src/{intt_ppc_asm.S → mlkem_intt_ppc_asm.S} +29 -5
  52. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/ppc64le/src/{ntt_ppc_asm.S → mlkem_ntt_ppc_asm.S} +22 -1
  53. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/ppc64le/src/{poly_tomont_ppc_asm.S → mlkem_poly_tomont_ppc_asm.S} +25 -3
  54. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/ppc64le/src/{reduce_ppc_asm.S → mlkem_reduce_ppc_asm.S} +22 -1
  55. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/riscv64/meta.h +4 -0
  56. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/riscv64/src/arith_native_riscv64.h +4 -0
  57. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/riscv64/src/rv64v_poly.c +6 -2
  58. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/meta.h +25 -5
  59. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/arith_native_x86_64.h +20 -20
  60. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/compress_consts.c +14 -3
  61. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/compress_consts.h +8 -0
  62. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{intt_avx2_asm.S → mlkem_intt_avx2_asm.S} +27 -3
  63. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{ntt_avx2_asm.S → mlkem_ntt_avx2_asm.S} +23 -1
  64. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{nttfrombytes_avx2_asm.S → mlkem_nttfrombytes_avx2_asm.S} +27 -3
  65. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{ntttobytes_avx2_asm.S → mlkem_ntttobytes_avx2_asm.S} +27 -3
  66. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{nttunpack_avx2_asm.S → mlkem_nttunpack_avx2_asm.S} +17 -1
  67. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{poly_compress_d10_avx2_asm.S → mlkem_poly_compress_d10_avx2_asm.S} +34 -3
  68. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{poly_compress_d11_avx2_asm.S → mlkem_poly_compress_d11_avx2_asm.S} +33 -2
  69. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{poly_compress_d4_avx2_asm.S → mlkem_poly_compress_d4_avx2_asm.S} +34 -3
  70. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{poly_compress_d5_avx2_asm.S → mlkem_poly_compress_d5_avx2_asm.S} +33 -2
  71. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{poly_decompress_d10_avx2_asm.S → mlkem_poly_decompress_d10_avx2_asm.S} +32 -3
  72. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{poly_decompress_d11_avx2_asm.S → mlkem_poly_decompress_d11_avx2_asm.S} +32 -2
  73. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{poly_decompress_d4_avx2_asm.S → mlkem_poly_decompress_d4_avx2_asm.S} +32 -3
  74. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{poly_decompress_d5_avx2_asm.S → mlkem_poly_decompress_d5_avx2_asm.S} +32 -2
  75. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{poly_mulcache_compute_avx2_asm.S → mlkem_poly_mulcache_compute_avx2_asm.S} +29 -1
  76. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{polyvec_basemul_acc_montgomery_cached_k2_avx2_asm.S → mlkem_polyvec_basemul_acc_montgomery_cached_k2_avx2_asm.S} +35 -1
  77. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{polyvec_basemul_acc_montgomery_cached_k3_avx2_asm.S → mlkem_polyvec_basemul_acc_montgomery_cached_k3_avx2_asm.S} +35 -1
  78. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{polyvec_basemul_acc_montgomery_cached_k4_avx2_asm.S → mlkem_polyvec_basemul_acc_montgomery_cached_k4_avx2_asm.S} +35 -1
  79. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{reduce_avx2_asm.S → mlkem_reduce_avx2_asm.S} +17 -1
  80. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{rej_uniform_avx2_asm.S → mlkem_rej_uniform_avx2_asm.S} +40 -7
  81. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{tomont_avx2_asm.S → mlkem_tomont_avx2_asm.S} +20 -3
  82. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/poly.c +7 -2
  83. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/poly.h +6 -0
  84. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/poly_k.c +25 -4
  85. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/poly_k.h +33 -4
  86. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/sampling.c +4 -0
  87. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/sampling.h +4 -0
  88. data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/sys.h +19 -1
  89. data/lib/pq_crypto/version.rb +1 -1
  90. data/script/vendor_libs.rb +3 -3
  91. metadata +40 -42
@@ -26,7 +26,10 @@
26
26
  #include "debug.h"
27
27
  #include "verify.h"
28
28
 
29
- #if defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || (MLKEM_K == 2 || MLKEM_K == 3)
29
+ #if (!defined(MLK_CONFIG_NO_ENCAPS_API) || \
30
+ !defined(MLK_CONFIG_NO_DECAPS_API)) && \
31
+ (defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_K == 2 || \
32
+ MLKEM_K == 3)
30
33
  /* Reference: `poly_compress()` in the reference implementation @[REF],
31
34
  * for ML-KEM-{512,768}.
32
35
  * - In contrast to the reference implementation, we assume
@@ -158,6 +161,7 @@ __contract__(
158
161
  mlk_poly_compress_d10_c(r, a);
159
162
  }
160
163
 
164
+ #if !defined(MLK_CONFIG_NO_DECAPS_API)
161
165
  /* Reference: `poly_decompress()` in the reference implementation @[REF],
162
166
  * for ML-KEM-{512,768}. */
163
167
  MLK_STATIC_TESTABLE void mlk_poly_decompress_d4_c(
@@ -268,9 +272,14 @@ __contract__(
268
272
 
269
273
  mlk_poly_decompress_d10_c(r, a);
270
274
  }
271
- #endif /* MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 2 || MLKEM_K == 3 */
272
-
273
- #if defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_K == 4
275
+ #endif /* !MLK_CONFIG_NO_DECAPS_API */
276
+ #endif /* (!MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API) && \
277
+ (MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 2 || MLKEM_K == 3) \
278
+ */
279
+
280
+ #if (!defined(MLK_CONFIG_NO_ENCAPS_API) || \
281
+ !defined(MLK_CONFIG_NO_DECAPS_API)) && \
282
+ (defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_K == 4)
274
283
  /* Reference: `poly_compress()` in the reference implementation @[REF],
275
284
  * for ML-KEM-1024.
276
285
  * - In contrast to the reference implementation, we assume
@@ -409,6 +418,7 @@ __contract__(
409
418
  mlk_poly_compress_d11_c(r, a);
410
419
  }
411
420
 
421
+ #if !defined(MLK_CONFIG_NO_DECAPS_API)
412
422
  /* Reference: `poly_decompress()` in the reference implementation @[REF],
413
423
  * for ML-KEM-1024. */
414
424
  MLK_STATIC_TESTABLE void mlk_poly_decompress_d5_c(
@@ -554,8 +564,11 @@ __contract__(
554
564
  mlk_poly_decompress_d11_c(r, a);
555
565
  }
556
566
 
557
- #endif /* MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 4 */
567
+ #endif /* !MLK_CONFIG_NO_DECAPS_API */
568
+ #endif /* (!MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API) && \
569
+ (MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 4) */
558
570
 
571
+ #if !defined(MLK_CONFIG_NO_KEYPAIR_API) || !defined(MLK_CONFIG_NO_ENCAPS_API)
559
572
  /* Reference: `poly_tobytes()` in the reference implementation @[REF].
560
573
  * - In contrast to the reference implementation, we assume
561
574
  * unsigned canonical coefficients here.
@@ -620,7 +633,9 @@ void mlk_poly_tobytes(uint8_t r[MLKEM_POLYBYTES], const mlk_poly *a)
620
633
 
621
634
  mlk_poly_tobytes_c(r, a);
622
635
  }
636
+ #endif /* !MLK_CONFIG_NO_KEYPAIR_API || !MLK_CONFIG_NO_ENCAPS_API */
623
637
 
638
+ #if !defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API)
624
639
  /* Reference: `poly_frombytes()` in the reference implementation @[REF]. */
625
640
  MLK_STATIC_TESTABLE void mlk_poly_frombytes_c(mlk_poly *r,
626
641
  const uint8_t a[MLKEM_POLYBYTES])
@@ -668,7 +683,9 @@ void mlk_poly_frombytes(mlk_poly *r, const uint8_t a[MLKEM_POLYBYTES])
668
683
 
669
684
  mlk_poly_frombytes_c(r, a);
670
685
  }
686
+ #endif /* !MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API */
671
687
 
688
+ #if !defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API)
672
689
  /* Reference: `poly_frommsg()` in the reference implementation @[REF].
673
690
  * - We use a value barrier around the bit-selection mask to
674
691
  * reduce the risk of compiler-introduced branches.
@@ -706,7 +723,9 @@ void mlk_poly_frommsg(mlk_poly *r, const uint8_t msg[MLKEM_INDCPA_MSGBYTES])
706
723
  }
707
724
  mlk_assert_abs_bound(r, MLKEM_N, MLKEM_Q);
708
725
  }
726
+ #endif /* !MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API */
709
727
 
728
+ #if !defined(MLK_CONFIG_NO_DECAPS_API)
710
729
  /* Reference: `poly_tomsg()` in the reference implementation @[REF].
711
730
  * - In contrast to the reference implementation, we assume
712
731
  * unsigned canonical coefficients here.
@@ -735,6 +754,7 @@ void mlk_poly_tomsg(uint8_t msg[MLKEM_INDCPA_MSGBYTES], const mlk_poly *r)
735
754
  }
736
755
  }
737
756
  }
757
+ #endif /* !MLK_CONFIG_NO_DECAPS_API */
738
758
 
739
759
  #else /* !MLK_CONFIG_MULTILEVEL_NO_SHARED */
740
760
 
@@ -333,6 +333,7 @@ __contract__(
333
333
  return (int16_t)((((uint32_t)u * MLKEM_Q) + 1024) >> 11);
334
334
  }
335
335
 
336
+ #if !defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API)
336
337
  #if defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || (MLKEM_K == 2 || MLKEM_K == 3)
337
338
  #define mlk_poly_compress_d4 MLK_NAMESPACE(poly_compress_d4)
338
339
  /**
@@ -374,6 +375,7 @@ MLK_INTERNAL_API
374
375
  void mlk_poly_compress_d10(uint8_t r[MLKEM_POLYCOMPRESSEDBYTES_D10],
375
376
  const mlk_poly *a);
376
377
 
378
+ #if !defined(MLK_CONFIG_NO_DECAPS_API)
377
379
  #define mlk_poly_decompress_d4 MLK_NAMESPACE(poly_decompress_d4)
378
380
  /**
379
381
  * De-serialization and subsequent decompression (4 bits) of a polynomial;
@@ -419,6 +421,7 @@ void mlk_poly_decompress_d4(mlk_poly *r,
419
421
  MLK_INTERNAL_API
420
422
  void mlk_poly_decompress_d10(mlk_poly *r,
421
423
  const uint8_t a[MLKEM_POLYCOMPRESSEDBYTES_D10]);
424
+ #endif /* !MLK_CONFIG_NO_DECAPS_API */
422
425
  #endif /* MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 2 || MLKEM_K == 3 */
423
426
 
424
427
  #if defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_K == 4
@@ -462,6 +465,7 @@ MLK_INTERNAL_API
462
465
  void mlk_poly_compress_d11(uint8_t r[MLKEM_POLYCOMPRESSEDBYTES_D11],
463
466
  const mlk_poly *a);
464
467
 
468
+ #if !defined(MLK_CONFIG_NO_DECAPS_API)
465
469
  #define mlk_poly_decompress_d5 MLK_NAMESPACE(poly_decompress_d5)
466
470
  /**
467
471
  * De-serialization and subsequent decompression (5 bits) of a polynomial;
@@ -507,8 +511,11 @@ void mlk_poly_decompress_d5(mlk_poly *r,
507
511
  MLK_INTERNAL_API
508
512
  void mlk_poly_decompress_d11(mlk_poly *r,
509
513
  const uint8_t a[MLKEM_POLYCOMPRESSEDBYTES_D11]);
514
+ #endif /* !MLK_CONFIG_NO_DECAPS_API */
510
515
  #endif /* MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 4 */
516
+ #endif /* !MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API */
511
517
 
518
+ #if !defined(MLK_CONFIG_NO_KEYPAIR_API) || !defined(MLK_CONFIG_NO_ENCAPS_API)
512
519
  #define mlk_poly_tobytes MLK_NAMESPACE(poly_tobytes)
513
520
  /**
514
521
  * Serialization of a polynomial. Signed coefficients are converted to
@@ -529,8 +536,10 @@ __contract__(
529
536
  requires(array_bound(a->coeffs, 0, MLKEM_N, 0, MLKEM_Q))
530
537
  assigns(memory_slice(r, MLKEM_POLYBYTES))
531
538
  );
539
+ #endif /* !MLK_CONFIG_NO_KEYPAIR_API || !MLK_CONFIG_NO_ENCAPS_API */
532
540
 
533
541
 
542
+ #if !defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API)
534
543
  #define mlk_poly_frombytes MLK_NAMESPACE(poly_frombytes)
535
544
  /**
536
545
  * De-serialization of a polynomial.
@@ -550,8 +559,10 @@ __contract__(
550
559
  assigns(memory_slice(r, sizeof(mlk_poly)))
551
560
  ensures(array_bound(r->coeffs, 0, MLKEM_N, 0, MLKEM_UINT12_LIMIT))
552
561
  );
562
+ #endif /* !MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API */
553
563
 
554
564
 
565
+ #if !defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API)
555
566
  #define mlk_poly_frommsg MLK_NAMESPACE(poly_frommsg)
556
567
  /**
557
568
  * Convert a 32-byte message to a polynomial.
@@ -573,7 +584,9 @@ __contract__(
573
584
  assigns(memory_slice(r, sizeof(mlk_poly)))
574
585
  ensures(array_bound(r->coeffs, 0, MLKEM_N, 0, MLKEM_Q))
575
586
  );
587
+ #endif /* !MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API */
576
588
 
589
+ #if !defined(MLK_CONFIG_NO_DECAPS_API)
577
590
  #define mlk_poly_tomsg MLK_NAMESPACE(poly_tomsg)
578
591
  /**
579
592
  * Convert a polynomial to a 32-byte message.
@@ -595,5 +608,6 @@ __contract__(
595
608
  requires(array_bound(r->coeffs, 0, MLKEM_N, 0, MLKEM_Q))
596
609
  assigns(memory_slice(msg, MLKEM_INDCPA_MSGBYTES))
597
610
  );
611
+ #endif /* !MLK_CONFIG_NO_DECAPS_API */
598
612
 
599
613
  #endif /* !MLK_COMPRESS_H */
@@ -0,0 +1,42 @@
1
+ /*
2
+ * Copyright (c) The mlkem-native project authors
3
+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT
4
+ */
5
+ #ifndef MLK_CONTEXT_H
6
+ #define MLK_CONTEXT_H
7
+
8
+ /* This header is included by common.h once the configuration has been pulled
9
+ * in; it is not meant to be included directly. */
10
+
11
+ /*
12
+ * If the integration wants to provide a context parameter for use in
13
+ * platform-specific hooks, then it should define this parameter.
14
+ *
15
+ * The MLK_CONTEXT_PARAMETERS_n macros are intended to be used with macros
16
+ * defining the function names and expand to either pass or discard the context
17
+ * argument as required by the current build. If there is no context parameter
18
+ * requested then these are removed from the prototypes and from all calls.
19
+ */
20
+ #ifdef MLK_CONFIG_CONTEXT_PARAMETER
21
+ #define MLK_CONTEXT_PARAMETERS_0(context) (context)
22
+ #define MLK_CONTEXT_PARAMETERS_1(arg0, context) (arg0, context)
23
+ #define MLK_CONTEXT_PARAMETERS_2(arg0, arg1, context) (arg0, arg1, context)
24
+ #define MLK_CONTEXT_PARAMETERS_3(arg0, arg1, arg2, context) \
25
+ (arg0, arg1, arg2, context)
26
+ #define MLK_CONTEXT_PARAMETERS_4(arg0, arg1, arg2, arg3, context) \
27
+ (arg0, arg1, arg2, arg3, context)
28
+ #else /* MLK_CONFIG_CONTEXT_PARAMETER */
29
+ #define MLK_CONTEXT_PARAMETERS_0(context) ()
30
+ #define MLK_CONTEXT_PARAMETERS_1(arg0, context) (arg0)
31
+ #define MLK_CONTEXT_PARAMETERS_2(arg0, arg1, context) (arg0, arg1)
32
+ #define MLK_CONTEXT_PARAMETERS_3(arg0, arg1, arg2, context) (arg0, arg1, arg2)
33
+ #define MLK_CONTEXT_PARAMETERS_4(arg0, arg1, arg2, arg3, context) \
34
+ (arg0, arg1, arg2, arg3)
35
+ #endif /* !MLK_CONFIG_CONTEXT_PARAMETER */
36
+
37
+ #if defined(MLK_CONFIG_CONTEXT_PARAMETER_TYPE) != \
38
+ defined(MLK_CONFIG_CONTEXT_PARAMETER)
39
+ #error MLK_CONFIG_CONTEXT_PARAMETER_TYPE must be defined if and only if MLK_CONFIG_CONTEXT_PARAMETER is defined
40
+ #endif
41
+
42
+ #endif /* !MLK_CONTEXT_H */
@@ -24,38 +24,44 @@
24
24
  /*
25
25
  * Keccak-f1600
26
26
  *
27
- * - On Arm-based Apple CPUs, we pick a pure Neon implementation.
27
+ * - On Arm-based Apple CPUs, or if MLK_SYS_AARCH64_FAST_SHA3 is set,
28
+ * we pick a pure Neon implementation.
28
29
  * - Otherwise, unless MLK_SYS_AARCH64_SLOW_BARREL_SHIFTER is set,
29
30
  * we use lazy-rotation scalar assembly from @[HYBRID].
30
31
  * - Otherwise, if MLK_SYS_AARCH64_SLOW_BARREL_SHIFTER is set, we
31
32
  * fall back to the standard C implementation.
32
33
  */
33
- #if defined(__ARM_FEATURE_SHA3) && defined(__APPLE__)
34
+ #if defined(__ARM_FEATURE_SHA3) && \
35
+ (defined(__APPLE__) || defined(MLK_SYS_AARCH64_FAST_SHA3))
34
36
  #include "x1_v84a.h"
35
37
  #elif !defined(MLK_SYS_AARCH64_SLOW_BARREL_SHIFTER)
36
38
  #include "x1_scalar.h"
37
39
  #endif
38
40
 
41
+ /* Batched, SIMD-based Keccak-f1600 implementations. */
42
+ #if defined(MLK_SYS_AARCH64_NEON)
43
+
39
44
  /*
40
45
  * Keccak-f1600x2/x4
41
46
  *
42
47
  * The optimal implementation is highly CPU-specific; see @[HYBRID].
43
48
  *
44
49
  * For now, if v8.4-A is not implemented, we fall back to Keccak-f1600.
45
- * If v8.4-A is implemented and we are on an Apple CPU, we use a plain
46
- * Neon-based implementation.
47
- * If v8.4-A is implemented and we are not on an Apple CPU, we use a
48
- * scalar/Neon/Neon hybrid.
49
- * The reason for this distinction is that Apple CPUs appear to implement
50
- * the SHA3 instructions on all SIMD units, while Arm CPUs prior to Cortex-X4
51
- * don't, and ordinary Neon instructions are still needed.
50
+ * If v8.4-A is implemented and we are on an Apple CPU or
51
+ * MLK_SYS_AARCH64_FAST_SHA3 is set, we use a plain Neon-based
52
+ * implementation.
53
+ * Otherwise, if v8.4-A is implemented, we use a scalar/Neon/Neon hybrid.
54
+ * The reason for this distinction is that Apple CPUs (and CPUs flagged with
55
+ * MLK_SYS_AARCH64_FAST_SHA3) implement the SHA3 instructions on all SIMD
56
+ * units, while Arm CPUs prior to Cortex-X4 don't, and ordinary Neon
57
+ * instructions are still needed.
52
58
  */
53
59
  #if defined(__ARM_FEATURE_SHA3)
54
60
  /*
55
- * For Apple-M cores, we use a plain implementation leveraging SHA3
56
- * instructions only.
61
+ * For Apple-M cores (and CPUs flagged with MLK_SYS_AARCH64_FAST_SHA3), we
62
+ * use a plain implementation leveraging SHA3 instructions only.
57
63
  */
58
- #if defined(__APPLE__)
64
+ #if defined(__APPLE__) || defined(MLK_SYS_AARCH64_FAST_SHA3)
59
65
  #include "x2_v84a.h"
60
66
  #else
61
67
  #include "x4_v8a_v84a_scalar.h"
@@ -67,4 +73,6 @@
67
73
 
68
74
  #endif /* !__ARM_FEATURE_SHA3 */
69
75
 
76
+ #endif /* MLK_SYS_AARCH64_NEON */
77
+
70
78
  #endif /* !MLK_FIPS202_NATIVE_AARCH64_AUTO_H */
@@ -13,6 +13,8 @@
13
13
  Description: AArch64 scalar implementation of Keccak-f[1600] permutation for single state
14
14
  Signature: void mlk_keccak_f1600_x1_scalar_aarch64_asm(uint64_t state[25], const uint64_t rc[24])
15
15
  ABI:
16
+ Architecture: aarch64
17
+ CallingConvention: AAPCS64
16
18
  x0:
17
19
  type: buffer
18
20
  size_bytes: 200
@@ -66,7 +68,7 @@ MLK_ASM_FN_SYMBOL(keccak_f1600_x1_scalar_aarch64_asm)
66
68
  .cfi_rel_offset x29, 0x70
67
69
  .cfi_rel_offset x30, 0x78
68
70
 
69
- Lkeccak_f1600_x1_scalar_initial:
71
+ Lmlk_keccak_f1600_x1_scalar_initial:
70
72
  mov x26, x1
71
73
  str x1, [sp, #0x8]
72
74
  ldp x1, x6, [x0]
@@ -191,7 +193,7 @@ Lkeccak_f1600_x1_scalar_initial:
191
193
  eor x14, x26, x6, ror #46
192
194
  eor x6, x27, x29, ror #41
193
195
 
194
- Lkeccak_f1600_x1_scalar_loop:
196
+ Lmlk_keccak_f1600_x1_scalar_loop:
195
197
  eor x0, x15, x11, ror #52
196
198
  eor x0, x0, x13, ror #48
197
199
  eor x26, x8, x9, ror #57
@@ -304,7 +306,7 @@ Lkeccak_f1600_x1_scalar_loop:
304
306
  bic x3, x0, x30, ror #5
305
307
  eor x23, x3, x26, ror #52
306
308
  eor x3, x29, x30, ror #24
307
- b.le Lkeccak_f1600_x1_scalar_loop
309
+ b.le Lmlk_keccak_f1600_x1_scalar_loop
308
310
  ror x6, x6, #0x2b
309
311
  ror x11, x11, #0x32
310
312
  ror x21, x21, #0x14
@@ -19,6 +19,9 @@
19
19
  Description: AArch64 ARMv8.4-A implementation of Keccak-f[1600] permutation for single state
20
20
  Signature: void mlk_keccak_f1600_x1_v84a_aarch64_asm(uint64_t state[25], const uint64_t rc[24])
21
21
  ABI:
22
+ Architecture: aarch64
23
+ CallingConvention: AAPCS64
24
+ Features: [NEON, SHA3]
22
25
  x0:
23
26
  type: buffer
24
27
  size_bytes: 200
@@ -91,7 +94,7 @@ MLK_ASM_FN_SYMBOL(keccak_f1600_x1_v84a_aarch64_asm)
91
94
  ldr d24, [x0, #0xc0]
92
95
  mov x2, #0x18 // =24
93
96
 
94
- Lkeccak_f1600_x1_v84a_loop:
97
+ Lmlk_keccak_f1600_x1_v84a_loop:
95
98
  eor3 v30.16b, v0.16b, v5.16b, v10.16b
96
99
  eor3 v29.16b, v1.16b, v6.16b, v11.16b
97
100
  eor3 v28.16b, v2.16b, v7.16b, v12.16b
@@ -160,7 +163,7 @@ Lkeccak_f1600_x1_v84a_loop:
160
163
  bcax v4.16b, v4.16b, v27.16b, v30.16b
161
164
  eor v0.16b, v0.16b, v31.16b
162
165
  sub x2, x2, #0x1
163
- cbnz x2, Lkeccak_f1600_x1_v84a_loop
166
+ cbnz x2, Lmlk_keccak_f1600_x1_v84a_loop
164
167
  stp d0, d1, [x0]
165
168
  stp d2, d3, [x0, #0x10]
166
169
  stp d4, d5, [x0, #0x20]
@@ -19,6 +19,9 @@
19
19
  Description: AArch64 ARMv8.4-A implementation of Keccak-f[1600] permutation for two sequential states
20
20
  Signature: void mlk_keccak_f1600_x2_v84a_aarch64_asm(uint64_t state[50], const uint64_t rc[24])
21
21
  ABI:
22
+ Architecture: aarch64
23
+ CallingConvention: AAPCS64
24
+ Features: [NEON, SHA3]
22
25
  x0:
23
26
  type: buffer
24
27
  size_bytes: 400
@@ -118,7 +121,7 @@ MLK_ASM_FN_SYMBOL(keccak_f1600_x2_v84a_aarch64_asm)
118
121
  trn1 v24.2d, v25.2d, v27.2d
119
122
  mov x2, #0x18 // =24
120
123
 
121
- Lkeccak_f1600_x2_v84a_loop:
124
+ Lmlk_keccak_f1600_x2_v84a_loop:
122
125
  eor3 v30.16b, v0.16b, v5.16b, v10.16b
123
126
  eor3 v29.16b, v1.16b, v6.16b, v11.16b
124
127
  eor3 v28.16b, v2.16b, v7.16b, v12.16b
@@ -187,7 +190,7 @@ Lkeccak_f1600_x2_v84a_loop:
187
190
  bcax v4.16b, v4.16b, v27.16b, v30.16b
188
191
  eor v0.16b, v0.16b, v31.16b
189
192
  sub x2, x2, #0x1
190
- cbnz x2, Lkeccak_f1600_x2_v84a_loop
193
+ cbnz x2, Lmlk_keccak_f1600_x2_v84a_loop
191
194
  sub x0, x0, #0xc0
192
195
  add x2, x0, #0xc8
193
196
  trn1 v25.2d, v0.2d, v1.2d
@@ -13,6 +13,9 @@
13
13
  Description: AArch64 hybrid scalar/vector implementation of Keccak-f[1600] permutation for four sequential states
14
14
  Signature: void mlk_keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm(uint64_t state[100], const uint64_t rc[24])
15
15
  ABI:
16
+ Architecture: aarch64
17
+ CallingConvention: AAPCS64
18
+ Features: [NEON]
16
19
  x0:
17
20
  type: buffer
18
21
  size_bytes: 800
@@ -140,7 +143,7 @@ MLK_ASM_FN_SYMBOL(keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm)
140
143
  ldr x25, [x0, #0xc0]
141
144
  sub x0, x0, #0x190
142
145
 
143
- Lkeccak_f1600_x4_v8a_scalar_hybrid_initial:
146
+ Lmlk_keccak_f1600_x4_v8a_scalar_hybrid_initial:
144
147
  eor x30, x24, x25
145
148
  eor x27, x9, x10
146
149
  eor v30.16b, v0.16b, v5.16b
@@ -523,7 +526,7 @@ Lkeccak_f1600_x4_v8a_scalar_hybrid_initial:
523
526
  str x30, [sp, #0x10]
524
527
  eor v0.16b, v0.16b, v28.16b
525
528
 
526
- Lkeccak_f1600_x4_v8a_scalar_hybrid_loop:
529
+ Lmlk_keccak_f1600_x4_v8a_scalar_hybrid_loop:
527
530
  eor x0, x15, x11, ror #52
528
531
  eor x0, x0, x13, ror #48
529
532
  eor v30.16b, v0.16b, v5.16b
@@ -911,8 +914,8 @@ Lkeccak_f1600_x4_v8a_scalar_hybrid_loop:
911
914
  str x30, [sp, #0x10]
912
915
  eor v0.16b, v0.16b, v28.16b
913
916
 
914
- Lkeccak_f1600_x4_v8a_scalar_hybrid_loop_end:
915
- b.le Lkeccak_f1600_x4_v8a_scalar_hybrid_loop
917
+ Lmlk_keccak_f1600_x4_v8a_scalar_hybrid_loop_end:
918
+ b.le Lmlk_keccak_f1600_x4_v8a_scalar_hybrid_loop
916
919
  ror x2, x2, #0x3d
917
920
  ror x3, x3, #0x27
918
921
  ror x4, x4, #0x36
@@ -938,7 +941,7 @@ Lkeccak_f1600_x4_v8a_scalar_hybrid_loop_end:
938
941
  ror x25, x25, #0x9
939
942
  ldr x30, [sp, #0x20]
940
943
  cmp x30, #0x1
941
- b.eq Lkeccak_f1600_x4_v8a_scalar_hybrid_done
944
+ b.eq Lmlk_keccak_f1600_x4_v8a_scalar_hybrid_done
942
945
  mov x30, #0x1 // =1
943
946
  str x30, [sp, #0x20]
944
947
  ldr x0, [sp]
@@ -972,9 +975,9 @@ Lkeccak_f1600_x4_v8a_scalar_hybrid_loop_end:
972
975
  ldp x15, x20, [x0, #0xb0]
973
976
  ldr x25, [x0, #0xc0]
974
977
  sub x0, x0, #0x258
975
- b Lkeccak_f1600_x4_v8a_scalar_hybrid_initial
978
+ b Lmlk_keccak_f1600_x4_v8a_scalar_hybrid_initial
976
979
 
977
- Lkeccak_f1600_x4_v8a_scalar_hybrid_done:
980
+ Lmlk_keccak_f1600_x4_v8a_scalar_hybrid_done:
978
981
  ldr x0, [sp]
979
982
  add x0, x0, #0x258
980
983
  stp x1, x6, [x0]
@@ -13,6 +13,9 @@
13
13
  Description: AArch64 hybrid scalar/vector implementation of Keccak-f[1600] permutation for four sequential states with ARMv8.4-A optimizations
14
14
  Signature: void mlk_keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm(uint64_t state[100], const uint64_t rc[24])
15
15
  ABI:
16
+ Architecture: aarch64
17
+ CallingConvention: AAPCS64
18
+ Features: [NEON, SHA3]
16
19
  x0:
17
20
  type: buffer
18
21
  size_bytes: 800
@@ -142,7 +145,7 @@ MLK_ASM_FN_SYMBOL(keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm)
142
145
  ldr x25, [x0, #0xc0]
143
146
  sub x0, x0, #0x190
144
147
 
145
- Lkeccak_f1600_x4_v8a_v84a_scalar_hybrid_initial:
148
+ Lmlk_keccak_f1600_x4_v8a_v84a_scalar_hybrid_initial:
146
149
  eor x30, x24, x25
147
150
  eor x27, x9, x10
148
151
  eor3 v30.16b, v0.16b, v5.16b, v10.16b
@@ -478,7 +481,7 @@ Lkeccak_f1600_x4_v8a_v84a_scalar_hybrid_initial:
478
481
  str x30, [sp, #0x10]
479
482
  eor v0.16b, v0.16b, v28.16b
480
483
 
481
- Lkeccak_f1600_x4_v8a_v84a_scalar_hybrid_loop:
484
+ Lmlk_keccak_f1600_x4_v8a_v84a_scalar_hybrid_loop:
482
485
  eor x0, x15, x11, ror #52
483
486
  eor x0, x0, x13, ror #48
484
487
  eor3 v30.16b, v0.16b, v5.16b, v10.16b
@@ -819,8 +822,8 @@ Lkeccak_f1600_x4_v8a_v84a_scalar_hybrid_loop:
819
822
  str x30, [sp, #0x10]
820
823
  eor v0.16b, v0.16b, v28.16b
821
824
 
822
- Lkeccak_f1600_x4_v8a_v84a_scalar_hybrid_loop_end:
823
- b.le Lkeccak_f1600_x4_v8a_v84a_scalar_hybrid_loop
825
+ Lmlk_keccak_f1600_x4_v8a_v84a_scalar_hybrid_loop_end:
826
+ b.le Lmlk_keccak_f1600_x4_v8a_v84a_scalar_hybrid_loop
824
827
  ror x2, x2, #0x3d
825
828
  ror x3, x3, #0x27
826
829
  ror x4, x4, #0x36
@@ -846,7 +849,7 @@ Lkeccak_f1600_x4_v8a_v84a_scalar_hybrid_loop_end:
846
849
  ror x25, x25, #0x9
847
850
  ldr x30, [sp, #0x20]
848
851
  cmp x30, #0x1
849
- b.eq Lkeccak_f1600_x4_v8a_v84a_scalar_hybrid_done
852
+ b.eq Lmlk_keccak_f1600_x4_v8a_v84a_scalar_hybrid_done
850
853
  mov x30, #0x1 // =1
851
854
  str x30, [sp, #0x20]
852
855
  ldr x0, [sp]
@@ -880,9 +883,9 @@ Lkeccak_f1600_x4_v8a_v84a_scalar_hybrid_loop_end:
880
883
  ldp x15, x20, [x0, #0xb0]
881
884
  ldr x25, [x0, #0xc0]
882
885
  sub x0, x0, #0x258
883
- b Lkeccak_f1600_x4_v8a_v84a_scalar_hybrid_initial
886
+ b Lmlk_keccak_f1600_x4_v8a_v84a_scalar_hybrid_initial
884
887
 
885
- Lkeccak_f1600_x4_v8a_v84a_scalar_hybrid_done:
888
+ Lmlk_keccak_f1600_x4_v8a_v84a_scalar_hybrid_done:
886
889
  ldr x0, [sp]
887
890
  add x0, x0, #0x258
888
891
  stp x1, x6, [x0]
@@ -21,7 +21,8 @@
21
21
  MLK_MUST_CHECK_RETURN_VALUE
22
22
  static MLK_INLINE int mlk_keccak_f1600_x1_native(uint64_t *state)
23
23
  {
24
- if (!mlk_sys_check_capability(MLK_SYS_CAP_SHA3))
24
+ if (!mlk_sys_check_capability(MLK_SYS_CAP_NEON) ||
25
+ !mlk_sys_check_capability(MLK_SYS_CAP_SHA3))
25
26
  {
26
27
  return MLK_NATIVE_FUNC_FALLBACK;
27
28
  }
@@ -21,7 +21,8 @@
21
21
  MLK_MUST_CHECK_RETURN_VALUE
22
22
  static MLK_INLINE int mlk_keccak_f1600_x4_native(uint64_t *state)
23
23
  {
24
- if (!mlk_sys_check_capability(MLK_SYS_CAP_SHA3))
24
+ if (!mlk_sys_check_capability(MLK_SYS_CAP_NEON) ||
25
+ !mlk_sys_check_capability(MLK_SYS_CAP_SHA3))
25
26
  {
26
27
  return MLK_NATIVE_FUNC_FALLBACK;
27
28
  }
@@ -17,6 +17,11 @@
17
17
  MLK_MUST_CHECK_RETURN_VALUE
18
18
  static MLK_INLINE int mlk_keccak_f1600_x4_native(uint64_t *state)
19
19
  {
20
+ if (!mlk_sys_check_capability(MLK_SYS_CAP_NEON))
21
+ {
22
+ return MLK_NATIVE_FUNC_FALLBACK;
23
+ }
24
+
20
25
  mlk_keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm(
21
26
  state, mlk_keccakf1600_round_constants);
22
27
  return MLK_NATIVE_FUNC_SUCCESS;
@@ -21,7 +21,8 @@
21
21
  MLK_MUST_CHECK_RETURN_VALUE
22
22
  static MLK_INLINE int mlk_keccak_f1600_x4_native(uint64_t *state)
23
23
  {
24
- if (!mlk_sys_check_capability(MLK_SYS_CAP_SHA3))
24
+ if (!mlk_sys_check_capability(MLK_SYS_CAP_NEON) ||
25
+ !mlk_sys_check_capability(MLK_SYS_CAP_SHA3))
25
26
  {
26
27
  return MLK_NATIVE_FUNC_FALLBACK;
27
28
  }
@@ -17,6 +17,7 @@
17
17
 
18
18
  #if !defined(__ASSEMBLER__)
19
19
  #include "../api.h"
20
+ #include "src/fips202_native_armv81m.h"
20
21
 
21
22
  /*
22
23
  * Native x4 permutation
@@ -35,42 +36,27 @@ static MLK_INLINE int mlk_keccak_f1600_x4_native(uint64_t *state)
35
36
  /*
36
37
  * Native x4 XOR bytes (with on-the-fly bit interleaving)
37
38
  */
38
- #define mlk_keccak_f1600_x4_state_xor_bytes \
39
- MLK_NAMESPACE(keccak_f1600_x4_state_xor_bytes_asm)
40
- void mlk_keccak_f1600_x4_state_xor_bytes(void *state, const uint8_t *data0,
41
- const uint8_t *data1,
42
- const uint8_t *data2,
43
- const uint8_t *data3, unsigned offset,
44
- unsigned length);
45
-
46
39
  MLK_MUST_CHECK_RETURN_VALUE
47
40
  static MLK_INLINE int mlk_keccakf1600_xor_bytes_x4_native(
48
41
  uint64_t *state, const uint8_t *data0, const uint8_t *data1,
49
42
  const uint8_t *data2, const uint8_t *data3, unsigned offset,
50
43
  unsigned length)
51
44
  {
52
- mlk_keccak_f1600_x4_state_xor_bytes(state, data0, data1, data2, data3, offset,
53
- length);
45
+ mlk_keccak_f1600_x4_state_xor_bytes_asm(state, data0, data1, data2, data3,
46
+ offset, length);
54
47
  return MLK_NATIVE_FUNC_SUCCESS;
55
48
  }
56
49
 
57
50
  /*
58
51
  * Native x4 extract bytes (with on-the-fly bit de-interleaving)
59
52
  */
60
- #define mlk_keccak_f1600_x4_state_extract_bytes \
61
- MLK_NAMESPACE(keccak_f1600_x4_state_extract_bytes_asm)
62
- void mlk_keccak_f1600_x4_state_extract_bytes(void *state, uint8_t *data0,
63
- uint8_t *data1, uint8_t *data2,
64
- uint8_t *data3, unsigned offset,
65
- unsigned length);
66
-
67
53
  MLK_MUST_CHECK_RETURN_VALUE
68
54
  static MLK_INLINE int mlk_keccakf1600_extract_bytes_x4_native(
69
55
  uint64_t *state, uint8_t *data0, uint8_t *data1, uint8_t *data2,
70
56
  uint8_t *data3, unsigned offset, unsigned length)
71
57
  {
72
- mlk_keccak_f1600_x4_state_extract_bytes(state, data0, data1, data2, data3,
73
- offset, length);
58
+ mlk_keccak_f1600_x4_state_extract_bytes_asm(state, data0, data1, data2, data3,
59
+ offset, length);
74
60
  return MLK_NATIVE_FUNC_SUCCESS;
75
61
  }
76
62
 
@@ -9,6 +9,9 @@
9
9
  Description: Armv8.1-M MVE implementation of batched (x4) Keccak-f[1600] permutation using bit-interleaved state
10
10
  Signature: void mlk_keccak_f1600_x4_mve_asm(void *state, void *tmpstate, const uint32_t *rc)
11
11
  ABI:
12
+ Architecture: armv81m
13
+ CallingConvention: AAPCS32
14
+ Features: [MVE]
12
15
  r0:
13
16
  type: buffer
14
17
  size_bytes: 800
@@ -111,9 +114,9 @@ MLK_ASM_FN_SYMBOL(keccak_f1600_x4_mve_asm)
111
114
  vldrw.u32 q0, [r3]
112
115
  vldrw.u32 q1, [r2]
113
116
  vldrw.u32 q2, [r2, #32]
114
- wls lr, lr, Lkeccak_f1600_x4_mve_asm_roundend @ imm = #0x8c0
117
+ wls lr, lr, Lmlk_keccak_f1600_x4_mve_asm_roundend @ imm = #0x8c0
115
118
 
116
- Lkeccak_f1600_x4_mve_asm_roundstart:
119
+ Lmlk_keccak_f1600_x4_mve_asm_roundstart:
117
120
  vldrw.u32 q6, [r2, #112]
118
121
  veor q7, q6, q2
119
122
  vldrw.u32 q2, [r2, #80]
@@ -674,10 +677,10 @@ Lkeccak_f1600_x4_mve_asm_roundstart:
674
677
  veor q0, q4, q6
675
678
  vstrw.32 q0, [r5]
676
679
 
677
- Lkeccak_f1600_x4_mve_asm_roundend_pre:
678
- le lr, Lkeccak_f1600_x4_mve_asm_roundstart @ imm = #-0x8c0
680
+ Lmlk_keccak_f1600_x4_mve_asm_roundend_pre:
681
+ le lr, Lmlk_keccak_f1600_x4_mve_asm_roundstart @ imm = #-0x8c0
679
682
 
680
- Lkeccak_f1600_x4_mve_asm_roundend:
683
+ Lmlk_keccak_f1600_x4_mve_asm_roundend:
681
684
  add sp, #0x80
682
685
  .cfi_adjust_cfa_offset -0x80
683
686
  vpop {d8, d9, d10, d11, d12, d13, d14, d15}