pq_crypto 0.6.4 → 0.6.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +46 -0
- data/README.md +5 -0
- data/ext/pqcrypto/pqcrypto_version.h +1 -1
- data/ext/pqcrypto/vendor/.vendored +4 -4
- data/ext/pqcrypto/vendor/mlkem-native/README.md +2 -4
- data/ext/pqcrypto/vendor/mlkem-native/RELEASE.md +98 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/mlkem_native.c +8 -7
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/mlkem_native.h +21 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/mlkem_native_asm.S +45 -44
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/mlkem_native_config.h +36 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/common.h +13 -30
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/compress.c +25 -5
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/compress.h +14 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/context.h +42 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/auto.h +20 -12
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/src/keccak_f1600_x1_scalar_aarch64_asm.S +5 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/src/keccak_f1600_x1_v84a_aarch64_asm.S +5 -2
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/src/keccak_f1600_x2_v84a_aarch64_asm.S +5 -2
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/src/keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm.S +10 -7
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/src/keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm.S +10 -7
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x1_v84a.h +2 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x2_v84a.h +2 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x4_v8a_scalar.h +5 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x4_v8a_v84a_scalar.h +2 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/armv81m/mve.h +5 -19
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/armv81m/src/keccak_f1600_x4_mve.S +8 -5
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/armv81m/src/{state_extract_bytes_x4_mve.S → keccak_f1600_x4_state_extract_bytes_mve.S} +14 -14
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/armv81m/src/{state_xor_bytes_x4_mve.S → keccak_f1600_x4_state_xor_bytes_mve.S} +12 -12
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/x86_64/src/keccak_f1600_x4_avx2_asm.S +36 -2
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/indcpa.c +22 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/indcpa.h +6 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/kem.c +11 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/kem.h +25 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/meta.h +44 -2
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/aarch64_zetas.c +2 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/arith_native_aarch64.h +11 -10
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{intt_aarch64_asm.S → mlkem_intt_aarch64_asm.S} +16 -9
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{ntt_aarch64_asm.S → mlkem_ntt_aarch64_asm.S} +10 -7
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{poly_mulcache_compute_aarch64_asm.S → mlkem_poly_mulcache_compute_aarch64_asm.S} +6 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{poly_reduce_aarch64_asm.S → mlkem_poly_reduce_aarch64_asm.S} +6 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{poly_tobytes_aarch64_asm.S → mlkem_poly_tobytes_aarch64_asm.S} +12 -5
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{poly_tomont_aarch64_asm.S → mlkem_poly_tomont_aarch64_asm.S} +11 -5
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{polyvec_basemul_acc_montgomery_cached_k2_aarch64_asm.S → mlkem_polyvec_basemul_acc_montgomery_cached_k2_aarch64_asm.S} +6 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{polyvec_basemul_acc_montgomery_cached_k3_aarch64_asm.S → mlkem_polyvec_basemul_acc_montgomery_cached_k3_aarch64_asm.S} +6 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{polyvec_basemul_acc_montgomery_cached_k4_aarch64_asm.S → mlkem_polyvec_basemul_acc_montgomery_cached_k4_aarch64_asm.S} +6 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{rej_uniform_aarch64_asm.S → mlkem_rej_uniform_aarch64_asm.S} +17 -17
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/api.h +21 -8
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/meta.h +1 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/ppc64le/meta.h +4 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/ppc64le/src/{intt_ppc_asm.S → mlkem_intt_ppc_asm.S} +29 -5
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/ppc64le/src/{ntt_ppc_asm.S → mlkem_ntt_ppc_asm.S} +22 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/ppc64le/src/{poly_tomont_ppc_asm.S → mlkem_poly_tomont_ppc_asm.S} +25 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/ppc64le/src/{reduce_ppc_asm.S → mlkem_reduce_ppc_asm.S} +22 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/riscv64/meta.h +4 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/riscv64/src/arith_native_riscv64.h +4 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/riscv64/src/rv64v_poly.c +6 -2
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/meta.h +25 -5
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/arith_native_x86_64.h +20 -20
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/compress_consts.c +14 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/compress_consts.h +8 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{intt_avx2_asm.S → mlkem_intt_avx2_asm.S} +27 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{ntt_avx2_asm.S → mlkem_ntt_avx2_asm.S} +23 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{nttfrombytes_avx2_asm.S → mlkem_nttfrombytes_avx2_asm.S} +27 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{ntttobytes_avx2_asm.S → mlkem_ntttobytes_avx2_asm.S} +27 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{nttunpack_avx2_asm.S → mlkem_nttunpack_avx2_asm.S} +17 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{poly_compress_d10_avx2_asm.S → mlkem_poly_compress_d10_avx2_asm.S} +34 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{poly_compress_d11_avx2_asm.S → mlkem_poly_compress_d11_avx2_asm.S} +33 -2
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{poly_compress_d4_avx2_asm.S → mlkem_poly_compress_d4_avx2_asm.S} +34 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{poly_compress_d5_avx2_asm.S → mlkem_poly_compress_d5_avx2_asm.S} +33 -2
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{poly_decompress_d10_avx2_asm.S → mlkem_poly_decompress_d10_avx2_asm.S} +32 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{poly_decompress_d11_avx2_asm.S → mlkem_poly_decompress_d11_avx2_asm.S} +32 -2
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{poly_decompress_d4_avx2_asm.S → mlkem_poly_decompress_d4_avx2_asm.S} +32 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{poly_decompress_d5_avx2_asm.S → mlkem_poly_decompress_d5_avx2_asm.S} +32 -2
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{poly_mulcache_compute_avx2_asm.S → mlkem_poly_mulcache_compute_avx2_asm.S} +29 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{polyvec_basemul_acc_montgomery_cached_k2_avx2_asm.S → mlkem_polyvec_basemul_acc_montgomery_cached_k2_avx2_asm.S} +35 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{polyvec_basemul_acc_montgomery_cached_k3_avx2_asm.S → mlkem_polyvec_basemul_acc_montgomery_cached_k3_avx2_asm.S} +35 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{polyvec_basemul_acc_montgomery_cached_k4_avx2_asm.S → mlkem_polyvec_basemul_acc_montgomery_cached_k4_avx2_asm.S} +35 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{reduce_avx2_asm.S → mlkem_reduce_avx2_asm.S} +17 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{rej_uniform_avx2_asm.S → mlkem_rej_uniform_avx2_asm.S} +40 -7
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{tomont_avx2_asm.S → mlkem_tomont_avx2_asm.S} +20 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/poly.c +7 -2
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/poly.h +6 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/poly_k.c +25 -4
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/poly_k.h +33 -4
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/sampling.c +4 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/sampling.h +4 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/sys.h +19 -1
- data/lib/pq_crypto/version.rb +1 -1
- data/script/vendor_libs.rb +3 -3
- metadata +40 -42
|
@@ -26,7 +26,10 @@
|
|
|
26
26
|
#include "debug.h"
|
|
27
27
|
#include "verify.h"
|
|
28
28
|
|
|
29
|
-
#if defined(
|
|
29
|
+
#if (!defined(MLK_CONFIG_NO_ENCAPS_API) || \
|
|
30
|
+
!defined(MLK_CONFIG_NO_DECAPS_API)) && \
|
|
31
|
+
(defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_K == 2 || \
|
|
32
|
+
MLKEM_K == 3)
|
|
30
33
|
/* Reference: `poly_compress()` in the reference implementation @[REF],
|
|
31
34
|
* for ML-KEM-{512,768}.
|
|
32
35
|
* - In contrast to the reference implementation, we assume
|
|
@@ -158,6 +161,7 @@ __contract__(
|
|
|
158
161
|
mlk_poly_compress_d10_c(r, a);
|
|
159
162
|
}
|
|
160
163
|
|
|
164
|
+
#if !defined(MLK_CONFIG_NO_DECAPS_API)
|
|
161
165
|
/* Reference: `poly_decompress()` in the reference implementation @[REF],
|
|
162
166
|
* for ML-KEM-{512,768}. */
|
|
163
167
|
MLK_STATIC_TESTABLE void mlk_poly_decompress_d4_c(
|
|
@@ -268,9 +272,14 @@ __contract__(
|
|
|
268
272
|
|
|
269
273
|
mlk_poly_decompress_d10_c(r, a);
|
|
270
274
|
}
|
|
271
|
-
#endif /*
|
|
272
|
-
|
|
273
|
-
|
|
275
|
+
#endif /* !MLK_CONFIG_NO_DECAPS_API */
|
|
276
|
+
#endif /* (!MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API) && \
|
|
277
|
+
(MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 2 || MLKEM_K == 3) \
|
|
278
|
+
*/
|
|
279
|
+
|
|
280
|
+
#if (!defined(MLK_CONFIG_NO_ENCAPS_API) || \
|
|
281
|
+
!defined(MLK_CONFIG_NO_DECAPS_API)) && \
|
|
282
|
+
(defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_K == 4)
|
|
274
283
|
/* Reference: `poly_compress()` in the reference implementation @[REF],
|
|
275
284
|
* for ML-KEM-1024.
|
|
276
285
|
* - In contrast to the reference implementation, we assume
|
|
@@ -409,6 +418,7 @@ __contract__(
|
|
|
409
418
|
mlk_poly_compress_d11_c(r, a);
|
|
410
419
|
}
|
|
411
420
|
|
|
421
|
+
#if !defined(MLK_CONFIG_NO_DECAPS_API)
|
|
412
422
|
/* Reference: `poly_decompress()` in the reference implementation @[REF],
|
|
413
423
|
* for ML-KEM-1024. */
|
|
414
424
|
MLK_STATIC_TESTABLE void mlk_poly_decompress_d5_c(
|
|
@@ -554,8 +564,11 @@ __contract__(
|
|
|
554
564
|
mlk_poly_decompress_d11_c(r, a);
|
|
555
565
|
}
|
|
556
566
|
|
|
557
|
-
#endif /*
|
|
567
|
+
#endif /* !MLK_CONFIG_NO_DECAPS_API */
|
|
568
|
+
#endif /* (!MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API) && \
|
|
569
|
+
(MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 4) */
|
|
558
570
|
|
|
571
|
+
#if !defined(MLK_CONFIG_NO_KEYPAIR_API) || !defined(MLK_CONFIG_NO_ENCAPS_API)
|
|
559
572
|
/* Reference: `poly_tobytes()` in the reference implementation @[REF].
|
|
560
573
|
* - In contrast to the reference implementation, we assume
|
|
561
574
|
* unsigned canonical coefficients here.
|
|
@@ -620,7 +633,9 @@ void mlk_poly_tobytes(uint8_t r[MLKEM_POLYBYTES], const mlk_poly *a)
|
|
|
620
633
|
|
|
621
634
|
mlk_poly_tobytes_c(r, a);
|
|
622
635
|
}
|
|
636
|
+
#endif /* !MLK_CONFIG_NO_KEYPAIR_API || !MLK_CONFIG_NO_ENCAPS_API */
|
|
623
637
|
|
|
638
|
+
#if !defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API)
|
|
624
639
|
/* Reference: `poly_frombytes()` in the reference implementation @[REF]. */
|
|
625
640
|
MLK_STATIC_TESTABLE void mlk_poly_frombytes_c(mlk_poly *r,
|
|
626
641
|
const uint8_t a[MLKEM_POLYBYTES])
|
|
@@ -668,7 +683,9 @@ void mlk_poly_frombytes(mlk_poly *r, const uint8_t a[MLKEM_POLYBYTES])
|
|
|
668
683
|
|
|
669
684
|
mlk_poly_frombytes_c(r, a);
|
|
670
685
|
}
|
|
686
|
+
#endif /* !MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API */
|
|
671
687
|
|
|
688
|
+
#if !defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API)
|
|
672
689
|
/* Reference: `poly_frommsg()` in the reference implementation @[REF].
|
|
673
690
|
* - We use a value barrier around the bit-selection mask to
|
|
674
691
|
* reduce the risk of compiler-introduced branches.
|
|
@@ -706,7 +723,9 @@ void mlk_poly_frommsg(mlk_poly *r, const uint8_t msg[MLKEM_INDCPA_MSGBYTES])
|
|
|
706
723
|
}
|
|
707
724
|
mlk_assert_abs_bound(r, MLKEM_N, MLKEM_Q);
|
|
708
725
|
}
|
|
726
|
+
#endif /* !MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API */
|
|
709
727
|
|
|
728
|
+
#if !defined(MLK_CONFIG_NO_DECAPS_API)
|
|
710
729
|
/* Reference: `poly_tomsg()` in the reference implementation @[REF].
|
|
711
730
|
* - In contrast to the reference implementation, we assume
|
|
712
731
|
* unsigned canonical coefficients here.
|
|
@@ -735,6 +754,7 @@ void mlk_poly_tomsg(uint8_t msg[MLKEM_INDCPA_MSGBYTES], const mlk_poly *r)
|
|
|
735
754
|
}
|
|
736
755
|
}
|
|
737
756
|
}
|
|
757
|
+
#endif /* !MLK_CONFIG_NO_DECAPS_API */
|
|
738
758
|
|
|
739
759
|
#else /* !MLK_CONFIG_MULTILEVEL_NO_SHARED */
|
|
740
760
|
|
|
@@ -333,6 +333,7 @@ __contract__(
|
|
|
333
333
|
return (int16_t)((((uint32_t)u * MLKEM_Q) + 1024) >> 11);
|
|
334
334
|
}
|
|
335
335
|
|
|
336
|
+
#if !defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API)
|
|
336
337
|
#if defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || (MLKEM_K == 2 || MLKEM_K == 3)
|
|
337
338
|
#define mlk_poly_compress_d4 MLK_NAMESPACE(poly_compress_d4)
|
|
338
339
|
/**
|
|
@@ -374,6 +375,7 @@ MLK_INTERNAL_API
|
|
|
374
375
|
void mlk_poly_compress_d10(uint8_t r[MLKEM_POLYCOMPRESSEDBYTES_D10],
|
|
375
376
|
const mlk_poly *a);
|
|
376
377
|
|
|
378
|
+
#if !defined(MLK_CONFIG_NO_DECAPS_API)
|
|
377
379
|
#define mlk_poly_decompress_d4 MLK_NAMESPACE(poly_decompress_d4)
|
|
378
380
|
/**
|
|
379
381
|
* De-serialization and subsequent decompression (4 bits) of a polynomial;
|
|
@@ -419,6 +421,7 @@ void mlk_poly_decompress_d4(mlk_poly *r,
|
|
|
419
421
|
MLK_INTERNAL_API
|
|
420
422
|
void mlk_poly_decompress_d10(mlk_poly *r,
|
|
421
423
|
const uint8_t a[MLKEM_POLYCOMPRESSEDBYTES_D10]);
|
|
424
|
+
#endif /* !MLK_CONFIG_NO_DECAPS_API */
|
|
422
425
|
#endif /* MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 2 || MLKEM_K == 3 */
|
|
423
426
|
|
|
424
427
|
#if defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_K == 4
|
|
@@ -462,6 +465,7 @@ MLK_INTERNAL_API
|
|
|
462
465
|
void mlk_poly_compress_d11(uint8_t r[MLKEM_POLYCOMPRESSEDBYTES_D11],
|
|
463
466
|
const mlk_poly *a);
|
|
464
467
|
|
|
468
|
+
#if !defined(MLK_CONFIG_NO_DECAPS_API)
|
|
465
469
|
#define mlk_poly_decompress_d5 MLK_NAMESPACE(poly_decompress_d5)
|
|
466
470
|
/**
|
|
467
471
|
* De-serialization and subsequent decompression (5 bits) of a polynomial;
|
|
@@ -507,8 +511,11 @@ void mlk_poly_decompress_d5(mlk_poly *r,
|
|
|
507
511
|
MLK_INTERNAL_API
|
|
508
512
|
void mlk_poly_decompress_d11(mlk_poly *r,
|
|
509
513
|
const uint8_t a[MLKEM_POLYCOMPRESSEDBYTES_D11]);
|
|
514
|
+
#endif /* !MLK_CONFIG_NO_DECAPS_API */
|
|
510
515
|
#endif /* MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 4 */
|
|
516
|
+
#endif /* !MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API */
|
|
511
517
|
|
|
518
|
+
#if !defined(MLK_CONFIG_NO_KEYPAIR_API) || !defined(MLK_CONFIG_NO_ENCAPS_API)
|
|
512
519
|
#define mlk_poly_tobytes MLK_NAMESPACE(poly_tobytes)
|
|
513
520
|
/**
|
|
514
521
|
* Serialization of a polynomial. Signed coefficients are converted to
|
|
@@ -529,8 +536,10 @@ __contract__(
|
|
|
529
536
|
requires(array_bound(a->coeffs, 0, MLKEM_N, 0, MLKEM_Q))
|
|
530
537
|
assigns(memory_slice(r, MLKEM_POLYBYTES))
|
|
531
538
|
);
|
|
539
|
+
#endif /* !MLK_CONFIG_NO_KEYPAIR_API || !MLK_CONFIG_NO_ENCAPS_API */
|
|
532
540
|
|
|
533
541
|
|
|
542
|
+
#if !defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API)
|
|
534
543
|
#define mlk_poly_frombytes MLK_NAMESPACE(poly_frombytes)
|
|
535
544
|
/**
|
|
536
545
|
* De-serialization of a polynomial.
|
|
@@ -550,8 +559,10 @@ __contract__(
|
|
|
550
559
|
assigns(memory_slice(r, sizeof(mlk_poly)))
|
|
551
560
|
ensures(array_bound(r->coeffs, 0, MLKEM_N, 0, MLKEM_UINT12_LIMIT))
|
|
552
561
|
);
|
|
562
|
+
#endif /* !MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API */
|
|
553
563
|
|
|
554
564
|
|
|
565
|
+
#if !defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API)
|
|
555
566
|
#define mlk_poly_frommsg MLK_NAMESPACE(poly_frommsg)
|
|
556
567
|
/**
|
|
557
568
|
* Convert a 32-byte message to a polynomial.
|
|
@@ -573,7 +584,9 @@ __contract__(
|
|
|
573
584
|
assigns(memory_slice(r, sizeof(mlk_poly)))
|
|
574
585
|
ensures(array_bound(r->coeffs, 0, MLKEM_N, 0, MLKEM_Q))
|
|
575
586
|
);
|
|
587
|
+
#endif /* !MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API */
|
|
576
588
|
|
|
589
|
+
#if !defined(MLK_CONFIG_NO_DECAPS_API)
|
|
577
590
|
#define mlk_poly_tomsg MLK_NAMESPACE(poly_tomsg)
|
|
578
591
|
/**
|
|
579
592
|
* Convert a polynomial to a 32-byte message.
|
|
@@ -595,5 +608,6 @@ __contract__(
|
|
|
595
608
|
requires(array_bound(r->coeffs, 0, MLKEM_N, 0, MLKEM_Q))
|
|
596
609
|
assigns(memory_slice(msg, MLKEM_INDCPA_MSGBYTES))
|
|
597
610
|
);
|
|
611
|
+
#endif /* !MLK_CONFIG_NO_DECAPS_API */
|
|
598
612
|
|
|
599
613
|
#endif /* !MLK_COMPRESS_H */
|
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
/*
|
|
2
|
+
* Copyright (c) The mlkem-native project authors
|
|
3
|
+
* SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT
|
|
4
|
+
*/
|
|
5
|
+
#ifndef MLK_CONTEXT_H
|
|
6
|
+
#define MLK_CONTEXT_H
|
|
7
|
+
|
|
8
|
+
/* This header is included by common.h once the configuration has been pulled
|
|
9
|
+
* in; it is not meant to be included directly. */
|
|
10
|
+
|
|
11
|
+
/*
|
|
12
|
+
* If the integration wants to provide a context parameter for use in
|
|
13
|
+
* platform-specific hooks, then it should define this parameter.
|
|
14
|
+
*
|
|
15
|
+
* The MLK_CONTEXT_PARAMETERS_n macros are intended to be used with macros
|
|
16
|
+
* defining the function names and expand to either pass or discard the context
|
|
17
|
+
* argument as required by the current build. If there is no context parameter
|
|
18
|
+
* requested then these are removed from the prototypes and from all calls.
|
|
19
|
+
*/
|
|
20
|
+
#ifdef MLK_CONFIG_CONTEXT_PARAMETER
|
|
21
|
+
#define MLK_CONTEXT_PARAMETERS_0(context) (context)
|
|
22
|
+
#define MLK_CONTEXT_PARAMETERS_1(arg0, context) (arg0, context)
|
|
23
|
+
#define MLK_CONTEXT_PARAMETERS_2(arg0, arg1, context) (arg0, arg1, context)
|
|
24
|
+
#define MLK_CONTEXT_PARAMETERS_3(arg0, arg1, arg2, context) \
|
|
25
|
+
(arg0, arg1, arg2, context)
|
|
26
|
+
#define MLK_CONTEXT_PARAMETERS_4(arg0, arg1, arg2, arg3, context) \
|
|
27
|
+
(arg0, arg1, arg2, arg3, context)
|
|
28
|
+
#else /* MLK_CONFIG_CONTEXT_PARAMETER */
|
|
29
|
+
#define MLK_CONTEXT_PARAMETERS_0(context) ()
|
|
30
|
+
#define MLK_CONTEXT_PARAMETERS_1(arg0, context) (arg0)
|
|
31
|
+
#define MLK_CONTEXT_PARAMETERS_2(arg0, arg1, context) (arg0, arg1)
|
|
32
|
+
#define MLK_CONTEXT_PARAMETERS_3(arg0, arg1, arg2, context) (arg0, arg1, arg2)
|
|
33
|
+
#define MLK_CONTEXT_PARAMETERS_4(arg0, arg1, arg2, arg3, context) \
|
|
34
|
+
(arg0, arg1, arg2, arg3)
|
|
35
|
+
#endif /* !MLK_CONFIG_CONTEXT_PARAMETER */
|
|
36
|
+
|
|
37
|
+
#if defined(MLK_CONFIG_CONTEXT_PARAMETER_TYPE) != \
|
|
38
|
+
defined(MLK_CONFIG_CONTEXT_PARAMETER)
|
|
39
|
+
#error MLK_CONFIG_CONTEXT_PARAMETER_TYPE must be defined if and only if MLK_CONFIG_CONTEXT_PARAMETER is defined
|
|
40
|
+
#endif
|
|
41
|
+
|
|
42
|
+
#endif /* !MLK_CONTEXT_H */
|
|
@@ -24,38 +24,44 @@
|
|
|
24
24
|
/*
|
|
25
25
|
* Keccak-f1600
|
|
26
26
|
*
|
|
27
|
-
* - On Arm-based Apple CPUs,
|
|
27
|
+
* - On Arm-based Apple CPUs, or if MLK_SYS_AARCH64_FAST_SHA3 is set,
|
|
28
|
+
* we pick a pure Neon implementation.
|
|
28
29
|
* - Otherwise, unless MLK_SYS_AARCH64_SLOW_BARREL_SHIFTER is set,
|
|
29
30
|
* we use lazy-rotation scalar assembly from @[HYBRID].
|
|
30
31
|
* - Otherwise, if MLK_SYS_AARCH64_SLOW_BARREL_SHIFTER is set, we
|
|
31
32
|
* fall back to the standard C implementation.
|
|
32
33
|
*/
|
|
33
|
-
#if defined(__ARM_FEATURE_SHA3) &&
|
|
34
|
+
#if defined(__ARM_FEATURE_SHA3) && \
|
|
35
|
+
(defined(__APPLE__) || defined(MLK_SYS_AARCH64_FAST_SHA3))
|
|
34
36
|
#include "x1_v84a.h"
|
|
35
37
|
#elif !defined(MLK_SYS_AARCH64_SLOW_BARREL_SHIFTER)
|
|
36
38
|
#include "x1_scalar.h"
|
|
37
39
|
#endif
|
|
38
40
|
|
|
41
|
+
/* Batched, SIMD-based Keccak-f1600 implementations. */
|
|
42
|
+
#if defined(MLK_SYS_AARCH64_NEON)
|
|
43
|
+
|
|
39
44
|
/*
|
|
40
45
|
* Keccak-f1600x2/x4
|
|
41
46
|
*
|
|
42
47
|
* The optimal implementation is highly CPU-specific; see @[HYBRID].
|
|
43
48
|
*
|
|
44
49
|
* For now, if v8.4-A is not implemented, we fall back to Keccak-f1600.
|
|
45
|
-
* If v8.4-A is implemented and we are on an Apple CPU
|
|
46
|
-
* Neon-based
|
|
47
|
-
*
|
|
48
|
-
* scalar/Neon/Neon hybrid.
|
|
49
|
-
* The reason for this distinction is that Apple CPUs
|
|
50
|
-
* the SHA3 instructions on all SIMD
|
|
51
|
-
* don't, and ordinary Neon
|
|
50
|
+
* If v8.4-A is implemented and we are on an Apple CPU or
|
|
51
|
+
* MLK_SYS_AARCH64_FAST_SHA3 is set, we use a plain Neon-based
|
|
52
|
+
* implementation.
|
|
53
|
+
* Otherwise, if v8.4-A is implemented, we use a scalar/Neon/Neon hybrid.
|
|
54
|
+
* The reason for this distinction is that Apple CPUs (and CPUs flagged with
|
|
55
|
+
* MLK_SYS_AARCH64_FAST_SHA3) implement the SHA3 instructions on all SIMD
|
|
56
|
+
* units, while Arm CPUs prior to Cortex-X4 don't, and ordinary Neon
|
|
57
|
+
* instructions are still needed.
|
|
52
58
|
*/
|
|
53
59
|
#if defined(__ARM_FEATURE_SHA3)
|
|
54
60
|
/*
|
|
55
|
-
* For Apple-M cores
|
|
56
|
-
* instructions only.
|
|
61
|
+
* For Apple-M cores (and CPUs flagged with MLK_SYS_AARCH64_FAST_SHA3), we
|
|
62
|
+
* use a plain implementation leveraging SHA3 instructions only.
|
|
57
63
|
*/
|
|
58
|
-
#if defined(__APPLE__)
|
|
64
|
+
#if defined(__APPLE__) || defined(MLK_SYS_AARCH64_FAST_SHA3)
|
|
59
65
|
#include "x2_v84a.h"
|
|
60
66
|
#else
|
|
61
67
|
#include "x4_v8a_v84a_scalar.h"
|
|
@@ -67,4 +73,6 @@
|
|
|
67
73
|
|
|
68
74
|
#endif /* !__ARM_FEATURE_SHA3 */
|
|
69
75
|
|
|
76
|
+
#endif /* MLK_SYS_AARCH64_NEON */
|
|
77
|
+
|
|
70
78
|
#endif /* !MLK_FIPS202_NATIVE_AARCH64_AUTO_H */
|
|
@@ -13,6 +13,8 @@
|
|
|
13
13
|
Description: AArch64 scalar implementation of Keccak-f[1600] permutation for single state
|
|
14
14
|
Signature: void mlk_keccak_f1600_x1_scalar_aarch64_asm(uint64_t state[25], const uint64_t rc[24])
|
|
15
15
|
ABI:
|
|
16
|
+
Architecture: aarch64
|
|
17
|
+
CallingConvention: AAPCS64
|
|
16
18
|
x0:
|
|
17
19
|
type: buffer
|
|
18
20
|
size_bytes: 200
|
|
@@ -66,7 +68,7 @@ MLK_ASM_FN_SYMBOL(keccak_f1600_x1_scalar_aarch64_asm)
|
|
|
66
68
|
.cfi_rel_offset x29, 0x70
|
|
67
69
|
.cfi_rel_offset x30, 0x78
|
|
68
70
|
|
|
69
|
-
|
|
71
|
+
Lmlk_keccak_f1600_x1_scalar_initial:
|
|
70
72
|
mov x26, x1
|
|
71
73
|
str x1, [sp, #0x8]
|
|
72
74
|
ldp x1, x6, [x0]
|
|
@@ -191,7 +193,7 @@ Lkeccak_f1600_x1_scalar_initial:
|
|
|
191
193
|
eor x14, x26, x6, ror #46
|
|
192
194
|
eor x6, x27, x29, ror #41
|
|
193
195
|
|
|
194
|
-
|
|
196
|
+
Lmlk_keccak_f1600_x1_scalar_loop:
|
|
195
197
|
eor x0, x15, x11, ror #52
|
|
196
198
|
eor x0, x0, x13, ror #48
|
|
197
199
|
eor x26, x8, x9, ror #57
|
|
@@ -304,7 +306,7 @@ Lkeccak_f1600_x1_scalar_loop:
|
|
|
304
306
|
bic x3, x0, x30, ror #5
|
|
305
307
|
eor x23, x3, x26, ror #52
|
|
306
308
|
eor x3, x29, x30, ror #24
|
|
307
|
-
b.le
|
|
309
|
+
b.le Lmlk_keccak_f1600_x1_scalar_loop
|
|
308
310
|
ror x6, x6, #0x2b
|
|
309
311
|
ror x11, x11, #0x32
|
|
310
312
|
ror x21, x21, #0x14
|
|
@@ -19,6 +19,9 @@
|
|
|
19
19
|
Description: AArch64 ARMv8.4-A implementation of Keccak-f[1600] permutation for single state
|
|
20
20
|
Signature: void mlk_keccak_f1600_x1_v84a_aarch64_asm(uint64_t state[25], const uint64_t rc[24])
|
|
21
21
|
ABI:
|
|
22
|
+
Architecture: aarch64
|
|
23
|
+
CallingConvention: AAPCS64
|
|
24
|
+
Features: [NEON, SHA3]
|
|
22
25
|
x0:
|
|
23
26
|
type: buffer
|
|
24
27
|
size_bytes: 200
|
|
@@ -91,7 +94,7 @@ MLK_ASM_FN_SYMBOL(keccak_f1600_x1_v84a_aarch64_asm)
|
|
|
91
94
|
ldr d24, [x0, #0xc0]
|
|
92
95
|
mov x2, #0x18 // =24
|
|
93
96
|
|
|
94
|
-
|
|
97
|
+
Lmlk_keccak_f1600_x1_v84a_loop:
|
|
95
98
|
eor3 v30.16b, v0.16b, v5.16b, v10.16b
|
|
96
99
|
eor3 v29.16b, v1.16b, v6.16b, v11.16b
|
|
97
100
|
eor3 v28.16b, v2.16b, v7.16b, v12.16b
|
|
@@ -160,7 +163,7 @@ Lkeccak_f1600_x1_v84a_loop:
|
|
|
160
163
|
bcax v4.16b, v4.16b, v27.16b, v30.16b
|
|
161
164
|
eor v0.16b, v0.16b, v31.16b
|
|
162
165
|
sub x2, x2, #0x1
|
|
163
|
-
cbnz x2,
|
|
166
|
+
cbnz x2, Lmlk_keccak_f1600_x1_v84a_loop
|
|
164
167
|
stp d0, d1, [x0]
|
|
165
168
|
stp d2, d3, [x0, #0x10]
|
|
166
169
|
stp d4, d5, [x0, #0x20]
|
|
@@ -19,6 +19,9 @@
|
|
|
19
19
|
Description: AArch64 ARMv8.4-A implementation of Keccak-f[1600] permutation for two sequential states
|
|
20
20
|
Signature: void mlk_keccak_f1600_x2_v84a_aarch64_asm(uint64_t state[50], const uint64_t rc[24])
|
|
21
21
|
ABI:
|
|
22
|
+
Architecture: aarch64
|
|
23
|
+
CallingConvention: AAPCS64
|
|
24
|
+
Features: [NEON, SHA3]
|
|
22
25
|
x0:
|
|
23
26
|
type: buffer
|
|
24
27
|
size_bytes: 400
|
|
@@ -118,7 +121,7 @@ MLK_ASM_FN_SYMBOL(keccak_f1600_x2_v84a_aarch64_asm)
|
|
|
118
121
|
trn1 v24.2d, v25.2d, v27.2d
|
|
119
122
|
mov x2, #0x18 // =24
|
|
120
123
|
|
|
121
|
-
|
|
124
|
+
Lmlk_keccak_f1600_x2_v84a_loop:
|
|
122
125
|
eor3 v30.16b, v0.16b, v5.16b, v10.16b
|
|
123
126
|
eor3 v29.16b, v1.16b, v6.16b, v11.16b
|
|
124
127
|
eor3 v28.16b, v2.16b, v7.16b, v12.16b
|
|
@@ -187,7 +190,7 @@ Lkeccak_f1600_x2_v84a_loop:
|
|
|
187
190
|
bcax v4.16b, v4.16b, v27.16b, v30.16b
|
|
188
191
|
eor v0.16b, v0.16b, v31.16b
|
|
189
192
|
sub x2, x2, #0x1
|
|
190
|
-
cbnz x2,
|
|
193
|
+
cbnz x2, Lmlk_keccak_f1600_x2_v84a_loop
|
|
191
194
|
sub x0, x0, #0xc0
|
|
192
195
|
add x2, x0, #0xc8
|
|
193
196
|
trn1 v25.2d, v0.2d, v1.2d
|
|
@@ -13,6 +13,9 @@
|
|
|
13
13
|
Description: AArch64 hybrid scalar/vector implementation of Keccak-f[1600] permutation for four sequential states
|
|
14
14
|
Signature: void mlk_keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm(uint64_t state[100], const uint64_t rc[24])
|
|
15
15
|
ABI:
|
|
16
|
+
Architecture: aarch64
|
|
17
|
+
CallingConvention: AAPCS64
|
|
18
|
+
Features: [NEON]
|
|
16
19
|
x0:
|
|
17
20
|
type: buffer
|
|
18
21
|
size_bytes: 800
|
|
@@ -140,7 +143,7 @@ MLK_ASM_FN_SYMBOL(keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm)
|
|
|
140
143
|
ldr x25, [x0, #0xc0]
|
|
141
144
|
sub x0, x0, #0x190
|
|
142
145
|
|
|
143
|
-
|
|
146
|
+
Lmlk_keccak_f1600_x4_v8a_scalar_hybrid_initial:
|
|
144
147
|
eor x30, x24, x25
|
|
145
148
|
eor x27, x9, x10
|
|
146
149
|
eor v30.16b, v0.16b, v5.16b
|
|
@@ -523,7 +526,7 @@ Lkeccak_f1600_x4_v8a_scalar_hybrid_initial:
|
|
|
523
526
|
str x30, [sp, #0x10]
|
|
524
527
|
eor v0.16b, v0.16b, v28.16b
|
|
525
528
|
|
|
526
|
-
|
|
529
|
+
Lmlk_keccak_f1600_x4_v8a_scalar_hybrid_loop:
|
|
527
530
|
eor x0, x15, x11, ror #52
|
|
528
531
|
eor x0, x0, x13, ror #48
|
|
529
532
|
eor v30.16b, v0.16b, v5.16b
|
|
@@ -911,8 +914,8 @@ Lkeccak_f1600_x4_v8a_scalar_hybrid_loop:
|
|
|
911
914
|
str x30, [sp, #0x10]
|
|
912
915
|
eor v0.16b, v0.16b, v28.16b
|
|
913
916
|
|
|
914
|
-
|
|
915
|
-
b.le
|
|
917
|
+
Lmlk_keccak_f1600_x4_v8a_scalar_hybrid_loop_end:
|
|
918
|
+
b.le Lmlk_keccak_f1600_x4_v8a_scalar_hybrid_loop
|
|
916
919
|
ror x2, x2, #0x3d
|
|
917
920
|
ror x3, x3, #0x27
|
|
918
921
|
ror x4, x4, #0x36
|
|
@@ -938,7 +941,7 @@ Lkeccak_f1600_x4_v8a_scalar_hybrid_loop_end:
|
|
|
938
941
|
ror x25, x25, #0x9
|
|
939
942
|
ldr x30, [sp, #0x20]
|
|
940
943
|
cmp x30, #0x1
|
|
941
|
-
b.eq
|
|
944
|
+
b.eq Lmlk_keccak_f1600_x4_v8a_scalar_hybrid_done
|
|
942
945
|
mov x30, #0x1 // =1
|
|
943
946
|
str x30, [sp, #0x20]
|
|
944
947
|
ldr x0, [sp]
|
|
@@ -972,9 +975,9 @@ Lkeccak_f1600_x4_v8a_scalar_hybrid_loop_end:
|
|
|
972
975
|
ldp x15, x20, [x0, #0xb0]
|
|
973
976
|
ldr x25, [x0, #0xc0]
|
|
974
977
|
sub x0, x0, #0x258
|
|
975
|
-
b
|
|
978
|
+
b Lmlk_keccak_f1600_x4_v8a_scalar_hybrid_initial
|
|
976
979
|
|
|
977
|
-
|
|
980
|
+
Lmlk_keccak_f1600_x4_v8a_scalar_hybrid_done:
|
|
978
981
|
ldr x0, [sp]
|
|
979
982
|
add x0, x0, #0x258
|
|
980
983
|
stp x1, x6, [x0]
|
|
@@ -13,6 +13,9 @@
|
|
|
13
13
|
Description: AArch64 hybrid scalar/vector implementation of Keccak-f[1600] permutation for four sequential states with ARMv8.4-A optimizations
|
|
14
14
|
Signature: void mlk_keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm(uint64_t state[100], const uint64_t rc[24])
|
|
15
15
|
ABI:
|
|
16
|
+
Architecture: aarch64
|
|
17
|
+
CallingConvention: AAPCS64
|
|
18
|
+
Features: [NEON, SHA3]
|
|
16
19
|
x0:
|
|
17
20
|
type: buffer
|
|
18
21
|
size_bytes: 800
|
|
@@ -142,7 +145,7 @@ MLK_ASM_FN_SYMBOL(keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm)
|
|
|
142
145
|
ldr x25, [x0, #0xc0]
|
|
143
146
|
sub x0, x0, #0x190
|
|
144
147
|
|
|
145
|
-
|
|
148
|
+
Lmlk_keccak_f1600_x4_v8a_v84a_scalar_hybrid_initial:
|
|
146
149
|
eor x30, x24, x25
|
|
147
150
|
eor x27, x9, x10
|
|
148
151
|
eor3 v30.16b, v0.16b, v5.16b, v10.16b
|
|
@@ -478,7 +481,7 @@ Lkeccak_f1600_x4_v8a_v84a_scalar_hybrid_initial:
|
|
|
478
481
|
str x30, [sp, #0x10]
|
|
479
482
|
eor v0.16b, v0.16b, v28.16b
|
|
480
483
|
|
|
481
|
-
|
|
484
|
+
Lmlk_keccak_f1600_x4_v8a_v84a_scalar_hybrid_loop:
|
|
482
485
|
eor x0, x15, x11, ror #52
|
|
483
486
|
eor x0, x0, x13, ror #48
|
|
484
487
|
eor3 v30.16b, v0.16b, v5.16b, v10.16b
|
|
@@ -819,8 +822,8 @@ Lkeccak_f1600_x4_v8a_v84a_scalar_hybrid_loop:
|
|
|
819
822
|
str x30, [sp, #0x10]
|
|
820
823
|
eor v0.16b, v0.16b, v28.16b
|
|
821
824
|
|
|
822
|
-
|
|
823
|
-
b.le
|
|
825
|
+
Lmlk_keccak_f1600_x4_v8a_v84a_scalar_hybrid_loop_end:
|
|
826
|
+
b.le Lmlk_keccak_f1600_x4_v8a_v84a_scalar_hybrid_loop
|
|
824
827
|
ror x2, x2, #0x3d
|
|
825
828
|
ror x3, x3, #0x27
|
|
826
829
|
ror x4, x4, #0x36
|
|
@@ -846,7 +849,7 @@ Lkeccak_f1600_x4_v8a_v84a_scalar_hybrid_loop_end:
|
|
|
846
849
|
ror x25, x25, #0x9
|
|
847
850
|
ldr x30, [sp, #0x20]
|
|
848
851
|
cmp x30, #0x1
|
|
849
|
-
b.eq
|
|
852
|
+
b.eq Lmlk_keccak_f1600_x4_v8a_v84a_scalar_hybrid_done
|
|
850
853
|
mov x30, #0x1 // =1
|
|
851
854
|
str x30, [sp, #0x20]
|
|
852
855
|
ldr x0, [sp]
|
|
@@ -880,9 +883,9 @@ Lkeccak_f1600_x4_v8a_v84a_scalar_hybrid_loop_end:
|
|
|
880
883
|
ldp x15, x20, [x0, #0xb0]
|
|
881
884
|
ldr x25, [x0, #0xc0]
|
|
882
885
|
sub x0, x0, #0x258
|
|
883
|
-
b
|
|
886
|
+
b Lmlk_keccak_f1600_x4_v8a_v84a_scalar_hybrid_initial
|
|
884
887
|
|
|
885
|
-
|
|
888
|
+
Lmlk_keccak_f1600_x4_v8a_v84a_scalar_hybrid_done:
|
|
886
889
|
ldr x0, [sp]
|
|
887
890
|
add x0, x0, #0x258
|
|
888
891
|
stp x1, x6, [x0]
|
|
@@ -21,7 +21,8 @@
|
|
|
21
21
|
MLK_MUST_CHECK_RETURN_VALUE
|
|
22
22
|
static MLK_INLINE int mlk_keccak_f1600_x1_native(uint64_t *state)
|
|
23
23
|
{
|
|
24
|
-
if (!mlk_sys_check_capability(
|
|
24
|
+
if (!mlk_sys_check_capability(MLK_SYS_CAP_NEON) ||
|
|
25
|
+
!mlk_sys_check_capability(MLK_SYS_CAP_SHA3))
|
|
25
26
|
{
|
|
26
27
|
return MLK_NATIVE_FUNC_FALLBACK;
|
|
27
28
|
}
|
|
@@ -21,7 +21,8 @@
|
|
|
21
21
|
MLK_MUST_CHECK_RETURN_VALUE
|
|
22
22
|
static MLK_INLINE int mlk_keccak_f1600_x4_native(uint64_t *state)
|
|
23
23
|
{
|
|
24
|
-
if (!mlk_sys_check_capability(
|
|
24
|
+
if (!mlk_sys_check_capability(MLK_SYS_CAP_NEON) ||
|
|
25
|
+
!mlk_sys_check_capability(MLK_SYS_CAP_SHA3))
|
|
25
26
|
{
|
|
26
27
|
return MLK_NATIVE_FUNC_FALLBACK;
|
|
27
28
|
}
|
|
@@ -17,6 +17,11 @@
|
|
|
17
17
|
MLK_MUST_CHECK_RETURN_VALUE
|
|
18
18
|
static MLK_INLINE int mlk_keccak_f1600_x4_native(uint64_t *state)
|
|
19
19
|
{
|
|
20
|
+
if (!mlk_sys_check_capability(MLK_SYS_CAP_NEON))
|
|
21
|
+
{
|
|
22
|
+
return MLK_NATIVE_FUNC_FALLBACK;
|
|
23
|
+
}
|
|
24
|
+
|
|
20
25
|
mlk_keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm(
|
|
21
26
|
state, mlk_keccakf1600_round_constants);
|
|
22
27
|
return MLK_NATIVE_FUNC_SUCCESS;
|
|
@@ -21,7 +21,8 @@
|
|
|
21
21
|
MLK_MUST_CHECK_RETURN_VALUE
|
|
22
22
|
static MLK_INLINE int mlk_keccak_f1600_x4_native(uint64_t *state)
|
|
23
23
|
{
|
|
24
|
-
if (!mlk_sys_check_capability(
|
|
24
|
+
if (!mlk_sys_check_capability(MLK_SYS_CAP_NEON) ||
|
|
25
|
+
!mlk_sys_check_capability(MLK_SYS_CAP_SHA3))
|
|
25
26
|
{
|
|
26
27
|
return MLK_NATIVE_FUNC_FALLBACK;
|
|
27
28
|
}
|
|
@@ -17,6 +17,7 @@
|
|
|
17
17
|
|
|
18
18
|
#if !defined(__ASSEMBLER__)
|
|
19
19
|
#include "../api.h"
|
|
20
|
+
#include "src/fips202_native_armv81m.h"
|
|
20
21
|
|
|
21
22
|
/*
|
|
22
23
|
* Native x4 permutation
|
|
@@ -35,42 +36,27 @@ static MLK_INLINE int mlk_keccak_f1600_x4_native(uint64_t *state)
|
|
|
35
36
|
/*
|
|
36
37
|
* Native x4 XOR bytes (with on-the-fly bit interleaving)
|
|
37
38
|
*/
|
|
38
|
-
#define mlk_keccak_f1600_x4_state_xor_bytes \
|
|
39
|
-
MLK_NAMESPACE(keccak_f1600_x4_state_xor_bytes_asm)
|
|
40
|
-
void mlk_keccak_f1600_x4_state_xor_bytes(void *state, const uint8_t *data0,
|
|
41
|
-
const uint8_t *data1,
|
|
42
|
-
const uint8_t *data2,
|
|
43
|
-
const uint8_t *data3, unsigned offset,
|
|
44
|
-
unsigned length);
|
|
45
|
-
|
|
46
39
|
MLK_MUST_CHECK_RETURN_VALUE
|
|
47
40
|
static MLK_INLINE int mlk_keccakf1600_xor_bytes_x4_native(
|
|
48
41
|
uint64_t *state, const uint8_t *data0, const uint8_t *data1,
|
|
49
42
|
const uint8_t *data2, const uint8_t *data3, unsigned offset,
|
|
50
43
|
unsigned length)
|
|
51
44
|
{
|
|
52
|
-
|
|
53
|
-
|
|
45
|
+
mlk_keccak_f1600_x4_state_xor_bytes_asm(state, data0, data1, data2, data3,
|
|
46
|
+
offset, length);
|
|
54
47
|
return MLK_NATIVE_FUNC_SUCCESS;
|
|
55
48
|
}
|
|
56
49
|
|
|
57
50
|
/*
|
|
58
51
|
* Native x4 extract bytes (with on-the-fly bit de-interleaving)
|
|
59
52
|
*/
|
|
60
|
-
#define mlk_keccak_f1600_x4_state_extract_bytes \
|
|
61
|
-
MLK_NAMESPACE(keccak_f1600_x4_state_extract_bytes_asm)
|
|
62
|
-
void mlk_keccak_f1600_x4_state_extract_bytes(void *state, uint8_t *data0,
|
|
63
|
-
uint8_t *data1, uint8_t *data2,
|
|
64
|
-
uint8_t *data3, unsigned offset,
|
|
65
|
-
unsigned length);
|
|
66
|
-
|
|
67
53
|
MLK_MUST_CHECK_RETURN_VALUE
|
|
68
54
|
static MLK_INLINE int mlk_keccakf1600_extract_bytes_x4_native(
|
|
69
55
|
uint64_t *state, uint8_t *data0, uint8_t *data1, uint8_t *data2,
|
|
70
56
|
uint8_t *data3, unsigned offset, unsigned length)
|
|
71
57
|
{
|
|
72
|
-
|
|
73
|
-
|
|
58
|
+
mlk_keccak_f1600_x4_state_extract_bytes_asm(state, data0, data1, data2, data3,
|
|
59
|
+
offset, length);
|
|
74
60
|
return MLK_NATIVE_FUNC_SUCCESS;
|
|
75
61
|
}
|
|
76
62
|
|
data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/armv81m/src/keccak_f1600_x4_mve.S
CHANGED
|
@@ -9,6 +9,9 @@
|
|
|
9
9
|
Description: Armv8.1-M MVE implementation of batched (x4) Keccak-f[1600] permutation using bit-interleaved state
|
|
10
10
|
Signature: void mlk_keccak_f1600_x4_mve_asm(void *state, void *tmpstate, const uint32_t *rc)
|
|
11
11
|
ABI:
|
|
12
|
+
Architecture: armv81m
|
|
13
|
+
CallingConvention: AAPCS32
|
|
14
|
+
Features: [MVE]
|
|
12
15
|
r0:
|
|
13
16
|
type: buffer
|
|
14
17
|
size_bytes: 800
|
|
@@ -111,9 +114,9 @@ MLK_ASM_FN_SYMBOL(keccak_f1600_x4_mve_asm)
|
|
|
111
114
|
vldrw.u32 q0, [r3]
|
|
112
115
|
vldrw.u32 q1, [r2]
|
|
113
116
|
vldrw.u32 q2, [r2, #32]
|
|
114
|
-
wls lr, lr,
|
|
117
|
+
wls lr, lr, Lmlk_keccak_f1600_x4_mve_asm_roundend @ imm = #0x8c0
|
|
115
118
|
|
|
116
|
-
|
|
119
|
+
Lmlk_keccak_f1600_x4_mve_asm_roundstart:
|
|
117
120
|
vldrw.u32 q6, [r2, #112]
|
|
118
121
|
veor q7, q6, q2
|
|
119
122
|
vldrw.u32 q2, [r2, #80]
|
|
@@ -674,10 +677,10 @@ Lkeccak_f1600_x4_mve_asm_roundstart:
|
|
|
674
677
|
veor q0, q4, q6
|
|
675
678
|
vstrw.32 q0, [r5]
|
|
676
679
|
|
|
677
|
-
|
|
678
|
-
le lr,
|
|
680
|
+
Lmlk_keccak_f1600_x4_mve_asm_roundend_pre:
|
|
681
|
+
le lr, Lmlk_keccak_f1600_x4_mve_asm_roundstart @ imm = #-0x8c0
|
|
679
682
|
|
|
680
|
-
|
|
683
|
+
Lmlk_keccak_f1600_x4_mve_asm_roundend:
|
|
681
684
|
add sp, #0x80
|
|
682
685
|
.cfi_adjust_cfa_offset -0x80
|
|
683
686
|
vpop {d8, d9, d10, d11, d12, d13, d14, d15}
|