pq_crypto 0.6.4 → 0.6.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +180 -0
- data/README.md +7 -0
- data/ext/pqcrypto/extconf.rb +4 -1
- data/ext/pqcrypto/pq_externalmu.c +35 -0
- data/ext/pqcrypto/pqcrypto_native_api.h +85 -75
- data/ext/pqcrypto/pqcrypto_ruby_secure.c +16 -9
- data/ext/pqcrypto/pqcrypto_secure.c +43 -33
- data/ext/pqcrypto/pqcrypto_secure.h +52 -47
- data/ext/pqcrypto/pqcrypto_version.h +1 -1
- data/ext/pqcrypto/vendor/.vendored +7 -7
- data/ext/pqcrypto/vendor/mldsa-native/BUILDING.md +5 -2
- data/ext/pqcrypto/vendor/mldsa-native/LICENSE +21 -2
- data/ext/pqcrypto/vendor/mldsa-native/README.md +20 -7
- data/ext/pqcrypto/vendor/mldsa-native/RELEASE.md +160 -0
- data/ext/pqcrypto/vendor/mldsa-native/SECURITY.md +1 -1
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/README.md +2 -2
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/mldsa_native.c +85 -59
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/mldsa_native.h +292 -348
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/mldsa_native_asm.S +122 -76
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/mldsa_native_config.h +184 -86
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/cbmc.h +49 -4
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/common.h +49 -81
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/context.h +152 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/ct.h +25 -12
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/debug.c +2 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/debug.h +2 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/fips202x4.c +2 -2
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/keccakf1600.c +9 -11
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/auto.h +19 -11
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x1_scalar_aarch64_asm.S +6 -4
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x1_v84a_aarch64_asm.S +7 -4
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x2_v84a_aarch64_asm.S +7 -4
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm.S +12 -9
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm.S +12 -9
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/x1_scalar.h +1 -1
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/x1_v84a.h +3 -2
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/x2_v84a.h +3 -2
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/x4_v8a_scalar.h +6 -1
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/aarch64/x4_v8a_v84a_scalar.h +3 -2
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/api.h +11 -11
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/armv81m/mve.h +9 -22
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/armv81m/src/keccak_f1600_x4_mve.S +8 -5
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/armv81m/src/keccak_f1600_x4_mve.c +1 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/armv81m/src/{state_extract_bytes_x4_mve.S → keccak_f1600_x4_state_extract_bytes_mve.S} +14 -14
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/armv81m/src/{state_xor_bytes_x4_mve.S → keccak_f1600_x4_state_xor_bytes_mve.S} +12 -12
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/auto.h +5 -4
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/x86_64/keccak_f1600_x4_avx2.h +2 -2
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/x86_64/src/fips202_native_x86_64.h +1 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/fips202/native/x86_64/src/keccak_f1600_x4_avx2_asm.S +36 -2
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/meta.h +62 -4
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/arith_native_aarch64.h +87 -54
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{intt_aarch64_asm.S → mldsa_intt_aarch64_asm.S} +39 -6
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{ntt_aarch64_asm.S → mldsa_ntt_aarch64_asm.S} +39 -6
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{pointwise_montgomery_aarch64_asm.S → mldsa_pointwise_montgomery_aarch64_asm.S} +25 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{poly_caddq_aarch64_asm.S → mldsa_poly_caddq_aarch64_asm.S} +19 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{poly_chknorm_aarch64_asm.S → mldsa_poly_chknorm_aarch64_asm.S} +24 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{poly_decompose_32_aarch64_asm.S → mldsa_poly_decompose_32_aarch64_asm.S} +25 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{poly_decompose_88_aarch64_asm.S → mldsa_poly_decompose_88_aarch64_asm.S} +25 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{poly_use_hint_32_aarch64_asm.S → mldsa_poly_use_hint_32_aarch64_asm.S} +25 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{poly_use_hint_88_aarch64_asm.S → mldsa_poly_use_hint_88_aarch64_asm.S} +25 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{mld_polyvecl_pointwise_acc_montgomery_l4_aarch64_asm.S → mldsa_polyvecl_pointwise_acc_montgomery_l4_aarch64_asm.S} +31 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{mld_polyvecl_pointwise_acc_montgomery_l5_aarch64_asm.S → mldsa_polyvecl_pointwise_acc_montgomery_l5_aarch64_asm.S} +31 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{mld_polyvecl_pointwise_acc_montgomery_l7_aarch64_asm.S → mldsa_polyvecl_pointwise_acc_montgomery_l7_aarch64_asm.S} +31 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{polyz_unpack_17_aarch64_asm.S → mldsa_polyz_unpack_17_aarch64_asm.S} +31 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{polyz_unpack_19_aarch64_asm.S → mldsa_polyz_unpack_19_aarch64_asm.S} +31 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{rej_uniform_aarch64_asm.S → mldsa_rej_uniform_aarch64_asm.S} +48 -15
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{rej_uniform_eta2_aarch64_asm.S → mldsa_rej_uniform_eta2_aarch64_asm.S} +42 -9
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/aarch64/src/{rej_uniform_eta4_aarch64_asm.S → mldsa_rej_uniform_eta4_aarch64_asm.S} +42 -9
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/api.h +11 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/meta.h +3 -2
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/meta.h +28 -28
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/arith_native_x86_64.h +171 -49
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{intt_avx2_asm.S → mldsa_intt_avx2_asm.S} +23 -1
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{ntt_avx2_asm.S → mldsa_ntt_avx2_asm.S} +23 -1
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{nttunpack_avx2_asm.S → mldsa_nttunpack_avx2_asm.S} +17 -1
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{pointwise_acc_l4_avx2_asm.S → mldsa_pointwise_acc_l4_avx2_asm.S} +37 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{pointwise_acc_l5_avx2_asm.S → mldsa_pointwise_acc_l5_avx2_asm.S} +37 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{pointwise_acc_l7_avx2_asm.S → mldsa_pointwise_acc_l7_avx2_asm.S} +37 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{pointwise_avx2_asm.S → mldsa_pointwise_avx2_asm.S} +31 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/{poly_caddq_avx2_asm.S → mldsa_poly_caddq_avx2_asm.S} +18 -9
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_poly_chknorm_avx2_asm.S +176 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_poly_decompose_32_avx2_asm.S +490 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_poly_decompose_88_avx2_asm.S +489 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_poly_use_hint_32_avx2_asm.S +123 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_poly_use_hint_88_avx2_asm.S +125 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_polyz_unpack_17_avx2_asm.S +355 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_polyz_unpack_19_avx2_asm.S +355 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_rej_uniform_avx2_asm.S +132 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_rej_uniform_eta2_avx2_asm.S +205 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/mldsa_rej_uniform_eta4_avx2_asm.S +176 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/packing.c +27 -36
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/packing.h +42 -8
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/params.h +93 -17
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/poly.c +74 -15
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/poly.h +97 -11
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/poly_kl.c +7 -38
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/poly_kl.h +49 -7
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/polyvec.c +16 -17
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/polyvec.h +26 -9
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/polyvec_lazy.c +3 -0
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/polyvec_lazy.h +18 -19
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/reduce.h +15 -3
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/rounding.h +28 -6
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/sign.c +311 -246
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/sign.h +245 -240
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/sys.h +64 -5
- data/ext/pqcrypto/vendor/mlkem-native/BUILDING.md +5 -2
- data/ext/pqcrypto/vendor/mlkem-native/LICENSE +21 -3
- data/ext/pqcrypto/vendor/mlkem-native/README.md +4 -6
- data/ext/pqcrypto/vendor/mlkem-native/RELEASE.md +211 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/README.md +2 -2
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/mlkem_native.c +25 -34
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/mlkem_native.h +79 -142
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/mlkem_native_asm.S +62 -71
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/mlkem_native_config.h +77 -39
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/cbmc.h +25 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/common.h +48 -34
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/compress.c +25 -5
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/compress.h +14 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/context.h +51 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/fips202.h +2 -2
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/keccakf1600.c +8 -8
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/auto.h +20 -12
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/src/keccak_f1600_x1_scalar_aarch64_asm.S +5 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/src/keccak_f1600_x1_v84a_aarch64_asm.S +5 -2
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/src/keccak_f1600_x2_v84a_aarch64_asm.S +5 -2
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/src/keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm.S +10 -7
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/src/keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm.S +10 -7
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x1_scalar.h +1 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x1_v84a.h +3 -2
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x2_v84a.h +3 -2
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x4_v8a_scalar.h +6 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/aarch64/x4_v8a_v84a_scalar.h +3 -2
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/api.h +11 -11
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/armv81m/mve.h +8 -22
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/armv81m/src/keccak_f1600_x4_mve.S +8 -5
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/armv81m/src/{state_extract_bytes_x4_mve.S → keccak_f1600_x4_state_extract_bytes_mve.S} +14 -14
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/armv81m/src/{state_xor_bytes_x4_mve.S → keccak_f1600_x4_state_xor_bytes_mve.S} +12 -12
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/x86_64/keccak_f1600_x4_avx2.h +2 -2
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/fips202/native/x86_64/src/keccak_f1600_x4_avx2_asm.S +36 -2
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/indcpa.c +22 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/indcpa.h +20 -11
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/kem.c +39 -11
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/kem.h +64 -15
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/meta.h +44 -2
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/aarch64_zetas.c +2 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/arith_native_aarch64.h +11 -10
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{intt_aarch64_asm.S → mlkem_intt_aarch64_asm.S} +16 -9
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{ntt_aarch64_asm.S → mlkem_ntt_aarch64_asm.S} +10 -7
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{poly_mulcache_compute_aarch64_asm.S → mlkem_poly_mulcache_compute_aarch64_asm.S} +6 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{poly_reduce_aarch64_asm.S → mlkem_poly_reduce_aarch64_asm.S} +6 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{poly_tobytes_aarch64_asm.S → mlkem_poly_tobytes_aarch64_asm.S} +12 -5
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{poly_tomont_aarch64_asm.S → mlkem_poly_tomont_aarch64_asm.S} +11 -5
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{polyvec_basemul_acc_montgomery_cached_k2_aarch64_asm.S → mlkem_polyvec_basemul_acc_montgomery_cached_k2_aarch64_asm.S} +6 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{polyvec_basemul_acc_montgomery_cached_k3_aarch64_asm.S → mlkem_polyvec_basemul_acc_montgomery_cached_k3_aarch64_asm.S} +6 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{polyvec_basemul_acc_montgomery_cached_k4_aarch64_asm.S → mlkem_polyvec_basemul_acc_montgomery_cached_k4_aarch64_asm.S} +6 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/aarch64/src/{rej_uniform_aarch64_asm.S → mlkem_rej_uniform_aarch64_asm.S} +17 -17
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/api.h +21 -8
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/meta.h +1 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/ppc64le/meta.h +4 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/ppc64le/src/{intt_ppc_asm.S → mlkem_intt_ppc_asm.S} +29 -5
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/ppc64le/src/{ntt_ppc_asm.S → mlkem_ntt_ppc_asm.S} +22 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/ppc64le/src/{poly_tomont_ppc_asm.S → mlkem_poly_tomont_ppc_asm.S} +25 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/ppc64le/src/{reduce_ppc_asm.S → mlkem_reduce_ppc_asm.S} +22 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/riscv64/meta.h +4 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/riscv64/src/arith_native_riscv64.h +4 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/riscv64/src/rv64v_poly.c +6 -2
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/meta.h +45 -25
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/arith_native_x86_64.h +20 -20
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/compress_consts.c +14 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/compress_consts.h +8 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{intt_avx2_asm.S → mlkem_intt_avx2_asm.S} +27 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{ntt_avx2_asm.S → mlkem_ntt_avx2_asm.S} +23 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{nttfrombytes_avx2_asm.S → mlkem_nttfrombytes_avx2_asm.S} +27 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{ntttobytes_avx2_asm.S → mlkem_ntttobytes_avx2_asm.S} +27 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{nttunpack_avx2_asm.S → mlkem_nttunpack_avx2_asm.S} +17 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{poly_compress_d10_avx2_asm.S → mlkem_poly_compress_d10_avx2_asm.S} +34 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{poly_compress_d11_avx2_asm.S → mlkem_poly_compress_d11_avx2_asm.S} +33 -2
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{poly_compress_d4_avx2_asm.S → mlkem_poly_compress_d4_avx2_asm.S} +34 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{poly_compress_d5_avx2_asm.S → mlkem_poly_compress_d5_avx2_asm.S} +33 -2
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{poly_decompress_d10_avx2_asm.S → mlkem_poly_decompress_d10_avx2_asm.S} +32 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{poly_decompress_d11_avx2_asm.S → mlkem_poly_decompress_d11_avx2_asm.S} +32 -2
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{poly_decompress_d4_avx2_asm.S → mlkem_poly_decompress_d4_avx2_asm.S} +32 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{poly_decompress_d5_avx2_asm.S → mlkem_poly_decompress_d5_avx2_asm.S} +32 -2
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{poly_mulcache_compute_avx2_asm.S → mlkem_poly_mulcache_compute_avx2_asm.S} +29 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{polyvec_basemul_acc_montgomery_cached_k2_avx2_asm.S → mlkem_polyvec_basemul_acc_montgomery_cached_k2_avx2_asm.S} +35 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{polyvec_basemul_acc_montgomery_cached_k3_avx2_asm.S → mlkem_polyvec_basemul_acc_montgomery_cached_k3_avx2_asm.S} +35 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{polyvec_basemul_acc_montgomery_cached_k4_avx2_asm.S → mlkem_polyvec_basemul_acc_montgomery_cached_k4_avx2_asm.S} +35 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{reduce_avx2_asm.S → mlkem_reduce_avx2_asm.S} +17 -1
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{rej_uniform_avx2_asm.S → mlkem_rej_uniform_avx2_asm.S} +40 -7
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/native/x86_64/src/{tomont_avx2_asm.S → mlkem_tomont_avx2_asm.S} +20 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/poly.c +7 -2
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/poly.h +6 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/poly_k.c +25 -4
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/poly_k.h +33 -4
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/sampling.c +4 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/sampling.h +4 -0
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/sys.h +21 -3
- data/ext/pqcrypto/vendor/mlkem-native/mlkem/src/verify.h +11 -10
- data/lib/pq_crypto/version.rb +1 -1
- data/script/vendor_libs.rb +6 -6
- metadata +79 -79
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/poly_chknorm_avx2.c +0 -52
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/poly_decompose_32_avx2.c +0 -157
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/poly_decompose_88_avx2.c +0 -157
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/poly_use_hint_32_avx2.c +0 -103
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/poly_use_hint_88_avx2.c +0 -105
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/polyz_unpack_17_avx2.c +0 -94
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/polyz_unpack_19_avx2.c +0 -96
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/rej_uniform_avx2.c +0 -126
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/rej_uniform_eta2_avx2.c +0 -157
- data/ext/pqcrypto/vendor/mldsa-native/mldsa/src/native/x86_64/src/rej_uniform_eta4_avx2.c +0 -141
|
@@ -14,20 +14,53 @@
|
|
|
14
14
|
* Becker, Hwang, Kannwischer, Yang, Yang
|
|
15
15
|
* https://eprint.iacr.org/2021/986
|
|
16
16
|
*
|
|
17
|
+
* - [NeonNTT_Autoformalised]
|
|
18
|
+
* Neon NTT - (Auto)formalised
|
|
19
|
+
* Hanno Becker
|
|
20
|
+
* https://eprint.iacr.org/2026/1223
|
|
21
|
+
*
|
|
17
22
|
* - [SLOTHY_Paper]
|
|
18
23
|
* Fast and Clean: Auditable high-performance assembly via constraint solving
|
|
19
24
|
* Abdulrahman, Becker, Kannwischer, Klein
|
|
20
25
|
* https://eprint.iacr.org/2022/1303
|
|
21
26
|
*/
|
|
22
27
|
|
|
23
|
-
/* AArch64 ML-DSA forward NTT following @[NeonNTT] and @[
|
|
28
|
+
/* AArch64 ML-DSA forward NTT following @[NeonNTT], @[SLOTHY_Paper], and @[NeonNTT_Autoformalised] */
|
|
29
|
+
|
|
30
|
+
/*yaml
|
|
31
|
+
Name: ntt_aarch64_asm
|
|
32
|
+
Description: AArch64 ML-DSA forward NTT
|
|
33
|
+
Signature: void mld_ntt_aarch64_asm(int32_t r[256], const int32_t zetas_l123456[144], const int32_t zetas_l78[384])
|
|
34
|
+
ABI:
|
|
35
|
+
Architecture: aarch64
|
|
36
|
+
CallingConvention: AAPCS64
|
|
37
|
+
Features: [NEON]
|
|
38
|
+
x0:
|
|
39
|
+
type: buffer
|
|
40
|
+
size_bytes: 1024
|
|
41
|
+
permissions: read/write
|
|
42
|
+
c_parameter: int32_t r[256]
|
|
43
|
+
description: Input/output polynomial (256 x int32_t)
|
|
44
|
+
x1:
|
|
45
|
+
type: buffer
|
|
46
|
+
size_bytes: 576
|
|
47
|
+
permissions: read-only
|
|
48
|
+
c_parameter: const int32_t zetas_l123456[144]
|
|
49
|
+
description: Twiddle factors for layers 1-6 (144 x int32_t)
|
|
50
|
+
x2:
|
|
51
|
+
type: buffer
|
|
52
|
+
size_bytes: 1536
|
|
53
|
+
permissions: read-only
|
|
54
|
+
c_parameter: const int32_t zetas_l78[384]
|
|
55
|
+
description: Twiddle factors for layers 7-8 (384 x int32_t)
|
|
56
|
+
*/
|
|
24
57
|
|
|
25
58
|
#include "../../../common.h"
|
|
26
59
|
#if defined(MLD_ARITH_BACKEND_AARCH64) && !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)
|
|
27
60
|
|
|
28
61
|
/*
|
|
29
62
|
* WARNING: This file is auto-derived from the mldsa-native source file
|
|
30
|
-
* dev/aarch64_opt/src/
|
|
63
|
+
* dev/aarch64_opt/src/mldsa_ntt_aarch64_asm.S using scripts/simpasm. Do not modify it directly.
|
|
31
64
|
*/
|
|
32
65
|
|
|
33
66
|
.text
|
|
@@ -138,7 +171,7 @@ MLD_ASM_FN_SYMBOL(ntt_aarch64_asm)
|
|
|
138
171
|
sub v15.4s, v10.4s, v26.4s
|
|
139
172
|
sub x4, x4, #0x2
|
|
140
173
|
|
|
141
|
-
|
|
174
|
+
Lmld_ntt_layer123_start:
|
|
142
175
|
add v31.4s, v10.4s, v26.4s
|
|
143
176
|
mul v17.4s, v19.4s, v1.s[2]
|
|
144
177
|
add v26.4s, v15.4s, v23.4s
|
|
@@ -216,7 +249,7 @@ Lntt_layer123_start:
|
|
|
216
249
|
sqrdmulh v12.4s, v19.4s, v1.s[3]
|
|
217
250
|
sub v15.4s, v10.4s, v26.4s
|
|
218
251
|
subs x4, x4, #0x1
|
|
219
|
-
cbnz x4,
|
|
252
|
+
cbnz x4, Lmld_ntt_layer123_start
|
|
220
253
|
add v13.4s, v10.4s, v26.4s
|
|
221
254
|
mls v18.4s, v5.4s, v7.s[0]
|
|
222
255
|
str q22, [x0, #0x180]
|
|
@@ -378,7 +411,7 @@ Lntt_layer123_start:
|
|
|
378
411
|
sub v31.4s, v6.4s, v9.4s
|
|
379
412
|
sub x4, x4, #0x1
|
|
380
413
|
|
|
381
|
-
|
|
414
|
+
Lmld_ntt_layer45678_start:
|
|
382
415
|
add v2.4s, v13.4s, v12.4s
|
|
383
416
|
sqrdmulh v5.4s, v30.4s, v20.4s
|
|
384
417
|
sub v25.4s, v13.4s, v12.4s
|
|
@@ -544,7 +577,7 @@ Lntt_layer45678_start:
|
|
|
544
577
|
trn2 v10.2d, v22.2d, v1.2d
|
|
545
578
|
mul v28.4s, v30.4s, v3.4s
|
|
546
579
|
subs x4, x4, #0x1
|
|
547
|
-
cbnz x4,
|
|
580
|
+
cbnz x4, Lmld_ntt_layer45678_start
|
|
548
581
|
add v9.4s, v6.4s, v9.4s
|
|
549
582
|
sqrdmulh v6.4s, v30.4s, v20.4s
|
|
550
583
|
ldur q24, [x2, #-0xa0]
|
|
@@ -2,6 +2,28 @@
|
|
|
2
2
|
* SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT
|
|
3
3
|
*/
|
|
4
4
|
|
|
5
|
+
/*yaml
|
|
6
|
+
Name: poly_pointwise_montgomery_aarch64_asm
|
|
7
|
+
Description: AArch64 pointwise Montgomery multiplication of two polynomials
|
|
8
|
+
Signature: void mld_poly_pointwise_montgomery_aarch64_asm(int32_t a[256], const int32_t b[256])
|
|
9
|
+
ABI:
|
|
10
|
+
Architecture: aarch64
|
|
11
|
+
CallingConvention: AAPCS64
|
|
12
|
+
Features: [NEON]
|
|
13
|
+
x0:
|
|
14
|
+
type: buffer
|
|
15
|
+
size_bytes: 1024
|
|
16
|
+
permissions: read/write
|
|
17
|
+
c_parameter: int32_t a[256]
|
|
18
|
+
description: Input/output polynomial (256 x int32_t)
|
|
19
|
+
x1:
|
|
20
|
+
type: buffer
|
|
21
|
+
size_bytes: 1024
|
|
22
|
+
permissions: read-only
|
|
23
|
+
c_parameter: const int32_t b[256]
|
|
24
|
+
description: Input polynomial (256 x int32_t)
|
|
25
|
+
*/
|
|
26
|
+
|
|
5
27
|
#include "../../../common.h"
|
|
6
28
|
#if defined(MLD_ARITH_BACKEND_AARCH64) && \
|
|
7
29
|
(!defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API) || \
|
|
@@ -10,7 +32,7 @@
|
|
|
10
32
|
|
|
11
33
|
/*
|
|
12
34
|
* WARNING: This file is auto-derived from the mldsa-native source file
|
|
13
|
-
* dev/aarch64_opt/src/
|
|
35
|
+
* dev/aarch64_opt/src/mldsa_pointwise_montgomery_aarch64_asm.S using scripts/simpasm. Do not modify it directly.
|
|
14
36
|
*/
|
|
15
37
|
|
|
16
38
|
.text
|
|
@@ -27,7 +49,7 @@ MLD_ASM_FN_SYMBOL(poly_pointwise_montgomery_aarch64_asm)
|
|
|
27
49
|
dup v1.4s, w3
|
|
28
50
|
mov x3, #0x40 // =64
|
|
29
51
|
|
|
30
|
-
|
|
52
|
+
Lmld_poly_pointwise_montgomery_loop_start:
|
|
31
53
|
ldr q16, [x0]
|
|
32
54
|
ldr q17, [x0, #0x10]
|
|
33
55
|
ldr q18, [x0, #0x20]
|
|
@@ -69,7 +91,7 @@ Lpoly_pointwise_montgomery_loop_start:
|
|
|
69
91
|
str q19, [x0, #0x30]
|
|
70
92
|
str q16, [x0], #0x40
|
|
71
93
|
subs x3, x3, #0x4
|
|
72
|
-
cbnz x3,
|
|
94
|
+
cbnz x3, Lmld_poly_pointwise_montgomery_loop_start
|
|
73
95
|
ret
|
|
74
96
|
.cfi_endproc
|
|
75
97
|
|
|
@@ -2,13 +2,29 @@
|
|
|
2
2
|
* Copyright (c) The mldsa-native project authors
|
|
3
3
|
* SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT
|
|
4
4
|
*/
|
|
5
|
+
/*yaml
|
|
6
|
+
Name: poly_caddq_aarch64_asm
|
|
7
|
+
Description: AArch64 conditional addition of q to each coefficient
|
|
8
|
+
Signature: void mld_poly_caddq_aarch64_asm(int32_t a[256])
|
|
9
|
+
ABI:
|
|
10
|
+
Architecture: aarch64
|
|
11
|
+
CallingConvention: AAPCS64
|
|
12
|
+
Features: [NEON]
|
|
13
|
+
x0:
|
|
14
|
+
type: buffer
|
|
15
|
+
size_bytes: 1024
|
|
16
|
+
permissions: read/write
|
|
17
|
+
c_parameter: int32_t a[256]
|
|
18
|
+
description: Input/output polynomial (256 x int32_t)
|
|
19
|
+
*/
|
|
20
|
+
|
|
5
21
|
#include "../../../common.h"
|
|
6
22
|
|
|
7
23
|
#if defined(MLD_ARITH_BACKEND_AARCH64) && !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)
|
|
8
24
|
|
|
9
25
|
/*
|
|
10
26
|
* WARNING: This file is auto-derived from the mldsa-native source file
|
|
11
|
-
* dev/aarch64_opt/src/
|
|
27
|
+
* dev/aarch64_opt/src/mldsa_poly_caddq_aarch64_asm.S using scripts/simpasm. Do not modify it directly.
|
|
12
28
|
*/
|
|
13
29
|
|
|
14
30
|
.text
|
|
@@ -22,7 +38,7 @@ MLD_ASM_FN_SYMBOL(poly_caddq_aarch64_asm)
|
|
|
22
38
|
dup v4.4s, w9
|
|
23
39
|
mov x1, #0x10 // =16
|
|
24
40
|
|
|
25
|
-
|
|
41
|
+
Lmld_poly_caddq_loop:
|
|
26
42
|
ldr q0, [x0]
|
|
27
43
|
ldr q1, [x0, #0x10]
|
|
28
44
|
ldr q2, [x0, #0x20]
|
|
@@ -40,7 +56,7 @@ Lpoly_caddq_loop:
|
|
|
40
56
|
str q3, [x0, #0x30]
|
|
41
57
|
str q0, [x0], #0x40
|
|
42
58
|
subs x1, x1, #0x1
|
|
43
|
-
b.ne
|
|
59
|
+
b.ne Lmld_poly_caddq_loop
|
|
44
60
|
ret
|
|
45
61
|
.cfi_endproc
|
|
46
62
|
|
|
@@ -2,13 +2,34 @@
|
|
|
2
2
|
* Copyright (c) The mldsa-native project authors
|
|
3
3
|
* SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT
|
|
4
4
|
*/
|
|
5
|
+
/*yaml
|
|
6
|
+
Name: poly_chknorm_aarch64_asm
|
|
7
|
+
Description: AArch64 infinity-norm bound check on polynomial coefficients
|
|
8
|
+
Signature: int mld_poly_chknorm_aarch64_asm(const int32_t a[256], int32_t B)
|
|
9
|
+
ABI:
|
|
10
|
+
Architecture: aarch64
|
|
11
|
+
CallingConvention: AAPCS64
|
|
12
|
+
Features: [NEON]
|
|
13
|
+
x0:
|
|
14
|
+
type: buffer
|
|
15
|
+
size_bytes: 1024
|
|
16
|
+
permissions: read-only
|
|
17
|
+
c_parameter: const int32_t a[256]
|
|
18
|
+
description: Input polynomial (256 x int32_t)
|
|
19
|
+
x1:
|
|
20
|
+
type: scalar
|
|
21
|
+
c_parameter: int32_t B
|
|
22
|
+
description: Norm bound
|
|
23
|
+
test_with: 131072 # representative non-negative bound (1 << 17)
|
|
24
|
+
*/
|
|
25
|
+
|
|
5
26
|
#include "../../../common.h"
|
|
6
27
|
|
|
7
28
|
#if defined(MLD_ARITH_BACKEND_AARCH64) && !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)
|
|
8
29
|
|
|
9
30
|
/*
|
|
10
31
|
* WARNING: This file is auto-derived from the mldsa-native source file
|
|
11
|
-
* dev/aarch64_opt/src/
|
|
32
|
+
* dev/aarch64_opt/src/mldsa_poly_chknorm_aarch64_asm.S using scripts/simpasm. Do not modify it directly.
|
|
12
33
|
*/
|
|
13
34
|
|
|
14
35
|
.text
|
|
@@ -21,7 +42,7 @@ MLD_ASM_FN_SYMBOL(poly_chknorm_aarch64_asm)
|
|
|
21
42
|
eor v21.16b, v21.16b, v21.16b
|
|
22
43
|
mov x2, #0x10 // =16
|
|
23
44
|
|
|
24
|
-
|
|
45
|
+
Lmld_poly_chknorm_loop:
|
|
25
46
|
ldr q1, [x0, #0x10]
|
|
26
47
|
ldr q2, [x0, #0x20]
|
|
27
48
|
ldr q3, [x0, #0x30]
|
|
@@ -39,7 +60,7 @@ Lpoly_chknorm_loop:
|
|
|
39
60
|
cmge v0.4s, v0.4s, v20.4s
|
|
40
61
|
orr v21.16b, v21.16b, v0.16b
|
|
41
62
|
subs x2, x2, #0x1
|
|
42
|
-
b.ne
|
|
63
|
+
b.ne Lmld_poly_chknorm_loop
|
|
43
64
|
umaxv s21, v21.4s
|
|
44
65
|
fmov w0, s21
|
|
45
66
|
and w0, w0, #0x1
|
|
@@ -2,6 +2,28 @@
|
|
|
2
2
|
* Copyright (c) The mldsa-native project authors
|
|
3
3
|
* SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT
|
|
4
4
|
*/
|
|
5
|
+
/*yaml
|
|
6
|
+
Name: poly_decompose_32_aarch64_asm
|
|
7
|
+
Description: AArch64 coefficient decomposition (alpha = (Q-1)/32)
|
|
8
|
+
Signature: void mld_poly_decompose_32_aarch64_asm(int32_t a1[256], int32_t a0[256])
|
|
9
|
+
ABI:
|
|
10
|
+
Architecture: aarch64
|
|
11
|
+
CallingConvention: AAPCS64
|
|
12
|
+
Features: [NEON]
|
|
13
|
+
x0:
|
|
14
|
+
type: buffer
|
|
15
|
+
size_bytes: 1024
|
|
16
|
+
permissions: write-only
|
|
17
|
+
c_parameter: int32_t a1[256]
|
|
18
|
+
description: Output high-part polynomial (256 x int32_t)
|
|
19
|
+
x1:
|
|
20
|
+
type: buffer
|
|
21
|
+
size_bytes: 1024
|
|
22
|
+
permissions: read/write
|
|
23
|
+
c_parameter: int32_t a0[256]
|
|
24
|
+
description: Input polynomial / output low-part (256 x int32_t)
|
|
25
|
+
*/
|
|
26
|
+
|
|
5
27
|
#include "../../../common.h"
|
|
6
28
|
|
|
7
29
|
#if defined(MLD_ARITH_BACKEND_AARCH64) && !defined(MLD_CONFIG_NO_SIGN_API) && !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \
|
|
@@ -9,7 +31,7 @@
|
|
|
9
31
|
|
|
10
32
|
/*
|
|
11
33
|
* WARNING: This file is auto-derived from the mldsa-native source file
|
|
12
|
-
* dev/aarch64_opt/src/
|
|
34
|
+
* dev/aarch64_opt/src/mldsa_poly_decompose_32_aarch64_asm.S using scripts/simpasm. Do not modify it directly.
|
|
13
35
|
*/
|
|
14
36
|
|
|
15
37
|
.text
|
|
@@ -32,7 +54,7 @@ MLD_ASM_FN_SYMBOL(poly_decompose_32_aarch64_asm)
|
|
|
32
54
|
dup v23.4s, w11
|
|
33
55
|
mov x3, #0x10 // =16
|
|
34
56
|
|
|
35
|
-
|
|
57
|
+
Lmld_poly_decompose_32_loop:
|
|
36
58
|
ldr q0, [x1]
|
|
37
59
|
ldr q1, [x1, #0x10]
|
|
38
60
|
ldr q2, [x1, #0x20]
|
|
@@ -70,7 +92,7 @@ Lpoly_decompose_32_loop:
|
|
|
70
92
|
str q3, [x1, #0x30]
|
|
71
93
|
str q0, [x1], #0x40
|
|
72
94
|
subs x3, x3, #0x1
|
|
73
|
-
b.ne
|
|
95
|
+
b.ne Lmld_poly_decompose_32_loop
|
|
74
96
|
ret
|
|
75
97
|
.cfi_endproc
|
|
76
98
|
|
|
@@ -2,6 +2,28 @@
|
|
|
2
2
|
* Copyright (c) The mldsa-native project authors
|
|
3
3
|
* SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT
|
|
4
4
|
*/
|
|
5
|
+
/*yaml
|
|
6
|
+
Name: poly_decompose_88_aarch64_asm
|
|
7
|
+
Description: AArch64 coefficient decomposition (alpha = (Q-1)/88)
|
|
8
|
+
Signature: void mld_poly_decompose_88_aarch64_asm(int32_t a1[256], int32_t a0[256])
|
|
9
|
+
ABI:
|
|
10
|
+
Architecture: aarch64
|
|
11
|
+
CallingConvention: AAPCS64
|
|
12
|
+
Features: [NEON]
|
|
13
|
+
x0:
|
|
14
|
+
type: buffer
|
|
15
|
+
size_bytes: 1024
|
|
16
|
+
permissions: write-only
|
|
17
|
+
c_parameter: int32_t a1[256]
|
|
18
|
+
description: Output high-part polynomial (256 x int32_t)
|
|
19
|
+
x1:
|
|
20
|
+
type: buffer
|
|
21
|
+
size_bytes: 1024
|
|
22
|
+
permissions: read/write
|
|
23
|
+
c_parameter: int32_t a0[256]
|
|
24
|
+
description: Input polynomial / output low-part (256 x int32_t)
|
|
25
|
+
*/
|
|
26
|
+
|
|
5
27
|
#include "../../../common.h"
|
|
6
28
|
|
|
7
29
|
#if defined(MLD_ARITH_BACKEND_AARCH64) && !defined(MLD_CONFIG_NO_SIGN_API) && !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \
|
|
@@ -9,7 +31,7 @@
|
|
|
9
31
|
|
|
10
32
|
/*
|
|
11
33
|
* WARNING: This file is auto-derived from the mldsa-native source file
|
|
12
|
-
* dev/aarch64_opt/src/
|
|
34
|
+
* dev/aarch64_opt/src/mldsa_poly_decompose_88_aarch64_asm.S using scripts/simpasm. Do not modify it directly.
|
|
13
35
|
*/
|
|
14
36
|
|
|
15
37
|
.text
|
|
@@ -32,7 +54,7 @@ MLD_ASM_FN_SYMBOL(poly_decompose_88_aarch64_asm)
|
|
|
32
54
|
dup v23.4s, w11
|
|
33
55
|
mov x3, #0x10 // =16
|
|
34
56
|
|
|
35
|
-
|
|
57
|
+
Lmld_poly_decompose_88_loop:
|
|
36
58
|
ldr q0, [x1]
|
|
37
59
|
ldr q1, [x1, #0x10]
|
|
38
60
|
ldr q2, [x1, #0x20]
|
|
@@ -70,7 +92,7 @@ Lpoly_decompose_88_loop:
|
|
|
70
92
|
str q3, [x1, #0x30]
|
|
71
93
|
str q0, [x1], #0x40
|
|
72
94
|
subs x3, x3, #0x1
|
|
73
|
-
b.ne
|
|
95
|
+
b.ne Lmld_poly_decompose_88_loop
|
|
74
96
|
ret
|
|
75
97
|
.cfi_endproc
|
|
76
98
|
|
|
@@ -2,6 +2,28 @@
|
|
|
2
2
|
* Copyright (c) The mldsa-native project authors
|
|
3
3
|
* SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT
|
|
4
4
|
*/
|
|
5
|
+
/*yaml
|
|
6
|
+
Name: poly_use_hint_32_aarch64_asm
|
|
7
|
+
Description: AArch64 hint application (alpha = (Q-1)/32)
|
|
8
|
+
Signature: void mld_poly_use_hint_32_aarch64_asm(int32_t a[256], const int32_t h[256])
|
|
9
|
+
ABI:
|
|
10
|
+
Architecture: aarch64
|
|
11
|
+
CallingConvention: AAPCS64
|
|
12
|
+
Features: [NEON]
|
|
13
|
+
x0:
|
|
14
|
+
type: buffer
|
|
15
|
+
size_bytes: 1024
|
|
16
|
+
permissions: read/write
|
|
17
|
+
c_parameter: int32_t a[256]
|
|
18
|
+
description: Input/output polynomial (256 x int32_t)
|
|
19
|
+
x1:
|
|
20
|
+
type: buffer
|
|
21
|
+
size_bytes: 1024
|
|
22
|
+
permissions: read-only
|
|
23
|
+
c_parameter: const int32_t h[256]
|
|
24
|
+
description: Hint polynomial (256 x int32_t)
|
|
25
|
+
*/
|
|
26
|
+
|
|
5
27
|
#include "../../../common.h"
|
|
6
28
|
|
|
7
29
|
#if defined(MLD_ARITH_BACKEND_AARCH64) && !defined(MLD_CONFIG_NO_VERIFY_API) && !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \
|
|
@@ -9,7 +31,7 @@
|
|
|
9
31
|
|
|
10
32
|
/*
|
|
11
33
|
* WARNING: This file is auto-derived from the mldsa-native source file
|
|
12
|
-
* dev/aarch64_opt/src/
|
|
34
|
+
* dev/aarch64_opt/src/mldsa_poly_use_hint_32_aarch64_asm.S using scripts/simpasm. Do not modify it directly.
|
|
13
35
|
*/
|
|
14
36
|
|
|
15
37
|
.text
|
|
@@ -33,7 +55,7 @@ MLD_ASM_FN_SYMBOL(poly_use_hint_32_aarch64_asm)
|
|
|
33
55
|
movi v24.4s, #0xf
|
|
34
56
|
mov x3, #0x10 // =16
|
|
35
57
|
|
|
36
|
-
|
|
58
|
+
Lmld_poly_use_hint_32_loop:
|
|
37
59
|
ldr q1, [x0, #0x10]
|
|
38
60
|
ldr q2, [x0, #0x20]
|
|
39
61
|
ldr q3, [x0, #0x30]
|
|
@@ -87,7 +109,7 @@ Lpoly_use_hint_32_loop:
|
|
|
87
109
|
str q19, [x0, #0x30]
|
|
88
110
|
str q16, [x0], #0x40
|
|
89
111
|
subs x3, x3, #0x1
|
|
90
|
-
b.ne
|
|
112
|
+
b.ne Lmld_poly_use_hint_32_loop
|
|
91
113
|
ret
|
|
92
114
|
.cfi_endproc
|
|
93
115
|
|
|
@@ -2,6 +2,28 @@
|
|
|
2
2
|
* Copyright (c) The mldsa-native project authors
|
|
3
3
|
* SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT
|
|
4
4
|
*/
|
|
5
|
+
/*yaml
|
|
6
|
+
Name: poly_use_hint_88_aarch64_asm
|
|
7
|
+
Description: AArch64 hint application (alpha = (Q-1)/88)
|
|
8
|
+
Signature: void mld_poly_use_hint_88_aarch64_asm(int32_t a[256], const int32_t h[256])
|
|
9
|
+
ABI:
|
|
10
|
+
Architecture: aarch64
|
|
11
|
+
CallingConvention: AAPCS64
|
|
12
|
+
Features: [NEON]
|
|
13
|
+
x0:
|
|
14
|
+
type: buffer
|
|
15
|
+
size_bytes: 1024
|
|
16
|
+
permissions: read/write
|
|
17
|
+
c_parameter: int32_t a[256]
|
|
18
|
+
description: Input/output polynomial (256 x int32_t)
|
|
19
|
+
x1:
|
|
20
|
+
type: buffer
|
|
21
|
+
size_bytes: 1024
|
|
22
|
+
permissions: read-only
|
|
23
|
+
c_parameter: const int32_t h[256]
|
|
24
|
+
description: Hint polynomial (256 x int32_t)
|
|
25
|
+
*/
|
|
26
|
+
|
|
5
27
|
#include "../../../common.h"
|
|
6
28
|
|
|
7
29
|
#if defined(MLD_ARITH_BACKEND_AARCH64) && !defined(MLD_CONFIG_NO_VERIFY_API) && !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \
|
|
@@ -9,7 +31,7 @@
|
|
|
9
31
|
|
|
10
32
|
/*
|
|
11
33
|
* WARNING: This file is auto-derived from the mldsa-native source file
|
|
12
|
-
* dev/aarch64_opt/src/
|
|
34
|
+
* dev/aarch64_opt/src/mldsa_poly_use_hint_88_aarch64_asm.S using scripts/simpasm. Do not modify it directly.
|
|
13
35
|
*/
|
|
14
36
|
|
|
15
37
|
.text
|
|
@@ -33,7 +55,7 @@ MLD_ASM_FN_SYMBOL(poly_use_hint_88_aarch64_asm)
|
|
|
33
55
|
movi v24.4s, #0x2b
|
|
34
56
|
mov x3, #0x10 // =16
|
|
35
57
|
|
|
36
|
-
|
|
58
|
+
Lmld_poly_use_hint_88_loop:
|
|
37
59
|
ldr q1, [x0, #0x10]
|
|
38
60
|
ldr q2, [x0, #0x20]
|
|
39
61
|
ldr q3, [x0, #0x30]
|
|
@@ -95,7 +117,7 @@ Lpoly_use_hint_88_loop:
|
|
|
95
117
|
str q19, [x0, #0x30]
|
|
96
118
|
str q16, [x0], #0x40
|
|
97
119
|
subs x3, x3, #0x1
|
|
98
|
-
b.ne
|
|
120
|
+
b.ne Lmld_poly_use_hint_88_loop
|
|
99
121
|
ret
|
|
100
122
|
.cfi_endproc
|
|
101
123
|
|
|
@@ -2,13 +2,41 @@
|
|
|
2
2
|
* SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT
|
|
3
3
|
*/
|
|
4
4
|
|
|
5
|
+
/*yaml
|
|
6
|
+
Name: polyvecl_pointwise_acc_montgomery_l4_aarch64_asm
|
|
7
|
+
Description: AArch64 pointwise multiply-accumulate of length-4 polynomial vectors
|
|
8
|
+
Signature: void mld_polyvecl_pointwise_acc_montgomery_l4_aarch64_asm(int32_t r[256], const int32_t a[4][256], const int32_t b[4][256])
|
|
9
|
+
ABI:
|
|
10
|
+
Architecture: aarch64
|
|
11
|
+
CallingConvention: AAPCS64
|
|
12
|
+
Features: [NEON]
|
|
13
|
+
x0:
|
|
14
|
+
type: buffer
|
|
15
|
+
size_bytes: 1024
|
|
16
|
+
permissions: write-only
|
|
17
|
+
c_parameter: int32_t r[256]
|
|
18
|
+
description: Output polynomial (256 x int32_t)
|
|
19
|
+
x1:
|
|
20
|
+
type: buffer
|
|
21
|
+
size_bytes: 4096
|
|
22
|
+
permissions: read-only
|
|
23
|
+
c_parameter: const int32_t a[4][256]
|
|
24
|
+
description: Input polynomial vector a (4 x 256 x int32_t)
|
|
25
|
+
x2:
|
|
26
|
+
type: buffer
|
|
27
|
+
size_bytes: 4096
|
|
28
|
+
permissions: read-only
|
|
29
|
+
c_parameter: const int32_t b[4][256]
|
|
30
|
+
description: Input polynomial vector b (4 x 256 x int32_t)
|
|
31
|
+
*/
|
|
32
|
+
|
|
5
33
|
#include "../../../common.h"
|
|
6
34
|
#if defined(MLD_ARITH_BACKEND_AARCH64) && !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \
|
|
7
35
|
(defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLDSA_L == 4)
|
|
8
36
|
|
|
9
37
|
/*
|
|
10
38
|
* WARNING: This file is auto-derived from the mldsa-native source file
|
|
11
|
-
* dev/aarch64_opt/src/
|
|
39
|
+
* dev/aarch64_opt/src/mldsa_polyvecl_pointwise_acc_montgomery_l4_aarch64_asm.S using scripts/simpasm. Do not modify it directly.
|
|
12
40
|
*/
|
|
13
41
|
|
|
14
42
|
.text
|
|
@@ -25,7 +53,7 @@ MLD_ASM_FN_SYMBOL(polyvecl_pointwise_acc_montgomery_l4_aarch64_asm)
|
|
|
25
53
|
dup v1.4s, w3
|
|
26
54
|
mov x3, #0x40 // =64
|
|
27
55
|
|
|
28
|
-
|
|
56
|
+
Lmld_polyvecl_pointwise_acc_montgomery_l4_loop_start:
|
|
29
57
|
ldr q17, [x1, #0x10]
|
|
30
58
|
ldr q18, [x1, #0x20]
|
|
31
59
|
ldr q19, [x1, #0x30]
|
|
@@ -115,7 +143,7 @@ Lpolyvecl_pointwise_acc_montgomery_l4_loop_start:
|
|
|
115
143
|
str q19, [x0, #0x30]
|
|
116
144
|
str q16, [x0], #0x40
|
|
117
145
|
subs x3, x3, #0x4
|
|
118
|
-
cbnz x3,
|
|
146
|
+
cbnz x3, Lmld_polyvecl_pointwise_acc_montgomery_l4_loop_start
|
|
119
147
|
ret
|
|
120
148
|
.cfi_endproc
|
|
121
149
|
|
|
@@ -2,13 +2,41 @@
|
|
|
2
2
|
* SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT
|
|
3
3
|
*/
|
|
4
4
|
|
|
5
|
+
/*yaml
|
|
6
|
+
Name: polyvecl_pointwise_acc_montgomery_l5_aarch64_asm
|
|
7
|
+
Description: AArch64 pointwise multiply-accumulate of length-5 polynomial vectors
|
|
8
|
+
Signature: void mld_polyvecl_pointwise_acc_montgomery_l5_aarch64_asm(int32_t r[256], const int32_t a[5][256], const int32_t b[5][256])
|
|
9
|
+
ABI:
|
|
10
|
+
Architecture: aarch64
|
|
11
|
+
CallingConvention: AAPCS64
|
|
12
|
+
Features: [NEON]
|
|
13
|
+
x0:
|
|
14
|
+
type: buffer
|
|
15
|
+
size_bytes: 1024
|
|
16
|
+
permissions: write-only
|
|
17
|
+
c_parameter: int32_t r[256]
|
|
18
|
+
description: Output polynomial (256 x int32_t)
|
|
19
|
+
x1:
|
|
20
|
+
type: buffer
|
|
21
|
+
size_bytes: 5120
|
|
22
|
+
permissions: read-only
|
|
23
|
+
c_parameter: const int32_t a[5][256]
|
|
24
|
+
description: Input polynomial vector a (5 x 256 x int32_t)
|
|
25
|
+
x2:
|
|
26
|
+
type: buffer
|
|
27
|
+
size_bytes: 5120
|
|
28
|
+
permissions: read-only
|
|
29
|
+
c_parameter: const int32_t b[5][256]
|
|
30
|
+
description: Input polynomial vector b (5 x 256 x int32_t)
|
|
31
|
+
*/
|
|
32
|
+
|
|
5
33
|
#include "../../../common.h"
|
|
6
34
|
#if defined(MLD_ARITH_BACKEND_AARCH64) && !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \
|
|
7
35
|
(defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLDSA_L == 5)
|
|
8
36
|
|
|
9
37
|
/*
|
|
10
38
|
* WARNING: This file is auto-derived from the mldsa-native source file
|
|
11
|
-
* dev/aarch64_opt/src/
|
|
39
|
+
* dev/aarch64_opt/src/mldsa_polyvecl_pointwise_acc_montgomery_l5_aarch64_asm.S using scripts/simpasm. Do not modify it directly.
|
|
12
40
|
*/
|
|
13
41
|
|
|
14
42
|
.text
|
|
@@ -25,7 +53,7 @@ MLD_ASM_FN_SYMBOL(polyvecl_pointwise_acc_montgomery_l5_aarch64_asm)
|
|
|
25
53
|
dup v1.4s, w3
|
|
26
54
|
mov x3, #0x40 // =64
|
|
27
55
|
|
|
28
|
-
|
|
56
|
+
Lmld_polyvecl_pointwise_acc_montgomery_l5_loop_start:
|
|
29
57
|
ldr q17, [x1, #0x10]
|
|
30
58
|
ldr q18, [x1, #0x20]
|
|
31
59
|
ldr q19, [x1, #0x30]
|
|
@@ -131,7 +159,7 @@ Lpolyvecl_pointwise_acc_montgomery_l5_loop_start:
|
|
|
131
159
|
str q19, [x0, #0x30]
|
|
132
160
|
str q16, [x0], #0x40
|
|
133
161
|
subs x3, x3, #0x4
|
|
134
|
-
cbnz x3,
|
|
162
|
+
cbnz x3, Lmld_polyvecl_pointwise_acc_montgomery_l5_loop_start
|
|
135
163
|
ret
|
|
136
164
|
.cfi_endproc
|
|
137
165
|
|
|
@@ -2,13 +2,41 @@
|
|
|
2
2
|
* SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT
|
|
3
3
|
*/
|
|
4
4
|
|
|
5
|
+
/*yaml
|
|
6
|
+
Name: polyvecl_pointwise_acc_montgomery_l7_aarch64_asm
|
|
7
|
+
Description: AArch64 pointwise multiply-accumulate of length-7 polynomial vectors
|
|
8
|
+
Signature: void mld_polyvecl_pointwise_acc_montgomery_l7_aarch64_asm(int32_t r[256], const int32_t a[7][256], const int32_t b[7][256])
|
|
9
|
+
ABI:
|
|
10
|
+
Architecture: aarch64
|
|
11
|
+
CallingConvention: AAPCS64
|
|
12
|
+
Features: [NEON]
|
|
13
|
+
x0:
|
|
14
|
+
type: buffer
|
|
15
|
+
size_bytes: 1024
|
|
16
|
+
permissions: write-only
|
|
17
|
+
c_parameter: int32_t r[256]
|
|
18
|
+
description: Output polynomial (256 x int32_t)
|
|
19
|
+
x1:
|
|
20
|
+
type: buffer
|
|
21
|
+
size_bytes: 7168
|
|
22
|
+
permissions: read-only
|
|
23
|
+
c_parameter: const int32_t a[7][256]
|
|
24
|
+
description: Input polynomial vector a (7 x 256 x int32_t)
|
|
25
|
+
x2:
|
|
26
|
+
type: buffer
|
|
27
|
+
size_bytes: 7168
|
|
28
|
+
permissions: read-only
|
|
29
|
+
c_parameter: const int32_t b[7][256]
|
|
30
|
+
description: Input polynomial vector b (7 x 256 x int32_t)
|
|
31
|
+
*/
|
|
32
|
+
|
|
5
33
|
#include "../../../common.h"
|
|
6
34
|
#if defined(MLD_ARITH_BACKEND_AARCH64) && !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \
|
|
7
35
|
(defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLDSA_L == 7)
|
|
8
36
|
|
|
9
37
|
/*
|
|
10
38
|
* WARNING: This file is auto-derived from the mldsa-native source file
|
|
11
|
-
* dev/aarch64_opt/src/
|
|
39
|
+
* dev/aarch64_opt/src/mldsa_polyvecl_pointwise_acc_montgomery_l7_aarch64_asm.S using scripts/simpasm. Do not modify it directly.
|
|
12
40
|
*/
|
|
13
41
|
|
|
14
42
|
.text
|
|
@@ -25,7 +53,7 @@ MLD_ASM_FN_SYMBOL(polyvecl_pointwise_acc_montgomery_l7_aarch64_asm)
|
|
|
25
53
|
dup v1.4s, w3
|
|
26
54
|
mov x3, #0x40 // =64
|
|
27
55
|
|
|
28
|
-
|
|
56
|
+
Lmld_polyvecl_pointwise_acc_montgomery_l7_loop_start:
|
|
29
57
|
ldr q17, [x1, #0x10]
|
|
30
58
|
ldr q18, [x1, #0x20]
|
|
31
59
|
ldr q19, [x1, #0x30]
|
|
@@ -163,7 +191,7 @@ Lpolyvecl_pointwise_acc_montgomery_l7_loop_start:
|
|
|
163
191
|
str q19, [x0, #0x30]
|
|
164
192
|
str q16, [x0], #0x40
|
|
165
193
|
subs x3, x3, #0x4
|
|
166
|
-
cbnz x3,
|
|
194
|
+
cbnz x3, Lmld_polyvecl_pointwise_acc_montgomery_l7_loop_start
|
|
167
195
|
ret
|
|
168
196
|
.cfi_endproc
|
|
169
197
|
|