crypton 1.1.5 → 2.1.9
raw patch · 703 files changed
This diff is very large; some files are shown as “too large to diff”. Download the raw patch for the complete diff.
Files
- CHANGELOG.md +485/−0
- Crypto/Cipher/AES.hs +20/−0
- Crypto/Cipher/AES/GCM.hs +256/−0
- Crypto/Cipher/AES/Primitive.hs +379/−80
- Crypto/Cipher/Blowfish/Box.hs +0/−303
- Crypto/Cipher/Blowfish/Primitive.hs +94/−228
- Crypto/Cipher/Camellia/Primitive.hs +50/−279
- Crypto/Cipher/ChaCha.hs +16/−3
- Crypto/Cipher/ChaCha/Poly1305.hs +335/−0
- Crypto/Cipher/ChaChaPoly1305.hs +30/−26
- Crypto/Cipher/DES.hs +9/−10
- Crypto/Cipher/DES/Primitive.hs +66/−553
- Crypto/Cipher/RC4.hs +8/−1
- Crypto/Cipher/Salsa.hs +17/−4
- Crypto/Cipher/TripleDES.hs +51/−29
- Crypto/Cipher/Twofish/Primitive.hs +34/−33
- Crypto/Cipher/Types/AEAD.hs +34/−0
- Crypto/Cipher/Types/Block.hs +103/−19
- Crypto/Cipher/Types/Utils.hs +11/−5
- Crypto/ConstructHash/MiyaguchiPreneel.hs +7/−7
- Crypto/Data/AFIS.hs +62/−9
- Crypto/Data/Padding.hs +54/−4
- Crypto/Debug.hs +52/−0
- Crypto/ECC.hs +37/−13
- Crypto/ECC/Simple/Prim.hs +244/−26
- Crypto/ECC/Simple/Types.hs +13/−0
- Crypto/Error/Types.hs +11/−0
- Crypto/Hash/Algorithms.hs +2/−0
- Crypto/Hash/SHAKE.hs +4/−4
- Crypto/Hash/Skein256.hs +39/−0
- Crypto/Hash/Skein512.hs +39/−0
- Crypto/Hash/Types.hs +20/−0
- Crypto/Internal/ByteArray.hs +68/−1
- Crypto/Internal/ECC.hs +524/−0
- Crypto/Internal/Nat.hs +3/−3
- Crypto/Internal/Poly1305.hs +37/−0
- Crypto/KDF/Argon2.hs +31/−20
- Crypto/KDF/BCrypt.hs +84/−82
- Crypto/KDF/BCryptPBKDF.hs +47/−45
- Crypto/KDF/HKDF.hs +30/−3
- Crypto/KDF/PBKDF2.hs +90/−5
- Crypto/KDF/Scrypt.hs +22/−7
- Crypto/KEM.hs +121/−0
- Crypto/MAC/CMAC.hs +40/−20
- Crypto/MAC/HMAC.hs +1/−1
- Crypto/MAC/KMAC.hs +1/−1
- Crypto/MAC/KeyedBlake2.hs +1/−1
- Crypto/MAC/Poly1305.hs +30/−19
- Crypto/Number/Basic.hs +15/−3
- Crypto/Number/F2m.hs +111/−17
- Crypto/Number/ModArithmetic.hs +170/−13
- Crypto/Number/Prime.hs +84/−16
- Crypto/Number/Serialize/Internal.hs +6/−1
- Crypto/OTP.hs +87/−13
- Crypto/PubKey/Curve25519.hs +19/−10
- Crypto/PubKey/Curve448.hs +4/−0
- Crypto/PubKey/DH.hs +42/−2
- Crypto/PubKey/DSA.hs +117/−23
- Crypto/PubKey/ECC/DH.hs +38/−5
- Crypto/PubKey/ECC/ECDSA.hs +50/−6
- Crypto/PubKey/ECC/P256.hs +15/−5
- Crypto/PubKey/ECC/Prim.hs +410/−25
- Crypto/PubKey/ECC/Types.hs +13/−0
- Crypto/PubKey/ECDSA.hs +82/−5
- Crypto/PubKey/Ed25519.hs +4/−0
- Crypto/PubKey/Ed448.hs +4/−0
- Crypto/PubKey/EdDSA.hs +8/−2
- Crypto/PubKey/ElGamal.hs +180/−49
- Crypto/PubKey/Internal.hs +2/−3
- Crypto/PubKey/MLDSA.hs +656/−0
- Crypto/PubKey/MLKEM.hs +448/−0
- Crypto/PubKey/RSA.hs +79/−9
- Crypto/PubKey/RSA/OAEP.hs +52/−11
- Crypto/PubKey/RSA/PKCS15.hs +178/−15
- Crypto/PubKey/RSA/PSS.hs +37/−3
- Crypto/PubKey/RSA/Types.hs +67/−4
- Crypto/PubKey/Rabin/Basic.hs +119/−25
- Crypto/PubKey/Rabin/Modified.hs +79/−18
- Crypto/PubKey/Rabin/OAEP.hs +41/−8
- Crypto/PubKey/Rabin/RW.hs +85/−16
- Crypto/PubKey/Rabin/Types.hs +1/−0
- Crypto/Random/Probabilistic.hs +37/−11
- Crypto/Random/Types.hs +13/−1
- Crypto/Tutorial.hs +5/−2
- LICENSE +1/−0
- README.md +225/−64
- benchs/Bench.hs +13/−8
- benchs/Number/F2m.hs +11/−8
- cbits/LICENSE.go +38/−0
- cbits/aes/LICENSE.fusion +29/−0
- cbits/aes/armv8.c +469/−0
- cbits/aes/armv8_impl.c +824/−0
- cbits/aes/block128.h +15/−1
- cbits/aes/gcm_fused_x86.c +1013/−0
- cbits/aes/gcm_fused_x86.h +55/−0
- cbits/aes/gcm_vaes512_x86.c +364/−0
- cbits/aes/gcm_vaes512_x86.h +37/−0
- cbits/aes/gcm_vaes_x86.c +325/−0
- cbits/aes/gcm_vaes_x86.h +35/−0
- cbits/aes/gcm_x86_asm.c +266/−0
- cbits/aes/gcm_x86_asm.h +72/−0
- cbits/aes/gf.c +16/−0
- cbits/aes/gf.h +1/−0
- cbits/aes/x86ni.c +558/−13
- cbits/aes/x86ni.h +19/−23
- cbits/aes/x86ni_impl.c +321/−10
- cbits/asm/LICENSE.cryptogams +36/−0
- cbits/asm/README.md +171/−0
- cbits/asm/aesni-gcm-x86_64-elf.S +814/−0
- cbits/asm/aesni-gcm-x86_64-macosx.S +805/−0
- cbits/asm/aesni-gcm-x86_64-mingw64.S +965/−0
- cbits/asm/aesni-gcm-x86_64.pl +974/−0
- cbits/asm/arm-xlate.pl +467/−0
- cbits/asm/arm_arch.h +101/−0
- cbits/asm/chacha-armv8-ios64.S +2053/−0
- cbits/asm/chacha-armv8-linux64.S +2055/−0
- cbits/asm/chacha-armv8.pl +1328/−0
- cbits/asm/chacha-x86_64-elf.S +2241/−0
- cbits/asm/chacha-x86_64-macosx.S +2232/−0
- cbits/asm/chacha-x86_64-mingw64.S +2556/−0
- cbits/asm/chacha-x86_64.pl +4044/−0
- cbits/asm/generate.sh +153/−0
- cbits/asm/keccak1600-armv8-ios64.S +841/−0
- cbits/asm/keccak1600-armv8-linux64.S +843/−0
- cbits/asm/keccak1600-armv8.pl +932/−0
- cbits/asm/keccak1600-x86_64-elf.S +538/−0
- cbits/asm/keccak1600-x86_64-macosx.S +529/−0
- cbits/asm/keccak1600-x86_64-mingw64.S +648/−0
- cbits/asm/keccak1600-x86_64.pl +601/−0
- cbits/asm/poly1305-armv8-ios64.S +844/−0
- cbits/asm/poly1305-armv8-linux64.S +846/−0
- cbits/asm/poly1305-armv8.pl +927/−0
- cbits/asm/poly1305-x86_64-elf.S +2033/−0
- cbits/asm/poly1305-x86_64-macosx.S +2024/−0
- cbits/asm/poly1305-x86_64-mingw64.S +2281/−0
- cbits/asm/poly1305-x86_64.pl +4333/−0
- cbits/asm/sha1-armv8-ios64.S +1216/−0
- cbits/asm/sha1-armv8-linux64.S +1218/−0
- cbits/asm/sha1-armv8.pl +362/−0
- cbits/asm/sha256-armv8-ios64.S +2051/−0
- cbits/asm/sha256-armv8-linux64.S +2053/−0
- cbits/asm/sha256-x86_64-elf.S +5463/−0
- cbits/asm/sha256-x86_64-macosx.S +5454/−0
- cbits/asm/sha256-x86_64-mingw64.S +5731/−0
- cbits/asm/sha512-armv8.pl +892/−0
- cbits/asm/sha512-x86_64-elf.S +5727/−0
- cbits/asm/sha512-x86_64-macosx.S +5718/−0
- cbits/asm/sha512-x86_64-mingw64.S +6016/−0
- cbits/asm/sha512-x86_64.pl +2519/−0
- cbits/asm/x86_64-xlate.pl +1943/−0
- cbits/chacha_avx2.c +146/−0
- cbits/chacha_neon.c +145/−0
- cbits/chacha_sse2.c +105/−0
- cbits/chacha_sse_impl.c +114/−0
- cbits/crypton_aes.c +491/−48
- cbits/crypton_aes.h +72/−2
- cbits/crypton_align.h +75/−61
- cbits/crypton_armv8_target.h +40/−0
- cbits/crypton_bignum.h +512/−0
- cbits/crypton_blowfish.c +456/−0
- cbits/crypton_blowfish.h +44/−0
- cbits/crypton_bzero.h +27/−0
- cbits/crypton_camellia.c +697/−0
- cbits/crypton_camellia.h +21/−0
- cbits/crypton_chacha.c +122/−7
- cbits/crypton_chacha.h +2/−2
- cbits/crypton_chachapoly.c +156/−0
- cbits/crypton_chachapoly.h +62/−0
- cbits/crypton_cpu.c +240/−0
- cbits/crypton_cpu.h +62/−0
- cbits/crypton_des.c +1325/−0
- cbits/crypton_des.h +20/−0
- cbits/crypton_ecc.c +542/−0
- cbits/crypton_ecc.h +57/−0
- cbits/crypton_ecc_s2n.c +204/−0
- cbits/crypton_ecc_s2n.h +27/−0
- cbits/crypton_ecc_s2n_curves.h +76/−0
- cbits/crypton_f2m.c +555/−0
- cbits/crypton_f2m.h +31/−0
- cbits/crypton_md4.c +11/−19
- cbits/crypton_md5.c +18/−19
- cbits/crypton_memxor.c +29/−0
- cbits/crypton_memxor.h +8/−0
- cbits/crypton_modinv.c +74/−0
- cbits/crypton_modinv.h +22/−0
- cbits/crypton_pbkdf2.c +21/−4
- cbits/crypton_poly1305.c +84/−14
- cbits/crypton_poly1305.h +16/−3
- cbits/crypton_powm.c +311/−0
- cbits/crypton_powm.h +23/−0
- cbits/crypton_ripemd.c +11/−19
- cbits/crypton_salsa.c +5/−5
- cbits/crypton_salsa.h +3/−3
- cbits/crypton_scrypt.c +2/−2
- cbits/crypton_sha1.c +137/−14
- cbits/crypton_sha256.c +105/−14
- cbits/crypton_sha256.h +12/−0
- cbits/crypton_sha3.c +83/−17
- cbits/crypton_sha512.c +74/−15
- cbits/crypton_sha512.h +12/−0
- cbits/crypton_skein256.c +29/−27
- cbits/crypton_skein256.h +3/−3
- cbits/crypton_skein512.c +34/−35
- cbits/crypton_skein512.h +3/−3
- cbits/crypton_tiger.c +9/−16
- cbits/crypton_xsalsa.c +6/−6
- cbits/curve25519/x25519.c +88/−0
- cbits/curve25519/x25519.h +16/−0
- cbits/decaf/ed448goldilocks/decaf.c +4/−1
- cbits/decaf/include/word.h +1/−1
- cbits/ed25519/ed25519.c +33/−7
- cbits/ed25519/ed25519_s2n.c +65/−0
- cbits/ed25519/ed25519_s2n.h +14/−0
- cbits/include32/p256/p256.h +2/−16
- cbits/include32/p256/p256_gf.h +95/−85
- cbits/include64/p256/p256.h +2/−16
- cbits/include64/p256/p256_gf.h +129/−88
- cbits/include64/p256/p256_s2n.h +25/−0
- cbits/mldsa/COMMIT +2/−0
- cbits/mldsa/LICENSE +305/−0
- cbits/mldsa/README.md +60/−0
- cbits/mldsa/crypton_mldsa.c +31/−0
- cbits/mldsa/crypton_mldsa.h +41/−0
- cbits/mldsa/crypton_mldsa_asm.S +14/−0
- cbits/mldsa/import.sh +40/−0
- cbits/mldsa/mldsa_native.c +803/−0
- cbits/mldsa/mldsa_native.h +956/−0
- cbits/mldsa/mldsa_native_asm.S +830/−0
- cbits/mldsa/mldsa_native_config.h +855/−0
- cbits/mldsa/src/cbmc.h +233/−0
- cbits/mldsa/src/common.h +301/−0
- cbits/mldsa/src/context.h +152/−0
- cbits/mldsa/src/ct.c +21/−0
- cbits/mldsa/src/ct.h +373/−0
- cbits/mldsa/src/debug.c +75/−0
- cbits/mldsa/src/debug.h +125/−0
- cbits/mldsa/src/fips202/fips202.c +270/−0
- cbits/mldsa/src/fips202/fips202.h +224/−0
- cbits/mldsa/src/fips202/fips202x4.c +187/−0
- cbits/mldsa/src/fips202/fips202x4.h +125/−0
- cbits/mldsa/src/fips202/keccakf1600.c +510/−0
- cbits/mldsa/src/fips202/keccakf1600.h +110/−0
- cbits/mldsa/src/fips202/native/aarch64/auto.h +85/−0
- cbits/mldsa/src/fips202/native/aarch64/src/fips202_native_aarch64.h +69/−0
- cbits/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x1_scalar_aarch64_asm.S +378/−0
- cbits/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x1_v84a_aarch64_asm.S +207/−0
- cbits/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x2_v84a_aarch64_asm.S +262/−0
- cbits/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm.S +1080/−0
- cbits/mldsa/src/fips202/native/aarch64/src/keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm.S +990/−0
- cbits/mldsa/src/fips202/native/aarch64/src/keccakf1600_round_constants.c +47/−0
- cbits/mldsa/src/fips202/native/aarch64/x1_scalar.h +27/−0
- cbits/mldsa/src/fips202/native/aarch64/x1_v84a.h +36/−0
- cbits/mldsa/src/fips202/native/aarch64/x2_v84a.h +40/−0
- cbits/mldsa/src/fips202/native/aarch64/x4_v8a_scalar.h +32/−0
- cbits/mldsa/src/fips202/native/aarch64/x4_v8a_v84a_scalar.h +37/−0
- cbits/mldsa/src/fips202/native/api.h +129/−0
- cbits/mldsa/src/fips202/native/auto.h +35/−0
- cbits/mldsa/src/fips202/native/x86_64/keccak_f1600_x4_avx2.h +34/−0
- cbits/mldsa/src/fips202/native/x86_64/src/fips202_native_x86_64.h +45/−0
- cbits/mldsa/src/fips202/native/x86_64/src/keccak_f1600_x4_avx2_asm.S +488/−0
- cbits/mldsa/src/fips202/native/x86_64/src/keccakf1600_constants.c +52/−0
- cbits/mldsa/src/native/aarch64/meta.h +314/−0
- cbits/mldsa/src/native/aarch64/src/aarch64_zetas.c +248/−0
- cbits/mldsa/src/native/aarch64/src/arith_native_aarch64.h +367/−0
- cbits/mldsa/src/native/aarch64/src/mldsa_intt_aarch64_asm.S +786/−0
- cbits/mldsa/src/native/aarch64/src/mldsa_ntt_aarch64_asm.S +686/−0
- cbits/mldsa/src/native/aarch64/src/mldsa_pointwise_montgomery_aarch64_asm.S +106/−0
- cbits/mldsa/src/native/aarch64/src/mldsa_poly_caddq_aarch64_asm.S +69/−0
- cbits/mldsa/src/native/aarch64/src/mldsa_poly_chknorm_aarch64_asm.S +76/−0
- cbits/mldsa/src/native/aarch64/src/mldsa_poly_decompose_32_aarch64_asm.S +108/−0
- cbits/mldsa/src/native/aarch64/src/mldsa_poly_decompose_88_aarch64_asm.S +108/−0
- cbits/mldsa/src/native/aarch64/src/mldsa_poly_use_hint_32_aarch64_asm.S +125/−0
- cbits/mldsa/src/native/aarch64/src/mldsa_poly_use_hint_88_aarch64_asm.S +133/−0
- cbits/mldsa/src/native/aarch64/src/mldsa_polyvecl_pointwise_acc_montgomery_l4_aarch64_asm.S +157/−0
- cbits/mldsa/src/native/aarch64/src/mldsa_polyvecl_pointwise_acc_montgomery_l5_aarch64_asm.S +173/−0
- cbits/mldsa/src/native/aarch64/src/mldsa_polyvecl_pointwise_acc_montgomery_l7_aarch64_asm.S +205/−0
- cbits/mldsa/src/native/aarch64/src/mldsa_polyz_unpack_17_aarch64_asm.S +103/−0
- cbits/mldsa/src/native/aarch64/src/mldsa_polyz_unpack_19_aarch64_asm.S +100/−0
- cbits/mldsa/src/native/aarch64/src/mldsa_rej_uniform_aarch64_asm.S +222/−0
- cbits/mldsa/src/native/aarch64/src/mldsa_rej_uniform_eta2_aarch64_asm.S +170/−0
- cbits/mldsa/src/native/aarch64/src/mldsa_rej_uniform_eta4_aarch64_asm.S +163/−0
- cbits/mldsa/src/native/aarch64/src/polyz_unpack_table.c +52/−0
- cbits/mldsa/src/native/aarch64/src/rej_uniform_eta_table.c +547/−0
- cbits/mldsa/src/native/aarch64/src/rej_uniform_table.c +63/−0
- cbits/mldsa/src/native/api.h +617/−0
- cbits/mldsa/src/native/meta.h +24/−0
- cbits/mldsa/src/native/x86_64/meta.h +323/−0
- cbits/mldsa/src/native/x86_64/src/arith_native_x86_64.h +330/−0
- cbits/mldsa/src/native/x86_64/src/consts.c +157/−0
- cbits/mldsa/src/native/x86_64/src/consts.h +27/−0
- cbits/mldsa/src/native/x86_64/src/mldsa_intt_avx2_asm.S +2333/−0
- cbits/mldsa/src/native/x86_64/src/mldsa_ntt_avx2_asm.S +2405/−0
- cbits/mldsa/src/native/x86_64/src/mldsa_nttunpack_avx2_asm.S +254/−0
- cbits/mldsa/src/native/x86_64/src/mldsa_pointwise_acc_l4_avx2_asm.S +173/−0
- cbits/mldsa/src/native/x86_64/src/mldsa_pointwise_acc_l5_avx2_asm.S +189/−0
- cbits/mldsa/src/native/x86_64/src/mldsa_pointwise_acc_l7_avx2_asm.S +221/−0
- cbits/mldsa/src/native/x86_64/src/mldsa_pointwise_avx2_asm.S +158/−0
- cbits/mldsa/src/native/x86_64/src/mldsa_poly_caddq_avx2_asm.S +199/−0
- cbits/mldsa/src/native/x86_64/src/mldsa_poly_chknorm_avx2_asm.S +176/−0
- cbits/mldsa/src/native/x86_64/src/mldsa_poly_decompose_32_avx2_asm.S +490/−0
- cbits/mldsa/src/native/x86_64/src/mldsa_poly_decompose_88_avx2_asm.S +489/−0
- cbits/mldsa/src/native/x86_64/src/mldsa_poly_use_hint_32_avx2_asm.S +123/−0
- cbits/mldsa/src/native/x86_64/src/mldsa_poly_use_hint_88_avx2_asm.S +125/−0
- cbits/mldsa/src/native/x86_64/src/mldsa_polyz_unpack_17_avx2_asm.S +355/−0
- cbits/mldsa/src/native/x86_64/src/mldsa_polyz_unpack_19_avx2_asm.S +355/−0
- cbits/mldsa/src/native/x86_64/src/mldsa_rej_uniform_avx2_asm.S +132/−0
- cbits/mldsa/src/native/x86_64/src/mldsa_rej_uniform_eta2_avx2_asm.S +205/−0
- cbits/mldsa/src/native/x86_64/src/mldsa_rej_uniform_eta4_avx2_asm.S +176/−0
- cbits/mldsa/src/native/x86_64/src/rej_uniform_table.c +161/−0
- cbits/mldsa/src/packing.c +213/−0
- cbits/mldsa/src/packing.h +277/−0
- cbits/mldsa/src/params.h +153/−0
- cbits/mldsa/src/poly.c +1066/−0
- cbits/mldsa/src/poly.h +464/−0
- cbits/mldsa/src/poly_kl.c +910/−0
- cbits/mldsa/src/poly_kl.h +367/−0
- cbits/mldsa/src/polyvec.c +509/−0
- cbits/mldsa/src/polyvec.h +435/−0
- cbits/mldsa/src/polyvec_lazy.c +311/−0
- cbits/mldsa/src/polyvec_lazy.h +652/−0
- cbits/mldsa/src/randombytes.h +26/−0
- cbits/mldsa/src/reduce.h +144/−0
- cbits/mldsa/src/rounding.h +265/−0
- cbits/mldsa/src/sign.c +1720/−0
- cbits/mldsa/src/sign.h +850/−0
- cbits/mldsa/src/symmetric.h +68/−0
- cbits/mldsa/src/sys.h +327/−0
- cbits/mldsa/src/zetas.inc +55/−0
- cbits/mlkem/COMMIT +2/−0
- cbits/mlkem/LICENSE +312/−0
- cbits/mlkem/README.md +59/−0
- cbits/mlkem/crypton_mlkem.c +31/−0
- cbits/mlkem/crypton_mlkem.h +41/−0
- cbits/mlkem/crypton_mlkem_asm.S +14/−0
- cbits/mlkem/import.sh +42/−0
- cbits/mlkem/mlkem_native.c +692/−0
- cbits/mlkem/mlkem_native.h +464/−0
- cbits/mlkem/mlkem_native_asm.S +716/−0
- cbits/mlkem/mlkem_native_config.h +683/−0
- cbits/mlkem/src/cbmc.h +222/−0
- cbits/mlkem/src/common.h +296/−0
- cbits/mlkem/src/compress.c +763/−0
- cbits/mlkem/src/compress.h +613/−0
- cbits/mlkem/src/context.h +51/−0
- cbits/mlkem/src/debug.c +64/−0
- cbits/mlkem/src/debug.h +121/−0
- cbits/mlkem/src/fips202/fips202.c +249/−0
- cbits/mlkem/src/fips202/fips202.h +144/−0
- cbits/mlkem/src/fips202/fips202x4.c +207/−0
- cbits/mlkem/src/fips202/fips202x4.h +81/−0
- cbits/mlkem/src/fips202/keccakf1600.c +499/−0
- cbits/mlkem/src/fips202/keccakf1600.h +98/−0
- cbits/mlkem/src/fips202/native/aarch64/auto.h +78/−0
- cbits/mlkem/src/fips202/native/aarch64/src/fips202_native_aarch64.h +80/−0
- cbits/mlkem/src/fips202/native/aarch64/src/keccak_f1600_x1_scalar_aarch64_asm.S +377/−0
- cbits/mlkem/src/fips202/native/aarch64/src/keccak_f1600_x1_v84a_aarch64_asm.S +206/−0
- cbits/mlkem/src/fips202/native/aarch64/src/keccak_f1600_x2_v84a_aarch64_asm.S +261/−0
- cbits/mlkem/src/fips202/native/aarch64/src/keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm.S +1079/−0
- cbits/mlkem/src/fips202/native/aarch64/src/keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm.S +989/−0
- cbits/mlkem/src/fips202/native/aarch64/src/keccakf1600_round_constants.c +47/−0
- cbits/mlkem/src/fips202/native/aarch64/x1_scalar.h +26/−0
- cbits/mlkem/src/fips202/native/aarch64/x1_v84a.h +35/−0
- cbits/mlkem/src/fips202/native/aarch64/x2_v84a.h +38/−0
- cbits/mlkem/src/fips202/native/aarch64/x4_v8a_scalar.h +31/−0
- cbits/mlkem/src/fips202/native/aarch64/x4_v8a_v84a_scalar.h +36/−0
- cbits/mlkem/src/fips202/native/api.h +117/−0
- cbits/mlkem/src/fips202/native/auto.h +29/−0
- cbits/mlkem/src/fips202/native/x86_64/keccak_f1600_x4_avx2.h +33/−0
- cbits/mlkem/src/fips202/native/x86_64/src/fips202_native_x86_64.h +44/−0
- cbits/mlkem/src/fips202/native/x86_64/src/keccak_f1600_x4_avx2_asm.S +487/−0
- cbits/mlkem/src/fips202/native/x86_64/src/keccakf1600_constants.c +52/−0
- cbits/mlkem/src/indcpa.c +675/−0
- cbits/mlkem/src/indcpa.h +172/−0
- cbits/mlkem/src/kem.c +474/−0
- cbits/mlkem/src/kem.h +353/−0
- cbits/mlkem/src/native/aarch64/README.md +16/−0
- cbits/mlkem/src/native/aarch64/meta.h +166/−0
- cbits/mlkem/src/native/aarch64/src/aarch64_zetas.c +184/−0
- cbits/mlkem/src/native/aarch64/src/arith_native_aarch64.h +184/−0
- cbits/mlkem/src/native/aarch64/src/mlkem_intt_aarch64_asm.S +635/−0
- cbits/mlkem/src/native/aarch64/src/mlkem_ntt_aarch64_asm.S +565/−0
- cbits/mlkem/src/native/aarch64/src/mlkem_poly_mulcache_compute_aarch64_asm.S +130/−0
- cbits/mlkem/src/native/aarch64/src/mlkem_poly_reduce_aarch64_asm.S +153/−0
- cbits/mlkem/src/native/aarch64/src/mlkem_poly_tobytes_aarch64_asm.S +124/−0
- cbits/mlkem/src/native/aarch64/src/mlkem_poly_tomont_aarch64_asm.S +102/−0
- cbits/mlkem/src/native/aarch64/src/mlkem_polyvec_basemul_acc_montgomery_cached_k2_aarch64_asm.S +264/−0
- cbits/mlkem/src/native/aarch64/src/mlkem_polyvec_basemul_acc_montgomery_cached_k3_aarch64_asm.S +317/−0
- cbits/mlkem/src/native/aarch64/src/mlkem_polyvec_basemul_acc_montgomery_cached_k4_aarch64_asm.S +371/−0
- cbits/mlkem/src/native/aarch64/src/mlkem_rej_uniform_aarch64_asm.S +226/−0
- cbits/mlkem/src/native/aarch64/src/rej_uniform_table.c +543/−0
- cbits/mlkem/src/native/api.h +651/−0
- cbits/mlkem/src/native/meta.h +30/−0
- cbits/mlkem/src/native/x86_64/README.md +4/−0
- cbits/mlkem/src/native/x86_64/meta.h +324/−0
- cbits/mlkem/src/native/x86_64/src/arith_native_x86_64.h +327/−0
- cbits/mlkem/src/native/x86_64/src/compress_consts.c +115/−0
- cbits/mlkem/src/native/x86_64/src/compress_consts.h +53/−0
- cbits/mlkem/src/native/x86_64/src/consts.c +102/−0
- cbits/mlkem/src/native/x86_64/src/consts.h +25/−0
- cbits/mlkem/src/native/x86_64/src/mlkem_intt_avx2_asm.S +743/−0
- cbits/mlkem/src/native/x86_64/src/mlkem_ntt_avx2_asm.S too large to diff
- cbits/mlkem/src/native/x86_64/src/mlkem_nttfrombytes_avx2_asm.S too large to diff
- cbits/mlkem/src/native/x86_64/src/mlkem_ntttobytes_avx2_asm.S too large to diff
- cbits/mlkem/src/native/x86_64/src/mlkem_nttunpack_avx2_asm.S too large to diff
- cbits/mlkem/src/native/x86_64/src/mlkem_poly_compress_d10_avx2_asm.S too large to diff
- cbits/mlkem/src/native/x86_64/src/mlkem_poly_compress_d11_avx2_asm.S too large to diff
- cbits/mlkem/src/native/x86_64/src/mlkem_poly_compress_d4_avx2_asm.S too large to diff
- cbits/mlkem/src/native/x86_64/src/mlkem_poly_compress_d5_avx2_asm.S too large to diff
- cbits/mlkem/src/native/x86_64/src/mlkem_poly_decompress_d10_avx2_asm.S too large to diff
- cbits/mlkem/src/native/x86_64/src/mlkem_poly_decompress_d11_avx2_asm.S too large to diff
- cbits/mlkem/src/native/x86_64/src/mlkem_poly_decompress_d4_avx2_asm.S too large to diff
- cbits/mlkem/src/native/x86_64/src/mlkem_poly_decompress_d5_avx2_asm.S too large to diff
- cbits/mlkem/src/native/x86_64/src/mlkem_poly_mulcache_compute_avx2_asm.S too large to diff
- cbits/mlkem/src/native/x86_64/src/mlkem_polyvec_basemul_acc_montgomery_cached_k2_avx2_asm.S too large to diff
- cbits/mlkem/src/native/x86_64/src/mlkem_polyvec_basemul_acc_montgomery_cached_k3_avx2_asm.S too large to diff
- cbits/mlkem/src/native/x86_64/src/mlkem_polyvec_basemul_acc_montgomery_cached_k4_avx2_asm.S too large to diff
- cbits/mlkem/src/native/x86_64/src/mlkem_reduce_avx2_asm.S too large to diff
- cbits/mlkem/src/native/x86_64/src/mlkem_rej_uniform_avx2_asm.S too large to diff
- cbits/mlkem/src/native/x86_64/src/mlkem_tomont_avx2_asm.S too large to diff
- cbits/mlkem/src/native/x86_64/src/rej_uniform_table.c too large to diff
- cbits/mlkem/src/params.h too large to diff
- cbits/mlkem/src/poly.c too large to diff
- cbits/mlkem/src/poly.h too large to diff
- cbits/mlkem/src/poly_k.c too large to diff
- cbits/mlkem/src/poly_k.h too large to diff
- cbits/mlkem/src/randombytes.h too large to diff
- cbits/mlkem/src/sampling.c too large to diff
- cbits/mlkem/src/sampling.h too large to diff
- cbits/mlkem/src/symmetric.h too large to diff
- cbits/mlkem/src/sys.h too large to diff
- cbits/mlkem/src/verify.c too large to diff
- cbits/mlkem/src/verify.h too large to diff
- cbits/mlkem/src/zetas.inc too large to diff
- cbits/p256/gen_base_table.py too large to diff
- cbits/p256/p256.c too large to diff
- cbits/p256/p256_base_table.c too large to diff
- cbits/p256/p256_ec.c too large to diff
- cbits/p256/p256_s2n.c too large to diff
- cbits/p256/p256_verify.c too large to diff
- cbits/p256/p256_verify.h too large to diff
- cbits/p256/p256_wnaf_table.c too large to diff
- cbits/s2n/COMMIT too large to diff
- cbits/s2n/LICENSE too large to diff
- cbits/s2n/README.md too large to diff
- cbits/s2n/arm/bignum_deamont_p384.S too large to diff
- cbits/s2n/arm/bignum_demont_p256.S too large to diff
- cbits/s2n/arm/bignum_inv_p521.S too large to diff
- cbits/s2n/arm/bignum_modinv.S too large to diff
- cbits/s2n/arm/bignum_montinv_p384.S too large to diff
- cbits/s2n/arm/bignum_montmul_p384.S too large to diff
- cbits/s2n/arm/bignum_montmul_p384_alt.S too large to diff
- cbits/s2n/arm/bignum_montsqr_p384.S too large to diff
- cbits/s2n/arm/bignum_montsqr_p384_alt.S too large to diff
- cbits/s2n/arm/bignum_mul_p521.S too large to diff
- cbits/s2n/arm/bignum_mul_p521_alt.S too large to diff
- cbits/s2n/arm/bignum_neg_p256.S too large to diff
- cbits/s2n/arm/bignum_sqr_p521.S too large to diff
- cbits/s2n/arm/bignum_sqr_p521_alt.S too large to diff
- cbits/s2n/arm/bignum_tomont_p256.S too large to diff
- cbits/s2n/arm/bignum_tomont_p384.S too large to diff
- cbits/s2n/arm/curve25519_x25519.S too large to diff
- cbits/s2n/arm/curve25519_x25519_alt.S too large to diff
- cbits/s2n/arm/curve25519_x25519base.S too large to diff
- cbits/s2n/arm/curve25519_x25519base_alt.S too large to diff
- cbits/s2n/arm/edwards25519_encode.S too large to diff
- cbits/s2n/arm/edwards25519_scalarmulbase.S too large to diff
- cbits/s2n/arm/edwards25519_scalarmulbase_alt.S too large to diff
- cbits/s2n/arm/p256_montjadd.S too large to diff
- cbits/s2n/arm/p256_montjadd_alt.S too large to diff
- cbits/s2n/arm/p256_montjdouble.S too large to diff
- cbits/s2n/arm/p256_montjdouble_alt.S too large to diff
- cbits/s2n/arm/p256_montjmixadd.S too large to diff
- cbits/s2n/arm/p256_montjmixadd_alt.S too large to diff
- cbits/s2n/arm/p256_scalarmul.S too large to diff
- cbits/s2n/arm/p256_scalarmul_alt.S too large to diff
- cbits/s2n/arm/p256_scalarmulbase.S too large to diff
- cbits/s2n/arm/p256_scalarmulbase_alt.S too large to diff
- cbits/s2n/arm/p384_montjscalarmul.S too large to diff
- cbits/s2n/arm/p384_montjscalarmul_alt.S too large to diff
- cbits/s2n/arm/p521_jscalarmul.S too large to diff
- cbits/s2n/arm/p521_jscalarmul_alt.S too large to diff
- cbits/s2n/import.sh too large to diff
- cbits/s2n/include/_internal_s2n_bignum_arm.h too large to diff
- cbits/s2n/include/_internal_s2n_bignum_x86_att.h too large to diff
- cbits/s2n/x86_att/bignum_deamont_p384.S too large to diff
- cbits/s2n/x86_att/bignum_deamont_p384_alt.S too large to diff
- cbits/s2n/x86_att/bignum_demont_p256.S too large to diff
- cbits/s2n/x86_att/bignum_demont_p256_alt.S too large to diff
- cbits/s2n/x86_att/bignum_emontredc_8n.S too large to diff
- cbits/s2n/x86_att/bignum_inv_p521.S too large to diff
- cbits/s2n/x86_att/bignum_kmul_16_32.S too large to diff
- cbits/s2n/x86_att/bignum_kmul_32_64.S too large to diff
- cbits/s2n/x86_att/bignum_ksqr_16_32.S too large to diff
- cbits/s2n/x86_att/bignum_ksqr_32_64.S too large to diff
- cbits/s2n/x86_att/bignum_modinv.S too large to diff
- cbits/s2n/x86_att/bignum_montinv_p384.S too large to diff
- cbits/s2n/x86_att/bignum_montmul_p384.S too large to diff
- cbits/s2n/x86_att/bignum_montmul_p384_alt.S too large to diff
- cbits/s2n/x86_att/bignum_montsqr_p384.S too large to diff
- cbits/s2n/x86_att/bignum_montsqr_p384_alt.S too large to diff
- cbits/s2n/x86_att/bignum_mul_p521.S too large to diff
- cbits/s2n/x86_att/bignum_mul_p521_alt.S too large to diff
- cbits/s2n/x86_att/bignum_neg_p256.S too large to diff
- cbits/s2n/x86_att/bignum_sqr_p521.S too large to diff
- cbits/s2n/x86_att/bignum_sqr_p521_alt.S too large to diff
- cbits/s2n/x86_att/bignum_tomont_p256.S too large to diff
- cbits/s2n/x86_att/bignum_tomont_p256_alt.S too large to diff
- cbits/s2n/x86_att/bignum_tomont_p384.S too large to diff
- cbits/s2n/x86_att/bignum_tomont_p384_alt.S too large to diff
- cbits/s2n/x86_att/curve25519_x25519.S too large to diff
- cbits/s2n/x86_att/curve25519_x25519_alt.S too large to diff
- cbits/s2n/x86_att/curve25519_x25519base.S too large to diff
- cbits/s2n/x86_att/curve25519_x25519base_alt.S too large to diff
- cbits/s2n/x86_att/edwards25519_encode.S too large to diff
- cbits/s2n/x86_att/edwards25519_scalarmulbase.S too large to diff
- cbits/s2n/x86_att/edwards25519_scalarmulbase_alt.S too large to diff
- cbits/s2n/x86_att/p256_montjadd.S too large to diff
- cbits/s2n/x86_att/p256_montjadd_alt.S too large to diff
- cbits/s2n/x86_att/p256_montjdouble.S too large to diff
- cbits/s2n/x86_att/p256_montjdouble_alt.S too large to diff
- cbits/s2n/x86_att/p256_montjmixadd.S too large to diff
- cbits/s2n/x86_att/p256_montjmixadd_alt.S too large to diff
- cbits/s2n/x86_att/p256_scalarmul.S too large to diff
- cbits/s2n/x86_att/p256_scalarmul_alt.S too large to diff
- cbits/s2n/x86_att/p256_scalarmulbase.S too large to diff
- cbits/s2n/x86_att/p256_scalarmulbase_alt.S too large to diff
- cbits/s2n/x86_att/p384_montjscalarmul.S too large to diff
- cbits/s2n/x86_att/p384_montjscalarmul_alt.S too large to diff
- cbits/s2n/x86_att/p521_jscalarmul.S too large to diff
- cbits/s2n/x86_att/p521_jscalarmul_alt.S too large to diff
- cbits/sha1_armv8.c too large to diff
- cbits/sha1_x86.c too large to diff
- cbits/sha256_armv8.c too large to diff
- cbits/sha3_armv8.c too large to diff
- cbits/sha512_armv8.c too large to diff
- cbits/tests/ct/README too large to diff
- cbits/tests/ct/ct.h too large to diff
- cbits/tests/ct/ct_aes.c too large to diff
- cbits/tests/ct/ct_aes_armv8.c too large to diff
- cbits/tests/ct/ct_canary.c too large to diff
- cbits/tests/ct/ct_chapoly.c too large to diff
- cbits/tests/ct/ct_decaf.c too large to diff
- cbits/tests/ct/ct_ed25519.c too large to diff
- cbits/tests/ct/ct_mlkem.c too large to diff
- cbits/tests/ct/ct_p256.c too large to diff
- cbits/tests/ct/ct_powm.c too large to diff
- cbits/tests/ct/ct_x25519.c too large to diff
- cbits/tests/ct/known.txt too large to diff
- cbits/tests/ct/run.sh too large to diff
- cbits/tests/endian/README too large to diff
- cbits/tests/endian/endian.c too large to diff
- cbits/tests/endian/run.sh too large to diff
- cbits/tests/endian/vectors.txt too large to diff
- cbits/tests/fuzz/README too large to diff
- cbits/tests/fuzz/corpus/aead-authentic.bin too large to diff
- cbits/tests/fuzz/corpus/ed25519-tampered.bin too large to diff
- cbits/tests/fuzz/corpus/ed25519-valid.bin too large to diff
- cbits/tests/fuzz/corpus/p256-on-curve.bin too large to diff
- cbits/tests/fuzz/fuzz.h too large to diff
- cbits/tests/fuzz/fuzz_aead.c too large to diff
- cbits/tests/fuzz/fuzz_canary.c too large to diff
- cbits/tests/fuzz/fuzz_decaf.c too large to diff
- cbits/tests/fuzz/fuzz_ed25519.c too large to diff
- cbits/tests/fuzz/fuzz_p256.c too large to diff
- cbits/tests/fuzz/run.sh too large to diff
- cbits/tests/fuzz/standalone.c too large to diff
- cbits/tests/perf/floors.txt too large to diff
- cbits/tests/perf/run.sh too large to diff
- cbits/tests/perf/throughput.c too large to diff
- cbits/tests/scrub/README too large to diff
- cbits/tests/scrub/known.txt too large to diff
- cbits/tests/scrub/run.sh too large to diff
- cbits/tests/scrub/scrub.c too large to diff
- cbits/tests/width/ed448_width.c too large to diff
- cbits/tests/width/p256_width.c too large to diff
- cbits/tests/width/run.sh too large to diff
- cbits/tests/width/x25519_width.c too large to diff
- crypton.cabal too large to diff
- tests/AFISSpec.hs too large to diff
- tests/BCrypt.hs too large to diff
- tests/BCryptPBKDF.hs too large to diff
- tests/BlockCipher.hs too large to diff
- tests/BlockCipher/AES/CBC.hs too large to diff
- tests/BlockCipher/AES/CCM.hs too large to diff
- tests/BlockCipher/AES/CTR.hs too large to diff
- tests/BlockCipher/AES/ECB.hs too large to diff
- tests/BlockCipher/AES/GCM.hs too large to diff
- tests/BlockCipher/AES/GCMLong.hs too large to diff
- tests/BlockCipher/AES/OCB3.hs too large to diff
- tests/BlockCipher/AES/XTS.hs too large to diff
- tests/BlockCipher/AESGCMSIVSpec.hs too large to diff
- tests/BlockCipher/AESSpec.hs too large to diff
- tests/BlockCipher/BlowfishSpec.hs too large to diff
- tests/BlockCipher/CAST5Spec.hs too large to diff
- tests/BlockCipher/CamelliaSpec.hs too large to diff
- tests/BlockCipher/DESSpec.hs too large to diff
- tests/BlockCipher/ModesSpec.hs too large to diff
- tests/BlockCipher/TripleDESSpec.hs too large to diff
- tests/BlockCipher/TwofishSpec.hs too large to diff
- tests/ChaCha.hs too large to diff
- tests/ChaChaPoly1305.hs too large to diff
- tests/ConstructHash/MiyaguchiPreneelSpec.hs too large to diff
- tests/Curve25519Spec.hs too large to diff
- tests/Curve448Spec.hs too large to diff
- tests/ECC.hs too large to diff
- tests/ECC/Edwards25519.hs too large to diff
- tests/ECC/Edwards25519Spec.hs too large to diff
- tests/ECCSpec.hs too large to diff
- tests/ECDSA.hs too large to diff
- tests/ECDSASpec.hs too large to diff
- tests/Ed25519Spec.hs too large to diff
- tests/Ed448Spec.hs too large to diff
- tests/EdDSASpec.hs too large to diff
- tests/Hash.hs too large to diff
- tests/HashSpec.hs too large to diff
- tests/Imports.hs too large to diff
- tests/KAT_AES.hs too large to diff
- tests/KAT_AES/KATCBC.hs too large to diff
- tests/KAT_AES/KATCCM.hs too large to diff
- tests/KAT_AES/KATECB.hs too large to diff
- tests/KAT_AES/KATGCM.hs too large to diff
- tests/KAT_AES/KATOCB3.hs too large to diff
- tests/KAT_AES/KATXTS.hs too large to diff
- tests/KAT_AESGCMSIV.hs too large to diff
- tests/KAT_AFIS.hs too large to diff
- tests/KAT_Argon2.hs too large to diff
- tests/KAT_Blake2.hs too large to diff
- tests/KAT_Blowfish.hs too large to diff
- tests/KAT_CAST5.hs too large to diff
- tests/KAT_CMAC.hs too large to diff
- tests/KAT_Camellia.hs too large to diff
- tests/KAT_Curve25519.hs too large to diff
- tests/KAT_Curve448.hs too large to diff
- tests/KAT_DES.hs too large to diff
- tests/KAT_Ed25519.hs too large to diff
- tests/KAT_Ed448.hs too large to diff
- tests/KAT_EdDSA.hs too large to diff
- tests/KAT_HKDF.hs too large to diff
- tests/KAT_HMAC.hs too large to diff
- tests/KAT_KMAC.hs too large to diff
- tests/KAT_MiyaguchiPreneel.hs too large to diff
- tests/KAT_OTP.hs too large to diff
- tests/KAT_PBKDF2.hs too large to diff
- tests/KAT_PubKey.hs too large to diff
- tests/KAT_PubKey/DSA.hs too large to diff
- tests/KAT_PubKey/ECC.hs too large to diff
- tests/KAT_PubKey/ECDSA.hs too large to diff
- tests/KAT_PubKey/OAEP.hs too large to diff
- tests/KAT_PubKey/P256.hs too large to diff
- tests/KAT_PubKey/PSS.hs too large to diff
- tests/KAT_PubKey/RSA.hs too large to diff
- tests/KAT_PubKey/Rabin.hs too large to diff
- tests/KAT_RC4.hs too large to diff
- tests/KAT_Scrypt.hs too large to diff
- tests/KAT_TripleDES.hs too large to diff
- tests/KAT_Twofish.hs too large to diff
- tests/KDF/Argon2Spec.hs too large to diff
- tests/KDF/BCryptPBKDFSpec.hs too large to diff
- tests/KDF/BCryptSpec.hs too large to diff
- tests/KDF/HKDFSpec.hs too large to diff
- tests/KDF/PBKDF2Spec.hs too large to diff
- tests/KDF/ScryptSpec.hs too large to diff
- tests/MAC/Blake2Spec.hs too large to diff
- tests/MAC/CMACSpec.hs too large to diff
- tests/MAC/HMACSpec.hs too large to diff
- tests/MAC/KMACSpec.hs too large to diff
- tests/MAC/Poly1305Spec.hs too large to diff
- tests/MAC/Poly1305Vectors.hs too large to diff
- tests/Number.hs too large to diff
- tests/Number/F2m.hs too large to diff
- tests/Number/F2mSpec.hs too large to diff
- tests/NumberSpec.hs too large to diff
- tests/OTPSpec.hs too large to diff
- tests/Padding.hs too large to diff
- tests/PaddingSpec.hs too large to diff
- tests/Poly1305.hs too large to diff
- tests/PubKey/DHSpec.hs too large to diff
- tests/PubKey/DSASpec.hs too large to diff
- tests/PubKey/ECCSpec.hs too large to diff
- tests/PubKey/ECDSASpec.hs too large to diff
- tests/PubKey/ElGamalSpec.hs too large to diff
- tests/PubKey/MGF1Spec.hs too large to diff
- tests/PubKey/MLDSASpec.hs too large to diff
- tests/PubKey/MLDSAVectors.hs too large to diff
- tests/PubKey/MLKEMSpec.hs too large to diff
- tests/PubKey/MLKEMVectors.hs too large to diff
- tests/PubKey/OAEPSpec.hs too large to diff
- tests/PubKey/P256Spec.hs too large to diff
- tests/PubKey/PSSSpec.hs too large to diff
- tests/PubKey/RSASpec.hs too large to diff
- tests/PubKey/RabinSpec.hs too large to diff
- tests/PubKey/SecrecySpec.hs too large to diff
- tests/RuntimeSpec.hs too large to diff
- tests/Salsa.hs too large to diff
- tests/Spec.hs too large to diff
- tests/StreamCipher/ChaChaPoly1305Spec.hs too large to diff
- tests/StreamCipher/ChaChaSpec.hs too large to diff
- tests/StreamCipher/RC4Spec.hs too large to diff
- tests/StreamCipher/SalsaSpec.hs too large to diff
- tests/StreamCipher/XSalsaSpec.hs too large to diff
- tests/Tests.hs too large to diff
- tests/Utils.hs too large to diff
- tests/XSalsa.hs too large to diff
@@ -1,5 +1,489 @@ # CHANGELOG for crypton +## 2.1.9++* feat(rsa): PKCS#1 v1.5 operations that take a digest, and ones that take a DigestInfo+ [#305](https://github.com/kazu-yamamoto/crypton/pull/305)+* feat(dsa): sign and verify over a digest+ [#305](https://github.com/kazu-yamamoto/crypton/pull/305)+* feat(elgamal): sign and verify over a digest+ [#305](https://github.com/kazu-yamamoto/crypton/pull/305)+* feat(rabin): Rabin-Williams and Modified Rabin sign and verify over a digest+ [#305](https://github.com/kazu-yamamoto/crypton/pull/305)++## 2.1.8++* chore: stop hiding foldl' from Prelude+ [#294](https://github.com/kazu-yamamoto/crypton/pull/294)+* chore: ask for hidden visibility only where the format has it+ [#295](https://github.com/kazu-yamamoto/crypton/pull/295)+* feat: ML-KEM and ML-DSA, through mlkem-native and mldsa-native+ [#297](https://github.com/kazu-yamamoto/crypton/pull/297)++## 2.1.7++RSA-PSS verification was accepting an encoding RFC 8017 says to refuse. It+is a conformance fault rather than a forgery: producing such a signature takes+the private key, so a third party holding a valid signature cannot turn it+into one of these.++* chore(p256): drop two declarations nothing defines+ [#292](https://github.com/kazu-yamamoto/crypton/pull/292)+* fix(pss): refuse an encoding with a bit set outside emBits+ [#293](https://github.com/kazu-yamamoto/crypton/pull/293)++## 2.1.6++The `license:` field now says what the tree holds -- `BSD-3-Clause AND MIT+AND ISC` -- and `license-files:` lists all five texts. **Nothing is required+of a user that was not required before**; the field was simply incomplete.++* Carry the MIT notice for the parts that follow fusion, and say what the tree holds+ [#267](https://github.com/kazu-yamamoto/crypton/pull/267)+* fix: two preconditions the Haskell layer did not enforce+ [#288](https://github.com/kazu-yamamoto/crypton/pull/288)+* fix(x86): stop casting a packed block to __m128i *+ [#289](https://github.com/kazu-yamamoto/crypton/pull/289)+* fix(c): say that the digest pointers are never null+ [#290](https://github.com/kazu-yamamoto/crypton/pull/290)+* fix(pbkdf2): refuse a digest larger than its block at compile time+ [#291](https://github.com/kazu-yamamoto/crypton/pull/291)++## 2.1.5++2.1.3 and 2.1.4 cannot be built with GCC 14 or newer. This release is+that fix.++* Watch for a primitive fallen off its fast path+ [#277](https://github.com/kazu-yamamoto/crypton/pull/277)+* The performance tables for 2.1.4+ [#278](https://github.com/kazu-yamamoto/crypton/pull/278)+* The x86-64 table, on the machine the README names+ [#279](https://github.com/kazu-yamamoto/crypton/pull/279)+* perf(armv8): GHASH against a twisted H, a quarter faster+ [#281](https://github.com/kazu-yamamoto/crypton/pull/281)+* ct: run the constant-time harness on AArch64, and put its AES to it+ [#283](https://github.com/kazu-yamamoto/crypton/pull/283)+* Fix crypton_sha1_x86_do_chunk signature+ [#284](https://github.com/kazu-yamamoto/crypton/pull/284)+* ci: build the C with a compiler stricter than this matrix has+ [#285](https://github.com/kazu-yamamoto/crypton/pull/285)++## 2.1.4++2.1.3 could not be built from Hackage in the default configuration, and is+deprecated there. This release is that fix.++* Add p256 header files to cabal extra-source-files+ [#271](https://github.com/kazu-yamamoto/crypton/pull/271)+* Build what is published, not only what is checked out+ [#272](https://github.com/kazu-yamamoto/crypton/pull/272)+* SHA-256 on AArch64 is 5.4x slower in 2.1.3 than in 2.1.2+ [#274](https://github.com/kazu-yamamoto/crypton/pull/274)+* The target attribute spelling gcc before 13 understands+ [#276](https://github.com/kazu-yamamoto/crypton/pull/276)++## 2.1.3++* fix(aead): reject missing and oversized authentication tags+ [#233](https://github.com/kazu-yamamoto/crypton/pull/233)+* fix(chacha): remove redundant Word8 import+ [#234](https://github.com/kazu-yamamoto/crypton/pull/234)+* test(number): cover the two modulus sizes the assembly runs at+ [#235](https://github.com/kazu-yamamoto/crypton/pull/235)+* perf(rsa): swap the buffers instead of copying them back+ [#236](https://github.com/kazu-yamamoto/crypton/pull/236)+* perf(rsa): scan the exponentiation's table four limbs at a time+ [#237](https://github.com/kazu-yamamoto/crypton/pull/237)+* fix(gcm): write the field doubling from its definition+ [#238](https://github.com/kazu-yamamoto/crypton/pull/238)+* doc(gcm): carry the MIT notice for the parts that follow fusion+ [#239](https://github.com/kazu-yamamoto/crypton/pull/239)+* perf(ed25519): the base point multiplication through s2n-bignum+ [#240](https://github.com/kazu-yamamoto/crypton/pull/240)+* perf(gcm): AES-GCM through the 512-bit VAES and VPCLMULQDQ+ [#241](https://github.com/kazu-yamamoto/crypton/pull/241)+* doc: the README said AVX-512 was not used, and it is+ [#242](https://github.com/kazu-yamamoto/crypton/pull/242)+* perf(ecdsa): P-256 verification multiplies both scalars at once+ [#243](https://github.com/kazu-yamamoto/crypton/pull/243)+* perf(rsa): four limbs and two carry chains on AArch64+ [#244](https://github.com/kazu-yamamoto/crypton/pull/244)+* perf(rsa): write out the ragged end of the AArch64 row+ [#245](https://github.com/kazu-yamamoto/crypton/pull/245)+* perf(rsa): build R^2 by squaring, not by doubling+ [#246](https://github.com/kazu-yamamoto/crypton/pull/246)+* perf(rsa): stop clearing the scratch a Montgomery multiply writes over+ [#247](https://github.com/kazu-yamamoto/crypton/pull/247)+* doc(gcm): write down what the AArch64 GHASH is short of+ [#248](https://github.com/kazu-yamamoto/crypton/pull/248)+* security(aes): refuse a nonce of no bytes in Crypto.Cipher.AES.GCM+ [#250](https://github.com/kazu-yamamoto/crypton/pull/250)+* ci: build and test the C the other architectures use+ [#251](https://github.com/kazu-yamamoto/crypton/pull/251)+* perf(p256): five teeth to a comb block, over the signed representation+ [#252](https://github.com/kazu-yamamoto/crypton/pull/252)+* security(cipher): stop truncating message lengths on the way to the C+ [#253](https://github.com/kazu-yamamoto/crypton/pull/253)+* fix(c): two left shifts the standard leaves undefined+ [#254](https://github.com/kazu-yamamoto/crypton/pull/254)+* ci: run the C under the sanitizers+ [#255](https://github.com/kazu-yamamoto/crypton/pull/255)+* fix(c): read words out of a block rather than pointing at it+ [#256](https://github.com/kazu-yamamoto/crypton/pull/256)+* fix(internal): drop an import nothing uses any more+ [#257](https://github.com/kazu-yamamoto/crypton/pull/257)+* fix(c): decide the dispatch table once, not on every key+ [#258](https://github.com/kazu-yamamoto/crypton/pull/258)+* Fix the three cabal flag settings that were broken+ [#259](https://github.com/kazu-yamamoto/crypton/pull/259)+* Run the C that only 32-bit architectures get, and fix what that found+ [#260](https://github.com/kazu-yamamoto/crypton/pull/260)+* Let the last addition of each scalar multiplication be a complete one+ [#261](https://github.com/kazu-yamamoto/crypton/pull/261)+* Ask whether the secrets decide anything+ [#262](https://github.com/kazu-yamamoto/crypton/pull/262)+* Ask a big-endian machine the same questions+ [#263](https://github.com/kazu-yamamoto/crypton/pull/263)+* Ask what the secrets leave behind+ [#264](https://github.com/kazu-yamamoto/crypton/pull/264)+* Stop taking eighteen runner slots to test three things+ [#265](https://github.com/kazu-yamamoto/crypton/pull/265)+* Make the scrubs ones the compiler cannot drop, and finish round ten+ [#266](https://github.com/kazu-yamamoto/crypton/pull/266)+* Feed the parsers bytes nobody chose+ [#268](https://github.com/kazu-yamamoto/crypton/pull/268)++## 2.1.2++* perf(p256): 255 squarings for the field inversion, not 287+ [#223](https://github.com/kazu-yamamoto/crypton/pull/223)+* perf(p256): ECDH through s2n-bignum, 2.7x+ [#224](https://github.com/kazu-yamamoto/crypton/pull/224)+* perf(ecc): P-384 and P-521 through s2n-bignum, 7x and 9x+ [#225](https://github.com/kazu-yamamoto/crypton/pull/225)+* perf(p256): ECDSA signing 2.4x and verification 2.2x+ [#226](https://github.com/kazu-yamamoto/crypton/pull/226)+* perf(rsa): the Montgomery multiplication through s2n-bignum on x86-64+ [#227](https://github.com/kazu-yamamoto/crypton/pull/227)+* perf(ecdsa): invert modulo the order in division steps, not an exponentiation+ [#228](https://github.com/kazu-yamamoto/crypton/pull/228)+* perf(x25519): X25519 through s2n-bignum, and a table for key generation+ [#229](https://github.com/kazu-yamamoto/crypton/pull/229)+* perf(gcm): AES-GCM through VAES and VPCLMULQDQ+ [#230](https://github.com/kazu-yamamoto/crypton/pull/230)+* perf(gcm): compile the wide loop once per key length+ [#231](https://github.com/kazu-yamamoto/crypton/pull/231)++## 2.1.1++* feat(ecdsa): RFC 6979 deterministic nonces for Crypto.PubKey.ECDSA+ [#219](https://github.com/kazu-yamamoto/crypton/pull/219)+* feat(gcm): a decrypt that hands back the tag instead of comparing it+ [#220](https://github.com/kazu-yamamoto/crypton/pull/220)+* feat(chachapoly): ChaCha20-Poly1305 a message at a time+ [#221](https://github.com/kazu-yamamoto/crypton/pull/221)+* docs(rsa): say in the haddock what the optional blinder covers+ [#222](https://github.com/kazu-yamamoto/crypton/pull/222)++## 2.1.0++* fix(cpu): stop reading Intel's SDBG bit as AMD's XOP+ [#204](https://github.com/kazu-yamamoto/crypton/pull/204)+* fix(bench): build the benchmark against the checked ChaCha20-Poly1305 key+ [#205](https://github.com/kazu-yamamoto/crypton/pull/205)+* ci: key the cache on the package version+ [#206](https://github.com/kazu-yamamoto/crypton/pull/206)+* ci: build the benchmarks+ [#207](https://github.com/kazu-yamamoto/crypton/pull/207)+* perf(gcm): a fused AES-GCM for x86-64+ [#208](https://github.com/kazu-yamamoto/crypton/pull/208)+* perf(gcm): a fused AES-GCM for AArch64+ [#209](https://github.com/kazu-yamamoto/crypton/pull/209)+* perf(gcm): build the counter in vector registers+ [#210](https://github.com/kazu-yamamoto/crypton/pull/210)+* perf(gcm): a spare lane for E(K,Y0), and a cheaper short block+ [#211](https://github.com/kazu-yamamoto/crypton/pull/211)+* perf(gcm): unroll the tail pass, and take its blocks from registers+ [#212](https://github.com/kazu-yamamoto/crypton/pull/212)+* perf(gcm): read a short block where it lies+ [#213](https://github.com/kazu-yamamoto/crypton/pull/213)+* perf(gcm): the length block and the counter, in registers+ [#214](https://github.com/kazu-yamamoto/crypton/pull/214)+* perf(gcm): let the one-call interface specialise+ [#215](https://github.com/kazu-yamamoto/crypton/pull/215)+* perf(gcm): decryption takes the fused path too+ [#216](https://github.com/kazu-yamamoto/crypton/pull/216)+* perf(gcm): GHASH takes the ciphertext from the output buffer+ [#217](https://github.com/kazu-yamamoto/crypton/pull/217)+* perf(p256): a signed five-bit window for the variable-point multiply+ [#218](https://github.com/kazu-yamamoto/crypton/pull/218)++## 2.0.1++* feat(hash): Skein with the digest size as a type parameter+ [#197](https://github.com/kazu-yamamoto/crypton/pull/197)+* fix(chachapoly1305): take a checked key, so that initializing cannot fail+ [#198](https://github.com/kazu-yamamoto/crypton/pull/198)+* feat(aes): Crypto.Cipher.AES.GCM, for many short messages under one key+ [#199](https://github.com/kazu-yamamoto/crypton/pull/199)+* build: say which platforms the fallback AES sources are for+ [#200](https://github.com/kazu-yamamoto/crypton/pull/200)+* feat(aes): encryptWithMask, for the QUIC header protection mask+ [#201](https://github.com/kazu-yamamoto/crypton/pull/201)+* fix(cpu): stop reading Intel's SDBG bit as AMD's XOP+ [#203](https://github.com/kazu-yamamoto/crypton/pull/203)++## 2.0.0++**Breaking changes.** Input that used to be accepted is now refused: a value+at or above an RSA or Rabin modulus, a signature of the wrong length or out of+range, a digest too short for HOTP's dynamic truncation, a non-canonical+Ed25519 signature, a PKCS#7 block size outside 1..255, and block cipher input+that is not a whole number of blocks. A refused KDF, Argon2 or bcrypt+parameter is reported as a `CryptoError` rather than raised as an `ErrorCall`,+and `CryptoError_ParameterInvalid` is appended to `CryptoError`;+`tryGetShared` is added beside `getShared`. No exported function changed its+signature.++**Deprecated.** The eighteen curves over a binary field in+`Crypto.ECC.Simple.Types`. They are obsolete, they are the curves whose+cofactor is not 1, and they will go in a later major version. Prefer a prime+curve, or X25519.++* Add GHC 9.14 to CI+ [#74](https://github.com/kazu-yamamoto/crypton/pull/74)+* fix(ed25519): reject non-canonical signatures+ [#81](https://github.com/kazu-yamamoto/crypton/pull/81)+* fix(hkdf): enforce RFC 5869 output limit+ [#82](https://github.com/kazu-yamamoto/crypton/pull/82)+* fix(p256): accept valid edge-case points+ [#83](https://github.com/kazu-yamamoto/crypton/pull/83)+* fix(ecc): accept zero-x P-256 shared secrets+ [#84](https://github.com/kazu-yamamoto/crypton/pull/84)+* fix(otp): require a digest long enough for dynamic truncation+ [#85](https://github.com/kazu-yamamoto/crypton/pull/85)+* fix(pkcs15): reject malformed PKCS#1 v1.5 signatures+ [#86](https://github.com/kazu-yamamoto/crypton/pull/86)+* fix(ecdh): validate the peer point before the exchange+ [#87](https://github.com/kazu-yamamoto/crypton/pull/87)+* fix(dsa): do not crash on non-invertible values+ [#88](https://github.com/kazu-yamamoto/crypton/pull/88)+* fix(dh): validate the peer public number+ [#89](https://github.com/kazu-yamamoto/crypton/pull/89)+* fix(argon2): report invalid options as CryptoFailed+ [#90](https://github.com/kazu-yamamoto/crypton/pull/90)+* fix(rsa): drop the early exits from PKCS#1 v1.5 and OAEP unpadding+ [#91](https://github.com/kazu-yamamoto/crypton/pull/91)+* fix(otp): compare TOTP candidates without an early exit+ [#92](https://github.com/kazu-yamamoto/crypton/pull/92)+* feat(dh): add getShared' reporting rejections as CryptoFailable+ [#93](https://github.com/kazu-yamamoto/crypton/pull/93)+* feat(aead): add aeadSimpleDecrypt' taking the tag length+ [#94](https://github.com/kazu-yamamoto/crypton/pull/94)+* test: move the suite to hspec, with hspec-discover+ [#95](https://github.com/kazu-yamamoto/crypton/pull/95)+* docs(bcrypt): say that only the first 72 bytes of a password count+ [#96](https://github.com/kazu-yamamoto/crypton/pull/96)+* feat(elgamal): fix and expose Crypto.PubKey.ElGamal+ [#97](https://github.com/kazu-yamamoto/crypton/pull/97)+* fix(padding): reject a PKCS7 block size outside 1..255+ [#98](https://github.com/kazu-yamamoto/crypton/pull/98)+* build(bench): move the benchmarks from gauge to tasty-bench+ [#99](https://github.com/kazu-yamamoto/crypton/pull/99)+* feat(aes): use the ARMv8 cryptographic extensions on AArch64+ [#100](https://github.com/kazu-yamamoto/crypton/pull/100)+* ci: stop throwing the cache away, and keep the build products in it+ [#101](https://github.com/kazu-yamamoto/crypton/pull/101)+* feat(aes): use PMULL for GHASH on AArch64+ [#102](https://github.com/kazu-yamamoto/crypton/pull/102)+* ci: cut the macOS queueing and supersede stale branch runs+ [#103](https://github.com/kazu-yamamoto/crypton/pull/103)+* feat(sha256): use the ARMv8 SHA-2 instructions on AArch64+ [#104](https://github.com/kazu-yamamoto/crypton/pull/104)+* perf(gcm): fold four GHASH blocks into one reduction+ [#105](https://github.com/kazu-yamamoto/crypton/pull/105)+* ci: build and test on aarch64 Linux+ [#106](https://github.com/kazu-yamamoto/crypton/pull/106)+* fix(cabal): build the AES-NI paths on Windows too+ [#107](https://github.com/kazu-yamamoto/crypton/pull/107)+* perf(aes): specialise by key size and interleave eight blocks on AArch64+ [#108](https://github.com/kazu-yamamoto/crypton/pull/108)+* perf(gcm): drive GCM from AArch64 rather than the generic loop+ [#109](https://github.com/kazu-yamamoto/crypton/pull/109)+* feat(sha512): use the ARMv8.2 SHA-512 instructions on AArch64+ [#110](https://github.com/kazu-yamamoto/crypton/pull/110)+* perf(chacha): do four blocks at a time with NEON on AArch64+ [#111](https://github.com/kazu-yamamoto/crypton/pull/111)+* perf(chacha): do four blocks at a time with SSE2 on x86-64+ [#112](https://github.com/kazu-yamamoto/crypton/pull/112)+* perf(chacha): take eight blocks with AVX2 where the machine has it+ [#113](https://github.com/kazu-yamamoto/crypton/pull/113)+* perf(gcm): give x86 its own decryption loop, and eight blocks either way+ [#114](https://github.com/kazu-yamamoto/crypton/pull/114)+* fix(padding): bound PKCS7 padding by the block, and check ZERO's size+ [#115](https://github.com/kazu-yamamoto/crypton/pull/115)+* ecc: say which curves branch on a secret scalar, and work in Jacobian coordinates+ [#116](https://github.com/kazu-yamamoto/crypton/pull/116)+* perf(poly1305): take four blocks at a time with AVX2 on x86-64+ [#117](https://github.com/kazu-yamamoto/crypton/pull/117)+* perf(xts): drive XTS eight blocks at a time, and dispatch its decryption+ [#118](https://github.com/kazu-yamamoto/crypton/pull/118)+* Report refused KDF parameters as CryptoError, and fix a PBKDF2 SIGBUS+ [#119](https://github.com/kazu-yamamoto/crypton/pull/119)+* Search the HOTP resynchronization window without early exits+ [#120](https://github.com/kazu-yamamoto/crypton/pull/120)+* Refuse an RSA representative that is not below the modulus+ [#121](https://github.com/kazu-yamamoto/crypton/pull/121)+* Give AFIS one answer for a parameter it cannot use+ [#122](https://github.com/kazu-yamamoto/crypton/pull/122)+* Say what ElGamal's signWith requires of k+ [#123](https://github.com/kazu-yamamoto/crypton/pull/123)+* Draw Miller-Rabin witnesses per number, not once per process+ [#124](https://github.com/kazu-yamamoto/crypton/pull/124)+* Refuse Rabin values that are not below the modulus, and keep the padding that was signed+ [#125](https://github.com/kazu-yamamoto/crypton/pull/125)+* Decode Rabin's OAEP without early exits+ [#126](https://github.com/kazu-yamamoto/crypton/pull/126)+* Make CMAC linear, and chain it through CBC+ [#127](https://github.com/kazu-yamamoto/crypton/pull/127)+* Put DES in C+ [#128](https://github.com/kazu-yamamoto/crypton/pull/128)+* Make the generic block cipher modes linear, and bulk where the blocks allow+ [#129](https://github.com/kazu-yamamoto/crypton/pull/129)+* Walk Twofish's blocks once, and carry them in words+ [#130](https://github.com/kazu-yamamoto/crypton/pull/130)+* Put Camellia in C+ [#131](https://github.com/kazu-yamamoto/crypton/pull/131)+* Route P-256 through the C implementation it already had+ [#132](https://github.com/kazu-yamamoto/crypton/pull/132)+* Fold instead of dividing in the generic curve arithmetic+ [#133](https://github.com/kazu-yamamoto/crypton/pull/133)+* Reduce the binary field by folding, and work a byte and a nibble at a time+ [#134](https://github.com/kazu-yamamoto/crypton/pull/134)+* Stop running a Fermat test Miller-Rabin subsumes+ [#135](https://github.com/kazu-yamamoto/crypton/pull/135)+* Make expSafe hide the exponent again+ [#136](https://github.com/kazu-yamamoto/crypton/pull/136)+* Square, and multiply, faster in expSafe+ [#137](https://github.com/kazu-yamamoto/crypton/pull/137)+* Invert the signing nonce without a side channel+ [#138](https://github.com/kazu-yamamoto/crypton/pull/138)+* Keep the P-256 signature out of Integer arithmetic+ [#139](https://github.com/kazu-yamamoto/crypton/pull/139)+* Add at every bit in the prime-curve multiplication, which laziness was skipping+ [#140](https://github.com/kazu-yamamoto/crypton/pull/140)+* Multiply points in C on curves over a prime field+ [#141](https://github.com/kazu-yamamoto/crypton/pull/141)+* A ladder for the curves over a binary field+ [#142](https://github.com/kazu-yamamoto/crypton/pull/142)+* Work RSA's qinv out without the extended Euclidean algorithm+ [#143](https://github.com/kazu-yamamoto/crypton/pull/143)+* Keep the RSA blinding factor out of the extended algorithm+ [#144](https://github.com/kazu-yamamoto/crypton/pull/144)+* Unroll the inner loop at four and two as well+ [#145](https://github.com/kazu-yamamoto/crypton/pull/145)+* Keep a table for each curve's base point+ [#146](https://github.com/kazu-yamamoto/crypton/pull/146)+* Start R squared at the top of the modulus, and why folding did not pay+ [#147](https://github.com/kazu-yamamoto/crypton/pull/147)+* Do the binary field arithmetic in C+ [#148](https://github.com/kazu-yamamoto/crypton/pull/148)+* Use the x86 carry-less multiply where the processor has it+ [#149](https://github.com/kazu-yamamoto/crypton/pull/149)+* Close the two testing gaps: one multiplication for both APIs, one place for each buffer's size+ [#150](https://github.com/kazu-yamamoto/crypton/pull/150)+* Ask aarch64 for its carry-less multiply as well+ [#151](https://github.com/kazu-yamamoto/crypton/pull/151)+* Work the RSA private exponent out without the extended algorithm+ [#152](https://github.com/kazu-yamamoto/crypton/pull/152)+* Fewer Miller-Rabin rounds for a candidate nobody chose+ [#153](https://github.com/kazu-yamamoto/crypton/pull/153)+* Blowfish, and the key setup bcrypt wraps it in, in C+ [#154](https://github.com/kazu-yamamoto/crypton/pull/154)+* perf(sha256): use the Intel SHA extensions on x86-64+ [#155](https://github.com/kazu-yamamoto/crypton/pull/155)+* perf(aes): AES-192 through the processor's AES instructions+ [#156](https://github.com/kazu-yamamoto/crypton/pull/156)+* perf(aes): build the AArch64 key schedule with AESE, not the S-box table+ [#157](https://github.com/kazu-yamamoto/crypton/pull/157)+* test(aes): run the XTS vectors, and OCB and CCM at 192 and 256 bits+ [#158](https://github.com/kazu-yamamoto/crypton/pull/158)+* perf(ocb): drive OCB through the ECB paths a group at a time+ [#159](https://github.com/kazu-yamamoto/crypton/pull/159)+* perf(gcm): take the GHASH of the group before, alongside this group's rounds+ [#160](https://github.com/kazu-yamamoto/crypton/pull/160)+* perf(sha): compute the message schedule in vector registers on x86+ [#161](https://github.com/kazu-yamamoto/crypton/pull/161)+* perf(chacha): combine as the keystream comes out of the registers+ [#162](https://github.com/kazu-yamamoto/crypton/pull/162)+* perf(poly1305): shorten the carry chain and stop spilling the loop+ [#163](https://github.com/kazu-yamamoto/crypton/pull/163)+* docs(sidechannel): say what the prime-field modules keep from the clock+ [#164](https://github.com/kazu-yamamoto/crypton/pull/164)+* perf(sha1): use the Intel SHA extensions on x86-64+ [#165](https://github.com/kazu-yamamoto/crypton/pull/165)+* refactor(aes): drop the keystream generator nobody can call+ [#166](https://github.com/kazu-yamamoto/crypton/pull/166)+* build: compile the C at -O3+ [#167](https://github.com/kazu-yamamoto/crypton/pull/167)+* perf(modes): stop the generic cipher modes allocating per byte+ [#168](https://github.com/kazu-yamamoto/crypton/pull/168)+* perf(poly1305): four blocks at a time with NEON+ [#169](https://github.com/kazu-yamamoto/crypton/pull/169)+* perf(sha1): use the ARMv8 SHA-1 instructions+ [#170](https://github.com/kazu-yamamoto/crypton/pull/170)+* perf(sha3): use the ARMv8.2 SHA-3 instructions+ [#171](https://github.com/kazu-yamamoto/crypton/pull/171)+* perf(gcm): the CRYPTOGAMS stitched AES-GCM on x86-64+ [#172](https://github.com/kazu-yamamoto/crypton/pull/172)+* perf(chacha): the CRYPTOGAMS ChaCha20 on AArch64+ [#173](https://github.com/kazu-yamamoto/crypton/pull/173)+* perf(poly1305): the CRYPTOGAMS Poly1305 on AArch64+ [#174](https://github.com/kazu-yamamoto/crypton/pull/174)+* perf(sha256): the CRYPTOGAMS SHA-256 on AArch64+ [#175](https://github.com/kazu-yamamoto/crypton/pull/175)+* perf(poly1305): the CRYPTOGAMS Poly1305 on x86-64 too+ [#176](https://github.com/kazu-yamamoto/crypton/pull/176)+* perf(chacha): the CRYPTOGAMS ChaCha20 on x86-64 too+ [#177](https://github.com/kazu-yamamoto/crypton/pull/177)+* perf(sha2): the CRYPTOGAMS SHA-256 and SHA-512 on x86-64+ [#178](https://github.com/kazu-yamamoto/crypton/pull/178)+* perf(sha1): hand the SHA-1 block loop a run of blocks, not one at a time+ [#179](https://github.com/kazu-yamamoto/crypton/pull/179)+* perf(xts): double the tweak in the integer registers+ [#180](https://github.com/kazu-yamamoto/crypton/pull/180)+* perf(sha3): take the CRYPTOGAMS Keccak for AArch64+ [#181](https://github.com/kazu-yamamoto/crypton/pull/181)+* perf(sha1): take the CRYPTOGAMS SHA-1 for AArch64+ [#182](https://github.com/kazu-yamamoto/crypton/pull/182)+* docs: put the performance tables in the README+ [#183](https://github.com/kazu-yamamoto/crypton/pull/183)+* perf(sha3): take the CRYPTOGAMS Keccak for x86-64 as well+ [#184](https://github.com/kazu-yamamoto/crypton/pull/184)+* docs: rebuild the performance tables+ [#185](https://github.com/kazu-yamamoto/crypton/pull/185)+* perf(ecc): stop sharing the doublings in the double multiplication+ [#186](https://github.com/kazu-yamamoto/crypton/pull/186)+* perf(number): count bytes from the bit count, not from base 256+ [#187](https://github.com/kazu-yamamoto/crypton/pull/187)+* perf(p256): inline the field arithmetic on AArch64+ [#188](https://github.com/kazu-yamamoto/crypton/pull/188)+* fix(api): name the reporting variants try..., not with an apostrophe+ [#189](https://github.com/kazu-yamamoto/crypton/pull/189)+* fix(ecc): require a public point to be in the prime-order subgroup+ [#190](https://github.com/kazu-yamamoto/crypton/pull/190)+* fix(pubkey): stop printing private keys, and add Crypto.Debug+ [#191](https://github.com/kazu-yamamoto/crypton/pull/191)+* fix(poly1305): take a checked key, so that initializing cannot fail+ [#192](https://github.com/kazu-yamamoto/crypton/pull/192)+* fix(bcrypt): refuse a cost bcrypt does not have rather than substituting one+ [#194](https://github.com/kazu-yamamoto/crypton/pull/194)+* chore: build without a warning+ [#195](https://github.com/kazu-yamamoto/crypton/pull/195)+* docs: build the documentation without a warning+ [#196](https://github.com/kazu-yamamoto/crypton/pull/196)+ ## 1.1.5 * fix(aead): reject undersized tags@@ -377,3 +861,4 @@ ## 0.1 * Initial release+
@@ -7,6 +7,26 @@ -- Maintainer : Vincent Hanquez <vincent@snarc.org> -- Stability : stable -- Portability : good+--+-- AES, in the modes "Crypto.Cipher.Types" defines.+--+-- == Which implementation runs+--+-- Where the processor has instructions for AES -- AES-NI on x86-64, the+-- cryptographic extensions on AArch64 -- every key size and every mode here+-- goes through them, and a block costs the same whatever the key and the data+-- are.+--+-- Where it does not, the fallback is the table-driven code in+-- @cbits\/aes\/generic.c@, which indexes a 256-byte table with bytes derived+-- from the key and from the block. That is the cache-timing exposure the+-- instructions exist to remove, and on such a machine AES here is not+-- constant time. Every x86-64 part since about 2010 and every AArch64 one in+-- ordinary use has the instructions.+--+-- 'Crypto.System.CPU.processorOptions' says which of the two a given machine+-- got: @AESNI@ in that list means the processor's AES instructions, on either+-- architecture. module Crypto.Cipher.AES ( AES128, AES192,
@@ -0,0 +1,256 @@+-- |+-- Module : Crypto.Cipher.AES.GCM+-- License : BSD-style+-- Maintainer : Kazu Yamamoto <kazu@iij.ad.jp>+-- Stability : experimental+-- Portability : unknown+--+-- AES-GCM for callers that send many short messages under one key, which is+-- what a datagram transport does.+--+-- The interface in "Crypto.Cipher.Types" builds a state from the key /and/+-- the nonce and then walks it through appending the additional data,+-- encrypting and finalizing, copying the state at each step. For a stream+-- that is nothing next to the encryption. For a QUIC packet it is most of+-- the work: the key schedule and the table of multiples of @H@ depend on the+-- key alone, and rebuilding them for every nonce costs more than encrypting+-- 1440 bytes.+--+-- So here a t'Context' is built from the key once and holds both, and+-- 'encrypt' takes a nonce and a whole message and answers in one call.+--+-- > ctx <- throwCryptoError <$> pure (newContext key)+-- > let packet = encrypt ctx nonce header plaintext 16+--+-- This runs on AES-NI and carry-less multiply, or on the ARMv8 cryptographic+-- extension, and makes no branch and no memory access that depends on the key+-- or on the data. Where the processor has neither, AES falls back to a table+-- driven implementation that is /not/ constant time; see the side channels+-- section of the README, and 'Crypto.System.CPU.processorOptions' for which is+-- in use.+--+-- The result is the ciphertext with the tag after it, which is the shape a+-- packet wants. 'decrypt' takes that shape back, compares the tag itself and+-- answers 'Nothing' when it does not match.+--+-- This computes the same thing as the general interface; the tests hold it to+-- that on the same vectors.+module Crypto.Cipher.AES.GCM (+ Context,+ newContext,+ encrypt,+ decrypt,+ decryptWithTag,++ -- * Header protection+ HeaderKey,+ newHeaderKey,+ encryptWithMask,+) where++import Crypto.Cipher.AES.Primitive (+ AES,+ AESGCMKey,+ gcmFullDecrypt,+ gcmFullDecryptTag,+ gcmFullEncrypt,+ gcmFullEncryptMask,+ gcmKeyInit,+ initAES,+ )+import Crypto.Cipher.Types (AuthTag)+import Crypto.Cipher.Types.AEAD (minimumTagLength)+import Crypto.Error+import Crypto.Internal.ByteArray (ByteArray, ByteArrayAccess)+import qualified Crypto.Internal.ByteArray as B+import Data.Word (Word8)+import Foreign.Ptr (Ptr)++-- | Everything a key determines: the AES key schedule and the table of+-- multiples of @H@. Build it once and encrypt as many messages under it as+-- the key is good for.+data Context = Context !AES !AESGCMKey++-- | Take a key of 16, 24 or 32 bytes. Any other length is reported as+-- 'CryptoError_KeySizeInvalid'.+newContext :: ByteArrayAccess key => key -> CryptoFailable Context+newContext k = do+ aes <- initAES k+ return $ Context aes (gcmKeyInit aes)++-- | Encrypt one message: the nonce, the additional data that is+-- authenticated but not encrypted, the plaintext, and how many bytes of tag+-- to produce, which GCM allows between 4 and 16. Any other length throws+-- 'CryptoError_AuthenticationTagSizeInvalid', and so does 'decryptWithTag'.+--+-- The answer is the ciphertext followed by the tag.+--+-- A nonce must not be used twice with the same t'Context'. Twelve bytes is+-- the size GCM is defined for and the only one that does not cost a further+-- pass. A nonce of no bytes is refused: it would hand out the key GCM+-- authenticates with, so 'encrypt' and 'decryptWithTag' throw+-- 'CryptoError_IvSizeInvalid' for it, 'decrypt' gives 'Nothing' and+-- 'encryptWithMask' gives 'False'.+{-# INLINABLE encrypt #-}+encrypt+ :: ( ByteArrayAccess nonce+ , ByteArrayAccess aad+ , ByteArrayAccess ba+ , ByteArray output+ )+ => Context+ -> nonce+ -> aad+ -> ba+ -> Int+ -> output+encrypt (Context aes gk) nonce aad input taglen+ | tooLongForC aad input =+ throwCryptoError (CryptoFailed CryptoError_ParameterInvalid)+ | badNonce nonce =+ throwCryptoError (CryptoFailed CryptoError_IvSizeInvalid)+ | badTagLength taglen =+ throwCryptoError (CryptoFailed CryptoError_AuthenticationTagSizeInvalid)+ | otherwise = gcmFullEncrypt aes gk nonce aad input taglen++-- | Decrypt one message, in the shape 'encrypt' produced: the ciphertext with+-- its tag after it. The tag is compared here, every byte of it whatever the+-- answer, and a message whose tag does not match gives 'Nothing' rather than+-- the plaintext.+--+-- 'Nothing' also comes back when the nonce has no bytes, the input is+-- shorter than the tag, or the tag length is outside 4 to 16.+{-# INLINABLE decrypt #-}+decrypt+ :: (ByteArrayAccess nonce, ByteArrayAccess aad, ByteArray ba)+ => Context+ -> nonce+ -> aad+ -> ba+ -> Int+ -> Maybe ba+decrypt (Context aes gk) nonce aad input taglen+ | tooLongForC aad input = Nothing+ | badNonce nonce || badTagLength taglen || B.length input < taglen = Nothing+ | otherwise = gcmFullDecrypt aes gk nonce aad body tag+ where+ (body, tag) = B.splitAt (B.length input - taglen) input++-- | Decrypt one message, the tag kept apart, and hand back the tag this end+-- computed.+--+-- For a caller whose protocol hands it the tag separately from the+-- ciphertext, so that 'decrypt' -- which wants the two together and compares+-- them itself -- does not fit. Compare the two tags with '=='; the 'Eq'+-- instance of t'AuthTag' is a constant-time comparison, and taking them apart+-- to compare the bytes is how this goes wrong.+--+-- Nothing here says whether the message is authentic. Until the comparison+-- is made and has come out equal, what this returns is not plaintext, it is+-- what the ciphertext turns into, and a caller must not act on it.+{-# INLINABLE decryptWithTag #-}+decryptWithTag+ :: (ByteArrayAccess nonce, ByteArrayAccess aad, ByteArray ba)+ => Context+ -> nonce+ -> aad+ -> ba+ -> Int+ -> (ba, AuthTag)+decryptWithTag (Context aes gk) nonce aad input taglen+ | tooLongForC aad input =+ throwCryptoError (CryptoFailed CryptoError_ParameterInvalid)+ | badNonce nonce =+ throwCryptoError (CryptoFailed CryptoError_IvSizeInvalid)+ | badTagLength taglen =+ throwCryptoError (CryptoFailed CryptoError_AuthenticationTagSizeInvalid)+ | otherwise = gcmFullDecryptTag aes gk nonce aad input taglen++----------------------------------------------------------------++-- | The key schedule for header protection, which QUIC keeps separately from+-- the one it encrypts with. Built once, like a t'Context'.+newtype HeaderKey = HeaderKey AES++-- | Take a header protection key of 16, 24 or 32 bytes.+newHeaderKey :: ByteArrayAccess key => key -> CryptoFailable HeaderKey+newHeaderKey k = HeaderKey <$> initAES k++-- | Encrypt one message and, from a sample of the ciphertext it just+-- produced, make the header protection mask -- in one call, into two buffers+-- the caller already has.+--+-- QUIC takes its sample from the ciphertext, so the mask cannot be had before+-- the encryption. It can be had before coming back, and with the buffers+-- already there nothing is allocated for either. On an Apple M4 the mask+-- then costs about 0.02 us, where asking for it separately costs 0.11.+--+-- The sealed message wants @length input + taglen@ bytes and the mask+-- sixteen. @sampleOffset@ says where the sixteen bytes of sample begin in+-- the sealed message, counting the tag as part of it.+--+-- 'False' comes back, and nothing is written, when the nonce has no bytes,+-- the sample would not fit, or the tag length is outside 4 to 16.+{-# INLINABLE encryptWithMask #-}+encryptWithMask+ :: (ByteArrayAccess nonce, ByteArrayAccess aad, ByteArrayAccess ba)+ => Context+ -> HeaderKey+ -> nonce+ -> aad+ -> ba+ -> Int+ -- ^ tag length+ -> Int+ -- ^ sample offset+ -> Ptr Word8+ -- ^ where the sealed message goes+ -> Ptr Word8+ -- ^ where the sixteen bytes of mask go+ -> IO Bool+encryptWithMask (Context aes gk) (HeaderKey hp) nonce aad input taglen off outp maskp+ | tooLongForC aad input = return False+ | badNonce nonce = return False+ | off < 0 || badTagLength taglen || off + 16 > B.length input + taglen =+ return False+ | otherwise = do+ gcmFullEncryptMask aes gk hp nonce aad input taglen off outp maskp+ return True++-- | The C behind all four takes its lengths as @uint32_t@, so a message or+-- its additional data from 2^32 bytes up cannot be handed to it: the length+-- would be truncated and most of the buffer left untouched, with nothing to+-- say so. There is no splitting the work here -- the C does the whole+-- message in one call, tag and all -- so such a message is refused.+tooLongForC+ :: (ByteArrayAccess aad, ByteArrayAccess ba) => aad -> ba -> Bool+tooLongForC aad input =+ B.overCLength (B.length aad) || B.overCLength (B.length input)++-- | SP 800-38D 5.2.1.1 asks for at least one byte of IV, and this is why.+--+-- GCM builds its pre-counter block from a nonce that is not twelve bytes as+-- @J0 = GHASH_H(IV || 0^s || [0]_64 || [len(IV)]_64)@. For an empty IV that+-- input is one block of zeros, so @J0@ is zero, and the tag of a message+-- becomes @GHASH_H(A, C) XOR E(K, 0^128)@ -- where @E(K, 0^128)@ is the+-- definition of @H@. The tag of an empty message under an empty nonce is+-- therefore @H@ itself, and any other full tag gives @H@ as the root of a+-- known polynomial.+--+-- @H@ belongs to the key, not to the nonce. An attacker holding it, plus one+-- genuine message under any nonce, has @E(K, J0)@ for that nonce and can make+-- a tag that verifies for data of their own -- under a correct twelve-byte+-- nonce, and for every nonce they have seen. One encryption with an empty+-- nonce and a full tag ends the authenticity of everything under the key.+--+-- 'Crypto.Cipher.AES.Primitive.gcmAeadInit' refuses the empty IV for this+-- reason, and these four have to as well. Only the empty one is refused:+-- SP 800-38D allows every length from one byte up.+badNonce :: ByteArrayAccess nonce => nonce -> Bool+badNonce nonce = B.length nonce == 0++-- | GCM makes a sixteen-byte tag and a shorter one is a prefix of it. Below+-- 'minimumTagLength' it authenticates next to nothing, and past sixteen the+-- C code would read beyond the tag it computed.+badTagLength :: Int -> Bool+badTagLength t = t < minimumTagLength || t > 16
@@ -21,8 +21,6 @@ initAES, -- * Miscellanea- genCTR,- genCounter, -- * Encryption encryptECB,@@ -42,6 +40,12 @@ -- * Incremental GCM gcmMode, gcmInit,+ AESGCMKey,+ gcmKeyInit,+ gcmFullEncrypt,+ gcmFullEncryptMask,+ gcmFullDecrypt,+ gcmFullDecryptTag, gcmAeadInit, -- * Incremental OCB@@ -160,6 +164,14 @@ sizeGCM :: Int sizeGCM = 320 +-- | The size of what a key determines, which is the 320 bytes above and the+-- powers of H the fused path reads: sixteen of them, and sixteen more for+-- the term the Karatsuba multiplication would otherwise work out every time.+-- The same on every platform, so that this is one number rather than one per+-- architecture; the powers are filled only where that path is compiled in.+sizeGCMKey :: Int+sizeGCMKey = 832+ sizeOCB :: Int sizeOCB = 160 @@ -172,12 +184,6 @@ ivToPtr :: ByteArrayAccess iv => iv -> (Ptr Word8 -> IO a) -> IO a ivToPtr iv f = withByteArray iv (f . castPtr) -ivCopyPtr :: IV AES -> (Ptr Word8 -> IO a) -> IO (a, IV AES)-ivCopyPtr (IV iv) f = (\(x, y) -> (x, IV y)) `fmap` copyAndModify iv f- where- copyAndModify :: ByteArray ba => ba -> (Ptr Word8 -> IO a) -> IO (a, ba)- copyAndModify ba f' = B.copyRet ba f'- withKeyAndIV :: ByteArrayAccess iv => AES -> iv -> (Ptr AES -> Ptr Word8 -> IO a) -> IO a withKeyAndIV ctx iv f = keyToPtr ctx $ \kptr -> ivToPtr iv $ \ivp -> f kptr ivp@@ -249,61 +255,6 @@ -- ^ ciphertext encryptCBC = doCBC c_aes_encrypt_cbc --- | generate a counter mode pad. this is generally xor-ed to an input--- to make the standard counter mode block operations.------ if the length requested is not a multiple of the block cipher size,--- more data will be returned, so that the returned bytearray is--- a multiple of the block cipher size.-{-# NOINLINE genCTR #-}-genCTR- :: ByteArray ba- => AES- -- ^ Cipher Key.- -> IV AES- -- ^ usually a 128 bit integer.- -> Int- -- ^ length of bytes required.- -> ba-genCTR ctx (IV iv) len- | len <= 0 = B.empty- | otherwise = B.allocAndFreeze (nbBlocks * 16) generate- where- generate o = withKeyAndIV ctx iv $ \k i -> c_aes_gen_ctr (castPtr o) k i (fromIntegral nbBlocks)- (nbBlocks', r) = len `quotRem` 16- nbBlocks = if r == 0 then nbBlocks' else nbBlocks' + 1---- | generate a counter mode pad. this is generally xor-ed to an input--- to make the standard counter mode block operations.------ if the length requested is not a multiple of the block cipher size,--- more data will be returned, so that the returned bytearray is--- a multiple of the block cipher size.------ Similiar to 'genCTR' but also return the next IV for continuation-{-# NOINLINE genCounter #-}-genCounter- :: ByteArray ba- => AES- -> IV AES- -> Int- -> (ba, IV AES)-genCounter ctx iv len- | len <= 0 = (B.empty, iv)- | otherwise = unsafeDoIO $- keyToPtr ctx $ \k ->- ivCopyPtr iv $ \i ->- B.alloc outputLength $ \o -> do- c_aes_gen_ctr_cont (castPtr o) k i (fromIntegral nbBlocks)- where- (nbBlocks', r) = len `quotRem` 16- nbBlocks = if r == 0 then nbBlocks' else nbBlocks' + 1- outputLength = nbBlocks * 16--{- TODO: when genCTR has same AESIV requirements for IV, add the following rules:- - RULES "snd . genCounter" forall ctx iv len . snd (genCounter ctx iv len) = genCTR ctx iv len- -}- -- | encrypt using Counter mode (CTR) -- -- in CTR mode encryption and decryption is the same operation.@@ -320,6 +271,7 @@ -- ^ ciphertext output encryptCTR ctx iv input | len <= 0 = B.empty+ | B.overCLength len = error tooLongMessage | B.length iv /= 16 = error $ "AES error: IV length must be block size (16). Its length is: "@@ -414,6 +366,20 @@ c_aes_encrypt_c32 (castPtr o) k v i (fromIntegral len) len = B.length input +-- | What the AES modes say when a message cannot be given to the C, whose+-- lengths are @uint32_t@. Above that the length is truncated on the way+-- down and most of the buffer is left as it was found, with nothing to say+-- so, which is worse than refusing.+--+-- Unlike the stream ciphers, these cannot be done in pieces: the C is handed+-- the IV and does not hand it back, so a second call would start from the+-- wrong place. ECB, CBC and XTS count blocks rather than bytes, so their+-- limit is sixteen times further out than CTR's.+tooLongMessage :: String+tooLongMessage =+ "AES error: message too long for this implementation, whose C takes its "+ ++ "lengths as uint32_t"+ {-# INLINE doECB #-} doECB :: ByteArray ba@@ -422,6 +388,7 @@ -> ba -> ba doECB f ctx input+ | B.overCLength nbBlocks = error tooLongMessage | len == 0 = B.empty | r /= 0 = error $@@ -445,6 +412,7 @@ -> ba -> ba doCBC f ctx (IV iv) input+ | B.overCLength nbBlocks = error tooLongMessage | len == 0 = B.empty | r /= 0 = error $@@ -468,6 +436,7 @@ -> ba -> ba doXTS f (key1, key2) iv spoint input+ | B.overCLength nbBlocks = error tooLongMessage | len == 0 = B.empty | r /= 0 = error $@@ -492,6 +461,190 @@ c_aes_gcm_init (castPtr gcmStPtr) k v (fromIntegral $ B.length iv) return $ AESGCM sm +-- | How long a message may be and still be handed to an unsafe foreign call.+-- Four kibibytes is about half a microsecond of work, and it takes in a+-- datagram of any size a network will carry.+shortMessage :: Int+shortMessage = 4096++-- | The part of a GCM state the key alone determines: H, which is the key+-- applied to a block of zeroes, and the table of its multiples. That is 256+-- of the 320 bytes of a GCM state, and it is the same for every message sent+-- under one key, so a caller that keeps a key can build this once rather than+-- once for every message.+newtype AESGCMKey = AESGCMKey ScrubbedBytes++-- | Build the key part of a GCM state.+{-# NOINLINE gcmKeyInit #-}+gcmKeyInit :: AES -> AESGCMKey+gcmKeyInit ctx = AESGCMKey $ B.allocAndFreeze sizeGCMKey $ \p ->+ keyToPtr ctx $ \k -> c_aes_gcm_key_init (castPtr p) k++-- | Authenticate and encrypt one message in a single call: the nonce, the+-- additional data, the plaintext and the tag, with no state crossing back+-- into Haskell in between. The result is the ciphertext followed by the tag.+{-# INLINABLE gcmFullEncrypt #-}+gcmFullEncrypt+ :: (ByteArrayAccess iv, ByteArrayAccess aad, ByteArrayAccess ba, ByteArray output)+ => AES -> AESGCMKey -> iv -> aad -> ba -> Int -> output+gcmFullEncrypt ctx (AESGCMKey gk) iv aad input taglen =+ B.allocAndFreeze (B.length input + taglen) $ \out ->+ B.withByteArray gk $ \gkp ->+ keyToPtr ctx $ \k ->+ B.withByteArray iv $ \ivp ->+ B.withByteArray aad $ \aadp ->+ B.withByteArray input $ \inp ->+ call+ out+ (castPtr gkp)+ k+ ivp+ (fromIntegral $ B.length iv)+ aadp+ (fromIntegral $ B.length aad)+ inp+ (fromIntegral $ B.length input)+ (fromIntegral taglen)+ where+ -- An unsafe call keeps a capability for as long as it runs, so it is only+ -- right for work that is over quickly. A message this side of+ -- 'shortMessage' is, and it is the short ones the saving matters for: a+ -- safe call costs about 0.075 us whatever the length, which is a fifth of+ -- a 1440-byte packet and a percent of a 16 KiB record.+ call+ | B.length input <= shortMessage = c_aes_gcm_full_encrypt_unsafe+ | otherwise = c_aes_gcm_full_encrypt++-- | Encrypt, and from a sample of the ciphertext just produced make the+-- header protection mask, into buffers the caller owns. QUIC takes its+-- sample from the ciphertext, so the mask cannot be had before the+-- encryption; it can be had before coming back, and with the buffers already+-- there nothing is allocated for either.+--+-- @sampleoff@ is where the sixteen bytes of sample begin in the output.+{-# INLINABLE gcmFullEncryptMask #-}+gcmFullEncryptMask+ :: (ByteArrayAccess iv, ByteArrayAccess aad, ByteArrayAccess ba)+ => AES+ -> AESGCMKey+ -> AES+ -> iv+ -> aad+ -> ba+ -> Int+ -> Int+ -> Ptr Word8+ -> Ptr Word8+ -> IO ()+gcmFullEncryptMask ctx (AESGCMKey gk) hpctx iv aad input taglen sampleoff outp maskp =+ B.withByteArray gk $ \gkp ->+ keyToPtr ctx $ \k ->+ keyToPtr hpctx $ \hk ->+ B.withByteArray iv $ \ivp ->+ B.withByteArray aad $ \aadp ->+ B.withByteArray input $ \inp ->+ call+ outp+ (castPtr gkp)+ k+ ivp+ (fromIntegral $ B.length iv)+ aadp+ (fromIntegral $ B.length aad)+ inp+ (fromIntegral $ B.length input)+ (fromIntegral taglen)+ hk+ (fromIntegral sampleoff)+ maskp+ where+ call+ | B.length input <= shortMessage = c_aes_gcm_full_encrypt_mask_unsafe+ | otherwise = c_aes_gcm_full_encrypt_mask++-- | The same the other way, with the tag compared here rather than by the+-- caller: 'Nothing' when it does not match, and every byte of it is looked at+-- either way. The ciphertext comes in without its tag, which is given+-- separately.+{-# INLINABLE gcmFullDecrypt #-}+gcmFullDecrypt+ :: ( ByteArrayAccess iv+ , ByteArrayAccess aad+ , ByteArrayAccess ba+ , ByteArrayAccess tag+ , ByteArray output+ )+ => AES -> AESGCMKey -> iv -> aad -> ba -> tag -> Maybe output+gcmFullDecrypt ctx (AESGCMKey gk) iv aad input tag = unsafeDoIO $ do+ (r, out) <- B.allocRet (B.length input) $ \outp ->+ B.withByteArray gk $ \gkp ->+ keyToPtr ctx $ \k ->+ B.withByteArray iv $ \ivp ->+ B.withByteArray aad $ \aadp ->+ B.withByteArray input $ \inp ->+ B.withByteArray tag $ \tagp ->+ call+ outp+ (castPtr gkp)+ k+ ivp+ (fromIntegral $ B.length iv)+ aadp+ (fromIntegral $ B.length aad)+ inp+ (fromIntegral $ B.length input)+ tagp+ (fromIntegral $ B.length tag)+ return $ if r /= 0 then Just out else Nothing+ where+ call+ | B.length input <= shortMessage = c_aes_gcm_full_decrypt_unsafe+ | otherwise = c_aes_gcm_full_decrypt++-- | Decrypt one message and hand back the tag that was computed over it,+-- rather than comparing it here.+--+-- For a caller that holds the expected tag in a form of its own and will+-- compare it itself. Compare the two t'AuthTag's with '==', whose instance+-- for that type is a constant-time comparison; taking them apart and+-- comparing the bytes is how this goes wrong.+--+-- Where the tag simply arrives after the ciphertext, 'gcmFullDecrypt' is the+-- one to use: it compares in C and never puts a tag in the caller's hands.+{-# INLINABLE gcmFullDecryptTag #-}+gcmFullDecryptTag+ :: ( ByteArrayAccess iv+ , ByteArrayAccess aad+ , ByteArrayAccess ba+ , ByteArray output+ )+ => AES -> AESGCMKey -> iv -> aad -> ba -> Int -> (output, AuthTag)+gcmFullDecryptTag ctx (AESGCMKey gk) iv aad input taglen = unsafeDoIO $ do+ (tagbs, out) <- B.allocRet (B.length input) $ \outp ->+ B.alloc taglen $ \tagp ->+ B.withByteArray gk $ \gkp ->+ keyToPtr ctx $ \k ->+ B.withByteArray iv $ \ivp ->+ B.withByteArray aad $ \aadp ->+ B.withByteArray input $ \inp ->+ call+ outp+ tagp+ (castPtr gkp)+ k+ ivp+ (fromIntegral $ B.length iv)+ aadp+ (fromIntegral $ B.length aad)+ inp+ (fromIntegral $ B.length input)+ (fromIntegral taglen)+ return (out, AuthTag $ B.convert (tagbs :: B.Bytes))+ where+ call+ | B.length input <= shortMessage = c_aes_gcm_full_decrypt_tag_unsafe+ | otherwise = c_aes_gcm_full_decrypt_tag+ -- | append data which is only going to be authenticated to the GCM context. -- -- needs to happen after initialization and before appending encryption/decryption data.@@ -502,7 +655,8 @@ doAppend = withNewGCMSt gcmSt $ \gcmStPtr -> withByteArray input $ \i ->- c_aes_gcm_aad gcmStPtr i (fromIntegral $ B.length input)+ B.inCLengths (B.length input) $ \off n ->+ c_aes_gcm_aad gcmStPtr (i `plusPtr` off) (fromIntegral n) -- | append data to encrypt and append to the GCM context --@@ -516,7 +670,13 @@ doEnc gcmStPtr aesPtr = B.alloc len $ \o -> withByteArray input $ \i ->- c_aes_gcm_encrypt (castPtr o) gcmStPtr aesPtr i (fromIntegral len)+ B.inCLengths len $ \off n ->+ c_aes_gcm_encrypt+ (castPtr o `plusPtr` off)+ gcmStPtr+ aesPtr+ (i `plusPtr` off)+ (fromIntegral n) -- | append data to decrypt and append to the GCM context --@@ -530,7 +690,13 @@ doDec gcmStPtr aesPtr = B.alloc len $ \o -> withByteArray input $ \i ->- c_aes_gcm_decrypt (castPtr o) gcmStPtr aesPtr i (fromIntegral len)+ B.inCLengths len $ \off n ->+ c_aes_gcm_decrypt+ (castPtr o `plusPtr` off)+ gcmStPtr+ aesPtr+ (i `plusPtr` off)+ (fromIntegral n) -- | Generate the Tag from GCM context {-# NOINLINE gcmFinish #-}@@ -563,9 +729,11 @@ -- The tag length is expressed in bytes and must be in [0..16]. -- The IV length must be in [1..15] bytes per RFC 7253. {-# NOINLINE ocbInitWithTagLength #-}-ocbInitWithTagLength :: ByteArrayAccess iv => AES -> iv -> Int -> CryptoFailable AESOCB+ocbInitWithTagLength+ :: ByteArrayAccess iv => AES -> iv -> Int -> CryptoFailable AESOCB ocbInitWithTagLength ctx iv taglen- | taglen < 0 || taglen > 16 = CryptoFailed CryptoError_AuthenticationTagSizeInvalid+ | taglen < 0 || taglen > 16 =+ CryptoFailed CryptoError_AuthenticationTagSizeInvalid | ivlen < 1 || ivlen > 15 = CryptoFailed CryptoError_IvSizeInvalid | otherwise = CryptoPassed $ unsafeDoIO $ do sm <- B.alloc sizeOCB $ \ocbStPtr ->@@ -585,7 +753,9 @@ -- need to happen after initialization and before appending encryption/decryption data. {-# NOINLINE ocbAppendAAD #-} ocbAppendAAD :: ByteArrayAccess aad => AES -> AESOCB -> aad -> AESOCB-ocbAppendAAD ctx ocb input = unsafeDoIO (snd `fmap` withOCBKeyAndCopySt ctx ocb doAppend)+ocbAppendAAD ctx ocb input+ | B.overCLength (B.length input) = error tooLongMessage+ | otherwise = unsafeDoIO (snd `fmap` withOCBKeyAndCopySt ctx ocb doAppend) where doAppend ocbStPtr aesPtr = withByteArray input $ \i ->@@ -597,7 +767,9 @@ -- need to happen after AAD appending, or after initialization if no AAD data. {-# NOINLINE ocbAppendEncrypt #-} ocbAppendEncrypt :: ByteArray ba => AES -> AESOCB -> ba -> (ba, AESOCB)-ocbAppendEncrypt ctx ocb input = unsafeDoIO $ withOCBKeyAndCopySt ctx ocb doEnc+ocbAppendEncrypt ctx ocb input+ | B.overCLength (B.length input) = error tooLongMessage+ | otherwise = unsafeDoIO $ withOCBKeyAndCopySt ctx ocb doEnc where len = B.length input doEnc ocbStPtr aesPtr =@@ -611,7 +783,9 @@ -- need to happen after AAD appending, or after initialization if no AAD data. {-# NOINLINE ocbAppendDecrypt #-} ocbAppendDecrypt :: ByteArray ba => AES -> AESOCB -> ba -> (ba, AESOCB)-ocbAppendDecrypt ctx ocb input = unsafeDoIO $ withOCBKeyAndCopySt ctx ocb doDec+ocbAppendDecrypt ctx ocb input+ | B.overCLength (B.length input) = error tooLongMessage+ | otherwise = unsafeDoIO $ withOCBKeyAndCopySt ctx ocb doDec where len = B.length input doDec ocbStPtr aesPtr =@@ -671,7 +845,9 @@ -- needs to happen after initialization and before appending encryption/decryption data. {-# NOINLINE ccmAppendAAD #-} ccmAppendAAD :: ByteArrayAccess aad => AES -> AESCCM -> aad -> AESCCM-ccmAppendAAD ctx ccm input = unsafeDoIO $ snd <$> withCCMKeyAndCopySt ctx ccm doAppend+ccmAppendAAD ctx ccm input+ | B.overCLength (B.length input) = error tooLongMessage+ | otherwise = unsafeDoIO $ snd <$> withCCMKeyAndCopySt ctx ccm doAppend where doAppend ccmStPtr aesPtr = withByteArray input $ \i -> c_aes_ccm_aad ccmStPtr aesPtr i (fromIntegral $ B.length input)@@ -682,7 +858,9 @@ -- needs to happen after AAD appending, or after initialization if no AAD data. {-# NOINLINE ccmEncrypt #-} ccmEncrypt :: ByteArray ba => AES -> AESCCM -> ba -> (ba, AESCCM)-ccmEncrypt ctx ccm input = unsafeDoIO $ withCCMKeyAndCopySt ctx ccm cbcmacAndIv+ccmEncrypt ctx ccm input+ | B.overCLength (B.length input) = error tooLongMessage+ | otherwise = unsafeDoIO $ withCCMKeyAndCopySt ctx ccm cbcmacAndIv where len = B.length input cbcmacAndIv ccmStPtr aesPtr =@@ -696,7 +874,9 @@ -- needs to happen after AAD appending, or after initialization if no AAD data. {-# NOINLINE ccmDecrypt #-} ccmDecrypt :: ByteArray ba => AES -> AESCCM -> ba -> (ba, AESCCM)-ccmDecrypt ctx ccm input = unsafeDoIO $ withCCMKeyAndCopySt ctx ccm cbcmacAndIv+ccmDecrypt ctx ccm input+ | B.overCLength (B.length input) = error tooLongMessage+ | otherwise = unsafeDoIO $ withCCMKeyAndCopySt ctx ccm cbcmacAndIv where len = B.length input cbcmacAndIv ccmStPtr aesPtr =@@ -738,12 +918,6 @@ c_aes_decrypt_xts :: CString -> Ptr AES -> Ptr AES -> Ptr Word8 -> CUInt -> CString -> CUInt -> IO () -foreign import ccall "crypton_aes.h crypton_aes_gen_ctr"- c_aes_gen_ctr :: CString -> Ptr AES -> Ptr Word8 -> CUInt -> IO ()--foreign import ccall unsafe "crypton_aes.h crypton_aes_gen_ctr_cont"- c_aes_gen_ctr_cont :: CString -> Ptr AES -> Ptr Word8 -> CUInt -> IO ()- foreign import ccall "crypton_aes.h crypton_aes_encrypt_ctr" c_aes_encrypt_ctr :: CString -> Ptr AES -> Ptr Word8 -> CString -> CUInt -> IO ()@@ -751,6 +925,131 @@ foreign import ccall "crypton_aes.h crypton_aes_encrypt_c32" c_aes_encrypt_c32 :: CString -> Ptr AES -> Ptr Word8 -> CString -> CUInt -> IO ()++foreign import ccall unsafe "crypton_aes.h crypton_aes_gcm_key_init"+ c_aes_gcm_key_init :: Ptr AESGCM -> Ptr AES -> IO ()++foreign import ccall "crypton_aes.h crypton_aes_gcm_full_encrypt"+ c_aes_gcm_full_encrypt+ :: Ptr Word8+ -> Ptr AESGCM+ -> Ptr AES+ -> Ptr Word8+ -> CUInt+ -> Ptr Word8+ -> CUInt+ -> Ptr Word8+ -> CUInt+ -> CUInt+ -> IO ()++foreign import ccall "crypton_aes.h crypton_aes_gcm_full_decrypt"+ c_aes_gcm_full_decrypt+ :: Ptr Word8+ -> Ptr AESGCM+ -> Ptr AES+ -> Ptr Word8+ -> CUInt+ -> Ptr Word8+ -> CUInt+ -> Ptr Word8+ -> CUInt+ -> Ptr Word8+ -> CUInt+ -> IO CInt++foreign import ccall "crypton_aes.h crypton_aes_gcm_full_decrypt_tag"+ c_aes_gcm_full_decrypt_tag+ :: Ptr Word8+ -> Ptr Word8+ -> Ptr AESGCM+ -> Ptr AES+ -> Ptr Word8+ -> CUInt+ -> Ptr Word8+ -> CUInt+ -> Ptr Word8+ -> CUInt+ -> CUInt+ -> IO ()++foreign import ccall unsafe "crypton_aes.h crypton_aes_gcm_full_decrypt_tag"+ c_aes_gcm_full_decrypt_tag_unsafe+ :: Ptr Word8+ -> Ptr Word8+ -> Ptr AESGCM+ -> Ptr AES+ -> Ptr Word8+ -> CUInt+ -> Ptr Word8+ -> CUInt+ -> Ptr Word8+ -> CUInt+ -> CUInt+ -> IO ()++foreign import ccall unsafe "crypton_aes.h crypton_aes_gcm_full_encrypt"+ c_aes_gcm_full_encrypt_unsafe+ :: Ptr Word8+ -> Ptr AESGCM+ -> Ptr AES+ -> Ptr Word8+ -> CUInt+ -> Ptr Word8+ -> CUInt+ -> Ptr Word8+ -> CUInt+ -> CUInt+ -> IO ()++foreign import ccall unsafe "crypton_aes.h crypton_aes_gcm_full_decrypt"+ c_aes_gcm_full_decrypt_unsafe+ :: Ptr Word8+ -> Ptr AESGCM+ -> Ptr AES+ -> Ptr Word8+ -> CUInt+ -> Ptr Word8+ -> CUInt+ -> Ptr Word8+ -> CUInt+ -> Ptr Word8+ -> CUInt+ -> IO CInt++foreign import ccall "crypton_aes.h crypton_aes_gcm_full_encrypt_mask"+ c_aes_gcm_full_encrypt_mask+ :: Ptr Word8+ -> Ptr AESGCM+ -> Ptr AES+ -> Ptr Word8+ -> CUInt+ -> Ptr Word8+ -> CUInt+ -> Ptr Word8+ -> CUInt+ -> CUInt+ -> Ptr AES+ -> CUInt+ -> Ptr Word8+ -> IO ()++foreign import ccall unsafe "crypton_aes.h crypton_aes_gcm_full_encrypt_mask"+ c_aes_gcm_full_encrypt_mask_unsafe+ :: Ptr Word8+ -> Ptr AESGCM+ -> Ptr AES+ -> Ptr Word8+ -> CUInt+ -> Ptr Word8+ -> CUInt+ -> Ptr Word8+ -> CUInt+ -> CUInt+ -> Ptr AES+ -> CUInt+ -> Ptr Word8+ -> IO () foreign import ccall "crypton_aes.h crypton_aes_gcm_init" c_aes_gcm_init :: Ptr AESGCM -> Ptr AES -> Ptr Word8 -> CUInt -> IO ()
@@ -1,303 +0,0 @@-{-# LANGUAGE MagicHash #-}---- |--- Module : Crypto.Cipher.Blowfish.Box--- License : BSD-style--- Stability : experimental--- Portability : Good-module Crypto.Cipher.Blowfish.Box (- KeySchedule (..),- createKeySchedule,- copyKeySchedule,-) where--import Crypto.Internal.WordArray (- MutableArray32,- mutableArray32FromAddrBE,- mutableArrayRead32,- mutableArrayWrite32,- )--newtype KeySchedule = KeySchedule MutableArray32---- | Copy the state of one key schedule into the other.--- The first parameter is the destination and the second the source.-copyKeySchedule :: KeySchedule -> KeySchedule -> IO ()-copyKeySchedule (KeySchedule dst) (KeySchedule src) = loop 0- where- loop 1042 = return ()- loop i = do- w32 <- mutableArrayRead32 src i- mutableArrayWrite32 dst i w32- loop (i + 1)---- | Create a key schedule mutable array of the pbox followed by--- all the sboxes.-createKeySchedule :: IO KeySchedule-createKeySchedule =- KeySchedule- `fmap` mutableArray32FromAddrBE- 1042- "\- \\x24\x3f\x6a\x88\x85\xa3\x08\xd3\x13\x19\x8a\x2e\x03\x70\x73\x44\- \\xa4\x09\x38\x22\x29\x9f\x31\xd0\x08\x2e\xfa\x98\xec\x4e\x6c\x89\- \\x45\x28\x21\xe6\x38\xd0\x13\x77\xbe\x54\x66\xcf\x34\xe9\x0c\x6c\- \\xc0\xac\x29\xb7\xc9\x7c\x50\xdd\x3f\x84\xd5\xb5\xb5\x47\x09\x17\- \\x92\x16\xd5\xd9\x89\x79\xfb\x1b\- \\xd1\x31\x0b\xa6\x98\xdf\xb5\xac\x2f\xfd\x72\xdb\xd0\x1a\xdf\xb7\- \\xb8\xe1\xaf\xed\x6a\x26\x7e\x96\xba\x7c\x90\x45\xf1\x2c\x7f\x99\- \\x24\xa1\x99\x47\xb3\x91\x6c\xf7\x08\x01\xf2\xe2\x85\x8e\xfc\x16\- \\x63\x69\x20\xd8\x71\x57\x4e\x69\xa4\x58\xfe\xa3\xf4\x93\x3d\x7e\- \\x0d\x95\x74\x8f\x72\x8e\xb6\x58\x71\x8b\xcd\x58\x82\x15\x4a\xee\- \\x7b\x54\xa4\x1d\xc2\x5a\x59\xb5\x9c\x30\xd5\x39\x2a\xf2\x60\x13\- \\xc5\xd1\xb0\x23\x28\x60\x85\xf0\xca\x41\x79\x18\xb8\xdb\x38\xef\- \\x8e\x79\xdc\xb0\x60\x3a\x18\x0e\x6c\x9e\x0e\x8b\xb0\x1e\x8a\x3e\- \\xd7\x15\x77\xc1\xbd\x31\x4b\x27\x78\xaf\x2f\xda\x55\x60\x5c\x60\- \\xe6\x55\x25\xf3\xaa\x55\xab\x94\x57\x48\x98\x62\x63\xe8\x14\x40\- \\x55\xca\x39\x6a\x2a\xab\x10\xb6\xb4\xcc\x5c\x34\x11\x41\xe8\xce\- \\xa1\x54\x86\xaf\x7c\x72\xe9\x93\xb3\xee\x14\x11\x63\x6f\xbc\x2a\- \\x2b\xa9\xc5\x5d\x74\x18\x31\xf6\xce\x5c\x3e\x16\x9b\x87\x93\x1e\- \\xaf\xd6\xba\x33\x6c\x24\xcf\x5c\x7a\x32\x53\x81\x28\x95\x86\x77\- \\x3b\x8f\x48\x98\x6b\x4b\xb9\xaf\xc4\xbf\xe8\x1b\x66\x28\x21\x93\- \\x61\xd8\x09\xcc\xfb\x21\xa9\x91\x48\x7c\xac\x60\x5d\xec\x80\x32\- \\xef\x84\x5d\x5d\xe9\x85\x75\xb1\xdc\x26\x23\x02\xeb\x65\x1b\x88\- \\x23\x89\x3e\x81\xd3\x96\xac\xc5\x0f\x6d\x6f\xf3\x83\xf4\x42\x39\- \\x2e\x0b\x44\x82\xa4\x84\x20\x04\x69\xc8\xf0\x4a\x9e\x1f\x9b\x5e\- \\x21\xc6\x68\x42\xf6\xe9\x6c\x9a\x67\x0c\x9c\x61\xab\xd3\x88\xf0\- \\x6a\x51\xa0\xd2\xd8\x54\x2f\x68\x96\x0f\xa7\x28\xab\x51\x33\xa3\- \\x6e\xef\x0b\x6c\x13\x7a\x3b\xe4\xba\x3b\xf0\x50\x7e\xfb\x2a\x98\- \\xa1\xf1\x65\x1d\x39\xaf\x01\x76\x66\xca\x59\x3e\x82\x43\x0e\x88\- \\x8c\xee\x86\x19\x45\x6f\x9f\xb4\x7d\x84\xa5\xc3\x3b\x8b\x5e\xbe\- \\xe0\x6f\x75\xd8\x85\xc1\x20\x73\x40\x1a\x44\x9f\x56\xc1\x6a\xa6\- \\x4e\xd3\xaa\x62\x36\x3f\x77\x06\x1b\xfe\xdf\x72\x42\x9b\x02\x3d\- \\x37\xd0\xd7\x24\xd0\x0a\x12\x48\xdb\x0f\xea\xd3\x49\xf1\xc0\x9b\- \\x07\x53\x72\xc9\x80\x99\x1b\x7b\x25\xd4\x79\xd8\xf6\xe8\xde\xf7\- \\xe3\xfe\x50\x1a\xb6\x79\x4c\x3b\x97\x6c\xe0\xbd\x04\xc0\x06\xba\- \\xc1\xa9\x4f\xb6\x40\x9f\x60\xc4\x5e\x5c\x9e\xc2\x19\x6a\x24\x63\- \\x68\xfb\x6f\xaf\x3e\x6c\x53\xb5\x13\x39\xb2\xeb\x3b\x52\xec\x6f\- \\x6d\xfc\x51\x1f\x9b\x30\x95\x2c\xcc\x81\x45\x44\xaf\x5e\xbd\x09\- \\xbe\xe3\xd0\x04\xde\x33\x4a\xfd\x66\x0f\x28\x07\x19\x2e\x4b\xb3\- \\xc0\xcb\xa8\x57\x45\xc8\x74\x0f\xd2\x0b\x5f\x39\xb9\xd3\xfb\xdb\- \\x55\x79\xc0\xbd\x1a\x60\x32\x0a\xd6\xa1\x00\xc6\x40\x2c\x72\x79\- \\x67\x9f\x25\xfe\xfb\x1f\xa3\xcc\x8e\xa5\xe9\xf8\xdb\x32\x22\xf8\- \\x3c\x75\x16\xdf\xfd\x61\x6b\x15\x2f\x50\x1e\xc8\xad\x05\x52\xab\- \\x32\x3d\xb5\xfa\xfd\x23\x87\x60\x53\x31\x7b\x48\x3e\x00\xdf\x82\- \\x9e\x5c\x57\xbb\xca\x6f\x8c\xa0\x1a\x87\x56\x2e\xdf\x17\x69\xdb\- \\xd5\x42\xa8\xf6\x28\x7e\xff\xc3\xac\x67\x32\xc6\x8c\x4f\x55\x73\- \\x69\x5b\x27\xb0\xbb\xca\x58\xc8\xe1\xff\xa3\x5d\xb8\xf0\x11\xa0\- \\x10\xfa\x3d\x98\xfd\x21\x83\xb8\x4a\xfc\xb5\x6c\x2d\xd1\xd3\x5b\- \\x9a\x53\xe4\x79\xb6\xf8\x45\x65\xd2\x8e\x49\xbc\x4b\xfb\x97\x90\- \\xe1\xdd\xf2\xda\xa4\xcb\x7e\x33\x62\xfb\x13\x41\xce\xe4\xc6\xe8\- \\xef\x20\xca\xda\x36\x77\x4c\x01\xd0\x7e\x9e\xfe\x2b\xf1\x1f\xb4\- \\x95\xdb\xda\x4d\xae\x90\x91\x98\xea\xad\x8e\x71\x6b\x93\xd5\xa0\- \\xd0\x8e\xd1\xd0\xaf\xc7\x25\xe0\x8e\x3c\x5b\x2f\x8e\x75\x94\xb7\- \\x8f\xf6\xe2\xfb\xf2\x12\x2b\x64\x88\x88\xb8\x12\x90\x0d\xf0\x1c\- \\x4f\xad\x5e\xa0\x68\x8f\xc3\x1c\xd1\xcf\xf1\x91\xb3\xa8\xc1\xad\- \\x2f\x2f\x22\x18\xbe\x0e\x17\x77\xea\x75\x2d\xfe\x8b\x02\x1f\xa1\- \\xe5\xa0\xcc\x0f\xb5\x6f\x74\xe8\x18\xac\xf3\xd6\xce\x89\xe2\x99\- \\xb4\xa8\x4f\xe0\xfd\x13\xe0\xb7\x7c\xc4\x3b\x81\xd2\xad\xa8\xd9\- \\x16\x5f\xa2\x66\x80\x95\x77\x05\x93\xcc\x73\x14\x21\x1a\x14\x77\- \\xe6\xad\x20\x65\x77\xb5\xfa\x86\xc7\x54\x42\xf5\xfb\x9d\x35\xcf\- \\xeb\xcd\xaf\x0c\x7b\x3e\x89\xa0\xd6\x41\x1b\xd3\xae\x1e\x7e\x49\- \\x00\x25\x0e\x2d\x20\x71\xb3\x5e\x22\x68\x00\xbb\x57\xb8\xe0\xaf\- \\x24\x64\x36\x9b\xf0\x09\xb9\x1e\x55\x63\x91\x1d\x59\xdf\xa6\xaa\- \\x78\xc1\x43\x89\xd9\x5a\x53\x7f\x20\x7d\x5b\xa2\x02\xe5\xb9\xc5\- \\x83\x26\x03\x76\x62\x95\xcf\xa9\x11\xc8\x19\x68\x4e\x73\x4a\x41\- \\xb3\x47\x2d\xca\x7b\x14\xa9\x4a\x1b\x51\x00\x52\x9a\x53\x29\x15\- \\xd6\x0f\x57\x3f\xbc\x9b\xc6\xe4\x2b\x60\xa4\x76\x81\xe6\x74\x00\- \\x08\xba\x6f\xb5\x57\x1b\xe9\x1f\xf2\x96\xec\x6b\x2a\x0d\xd9\x15\- \\xb6\x63\x65\x21\xe7\xb9\xf9\xb6\xff\x34\x05\x2e\xc5\x85\x56\x64\- \\x53\xb0\x2d\x5d\xa9\x9f\x8f\xa1\x08\xba\x47\x99\x6e\x85\x07\x6a\- \\x4b\x7a\x70\xe9\xb5\xb3\x29\x44\xdb\x75\x09\x2e\xc4\x19\x26\x23\- \\xad\x6e\xa6\xb0\x49\xa7\xdf\x7d\x9c\xee\x60\xb8\x8f\xed\xb2\x66\- \\xec\xaa\x8c\x71\x69\x9a\x17\xff\x56\x64\x52\x6c\xc2\xb1\x9e\xe1\- \\x19\x36\x02\xa5\x75\x09\x4c\x29\xa0\x59\x13\x40\xe4\x18\x3a\x3e\- \\x3f\x54\x98\x9a\x5b\x42\x9d\x65\x6b\x8f\xe4\xd6\x99\xf7\x3f\xd6\- \\xa1\xd2\x9c\x07\xef\xe8\x30\xf5\x4d\x2d\x38\xe6\xf0\x25\x5d\xc1\- \\x4c\xdd\x20\x86\x84\x70\xeb\x26\x63\x82\xe9\xc6\x02\x1e\xcc\x5e\- \\x09\x68\x6b\x3f\x3e\xba\xef\xc9\x3c\x97\x18\x14\x6b\x6a\x70\xa1\- \\x68\x7f\x35\x84\x52\xa0\xe2\x86\xb7\x9c\x53\x05\xaa\x50\x07\x37\- \\x3e\x07\x84\x1c\x7f\xde\xae\x5c\x8e\x7d\x44\xec\x57\x16\xf2\xb8\- \\xb0\x3a\xda\x37\xf0\x50\x0c\x0d\xf0\x1c\x1f\x04\x02\x00\xb3\xff\- \\xae\x0c\xf5\x1a\x3c\xb5\x74\xb2\x25\x83\x7a\x58\xdc\x09\x21\xbd\- \\xd1\x91\x13\xf9\x7c\xa9\x2f\xf6\x94\x32\x47\x73\x22\xf5\x47\x01\- \\x3a\xe5\xe5\x81\x37\xc2\xda\xdc\xc8\xb5\x76\x34\x9a\xf3\xdd\xa7\- \\xa9\x44\x61\x46\x0f\xd0\x03\x0e\xec\xc8\xc7\x3e\xa4\x75\x1e\x41\- \\xe2\x38\xcd\x99\x3b\xea\x0e\x2f\x32\x80\xbb\xa1\x18\x3e\xb3\x31\- \\x4e\x54\x8b\x38\x4f\x6d\xb9\x08\x6f\x42\x0d\x03\xf6\x0a\x04\xbf\- \\x2c\xb8\x12\x90\x24\x97\x7c\x79\x56\x79\xb0\x72\xbc\xaf\x89\xaf\- \\xde\x9a\x77\x1f\xd9\x93\x08\x10\xb3\x8b\xae\x12\xdc\xcf\x3f\x2e\- \\x55\x12\x72\x1f\x2e\x6b\x71\x24\x50\x1a\xdd\xe6\x9f\x84\xcd\x87\- \\x7a\x58\x47\x18\x74\x08\xda\x17\xbc\x9f\x9a\xbc\xe9\x4b\x7d\x8c\- \\xec\x7a\xec\x3a\xdb\x85\x1d\xfa\x63\x09\x43\x66\xc4\x64\xc3\xd2\- \\xef\x1c\x18\x47\x32\x15\xd9\x08\xdd\x43\x3b\x37\x24\xc2\xba\x16\- \\x12\xa1\x4d\x43\x2a\x65\xc4\x51\x50\x94\x00\x02\x13\x3a\xe4\xdd\- \\x71\xdf\xf8\x9e\x10\x31\x4e\x55\x81\xac\x77\xd6\x5f\x11\x19\x9b\- \\x04\x35\x56\xf1\xd7\xa3\xc7\x6b\x3c\x11\x18\x3b\x59\x24\xa5\x09\- \\xf2\x8f\xe6\xed\x97\xf1\xfb\xfa\x9e\xba\xbf\x2c\x1e\x15\x3c\x6e\- \\x86\xe3\x45\x70\xea\xe9\x6f\xb1\x86\x0e\x5e\x0a\x5a\x3e\x2a\xb3\- \\x77\x1f\xe7\x1c\x4e\x3d\x06\xfa\x29\x65\xdc\xb9\x99\xe7\x1d\x0f\- \\x80\x3e\x89\xd6\x52\x66\xc8\x25\x2e\x4c\xc9\x78\x9c\x10\xb3\x6a\- \\xc6\x15\x0e\xba\x94\xe2\xea\x78\xa5\xfc\x3c\x53\x1e\x0a\x2d\xf4\- \\xf2\xf7\x4e\xa7\x36\x1d\x2b\x3d\x19\x39\x26\x0f\x19\xc2\x79\x60\- \\x52\x23\xa7\x08\xf7\x13\x12\xb6\xeb\xad\xfe\x6e\xea\xc3\x1f\x66\- \\xe3\xbc\x45\x95\xa6\x7b\xc8\x83\xb1\x7f\x37\xd1\x01\x8c\xff\x28\- \\xc3\x32\xdd\xef\xbe\x6c\x5a\xa5\x65\x58\x21\x85\x68\xab\x98\x02\- \\xee\xce\xa5\x0f\xdb\x2f\x95\x3b\x2a\xef\x7d\xad\x5b\x6e\x2f\x84\- \\x15\x21\xb6\x28\x29\x07\x61\x70\xec\xdd\x47\x75\x61\x9f\x15\x10\- \\x13\xcc\xa8\x30\xeb\x61\xbd\x96\x03\x34\xfe\x1e\xaa\x03\x63\xcf\- \\xb5\x73\x5c\x90\x4c\x70\xa2\x39\xd5\x9e\x9e\x0b\xcb\xaa\xde\x14\- \\xee\xcc\x86\xbc\x60\x62\x2c\xa7\x9c\xab\x5c\xab\xb2\xf3\x84\x6e\- \\x64\x8b\x1e\xaf\x19\xbd\xf0\xca\xa0\x23\x69\xb9\x65\x5a\xbb\x50\- \\x40\x68\x5a\x32\x3c\x2a\xb4\xb3\x31\x9e\xe9\xd5\xc0\x21\xb8\xf7\- \\x9b\x54\x0b\x19\x87\x5f\xa0\x99\x95\xf7\x99\x7e\x62\x3d\x7d\xa8\- \\xf8\x37\x88\x9a\x97\xe3\x2d\x77\x11\xed\x93\x5f\x16\x68\x12\x81\- \\x0e\x35\x88\x29\xc7\xe6\x1f\xd6\x96\xde\xdf\xa1\x78\x58\xba\x99\- \\x57\xf5\x84\xa5\x1b\x22\x72\x63\x9b\x83\xc3\xff\x1a\xc2\x46\x96\- \\xcd\xb3\x0a\xeb\x53\x2e\x30\x54\x8f\xd9\x48\xe4\x6d\xbc\x31\x28\- \\x58\xeb\xf2\xef\x34\xc6\xff\xea\xfe\x28\xed\x61\xee\x7c\x3c\x73\- \\x5d\x4a\x14\xd9\xe8\x64\xb7\xe3\x42\x10\x5d\x14\x20\x3e\x13\xe0\- \\x45\xee\xe2\xb6\xa3\xaa\xab\xea\xdb\x6c\x4f\x15\xfa\xcb\x4f\xd0\- \\xc7\x42\xf4\x42\xef\x6a\xbb\xb5\x65\x4f\x3b\x1d\x41\xcd\x21\x05\- \\xd8\x1e\x79\x9e\x86\x85\x4d\xc7\xe4\x4b\x47\x6a\x3d\x81\x62\x50\- \\xcf\x62\xa1\xf2\x5b\x8d\x26\x46\xfc\x88\x83\xa0\xc1\xc7\xb6\xa3\- \\x7f\x15\x24\xc3\x69\xcb\x74\x92\x47\x84\x8a\x0b\x56\x92\xb2\x85\- \\x09\x5b\xbf\x00\xad\x19\x48\x9d\x14\x62\xb1\x74\x23\x82\x0e\x00\- \\x58\x42\x8d\x2a\x0c\x55\xf5\xea\x1d\xad\xf4\x3e\x23\x3f\x70\x61\- \\x33\x72\xf0\x92\x8d\x93\x7e\x41\xd6\x5f\xec\xf1\x6c\x22\x3b\xdb\- \\x7c\xde\x37\x59\xcb\xee\x74\x60\x40\x85\xf2\xa7\xce\x77\x32\x6e\- \\xa6\x07\x80\x84\x19\xf8\x50\x9e\xe8\xef\xd8\x55\x61\xd9\x97\x35\- \\xa9\x69\xa7\xaa\xc5\x0c\x06\xc2\x5a\x04\xab\xfc\x80\x0b\xca\xdc\- \\x9e\x44\x7a\x2e\xc3\x45\x34\x84\xfd\xd5\x67\x05\x0e\x1e\x9e\xc9\- \\xdb\x73\xdb\xd3\x10\x55\x88\xcd\x67\x5f\xda\x79\xe3\x67\x43\x40\- \\xc5\xc4\x34\x65\x71\x3e\x38\xd8\x3d\x28\xf8\x9e\xf1\x6d\xff\x20\- \\x15\x3e\x21\xe7\x8f\xb0\x3d\x4a\xe6\xe3\x9f\x2b\xdb\x83\xad\xf7\- \\xe9\x3d\x5a\x68\x94\x81\x40\xf7\xf6\x4c\x26\x1c\x94\x69\x29\x34\- \\x41\x15\x20\xf7\x76\x02\xd4\xf7\xbc\xf4\x6b\x2e\xd4\xa2\x00\x68\- \\xd4\x08\x24\x71\x33\x20\xf4\x6a\x43\xb7\xd4\xb7\x50\x00\x61\xaf\- \\x1e\x39\xf6\x2e\x97\x24\x45\x46\x14\x21\x4f\x74\xbf\x8b\x88\x40\- \\x4d\x95\xfc\x1d\x96\xb5\x91\xaf\x70\xf4\xdd\xd3\x66\xa0\x2f\x45\- \\xbf\xbc\x09\xec\x03\xbd\x97\x85\x7f\xac\x6d\xd0\x31\xcb\x85\x04\- \\x96\xeb\x27\xb3\x55\xfd\x39\x41\xda\x25\x47\xe6\xab\xca\x0a\x9a\- \\x28\x50\x78\x25\x53\x04\x29\xf4\x0a\x2c\x86\xda\xe9\xb6\x6d\xfb\- \\x68\xdc\x14\x62\xd7\x48\x69\x00\x68\x0e\xc0\xa4\x27\xa1\x8d\xee\- \\x4f\x3f\xfe\xa2\xe8\x87\xad\x8c\xb5\x8c\xe0\x06\x7a\xf4\xd6\xb6\- \\xaa\xce\x1e\x7c\xd3\x37\x5f\xec\xce\x78\xa3\x99\x40\x6b\x2a\x42\- \\x20\xfe\x9e\x35\xd9\xf3\x85\xb9\xee\x39\xd7\xab\x3b\x12\x4e\x8b\- \\x1d\xc9\xfa\xf7\x4b\x6d\x18\x56\x26\xa3\x66\x31\xea\xe3\x97\xb2\- \\x3a\x6e\xfa\x74\xdd\x5b\x43\x32\x68\x41\xe7\xf7\xca\x78\x20\xfb\- \\xfb\x0a\xf5\x4e\xd8\xfe\xb3\x97\x45\x40\x56\xac\xba\x48\x95\x27\- \\x55\x53\x3a\x3a\x20\x83\x8d\x87\xfe\x6b\xa9\xb7\xd0\x96\x95\x4b\- \\x55\xa8\x67\xbc\xa1\x15\x9a\x58\xcc\xa9\x29\x63\x99\xe1\xdb\x33\- \\xa6\x2a\x4a\x56\x3f\x31\x25\xf9\x5e\xf4\x7e\x1c\x90\x29\x31\x7c\- \\xfd\xf8\xe8\x02\x04\x27\x2f\x70\x80\xbb\x15\x5c\x05\x28\x2c\xe3\- \\x95\xc1\x15\x48\xe4\xc6\x6d\x22\x48\xc1\x13\x3f\xc7\x0f\x86\xdc\- \\x07\xf9\xc9\xee\x41\x04\x1f\x0f\x40\x47\x79\xa4\x5d\x88\x6e\x17\- \\x32\x5f\x51\xeb\xd5\x9b\xc0\xd1\xf2\xbc\xc1\x8f\x41\x11\x35\x64\- \\x25\x7b\x78\x34\x60\x2a\x9c\x60\xdf\xf8\xe8\xa3\x1f\x63\x6c\x1b\- \\x0e\x12\xb4\xc2\x02\xe1\x32\x9e\xaf\x66\x4f\xd1\xca\xd1\x81\x15\- \\x6b\x23\x95\xe0\x33\x3e\x92\xe1\x3b\x24\x0b\x62\xee\xbe\xb9\x22\- \\x85\xb2\xa2\x0e\xe6\xba\x0d\x99\xde\x72\x0c\x8c\x2d\xa2\xf7\x28\- \\xd0\x12\x78\x45\x95\xb7\x94\xfd\x64\x7d\x08\x62\xe7\xcc\xf5\xf0\- \\x54\x49\xa3\x6f\x87\x7d\x48\xfa\xc3\x9d\xfd\x27\xf3\x3e\x8d\x1e\- \\x0a\x47\x63\x41\x99\x2e\xff\x74\x3a\x6f\x6e\xab\xf4\xf8\xfd\x37\- \\xa8\x12\xdc\x60\xa1\xeb\xdd\xf8\x99\x1b\xe1\x4c\xdb\x6e\x6b\x0d\- \\xc6\x7b\x55\x10\x6d\x67\x2c\x37\x27\x65\xd4\x3b\xdc\xd0\xe8\x04\- \\xf1\x29\x0d\xc7\xcc\x00\xff\xa3\xb5\x39\x0f\x92\x69\x0f\xed\x0b\- \\x66\x7b\x9f\xfb\xce\xdb\x7d\x9c\xa0\x91\xcf\x0b\xd9\x15\x5e\xa3\- \\xbb\x13\x2f\x88\x51\x5b\xad\x24\x7b\x94\x79\xbf\x76\x3b\xd6\xeb\- \\x37\x39\x2e\xb3\xcc\x11\x59\x79\x80\x26\xe2\x97\xf4\x2e\x31\x2d\- \\x68\x42\xad\xa7\xc6\x6a\x2b\x3b\x12\x75\x4c\xcc\x78\x2e\xf1\x1c\- \\x6a\x12\x42\x37\xb7\x92\x51\xe7\x06\xa1\xbb\xe6\x4b\xfb\x63\x50\- \\x1a\x6b\x10\x18\x11\xca\xed\xfa\x3d\x25\xbd\xd8\xe2\xe1\xc3\xc9\- \\x44\x42\x16\x59\x0a\x12\x13\x86\xd9\x0c\xec\x6e\xd5\xab\xea\x2a\- \\x64\xaf\x67\x4e\xda\x86\xa8\x5f\xbe\xbf\xe9\x88\x64\xe4\xc3\xfe\- \\x9d\xbc\x80\x57\xf0\xf7\xc0\x86\x60\x78\x7b\xf8\x60\x03\x60\x4d\- \\xd1\xfd\x83\x46\xf6\x38\x1f\xb0\x77\x45\xae\x04\xd7\x36\xfc\xcc\- \\x83\x42\x6b\x33\xf0\x1e\xab\x71\xb0\x80\x41\x87\x3c\x00\x5e\x5f\- \\x77\xa0\x57\xbe\xbd\xe8\xae\x24\x55\x46\x42\x99\xbf\x58\x2e\x61\- \\x4e\x58\xf4\x8f\xf2\xdd\xfd\xa2\xf4\x74\xef\x38\x87\x89\xbd\xc2\- \\x53\x66\xf9\xc3\xc8\xb3\x8e\x74\xb4\x75\xf2\x55\x46\xfc\xd9\xb9\- \\x7a\xeb\x26\x61\x8b\x1d\xdf\x84\x84\x6a\x0e\x79\x91\x5f\x95\xe2\- \\x46\x6e\x59\x8e\x20\xb4\x57\x70\x8c\xd5\x55\x91\xc9\x02\xde\x4c\- \\xb9\x0b\xac\xe1\xbb\x82\x05\xd0\x11\xa8\x62\x48\x75\x74\xa9\x9e\- \\xb7\x7f\x19\xb6\xe0\xa9\xdc\x09\x66\x2d\x09\xa1\xc4\x32\x46\x33\- \\xe8\x5a\x1f\x02\x09\xf0\xbe\x8c\x4a\x99\xa0\x25\x1d\x6e\xfe\x10\- \\x1a\xb9\x3d\x1d\x0b\xa5\xa4\xdf\xa1\x86\xf2\x0f\x28\x68\xf1\x69\- \\xdc\xb7\xda\x83\x57\x39\x06\xfe\xa1\xe2\xce\x9b\x4f\xcd\x7f\x52\- \\x50\x11\x5e\x01\xa7\x06\x83\xfa\xa0\x02\xb5\xc4\x0d\xe6\xd0\x27\- \\x9a\xf8\x8c\x27\x77\x3f\x86\x41\xc3\x60\x4c\x06\x61\xa8\x06\xb5\- \\xf0\x17\x7a\x28\xc0\xf5\x86\xe0\x00\x60\x58\xaa\x30\xdc\x7d\x62\- \\x11\xe6\x9e\xd7\x23\x38\xea\x63\x53\xc2\xdd\x94\xc2\xc2\x16\x34\- \\xbb\xcb\xee\x56\x90\xbc\xb6\xde\xeb\xfc\x7d\xa1\xce\x59\x1d\x76\- \\x6f\x05\xe4\x09\x4b\x7c\x01\x88\x39\x72\x0a\x3d\x7c\x92\x7c\x24\- \\x86\xe3\x72\x5f\x72\x4d\x9d\xb9\x1a\xc1\x5b\xb4\xd3\x9e\xb8\xfc\- \\xed\x54\x55\x78\x08\xfc\xa5\xb5\xd8\x3d\x7c\xd3\x4d\xad\x0f\xc4\- \\x1e\x50\xef\x5e\xb1\x61\xe6\xf8\xa2\x85\x14\xd9\x6c\x51\x13\x3c\- \\x6f\xd5\xc7\xe7\x56\xe1\x4e\xc4\x36\x2a\xbf\xce\xdd\xc6\xc8\x37\- \\xd7\x9a\x32\x34\x92\x63\x82\x12\x67\x0e\xfa\x8e\x40\x60\x00\xe0\- \\x3a\x39\xce\x37\xd3\xfa\xf5\xcf\xab\xc2\x77\x37\x5a\xc5\x2d\x1b\- \\x5c\xb0\x67\x9e\x4f\xa3\x37\x42\xd3\x82\x27\x40\x99\xbc\x9b\xbe\- \\xd5\x11\x8e\x9d\xbf\x0f\x73\x15\xd6\x2d\x1c\x7e\xc7\x00\xc4\x7b\- \\xb7\x8c\x1b\x6b\x21\xa1\x90\x45\xb2\x6e\xb1\xbe\x6a\x36\x6e\xb4\- \\x57\x48\xab\x2f\xbc\x94\x6e\x79\xc6\xa3\x76\xd2\x65\x49\xc2\xc8\- \\x53\x0f\xf8\xee\x46\x8d\xde\x7d\xd5\x73\x0a\x1d\x4c\xd0\x4d\xc6\- \\x29\x39\xbb\xdb\xa9\xba\x46\x50\xac\x95\x26\xe8\xbe\x5e\xe3\x04\- \\xa1\xfa\xd5\xf0\x6a\x2d\x51\x9a\x63\xef\x8c\xe2\x9a\x86\xee\x22\- \\xc0\x89\xc2\xb8\x43\x24\x2e\xf6\xa5\x1e\x03\xaa\x9c\xf2\xd0\xa4\- \\x83\xc0\x61\xba\x9b\xe9\x6a\x4d\x8f\xe5\x15\x50\xba\x64\x5b\xd6\- \\x28\x26\xa2\xf9\xa7\x3a\x3a\xe1\x4b\xa9\x95\x86\xef\x55\x62\xe9\- \\xc7\x2f\xef\xd3\xf7\x52\xf7\xda\x3f\x04\x6f\x69\x77\xfa\x0a\x59\- \\x80\xe4\xa9\x15\x87\xb0\x86\x01\x9b\x09\xe6\xad\x3b\x3e\xe5\x93\- \\xe9\x90\xfd\x5a\x9e\x34\xd7\x97\x2c\xf0\xb7\xd9\x02\x2b\x8b\x51\- \\x96\xd5\xac\x3a\x01\x7d\xa6\x7d\xd1\xcf\x3e\xd6\x7c\x7d\x2d\x28\- \\x1f\x9f\x25\xcf\xad\xf2\xb8\x9b\x5a\xd6\xb4\x72\x5a\x88\xf5\x4c\- \\xe0\x29\xac\x71\xe0\x19\xa5\xe6\x47\xb0\xac\xfd\xed\x93\xfa\x9b\- \\xe8\xd3\xc4\x8d\x28\x3b\x57\xcc\xf8\xd5\x66\x29\x79\x13\x2e\x28\- \\x78\x5f\x01\x91\xed\x75\x60\x55\xf7\x96\x0e\x44\xe3\xd3\x5e\x8c\- \\x15\x05\x6d\xd4\x88\xf4\x6d\xba\x03\xa1\x61\x25\x05\x64\xf0\xbd\- \\xc3\xeb\x9e\x15\x3c\x90\x57\xa2\x97\x27\x1a\xec\xa9\x3a\x07\x2a\- \\x1b\x3f\x6d\x9b\x1e\x63\x21\xf5\xf5\x9c\x66\xfb\x26\xdc\xf3\x19\- \\x75\x33\xd9\x28\xb1\x55\xfd\xf5\x03\x56\x34\x82\x8a\xba\x3c\xbb\- \\x28\x51\x77\x11\xc2\x0a\xd9\xf8\xab\xcc\x51\x67\xcc\xad\x92\x5f\- \\x4d\xe8\x17\x51\x38\x30\xdc\x8e\x37\x9d\x58\x62\x93\x20\xf9\x91\- \\xea\x7a\x90\xc2\xfb\x3e\x7b\xce\x51\x21\xce\x64\x77\x4f\xbe\x32\- \\xa8\xb6\xe3\x7e\xc3\x29\x3d\x46\x48\xde\x53\x69\x64\x13\xe6\x80\- \\xa2\xae\x08\x10\xdd\x6d\xb2\x24\x69\x85\x2d\xfd\x09\x07\x21\x66\- \\xb3\x9a\x46\x0a\x64\x45\xc0\xdd\x58\x6c\xde\xcf\x1c\x20\xc8\xae\- \\x5b\xbe\xf7\xdd\x1b\x58\x8d\x40\xcc\xd2\x01\x7f\x6b\xb4\xe3\xbb\- \\xdd\xa2\x6a\x7e\x3a\x59\xff\x45\x3e\x35\x0a\x44\xbc\xb4\xcd\xd5\- \\x72\xea\xce\xa8\xfa\x64\x84\xbb\x8d\x66\x12\xae\xbf\x3c\x6f\x47\- \\xd2\x9b\xe4\x63\x54\x2f\x5d\x9e\xae\xc2\x77\x1b\xf6\x4e\x63\x70\- \\x74\x0e\x0d\x8d\xe7\x5b\x13\x57\xf8\x72\x16\x71\xaf\x53\x7d\x5d\- \\x40\x40\xcb\x08\x4e\xb4\xe2\xcc\x34\xd2\x46\x6a\x01\x15\xaf\x84\- \\xe1\xb0\x04\x28\x95\x98\x3a\x1d\x06\xb8\x9f\xb4\xce\x6e\xa0\x48\- \\x6f\x3f\x3b\x82\x35\x20\xab\x82\x01\x1a\x1d\x4b\x27\x72\x27\xf8\- \\x61\x15\x60\xb1\xe7\x93\x3f\xdc\xbb\x3a\x79\x2b\x34\x45\x25\xbd\- \\xa0\x88\x39\xe1\x51\xce\x79\x4b\x2f\x32\xc9\xb7\xa0\x1f\xba\xc9\- \\xe0\x1c\xc8\x7e\xbc\xc7\xd1\xf6\xcf\x01\x11\xc3\xa1\xe8\xaa\xc7\- \\x1a\x90\x87\x49\xd4\x4f\xbd\x9a\xd0\xda\xde\xcb\xd5\x0a\xda\x38\- \\x03\x39\xc3\x2a\xc6\x91\x36\x67\x8d\xf9\x31\x7c\xe0\xb1\x2b\x4f\- \\xf7\x9e\x59\xb7\x43\xf5\xbb\x3a\xf2\xd5\x19\xff\x27\xd9\x45\x9c\- \\xbf\x97\x22\x2c\x15\xe6\xfc\x2a\x0f\x91\xfc\x71\x9b\x94\x15\x25\- \\xfa\xe5\x93\x61\xce\xb6\x9c\xeb\xc2\xa8\x64\x59\x12\xba\xa8\xd1\- \\xb6\xc1\x07\x5e\xe3\x05\x6a\x0c\x10\xd2\x50\x65\xcb\x03\xa4\x42\- \\xe0\xec\x6e\x0e\x16\x98\xdb\x3b\x4c\x98\xa0\xbe\x32\x78\xe9\x64\- \\x9f\x1f\x95\x32\xe0\xd3\x92\xdf\xd3\xa0\x34\x2b\x89\x71\xf2\x1e\- \\x1b\x0a\x74\x41\x4b\xa3\x34\x8c\xc5\xbe\x71\x20\xc3\x76\x32\xd8\- \\xdf\x35\x9f\x8d\x9b\x99\x2f\x2e\xe6\x0b\x6f\x47\x0f\xe3\xf1\x1d\- \\xe5\x4c\xda\x54\x1e\xda\xd8\x91\xce\x62\x79\xcf\xcd\x3e\x7e\x6f\- \\x16\x18\xb1\x66\xfd\x2c\x1d\x05\x84\x8f\xd2\xc5\xf6\xfb\x22\x99\- \\xf5\x23\xf3\x57\xa6\x32\x76\x23\x93\xa8\x35\x31\x56\xcc\xcd\x02\- \\xac\xf0\x81\x62\x5a\x75\xeb\xb5\x6e\x16\x36\x97\x88\xd2\x73\xcc\- \\xde\x96\x62\x92\x81\xb9\x49\xd0\x4c\x50\x90\x1b\x71\xc6\x56\x14\- \\xe6\xc6\xc7\xbd\x32\x7a\x14\x0a\x45\xe1\xd0\x06\xc3\xf2\x7b\x9a\- \\xc9\xaa\x53\xfd\x62\xa8\x0f\x00\xbb\x25\xbf\xe2\x35\xbd\xd2\xf6\- \\x71\x12\x69\x05\xb2\x04\x02\x22\xb6\xcb\xcf\x7c\xcd\x76\x9c\x2b\- \\x53\x11\x3e\xc0\x16\x40\xe3\xd3\x38\xab\xbd\x60\x25\x47\xad\xf0\- \\xba\x38\x20\x9c\xf7\x46\xce\x76\x77\xaf\xa1\xc5\x20\x75\x60\x60\- \\x85\xcb\xfe\x4e\x8a\xe8\x8d\xd8\x7a\xaa\xf9\xb0\x4c\xf9\xaa\x7e\- \\x19\x48\xc2\x5c\x02\xfb\x8a\x8c\x01\xc3\x6a\xe4\xd6\xeb\xe1\xf9\- \\x90\xd4\xf8\x69\xa6\x5c\xde\xa0\x3f\x09\x25\x2d\xc2\x08\xe6\x9f\- \\xb7\x4e\x61\x32\xce\x77\xe2\x5b\x57\x8f\xdf\xe3\x3a\xc3\x72\xe6\- \"#
@@ -12,273 +12,139 @@ -- License : BSD-style -- Stability : experimental -- Portability : Good+--+-- The cipher itself is in C, as is the key setup bcrypt wraps around it:+-- what the schedule costs is the whole of what bcrypt is for, and in Haskell+-- it cost about twice what the usual implementations do. module Crypto.Cipher.Blowfish.Primitive ( Context, initBlowfish, encrypt, decrypt,- KeySchedule,- createKeySchedule,- freezeKeySchedule,- expandKey,- expandKeyWithSalt,- cipherBlockMutable,+ bcryptHash,+ bcryptPbkdfHash, ) where -import Control.Monad (when)-import Data.Bits-import Data.Memory.Endian-import Data.Word--import Crypto.Cipher.Blowfish.Box import Crypto.Error-import Crypto.Internal.ByteArray (ByteArray, ByteArrayAccess)+import Crypto.Internal.ByteArray (ByteArray, ByteArrayAccess, ScrubbedBytes) import qualified Crypto.Internal.ByteArray as B import Crypto.Internal.Compat import Crypto.Internal.Imports-import Crypto.Internal.WordArray+import Foreign.C.Types (CInt (..))+import Foreign.Ptr (Ptr) -newtype Context = Context Array32+-- | The key schedule: the P array and the four S boxes, as the C keeps them.+newtype Context = Context ScrubbedBytes instance NFData Context where rnf a = a `seq` () +-- | How many bytes of schedule the C wants: eighteen words and four boxes of+-- two hundred and fifty-six.+contextSize :: Int+contextSize = (18 + 4 * 256) * 4+ -- | Initialize a new Blowfish context from a key. -- -- key needs to be between 0 and 448 bits. initBlowfish :: ByteArrayAccess key => key -> CryptoFailable Context initBlowfish key | B.length key > (448 `div` 8) = CryptoFailed CryptoError_KeySizeInvalid- | otherwise = CryptoPassed $ unsafeDoIO $ do- ks <- createKeySchedule- expandKey ks key- freezeKeySchedule ks---- | Get an immutable Blowfish context by freezing a mutable key schedule.-freezeKeySchedule :: KeySchedule -> IO Context-freezeKeySchedule (KeySchedule ma) = Context `fmap` mutableArray32Freeze ma--expandKey :: ByteArrayAccess key => KeySchedule -> key -> IO ()-expandKey ks@(KeySchedule ma) key = do- when (B.length key > 0) $ iterKeyStream key 0 0 $ \i l r a0 a1 cont -> do- mutableArrayWriteXor32 ma i l- mutableArrayWriteXor32 ma (i + 1) r- when (i + 2 < 18) (cont a0 a1)- loop 0 0 0- where- loop i l r = do- n <- cipherBlockMutable ks (fromIntegral l `shiftL` 32 .|. fromIntegral r)- let nl = fromIntegral (n `shiftR` 32)- nr = fromIntegral (n .&. 0xffffffff)- mutableArrayWrite32 ma i nl- mutableArrayWrite32 ma (i + 1) nr- when (i < 18 + 1024) (loop (i + 2) nl nr)--expandKeyWithSalt- :: (ByteArrayAccess key, ByteArrayAccess salt)- => KeySchedule- -> key- -> salt- -> IO ()-expandKeyWithSalt ks key salt- | B.length salt == 16 =- expandKeyWithSalt128- ks- key- (fromBE $ B.toW64BE salt 0)- (fromBE $ B.toW64BE salt 8)- | otherwise = expandKeyWithSaltAny ks key salt--expandKeyWithSaltAny- :: (ByteArrayAccess key, ByteArrayAccess salt)- => KeySchedule- -- ^ The key schedule- -> key- -- ^ The key- -> salt- -- ^ The salt- -> IO ()-expandKeyWithSaltAny ks@(KeySchedule ma) key salt = do- when (B.length key > 0) $ iterKeyStream key 0 0 $ \i l r a0 a1 cont -> do- mutableArrayWriteXor32 ma i l- mutableArrayWriteXor32 ma (i + 1) r- when (i + 2 < 18) (cont a0 a1)- -- Go through the entire key schedule overwriting the P-Array and S-Boxes- when (B.length salt > 0) $ iterKeyStream salt 0 0 $ \i l r a0 a1 cont -> do- let l' = xor l a0- let r' = xor r a1- n <- cipherBlockMutable ks (fromIntegral l' `shiftL` 32 .|. fromIntegral r')- let nl = fromIntegral (n `shiftR` 32)- nr = fromIntegral (n .&. 0xffffffff)- mutableArrayWrite32 ma i nl- mutableArrayWrite32 ma (i + 1) nr- when (i + 2 < 18 + 1024) (cont nl nr)--expandKeyWithSalt128- :: ByteArrayAccess ba- => KeySchedule- -- ^ The key schedule- -> ba- -- ^ The key- -> Word64- -- ^ First word of the salt- -> Word64- -- ^ Second word of the salt- -> IO ()-expandKeyWithSalt128 ks@(KeySchedule ma) key salt1 salt2 = do- when (B.length key > 0) $ iterKeyStream key 0 0 $ \i l r a0 a1 cont -> do- mutableArrayWriteXor32 ma i l- mutableArrayWriteXor32 ma (i + 1) r- when (i + 2 < 18) (cont a0 a1)- -- Go through the entire key schedule overwriting the P-Array and S-Boxes- loop 0 salt1 salt1 salt2- where- loop i input slt1 slt2- | i == 1042 = return ()- | otherwise = do- n <- cipherBlockMutable ks input- let nl = fromIntegral (n `shiftR` 32)- nr = fromIntegral (n .&. 0xffffffff)- mutableArrayWrite32 ma i nl- mutableArrayWrite32 ma (i + 1) nr- loop (i + 2) (n `xor` slt2) slt2 slt1+ | otherwise = CryptoPassed $+ unsafeDoIO $+ fmap Context $+ B.alloc contextSize $ \ctx ->+ B.withByteArray key $ \k ->+ c_blowfish_init ctx k (fromIntegral (B.length key)) -- | Encrypt blocks -- -- Input need to be a multiple of 8 bytes encrypt :: ByteArray ba => Context -> ba -> ba-encrypt ctx ba- | B.length ba == 0 = B.empty- | B.length ba `mod` 8 /= 0 = error "invalid data length"- | otherwise = B.mapAsWord64 (cipherBlock ctx False) ba+encrypt = through c_blowfish_encrypt -- | Decrypt blocks -- -- Input need to be a multiple of 8 bytes decrypt :: ByteArray ba => Context -> ba -> ba-decrypt ctx ba- | B.length ba == 0 = B.empty- | B.length ba `mod` 8 /= 0 = error "invalid data length"- | otherwise = B.mapAsWord64 (cipherBlock ctx True) ba+decrypt = through c_blowfish_decrypt --- | Encrypt or decrypt a single block of 64 bits.------ The inverse argument decides whether to encrypt or decrypt.-cipherBlock :: Context -> Bool -> Word64 -> Word64-cipherBlock (Context ar) inverse input = doRound input 0+through+ :: ByteArray ba+ => (Ptr Word8 -> Ptr Word8 -> Ptr Word8 -> Word32 -> IO ())+ -> Context+ -> ba+ -> ba+through f (Context ctx) input+ | len `mod` 8 /= 0 =+ error "Crypto.Cipher.Blowfish: input length must be a multiple of 8"+ | otherwise = unsafeDoIO $+ B.alloc len $ \out ->+ B.withByteArray ctx $ \c ->+ B.withByteArray input $ \i -> f c out i (fromIntegral len) where- -- \| Transform the input over 16 rounds- doRound :: Word64 -> Int -> Word64- doRound !i roundIndex- | roundIndex == 16 =- let final = (fromIntegral (p 16) `shiftL` 32) .|. fromIntegral (p 17)- in rotateL (i `xor` final) 32- | otherwise =- let newr = fromIntegral (i `shiftR` 32) `xor` p roundIndex- newi = ((i `shiftL` 32) `xor` f newr) .|. fromIntegral newr- in doRound newi (roundIndex + 1)+ len = B.length input - -- \| The Blowfish Feistel function F- f :: Word32 -> Word64- f t =- let a = s0 (0xff .&. (t `shiftR` 24))- b = s1 (0xff .&. (t `shiftR` 16))- c = s2 (0xff .&. (t `shiftR` 8))- d = s3 (0xff .&. t)- in fromIntegral (((a + b) `xor` c) + d) `shiftL` 32+-- | What bcrypt does with Blowfish: the key setup that costs what the cost+-- says, and then the sixty-four encryptions. The answer is 24 bytes, of+-- which bcrypt keeps 23.+--+-- The salt has to be 16 bytes and the key 1 to 73, which is a password of at+-- most 72 with the zero byte the original implementation appends. 'Nothing'+-- means it was given something else.+bcryptHash+ :: (ByteArrayAccess salt, ByteArrayAccess key, ByteArray output)+ => Int+ -- ^ the cost, between 4 and 31+ -> salt+ -> key+ -> Maybe output+bcryptHash cost salt key+ | cost < 4 || cost > 31 = Nothing+ | B.length salt /= 16 = Nothing+ | B.length key < 1 || B.length key > 73 = Nothing+ | otherwise = unsafeDoIO $ do+ (r, out) <- B.allocRet 24 $ \o ->+ B.withByteArray salt $ \s ->+ B.withByteArray key $ \k ->+ c_bcrypt o (fromIntegral cost) s k (fromIntegral (B.length key))+ return $ if r == 0 then Just out else Nothing - -- \| S-Box arrays, each containing 256 32-bit words- -- The first 18 words contain the P-Array of subkeys- s0, s1, s2, s3 :: Word32 -> Word32- s0 i = arrayRead32 ar (fromIntegral i + 18)- s1 i = arrayRead32 ar (fromIntegral i + 274)- s2 i = arrayRead32 ar (fromIntegral i + 530)- s3 i = arrayRead32 ar (fromIntegral i + 786)- p :: Int -> Word32- p i- | inverse = arrayRead32 ar (17 - i)- | otherwise = arrayRead32 ar i+foreign import ccall unsafe "crypton_blowfish_init"+ c_blowfish_init :: Ptr Word8 -> Ptr Word8 -> Word32 -> IO () --- | Blowfish encrypt a Word using the current state of the key schedule-cipherBlockMutable :: KeySchedule -> Word64 -> IO Word64-cipherBlockMutable (KeySchedule ma) input = doRound input 0- where- -- \| Transform the input over 16 rounds- doRound !i roundIndex- | roundIndex == 16 = do- pVal1 <- mutableArrayRead32 ma 16- pVal2 <- mutableArrayRead32 ma 17- let final = (fromIntegral pVal1 `shiftL` 32) .|. fromIntegral pVal2- return $ rotateL (i `xor` final) 32- | otherwise = do- pVal <- mutableArrayRead32 ma roundIndex- let newr = fromIntegral (i `shiftR` 32) `xor` pVal- newr' <- f newr- let newi = ((i `shiftL` 32) `xor` newr') .|. fromIntegral newr- doRound newi (roundIndex + 1)+foreign import ccall unsafe "crypton_blowfish_encrypt"+ c_blowfish_encrypt :: Ptr Word8 -> Ptr Word8 -> Ptr Word8 -> Word32 -> IO () - -- \| The Blowfish Feistel function F- f :: Word32 -> IO Word64- f t = do- a <- s0 (0xff .&. (t `shiftR` 24))- b <- s1 (0xff .&. (t `shiftR` 16))- c <- s2 (0xff .&. (t `shiftR` 8))- d <- s3 (0xff .&. t)- return (fromIntegral (((a + b) `xor` c) + d) `shiftL` 32)+foreign import ccall unsafe "crypton_blowfish_decrypt"+ c_blowfish_decrypt :: Ptr Word8 -> Ptr Word8 -> Ptr Word8 -> Word32 -> IO () - -- \| S-Box arrays, each containing 256 32-bit words- -- The first 18 words contain the P-Array of subkeys- s0, s1, s2, s3 :: Word32 -> IO Word32- s0 i = mutableArrayRead32 ma (fromIntegral i + 18)- s1 i = mutableArrayRead32 ma (fromIntegral i + 274)- s2 i = mutableArrayRead32 ma (fromIntegral i + 530)- s3 i = mutableArrayRead32 ma (fromIntegral i + 786)+-- the work is what the cost says, so this one may take a while: it is a safe+-- call, which lets the other capabilities carry on while it does+foreign import ccall safe "crypton_bcrypt"+ c_bcrypt :: Ptr Word8 -> Word32 -> Ptr Word8 -> Ptr Word8 -> Word32 -> IO CInt -iterKeyStream- :: ByteArrayAccess x- => x- -> Word32- -> Word32- -> ( Int- -> Word32- -> Word32- -> Word32- -> Word32- -> (Word32 -> Word32 -> IO ())- -> IO ()- )+-- | What bcrypt_pbkdf does with Blowfish: the same key setup sixty-four times+-- over, and then the four blocks of its own magic. Writes 32 bytes where it+-- is pointed, which is what the caller of this one wants.+bcryptPbkdfHash+ :: (ByteArrayAccess pass, ByteArrayAccess salt)+ => pass+ -> salt+ -> Ptr Word8 -> IO ()-iterKeyStream x a0 a1 g = f 0 0 a0 a1- where- len = B.length x- -- Avoiding the modulo operation when interating over the ring- -- buffer is assumed to be more efficient here. All other- -- implementations do this, too. The branch prediction shall prefer- -- the branch with the increment.- n j = if j + 1 >= len then 0 else j + 1- f i j0 b0 b1 = g i l r b0 b1 (f (i + 2) j8)- where- j1 = n j0- j2 = n j1- j3 = n j2- j4 = n j3- j5 = n j4- j6 = n j5- j7 = n j6- j8 = n j7- x0 = fromIntegral (B.index x j0)- x1 = fromIntegral (B.index x j1)- x2 = fromIntegral (B.index x j2)- x3 = fromIntegral (B.index x j3)- x4 = fromIntegral (B.index x j4)- x5 = fromIntegral (B.index x j5)- x6 = fromIntegral (B.index x j6)- x7 = fromIntegral (B.index x j7)- l = shiftL x0 24 .|. shiftL x1 16 .|. shiftL x2 8 .|. x3- r = shiftL x4 24 .|. shiftL x5 16 .|. shiftL x6 8 .|. x7-{-# INLINE iterKeyStream #-}+bcryptPbkdfHash pass salt out =+ B.withByteArray pass $ \p ->+ B.withByteArray salt $ \s -> do+ _ <-+ c_bcrypt_pbkdf_hash+ out+ p+ (fromIntegral (B.length pass))+ s+ (fromIntegral (B.length salt))+ return () --- Benchmarking shows that GHC considers this function too big to inline--- although forcing inlining causes an actual improvement.--- It is assumed that all function calls (especially the continuation)--- collapse into a tight loop after inlining.+foreign import ccall safe "crypton_bcrypt_pbkdf_hash"+ c_bcrypt_pbkdf_hash+ :: Ptr Word8 -> Ptr Word8 -> Word32 -> Ptr Word8 -> Word32 -> IO CInt
@@ -1,4 +1,4 @@-{-# LANGUAGE MagicHash #-}+{-# LANGUAGE ForeignFunctionInterface #-} -- | -- Module : Crypto.Cipher.Camellia.Primitive@@ -7,6 +7,8 @@ -- Stability : experimental -- Portability : Good --+-- Camellia with a 128-bit key, over the C in @cbits/crypton_camellia.c@.+-- -- This only cover Camellia 128 bits for now. The API will change once -- 192 and 256 mode are implemented too. module Crypto.Cipher.Camellia.Primitive (@@ -16,296 +18,65 @@ decrypt, ) where -import Data.Bits-import Data.Word- import Crypto.Error-import Crypto.Internal.ByteArray (ByteArray, ByteArrayAccess)+import Crypto.Internal.ByteArray (ByteArray, ByteArrayAccess, Bytes) import qualified Crypto.Internal.ByteArray as B-import Crypto.Internal.WordArray-import Crypto.Internal.Words-import Data.Memory.Endian--data Mode = Decrypt | Encrypt--w64tow128 :: (Word64, Word64) -> Word128-w64tow128 (x1, x2) = Word128 x1 x2--w64tow8 :: Word64 -> (Word8, Word8, Word8, Word8, Word8, Word8, Word8, Word8)-w64tow8 x = (t1, t2, t3, t4, t5, t6, t7, t8)- where- t1 = fromIntegral (x `shiftR` 56)- t2 = fromIntegral (x `shiftR` 48)- t3 = fromIntegral (x `shiftR` 40)- t4 = fromIntegral (x `shiftR` 32)- t5 = fromIntegral (x `shiftR` 24)- t6 = fromIntegral (x `shiftR` 16)- t7 = fromIntegral (x `shiftR` 8)- t8 = fromIntegral (x)--w8tow64 :: (Word8, Word8, Word8, Word8, Word8, Word8, Word8, Word8) -> Word64-w8tow64 (t1, t2, t3, t4, t5, t6, t7, t8) =- (fromIntegral t1 `shiftL` 56)- .|. (fromIntegral t2 `shiftL` 48)- .|. (fromIntegral t3 `shiftL` 40)- .|. (fromIntegral t4 `shiftL` 32)- .|. (fromIntegral t5 `shiftL` 24)- .|. (fromIntegral t6 `shiftL` 16)- .|. (fromIntegral t7 `shiftL` 8)- .|. (fromIntegral t8)--sbox :: Int -> Word8-sbox = arrayRead8 t- where- t =- array8- "\x70\x82\x2c\xec\xb3\x27\xc0\xe5\xe4\x85\x57\x35\xea\x0c\xae\x41\- \\x23\xef\x6b\x93\x45\x19\xa5\x21\xed\x0e\x4f\x4e\x1d\x65\x92\xbd\- \\x86\xb8\xaf\x8f\x7c\xeb\x1f\xce\x3e\x30\xdc\x5f\x5e\xc5\x0b\x1a\- \\xa6\xe1\x39\xca\xd5\x47\x5d\x3d\xd9\x01\x5a\xd6\x51\x56\x6c\x4d\- \\x8b\x0d\x9a\x66\xfb\xcc\xb0\x2d\x74\x12\x2b\x20\xf0\xb1\x84\x99\- \\xdf\x4c\xcb\xc2\x34\x7e\x76\x05\x6d\xb7\xa9\x31\xd1\x17\x04\xd7\- \\x14\x58\x3a\x61\xde\x1b\x11\x1c\x32\x0f\x9c\x16\x53\x18\xf2\x22\- \\xfe\x44\xcf\xb2\xc3\xb5\x7a\x91\x24\x08\xe8\xa8\x60\xfc\x69\x50\- \\xaa\xd0\xa0\x7d\xa1\x89\x62\x97\x54\x5b\x1e\x95\xe0\xff\x64\xd2\- \\x10\xc4\x00\x48\xa3\xf7\x75\xdb\x8a\x03\xe6\xda\x09\x3f\xdd\x94\- \\x87\x5c\x83\x02\xcd\x4a\x90\x33\x73\x67\xf6\xf3\x9d\x7f\xbf\xe2\- \\x52\x9b\xd8\x26\xc8\x37\xc6\x3b\x81\x96\x6f\x4b\x13\xbe\x63\x2e\- \\xe9\x79\xa7\x8c\x9f\x6e\xbc\x8e\x29\xf5\xf9\xb6\x2f\xfd\xb4\x59\- \\x78\x98\x06\x6a\xe7\x46\x71\xba\xd4\x25\xab\x42\x88\xa2\x8d\xfa\- \\x72\x07\xb9\x55\xf8\xee\xac\x0a\x36\x49\x2a\x68\x3c\x38\xf1\xa4\- \\x40\x28\xd3\x7b\xbb\xc9\x43\xc1\x15\xe3\xad\xf4\x77\xc7\x80\x9e"#--sbox1 :: Word8 -> Word8-sbox1 x = sbox (fromIntegral x)--sbox2 :: Word8 -> Word8-sbox2 x = sbox1 x `rotateL` 1--sbox3 :: Word8 -> Word8-sbox3 x = sbox1 x `rotateL` 7--sbox4 :: Word8 -> Word8-sbox4 x = sbox1 (x `rotateL` 1)--sigma1, sigma2, sigma3, sigma4, sigma5, sigma6 :: Word64-sigma1 = 0xA09E667F3BCC908B-sigma2 = 0xB67AE8584CAA73B2-sigma3 = 0xC6EF372FE94F82BE-sigma4 = 0x54FF53A5F1D36F1C-sigma5 = 0x10E527FADE682D1D-sigma6 = 0xB05688C2B3E6C1FD--rotl128 :: Word128 -> Int -> Word128-rotl128 v 0 = v-rotl128 (Word128 x1 x2) 64 = Word128 x2 x1-rotl128 v@(Word128 x1 x2) w- | w > 64 = (v `rotl128` 64) `rotl128` (w - 64)- | otherwise = Word128 (x1high .|. x2low) (x2high .|. x1low)- where- splitBits i = (i .&. complement x, i .&. x)- where- x = 2 ^ w - 1- (x1high, x1low) = splitBits (x1 `rotateL` w)- (x2high, x2low) = splitBits (x2 `rotateL` w)---- | Camellia context-data Camellia = Camellia- { k :: Array64- , kw :: Array64- , ke :: Array64- }+import Crypto.Internal.Compat (unsafeDoIO)+import Data.Word+import Foreign.Ptr (Ptr) -setKeyInterim- :: ByteArrayAccess key => key -> (Word128, Word128, Word128, Word128)-setKeyInterim keyseed = (w64tow128 kL, w64tow128 kR, w64tow128 kA, w64tow128 kB)- where- kL = (fromBE $ B.toW64BE keyseed 0, fromBE $ B.toW64BE keyseed 8)- kR = (0, 0)+-- | The subkeys of RFC 3713 section 2.2: kw, k and ke, as 26 64-bit words.+newtype Camellia = Camellia Bytes+ deriving (Eq) - kA =- let d1 = (fst kL `xor` fst kR)- d2 = (snd kL `xor` snd kR)- d3 = d2 `xor` feistel d1 sigma1- d4 = d1 `xor` feistel d3 sigma2- d5 = d4 `xor` (fst kL)- d6 = d3 `xor` (snd kL)- d7 = d6 `xor` feistel d5 sigma3- d8 = d5 `xor` feistel d7 sigma4- in (d8, d7)+scheduleSize :: Int+scheduleSize = 26 * 8 - kB =- let d1 = (fst kA `xor` fst kR)- d2 = (snd kA `xor` snd kR)- d3 = d2 `xor` feistel d1 sigma5- d4 = d1 `xor` feistel d3 sigma6- in (d4, d3)+blockBytes :: Int+blockBytes = 16 --- | Initialize a 128-bit key------ Return the initialized key or a error message if the given--- keyseed was not 16-bytes in length.-initCamellia- :: ByteArray key- => key- -- ^ The key to create the camellia context- -> CryptoFailable Camellia+-- | Initialize a 128-bit key.+initCamellia :: ByteArrayAccess key => key -> CryptoFailable Camellia initCamellia key- | B.length key /= 16 = CryptoFailed $ CryptoError_KeySizeInvalid+ | B.length key /= 16 = CryptoFailed CryptoError_KeySizeInvalid | otherwise =- let (kL, _, kA, _) = setKeyInterim key- in let (Word128 kw1 kw2) = (kL `rotl128` 0)- in let (Word128 k1 k2) = (kA `rotl128` 0)- in let (Word128 k3 k4) = (kL `rotl128` 15)- in let (Word128 k5 k6) = (kA `rotl128` 15)- in let (Word128 ke1 ke2) = (kA `rotl128` 30) -- ke1 = (KA <<< 30) >> 64; ke2 = (KA <<< 30) & MASK64;- in let (Word128 k7 k8) = (kL `rotl128` 45) -- k7 = (KL <<< 45) >> 64; k8 = (KL <<< 45) & MASK64;- in let (Word128 k9 _) = (kA `rotl128` 45) -- k9 = (KA <<< 45) >> 64;- in let (Word128 _ k10) = (kL `rotl128` 60)- in let (Word128 k11 k12) = (kA `rotl128` 60)- in let (Word128 ke3 ke4) = (kL `rotl128` 77)- in let (Word128 k13 k14) = (kL `rotl128` 94)- in let (Word128 k15 k16) = (kA `rotl128` 94)- in let (Word128 k17 k18) = (kL `rotl128` 111)- in let (Word128 kw3 kw4) = (kA `rotl128` 111)- in CryptoPassed $- Camellia- { kw = array64 4 [kw1, kw2, kw3, kw4]- , ke = array64 4 [ke1, ke2, ke3, ke4]- , k =- array64- 18- [ k1- , k2- , k3- , k4- , k5- , k6- , k7- , k8- , k9- , k10- , k11- , k12- , k13- , k14- , k15- , k16- , k17- , k18- ]- }--feistel :: Word64 -> Word64 -> Word64-feistel fin sk =- let x = fin `xor` sk- in let (t1, t2, t3, t4, t5, t6, t7, t8) = w64tow8 x- in let t1' = sbox1 t1- in let t2' = sbox2 t2- in let t3' = sbox3 t3- in let t4' = sbox4 t4- in let t5' = sbox2 t5- in let t6' = sbox3 t6- in let t7' = sbox4 t7- in let t8' = sbox1 t8- in let y1 = t1' `xor` t3' `xor` t4' `xor` t6' `xor` t7' `xor` t8'- in let y2 = t1' `xor` t2' `xor` t4' `xor` t5' `xor` t7' `xor` t8'- in let y3 = t1' `xor` t2' `xor` t3' `xor` t5' `xor` t6' `xor` t8'- in let y4 = t2' `xor` t3' `xor` t4' `xor` t5' `xor` t6' `xor` t7'- in let y5 = t1' `xor` t2' `xor` t6' `xor` t7' `xor` t8'- in let y6 = t2' `xor` t3' `xor` t5' `xor` t7' `xor` t8'- in let y7 = t3' `xor` t4' `xor` t5' `xor` t6' `xor` t8'- in let y8 = t1' `xor` t4' `xor` t5' `xor` t6' `xor` t7'- in w8tow64 (y1, y2, y3, y4, y5, y6, y7, y8)--fl :: Word64 -> Word64 -> Word64-fl fin sk =- let (x1, x2) = w64to32 fin- in let (k1, k2) = w64to32 sk- in let y2 = x2 `xor` ((x1 .&. k1) `rotateL` 1)- in let y1 = x1 `xor` (y2 .|. k2)- in w32to64 (y1, y2)--flinv :: Word64 -> Word64 -> Word64-flinv fin sk =- let (y1, y2) = w64to32 fin- in let (k1, k2) = w64to32 sk- in let x1 = y1 `xor` (y2 .|. k2)- in let x2 = y2 `xor` ((x1 .&. k1) `rotateL` 1)- in w32to64 (x1, x2)--{- in decrypt mode 0->17 1->16 ... -}-getKeyK :: Mode -> Camellia -> Int -> Word64-getKeyK Encrypt key i = k key `arrayRead64` i-getKeyK Decrypt key i = k key `arrayRead64` (17 - i)--{- in decrypt mode 0->3 1->2 2->1 3->0 -}-getKeyKe :: Mode -> Camellia -> Int -> Word64-getKeyKe Encrypt key i = ke key `arrayRead64` i-getKeyKe Decrypt key i = ke key `arrayRead64` (3 - i)--{- in decrypt mode 0->2 1->3 2->0 3->1 -}-getKeyKw :: Mode -> Camellia -> Int -> Word64-getKeyKw Encrypt key i = (kw key) `arrayRead64` i-getKeyKw Decrypt key i = (kw key) `arrayRead64` ((i + 2) `mod` 4)--{- perform the following- D2 = D2 ^ F(D1, k1); // Round 1- D1 = D1 ^ F(D2, k2); // Round 2- D2 = D2 ^ F(D1, k3); // Round 3- D1 = D1 ^ F(D2, k4); // Round 4- D2 = D2 ^ F(D1, k5); // Round 5- D1 = D1 ^ F(D2, k6); // Round 6- -}-doBlockRound :: Mode -> Camellia -> Word64 -> Word64 -> Int -> (Word64, Word64)-doBlockRound mode key d1 d2 i =- let r1 = d2 `xor` feistel d1 (getKeyK mode key (0 + i {- Round 1+i -}))- in let r2 = d1 `xor` feistel r1 (getKeyK mode key (1 + i {- Round 2+i -}))- in let r3 = r1 `xor` feistel r2 (getKeyK mode key (2 + i {- Round 3+i -}))- in let r4 = r2 `xor` feistel r3 (getKeyK mode key (3 + i {- Round 4+i -}))- in let r5 = r3 `xor` feistel r4 (getKeyK mode key (4 + i {- Round 5+i -}))- in let r6 = r4 `xor` feistel r5 (getKeyK mode key (5 + i {- Round 6+i -}))- in (r6, r5)--doBlock :: Mode -> Camellia -> Word128 -> Word128-doBlock mode key (Word128 d1 d2) =- let d1a = d1 `xor` (getKeyKw mode key 0 {- Prewhitening -})- in let d2a = d2 `xor` (getKeyKw mode key 1)- in let (d1b, d2b) = doBlockRound mode key d1a d2a 0- in let d1c = fl d1b (getKeyKe mode key 0 {- FL -})- in let d2c = flinv d2b (getKeyKe mode key 1 {- FLINV -})- in let (d1d, d2d) = doBlockRound mode key d1c d2c 6- in let d1e = fl d1d (getKeyKe mode key 2 {- FL -})- in let d2e = flinv d2d (getKeyKe mode key 3 {- FLINV -})- in let (d1f, d2f) = doBlockRound mode key d1e d2e 12- in let d2g = d2f `xor` (getKeyKw mode key 2 {- Postwhitening -})- in let d1g = d1f `xor` (getKeyKw mode key 3)- in w64tow128 (d2g, d1g)+ CryptoPassed $+ Camellia $+ B.allocAndFreeze scheduleSize $ \ks ->+ B.withByteArray key $ \k -> c_camellia_init ks k -{- encryption for 128 bits blocks -}-encryptBlock :: Camellia -> Word128 -> Word128-encryptBlock = doBlock Encrypt+-- | Encrypt the given input, which has to be a whole number of blocks.+encrypt :: ByteArray ba => Camellia -> ba -> ba+encrypt = run c_camellia_encrypt -{- decryption for 128 bits blocks -}-decryptBlock :: Camellia -> Word128 -> Word128-decryptBlock = doBlock Decrypt+-- | Decrypt the given input, which has to be a whole number of blocks.+decrypt :: ByteArray ba => Camellia -> ba -> ba+decrypt = run c_camellia_decrypt --- | Encrypts the given ByteString using the given Key-encrypt+run :: ByteArray ba- => Camellia- -- ^ The key to use+ => (Ptr Word8 -> Ptr Word8 -> Ptr Word8 -> Word32 -> IO ())+ -> Camellia -> ba- -- ^ The data to encrypt -> ba-encrypt key = B.mapAsWord128 (encryptBlock key)+run f (Camellia sched) input+ | len `mod` blockBytes /= 0 =+ error $+ "Crypto.Cipher.Camellia: input length must be a multiple of block size (16). Its length is: "+ ++ show len+ | otherwise = unsafeDoIO $+ B.alloc len $ \out ->+ B.withByteArray sched $ \ks ->+ B.withByteArray input $ \inp ->+ f out ks inp (fromIntegral (len `div` blockBytes))+ where+ len = B.length input --- | Decrypts the given ByteString using the given Key-decrypt- :: ByteArray ba- => Camellia- -- ^ The key to use- -> ba- -- ^ The data to decrypt- -> ba-decrypt key = B.mapAsWord128 (decryptBlock key)+foreign import ccall unsafe "crypton_camellia.h crypton_camellia_init"+ c_camellia_init :: Ptr Word8 -> Ptr Word8 -> IO ()++foreign import ccall unsafe "crypton_camellia.h crypton_camellia_encrypt"+ c_camellia_encrypt :: Ptr Word8 -> Ptr Word8 -> Ptr Word8 -> Word32 -> IO ()++foreign import ccall unsafe "crypton_camellia.h crypton_camellia_decrypt"+ c_camellia_decrypt :: Ptr Word8 -> Ptr Word8 -> Ptr Word8 -> Word32 -> IO ()
@@ -183,10 +183,19 @@ | B.null src = (B.empty, prevSt) | otherwise = unsafeDoIO $ do (out, st) <- B.copyRet prevStMem $ \ctx ->- B.alloc (B.length src) $ \dstPtr ->+ B.alloc n $ \dstPtr -> B.withByteArray src $ \srcPtr ->- ccrypton_chacha_combine dstPtr ctx srcPtr (fromIntegral $ B.length src)+ -- in pieces the C's uint32_t length can hold; it carries+ -- the state in ctx, so it can simply be called again+ B.inCLengths n $ \off len ->+ ccrypton_chacha_combine+ (dstPtr `plusPtr` off)+ ctx+ (srcPtr `plusPtr` off)+ (fromIntegral len) return (out, State st)+ where+ n = B.length src -- | Generate a number of bytes from the ChaCha output directly generate@@ -201,7 +210,11 @@ | otherwise = unsafeDoIO $ do (out, st) <- B.copyRet prevStMem $ \ctx -> B.alloc len $ \dstPtr ->- ccrypton_chacha_generate dstPtr ctx (fromIntegral len)+ B.inCLengths len $ \off n ->+ ccrypton_chacha_generate+ (dstPtr `plusPtr` off)+ ctx+ (fromIntegral n) return (out, State st) -- | similar to 'generate' but assume certains values
@@ -0,0 +1,335 @@+-- |+-- Module : Crypto.Cipher.ChaCha.Poly1305+-- License : BSD-style+-- Maintainer : Kazu Yamamoto <kazu@iij.ad.jp>+-- Stability : experimental+-- Portability : Good+--+-- ChaCha20-Poly1305 (RFC 8439) a message at a time.+--+-- "Crypto.Cipher.ChaChaPoly1305" takes a message in pieces: a state is+-- started, the additional data appended, the body encrypted and the tag+-- taken, each a step of its own. That is what a protocol wants when the+-- message arrives in pieces, and it is eight foreign calls and the+-- allocations between them when the message was already whole.+--+-- Here the whole message goes in one call.+--+-- The functions are the same shape as "Crypto.Cipher.AES.GCM", so a protocol+-- that offers both ciphers can hold them the same way.+module Crypto.Cipher.ChaCha.Poly1305 (+ Context,+ newContext,+ encrypt,+ decrypt,+ decryptWithTag,+) where++import Crypto.Cipher.Types (AuthTag (..))+import Crypto.Error+import Crypto.Internal.ByteArray (ByteArray, ByteArrayAccess)+import qualified Crypto.Internal.ByteArray as B+import Crypto.Internal.Compat (unsafeDoIO)+import Crypto.Internal.Imports+import Foreign.C.Types (CInt (..), CUInt (..))+import Foreign.Ptr (Ptr, plusPtr)++-- | A key, checked once.+--+-- ChaCha20-Poly1305 has nothing to precompute from a key: the one-time+-- Poly1305 key comes from the nonce, so it differs for every message. This+-- holds the thirty-two bytes and the knowledge that they are thirty-two, and+-- exists so that the interface is the one "Crypto.Cipher.AES.GCM" has.+newtype Context = Context B.ScrubbedBytes++instance NFData Context where+ rnf (Context k) = k `seq` ()++-- | Take a key of 32 bytes. Any other length is reported as+-- 'CryptoError_KeySizeInvalid'.+newContext :: ByteArrayAccess key => key -> CryptoFailable Context+newContext k+ | B.length k /= 32 = CryptoFailed CryptoError_KeySizeInvalid+ | otherwise = CryptoPassed $ Context (B.convert k)+{-# INLINABLE newContext #-}++-- | Encrypt one message. The result is the ciphertext with the tag after it,+-- which is the shape 'decrypt' expects.+--+-- The nonce is the twelve bytes RFC 8439 defines; any other length gives+-- 'CryptoError_IvSizeInvalid'. RFC 8439 requires a 16-byte tag.+{-# INLINABLE encrypt #-}+encrypt+ :: ( ByteArrayAccess nonce+ , ByteArrayAccess aad+ , ByteArrayAccess ba+ , ByteArray output+ )+ => Context+ -> nonce+ -> aad+ -> ba+ -> Int+ -> CryptoFailable output+encrypt (Context k) nonce aad input taglen+ | tooLongForC aad input = CryptoFailed CryptoError_ParameterInvalid+ | not (validNonce nonce) = CryptoFailed CryptoError_IvSizeInvalid+ | badTag taglen = CryptoFailed CryptoError_AuthenticationTagSizeInvalid+ | otherwise =+ CryptoPassed $+ unsafeDoIO $+ B.alloc (B.length input + taglen) $ \out ->+ B.withByteArray k $ \kp ->+ B.withByteArray nonce $ \np ->+ B.withByteArray aad $ \ap ->+ B.withByteArray input $ \ip ->+ (callE (B.length input))+ out+ (out `plusPtr` B.length input)+ (fromIntegral taglen)+ kp+ np+ (fromIntegral $ B.length nonce)+ ap+ (fromIntegral $ B.length aad)+ ip+ (fromIntegral $ B.length input)++-- | Decrypt one message, in the shape 'encrypt' produced: the ciphertext with+-- its tag after it. The tag is compared here, every byte of it whatever the+-- answer, and a message whose tag does not match gives 'Nothing' rather than+-- the plaintext.+--+-- 'Nothing' also comes back when the input is shorter than the tag, or the+-- nonce is not twelve bytes.+{-# INLINABLE decrypt #-}+decrypt+ :: (ByteArrayAccess nonce, ByteArrayAccess aad, ByteArray ba)+ => Context+ -> nonce+ -> aad+ -> ba+ -> Int+ -> Maybe ba+decrypt (Context k) nonce aad input taglen+ | tooLongForC aad input = Nothing+ | not (validNonce nonce) = Nothing+ | badTag taglen || B.length input < taglen = Nothing+ | otherwise = unsafeDoIO $ do+ (r, out) <- B.allocRet bodylen $ \outp ->+ B.withByteArray k $ \kp ->+ B.withByteArray nonce $ \np ->+ B.withByteArray aad $ \ap ->+ B.withByteArray body $ \ip ->+ B.withByteArray tag $ \tp ->+ (callD bodylen)+ outp+ tp+ (fromIntegral taglen)+ kp+ np+ (fromIntegral $ B.length nonce)+ ap+ (fromIntegral $ B.length aad)+ ip+ (fromIntegral bodylen)+ return $ if r /= 0 then Just out else Nothing+ where+ bodylen = B.length input - taglen+ (body, tag) = B.splitAt bodylen input++-- | Decrypt one message, the tag kept apart, and hand back the tag this end+-- computed.+--+-- For a caller whose protocol carries the tag separately from the ciphertext,+-- so that 'decrypt' -- which wants the two together and compares them itself+-- -- does not fit. Compare the two tags with '=='; the 'Eq' instance of+-- t'AuthTag' is a constant-time comparison, and taking them apart to compare+-- the bytes is how this goes wrong.+--+-- Nothing here says whether the message is authentic. Until the comparison+-- is made and has come out equal, what this returns is not plaintext, it is+-- what the ciphertext turns into, and a caller must not act on it.+{-# INLINABLE decryptWithTag #-}+decryptWithTag+ :: (ByteArrayAccess nonce, ByteArrayAccess aad, ByteArray ba)+ => Context+ -> nonce+ -> aad+ -> ba+ -> Int+ -> CryptoFailable (ba, AuthTag)+decryptWithTag (Context k) nonce aad input taglen+ | tooLongForC aad input = CryptoFailed CryptoError_ParameterInvalid+ | not (validNonce nonce) = CryptoFailed CryptoError_IvSizeInvalid+ | badTag taglen = CryptoFailed CryptoError_AuthenticationTagSizeInvalid+ | otherwise = CryptoPassed $ unsafeDoIO $ do+ (tagbs, out) <- B.allocRet (B.length input) $ \outp ->+ B.alloc taglen $ \tagp ->+ B.withByteArray k $ \kp ->+ B.withByteArray nonce $ \np ->+ B.withByteArray aad $ \ap ->+ B.withByteArray input $ \ip ->+ (callT (B.length input))+ outp+ tagp+ (fromIntegral taglen)+ kp+ np+ (fromIntegral $ B.length nonce)+ ap+ (fromIntegral $ B.length aad)+ ip+ (fromIntegral $ B.length input)+ return (out, AuthTag $ B.convert (tagbs :: B.Bytes))++-- | The C takes its lengths as @uint32_t@, so a message or its additional+-- data from 2^32 bytes up cannot be handed to it: the length would be+-- truncated and most of the buffer left untouched, with nothing to say so.+-- The C does the whole message in one call, so there is no splitting it.+tooLongForC+ :: (ByteArrayAccess aad, ByteArrayAccess ba) => aad -> ba -> Bool+tooLongForC aad input =+ B.overCLength (B.length aad) || B.overCLength (B.length input)++-- RFC 8439 is the twelve-byte nonce. ChaCha20 will take eight, but that is+-- the other construction, with a 64-bit block counter, and it is not what+-- this AEAD is defined over -- so it is refused here rather than quietly+-- encrypting under a scheme nobody asked for.+validNonce :: ByteArrayAccess nonce => nonce -> Bool+validNonce n = B.length n == 12++badTag :: Int -> Bool+badTag t = t /= 16++-- | An unsafe call keeps a capability for as long as it runs, so it is only+-- for a message short enough that the run is short. Four kibibytes is what+-- the AES side uses, and it takes in a datagram of any size a network will+-- carry.+shortMessage :: Int+shortMessage = 4096++callE :: Int -> CEncrypt+callE n+ | n <= shortMessage = c_chachapoly_encrypt_unsafe+ | otherwise = c_chachapoly_encrypt++callD :: Int -> CDecrypt+callD n+ | n <= shortMessage = c_chachapoly_decrypt_unsafe+ | otherwise = c_chachapoly_decrypt++callT :: Int -> CEncrypt+callT n+ | n <= shortMessage = c_chachapoly_decrypt_tag_unsafe+ | otherwise = c_chachapoly_decrypt_tag++type CEncrypt =+ Ptr Word8+ -> Ptr Word8+ -> CUInt+ -> Ptr Word8+ -> Ptr Word8+ -> CUInt+ -> Ptr Word8+ -> CUInt+ -> Ptr Word8+ -> CUInt+ -> IO ()++type CDecrypt =+ Ptr Word8+ -> Ptr Word8+ -> CUInt+ -> Ptr Word8+ -> Ptr Word8+ -> CUInt+ -> Ptr Word8+ -> CUInt+ -> Ptr Word8+ -> CUInt+ -> IO CInt++foreign import ccall "crypton_chachapoly.h crypton_chachapoly_encrypt"+ c_chachapoly_encrypt+ :: Ptr Word8+ -> Ptr Word8+ -> CUInt+ -> Ptr Word8+ -> Ptr Word8+ -> CUInt+ -> Ptr Word8+ -> CUInt+ -> Ptr Word8+ -> CUInt+ -> IO ()++foreign import ccall unsafe "crypton_chachapoly.h crypton_chachapoly_encrypt"+ c_chachapoly_encrypt_unsafe+ :: Ptr Word8+ -> Ptr Word8+ -> CUInt+ -> Ptr Word8+ -> Ptr Word8+ -> CUInt+ -> Ptr Word8+ -> CUInt+ -> Ptr Word8+ -> CUInt+ -> IO ()++foreign import ccall "crypton_chachapoly.h crypton_chachapoly_decrypt"+ c_chachapoly_decrypt+ :: Ptr Word8+ -> Ptr Word8+ -> CUInt+ -> Ptr Word8+ -> Ptr Word8+ -> CUInt+ -> Ptr Word8+ -> CUInt+ -> Ptr Word8+ -> CUInt+ -> IO CInt++foreign import ccall unsafe "crypton_chachapoly.h crypton_chachapoly_decrypt"+ c_chachapoly_decrypt_unsafe+ :: Ptr Word8+ -> Ptr Word8+ -> CUInt+ -> Ptr Word8+ -> Ptr Word8+ -> CUInt+ -> Ptr Word8+ -> CUInt+ -> Ptr Word8+ -> CUInt+ -> IO CInt++foreign import ccall "crypton_chachapoly.h crypton_chachapoly_decrypt_tag"+ c_chachapoly_decrypt_tag+ :: Ptr Word8+ -> Ptr Word8+ -> CUInt+ -> Ptr Word8+ -> Ptr Word8+ -> CUInt+ -> Ptr Word8+ -> CUInt+ -> Ptr Word8+ -> CUInt+ -> IO ()++foreign import ccall unsafe "crypton_chachapoly.h crypton_chachapoly_decrypt_tag"+ c_chachapoly_decrypt_tag_unsafe+ :: Ptr Word8+ -> Ptr Word8+ -> CUInt+ -> Ptr Word8+ -> Ptr Word8+ -> CUInt+ -> Ptr Word8+ -> CUInt+ -> Ptr Word8+ -> CUInt+ -> IO ()
@@ -1,3 +1,5 @@+{-# LANGUAGE GeneralizedNewtypeDeriving #-}+ -- | -- Module : Crypto.Cipher.ChaChaPoly1305 -- License : BSD-style@@ -28,7 +30,7 @@ -- > -> ByteString -- input plaintext to be encrypted -- > -> CryptoFailable ByteString -- ciphertext with a 128-bit tag attached -- >encrypt nonce key header plaintext = do--- > st1 <- C.nonce12 nonce >>= C.initialize key+-- > st1 <- C.initialize <$> C.key key <*> C.nonce12 nonce -- > let -- > st2 = C.finalizeAAD $ C.appendAAD header st1 -- > (out, st3) = C.encrypt plaintext st2@@ -41,6 +43,8 @@ -- * Low level State,+ Key,+ key, Nonce, XNonce, nonce12,@@ -68,6 +72,7 @@ ) import qualified Crypto.Internal.ByteArray as B import Crypto.Internal.Imports+import qualified Crypto.Internal.Poly1305 as PolyKey import qualified Crypto.MAC.Poly1305 as Poly1305 import qualified Data.ByteArray.Pack as P import Data.Memory.Endian@@ -174,40 +179,39 @@ -- -- The key length need to be 256 bits, and the nonce -- procured using either `nonce8` or `nonce12`-initialize- :: ByteArrayAccess key- => key -> Nonce -> CryptoFailable State-initialize key (Nonce8 nonce) = initialize' key nonce-initialize key (Nonce12 nonce) = initialize' key nonce+-- | A ChaCha20Poly1305 key: thirty-two bytes, checked once here rather than+-- at every use, so that 'initialize' and 'initializeX' cannot fail.+newtype Key = Key ScrubbedBytes+ deriving (ByteArrayAccess, Eq, NFData) -initialize'- :: ByteArrayAccess key- => key -> Bytes -> CryptoFailable State-initialize' key nonce- | B.length key /= 32 = CryptoFailed CryptoError_KeySizeInvalid- | otherwise = CryptoPassed $ initFromRootState rootState- where- rootState = ChaCha.initialize 20 key nonce+-- | Take thirty-two bytes for a key. A different length is reported as+-- 'CryptoError_KeySizeInvalid'; nothing else about a key can be wrong.+key :: ByteArrayAccess ba => ba -> CryptoFailable Key+key k+ | B.length k /= 32 = CryptoFailed CryptoError_KeySizeInvalid+ | otherwise = CryptoPassed $ Key $ B.convert k +initialize :: Key -> Nonce -> State+initialize k (Nonce8 nonce) = initialize' k nonce+initialize k (Nonce12 nonce) = initialize' k nonce++initialize' :: Key -> Bytes -> State+initialize' k nonce = initFromRootState (ChaCha.initialize 20 k nonce)+ initFromRootState :: ChaCha.State -> State initFromRootState rootState = State encState polyState 0 0 where (polyKey, encState) = ChaCha.generate rootState 64- polyState =- throwCryptoError $ Poly1305.initialize (B.take 32 polyKey :: ScrubbedBytes)+ -- 64 bytes are generated so the ChaCha state advances a whole block, and+ -- the first 32 of them are the key, so there is no length left to check+ polyState = Poly1305.initialize (PolyKey.Key (B.take 32 polyKey)) -- | Initialize a new XChaChaPoly1305 State -- -- The key length needs to be 256 bits, and the nonce -- procured using `nonce24`.-initializeX- :: ByteArrayAccess key- => key -> XNonce -> CryptoFailable State-initializeX key (Nonce24 nonce)- | B.length key /= 32 = CryptoFailed CryptoError_KeySizeInvalid- | otherwise = CryptoPassed $ initFromRootState rootState- where- rootState = ChaCha.initializeX 20 key nonce+initializeX :: Key -> XNonce -> State+initializeX k (Nonce24 nonce) = initFromRootState (ChaCha.initializeX 20 k nonce) -- | Append Authenticated Data to the State and return -- the new modified State.@@ -263,8 +267,8 @@ aeadChacha20poly1305Init :: (ByteArrayAccess k, ByteArrayAccess n) => k -> n -> CryptoFailable (AEAD ChaCha20Poly1305)-aeadChacha20poly1305Init key nonce = do- st0 <- nonce12 nonce >>= initialize key+aeadChacha20poly1305Init k nonce = do+ st0 <- initialize <$> key k <*> nonce12 nonce return $ AEAD model st0 where model =
@@ -4,6 +4,9 @@ -- Maintainer : Vincent Hanquez <vincent@snarc.org> -- Stability : stable -- Portability : good+--+-- DES, which is here because callers still meet it rather than because it+-- should be chosen: its 56-bit key is exhaustible. Prefer "Crypto.Cipher.AES". module Crypto.Cipher.DES ( DES, ) where@@ -13,11 +16,9 @@ import Crypto.Error import Crypto.Internal.ByteArray (ByteArrayAccess) import qualified Crypto.Internal.ByteArray as B-import Data.Memory.Endian-import Data.Word -- | DES Context-data DES = DES Word64+data DES = DES Schedule Schedule deriving (Eq) instance Cipher DES where@@ -27,13 +28,11 @@ instance BlockCipher DES where blockSize _ = 8- ecbEncrypt (DES key) = B.mapAsWord64 (unBlock . encrypt key . Block)- ecbDecrypt (DES key) = B.mapAsWord64 (unBlock . decrypt key . Block)+ ecbEncrypt (DES enc _) = ecb enc+ ecbDecrypt (DES _ dec) = ecb dec initDES :: ByteArrayAccess key => key -> CryptoFailable DES initDES k- | len == 8 = CryptoPassed $ DES key- | otherwise = CryptoFailed $ CryptoError_KeySizeInvalid- where- len = B.length k- key = fromBE $ B.toW64BE k 0+ | B.length k == 8 =+ CryptoPassed $ DES (schedule [(Encrypt, k)]) (schedule [(Decrypt, k)])+ | otherwise = CryptoFailed CryptoError_KeySizeInvalid
@@ -1,570 +1,83 @@-{-# LANGUAGE FlexibleInstances #-}--------------------------------------------------------------------------------------------------------------------------------------------------------------+{-# LANGUAGE ForeignFunctionInterface #-} -- |--- Module : Crypto.Cipher.DES.Primitive--- License : BSD-style+-- Module : Crypto.Cipher.DES.Primitive+-- License : BSD-style+-- Stability : experimental+-- Portability : Good ----- This module is copy of DES module from Crypto package.--- http://hackage.haskell.org/package/Crypto+-- The DES block operation, as FIPS 46-3 defines it, over the C in+-- @cbits/crypton_des.c@.+--+-- A t'Schedule' holds the round keys of one or more stages in the order they+-- are applied, which is what lets single DES and the three stage constructions+-- share one entry point. module Crypto.Cipher.DES.Primitive (- encrypt,- decrypt,- Block (..),+ Schedule,+ Direction (..),+ schedule,+ ecb, ) where -import Data.Bits+import Crypto.Internal.ByteArray (ByteArray, ByteArrayAccess, Bytes)+import qualified Crypto.Internal.ByteArray as B+import Crypto.Internal.Compat (unsafeDoIO) import Data.Word---- | a DES block (64 bits)-newtype Block = Block {unBlock :: Word64}--type Rotation = Int-type Key = Word64--type Bits4 = [Bool]-type Bits6 = [Bool]-type Bits32 = [Bool]-type Bits48 = [Bool]-type Bits56 = [Bool]-type Bits64 = [Bool]--desXor :: [Bool] -> [Bool] -> [Bool]-desXor a b = zipWith (/=) a b--desRotate :: [Bool] -> Int -> [Bool]-desRotate bits rot = drop rot' bits ++ take rot' bits- where- rot' = rot `mod` length bits--bitify :: Word64 -> Bits64-bitify w = map (\b -> w .&. (shiftL 1 b) /= 0) [63, 62 .. 0]--unbitify :: Bits64 -> Word64-unbitify bs = foldl (\i b -> if b then 1 + shiftL i 1 else shiftL i 1) 0 bs--initial_permutation :: Bits64 -> Bits64-initial_permutation mb = map ((!!) mb) i- where- i =- [ 57- , 49- , 41- , 33- , 25- , 17- , 9- , 1- , 59- , 51- , 43- , 35- , 27- , 19- , 11- , 3- , 61- , 53- , 45- , 37- , 29- , 21- , 13- , 5- , 63- , 55- , 47- , 39- , 31- , 23- , 15- , 7- , 56- , 48- , 40- , 32- , 24- , 16- , 8- , 0- , 58- , 50- , 42- , 34- , 26- , 18- , 10- , 2- , 60- , 52- , 44- , 36- , 28- , 20- , 12- , 4- , 62- , 54- , 46- , 38- , 30- , 22- , 14- , 6- ]--{--"\x39\x31\x29\x21\x19\x11\x09\x01\x3b\x33\x2b\x23\x1b\x13\-\\x0b\x03\x3d\x35\x2d\x25\x1d\x15\x0d\x05\x3f\x37\x2f\x27\-\\x1f\x17\x0f\x07\x38\x30\x28\x20\x18\x10\x08\x00\x3a\x32\-\\x2a\x22\x1a\x12\x0a\x02\x3c\x34\x2c\x24\x1c\x14\x0c\x04\-\\x3e\x36\x2e\x26\x1e\x16\x0e\x06"--}--key_transformation :: Bits64 -> Bits56-key_transformation kb = map ((!!) kb) i- where- i =- [ 56- , 48- , 40- , 32- , 24- , 16- , 8- , 0- , 57- , 49- , 41- , 33- , 25- , 17- , 9- , 1- , 58- , 50- , 42- , 34- , 26- , 18- , 10- , 2- , 59- , 51- , 43- , 35- , 62- , 54- , 46- , 38- , 30- , 22- , 14- , 6- , 61- , 53- , 45- , 37- , 29- , 21- , 13- , 5- , 60- , 52- , 44- , 36- , 28- , 20- , 12- , 4- , 27- , 19- , 11- , 3- ]--{--"\x38\x30\x28\x20\x18\x10\x08\x00\x39\x31\x29\x21\x19\x11\-\\x09\x01\x3a\x32\x2a\x22\x1a\x12\x0a\x02\x3b\x33\x2b\x23\-\\x3e\x36\x2e\x26\x1e\x16\x0e\x06\x3d\x35\x2d\x25\x1d\x15\-\\x0d\x05\x3c\x34\x2c\x24\x1c\x14\x0c\x04\x1b\x13\x0b\x03"--}--des_enc :: Block -> Key -> Block-des_enc = do_des [1, 2, 4, 6, 8, 10, 12, 14, 15, 17, 19, 21, 23, 25, 27, 28]--des_dec :: Block -> Key -> Block-des_dec = do_des [28, 27, 25, 23, 21, 19, 17, 15, 14, 12, 10, 8, 6, 4, 2, 1]--do_des :: [Rotation] -> Block -> Key -> Block-do_des rots (Block m) k = Block $ des_work rots (takeDrop 32 mb) kb- where- kb = key_transformation $ bitify k- mb = initial_permutation $ bitify m--des_work :: [Rotation] -> (Bits32, Bits32) -> Bits56 -> Word64-des_work [] (ml, mr) _ = unbitify $ final_perm $ (mr ++ ml)-des_work (r : rs) mb kb = des_work rs mb' kb- where- mb' = do_round r mb kb--do_round :: Rotation -> (Bits32, Bits32) -> Bits56 -> (Bits32, Bits32)-do_round r (ml, mr) kb = (mr, m')- where- kb' = get_key kb r- comp_kb = compression_permutation kb'- expa_mr = expansion_permutation mr- res = comp_kb `desXor` expa_mr- res' = drop 1 $ iterate (trans 6) ([], res)- trans n (_, b) = (take n b, drop n b)- res_s =- concat $- zipWith- (\f (x, _) -> f x)- [ s_box_1- , s_box_2- , s_box_3- , s_box_4- , s_box_5- , s_box_6- , s_box_7- , s_box_8- ]- res'- res_p = p_box res_s- m' = res_p `desXor` ml--get_key :: Bits56 -> Rotation -> Bits56-get_key kb r = kb'- where- (kl, kr) = takeDrop 28 kb- kb' = desRotate kl r ++ desRotate kr r--compression_permutation :: Bits56 -> Bits48-compression_permutation kb = map ((!!) kb) i- where- i =- [ 13- , 16- , 10- , 23- , 0- , 4- , 2- , 27- , 14- , 5- , 20- , 9- , 22- , 18- , 11- , 3- , 25- , 7- , 15- , 6- , 26- , 19- , 12- , 1- , 40- , 51- , 30- , 36- , 46- , 54- , 29- , 39- , 50- , 44- , 32- , 47- , 43- , 48- , 38- , 55- , 33- , 52- , 45- , 41- , 49- , 35- , 28- , 31- ]--expansion_permutation :: Bits32 -> Bits48-expansion_permutation mb = map ((!!) mb) i- where- i =- [ 31- , 0- , 1- , 2- , 3- , 4- , 3- , 4- , 5- , 6- , 7- , 8- , 7- , 8- , 9- , 10- , 11- , 12- , 11- , 12- , 13- , 14- , 15- , 16- , 15- , 16- , 17- , 18- , 19- , 20- , 19- , 20- , 21- , 22- , 23- , 24- , 23- , 24- , 25- , 26- , 27- , 28- , 27- , 28- , 29- , 30- , 31- , 0- ]--s_box :: [[Word8]] -> Bits6 -> Bits4-s_box s [a, b, c, d, e, f] = to_bool 4 $ (s !! row) !! col- where- row = sum $ zipWith numericise [a, f] [1, 0]- col = sum $ zipWith numericise [b, c, d, e] [3, 2, 1, 0]- numericise :: Bool -> Int -> Int- numericise = (\x y -> if x then 2 ^ y else 0)-- to_bool :: Int -> Word8 -> [Bool]- to_bool 0 _ = []- to_bool n i = ((i .&. 8) == 8) : to_bool (n - 1) (shiftL i 1)-s_box _ _ = error "DES: internal error bits6 more than 6 elements"--s_box_1 :: Bits6 -> Bits4-s_box_1 = s_box i- where- i =- [ [14, 4, 13, 1, 2, 15, 11, 8, 3, 10, 6, 12, 5, 9, 0, 7]- , [0, 15, 7, 4, 14, 2, 13, 1, 10, 6, 12, 11, 9, 5, 3, 8]- , [4, 1, 14, 8, 13, 6, 2, 11, 15, 12, 9, 7, 3, 10, 5, 0]- , [15, 12, 8, 2, 4, 9, 1, 7, 5, 11, 3, 14, 10, 0, 6, 13]- ]--s_box_2 :: Bits6 -> Bits4-s_box_2 = s_box i- where- i =- [ [15, 1, 8, 14, 6, 11, 3, 4, 9, 7, 2, 13, 12, 0, 5, 10]- , [3, 13, 4, 7, 15, 2, 8, 14, 12, 0, 1, 10, 6, 9, 11, 5]- , [0, 14, 7, 11, 10, 4, 13, 1, 5, 8, 12, 6, 9, 3, 2, 15]- , [13, 8, 10, 1, 3, 15, 4, 2, 11, 6, 7, 12, 0, 5, 14, 9]- ]--s_box_3 :: Bits6 -> Bits4-s_box_3 = s_box i- where- i =- [ [10, 0, 9, 14, 6, 3, 15, 5, 1, 13, 12, 7, 11, 4, 2, 8]- , [13, 7, 0, 9, 3, 4, 6, 10, 2, 8, 5, 14, 12, 11, 15, 1]- , [13, 6, 4, 9, 8, 15, 3, 0, 11, 1, 2, 12, 5, 10, 14, 7]- , [1, 10, 13, 0, 6, 9, 8, 7, 4, 15, 14, 3, 11, 5, 2, 12]- ]--s_box_4 :: Bits6 -> Bits4-s_box_4 = s_box i- where- i =- [ [7, 13, 14, 3, 0, 6, 9, 10, 1, 2, 8, 5, 11, 12, 4, 15]- , [13, 8, 11, 5, 6, 15, 0, 3, 4, 7, 2, 12, 1, 10, 14, 9]- , [10, 6, 9, 0, 12, 11, 7, 13, 15, 1, 3, 14, 5, 2, 8, 4]- , [3, 15, 0, 6, 10, 1, 13, 8, 9, 4, 5, 11, 12, 7, 2, 14]- ]--s_box_5 :: Bits6 -> Bits4-s_box_5 = s_box i- where- i =- [ [2, 12, 4, 1, 7, 10, 11, 6, 8, 5, 3, 15, 13, 0, 14, 9]- , [14, 11, 2, 12, 4, 7, 13, 1, 5, 0, 15, 10, 3, 9, 8, 6]- , [4, 2, 1, 11, 10, 13, 7, 8, 15, 9, 12, 5, 6, 3, 0, 14]- , [11, 8, 12, 7, 1, 14, 2, 13, 6, 15, 0, 9, 10, 4, 5, 3]- ]+import Foreign.C.Types (CInt (..))+import Foreign.Ptr (Ptr, plusPtr) -s_box_6 :: Bits6 -> Bits4-s_box_6 = s_box i- where- i =- [ [12, 1, 10, 15, 9, 2, 6, 8, 0, 13, 3, 4, 14, 7, 5, 11]- , [10, 15, 4, 2, 7, 12, 9, 5, 6, 1, 13, 14, 0, 11, 3, 8]- , [9, 14, 15, 5, 2, 8, 12, 3, 7, 0, 4, 10, 1, 13, 11, 6]- , [4, 3, 2, 12, 9, 5, 15, 10, 11, 14, 1, 7, 6, 0, 8, 13]- ]+-- | Which way a stage runs.+data Direction = Encrypt | Decrypt+ deriving (Show, Eq) -s_box_7 :: Bits6 -> Bits4-s_box_7 = s_box i- where- i =- [ [4, 11, 2, 14, 15, 0, 8, 13, 3, 12, 9, 7, 5, 10, 6, 1]- , [13, 0, 11, 7, 4, 9, 1, 10, 14, 3, 5, 12, 2, 15, 8, 6]- , [1, 4, 11, 13, 12, 3, 7, 14, 10, 15, 6, 8, 0, 5, 9, 2]- , [6, 11, 13, 8, 1, 4, 10, 7, 9, 5, 0, 15, 14, 2, 3, 12]- ]+-- | The round keys of one or more stages, in the order they are applied.+newtype Schedule = Schedule Bytes+ deriving (Eq) -s_box_8 :: Bits6 -> Bits4-s_box_8 = s_box i- where- i =- [ [13, 2, 8, 4, 6, 15, 11, 1, 10, 9, 3, 14, 5, 0, 12, 7]- , [1, 15, 13, 8, 10, 3, 7, 4, 12, 5, 6, 11, 0, 14, 9, 2]- , [7, 11, 4, 1, 9, 12, 14, 2, 0, 6, 10, 13, 15, 3, 5, 8]- , [2, 1, 14, 7, 4, 10, 8, 13, 15, 12, 9, 0, 3, 5, 6, 11]- ]+-- | Bytes per stage: sixteen rounds of eight six-bit values.+stageSize :: Int+stageSize = 16 * 8 -p_box :: Bits32 -> Bits32-p_box kb = map ((!!) kb) i- where- i =- [ 15- , 6- , 19- , 20- , 28- , 11- , 27- , 16- , 0- , 14- , 22- , 25- , 4- , 17- , 30- , 9- , 1- , 7- , 23- , 13- , 31- , 26- , 2- , 8- , 18- , 12- , 29- , 5- , 21- , 10- , 3- , 24- ]+-- | The block size DES works in.+blockBytes :: Int+blockBytes = 8 -final_perm :: Bits64 -> Bits64-final_perm kb = map ((!!) kb) i+-- | Build the schedule for a sequence of stages, each an eight byte key and+-- the direction that stage runs in. Shorter keys are rejected by the callers,+-- which know their own size; the bytes past the eighth are not read.+schedule :: ByteArrayAccess key => [(Direction, key)] -> Schedule+schedule stages =+ Schedule $ B.allocAndFreeze (stageSize * length stages) $ \dst ->+ mapM_ (uncurry (one dst)) (zip [0 ..] stages) where- i =- [ 39- , 7- , 47- , 15- , 55- , 23- , 63- , 31- , 38- , 6- , 46- , 14- , 54- , 22- , 62- , 30- , 37- , 5- , 45- , 13- , 53- , 21- , 61- , 29- , 36- , 4- , 44- , 12- , 52- , 20- , 60- , 28- , 35- , 3- , 43- , 11- , 51- , 19- , 59- , 27- , 34- , 2- , 42- , 10- , 50- , 18- , 58- , 26- , 33- , 1- , 41- , 9- , 49- , 17- , 57- , 25- , 32- , 0- , 40- , 8- , 48- , 16- , 56- , 24- ]+ one dst i (dir, key) =+ B.withByteArray key $ \k ->+ c_des_init (dst `plusPtr` (i * stageSize)) k (reverseFlag dir)+ reverseFlag Encrypt = 0+ reverseFlag Decrypt = 1 -takeDrop :: Int -> [a] -> ([a], [a])-takeDrop _ [] = ([], [])-takeDrop 0 xs = ([], xs)-takeDrop n (x : xs) = (x : ys, zs)+-- | Apply every stage of the schedule, in order, to each block of the input.+ecb :: ByteArray ba => Schedule -> ba -> ba+ecb (Schedule sched) input+ | len `mod` blockBytes /= 0 =+ error $+ "Crypto.Cipher.DES: input length must be a multiple of block size (8). Its length is: "+ ++ show len+ | otherwise = unsafeDoIO $+ B.alloc len $ \out ->+ B.withByteArray sched $ \ks ->+ B.withByteArray input $ \inp ->+ c_des_ecb+ out+ ks+ (fromIntegral (B.length sched `div` stageSize))+ inp+ (fromIntegral (len `div` blockBytes)) where- (ys, zs) = takeDrop (n - 1) xs+ len = B.length input --- | Basic DES encryption which takes a key and a block of plaintext--- and returns the encrypted block of ciphertext according to the standard.-encrypt :: Word64 -> Block -> Block-encrypt = flip des_enc+foreign import ccall unsafe "crypton_des.h crypton_des_init"+ c_des_init :: Ptr Word8 -> Ptr Word8 -> CInt -> IO () --- | Basic DES decryption which takes a key and a block of ciphertext and--- returns the decrypted block of plaintext according to the standard.-decrypt :: Word64 -> Block -> Block-decrypt = flip des_dec+foreign import ccall unsafe "crypton_des.h crypton_des_ecb"+ c_des_ecb :: Ptr Word8 -> Ptr Word8 -> Word32 -> Ptr Word8 -> Word32 -> IO ()
@@ -98,7 +98,14 @@ B.allocRet len $ \outptr -> B.withByteArray clearText $ \clearPtr -> do st <- B.copy prevSt $ \stPtr ->- c_rc4_combine (castPtr stPtr) clearPtr (fromIntegral len) outptr+ -- in pieces the C's uint32_t length can hold; the state it+ -- keeps means it can simply be called again+ B.inCLengths len $ \off n ->+ c_rc4_combine+ (castPtr stPtr)+ (clearPtr `plusPtr` off)+ (fromIntegral n)+ (outptr `plusPtr` off) return $! State st where -- return $! (State st, B.PS outfptr 0 len)
@@ -70,10 +70,19 @@ | B.null src = (B.empty, prevSt) | otherwise = unsafeDoIO $ do (out, st) <- B.copyRet prevStMem $ \ctx ->- B.alloc (B.length src) $ \dstPtr ->- B.withByteArray src $ \srcPtr -> do- ccrypton_salsa_combine dstPtr ctx srcPtr (fromIntegral $ B.length src)+ B.alloc n $ \dstPtr ->+ B.withByteArray src $ \srcPtr ->+ -- in pieces the C's uint32_t length can hold; it carries+ -- the state in ctx, so it can simply be called again+ B.inCLengths n $ \off len ->+ ccrypton_salsa_combine+ (dstPtr `plusPtr` off)+ ctx+ (srcPtr `plusPtr` off)+ (fromIntegral len) return (out, State st)+ where+ n = B.length src -- | Generate a number of bytes from the Salsa output directly generate@@ -88,7 +97,11 @@ | otherwise = unsafeDoIO $ do (out, st) <- B.copyRet prevStMem $ \ctx -> B.alloc len $ \dstPtr ->- ccrypton_salsa_generate dstPtr ctx (fromIntegral len)+ B.inCLengths len $ \off n ->+ ccrypton_salsa_generate+ (dstPtr `plusPtr` off)+ ctx+ (fromIntegral n) return (out, State st) foreign import ccall "crypton_salsa_init"
@@ -13,82 +13,104 @@ import Crypto.Cipher.DES.Primitive import Crypto.Cipher.Types import Crypto.Error-import Crypto.Internal.ByteArray (ByteArrayAccess)+import Crypto.Internal.ByteArray (ByteArrayAccess, ScrubbedBytes) import qualified Crypto.Internal.ByteArray as B-import Data.Memory.Endian-import Data.Word -- | 3DES with 3 different keys used all in the same direction-data DES_EEE3 = DES_EEE3 Word64 Word64 Word64+data DES_EEE3 = DES_EEE3 Schedule Schedule deriving (Eq) -- | 3DES with 3 different keys used in alternative direction-data DES_EDE3 = DES_EDE3 Word64 Word64 Word64+data DES_EDE3 = DES_EDE3 Schedule Schedule deriving (Eq) -- | 3DES where the first and third keys are equal, used in the same direction-data DES_EEE2 = DES_EEE2 Word64 Word64 -- key1 and key3 are equal+data DES_EEE2 = DES_EEE2 Schedule Schedule deriving (Eq) -- | 3DES where the first and third keys are equal, used in alternative direction-data DES_EDE2 = DES_EDE2 Word64 Word64 -- key1 and key3 are equal+data DES_EDE2 = DES_EDE2 Schedule Schedule deriving (Eq) instance Cipher DES_EEE3 where cipherName _ = "3DES_EEE" cipherKeySize _ = KeySizeFixed 24- cipherInit k = init3DES DES_EEE3 k+ cipherInit k = init3DES DES_EEE3 Encrypt k instance Cipher DES_EDE3 where cipherName _ = "3DES_EDE" cipherKeySize _ = KeySizeFixed 24- cipherInit k = init3DES DES_EDE3 k+ cipherInit k = init3DES DES_EDE3 Decrypt k instance Cipher DES_EDE2 where cipherName _ = "2DES_EDE" cipherKeySize _ = KeySizeFixed 16- cipherInit k = init2DES DES_EDE2 k+ cipherInit k = init2DES DES_EDE2 Decrypt k instance Cipher DES_EEE2 where cipherName _ = "2DES_EEE" cipherKeySize _ = KeySizeFixed 16- cipherInit k = init2DES DES_EEE2 k+ cipherInit k = init2DES DES_EEE2 Encrypt k instance BlockCipher DES_EEE3 where blockSize _ = 8- ecbEncrypt (DES_EEE3 k1 k2 k3) = B.mapAsWord64 (unBlock . (encrypt k3 . encrypt k2 . encrypt k1) . Block)- ecbDecrypt (DES_EEE3 k1 k2 k3) = B.mapAsWord64 (unBlock . (decrypt k1 . decrypt k2 . decrypt k3) . Block)+ ecbEncrypt (DES_EEE3 enc _) = ecb enc+ ecbDecrypt (DES_EEE3 _ dec) = ecb dec instance BlockCipher DES_EDE3 where blockSize _ = 8- ecbEncrypt (DES_EDE3 k1 k2 k3) = B.mapAsWord64 (unBlock . (encrypt k3 . decrypt k2 . encrypt k1) . Block)- ecbDecrypt (DES_EDE3 k1 k2 k3) = B.mapAsWord64 (unBlock . (decrypt k1 . encrypt k2 . decrypt k3) . Block)+ ecbEncrypt (DES_EDE3 enc _) = ecb enc+ ecbDecrypt (DES_EDE3 _ dec) = ecb dec instance BlockCipher DES_EEE2 where blockSize _ = 8- ecbEncrypt (DES_EEE2 k1 k2) = B.mapAsWord64 (unBlock . (encrypt k1 . encrypt k2 . encrypt k1) . Block)- ecbDecrypt (DES_EEE2 k1 k2) = B.mapAsWord64 (unBlock . (decrypt k1 . decrypt k2 . decrypt k1) . Block)+ ecbEncrypt (DES_EEE2 enc _) = ecb enc+ ecbDecrypt (DES_EEE2 _ dec) = ecb dec instance BlockCipher DES_EDE2 where blockSize _ = 8- ecbEncrypt (DES_EDE2 k1 k2) = B.mapAsWord64 (unBlock . (encrypt k1 . decrypt k2 . encrypt k1) . Block)- ecbDecrypt (DES_EDE2 k1 k2) = B.mapAsWord64 (unBlock . (decrypt k1 . encrypt k2 . decrypt k1) . Block)+ ecbEncrypt (DES_EDE2 enc _) = ecb enc+ ecbDecrypt (DES_EDE2 _ dec) = ecb dec +-- | The schedules of a three stage cipher, for both directions.+--+-- The outer stages encrypt and the middle one goes whichever way the+-- construction says; decrypting is the same three stages in the opposite+-- order, each the other way round.+stages+ :: ByteArrayAccess key+ => Direction+ -- ^ the direction of the middle stage when encrypting+ -> (key, key, key)+ -> (Schedule, Schedule)+stages mid (k1, k2, k3) =+ ( schedule [(Encrypt, k1), (mid, k2), (Encrypt, k3)]+ , schedule [(Decrypt, k3), (opposite mid, k2), (Decrypt, k1)]+ )+ where+ opposite Encrypt = Decrypt+ opposite Decrypt = Encrypt+ init3DES :: ByteArrayAccess key- => (Word64 -> Word64 -> Word64 -> a) -> key -> CryptoFailable a-init3DES constr k- | len == 24 = CryptoPassed $ constr k1 k2 k3+ => (Schedule -> Schedule -> a) -> Direction -> key -> CryptoFailable a+init3DES constr mid k+ | B.length k == 24 =+ CryptoPassed $ uncurry constr $ stages mid (part 0, part 8, part 16) | otherwise = CryptoFailed CryptoError_KeySizeInvalid where- len = B.length k- (k1, k2, k3) = (fromBE $ B.toW64BE k 0, fromBE $ B.toW64BE k 8, fromBE $ B.toW64BE k 16)+ part = keyPart k init2DES- :: ByteArrayAccess key => (Word64 -> Word64 -> a) -> key -> CryptoFailable a-init2DES constr k- | len == 16 = CryptoPassed $ constr k1 k2+ :: ByteArrayAccess key+ => (Schedule -> Schedule -> a) -> Direction -> key -> CryptoFailable a+init2DES constr mid k+ | B.length k == 16 =+ CryptoPassed $ uncurry constr $ stages mid (part 0, part 8, part 0) | otherwise = CryptoFailed CryptoError_KeySizeInvalid where- len = B.length k- (k1, k2) = (fromBE $ B.toW64BE k 0, fromBE $ B.toW64BE k 8)+ part = keyPart k++-- | The eight bytes of a key that start at the given offset.+keyPart :: ByteArrayAccess key => key -> Int -> ScrubbedBytes+keyPart k i = B.take 8 $ B.drop i (B.convert k :: ScrubbedBytes)
@@ -13,10 +13,10 @@ import Crypto.Internal.ByteArray (ByteArray) import qualified Crypto.Internal.ByteArray as B import Crypto.Internal.WordArray+import Crypto.Internal.Words (Word128 (..)) import Data.Bits-import Data.List (foldl')+import qualified Data.List as L import Data.Word-import Prelude hiding (foldl') -- Based on the Golang referance implementation -- https://github.com/golang/crypto/blob/master/twofish/twofish.go@@ -65,13 +65,34 @@ generatedK = array32 40 $ genK keyPackage generatedS = genSboxes keyPackage $ sWords key -mapBlocks :: ByteArray ba => (ba -> ba) -> ba -> ba+-- | Run a block operation over every block of the input.+--+-- 'B.mapAsWord128' walks the input and the output once each, where taking a+-- block off the front and appending the result copied the whole of both, once+-- per block.+mapBlocks :: ByteArray ba => (Word128 -> Word128) -> ba -> ba mapBlocks operation input- | B.null rest = blockOutput- | otherwise = blockOutput `B.append` mapBlocks operation rest+ | B.length input `mod` blockSize /= 0 =+ error $+ "Crypto.Cipher.Twofish: input length must be a multiple of block size (16). Its length is: "+ ++ show (B.length input)+ | otherwise = B.mapAsWord128 operation input++-- | The four little-endian words of a block, from the two big-endian words+-- t'Word128' is read as.+load32ls :: Word128 -> (Word32, Word32, Word32, Word32)+load32ls (Word128 hi lo) =+ ( byteSwap32 (fromIntegral (hi `shiftR` 32))+ , byteSwap32 (fromIntegral hi)+ , byteSwap32 (fromIntegral (lo `shiftR` 32))+ , byteSwap32 (fromIntegral lo)+ )++store32ls :: (Word32, Word32, Word32, Word32) -> Word128+store32ls (a, b, c, d) = Word128 (pair a b) (pair c d) where- (block, rest) = B.splitAt blockSize input- blockOutput = operation block+ pair x y =+ (fromIntegral (byteSwap32 x) `shiftL` 32) .|. fromIntegral (byteSwap32 y) -- | Encrypts the given ByteString using the given Key encrypt@@ -83,7 +104,7 @@ -> ba encrypt cipher = mapBlocks (encryptBlock cipher) -encryptBlock :: ByteArray ba => Twofish -> ba -> ba+encryptBlock :: Twofish -> Word128 -> Word128 encryptBlock Twofish{s = (s1, s2, s3, s4), k = ks} message = store32ls ts where (a, b, c, d) = load32ls message@@ -91,7 +112,7 @@ b' = b `xor` arrayRead32 ks 1 c' = c `xor` arrayRead32 ks 2 d' = d `xor` arrayRead32 ks 3- (!a'', !b'', !c'', !d'') = foldl' shuffle (a', b', c', d') [0 .. 7]+ (!a'', !b'', !c'', !d'') = L.foldl' shuffle (a', b', c', d') [0 .. 7] ts = ( c'' `xor` arrayRead32 ks 4 , d'' `xor` arrayRead32 ks 5@@ -150,7 +171,7 @@ decrypt cipher = mapBlocks (decryptBlock cipher) {- decryption for 128 bits blocks -}-decryptBlock :: ByteArray ba => Twofish -> ba -> ba+decryptBlock :: Twofish -> Word128 -> Word128 decryptBlock Twofish{s = (s1, s2, s3, s4), k = ks} message = store32ls ixs where (a, b, c, d) = load32ls message@@ -158,7 +179,7 @@ b' = d `xor` arrayRead32 ks 7 c' = a `xor` arrayRead32 ks 4 d' = b `xor` arrayRead32 ks 5- (!a'', !b'', !c'', !d'') = foldl' unshuffle (a', b', c', d') [8, 7 .. 1]+ (!a'', !b'', !c'', !d'') = L.foldl' unshuffle (a', b', c', d') [8, 7 .. 1] ixs = ( a'' `xor` arrayRead32 ks 0 , b'' `xor` arrayRead32 ks 1@@ -252,26 +273,6 @@ , [0xA4, 0x55, 0x87, 0x5A, 0x58, 0xDB, 0x9E, 0x03] ] -load32ls :: ByteArray ba => ba -> (Word32, Word32, Word32, Word32)-load32ls message = (intify q1, intify q2, intify q3, intify q4)- where- (half1, half2) = B.splitAt 8 message- (q1, q2) = B.splitAt 4 half1- (q3, q4) = B.splitAt 4 half2-- intify :: ByteArray ba => ba -> Word32- intify bytes =- foldl'- (\int (!word, !ind) -> int .|. shiftL (fromIntegral word) (ind * 8))- 0- (zip (B.unpack bytes) [0 ..])--store32ls :: ByteArray ba => (Word32, Word32, Word32, Word32) -> ba-store32ls (a, b, c, d) = B.pack $ concatMap splitWordl [a, b, c, d]- where- splitWordl :: Word32 -> [Word8]- splitWordl w = fmap (\ind -> fromIntegral $ shiftR w (8 * ind)) [0 .. 3]- -- Create S words sWords :: ByteArray ba => ba -> [Word8] sWords key = sWord@@ -282,7 +283,7 @@ ( \wordIndex -> map ( \rsRow ->- foldl'+ L.foldl' ( \acc (!rsVal, !colIndex) -> acc `xor` gfMult rsPolynomial (B.index key $ 8 * wordIndex + colIndex) rsVal )@@ -441,7 +442,7 @@ b' = rotateL b 8 h :: ByteArray ba => [Word8] -> KeyPackage ba -> Int -> Word32-h input keyPackage offset = foldl' xorMdsColMult 0 $ zip [y0f, y1f, y2f, y3f] $ enumFrom Zero+h input keyPackage offset = L.foldl' xorMdsColMult 0 $ zip [y0f, y1f, y2f, y3f] $ enumFrom Zero where key = rawKeyBytes keyPackage [y0, y1, y2, y3] = take 4 input
@@ -105,3 +105,37 @@ aead = aeadAppendHeader aeadIni header (output, aeadFinal) = aeadDecrypt aead input tag = aeadFinalize aeadFinal (B.length authTag)++-- | Simple AEAD decryption with the tag length given by the caller.+--+-- 'aeadSimpleDecrypt' authenticates as many octets as the tag it is handed is+-- long. That is the caller's choice only for as long as the tag is: one read+-- off the wire is the peer's, and an attacker who truncates it picks how much+-- of it gets verified, down to 'minimumTagLength'.+--+-- Here the length is a separate argument and a tag that is not exactly that+-- long is refused before anything is compared, so the peer cannot weaken the+-- check. Prefer this wherever the tag is attacker reachable.+tryAeadSimpleDecrypt+ :: (ByteArrayAccess aad, ByteArray ba)+ => AEAD a+ -- ^ An AEAD Context+ -> aad+ -- ^ Associated\/additional data+ -> ba+ -- ^ Ciphertext+ -> Int+ -- ^ The tag length to authenticate, which the tag must match+ -> AuthTag+ -- ^ The authentication tag+ -> Maybe ba+ -- ^ Plaintext+tryAeadSimpleDecrypt aeadIni header input taglen authTag+ | taglen < minimumTagLength = Nothing+ | B.length authTag /= taglen = Nothing+ | tag == authTag = Just output+ | otherwise = Nothing+ where+ aead = aeadAppendHeader aeadIni header+ (output, aeadFinal) = aeadDecrypt aead input+ tag = aeadFinalize aeadFinal taglen
@@ -43,7 +43,6 @@ import Crypto.Cipher.Types.AEAD import Crypto.Cipher.Types.Base import Crypto.Cipher.Types.GF-import Crypto.Cipher.Types.Utils import Crypto.Error import Data.Word @@ -54,7 +53,10 @@ withByteArray, ) import qualified Crypto.Internal.ByteArray as B+import Data.ByteString (ByteString)+import qualified Data.ByteString as S +import Foreign.Marshal.Utils (copyBytes) import Foreign.Ptr import Foreign.Storable @@ -211,49 +213,131 @@ cbcEncryptGeneric :: (ByteArray ba, BlockCipher cipher) => cipher -> IV cipher -> ba -> ba-cbcEncryptGeneric cipher ivini input = mconcat $ doEnc ivini $ chunk (blockSize cipher) input+cbcEncryptGeneric cipher ivini input =+ B.concat $ doEnc ivini $ slices (blockSize cipher) input where+ -- the blocks of the message as shared slices rather than copies: each+ -- block already costs an exclusive or and a call into the cipher, both of+ -- which allocate, and the chain makes it one block at a time doEnc _ [] = [] doEnc iv (i : is) =- let o = ecbEncrypt cipher $ B.xor iv i+ let o = ecbEncrypt cipher (B.bxor iv i) `asTypeOf` input in o : doEnc (IV o) is +-- | How many blocks to hand the cipher at a time in the modes whose blocks do+-- not depend on one another. Enough that the cost of a call disappears, few+-- enough that what it copies stays in cache.+blocksPerCall :: Int+blocksPerCall = 2048++-- | The input in slices of that many blocks. A ByteString shares where+-- 'B.splitAt' copies the rest of the message, once per slice.+slices :: ByteArray ba => Int -> ba -> [ByteString]+slices bytes input = go (B.convert input)+ where+ go bs+ | S.null bs = []+ | otherwise = let (hd, tl) = S.splitAt bytes bs in hd : go tl++-- | The previous ciphertext block of every block in a slice: the incoming IV,+-- and then the slice itself one block short.+shiftedBy :: BlockCipher cipher => Int -> IV cipher -> ByteString -> ByteString+shiftedBy bsz iv c = S.append (B.convert iv) (S.take (S.length c - bsz) c)++-- | The last whole block of a slice, which is where the next one carries on+-- from.+lastBlockOf :: Int -> ByteString -> IV cipher+lastBlockOf bsz c = IV (B.convert (S.drop (S.length c - bsz) c) :: Bytes)++-- | Decryption does not chain: @P_i@ is @D(C_i)@ exclusive-ored with+-- @C_(i-1)@, so a whole slice is decrypted in one call and exclusive-ored with+-- the ciphertext moved along by a block. cbcDecryptGeneric :: (ByteArray ba, BlockCipher cipher) => cipher -> IV cipher -> ba -> ba-cbcDecryptGeneric cipher ivini input = mconcat $ doDec ivini $ chunk (blockSize cipher) input+cbcDecryptGeneric cipher ivini input =+ B.concat $ doDec ivini $ slices (blocksPerCall * bsz) input where+ bsz = blockSize cipher+ conv x = B.convert x `asTypeOf` input+ xorB a b = B.bxor a b `asTypeOf` input doDec _ [] = []- doDec iv (i : is) =- let o = B.xor iv $ ecbDecrypt cipher i- in o : doDec (IV i) is+ doDec iv (c : cs) =+ xorB (ecbDecrypt cipher (conv c)) (conv (shiftedBy bsz iv c))+ : doDec (lastBlockOf bsz c) cs cfbEncryptGeneric :: (ByteArray ba, BlockCipher cipher) => cipher -> IV cipher -> ba -> ba-cfbEncryptGeneric cipher ivini input = mconcat $ doEnc ivini $ chunk (blockSize cipher) input+cfbEncryptGeneric cipher ivini input =+ B.concat $ doEnc ivini $ slices (blockSize cipher) input where doEnc _ [] = [] doEnc (IV iv) (i : is) =- let o = B.xor i $ ecbEncrypt cipher iv+ let o = B.bxor i (ecbEncrypt cipher iv) `asTypeOf` input in o : doEnc (IV o) is +-- | Nor does this one: @P_i@ is @C_i@ exclusive-ored with @E(C_(i-1))@, and+-- what gets encrypted is again the ciphertext moved along by a block. cfbDecryptGeneric :: (ByteArray ba, BlockCipher cipher) => cipher -> IV cipher -> ba -> ba-cfbDecryptGeneric cipher ivini input = mconcat $ doDec ivini $ chunk (blockSize cipher) input+cfbDecryptGeneric cipher ivini input =+ B.concat $ doDec ivini $ slices (blocksPerCall * bsz) input where+ bsz = blockSize cipher+ conv x = B.convert x `asTypeOf` input+ xorB a b = B.bxor a b `asTypeOf` input doDec _ [] = []- doDec (IV iv) (i : is) =- let o = B.xor i $ ecbEncrypt cipher iv- in o : doDec (IV i) is+ doDec iv (c : cs) =+ xorB (conv c) (ecbEncrypt cipher (conv (shiftedBy bsz iv c)))+ : doDec (lastBlockOf bsz c) cs +-- | The counters do not depend on the message at all, so a slice of them is+-- built and encrypted in one call. ctrCombineGeneric :: (ByteArray ba, BlockCipher cipher) => cipher -> IV cipher -> ba -> ba-ctrCombineGeneric cipher ivini input = mconcat $ doCnt ivini $ chunk (blockSize cipher) input+ctrCombineGeneric cipher ivini input =+ B.concat $ doCnt ivini $ slices (blocksPerCall * bsz) input where+ bsz = blockSize cipher+ conv x = B.convert x `asTypeOf` input+ xorB a b = B.bxor a b `asTypeOf` input doCnt _ [] = []- doCnt iv@(IV ivd) (i : is) =- let ivEnc = ecbEncrypt cipher ivd- in B.xor i ivEnc : doCnt (ivAdd iv 1) is+ doCnt iv (m : ms) =+ xorB (conv m) (ecbEncrypt cipher (counters iv n `asTypeOf` input))+ : doCnt (ivAdd iv n) ms+ where+ n = (S.length m + bsz - 1) `div` bsz +-- | The counters for a slice: the given one, then each next as the one before+-- it plus one.+--+-- One buffer, filled in place. Asking 'ivAdd' for each of them separately+-- allocated a block per block and walked the whole width of the counter from+-- the original every time, which cost more than the cipher did: counter mode+-- ran at a quarter of what the same cipher managed in ECB, and at an eighth+-- for Blowfish.+counters :: (ByteArray ba, BlockCipher cipher) => IV cipher -> Int -> ba+counters iv n = B.allocAndFreeze (n * bsz) fill+ where+ bsz = B.length iv++ fill p = do+ B.copyByteArrayToPtr iv p+ let go k prev+ | k >= n = return ()+ | otherwise = do+ let this = prev `plusPtr` bsz+ copyBytes this prev bsz+ increment this (bsz - 1)+ go (k + 1) this+ go 1 p++ increment p ofs+ | ofs < 0 = return ()+ | otherwise = do+ v <- peek (p `plusPtr` ofs) :: IO Word8+ poke (p `plusPtr` ofs) (v + 1)+ if v == 0xff then increment p (ofs - 1) else return ()+ xtsEncryptGeneric :: (ByteArray ba, BlockCipher128 cipher) => XTS ba cipher xtsEncryptGeneric = xtsGeneric ecbEncrypt @@ -269,13 +353,13 @@ -> ba -> ba xtsGeneric f (cipher, tweakCipher) (IV iv) sPoint input =- mconcat $ doXts iniTweak $ chunk (blockSize cipher) input+ B.concat $ doXts iniTweak $ slices (blockSize cipher) input where encTweak = ecbEncrypt tweakCipher iv iniTweak = iterate xtsGFMul encTweak !! fromIntegral sPoint doXts _ [] = [] doXts tweak (i : is) =- let o = B.xor (f cipher $ B.xor i tweak) tweak+ let o = B.bxor (f cipher (B.bxor i tweak)) tweak `asTypeOf` input in o : doXts (xtsGFMul tweak) is {-
@@ -10,13 +10,19 @@ import Crypto.Internal.ByteArray (ByteArray) import qualified Crypto.Internal.ByteArray as B+import Data.ByteString (ByteString)+import qualified Data.ByteString as S -- | Chunk some input byte array into @sz byte list of byte array.+--+-- The input is held as a 'ByteString' while it is cut up, because+-- 'Crypto.Internal.ByteArray.splitAt' copies both halves whatever the type+-- underneath: cutting a block off the front that way copies the rest of the+-- message, once per block, and so the message about n/2 times. A ByteString+-- shares instead, and only the blocks themselves are copied out. chunk :: ByteArray b => Int -> b -> [b]-chunk sz bs = split bs+chunk sz bs = map B.convert (split (B.convert bs :: ByteString)) where split b- | B.length b <= sz = [b]- | otherwise =- let (b1, b2) = B.splitAt sz b- in b1 : split b2+ | S.length b <= sz = [b]+ | otherwise = let (b1, b2) = S.splitAt sz b in b1 : split b2
@@ -15,10 +15,10 @@ MiyaguchiPreneel, ) where -import Data.List (foldl')-import Prelude hiding (foldl')+import qualified Data.List as L import Crypto.Cipher.Types+import Crypto.Cipher.Types.Utils (chunk) import Crypto.Data.Padding (Format (ZERO), pad) import Crypto.Error (throwCryptoError) import Crypto.Internal.ByteArray (ByteArray, ByteArrayAccess, Bytes)@@ -40,14 +40,14 @@ -> MiyaguchiPreneel cipher -- ^ output tag compute' g =- MP . foldl' (step $ g) (B.replicate bsz 0) . chunks . pad (ZERO bsz) . B.convert+ MP . L.foldl' (step $ g) (B.replicate bsz 0) . chunks . pad (ZERO bsz) . B.convert where bsz = blockSize (g B.empty {- dummy to get block size -})+ -- 'chunk' slices rather than splitting the message, which copied whatever+ -- was left of it once per block chunks msg | B.null msg = []- | otherwise = (hd :: Bytes) : chunks tl- where- (hd, tl) = B.splitAt bsz msg+ | otherwise = chunk bsz (msg :: Bytes) -- | Compute Miyaguchi-Preneel one way compress using the inferred block cipher. -- Only safe when KEY-SIZE equals to BLOCK-SIZE.@@ -74,4 +74,4 @@ k = g iv bxor :: ByteArray ba => ba -> ba -> ba-bxor = B.xor+bxor = B.bxor
@@ -15,10 +15,13 @@ -- destroy a key stored on disk. module Crypto.Data.AFIS ( split,+ trySplit, merge,+ tryMerge, ) where import Control.Monad (foldM, forM_)+import Crypto.Error import Crypto.Hash import Crypto.Internal.Compat import Crypto.Random.Types@@ -49,6 +52,10 @@ -- -- where acc is : -- acc(n+1) = hash (n ++ rand(n)) ^ acc(n)+--+-- The data has to be at least one byte long and the number of times to diffuse+-- it at least two; anything else raises 'CryptoError_ParameterInvalid', which+-- 'trySplit' reports as 'CryptoFailed' instead. split :: (ByteArray ba, HashAlgorithm hash, DRG rng) => hash@@ -61,10 +68,31 @@ -- ^ original data to diffuse. -> (ba, rng) -- ^ The diffused data-{-# NOINLINE split #-}-split hashAlg rng expandTimes src- | expandTimes <= 1 = error "invalid expandTimes value"- | otherwise = unsafeDoIO $ do+split hashAlg rng expandTimes src =+ throwCryptoError (trySplit hashAlg rng expandTimes src)++-- | Split data to diffused data, reporting parameters the splitter cannot work+-- with rather than raising.+--+-- See 'split'.+trySplit+ :: (ByteArray ba, HashAlgorithm hash, DRG rng)+ => hash+ -- ^ Hash algorithm to use as diffuser+ -> rng+ -- ^ Random generator to use+ -> Int+ -- ^ Number of times to diffuse the data.+ -> ba+ -- ^ original data to diffuse.+ -> CryptoFailable (ba, rng)+ -- ^ The diffused data+{-# NOINLINE trySplit #-}+trySplit hashAlg rng expandTimes src+ | expandTimes < 2 = CryptoFailed CryptoError_ParameterInvalid+ -- an empty secret splits into nothing at all, which merge cannot undo+ | blockSize == 0 = CryptoFailed CryptoError_ParameterInvalid+ | otherwise = CryptoPassed $ unsafeDoIO $ do (rng', bs) <- B.allocRet diffusedLen runOp return (bs, rng') where@@ -87,6 +115,11 @@ return g' -- | Merge previously diffused data back to the original data.+--+-- The diffused data has to be a non-empty multiple of the number of times it+-- was diffused, and that number at least two -- the same values 'split'+-- accepts. Anything else raises 'CryptoError_ParameterInvalid', which+-- 'tryMerge' reports as 'CryptoFailed' instead. merge :: (ByteArray ba, HashAlgorithm hash) => hash@@ -97,11 +130,31 @@ -- ^ Diffused data -> ba -- ^ Original data-{-# NOINLINE merge #-}-merge hashAlg expandTimes bs- | r /= 0 = error "diffused data not a multiple of expandTimes"- | originalSize <= 0 = error "diffused data null"- | otherwise = B.allocAndFreeze originalSize $ \dstPtr ->+merge hashAlg expandTimes bs =+ throwCryptoError (tryMerge hashAlg expandTimes bs)++-- | Merge previously diffused data back to the original data, reporting+-- parameters the merger cannot work with rather than raising.+--+-- See 'merge'.+tryMerge+ :: (ByteArray ba, HashAlgorithm hash)+ => hash+ -- ^ Hash algorithm used as diffuser+ -> Int+ -- ^ Number of times to un-diffuse the data+ -> ba+ -- ^ Diffused data+ -> CryptoFailable ba+ -- ^ Original data+{-# NOINLINE tryMerge #-}+tryMerge hashAlg expandTimes bs+ -- guards the quotRem below, which for zero would divide by zero; a count+ -- of one would return the diffused data itself as the secret+ | expandTimes < 2 = CryptoFailed CryptoError_ParameterInvalid+ | r /= 0 = CryptoFailed CryptoError_ParameterInvalid+ | originalSize <= 0 = CryptoFailed CryptoError_ParameterInvalid+ | otherwise = CryptoPassed $ B.allocAndFreeze originalSize $ \dstPtr -> B.withByteArray bs $ \srcPtr -> do memSet dstPtr 0 originalSize forM_ [0 .. (expandTimes - 2)] $ \i -> do
@@ -22,18 +22,54 @@ PKCS5 | -- | PKCS7 with padding size between 1 and 255 PKCS7 Int- | -- | zero padding with block size+ | -- | Zero padding with block size, which must be at least 1.+ --+ -- Zero padding does not say how much of it there is, so 'unpad' cannot+ -- undo 'pad': see 'unpad'. ZERO Int deriving (Show, Eq) +-- | Is this a block size PKCS7 can describe?+--+-- The padding octet carries the number of octets added, so it cannot describe+-- a block longer than 255, and a block of zero has nothing to describe.+-- Outside that range the octet would be computed as an 'Int' and then narrowed+-- to a 'Data.Word.Word8', which wraps: 'pad' and 'unpad' would agree on the+-- wrapped value and hand back something other than what was padded.+pkcs7SizeValid :: Int -> Bool+pkcs7SizeValid sz = sz >= 1 && sz <= 255++-- | Is this a block size 'ZERO' can use?+--+-- Nothing is written into the padding, so there is no upper bound to match+-- the one 'PKCS7' has; but a block of zero or fewer octets is not a block,+-- and the length is taken modulo it.+zeroSizeValid :: Int -> Bool+zeroSizeValid sz = sz >= 1+ -- | Apply some pad to a bytearray+--+-- A 'PKCS7' block size outside 1..255, or a 'ZERO' block size below 1, raises+-- an 'error'; 'unpad' reports the same condition as 'Nothing'. pad :: ByteArray byteArray => Format -> byteArray -> byteArray pad PKCS5 bin = pad (PKCS7 8) bin-pad (PKCS7 sz) bin = bin `B.append` paddingString+pad (PKCS7 sz) bin+ | not (pkcs7SizeValid sz) =+ error $+ "Crypto.Data.Padding: PKCS7 block size "+ ++ show sz+ ++ " is not between 1 and 255"+ | otherwise = bin `B.append` paddingString where paddingString = B.replicate paddingByte (fromIntegral paddingByte) paddingByte = sz - (B.length bin `mod` sz)-pad (ZERO sz) bin = bin `B.append` paddingString+pad (ZERO sz) bin+ | not (zeroSizeValid sz) =+ error $+ "Crypto.Data.Padding: ZERO block size "+ ++ show sz+ ++ " is not at least 1"+ | otherwise = bin `B.append` paddingString where paddingString = B.replicate paddingSz 0 paddingSz@@ -44,12 +80,25 @@ len = B.length bin -- | Try to remove some padding from a bytearray.+--+-- 'PKCS7' padding says how long it is, so this undoes 'pad' exactly.+--+-- 'ZERO' padding says nothing, and 'pad' adds none at all when the input is+-- already a multiple of the block size, so there is no way to tell padding+-- from data that happens to end in zero octets. This therefore does not undo+-- 'pad': it returns the input unchanged when the last octet is not zero, and+-- 'Nothing' when it is, rather than guess and hand back less than it was+-- given. Zero padding is only usable where the original length is known by+-- other means. unpad :: ByteArray byteArray => Format -> byteArray -> Maybe byteArray unpad PKCS5 bin = unpad (PKCS7 8) bin unpad (PKCS7 sz) bin+ | not (pkcs7SizeValid sz) = Nothing | len == 0 = Nothing | (len `mod` sz) /= 0 = Nothing- | paddingSz < 1 || paddingSz > len = Nothing+ -- the padded length is a multiple of the block size and the padding is+ -- what was added to reach it, so it is never more than one block+ | paddingSz < 1 || paddingSz > sz = Nothing | paddingWitness `B.constEq` padding = Just content | otherwise = Nothing where@@ -59,6 +108,7 @@ (content, padding) = B.splitAt (len - paddingSz) bin paddingWitness = B.replicate paddingSz paddingByte :: Bytes unpad (ZERO sz) bin+ | not (zeroSizeValid sz) = Nothing | len == 0 = Nothing | (len `mod` sz) /= 0 = Nothing | B.index bin (len - 1) /= 0 = Just bin
@@ -0,0 +1,52 @@+-- |+-- Module : Crypto.Debug+-- License : BSD-style+-- Maintainer : Kazu Yamamoto <kazu@iij.ad.jp>+-- Stability : experimental+-- Portability : unknown+--+-- Printing secret key material, on purpose.+--+-- The 'Show' instance of a type that holds a secret does not print it. That+-- is deliberate: 'Show' is what @print@, a message built with @error@, an+-- exception and a test framework's failure output all reach for, and a+-- private key reaching a log or a bug report that way is an accident nobody+-- asked for. Those instances render the public part and write @\<secret\>@+-- for the rest.+--+-- This module is how you print one when printing it is what you mean. What+-- 'debugShow' returns is what the derived 'Show' used to return, so for the+-- types that still have a 'Read' instance+--+-- > read (debugShow k) == k+--+-- and a call site that was serializing a key through @show@ moves by one+-- word.+--+-- Needing 'debugShow' in scope is the record of the intent: nothing here is+-- exported anywhere else, so a search for this module finds every place a key+-- can be revealed. Do not leave a call to it where production code runs.+module Crypto.Debug (+ DebugShow (..),+ debugShowBytes,+) where++import Data.Bits (shiftR, (.&.))+import qualified Data.ByteArray as BA+import Data.Word (Word8)++-- | Rendering a value with its secret in place.+class DebugShow a where+ -- | Render the value, secret included.+ debugShow :: a -> String++-- | Render a secret that is held as bytes, in hexadecimal. The secret keys+-- that keep theirs in a @ScrubbedBytes@ never had a 'Show' that printed it,+-- so unlike the rest of this module what comes back is for reading and not+-- for 'Prelude.read'.+debugShowBytes :: BA.ByteArrayAccess ba => String -> ba -> String+debugShowBytes con b = con ++ (' ' : concatMap hex (BA.unpack b))+ where+ hex :: Word8 -> String+ hex w = [digit (w `shiftR` 4), digit (w .&. 0x0f)]+ digit n = "0123456789abcdef" !! fromIntegral n
@@ -12,6 +12,24 @@ -- Portability : unknown -- -- Elliptic Curve Cryptography+--+-- == Timing+--+-- t'Curve_P256R1' reaches a dedicated implementation whose scalar+-- multiplication does not branch on the scalar. t'Curve_P384R1' and+-- t'Curve_P521R1' do not: they are built on "Crypto.ECC.Simple.Prim", whose+-- scalar multiplication is a double-and-add over @Integer@ and is+-- documented there as vulnerable to timing attacks.+--+-- That matters wherever the scalar is secret, which is both operations that+-- have one: 'ecdh', which multiplies by the private key, and ECDSA signing,+-- which multiplies by the secret nonce. Verification and public-key+-- derivation work on values an attacker already has, so they are unaffected.+--+-- Note also that @Integer@ arithmetic is variable-time underneath, so no+-- curve built on "Crypto.ECC.Simple.Prim" can be made constant-time without+-- leaving it. Where that matters, use t'Curve_P256R1', t'Curve_X25519',+-- t'Curve_X448' or t'Curve_Edwards25519'. module Crypto.ECC ( Curve_P256R1 (..), Curve_P384R1 (..),@@ -31,13 +49,12 @@ import qualified Crypto.ECC.Simple.Prim as Simple import qualified Crypto.ECC.Simple.Types as Simple import Crypto.Error+import Crypto.KEM (SharedSecret (..)) import Crypto.Internal.ByteArray ( ByteArray, ByteArrayAccess,- ScrubbedBytes, ) import qualified Crypto.Internal.ByteArray as B-import Crypto.Internal.Imports import Crypto.Number.Basic (numBits) import Crypto.Number.Serialize (i2ospOf_, os2ip) import qualified Crypto.Number.Serialize.LE as LE@@ -57,16 +74,6 @@ , keypairGetPrivate :: !(Scalar curve) } --- | Secret shared via key exchange-newtype SharedSecret = SharedSecret ScrubbedBytes- deriving (Eq, ByteArrayAccess, NFData)--instance Semigroup SharedSecret where- SharedSecret x <> SharedSecret y = SharedSecret (x <> y)--instance Monoid SharedSecret where- mempty = SharedSecret mempty- class EllipticCurve curve where -- | Point on an Elliptic Curve type Point curve :: Type@@ -204,8 +211,17 @@ instance EllipticCurveDH Curve_P256R1 where ecdhRaw _ s p = SharedSecret $ P256.pointDh s p- ecdh prx s p = checkNonZeroDH (ecdhRaw prx s p) + -- An all-zero x-coordinate can be valid. Since P-256's group has prime+ -- order n, s * P is the identity only when P is the identity or the+ -- 256-bit scalar s is zero or n.+ ecdh _ s p+ | P256.pointIsAtInfinity p+ || P256.scalarIsZero s+ || P256.scalarCmp s P256.scalarN == EQ =+ CryptoFailed CryptoError_ScalarMultiplicationInvalid+ | otherwise = CryptoPassed $ SharedSecret $ P256.pointDh s p+ instance EllipticCurveBasepointArith Curve_P256R1 where curveOrderBits _ = 256 pointBaseSmul _ = P256.toPoint@@ -215,6 +231,10 @@ scalarAdd _ = P256.scalarAdd scalarMul _ = P256.scalarMul +-- | NIST P-384.+--+-- Scalar multiplication branches on the scalar; see the note on timing+-- at the head of this module. data Curve_P384R1 = Curve_P384R1 deriving (Show, Data) @@ -251,6 +271,10 @@ scalarAdd _ = ecScalarAdd scalarMul _ = ecScalarMul +-- | NIST P-521.+--+-- Scalar multiplication branches on the scalar; see the note on timing+-- at the head of this module. data Curve_P521R1 = Curve_P521R1 deriving (Show, Data)
@@ -1,3 +1,4 @@+{-# LANGUAGE BangPatterns #-} {-# LANGUAGE ScopedTypeVariables #-} -- | Elliptic Curve Arithmetic.@@ -15,14 +16,19 @@ pointFromIntegers, isPointAtInfinity, isPointValid,+ isPointInSubgroup, ) where import Crypto.ECC.Simple.Types import Crypto.Error+import Crypto.Internal.ECC (CurveField (..), MulResult (..), curveMul)+import Crypto.Number.Basic (numBits) import Crypto.Number.F2m import Crypto.Number.Generate (generateBetween) import Crypto.Number.ModArithmetic import Crypto.Random+import Data.Bits (shiftL, shiftR, testBit, (.&.))+ import Data.Maybe import Data.Proxy @@ -125,41 +131,231 @@ pointBaseMul :: Curve curve => Scalar curve -> Point curve pointBaseMul n = pointMul n (curveEccG $ curveParameters (Proxy :: Proxy curve)) --- | Elliptic curve point multiplication (double and add algorithm).+-- | Elliptic curve point multiplication. ----- /WARNING:/ Vulnerable to timing attacks.-pointMul :: Curve curve => Scalar curve -> Point curve -> Point curve+-- Over a prime field this goes to C, four bits of scalar at a time, with the+-- multiple to add taken from a table read by touching every entry of it.+-- Over a binary field it also goes to C, as Montgomery's ladder: it carries+-- the x coordinates of two consecutive multiples -- their difference being+-- the point is what lets it carry no more than that -- and spends one+-- addition and one doubling on every bit whichever way the bit goes, with the+-- two exchanged by a mask rather than chosen by a branch. Either way the work+-- follows the width of the curve's order and not the scalar.+--+-- What falls back on the 'Integer' arithmetic below is a point that is not on+-- the curve, the one point of a binary curve that has no x, and a prime the C+-- will not take.+--+-- Multiplying the base point of a curve over a prime field -- which is what+-- signing and making a key do, and nothing else does -- goes through a table+-- of its multiples, built when that curve is first asked for one and kept+-- afterwards. The build is a few milliseconds and the table a few hundred+-- kilobytes, and a multiplication that uses it takes about a third of what+-- one without it takes.+--+-- /WARNING:/ What is left of the 'Integer' arithmetic below -- a point off+-- the curve, the one point of a binary curve with no x, a prime or a+-- polynomial the C will not take -- has uniform operation counts at best, and+-- uniform operation counts are not constant time: those operations cost what+-- the values they are given cost. See the note in+-- "Crypto.ECC".+pointMul+ :: forall curve. Curve curve => Scalar curve -> Point curve -> Point curve pointMul _ PointO = PointO pointMul (Scalar n) p | n == 0 = PointO- | n == 1 = p- | odd n = pointAdd p (pointMul (Scalar (n - 1)) p)- | otherwise = pointMul (Scalar (n `div` 2)) (pointDouble p)+ | n < 0 = pointNegate (pointMul (Scalar (negate n) :: Scalar curve) p)+ | otherwise =+ case curveType (Proxy :: Proxy curve) of+ CurvePrime (CurvePrimeParam pr) -> primeMul pr+ CurveBinary (CurveBinaryParam fx) -> binaryMul fx+ where+ cc = curveParameters (Proxy :: Proxy curve)+ a = curveEccA cc+ -- Count to the width of the order, which is public, so a scalar in range+ -- -- which is every secret one -- takes the same number of steps whatever+ -- it is. A scalar may still be given out of range, and then the count has+ -- to follow it or the high bits would be dropped.+ bits = max (integerBits n) (integerBits (curveEccN cc)) --- | Elliptic curve double-scalar multiplication (uses Shamir's trick).+ -- The C answers for a point on the curve; anything else keeps the+ -- answers it has always had from the code below.+ primeMul pr = case p of+ Point px py+ | isPointValid (Proxy :: Proxy curve) px py ->+ answer slow $+ curveMul+ (Prime pr a (curveEccB cc))+ (curveEccN cc)+ n+ px+ py+ (p == curveEccG cc)+ _ -> slow+ where+ slow = jacobianMul pr a bits n p++ -- The ladder answers for a point on the curve that has an x; the one+ -- point with no x, and anything off the curve, keep what they had.+ binaryMul fx = case p of+ Point px py+ | isPointValid (Proxy :: Proxy curve) px py ->+ answer (affineMul n p) $+ curveMul (Binary fx (curveEccB cc)) (curveEccN cc) n px py False+ _ -> affineMul n p++ -- what the C could not take goes back to the code that was here before+ answer fallback r = case r of+ MulPoint x y -> Point x y+ MulInfinity -> PointO+ MulUnsupported -> fallback++ affineMul k q+ | k == 0 = PointO+ | k == 1 = q+ | odd k = pointAdd q (affineMul (k - 1) q)+ | otherwise = affineMul (k `div` 2) (pointDouble q)++-- | Number of bits needed to write n, for n > 0.+integerBits :: Integer -> Int+integerBits = go 0+ where+ go acc 0 = acc+ go acc k = go (acc + 1) (k `div` 2)++-- | A point in Jacobian coordinates: @(X, Y, Z)@ stands for the affine+-- @(X\/Z^2, Y\/Z^3)@, and @JPointO@ for the point at infinity. Only ever+-- used inside this module, since t'Point' is what the curve exposes.+data JPoint = JPointO | JPoint !Integer !Integer !Integer++-- | The prime, the width to fold at, and what to fold back in. A @c@ of zero+-- says to divide instead, either because the prime has no such shape or+-- because it is too small for folding to pay: @c@ has to be under half the+-- width, or folding would not shrink the number, and below 256 bits the+-- handful of 'Integer' operations folding takes costs more than the division+-- it saves -- measured on P-192, where folding is 14% slower. --+-- Most curve primes are @2^k - c@ with @c@ far smaller than the prime, and+-- then reducing is a shift, a multiplication by @c@ and an addition, where+-- dividing a number twice the width costs about four times as much.+data Field = Field !Integer !Int !Integer++mkField :: Integer -> Field+mkField p+ | p > 0 && c > 0 && 2 * numBits c <= k && k >= 256 = Field p k c+ | otherwise = Field p 0 0+ where+ k = numBits p+ c = (1 `shiftL` k) - p++fieldPrime :: Field -> Integer+fieldPrime (Field p _ _) = p++fieldReduce :: Field -> Integer -> Integer+fieldReduce (Field p k c) x+ | c == 0 || x < 0 = x `mod` p+ | otherwise = trim (fold x)+ where+ mask = (1 `shiftL` k) - 1+ fold v+ | v > mask = fold ((v `shiftR` k) * c + (v .&. mask))+ | otherwise = v+ trim v+ | v >= p = trim (v - p)+ | otherwise = v+{-# INLINE fieldReduce #-}++jacobianMul+ :: Integer -> Integer -> Int -> Integer -> Point curve -> Point curve+jacobianMul _ _ _ _ PointO = PointO+jacobianMul pr a bits n (Point px py) = fromJacobian f (go (bits - 1) JPointO)+ where+ f = mkField pr++ -- The bangs are what make the addition happen at every bit. Without+ -- them the one that is not taken stays a thunk and is never worked out,+ -- so the multiplication costs a step for every bit that is set rather+ -- than for every bit there is, and a single measurement tells an attacker+ -- how many bits of the scalar are set.+ go i acc+ | i < 0 = acc+ | otherwise =+ let !d = jDouble f a acc+ !s = jAddAffine f a d px py+ in go (i - 1) (if testBit n i then s else d)++jDouble :: Field -> Integer -> JPoint -> JPoint+jDouble _ _ JPointO = JPointO+jDouble f a (JPoint x y z)+ | y == 0 = JPointO+ | otherwise = JPoint x3 y3 z3+ where+ red = fieldReduce f+ yy = red (y * y)+ delta = red (4 * x * yy)+ zz = red (z * z)+ m = red (3 * x * x + a * zz * zz)+ x3 = red (m * m - 2 * delta)+ y3 = red (m * (delta - x3) - 8 * yy * yy)+ z3 = red (2 * y * z)++-- | Add a point whose z is one, which is what a scalar multiplication always+-- adds: u1 is x1, s1 is y1, and z3 is one multiplication rather than two.+jAddAffine :: Field -> Integer -> JPoint -> Integer -> Integer -> JPoint+jAddAffine _ _ JPointO x2 y2 = JPoint x2 y2 1+jAddAffine f a p@(JPoint x1 y1 z1) x2 y2+ | h /= 0 = JPoint x3 y3 z3+ | r /= 0 = JPointO+ | otherwise = jDouble f a p+ where+ red = fieldReduce f+ z1s = red (z1 * z1)+ u2 = red (x2 * z1s)+ s2 = red (y2 * z1s * z1)+ h = red (u2 - x1)+ r = red (s2 - y1)+ h2 = red (h * h)+ h3 = red (h2 * h)+ x3 = red (r * r - h3 - 2 * x1 * h2)+ y3 = red (r * (x1 * h2 - x3) - y1 * h3)+ z3 = red (h * z1)++fromJacobian :: Field -> JPoint -> Point curve+fromJacobian _ JPointO = PointO+fromJacobian f (JPoint x y z) =+ case inverse z (fieldPrime f) of+ Nothing -> PointO+ Just zi ->+ let red = fieldReduce f+ zi2 = red (zi * zi)+ in Point (red (x * zi2)) (red (y * zi2 * zi))++-- | Elliptic curve double-scalar multiplication.+-- -- > pointAddTwoMuls n1 p1 n2 p2 == pointAdd (pointMul n1 p1) -- > (pointMul n2 p2) --+-- which is how it is done: the two multiplications separately, and then one+-- addition.+--+-- This used to be Shamir's trick, one pass over the bits of both scalars at+-- once, which shares the doublings between them and is the right thing to do+-- when the two multiplications would cost the same. They no longer do.+-- 'pointMul' goes to C, and over a prime field it multiplies the base point+-- through a table of its multiples, which is a third of the price of an+-- ordinary multiplication -- and the base point is one of the two here,+-- since ECDSA verification is what asks for this. Sharing the doublings+-- with a pass in "Integer" arithmetic gives that up and more: on P-384 it+-- costs twice what two multiplications in C cost, and on the curves over a+-- binary field, whose addition needs an inversion where C has a ladder that+-- needs none, it costs two hundred times as much.+-- -- /WARNING:/ Vulnerable to timing attacks. pointAddTwoMuls- :: Curve curve+ :: forall curve+ . Curve curve => Scalar curve -> Point curve -> Scalar curve -> Point curve -> Point curve-pointAddTwoMuls _ PointO _ PointO = PointO-pointAddTwoMuls _ PointO n2 p2 = pointMul n2 p2-pointAddTwoMuls n1 p1 _ PointO = pointMul n1 p1-pointAddTwoMuls (Scalar n1) p1 (Scalar n2) p2 = go (n1, n2)- where- p0 = pointAdd p1 p2-- go (0, 0) = PointO- go (k1, k2) =- let q = pointDouble $ go (k1 `div` 2, k2 `div` 2)- in case (odd k1, odd k2) of- (True, True) -> pointAdd p0 q- (True, False) -> pointAdd p1 q- (False, True) -> pointAdd p2 q- (False, False) -> q+pointAddTwoMuls n1 p1 n2 p2 = pointAdd (pointMul n1 p1) (pointMul n2 p2) -- | Check if a point is the point at infinity. isPointAtInfinity :: Point curve -> Bool@@ -173,9 +369,13 @@ pointFromIntegers :: forall curve. Curve curve => (Integer, Integer) -> CryptoFailable (Point curve) pointFromIntegers (x, y)- | isPointValid (Proxy :: Proxy curve) x y = CryptoPassed $ Point x y- | otherwise =- CryptoFailed $ CryptoError_PointCoordinatesInvalid+ | not (isPointValid (Proxy :: Proxy curve) x y) =+ CryptoFailed CryptoError_PointCoordinatesInvalid+ | not (isPointInSubgroup (Proxy :: Proxy curve) p) =+ CryptoFailed CryptoError_PointSubgroupInvalid+ | otherwise = CryptoPassed p+ where+ p = Point x y -- | check if a point is on specific curve --@@ -206,6 +406,24 @@ ] where ty = curveType proxy+ cc = curveParameters proxy++-- | Check that a point is in the subgroup the base point generates, which is+-- the further check 'isPointValid' does not make. A point that is on the+-- curve but outside that subgroup answers a multiplication modulo an order+-- smaller than the group's, so the multiplier -- a private number, where the+-- point came from a peer -- is revealed modulo that small order.+--+-- Where the cofactor is 1 the subgroup is the whole curve group and the+-- answer is 'True' for any point on the curve, at no cost. Otherwise the+-- point is multiplied by the group order and the answer is whether that+-- reaches the point at infinity, which costs one scalar multiplication.+isPointInSubgroup+ :: forall proxy curve. Curve curve => proxy curve -> Point curve -> Bool+isPointInSubgroup proxy p+ | curveEccH cc == 1 = True+ | otherwise = pointMul (Scalar (curveEccN cc) :: Scalar curve) p == PointO+ where cc = curveParameters proxy -- | div and mod
@@ -153,6 +153,19 @@ data SEC_t571k1 = SEC_t571k1 deriving (Show, Read, Eq) data SEC_t571r1 = SEC_t571r1 deriving (Show, Read, Eq) +{-# DEPRECATED+ SEC_t113r1, SEC_t113r2, SEC_t131r1, SEC_t131r2, SEC_t163k1, SEC_t163r1,+ SEC_t163r2, SEC_t193r1, SEC_t193r2, SEC_t233k1, SEC_t233r1, SEC_t239k1,+ SEC_t283k1, SEC_t283r1, SEC_t409k1, SEC_t409r1, SEC_t571k1, SEC_t571r1+ [ "This curve is over a binary field, and those are obsolete."+ , "They are also the curves whose cofactor is not 1, so a point from"+ , "a peer needs the subgroup check that costs a further scalar"+ , "multiplication; pyca/cryptography deprecated them for removal in"+ , "the release that fixed CVE-2026-26007. This one will go in a"+ , "later major version of crypton. Prefer a prime curve, or X25519."+ ]+ #-}+ -- | Define names for known recommended curves. instance Curve SEC_p112r1 where curveType _ = typeSEC_p112r1
@@ -50,6 +50,17 @@ CryptoError_SaltTooSmall | CryptoError_OutputLengthTooSmall | CryptoError_OutputLengthTooBig+ | -- | A parameter is outside the range the algorithm accepts. Appended to+ -- keep the 'Enum' values of the constructors above unchanged.+ CryptoError_ParameterInvalid+ | -- | A point satisfies the curve equation but lies outside the subgroup+ -- the base point generates, so multiplying it would answer modulo a+ -- small order. Appended for the same reason as the constructor above.+ CryptoError_PointSubgroupInvalid+ | -- | A public key is the right length but is not a well-formed encoding+ -- of one, so no honest party produced it. Appended for the same+ -- reason as the two constructors above.+ CryptoError_PublicKeyStructureInvalid deriving (Show, Eq, Enum, Data) instance E.Exception CryptoError
@@ -48,8 +48,10 @@ Blake2bp (..), Blake2s (..), Blake2sp (..),+ Skein256 (..), Skein256_224 (..), Skein256_256 (..),+ Skein512 (..), Skein512_224 (..), Skein512_256 (..), Skein512_384 (..),
@@ -44,9 +44,9 @@ -- | SHAKE128 (128 bits) extendable output function. Supports an arbitrary -- digest size, to be specified as a type parameter of kind 'Nat'. ----- Note: outputs from @'SHAKE128' n@ and @'SHAKE128' m@ for the same input are+-- Note: outputs from @t'SHAKE128' n@ and @t'SHAKE128' m@ for the same input are -- correlated (one being a prefix of the other). Results are unrelated to--- 'SHAKE256' results.+-- t'SHAKE256' results. data SHAKE128 (bitlen :: Nat) = SHAKE128 deriving (Show, Data) @@ -68,9 +68,9 @@ -- | SHAKE256 (256 bits) extendable output function. Supports an arbitrary -- digest size, to be specified as a type parameter of kind 'Nat'. ----- Note: outputs from @'SHAKE256' n@ and @'SHAKE256' m@ for the same input are+-- Note: outputs from @t'SHAKE256' n@ and @t'SHAKE256' m@ for the same input are -- correlated (one being a prefix of the other). Results are unrelated to--- 'SHAKE128' results.+-- t'SHAKE128' results. data SHAKE256 (bitlen :: Nat) = SHAKE256 deriving (Show, Data)
@@ -1,7 +1,11 @@ {-# LANGUAGE DataKinds #-} {-# LANGUAGE DeriveDataTypeable #-} {-# LANGUAGE ForeignFunctionInterface #-}+{-# LANGUAGE KindSignatures #-}+{-# LANGUAGE ScopedTypeVariables #-} {-# LANGUAGE TypeFamilies #-}+{-# LANGUAGE TypeOperators #-}+{-# LANGUAGE UndecidableInstances #-} -- | -- Module : Crypto.Hash.Skein256@@ -13,14 +17,17 @@ -- Module containing the binding functions to work with the -- Skein256 cryptographic hash. module Crypto.Hash.Skein256 (+ Skein256 (..), Skein256_224 (..), Skein256_256 (..), ) where import Crypto.Hash.Types+import Crypto.Internal.Nat import Data.Data import Data.Word (Word32, Word8) import Foreign.Ptr (Ptr)+import GHC.TypeLits (KnownNat, Nat, type (+)) -- | Skein256 (224 bits) cryptographic hash algorithm data Skein256_224 = Skein256_224@@ -51,6 +58,38 @@ hashInternalInit p = c_skein256_init p 256 hashInternalUpdate = c_skein256_update hashInternalFinalize p = c_skein256_finalize p 256++-- | Skein256 with the digest size given as a type parameter of kind 'Nat',+-- in bits. @t'Skein256' 256@ is @t'Skein256_256'@; the sizes with a type of+-- their own+-- above are there for their names, and this one also takes the sizes that+-- have none.+--+-- A size that is not a whole number of bytes is rounded up to the next one,+-- as the implementation underneath does.+--+-- The output is produced in counter mode, a block of it per Threefish call,+-- so one large digest is a good deal cheaper than the same number of bytes+-- taken from repeated small ones: on an Apple M4, 512 KiB arrives at 947 MB/s+-- in one digest against 172 MB/s as 8192 separate @t'Skein256_256'@ ones.+--+-- Note the digest size goes into the configuration block, so it changes the+-- value the message is hashed from: a longer digest is /not/ an extension of+-- a shorter one. That is the opposite of how t'Crypto.Hash.SHAKE.SHAKE128'+-- behaves.+data Skein256 (bitlen :: Nat) = Skein256+ deriving (Show, Data)++instance KnownNat bitlen => HashAlgorithm (Skein256 bitlen) where+ type HashBlockSize (Skein256 bitlen) = 32+ type HashDigestSize (Skein256 bitlen) = Div8 (bitlen + 7)+ type HashInternalContextSize (Skein256 bitlen) = 96+ hashBlockSize _ = 32+ hashDigestSize _ = byteLen (Proxy :: Proxy bitlen)+ hashInternalContextSize _ = 96+ hashInternalInit p = c_skein256_init p (integralNatVal (Proxy :: Proxy bitlen))+ hashInternalUpdate = c_skein256_update+ hashInternalFinalize p = c_skein256_finalize p (integralNatVal (Proxy :: Proxy bitlen)) foreign import ccall unsafe "crypton_skein256_init" c_skein256_init :: Ptr (Context a) -> Word32 -> IO ()
@@ -1,7 +1,11 @@ {-# LANGUAGE DataKinds #-} {-# LANGUAGE DeriveDataTypeable #-} {-# LANGUAGE ForeignFunctionInterface #-}+{-# LANGUAGE KindSignatures #-}+{-# LANGUAGE ScopedTypeVariables #-} {-# LANGUAGE TypeFamilies #-}+{-# LANGUAGE TypeOperators #-}+{-# LANGUAGE UndecidableInstances #-} -- | -- Module : Crypto.Hash.Skein512@@ -13,6 +17,7 @@ -- Module containing the binding functions to work with the -- Skein512 cryptographic hash. module Crypto.Hash.Skein512 (+ Skein512 (..), Skein512_224 (..), Skein512_256 (..), Skein512_384 (..),@@ -20,9 +25,11 @@ ) where import Crypto.Hash.Types+import Crypto.Internal.Nat import Data.Data import Data.Word (Word32, Word8) import Foreign.Ptr (Ptr)+import GHC.TypeLits (KnownNat, Nat, type (+)) -- | Skein512 (224 bits) cryptographic hash algorithm data Skein512_224 = Skein512_224@@ -83,6 +90,38 @@ hashInternalInit p = c_skein512_init p 512 hashInternalUpdate = c_skein512_update hashInternalFinalize p = c_skein512_finalize p 512++-- | Skein512 with the digest size given as a type parameter of kind 'Nat',+-- in bits. @t'Skein512' 512@ is @t'Skein512_512'@; the sizes with a type of+-- their own+-- above are there for their names, and this one also takes the sizes that+-- have none.+--+-- A size that is not a whole number of bytes is rounded up to the next one,+-- as the implementation underneath does.+--+-- The output is produced in counter mode, a block of it per Threefish call,+-- so one large digest is a good deal cheaper than the same number of bytes+-- taken from repeated small ones: on an Apple M4, 512 KiB arrives at 947 MB/s+-- in one digest against 172 MB/s as 8192 separate @t'Skein512_512'@ ones.+--+-- Note the digest size goes into the configuration block, so it changes the+-- value the message is hashed from: a longer digest is /not/ an extension of+-- a shorter one. That is the opposite of how t'Crypto.Hash.SHAKE.SHAKE128'+-- behaves.+data Skein512 (bitlen :: Nat) = Skein512+ deriving (Show, Data)++instance KnownNat bitlen => HashAlgorithm (Skein512 bitlen) where+ type HashBlockSize (Skein512 bitlen) = 64+ type HashDigestSize (Skein512 bitlen) = Div8 (bitlen + 7)+ type HashInternalContextSize (Skein512 bitlen) = 160+ hashBlockSize _ = 64+ hashDigestSize _ = byteLen (Proxy :: Proxy bitlen)+ hashInternalContextSize _ = 160+ hashInternalInit p = c_skein512_init p (integralNatVal (Proxy :: Proxy bitlen))+ hashInternalUpdate = c_skein512_update+ hashInternalFinalize p = c_skein512_finalize p (integralNatVal (Proxy :: Proxy bitlen)) foreign import ccall unsafe "crypton_skein512_init" c_skein512_init :: Ptr (Context a) -> Word32 -> IO ()
@@ -103,6 +103,26 @@ -- layout is architecture dependent, may contain uninitialized data fragments, -- and change in future versions. The bytearray should not be used as input to -- cryptographic algorithms.+--+-- __A context is not erased when it is finished with.__ A hash algorithm+-- buffers its input a block at a time, and finalizing does not clear what is+-- left there. How much survives depends on where the message ended relative+-- to the block: with SHA-256, a 32-byte message is still in the context in+-- full afterwards, and a 100-byte one leaves its last 36 bytes.+-- @hashFinalize@ works on a copy, so the caller's own context keeps what it+-- had as well. Nothing clears either of them: this is 'Bytes' rather than+-- @ScrubbedBytes@, and the C clears nothing. They go to the garbage+-- collector as they are, and a core file, a crash dump or a swapped page can+-- carry them away afterwards.+--+-- That is a deliberate trade rather than an oversight, and there is no way+-- to ask for the other side of it: no operation here clears a context.+-- Scrubbing them all was measured at about 70% of a 32-byte hash and a third+-- of an incremental one, because the allocation is most of the work when the+-- message is short -- and short hashes are the common case, in HMAC, in+-- HKDF, and anywhere a key or an identifier is hashed. A 64 KB hash does+-- not notice it. Anything that must not be left in memory this way is+-- better not hashed through this interface at all. newtype Context a = Context Bytes deriving (ByteArrayAccess, NFData)
@@ -14,8 +14,11 @@ module Data.ByteArray.Mapping, module Data.ByteArray.Encoding, constAllZero,+ overCLength,+ inCLengths, allocAndFreezePrimIO, allocAndFreezePrim,+ bxor, ) where import Data.ByteArray@@ -24,12 +27,57 @@ import Data.Bits ((.|.)) import qualified Data.Primitive.ByteArray as Prim-import Data.Word (Word8)+import Data.Word (Word32, Word8) import Foreign.Ptr (Ptr, castPtr) import Foreign.Storable (peekByteOff) import Crypto.Internal.Compat (unsafeDoIO) +-- | Whether a length is too large to reach the C, which takes its lengths as+-- @uint32_t@.+--+-- From 2^32 up the value is truncated on the way down, and the C then works+-- on the low bits of it and leaves the rest of the buffer as it found it --+-- which for a fresh allocation is zeros. What comes back is as long as the+-- caller asked for, with nothing to say that most of it was never written:+-- a 4 GiB message through Crypto.Cipher.ChaCha.combine came back with 2^32+-- bytes of zeros where the ciphertext should have been.+--+-- Every place that hands a caller's length to the C either turns it away+-- with this or cuts the work into pieces small enough to pass.+--+-- The round trip through 'Word32' rather than a comparison against 2^32,+-- which a 32-bit 'Int' cannot hold. There every non-negative 'Int' passes,+-- which is the right answer: there is no such buffer to be had.+overCLength :: Int -> Bool+overCLength n = fromIntegral (fromIntegral n :: Word32) /= n++-- | Walk a length in pieces small enough to reach the C, calling the action+-- with the offset and the size of each.+--+-- The same 2 GiB step "Crypto.Hash" has always taken, and for the same+-- reason: the C takes its lengths as @uint32_t@, and a 32-bit 'Int' cannot+-- hold a whole one either. This is for the C that keeps its state in a+-- context and can simply be called again -- the stream ciphers, the MACs --+-- where a long message can be enciphered in pieces rather than refused.+inCLengths :: Int -> (Int -> Int -> IO ()) -> IO ()+inCLengths total f = go 0+ where+ go !off+ | off >= total = return ()+ | otherwise = f off n >> go (off + n)+ where+ !n = min (total - off) cChunk++-- | The step 'inCLengths' takes: the largest multiple of 64 that a signed+-- 32-bit integer holds.+--+-- Under 2^31 because a 32-bit 'Int' cannot hold more, and a+-- multiple of 64 because some of the C this feeds -- the AEAD modes -- will+-- take a piece that is not a whole number of blocks only as the last one.+cChunk :: Int+cChunk = 0x7fffffc0+ -- | Allocate a pinned 'Prim.ByteArray' of the given size, populate it via a -- 'Ptr', then freeze and return it. The pointer must not be retained after -- the action returns.@@ -54,3 +102,22 @@ e <- peekByteOff p i loop p (i + 1) (acc .|. e) len = Data.ByteArray.length b++-- | @a@ exclusive-ored with @b@, as long as the shorter of the two.+--+-- 'Data.ByteArray.xor' does this a byte at a time through an IO applicative,+-- which allocates about fifty bytes of heap for every byte it produces. That+-- is more than a block cipher costs: it was four fifths of the time counter+-- mode spent on anything but AES, whose modes are in C and do not come this+-- way.+bxor :: (ByteArrayAccess a, ByteArrayAccess b, ByteArray c) => a -> b -> c+bxor a b = unsafeDoIO $+ alloc n $ \pd ->+ withByteArray a $ \pa ->+ withByteArray b $ \pb ->+ c_memxor pd pa pb (fromIntegral n)+ where+ n = min (Data.ByteArray.length a) (Data.ByteArray.length b)++foreign import ccall unsafe "crypton_memxor.h crypton_memxor"+ c_memxor :: Ptr Word8 -> Ptr Word8 -> Ptr Word8 -> Word32 -> IO ()
@@ -0,0 +1,524 @@+{-# LANGUAGE BangPatterns #-}++-- |+-- Module : Crypto.Internal.ECC+-- License : BSD-style+-- Maintainer : Kazu Yamamoto <kazu@iij.ad.jp>+-- Stability : experimental+-- Portability : Good+--+-- The C scalar multiplication for curves over a prime field, which both of+-- the elliptic curve APIs reach for.+module Crypto.Internal.ECC (+ MulResult (..),+ CurveField (..),+ curveMul,+ primeCurveMul,+ primeCurveTableMul,+ baseTable,+ binaryCurveMul,+ binaryCurveC,+) where++import Crypto.Internal.Compat (unsafeDoIO)+import Crypto.Number.Basic (numBits, numBytes)+import Crypto.Number.F2m (addF2m, divF2m, mulF2m, squareF2m)+import qualified Crypto.Number.Serialize.Internal as Internal+import Crypto.PubKey.ECC.Types (+ Curve (..),+ CurveCommon (..),+ CurveName,+ CurvePrime (..),+ Point (..),+ getCurveByName,+ )+import Data.Bits (testBit)+import Data.Word (Word32, Word8)+import Foreign.C.Types (CInt (..))+import Foreign.ForeignPtr (ForeignPtr, mallocForeignPtrBytes, withForeignPtr)+import Foreign.Marshal.Alloc (allocaBytes)+import Foreign.Ptr (Ptr, plusPtr)++-- | What the C made of it.+data MulResult+ = -- | the point it arrived at+ MulPoint !Integer !Integer+ | -- | the point at infinity, which has no coordinates+ MulInfinity+ | -- | not something the C works with, so the caller has to+ MulUnsupported+ deriving (Show, Eq)++-- | Multiply a point by a scalar on the curve @y^2 = x^3 + a*x + b@ over the+-- field of @p@, which has to be an odd prime. The point has to be on the+-- curve and not the point at infinity, and its coordinates, @a@ and @b@ have+-- to be under @p@; the caller has all of that to hand and the C does not+-- check it.+--+-- The scalar is walked four bits at a time over the whole of the width asked+-- for, so its value is hidden but that width is not. Ask for the width of+-- the curve's order, which is public, and every scalar in range costs the+-- same.+primeCurveMul+ :: Integer+ -- ^ p+ -> Integer+ -- ^ a+ -> Integer+ -- ^ b+ -> Int+ -- ^ how many bytes of scalar to walk+ -> Integer+ -- ^ the scalar+ -> Integer+ -- ^ the point's x+ -> Integer+ -- ^ the point's y+ -> MulResult+primeCurveMul p a b klen k px py+ | p <= 0 || even p || klen <= 0 || k < 0 = MulUnsupported+ | otherwise = unsafeDoIO $+ allocaBytes (sum widths) $ \base -> case scanl plusPtr base widths of+ (outx : outy : cx : cy : ca : cb : cp : ck : _) -> do+ _ <- Internal.i2ospOf px cx plen+ _ <- Internal.i2ospOf py cy plen+ _ <- Internal.i2ospOf a ca plen+ _ <- Internal.i2ospOf b cb plen+ _ <- Internal.i2ospOf p cp plen+ _ <- Internal.i2ospOf k ck klen+ r <-+ c_ecc_mul+ outx+ outy+ cx+ cy+ ck+ (fromIntegral klen)+ ca+ cb+ cp+ (fromIntegral plen)+ -- the scalar is the caller's secret, and this is the last place+ -- it is written out in the clear+ Internal.i2ospOf 0 ck klen >> return ()+ case r of+ 0 -> do+ !x <- Internal.os2ip outx plen+ !y <- Internal.os2ip outy plen+ return (MulPoint x y)+ 1 -> return MulInfinity+ _ -> return MulUnsupported+ _ -> return MulUnsupported -- there are eight, but say so anyway+ where+ !plen = numBytes p+ -- What the buffer holds, in this order: the two coordinates out, the two+ -- in, a, b, the prime, and the scalar. The room to take and where each+ -- one starts both come from here, so they cannot drift apart.+ --+ -- They did once, and nothing caught it: the memory is a pinned array on+ -- the GHC heap, so writing past it is invisible to valgrind, which sees+ -- one large allocation, and to the sanity checks of the debug RTS, which+ -- found nothing when the mistake was put back to try them. One runner+ -- out of eighteen died of it and the rest went green. The way to be+ -- right about this is not to have two numbers to keep the same.+ widths = [plen, plen, plen, plen, plen, plen, plen, klen]++foreign import ccall unsafe "crypton_ecc_table_size"+ c_ecc_table_size :: Word32 -> Word32 -> Word32++foreign import ccall safe "crypton_ecc_table_build"+ c_ecc_table_build+ :: Ptr Word8+ -> Ptr Word8+ -> Ptr Word8+ -> Word32+ -> Ptr Word8+ -> Ptr Word8+ -> Ptr Word8+ -> Word32+ -> IO CInt++foreign import ccall safe "crypton_ecc_table_mul"+ c_ecc_table_mul+ :: Ptr Word8+ -> Ptr Word8+ -> Ptr Word8+ -> Ptr Word8+ -> Word32+ -> Ptr Word8+ -> Ptr Word8+ -> Ptr Word8+ -> Word32+ -> IO CInt++foreign import ccall safe "crypton_ecc_mul"+ c_ecc_mul+ :: Ptr Word8+ -> Ptr Word8+ -> Ptr Word8+ -> Ptr Word8+ -> Ptr Word8+ -> Word32+ -> Ptr Word8+ -> Ptr Word8+ -> Ptr Word8+ -> Word32+ -> IO CInt++-- | What a curve is made of, as much of it as a multiplication needs.+data CurveField+ = -- | over a prime field: the prime, a and b+ Prime !Integer !Integer !Integer+ | -- | over a binary field: the polynomial and b+ Binary !Integer !Integer+ deriving (Show, Eq)++-- | Multiply a point by a scalar, through the C wherever the C takes it.+--+-- Both elliptic curve APIs come here, so that the decision -- the table for a+-- base point, the C for anything else, what is left over -- is made once and+-- in one place. One of those APIs cannot be reached from outside the library+-- on a curve over a binary field, and this is how that copy stays the same+-- code as the copy everybody runs.+--+-- The caller has seen to it that the point is on the curve, which is what the+-- C takes for granted, and deals with 'MulUnsupported' in whatever way it+-- has.+curveMul+ :: CurveField+ -> Integer+ -- ^ the order of the curve+ -> Integer+ -- ^ the scalar+ -> Integer+ -- ^ the point's x+ -> Integer+ -- ^ the point's y+ -> Bool+ -- ^ whether that point is the curve's base point+ -> MulResult+curveMul field order k px py isBase = case field of+ Prime p a b+ | isBase+ , klen == numBytes order+ , Just table <- baseTable p a b klen px py ->+ primeCurveTableMul table p a b klen k+ | otherwise -> primeCurveMul p a b klen k px py+ Binary fx b+ | px == 0 -> MulUnsupported -- its own negation, and easier the long way+ | otherwise -> case binaryCurveC fx b klen k px py of+ -- the ladder in Haskell, for a field the C will not take+ MulUnsupported -> binaryCurveMul fx b (klen * 8) k px py+ r -> r+ where+ -- Walk the width of the order, which is public, so a scalar in range --+ -- which is every secret one -- costs the same whatever it is. A scalar+ -- may still be given out of range, and then the width has to follow it or+ -- the high bits would be dropped.+ !klen = max (numBytes k) (numBytes order)++-- | The table for the base point of a curve the library knows, which is the+-- point signing and making a key multiply and the only point worth keeping a+-- table for. The curves are told apart by their numbers, which are public,+-- so both of the elliptic curve APIs find the same table.+--+-- Each is built when it is first wanted and kept for as long as the program+-- runs, and a curve nobody multiplies the base point of never has one built.+-- Building costs 2.8 ms for secp256k1, 5.5 for secp384r1 and 10.6 for+-- secp521r1, and the last two take 221 KB and 456 KB. A multiplication with+-- the table takes about a third of what one without it takes, so the build+-- pays for itself after about fifteen of them: a program that signs many+-- times wins, and one that signs once and exits does not.+baseTable+ :: Integer+ -- ^ p+ -> Integer+ -- ^ a+ -> Integer+ -- ^ b+ -> Int+ -- ^ how many bytes of scalar are wanted+ -> Integer+ -- ^ the base point's x+ -> Integer+ -- ^ the base point's y+ -> Maybe (ForeignPtr Word8)+baseTable p a b klen gx gy =+ case lookup (p, a, b, klen, gx, gy) baseTables of+ Just table -> table+ Nothing -> Nothing++type TableKey = (Integer, Integer, Integer, Int, Integer, Integer)++baseTables :: [(TableKey, Maybe (ForeignPtr Word8))]+baseTables =+ [ ((p, a, b, klen, gx, gy), primeCurveTable p a b klen gx gy)+ | name <- [minBound .. maxBound] :: [CurveName]+ , CurveFP (CurvePrime p cc) <- [getCurveByName name]+ , Point gx gy <- [ecc_g cc]+ , let a = ecc_a cc+ , let b = ecc_b cc+ , let klen = numBytes (ecc_n cc)+ ]+{-# NOINLINE baseTables #-}++-- | The multiples of a point that 'primeCurveTableMul' wants: for every four+-- bits of a scalar, the sixteen points those bits can call for. Building it+-- costs a few thousand point operations, and what it saves is all the+-- doublings of every multiplication that uses it, so it is worth keeping for+-- as long as the point is -- which for a curve's base point is forever.+--+-- The arguments are as for 'primeCurveMul'. 'Nothing' means the C would not+-- take them.+primeCurveTable+ :: Integer+ -- ^ p+ -> Integer+ -- ^ a+ -> Integer+ -- ^ b+ -> Int+ -- ^ how many bytes of scalar the table is to cover+ -> Integer+ -- ^ the point's x+ -> Integer+ -- ^ the point's y+ -> Maybe (ForeignPtr Word8)+primeCurveTable p a b klen px py+ | p <= 0 || even p || klen <= 0 || size == 0 = Nothing+ | otherwise = unsafeDoIO $ do+ table <- mallocForeignPtrBytes (fromIntegral size)+ allocaBytes (sum widths) $ \base -> case scanl plusPtr base widths of+ (cx : cy : ca : cb : cp : _) -> do+ _ <- Internal.i2ospOf px cx plen+ _ <- Internal.i2ospOf py cy plen+ _ <- Internal.i2ospOf a ca plen+ _ <- Internal.i2ospOf b cb plen+ _ <- Internal.i2ospOf p cp plen+ r <- withForeignPtr table $ \t ->+ c_ecc_table_build+ t+ cx+ cy+ (fromIntegral klen)+ ca+ cb+ cp+ (fromIntegral plen)+ return $ if r == 0 then Just table else Nothing+ _ -> return Nothing -- there are five, but say so anyway+ where+ !plen = numBytes p+ !size = c_ecc_table_size (fromIntegral plen) (fromIntegral klen)+ -- the point, a, b and the prime, all of the prime's width. The room to+ -- take and where each one starts both come from here, so they cannot+ -- drift apart: they did once, and nothing caught it -- see the note on+ -- primeCurveMul.+ widths = [plen, plen, plen, plen, plen]++-- | Multiply the point a table was built for by a scalar of the width the+-- table was built for. One addition for every four bits and no doublings.+primeCurveTableMul+ :: ForeignPtr Word8+ -- ^ the table+ -> Integer+ -- ^ p+ -> Integer+ -- ^ a+ -> Integer+ -- ^ b+ -> Int+ -- ^ the width the table was built for+ -> Integer+ -- ^ the scalar+ -> MulResult+primeCurveTableMul table p a b klen k+ | p <= 0 || even p || klen <= 0 || k < 0 = MulUnsupported+ | otherwise = unsafeDoIO $+ allocaBytes (sum widths) $ \base -> case scanl plusPtr base widths of+ (outx : outy : ca : cb : cp : ck : _) -> do+ _ <- Internal.i2ospOf a ca plen+ _ <- Internal.i2ospOf b cb plen+ _ <- Internal.i2ospOf p cp plen+ _ <- Internal.i2ospOf k ck klen+ r <- withForeignPtr table $ \t ->+ c_ecc_table_mul+ outx+ outy+ t+ ck+ (fromIntegral klen)+ ca+ cb+ cp+ (fromIntegral plen)+ Internal.i2ospOf 0 ck klen >> return ()+ case r of+ 0 -> do+ !x <- Internal.os2ip outx plen+ !y <- Internal.os2ip outy plen+ return (MulPoint x y)+ 1 -> return MulInfinity+ _ -> return MulUnsupported+ _ -> return MulUnsupported -- there are six, but say so anyway+ where+ !plen = numBytes p+ widths = [plen, plen, plen, plen, plen, klen]++-- | Multiply a point by a scalar on the curve @y^2 + x*y = x^3 + a*x^2 + b@+-- over the binary field of @fx@, by Montgomery's ladder.+--+-- The ladder carries the multiples of two consecutive numbers, whose+-- difference is therefore the point itself, and every bit of the scalar costs+-- one addition and one doubling of them whichever way it goes. Only the x+-- coordinates are carried -- the difference being known is what lets them be+-- -- and the y is worked out at the end from the two of them, which is what+-- makes the coordinates projective: one division for the whole+-- multiplication rather than one for every step.+--+-- The point has to be on the curve and to have an x, which is what the+-- caller has to hand: the one point with no x is its own negation and is+-- easier multiplied the long way. The scalar is walked over the whole of+-- the width asked for, so its value is hidden but that width is not.+binaryCurveMul+ :: Integer+ -- ^ the polynomial the field is over+ -> Integer+ -- ^ b+ -> Int+ -- ^ how many bits of scalar to walk+ -> Integer+ -- ^ the scalar+ -> Integer+ -- ^ the point's x+ -> Integer+ -- ^ the point's y+ -> MulResult+binaryCurveMul fx b bits k x y+ | bits <= 0 || k < 0 || x == 0 = MulUnsupported+ | otherwise = recover (go (bits - 1) (1, 0) (x, 1))+ where+ infixl 6 .+.+ (.+.) = addF2m+ sqr = squareF2m fx+ mul = mulF2m fx++ -- The two of them added, which the difference between them being the+ -- point makes possible from their x coordinates alone. It does not+ -- matter which way round they come.+ madd (xa, za) (xb, zb) =+ let t1 = mul xa zb+ t2 = mul xb za+ z = sqr (t1 .+. t2)+ in (mul x z .+. mul t1 t2, z)++ -- One of them doubled.+ mdouble (xa, za) =+ let xa2 = sqr xa+ za2 = sqr za+ in (sqr xa2 .+. mul b (sqr za2), mul xa2 za2)++ -- Nothing is at infinity to begin with and the point is next to it, and+ -- from there each bit takes the pair to twice where it was. The bangs+ -- are what make both halves happen: without them the one the bit does not+ -- call for would stay a thunk, and the work would follow the scalar.+ go i p1 p2+ | i < 0 = (p1, p2)+ | testBit k i =+ let !s = madd p1 p2+ !d = mdouble p2+ in go (i - 1) s d+ | otherwise =+ let !s = madd p1 p2+ !d = mdouble p1+ in go (i - 1) d s++ -- x1 is the answer and x2 is one point further on; together with the+ -- point they give the y that the ladder does not carry.+ recover ((x1, z1), (x2, z2))+ | z1 == 0 = MulInfinity -- the multiple is at infinity+ | z2 == 0 = MulPoint x (x .+. y) -- the one after it is, so this is -P+ | otherwise = case (divF2m fx x1 z1, divF2m fx x2 z2) of+ (Just xa, Just xb) ->+ let u = xa .+. x+ v = xb .+. x+ inner = mul u v .+. sqr x .+. y+ in case divF2m fx (mul u inner) x of+ Just w -> MulPoint xa (w .+. y)+ Nothing -> MulUnsupported+ _ -> MulUnsupported++-- | Multiply a point by a scalar on a curve over a binary field, in C.+--+-- The ladder is the same one 'binaryCurveMul' walks, but the field arithmetic+-- is carry-less multiplication -- the processor's where it has it, and four+-- interleaved groups of bits where it does not -- rather than 'Integer'+-- shifts and exclusive ors, and nothing in it branches on the scalar or+-- indexes memory with it.+--+-- The point has to be on the curve and to have an x, and the scalar is walked+-- over the whole of the width asked for, as for 'primeCurveMul'.+binaryCurveC+ :: Integer+ -- ^ the polynomial the field is over+ -> Integer+ -- ^ b+ -> Int+ -- ^ how many bytes of scalar to walk+ -> Integer+ -- ^ the scalar+ -> Integer+ -- ^ the point's x+ -> Integer+ -- ^ the point's y+ -> MulResult+binaryCurveC fx b klen k px py+ | fx <= 1 || klen <= 0 || k < 0 || px <= 0 || flen <= 0 = MulUnsupported+ | otherwise = unsafeDoIO $+ allocaBytes (sum widths) $ \base -> case scanl plusPtr base widths of+ (outx : outy : cx : cy : cb : cf : ck : _) -> do+ _ <- Internal.i2ospOf px cx flen+ _ <- Internal.i2ospOf py cy flen+ _ <- Internal.i2ospOf b cb flen+ _ <- Internal.i2ospOf fx cf fxlen+ _ <- Internal.i2ospOf k ck klen+ r <-+ c_f2m_mul+ outx+ outy+ cx+ cy+ ck+ (fromIntegral klen)+ cb+ (fromIntegral flen)+ cf+ (fromIntegral fxlen)+ Internal.i2ospOf 0 ck klen >> return ()+ case r of+ 0 -> do+ !x <- Internal.os2ip outx flen+ !y <- Internal.os2ip outy flen+ return (MulPoint x y)+ 1 -> return MulInfinity+ _ -> return MulUnsupported+ _ -> return MulUnsupported -- there are seven, but say so anyway+ where+ -- the field is the degree of the polynomial, which is one under its width+ !flen = (numBits fx - 1 + 7) `div` 8+ !fxlen = numBytes fx+ widths = [flen, flen, flen, flen, flen, fxlen, klen]++foreign import ccall safe "crypton_f2m_mul"+ c_f2m_mul+ :: Ptr Word8+ -> Ptr Word8+ -> Ptr Word8+ -> Ptr Word8+ -> Ptr Word8+ -> Word32+ -> Ptr Word8+ -> Word32+ -> Ptr Word8+ -> Word32+ -> IO CInt
@@ -33,7 +33,7 @@ IsLE bitlen n 'False = 'False #endif --- | ensure the given `bitlen` is lesser or equal to `n`+-- | ensure the given @bitlen@ is lesser or equal to @n@ -- type IsAtMost (bitlen :: Nat) (n :: Nat) = IsLE bitlen n (bitlen <=? n) ~ 'True @@ -48,7 +48,7 @@ IsGE bitlen n 'False = 'False #endif --- | ensure the given `bitlen` is greater or equal to `n`+-- | ensure the given @bitlen@ is greater or equal to @n@ -- type IsAtLeast (bitlen :: Nat) (n :: Nat) = IsGE bitlen n (n <=? bitlen) ~ 'True @@ -208,6 +208,6 @@ Mod8 63 = 7 Mod8 n = Mod8 (n - 64) --- | ensure the given `bitlen` is divisible by 8+-- | ensure the given @bitlen@ is divisible by 8 -- type IsDivisibleBy8 bitLen = IsDiv8 bitLen bitLen ~ 'True
@@ -0,0 +1,37 @@+{-# LANGUAGE GeneralizedNewtypeDeriving #-}++-- |+-- Module : Crypto.Internal.Poly1305+-- License : BSD-style+-- Maintainer : Kazu Yamamoto <kazu@iij.ad.jp>+-- Stability : experimental+-- Portability : unknown+--+-- The Poly1305 key with its constructor, for the modules here that build one+-- from bytes whose length they already know. "Crypto.MAC.Poly1305" exports+-- the type without the constructor, so that outside this library a key can+-- only be made by 'key', which checks.+module Crypto.Internal.Poly1305 (+ Key (..),+ key,+) where++import Crypto.Error+import Crypto.Internal.ByteArray (ByteArrayAccess, ScrubbedBytes)+import qualified Crypto.Internal.ByteArray as B+import Crypto.Internal.DeepSeq++-- | A Poly1305 key: thirty-two bytes, and the length is checked here rather+-- than at every use. 'Crypto.MAC.Poly1305.initialize' and+-- 'Crypto.MAC.Poly1305.auth' take one of these and cannot fail, so a caller+-- that holds a key does not carry an error case for a length it already knows+-- is right.+newtype Key = Key ScrubbedBytes+ deriving (ByteArrayAccess, Eq, NFData)++-- | Take thirty-two bytes for a key. A different length is reported as+-- 'CryptoError_MacKeyInvalid'; nothing else about a key can be wrong.+key :: ByteArrayAccess ba => ba -> CryptoFailable Key+key k+ | B.length k /= 32 = CryptoFailed CryptoError_MacKeyInvalid+ | otherwise = CryptoPassed $ Key $ B.convert k
@@ -24,10 +24,10 @@ hash, ) where -import Control.Monad (when) import Crypto.Error import Crypto.Internal.ByteArray (ByteArray, ByteArrayAccess) import qualified Crypto.Internal.ByteArray as B+import Crypto.Internal.Compat (unsafeDoIO) import Data.Word import Foreign.C import Foreign.Ptr@@ -57,17 +57,17 @@ -- | The time cost, which defines the amount of computation realized and therefore the execution time, given in number of iterations. ----- 'FFI.ARGON2_MIN_TIME' <= 'hashIterations' <= 'FFI.ARGON2_MAX_TIME'+-- 'FFI.ARGON2_MIN_TIME' <= 'iterations' <= 'FFI.ARGON2_MAX_TIME' type TimeCost = Word32 -- | The memory cost, which defines the memory usage, given in kibibytes. ----- max 'FFI.ARGON2_MIN_MEMORY' (8 * 'hashParallelism') <= 'hashMemory' <= 'FFI.ARGON2_MAX_MEMORY'+-- max 'FFI.ARGON2_MIN_MEMORY' (8 * 'parallelism') <= 'memory' <= 'FFI.ARGON2_MAX_MEMORY' type MemoryCost = Word32 -- | A parallelism degree, which defines the number of parallel threads. ----- 'FFI.ARGON2_MIN_LANES' <= 'hashParallelism' <= 'FFI.ARGON2_MAX_LANES' && 'FFI.ARGON_MIN_THREADS' <= 'hashParallelism' <= 'FFI.ARGON2_MAX_THREADS'+-- 'FFI.ARGON2_MIN_LANES' <= 'parallelism' <= 'FFI.ARGON2_MAX_LANES' && 'FFI.ARGON_MIN_THREADS' <= 'parallelism' <= 'FFI.ARGON2_MAX_THREADS' type Parallelism = Word32 -- | Parameters that can be adjusted to change the runtime performance of the@@ -104,6 +104,11 @@ , version = Version13 } +-- | Hash a password with Argon2.+--+-- Options the underlying implementation refuses -- iterations, memory or+-- parallelism outside the range it accepts -- are reported as+-- 'CryptoError_ParameterInvalid'. hash :: (ByteArrayAccess password, ByteArrayAccess salt, ByteArray out) => Options@@ -115,22 +120,28 @@ | saltLen < saltMinLength = CryptoFailed CryptoError_SaltTooSmall | outLen < outputMinLength = CryptoFailed CryptoError_OutputLengthTooSmall | outLen > outputMaxLength = CryptoFailed CryptoError_OutputLengthTooBig- | otherwise = CryptoPassed $ B.allocAndFreeze outLen $ \out -> do- res <- B.withByteArray password $ \pPass ->- B.withByteArray salt $ \pSalt ->- argon2_hash- (iterations options)- (memory options)- (parallelism options)- pPass- (csizeOfInt passwordLen)- pSalt- (csizeOfInt saltLen)- out- (csizeOfInt outLen)- (cOfVariant $ variant options)- (cOfVersion $ version options)- when (res /= 0) $ error "argon2: hash: internal error"+ | otherwise = unsafeDoIO $ do+ -- the bounds on iterations, memory and parallelism are checked by the+ -- C implementation, which reports them through its return code+ (res, out) <- B.allocRet outLen $ \pOut ->+ B.withByteArray password $ \pPass ->+ B.withByteArray salt $ \pSalt ->+ argon2_hash+ (iterations options)+ (memory options)+ (parallelism options)+ pPass+ (csizeOfInt passwordLen)+ pSalt+ (csizeOfInt saltLen)+ pOut+ (csizeOfInt outLen)+ (cOfVariant $ variant options)+ (cOfVersion $ version options)+ return $+ if res == 0+ then CryptoPassed out+ else CryptoFailed CryptoError_ParameterInvalid where saltLen = B.length salt passwordLen = B.length password
@@ -37,6 +37,16 @@ -- if passwords are UTF-8 encoded (which they should be) and less than 256 -- characters long. --+-- Only the first 72 bytes of a password are used. The rest is silently+-- ignored, so two passwords sharing a 72-byte prefix produce the same hash and+-- validate against each other. That is what the original implementation does+-- and is kept for compatibility, but it means a longer+-- passphrase buys nothing past that point, and the limit is on /bytes/ rather+-- than characters -- a UTF-8 passphrase reaches it sooner than its length in+-- characters suggests. Where passwords may be longer, hash them to a fixed+-- size first, or use "Crypto.KDF.Argon2" or "Crypto.KDF.Scrypt", which have no+-- such limit.+-- -- The cost parameter can be between 4 and 31 inclusive, but anything less than -- 10 is probably not strong enough. High values may be prohibitively slow -- depending on your hardware. Choose the highest value you can without having@@ -44,22 +54,17 @@ -- depending on the account, since it is unique to an individual hash. module Crypto.KDF.BCrypt ( hashPassword,+ tryHashPassword, validatePassword, validatePasswordEither, bcrypt,+ tryBcrypt, ) where -import Control.Monad (forM_, unless, when)-import Crypto.Cipher.Blowfish.Primitive (- Context,- createKeySchedule,- encrypt,- expandKey,- expandKeyWithSalt,- freezeKeySchedule,- )-import Crypto.Internal.Compat+import Control.Monad (unless, when)+import Crypto.Cipher.Blowfish.Primitive (bcryptHash)+import Crypto.Error import Crypto.Random (MonadRandom, getRandomBytes) import Data.ByteArray ( ByteArray,@@ -78,48 +83,94 @@ -- -- Each increment of the cost approximately doubles the time taken. -- The 16 bytes of random salt will be generated internally.+--+-- A cost outside 4 to 31 raises 'CryptoError_ParameterInvalid';+-- 'tryHashPassword' reports it instead. hashPassword :: (MonadRandom m, ByteArray password, ByteArray hash) => Int- -- ^ The cost parameter. Should be between 4 and 31 (inclusive).- -- Values which lie outside this range will be adjusted accordingly.+ -- ^ The cost parameter. Must be between 4 and 31 inclusive; anything+ -- else is refused. -> password -- ^ The password. Should be the UTF-8 encoded bytes of the password text.+ -- Only the first 72 bytes are used; see the module documentation. -> m hash -- ^ The bcrypt hash in standard format.-hashPassword cost password = do+hashPassword cost password = throwCryptoError <$> tryHashPassword cost password++-- | Create a bcrypt hash for a password with a provided cost value,+-- reporting a cost the implementation refuses rather than raising.+--+-- The salt is generated internally and is always the right length, so the+-- cost is the only thing here that can be wrong.+tryHashPassword+ :: (MonadRandom m, ByteArray password, ByteArray hash)+ => Int+ -- ^ The cost parameter. Must be between 4 and 31 inclusive; anything+ -- else is reported.+ -> password+ -- ^ The password. Should be the UTF-8 encoded bytes of the password text.+ -- Only the first 72 bytes are used; see the module documentation.+ -> m (CryptoFailable hash)+ -- ^ The bcrypt hash in standard format.+tryHashPassword cost password = do salt <- getRandomBytes 16- return $ bcrypt cost (salt :: Bytes) password+ return $ tryBcrypt cost (salt :: Bytes) password -- | Create a bcrypt hash for a password with a provided cost value and salt. ----- Cost value under 4 will be automatically adjusted back to 10 for safety reason.+-- A cost outside 4 to 31, or a salt that is not 16 bytes long, raises+-- 'CryptoError_ParameterInvalid'; 'tryBcrypt' reports the same conditions as+-- 'CryptoFailed'. bcrypt :: (ByteArray salt, ByteArray password, ByteArray output) => Int- -- ^ The cost parameter. Should be between 4 and 31 (inclusive).- -- Values which lie outside this range will be adjusted accordingly.+ -- ^ The cost parameter. Must be between 4 and 31 inclusive; anything+ -- else is refused. -> salt -- ^ The salt. Must be 16 bytes in length or an error will be raised. -> password -- ^ The password. Should be the UTF-8 encoded bytes of the password text.+ -- Only the first 72 bytes are used; see the module documentation. -> output -- ^ The bcrypt hash in standard format.-bcrypt cost salt password = B.concat [header, B.snoc costBytes dollar, b64 salt, b64 hash]+bcrypt cost salt password = throwCryptoError (tryBcrypt cost salt password)++-- | Create a bcrypt hash for a password with a provided cost value and salt,+-- reporting a parameter the implementation refuses rather than raising.+--+-- bcrypt is defined for a cost of 4 to 31, and a cost outside that is+-- reported rather than replaced by one inside it: a caller that asks for+-- something this does not do should hear so, not receive a hash at a cost it+-- did not choose.+tryBcrypt+ :: (ByteArray salt, ByteArray password, ByteArray output)+ => Int+ -- ^ The cost parameter. Must be between 4 and 31 inclusive; anything+ -- else is refused.+ -> salt+ -- ^ The salt. Must be 16 bytes in length.+ -> password+ -- ^ The password. Should be the UTF-8 encoded bytes of the password text.+ -- Only the first 72 bytes are used; see the module documentation.+ -> CryptoFailable output+ -- ^ The bcrypt hash in standard format.+tryBcrypt cost salt password+ | cost < 4 || cost > 31 = CryptoFailed CryptoError_ParameterInvalid+ | B.length salt /= 16 = CryptoFailed CryptoError_ParameterInvalid+ | otherwise =+ CryptoPassed $+ B.concat [header, B.snoc costBytes dollar, b64 salt, b64 hash] where- hash = rawHash 'b' realCost salt password+ hash = rawHash 'b' cost salt password header = B.pack [dollar, fromIntegral (ord '2'), fromIntegral (ord 'b'), dollar] dollar = fromIntegral (ord '$') zero = fromIntegral (ord '0') costBytes = B.pack- [ zero + fromIntegral (realCost `div` 10)- , zero + fromIntegral (realCost `mod` 10)+ [ zero + fromIntegral (cost `div` 10)+ , zero + fromIntegral (cost `mod` 10) ]- realCost- | cost < 4 = 10 -- 4 is virtually pointless so go for 10- | cost > 31 = 31- | otherwise = cost b64 :: ByteArray ba => ba -> ba b64 = convertToBase Base64OpenBSD@@ -128,6 +179,9 @@ -- -- Returns @False@ if the password doesn't match the hash, or if the hash is -- invalid or an unsupported version.+--+-- Only the first 72 bytes of the password are compared; see the module+-- documentation. validatePassword :: (ByteArray password, ByteArray hash) => password -> hash -> Bool validatePassword password bcHash = either (const False) id (validatePasswordEither password bcHash)@@ -135,7 +189,7 @@ -- | Check a password against a bcrypt hash -- -- As for @validatePassword@ but will provide error information if the hash is invalid or--- an unsupported version.+-- an unsupported version. The same 72-byte limit applies. validatePasswordEither :: (ByteArray password, ByteArray hash) => password -> hash -> Either String Bool validatePasswordEither password bcHash = do@@ -145,47 +199,12 @@ rawHash :: (ByteArrayAccess salt, ByteArray password, ByteArray output) => Char -> Int -> salt -> password -> output-rawHash _ cost salt password = B.take 23 hash -- Another compatibility bug. Ignore last byte of hash+rawHash _ cost salt password = case bcryptHash cost salt key of+ Just hash -> B.take 23 hash -- Another compatibility bug. Ignore last byte of hash+ Nothing -> error "bcrypt: the cost or the salt is not one bcrypt takes" where- hash = loop (0 :: Int) orpheanBeholder-- loop i input- | i < 64 = loop (i + 1) (encrypt ctx input)- | otherwise = input- -- Truncate the password if necessary and append a null byte for C compatibility- key = B.snoc (B.take 72 password) 0-- ctx = expensiveBlowfishContext key salt cost-- -- The BCrypt plaintext: "OrpheanBeholderScryDoubt"- orpheanBeholder =- B.pack- [ 79- , 114- , 112- , 104- , 101- , 97- , 110- , 66- , 101- , 104- , 111- , 108- , 100- , 101- , 114- , 83- , 99- , 114- , 121- , 68- , 111- , 117- , 98- , 116- ]+ key = B.snoc (B.take 72 (B.convert password :: Bytes)) 0 -- "$2a$10$XajjQvNhvvRt5GSeFk1xFeyqRrsxkhBkUiQeg0dt.wU1qD4aFDcga" parseBCryptHash :: ByteArray ba => ba -> Either String BCryptHash@@ -217,20 +236,3 @@ salt <- convertFromBase Base64OpenBSD s hash <- convertFromBase Base64OpenBSD h return (salt, hash)---- | Create a key schedule for the BCrypt "EKS" version.------ Salt must be a 128-bit byte array.--- Cost must be between 4 and 31 inclusive--- See <https://www.usenix.org/conference/1999-usenix-annual-technical-conference/future-adaptable-password-scheme>-expensiveBlowfishContext- :: (ByteArrayAccess key, ByteArrayAccess salt) => key -> salt -> Int -> Context-expensiveBlowfishContext keyBytes saltBytes cost- | B.length saltBytes /= 16 = error "bcrypt salt must be 16 bytes"- | otherwise = unsafeDoIO $ do- ks <- createKeySchedule- expandKeyWithSalt ks keyBytes saltBytes- forM_ [1 .. 2 ^ cost :: Int] $ \_ -> do- expandKey ks keyBytes- expandKey ks saltBytes- freezeKeySchedule ks
@@ -9,14 +9,16 @@ module Crypto.KDF.BCryptPBKDF ( Parameters (..), generate,+ tryGenerate, hashInternal,+ tryHashInternal, ) where import qualified Control.Exception as E import Control.Monad (when)-import qualified Crypto.Cipher.Blowfish.Box as Blowfish-import qualified Crypto.Cipher.Blowfish.Primitive as Blowfish+import Crypto.Cipher.Blowfish.Primitive (bcryptPbkdfHash)+import Crypto.Error import Crypto.Hash.Algorithms (SHA512 (..)) import Crypto.Hash.Types ( Context,@@ -48,17 +50,30 @@ deriving (Eq, Ord, Show) -- | Derive a key of specified length using the bcrypt_pbkdf algorithm.+--+-- Parameters outside the ranges documented for t'Parameters' raise+-- 'CryptoError_ParameterInvalid'; 'tryGenerate' reports the same condition as+-- 'CryptoFailed'. generate :: (B.ByteArray pass, B.ByteArray salt, B.ByteArray output) => Parameters -> pass -> salt -> output-generate params pass salt- | iterCounts params < 1 = error "BCryptPBKDF: iterCounts must be > 0"- | keyLen < 1 || keyLen > 1024 =- error "BCryptPBKDF: outputLength must be in 1..1024"- | otherwise = B.unsafeCreate keyLen deriveKey+generate params pass salt = throwCryptoError (tryGenerate params pass salt)++-- | Derive a key of specified length using the bcrypt_pbkdf algorithm,+-- reporting parameters the implementation refuses rather than raising.+tryGenerate+ :: (B.ByteArray pass, B.ByteArray salt, B.ByteArray output)+ => Parameters+ -> pass+ -> salt+ -> CryptoFailable output+tryGenerate params pass salt+ | iterCounts params < 1 = CryptoFailed CryptoError_ParameterInvalid+ | keyLen < 1 || keyLen > 1024 = CryptoFailed CryptoError_ParameterInvalid+ | otherwise = CryptoPassed $ B.unsafeCreate keyLen deriveKey where outLen, tmpLen, blkLen, keyLen, passLen, saltLen, ctxLen, hashLen, blocks :: Int outLen = 32@@ -76,8 +91,6 @@ -- Allocate all necessary memory. The algorithm shall not allocate -- any more dynamic memory after this point. ForeignPtrs allocate -- pinned memory, so raw pointers to them are stable.- ksClean <- Blowfish.createKeySchedule- ksDirty <- Blowfish.createKeySchedule ctxFP <- mallocForeignPtrBytes ctxLen :: IO (ForeignPtr Word8) outFP <- mallocForeignPtrBytes outLen :: IO (ForeignPtr Word8) tmpFP <- mallocForeignPtrBytes tmpLen :: IO (ForeignPtr Word8)@@ -116,8 +129,7 @@ hashInternalUpdate shaPtr blkPtr (fromIntegral blkLen) hashInternalFinalize shaPtr (castPtr saltHashPtr) let saltHashBS = BSI.fromForeignPtr saltHashFP 0 hashLen- Blowfish.copyKeySchedule ksDirty ksClean- hashInternalMutable ksDirty passHashBS saltHashBS tmpPtr+ hashInternalMutable passHashBS saltHashBS tmpPtr memCopy outPtr tmpPtr outLen -- Remaining rounds. forM_ [2 .. iterCounts params] $ const $ do@@ -125,8 +137,7 @@ hashInternalUpdate shaPtr tmpPtr (fromIntegral tmpLen) hashInternalFinalize shaPtr (castPtr saltHashPtr) let saltHashBS2 = BSI.fromForeignPtr saltHashFP 0 hashLen- Blowfish.copyKeySchedule ksDirty ksClean- hashInternalMutable ksDirty passHashBS saltHashBS2 tmpPtr+ hashInternalMutable passHashBS saltHashBS2 tmpPtr memXor outPtr outPtr tmpPtr outLen -- Spread the current out buffer evenly over the key buffer. -- After both loops have run every byte of the key buffer@@ -141,49 +152,40 @@ -- | Internal hash function used by `generate`. -- -- Normal users should not need this.+--+-- Inputs that are not 512 bits long raise 'CryptoError_ParameterInvalid';+-- 'tryHashInternal' reports the same condition as 'CryptoFailed'. hashInternal :: (B.ByteArrayAccess pass, B.ByteArrayAccess salt, B.ByteArray output) => pass -> salt -> output-hashInternal passHash saltHash- | B.length passHash /= 64 = error "passHash must be 512 bits"- | B.length saltHash /= 64 = error "saltHash must be 512 bits"- | otherwise = unsafeDoIO $ do- ks0 <- Blowfish.createKeySchedule- B.alloc 32 $ \outPtr -> hashInternalMutable ks0 passHash saltHash outPtr+hashInternal passHash saltHash =+ throwCryptoError (tryHashInternal passHash saltHash) +-- | Internal hash function used by 'tryGenerate', reporting inputs the+-- implementation refuses rather than raising.+--+-- Normal users should not need this.+tryHashInternal+ :: (B.ByteArrayAccess pass, B.ByteArrayAccess salt, B.ByteArray output)+ => pass+ -> salt+ -> CryptoFailable output+tryHashInternal passHash saltHash+ | B.length passHash /= 64 = CryptoFailed CryptoError_ParameterInvalid+ | B.length saltHash /= 64 = CryptoFailed CryptoError_ParameterInvalid+ | otherwise = CryptoPassed $ unsafeDoIO $ do+ B.alloc 32 $ \outPtr -> hashInternalMutable passHash saltHash outPtr+ hashInternalMutable :: (B.ByteArrayAccess pass, B.ByteArrayAccess salt)- => Blowfish.KeySchedule- -> pass+ => pass -> salt -> Ptr Word8 -> IO ()-hashInternalMutable bfks passHash saltHash outPtr = do- Blowfish.expandKeyWithSalt bfks passHash saltHash- forM_ [0 .. 63 :: Int] $ const $ do- Blowfish.expandKey bfks saltHash- Blowfish.expandKey bfks passHash- -- "OxychromaticBlowfishSwatDynamite" represented as 4 Word64 in big-endian.- store 0 =<< cipher 64 0x4f78796368726f6d- store 8 =<< cipher 64 0x61746963426c6f77- store 16 =<< cipher 64 0x6669736853776174- store 24 =<< cipher 64 0x44796e616d697465- where- store :: Int -> Word64 -> IO ()- store o w64 = do- pokeByteOff outPtr (o + 0) (fromIntegral (w64 `shiftR` 32) :: Word8)- pokeByteOff outPtr (o + 1) (fromIntegral (w64 `shiftR` 40) :: Word8)- pokeByteOff outPtr (o + 2) (fromIntegral (w64 `shiftR` 48) :: Word8)- pokeByteOff outPtr (o + 3) (fromIntegral (w64 `shiftR` 56) :: Word8)- pokeByteOff outPtr (o + 4) (fromIntegral (w64 `shiftR` 0) :: Word8)- pokeByteOff outPtr (o + 5) (fromIntegral (w64 `shiftR` 8) :: Word8)- pokeByteOff outPtr (o + 6) (fromIntegral (w64 `shiftR` 16) :: Word8)- pokeByteOff outPtr (o + 7) (fromIntegral (w64 `shiftR` 24) :: Word8)- cipher :: Int -> Word64 -> IO Word64- cipher 0 block = return block- cipher i block = Blowfish.cipherBlockMutable bfks block >>= cipher (i - 1)+hashInternalMutable passHash saltHash outPtr =+ bcryptPbkdfHash passHash saltHash outPtr finallyErase :: ForeignPtr Word8 -> Int -> IO () -> IO () finallyErase fp len action =
@@ -1,4 +1,5 @@ {-# LANGUAGE BangPatterns #-}+{-# LANGUAGE ScopedTypeVariables #-} -- | -- Module : Crypto.KDF.HKDF@@ -15,9 +16,11 @@ extract, extractSkip, expand,+ tryExpand, toPRK, ) where +import Crypto.Error import Crypto.Hash import Crypto.Internal.ByteArray ( ByteArray,@@ -60,8 +63,13 @@ extractSkip ikm = PRK_NoExpand $ B.convert ikm -- | Expand key material of specific length out of the parameters+--+-- Requests exceeding the RFC 5869 limit of @255 * HashLen@ raise+-- 'CryptoError_OutputLengthTooBig'; 'tryExpand' reports the same condition as+-- 'CryptoFailed'. expand- :: (HashAlgorithm a, ByteArrayAccess info, ByteArray out)+ :: forall a info out+ . (HashAlgorithm a, ByteArrayAccess info, ByteArray out) => PRK a -- ^ Pseudo Random Key -> info@@ -71,8 +79,27 @@ -> out -- ^ Output data expand prkAt infoAt outputLength =- let hF = hFGet prkAt- in B.concat $ loop hF B.empty outputLength 1+ throwCryptoError (tryExpand prkAt infoAt outputLength)++-- | Expand key material of specific length out of the parameters, reporting a+-- length the RFC refuses rather than raising.+tryExpand+ :: forall a info out+ . (HashAlgorithm a, ByteArrayAccess info, ByteArray out)+ => PRK a+ -- ^ Pseudo Random Key+ -> info+ -- ^ Optional context and application specific information+ -> Int+ -- ^ Output length in bytes+ -> CryptoFailable out+ -- ^ Output data+tryExpand prkAt infoAt outputLength+ | outputLength > 255 * hashDigestSize (undefined :: a) =+ CryptoFailed CryptoError_OutputLengthTooBig+ | otherwise =+ let hF = hFGet prkAt+ in CryptoPassed $ B.concat $ loop hF B.empty outputLength 1 where hFGet :: (HashAlgorithm a, ByteArrayAccess b) => PRK a -> (b -> HMAC a) hFGet prk = case prk of
@@ -14,9 +14,13 @@ prfHMAC, Parameters (..), generate,+ tryGenerate, fastPBKDF2_SHA1,+ tryFastPBKDF2_SHA1, fastPBKDF2_SHA256,+ tryFastPBKDF2_SHA256, fastPBKDF2_SHA512,+ tryFastPBKDF2_SHA512, ) where import Data.Bits@@ -25,6 +29,7 @@ import Foreign.Marshal.Alloc import Foreign.Ptr (Ptr, plusPtr) +import Crypto.Error import Crypto.Hash (HashAlgorithm) import qualified Crypto.MAC.HMAC as HMAC @@ -55,11 +60,27 @@ data Parameters = Parameters { iterCounts :: Int -- ^ the number of user-defined iterations for the algorithms. e.g. WPA2 uses 4000.+ -- (must be > 0) , outputLength :: Int -- ^ the number of bytes to generate out of PBKDF2+ -- (must not be negative) } +-- | Report parameters no PBKDF2 entry point accepts.+--+-- An iteration count below one derives a key that is not a key at all, and a+-- negative output length asks for a buffer that cannot be allocated.+validateParameters :: Parameters -> Maybe CryptoError+validateParameters params+ | iterCounts params < 1 = Just CryptoError_ParameterInvalid+ | outputLength params < 0 = Just CryptoError_ParameterInvalid+ | otherwise = Nothing+ -- | generate the pbkdf2 key derivation function from the output+--+-- Parameters outside the ranges documented for t'Parameters' raise+-- 'CryptoError_ParameterInvalid'; 'tryGenerate' reports the same condition as+-- 'CryptoFailed'. generate :: (ByteArrayAccess password, ByteArrayAccess salt, ByteArray ba) => PRF password@@ -68,7 +89,20 @@ -> salt -> ba generate prf params password salt =- B.allocAndFreeze (outputLength params) $ \p -> do+ throwCryptoError (tryGenerate prf params password salt)++-- | generate the pbkdf2 key derivation function from the output, reporting+-- parameters the implementation refuses rather than raising.+tryGenerate+ :: (ByteArrayAccess password, ByteArrayAccess salt, ByteArray ba)+ => PRF password+ -> Parameters+ -> password+ -> salt+ -> CryptoFailable ba+tryGenerate prf params password salt+ | Just err <- validateParameters params = CryptoFailed err+ | otherwise = CryptoPassed $ B.allocAndFreeze (outputLength params) $ \p -> do memSet p 0 (outputLength params) loop 1 (outputLength params) p where@@ -113,8 +147,13 @@ b = fromIntegral ((w `shiftR` 16) .&. 0xff) c = fromIntegral ((w `shiftR` 8) .&. 0xff) d = fromIntegral (w .&. 0xff)-{-# NOINLINE generate #-}+{-# NOINLINE tryGenerate #-} +-- | PBKDF2 with HMAC-SHA1, using the bundled C implementation.+--+-- Parameters outside the ranges documented for t'Parameters' raise+-- 'CryptoError_ParameterInvalid'; 'tryFastPBKDF2_SHA1' reports the same condition+-- as 'CryptoFailed'. fastPBKDF2_SHA1 :: (ByteArrayAccess password, ByteArrayAccess salt, ByteArray out) => Parameters@@ -122,7 +161,19 @@ -> salt -> out fastPBKDF2_SHA1 params password salt =- B.allocAndFreeze (outputLength params) $ \outPtr ->+ throwCryptoError (tryFastPBKDF2_SHA1 params password salt)++-- | PBKDF2 with HMAC-SHA1, reporting parameters the implementation refuses+-- rather than raising.+tryFastPBKDF2_SHA1+ :: (ByteArrayAccess password, ByteArrayAccess salt, ByteArray out)+ => Parameters+ -> password+ -> salt+ -> CryptoFailable out+tryFastPBKDF2_SHA1 params password salt+ | Just err <- validateParameters params = CryptoFailed err+ | otherwise = CryptoPassed $ B.allocAndFreeze (outputLength params) $ \outPtr -> B.withByteArray password $ \passPtr -> B.withByteArray salt $ \saltPtr -> c_crypton_fastpbkdf2_hmac_sha1@@ -134,6 +185,11 @@ outPtr (fromIntegral $ outputLength params) +-- | PBKDF2 with HMAC-SHA256, using the bundled C implementation.+--+-- Parameters outside the ranges documented for t'Parameters' raise+-- 'CryptoError_ParameterInvalid'; 'tryFastPBKDF2_SHA256' reports the same condition+-- as 'CryptoFailed'. fastPBKDF2_SHA256 :: (ByteArrayAccess password, ByteArrayAccess salt, ByteArray out) => Parameters@@ -141,7 +197,19 @@ -> salt -> out fastPBKDF2_SHA256 params password salt =- B.allocAndFreeze (outputLength params) $ \outPtr ->+ throwCryptoError (tryFastPBKDF2_SHA256 params password salt)++-- | PBKDF2 with HMAC-SHA256, reporting parameters the implementation refuses+-- rather than raising.+tryFastPBKDF2_SHA256+ :: (ByteArrayAccess password, ByteArrayAccess salt, ByteArray out)+ => Parameters+ -> password+ -> salt+ -> CryptoFailable out+tryFastPBKDF2_SHA256 params password salt+ | Just err <- validateParameters params = CryptoFailed err+ | otherwise = CryptoPassed $ B.allocAndFreeze (outputLength params) $ \outPtr -> B.withByteArray password $ \passPtr -> B.withByteArray salt $ \saltPtr -> c_crypton_fastpbkdf2_hmac_sha256@@ -153,6 +221,11 @@ outPtr (fromIntegral $ outputLength params) +-- | PBKDF2 with HMAC-SHA512, using the bundled C implementation.+--+-- Parameters outside the ranges documented for t'Parameters' raise+-- 'CryptoError_ParameterInvalid'; 'tryFastPBKDF2_SHA512' reports the same condition+-- as 'CryptoFailed'. fastPBKDF2_SHA512 :: (ByteArrayAccess password, ByteArrayAccess salt, ByteArray out) => Parameters@@ -160,7 +233,19 @@ -> salt -> out fastPBKDF2_SHA512 params password salt =- B.allocAndFreeze (outputLength params) $ \outPtr ->+ throwCryptoError (tryFastPBKDF2_SHA512 params password salt)++-- | PBKDF2 with HMAC-SHA512, reporting parameters the implementation refuses+-- rather than raising.+tryFastPBKDF2_SHA512+ :: (ByteArrayAccess password, ByteArrayAccess salt, ByteArray out)+ => Parameters+ -> password+ -> salt+ -> CryptoFailable out+tryFastPBKDF2_SHA512 params password salt+ | Just err <- validateParameters params = CryptoFailed err+ | otherwise = CryptoPassed $ B.allocAndFreeze (outputLength params) $ \outPtr -> B.withByteArray password $ \passPtr -> B.withByteArray salt $ \saltPtr -> c_crypton_fastpbkdf2_hmac_sha512
@@ -14,6 +14,7 @@ module Crypto.KDF.Scrypt ( Parameters (..), generate,+ tryGenerate, ) where import Control.Monad (forM_)@@ -21,6 +22,7 @@ import Foreign.Marshal.Alloc import Foreign.Ptr (Ptr, plusPtr) +import Crypto.Error import Crypto.Hash (SHA256 (..)) import Crypto.Internal.ByteArray (ByteArray, ByteArrayAccess) import qualified Crypto.Internal.ByteArray as B@@ -44,18 +46,31 @@ :: Ptr Word8 -> Word32 -> Word64 -> Ptr Word8 -> Ptr Word8 -> IO () -- | Generate the scrypt key derivation data+--+-- Parameters the implementation refuses raise a 'CryptoError'; 'tryGenerate'+-- reports the same condition as 'CryptoFailed'. generate :: (ByteArrayAccess password, ByteArrayAccess salt, ByteArray output) => Parameters -> password -> salt -> output-generate params password salt- | r params * p params >= 0x40000000 =- error "Scrypt: invalid parameters: r and p constraint"- | popCount (n params) /= 1 =- error "Scrypt: invalid parameters: n not a power of 2"- | otherwise = unsafeDoIO $ do+generate params password salt = throwCryptoError (tryGenerate params password salt)++-- | Generate the scrypt key derivation data, reporting parameters the+-- implementation refuses rather than raising.+--+-- @n@ has to be a power of two, and @r@ times @p@ has to stay below 2^30.+tryGenerate+ :: (ByteArrayAccess password, ByteArrayAccess salt, ByteArray output)+ => Parameters+ -> password+ -> salt+ -> CryptoFailable output+tryGenerate params password salt+ | r params * p params >= 0x40000000 = CryptoFailed CryptoError_ParameterInvalid+ | popCount (n params) /= 1 = CryptoFailed CryptoError_ParameterInvalid+ | otherwise = CryptoPassed $ unsafeDoIO $ do let b = PBKDF2.generate prf (PBKDF2.Parameters 1 intLen) password salt :: B.Bytes newSalt <- B.copy b $ \bPtr -> allocaBytesAligned (128 * (fromIntegral $ n params) * (r params)) 8 $ \v ->@@ -77,4 +92,4 @@ where prf = PBKDF2.prfHMAC SHA256 intLen = p params * 128 * r params-{-# NOINLINE generate #-}+{-# NOINLINE tryGenerate #-}
@@ -0,0 +1,121 @@+-- |+-- Module : Crypto.KEM+-- License : BSD-style+-- Maintainer : Kazu Yamamoto <kazu@iij.ad.jp>+-- Stability : experimental+-- Portability : unknown+--+-- Key encapsulation: one side publishes a key, the other draws a secret and+-- returns something only the first side can turn back into it.+--+-- The shape is not ML-KEM's alone, which is why this is a class in a module+-- of its own rather than part of any one algorithm: 'Crypto.PubKey.MLKEM'+-- is one instance of it, and a caller that does not care which it has can+-- be written against this.+--+-- A bare Diffie-Hellman exchange is deliberately __not__ an instance, even+-- though the shapes line up. The KEM a Diffie-Hellman group gives is+-- DHKEM, of RFC 9180 section 4.1, and that is not the raw exchange: its+-- shared secret is the exchange's output run through HKDF with the+-- ephemeral and the recipient public keys as context, under labels that+-- name the ciphersuite. The binding to those two keys, and the separation+-- between suites, are what the KEM security argument rests on; the raw+-- value has neither. A protocol can supply the binding at its own level --+-- TLS 1.3 does, in the key schedule -- but then the binding belongs to the+-- protocol, not to this class, and offering the raw exchange here would+-- invite its use somewhere that supplies nothing.+--+-- DHKEM proper is an instance, and it is not here either: the labels it+-- derives under carry the HPKE ciphersuite identifier, which is an IANA+-- registry value rather than anything a primitive knows, so the instances+-- live beside the registry in the @hpke@ package.+{-# LANGUAGE GeneralizedNewtypeDeriving #-}+{-# LANGUAGE TypeFamilies #-}++module Crypto.KEM (+ KEM (..),+ SharedSecret (..),+) where++import Data.Kind (Type)++import Crypto.Error (CryptoFailable)+import Crypto.Internal.ByteArray (ByteArrayAccess, ScrubbedBytes)+import Crypto.Internal.Imports+import Crypto.Random (MonadRandom)++-- | Secret shared via key exchange.+newtype SharedSecret = SharedSecret ScrubbedBytes+ deriving (Eq, ByteArrayAccess, NFData)++instance Show SharedSecret where+ show _ = "SharedSecret <redacted>"++instance Semigroup SharedSecret where+ SharedSecret x <> SharedSecret y = SharedSecret (x <> y)++instance Monoid SharedSecret where+ mempty = SharedSecret mempty++-- | A key encapsulation mechanism.+--+-- The three values are named for what they do rather than for what they are+-- in any one algorithm. In ML-KEM the encapsulation key and the ciphertext+-- are what the names say; in a Diffie-Hellman exchange both are public+-- values of the group, and the \"ciphertext\" is the ephemeral one the+-- encapsulating side generates. Keeping them apart in the types is what+-- stops one being passed where the other belongs, which they are not+-- interchangeable for even where they have the same representation.+class KEM kem where+ -- | What the encapsulating side is given.+ type EncapsulationKey kem :: Type++ -- | What the decapsulating side keeps.+ type DecapsulationKey kem :: Type++ -- | What travels back, and is decapsulated.+ type Ciphertext kem :: Type++ -- | The randomness 'encapsulate' draws, for the instances that let a+ -- caller supply it. In ML-KEM it is @m@ of FIPS 203, a string of+ -- bytes; in DHKEM it is the ephemeral secret key, a scalar of the+ -- group. Both are secret, and both determine the shared secret+ -- completely.+ type Coins kem :: Type++ -- | Generate a key pair for the decapsulating side.+ generateKeyPair+ :: MonadRandom m+ => proxy kem -> m (EncapsulationKey kem, DecapsulationKey kem)++ -- | Draw a secret and encapsulate it against the key.+ --+ -- This can fail, and does for some instances: a Diffie-Hellman exchange+ -- refuses a peer value that would make the secret degenerate, where+ -- ML-KEM has nothing to refuse.+ encapsulate+ :: MonadRandom m+ => proxy kem+ -> EncapsulationKey kem+ -> m (CryptoFailable (Ciphertext kem, SharedSecret))++ -- | Encapsulate with the randomness supplied rather than drawn.+ --+ -- The secret this produces is a deterministic function of the key and+ -- these coins, so they must come from a source no other party can+ -- predict or repeat, and must not be used twice. 'encapsulate' is the+ -- entry point for ordinary use; this one is for test vectors, and for a+ -- protocol that has to name the ephemeral value it used -- HPKE lets a+ -- sender supply its own, and the RFC 9180 vectors are written that way.+ encapsulateWith+ :: proxy kem+ -> EncapsulationKey kem+ -> Coins kem+ -> CryptoFailable (Ciphertext kem, SharedSecret)++ -- | Recover the secret.+ decapsulate+ :: proxy kem+ -> DecapsulationKey kem+ -> Ciphertext kem+ -> CryptoFailable SharedSecret
@@ -1,3 +1,4 @@+{-# LANGUAGE BangPatterns #-} {-# LANGUAGE GeneralizedNewtypeDeriving #-} -- |@@ -17,11 +18,12 @@ ) where import Data.Bits (setBit, shiftL, testBit)-import Data.List (foldl')+import Data.ByteString (ByteString)+import qualified Data.ByteString as S import Data.Word-import Prelude hiding (foldl') import Crypto.Cipher.Types+import Crypto.Cipher.Types.Block (IV (..)) import Crypto.Internal.ByteArray (ByteArray, ByteArrayAccess, Bytes) import qualified Crypto.Internal.ByteArray as B @@ -41,28 +43,46 @@ -- ^ input message -> CMAC cipher -- ^ output tag-cmac k msg =- CMAC $ foldl' (\c m -> ecbEncrypt k $ bxor c m) zeroV ms+cmac k msg = CMAC $ B.convert $ step (chain zeroV whole) final where bytes = blockSize k- zeroV = B.replicate bytes 0 :: Bytes+ zeroV = S.replicate bytes 0 (k1, k2) = subKeys k- ms = cmacChunks k k1 k2 $ B.convert msg -cmacChunks :: (BlockCipher k, ByteArray ba) => k -> ba -> ba -> ba -> [ba]-cmacChunks k k1 k2 = rec'- where- rec' msg- | B.null tl =- if lack == 0- then [bxor k1 hd]- else [bxor k2 $ hd `B.append` B.pack (0x80 : replicate (lack - 1) 0)]- | otherwise = hd : rec' tl- where- bytes = blockSize k- (hd, tl) = B.splitAt bytes msg- lack = bytes - B.length hd+ -- The message is held as a ByteString and sliced, never consumed. 'Bytes'+ -- has no shared representation, so splitting one repeatedly -- which is+ -- what this used to do, once per block -- copied whatever was left of the+ -- message each time, and so the message about n/2 times in all.+ msgBytes = B.convert msg :: ByteString+ msgLen = S.length msgBytes + -- the last block is the one the subkeys are for, and it is a whole block+ -- only when there is one to be had+ lastLen+ | msgLen > 0 && msgLen `mod` bytes == 0 = bytes+ | otherwise = msgLen `mod` bytes+ (whole, rest) = S.splitAt (msgLen - lastLen) msgBytes+ final+ | lastLen == bytes = bxor k1 rest+ | otherwise =+ bxor k2 $+ S.concat [rest, S.singleton 0x80, S.replicate (bytes - lastLen - 1) 0]++ -- CMAC chains its blocks the way CBC does, so the running state is the+ -- last ciphertext block of a CBC encryption. Handing the cipher a chunk+ -- at a time rather than a block at a time is what makes that worth saying:+ -- for AES it reaches the C implementation of CBC, where a block at a time+ -- reached a foreign call per sixteen bytes.+ chunkBytes = bytes * 2048+ chain !c bs+ | S.null bs = c+ | otherwise =+ let (hd, tl) = S.splitAt chunkBytes bs+ out = cbcEncrypt k (IV c) hd+ in chain (S.drop (S.length hd - bytes) out) tl++ step c m = ecbEncrypt k (bxor c m) :: ByteString+ -- | make sub-keys used in CMAC subKeys :: (BlockCipher k, ByteArray ba)@@ -102,7 +122,7 @@ sl1 = shiftL x 1 bxor :: ByteArray ba => ba -> ba -> ba-bxor = B.xor+bxor = B.bxor -----
@@ -63,7 +63,7 @@ hmacLazy secret msg = finalize $ updates (initialize secret) (L.toChunks msg) -- | Represent an ongoing HMAC state, that can be appended with 'update'--- and finalize to an HMAC with 'hmacFinalize'+-- and finalize to an HMAC with 'finalize' data Context hashalg = Context !(Hash.Context hashalg) !(Hash.Context hashalg) -- | Initialize a new incremental HMAC context
@@ -99,7 +99,7 @@ kmac str key msg = finalize $ updates (initialize str key) [msg] -- | Represent an ongoing KMAC state, that can be appended with 'update' and--- finalized to a 'KMAC' with 'finalize'.+-- finalized to a t'KMAC' with 'finalize'. newtype Context a = Context (H.Context a) -- | Initialize a new incremental KMAC context with the supplied customization
@@ -48,7 +48,7 @@ KeyedBlake2 x == KeyedBlake2 y = B.constEq x y -- | Represent an ongoing Blake2 state, that can be appended with 'update' and--- finalized to a 'KeyedBlake2' with 'finalize'.+-- finalized to a t'KeyedBlake2' with 'finalize'. newtype Context a = Context (H.Context a) -- | Initialize a new incremental keyed Blake2 context with the supplied key.
@@ -12,6 +12,8 @@ module Crypto.MAC.Poly1305 ( Ctx, State,+ Key,+ key, Auth (..), authTag, @@ -33,6 +35,7 @@ ) import qualified Crypto.Internal.ByteArray as B import Crypto.Internal.DeepSeq+import Crypto.Internal.Poly1305 (Key (..), key) import Data.Word import Foreign.C.Types import Foreign.Ptr@@ -63,6 +66,12 @@ instance Eq Auth where (Auth a1) == (Auth a2) = B.constEq a1 a2 +-- | @sizeof(poly1305_ctx)@: the accumulator and the key, either as the+-- limbs the C implementation works in or as the state the assembly keeps,+-- and the buffer for a partial block. See @cbits/crypton_poly1305.h@.+sizeCtx :: Int+sizeCtx = 232+ foreign import ccall unsafe "crypton_poly1305.h crypton_poly1305_init" c_poly1305_init :: Ptr State -> Ptr Word8 -> IO () @@ -73,22 +82,20 @@ c_poly1305_finalize :: Ptr Word8 -> Ptr State -> IO () -- | initialize a Poly1305 context-initialize- :: ByteArrayAccess key- => key- -> CryptoFailable State-initialize key- | B.length key /= 32 = CryptoFailed $ CryptoError_MacKeyInvalid- | otherwise = CryptoPassed $ State $ B.allocAndFreeze 84 $ \ctxPtr ->- B.withByteArray key $ \keyPtr ->- c_poly1305_init (castPtr ctxPtr) keyPtr+initialize :: Key -> State+initialize k = State $ B.allocAndFreeze sizeCtx $ \ctxPtr ->+ B.withByteArray k $ \keyPtr ->+ c_poly1305_init (castPtr ctxPtr) keyPtr {-# NOINLINE initialize #-} -- | update a context with a bytestring update :: ByteArrayAccess ba => State -> ba -> State update (State prevCtx) d = State $ B.copyAndFreeze prevCtx $ \ctxPtr -> B.withByteArray d $ \dataPtr ->- c_poly1305_update (castPtr ctxPtr) dataPtr (fromIntegral $ B.length d)+ -- in pieces the C's uint32_t length can hold; the context carries+ -- across, so it can simply be called again+ B.inCLengths (B.length d) $ \off n ->+ c_poly1305_update (castPtr ctxPtr) (dataPtr `plusPtr` off) (fromIntegral n) {-# NOINLINE update #-} -- | updates a context with multiples bytestring@@ -97,7 +104,9 @@ where loop [] _ = return () loop (x : xs) ctxPtr = do- B.withByteArray x $ \dataPtr -> c_poly1305_update ctxPtr dataPtr (fromIntegral $ B.length x)+ B.withByteArray x $ \dataPtr ->+ B.inCLengths (B.length x) $ \off n ->+ c_poly1305_update ctxPtr (dataPtr `plusPtr` off) (fromIntegral n) loop xs ctxPtr {-# NOINLINE updates #-} @@ -111,16 +120,18 @@ {-# NOINLINE finalize #-} -- | One-pass authorization creation-auth :: (ByteArrayAccess key, ByteArrayAccess ba) => key -> ba -> Auth-auth key d- | B.length key /= 32 = error "Poly1305: key length expected 32 bytes"- | otherwise = Auth $ B.allocAndFreeze 16 $ \dst -> do- _ <- B.alloc 84 (onCtx dst) :: IO ScrubbedBytes- return ()+auth :: ByteArrayAccess ba => Key -> ba -> Auth+auth k d = Auth $ B.allocAndFreeze 16 $ \dst -> do+ _ <- B.alloc sizeCtx (onCtx dst) :: IO ScrubbedBytes+ return () where onCtx dst ctxPtr =- B.withByteArray key $ \keyPtr -> do+ B.withByteArray k $ \keyPtr -> do c_poly1305_init (castPtr ctxPtr) keyPtr B.withByteArray d $ \dataPtr ->- c_poly1305_update (castPtr ctxPtr) dataPtr (fromIntegral $ B.length d)+ B.inCLengths (B.length d) $ \off n ->+ c_poly1305_update+ (castPtr ctxPtr)+ (dataPtr `plusPtr` off)+ (fromIntegral n) c_poly1305_finalize dst (castPtr ctxPtr)
@@ -56,7 +56,7 @@ -- | Get the extended GCD of two integer using integer divMod ----- gcde 'a' 'b' find (x,y,gcd(a,b)) where ax + by = d+-- gcde @a@ @b@ find (x,y,gcd(a,b)) where ax + by = d gcde :: Integer -> Integer -> (Integer, Integer, Integer) gcde a b = onGmpUnsupported (gmpGcde a b) $@@ -85,7 +85,12 @@ -- | Compute the number of bits for an integer numBits :: Integer -> Int-numBits n = gmpSizeInBits n `onGmpUnsupported` (if n == 0 then 1 else computeBits 0 n)+-- GMP sizes the magnitude and calls zero zero bits, and every caller here --+-- 'numBytes' above all -- is written against that. The fallback used to+-- answer 1 for zero, and to divide a negative number by 256 forever, because+-- the quotient never reaches zero.+numBits n =+ gmpSizeInBits n `onGmpUnsupported` (if n == 0 then 0 else computeBits 0 (abs n)) where computeBits !acc i | q == 0 =@@ -118,8 +123,15 @@ (q, r) = i `divMod` 256 -- | Compute the number of bytes for an integer+--+-- Out of 'numBits' rather than out of GMP's own count in base 256. GHC's+-- bignum sizes a number in base two by looking at its highest limb, and in+-- any other base -- 256 included -- by dividing the number down to nothing:+-- on a 2048-bit modulus that is 1.65 us against 0.01, and every serialization+-- here asks for the size before it allocates. Eight bits to the byte does+-- the rest. numBytes :: Integer -> Int-numBytes n = gmpSizeInBytes n `onGmpUnsupported` ((numBits n + 7) `div` 8)+numBytes n = (numBits n + 7) `div` 8 -- | Express an integer as an odd number and a power of 2 asPowerOf2AndOdd :: Integer -> (Int, Integer)
@@ -7,7 +7,7 @@ -- -- This module provides basic arithmetic operations over F₂m. Performance is -- not optimal and it doesn't provide protection against timing--- attacks. The 'm' parameter is implicitly derived from the irreducible+-- attacks. The @m@ parameter is implicitly derived from the irreducible -- polynomial where applicable. module Crypto.Number.F2m ( BinaryPolynomial,@@ -23,10 +23,21 @@ quadraticF2m, ) where +import Crypto.Internal.WordArray import Crypto.Number.Basic-import Data.Bits (setBit, shift, testBit, unsafeShiftR, xor)-import Data.List (foldl')-import Prelude hiding (foldl')+import Data.Bits (+ setBit,+ shift,+ shiftL,+ shiftR,+ testBit,+ unsafeShiftR,+ xor,+ (.&.),+ (.|.),+ )+import qualified Data.List as L+import Data.Word (Word32) -- | Binary Polynomial represented by an integer type BinaryPolynomial = Integer@@ -53,17 +64,50 @@ error "modF2m: negative number represent no binary polynomial" | fx == 0 = error "modF2m: cannot divide by zero polynomial" | fx == 1 = 0- | otherwise = go i+ | otherwise = case tailExponents fx of+ Just es -> fold es i+ Nothing -> go i where lfx = log2 fx+ -- one bit at a time, for a modulus with too many terms to be worth the+ -- other way go n | s == 0 = n `addF2m` fx | s < 0 = n | otherwise = go $ n `addF2m` shift fx s where s = log2 n - lfx++ -- x^m is the rest of the modulus, so everything above bit m folds back in+ -- as a copy of the number's top shifted by each of the modulus's lower+ -- exponents: a handful of shifts, where the loop above takes one step per+ -- bit of excess+ mask = (1 `shiftL` lfx) - 1+ fold es n+ | n <= mask = n+ | otherwise =+ fold+ es+ (L.foldl' (\acc e -> acc `xor` (hi `shiftL` e)) (n .&. mask) es)+ where+ hi = n `shiftR` lfx {-# INLINE modF2m #-} +-- | The exponents of a modulus below its leading term, when there are few+-- enough of them to reduce with.+--+-- Every binary curve in use has a trinomial or a pentanomial here, which is+-- three or five exponents; sixteen is the point past which folding stops being+-- the cheaper way.+tailExponents :: BinaryPolynomial -> Maybe [Int]+tailExponents fx = go (log2 fx - 1) 0 []+ where+ go i n acc+ | i < 0 = Just acc+ | n > 16 = Nothing+ | testBit fx i = go (i - 1) (n + 1 :: Int) (i : acc)+ | otherwise = go (i - 1) n acc+ -- | Multiplication over F₂m. -- -- This function is undefined for negative arguments, because their bit@@ -80,14 +124,46 @@ || n2 < 0 = error "mulF2m: negative number represent no binary polynomial" | fx == 0 = error "mulF2m: cannot multiply modulo zero polynomial"- | otherwise = modF2m fx $ go (if n2 `mod` 2 == 1 then n1 else 0) (log2 n2)+ | otherwise = modF2m fx (go n2 0 0) where- go n s- | s == 0 = n- | otherwise =- if testBit n2 s- then go (n `addF2m` shift n1 s) (s - 1)- else go n (s - 1)+ -- Four bits of the multiplier at a time, against the sixteen multiples of+ -- n1 that four bits can ask for. A bit at a time is four times the+ -- shifting and exclusive-oring, and each of those allocates.+ go 0 _ acc = acc+ go v sh acc =+ go (v `shiftR` 4) (sh + 4) (acc `xor` (multiple (v .&. 0xf) `shiftL` sh))++ m2 = n1 `shiftL` 1+ m4 = n1 `shiftL` 2+ m8 = n1 `shiftL` 3+ m3 = m2 `xor` n1+ m5 = m4 `xor` n1+ m6 = m4 `xor` m2+ m7 = m6 `xor` n1+ m9 = m8 `xor` n1+ m10 = m8 `xor` m2+ m11 = m10 `xor` n1+ m12 = m8 `xor` m4+ m13 = m12 `xor` n1+ m14 = m12 `xor` m2+ m15 = m14 `xor` n1++ multiple 1 = n1+ multiple 2 = m2+ multiple 3 = m3+ multiple 4 = m4+ multiple 5 = m5+ multiple 6 = m6+ multiple 7 = m7+ multiple 8 = m8+ multiple 9 = m9+ multiple 10 = m10+ multiple 11 = m11+ multiple 12 = m12+ multiple 13 = m13+ multiple 14 = m14+ multiple 15 = m15+ multiple _ = 0 {-# INLINEABLE mulF2m #-} -- | Squaring over F₂m.@@ -114,11 +190,29 @@ -> Integer squareF2m' n | n < 0 = error "mulF2m: negative number represent no binary polynomial"- | otherwise =- foldl'- (\acc s -> if testBit n s then setBit acc (2 * s) else acc)- 0- [0 .. log2 n]+ | otherwise = go n 0 0+ where+ -- A byte at a time, through a table of the sixteen-bit patterns a byte+ -- spreads into. A bit at a time is eight times the work, and setting a+ -- bit of an Integer allocates another one.+ go 0 _ acc = acc+ go v sh acc =+ go+ (v `shiftR` 8)+ (sh + 16)+ ( acc+ .|. (fromIntegral (arrayRead32 spreadTable (fromIntegral (v .&. 0xff))) `shiftL` sh)+ )++-- | Each byte, with a zero inserted between every pair of its bits.+spreadTable :: Array32+spreadTable = array32 256 [spread b | b <- [0 .. 255]]+ where+ spread :: Int -> Word32+ spread b =+ L.foldl' (\acc i -> if testBit b i then setBit acc (2 * i) else acc) 0 [0 .. 7]+{-# NOINLINE spreadTable #-}+ {-# INLINE squareF2m' #-} -- | Exponentiation in F₂m by computing @a^b mod fx@.
@@ -6,13 +6,31 @@ -- Maintainer : Vincent Hanquez <vincent@snarc.org> -- Stability : experimental -- Portability : Good+--+-- Modular arithmetic on 'Integer'.+--+-- == What an 'Integer' shows+--+-- An 'Integer' is as long as its value needs, and every operation on one+-- costs what that length says. A secret that happens to be short is+-- multiplied, reduced and compared in fewer words than a full-length one, and+-- the difference is there to be measured. 'expSafe' and 'inverseSafe' keep+-- the /value/ of an exponent or of a number being inverted out of the work+-- they do, and that is as far as an 'Integer' can be taken: hiding the length+-- as well means a fixed-width representation, which is what the curve modules+-- and 'expSafe' itself use underneath. module Crypto.Number.ModArithmetic (+ -- * Exceptions+ CoprimesAssertionError (..),+ ModulusAssertionError (..),+ -- * Exponentiation expSafe, expFast, -- * Inverse computing inverse,+ inverseSafe, inverseCoprimes, inverseFermat, @@ -22,8 +40,15 @@ ) where import qualified Control.Exception as E+import Crypto.Internal.Compat (unsafeDoIO) import Crypto.Number.Basic import Crypto.Number.Compat+import qualified Crypto.Number.Serialize.Internal as Internal+import Data.Memory.PtrMethods (memSet)+import Data.Word (Word32, Word8)+import Foreign.C.Types (CInt (..))+import Foreign.Marshal.Alloc (allocaBytes)+import Foreign.Ptr (Ptr, plusPtr) -- | Raised when two numbers are supposed to be coprimes but are not. data CoprimesAssertionError = CoprimesAssertionError@@ -37,12 +62,25 @@ -- Modulo need to be odd otherwise the normal fast modular exponentiation -- is used. ----- When used with integer-simple, this function is not different--- from expFast, and thus provide the same unstudied and dubious--- timing and side channels claims.+-- With an odd modulo the work is done in C, four bits of exponent at a time:+-- four squarings and one multiplication by a small power of the base, taken+-- from a table of sixteen which is read by touching every entry and keeping+-- one of them with a mask. So each group of four bits costs the same five+-- multiplications and the same sixteen reads whatever those bits are, and+-- nothing branches on the exponent or indexes memory with it. ----- Before GHC 8.4.2, powModSecInteger is missing from integer-gmp,--- so expSafe has the same security as expFast.+-- What the exponent still shows is its length: it is rounded up to a whole+-- 64-bit word and every bit of that is walked over, so its value is hidden+-- but its size is not. The @mpz_powm_sec@ of GMP, which GHC stopped+-- offering in integer-gmp 1.1 and which this replaces, hides exactly as much.+--+-- The base is taken to be public -- in this library it is a ciphertext, a+-- public value from a peer, or a generator -- and is reduced modulo the+-- modulus in the ordinary way first.+--+-- Hiding the exponent has a price: against the windowed exponentiation of+-- GMP, which is what this function used to end up calling, a 2048-bit+-- modulus costs somewhat over twice as much. expSafe :: Integer -- ^ base@@ -53,15 +91,67 @@ -> Integer -- ^ result expSafe b e m- | odd m =- gmpPowModSecInteger b e m- `onGmpUnsupported` ( gmpPowModInteger b e m- `onGmpUnsupported` exponentiation b e m- )+ | odd m && m > 1 && e >= 0 =+ gmpPowModSecInteger b e m `onGmpUnsupported` expSec (b `mod` m) e m+ -- a modulus of one, and a negative exponent asking for an inverse, are+ -- left to the path they have always taken | otherwise = gmpPowModInteger b e m `onGmpUnsupported` exponentiation b e m +-- | The windowed exponentiation itself, in C. The base has to be reduced+-- already, the exponent to be zero or more, and the modulus odd and above+-- one.+expSec :: Integer -> Integer -> Integer -> Integer+expSec b e m = unsafeDoIO $+ allocaBytes (sum widths) $ \start -> case scanl plusPtr start widths of+ (out : base : expo : modu : _) -> do+ _ <- Internal.i2ospOf b base mLen+ _ <- Internal.i2ospOf e expo eLen+ _ <- Internal.i2ospOf m modu mLen+ r <-+ c_powm_sec+ out+ base+ (fromIntegral mLen)+ expo+ (fromIntegral eLen)+ modu+ (fromIntegral mLen)+ -- the exponent is the caller's secret, and this is the last place it+ -- is written out in the clear+ memSet expo 0 eLen+ if r == 0+ then do+ !v <- Internal.os2ip out mLen+ return v+ else+ return+ ( gmpPowModInteger b e m+ `onGmpUnsupported` exponentiation b e m+ )+ _ -> return 0 -- there are four, but say so anyway+ where+ !mLen = numBytes m+ -- the answer, the base, the exponent and the modulus. The room to take+ -- and where each one starts both come from here, so they cannot drift+ -- apart.+ widths = [mLen, mLen, eLen, mLen]+ -- whole words of exponent, so that the count of them says as little as+ -- what GMP's own secure exponentiation lets slip+ !eLen = 8 * ((numBytes e + 7) `div` 8)++foreign import ccall safe "crypton_powm_sec"+ c_powm_sec+ :: Ptr Word8+ -> Ptr Word8+ -> Word32+ -> Ptr Word8+ -> Word32+ -> Ptr Word8+ -> Word32+ -> IO CInt+ -- | Compute the modular exponentiation of base^exponent using -- the fastest algorithm without any consideration for -- hiding parameters.@@ -81,9 +171,19 @@ -- | @exponentiation@ computes modular exponentiation as /b^e mod m/ -- using repetitive squaring.+--+-- The corner cases are held to what GMP answers, since that is what this+-- computes on every build that has it: a modulus of one is zero whatever+-- else is asked, and a negative exponent is a request for the inverse of+-- the base raised to its magnitude, which is zero when no inverse exists.+-- Read literally, the recursion below walked a negative exponent from -1 to+-- -2 and back for as long as the stack held. exponentiation :: Integer -> Integer -> Integer -> Integer exponentiation b e m- | b == 1 = b+ | m == 1 = 0+ | e < 0 =+ maybe 0 (\bInv -> exponentiation bInv (negate e) m) (inverse (b `mod` m) m)+ | b == 1 = 1 | e == 0 = 1 | e == 1 = b `mod` m | even e =@@ -105,7 +205,7 @@ -- is known to exists. -- -- If the numbers are not defined as coprime, this function--- will raise a 'CoprimesAssertionError'.+-- will raise a t'CoprimesAssertionError'. inverseCoprimes :: Integer -> Integer -> Integer inverseCoprimes g m = case inverse g m of@@ -144,6 +244,63 @@ inverseFermat :: Integer -> Integer -> Integer inverseFermat g p = expSafe g (p - 2) p +-- | @inverseSafe@ computes the modular inverse without letting the number+-- being inverted steer how long the work takes, which is what 'inverse' does:+-- the extended Euclidean algorithm takes a number of steps that follows the+-- bits it is given, and a nonce inverted that way has been taken apart before+-- by watching the steps go by.+--+-- The answer comes from a fixed number of division steps where the assembly+-- for them is built, and from 'inverseFermat' where it is not. Either way it+-- is checked here by multiplying out: neither one says when the number has no+-- inverse -- the first returns something that is not one and the second+-- returns something that is not one either -- so the check is what makes this+-- agree with 'inverse' on every input, and 'inverse' is asked when it fails.+-- That fallback is reached only by parameters that are already broken.+--+-- The division steps cost about a twentieth of the exponentiation: on an+-- Apple M4, inverting modulo the P-256 group order is 0.80 microseconds+-- against 6.02, and modulo the P-521 one 2.05 against 63.2.+inverseSafe :: Integer -> Integer -> Maybe Integer+inverseSafe g m+ | m > 1 && (g * r) `mod` m == 1 = Just r+ | otherwise = inverse g m+ where+ r = case inverseSec g m of+ Just v -> v+ Nothing -> inverseFermat g m++-- | The inverse in a fixed number of division steps, from the vendored+-- assembly. 'Nothing' when that is not built, when the modulus is even --+-- where the routine answers without saying it cannot -- or when the numbers+-- are larger than it keeps room for. The answer is not checked here; the+-- caller does that.+inverseSec :: Integer -> Integer -> Maybe Integer+inverseSec g m+ | m <= 1 || even m || g < 0 = Nothing+ | otherwise = unsafeDoIO $+ allocaBytes (3 * mLen) $ \out -> do+ let gp = out `plusPtr` mLen+ mp = gp `plusPtr` mLen+ _ <- Internal.i2ospOf (g `mod` m) gp mLen+ _ <- Internal.i2ospOf m mp mLen+ r <- c_modinv_sec out gp mp (fromIntegral mLen)+ if r == 0+ then do+ !v <- Internal.os2ip out mLen+ return (Just v)+ else return Nothing+ where+ !mLen = numBytes m++foreign import ccall unsafe "crypton_modinv_sec"+ c_modinv_sec+ :: Ptr Word8+ -> Ptr Word8+ -> Ptr Word8+ -> Word32+ -> IO CInt+ -- | Raised when the assumption about the modulus is invalid. data ModulusAssertionError = ModulusAssertionError deriving (Show)@@ -153,7 +310,7 @@ -- | Modular square root of @g@ modulo a prime @p@. -- -- If the modulus is found not to be prime, the function will raise a--- 'ModulusAssertionError'.+-- t'ModulusAssertionError'. -- -- This implementation is variable time and should be used with public -- parameters only.
@@ -23,23 +23,69 @@ import Crypto.Number.Compat import Crypto.Number.Generate import Crypto.Number.ModArithmetic (expSafe)+import Crypto.Number.Serialize (i2osp) import Crypto.Random.Probabilistic import Crypto.Random.Types +import Crypto.Internal.ByteArray (Bytes)+ import Data.Bits -- | Returns if the number is probably prime.--- First a list of small primes are implicitely tested for divisibility,--- then a fermat primality test is used with arbitrary numbers and--- then the Miller Rabin algorithm is used with an accuracy of 30 recursions.+--+-- The small primes are tested for divisibility first, and then the+-- Miller-Rabin algorithm with an accuracy of 30 rounds.+--+-- A Fermat test of fifty consecutive bases used to run between the two. It+-- ruled out nothing Miller-Rabin does not: a strong probable prime to a base+-- is a Fermat probable prime to that base, and the converse is what Carmichael+-- numbers are. What it cost was fifty modular exponentiations on every number+-- that turned out to be prime -- 2167 of the 3480 microseconds spent on a+-- 512-bit prime, and about two thirds of the time to generate one. isProbablyPrime :: Integer -> Bool-isProbablyPrime !n+isProbablyPrime = probablyPrime 30++-- | The same, with the number of rounds said outright.+--+-- Thirty rounds is what a number from anywhere gets: whoever handed it over+-- may have built it to pass, and against that the only thing to go on is that+-- each round with a base drawn at random catches three quarters of the+-- composites there are, whatever the number is. Thirty of them leave one+-- chance in 2^60.+probablyPrime :: Int -> Integer -> Bool+probablyPrime rounds !n+ | n < 2 = False | any (\p -> p `divides` n) (filter (< n) firstPrimes) = False- | n >= 2 && n <= 2903 = True- | primalityTestFermat 50 (n `div` 2) n =- primalityTestMillerRabin 30 n- | otherwise = False+ | n <= 2903 = True+ | otherwise = primalityTestMillerRabin rounds n +-- | How many rounds a candidate drawn here needs.+--+-- A number nobody chose is a different matter from one somebody did. The+-- composites that survive a round are rare, and the ones that survive several+-- are rarer than the bound above says: Damgard, Landrock and Pomerance+-- worked out how much rarer for a candidate drawn at random, and Table 4.4 of+-- the Handbook of Applied Cryptography puts their numbers in a table -- two+-- rounds at 1300 bits, three at 850, five at 550, and so on, for one chance+-- in 2^80.+--+-- This is twice that, and never more than the thirty a number from anywhere+-- gets, which leaves the chance far under one in 2^100 at every size. It is+-- what makes generating a prime worth doing: the thirty rounds were half the+-- time it took.+roundsForDrawn :: Int -> Int+roundsForDrawn bits+ | bits >= 1300 = 6+ | bits >= 850 = 8+ | bits >= 650 = 10+ | bits >= 550 = 12+ | bits >= 450 = 14+ | bits >= 400 = 16+ | bits >= 350 = 18+ | bits >= 300 = 20+ | bits >= 250 = 24+ | otherwise = 30+ -- | Generate a prime number of the required bitsize (i.e. in the range -- [2^(b-1)+2^(b-2), 2^b)). --@@ -55,7 +101,7 @@ throwCryptoError $ CryptoFailed $ CryptoError_PrimeSizeInvalid else do sp <- generateParams bits (Just SetTwoHighest) True- let prime = findPrimeFrom sp+ let prime = findPrimeFromDrawn (roundsForDrawn bits) sp if prime < 1 `shiftL` bits then return $ prime@@ -76,7 +122,12 @@ throwCryptoError $ CryptoFailed $ CryptoError_PrimeSizeInvalid else do sp <- generateParams bits (Just SetTwoHighest) True- let p = findPrimeFromWith (\i -> isProbablyPrime (2 * i + 1)) (sp `div` 2)+ let rounds = roundsForDrawn bits+ p =+ findPrimeFromWithRounds+ rounds+ (\i -> probablyPrime rounds (2 * i + 1))+ (sp `div` 2) let val = 2 * p + 1 if val < 1 `shiftL` bits then@@ -85,16 +136,26 @@ -- | Find a prime from a starting point where the property hold. findPrimeFromWith :: (Integer -> Bool) -> Integer -> Integer-findPrimeFromWith prop !n- | even n = findPrimeFromWith prop (n + 1)+findPrimeFromWith = findPrimeFromWithRounds 30++-- | The same, with the number of rounds said outright: the walk starts where+-- the caller says, and only a caller that drew that starting point itself is+-- entitled to the smaller number.+findPrimeFromWithRounds :: Int -> (Integer -> Bool) -> Integer -> Integer+findPrimeFromWithRounds rounds prop !n+ | even n = findPrimeFromWithRounds rounds prop (n + 1) | otherwise =- if not (isProbablyPrime n)- then findPrimeFromWith prop (n + 2)+ if not (probablyPrime rounds n)+ then findPrimeFromWithRounds rounds prop (n + 2) else if prop n then n- else findPrimeFromWith prop (n + 2)+ else findPrimeFromWithRounds rounds prop (n + 2) +-- | Find a prime from a starting point that the caller drew itself.+findPrimeFromDrawn :: Int -> Integer -> Integer+findPrimeFromDrawn rounds = findPrimeFromWithRounds rounds (\_ -> True)+ -- | Find a prime from a starting point with no specific property. findPrimeFrom :: Integer -> Integer findPrimeFrom n =@@ -104,11 +165,18 @@ -- | Miller Rabin algorithm return if the number is probably prime or composite. -- the tries parameter is the number of recursion, that determines the accuracy of the test.+--+-- The witnesses are drawn from a generator derived from @n@ itself and from a+-- secret drawn once per process: testing the same number twice gives the same+-- answer, testing two numbers draws independent witnesses for each, and an+-- attacker choosing the number cannot tell which witnesses it will face. primalityTestMillerRabin :: Int -> Integer -> Bool primalityTestMillerRabin tries !n = case gmpTestPrimeMillerRabin tries n of GmpSupported b -> b- GmpUnsupported -> probabilistic run+ -- the material is forced only once a witness is drawn, which the+ -- guards in run reach only for an odd n above 3+ GmpUnsupported -> probabilisticFrom (i2osp n :: Bytes) run where run | n <= 3 = error "Miller-Rabin requires tested value to be > 3"
@@ -53,7 +53,12 @@ !padSz = ptrSz - sz fillPtr :: Ptr Word8 -> Int -> Integer -> IO ()-fillPtr p sz m = gmpExportInteger m p `onGmpUnsupported` export (sz - 1) m+fillPtr p sz m+ -- zero is no bytes wide, and the callers above have already written the+ -- room out as zeros. Without this the loop below starts at offset -1,+ -- never meets the 0 it stops at, and walks backwards out of the buffer.+ | sz <= 0 = return ()+ | otherwise = gmpExportInteger m p `onGmpUnsupported` export (sz - 1) m where export ofs i | ofs == 0 = pokeByteOff p ofs (fromIntegral i :: Word8)
@@ -1,3 +1,4 @@+{-# LANGUAGE BangPatterns #-} {-# LANGUAGE ScopedTypeVariables #-} -- | One-time password implementation as defined by the@@ -29,6 +30,7 @@ OTP, OTPDigits (..), OTPTime,+ minimumDigestSize, hotp, resynchronize, totp,@@ -41,13 +43,13 @@ where import Control.Monad (unless)-import Crypto.Hash (HashAlgorithm, SHA1 (..))+import Crypto.Hash (HashAlgorithm, SHA1 (..), hashDigestSize) import Crypto.Internal.ByteArray (ByteArrayAccess, Bytes) import qualified Crypto.Internal.ByteArray as B import Crypto.MAC.HMAC-import Data.Bits (shiftL, (.&.), (.|.))+import Data.Bits (complement, shiftL, shiftR, xor, (.&.), (.|.)) import Data.ByteArray.Mapping (fromW64BE)-import Data.List (elemIndex)+import qualified Data.List as L import Data.Word -- | A one-time password which is a sequence of 4 to 9 digits.@@ -60,6 +62,21 @@ -- | An integral time value in seconds. type OTPTime = Word64 +-- | The smallest hash digest 'hotp' can be used with, in bytes.+--+-- RFC 4226 section 5.3 defines dynamic truncation over the 20-byte HMAC-SHA-1+-- output: the offset is the low four bits of the last byte, so it selects any+-- of the first 16 bytes, and four bytes are then read starting there. The+-- highest byte that can be reached is therefore byte 18, and a shorter digest+-- would make that read run off the end of the MAC.+minimumDigestSize :: Int+minimumDigestSize = 20++-- | Calculate an HOTP value as defined by RFC 4226.+--+-- The hash must produce a digest of at least 'minimumDigestSize' bytes, which+-- is what the dynamic truncation step is defined over; 'error' is raised+-- otherwise. hotp :: forall hash key . (HashAlgorithm hash, ByteArrayAccess key)@@ -72,10 +89,19 @@ -- ^ Counter value synchronized between the client and server -> OTP -- ^ The HOTP value-hotp _ d k c = dt `mod` digitsPower d+hotp _ d k c+ | macLen < minimumDigestSize =+ error $+ "Crypto.OTP.hotp: hash digest is "+ ++ show macLen+ ++ " bytes, but at least "+ ++ show minimumDigestSize+ ++ " are required"+ | otherwise = dt `mod` digitsPower d where mac = hmac k (fromW64BE c :: Bytes) :: HMAC hash- offset = fromIntegral (B.index mac (B.length mac - 1) .&. 0xf)+ macLen = B.length mac+ offset = fromIntegral (B.index mac (macLen - 1) .&. 0xf) dt = (fromIntegral (B.index mac offset .&. 0x7f) `shiftL` 24) .|. (fromIntegral (B.index mac (offset + 1) .&. 0xff) `shiftL` 16)@@ -84,6 +110,12 @@ -- | Attempt to resynchronize the server's counter value -- with the client, given a sequence of HOTP values.+--+-- Every counter in the window is tried and every submitted value is compared,+-- whatever matches, so the time taken does not depend on where in the window+-- the client's counter was found, nor on how many of the submitted values were+-- right. The cost of a call is therefore one HMAC per counter in the window+-- plus one per extra value, every time. resynchronize :: (HashAlgorithm hash, ByteArrayAccess key) => hash@@ -101,17 +133,46 @@ -> Maybe Word64 -- ^ The new counter value, synchronized with the client's current counter -- or Nothing if the submitted OTP values didn't match anywhere within the window-resynchronize h d s k c (p1, extras) = do- offBy <- fmap fromIntegral (elemIndex p1 range)- checkExtraOtps (c + offBy + 1) extras+resynchronize h d s k c (p1, extras)+ | accepted == 0 = Nothing+ | otherwise = Just (afterFirst + fromIntegral (length extras)) where- checkExtraOtps ctr [] = Just ctr- checkExtraOtps ctr (p : ps)- | hotp h d k ctr /= p = Nothing- | otherwise = checkExtraOtps (ctr + 1) ps+ -- Every counter in the window is tried and every extra value is compared,+ -- whatever matches: the search does not stop at the first hit and the+ -- check of the extra values does not stop at the first miss. Each skipped+ -- counter used to save an HMAC, so the time taken revealed where in the+ -- window the client's counter sat and how many of its extra values were+ -- right -- the second of which a client that submits guesses cannot learn+ -- from the answer itself, since that is 'Nothing' either way.+ accepted = matched .&. extrasMatched range = map (hotp h d k) [c .. c + fromIntegral s] + -- the offset of the first match, accumulated without stopping there+ (matched, offset) = L.foldl' pick (0, 0) (zip [0 ..] range)+ pick (!m, !off) (i, candidate) = (m .|. hit, off .|. (hit .&. i))+ where+ -- zero once something has matched, so only the first match counts+ hit = eqMask candidate p1 .&. complement m++ -- the counter the first submitted value matched, plus one+ afterFirst = c + offset + 1++ -- the counters continue past the window, and wrap where the old+ -- 'checkExtraOtps' wrapped+ extrasMatched =+ L.foldl' step (complement 0) (zip (iterate (+ 1) afterFirst) extras)+ step acc (ctr, p) = acc .&. eqMask (hotp h d k ctr) p++-- | All ones when the two values are equal, zero otherwise, without branching+-- on either of them.+eqMask :: OTP -> OTP -> Word64+eqMask a b = negate (fromIntegral (1 - nonZero))+ where+ v = a `xor` b+ -- 0 when v is zero, 1 otherwise+ nonZero = (v .|. negate v) `shiftR` 31+ digitsPower :: OTPDigits -> Word32 digitsPower OTP4 = 10000 digitsPower OTP5 = 100000@@ -148,6 +209,13 @@ mkTOTPParams h t0 x d skew = do unless (x > 0) (Left "Time step must be greater than zero") unless (x <= 300) (Left "Time step cannot be greater than 300 seconds")+ unless+ (hashDigestSize h >= minimumDigestSize)+ ( Left $+ "Hash digest must be at least "+ ++ show minimumDigestSize+ ++ " bytes"+ ) return (TP h t0 x d skew) -- | Calculate a totp value for the given time.@@ -171,12 +239,18 @@ -> OTPTime -> OTP -> Bool-totpVerify (TP h t0 x d skew) k now otp = otp `elem` map (hotp h d k) (range window [])+totpVerify (TP h t0 x d skew) k now otp = matched /= 0 where t = timeToCounter now t0 x window = fromIntegral (fromEnum skew) range 0 acc = t : acc range n acc = range (n - 1) ((t - n) : (t + n) : acc)++ -- every candidate is compared, and none of the comparisons stops early, so+ -- neither which step matched nor how far a mismatch got is visible in how+ -- long this takes+ matched = L.foldl' step 0 (map (hotp h d k) (range window []))+ step acc candidate = acc .|. eqMask candidate otp timeToCounter :: Word64 -> Word64 -> Word16 -> Word64 timeToCounter now t0 x = (now - t0) `div` fromIntegral x
@@ -1,5 +1,4 @@ {-# LANGUAGE GeneralizedNewtypeDeriving #-}-{-# LANGUAGE MagicHash #-} {-# LANGUAGE ScopedTypeVariables #-} -- |@@ -26,11 +25,11 @@ generateSecretKey, ) where +import Crypto.Debug (DebugShow (..), debugShowBytes) import Data.Bits import Data.Word import Foreign.Ptr import Foreign.Storable-import GHC.Ptr import Crypto.Error import Crypto.Internal.ByteArray (@@ -48,6 +47,9 @@ newtype SecretKey = SecretKey ScrubbedBytes deriving (Show, Eq, ByteArrayAccess, NFData) +instance DebugShow SecretKey where+ debugShow = debugShowBytes "SecretKey"+ -- | A Curve25519 public key newtype PublicKey = PublicKey Bytes deriving (Show, Eq, ByteArrayAccess, NFData)@@ -110,20 +112,19 @@ $ \result -> withByteArray sec $ \psec -> withByteArray pub $ \ppub ->- ccrypton_curve25519 result psec ppub+ ccrypton_x25519 result psec ppub {-# NOINLINE dh #-} -- | Create a public key from a secret key+-- The base point does not go in: where the assembly is built there is a table+-- for this, and it is four to five times less work than multiplying the point+-- 9 the general way. toPublic :: SecretKey -> PublicKey toPublic (SecretKey sec) = PublicKey <$> B.allocAndFreeze 32 $ \result -> withByteArray sec $ \psec ->- ccrypton_curve25519 result psec basePoint- where- basePoint =- Ptr- "\x09\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00"#+ ccrypton_x25519_base result psec {-# NOINLINE toPublic #-} -- | Generate a secret key.@@ -138,12 +139,20 @@ modifyByte :: Ptr Word8 -> Int -> (Word8 -> Word8) -> IO () modifyByte p n f = peekByteOff p n >>= pokeByteOff p n . f -foreign import ccall "crypton_curve25519_donna"- ccrypton_curve25519+foreign import ccall "crypton_x25519"+ ccrypton_x25519 :: Ptr Word8 -- ^ public -> Ptr Word8 -- ^ secret -> Ptr Word8 -- ^ basepoint+ -> IO ()++foreign import ccall "crypton_x25519_base"+ ccrypton_x25519_base+ :: Ptr Word8+ -- ^ public+ -> Ptr Word8+ -- ^ secret -> IO ()
@@ -28,6 +28,7 @@ generateSecretKey, ) where +import Crypto.Debug (DebugShow (..), debugShowBytes) import Data.Word import Foreign.Ptr @@ -46,6 +47,9 @@ -- | A Curve448 Secret key newtype SecretKey = SecretKey ScrubbedBytes deriving (Show, Eq, ByteArrayAccess, NFData)++instance DebugShow SecretKey where+ debugShow = debugShowBytes "SecretKey" -- | A Curve448 public key newtype PublicKey = PublicKey Bytes
@@ -17,9 +17,17 @@ calculatePublic, generatePublic, getShared,+ tryGetShared, ) where +import Crypto.Debug (DebugShow (..))+import Crypto.Error (+ CryptoError (..),+ CryptoFailable (..),+ throwCryptoError,+ ) import Crypto.Internal.Imports+import Crypto.Number.Basic (numBytes) import Crypto.Number.Generate (generateMax) import Crypto.Number.ModArithmetic (expSafe) import Crypto.Number.Prime (generateSafePrime)@@ -45,8 +53,16 @@ -- | Represent Diffie Hellman private number X. newtype PrivateNumber = PrivateNumber Integer- deriving (Show, Read, Eq, Enum, Real, Num, Ord, NFData)+ deriving (Read, Eq, Enum, Real, Num, Ord, NFData) +-- | The number is not shown. Use 'Crypto.Debug.debugShow' to see it.+instance Show PrivateNumber where+ show _ = "PrivateNumber <secret>"++instance DebugShow PrivateNumber where+ debugShow (PrivateNumber n) =+ showString "PrivateNumber " . showsPrec 11 n $ ""+ -- | Represent Diffie Hellman shared secret. newtype SharedKey = SharedKey ScrubbedBytes deriving (Show, Eq, ByteArrayAccess, NFData)@@ -83,5 +99,29 @@ -- commented until 0.3 {-# DEPRECATED generatePublic "use calculatePublic" #-} -- | generate a shared key using our private number and the other party public number+--+-- This raises the 'CryptoError' that 'tryGetShared' reports. Use 'tryGetShared'+-- where the failure has to be handled. getShared :: Params -> PrivateNumber -> PublicNumber -> SharedKey-getShared (Params p _ bits) (PrivateNumber x) (PublicNumber y) = SharedKey $ i2ospOf_ ((bits + 7) `div` 8) $ expSafe y x p+getShared params x y = throwCryptoError $ tryGetShared params x y++-- | generate a shared key using our private number and the other party public+-- number, reporting a rejected public number instead of raising.+--+-- The public number comes from the other party, so it is checked to satisfy+-- @1 < y < p-1@ as RFC 7919 section 5.1 requires. The excluded values+-- generate the subgroup @{1}@ or @{1, p-1}@, so the shared secret they produce+-- is one of a handful of constants and carries none of our private number's+-- secrecy. A value outside that range is reported as+-- 'CryptoError_ParameterInvalid'.+--+-- Note this is the only check made here: it does not establish that @y@ lies+-- in the subgroup generated by @g@, which needs the subgroup order that+-- t'Params' does not carry.+tryGetShared+ :: Params -> PrivateNumber -> PublicNumber -> CryptoFailable SharedKey+tryGetShared (Params p _ _) (PrivateNumber x) (PublicNumber y)+ | y <= 1 || y >= p - 1 = CryptoFailed CryptoError_ParameterInvalid+ -- the size of p, not params_bits: only p and g travel on the wire, so a+ -- caller-supplied bit size can disagree with p+ | otherwise = CryptoPassed $ SharedKey $ i2ospOf_ (numBytes p) $ expSafe y x p
@@ -8,6 +8,24 @@ -- Portability : Good -- -- An implementation of the Digital Signature Algorithm (DSA)+--+-- == What is kept from the clock, and what is not+--+-- Signing keeps the private number and the ephemeral @k@ out of the two+-- places whose duration would otherwise follow them: the exponentiation is+-- 'Crypto.Number.ModArithmetic.expSafe', which walks the exponent a fixed+-- four bits at a time, and @k@ is inverted by Fermat's little theorem rather+-- than by the extended Euclidean algorithm, whose number of steps follows the+-- bits it is given.+--+-- What is left is the arithmetic around them. @x * r@, the addition and the+-- reduction modulo @q@ are 'Integer' operations, and an 'Integer' costs what+-- its size says: a private number that happens to be short is multiplied in+-- fewer words than a full-length one. The same holds in+-- "Crypto.PubKey.ElGamal" and "Crypto.PubKey.Rabin.Basic". Removing it means+-- leaving 'Integer' for a fixed-width representation, which is what+-- "Crypto.PubKey.RSA" does for its exponentiation and the curve modules do+-- throughout; there is nothing a caller can do about it from here. module Crypto.PubKey.DSA ( Params (..), Signature (..),@@ -23,9 +41,12 @@ -- * Signature primitive sign, signWith,+ signDigest,+ signDigestWith, -- * Verification primitive verify,+ verifyDigest, -- * Key pair KeyPair (..),@@ -33,15 +54,15 @@ toPrivateKey, ) where +import Crypto.Debug (DebugShow (..)) import Data.Data-import Data.Maybe import Crypto.Hash import Crypto.Internal.ByteArray (ByteArrayAccess) import Crypto.Internal.Imports import Crypto.Number.Generate-import Crypto.Number.ModArithmetic (expFast, expSafe, inverse)-import Crypto.PubKey.Internal (dsaTruncHash)+import Crypto.Number.ModArithmetic (expFast, expSafe, inverse, inverseSafe)+import Crypto.PubKey.Internal (dsaTruncHashDigest) import Crypto.Random.Types -- | DSA Public Number, usually embedded in DSA Public Key@@ -98,15 +119,52 @@ , private_x :: PrivateNumber -- ^ DSA private X }- deriving (Show, Read, Eq, Data)+ deriving (Read, Eq, Data) +-- | The parameters are shown; @private_x@ is not. Use+-- 'Crypto.Debug.debugShow' to see it.+instance Show PrivateKey where+ showsPrec d k =+ showParen (d > 10) $+ showString "PrivateKey {private_params = "+ . shows (private_params k)+ . showString ", private_x = <secret>}"++instance DebugShow PrivateKey where+ debugShow k =+ showString "PrivateKey {private_params = "+ . shows (private_params k)+ . showString ", private_x = "+ . shows (private_x k)+ . showChar '}'+ $ ""+ instance NFData PrivateKey where rnf (PrivateKey params x) = x `seq` params `seq` () -- | Represent a DSA key pair data KeyPair = KeyPair Params PublicNumber PrivateNumber- deriving (Show, Read, Eq, Data)+ deriving (Read, Eq, Data) +instance Show KeyPair where+ showsPrec d (KeyPair params y _) =+ showParen (d > 10) $+ showString "KeyPair "+ . showsPrec 11 params+ . showChar ' '+ . showsPrec 11 y+ . showString " <secret>"++instance DebugShow KeyPair where+ debugShow (KeyPair params y x) =+ showString "KeyPair "+ . showsPrec 11 params+ . showChar ' '+ . showsPrec 11 y+ . showChar ' '+ . showsPrec 11 x+ $ ""+ instance NFData KeyPair where rnf (KeyPair params y x) = x `seq` y `seq` params `seq` () @@ -139,27 +197,54 @@ -> msg -- ^ message to sign -> Maybe Signature-signWith k pk hashAlg msg- | r == 0 || s == 0 = Nothing- | otherwise = Just $ Signature r s+signWith k pk hashAlg msg = signDigestWith k pk (hashWith hashAlg msg)++-- | Sign a digest using the private key and an explicit k number.+--+-- The digest's type says which algorithm made it, so this needs nothing else+-- to name one. 'signWith' takes a @hash@ value and never reads it -- it is+-- there to fix the type -- which leaves a caller that is itself polymorphic+-- in the algorithm with nothing to pass.+signDigestWith+ :: HashAlgorithm hash+ => Integer+ -- ^ k random number+ -> PrivateKey+ -- ^ private key+ -> Digest hash+ -- ^ digest of the message to sign+ -> Maybe Signature+signDigestWith k pk digest = do+ -- k comes from the caller and is only invertible when it is coprime with+ -- q, which the caller cannot check without knowing q is prime. It is also+ -- a secret worth as much as the private key, so it is inverted without+ -- the extended Euclidean algorithm, whose steps follow the bits of what+ -- it is given+ kInv <- inverseSafe k q+ let hm = dsaTruncHashDigest digest q+ r = expSafe g k p `mod` q+ s = (kInv * (hm + x * r)) `mod` q+ if r == 0 || s == 0 then Nothing else Just $ Signature r s where -- parameters (Params p g q) = private_params pk x = private_x pk- -- compute r,s- kInv = fromJust $ inverse k q- hm = dsaTruncHash hashAlg msg q- r = expSafe g k p `mod` q- s = (kInv * (hm + x * r)) `mod` q -- | sign message using the private key. sign :: (ByteArrayAccess msg, HashAlgorithm hash, MonadRandom m) => PrivateKey -> hash -> msg -> m Signature-sign pk hashAlg msg = do+sign pk hashAlg msg = signDigest pk (hashWith hashAlg msg)++-- | Sign a digest using the private key. See 'signDigestWith' for why a+-- digest rather than a @hash@ value.+signDigest+ :: (HashAlgorithm hash, MonadRandom m)+ => PrivateKey -> Digest hash -> m Signature+signDigest pk digest = do k <- generateMax q- case signWith k pk hashAlg msg of- Nothing -> sign pk hashAlg msg+ case signDigestWith k pk digest of+ Nothing -> signDigest pk digest Just sig -> return sig where (Params _ _ q) = private_params pk@@ -168,15 +253,24 @@ verify :: (ByteArrayAccess msg, HashAlgorithm hash) => hash -> PublicKey -> Signature -> msg -> Bool-verify hashAlg pk (Signature r s) m+verify hashAlg pk sig m = verifyDigest pk sig (hashWith hashAlg m)++-- | Verify a signature over a digest. See 'signDigestWith' for why a digest+-- rather than a @hash@ value.+verifyDigest+ :: HashAlgorithm hash => PublicKey -> Signature -> Digest hash -> Bool+verifyDigest pk (Signature r s) digest -- Reject the signature if either 0 < r < q or 0 < s < q is not satisfied. | r <= 0 || r >= q || s <= 0 || s >= q = False- | otherwise = v == r+ -- s is invertible for every 0 < s < q when q is prime, but the parameters+ -- arrive with the public key and a composite q admits an s that is not+ | otherwise = maybe False (r ==) v where (Params p g q) = public_params pk y = public_y pk- hm = dsaTruncHash hashAlg m q- w = fromJust $ inverse s q- u1 = (hm * w) `mod` q- u2 = (r * w) `mod` q- v = ((expFast g u1 p) * (expFast y u2 p)) `mod` p `mod` q+ hm = dsaTruncHashDigest digest q+ v = do+ w <- inverse s q+ let u1 = (hm * w) `mod` q+ u2 = (r * w) `mod` q+ return $ ((expFast g u1 p) * (expFast y u2 p)) `mod` p `mod` q
@@ -14,12 +14,18 @@ generatePrivate, calculatePublic, getShared,+ tryGetShared, ) where +import Crypto.Error (+ CryptoError (..),+ CryptoFailable (..),+ throwCryptoError,+ ) import Crypto.Number.Generate (generateMax) import Crypto.Number.Serialize (i2ospOf_) import Crypto.PubKey.DH (SharedKey (..))-import Crypto.PubKey.ECC.Prim (pointMul)+import Crypto.PubKey.ECC.Prim (isPointInSubgroup, isPointValid, pointMul) import Crypto.PubKey.ECC.Types ( Curve, Point (..),@@ -47,10 +53,37 @@ -- | Generating a shared key using our private number and -- the other party public point.+--+-- This raises the 'Crypto.Error.CryptoError' that 'tryGetShared' reports. Use+-- 'tryGetShared' where the failure has to be handled. getShared :: Curve -> PrivateNumber -> PublicPoint -> SharedKey-getShared curve db qa = SharedKey $ i2ospOf_ ((nbBits + 7) `div` 8) x+getShared curve db qa = throwCryptoError $ tryGetShared curve db qa++-- | Generating a shared key using our private number and the other party+-- public point, reporting a rejected point instead of raising.+--+-- The public point comes from the other party, so it is checked before it is+-- multiplied. A point that does not satisfy the curve equation is reported as+-- 'CryptoError_PointCoordinatesInvalid'.+--+-- Satisfying the equation is not by itself membership of the subgroup the base+-- point generates; the two coincide only when the cofactor is 1. On a curve+-- whose cofactor is above 1 the other party can offer a point of small order,+-- and the value that comes back then depends on our private number only+-- through its residue modulo that order, which hands them those bits. So the+-- point is also required to be in the subgroup, by 'isPointInSubgroup', and is+-- reported as 'CryptoError_PointSubgroupInvalid' when it is not. That check+-- costs one further scalar multiplication, and is skipped where the cofactor+-- is 1 and it cannot fail.+--+-- An exchange that yields the point at infinity, and so has no x coordinate to+-- derive the key from, is reported as 'CryptoError_ScalarMultiplicationInvalid'.+tryGetShared :: Curve -> PrivateNumber -> PublicPoint -> CryptoFailable SharedKey+tryGetShared curve db qa+ | not (isPointValid curve qa) = CryptoFailed CryptoError_PointCoordinatesInvalid+ | not (isPointInSubgroup curve qa) = CryptoFailed CryptoError_PointSubgroupInvalid+ | otherwise = case pointMul curve db qa of+ Point x _ -> CryptoPassed $ SharedKey $ i2ospOf_ ((nbBits + 7) `div` 8) x+ PointO -> CryptoFailed CryptoError_ScalarMultiplicationInvalid where- x = case pointMul curve db qa of- Point x' _ -> x'- _ -> error "getShared" nbBits = curveSizeBits curve
@@ -1,7 +1,11 @@ {-# LANGUAGE DeriveDataTypeable #-} --- | /WARNING:/ Signature operations may leak the private key. Signature verification--- should be safe.+-- | /WARNING:/ Signature operations may leak the private key. The nonce is+-- inverted without a side channel on every curve, and on P-256 the scalar+-- multiplication is the constant-time C implementation, but what surrounds+-- them is 'Integer' arithmetic, whose cost follows the values it is given, and+-- on every other curve the multiplication follows the nonce as well.+-- Signature verification takes only public values and should be safe. module Crypto.PubKey.ECC.ECDSA ( Signature (..), ExtendedSignature (..),@@ -25,6 +29,7 @@ deterministicNonce, ) where +import Crypto.Debug (DebugShow (..)) import Control.Monad import Data.Bits import Data.ByteArray (ByteArrayAccess, ScrubbedBytes)@@ -66,8 +71,26 @@ { private_curve :: Curve , private_d :: PrivateNumber }- deriving (Show, Read, Eq, Data)+ deriving (Read, Eq, Data) +-- | The curve is shown; @private_d@ is not. Use+-- 'Crypto.Debug.debugShow' to see it.+instance Show PrivateKey where+ showsPrec d k =+ showParen (d > 10) $+ showString "PrivateKey {private_curve = "+ . shows (private_curve k)+ . showString ", private_d = <secret>}"++instance DebugShow PrivateKey where+ debugShow k =+ showString "PrivateKey {private_curve = "+ . shows (private_curve k)+ . showString ", private_d = "+ . shows (private_d k)+ . showChar '}'+ $ ""+ -- | ECDSA Public Key. data PublicKey = PublicKey { public_curve :: Curve@@ -77,8 +100,27 @@ -- | ECDSA Key Pair. data KeyPair = KeyPair Curve PublicPoint PrivateNumber- deriving (Show, Read, Eq, Data)+ deriving (Read, Eq, Data) +instance Show KeyPair where+ showsPrec d (KeyPair c q _) =+ showParen (d > 10) $+ showString "KeyPair "+ . showsPrec 11 c+ . showChar ' '+ . showsPrec 11 q+ . showString " <secret>"++instance DebugShow KeyPair where+ debugShow (KeyPair c q x) =+ showString "KeyPair "+ . showsPrec 11 c+ . showChar ' '+ . showsPrec 11 q+ . showChar ' '+ . showsPrec 11 x+ $ ""+ -- | Public key of a ECDSA Key pair. toPublicKey :: KeyPair -> PublicKey toPublicKey (KeyPair curve pub _) = PublicKey curve pub@@ -103,8 +145,10 @@ let z = dsaTruncHashDigest digest n CurveCommon _ _ g n _ = common_curve curve (i, r, p) <- pointDecompose curve $ pointMul curve k g- kInv <- inverse k n- let s = kInv * (z + r * d) `mod` n+ kInv <- scalarInverse curve k+ -- kInv and d are secret, so the arithmetic that mixes them goes through+ -- the curve's own, which on P-256 is the C implementation's+ let s = scalarMul curve kInv (scalarAdd curve z (scalarMul curve r d)) when (r == 0 || s == 0) Nothing return $ if s <= n `unsafeShiftR` 1
@@ -36,6 +36,7 @@ scalarZero, scalarN, scalarIsZero,+ scalarReduce, scalarAdd, scalarSub, scalarMul,@@ -150,17 +151,17 @@ withScalar n1 $ \pn1 -> withScalar n2 $ \pn2 -> withPoint p $ \px py -> ccrypton_p256_points_mul_vartime pn1 pn2 px py dx dy --- | Check if a 'Point' is valid+-- | Check if a t'Point' is valid pointIsValid :: Point -> Bool pointIsValid p = unsafeDoIO $ withPoint p $ \px py -> do r <- ccrypton_p256_is_valid_point px py return (r /= 0) --- | Check if a 'Point' is the point at infinity+-- | Check if a t'Point' is the point at infinity pointIsAtInfinity :: Point -> Bool pointIsAtInfinity (Point b) = constAllZero b --- | Return the x coordinate as a 'Scalar' if the point is not at infinity+-- | Return the x coordinate as a t'Scalar' if the point is not at infinity pointX :: Point -> Maybe Scalar pointX p | pointIsAtInfinity p = Nothing@@ -251,6 +252,17 @@ result <- ccrypton_p256_is_zero d return $ result /= 0 +-- | Bring a scalar below the order of the curve+--+-- 'scalarFromInteger' and 'scalarFromBinary' take any 256 bits, so a scalar+-- can arrive above the order; the arithmetic below wants it brought down+-- first. Twice the order is more than 256 bits hold, so this is a single+-- subtraction, taken or not through a mask rather than a branch.+scalarReduce :: Scalar -> Scalar+scalarReduce a =+ withNewScalarFreeze $ \d -> withScalar a $ \pa ->+ ccrypton_p256_mod ccrypton_SECP256r1_n pa d+ -- | Perform addition between two scalars -- -- > a + b@@ -401,8 +413,6 @@ foreign import ccall "crypton_p256e_scalar_invert" ccrypton_p256e_scalar_invert :: Ptr P256Scalar -> Ptr P256Scalar -> IO () --- foreign import ccall "crypton_p256_modinv"--- ccrypton_p256_modinv :: Ptr P256Scalar -> Ptr P256Scalar -> Ptr P256Scalar -> IO () foreign import ccall "crypton_p256_modinv_vartime" ccrypton_p256_modinv_vartime :: Ptr P256Scalar -> Ptr P256Scalar -> Ptr P256Scalar -> IO ()
@@ -1,8 +1,15 @@+{-# LANGUAGE BangPatterns #-}+ -- | Elliptic Curve Arithmetic. ----- /WARNING:/ These functions are vulnerable to timing attacks.+-- /WARNING:/ These functions are vulnerable to timing attacks, except on+-- P-256, whose multiplications go to the C implementation in+-- "Crypto.PubKey.ECC.P256". module Crypto.PubKey.ECC.Prim ( scalarGenerate,+ scalarInverse,+ scalarAdd,+ scalarMul, pointAdd, pointNegate, pointDouble,@@ -13,21 +20,137 @@ pointCompose, isPointAtInfinity, isPointValid,+ isPointInSubgroup, ) where +import Crypto.Error (maybeCryptoError)+import Crypto.Internal.ECC (CurveField (..), MulResult (..), curveMul)+import Crypto.Number.Basic (numBits) import Crypto.Number.F2m import Crypto.Number.Generate (generateBetween) import Crypto.Number.ModArithmetic+import qualified Crypto.PubKey.ECC.P256 as P256 import Crypto.PubKey.ECC.Types import Crypto.Random+import Data.Bits (shiftL, shiftR, testBit, (.&.)) import Data.Maybe +-- | P-256, the one curve here that has a C implementation: 'SEC_p256r1', also+-- known as NIST P-256 and prime256v1.+--+-- A 'Curve' carries its parameters rather than a name, so this compares the+-- parameters. They are public, so the comparison tells an attacker nothing.+p256Curve :: Curve+p256Curve = getCurveByName SEC_p256r1+{-# NOINLINE p256Curve #-}++p256Order :: Integer+p256Order = ecc_n (common_curve p256Curve)++p256Base :: Point+p256Base = ecc_g (common_curve p256Curve)++-- | A point the C implementation will take: in range, on the curve, and not+-- the point at infinity, which it does not represent. Anything else is left+-- to the generic code, which answers for points off the curve too.+toP256 :: Point -> Maybe P256.Point+toP256 PointO = Nothing+toP256 (Point x y)+ | x < 0 || y < 0 || x >= limit || y >= limit = Nothing+ | P256.pointIsValid p = Just p+ | otherwise = Nothing+ where+ limit = 1 `shiftL` 256+ p = P256.pointFromIntegers (x, y)++fromP256 :: P256.Point -> Point+fromP256 p+ | P256.pointIsAtInfinity p = PointO+ | otherwise = uncurry Point (P256.pointToIntegers p)++-- | Any 256-bit number as a scalar.+--+-- The arithmetic below takes them as they come: a 256-bit value is barely+-- over the order, and both the multiplication and the addition bring their+-- answer back under it. 'Nothing' is for what does not fit in 256 bits,+-- which no scalar anybody signs with does.+p256Scalar :: Integer -> Maybe P256.Scalar+p256Scalar n+ | n < 0 || n >= 1 `shiftL` 256 = Nothing+ | otherwise = maybeCryptoError (P256.scalarFromInteger n)++-- | The scalar reduced into the range the C implementation takes.+--+-- Every point it accepts has the curve's order, so reducing changes no+-- answer; 'Nothing' means the multiple is the point at infinity, which is the+-- generic code's business.+-- The reduction is a single masked subtraction, so a secret scalar does not+-- steer it, which taking the remainder would: dividing takes a number of+-- steps that follows the number being divided. Anything wider than 256 bits+-- has to go through a division first, but a scalar that wide is not one+-- anybody signs with.+toP256Scalar :: Integer -> Maybe P256.Scalar+toP256Scalar n = case P256.scalarReduce <$> p256Scalar n of+ Nothing -> toP256Scalar (n `mod` p256Order) -- wider than 256 bits, or below zero+ Just s+ | P256.scalarIsZero s -> Nothing+ | otherwise -> Just s++-- | @n1 * p1 + n2 * p2@ through the C implementation, when one of the points+-- is the base point. That is the shape signature verification uses.+p256AddTwoMuls :: Integer -> Point -> Integer -> Point -> Maybe Point+p256AddTwoMuls n1 p1 n2 p2+ | p1 == p256Base = withBase n1 n2 p2+ | p2 == p256Base = withBase n2 n1 p1+ | otherwise = Nothing+ where+ withBase a b q =+ fromP256+ <$> (P256.pointsMulVarTime <$> toP256Scalar a <*> toP256Scalar b <*> toP256 q)+ -- | Generate a valid scalar for a specific Curve scalarGenerate :: MonadRandom randomly => Curve -> randomly PrivateNumber scalarGenerate curve = generateBetween 1 (n - 1) where n = ecc_n $ common_curve curve +-- | The inverse of a scalar modulo the order of the curve, without letting+-- the scalar steer how long the work takes. This is what signing needs for+-- its nonce, which is as worth hiding as the private key itself: a handful of+-- signatures whose nonces are partly known give the key away.+--+-- On P-256 the C implementation does it; elsewhere it is 'inverseSafe'.+-- 'Nothing' means the scalar has no inverse, which for the curves in use here+-- means it was a multiple of the order.+scalarInverse :: Curve -> Integer -> Maybe Integer+scalarInverse c k+ | c == p256Curve+ , Just s <- toP256Scalar k =+ Just (P256.scalarToInteger (P256.scalarInvSafe s))+ | otherwise = inverseSafe k (ecc_n $ common_curve c)++-- | Addition modulo the order of the curve.+--+-- On P-256 this is the C implementation's arithmetic, which works in a fixed+-- width and so does not let the values steer it; elsewhere it is 'Integer'+-- arithmetic, whose cost follows the values.+scalarAdd :: Curve -> Integer -> Integer -> Integer+scalarAdd c a b+ | c == p256Curve+ , Just x <- p256Scalar a+ , Just y <- p256Scalar b =+ P256.scalarToInteger (P256.scalarAdd x y)+ | otherwise = (a + b) `mod` ecc_n (common_curve c)++-- | Multiplication modulo the order of the curve, as 'scalarAdd'.+scalarMul :: Curve -> Integer -> Integer -> Integer+scalarMul c a b+ | c == p256Curve+ , Just x <- p256Scalar a+ , Just y <- p256Scalar b =+ P256.scalarToInteger (P256.scalarMul x y)+ | otherwise = (a * b) `mod` ecc_n (common_curve c)+ -- TODO: Extract helper function for `fromMaybe PointO...` -- | Elliptic Curve point negation:@@ -98,47 +221,258 @@ -- | Elliptic curve point multiplication using the base ----- /WARNING:/ Vulnerable to timing attacks.+-- On P-256 this reaches the C implementation, which multiplies the base point+-- through a table of its own.+--+-- /WARNING:/ On every other curve, vulnerable to timing attacks. pointBaseMul :: Curve -> Integer -> Point pointBaseMul c n = pointMul c n (ecc_g $ common_curve c) --- | Elliptic curve point multiplication (double and add algorithm).+-- | Elliptic curve point multiplication. ----- /WARNING:/ Vulnerable to timing attacks.+-- Over a prime field this goes to C, four bits of scalar at a time, with the+-- multiple to add taken from a table read by touching every entry of it.+-- Over a binary field it also goes to C, as Montgomery's ladder: it carries+-- the x coordinates of two consecutive multiples -- their difference being+-- the point is what lets it carry no more than that -- and spends one+-- addition and one doubling on every bit whichever way the bit goes, with the+-- two exchanged by a mask rather than chosen by a branch. Either way the work+-- follows the width of the curve's order and not the scalar.+--+-- What falls back on the 'Integer' arithmetic below is a point that is not on+-- the curve, the one point of a binary curve that has no x, and a prime the C+-- will not take.+--+-- Multiplying the base point of a curve over a prime field -- which is what+-- signing and making a key do, and nothing else does -- goes through a table+-- of its multiples, built when that curve is first asked for one and kept+-- afterwards. The build is a few milliseconds and the table a few hundred+-- kilobytes, and a multiplication that uses it takes about a third of what+-- one without it takes.+--+-- On P-256 the multiplication goes to the C implementation in+-- "Crypto.PubKey.ECC.P256", which has a table for the base point.+--+-- /WARNING:/ What is left of the 'Integer' arithmetic below -- a point off+-- the curve, the one point of a binary curve with no x, a prime or a+-- polynomial the C will not take -- has uniform operation counts at best, and+-- uniform operation counts are not constant time: those operations cost what+-- the values they are given cost. pointMul :: Curve -> Integer -> Point -> Point pointMul _ _ PointO = PointO pointMul c n p+ -- the base point has a table of its own in the C, which is what makes key+ -- generation and signing quicker than multiplying any other point+ | c == p256Curve+ , p == p256Base =+ maybe PointO (fromP256 . P256.toPoint) (toP256Scalar n)+ | c == p256Curve+ , Just q <- toP256 p =+ maybe PointO (\s -> fromP256 (P256.pointMul s q)) (toP256Scalar n) | n < 0 = pointMul c (-n) (pointNegate c p) | n == 0 = PointO- | n == 1 = p- | odd n = pointAdd c p (pointMul c (n - 1) p)- | otherwise = pointMul c (n `div` 2) (pointDouble c p)+ | otherwise =+ case c of+ CurveFP (CurvePrime pr cc) -> primeMul pr cc+ CurveF2m (CurveBinary fx cc) -> binaryMul fx cc+ where+ -- The C answers for a point on the curve; anything else keeps the+ -- answers it has always had from the code below. Multiplying the base+ -- point, which is what signing and making a key do, goes through the+ -- table kept for it.+ primeMul pr cc = case p of+ Point px py+ | isPointValid c p ->+ answer slow $+ curveMul+ (Prime pr (ecc_a cc) (ecc_b cc))+ (ecc_n cc)+ n+ px+ py+ (p == ecc_g cc)+ _ -> slow+ where+ slow =+ jacobianMul+ pr+ (ecc_a cc)+ (max (integerBits n) (integerBits (ecc_n cc)))+ n+ p --- | Elliptic curve double-scalar multiplication (uses Shamir's trick).+ -- The ladder answers for a point on the curve that has an x; the one+ -- point with no x, and anything off the curve, keep what they had.+ binaryMul fx cc = case p of+ Point px py+ | isPointValid c p ->+ answer (affineMul n p) $+ curveMul (Binary fx (ecc_b cc)) (ecc_n cc) n px py False+ _ -> affineMul n p++ -- what the C could not take goes back to the code that was here before+ answer fallback r = case r of+ MulPoint x y -> Point x y+ MulInfinity -> PointO+ MulUnsupported -> fallback++ affineMul k q+ | k == 0 = PointO+ | k == 1 = q+ | odd k = pointAdd c q (affineMul (k - 1) q)+ | otherwise = affineMul (k `div` 2) (pointDouble c q)++-- | Number of bits needed to write n, for n > 0.+integerBits :: Integer -> Int+integerBits = go 0+ where+ go acc 0 = acc+ go acc k = go (acc + 1) (k `div` 2)++-- | A point in Jacobian coordinates: @(X, Y, Z)@ stands for the affine+-- @(X\/Z^2, Y\/Z^3)@, and @JPointO@ for the point at infinity.+data JPoint = JPointO | JPoint !Integer !Integer !Integer++-- | The field a prime curve works in, and how to reduce into it. --+-- Most curve primes are @2^k - c@ with @c@ far smaller than the prime.+-- Reducing is then a shift, a multiplication by @c@ and an addition, where+-- dividing a number twice the width costs about four times as much: 227ns+-- against 183 for a P-384 multiplication, and 226 against 89 for P-521, whose+-- @c@ is one.+-- | The prime, the width to fold at, and what to fold back in. A @c@ of zero+-- says to divide instead, either because the prime has no such shape or+-- because it is too small for folding to pay: @c@ has to be under half the+-- width, or folding would not shrink the number, and below 256 bits the+-- handful of 'Integer' operations folding takes costs more than the division+-- it saves -- measured on P-192, where folding is 14% slower.+data Field = Field !Integer !Int !Integer++mkField :: Integer -> Field+mkField p+ | p > 0 && c > 0 && 2 * numBits c <= k && k >= 256 = Field p k c+ | otherwise = Field p 0 0+ where+ k = numBits p+ c = (1 `shiftL` k) - p++fieldPrime :: Field -> Integer+fieldPrime (Field p _ _) = p++fieldReduce :: Field -> Integer -> Integer+fieldReduce (Field p k c) x+ | c == 0 || x < 0 = x `mod` p+ | otherwise = trim (fold x)+ where+ mask = (1 `shiftL` k) - 1+ fold v+ | v > mask = fold ((v `shiftR` k) * c + (v .&. mask))+ | otherwise = v+ trim v+ | v >= p = trim (v - p)+ | otherwise = v+{-# INLINE fieldReduce #-}++-- | A point in affine coordinates: the second operand of every addition a+-- scalar multiplication makes, where knowing that z is one saves four+-- multiplications of the sixteen.+data Affine = AffineO | Affine !Integer !Integer++jacobianMul :: Integer -> Integer -> Int -> Integer -> Point -> Point+jacobianMul _ _ _ _ PointO = PointO+jacobianMul pr a bits n (Point px py) = fromJacobian f (go (bits - 1) JPointO)+ where+ f = mkField pr+ base = Affine px py++ -- The bangs are what make the addition happen at every bit. Without+ -- them the one that is not taken stays a thunk and is never worked out,+ -- so the multiplication costs a step for every bit that is set rather+ -- than for every bit there is, and a single measurement tells an attacker+ -- how many bits of the scalar are set.+ go i acc+ | i < 0 = acc+ | otherwise =+ let !d = jDouble f a acc+ !s = jAddAffine f a d base+ in go (i - 1) (if testBit n i then s else d)++jDouble :: Field -> Integer -> JPoint -> JPoint+jDouble _ _ JPointO = JPointO+jDouble f a (JPoint x y z)+ | y == 0 = JPointO+ | otherwise = JPoint x3 y3 z3+ where+ red = fieldReduce f+ yy = red (y * y)+ delta = red (4 * x * yy)+ zz = red (z * z)+ m = red (3 * x * x + a * zz * zz)+ x3 = red (m * m - 2 * delta)+ y3 = red (m * (delta - x3) - 8 * yy * yy)+ z3 = red (2 * y * z)++-- | Add a point whose z is one, which is what a scalar multiplication always+-- adds: u1 is x1, s1 is y1, and z3 is one multiplication rather than two.+jAddAffine :: Field -> Integer -> JPoint -> Affine -> JPoint+jAddAffine _ _ p AffineO = p+jAddAffine _ _ JPointO (Affine x2 y2) = JPoint x2 y2 1+jAddAffine f a p@(JPoint x1 y1 z1) (Affine x2 y2)+ | h /= 0 = JPoint x3 y3 z3+ | r /= 0 = JPointO+ | otherwise = jDouble f a p+ where+ red = fieldReduce f+ z1s = red (z1 * z1)+ u2 = red (x2 * z1s)+ s2 = red (y2 * z1s * z1)+ h = red (u2 - x1)+ r = red (s2 - y1)+ h2 = red (h * h)+ h3 = red (h2 * h)+ x3 = red (r * r - h3 - 2 * x1 * h2)+ y3 = red (r * (x1 * h2 - x3) - y1 * h3)+ z3 = red (h * z1)++fromJacobian :: Field -> JPoint -> Point+fromJacobian _ JPointO = PointO+fromJacobian f (JPoint x y z) =+ case inverse z (fieldPrime f) of+ Nothing -> PointO+ Just zi ->+ let red = fieldReduce f+ zi2 = red (zi * zi)+ in Point (red (x * zi2)) (red (y * zi2 * zi))++-- | Elliptic curve double-scalar multiplication.+-- -- > pointAddTwoMuls c n1 p1 n2 p2 == pointAdd c (pointMul c n1 p1) -- > (pointMul c n2 p2) --+-- which, apart from P-256, is how it is done: the two multiplications+-- separately, and then one addition. P-256 has a double multiplication of+-- its own in C and takes it.+--+-- This used to be Shamir's trick, one pass over the bits of both scalars at+-- once, which shares the doublings between them and is the right thing to do+-- when the two multiplications would cost the same. They no longer do.+-- 'pointMul' goes to C, and over a prime field it multiplies the base point+-- through a table of its multiples, which is a third of the price of an+-- ordinary multiplication -- and the base point is one of the two here, since+-- signature verification is what asks for this. Sharing the doublings with a+-- pass in 'Integer' arithmetic gives that up and more: on P-384 it costs+-- twice what two multiplications in C cost, and on the curves over a binary+-- field, whose addition needs an inversion where the C has a ladder that+-- needs none, it costs two hundred times as much.+--+-- Both scalars are public wherever this is called from, so nothing here is+-- meant to hide them.+-- -- /WARNING:/ Vulnerable to timing attacks. pointAddTwoMuls :: Curve -> Integer -> Point -> Integer -> Point -> Point-pointAddTwoMuls _ _ PointO _ PointO = PointO-pointAddTwoMuls c _ PointO n2 p2 = pointMul c n2 p2-pointAddTwoMuls c n1 p1 _ PointO = pointMul c n1 p1 pointAddTwoMuls c n1 p1 n2 p2- | n1 < 0 = pointAddTwoMuls c (-n1) (pointNegate c p1) n2 p2- | n2 < 0 = pointAddTwoMuls c n1 p1 (-n2) (pointNegate c p2)- | otherwise = go (n1, n2)- where- p0 = pointAdd c p1 p2-- go (0, 0) = PointO- go (k1, k2) =- let q = pointDouble c $ go (k1 `div` 2, k2 `div` 2)- in case (odd k1, odd k2) of- (True, True) -> pointAdd c p0 q- (True, False) -> pointAdd c p1 q- (False, True) -> pointAdd c p2 q- (False, False) -> q+ | c == p256Curve, Just r <- p256AddTwoMuls n1 p1 n2 p2 = r+ | otherwise = pointAdd c (pointMul c n1 p1) (pointMul c n2 p2) -- | Decompose a point into index, residue, and parity. --@@ -184,6 +518,33 @@ -- * x is not out of range -- * y is not out of range -- * the equation @y^2 = x^3 + a*x + b (mod p)@ holds+--+-- over a prime curve, and the corresponding checks over a binary curve: the+-- coordinates reduce to themselves in the field, and+-- @y^2 + x*y = x^3 + a*x^2 + b@ holds.+--+-- This is the check to make on a point that arrives from elsewhere, before+-- multiplying it by a private number. Without it the multiplication is+-- carried out in whatever group the supplied point generates rather than the+-- curve group, and if that group is small the private number can be recovered+-- from the result.+--+-- Two things it does not establish:+--+-- * The point at infinity is reported as valid, since it is a member of the+-- curve group. It is not a usable peer value: multiplying it by anything+-- yields the point at infinity again, which has no coordinates. Reject it+-- separately where a peer is not allowed to send it.+--+-- * Being on the curve is not membership of the subgroup generated by the base+-- point. The two coincide only when the cofactor is 1. Of the curves in+-- 'Crypto.PubKey.ECC.Types.CurveName' that holds for every prime curve+-- except @SEC_p112r2@ and @SEC_p128r2@, whose cofactor is 4, and for no+-- binary curve, whose cofactor is 2 or 4. Where the cofactor is above 1 a+-- point on the curve may still generate a small subgroup, and ruling that+-- out needs a further check -- multiplying by the group order and requiring+-- the point at infinity, or clearing the cofactor -- that this function does+-- not make. isPointValid :: Curve -> Point -> Bool isPointValid _ PointO = True isPointValid (CurveFP (CurvePrime p cc)) (Point x y) =@@ -205,6 +566,30 @@ add = addF2m mul = mulF2m fx isValid e = modF2m fx e == e++-- | Check that a point is in the subgroup the base point generates, which is+-- the further check 'isPointValid' does not make. A point that is on the+-- curve but outside that subgroup answers a multiplication modulo an order+-- smaller than the group's, so the multiplier -- a private number, where the+-- point came from a peer -- is revealed modulo that small order.+--+-- Where the cofactor is 1 the subgroup is the whole curve group and the+-- answer is 'True' for any point on the curve, at no cost. Otherwise the+-- point is multiplied by the group order and the answer is whether that+-- reaches the point at infinity, which costs one scalar multiplication. This+-- is the check OpenSSL's @EC_KEY_check_key@ makes.+--+-- The point at infinity is reported as in the subgroup, as 'isPointValid'+-- reports it valid; it is a member, and unusable for other reasons.+--+-- A point that is not on the curve at all has no meaningful answer here, so+-- check 'isPointValid' first.+isPointInSubgroup :: Curve -> Point -> Bool+isPointInSubgroup curve p+ | ecc_h cc == 1 = True+ | otherwise = pointMul curve (ecc_n cc) p == PointO+ where+ cc = common_curve curve -- | div and mod divmod :: Integer -> Integer -> Integer -> Maybe Integer
@@ -135,6 +135,19 @@ | SEC_t571r1 deriving (Show, Read, Eq, Ord, Enum, Bounded, Data) +{-# DEPRECATED+ SEC_t113r1, SEC_t113r2, SEC_t131r1, SEC_t131r2, SEC_t163k1, SEC_t163r1,+ SEC_t163r2, SEC_t193r1, SEC_t193r2, SEC_t233k1, SEC_t233r1, SEC_t239k1,+ SEC_t283k1, SEC_t283r1, SEC_t409k1, SEC_t409r1, SEC_t571k1, SEC_t571r1+ [ "This curve is over a binary field, and those are obsolete."+ , "They are also the curves whose cofactor is not 1, so a point from"+ , "a peer needs the subgroup check that costs a further scalar"+ , "multiplication; pyca/cryptography deprecated them for removal in"+ , "the release that fixed CVE-2026-26007. This one will go in a"+ , "later major version of crypton. Prefer a prime curve, or X25519."+ ]+ #-}+ {- curvesOIDs :: [ (CurveName, [Integer]) ] curvesOIDs =
@@ -48,6 +48,11 @@ signDigest, verify, verifyDigest,++ -- * Deterministic nonces+ deterministicNonce,+ signDeterministic,+ signDigestDeterministic, ) where import Control.Monad@@ -58,8 +63,10 @@ import Crypto.Hash import Crypto.Internal.ByteArray (ByteArray, ByteArrayAccess) import Crypto.Internal.Imports-import Crypto.Number.ModArithmetic (inverseFermat)+import Crypto.Number.Generate (generatePrefix)+import Crypto.Number.ModArithmetic (inverseSafe) import qualified Crypto.PubKey.ECC.P256 as P256+import Crypto.Random.HmacDRG (initial, update) import Crypto.Random.Types import Data.Bits@@ -257,6 +264,75 @@ verify prx hashAlg q sig msg = verifyDigest prx q sig (hashWith hashAlg msg) -- | Truncate a digest based on curve order size.+-- | Deterministic nonce generation according to RFC 6979.+--+-- The nonce is derived from the private key and the message alone, so a+-- signature made this way needs no random number generator and cannot be the+-- one that repeats a nonce -- which, for ECDSA, hands over the private key.+--+-- The hash used to seed the generator is given separately from the one the+-- message was digested with, as RFC 6979 allows.+--+-- The last argument is what to do with a candidate nonce. It may answer+-- 'Nothing', in which case another candidate is drawn, which is what+-- 'signDigestDeterministic' does for the r or s that comes out zero:+--+-- > deterministicNonce prx SHA256 priv digest (\k -> signDigestWith prx k priv digest)+deterministicNonce+ :: (EllipticCurveECDSA curve, HashAlgorithm hashDRG, HashAlgorithm hashDigest)+ => proxy curve+ -> hashDRG+ -> PrivateKey curve+ -> Digest hashDigest+ -> (Scalar curve -> Maybe a)+ -> a+deterministicNonce prx alg d digest go = fst $ withDRG state run+ where+ state = update seed $ initial alg+ -- RFC 6979 section 3.2 step d: int2octets(x) || bits2octets(h1). The+ -- second is the truncated digest taken modulo the order, which is what+ -- scalarAdd with zero does, its contract being to reduce there.+ seed =+ B.append (encodeScalar prx d) (encodeScalar prx z)+ :: B.ScrubbedBytes+ z = scalarAdd prx (tHashDigest prx digest) zeroScalar+ zeroScalar = throwCryptoError $ scalarFromInteger prx 0+ run = do+ k <- generatePrefix (curveOrderBits prx)+ case scalarFromInteger prx k of+ CryptoPassed s+ | scalarIsValid prx s -> maybe run pure (go s)+ _ -> run++-- | Sign a digest with a nonce derived from the private key and the digest,+-- as RFC 6979 says, rather than from a random number generator.+signDigestDeterministic+ :: (EllipticCurveECDSA curve, HashAlgorithm hashDRG, HashAlgorithm hashDigest)+ => proxy curve+ -> hashDRG+ -> PrivateKey curve+ -> Digest hashDigest+ -> Signature curve+signDigestDeterministic prx alg d digest =+ deterministicNonce prx alg d digest $ \k -> signDigestWith prx k d digest++-- | Sign a message with a nonce derived from the private key and the message,+-- as RFC 6979 says, rather than from a random number generator.+signDeterministic+ :: ( EllipticCurveECDSA curve+ , HashAlgorithm hashDRG+ , HashAlgorithm hash+ , ByteArrayAccess msg+ )+ => proxy curve+ -> hashDRG+ -> PrivateKey curve+ -> hash+ -> msg+ -> Signature curve+signDeterministic prx alg d hashAlg msg =+ signDigestDeterministic prx alg d (hashWith hashAlg msg)+ tHashDigest :: (EllipticCurveECDSA curve, HashAlgorithm hash) => proxy curve -> Digest hash -> Scalar curve@@ -296,15 +372,16 @@ => Simple.Scalar curve -> Bool ecScalarIsZero (Simple.Scalar a) = a == 0 +-- | 'inverseSafe' is the one that takes a fixed number of division steps+-- where the assembly for them is built, and checks whatever it gets by+-- multiplying out. It answers 'Nothing' exactly where the exponentiation+-- this used to do answered zero. ecScalarInv :: Simple.Curve c => proxy c -> Simple.Scalar c -> Maybe (Simple.Scalar c)-ecScalarInv prx (Simple.Scalar s)- | i == 0 = Nothing- | otherwise = Just $ Simple.Scalar i+ecScalarInv prx (Simple.Scalar s) = Simple.Scalar <$> inverseSafe s n where n = Simple.curveEccN $ Simple.curveParameters prx- i = inverseFermat s n ecPointX :: Simple.Curve c
@@ -32,6 +32,7 @@ generateSecretKey, ) where +import Crypto.Debug (DebugShow (..), debugShowBytes) import Data.Word import Foreign.C.Types import Foreign.Ptr@@ -51,6 +52,9 @@ -- | An Ed25519 Secret key newtype SecretKey = SecretKey ScrubbedBytes deriving (Show, Eq, ByteArrayAccess, NFData)++instance DebugShow SecretKey where+ debugShow = debugShowBytes "SecretKey" -- | An Ed25519 public key newtype PublicKey = PublicKey Bytes
@@ -36,6 +36,7 @@ generateSecretKey, ) where +import Crypto.Debug (DebugShow (..), debugShowBytes) import Data.Word import Foreign.C.Types import Foreign.Ptr@@ -55,6 +56,9 @@ -- | An Ed448 Secret key newtype SecretKey = SecretKey ScrubbedBytes deriving (Show, Eq, ByteArrayAccess, NFData)++instance DebugShow SecretKey where+ debugShow = debugShowBytes "SecretKey" -- | An Ed448 public key newtype PublicKey = PublicKey Bytes
@@ -51,6 +51,7 @@ generateSecretKey, ) where +import Crypto.Debug (DebugShow (..), debugShowBytes) import Data.Bits import Data.ByteArray ( ByteArray,@@ -85,6 +86,9 @@ newtype SecretKey curve = SecretKey ScrubbedBytes deriving (Show, Eq, ByteArrayAccess, NFData) +instance DebugShow (SecretKey curve) where+ debugShow = debugShowBytes "SecretKey"+ -- | An EdDSA public key newtype PublicKey curve hash = PublicKey Bytes deriving (Show, Eq, ByteArrayAccess, NFData)@@ -390,11 +394,13 @@ => proxy curve -> Signature curve hash -> CryptoFailable (Bytes, Point curve, Scalar curve)-decodeSignature prx (Signature bs) = do+decodeSignature prx sig@(Signature bs) = do let (bsR, bsS) = B.splitAt (publicKeySize prx) bs pR <- decodePoint prx bsR sS <- decodeScalarLE prx bsS- return (bsR, pR, sS)+ if encodeSignature prx (encodePoint prx pR, pR, sS) == sig+ then return (bsR, pR, sS)+ else CryptoFailed CryptoError_PointFormatInvalid -- implementations are supposed to decode any scalar up to the size of the digest decodeScalarNoErr
@@ -1,3 +1,4 @@+{-# LANGUAGE DeriveDataTypeable #-} {-# LANGUAGE GeneralizedNewtypeDeriving #-} -- |@@ -7,18 +8,43 @@ -- Stability : experimental -- Portability : Good ----- This module is a work in progress. do not use:--- it might eat your dog, your data or even both.+-- ElGamal encryption and signature over the multiplicative group of integers+-- modulo a prime, reusing the parameters of "Crypto.PubKey.DH". ----- TODO: provide a mapping between integer and ciphertext--- generate numbers correctly+-- /These are raw primitives, not a scheme./ The encryption here is textbook+-- ElGamal: it applies no padding, so it is malleable by construction --+-- multiplying a ciphertext's second component by @t@ multiplies the plaintext+-- by @t@ -- and it is not IND-CCA secure. A message is an 'Integer' below the+-- modulus rather than a byte string, and nothing here maps one to the other.+-- Use it to build a scheme that adds those, or prefer+-- "Crypto.PubKey.RSA.OAEP" or "Crypto.PubKey.ECIES" where a scheme is what is+-- wanted.+--+-- The signature primitive is likewise raw, and an ephemeral value must never+-- be reused between signatures: two signatures under the same @k@ reveal the+-- private key.+--+-- == What is kept from the clock, and what is not+--+-- Every exponentiation with a secret exponent is+-- 'Crypto.Number.ModArithmetic.expSafe'. Decryption inverts the shared+-- secret by Fermat's little theorem rather than by the extended Euclidean+-- algorithm, whose steps follow the bits it is given. 'sign' cannot do that+-- -- @k@ is inverted modulo @p-1@, which is even -- so it blinds instead: the+-- algorithm is handed @k@ times a fresh random unit, and the blinder is+-- divided out afterwards. 'signWith', having no randomness of its own, hands+-- it @k@.+--+-- What is left is the 'Integer' arithmetic around all of that, whose cost+-- follows the size of the numbers. See "Crypto.PubKey.DSA" for the same note+-- at more length. module Crypto.PubKey.ElGamal ( Params, PublicNumber, PrivateNumber, EphemeralKey (..), SharedKey,- Signature,+ Signature (..), -- * Generation generatePrivate,@@ -31,18 +57,22 @@ -- * Signature primitives signWith,+ signDigestWith, sign,+ signDigest, -- * Verification primitives verify,+ verifyDigest, ) where +import Crypto.Error import Crypto.Hash import Crypto.Internal.ByteArray (ByteArrayAccess) import Crypto.Internal.Imports import Crypto.Number.Basic (gcde)-import Crypto.Number.Generate (generateMax)-import Crypto.Number.ModArithmetic (expFast, expSafe, inverse)+import Crypto.Number.Generate (generateBetween, generateMax)+import Crypto.Number.ModArithmetic (expFast, expSafe, inverseSafe) import Crypto.Number.Serialize (os2ip) import Crypto.PubKey.DH ( Params (..),@@ -51,38 +81,59 @@ SharedKey (..), ) import Crypto.Random.Types-import Data.Maybe (fromJust)+import Data.Data -- | ElGamal Signature-data Signature = Signature (Integer, Integer)+data Signature = Signature+ { sign_r :: Integer+ -- ^ ElGamal r+ , sign_s :: Integer+ -- ^ ElGamal s+ }+ deriving (Show, Read, Eq, Data) +instance NFData Signature where+ rnf (Signature r s) = r `seq` s `seq` ()+ -- | ElGamal Ephemeral key. also called Temporary key. newtype EphemeralKey = EphemeralKey Integer deriving (NFData) --- | generate a private number with no specific property--- this number is usually called a and need to be between--- 0 and q (order of the group G).+-- | generate a private number, in @[1, q-1]@ where @q@ is the order of the+-- group. Zero is excluded: it would make the public number 1 and the shared+-- value constant. generatePrivate :: MonadRandom m => Integer -> m PrivateNumber-generatePrivate q = PrivateNumber <$> generateMax q---- | generate an ephemeral key which is a number with no specific property,--- and need to be between 0 and q (order of the group G).-generateEphemeral :: MonadRandom m => Integer -> m EphemeralKey-generateEphemeral q = toEphemeral <$> generatePrivate q- where- toEphemeral (PrivateNumber n) = EphemeralKey n+generatePrivate q = PrivateNumber <$> generateBetween 1 (q - 1) -- | generate a public number that is for the other party benefits. -- this number is usually called h=g^a generatePublic :: Params -> PrivateNumber -> PublicNumber generatePublic (Params p g _) (PrivateNumber a) = PublicNumber $ expSafe g a p +-- | Is the other party's public number usable?+--+-- @1@ and @p-1@ generate a group of one or two elements, so the value they+-- mask the message with is one of a handful of constants.+validPublic :: Integer -> Integer -> Bool+validPublic p h = h > 1 && h < p - 1+ -- | encrypt with a specified ephemeral key--- do not reuse ephemeral key.+--+-- The ephemeral key must lie in @[1, p-2]@ and must never be reused: zero+-- would leave the message unmasked, and a repeat lets anyone who learns one+-- plaintext recover the other. A message must be below the modulus, or+-- decryption would return it reduced. encryptWith- :: EphemeralKey -> Params -> PublicNumber -> Integer -> (Integer, Integer)-encryptWith (EphemeralKey b) (Params p g _) (PublicNumber h) m = (c1, c2)+ :: EphemeralKey+ -> Params+ -> PublicNumber+ -> Integer+ -> CryptoFailable (Integer, Integer)+encryptWith (EphemeralKey b) (Params p g _) (PublicNumber h) m+ | b < 1 || b > p - 2 = CryptoFailed CryptoError_ParameterInvalid+ | not (validPublic p h) = CryptoFailed CryptoError_ParameterInvalid+ | m < 0 || m >= p = CryptoFailed CryptoError_ParameterInvalid+ | otherwise = CryptoPassed (c1, c2) where s = expSafe h b p c1 = expSafe g b p@@ -91,29 +142,52 @@ -- | encrypt a message using params and public keys -- will generate b (called the ephemeral key) encrypt- :: MonadRandom m => Params -> PublicNumber -> Integer -> m (Integer, Integer)-encrypt params@(Params p _ _) public m = (\b -> encryptWith b params public m) <$> generateEphemeral q- where- q = p - 1 -- p is prime, hence order of the group is p-1+ :: MonadRandom m+ => Params+ -> PublicNumber+ -> Integer+ -> m (CryptoFailable (Integer, Integer))+encrypt params@(Params p _ _) public m+ | p < 5 = return (CryptoFailed CryptoError_ParameterInvalid)+ | otherwise = do+ b <- generateBetween 1 (p - 2)+ return $ encryptWith (EphemeralKey b) params public m -- | decrypt message-decrypt :: Params -> PrivateNumber -> (Integer, Integer) -> Integer-decrypt (Params p _ _) (PrivateNumber a) (c1, c2) = (c2 * sm1) `mod` p+--+-- @c1@ must be a unit modulo @p@; a ciphertext whose first component is zero+-- or out of range is rejected rather than raising.+decrypt+ :: Params -> PrivateNumber -> (Integer, Integer) -> CryptoFailable Integer+decrypt (Params p _ _) (PrivateNumber a) (c1, c2)+ | c1 <= 0 || c1 >= p = CryptoFailed CryptoError_ParameterInvalid+ | c2 < 0 || c2 >= p = CryptoFailed CryptoError_ParameterInvalid+ | otherwise = case inverseSafe s p of+ Nothing -> CryptoFailed CryptoError_ParameterInvalid+ Just sm1 -> CryptoPassed ((c2 * sm1) `mod` p) where+ -- the shared secret, which the extended Euclidean algorithm would take+ -- apart: its steps follow the bits of what it is given, and this one is+ -- worth the private number. p is prime, so Fermat gives the inverse+ -- without reading it s = expSafe c1 a p- sm1 = fromJust $ inverse s p -- always inversible in Zp --- | sign a message with an explicit k number+-- | sign a message with an explicit ephemeral value ----- if k is not appropriate, then no signature is returned.+-- @k@ has to lie in @[1, p-2]@ and be coprime with @p-1@. 'Nothing' says the+-- value handed in cannot be used: either it fails one of those two conditions,+-- or it is one of the few that produce a second component of zero. Either way+-- the answer is to draw another @k@, which is what 'sign' does. ----- with some appropriate value of k, the signature generation can fail,--- and no signature is returned. User of this function need to retry--- with a different k value.+-- @k@ is an ephemeral private key. It has to be drawn uniformly at random,+-- kept secret, and used for one signature only: the private number follows+-- from a signature and its @k@, and equally from two signatures made with the+-- same @k@. None of that is visible to this function, which is why it takes+-- @k@ from the caller and checks only what it can. signWith :: (ByteArrayAccess msg, HashAlgorithm hash) => Integer- -- ^ random number k, between 0 and p-1 and gcd(k,p-1)=1+ -- ^ ephemeral value k, in [1, p-2] and coprime with p-1 -> Params -- ^ DH params (p,g) -> PrivateNumber@@ -123,21 +197,60 @@ -> msg -- ^ message to sign -> Maybe Signature-signWith k (Params p g _) (PrivateNumber x) hashAlg msg- | k >= p - 1 || d > 1 = Nothing -- gcd(k,p-1) is not 1+signWith k params priv hashAlg msg =+ signDigestWith k params priv (hashWith hashAlg msg)++-- | Sign a digest with an explicit ephemeral value.+--+-- The digest's type says which algorithm made it, so this needs nothing else+-- to name one. 'signWith' takes a @hash@ value and never reads it -- it is+-- there to fix the type -- which leaves a caller that is itself polymorphic+-- in the algorithm with nothing to pass.+signDigestWith+ :: HashAlgorithm hash+ => Integer+ -- ^ ephemeral value k, in [1, p-2] and coprime with p-1+ -> Params+ -- ^ DH params (p,g)+ -> PrivateNumber+ -- ^ DH private key+ -> Digest hash+ -- ^ digest of the message to sign+ -> Maybe Signature+signDigestWith = signWithBlinder 1++-- | The same with a blinder for the inversion of @k@.+--+-- @k@ is inverted modulo @p-1@, which is even, so Fermat's little theorem+-- does not reach it the way it reaches DSA's @k@ modulo a prime order: the+-- extended Euclidean algorithm is the only way there, and its steps follow+-- the bits of what it is given. What can be done instead is to hand it+-- something else: for a unit @b@, the inverse of @k*b@ times @b@ is the+-- inverse of @k@, and the steps then follow @k*b@, which is a fresh random+-- number. A blinder of 1 is no blinding, which is what the exported+-- 'signWith' has to do, having no randomness of its own.+--+-- When @b@ shares a factor with @p-1@ the algorithm reports it the same way+-- it reports one in @k@, and the answer is the same: draw again.+signWithBlinder+ :: HashAlgorithm hash+ => Integer -> Integer -> Params -> PrivateNumber -> Digest hash -> Maybe Signature+signWithBlinder b k (Params p g _) (PrivateNumber x) digest+ | k <= 0 || k >= p - 1 || b <= 0 || d > 1 = Nothing | s == 0 = Nothing- | otherwise = Just $ Signature (r, s)+ | otherwise = Just $ Signature r s where r = expSafe g k p- h = os2ip $ hashWith hashAlg msg+ h = os2ip digest s = ((h - x * r) * kInv) `mod` (p - 1)- (kInv, _, d) = gcde k (p - 1)+ kInv = (kbInv * b) `mod` (p - 1)+ (kbInv, _, d) = gcde ((k * b) `mod` (p - 1)) (p - 1) -- | sign message ----- This function will generate a random number, however--- as the signature might fail, the function will automatically retry--- until a proper signature has been created.+-- This function draws the ephemeral value itself, and draws a fresh one on+-- each attempt until 'signWith' accepts it, so a caller who has no particular+-- @k@ in mind should use this rather than 'signWith'. sign :: (ByteArrayAccess msg, HashAlgorithm hash, MonadRandom m) => Params@@ -149,10 +262,20 @@ -> msg -- ^ message to sign -> m Signature-sign params@(Params p _ _) priv hashAlg msg = do+sign params priv hashAlg msg = signDigest params priv (hashWith hashAlg msg)++-- | Sign a digest, drawing the ephemeral value. See 'signDigestWith' for+-- why a digest rather than a @hash@ value.+signDigest+ :: (HashAlgorithm hash, MonadRandom m)+ => Params -> PrivateNumber -> Digest hash -> m Signature+signDigest params@(Params p _ _) priv digest = do k <- generateMax (p - 1)- case signWith k params priv hashAlg msg of- Nothing -> sign params priv hashAlg msg+ -- and a blinder for the inversion of k, which is the one step here that+ -- the extended Euclidean algorithm has to do+ b <- generateMax (p - 1)+ case signWithBlinder b k params priv digest of+ Nothing -> signDigest params priv digest Just sig -> return sig -- | verify a signature@@ -164,10 +287,18 @@ -> msg -> Signature -> Bool-verify (Params p g _) (PublicNumber y) hashAlg msg (Signature (r, s))+verify params pub hashAlg msg sig =+ verifyDigest params pub (hashWith hashAlg msg) sig++-- | Verify a signature over a digest. See 'signDigestWith' for why a digest+-- rather than a @hash@ value.+verifyDigest+ :: HashAlgorithm hash+ => Params -> PublicNumber -> Digest hash -> Signature -> Bool+verifyDigest (Params p g _) (PublicNumber y) digest (Signature r s) | or [r <= 0, r >= p, s <= 0, s >= (p - 1)] = False | otherwise = lhs == rhs where- h = os2ip $ hashWith hashAlg msg+ h = os2ip digest lhs = expFast g h p rhs = (expFast y r p * expFast r s p) `mod` p
@@ -12,8 +12,7 @@ ) where import Data.Bits (shiftR)-import Data.List (foldl')-import Prelude hiding (foldl')+import qualified Data.List as L import Crypto.Hash import Crypto.Internal.ByteArray (ByteArrayAccess)@@ -22,7 +21,7 @@ -- | This is a strict version of and and' :: [Bool] -> Bool-and' l = foldl' (&&!) True l+and' l = L.foldl' (&&!) True l -- | This is a strict version of &&. (&&!) :: Bool -> Bool -> Bool
@@ -0,0 +1,656 @@+-- |+-- Module : Crypto.PubKey.MLDSA+-- License : BSD-style+-- Maintainer : Kazu Yamamoto <kazu@iij.ad.jp>+-- Stability : experimental+-- Portability : unknown+--+-- ML-DSA, the Module-Lattice-Based Digital Signature Algorithm of+-- <https://csrc.nist.gov/pubs/fips/204/final FIPS 204>, in all three+-- parameter sets.+--+-- > (vk, sk) <- generateKeyPair MLDSA65+-- > sig <- sign sk emptyContext message+-- > verify vk emptyContext message sig+--+-- What 'generateKeyPair' and 'sign' draw their randomness from is the+-- 'Crypto.Random.MonadRandom' instance in use. Its documentation says what+-- an instance of your own has to be.+--+-- The parameter set is a type, so an ML-DSA-65 key cannot be passed where+-- an ML-DSA-87 one is expected. The three are fixed by FIPS 204 and the+-- class has no other instances.+--+-- This is pure ML-DSA: the message goes in whole. The pre-hash variant+-- (HashML-DSA) is a different algorithm with a different domain separator+-- and is not offered here.+{-# LANGUAGE DataKinds #-}+{-# LANGUAGE GeneralizedNewtypeDeriving #-}+{-# LANGUAGE ScopedTypeVariables #-}++module Crypto.PubKey.MLDSA (+ -- * Parameter sets+ MLDSA44 (..),+ MLDSA65 (..),+ MLDSA87 (..),+ MLDSA (verificationKeySize, signingKeySize, signatureSize),++ -- * Keys and signatures+ VerificationKey,+ SigningKey,+ Signature,++ -- * Smart constructors+ verificationKey,+ signingKey,+ signature,++ -- * Generating a key pair+ generateKeyPair,+ generateKeyPairAndSeed,+ keyPairFromSeed,+ toPublic,++ -- * The context string+ Context,+ context,+ emptyContext,++ -- * The message representative+ Mu,+ mu,+ messageRepresentative,++ -- ** A message that does not arrive in one piece+ MuContext,+ muInit,+ muUpdate,+ muUpdates,+ muFinalize,++ -- * Signing and verifying+ sign,+ signWith,+ signDeterministic,+ verify,++ -- * Signing and verifying a message representative+ signExternalMu,+ signExternalMuWith,+ signExternalMuDeterministic,+ verifyExternalMu,++ -- * Sizes+ seedSize,+ signingRandomnessSize,+ maxContextLength,+ muSize,+) where++import Data.Proxy (Proxy (..))+import Foreign.C.Types (CInt (..), CSize (..))+import Foreign.Ptr (Ptr, nullPtr)++import Crypto.Debug (DebugShow (..), debugShowBytes)+import Crypto.Hash (Digest, hash, hashFinalize, hashInit, hashUpdate, hashUpdates)+import qualified Crypto.Hash as Hash (Context)+import Crypto.Hash.Algorithms (SHAKE256 (..))+import Crypto.Error+import Crypto.Internal.ByteArray (+ ByteArrayAccess,+ Bytes,+ ScrubbedBytes,+ withByteArray,+ )+import qualified Crypto.Internal.ByteArray as B+import Crypto.Internal.Compat (unsafeDoIO)+import Crypto.Internal.Imports+import Crypto.Random (MonadRandom, getRandomBytes)++-- | ML-DSA-44.+data MLDSA44 = MLDSA44 deriving (Show, Eq)++-- | ML-DSA-65.+data MLDSA65 = MLDSA65 deriving (Show, Eq)++-- | ML-DSA-87.+data MLDSA87 = MLDSA87 deriving (Show, Eq)++-- | The three parameter sets of FIPS 204.+--+-- Named for the algorithm rather than \"DSA\", which is a different one that+-- crypton also has, in "Crypto.PubKey.DSA". It is the three sets FIPS 204+-- defines, closed, carrying their sizes and the calls into the+-- implementation; only the sizes are exported.+class MLDSA p where+ -- | Size in bytes of a 'VerificationKey' of this parameter set.+ verificationKeySize :: proxy p -> Int++ -- | Size in bytes of a 'SigningKey' of this parameter set.+ signingKeySize :: proxy p -> Int++ -- | Size in bytes of a 'Signature' of this parameter set.+ signatureSize :: proxy p -> Int++ c_keypair :: proxy p -> Ptr Word8 -> Ptr Word8 -> Ptr Word8 -> IO CInt+ c_sign+ :: proxy p+ -> Ptr Word8 -> Ptr Word8 -> CSize -> Ptr Word8 -> CSize+ -> Ptr Word8 -> Ptr Word8 -> CInt -> IO CInt+ c_verify+ :: proxy p+ -> Ptr Word8 -> Ptr Word8 -> CSize -> Ptr Word8 -> CSize+ -> Ptr Word8 -> CInt -> IO CInt+ c_pkFromSk :: proxy p -> Ptr Word8 -> Ptr Word8 -> IO CInt++-- | A public verification key.+newtype VerificationKey p = VerificationKey Bytes+ deriving (Show, Eq, ByteArrayAccess, NFData)++-- | A private signing key.+newtype SigningKey p = SigningKey ScrubbedBytes+ deriving (Eq, ByteArrayAccess, NFData)++instance Show (SigningKey p) where+ show _ = "SigningKey <redacted>"++instance DebugShow (SigningKey p) where+ debugShow = debugShowBytes "SigningKey"++-- | A signature.+newtype Signature p = Signature Bytes+ deriving (Show, Eq, ByteArrayAccess, NFData)++-- | The context string a signature is bound to, at most+-- 'maxContextLength' bytes.+--+-- FIPS 204 mixes it into what is signed, so a signature made under one+-- context does not verify under another. Two uses of one key that don't+-- agree on a context string cannot be made to accept each other's+-- signatures. Use 'emptyContext' where there is nothing to separate -- TLS,+-- for one, signs with an empty context.+newtype Context = Context Bytes+ deriving (Show, Eq, ByteArrayAccess, NFData)++-- | The context string of length zero, which is what to sign under when+-- there is nothing to separate.+--+-- It is a context and not the absence of one: 'sign' always takes one, and+-- what this separates from is every non-empty context there is.+emptyContext :: Context+emptyContext = Context B.empty++-- | Try to build a context string.+context :: ByteArrayAccess ba => ba -> CryptoFailable Context+context bs+ | B.length bs <= maxContextLength =+ CryptoPassed $ Context $ B.copyAndFreeze bs (\_ -> return ())+ | otherwise = CryptoFailed CryptoError_ParameterInvalid++-- | The longest context string FIPS 204 allows, which is 255 bytes because+-- its length is encoded in one byte.+maxContextLength :: Int+maxContextLength = 255++-- | The message representative, @mu@ in FIPS 204: a 64-byte commitment to+-- the verification key, the context string and the message, and the only+-- part of them that signing and verification actually read.+--+-- Signing it directly is the "external mu" interface. It is for a caller+-- that has the representative without having the message in one piece: a+-- message arriving as a stream, or hashed on another machine, or by a+-- device that holds the key and is handed only this. TLS does not need it.+newtype Mu = Mu Bytes+ deriving (Show, Eq, ByteArrayAccess, NFData)++-- | Size in bytes of a 'Mu'.+muSize :: Int+muSize = 64++-- | Try to read a message representative.+mu :: ByteArrayAccess ba => ba -> CryptoFailable Mu+mu bs+ | B.length bs == muSize = CryptoPassed $ Mu $ B.copyAndFreeze bs (\_ -> return ())+ | otherwise = CryptoFailed CryptoError_ParameterInvalid++-- | Compute the message representative, for a caller that wants to make it+-- here and sign it later, or sign it elsewhere.+--+-- @'signExternalMuDeterministic' sk ('messageRepresentative' ('toPublic' sk) ctx msg)@+-- and @'signDeterministic' sk ctx msg@ are the same signature.+messageRepresentative+ :: (MLDSA p, ByteArrayAccess msg)+ => VerificationKey p -> Context -> msg -> Mu+messageRepresentative vk ctx msg = muFinalize (muUpdate (muInit vk ctx) msg)++-- | A 'Mu' being computed, with the message going in a piece at a time.+--+-- The name is not 'Context': that is ML-DSA's context string, which this+-- is built from and is not.+newtype MuContext = MuContext (Hash.Context (SHAKE256 512))++-- | Begin a message representative. The key and the context string are+-- what it is bound to, and they are all that is needed before the message.+--+-- > muFinalize (muUpdates (muInit vk ctx) chunks)+--+-- is 'messageRepresentative' of the chunks joined, so a message too large+-- to hold at once never has to be.+muInit :: MLDSA p => VerificationKey p -> Context -> MuContext+muInit vk ctx =+ -- FIPS 204: tr <- H(pk, 64) at key generation, and mu <- H(tr || M', 64)+ -- when signing, with M' the domain-separated message. Everything up to+ -- the message itself is absorbed here.+ MuContext $ hashUpdates hashInit [tr, domainPrefix ctx]+ where+ tr = B.convert (shake64 (B.convert vk :: Bytes)) :: Bytes++-- | Absorb a piece of the message.+muUpdate :: ByteArrayAccess msg => MuContext -> msg -> MuContext+muUpdate (MuContext c) msg = MuContext (hashUpdate c msg)++-- | Absorb several pieces, which is 'muUpdate' one after the other.+muUpdates :: ByteArrayAccess msg => MuContext -> [msg] -> MuContext+muUpdates (MuContext c) msgs = MuContext (hashUpdates c msgs)++-- | The message representative of everything absorbed so far.+muFinalize :: MuContext -> Mu+muFinalize (MuContext c) = Mu (B.convert (hashFinalize c :: Digest (SHAKE256 512)))++shake64 :: ByteArrayAccess ba => ba -> Digest (SHAKE256 512)+shake64 = hash++-- | Size in bytes of the seed 'keyPairFromSeed' takes, @xi@ in FIPS 204.+seedSize :: Int+seedSize = 32++-- | Size in bytes of the randomness 'signWith' takes.+signingRandomnessSize :: Int+signingRandomnessSize = 32++-- | Try to read a verification key. Only the length is checked: a+-- verification key is a packed encoding with no redundancy to test, and one+-- that is not a real key simply verifies nothing.+verificationKey+ :: forall p ba+ . (MLDSA p, ByteArrayAccess ba)+ => ba -> CryptoFailable (VerificationKey p)+verificationKey bs+ | B.length bs == verificationKeySize (Proxy :: Proxy p) =+ CryptoPassed $ VerificationKey $ B.copyAndFreeze bs (\_ -> return ())+ | otherwise = CryptoFailed CryptoError_PublicKeySizeInvalid++-- | Try to read a signing key.+--+-- Beyond the length this runs the validity checks of the implementation:+-- the secret polynomials must have coefficients in range, and the+-- commitment and the public-key hash the key carries must match what is+-- recomputed from the rest of it. A key that fails has been damaged or was+-- never a key, and signing with it would produce signatures nothing+-- verifies.+signingKey+ :: forall p ba+ . (MLDSA p, ByteArrayAccess ba)+ => ba -> CryptoFailable (SigningKey p)+signingKey bs+ | B.length bs /= signingKeySize p = CryptoFailed CryptoError_SecretKeySizeInvalid+ | otherwise = unsafeDoIO $ do+ (r, _ :: Bytes) <- B.allocRet (verificationKeySize p) $ \ppk ->+ withByteArray bs $ \psk -> c_pkFromSk p ppk psk+ return $+ if r == 0+ then CryptoPassed $ SigningKey $ B.copyAndFreeze bs (\_ -> return ())+ else CryptoFailed CryptoError_SecretKeyStructureInvalid+ where+ p = Proxy :: Proxy p+{-# NOINLINE signingKey #-}++-- | Try to read a signature. Only the length is checked; whether it is a+-- signature of anything is what 'verify' answers.+signature+ :: forall p ba+ . (MLDSA p, ByteArrayAccess ba)+ => ba -> CryptoFailable (Signature p)+signature bs+ | B.length bs == signatureSize (Proxy :: Proxy p) =+ CryptoPassed $ Signature $ B.copyAndFreeze bs (\_ -> return ())+ | otherwise = CryptoFailed CryptoError_ParameterInvalid++-- | Recover the verification key a signing key was made with.+toPublic :: forall p. MLDSA p => SigningKey p -> VerificationKey p+toPublic sk = VerificationKey $ unsafeDoIO $ do+ (_ :: CInt, pk) <- B.allocRet (verificationKeySize p) $ \ppk ->+ withByteArray sk $ \psk -> c_pkFromSk p ppk psk+ return pk+ where+ p = Proxy :: Proxy p+{-# NOINLINE toPublic #-}++-- | Generate a key pair.+--+-- The seed it is derived from is drawn here and thrown away. Use+-- 'generateKeyPairAndSeed' where it has to be kept.+generateKeyPair+ :: forall p proxy m+ . (MLDSA p, MonadRandom m)+ => proxy p -> m (VerificationKey p, SigningKey p)+generateKeyPair p = do+ (vk, sk, _) <- generateKeyPairAndSeed p+ return (vk, sk)++-- | Generate a key pair and hand back the seed it was derived from, @xi@+-- in FIPS 204.+--+-- A 'SigningKey' is the expanded key and nothing else, so the seed cannot+-- be recovered from a pair afterwards. An application that has to write+-- the key out in a form that keeps the seed -- RFC 9881 lets an ML-DSA+-- private key be the seed, the expanded key, or both -- has to generate it+-- here:+--+-- > (vk, sk, seed) <- generateKeyPairAndSeed MLDSA65+--+-- The seed is as secret as the signing key: 'keyPairFromSeed' turns it+-- back into the same pair.+generateKeyPairAndSeed+ :: forall p proxy m+ . (MLDSA p, MonadRandom m)+ => proxy p -> m (VerificationKey p, SigningKey p, ScrubbedBytes)+generateKeyPairAndSeed p = do+ seed <- getRandomBytes seedSize :: m ScrubbedBytes+ case keyPairFromSeed p seed of+ CryptoPassed (vk, sk) -> return (vk, sk, seed)+ CryptoFailed e ->+ error ("Crypto.PubKey.MLDSA.generateKeyPairAndSeed: " ++ show e)++-- | Derive a key pair from a seed, @xi@ in FIPS 204, which must be+-- 'seedSize' bytes.+keyPairFromSeed+ :: forall p proxy ba+ . (MLDSA p, ByteArrayAccess ba)+ => proxy p -> ba -> CryptoFailable (VerificationKey p, SigningKey p)+keyPairFromSeed p seed+ | B.length seed /= seedSize = CryptoFailed CryptoError_SeedSizeInvalid+ | otherwise = unsafeDoIO $ do+ -- Not zeroed, and does not need to be: the C writes the whole+ -- buffer, and on a non-zero return the result is discarded without+ -- being read. Anything that is *read* before being written has to+ -- use B.zero instead -- see signInternal in Crypto.PubKey.MLDSA.+ sk <- B.alloc (signingKeySize p) (\_ -> return ()) :: IO ScrubbedBytes+ (r, vk) <- B.allocRet (verificationKeySize p) $ \pvk ->+ withByteArray sk $ \psk ->+ withByteArray seed $ \pseed ->+ c_keypair p pvk psk pseed+ return $+ if r == 0+ then CryptoPassed (VerificationKey vk, SigningKey sk)+ else CryptoFailed CryptoError_ParameterInvalid+{-# NOINLINE keyPairFromSeed #-}++-- | Sign a message.+--+-- This is the hedged signing FIPS 204 recommends: fresh randomness goes in+-- alongside the key and the message, so two signatures of one message+-- differ and a fault in one reveals less. Verification does not care which+-- of the three entry points made the signature.+sign+ :: forall p m msg+ . (MLDSA p, MonadRandom m, ByteArrayAccess msg)+ => SigningKey p -> Context -> msg -> m (Signature p)+sign sk ctx msg = do+ rnd <- getRandomBytes signingRandomnessSize :: m ScrubbedBytes+ case signWith sk ctx msg rnd of+ CryptoPassed s -> return s+ CryptoFailed e -> error ("Crypto.PubKey.MLDSA.sign: " ++ show e)++-- | Sign with the randomness supplied, which must be+-- 'signingRandomnessSize' bytes.+--+-- For test vectors, and for callers who draw their own randomness. Ordinary+-- use wants 'sign'.+signWith+ :: forall p msg rnd+ . (MLDSA p, ByteArrayAccess msg, ByteArrayAccess rnd)+ => SigningKey p -> Context -> msg -> rnd -> CryptoFailable (Signature p)+signWith sk ctx msg rnd+ | B.length rnd /= signingRandomnessSize = CryptoFailed CryptoError_SeedSizeInvalid+ | otherwise = signInternal sk ctx msg (Just rnd)++-- | Sign deterministically, as FIPS 204 section 3.4 allows: the randomness+-- is replaced by zeroes, so one key and one message always give one+-- signature.+--+-- This is what test vectors are written against, and what to use where the+-- signature must be reproducible. It gives up what hedging buys, so where+-- there is a usable random source 'sign' is the better default.+signDeterministic+ :: forall p msg+ . (MLDSA p, ByteArrayAccess msg)+ => SigningKey p -> Context -> msg -> Signature p+signDeterministic sk ctx msg =+ case signInternal sk ctx msg (Nothing :: Maybe Bytes) of+ CryptoPassed s -> s+ CryptoFailed e -> error ("Crypto.PubKey.MLDSA.signDeterministic: " ++ show e)++signInternal+ :: forall p msg rnd+ . (MLDSA p, ByteArrayAccess msg, ByteArrayAccess rnd)+ => SigningKey p -> Context -> msg -> Maybe rnd -> CryptoFailable (Signature p)+signInternal sk ctx msg mrnd = unsafeDoIO $ do+ -- B.zero, not B.alloc with an empty action: alloc hands back whatever+ -- was in the memory. That made signDeterministic sign with the last+ -- caller's bytes and produce a different signature every time, which the+ -- ACVP vectors caught only once the whole suite ran and the allocator+ -- stopped handing out fresh zeroed pages.+ let zeroes = B.zero signingRandomnessSize :: ScrubbedBytes+ withRnd f = case mrnd of+ Just r -> withByteArray r f+ Nothing -> withByteArray zeroes f+ (r, sig) <- B.allocRet (signatureSize p) $ \psig ->+ withByteArray msg $ \pmsg ->+ withByteArray pre $ \ppre ->+ withRnd $ \prnd ->+ withByteArray sk $ \psk ->+ c_sign+ p+ psig+ pmsg+ (fromIntegral (B.length msg))+ ppre+ (fromIntegral (B.length pre))+ prnd+ psk+ 0+ return $+ if r == 0+ then CryptoPassed (Signature sig)+ else CryptoFailed CryptoError_ParameterInvalid+ where+ p = Proxy :: Proxy p+ pre = domainPrefix ctx+{-# NOINLINE signInternal #-}++-- | Sign a message representative, drawing the randomness.+--+-- The context string is already inside the representative, which is why+-- this does not take one.+signExternalMu+ :: forall p m+ . (MLDSA p, MonadRandom m)+ => SigningKey p -> Mu -> m (Signature p)+signExternalMu sk m = do+ rnd <- getRandomBytes signingRandomnessSize :: m ScrubbedBytes+ case signExternalMuWith sk m rnd of+ CryptoPassed s -> return s+ CryptoFailed e -> error ("Crypto.PubKey.MLDSA.signExternalMu: " ++ show e)++-- | Sign a message representative with the randomness supplied.+signExternalMuWith+ :: (MLDSA p, ByteArrayAccess rnd)+ => SigningKey p -> Mu -> rnd -> CryptoFailable (Signature p)+signExternalMuWith sk m rnd+ | B.length rnd /= signingRandomnessSize = CryptoFailed CryptoError_SeedSizeInvalid+ | otherwise = signMu sk m (Just rnd)++-- | Sign a message representative deterministically.+signExternalMuDeterministic+ :: MLDSA p => SigningKey p -> Mu -> Signature p+signExternalMuDeterministic sk m =+ case signMu sk m (Nothing :: Maybe Bytes) of+ CryptoPassed s -> s+ CryptoFailed e ->+ error ("Crypto.PubKey.MLDSA.signExternalMuDeterministic: " ++ show e)++-- | Verify a signature of a message representative.+verifyExternalMu+ :: forall p. MLDSA p => VerificationKey p -> Mu -> Signature p -> Bool+verifyExternalMu vk m sig+ | B.length sig /= signatureSize p = False+ | otherwise = unsafeDoIO $+ withByteArray sig $ \psig ->+ withByteArray m $ \pmu ->+ withByteArray vk $ \pvk -> do+ r <-+ c_verify+ p+ psig+ pmu+ (fromIntegral muSize)+ nullPtr+ 0+ pvk+ 1+ return (r == 0)+ where+ p = Proxy :: Proxy p+{-# NOINLINE verifyExternalMu #-}++-- The external-mu entry points are the ordinary ones with the last argument+-- set: the representative goes in where the message would, there is no+-- domain separation prefix to prepend because it is already inside, and the+-- implementation is told so.+signMu+ :: forall p rnd+ . (MLDSA p, ByteArrayAccess rnd)+ => SigningKey p -> Mu -> Maybe rnd -> CryptoFailable (Signature p)+signMu sk m mrnd = unsafeDoIO $ do+ let zeroes = B.zero signingRandomnessSize :: ScrubbedBytes+ withRnd f = case mrnd of+ Just r -> withByteArray r f+ Nothing -> withByteArray zeroes f+ (r, sig) <- B.allocRet (signatureSize p) $ \psig ->+ withByteArray m $ \pmu ->+ withRnd $ \prnd ->+ withByteArray sk $ \psk ->+ c_sign p psig pmu (fromIntegral muSize) nullPtr 0 prnd psk 1+ return $+ if r == 0+ then CryptoPassed (Signature sig)+ else CryptoFailed CryptoError_ParameterInvalid+ where+ p = Proxy :: Proxy p+{-# NOINLINE signMu #-}++-- | Verify a signature.+--+-- The context must be the one it was signed under; anything else is a+-- rejection, which is what the context is for.+verify+ :: forall p msg+ . (MLDSA p, ByteArrayAccess msg)+ => VerificationKey p -> Context -> msg -> Signature p -> Bool+verify vk ctx msg sig+ | B.length sig /= signatureSize p = False+ | otherwise = unsafeDoIO $+ withByteArray sig $ \psig ->+ withByteArray msg $ \pmsg ->+ withByteArray pre $ \ppre ->+ withByteArray vk $ \pvk -> do+ r <-+ c_verify+ p+ psig+ pmsg+ (fromIntegral (B.length msg))+ ppre+ (fromIntegral (B.length pre))+ pvk+ 0+ return (r == 0)+ where+ p = Proxy :: Proxy p+ pre = domainPrefix ctx+{-# NOINLINE verify #-}++-- | The domain separation prefix of FIPS 204 for pure ML-DSA, which is a+-- zero byte, the context's length and the context itself. It is built here+-- rather than taken from the implementation because it is three bytes of+-- concatenation and doing it here keeps one fewer foreign call.+domainPrefix :: Context -> Bytes+domainPrefix (Context ctx) =+ B.concat [B.pack [0, fromIntegral (B.length ctx)] :: Bytes, B.convert ctx]++instance MLDSA MLDSA44 where+ verificationKeySize _ = 1312+ signingKeySize _ = 2560+ signatureSize _ = 2420+ c_keypair _ = c_mldsa44_keypair+ c_sign _ = c_mldsa44_sign+ c_verify _ = c_mldsa44_verify+ c_pkFromSk _ = c_mldsa44_pk_from_sk++instance MLDSA MLDSA65 where+ verificationKeySize _ = 1952+ signingKeySize _ = 4032+ signatureSize _ = 3309+ c_keypair _ = c_mldsa65_keypair+ c_sign _ = c_mldsa65_sign+ c_verify _ = c_mldsa65_verify+ c_pkFromSk _ = c_mldsa65_pk_from_sk++instance MLDSA MLDSA87 where+ verificationKeySize _ = 2592+ signingKeySize _ = 4896+ signatureSize _ = 4627+ c_keypair _ = c_mldsa87_keypair+ c_sign _ = c_mldsa87_sign+ c_verify _ = c_mldsa87_verify+ c_pkFromSk _ = c_mldsa87_pk_from_sk++foreign import ccall unsafe "crypton_mldsa44_keypair_internal"+ c_mldsa44_keypair :: Ptr Word8 -> Ptr Word8 -> Ptr Word8 -> IO CInt+foreign import ccall unsafe "crypton_mldsa44_signature_internal"+ c_mldsa44_sign+ :: Ptr Word8 -> Ptr Word8 -> CSize -> Ptr Word8 -> CSize+ -> Ptr Word8 -> Ptr Word8 -> CInt -> IO CInt+foreign import ccall unsafe "crypton_mldsa44_verify_internal"+ c_mldsa44_verify+ :: Ptr Word8 -> Ptr Word8 -> CSize -> Ptr Word8 -> CSize+ -> Ptr Word8 -> CInt -> IO CInt+foreign import ccall unsafe "crypton_mldsa44_pk_from_sk"+ c_mldsa44_pk_from_sk :: Ptr Word8 -> Ptr Word8 -> IO CInt++foreign import ccall unsafe "crypton_mldsa65_keypair_internal"+ c_mldsa65_keypair :: Ptr Word8 -> Ptr Word8 -> Ptr Word8 -> IO CInt+foreign import ccall unsafe "crypton_mldsa65_signature_internal"+ c_mldsa65_sign+ :: Ptr Word8 -> Ptr Word8 -> CSize -> Ptr Word8 -> CSize+ -> Ptr Word8 -> Ptr Word8 -> CInt -> IO CInt+foreign import ccall unsafe "crypton_mldsa65_verify_internal"+ c_mldsa65_verify+ :: Ptr Word8 -> Ptr Word8 -> CSize -> Ptr Word8 -> CSize+ -> Ptr Word8 -> CInt -> IO CInt+foreign import ccall unsafe "crypton_mldsa65_pk_from_sk"+ c_mldsa65_pk_from_sk :: Ptr Word8 -> Ptr Word8 -> IO CInt++foreign import ccall unsafe "crypton_mldsa87_keypair_internal"+ c_mldsa87_keypair :: Ptr Word8 -> Ptr Word8 -> Ptr Word8 -> IO CInt+foreign import ccall unsafe "crypton_mldsa87_signature_internal"+ c_mldsa87_sign+ :: Ptr Word8 -> Ptr Word8 -> CSize -> Ptr Word8 -> CSize+ -> Ptr Word8 -> Ptr Word8 -> CInt -> IO CInt+foreign import ccall unsafe "crypton_mldsa87_verify_internal"+ c_mldsa87_verify+ :: Ptr Word8 -> Ptr Word8 -> CSize -> Ptr Word8 -> CSize+ -> Ptr Word8 -> CInt -> IO CInt+foreign import ccall unsafe "crypton_mldsa87_pk_from_sk"+ c_mldsa87_pk_from_sk :: Ptr Word8 -> Ptr Word8 -> IO CInt
@@ -0,0 +1,448 @@+-- |+-- Module : Crypto.PubKey.MLKEM+-- License : BSD-style+-- Maintainer : Kazu Yamamoto <kazu@iij.ad.jp>+-- Stability : experimental+-- Portability : unknown+--+-- ML-KEM, the Module-Lattice-Based Key-Encapsulation Mechanism of+-- <https://csrc.nist.gov/pubs/fips/203/final FIPS 203>, in all three+-- parameter sets.+--+-- A key encapsulation mechanism is not a Diffie-Hellman: there is no shared+-- secret to be computed from two key pairs. One side publishes an+-- 'EncapsulationKey'; the other calls 'encapsulate' on it, which draws a+-- fresh secret and returns it along with a 'Ciphertext' that only the holder+-- of the matching 'DecapsulationKey' can turn back into that secret.+--+-- > (ek, dk) <- generateKeyPair MLKEM768 -- the receiver+-- > (ct, ss) <- encapsulate ek -- the sender+-- > let ss' = decapsulate dk ct -- the receiver, again+-- > ss == ss'+--+-- What 'generateKeyPair' and 'encapsulate' draw their randomness from is+-- the 'Crypto.Random.MonadRandom' instance in use. Its documentation says+-- what an instance of your own has to be.+--+-- The parameter set is a type, so an ML-KEM-768 key cannot be passed where+-- an ML-KEM-1024 one is expected. The three are fixed by FIPS 203 and the+-- class has no other instances.+{-# LANGUAGE GeneralizedNewtypeDeriving #-}+{-# LANGUAGE TypeFamilies #-}+{-# LANGUAGE TypeOperators #-}+{-# LANGUAGE ScopedTypeVariables #-}++module Crypto.PubKey.MLKEM (+ -- * Parameter sets+ MLKEM512 (..),+ MLKEM768 (..),+ MLKEM1024 (..),+ MLKEM (encapsulationKeySize, decapsulationKeySize, ciphertextSize),++ -- * Keys, ciphertexts and shared secrets+ --+ -- | These are the associated types of 'KEM', re-exported so that a+ -- caller of this module alone has them.+ KEM (..),+ SharedSecret (..),++ -- * Smart constructors+ encapsulationKey,+ decapsulationKey,+ ciphertext,++ -- * What ML-KEM has beyond the class+ generateKeyPairAndSeed,+ keyPairFromSeed,++ -- * Sizes+ seedSize,+ encapsulationCoinsSize,+ sharedSecretSize,+) where++import Data.Proxy (Proxy (..))+import Foreign.C.Types (CInt (..))+import Foreign.Ptr (Ptr)++import Crypto.Debug (DebugShow (..), debugShowBytes)+import Crypto.Error+import Crypto.KEM+import Crypto.Internal.ByteArray (+ ByteArrayAccess,+ Bytes,+ ScrubbedBytes,+ withByteArray,+ )+import qualified Crypto.Internal.ByteArray as B+import Crypto.Internal.Compat (unsafeDoIO)+import Crypto.Internal.Imports+import Crypto.Random (MonadRandom, getRandomBytes)++-- | ML-KEM-512.+data MLKEM512 = MLKEM512 deriving (Show, Eq)++-- | ML-KEM-768. This is the set TLS uses, on its own and as the+-- lattice half of the hybrid groups.+data MLKEM768 = MLKEM768 deriving (Show, Eq)++-- | ML-KEM-1024.+data MLKEM1024 = MLKEM1024 deriving (Show, Eq)++-- | The three parameter sets of FIPS 203.+--+-- This is not an abstract KEM interface and does not try to be: it is the+-- three sets FIPS 203 defines, closed, carrying their sizes and the calls+-- into the implementation. Only the sizes are exported. If crypton grows+-- a second KEM and an interface common to both is wanted, that belongs in+-- a module of its own, with this as one of its instances.+class+ ( KEM p+ , EncapsulationKey p ~ MLKEMEncapsulationKey p+ , DecapsulationKey p ~ MLKEMDecapsulationKey p+ , Ciphertext p ~ MLKEMCiphertext p+ , Coins p ~ ScrubbedBytes+ ) =>+ MLKEM p+ where+ -- | Size in bytes of an 'EncapsulationKey' of this parameter set.+ encapsulationKeySize :: proxy p -> Int++ -- | Size in bytes of a 'DecapsulationKey' of this parameter set.+ decapsulationKeySize :: proxy p -> Int++ -- | Size in bytes of a 'Ciphertext' of this parameter set.+ ciphertextSize :: proxy p -> Int++ c_keypair :: proxy p -> Ptr Word8 -> Ptr Word8 -> Ptr Word8 -> IO CInt+ c_enc :: proxy p -> Ptr Word8 -> Ptr Word8 -> Ptr Word8 -> Ptr Word8 -> IO CInt+ c_dec :: proxy p -> Ptr Word8 -> Ptr Word8 -> Ptr Word8 -> IO CInt+ c_checkPk :: proxy p -> Ptr Word8 -> IO CInt+ c_checkSk :: proxy p -> Ptr Word8 -> IO CInt++-- | A public encapsulation key, @ek@ in FIPS 203.+newtype MLKEMEncapsulationKey p = MLKEMEncapsulationKey Bytes+ deriving (Show, Eq, ByteArrayAccess, NFData)++-- | A private decapsulation key, @dk@ in FIPS 203. It embeds the matching+-- encapsulation key, which is why it is the larger of the two.+newtype MLKEMDecapsulationKey p = MLKEMDecapsulationKey ScrubbedBytes+ deriving (Eq, ByteArrayAccess, NFData)++instance Show (MLKEMDecapsulationKey p) where+ show _ = "DecapsulationKey <redacted>"++instance DebugShow (MLKEMDecapsulationKey p) where+ debugShow = debugShowBytes "DecapsulationKey"++-- | The value 'encapsulate' produces and 'decapsulate' consumes.+newtype MLKEMCiphertext p = MLKEMCiphertext Bytes+ deriving (Show, Eq, ByteArrayAccess, NFData)++-- | Size in bytes of the seed 'keyPairFromSeed' takes, which is @d@ and @z@+-- of FIPS 203 one after the other.+seedSize :: Int+seedSize = 64++-- | Size in bytes of the randomness 'encapsulateWith' takes, @m@ in+-- FIPS 203.+encapsulationCoinsSize :: Int+encapsulationCoinsSize = 32++-- | Size in bytes of a 'SharedSecret'.+sharedSecretSize :: Int+sharedSecretSize = 32++-- | Try to read an encapsulation key.+--+-- Beyond the length this runs the check of FIPS 203 section 7.2: the key+-- must be the encoding of coefficients that are all in range, which is to+-- say it must survive a decode and re-encode unchanged. A key that fails+-- it is not one any honest party produced.+encapsulationKey+ :: forall p ba+ . (MLKEM p, ByteArrayAccess ba)+ => ba -> CryptoFailable (EncapsulationKey p)+encapsulationKey bs+ | B.length bs /= encapsulationKeySize p = CryptoFailed CryptoError_PublicKeySizeInvalid+ | otherwise = unsafeDoIO $ withByteArray bs $ \inp -> do+ r <- c_checkPk p inp+ return $+ if r == 0+ then CryptoPassed $ MLKEMEncapsulationKey $ B.copyAndFreeze bs (\_ -> return ())+ else CryptoFailed CryptoError_PublicKeyStructureInvalid+ where+ p = Proxy :: Proxy p+{-# NOINLINE encapsulationKey #-}++-- | Try to read a decapsulation key.+--+-- Beyond the length this runs the check of FIPS 203 section 7.3: the hash+-- of the encapsulation key the private key embeds must match the copy of+-- that hash it also embeds. The two disagreeing means the key was not+-- produced as a pair, and decapsulating with it would silently answer with+-- the implicit rejection every time.+decapsulationKey+ :: forall p ba+ . (MLKEM p, ByteArrayAccess ba)+ => ba -> CryptoFailable (DecapsulationKey p)+decapsulationKey bs+ | B.length bs /= decapsulationKeySize p = CryptoFailed CryptoError_SecretKeySizeInvalid+ | otherwise = unsafeDoIO $ withByteArray bs $ \inp -> do+ r <- c_checkSk p inp+ return $+ if r == 0+ then CryptoPassed $ MLKEMDecapsulationKey $ B.copyAndFreeze bs (\_ -> return ())+ else CryptoFailed CryptoError_SecretKeyStructureInvalid+ where+ p = Proxy :: Proxy p+{-# NOINLINE decapsulationKey #-}++-- | Try to read a ciphertext. Only the length is checked; every string of+-- the right length is a ciphertext that 'decapsulate' will answer.+ciphertext+ :: forall p ba+ . (MLKEM p, ByteArrayAccess ba)+ => ba -> CryptoFailable (Ciphertext p)+ciphertext bs+ | B.length bs == ciphertextSize (Proxy :: Proxy p) =+ CryptoPassed $ MLKEMCiphertext $ B.copyAndFreeze bs (\_ -> return ())+ | otherwise = CryptoFailed CryptoError_PointSizeInvalid++-- | Generate a key pair.+--+-- The seed it is derived from is drawn here and thrown away. Use+-- 'generateKeyPairAndSeed' where it has to be kept.+mlkemGenerateKeyPair+ :: forall p proxy m+ . (MLKEM p, MonadRandom m)+ => proxy p -> m (MLKEMEncapsulationKey p, DecapsulationKey p)+mlkemGenerateKeyPair p = do+ (ek, dk, _) <- generateKeyPairAndSeed p+ return (ek, dk)++-- | Generate a key pair and hand back the seed it was derived from, @d@+-- and @z@ of FIPS 203 one after the other.+--+-- A 'DecapsulationKey' is the expanded key and nothing else, so the seed+-- cannot be recovered from a pair afterwards. An application that has to+-- write the key out in a form that keeps the seed has to generate it here:+--+-- > (ek, dk, seed) <- generateKeyPairAndSeed MLKEM768+--+-- The seed is as secret as the decapsulation key: 'keyPairFromSeed' turns+-- it back into the same pair.+generateKeyPairAndSeed+ :: forall p proxy m+ . (MLKEM p, MonadRandom m)+ => proxy p+ -> m (MLKEMEncapsulationKey p, DecapsulationKey p, ScrubbedBytes)+generateKeyPairAndSeed p = do+ seed <- getRandomBytes seedSize :: m ScrubbedBytes+ case keyPairFromSeed p seed of+ CryptoPassed (ek, dk) -> return (ek, dk, seed)+ CryptoFailed e ->+ error ("Crypto.PubKey.MLKEM.generateKeyPairAndSeed: " ++ show e)++-- | Derive a key pair from a seed, which is @d@ and @z@ of FIPS 203 one+-- after the other and must be 'seedSize' bytes.+--+-- This is the entry point to use when the seed comes from somewhere+-- particular -- a test vector, or a store that keeps seeds rather than+-- expanded keys. For an ordinary key, 'generateKeyPair' draws the seed+-- itself.+keyPairFromSeed+ :: forall p proxy ba+ . (MLKEM p, ByteArrayAccess ba)+ => proxy p+ -> ba+ -> CryptoFailable (MLKEMEncapsulationKey p, DecapsulationKey p)+keyPairFromSeed p seed+ | B.length seed /= seedSize = CryptoFailed CryptoError_SeedSizeInvalid+ | otherwise = unsafeDoIO $ do+ -- Not zeroed, and does not need to be: the C writes the whole+ -- buffer, and on a non-zero return the result is discarded without+ -- being read. Anything that is *read* before being written has to+ -- use B.zero instead -- see signInternal in Crypto.PubKey.MLDSA.+ dk <- B.alloc (decapsulationKeySize p) (\_ -> return ()) :: IO ScrubbedBytes+ (r, ek) <- B.allocRet (encapsulationKeySize p) $ \pek ->+ withByteArray dk $ \pdk ->+ withByteArray seed $ \pseed ->+ c_keypair p pek pdk pseed+ return $+ if r == 0+ then CryptoPassed (MLKEMEncapsulationKey ek, MLKEMDecapsulationKey dk)+ else CryptoFailed CryptoError_ParameterInvalid+{-# NOINLINE keyPairFromSeed #-}++-- | Encapsulate against a public key, drawing the randomness.+mlkemEncapsulate+ :: forall p m+ . (MLKEM p, MonadRandom m)+ => MLKEMEncapsulationKey p+ -> m (CryptoFailable (Ciphertext p, SharedSecret))+mlkemEncapsulate ek = do+ coins <- getRandomBytes encapsulationCoinsSize :: m ScrubbedBytes+ return (mlkemEncapsulateWith ek coins)++-- The class's 'encapsulateWith' for ML-KEM, where the coins are @m@ of+-- FIPS 203 and must be 'encapsulationCoinsSize' bytes.+mlkemEncapsulateWith+ :: forall p+ . MLKEM p+ => MLKEMEncapsulationKey p+ -> ScrubbedBytes+ -> CryptoFailable (Ciphertext p, SharedSecret)+mlkemEncapsulateWith ek coins+ | B.length coins /= encapsulationCoinsSize = CryptoFailed CryptoError_SeedSizeInvalid+ | otherwise = unsafeDoIO $ do+ ss <- B.alloc sharedSecretSize (\_ -> return ()) :: IO ScrubbedBytes+ (r, ct) <- B.allocRet (ciphertextSize p) $ \pct ->+ withByteArray ss $ \pss ->+ withByteArray ek $ \pek ->+ withByteArray coins $ \pcoins ->+ c_enc p pct pss pek pcoins+ return $+ if r == 0+ then CryptoPassed (MLKEMCiphertext ct, SharedSecret ss)+ else CryptoFailed CryptoError_ParameterInvalid+ where+ p = Proxy :: Proxy p+{-# NOINLINE mlkemEncapsulateWith #-}++-- | Recover the shared secret from a ciphertext.+--+-- A ciphertext that was not produced by encapsulating against the matching+-- key is not an error. ML-KEM rejects implicitly: it yields a secret+-- derived from the private key and the ciphertext, and the caller cannot+-- tell that case from the other one, which is the point -- telling them+-- apart is what a chosen-ciphertext attack needs. A ciphertext that does+-- not belong here shows up later, as the two sides failing to agree on+-- anything.+--+-- The checks FIPS 203 does require are at the point where bytes become a+-- value of these types, which is where they can be reported:+--+-- * The ciphertext type check of section 7.3 is its length, and+-- 'ciphertext' is the only way to build a 'Ciphertext' from bytes. There+-- is nothing else to check: a ciphertext's coefficients are compressed to+-- fewer than twelve bits, so every bit pattern decodes to a value in+-- range.+-- * The hash check of section 7.3 is on the decapsulation key, and+-- 'decapsulationKey' runs it; a key from 'generateKeyPair' or+-- 'keyPairFromSeed' satisfies it by construction.+--+-- So the result is 'CryptoPassed' for every key and ciphertext this module+-- can produce. It is 'CryptoFailable' rather than a bare 'SharedSecret'+-- because the implementation checks the key again on its way through, and+-- what it finds is better reported than turned into an exception.+mlkemDecapsulate+ :: forall p+ . MLKEM p+ => DecapsulationKey p -> Ciphertext p -> CryptoFailable SharedSecret+mlkemDecapsulate dk ct = unsafeDoIO $ do+ (r, ss) <- B.allocRet sharedSecretSize $ \pss ->+ withByteArray ct $ \pct ->+ withByteArray dk $ \pdk ->+ c_dec (Proxy :: Proxy p) pss pct pdk+ return $+ if r == (0 :: CInt)+ then CryptoPassed (SharedSecret ss)+ else CryptoFailed CryptoError_SecretKeyStructureInvalid+{-# NOINLINE mlkemDecapsulate #-}++-- The class's view of the three sets. The operations are the ones above;+-- only the shape of the arguments differs, because the class takes the+-- mechanism as a proxy.+instance KEM MLKEM512 where+ type EncapsulationKey MLKEM512 = MLKEMEncapsulationKey MLKEM512+ type DecapsulationKey MLKEM512 = MLKEMDecapsulationKey MLKEM512+ type Ciphertext MLKEM512 = MLKEMCiphertext MLKEM512+ type Coins MLKEM512 = ScrubbedBytes+ generateKeyPair = mlkemGenerateKeyPair+ encapsulate _ = mlkemEncapsulate+ encapsulateWith _ = mlkemEncapsulateWith+ decapsulate _ = mlkemDecapsulate++instance KEM MLKEM768 where+ type EncapsulationKey MLKEM768 = MLKEMEncapsulationKey MLKEM768+ type DecapsulationKey MLKEM768 = MLKEMDecapsulationKey MLKEM768+ type Ciphertext MLKEM768 = MLKEMCiphertext MLKEM768+ type Coins MLKEM768 = ScrubbedBytes+ generateKeyPair = mlkemGenerateKeyPair+ encapsulate _ = mlkemEncapsulate+ encapsulateWith _ = mlkemEncapsulateWith+ decapsulate _ = mlkemDecapsulate++instance KEM MLKEM1024 where+ type EncapsulationKey MLKEM1024 = MLKEMEncapsulationKey MLKEM1024+ type DecapsulationKey MLKEM1024 = MLKEMDecapsulationKey MLKEM1024+ type Ciphertext MLKEM1024 = MLKEMCiphertext MLKEM1024+ type Coins MLKEM1024 = ScrubbedBytes+ generateKeyPair = mlkemGenerateKeyPair+ encapsulate _ = mlkemEncapsulate+ encapsulateWith _ = mlkemEncapsulateWith+ decapsulate _ = mlkemDecapsulate++instance MLKEM MLKEM512 where+ encapsulationKeySize _ = 800+ decapsulationKeySize _ = 1632+ ciphertextSize _ = 768+ c_keypair _ = c_mlkem512_keypair+ c_enc _ = c_mlkem512_enc+ c_dec _ = c_mlkem512_dec+ c_checkPk _ = c_mlkem512_check_pk+ c_checkSk _ = c_mlkem512_check_sk++instance MLKEM MLKEM768 where+ encapsulationKeySize _ = 1184+ decapsulationKeySize _ = 2400+ ciphertextSize _ = 1088+ c_keypair _ = c_mlkem768_keypair+ c_enc _ = c_mlkem768_enc+ c_dec _ = c_mlkem768_dec+ c_checkPk _ = c_mlkem768_check_pk+ c_checkSk _ = c_mlkem768_check_sk++instance MLKEM MLKEM1024 where+ encapsulationKeySize _ = 1568+ decapsulationKeySize _ = 3168+ ciphertextSize _ = 1568+ c_keypair _ = c_mlkem1024_keypair+ c_enc _ = c_mlkem1024_enc+ c_dec _ = c_mlkem1024_dec+ c_checkPk _ = c_mlkem1024_check_pk+ c_checkSk _ = c_mlkem1024_check_sk++foreign import ccall unsafe "crypton_mlkem512_keypair_derand"+ c_mlkem512_keypair :: Ptr Word8 -> Ptr Word8 -> Ptr Word8 -> IO CInt+foreign import ccall unsafe "crypton_mlkem512_enc_derand"+ c_mlkem512_enc :: Ptr Word8 -> Ptr Word8 -> Ptr Word8 -> Ptr Word8 -> IO CInt+foreign import ccall unsafe "crypton_mlkem512_dec"+ c_mlkem512_dec :: Ptr Word8 -> Ptr Word8 -> Ptr Word8 -> IO CInt+foreign import ccall unsafe "crypton_mlkem512_check_pk"+ c_mlkem512_check_pk :: Ptr Word8 -> IO CInt+foreign import ccall unsafe "crypton_mlkem512_check_sk"+ c_mlkem512_check_sk :: Ptr Word8 -> IO CInt++foreign import ccall unsafe "crypton_mlkem768_keypair_derand"+ c_mlkem768_keypair :: Ptr Word8 -> Ptr Word8 -> Ptr Word8 -> IO CInt+foreign import ccall unsafe "crypton_mlkem768_enc_derand"+ c_mlkem768_enc :: Ptr Word8 -> Ptr Word8 -> Ptr Word8 -> Ptr Word8 -> IO CInt+foreign import ccall unsafe "crypton_mlkem768_dec"+ c_mlkem768_dec :: Ptr Word8 -> Ptr Word8 -> Ptr Word8 -> IO CInt+foreign import ccall unsafe "crypton_mlkem768_check_pk"+ c_mlkem768_check_pk :: Ptr Word8 -> IO CInt+foreign import ccall unsafe "crypton_mlkem768_check_sk"+ c_mlkem768_check_sk :: Ptr Word8 -> IO CInt++foreign import ccall unsafe "crypton_mlkem1024_keypair_derand"+ c_mlkem1024_keypair :: Ptr Word8 -> Ptr Word8 -> Ptr Word8 -> IO CInt+foreign import ccall unsafe "crypton_mlkem1024_enc_derand"+ c_mlkem1024_enc :: Ptr Word8 -> Ptr Word8 -> Ptr Word8 -> Ptr Word8 -> IO CInt+foreign import ccall unsafe "crypton_mlkem1024_dec"+ c_mlkem1024_dec :: Ptr Word8 -> Ptr Word8 -> Ptr Word8 -> IO CInt+foreign import ccall unsafe "crypton_mlkem1024_check_pk"+ c_mlkem1024_check_pk :: Ptr Word8 -> IO CInt+foreign import ccall unsafe "crypton_mlkem1024_check_sk"+ c_mlkem1024_check_sk :: Ptr Word8 -> IO CInt
@@ -1,3 +1,5 @@+{-# LANGUAGE ScopedTypeVariables #-}+ -- | -- Module : Crypto.PubKey.RSA -- License : BSD-style@@ -16,9 +18,16 @@ generateBlinder, ) where +import Crypto.Internal.ByteArray (ScrubbedBytes) import Crypto.Number.Generate (generateMax)-import Crypto.Number.ModArithmetic (inverse, inverseCoprimes)+import Crypto.Number.ModArithmetic (+ expSafe,+ inverse,+ inverseCoprimes,+ inverseSafe,+ ) import Crypto.Number.Prime (generatePrime)+import Crypto.Number.Serialize (os2ip) import Crypto.PubKey.RSA.Types import Crypto.Random.Types @@ -51,23 +60,64 @@ -- * e=0x10001 is a popular choice -- -- * e=3 is popular as well, but proven to not be as secure for some cases.+--+-- /WARNING:/ Making a key is not constant time, and cannot be: the search for+-- the two primes takes as long as it takes, and 'Crypto.Number.Prime' is not+-- constant time either. What that leaks is about the search rather than+-- about the primes it settles on. Of the arithmetic that does touch them,+-- the inverse of one prime modulo the other is worked out without a side+-- channel, and so is the private exponent, which is the inverse of @e@ modulo+-- @(p-1)*(q-1)@: @e@ being public lets that be worked out as a remainder, an+-- inverse modulo @e@ itself, and an exact division, none of which follows the+-- number being inverted. An @e@ that is not prime keeps the extended+-- Euclidean algorithm, which for a public @e@ is one division by a small+-- number and then a few steps on numbers under it. generateWith :: (Integer, Integer) -- ^ chosen distinct primes p and q -> Int -- ^ size in bytes -> Integer- -- ^ RSA public exponent 'e'+ -- ^ RSA public exponent @e@ -> Maybe (PublicKey, PrivateKey) generateWith (p, q) size e =- case inverse e phi of+ case privateExponent of Nothing -> Nothing Just d -> Just (pub, priv d) where n = p * q phi = (p - 1) * (q - 1)- -- q and p should be *distinct* *prime* numbers, hence always coprime- qinv = inverseCoprimes q p+ -- The private exponent is the inverse of e modulo phi, and phi is the+ -- key. The extended Euclidean algorithm would take a number of steps+ -- that follows it; e being public lets the work be about e instead.+ --+ -- Whatever d is, e * d = 1 + k * phi for some k under e, and reading that+ -- modulo e gives k = -phi^-1 mod e -- an inverse modulo a number of a+ -- handful of bits, which for a prime e is Fermat. Then d is an exact+ -- division by e. Nothing in that follows phi: the remainder and the+ -- division are one pass each over its limbs, and the rest is arithmetic+ -- the size of e.+ --+ -- Fermat wants a prime e, and rather than ask whether e is one -- which+ -- costs more than everything else here -- the k it gives is checked,+ -- which is arithmetic the size of e. A composite e that fails the check+ -- keeps the algorithm it had.+ privateExponent+ | e <= 1 = Nothing+ | t == 0 = Nothing -- e divides phi, so there is no inverse+ | (k * t) `mod` e == e - 1 = Just ((1 + k * phi) `div` e)+ | otherwise = inverse e phi+ where+ t = phi `mod` e+ k = (e - expSafe t (e - 2) e) `mod` e+ -- q and p should be *distinct* *prime* numbers, hence always coprime.+ -- Both of them are the key itself, so the inverse is worked out through+ -- Fermat's little theorem rather than the extended Euclidean algorithm,+ -- whose steps follow the numbers it is given. It falls back on the one+ -- that raises, which is what a p that is not prime deserves.+ qinv = case inverseSafe q p of+ Just i -> i+ Nothing -> inverseCoprimes q p pub = PublicKey { public_size = size@@ -91,7 +141,7 @@ => Int -- ^ size in bytes -> Integer- -- ^ RSA public exponent 'e'+ -- ^ RSA public exponent @e@ -> m (PublicKey, PrivateKey) generate size e = loop where@@ -113,10 +163,30 @@ -- -- the unique parameter apart from the random number generator is the -- public key value N.+--+-- The blinder holds a random number and its inverse. N is composite, so+-- Fermat has no answer for the inverse and it goes through the extended+-- Euclidean algorithm, whose steps follow the number handed to it -- which+-- would be the number the blinding rests on. So the algorithm is handed that+-- number multiplied by another random one instead, and its answer multiplied+-- by that number again, which leaves the inverse wanted and shows the+-- algorithm nothing that has anything to do with it. generateBlinder- :: MonadRandom m+ :: forall m+ . MonadRandom m => Integer -- ^ RSA public N parameter. -> m Blinder-generateBlinder n =- (\r -> Blinder r (inverseCoprimes r n)) <$> generateMax n+generateBlinder n = do+ r <- generateMax n+ -- The inverse goes through the extended Euclidean algorithm, whose steps+ -- follow the number handed to it, and r is what the blinding rests on.+ -- So another random number goes with it: the product is uniform and says+ -- nothing about r on its own, and multiplying its inverse by that number+ -- again leaves the inverse of r. Sixteen bytes are enough to hide it and+ -- are under either prime, so the product is coprime with n whenever r is,+ -- as it was before.+ u <- os2ip <$> (getRandomBytes 16 :: m ScrubbedBytes)+ let v = (r * u) `mod` n+ rm1 = (inverseCoprimes v n * u) `mod` n+ return $ Blinder r rm1
@@ -21,18 +21,21 @@ ) where import Crypto.Hash+import Crypto.Number.Serialize (os2ip) import Crypto.PubKey.Internal (and') import Crypto.PubKey.MaskGenFunction import Crypto.PubKey.RSA (generateBlinder) import Crypto.PubKey.RSA.Prim import Crypto.PubKey.RSA.Types import Crypto.Random.Types-import Data.Bits (xor)+import Data.Bits (complement, shiftR, xor, (.&.), (.|.)) import Data.ByteString (ByteString) import qualified Data.ByteString as B+import qualified Data.List as L+import Data.Word (Word32) import Crypto.Internal.ByteArray (ByteArray, ByteArrayAccess)-import qualified Crypto.Internal.ByteArray as B (convert)+import qualified Crypto.Internal.ByteArray as B (constEq, convert) -- | Parameters for OAEP encryption/decryption data OAEPParams hash seed output = OAEPParams@@ -109,6 +112,16 @@ -- | un-pad a OAEP encoded message. -- -- It doesn't apply the RSA decryption primitive+--+-- The data block is scanned in full rather than up to the 01 octet separating+-- the padding from the message, and the label hash and the leading octet are+-- compared without an early exit, so neither the length of the padding nor+-- where a comparison first differs shows up in how long this takes.+--+-- What remains visible is the result itself: whether the block was well formed,+-- and the length of the message when it was. That is the signal Manger's+-- attack needs, so a caller that decrypts attacker-supplied ciphertext must not+-- pass the distinction on. unpad :: HashAlgorithm hash => OAEPParams hash ByteString ByteString@@ -124,7 +137,9 @@ where -- parameters mgf = oaepMaskGenAlg oaep- labelHash = B.convert $ hashWith (oaepHash oaep) (maybe B.empty id $ oaepLabel oaep)+ labelHash =+ B.convert $ hashWith (oaepHash oaep) (maybe B.empty id $ oaepLabel oaep)+ :: ByteString hashLen = hashDigestSize (oaepHash oaep) -- getting em's fields (pb, em0) = B.splitAt 1 em@@ -135,22 +150,47 @@ db = B.pack $ B.zipWith xor maskedDB dbmask -- getting db's fields (labelHash', db1) = B.splitAt hashLen db- (_, db2) = B.break (/= 0) db1- (ps1, msg) = B.splitAt 1 db2 + -- index of the first nonzero octet in db1, or its length when every octet+ -- is zero; all of them are looked at either way+ oneIndex =+ fst $+ L.foldl'+ step+ (fromIntegral (B.length db1) :: Word32, 1 :: Word32)+ (zip [0 ..] (B.unpack db1))+ step (idx, unseen) (i, b) = (select found i idx, unseen .&. complement found)+ where+ w = fromIntegral b :: Word32+ -- 0 when b is zero, 1 otherwise+ nonZero = (w .|. negate w) `shiftR` 31+ -- all ones at the first nonzero octet only+ found = negate (unseen .&. nonZero)+ select mask a b = (a .&. mask) .|. (b .&. complement mask)++ ps1 = B.take 1 $ B.drop (fromIntegral oneIndex) db1+ msg = B.drop (fromIntegral oneIndex + 1) db1+ paddingSuccess = and'- [ labelHash' == labelHash -- no need for constant eq- , ps1 == B.replicate 1 0x1- , pb == B.replicate 1 0x0+ [ labelHash' `B.constEq` labelHash+ , ps1 `B.constEq` B.replicate 1 0x1+ , pb `B.constEq` B.replicate 1 0x0 ] -- | Decrypt a ciphertext using OAEP ----- When the signature is not in a context where an attacker could gain--- information from the timing of the operation, the blinder can be set to None.+-- The blinder is optional and 'Nothing' is accepted, but see t'Blinder' for+-- what it covers and when leaving it out is a decision rather than a default.+-- 'decryptSafer' generates one for you. ----- If unsure always set a blinder or use decryptSafer+-- Following RFC 8017, the ciphertext is rejected unless it is exactly as long+-- as the modulus (section 7.1.2, step 1) and its integer representative is+-- below the modulus (RSADP, section 5.1.2, step 1). The decryption primitive+-- normalises any multiple of the modulus away, so without the second check+-- @c + n@ would decrypt to the same message as @c@, and a ciphertext would not+-- be unique to its plaintext. Both checks are made on the ciphertext alone,+-- which is public, and report 'MessageSizeIncorrect'. decrypt :: HashAlgorithm hash => Maybe Blinder@@ -164,6 +204,7 @@ -> Either Error ByteString decrypt blinder oaep pk cipher | B.length cipher /= k = Left MessageSizeIncorrect+ | os2ip cipher >= private_n pk = Left MessageSizeIncorrect | k < 2 * hashLen + 2 = Left InvalidParameters | otherwise = unpad oaep (private_size pk) $ dp blinder pk cipher where
@@ -15,27 +15,36 @@ decryptSafer, sign, signSafer,+ signDigest,+ signSaferDigest,+ signDigestInfo,+ signSaferDigestInfo, -- * Public key operations encrypt, verify,+ verifyDigest,+ verifyDigestInfo, -- * Hash ASN1 description HashAlgorithmASN1, ) where import Crypto.Hash+import Crypto.Number.Serialize (os2ip) import Crypto.PubKey.Internal (and') import Crypto.PubKey.RSA (generateBlinder) import Crypto.PubKey.RSA.Prim import Crypto.PubKey.RSA.Types import Crypto.Random.Types +import Data.Bits (complement, shiftR, (.&.), (.|.)) import Data.ByteString (ByteString) import Data.Word import Crypto.Internal.ByteArray (ByteArray, Bytes) import qualified Crypto.Internal.ByteArray as B+import qualified Data.List as L -- | A specialized class for hash algorithm that can product -- a ASN1 wrapped description the algorithm plus the content@@ -407,29 +416,61 @@ padding = 0 : 1 : (replicate (klen - siglen - 3) 0xff ++ [0]) -- | Try to remove a standard PKCS1.5 encryption padding.+--+-- The block is scanned in full rather than up to the octet ending the padding+-- string, so how long that string is does not show up in how long this takes.+--+-- What remains visible is the result itself: whether the padding was well+-- formed, and the length of the message when it was. That is inherent to the+-- scheme, and it is the signal Bleichenbacher's attack needs, so a caller that+-- decrypts attacker-supplied ciphertext must not pass the distinction on --+-- TLS, for instance, continues with a random premaster secret and reports+-- nothing. unpad :: ByteArray bytearray => bytearray -> Either Error bytearray unpad packed | paddingSuccess = Right m | otherwise = Left MessageNotRecognized where+ len = B.length packed (zt, ps0m) = B.splitAt 2 packed- (ps, zm) = B.span (/= 0) ps0m- (z, m) = B.splitAt 1 zm++ -- index of the first zero octet in ps0m, counted from the start of packed,+ -- or len when there is none; every octet is looked at either way+ zeroIndex = fst $ L.foldl' step (fromIntegral len :: Word32, 1 :: Word32) indexed+ indexed = zip [2 ..] (B.unpack ps0m)+ step (idx, unseen) (i, b) = (select found i idx, unseen .&. complement found)+ where+ w = fromIntegral b :: Word32+ -- 0 when b is zero, 1 otherwise+ nonZero = (w .|. negate w) `shiftR` 31+ -- all ones at the first zero octet only+ found = negate (unseen .&. complement nonZero)+ select mask a b = (a .&. mask) .|. (b .&. complement mask)++ psLength = fromIntegral zeroIndex - 2 :: Int+ m = B.drop (fromIntegral zeroIndex + 1) packed paddingSuccess = and' [ zt `B.constEq` (B.pack [0, 2] :: Bytes)- , z == B.zero 1- , B.length ps >= 8+ , fromIntegral zeroIndex < len+ , psLength >= 8 ] -- | decrypt message using the private key. ----- When the decryption is not in a context where an attacker could gain--- information from the timing of the operation, the blinder can be set to None.------ If unsure always set a blinder or use decryptSafer+-- The blinder is optional and 'Nothing' is accepted, but see t'Blinder' for+-- what it covers and when leaving it out is a decision rather than a default.+-- 'decryptSafer' generates one for you. -- -- The message is returned un-padded.+--+-- Following RFC 8017, the ciphertext is rejected unless it is exactly as long+-- as the modulus (section 7.2.2, step 1) and its integer representative is+-- below the modulus (RSADP, section 5.1.2, step 1). The decryption primitive+-- normalises any multiple of the modulus away, so without the second check+-- @c + n@ would decrypt to the same message as @c@, and a ciphertext would not+-- be unique to its plaintext. Both checks are made on the ciphertext alone,+-- which is public, and report 'MessageSizeIncorrect'. decrypt :: ByteArray ba => Maybe Blinder@@ -441,6 +482,7 @@ -> Either Error ba decrypt blinder pk c | B.length c /= (private_size pk) = Left MessageSizeIncorrect+ | os2ip c >= private_n pk = Left MessageSizeIncorrect -- "convert" must be apply to "c". | otherwise = unpad $ dp blinder pk $ B.convert c @@ -470,10 +512,9 @@ -- | sign message using private key, a hash and its ASN1 description ----- When the signature is not in a context where an attacker could gain--- information from the timing of the operation, the blinder can be set to None.------ If unsure always set a blinder or use signSafer+-- The blinder is optional and 'Nothing' is accepted, but see t'Blinder' for+-- what it covers and when leaving it out is a decision rather than a default.+-- 'signSafer' generates one for you. sign :: HashAlgorithmASN1 hashAlg => Maybe Blinder@@ -502,6 +543,14 @@ return (sign (Just blinder) hashAlg pk m) -- | verify message with the signed message+--+-- Following RFC 8017, the signature is rejected unless it is exactly as long+-- as the modulus (section 8.2.2, step 1) and its integer representative is+-- below the modulus (section 5.2.2, step 1). Verification works by+-- re-encoding the expected signature and comparing it with the result of the+-- public-key operation, and that operation normalises away both the length of+-- the encoding and any multiple of the modulus; without these checks a+-- zero-padded signature, or @s + n@, would verify just as well as @s@. verify :: HashAlgorithmASN1 hashAlg => Maybe hashAlg@@ -512,9 +561,123 @@ -- ^ Signature -> Bool verify hashAlg pk m sm =- case makeSignature hashAlg (public_size pk) m of- Left _ -> False- Right s -> s == (ep pk sm)+ verifyEncoded pk (makeSignature hashAlg (public_size pk) m) sm++-- | The two checks of RFC 8017 and the comparison, shared by the three+-- verification entry points. The expected encoding is a thunk and is+-- forced only once the checks have passed, as it was when this was written+-- out inside 'verify'.+verifyEncoded+ :: PublicKey+ -> Either Error ByteString+ -- ^ the encoding a signature of this message would have+ -> ByteString+ -- ^ signature+ -> Bool+verifyEncoded pk expected sm+ | B.length sm /= public_size pk = False+ | os2ip sm >= public_n pk = False+ | otherwise =+ case expected of+ Left _ -> False+ Right s -> s == ep pk sm++-- | Sign a digest.+--+-- The digest's type says which algorithm made it, so this needs nothing+-- else to name one: the ASN.1 DigestInfo prefix comes from the+-- 'HashAlgorithmASN1' instance that type selects. 'sign' takes @Maybe+-- hashAlg@ and never reads the value inside the @Just@ -- it is there only+-- to fix the type -- which is no burden when the caller has a value and+-- leaves nothing to pass when the caller is itself polymorphic in the+-- algorithm.+--+-- > signDigest blinder key (hashWith SHA256 message)+--+-- The blinder is optional and 'Nothing' is accepted, but see t'Blinder' for+-- what it covers and when leaving it out is a decision rather than a+-- default. 'signSaferDigest' generates one for you.+signDigest+ :: HashAlgorithmASN1 hashAlg+ => Maybe Blinder+ -- ^ optional blinder+ -> PrivateKey+ -- ^ private key+ -> Digest hashAlg+ -- ^ digest of the message to sign+ -> Either Error ByteString+signDigest blinder pk digest =+ dp blinder pk `fmap` padSignature (private_size pk) (hashDigestASN1 digest)++-- | 'signDigest' with a blinder generated for the occasion, as 'signSafer'+-- is to 'sign'.+signSaferDigest+ :: (HashAlgorithmASN1 hashAlg, MonadRandom m)+ => PrivateKey+ -- ^ private key+ -> Digest hashAlg+ -- ^ digest of the message to sign+ -> m (Either Error ByteString)+signSaferDigest pk digest = do+ blinder <- generateBlinder (private_n pk)+ return (signDigest (Just blinder) pk digest)++-- | Sign something that is already a DigestInfo, the ASN.1 structure+-- naming a hash algorithm and carrying a digest under it.+--+-- This is what @'sign' blinder 'Nothing'@ does. It needs no+-- 'HashAlgorithmASN1' constraint, because nothing here hashes or encodes:+-- the caller has done both. @'sign' blinder 'Nothing'@ carries the+-- constraint anyway, and since the type variable then appears nowhere else+-- the caller has to name an algorithm that is never used --+-- @'sign' blinder ('Nothing' :: 'Maybe' 'Crypto.Hash.SHA256')@ -- to say+-- which one it is not using.+signDigestInfo+ :: Maybe Blinder+ -- ^ optional blinder+ -> PrivateKey+ -- ^ private key+ -> ByteString+ -- ^ a DigestInfo, encoded+ -> Either Error ByteString+signDigestInfo blinder pk di =+ dp blinder pk `fmap` padSignature (private_size pk) di++-- | 'signDigestInfo' with a blinder generated for the occasion.+signSaferDigestInfo+ :: MonadRandom m+ => PrivateKey+ -- ^ private key+ -> ByteString+ -- ^ a DigestInfo, encoded+ -> m (Either Error ByteString)+signSaferDigestInfo pk di = do+ blinder <- generateBlinder (private_n pk)+ return (signDigestInfo (Just blinder) pk di)++-- | Verify a signature over a digest. The checks are 'verify's.+verifyDigest+ :: HashAlgorithmASN1 hashAlg+ => PublicKey+ -> Digest hashAlg+ -- ^ digest of the message+ -> ByteString+ -- ^ signature+ -> Bool+verifyDigest pk digest sm =+ verifyEncoded pk (padSignature (public_size pk) (hashDigestASN1 digest)) sm++-- | Verify a signature over something that is already a DigestInfo, which+-- is what @'verify' 'Nothing'@ does.+verifyDigestInfo+ :: PublicKey+ -> ByteString+ -- ^ a DigestInfo, encoded+ -> ByteString+ -- ^ signature+ -> Bool+verifyDigestInfo pk di sm =+ verifyEncoded pk (padSignature (public_size pk) di) sm -- | make signature digest, used in 'sign' and 'verify' makeSignature
@@ -22,12 +22,13 @@ import Crypto.Hash import Crypto.Number.Basic (numBits)+import Crypto.Number.Serialize (os2ip) import Crypto.PubKey.MaskGenFunction import Crypto.PubKey.RSA (generateBlinder) import Crypto.PubKey.RSA.Prim import Crypto.PubKey.RSA.Types import Crypto.Random.Types-import Data.Bits (shiftR, xor, (.&.))+import Data.Bits (complement, shiftR, xor, (.&.)) import Data.Word import Crypto.Internal.ByteArray (ByteArray, ByteArrayAccess)@@ -68,6 +69,9 @@ -- | Sign using the PSS parameters and the salt explicitely passed as parameters. -- -- the function ignore SaltLength from the PSS Parameters+--+-- See t'Blinder' for what the optional blinder covers and when leaving it out+-- is a decision rather than a default. 'signSafer' generates one for you. signDigestWithSalt :: HashAlgorithm hash => ByteString@@ -102,6 +106,9 @@ -- | Sign using the PSS parameters and the salt explicitely passed as parameters. -- -- the function ignore SaltLength from the PSS Parameters+--+-- See t'Blinder' for what the optional blinder covers and when leaving it out+-- is a decision rather than a default. 'signSafer' generates one for you. signWithSalt :: HashAlgorithm hash => ByteString@@ -120,6 +127,9 @@ mHash = hashWith (pssHash params) m -- | Sign using the PSS Parameters+--+-- See t'Blinder' for what the optional blinder covers and when leaving it out+-- is a decision rather than a default. 'signSafer' generates one for you. sign :: (HashAlgorithm hash, MonadRandom m) => Maybe Blinder@@ -136,6 +146,9 @@ return (signWithSalt salt blinder params pk m) -- | Sign using the PSS Parameters+--+-- See t'Blinder' for what the optional blinder covers and when leaving it out+-- is a decision rather than a default. 'signSafer' generates one for you. signDigest :: (HashAlgorithm hash, MonadRandom m) => Maybe Blinder@@ -197,6 +210,13 @@ mHash = hashWith (pssHash params) m -- | Verify a signature using the PSS Parameters+--+-- Following RFC 8017, the signature is rejected unless it is exactly as long+-- as the modulus (section 8.1.2, step 1) and its integer representative is+-- below the modulus (RSAVP1, section 5.2.2, step 1). The public-key operation+-- normalises any multiple of the modulus away, so without the second check+-- @s + n@ would verify as readily as @s@, and a third party could turn one+-- valid signature into another without the private key. verifyDigest :: HashAlgorithm hash => PSSParams hash ByteString ByteString@@ -211,7 +231,9 @@ -> Bool verifyDigest params pk digest s | B.length s /= k = False+ | os2ip s >= public_n pk = False | B.any (/= 0) pre = False+ | B.any (\x -> x .&. topBits /= 0) (B.take 1 maskedDB) = False | B.last em /= pssTrailerField params = False | B.any (/= 0) ps0 = False | b1 /= B.singleton 1 = False@@ -225,6 +247,13 @@ emLen = if emTruncate pubBits then k - 1 else k dbLen = emLen - hashLen - 1 pubBits = numBits (public_n pk)+ -- RFC 8017 9.1.2 step 6: the leftmost 8*emLen - emBits bits of the+ -- leftmost octet of maskedDB have to be zero already. Step 9 clears+ -- them in DB, which is what normalizeToKeySize does below, and clearing+ -- is not checking: without this an encoding with the top bit set -- one+ -- the standard calls inconsistent -- verifies as though it were sound,+ -- because the bit that made it wrong is thrown away before it is read.+ topBits = complement (normalizeMask pubBits) -- unmarshall fields (pre, em) = B.splitAt (k - emLen) (ep pk s) -- drop 0..1 byte maskedDB = B.take dbLen em@@ -242,7 +271,12 @@ normalizeToKeySize :: Int -> [Word8] -> [Word8] normalizeToKeySize _ [] = [] -- very unlikely-normalizeToKeySize bits (x : xs) = x .&. mask : xs+normalizeToKeySize bits (x : xs) = x .&. normalizeMask bits : xs++-- | The bits of the leftmost octet that belong to the encoding: the low+-- @emBits `mod` 8@ of them, or all eight when that is zero. Its complement+-- is the bits RFC 8017 requires to be zero.+normalizeMask :: Int -> Word8+normalizeMask bits = if sh > 0 then 0xff `shiftR` (8 - sh) else 0xff where- mask = if sh > 0 then 0xff `shiftR` (8 - sh) else 0xff sh = (bits - 1) .&. 0x7
@@ -21,13 +21,38 @@ private_e, ) where +import Crypto.Debug (DebugShow (..)) import Crypto.Internal.Imports import Data.Data import GHC.Generics --- | Blinder which is used to obfuscate the timing--- of the decryption primitive (used by decryption and signing).+-- | A blinder, which keeps the timing of the private key operation from+-- saying anything about the number it was given.+--+-- The private exponent is not what is at risk. 'Crypto.Number.ModArithmetic.expSafe',+-- which the exponentiation goes through, keeps the /value/ of an exponent out+-- of the work it does.+--+-- What a blinder covers is the other side. Without one, the operation runs+-- on the ciphertext as it arrived, so how long it takes depends on a number+-- an attacker may have chosen and can vary -- which is what a remote timing+-- attack on RSA needs. With one, the input is multiplied by a random value+-- first and that value divided out afterwards, so the timing carries nothing+-- an attacker can steer.+--+-- Every private key operation here takes a @'Maybe' t'Blinder'@. The+-- @Safer@ form of each -- 'Crypto.PubKey.RSA.PKCS15.decryptSafer',+-- 'Crypto.PubKey.RSA.PKCS15.signSafer' and their kind -- generates one and is+-- the one to reach for. Pass 'Nothing' only where the input is not attacker+-- controlled and you have decided that it is not.+--+-- A blinder costs one more exponentiation, by the public exponent, which is+-- the cheap direction: measured on an Apple M4, PKCS#1 v1.5 signing goes from+-- about 601 to about 620 microseconds.+--+-- Use a blinder once. 'Crypto.PubKey.RSA.generateBlinder' makes a fresh one;+-- carrying one across operations is not what it is for. data Blinder = Blinder !Integer !Integer deriving (Show, Eq) @@ -84,8 +109,39 @@ , private_qinv :: Integer -- ^ q^(-1) mod p }- deriving (Show, Read, Eq, Data, Generic)+ deriving (Read, Eq, Data, Generic) +-- | The public part is shown; the secret fields are not. Use+-- 'Crypto.Debug.debugShow' to see them.+instance Show PrivateKey where+ showsPrec d k =+ showParen (d > 10) $+ showString "PrivateKey {private_pub = "+ . shows (private_pub k)+ . showString+ ", private_d = <secret>, private_p = <secret>\+ \, private_q = <secret>, private_dP = <secret>\+ \, private_dQ = <secret>, private_qinv = <secret>}"++instance DebugShow PrivateKey where+ debugShow k =+ showString "PrivateKey {private_pub = "+ . shows (private_pub k)+ . showString ", private_d = "+ . shows (private_d k)+ . showString ", private_p = "+ . shows (private_p k)+ . showString ", private_q = "+ . shows (private_q k)+ . showString ", private_dP = "+ . shows (private_dP k)+ . showString ", private_dQ = "+ . shows (private_dQ k)+ . showString ", private_qinv = "+ . shows (private_qinv k)+ . showChar '}'+ $ ""+ instance NFData PrivateKey where rnf (PrivateKey pub d p q dp dq qinv) = rnf pub `seq`@@ -113,7 +169,14 @@ -- -- note the RSA private key contains already an instance of public key for efficiency newtype KeyPair = KeyPair PrivateKey- deriving (Show, Read, Eq, Data, NFData)+ deriving (Read, Eq, Data, NFData)++instance Show KeyPair where+ showsPrec d (KeyPair k) =+ showParen (d > 10) $ showString "KeyPair " . showsPrec 11 k++instance DebugShow KeyPair where+ debugShow (KeyPair k) = "KeyPair (" ++ debugShow k ++ ")" -- | Public key of a RSA KeyPair toPublicKey :: KeyPair -> PublicKey
@@ -8,6 +8,55 @@ -- Portability : unknown -- -- Rabin cryptosystem for public-key cryptography and digital signature.+--+-- == What is kept from the clock, and what is not+--+-- The square roots modulo the secret primes are taken with+-- 'Crypto.Number.ModArithmetic.expSafe', which does not read the exponent it+-- is given. Two things here do read what they are given.+--+-- Signing asks for the Jacobi symbol of the hash modulo each of the two+-- private primes, and the Jacobi symbol is computed by a sequence of+-- reductions whose number follows both of its arguments -- so the work done+-- per signature follows the primes. Key generation runs the extended+-- Euclidean algorithm on the two primes for the same reason. Neither has a+-- drop-in replacement here: a Jacobi symbol that does not read its arguments+-- is a different algorithm, not a different call.+--+-- Around all of that is 'Integer' arithmetic, whose cost follows the size of+-- the numbers; see "Crypto.PubKey.DSA" for that note at more length.+--+-- == The hash algorithm is passed as a value, and that is not going to change+--+-- 'sign', 'signWith' and 'verify' take a @hash@ argument whose value they+-- never read. Every 'Crypto.Hash.HashAlgorithm' instance is a nullary+-- constructor, and the algorithm comes from the type; the value is there to+-- carry the type and nothing else. A caller holding a value loses nothing+-- by it, and a caller that is itself polymorphic in the algorithm has no+-- value to pass.+--+-- Elsewhere in crypton that is answered by taking a+-- 'Crypto.Hash.Digest' instead: the digest carries the algorithm in its+-- type and the bytes in its value, so nothing is passed only to name a+-- type. "Crypto.PubKey.RSA.PKCS15", "Crypto.PubKey.DSA",+-- "Crypto.PubKey.ElGamal", "Crypto.PubKey.Rabin.RW" and+-- "Crypto.PubKey.Rabin.Modified" all do that.+--+-- This module cannot. What is hashed here is not the message:+--+-- > h = os2ip $ hashWith hashAlg $ B.append padding m+--+-- and the padding is not the caller's either. 'sign' searches for one,+-- drawing eight bytes at a time until the first octet is non-zero and the+-- Jacobi symbols of the hash modulo each private prime are both 1. Which+-- bytes get hashed is therefore decided inside the signing operation, after+-- the caller has handed over the message, so there is no digest for a+-- caller to compute in advance.+--+-- A proxy argument would work where a digest cannot. It is deliberately+-- not added: it would be the one exception to a rule the rest of the+-- library now follows, and this module has no callers asking for it. See+-- the survey in <https://github.com/kazu-yamamoto/crypton/issues/304>. module Crypto.PubKey.Rabin.Basic ( PublicKey (..), PrivateKey (..),@@ -21,6 +70,7 @@ verify, ) where +import Crypto.Debug (DebugShow (..)) import Data.ByteString (ByteString) import qualified Data.ByteString as B import Data.Data@@ -53,8 +103,32 @@ , private_a :: Integer , private_b :: Integer }- deriving (Show, Read, Eq, Data)+ deriving (Read, Eq, Data) +-- | The public part is shown; the secret fields are not. Use+-- 'Crypto.Debug.debugShow' to see them.+instance Show PrivateKey where+ showsPrec d k =+ showParen (d > 10) $+ showString "PrivateKey {private_pub = "+ . shows (private_pub k)+ . showString ", private_p = <secret>, private_q = <secret>, private_a = <secret>, private_b = <secret>}"++instance DebugShow PrivateKey where+ debugShow k =+ showString "PrivateKey {private_pub = "+ . shows (private_pub k)+ . showString ", private_p = "+ . shows (private_p k)+ . showString ", private_q = "+ . shows (private_q k)+ . showString ", private_a = "+ . shows (private_a k)+ . showString ", private_b = "+ . shows (private_b k)+ . showChar '}'+ $ ""+ -- | Rabin Signature. data Signature = Signature (Integer, Integer) deriving (Show, Read, Eq, Data) @@ -128,6 +202,12 @@ -- | Decrypt ciphertext using private key. --+-- The ciphertext has to be what 'encrypt' produces: the big-endian encoding,+-- with no leading zero octet, of a value below the modulus. Squaring and the+-- square roots that undo it work modulo n, so without that condition @c@ and+-- @c + n@ -- and @c@ with a zero octet in front of it -- would all decrypt to+-- the same message, and a ciphertext would not be unique to its plaintext.+-- -- See algorithm 8.12 in "Handbook of Applied Cryptography" by Alfred J. Menezes et al. decrypt :: HashAlgorithm hash@@ -138,18 +218,21 @@ -> ByteString -- ^ ciphertext -> Maybe ByteString-decrypt oaep pk c =- let p = private_p pk- q = private_q pk- a = private_a pk- b = private_b pk- n = public_n $ private_pub pk- k = numBytes n- c' = os2ip c- solutions = rights $ toList $ mapTuple (unpad oaep k . i2ospOf_ k) $ sqroot' c' p q a b n- in case solutions of- [x] -> Just x- _ -> Nothing+decrypt oaep pk c+ | os2ip c >= public_n (private_pub pk) = Nothing+ | c /= (i2osp (os2ip c) :: ByteString) = Nothing+ | otherwise =+ let p = private_p pk+ q = private_q pk+ a = private_a pk+ b = private_b pk+ n = public_n $ private_pub pk+ k = numBytes n+ c' = os2ip c+ solutions = rights $ toList $ mapTuple (unpad oaep k . i2ospOf_ k) $ sqroot' c' p q a b n+ in case solutions of+ [x] -> Just x+ _ -> Nothing where toList (w, x, y, z) = w : x : y : z : [] mapTuple f (w, x, y, z) = (f w, f x, f y, f z)@@ -168,10 +251,14 @@ -> ByteString -- ^ message to sign -> Either Error Signature-signWith padding pk hashAlg m = do- h <- calculateHash padding pk hashAlg m- signature <- calculateSignature h- return signature+signWith padding pk hashAlg m+ -- the signature carries the padding as an integer, so a leading zero octet+ -- would not survive it: verify would hash one octet less than was signed+ | B.null padding || B.index padding 0 == 0 = Left InvalidParameters+ | otherwise = do+ h <- calculateHash padding pk hashAlg m+ signature <- calculateSignature h+ return signature where calculateSignature h = let p = private_p pk@@ -203,8 +290,10 @@ where findPadding = do padding <- getRandomBytes 8- case calculateHash padding pk hashAlg m of- Right _ -> return padding+ case (B.index padding 0, calculateHash padding pk hashAlg m) of+ -- a padding that starts with a zero octet is one signWith refuses+ (0, _) -> findPadding+ (_, Right _) -> return padding _ -> findPadding -- | Calculate hash of message and padding.@@ -242,12 +331,17 @@ -> Signature -- ^ signature -> Bool-verify pk hashAlg m (Signature (padding, s)) =- let n = public_n pk- p = i2osp padding- h = os2ip $ hashWith hashAlg $ B.append p m- h' = expSafe s 2 n- in h' == h+verify pk hashAlg m (Signature (padding, s))+ -- squaring works modulo n, so s + n and -s would verify wherever s does+ | s < 0 || s >= n = False+ | padding < 0 = False+ | otherwise =+ let p = i2osp padding+ h = os2ip $ hashWith hashAlg $ B.append p m+ h' = expSafe s 2 n+ in h' == h+ where+ n = public_n pk -- | Square roots modulo prime p where p is congruent 3 mod 4 -- Value a must be a quadratic residue modulo p (i.e. jacobi symbol (a/n) = 1).
@@ -9,14 +9,20 @@ -- -- Modified-Rabin public-key digital signature algorithm. -- See algorithm 11.30 in "Handbook of Applied Cryptography" by Alfred J. Menezes et al.+-- The Jacobi symbols here are taken modulo the public modulus, not the+-- private primes, so what "Crypto.PubKey.Rabin.Basic" says about that does+-- not apply; the note there about 'Integer' arithmetic does. module Crypto.PubKey.Rabin.Modified ( PublicKey (..), PrivateKey (..), generate, sign,+ signDigest, verify,+ verifyDigest, ) where +import Crypto.Debug (DebugShow (..)) import Data.ByteString import Data.Data @@ -44,8 +50,30 @@ -- ^ q prime number , private_d :: Integer }- deriving (Show, Read, Eq, Data)+ deriving (Read, Eq, Data) +-- | The public part is shown; the secret fields are not. Use+-- 'Crypto.Debug.debugShow' to see them.+instance Show PrivateKey where+ showsPrec d k =+ showParen (d > 10) $+ showString "PrivateKey {private_pub = "+ . shows (private_pub k)+ . showString ", private_p = <secret>, private_q = <secret>, private_d = <secret>}"++instance DebugShow PrivateKey where+ debugShow k =+ showString "PrivateKey {private_pub = "+ . shows (private_pub k)+ . showString ", private_p = "+ . shows (private_p k)+ . showString ", private_q = "+ . shows (private_q k)+ . showString ", private_d = "+ . shows (private_d k)+ . showChar '}'+ $ ""+ -- | Generate a pair of (private, public) key of size in bytes. -- Prime p is congruent 3 mod 8 and prime q is congruent 7 mod 8. generate@@ -83,10 +111,25 @@ -> ByteString -- ^ message to sign -> Either Error Integer-sign pk hashAlg m =+sign pk hashAlg m = signDigest pk (hashWith hashAlg m)++-- | Sign a digest using the private key.+--+-- The digest's type says which algorithm made it, so this needs nothing else+-- to name one. 'sign' takes a @hash@ value and never reads it -- it is there+-- to fix the type -- which leaves a caller that is itself polymorphic in the+-- algorithm with nothing to pass.+signDigest+ :: HashAlgorithm hash+ => PrivateKey+ -- ^ private key+ -> Digest hash+ -- ^ digest of the message to sign+ -> Either Error Integer+signDigest pk digest = let d = private_d pk n = public_n $ private_pub pk- h = os2ip $ hashWith hashAlg m+ h = os2ip digest limit = (n - 6) `div` 16 in if h > limit then Left MessageTooLong@@ -109,18 +152,36 @@ -> Integer -- ^ signature -> Bool-verify pk hashAlg m s =- let n = public_n pk- h = os2ip $ hashWith hashAlg m- s' = expSafe s 2 n- s'' = case s' `mod` 8 of- 6 -> s'- 3 -> 2 * s'- 7 -> n - s'- 2 -> 2 * (n - s')- _ -> 0- in case s'' `mod` 16 of- 6 ->- let h' = (s'' - 6) `div` 16- in h' == h- _ -> False+verify pk hashAlg m s = verifyDigest pk (hashWith hashAlg m) s++-- | Verify a signature over a digest. See 'signDigest' for why a digest+-- rather than a @hash@ value.+verifyDigest+ :: HashAlgorithm hash+ => PublicKey+ -- ^ public key+ -> Digest hash+ -- ^ digest of the message+ -> Integer+ -- ^ signature+ -> Bool+verifyDigest pk digest s+ -- squaring works modulo n, so s + n and -s would verify wherever s does+ | s < 0 || s >= n = False+ | otherwise = go+ where+ n = public_n pk+ go =+ let h = os2ip digest+ s' = expSafe s 2 n+ s'' = case s' `mod` 8 of+ 6 -> s'+ 3 -> 2 * s'+ 7 -> n - s'+ 2 -> 2 * (n - s')+ _ -> 0+ in case s'' `mod` 16 of+ 6 ->+ let h' = (s'' - 6) `div` 16+ in h' == h+ _ -> False
@@ -14,13 +14,15 @@ unpad, ) where -import Data.Bits (xor)+import Data.Bits (complement, shiftR, xor, (.&.), (.|.)) import Data.ByteString (ByteString) import qualified Data.ByteString as B+import qualified Data.List as L+import Data.Word (Word32) import Crypto.Hash import Crypto.Internal.ByteArray (ByteArray, ByteArrayAccess)-import qualified Crypto.Internal.ByteArray as B (convert)+import qualified Crypto.Internal.ByteArray as B (constEq, convert) import Crypto.PubKey.Internal (and') import Crypto.PubKey.MaskGenFunction import Crypto.PubKey.Rabin.Types@@ -80,6 +82,17 @@ em = B.concat [B.singleton 0x0, maskedSeed, maskedDB] -- | Un-pad a OAEP encoded message.+--+-- The data block is scanned in full rather than up to the 01 octet separating+-- the padding from the message, and the label hash and the leading octet are+-- compared without an early exit, so neither the length of the padding nor+-- where a comparison first differs shows up in how long this takes. This is+-- what "Crypto.PubKey.RSA.OAEP" does with the same block.+--+-- What remains visible is the result itself: whether the block was well formed,+-- and the length of the message when it was. That is the signal Manger's+-- attack needs, so a caller that decrypts attacker-supplied ciphertext must not+-- pass the distinction on. unpad :: HashAlgorithm hash => OAEPParams hash ByteString ByteString@@ -95,7 +108,9 @@ where -- parameters mgf = oaepMaskGenAlg oaep- labelHash = B.convert $ hashWith (oaepHash oaep) (maybe B.empty id $ oaepLabel oaep)+ labelHash =+ B.convert $ hashWith (oaepHash oaep) (maybe B.empty id $ oaepLabel oaep)+ :: ByteString hashLen = hashDigestSize (oaepHash oaep) -- getting em's fields (pb, em0) = B.splitAt 1 em@@ -106,12 +121,30 @@ db = B.pack $ B.zipWith xor maskedDB dbmask -- getting db's fields (labelHash', db1) = B.splitAt hashLen db- (_, db2) = B.break (/= 0) db1- (ps1, msg) = B.splitAt 1 db2 + -- index of the first nonzero octet in db1, or its length when every octet+ -- is zero; all of them are looked at either way+ oneIndex =+ fst $+ L.foldl'+ step+ (fromIntegral (B.length db1) :: Word32, 1 :: Word32)+ (zip [0 ..] (B.unpack db1))+ step (idx, unseen) (i, b) = (select found i idx, unseen .&. complement found)+ where+ w = fromIntegral b :: Word32+ -- 0 when b is zero, 1 otherwise+ nonZero = (w .|. negate w) `shiftR` 31+ -- all ones at the first nonzero octet only+ found = negate (unseen .&. nonZero)+ select mask a b = (a .&. mask) .|. (b .&. complement mask)++ ps1 = B.take 1 $ B.drop (fromIntegral oneIndex) db1+ msg = B.drop (fromIntegral oneIndex + 1) db1+ paddingSuccess = and'- [ labelHash' == labelHash -- no need for constant eq- , ps1 == B.replicate 1 0x1- , pb == B.replicate 1 0x0+ [ labelHash' `B.constEq` labelHash+ , ps1 `B.constEq` B.replicate 1 0x1+ , pb `B.constEq` B.replicate 1 0x0 ]
@@ -10,6 +10,9 @@ -- Rabin-Williams cryptosystem for public-key encryption and digital signature. -- See pages 323 - 324 in "Computational Number Theory and Modern Cryptography" by Song Y. Yan. -- Also inspired by https://github.com/vanilala/vncrypt/blob/master/vncrypt/vnrw_gmp.c.+-- The Jacobi symbols here are taken modulo the public modulus, not the+-- private primes, so what "Crypto.PubKey.Rabin.Basic" says about that does+-- not apply; the note there about 'Integer' arithmetic does. module Crypto.PubKey.Rabin.RW ( PublicKey (..), PrivateKey (..),@@ -18,9 +21,12 @@ encryptWithSeed, decrypt, sign,+ signDigest, verify,+ verifyDigest, ) where +import Crypto.Debug (DebugShow (..)) import Data.ByteString import Data.Data @@ -50,8 +56,30 @@ -- ^ q prime number , private_d :: Integer }- deriving (Show, Read, Eq, Data)+ deriving (Read, Eq, Data) +-- | The public part is shown; the secret fields are not. Use+-- 'Crypto.Debug.debugShow' to see them.+instance Show PrivateKey where+ showsPrec d k =+ showParen (d > 10) $+ showString "PrivateKey {private_pub = "+ . shows (private_pub k)+ . showString ", private_p = <secret>, private_q = <secret>, private_d = <secret>}"++instance DebugShow PrivateKey where+ debugShow k =+ showString "PrivateKey {private_pub = "+ . shows (private_pub k)+ . showString ", private_p = "+ . shows (private_p k)+ . showString ", private_q = "+ . shows (private_q k)+ . showString ", private_d = "+ . shows (private_d k)+ . showChar '}'+ $ ""+ -- | Generate a pair of (private, public) key of size in bytes. -- Prime p is congruent 3 mod 8 and prime q is congruent 7 mod 8. generate@@ -118,6 +146,12 @@ hashLen = hashDigestSize (oaepHash oaep) -- | Decrypt ciphertext using private key.+--+-- The ciphertext has to be what 'encrypt' produces: the big-endian encoding,+-- with no leading zero octet, of a value below the modulus. The primitives+-- work modulo n, so without that condition @c@ and @c + n@ -- and @c@ with a+-- zero octet in front of it -- would all decrypt to the same message, and a+-- ciphertext would not be unique to its plaintext. decrypt :: HashAlgorithm hash => OAEPParams hash ByteString ByteString@@ -127,14 +161,17 @@ -> ByteString -- ^ ciphertext -> Maybe ByteString-decrypt oaep pk c =- let d = private_d pk- n = public_n $ private_pub pk- k = numBytes n- c' = i2ospOf_ k $ dp2 n $ dp1 d n $ os2ip c- in case unpad oaep k c' of- Left _ -> Nothing- Right p -> Just p+decrypt oaep pk c+ | os2ip c >= public_n (private_pub pk) = Nothing+ | c /= (i2osp (os2ip c) :: ByteString) = Nothing+ | otherwise =+ let d = private_d pk+ n = public_n $ private_pub pk+ k = numBytes n+ c' = i2ospOf_ k $ dp2 n $ dp1 d n $ os2ip c+ in case unpad oaep k c' of+ Left _ -> Nothing+ Right p -> Just p -- | Sign message using hash algorithm and private key. sign@@ -146,11 +183,26 @@ -> ByteString -- ^ message to sign -> Either Error Integer-sign pk hashAlg m =+sign pk hashAlg m = signDigest pk (hashWith hashAlg m)++-- | Sign a digest using the private key.+--+-- The digest's type says which algorithm made it, so this needs nothing else+-- to name one. 'sign' takes a @hash@ value and never reads it -- it is there+-- to fix the type -- which leaves a caller that is itself polymorphic in the+-- algorithm with nothing to pass.+signDigest+ :: HashAlgorithm hash+ => PrivateKey+ -- ^ private key+ -> Digest hash+ -- ^ digest of the message to sign+ -> Either Error Integer+signDigest pk digest = let d = private_d pk n = public_n $ private_pub pk in do- m' <- ep1 n $ os2ip $ hashWith hashAlg m+ m' <- ep1 n $ os2ip digest return $ dp1 d n m' -- | Verify signature using hash algorithm and public key.@@ -165,11 +217,28 @@ -> Integer -- ^ signature -> Bool-verify pk hashAlg m s =- let n = public_n pk- h = os2ip $ hashWith hashAlg m- h' = dp2 n $ ep2 n s- in h' == h+verify pk hashAlg m s = verifyDigest pk (hashWith hashAlg m) s++-- | Verify a signature over a digest. See 'signDigest' for why a digest+-- rather than a @hash@ value.+verifyDigest+ :: HashAlgorithm hash+ => PublicKey+ -- ^ public key+ -> Digest hash+ -- ^ digest of the message+ -> Integer+ -- ^ signature+ -> Bool+verifyDigest pk digest s+ -- squaring works modulo n, so s + n and -s would verify wherever s does+ | s < 0 || s >= n = False+ | otherwise =+ let h = os2ip digest+ h' = dp2 n $ ep2 n s+ in h' == h+ where+ n = public_n pk -- | Encryption primitive 1 ep1 :: Integer -> Integer -> Either Error Integer
@@ -6,6 +6,7 @@ -- Portability : unknown module Crypto.PubKey.Rabin.Types ( Error (..),+ PrimeCondition, generatePrimes, ) where
@@ -5,23 +5,49 @@ -- Stability : experimental -- Portability : Good module Crypto.Random.Probabilistic (- probabilistic,+ probabilisticFrom, ) where +import Crypto.Hash (SHA512 (..), hashWith)+import Crypto.Internal.ByteArray (ByteArrayAccess, ScrubbedBytes)+import qualified Crypto.Internal.ByteArray as B import Crypto.Internal.Compat import Crypto.Random+import Crypto.Random.ChaChaDRG (initialize) --- | This create a random number generator out of thin air with--- the system entropy; don't generally use as the IO is not exposed--- this can have unexpected random for.+-- | Run a probabilistic algorithm on a generator derived from the value it is+-- about to work on, and from a secret this process drew once. ----- This is useful for probabilistic algorithm like Miller Rabin--- probably prime algorithm, given appropriate choice of the heuristic+-- This is useful for a probabilistic algorithm like the Miller-Rabin primality+-- test, where the caller is a pure function and has to behave like one: the+-- same value has to give the same answer for as long as the process lives.+-- Deriving the generator from the value gives that much, and it keeps the+-- draws made for two different values independent of each other -- one+-- generator made once and shared by every call would make the witnesses drawn+-- for one value the witnesses for every value. --+-- The process secret is what makes the derivation unpredictable. The values+-- worked on may come from wherever the caller's input comes from, so the+-- generator must not be something that can be worked out from them.+--+-- The IO is not exposed and the result is not reproducible between processes. -- Generally, it's advised not to use this function.-probabilistic :: MonadPseudoRandom ChaChaDRG a -> a-probabilistic f = fst $ withDRG drg f+probabilisticFrom+ :: ByteArrayAccess seed+ => seed+ -- ^ the value being worked on, as bytes+ -> MonadPseudoRandom ChaChaDRG a+ -> a+probabilisticFrom material f = fst $ withDRG drg f where- {-# NOINLINE drg #-}- drg = unsafeDoIO drgNew-{-# NOINLINE probabilistic #-}+ drg = initialize (B.take seedLength (B.convert digest :: ScrubbedBytes))+ digest = hashWith SHA512 (B.append secret (B.convert material) :: ScrubbedBytes)+ -- what Crypto.Random.ChaChaDRG.initialize wants, and no more than SHA-512+ -- produces+ seedLength = 40++-- | Drawn once, for the lifetime of the process: it is the only part of the+-- derivation above that an attacker supplying values cannot see.+secret :: ScrubbedBytes+secret = unsafeDoIO (getRandomBytes 32)+{-# NOINLINE secret #-}
@@ -15,6 +15,18 @@ import Crypto.Random.Entropy -- | A monad constraint that allows to generate random bytes+--+-- Everything in this library that draws a key, a nonce or a signature's+-- randomness draws it through this class, and cannot tell a strong source+-- from a weak one. The two instances below are the ones to reach for:+-- @IO@ reads the system entropy source, and @MonadPseudoRandom@ runs a+-- 'DRG' seeded from it.+--+-- An instance of your own is held to the same standard. A generator that+-- another party can predict, or that repeats, yields keys and signatures+-- that give away what they are meant to keep -- so the deliberately+-- repeatable instance that makes a test reproducible is not one to ship+-- with. class Monad m => MonadRandom m where getRandomBytes :: ByteArray byteArray => Int -> m byteArray @@ -53,6 +65,6 @@ getRandomBytes n = MonadPseudoRandom (randomBytesGenerate n) -- | Run a pure computation with a Deterministic Random Generator--- in the 'MonadPseudoRandom'+-- in the t'MonadPseudoRandom' withDRG :: DRG gen => gen -> MonadPseudoRandom gen a -> (a, gen) withDRG gen m = runPseudoRandom m gen
@@ -160,6 +160,7 @@ -- > import Data.ByteString (ByteString) -- > import qualified Data.ByteString as B -- >+-- > import Crypto.Error (throwCryptoError) -- > import qualified Crypto.Cipher.XSalsa as XSalsa -- > import qualified Crypto.MAC.Poly1305 as Poly1305 -- > import qualified Crypto.PubKey.Curve25519 as X25519@@ -175,7 +176,8 @@ -- > state1 = XSalsa.derive state0 iv1 -- > (rs, state2) = XSalsa.generate state1 32 -- > (c, _) = XSalsa.combine state2 content--- > tag = Poly1305.auth (rs :: ByteString) c+-- > macKey = throwCryptoError (Poly1305.key (rs :: ByteString))+-- > tag = Poly1305.auth macKey c -- > -- > -- | Try to open a @crypto_box@ packet and recover the content using the -- > -- 192-bit nonce, sender public key and receiver private key.@@ -192,4 +194,5 @@ -- > state1 = XSalsa.derive state0 iv1 -- > (rs, state2) = XSalsa.generate state1 32 -- > (content, _) = XSalsa.combine state2 c--- > tag = Poly1305.auth (rs :: ByteString) c+-- > macKey = throwCryptoError (Poly1305.key (rs :: ByteString))+-- > tag = Poly1305.auth macKey c
@@ -1,4 +1,5 @@ Copyright (c) 2006-2015 Vincent Hanquez <vincent@snarc.org>+Copyright (c) 2023-2026 Kazu Yamamoto <kazu@iij.ad.jp> All rights reserved.
@@ -3,91 +3,252 @@ crypton ========== -Crypton is a fork from cryptonite with the original author's permission.+`crypton` is a fork from `cryptonite` with the original author's permission. -Crypton is a haskell repository of cryptographic primitives. Each crypto-algorithm has specificities that are hard to wrap in common APIs and types,-so instead of trying to provide a common ground for algorithms, this package-provides a non-consistent low-level API. -If you have no idea what you're doing, please do not use this directly.-Instead, rely on higher level protocols or implementations.+`crypton` is a low-level cryptography library. To achieve high+performance, it utilizes C and assembly language to define FFI+bindings, structuring them in a way that makes them easy to use. -Documentation: [crypton on hackage](http://hackage.haskell.org/package/crypton) -Stability----------+Side channels+------------- -Crypton APIs are stable, and we only strive to add, not change or remove.-Note that because the API exposed is wide and also expose internals things (for-power users and flexibility), certains APIs can be revised in extreme cases-where we can't just add.+AES is where this matters most, and which implementation runs is decided at+runtime from what the processor has. -Versioning-----------+On x86-64 with AES-NI and carry-less multiply, and on AArch64 with the ARMv8+cryptographic extension, AES and GHASH are instructions rather than tables.+crypton's AES and AES-GCM then make no branch and no memory access that+depends on the key or on the data: the secrets stay in vector registers and+never reach one a branch can test, which the generated code is checked+against. Every x86-64 part since about 2010 and every AArch64 part in+ordinary use has these. -Next version of `0.x` is `0.(x+1)`. There's no exceptions, or API related meaning-behind the numbers.+Where neither is present crypton falls back to a table-driven AES, which+indexes a 256-byte substitution table with data derived from the key and the+input. **That is not constant time**, and on a machine where an attacker can+observe the cache it is open to a timing attack. The fallback exists so that+the library builds and runs everywhere; it is not meant for a setting where+that matters. -Coding Style-------------+`Crypto.System.CPU.processorOptions` says which is in use. `AESNI` in that+list means the instruction path, and `PCLMUL` that GHASH has its instruction+too; without `AESNI` it is the tables. The list also reports `RDRAND`, which+is unrelated to this. -The coding style of this project mostly follows:-[haskell-style](https://github.com/tibbe/haskell-style-guide/blob/master/haskell-style.md)+ ghci> import Crypto.System.CPU+ ghci> processorOptions+ [AESNI,PCLMUL] -Support--------+RSA is the other place to know about, and there the choice is the caller's.+The private key operations in `Crypto.PubKey.RSA.PKCS15`, `.OAEP` and `.PSS`+take a `Maybe Blinder`, and `Nothing` is no harder to write than the safe+form: -See [Haskell packages guidelines](https://github.com/vincenthz/haskell-pkg-guidelines/blob/master/README.md#support)+ decrypt :: Maybe Blinder -> PrivateKey -> ByteString -> ...+ decryptSafer :: MonadRandom m => PrivateKey -> ByteString -> m ... -Known Building Issues----------------------+The exponent itself is not what is at risk. `expSafe` keeps the *value* of an+exponent out of the work it does, so the private exponent does not leak+through the exponentiation. What a blinder covers is the other side: without+one, the operation runs on the ciphertext the caller was handed, so how long+it takes depends on a number an attacker may have chosen and can vary. That+is what a remote timing attack on RSA needs. With a blinder the input is+multiplied by a random value first and the result divided out afterwards, so+the timing carries nothing an attacker can steer. -On OSX <= 10.7, the system compiler doesn't understand the '-maes' option, and-with the lack of autodetection feature builtin in .cabal file, it is left on-the user to disable the aesni. See the [Disabling AESNI] section+`decryptSafer` and `signSafer` generate the blinder themselves and are the+ones to reach for. Pass `Nothing` only where the input is not attacker+controlled and you have decided that it is not. -On CentOS 7 the default C compiler includes intrinsic header files incompatible-with per-function target options. Solutions are to use GCC >= 4.9 or disable-flag *use_target_attributes* (see flag configuration examples below).+The RSA rows in the tables below are the unblinded path. A blinder costs one+more exponentiation, by the public exponent, which is the cheap direction:+measured on the M4, signing goes from about 460 to about 476 microseconds,+under four per cent. -Disabling AESNI----------------+Performance+----------- -It may be useful to disable AESNI for building, testing or runtime purposes.-This is achieved with the *support_aesni* flag.+The algorithms a TLS connection uses, measured against the last release+before the rewrite and against OpenSSL on the same machine. Throughput is+over 16 KiB messages; the public key operations are one operation each; every+figure is the best of several runs, and crypton and OpenSSL are run+alternately so that neither gets the quieter machine. -As part of configure of crypton:+Bulk encryption and hashing are measured through crypton's C layer, as+`openssl speed` measures OpenSSL's. The public key operations are measured+through crypton's Haskell API, since that is where ECDSA and RSA live and it+is what a program actually calls; the Haskell layer adds well under a+microsecond, which the X25519 and ECDH P-256 rows confirm by agreeing with a+C-level measurement to within a percent. Both releases of crypton are built+the same way -- `-optc-O3`, which is what each asks for -- and by+`cabal build`, since a copy of the sources compiled by hand does not measure+what a program linking the library gets, and leaves out whole implementations+without saying so. Each column of a table comes from one run on the machine+named above it. -```- cabal configure --flag='-support_aesni'-```+### x86-64 -or as part of an installation:+An AMD EPYC 7763, which has AES-NI, PCLMULQDQ, AVX2, ADX, VAES, VPCLMULQDQ+and the SHA extensions, against OpenSSL 4.0.3. -```- cabal install --constraint="crypton -support_aesni"-```+Throughput in MB/s, **higher is better**: -For help with cabal flags, see: [stackoverflow : is there a way to define flags for cabal](http://stackoverflow.com/questions/23523869/is-there-any-way-to-define-flags-for-cabal-dependencies)+| | crypton 1.1.5 | crypton 2.1.5 | OpenSSL | 2.1.5 / OpenSSL |+| --- | ---: | ---: | ---: | ---: |+| AES-128-GCM | 1362 | **6038** | 4055 | 1.49 |+| AES-256-GCM | 1093 | **5462** | 3770 | 1.45 |+| ChaCha20-Poly1305 | 399 | 2211 | 2229 | 0.99 |+| SHA-1 | 727 | 1678 | 1673 | 1.00 |+| SHA-256 | 290 | 1585 | 1579 | 1.00 |+| SHA-512 | 463 | 804 | 751 | 1.07 |+| SHA3-256 | 109 | 424 | 425 | 1.00 | -Links------+Time per operation in microseconds, **lower is better**: -* [ChaCha](http://cr.yp.to/chacha.html)-* [ChaCha-test-vectors](https://github.com/secworks/chacha_testvectors.git)-* [Poly1305](http://cr.yp.to/mac.html)-* [Poly1305-test-vectors](http://tools.ietf.org/html/draft-nir-cfrg-chacha20-poly1305-06#page-12)-* [Salsa](http://cr.yp.to/snuffle.html)-* [Salsa128-test-vectors](https://github.com/alexwebr/salsa20/blob/master/test_vectors.128)-* [Salsa256-test-vectors](https://github.com/alexwebr/salsa20/blob/master/test_vectors.256)-* [XSalsa](https://cr.yp.to/snuffle/xsalsa-20081128.pdf)-* [PBKDF2](http://tools.ietf.org/html/rfc2898)-* [PBKDF2-test-vectors](http://www.ietf.org/rfc/rfc6070.txt)-* [Scrypt](http://www.tarsnap.com/scrypt.html)-* [Curve25519](http://cr.yp.to/ecdh.html)-* [Ed25519](http://ed25519.cr.yp.to/papers.html)-* [Ed448-Goldilocks](http://ed448goldilocks.sourceforge.net/)-* [EdDSA-test-vectors](http://www.ietf.org/rfc/rfc8032.txt)-* [AFIS](http://clemens.endorphin.org/cryptography)+| | crypton 1.1.5 | crypton 2.1.5 | OpenSSL | OpenSSL / 2.1.5 |+| --- | ---: | ---: | ---: | ---: |+| X25519 | 45.31 | 28.41 | 36.48 | 1.28 |+| ECDH P-256 | 165.4 | 51.14 | 51.65 | 1.01 |+| ECDH P-384 | 2278 | **165.0** | 847.5 | 5.13 |+| Ed25519 sign | 30.03 | 18.62 | 33.71 | 1.81 |+| Ed25519 verify | 48.05 | 47.66 | 110.6 | 2.32 |+| ECDSA P-256 sign | 81.70 | 18.96 | 21.87 | 1.15 |+| ECDSA P-256 verify | 233.3 | 70.47 | 67.52 | 0.96 |+| ECDSA P-384 sign | 2264 | **303.4** | 890.1 | 2.93 |+| ECDSA P-384 verify | 2676 | **471.4** | 721.5 | 1.53 |+| RSA-2048 sign/decrypt | 759.4 | 612.0 | 659.4 | 1.08 |+| RSA-2048 verify/encrypt | 33.56 | 30.21 | 18.86 | 0.62 | +### AArch64++An Apple M4, which has the AES, PMULL, SHA-1, SHA-2, SHA-512 and SHA-3+instructions, against OpenSSL 4.0.3.++Throughput in MB/s, **higher is better**:++| | crypton 1.1.5 | crypton 2.1.5 | OpenSSL | 2.1.5 / OpenSSL |+| --- | ---: | ---: | ---: | ---: |+| AES-128-GCM | 127 | **12422** | 10846 | 1.15 |+| AES-256-GCM | 98 | **9721** | 9197 | 1.06 |+| ChaCha20-Poly1305 | 771 | 2319 | 2250 | 1.03 |+| SHA-1 | 1209 | 3389 | 3361 | 1.01 |+| SHA-256 | 474 | 3400 | 3362 | 1.01 |+| SHA-512 | 730 | 1880 | 1883 | 1.00 |+| SHA3-256 | 550 | 1075 | 1065 | 1.01 |++Time per operation in microseconds, **lower is better**:++| | crypton 1.1.5 | crypton 2.1.5 | OpenSSL | OpenSSL / 2.1.5 |+| --- | ---: | ---: | ---: | ---: |+| X25519 | 18.27 | **12.22** | 15.53 | 1.27 |+| ECDH P-256 | 68.70 | **20.43** | 24.77 | 1.21 |+| ECDH P-384 | 3328 | **73.50** | 372.6 | 5.07 |+| Ed25519 sign | 13.58 | **7.75** | 13.23 | 1.71 |+| Ed25519 verify | 18.28 | 18.17 | 34.76 | 1.91 |+| ECDSA P-256 sign | 31.97 | **6.55** | 10.92 | 1.67 |+| ECDSA P-256 verify | 95.63 | **26.80** | 32.68 | 1.22 |+| ECDSA P-384 sign | 3219 | **124.1** | 394.2 | 3.18 |+| ECDSA P-384 verify | 3870 | **203.1** | 326.5 | 1.61 |+| RSA-2048 sign/decrypt | 447.9 | 460.1 | 319.9 | 0.70 |+| RSA-2048 verify/encrypt | 18.23 | 15.12 | 8.405 | 0.56 |++### What the numbers say++There are two changes behind the 1.1.5 column and the 2.1.5 one, not a+single steady improvement.++The first, in 2.0.0, was a rewrite: the bulk algorithms moved into C, the+curves other than P-256 moved out of Haskell `Integer` arithmetic, and+everything that touches a secret was made to take the same time whatever the+secret is. 1.1.5 had no AArch64 code of its own at all, which is why AES-GCM+there is close to a hundred times what it was, and on x86-64 it had AES-NI+and nothing else.++The second, from 2.1.0 onwards, is assembly, for the operations where C+cannot reach. Which of the two a row owes its gain to is not the same+everywhere: ECDSA P-384 signing took nineteenfold from the rewrite and a+further fifth from the assembly, while X25519 waited for the assembly+entirely and ECDH P-384 is almost all of it.++Most of that assembly is not crypton's. The prime curves, the inverse modulo+a group order, X25519, and RSA's Montgomery multiplication on x86-64 go+through [s2n-bignum](https://github.com/awslabs/s2n-bignum), vendored in+`cbits/s2n`. Every routine in it carries a machine-checked proof in+HOL-Light that it computes what it says, and is written in a constant-time+style. It is `Apache-2.0 OR ISC OR MIT-0`, and crypton takes it under ISC.++That licence is why any of this was possible. The obvious assembly to reach+for is OpenSSL's and BoringSSL's `ecp_nistz256`, and it cannot be used here:+it is Apache-2.0 only, and Intel and CloudFlare hold copyright in it besides+OpenSSL, so nobody is in a position to relicense it.++Where crypton is behind, which is now the RSA rows on both architectures and+ECDSA P-256 verification on x86-64, there is one reason. The AES-GCM rows+were the other half of this section until 2.1.5; they are ahead on both+machines now, and what the instructions do is still worth setting out.++*RSA.* 2.0.0 made signing slower than 1.1.5 on purpose: its modular+exponentiation stopped indexing a table with the bits of the exponent, and+hiding the exponent is what the difference bought. On x86-64 that cost is+more than repaid -- s2n-bignum's Montgomery multiplication is twice the C's,+because the C cannot form the two carry chains `ADCX` and `ADOX` give, and+2.1.5 signs in less than 1.1.5 took while keeping what 2.0.0 gained. On+AArch64 there is nothing to use: s2n-bignum has no generic routine for it,+and the same five that help on x86-64 measure level with the C there, so the+C stays and the gap with it. No portable C closes that gap either -- the+measurements are in+[#275](https://github.com/kazu-yamamoto/crypton/issues/275). Verification+does not move much either way: its exponent is 65537, seventeen bits, and+there is no exponentiation to speak of.++*The wide AES instructions.* `VAES` and `VPCLMULQDQ` do two blocks where+`AES-NI` and `PCLMULQDQ` do one, and four in their 512-bit form. crypton uses+the 256-bit form where the processor has it, which is Zen 3 and Ice Lake+onwards, and the 512-bit form where that is worth having, which is Ice Lake and+Zen 5 onwards. There was nothing to borrow: the wide AES-GCM in OpenSSL,+BoringSSL and AWS-LC is Apache-2.0 and s2n-bignum has no GCM, so both files are+crypton's own.++Having the 256-bit one is where the 1.49 in the x86-64 table comes from, and+it is narrower than it sounds. The EPYC 7763 is Zen 3: VAES and VPCLMULQDQ,+no AVX-512. OpenSSL's x86-64 AES-GCM is `aesni-gcm-x86_64.pl`, which is+128-bit -- its `vaesenc`s are the VEX encoding of `AESENC` on `xmm`, and+there is not one `ymm` in the file -- or `aes-gcm-avx512.pl`, which wants+`AVX512VAES`. There is no rung between them, so on this processor OpenSSL+takes a block at a time where crypton takes two. The same idea as theirs,+one step further down the feature ladder; not a better one.++The 512-bit path arrived after 2.1.2, so it is in the 2.1.5 column -- but+neither machine in the tables above has AVX-512, so neither column shows it.+On the runners that do, measured over 16 KiB in MB/s: an EPYC 9V45 (Zen 5)+goes from 9616 to 14268 with it, a Xeon 6973P-C from 8095 to 9848, a Xeon+8573C from 6983 to 8447. OpenSSL on those machines is ahead still -- 25760+on the first of them -- because it interleaves the GHASH with the AES where+crypton does them in turn. Zen 4 keeps the 256-bit path: its 512-bit+instructions are two passes through a 256-bit datapath, so the wider encoding+buys nothing there and costs a little.++AArch64 has no counterpart to any of these: one AES block and one GHASH+multiplication at a time is all the instruction set offers. Its AES-GCM+rows were 0.85 and 0.87 until 2.1.5, for that reason. What closed it was+not width but the GHASH's representation -- H is twisted once at key setup+so that GCM's bit reflection is already undone, which turns a reduction of+some twenty-five shifts and XORs into two PMULL and six EOR and makes+Karatsuba worth taking. The scheme is ARM's, from the BSD-3-Clause part of+[AArch64cryptolib](https://github.com/ARM-software/AArch64cryptolib),+written out in crypton's own intrinsics. The AES there is ahead of+OpenSSL's and always was; it was the GHASH beside it that was behind.++One row wants a word of its own: crypton's `Ed25519.sign` derives the public+key from the secret key every time it signs, so that a caller who passes a+public key that does not match cannot be made to leak the private one. That+costs a second scalar multiplication, which OpenSSL's signing does not pay --+and the row is still 1.71 on AArch64 and 1.81 on x86-64, so the safety is had+for nothing here rather than paid for.++SHA-1 is in the tables because a number of protocols and file formats still+ask for it, not because it is a good choice for anything new. The algorithms+that nothing should ask for any more -- MD5, 3DES, RC4, CBC mode -- are left+out.
@@ -5,7 +5,7 @@ module Main where -import Gauge.Main+import Test.Tasty.Bench import Crypto.Cipher.AES import qualified Crypto.Cipher.AESGCMSIV as AESGCMSIV@@ -195,7 +195,9 @@ where cp k (ini, plain) = let iniState =- throwCryptoError $ CP.initialize k (throwCryptoError $ CP.nonce12 nonce12)+ CP.initialize+ (throwCryptoError $ CP.key k)+ (throwCryptoError $ CP.nonce12 nonce12) afterAAD = CP.finalizeAAD (CP.appendAAD ini iniState) (out, afterEncrypt) = CP.encrypt plain afterAAD outtag = CP.finalize afterEncrypt@@ -386,19 +388,22 @@ , bgroup "Ed25519" benchEd25519 ] where+ -- the environment is a key pair and a signature that the benchmarked+ -- operation only reads, so building it once outside the timed region is+ -- the same measurement gauge's perBatchEnv made benchGen prx alg =- [ bench "sign" $ perBatchEnv (genEnv prx alg) (run_gen_sign prx)- , bench "verify" $ perBatchEnv (genEnv prx alg) (run_gen_verify prx)+ [ env (genEnv prx alg) $ bench "sign" . nfIO . run_gen_sign prx+ , env (genEnv prx alg) $ bench "verify" . nfIO . run_gen_verify prx ] benchGenEd25519 = benchGen (Just Curve_Edwards25519) SHA512 benchEd25519 =- [ bench "sign" $ perBatchEnv ed25519Env run_ed25519_sign- , bench "verify" $ perBatchEnv ed25519Env run_ed25519_verify+ [ env ed25519Env $ bench "sign" . nfIO . run_ed25519_sign+ , env ed25519Env $ bench "verify" . nfIO . run_ed25519_verify ] msg = B.empty -- empty message = worst-case scenario showing API overhead- genEnv prx alg _ = do+ genEnv prx alg = do sec <- EdDSA.generateSecretKey prx let pub = EdDSA.toPublic prx alg sec sig = EdDSA.sign prx sec pub msg@@ -408,7 +413,7 @@ run_gen_verify prx (_, pub, sig) = return (EdDSA.verify prx pub msg sig) - ed25519Env _ = do+ ed25519Env = do sec <- Ed25519.generateSecretKey let pub = Ed25519.toPublic sec sig = Ed25519.sign sec pub msg
@@ -2,20 +2,23 @@ module Number.F2m (benchF2m) where -import Gauge.Main import System.Random+import Test.Tasty.Bench import Crypto.Number.Basic (log2) import Crypto.Number.F2m genInteger :: Int -> Int -> Integer-genInteger salt bits =- head- . dropWhile ((< bits) . log2)- . scanl (\a r -> a * 2 ^ (31 :: Int) + abs r) 0- . randoms- . mkStdGen- $ salt + bits+genInteger salt bits = case candidates of+ x : _ -> x+ [] -> error "genInteger: the stream of candidates ran out"+ where+ candidates =+ dropWhile ((< bits) . log2)+ . scanl (\a r -> a * 2 ^ (31 :: Int) + abs r) 0+ . randoms+ . mkStdGen+ $ salt + bits benchMod :: Int -> Benchmark benchMod bits = bench (show bits) $ nf (modF2m m) a
@@ -0,0 +1,38 @@+The arrangement of the AArch64 multiply-accumulate loop in+cbits/crypton_bignum.h -- four limbs to an iteration, the low halves of the+products and the high halves accumulated in two chains -- follows+addMulVVWx in Go's crypto/internal/fips140/bigmod/nat_arm64.s, written out+in the assembler that file's compiler speaks. Go is at+https://github.com/golang/go and carries the licence below. That file's+own header reads "Copyright 2013 The Go Authors. All rights reserved. Use+of this source code is governed by a BSD-style license that can be found in+the LICENSE file", and the LICENSE file it means is this one.+++Copyright 2009 The Go Authors.++Redistribution and use in source and binary forms, with or without+modification, are permitted provided that the following conditions are+met:++ * Redistributions of source code must retain the above copyright+notice, this list of conditions and the following disclaimer.+ * Redistributions in binary form must reproduce the above+copyright notice, this list of conditions and the following disclaimer+in the documentation and/or other materials provided with the+distribution.+ * Neither the name of Google LLC nor the names of its+contributors may be used to endorse or promote products derived from+this software without specific prior written permission.++THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS+"AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT+LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR+A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT+OWNER OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL,+SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT+LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE,+DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY+THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT+(INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE+OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
@@ -0,0 +1,29 @@+Parts of cbits/aes/gcm_fused_x86.c follow the AES-GCM implementation in+picotls, lib/fusion.c, which is under the MIT license reproduced below.+The design is described by its author at++ http://blog.kazuhooku.com/2020/06/quicaes-gcm-12.html+ http://blog.kazuhooku.com/2020/06/quicaes-gcm-22.html++and the source is at https://github.com/h2o/picotls.+++Copyright (c) 2020-2022 Fastly, Kazuho Oku++Permission is hereby granted, free of charge, to any person obtaining a copy+of this software and associated documentation files (the "Software"), to+deal in the Software without restriction, including without limitation the+rights to use, copy, modify, merge, publish, distribute, sublicense, and/or+sell copies of the Software, and to permit persons to whom the Software is+furnished to do so, subject to the following conditions:++The above copyright notice and this permission notice shall be included in+all copies or substantial portions of the Software.++THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR+IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,+FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE+AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER+LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING+FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS+IN THE SOFTWARE.
@@ -0,0 +1,469 @@+/*+ * AES using the ARMv8-A Cryptographic Extensions.+ *+ * The generic code in aes/generic.c is S-box table driven, which on AArch64+ * was the only thing available: crypton_aes.c only ever swapped in the AES-NI+ * implementation, and that is gated on x86. This provides the AArch64+ * equivalent.+ *+ * The key schedule is laid out exactly as x86ni.c lays it out, because+ * crypton_aes.c leaves some operations -- OCB and CCM -- pointing at the+ * generic implementation even once the accelerated table is installed, and+ * those read the forward schedule. So: the forward round keys k[0..nbr]+ * first, in the order crypton_aes_generic_init writes them, then+ * InvMixColumns(k[nbr-1]) down to InvMixColumns(k[1]) for decryption. The two+ * ends of the decryption schedule, k[nbr] and k[0], are read back out of the+ * forward half rather than stored twice, which is what makes AES-256 fit in+ * the 16*14*2 bytes of aes_key.data.+ */++#include <stdint.h>+#include <string.h>+#include <arm_neon.h>+#if defined(__linux__)+#include <sys/auxv.h>+#include <asm/hwcap.h>+#endif+#include "crypton_aes.h"+#include "crypton_bitfn.h"++/*+ * The AES and PMULL instructions are extensions, so a translation unit+ * compiled for baseline ARMv8-A may not use them. Mark the functions that do,+ * the way cbits/aes/x86ni.h marks their x86 counterparts, rather than raising+ * -march for every file in the library: the flag use_target_attributes picks+ * between the two, and with it set -- which is the default -- nothing else+ * enables the extensions, so without these the file does not compile at all on+ * a toolchain whose baseline lacks them. Apple's does not lack them, which is+ * why only Linux noticed.+ *+ * "+crypto" rather than "crypto": GCC rejects the latter.+ */+#include "crypton_armv8_target.h"++/* forward round keys: nbr + 1 of them, written by the generic key expansion */+#define FWD(key) ((const uint8_t *) (key)->data)+/* InvMixColumns(k[nbr-1]) .. InvMixColumns(k[1]): nbr - 1 of them */+#define INV(key) (((const uint8_t *) (key)->data) + 16 * ((key)->nbr + 1))++/*+ * The key schedule of FIPS 197 5.2, with the S-box the schedule needs coming+ * from the instructions rather than a table in memory.+ *+ * AArch64 has no counterpart to x86's AESKEYGENASSIST, but AESE is+ * AddRoundKey, SubBytes and ShiftRows together, so against a zero key it is+ * SubBytes and ShiftRows. Give it a word in all four columns and ShiftRows+ * only moves identical bytes between them, which leaves every column holding+ * SubWord of that word. RotWord is then a byte rotation, and on a register+ * whose four words are equal a rotation of the whole register by one byte+ * rotates each word.+ *+ * The words stay in vector registers throughout: a word moved to a general+ * register and back costs more than the instruction it is moved for.+ *+ * The exposure this removes is a small one -- sixteen lookups at addresses+ * derived from the key, once per key, against the per-block indexing the+ * instructions exist to remove -- but a key schedule is the one thing an+ * attacker most wants and it costs little to keep it out of the cache.+ */+CRYPTON_TARGET_ARMV8_CRYPTO+static uint32x4_t sub_word(uint32x4_t w)+{+ return vreinterpretq_u32_u8(+ vaeseq_u8(vreinterpretq_u8_u32(w), vdupq_n_u8(0)));+}++CRYPTON_TARGET_ARMV8_CRYPTO+static uint32x4_t sub_rot_word(uint32x4_t w)+{+ const uint8x16_t s = vreinterpretq_u8_u32(sub_word(w));++ return vreinterpretq_u32_u8(vextq_u8(s, s, 1));+}++CRYPTON_TARGET_ARMV8_CRYPTO+void crypton_aes_armv8_init(aes_key *key, uint8_t *origkey, uint8_t size)+{+ /* 2^0 .. 2^9 in GF(2^8), which is as far as any key size reaches */+ static const uint32_t rcon[10] = {+ 0x01, 0x02, 0x04, 0x08, 0x10, 0x20, 0x40, 0x80, 0x1b, 0x36,+ };+ uint32_t *w = (uint32_t *) key->data;+ uint8_t *inv;+ int nk, nw, i;++ switch (size) {+ case 16: key->nbr = 10; break;+ case 24: key->nbr = 12; break;+ case 32: key->nbr = 14; break;+ default: return;+ }+ nk = size / 4; /* words of key */+ nw = 4 * (key->nbr + 1); /* words of schedule */++ memcpy(w, origkey, size);+ for (i = nk; i < nw; i++) {+ uint32x4_t t = vld1q_dup_u32(w + i - 1);++ if (i % nk == 0)+ t = veorq_u32(sub_rot_word(t),+ vdupq_n_u32(rcon[i / nk - 1]));+ else if (nk > 6 && i % nk == 4)+ t = sub_word(t);+ vst1q_lane_u32(w + i, veorq_u32(t, vld1q_dup_u32(w + i - nk)), 0);+ }++ /* and the inverted round keys the decryption modes read */+ inv = ((uint8_t *) key->data) + 16 * (key->nbr + 1);+ for (i = 1; i < key->nbr; i++) {+ uint8x16_t rk =+ vld1q_u8(((const uint8_t *) key->data) + 16 * (key->nbr - i));+ vst1q_u8(inv + 16 * (i - 1), vaesimcq_u8(rk));+ }+}++/*+ * Whether the extensions are actually present.+ *+ * They are mandatory on Apple silicon, and on other AArch64 systems the+ * kernel reports them through the auxiliary vector. A system without them+ * keeps the generic implementation.+ */+int crypton_aes_armv8_available(void)+{+#if defined(__APPLE__)+ return 1;+#elif defined(__linux__)+ return (getauxval(AT_HWCAP) & HWCAP_AES) != 0;+#else+ return 0;+#endif+}++/*+ * GHASH using PMULL, the AArch64 counterpart to PCLMULQDQ.+ *+ * Not the transliteration of gfmul_pclmuldq in x86ni.c this used to be. The+ * x86 formulation keeps H in the order GCM writes it and pays, at the end of+ * every batch, a reduction that first has to undo GCM's bit reflection: some+ * twenty-five shifts and XORs, and a batch of eight costs it once.+ *+ * Instead H is twisted once, at key setup, so that the reflection is already+ * undone and a reversed-polynomial multiply lands in the right place. Two+ * things follow. The reduction becomes two PMULL against 0xC2000..0 and six+ * EOR, a third of what it was. And because nothing has to be byte-reversed+ * back and forth, Karatsuba pays: three PMULL a block rather than four, with+ * the middle terms accumulated in a third register and tidied up once per+ * batch.+ *+ * crypton tried Karatsuba in the old representation and measured it 1.6 per+ * cent slower -- the saved PMULL did not cover the extra EOR when the+ * reduction stayed as expensive as it was. It is the pair that pays.+ * Measured over 16 KiB messages, this against the old code:+ *+ * Apple M4 Neoverse N2+ * AES-128-GCM 1.30 1.25+ * AES-192-GCM 1.22 1.25+ * AES-256-GCM 1.19 1.24+ *+ * The scheme is ARM's, from the 'big' AES-GCM kernel of+ * https://github.com/ARM-software/AArch64cryptolib, which is BSD-3-Clause,+ * (c) 2018-2019 ARM Limited. Their kernels under AArch64cryptolib_opt_bigger+ * are faster again and are NOT under that licence, whatever the repository's+ * LICENSE.md says; nothing here comes from those files.+ *+ * The table holds the twisted powers H^1 .. H^8 at htable[0 .. 7] and the+ * Karatsuba half of each -- its high 64 bits XOR its low -- in the first+ * eight bytes of htable[8 .. 15]. The running tag is kept the way GCM+ * writes it at every boundary, and swapped into the internal form on the way+ * in and out, which is two instructions.+ */++#define GHASH_MODC ((poly64_t) 0xC200000000000000ul)++/* the internal accumulator form, and back again: its own inverse */+CRYPTON_TARGET_ARMV8_CRYPTO+static inline uint8x16_t ghash_swap(uint8x16_t t)+{+ t = vrev64q_u8(t);+ return vextq_u8(t, t, 8);+}++/* the high and low halves XORed together, which is what Karatsuba wants */+CRYPTON_TARGET_ARMV8_CRYPTO+static inline poly64_t ghash_karat(poly64x2_t v)+{+ return (poly64_t) veor_u64(vget_high_u64(vreinterpretq_u64_p64(v)),+ vget_low_u64(vreinterpretq_u64_p64(v)));+}++/* the twisted power H^(i+1) */+#define GHASH_POW(ht, i) \+ vreinterpretq_p64_u8(vld1q_u8((const uint8_t *) &(ht)[i]))+/* and its Karatsuba half */+#define GHASH_KARAT(ht, i) \+ ((poly64_t) vgetq_lane_u64( \+ vreinterpretq_u64_u8(vld1q_u8((const uint8_t *) &(ht)[8 + (i)])), 0))++/*+ * One block's three partial products, XORed into the accumulators. b is+ * already in the internal form; hp and hk are the power it is to meet.+ */+#define GHASH_MUL(b, hp, hk, H, M, L) \+ do { \+ poly64x2_t b__ = (b); \+ poly64x2_t hp__ = (hp); \+ (H) = veorq_u64((H), vreinterpretq_u64_p128( \+ vmull_high_p64(b__, hp__))); \+ (L) = veorq_u64((L), vreinterpretq_u64_p128(vmull_p64( \+ (poly64_t) vgetq_lane_u64(vreinterpretq_u64_p64(b__), 0), \+ (poly64_t) vgetq_lane_u64(vreinterpretq_u64_p64(hp__), 0)))); \+ (M) = veorq_u64((M), vreinterpretq_u64_p128( \+ vmull_p64(ghash_karat(b__), (hk)))); \+ } while (0)++/*+ * Finish the Karatsuba -- the middle accumulator still holds only the+ * (ah^al)(bh^bl) terms and wants the other two taken out of it -- and reduce+ * the 256 bits modulo the GCM polynomial. The result is in internal form.+ */+CRYPTON_TARGET_ARMV8_CRYPTO+static inline uint64x2_t ghash_reduce(uint64x2_t H, uint64x2_t M, uint64x2_t L)+{+ uint64x2_t t;++ M = veorq_u64(M, H);+ M = veorq_u64(M, L);++ t = vreinterpretq_u64_p128(vmull_p64(+ (poly64_t) vgetq_lane_u64(H, 0), GHASH_MODC));+ H = vreinterpretq_u64_u8(vextq_u8(vreinterpretq_u8_u64(H),+ vreinterpretq_u8_u64(H), 8));+ M = veorq_u64(M, t);+ M = veorq_u64(M, H);++ t = vreinterpretq_u64_p128(vmull_p64(+ (poly64_t) vgetq_lane_u64(M, 0), GHASH_MODC));+ M = vreinterpretq_u64_u8(vextq_u8(vreinterpretq_u8_u64(M),+ vreinterpretq_u8_u64(M), 8));+ L = veorq_u64(L, t);+ return veorq_u64(L, M);+}++/* a single block against H^1, accumulator in internal form */+CRYPTON_TARGET_ARMV8_CRYPTO+static inline uint64x2_t ghash_one(uint64x2_t acc, uint8x16_t blk,+ const block128 *ht)+{+ uint64x2_t H = vdupq_n_u64(0), M = H, L = H;+ poly64x2_t b;++ acc = vreinterpretq_u64_u8(vextq_u8(vreinterpretq_u8_u64(acc),+ vreinterpretq_u8_u64(acc), 8));+ b = vreinterpretq_p64_u64(veorq_u64(+ vreinterpretq_u64_u8(vrev64q_u8(blk)), acc));+ GHASH_MUL(b, GHASH_POW(ht, 0), GHASH_KARAT(ht, 0), H, M, L);+ return ghash_reduce(H, M, L);+}++/*+ * Twist H and raise it to the powers a batch needs.+ *+ * The twist is a shift left by one with 0xC2000..01 folded back in when a+ * bit falls off the top -- the same correction the old reduction applied to+ * every product, done once here instead. Each further power is one multiply+ * in the twisted domain; the result comes out of ghash_reduce with its+ * halves swapped, which a batch undoes on the way in, so a stored power has+ * to be swapped back.+ */+CRYPTON_TARGET_ARMV8_CRYPTO+void crypton_aes_armv8_hinit_pmull(block128 *htable, const block128 *h)+{+ uint8x16_t hk = vrev64q_u8(vld1q_u8((const uint8_t *) h));+ uint64x2_t shl = vshlq_n_u64(vreinterpretq_u64_u8(hk), 1);+ uint64x2_t shr = vreinterpretq_u64_s64(+ vshrq_n_s64(vreinterpretq_s64_u8(hk), 63));+ uint8x16_t mask = vextq_u8(vreinterpretq_u8_u64(shr),+ vreinterpretq_u8_u64(shr), 12);+ uint64x2_t tc = vdupq_n_u64(0);+ poly64x2_t base, p;+ int i;++ tc = vsetq_lane_u64(0xC200000000000001ul, tc, 0);+ tc = vsetq_lane_u64(1, tc, 1);+ tc = vandq_u64(vreinterpretq_u64_u8(mask), tc);+ base = vreinterpretq_p64_u64(veorq_u64(tc, shl));++ p = base;+ for (i = 0; i < 8; i++) {+ uint64x2_t H = vdupq_n_u64(0), M = H, L = H, r;++ vst1q_u8((uint8_t *) &htable[i],+ vreinterpretq_u8_p64(p));+ vst1q_u8((uint8_t *) &htable[8 + i],+ vreinterpretq_u8_u64(+ vdupq_n_u64((uint64_t) ghash_karat(p))));++ GHASH_MUL(p, base, ghash_karat(base), H, M, L);+ r = ghash_reduce(H, M, L);+ p = vreinterpretq_p64_u8(vextq_u8(vreinterpretq_u8_u64(r),+ vreinterpretq_u8_u64(r), 8));+ }+}++CRYPTON_TARGET_ARMV8_CRYPTO+void crypton_aes_armv8_gf_mul_pmull(block128 *a, const block128 *htable)+{+ uint64x2_t acc = vreinterpretq_u64_u8(+ ghash_swap(vld1q_u8((const uint8_t *) a)));++ acc = ghash_one(acc, vdupq_n_u8(0), htable);+ vst1q_u8((uint8_t *) a, ghash_swap(vreinterpretq_u8_u64(acc)));+}++/*+ * Four GHASH steps -- ((((a^b0)H ^ b1)H ^ b2)H ^ b3)H -- with a single+ * reduction. Expanded that is (a^b0)H^4 ^ b1*H^3 ^ b2*H^2 ^ b3*H, so the+ * four products can be summed first and reduced once, which is where the+ * time goes. Aggregated reduction, from the Intel GCM paper.+ */+CRYPTON_TARGET_ARMV8_CRYPTO+void crypton_aes_armv8_gf_mul4_pmull(block128 *a, const block128 *blocks,+ const block128 *htable)+{+ uint64x2_t acc = vreinterpretq_u64_u8(+ ghash_swap(vld1q_u8((const uint8_t *) a)));+ uint64x2_t H = vdupq_n_u64(0), M = H, L = H;+ poly64x2_t b;+ int i;++ acc = vreinterpretq_u64_u8(vextq_u8(vreinterpretq_u8_u64(acc),+ vreinterpretq_u8_u64(acc), 8));+ b = vreinterpretq_p64_u64(veorq_u64(+ vreinterpretq_u64_u8(vrev64q_u8(+ vld1q_u8((const uint8_t *) &blocks[0]))), acc));+ GHASH_MUL(b, GHASH_POW(htable, 3), GHASH_KARAT(htable, 3), H, M, L);++ for (i = 1; i < 4; i++) {+ b = vreinterpretq_p64_u8(vrev64q_u8(+ vld1q_u8((const uint8_t *) &blocks[i])));+ GHASH_MUL(b, GHASH_POW(htable, 3 - i),+ GHASH_KARAT(htable, 3 - i), H, M, L);+ }++ acc = ghash_reduce(H, M, L);+ vst1q_u8((uint8_t *) a, ghash_swap(vreinterpretq_u8_u64(acc)));+}+++int crypton_aes_armv8_pmull_available(void)+{+#if defined(__APPLE__)+ return 1;+#elif defined(__linux__)+ return (getauxval(AT_HWCAP) & HWCAP_PMULL) != 0;+#else+ return 0;+#endif+}++/*+ * The XTS tweak advances by doubling in GF(2^128), which+ * crypton_aes_generic_gf_mulx does through memory. Here it stays in a+ * register: shift both halves left by one, carry the low half's top bit into+ * the high half, and fold the bit that leaves the top back in as 0x87. The+ * block is little-endian, so lane 0 is the low half.+ */+CRYPTON_TARGET_ARMV8_CRYPTO+static inline uint8x16_t gfmulx_neon(uint8x16_t v)+{+ const uint64x2_t x = vreinterpretq_u64_u8(v);+ const uint64x2_t zero = vdupq_n_u64(0);+ const uint64x2_t carry = vshrq_n_u64(x, 63);+ /* the low half's carry becomes the high half's bit 0 */+ const uint64x2_t into_hi = vextq_u64(zero, carry, 1);+ /* and the high half's becomes all ones, or nothing, in the low half */+ const uint64x2_t out = vsubq_u64(zero, vextq_u64(carry, zero, 1));+ const uint64x2_t poly = vsetq_lane_u64(0x87, zero, 0);++ return vreinterpretq_u8_u64(veorq_u64(+ vorrq_u64(vshlq_n_u64(x, 1), into_hi), vandq_u64(out, poly)));+}++/*+ * The modes, generated once per key size. See armv8_impl.c for why the+ * round count has to be a compile-time constant.+ */+#define SIZED(m) m##128+#define NBR 10+#include <aes/armv8_impl.c>+#undef SIZED+#undef NBR++#define SIZED(m) m##192+#define NBR 12+#include <aes/armv8_impl.c>+#undef SIZED+#undef NBR++#define SIZED(m) m##256+#define NBR 14+#include <aes/armv8_impl.c>+#undef SIZED+#undef NBR++/*+ * The fused entry point, over the three key sizes. Each was generated with+ * its round count fixed, which is what lets the eight chains stay in+ * registers; the choice between them is made once per message here.+ */+CRYPTON_TARGET_ARMV8_CRYPTO+void crypton_aes_armv8_gcm_fused(uint8_t *out, const block128 *ht,+ aes_key *key, const uint8_t *nonce,+ const uint8_t *aad, uint32_t aadlen,+ const uint8_t *in, uint32_t inlen,+ uint32_t taglen, aes_key *hpkey,+ uint32_t sampleoff, uint8_t *mask)+{+ switch (key->strength) {+ case 0:+ crypton_aes_armv8_gcm_fused128(out, ht, key, nonce, aad, aadlen,+ in, inlen, taglen, hpkey,+ sampleoff, mask);+ break;+ case 1:+ crypton_aes_armv8_gcm_fused192(out, ht, key, nonce, aad, aadlen,+ in, inlen, taglen, hpkey,+ sampleoff, mask);+ break;+ default:+ crypton_aes_armv8_gcm_fused256(out, ht, key, nonce, aad, aadlen,+ in, inlen, taglen, hpkey,+ sampleoff, mask);+ break;+ }+}++CRYPTON_TARGET_ARMV8_CRYPTO+int crypton_aes_armv8_gcm_fused_dec(uint8_t *out, const block128 *ht,+ aes_key *key, const uint8_t *nonce,+ const uint8_t *aad, uint32_t aadlen,+ const uint8_t *in, uint32_t inlen,+ const uint8_t *tag, uint32_t taglen,+ uint8_t *outtag)+{+ switch (key->strength) {+ case 0:+ return crypton_aes_armv8_gcm_fused_dec128(out, ht, key, nonce,+ aad, aadlen, in, inlen,+ tag, taglen, outtag);+ case 1:+ return crypton_aes_armv8_gcm_fused_dec192(out, ht, key, nonce,+ aad, aadlen, in, inlen,+ tag, taglen, outtag);+ default:+ return crypton_aes_armv8_gcm_fused_dec256(out, ht, key, nonce,+ aad, aadlen, in, inlen,+ tag, taglen, outtag);+ }+}
@@ -0,0 +1,824 @@+/*+ * Included from armv8.c once per key size, with NBR set to the number of+ * rounds and SIZED() naming the functions. This mirrors x86ni_impl.c.+ *+ * Two things here want compile-time constants, and both are worth having.+ * With the round count fixed the compiler keeps the round keys scheduled+ * instead of reloading them against a count read out of the key. With the+ * blocks in flight fixed it interleaves that many independent chains, which+ * is what covers the latency of AESE and AESMC -- one block at a time leaves+ * the pipeline waiting on itself. On Apple silicon the two together are+ * worth about four times a loop that does one block with a round count from+ * memory.+ *+ * The blocks are named by constant index throughout, and every step is+ * written out one per block rather than left to a loop over s[i]. Such a+ * loop is only as good as the compiler's willingness to unroll it, and GCC+ * at -O2 declines: s[] then lives on the stack and each round turns into a+ * load and a store, which measured slower than the one-block code this+ * replaces. Spelling the steps out costs nothing and leaves nothing to+ * decide.+ */++/* Eight chains is where the return flattens out on the cores measured. */+#define WAY 8++#define EACH1(m) m(0)+#define EACH8(m) m(0) m(1) m(2) m(3) m(4) m(5) m(6) m(7)++#define LOAD_IN(i) s[i] = vld1q_u8((const uint8_t *) (input + (i)));+#define STORE_OUT(i) vst1q_u8((uint8_t *) (output + (i)), s[i]);++#define ENC_STEP(i) s[i] = vaesmcq_u8(vaeseq_u8(s[i], k_));+#define ENC_LAST(i) s[i] = veorq_u8(vaeseq_u8(s[i], k_), l_);+#define DEC_STEP(i) s[i] = vaesimcq_u8(vaesdq_u8(s[i], k_));+#define DEC_LAST(i) s[i] = veorq_u8(vaesdq_u8(s[i], k_), l_);++/* Encrypt the blocks EACH names, in place in s[]. rk must be in scope. */+#define ENC_ROUNDS(EACH) \+ do { \+ int r_; \+ for (r_ = 0; r_ < NBR - 1; r_++) { \+ const uint8x16_t k_ = vld1q_u8(rk + 16 * r_); \+ EACH(ENC_STEP) \+ } \+ { \+ const uint8x16_t k_ = vld1q_u8(rk + 16 * (NBR - 1)); \+ const uint8x16_t l_ = vld1q_u8(rk + 16 * NBR); \+ EACH(ENC_LAST) \+ } \+ } while (0)++/*+ * Decrypt them. fwd and inv must be in scope: the schedule is k[nbr],+ * imc(k[nbr-1]) .. imc(k[1]), k[0], so the two ends come from the forward+ * keys and the middle from the inverted ones.+ */+#define DEC_ROUNDS(EACH) \+ do { \+ int r_; \+ { \+ const uint8x16_t k_ = vld1q_u8(fwd + 16 * NBR); \+ EACH(DEC_STEP) \+ } \+ for (r_ = 0; r_ < NBR - 2; r_++) { \+ const uint8x16_t k_ = vld1q_u8(inv + 16 * r_); \+ EACH(DEC_STEP) \+ } \+ { \+ const uint8x16_t k_ = vld1q_u8(inv + 16 * (NBR - 2));\+ const uint8x16_t l_ = vld1q_u8(fwd); \+ EACH(DEC_LAST) \+ } \+ } while (0)++CRYPTON_TARGET_ARMV8_CRYPTO+void SIZED(crypton_aes_armv8_encrypt_block)(aes_block *output, aes_key *key, aes_block *input)+{+ const uint8_t *rk = FWD(key);+ uint8x16_t s[1];++ EACH1(LOAD_IN);+ ENC_ROUNDS(EACH1);+ EACH1(STORE_OUT);+}++CRYPTON_TARGET_ARMV8_CRYPTO+void SIZED(crypton_aes_armv8_decrypt_block)(aes_block *output, aes_key *key, aes_block *input)+{+ const uint8_t *fwd = FWD(key);+ const uint8_t *inv = INV(key);+ uint8x16_t s[1];++ EACH1(LOAD_IN);+ DEC_ROUNDS(EACH1);+ EACH1(STORE_OUT);+}++CRYPTON_TARGET_ARMV8_CRYPTO+void SIZED(crypton_aes_armv8_encrypt_ecb)(aes_block *output, aes_key *key, aes_block *input, uint32_t nb_blocks)+{+ const uint8_t *rk = FWD(key);+ uint8x16_t s[WAY];++ for (; nb_blocks >= WAY; nb_blocks -= WAY, input += WAY, output += WAY) {+ EACH8(LOAD_IN);+ ENC_ROUNDS(EACH8);+ EACH8(STORE_OUT);+ }+ for (; nb_blocks > 0; nb_blocks--, input++, output++) {+ EACH1(LOAD_IN);+ ENC_ROUNDS(EACH1);+ EACH1(STORE_OUT);+ }+}++CRYPTON_TARGET_ARMV8_CRYPTO+void SIZED(crypton_aes_armv8_decrypt_ecb)(aes_block *output, aes_key *key, aes_block *input, uint32_t nb_blocks)+{+ const uint8_t *fwd = FWD(key);+ const uint8_t *inv = INV(key);+ uint8x16_t s[WAY];++ for (; nb_blocks >= WAY; nb_blocks -= WAY, input += WAY, output += WAY) {+ EACH8(LOAD_IN);+ DEC_ROUNDS(EACH8);+ EACH8(STORE_OUT);+ }+ for (; nb_blocks > 0; nb_blocks--, input++, output++) {+ EACH1(LOAD_IN);+ DEC_ROUNDS(EACH1);+ EACH1(STORE_OUT);+ }+}++/* CBC encryption chains, so there is nothing to interleave. It still gains+ * the round keys staying put. */+CRYPTON_TARGET_ARMV8_CRYPTO+void SIZED(crypton_aes_armv8_encrypt_cbc)(aes_block *output, aes_key *key, aes_block *_iv, aes_block *input, uint32_t nb_blocks)+{+ const uint8_t *rk = FWD(key);+ uint8x16_t iv = vld1q_u8((const uint8_t *) _iv);+ uint8x16_t s[1];++ for (; nb_blocks-- > 0; input++, output++) {+ s[0] = veorq_u8(iv, vld1q_u8((const uint8_t *) input));+ ENC_ROUNDS(EACH1);+ iv = s[0];+ EACH1(STORE_OUT);+ }+}++/* Decryption does not chain: each block is deciphered on its own and then+ * XORed with the ciphertext before it, so it interleaves like ECB. */+/* c[] holds the previous block at index 0 and this group's ciphertext after+ * it, so block i is XORed with c[i] and the next group starts from c[WAY]. */+#define CBC_KEEP(i) c[(i) + 1] = s[i];+#define CBC_XOR(i) vst1q_u8((uint8_t *) (output + (i)), veorq_u8(s[i], c[i]));++CRYPTON_TARGET_ARMV8_CRYPTO+void SIZED(crypton_aes_armv8_decrypt_cbc)(aes_block *output, aes_key *key, aes_block *_iv, aes_block *input, uint32_t nb_blocks)+{+ const uint8_t *fwd = FWD(key);+ const uint8_t *inv = INV(key);+ uint8x16_t iv = vld1q_u8((const uint8_t *) _iv);+ uint8x16_t s[WAY], c[WAY + 1];++ for (; nb_blocks >= WAY; nb_blocks -= WAY, input += WAY, output += WAY) {+ EACH8(LOAD_IN);+ c[0] = iv;+ EACH8(CBC_KEEP);+ DEC_ROUNDS(EACH8);+ EACH8(CBC_XOR);+ iv = c[WAY];+ }+ for (; nb_blocks > 0; nb_blocks--, input++, output++) {+ EACH1(LOAD_IN);+ c[1] = s[0];+ DEC_ROUNDS(EACH1);+ vst1q_u8((uint8_t *) output, veorq_u8(s[0], iv));+ iv = c[1];+ }+}++/*+ * CTR counts the whole 128 bits big-endian, with the carry crossing the+ * halves. The arithmetic is kept identical to+ * crypton_aes_generic_encrypt_ctr, which also leaves the caller's IV alone.+ */+#define CTR_SET(i) s[i] = vreinterpretq_u8_u64(vsetq_lane_u64(cpu_to_be64(lo + (i)), base, 1));+#define CTR_XOR(i) vst1q_u8(output + 16 * (i), \+ veorq_u8(s[i], vld1q_u8(input + 16 * (i))));++CRYPTON_TARGET_ARMV8_CRYPTO+void SIZED(crypton_aes_armv8_encrypt_ctr)(uint8_t *output, aes_key *key, aes_block *iv, uint8_t *input, uint32_t len)+{+ const uint8_t *rk = FWD(key);+ uint32_t nb_blocks = len / 16;+ uint32_t remaining = len % 16;+ aes_block ctr;+ uint8x16_t s[WAY];+ uint32_t i;++ block128_copy(&ctr, iv);++ /*+ * The counter goes through memory only when its low half is about to+ * wrap. Otherwise it stays in registers: the top eight bytes do not+ * change and the bottom eight are one add away. That matters -- with+ * a store and a reload for every block, CTR ran at the same speed for+ * 128-bit and 256-bit keys, which is the giveaway that the cipher was+ * not what it was waiting for.+ */+ for (; nb_blocks >= WAY; nb_blocks -= WAY, input += 16 * WAY, output += 16 * WAY) {+ uint64_t lo = be64_to_cpu(ctr.q[1]);++ if (lo + (WAY - 1) < lo) {+ /* a block in this group carries into the top half;+ * let the scalar increment deal with it */+ for (i = 0; i < WAY; i++, block128_inc_be(&ctr))+ s[i] = vld1q_u8((const uint8_t *) &ctr);+ } else {+ const uint64x2_t base =+ vreinterpretq_u64_u8(vld1q_u8((const uint8_t *) &ctr));++ EACH8(CTR_SET);++ /* no block above needed a carry, but the counter left+ * for the next group still can */+ ctr.q[1] = cpu_to_be64(lo + WAY);+ if (lo + WAY < lo)+ ctr.q[0] = cpu_to_be64(be64_to_cpu(ctr.q[0]) + 1);+ }+ ENC_ROUNDS(EACH8);+ EACH8(CTR_XOR);+ }+ for (; nb_blocks > 0; nb_blocks--, input += 16, output += 16) {+ s[0] = vld1q_u8((const uint8_t *) &ctr);+ block128_inc_be(&ctr);+ ENC_ROUNDS(EACH1);+ vst1q_u8(output, veorq_u8(s[0], vld1q_u8(input)));+ }+ if (remaining) {+ aes_block o;++ s[0] = vld1q_u8((const uint8_t *) &ctr);+ ENC_ROUNDS(EACH1);+ vst1q_u8((uint8_t *) &o, s[0]);+ for (i = 0; i < remaining; i++)+ output[i] = o.b[i] ^ input[i];+ }+}+++/*+ * GCM, rather than the generic loop calling the block function once per+ * block through the branch table. Eight counter blocks go through the+ * rounds together, and their GHASH folds into a single reduction with+ * H^8 .. H^1, so a group costs one reduction instead of eight. The tag+ * and the counter stay in registers across the whole run.+ *+ * GCM's counter is the low 32 bits only and wraps there, so unlike CTR+ * there is no carry to chase: the top twelve bytes never move.+ *+ * What this loop is short of is not overlap but instructions. A group of+ * eight blocks compiles to 383 of them on an Apple M4, and only 152 are the+ * cipher -- eighty AESE and seventy-two AESMC. The rest is the GHASH and+ * the counters. Measured against OpenSSL on the same machine, 16 KiB+ * messages under AES-128:+ *+ * crypton, this loop with the GHASH taken out 19883 MB/s+ * OpenSSL, AES-128-CTR 17176+ * OpenSSL, AES-128-GCM 9869+ * crypton, AES-128-GCM 8641+ *+ * The cipher here is ahead of OpenSSL's; all of the 0.87 is the GHASH.+ * Four ways of closing it were tried and every one measured worse, so they+ * are written down here rather than tried again. Each was checked to give+ * the same ciphertext and tag as this code for three key sizes and+ * thirty-five lengths, encrypt and decrypt, before being timed:+ *+ * the GHASH held back a group and spread through the next -1.3%+ * group's AES rounds, so that the two do not queue to -3.6%+ * Karatsuba: three multiplications for the two halves -1.6%+ * instead of four+ * the same with every product on a lane the instruction 0.0%+ * reaches, which does remove sixteen fmov a group+ * the same again with the H powers' halves added together -3%+ * already, in the spare half of htable+ *+ * The first fails because there was nothing to gain: a group's AES depends+ * on nothing in the group before it, so a wide out-of-order core already+ * runs the two together, and holding a group back only adds copies. The+ * rest fail for one reason -- none of them makes the loop shorter.+ * Karatsuba buys a PMULL for two EOR and an EXT, and the folded table turns+ * sixteen `dup` into sixteen `ld1r` and sixteen more address adds.+ *+ * That last sentence used to end by saying a count which does come down+ * wants the data laid out differently. It does, and since the GHASH was+ * rewritten against a twisted H -- see cbits/aes/armv8.c -- it is laid out+ * differently, so the figures above are what the *previous* GHASH gave.+ * Karatsuba pays now that the reduction it has to carry is a third of what+ * it was; the other three are untried in the new representation and the+ * first of them has no more reason to work than it had.+ */+#define GCM_CTR(i) s[i] = vreinterpretq_u8_u32(vsetq_lane_u32(cpu_to_be32(c + 1 + (i)), base, 3));+#define GCM_ENC(i) { const uint8x16_t m_ = vld1q_u8(input + 16 * (i)); \+ s[i] = veorq_u8(s[i], m_); \+ vst1q_u8(output + 16 * (i), s[i]); }+#define GCM_DEC(i) { const uint8x16_t m_ = vld1q_u8(input + 16 * (i)); \+ vst1q_u8(output + 16 * (i), veorq_u8(s[i], m_)); \+ s[i] = m_; }+#define GCM_GHASH(i) \+ { \+ poly64x2_t b_ = vreinterpretq_p64_u8(vrev64q_u8(s[i])); \+ if ((i) == 0) \+ b_ = vreinterpretq_p64_u64(veorq_u64( \+ vreinterpretq_u64_p64(b_), acc)); \+ GHASH_MUL(b_, GHASH_POW(ht, WAY - 1 - (i)), \+ GHASH_KARAT(ht, WAY - 1 - (i)), gH, gM, gL); \+ }++/* the eight blocks now in s[] are the ciphertext; fold them into the tag */+#define GCM_FOLD() \+ do { \+ uint64x2_t gH = vdupq_n_u64(0), gM = gH, gL = gH; \+ acc = vreinterpretq_u64_u8( \+ vextq_u8(vreinterpretq_u8_u64(acc), \+ vreinterpretq_u8_u64(acc), 8)); \+ EACH8(GCM_GHASH) \+ acc = ghash_reduce(gH, gM, gL); \+ } while (0)++#define GCM_PROLOGUE \+ const uint8_t *rk = FWD(key); \+ const block128 *ht = gcm->htable; \+ uint8x16_t s[WAY]; \+ uint64x2_t acc = vreinterpretq_u64_u8( \+ ghash_swap(vld1q_u8((const uint8_t *) &gcm->tag))); \+ uint32_t c = be32_to_cpu(gcm->civ.d[3]); \+ uint32x4_t base = vreinterpretq_u32_u8(vld1q_u8((const uint8_t *) &gcm->civ))++/* one block, for what is left after the last group of eight */+#define GCM_ONE(load_m, store_c, ghash_of) \+ do { \+ const uint8x16_t m_ = (load_m); \+ c++; \+ s[0] = vreinterpretq_u8_u32(vsetq_lane_u32(cpu_to_be32(c), base, 3)); \+ ENC_ROUNDS(EACH1); \+ s[0] = veorq_u8(s[0], m_); \+ (store_c); \+ acc = ghash_one(acc, (ghash_of), ht); \+ } while (0)++#define GCM_EPILOGUE \+ do { \+ gcm->civ.d[3] = cpu_to_be32(c); \+ vst1q_u8((uint8_t *) &gcm->tag, \+ ghash_swap(vreinterpretq_u8_u64(acc))); \+ } while (0)++CRYPTON_TARGET_ARMV8_CRYPTO+void SIZED(crypton_aes_armv8_gcm_encrypt)(uint8_t *output, aes_gcm *gcm, aes_key *key, uint8_t *input, uint32_t length)+{+ GCM_PROLOGUE;+ uint32_t i;++ gcm->length_input += length;++ for (; length >= 16 * WAY; input += 16 * WAY, output += 16 * WAY, length -= 16 * WAY) {+ EACH8(GCM_CTR);+ c += WAY;+ ENC_ROUNDS(EACH8);+ EACH8(GCM_ENC);+ GCM_FOLD();+ }+ for (; length >= 16; input += 16, output += 16, length -= 16) {+ GCM_ONE(vld1q_u8(input), vst1q_u8(output, s[0]), s[0]);+ }+ if (length) {+ aes_block m, o;++ block128_zero(&m);+ block128_copy_bytes(&m, input, length);+ c++;+ s[0] = vreinterpretq_u8_u32(vsetq_lane_u32(cpu_to_be32(c), base, 3));+ ENC_ROUNDS(EACH1);+ s[0] = veorq_u8(s[0], vld1q_u8((const uint8_t *) &m));+ vst1q_u8((uint8_t *) &o, s[0]);+ block128_zero(&m);+ for (i = 0; i < length; i++)+ output[i] = m.b[i] = o.b[i];+ acc = ghash_one(acc, vld1q_u8((const uint8_t *) &m), ht);+ }+ GCM_EPILOGUE;+}++CRYPTON_TARGET_ARMV8_CRYPTO+void SIZED(crypton_aes_armv8_gcm_decrypt)(uint8_t *output, aes_gcm *gcm, aes_key *key, uint8_t *input, uint32_t length)+{+ GCM_PROLOGUE;+ uint32_t i;++ gcm->length_input += length;++ for (; length >= 16 * WAY; input += 16 * WAY, output += 16 * WAY, length -= 16 * WAY) {+ EACH8(GCM_CTR);+ c += WAY;+ ENC_ROUNDS(EACH8);+ EACH8(GCM_DEC);+ GCM_FOLD();+ }+ for (; length >= 16; input += 16, output += 16, length -= 16) {+ const uint8x16_t ct = vld1q_u8(input);++ GCM_ONE(ct, vst1q_u8(output, s[0]), ct);+ }+ if (length) {+ aes_block m, o;++ block128_zero(&m);+ block128_copy_bytes(&m, input, length);+ c++;+ s[0] = vreinterpretq_u8_u32(vsetq_lane_u32(cpu_to_be32(c), base, 3));+ ENC_ROUNDS(EACH1);+ s[0] = veorq_u8(s[0], vld1q_u8((const uint8_t *) &m));+ vst1q_u8((uint8_t *) &o, s[0]);+ for (i = 0; i < length; i++)+ output[i] = o.b[i];+ acc = ghash_one(acc, vld1q_u8((const uint8_t *) &m), ht);+ }+ GCM_EPILOGUE;+}+++/*+ * XTS. The tweak for each block is the one before it doubled, so a group's+ * eight tweaks are a short chain that runs while the eight AES chains are in+ * flight. The first tweak is the data unit number enciphered under the+ * second key; spoint skips that many blocks into the unit.+ */+#define XTS_IN(i) s[i] = veorq_u8(vld1q_u8((const uint8_t *) (input + (i))), t[i]);+#define XTS_OUT(i) vst1q_u8((uint8_t *) (output + (i)), veorq_u8(s[i], t[i]));+/*+ * The tweak is kept in general-purpose registers and moved into a vector+ * one per block. Doubling it costs three integer operations, and the+ * integer units have nothing else to do here, where the vector ones are+ * busy with the rounds and the exclusive ors: done in vector registers,+ * which is what this did, the eight doublings of a group take about as+ * long as the eight blocks of AES they are for.+ */+#define XTS_TWEAK(i) do { \+ t[i] = vreinterpretq_u8_u64( \+ vcombine_u64(vcreate_u64(tlo), vcreate_u64(thi))); \+ { \+ const uint64_t _c = thi >> 63; \+ thi = (thi << 1) | (tlo >> 63); \+ tlo = (tlo << 1) ^ (_c ? 0x87 : 0); \+ } \+} while (0);+/*+ * The group after this one's. Doubling is a chain -- each tweak waits for+ * the one before it -- and eight of them in front of the rounds that want+ * them is time in which nothing else happens, which on a processor whose+ * AES is this fast is most of the block. Worked out a group early they+ * have nothing to wait for and go through the rounds of the group before,+ * which do not want the same units. There are registers enough here for+ * both groups at once.+ */+#define XTS_TWEAK_NEXT(i) do { \+ tn[i] = vreinterpretq_u8_u64( \+ vcombine_u64(vcreate_u64(tlo), vcreate_u64(thi))); \+ { \+ const uint64_t _c = thi >> 63; \+ thi = (thi << 1) | (tlo >> 63); \+ tlo = (tlo << 1) ^ (_c ? 0x87 : 0); \+ } \+} while (0);+#define XTS_TWEAK_ROLL(i) do { t[i] = tn[i]; } while (0);++CRYPTON_TARGET_ARMV8_CRYPTO+void SIZED(crypton_aes_armv8_encrypt_xts)(aes_block *output, aes_key *key, aes_key *key2, aes_block *dataunit, uint32_t spoint, aes_block *input, uint32_t nb_blocks)+{+ const uint8_t *rk = FWD(key);+ uint8x16_t s[WAY], t[WAY], tn[WAY];+ uint64_t tlo, thi;++ {+ aes_block first;++ SIZED(crypton_aes_armv8_encrypt_block)(&first, key2, dataunit);+ tlo = first.q[0];+ thi = first.q[1];+ }+ while (spoint-- > 0) {+ const uint64_t c = thi >> 63;++ thi = (thi << 1) | (tlo >> 63);+ tlo = (tlo << 1) ^ (c ? 0x87 : 0);+ }++ EACH8(XTS_TWEAK);+ for (; nb_blocks >= WAY; nb_blocks -= WAY, input += WAY, output += WAY) {+ EACH8(XTS_IN);+ EACH8(XTS_TWEAK_NEXT);+ ENC_ROUNDS(EACH8);+ EACH8(XTS_OUT);+ EACH8(XTS_TWEAK_ROLL);+ }+ /* the group that was made ready and not used */+ {+ const uint64x2_t back = vreinterpretq_u64_u8(t[0]);++ tlo = vgetq_lane_u64(back, 0);+ thi = vgetq_lane_u64(back, 1);+ }+ for (; nb_blocks > 0; nb_blocks--, input++, output++) {+ EACH1(XTS_TWEAK);+ EACH1(XTS_IN);+ ENC_ROUNDS(EACH1);+ EACH1(XTS_OUT);+ }+}++CRYPTON_TARGET_ARMV8_CRYPTO+void SIZED(crypton_aes_armv8_decrypt_xts)(aes_block *output, aes_key *key, aes_key *key2, aes_block *dataunit, uint32_t spoint, aes_block *input, uint32_t nb_blocks)+{+ const uint8_t *fwd = FWD(key);+ const uint8_t *inv = INV(key);+ uint8x16_t s[WAY], t[WAY], tn[WAY];+ uint64_t tlo, thi;++ {+ aes_block first;++ /* the tweak is always enciphered, whichever way the data goes */+ SIZED(crypton_aes_armv8_encrypt_block)(&first, key2, dataunit);+ tlo = first.q[0];+ thi = first.q[1];+ }+ while (spoint-- > 0) {+ const uint64_t c = thi >> 63;++ thi = (thi << 1) | (tlo >> 63);+ tlo = (tlo << 1) ^ (c ? 0x87 : 0);+ }++ EACH8(XTS_TWEAK);+ for (; nb_blocks >= WAY; nb_blocks -= WAY, input += WAY, output += WAY) {+ EACH8(XTS_IN);+ EACH8(XTS_TWEAK_NEXT);+ DEC_ROUNDS(EACH8);+ EACH8(XTS_OUT);+ EACH8(XTS_TWEAK_ROLL);+ }+ /* the group that was made ready and not used */+ {+ const uint64x2_t back = vreinterpretq_u64_u8(t[0]);++ tlo = vgetq_lane_u64(back, 0);+ thi = vgetq_lane_u64(back, 1);+ }+ for (; nb_blocks > 0; nb_blocks--, input++, output++) {+ EACH1(XTS_TWEAK);+ EACH1(XTS_IN);+ DEC_ROUNDS(EACH1);+ EACH1(XTS_OUT);+ }+}++/*+ * One message, one call: the additional data, the counter-mode encryption,+ * the tag and the QUIC header protection mask, with the running tag and the+ * counter kept in registers from end to end.+ *+ * What this saves over composing crypton_aes_gcm_aad, _encrypt and _finish is+ * not the arithmetic but the boundaries. Each of those reaches its+ * primitives through a branch table, so the 128-bit state goes back to memory+ * at every step and a header of one block pays a reduction of its own. On an+ * Apple M4 that framing was 0.07 of the 0.112 microseconds a 100-byte packet+ * cost -- more than the encryption of the packet itself.+ *+ * The GHASH is taken in batches of WAY against the powers of H the key+ * already holds, so a batch costs one reduction rather than one per block,+ * and the additional data and the length block ride in the same batches as+ * the ciphertext instead of being multiplied on their own.+ */++/* start a batch, or continue one; blen is how many blocks this batch holds */+#define FG_ABSORB(blk) \+ do { \+ poly64x2_t b_ = vreinterpretq_p64_u8(vrev64q_u8(blk)); \+ if (bn == 0) { \+ uint32_t left_ = gtotal - gidx; \+ blen = left_ < WAY ? left_ : WAY; \+ acc = vreinterpretq_u64_u8( \+ vextq_u8(vreinterpretq_u8_u64(acc), \+ vreinterpretq_u8_u64(acc), 8)); \+ b_ = vreinterpretq_p64_u64(veorq_u64( \+ vreinterpretq_u64_p64(b_), acc)); \+ gH = vdupq_n_u64(0); \+ gM = gH; \+ gL = gH; \+ } \+ GHASH_MUL(b_, GHASH_POW(ht, blen - bn - 1), \+ GHASH_KARAT(ht, blen - bn - 1), gH, gM, gL); \+ gidx++; bn++; \+ if (bn == blen) { acc = ghash_reduce(gH, gM, gL); bn = 0; } \+ } while (0)++/* a block that is short, zero padded, as GHASH wants it */+#define FG_PARTIAL(p, n) \+ ({ uint8_t buf_[16]; memset(buf_, 0, 16); memcpy(buf_, (p), (n)); \+ vld1q_u8(buf_); })++CRYPTON_TARGET_ARMV8_CRYPTO+void SIZED(crypton_aes_armv8_gcm_fused)(uint8_t *out, const block128 *ht,+ aes_key *key, const uint8_t *nonce,+ const uint8_t *aad, uint32_t aadlen,+ const uint8_t *in, uint32_t inlen,+ uint32_t taglen, aes_key *hpkey,+ uint32_t sampleoff, uint8_t *mask)+{+ const uint8_t *rk = FWD(key);+ uint8x16_t s[WAY];+ uint8x16_t ek0;+ uint64x2_t acc = vdupq_n_u64(0), gH = acc, gM = acc, gL = acc;+ uint32x4_t base;+ uint32_t c = 1, bn = 0, blen = 0, gidx = 0;+ uint32_t gtotal = (aadlen + 15) / 16 + (inlen + 15) / 16 + 1;+ uint32_t i, done;+ uint8_t y0[16], lenb[16];+ uint64_t la, lc;++ memcpy(y0, nonce, 12);+ y0[12] = 0; y0[13] = 0; y0[14] = 0; y0[15] = 1;+ base = vreinterpretq_u32_u8(vld1q_u8(y0));++ s[0] = vld1q_u8(y0);+ ENC_ROUNDS(EACH1);+ ek0 = s[0];++ for (i = 0; i + 16 <= aadlen; i += 16)+ FG_ABSORB(vld1q_u8(aad + i));+ if (i < aadlen)+ FG_ABSORB(FG_PARTIAL(aad + i, aadlen - i));++ for (done = 0; done + 16 * WAY <= inlen; done += 16 * WAY) {+ const uint8_t *p = in + done;+ uint8_t *q = out + done;++ EACH8(GCM_CTR);+ c += WAY;+ ENC_ROUNDS(EACH8);+ {+ const uint8_t *input = p;+ uint8_t *output = q;+ EACH8(GCM_ENC);+ }+ FG_ABSORB(s[0]); FG_ABSORB(s[1]); FG_ABSORB(s[2]); FG_ABSORB(s[3]);+ FG_ABSORB(s[4]); FG_ABSORB(s[5]); FG_ABSORB(s[6]); FG_ABSORB(s[7]);+ }++ for (; done < inlen; done += 16) {+ uint32_t n = inlen - done < 16 ? inlen - done : 16;+ uint8x16_t m_ = n == 16 ? vld1q_u8(in + done)+ : FG_PARTIAL(in + done, n);+ c++;+ s[0] = vreinterpretq_u8_u32(vsetq_lane_u32(cpu_to_be32(c), base, 3));+ ENC_ROUNDS(EACH1);+ s[0] = veorq_u8(s[0], m_);+ if (n == 16) {+ vst1q_u8(out + done, s[0]);+ } else {+ uint8_t buf_[16];+ vst1q_u8(buf_, s[0]);+ memcpy(out + done, buf_, n);+ memset(buf_ + n, 0, 16 - n);+ s[0] = vld1q_u8(buf_);+ }+ FG_ABSORB(s[0]);+ }++ la = (uint64_t) aadlen << 3;+ lc = (uint64_t) inlen << 3;+ for (i = 0; i < 8; i++) lenb[i] = (uint8_t) (la >> (56 - 8 * i));+ for (i = 0; i < 8; i++) lenb[8 + i] = (uint8_t) (lc >> (56 - 8 * i));+ FG_ABSORB(vld1q_u8(lenb));++ {+ uint8_t tbuf[16];+ vst1q_u8(tbuf, veorq_u8(ghash_swap(vreinterpretq_u8_u64(acc)), ek0));+ memcpy(out + inlen, tbuf, taglen);+ }++ if (hpkey != 0 && mask != 0) {+ block128 sample, m;+ memcpy(&sample, out + sampleoff, 16);+ crypton_aes_encrypt_ecb(&m, hpkey, &sample, 1);+ memcpy(mask, &m, 16);+ }+}+++/*+ * The same for decryption. GCM_DEC leaves the ciphertext in s[] once it has+ * written the plaintext out, which is what GHASH wants, so the only other+ * difference is the end: the tag is compared here rather than written, every+ * byte of it whichever way the answer goes.+ */+CRYPTON_TARGET_ARMV8_CRYPTO+int SIZED(crypton_aes_armv8_gcm_fused_dec)(uint8_t *out, const block128 *ht,+ aes_key *key, const uint8_t *nonce,+ const uint8_t *aad, uint32_t aadlen,+ const uint8_t *in, uint32_t inlen,+ const uint8_t *tagp, uint32_t taglen,+ uint8_t *outtag)+{+ const uint8_t *rk = FWD(key);+ uint8x16_t s[WAY];+ uint8x16_t ek0;+ uint64x2_t acc = vdupq_n_u64(0), gH = acc, gM = acc, gL = acc;+ uint32x4_t base;+ uint32_t c = 1, bn = 0, blen = 0, gidx = 0;+ uint32_t gtotal = (aadlen + 15) / 16 + (inlen + 15) / 16 + 1;+ uint32_t i, done;+ uint8_t y0[16], lenb[16], want[16];+ uint64_t la, lc;+ uint8_t diff = 0;++ memcpy(y0, nonce, 12);+ y0[12] = 0; y0[13] = 0; y0[14] = 0; y0[15] = 1;+ base = vreinterpretq_u32_u8(vld1q_u8(y0));++ s[0] = vld1q_u8(y0);+ ENC_ROUNDS(EACH1);+ ek0 = s[0];++ for (i = 0; i + 16 <= aadlen; i += 16)+ FG_ABSORB(vld1q_u8(aad + i));+ if (i < aadlen)+ FG_ABSORB(FG_PARTIAL(aad + i, aadlen - i));++ for (done = 0; done + 16 * WAY <= inlen; done += 16 * WAY) {+ const uint8_t *p = in + done;+ uint8_t *q = out + done;++ EACH8(GCM_CTR);+ c += WAY;+ ENC_ROUNDS(EACH8);+ {+ const uint8_t *input = p;+ uint8_t *output = q;+ EACH8(GCM_DEC);+ }+ FG_ABSORB(s[0]); FG_ABSORB(s[1]); FG_ABSORB(s[2]); FG_ABSORB(s[3]);+ FG_ABSORB(s[4]); FG_ABSORB(s[5]); FG_ABSORB(s[6]); FG_ABSORB(s[7]);+ }++ for (; done < inlen; done += 16) {+ uint32_t n = inlen - done < 16 ? inlen - done : 16;+ uint8x16_t m_ = n == 16 ? vld1q_u8(in + done)+ : FG_PARTIAL(in + done, n);+ c++;+ s[0] = vreinterpretq_u8_u32(vsetq_lane_u32(cpu_to_be32(c), base, 3));+ ENC_ROUNDS(EACH1);+ {+ uint8x16_t pl = veorq_u8(s[0], m_);+ if (n == 16) {+ vst1q_u8(out + done, pl);+ } else {+ uint8_t buf_[16];+ vst1q_u8(buf_, pl);+ memcpy(out + done, buf_, n);+ }+ }+ FG_ABSORB(m_);+ }++ la = (uint64_t) aadlen << 3;+ lc = (uint64_t) inlen << 3;+ for (i = 0; i < 8; i++) lenb[i] = (uint8_t) (la >> (56 - 8 * i));+ for (i = 0; i < 8; i++) lenb[8 + i] = (uint8_t) (lc >> (56 - 8 * i));+ FG_ABSORB(vld1q_u8(lenb));++ vst1q_u8(want, veorq_u8(ghash_swap(vreinterpretq_u8_u64(acc)), ek0));+ if (outtag) {+ /* The caller holds the expected tag and will compare it itself. */+ memcpy(outtag, want, taglen);+ return 1;+ }+ for (i = 0; i < taglen; i++)+ diff |= (uint8_t) (want[i] ^ tagp[i]);+ return diff == 0;+}++#undef FG_ABSORB+#undef FG_PARTIAL++#undef WAY+#undef EACH1+#undef EACH8+#undef LOAD_IN+#undef STORE_OUT+#undef ENC_STEP+#undef ENC_LAST+#undef DEC_STEP+#undef DEC_LAST+#undef ENC_ROUNDS+#undef DEC_ROUNDS+#undef CBC_KEEP+#undef CBC_XOR+#undef CTR_SET+#undef CTR_XOR+#undef XTS_IN+#undef XTS_OUT+#undef XTS_TWEAK+#undef GCM_CTR+#undef GCM_ENC+#undef GCM_DEC+#undef GCM_GHASH+#undef GCM_FOLD+#undef GCM_PROLOGUE+#undef GCM_ONE+#undef GCM_EPILOGUE
@@ -34,7 +34,21 @@ #include <crypton_bitfn.h> #include <crypton_align.h> -typedef union {+/* Packed, so that the union asks nothing of the address it is at.+ *+ * Several callers here make one of these out of a pointer of their own -- a+ * ciphertext, a tag, a nonce -- and without this that cast produces a pointer+ * the standard says may not exist, and reading q[] out of it is a member+ * access at an address the type is not aligned for. crypton_align.h used to+ * answer that with need_alignment, which is zero on i386 and x86-64: the two+ * places the access is architecturally fine and the standard still says+ * nothing about it. UndefinedBehaviorSanitizer reported it.+ *+ * With the alignment declared to be one, the compiler is the one that knows+ * what the target can do: on i386, x86-64 and AArch64 it emits the same load+ * and store it emitted before, and where a target cannot read a word off an+ * odd address it emits what that target needs. */+typedef union __attribute__((packed)) { uint64_t q[2]; uint32_t d[4]; uint16_t w[8];
@@ -0,0 +1,1013 @@+/*+ * A fused AES-GCM for x86-64, following the design Kazuho Oku sets out in+ * "QUICむけにAES-GCM実装を最適化した話" and implements in picotls's+ * lib/fusion.c: keep AES-NI issuing every clock and fit everything else --+ * the additional data, the tag, the QUIC header protection mask -- into the+ * gaps it leaves. Written in C with intrinsics rather than assembly, for+ * the same reason he gives: the scheduling is what is complicated here, and+ * it has to stay readable to stay correct.+ *+ * Parts of this file follow fusion closely enough to say so: `loadn` and the+ * two tables it reads, `loadn_page_end` and `storen` are its `loadn128`,+ * `loadn_end_of_page` and `storen128` in another spelling, and the+ * reduction of a 256-bit product is the sequence fusion takes from Gueron's+ * "AES-GCM for Efficient Authenticated Encryption". fusion is under the MIT+ * license, which is in cbits/aes/LICENSE.fusion beside this file.+ *+ * The powers of H are built once per key, so the additional data, the+ * ciphertext and the length block are absorbed against them in batches that+ * share one reduction, rather than each block paying for a reduction of its+ * own. How many powers, and so how large a batch, is+ * CRYPTON_GCM_FUSED_POWERS in crypton_aes.h.+ *+ * Only messages shorter than CRYPTON_GCM_FUSED_MAX_MESSAGE come here. Above+ * that the stitched assembly in cbits/asm is faster, and crypton_aes.c sends+ * them there instead; below it, that assembly will not start at all.+ */++#include <crypton_cpu.h>++#ifdef WITH_GCM_FUSED++#include <stdint.h>+#include <string.h>+#include <wmmintrin.h>+#include <smmintrin.h>+#include <tmmintrin.h>++#include <crypton_aes.h>+#include <aes/gcm_fused_x86.h>++/*+ * aes_key is a struct of bytes, so its round keys sit wherever the members+ * before them leave them -- eight bytes in, as it happens. A __m128i *+ * pointed at that gets an aligned load and a fault, so each round key is+ * fetched with an unaligned load instead. Copying them somewhere aligned+ * would cost a copy per call, which at these message lengths is a tenth of+ * the whole; the load is free from L1 and the round key is fetched once for+ * all six lanes.+ */+#define RK(p, i) \+ _mm_loadu_si128((const __m128i *) ((const uint8_t *) (p) + 16 * (size_t) (i)))++/* a full sixteen-byte reversal: mask bytes 15,14,...,0 */+static const __m128i BSWAP = {0x08090a0b0c0d0e0fLL, 0x0001020304050607LL};+++#define TGT __attribute__((target("aes,pclmul,sse4.1")))++/* the two halves of a value added together: the term Karatsuba needs, and it+ * does not depend on what the value is multiplied by */+TGT static inline __m128i fold(__m128i a)+{+ return _mm_xor_si128(a, _mm_unpackhi_epi64(a, a));+}++/* H multiplied by x in the field, which is what puts H and its powers one+ * bit up: GCM numbers the bits of a field element the other way round from+ * the way the carry-less multiply does, and pre-shifting is what saves the+ * correction after every multiply.+ *+ * Written from the definition. The field is GF(2)[x] modulo x^128 + x^127 ++ * x^126 + x^121 + 1, so multiplying by x is a shift of one place, and the+ * term that leaves the top comes back as the other four. */+TGT static __m128i twist(__m128i h)+{+ /* x^127 + x^126 + x^121 + 1, the terms x^128 is congruent to */+ const __m128i poly = _mm_set_epi64x(0xc200000000000000ULL, 1);+ /* the top bit of each half */+ __m128i tops = _mm_srli_epi64(h, 63);+ /* doubling a polynomial is a shift by one: each half doubles, and the+ * low half's top bit becomes the high half's bottom bit */+ __m128i doubled = _mm_or_si128(_mm_add_epi64(h, h),+ _mm_slli_si128(tops, 8));+ /* bit 127 is the one that leaves the field; spread it to a mask by+ * subtracting it from zero, and it brings the four terms back */+ __m128i mask = _mm_sub_epi64(_mm_setzero_si128(),+ _mm_unpackhi_epi64(tops, tops));++ return _mm_xor_si128(doubled, _mm_and_si128(mask, poly));+}++/* one reduction of a 256-bit product back into the field */+TGT static __m128i reduce256(__m128i lo, __m128i hi)+{+ const __m128i poly = _mm_set_epi64x(0xc200000000000000ULL, 1);+ __m128i t;++ t = _mm_clmulepi64_si128(lo, poly, 0x10);+ lo = _mm_xor_si128(_mm_shuffle_epi32(lo, 0x4e), t);+ t = _mm_clmulepi64_si128(lo, poly, 0x10);+ lo = _mm_xor_si128(_mm_shuffle_epi32(lo, 0x4e), t);+ return _mm_xor_si128(hi, lo);+}++/*+ * One multiply in exactly the form the hot loop uses it: the left operand+ * plain, the right one already twisted. Building the table with the same+ * multiply that consumes it is the only way the two conventions cannot drift+ * apart.+ */+TGT static __m128i mul_twisted(__m128i a, __m128i ht)+{+ __m128i lo = _mm_clmulepi64_si128(a, ht, 0x00);+ __m128i hi = _mm_clmulepi64_si128(a, ht, 0x11);+ __m128i mid = _mm_clmulepi64_si128(fold(a), fold(ht), 0x00);++ mid = _mm_xor_si128(mid, _mm_xor_si128(lo, hi));+ lo = _mm_xor_si128(lo, _mm_slli_si128(mid, 8));+ hi = _mm_xor_si128(hi, _mm_srli_si128(mid, 8));+ return reduce256(lo, hi);+}++TGT void crypton_gcm_fused_key_init(aes_gcm_fused *fk, const aes_key *key)+{+ const uint8_t *rk = key->data;+ const int rounds = key->nbr;+ __m128i h, p;+ int i;++ /* H = E_K(0) */+ h = RK(rk, 0);+ for (i = 1; i < rounds; i++) h = _mm_aesenc_si128(h, RK(rk, i));+ h = _mm_aesenclast_si128(h, RK(rk, rounds));+ h = _mm_shuffle_epi8(h, BSWAP);++ {+ __m128i ht = twist(h);+ p = h;+ for (i = 0; i < CRYPTON_GCM_FUSED_POWERS; i++) {+ __m128i t = twist(p);+ _mm_storeu_si128((__m128i *) &fk->p[i].h, t);+ _mm_storeu_si128((__m128i *) &fk->p[i].r, fold(t));+ p = mul_twisted(p, ht);+ }+ }+}++/*+ * The running product. Three plain locals and a macro, not a struct behind+ * a pointer: taking the address of the accumulators is enough to keep them+ * out of registers, and then every multiply reloads and restores them. That+ * is the same mistake as reaching a table through an index the compiler+ * cannot fold, and it costs more here because it is on the inner path.+ */+#define GHASH_DECL __m128i glo = _mm_setzero_si128(), \+ ghi = _mm_setzero_si128(), \+ gmid = _mm_setzero_si128(), \+ gtag = _mm_setzero_si128(); \+ int gidx = 0, gblen = 0, gbpos = 0++/*+ * Absorb one block. Blocks are taken in batches of at most CRYPTON_GCM_FUSED_POWERS: the+ * first of a batch carries in the value the batch before it reduced to, the+ * rest go in against descending powers, and the batch ends with the one+ * reduction they share. With the batch as long as the message this is+ * picotls's single reduction; with it fixed, the state stays a fixed size.+ */+#define GHASH_ONE(blk, unused_power) \+ do { \+ __m128i _b = (blk); \+ __m128i _h, _r; \+ if (gbpos == 0) { \+ int _left = gtotal - gidx; \+ gblen = _left < CRYPTON_GCM_FUSED_POWERS ? _left : CRYPTON_GCM_FUSED_POWERS; \+ _b = _mm_xor_si128(_b, gtag); \+ glo = ghi = gmid = _mm_setzero_si128(); \+ } \+ _h = _mm_loadu_si128((const __m128i *) &fk->p[gblen-gbpos-1].h); \+ _r = _mm_loadu_si128((const __m128i *) &fk->p[gblen-gbpos-1].r); \+ glo = _mm_xor_si128(glo, _mm_clmulepi64_si128(_b, _h, 0x00)); \+ ghi = _mm_xor_si128(ghi, _mm_clmulepi64_si128(_b, _h, 0x11)); \+ gmid = _mm_xor_si128(gmid, \+ _mm_clmulepi64_si128(fold(_b), _r, 0x00)); \+ gidx++; gbpos++; \+ if (gbpos == gblen) { \+ gtag = ghash_reduce(glo, ghi, gmid); \+ gbpos = 0; \+ } \+ } while (0)++TGT static __m128i ghash_reduce(__m128i glo, __m128i ghi, __m128i gmid)+{+ __m128i mid = _mm_xor_si128(gmid, _mm_xor_si128(glo, ghi));+ __m128i lo = _mm_xor_si128(glo, _mm_slli_si128(mid, 8));+ __m128i hi = _mm_xor_si128(ghi, _mm_srli_si128(mid, 8));++ return reduce256(lo, hi);+}++/* zero every byte from n onwards, so a partial block can be fed to GHASH+ * without being written out and read back */+TGT static __m128i clampn(__m128i v, size_t n)+{+ const __m128i idx = {0x0706050403020100LL, 0x0f0e0d0c0b0a0908LL};+ return _mm_and_si128(v, _mm_cmpgt_epi8(_mm_set1_epi8((char) n), idx));+}++/*+ * A short block, zero padded, without going through the stack.+ *+ * The obvious way -- zero sixteen bytes, copy n in, load them back -- is+ * three trips to memory with a store the load has to wait for, and at these+ * lengths that is a tenth of the whole call. Reading the sixteen bytes and+ * masking off what is above n is two instructions, but it reads past the end+ * of what the caller gave, so it has to be sure those bytes exist.+ *+ * They do unless the block ends a page. A load of sixteen bytes that starts+ * at least sixteen from the end of a page stays inside it; and if the n bytes+ * asked for themselves cross the boundary then the next page is there to be+ * read as well. What is left is a block near the end of a page whose own+ * bytes stop short of it, and that is read aligned -- which cannot leave the+ * page -- and shuffled down. This is how picotls's fusion does it.+ */++/* thirty-two bytes of ones, then thirty-one of zeros: sixteen loaded from+ * 32 - n give n bytes of ones and the rest zeros */+static const uint8_t loadn_mask[63] = {+ 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff,+ 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff,+ 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff,+ 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff};++/* the first sixteen map to byte offsets, the rest to zero */+static const uint8_t loadn_shuffle[31] = {+ 0x00, 0x01, 0x02, 0x03, 0x04, 0x05, 0x06, 0x07,+ 0x08, 0x09, 0x0a, 0x0b, 0x0c, 0x0d, 0x0e, 0x0f,+ 0x80, 0x80, 0x80, 0x80, 0x80, 0x80, 0x80, 0x80,+ 0x80, 0x80, 0x80, 0x80, 0x80, 0x80, 0x80};++#if defined(__has_feature)+#if __has_feature(address_sanitizer)+#define NO_ASAN __attribute__((no_sanitize_address))+#endif+#elif defined(__SANITIZE_ADDRESS__)+#define NO_ASAN __attribute__((no_sanitize_address))+#endif+#ifndef NO_ASAN+#define NO_ASAN+#endif++TGT NO_ASAN static __m128i loadn_page_end(const uint8_t *p, size_t n)+{+ uintptr_t shift = (uintptr_t) p & 15;+ __m128i pattern = _mm_loadu_si128((const __m128i *) (loadn_shuffle + shift));++ (void) n;+ return _mm_shuffle_epi8(+ _mm_load_si128((const __m128i *) ((uintptr_t) p - shift)), pattern);+}++TGT NO_ASAN static __m128i loadn(const uint8_t *p, size_t n)+{+ __m128i mask = _mm_loadu_si128((const __m128i *) (loadn_mask + 32 - n));+ uintptr_t mod4k = (uintptr_t) p % 4096;+ __m128i v;++ if (mod4k <= 4096 - 16 || mod4k + n > 4096)+ v = _mm_loadu_si128((const __m128i *) p);+ else+ v = loadn_page_end(p, n);+ return _mm_and_si128(v, mask);+}++TGT static void storen(uint8_t *p, __m128i v, size_t n)+{+ uint8_t buf[16];+ _mm_storeu_si128((__m128i *) buf, v);+ memcpy(p, buf, n);+}++/* One block. The ten rounds of the common case are written out for the+ * same reason the six-wide group is: a loop over a round count that lives in+ * the key leaves every round key fetched through an index the compiler+ * cannot fold, and adds a branch to a chain that is already latency-bound. */+TGT static __m128i aes_one_block(const uint8_t *rk, int rounds, __m128i v)+{+ const uint8_t *k = rk;+ int i;++ if (rounds == 10) {+ v = _mm_xor_si128(v, RK(k, 0));+ v = _mm_aesenc_si128(v, RK(k, 1));+ v = _mm_aesenc_si128(v, RK(k, 2));+ v = _mm_aesenc_si128(v, RK(k, 3));+ v = _mm_aesenc_si128(v, RK(k, 4));+ v = _mm_aesenc_si128(v, RK(k, 5));+ v = _mm_aesenc_si128(v, RK(k, 6));+ v = _mm_aesenc_si128(v, RK(k, 7));+ v = _mm_aesenc_si128(v, RK(k, 8));+ v = _mm_aesenc_si128(v, RK(k, 9));+ return _mm_aesenclast_si128(v, RK(k, 10));+ }+ v = _mm_xor_si128(v, RK(k, 0));+ for (i = 1; i < rounds; i++) v = _mm_aesenc_si128(v, RK(k, i));+ return _mm_aesenclast_si128(v, RK(k, rounds));+}++/*+ * Six blocks at once, with lane 5 free to run a different key schedule from+ * the rest. Which schedule that lane uses is chosen once, into a pointer,+ * rather than tested inside the rounds, and the six live in named variables+ * rather than an array -- an array indexed by a running variable goes to+ * memory, and then every round is a load and a store instead of a register+ * to register operation, which is the whole of what this is trying to avoid.+ *+ * Lanes beyond what the caller needs still run. Six are in flight whatever+ * the message length, so the spare ones cost nothing, and that is exactly+ * why the header protection mask and E(K,Y0) are worth putting in them+ * instead of giving each a dependent chain of its own.+ */+#define WIDE6(alt) \+ do { \+ const uint8_t *ak = (alt); \+ int r; \+ t0 = _mm_xor_si128(t0, RK(rk, 0)); \+ t1 = _mm_xor_si128(t1, RK(rk, 0)); \+ t2 = _mm_xor_si128(t2, RK(rk, 0)); \+ t3 = _mm_xor_si128(t3, RK(rk, 0)); \+ t4 = _mm_xor_si128(t4, RK(rk, 0)); \+ t5 = _mm_xor_si128(t5, RK(ak, 0)); \+ for (r = 1; r < rounds; r++) { \+ __m128i k = RK(rk, r); \+ t0 = _mm_aesenc_si128(t0, k); \+ t1 = _mm_aesenc_si128(t1, k); \+ t2 = _mm_aesenc_si128(t2, k); \+ t3 = _mm_aesenc_si128(t3, k); \+ t4 = _mm_aesenc_si128(t4, k); \+ t5 = _mm_aesenc_si128(t5, RK(ak, r)); \+ GSTEP(); \+ } \+ { \+ __m128i k = RK(rk, rounds); \+ t0 = _mm_aesenclast_si128(t0, k); \+ t1 = _mm_aesenclast_si128(t1, k); \+ t2 = _mm_aesenclast_si128(t2, k); \+ t3 = _mm_aesenclast_si128(t3, k); \+ t4 = _mm_aesenclast_si128(t4, k); \+ t5 = _mm_aesenclast_si128(t5, RK(ak, rounds)); \+ } \+ } while (0)++/* the same with nothing queued to absorb: the decryption side takes its+ * GHASH straight from the input and has no queue to drain */+#define WIDE6_NOQ(alt) \+ do { \+ const uint8_t *ak = (alt); \+ int r; \+ t0 = _mm_xor_si128(t0, RK(rk, 0)); \+ t1 = _mm_xor_si128(t1, RK(rk, 0)); \+ t2 = _mm_xor_si128(t2, RK(rk, 0)); \+ t3 = _mm_xor_si128(t3, RK(rk, 0)); \+ t4 = _mm_xor_si128(t4, RK(rk, 0)); \+ t5 = _mm_xor_si128(t5, RK(ak, 0)); \+ for (r = 1; r < rounds; r++) { \+ __m128i k = RK(rk, r); \+ t0 = _mm_aesenc_si128(t0, k); \+ t1 = _mm_aesenc_si128(t1, k); \+ t2 = _mm_aesenc_si128(t2, k); \+ t3 = _mm_aesenc_si128(t3, k); \+ t4 = _mm_aesenc_si128(t4, k); \+ t5 = _mm_aesenc_si128(t5, RK(ak, r)); \+ } \+ { \+ __m128i k = RK(rk, rounds); \+ t0 = _mm_aesenclast_si128(t0, k); \+ t1 = _mm_aesenclast_si128(t1, k); \+ t2 = _mm_aesenclast_si128(t2, k); \+ t3 = _mm_aesenclast_si128(t3, k); \+ t4 = _mm_aesenclast_si128(t4, k); \+ t5 = _mm_aesenclast_si128(t5, RK(ak, rounds)); \+ } \+ } while (0)++/*+ * The same written out for the ten rounds of AES-128, with a slot for one+ * queued multiply between each of the first six. The loop above cannot take+ * them: the round count is a value in the key, so there is no place the+ * compiler knows is a round apart from the next, and a test before each+ * multiply would end the basic block the scheduler works inside. This pass+ * runs with a group's multiplies still waiting, and on a short message that+ * is most of what it has to do.+ */+#define WROUND6(r, ak) \+ do { \+ __m128i k = RK(rk, r); \+ t0 = _mm_aesenc_si128(t0, k); \+ t1 = _mm_aesenc_si128(t1, k); \+ t2 = _mm_aesenc_si128(t2, k); \+ t3 = _mm_aesenc_si128(t3, k); \+ t4 = _mm_aesenc_si128(t4, k); \+ t5 = _mm_aesenc_si128(t5, RK(ak, r)); \+ } while (0)++#define WIDE6_10(alt) \+ do { \+ const uint8_t *ak = (alt); \+ t0 = _mm_xor_si128(t0, RK(rk, 0)); \+ t1 = _mm_xor_si128(t1, RK(rk, 0)); \+ t2 = _mm_xor_si128(t2, RK(rk, 0)); \+ t3 = _mm_xor_si128(t3, RK(rk, 0)); \+ t4 = _mm_xor_si128(t4, RK(rk, 0)); \+ t5 = _mm_xor_si128(t5, RK(ak, 0)); \+ WROUND6(1, ak); GAT(0); \+ WROUND6(2, ak); GAT(1); \+ WROUND6(3, ak); GAT(2); \+ WROUND6(4, ak); GAT(3); \+ WROUND6(5, ak); GAT(4); \+ WROUND6(6, ak); GAT(5); \+ WROUND6(7, ak); \+ WROUND6(8, ak); \+ WROUND6(9, ak); \+ { \+ __m128i k = RK(rk, 10); \+ t0 = _mm_aesenclast_si128(t0, k); \+ t1 = _mm_aesenclast_si128(t1, k); \+ t2 = _mm_aesenclast_si128(t2, k); \+ t3 = _mm_aesenclast_si128(t3, k); \+ t4 = _mm_aesenclast_si128(t4, k); \+ t5 = _mm_aesenclast_si128(t5, RK(ak, 10)); \+ } \+ } while (0)++/*+ * v2: six blocks of AES in flight at once, so the ten rounds of one block no+ * longer wait on each other -- AES-NI is pipelined and will take one+ * instruction a clock as long as the instructions in flight are independent.+ * The GHASH multiplies of the group just finished are issued between the+ * rounds of the group now running, which is the stitching: they do not want+ * the same execution port, so held against each other they cost about what+ * the rounds alone cost.+ */++/*+ * The counter block, built without leaving the vector registers.+ *+ * GCM counts in the low 32 bits of the block, big endian, and wraps there.+ * ctr holds the block with its bytes reversed, so those four bytes are the+ * low lane and _mm_add_epi32 steps them without carrying into the nonce+ * above -- which is the wrap GCM asks for. A shuffle puts the bytes back.+ *+ * The obvious way -- increment a uint32_t, byte swap it, pinsrd it in --+ * costs a move from a general register to a vector one for every lane, six+ * to a group, and those do not come free.+ */+#define CTR6(j) \+ do { \+ ctr = _mm_add_epi32(ctr, one32); \+ b##j = _mm_xor_si128(_mm_shuffle_epi8(ctr, BSWAP), RK(rk, 0)); \+ } while (0)++#define ROUND6(r) \+ do { \+ __m128i k = RK(rk, r); \+ b0 = _mm_aesenc_si128(b0, k); \+ b1 = _mm_aesenc_si128(b1, k); \+ b2 = _mm_aesenc_si128(b2, k); \+ b3 = _mm_aesenc_si128(b3, k); \+ b4 = _mm_aesenc_si128(b4, k); \+ b5 = _mm_aesenc_si128(b5, k); \+ } while (0)++#define LAST6(r) \+ do { \+ __m128i k = RK(rk, r); \+ b0 = _mm_aesenclast_si128(b0, k); \+ b1 = _mm_aesenclast_si128(b1, k); \+ b2 = _mm_aesenclast_si128(b2, k); \+ b3 = _mm_aesenclast_si128(b3, k); \+ b4 = _mm_aesenclast_si128(b4, k); \+ b5 = _mm_aesenclast_si128(b5, k); \+ } while (0)++/* one GHASH multiply, taken from a queue of blocks waiting to be absorbed,+ * to be issued in the gaps between AES rounds */+/* A ring, so that a block queued while others are still waiting costs an+ * index and not a move: the queue is walked from both ends and never+ * compacted. */+#define GQ_MASK 15++#define GPUSH(v) \+ do { gq[gw] = (v); gw++; gn++; } while (0)++#define GSTEP() \+ do { \+ if (gn > 0) { \+ GHASH_ONE(gq[gi], gp); \+ gi++; gp--; gn--; \+ } \+ } while (0)++/* the same at a slot the compiler can see, for the unrolled group below */+/*+ * One queued block, at a slot the compiler can see and with nothing to test+ * before it. A test here would end the basic block, and the scheduler works+ * inside one: six tests turn the group into twelve blocks and the multiplies+ * can no longer be moved up among the rounds, which is the whole point of+ * writing them there. The group below is entered only when the queue is+ * full, so there is nothing to test.+ */+/*+ * The block a slot names, read from the output where the group before it+ * left the ciphertext rather than from a copy kept beside it. The copy+ * cost six stores a group for bytes already in memory; picotls's fusion+ * points its GHASH at the output it has just written for the same reason.+ */+#define GAT(j) GHASH_ONE(_mm_shuffle_epi8( \+ _mm_loadu_si128((const __m128i *) (prev + 16 * (j))), BSWAP), gp - (j))++TGT void crypton_gcm_fused_encrypt(uint8_t *out, const aes_gcm_fused *fk,+ const aes_key *key, const uint8_t *nonce,+ const uint8_t *aad, size_t aadlen,+ const uint8_t *in, size_t inlen, size_t taglen,+ const aes_key *hpkey, size_t sampleoff,+ uint8_t *mask)+{+ const uint8_t *rk = key->data;+ const uint8_t *hprk = hpkey != 0 ? hpkey->data : rk;+ const int rounds = key->nbr;+ const int hprounds = hpkey != 0 ? hpkey->nbr : rounds;+ GHASH_DECL;+ __m128i ctrbase, ctr, one32, ek0, tag, b0, b1, b2, b3, b4, b5;+ const int ntail_pre = (int) ((inlen % 96 + 15) / 16);+ int lane_ek0;+ __m128i gq[6];+ unsigned gi = 0, gw = 0;+ int gn = 0;+ size_t nblk = (aadlen + 15) / 16 + (inlen + 15) / 16 + 1;+ const int gtotal = (int) nblk;+ int gp = (int) nblk;+ size_t i;+ size_t done;+ int lane_mask = 0;++ /*+ * Y0: the twelve bytes of nonce and a counter of one. loadn reads the+ * nonce where it lies and masks what is above it, so this is one load+ * rather than three of four bytes each and a set built from them.+ */+ ctrbase = _mm_insert_epi32(loadn(nonce, 12), (int) __builtin_bswap32(1), 3);+ ctr = _mm_shuffle_epi8(ctrbase, BSWAP);+ one32 = _mm_set_epi32(0, 0, 0, 1);++ lane_ek0 = ntail_pre > 0 && ntail_pre <= 4;+ if (!lane_ek0)+ ek0 = aes_one_block(rk, rounds, ctrbase);++ /* The additional data goes in first and takes the highest powers, but it+ * is only queued here: absorbing it takes multiplies, and the multiplies+ * belong in the gaps between the AES rounds below rather than in front+ * of them where nothing else is running. */+ /*+ * The additional data goes in first and takes the highest powers. It is+ * absorbed here rather than queued: what the queue is for is giving the+ * rounds below something to interleave with, and a queue that sometimes+ * holds the header and sometimes does not forces a test before every+ * multiply -- which is what stopped the interleaving from happening at+ * all. See the peeled first group below.+ */+ {+ size_t nfull = aadlen / 16;+ size_t rest = aadlen % 16;++ for (i = 0; i < nfull; i++)+ GHASH_ONE(_mm_shuffle_epi8(+ _mm_loadu_si128((const __m128i *) (aad + i * 16)), BSWAP), 0);+ if (rest)+ GHASH_ONE(_mm_shuffle_epi8(loadn(aad + nfull * 16, rest), BSWAP), 0);+ }++ /* Whole groups of six. The rounds are written out rather than looped:+ * the number of them is a value in the key, so a loop over it leaves the+ * compiler fetching each round key through an index it cannot fold, and+ * the six lanes go to memory with them. Written out, the whole group+ * stays in registers, and the six multiplies of the group before can be+ * placed between the rounds by hand -- which is the stitching: AES-NI+ * and PCLMULQDQ do not contend for the same port, so the multiplies are+ * very nearly free.+ */+ done = 0;+ if (rounds == 10) {+ if (done + 96 <= inlen) {+ const uint8_t *p = in + done;+ uint8_t *q = out + done;++ CTR6(0); CTR6(1); CTR6(2); CTR6(3); CTR6(4); CTR6(5);+ ROUND6(1);+ ROUND6(2);+ ROUND6(3);+ ROUND6(4);+ ROUND6(5);+ ROUND6(6);+ ROUND6(7);+ ROUND6(8);+ ROUND6(9);+ LAST6(10);+ gn = 0; gi = 0; gw = 0;++ b0 = _mm_xor_si128(b0, _mm_loadu_si128((const __m128i *) p));+ b1 = _mm_xor_si128(b1, _mm_loadu_si128((const __m128i *) (p + 16)));+ b2 = _mm_xor_si128(b2, _mm_loadu_si128((const __m128i *) (p + 32)));+ b3 = _mm_xor_si128(b3, _mm_loadu_si128((const __m128i *) (p + 48)));+ b4 = _mm_xor_si128(b4, _mm_loadu_si128((const __m128i *) (p + 64)));+ b5 = _mm_xor_si128(b5, _mm_loadu_si128((const __m128i *) (p + 80)));+ _mm_storeu_si128((__m128i *) q, b0);+ _mm_storeu_si128((__m128i *) (q + 16), b1);+ _mm_storeu_si128((__m128i *) (q + 32), b2);+ _mm_storeu_si128((__m128i *) (q + 48), b3);+ _mm_storeu_si128((__m128i *) (q + 64), b4);+ _mm_storeu_si128((__m128i *) (q + 80), b5);++ done += 96;+ }+ for (; done + 96 <= inlen; done += 96) {+ const uint8_t *p = in + done;+ uint8_t *q = out + done;+ const uint8_t *prev = out + done - 96;++ CTR6(0); CTR6(1); CTR6(2); CTR6(3); CTR6(4); CTR6(5);+ ROUND6(1); GAT(0);+ ROUND6(2); GAT(1);+ ROUND6(3); GAT(2);+ ROUND6(4); GAT(3);+ ROUND6(5); GAT(4);+ ROUND6(6); GAT(5);+ ROUND6(7);+ ROUND6(8);+ ROUND6(9);+ LAST6(10);+ gp -= 6;++ b0 = _mm_xor_si128(b0, _mm_loadu_si128((const __m128i *) p));+ b1 = _mm_xor_si128(b1, _mm_loadu_si128((const __m128i *) (p + 16)));+ b2 = _mm_xor_si128(b2, _mm_loadu_si128((const __m128i *) (p + 32)));+ b3 = _mm_xor_si128(b3, _mm_loadu_si128((const __m128i *) (p + 48)));+ b4 = _mm_xor_si128(b4, _mm_loadu_si128((const __m128i *) (p + 64)));+ b5 = _mm_xor_si128(b5, _mm_loadu_si128((const __m128i *) (p + 80)));+ _mm_storeu_si128((__m128i *) q, b0);+ _mm_storeu_si128((__m128i *) (q + 16), b1);+ _mm_storeu_si128((__m128i *) (q + 32), b2);+ _mm_storeu_si128((__m128i *) (q + 48), b3);+ _mm_storeu_si128((__m128i *) (q + 64), b4);+ _mm_storeu_si128((__m128i *) (q + 80), b5);++ /* straight from the registers the last round left them in: the+ * queue existed only to hold them until the next group's rounds+ * could hide the multiplies, and that is 192 bytes of store and+ * load per 96 bytes of payload */+ }++ } else {+ for (done = 0; done + 96 <= inlen; done += 96) {+ const uint8_t *p = in + done;+ uint8_t *q = out + done;+ int r;++ CTR6(0); CTR6(1); CTR6(2); CTR6(3); CTR6(4); CTR6(5);+ for (r = 1; r < rounds; r++) {+ ROUND6(r);+ GSTEP();+ }+ LAST6(rounds);++ b0 = _mm_xor_si128(b0, _mm_loadu_si128((const __m128i *) p));+ b1 = _mm_xor_si128(b1, _mm_loadu_si128((const __m128i *) (p + 16)));+ b2 = _mm_xor_si128(b2, _mm_loadu_si128((const __m128i *) (p + 32)));+ b3 = _mm_xor_si128(b3, _mm_loadu_si128((const __m128i *) (p + 48)));+ b4 = _mm_xor_si128(b4, _mm_loadu_si128((const __m128i *) (p + 64)));+ b5 = _mm_xor_si128(b5, _mm_loadu_si128((const __m128i *) (p + 80)));+ _mm_storeu_si128((__m128i *) q, b0);+ _mm_storeu_si128((__m128i *) (q + 16), b1);+ _mm_storeu_si128((__m128i *) (q + 32), b2);+ _mm_storeu_si128((__m128i *) (q + 48), b3);+ _mm_storeu_si128((__m128i *) (q + 64), b4);+ _mm_storeu_si128((__m128i *) (q + 80), b5);++ while (gn > 0) GSTEP();+ gi = 0; gw = 0;+ gq[0] = _mm_shuffle_epi8(b0, BSWAP);+ gq[1] = _mm_shuffle_epi8(b1, BSWAP);+ gq[2] = _mm_shuffle_epi8(b2, BSWAP);+ gq[3] = _mm_shuffle_epi8(b3, BSWAP);+ gq[4] = _mm_shuffle_epi8(b4, BSWAP);+ gq[5] = _mm_shuffle_epi8(b5, BSWAP);+ gn = 6;+ }+ }++ /* The tail. The wide pass below runs only when there are blocks for it:+ * sixty AES instructions to fill one lane is not worth it, so a message+ * that ends on a group boundary leaves E(K,Y0) the chain it was given+ * above and takes the mask on one of its own.+ *+ * The spare lane carries the mask only when two things hold. The sample+ * has to lie entirely in output the groups above have already written:+ * this pass reads it while it runs, and the blocks it is itself+ * computing are stored after it, so a sample reaching into them would be+ * read before it exists. And the two key schedules have to have the+ * same number of rounds, because the lanes share the loop that counts+ * them and a shorter schedule would be read past its end. TLS and QUIC+ * satisfy both; anything else gets the mask on a chain of its own, which+ * is what it would have had anyway. */+ {+ __m128i t0, t1, t2, t3, t4, t5;+ __m128i tv[6];+ size_t toff[6];+ int ntail = 0, j;++ if (done >= inlen) goto no_tail;++ for (i = done; i < inlen; i += 16) {+ toff[ntail] = i;+ ctr = _mm_add_epi32(ctr, one32);+ tv[ntail] = _mm_shuffle_epi8(ctr, BSWAP);+ ntail++;+ }++ lane_mask = ntail > 0 && ntail <= 5 && hpkey != 0 && mask != 0+ && hprounds == rounds+ && sampleoff + 16 <= done;++ if (ntail > 0) {+ t0 = tv[0];+ t1 = ntail > 1 ? tv[1] : ctrbase;+ t2 = ntail > 2 ? tv[2] : ctrbase;+ t3 = ntail > 3 ? tv[3] : ctrbase;+ t4 = lane_ek0 ? ctrbase : (ntail > 4 ? tv[4] : ctrbase);+ t5 = lane_mask+ ? _mm_loadu_si128((const __m128i *) (out + sampleoff))+ : (ntail > 5 ? tv[5] : ctrbase);+ /* A group leaves exactly six queued, which is what lets the+ * slots below be named at compile time. Where no group ran+ * there is nothing to place and the plain pass will do. */+ if (rounds == 10 && done >= 96) {+ const uint8_t *prev = out + done - 96;+ WIDE6_10(lane_mask ? hprk : rk);+ } else {+ WIDE6(lane_mask ? hprk : rk);+ }+ if (lane_ek0)+ ek0 = t4;+ else+ tv[4] = t4;+ if (lane_mask)+ _mm_storeu_si128((__m128i *) mask, t5);+ else if (ntail > 5)+ tv[5] = t5;+ }+no_tail:+ while (gn > 0) GSTEP();++ /* The last group's ciphertext is absorbed by the pass above where+ * there is one to absorb it. A message that ends on a group+ * boundary has no such pass, so it is taken here. */+ if (rounds == 10 && done >= 96 && ntail == 0) {+ const uint8_t *prev = out + done - 96;+ GAT(0); GAT(1); GAT(2); GAT(3); GAT(4); GAT(5);+ }++ /*+ * The tail blocks, from the registers the pass left them in. They+ * were going through an array indexed by the loop variable, which+ * the compiler cannot see through and so keeps in memory: every+ * block then reloaded its own keystream. Named one per block and+ * reached by a test on a count instead, they stay where they are.+ */+#define TAILBLK(j, reg) \+ do { \+ size_t off = done + 16 * (j); \+ size_t n = inlen - off < 16 ? inlen - off : 16; \+ __m128i c = _mm_xor_si128(reg, n == 16 \+ ? _mm_loadu_si128((const __m128i *) (in + off)) \+ : loadn(in + off, n)); \+ if (n == 16) { \+ _mm_storeu_si128((__m128i *) (out + off), c); \+ } else { \+ /* the tag goes in directly above, so those bytes are \+ * written over anyway where there are sixteen to spare */ \+ if (n + taglen >= 16) \+ _mm_storeu_si128((__m128i *) (out + off), c); \+ else \+ storen(out + off, c, n); \+ c = clampn(c, n); \+ } \+ GHASH_ONE(_mm_shuffle_epi8(c, BSWAP), gp); \+ gp--; \+ } while (0)++ if (ntail > 0) TAILBLK(0, t0);+ if (ntail > 1) TAILBLK(1, t1);+ if (ntail > 2) TAILBLK(2, t2);+ if (ntail > 3) TAILBLK(3, t3);+ if (ntail > 4) TAILBLK(4, tv[4]);+ if (ntail > 5) TAILBLK(5, tv[5]);+#undef TAILBLK+ }++ {+ /*+ * The length block: the additional data's bit count and the+ * message's, each big endian in a half, and then reversed like+ * every other block on its way to GHASH.+ *+ * Reversed, that block is the two counts as ordinary little endian+ * words with the message's in the low half -- which is one set, and+ * no shuffle. Sixteen byte stores to the stack and a load back is+ * what it cost before.+ */+ GHASH_ONE(_mm_set_epi64x((long long) ((uint64_t) aadlen << 3),+ (long long) ((uint64_t) inlen << 3)), gp);+ }++ tag = _mm_shuffle_epi8(gtag, BSWAP);+ tag = _mm_xor_si128(tag, ek0);+ if (taglen == 16)+ _mm_storeu_si128((__m128i *) (out + inlen), tag);+ else+ storen(out + inlen, tag, taglen);++ /* A sample the pass above could not reach -- because it covered blocks+ * that pass was still computing, or the tag, which is written just now+ * -- is taken here instead, where everything it can cover exists. */+ if (!lane_mask && hpkey != 0 && mask != 0)+ _mm_storeu_si128((__m128i *) mask,+ aes_one_block(hprk, hprounds, _mm_loadu_si128(+ (const __m128i *) (out + sampleoff))));+}+++/*+ * The same for decryption, which is the simpler of the two.+ *+ * What GHASH absorbs here is the ciphertext, and the ciphertext is the+ * input: it is there before any of the AES has run. So there is no queue --+ * the multiplies of a group go between the rounds of that same group rather+ * than waiting for the one after, and nothing is stored and loaded back to+ * carry them across.+ *+ * The tag is compared here, every byte of it whichever way the answer goes,+ * and the answer is 1 for a message whose tag matched.+ */++/* one ciphertext block straight from the input, at a slot named here */+#define DAT(j) GHASH_ONE(_mm_shuffle_epi8( \+ _mm_loadu_si128((const __m128i *) (p + 16 * (j))), BSWAP), 0)++/* the next counter block, into a named register */+#define CTRT(t) \+ do { ctr = _mm_add_epi32(ctr, one32); \+ t = _mm_shuffle_epi8(ctr, BSWAP); } while (0)++TGT int crypton_gcm_fused_decrypt(uint8_t *out, const aes_gcm_fused *fk,+ const aes_key *key, const uint8_t *nonce,+ const uint8_t *aad, size_t aadlen,+ const uint8_t *in, size_t inlen,+ const uint8_t *tag, size_t taglen,+ uint8_t *outtag)+{+ const uint8_t *rk = key->data;+ const int rounds = key->nbr;+ GHASH_DECL;+ __m128i ctrbase, ctr, one32, ek0, want, b0, b1, b2, b3, b4, b5;+ const int ntail_pre = (int) ((inlen % 96 + 15) / 16);+ int lane_ek0, gp = 0;+ const int gtotal = (int) ((aadlen + 15) / 16 + (inlen + 15) / 16 + 1);+ size_t i;+ size_t done;+ uint8_t diff = 0;++ ctrbase = _mm_insert_epi32(loadn(nonce, 12), (int) __builtin_bswap32(1), 3);+ ctr = _mm_shuffle_epi8(ctrbase, BSWAP);+ one32 = _mm_set_epi32(0, 0, 0, 1);++ lane_ek0 = ntail_pre > 0 && ntail_pre <= 5;+ if (!lane_ek0)+ ek0 = aes_one_block(rk, rounds, ctrbase);++ {+ size_t nfull = aadlen / 16;+ size_t rest = aadlen % 16;++ for (i = 0; i < nfull; i++)+ GHASH_ONE(_mm_shuffle_epi8(+ _mm_loadu_si128((const __m128i *) (aad + i * 16)), BSWAP), 0);+ if (rest)+ GHASH_ONE(_mm_shuffle_epi8(loadn(aad + nfull * 16, rest), BSWAP), 0);+ }++ done = 0;+ if (rounds == 10) {+ for (; done + 96 <= inlen; done += 96) {+ const uint8_t *p = in + done;+ uint8_t *q = out + done;++ CTR6(0); CTR6(1); CTR6(2); CTR6(3); CTR6(4); CTR6(5);+ ROUND6(1); DAT(0);+ ROUND6(2); DAT(1);+ ROUND6(3); DAT(2);+ ROUND6(4); DAT(3);+ ROUND6(5); DAT(4);+ ROUND6(6); DAT(5);+ ROUND6(7);+ ROUND6(8);+ ROUND6(9);+ LAST6(10);++ _mm_storeu_si128((__m128i *) q,+ _mm_xor_si128(b0, _mm_loadu_si128((const __m128i *) p)));+ _mm_storeu_si128((__m128i *) (q + 16),+ _mm_xor_si128(b1, _mm_loadu_si128((const __m128i *) (p + 16))));+ _mm_storeu_si128((__m128i *) (q + 32),+ _mm_xor_si128(b2, _mm_loadu_si128((const __m128i *) (p + 32))));+ _mm_storeu_si128((__m128i *) (q + 48),+ _mm_xor_si128(b3, _mm_loadu_si128((const __m128i *) (p + 48))));+ _mm_storeu_si128((__m128i *) (q + 64),+ _mm_xor_si128(b4, _mm_loadu_si128((const __m128i *) (p + 64))));+ _mm_storeu_si128((__m128i *) (q + 80),+ _mm_xor_si128(b5, _mm_loadu_si128((const __m128i *) (p + 80))));+ }+ } else {+ for (; done + 96 <= inlen; done += 96) {+ const uint8_t *p = in + done;+ uint8_t *q = out + done;+ int r;++ CTR6(0); CTR6(1); CTR6(2); CTR6(3); CTR6(4); CTR6(5);+ for (r = 1; r < rounds; r++) {+ ROUND6(r);+ if (r <= 6) DAT(r - 1);+ }+ LAST6(rounds);++ _mm_storeu_si128((__m128i *) q,+ _mm_xor_si128(b0, _mm_loadu_si128((const __m128i *) p)));+ _mm_storeu_si128((__m128i *) (q + 16),+ _mm_xor_si128(b1, _mm_loadu_si128((const __m128i *) (p + 16))));+ _mm_storeu_si128((__m128i *) (q + 32),+ _mm_xor_si128(b2, _mm_loadu_si128((const __m128i *) (p + 32))));+ _mm_storeu_si128((__m128i *) (q + 48),+ _mm_xor_si128(b3, _mm_loadu_si128((const __m128i *) (p + 48))));+ _mm_storeu_si128((__m128i *) (q + 64),+ _mm_xor_si128(b4, _mm_loadu_si128((const __m128i *) (p + 64))));+ _mm_storeu_si128((__m128i *) (q + 80),+ _mm_xor_si128(b5, _mm_loadu_si128((const __m128i *) (p + 80))));+ }+ }++ /* the tail, with E(K,Y0) in a lane the length leaves idle */+ {+ __m128i t0, t1, t2, t3, t4, t5;+ int ntail = (int) ((inlen - done + 15) / 16), j;++ /* the counter is where the groups left it */+ t0 = t1 = t2 = t3 = t4 = t5 = ctrbase;+ if (ntail > 0) CTRT(t0);+ if (ntail > 1) CTRT(t1);+ if (ntail > 2) CTRT(t2);+ if (ntail > 3) CTRT(t3);+ if (ntail > 4) CTRT(t4);+ if (ntail > 5) CTRT(t5);++ if (ntail > 0) {+ WIDE6_NOQ(rk);+ if (lane_ek0) ek0 = t5;+ }++ for (j = 0; j < ntail; j++) {+ size_t off = done + 16 * (size_t) j;+ size_t n = inlen - off < 16 ? inlen - off : 16;+ __m128i c = n == 16 ? _mm_loadu_si128((const __m128i *) (in + off))+ : loadn(in + off, n);+ __m128i ks = j == 0 ? t0 : j == 1 ? t1 : j == 2 ? t2+ : j == 3 ? t3 : j == 4 ? t4 : t5;++ GHASH_ONE(_mm_shuffle_epi8(c, BSWAP), 0);+ {+ __m128i pl = _mm_xor_si128(ks, c);+ if (n == 16)+ _mm_storeu_si128((__m128i *) (out + off), pl);+ else+ storen(out + off, pl, n);+ }+ }+ }++ GHASH_ONE(_mm_set_epi64x((long long) ((uint64_t) aadlen << 3),+ (long long) ((uint64_t) inlen << 3)), gp);++ want = _mm_xor_si128(_mm_shuffle_epi8(gtag, BSWAP), ek0);+ {+ uint8_t got[16];+ _mm_storeu_si128((__m128i *) got, want);+ if (outtag) {+ /* The caller holds the expected tag and will compare it itself. */+ memcpy(outtag, got, taglen);+ return 1;+ }+ for (i = 0; i < taglen; i++)+ diff |= (uint8_t) (got[i] ^ tag[i]);+ }+ return diff == 0;+}++#endif /* WITH_GCM_FUSED */
@@ -0,0 +1,55 @@+/*+ * Copyright (c) 2026 Kazu Yamamoto <kazu@iij.ad.jp>+ *+ * All rights reserved.+ *+ * Redistribution and use in source and binary forms, with or without+ * modification, are permitted provided that the following conditions+ * are met:+ * 1. Redistributions of source code must retain the above copyright+ * notice, this list of conditions and the following disclaimer.+ * 2. Redistributions in binary form must reproduce the above copyright+ * notice, this list of conditions and the following disclaimer in the+ * documentation and/or other materials provided with the distribution.+ * 3. Neither the name of the author nor the names of his contributors+ * may be used to endorse or promote products derived from this software+ * without specific prior written permission.+ *+ * THIS SOFTWARE IS PROVIDED BY THE REGENTS AND CONTRIBUTORS ``AS IS'' AND+ * ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE+ * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE+ * ARE DISCLAIMED. IN NO EVENT SHALL THE AUTHORS OR CONTRIBUTORS BE LIABLE+ * FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL+ * DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS+ * OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)+ * HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT+ * LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY+ * OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF+ * SUCH DAMAGE.+ */++#ifndef CRYPTON_GCM_FUSED_X86_H+#define CRYPTON_GCM_FUSED_X86_H++#include <stdint.h>+#include <stddef.h>+#include <crypton_aes.h>++void crypton_gcm_fused_key_init(aes_gcm_fused *fk, const aes_key *key);++void crypton_gcm_fused_encrypt(uint8_t *out, const aes_gcm_fused *fk,+ const aes_key *key,+ const uint8_t *nonce,+ const uint8_t *aad, size_t aadlen,+ const uint8_t *in, size_t inlen, size_t taglen,+ const aes_key *hpkey, size_t sampleoff,+ uint8_t *mask);++int crypton_gcm_fused_decrypt(uint8_t *out, const aes_gcm_fused *fk,+ const aes_key *key, const uint8_t *nonce,+ const uint8_t *aad, size_t aadlen,+ const uint8_t *in, size_t inlen,+ const uint8_t *tag, size_t taglen,+ uint8_t *outtag);++#endif
@@ -0,0 +1,364 @@+/*+ * AES-GCM through VAES and VPCLMULQDQ in their 512-bit form, which takes+ * four blocks where the 256-bit form in cbits/aes/gcm_vaes_x86.c takes two+ * and the 128-bit one takes one. The instruction rate is the same, so the+ * work per group halves again.+ *+ * This is that file widened and nothing else: the same group of sixteen+ * blocks, the same descending powers of H sharing one reduction, the same+ * round keys read from memory rather than held in registers. Four blocks to+ * a register means the group fills four of them rather than eight, which is+ * what leaves room for the group's own ciphertext to be kept for the GHASH+ * when encrypting.+ *+ * Nothing here is borrowed. OpenSSL's and BoringSSL's AVX-512 AES-GCM is+ * Apache-2.0 and s2n-bignum has no GCM at all.+ *+ * The reduction at the end is a copy of the one in gcm_vaes_x86.c rather+ * than a call to it: the two files are compiled for different instruction+ * sets, and a function compiled for one cannot be inlined into the other.+ */+#include "aes/gcm_vaes512_x86.h"++#ifdef WITH_GCM_VAES512++#include <string.h>+#include <immintrin.h>++#include <aes/gf.h>+#include <aes/block128.h>++#if defined(__clang__) || defined(__GNUC__)+#define V512_TARGET \+ __attribute__((target("avx512f,avx512bw,avx512vl,aes,pclmul,vaes,vpclmulqdq")))+#else+#define V512_TARGET+#endif++/*+ * Thirty-two blocks to a group, four to a register, so eight registers are+ * in flight. The number of registers is what matters as much as the blocks+ * per instruction: AES-NI has a latency of four cycles against a throughput+ * of one, so it takes eight independent chains to keep two ports busy. Four+ * registers of four blocks was written first and measured *slower* than the+ * 256-bit path -- the blocks per instruction had doubled and the chains had+ * halved.+ *+ * The table holds sixteen powers of H, so the GHASH of a group is two passes+ * of sixteen blocks, the second picking up the tag the first leaves.+ */+#define V512WIDE 8+#define V512HALF 4+#define V512BYTES 512++/*+ * The 128-bit multiply of cbits/aes/x86ni.c, done in all four lanes at once.+ * Every shuffle and shift here works inside its own 128-bit lane, so the+ * four products never mix: what comes out is four independent carry-less+ * products, accumulated by the caller and reduced together at the end.+ */+V512_TARGET+static inline void clmul512(__m512i a, __m512i b, __m512i *lo, __m512i *hi)+{+ const __m512i bswap = _mm512_set4_epi32(+ 0x00010203, 0x04050607, 0x08090a0b, 0x0c0d0e0f);+ __m512i t3, t4, t5, t6;++ a = _mm512_shuffle_epi8(a, bswap);++ /* Karatsuba, as in the 128-bit one: three multiplies, not four */+ t3 = _mm512_clmulepi64_epi128(a, b, 0x00);+ t6 = _mm512_clmulepi64_epi128(a, b, 0x11);+ t4 = _mm512_clmulepi64_epi128(+ _mm512_xor_si512(a, _mm512_shuffle_epi32(a, _MM_PERM_BADC)),+ _mm512_xor_si512(b, _mm512_shuffle_epi32(b, _MM_PERM_BADC)),+ 0x00);+ t4 = _mm512_xor_si512(t4, _mm512_xor_si512(t3, t6));++ t5 = _mm512_bslli_epi128(t4, 8);+ t4 = _mm512_bsrli_epi128(t4, 8);++ *lo = _mm512_xor_si512(t3, t5);+ *hi = _mm512_xor_si512(t6, t4);+}++/*+ * The reduction of cbits/aes/x86ni.c. By the time it runs the four lanes+ * have been folded into one, so there is one 256-bit product to reduce.+ */+V512_TARGET+static inline __m128i gfred512(__m128i t3, __m128i t6)+{+ const __m128i bswap = _mm_set_epi8(0,1,2,3,4,5,6,7,8,9,10,11,12,13,14,15);+ __m128i t2, t4, t5, t7, t8, t9;++ t7 = _mm_srli_epi32(t3, 31);+ t8 = _mm_srli_epi32(t6, 31);+ t3 = _mm_slli_epi32(t3, 1);+ t6 = _mm_slli_epi32(t6, 1);++ t9 = _mm_srli_si128(t7, 12);+ t8 = _mm_slli_si128(t8, 4);+ t7 = _mm_slli_si128(t7, 4);+ t3 = _mm_or_si128(t3, t7);+ t6 = _mm_or_si128(t6, t8);+ t6 = _mm_or_si128(t6, t9);++ t7 = _mm_slli_epi32(t3, 31);+ t8 = _mm_slli_epi32(t3, 30);+ t9 = _mm_slli_epi32(t3, 25);++ t7 = _mm_xor_si128(t7, t8);+ t7 = _mm_xor_si128(t7, t9);+ t8 = _mm_srli_si128(t7, 4);+ t7 = _mm_slli_si128(t7, 12);+ t3 = _mm_xor_si128(t3, t7);++ t2 = _mm_srli_epi32(t3, 1);+ t4 = _mm_srli_epi32(t3, 2);+ t5 = _mm_srli_epi32(t3, 7);+ t2 = _mm_xor_si128(t2, t4);+ t2 = _mm_xor_si128(t2, t5);+ t2 = _mm_xor_si128(t2, t8);+ t3 = _mm_xor_si128(t3, t2);+ t6 = _mm_xor_si128(t6, t3);++ return _mm_shuffle_epi8(t6, bswap);+}++/* the four 128-bit lanes of a register added together */+V512_TARGET+static inline __m128i fold512(__m512i v)+{+ __m256i h = _mm256_xor_si256(_mm512_castsi512_si256(v),+ _mm512_extracti64x4_epi64(v, 1));++ return _mm_xor_si128(_mm256_castsi256_si128(h),+ _mm256_extracti128_si256(h, 1));+}++/*+ * Sixteen blocks against H^16 .. H^1, one reduction. v[j] holds blocks 4j+ * to 4j+3 in its four lanes, so the powers for it are H^(16-4j) down to+ * H^(13-4j) -- the table's own order the other way round, hence the four+ * 128-bit loads rather than one 512-bit one.+ */+V512_TARGET+static inline __m128i ghash16(__m128i tag, const table_4bit htable,+ const __m512i *v, int fromwire)+{+ __m512i lo = _mm512_setzero_si512(), hi = _mm512_setzero_si512();+ __m512i l, h, b;+ int j;++ for (j = 0; j < V512HALF; j++) {+ const __m128i p0 =+ _mm_loadu_si128((const __m128i *) &htable[15 - 4 * j]);+ const __m128i p1 =+ _mm_loadu_si128((const __m128i *) &htable[14 - 4 * j]);+ const __m128i p2 =+ _mm_loadu_si128((const __m128i *) &htable[13 - 4 * j]);+ const __m128i p3 =+ _mm_loadu_si128((const __m128i *) &htable[12 - 4 * j]);+ __m512i hp = _mm512_castsi128_si512(p0);++ hp = _mm512_inserti32x4(hp, p1, 1);+ hp = _mm512_inserti32x4(hp, p2, 2);+ hp = _mm512_inserti32x4(hp, p3, 3);++ b = fromwire ? _mm512_loadu_si512(v + j) : v[j];+ if (j == 0) /* the running tag joins the first block */+ b = _mm512_xor_si512(+ b, _mm512_inserti32x4(+ _mm512_setzero_si512(), tag, 0));+ clmul512(b, hp, &l, &h);+ lo = _mm512_xor_si512(lo, l);+ hi = _mm512_xor_si512(hi, h);+ }++ /* the four lanes are independent products of the same sum: fold them */+ return gfred512(fold512(lo), fold512(hi));+}++#define KK512(r) _mm512_broadcast_i32x4(_mm_loadu_si128(k_ + (r)))++#define AESENC32(K) \+ do { \+ const __m512i rk = (K); \+ v[0] = _mm512_aesenc_epi128(v[0], rk); \+ v[1] = _mm512_aesenc_epi128(v[1], rk); \+ v[2] = _mm512_aesenc_epi128(v[2], rk); \+ v[3] = _mm512_aesenc_epi128(v[3], rk); \+ v[4] = _mm512_aesenc_epi128(v[4], rk); \+ v[5] = _mm512_aesenc_epi128(v[5], rk); \+ v[6] = _mm512_aesenc_epi128(v[6], rk); \+ v[7] = _mm512_aesenc_epi128(v[7], rk); \+ } while (0)++#define AESLAST32(K) \+ do { \+ const __m512i rk = (K); \+ v[0] = _mm512_aesenclast_epi128(v[0], rk); \+ v[1] = _mm512_aesenclast_epi128(v[1], rk); \+ v[2] = _mm512_aesenclast_epi128(v[2], rk); \+ v[3] = _mm512_aesenclast_epi128(v[3], rk); \+ v[4] = _mm512_aesenclast_epi128(v[4], rk); \+ v[5] = _mm512_aesenclast_epi128(v[5], rk); \+ v[6] = _mm512_aesenclast_epi128(v[6], rk); \+ v[7] = _mm512_aesenclast_epi128(v[7], rk); \+ } while (0)++#define XOR32(K) \+ do { \+ const __m512i rk = (K); \+ v[0] = _mm512_xor_si512(v[0], rk); \+ v[1] = _mm512_xor_si512(v[1], rk); \+ v[2] = _mm512_xor_si512(v[2], rk); \+ v[3] = _mm512_xor_si512(v[3], rk); \+ v[4] = _mm512_xor_si512(v[4], rk); \+ v[5] = _mm512_xor_si512(v[5], rk); \+ v[6] = _mm512_xor_si512(v[6], rk); \+ v[7] = _mm512_xor_si512(v[7], rk); \+ } while (0)++/*+ * The rounds are written out rather than looped for the reason the 128-bit+ * loop gives: the count is a value in the key, and a loop over it leaves the+ * round key reached through an index the compiler cannot fold.+ */+V512_TARGET+static inline __attribute__((always_inline)) void+rounds32(__m512i *v, const uint8_t *k, const int nbr)+{+ const __m128i *k_ = (const __m128i *) k;++ XOR32(KK512(0));+ AESENC32(KK512(1)); AESENC32(KK512(2)); AESENC32(KK512(3));+ AESENC32(KK512(4)); AESENC32(KK512(5)); AESENC32(KK512(6));+ AESENC32(KK512(7)); AESENC32(KK512(8)); AESENC32(KK512(9));+ if (nbr > 10) {+ AESENC32(KK512(10)); AESENC32(KK512(11));+ if (nbr > 12) {+ AESENC32(KK512(12)); AESENC32(KK512(13));+ }+ }+ AESLAST32(_mm512_broadcast_i32x4(_mm_loadu_si128(k_ + nbr)));+}++/* sixteen consecutive counters, four to a register. GCM counts in the low+ * thirty-two bits and wraps there, which is what _mm_add_epi32 does. */+V512_TARGET+static inline __m128i counters32(__m512i *v, __m128i iv, __m128i one,+ __m128i bswap)+{+ int j;++ for (j = 0; j < V512WIDE; j++) {+ __m128i c0, c1, c2, c3;+ __m512i c;++ iv = _mm_add_epi32(iv, one);+ c0 = _mm_shuffle_epi8(iv, bswap);+ iv = _mm_add_epi32(iv, one);+ c1 = _mm_shuffle_epi8(iv, bswap);+ iv = _mm_add_epi32(iv, one);+ c2 = _mm_shuffle_epi8(iv, bswap);+ iv = _mm_add_epi32(iv, one);+ c3 = _mm_shuffle_epi8(iv, bswap);++ c = _mm512_castsi128_si512(c0);+ c = _mm512_inserti32x4(c, c1, 1);+ c = _mm512_inserti32x4(c, c2, 2);+ c = _mm512_inserti32x4(c, c3, 3);+ v[j] = c;+ }+ return iv;+}++/*+ * Inlined into three callers with the round count a constant in each, which+ * folds away the tests inside the group loop -- the same reason the 256-bit+ * file gives, where without it AES-256 lost what AES-128 gained.+ */+V512_TARGET+static inline __attribute__((always_inline)) uint32_t+bulk_n(uint8_t *output, aes_gcm *gcm, const aes_key *key,+ const uint8_t *input, uint32_t length, int decrypt, const int nbr)+{+ const __m128i bswap = _mm_setr_epi8(7,6,5,4,3,2,1,0,15,14,13,12,11,10,9,8);+ const __m128i one = _mm_set_epi32(0, 1, 0, 0);+ __m512i v[V512WIDE];+ __m128i iv, tag;+ uint32_t groups = length / V512BYTES;+ uint32_t done = 0;+ uint32_t g;+ int j;++ if (groups == 0)+ return 0;++ iv = _mm_shuffle_epi8(_mm_loadu_si128((const __m128i *) &gcm->civ), bswap);+ tag = _mm_loadu_si128((const __m128i *) &gcm->tag);++ for (g = 0; g < groups; g++, input += V512BYTES, output += V512BYTES,+ done += V512BYTES) {+ iv = counters32(v, iv, one, bswap);+ rounds32(v, key->data, nbr);++ for (j = 0; j < V512WIDE; j++) {+ const __m512i in =+ _mm512_loadu_si512((const __m512i *) (input + 64 * j));++ v[j] = _mm512_xor_si512(v[j], in);+ _mm512_storeu_si512((__m512i *) (output + 64 * j), v[j]);+ }+ /* sixteen blocks to a pass, since that is how many powers of+ * H the table holds; the second picks up the tag the first+ * leaves */+ tag = ghash16(tag, gcm->htable,+ decrypt ? (const __m512i *) input : v,+ decrypt);+ tag = ghash16(tag, gcm->htable,+ decrypt ? (const __m512i *) (input + 256)+ : v + V512HALF,+ decrypt);+ }++ _mm_storeu_si128((__m128i *) &gcm->civ, _mm_shuffle_epi8(iv, bswap));+ _mm_storeu_si128((__m128i *) &gcm->tag, tag);+ return done;+}++V512_TARGET+static uint32_t bulk(uint8_t *output, aes_gcm *gcm, const aes_key *key,+ const uint8_t *input, uint32_t length, int decrypt)+{+ switch (key->nbr) {+ case 10:+ return bulk_n(output, gcm, key, input, length, decrypt, 10);+ case 12:+ return bulk_n(output, gcm, key, input, length, decrypt, 12);+ case 14:+ return bulk_n(output, gcm, key, input, length, decrypt, 14);+ default:+ return 0; /* not a key length AES has */+ }+}++uint32_t crypton_gcm_vaes512_bulk_encrypt(uint8_t *output, aes_gcm *gcm,+ const aes_key *key,+ const uint8_t *input,+ uint32_t length)+{+ return bulk(output, gcm, key, input, length, 0);+}++uint32_t crypton_gcm_vaes512_bulk_decrypt(uint8_t *output, aes_gcm *gcm,+ const aes_key *key,+ const uint8_t *input,+ uint32_t length)+{+ return bulk(output, gcm, key, input, length, 1);+}++#endif
@@ -0,0 +1,37 @@+/*+ * AES-GCM in the 512-bit form of the AES and carry-less multiply+ * instructions, which do four blocks where the 128-bit ones do one and the+ * 256-bit ones in cbits/aes/gcm_vaes_x86.c do two.+ */+#ifndef CRYPTON_GCM_VAES512_X86_H+#define CRYPTON_GCM_VAES512_X86_H++#include <crypton_cpu.h>++#if defined(ARCH_X86) && defined(__x86_64__) && defined(WITH_AESNI) \+ && defined(WITH_PCLMUL)+#define WITH_GCM_VAES512+#endif++#ifdef WITH_GCM_VAES512++#include <stdint.h>+#include <crypton_aes.h>++/* The same contract as the 256-bit pair: whole groups off the front, the+ * counter left in gcm->civ and the running tag in gcm->tag, and the number+ * of bytes taken returned, a multiple of 512 and possibly zero. */+uint32_t crypton_gcm_vaes512_bulk_encrypt(uint8_t *output, aes_gcm *gcm,+ const aes_key *key,+ const uint8_t *input,+ uint32_t length);+uint32_t crypton_gcm_vaes512_bulk_decrypt(uint8_t *output, aes_gcm *gcm,+ const aes_key *key,+ const uint8_t *input,+ uint32_t length);++/* Thirty-two blocks is the least it will start on. */+#define GCM_VAES512_MIN_BLOCKS 32++#endif+#endif
@@ -0,0 +1,325 @@+/*+ * AES-GCM through VAES and VPCLMULQDQ: the same AES and carry-less multiply+ * instructions the rest of this directory uses, in their 256-bit form, which+ * takes two blocks where the 128-bit form takes one. The instruction rate is+ * the same, so the throughput is twice -- measured at 2.00 on an EPYC 9V74,+ * for both halves, with nothing else in the loop.+ *+ * Nothing here is borrowed. OpenSSL's and BoringSSL's wide AES-GCM is+ * Apache-2.0, s2n-bignum has no GCM at all, and the CRYPTOGAMS assembly in+ * cbits/asm is 128-bit throughout -- its `vaesenc` is the VEX encoding of+ * AESENC on XMM, not the VAES extension. So this is the 128-bit loop in+ * cbits/aes/x86ni_impl.c widened, and it keeps that loop's shape: a group of+ * counters through the rounds together, the round keys read from memory+ * rather than held in registers, and the group's GHASH folded against+ * descending powers of H so that sixteen blocks share one reduction.+ *+ * The powers come from the table crypton_aesni_hinit_pclmul fills. It has+ * sixteen slots and the 128-bit loop uses eight of them; this uses all+ * sixteen, which is why that function now fills them.+ */+#include "aes/gcm_vaes_x86.h"++#ifdef WITH_GCM_VAES++#include <string.h>+#include <immintrin.h>++#include <aes/gf.h>+#include <aes/block128.h>++#if defined(__clang__) || defined(__GNUC__)+#define VAES_TARGET __attribute__((target("avx2,aes,pclmul,vaes,vpclmulqdq")))+#else+#define VAES_TARGET+#endif++/* sixteen blocks to a group, two to a register */+#define VWIDE 8++/*+ * The 128-bit multiply of cbits/aes/x86ni.c, done in both lanes at once.+ * Every shuffle and shift here works inside its own 128-bit half, so the two+ * products never mix: what comes out is two independent carry-less products,+ * accumulated by the caller and reduced together at the end.+ */+VAES_TARGET+static inline void clmul256(__m256i a, __m256i b, __m256i *lo, __m256i *hi)+{+ const __m256i bswap = _mm256_setr_epi8(+ 15,14,13,12,11,10,9,8,7,6,5,4,3,2,1,0,+ 15,14,13,12,11,10,9,8,7,6,5,4,3,2,1,0);+ __m256i t3, t4, t5, t6;++ a = _mm256_shuffle_epi8(a, bswap);++ /* Karatsuba, as in the 128-bit one: three multiplies, not four */+ t3 = _mm256_clmulepi64_epi128(a, b, 0x00);+ t6 = _mm256_clmulepi64_epi128(a, b, 0x11);+ t4 = _mm256_clmulepi64_epi128(+ _mm256_xor_si256(a, _mm256_shuffle_epi32(a, 0x4e)),+ _mm256_xor_si256(b, _mm256_shuffle_epi32(b, 0x4e)), 0x00);+ t4 = _mm256_xor_si256(t4, _mm256_xor_si256(t3, t6));++ t5 = _mm256_slli_si256(t4, 8);+ t4 = _mm256_srli_si256(t4, 8);++ *lo = _mm256_xor_si256(t3, t5);+ *hi = _mm256_xor_si256(t6, t4);+}++/*+ * The reduction of cbits/aes/x86ni.c, unchanged: by the time it runs the two+ * lanes have been folded into one, so there is one 256-bit product to reduce+ * and no reason to do it twice.+ */+VAES_TARGET+static inline __m128i gfred(__m128i t3, __m128i t6)+{+ const __m128i bswap = _mm_set_epi8(0,1,2,3,4,5,6,7,8,9,10,11,12,13,14,15);+ __m128i t2, t4, t5, t7, t8, t9;++ t7 = _mm_srli_epi32(t3, 31);+ t8 = _mm_srli_epi32(t6, 31);+ t3 = _mm_slli_epi32(t3, 1);+ t6 = _mm_slli_epi32(t6, 1);++ t9 = _mm_srli_si128(t7, 12);+ t8 = _mm_slli_si128(t8, 4);+ t7 = _mm_slli_si128(t7, 4);+ t3 = _mm_or_si128(t3, t7);+ t6 = _mm_or_si128(t6, t8);+ t6 = _mm_or_si128(t6, t9);++ t7 = _mm_slli_epi32(t3, 31);+ t8 = _mm_slli_epi32(t3, 30);+ t9 = _mm_slli_epi32(t3, 25);++ t7 = _mm_xor_si128(t7, t8);+ t7 = _mm_xor_si128(t7, t9);+ t8 = _mm_srli_si128(t7, 4);+ t7 = _mm_slli_si128(t7, 12);+ t3 = _mm_xor_si128(t3, t7);++ t2 = _mm_srli_epi32(t3, 1);+ t4 = _mm_srli_epi32(t3, 2);+ t5 = _mm_srli_epi32(t3, 7);+ t2 = _mm_xor_si128(t2, t4);+ t2 = _mm_xor_si128(t2, t5);+ t2 = _mm_xor_si128(t2, t8);+ t3 = _mm_xor_si128(t3, t2);+ t6 = _mm_xor_si128(t6, t3);++ return _mm_shuffle_epi8(t6, bswap);+}++/*+ * Sixteen blocks against H^16 .. H^1, one reduction. v[j] holds blocks 2j+ * and 2j+1 in its low and high halves, so the powers for it are H^(16-2j)+ * low and H^(15-2j) high -- the table's own order the other way round, hence+ * the pair of 128-bit loads rather than one 256-bit one.+ */+VAES_TARGET+static inline __m128i ghash16(__m128i tag, const table_4bit htable,+ const __m256i *v, int fromwire)+{+ __m256i lo = _mm256_setzero_si256(), hi = _mm256_setzero_si256();+ __m256i l, h, b;+ int j;++ for (j = 0; j < VWIDE; j++) {+ const __m256i hp = _mm256_set_m128i(+ _mm_loadu_si128((const __m128i *) &htable[14 - 2 * j]),+ _mm_loadu_si128((const __m128i *) &htable[15 - 2 * j]));++ b = fromwire ? _mm256_loadu_si256(v + j) : v[j];+ if (j == 0) /* the running tag joins the first block */+ b = _mm256_xor_si256(+ b, _mm256_inserti128_si256(+ _mm256_setzero_si256(), tag, 0));+ clmul256(b, hp, &l, &h);+ lo = _mm256_xor_si256(lo, l);+ hi = _mm256_xor_si256(hi, h);+ }++ /* the two lanes are independent products of the same sum: fold them */+ return gfred(_mm_xor_si128(_mm256_castsi256_si128(lo),+ _mm256_extracti128_si256(lo, 1)),+ _mm_xor_si128(_mm256_castsi256_si128(hi),+ _mm256_extracti128_si256(hi, 1)));+}++#define KK(r) _mm256_broadcastsi128_si256(_mm_loadu_si128(k_ + (r)))++#define AESENC16(K) \+ do { \+ const __m256i rk = (K); \+ v[0] = _mm256_aesenc_epi128(v[0], rk); \+ v[1] = _mm256_aesenc_epi128(v[1], rk); \+ v[2] = _mm256_aesenc_epi128(v[2], rk); \+ v[3] = _mm256_aesenc_epi128(v[3], rk); \+ v[4] = _mm256_aesenc_epi128(v[4], rk); \+ v[5] = _mm256_aesenc_epi128(v[5], rk); \+ v[6] = _mm256_aesenc_epi128(v[6], rk); \+ v[7] = _mm256_aesenc_epi128(v[7], rk); \+ } while (0)++#define AESLAST16(K) \+ do { \+ const __m256i rk = (K); \+ v[0] = _mm256_aesenclast_epi128(v[0], rk); \+ v[1] = _mm256_aesenclast_epi128(v[1], rk); \+ v[2] = _mm256_aesenclast_epi128(v[2], rk); \+ v[3] = _mm256_aesenclast_epi128(v[3], rk); \+ v[4] = _mm256_aesenclast_epi128(v[4], rk); \+ v[5] = _mm256_aesenclast_epi128(v[5], rk); \+ v[6] = _mm256_aesenclast_epi128(v[6], rk); \+ v[7] = _mm256_aesenclast_epi128(v[7], rk); \+ } while (0)++#define XOR16(K) \+ do { \+ const __m256i rk = (K); \+ v[0] = _mm256_xor_si256(v[0], rk); \+ v[1] = _mm256_xor_si256(v[1], rk); \+ v[2] = _mm256_xor_si256(v[2], rk); \+ v[3] = _mm256_xor_si256(v[3], rk); \+ v[4] = _mm256_xor_si256(v[4], rk); \+ v[5] = _mm256_xor_si256(v[5], rk); \+ v[6] = _mm256_xor_si256(v[6], rk); \+ v[7] = _mm256_xor_si256(v[7], rk); \+ } while (0)++/*+ * The rounds are written out rather than looped for the reason the 128-bit+ * loop gives: the count is a value in the key, and a loop over it leaves the+ * round key reached through an index the compiler cannot fold.+ */+VAES_TARGET+static inline __attribute__((always_inline)) void+rounds16(__m256i *v, const uint8_t *k, const int nbr)+{+ const __m128i *k_ = (const __m128i *) k;++ XOR16(KK(0));+ AESENC16(KK(1)); AESENC16(KK(2)); AESENC16(KK(3));+ AESENC16(KK(4)); AESENC16(KK(5)); AESENC16(KK(6));+ AESENC16(KK(7)); AESENC16(KK(8)); AESENC16(KK(9));+ if (nbr > 10) {+ AESENC16(KK(10)); AESENC16(KK(11));+ if (nbr > 12) {+ AESENC16(KK(12)); AESENC16(KK(13));+ }+ }+ AESLAST16(_mm256_broadcastsi128_si256(_mm_loadu_si128(k_ + nbr)));+}++/* sixteen consecutive counters, two to a register. GCM counts in the low+ * thirty-two bits and wraps there, which is what _mm_add_epi32 does. */+VAES_TARGET+static inline __m128i counters16(__m256i *v, __m128i iv, __m128i one,+ __m128i bswap)+{+ int j;++ for (j = 0; j < VWIDE; j++) {+ __m128i c0, c1;++ iv = _mm_add_epi32(iv, one);+ c0 = _mm_shuffle_epi8(iv, bswap);+ iv = _mm_add_epi32(iv, one);+ c1 = _mm_shuffle_epi8(iv, bswap);+ v[j] = _mm256_set_m128i(c1, c0);+ }+ return iv;+}++/*+ * The round count is a value in the key, and a test on it inside the group+ * loop is a branch the 128-bit path does not have: that one compiles a+ * separate function for each key length through the SIZED macro. This does+ * the same thing by being inlined into three callers with the count a+ * constant in each, which folds the tests away. Without it AES-256 lost+ * what AES-128 gained.+ */+VAES_TARGET+static inline __attribute__((always_inline)) uint32_t+bulk_n(uint8_t *output, aes_gcm *gcm, const aes_key *key,+ const uint8_t *input, uint32_t length, int decrypt, const int nbr)+{+ const __m128i bswap = _mm_setr_epi8(7,6,5,4,3,2,1,0,15,14,13,12,11,10,9,8);+ const __m128i one = _mm_set_epi32(0, 1, 0, 0);+ __m256i v[VWIDE];+ __m128i iv, tag;+ uint32_t groups = length / 256;+ uint32_t done = 0;+ uint32_t g;+ int j;++ if (groups == 0)+ return 0;++ iv = _mm_shuffle_epi8(_mm_loadu_si128((const __m128i *) &gcm->civ), bswap);+ tag = _mm_loadu_si128((const __m128i *) &gcm->tag);++ for (g = 0; g < groups; g++, input += 256, output += 256, done += 256) {+ iv = counters16(v, iv, one, bswap);+ rounds16(v, key->data, nbr);++ /*+ * The ciphertext is what the tag is taken over, and after+ * this exclusive or it is in v itself when encrypting. When+ * decrypting it is the input, which the GHASH below reads+ * again rather than keeping: there are sixteen vector+ * registers, the group fills eight of them, and a second+ * eight held aside is what makes the compiler spill. The+ * input is in L1 from the load a moment ago.+ */+ for (j = 0; j < VWIDE; j++) {+ const __m256i in =+ _mm256_loadu_si256((const __m256i *) (input + 32 * j));++ v[j] = _mm256_xor_si256(v[j], in);+ _mm256_storeu_si256((__m256i *) (output + 32 * j), v[j]);+ }+ tag = ghash16(tag, gcm->htable,+ decrypt ? (const __m256i *) input : v,+ decrypt);+ }++ _mm_storeu_si128((__m128i *) &gcm->civ, _mm_shuffle_epi8(iv, bswap));+ _mm_storeu_si128((__m128i *) &gcm->tag, tag);+ return done;+}++VAES_TARGET+static uint32_t bulk(uint8_t *output, aes_gcm *gcm, const aes_key *key,+ const uint8_t *input, uint32_t length, int decrypt)+{+ switch (key->nbr) {+ case 10:+ return bulk_n(output, gcm, key, input, length, decrypt, 10);+ case 12:+ return bulk_n(output, gcm, key, input, length, decrypt, 12);+ case 14:+ return bulk_n(output, gcm, key, input, length, decrypt, 14);+ default:+ return 0; /* not a key length AES has */+ }+}++uint32_t crypton_gcm_vaes_bulk_encrypt(uint8_t *output, aes_gcm *gcm,+ const aes_key *key,+ const uint8_t *input, uint32_t length)+{+ return bulk(output, gcm, key, input, length, 0);+}++uint32_t crypton_gcm_vaes_bulk_decrypt(uint8_t *output, aes_gcm *gcm,+ const aes_key *key,+ const uint8_t *input, uint32_t length)+{+ return bulk(output, gcm, key, input, length, 1);+}++#endif
@@ -0,0 +1,35 @@+/*+ * AES-GCM in the 256-bit form of the AES and carry-less multiply+ * instructions, which do two blocks where the 128-bit ones do one.+ */+#ifndef CRYPTON_GCM_VAES_X86_H+#define CRYPTON_GCM_VAES_X86_H++#include <crypton_cpu.h>++#if defined(ARCH_X86) && defined(__x86_64__) && defined(WITH_AESNI) \+ && defined(WITH_PCLMUL)+#define WITH_GCM_VAES+#endif++#ifdef WITH_GCM_VAES++#include <stdint.h>+#include <crypton_aes.h>++/* The bulk of a message in whole groups of sixteen blocks, leaving the+ * counter in gcm->civ and the running tag in gcm->tag where the caller's own+ * loop expects to find them. Returns the number of bytes taken, which is a+ * multiple of 256 and may be zero. */+uint32_t crypton_gcm_vaes_bulk_encrypt(uint8_t *output, aes_gcm *gcm,+ const aes_key *key,+ const uint8_t *input, uint32_t length);+uint32_t crypton_gcm_vaes_bulk_decrypt(uint8_t *output, aes_gcm *gcm,+ const aes_key *key,+ const uint8_t *input, uint32_t length);++/* Sixteen blocks is the least it will start on. */+#define GCM_VAES_MIN_BLOCKS 16++#endif+#endif
@@ -0,0 +1,266 @@+/*+ * Copyright (c) 2026 Kazu Yamamoto <kazu@iij.ad.jp>+ *+ * All rights reserved.+ *+ * Redistribution and use in source and binary forms, with or without+ * modification, are permitted provided that the following conditions+ * are met:+ * 1. Redistributions of source code must retain the above copyright+ * notice, this list of conditions and the following disclaimer.+ * 2. Redistributions in binary form must reproduce the above copyright+ * notice, this list of conditions and the following disclaimer in the+ * documentation and/or other materials provided with the distribution.+ * 3. Neither the name of the author nor the names of his contributors+ * may be used to endorse or promote products derived from this software+ * without specific prior written permission.+ *+ * THIS SOFTWARE IS PROVIDED BY THE REGENTS AND CONTRIBUTORS ``AS IS'' AND+ * ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE+ * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE+ * ARE DISCLAIMED. IN NO EVENT SHALL THE AUTHORS OR CONTRIBUTORS BE LIABLE+ * FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL+ * DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS+ * OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)+ * HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT+ * LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY+ * OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF+ * SUCH DAMAGE.+ *+ * What the stitched AES-GCM assembly in cbits/asm needs in order to be+ * called: the two pieces of state it reads are laid out the way OpenSSL+ * lays them out, which is not the way crypton does, and neither is worth+ * changing the rest of the library for. Both are built here, per message,+ * from the key schedule and the H that crypton already has.+ */++#include "crypton_cpu.h"++#ifdef WITH_X86_GCM_ASM++#include <stddef.h>+#include <stdint.h>+#include <string.h>+#include <wmmintrin.h>+#include <crypton_aes.h>+#include <aes/gcm_x86_asm.h>++#ifdef WITH_TARGET_ATTRIBUTES+#define TARGET_PCLMUL __attribute__((target("sse4.1,pclmul")))+#else+#define TARGET_PCLMUL+#endif++#define ALIGNMENT(n) __attribute__((aligned(n)))++/*+ * cbits/asm/aesni-gcm-x86_64-*.S. Both answer how many bytes they got+ * through, which is a multiple of six blocks and is zero if the message is+ * shorter than they are willing to start on.+ */+size_t crypton_gcm_asm_encrypt(const void *in, void *out, size_t len,+ const void *key, uint8_t ivec[16], void *Xi);+size_t crypton_gcm_asm_decrypt(const void *in, void *out, size_t len,+ const void *key, uint8_t ivec[16], void *Xi);++/*+ * The key schedule as the assembly reads it: the encryption round keys,+ * and at offset 240 the number of rounds less one, which is the count+ * OpenSSL's AES-NI key setup leaves there -- 9, 11 and 13 -- and what the+ * assembly compares against to tell the three key sizes apart.+ */+struct asm_key {+ uint8_t rd_key[240];+ uint32_t rounds;+};++/*+ * The assembly reads the running tag from the front of this and the powers+ * of H from 32 bytes in, which is where they sit in OpenSSL's GCM context+ * -- the 16 bytes between them hold H itself there and nothing here.+ * Powers up to the sixth are used, since the loop takes six blocks at a+ * time, and each pair of them is followed by the halves the Karatsuba+ * multiplication would otherwise have to add up again.+ */+struct asm_gcm {+ block128 xi;+ block128 unused;+ block128 htable[9];+};++/*+ * H, and every power of it, is kept shifted up by one bit: GCM numbers the+ * bits of a field element the other way round from the way the carry-less+ * multiply does, and pre-shifting the operand is what saves the correction+ * after each multiply.+ *+ * Written from the definition. The field is GF(2)[x] modulo x^128 + x^127 ++ * x^126 + x^121 + 1, so multiplying by x is a shift of one place, and the+ * term that leaves the top comes back as the other four.+ */+TARGET_PCLMUL+static __m128i twist(__m128i h)+{+ /* x^127 + x^126 + x^121 + 1, the terms x^128 is congruent to */+ const __m128i poly = _mm_set_epi64x(0xc200000000000000ULL, 1);+ /* the top bit of each half */+ __m128i tops = _mm_srli_epi64(h, 63);+ /* doubling a polynomial is a shift by one: each half doubles, and the+ * low half's top bit becomes the high half's bottom bit */+ __m128i doubled = _mm_or_si128(_mm_add_epi64(h, h),+ _mm_slli_si128(tops, 8));+ /* bit 127 is the one that leaves the field; spread it to a mask by+ * subtracting it from zero, and it brings the four terms back */+ __m128i mask = _mm_sub_epi64(_mm_setzero_si128(),+ _mm_unpackhi_epi64(tops, tops));++ return _mm_xor_si128(doubled, _mm_and_si128(mask, poly));+}++/* the two halves of a value added together, which is the term Karatsuba+ * needs and which does not depend on what it is multiplied by */+TARGET_PCLMUL+static __m128i fold(__m128i a)+{+ return _mm_xor_si128(a, _mm_unpackhi_epi64(a, a));+}++/*+ * The table the assembly reads: the first six powers of H, each shifted up+ * by one, and after each pair the two halves of both of them added+ * together, which is the term the Karatsuba multiplication would otherwise+ * work out for itself every time.+ *+ * The powers are not computed here. crypton's own table already holds+ * H^1 to H^8, in the byte order the multiply wants and unshifted, so+ * twisting each one is the whole of the work -- which is why this is worth+ * doing per message rather than keeping a second table in the context.+ */+TARGET_PCLMUL+static void init_htable(struct asm_gcm *st, const aes_gcm *gcm)+{+ int i;++ for (i = 0; i < 3; i++) {+ __m128i odd = twist(_mm_loadu_si128(+ (const __m128i *) &gcm->htable[2 * i]));+ __m128i even = twist(_mm_loadu_si128(+ (const __m128i *) &gcm->htable[2 * i + 1]));++ _mm_storeu_si128((__m128i *) &st->htable[3 * i + 0], odd);+ _mm_storeu_si128((__m128i *) &st->htable[3 * i + 1], even);+ _mm_storeu_si128((__m128i *) &st->htable[3 * i + 2],+ _mm_unpacklo_epi64(fold(odd), fold(even)));+ }+}++/*+ * The counter block, whose bottom 32 bits are what counts, as GCM has it.+ * crypton keeps the value it last used and the assembly wants the one it+ * is to use next, so this steps between the two conventions at each end.+ */+static void ctr32_bump(uint8_t ivec[16], uint32_t delta)+{+ uint32_t c = ((uint32_t) ivec[12] << 24) | ((uint32_t) ivec[13] << 16)+ | ((uint32_t) ivec[14] << 8) | (uint32_t) ivec[15];++ c += delta;+ ivec[12] = (uint8_t) (c >> 24);+ ivec[13] = (uint8_t) (c >> 16);+ ivec[14] = (uint8_t) (c >> 8);+ ivec[15] = (uint8_t) c;+}++/*+ * How much of the message to hand over. The assembly works in groups of+ * six blocks, and what it leaves behind goes to a loop that works in groups+ * of eight and then one at a time. Handing over every group it could take+ * often leaves two or four blocks to go through one at a time, which at a+ * multiply apiece costs more than the three groups it takes to line the+ * remainder up on eight. So the length is rounded down to whichever number+ * of six-block groups within reach leaves the least behind, modulo eight.+ */+static uint32_t handover(uint32_t blocks)+{+ uint32_t groups = blocks / 6;+ uint32_t best = groups;+ uint32_t least = (blocks - 6 * groups) % 8;+ uint32_t i;++ for (i = 1; i <= 3 && groups >= i; i++) {+ uint32_t left = (blocks - 6 * (groups - i)) % 8;++ if (left < least) {+ least = left;+ best = groups - i;+ }+ }+ return best * 6 * 16;+}++int crypton_gcm_asm_usable(void)+{+ static int resolved = 0;+ static int usable = 0;++ if (!resolved) {+ const uint32_t need = CRYPTON_X86_AVX | CRYPTON_X86_MOVBE+ | CRYPTON_X86_PCLMUL;++ usable = (crypton_x86_simd_features() & need) == need;+ resolved = 1;+ }+ return usable;+}++TARGET_PCLMUL+static uint32_t bulk(int encrypt, uint8_t *output, aes_gcm *gcm, aes_key *key,+ const uint8_t *input, uint32_t length)+{+ struct asm_gcm st ALIGNMENT(16);+ struct asm_key k ALIGNMENT(16);+ uint8_t ivec[16] ALIGNMENT(16);+ uint32_t hand;+ size_t done;++ if (!crypton_gcm_asm_usable())+ return 0;++ /* below its own minimum the assembly does nothing, so in that case+ * give it everything and let it decide */+ hand = handover(length / 16);+ if (hand < (encrypt ? 0x60 * 3 : 0x60))+ hand = length;++ memcpy(k.rd_key, key->data, 16 * (size_t) (key->nbr + 1));+ k.rounds = (uint32_t) key->nbr - 1;+ memcpy(&st.xi, &gcm->tag, 16);+ memcpy(ivec, &gcm->civ, 16);+ ctr32_bump(ivec, 1);+ init_htable(&st, gcm);++ done = encrypt+ ? crypton_gcm_asm_encrypt(input, output, hand, &k, ivec, &st.xi)+ : crypton_gcm_asm_decrypt(input, output, hand, &k, ivec, &st.xi);++ if (done > 0) {+ ctr32_bump(ivec, 0xffffffff);+ memcpy(&gcm->tag, &st.xi, 16);+ memcpy(&gcm->civ, ivec, 16);+ }+ return (uint32_t) done;+}++uint32_t crypton_gcm_asm_bulk_encrypt(uint8_t *output, aes_gcm *gcm, aes_key *key,+ const uint8_t *input, uint32_t length)+{+ return bulk(1, output, gcm, key, input, length);+}++uint32_t crypton_gcm_asm_bulk_decrypt(uint8_t *output, aes_gcm *gcm, aes_key *key,+ const uint8_t *input, uint32_t length)+{+ return bulk(0, output, gcm, key, input, length);+}++#endif
@@ -0,0 +1,72 @@+/*+ * Copyright (c) 2026 Kazu Yamamoto <kazu@iij.ad.jp>+ *+ * All rights reserved.+ *+ * Redistribution and use in source and binary forms, with or without+ * modification, are permitted provided that the following conditions+ * are met:+ * 1. Redistributions of source code must retain the above copyright+ * notice, this list of conditions and the following disclaimer.+ * 2. Redistributions in binary form must reproduce the above copyright+ * notice, this list of conditions and the following disclaimer in the+ * documentation and/or other materials provided with the distribution.+ * 3. Neither the name of the author nor the names of his contributors+ * may be used to endorse or promote products derived from this software+ * without specific prior written permission.+ *+ * THIS SOFTWARE IS PROVIDED BY THE REGENTS AND CONTRIBUTORS ``AS IS'' AND+ * ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE+ * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE+ * ARE DISCLAIMED. IN NO EVENT SHALL THE AUTHORS OR CONTRIBUTORS BE LIABLE+ * FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL+ * DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS+ * OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)+ * HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT+ * LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY+ * OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF+ * SUCH DAMAGE.+ */+#ifndef CRYPTON_AES_GCM_X86_ASM_H+#define CRYPTON_AES_GCM_X86_ASM_H++#ifdef WITH_X86_GCM_ASM++#include <stdint.h>+#include <crypton_aes.h>++/*+ * How long a message has to be before it is handed over. The assembly+ * needs the powers of H in a layout of its own, and what does not fill six+ * blocks is left to the loop that would otherwise have taken eight at a+ * time, so a short message pays for the setup and for a tail that goes+ * through one block at a time. Encryption also spends its first twelve+ * blocks in plain counter mode before the stitched loop starts, which is+ * why it has to be given a good deal more before it comes out ahead.+ *+ * Measured on a Haswell-generation x86-64: decryption is ahead from 288+ * bytes up, by 5 to 20 per cent, and below that loses by about as much.+ * Encryption between 288 and 1024 bytes is a wash -- it swings either way+ * by up to ten per cent depending on how the length divides into groups --+ * and from 1152 bytes it is ahead by 9 per cent or more, reaching 25 to 40+ * per cent once the message is a few kilobytes.+ */+#define GCM_ASM_MIN_BLOCKS_ENC 72+#define GCM_ASM_MIN_BLOCKS_DEC 18++/* whether the processor has what cbits/asm/aesni-gcm-x86_64-*.S needs */+int crypton_gcm_asm_usable(void);++/*+ * Encrypt or decrypt from the front of the message, hashing as it goes, and+ * answer how much was done -- a multiple of 96 bytes, possibly none of it.+ * The counter and the running tag in *gcm are brought forward by that much.+ */+uint32_t crypton_gcm_asm_bulk_encrypt(uint8_t *output, aes_gcm *gcm, aes_key *key,+ const uint8_t *input, uint32_t length);+uint32_t crypton_gcm_asm_bulk_decrypt(uint8_t *output, aes_gcm *gcm, aes_key *key,+ const uint8_t *input, uint32_t length);++#endif++#endif
@@ -144,3 +144,19 @@ block128_cpu_swap_be(a, &b); /* restore BE order when done */ } }++/*+ * Four GHASH steps at once. The generic table-driven multiply has no cheaper+ * way to do this than one block at a time; the point of the entry is that the+ * PMULL and PCLMUL versions can fold the four products into one reduction, so+ * the GCM loops hand over four blocks whenever they have them.+ */+void crypton_aes_generic_gf_mul4(block128 *a, const block128 *blocks, const table_4bit htable)+{+ int i;++ for (i = 0; i < 4; i++) {+ block128_xor(a, &blocks[i]);+ crypton_aes_generic_gf_mul(a, htable);+ }+}
@@ -38,5 +38,6 @@ void crypton_aes_generic_hinit(table_4bit htable, const block128 *h); void crypton_aes_generic_gf_mul(block128 *a, const table_4bit htable);+void crypton_aes_generic_gf_mul4(block128 *a, const block128 *blocks, const table_4bit htable); #endif
@@ -37,7 +37,10 @@ #include <crypton_cpu.h> #include <aes/gf.h> #include <aes/x86ni.h>+#include <aes/gcm_vaes_x86.h>+#include <aes/gcm_vaes512_x86.h> #include <aes/block128.h>+#include <aes/gcm_x86_asm.h> #ifdef ARCH_X86 #define ALIGN_UP(addr, size) (((addr) + ((size) - 1)) & (~((size) - 1)))@@ -56,7 +59,23 @@ return _mm_xor_si128(key, keygened); } +/*+ * SubWord(RotWord(w)), which is the one part of a key schedule that would+ * otherwise want the S-box out of a table. AESKEYGENASSIST computes it for+ * the words in lanes 1 and 3 and exclusive-ors the round constant into the+ * result; the constant is an immediate, so it is left at zero here and+ * applied by the caller, which keeps the 192-bit schedule a loop.+ */ TARGET_AESNI+static uint32_t key_sub_rot(uint32_t w)+{+ const __m128i t =+ _mm_aeskeygenassist_si128(_mm_setr_epi32(0, (int) w, 0, 0), 0x00);++ return (uint32_t) _mm_cvtsi128_si32(_mm_srli_si128(t, 4));+}++TARGET_AESNI static __m128i aes_128_key_expansion_aa(__m128i key, __m128i keygened) { keygened = _mm_shuffle_epi32(keygened, 0xaa);@@ -105,6 +124,34 @@ for (i = 0; i < 20; i++) _mm_storeu_si128(((__m128i *) out) + i, k[i]); break;+ case 24: {+ /*+ * The 192-bit schedule takes six words at a time where a round+ * key is four, so it does not fall into 128-bit pieces the way+ * the other two do; it is built a word at a time instead.+ * Thirteen round keys, then the eleven inverted ones.+ */+ static const uint32_t rcon[8] = {+ 0x01, 0x02, 0x04, 0x08, 0x10, 0x20, 0x40, 0x80,+ };+ uint32_t w[52];++ memcpy(w, ikey, 24);+ for (i = 6; i < 52; i++) {+ uint32_t t = w[i - 1];++ if (i % 6 == 0)+ t = key_sub_rot(t) ^ rcon[i / 6 - 1];+ w[i] = w[i - 6] ^ t;+ }+ memcpy(out, w, sizeof(w));++ for (i = 1; i < 12; i++)+ _mm_storeu_si128(((__m128i *) out) + 12 + i,+ _mm_aesimc_si128(_mm_loadu_si128(+ ((const __m128i *) w) + (12 - i))));+ break;+ } case 32: #define AES_256_key_exp_1(K1, K2, RCON) aes_128_key_expansion_ff(K1, _mm_aeskeygenassist_si128(K2, RCON)) #define AES_256_key_exp_2(K1, K2) aes_128_key_expansion_aa(K1, _mm_aeskeygenassist_si128(K2, 0x00))@@ -162,46 +209,109 @@ return v; } +/* memcpy rather than a cast, as everything else that moves bytes between a+ * crypton structure and a word does since block128 was packed. The cast+ * this replaces was written in 2014, when block128 was a plain union and+ * taking a __m128i * to one promised nothing the type did not already+ * offer. Packing it dropped its alignment to one, and the promise with it:+ * gcc has reported the cast ever since, and it is right to -- the attribute+ * on the local below is what makes the promise true, and nothing obliges+ * the next edit to keep it. Sixteen bytes of memcpy between a __m128i and+ * a sixteen-byte object is one movdqu, or nothing at all when both stay in+ * registers. */ TARGET_AESNI static __m128i gfmul_generic(__m128i tag, const table_4bit htable) {- aes_block _t ALIGNMENT(16);- _mm_store_si128((__m128i *) &_t, tag);+ aes_block _t;+ memcpy(&_t, &tag, sizeof _t); crypton_aes_generic_gf_mul(&_t, htable);- tag = _mm_load_si128((__m128i *) &_t);+ memcpy(&tag, &_t, sizeof tag); return tag; } +/* Four or eight GHASH steps. The table-driven multiply gains nothing from+ * seeing them together; the PCLMUL versions below fold them into one+ * reduction. */+TARGET_AESNI+static __m128i gfmul4_generic(__m128i tag, const table_4bit htable, const __m128i *m)+{+ int i;++ for (i = 0; i < 4; i++)+ tag = gfmul_generic(_mm_xor_si128(tag, m[i]), htable);+ return tag;+}++TARGET_AESNI+static __m128i gfmul8_generic(__m128i tag, const table_4bit htable, const __m128i *m)+{+ int i;++ for (i = 0; i < 8; i++)+ tag = gfmul_generic(_mm_xor_si128(tag, m[i]), htable);+ return tag;+}+ #ifdef WITH_PCLMUL __m128i (*crypton_gfmul_branch_ptr)(__m128i a, const table_4bit t) = gfmul_generic; #define gfmul(a,t) ((*crypton_gfmul_branch_ptr)(a,t)) +__m128i (*crypton_gfmul4_branch_ptr)(__m128i a, const table_4bit t, const __m128i *m) = gfmul4_generic;+#define gfmul4(a,t,m) ((*crypton_gfmul4_branch_ptr)(a,t,m))++__m128i (*crypton_gfmul8_branch_ptr)(__m128i a, const table_4bit t, const __m128i *m) = gfmul8_generic;+#define gfmul8(a,t,m) ((*crypton_gfmul8_branch_ptr)(a,t,m))+ /* See Intel carry-less-multiplication-instruction-in-gcm-mode-paper.pdf * * Adapted from figure 5, with additional byte swapping so that interface * is simimar to crypton_aes_generic_gf_mul. */+/*+ * The 256-bit carry-less product, before the reflection fixup and the+ * reduction. Split out from the reduction because both of those are linear+ * over XOR: several products can be added together and fixed up just once,+ * which is what gf_mul4 below does.+ */ TARGET_AESNI_PCLMUL-static __m128i gfmul_pclmuldq(__m128i a, const table_4bit htable)+static inline void clmul_pclmuldq(__m128i a, __m128i b, __m128i *lo, __m128i *hi) {- __m128i b, tmp2, tmp3, tmp4, tmp5, tmp6, tmp7, tmp8, tmp9;+ __m128i tmp3, tmp4, tmp5, tmp6; __m128i bswap_mask = _mm_set_epi8(0,1,2,3,4,5,6,7,8,9,10,11,12,13,14,15); a = _mm_shuffle_epi8(a, bswap_mask);- b = _mm_loadu_si128((__m128i *) htable); + /*+ * Karatsuba: the middle term of the product is+ * (a0^a1)(b0^b1) ^ a0b0 ^ a1b1, which is one carry-less multiply+ * where the direct form needs two. Three PCLMULQDQ rather than+ * four, at the cost of a few shuffles and exclusive ors -- worth it+ * wherever the multiply is the narrower port, which is every part+ * this has been measured on.+ */ tmp3 = _mm_clmulepi64_si128(a, b, 0x00);- tmp4 = _mm_clmulepi64_si128(a, b, 0x10);- tmp5 = _mm_clmulepi64_si128(a, b, 0x01); tmp6 = _mm_clmulepi64_si128(a, b, 0x11);+ tmp4 = _mm_clmulepi64_si128(_mm_xor_si128(a, _mm_shuffle_epi32(a, 0x4e)),+ _mm_xor_si128(b, _mm_shuffle_epi32(b, 0x4e)),+ 0x00);+ tmp4 = _mm_xor_si128(tmp4, _mm_xor_si128(tmp3, tmp6)); - tmp4 = _mm_xor_si128(tmp4, tmp5); tmp5 = _mm_slli_si128(tmp4, 8); tmp4 = _mm_srli_si128(tmp4, 8);- tmp3 = _mm_xor_si128(tmp3, tmp5);- tmp6 = _mm_xor_si128(tmp6, tmp4); + *lo = _mm_xor_si128(tmp3, tmp5);+ *hi = _mm_xor_si128(tmp6, tmp4);+}++/* Shift the 256-bit product left by one to undo GCM's bit reflection, then+ * reduce modulo the GCM polynomial. This is the expensive half. */+TARGET_AESNI_PCLMUL+static inline __m128i gfred_pclmuldq(__m128i tmp3, __m128i tmp6)+{+ __m128i tmp2, tmp4, tmp5, tmp7, tmp8, tmp9;+ __m128i bswap_mask = _mm_set_epi8(0,1,2,3,4,5,6,7,8,9,10,11,12,13,14,15);+ tmp7 = _mm_srli_epi32(tmp3, 31); tmp8 = _mm_srli_epi32(tmp6, 31); tmp3 = _mm_slli_epi32(tmp3, 1);@@ -236,14 +346,41 @@ return _mm_shuffle_epi8(tmp6, bswap_mask); } +TARGET_AESNI_PCLMUL+static __m128i gfmul_pclmuldq(__m128i a, const table_4bit htable)+{+ __m128i lo, hi;++ clmul_pclmuldq(a, _mm_loadu_si128((__m128i *) htable), &lo, &hi);+ return gfred_pclmuldq(lo, hi);+}++TARGET_AESNI_PCLMUL void crypton_aesni_hinit_pclmul(table_4bit htable, const block128 *h) {+ __m128i bswap_mask = _mm_set_epi8(0,1,2,3,4,5,6,7,8,9,10,11,12,13,14,15);+ __m128i p;+ int i;+ /* When pclmul is active we don't need to fill the table. Instead we just * store H at index 0. It is written in reverse order, so function * gfmul_pclmuldq will not byte-swap this value. */- htable->q[0] = bitfn_swap64(h->q[1]);- htable->q[1] = bitfn_swap64(h->q[0]);+ htable[0].q[0] = bitfn_swap64(h->q[1]);+ htable[0].q[1] = bitfn_swap64(h->q[0]);++ /* Indices 1..15 get H^2 .. H^16, which is what lets a group of blocks+ * fold into one reduction: gf_mul4 uses the first four, the 128-bit+ * GCM loop eight, and the 256-bit one all sixteen. The table has+ * sixteen slots and now they are all used. Filling the upper half+ * costs eight multiplies once per key, which is nothing beside a+ * message. */+ p = _mm_loadu_si128((const __m128i *) h);+ for (i = 1; i < 16; i++) {+ p = gfmul_pclmuldq(p, htable);+ _mm_storeu_si128((__m128i *) &htable[i],+ _mm_shuffle_epi8(p, bswap_mask));+ } } TARGET_AESNI_PCLMUL@@ -255,13 +392,74 @@ _mm_storeu_si128((__m128i *) a, _b); } +/*+ * Four GHASH steps -- ((((a^b0)H ^ b1)H ^ b2)H ^ b3)H -- with a single+ * reduction. Expanded that is (a^b0)H^4 ^ b1*H^3 ^ b2*H^2 ^ b3*H, so the+ * four products can be summed first and reduced once, which is where the+ * time goes. Aggregated reduction, from the Intel GCM paper.+ */+TARGET_AESNI_PCLMUL+static __m128i gfmul4_pclmul(__m128i tag, const table_4bit htable, const __m128i *m)+{+ __m128i lo, hi, l, h;+ int i;++ clmul_pclmuldq(_mm_xor_si128(tag, m[0]),+ _mm_loadu_si128((const __m128i *) &htable[3]), &lo, &hi);++ for (i = 1; i < 4; i++) {+ clmul_pclmuldq(m[i], _mm_loadu_si128((const __m128i *) &htable[3 - i]),+ &l, &h);+ lo = _mm_xor_si128(lo, l);+ hi = _mm_xor_si128(hi, h);+ }++ return gfred_pclmuldq(lo, hi);+}++TARGET_AESNI_PCLMUL+static __m128i gfmul8_pclmul(__m128i tag, const table_4bit htable, const __m128i *m)+{+ __m128i lo, hi, l, h;+ int i;++ clmul_pclmuldq(_mm_xor_si128(tag, m[0]),+ _mm_loadu_si128((const __m128i *) &htable[7]), &lo, &hi);++ for (i = 1; i < 8; i++) {+ clmul_pclmuldq(m[i], _mm_loadu_si128((const __m128i *) &htable[7 - i]),+ &l, &h);+ lo = _mm_xor_si128(lo, l);+ hi = _mm_xor_si128(hi, h);+ }++ return gfred_pclmuldq(lo, hi);+}++TARGET_AESNI_PCLMUL+void crypton_aesni_gf_mul4_pclmul(block128 *a, const block128 *blocks, const table_4bit htable)+{+ __m128i m[4];+ int i;++ for (i = 0; i < 4; i++)+ m[i] = _mm_loadu_si128((const __m128i *) &blocks[i]);++ _mm_storeu_si128((__m128i *) a,+ gfmul4_pclmul(_mm_loadu_si128((const __m128i *) a), htable, m));+}+ void crypton_aesni_init_pclmul(void) { crypton_gfmul_branch_ptr = gfmul_pclmuldq;+ crypton_gfmul4_branch_ptr = gfmul4_pclmul;+ crypton_gfmul8_branch_ptr = gfmul8_pclmul; } #else #define gfmul(a,t) (gfmul_generic(a,t))+#define gfmul4(a,t,m) (gfmul4_generic(a,t,m))+#define gfmul8(a,t,m) (gfmul8_generic(a,t,m)) #endif TARGET_AESNI@@ -271,6 +469,164 @@ return gfmul(tag, htable); } +TARGET_AESNI+static inline __m128i ghash_add4(__m128i tag, const table_4bit htable, const __m128i *m)+{+ return gfmul4(tag, htable, m);+}++TARGET_AESNI+static inline __m128i ghash_add8(__m128i tag, const table_4bit htable, const __m128i *m)+{+ return gfmul8(tag, htable, m);+}++/*+ * Eight blocks through the rounds with the round keys read from memory rather+ * than held in registers.+ *+ * There are sixteen vector registers. Eight blocks and eleven to fifteen+ * round keys do not fit in them, and when the GCM loop preloaded the keys the+ * compiler spilled: ninety-six stack accesses around a hundred AESENCs, which+ * cost more than half the loop's throughput. AESENC takes a memory operand,+ * and the round keys are in L1 from one group to the next, so reading them+ * each round costs nothing and leaves the registers for the blocks.+ */+/*+ * Eight blocks through the rounds with the round keys read from memory rather+ * than held in registers.+ *+ * There are sixteen vector registers. Eight blocks and eleven to fifteen+ * round keys do not fit in them, and when the GCM loop preloaded the keys the+ * compiler spilled: ninety-six stack accesses around a hundred AESENCs.+ * AESENC takes a memory operand and the round keys stay in L1 from one group+ * to the next, so reading them costs nothing and leaves the registers for the+ * blocks.+ *+ * The rounds are written out rather than looped: the loop cost a fifth of the+ * throughput, which is what -funroll-loops was recovering.+ */+#define K_(r) _mm_loadu_si128(k_ + (r))++/* the rounds beyond the tenth, which only a longer key has */+#define ROUNDS8_EXTRA_128+#define ROUNDS8_EXTRA_192 AESENC8(K_(10)) AESENC8(K_(11))+#define ROUNDS8_EXTRA_256 \+ AESENC8(K_(10)) AESENC8(K_(11)) AESENC8(K_(12)) AESENC8(K_(13))++#define DO_ENC_BLOCK8_MEM(m, k, nbr, EXTRA) \+ do { \+ const __m128i *k_ = (const __m128i *) (k); \+ XOR8(K_(0)) \+ AESENC8(K_(1)) AESENC8(K_(2)) AESENC8(K_(3)) \+ AESENC8(K_(4)) AESENC8(K_(5)) AESENC8(K_(6)) \+ AESENC8(K_(7)) AESENC8(K_(8)) AESENC8(K_(9)) \+ EXTRA \+ AESENCLAST8(K_(nbr)) \+ } while (0)++#define DO_ENC_BLOCK_MEM(m, k, nbr) \+ do { \+ const __m128i *k_ = (const __m128i *) (k); \+ int r_; \+ m = _mm_xor_si128(m, K_(0)); \+ for (r_ = 1; r_ < (nbr); r_++) \+ m = _mm_aesenc_si128(m, K_(r_)); \+ m = _mm_aesenclast_si128(m, K_(nbr)); \+ } while (0)++/*+ * GCM's GHASH, called directly rather than through the branch pointer the+ * other callers use: the pointer is a call the compiler cannot see through,+ * and these want to be scheduled against the rounds around them. The cost is+ * that the GCM loops are compiled with the instruction and so may only be+ * installed where the processor has it, which crypton_aes.c sees to, as it+ * already does for the AArch64 ones.+ */+#ifdef WITH_PCLMUL++#define GCM_TARGET TARGET_AESNI_PCLMUL++TARGET_AESNI_PCLMUL+static inline __m128i gcm_ghash_add(__m128i tag, const table_4bit htable, __m128i m)+{+ return gfmul_pclmuldq(_mm_xor_si128(tag, m), htable);+}++TARGET_AESNI_PCLMUL+static inline __m128i gcm_ghash_add8(__m128i tag, const table_4bit htable, const __m128i *m)+{+ return gfmul8_pclmul(tag, htable, m);+}++/*+ * One block's carry-less multiply, accumulated rather than reduced, so that+ * the eight of a group can be spread between the rounds of the next group's+ * AES.+ */+TARGET_AESNI_PCLMUL+static inline void ghash_fold(__m128i *lo, __m128i *hi, __m128i b,+ const table_4bit htable, int i)+{+ __m128i l, h;++ clmul_pclmuldq(b, _mm_loadu_si128((const __m128i *) &htable[i]), &l, &h);+ *lo = _mm_xor_si128(*lo, l);+ *hi = _mm_xor_si128(*hi, h);+}++#else++#define GCM_TARGET TARGET_AESNI+#define gcm_ghash_add(t, h, m) ghash_add((t), (h), (m))+#define gcm_ghash_add8(t, h, m) ghash_add8((t), (h), (m))++#endif++/*+ * A group of eight encrypted, with the previous group's GHASH folded in+ * between the rounds where the build has the carry-less multiply: GH(j) after+ * round j + 1, and the reduction after round nine, which every key size+ * reaches. The names are the ones the GCM loops use.+ */+#ifdef WITH_PCLMUL++#define GCM_GH(j) \+ ghash_fold(&glo_, &ghi_, \+ (j) == 0 ? _mm_xor_si128(tag, pending[0]) : pending[j], \+ gcm->htable, 7 - (j));++#define GCM_GHRED tag = gfred_pclmuldq(glo_, ghi_);++#define GCM_GROUP8(m, k, nbr, EXTRA) \+ do { \+ const __m128i *k_ = (const __m128i *) (k); \+ __m128i glo_ = _mm_setzero_si128(); \+ __m128i ghi_ = _mm_setzero_si128(); \+ XOR8(K_(0)) \+ AESENC8(K_(1)) GCM_GH(0) \+ AESENC8(K_(2)) GCM_GH(1) \+ AESENC8(K_(3)) GCM_GH(2) \+ AESENC8(K_(4)) GCM_GH(3) \+ AESENC8(K_(5)) GCM_GH(4) \+ AESENC8(K_(6)) GCM_GH(5) \+ AESENC8(K_(7)) GCM_GH(6) \+ AESENC8(K_(8)) GCM_GH(7) \+ AESENC8(K_(9)) GCM_GHRED \+ EXTRA \+ AESENCLAST8(K_(nbr)) \+ } while (0)++#else++#define GCM_GROUP8(m, k, nbr, EXTRA) \+ do { \+ DO_ENC_BLOCK8_MEM(m, k, nbr, EXTRA); \+ tag = ghash_add8(tag, gcm->htable, pending); \+ } while (0)++#endif+ #define PRELOAD_ENC_KEYS128(k) \ __m128i K0 = _mm_loadu_si128(((__m128i *) k)+0); \ __m128i K1 = _mm_loadu_si128(((__m128i *) k)+1); \@@ -284,6 +640,11 @@ __m128i K9 = _mm_loadu_si128(((__m128i *) k)+9); \ __m128i K10 = _mm_loadu_si128(((__m128i *) k)+10); +#define PRELOAD_ENC_KEYS192(k) \+ PRELOAD_ENC_KEYS128(k) \+ __m128i K11 = _mm_loadu_si128(((__m128i *) k)+11); \+ __m128i K12 = _mm_loadu_si128(((__m128i *) k)+12);+ #define PRELOAD_ENC_KEYS256(k) \ PRELOAD_ENC_KEYS128(k) \ __m128i K11 = _mm_loadu_si128(((__m128i *) k)+11); \@@ -304,6 +665,21 @@ m = _mm_aesenc_si128(m, K9); \ m = _mm_aesenclast_si128(m, K10); +#define DO_ENC_BLOCK192(m) \+ m = _mm_xor_si128(m, K0); \+ m = _mm_aesenc_si128(m, K1); \+ m = _mm_aesenc_si128(m, K2); \+ m = _mm_aesenc_si128(m, K3); \+ m = _mm_aesenc_si128(m, K4); \+ m = _mm_aesenc_si128(m, K5); \+ m = _mm_aesenc_si128(m, K6); \+ m = _mm_aesenc_si128(m, K7); \+ m = _mm_aesenc_si128(m, K8); \+ m = _mm_aesenc_si128(m, K9); \+ m = _mm_aesenc_si128(m, K10); \+ m = _mm_aesenc_si128(m, K11); \+ m = _mm_aesenclast_si128(m, K12);+ #define DO_ENC_BLOCK256(m) \ m = _mm_xor_si128(m, K0); \ m = _mm_aesenc_si128(m, K1); \@@ -334,10 +710,55 @@ __m128i K8 = _mm_loadu_si128(((__m128i *) k)+at+8); \ __m128i K9 = _mm_loadu_si128(((__m128i *) k)+at+9); \ +/*+ * Eight blocks through the rounds together, which is what covers the+ * latency of AESENC. Written out one line per block rather than left to a+ * loop over m[i]: a loop is only as good as the compiler's willingness to+ * unroll it, and when it declines the blocks go to the stack and each+ * round becomes a load and a store.+ */+#define XOR8(KK) \+ m[0] = _mm_xor_si128(m[0], KK); m[1] = _mm_xor_si128(m[1], KK); \+ m[2] = _mm_xor_si128(m[2], KK); m[3] = _mm_xor_si128(m[3], KK); \+ m[4] = _mm_xor_si128(m[4], KK); m[5] = _mm_xor_si128(m[5], KK); \+ m[6] = _mm_xor_si128(m[6], KK); m[7] = _mm_xor_si128(m[7], KK);++#define AESENC8(KK) \+ m[0] = _mm_aesenc_si128(m[0], KK); m[1] = _mm_aesenc_si128(m[1], KK); \+ m[2] = _mm_aesenc_si128(m[2], KK); m[3] = _mm_aesenc_si128(m[3], KK); \+ m[4] = _mm_aesenc_si128(m[4], KK); m[5] = _mm_aesenc_si128(m[5], KK); \+ m[6] = _mm_aesenc_si128(m[6], KK); m[7] = _mm_aesenc_si128(m[7], KK);++#define AESENCLAST8(KK) \+ m[0] = _mm_aesenclast_si128(m[0], KK); m[1] = _mm_aesenclast_si128(m[1], KK); \+ m[2] = _mm_aesenclast_si128(m[2], KK); m[3] = _mm_aesenclast_si128(m[3], KK); \+ m[4] = _mm_aesenclast_si128(m[4], KK); m[5] = _mm_aesenclast_si128(m[5], KK); \+ m[6] = _mm_aesenclast_si128(m[6], KK); m[7] = _mm_aesenclast_si128(m[7], KK);++#define DO_ENC_BLOCK8_128(m) \+ XOR8(K0) AESENC8(K1) AESENC8(K2) AESENC8(K3) AESENC8(K4) AESENC8(K5) \+ AESENC8(K6) AESENC8(K7) AESENC8(K8) AESENC8(K9) AESENCLAST8(K10)++#define DO_ENC_BLOCK8_192(m) \+ XOR8(K0) AESENC8(K1) AESENC8(K2) AESENC8(K3) AESENC8(K4) AESENC8(K5) \+ AESENC8(K6) AESENC8(K7) AESENC8(K8) AESENC8(K9) AESENC8(K10) \+ AESENC8(K11) AESENCLAST8(K12)++#define DO_ENC_BLOCK8_256(m) \+ XOR8(K0) AESENC8(K1) AESENC8(K2) AESENC8(K3) AESENC8(K4) AESENC8(K5) \+ AESENC8(K6) AESENC8(K7) AESENC8(K8) AESENC8(K9) AESENC8(K10) \+ AESENC8(K11) AESENC8(K12) AESENC8(K13) AESENCLAST8(K14)+ #define PRELOAD_DEC_KEYS128(k) \ PRELOAD_DEC_KEYS_AT(k, 10) \ __m128i K10 = _mm_loadu_si128(((__m128i *) k)+0); +#define PRELOAD_DEC_KEYS192(k) \+ PRELOAD_DEC_KEYS_AT(k, 12) \+ __m128i K10 = _mm_loadu_si128(((__m128i *) k)+12+10); \+ __m128i K11 = _mm_loadu_si128(((__m128i *) k)+12+11); \+ __m128i K12 = _mm_loadu_si128(((__m128i *) k)+0);+ #define PRELOAD_DEC_KEYS256(k) \ PRELOAD_DEC_KEYS_AT(k, 14) \ __m128i K10 = _mm_loadu_si128(((__m128i *) k)+14+10); \@@ -346,6 +767,76 @@ __m128i K13 = _mm_loadu_si128(((__m128i *) k)+14+13); \ __m128i K14 = _mm_loadu_si128(((__m128i *) k)+0); +#define AESDEC8(KK) \+ m[0] = _mm_aesdec_si128(m[0], KK); m[1] = _mm_aesdec_si128(m[1], KK); \+ m[2] = _mm_aesdec_si128(m[2], KK); m[3] = _mm_aesdec_si128(m[3], KK); \+ m[4] = _mm_aesdec_si128(m[4], KK); m[5] = _mm_aesdec_si128(m[5], KK); \+ m[6] = _mm_aesdec_si128(m[6], KK); m[7] = _mm_aesdec_si128(m[7], KK);++#define AESDECLAST8(KK) \+ m[0] = _mm_aesdeclast_si128(m[0], KK); m[1] = _mm_aesdeclast_si128(m[1], KK); \+ m[2] = _mm_aesdeclast_si128(m[2], KK); m[3] = _mm_aesdeclast_si128(m[3], KK); \+ m[4] = _mm_aesdeclast_si128(m[4], KK); m[5] = _mm_aesdeclast_si128(m[5], KK); \+ m[6] = _mm_aesdeclast_si128(m[6], KK); m[7] = _mm_aesdeclast_si128(m[7], KK);++#define DO_DEC_BLOCK8_128(m) \+ XOR8(K0) AESDEC8(K1) AESDEC8(K2) AESDEC8(K3) AESDEC8(K4) AESDEC8(K5) \+ AESDEC8(K6) AESDEC8(K7) AESDEC8(K8) AESDEC8(K9) AESDECLAST8(K10)++#define DO_DEC_BLOCK8_192(m) \+ XOR8(K0) AESDEC8(K1) AESDEC8(K2) AESDEC8(K3) AESDEC8(K4) AESDEC8(K5) \+ AESDEC8(K6) AESDEC8(K7) AESDEC8(K8) AESDEC8(K9) AESDEC8(K10) \+ AESDEC8(K11) AESDECLAST8(K12)++#define DO_DEC_BLOCK8_256(m) \+ XOR8(K0) AESDEC8(K1) AESDEC8(K2) AESDEC8(K3) AESDEC8(K4) AESDEC8(K5) \+ AESDEC8(K6) AESDEC8(K7) AESDEC8(K8) AESDEC8(K9) AESDEC8(K10) \+ AESDEC8(K11) AESDEC8(K12) AESDEC8(K13) AESDECLAST8(K14)++/*+ * The XTS tweak advances by doubling in GF(2^128). gfmulx above does that+ * through memory; this keeps it in a register, which matters once eight+ * tweaks are wanted per group. The block is little-endian, so the low+ * 64-bit half is first.+ */+TARGET_AESNI+static inline __m128i gfmulx_sse(__m128i v)+{+ const __m128i poly = _mm_set_epi64x(0, 0x87);+ const __m128i carry = _mm_srli_epi64(v, 63);+ /* the low half's carry becomes the high half's bit 0 */+ const __m128i into_hi = _mm_slli_si128(carry, 8);+ /* and the high half's becomes all ones, or nothing, in the low half */+ const __m128i out = _mm_sub_epi64(_mm_setzero_si128(), _mm_srli_si128(carry, 8));++ return _mm_xor_si128(_mm_or_si128(_mm_slli_epi64(v, 1), into_hi),+ _mm_and_si128(out, poly));+}++/*+ * The tweak doubles in a pair of general-purpose registers and is moved+ * into a vector one per block. The doubling is three integer operations,+ * and the integer units have nothing else to do here, where there are only+ * sixteen vector registers and the rounds want as many of them as they can+ * get: done in vector registers, which is what this did, the eight+ * doublings of a group both lengthen the critical path and push the round+ * keys out to memory.+ */+#define XTS_TWEAK_STEP(lo, hi) do { \+ const uint64_t _c = (hi) >> 63; \+ (hi) = ((hi) << 1) | ((lo) >> 63); \+ (lo) = ((lo) << 1) ^ (_c ? 0x87 : 0); \+} while (0)++/* the eight tweaks a group needs, from the one it starts at */+#define XTS_TWEAKS8(dst, lo, hi) do { \+ int _i; \+ for (_i = 0; _i < 8; _i++) { \+ (dst)[_i] = _mm_set_epi64x((long long) (hi), (long long) (lo)); \+ XTS_TWEAK_STEP(lo, hi); \+ } \+} while (0)+ #define DO_DEC_BLOCK128(m) \ m = _mm_xor_si128(m, K0); \ m = _mm_aesdec_si128(m, K1); \@@ -359,6 +850,21 @@ m = _mm_aesdec_si128(m, K9); \ m = _mm_aesdeclast_si128(m, K10); +#define DO_DEC_BLOCK192(m) \+ m = _mm_xor_si128(m, K0); \+ m = _mm_aesdec_si128(m, K1); \+ m = _mm_aesdec_si128(m, K2); \+ m = _mm_aesdec_si128(m, K3); \+ m = _mm_aesdec_si128(m, K4); \+ m = _mm_aesdec_si128(m, K5); \+ m = _mm_aesdec_si128(m, K6); \+ m = _mm_aesdec_si128(m, K7); \+ m = _mm_aesdec_si128(m, K8); \+ m = _mm_aesdec_si128(m, K9); \+ m = _mm_aesdec_si128(m, K10); \+ m = _mm_aesdec_si128(m, K11); \+ m = _mm_aesdeclast_si128(m, K12);+ #define DO_DEC_BLOCK256(m) \ m = _mm_xor_si128(m, K0); \ m = _mm_aesdec_si128(m, K1); \@@ -377,34 +883,73 @@ m = _mm_aesdeclast_si128(m, K14); #define SIZE 128+#define NBR 10+#define ROUNDS8_EXTRA ROUNDS8_EXTRA_128 #define SIZED(m) m##128 #define PRELOAD_ENC PRELOAD_ENC_KEYS128 #define DO_ENC_BLOCK DO_ENC_BLOCK128+#define DO_ENC_BLOCK8 DO_ENC_BLOCK8_128 #define PRELOAD_DEC PRELOAD_DEC_KEYS128 #define DO_DEC_BLOCK DO_DEC_BLOCK128+#define DO_DEC_BLOCK8 DO_DEC_BLOCK8_128 #include <aes/x86ni_impl.c> #undef SIZE+#undef NBR+#undef ROUNDS8_EXTRA #undef SIZED #undef PRELOAD_ENC #undef PRELOAD_DEC #undef DO_ENC_BLOCK+#undef DO_ENC_BLOCK8 #undef DO_DEC_BLOCK+#undef DO_DEC_BLOCK8 +#define SIZED(m) m##192+#define SIZE 192+#define NBR 12+#define ROUNDS8_EXTRA ROUNDS8_EXTRA_192+#define PRELOAD_ENC PRELOAD_ENC_KEYS192+#define DO_ENC_BLOCK DO_ENC_BLOCK192+#define DO_ENC_BLOCK8 DO_ENC_BLOCK8_192+#define PRELOAD_DEC PRELOAD_DEC_KEYS192+#define DO_DEC_BLOCK DO_DEC_BLOCK192+#define DO_DEC_BLOCK8 DO_DEC_BLOCK8_192+#include <aes/x86ni_impl.c>++#undef SIZE+#undef NBR+#undef ROUNDS8_EXTRA+#undef SIZED+#undef PRELOAD_ENC+#undef PRELOAD_DEC+#undef DO_ENC_BLOCK+#undef DO_ENC_BLOCK8+#undef DO_DEC_BLOCK+#undef DO_DEC_BLOCK8+ #define SIZED(m) m##256 #define SIZE 256+#define NBR 14+#define ROUNDS8_EXTRA ROUNDS8_EXTRA_256 #define PRELOAD_ENC PRELOAD_ENC_KEYS256 #define DO_ENC_BLOCK DO_ENC_BLOCK256+#define DO_ENC_BLOCK8 DO_ENC_BLOCK8_256 #define PRELOAD_DEC PRELOAD_DEC_KEYS256 #define DO_DEC_BLOCK DO_DEC_BLOCK256+#define DO_DEC_BLOCK8 DO_DEC_BLOCK8_256 #include <aes/x86ni_impl.c> #undef SIZE+#undef NBR+#undef ROUNDS8_EXTRA #undef SIZED #undef PRELOAD_ENC #undef PRELOAD_DEC #undef DO_ENC_BLOCK+#undef DO_ENC_BLOCK8 #undef DO_DEC_BLOCK+#undef DO_DEC_BLOCK8 #endif
@@ -59,34 +59,30 @@ #endif void crypton_aesni_init(aes_key *key, uint8_t *origkey, uint8_t size);-void crypton_aesni_encrypt_block128(aes_block *out, aes_key *key, aes_block *in);-void crypton_aesni_encrypt_block256(aes_block *out, aes_key *key, aes_block *in);-void crypton_aesni_decrypt_block128(aes_block *out, aes_key *key, aes_block *in);-void crypton_aesni_decrypt_block256(aes_block *out, aes_key *key, aes_block *in);-void crypton_aesni_encrypt_ecb128(aes_block *out, aes_key *key, aes_block *in, uint32_t blocks);-void crypton_aesni_encrypt_ecb256(aes_block *out, aes_key *key, aes_block *in, uint32_t blocks);-void crypton_aesni_decrypt_ecb128(aes_block *out, aes_key *key, aes_block *in, uint32_t blocks);-void crypton_aesni_decrypt_ecb256(aes_block *out, aes_key *key, aes_block *in, uint32_t blocks);-void crypton_aesni_encrypt_cbc128(aes_block *out, aes_key *key, aes_block *_iv, aes_block *in, uint32_t blocks);-void crypton_aesni_encrypt_cbc256(aes_block *out, aes_key *key, aes_block *_iv, aes_block *in, uint32_t blocks);-void crypton_aesni_decrypt_cbc128(aes_block *out, aes_key *key, aes_block *_iv, aes_block *in, uint32_t blocks);-void crypton_aesni_decrypt_cbc256(aes_block *out, aes_key *key, aes_block *_iv, aes_block *in, uint32_t blocks);-void crypton_aesni_encrypt_ctr128(uint8_t *out, aes_key *key, aes_block *_iv, uint8_t *in, uint32_t length);-void crypton_aesni_encrypt_ctr256(uint8_t *out, aes_key *key, aes_block *_iv, uint8_t *in, uint32_t length);-void crypton_aesni_encrypt_c32_128(uint8_t *out, aes_key *key, aes_block *_iv, uint8_t *in, uint32_t length);-void crypton_aesni_encrypt_c32_256(uint8_t *out, aes_key *key, aes_block *_iv, uint8_t *in, uint32_t length);-void crypton_aesni_encrypt_xts128(aes_block *out, aes_key *key1, aes_key *key2,- aes_block *_tweak, uint32_t spoint, aes_block *in, uint32_t blocks);-void crypton_aesni_encrypt_xts256(aes_block *out, aes_key *key1, aes_key *key2,- aes_block *_tweak, uint32_t spoint, aes_block *in, uint32_t blocks);--void crypton_aesni_gcm_encrypt128(uint8_t *out, aes_gcm *gcm, aes_key *key, uint8_t *in, uint32_t length);-void crypton_aesni_gcm_encrypt256(uint8_t *out, aes_gcm *gcm, aes_key *key, uint8_t *in, uint32_t length);+#define AESNI_DECLS(sz) \+ void crypton_aesni_encrypt_block##sz(aes_block *out, aes_key *key, aes_block *in); \+ void crypton_aesni_decrypt_block##sz(aes_block *out, aes_key *key, aes_block *in); \+ void crypton_aesni_encrypt_ecb##sz(aes_block *out, aes_key *key, aes_block *in, uint32_t blocks); \+ void crypton_aesni_decrypt_ecb##sz(aes_block *out, aes_key *key, aes_block *in, uint32_t blocks); \+ void crypton_aesni_encrypt_cbc##sz(aes_block *out, aes_key *key, aes_block *_iv, aes_block *in, uint32_t blocks); \+ void crypton_aesni_decrypt_cbc##sz(aes_block *out, aes_key *key, aes_block *_iv, aes_block *in, uint32_t blocks); \+ void crypton_aesni_encrypt_ctr##sz(uint8_t *out, aes_key *key, aes_block *_iv, uint8_t *in, uint32_t length); \+ void crypton_aesni_encrypt_c32_##sz(uint8_t *out, aes_key *key, aes_block *_iv, uint8_t *in, uint32_t length); \+ void crypton_aesni_encrypt_xts##sz(aes_block *out, aes_key *key1, aes_key *key2, \+ aes_block *_tweak, uint32_t spoint, aes_block *in, uint32_t blocks); \+ void crypton_aesni_decrypt_xts##sz(aes_block *out, aes_key *key1, aes_key *key2, \+ aes_block *_tweak, uint32_t spoint, aes_block *in, uint32_t blocks); \+ void crypton_aesni_gcm_encrypt##sz(uint8_t *out, aes_gcm *gcm, aes_key *key, uint8_t *in, uint32_t length); \+ void crypton_aesni_gcm_decrypt##sz(uint8_t *out, aes_gcm *gcm, aes_key *key, uint8_t *in, uint32_t length);+AESNI_DECLS(128)+AESNI_DECLS(192)+AESNI_DECLS(256) #ifdef WITH_PCLMUL void crypton_aesni_init_pclmul(void); void crypton_aesni_hinit_pclmul(table_4bit htable, const block128 *h); void crypton_aesni_gf_mul_pclmul(block128 *a, const table_4bit htable);+void crypton_aesni_gf_mul4_pclmul(block128 *a, const block128 *blocks, const table_4bit htable); #endif #endif
@@ -204,34 +204,124 @@ void SIZED(crypton_aesni_encrypt_xts)(aes_block *out, aes_key *key1, aes_key *key2, aes_block *_tweak, uint32_t spoint, aes_block *in, uint32_t blocks) {- __m128i tweak = _mm_loadu_si128((__m128i *) _tweak);+ uint64_t tlo, thi; do { __m128i *k2 = (__m128i *) key2->data;+ __m128i tweak = _mm_loadu_si128((__m128i *) _tweak);+ aes_block first ALIGNMENT(16);+ PRELOAD_ENC(k2); DO_ENC_BLOCK(tweak);+ _mm_storeu_si128((__m128i *) &first, tweak);+ tlo = first.q[0];+ thi = first.q[1]; while (spoint-- > 0)- tweak = gfmulx(tweak);+ XTS_TWEAK_STEP(tlo, thi); } while (0) ; do { __m128i *k1 = (__m128i *) key1->data;- PRELOAD_ENC(k1); - for ( ; blocks-- > 0; in += 1, out += 1, tweak = gfmulx(tweak)) {+ /*+ * Eight at a time. The eight tweaks are kept from one group+ * to the next and each is advanced by eight doublings at+ * once, which is a single multiplication and does not wait+ * for the other seven; doubling along the group instead,+ * which is what this did, puts a chain of eight in front of+ * every set of rounds, and on a processor whose AES is fast+ * that chain is most of the block.+ */+ for ( ; blocks >= 8; blocks -= 8, in += 8, out += 8) {+ __m128i m[8], t[8];+ int i;++ XTS_TWEAKS8(t, tlo, thi);+ for (i = 0; i < 8; i++)+ m[i] = _mm_xor_si128(+ _mm_loadu_si128((__m128i *) (in + i)), t[i]);+ DO_ENC_BLOCK8_MEM(m, k1, NBR, ROUNDS8_EXTRA);+ for (i = 0; i < 8; i++)+ _mm_storeu_si128((__m128i *) (out + i),+ _mm_xor_si128(m[i], t[i]));+ }+ for ( ; blocks-- > 0; in += 1, out += 1) {+ const __m128i tweak =+ _mm_set_epi64x((long long) thi, (long long) tlo); __m128i m = _mm_loadu_si128((__m128i *) in); m = _mm_xor_si128(m, tweak);- DO_ENC_BLOCK(m);+ DO_ENC_BLOCK_MEM(m, k1, NBR); m = _mm_xor_si128(m, tweak); _mm_storeu_si128((__m128i *) out, m);+ XTS_TWEAK_STEP(tlo, thi); } } while (0); } +/*+ * XTS the other way, which until now fell to the generic loop -- and which+ * nothing reached at all, since crypton_aes_decrypt_xts called the generic+ * function directly rather than through the branch table. The tweak is+ * enciphered whichever way the data goes; only the data is deciphered.+ */ TARGET_AESNI+void SIZED(crypton_aesni_decrypt_xts)(aes_block *out, aes_key *key1, aes_key *key2,+ aes_block *_tweak, uint32_t spoint, aes_block *in, uint32_t blocks)+{+ uint64_t tlo, thi;++ do {+ __m128i *k2 = (__m128i *) key2->data;+ __m128i tweak = _mm_loadu_si128((__m128i *) _tweak);+ aes_block first ALIGNMENT(16);++ PRELOAD_ENC(k2);+ DO_ENC_BLOCK(tweak);+ _mm_storeu_si128((__m128i *) &first, tweak);+ tlo = first.q[0];+ thi = first.q[1];++ while (spoint-- > 0)+ XTS_TWEAK_STEP(tlo, thi);+ } while (0) ;++ do {+ __m128i *k1 = (__m128i *) key1->data;+ PRELOAD_DEC(k1);++ /* the tweaks kept and advanced, as encryption has them */+ for ( ; blocks >= 8; blocks -= 8, in += 8, out += 8) {+ __m128i m[8], t[8];+ int i;++ XTS_TWEAKS8(t, tlo, thi);+ for (i = 0; i < 8; i++)+ m[i] = _mm_xor_si128(+ _mm_loadu_si128((__m128i *) (in + i)), t[i]);+ DO_DEC_BLOCK8(m);+ for (i = 0; i < 8; i++)+ _mm_storeu_si128((__m128i *) (out + i),+ _mm_xor_si128(m[i], t[i]));+ }+ for ( ; blocks-- > 0; in += 1, out += 1) {+ const __m128i tweak =+ _mm_set_epi64x((long long) thi, (long long) tlo);+ __m128i m = _mm_loadu_si128((__m128i *) in);++ m = _mm_xor_si128(m, tweak);+ DO_DEC_BLOCK(m);+ m = _mm_xor_si128(m, tweak);++ _mm_storeu_si128((__m128i *) out, m);+ XTS_TWEAK_STEP(tlo, thi);+ }+ } while (0);+}++GCM_TARGET void SIZED(crypton_aesni_gcm_encrypt)(uint8_t *output, aes_gcm *gcm, aes_key *key, uint8_t *input, uint32_t length) { __m128i *k = (__m128i *) key->data;@@ -239,15 +329,107 @@ __m128i one = _mm_set_epi32(0,1,0,0); uint32_t nb_blocks = length / 16; uint32_t part_block_len = length % 16;+ /* the group of ciphertext whose GHASH has not been taken yet */+ __m128i pending[8];+ int held = 0; gcm->length_input += length; +#ifdef WITH_GCM_VAES512+ /*+ * The widest form the processor has, first: four blocks to an+ * instruction where the 256-bit one below takes two and the assembly+ * after that takes one. Same contract throughout -- whole groups off+ * the front, the counter and the tag left behind.+ */+ if (nb_blocks >= GCM_VAES512_MIN_BLOCKS+ && (crypton_x86_simd_features() & CRYPTON_X86_VAES512)) {+ uint32_t done = crypton_gcm_vaes512_bulk_encrypt(+ output, gcm, key, input, nb_blocks * 16);++ output += done;+ input += done;+ nb_blocks -= done / 16;+ }+#endif+#ifdef WITH_GCM_VAES+ /*+ * The 256-bit instructions next: they take two blocks where the ones+ * below take one, and the assembly that follows is 128-bit+ * throughout.+ */+ if (nb_blocks >= GCM_VAES_MIN_BLOCKS+ && (crypton_x86_simd_features() & CRYPTON_X86_VAES)) {+ uint32_t done = crypton_gcm_vaes_bulk_encrypt(output, gcm, key,+ input,+ nb_blocks * 16);++ output += done;+ input += done;+ nb_blocks -= done / 16;+ }+#endif+#if defined(WITH_X86_GCM_ASM) && defined(WITH_PCLMUL)+ /*+ * The stitched assembly next, which takes whole groups of six+ * blocks off the front of the message and leaves the counter and the+ * running tag where the loop below expects to find them. It wants+ * eighteen blocks before it will start, and answers with what it did.+ */+ if (nb_blocks >= GCM_ASM_MIN_BLOCKS_ENC) {+ uint32_t done = crypton_gcm_asm_bulk_encrypt(output, gcm, key,+ input, nb_blocks * 16);++ output += done;+ input += done;+ nb_blocks -= done / 16;+ }+#endif+ __m128i tag = _mm_loadu_si128((__m128i *) &gcm->tag); __m128i iv = _mm_loadu_si128((__m128i *) &gcm->civ); iv = _mm_shuffle_epi8(iv, bswap_mask); - PRELOAD_ENC(k); + /*+ * Eight blocks at a time: the counters go through the rounds together+ * so the pipeline has something to do while AESENC is in flight, and+ * their GHASH folds into one reduction against H^8 .. H^1 rather than+ * eight.+ *+ * The GHASH is of the group before, not this one. Taken in step the+ * two halves cannot overlap at all: the multiply of a block waits for+ * the rounds that produced it, and on this processor they do not even+ * want the same port -- AESENC and PCLMULQDQ issue to different ones,+ * so held a group apart they run through each other. It costs one+ * group's worth of ciphertext kept aside and a last GHASH after the+ * loop.+ */+ for (; nb_blocks >= 8; nb_blocks -= 8, output += 128, input += 128) {+ __m128i m[8];+ int i;++ for (i = 0; i < 8; i++) {+ /* iv += 1, put back in big endian */+ iv = _mm_add_epi32(iv, one);+ m[i] = _mm_shuffle_epi8(iv, bswap_mask);+ }+ if (held)+ GCM_GROUP8(m, k, NBR, ROUNDS8_EXTRA);+ else+ DO_ENC_BLOCK8_MEM(m, k, NBR, ROUNDS8_EXTRA);++ for (i = 0; i < 8; i++) {+ m[i] = _mm_xor_si128(m[i],+ _mm_loadu_si128((__m128i *) (input + 16 * i)));+ _mm_storeu_si128((__m128i *) (output + 16 * i), m[i]);+ }+ for (i = 0; i < 8; i++)+ pending[i] = m[i];+ held = 1;+ }+ if (held)+ tag = gcm_ghash_add8(tag, gcm->htable, pending); for (; nb_blocks-- > 0; output += 16, input += 16) { /* iv += 1 */ iv = _mm_add_epi32(iv, one);@@ -255,11 +437,11 @@ /* put back iv in big endian, encrypt it, * and xor it to input */ __m128i tmp = _mm_shuffle_epi8(iv, bswap_mask);- DO_ENC_BLOCK(tmp);+ DO_ENC_BLOCK_MEM(tmp, k, NBR); __m128i m = _mm_loadu_si128((__m128i *) input); m = _mm_xor_si128(m, tmp); - tag = ghash_add(tag, gcm->htable, m);+ tag = gcm_ghash_add(tag, gcm->htable, m); /* store it out */ _mm_storeu_si128((__m128i *) output, m);@@ -294,16 +476,145 @@ /* put back iv in big endian mode, encrypt it and xor it with input */ __m128i tmp = _mm_shuffle_epi8(iv, bswap_mask);- DO_ENC_BLOCK(tmp);+ DO_ENC_BLOCK_MEM(tmp, k, NBR); __m128i m = _mm_loadu_si128((__m128i *) &block); m = _mm_xor_si128(m, tmp); m = _mm_shuffle_epi8(m, mask); - tag = ghash_add(tag, gcm->htable, m);+ tag = gcm_ghash_add(tag, gcm->htable, m); /* make output */ _mm_storeu_si128((__m128i *) &block.b, m);+ memcpy(output, &block.b, part_block_len);+ }+ /* store back IV & tag */+ __m128i tmp = _mm_shuffle_epi8(iv, bswap_mask);+ _mm_storeu_si128((__m128i *) &gcm->civ, tmp);+ _mm_storeu_si128((__m128i *) &gcm->tag, tag);+}++/*+ * GCM decryption, which until now fell to the generic loop: that advances+ * the counter and calls the block function once per block through the+ * branch table, and measured a quarter the speed of encryption on the same+ * machine. The shape is the encryption loop with two differences -- the+ * tag is taken over the ciphertext, which is the input rather than the+ * output, and the ciphertext is read before anything is written, since+ * output may be input.+ */+GCM_TARGET+void SIZED(crypton_aesni_gcm_decrypt)(uint8_t *output, aes_gcm *gcm, aes_key *key, uint8_t *input, uint32_t length)+{+ __m128i *k = (__m128i *) key->data;+ __m128i bswap_mask = _mm_setr_epi8(7,6,5,4,3,2,1,0,15,14,13,12,11,10,9,8);+ __m128i one = _mm_set_epi32(0,1,0,0);+ uint32_t nb_blocks = length / 16;+ uint32_t part_block_len = length % 16;+ /* the group of ciphertext whose GHASH has not been taken yet */+ __m128i pending[8];+ int held = 0;++ gcm->length_input += length;++#ifdef WITH_GCM_VAES512+ /* as in encryption; the tag is taken over the input here */+ if (nb_blocks >= GCM_VAES512_MIN_BLOCKS+ && (crypton_x86_simd_features() & CRYPTON_X86_VAES512)) {+ uint32_t done = crypton_gcm_vaes512_bulk_decrypt(+ output, gcm, key, input, nb_blocks * 16);++ output += done;+ input += done;+ nb_blocks -= done / 16;+ }+#endif+#ifdef WITH_GCM_VAES+ /* as in encryption; the tag is taken over the input here */+ if (nb_blocks >= GCM_VAES_MIN_BLOCKS+ && (crypton_x86_simd_features() & CRYPTON_X86_VAES)) {+ uint32_t done = crypton_gcm_vaes_bulk_decrypt(output, gcm, key,+ input,+ nb_blocks * 16);++ output += done;+ input += done;+ nb_blocks -= done / 16;+ }+#endif+#if defined(WITH_X86_GCM_ASM) && defined(WITH_PCLMUL)+ /* the same as encryption, except that decryption has nothing to+ * hold back and so will start on six blocks */+ if (nb_blocks >= GCM_ASM_MIN_BLOCKS_DEC) {+ uint32_t done = crypton_gcm_asm_bulk_decrypt(output, gcm, key,+ input, nb_blocks * 16);++ output += done;+ input += done;+ nb_blocks -= done / 16;+ }+#endif++ __m128i tag = _mm_loadu_si128((__m128i *) &gcm->tag);+ __m128i iv = _mm_loadu_si128((__m128i *) &gcm->civ);+ iv = _mm_shuffle_epi8(iv, bswap_mask);+++ /* the group before's GHASH, alongside this group's rounds, as+ * encryption does it */+ for (; nb_blocks >= 8; nb_blocks -= 8, output += 128, input += 128) {+ __m128i m[8], c[8];+ int i;++ for (i = 0; i < 8; i++) {+ /* iv += 1, put back in big endian */+ iv = _mm_add_epi32(iv, one);+ m[i] = _mm_shuffle_epi8(iv, bswap_mask);+ }+ for (i = 0; i < 8; i++)+ c[i] = _mm_loadu_si128((__m128i *) (input + 16 * i));+ if (held)+ GCM_GROUP8(m, k, NBR, ROUNDS8_EXTRA);+ else+ DO_ENC_BLOCK8_MEM(m, k, NBR, ROUNDS8_EXTRA);++ for (i = 0; i < 8; i++)+ _mm_storeu_si128((__m128i *) (output + 16 * i),+ _mm_xor_si128(m[i], c[i]));+ for (i = 0; i < 8; i++)+ pending[i] = c[i];+ held = 1;+ }+ if (held)+ tag = gcm_ghash_add8(tag, gcm->htable, pending);+ for (; nb_blocks-- > 0; output += 16, input += 16) {+ __m128i c = _mm_loadu_si128((__m128i *) input);++ iv = _mm_add_epi32(iv, one);+ __m128i tmp = _mm_shuffle_epi8(iv, bswap_mask);+ DO_ENC_BLOCK_MEM(tmp, k, NBR);++ tag = gcm_ghash_add(tag, gcm->htable, c);+ _mm_storeu_si128((__m128i *) output, _mm_xor_si128(tmp, c));+ }+ if (part_block_len > 0) {+ aes_block block;++ /* the ciphertext padded with zeros is what the tag is taken+ * over, so no mask is needed the way encryption needs one */+ block128_zero(&block);+ block128_copy_bytes(&block, input, part_block_len);+ __m128i c = _mm_loadu_si128((__m128i *) &block);++ /* iv += 1 */+ iv = _mm_add_epi32(iv, one);++ __m128i tmp = _mm_shuffle_epi8(iv, bswap_mask);+ DO_ENC_BLOCK_MEM(tmp, k, NBR);++ tag = gcm_ghash_add(tag, gcm->htable, c);++ _mm_storeu_si128((__m128i *) &block.b, _mm_xor_si128(tmp, c)); memcpy(output, &block.b, part_block_len); } /* store back IV & tag */
@@ -0,0 +1,36 @@+Copyright (c) 2006, CRYPTOGAMS by <appro@openssl.org>+All rights reserved.++Redistribution and use in source and binary forms, with or without+modification, are permitted provided that the following conditions+are met:++ * Redistributions of source code must retain copyright notices,+ this list of conditions and the following disclaimer.++ * Redistributions in binary form must reproduce the above+ copyright notice, this list of conditions and the following+ disclaimer in the documentation and/or other materials+ provided with the distribution.++ * Neither the name of the CRYPTOGAMS nor the names of its+ copyright holder and contributors may be used to endorse or+ promote products derived from this software without specific+ prior written permission.++ALTERNATIVELY, provided that this notice is retained in full, this+product may be distributed under the terms of the GNU General Public+License (GPL), in which case the provisions of the GPL apply INSTEAD OF+those given above.++THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDER AND CONTRIBUTORS+"AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT+LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR+A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT+OWNER OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL,+SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT+LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE,+DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY+THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT+(INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE+OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
@@ -0,0 +1,171 @@+# Vendored assembly++## What is here++Two modules from [CRYPTOGAMS](https://github.com/dot-asm/cryptogams), by Andy+Polyakov, checked in unmodified together with the translators they need:++| generator | what it is |+| --- | --- |+| `aesni-gcm-x86_64.pl` | AES-NI/PCLMULQDQ stitched AES-GCM for x86-64 |+| `chacha-x86_64.pl` | ChaCha20 for x86-64 |+| `poly1305-x86_64.pl` | Poly1305 for x86-64 |+| `sha512-x86_64.pl` | SHA-256 and SHA-512 for x86-64 (one generator, two outputs, as on AArch64) |+| `keccak1600-x86_64.pl` | Keccak for x86-64 |+| `chacha-armv8.pl` | ChaCha20 for AArch64 |+| `poly1305-armv8.pl` | Poly1305 for AArch64 |+| `sha1-armv8.pl` | SHA-1 for AArch64 |+| `sha512-armv8.pl` | SHA-256 for AArch64 (the generator emits SHA-512 or SHA-256 according to the name it is given, and only the latter is wanted) |+| `keccak1600-armv8.pl` | Keccak for AArch64 |++`x86_64-xlate.pl`, `arm-xlate.pl` and `arm_arch.h` are the machinery those+modules use. `generate.sh` runs the generators to produce the `.S` files, which are+what crypton actually compiles -- one per object format, since the calling+convention and the assembler syntax differ. The names of the entry points are+changed on the way through, and the ELF output is given the note that says the+code does not want an executable stack.++Keeping the generated files in the tree means building crypton needs no perl.++## Why++Both do something the compiler will not do with intrinsics.++**AES-GCM.** Counter-mode AES and GHASH do not compete for the same execution+ports, so a loop that interleaves them at instruction granularity runs both in+the time one of them would take. Written in C the interleaving does not+survive: given the AES rounds and the multiplies of a group of blocks, GCC and+Clang schedule all the multiplies after all the rounds, which is the sum of the+two rather than the maximum.++**ChaCha20.** On AArch64 the vector registers hold four ChaCha states and there+is no room for a fifth, so further parallelism has to come from the integer+side: that module runs a fifth block through the general registers alongside+four in the vector ones, and above 512 bytes two alongside six. Which register+holds which word is the whole trick, and that is not something C says. The+x86-64 module is worth taking for a different reason -- it has vector code for+lengths the C here still takes a block at a time, so a 256-byte message more+than doubles -- and is a few per cent ahead in bulk besides.++**Poly1305.** One multiplication modulo 2^130 - 5 depends on the one before it,+so what there is to win is in how the multiplies and the carries are laid+against each other, and in keeping the accumulator in whichever base costs+less: these modules work in base 2^64 while the message is short and switch to+base 2^26 for the vector loop, which is a decision no compiler will make for+you. On AArch64 that is twice the speed of the C here at 16 KiB and three and+a half times at 64 bytes; on x86-64, a quarter faster at 16 KiB and nearly+three times at 64.++The x86-64 module also has paths for AVX-512, which are **not** taken. What+the generator emits is chosen from the version of the assembler it is told+about, and `generate.sh` tells it one that predates AVX-512: no machine here+can run those paths, an assembler old enough to be in use cannot always+assemble them, and a path nothing has executed is not worth the few per cent+it might be worth.++**SHA-1, SHA-256, SHA-512 and Keccak.** On AArch64 the instructions are the same ones the+intrinsics here already use. What the module does is schedule them across a+whole run of blocks instead of one at a time, and keep the message schedule of+the next block moving while the rounds of this one are still going, which a+per-block C function cannot do at all. A quarter faster, and it needs no+alignment and no copy since it reads the message as bytes. The x86-64 module+is the same idea with more paths to choose from -- the SHA extensions, AVX2,+AVX, SSSE3 -- and is about a fifth faster than the C on a machine with AVX2+and no SHA extensions. The AArch64 SHA-1 and Keccak modules are the same+story again -- the instructions are the ones the intrinsics here use, and what+the modules add is the arrangement: the schedule of the next four SHA-1 rounds+against the rounds of this one, and one Keccak round against the next. The+x86-64 Keccak is there for a different reason: nothing on that side has+instructions for this permutation, and what the module has over the C is that+its twenty-five lanes stay in registers across a round, where a compiler given+the C spills them. Two and a half times, and level with openssl.++## Interfaces++ size_t crypton_gcm_asm_encrypt(const void *in, void *out, size_t len,+ const void *key, unsigned char ivec[16],+ void *Xi);+ size_t crypton_gcm_asm_decrypt(... the same ...);++Both return the number of bytes processed, which is a multiple of 96 and may be+zero: encryption wants at least 288 bytes to start, decryption at least 96.+Whatever is left over is the caller's to finish.++`key` is the AES key schedule in the layout the OpenSSL assembly expects -- the+round keys, and at offset 240 one less than the number of rounds, which is what+OpenSSL's own AES-NI key setup puts there -- and `Xi` points at the running+GHASH state, with the table of powers of H, in the layout `gcm_init_avx` leaves+behind, 32 bytes past it. `cbits/aes/gcm_x86_asm.c` builds both, and the+multiplication that fills that table is written there in C rather than taken+from `ghash-x86_64.pl`: it runs once per message, so it is not worth a second+vendored file, and the one in OpenSSL is under a licence this package does not+use.++The code needs AES-NI, PCLMULQDQ, AVX and MOVBE, which+`crypton_x86_simd_features()` is asked about before any of it is called.++ void crypton_chacha20_ctr32(unsigned char *out, const unsigned char *in,+ size_t len, const unsigned int key[8],+ const unsigned int counter[4]);++Twenty rounds, the constants that go with a 256-bit key, and a 32-bit counter+which it does not write back: the caller advances it by the number of blocks.+Any length is accepted; the AArch64 module's vector path starts at 192 bytes,+the x86-64 one's rather lower. They ask `crypton_armcap_P` and+`crypton_ia32cap_P` respectively what the processor has, and+`cbits/crypton_chacha.c` calls them only for the states they fit -- twenty+rounds, a 256-bit key, and only as many blocks as the 32-bit counter has room+for.++ int crypton_poly1305_asm_init(void *ctx, const unsigned char key[16],+ void *func[2]);+ void crypton_poly1305_asm_blocks(void *ctx, const unsigned char *inp,+ size_t len, unsigned int padbit);+ void crypton_poly1305_asm_emit(void *ctx, unsigned char mac[16],+ const unsigned int nonce[4]);++`ctx` is 192 bytes of state the module keeps for itself -- its accumulator, the+clamped key and the powers of it -- and `key` is the first half of the Poly1305+key, the second half being handed to `emit` as `nonce`. `padbit` is the bit+above each block, set for the blocks of the message and clear for the padded+last one. `len` is a whole number of blocks.++Initialisation hands back through `func` the pair of functions its own dispatch+would use, the vector entry point not being exported, and+`cbits/crypton_poly1305.c` calls those. It reads `crypton_armcap_P` to choose+between them; `cbits/crypton_cpu.c` defines that.++The x86-64 Poly1305 module presents the same three functions, and reads+`crypton_ia32cap_P` -- cpuid's own words, in the order OpenSSL keeps them --+where the AArch64 one reads `crypton_armcap_P`. `cbits/crypton_cpu.c` fills+it, with the bits for anything the operating system will not preserve cleared,+and the AVX-512 ones cleared whatever the processor says.++ void crypton_sha1_asm_block_data_order(unsigned int state[5],+ const void *data, size_t blocks);+ void crypton_sha256_asm_block_data_order(unsigned int state[8],+ const void *data, size_t blocks);+ size_t crypton_keccak_asm_absorb_cext(unsigned long long state[25],+ const void *inp, size_t len,+ size_t bsz);++The state is the words of the digest in host order and `blocks` whole blocks of+64 bytes. Each entry point picks between the SHA-2 instructions, NEON and+plain integer code from `crypton_armcap_P`, whose SHA-1 and SHA-256 bits+`cbits/crypton_sha1.c` and `cbits/crypton_sha256.c` set once they have asked+the operating system whether the processor has them -- they are optional in+ARMv8.0.++Keccak's absorb takes the state as its twenty-five lanes, `bsz` as the rate in+bytes, and answers with what was left over. On AArch64 the `_cext` entry point+is the one that uses the SHA-3 instructions, and `cbits/crypton_sha3.c` calls+it only where its own runtime check has found them; the x86-64 module asks+nothing of the processor beyond the baseline and is called wherever it is+compiled in.++## Licence++`LICENSE.cryptogams` is the licence the CRYPTOGAMS files are distributed under.+It is the three-clause BSD licence, with the GNU General Public Licence offered+as an alternative; crypton takes the former, which is the licence of the rest+of this package.
@@ -0,0 +1,814 @@+.text ++.type _crypton_gcm_asm_ctr32_ghash_6x,@function+.align 32+_crypton_gcm_asm_ctr32_ghash_6x:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa+++ vmovdqu 32(%r11),%xmm2+ subq $6,%rdx+ vpxor %xmm4,%xmm4,%xmm4+ vmovdqu 0-128(%rcx),%xmm15+ vpaddb %xmm2,%xmm1,%xmm10+ vpaddb %xmm2,%xmm10,%xmm11+ vpaddb %xmm2,%xmm11,%xmm12+ vpaddb %xmm2,%xmm12,%xmm13+ vpaddb %xmm2,%xmm13,%xmm14+ vpxor %xmm15,%xmm1,%xmm9+ vmovdqu %xmm4,16+8(%rsp)+ jmp .Loop6x++.align 32+.Loop6x:+ addl $100663296,%ebx+ jc .Lhandle_ctr32+ vmovdqu 0-32(%r9),%xmm3+ vpaddb %xmm2,%xmm14,%xmm1+ vpxor %xmm15,%xmm10,%xmm10+ vpxor %xmm15,%xmm11,%xmm11++.Lresume_ctr32:+ vmovdqu %xmm1,(%r8)+ vpclmulqdq $0x10,%xmm3,%xmm7,%xmm5+ vpxor %xmm15,%xmm12,%xmm12+ vmovups 16-128(%rcx),%xmm2+ vpclmulqdq $0x01,%xmm3,%xmm7,%xmm6+ xorq %r12,%r12+ cmpq %r14,%r15++ vaesenc %xmm2,%xmm9,%xmm9+ vmovdqu 48+8(%rsp),%xmm0+ vpxor %xmm15,%xmm13,%xmm13+ vpclmulqdq $0x00,%xmm3,%xmm7,%xmm1+ vaesenc %xmm2,%xmm10,%xmm10+ vpxor %xmm15,%xmm14,%xmm14+ setnc %r12b+ vpclmulqdq $0x11,%xmm3,%xmm7,%xmm7+ vaesenc %xmm2,%xmm11,%xmm11+ vmovdqu 16-32(%r9),%xmm3+ negq %r12+ vaesenc %xmm2,%xmm12,%xmm12+ vpxor %xmm5,%xmm6,%xmm6+ vpclmulqdq $0x00,%xmm3,%xmm0,%xmm5+ vpxor %xmm4,%xmm8,%xmm8+ vaesenc %xmm2,%xmm13,%xmm13+ vpxor %xmm5,%xmm1,%xmm4+ andq $0x60,%r12+ vmovups 32-128(%rcx),%xmm15+ vpclmulqdq $0x10,%xmm3,%xmm0,%xmm1+ vaesenc %xmm2,%xmm14,%xmm14++ vpclmulqdq $0x01,%xmm3,%xmm0,%xmm2+ leaq (%r14,%r12,1),%r14+ vaesenc %xmm15,%xmm9,%xmm9+ vpxor 16+8(%rsp),%xmm8,%xmm8+ vpclmulqdq $0x11,%xmm3,%xmm0,%xmm3+ vmovdqu 64+8(%rsp),%xmm0+ vaesenc %xmm15,%xmm10,%xmm10+ movbeq 88(%r14),%r13+ vaesenc %xmm15,%xmm11,%xmm11+ movbeq 80(%r14),%r12+ vaesenc %xmm15,%xmm12,%xmm12+ movq %r13,32+8(%rsp)+ vaesenc %xmm15,%xmm13,%xmm13+ movq %r12,40+8(%rsp)+ vmovdqu 48-32(%r9),%xmm5+ vaesenc %xmm15,%xmm14,%xmm14++ vmovups 48-128(%rcx),%xmm15+ vpxor %xmm1,%xmm6,%xmm6+ vpclmulqdq $0x00,%xmm5,%xmm0,%xmm1+ vaesenc %xmm15,%xmm9,%xmm9+ vpxor %xmm2,%xmm6,%xmm6+ vpclmulqdq $0x10,%xmm5,%xmm0,%xmm2+ vaesenc %xmm15,%xmm10,%xmm10+ vpxor %xmm3,%xmm7,%xmm7+ vpclmulqdq $0x01,%xmm5,%xmm0,%xmm3+ vaesenc %xmm15,%xmm11,%xmm11+ vpclmulqdq $0x11,%xmm5,%xmm0,%xmm5+ vmovdqu 80+8(%rsp),%xmm0+ vaesenc %xmm15,%xmm12,%xmm12+ vaesenc %xmm15,%xmm13,%xmm13+ vpxor %xmm1,%xmm4,%xmm4+ vmovdqu 64-32(%r9),%xmm1+ vaesenc %xmm15,%xmm14,%xmm14++ vmovups 64-128(%rcx),%xmm15+ vpxor %xmm2,%xmm6,%xmm6+ vpclmulqdq $0x00,%xmm1,%xmm0,%xmm2+ vaesenc %xmm15,%xmm9,%xmm9+ vpxor %xmm3,%xmm6,%xmm6+ vpclmulqdq $0x10,%xmm1,%xmm0,%xmm3+ vaesenc %xmm15,%xmm10,%xmm10+ movbeq 72(%r14),%r13+ vpxor %xmm5,%xmm7,%xmm7+ vpclmulqdq $0x01,%xmm1,%xmm0,%xmm5+ vaesenc %xmm15,%xmm11,%xmm11+ movbeq 64(%r14),%r12+ vpclmulqdq $0x11,%xmm1,%xmm0,%xmm1+ vmovdqu 96+8(%rsp),%xmm0+ vaesenc %xmm15,%xmm12,%xmm12+ movq %r13,48+8(%rsp)+ vaesenc %xmm15,%xmm13,%xmm13+ movq %r12,56+8(%rsp)+ vpxor %xmm2,%xmm4,%xmm4+ vmovdqu 96-32(%r9),%xmm2+ vaesenc %xmm15,%xmm14,%xmm14++ vmovups 80-128(%rcx),%xmm15+ vpxor %xmm3,%xmm6,%xmm6+ vpclmulqdq $0x00,%xmm2,%xmm0,%xmm3+ vaesenc %xmm15,%xmm9,%xmm9+ vpxor %xmm5,%xmm6,%xmm6+ vpclmulqdq $0x10,%xmm2,%xmm0,%xmm5+ vaesenc %xmm15,%xmm10,%xmm10+ movbeq 56(%r14),%r13+ vpxor %xmm1,%xmm7,%xmm7+ vpclmulqdq $0x01,%xmm2,%xmm0,%xmm1+ vpxor 112+8(%rsp),%xmm8,%xmm8+ vaesenc %xmm15,%xmm11,%xmm11+ movbeq 48(%r14),%r12+ vpclmulqdq $0x11,%xmm2,%xmm0,%xmm2+ vaesenc %xmm15,%xmm12,%xmm12+ movq %r13,64+8(%rsp)+ vaesenc %xmm15,%xmm13,%xmm13+ movq %r12,72+8(%rsp)+ vpxor %xmm3,%xmm4,%xmm4+ vmovdqu 112-32(%r9),%xmm3+ vaesenc %xmm15,%xmm14,%xmm14++ vmovups 96-128(%rcx),%xmm15+ vpxor %xmm5,%xmm6,%xmm6+ vpclmulqdq $0x10,%xmm3,%xmm8,%xmm5+ vaesenc %xmm15,%xmm9,%xmm9+ vpxor %xmm1,%xmm6,%xmm6+ vpclmulqdq $0x01,%xmm3,%xmm8,%xmm1+ vaesenc %xmm15,%xmm10,%xmm10+ movbeq 40(%r14),%r13+ vpxor %xmm2,%xmm7,%xmm7+ vpclmulqdq $0x00,%xmm3,%xmm8,%xmm2+ vaesenc %xmm15,%xmm11,%xmm11+ movbeq 32(%r14),%r12+ vpclmulqdq $0x11,%xmm3,%xmm8,%xmm8+ vaesenc %xmm15,%xmm12,%xmm12+ movq %r13,80+8(%rsp)+ vaesenc %xmm15,%xmm13,%xmm13+ movq %r12,88+8(%rsp)+ vpxor %xmm5,%xmm6,%xmm6+ vaesenc %xmm15,%xmm14,%xmm14+ vpxor %xmm1,%xmm6,%xmm6++ vmovups 112-128(%rcx),%xmm15+ vpslldq $8,%xmm6,%xmm5+ vpxor %xmm2,%xmm4,%xmm4+ vmovdqu 16(%r11),%xmm3++ vaesenc %xmm15,%xmm9,%xmm9+ vpxor %xmm8,%xmm7,%xmm7+ vaesenc %xmm15,%xmm10,%xmm10+ vpxor %xmm5,%xmm4,%xmm4+ movbeq 24(%r14),%r13+ vaesenc %xmm15,%xmm11,%xmm11+ movbeq 16(%r14),%r12+ vpalignr $8,%xmm4,%xmm4,%xmm0+ vpclmulqdq $0x10,%xmm3,%xmm4,%xmm4+ movq %r13,96+8(%rsp)+ vaesenc %xmm15,%xmm12,%xmm12+ movq %r12,104+8(%rsp)+ vaesenc %xmm15,%xmm13,%xmm13+ vmovups 128-128(%rcx),%xmm1+ vaesenc %xmm15,%xmm14,%xmm14++ vaesenc %xmm1,%xmm9,%xmm9+ vmovups 144-128(%rcx),%xmm15+ vaesenc %xmm1,%xmm10,%xmm10+ vpsrldq $8,%xmm6,%xmm6+ vaesenc %xmm1,%xmm11,%xmm11+ vpxor %xmm6,%xmm7,%xmm7+ vaesenc %xmm1,%xmm12,%xmm12+ vpxor %xmm0,%xmm4,%xmm4+ movbeq 8(%r14),%r13+ vaesenc %xmm1,%xmm13,%xmm13+ movbeq 0(%r14),%r12+ vaesenc %xmm1,%xmm14,%xmm14+ vmovups 160-128(%rcx),%xmm1+ cmpl $11,%r10d+ jb .Lenc_tail++ vaesenc %xmm15,%xmm9,%xmm9+ vaesenc %xmm15,%xmm10,%xmm10+ vaesenc %xmm15,%xmm11,%xmm11+ vaesenc %xmm15,%xmm12,%xmm12+ vaesenc %xmm15,%xmm13,%xmm13+ vaesenc %xmm15,%xmm14,%xmm14++ vaesenc %xmm1,%xmm9,%xmm9+ vaesenc %xmm1,%xmm10,%xmm10+ vaesenc %xmm1,%xmm11,%xmm11+ vaesenc %xmm1,%xmm12,%xmm12+ vaesenc %xmm1,%xmm13,%xmm13+ vmovups 176-128(%rcx),%xmm15+ vaesenc %xmm1,%xmm14,%xmm14+ vmovups 192-128(%rcx),%xmm1+ je .Lenc_tail++ vaesenc %xmm15,%xmm9,%xmm9+ vaesenc %xmm15,%xmm10,%xmm10+ vaesenc %xmm15,%xmm11,%xmm11+ vaesenc %xmm15,%xmm12,%xmm12+ vaesenc %xmm15,%xmm13,%xmm13+ vaesenc %xmm15,%xmm14,%xmm14++ vaesenc %xmm1,%xmm9,%xmm9+ vaesenc %xmm1,%xmm10,%xmm10+ vaesenc %xmm1,%xmm11,%xmm11+ vaesenc %xmm1,%xmm12,%xmm12+ vaesenc %xmm1,%xmm13,%xmm13+ vmovups 208-128(%rcx),%xmm15+ vaesenc %xmm1,%xmm14,%xmm14+ vmovups 224-128(%rcx),%xmm1+ jmp .Lenc_tail++.align 32+.Lhandle_ctr32:+ vmovdqu (%r11),%xmm0+ vpshufb %xmm0,%xmm1,%xmm6+ vmovdqu 48(%r11),%xmm5+ vpaddd 64(%r11),%xmm6,%xmm10+ vpaddd %xmm5,%xmm6,%xmm11+ vmovdqu 0-32(%r9),%xmm3+ vpaddd %xmm5,%xmm10,%xmm12+ vpshufb %xmm0,%xmm10,%xmm10+ vpaddd %xmm5,%xmm11,%xmm13+ vpshufb %xmm0,%xmm11,%xmm11+ vpxor %xmm15,%xmm10,%xmm10+ vpaddd %xmm5,%xmm12,%xmm14+ vpshufb %xmm0,%xmm12,%xmm12+ vpxor %xmm15,%xmm11,%xmm11+ vpaddd %xmm5,%xmm13,%xmm1+ vpshufb %xmm0,%xmm13,%xmm13+ vpshufb %xmm0,%xmm14,%xmm14+ vpshufb %xmm0,%xmm1,%xmm1+ jmp .Lresume_ctr32++.align 32+.Lenc_tail:+ vaesenc %xmm15,%xmm9,%xmm9+ vmovdqu %xmm7,16+8(%rsp)+ vpalignr $8,%xmm4,%xmm4,%xmm8+ vaesenc %xmm15,%xmm10,%xmm10+ vpclmulqdq $0x10,%xmm3,%xmm4,%xmm4+ vpxor 0(%rdi),%xmm1,%xmm2+ vaesenc %xmm15,%xmm11,%xmm11+ vpxor 16(%rdi),%xmm1,%xmm0+ vaesenc %xmm15,%xmm12,%xmm12+ vpxor 32(%rdi),%xmm1,%xmm5+ vaesenc %xmm15,%xmm13,%xmm13+ vpxor 48(%rdi),%xmm1,%xmm6+ vaesenc %xmm15,%xmm14,%xmm14+ vpxor 64(%rdi),%xmm1,%xmm7+ vpxor 80(%rdi),%xmm1,%xmm3+ vmovdqu (%r8),%xmm1++ vaesenclast %xmm2,%xmm9,%xmm9+ vmovdqu 32(%r11),%xmm2+ vaesenclast %xmm0,%xmm10,%xmm10+ vpaddb %xmm2,%xmm1,%xmm0+ movq %r13,112+8(%rsp)+ leaq 96(%rdi),%rdi+ vaesenclast %xmm5,%xmm11,%xmm11+ vpaddb %xmm2,%xmm0,%xmm5+ movq %r12,120+8(%rsp)+ leaq 96(%rsi),%rsi+ vmovdqu 0-128(%rcx),%xmm15+ vaesenclast %xmm6,%xmm12,%xmm12+ vpaddb %xmm2,%xmm5,%xmm6+ vaesenclast %xmm7,%xmm13,%xmm13+ vpaddb %xmm2,%xmm6,%xmm7+ vaesenclast %xmm3,%xmm14,%xmm14+ vpaddb %xmm2,%xmm7,%xmm3++ addq $0x60,%rax+ subq $0x6,%rdx+ jc .L6x_done++ vmovups %xmm9,-96(%rsi)+ vpxor %xmm15,%xmm1,%xmm9+ vmovups %xmm10,-80(%rsi)+ vmovdqa %xmm0,%xmm10+ vmovups %xmm11,-64(%rsi)+ vmovdqa %xmm5,%xmm11+ vmovups %xmm12,-48(%rsi)+ vmovdqa %xmm6,%xmm12+ vmovups %xmm13,-32(%rsi)+ vmovdqa %xmm7,%xmm13+ vmovups %xmm14,-16(%rsi)+ vmovdqa %xmm3,%xmm14+ vmovdqu 32+8(%rsp),%xmm7+ jmp .Loop6x++.L6x_done:+ vpxor 16+8(%rsp),%xmm8,%xmm8+ vpxor %xmm4,%xmm8,%xmm8++ .byte 0xf3,0xc3+.cfi_endproc +.size _crypton_gcm_asm_ctr32_ghash_6x,.-_crypton_gcm_asm_ctr32_ghash_6x+.globl crypton_gcm_asm_decrypt+.type crypton_gcm_asm_decrypt,@function+.align 32+crypton_gcm_asm_decrypt:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa+++ pushq %rbp+.cfi_adjust_cfa_offset 8+.cfi_offset %rbp,-16+ movq %rsp,%rbp+.cfi_def_cfa_register %rbp+ xorq %rax,%rax+ cmpq $0x60,%rdx+ jb .Lgcm_dec_abort++ pushq %rbx+.cfi_offset %rbx,-24+ pushq %r12+.cfi_offset %r12,-32+ pushq %r13+.cfi_offset %r13,-40+ pushq %r14+.cfi_offset %r14,-48+ pushq %r15+.cfi_offset %r15,-56++ vzeroupper++ vmovdqu (%r8),%xmm1+ addq $-128,%rsp+ movl 12(%r8),%ebx+ leaq .Lbswap_mask(%rip),%r11+ leaq -128(%rcx),%r14+ movq $0xf80,%r15+ vmovdqu (%r9),%xmm8+ andq $-128,%rsp+ vmovdqu (%r11),%xmm0+ leaq 128(%rcx),%rcx+ leaq 32+32(%r9),%r9+ movl 240-128(%rcx),%r10d+ vpshufb %xmm0,%xmm8,%xmm8++ andq %r15,%r14+ andq %rsp,%r15+ subq %r14,%r15+ jc .Ldec_no_key_aliasing+ cmpq $768,%r15+ jnc .Ldec_no_key_aliasing+ subq %r15,%rsp+.Ldec_no_key_aliasing:++ vmovdqu 80(%rdi),%xmm7+ leaq (%rdi),%r14+ vmovdqu 64(%rdi),%xmm4+ leaq -192(%rdi,%rdx,1),%r15+ vmovdqu 48(%rdi),%xmm5+ shrq $4,%rdx+ xorq %rax,%rax+ vmovdqu 32(%rdi),%xmm6+ vpshufb %xmm0,%xmm7,%xmm7+ vmovdqu 16(%rdi),%xmm2+ vpshufb %xmm0,%xmm4,%xmm4+ vmovdqu (%rdi),%xmm3+ vpshufb %xmm0,%xmm5,%xmm5+ vmovdqu %xmm4,48(%rsp)+ vpshufb %xmm0,%xmm6,%xmm6+ vmovdqu %xmm5,64(%rsp)+ vpshufb %xmm0,%xmm2,%xmm2+ vmovdqu %xmm6,80(%rsp)+ vpshufb %xmm0,%xmm3,%xmm3+ vmovdqu %xmm2,96(%rsp)+ vmovdqu %xmm3,112(%rsp)++ call _crypton_gcm_asm_ctr32_ghash_6x++ vmovups %xmm9,-96(%rsi)+ vmovups %xmm10,-80(%rsi)+ vmovups %xmm11,-64(%rsi)+ vmovups %xmm12,-48(%rsi)+ vmovups %xmm13,-32(%rsi)+ vmovups %xmm14,-16(%rsi)++ vpshufb (%r11),%xmm8,%xmm8+ vmovdqu %xmm8,-64(%r9)++ vzeroupper+ movq -40(%rbp),%r15+ movq -32(%rbp),%r14+ movq -24(%rbp),%r13+ movq -16(%rbp),%r12+ movq -8(%rbp),%rbx+ movq %rbp,%rsp+.cfi_def_cfa_register %rsp+.Lgcm_dec_abort:+ popq %rbp+.cfi_adjust_cfa_offset -8+.cfi_restore %rbp+.cfi_restore %r12+.cfi_restore %r13+.cfi_restore %r14+.cfi_restore %r15+.cfi_restore %rbx+ .byte 0xf3,0xc3+.cfi_endproc +.size crypton_gcm_asm_decrypt,.-crypton_gcm_asm_decrypt+.type _crypton_gcm_asm_ctr32_6x,@function+.align 32+_crypton_gcm_asm_ctr32_6x:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa+++ vmovdqu 0-128(%rcx),%xmm4+ vmovdqu 32(%r11),%xmm2+ leaq -1(%r10),%r13+ vmovups 16-128(%rcx),%xmm15+ leaq 32-128(%rcx),%r12+ vpxor %xmm4,%xmm1,%xmm9+ addl $100663296,%ebx+ jc .Lhandle_ctr32_2+ vpaddb %xmm2,%xmm1,%xmm10+ vpaddb %xmm2,%xmm10,%xmm11+ vpxor %xmm4,%xmm10,%xmm10+ vpaddb %xmm2,%xmm11,%xmm12+ vpxor %xmm4,%xmm11,%xmm11+ vpaddb %xmm2,%xmm12,%xmm13+ vpxor %xmm4,%xmm12,%xmm12+ vpaddb %xmm2,%xmm13,%xmm14+ vpxor %xmm4,%xmm13,%xmm13+ vpaddb %xmm2,%xmm14,%xmm1+ vpxor %xmm4,%xmm14,%xmm14+ jmp .Loop_ctr32++.align 16+.Loop_ctr32:+ vaesenc %xmm15,%xmm9,%xmm9+ vaesenc %xmm15,%xmm10,%xmm10+ vaesenc %xmm15,%xmm11,%xmm11+ vaesenc %xmm15,%xmm12,%xmm12+ vaesenc %xmm15,%xmm13,%xmm13+ vaesenc %xmm15,%xmm14,%xmm14+ vmovups (%r12),%xmm15+ leaq 16(%r12),%r12+ decl %r13d+ jnz .Loop_ctr32++ vmovdqu (%r12),%xmm3+ vaesenc %xmm15,%xmm9,%xmm9+ vpxor 0(%rdi),%xmm3,%xmm4+ vaesenc %xmm15,%xmm10,%xmm10+ vpxor 16(%rdi),%xmm3,%xmm5+ vaesenc %xmm15,%xmm11,%xmm11+ vpxor 32(%rdi),%xmm3,%xmm6+ vaesenc %xmm15,%xmm12,%xmm12+ vpxor 48(%rdi),%xmm3,%xmm8+ vaesenc %xmm15,%xmm13,%xmm13+ vpxor 64(%rdi),%xmm3,%xmm2+ vaesenc %xmm15,%xmm14,%xmm14+ vpxor 80(%rdi),%xmm3,%xmm3+ leaq 96(%rdi),%rdi++ vaesenclast %xmm4,%xmm9,%xmm9+ vaesenclast %xmm5,%xmm10,%xmm10+ vaesenclast %xmm6,%xmm11,%xmm11+ vaesenclast %xmm8,%xmm12,%xmm12+ vaesenclast %xmm2,%xmm13,%xmm13+ vaesenclast %xmm3,%xmm14,%xmm14+ vmovups %xmm9,0(%rsi)+ vmovups %xmm10,16(%rsi)+ vmovups %xmm11,32(%rsi)+ vmovups %xmm12,48(%rsi)+ vmovups %xmm13,64(%rsi)+ vmovups %xmm14,80(%rsi)+ leaq 96(%rsi),%rsi++ .byte 0xf3,0xc3+.align 32+.Lhandle_ctr32_2:+ vpshufb %xmm0,%xmm1,%xmm6+ vmovdqu 48(%r11),%xmm5+ vpaddd 64(%r11),%xmm6,%xmm10+ vpaddd %xmm5,%xmm6,%xmm11+ vpaddd %xmm5,%xmm10,%xmm12+ vpshufb %xmm0,%xmm10,%xmm10+ vpaddd %xmm5,%xmm11,%xmm13+ vpshufb %xmm0,%xmm11,%xmm11+ vpxor %xmm4,%xmm10,%xmm10+ vpaddd %xmm5,%xmm12,%xmm14+ vpshufb %xmm0,%xmm12,%xmm12+ vpxor %xmm4,%xmm11,%xmm11+ vpaddd %xmm5,%xmm13,%xmm1+ vpshufb %xmm0,%xmm13,%xmm13+ vpxor %xmm4,%xmm12,%xmm12+ vpshufb %xmm0,%xmm14,%xmm14+ vpxor %xmm4,%xmm13,%xmm13+ vpshufb %xmm0,%xmm1,%xmm1+ vpxor %xmm4,%xmm14,%xmm14+ jmp .Loop_ctr32+.cfi_endproc +.size _crypton_gcm_asm_ctr32_6x,.-_crypton_gcm_asm_ctr32_6x++.globl crypton_gcm_asm_encrypt+.type crypton_gcm_asm_encrypt,@function+.align 32+crypton_gcm_asm_encrypt:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa+++ pushq %rbp+.cfi_adjust_cfa_offset 8+.cfi_offset %rbp,-16+ movq %rsp,%rbp+.cfi_def_cfa_register %rbp+ xorq %rax,%rax+ cmpq $288,%rdx+ jb .Lgcm_enc_abort++ pushq %rbx+.cfi_offset %rbx,-24+ pushq %r12+.cfi_offset %r12,-32+ pushq %r13+.cfi_offset %r13,-40+ pushq %r14+.cfi_offset %r14,-48+ pushq %r15+.cfi_offset %r15,-56++ vzeroupper++ vmovdqu (%r8),%xmm1+ addq $-128,%rsp+ movl 12(%r8),%ebx+ leaq .Lbswap_mask(%rip),%r11+ leaq -128(%rcx),%r14+ movq $0xf80,%r15+ leaq 128(%rcx),%rcx+ vmovdqu (%r11),%xmm0+ andq $-128,%rsp+ movl 240-128(%rcx),%r10d++ andq %r15,%r14+ andq %rsp,%r15+ subq %r14,%r15+ jc .Lenc_no_key_aliasing+ cmpq $768,%r15+ jnc .Lenc_no_key_aliasing+ subq %r15,%rsp+.Lenc_no_key_aliasing:++ leaq (%rsi),%r14+ leaq -192(%rsi,%rdx,1),%r15+ shrq $4,%rdx++ call _crypton_gcm_asm_ctr32_6x+ vpshufb %xmm0,%xmm9,%xmm8+ vpshufb %xmm0,%xmm10,%xmm2+ vmovdqu %xmm8,112(%rsp)+ vpshufb %xmm0,%xmm11,%xmm4+ vmovdqu %xmm2,96(%rsp)+ vpshufb %xmm0,%xmm12,%xmm5+ vmovdqu %xmm4,80(%rsp)+ vpshufb %xmm0,%xmm13,%xmm6+ vmovdqu %xmm5,64(%rsp)+ vpshufb %xmm0,%xmm14,%xmm7+ vmovdqu %xmm6,48(%rsp)++ call _crypton_gcm_asm_ctr32_6x++ vmovdqu (%r9),%xmm8+ leaq 32+32(%r9),%r9+ subq $12,%rdx+ movq $192,%rax+ vpshufb %xmm0,%xmm8,%xmm8++ call _crypton_gcm_asm_ctr32_ghash_6x+ vmovdqu 32(%rsp),%xmm7+ vmovdqu (%r11),%xmm0+ vmovdqu 0-32(%r9),%xmm3+ vpunpckhqdq %xmm7,%xmm7,%xmm1+ vmovdqu 32-32(%r9),%xmm15+ vmovups %xmm9,-96(%rsi)+ vpshufb %xmm0,%xmm9,%xmm9+ vpxor %xmm7,%xmm1,%xmm1+ vmovups %xmm10,-80(%rsi)+ vpshufb %xmm0,%xmm10,%xmm10+ vmovups %xmm11,-64(%rsi)+ vpshufb %xmm0,%xmm11,%xmm11+ vmovups %xmm12,-48(%rsi)+ vpshufb %xmm0,%xmm12,%xmm12+ vmovups %xmm13,-32(%rsi)+ vpshufb %xmm0,%xmm13,%xmm13+ vmovups %xmm14,-16(%rsi)+ vpshufb %xmm0,%xmm14,%xmm14+ vmovdqu %xmm9,16(%rsp)+ vmovdqu 48(%rsp),%xmm6+ vmovdqu 16-32(%r9),%xmm0+ vpunpckhqdq %xmm6,%xmm6,%xmm2+ vpclmulqdq $0x00,%xmm3,%xmm7,%xmm5+ vpxor %xmm6,%xmm2,%xmm2+ vpclmulqdq $0x11,%xmm3,%xmm7,%xmm7+ vpclmulqdq $0x00,%xmm15,%xmm1,%xmm1++ vmovdqu 64(%rsp),%xmm9+ vpclmulqdq $0x00,%xmm0,%xmm6,%xmm4+ vmovdqu 48-32(%r9),%xmm3+ vpxor %xmm5,%xmm4,%xmm4+ vpunpckhqdq %xmm9,%xmm9,%xmm5+ vpclmulqdq $0x11,%xmm0,%xmm6,%xmm6+ vpxor %xmm9,%xmm5,%xmm5+ vpxor %xmm7,%xmm6,%xmm6+ vpclmulqdq $0x10,%xmm15,%xmm2,%xmm2+ vmovdqu 80-32(%r9),%xmm15+ vpxor %xmm1,%xmm2,%xmm2++ vmovdqu 80(%rsp),%xmm1+ vpclmulqdq $0x00,%xmm3,%xmm9,%xmm7+ vmovdqu 64-32(%r9),%xmm0+ vpxor %xmm4,%xmm7,%xmm7+ vpunpckhqdq %xmm1,%xmm1,%xmm4+ vpclmulqdq $0x11,%xmm3,%xmm9,%xmm9+ vpxor %xmm1,%xmm4,%xmm4+ vpxor %xmm6,%xmm9,%xmm9+ vpclmulqdq $0x00,%xmm15,%xmm5,%xmm5+ vpxor %xmm2,%xmm5,%xmm5++ vmovdqu 96(%rsp),%xmm2+ vpclmulqdq $0x00,%xmm0,%xmm1,%xmm6+ vmovdqu 96-32(%r9),%xmm3+ vpxor %xmm7,%xmm6,%xmm6+ vpunpckhqdq %xmm2,%xmm2,%xmm7+ vpclmulqdq $0x11,%xmm0,%xmm1,%xmm1+ vpxor %xmm2,%xmm7,%xmm7+ vpxor %xmm9,%xmm1,%xmm1+ vpclmulqdq $0x10,%xmm15,%xmm4,%xmm4+ vmovdqu 128-32(%r9),%xmm15+ vpxor %xmm5,%xmm4,%xmm4++ vpxor 112(%rsp),%xmm8,%xmm8+ vpclmulqdq $0x00,%xmm3,%xmm2,%xmm5+ vmovdqu 112-32(%r9),%xmm0+ vpunpckhqdq %xmm8,%xmm8,%xmm9+ vpxor %xmm6,%xmm5,%xmm5+ vpclmulqdq $0x11,%xmm3,%xmm2,%xmm2+ vpxor %xmm8,%xmm9,%xmm9+ vpxor %xmm1,%xmm2,%xmm2+ vpclmulqdq $0x00,%xmm15,%xmm7,%xmm7+ vpxor %xmm4,%xmm7,%xmm4++ vpclmulqdq $0x00,%xmm0,%xmm8,%xmm6+ vmovdqu 0-32(%r9),%xmm3+ vpunpckhqdq %xmm14,%xmm14,%xmm1+ vpclmulqdq $0x11,%xmm0,%xmm8,%xmm8+ vpxor %xmm14,%xmm1,%xmm1+ vpxor %xmm5,%xmm6,%xmm5+ vpclmulqdq $0x10,%xmm15,%xmm9,%xmm9+ vmovdqu 32-32(%r9),%xmm15+ vpxor %xmm2,%xmm8,%xmm7+ vpxor %xmm4,%xmm9,%xmm6++ vmovdqu 16-32(%r9),%xmm0+ vpxor %xmm5,%xmm7,%xmm9+ vpclmulqdq $0x00,%xmm3,%xmm14,%xmm4+ vpxor %xmm9,%xmm6,%xmm6+ vpunpckhqdq %xmm13,%xmm13,%xmm2+ vpclmulqdq $0x11,%xmm3,%xmm14,%xmm14+ vpxor %xmm13,%xmm2,%xmm2+ vpslldq $8,%xmm6,%xmm9+ vpclmulqdq $0x00,%xmm15,%xmm1,%xmm1+ vpxor %xmm9,%xmm5,%xmm8+ vpsrldq $8,%xmm6,%xmm6+ vpxor %xmm6,%xmm7,%xmm7++ vpclmulqdq $0x00,%xmm0,%xmm13,%xmm5+ vmovdqu 48-32(%r9),%xmm3+ vpxor %xmm4,%xmm5,%xmm5+ vpunpckhqdq %xmm12,%xmm12,%xmm9+ vpclmulqdq $0x11,%xmm0,%xmm13,%xmm13+ vpxor %xmm12,%xmm9,%xmm9+ vpxor %xmm14,%xmm13,%xmm13+ vpalignr $8,%xmm8,%xmm8,%xmm14+ vpclmulqdq $0x10,%xmm15,%xmm2,%xmm2+ vmovdqu 80-32(%r9),%xmm15+ vpxor %xmm1,%xmm2,%xmm2++ vpclmulqdq $0x00,%xmm3,%xmm12,%xmm4+ vmovdqu 64-32(%r9),%xmm0+ vpxor %xmm5,%xmm4,%xmm4+ vpunpckhqdq %xmm11,%xmm11,%xmm1+ vpclmulqdq $0x11,%xmm3,%xmm12,%xmm12+ vpxor %xmm11,%xmm1,%xmm1+ vpxor %xmm13,%xmm12,%xmm12+ vxorps 16(%rsp),%xmm7,%xmm7+ vpclmulqdq $0x00,%xmm15,%xmm9,%xmm9+ vpxor %xmm2,%xmm9,%xmm9++ vpclmulqdq $0x10,16(%r11),%xmm8,%xmm8+ vxorps %xmm14,%xmm8,%xmm8++ vpclmulqdq $0x00,%xmm0,%xmm11,%xmm5+ vmovdqu 96-32(%r9),%xmm3+ vpxor %xmm4,%xmm5,%xmm5+ vpunpckhqdq %xmm10,%xmm10,%xmm2+ vpclmulqdq $0x11,%xmm0,%xmm11,%xmm11+ vpxor %xmm10,%xmm2,%xmm2+ vpalignr $8,%xmm8,%xmm8,%xmm14+ vpxor %xmm12,%xmm11,%xmm11+ vpclmulqdq $0x10,%xmm15,%xmm1,%xmm1+ vmovdqu 128-32(%r9),%xmm15+ vpxor %xmm9,%xmm1,%xmm1++ vxorps %xmm7,%xmm14,%xmm14+ vpclmulqdq $0x10,16(%r11),%xmm8,%xmm8+ vxorps %xmm14,%xmm8,%xmm8++ vpclmulqdq $0x00,%xmm3,%xmm10,%xmm4+ vmovdqu 112-32(%r9),%xmm0+ vpxor %xmm5,%xmm4,%xmm4+ vpunpckhqdq %xmm8,%xmm8,%xmm9+ vpclmulqdq $0x11,%xmm3,%xmm10,%xmm10+ vpxor %xmm8,%xmm9,%xmm9+ vpxor %xmm11,%xmm10,%xmm10+ vpclmulqdq $0x00,%xmm15,%xmm2,%xmm2+ vpxor %xmm1,%xmm2,%xmm2++ vpclmulqdq $0x00,%xmm0,%xmm8,%xmm5+ vpclmulqdq $0x11,%xmm0,%xmm8,%xmm7+ vpxor %xmm4,%xmm5,%xmm5+ vpclmulqdq $0x10,%xmm15,%xmm9,%xmm6+ vpxor %xmm10,%xmm7,%xmm7+ vpxor %xmm2,%xmm6,%xmm6++ vpxor %xmm5,%xmm7,%xmm4+ vpxor %xmm4,%xmm6,%xmm6+ vpslldq $8,%xmm6,%xmm1+ vmovdqu 16(%r11),%xmm3+ vpsrldq $8,%xmm6,%xmm6+ vpxor %xmm1,%xmm5,%xmm8+ vpxor %xmm6,%xmm7,%xmm7++ vpalignr $8,%xmm8,%xmm8,%xmm2+ vpclmulqdq $0x10,%xmm3,%xmm8,%xmm8+ vpxor %xmm2,%xmm8,%xmm8++ vpalignr $8,%xmm8,%xmm8,%xmm2+ vpclmulqdq $0x10,%xmm3,%xmm8,%xmm8+ vpxor %xmm7,%xmm2,%xmm2+ vpxor %xmm2,%xmm8,%xmm8+ vpshufb (%r11),%xmm8,%xmm8+ vmovdqu %xmm8,-64(%r9)++ vzeroupper+ movq -40(%rbp),%r15+ movq -32(%rbp),%r14+ movq -24(%rbp),%r13+ movq -16(%rbp),%r12+ movq -8(%rbp),%rbx+ movq %rbp,%rsp+.cfi_def_cfa_register %rsp+.Lgcm_enc_abort:+ popq %rbp+.cfi_adjust_cfa_offset -8+.cfi_restore %rbp+.cfi_restore %r12+.cfi_restore %r13+.cfi_restore %r14+.cfi_restore %r15+.cfi_restore %rbx+ .byte 0xf3,0xc3+.cfi_endproc +.size crypton_gcm_asm_encrypt,.-crypton_gcm_asm_encrypt+.align 64+.Lbswap_mask:+.byte 15,14,13,12,11,10,9,8,7,6,5,4,3,2,1,0+.Lpoly:+.byte 0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0xc2+.Lone_msb:+.byte 0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,1+.Ltwo_lsb:+.byte 2,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0+.Lone_lsb:+.byte 1,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0+.byte 65,69,83,45,78,73,32,71,67,77,32,109,111,100,117,108,101,32,102,111,114,32,120,56,54,95,54,52,44,32,67,82,89,80,84,79,71,65,77,83,32,98,121,32,60,97,112,112,114,111,64,111,112,101,110,115,115,108,46,111,114,103,62,0+.align 64++.section .note.gnu.property,"a",@note+ .long 4,2f-1f,5+ .byte 0x47,0x4E,0x55,0+1: .long 0xc0000002,4,3+.align 8+2:++.section .note.GNU-stack,"",@progbits
@@ -0,0 +1,805 @@+.text +++.p2align 5+_crypton_gcm_asm_ctr32_ghash_6x:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa+++ vmovdqu 32(%r11),%xmm2+ subq $6,%rdx+ vpxor %xmm4,%xmm4,%xmm4+ vmovdqu 0-128(%rcx),%xmm15+ vpaddb %xmm2,%xmm1,%xmm10+ vpaddb %xmm2,%xmm10,%xmm11+ vpaddb %xmm2,%xmm11,%xmm12+ vpaddb %xmm2,%xmm12,%xmm13+ vpaddb %xmm2,%xmm13,%xmm14+ vpxor %xmm15,%xmm1,%xmm9+ vmovdqu %xmm4,16+8(%rsp)+ jmp L$oop6x++.p2align 5+L$oop6x:+ addl $100663296,%ebx+ jc L$handle_ctr32+ vmovdqu 0-32(%r9),%xmm3+ vpaddb %xmm2,%xmm14,%xmm1+ vpxor %xmm15,%xmm10,%xmm10+ vpxor %xmm15,%xmm11,%xmm11++L$resume_ctr32:+ vmovdqu %xmm1,(%r8)+ vpclmulqdq $0x10,%xmm3,%xmm7,%xmm5+ vpxor %xmm15,%xmm12,%xmm12+ vmovups 16-128(%rcx),%xmm2+ vpclmulqdq $0x01,%xmm3,%xmm7,%xmm6+ xorq %r12,%r12+ cmpq %r14,%r15++ vaesenc %xmm2,%xmm9,%xmm9+ vmovdqu 48+8(%rsp),%xmm0+ vpxor %xmm15,%xmm13,%xmm13+ vpclmulqdq $0x00,%xmm3,%xmm7,%xmm1+ vaesenc %xmm2,%xmm10,%xmm10+ vpxor %xmm15,%xmm14,%xmm14+ setnc %r12b+ vpclmulqdq $0x11,%xmm3,%xmm7,%xmm7+ vaesenc %xmm2,%xmm11,%xmm11+ vmovdqu 16-32(%r9),%xmm3+ negq %r12+ vaesenc %xmm2,%xmm12,%xmm12+ vpxor %xmm5,%xmm6,%xmm6+ vpclmulqdq $0x00,%xmm3,%xmm0,%xmm5+ vpxor %xmm4,%xmm8,%xmm8+ vaesenc %xmm2,%xmm13,%xmm13+ vpxor %xmm5,%xmm1,%xmm4+ andq $0x60,%r12+ vmovups 32-128(%rcx),%xmm15+ vpclmulqdq $0x10,%xmm3,%xmm0,%xmm1+ vaesenc %xmm2,%xmm14,%xmm14++ vpclmulqdq $0x01,%xmm3,%xmm0,%xmm2+ leaq (%r14,%r12,1),%r14+ vaesenc %xmm15,%xmm9,%xmm9+ vpxor 16+8(%rsp),%xmm8,%xmm8+ vpclmulqdq $0x11,%xmm3,%xmm0,%xmm3+ vmovdqu 64+8(%rsp),%xmm0+ vaesenc %xmm15,%xmm10,%xmm10+ movbeq 88(%r14),%r13+ vaesenc %xmm15,%xmm11,%xmm11+ movbeq 80(%r14),%r12+ vaesenc %xmm15,%xmm12,%xmm12+ movq %r13,32+8(%rsp)+ vaesenc %xmm15,%xmm13,%xmm13+ movq %r12,40+8(%rsp)+ vmovdqu 48-32(%r9),%xmm5+ vaesenc %xmm15,%xmm14,%xmm14++ vmovups 48-128(%rcx),%xmm15+ vpxor %xmm1,%xmm6,%xmm6+ vpclmulqdq $0x00,%xmm5,%xmm0,%xmm1+ vaesenc %xmm15,%xmm9,%xmm9+ vpxor %xmm2,%xmm6,%xmm6+ vpclmulqdq $0x10,%xmm5,%xmm0,%xmm2+ vaesenc %xmm15,%xmm10,%xmm10+ vpxor %xmm3,%xmm7,%xmm7+ vpclmulqdq $0x01,%xmm5,%xmm0,%xmm3+ vaesenc %xmm15,%xmm11,%xmm11+ vpclmulqdq $0x11,%xmm5,%xmm0,%xmm5+ vmovdqu 80+8(%rsp),%xmm0+ vaesenc %xmm15,%xmm12,%xmm12+ vaesenc %xmm15,%xmm13,%xmm13+ vpxor %xmm1,%xmm4,%xmm4+ vmovdqu 64-32(%r9),%xmm1+ vaesenc %xmm15,%xmm14,%xmm14++ vmovups 64-128(%rcx),%xmm15+ vpxor %xmm2,%xmm6,%xmm6+ vpclmulqdq $0x00,%xmm1,%xmm0,%xmm2+ vaesenc %xmm15,%xmm9,%xmm9+ vpxor %xmm3,%xmm6,%xmm6+ vpclmulqdq $0x10,%xmm1,%xmm0,%xmm3+ vaesenc %xmm15,%xmm10,%xmm10+ movbeq 72(%r14),%r13+ vpxor %xmm5,%xmm7,%xmm7+ vpclmulqdq $0x01,%xmm1,%xmm0,%xmm5+ vaesenc %xmm15,%xmm11,%xmm11+ movbeq 64(%r14),%r12+ vpclmulqdq $0x11,%xmm1,%xmm0,%xmm1+ vmovdqu 96+8(%rsp),%xmm0+ vaesenc %xmm15,%xmm12,%xmm12+ movq %r13,48+8(%rsp)+ vaesenc %xmm15,%xmm13,%xmm13+ movq %r12,56+8(%rsp)+ vpxor %xmm2,%xmm4,%xmm4+ vmovdqu 96-32(%r9),%xmm2+ vaesenc %xmm15,%xmm14,%xmm14++ vmovups 80-128(%rcx),%xmm15+ vpxor %xmm3,%xmm6,%xmm6+ vpclmulqdq $0x00,%xmm2,%xmm0,%xmm3+ vaesenc %xmm15,%xmm9,%xmm9+ vpxor %xmm5,%xmm6,%xmm6+ vpclmulqdq $0x10,%xmm2,%xmm0,%xmm5+ vaesenc %xmm15,%xmm10,%xmm10+ movbeq 56(%r14),%r13+ vpxor %xmm1,%xmm7,%xmm7+ vpclmulqdq $0x01,%xmm2,%xmm0,%xmm1+ vpxor 112+8(%rsp),%xmm8,%xmm8+ vaesenc %xmm15,%xmm11,%xmm11+ movbeq 48(%r14),%r12+ vpclmulqdq $0x11,%xmm2,%xmm0,%xmm2+ vaesenc %xmm15,%xmm12,%xmm12+ movq %r13,64+8(%rsp)+ vaesenc %xmm15,%xmm13,%xmm13+ movq %r12,72+8(%rsp)+ vpxor %xmm3,%xmm4,%xmm4+ vmovdqu 112-32(%r9),%xmm3+ vaesenc %xmm15,%xmm14,%xmm14++ vmovups 96-128(%rcx),%xmm15+ vpxor %xmm5,%xmm6,%xmm6+ vpclmulqdq $0x10,%xmm3,%xmm8,%xmm5+ vaesenc %xmm15,%xmm9,%xmm9+ vpxor %xmm1,%xmm6,%xmm6+ vpclmulqdq $0x01,%xmm3,%xmm8,%xmm1+ vaesenc %xmm15,%xmm10,%xmm10+ movbeq 40(%r14),%r13+ vpxor %xmm2,%xmm7,%xmm7+ vpclmulqdq $0x00,%xmm3,%xmm8,%xmm2+ vaesenc %xmm15,%xmm11,%xmm11+ movbeq 32(%r14),%r12+ vpclmulqdq $0x11,%xmm3,%xmm8,%xmm8+ vaesenc %xmm15,%xmm12,%xmm12+ movq %r13,80+8(%rsp)+ vaesenc %xmm15,%xmm13,%xmm13+ movq %r12,88+8(%rsp)+ vpxor %xmm5,%xmm6,%xmm6+ vaesenc %xmm15,%xmm14,%xmm14+ vpxor %xmm1,%xmm6,%xmm6++ vmovups 112-128(%rcx),%xmm15+ vpslldq $8,%xmm6,%xmm5+ vpxor %xmm2,%xmm4,%xmm4+ vmovdqu 16(%r11),%xmm3++ vaesenc %xmm15,%xmm9,%xmm9+ vpxor %xmm8,%xmm7,%xmm7+ vaesenc %xmm15,%xmm10,%xmm10+ vpxor %xmm5,%xmm4,%xmm4+ movbeq 24(%r14),%r13+ vaesenc %xmm15,%xmm11,%xmm11+ movbeq 16(%r14),%r12+ vpalignr $8,%xmm4,%xmm4,%xmm0+ vpclmulqdq $0x10,%xmm3,%xmm4,%xmm4+ movq %r13,96+8(%rsp)+ vaesenc %xmm15,%xmm12,%xmm12+ movq %r12,104+8(%rsp)+ vaesenc %xmm15,%xmm13,%xmm13+ vmovups 128-128(%rcx),%xmm1+ vaesenc %xmm15,%xmm14,%xmm14++ vaesenc %xmm1,%xmm9,%xmm9+ vmovups 144-128(%rcx),%xmm15+ vaesenc %xmm1,%xmm10,%xmm10+ vpsrldq $8,%xmm6,%xmm6+ vaesenc %xmm1,%xmm11,%xmm11+ vpxor %xmm6,%xmm7,%xmm7+ vaesenc %xmm1,%xmm12,%xmm12+ vpxor %xmm0,%xmm4,%xmm4+ movbeq 8(%r14),%r13+ vaesenc %xmm1,%xmm13,%xmm13+ movbeq 0(%r14),%r12+ vaesenc %xmm1,%xmm14,%xmm14+ vmovups 160-128(%rcx),%xmm1+ cmpl $11,%r10d+ jb L$enc_tail++ vaesenc %xmm15,%xmm9,%xmm9+ vaesenc %xmm15,%xmm10,%xmm10+ vaesenc %xmm15,%xmm11,%xmm11+ vaesenc %xmm15,%xmm12,%xmm12+ vaesenc %xmm15,%xmm13,%xmm13+ vaesenc %xmm15,%xmm14,%xmm14++ vaesenc %xmm1,%xmm9,%xmm9+ vaesenc %xmm1,%xmm10,%xmm10+ vaesenc %xmm1,%xmm11,%xmm11+ vaesenc %xmm1,%xmm12,%xmm12+ vaesenc %xmm1,%xmm13,%xmm13+ vmovups 176-128(%rcx),%xmm15+ vaesenc %xmm1,%xmm14,%xmm14+ vmovups 192-128(%rcx),%xmm1+ je L$enc_tail++ vaesenc %xmm15,%xmm9,%xmm9+ vaesenc %xmm15,%xmm10,%xmm10+ vaesenc %xmm15,%xmm11,%xmm11+ vaesenc %xmm15,%xmm12,%xmm12+ vaesenc %xmm15,%xmm13,%xmm13+ vaesenc %xmm15,%xmm14,%xmm14++ vaesenc %xmm1,%xmm9,%xmm9+ vaesenc %xmm1,%xmm10,%xmm10+ vaesenc %xmm1,%xmm11,%xmm11+ vaesenc %xmm1,%xmm12,%xmm12+ vaesenc %xmm1,%xmm13,%xmm13+ vmovups 208-128(%rcx),%xmm15+ vaesenc %xmm1,%xmm14,%xmm14+ vmovups 224-128(%rcx),%xmm1+ jmp L$enc_tail++.p2align 5+L$handle_ctr32:+ vmovdqu (%r11),%xmm0+ vpshufb %xmm0,%xmm1,%xmm6+ vmovdqu 48(%r11),%xmm5+ vpaddd 64(%r11),%xmm6,%xmm10+ vpaddd %xmm5,%xmm6,%xmm11+ vmovdqu 0-32(%r9),%xmm3+ vpaddd %xmm5,%xmm10,%xmm12+ vpshufb %xmm0,%xmm10,%xmm10+ vpaddd %xmm5,%xmm11,%xmm13+ vpshufb %xmm0,%xmm11,%xmm11+ vpxor %xmm15,%xmm10,%xmm10+ vpaddd %xmm5,%xmm12,%xmm14+ vpshufb %xmm0,%xmm12,%xmm12+ vpxor %xmm15,%xmm11,%xmm11+ vpaddd %xmm5,%xmm13,%xmm1+ vpshufb %xmm0,%xmm13,%xmm13+ vpshufb %xmm0,%xmm14,%xmm14+ vpshufb %xmm0,%xmm1,%xmm1+ jmp L$resume_ctr32++.p2align 5+L$enc_tail:+ vaesenc %xmm15,%xmm9,%xmm9+ vmovdqu %xmm7,16+8(%rsp)+ vpalignr $8,%xmm4,%xmm4,%xmm8+ vaesenc %xmm15,%xmm10,%xmm10+ vpclmulqdq $0x10,%xmm3,%xmm4,%xmm4+ vpxor 0(%rdi),%xmm1,%xmm2+ vaesenc %xmm15,%xmm11,%xmm11+ vpxor 16(%rdi),%xmm1,%xmm0+ vaesenc %xmm15,%xmm12,%xmm12+ vpxor 32(%rdi),%xmm1,%xmm5+ vaesenc %xmm15,%xmm13,%xmm13+ vpxor 48(%rdi),%xmm1,%xmm6+ vaesenc %xmm15,%xmm14,%xmm14+ vpxor 64(%rdi),%xmm1,%xmm7+ vpxor 80(%rdi),%xmm1,%xmm3+ vmovdqu (%r8),%xmm1++ vaesenclast %xmm2,%xmm9,%xmm9+ vmovdqu 32(%r11),%xmm2+ vaesenclast %xmm0,%xmm10,%xmm10+ vpaddb %xmm2,%xmm1,%xmm0+ movq %r13,112+8(%rsp)+ leaq 96(%rdi),%rdi+ vaesenclast %xmm5,%xmm11,%xmm11+ vpaddb %xmm2,%xmm0,%xmm5+ movq %r12,120+8(%rsp)+ leaq 96(%rsi),%rsi+ vmovdqu 0-128(%rcx),%xmm15+ vaesenclast %xmm6,%xmm12,%xmm12+ vpaddb %xmm2,%xmm5,%xmm6+ vaesenclast %xmm7,%xmm13,%xmm13+ vpaddb %xmm2,%xmm6,%xmm7+ vaesenclast %xmm3,%xmm14,%xmm14+ vpaddb %xmm2,%xmm7,%xmm3++ addq $0x60,%rax+ subq $0x6,%rdx+ jc L$6x_done++ vmovups %xmm9,-96(%rsi)+ vpxor %xmm15,%xmm1,%xmm9+ vmovups %xmm10,-80(%rsi)+ vmovdqa %xmm0,%xmm10+ vmovups %xmm11,-64(%rsi)+ vmovdqa %xmm5,%xmm11+ vmovups %xmm12,-48(%rsi)+ vmovdqa %xmm6,%xmm12+ vmovups %xmm13,-32(%rsi)+ vmovdqa %xmm7,%xmm13+ vmovups %xmm14,-16(%rsi)+ vmovdqa %xmm3,%xmm14+ vmovdqu 32+8(%rsp),%xmm7+ jmp L$oop6x++L$6x_done:+ vpxor 16+8(%rsp),%xmm8,%xmm8+ vpxor %xmm4,%xmm8,%xmm8++ .byte 0xf3,0xc3+.cfi_endproc ++.globl _crypton_gcm_asm_decrypt++.p2align 5+_crypton_gcm_asm_decrypt:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa+++ pushq %rbp+.cfi_adjust_cfa_offset 8+.cfi_offset %rbp,-16+ movq %rsp,%rbp+.cfi_def_cfa_register %rbp+ xorq %rax,%rax+ cmpq $0x60,%rdx+ jb L$gcm_dec_abort++ pushq %rbx+.cfi_offset %rbx,-24+ pushq %r12+.cfi_offset %r12,-32+ pushq %r13+.cfi_offset %r13,-40+ pushq %r14+.cfi_offset %r14,-48+ pushq %r15+.cfi_offset %r15,-56++ vzeroupper++ vmovdqu (%r8),%xmm1+ addq $-128,%rsp+ movl 12(%r8),%ebx+ leaq L$bswap_mask(%rip),%r11+ leaq -128(%rcx),%r14+ movq $0xf80,%r15+ vmovdqu (%r9),%xmm8+ andq $-128,%rsp+ vmovdqu (%r11),%xmm0+ leaq 128(%rcx),%rcx+ leaq 32+32(%r9),%r9+ movl 240-128(%rcx),%r10d+ vpshufb %xmm0,%xmm8,%xmm8++ andq %r15,%r14+ andq %rsp,%r15+ subq %r14,%r15+ jc L$dec_no_key_aliasing+ cmpq $768,%r15+ jnc L$dec_no_key_aliasing+ subq %r15,%rsp+L$dec_no_key_aliasing:++ vmovdqu 80(%rdi),%xmm7+ leaq (%rdi),%r14+ vmovdqu 64(%rdi),%xmm4+ leaq -192(%rdi,%rdx,1),%r15+ vmovdqu 48(%rdi),%xmm5+ shrq $4,%rdx+ xorq %rax,%rax+ vmovdqu 32(%rdi),%xmm6+ vpshufb %xmm0,%xmm7,%xmm7+ vmovdqu 16(%rdi),%xmm2+ vpshufb %xmm0,%xmm4,%xmm4+ vmovdqu (%rdi),%xmm3+ vpshufb %xmm0,%xmm5,%xmm5+ vmovdqu %xmm4,48(%rsp)+ vpshufb %xmm0,%xmm6,%xmm6+ vmovdqu %xmm5,64(%rsp)+ vpshufb %xmm0,%xmm2,%xmm2+ vmovdqu %xmm6,80(%rsp)+ vpshufb %xmm0,%xmm3,%xmm3+ vmovdqu %xmm2,96(%rsp)+ vmovdqu %xmm3,112(%rsp)++ call _crypton_gcm_asm_ctr32_ghash_6x++ vmovups %xmm9,-96(%rsi)+ vmovups %xmm10,-80(%rsi)+ vmovups %xmm11,-64(%rsi)+ vmovups %xmm12,-48(%rsi)+ vmovups %xmm13,-32(%rsi)+ vmovups %xmm14,-16(%rsi)++ vpshufb (%r11),%xmm8,%xmm8+ vmovdqu %xmm8,-64(%r9)++ vzeroupper+ movq -40(%rbp),%r15+ movq -32(%rbp),%r14+ movq -24(%rbp),%r13+ movq -16(%rbp),%r12+ movq -8(%rbp),%rbx+ movq %rbp,%rsp+.cfi_def_cfa_register %rsp+L$gcm_dec_abort:+ popq %rbp+.cfi_adjust_cfa_offset -8+.cfi_restore %rbp+.cfi_restore %r12+.cfi_restore %r13+.cfi_restore %r14+.cfi_restore %r15+.cfi_restore %rbx+ .byte 0xf3,0xc3+.cfi_endproc +++.p2align 5+_crypton_gcm_asm_ctr32_6x:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa+++ vmovdqu 0-128(%rcx),%xmm4+ vmovdqu 32(%r11),%xmm2+ leaq -1(%r10),%r13+ vmovups 16-128(%rcx),%xmm15+ leaq 32-128(%rcx),%r12+ vpxor %xmm4,%xmm1,%xmm9+ addl $100663296,%ebx+ jc L$handle_ctr32_2+ vpaddb %xmm2,%xmm1,%xmm10+ vpaddb %xmm2,%xmm10,%xmm11+ vpxor %xmm4,%xmm10,%xmm10+ vpaddb %xmm2,%xmm11,%xmm12+ vpxor %xmm4,%xmm11,%xmm11+ vpaddb %xmm2,%xmm12,%xmm13+ vpxor %xmm4,%xmm12,%xmm12+ vpaddb %xmm2,%xmm13,%xmm14+ vpxor %xmm4,%xmm13,%xmm13+ vpaddb %xmm2,%xmm14,%xmm1+ vpxor %xmm4,%xmm14,%xmm14+ jmp L$oop_ctr32++.p2align 4+L$oop_ctr32:+ vaesenc %xmm15,%xmm9,%xmm9+ vaesenc %xmm15,%xmm10,%xmm10+ vaesenc %xmm15,%xmm11,%xmm11+ vaesenc %xmm15,%xmm12,%xmm12+ vaesenc %xmm15,%xmm13,%xmm13+ vaesenc %xmm15,%xmm14,%xmm14+ vmovups (%r12),%xmm15+ leaq 16(%r12),%r12+ decl %r13d+ jnz L$oop_ctr32++ vmovdqu (%r12),%xmm3+ vaesenc %xmm15,%xmm9,%xmm9+ vpxor 0(%rdi),%xmm3,%xmm4+ vaesenc %xmm15,%xmm10,%xmm10+ vpxor 16(%rdi),%xmm3,%xmm5+ vaesenc %xmm15,%xmm11,%xmm11+ vpxor 32(%rdi),%xmm3,%xmm6+ vaesenc %xmm15,%xmm12,%xmm12+ vpxor 48(%rdi),%xmm3,%xmm8+ vaesenc %xmm15,%xmm13,%xmm13+ vpxor 64(%rdi),%xmm3,%xmm2+ vaesenc %xmm15,%xmm14,%xmm14+ vpxor 80(%rdi),%xmm3,%xmm3+ leaq 96(%rdi),%rdi++ vaesenclast %xmm4,%xmm9,%xmm9+ vaesenclast %xmm5,%xmm10,%xmm10+ vaesenclast %xmm6,%xmm11,%xmm11+ vaesenclast %xmm8,%xmm12,%xmm12+ vaesenclast %xmm2,%xmm13,%xmm13+ vaesenclast %xmm3,%xmm14,%xmm14+ vmovups %xmm9,0(%rsi)+ vmovups %xmm10,16(%rsi)+ vmovups %xmm11,32(%rsi)+ vmovups %xmm12,48(%rsi)+ vmovups %xmm13,64(%rsi)+ vmovups %xmm14,80(%rsi)+ leaq 96(%rsi),%rsi++ .byte 0xf3,0xc3+.p2align 5+L$handle_ctr32_2:+ vpshufb %xmm0,%xmm1,%xmm6+ vmovdqu 48(%r11),%xmm5+ vpaddd 64(%r11),%xmm6,%xmm10+ vpaddd %xmm5,%xmm6,%xmm11+ vpaddd %xmm5,%xmm10,%xmm12+ vpshufb %xmm0,%xmm10,%xmm10+ vpaddd %xmm5,%xmm11,%xmm13+ vpshufb %xmm0,%xmm11,%xmm11+ vpxor %xmm4,%xmm10,%xmm10+ vpaddd %xmm5,%xmm12,%xmm14+ vpshufb %xmm0,%xmm12,%xmm12+ vpxor %xmm4,%xmm11,%xmm11+ vpaddd %xmm5,%xmm13,%xmm1+ vpshufb %xmm0,%xmm13,%xmm13+ vpxor %xmm4,%xmm12,%xmm12+ vpshufb %xmm0,%xmm14,%xmm14+ vpxor %xmm4,%xmm13,%xmm13+ vpshufb %xmm0,%xmm1,%xmm1+ vpxor %xmm4,%xmm14,%xmm14+ jmp L$oop_ctr32+.cfi_endproc +++.globl _crypton_gcm_asm_encrypt++.p2align 5+_crypton_gcm_asm_encrypt:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa+++ pushq %rbp+.cfi_adjust_cfa_offset 8+.cfi_offset %rbp,-16+ movq %rsp,%rbp+.cfi_def_cfa_register %rbp+ xorq %rax,%rax+ cmpq $288,%rdx+ jb L$gcm_enc_abort++ pushq %rbx+.cfi_offset %rbx,-24+ pushq %r12+.cfi_offset %r12,-32+ pushq %r13+.cfi_offset %r13,-40+ pushq %r14+.cfi_offset %r14,-48+ pushq %r15+.cfi_offset %r15,-56++ vzeroupper++ vmovdqu (%r8),%xmm1+ addq $-128,%rsp+ movl 12(%r8),%ebx+ leaq L$bswap_mask(%rip),%r11+ leaq -128(%rcx),%r14+ movq $0xf80,%r15+ leaq 128(%rcx),%rcx+ vmovdqu (%r11),%xmm0+ andq $-128,%rsp+ movl 240-128(%rcx),%r10d++ andq %r15,%r14+ andq %rsp,%r15+ subq %r14,%r15+ jc L$enc_no_key_aliasing+ cmpq $768,%r15+ jnc L$enc_no_key_aliasing+ subq %r15,%rsp+L$enc_no_key_aliasing:++ leaq (%rsi),%r14+ leaq -192(%rsi,%rdx,1),%r15+ shrq $4,%rdx++ call _crypton_gcm_asm_ctr32_6x+ vpshufb %xmm0,%xmm9,%xmm8+ vpshufb %xmm0,%xmm10,%xmm2+ vmovdqu %xmm8,112(%rsp)+ vpshufb %xmm0,%xmm11,%xmm4+ vmovdqu %xmm2,96(%rsp)+ vpshufb %xmm0,%xmm12,%xmm5+ vmovdqu %xmm4,80(%rsp)+ vpshufb %xmm0,%xmm13,%xmm6+ vmovdqu %xmm5,64(%rsp)+ vpshufb %xmm0,%xmm14,%xmm7+ vmovdqu %xmm6,48(%rsp)++ call _crypton_gcm_asm_ctr32_6x++ vmovdqu (%r9),%xmm8+ leaq 32+32(%r9),%r9+ subq $12,%rdx+ movq $192,%rax+ vpshufb %xmm0,%xmm8,%xmm8++ call _crypton_gcm_asm_ctr32_ghash_6x+ vmovdqu 32(%rsp),%xmm7+ vmovdqu (%r11),%xmm0+ vmovdqu 0-32(%r9),%xmm3+ vpunpckhqdq %xmm7,%xmm7,%xmm1+ vmovdqu 32-32(%r9),%xmm15+ vmovups %xmm9,-96(%rsi)+ vpshufb %xmm0,%xmm9,%xmm9+ vpxor %xmm7,%xmm1,%xmm1+ vmovups %xmm10,-80(%rsi)+ vpshufb %xmm0,%xmm10,%xmm10+ vmovups %xmm11,-64(%rsi)+ vpshufb %xmm0,%xmm11,%xmm11+ vmovups %xmm12,-48(%rsi)+ vpshufb %xmm0,%xmm12,%xmm12+ vmovups %xmm13,-32(%rsi)+ vpshufb %xmm0,%xmm13,%xmm13+ vmovups %xmm14,-16(%rsi)+ vpshufb %xmm0,%xmm14,%xmm14+ vmovdqu %xmm9,16(%rsp)+ vmovdqu 48(%rsp),%xmm6+ vmovdqu 16-32(%r9),%xmm0+ vpunpckhqdq %xmm6,%xmm6,%xmm2+ vpclmulqdq $0x00,%xmm3,%xmm7,%xmm5+ vpxor %xmm6,%xmm2,%xmm2+ vpclmulqdq $0x11,%xmm3,%xmm7,%xmm7+ vpclmulqdq $0x00,%xmm15,%xmm1,%xmm1++ vmovdqu 64(%rsp),%xmm9+ vpclmulqdq $0x00,%xmm0,%xmm6,%xmm4+ vmovdqu 48-32(%r9),%xmm3+ vpxor %xmm5,%xmm4,%xmm4+ vpunpckhqdq %xmm9,%xmm9,%xmm5+ vpclmulqdq $0x11,%xmm0,%xmm6,%xmm6+ vpxor %xmm9,%xmm5,%xmm5+ vpxor %xmm7,%xmm6,%xmm6+ vpclmulqdq $0x10,%xmm15,%xmm2,%xmm2+ vmovdqu 80-32(%r9),%xmm15+ vpxor %xmm1,%xmm2,%xmm2++ vmovdqu 80(%rsp),%xmm1+ vpclmulqdq $0x00,%xmm3,%xmm9,%xmm7+ vmovdqu 64-32(%r9),%xmm0+ vpxor %xmm4,%xmm7,%xmm7+ vpunpckhqdq %xmm1,%xmm1,%xmm4+ vpclmulqdq $0x11,%xmm3,%xmm9,%xmm9+ vpxor %xmm1,%xmm4,%xmm4+ vpxor %xmm6,%xmm9,%xmm9+ vpclmulqdq $0x00,%xmm15,%xmm5,%xmm5+ vpxor %xmm2,%xmm5,%xmm5++ vmovdqu 96(%rsp),%xmm2+ vpclmulqdq $0x00,%xmm0,%xmm1,%xmm6+ vmovdqu 96-32(%r9),%xmm3+ vpxor %xmm7,%xmm6,%xmm6+ vpunpckhqdq %xmm2,%xmm2,%xmm7+ vpclmulqdq $0x11,%xmm0,%xmm1,%xmm1+ vpxor %xmm2,%xmm7,%xmm7+ vpxor %xmm9,%xmm1,%xmm1+ vpclmulqdq $0x10,%xmm15,%xmm4,%xmm4+ vmovdqu 128-32(%r9),%xmm15+ vpxor %xmm5,%xmm4,%xmm4++ vpxor 112(%rsp),%xmm8,%xmm8+ vpclmulqdq $0x00,%xmm3,%xmm2,%xmm5+ vmovdqu 112-32(%r9),%xmm0+ vpunpckhqdq %xmm8,%xmm8,%xmm9+ vpxor %xmm6,%xmm5,%xmm5+ vpclmulqdq $0x11,%xmm3,%xmm2,%xmm2+ vpxor %xmm8,%xmm9,%xmm9+ vpxor %xmm1,%xmm2,%xmm2+ vpclmulqdq $0x00,%xmm15,%xmm7,%xmm7+ vpxor %xmm4,%xmm7,%xmm4++ vpclmulqdq $0x00,%xmm0,%xmm8,%xmm6+ vmovdqu 0-32(%r9),%xmm3+ vpunpckhqdq %xmm14,%xmm14,%xmm1+ vpclmulqdq $0x11,%xmm0,%xmm8,%xmm8+ vpxor %xmm14,%xmm1,%xmm1+ vpxor %xmm5,%xmm6,%xmm5+ vpclmulqdq $0x10,%xmm15,%xmm9,%xmm9+ vmovdqu 32-32(%r9),%xmm15+ vpxor %xmm2,%xmm8,%xmm7+ vpxor %xmm4,%xmm9,%xmm6++ vmovdqu 16-32(%r9),%xmm0+ vpxor %xmm5,%xmm7,%xmm9+ vpclmulqdq $0x00,%xmm3,%xmm14,%xmm4+ vpxor %xmm9,%xmm6,%xmm6+ vpunpckhqdq %xmm13,%xmm13,%xmm2+ vpclmulqdq $0x11,%xmm3,%xmm14,%xmm14+ vpxor %xmm13,%xmm2,%xmm2+ vpslldq $8,%xmm6,%xmm9+ vpclmulqdq $0x00,%xmm15,%xmm1,%xmm1+ vpxor %xmm9,%xmm5,%xmm8+ vpsrldq $8,%xmm6,%xmm6+ vpxor %xmm6,%xmm7,%xmm7++ vpclmulqdq $0x00,%xmm0,%xmm13,%xmm5+ vmovdqu 48-32(%r9),%xmm3+ vpxor %xmm4,%xmm5,%xmm5+ vpunpckhqdq %xmm12,%xmm12,%xmm9+ vpclmulqdq $0x11,%xmm0,%xmm13,%xmm13+ vpxor %xmm12,%xmm9,%xmm9+ vpxor %xmm14,%xmm13,%xmm13+ vpalignr $8,%xmm8,%xmm8,%xmm14+ vpclmulqdq $0x10,%xmm15,%xmm2,%xmm2+ vmovdqu 80-32(%r9),%xmm15+ vpxor %xmm1,%xmm2,%xmm2++ vpclmulqdq $0x00,%xmm3,%xmm12,%xmm4+ vmovdqu 64-32(%r9),%xmm0+ vpxor %xmm5,%xmm4,%xmm4+ vpunpckhqdq %xmm11,%xmm11,%xmm1+ vpclmulqdq $0x11,%xmm3,%xmm12,%xmm12+ vpxor %xmm11,%xmm1,%xmm1+ vpxor %xmm13,%xmm12,%xmm12+ vxorps 16(%rsp),%xmm7,%xmm7+ vpclmulqdq $0x00,%xmm15,%xmm9,%xmm9+ vpxor %xmm2,%xmm9,%xmm9++ vpclmulqdq $0x10,16(%r11),%xmm8,%xmm8+ vxorps %xmm14,%xmm8,%xmm8++ vpclmulqdq $0x00,%xmm0,%xmm11,%xmm5+ vmovdqu 96-32(%r9),%xmm3+ vpxor %xmm4,%xmm5,%xmm5+ vpunpckhqdq %xmm10,%xmm10,%xmm2+ vpclmulqdq $0x11,%xmm0,%xmm11,%xmm11+ vpxor %xmm10,%xmm2,%xmm2+ vpalignr $8,%xmm8,%xmm8,%xmm14+ vpxor %xmm12,%xmm11,%xmm11+ vpclmulqdq $0x10,%xmm15,%xmm1,%xmm1+ vmovdqu 128-32(%r9),%xmm15+ vpxor %xmm9,%xmm1,%xmm1++ vxorps %xmm7,%xmm14,%xmm14+ vpclmulqdq $0x10,16(%r11),%xmm8,%xmm8+ vxorps %xmm14,%xmm8,%xmm8++ vpclmulqdq $0x00,%xmm3,%xmm10,%xmm4+ vmovdqu 112-32(%r9),%xmm0+ vpxor %xmm5,%xmm4,%xmm4+ vpunpckhqdq %xmm8,%xmm8,%xmm9+ vpclmulqdq $0x11,%xmm3,%xmm10,%xmm10+ vpxor %xmm8,%xmm9,%xmm9+ vpxor %xmm11,%xmm10,%xmm10+ vpclmulqdq $0x00,%xmm15,%xmm2,%xmm2+ vpxor %xmm1,%xmm2,%xmm2++ vpclmulqdq $0x00,%xmm0,%xmm8,%xmm5+ vpclmulqdq $0x11,%xmm0,%xmm8,%xmm7+ vpxor %xmm4,%xmm5,%xmm5+ vpclmulqdq $0x10,%xmm15,%xmm9,%xmm6+ vpxor %xmm10,%xmm7,%xmm7+ vpxor %xmm2,%xmm6,%xmm6++ vpxor %xmm5,%xmm7,%xmm4+ vpxor %xmm4,%xmm6,%xmm6+ vpslldq $8,%xmm6,%xmm1+ vmovdqu 16(%r11),%xmm3+ vpsrldq $8,%xmm6,%xmm6+ vpxor %xmm1,%xmm5,%xmm8+ vpxor %xmm6,%xmm7,%xmm7++ vpalignr $8,%xmm8,%xmm8,%xmm2+ vpclmulqdq $0x10,%xmm3,%xmm8,%xmm8+ vpxor %xmm2,%xmm8,%xmm8++ vpalignr $8,%xmm8,%xmm8,%xmm2+ vpclmulqdq $0x10,%xmm3,%xmm8,%xmm8+ vpxor %xmm7,%xmm2,%xmm2+ vpxor %xmm2,%xmm8,%xmm8+ vpshufb (%r11),%xmm8,%xmm8+ vmovdqu %xmm8,-64(%r9)++ vzeroupper+ movq -40(%rbp),%r15+ movq -32(%rbp),%r14+ movq -24(%rbp),%r13+ movq -16(%rbp),%r12+ movq -8(%rbp),%rbx+ movq %rbp,%rsp+.cfi_def_cfa_register %rsp+L$gcm_enc_abort:+ popq %rbp+.cfi_adjust_cfa_offset -8+.cfi_restore %rbp+.cfi_restore %r12+.cfi_restore %r13+.cfi_restore %r14+.cfi_restore %r15+.cfi_restore %rbx+ .byte 0xf3,0xc3+.cfi_endproc ++.p2align 6+L$bswap_mask:+.byte 15,14,13,12,11,10,9,8,7,6,5,4,3,2,1,0+L$poly:+.byte 0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0xc2+L$one_msb:+.byte 0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,1+L$two_lsb:+.byte 2,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0+L$one_lsb:+.byte 1,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0+.byte 65,69,83,45,78,73,32,71,67,77,32,109,111,100,117,108,101,32,102,111,114,32,120,56,54,95,54,52,44,32,67,82,89,80,84,79,71,65,77,83,32,98,121,32,60,97,112,112,114,111,64,111,112,101,110,115,115,108,46,111,114,103,62,0+.p2align 6
@@ -0,0 +1,965 @@+.text ++.def _crypton_gcm_asm_ctr32_ghash_6x; .scl 3; .type 32; .endef+.p2align 5+_crypton_gcm_asm_ctr32_ghash_6x:+ .byte 0xf3,0x0f,0x1e,0xfa+++ vmovdqu 32(%r11),%xmm2+ subq $6,%rdx+ vpxor %xmm4,%xmm4,%xmm4+ vmovdqu 0-128(%rcx),%xmm15+ vpaddb %xmm2,%xmm1,%xmm10+ vpaddb %xmm2,%xmm10,%xmm11+ vpaddb %xmm2,%xmm11,%xmm12+ vpaddb %xmm2,%xmm12,%xmm13+ vpaddb %xmm2,%xmm13,%xmm14+ vpxor %xmm15,%xmm1,%xmm9+ vmovdqu %xmm4,16+8(%rsp)+ jmp .Loop6x++.p2align 5+.Loop6x:+ addl $100663296,%ebx+ jc .Lhandle_ctr32+ vmovdqu 0-32(%r9),%xmm3+ vpaddb %xmm2,%xmm14,%xmm1+ vpxor %xmm15,%xmm10,%xmm10+ vpxor %xmm15,%xmm11,%xmm11++.Lresume_ctr32:+ vmovdqu %xmm1,(%r8)+ vpclmulqdq $0x10,%xmm3,%xmm7,%xmm5+ vpxor %xmm15,%xmm12,%xmm12+ vmovups 16-128(%rcx),%xmm2+ vpclmulqdq $0x01,%xmm3,%xmm7,%xmm6+ xorq %r12,%r12+ cmpq %r14,%r15++ vaesenc %xmm2,%xmm9,%xmm9+ vmovdqu 48+8(%rsp),%xmm0+ vpxor %xmm15,%xmm13,%xmm13+ vpclmulqdq $0x00,%xmm3,%xmm7,%xmm1+ vaesenc %xmm2,%xmm10,%xmm10+ vpxor %xmm15,%xmm14,%xmm14+ setnc %r12b+ vpclmulqdq $0x11,%xmm3,%xmm7,%xmm7+ vaesenc %xmm2,%xmm11,%xmm11+ vmovdqu 16-32(%r9),%xmm3+ negq %r12+ vaesenc %xmm2,%xmm12,%xmm12+ vpxor %xmm5,%xmm6,%xmm6+ vpclmulqdq $0x00,%xmm3,%xmm0,%xmm5+ vpxor %xmm4,%xmm8,%xmm8+ vaesenc %xmm2,%xmm13,%xmm13+ vpxor %xmm5,%xmm1,%xmm4+ andq $0x60,%r12+ vmovups 32-128(%rcx),%xmm15+ vpclmulqdq $0x10,%xmm3,%xmm0,%xmm1+ vaesenc %xmm2,%xmm14,%xmm14++ vpclmulqdq $0x01,%xmm3,%xmm0,%xmm2+ leaq (%r14,%r12,1),%r14+ vaesenc %xmm15,%xmm9,%xmm9+ vpxor 16+8(%rsp),%xmm8,%xmm8+ vpclmulqdq $0x11,%xmm3,%xmm0,%xmm3+ vmovdqu 64+8(%rsp),%xmm0+ vaesenc %xmm15,%xmm10,%xmm10+ movbeq 88(%r14),%r13+ vaesenc %xmm15,%xmm11,%xmm11+ movbeq 80(%r14),%r12+ vaesenc %xmm15,%xmm12,%xmm12+ movq %r13,32+8(%rsp)+ vaesenc %xmm15,%xmm13,%xmm13+ movq %r12,40+8(%rsp)+ vmovdqu 48-32(%r9),%xmm5+ vaesenc %xmm15,%xmm14,%xmm14++ vmovups 48-128(%rcx),%xmm15+ vpxor %xmm1,%xmm6,%xmm6+ vpclmulqdq $0x00,%xmm5,%xmm0,%xmm1+ vaesenc %xmm15,%xmm9,%xmm9+ vpxor %xmm2,%xmm6,%xmm6+ vpclmulqdq $0x10,%xmm5,%xmm0,%xmm2+ vaesenc %xmm15,%xmm10,%xmm10+ vpxor %xmm3,%xmm7,%xmm7+ vpclmulqdq $0x01,%xmm5,%xmm0,%xmm3+ vaesenc %xmm15,%xmm11,%xmm11+ vpclmulqdq $0x11,%xmm5,%xmm0,%xmm5+ vmovdqu 80+8(%rsp),%xmm0+ vaesenc %xmm15,%xmm12,%xmm12+ vaesenc %xmm15,%xmm13,%xmm13+ vpxor %xmm1,%xmm4,%xmm4+ vmovdqu 64-32(%r9),%xmm1+ vaesenc %xmm15,%xmm14,%xmm14++ vmovups 64-128(%rcx),%xmm15+ vpxor %xmm2,%xmm6,%xmm6+ vpclmulqdq $0x00,%xmm1,%xmm0,%xmm2+ vaesenc %xmm15,%xmm9,%xmm9+ vpxor %xmm3,%xmm6,%xmm6+ vpclmulqdq $0x10,%xmm1,%xmm0,%xmm3+ vaesenc %xmm15,%xmm10,%xmm10+ movbeq 72(%r14),%r13+ vpxor %xmm5,%xmm7,%xmm7+ vpclmulqdq $0x01,%xmm1,%xmm0,%xmm5+ vaesenc %xmm15,%xmm11,%xmm11+ movbeq 64(%r14),%r12+ vpclmulqdq $0x11,%xmm1,%xmm0,%xmm1+ vmovdqu 96+8(%rsp),%xmm0+ vaesenc %xmm15,%xmm12,%xmm12+ movq %r13,48+8(%rsp)+ vaesenc %xmm15,%xmm13,%xmm13+ movq %r12,56+8(%rsp)+ vpxor %xmm2,%xmm4,%xmm4+ vmovdqu 96-32(%r9),%xmm2+ vaesenc %xmm15,%xmm14,%xmm14++ vmovups 80-128(%rcx),%xmm15+ vpxor %xmm3,%xmm6,%xmm6+ vpclmulqdq $0x00,%xmm2,%xmm0,%xmm3+ vaesenc %xmm15,%xmm9,%xmm9+ vpxor %xmm5,%xmm6,%xmm6+ vpclmulqdq $0x10,%xmm2,%xmm0,%xmm5+ vaesenc %xmm15,%xmm10,%xmm10+ movbeq 56(%r14),%r13+ vpxor %xmm1,%xmm7,%xmm7+ vpclmulqdq $0x01,%xmm2,%xmm0,%xmm1+ vpxor 112+8(%rsp),%xmm8,%xmm8+ vaesenc %xmm15,%xmm11,%xmm11+ movbeq 48(%r14),%r12+ vpclmulqdq $0x11,%xmm2,%xmm0,%xmm2+ vaesenc %xmm15,%xmm12,%xmm12+ movq %r13,64+8(%rsp)+ vaesenc %xmm15,%xmm13,%xmm13+ movq %r12,72+8(%rsp)+ vpxor %xmm3,%xmm4,%xmm4+ vmovdqu 112-32(%r9),%xmm3+ vaesenc %xmm15,%xmm14,%xmm14++ vmovups 96-128(%rcx),%xmm15+ vpxor %xmm5,%xmm6,%xmm6+ vpclmulqdq $0x10,%xmm3,%xmm8,%xmm5+ vaesenc %xmm15,%xmm9,%xmm9+ vpxor %xmm1,%xmm6,%xmm6+ vpclmulqdq $0x01,%xmm3,%xmm8,%xmm1+ vaesenc %xmm15,%xmm10,%xmm10+ movbeq 40(%r14),%r13+ vpxor %xmm2,%xmm7,%xmm7+ vpclmulqdq $0x00,%xmm3,%xmm8,%xmm2+ vaesenc %xmm15,%xmm11,%xmm11+ movbeq 32(%r14),%r12+ vpclmulqdq $0x11,%xmm3,%xmm8,%xmm8+ vaesenc %xmm15,%xmm12,%xmm12+ movq %r13,80+8(%rsp)+ vaesenc %xmm15,%xmm13,%xmm13+ movq %r12,88+8(%rsp)+ vpxor %xmm5,%xmm6,%xmm6+ vaesenc %xmm15,%xmm14,%xmm14+ vpxor %xmm1,%xmm6,%xmm6++ vmovups 112-128(%rcx),%xmm15+ vpslldq $8,%xmm6,%xmm5+ vpxor %xmm2,%xmm4,%xmm4+ vmovdqu 16(%r11),%xmm3++ vaesenc %xmm15,%xmm9,%xmm9+ vpxor %xmm8,%xmm7,%xmm7+ vaesenc %xmm15,%xmm10,%xmm10+ vpxor %xmm5,%xmm4,%xmm4+ movbeq 24(%r14),%r13+ vaesenc %xmm15,%xmm11,%xmm11+ movbeq 16(%r14),%r12+ vpalignr $8,%xmm4,%xmm4,%xmm0+ vpclmulqdq $0x10,%xmm3,%xmm4,%xmm4+ movq %r13,96+8(%rsp)+ vaesenc %xmm15,%xmm12,%xmm12+ movq %r12,104+8(%rsp)+ vaesenc %xmm15,%xmm13,%xmm13+ vmovups 128-128(%rcx),%xmm1+ vaesenc %xmm15,%xmm14,%xmm14++ vaesenc %xmm1,%xmm9,%xmm9+ vmovups 144-128(%rcx),%xmm15+ vaesenc %xmm1,%xmm10,%xmm10+ vpsrldq $8,%xmm6,%xmm6+ vaesenc %xmm1,%xmm11,%xmm11+ vpxor %xmm6,%xmm7,%xmm7+ vaesenc %xmm1,%xmm12,%xmm12+ vpxor %xmm0,%xmm4,%xmm4+ movbeq 8(%r14),%r13+ vaesenc %xmm1,%xmm13,%xmm13+ movbeq 0(%r14),%r12+ vaesenc %xmm1,%xmm14,%xmm14+ vmovups 160-128(%rcx),%xmm1+ cmpl $11,%r10d+ jb .Lenc_tail++ vaesenc %xmm15,%xmm9,%xmm9+ vaesenc %xmm15,%xmm10,%xmm10+ vaesenc %xmm15,%xmm11,%xmm11+ vaesenc %xmm15,%xmm12,%xmm12+ vaesenc %xmm15,%xmm13,%xmm13+ vaesenc %xmm15,%xmm14,%xmm14++ vaesenc %xmm1,%xmm9,%xmm9+ vaesenc %xmm1,%xmm10,%xmm10+ vaesenc %xmm1,%xmm11,%xmm11+ vaesenc %xmm1,%xmm12,%xmm12+ vaesenc %xmm1,%xmm13,%xmm13+ vmovups 176-128(%rcx),%xmm15+ vaesenc %xmm1,%xmm14,%xmm14+ vmovups 192-128(%rcx),%xmm1+ je .Lenc_tail++ vaesenc %xmm15,%xmm9,%xmm9+ vaesenc %xmm15,%xmm10,%xmm10+ vaesenc %xmm15,%xmm11,%xmm11+ vaesenc %xmm15,%xmm12,%xmm12+ vaesenc %xmm15,%xmm13,%xmm13+ vaesenc %xmm15,%xmm14,%xmm14++ vaesenc %xmm1,%xmm9,%xmm9+ vaesenc %xmm1,%xmm10,%xmm10+ vaesenc %xmm1,%xmm11,%xmm11+ vaesenc %xmm1,%xmm12,%xmm12+ vaesenc %xmm1,%xmm13,%xmm13+ vmovups 208-128(%rcx),%xmm15+ vaesenc %xmm1,%xmm14,%xmm14+ vmovups 224-128(%rcx),%xmm1+ jmp .Lenc_tail++.p2align 5+.Lhandle_ctr32:+ vmovdqu (%r11),%xmm0+ vpshufb %xmm0,%xmm1,%xmm6+ vmovdqu 48(%r11),%xmm5+ vpaddd 64(%r11),%xmm6,%xmm10+ vpaddd %xmm5,%xmm6,%xmm11+ vmovdqu 0-32(%r9),%xmm3+ vpaddd %xmm5,%xmm10,%xmm12+ vpshufb %xmm0,%xmm10,%xmm10+ vpaddd %xmm5,%xmm11,%xmm13+ vpshufb %xmm0,%xmm11,%xmm11+ vpxor %xmm15,%xmm10,%xmm10+ vpaddd %xmm5,%xmm12,%xmm14+ vpshufb %xmm0,%xmm12,%xmm12+ vpxor %xmm15,%xmm11,%xmm11+ vpaddd %xmm5,%xmm13,%xmm1+ vpshufb %xmm0,%xmm13,%xmm13+ vpshufb %xmm0,%xmm14,%xmm14+ vpshufb %xmm0,%xmm1,%xmm1+ jmp .Lresume_ctr32++.p2align 5+.Lenc_tail:+ vaesenc %xmm15,%xmm9,%xmm9+ vmovdqu %xmm7,16+8(%rsp)+ vpalignr $8,%xmm4,%xmm4,%xmm8+ vaesenc %xmm15,%xmm10,%xmm10+ vpclmulqdq $0x10,%xmm3,%xmm4,%xmm4+ vpxor 0(%rdi),%xmm1,%xmm2+ vaesenc %xmm15,%xmm11,%xmm11+ vpxor 16(%rdi),%xmm1,%xmm0+ vaesenc %xmm15,%xmm12,%xmm12+ vpxor 32(%rdi),%xmm1,%xmm5+ vaesenc %xmm15,%xmm13,%xmm13+ vpxor 48(%rdi),%xmm1,%xmm6+ vaesenc %xmm15,%xmm14,%xmm14+ vpxor 64(%rdi),%xmm1,%xmm7+ vpxor 80(%rdi),%xmm1,%xmm3+ vmovdqu (%r8),%xmm1++ vaesenclast %xmm2,%xmm9,%xmm9+ vmovdqu 32(%r11),%xmm2+ vaesenclast %xmm0,%xmm10,%xmm10+ vpaddb %xmm2,%xmm1,%xmm0+ movq %r13,112+8(%rsp)+ leaq 96(%rdi),%rdi+ vaesenclast %xmm5,%xmm11,%xmm11+ vpaddb %xmm2,%xmm0,%xmm5+ movq %r12,120+8(%rsp)+ leaq 96(%rsi),%rsi+ vmovdqu 0-128(%rcx),%xmm15+ vaesenclast %xmm6,%xmm12,%xmm12+ vpaddb %xmm2,%xmm5,%xmm6+ vaesenclast %xmm7,%xmm13,%xmm13+ vpaddb %xmm2,%xmm6,%xmm7+ vaesenclast %xmm3,%xmm14,%xmm14+ vpaddb %xmm2,%xmm7,%xmm3++ addq $0x60,%rax+ subq $0x6,%rdx+ jc .L6x_done++ vmovups %xmm9,-96(%rsi)+ vpxor %xmm15,%xmm1,%xmm9+ vmovups %xmm10,-80(%rsi)+ vmovdqa %xmm0,%xmm10+ vmovups %xmm11,-64(%rsi)+ vmovdqa %xmm5,%xmm11+ vmovups %xmm12,-48(%rsi)+ vmovdqa %xmm6,%xmm12+ vmovups %xmm13,-32(%rsi)+ vmovdqa %xmm7,%xmm13+ vmovups %xmm14,-16(%rsi)+ vmovdqa %xmm3,%xmm14+ vmovdqu 32+8(%rsp),%xmm7+ jmp .Loop6x++.L6x_done:+ vpxor 16+8(%rsp),%xmm8,%xmm8+ vpxor %xmm4,%xmm8,%xmm8++ .byte 0xf3,0xc3+++.globl crypton_gcm_asm_decrypt+.def crypton_gcm_asm_decrypt; .scl 2; .type 32; .endef+.p2align 5+crypton_gcm_asm_decrypt:+ .byte 0xf3,0x0f,0x1e,0xfa+ movq %rdi,8(%rsp)+ movq %rsi,16(%rsp)+ movq %rsp,%r11+.LSEH_begin_crypton_gcm_asm_decrypt:+++ pushq %rbp++ movq %rsp,%rbp++ movq %rcx,%rdi+ movq %rdx,%rsi+ movq %r8,%rdx+ movq %r9,%rcx+ movq 48(%rsp),%r8+ movq 56(%rsp),%r9+ xorq %rax,%rax+ cmpq $0x60,%rdx+ jb .Lgcm_dec_abort++ pushq %rbx++ pushq %r12++ pushq %r13++ pushq %r14++ pushq %r15++ leaq -168(%rsp),%rsp++ movaps %xmm6,-208(%rbp)+ movaps %xmm7,-192(%rbp)+ movaps %xmm8,-176(%rbp)+ movaps %xmm9,-160(%rbp)+ movaps %xmm10,-144(%rbp)+ movaps %xmm11,-128(%rbp)+ movaps %xmm12,-112(%rbp)+ movaps %xmm13,-96(%rbp)+ movaps %xmm14,-80(%rbp)+ movaps %xmm15,-64(%rbp)++.LSEH_body_crypton_gcm_asm_decrypt:++ vzeroupper++ vmovdqu (%r8),%xmm1+ addq $-128,%rsp+ movl 12(%r8),%ebx+ leaq .Lbswap_mask(%rip),%r11+ leaq -128(%rcx),%r14+ movq $0xf80,%r15+ vmovdqu (%r9),%xmm8+ andq $-128,%rsp+ vmovdqu (%r11),%xmm0+ leaq 128(%rcx),%rcx+ leaq 32+32(%r9),%r9+ movl 240-128(%rcx),%r10d+ vpshufb %xmm0,%xmm8,%xmm8++ andq %r15,%r14+ andq %rsp,%r15+ subq %r14,%r15+ jc .Ldec_no_key_aliasing+ cmpq $768,%r15+ jnc .Ldec_no_key_aliasing+ subq %r15,%rsp+.Ldec_no_key_aliasing:++ vmovdqu 80(%rdi),%xmm7+ leaq (%rdi),%r14+ vmovdqu 64(%rdi),%xmm4+ leaq -192(%rdi,%rdx,1),%r15+ vmovdqu 48(%rdi),%xmm5+ shrq $4,%rdx+ xorq %rax,%rax+ vmovdqu 32(%rdi),%xmm6+ vpshufb %xmm0,%xmm7,%xmm7+ vmovdqu 16(%rdi),%xmm2+ vpshufb %xmm0,%xmm4,%xmm4+ vmovdqu (%rdi),%xmm3+ vpshufb %xmm0,%xmm5,%xmm5+ vmovdqu %xmm4,48(%rsp)+ vpshufb %xmm0,%xmm6,%xmm6+ vmovdqu %xmm5,64(%rsp)+ vpshufb %xmm0,%xmm2,%xmm2+ vmovdqu %xmm6,80(%rsp)+ vpshufb %xmm0,%xmm3,%xmm3+ vmovdqu %xmm2,96(%rsp)+ vmovdqu %xmm3,112(%rsp)++ call _crypton_gcm_asm_ctr32_ghash_6x++ vmovups %xmm9,-96(%rsi)+ vmovups %xmm10,-80(%rsi)+ vmovups %xmm11,-64(%rsi)+ vmovups %xmm12,-48(%rsi)+ vmovups %xmm13,-32(%rsi)+ vmovups %xmm14,-16(%rsi)++ vpshufb (%r11),%xmm8,%xmm8+ vmovdqu %xmm8,-64(%r9)++ vzeroupper+ movaps -208(%rbp),%xmm6+ movaps -192(%rbp),%xmm7+ movaps -176(%rbp),%xmm8+ movaps -160(%rbp),%xmm9+ movaps -144(%rbp),%xmm10+ movaps -128(%rbp),%xmm11+ movaps -112(%rbp),%xmm12+ movaps -96(%rbp),%xmm13+ movaps -80(%rbp),%xmm14+ movaps -64(%rbp),%xmm15+ movq -40(%rbp),%r15+ movq -32(%rbp),%r14+ movq -24(%rbp),%r13+ movq -16(%rbp),%r12+ movq -8(%rbp),%rbx+ movq %rbp,%rsp++.Lgcm_dec_abort:+ popq %rbp++.LSEH_epilogue_crypton_gcm_asm_decrypt:+ mov 8(%rsp),%rdi+ mov 16(%rsp),%rsi++ .byte 0xf3,0xc3++.LSEH_end_crypton_gcm_asm_decrypt:+.def _crypton_gcm_asm_ctr32_6x; .scl 3; .type 32; .endef+.p2align 5+_crypton_gcm_asm_ctr32_6x:+ .byte 0xf3,0x0f,0x1e,0xfa+++ vmovdqu 0-128(%rcx),%xmm4+ vmovdqu 32(%r11),%xmm2+ leaq -1(%r10),%r13+ vmovups 16-128(%rcx),%xmm15+ leaq 32-128(%rcx),%r12+ vpxor %xmm4,%xmm1,%xmm9+ addl $100663296,%ebx+ jc .Lhandle_ctr32_2+ vpaddb %xmm2,%xmm1,%xmm10+ vpaddb %xmm2,%xmm10,%xmm11+ vpxor %xmm4,%xmm10,%xmm10+ vpaddb %xmm2,%xmm11,%xmm12+ vpxor %xmm4,%xmm11,%xmm11+ vpaddb %xmm2,%xmm12,%xmm13+ vpxor %xmm4,%xmm12,%xmm12+ vpaddb %xmm2,%xmm13,%xmm14+ vpxor %xmm4,%xmm13,%xmm13+ vpaddb %xmm2,%xmm14,%xmm1+ vpxor %xmm4,%xmm14,%xmm14+ jmp .Loop_ctr32++.p2align 4+.Loop_ctr32:+ vaesenc %xmm15,%xmm9,%xmm9+ vaesenc %xmm15,%xmm10,%xmm10+ vaesenc %xmm15,%xmm11,%xmm11+ vaesenc %xmm15,%xmm12,%xmm12+ vaesenc %xmm15,%xmm13,%xmm13+ vaesenc %xmm15,%xmm14,%xmm14+ vmovups (%r12),%xmm15+ leaq 16(%r12),%r12+ decl %r13d+ jnz .Loop_ctr32++ vmovdqu (%r12),%xmm3+ vaesenc %xmm15,%xmm9,%xmm9+ vpxor 0(%rdi),%xmm3,%xmm4+ vaesenc %xmm15,%xmm10,%xmm10+ vpxor 16(%rdi),%xmm3,%xmm5+ vaesenc %xmm15,%xmm11,%xmm11+ vpxor 32(%rdi),%xmm3,%xmm6+ vaesenc %xmm15,%xmm12,%xmm12+ vpxor 48(%rdi),%xmm3,%xmm8+ vaesenc %xmm15,%xmm13,%xmm13+ vpxor 64(%rdi),%xmm3,%xmm2+ vaesenc %xmm15,%xmm14,%xmm14+ vpxor 80(%rdi),%xmm3,%xmm3+ leaq 96(%rdi),%rdi++ vaesenclast %xmm4,%xmm9,%xmm9+ vaesenclast %xmm5,%xmm10,%xmm10+ vaesenclast %xmm6,%xmm11,%xmm11+ vaesenclast %xmm8,%xmm12,%xmm12+ vaesenclast %xmm2,%xmm13,%xmm13+ vaesenclast %xmm3,%xmm14,%xmm14+ vmovups %xmm9,0(%rsi)+ vmovups %xmm10,16(%rsi)+ vmovups %xmm11,32(%rsi)+ vmovups %xmm12,48(%rsi)+ vmovups %xmm13,64(%rsi)+ vmovups %xmm14,80(%rsi)+ leaq 96(%rsi),%rsi++ .byte 0xf3,0xc3+.p2align 5+.Lhandle_ctr32_2:+ vpshufb %xmm0,%xmm1,%xmm6+ vmovdqu 48(%r11),%xmm5+ vpaddd 64(%r11),%xmm6,%xmm10+ vpaddd %xmm5,%xmm6,%xmm11+ vpaddd %xmm5,%xmm10,%xmm12+ vpshufb %xmm0,%xmm10,%xmm10+ vpaddd %xmm5,%xmm11,%xmm13+ vpshufb %xmm0,%xmm11,%xmm11+ vpxor %xmm4,%xmm10,%xmm10+ vpaddd %xmm5,%xmm12,%xmm14+ vpshufb %xmm0,%xmm12,%xmm12+ vpxor %xmm4,%xmm11,%xmm11+ vpaddd %xmm5,%xmm13,%xmm1+ vpshufb %xmm0,%xmm13,%xmm13+ vpxor %xmm4,%xmm12,%xmm12+ vpshufb %xmm0,%xmm14,%xmm14+ vpxor %xmm4,%xmm13,%xmm13+ vpshufb %xmm0,%xmm1,%xmm1+ vpxor %xmm4,%xmm14,%xmm14+ jmp .Loop_ctr32++++.globl crypton_gcm_asm_encrypt+.def crypton_gcm_asm_encrypt; .scl 2; .type 32; .endef+.p2align 5+crypton_gcm_asm_encrypt:+ .byte 0xf3,0x0f,0x1e,0xfa+ movq %rdi,8(%rsp)+ movq %rsi,16(%rsp)+ movq %rsp,%r11+.LSEH_begin_crypton_gcm_asm_encrypt:+++ pushq %rbp++ movq %rsp,%rbp++ movq %rcx,%rdi+ movq %rdx,%rsi+ movq %r8,%rdx+ movq %r9,%rcx+ movq 48(%rsp),%r8+ movq 56(%rsp),%r9+ xorq %rax,%rax+ cmpq $288,%rdx+ jb .Lgcm_enc_abort++ pushq %rbx++ pushq %r12++ pushq %r13++ pushq %r14++ pushq %r15++ leaq -168(%rsp),%rsp++ movaps %xmm6,-208(%rbp)+ movaps %xmm7,-192(%rbp)+ movaps %xmm8,-176(%rbp)+ movaps %xmm9,-160(%rbp)+ movaps %xmm10,-144(%rbp)+ movaps %xmm11,-128(%rbp)+ movaps %xmm12,-112(%rbp)+ movaps %xmm13,-96(%rbp)+ movaps %xmm14,-80(%rbp)+ movaps %xmm15,-64(%rbp)++.LSEH_body_crypton_gcm_asm_encrypt:++ vzeroupper++ vmovdqu (%r8),%xmm1+ addq $-128,%rsp+ movl 12(%r8),%ebx+ leaq .Lbswap_mask(%rip),%r11+ leaq -128(%rcx),%r14+ movq $0xf80,%r15+ leaq 128(%rcx),%rcx+ vmovdqu (%r11),%xmm0+ andq $-128,%rsp+ movl 240-128(%rcx),%r10d++ andq %r15,%r14+ andq %rsp,%r15+ subq %r14,%r15+ jc .Lenc_no_key_aliasing+ cmpq $768,%r15+ jnc .Lenc_no_key_aliasing+ subq %r15,%rsp+.Lenc_no_key_aliasing:++ leaq (%rsi),%r14+ leaq -192(%rsi,%rdx,1),%r15+ shrq $4,%rdx++ call _crypton_gcm_asm_ctr32_6x+ vpshufb %xmm0,%xmm9,%xmm8+ vpshufb %xmm0,%xmm10,%xmm2+ vmovdqu %xmm8,112(%rsp)+ vpshufb %xmm0,%xmm11,%xmm4+ vmovdqu %xmm2,96(%rsp)+ vpshufb %xmm0,%xmm12,%xmm5+ vmovdqu %xmm4,80(%rsp)+ vpshufb %xmm0,%xmm13,%xmm6+ vmovdqu %xmm5,64(%rsp)+ vpshufb %xmm0,%xmm14,%xmm7+ vmovdqu %xmm6,48(%rsp)++ call _crypton_gcm_asm_ctr32_6x++ vmovdqu (%r9),%xmm8+ leaq 32+32(%r9),%r9+ subq $12,%rdx+ movq $192,%rax+ vpshufb %xmm0,%xmm8,%xmm8++ call _crypton_gcm_asm_ctr32_ghash_6x+ vmovdqu 32(%rsp),%xmm7+ vmovdqu (%r11),%xmm0+ vmovdqu 0-32(%r9),%xmm3+ vpunpckhqdq %xmm7,%xmm7,%xmm1+ vmovdqu 32-32(%r9),%xmm15+ vmovups %xmm9,-96(%rsi)+ vpshufb %xmm0,%xmm9,%xmm9+ vpxor %xmm7,%xmm1,%xmm1+ vmovups %xmm10,-80(%rsi)+ vpshufb %xmm0,%xmm10,%xmm10+ vmovups %xmm11,-64(%rsi)+ vpshufb %xmm0,%xmm11,%xmm11+ vmovups %xmm12,-48(%rsi)+ vpshufb %xmm0,%xmm12,%xmm12+ vmovups %xmm13,-32(%rsi)+ vpshufb %xmm0,%xmm13,%xmm13+ vmovups %xmm14,-16(%rsi)+ vpshufb %xmm0,%xmm14,%xmm14+ vmovdqu %xmm9,16(%rsp)+ vmovdqu 48(%rsp),%xmm6+ vmovdqu 16-32(%r9),%xmm0+ vpunpckhqdq %xmm6,%xmm6,%xmm2+ vpclmulqdq $0x00,%xmm3,%xmm7,%xmm5+ vpxor %xmm6,%xmm2,%xmm2+ vpclmulqdq $0x11,%xmm3,%xmm7,%xmm7+ vpclmulqdq $0x00,%xmm15,%xmm1,%xmm1++ vmovdqu 64(%rsp),%xmm9+ vpclmulqdq $0x00,%xmm0,%xmm6,%xmm4+ vmovdqu 48-32(%r9),%xmm3+ vpxor %xmm5,%xmm4,%xmm4+ vpunpckhqdq %xmm9,%xmm9,%xmm5+ vpclmulqdq $0x11,%xmm0,%xmm6,%xmm6+ vpxor %xmm9,%xmm5,%xmm5+ vpxor %xmm7,%xmm6,%xmm6+ vpclmulqdq $0x10,%xmm15,%xmm2,%xmm2+ vmovdqu 80-32(%r9),%xmm15+ vpxor %xmm1,%xmm2,%xmm2++ vmovdqu 80(%rsp),%xmm1+ vpclmulqdq $0x00,%xmm3,%xmm9,%xmm7+ vmovdqu 64-32(%r9),%xmm0+ vpxor %xmm4,%xmm7,%xmm7+ vpunpckhqdq %xmm1,%xmm1,%xmm4+ vpclmulqdq $0x11,%xmm3,%xmm9,%xmm9+ vpxor %xmm1,%xmm4,%xmm4+ vpxor %xmm6,%xmm9,%xmm9+ vpclmulqdq $0x00,%xmm15,%xmm5,%xmm5+ vpxor %xmm2,%xmm5,%xmm5++ vmovdqu 96(%rsp),%xmm2+ vpclmulqdq $0x00,%xmm0,%xmm1,%xmm6+ vmovdqu 96-32(%r9),%xmm3+ vpxor %xmm7,%xmm6,%xmm6+ vpunpckhqdq %xmm2,%xmm2,%xmm7+ vpclmulqdq $0x11,%xmm0,%xmm1,%xmm1+ vpxor %xmm2,%xmm7,%xmm7+ vpxor %xmm9,%xmm1,%xmm1+ vpclmulqdq $0x10,%xmm15,%xmm4,%xmm4+ vmovdqu 128-32(%r9),%xmm15+ vpxor %xmm5,%xmm4,%xmm4++ vpxor 112(%rsp),%xmm8,%xmm8+ vpclmulqdq $0x00,%xmm3,%xmm2,%xmm5+ vmovdqu 112-32(%r9),%xmm0+ vpunpckhqdq %xmm8,%xmm8,%xmm9+ vpxor %xmm6,%xmm5,%xmm5+ vpclmulqdq $0x11,%xmm3,%xmm2,%xmm2+ vpxor %xmm8,%xmm9,%xmm9+ vpxor %xmm1,%xmm2,%xmm2+ vpclmulqdq $0x00,%xmm15,%xmm7,%xmm7+ vpxor %xmm4,%xmm7,%xmm4++ vpclmulqdq $0x00,%xmm0,%xmm8,%xmm6+ vmovdqu 0-32(%r9),%xmm3+ vpunpckhqdq %xmm14,%xmm14,%xmm1+ vpclmulqdq $0x11,%xmm0,%xmm8,%xmm8+ vpxor %xmm14,%xmm1,%xmm1+ vpxor %xmm5,%xmm6,%xmm5+ vpclmulqdq $0x10,%xmm15,%xmm9,%xmm9+ vmovdqu 32-32(%r9),%xmm15+ vpxor %xmm2,%xmm8,%xmm7+ vpxor %xmm4,%xmm9,%xmm6++ vmovdqu 16-32(%r9),%xmm0+ vpxor %xmm5,%xmm7,%xmm9+ vpclmulqdq $0x00,%xmm3,%xmm14,%xmm4+ vpxor %xmm9,%xmm6,%xmm6+ vpunpckhqdq %xmm13,%xmm13,%xmm2+ vpclmulqdq $0x11,%xmm3,%xmm14,%xmm14+ vpxor %xmm13,%xmm2,%xmm2+ vpslldq $8,%xmm6,%xmm9+ vpclmulqdq $0x00,%xmm15,%xmm1,%xmm1+ vpxor %xmm9,%xmm5,%xmm8+ vpsrldq $8,%xmm6,%xmm6+ vpxor %xmm6,%xmm7,%xmm7++ vpclmulqdq $0x00,%xmm0,%xmm13,%xmm5+ vmovdqu 48-32(%r9),%xmm3+ vpxor %xmm4,%xmm5,%xmm5+ vpunpckhqdq %xmm12,%xmm12,%xmm9+ vpclmulqdq $0x11,%xmm0,%xmm13,%xmm13+ vpxor %xmm12,%xmm9,%xmm9+ vpxor %xmm14,%xmm13,%xmm13+ vpalignr $8,%xmm8,%xmm8,%xmm14+ vpclmulqdq $0x10,%xmm15,%xmm2,%xmm2+ vmovdqu 80-32(%r9),%xmm15+ vpxor %xmm1,%xmm2,%xmm2++ vpclmulqdq $0x00,%xmm3,%xmm12,%xmm4+ vmovdqu 64-32(%r9),%xmm0+ vpxor %xmm5,%xmm4,%xmm4+ vpunpckhqdq %xmm11,%xmm11,%xmm1+ vpclmulqdq $0x11,%xmm3,%xmm12,%xmm12+ vpxor %xmm11,%xmm1,%xmm1+ vpxor %xmm13,%xmm12,%xmm12+ vxorps 16(%rsp),%xmm7,%xmm7+ vpclmulqdq $0x00,%xmm15,%xmm9,%xmm9+ vpxor %xmm2,%xmm9,%xmm9++ vpclmulqdq $0x10,16(%r11),%xmm8,%xmm8+ vxorps %xmm14,%xmm8,%xmm8++ vpclmulqdq $0x00,%xmm0,%xmm11,%xmm5+ vmovdqu 96-32(%r9),%xmm3+ vpxor %xmm4,%xmm5,%xmm5+ vpunpckhqdq %xmm10,%xmm10,%xmm2+ vpclmulqdq $0x11,%xmm0,%xmm11,%xmm11+ vpxor %xmm10,%xmm2,%xmm2+ vpalignr $8,%xmm8,%xmm8,%xmm14+ vpxor %xmm12,%xmm11,%xmm11+ vpclmulqdq $0x10,%xmm15,%xmm1,%xmm1+ vmovdqu 128-32(%r9),%xmm15+ vpxor %xmm9,%xmm1,%xmm1++ vxorps %xmm7,%xmm14,%xmm14+ vpclmulqdq $0x10,16(%r11),%xmm8,%xmm8+ vxorps %xmm14,%xmm8,%xmm8++ vpclmulqdq $0x00,%xmm3,%xmm10,%xmm4+ vmovdqu 112-32(%r9),%xmm0+ vpxor %xmm5,%xmm4,%xmm4+ vpunpckhqdq %xmm8,%xmm8,%xmm9+ vpclmulqdq $0x11,%xmm3,%xmm10,%xmm10+ vpxor %xmm8,%xmm9,%xmm9+ vpxor %xmm11,%xmm10,%xmm10+ vpclmulqdq $0x00,%xmm15,%xmm2,%xmm2+ vpxor %xmm1,%xmm2,%xmm2++ vpclmulqdq $0x00,%xmm0,%xmm8,%xmm5+ vpclmulqdq $0x11,%xmm0,%xmm8,%xmm7+ vpxor %xmm4,%xmm5,%xmm5+ vpclmulqdq $0x10,%xmm15,%xmm9,%xmm6+ vpxor %xmm10,%xmm7,%xmm7+ vpxor %xmm2,%xmm6,%xmm6++ vpxor %xmm5,%xmm7,%xmm4+ vpxor %xmm4,%xmm6,%xmm6+ vpslldq $8,%xmm6,%xmm1+ vmovdqu 16(%r11),%xmm3+ vpsrldq $8,%xmm6,%xmm6+ vpxor %xmm1,%xmm5,%xmm8+ vpxor %xmm6,%xmm7,%xmm7++ vpalignr $8,%xmm8,%xmm8,%xmm2+ vpclmulqdq $0x10,%xmm3,%xmm8,%xmm8+ vpxor %xmm2,%xmm8,%xmm8++ vpalignr $8,%xmm8,%xmm8,%xmm2+ vpclmulqdq $0x10,%xmm3,%xmm8,%xmm8+ vpxor %xmm7,%xmm2,%xmm2+ vpxor %xmm2,%xmm8,%xmm8+ vpshufb (%r11),%xmm8,%xmm8+ vmovdqu %xmm8,-64(%r9)++ vzeroupper+ movaps -208(%rbp),%xmm6+ movaps -192(%rbp),%xmm7+ movaps -176(%rbp),%xmm8+ movaps -160(%rbp),%xmm9+ movaps -144(%rbp),%xmm10+ movaps -128(%rbp),%xmm11+ movaps -112(%rbp),%xmm12+ movaps -96(%rbp),%xmm13+ movaps -80(%rbp),%xmm14+ movaps -64(%rbp),%xmm15+ movq -40(%rbp),%r15+ movq -32(%rbp),%r14+ movq -24(%rbp),%r13+ movq -16(%rbp),%r12+ movq -8(%rbp),%rbx+ movq %rbp,%rsp++.Lgcm_enc_abort:+ popq %rbp++.LSEH_epilogue_crypton_gcm_asm_encrypt:+ mov 8(%rsp),%rdi+ mov 16(%rsp),%rsi++ .byte 0xf3,0xc3++.LSEH_end_crypton_gcm_asm_encrypt:+.p2align 6+.Lbswap_mask:+.byte 15,14,13,12,11,10,9,8,7,6,5,4,3,2,1,0+.Lpoly:+.byte 0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0xc2+.Lone_msb:+.byte 0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,1+.Ltwo_lsb:+.byte 2,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0+.Lone_lsb:+.byte 1,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0+.byte 65,69,83,45,78,73,32,71,67,77,32,109,111,100,117,108,101,32,102,111,114,32,120,56,54,95,54,52,44,32,67,82,89,80,84,79,71,65,77,83,32,98,121,32,60,97,112,112,114,111,64,111,112,101,110,115,115,108,46,111,114,103,62,0+.p2align 6+.section .pdata+.p2align 2+.rva .LSEH_begin_crypton_gcm_asm_decrypt+.rva .LSEH_body_crypton_gcm_asm_decrypt+.rva .LSEH_info_crypton_gcm_asm_decrypt_prologue++.rva .LSEH_body_crypton_gcm_asm_decrypt+.rva .LSEH_epilogue_crypton_gcm_asm_decrypt+.rva .LSEH_info_crypton_gcm_asm_decrypt_body++.rva .LSEH_epilogue_crypton_gcm_asm_decrypt+.rva .LSEH_end_crypton_gcm_asm_decrypt+.rva .LSEH_info_crypton_gcm_asm_decrypt_epilogue++.rva .LSEH_begin_crypton_gcm_asm_encrypt+.rva .LSEH_body_crypton_gcm_asm_encrypt+.rva .LSEH_info_crypton_gcm_asm_encrypt_prologue++.rva .LSEH_body_crypton_gcm_asm_encrypt+.rva .LSEH_epilogue_crypton_gcm_asm_encrypt+.rva .LSEH_info_crypton_gcm_asm_encrypt_body++.rva .LSEH_epilogue_crypton_gcm_asm_encrypt+.rva .LSEH_end_crypton_gcm_asm_encrypt+.rva .LSEH_info_crypton_gcm_asm_encrypt_epilogue++.section .xdata+.p2align 3+.LSEH_info_crypton_gcm_asm_decrypt_prologue:+.byte 1,4,6,0x05+.byte 4,0x74,2,0+.byte 4,0x64,3,0+.byte 4,0x53+.byte 1,0x50+.long 0,0+.LSEH_info_crypton_gcm_asm_decrypt_body:+.byte 1,0,38,213+.byte 0x00,0x68,0x00,0x00+.byte 0x00,0x78,0x01,0x00+.byte 0x00,0x88,0x02,0x00+.byte 0x00,0x98,0x03,0x00+.byte 0x00,0xa8,0x04,0x00+.byte 0x00,0xb8,0x05,0x00+.byte 0x00,0xc8,0x06,0x00+.byte 0x00,0xd8,0x07,0x00+.byte 0x00,0xe8,0x08,0x00+.byte 0x00,0xf8,0x09,0x00+.byte 0x00,0xf4,0x15,0x00+.byte 0x00,0xe4,0x16,0x00+.byte 0x00,0xd4,0x17,0x00+.byte 0x00,0xc4,0x18,0x00+.byte 0x00,0x34,0x19,0x00+.byte 0x00,0x74,0x1c,0x00+.byte 0x00,0x64,0x1d,0x00+.byte 0x00,0x53+.byte 0x00,0x01,0x1a,0x00+.byte 0x00,0x50+.byte 0x00,0x00,0x00,0x00+.byte 0x00,0x00,0x00,0x00+.LSEH_info_crypton_gcm_asm_decrypt_epilogue:+.byte 1,0,4,0+.byte 0x00,0x74,0x01,0x00+.byte 0x00,0x64,0x02,0x00+.byte 0x00,0x00,0x00,0x00++.LSEH_info_crypton_gcm_asm_encrypt_prologue:+.byte 1,4,6,0x05+.byte 4,0x74,2,0+.byte 4,0x64,3,0+.byte 4,0x53+.byte 1,0x50+.long 0,0+.LSEH_info_crypton_gcm_asm_encrypt_body:+.byte 1,0,38,213+.byte 0x00,0x68,0x00,0x00+.byte 0x00,0x78,0x01,0x00+.byte 0x00,0x88,0x02,0x00+.byte 0x00,0x98,0x03,0x00+.byte 0x00,0xa8,0x04,0x00+.byte 0x00,0xb8,0x05,0x00+.byte 0x00,0xc8,0x06,0x00+.byte 0x00,0xd8,0x07,0x00+.byte 0x00,0xe8,0x08,0x00+.byte 0x00,0xf8,0x09,0x00+.byte 0x00,0xf4,0x15,0x00+.byte 0x00,0xe4,0x16,0x00+.byte 0x00,0xd4,0x17,0x00+.byte 0x00,0xc4,0x18,0x00+.byte 0x00,0x34,0x19,0x00+.byte 0x00,0x74,0x1c,0x00+.byte 0x00,0x64,0x1d,0x00+.byte 0x00,0x53+.byte 0x00,0x01,0x1a,0x00+.byte 0x00,0x50+.byte 0x00,0x00,0x00,0x00+.byte 0x00,0x00,0x00,0x00+.LSEH_info_crypton_gcm_asm_encrypt_epilogue:+.byte 1,0,4,0+.byte 0x00,0x74,0x01,0x00+.byte 0x00,0x64,0x02,0x00+.byte 0x00,0x00,0x00,0x00+
@@ -0,0 +1,974 @@+#! /usr/bin/env perl+#+# ====================================================================+# Written by Andy Polyakov <appro@openssl.org> for the OpenSSL+# project. The module is, however, dual licensed under OpenSSL and+# CRYPTOGAMS licenses depending on where you obtain it. For further+# details see http://www.openssl.org/~appro/cryptogams/.+# ====================================================================+#+#+# AES-NI-CTR+GHASH stitch.+#+# February 2013+#+# OpenSSL GCM implementation is organized in such way that its+# performance is rather close to the sum of its streamed components,+# in the context parallelized AES-NI CTR and modulo-scheduled+# PCLMULQDQ-enabled GHASH. Unfortunately, as no stitch implementation+# was observed to perform significantly better than the sum of the+# components on contemporary CPUs, the effort was deemed impossible to+# justify. This module is based on combination of Intel submissions,+# [1] and [2], with MOVBE twist suggested by Ilya Albrekht and Max+# Locktyukhin of Intel Corp. who verified that it reduces shuffles+# pressure with notable relative improvement, achieving 1.0 cycle per+# byte processed with 128-bit key on Haswell processor, 0.74 - on+# Broadwell, 0.63 - on Skylake... [Mentioned results are raw profiled+# measurements for favourable packet size, one divisible by 96.+# Applications using the EVP interface will observe a few percent+# worse performance.]+#+# Knights Landing processes 1 byte in 1.25 cycles (measured with EVP).+#+# [1] http://rt.openssl.org/Ticket/Display.html?id=2900&user=guest&pass=guest+# [2] http://www.intel.com/content/dam/www/public/us/en/documents/software-support/enabling-high-performance-gcm.pdf++$flavour = shift;+$output = shift;+if ($flavour =~ /\./) { $output = $flavour; undef $flavour; }++$win64=0; $win64=1 if ($flavour =~ /[nm]asm|mingw64/ || $output =~ /\.asm$/);++$0 =~ m/(.*[\/\\])[^\/\\]+$/; $dir=$1;+( $xlate="${dir}x86_64-xlate.pl" and -f $xlate ) or+( $xlate="${dir}../../perlasm/x86_64-xlate.pl" and -f $xlate) or+die "can't locate x86_64-xlate.pl";++$ENV{CC} //= "cc";+if (`$ENV{CC} -Wa,-v -c -o /dev/null -x assembler /dev/null 2>&1`+ =~ /GNU assembler version ([2-9]\.[0-9]+)/) {+ $avx = ($1>=2.20) + ($1>=2.22);+}++if (!$avx && $win64 && ($flavour =~ /nasm/ || $ENV{ASM} =~ /nasm/) &&+ `nasm -v 2>&1` =~ /NASM version ([2-9]\.[0-9]+)/) {+ $avx = ($1>=2.09) + ($1>=2.10);+}++if (!$avx && $win64 && ($flavour =~ /masm/ || $ENV{ASM} =~ /ml64/) &&+ `ml64 2>&1` =~ /Version ([0-9]+)\./) {+ $avx = ($1>=10) + ($1>=11);+}++if (!$avx && `$ENV{CC} -v 2>&1` =~ /((?:clang|LLVM) version|.*based on LLVM) ([0-9]+\.[0-9]+)/) {+ $avx = ($2>=3.0) + ($2>3.0);+}++open OUT,"| \"$^X\" \"$xlate\" $flavour \"$output\"";+*STDOUT=*OUT;++if ($avx>1) {{{++($inp,$out,$len,$key,$ivp,$Xip)=("%rdi","%rsi","%rdx","%rcx","%r8","%r9");++($Ii,$T1,$T2,$Hkey,+ $Z0,$Z1,$Z2,$Z3,$Xi) = map("%xmm$_",(0..8));++($inout0,$inout1,$inout2,$inout3,$inout4,$inout5,$rndkey) = map("%xmm$_",(9..15));++($counter,$rounds,$ret,$const,$in0,$end0)=("%ebx","%r10d","%rax","%r11","%r14","%r15");++$code=<<___;+.text++.type _aesni_ctr32_ghash_6x,\@abi-omnipotent+.align 32+_aesni_ctr32_ghash_6x:+.cfi_startproc+ vmovdqu 0x20($const),$T2 # borrow $T2, .Lone_msb+ sub \$6,$len+ vpxor $Z0,$Z0,$Z0 # $Z0 = 0+ vmovdqu 0x00-0x80($key),$rndkey+ vpaddb $T2,$T1,$inout1+ vpaddb $T2,$inout1,$inout2+ vpaddb $T2,$inout2,$inout3+ vpaddb $T2,$inout3,$inout4+ vpaddb $T2,$inout4,$inout5+ vpxor $rndkey,$T1,$inout0+ vmovdqu $Z0,16+8(%rsp) # "$Z3" = 0+ jmp .Loop6x++.align 32+.Loop6x:+ add \$`6<<24`,$counter+ jc .Lhandle_ctr32 # discard $inout[1-5]?+ vmovdqu 0x00-0x20($Xip),$Hkey # $Hkey^1+ vpaddb $T2,$inout5,$T1 # next counter value+ vpxor $rndkey,$inout1,$inout1+ vpxor $rndkey,$inout2,$inout2++.Lresume_ctr32:+ vmovdqu $T1,($ivp) # save next counter value+ vpclmulqdq \$0x10,$Hkey,$Z3,$Z1+ vpxor $rndkey,$inout3,$inout3+ vmovups 0x10-0x80($key),$T2 # borrow $T2 for $rndkey+ vpclmulqdq \$0x01,$Hkey,$Z3,$Z2+ xor %r12,%r12+ cmp $in0,$end0++ vaesenc $T2,$inout0,$inout0+ vmovdqu 0x30+8(%rsp),$Ii # I[4]+ vpxor $rndkey,$inout4,$inout4+ vpclmulqdq \$0x00,$Hkey,$Z3,$T1+ vaesenc $T2,$inout1,$inout1+ vpxor $rndkey,$inout5,$inout5+ setnc %r12b+ vpclmulqdq \$0x11,$Hkey,$Z3,$Z3+ vaesenc $T2,$inout2,$inout2+ vmovdqu 0x10-0x20($Xip),$Hkey # $Hkey^2+ neg %r12+ vaesenc $T2,$inout3,$inout3+ vpxor $Z1,$Z2,$Z2+ vpclmulqdq \$0x00,$Hkey,$Ii,$Z1+ vpxor $Z0,$Xi,$Xi # modulo-scheduled+ vaesenc $T2,$inout4,$inout4+ vpxor $Z1,$T1,$Z0+ and \$0x60,%r12+ vmovups 0x20-0x80($key),$rndkey+ vpclmulqdq \$0x10,$Hkey,$Ii,$T1+ vaesenc $T2,$inout5,$inout5++ vpclmulqdq \$0x01,$Hkey,$Ii,$T2+ lea ($in0,%r12),$in0+ vaesenc $rndkey,$inout0,$inout0+ vpxor 16+8(%rsp),$Xi,$Xi # modulo-scheduled [vpxor $Z3,$Xi,$Xi]+ vpclmulqdq \$0x11,$Hkey,$Ii,$Hkey+ vmovdqu 0x40+8(%rsp),$Ii # I[3]+ vaesenc $rndkey,$inout1,$inout1+ movbe 0x58($in0),%r13+ vaesenc $rndkey,$inout2,$inout2+ movbe 0x50($in0),%r12+ vaesenc $rndkey,$inout3,$inout3+ mov %r13,0x20+8(%rsp)+ vaesenc $rndkey,$inout4,$inout4+ mov %r12,0x28+8(%rsp)+ vmovdqu 0x30-0x20($Xip),$Z1 # borrow $Z1 for $Hkey^3+ vaesenc $rndkey,$inout5,$inout5++ vmovups 0x30-0x80($key),$rndkey+ vpxor $T1,$Z2,$Z2+ vpclmulqdq \$0x00,$Z1,$Ii,$T1+ vaesenc $rndkey,$inout0,$inout0+ vpxor $T2,$Z2,$Z2+ vpclmulqdq \$0x10,$Z1,$Ii,$T2+ vaesenc $rndkey,$inout1,$inout1+ vpxor $Hkey,$Z3,$Z3+ vpclmulqdq \$0x01,$Z1,$Ii,$Hkey+ vaesenc $rndkey,$inout2,$inout2+ vpclmulqdq \$0x11,$Z1,$Ii,$Z1+ vmovdqu 0x50+8(%rsp),$Ii # I[2]+ vaesenc $rndkey,$inout3,$inout3+ vaesenc $rndkey,$inout4,$inout4+ vpxor $T1,$Z0,$Z0+ vmovdqu 0x40-0x20($Xip),$T1 # borrow $T1 for $Hkey^4+ vaesenc $rndkey,$inout5,$inout5++ vmovups 0x40-0x80($key),$rndkey+ vpxor $T2,$Z2,$Z2+ vpclmulqdq \$0x00,$T1,$Ii,$T2+ vaesenc $rndkey,$inout0,$inout0+ vpxor $Hkey,$Z2,$Z2+ vpclmulqdq \$0x10,$T1,$Ii,$Hkey+ vaesenc $rndkey,$inout1,$inout1+ movbe 0x48($in0),%r13+ vpxor $Z1,$Z3,$Z3+ vpclmulqdq \$0x01,$T1,$Ii,$Z1+ vaesenc $rndkey,$inout2,$inout2+ movbe 0x40($in0),%r12+ vpclmulqdq \$0x11,$T1,$Ii,$T1+ vmovdqu 0x60+8(%rsp),$Ii # I[1]+ vaesenc $rndkey,$inout3,$inout3+ mov %r13,0x30+8(%rsp)+ vaesenc $rndkey,$inout4,$inout4+ mov %r12,0x38+8(%rsp)+ vpxor $T2,$Z0,$Z0+ vmovdqu 0x60-0x20($Xip),$T2 # borrow $T2 for $Hkey^5+ vaesenc $rndkey,$inout5,$inout5++ vmovups 0x50-0x80($key),$rndkey+ vpxor $Hkey,$Z2,$Z2+ vpclmulqdq \$0x00,$T2,$Ii,$Hkey+ vaesenc $rndkey,$inout0,$inout0+ vpxor $Z1,$Z2,$Z2+ vpclmulqdq \$0x10,$T2,$Ii,$Z1+ vaesenc $rndkey,$inout1,$inout1+ movbe 0x38($in0),%r13+ vpxor $T1,$Z3,$Z3+ vpclmulqdq \$0x01,$T2,$Ii,$T1+ vpxor 0x70+8(%rsp),$Xi,$Xi # accumulate I[0]+ vaesenc $rndkey,$inout2,$inout2+ movbe 0x30($in0),%r12+ vpclmulqdq \$0x11,$T2,$Ii,$T2+ vaesenc $rndkey,$inout3,$inout3+ mov %r13,0x40+8(%rsp)+ vaesenc $rndkey,$inout4,$inout4+ mov %r12,0x48+8(%rsp)+ vpxor $Hkey,$Z0,$Z0+ vmovdqu 0x70-0x20($Xip),$Hkey # $Hkey^6+ vaesenc $rndkey,$inout5,$inout5++ vmovups 0x60-0x80($key),$rndkey+ vpxor $Z1,$Z2,$Z2+ vpclmulqdq \$0x10,$Hkey,$Xi,$Z1+ vaesenc $rndkey,$inout0,$inout0+ vpxor $T1,$Z2,$Z2+ vpclmulqdq \$0x01,$Hkey,$Xi,$T1+ vaesenc $rndkey,$inout1,$inout1+ movbe 0x28($in0),%r13+ vpxor $T2,$Z3,$Z3+ vpclmulqdq \$0x00,$Hkey,$Xi,$T2+ vaesenc $rndkey,$inout2,$inout2+ movbe 0x20($in0),%r12+ vpclmulqdq \$0x11,$Hkey,$Xi,$Xi+ vaesenc $rndkey,$inout3,$inout3+ mov %r13,0x50+8(%rsp)+ vaesenc $rndkey,$inout4,$inout4+ mov %r12,0x58+8(%rsp)+ vpxor $Z1,$Z2,$Z2+ vaesenc $rndkey,$inout5,$inout5+ vpxor $T1,$Z2,$Z2++ vmovups 0x70-0x80($key),$rndkey+ vpslldq \$8,$Z2,$Z1+ vpxor $T2,$Z0,$Z0+ vmovdqu 0x10($const),$Hkey # .Lpoly++ vaesenc $rndkey,$inout0,$inout0+ vpxor $Xi,$Z3,$Z3+ vaesenc $rndkey,$inout1,$inout1+ vpxor $Z1,$Z0,$Z0+ movbe 0x18($in0),%r13+ vaesenc $rndkey,$inout2,$inout2+ movbe 0x10($in0),%r12+ vpalignr \$8,$Z0,$Z0,$Ii # 1st phase+ vpclmulqdq \$0x10,$Hkey,$Z0,$Z0+ mov %r13,0x60+8(%rsp)+ vaesenc $rndkey,$inout3,$inout3+ mov %r12,0x68+8(%rsp)+ vaesenc $rndkey,$inout4,$inout4+ vmovups 0x80-0x80($key),$T1 # borrow $T1 for $rndkey+ vaesenc $rndkey,$inout5,$inout5++ vaesenc $T1,$inout0,$inout0+ vmovups 0x90-0x80($key),$rndkey+ vaesenc $T1,$inout1,$inout1+ vpsrldq \$8,$Z2,$Z2+ vaesenc $T1,$inout2,$inout2+ vpxor $Z2,$Z3,$Z3+ vaesenc $T1,$inout3,$inout3+ vpxor $Ii,$Z0,$Z0+ movbe 0x08($in0),%r13+ vaesenc $T1,$inout4,$inout4+ movbe 0x00($in0),%r12+ vaesenc $T1,$inout5,$inout5+ vmovups 0xa0-0x80($key),$T1+ cmp \$11,$rounds+ jb .Lenc_tail # 128-bit key++ vaesenc $rndkey,$inout0,$inout0+ vaesenc $rndkey,$inout1,$inout1+ vaesenc $rndkey,$inout2,$inout2+ vaesenc $rndkey,$inout3,$inout3+ vaesenc $rndkey,$inout4,$inout4+ vaesenc $rndkey,$inout5,$inout5++ vaesenc $T1,$inout0,$inout0+ vaesenc $T1,$inout1,$inout1+ vaesenc $T1,$inout2,$inout2+ vaesenc $T1,$inout3,$inout3+ vaesenc $T1,$inout4,$inout4+ vmovups 0xb0-0x80($key),$rndkey+ vaesenc $T1,$inout5,$inout5+ vmovups 0xc0-0x80($key),$T1+ je .Lenc_tail # 192-bit key++ vaesenc $rndkey,$inout0,$inout0+ vaesenc $rndkey,$inout1,$inout1+ vaesenc $rndkey,$inout2,$inout2+ vaesenc $rndkey,$inout3,$inout3+ vaesenc $rndkey,$inout4,$inout4+ vaesenc $rndkey,$inout5,$inout5++ vaesenc $T1,$inout0,$inout0+ vaesenc $T1,$inout1,$inout1+ vaesenc $T1,$inout2,$inout2+ vaesenc $T1,$inout3,$inout3+ vaesenc $T1,$inout4,$inout4+ vmovups 0xd0-0x80($key),$rndkey+ vaesenc $T1,$inout5,$inout5+ vmovups 0xe0-0x80($key),$T1+ jmp .Lenc_tail # 256-bit key++.align 32+.Lhandle_ctr32:+ vmovdqu ($const),$Ii # borrow $Ii for .Lbswap_mask+ vpshufb $Ii,$T1,$Z2 # byte-swap counter+ vmovdqu 0x30($const),$Z1 # borrow $Z1, .Ltwo_lsb+ vpaddd 0x40($const),$Z2,$inout1 # .Lone_lsb+ vpaddd $Z1,$Z2,$inout2+ vmovdqu 0x00-0x20($Xip),$Hkey # $Hkey^1+ vpaddd $Z1,$inout1,$inout3+ vpshufb $Ii,$inout1,$inout1+ vpaddd $Z1,$inout2,$inout4+ vpshufb $Ii,$inout2,$inout2+ vpxor $rndkey,$inout1,$inout1+ vpaddd $Z1,$inout3,$inout5+ vpshufb $Ii,$inout3,$inout3+ vpxor $rndkey,$inout2,$inout2+ vpaddd $Z1,$inout4,$T1 # byte-swapped next counter value+ vpshufb $Ii,$inout4,$inout4+ vpshufb $Ii,$inout5,$inout5+ vpshufb $Ii,$T1,$T1 # next counter value+ jmp .Lresume_ctr32++.align 32+.Lenc_tail:+ vaesenc $rndkey,$inout0,$inout0+ vmovdqu $Z3,16+8(%rsp) # postpone vpxor $Z3,$Xi,$Xi+ vpalignr \$8,$Z0,$Z0,$Xi # 2nd phase+ vaesenc $rndkey,$inout1,$inout1+ vpclmulqdq \$0x10,$Hkey,$Z0,$Z0+ vpxor 0x00($inp),$T1,$T2+ vaesenc $rndkey,$inout2,$inout2+ vpxor 0x10($inp),$T1,$Ii+ vaesenc $rndkey,$inout3,$inout3+ vpxor 0x20($inp),$T1,$Z1+ vaesenc $rndkey,$inout4,$inout4+ vpxor 0x30($inp),$T1,$Z2+ vaesenc $rndkey,$inout5,$inout5+ vpxor 0x40($inp),$T1,$Z3+ vpxor 0x50($inp),$T1,$Hkey+ vmovdqu ($ivp),$T1 # load next counter value++ vaesenclast $T2,$inout0,$inout0+ vmovdqu 0x20($const),$T2 # borrow $T2, .Lone_msb+ vaesenclast $Ii,$inout1,$inout1+ vpaddb $T2,$T1,$Ii+ mov %r13,0x70+8(%rsp)+ lea 0x60($inp),$inp+ vaesenclast $Z1,$inout2,$inout2+ vpaddb $T2,$Ii,$Z1+ mov %r12,0x78+8(%rsp)+ lea 0x60($out),$out+ vmovdqu 0x00-0x80($key),$rndkey+ vaesenclast $Z2,$inout3,$inout3+ vpaddb $T2,$Z1,$Z2+ vaesenclast $Z3, $inout4,$inout4+ vpaddb $T2,$Z2,$Z3+ vaesenclast $Hkey,$inout5,$inout5+ vpaddb $T2,$Z3,$Hkey++ add \$0x60,$ret+ sub \$0x6,$len+ jc .L6x_done++ vmovups $inout0,-0x60($out) # save output+ vpxor $rndkey,$T1,$inout0+ vmovups $inout1,-0x50($out)+ vmovdqa $Ii,$inout1 # 0 latency+ vmovups $inout2,-0x40($out)+ vmovdqa $Z1,$inout2 # 0 latency+ vmovups $inout3,-0x30($out)+ vmovdqa $Z2,$inout3 # 0 latency+ vmovups $inout4,-0x20($out)+ vmovdqa $Z3,$inout4 # 0 latency+ vmovups $inout5,-0x10($out)+ vmovdqa $Hkey,$inout5 # 0 latency+ vmovdqu 0x20+8(%rsp),$Z3 # I[5]+ jmp .Loop6x++.L6x_done:+ vpxor 16+8(%rsp),$Xi,$Xi # modulo-scheduled+ vpxor $Z0,$Xi,$Xi # modulo-scheduled++ ret+.cfi_endproc+.size _aesni_ctr32_ghash_6x,.-_aesni_ctr32_ghash_6x+___+######################################################################+#+# size_t aesni_gcm_[en|de]crypt(const void *inp, void *out, size_t len,+# const AES_KEY *key, unsigned char iv[16],+# struct { u128 Xi,H,Htbl[9]; } *Xip);+$code.=<<___;+.globl aesni_gcm_decrypt+.type aesni_gcm_decrypt,\@function,6,"unwind"+.align 32+aesni_gcm_decrypt:+.cfi_startproc+ push %rbp+.cfi_push %rbp+ mov %rsp,%rbp+.cfi_def_cfa_register %rbp+ xor $ret,$ret+ cmp \$0x60,$len # minimal accepted length+ jb .Lgcm_dec_abort++ push %rbx+.cfi_push %rbx+ push %r12+.cfi_push %r12+ push %r13+.cfi_push %r13+ push %r14+.cfi_push %r14+ push %r15+.cfi_push %r15+___+$code.=<<___ if ($win64);+ lea -0xa8(%rsp),%rsp+.cfi_alloca 0xa8+ movaps %xmm6,-0xd0(%rbp)+ movaps %xmm7,-0xc0(%rbp)+ movaps %xmm8,-0xb0(%rbp)+ movaps %xmm9,-0xa0(%rbp)+ movaps %xmm10,-0x90(%rbp)+ movaps %xmm11,-0x80(%rbp)+ movaps %xmm12,-0x70(%rbp)+ movaps %xmm13,-0x60(%rbp)+ movaps %xmm14,-0x50(%rbp)+ movaps %xmm15,-0x40(%rbp)+.cfi_offset %xmm6-%xmm15,-0xe0+___+$code.=<<___;+.cfi_end_prologue+ vzeroupper++ vmovdqu ($ivp),$T1 # input counter value+ add \$-128,%rsp+ mov 12($ivp),$counter+ lea .Lbswap_mask(%rip),$const+ lea -0x80($key),$in0 # borrow $in0+ mov \$0xf80,$end0 # borrow $end0+ vmovdqu ($Xip),$Xi # load Xi+ and \$-128,%rsp # ensure stack alignment+ vmovdqu ($const),$Ii # borrow $Ii for .Lbswap_mask+ lea 0x80($key),$key # size optimization+ lea 0x20+0x20($Xip),$Xip # size optimization+ mov 0xf0-0x80($key),$rounds+ vpshufb $Ii,$Xi,$Xi++ and $end0,$in0+ and %rsp,$end0+ sub $in0,$end0+ jc .Ldec_no_key_aliasing+ cmp \$768,$end0+ jnc .Ldec_no_key_aliasing+ sub $end0,%rsp # avoid aliasing with key+.Ldec_no_key_aliasing:++ vmovdqu 0x50($inp),$Z3 # I[5]+ lea ($inp),$in0+ vmovdqu 0x40($inp),$Z0+ lea -0xc0($inp,$len),$end0+ vmovdqu 0x30($inp),$Z1+ shr \$4,$len+ xor $ret,$ret+ vmovdqu 0x20($inp),$Z2+ vpshufb $Ii,$Z3,$Z3 # passed to _aesni_ctr32_ghash_6x+ vmovdqu 0x10($inp),$T2+ vpshufb $Ii,$Z0,$Z0+ vmovdqu ($inp),$Hkey+ vpshufb $Ii,$Z1,$Z1+ vmovdqu $Z0,0x30(%rsp)+ vpshufb $Ii,$Z2,$Z2+ vmovdqu $Z1,0x40(%rsp)+ vpshufb $Ii,$T2,$T2+ vmovdqu $Z2,0x50(%rsp)+ vpshufb $Ii,$Hkey,$Hkey+ vmovdqu $T2,0x60(%rsp)+ vmovdqu $Hkey,0x70(%rsp)++ call _aesni_ctr32_ghash_6x++ vmovups $inout0,-0x60($out) # save output+ vmovups $inout1,-0x50($out)+ vmovups $inout2,-0x40($out)+ vmovups $inout3,-0x30($out)+ vmovups $inout4,-0x20($out)+ vmovups $inout5,-0x10($out)++ vpshufb ($const),$Xi,$Xi # .Lbswap_mask+ vmovdqu $Xi,-0x40($Xip) # output Xi++ vzeroupper+___+$code.=<<___ if ($win64);+ movaps -0xd0(%rbp),%xmm6+ movaps -0xc0(%rbp),%xmm7+ movaps -0xb0(%rbp),%xmm8+ movaps -0xa0(%rbp),%xmm9+ movaps -0x90(%rbp),%xmm10+ movaps -0x80(%rbp),%xmm11+ movaps -0x70(%rbp),%xmm12+ movaps -0x60(%rbp),%xmm13+ movaps -0x50(%rbp),%xmm14+ movaps -0x40(%rbp),%xmm15+___+$code.=<<___;+ mov -0x28(%rbp),%r15+ mov -0x20(%rbp),%r14+ mov -0x18(%rbp),%r13+ mov -0x10(%rbp),%r12+ mov -0x08(%rbp),%rbx+ mov %rbp,%rsp # restore %rsp+.cfi_def_cfa_register %rsp+.Lgcm_dec_abort:+ pop %rbp+.cfi_pop %rbp+.cfi_epilogue+ ret+.cfi_endproc+.size aesni_gcm_decrypt,.-aesni_gcm_decrypt+___++$code.=<<___;+.type _aesni_ctr32_6x,\@abi-omnipotent+.align 32+_aesni_ctr32_6x:+.cfi_startproc+ vmovdqu 0x00-0x80($key),$Z0 # borrow $Z0 for $rndkey+ vmovdqu 0x20($const),$T2 # borrow $T2, .Lone_msb+ lea -1($rounds),%r13+ vmovups 0x10-0x80($key),$rndkey+ lea 0x20-0x80($key),%r12+ vpxor $Z0,$T1,$inout0+ add \$`6<<24`,$counter+ jc .Lhandle_ctr32_2+ vpaddb $T2,$T1,$inout1+ vpaddb $T2,$inout1,$inout2+ vpxor $Z0,$inout1,$inout1+ vpaddb $T2,$inout2,$inout3+ vpxor $Z0,$inout2,$inout2+ vpaddb $T2,$inout3,$inout4+ vpxor $Z0,$inout3,$inout3+ vpaddb $T2,$inout4,$inout5+ vpxor $Z0,$inout4,$inout4+ vpaddb $T2,$inout5,$T1+ vpxor $Z0,$inout5,$inout5+ jmp .Loop_ctr32++.align 16+.Loop_ctr32:+ vaesenc $rndkey,$inout0,$inout0+ vaesenc $rndkey,$inout1,$inout1+ vaesenc $rndkey,$inout2,$inout2+ vaesenc $rndkey,$inout3,$inout3+ vaesenc $rndkey,$inout4,$inout4+ vaesenc $rndkey,$inout5,$inout5+ vmovups (%r12),$rndkey+ lea 0x10(%r12),%r12+ dec %r13d+ jnz .Loop_ctr32++ vmovdqu (%r12),$Hkey # last round key+ vaesenc $rndkey,$inout0,$inout0+ vpxor 0x00($inp),$Hkey,$Z0+ vaesenc $rndkey,$inout1,$inout1+ vpxor 0x10($inp),$Hkey,$Z1+ vaesenc $rndkey,$inout2,$inout2+ vpxor 0x20($inp),$Hkey,$Z2+ vaesenc $rndkey,$inout3,$inout3+ vpxor 0x30($inp),$Hkey,$Xi+ vaesenc $rndkey,$inout4,$inout4+ vpxor 0x40($inp),$Hkey,$T2+ vaesenc $rndkey,$inout5,$inout5+ vpxor 0x50($inp),$Hkey,$Hkey+ lea 0x60($inp),$inp++ vaesenclast $Z0,$inout0,$inout0+ vaesenclast $Z1,$inout1,$inout1+ vaesenclast $Z2,$inout2,$inout2+ vaesenclast $Xi,$inout3,$inout3+ vaesenclast $T2,$inout4,$inout4+ vaesenclast $Hkey,$inout5,$inout5+ vmovups $inout0,0x00($out)+ vmovups $inout1,0x10($out)+ vmovups $inout2,0x20($out)+ vmovups $inout3,0x30($out)+ vmovups $inout4,0x40($out)+ vmovups $inout5,0x50($out)+ lea 0x60($out),$out++ ret+.align 32+.Lhandle_ctr32_2:+ vpshufb $Ii,$T1,$Z2 # byte-swap counter+ vmovdqu 0x30($const),$Z1 # borrow $Z1, .Ltwo_lsb+ vpaddd 0x40($const),$Z2,$inout1 # .Lone_lsb+ vpaddd $Z1,$Z2,$inout2+ vpaddd $Z1,$inout1,$inout3+ vpshufb $Ii,$inout1,$inout1+ vpaddd $Z1,$inout2,$inout4+ vpshufb $Ii,$inout2,$inout2+ vpxor $Z0,$inout1,$inout1+ vpaddd $Z1,$inout3,$inout5+ vpshufb $Ii,$inout3,$inout3+ vpxor $Z0,$inout2,$inout2+ vpaddd $Z1,$inout4,$T1 # byte-swapped next counter value+ vpshufb $Ii,$inout4,$inout4+ vpxor $Z0,$inout3,$inout3+ vpshufb $Ii,$inout5,$inout5+ vpxor $Z0,$inout4,$inout4+ vpshufb $Ii,$T1,$T1 # next counter value+ vpxor $Z0,$inout5,$inout5+ jmp .Loop_ctr32+.cfi_endproc+.size _aesni_ctr32_6x,.-_aesni_ctr32_6x++.globl aesni_gcm_encrypt+.type aesni_gcm_encrypt,\@function,6,"unwind"+.align 32+aesni_gcm_encrypt:+.cfi_startproc+ push %rbp+.cfi_push %rbp+ mov %rsp,%rbp+.cfi_def_cfa_register %rbp+ xor $ret,$ret+ cmp \$0x60*3,$len # minimal accepted length+ jb .Lgcm_enc_abort++ push %rbx+.cfi_push %rbx+ push %r12+.cfi_push %r12+ push %r13+.cfi_push %r13+ push %r14+.cfi_push %r14+ push %r15+.cfi_push %r15+___+$code.=<<___ if ($win64);+ lea -0xa8(%rsp),%rsp+.cfi_alloca 0xa8+ movaps %xmm6,-0xd0(%rbp)+ movaps %xmm7,-0xc0(%rbp)+ movaps %xmm8,-0xb0(%rbp)+ movaps %xmm9,-0xa0(%rbp)+ movaps %xmm10,-0x90(%rbp)+ movaps %xmm11,-0x80(%rbp)+ movaps %xmm12,-0x70(%rbp)+ movaps %xmm13,-0x60(%rbp)+ movaps %xmm14,-0x50(%rbp)+ movaps %xmm15,-0x40(%rbp)+.cfi_offset %xmm6-%xmm15,-0xe0+___+$code.=<<___;+.cfi_end_prologue+ vzeroupper++ vmovdqu ($ivp),$T1 # input counter value+ add \$-128,%rsp+ mov 12($ivp),$counter+ lea .Lbswap_mask(%rip),$const+ lea -0x80($key),$in0 # borrow $in0+ mov \$0xf80,$end0 # borrow $end0+ lea 0x80($key),$key # size optimization+ vmovdqu ($const),$Ii # borrow $Ii for .Lbswap_mask+ and \$-128,%rsp # ensure stack alignment+ mov 0xf0-0x80($key),$rounds++ and $end0,$in0+ and %rsp,$end0+ sub $in0,$end0+ jc .Lenc_no_key_aliasing+ cmp \$768,$end0+ jnc .Lenc_no_key_aliasing+ sub $end0,%rsp # avoid aliasing with key+.Lenc_no_key_aliasing:++ lea ($out),$in0+ lea -0xc0($out,$len),$end0+ shr \$4,$len++ call _aesni_ctr32_6x+ vpshufb $Ii,$inout0,$Xi # save bswapped output on stack+ vpshufb $Ii,$inout1,$T2+ vmovdqu $Xi,0x70(%rsp)+ vpshufb $Ii,$inout2,$Z0+ vmovdqu $T2,0x60(%rsp)+ vpshufb $Ii,$inout3,$Z1+ vmovdqu $Z0,0x50(%rsp)+ vpshufb $Ii,$inout4,$Z2+ vmovdqu $Z1,0x40(%rsp)+ vpshufb $Ii,$inout5,$Z3 # passed to _aesni_ctr32_ghash_6x+ vmovdqu $Z2,0x30(%rsp)++ call _aesni_ctr32_6x++ vmovdqu ($Xip),$Xi # load Xi+ lea 0x20+0x20($Xip),$Xip # size optimization+ sub \$12,$len+ mov \$0x60*2,$ret+ vpshufb $Ii,$Xi,$Xi++ call _aesni_ctr32_ghash_6x+ vmovdqu 0x20(%rsp),$Z3 # I[5]+ vmovdqu ($const),$Ii # borrow $Ii for .Lbswap_mask+ vmovdqu 0x00-0x20($Xip),$Hkey # $Hkey^1+ vpunpckhqdq $Z3,$Z3,$T1+ vmovdqu 0x20-0x20($Xip),$rndkey # borrow $rndkey for $HK+ vmovups $inout0,-0x60($out) # save output+ vpshufb $Ii,$inout0,$inout0 # but keep bswapped copy+ vpxor $Z3,$T1,$T1+ vmovups $inout1,-0x50($out)+ vpshufb $Ii,$inout1,$inout1+ vmovups $inout2,-0x40($out)+ vpshufb $Ii,$inout2,$inout2+ vmovups $inout3,-0x30($out)+ vpshufb $Ii,$inout3,$inout3+ vmovups $inout4,-0x20($out)+ vpshufb $Ii,$inout4,$inout4+ vmovups $inout5,-0x10($out)+ vpshufb $Ii,$inout5,$inout5+ vmovdqu $inout0,0x10(%rsp) # free $inout0+___+{ my ($HK,$T3)=($rndkey,$inout0);++$code.=<<___;+ vmovdqu 0x30(%rsp),$Z2 # I[4]+ vmovdqu 0x10-0x20($Xip),$Ii # borrow $Ii for $Hkey^2+ vpunpckhqdq $Z2,$Z2,$T2+ vpclmulqdq \$0x00,$Hkey,$Z3,$Z1+ vpxor $Z2,$T2,$T2+ vpclmulqdq \$0x11,$Hkey,$Z3,$Z3+ vpclmulqdq \$0x00,$HK,$T1,$T1++ vmovdqu 0x40(%rsp),$T3 # I[3]+ vpclmulqdq \$0x00,$Ii,$Z2,$Z0+ vmovdqu 0x30-0x20($Xip),$Hkey # $Hkey^3+ vpxor $Z1,$Z0,$Z0+ vpunpckhqdq $T3,$T3,$Z1+ vpclmulqdq \$0x11,$Ii,$Z2,$Z2+ vpxor $T3,$Z1,$Z1+ vpxor $Z3,$Z2,$Z2+ vpclmulqdq \$0x10,$HK,$T2,$T2+ vmovdqu 0x50-0x20($Xip),$HK+ vpxor $T1,$T2,$T2++ vmovdqu 0x50(%rsp),$T1 # I[2]+ vpclmulqdq \$0x00,$Hkey,$T3,$Z3+ vmovdqu 0x40-0x20($Xip),$Ii # borrow $Ii for $Hkey^4+ vpxor $Z0,$Z3,$Z3+ vpunpckhqdq $T1,$T1,$Z0+ vpclmulqdq \$0x11,$Hkey,$T3,$T3+ vpxor $T1,$Z0,$Z0+ vpxor $Z2,$T3,$T3+ vpclmulqdq \$0x00,$HK,$Z1,$Z1+ vpxor $T2,$Z1,$Z1++ vmovdqu 0x60(%rsp),$T2 # I[1]+ vpclmulqdq \$0x00,$Ii,$T1,$Z2+ vmovdqu 0x60-0x20($Xip),$Hkey # $Hkey^5+ vpxor $Z3,$Z2,$Z2+ vpunpckhqdq $T2,$T2,$Z3+ vpclmulqdq \$0x11,$Ii,$T1,$T1+ vpxor $T2,$Z3,$Z3+ vpxor $T3,$T1,$T1+ vpclmulqdq \$0x10,$HK,$Z0,$Z0+ vmovdqu 0x80-0x20($Xip),$HK+ vpxor $Z1,$Z0,$Z0++ vpxor 0x70(%rsp),$Xi,$Xi # accumulate I[0]+ vpclmulqdq \$0x00,$Hkey,$T2,$Z1+ vmovdqu 0x70-0x20($Xip),$Ii # borrow $Ii for $Hkey^6+ vpunpckhqdq $Xi,$Xi,$T3+ vpxor $Z2,$Z1,$Z1+ vpclmulqdq \$0x11,$Hkey,$T2,$T2+ vpxor $Xi,$T3,$T3+ vpxor $T1,$T2,$T2+ vpclmulqdq \$0x00,$HK,$Z3,$Z3+ vpxor $Z0,$Z3,$Z0++ vpclmulqdq \$0x00,$Ii,$Xi,$Z2+ vmovdqu 0x00-0x20($Xip),$Hkey # $Hkey^1+ vpunpckhqdq $inout5,$inout5,$T1+ vpclmulqdq \$0x11,$Ii,$Xi,$Xi+ vpxor $inout5,$T1,$T1+ vpxor $Z1,$Z2,$Z1+ vpclmulqdq \$0x10,$HK,$T3,$T3+ vmovdqu 0x20-0x20($Xip),$HK+ vpxor $T2,$Xi,$Z3+ vpxor $Z0,$T3,$Z2++ vmovdqu 0x10-0x20($Xip),$Ii # borrow $Ii for $Hkey^2+ vpxor $Z1,$Z3,$T3 # aggregated Karatsuba post-processing+ vpclmulqdq \$0x00,$Hkey,$inout5,$Z0+ vpxor $T3,$Z2,$Z2+ vpunpckhqdq $inout4,$inout4,$T2+ vpclmulqdq \$0x11,$Hkey,$inout5,$inout5+ vpxor $inout4,$T2,$T2+ vpslldq \$8,$Z2,$T3+ vpclmulqdq \$0x00,$HK,$T1,$T1+ vpxor $T3,$Z1,$Xi+ vpsrldq \$8,$Z2,$Z2+ vpxor $Z2,$Z3,$Z3++ vpclmulqdq \$0x00,$Ii,$inout4,$Z1+ vmovdqu 0x30-0x20($Xip),$Hkey # $Hkey^3+ vpxor $Z0,$Z1,$Z1+ vpunpckhqdq $inout3,$inout3,$T3+ vpclmulqdq \$0x11,$Ii,$inout4,$inout4+ vpxor $inout3,$T3,$T3+ vpxor $inout5,$inout4,$inout4+ vpalignr \$8,$Xi,$Xi,$inout5 # 1st phase+ vpclmulqdq \$0x10,$HK,$T2,$T2+ vmovdqu 0x50-0x20($Xip),$HK+ vpxor $T1,$T2,$T2++ vpclmulqdq \$0x00,$Hkey,$inout3,$Z0+ vmovdqu 0x40-0x20($Xip),$Ii # borrow $Ii for $Hkey^4+ vpxor $Z1,$Z0,$Z0+ vpunpckhqdq $inout2,$inout2,$T1+ vpclmulqdq \$0x11,$Hkey,$inout3,$inout3+ vpxor $inout2,$T1,$T1+ vpxor $inout4,$inout3,$inout3+ vxorps 0x10(%rsp),$Z3,$Z3 # accumulate $inout0+ vpclmulqdq \$0x00,$HK,$T3,$T3+ vpxor $T2,$T3,$T3++ vpclmulqdq \$0x10,0x10($const),$Xi,$Xi+ vxorps $inout5,$Xi,$Xi++ vpclmulqdq \$0x00,$Ii,$inout2,$Z1+ vmovdqu 0x60-0x20($Xip),$Hkey # $Hkey^5+ vpxor $Z0,$Z1,$Z1+ vpunpckhqdq $inout1,$inout1,$T2+ vpclmulqdq \$0x11,$Ii,$inout2,$inout2+ vpxor $inout1,$T2,$T2+ vpalignr \$8,$Xi,$Xi,$inout5 # 2nd phase+ vpxor $inout3,$inout2,$inout2+ vpclmulqdq \$0x10,$HK,$T1,$T1+ vmovdqu 0x80-0x20($Xip),$HK+ vpxor $T3,$T1,$T1++ vxorps $Z3,$inout5,$inout5+ vpclmulqdq \$0x10,0x10($const),$Xi,$Xi+ vxorps $inout5,$Xi,$Xi++ vpclmulqdq \$0x00,$Hkey,$inout1,$Z0+ vmovdqu 0x70-0x20($Xip),$Ii # borrow $Ii for $Hkey^6+ vpxor $Z1,$Z0,$Z0+ vpunpckhqdq $Xi,$Xi,$T3+ vpclmulqdq \$0x11,$Hkey,$inout1,$inout1+ vpxor $Xi,$T3,$T3+ vpxor $inout2,$inout1,$inout1+ vpclmulqdq \$0x00,$HK,$T2,$T2+ vpxor $T1,$T2,$T2++ vpclmulqdq \$0x00,$Ii,$Xi,$Z1+ vpclmulqdq \$0x11,$Ii,$Xi,$Z3+ vpxor $Z0,$Z1,$Z1+ vpclmulqdq \$0x10,$HK,$T3,$Z2+ vpxor $inout1,$Z3,$Z3+ vpxor $T2,$Z2,$Z2++ vpxor $Z1,$Z3,$Z0 # aggregated Karatsuba post-processing+ vpxor $Z0,$Z2,$Z2+ vpslldq \$8,$Z2,$T1+ vmovdqu 0x10($const),$Hkey # .Lpoly+ vpsrldq \$8,$Z2,$Z2+ vpxor $T1,$Z1,$Xi+ vpxor $Z2,$Z3,$Z3++ vpalignr \$8,$Xi,$Xi,$T2 # 1st phase+ vpclmulqdq \$0x10,$Hkey,$Xi,$Xi+ vpxor $T2,$Xi,$Xi++ vpalignr \$8,$Xi,$Xi,$T2 # 2nd phase+ vpclmulqdq \$0x10,$Hkey,$Xi,$Xi+ vpxor $Z3,$T2,$T2+ vpxor $T2,$Xi,$Xi+___+}+$code.=<<___;+ vpshufb ($const),$Xi,$Xi # .Lbswap_mask+ vmovdqu $Xi,-0x40($Xip) # output Xi++ vzeroupper+___+$code.=<<___ if ($win64);+ movaps -0xd0(%rbp),%xmm6+ movaps -0xc0(%rbp),%xmm7+ movaps -0xb0(%rbp),%xmm8+ movaps -0xa0(%rbp),%xmm9+ movaps -0x90(%rbp),%xmm10+ movaps -0x80(%rbp),%xmm11+ movaps -0x70(%rbp),%xmm12+ movaps -0x60(%rbp),%xmm13+ movaps -0x50(%rbp),%xmm14+ movaps -0x40(%rbp),%xmm15+___+$code.=<<___;+ mov -0x28(%rbp),%r15+ mov -0x20(%rbp),%r14+ mov -0x18(%rbp),%r13+ mov -0x10(%rbp),%r12+ mov -0x08(%rbp),%rbx+ mov %rbp,%rsp # restore %rsp+.cfi_def_cfa_register %rsp+.Lgcm_enc_abort:+ pop %rbp+.cfi_pop %rbp+.cfi_epilogue+ ret+.cfi_endproc+.size aesni_gcm_encrypt,.-aesni_gcm_encrypt+___++$code.=<<___;+.align 64+.Lbswap_mask:+ .byte 15,14,13,12,11,10,9,8,7,6,5,4,3,2,1,0+.Lpoly:+ .byte 0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0xc2+.Lone_msb:+ .byte 0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,1+.Ltwo_lsb:+ .byte 2,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0+.Lone_lsb:+ .byte 1,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0+.asciz "AES-NI GCM module for x86_64, CRYPTOGAMS by <appro\@openssl.org>"+.align 64+___+}}} else {{{+$code=<<___; # assembler is too old+.text++.globl aesni_gcm_encrypt+.type aesni_gcm_encrypt,\@abi-omnipotent+aesni_gcm_encrypt:+.cfi_startproc+ xor %eax,%eax+ ret+.cfi_endproc+.size aesni_gcm_encrypt,.-aesni_gcm_encrypt++.globl aesni_gcm_decrypt+.type aesni_gcm_decrypt,\@abi-omnipotent+aesni_gcm_decrypt:+.cfi_startproc+ xor %eax,%eax+ ret+.cfi_endproc+.size aesni_gcm_decrypt,.-aesni_gcm_decrypt+___+}}}++$code =~ s/\`([^\`]*)\`/eval($1)/gem;++print $code;++close STDOUT or die "error closing STDOUT: $!";
@@ -0,0 +1,467 @@+#! /usr/bin/env perl+#+# ARM assembler distiller/adapter by \@dot-asm.++use strict;++################################################################+# Recognized "flavour"-s are:+#+# linux[32|64] GNU assembler, effectively pass-through+# ios[32|64] global symbols' decorations, PIC tweaks, etc.+# win[32|64] Visual Studio armasm-specific directives+# coff[32|64] e.g. clang --target=arm-windows ...+# cheri64 L64P128 platform+#+my $flavour = shift;+ $flavour = "linux" if (!$flavour or $flavour eq "void");++my $output = shift;+open STDOUT,">$output" || die "can't open $output: $!";++my %GLOBALS;+my $dotinlocallabels = ($flavour !~ /ios/) ? 1 : 0;+my $in_proc; # used with 'windows' flavour++################################################################+# directives which need special treatment on different platforms+################################################################+my $arch = sub { } if ($flavour !~ /linux|coff64/);# omit .arch+my $fpu = sub { } if ($flavour !~ /linux/); # omit .fpu++my $rodata = sub {+ SWITCH: for ($flavour) {+ /linux|cheri/ && return ".section\t.rodata";+ /ios/ && return ".section\t__TEXT,__const";+ /coff/ && return ".section\t.rdata,\"dr\"";+ /win/ && return "\tAREA\t|.rdata|,DATA,READONLY,ALIGN=8";+ last;+ }+};++my $hidden = sub {+ if ($flavour =~ /ios/) { ".private_extern\t".join(',',@_); }+} if ($flavour !~ /linux|cheri/);++my $comm = sub {+ my @args = split(/,\s*/,shift);+ my $name = @args[0];+ my $global = \$GLOBALS{$name};+ my $ret;++ if ($flavour =~ /ios32/) {+ $ret = ".comm\t_$name,@args[1]\n";+ $ret .= ".non_lazy_symbol_pointer\n";+ $ret .= "$name:\n";+ $ret .= ".indirect_symbol\t_$name\n";+ $ret .= ".long\t0\n";+ $ret .= ".previous";+ $name = "_$name";+ } elsif ($flavour =~ /ios64/) {+ $name = "_$name";+ $ret = ".comm\t$name,@args[1]";+ } elsif ($flavour =~ /win/) {+ $ret = "\tCOMMON\t|$name|,@args[1]";+ } elsif ($flavour =~ /coff/) {+ $ret = ".comm\t$name,@args[1]";+ } else {+ $ret = ".comm\t".join(',',@args);+ }++ $$global = $name;+ $ret;+};++my $globl = sub {+ my $name = shift;+ my $global = \$GLOBALS{$name};+ my $ret;++ SWITCH: for ($flavour) {+ /ios/ && do { $name = "_$name"; last; };+ /win/ && do { $ret = ""; last; };+ }++ $ret = ".globl $name" if (!defined($ret));+ $$global = $name;+ $ret;+};+my $global = $globl;++my $extern = sub {+ &$globl(@_);+ if ($flavour =~ /win/) {+ return "\tEXTERN\t@_";+ }+ return; # return nothing+};++my $type = sub {+ my $arg = join(',',@_);+ my $ret;++ SWITCH: for ($flavour) {+ /ios32/ && do { if ($arg =~ /(\w+),\s*%function/) {+ $ret = "#ifdef __thumb2__\n" .+ ".thumb_func $1\n" .+ "#endif";+ }+ last;+ };+ /win/ && do { if ($arg =~ /(\w+),\s*%(function|object)/) {+ my $type = "[DATA]";+ if ($2 eq "function") {+ $in_proc = $1;+ $type = "[FUNC]";+ }+ $ret = $GLOBALS{$1} ? "\tEXPORT\t|$1|$type"+ : "";+ }+ last;+ };+ /coff/ && do { if ($arg =~ /(\w+),\s*%function/) {+ $ret = ".def $1;\n".+ ".type 32;\n".+ ".endef";+ }+ last;+ };+ }+ return $ret;+} if ($flavour !~ /linux|cheri/);++my $size = sub {+ if ($in_proc && $flavour =~ /win/) {+ $in_proc = undef;+ return "\tENDP";+ }+} if ($flavour !~ /linux|cheri/);++my $inst = sub {+ if ($flavour =~ /win/) { "\tDCDU\t".join(',',@_); }+ else { ".long\t".join(',',@_); }+} if ($flavour !~ /linux|cheri/);++my $asciz = sub {+ my $line = join(",",@_);+ if ($line =~ /^"(.*)"$/)+ { if ($flavour =~ /win/) {+ "\tDCB\t$line,0\n\tALIGN\t4";+ } else {+ ".byte " . join(",",unpack("C*",$1),0) . "\n.align 2";+ }+ } else { ""; }+};++my $align = sub {+ "\tALIGN\t".2**@_[0];+} if ($flavour =~ /win/);+ $align = sub {+ ".p2align\t".@_[0];+} if ($flavour =~ /coff/);++my $byte = sub {+ "\tDCB\t".join(',',@_);+} if ($flavour =~ /win/);++my $short = sub {+ "\tDCWU\t".join(',',@_);+} if ($flavour =~ /win/);++my $word = sub {+ "\tDCDU\t".join(',',@_);+} if ($flavour =~ /win/);++my $long = $word if ($flavour =~ /win/);++my $quad = sub {+ "\tDCQU\t".join(',',@_);+} if ($flavour =~ /win/);++my $skip = sub {+ "\tSPACE\t".shift;+} if ($flavour =~ /win/);++my $code = sub {+ "\tCODE@_[0]";+} if ($flavour =~ /win/);++my $thumb = sub { # .thumb should appear prior .text in source+ "# define ARM THUMB\n" .+ "\tTHUMB";+} if ($flavour =~ /win/);++my $text = sub {+ "\tAREA\t|.text|,CODE,ALIGN=8,".($flavour =~ /64/ ? "ARM64" : "ARM");+} if ($flavour =~ /win/);++my $syntax = sub {} if ($flavour =~ /win/); # omit .syntax++my $rva = sub {+ # .rva directive comes in handy only on 32-bit Windows, i.e. it can+ # be used only in '#if defined(_WIN32) && !defined(_WIN64)' sections.+ # However! Corresponding compilers don't seem to bet on PIC, which+ # raises the question why would assembler programmer have to jump+ # through the hoops? But just in case, it would go as following:+ #+ # ldr r1,.LOPENSSL_armcap+ # ldr r2,.LOPENSSL_armcap+4+ # adr r0,.LOPENSSL_armcap+ # bic r1,r1,#1 ; de-thumb-ify link.exe's ideas+ # sub r0,r0,r1 ; r0 is image base now+ # ldr r0,[r0,r2]+ # ...+ #.LOPENSSL_armcap:+ # .rva .LOPENSSL_armcap ; self-reference+ # .rva OPENSSL_armcap_P ; real target+ #+ # Non-position-independent [and ISA-neutral] alternative is so much+ # simpler:+ #+ # ldr r0,.LOPENSSL_armcap+ # ldr r0,[r0]+ # ...+ #.LOPENSSL_armcap:+ # .long OPENSSL_armcap_P+ #+ "\tDCDU\t@_[0]\n\tRELOC\t2"+} if ($flavour =~ /win(?!64)/);++################################################################+# some broken instructions in Visual Studio armasm[64]...++my $it = sub {} if ($flavour =~ /win32/); # omit 'it'++my $ext = sub {+ "\text8\t".join(',',@_);+} if ($flavour =~ /win64/);++my $csel = sub {+ my ($args,$comment) = split(m|\s*//|,shift);+ my @regs = split(m|,\s*|,$args);+ my $cond = pop(@regs);++ "\tcsel$cond\t".join(',',@regs);+} if ($flavour =~ /win64/);++my $csetm = sub {+ my ($args,$comment) = split(m|\s*//|,shift);+ my @regs = split(m|,\s*|,$args);+ my $cond = pop(@regs);++ "\tcsetm$cond\t".join(',',@regs);+} if ($flavour =~ /win64/);++# ... then conditional branch instructions are also broken, but+# maintaining all the variants is tedious, so I kludge-fix it+# elsewhere...++################################################################+# CHERI-specific synthetic instructions+my $scvalue = sub {+ my ($args,$comment) = split(m|\s*//|,shift);+ $args =~ s/\b(?:x([0-9]+)|(sp))\b/c$1$2/g;+ my @regs = split(m|,\s*|,$args);+ @regs[2] =~ s/\bc([0-9])\b/x$1/;++ "\tscvalue\t".join(',',@regs);+};++my $cadd = sub {+ my ($args,$comment) = split(m|\s*//|,shift);+ if ($flavour =~ /cheri/) {+ $args =~ s/\b(?:x([0-9]+)|(sp))\b/c$1$2/g;+ } else {+ $args =~ s/\bc([0-9]+)\b/x$1/g;+ }+ my @regs = split(m|,\s*|,$args);+ @regs[2] =~ s/c([0-9])/x$1/;++ "\tadd\t".join(',',@regs);+};++my $csub = sub {+ my ($args,$comment) = split(m|\s*//|,shift);+ if ($flavour =~ /cheri/) {+ $args =~ s/\b(?:x([0-9]+)|(sp))\b/c$1$2/g;+ } else {+ $args =~ s/\bc([0-9]+)\b/x$1/g;+ }+ my @regs = split(m|,\s*|,$args);+ @regs[2] =~ s/c([0-9])/x$1/;++ "\tsub\t".join(',',@regs);+};++my $cmov = sub {+ my $args = shift;+ if ($flavour =~ /cheri/) {+ $args =~ s/\b(?:x([0-9]+)|(sp))\b/c$1$2/g;+ } else {+ $args =~ s/\bc([0-9]+)\b/x$1/g;+ }++ "\tmov\t".$args;+};++my $adr = sub {+ my $args = shift;+ $args =~ s/\bx([0-9]+)\b/c$1/g;++ "\tadr\t".$args;+} if ($flavour =~ /cheri/);++################################################################+my $adrp = sub {+ my ($args,$comment) = split(m|\s*//|,shift);+ "\tadrp\t$args\@PAGE";+} if ($flavour =~ /ios64/);++my $paciasp = sub {+ ($flavour =~ /linux|cheri/) ? "\t.inst\t0xd503233f"+ : &$inst(0xd503233f);+};++my $autiasp = sub {+ ($flavour =~ /linux|cheri/) ? "\t.inst\t0xd50323bf"+ : &$inst(0xd50323bf);+};++sub range {+ my ($r,$sfx,$start,$end) = @_;++ join(",",map("$r$_$sfx",($start..$end)));+}++sub expand_line {+ my $line = shift;+ my @ret = ();++ pos($line)=0;++ while ($line =~ m/\G[^@\/\{\"]*/g) {+ if ($line =~ m/\G(@|\/\/|$)/gc) {+ last;+ }+ elsif ($line =~ m/\G\{/gc) {+ my $saved_pos = pos($line);+ $line =~ s/\G([rdqv])([0-9]+)([^\-]*)\-\1([0-9]+)\3/range($1,$3,$2,$4)/e;+ pos($line) = $saved_pos;+ $line =~ m/\G[^\}]*\}/g;+ }+ elsif ($line =~ m/\G\"/gc) {+ $line =~ m/\G[^\"]*\"/g;+ }+ }++ $line =~ s/\b(\w+)/$GLOBALS{$1} or $1/ge;++ if ($flavour =~ /cheri/) {+ $line =~ s/\[\s*(?:x([0-9]+)|(sp))\s*(,?.*)\]/[c$1$2$3]/;+ } else {+ $line =~ s/\bc((?:[0-9]+|zr))\b/x$1/g;+ $line =~ s/\bcsp\b/sp/g;+ }++ if ($flavour =~ /win/) {+ # adjust alignment hints, "[rN,:32]" -> "[rN@32]"+ $line =~ s/(\[\s*(?:r[0-9]+|sp))\s*,?\s*:([0-9]+\s*\])/$1\@$2/;+ # adjust local labels, ".Lwhatever" -> "|$Lwhatever|"+ $line =~ s/\.(L\w{2,})/|\$$1|/g;+ # omit "#:lo12:" on win64+ $line =~ s/#:lo12://;+ } elsif ($flavour =~ /coff(?!64)/) {+ $line =~ s/\.L(\w{2,})/(\$ML$1)/g;+ } elsif ($flavour =~ /ios64/) {+ $line =~ s/#:lo12:(\w+)/$1\@PAGEOFF/;+ }++ if ($flavour =~ /64/) {+ # "vX.Md[N]" -> "vX.d[N]+ $line =~ s/\b(v[0-9]+)\.[1-9]+([bhsd]\[[0-9]+\])/$1.$2/;+ }++ return $line;+}++if ($flavour =~ /win(32|64)/) {+ print<<___;+ GBLA __SIZEOF_POINTER__+__SIZEOF_POINTER__ SETA $1/8+___+}++while(my $line=<>) {++ if ($flavour =~ /win/) {+ if ($line =~ m/^#\s*(ifdef|ifndef|else|endif)\b(.*)/) {+ my ($op, $arg) = ($1, $2);+ $op = "if :def:" if ($op eq "ifdef");+ $op = "if :lnot::def:" if ($op eq "ifndef");+ print " ".$op.$arg."\n";+ next;+ }+ $line =~ s|//.*||;+ }++ # fix up assembler-specific commentary delimiter+ $line =~ s/@(?=[\s@])/\;/g if ($flavour =~ /win|coff/);++ if ($line =~ m/^\s*(#|@|;|\/\/)/) { print $line; next; }++ $line =~ s|/\*.*\*/||; # get rid of C-style comments...+ $line =~ s|^\s+||; # ... and skip white spaces in beginning...+ $line =~ s|\s+$||; # ... and at the end++ {+ $line =~ s|[\b\.]L(\w{2,})|L$1|g; # common denominator for Locallabel+ $line =~ s|\bL(\w{2,})|\.L$1|g if ($dotinlocallabels);+ }++ {+ $line =~ s|(^[\.\w]+)\:\s*||;+ my $label = $1;+ if ($label) {+ $label = ($GLOBALS{$label} or $label);+ if ($flavour =~ /win/) {+ $label =~ s|^\.L(?=\w)|\$L|;+ printf "|%s|%s", $label, ($label eq $in_proc ? " PROC" : "");+ } else {+ $label =~ s|^\.L(?=\w)|\$ML| if ($flavour =~ /coff(?!64)/);+ printf "%s:", $label;+ }+ }+ }++ if ($line !~ m/^[#@;]/) {+ $line =~ s|^\s*(\.?)(\S+)\s*||;+ my $c = $1; $c = "\t" if ($c eq "");+ my $mnemonic = $2;+ my $opcode;+ if ($mnemonic =~ m/([^\.]+)\.([^\.]+)/) {+ $opcode = eval("\$$1_$2");+ } else {+ $opcode = eval("\$$mnemonic");+ }++ my $arg=expand_line($line);++ if (ref($opcode) eq 'CODE') {+ $line = &$opcode($arg);+ } elsif ($mnemonic) {+ if ($flavour =~ /win64/) {+ # "b.cond" -> "bcond", kludge-fix:-(+ $mnemonic =~ s/^b\.([a-z]{2}$)/b$1/;+ }+ $line = $c.$mnemonic;+ $line.= "\t$arg" if ($arg ne "");+ }+ }++ print $line if ($line);+ print "\n";+}++print "\tEND\n" if ($flavour =~ /win/);++close STDOUT;
@@ -0,0 +1,101 @@+#ifndef __ARM_ARCH_H__+#define __ARM_ARCH_H__++#if !defined(__ARM_ARCH__)+# if defined(__CC_ARM)+# if __TARGET_ARCH_THUMB+# define __thumb__+# if __TARGET_ARCH_THUMB >= 4+# define __thumb2__+# endif+# endif+# if __TARGET_ARCH_ARM+# define __ARM_ARCH__ __TARGET_ARCH_ARM+# else+# define __ARM_ARCH__ (__TARGET_ARCH_THUMB + 3)+# endif+# if defined(__BIG_ENDIAN)+# define __ARMEB__+# else+# define __ARMEL__+# endif+# elif defined(__GNUC__) || defined(__clang__)+# if defined(__aarch64__)+# define __ARM_ARCH__ 8+# ifdef __AARCH64EB__+# define __ARMEB__+# else+# define __ARMEL__+# endif+# elif defined(__ARM_ARCH)+# define __ARM_ARCH__ __ARM_ARCH+ /*+ * Why didn't gcc define __ARM_ARCH from start? Instead it defined+ * bunch of below macros. See all_architectures[] table in+ * gcc/config/arm/arm.c. On a side note it defines+ * __ARMEL__/__ARMEB__ for little-/big-endian.+ */+# elif defined(__ARM_ARCH_8A__)+# define __ARM_ARCH__ 8+# elif defined(__ARM_ARCH_7__) || defined(__ARM_ARCH_7A__) || \+ defined(__ARM_ARCH_7R__)|| defined(__ARM_ARCH_7M__) || \+ defined(__ARM_ARCH_7EM__)+# define __ARM_ARCH__ 7+# elif defined(__ARM_ARCH_6__) || defined(__ARM_ARCH_6J__) || \+ defined(__ARM_ARCH_6K__)|| defined(__ARM_ARCH_6M__) || \+ defined(__ARM_ARCH_6Z__)|| defined(__ARM_ARCH_6ZK__) || \+ defined(__ARM_ARCH_6T2__)+# define __ARM_ARCH__ 6+# elif defined(__ARM_ARCH_5__) || defined(__ARM_ARCH_5T__) || \+ defined(__ARM_ARCH_5E__)|| defined(__ARM_ARCH_5TE__) || \+ defined(__ARM_ARCH_5TEJ__)+# define __ARM_ARCH__ 5+# elif defined(__ARM_ARCH_4__) || defined(__ARM_ARCH_4T__)+# define __ARM_ARCH__ 4+# else+# error "unsupported ARM architecture"+# endif+# elif defined(_MSC_VER)+# define __ARMEL__+# if defined(_M_ARM)+# define __ARM_ARCH__ _M_ARM+# if defined(_M_THUMB)+# define __thumb__+# if _M_THUMB >= 7+# define __thumb2__+# endif+# endif+# elif defined(_M_ARM64)+# define __AARCH64EL__+# define __ARM_ARCH__ 8+# else+# error "unsupported ARM architecture"+# endif+# endif+#endif++#if !defined(__ARM_MAX_ARCH__)+# define __ARM_MAX_ARCH__ __ARM_ARCH__+#endif++#if __ARM_MAX_ARCH__<__ARM_ARCH__+# error "__ARM_MAX_ARCH__ can't be less than __ARM_ARCH__"+#elif __ARM_MAX_ARCH__!=__ARM_ARCH__+# if __ARM_ARCH__<7 && __ARM_MAX_ARCH__>=7 && defined(__ARMEB__)+# error "can't build universal big-endian binary"+# endif+#endif++#ifndef __ASSEMBLER__+extern unsigned int OPENSSL_armcap_P;+#endif++#define ARMV7_NEON (1<<0)+#define ARMV7_TICK (1<<1)+#define ARMV8_AES (1<<2)+#define ARMV8_SHA1 (1<<3)+#define ARMV8_SHA256 (1<<4)+#define ARMV8_PMULL (1<<5)+#define ARMV8_SHA512 (1<<6)++#endif
@@ -0,0 +1,2053 @@+#ifndef __KERNEL__+# include "arm_arch.h"++#endif++.text++.align 5+Lsigma:+.quad 0x3320646e61707865,0x6b20657479622d32 // endian-neutral+Lone:+.long 1,2,3,4+Lrot24:+.long 0x02010003,0x06050407,0x0a09080b,0x0e0d0c0f+.byte 67,104,97,67,104,97,50,48,32,102,111,114,32,65,82,77,118,56,44,32,67,82,89,80,84,79,71,65,77,83,32,98,121,32,64,100,111,116,45,97,115,109,0+.align 2++.globl _crypton_chacha20_asm_ctr32++.align 5+_crypton_chacha20_asm_ctr32:+ cbz x2,Labort+ cmp x2,#192+ b.lo Lshort++#ifndef __KERNEL__+ adrp x17,_crypton_armcap_P@PAGE+ ldr w17,[x17,_crypton_armcap_P@PAGEOFF]+ tst w17,#ARMV7_NEON+ b.ne Lcrypton_chacha20_asm_neon+#endif++Lshort:+.long 0xd503233f // paciasp+ stp x29,x30,[sp,#-12*__SIZEOF_POINTER__]!+ add x29,sp,#0++ adr x5,Lsigma+ stp x19,x20,[sp,#2*__SIZEOF_POINTER__]+ stp x21,x22,[sp,#4*__SIZEOF_POINTER__]+ stp x23,x24,[sp,#6*__SIZEOF_POINTER__]+ stp x25,x26,[sp,#8*__SIZEOF_POINTER__]+ stp x27,x28,[sp,#10*__SIZEOF_POINTER__]+ sub sp,sp,#64++ ldp x22,x23,[x5] // load sigma+ ldp x24,x25,[x3] // load key+ ldp x26,x27,[x3,#16]+ ldp x28,x30,[x4] // load counter+#ifdef __AARCH64EB__+ ror x24,x24,#32+ ror x25,x25,#32+ ror x26,x26,#32+ ror x27,x27,#32+ ror x28,x28,#32+ ror x30,x30,#32+#endif++Loop_outer:+ mov w5,w22 // unpack key block+ lsr x6,x22,#32+ mov w7,w23+ lsr x8,x23,#32+ mov w9,w24+ lsr x10,x24,#32+ mov w11,w25+ lsr x12,x25,#32+ mov w13,w26+ lsr x14,x26,#32+ mov w15,w27+ lsr x16,x27,#32+ mov w17,w28+ lsr x19,x28,#32+ mov w20,w30+ lsr x21,x30,#32++ mov x4,#10+ subs x2,x2,#64+Loop:+ sub x4,x4,#1+ add w5,w5,w9+ add w6,w6,w10+ add w7,w7,w11+ add w8,w8,w12+ eor w17,w17,w5+ eor w19,w19,w6+ eor w20,w20,w7+ eor w21,w21,w8+ ror w17,w17,#16+ ror w19,w19,#16+ ror w20,w20,#16+ ror w21,w21,#16+ add w13,w13,w17+ add w14,w14,w19+ add w15,w15,w20+ add w16,w16,w21+ eor w9,w9,w13+ eor w10,w10,w14+ eor w11,w11,w15+ eor w12,w12,w16+ ror w9,w9,#20+ ror w10,w10,#20+ ror w11,w11,#20+ ror w12,w12,#20+ add w5,w5,w9+ add w6,w6,w10+ add w7,w7,w11+ add w8,w8,w12+ eor w17,w17,w5+ eor w19,w19,w6+ eor w20,w20,w7+ eor w21,w21,w8+ ror w17,w17,#24+ ror w19,w19,#24+ ror w20,w20,#24+ ror w21,w21,#24+ add w13,w13,w17+ add w14,w14,w19+ add w15,w15,w20+ add w16,w16,w21+ eor w9,w9,w13+ eor w10,w10,w14+ eor w11,w11,w15+ eor w12,w12,w16+ ror w9,w9,#25+ ror w10,w10,#25+ ror w11,w11,#25+ ror w12,w12,#25+ add w5,w5,w10+ add w6,w6,w11+ add w7,w7,w12+ add w8,w8,w9+ eor w21,w21,w5+ eor w17,w17,w6+ eor w19,w19,w7+ eor w20,w20,w8+ ror w21,w21,#16+ ror w17,w17,#16+ ror w19,w19,#16+ ror w20,w20,#16+ add w15,w15,w21+ add w16,w16,w17+ add w13,w13,w19+ add w14,w14,w20+ eor w10,w10,w15+ eor w11,w11,w16+ eor w12,w12,w13+ eor w9,w9,w14+ ror w10,w10,#20+ ror w11,w11,#20+ ror w12,w12,#20+ ror w9,w9,#20+ add w5,w5,w10+ add w6,w6,w11+ add w7,w7,w12+ add w8,w8,w9+ eor w21,w21,w5+ eor w17,w17,w6+ eor w19,w19,w7+ eor w20,w20,w8+ ror w21,w21,#24+ ror w17,w17,#24+ ror w19,w19,#24+ ror w20,w20,#24+ add w15,w15,w21+ add w16,w16,w17+ add w13,w13,w19+ add w14,w14,w20+ eor w10,w10,w15+ eor w11,w11,w16+ eor w12,w12,w13+ eor w9,w9,w14+ ror w10,w10,#25+ ror w11,w11,#25+ ror w12,w12,#25+ ror w9,w9,#25+ cbnz x4,Loop++ add w5,w5,w22 // accumulate key block+ add x6,x6,x22,lsr#32+ add w7,w7,w23+ add x8,x8,x23,lsr#32+ add w9,w9,w24+ add x10,x10,x24,lsr#32+ add w11,w11,w25+ add x12,x12,x25,lsr#32+ add w13,w13,w26+ add x14,x14,x26,lsr#32+ add w15,w15,w27+ add x16,x16,x27,lsr#32+ add w17,w17,w28+ add x19,x19,x28,lsr#32+ add w20,w20,w30+ add x21,x21,x30,lsr#32++ b.lo Ltail++ add x5,x5,x6,lsl#32 // pack+ add x7,x7,x8,lsl#32+ ldp x6,x8,[x1,#0] // load input+ add x9,x9,x10,lsl#32+ add x11,x11,x12,lsl#32+ ldp x10,x12,[x1,#16]+ add x13,x13,x14,lsl#32+ add x15,x15,x16,lsl#32+ ldp x14,x16,[x1,#32]+ add x17,x17,x19,lsl#32+ add x20,x20,x21,lsl#32+ ldp x19,x21,[x1,#48]+ add x1,x1,#64+#ifdef __AARCH64EB__+ rev x5,x5+ rev x7,x7+ rev x9,x9+ rev x11,x11+ rev x13,x13+ rev x15,x15+ rev x17,x17+ rev x20,x20+#endif+ eor x5,x5,x6+ eor x7,x7,x8+ eor x9,x9,x10+ eor x11,x11,x12+ eor x13,x13,x14+ eor x15,x15,x16+ eor x17,x17,x19+ eor x20,x20,x21++ stp x5,x7,[x0,#0] // store output+ add x28,x28,#1 // increment counter+ stp x9,x11,[x0,#16]+ stp x13,x15,[x0,#32]+ stp x17,x20,[x0,#48]+ add x0,x0,#64++ b.hi Loop_outer++ ldp x19,x20,[x29,#2*__SIZEOF_POINTER__]+ add sp,sp,#64+ ldp x21,x22,[x29,#4*__SIZEOF_POINTER__]+ ldp x23,x24,[x29,#6*__SIZEOF_POINTER__]+ ldp x25,x26,[x29,#8*__SIZEOF_POINTER__]+ ldp x27,x28,[x29,#10*__SIZEOF_POINTER__]+ ldp x29,x30,[sp],#12*__SIZEOF_POINTER__+.long 0xd50323bf // autiasp+Labort:+ ret++.align 4+Ltail:+ add x2,x2,#64+Less_than_64:+ sub x0,x0,#1+ add x1,x1,x2+ add x0,x0,x2+ add x4,sp,x2+ neg x2,x2++ add x5,x5,x6,lsl#32 // pack+ add x7,x7,x8,lsl#32+ add x9,x9,x10,lsl#32+ add x11,x11,x12,lsl#32+ add x13,x13,x14,lsl#32+ add x15,x15,x16,lsl#32+ add x17,x17,x19,lsl#32+ add x20,x20,x21,lsl#32+#ifdef __AARCH64EB__+ rev x5,x5+ rev x7,x7+ rev x9,x9+ rev x11,x11+ rev x13,x13+ rev x15,x15+ rev x17,x17+ rev x20,x20+#endif+ stp x5,x7,[sp,#0] // off-load complete block+ stp x9,x11,[sp,#16]+ stp x13,x15,[sp,#32]+ stp x17,x20,[sp,#48]++Loop_tail:+ ldrb w10,[x1,x2]+ ldrb w11,[x4,x2]+ add x2,x2,#1+ eor w10,w10,w11+ strb w10,[x0,x2]+ cbnz x2,Loop_tail++ stp xzr,xzr,[sp,#0] // wipe off-load area+ stp xzr,xzr,[sp,#16]+ stp xzr,xzr,[sp,#32]+ stp xzr,xzr,[sp,#48]++ ldp x19,x20,[x29,#2*__SIZEOF_POINTER__]+ add sp,sp,#64+ ldp x21,x22,[x29,#4*__SIZEOF_POINTER__]+ ldp x23,x24,[x29,#6*__SIZEOF_POINTER__]+ ldp x25,x26,[x29,#8*__SIZEOF_POINTER__]+ ldp x27,x28,[x29,#10*__SIZEOF_POINTER__]+ ldp x29,x30,[sp],#12*__SIZEOF_POINTER__+.long 0xd50323bf // autiasp+ ret+++#ifdef __KERNEL__+.globl _crypton_chacha20_asm_neon+#endif++.align 5+_crypton_chacha20_asm_neon:+Lcrypton_chacha20_asm_neon:+.long 0xd503233f // paciasp+ stp x29,x30,[sp,#-12*__SIZEOF_POINTER__]!+ add x29,sp,#0++ adr x5,Lsigma+ stp x19,x20,[sp,#2*__SIZEOF_POINTER__]+ stp x21,x22,[sp,#4*__SIZEOF_POINTER__]+ stp x23,x24,[sp,#6*__SIZEOF_POINTER__]+ stp x25,x26,[sp,#8*__SIZEOF_POINTER__]+ stp x27,x28,[sp,#10*__SIZEOF_POINTER__]+ cmp x2,#512+ b.hs L512_or_more_neon++ sub sp,sp,#64++ ldp x22,x23,[x5] // load sigma+ ld1 {v0.4s},[x5],#16+ ldp x24,x25,[x3] // load key+ ldp x26,x27,[x3,#16]+ ld1 {v1.4s,v2.4s},[x3]+ ldp x28,x30,[x4] // load counter+ ld1 {v3.4s},[x4]+ stp d8,d9,[sp] // meet ABI requirements+ ld1 {v8.4s,v9.4s},[x5]+#ifdef __AARCH64EB__+ rev64 v0.4s,v0.4s+ ror x24,x24,#32+ ror x25,x25,#32+ ror x26,x26,#32+ ror x27,x27,#32+ ror x28,x28,#32+ ror x30,x30,#32+#endif++Loop_outer_neon:+ dup v16.4s,v0.s[0] // unpack key block+ mov w5,w22+ dup v20.4s,v0.s[1]+ lsr x6,x22,#32+ dup v24.4s,v0.s[2]+ mov w7,w23+ dup v28.4s,v0.s[3]+ lsr x8,x23,#32+ dup v17.4s,v1.s[0]+ mov w9,w24+ dup v21.4s,v1.s[1]+ lsr x10,x24,#32+ dup v25.4s,v1.s[2]+ mov w11,w25+ dup v29.4s,v1.s[3]+ lsr x12,x25,#32+ dup v19.4s,v3.s[0]+ mov w13,w26+ dup v23.4s,v3.s[1]+ lsr x14,x26,#32+ dup v27.4s,v3.s[2]+ mov w15,w27+ dup v31.4s,v3.s[3]+ lsr x16,x27,#32+ add v19.4s,v19.4s,v8.4s+ mov w17,w28+ dup v18.4s,v2.s[0]+ lsr x19,x28,#32+ dup v22.4s,v2.s[1]+ mov w20,w30+ dup v26.4s,v2.s[2]+ lsr x21,x30,#32+ dup v30.4s,v2.s[3]++ mov x4,#10+ subs x2,x2,#320+Loop_neon:+ sub x4,x4,#1+ add v16.4s,v16.4s,v17.4s+ add v20.4s,v20.4s,v21.4s+ add v24.4s,v24.4s,v25.4s+ add v28.4s,v28.4s,v29.4s+ eor v19.16b,v19.16b,v16.16b+ eor v23.16b,v23.16b,v20.16b+ eor v27.16b,v27.16b,v24.16b+ eor v31.16b,v31.16b,v28.16b+ add w5,w5,w9+ rev32 v19.8h,v19.8h+ add w6,w6,w10+ rev32 v23.8h,v23.8h+ add w7,w7,w11+ rev32 v27.8h,v27.8h+ add w8,w8,w12+ rev32 v31.8h,v31.8h+ eor w17,w17,w5+ add v18.4s,v18.4s,v19.4s+ eor w19,w19,w6+ add v22.4s,v22.4s,v23.4s+ eor w20,w20,w7+ add v26.4s,v26.4s,v27.4s+ eor w21,w21,w8+ add v30.4s,v30.4s,v31.4s+ ror w17,w17,#16+ eor v4.16b,v17.16b,v18.16b+ ror w19,w19,#16+ eor v5.16b,v21.16b,v22.16b+ ror w20,w20,#16+ eor v6.16b,v25.16b,v26.16b+ ror w21,w21,#16+ eor v7.16b,v29.16b,v30.16b+ add w13,w13,w17+ ushr v17.4s,v4.4s,#20+ add w14,w14,w19+ ushr v21.4s,v5.4s,#20+ add w15,w15,w20+ ushr v25.4s,v6.4s,#20+ add w16,w16,w21+ ushr v29.4s,v7.4s,#20+ eor w9,w9,w13+ sli v17.4s,v4.4s,#12+ eor w10,w10,w14+ sli v21.4s,v5.4s,#12+ eor w11,w11,w15+ sli v25.4s,v6.4s,#12+ eor w12,w12,w16+ sli v29.4s,v7.4s,#12+ ror w9,w9,#20+ add v16.4s,v16.4s,v17.4s+ ror w10,w10,#20+ add v20.4s,v20.4s,v21.4s+ ror w11,w11,#20+ add v24.4s,v24.4s,v25.4s+ ror w12,w12,#20+ add v28.4s,v28.4s,v29.4s+ add w5,w5,w9+ eor v4.16b,v19.16b,v16.16b+ add w6,w6,w10+ eor v5.16b,v23.16b,v20.16b+ add w7,w7,w11+ eor v6.16b,v27.16b,v24.16b+ add w8,w8,w12+ eor v7.16b,v31.16b,v28.16b+ eor w17,w17,w5+ tbl v19.16b,{v4.16b},v9.16b+ eor w19,w19,w6+ tbl v23.16b,{v5.16b},v9.16b+ eor w20,w20,w7+ tbl v27.16b,{v6.16b},v9.16b+ eor w21,w21,w8+ tbl v31.16b,{v7.16b},v9.16b+ ror w17,w17,#24+ add v18.4s,v18.4s,v19.4s+ ror w19,w19,#24+ add v22.4s,v22.4s,v23.4s+ ror w20,w20,#24+ add v26.4s,v26.4s,v27.4s+ ror w21,w21,#24+ add v30.4s,v30.4s,v31.4s+ add w13,w13,w17+ eor v4.16b,v17.16b,v18.16b+ add w14,w14,w19+ eor v5.16b,v21.16b,v22.16b+ add w15,w15,w20+ eor v6.16b,v25.16b,v26.16b+ add w16,w16,w21+ eor v7.16b,v29.16b,v30.16b+ eor w9,w9,w13+ ushr v17.4s,v4.4s,#25+ eor w10,w10,w14+ ushr v21.4s,v5.4s,#25+ eor w11,w11,w15+ ushr v25.4s,v6.4s,#25+ eor w12,w12,w16+ ushr v29.4s,v7.4s,#25+ ror w9,w9,#25+ sli v17.4s,v4.4s,#7+ ror w10,w10,#25+ sli v21.4s,v5.4s,#7+ ror w11,w11,#25+ sli v25.4s,v6.4s,#7+ ror w12,w12,#25+ sli v29.4s,v7.4s,#7+ add v16.4s,v16.4s,v21.4s+ add v20.4s,v20.4s,v25.4s+ add v24.4s,v24.4s,v29.4s+ add v28.4s,v28.4s,v17.4s+ eor v31.16b,v31.16b,v16.16b+ eor v19.16b,v19.16b,v20.16b+ eor v23.16b,v23.16b,v24.16b+ eor v27.16b,v27.16b,v28.16b+ add w5,w5,w10+ rev32 v31.8h,v31.8h+ add w6,w6,w11+ rev32 v19.8h,v19.8h+ add w7,w7,w12+ rev32 v23.8h,v23.8h+ add w8,w8,w9+ rev32 v27.8h,v27.8h+ eor w21,w21,w5+ add v26.4s,v26.4s,v31.4s+ eor w17,w17,w6+ add v30.4s,v30.4s,v19.4s+ eor w19,w19,w7+ add v18.4s,v18.4s,v23.4s+ eor w20,w20,w8+ add v22.4s,v22.4s,v27.4s+ ror w21,w21,#16+ eor v4.16b,v21.16b,v26.16b+ ror w17,w17,#16+ eor v5.16b,v25.16b,v30.16b+ ror w19,w19,#16+ eor v6.16b,v29.16b,v18.16b+ ror w20,w20,#16+ eor v7.16b,v17.16b,v22.16b+ add w15,w15,w21+ ushr v21.4s,v4.4s,#20+ add w16,w16,w17+ ushr v25.4s,v5.4s,#20+ add w13,w13,w19+ ushr v29.4s,v6.4s,#20+ add w14,w14,w20+ ushr v17.4s,v7.4s,#20+ eor w10,w10,w15+ sli v21.4s,v4.4s,#12+ eor w11,w11,w16+ sli v25.4s,v5.4s,#12+ eor w12,w12,w13+ sli v29.4s,v6.4s,#12+ eor w9,w9,w14+ sli v17.4s,v7.4s,#12+ ror w10,w10,#20+ add v16.4s,v16.4s,v21.4s+ ror w11,w11,#20+ add v20.4s,v20.4s,v25.4s+ ror w12,w12,#20+ add v24.4s,v24.4s,v29.4s+ ror w9,w9,#20+ add v28.4s,v28.4s,v17.4s+ add w5,w5,w10+ eor v4.16b,v31.16b,v16.16b+ add w6,w6,w11+ eor v5.16b,v19.16b,v20.16b+ add w7,w7,w12+ eor v6.16b,v23.16b,v24.16b+ add w8,w8,w9+ eor v7.16b,v27.16b,v28.16b+ eor w21,w21,w5+ tbl v31.16b,{v4.16b},v9.16b+ eor w17,w17,w6+ tbl v19.16b,{v5.16b},v9.16b+ eor w19,w19,w7+ tbl v23.16b,{v6.16b},v9.16b+ eor w20,w20,w8+ tbl v27.16b,{v7.16b},v9.16b+ ror w21,w21,#24+ add v26.4s,v26.4s,v31.4s+ ror w17,w17,#24+ add v30.4s,v30.4s,v19.4s+ ror w19,w19,#24+ add v18.4s,v18.4s,v23.4s+ ror w20,w20,#24+ add v22.4s,v22.4s,v27.4s+ add w15,w15,w21+ eor v4.16b,v21.16b,v26.16b+ add w16,w16,w17+ eor v5.16b,v25.16b,v30.16b+ add w13,w13,w19+ eor v6.16b,v29.16b,v18.16b+ add w14,w14,w20+ eor v7.16b,v17.16b,v22.16b+ eor w10,w10,w15+ ushr v21.4s,v4.4s,#25+ eor w11,w11,w16+ ushr v25.4s,v5.4s,#25+ eor w12,w12,w13+ ushr v29.4s,v6.4s,#25+ eor w9,w9,w14+ ushr v17.4s,v7.4s,#25+ ror w10,w10,#25+ sli v21.4s,v4.4s,#7+ ror w11,w11,#25+ sli v25.4s,v5.4s,#7+ ror w12,w12,#25+ sli v29.4s,v6.4s,#7+ ror w9,w9,#25+ sli v17.4s,v7.4s,#7+ cbnz x4,Loop_neon++ add v19.4s,v19.4s,v8.4s++ zip1 v4.4s,v16.4s,v20.4s // transpose data+ zip1 v5.4s,v24.4s,v28.4s+ zip2 v6.4s,v16.4s,v20.4s+ zip2 v7.4s,v24.4s,v28.4s+ zip1 v16.2d,v4.2d,v5.2d+ zip2 v20.2d,v4.2d,v5.2d+ zip1 v24.2d,v6.2d,v7.2d+ zip2 v28.2d,v6.2d,v7.2d++ zip1 v4.4s,v17.4s,v21.4s+ zip1 v5.4s,v25.4s,v29.4s+ zip2 v6.4s,v17.4s,v21.4s+ zip2 v7.4s,v25.4s,v29.4s+ zip1 v17.2d,v4.2d,v5.2d+ zip2 v21.2d,v4.2d,v5.2d+ zip1 v25.2d,v6.2d,v7.2d+ zip2 v29.2d,v6.2d,v7.2d++ zip1 v4.4s,v18.4s,v22.4s+ add w5,w5,w22 // accumulate key block+ zip1 v5.4s,v26.4s,v30.4s+ add x6,x6,x22,lsr#32+ zip2 v6.4s,v18.4s,v22.4s+ add w7,w7,w23+ zip2 v7.4s,v26.4s,v30.4s+ add x8,x8,x23,lsr#32+ zip1 v18.2d,v4.2d,v5.2d+ add w9,w9,w24+ zip2 v22.2d,v4.2d,v5.2d+ add x10,x10,x24,lsr#32+ zip1 v26.2d,v6.2d,v7.2d+ add w11,w11,w25+ zip2 v30.2d,v6.2d,v7.2d+ add x12,x12,x25,lsr#32++ zip1 v4.4s,v19.4s,v23.4s+ add w13,w13,w26+ zip1 v5.4s,v27.4s,v31.4s+ add x14,x14,x26,lsr#32+ zip2 v6.4s,v19.4s,v23.4s+ add w15,w15,w27+ zip2 v7.4s,v27.4s,v31.4s+ add x16,x16,x27,lsr#32+ zip1 v19.2d,v4.2d,v5.2d+ add w17,w17,w28+ zip2 v23.2d,v4.2d,v5.2d+ add x19,x19,x28,lsr#32+ zip1 v27.2d,v6.2d,v7.2d+ add w20,w20,w30+ zip2 v31.2d,v6.2d,v7.2d+ add x21,x21,x30,lsr#32++ b.lo Ltail_neon++ add x5,x5,x6,lsl#32 // pack+ add x7,x7,x8,lsl#32+ ldp x6,x8,[x1,#0] // load input+ add v16.4s,v16.4s,v0.4s // accumulate key block+ add x9,x9,x10,lsl#32+ add x11,x11,x12,lsl#32+ ldp x10,x12,[x1,#16]+ add v17.4s,v17.4s,v1.4s+ add x13,x13,x14,lsl#32+ add x15,x15,x16,lsl#32+ ldp x14,x16,[x1,#32]+ add v18.4s,v18.4s,v2.4s+ add x17,x17,x19,lsl#32+ add x20,x20,x21,lsl#32+ ldp x19,x21,[x1,#48]+ add v19.4s,v19.4s,v3.4s+ add x1,x1,#64+#ifdef __AARCH64EB__+ rev x5,x5+ rev x7,x7+ rev x9,x9+ rev x11,x11+ rev x13,x13+ rev x15,x15+ rev x17,x17+ rev x20,x20+#endif+ ld1 {v4.16b,v5.16b,v6.16b,v7.16b},[x1],#64+ eor x5,x5,x6+ add v20.4s,v20.4s,v0.4s+ eor x7,x7,x8+ add v21.4s,v21.4s,v1.4s+ eor x9,x9,x10+ add v22.4s,v22.4s,v2.4s+ eor x11,x11,x12+ add v23.4s,v23.4s,v3.4s+ eor x13,x13,x14+ eor v16.16b,v16.16b,v4.16b+ movi v4.4s,#5+ eor x15,x15,x16+ eor v17.16b,v17.16b,v5.16b+ eor x17,x17,x19+ eor v18.16b,v18.16b,v6.16b+ eor x20,x20,x21+ eor v19.16b,v19.16b,v7.16b+ add v8.4s,v8.4s,v4.4s // += 5+ ld1 {v4.16b,v5.16b,v6.16b,v7.16b},[x1],#64++ stp x5,x7,[x0,#0] // store output+ add x28,x28,#5 // increment counter+ stp x9,x11,[x0,#16]+ stp x13,x15,[x0,#32]+ stp x17,x20,[x0,#48]+ add x0,x0,#64++ st1 {v16.16b,v17.16b,v18.16b,v19.16b},[x0],#64+ add v24.4s,v24.4s,v0.4s+ add v25.4s,v25.4s,v1.4s+ add v26.4s,v26.4s,v2.4s+ add v27.4s,v27.4s,v3.4s+ ld1 {v16.16b,v17.16b,v18.16b,v19.16b},[x1],#64++ eor v20.16b,v20.16b,v4.16b+ eor v21.16b,v21.16b,v5.16b+ eor v22.16b,v22.16b,v6.16b+ eor v23.16b,v23.16b,v7.16b+ st1 {v20.16b,v21.16b,v22.16b,v23.16b},[x0],#64+ add v28.4s,v28.4s,v0.4s+ add v29.4s,v29.4s,v1.4s+ add v30.4s,v30.4s,v2.4s+ add v31.4s,v31.4s,v3.4s+ ld1 {v20.16b,v21.16b,v22.16b,v23.16b},[x1],#64++ eor v24.16b,v24.16b,v16.16b+ eor v25.16b,v25.16b,v17.16b+ eor v26.16b,v26.16b,v18.16b+ eor v27.16b,v27.16b,v19.16b+ st1 {v24.16b,v25.16b,v26.16b,v27.16b},[x0],#64++ eor v28.16b,v28.16b,v20.16b+ eor v29.16b,v29.16b,v21.16b+ eor v30.16b,v30.16b,v22.16b+ eor v31.16b,v31.16b,v23.16b+ st1 {v28.16b,v29.16b,v30.16b,v31.16b},[x0],#64++ b.hi Loop_outer_neon++ ldp d8,d9,[sp] // meet ABI requirements+ eor v1.16b,v1.16b,v1.16b // cleanse key and nonce+ eor v2.16b,v2.16b,v2.16b+ eor v3.16b,v3.16b,v3.16b++ ldp x19,x20,[x29,#2*__SIZEOF_POINTER__]+ add sp,sp,#64+ ldp x21,x22,[x29,#4*__SIZEOF_POINTER__]+ ldp x23,x24,[x29,#6*__SIZEOF_POINTER__]+ ldp x25,x26,[x29,#8*__SIZEOF_POINTER__]+ ldp x27,x28,[x29,#10*__SIZEOF_POINTER__]+ ldp x29,x30,[sp],#12*__SIZEOF_POINTER__+.long 0xd50323bf // autiasp+ ret++.align 4+Ltail_neon:+ add x2,x2,#320+ ldp d8,d9,[sp] // meet ABI requirements+ cmp x2,#64+ b.lo Less_than_64_neon++ add x5,x5,x6,lsl#32 // pack+ add x7,x7,x8,lsl#32+ ldp x6,x8,[x1,#0] // load input+ add x9,x9,x10,lsl#32+ add x11,x11,x12,lsl#32+ ldp x10,x12,[x1,#16]+ add x13,x13,x14,lsl#32+ add x15,x15,x16,lsl#32+ ldp x14,x16,[x1,#32]+ add x17,x17,x19,lsl#32+ add x20,x20,x21,lsl#32+ ldp x19,x21,[x1,#48]+ add x1,x1,#64+#ifdef __AARCH64EB__+ rev x5,x5+ rev x7,x7+ rev x9,x9+ rev x11,x11+ rev x13,x13+ rev x15,x15+ rev x17,x17+ rev x20,x20+#endif+ eor x5,x5,x6+ eor x7,x7,x8+ eor x9,x9,x10+ eor x11,x11,x12+ eor x13,x13,x14+ eor x15,x15,x16+ eor x17,x17,x19+ eor x20,x20,x21++ stp x5,x7,[x0,#0] // store output+ add v16.4s,v16.4s,v0.4s // accumulate key block+ stp x9,x11,[x0,#16]+ add v17.4s,v17.4s,v1.4s+ stp x13,x15,[x0,#32]+ add v18.4s,v18.4s,v2.4s+ stp x17,x20,[x0,#48]+ add v19.4s,v19.4s,v3.4s+ add x0,x0,#64+ b.eq Ldone_neon+ sub x2,x2,#64+ cmp x2,#64+ b.lo Last_neon++ ld1 {v4.16b,v5.16b,v6.16b,v7.16b},[x1],#64+ eor v16.16b,v16.16b,v4.16b+ eor v17.16b,v17.16b,v5.16b+ eor v18.16b,v18.16b,v6.16b+ eor v19.16b,v19.16b,v7.16b+ st1 {v16.16b,v17.16b,v18.16b,v19.16b},[x0],#64+ b.eq Ldone_neon++ add v16.4s,v20.4s,v0.4s+ add v17.4s,v21.4s,v1.4s+ sub x2,x2,#64+ add v18.4s,v22.4s,v2.4s+ cmp x2,#64+ add v19.4s,v23.4s,v3.4s+ b.lo Last_neon++ ld1 {v4.16b,v5.16b,v6.16b,v7.16b},[x1],#64+ eor v20.16b,v16.16b,v4.16b+ eor v21.16b,v17.16b,v5.16b+ eor v22.16b,v18.16b,v6.16b+ eor v23.16b,v19.16b,v7.16b+ st1 {v20.16b,v21.16b,v22.16b,v23.16b},[x0],#64+ b.eq Ldone_neon++ add v16.4s,v24.4s,v0.4s+ add v17.4s,v25.4s,v1.4s+ sub x2,x2,#64+ add v18.4s,v26.4s,v2.4s+ cmp x2,#64+ add v19.4s,v27.4s,v3.4s+ b.lo Last_neon++ ld1 {v4.16b,v5.16b,v6.16b,v7.16b},[x1],#64+ eor v24.16b,v16.16b,v4.16b+ eor v25.16b,v17.16b,v5.16b+ eor v26.16b,v18.16b,v6.16b+ eor v27.16b,v19.16b,v7.16b+ st1 {v24.16b,v25.16b,v26.16b,v27.16b},[x0],#64+ b.eq Ldone_neon++ add v16.4s,v28.4s,v0.4s+ add v17.4s,v29.4s,v1.4s+ add v18.4s,v30.4s,v2.4s+ add v19.4s,v31.4s,v3.4s+ sub x2,x2,#64++Last_neon:+ st1 {v16.16b,v17.16b,v18.16b,v19.16b},[sp] // off-load complete block++ sub x0,x0,#1+ add x1,x1,x2+ add x0,x0,x2+ add x4,sp,x2+ neg x2,x2++Loop_tail_neon:+ ldrb w10,[x1,x2]+ ldrb w11,[x4,x2]+ add x2,x2,#1+ eor w10,w10,w11+ strb w10,[x0,x2]+ cbnz x2,Loop_tail_neon++ stp q0,q0,[sp,#0] // wipe off-load area+ stp q0,q0,[sp,#32] // [with known constant]++Ldone_neon:+ eor v1.16b,v1.16b,v1.16b // cleanse key and nonce+ eor v2.16b,v2.16b,v2.16b+ eor v3.16b,v3.16b,v3.16b++ ldp x19,x20,[x29,#2*__SIZEOF_POINTER__]+ add sp,sp,#64+ ldp x21,x22,[x29,#4*__SIZEOF_POINTER__]+ ldp x23,x24,[x29,#6*__SIZEOF_POINTER__]+ ldp x25,x26,[x29,#8*__SIZEOF_POINTER__]+ ldp x27,x28,[x29,#10*__SIZEOF_POINTER__]+ ldp x29,x30,[sp],#12*__SIZEOF_POINTER__+.long 0xd50323bf // autiasp+ ret++.align 4+Less_than_64_neon:+ eor v1.16b,v1.16b,v1.16b // cleanse key and nonce+ eor v2.16b,v2.16b,v2.16b+ eor v3.16b,v3.16b,v3.16b+ b Less_than_64+++.align 5+crypton_chacha20_asm_512_neon:+.long 0xd503233f // paciasp+ stp x29,x30,[sp,#-12*__SIZEOF_POINTER__]!+ add x29,sp,#0++ adr x5,Lsigma+ stp x19,x20,[sp,#2*__SIZEOF_POINTER__]+ stp x21,x22,[sp,#4*__SIZEOF_POINTER__]+ stp x23,x24,[sp,#6*__SIZEOF_POINTER__]+ stp x25,x26,[sp,#8*__SIZEOF_POINTER__]+ stp x27,x28,[sp,#10*__SIZEOF_POINTER__]++L512_or_more_neon:+ sub sp,sp,#128+64++ eor v7.16b,v7.16b,v7.16b+ ldp x22,x23,[x5] // load sigma+ ld1 {v0.4s},[x5],#16+ ldp x24,x25,[x3] // load key+ ldp x26,x27,[x3,#16]+ ld1 {v1.4s,v2.4s},[x3]+ ldp x28,x30,[x4] // load counter+ ld1 {v3.4s},[x4]+ ld1 {v7.s}[0],[x5]+ add x3,x5,#16+#ifdef __AARCH64EB__+ rev64 v0.4s,v0.4s+ ror x24,x24,#32+ ror x25,x25,#32+ ror x26,x26,#32+ ror x27,x27,#32+ ror x28,x28,#32+ ror x30,x30,#32+#endif+ add v3.4s,v3.4s,v7.4s // += 1+ stp q0,q1,[sp,#0] // off-load key block, invariant part+ add v3.4s,v3.4s,v7.4s // not typo+ str q2,[sp,#32]+ add v4.4s,v3.4s,v7.4s+ add v5.4s,v4.4s,v7.4s+ add v6.4s,v5.4s,v7.4s+ shl v7.4s,v7.4s,#2 // 1 -> 4++ stp d8,d9,[sp,#128+0] // meet ABI requirements+ stp d10,d11,[sp,#128+16]+ stp d12,d13,[sp,#128+32]+ stp d14,d15,[sp,#128+48]++ sub x2,x2,#512 // not typo++Loop_outer_512_neon:+ mov v8.16b,v0.16b+ mov v12.16b,v0.16b+ mov v16.16b,v0.16b+ mov v20.16b,v0.16b+ mov v24.16b,v0.16b+ mov v28.16b,v0.16b+ mov v9.16b,v1.16b+ mov w5,w22 // unpack key block+ mov v13.16b,v1.16b+ lsr x6,x22,#32+ mov v17.16b,v1.16b+ mov w7,w23+ mov v21.16b,v1.16b+ lsr x8,x23,#32+ mov v25.16b,v1.16b+ mov w9,w24+ mov v29.16b,v1.16b+ lsr x10,x24,#32+ mov v11.16b,v3.16b+ mov w11,w25+ mov v15.16b,v4.16b+ lsr x12,x25,#32+ mov v19.16b,v5.16b+ mov w13,w26+ mov v23.16b,v6.16b+ lsr x14,x26,#32+ mov v10.16b,v2.16b+ mov w15,w27+ mov v14.16b,v2.16b+ lsr x16,x27,#32+ add v27.4s,v11.4s,v7.4s // +4+ mov w17,w28+ add v31.4s,v15.4s,v7.4s // +4+ lsr x19,x28,#32+ mov v18.16b,v2.16b+ mov w20,w30+ mov v22.16b,v2.16b+ lsr x21,x30,#32+ mov v26.16b,v2.16b+ stp q3,q4,[sp,#48] // off-load key block, variable part+ mov v30.16b,v2.16b+ stp q5,q6,[sp,#80]++ mov x4,#5+ ld1 {v6.4s},[x3]+ subs x2,x2,#512+Loop_upper_neon:+ sub x4,x4,#1+ add v8.4s,v8.4s,v9.4s+ add w5,w5,w9+ add v12.4s,v12.4s,v13.4s+ add w6,w6,w10+ add v16.4s,v16.4s,v17.4s+ add w7,w7,w11+ add v20.4s,v20.4s,v21.4s+ add w8,w8,w12+ add v24.4s,v24.4s,v25.4s+ eor w17,w17,w5+ add v28.4s,v28.4s,v29.4s+ eor w19,w19,w6+ eor v11.16b,v11.16b,v8.16b+ eor w20,w20,w7+ eor v15.16b,v15.16b,v12.16b+ eor w21,w21,w8+ eor v19.16b,v19.16b,v16.16b+ ror w17,w17,#16+ eor v23.16b,v23.16b,v20.16b+ ror w19,w19,#16+ eor v27.16b,v27.16b,v24.16b+ ror w20,w20,#16+ eor v31.16b,v31.16b,v28.16b+ ror w21,w21,#16+ rev32 v11.8h,v11.8h+ add w13,w13,w17+ rev32 v15.8h,v15.8h+ add w14,w14,w19+ rev32 v19.8h,v19.8h+ add w15,w15,w20+ rev32 v23.8h,v23.8h+ add w16,w16,w21+ rev32 v27.8h,v27.8h+ eor w9,w9,w13+ rev32 v31.8h,v31.8h+ eor w10,w10,w14+ add v10.4s,v10.4s,v11.4s+ eor w11,w11,w15+ add v14.4s,v14.4s,v15.4s+ eor w12,w12,w16+ add v18.4s,v18.4s,v19.4s+ ror w9,w9,#20+ add v22.4s,v22.4s,v23.4s+ ror w10,w10,#20+ add v26.4s,v26.4s,v27.4s+ ror w11,w11,#20+ add v30.4s,v30.4s,v31.4s+ ror w12,w12,#20+ eor v0.16b,v9.16b,v10.16b+ add w5,w5,w9+ eor v1.16b,v13.16b,v14.16b+ add w6,w6,w10+ eor v2.16b,v17.16b,v18.16b+ add w7,w7,w11+ eor v3.16b,v21.16b,v22.16b+ add w8,w8,w12+ eor v4.16b,v25.16b,v26.16b+ eor w17,w17,w5+ eor v5.16b,v29.16b,v30.16b+ eor w19,w19,w6+ ushr v9.4s,v0.4s,#20+ eor w20,w20,w7+ ushr v13.4s,v1.4s,#20+ eor w21,w21,w8+ ushr v17.4s,v2.4s,#20+ ror w17,w17,#24+ ushr v21.4s,v3.4s,#20+ ror w19,w19,#24+ ushr v25.4s,v4.4s,#20+ ror w20,w20,#24+ ushr v29.4s,v5.4s,#20+ ror w21,w21,#24+ sli v9.4s,v0.4s,#12+ add w13,w13,w17+ sli v13.4s,v1.4s,#12+ add w14,w14,w19+ sli v17.4s,v2.4s,#12+ add w15,w15,w20+ sli v21.4s,v3.4s,#12+ add w16,w16,w21+ sli v25.4s,v4.4s,#12+ eor w9,w9,w13+ sli v29.4s,v5.4s,#12+ eor w10,w10,w14+ add v8.4s,v8.4s,v9.4s+ eor w11,w11,w15+ add v12.4s,v12.4s,v13.4s+ eor w12,w12,w16+ add v16.4s,v16.4s,v17.4s+ ror w9,w9,#25+ add v20.4s,v20.4s,v21.4s+ ror w10,w10,#25+ add v24.4s,v24.4s,v25.4s+ ror w11,w11,#25+ add v28.4s,v28.4s,v29.4s+ ror w12,w12,#25+ eor v11.16b,v11.16b,v8.16b+ add w5,w5,w10+ eor v15.16b,v15.16b,v12.16b+ add w6,w6,w11+ eor v19.16b,v19.16b,v16.16b+ add w7,w7,w12+ eor v23.16b,v23.16b,v20.16b+ add w8,w8,w9+ eor v27.16b,v27.16b,v24.16b+ eor w21,w21,w5+ eor v31.16b,v31.16b,v28.16b+ eor w17,w17,w6+ tbl v11.16b,{v11.16b},v6.16b+ eor w19,w19,w7+ tbl v15.16b,{v15.16b},v6.16b+ eor w20,w20,w8+ tbl v19.16b,{v19.16b},v6.16b+ ror w21,w21,#16+ tbl v23.16b,{v23.16b},v6.16b+ ror w17,w17,#16+ tbl v27.16b,{v27.16b},v6.16b+ ror w19,w19,#16+ tbl v31.16b,{v31.16b},v6.16b+ ror w20,w20,#16+ add v10.4s,v10.4s,v11.4s+ add w15,w15,w21+ add v14.4s,v14.4s,v15.4s+ add w16,w16,w17+ add v18.4s,v18.4s,v19.4s+ add w13,w13,w19+ add v22.4s,v22.4s,v23.4s+ add w14,w14,w20+ add v26.4s,v26.4s,v27.4s+ eor w10,w10,w15+ add v30.4s,v30.4s,v31.4s+ eor w11,w11,w16+ eor v0.16b,v9.16b,v10.16b+ eor w12,w12,w13+ eor v1.16b,v13.16b,v14.16b+ eor w9,w9,w14+ eor v2.16b,v17.16b,v18.16b+ ror w10,w10,#20+ eor v3.16b,v21.16b,v22.16b+ ror w11,w11,#20+ eor v4.16b,v25.16b,v26.16b+ ror w12,w12,#20+ eor v5.16b,v29.16b,v30.16b+ ror w9,w9,#20+ ushr v9.4s,v0.4s,#25+ add w5,w5,w10+ ushr v13.4s,v1.4s,#25+ add w6,w6,w11+ ushr v17.4s,v2.4s,#25+ add w7,w7,w12+ ushr v21.4s,v3.4s,#25+ add w8,w8,w9+ ushr v25.4s,v4.4s,#25+ eor w21,w21,w5+ ushr v29.4s,v5.4s,#25+ eor w17,w17,w6+ sli v9.4s,v0.4s,#7+ eor w19,w19,w7+ sli v13.4s,v1.4s,#7+ eor w20,w20,w8+ sli v17.4s,v2.4s,#7+ ror w21,w21,#24+ sli v21.4s,v3.4s,#7+ ror w17,w17,#24+ sli v25.4s,v4.4s,#7+ ror w19,w19,#24+ sli v29.4s,v5.4s,#7+ ror w20,w20,#24+ ext v10.16b,v10.16b,v10.16b,#8+ add w15,w15,w21+ ext v14.16b,v14.16b,v14.16b,#8+ add w16,w16,w17+ ext v18.16b,v18.16b,v18.16b,#8+ add w13,w13,w19+ ext v22.16b,v22.16b,v22.16b,#8+ add w14,w14,w20+ ext v26.16b,v26.16b,v26.16b,#8+ eor w10,w10,w15+ ext v30.16b,v30.16b,v30.16b,#8+ eor w11,w11,w16+ ext v11.16b,v11.16b,v11.16b,#12+ eor w12,w12,w13+ ext v15.16b,v15.16b,v15.16b,#12+ eor w9,w9,w14+ ext v19.16b,v19.16b,v19.16b,#12+ ror w10,w10,#25+ ext v23.16b,v23.16b,v23.16b,#12+ ror w11,w11,#25+ ext v27.16b,v27.16b,v27.16b,#12+ ror w12,w12,#25+ ext v31.16b,v31.16b,v31.16b,#12+ ror w9,w9,#25+ ext v9.16b,v9.16b,v9.16b,#4+ ext v13.16b,v13.16b,v13.16b,#4+ ext v17.16b,v17.16b,v17.16b,#4+ ext v21.16b,v21.16b,v21.16b,#4+ ext v25.16b,v25.16b,v25.16b,#4+ ext v29.16b,v29.16b,v29.16b,#4+ add v8.4s,v8.4s,v9.4s+ add w5,w5,w9+ add v12.4s,v12.4s,v13.4s+ add w6,w6,w10+ add v16.4s,v16.4s,v17.4s+ add w7,w7,w11+ add v20.4s,v20.4s,v21.4s+ add w8,w8,w12+ add v24.4s,v24.4s,v25.4s+ eor w17,w17,w5+ add v28.4s,v28.4s,v29.4s+ eor w19,w19,w6+ eor v11.16b,v11.16b,v8.16b+ eor w20,w20,w7+ eor v15.16b,v15.16b,v12.16b+ eor w21,w21,w8+ eor v19.16b,v19.16b,v16.16b+ ror w17,w17,#16+ eor v23.16b,v23.16b,v20.16b+ ror w19,w19,#16+ eor v27.16b,v27.16b,v24.16b+ ror w20,w20,#16+ eor v31.16b,v31.16b,v28.16b+ ror w21,w21,#16+ rev32 v11.8h,v11.8h+ add w13,w13,w17+ rev32 v15.8h,v15.8h+ add w14,w14,w19+ rev32 v19.8h,v19.8h+ add w15,w15,w20+ rev32 v23.8h,v23.8h+ add w16,w16,w21+ rev32 v27.8h,v27.8h+ eor w9,w9,w13+ rev32 v31.8h,v31.8h+ eor w10,w10,w14+ add v10.4s,v10.4s,v11.4s+ eor w11,w11,w15+ add v14.4s,v14.4s,v15.4s+ eor w12,w12,w16+ add v18.4s,v18.4s,v19.4s+ ror w9,w9,#20+ add v22.4s,v22.4s,v23.4s+ ror w10,w10,#20+ add v26.4s,v26.4s,v27.4s+ ror w11,w11,#20+ add v30.4s,v30.4s,v31.4s+ ror w12,w12,#20+ eor v0.16b,v9.16b,v10.16b+ add w5,w5,w9+ eor v1.16b,v13.16b,v14.16b+ add w6,w6,w10+ eor v2.16b,v17.16b,v18.16b+ add w7,w7,w11+ eor v3.16b,v21.16b,v22.16b+ add w8,w8,w12+ eor v4.16b,v25.16b,v26.16b+ eor w17,w17,w5+ eor v5.16b,v29.16b,v30.16b+ eor w19,w19,w6+ ushr v9.4s,v0.4s,#20+ eor w20,w20,w7+ ushr v13.4s,v1.4s,#20+ eor w21,w21,w8+ ushr v17.4s,v2.4s,#20+ ror w17,w17,#24+ ushr v21.4s,v3.4s,#20+ ror w19,w19,#24+ ushr v25.4s,v4.4s,#20+ ror w20,w20,#24+ ushr v29.4s,v5.4s,#20+ ror w21,w21,#24+ sli v9.4s,v0.4s,#12+ add w13,w13,w17+ sli v13.4s,v1.4s,#12+ add w14,w14,w19+ sli v17.4s,v2.4s,#12+ add w15,w15,w20+ sli v21.4s,v3.4s,#12+ add w16,w16,w21+ sli v25.4s,v4.4s,#12+ eor w9,w9,w13+ sli v29.4s,v5.4s,#12+ eor w10,w10,w14+ add v8.4s,v8.4s,v9.4s+ eor w11,w11,w15+ add v12.4s,v12.4s,v13.4s+ eor w12,w12,w16+ add v16.4s,v16.4s,v17.4s+ ror w9,w9,#25+ add v20.4s,v20.4s,v21.4s+ ror w10,w10,#25+ add v24.4s,v24.4s,v25.4s+ ror w11,w11,#25+ add v28.4s,v28.4s,v29.4s+ ror w12,w12,#25+ eor v11.16b,v11.16b,v8.16b+ add w5,w5,w10+ eor v15.16b,v15.16b,v12.16b+ add w6,w6,w11+ eor v19.16b,v19.16b,v16.16b+ add w7,w7,w12+ eor v23.16b,v23.16b,v20.16b+ add w8,w8,w9+ eor v27.16b,v27.16b,v24.16b+ eor w21,w21,w5+ eor v31.16b,v31.16b,v28.16b+ eor w17,w17,w6+ tbl v11.16b,{v11.16b},v6.16b+ eor w19,w19,w7+ tbl v15.16b,{v15.16b},v6.16b+ eor w20,w20,w8+ tbl v19.16b,{v19.16b},v6.16b+ ror w21,w21,#16+ tbl v23.16b,{v23.16b},v6.16b+ ror w17,w17,#16+ tbl v27.16b,{v27.16b},v6.16b+ ror w19,w19,#16+ tbl v31.16b,{v31.16b},v6.16b+ ror w20,w20,#16+ add v10.4s,v10.4s,v11.4s+ add w15,w15,w21+ add v14.4s,v14.4s,v15.4s+ add w16,w16,w17+ add v18.4s,v18.4s,v19.4s+ add w13,w13,w19+ add v22.4s,v22.4s,v23.4s+ add w14,w14,w20+ add v26.4s,v26.4s,v27.4s+ eor w10,w10,w15+ add v30.4s,v30.4s,v31.4s+ eor w11,w11,w16+ eor v0.16b,v9.16b,v10.16b+ eor w12,w12,w13+ eor v1.16b,v13.16b,v14.16b+ eor w9,w9,w14+ eor v2.16b,v17.16b,v18.16b+ ror w10,w10,#20+ eor v3.16b,v21.16b,v22.16b+ ror w11,w11,#20+ eor v4.16b,v25.16b,v26.16b+ ror w12,w12,#20+ eor v5.16b,v29.16b,v30.16b+ ror w9,w9,#20+ ushr v9.4s,v0.4s,#25+ add w5,w5,w10+ ushr v13.4s,v1.4s,#25+ add w6,w6,w11+ ushr v17.4s,v2.4s,#25+ add w7,w7,w12+ ushr v21.4s,v3.4s,#25+ add w8,w8,w9+ ushr v25.4s,v4.4s,#25+ eor w21,w21,w5+ ushr v29.4s,v5.4s,#25+ eor w17,w17,w6+ sli v9.4s,v0.4s,#7+ eor w19,w19,w7+ sli v13.4s,v1.4s,#7+ eor w20,w20,w8+ sli v17.4s,v2.4s,#7+ ror w21,w21,#24+ sli v21.4s,v3.4s,#7+ ror w17,w17,#24+ sli v25.4s,v4.4s,#7+ ror w19,w19,#24+ sli v29.4s,v5.4s,#7+ ror w20,w20,#24+ ext v10.16b,v10.16b,v10.16b,#8+ add w15,w15,w21+ ext v14.16b,v14.16b,v14.16b,#8+ add w16,w16,w17+ ext v18.16b,v18.16b,v18.16b,#8+ add w13,w13,w19+ ext v22.16b,v22.16b,v22.16b,#8+ add w14,w14,w20+ ext v26.16b,v26.16b,v26.16b,#8+ eor w10,w10,w15+ ext v30.16b,v30.16b,v30.16b,#8+ eor w11,w11,w16+ ext v11.16b,v11.16b,v11.16b,#4+ eor w12,w12,w13+ ext v15.16b,v15.16b,v15.16b,#4+ eor w9,w9,w14+ ext v19.16b,v19.16b,v19.16b,#4+ ror w10,w10,#25+ ext v23.16b,v23.16b,v23.16b,#4+ ror w11,w11,#25+ ext v27.16b,v27.16b,v27.16b,#4+ ror w12,w12,#25+ ext v31.16b,v31.16b,v31.16b,#4+ ror w9,w9,#25+ ext v9.16b,v9.16b,v9.16b,#12+ ext v13.16b,v13.16b,v13.16b,#12+ ext v17.16b,v17.16b,v17.16b,#12+ ext v21.16b,v21.16b,v21.16b,#12+ ext v25.16b,v25.16b,v25.16b,#12+ ext v29.16b,v29.16b,v29.16b,#12+ cbnz x4,Loop_upper_neon++ add w5,w5,w22 // accumulate key block+ add x6,x6,x22,lsr#32+ add w7,w7,w23+ add x8,x8,x23,lsr#32+ add w9,w9,w24+ add x10,x10,x24,lsr#32+ add w11,w11,w25+ add x12,x12,x25,lsr#32+ add w13,w13,w26+ add x14,x14,x26,lsr#32+ add w15,w15,w27+ add x16,x16,x27,lsr#32+ add w17,w17,w28+ add x19,x19,x28,lsr#32+ add w20,w20,w30+ add x21,x21,x30,lsr#32++ add x5,x5,x6,lsl#32 // pack+ add x7,x7,x8,lsl#32+ ldp x6,x8,[x1,#0] // load input+ add x9,x9,x10,lsl#32+ add x11,x11,x12,lsl#32+ ldp x10,x12,[x1,#16]+ add x13,x13,x14,lsl#32+ add x15,x15,x16,lsl#32+ ldp x14,x16,[x1,#32]+ add x17,x17,x19,lsl#32+ add x20,x20,x21,lsl#32+ ldp x19,x21,[x1,#48]+ add x1,x1,#64+#ifdef __AARCH64EB__+ rev x5,x5+ rev x7,x7+ rev x9,x9+ rev x11,x11+ rev x13,x13+ rev x15,x15+ rev x17,x17+ rev x20,x20+#endif+ eor x5,x5,x6+ eor x7,x7,x8+ eor x9,x9,x10+ eor x11,x11,x12+ eor x13,x13,x14+ eor x15,x15,x16+ eor x17,x17,x19+ eor x20,x20,x21++ stp x5,x7,[x0,#0] // store output+ add x28,x28,#1 // increment counter+ mov w5,w22 // unpack key block+ lsr x6,x22,#32+ stp x9,x11,[x0,#16]+ mov w7,w23+ lsr x8,x23,#32+ stp x13,x15,[x0,#32]+ mov w9,w24+ lsr x10,x24,#32+ stp x17,x20,[x0,#48]+ add x0,x0,#64+ mov w11,w25+ lsr x12,x25,#32+ mov w13,w26+ lsr x14,x26,#32+ mov w15,w27+ lsr x16,x27,#32+ mov w17,w28+ lsr x19,x28,#32+ mov w20,w30+ lsr x21,x30,#32++ mov x4,#5+Loop_lower_neon:+ sub x4,x4,#1+ add v8.4s,v8.4s,v9.4s+ add w5,w5,w9+ add v12.4s,v12.4s,v13.4s+ add w6,w6,w10+ add v16.4s,v16.4s,v17.4s+ add w7,w7,w11+ add v20.4s,v20.4s,v21.4s+ add w8,w8,w12+ add v24.4s,v24.4s,v25.4s+ eor w17,w17,w5+ add v28.4s,v28.4s,v29.4s+ eor w19,w19,w6+ eor v11.16b,v11.16b,v8.16b+ eor w20,w20,w7+ eor v15.16b,v15.16b,v12.16b+ eor w21,w21,w8+ eor v19.16b,v19.16b,v16.16b+ ror w17,w17,#16+ eor v23.16b,v23.16b,v20.16b+ ror w19,w19,#16+ eor v27.16b,v27.16b,v24.16b+ ror w20,w20,#16+ eor v31.16b,v31.16b,v28.16b+ ror w21,w21,#16+ rev32 v11.8h,v11.8h+ add w13,w13,w17+ rev32 v15.8h,v15.8h+ add w14,w14,w19+ rev32 v19.8h,v19.8h+ add w15,w15,w20+ rev32 v23.8h,v23.8h+ add w16,w16,w21+ rev32 v27.8h,v27.8h+ eor w9,w9,w13+ rev32 v31.8h,v31.8h+ eor w10,w10,w14+ add v10.4s,v10.4s,v11.4s+ eor w11,w11,w15+ add v14.4s,v14.4s,v15.4s+ eor w12,w12,w16+ add v18.4s,v18.4s,v19.4s+ ror w9,w9,#20+ add v22.4s,v22.4s,v23.4s+ ror w10,w10,#20+ add v26.4s,v26.4s,v27.4s+ ror w11,w11,#20+ add v30.4s,v30.4s,v31.4s+ ror w12,w12,#20+ eor v0.16b,v9.16b,v10.16b+ add w5,w5,w9+ eor v1.16b,v13.16b,v14.16b+ add w6,w6,w10+ eor v2.16b,v17.16b,v18.16b+ add w7,w7,w11+ eor v3.16b,v21.16b,v22.16b+ add w8,w8,w12+ eor v4.16b,v25.16b,v26.16b+ eor w17,w17,w5+ eor v5.16b,v29.16b,v30.16b+ eor w19,w19,w6+ ushr v9.4s,v0.4s,#20+ eor w20,w20,w7+ ushr v13.4s,v1.4s,#20+ eor w21,w21,w8+ ushr v17.4s,v2.4s,#20+ ror w17,w17,#24+ ushr v21.4s,v3.4s,#20+ ror w19,w19,#24+ ushr v25.4s,v4.4s,#20+ ror w20,w20,#24+ ushr v29.4s,v5.4s,#20+ ror w21,w21,#24+ sli v9.4s,v0.4s,#12+ add w13,w13,w17+ sli v13.4s,v1.4s,#12+ add w14,w14,w19+ sli v17.4s,v2.4s,#12+ add w15,w15,w20+ sli v21.4s,v3.4s,#12+ add w16,w16,w21+ sli v25.4s,v4.4s,#12+ eor w9,w9,w13+ sli v29.4s,v5.4s,#12+ eor w10,w10,w14+ add v8.4s,v8.4s,v9.4s+ eor w11,w11,w15+ add v12.4s,v12.4s,v13.4s+ eor w12,w12,w16+ add v16.4s,v16.4s,v17.4s+ ror w9,w9,#25+ add v20.4s,v20.4s,v21.4s+ ror w10,w10,#25+ add v24.4s,v24.4s,v25.4s+ ror w11,w11,#25+ add v28.4s,v28.4s,v29.4s+ ror w12,w12,#25+ eor v11.16b,v11.16b,v8.16b+ add w5,w5,w10+ eor v15.16b,v15.16b,v12.16b+ add w6,w6,w11+ eor v19.16b,v19.16b,v16.16b+ add w7,w7,w12+ eor v23.16b,v23.16b,v20.16b+ add w8,w8,w9+ eor v27.16b,v27.16b,v24.16b+ eor w21,w21,w5+ eor v31.16b,v31.16b,v28.16b+ eor w17,w17,w6+ tbl v11.16b,{v11.16b},v6.16b+ eor w19,w19,w7+ tbl v15.16b,{v15.16b},v6.16b+ eor w20,w20,w8+ tbl v19.16b,{v19.16b},v6.16b+ ror w21,w21,#16+ tbl v23.16b,{v23.16b},v6.16b+ ror w17,w17,#16+ tbl v27.16b,{v27.16b},v6.16b+ ror w19,w19,#16+ tbl v31.16b,{v31.16b},v6.16b+ ror w20,w20,#16+ add v10.4s,v10.4s,v11.4s+ add w15,w15,w21+ add v14.4s,v14.4s,v15.4s+ add w16,w16,w17+ add v18.4s,v18.4s,v19.4s+ add w13,w13,w19+ add v22.4s,v22.4s,v23.4s+ add w14,w14,w20+ add v26.4s,v26.4s,v27.4s+ eor w10,w10,w15+ add v30.4s,v30.4s,v31.4s+ eor w11,w11,w16+ eor v0.16b,v9.16b,v10.16b+ eor w12,w12,w13+ eor v1.16b,v13.16b,v14.16b+ eor w9,w9,w14+ eor v2.16b,v17.16b,v18.16b+ ror w10,w10,#20+ eor v3.16b,v21.16b,v22.16b+ ror w11,w11,#20+ eor v4.16b,v25.16b,v26.16b+ ror w12,w12,#20+ eor v5.16b,v29.16b,v30.16b+ ror w9,w9,#20+ ushr v9.4s,v0.4s,#25+ add w5,w5,w10+ ushr v13.4s,v1.4s,#25+ add w6,w6,w11+ ushr v17.4s,v2.4s,#25+ add w7,w7,w12+ ushr v21.4s,v3.4s,#25+ add w8,w8,w9+ ushr v25.4s,v4.4s,#25+ eor w21,w21,w5+ ushr v29.4s,v5.4s,#25+ eor w17,w17,w6+ sli v9.4s,v0.4s,#7+ eor w19,w19,w7+ sli v13.4s,v1.4s,#7+ eor w20,w20,w8+ sli v17.4s,v2.4s,#7+ ror w21,w21,#24+ sli v21.4s,v3.4s,#7+ ror w17,w17,#24+ sli v25.4s,v4.4s,#7+ ror w19,w19,#24+ sli v29.4s,v5.4s,#7+ ror w20,w20,#24+ ext v10.16b,v10.16b,v10.16b,#8+ add w15,w15,w21+ ext v14.16b,v14.16b,v14.16b,#8+ add w16,w16,w17+ ext v18.16b,v18.16b,v18.16b,#8+ add w13,w13,w19+ ext v22.16b,v22.16b,v22.16b,#8+ add w14,w14,w20+ ext v26.16b,v26.16b,v26.16b,#8+ eor w10,w10,w15+ ext v30.16b,v30.16b,v30.16b,#8+ eor w11,w11,w16+ ext v11.16b,v11.16b,v11.16b,#12+ eor w12,w12,w13+ ext v15.16b,v15.16b,v15.16b,#12+ eor w9,w9,w14+ ext v19.16b,v19.16b,v19.16b,#12+ ror w10,w10,#25+ ext v23.16b,v23.16b,v23.16b,#12+ ror w11,w11,#25+ ext v27.16b,v27.16b,v27.16b,#12+ ror w12,w12,#25+ ext v31.16b,v31.16b,v31.16b,#12+ ror w9,w9,#25+ ext v9.16b,v9.16b,v9.16b,#4+ ext v13.16b,v13.16b,v13.16b,#4+ ext v17.16b,v17.16b,v17.16b,#4+ ext v21.16b,v21.16b,v21.16b,#4+ ext v25.16b,v25.16b,v25.16b,#4+ ext v29.16b,v29.16b,v29.16b,#4+ add v8.4s,v8.4s,v9.4s+ add w5,w5,w9+ add v12.4s,v12.4s,v13.4s+ add w6,w6,w10+ add v16.4s,v16.4s,v17.4s+ add w7,w7,w11+ add v20.4s,v20.4s,v21.4s+ add w8,w8,w12+ add v24.4s,v24.4s,v25.4s+ eor w17,w17,w5+ add v28.4s,v28.4s,v29.4s+ eor w19,w19,w6+ eor v11.16b,v11.16b,v8.16b+ eor w20,w20,w7+ eor v15.16b,v15.16b,v12.16b+ eor w21,w21,w8+ eor v19.16b,v19.16b,v16.16b+ ror w17,w17,#16+ eor v23.16b,v23.16b,v20.16b+ ror w19,w19,#16+ eor v27.16b,v27.16b,v24.16b+ ror w20,w20,#16+ eor v31.16b,v31.16b,v28.16b+ ror w21,w21,#16+ rev32 v11.8h,v11.8h+ add w13,w13,w17+ rev32 v15.8h,v15.8h+ add w14,w14,w19+ rev32 v19.8h,v19.8h+ add w15,w15,w20+ rev32 v23.8h,v23.8h+ add w16,w16,w21+ rev32 v27.8h,v27.8h+ eor w9,w9,w13+ rev32 v31.8h,v31.8h+ eor w10,w10,w14+ add v10.4s,v10.4s,v11.4s+ eor w11,w11,w15+ add v14.4s,v14.4s,v15.4s+ eor w12,w12,w16+ add v18.4s,v18.4s,v19.4s+ ror w9,w9,#20+ add v22.4s,v22.4s,v23.4s+ ror w10,w10,#20+ add v26.4s,v26.4s,v27.4s+ ror w11,w11,#20+ add v30.4s,v30.4s,v31.4s+ ror w12,w12,#20+ eor v0.16b,v9.16b,v10.16b+ add w5,w5,w9+ eor v1.16b,v13.16b,v14.16b+ add w6,w6,w10+ eor v2.16b,v17.16b,v18.16b+ add w7,w7,w11+ eor v3.16b,v21.16b,v22.16b+ add w8,w8,w12+ eor v4.16b,v25.16b,v26.16b+ eor w17,w17,w5+ eor v5.16b,v29.16b,v30.16b+ eor w19,w19,w6+ ushr v9.4s,v0.4s,#20+ eor w20,w20,w7+ ushr v13.4s,v1.4s,#20+ eor w21,w21,w8+ ushr v17.4s,v2.4s,#20+ ror w17,w17,#24+ ushr v21.4s,v3.4s,#20+ ror w19,w19,#24+ ushr v25.4s,v4.4s,#20+ ror w20,w20,#24+ ushr v29.4s,v5.4s,#20+ ror w21,w21,#24+ sli v9.4s,v0.4s,#12+ add w13,w13,w17+ sli v13.4s,v1.4s,#12+ add w14,w14,w19+ sli v17.4s,v2.4s,#12+ add w15,w15,w20+ sli v21.4s,v3.4s,#12+ add w16,w16,w21+ sli v25.4s,v4.4s,#12+ eor w9,w9,w13+ sli v29.4s,v5.4s,#12+ eor w10,w10,w14+ add v8.4s,v8.4s,v9.4s+ eor w11,w11,w15+ add v12.4s,v12.4s,v13.4s+ eor w12,w12,w16+ add v16.4s,v16.4s,v17.4s+ ror w9,w9,#25+ add v20.4s,v20.4s,v21.4s+ ror w10,w10,#25+ add v24.4s,v24.4s,v25.4s+ ror w11,w11,#25+ add v28.4s,v28.4s,v29.4s+ ror w12,w12,#25+ eor v11.16b,v11.16b,v8.16b+ add w5,w5,w10+ eor v15.16b,v15.16b,v12.16b+ add w6,w6,w11+ eor v19.16b,v19.16b,v16.16b+ add w7,w7,w12+ eor v23.16b,v23.16b,v20.16b+ add w8,w8,w9+ eor v27.16b,v27.16b,v24.16b+ eor w21,w21,w5+ eor v31.16b,v31.16b,v28.16b+ eor w17,w17,w6+ tbl v11.16b,{v11.16b},v6.16b+ eor w19,w19,w7+ tbl v15.16b,{v15.16b},v6.16b+ eor w20,w20,w8+ tbl v19.16b,{v19.16b},v6.16b+ ror w21,w21,#16+ tbl v23.16b,{v23.16b},v6.16b+ ror w17,w17,#16+ tbl v27.16b,{v27.16b},v6.16b+ ror w19,w19,#16+ tbl v31.16b,{v31.16b},v6.16b+ ror w20,w20,#16+ add v10.4s,v10.4s,v11.4s+ add w15,w15,w21+ add v14.4s,v14.4s,v15.4s+ add w16,w16,w17+ add v18.4s,v18.4s,v19.4s+ add w13,w13,w19+ add v22.4s,v22.4s,v23.4s+ add w14,w14,w20+ add v26.4s,v26.4s,v27.4s+ eor w10,w10,w15+ add v30.4s,v30.4s,v31.4s+ eor w11,w11,w16+ eor v0.16b,v9.16b,v10.16b+ eor w12,w12,w13+ eor v1.16b,v13.16b,v14.16b+ eor w9,w9,w14+ eor v2.16b,v17.16b,v18.16b+ ror w10,w10,#20+ eor v3.16b,v21.16b,v22.16b+ ror w11,w11,#20+ eor v4.16b,v25.16b,v26.16b+ ror w12,w12,#20+ eor v5.16b,v29.16b,v30.16b+ ror w9,w9,#20+ ushr v9.4s,v0.4s,#25+ add w5,w5,w10+ ushr v13.4s,v1.4s,#25+ add w6,w6,w11+ ushr v17.4s,v2.4s,#25+ add w7,w7,w12+ ushr v21.4s,v3.4s,#25+ add w8,w8,w9+ ushr v25.4s,v4.4s,#25+ eor w21,w21,w5+ ushr v29.4s,v5.4s,#25+ eor w17,w17,w6+ sli v9.4s,v0.4s,#7+ eor w19,w19,w7+ sli v13.4s,v1.4s,#7+ eor w20,w20,w8+ sli v17.4s,v2.4s,#7+ ror w21,w21,#24+ sli v21.4s,v3.4s,#7+ ror w17,w17,#24+ sli v25.4s,v4.4s,#7+ ror w19,w19,#24+ sli v29.4s,v5.4s,#7+ ror w20,w20,#24+ ext v10.16b,v10.16b,v10.16b,#8+ add w15,w15,w21+ ext v14.16b,v14.16b,v14.16b,#8+ add w16,w16,w17+ ext v18.16b,v18.16b,v18.16b,#8+ add w13,w13,w19+ ext v22.16b,v22.16b,v22.16b,#8+ add w14,w14,w20+ ext v26.16b,v26.16b,v26.16b,#8+ eor w10,w10,w15+ ext v30.16b,v30.16b,v30.16b,#8+ eor w11,w11,w16+ ext v11.16b,v11.16b,v11.16b,#4+ eor w12,w12,w13+ ext v15.16b,v15.16b,v15.16b,#4+ eor w9,w9,w14+ ext v19.16b,v19.16b,v19.16b,#4+ ror w10,w10,#25+ ext v23.16b,v23.16b,v23.16b,#4+ ror w11,w11,#25+ ext v27.16b,v27.16b,v27.16b,#4+ ror w12,w12,#25+ ext v31.16b,v31.16b,v31.16b,#4+ ror w9,w9,#25+ ext v9.16b,v9.16b,v9.16b,#12+ ext v13.16b,v13.16b,v13.16b,#12+ ext v17.16b,v17.16b,v17.16b,#12+ ext v21.16b,v21.16b,v21.16b,#12+ ext v25.16b,v25.16b,v25.16b,#12+ ext v29.16b,v29.16b,v29.16b,#12+ cbnz x4,Loop_lower_neon++ add w5,w5,w22 // accumulate key block+ ldp q0,q1,[sp,#0]+ add x6,x6,x22,lsr#32+ ldp q2,q3,[sp,#32]+ add w7,w7,w23+ ldp q4,q5,[sp,#64]+ add x8,x8,x23,lsr#32+ ldr q6,[sp,#96]+ add v8.4s,v8.4s,v0.4s+ add w9,w9,w24+ add v12.4s,v12.4s,v0.4s+ add x10,x10,x24,lsr#32+ add v16.4s,v16.4s,v0.4s+ add w11,w11,w25+ add v20.4s,v20.4s,v0.4s+ add x12,x12,x25,lsr#32+ add v24.4s,v24.4s,v0.4s+ add w13,w13,w26+ add v28.4s,v28.4s,v0.4s+ add x14,x14,x26,lsr#32+ add v10.4s,v10.4s,v2.4s+ add w15,w15,w27+ add v14.4s,v14.4s,v2.4s+ add x16,x16,x27,lsr#32+ add v18.4s,v18.4s,v2.4s+ add w17,w17,w28+ add v22.4s,v22.4s,v2.4s+ add x19,x19,x28,lsr#32+ add v26.4s,v26.4s,v2.4s+ add w20,w20,w30+ add v30.4s,v30.4s,v2.4s+ add x21,x21,x30,lsr#32+ add v27.4s,v27.4s,v7.4s // +4+ add x5,x5,x6,lsl#32 // pack+ add v31.4s,v31.4s,v7.4s // +4+ add x7,x7,x8,lsl#32+ add v11.4s,v11.4s,v3.4s+ ldp x6,x8,[x1,#0] // load input+ add v15.4s,v15.4s,v4.4s+ add x9,x9,x10,lsl#32+ add v19.4s,v19.4s,v5.4s+ add x11,x11,x12,lsl#32+ add v23.4s,v23.4s,v6.4s+ ldp x10,x12,[x1,#16]+ add v27.4s,v27.4s,v3.4s+ add x13,x13,x14,lsl#32+ add v31.4s,v31.4s,v4.4s+ add x15,x15,x16,lsl#32+ add v9.4s,v9.4s,v1.4s+ ldp x14,x16,[x1,#32]+ add v13.4s,v13.4s,v1.4s+ add x17,x17,x19,lsl#32+ add v17.4s,v17.4s,v1.4s+ add x20,x20,x21,lsl#32+ add v21.4s,v21.4s,v1.4s+ ldp x19,x21,[x1,#48]+ add v25.4s,v25.4s,v1.4s+ add x1,x1,#64+ add v29.4s,v29.4s,v1.4s++#ifdef __AARCH64EB__+ rev x5,x5+ rev x7,x7+ rev x9,x9+ rev x11,x11+ rev x13,x13+ rev x15,x15+ rev x17,x17+ rev x20,x20+#endif+ ld1 {v0.16b,v1.16b,v2.16b,v3.16b},[x1],#64+ eor x5,x5,x6+ eor x7,x7,x8+ eor x9,x9,x10+ eor x11,x11,x12+ eor x13,x13,x14+ eor v8.16b,v8.16b,v0.16b+ eor x15,x15,x16+ eor v9.16b,v9.16b,v1.16b+ eor x17,x17,x19+ eor v10.16b,v10.16b,v2.16b+ eor x20,x20,x21+ eor v11.16b,v11.16b,v3.16b+ ld1 {v0.16b,v1.16b,v2.16b,v3.16b},[x1],#64++ stp x5,x7,[x0,#0] // store output+ add x28,x28,#7 // increment counter+ stp x9,x11,[x0,#16]+ stp x13,x15,[x0,#32]+ stp x17,x20,[x0,#48]+ add x0,x0,#64+ st1 {v8.16b,v9.16b,v10.16b,v11.16b},[x0],#64++ ld1 {v8.16b,v9.16b,v10.16b,v11.16b},[x1],#64+ eor v12.16b,v12.16b,v0.16b+ eor v13.16b,v13.16b,v1.16b+ eor v14.16b,v14.16b,v2.16b+ eor v15.16b,v15.16b,v3.16b+ st1 {v12.16b,v13.16b,v14.16b,v15.16b},[x0],#64++ ld1 {v12.16b,v13.16b,v14.16b,v15.16b},[x1],#64+ eor v16.16b,v16.16b,v8.16b+ ldp q0,q1,[sp,#0]+ eor v17.16b,v17.16b,v9.16b+ ldp q2,q3,[sp,#32]+ eor v18.16b,v18.16b,v10.16b+ eor v19.16b,v19.16b,v11.16b+ st1 {v16.16b,v17.16b,v18.16b,v19.16b},[x0],#64++ ld1 {v16.16b,v17.16b,v18.16b,v19.16b},[x1],#64+ eor v20.16b,v20.16b,v12.16b+ eor v21.16b,v21.16b,v13.16b+ eor v22.16b,v22.16b,v14.16b+ eor v23.16b,v23.16b,v15.16b+ st1 {v20.16b,v21.16b,v22.16b,v23.16b},[x0],#64++ ld1 {v20.16b,v21.16b,v22.16b,v23.16b},[x1],#64+ eor v24.16b,v24.16b,v16.16b+ eor v25.16b,v25.16b,v17.16b+ eor v26.16b,v26.16b,v18.16b+ eor v27.16b,v27.16b,v19.16b+ st1 {v24.16b,v25.16b,v26.16b,v27.16b},[x0],#64++ shl v8.4s,v7.4s,#1 // 4 -> 8+ eor v28.16b,v28.16b,v20.16b+ eor v29.16b,v29.16b,v21.16b+ eor v30.16b,v30.16b,v22.16b+ eor v31.16b,v31.16b,v23.16b+ st1 {v28.16b,v29.16b,v30.16b,v31.16b},[x0],#64++ add v3.4s,v3.4s,v8.4s // += 8+ add v4.4s,v4.4s,v8.4s+ add v5.4s,v5.4s,v8.4s+ add v6.4s,v6.4s,v8.4s++ b.hs Loop_outer_512_neon++ adds x2,x2,#512+ ushr v7.4s,v7.4s,#1 // 4 -> 2++ ldp d10,d11,[sp,#128+16] // meet ABI requirements+ ldp d12,d13,[sp,#128+32]+ ldp d14,d15,[sp,#128+48]++ stp q0,q0,[sp,#16] // wipe key off-load area+ stp q0,q0,[sp,#48] // [with known constant]+ stp q0,q0,[sp,#80]++ b.eq Ldone_512_neon++ // we have <512 bytes tail, harmonize state with other contexts+ sub x3,x3,#16+ cmp x2,#192+ add sp,sp,#128+ sub v3.4s,v3.4s,v7.4s // -= 2+ ld1 {v8.4s,v9.4s},[x3]+ b.hs Loop_outer_neon++ ldp d8,d9,[sp,#0] // meet ABI requirements+ eor v1.16b,v1.16b,v1.16b // cleanse key and nonce+ eor v2.16b,v2.16b,v2.16b+ eor v3.16b,v3.16b,v3.16b+ eor v4.16b,v4.16b,v4.16b+ eor v5.16b,v5.16b,v5.16b+ eor v6.16b,v6.16b,v6.16b+ b Loop_outer++Ldone_512_neon:+ ldp d8,d9,[sp,#128+0] // meet ABI requirements+ eor v1.16b,v1.16b,v1.16b // cleanse key and nonce+ eor v2.16b,v2.16b,v2.16b+ eor v3.16b,v3.16b,v3.16b+ eor v4.16b,v4.16b,v4.16b+ eor v5.16b,v5.16b,v5.16b+ eor v6.16b,v6.16b,v6.16b++ ldp x19,x20,[x29,#2*__SIZEOF_POINTER__]+ add sp,sp,#128+64+ ldp x21,x22,[x29,#4*__SIZEOF_POINTER__]+ ldp x23,x24,[x29,#6*__SIZEOF_POINTER__]+ ldp x25,x26,[x29,#8*__SIZEOF_POINTER__]+ ldp x27,x28,[x29,#10*__SIZEOF_POINTER__]+ ldp x29,x30,[sp],#12*__SIZEOF_POINTER__+.long 0xd50323bf // autiasp+ ret+
@@ -0,0 +1,2055 @@+#ifndef __KERNEL__+# include "arm_arch.h"++#endif++.text++.align 5+.Lsigma:+.quad 0x3320646e61707865,0x6b20657479622d32 // endian-neutral+.Lone:+.long 1,2,3,4+.Lrot24:+.long 0x02010003,0x06050407,0x0a09080b,0x0e0d0c0f+.byte 67,104,97,67,104,97,50,48,32,102,111,114,32,65,82,77,118,56,44,32,67,82,89,80,84,79,71,65,77,83,32,98,121,32,64,100,111,116,45,97,115,109,0+.align 2++.globl crypton_chacha20_asm_ctr32+.type crypton_chacha20_asm_ctr32,%function+.align 5+crypton_chacha20_asm_ctr32:+ cbz x2,.Labort+ cmp x2,#192+ b.lo .Lshort++#ifndef __KERNEL__+ adrp x17,crypton_armcap_P+ ldr w17,[x17,#:lo12:crypton_armcap_P]+ tst w17,#ARMV7_NEON+ b.ne .Lcrypton_chacha20_asm_neon+#endif++.Lshort:+.inst 0xd503233f // paciasp+ stp x29,x30,[sp,#-12*__SIZEOF_POINTER__]!+ add x29,sp,#0++ adr x5,.Lsigma+ stp x19,x20,[sp,#2*__SIZEOF_POINTER__]+ stp x21,x22,[sp,#4*__SIZEOF_POINTER__]+ stp x23,x24,[sp,#6*__SIZEOF_POINTER__]+ stp x25,x26,[sp,#8*__SIZEOF_POINTER__]+ stp x27,x28,[sp,#10*__SIZEOF_POINTER__]+ sub sp,sp,#64++ ldp x22,x23,[x5] // load sigma+ ldp x24,x25,[x3] // load key+ ldp x26,x27,[x3,#16]+ ldp x28,x30,[x4] // load counter+#ifdef __AARCH64EB__+ ror x24,x24,#32+ ror x25,x25,#32+ ror x26,x26,#32+ ror x27,x27,#32+ ror x28,x28,#32+ ror x30,x30,#32+#endif++.Loop_outer:+ mov w5,w22 // unpack key block+ lsr x6,x22,#32+ mov w7,w23+ lsr x8,x23,#32+ mov w9,w24+ lsr x10,x24,#32+ mov w11,w25+ lsr x12,x25,#32+ mov w13,w26+ lsr x14,x26,#32+ mov w15,w27+ lsr x16,x27,#32+ mov w17,w28+ lsr x19,x28,#32+ mov w20,w30+ lsr x21,x30,#32++ mov x4,#10+ subs x2,x2,#64+.Loop:+ sub x4,x4,#1+ add w5,w5,w9+ add w6,w6,w10+ add w7,w7,w11+ add w8,w8,w12+ eor w17,w17,w5+ eor w19,w19,w6+ eor w20,w20,w7+ eor w21,w21,w8+ ror w17,w17,#16+ ror w19,w19,#16+ ror w20,w20,#16+ ror w21,w21,#16+ add w13,w13,w17+ add w14,w14,w19+ add w15,w15,w20+ add w16,w16,w21+ eor w9,w9,w13+ eor w10,w10,w14+ eor w11,w11,w15+ eor w12,w12,w16+ ror w9,w9,#20+ ror w10,w10,#20+ ror w11,w11,#20+ ror w12,w12,#20+ add w5,w5,w9+ add w6,w6,w10+ add w7,w7,w11+ add w8,w8,w12+ eor w17,w17,w5+ eor w19,w19,w6+ eor w20,w20,w7+ eor w21,w21,w8+ ror w17,w17,#24+ ror w19,w19,#24+ ror w20,w20,#24+ ror w21,w21,#24+ add w13,w13,w17+ add w14,w14,w19+ add w15,w15,w20+ add w16,w16,w21+ eor w9,w9,w13+ eor w10,w10,w14+ eor w11,w11,w15+ eor w12,w12,w16+ ror w9,w9,#25+ ror w10,w10,#25+ ror w11,w11,#25+ ror w12,w12,#25+ add w5,w5,w10+ add w6,w6,w11+ add w7,w7,w12+ add w8,w8,w9+ eor w21,w21,w5+ eor w17,w17,w6+ eor w19,w19,w7+ eor w20,w20,w8+ ror w21,w21,#16+ ror w17,w17,#16+ ror w19,w19,#16+ ror w20,w20,#16+ add w15,w15,w21+ add w16,w16,w17+ add w13,w13,w19+ add w14,w14,w20+ eor w10,w10,w15+ eor w11,w11,w16+ eor w12,w12,w13+ eor w9,w9,w14+ ror w10,w10,#20+ ror w11,w11,#20+ ror w12,w12,#20+ ror w9,w9,#20+ add w5,w5,w10+ add w6,w6,w11+ add w7,w7,w12+ add w8,w8,w9+ eor w21,w21,w5+ eor w17,w17,w6+ eor w19,w19,w7+ eor w20,w20,w8+ ror w21,w21,#24+ ror w17,w17,#24+ ror w19,w19,#24+ ror w20,w20,#24+ add w15,w15,w21+ add w16,w16,w17+ add w13,w13,w19+ add w14,w14,w20+ eor w10,w10,w15+ eor w11,w11,w16+ eor w12,w12,w13+ eor w9,w9,w14+ ror w10,w10,#25+ ror w11,w11,#25+ ror w12,w12,#25+ ror w9,w9,#25+ cbnz x4,.Loop++ add w5,w5,w22 // accumulate key block+ add x6,x6,x22,lsr#32+ add w7,w7,w23+ add x8,x8,x23,lsr#32+ add w9,w9,w24+ add x10,x10,x24,lsr#32+ add w11,w11,w25+ add x12,x12,x25,lsr#32+ add w13,w13,w26+ add x14,x14,x26,lsr#32+ add w15,w15,w27+ add x16,x16,x27,lsr#32+ add w17,w17,w28+ add x19,x19,x28,lsr#32+ add w20,w20,w30+ add x21,x21,x30,lsr#32++ b.lo .Ltail++ add x5,x5,x6,lsl#32 // pack+ add x7,x7,x8,lsl#32+ ldp x6,x8,[x1,#0] // load input+ add x9,x9,x10,lsl#32+ add x11,x11,x12,lsl#32+ ldp x10,x12,[x1,#16]+ add x13,x13,x14,lsl#32+ add x15,x15,x16,lsl#32+ ldp x14,x16,[x1,#32]+ add x17,x17,x19,lsl#32+ add x20,x20,x21,lsl#32+ ldp x19,x21,[x1,#48]+ add x1,x1,#64+#ifdef __AARCH64EB__+ rev x5,x5+ rev x7,x7+ rev x9,x9+ rev x11,x11+ rev x13,x13+ rev x15,x15+ rev x17,x17+ rev x20,x20+#endif+ eor x5,x5,x6+ eor x7,x7,x8+ eor x9,x9,x10+ eor x11,x11,x12+ eor x13,x13,x14+ eor x15,x15,x16+ eor x17,x17,x19+ eor x20,x20,x21++ stp x5,x7,[x0,#0] // store output+ add x28,x28,#1 // increment counter+ stp x9,x11,[x0,#16]+ stp x13,x15,[x0,#32]+ stp x17,x20,[x0,#48]+ add x0,x0,#64++ b.hi .Loop_outer++ ldp x19,x20,[x29,#2*__SIZEOF_POINTER__]+ add sp,sp,#64+ ldp x21,x22,[x29,#4*__SIZEOF_POINTER__]+ ldp x23,x24,[x29,#6*__SIZEOF_POINTER__]+ ldp x25,x26,[x29,#8*__SIZEOF_POINTER__]+ ldp x27,x28,[x29,#10*__SIZEOF_POINTER__]+ ldp x29,x30,[sp],#12*__SIZEOF_POINTER__+.inst 0xd50323bf // autiasp+.Labort:+ ret++.align 4+.Ltail:+ add x2,x2,#64+.Less_than_64:+ sub x0,x0,#1+ add x1,x1,x2+ add x0,x0,x2+ add x4,sp,x2+ neg x2,x2++ add x5,x5,x6,lsl#32 // pack+ add x7,x7,x8,lsl#32+ add x9,x9,x10,lsl#32+ add x11,x11,x12,lsl#32+ add x13,x13,x14,lsl#32+ add x15,x15,x16,lsl#32+ add x17,x17,x19,lsl#32+ add x20,x20,x21,lsl#32+#ifdef __AARCH64EB__+ rev x5,x5+ rev x7,x7+ rev x9,x9+ rev x11,x11+ rev x13,x13+ rev x15,x15+ rev x17,x17+ rev x20,x20+#endif+ stp x5,x7,[sp,#0] // off-load complete block+ stp x9,x11,[sp,#16]+ stp x13,x15,[sp,#32]+ stp x17,x20,[sp,#48]++.Loop_tail:+ ldrb w10,[x1,x2]+ ldrb w11,[x4,x2]+ add x2,x2,#1+ eor w10,w10,w11+ strb w10,[x0,x2]+ cbnz x2,.Loop_tail++ stp xzr,xzr,[sp,#0] // wipe off-load area+ stp xzr,xzr,[sp,#16]+ stp xzr,xzr,[sp,#32]+ stp xzr,xzr,[sp,#48]++ ldp x19,x20,[x29,#2*__SIZEOF_POINTER__]+ add sp,sp,#64+ ldp x21,x22,[x29,#4*__SIZEOF_POINTER__]+ ldp x23,x24,[x29,#6*__SIZEOF_POINTER__]+ ldp x25,x26,[x29,#8*__SIZEOF_POINTER__]+ ldp x27,x28,[x29,#10*__SIZEOF_POINTER__]+ ldp x29,x30,[sp],#12*__SIZEOF_POINTER__+.inst 0xd50323bf // autiasp+ ret+.size crypton_chacha20_asm_ctr32,.-crypton_chacha20_asm_ctr32++#ifdef __KERNEL__+.globl crypton_chacha20_asm_neon+#endif+.type crypton_chacha20_asm_neon,%function+.align 5+crypton_chacha20_asm_neon:+.Lcrypton_chacha20_asm_neon:+.inst 0xd503233f // paciasp+ stp x29,x30,[sp,#-12*__SIZEOF_POINTER__]!+ add x29,sp,#0++ adr x5,.Lsigma+ stp x19,x20,[sp,#2*__SIZEOF_POINTER__]+ stp x21,x22,[sp,#4*__SIZEOF_POINTER__]+ stp x23,x24,[sp,#6*__SIZEOF_POINTER__]+ stp x25,x26,[sp,#8*__SIZEOF_POINTER__]+ stp x27,x28,[sp,#10*__SIZEOF_POINTER__]+ cmp x2,#512+ b.hs .L512_or_more_neon++ sub sp,sp,#64++ ldp x22,x23,[x5] // load sigma+ ld1 {v0.4s},[x5],#16+ ldp x24,x25,[x3] // load key+ ldp x26,x27,[x3,#16]+ ld1 {v1.4s,v2.4s},[x3]+ ldp x28,x30,[x4] // load counter+ ld1 {v3.4s},[x4]+ stp d8,d9,[sp] // meet ABI requirements+ ld1 {v8.4s,v9.4s},[x5]+#ifdef __AARCH64EB__+ rev64 v0.4s,v0.4s+ ror x24,x24,#32+ ror x25,x25,#32+ ror x26,x26,#32+ ror x27,x27,#32+ ror x28,x28,#32+ ror x30,x30,#32+#endif++.Loop_outer_neon:+ dup v16.4s,v0.s[0] // unpack key block+ mov w5,w22+ dup v20.4s,v0.s[1]+ lsr x6,x22,#32+ dup v24.4s,v0.s[2]+ mov w7,w23+ dup v28.4s,v0.s[3]+ lsr x8,x23,#32+ dup v17.4s,v1.s[0]+ mov w9,w24+ dup v21.4s,v1.s[1]+ lsr x10,x24,#32+ dup v25.4s,v1.s[2]+ mov w11,w25+ dup v29.4s,v1.s[3]+ lsr x12,x25,#32+ dup v19.4s,v3.s[0]+ mov w13,w26+ dup v23.4s,v3.s[1]+ lsr x14,x26,#32+ dup v27.4s,v3.s[2]+ mov w15,w27+ dup v31.4s,v3.s[3]+ lsr x16,x27,#32+ add v19.4s,v19.4s,v8.4s+ mov w17,w28+ dup v18.4s,v2.s[0]+ lsr x19,x28,#32+ dup v22.4s,v2.s[1]+ mov w20,w30+ dup v26.4s,v2.s[2]+ lsr x21,x30,#32+ dup v30.4s,v2.s[3]++ mov x4,#10+ subs x2,x2,#320+.Loop_neon:+ sub x4,x4,#1+ add v16.4s,v16.4s,v17.4s+ add v20.4s,v20.4s,v21.4s+ add v24.4s,v24.4s,v25.4s+ add v28.4s,v28.4s,v29.4s+ eor v19.16b,v19.16b,v16.16b+ eor v23.16b,v23.16b,v20.16b+ eor v27.16b,v27.16b,v24.16b+ eor v31.16b,v31.16b,v28.16b+ add w5,w5,w9+ rev32 v19.8h,v19.8h+ add w6,w6,w10+ rev32 v23.8h,v23.8h+ add w7,w7,w11+ rev32 v27.8h,v27.8h+ add w8,w8,w12+ rev32 v31.8h,v31.8h+ eor w17,w17,w5+ add v18.4s,v18.4s,v19.4s+ eor w19,w19,w6+ add v22.4s,v22.4s,v23.4s+ eor w20,w20,w7+ add v26.4s,v26.4s,v27.4s+ eor w21,w21,w8+ add v30.4s,v30.4s,v31.4s+ ror w17,w17,#16+ eor v4.16b,v17.16b,v18.16b+ ror w19,w19,#16+ eor v5.16b,v21.16b,v22.16b+ ror w20,w20,#16+ eor v6.16b,v25.16b,v26.16b+ ror w21,w21,#16+ eor v7.16b,v29.16b,v30.16b+ add w13,w13,w17+ ushr v17.4s,v4.4s,#20+ add w14,w14,w19+ ushr v21.4s,v5.4s,#20+ add w15,w15,w20+ ushr v25.4s,v6.4s,#20+ add w16,w16,w21+ ushr v29.4s,v7.4s,#20+ eor w9,w9,w13+ sli v17.4s,v4.4s,#12+ eor w10,w10,w14+ sli v21.4s,v5.4s,#12+ eor w11,w11,w15+ sli v25.4s,v6.4s,#12+ eor w12,w12,w16+ sli v29.4s,v7.4s,#12+ ror w9,w9,#20+ add v16.4s,v16.4s,v17.4s+ ror w10,w10,#20+ add v20.4s,v20.4s,v21.4s+ ror w11,w11,#20+ add v24.4s,v24.4s,v25.4s+ ror w12,w12,#20+ add v28.4s,v28.4s,v29.4s+ add w5,w5,w9+ eor v4.16b,v19.16b,v16.16b+ add w6,w6,w10+ eor v5.16b,v23.16b,v20.16b+ add w7,w7,w11+ eor v6.16b,v27.16b,v24.16b+ add w8,w8,w12+ eor v7.16b,v31.16b,v28.16b+ eor w17,w17,w5+ tbl v19.16b,{v4.16b},v9.16b+ eor w19,w19,w6+ tbl v23.16b,{v5.16b},v9.16b+ eor w20,w20,w7+ tbl v27.16b,{v6.16b},v9.16b+ eor w21,w21,w8+ tbl v31.16b,{v7.16b},v9.16b+ ror w17,w17,#24+ add v18.4s,v18.4s,v19.4s+ ror w19,w19,#24+ add v22.4s,v22.4s,v23.4s+ ror w20,w20,#24+ add v26.4s,v26.4s,v27.4s+ ror w21,w21,#24+ add v30.4s,v30.4s,v31.4s+ add w13,w13,w17+ eor v4.16b,v17.16b,v18.16b+ add w14,w14,w19+ eor v5.16b,v21.16b,v22.16b+ add w15,w15,w20+ eor v6.16b,v25.16b,v26.16b+ add w16,w16,w21+ eor v7.16b,v29.16b,v30.16b+ eor w9,w9,w13+ ushr v17.4s,v4.4s,#25+ eor w10,w10,w14+ ushr v21.4s,v5.4s,#25+ eor w11,w11,w15+ ushr v25.4s,v6.4s,#25+ eor w12,w12,w16+ ushr v29.4s,v7.4s,#25+ ror w9,w9,#25+ sli v17.4s,v4.4s,#7+ ror w10,w10,#25+ sli v21.4s,v5.4s,#7+ ror w11,w11,#25+ sli v25.4s,v6.4s,#7+ ror w12,w12,#25+ sli v29.4s,v7.4s,#7+ add v16.4s,v16.4s,v21.4s+ add v20.4s,v20.4s,v25.4s+ add v24.4s,v24.4s,v29.4s+ add v28.4s,v28.4s,v17.4s+ eor v31.16b,v31.16b,v16.16b+ eor v19.16b,v19.16b,v20.16b+ eor v23.16b,v23.16b,v24.16b+ eor v27.16b,v27.16b,v28.16b+ add w5,w5,w10+ rev32 v31.8h,v31.8h+ add w6,w6,w11+ rev32 v19.8h,v19.8h+ add w7,w7,w12+ rev32 v23.8h,v23.8h+ add w8,w8,w9+ rev32 v27.8h,v27.8h+ eor w21,w21,w5+ add v26.4s,v26.4s,v31.4s+ eor w17,w17,w6+ add v30.4s,v30.4s,v19.4s+ eor w19,w19,w7+ add v18.4s,v18.4s,v23.4s+ eor w20,w20,w8+ add v22.4s,v22.4s,v27.4s+ ror w21,w21,#16+ eor v4.16b,v21.16b,v26.16b+ ror w17,w17,#16+ eor v5.16b,v25.16b,v30.16b+ ror w19,w19,#16+ eor v6.16b,v29.16b,v18.16b+ ror w20,w20,#16+ eor v7.16b,v17.16b,v22.16b+ add w15,w15,w21+ ushr v21.4s,v4.4s,#20+ add w16,w16,w17+ ushr v25.4s,v5.4s,#20+ add w13,w13,w19+ ushr v29.4s,v6.4s,#20+ add w14,w14,w20+ ushr v17.4s,v7.4s,#20+ eor w10,w10,w15+ sli v21.4s,v4.4s,#12+ eor w11,w11,w16+ sli v25.4s,v5.4s,#12+ eor w12,w12,w13+ sli v29.4s,v6.4s,#12+ eor w9,w9,w14+ sli v17.4s,v7.4s,#12+ ror w10,w10,#20+ add v16.4s,v16.4s,v21.4s+ ror w11,w11,#20+ add v20.4s,v20.4s,v25.4s+ ror w12,w12,#20+ add v24.4s,v24.4s,v29.4s+ ror w9,w9,#20+ add v28.4s,v28.4s,v17.4s+ add w5,w5,w10+ eor v4.16b,v31.16b,v16.16b+ add w6,w6,w11+ eor v5.16b,v19.16b,v20.16b+ add w7,w7,w12+ eor v6.16b,v23.16b,v24.16b+ add w8,w8,w9+ eor v7.16b,v27.16b,v28.16b+ eor w21,w21,w5+ tbl v31.16b,{v4.16b},v9.16b+ eor w17,w17,w6+ tbl v19.16b,{v5.16b},v9.16b+ eor w19,w19,w7+ tbl v23.16b,{v6.16b},v9.16b+ eor w20,w20,w8+ tbl v27.16b,{v7.16b},v9.16b+ ror w21,w21,#24+ add v26.4s,v26.4s,v31.4s+ ror w17,w17,#24+ add v30.4s,v30.4s,v19.4s+ ror w19,w19,#24+ add v18.4s,v18.4s,v23.4s+ ror w20,w20,#24+ add v22.4s,v22.4s,v27.4s+ add w15,w15,w21+ eor v4.16b,v21.16b,v26.16b+ add w16,w16,w17+ eor v5.16b,v25.16b,v30.16b+ add w13,w13,w19+ eor v6.16b,v29.16b,v18.16b+ add w14,w14,w20+ eor v7.16b,v17.16b,v22.16b+ eor w10,w10,w15+ ushr v21.4s,v4.4s,#25+ eor w11,w11,w16+ ushr v25.4s,v5.4s,#25+ eor w12,w12,w13+ ushr v29.4s,v6.4s,#25+ eor w9,w9,w14+ ushr v17.4s,v7.4s,#25+ ror w10,w10,#25+ sli v21.4s,v4.4s,#7+ ror w11,w11,#25+ sli v25.4s,v5.4s,#7+ ror w12,w12,#25+ sli v29.4s,v6.4s,#7+ ror w9,w9,#25+ sli v17.4s,v7.4s,#7+ cbnz x4,.Loop_neon++ add v19.4s,v19.4s,v8.4s++ zip1 v4.4s,v16.4s,v20.4s // transpose data+ zip1 v5.4s,v24.4s,v28.4s+ zip2 v6.4s,v16.4s,v20.4s+ zip2 v7.4s,v24.4s,v28.4s+ zip1 v16.2d,v4.2d,v5.2d+ zip2 v20.2d,v4.2d,v5.2d+ zip1 v24.2d,v6.2d,v7.2d+ zip2 v28.2d,v6.2d,v7.2d++ zip1 v4.4s,v17.4s,v21.4s+ zip1 v5.4s,v25.4s,v29.4s+ zip2 v6.4s,v17.4s,v21.4s+ zip2 v7.4s,v25.4s,v29.4s+ zip1 v17.2d,v4.2d,v5.2d+ zip2 v21.2d,v4.2d,v5.2d+ zip1 v25.2d,v6.2d,v7.2d+ zip2 v29.2d,v6.2d,v7.2d++ zip1 v4.4s,v18.4s,v22.4s+ add w5,w5,w22 // accumulate key block+ zip1 v5.4s,v26.4s,v30.4s+ add x6,x6,x22,lsr#32+ zip2 v6.4s,v18.4s,v22.4s+ add w7,w7,w23+ zip2 v7.4s,v26.4s,v30.4s+ add x8,x8,x23,lsr#32+ zip1 v18.2d,v4.2d,v5.2d+ add w9,w9,w24+ zip2 v22.2d,v4.2d,v5.2d+ add x10,x10,x24,lsr#32+ zip1 v26.2d,v6.2d,v7.2d+ add w11,w11,w25+ zip2 v30.2d,v6.2d,v7.2d+ add x12,x12,x25,lsr#32++ zip1 v4.4s,v19.4s,v23.4s+ add w13,w13,w26+ zip1 v5.4s,v27.4s,v31.4s+ add x14,x14,x26,lsr#32+ zip2 v6.4s,v19.4s,v23.4s+ add w15,w15,w27+ zip2 v7.4s,v27.4s,v31.4s+ add x16,x16,x27,lsr#32+ zip1 v19.2d,v4.2d,v5.2d+ add w17,w17,w28+ zip2 v23.2d,v4.2d,v5.2d+ add x19,x19,x28,lsr#32+ zip1 v27.2d,v6.2d,v7.2d+ add w20,w20,w30+ zip2 v31.2d,v6.2d,v7.2d+ add x21,x21,x30,lsr#32++ b.lo .Ltail_neon++ add x5,x5,x6,lsl#32 // pack+ add x7,x7,x8,lsl#32+ ldp x6,x8,[x1,#0] // load input+ add v16.4s,v16.4s,v0.4s // accumulate key block+ add x9,x9,x10,lsl#32+ add x11,x11,x12,lsl#32+ ldp x10,x12,[x1,#16]+ add v17.4s,v17.4s,v1.4s+ add x13,x13,x14,lsl#32+ add x15,x15,x16,lsl#32+ ldp x14,x16,[x1,#32]+ add v18.4s,v18.4s,v2.4s+ add x17,x17,x19,lsl#32+ add x20,x20,x21,lsl#32+ ldp x19,x21,[x1,#48]+ add v19.4s,v19.4s,v3.4s+ add x1,x1,#64+#ifdef __AARCH64EB__+ rev x5,x5+ rev x7,x7+ rev x9,x9+ rev x11,x11+ rev x13,x13+ rev x15,x15+ rev x17,x17+ rev x20,x20+#endif+ ld1 {v4.16b,v5.16b,v6.16b,v7.16b},[x1],#64+ eor x5,x5,x6+ add v20.4s,v20.4s,v0.4s+ eor x7,x7,x8+ add v21.4s,v21.4s,v1.4s+ eor x9,x9,x10+ add v22.4s,v22.4s,v2.4s+ eor x11,x11,x12+ add v23.4s,v23.4s,v3.4s+ eor x13,x13,x14+ eor v16.16b,v16.16b,v4.16b+ movi v4.4s,#5+ eor x15,x15,x16+ eor v17.16b,v17.16b,v5.16b+ eor x17,x17,x19+ eor v18.16b,v18.16b,v6.16b+ eor x20,x20,x21+ eor v19.16b,v19.16b,v7.16b+ add v8.4s,v8.4s,v4.4s // += 5+ ld1 {v4.16b,v5.16b,v6.16b,v7.16b},[x1],#64++ stp x5,x7,[x0,#0] // store output+ add x28,x28,#5 // increment counter+ stp x9,x11,[x0,#16]+ stp x13,x15,[x0,#32]+ stp x17,x20,[x0,#48]+ add x0,x0,#64++ st1 {v16.16b,v17.16b,v18.16b,v19.16b},[x0],#64+ add v24.4s,v24.4s,v0.4s+ add v25.4s,v25.4s,v1.4s+ add v26.4s,v26.4s,v2.4s+ add v27.4s,v27.4s,v3.4s+ ld1 {v16.16b,v17.16b,v18.16b,v19.16b},[x1],#64++ eor v20.16b,v20.16b,v4.16b+ eor v21.16b,v21.16b,v5.16b+ eor v22.16b,v22.16b,v6.16b+ eor v23.16b,v23.16b,v7.16b+ st1 {v20.16b,v21.16b,v22.16b,v23.16b},[x0],#64+ add v28.4s,v28.4s,v0.4s+ add v29.4s,v29.4s,v1.4s+ add v30.4s,v30.4s,v2.4s+ add v31.4s,v31.4s,v3.4s+ ld1 {v20.16b,v21.16b,v22.16b,v23.16b},[x1],#64++ eor v24.16b,v24.16b,v16.16b+ eor v25.16b,v25.16b,v17.16b+ eor v26.16b,v26.16b,v18.16b+ eor v27.16b,v27.16b,v19.16b+ st1 {v24.16b,v25.16b,v26.16b,v27.16b},[x0],#64++ eor v28.16b,v28.16b,v20.16b+ eor v29.16b,v29.16b,v21.16b+ eor v30.16b,v30.16b,v22.16b+ eor v31.16b,v31.16b,v23.16b+ st1 {v28.16b,v29.16b,v30.16b,v31.16b},[x0],#64++ b.hi .Loop_outer_neon++ ldp d8,d9,[sp] // meet ABI requirements+ eor v1.16b,v1.16b,v1.16b // cleanse key and nonce+ eor v2.16b,v2.16b,v2.16b+ eor v3.16b,v3.16b,v3.16b++ ldp x19,x20,[x29,#2*__SIZEOF_POINTER__]+ add sp,sp,#64+ ldp x21,x22,[x29,#4*__SIZEOF_POINTER__]+ ldp x23,x24,[x29,#6*__SIZEOF_POINTER__]+ ldp x25,x26,[x29,#8*__SIZEOF_POINTER__]+ ldp x27,x28,[x29,#10*__SIZEOF_POINTER__]+ ldp x29,x30,[sp],#12*__SIZEOF_POINTER__+.inst 0xd50323bf // autiasp+ ret++.align 4+.Ltail_neon:+ add x2,x2,#320+ ldp d8,d9,[sp] // meet ABI requirements+ cmp x2,#64+ b.lo .Less_than_64_neon++ add x5,x5,x6,lsl#32 // pack+ add x7,x7,x8,lsl#32+ ldp x6,x8,[x1,#0] // load input+ add x9,x9,x10,lsl#32+ add x11,x11,x12,lsl#32+ ldp x10,x12,[x1,#16]+ add x13,x13,x14,lsl#32+ add x15,x15,x16,lsl#32+ ldp x14,x16,[x1,#32]+ add x17,x17,x19,lsl#32+ add x20,x20,x21,lsl#32+ ldp x19,x21,[x1,#48]+ add x1,x1,#64+#ifdef __AARCH64EB__+ rev x5,x5+ rev x7,x7+ rev x9,x9+ rev x11,x11+ rev x13,x13+ rev x15,x15+ rev x17,x17+ rev x20,x20+#endif+ eor x5,x5,x6+ eor x7,x7,x8+ eor x9,x9,x10+ eor x11,x11,x12+ eor x13,x13,x14+ eor x15,x15,x16+ eor x17,x17,x19+ eor x20,x20,x21++ stp x5,x7,[x0,#0] // store output+ add v16.4s,v16.4s,v0.4s // accumulate key block+ stp x9,x11,[x0,#16]+ add v17.4s,v17.4s,v1.4s+ stp x13,x15,[x0,#32]+ add v18.4s,v18.4s,v2.4s+ stp x17,x20,[x0,#48]+ add v19.4s,v19.4s,v3.4s+ add x0,x0,#64+ b.eq .Ldone_neon+ sub x2,x2,#64+ cmp x2,#64+ b.lo .Last_neon++ ld1 {v4.16b,v5.16b,v6.16b,v7.16b},[x1],#64+ eor v16.16b,v16.16b,v4.16b+ eor v17.16b,v17.16b,v5.16b+ eor v18.16b,v18.16b,v6.16b+ eor v19.16b,v19.16b,v7.16b+ st1 {v16.16b,v17.16b,v18.16b,v19.16b},[x0],#64+ b.eq .Ldone_neon++ add v16.4s,v20.4s,v0.4s+ add v17.4s,v21.4s,v1.4s+ sub x2,x2,#64+ add v18.4s,v22.4s,v2.4s+ cmp x2,#64+ add v19.4s,v23.4s,v3.4s+ b.lo .Last_neon++ ld1 {v4.16b,v5.16b,v6.16b,v7.16b},[x1],#64+ eor v20.16b,v16.16b,v4.16b+ eor v21.16b,v17.16b,v5.16b+ eor v22.16b,v18.16b,v6.16b+ eor v23.16b,v19.16b,v7.16b+ st1 {v20.16b,v21.16b,v22.16b,v23.16b},[x0],#64+ b.eq .Ldone_neon++ add v16.4s,v24.4s,v0.4s+ add v17.4s,v25.4s,v1.4s+ sub x2,x2,#64+ add v18.4s,v26.4s,v2.4s+ cmp x2,#64+ add v19.4s,v27.4s,v3.4s+ b.lo .Last_neon++ ld1 {v4.16b,v5.16b,v6.16b,v7.16b},[x1],#64+ eor v24.16b,v16.16b,v4.16b+ eor v25.16b,v17.16b,v5.16b+ eor v26.16b,v18.16b,v6.16b+ eor v27.16b,v19.16b,v7.16b+ st1 {v24.16b,v25.16b,v26.16b,v27.16b},[x0],#64+ b.eq .Ldone_neon++ add v16.4s,v28.4s,v0.4s+ add v17.4s,v29.4s,v1.4s+ add v18.4s,v30.4s,v2.4s+ add v19.4s,v31.4s,v3.4s+ sub x2,x2,#64++.Last_neon:+ st1 {v16.16b,v17.16b,v18.16b,v19.16b},[sp] // off-load complete block++ sub x0,x0,#1+ add x1,x1,x2+ add x0,x0,x2+ add x4,sp,x2+ neg x2,x2++.Loop_tail_neon:+ ldrb w10,[x1,x2]+ ldrb w11,[x4,x2]+ add x2,x2,#1+ eor w10,w10,w11+ strb w10,[x0,x2]+ cbnz x2,.Loop_tail_neon++ stp q0,q0,[sp,#0] // wipe off-load area+ stp q0,q0,[sp,#32] // [with known constant]++.Ldone_neon:+ eor v1.16b,v1.16b,v1.16b // cleanse key and nonce+ eor v2.16b,v2.16b,v2.16b+ eor v3.16b,v3.16b,v3.16b++ ldp x19,x20,[x29,#2*__SIZEOF_POINTER__]+ add sp,sp,#64+ ldp x21,x22,[x29,#4*__SIZEOF_POINTER__]+ ldp x23,x24,[x29,#6*__SIZEOF_POINTER__]+ ldp x25,x26,[x29,#8*__SIZEOF_POINTER__]+ ldp x27,x28,[x29,#10*__SIZEOF_POINTER__]+ ldp x29,x30,[sp],#12*__SIZEOF_POINTER__+.inst 0xd50323bf // autiasp+ ret++.align 4+.Less_than_64_neon:+ eor v1.16b,v1.16b,v1.16b // cleanse key and nonce+ eor v2.16b,v2.16b,v2.16b+ eor v3.16b,v3.16b,v3.16b+ b .Less_than_64+.size crypton_chacha20_asm_neon,.-crypton_chacha20_asm_neon+.type crypton_chacha20_asm_512_neon,%function+.align 5+crypton_chacha20_asm_512_neon:+.inst 0xd503233f // paciasp+ stp x29,x30,[sp,#-12*__SIZEOF_POINTER__]!+ add x29,sp,#0++ adr x5,.Lsigma+ stp x19,x20,[sp,#2*__SIZEOF_POINTER__]+ stp x21,x22,[sp,#4*__SIZEOF_POINTER__]+ stp x23,x24,[sp,#6*__SIZEOF_POINTER__]+ stp x25,x26,[sp,#8*__SIZEOF_POINTER__]+ stp x27,x28,[sp,#10*__SIZEOF_POINTER__]++.L512_or_more_neon:+ sub sp,sp,#128+64++ eor v7.16b,v7.16b,v7.16b+ ldp x22,x23,[x5] // load sigma+ ld1 {v0.4s},[x5],#16+ ldp x24,x25,[x3] // load key+ ldp x26,x27,[x3,#16]+ ld1 {v1.4s,v2.4s},[x3]+ ldp x28,x30,[x4] // load counter+ ld1 {v3.4s},[x4]+ ld1 {v7.s}[0],[x5]+ add x3,x5,#16+#ifdef __AARCH64EB__+ rev64 v0.4s,v0.4s+ ror x24,x24,#32+ ror x25,x25,#32+ ror x26,x26,#32+ ror x27,x27,#32+ ror x28,x28,#32+ ror x30,x30,#32+#endif+ add v3.4s,v3.4s,v7.4s // += 1+ stp q0,q1,[sp,#0] // off-load key block, invariant part+ add v3.4s,v3.4s,v7.4s // not typo+ str q2,[sp,#32]+ add v4.4s,v3.4s,v7.4s+ add v5.4s,v4.4s,v7.4s+ add v6.4s,v5.4s,v7.4s+ shl v7.4s,v7.4s,#2 // 1 -> 4++ stp d8,d9,[sp,#128+0] // meet ABI requirements+ stp d10,d11,[sp,#128+16]+ stp d12,d13,[sp,#128+32]+ stp d14,d15,[sp,#128+48]++ sub x2,x2,#512 // not typo++.Loop_outer_512_neon:+ mov v8.16b,v0.16b+ mov v12.16b,v0.16b+ mov v16.16b,v0.16b+ mov v20.16b,v0.16b+ mov v24.16b,v0.16b+ mov v28.16b,v0.16b+ mov v9.16b,v1.16b+ mov w5,w22 // unpack key block+ mov v13.16b,v1.16b+ lsr x6,x22,#32+ mov v17.16b,v1.16b+ mov w7,w23+ mov v21.16b,v1.16b+ lsr x8,x23,#32+ mov v25.16b,v1.16b+ mov w9,w24+ mov v29.16b,v1.16b+ lsr x10,x24,#32+ mov v11.16b,v3.16b+ mov w11,w25+ mov v15.16b,v4.16b+ lsr x12,x25,#32+ mov v19.16b,v5.16b+ mov w13,w26+ mov v23.16b,v6.16b+ lsr x14,x26,#32+ mov v10.16b,v2.16b+ mov w15,w27+ mov v14.16b,v2.16b+ lsr x16,x27,#32+ add v27.4s,v11.4s,v7.4s // +4+ mov w17,w28+ add v31.4s,v15.4s,v7.4s // +4+ lsr x19,x28,#32+ mov v18.16b,v2.16b+ mov w20,w30+ mov v22.16b,v2.16b+ lsr x21,x30,#32+ mov v26.16b,v2.16b+ stp q3,q4,[sp,#48] // off-load key block, variable part+ mov v30.16b,v2.16b+ stp q5,q6,[sp,#80]++ mov x4,#5+ ld1 {v6.4s},[x3]+ subs x2,x2,#512+.Loop_upper_neon:+ sub x4,x4,#1+ add v8.4s,v8.4s,v9.4s+ add w5,w5,w9+ add v12.4s,v12.4s,v13.4s+ add w6,w6,w10+ add v16.4s,v16.4s,v17.4s+ add w7,w7,w11+ add v20.4s,v20.4s,v21.4s+ add w8,w8,w12+ add v24.4s,v24.4s,v25.4s+ eor w17,w17,w5+ add v28.4s,v28.4s,v29.4s+ eor w19,w19,w6+ eor v11.16b,v11.16b,v8.16b+ eor w20,w20,w7+ eor v15.16b,v15.16b,v12.16b+ eor w21,w21,w8+ eor v19.16b,v19.16b,v16.16b+ ror w17,w17,#16+ eor v23.16b,v23.16b,v20.16b+ ror w19,w19,#16+ eor v27.16b,v27.16b,v24.16b+ ror w20,w20,#16+ eor v31.16b,v31.16b,v28.16b+ ror w21,w21,#16+ rev32 v11.8h,v11.8h+ add w13,w13,w17+ rev32 v15.8h,v15.8h+ add w14,w14,w19+ rev32 v19.8h,v19.8h+ add w15,w15,w20+ rev32 v23.8h,v23.8h+ add w16,w16,w21+ rev32 v27.8h,v27.8h+ eor w9,w9,w13+ rev32 v31.8h,v31.8h+ eor w10,w10,w14+ add v10.4s,v10.4s,v11.4s+ eor w11,w11,w15+ add v14.4s,v14.4s,v15.4s+ eor w12,w12,w16+ add v18.4s,v18.4s,v19.4s+ ror w9,w9,#20+ add v22.4s,v22.4s,v23.4s+ ror w10,w10,#20+ add v26.4s,v26.4s,v27.4s+ ror w11,w11,#20+ add v30.4s,v30.4s,v31.4s+ ror w12,w12,#20+ eor v0.16b,v9.16b,v10.16b+ add w5,w5,w9+ eor v1.16b,v13.16b,v14.16b+ add w6,w6,w10+ eor v2.16b,v17.16b,v18.16b+ add w7,w7,w11+ eor v3.16b,v21.16b,v22.16b+ add w8,w8,w12+ eor v4.16b,v25.16b,v26.16b+ eor w17,w17,w5+ eor v5.16b,v29.16b,v30.16b+ eor w19,w19,w6+ ushr v9.4s,v0.4s,#20+ eor w20,w20,w7+ ushr v13.4s,v1.4s,#20+ eor w21,w21,w8+ ushr v17.4s,v2.4s,#20+ ror w17,w17,#24+ ushr v21.4s,v3.4s,#20+ ror w19,w19,#24+ ushr v25.4s,v4.4s,#20+ ror w20,w20,#24+ ushr v29.4s,v5.4s,#20+ ror w21,w21,#24+ sli v9.4s,v0.4s,#12+ add w13,w13,w17+ sli v13.4s,v1.4s,#12+ add w14,w14,w19+ sli v17.4s,v2.4s,#12+ add w15,w15,w20+ sli v21.4s,v3.4s,#12+ add w16,w16,w21+ sli v25.4s,v4.4s,#12+ eor w9,w9,w13+ sli v29.4s,v5.4s,#12+ eor w10,w10,w14+ add v8.4s,v8.4s,v9.4s+ eor w11,w11,w15+ add v12.4s,v12.4s,v13.4s+ eor w12,w12,w16+ add v16.4s,v16.4s,v17.4s+ ror w9,w9,#25+ add v20.4s,v20.4s,v21.4s+ ror w10,w10,#25+ add v24.4s,v24.4s,v25.4s+ ror w11,w11,#25+ add v28.4s,v28.4s,v29.4s+ ror w12,w12,#25+ eor v11.16b,v11.16b,v8.16b+ add w5,w5,w10+ eor v15.16b,v15.16b,v12.16b+ add w6,w6,w11+ eor v19.16b,v19.16b,v16.16b+ add w7,w7,w12+ eor v23.16b,v23.16b,v20.16b+ add w8,w8,w9+ eor v27.16b,v27.16b,v24.16b+ eor w21,w21,w5+ eor v31.16b,v31.16b,v28.16b+ eor w17,w17,w6+ tbl v11.16b,{v11.16b},v6.16b+ eor w19,w19,w7+ tbl v15.16b,{v15.16b},v6.16b+ eor w20,w20,w8+ tbl v19.16b,{v19.16b},v6.16b+ ror w21,w21,#16+ tbl v23.16b,{v23.16b},v6.16b+ ror w17,w17,#16+ tbl v27.16b,{v27.16b},v6.16b+ ror w19,w19,#16+ tbl v31.16b,{v31.16b},v6.16b+ ror w20,w20,#16+ add v10.4s,v10.4s,v11.4s+ add w15,w15,w21+ add v14.4s,v14.4s,v15.4s+ add w16,w16,w17+ add v18.4s,v18.4s,v19.4s+ add w13,w13,w19+ add v22.4s,v22.4s,v23.4s+ add w14,w14,w20+ add v26.4s,v26.4s,v27.4s+ eor w10,w10,w15+ add v30.4s,v30.4s,v31.4s+ eor w11,w11,w16+ eor v0.16b,v9.16b,v10.16b+ eor w12,w12,w13+ eor v1.16b,v13.16b,v14.16b+ eor w9,w9,w14+ eor v2.16b,v17.16b,v18.16b+ ror w10,w10,#20+ eor v3.16b,v21.16b,v22.16b+ ror w11,w11,#20+ eor v4.16b,v25.16b,v26.16b+ ror w12,w12,#20+ eor v5.16b,v29.16b,v30.16b+ ror w9,w9,#20+ ushr v9.4s,v0.4s,#25+ add w5,w5,w10+ ushr v13.4s,v1.4s,#25+ add w6,w6,w11+ ushr v17.4s,v2.4s,#25+ add w7,w7,w12+ ushr v21.4s,v3.4s,#25+ add w8,w8,w9+ ushr v25.4s,v4.4s,#25+ eor w21,w21,w5+ ushr v29.4s,v5.4s,#25+ eor w17,w17,w6+ sli v9.4s,v0.4s,#7+ eor w19,w19,w7+ sli v13.4s,v1.4s,#7+ eor w20,w20,w8+ sli v17.4s,v2.4s,#7+ ror w21,w21,#24+ sli v21.4s,v3.4s,#7+ ror w17,w17,#24+ sli v25.4s,v4.4s,#7+ ror w19,w19,#24+ sli v29.4s,v5.4s,#7+ ror w20,w20,#24+ ext v10.16b,v10.16b,v10.16b,#8+ add w15,w15,w21+ ext v14.16b,v14.16b,v14.16b,#8+ add w16,w16,w17+ ext v18.16b,v18.16b,v18.16b,#8+ add w13,w13,w19+ ext v22.16b,v22.16b,v22.16b,#8+ add w14,w14,w20+ ext v26.16b,v26.16b,v26.16b,#8+ eor w10,w10,w15+ ext v30.16b,v30.16b,v30.16b,#8+ eor w11,w11,w16+ ext v11.16b,v11.16b,v11.16b,#12+ eor w12,w12,w13+ ext v15.16b,v15.16b,v15.16b,#12+ eor w9,w9,w14+ ext v19.16b,v19.16b,v19.16b,#12+ ror w10,w10,#25+ ext v23.16b,v23.16b,v23.16b,#12+ ror w11,w11,#25+ ext v27.16b,v27.16b,v27.16b,#12+ ror w12,w12,#25+ ext v31.16b,v31.16b,v31.16b,#12+ ror w9,w9,#25+ ext v9.16b,v9.16b,v9.16b,#4+ ext v13.16b,v13.16b,v13.16b,#4+ ext v17.16b,v17.16b,v17.16b,#4+ ext v21.16b,v21.16b,v21.16b,#4+ ext v25.16b,v25.16b,v25.16b,#4+ ext v29.16b,v29.16b,v29.16b,#4+ add v8.4s,v8.4s,v9.4s+ add w5,w5,w9+ add v12.4s,v12.4s,v13.4s+ add w6,w6,w10+ add v16.4s,v16.4s,v17.4s+ add w7,w7,w11+ add v20.4s,v20.4s,v21.4s+ add w8,w8,w12+ add v24.4s,v24.4s,v25.4s+ eor w17,w17,w5+ add v28.4s,v28.4s,v29.4s+ eor w19,w19,w6+ eor v11.16b,v11.16b,v8.16b+ eor w20,w20,w7+ eor v15.16b,v15.16b,v12.16b+ eor w21,w21,w8+ eor v19.16b,v19.16b,v16.16b+ ror w17,w17,#16+ eor v23.16b,v23.16b,v20.16b+ ror w19,w19,#16+ eor v27.16b,v27.16b,v24.16b+ ror w20,w20,#16+ eor v31.16b,v31.16b,v28.16b+ ror w21,w21,#16+ rev32 v11.8h,v11.8h+ add w13,w13,w17+ rev32 v15.8h,v15.8h+ add w14,w14,w19+ rev32 v19.8h,v19.8h+ add w15,w15,w20+ rev32 v23.8h,v23.8h+ add w16,w16,w21+ rev32 v27.8h,v27.8h+ eor w9,w9,w13+ rev32 v31.8h,v31.8h+ eor w10,w10,w14+ add v10.4s,v10.4s,v11.4s+ eor w11,w11,w15+ add v14.4s,v14.4s,v15.4s+ eor w12,w12,w16+ add v18.4s,v18.4s,v19.4s+ ror w9,w9,#20+ add v22.4s,v22.4s,v23.4s+ ror w10,w10,#20+ add v26.4s,v26.4s,v27.4s+ ror w11,w11,#20+ add v30.4s,v30.4s,v31.4s+ ror w12,w12,#20+ eor v0.16b,v9.16b,v10.16b+ add w5,w5,w9+ eor v1.16b,v13.16b,v14.16b+ add w6,w6,w10+ eor v2.16b,v17.16b,v18.16b+ add w7,w7,w11+ eor v3.16b,v21.16b,v22.16b+ add w8,w8,w12+ eor v4.16b,v25.16b,v26.16b+ eor w17,w17,w5+ eor v5.16b,v29.16b,v30.16b+ eor w19,w19,w6+ ushr v9.4s,v0.4s,#20+ eor w20,w20,w7+ ushr v13.4s,v1.4s,#20+ eor w21,w21,w8+ ushr v17.4s,v2.4s,#20+ ror w17,w17,#24+ ushr v21.4s,v3.4s,#20+ ror w19,w19,#24+ ushr v25.4s,v4.4s,#20+ ror w20,w20,#24+ ushr v29.4s,v5.4s,#20+ ror w21,w21,#24+ sli v9.4s,v0.4s,#12+ add w13,w13,w17+ sli v13.4s,v1.4s,#12+ add w14,w14,w19+ sli v17.4s,v2.4s,#12+ add w15,w15,w20+ sli v21.4s,v3.4s,#12+ add w16,w16,w21+ sli v25.4s,v4.4s,#12+ eor w9,w9,w13+ sli v29.4s,v5.4s,#12+ eor w10,w10,w14+ add v8.4s,v8.4s,v9.4s+ eor w11,w11,w15+ add v12.4s,v12.4s,v13.4s+ eor w12,w12,w16+ add v16.4s,v16.4s,v17.4s+ ror w9,w9,#25+ add v20.4s,v20.4s,v21.4s+ ror w10,w10,#25+ add v24.4s,v24.4s,v25.4s+ ror w11,w11,#25+ add v28.4s,v28.4s,v29.4s+ ror w12,w12,#25+ eor v11.16b,v11.16b,v8.16b+ add w5,w5,w10+ eor v15.16b,v15.16b,v12.16b+ add w6,w6,w11+ eor v19.16b,v19.16b,v16.16b+ add w7,w7,w12+ eor v23.16b,v23.16b,v20.16b+ add w8,w8,w9+ eor v27.16b,v27.16b,v24.16b+ eor w21,w21,w5+ eor v31.16b,v31.16b,v28.16b+ eor w17,w17,w6+ tbl v11.16b,{v11.16b},v6.16b+ eor w19,w19,w7+ tbl v15.16b,{v15.16b},v6.16b+ eor w20,w20,w8+ tbl v19.16b,{v19.16b},v6.16b+ ror w21,w21,#16+ tbl v23.16b,{v23.16b},v6.16b+ ror w17,w17,#16+ tbl v27.16b,{v27.16b},v6.16b+ ror w19,w19,#16+ tbl v31.16b,{v31.16b},v6.16b+ ror w20,w20,#16+ add v10.4s,v10.4s,v11.4s+ add w15,w15,w21+ add v14.4s,v14.4s,v15.4s+ add w16,w16,w17+ add v18.4s,v18.4s,v19.4s+ add w13,w13,w19+ add v22.4s,v22.4s,v23.4s+ add w14,w14,w20+ add v26.4s,v26.4s,v27.4s+ eor w10,w10,w15+ add v30.4s,v30.4s,v31.4s+ eor w11,w11,w16+ eor v0.16b,v9.16b,v10.16b+ eor w12,w12,w13+ eor v1.16b,v13.16b,v14.16b+ eor w9,w9,w14+ eor v2.16b,v17.16b,v18.16b+ ror w10,w10,#20+ eor v3.16b,v21.16b,v22.16b+ ror w11,w11,#20+ eor v4.16b,v25.16b,v26.16b+ ror w12,w12,#20+ eor v5.16b,v29.16b,v30.16b+ ror w9,w9,#20+ ushr v9.4s,v0.4s,#25+ add w5,w5,w10+ ushr v13.4s,v1.4s,#25+ add w6,w6,w11+ ushr v17.4s,v2.4s,#25+ add w7,w7,w12+ ushr v21.4s,v3.4s,#25+ add w8,w8,w9+ ushr v25.4s,v4.4s,#25+ eor w21,w21,w5+ ushr v29.4s,v5.4s,#25+ eor w17,w17,w6+ sli v9.4s,v0.4s,#7+ eor w19,w19,w7+ sli v13.4s,v1.4s,#7+ eor w20,w20,w8+ sli v17.4s,v2.4s,#7+ ror w21,w21,#24+ sli v21.4s,v3.4s,#7+ ror w17,w17,#24+ sli v25.4s,v4.4s,#7+ ror w19,w19,#24+ sli v29.4s,v5.4s,#7+ ror w20,w20,#24+ ext v10.16b,v10.16b,v10.16b,#8+ add w15,w15,w21+ ext v14.16b,v14.16b,v14.16b,#8+ add w16,w16,w17+ ext v18.16b,v18.16b,v18.16b,#8+ add w13,w13,w19+ ext v22.16b,v22.16b,v22.16b,#8+ add w14,w14,w20+ ext v26.16b,v26.16b,v26.16b,#8+ eor w10,w10,w15+ ext v30.16b,v30.16b,v30.16b,#8+ eor w11,w11,w16+ ext v11.16b,v11.16b,v11.16b,#4+ eor w12,w12,w13+ ext v15.16b,v15.16b,v15.16b,#4+ eor w9,w9,w14+ ext v19.16b,v19.16b,v19.16b,#4+ ror w10,w10,#25+ ext v23.16b,v23.16b,v23.16b,#4+ ror w11,w11,#25+ ext v27.16b,v27.16b,v27.16b,#4+ ror w12,w12,#25+ ext v31.16b,v31.16b,v31.16b,#4+ ror w9,w9,#25+ ext v9.16b,v9.16b,v9.16b,#12+ ext v13.16b,v13.16b,v13.16b,#12+ ext v17.16b,v17.16b,v17.16b,#12+ ext v21.16b,v21.16b,v21.16b,#12+ ext v25.16b,v25.16b,v25.16b,#12+ ext v29.16b,v29.16b,v29.16b,#12+ cbnz x4,.Loop_upper_neon++ add w5,w5,w22 // accumulate key block+ add x6,x6,x22,lsr#32+ add w7,w7,w23+ add x8,x8,x23,lsr#32+ add w9,w9,w24+ add x10,x10,x24,lsr#32+ add w11,w11,w25+ add x12,x12,x25,lsr#32+ add w13,w13,w26+ add x14,x14,x26,lsr#32+ add w15,w15,w27+ add x16,x16,x27,lsr#32+ add w17,w17,w28+ add x19,x19,x28,lsr#32+ add w20,w20,w30+ add x21,x21,x30,lsr#32++ add x5,x5,x6,lsl#32 // pack+ add x7,x7,x8,lsl#32+ ldp x6,x8,[x1,#0] // load input+ add x9,x9,x10,lsl#32+ add x11,x11,x12,lsl#32+ ldp x10,x12,[x1,#16]+ add x13,x13,x14,lsl#32+ add x15,x15,x16,lsl#32+ ldp x14,x16,[x1,#32]+ add x17,x17,x19,lsl#32+ add x20,x20,x21,lsl#32+ ldp x19,x21,[x1,#48]+ add x1,x1,#64+#ifdef __AARCH64EB__+ rev x5,x5+ rev x7,x7+ rev x9,x9+ rev x11,x11+ rev x13,x13+ rev x15,x15+ rev x17,x17+ rev x20,x20+#endif+ eor x5,x5,x6+ eor x7,x7,x8+ eor x9,x9,x10+ eor x11,x11,x12+ eor x13,x13,x14+ eor x15,x15,x16+ eor x17,x17,x19+ eor x20,x20,x21++ stp x5,x7,[x0,#0] // store output+ add x28,x28,#1 // increment counter+ mov w5,w22 // unpack key block+ lsr x6,x22,#32+ stp x9,x11,[x0,#16]+ mov w7,w23+ lsr x8,x23,#32+ stp x13,x15,[x0,#32]+ mov w9,w24+ lsr x10,x24,#32+ stp x17,x20,[x0,#48]+ add x0,x0,#64+ mov w11,w25+ lsr x12,x25,#32+ mov w13,w26+ lsr x14,x26,#32+ mov w15,w27+ lsr x16,x27,#32+ mov w17,w28+ lsr x19,x28,#32+ mov w20,w30+ lsr x21,x30,#32++ mov x4,#5+.Loop_lower_neon:+ sub x4,x4,#1+ add v8.4s,v8.4s,v9.4s+ add w5,w5,w9+ add v12.4s,v12.4s,v13.4s+ add w6,w6,w10+ add v16.4s,v16.4s,v17.4s+ add w7,w7,w11+ add v20.4s,v20.4s,v21.4s+ add w8,w8,w12+ add v24.4s,v24.4s,v25.4s+ eor w17,w17,w5+ add v28.4s,v28.4s,v29.4s+ eor w19,w19,w6+ eor v11.16b,v11.16b,v8.16b+ eor w20,w20,w7+ eor v15.16b,v15.16b,v12.16b+ eor w21,w21,w8+ eor v19.16b,v19.16b,v16.16b+ ror w17,w17,#16+ eor v23.16b,v23.16b,v20.16b+ ror w19,w19,#16+ eor v27.16b,v27.16b,v24.16b+ ror w20,w20,#16+ eor v31.16b,v31.16b,v28.16b+ ror w21,w21,#16+ rev32 v11.8h,v11.8h+ add w13,w13,w17+ rev32 v15.8h,v15.8h+ add w14,w14,w19+ rev32 v19.8h,v19.8h+ add w15,w15,w20+ rev32 v23.8h,v23.8h+ add w16,w16,w21+ rev32 v27.8h,v27.8h+ eor w9,w9,w13+ rev32 v31.8h,v31.8h+ eor w10,w10,w14+ add v10.4s,v10.4s,v11.4s+ eor w11,w11,w15+ add v14.4s,v14.4s,v15.4s+ eor w12,w12,w16+ add v18.4s,v18.4s,v19.4s+ ror w9,w9,#20+ add v22.4s,v22.4s,v23.4s+ ror w10,w10,#20+ add v26.4s,v26.4s,v27.4s+ ror w11,w11,#20+ add v30.4s,v30.4s,v31.4s+ ror w12,w12,#20+ eor v0.16b,v9.16b,v10.16b+ add w5,w5,w9+ eor v1.16b,v13.16b,v14.16b+ add w6,w6,w10+ eor v2.16b,v17.16b,v18.16b+ add w7,w7,w11+ eor v3.16b,v21.16b,v22.16b+ add w8,w8,w12+ eor v4.16b,v25.16b,v26.16b+ eor w17,w17,w5+ eor v5.16b,v29.16b,v30.16b+ eor w19,w19,w6+ ushr v9.4s,v0.4s,#20+ eor w20,w20,w7+ ushr v13.4s,v1.4s,#20+ eor w21,w21,w8+ ushr v17.4s,v2.4s,#20+ ror w17,w17,#24+ ushr v21.4s,v3.4s,#20+ ror w19,w19,#24+ ushr v25.4s,v4.4s,#20+ ror w20,w20,#24+ ushr v29.4s,v5.4s,#20+ ror w21,w21,#24+ sli v9.4s,v0.4s,#12+ add w13,w13,w17+ sli v13.4s,v1.4s,#12+ add w14,w14,w19+ sli v17.4s,v2.4s,#12+ add w15,w15,w20+ sli v21.4s,v3.4s,#12+ add w16,w16,w21+ sli v25.4s,v4.4s,#12+ eor w9,w9,w13+ sli v29.4s,v5.4s,#12+ eor w10,w10,w14+ add v8.4s,v8.4s,v9.4s+ eor w11,w11,w15+ add v12.4s,v12.4s,v13.4s+ eor w12,w12,w16+ add v16.4s,v16.4s,v17.4s+ ror w9,w9,#25+ add v20.4s,v20.4s,v21.4s+ ror w10,w10,#25+ add v24.4s,v24.4s,v25.4s+ ror w11,w11,#25+ add v28.4s,v28.4s,v29.4s+ ror w12,w12,#25+ eor v11.16b,v11.16b,v8.16b+ add w5,w5,w10+ eor v15.16b,v15.16b,v12.16b+ add w6,w6,w11+ eor v19.16b,v19.16b,v16.16b+ add w7,w7,w12+ eor v23.16b,v23.16b,v20.16b+ add w8,w8,w9+ eor v27.16b,v27.16b,v24.16b+ eor w21,w21,w5+ eor v31.16b,v31.16b,v28.16b+ eor w17,w17,w6+ tbl v11.16b,{v11.16b},v6.16b+ eor w19,w19,w7+ tbl v15.16b,{v15.16b},v6.16b+ eor w20,w20,w8+ tbl v19.16b,{v19.16b},v6.16b+ ror w21,w21,#16+ tbl v23.16b,{v23.16b},v6.16b+ ror w17,w17,#16+ tbl v27.16b,{v27.16b},v6.16b+ ror w19,w19,#16+ tbl v31.16b,{v31.16b},v6.16b+ ror w20,w20,#16+ add v10.4s,v10.4s,v11.4s+ add w15,w15,w21+ add v14.4s,v14.4s,v15.4s+ add w16,w16,w17+ add v18.4s,v18.4s,v19.4s+ add w13,w13,w19+ add v22.4s,v22.4s,v23.4s+ add w14,w14,w20+ add v26.4s,v26.4s,v27.4s+ eor w10,w10,w15+ add v30.4s,v30.4s,v31.4s+ eor w11,w11,w16+ eor v0.16b,v9.16b,v10.16b+ eor w12,w12,w13+ eor v1.16b,v13.16b,v14.16b+ eor w9,w9,w14+ eor v2.16b,v17.16b,v18.16b+ ror w10,w10,#20+ eor v3.16b,v21.16b,v22.16b+ ror w11,w11,#20+ eor v4.16b,v25.16b,v26.16b+ ror w12,w12,#20+ eor v5.16b,v29.16b,v30.16b+ ror w9,w9,#20+ ushr v9.4s,v0.4s,#25+ add w5,w5,w10+ ushr v13.4s,v1.4s,#25+ add w6,w6,w11+ ushr v17.4s,v2.4s,#25+ add w7,w7,w12+ ushr v21.4s,v3.4s,#25+ add w8,w8,w9+ ushr v25.4s,v4.4s,#25+ eor w21,w21,w5+ ushr v29.4s,v5.4s,#25+ eor w17,w17,w6+ sli v9.4s,v0.4s,#7+ eor w19,w19,w7+ sli v13.4s,v1.4s,#7+ eor w20,w20,w8+ sli v17.4s,v2.4s,#7+ ror w21,w21,#24+ sli v21.4s,v3.4s,#7+ ror w17,w17,#24+ sli v25.4s,v4.4s,#7+ ror w19,w19,#24+ sli v29.4s,v5.4s,#7+ ror w20,w20,#24+ ext v10.16b,v10.16b,v10.16b,#8+ add w15,w15,w21+ ext v14.16b,v14.16b,v14.16b,#8+ add w16,w16,w17+ ext v18.16b,v18.16b,v18.16b,#8+ add w13,w13,w19+ ext v22.16b,v22.16b,v22.16b,#8+ add w14,w14,w20+ ext v26.16b,v26.16b,v26.16b,#8+ eor w10,w10,w15+ ext v30.16b,v30.16b,v30.16b,#8+ eor w11,w11,w16+ ext v11.16b,v11.16b,v11.16b,#12+ eor w12,w12,w13+ ext v15.16b,v15.16b,v15.16b,#12+ eor w9,w9,w14+ ext v19.16b,v19.16b,v19.16b,#12+ ror w10,w10,#25+ ext v23.16b,v23.16b,v23.16b,#12+ ror w11,w11,#25+ ext v27.16b,v27.16b,v27.16b,#12+ ror w12,w12,#25+ ext v31.16b,v31.16b,v31.16b,#12+ ror w9,w9,#25+ ext v9.16b,v9.16b,v9.16b,#4+ ext v13.16b,v13.16b,v13.16b,#4+ ext v17.16b,v17.16b,v17.16b,#4+ ext v21.16b,v21.16b,v21.16b,#4+ ext v25.16b,v25.16b,v25.16b,#4+ ext v29.16b,v29.16b,v29.16b,#4+ add v8.4s,v8.4s,v9.4s+ add w5,w5,w9+ add v12.4s,v12.4s,v13.4s+ add w6,w6,w10+ add v16.4s,v16.4s,v17.4s+ add w7,w7,w11+ add v20.4s,v20.4s,v21.4s+ add w8,w8,w12+ add v24.4s,v24.4s,v25.4s+ eor w17,w17,w5+ add v28.4s,v28.4s,v29.4s+ eor w19,w19,w6+ eor v11.16b,v11.16b,v8.16b+ eor w20,w20,w7+ eor v15.16b,v15.16b,v12.16b+ eor w21,w21,w8+ eor v19.16b,v19.16b,v16.16b+ ror w17,w17,#16+ eor v23.16b,v23.16b,v20.16b+ ror w19,w19,#16+ eor v27.16b,v27.16b,v24.16b+ ror w20,w20,#16+ eor v31.16b,v31.16b,v28.16b+ ror w21,w21,#16+ rev32 v11.8h,v11.8h+ add w13,w13,w17+ rev32 v15.8h,v15.8h+ add w14,w14,w19+ rev32 v19.8h,v19.8h+ add w15,w15,w20+ rev32 v23.8h,v23.8h+ add w16,w16,w21+ rev32 v27.8h,v27.8h+ eor w9,w9,w13+ rev32 v31.8h,v31.8h+ eor w10,w10,w14+ add v10.4s,v10.4s,v11.4s+ eor w11,w11,w15+ add v14.4s,v14.4s,v15.4s+ eor w12,w12,w16+ add v18.4s,v18.4s,v19.4s+ ror w9,w9,#20+ add v22.4s,v22.4s,v23.4s+ ror w10,w10,#20+ add v26.4s,v26.4s,v27.4s+ ror w11,w11,#20+ add v30.4s,v30.4s,v31.4s+ ror w12,w12,#20+ eor v0.16b,v9.16b,v10.16b+ add w5,w5,w9+ eor v1.16b,v13.16b,v14.16b+ add w6,w6,w10+ eor v2.16b,v17.16b,v18.16b+ add w7,w7,w11+ eor v3.16b,v21.16b,v22.16b+ add w8,w8,w12+ eor v4.16b,v25.16b,v26.16b+ eor w17,w17,w5+ eor v5.16b,v29.16b,v30.16b+ eor w19,w19,w6+ ushr v9.4s,v0.4s,#20+ eor w20,w20,w7+ ushr v13.4s,v1.4s,#20+ eor w21,w21,w8+ ushr v17.4s,v2.4s,#20+ ror w17,w17,#24+ ushr v21.4s,v3.4s,#20+ ror w19,w19,#24+ ushr v25.4s,v4.4s,#20+ ror w20,w20,#24+ ushr v29.4s,v5.4s,#20+ ror w21,w21,#24+ sli v9.4s,v0.4s,#12+ add w13,w13,w17+ sli v13.4s,v1.4s,#12+ add w14,w14,w19+ sli v17.4s,v2.4s,#12+ add w15,w15,w20+ sli v21.4s,v3.4s,#12+ add w16,w16,w21+ sli v25.4s,v4.4s,#12+ eor w9,w9,w13+ sli v29.4s,v5.4s,#12+ eor w10,w10,w14+ add v8.4s,v8.4s,v9.4s+ eor w11,w11,w15+ add v12.4s,v12.4s,v13.4s+ eor w12,w12,w16+ add v16.4s,v16.4s,v17.4s+ ror w9,w9,#25+ add v20.4s,v20.4s,v21.4s+ ror w10,w10,#25+ add v24.4s,v24.4s,v25.4s+ ror w11,w11,#25+ add v28.4s,v28.4s,v29.4s+ ror w12,w12,#25+ eor v11.16b,v11.16b,v8.16b+ add w5,w5,w10+ eor v15.16b,v15.16b,v12.16b+ add w6,w6,w11+ eor v19.16b,v19.16b,v16.16b+ add w7,w7,w12+ eor v23.16b,v23.16b,v20.16b+ add w8,w8,w9+ eor v27.16b,v27.16b,v24.16b+ eor w21,w21,w5+ eor v31.16b,v31.16b,v28.16b+ eor w17,w17,w6+ tbl v11.16b,{v11.16b},v6.16b+ eor w19,w19,w7+ tbl v15.16b,{v15.16b},v6.16b+ eor w20,w20,w8+ tbl v19.16b,{v19.16b},v6.16b+ ror w21,w21,#16+ tbl v23.16b,{v23.16b},v6.16b+ ror w17,w17,#16+ tbl v27.16b,{v27.16b},v6.16b+ ror w19,w19,#16+ tbl v31.16b,{v31.16b},v6.16b+ ror w20,w20,#16+ add v10.4s,v10.4s,v11.4s+ add w15,w15,w21+ add v14.4s,v14.4s,v15.4s+ add w16,w16,w17+ add v18.4s,v18.4s,v19.4s+ add w13,w13,w19+ add v22.4s,v22.4s,v23.4s+ add w14,w14,w20+ add v26.4s,v26.4s,v27.4s+ eor w10,w10,w15+ add v30.4s,v30.4s,v31.4s+ eor w11,w11,w16+ eor v0.16b,v9.16b,v10.16b+ eor w12,w12,w13+ eor v1.16b,v13.16b,v14.16b+ eor w9,w9,w14+ eor v2.16b,v17.16b,v18.16b+ ror w10,w10,#20+ eor v3.16b,v21.16b,v22.16b+ ror w11,w11,#20+ eor v4.16b,v25.16b,v26.16b+ ror w12,w12,#20+ eor v5.16b,v29.16b,v30.16b+ ror w9,w9,#20+ ushr v9.4s,v0.4s,#25+ add w5,w5,w10+ ushr v13.4s,v1.4s,#25+ add w6,w6,w11+ ushr v17.4s,v2.4s,#25+ add w7,w7,w12+ ushr v21.4s,v3.4s,#25+ add w8,w8,w9+ ushr v25.4s,v4.4s,#25+ eor w21,w21,w5+ ushr v29.4s,v5.4s,#25+ eor w17,w17,w6+ sli v9.4s,v0.4s,#7+ eor w19,w19,w7+ sli v13.4s,v1.4s,#7+ eor w20,w20,w8+ sli v17.4s,v2.4s,#7+ ror w21,w21,#24+ sli v21.4s,v3.4s,#7+ ror w17,w17,#24+ sli v25.4s,v4.4s,#7+ ror w19,w19,#24+ sli v29.4s,v5.4s,#7+ ror w20,w20,#24+ ext v10.16b,v10.16b,v10.16b,#8+ add w15,w15,w21+ ext v14.16b,v14.16b,v14.16b,#8+ add w16,w16,w17+ ext v18.16b,v18.16b,v18.16b,#8+ add w13,w13,w19+ ext v22.16b,v22.16b,v22.16b,#8+ add w14,w14,w20+ ext v26.16b,v26.16b,v26.16b,#8+ eor w10,w10,w15+ ext v30.16b,v30.16b,v30.16b,#8+ eor w11,w11,w16+ ext v11.16b,v11.16b,v11.16b,#4+ eor w12,w12,w13+ ext v15.16b,v15.16b,v15.16b,#4+ eor w9,w9,w14+ ext v19.16b,v19.16b,v19.16b,#4+ ror w10,w10,#25+ ext v23.16b,v23.16b,v23.16b,#4+ ror w11,w11,#25+ ext v27.16b,v27.16b,v27.16b,#4+ ror w12,w12,#25+ ext v31.16b,v31.16b,v31.16b,#4+ ror w9,w9,#25+ ext v9.16b,v9.16b,v9.16b,#12+ ext v13.16b,v13.16b,v13.16b,#12+ ext v17.16b,v17.16b,v17.16b,#12+ ext v21.16b,v21.16b,v21.16b,#12+ ext v25.16b,v25.16b,v25.16b,#12+ ext v29.16b,v29.16b,v29.16b,#12+ cbnz x4,.Loop_lower_neon++ add w5,w5,w22 // accumulate key block+ ldp q0,q1,[sp,#0]+ add x6,x6,x22,lsr#32+ ldp q2,q3,[sp,#32]+ add w7,w7,w23+ ldp q4,q5,[sp,#64]+ add x8,x8,x23,lsr#32+ ldr q6,[sp,#96]+ add v8.4s,v8.4s,v0.4s+ add w9,w9,w24+ add v12.4s,v12.4s,v0.4s+ add x10,x10,x24,lsr#32+ add v16.4s,v16.4s,v0.4s+ add w11,w11,w25+ add v20.4s,v20.4s,v0.4s+ add x12,x12,x25,lsr#32+ add v24.4s,v24.4s,v0.4s+ add w13,w13,w26+ add v28.4s,v28.4s,v0.4s+ add x14,x14,x26,lsr#32+ add v10.4s,v10.4s,v2.4s+ add w15,w15,w27+ add v14.4s,v14.4s,v2.4s+ add x16,x16,x27,lsr#32+ add v18.4s,v18.4s,v2.4s+ add w17,w17,w28+ add v22.4s,v22.4s,v2.4s+ add x19,x19,x28,lsr#32+ add v26.4s,v26.4s,v2.4s+ add w20,w20,w30+ add v30.4s,v30.4s,v2.4s+ add x21,x21,x30,lsr#32+ add v27.4s,v27.4s,v7.4s // +4+ add x5,x5,x6,lsl#32 // pack+ add v31.4s,v31.4s,v7.4s // +4+ add x7,x7,x8,lsl#32+ add v11.4s,v11.4s,v3.4s+ ldp x6,x8,[x1,#0] // load input+ add v15.4s,v15.4s,v4.4s+ add x9,x9,x10,lsl#32+ add v19.4s,v19.4s,v5.4s+ add x11,x11,x12,lsl#32+ add v23.4s,v23.4s,v6.4s+ ldp x10,x12,[x1,#16]+ add v27.4s,v27.4s,v3.4s+ add x13,x13,x14,lsl#32+ add v31.4s,v31.4s,v4.4s+ add x15,x15,x16,lsl#32+ add v9.4s,v9.4s,v1.4s+ ldp x14,x16,[x1,#32]+ add v13.4s,v13.4s,v1.4s+ add x17,x17,x19,lsl#32+ add v17.4s,v17.4s,v1.4s+ add x20,x20,x21,lsl#32+ add v21.4s,v21.4s,v1.4s+ ldp x19,x21,[x1,#48]+ add v25.4s,v25.4s,v1.4s+ add x1,x1,#64+ add v29.4s,v29.4s,v1.4s++#ifdef __AARCH64EB__+ rev x5,x5+ rev x7,x7+ rev x9,x9+ rev x11,x11+ rev x13,x13+ rev x15,x15+ rev x17,x17+ rev x20,x20+#endif+ ld1 {v0.16b,v1.16b,v2.16b,v3.16b},[x1],#64+ eor x5,x5,x6+ eor x7,x7,x8+ eor x9,x9,x10+ eor x11,x11,x12+ eor x13,x13,x14+ eor v8.16b,v8.16b,v0.16b+ eor x15,x15,x16+ eor v9.16b,v9.16b,v1.16b+ eor x17,x17,x19+ eor v10.16b,v10.16b,v2.16b+ eor x20,x20,x21+ eor v11.16b,v11.16b,v3.16b+ ld1 {v0.16b,v1.16b,v2.16b,v3.16b},[x1],#64++ stp x5,x7,[x0,#0] // store output+ add x28,x28,#7 // increment counter+ stp x9,x11,[x0,#16]+ stp x13,x15,[x0,#32]+ stp x17,x20,[x0,#48]+ add x0,x0,#64+ st1 {v8.16b,v9.16b,v10.16b,v11.16b},[x0],#64++ ld1 {v8.16b,v9.16b,v10.16b,v11.16b},[x1],#64+ eor v12.16b,v12.16b,v0.16b+ eor v13.16b,v13.16b,v1.16b+ eor v14.16b,v14.16b,v2.16b+ eor v15.16b,v15.16b,v3.16b+ st1 {v12.16b,v13.16b,v14.16b,v15.16b},[x0],#64++ ld1 {v12.16b,v13.16b,v14.16b,v15.16b},[x1],#64+ eor v16.16b,v16.16b,v8.16b+ ldp q0,q1,[sp,#0]+ eor v17.16b,v17.16b,v9.16b+ ldp q2,q3,[sp,#32]+ eor v18.16b,v18.16b,v10.16b+ eor v19.16b,v19.16b,v11.16b+ st1 {v16.16b,v17.16b,v18.16b,v19.16b},[x0],#64++ ld1 {v16.16b,v17.16b,v18.16b,v19.16b},[x1],#64+ eor v20.16b,v20.16b,v12.16b+ eor v21.16b,v21.16b,v13.16b+ eor v22.16b,v22.16b,v14.16b+ eor v23.16b,v23.16b,v15.16b+ st1 {v20.16b,v21.16b,v22.16b,v23.16b},[x0],#64++ ld1 {v20.16b,v21.16b,v22.16b,v23.16b},[x1],#64+ eor v24.16b,v24.16b,v16.16b+ eor v25.16b,v25.16b,v17.16b+ eor v26.16b,v26.16b,v18.16b+ eor v27.16b,v27.16b,v19.16b+ st1 {v24.16b,v25.16b,v26.16b,v27.16b},[x0],#64++ shl v8.4s,v7.4s,#1 // 4 -> 8+ eor v28.16b,v28.16b,v20.16b+ eor v29.16b,v29.16b,v21.16b+ eor v30.16b,v30.16b,v22.16b+ eor v31.16b,v31.16b,v23.16b+ st1 {v28.16b,v29.16b,v30.16b,v31.16b},[x0],#64++ add v3.4s,v3.4s,v8.4s // += 8+ add v4.4s,v4.4s,v8.4s+ add v5.4s,v5.4s,v8.4s+ add v6.4s,v6.4s,v8.4s++ b.hs .Loop_outer_512_neon++ adds x2,x2,#512+ ushr v7.4s,v7.4s,#1 // 4 -> 2++ ldp d10,d11,[sp,#128+16] // meet ABI requirements+ ldp d12,d13,[sp,#128+32]+ ldp d14,d15,[sp,#128+48]++ stp q0,q0,[sp,#16] // wipe key off-load area+ stp q0,q0,[sp,#48] // [with known constant]+ stp q0,q0,[sp,#80]++ b.eq .Ldone_512_neon++ // we have <512 bytes tail, harmonize state with other contexts+ sub x3,x3,#16+ cmp x2,#192+ add sp,sp,#128+ sub v3.4s,v3.4s,v7.4s // -= 2+ ld1 {v8.4s,v9.4s},[x3]+ b.hs .Loop_outer_neon++ ldp d8,d9,[sp,#0] // meet ABI requirements+ eor v1.16b,v1.16b,v1.16b // cleanse key and nonce+ eor v2.16b,v2.16b,v2.16b+ eor v3.16b,v3.16b,v3.16b+ eor v4.16b,v4.16b,v4.16b+ eor v5.16b,v5.16b,v5.16b+ eor v6.16b,v6.16b,v6.16b+ b .Loop_outer++.Ldone_512_neon:+ ldp d8,d9,[sp,#128+0] // meet ABI requirements+ eor v1.16b,v1.16b,v1.16b // cleanse key and nonce+ eor v2.16b,v2.16b,v2.16b+ eor v3.16b,v3.16b,v3.16b+ eor v4.16b,v4.16b,v4.16b+ eor v5.16b,v5.16b,v5.16b+ eor v6.16b,v6.16b,v6.16b++ ldp x19,x20,[x29,#2*__SIZEOF_POINTER__]+ add sp,sp,#128+64+ ldp x21,x22,[x29,#4*__SIZEOF_POINTER__]+ ldp x23,x24,[x29,#6*__SIZEOF_POINTER__]+ ldp x25,x26,[x29,#8*__SIZEOF_POINTER__]+ ldp x27,x28,[x29,#10*__SIZEOF_POINTER__]+ ldp x29,x30,[sp],#12*__SIZEOF_POINTER__+.inst 0xd50323bf // autiasp+ ret+.size crypton_chacha20_asm_512_neon,.-crypton_chacha20_asm_512_neon++.section .note.GNU-stack,"",%progbits
@@ -0,0 +1,1328 @@+#!/usr/bin/env perl+#+# ====================================================================+# Written by Andy Polyakov, @dot-asm, initially for the OpenSSL+# project.+# ====================================================================+#+# June 2015+#+# ChaCha20 for ARMv8.+#+# April 2019+#+# Replace 3xNEON+1xIALU code path with 4+1. 4+1 is actually fastest+# option on most(*), but not all, processors, yet 6+2 is retained.+# This is because penalties are considered tolerable in comparison to+# improvement on processors where 6+2 helps. Most notably +37% on+# ThunderX2. It's server-oriented processor which will have to serve+# as many requests as possible. While others are mostly clients, when+# performance doesn't have to be absolute top-notch, just fast enough,+# as majority of time is spent "entertaining" relatively slow human.+#+# Performance in cycles per byte out of large buffer.+#+# IALU/gcc-4.9 4xNEON+1xIALU 6xNEON+2xIALU+#+# Apple A7 5.50/+49% 2.72 1.60+# Apple A14/M1 4.50/+27% 1.84 1.27+# Cortex-A53 8.40/+80% 4.06 4.45(*)+# Cortex-A57 8.06/+43% 4.08 4.40(*)+# Cortex-A76 5.52 2.90 2.40+# Cortex-X2 4.35 2.53 1.62+# Cortex-X925 3.94 1.79 1.30+# Denver 4.50/+82% 2.30 2.70(*)+# X-Gene 9.50/+46% 8.20 8.90(*)+# Mongoose 8.00/+44% 2.74 3.12(*)+# Kryo 8.17/+50% 4.47 4.65(*)+# ThunderX2 7.22/+48% 5.64 4.10+# Snapdragon X 3.90 1.79 1.25+#+# (*) slower than 4+1:-(++$flavour=shift;+$output=shift;++if ($flavour && $flavour ne "void") {+ $0 =~ m/(.*[\/\\])[^\/\\]+$/; $dir=$1;+ ( $xlate="${dir}arm-xlate.pl" and -f $xlate ) or+ ( $xlate="${dir}../../perlasm/arm-xlate.pl" and -f $xlate) or+ die "can't locate arm-xlate.pl";++ open STDOUT,"| \"$^X\" $xlate $flavour $output";+} else {+ open STDOUT,">$output";+}++sub AUTOLOAD() # thunk [simplified] x86-style perlasm+{ my $opcode = $AUTOLOAD; $opcode =~ s/.*:://; $opcode =~ s/_/\./;+ my $arg = pop;+ $arg = "#$arg" if ($arg*1 eq $arg);+ $code .= "\t$opcode\t".join(',',@_,$arg)."\n";+}++my ($out,$inp,$len,$key,$ctr) = map("x$_",(0..4));++my @x=map("x$_",(5..17,19..21));+my @d=map("x$_",(22..28,30));++sub ROUND {+my ($a0,$b0,$c0,$d0)=@_;+my ($a1,$b1,$c1,$d1)=map(($_&~3)+(($_+1)&3),($a0,$b0,$c0,$d0));+my ($a2,$b2,$c2,$d2)=map(($_&~3)+(($_+1)&3),($a1,$b1,$c1,$d1));+my ($a3,$b3,$c3,$d3)=map(($_&~3)+(($_+1)&3),($a2,$b2,$c2,$d2));++ (+ "&add_32 (@x[$a0],@x[$a0],@x[$b0])",+ "&add_32 (@x[$a1],@x[$a1],@x[$b1])",+ "&add_32 (@x[$a2],@x[$a2],@x[$b2])",+ "&add_32 (@x[$a3],@x[$a3],@x[$b3])",+ "&eor_32 (@x[$d0],@x[$d0],@x[$a0])",+ "&eor_32 (@x[$d1],@x[$d1],@x[$a1])",+ "&eor_32 (@x[$d2],@x[$d2],@x[$a2])",+ "&eor_32 (@x[$d3],@x[$d3],@x[$a3])",+ "&ror_32 (@x[$d0],@x[$d0],16)",+ "&ror_32 (@x[$d1],@x[$d1],16)",+ "&ror_32 (@x[$d2],@x[$d2],16)",+ "&ror_32 (@x[$d3],@x[$d3],16)",++ "&add_32 (@x[$c0],@x[$c0],@x[$d0])",+ "&add_32 (@x[$c1],@x[$c1],@x[$d1])",+ "&add_32 (@x[$c2],@x[$c2],@x[$d2])",+ "&add_32 (@x[$c3],@x[$c3],@x[$d3])",+ "&eor_32 (@x[$b0],@x[$b0],@x[$c0])",+ "&eor_32 (@x[$b1],@x[$b1],@x[$c1])",+ "&eor_32 (@x[$b2],@x[$b2],@x[$c2])",+ "&eor_32 (@x[$b3],@x[$b3],@x[$c3])",+ "&ror_32 (@x[$b0],@x[$b0],20)",+ "&ror_32 (@x[$b1],@x[$b1],20)",+ "&ror_32 (@x[$b2],@x[$b2],20)",+ "&ror_32 (@x[$b3],@x[$b3],20)",++ "&add_32 (@x[$a0],@x[$a0],@x[$b0])",+ "&add_32 (@x[$a1],@x[$a1],@x[$b1])",+ "&add_32 (@x[$a2],@x[$a2],@x[$b2])",+ "&add_32 (@x[$a3],@x[$a3],@x[$b3])",+ "&eor_32 (@x[$d0],@x[$d0],@x[$a0])",+ "&eor_32 (@x[$d1],@x[$d1],@x[$a1])",+ "&eor_32 (@x[$d2],@x[$d2],@x[$a2])",+ "&eor_32 (@x[$d3],@x[$d3],@x[$a3])",+ "&ror_32 (@x[$d0],@x[$d0],24)",+ "&ror_32 (@x[$d1],@x[$d1],24)",+ "&ror_32 (@x[$d2],@x[$d2],24)",+ "&ror_32 (@x[$d3],@x[$d3],24)",++ "&add_32 (@x[$c0],@x[$c0],@x[$d0])",+ "&add_32 (@x[$c1],@x[$c1],@x[$d1])",+ "&add_32 (@x[$c2],@x[$c2],@x[$d2])",+ "&add_32 (@x[$c3],@x[$c3],@x[$d3])",+ "&eor_32 (@x[$b0],@x[$b0],@x[$c0])",+ "&eor_32 (@x[$b1],@x[$b1],@x[$c1])",+ "&eor_32 (@x[$b2],@x[$b2],@x[$c2])",+ "&eor_32 (@x[$b3],@x[$b3],@x[$c3])",+ "&ror_32 (@x[$b0],@x[$b0],25)",+ "&ror_32 (@x[$b1],@x[$b1],25)",+ "&ror_32 (@x[$b2],@x[$b2],25)",+ "&ror_32 (@x[$b3],@x[$b3],25)"+ );+}++$code.=<<___;+#ifndef __KERNEL__+# include "arm_arch.h"+.extern OPENSSL_armcap_P+#endif++.text++.align 5+.Lsigma:+.quad 0x3320646e61707865,0x6b20657479622d32 // endian-neutral+.Lone:+.long 1,2,3,4+.Lrot24:+.long 0x02010003,0x06050407,0x0a09080b,0x0e0d0c0f+.asciz "ChaCha20 for ARMv8, CRYPTOGAMS by \@dot-asm"++.globl ChaCha20_ctr32+.type ChaCha20_ctr32,%function+.align 5+ChaCha20_ctr32:+ cbz $len,.Labort+ cmp $len,#192+ b.lo .Lshort++#ifndef __KERNEL__+ adrp c17,OPENSSL_armcap_P+ ldr w17,[c17,#:lo12:OPENSSL_armcap_P]+ tst w17,#ARMV7_NEON+ b.ne .LChaCha20_neon+#endif++.Lshort:+ .inst 0xd503233f // paciasp+ stp c29,c30,[sp,#-12*__SIZEOF_POINTER__]!+ add c29,csp,#0++ adr @x[0],.Lsigma+ stp c19,c20,[csp,#2*__SIZEOF_POINTER__]+ stp c21,c22,[csp,#4*__SIZEOF_POINTER__]+ stp c23,c24,[csp,#6*__SIZEOF_POINTER__]+ stp c25,c26,[csp,#8*__SIZEOF_POINTER__]+ stp c27,c28,[csp,#10*__SIZEOF_POINTER__]+ sub csp,csp,#64++ ldp @d[0],@d[1],[@x[0]] // load sigma+ ldp @d[2],@d[3],[$key] // load key+ ldp @d[4],@d[5],[$key,#16]+ ldp @d[6],@d[7],[$ctr] // load counter+#ifdef __AARCH64EB__+ ror @d[2],@d[2],#32+ ror @d[3],@d[3],#32+ ror @d[4],@d[4],#32+ ror @d[5],@d[5],#32+ ror @d[6],@d[6],#32+ ror @d[7],@d[7],#32+#endif++.Loop_outer:+ mov.32 @x[0],@d[0] // unpack key block+ lsr @x[1],@d[0],#32+ mov.32 @x[2],@d[1]+ lsr @x[3],@d[1],#32+ mov.32 @x[4],@d[2]+ lsr @x[5],@d[2],#32+ mov.32 @x[6],@d[3]+ lsr @x[7],@d[3],#32+ mov.32 @x[8],@d[4]+ lsr @x[9],@d[4],#32+ mov.32 @x[10],@d[5]+ lsr @x[11],@d[5],#32+ mov.32 @x[12],@d[6]+ lsr @x[13],@d[6],#32+ mov.32 @x[14],@d[7]+ lsr @x[15],@d[7],#32++ mov $ctr,#10+ subs $len,$len,#64+.Loop:+ sub $ctr,$ctr,#1+___+ foreach (&ROUND(0, 4, 8,12)) { eval; }+ foreach (&ROUND(0, 5,10,15)) { eval; }+$code.=<<___;+ cbnz $ctr,.Loop++ add.32 @x[0],@x[0],@d[0] // accumulate key block+ add @x[1],@x[1],@d[0],lsr#32+ add.32 @x[2],@x[2],@d[1]+ add @x[3],@x[3],@d[1],lsr#32+ add.32 @x[4],@x[4],@d[2]+ add @x[5],@x[5],@d[2],lsr#32+ add.32 @x[6],@x[6],@d[3]+ add @x[7],@x[7],@d[3],lsr#32+ add.32 @x[8],@x[8],@d[4]+ add @x[9],@x[9],@d[4],lsr#32+ add.32 @x[10],@x[10],@d[5]+ add @x[11],@x[11],@d[5],lsr#32+ add.32 @x[12],@x[12],@d[6]+ add @x[13],@x[13],@d[6],lsr#32+ add.32 @x[14],@x[14],@d[7]+ add @x[15],@x[15],@d[7],lsr#32++ b.lo .Ltail++ add @x[0],@x[0],@x[1],lsl#32 // pack+ add @x[2],@x[2],@x[3],lsl#32+ ldp @x[1],@x[3],[$inp,#0] // load input+ add @x[4],@x[4],@x[5],lsl#32+ add @x[6],@x[6],@x[7],lsl#32+ ldp @x[5],@x[7],[$inp,#16]+ add @x[8],@x[8],@x[9],lsl#32+ add @x[10],@x[10],@x[11],lsl#32+ ldp @x[9],@x[11],[$inp,#32]+ add @x[12],@x[12],@x[13],lsl#32+ add @x[14],@x[14],@x[15],lsl#32+ ldp @x[13],@x[15],[$inp,#48]+ cadd $inp,$inp,#64+#ifdef __AARCH64EB__+ rev @x[0],@x[0]+ rev @x[2],@x[2]+ rev @x[4],@x[4]+ rev @x[6],@x[6]+ rev @x[8],@x[8]+ rev @x[10],@x[10]+ rev @x[12],@x[12]+ rev @x[14],@x[14]+#endif+ eor @x[0],@x[0],@x[1]+ eor @x[2],@x[2],@x[3]+ eor @x[4],@x[4],@x[5]+ eor @x[6],@x[6],@x[7]+ eor @x[8],@x[8],@x[9]+ eor @x[10],@x[10],@x[11]+ eor @x[12],@x[12],@x[13]+ eor @x[14],@x[14],@x[15]++ stp @x[0],@x[2],[$out,#0] // store output+ add @d[6],@d[6],#1 // increment counter+ stp @x[4],@x[6],[$out,#16]+ stp @x[8],@x[10],[$out,#32]+ stp @x[12],@x[14],[$out,#48]+ cadd $out,$out,#64++ b.hi .Loop_outer++ ldp c19,c20,[c29,#2*__SIZEOF_POINTER__]+ add csp,csp,#64+ ldp c21,c22,[c29,#4*__SIZEOF_POINTER__]+ ldp c23,c24,[c29,#6*__SIZEOF_POINTER__]+ ldp c25,c26,[c29,#8*__SIZEOF_POINTER__]+ ldp c27,c28,[c29,#10*__SIZEOF_POINTER__]+ ldp c29,c30,[csp],#12*__SIZEOF_POINTER__+ .inst 0xd50323bf // autiasp+.Labort:+ ret++.align 4+.Ltail:+ add $len,$len,#64+.Less_than_64:+ csub $out,$out,#1+ cadd $inp,$inp,$len+ cadd $out,$out,$len+ cadd $ctr,sp,$len+ neg $len,$len++ add @x[0],@x[0],@x[1],lsl#32 // pack+ add @x[2],@x[2],@x[3],lsl#32+ add @x[4],@x[4],@x[5],lsl#32+ add @x[6],@x[6],@x[7],lsl#32+ add @x[8],@x[8],@x[9],lsl#32+ add @x[10],@x[10],@x[11],lsl#32+ add @x[12],@x[12],@x[13],lsl#32+ add @x[14],@x[14],@x[15],lsl#32+#ifdef __AARCH64EB__+ rev @x[0],@x[0]+ rev @x[2],@x[2]+ rev @x[4],@x[4]+ rev @x[6],@x[6]+ rev @x[8],@x[8]+ rev @x[10],@x[10]+ rev @x[12],@x[12]+ rev @x[14],@x[14]+#endif+ stp @x[0],@x[2],[sp,#0] // off-load complete block+ stp @x[4],@x[6],[sp,#16]+ stp @x[8],@x[10],[sp,#32]+ stp @x[12],@x[14],[sp,#48]++.Loop_tail:+ ldrb w10,[$inp,$len]+ ldrb w11,[$ctr,$len]+ add $len,$len,#1+ eor w10,w10,w11+ strb w10,[$out,$len]+ cbnz $len,.Loop_tail++ stp xzr,xzr,[sp,#0] // wipe off-load area+ stp xzr,xzr,[sp,#16]+ stp xzr,xzr,[sp,#32]+ stp xzr,xzr,[sp,#48]++ ldp c19,c20,[c29,#2*__SIZEOF_POINTER__]+ add csp,csp,#64+ ldp c21,c22,[c29,#4*__SIZEOF_POINTER__]+ ldp c23,c24,[c29,#6*__SIZEOF_POINTER__]+ ldp c25,c26,[c29,#8*__SIZEOF_POINTER__]+ ldp c27,c28,[c29,#10*__SIZEOF_POINTER__]+ ldp c29,c30,[csp],#12*__SIZEOF_POINTER__+ .inst 0xd50323bf // autiasp+ ret+.size ChaCha20_ctr32,.-ChaCha20_ctr32+___++{{{+########################################################################+# 4x"vertical" layout reduces *total* amount of instructions by trading+# 60 "horizontal" permutations in inner loop for 32-instruction diagonal+# transposition at the loop exit. And since NEON instruction issue rate+# is customarily limited, it's possible to process one additional block+# with scalar instructions at no additional cost. Hence the "4+1"+# description...++my @K = map("v$_.4s",(0..3));+my ($xt0,$xt1,$xt2,$xt3, $CTR,$ROT24) = map("v$_.4s",(4..9));+my @X = map("v$_.4s",(16,20,24,28, 17,21,25,29, 18,22,26,30, 19,23,27,31));+my ($xa0,$xa1,$xa2,$xa3, $xb0,$xb1,$xb2,$xb3,+ $xc0,$xc1,$xc2,$xc3, $xd0,$xd1,$xd2,$xd3) = @X;++sub NEON_lane_ROUND {+my ($a0,$b0,$c0,$d0)=@_;+my ($a1,$b1,$c1,$d1)=map(($_&~3)+(($_+1)&3),($a0,$b0,$c0,$d0));+my ($a2,$b2,$c2,$d2)=map(($_&~3)+(($_+1)&3),($a1,$b1,$c1,$d1));+my ($a3,$b3,$c3,$d3)=map(($_&~3)+(($_+1)&3),($a2,$b2,$c2,$d2));+my @x=map("'$_'",@X);++ (+ "&add (@x[$a0],@x[$a0],@x[$b0])", # Q1+ "&add (@x[$a1],@x[$a1],@x[$b1])", # Q2+ "&add (@x[$a2],@x[$a2],@x[$b2])", # Q3+ "&add (@x[$a3],@x[$a3],@x[$b3])", # Q4+ "&eor (@x[$d0],@x[$d0],@x[$a0])",+ "&eor (@x[$d1],@x[$d1],@x[$a1])",+ "&eor (@x[$d2],@x[$d2],@x[$a2])",+ "&eor (@x[$d3],@x[$d3],@x[$a3])",+ "&rev32_16 (@x[$d0],@x[$d0])",+ "&rev32_16 (@x[$d1],@x[$d1])",+ "&rev32_16 (@x[$d2],@x[$d2])",+ "&rev32_16 (@x[$d3],@x[$d3])",++ "&add (@x[$c0],@x[$c0],@x[$d0])",+ "&add (@x[$c1],@x[$c1],@x[$d1])",+ "&add (@x[$c2],@x[$c2],@x[$d2])",+ "&add (@x[$c3],@x[$c3],@x[$d3])",+ "&eor ('$xt0',@x[$b0],@x[$c0])",+ "&eor ('$xt1',@x[$b1],@x[$c1])",+ "&eor ('$xt2',@x[$b2],@x[$c2])",+ "&eor ('$xt3',@x[$b3],@x[$c3])",+ "&ushr (@x[$b0],'$xt0',20)",+ "&ushr (@x[$b1],'$xt1',20)",+ "&ushr (@x[$b2],'$xt2',20)",+ "&ushr (@x[$b3],'$xt3',20)",+ "&sli (@x[$b0],'$xt0',12)",+ "&sli (@x[$b1],'$xt1',12)",+ "&sli (@x[$b2],'$xt2',12)",+ "&sli (@x[$b3],'$xt3',12)",++ "&add (@x[$a0],@x[$a0],@x[$b0])",+ "&add (@x[$a1],@x[$a1],@x[$b1])",+ "&add (@x[$a2],@x[$a2],@x[$b2])",+ "&add (@x[$a3],@x[$a3],@x[$b3])",+ "&eor ('$xt0',@x[$d0],@x[$a0])",+ "&eor ('$xt1',@x[$d1],@x[$a1])",+ "&eor ('$xt2',@x[$d2],@x[$a2])",+ "&eor ('$xt3',@x[$d3],@x[$a3])",+ "&tbl (@x[$d0],'{$xt0}','$ROT24')",+ "&tbl (@x[$d1],'{$xt1}','$ROT24')",+ "&tbl (@x[$d2],'{$xt2}','$ROT24')",+ "&tbl (@x[$d3],'{$xt3}','$ROT24')",++ "&add (@x[$c0],@x[$c0],@x[$d0])",+ "&add (@x[$c1],@x[$c1],@x[$d1])",+ "&add (@x[$c2],@x[$c2],@x[$d2])",+ "&add (@x[$c3],@x[$c3],@x[$d3])",+ "&eor ('$xt0',@x[$b0],@x[$c0])",+ "&eor ('$xt1',@x[$b1],@x[$c1])",+ "&eor ('$xt2',@x[$b2],@x[$c2])",+ "&eor ('$xt3',@x[$b3],@x[$c3])",+ "&ushr (@x[$b0],'$xt0',25)",+ "&ushr (@x[$b1],'$xt1',25)",+ "&ushr (@x[$b2],'$xt2',25)",+ "&ushr (@x[$b3],'$xt3',25)",+ "&sli (@x[$b0],'$xt0',7)",+ "&sli (@x[$b1],'$xt1',7)",+ "&sli (@x[$b2],'$xt2',7)",+ "&sli (@x[$b3],'$xt3',7)"+ );+}++$code.=<<___;++#ifdef __KERNEL__+.globl ChaCha20_neon+#endif+.type ChaCha20_neon,%function+.align 5+ChaCha20_neon:+.LChaCha20_neon:+ .inst 0xd503233f // paciasp+ stp c29,c30,[csp,#-12*__SIZEOF_POINTER__]!+ add c29,csp,#0++ adr @x[0],.Lsigma+ stp c19,c20,[csp,#2*__SIZEOF_POINTER__]+ stp c21,c22,[csp,#4*__SIZEOF_POINTER__]+ stp c23,c24,[csp,#6*__SIZEOF_POINTER__]+ stp c25,c26,[csp,#8*__SIZEOF_POINTER__]+ stp c27,c28,[csp,#10*__SIZEOF_POINTER__]+ cmp $len,#512+ b.hs .L512_or_more_neon++ sub csp,csp,#64++ ldp @d[0],@d[1],[@x[0]] // load sigma+ ld1 {@K[0]},[@x[0]],#16+ ldp @d[2],@d[3],[$key] // load key+ ldp @d[4],@d[5],[$key,#16]+ ld1 {@K[1],@K[2]},[$key]+ ldp @d[6],@d[7],[$ctr] // load counter+ ld1 {@K[3]},[$ctr]+ stp d8,d9,[sp] // meet ABI requirements+ ld1 {$CTR,$ROT24},[@x[0]]+#ifdef __AARCH64EB__+ rev64 @K[0],@K[0]+ ror @d[2],@d[2],#32+ ror @d[3],@d[3],#32+ ror @d[4],@d[4],#32+ ror @d[5],@d[5],#32+ ror @d[6],@d[6],#32+ ror @d[7],@d[7],#32+#endif++.Loop_outer_neon:+ dup $xa0,@{K[0]}[0] // unpack key block+ mov.32 @x[0],@d[0]+ dup $xa1,@{K[0]}[1]+ lsr @x[1],@d[0],#32+ dup $xa2,@{K[0]}[2]+ mov.32 @x[2],@d[1]+ dup $xa3,@{K[0]}[3]+ lsr @x[3],@d[1],#32+ dup $xb0,@{K[1]}[0]+ mov.32 @x[4],@d[2]+ dup $xb1,@{K[1]}[1]+ lsr @x[5],@d[2],#32+ dup $xb2,@{K[1]}[2]+ mov.32 @x[6],@d[3]+ dup $xb3,@{K[1]}[3]+ lsr @x[7],@d[3],#32+ dup $xd0,@{K[3]}[0]+ mov.32 @x[8],@d[4]+ dup $xd1,@{K[3]}[1]+ lsr @x[9],@d[4],#32+ dup $xd2,@{K[3]}[2]+ mov.32 @x[10],@d[5]+ dup $xd3,@{K[3]}[3]+ lsr @x[11],@d[5],#32+ add $xd0,$xd0,$CTR+ mov.32 @x[12],@d[6]+ dup $xc0,@{K[2]}[0]+ lsr @x[13],@d[6],#32+ dup $xc1,@{K[2]}[1]+ mov.32 @x[14],@d[7]+ dup $xc2,@{K[2]}[2]+ lsr @x[15],@d[7],#32+ dup $xc3,@{K[2]}[3]++ mov $ctr,#10+ subs $len,$len,#320+.Loop_neon:+ sub $ctr,$ctr,#1+___+ my @plus_one=&ROUND(0,4,8,12); my $i=0;+ foreach (&NEON_lane_ROUND(0,4,8,12)) { eval; eval(shift(@plus_one)) if ($i++ > 6); }+ foreach (@plus_one) { eval; }++ @plus_one=&ROUND(0,5,10,15); $i=0;+ foreach (&NEON_lane_ROUND(0,5,10,15)) { eval; eval(shift(@plus_one)) if ($i++ > 6); }+ foreach (@plus_one) { eval; }+$code.=<<___;+ cbnz $ctr,.Loop_neon++ add $xd0,$xd0,$CTR++ zip1 $xt0,$xa0,$xa1 // transpose data+ zip1 $xt1,$xa2,$xa3+ zip2 $xt2,$xa0,$xa1+ zip2 $xt3,$xa2,$xa3+ zip1.64 $xa0,$xt0,$xt1+ zip2.64 $xa1,$xt0,$xt1+ zip1.64 $xa2,$xt2,$xt3+ zip2.64 $xa3,$xt2,$xt3++ zip1 $xt0,$xb0,$xb1+ zip1 $xt1,$xb2,$xb3+ zip2 $xt2,$xb0,$xb1+ zip2 $xt3,$xb2,$xb3+ zip1.64 $xb0,$xt0,$xt1+ zip2.64 $xb1,$xt0,$xt1+ zip1.64 $xb2,$xt2,$xt3+ zip2.64 $xb3,$xt2,$xt3++ zip1 $xt0,$xc0,$xc1+ add.32 @x[0],@x[0],@d[0] // accumulate key block+ zip1 $xt1,$xc2,$xc3+ add @x[1],@x[1],@d[0],lsr#32+ zip2 $xt2,$xc0,$xc1+ add.32 @x[2],@x[2],@d[1]+ zip2 $xt3,$xc2,$xc3+ add @x[3],@x[3],@d[1],lsr#32+ zip1.64 $xc0,$xt0,$xt1+ add.32 @x[4],@x[4],@d[2]+ zip2.64 $xc1,$xt0,$xt1+ add @x[5],@x[5],@d[2],lsr#32+ zip1.64 $xc2,$xt2,$xt3+ add.32 @x[6],@x[6],@d[3]+ zip2.64 $xc3,$xt2,$xt3+ add @x[7],@x[7],@d[3],lsr#32++ zip1 $xt0,$xd0,$xd1+ add.32 @x[8],@x[8],@d[4]+ zip1 $xt1,$xd2,$xd3+ add @x[9],@x[9],@d[4],lsr#32+ zip2 $xt2,$xd0,$xd1+ add.32 @x[10],@x[10],@d[5]+ zip2 $xt3,$xd2,$xd3+ add @x[11],@x[11],@d[5],lsr#32+ zip1.64 $xd0,$xt0,$xt1+ add.32 @x[12],@x[12],@d[6]+ zip2.64 $xd1,$xt0,$xt1+ add @x[13],@x[13],@d[6],lsr#32+ zip1.64 $xd2,$xt2,$xt3+ add.32 @x[14],@x[14],@d[7]+ zip2.64 $xd3,$xt2,$xt3+ add @x[15],@x[15],@d[7],lsr#32++ b.lo .Ltail_neon++ add @x[0],@x[0],@x[1],lsl#32 // pack+ add @x[2],@x[2],@x[3],lsl#32+ ldp @x[1],@x[3],[$inp,#0] // load input+ add $xa0,$xa0,@K[0] // accumulate key block+ add @x[4],@x[4],@x[5],lsl#32+ add @x[6],@x[6],@x[7],lsl#32+ ldp @x[5],@x[7],[$inp,#16]+ add $xb0,$xb0,@K[1]+ add @x[8],@x[8],@x[9],lsl#32+ add @x[10],@x[10],@x[11],lsl#32+ ldp @x[9],@x[11],[$inp,#32]+ add $xc0,$xc0,@K[2]+ add @x[12],@x[12],@x[13],lsl#32+ add @x[14],@x[14],@x[15],lsl#32+ ldp @x[13],@x[15],[$inp,#48]+ add $xd0,$xd0,@K[3]+ cadd $inp,$inp,#64+#ifdef __AARCH64EB__+ rev @x[0],@x[0]+ rev @x[2],@x[2]+ rev @x[4],@x[4]+ rev @x[6],@x[6]+ rev @x[8],@x[8]+ rev @x[10],@x[10]+ rev @x[12],@x[12]+ rev @x[14],@x[14]+#endif+ ld1.8 {$xt0-$xt3},[$inp],#64+ eor @x[0],@x[0],@x[1]+ add $xa1,$xa1,@K[0]+ eor @x[2],@x[2],@x[3]+ add $xb1,$xb1,@K[1]+ eor @x[4],@x[4],@x[5]+ add $xc1,$xc1,@K[2]+ eor @x[6],@x[6],@x[7]+ add $xd1,$xd1,@K[3]+ eor @x[8],@x[8],@x[9]+ eor $xa0,$xa0,$xt0+ movi $xt0,#5+ eor @x[10],@x[10],@x[11]+ eor $xb0,$xb0,$xt1+ eor @x[12],@x[12],@x[13]+ eor $xc0,$xc0,$xt2+ eor @x[14],@x[14],@x[15]+ eor $xd0,$xd0,$xt3+ add $CTR,$CTR,$xt0 // += 5+ ld1.8 {$xt0-$xt3},[$inp],#64++ stp @x[0],@x[2],[$out,#0] // store output+ add @d[6],@d[6],#5 // increment counter+ stp @x[4],@x[6],[$out,#16]+ stp @x[8],@x[10],[$out,#32]+ stp @x[12],@x[14],[$out,#48]+ cadd $out,$out,#64++ st1.8 {$xa0-$xd0},[$out],#64+ add $xa2,$xa2,@K[0]+ add $xb2,$xb2,@K[1]+ add $xc2,$xc2,@K[2]+ add $xd2,$xd2,@K[3]+ ld1.8 {$xa0-$xd0},[$inp],#64++ eor $xa1,$xa1,$xt0+ eor $xb1,$xb1,$xt1+ eor $xc1,$xc1,$xt2+ eor $xd1,$xd1,$xt3+ st1.8 {$xa1-$xd1},[$out],#64+ add $xa3,$xa3,@K[0]+ add $xb3,$xb3,@K[1]+ add $xc3,$xc3,@K[2]+ add $xd3,$xd3,@K[3]+ ld1.8 {$xa1-$xd1},[$inp],#64++ eor $xa2,$xa2,$xa0+ eor $xb2,$xb2,$xb0+ eor $xc2,$xc2,$xc0+ eor $xd2,$xd2,$xd0+ st1.8 {$xa2-$xd2},[$out],#64++ eor $xa3,$xa3,$xa1+ eor $xb3,$xb3,$xb1+ eor $xc3,$xc3,$xc1+ eor $xd3,$xd3,$xd1+ st1.8 {$xa3-$xd3},[$out],#64++ b.hi .Loop_outer_neon++ ldp d8,d9,[sp] // meet ABI requirements+ eor @K[1],@K[1],@K[1] // cleanse key and nonce+ eor @K[2],@K[2],@K[2]+ eor @K[3],@K[3],@K[3]++ ldp c19,c20,[c29,#2*__SIZEOF_POINTER__]+ add csp,csp,#64+ ldp c21,c22,[c29,#4*__SIZEOF_POINTER__]+ ldp c23,c24,[c29,#6*__SIZEOF_POINTER__]+ ldp c25,c26,[c29,#8*__SIZEOF_POINTER__]+ ldp c27,c28,[c29,#10*__SIZEOF_POINTER__]+ ldp c29,c30,[csp],#12*__SIZEOF_POINTER__+ .inst 0xd50323bf // autiasp+ ret++.align 4+.Ltail_neon:+ add $len,$len,#320+ ldp d8,d9,[sp] // meet ABI requirements+ cmp $len,#64+ b.lo .Less_than_64_neon++ add @x[0],@x[0],@x[1],lsl#32 // pack+ add @x[2],@x[2],@x[3],lsl#32+ ldp @x[1],@x[3],[$inp,#0] // load input+ add @x[4],@x[4],@x[5],lsl#32+ add @x[6],@x[6],@x[7],lsl#32+ ldp @x[5],@x[7],[$inp,#16]+ add @x[8],@x[8],@x[9],lsl#32+ add @x[10],@x[10],@x[11],lsl#32+ ldp @x[9],@x[11],[$inp,#32]+ add @x[12],@x[12],@x[13],lsl#32+ add @x[14],@x[14],@x[15],lsl#32+ ldp @x[13],@x[15],[$inp,#48]+ cadd $inp,$inp,#64+#ifdef __AARCH64EB__+ rev @x[0],@x[0]+ rev @x[2],@x[2]+ rev @x[4],@x[4]+ rev @x[6],@x[6]+ rev @x[8],@x[8]+ rev @x[10],@x[10]+ rev @x[12],@x[12]+ rev @x[14],@x[14]+#endif+ eor @x[0],@x[0],@x[1]+ eor @x[2],@x[2],@x[3]+ eor @x[4],@x[4],@x[5]+ eor @x[6],@x[6],@x[7]+ eor @x[8],@x[8],@x[9]+ eor @x[10],@x[10],@x[11]+ eor @x[12],@x[12],@x[13]+ eor @x[14],@x[14],@x[15]++ stp @x[0],@x[2],[$out,#0] // store output+ add $xa0,$xa0,@K[0] // accumulate key block+ stp @x[4],@x[6],[$out,#16]+ add $xb0,$xb0,@K[1]+ stp @x[8],@x[10],[$out,#32]+ add $xc0,$xc0,@K[2]+ stp @x[12],@x[14],[$out,#48]+ add $xd0,$xd0,@K[3]+ cadd $out,$out,#64+ b.eq .Ldone_neon+ sub $len,$len,#64+ cmp $len,#64+ b.lo .Last_neon++ ld1.8 {$xt0-$xt3},[$inp],#64+ eor $xa0,$xa0,$xt0+ eor $xb0,$xb0,$xt1+ eor $xc0,$xc0,$xt2+ eor $xd0,$xd0,$xt3+ st1.8 {$xa0-$xd0},[$out],#64+ b.eq .Ldone_neon++ add $xa0,$xa1,@K[0]+ add $xb0,$xb1,@K[1]+ sub $len,$len,#64+ add $xc0,$xc1,@K[2]+ cmp $len,#64+ add $xd0,$xd1,@K[3]+ b.lo .Last_neon++ ld1.8 {$xt0-$xt3},[$inp],#64+ eor $xa1,$xa0,$xt0+ eor $xb1,$xb0,$xt1+ eor $xc1,$xc0,$xt2+ eor $xd1,$xd0,$xt3+ st1.8 {$xa1-$xd1},[$out],#64+ b.eq .Ldone_neon++ add $xa0,$xa2,@K[0]+ add $xb0,$xb2,@K[1]+ sub $len,$len,#64+ add $xc0,$xc2,@K[2]+ cmp $len,#64+ add $xd0,$xd2,@K[3]+ b.lo .Last_neon++ ld1.8 {$xt0-$xt3},[$inp],#64+ eor $xa2,$xa0,$xt0+ eor $xb2,$xb0,$xt1+ eor $xc2,$xc0,$xt2+ eor $xd2,$xd0,$xt3+ st1.8 {$xa2-$xd2},[$out],#64+ b.eq .Ldone_neon++ add $xa0,$xa3,@K[0]+ add $xb0,$xb3,@K[1]+ add $xc0,$xc3,@K[2]+ add $xd0,$xd3,@K[3]+ sub $len,$len,#64++.Last_neon:+ st1.8 {$xa0-$xd0},[sp] // off-load complete block++ csub $out,$out,#1+ cadd $inp,$inp,$len+ cadd $out,$out,$len+ cadd $ctr,sp,$len+ neg $len,$len++.Loop_tail_neon:+ ldrb w10,[$inp,$len]+ ldrb w11,[$ctr,$len]+ add $len,$len,#1+ eor w10,w10,w11+ strb w10,[$out,$len]+ cbnz $len,.Loop_tail_neon++ stp @K[0],@K[0],[sp,#0] // wipe off-load area+ stp @K[0],@K[0],[sp,#32] // [with known constant]++.Ldone_neon:+ eor @K[1],@K[1],@K[1] // cleanse key and nonce+ eor @K[2],@K[2],@K[2]+ eor @K[3],@K[3],@K[3]++ ldp c19,c20,[c29,#2*__SIZEOF_POINTER__]+ add csp,csp,#64+ ldp c21,c22,[c29,#4*__SIZEOF_POINTER__]+ ldp c23,c24,[c29,#6*__SIZEOF_POINTER__]+ ldp c25,c26,[c29,#8*__SIZEOF_POINTER__]+ ldp c27,c28,[c29,#10*__SIZEOF_POINTER__]+ ldp c29,c30,[csp],#12*__SIZEOF_POINTER__+ .inst 0xd50323bf // autiasp+ ret++.align 4+.Less_than_64_neon:+ eor @K[1],@K[1],@K[1] // cleanse key and nonce+ eor @K[2],@K[2],@K[2]+ eor @K[3],@K[3],@K[3]+ b .Less_than_64+.size ChaCha20_neon,.-ChaCha20_neon+___+{+########################################################################+# While "vertical" layout minimizes total amount of instructions, number+# of blocks processed in parallel is limited to 4x. And trouble is that+# if NEON instructions are high-latency enough, algorithmic dependencies+# will manifest themselves as idle/wasted cycles. 6x"horizontal" avoids+# these gaps and achieves better performance. Since NEON instruction+# sequence is >2x longer, it's possible to slip in two additional blocks+# processed with scalar code path at no additional cost. Hence the "6+2"+# description...++my @K = map("v$_.4s",(0..6));+my ($T0,$T1,$T2,$T3,$T4,$T5)=@K;+my ($A0,$B0,$C0,$D0,$A1,$B1,$C1,$D1,$A2,$B2,$C2,$D2,+ $A3,$B3,$C3,$D3,$A4,$B4,$C4,$D4,$A5,$B5,$C5,$D5) = map("v$_.4s",(8..31));+my $rot24 = @K[6];+my $ONE = "v7.4s";++sub NEONROUND {+my $odd = pop;+my ($a,$b,$c,$d,$t)=@_;++ (+ "&add ('$a','$a','$b')",+ "&eor ('$d','$d','$a')",+ "&rev32_16 ('$d','$d')", # vrot ($d,16)++ "&add ('$c','$c','$d')",+ "&eor ('$t','$b','$c')",+ "&ushr ('$b','$t',20)",+ "&sli ('$b','$t',12)",++ "&add ('$a','$a','$b')",+ "&eor ('$d','$d','$a')",+ "&tbl ('$d','{$d}','$rot24')",++ "&add ('$c','$c','$d')",+ "&eor ('$t','$b','$c')",+ "&ushr ('$b','$t',25)",+ "&sli ('$b','$t',7)",++ "&ext ('$c','$c','$c',8)",+ "&ext ('$d','$d','$d',$odd?4:12)",+ "&ext ('$b','$b','$b',$odd?12:4)"+ );+}++$code.=<<___;+.type ChaCha20_512_neon,%function+.align 5+ChaCha20_512_neon:+ .inst 0xd503233f // paciasp+ stp c29,c30,[csp,#-12*__SIZEOF_POINTER__]!+ add c29,csp,#0++ adr @x[0],.Lsigma+ stp c19,c20,[csp,#2*__SIZEOF_POINTER__]+ stp c21,c22,[csp,#4*__SIZEOF_POINTER__]+ stp c23,c24,[csp,#6*__SIZEOF_POINTER__]+ stp c25,c26,[csp,#8*__SIZEOF_POINTER__]+ stp c27,c28,[csp,#10*__SIZEOF_POINTER__]++.L512_or_more_neon:+ sub csp,csp,#128+64++ eor $ONE,$ONE,$ONE+ ldp @d[0],@d[1],[@x[0]] // load sigma+ ld1 {@K[0]},[@x[0]],#16+ ldp @d[2],@d[3],[$key] // load key+ ldp @d[4],@d[5],[$key,#16]+ ld1 {@K[1],@K[2]},[$key]+ ldp @d[6],@d[7],[$ctr] // load counter+ ld1 {@K[3]},[$ctr]+ ld1 {$ONE}[0],[@x[0]]+ cadd $key,@x[0],#16 // .Lrot24+#ifdef __AARCH64EB__+ rev64 @K[0],@K[0]+ ror @d[2],@d[2],#32+ ror @d[3],@d[3],#32+ ror @d[4],@d[4],#32+ ror @d[5],@d[5],#32+ ror @d[6],@d[6],#32+ ror @d[7],@d[7],#32+#endif+ add @K[3],@K[3],$ONE // += 1+ stp @K[0],@K[1],[sp,#0] // off-load key block, invariant part+ add @K[3],@K[3],$ONE // not typo+ str @K[2],[sp,#32]+ add @K[4],@K[3],$ONE+ add @K[5],@K[4],$ONE+ add @K[6],@K[5],$ONE+ shl $ONE,$ONE,#2 // 1 -> 4++ stp d8,d9,[sp,#128+0] // meet ABI requirements+ stp d10,d11,[sp,#128+16]+ stp d12,d13,[sp,#128+32]+ stp d14,d15,[sp,#128+48]++ sub $len,$len,#512 // not typo++.Loop_outer_512_neon:+ mov $A0,@K[0]+ mov $A1,@K[0]+ mov $A2,@K[0]+ mov $A3,@K[0]+ mov $A4,@K[0]+ mov $A5,@K[0]+ mov $B0,@K[1]+ mov.32 @x[0],@d[0] // unpack key block+ mov $B1,@K[1]+ lsr @x[1],@d[0],#32+ mov $B2,@K[1]+ mov.32 @x[2],@d[1]+ mov $B3,@K[1]+ lsr @x[3],@d[1],#32+ mov $B4,@K[1]+ mov.32 @x[4],@d[2]+ mov $B5,@K[1]+ lsr @x[5],@d[2],#32+ mov $D0,@K[3]+ mov.32 @x[6],@d[3]+ mov $D1,@K[4]+ lsr @x[7],@d[3],#32+ mov $D2,@K[5]+ mov.32 @x[8],@d[4]+ mov $D3,@K[6]+ lsr @x[9],@d[4],#32+ mov $C0,@K[2]+ mov.32 @x[10],@d[5]+ mov $C1,@K[2]+ lsr @x[11],@d[5],#32+ add $D4,$D0,$ONE // +4+ mov.32 @x[12],@d[6]+ add $D5,$D1,$ONE // +4+ lsr @x[13],@d[6],#32+ mov $C2,@K[2]+ mov.32 @x[14],@d[7]+ mov $C3,@K[2]+ lsr @x[15],@d[7],#32+ mov $C4,@K[2]+ stp @K[3],@K[4],[sp,#48] // off-load key block, variable part+ mov $C5,@K[2]+ stp @K[5],@K[6],[sp,#80]++ mov $ctr,#5+ ld1 {$rot24},[$key]+ subs $len,$len,#512+.Loop_upper_neon:+ sub $ctr,$ctr,#1+___+ my @thread0=&NEONROUND($A0,$B0,$C0,$D0,$T0,0);+ my @thread1=&NEONROUND($A1,$B1,$C1,$D1,$T1,0);+ my @thread2=&NEONROUND($A2,$B2,$C2,$D2,$T2,0);+ my @thread3=&NEONROUND($A3,$B3,$C3,$D3,$T3,0);+ my @thread4=&NEONROUND($A4,$B4,$C4,$D4,$T4,0);+ my @thread5=&NEONROUND($A5,$B5,$C5,$D5,$T5,0);+ my @thread67=(&ROUND(0,4,8,12),&ROUND(0,5,10,15));+ my $diff = ($#thread0+1)*6 - $#thread67 - 1;+ my $i = 0;++ foreach (@thread0) {+ eval; eval(shift(@thread67));+ eval(shift(@thread1)); eval(shift(@thread67));+ eval(shift(@thread2)); eval(shift(@thread67));+ eval(shift(@thread3)); eval(shift(@thread67));+ eval(shift(@thread4)); eval(shift(@thread67));+ eval(shift(@thread5)); eval(shift(@thread67));+ }++ @thread0=&NEONROUND($A0,$B0,$C0,$D0,$T0,1);+ @thread1=&NEONROUND($A1,$B1,$C1,$D1,$T1,1);+ @thread2=&NEONROUND($A2,$B2,$C2,$D2,$T2,1);+ @thread3=&NEONROUND($A3,$B3,$C3,$D3,$T3,1);+ @thread4=&NEONROUND($A4,$B4,$C4,$D4,$T4,1);+ @thread5=&NEONROUND($A5,$B5,$C5,$D5,$T5,1);+ @thread67=(&ROUND(0,4,8,12),&ROUND(0,5,10,15));++ foreach (@thread0) {+ eval; eval(shift(@thread67));+ eval(shift(@thread1)); eval(shift(@thread67));+ eval(shift(@thread2)); eval(shift(@thread67));+ eval(shift(@thread3)); eval(shift(@thread67));+ eval(shift(@thread4)); eval(shift(@thread67));+ eval(shift(@thread5)); eval(shift(@thread67));+ }+$code.=<<___;+ cbnz $ctr,.Loop_upper_neon++ add.32 @x[0],@x[0],@d[0] // accumulate key block+ add @x[1],@x[1],@d[0],lsr#32+ add.32 @x[2],@x[2],@d[1]+ add @x[3],@x[3],@d[1],lsr#32+ add.32 @x[4],@x[4],@d[2]+ add @x[5],@x[5],@d[2],lsr#32+ add.32 @x[6],@x[6],@d[3]+ add @x[7],@x[7],@d[3],lsr#32+ add.32 @x[8],@x[8],@d[4]+ add @x[9],@x[9],@d[4],lsr#32+ add.32 @x[10],@x[10],@d[5]+ add @x[11],@x[11],@d[5],lsr#32+ add.32 @x[12],@x[12],@d[6]+ add @x[13],@x[13],@d[6],lsr#32+ add.32 @x[14],@x[14],@d[7]+ add @x[15],@x[15],@d[7],lsr#32++ add @x[0],@x[0],@x[1],lsl#32 // pack+ add @x[2],@x[2],@x[3],lsl#32+ ldp @x[1],@x[3],[$inp,#0] // load input+ add @x[4],@x[4],@x[5],lsl#32+ add @x[6],@x[6],@x[7],lsl#32+ ldp @x[5],@x[7],[$inp,#16]+ add @x[8],@x[8],@x[9],lsl#32+ add @x[10],@x[10],@x[11],lsl#32+ ldp @x[9],@x[11],[$inp,#32]+ add @x[12],@x[12],@x[13],lsl#32+ add @x[14],@x[14],@x[15],lsl#32+ ldp @x[13],@x[15],[$inp,#48]+ cadd $inp,$inp,#64+#ifdef __AARCH64EB__+ rev @x[0],@x[0]+ rev @x[2],@x[2]+ rev @x[4],@x[4]+ rev @x[6],@x[6]+ rev @x[8],@x[8]+ rev @x[10],@x[10]+ rev @x[12],@x[12]+ rev @x[14],@x[14]+#endif+ eor @x[0],@x[0],@x[1]+ eor @x[2],@x[2],@x[3]+ eor @x[4],@x[4],@x[5]+ eor @x[6],@x[6],@x[7]+ eor @x[8],@x[8],@x[9]+ eor @x[10],@x[10],@x[11]+ eor @x[12],@x[12],@x[13]+ eor @x[14],@x[14],@x[15]++ stp @x[0],@x[2],[$out,#0] // store output+ add @d[6],@d[6],#1 // increment counter+ mov.32 @x[0],@d[0] // unpack key block+ lsr @x[1],@d[0],#32+ stp @x[4],@x[6],[$out,#16]+ mov.32 @x[2],@d[1]+ lsr @x[3],@d[1],#32+ stp @x[8],@x[10],[$out,#32]+ mov.32 @x[4],@d[2]+ lsr @x[5],@d[2],#32+ stp @x[12],@x[14],[$out,#48]+ cadd $out,$out,#64+ mov.32 @x[6],@d[3]+ lsr @x[7],@d[3],#32+ mov.32 @x[8],@d[4]+ lsr @x[9],@d[4],#32+ mov.32 @x[10],@d[5]+ lsr @x[11],@d[5],#32+ mov.32 @x[12],@d[6]+ lsr @x[13],@d[6],#32+ mov.32 @x[14],@d[7]+ lsr @x[15],@d[7],#32++ mov $ctr,#5+.Loop_lower_neon:+ sub $ctr,$ctr,#1+___+ @thread0=&NEONROUND($A0,$B0,$C0,$D0,$T0,0);+ @thread1=&NEONROUND($A1,$B1,$C1,$D1,$T1,0);+ @thread2=&NEONROUND($A2,$B2,$C2,$D2,$T2,0);+ @thread3=&NEONROUND($A3,$B3,$C3,$D3,$T3,0);+ @thread4=&NEONROUND($A4,$B4,$C4,$D4,$T4,0);+ @thread5=&NEONROUND($A5,$B5,$C5,$D5,$T5,0);+ @thread67=(&ROUND(0,4,8,12),&ROUND(0,5,10,15));++ foreach (@thread0) {+ eval; eval(shift(@thread67));+ eval(shift(@thread1)); eval(shift(@thread67));+ eval(shift(@thread2)); eval(shift(@thread67));+ eval(shift(@thread3)); eval(shift(@thread67));+ eval(shift(@thread4)); eval(shift(@thread67));+ eval(shift(@thread5)); eval(shift(@thread67));+ }++ @thread0=&NEONROUND($A0,$B0,$C0,$D0,$T0,1);+ @thread1=&NEONROUND($A1,$B1,$C1,$D1,$T1,1);+ @thread2=&NEONROUND($A2,$B2,$C2,$D2,$T2,1);+ @thread3=&NEONROUND($A3,$B3,$C3,$D3,$T3,1);+ @thread4=&NEONROUND($A4,$B4,$C4,$D4,$T4,1);+ @thread5=&NEONROUND($A5,$B5,$C5,$D5,$T5,1);+ @thread67=(&ROUND(0,4,8,12),&ROUND(0,5,10,15));++ foreach (@thread0) {+ eval; eval(shift(@thread67));+ eval(shift(@thread1)); eval(shift(@thread67));+ eval(shift(@thread2)); eval(shift(@thread67));+ eval(shift(@thread3)); eval(shift(@thread67));+ eval(shift(@thread4)); eval(shift(@thread67));+ eval(shift(@thread5)); eval(shift(@thread67));+ }+$code.=<<___;+ cbnz $ctr,.Loop_lower_neon++ add.32 @x[0],@x[0],@d[0] // accumulate key block+ ldp @K[0],@K[1],[sp,#0]+ add @x[1],@x[1],@d[0],lsr#32+ ldp @K[2],@K[3],[sp,#32]+ add.32 @x[2],@x[2],@d[1]+ ldp @K[4],@K[5],[sp,#64]+ add @x[3],@x[3],@d[1],lsr#32+ ldr @K[6],[sp,#96]+ add $A0,$A0,@K[0]+ add.32 @x[4],@x[4],@d[2]+ add $A1,$A1,@K[0]+ add @x[5],@x[5],@d[2],lsr#32+ add $A2,$A2,@K[0]+ add.32 @x[6],@x[6],@d[3]+ add $A3,$A3,@K[0]+ add @x[7],@x[7],@d[3],lsr#32+ add $A4,$A4,@K[0]+ add.32 @x[8],@x[8],@d[4]+ add $A5,$A5,@K[0]+ add @x[9],@x[9],@d[4],lsr#32+ add $C0,$C0,@K[2]+ add.32 @x[10],@x[10],@d[5]+ add $C1,$C1,@K[2]+ add @x[11],@x[11],@d[5],lsr#32+ add $C2,$C2,@K[2]+ add.32 @x[12],@x[12],@d[6]+ add $C3,$C3,@K[2]+ add @x[13],@x[13],@d[6],lsr#32+ add $C4,$C4,@K[2]+ add.32 @x[14],@x[14],@d[7]+ add $C5,$C5,@K[2]+ add @x[15],@x[15],@d[7],lsr#32+ add $D4,$D4,$ONE // +4+ add @x[0],@x[0],@x[1],lsl#32 // pack+ add $D5,$D5,$ONE // +4+ add @x[2],@x[2],@x[3],lsl#32+ add $D0,$D0,@K[3]+ ldp @x[1],@x[3],[$inp,#0] // load input+ add $D1,$D1,@K[4]+ add @x[4],@x[4],@x[5],lsl#32+ add $D2,$D2,@K[5]+ add @x[6],@x[6],@x[7],lsl#32+ add $D3,$D3,@K[6]+ ldp @x[5],@x[7],[$inp,#16]+ add $D4,$D4,@K[3]+ add @x[8],@x[8],@x[9],lsl#32+ add $D5,$D5,@K[4]+ add @x[10],@x[10],@x[11],lsl#32+ add $B0,$B0,@K[1]+ ldp @x[9],@x[11],[$inp,#32]+ add $B1,$B1,@K[1]+ add @x[12],@x[12],@x[13],lsl#32+ add $B2,$B2,@K[1]+ add @x[14],@x[14],@x[15],lsl#32+ add $B3,$B3,@K[1]+ ldp @x[13],@x[15],[$inp,#48]+ add $B4,$B4,@K[1]+ cadd $inp,$inp,#64+ add $B5,$B5,@K[1]++#ifdef __AARCH64EB__+ rev @x[0],@x[0]+ rev @x[2],@x[2]+ rev @x[4],@x[4]+ rev @x[6],@x[6]+ rev @x[8],@x[8]+ rev @x[10],@x[10]+ rev @x[12],@x[12]+ rev @x[14],@x[14]+#endif+ ld1.8 {$T0-$T3},[$inp],#64+ eor @x[0],@x[0],@x[1]+ eor @x[2],@x[2],@x[3]+ eor @x[4],@x[4],@x[5]+ eor @x[6],@x[6],@x[7]+ eor @x[8],@x[8],@x[9]+ eor $A0,$A0,$T0+ eor @x[10],@x[10],@x[11]+ eor $B0,$B0,$T1+ eor @x[12],@x[12],@x[13]+ eor $C0,$C0,$T2+ eor @x[14],@x[14],@x[15]+ eor $D0,$D0,$T3+ ld1.8 {$T0-$T3},[$inp],#64++ stp @x[0],@x[2],[$out,#0] // store output+ add @d[6],@d[6],#7 // increment counter+ stp @x[4],@x[6],[$out,#16]+ stp @x[8],@x[10],[$out,#32]+ stp @x[12],@x[14],[$out,#48]+ cadd $out,$out,#64+ st1.8 {$A0-$D0},[$out],#64++ ld1.8 {$A0-$D0},[$inp],#64+ eor $A1,$A1,$T0+ eor $B1,$B1,$T1+ eor $C1,$C1,$T2+ eor $D1,$D1,$T3+ st1.8 {$A1-$D1},[$out],#64++ ld1.8 {$A1-$D1},[$inp],#64+ eor $A2,$A2,$A0+ ldp @K[0],@K[1],[sp,#0]+ eor $B2,$B2,$B0+ ldp @K[2],@K[3],[sp,#32]+ eor $C2,$C2,$C0+ eor $D2,$D2,$D0+ st1.8 {$A2-$D2},[$out],#64++ ld1.8 {$A2-$D2},[$inp],#64+ eor $A3,$A3,$A1+ eor $B3,$B3,$B1+ eor $C3,$C3,$C1+ eor $D3,$D3,$D1+ st1.8 {$A3-$D3},[$out],#64++ ld1.8 {$A3-$D3},[$inp],#64+ eor $A4,$A4,$A2+ eor $B4,$B4,$B2+ eor $C4,$C4,$C2+ eor $D4,$D4,$D2+ st1.8 {$A4-$D4},[$out],#64++ shl $A0,$ONE,#1 // 4 -> 8+ eor $A5,$A5,$A3+ eor $B5,$B5,$B3+ eor $C5,$C5,$C3+ eor $D5,$D5,$D3+ st1.8 {$A5-$D5},[$out],#64++ add @K[3],@K[3],$A0 // += 8+ add @K[4],@K[4],$A0+ add @K[5],@K[5],$A0+ add @K[6],@K[6],$A0++ b.hs .Loop_outer_512_neon++ adds $len,$len,#512+ ushr $ONE,$ONE,#1 // 4 -> 2++ ldp d10,d11,[sp,#128+16] // meet ABI requirements+ ldp d12,d13,[sp,#128+32]+ ldp d14,d15,[sp,#128+48]++ stp @K[0],@K[0],[sp,#16] // wipe key off-load area+ stp @K[0],@K[0],[sp,#48] // [with known constant]+ stp @K[0],@K[0],[sp,#80]++ b.eq .Ldone_512_neon++ // we have <512 bytes tail, harmonize state with other contexts+ csub $key,$key,#16 // .Lone+ cmp $len,#192+ cadd sp,sp,#128+ sub @K[3],@K[3],$ONE // -= 2+ ld1 {$CTR,$ROT24},[$key]+ b.hs .Loop_outer_neon++ ldp d8,d9,[sp,#0] // meet ABI requirements+ eor @K[1],@K[1],@K[1] // cleanse key and nonce+ eor @K[2],@K[2],@K[2]+ eor @K[3],@K[3],@K[3]+ eor @K[4],@K[4],@K[4]+ eor @K[5],@K[5],@K[5]+ eor @K[6],@K[6],@K[6]+ b .Loop_outer++.Ldone_512_neon:+ ldp d8,d9,[sp,#128+0] // meet ABI requirements+ eor @K[1],@K[1],@K[1] // cleanse key and nonce+ eor @K[2],@K[2],@K[2]+ eor @K[3],@K[3],@K[3]+ eor @K[4],@K[4],@K[4]+ eor @K[5],@K[5],@K[5]+ eor @K[6],@K[6],@K[6]++ ldp c19,c20,[c29,#2*__SIZEOF_POINTER__]+ add csp,csp,#128+64+ ldp c21,c22,[c29,#4*__SIZEOF_POINTER__]+ ldp c23,c24,[c29,#6*__SIZEOF_POINTER__]+ ldp c25,c26,[c29,#8*__SIZEOF_POINTER__]+ ldp c27,c28,[c29,#10*__SIZEOF_POINTER__]+ ldp c29,c30,[csp],#12*__SIZEOF_POINTER__+ .inst 0xd50323bf // autiasp+ ret+.size ChaCha20_512_neon,.-ChaCha20_512_neon+___+}+}}}++foreach (split("\n",$code)) {+ s/\`([^\`]*)\`/eval $1/geo;++ (s/\b([a-z]+)\.32\b/$1/ and (s/x([0-9]+)/w$1/g or 1)) or+ (m/\b(eor|ext|mov|tbl)\b/ and (s/\.4s/\.16b/g or 1)) or+ (s/\b((?:ld|st)1)\.8\b/$1/ and (s/\.4s/\.16b/g or 1)) or+ (m/\b(ld|st)[rp]\b/ and (s/v([0-9]+)\.4s/q$1/g or 1)) or+ (m/\b(dup|ld1)\b/ and (s/\.4(s}?\[[0-3]\])/.$1/g or 1)) or+ (s/\b(zip[12])\.64\b/$1/ and (s/\.4s/\.2d/g or 1)) or+ (s/\brev32\.16\b/rev32/ and (s/\.4s/\.8h/g or 1));++ #s/\bq([0-9]+)#(lo|hi)/sprintf "d%d",2*$1+($2 eq "hi")/geo;++ print $_,"\n";+}+close STDOUT; # flush
@@ -0,0 +1,2241 @@+.text ++++.align 64+.Lzero:+.long 0,0,0,0+.Lone:+.long 1,0,0,0+.Linc:+.long 0,1,2,3+.Lfour:+.long 4,4,4,4+.Lincy:+.long 0,2,4,6,1,3,5,7+.Leight:+.long 8,8,8,8,8,8,8,8+.Lrot16:+.byte 0x2,0x3,0x0,0x1, 0x6,0x7,0x4,0x5, 0xa,0xb,0x8,0x9, 0xe,0xf,0xc,0xd+.Lrot24:+.byte 0x3,0x0,0x1,0x2, 0x7,0x4,0x5,0x6, 0xb,0x8,0x9,0xa, 0xf,0xc,0xd,0xe+.Ltwoy:+.long 2,0,0,0, 2,0,0,0+.align 64+.Lzeroz:+.long 0,0,0,0, 1,0,0,0, 2,0,0,0, 3,0,0,0+.Lfourz:+.long 4,0,0,0, 4,0,0,0, 4,0,0,0, 4,0,0,0+.Lincz:+.long 0,1,2,3,4,5,6,7,8,9,10,11,12,13,14,15+.Lsixteen:+.long 16,16,16,16,16,16,16,16,16,16,16,16,16,16,16,16+.Lsigma:+.byte 101,120,112,97,110,100,32,51,50,45,98,121,116,101,32,107,0+.byte 67,104,97,67,104,97,50,48,32,102,111,114,32,120,56,54,95,54,52,44,32,67,82,89,80,84,79,71,65,77,83,32,98,121,32,64,100,111,116,45,97,115,109,0+.globl crypton_chacha20_asm_ctr32+.type crypton_chacha20_asm_ctr32,@function+.align 64+crypton_chacha20_asm_ctr32:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa+++ cmpq $0,%rdx+ je .Lno_data+ movq crypton_ia32cap_P+4(%rip),%r9+ testl $512,%r9d+ jnz .Lcrypton_chacha20_asm_ssse3+ pushq %rbx+.cfi_adjust_cfa_offset 8+.cfi_offset %rbx,-16+ pushq %rbp+.cfi_adjust_cfa_offset 8+.cfi_offset %rbp,-24+ pushq %r12+.cfi_adjust_cfa_offset 8+.cfi_offset %r12,-32+ pushq %r13+.cfi_adjust_cfa_offset 8+.cfi_offset %r13,-40+ pushq %r14+.cfi_adjust_cfa_offset 8+.cfi_offset %r14,-48+ pushq %r15+.cfi_adjust_cfa_offset 8+.cfi_offset %r15,-56+ subq $64+24,%rsp+.cfi_adjust_cfa_offset 88+.Lctr32_body:++ movq %rdx,%rbp++ movq 0(%rcx),%r12+ movq 8(%rcx),%r13+ movq 16(%rcx),%r14+ movq 24(%rcx),%r15+ movq 0(%r8),%rax+ movq 8(%r8),%rdx+ movq %r12,16(%rsp)+ movq %r13,24(%rsp)+ movq %r14,0(%rsp)+ movq %r15,8(%rsp)+ movq %rax,48(%rsp)+ movq %rdx,56(%rsp)+ jmp .Loop_outer++.align 32+.Loop_outer:+ movl $0x61707865,%eax+ movl $0x3320646e,%ebx+ movl $0x79622d32,%ecx+ movl $0x6b206574,%edx+ movl 16(%rsp),%r8d+ movl 20(%rsp),%r9d+ movl 24(%rsp),%r10d+ movl 28(%rsp),%r11d+ movl 48(%rsp),%r12d+ movl 52(%rsp),%r13d+ movl 56(%rsp),%r14d+ movq %r15,40(%rsp)+ movl 60(%rsp),%r15d++ movq %rbp,64+0(%rsp)+ movq %rsi,64+8(%rsp)+ movl 0(%rsp),%esi+ movq %rdi,64+16(%rsp)+ movl 4(%rsp),%edi+ movl $10,%ebp+ jmp .Loop++.align 32+.Loop:+ addl %r8d,%eax+ xorl %eax,%r12d+ roll $16,%r12d+ addl %r9d,%ebx+ xorl %ebx,%r13d+ roll $16,%r13d+ addl %r12d,%esi+ xorl %esi,%r8d+ roll $12,%r8d+ addl %r13d,%edi+ xorl %edi,%r9d+ roll $12,%r9d+ addl %r8d,%eax+ xorl %eax,%r12d+ roll $8,%r12d+ addl %r9d,%ebx+ xorl %ebx,%r13d+ roll $8,%r13d+ addl %r12d,%esi+ xorl %esi,%r8d+ roll $7,%r8d+ addl %r13d,%edi+ xorl %edi,%r9d+ roll $7,%r9d+ movl %esi,32(%rsp)+ movl %edi,36(%rsp)+ movl 40(%rsp),%esi+ movl 44(%rsp),%edi+ addl %r10d,%ecx+ xorl %ecx,%r14d+ roll $16,%r14d+ addl %r11d,%edx+ xorl %edx,%r15d+ roll $16,%r15d+ addl %r14d,%esi+ xorl %esi,%r10d+ roll $12,%r10d+ addl %r15d,%edi+ xorl %edi,%r11d+ roll $12,%r11d+ addl %r10d,%ecx+ xorl %ecx,%r14d+ roll $8,%r14d+ addl %r11d,%edx+ xorl %edx,%r15d+ roll $8,%r15d+ addl %r14d,%esi+ xorl %esi,%r10d+ roll $7,%r10d+ addl %r15d,%edi+ xorl %edi,%r11d+ roll $7,%r11d+ addl %r9d,%eax+ xorl %eax,%r15d+ roll $16,%r15d+ addl %r10d,%ebx+ xorl %ebx,%r12d+ roll $16,%r12d+ addl %r15d,%esi+ xorl %esi,%r9d+ roll $12,%r9d+ addl %r12d,%edi+ xorl %edi,%r10d+ roll $12,%r10d+ addl %r9d,%eax+ xorl %eax,%r15d+ roll $8,%r15d+ addl %r10d,%ebx+ xorl %ebx,%r12d+ roll $8,%r12d+ addl %r15d,%esi+ xorl %esi,%r9d+ roll $7,%r9d+ addl %r12d,%edi+ xorl %edi,%r10d+ roll $7,%r10d+ movl %esi,40(%rsp)+ movl %edi,44(%rsp)+ movl 32(%rsp),%esi+ movl 36(%rsp),%edi+ addl %r11d,%ecx+ xorl %ecx,%r13d+ roll $16,%r13d+ addl %r8d,%edx+ xorl %edx,%r14d+ roll $16,%r14d+ addl %r13d,%esi+ xorl %esi,%r11d+ roll $12,%r11d+ addl %r14d,%edi+ xorl %edi,%r8d+ roll $12,%r8d+ addl %r11d,%ecx+ xorl %ecx,%r13d+ roll $8,%r13d+ addl %r8d,%edx+ xorl %edx,%r14d+ roll $8,%r14d+ addl %r13d,%esi+ xorl %esi,%r11d+ roll $7,%r11d+ addl %r14d,%edi+ xorl %edi,%r8d+ roll $7,%r8d+ decl %ebp+ jnz .Loop+ addl 0(%rsp),%esi+ addl 4(%rsp),%edi+ movq 64(%rsp),%rbp+ movl %esi,32(%rsp)+ movq 64+8(%rsp),%rsi+ movl %edi,36(%rsp)+ movq 64+16(%rsp),%rdi++ addl $0x61707865,%eax+ addl $0x3320646e,%ebx+ addl $0x79622d32,%ecx+ addl $0x6b206574,%edx+ addl 16(%rsp),%r8d+ addl 20(%rsp),%r9d+ addl 24(%rsp),%r10d+ addl 28(%rsp),%r11d+ addl 48(%rsp),%r12d+ addl 52(%rsp),%r13d+ addl 56(%rsp),%r14d+ addl 60(%rsp),%r15d++ cmpq $64,%rbp+ jb .Ltail++ xorl 0(%rsi),%eax+ xorl 4(%rsi),%ebx+ xorl 8(%rsi),%ecx+ xorl 12(%rsi),%edx+ movl %eax,0(%rdi)+ movl 32(%rsp),%eax+ movl %ebx,4(%rdi)+ movl 36(%rsp),%ebx+ movl %ecx,8(%rdi)+ movl 40(%rsp),%ecx+ movl %edx,12(%rdi)+ movl 44(%rsp),%edx+ xorl 16(%rsi),%r8d+ addl 8(%rsp),%ecx+ xorl 20(%rsi),%r9d+ addl 12(%rsp),%edx+ xorl 24(%rsi),%r10d+ xorl 28(%rsi),%r11d+ xorl 32(%rsi),%eax+ xorl 36(%rsi),%ebx+ xorl 40(%rsi),%ecx+ xorl 44(%rsi),%edx+ xorl 48(%rsi),%r12d+ xorl 52(%rsi),%r13d+ xorl 56(%rsi),%r14d+ xorl 60(%rsi),%r15d+ leaq 64(%rsi),%rsi++ addl $1,48(%rsp)++ movl %r8d,16(%rdi)+ movl %r9d,20(%rdi)+ movl %r10d,24(%rdi)+ movl %r11d,28(%rdi)+ movl %eax,32(%rdi)+ movl %ebx,36(%rdi)+ movl %ecx,40(%rdi)+ movl %edx,44(%rdi)+ movl %r12d,48(%rdi)+ movl %r13d,52(%rdi)+ movl %r14d,56(%rdi)+ movl %r15d,60(%rdi)+ leaq 64(%rdi),%rdi+ movq 8(%rsp),%r15++ subq $64,%rbp+ jnz .Loop_outer++ jmp .Ldone++.align 16+.Ltail:+ movl %eax,0(%rsp)+ movl 8(%rsp),%eax+ movl %ebx,4(%rsp)+ movl 12(%rsp),%ebx+ movl %ecx,8(%rsp)+ addl 40(%rsp),%eax+ movl %edx,12(%rsp)+ addl 44(%rsp),%ebx+ movl %r8d,16(%rsp)+ movl %r9d,20(%rsp)+ movl %r10d,24(%rsp)+ movl %r11d,28(%rsp)+ movl %eax,40(%rsp)+ movl %ebx,44(%rsp)+ xorq %rbx,%rbx+ movl %r12d,48(%rsp)+ movl %r13d,52(%rsp)+ movl %r14d,56(%rsp)+ movl %r15d,60(%rsp)++.Loop_tail:+ movzbl (%rsi,%rbx,1),%eax+ movzbl (%rsp,%rbx,1),%edx+ leaq 1(%rbx),%rbx+ xorl %edx,%eax+ movb %al,-1(%rdi,%rbx,1)+ decq %rbp+ jnz .Loop_tail++.Ldone:+ leaq 64+24+48(%rsp),%rsi+.cfi_def_cfa %rsi,8+ movq -48(%rsi),%r15+.cfi_restore %r15+ movq -40(%rsi),%r14+.cfi_restore %r14+ movq -32(%rsi),%r13+.cfi_restore %r13+ movq -24(%rsi),%r12+.cfi_restore %r12+ movq -16(%rsi),%rbp+.cfi_restore %rbp+ movq -8(%rsi),%rbx+.cfi_restore %rbx+ leaq (%rsi),%rsp+.cfi_def_cfa_register %rsp+.Lno_data:+ .byte 0xf3,0xc3+.cfi_endproc +.size crypton_chacha20_asm_ctr32,.-crypton_chacha20_asm_ctr32+.type crypton_chacha20_asm_ssse3,@function+.align 32+crypton_chacha20_asm_ssse3:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa+++.Lcrypton_chacha20_asm_ssse3:+ movq %rsp,%r10+.cfi_def_cfa_register %r10+ testl $2048,%r9d+ jnz .Lcrypton_chacha20_asm_4xop+ cmpq $128,%rdx+ je .Lcrypton_chacha20_asm_128+ ja .Lcrypton_chacha20_asm_4x++.Ldo_sse3_after_all:+ subq $64+8,%rsp+ andq $-16,%rsp+ movdqa .Lsigma(%rip),%xmm0+ movdqu (%rcx),%xmm1+ movdqu 16(%rcx),%xmm2+ movdqu (%r8),%xmm3+ movdqa .Lrot16(%rip),%xmm6+ movdqa .Lrot24(%rip),%xmm7++ movdqa %xmm0,0(%rsp)+ movdqa %xmm1,16(%rsp)+ movdqa %xmm2,32(%rsp)+ movdqa %xmm3,48(%rsp)+ movq $10,%r8+ jmp .Loop_ssse3++.align 32+.Loop_outer_ssse3:+ movdqa .Lone(%rip),%xmm3+ movdqa 0(%rsp),%xmm0+ movdqa 16(%rsp),%xmm1+ movdqa 32(%rsp),%xmm2+ paddd 48(%rsp),%xmm3+ movq $10,%r8+ movdqa %xmm3,48(%rsp)+ jmp .Loop_ssse3++.align 32+.Loop_ssse3:+ paddd %xmm1,%xmm0+ pxor %xmm0,%xmm3+.byte 102,15,56,0,222+ paddd %xmm3,%xmm2+ pxor %xmm2,%xmm1+ movdqa %xmm1,%xmm4+ psrld $20,%xmm1+ pslld $12,%xmm4+ por %xmm4,%xmm1+ paddd %xmm1,%xmm0+ pxor %xmm0,%xmm3+.byte 102,15,56,0,223+ paddd %xmm3,%xmm2+ pxor %xmm2,%xmm1+ movdqa %xmm1,%xmm4+ psrld $25,%xmm1+ pslld $7,%xmm4+ por %xmm4,%xmm1+ pshufd $78,%xmm2,%xmm2+ pshufd $57,%xmm1,%xmm1+ pshufd $147,%xmm3,%xmm3+ nop+ paddd %xmm1,%xmm0+ pxor %xmm0,%xmm3+.byte 102,15,56,0,222+ paddd %xmm3,%xmm2+ pxor %xmm2,%xmm1+ movdqa %xmm1,%xmm4+ psrld $20,%xmm1+ pslld $12,%xmm4+ por %xmm4,%xmm1+ paddd %xmm1,%xmm0+ pxor %xmm0,%xmm3+.byte 102,15,56,0,223+ paddd %xmm3,%xmm2+ pxor %xmm2,%xmm1+ movdqa %xmm1,%xmm4+ psrld $25,%xmm1+ pslld $7,%xmm4+ por %xmm4,%xmm1+ pshufd $78,%xmm2,%xmm2+ pshufd $147,%xmm1,%xmm1+ pshufd $57,%xmm3,%xmm3+ decq %r8+ jnz .Loop_ssse3+ paddd 0(%rsp),%xmm0+ paddd 16(%rsp),%xmm1+ paddd 32(%rsp),%xmm2+ paddd 48(%rsp),%xmm3++ cmpq $64,%rdx+ jb .Ltail_ssse3++ movdqu 0(%rsi),%xmm4+ movdqu 16(%rsi),%xmm5+ pxor %xmm4,%xmm0+ movdqu 32(%rsi),%xmm4+ pxor %xmm5,%xmm1+ movdqu 48(%rsi),%xmm5+ leaq 64(%rsi),%rsi+ pxor %xmm4,%xmm2+ pxor %xmm5,%xmm3++ movdqu %xmm0,0(%rdi)+ movdqu %xmm1,16(%rdi)+ movdqu %xmm2,32(%rdi)+ movdqu %xmm3,48(%rdi)+ leaq 64(%rdi),%rdi++ subq $64,%rdx+ jnz .Loop_outer_ssse3++ jmp .Ldone_ssse3++.align 16+.Ltail_ssse3:+ movdqa %xmm0,0(%rsp)+ movdqa %xmm1,16(%rsp)+ movdqa %xmm2,32(%rsp)+ movdqa %xmm3,48(%rsp)+ xorq %r8,%r8++.Loop_tail_ssse3:+ movzbl (%rsi,%r8,1),%eax+ movzbl (%rsp,%r8,1),%ecx+ leaq 1(%r8),%r8+ xorl %ecx,%eax+ movb %al,-1(%rdi,%r8,1)+ decq %rdx+ jnz .Loop_tail_ssse3++.Ldone_ssse3:+ leaq (%r10),%rsp+.cfi_def_cfa_register %rsp+.Lssse3_epilogue:+ .byte 0xf3,0xc3+.cfi_endproc +.size crypton_chacha20_asm_ssse3,.-crypton_chacha20_asm_ssse3+.type crypton_chacha20_asm_128,@function+.align 32+crypton_chacha20_asm_128:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa+++.Lcrypton_chacha20_asm_128:+ movq %rsp,%r10+.cfi_def_cfa_register %r10+ subq $64+8,%rsp+ andq $-16,%rsp+ movdqa .Lsigma(%rip),%xmm8+ movdqu (%rcx),%xmm9+ movdqu 16(%rcx),%xmm2+ movdqu (%r8),%xmm3+ movdqa .Lone(%rip),%xmm1+ movdqa .Lrot16(%rip),%xmm6+ movdqa .Lrot24(%rip),%xmm7++ movdqa %xmm8,%xmm10+ movdqa %xmm8,0(%rsp)+ movdqa %xmm9,%xmm11+ movdqa %xmm9,16(%rsp)+ movdqa %xmm2,%xmm0+ movdqa %xmm2,32(%rsp)+ paddd %xmm3,%xmm1+ movdqa %xmm3,48(%rsp)+ movq $10,%r8+ jmp .Loop_128++.align 32+.Loop_128:+ paddd %xmm9,%xmm8+ pxor %xmm8,%xmm3+ paddd %xmm11,%xmm10+ pxor %xmm10,%xmm1+.byte 102,15,56,0,222+.byte 102,15,56,0,206+ paddd %xmm3,%xmm2+ paddd %xmm1,%xmm0+ pxor %xmm2,%xmm9+ pxor %xmm0,%xmm11+ movdqa %xmm9,%xmm4+ psrld $20,%xmm9+ movdqa %xmm11,%xmm5+ pslld $12,%xmm4+ psrld $20,%xmm11+ por %xmm4,%xmm9+ pslld $12,%xmm5+ por %xmm5,%xmm11+ paddd %xmm9,%xmm8+ pxor %xmm8,%xmm3+ paddd %xmm11,%xmm10+ pxor %xmm10,%xmm1+.byte 102,15,56,0,223+.byte 102,15,56,0,207+ paddd %xmm3,%xmm2+ paddd %xmm1,%xmm0+ pxor %xmm2,%xmm9+ pxor %xmm0,%xmm11+ movdqa %xmm9,%xmm4+ psrld $25,%xmm9+ movdqa %xmm11,%xmm5+ pslld $7,%xmm4+ psrld $25,%xmm11+ por %xmm4,%xmm9+ pslld $7,%xmm5+ por %xmm5,%xmm11+ pshufd $78,%xmm2,%xmm2+ pshufd $57,%xmm9,%xmm9+ pshufd $147,%xmm3,%xmm3+ pshufd $78,%xmm0,%xmm0+ pshufd $57,%xmm11,%xmm11+ pshufd $147,%xmm1,%xmm1+ paddd %xmm9,%xmm8+ pxor %xmm8,%xmm3+ paddd %xmm11,%xmm10+ pxor %xmm10,%xmm1+.byte 102,15,56,0,222+.byte 102,15,56,0,206+ paddd %xmm3,%xmm2+ paddd %xmm1,%xmm0+ pxor %xmm2,%xmm9+ pxor %xmm0,%xmm11+ movdqa %xmm9,%xmm4+ psrld $20,%xmm9+ movdqa %xmm11,%xmm5+ pslld $12,%xmm4+ psrld $20,%xmm11+ por %xmm4,%xmm9+ pslld $12,%xmm5+ por %xmm5,%xmm11+ paddd %xmm9,%xmm8+ pxor %xmm8,%xmm3+ paddd %xmm11,%xmm10+ pxor %xmm10,%xmm1+.byte 102,15,56,0,223+.byte 102,15,56,0,207+ paddd %xmm3,%xmm2+ paddd %xmm1,%xmm0+ pxor %xmm2,%xmm9+ pxor %xmm0,%xmm11+ movdqa %xmm9,%xmm4+ psrld $25,%xmm9+ movdqa %xmm11,%xmm5+ pslld $7,%xmm4+ psrld $25,%xmm11+ por %xmm4,%xmm9+ pslld $7,%xmm5+ por %xmm5,%xmm11+ pshufd $78,%xmm2,%xmm2+ pshufd $147,%xmm9,%xmm9+ pshufd $57,%xmm3,%xmm3+ pshufd $78,%xmm0,%xmm0+ pshufd $147,%xmm11,%xmm11+ pshufd $57,%xmm1,%xmm1+ decq %r8+ jnz .Loop_128+ paddd 0(%rsp),%xmm8+ paddd 16(%rsp),%xmm9+ paddd 32(%rsp),%xmm2+ paddd 48(%rsp),%xmm3+ paddd .Lone(%rip),%xmm1+ paddd 0(%rsp),%xmm10+ paddd 16(%rsp),%xmm11+ paddd 32(%rsp),%xmm0+ paddd 48(%rsp),%xmm1++ movdqu 0(%rsi),%xmm4+ movdqu 16(%rsi),%xmm5+ pxor %xmm4,%xmm8+ movdqu 32(%rsi),%xmm4+ pxor %xmm5,%xmm9+ movdqu 48(%rsi),%xmm5+ pxor %xmm4,%xmm2+ movdqu 64(%rsi),%xmm4+ pxor %xmm5,%xmm3+ movdqu 80(%rsi),%xmm5+ pxor %xmm4,%xmm10+ movdqu 96(%rsi),%xmm4+ pxor %xmm5,%xmm11+ movdqu 112(%rsi),%xmm5+ pxor %xmm4,%xmm0+ pxor %xmm5,%xmm1++ movdqu %xmm8,0(%rdi)+ movdqu %xmm9,16(%rdi)+ movdqu %xmm2,32(%rdi)+ movdqu %xmm3,48(%rdi)+ movdqu %xmm10,64(%rdi)+ movdqu %xmm11,80(%rdi)+ movdqu %xmm0,96(%rdi)+ movdqu %xmm1,112(%rdi)+ leaq (%r10),%rsp+.cfi_def_cfa_register %rsp+.L128_epilogue:+ .byte 0xf3,0xc3+.cfi_endproc +.size crypton_chacha20_asm_128,.-crypton_chacha20_asm_128+.type crypton_chacha20_asm_4x,@function+.align 32+crypton_chacha20_asm_4x:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa+++.Lcrypton_chacha20_asm_4x:+ movq %rsp,%r10+.cfi_def_cfa_register %r10+ movq %r9,%r11+ shrq $32,%r9+ testq $32,%r9+ jnz .Lcrypton_chacha20_asm_8x+ cmpq $192,%rdx+ ja .Lproceed4x++ andq $71303168,%r11+ cmpq $4194304,%r11+ je .Ldo_sse3_after_all++.Lproceed4x:+ subq $0x140+8,%rsp+ andq $-16,%rsp+ movdqa .Lsigma(%rip),%xmm11+ movdqu (%rcx),%xmm15+ movdqu 16(%rcx),%xmm7+ movdqu (%r8),%xmm3+ leaq 256(%rsp),%rcx+ leaq .Lrot16(%rip),%r9+ leaq .Lrot24(%rip),%r11++ pshufd $0x00,%xmm11,%xmm8+ pshufd $0x55,%xmm11,%xmm9+ movdqa %xmm8,64(%rsp)+ pshufd $0xaa,%xmm11,%xmm10+ movdqa %xmm9,80(%rsp)+ pshufd $0xff,%xmm11,%xmm11+ movdqa %xmm10,96(%rsp)+ movdqa %xmm11,112(%rsp)++ pshufd $0x00,%xmm15,%xmm12+ pshufd $0x55,%xmm15,%xmm13+ movdqa %xmm12,128-256(%rcx)+ pshufd $0xaa,%xmm15,%xmm14+ movdqa %xmm13,144-256(%rcx)+ pshufd $0xff,%xmm15,%xmm15+ movdqa %xmm14,160-256(%rcx)+ movdqa %xmm15,176-256(%rcx)++ pshufd $0x00,%xmm7,%xmm4+ pshufd $0x55,%xmm7,%xmm5+ movdqa %xmm4,192-256(%rcx)+ pshufd $0xaa,%xmm7,%xmm6+ movdqa %xmm5,208-256(%rcx)+ pshufd $0xff,%xmm7,%xmm7+ movdqa %xmm6,224-256(%rcx)+ movdqa %xmm7,240-256(%rcx)++ pshufd $0x00,%xmm3,%xmm0+ pshufd $0x55,%xmm3,%xmm1+ paddd .Linc(%rip),%xmm0+ pshufd $0xaa,%xmm3,%xmm2+ movdqa %xmm1,272-256(%rcx)+ pshufd $0xff,%xmm3,%xmm3+ movdqa %xmm2,288-256(%rcx)+ movdqa %xmm3,304-256(%rcx)++ jmp .Loop_enter4x++.align 32+.Loop_outer4x:+ movdqa 64(%rsp),%xmm8+ movdqa 80(%rsp),%xmm9+ movdqa 96(%rsp),%xmm10+ movdqa 112(%rsp),%xmm11+ movdqa 128-256(%rcx),%xmm12+ movdqa 144-256(%rcx),%xmm13+ movdqa 160-256(%rcx),%xmm14+ movdqa 176-256(%rcx),%xmm15+ movdqa 192-256(%rcx),%xmm4+ movdqa 208-256(%rcx),%xmm5+ movdqa 224-256(%rcx),%xmm6+ movdqa 240-256(%rcx),%xmm7+ movdqa 256-256(%rcx),%xmm0+ movdqa 272-256(%rcx),%xmm1+ movdqa 288-256(%rcx),%xmm2+ movdqa 304-256(%rcx),%xmm3+ paddd .Lfour(%rip),%xmm0++.Loop_enter4x:+ movdqa %xmm6,32(%rsp)+ movdqa %xmm7,48(%rsp)+ movdqa (%r9),%xmm7+ movl $10,%eax+ movdqa %xmm0,256-256(%rcx)+ jmp .Loop4x++.align 32+.Loop4x:+ paddd %xmm12,%xmm8+ paddd %xmm13,%xmm9+ pxor %xmm8,%xmm0+ pxor %xmm9,%xmm1+.byte 102,15,56,0,199+.byte 102,15,56,0,207+ paddd %xmm0,%xmm4+ paddd %xmm1,%xmm5+ pxor %xmm4,%xmm12+ pxor %xmm5,%xmm13+ movdqa %xmm12,%xmm6+ pslld $12,%xmm12+ psrld $20,%xmm6+ movdqa %xmm13,%xmm7+ pslld $12,%xmm13+ por %xmm6,%xmm12+ psrld $20,%xmm7+ movdqa (%r11),%xmm6+ por %xmm7,%xmm13+ paddd %xmm12,%xmm8+ paddd %xmm13,%xmm9+ pxor %xmm8,%xmm0+ pxor %xmm9,%xmm1+.byte 102,15,56,0,198+.byte 102,15,56,0,206+ paddd %xmm0,%xmm4+ paddd %xmm1,%xmm5+ pxor %xmm4,%xmm12+ pxor %xmm5,%xmm13+ movdqa %xmm12,%xmm7+ pslld $7,%xmm12+ psrld $25,%xmm7+ movdqa %xmm13,%xmm6+ pslld $7,%xmm13+ por %xmm7,%xmm12+ psrld $25,%xmm6+ movdqa (%r9),%xmm7+ por %xmm6,%xmm13+ movdqa %xmm4,0(%rsp)+ movdqa %xmm5,16(%rsp)+ movdqa 32(%rsp),%xmm4+ movdqa 48(%rsp),%xmm5+ paddd %xmm14,%xmm10+ paddd %xmm15,%xmm11+ pxor %xmm10,%xmm2+ pxor %xmm11,%xmm3+.byte 102,15,56,0,215+.byte 102,15,56,0,223+ paddd %xmm2,%xmm4+ paddd %xmm3,%xmm5+ pxor %xmm4,%xmm14+ pxor %xmm5,%xmm15+ movdqa %xmm14,%xmm6+ pslld $12,%xmm14+ psrld $20,%xmm6+ movdqa %xmm15,%xmm7+ pslld $12,%xmm15+ por %xmm6,%xmm14+ psrld $20,%xmm7+ movdqa (%r11),%xmm6+ por %xmm7,%xmm15+ paddd %xmm14,%xmm10+ paddd %xmm15,%xmm11+ pxor %xmm10,%xmm2+ pxor %xmm11,%xmm3+.byte 102,15,56,0,214+.byte 102,15,56,0,222+ paddd %xmm2,%xmm4+ paddd %xmm3,%xmm5+ pxor %xmm4,%xmm14+ pxor %xmm5,%xmm15+ movdqa %xmm14,%xmm7+ pslld $7,%xmm14+ psrld $25,%xmm7+ movdqa %xmm15,%xmm6+ pslld $7,%xmm15+ por %xmm7,%xmm14+ psrld $25,%xmm6+ movdqa (%r9),%xmm7+ por %xmm6,%xmm15+ paddd %xmm13,%xmm8+ paddd %xmm14,%xmm9+ pxor %xmm8,%xmm3+ pxor %xmm9,%xmm0+.byte 102,15,56,0,223+.byte 102,15,56,0,199+ paddd %xmm3,%xmm4+ paddd %xmm0,%xmm5+ pxor %xmm4,%xmm13+ pxor %xmm5,%xmm14+ movdqa %xmm13,%xmm6+ pslld $12,%xmm13+ psrld $20,%xmm6+ movdqa %xmm14,%xmm7+ pslld $12,%xmm14+ por %xmm6,%xmm13+ psrld $20,%xmm7+ movdqa (%r11),%xmm6+ por %xmm7,%xmm14+ paddd %xmm13,%xmm8+ paddd %xmm14,%xmm9+ pxor %xmm8,%xmm3+ pxor %xmm9,%xmm0+.byte 102,15,56,0,222+.byte 102,15,56,0,198+ paddd %xmm3,%xmm4+ paddd %xmm0,%xmm5+ pxor %xmm4,%xmm13+ pxor %xmm5,%xmm14+ movdqa %xmm13,%xmm7+ pslld $7,%xmm13+ psrld $25,%xmm7+ movdqa %xmm14,%xmm6+ pslld $7,%xmm14+ por %xmm7,%xmm13+ psrld $25,%xmm6+ movdqa (%r9),%xmm7+ por %xmm6,%xmm14+ movdqa %xmm4,32(%rsp)+ movdqa %xmm5,48(%rsp)+ movdqa 0(%rsp),%xmm4+ movdqa 16(%rsp),%xmm5+ paddd %xmm15,%xmm10+ paddd %xmm12,%xmm11+ pxor %xmm10,%xmm1+ pxor %xmm11,%xmm2+.byte 102,15,56,0,207+.byte 102,15,56,0,215+ paddd %xmm1,%xmm4+ paddd %xmm2,%xmm5+ pxor %xmm4,%xmm15+ pxor %xmm5,%xmm12+ movdqa %xmm15,%xmm6+ pslld $12,%xmm15+ psrld $20,%xmm6+ movdqa %xmm12,%xmm7+ pslld $12,%xmm12+ por %xmm6,%xmm15+ psrld $20,%xmm7+ movdqa (%r11),%xmm6+ por %xmm7,%xmm12+ paddd %xmm15,%xmm10+ paddd %xmm12,%xmm11+ pxor %xmm10,%xmm1+ pxor %xmm11,%xmm2+.byte 102,15,56,0,206+.byte 102,15,56,0,214+ paddd %xmm1,%xmm4+ paddd %xmm2,%xmm5+ pxor %xmm4,%xmm15+ pxor %xmm5,%xmm12+ movdqa %xmm15,%xmm7+ pslld $7,%xmm15+ psrld $25,%xmm7+ movdqa %xmm12,%xmm6+ pslld $7,%xmm12+ por %xmm7,%xmm15+ psrld $25,%xmm6+ movdqa (%r9),%xmm7+ por %xmm6,%xmm12+ decl %eax+ jnz .Loop4x++ paddd 64(%rsp),%xmm8+ paddd 80(%rsp),%xmm9+ paddd 96(%rsp),%xmm10+ paddd 112(%rsp),%xmm11++ movdqa %xmm8,%xmm6+ punpckldq %xmm9,%xmm8+ movdqa %xmm10,%xmm7+ punpckldq %xmm11,%xmm10+ punpckhdq %xmm9,%xmm6+ punpckhdq %xmm11,%xmm7+ movdqa %xmm8,%xmm9+ punpcklqdq %xmm10,%xmm8+ movdqa %xmm6,%xmm11+ punpcklqdq %xmm7,%xmm6+ punpckhqdq %xmm10,%xmm9+ punpckhqdq %xmm7,%xmm11+ paddd 128-256(%rcx),%xmm12+ paddd 144-256(%rcx),%xmm13+ paddd 160-256(%rcx),%xmm14+ paddd 176-256(%rcx),%xmm15++ movdqa %xmm8,0(%rsp)+ movdqa %xmm9,16(%rsp)+ movdqa 32(%rsp),%xmm8+ movdqa 48(%rsp),%xmm9++ movdqa %xmm12,%xmm10+ punpckldq %xmm13,%xmm12+ movdqa %xmm14,%xmm7+ punpckldq %xmm15,%xmm14+ punpckhdq %xmm13,%xmm10+ punpckhdq %xmm15,%xmm7+ movdqa %xmm12,%xmm13+ punpcklqdq %xmm14,%xmm12+ movdqa %xmm10,%xmm15+ punpcklqdq %xmm7,%xmm10+ punpckhqdq %xmm14,%xmm13+ punpckhqdq %xmm7,%xmm15+ paddd 192-256(%rcx),%xmm4+ paddd 208-256(%rcx),%xmm5+ paddd 224-256(%rcx),%xmm8+ paddd 240-256(%rcx),%xmm9++ movdqa %xmm6,32(%rsp)+ movdqa %xmm11,48(%rsp)++ movdqa %xmm4,%xmm14+ punpckldq %xmm5,%xmm4+ movdqa %xmm8,%xmm7+ punpckldq %xmm9,%xmm8+ punpckhdq %xmm5,%xmm14+ punpckhdq %xmm9,%xmm7+ movdqa %xmm4,%xmm5+ punpcklqdq %xmm8,%xmm4+ movdqa %xmm14,%xmm9+ punpcklqdq %xmm7,%xmm14+ punpckhqdq %xmm8,%xmm5+ punpckhqdq %xmm7,%xmm9+ paddd 256-256(%rcx),%xmm0+ paddd 272-256(%rcx),%xmm1+ paddd 288-256(%rcx),%xmm2+ paddd 304-256(%rcx),%xmm3++ movdqa %xmm0,%xmm8+ punpckldq %xmm1,%xmm0+ movdqa %xmm2,%xmm7+ punpckldq %xmm3,%xmm2+ punpckhdq %xmm1,%xmm8+ punpckhdq %xmm3,%xmm7+ movdqa %xmm0,%xmm1+ punpcklqdq %xmm2,%xmm0+ movdqa %xmm8,%xmm3+ punpcklqdq %xmm7,%xmm8+ punpckhqdq %xmm2,%xmm1+ punpckhqdq %xmm7,%xmm3+ cmpq $256,%rdx+ jb .Ltail4x++ movdqu 0(%rsi),%xmm6+ movdqu 16(%rsi),%xmm11+ movdqu 32(%rsi),%xmm2+ movdqu 48(%rsi),%xmm7+ pxor 0(%rsp),%xmm6+ pxor %xmm12,%xmm11+ pxor %xmm4,%xmm2+ pxor %xmm0,%xmm7++ movdqu %xmm6,0(%rdi)+ movdqu 64(%rsi),%xmm6+ movdqu %xmm11,16(%rdi)+ movdqu 80(%rsi),%xmm11+ movdqu %xmm2,32(%rdi)+ movdqu 96(%rsi),%xmm2+ movdqu %xmm7,48(%rdi)+ movdqu 112(%rsi),%xmm7+ leaq 128(%rsi),%rsi+ pxor 16(%rsp),%xmm6+ pxor %xmm13,%xmm11+ pxor %xmm5,%xmm2+ pxor %xmm1,%xmm7++ movdqu %xmm6,64(%rdi)+ movdqu 0(%rsi),%xmm6+ movdqu %xmm11,80(%rdi)+ movdqu 16(%rsi),%xmm11+ movdqu %xmm2,96(%rdi)+ movdqu 32(%rsi),%xmm2+ movdqu %xmm7,112(%rdi)+ leaq 128(%rdi),%rdi+ movdqu 48(%rsi),%xmm7+ pxor 32(%rsp),%xmm6+ pxor %xmm10,%xmm11+ pxor %xmm14,%xmm2+ pxor %xmm8,%xmm7++ movdqu %xmm6,0(%rdi)+ movdqu 64(%rsi),%xmm6+ movdqu %xmm11,16(%rdi)+ movdqu 80(%rsi),%xmm11+ movdqu %xmm2,32(%rdi)+ movdqu 96(%rsi),%xmm2+ movdqu %xmm7,48(%rdi)+ movdqu 112(%rsi),%xmm7+ leaq 128(%rsi),%rsi+ pxor 48(%rsp),%xmm6+ pxor %xmm15,%xmm11+ pxor %xmm9,%xmm2+ pxor %xmm3,%xmm7+ movdqu %xmm6,64(%rdi)+ movdqu %xmm11,80(%rdi)+ movdqu %xmm2,96(%rdi)+ movdqu %xmm7,112(%rdi)+ leaq 128(%rdi),%rdi++ subq $256,%rdx+ jnz .Loop_outer4x++ jmp .Ldone4x++.Ltail4x:+ cmpq $192,%rdx+ jae .L192_or_more4x+ cmpq $128,%rdx+ jae .L128_or_more4x+ cmpq $64,%rdx+ jae .L64_or_more4x+++ xorq %r9,%r9++ movdqa %xmm12,16(%rsp)+ movdqa %xmm4,32(%rsp)+ movdqa %xmm0,48(%rsp)+ jmp .Loop_tail4x++.align 32+.L64_or_more4x:+ movdqu 0(%rsi),%xmm6+ movdqu 16(%rsi),%xmm11+ movdqu 32(%rsi),%xmm2+ movdqu 48(%rsi),%xmm7+ pxor 0(%rsp),%xmm6+ pxor %xmm12,%xmm11+ pxor %xmm4,%xmm2+ pxor %xmm0,%xmm7+ movdqu %xmm6,0(%rdi)+ movdqu %xmm11,16(%rdi)+ movdqu %xmm2,32(%rdi)+ movdqu %xmm7,48(%rdi)+ je .Ldone4x++ movdqa 16(%rsp),%xmm6+ leaq 64(%rsi),%rsi+ xorq %r9,%r9+ movdqa %xmm6,0(%rsp)+ movdqa %xmm13,16(%rsp)+ leaq 64(%rdi),%rdi+ movdqa %xmm5,32(%rsp)+ subq $64,%rdx+ movdqa %xmm1,48(%rsp)+ jmp .Loop_tail4x++.align 32+.L128_or_more4x:+ movdqu 0(%rsi),%xmm6+ movdqu 16(%rsi),%xmm11+ movdqu 32(%rsi),%xmm2+ movdqu 48(%rsi),%xmm7+ pxor 0(%rsp),%xmm6+ pxor %xmm12,%xmm11+ pxor %xmm4,%xmm2+ pxor %xmm0,%xmm7++ movdqu %xmm6,0(%rdi)+ movdqu 64(%rsi),%xmm6+ movdqu %xmm11,16(%rdi)+ movdqu 80(%rsi),%xmm11+ movdqu %xmm2,32(%rdi)+ movdqu 96(%rsi),%xmm2+ movdqu %xmm7,48(%rdi)+ movdqu 112(%rsi),%xmm7+ pxor 16(%rsp),%xmm6+ pxor %xmm13,%xmm11+ pxor %xmm5,%xmm2+ pxor %xmm1,%xmm7+ movdqu %xmm6,64(%rdi)+ movdqu %xmm11,80(%rdi)+ movdqu %xmm2,96(%rdi)+ movdqu %xmm7,112(%rdi)+ je .Ldone4x++ movdqa 32(%rsp),%xmm6+ leaq 128(%rsi),%rsi+ xorq %r9,%r9+ movdqa %xmm6,0(%rsp)+ movdqa %xmm10,16(%rsp)+ leaq 128(%rdi),%rdi+ movdqa %xmm14,32(%rsp)+ subq $128,%rdx+ movdqa %xmm8,48(%rsp)+ jmp .Loop_tail4x++.align 32+.L192_or_more4x:+ movdqu 0(%rsi),%xmm6+ movdqu 16(%rsi),%xmm11+ movdqu 32(%rsi),%xmm2+ movdqu 48(%rsi),%xmm7+ pxor 0(%rsp),%xmm6+ pxor %xmm12,%xmm11+ pxor %xmm4,%xmm2+ pxor %xmm0,%xmm7++ movdqu %xmm6,0(%rdi)+ movdqu 64(%rsi),%xmm6+ movdqu %xmm11,16(%rdi)+ movdqu 80(%rsi),%xmm11+ movdqu %xmm2,32(%rdi)+ movdqu 96(%rsi),%xmm2+ movdqu %xmm7,48(%rdi)+ movdqu 112(%rsi),%xmm7+ leaq 128(%rsi),%rsi+ pxor 16(%rsp),%xmm6+ pxor %xmm13,%xmm11+ pxor %xmm5,%xmm2+ pxor %xmm1,%xmm7++ movdqu %xmm6,64(%rdi)+ movdqu 0(%rsi),%xmm6+ movdqu %xmm11,80(%rdi)+ movdqu 16(%rsi),%xmm11+ movdqu %xmm2,96(%rdi)+ movdqu 32(%rsi),%xmm2+ movdqu %xmm7,112(%rdi)+ leaq 128(%rdi),%rdi+ movdqu 48(%rsi),%xmm7+ pxor 32(%rsp),%xmm6+ pxor %xmm10,%xmm11+ pxor %xmm14,%xmm2+ pxor %xmm8,%xmm7+ movdqu %xmm6,0(%rdi)+ movdqu %xmm11,16(%rdi)+ movdqu %xmm2,32(%rdi)+ movdqu %xmm7,48(%rdi)+ je .Ldone4x++ movdqa 48(%rsp),%xmm6+ leaq 64(%rsi),%rsi+ xorq %r9,%r9+ movdqa %xmm6,0(%rsp)+ movdqa %xmm15,16(%rsp)+ leaq 64(%rdi),%rdi+ movdqa %xmm9,32(%rsp)+ subq $192,%rdx+ movdqa %xmm3,48(%rsp)++.Loop_tail4x:+ movzbl (%rsi,%r9,1),%eax+ movzbl (%rsp,%r9,1),%ecx+ leaq 1(%r9),%r9+ xorl %ecx,%eax+ movb %al,-1(%rdi,%r9,1)+ decq %rdx+ jnz .Loop_tail4x++.Ldone4x:+ leaq (%r10),%rsp+.cfi_def_cfa_register %rsp+.L4x_epilogue:+ .byte 0xf3,0xc3+.cfi_endproc +.size crypton_chacha20_asm_4x,.-crypton_chacha20_asm_4x+.type crypton_chacha20_asm_4xop,@function+.align 32+crypton_chacha20_asm_4xop:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa+++.Lcrypton_chacha20_asm_4xop:+ movq %rsp,%r10+.cfi_def_cfa_register %r10+ subq $0x140+8,%rsp+ andq $-16,%rsp+ vzeroupper++ vmovdqa .Lsigma(%rip),%xmm11+ vmovdqu (%rcx),%xmm3+ vmovdqu 16(%rcx),%xmm15+ vmovdqu (%r8),%xmm7+ leaq 256(%rsp),%rcx++ vpshufd $0x00,%xmm11,%xmm8+ vpshufd $0x55,%xmm11,%xmm9+ vmovdqa %xmm8,64(%rsp)+ vpshufd $0xaa,%xmm11,%xmm10+ vmovdqa %xmm9,80(%rsp)+ vpshufd $0xff,%xmm11,%xmm11+ vmovdqa %xmm10,96(%rsp)+ vmovdqa %xmm11,112(%rsp)++ vpshufd $0x00,%xmm3,%xmm0+ vpshufd $0x55,%xmm3,%xmm1+ vmovdqa %xmm0,128-256(%rcx)+ vpshufd $0xaa,%xmm3,%xmm2+ vmovdqa %xmm1,144-256(%rcx)+ vpshufd $0xff,%xmm3,%xmm3+ vmovdqa %xmm2,160-256(%rcx)+ vmovdqa %xmm3,176-256(%rcx)++ vpshufd $0x00,%xmm15,%xmm12+ vpshufd $0x55,%xmm15,%xmm13+ vmovdqa %xmm12,192-256(%rcx)+ vpshufd $0xaa,%xmm15,%xmm14+ vmovdqa %xmm13,208-256(%rcx)+ vpshufd $0xff,%xmm15,%xmm15+ vmovdqa %xmm14,224-256(%rcx)+ vmovdqa %xmm15,240-256(%rcx)++ vpshufd $0x00,%xmm7,%xmm4+ vpshufd $0x55,%xmm7,%xmm5+ vpaddd .Linc(%rip),%xmm4,%xmm4+ vpshufd $0xaa,%xmm7,%xmm6+ vmovdqa %xmm5,272-256(%rcx)+ vpshufd $0xff,%xmm7,%xmm7+ vmovdqa %xmm6,288-256(%rcx)+ vmovdqa %xmm7,304-256(%rcx)++ jmp .Loop_enter4xop++.align 32+.Loop_outer4xop:+ vmovdqa 64(%rsp),%xmm8+ vmovdqa 80(%rsp),%xmm9+ vmovdqa 96(%rsp),%xmm10+ vmovdqa 112(%rsp),%xmm11+ vmovdqa 128-256(%rcx),%xmm0+ vmovdqa 144-256(%rcx),%xmm1+ vmovdqa 160-256(%rcx),%xmm2+ vmovdqa 176-256(%rcx),%xmm3+ vmovdqa 192-256(%rcx),%xmm12+ vmovdqa 208-256(%rcx),%xmm13+ vmovdqa 224-256(%rcx),%xmm14+ vmovdqa 240-256(%rcx),%xmm15+ vmovdqa 256-256(%rcx),%xmm4+ vmovdqa 272-256(%rcx),%xmm5+ vmovdqa 288-256(%rcx),%xmm6+ vmovdqa 304-256(%rcx),%xmm7+ vpaddd .Lfour(%rip),%xmm4,%xmm4++.Loop_enter4xop:+ movl $10,%eax+ vmovdqa %xmm4,256-256(%rcx)+ jmp .Loop4xop++.align 32+.Loop4xop:+ vpaddd %xmm0,%xmm8,%xmm8+ vpaddd %xmm1,%xmm9,%xmm9+ vpaddd %xmm2,%xmm10,%xmm10+ vpaddd %xmm3,%xmm11,%xmm11+ vpxor %xmm4,%xmm8,%xmm4+ vpxor %xmm5,%xmm9,%xmm5+ vpxor %xmm6,%xmm10,%xmm6+ vpxor %xmm7,%xmm11,%xmm7+.byte 143,232,120,194,228,16+.byte 143,232,120,194,237,16+.byte 143,232,120,194,246,16+.byte 143,232,120,194,255,16+ vpaddd %xmm4,%xmm12,%xmm12+ vpaddd %xmm5,%xmm13,%xmm13+ vpaddd %xmm6,%xmm14,%xmm14+ vpaddd %xmm7,%xmm15,%xmm15+ vpxor %xmm0,%xmm12,%xmm0+ vpxor %xmm1,%xmm13,%xmm1+ vpxor %xmm14,%xmm2,%xmm2+ vpxor %xmm15,%xmm3,%xmm3+.byte 143,232,120,194,192,12+.byte 143,232,120,194,201,12+.byte 143,232,120,194,210,12+.byte 143,232,120,194,219,12+ vpaddd %xmm8,%xmm0,%xmm8+ vpaddd %xmm9,%xmm1,%xmm9+ vpaddd %xmm2,%xmm10,%xmm10+ vpaddd %xmm3,%xmm11,%xmm11+ vpxor %xmm4,%xmm8,%xmm4+ vpxor %xmm5,%xmm9,%xmm5+ vpxor %xmm6,%xmm10,%xmm6+ vpxor %xmm7,%xmm11,%xmm7+.byte 143,232,120,194,228,8+.byte 143,232,120,194,237,8+.byte 143,232,120,194,246,8+.byte 143,232,120,194,255,8+ vpaddd %xmm4,%xmm12,%xmm12+ vpaddd %xmm5,%xmm13,%xmm13+ vpaddd %xmm6,%xmm14,%xmm14+ vpaddd %xmm7,%xmm15,%xmm15+ vpxor %xmm0,%xmm12,%xmm0+ vpxor %xmm1,%xmm13,%xmm1+ vpxor %xmm14,%xmm2,%xmm2+ vpxor %xmm15,%xmm3,%xmm3+.byte 143,232,120,194,192,7+.byte 143,232,120,194,201,7+.byte 143,232,120,194,210,7+.byte 143,232,120,194,219,7+ vpaddd %xmm1,%xmm8,%xmm8+ vpaddd %xmm2,%xmm9,%xmm9+ vpaddd %xmm3,%xmm10,%xmm10+ vpaddd %xmm0,%xmm11,%xmm11+ vpxor %xmm7,%xmm8,%xmm7+ vpxor %xmm4,%xmm9,%xmm4+ vpxor %xmm5,%xmm10,%xmm5+ vpxor %xmm6,%xmm11,%xmm6+.byte 143,232,120,194,255,16+.byte 143,232,120,194,228,16+.byte 143,232,120,194,237,16+.byte 143,232,120,194,246,16+ vpaddd %xmm7,%xmm14,%xmm14+ vpaddd %xmm4,%xmm15,%xmm15+ vpaddd %xmm5,%xmm12,%xmm12+ vpaddd %xmm6,%xmm13,%xmm13+ vpxor %xmm1,%xmm14,%xmm1+ vpxor %xmm2,%xmm15,%xmm2+ vpxor %xmm12,%xmm3,%xmm3+ vpxor %xmm13,%xmm0,%xmm0+.byte 143,232,120,194,201,12+.byte 143,232,120,194,210,12+.byte 143,232,120,194,219,12+.byte 143,232,120,194,192,12+ vpaddd %xmm8,%xmm1,%xmm8+ vpaddd %xmm9,%xmm2,%xmm9+ vpaddd %xmm3,%xmm10,%xmm10+ vpaddd %xmm0,%xmm11,%xmm11+ vpxor %xmm7,%xmm8,%xmm7+ vpxor %xmm4,%xmm9,%xmm4+ vpxor %xmm5,%xmm10,%xmm5+ vpxor %xmm6,%xmm11,%xmm6+.byte 143,232,120,194,255,8+.byte 143,232,120,194,228,8+.byte 143,232,120,194,237,8+.byte 143,232,120,194,246,8+ vpaddd %xmm7,%xmm14,%xmm14+ vpaddd %xmm4,%xmm15,%xmm15+ vpaddd %xmm5,%xmm12,%xmm12+ vpaddd %xmm6,%xmm13,%xmm13+ vpxor %xmm1,%xmm14,%xmm1+ vpxor %xmm2,%xmm15,%xmm2+ vpxor %xmm12,%xmm3,%xmm3+ vpxor %xmm13,%xmm0,%xmm0+.byte 143,232,120,194,201,7+.byte 143,232,120,194,210,7+.byte 143,232,120,194,219,7+.byte 143,232,120,194,192,7+ decl %eax+ jnz .Loop4xop++ vpaddd 64(%rsp),%xmm8,%xmm8+ vpaddd 80(%rsp),%xmm9,%xmm9+ vpaddd 96(%rsp),%xmm10,%xmm10+ vpaddd 112(%rsp),%xmm11,%xmm11++ vmovdqa %xmm14,32(%rsp)+ vmovdqa %xmm15,48(%rsp)++ vpunpckldq %xmm9,%xmm8,%xmm14+ vpunpckldq %xmm11,%xmm10,%xmm15+ vpunpckhdq %xmm9,%xmm8,%xmm8+ vpunpckhdq %xmm11,%xmm10,%xmm10+ vpunpcklqdq %xmm15,%xmm14,%xmm9+ vpunpckhqdq %xmm15,%xmm14,%xmm14+ vpunpcklqdq %xmm10,%xmm8,%xmm11+ vpunpckhqdq %xmm10,%xmm8,%xmm8+ vpaddd 128-256(%rcx),%xmm0,%xmm0+ vpaddd 144-256(%rcx),%xmm1,%xmm1+ vpaddd 160-256(%rcx),%xmm2,%xmm2+ vpaddd 176-256(%rcx),%xmm3,%xmm3++ vmovdqa %xmm9,0(%rsp)+ vmovdqa %xmm14,16(%rsp)+ vmovdqa 32(%rsp),%xmm9+ vmovdqa 48(%rsp),%xmm14++ vpunpckldq %xmm1,%xmm0,%xmm10+ vpunpckldq %xmm3,%xmm2,%xmm15+ vpunpckhdq %xmm1,%xmm0,%xmm0+ vpunpckhdq %xmm3,%xmm2,%xmm2+ vpunpcklqdq %xmm15,%xmm10,%xmm1+ vpunpckhqdq %xmm15,%xmm10,%xmm10+ vpunpcklqdq %xmm2,%xmm0,%xmm3+ vpunpckhqdq %xmm2,%xmm0,%xmm0+ vpaddd 192-256(%rcx),%xmm12,%xmm12+ vpaddd 208-256(%rcx),%xmm13,%xmm13+ vpaddd 224-256(%rcx),%xmm9,%xmm9+ vpaddd 240-256(%rcx),%xmm14,%xmm14++ vpunpckldq %xmm13,%xmm12,%xmm2+ vpunpckldq %xmm14,%xmm9,%xmm15+ vpunpckhdq %xmm13,%xmm12,%xmm12+ vpunpckhdq %xmm14,%xmm9,%xmm9+ vpunpcklqdq %xmm15,%xmm2,%xmm13+ vpunpckhqdq %xmm15,%xmm2,%xmm2+ vpunpcklqdq %xmm9,%xmm12,%xmm14+ vpunpckhqdq %xmm9,%xmm12,%xmm12+ vpaddd 256-256(%rcx),%xmm4,%xmm4+ vpaddd 272-256(%rcx),%xmm5,%xmm5+ vpaddd 288-256(%rcx),%xmm6,%xmm6+ vpaddd 304-256(%rcx),%xmm7,%xmm7++ vpunpckldq %xmm5,%xmm4,%xmm9+ vpunpckldq %xmm7,%xmm6,%xmm15+ vpunpckhdq %xmm5,%xmm4,%xmm4+ vpunpckhdq %xmm7,%xmm6,%xmm6+ vpunpcklqdq %xmm15,%xmm9,%xmm5+ vpunpckhqdq %xmm15,%xmm9,%xmm9+ vpunpcklqdq %xmm6,%xmm4,%xmm7+ vpunpckhqdq %xmm6,%xmm4,%xmm4+ vmovdqa 0(%rsp),%xmm6+ vmovdqa 16(%rsp),%xmm15++ cmpq $256,%rdx+ jb .Ltail4xop++ vpxor 0(%rsi),%xmm6,%xmm6+ vpxor 16(%rsi),%xmm1,%xmm1+ vpxor 32(%rsi),%xmm13,%xmm13+ vpxor 48(%rsi),%xmm5,%xmm5+ vpxor 64(%rsi),%xmm15,%xmm15+ vpxor 80(%rsi),%xmm10,%xmm10+ vpxor 96(%rsi),%xmm2,%xmm2+ vpxor 112(%rsi),%xmm9,%xmm9+ leaq 128(%rsi),%rsi+ vpxor 0(%rsi),%xmm11,%xmm11+ vpxor 16(%rsi),%xmm3,%xmm3+ vpxor 32(%rsi),%xmm14,%xmm14+ vpxor 48(%rsi),%xmm7,%xmm7+ vpxor 64(%rsi),%xmm8,%xmm8+ vpxor 80(%rsi),%xmm0,%xmm0+ vpxor 96(%rsi),%xmm12,%xmm12+ vpxor 112(%rsi),%xmm4,%xmm4+ leaq 128(%rsi),%rsi++ vmovdqu %xmm6,0(%rdi)+ vmovdqu %xmm1,16(%rdi)+ vmovdqu %xmm13,32(%rdi)+ vmovdqu %xmm5,48(%rdi)+ vmovdqu %xmm15,64(%rdi)+ vmovdqu %xmm10,80(%rdi)+ vmovdqu %xmm2,96(%rdi)+ vmovdqu %xmm9,112(%rdi)+ leaq 128(%rdi),%rdi+ vmovdqu %xmm11,0(%rdi)+ vmovdqu %xmm3,16(%rdi)+ vmovdqu %xmm14,32(%rdi)+ vmovdqu %xmm7,48(%rdi)+ vmovdqu %xmm8,64(%rdi)+ vmovdqu %xmm0,80(%rdi)+ vmovdqu %xmm12,96(%rdi)+ vmovdqu %xmm4,112(%rdi)+ leaq 128(%rdi),%rdi++ subq $256,%rdx+ jnz .Loop_outer4xop++ jmp .Ldone4xop++.align 32+.Ltail4xop:+ cmpq $192,%rdx+ jae .L192_or_more4xop+ cmpq $128,%rdx+ jae .L128_or_more4xop+ cmpq $64,%rdx+ jae .L64_or_more4xop++ xorq %r9,%r9+ vmovdqa %xmm6,0(%rsp)+ vmovdqa %xmm1,16(%rsp)+ vmovdqa %xmm13,32(%rsp)+ vmovdqa %xmm5,48(%rsp)+ jmp .Loop_tail4xop++.align 32+.L64_or_more4xop:+ vpxor 0(%rsi),%xmm6,%xmm6+ vpxor 16(%rsi),%xmm1,%xmm1+ vpxor 32(%rsi),%xmm13,%xmm13+ vpxor 48(%rsi),%xmm5,%xmm5+ vmovdqu %xmm6,0(%rdi)+ vmovdqu %xmm1,16(%rdi)+ vmovdqu %xmm13,32(%rdi)+ vmovdqu %xmm5,48(%rdi)+ je .Ldone4xop++ leaq 64(%rsi),%rsi+ vmovdqa %xmm15,0(%rsp)+ xorq %r9,%r9+ vmovdqa %xmm10,16(%rsp)+ leaq 64(%rdi),%rdi+ vmovdqa %xmm2,32(%rsp)+ subq $64,%rdx+ vmovdqa %xmm9,48(%rsp)+ jmp .Loop_tail4xop++.align 32+.L128_or_more4xop:+ vpxor 0(%rsi),%xmm6,%xmm6+ vpxor 16(%rsi),%xmm1,%xmm1+ vpxor 32(%rsi),%xmm13,%xmm13+ vpxor 48(%rsi),%xmm5,%xmm5+ vpxor 64(%rsi),%xmm15,%xmm15+ vpxor 80(%rsi),%xmm10,%xmm10+ vpxor 96(%rsi),%xmm2,%xmm2+ vpxor 112(%rsi),%xmm9,%xmm9++ vmovdqu %xmm6,0(%rdi)+ vmovdqu %xmm1,16(%rdi)+ vmovdqu %xmm13,32(%rdi)+ vmovdqu %xmm5,48(%rdi)+ vmovdqu %xmm15,64(%rdi)+ vmovdqu %xmm10,80(%rdi)+ vmovdqu %xmm2,96(%rdi)+ vmovdqu %xmm9,112(%rdi)+ je .Ldone4xop++ leaq 128(%rsi),%rsi+ vmovdqa %xmm11,0(%rsp)+ xorq %r9,%r9+ vmovdqa %xmm3,16(%rsp)+ leaq 128(%rdi),%rdi+ vmovdqa %xmm14,32(%rsp)+ subq $128,%rdx+ vmovdqa %xmm7,48(%rsp)+ jmp .Loop_tail4xop++.align 32+.L192_or_more4xop:+ vpxor 0(%rsi),%xmm6,%xmm6+ vpxor 16(%rsi),%xmm1,%xmm1+ vpxor 32(%rsi),%xmm13,%xmm13+ vpxor 48(%rsi),%xmm5,%xmm5+ vpxor 64(%rsi),%xmm15,%xmm15+ vpxor 80(%rsi),%xmm10,%xmm10+ vpxor 96(%rsi),%xmm2,%xmm2+ vpxor 112(%rsi),%xmm9,%xmm9+ leaq 128(%rsi),%rsi+ vpxor 0(%rsi),%xmm11,%xmm11+ vpxor 16(%rsi),%xmm3,%xmm3+ vpxor 32(%rsi),%xmm14,%xmm14+ vpxor 48(%rsi),%xmm7,%xmm7++ vmovdqu %xmm6,0(%rdi)+ vmovdqu %xmm1,16(%rdi)+ vmovdqu %xmm13,32(%rdi)+ vmovdqu %xmm5,48(%rdi)+ vmovdqu %xmm15,64(%rdi)+ vmovdqu %xmm10,80(%rdi)+ vmovdqu %xmm2,96(%rdi)+ vmovdqu %xmm9,112(%rdi)+ leaq 128(%rdi),%rdi+ vmovdqu %xmm11,0(%rdi)+ vmovdqu %xmm3,16(%rdi)+ vmovdqu %xmm14,32(%rdi)+ vmovdqu %xmm7,48(%rdi)+ je .Ldone4xop++ leaq 64(%rsi),%rsi+ vmovdqa %xmm8,0(%rsp)+ xorq %r9,%r9+ vmovdqa %xmm0,16(%rsp)+ leaq 64(%rdi),%rdi+ vmovdqa %xmm12,32(%rsp)+ subq $192,%rdx+ vmovdqa %xmm4,48(%rsp)++.Loop_tail4xop:+ movzbl (%rsi,%r9,1),%eax+ movzbl (%rsp,%r9,1),%ecx+ leaq 1(%r9),%r9+ xorl %ecx,%eax+ movb %al,-1(%rdi,%r9,1)+ decq %rdx+ jnz .Loop_tail4xop++.Ldone4xop:+ vzeroupper+ leaq (%r10),%rsp+.cfi_def_cfa_register %rsp+.L4xop_epilogue:+ .byte 0xf3,0xc3+.cfi_endproc +.size crypton_chacha20_asm_4xop,.-crypton_chacha20_asm_4xop+.type crypton_chacha20_asm_avx2,@function+.align 32+crypton_chacha20_asm_avx2:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa+++.Lcrypton_chacha20_asm_8x:+ movq %rsp,%r10+.cfi_def_cfa_register %r10+ subq $0x280+8,%rsp+ andq $-32,%rsp+ vzeroupper+++++++++++ vbroadcasti128 .Lsigma(%rip),%ymm11+ vbroadcasti128 (%rcx),%ymm3+ vbroadcasti128 16(%rcx),%ymm15+ vbroadcasti128 (%r8),%ymm7+ leaq 256(%rsp),%rcx+ leaq 512(%rsp),%rax+ leaq .Lrot16(%rip),%r9+ leaq .Lrot24(%rip),%r11++ vpshufd $0x00,%ymm11,%ymm8+ vpshufd $0x55,%ymm11,%ymm9+ vmovdqa %ymm8,128-256(%rcx)+ vpshufd $0xaa,%ymm11,%ymm10+ vmovdqa %ymm9,160-256(%rcx)+ vpshufd $0xff,%ymm11,%ymm11+ vmovdqa %ymm10,192-256(%rcx)+ vmovdqa %ymm11,224-256(%rcx)++ vpshufd $0x00,%ymm3,%ymm0+ vpshufd $0x55,%ymm3,%ymm1+ vmovdqa %ymm0,256-256(%rcx)+ vpshufd $0xaa,%ymm3,%ymm2+ vmovdqa %ymm1,288-256(%rcx)+ vpshufd $0xff,%ymm3,%ymm3+ vmovdqa %ymm2,320-256(%rcx)+ vmovdqa %ymm3,352-256(%rcx)++ vpshufd $0x00,%ymm15,%ymm12+ vpshufd $0x55,%ymm15,%ymm13+ vmovdqa %ymm12,384-512(%rax)+ vpshufd $0xaa,%ymm15,%ymm14+ vmovdqa %ymm13,416-512(%rax)+ vpshufd $0xff,%ymm15,%ymm15+ vmovdqa %ymm14,448-512(%rax)+ vmovdqa %ymm15,480-512(%rax)++ vpshufd $0x00,%ymm7,%ymm4+ vpshufd $0x55,%ymm7,%ymm5+ vpaddd .Lincy(%rip),%ymm4,%ymm4+ vpshufd $0xaa,%ymm7,%ymm6+ vmovdqa %ymm5,544-512(%rax)+ vpshufd $0xff,%ymm7,%ymm7+ vmovdqa %ymm6,576-512(%rax)+ vmovdqa %ymm7,608-512(%rax)++ jmp .Loop_enter8x++.align 32+.Loop_outer8x:+ vmovdqa 128-256(%rcx),%ymm8+ vmovdqa 160-256(%rcx),%ymm9+ vmovdqa 192-256(%rcx),%ymm10+ vmovdqa 224-256(%rcx),%ymm11+ vmovdqa 256-256(%rcx),%ymm0+ vmovdqa 288-256(%rcx),%ymm1+ vmovdqa 320-256(%rcx),%ymm2+ vmovdqa 352-256(%rcx),%ymm3+ vmovdqa 384-512(%rax),%ymm12+ vmovdqa 416-512(%rax),%ymm13+ vmovdqa 448-512(%rax),%ymm14+ vmovdqa 480-512(%rax),%ymm15+ vmovdqa 512-512(%rax),%ymm4+ vmovdqa 544-512(%rax),%ymm5+ vmovdqa 576-512(%rax),%ymm6+ vmovdqa 608-512(%rax),%ymm7+ vpaddd .Leight(%rip),%ymm4,%ymm4++.Loop_enter8x:+ vmovdqa %ymm14,64(%rsp)+ vmovdqa %ymm15,96(%rsp)+ vbroadcasti128 (%r9),%ymm15+ vmovdqa %ymm4,512-512(%rax)+ movl $10,%eax+ jmp .Loop8x++.align 32+.Loop8x:+ vpaddd %ymm0,%ymm8,%ymm8+ vpxor %ymm4,%ymm8,%ymm4+ vpshufb %ymm15,%ymm4,%ymm4+ vpaddd %ymm1,%ymm9,%ymm9+ vpxor %ymm5,%ymm9,%ymm5+ vpshufb %ymm15,%ymm5,%ymm5+ vpaddd %ymm4,%ymm12,%ymm12+ vpxor %ymm0,%ymm12,%ymm0+ vpslld $12,%ymm0,%ymm14+ vpsrld $20,%ymm0,%ymm0+ vpor %ymm0,%ymm14,%ymm0+ vbroadcasti128 (%r11),%ymm14+ vpaddd %ymm5,%ymm13,%ymm13+ vpxor %ymm1,%ymm13,%ymm1+ vpslld $12,%ymm1,%ymm15+ vpsrld $20,%ymm1,%ymm1+ vpor %ymm1,%ymm15,%ymm1+ vpaddd %ymm0,%ymm8,%ymm8+ vpxor %ymm4,%ymm8,%ymm4+ vpshufb %ymm14,%ymm4,%ymm4+ vpaddd %ymm1,%ymm9,%ymm9+ vpxor %ymm5,%ymm9,%ymm5+ vpshufb %ymm14,%ymm5,%ymm5+ vpaddd %ymm4,%ymm12,%ymm12+ vpxor %ymm0,%ymm12,%ymm0+ vpslld $7,%ymm0,%ymm15+ vpsrld $25,%ymm0,%ymm0+ vpor %ymm0,%ymm15,%ymm0+ vbroadcasti128 (%r9),%ymm15+ vpaddd %ymm5,%ymm13,%ymm13+ vpxor %ymm1,%ymm13,%ymm1+ vpslld $7,%ymm1,%ymm14+ vpsrld $25,%ymm1,%ymm1+ vpor %ymm1,%ymm14,%ymm1+ vmovdqa %ymm12,0(%rsp)+ vmovdqa %ymm13,32(%rsp)+ vmovdqa 64(%rsp),%ymm12+ vmovdqa 96(%rsp),%ymm13+ vpaddd %ymm2,%ymm10,%ymm10+ vpxor %ymm6,%ymm10,%ymm6+ vpshufb %ymm15,%ymm6,%ymm6+ vpaddd %ymm3,%ymm11,%ymm11+ vpxor %ymm7,%ymm11,%ymm7+ vpshufb %ymm15,%ymm7,%ymm7+ vpaddd %ymm6,%ymm12,%ymm12+ vpxor %ymm2,%ymm12,%ymm2+ vpslld $12,%ymm2,%ymm14+ vpsrld $20,%ymm2,%ymm2+ vpor %ymm2,%ymm14,%ymm2+ vbroadcasti128 (%r11),%ymm14+ vpaddd %ymm7,%ymm13,%ymm13+ vpxor %ymm3,%ymm13,%ymm3+ vpslld $12,%ymm3,%ymm15+ vpsrld $20,%ymm3,%ymm3+ vpor %ymm3,%ymm15,%ymm3+ vpaddd %ymm2,%ymm10,%ymm10+ vpxor %ymm6,%ymm10,%ymm6+ vpshufb %ymm14,%ymm6,%ymm6+ vpaddd %ymm3,%ymm11,%ymm11+ vpxor %ymm7,%ymm11,%ymm7+ vpshufb %ymm14,%ymm7,%ymm7+ vpaddd %ymm6,%ymm12,%ymm12+ vpxor %ymm2,%ymm12,%ymm2+ vpslld $7,%ymm2,%ymm15+ vpsrld $25,%ymm2,%ymm2+ vpor %ymm2,%ymm15,%ymm2+ vbroadcasti128 (%r9),%ymm15+ vpaddd %ymm7,%ymm13,%ymm13+ vpxor %ymm3,%ymm13,%ymm3+ vpslld $7,%ymm3,%ymm14+ vpsrld $25,%ymm3,%ymm3+ vpor %ymm3,%ymm14,%ymm3+ vpaddd %ymm1,%ymm8,%ymm8+ vpxor %ymm7,%ymm8,%ymm7+ vpshufb %ymm15,%ymm7,%ymm7+ vpaddd %ymm2,%ymm9,%ymm9+ vpxor %ymm4,%ymm9,%ymm4+ vpshufb %ymm15,%ymm4,%ymm4+ vpaddd %ymm7,%ymm12,%ymm12+ vpxor %ymm1,%ymm12,%ymm1+ vpslld $12,%ymm1,%ymm14+ vpsrld $20,%ymm1,%ymm1+ vpor %ymm1,%ymm14,%ymm1+ vbroadcasti128 (%r11),%ymm14+ vpaddd %ymm4,%ymm13,%ymm13+ vpxor %ymm2,%ymm13,%ymm2+ vpslld $12,%ymm2,%ymm15+ vpsrld $20,%ymm2,%ymm2+ vpor %ymm2,%ymm15,%ymm2+ vpaddd %ymm1,%ymm8,%ymm8+ vpxor %ymm7,%ymm8,%ymm7+ vpshufb %ymm14,%ymm7,%ymm7+ vpaddd %ymm2,%ymm9,%ymm9+ vpxor %ymm4,%ymm9,%ymm4+ vpshufb %ymm14,%ymm4,%ymm4+ vpaddd %ymm7,%ymm12,%ymm12+ vpxor %ymm1,%ymm12,%ymm1+ vpslld $7,%ymm1,%ymm15+ vpsrld $25,%ymm1,%ymm1+ vpor %ymm1,%ymm15,%ymm1+ vbroadcasti128 (%r9),%ymm15+ vpaddd %ymm4,%ymm13,%ymm13+ vpxor %ymm2,%ymm13,%ymm2+ vpslld $7,%ymm2,%ymm14+ vpsrld $25,%ymm2,%ymm2+ vpor %ymm2,%ymm14,%ymm2+ vmovdqa %ymm12,64(%rsp)+ vmovdqa %ymm13,96(%rsp)+ vmovdqa 0(%rsp),%ymm12+ vmovdqa 32(%rsp),%ymm13+ vpaddd %ymm3,%ymm10,%ymm10+ vpxor %ymm5,%ymm10,%ymm5+ vpshufb %ymm15,%ymm5,%ymm5+ vpaddd %ymm0,%ymm11,%ymm11+ vpxor %ymm6,%ymm11,%ymm6+ vpshufb %ymm15,%ymm6,%ymm6+ vpaddd %ymm5,%ymm12,%ymm12+ vpxor %ymm3,%ymm12,%ymm3+ vpslld $12,%ymm3,%ymm14+ vpsrld $20,%ymm3,%ymm3+ vpor %ymm3,%ymm14,%ymm3+ vbroadcasti128 (%r11),%ymm14+ vpaddd %ymm6,%ymm13,%ymm13+ vpxor %ymm0,%ymm13,%ymm0+ vpslld $12,%ymm0,%ymm15+ vpsrld $20,%ymm0,%ymm0+ vpor %ymm0,%ymm15,%ymm0+ vpaddd %ymm3,%ymm10,%ymm10+ vpxor %ymm5,%ymm10,%ymm5+ vpshufb %ymm14,%ymm5,%ymm5+ vpaddd %ymm0,%ymm11,%ymm11+ vpxor %ymm6,%ymm11,%ymm6+ vpshufb %ymm14,%ymm6,%ymm6+ vpaddd %ymm5,%ymm12,%ymm12+ vpxor %ymm3,%ymm12,%ymm3+ vpslld $7,%ymm3,%ymm15+ vpsrld $25,%ymm3,%ymm3+ vpor %ymm3,%ymm15,%ymm3+ vbroadcasti128 (%r9),%ymm15+ vpaddd %ymm6,%ymm13,%ymm13+ vpxor %ymm0,%ymm13,%ymm0+ vpslld $7,%ymm0,%ymm14+ vpsrld $25,%ymm0,%ymm0+ vpor %ymm0,%ymm14,%ymm0+ decl %eax+ jnz .Loop8x++ leaq 512(%rsp),%rax+ vpaddd 128-256(%rcx),%ymm8,%ymm8+ vpaddd 160-256(%rcx),%ymm9,%ymm9+ vpaddd 192-256(%rcx),%ymm10,%ymm10+ vpaddd 224-256(%rcx),%ymm11,%ymm11++ vpunpckldq %ymm9,%ymm8,%ymm14+ vpunpckldq %ymm11,%ymm10,%ymm15+ vpunpckhdq %ymm9,%ymm8,%ymm8+ vpunpckhdq %ymm11,%ymm10,%ymm10+ vpunpcklqdq %ymm15,%ymm14,%ymm9+ vpunpckhqdq %ymm15,%ymm14,%ymm14+ vpunpcklqdq %ymm10,%ymm8,%ymm11+ vpunpckhqdq %ymm10,%ymm8,%ymm8+ vpaddd 256-256(%rcx),%ymm0,%ymm0+ vpaddd 288-256(%rcx),%ymm1,%ymm1+ vpaddd 320-256(%rcx),%ymm2,%ymm2+ vpaddd 352-256(%rcx),%ymm3,%ymm3++ vpunpckldq %ymm1,%ymm0,%ymm10+ vpunpckldq %ymm3,%ymm2,%ymm15+ vpunpckhdq %ymm1,%ymm0,%ymm0+ vpunpckhdq %ymm3,%ymm2,%ymm2+ vpunpcklqdq %ymm15,%ymm10,%ymm1+ vpunpckhqdq %ymm15,%ymm10,%ymm10+ vpunpcklqdq %ymm2,%ymm0,%ymm3+ vpunpckhqdq %ymm2,%ymm0,%ymm0+ vperm2i128 $0x20,%ymm1,%ymm9,%ymm15+ vperm2i128 $0x31,%ymm1,%ymm9,%ymm1+ vperm2i128 $0x20,%ymm10,%ymm14,%ymm9+ vperm2i128 $0x31,%ymm10,%ymm14,%ymm10+ vperm2i128 $0x20,%ymm3,%ymm11,%ymm14+ vperm2i128 $0x31,%ymm3,%ymm11,%ymm3+ vperm2i128 $0x20,%ymm0,%ymm8,%ymm11+ vperm2i128 $0x31,%ymm0,%ymm8,%ymm0+ vmovdqa %ymm15,0(%rsp)+ vmovdqa %ymm9,32(%rsp)+ vmovdqa 64(%rsp),%ymm15+ vmovdqa 96(%rsp),%ymm9++ vpaddd 384-512(%rax),%ymm12,%ymm12+ vpaddd 416-512(%rax),%ymm13,%ymm13+ vpaddd 448-512(%rax),%ymm15,%ymm15+ vpaddd 480-512(%rax),%ymm9,%ymm9++ vpunpckldq %ymm13,%ymm12,%ymm2+ vpunpckldq %ymm9,%ymm15,%ymm8+ vpunpckhdq %ymm13,%ymm12,%ymm12+ vpunpckhdq %ymm9,%ymm15,%ymm15+ vpunpcklqdq %ymm8,%ymm2,%ymm13+ vpunpckhqdq %ymm8,%ymm2,%ymm2+ vpunpcklqdq %ymm15,%ymm12,%ymm9+ vpunpckhqdq %ymm15,%ymm12,%ymm12+ vpaddd 512-512(%rax),%ymm4,%ymm4+ vpaddd 544-512(%rax),%ymm5,%ymm5+ vpaddd 576-512(%rax),%ymm6,%ymm6+ vpaddd 608-512(%rax),%ymm7,%ymm7++ vpunpckldq %ymm5,%ymm4,%ymm15+ vpunpckldq %ymm7,%ymm6,%ymm8+ vpunpckhdq %ymm5,%ymm4,%ymm4+ vpunpckhdq %ymm7,%ymm6,%ymm6+ vpunpcklqdq %ymm8,%ymm15,%ymm5+ vpunpckhqdq %ymm8,%ymm15,%ymm15+ vpunpcklqdq %ymm6,%ymm4,%ymm7+ vpunpckhqdq %ymm6,%ymm4,%ymm4+ vperm2i128 $0x20,%ymm5,%ymm13,%ymm8+ vperm2i128 $0x31,%ymm5,%ymm13,%ymm5+ vperm2i128 $0x20,%ymm15,%ymm2,%ymm13+ vperm2i128 $0x31,%ymm15,%ymm2,%ymm15+ vperm2i128 $0x20,%ymm7,%ymm9,%ymm2+ vperm2i128 $0x31,%ymm7,%ymm9,%ymm7+ vperm2i128 $0x20,%ymm4,%ymm12,%ymm9+ vperm2i128 $0x31,%ymm4,%ymm12,%ymm4+ vmovdqa 0(%rsp),%ymm6+ vmovdqa 32(%rsp),%ymm12++ cmpq $512,%rdx+ jb .Ltail8x++ vpxor 0(%rsi),%ymm6,%ymm6+ vpxor 32(%rsi),%ymm8,%ymm8+ vpxor 64(%rsi),%ymm1,%ymm1+ vpxor 96(%rsi),%ymm5,%ymm5+ leaq 128(%rsi),%rsi+ vmovdqu %ymm6,0(%rdi)+ vmovdqu %ymm8,32(%rdi)+ vmovdqu %ymm1,64(%rdi)+ vmovdqu %ymm5,96(%rdi)+ leaq 128(%rdi),%rdi++ vpxor 0(%rsi),%ymm12,%ymm12+ vpxor 32(%rsi),%ymm13,%ymm13+ vpxor 64(%rsi),%ymm10,%ymm10+ vpxor 96(%rsi),%ymm15,%ymm15+ leaq 128(%rsi),%rsi+ vmovdqu %ymm12,0(%rdi)+ vmovdqu %ymm13,32(%rdi)+ vmovdqu %ymm10,64(%rdi)+ vmovdqu %ymm15,96(%rdi)+ leaq 128(%rdi),%rdi++ vpxor 0(%rsi),%ymm14,%ymm14+ vpxor 32(%rsi),%ymm2,%ymm2+ vpxor 64(%rsi),%ymm3,%ymm3+ vpxor 96(%rsi),%ymm7,%ymm7+ leaq 128(%rsi),%rsi+ vmovdqu %ymm14,0(%rdi)+ vmovdqu %ymm2,32(%rdi)+ vmovdqu %ymm3,64(%rdi)+ vmovdqu %ymm7,96(%rdi)+ leaq 128(%rdi),%rdi++ vpxor 0(%rsi),%ymm11,%ymm11+ vpxor 32(%rsi),%ymm9,%ymm9+ vpxor 64(%rsi),%ymm0,%ymm0+ vpxor 96(%rsi),%ymm4,%ymm4+ leaq 128(%rsi),%rsi+ vmovdqu %ymm11,0(%rdi)+ vmovdqu %ymm9,32(%rdi)+ vmovdqu %ymm0,64(%rdi)+ vmovdqu %ymm4,96(%rdi)+ leaq 128(%rdi),%rdi++ subq $512,%rdx+ jnz .Loop_outer8x++ jmp .Ldone8x++.Ltail8x:+ cmpq $448,%rdx+ jae .L448_or_more8x+ cmpq $384,%rdx+ jae .L384_or_more8x+ cmpq $320,%rdx+ jae .L320_or_more8x+ cmpq $256,%rdx+ jae .L256_or_more8x+ cmpq $192,%rdx+ jae .L192_or_more8x+ cmpq $128,%rdx+ jae .L128_or_more8x+ cmpq $64,%rdx+ jae .L64_or_more8x++ xorq %r9,%r9+ vmovdqa %ymm6,0(%rsp)+ vmovdqa %ymm8,32(%rsp)+ jmp .Loop_tail8x++.align 32+.L64_or_more8x:+ vpxor 0(%rsi),%ymm6,%ymm6+ vpxor 32(%rsi),%ymm8,%ymm8+ vmovdqu %ymm6,0(%rdi)+ vmovdqu %ymm8,32(%rdi)+ je .Ldone8x++ leaq 64(%rsi),%rsi+ xorq %r9,%r9+ vmovdqa %ymm1,0(%rsp)+ leaq 64(%rdi),%rdi+ subq $64,%rdx+ vmovdqa %ymm5,32(%rsp)+ jmp .Loop_tail8x++.align 32+.L128_or_more8x:+ vpxor 0(%rsi),%ymm6,%ymm6+ vpxor 32(%rsi),%ymm8,%ymm8+ vpxor 64(%rsi),%ymm1,%ymm1+ vpxor 96(%rsi),%ymm5,%ymm5+ vmovdqu %ymm6,0(%rdi)+ vmovdqu %ymm8,32(%rdi)+ vmovdqu %ymm1,64(%rdi)+ vmovdqu %ymm5,96(%rdi)+ je .Ldone8x++ leaq 128(%rsi),%rsi+ xorq %r9,%r9+ vmovdqa %ymm12,0(%rsp)+ leaq 128(%rdi),%rdi+ subq $128,%rdx+ vmovdqa %ymm13,32(%rsp)+ jmp .Loop_tail8x++.align 32+.L192_or_more8x:+ vpxor 0(%rsi),%ymm6,%ymm6+ vpxor 32(%rsi),%ymm8,%ymm8+ vpxor 64(%rsi),%ymm1,%ymm1+ vpxor 96(%rsi),%ymm5,%ymm5+ vpxor 128(%rsi),%ymm12,%ymm12+ vpxor 160(%rsi),%ymm13,%ymm13+ vmovdqu %ymm6,0(%rdi)+ vmovdqu %ymm8,32(%rdi)+ vmovdqu %ymm1,64(%rdi)+ vmovdqu %ymm5,96(%rdi)+ vmovdqu %ymm12,128(%rdi)+ vmovdqu %ymm13,160(%rdi)+ je .Ldone8x++ leaq 192(%rsi),%rsi+ xorq %r9,%r9+ vmovdqa %ymm10,0(%rsp)+ leaq 192(%rdi),%rdi+ subq $192,%rdx+ vmovdqa %ymm15,32(%rsp)+ jmp .Loop_tail8x++.align 32+.L256_or_more8x:+ vpxor 0(%rsi),%ymm6,%ymm6+ vpxor 32(%rsi),%ymm8,%ymm8+ vpxor 64(%rsi),%ymm1,%ymm1+ vpxor 96(%rsi),%ymm5,%ymm5+ vpxor 128(%rsi),%ymm12,%ymm12+ vpxor 160(%rsi),%ymm13,%ymm13+ vpxor 192(%rsi),%ymm10,%ymm10+ vpxor 224(%rsi),%ymm15,%ymm15+ vmovdqu %ymm6,0(%rdi)+ vmovdqu %ymm8,32(%rdi)+ vmovdqu %ymm1,64(%rdi)+ vmovdqu %ymm5,96(%rdi)+ vmovdqu %ymm12,128(%rdi)+ vmovdqu %ymm13,160(%rdi)+ vmovdqu %ymm10,192(%rdi)+ vmovdqu %ymm15,224(%rdi)+ je .Ldone8x++ leaq 256(%rsi),%rsi+ xorq %r9,%r9+ vmovdqa %ymm14,0(%rsp)+ leaq 256(%rdi),%rdi+ subq $256,%rdx+ vmovdqa %ymm2,32(%rsp)+ jmp .Loop_tail8x++.align 32+.L320_or_more8x:+ vpxor 0(%rsi),%ymm6,%ymm6+ vpxor 32(%rsi),%ymm8,%ymm8+ vpxor 64(%rsi),%ymm1,%ymm1+ vpxor 96(%rsi),%ymm5,%ymm5+ vpxor 128(%rsi),%ymm12,%ymm12+ vpxor 160(%rsi),%ymm13,%ymm13+ vpxor 192(%rsi),%ymm10,%ymm10+ vpxor 224(%rsi),%ymm15,%ymm15+ vpxor 256(%rsi),%ymm14,%ymm14+ vpxor 288(%rsi),%ymm2,%ymm2+ vmovdqu %ymm6,0(%rdi)+ vmovdqu %ymm8,32(%rdi)+ vmovdqu %ymm1,64(%rdi)+ vmovdqu %ymm5,96(%rdi)+ vmovdqu %ymm12,128(%rdi)+ vmovdqu %ymm13,160(%rdi)+ vmovdqu %ymm10,192(%rdi)+ vmovdqu %ymm15,224(%rdi)+ vmovdqu %ymm14,256(%rdi)+ vmovdqu %ymm2,288(%rdi)+ je .Ldone8x++ leaq 320(%rsi),%rsi+ xorq %r9,%r9+ vmovdqa %ymm3,0(%rsp)+ leaq 320(%rdi),%rdi+ subq $320,%rdx+ vmovdqa %ymm7,32(%rsp)+ jmp .Loop_tail8x++.align 32+.L384_or_more8x:+ vpxor 0(%rsi),%ymm6,%ymm6+ vpxor 32(%rsi),%ymm8,%ymm8+ vpxor 64(%rsi),%ymm1,%ymm1+ vpxor 96(%rsi),%ymm5,%ymm5+ vpxor 128(%rsi),%ymm12,%ymm12+ vpxor 160(%rsi),%ymm13,%ymm13+ vpxor 192(%rsi),%ymm10,%ymm10+ vpxor 224(%rsi),%ymm15,%ymm15+ vpxor 256(%rsi),%ymm14,%ymm14+ vpxor 288(%rsi),%ymm2,%ymm2+ vpxor 320(%rsi),%ymm3,%ymm3+ vpxor 352(%rsi),%ymm7,%ymm7+ vmovdqu %ymm6,0(%rdi)+ vmovdqu %ymm8,32(%rdi)+ vmovdqu %ymm1,64(%rdi)+ vmovdqu %ymm5,96(%rdi)+ vmovdqu %ymm12,128(%rdi)+ vmovdqu %ymm13,160(%rdi)+ vmovdqu %ymm10,192(%rdi)+ vmovdqu %ymm15,224(%rdi)+ vmovdqu %ymm14,256(%rdi)+ vmovdqu %ymm2,288(%rdi)+ vmovdqu %ymm3,320(%rdi)+ vmovdqu %ymm7,352(%rdi)+ je .Ldone8x++ leaq 384(%rsi),%rsi+ xorq %r9,%r9+ vmovdqa %ymm11,0(%rsp)+ leaq 384(%rdi),%rdi+ subq $384,%rdx+ vmovdqa %ymm9,32(%rsp)+ jmp .Loop_tail8x++.align 32+.L448_or_more8x:+ vpxor 0(%rsi),%ymm6,%ymm6+ vpxor 32(%rsi),%ymm8,%ymm8+ vpxor 64(%rsi),%ymm1,%ymm1+ vpxor 96(%rsi),%ymm5,%ymm5+ vpxor 128(%rsi),%ymm12,%ymm12+ vpxor 160(%rsi),%ymm13,%ymm13+ vpxor 192(%rsi),%ymm10,%ymm10+ vpxor 224(%rsi),%ymm15,%ymm15+ vpxor 256(%rsi),%ymm14,%ymm14+ vpxor 288(%rsi),%ymm2,%ymm2+ vpxor 320(%rsi),%ymm3,%ymm3+ vpxor 352(%rsi),%ymm7,%ymm7+ vpxor 384(%rsi),%ymm11,%ymm11+ vpxor 416(%rsi),%ymm9,%ymm9+ vmovdqu %ymm6,0(%rdi)+ vmovdqu %ymm8,32(%rdi)+ vmovdqu %ymm1,64(%rdi)+ vmovdqu %ymm5,96(%rdi)+ vmovdqu %ymm12,128(%rdi)+ vmovdqu %ymm13,160(%rdi)+ vmovdqu %ymm10,192(%rdi)+ vmovdqu %ymm15,224(%rdi)+ vmovdqu %ymm14,256(%rdi)+ vmovdqu %ymm2,288(%rdi)+ vmovdqu %ymm3,320(%rdi)+ vmovdqu %ymm7,352(%rdi)+ vmovdqu %ymm11,384(%rdi)+ vmovdqu %ymm9,416(%rdi)+ je .Ldone8x++ leaq 448(%rsi),%rsi+ xorq %r9,%r9+ vmovdqa %ymm0,0(%rsp)+ leaq 448(%rdi),%rdi+ subq $448,%rdx+ vmovdqa %ymm4,32(%rsp)++.Loop_tail8x:+ movzbl (%rsi,%r9,1),%eax+ movzbl (%rsp,%r9,1),%ecx+ leaq 1(%r9),%r9+ xorl %ecx,%eax+ movb %al,-1(%rdi,%r9,1)+ decq %rdx+ jnz .Loop_tail8x++.Ldone8x:+ vzeroall+ leaq (%r10),%rsp+.cfi_def_cfa_register %rsp+.Lavx2_epilogue:+ .byte 0xf3,0xc3+.cfi_endproc +.size crypton_chacha20_asm_avx2,.-crypton_chacha20_asm_avx2++.section .note.gnu.property,"a",@note+ .long 4,2f-1f,5+ .byte 0x47,0x4E,0x55,0+1: .long 0xc0000002,4,3+.align 8+2:++.section .note.GNU-stack,"",@progbits
@@ -0,0 +1,2232 @@+.text ++++.p2align 6+L$zero:+.long 0,0,0,0+L$one:+.long 1,0,0,0+L$inc:+.long 0,1,2,3+L$four:+.long 4,4,4,4+L$incy:+.long 0,2,4,6,1,3,5,7+L$eight:+.long 8,8,8,8,8,8,8,8+L$rot16:+.byte 0x2,0x3,0x0,0x1, 0x6,0x7,0x4,0x5, 0xa,0xb,0x8,0x9, 0xe,0xf,0xc,0xd+L$rot24:+.byte 0x3,0x0,0x1,0x2, 0x7,0x4,0x5,0x6, 0xb,0x8,0x9,0xa, 0xf,0xc,0xd,0xe+L$twoy:+.long 2,0,0,0, 2,0,0,0+.p2align 6+L$zeroz:+.long 0,0,0,0, 1,0,0,0, 2,0,0,0, 3,0,0,0+L$fourz:+.long 4,0,0,0, 4,0,0,0, 4,0,0,0, 4,0,0,0+L$incz:+.long 0,1,2,3,4,5,6,7,8,9,10,11,12,13,14,15+L$sixteen:+.long 16,16,16,16,16,16,16,16,16,16,16,16,16,16,16,16+L$sigma:+.byte 101,120,112,97,110,100,32,51,50,45,98,121,116,101,32,107,0+.byte 67,104,97,67,104,97,50,48,32,102,111,114,32,120,56,54,95,54,52,44,32,67,82,89,80,84,79,71,65,77,83,32,98,121,32,64,100,111,116,45,97,115,109,0+.globl _crypton_chacha20_asm_ctr32++.p2align 6+_crypton_chacha20_asm_ctr32:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa+++ cmpq $0,%rdx+ je L$no_data+ movq _crypton_ia32cap_P+4(%rip),%r9+ testl $512,%r9d+ jnz L$crypton_chacha20_asm_ssse3+ pushq %rbx+.cfi_adjust_cfa_offset 8+.cfi_offset %rbx,-16+ pushq %rbp+.cfi_adjust_cfa_offset 8+.cfi_offset %rbp,-24+ pushq %r12+.cfi_adjust_cfa_offset 8+.cfi_offset %r12,-32+ pushq %r13+.cfi_adjust_cfa_offset 8+.cfi_offset %r13,-40+ pushq %r14+.cfi_adjust_cfa_offset 8+.cfi_offset %r14,-48+ pushq %r15+.cfi_adjust_cfa_offset 8+.cfi_offset %r15,-56+ subq $64+24,%rsp+.cfi_adjust_cfa_offset 88+L$ctr32_body:++ movq %rdx,%rbp++ movq 0(%rcx),%r12+ movq 8(%rcx),%r13+ movq 16(%rcx),%r14+ movq 24(%rcx),%r15+ movq 0(%r8),%rax+ movq 8(%r8),%rdx+ movq %r12,16(%rsp)+ movq %r13,24(%rsp)+ movq %r14,0(%rsp)+ movq %r15,8(%rsp)+ movq %rax,48(%rsp)+ movq %rdx,56(%rsp)+ jmp L$oop_outer++.p2align 5+L$oop_outer:+ movl $0x61707865,%eax+ movl $0x3320646e,%ebx+ movl $0x79622d32,%ecx+ movl $0x6b206574,%edx+ movl 16(%rsp),%r8d+ movl 20(%rsp),%r9d+ movl 24(%rsp),%r10d+ movl 28(%rsp),%r11d+ movl 48(%rsp),%r12d+ movl 52(%rsp),%r13d+ movl 56(%rsp),%r14d+ movq %r15,40(%rsp)+ movl 60(%rsp),%r15d++ movq %rbp,64+0(%rsp)+ movq %rsi,64+8(%rsp)+ movl 0(%rsp),%esi+ movq %rdi,64+16(%rsp)+ movl 4(%rsp),%edi+ movl $10,%ebp+ jmp L$oop++.p2align 5+L$oop:+ addl %r8d,%eax+ xorl %eax,%r12d+ roll $16,%r12d+ addl %r9d,%ebx+ xorl %ebx,%r13d+ roll $16,%r13d+ addl %r12d,%esi+ xorl %esi,%r8d+ roll $12,%r8d+ addl %r13d,%edi+ xorl %edi,%r9d+ roll $12,%r9d+ addl %r8d,%eax+ xorl %eax,%r12d+ roll $8,%r12d+ addl %r9d,%ebx+ xorl %ebx,%r13d+ roll $8,%r13d+ addl %r12d,%esi+ xorl %esi,%r8d+ roll $7,%r8d+ addl %r13d,%edi+ xorl %edi,%r9d+ roll $7,%r9d+ movl %esi,32(%rsp)+ movl %edi,36(%rsp)+ movl 40(%rsp),%esi+ movl 44(%rsp),%edi+ addl %r10d,%ecx+ xorl %ecx,%r14d+ roll $16,%r14d+ addl %r11d,%edx+ xorl %edx,%r15d+ roll $16,%r15d+ addl %r14d,%esi+ xorl %esi,%r10d+ roll $12,%r10d+ addl %r15d,%edi+ xorl %edi,%r11d+ roll $12,%r11d+ addl %r10d,%ecx+ xorl %ecx,%r14d+ roll $8,%r14d+ addl %r11d,%edx+ xorl %edx,%r15d+ roll $8,%r15d+ addl %r14d,%esi+ xorl %esi,%r10d+ roll $7,%r10d+ addl %r15d,%edi+ xorl %edi,%r11d+ roll $7,%r11d+ addl %r9d,%eax+ xorl %eax,%r15d+ roll $16,%r15d+ addl %r10d,%ebx+ xorl %ebx,%r12d+ roll $16,%r12d+ addl %r15d,%esi+ xorl %esi,%r9d+ roll $12,%r9d+ addl %r12d,%edi+ xorl %edi,%r10d+ roll $12,%r10d+ addl %r9d,%eax+ xorl %eax,%r15d+ roll $8,%r15d+ addl %r10d,%ebx+ xorl %ebx,%r12d+ roll $8,%r12d+ addl %r15d,%esi+ xorl %esi,%r9d+ roll $7,%r9d+ addl %r12d,%edi+ xorl %edi,%r10d+ roll $7,%r10d+ movl %esi,40(%rsp)+ movl %edi,44(%rsp)+ movl 32(%rsp),%esi+ movl 36(%rsp),%edi+ addl %r11d,%ecx+ xorl %ecx,%r13d+ roll $16,%r13d+ addl %r8d,%edx+ xorl %edx,%r14d+ roll $16,%r14d+ addl %r13d,%esi+ xorl %esi,%r11d+ roll $12,%r11d+ addl %r14d,%edi+ xorl %edi,%r8d+ roll $12,%r8d+ addl %r11d,%ecx+ xorl %ecx,%r13d+ roll $8,%r13d+ addl %r8d,%edx+ xorl %edx,%r14d+ roll $8,%r14d+ addl %r13d,%esi+ xorl %esi,%r11d+ roll $7,%r11d+ addl %r14d,%edi+ xorl %edi,%r8d+ roll $7,%r8d+ decl %ebp+ jnz L$oop+ addl 0(%rsp),%esi+ addl 4(%rsp),%edi+ movq 64(%rsp),%rbp+ movl %esi,32(%rsp)+ movq 64+8(%rsp),%rsi+ movl %edi,36(%rsp)+ movq 64+16(%rsp),%rdi++ addl $0x61707865,%eax+ addl $0x3320646e,%ebx+ addl $0x79622d32,%ecx+ addl $0x6b206574,%edx+ addl 16(%rsp),%r8d+ addl 20(%rsp),%r9d+ addl 24(%rsp),%r10d+ addl 28(%rsp),%r11d+ addl 48(%rsp),%r12d+ addl 52(%rsp),%r13d+ addl 56(%rsp),%r14d+ addl 60(%rsp),%r15d++ cmpq $64,%rbp+ jb L$tail++ xorl 0(%rsi),%eax+ xorl 4(%rsi),%ebx+ xorl 8(%rsi),%ecx+ xorl 12(%rsi),%edx+ movl %eax,0(%rdi)+ movl 32(%rsp),%eax+ movl %ebx,4(%rdi)+ movl 36(%rsp),%ebx+ movl %ecx,8(%rdi)+ movl 40(%rsp),%ecx+ movl %edx,12(%rdi)+ movl 44(%rsp),%edx+ xorl 16(%rsi),%r8d+ addl 8(%rsp),%ecx+ xorl 20(%rsi),%r9d+ addl 12(%rsp),%edx+ xorl 24(%rsi),%r10d+ xorl 28(%rsi),%r11d+ xorl 32(%rsi),%eax+ xorl 36(%rsi),%ebx+ xorl 40(%rsi),%ecx+ xorl 44(%rsi),%edx+ xorl 48(%rsi),%r12d+ xorl 52(%rsi),%r13d+ xorl 56(%rsi),%r14d+ xorl 60(%rsi),%r15d+ leaq 64(%rsi),%rsi++ addl $1,48(%rsp)++ movl %r8d,16(%rdi)+ movl %r9d,20(%rdi)+ movl %r10d,24(%rdi)+ movl %r11d,28(%rdi)+ movl %eax,32(%rdi)+ movl %ebx,36(%rdi)+ movl %ecx,40(%rdi)+ movl %edx,44(%rdi)+ movl %r12d,48(%rdi)+ movl %r13d,52(%rdi)+ movl %r14d,56(%rdi)+ movl %r15d,60(%rdi)+ leaq 64(%rdi),%rdi+ movq 8(%rsp),%r15++ subq $64,%rbp+ jnz L$oop_outer++ jmp L$done++.p2align 4+L$tail:+ movl %eax,0(%rsp)+ movl 8(%rsp),%eax+ movl %ebx,4(%rsp)+ movl 12(%rsp),%ebx+ movl %ecx,8(%rsp)+ addl 40(%rsp),%eax+ movl %edx,12(%rsp)+ addl 44(%rsp),%ebx+ movl %r8d,16(%rsp)+ movl %r9d,20(%rsp)+ movl %r10d,24(%rsp)+ movl %r11d,28(%rsp)+ movl %eax,40(%rsp)+ movl %ebx,44(%rsp)+ xorq %rbx,%rbx+ movl %r12d,48(%rsp)+ movl %r13d,52(%rsp)+ movl %r14d,56(%rsp)+ movl %r15d,60(%rsp)++L$oop_tail:+ movzbl (%rsi,%rbx,1),%eax+ movzbl (%rsp,%rbx,1),%edx+ leaq 1(%rbx),%rbx+ xorl %edx,%eax+ movb %al,-1(%rdi,%rbx,1)+ decq %rbp+ jnz L$oop_tail++L$done:+ leaq 64+24+48(%rsp),%rsi+.cfi_def_cfa %rsi,8+ movq -48(%rsi),%r15+.cfi_restore %r15+ movq -40(%rsi),%r14+.cfi_restore %r14+ movq -32(%rsi),%r13+.cfi_restore %r13+ movq -24(%rsi),%r12+.cfi_restore %r12+ movq -16(%rsi),%rbp+.cfi_restore %rbp+ movq -8(%rsi),%rbx+.cfi_restore %rbx+ leaq (%rsi),%rsp+.cfi_def_cfa_register %rsp+L$no_data:+ .byte 0xf3,0xc3+.cfi_endproc +++.p2align 5+crypton_chacha20_asm_ssse3:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa+++L$crypton_chacha20_asm_ssse3:+ movq %rsp,%r10+.cfi_def_cfa_register %r10+ testl $2048,%r9d+ jnz L$crypton_chacha20_asm_4xop+ cmpq $128,%rdx+ je L$crypton_chacha20_asm_128+ ja L$crypton_chacha20_asm_4x++L$do_sse3_after_all:+ subq $64+8,%rsp+ andq $-16,%rsp+ movdqa L$sigma(%rip),%xmm0+ movdqu (%rcx),%xmm1+ movdqu 16(%rcx),%xmm2+ movdqu (%r8),%xmm3+ movdqa L$rot16(%rip),%xmm6+ movdqa L$rot24(%rip),%xmm7++ movdqa %xmm0,0(%rsp)+ movdqa %xmm1,16(%rsp)+ movdqa %xmm2,32(%rsp)+ movdqa %xmm3,48(%rsp)+ movq $10,%r8+ jmp L$oop_ssse3++.p2align 5+L$oop_outer_ssse3:+ movdqa L$one(%rip),%xmm3+ movdqa 0(%rsp),%xmm0+ movdqa 16(%rsp),%xmm1+ movdqa 32(%rsp),%xmm2+ paddd 48(%rsp),%xmm3+ movq $10,%r8+ movdqa %xmm3,48(%rsp)+ jmp L$oop_ssse3++.p2align 5+L$oop_ssse3:+ paddd %xmm1,%xmm0+ pxor %xmm0,%xmm3+.byte 102,15,56,0,222+ paddd %xmm3,%xmm2+ pxor %xmm2,%xmm1+ movdqa %xmm1,%xmm4+ psrld $20,%xmm1+ pslld $12,%xmm4+ por %xmm4,%xmm1+ paddd %xmm1,%xmm0+ pxor %xmm0,%xmm3+.byte 102,15,56,0,223+ paddd %xmm3,%xmm2+ pxor %xmm2,%xmm1+ movdqa %xmm1,%xmm4+ psrld $25,%xmm1+ pslld $7,%xmm4+ por %xmm4,%xmm1+ pshufd $78,%xmm2,%xmm2+ pshufd $57,%xmm1,%xmm1+ pshufd $147,%xmm3,%xmm3+ nop+ paddd %xmm1,%xmm0+ pxor %xmm0,%xmm3+.byte 102,15,56,0,222+ paddd %xmm3,%xmm2+ pxor %xmm2,%xmm1+ movdqa %xmm1,%xmm4+ psrld $20,%xmm1+ pslld $12,%xmm4+ por %xmm4,%xmm1+ paddd %xmm1,%xmm0+ pxor %xmm0,%xmm3+.byte 102,15,56,0,223+ paddd %xmm3,%xmm2+ pxor %xmm2,%xmm1+ movdqa %xmm1,%xmm4+ psrld $25,%xmm1+ pslld $7,%xmm4+ por %xmm4,%xmm1+ pshufd $78,%xmm2,%xmm2+ pshufd $147,%xmm1,%xmm1+ pshufd $57,%xmm3,%xmm3+ decq %r8+ jnz L$oop_ssse3+ paddd 0(%rsp),%xmm0+ paddd 16(%rsp),%xmm1+ paddd 32(%rsp),%xmm2+ paddd 48(%rsp),%xmm3++ cmpq $64,%rdx+ jb L$tail_ssse3++ movdqu 0(%rsi),%xmm4+ movdqu 16(%rsi),%xmm5+ pxor %xmm4,%xmm0+ movdqu 32(%rsi),%xmm4+ pxor %xmm5,%xmm1+ movdqu 48(%rsi),%xmm5+ leaq 64(%rsi),%rsi+ pxor %xmm4,%xmm2+ pxor %xmm5,%xmm3++ movdqu %xmm0,0(%rdi)+ movdqu %xmm1,16(%rdi)+ movdqu %xmm2,32(%rdi)+ movdqu %xmm3,48(%rdi)+ leaq 64(%rdi),%rdi++ subq $64,%rdx+ jnz L$oop_outer_ssse3++ jmp L$done_ssse3++.p2align 4+L$tail_ssse3:+ movdqa %xmm0,0(%rsp)+ movdqa %xmm1,16(%rsp)+ movdqa %xmm2,32(%rsp)+ movdqa %xmm3,48(%rsp)+ xorq %r8,%r8++L$oop_tail_ssse3:+ movzbl (%rsi,%r8,1),%eax+ movzbl (%rsp,%r8,1),%ecx+ leaq 1(%r8),%r8+ xorl %ecx,%eax+ movb %al,-1(%rdi,%r8,1)+ decq %rdx+ jnz L$oop_tail_ssse3++L$done_ssse3:+ leaq (%r10),%rsp+.cfi_def_cfa_register %rsp+L$ssse3_epilogue:+ .byte 0xf3,0xc3+.cfi_endproc +++.p2align 5+crypton_chacha20_asm_128:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa+++L$crypton_chacha20_asm_128:+ movq %rsp,%r10+.cfi_def_cfa_register %r10+ subq $64+8,%rsp+ andq $-16,%rsp+ movdqa L$sigma(%rip),%xmm8+ movdqu (%rcx),%xmm9+ movdqu 16(%rcx),%xmm2+ movdqu (%r8),%xmm3+ movdqa L$one(%rip),%xmm1+ movdqa L$rot16(%rip),%xmm6+ movdqa L$rot24(%rip),%xmm7++ movdqa %xmm8,%xmm10+ movdqa %xmm8,0(%rsp)+ movdqa %xmm9,%xmm11+ movdqa %xmm9,16(%rsp)+ movdqa %xmm2,%xmm0+ movdqa %xmm2,32(%rsp)+ paddd %xmm3,%xmm1+ movdqa %xmm3,48(%rsp)+ movq $10,%r8+ jmp L$oop_128++.p2align 5+L$oop_128:+ paddd %xmm9,%xmm8+ pxor %xmm8,%xmm3+ paddd %xmm11,%xmm10+ pxor %xmm10,%xmm1+.byte 102,15,56,0,222+.byte 102,15,56,0,206+ paddd %xmm3,%xmm2+ paddd %xmm1,%xmm0+ pxor %xmm2,%xmm9+ pxor %xmm0,%xmm11+ movdqa %xmm9,%xmm4+ psrld $20,%xmm9+ movdqa %xmm11,%xmm5+ pslld $12,%xmm4+ psrld $20,%xmm11+ por %xmm4,%xmm9+ pslld $12,%xmm5+ por %xmm5,%xmm11+ paddd %xmm9,%xmm8+ pxor %xmm8,%xmm3+ paddd %xmm11,%xmm10+ pxor %xmm10,%xmm1+.byte 102,15,56,0,223+.byte 102,15,56,0,207+ paddd %xmm3,%xmm2+ paddd %xmm1,%xmm0+ pxor %xmm2,%xmm9+ pxor %xmm0,%xmm11+ movdqa %xmm9,%xmm4+ psrld $25,%xmm9+ movdqa %xmm11,%xmm5+ pslld $7,%xmm4+ psrld $25,%xmm11+ por %xmm4,%xmm9+ pslld $7,%xmm5+ por %xmm5,%xmm11+ pshufd $78,%xmm2,%xmm2+ pshufd $57,%xmm9,%xmm9+ pshufd $147,%xmm3,%xmm3+ pshufd $78,%xmm0,%xmm0+ pshufd $57,%xmm11,%xmm11+ pshufd $147,%xmm1,%xmm1+ paddd %xmm9,%xmm8+ pxor %xmm8,%xmm3+ paddd %xmm11,%xmm10+ pxor %xmm10,%xmm1+.byte 102,15,56,0,222+.byte 102,15,56,0,206+ paddd %xmm3,%xmm2+ paddd %xmm1,%xmm0+ pxor %xmm2,%xmm9+ pxor %xmm0,%xmm11+ movdqa %xmm9,%xmm4+ psrld $20,%xmm9+ movdqa %xmm11,%xmm5+ pslld $12,%xmm4+ psrld $20,%xmm11+ por %xmm4,%xmm9+ pslld $12,%xmm5+ por %xmm5,%xmm11+ paddd %xmm9,%xmm8+ pxor %xmm8,%xmm3+ paddd %xmm11,%xmm10+ pxor %xmm10,%xmm1+.byte 102,15,56,0,223+.byte 102,15,56,0,207+ paddd %xmm3,%xmm2+ paddd %xmm1,%xmm0+ pxor %xmm2,%xmm9+ pxor %xmm0,%xmm11+ movdqa %xmm9,%xmm4+ psrld $25,%xmm9+ movdqa %xmm11,%xmm5+ pslld $7,%xmm4+ psrld $25,%xmm11+ por %xmm4,%xmm9+ pslld $7,%xmm5+ por %xmm5,%xmm11+ pshufd $78,%xmm2,%xmm2+ pshufd $147,%xmm9,%xmm9+ pshufd $57,%xmm3,%xmm3+ pshufd $78,%xmm0,%xmm0+ pshufd $147,%xmm11,%xmm11+ pshufd $57,%xmm1,%xmm1+ decq %r8+ jnz L$oop_128+ paddd 0(%rsp),%xmm8+ paddd 16(%rsp),%xmm9+ paddd 32(%rsp),%xmm2+ paddd 48(%rsp),%xmm3+ paddd L$one(%rip),%xmm1+ paddd 0(%rsp),%xmm10+ paddd 16(%rsp),%xmm11+ paddd 32(%rsp),%xmm0+ paddd 48(%rsp),%xmm1++ movdqu 0(%rsi),%xmm4+ movdqu 16(%rsi),%xmm5+ pxor %xmm4,%xmm8+ movdqu 32(%rsi),%xmm4+ pxor %xmm5,%xmm9+ movdqu 48(%rsi),%xmm5+ pxor %xmm4,%xmm2+ movdqu 64(%rsi),%xmm4+ pxor %xmm5,%xmm3+ movdqu 80(%rsi),%xmm5+ pxor %xmm4,%xmm10+ movdqu 96(%rsi),%xmm4+ pxor %xmm5,%xmm11+ movdqu 112(%rsi),%xmm5+ pxor %xmm4,%xmm0+ pxor %xmm5,%xmm1++ movdqu %xmm8,0(%rdi)+ movdqu %xmm9,16(%rdi)+ movdqu %xmm2,32(%rdi)+ movdqu %xmm3,48(%rdi)+ movdqu %xmm10,64(%rdi)+ movdqu %xmm11,80(%rdi)+ movdqu %xmm0,96(%rdi)+ movdqu %xmm1,112(%rdi)+ leaq (%r10),%rsp+.cfi_def_cfa_register %rsp+L$128_epilogue:+ .byte 0xf3,0xc3+.cfi_endproc +++.p2align 5+crypton_chacha20_asm_4x:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa+++L$crypton_chacha20_asm_4x:+ movq %rsp,%r10+.cfi_def_cfa_register %r10+ movq %r9,%r11+ shrq $32,%r9+ testq $32,%r9+ jnz L$crypton_chacha20_asm_8x+ cmpq $192,%rdx+ ja L$proceed4x++ andq $71303168,%r11+ cmpq $4194304,%r11+ je L$do_sse3_after_all++L$proceed4x:+ subq $0x140+8,%rsp+ andq $-16,%rsp+ movdqa L$sigma(%rip),%xmm11+ movdqu (%rcx),%xmm15+ movdqu 16(%rcx),%xmm7+ movdqu (%r8),%xmm3+ leaq 256(%rsp),%rcx+ leaq L$rot16(%rip),%r9+ leaq L$rot24(%rip),%r11++ pshufd $0x00,%xmm11,%xmm8+ pshufd $0x55,%xmm11,%xmm9+ movdqa %xmm8,64(%rsp)+ pshufd $0xaa,%xmm11,%xmm10+ movdqa %xmm9,80(%rsp)+ pshufd $0xff,%xmm11,%xmm11+ movdqa %xmm10,96(%rsp)+ movdqa %xmm11,112(%rsp)++ pshufd $0x00,%xmm15,%xmm12+ pshufd $0x55,%xmm15,%xmm13+ movdqa %xmm12,128-256(%rcx)+ pshufd $0xaa,%xmm15,%xmm14+ movdqa %xmm13,144-256(%rcx)+ pshufd $0xff,%xmm15,%xmm15+ movdqa %xmm14,160-256(%rcx)+ movdqa %xmm15,176-256(%rcx)++ pshufd $0x00,%xmm7,%xmm4+ pshufd $0x55,%xmm7,%xmm5+ movdqa %xmm4,192-256(%rcx)+ pshufd $0xaa,%xmm7,%xmm6+ movdqa %xmm5,208-256(%rcx)+ pshufd $0xff,%xmm7,%xmm7+ movdqa %xmm6,224-256(%rcx)+ movdqa %xmm7,240-256(%rcx)++ pshufd $0x00,%xmm3,%xmm0+ pshufd $0x55,%xmm3,%xmm1+ paddd L$inc(%rip),%xmm0+ pshufd $0xaa,%xmm3,%xmm2+ movdqa %xmm1,272-256(%rcx)+ pshufd $0xff,%xmm3,%xmm3+ movdqa %xmm2,288-256(%rcx)+ movdqa %xmm3,304-256(%rcx)++ jmp L$oop_enter4x++.p2align 5+L$oop_outer4x:+ movdqa 64(%rsp),%xmm8+ movdqa 80(%rsp),%xmm9+ movdqa 96(%rsp),%xmm10+ movdqa 112(%rsp),%xmm11+ movdqa 128-256(%rcx),%xmm12+ movdqa 144-256(%rcx),%xmm13+ movdqa 160-256(%rcx),%xmm14+ movdqa 176-256(%rcx),%xmm15+ movdqa 192-256(%rcx),%xmm4+ movdqa 208-256(%rcx),%xmm5+ movdqa 224-256(%rcx),%xmm6+ movdqa 240-256(%rcx),%xmm7+ movdqa 256-256(%rcx),%xmm0+ movdqa 272-256(%rcx),%xmm1+ movdqa 288-256(%rcx),%xmm2+ movdqa 304-256(%rcx),%xmm3+ paddd L$four(%rip),%xmm0++L$oop_enter4x:+ movdqa %xmm6,32(%rsp)+ movdqa %xmm7,48(%rsp)+ movdqa (%r9),%xmm7+ movl $10,%eax+ movdqa %xmm0,256-256(%rcx)+ jmp L$oop4x++.p2align 5+L$oop4x:+ paddd %xmm12,%xmm8+ paddd %xmm13,%xmm9+ pxor %xmm8,%xmm0+ pxor %xmm9,%xmm1+.byte 102,15,56,0,199+.byte 102,15,56,0,207+ paddd %xmm0,%xmm4+ paddd %xmm1,%xmm5+ pxor %xmm4,%xmm12+ pxor %xmm5,%xmm13+ movdqa %xmm12,%xmm6+ pslld $12,%xmm12+ psrld $20,%xmm6+ movdqa %xmm13,%xmm7+ pslld $12,%xmm13+ por %xmm6,%xmm12+ psrld $20,%xmm7+ movdqa (%r11),%xmm6+ por %xmm7,%xmm13+ paddd %xmm12,%xmm8+ paddd %xmm13,%xmm9+ pxor %xmm8,%xmm0+ pxor %xmm9,%xmm1+.byte 102,15,56,0,198+.byte 102,15,56,0,206+ paddd %xmm0,%xmm4+ paddd %xmm1,%xmm5+ pxor %xmm4,%xmm12+ pxor %xmm5,%xmm13+ movdqa %xmm12,%xmm7+ pslld $7,%xmm12+ psrld $25,%xmm7+ movdqa %xmm13,%xmm6+ pslld $7,%xmm13+ por %xmm7,%xmm12+ psrld $25,%xmm6+ movdqa (%r9),%xmm7+ por %xmm6,%xmm13+ movdqa %xmm4,0(%rsp)+ movdqa %xmm5,16(%rsp)+ movdqa 32(%rsp),%xmm4+ movdqa 48(%rsp),%xmm5+ paddd %xmm14,%xmm10+ paddd %xmm15,%xmm11+ pxor %xmm10,%xmm2+ pxor %xmm11,%xmm3+.byte 102,15,56,0,215+.byte 102,15,56,0,223+ paddd %xmm2,%xmm4+ paddd %xmm3,%xmm5+ pxor %xmm4,%xmm14+ pxor %xmm5,%xmm15+ movdqa %xmm14,%xmm6+ pslld $12,%xmm14+ psrld $20,%xmm6+ movdqa %xmm15,%xmm7+ pslld $12,%xmm15+ por %xmm6,%xmm14+ psrld $20,%xmm7+ movdqa (%r11),%xmm6+ por %xmm7,%xmm15+ paddd %xmm14,%xmm10+ paddd %xmm15,%xmm11+ pxor %xmm10,%xmm2+ pxor %xmm11,%xmm3+.byte 102,15,56,0,214+.byte 102,15,56,0,222+ paddd %xmm2,%xmm4+ paddd %xmm3,%xmm5+ pxor %xmm4,%xmm14+ pxor %xmm5,%xmm15+ movdqa %xmm14,%xmm7+ pslld $7,%xmm14+ psrld $25,%xmm7+ movdqa %xmm15,%xmm6+ pslld $7,%xmm15+ por %xmm7,%xmm14+ psrld $25,%xmm6+ movdqa (%r9),%xmm7+ por %xmm6,%xmm15+ paddd %xmm13,%xmm8+ paddd %xmm14,%xmm9+ pxor %xmm8,%xmm3+ pxor %xmm9,%xmm0+.byte 102,15,56,0,223+.byte 102,15,56,0,199+ paddd %xmm3,%xmm4+ paddd %xmm0,%xmm5+ pxor %xmm4,%xmm13+ pxor %xmm5,%xmm14+ movdqa %xmm13,%xmm6+ pslld $12,%xmm13+ psrld $20,%xmm6+ movdqa %xmm14,%xmm7+ pslld $12,%xmm14+ por %xmm6,%xmm13+ psrld $20,%xmm7+ movdqa (%r11),%xmm6+ por %xmm7,%xmm14+ paddd %xmm13,%xmm8+ paddd %xmm14,%xmm9+ pxor %xmm8,%xmm3+ pxor %xmm9,%xmm0+.byte 102,15,56,0,222+.byte 102,15,56,0,198+ paddd %xmm3,%xmm4+ paddd %xmm0,%xmm5+ pxor %xmm4,%xmm13+ pxor %xmm5,%xmm14+ movdqa %xmm13,%xmm7+ pslld $7,%xmm13+ psrld $25,%xmm7+ movdqa %xmm14,%xmm6+ pslld $7,%xmm14+ por %xmm7,%xmm13+ psrld $25,%xmm6+ movdqa (%r9),%xmm7+ por %xmm6,%xmm14+ movdqa %xmm4,32(%rsp)+ movdqa %xmm5,48(%rsp)+ movdqa 0(%rsp),%xmm4+ movdqa 16(%rsp),%xmm5+ paddd %xmm15,%xmm10+ paddd %xmm12,%xmm11+ pxor %xmm10,%xmm1+ pxor %xmm11,%xmm2+.byte 102,15,56,0,207+.byte 102,15,56,0,215+ paddd %xmm1,%xmm4+ paddd %xmm2,%xmm5+ pxor %xmm4,%xmm15+ pxor %xmm5,%xmm12+ movdqa %xmm15,%xmm6+ pslld $12,%xmm15+ psrld $20,%xmm6+ movdqa %xmm12,%xmm7+ pslld $12,%xmm12+ por %xmm6,%xmm15+ psrld $20,%xmm7+ movdqa (%r11),%xmm6+ por %xmm7,%xmm12+ paddd %xmm15,%xmm10+ paddd %xmm12,%xmm11+ pxor %xmm10,%xmm1+ pxor %xmm11,%xmm2+.byte 102,15,56,0,206+.byte 102,15,56,0,214+ paddd %xmm1,%xmm4+ paddd %xmm2,%xmm5+ pxor %xmm4,%xmm15+ pxor %xmm5,%xmm12+ movdqa %xmm15,%xmm7+ pslld $7,%xmm15+ psrld $25,%xmm7+ movdqa %xmm12,%xmm6+ pslld $7,%xmm12+ por %xmm7,%xmm15+ psrld $25,%xmm6+ movdqa (%r9),%xmm7+ por %xmm6,%xmm12+ decl %eax+ jnz L$oop4x++ paddd 64(%rsp),%xmm8+ paddd 80(%rsp),%xmm9+ paddd 96(%rsp),%xmm10+ paddd 112(%rsp),%xmm11++ movdqa %xmm8,%xmm6+ punpckldq %xmm9,%xmm8+ movdqa %xmm10,%xmm7+ punpckldq %xmm11,%xmm10+ punpckhdq %xmm9,%xmm6+ punpckhdq %xmm11,%xmm7+ movdqa %xmm8,%xmm9+ punpcklqdq %xmm10,%xmm8+ movdqa %xmm6,%xmm11+ punpcklqdq %xmm7,%xmm6+ punpckhqdq %xmm10,%xmm9+ punpckhqdq %xmm7,%xmm11+ paddd 128-256(%rcx),%xmm12+ paddd 144-256(%rcx),%xmm13+ paddd 160-256(%rcx),%xmm14+ paddd 176-256(%rcx),%xmm15++ movdqa %xmm8,0(%rsp)+ movdqa %xmm9,16(%rsp)+ movdqa 32(%rsp),%xmm8+ movdqa 48(%rsp),%xmm9++ movdqa %xmm12,%xmm10+ punpckldq %xmm13,%xmm12+ movdqa %xmm14,%xmm7+ punpckldq %xmm15,%xmm14+ punpckhdq %xmm13,%xmm10+ punpckhdq %xmm15,%xmm7+ movdqa %xmm12,%xmm13+ punpcklqdq %xmm14,%xmm12+ movdqa %xmm10,%xmm15+ punpcklqdq %xmm7,%xmm10+ punpckhqdq %xmm14,%xmm13+ punpckhqdq %xmm7,%xmm15+ paddd 192-256(%rcx),%xmm4+ paddd 208-256(%rcx),%xmm5+ paddd 224-256(%rcx),%xmm8+ paddd 240-256(%rcx),%xmm9++ movdqa %xmm6,32(%rsp)+ movdqa %xmm11,48(%rsp)++ movdqa %xmm4,%xmm14+ punpckldq %xmm5,%xmm4+ movdqa %xmm8,%xmm7+ punpckldq %xmm9,%xmm8+ punpckhdq %xmm5,%xmm14+ punpckhdq %xmm9,%xmm7+ movdqa %xmm4,%xmm5+ punpcklqdq %xmm8,%xmm4+ movdqa %xmm14,%xmm9+ punpcklqdq %xmm7,%xmm14+ punpckhqdq %xmm8,%xmm5+ punpckhqdq %xmm7,%xmm9+ paddd 256-256(%rcx),%xmm0+ paddd 272-256(%rcx),%xmm1+ paddd 288-256(%rcx),%xmm2+ paddd 304-256(%rcx),%xmm3++ movdqa %xmm0,%xmm8+ punpckldq %xmm1,%xmm0+ movdqa %xmm2,%xmm7+ punpckldq %xmm3,%xmm2+ punpckhdq %xmm1,%xmm8+ punpckhdq %xmm3,%xmm7+ movdqa %xmm0,%xmm1+ punpcklqdq %xmm2,%xmm0+ movdqa %xmm8,%xmm3+ punpcklqdq %xmm7,%xmm8+ punpckhqdq %xmm2,%xmm1+ punpckhqdq %xmm7,%xmm3+ cmpq $256,%rdx+ jb L$tail4x++ movdqu 0(%rsi),%xmm6+ movdqu 16(%rsi),%xmm11+ movdqu 32(%rsi),%xmm2+ movdqu 48(%rsi),%xmm7+ pxor 0(%rsp),%xmm6+ pxor %xmm12,%xmm11+ pxor %xmm4,%xmm2+ pxor %xmm0,%xmm7++ movdqu %xmm6,0(%rdi)+ movdqu 64(%rsi),%xmm6+ movdqu %xmm11,16(%rdi)+ movdqu 80(%rsi),%xmm11+ movdqu %xmm2,32(%rdi)+ movdqu 96(%rsi),%xmm2+ movdqu %xmm7,48(%rdi)+ movdqu 112(%rsi),%xmm7+ leaq 128(%rsi),%rsi+ pxor 16(%rsp),%xmm6+ pxor %xmm13,%xmm11+ pxor %xmm5,%xmm2+ pxor %xmm1,%xmm7++ movdqu %xmm6,64(%rdi)+ movdqu 0(%rsi),%xmm6+ movdqu %xmm11,80(%rdi)+ movdqu 16(%rsi),%xmm11+ movdqu %xmm2,96(%rdi)+ movdqu 32(%rsi),%xmm2+ movdqu %xmm7,112(%rdi)+ leaq 128(%rdi),%rdi+ movdqu 48(%rsi),%xmm7+ pxor 32(%rsp),%xmm6+ pxor %xmm10,%xmm11+ pxor %xmm14,%xmm2+ pxor %xmm8,%xmm7++ movdqu %xmm6,0(%rdi)+ movdqu 64(%rsi),%xmm6+ movdqu %xmm11,16(%rdi)+ movdqu 80(%rsi),%xmm11+ movdqu %xmm2,32(%rdi)+ movdqu 96(%rsi),%xmm2+ movdqu %xmm7,48(%rdi)+ movdqu 112(%rsi),%xmm7+ leaq 128(%rsi),%rsi+ pxor 48(%rsp),%xmm6+ pxor %xmm15,%xmm11+ pxor %xmm9,%xmm2+ pxor %xmm3,%xmm7+ movdqu %xmm6,64(%rdi)+ movdqu %xmm11,80(%rdi)+ movdqu %xmm2,96(%rdi)+ movdqu %xmm7,112(%rdi)+ leaq 128(%rdi),%rdi++ subq $256,%rdx+ jnz L$oop_outer4x++ jmp L$done4x++L$tail4x:+ cmpq $192,%rdx+ jae L$192_or_more4x+ cmpq $128,%rdx+ jae L$128_or_more4x+ cmpq $64,%rdx+ jae L$64_or_more4x+++ xorq %r9,%r9++ movdqa %xmm12,16(%rsp)+ movdqa %xmm4,32(%rsp)+ movdqa %xmm0,48(%rsp)+ jmp L$oop_tail4x++.p2align 5+L$64_or_more4x:+ movdqu 0(%rsi),%xmm6+ movdqu 16(%rsi),%xmm11+ movdqu 32(%rsi),%xmm2+ movdqu 48(%rsi),%xmm7+ pxor 0(%rsp),%xmm6+ pxor %xmm12,%xmm11+ pxor %xmm4,%xmm2+ pxor %xmm0,%xmm7+ movdqu %xmm6,0(%rdi)+ movdqu %xmm11,16(%rdi)+ movdqu %xmm2,32(%rdi)+ movdqu %xmm7,48(%rdi)+ je L$done4x++ movdqa 16(%rsp),%xmm6+ leaq 64(%rsi),%rsi+ xorq %r9,%r9+ movdqa %xmm6,0(%rsp)+ movdqa %xmm13,16(%rsp)+ leaq 64(%rdi),%rdi+ movdqa %xmm5,32(%rsp)+ subq $64,%rdx+ movdqa %xmm1,48(%rsp)+ jmp L$oop_tail4x++.p2align 5+L$128_or_more4x:+ movdqu 0(%rsi),%xmm6+ movdqu 16(%rsi),%xmm11+ movdqu 32(%rsi),%xmm2+ movdqu 48(%rsi),%xmm7+ pxor 0(%rsp),%xmm6+ pxor %xmm12,%xmm11+ pxor %xmm4,%xmm2+ pxor %xmm0,%xmm7++ movdqu %xmm6,0(%rdi)+ movdqu 64(%rsi),%xmm6+ movdqu %xmm11,16(%rdi)+ movdqu 80(%rsi),%xmm11+ movdqu %xmm2,32(%rdi)+ movdqu 96(%rsi),%xmm2+ movdqu %xmm7,48(%rdi)+ movdqu 112(%rsi),%xmm7+ pxor 16(%rsp),%xmm6+ pxor %xmm13,%xmm11+ pxor %xmm5,%xmm2+ pxor %xmm1,%xmm7+ movdqu %xmm6,64(%rdi)+ movdqu %xmm11,80(%rdi)+ movdqu %xmm2,96(%rdi)+ movdqu %xmm7,112(%rdi)+ je L$done4x++ movdqa 32(%rsp),%xmm6+ leaq 128(%rsi),%rsi+ xorq %r9,%r9+ movdqa %xmm6,0(%rsp)+ movdqa %xmm10,16(%rsp)+ leaq 128(%rdi),%rdi+ movdqa %xmm14,32(%rsp)+ subq $128,%rdx+ movdqa %xmm8,48(%rsp)+ jmp L$oop_tail4x++.p2align 5+L$192_or_more4x:+ movdqu 0(%rsi),%xmm6+ movdqu 16(%rsi),%xmm11+ movdqu 32(%rsi),%xmm2+ movdqu 48(%rsi),%xmm7+ pxor 0(%rsp),%xmm6+ pxor %xmm12,%xmm11+ pxor %xmm4,%xmm2+ pxor %xmm0,%xmm7++ movdqu %xmm6,0(%rdi)+ movdqu 64(%rsi),%xmm6+ movdqu %xmm11,16(%rdi)+ movdqu 80(%rsi),%xmm11+ movdqu %xmm2,32(%rdi)+ movdqu 96(%rsi),%xmm2+ movdqu %xmm7,48(%rdi)+ movdqu 112(%rsi),%xmm7+ leaq 128(%rsi),%rsi+ pxor 16(%rsp),%xmm6+ pxor %xmm13,%xmm11+ pxor %xmm5,%xmm2+ pxor %xmm1,%xmm7++ movdqu %xmm6,64(%rdi)+ movdqu 0(%rsi),%xmm6+ movdqu %xmm11,80(%rdi)+ movdqu 16(%rsi),%xmm11+ movdqu %xmm2,96(%rdi)+ movdqu 32(%rsi),%xmm2+ movdqu %xmm7,112(%rdi)+ leaq 128(%rdi),%rdi+ movdqu 48(%rsi),%xmm7+ pxor 32(%rsp),%xmm6+ pxor %xmm10,%xmm11+ pxor %xmm14,%xmm2+ pxor %xmm8,%xmm7+ movdqu %xmm6,0(%rdi)+ movdqu %xmm11,16(%rdi)+ movdqu %xmm2,32(%rdi)+ movdqu %xmm7,48(%rdi)+ je L$done4x++ movdqa 48(%rsp),%xmm6+ leaq 64(%rsi),%rsi+ xorq %r9,%r9+ movdqa %xmm6,0(%rsp)+ movdqa %xmm15,16(%rsp)+ leaq 64(%rdi),%rdi+ movdqa %xmm9,32(%rsp)+ subq $192,%rdx+ movdqa %xmm3,48(%rsp)++L$oop_tail4x:+ movzbl (%rsi,%r9,1),%eax+ movzbl (%rsp,%r9,1),%ecx+ leaq 1(%r9),%r9+ xorl %ecx,%eax+ movb %al,-1(%rdi,%r9,1)+ decq %rdx+ jnz L$oop_tail4x++L$done4x:+ leaq (%r10),%rsp+.cfi_def_cfa_register %rsp+L$4x_epilogue:+ .byte 0xf3,0xc3+.cfi_endproc +++.p2align 5+crypton_chacha20_asm_4xop:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa+++L$crypton_chacha20_asm_4xop:+ movq %rsp,%r10+.cfi_def_cfa_register %r10+ subq $0x140+8,%rsp+ andq $-16,%rsp+ vzeroupper++ vmovdqa L$sigma(%rip),%xmm11+ vmovdqu (%rcx),%xmm3+ vmovdqu 16(%rcx),%xmm15+ vmovdqu (%r8),%xmm7+ leaq 256(%rsp),%rcx++ vpshufd $0x00,%xmm11,%xmm8+ vpshufd $0x55,%xmm11,%xmm9+ vmovdqa %xmm8,64(%rsp)+ vpshufd $0xaa,%xmm11,%xmm10+ vmovdqa %xmm9,80(%rsp)+ vpshufd $0xff,%xmm11,%xmm11+ vmovdqa %xmm10,96(%rsp)+ vmovdqa %xmm11,112(%rsp)++ vpshufd $0x00,%xmm3,%xmm0+ vpshufd $0x55,%xmm3,%xmm1+ vmovdqa %xmm0,128-256(%rcx)+ vpshufd $0xaa,%xmm3,%xmm2+ vmovdqa %xmm1,144-256(%rcx)+ vpshufd $0xff,%xmm3,%xmm3+ vmovdqa %xmm2,160-256(%rcx)+ vmovdqa %xmm3,176-256(%rcx)++ vpshufd $0x00,%xmm15,%xmm12+ vpshufd $0x55,%xmm15,%xmm13+ vmovdqa %xmm12,192-256(%rcx)+ vpshufd $0xaa,%xmm15,%xmm14+ vmovdqa %xmm13,208-256(%rcx)+ vpshufd $0xff,%xmm15,%xmm15+ vmovdqa %xmm14,224-256(%rcx)+ vmovdqa %xmm15,240-256(%rcx)++ vpshufd $0x00,%xmm7,%xmm4+ vpshufd $0x55,%xmm7,%xmm5+ vpaddd L$inc(%rip),%xmm4,%xmm4+ vpshufd $0xaa,%xmm7,%xmm6+ vmovdqa %xmm5,272-256(%rcx)+ vpshufd $0xff,%xmm7,%xmm7+ vmovdqa %xmm6,288-256(%rcx)+ vmovdqa %xmm7,304-256(%rcx)++ jmp L$oop_enter4xop++.p2align 5+L$oop_outer4xop:+ vmovdqa 64(%rsp),%xmm8+ vmovdqa 80(%rsp),%xmm9+ vmovdqa 96(%rsp),%xmm10+ vmovdqa 112(%rsp),%xmm11+ vmovdqa 128-256(%rcx),%xmm0+ vmovdqa 144-256(%rcx),%xmm1+ vmovdqa 160-256(%rcx),%xmm2+ vmovdqa 176-256(%rcx),%xmm3+ vmovdqa 192-256(%rcx),%xmm12+ vmovdqa 208-256(%rcx),%xmm13+ vmovdqa 224-256(%rcx),%xmm14+ vmovdqa 240-256(%rcx),%xmm15+ vmovdqa 256-256(%rcx),%xmm4+ vmovdqa 272-256(%rcx),%xmm5+ vmovdqa 288-256(%rcx),%xmm6+ vmovdqa 304-256(%rcx),%xmm7+ vpaddd L$four(%rip),%xmm4,%xmm4++L$oop_enter4xop:+ movl $10,%eax+ vmovdqa %xmm4,256-256(%rcx)+ jmp L$oop4xop++.p2align 5+L$oop4xop:+ vpaddd %xmm0,%xmm8,%xmm8+ vpaddd %xmm1,%xmm9,%xmm9+ vpaddd %xmm2,%xmm10,%xmm10+ vpaddd %xmm3,%xmm11,%xmm11+ vpxor %xmm4,%xmm8,%xmm4+ vpxor %xmm5,%xmm9,%xmm5+ vpxor %xmm6,%xmm10,%xmm6+ vpxor %xmm7,%xmm11,%xmm7+.byte 143,232,120,194,228,16+.byte 143,232,120,194,237,16+.byte 143,232,120,194,246,16+.byte 143,232,120,194,255,16+ vpaddd %xmm4,%xmm12,%xmm12+ vpaddd %xmm5,%xmm13,%xmm13+ vpaddd %xmm6,%xmm14,%xmm14+ vpaddd %xmm7,%xmm15,%xmm15+ vpxor %xmm0,%xmm12,%xmm0+ vpxor %xmm1,%xmm13,%xmm1+ vpxor %xmm14,%xmm2,%xmm2+ vpxor %xmm15,%xmm3,%xmm3+.byte 143,232,120,194,192,12+.byte 143,232,120,194,201,12+.byte 143,232,120,194,210,12+.byte 143,232,120,194,219,12+ vpaddd %xmm8,%xmm0,%xmm8+ vpaddd %xmm9,%xmm1,%xmm9+ vpaddd %xmm2,%xmm10,%xmm10+ vpaddd %xmm3,%xmm11,%xmm11+ vpxor %xmm4,%xmm8,%xmm4+ vpxor %xmm5,%xmm9,%xmm5+ vpxor %xmm6,%xmm10,%xmm6+ vpxor %xmm7,%xmm11,%xmm7+.byte 143,232,120,194,228,8+.byte 143,232,120,194,237,8+.byte 143,232,120,194,246,8+.byte 143,232,120,194,255,8+ vpaddd %xmm4,%xmm12,%xmm12+ vpaddd %xmm5,%xmm13,%xmm13+ vpaddd %xmm6,%xmm14,%xmm14+ vpaddd %xmm7,%xmm15,%xmm15+ vpxor %xmm0,%xmm12,%xmm0+ vpxor %xmm1,%xmm13,%xmm1+ vpxor %xmm14,%xmm2,%xmm2+ vpxor %xmm15,%xmm3,%xmm3+.byte 143,232,120,194,192,7+.byte 143,232,120,194,201,7+.byte 143,232,120,194,210,7+.byte 143,232,120,194,219,7+ vpaddd %xmm1,%xmm8,%xmm8+ vpaddd %xmm2,%xmm9,%xmm9+ vpaddd %xmm3,%xmm10,%xmm10+ vpaddd %xmm0,%xmm11,%xmm11+ vpxor %xmm7,%xmm8,%xmm7+ vpxor %xmm4,%xmm9,%xmm4+ vpxor %xmm5,%xmm10,%xmm5+ vpxor %xmm6,%xmm11,%xmm6+.byte 143,232,120,194,255,16+.byte 143,232,120,194,228,16+.byte 143,232,120,194,237,16+.byte 143,232,120,194,246,16+ vpaddd %xmm7,%xmm14,%xmm14+ vpaddd %xmm4,%xmm15,%xmm15+ vpaddd %xmm5,%xmm12,%xmm12+ vpaddd %xmm6,%xmm13,%xmm13+ vpxor %xmm1,%xmm14,%xmm1+ vpxor %xmm2,%xmm15,%xmm2+ vpxor %xmm12,%xmm3,%xmm3+ vpxor %xmm13,%xmm0,%xmm0+.byte 143,232,120,194,201,12+.byte 143,232,120,194,210,12+.byte 143,232,120,194,219,12+.byte 143,232,120,194,192,12+ vpaddd %xmm8,%xmm1,%xmm8+ vpaddd %xmm9,%xmm2,%xmm9+ vpaddd %xmm3,%xmm10,%xmm10+ vpaddd %xmm0,%xmm11,%xmm11+ vpxor %xmm7,%xmm8,%xmm7+ vpxor %xmm4,%xmm9,%xmm4+ vpxor %xmm5,%xmm10,%xmm5+ vpxor %xmm6,%xmm11,%xmm6+.byte 143,232,120,194,255,8+.byte 143,232,120,194,228,8+.byte 143,232,120,194,237,8+.byte 143,232,120,194,246,8+ vpaddd %xmm7,%xmm14,%xmm14+ vpaddd %xmm4,%xmm15,%xmm15+ vpaddd %xmm5,%xmm12,%xmm12+ vpaddd %xmm6,%xmm13,%xmm13+ vpxor %xmm1,%xmm14,%xmm1+ vpxor %xmm2,%xmm15,%xmm2+ vpxor %xmm12,%xmm3,%xmm3+ vpxor %xmm13,%xmm0,%xmm0+.byte 143,232,120,194,201,7+.byte 143,232,120,194,210,7+.byte 143,232,120,194,219,7+.byte 143,232,120,194,192,7+ decl %eax+ jnz L$oop4xop++ vpaddd 64(%rsp),%xmm8,%xmm8+ vpaddd 80(%rsp),%xmm9,%xmm9+ vpaddd 96(%rsp),%xmm10,%xmm10+ vpaddd 112(%rsp),%xmm11,%xmm11++ vmovdqa %xmm14,32(%rsp)+ vmovdqa %xmm15,48(%rsp)++ vpunpckldq %xmm9,%xmm8,%xmm14+ vpunpckldq %xmm11,%xmm10,%xmm15+ vpunpckhdq %xmm9,%xmm8,%xmm8+ vpunpckhdq %xmm11,%xmm10,%xmm10+ vpunpcklqdq %xmm15,%xmm14,%xmm9+ vpunpckhqdq %xmm15,%xmm14,%xmm14+ vpunpcklqdq %xmm10,%xmm8,%xmm11+ vpunpckhqdq %xmm10,%xmm8,%xmm8+ vpaddd 128-256(%rcx),%xmm0,%xmm0+ vpaddd 144-256(%rcx),%xmm1,%xmm1+ vpaddd 160-256(%rcx),%xmm2,%xmm2+ vpaddd 176-256(%rcx),%xmm3,%xmm3++ vmovdqa %xmm9,0(%rsp)+ vmovdqa %xmm14,16(%rsp)+ vmovdqa 32(%rsp),%xmm9+ vmovdqa 48(%rsp),%xmm14++ vpunpckldq %xmm1,%xmm0,%xmm10+ vpunpckldq %xmm3,%xmm2,%xmm15+ vpunpckhdq %xmm1,%xmm0,%xmm0+ vpunpckhdq %xmm3,%xmm2,%xmm2+ vpunpcklqdq %xmm15,%xmm10,%xmm1+ vpunpckhqdq %xmm15,%xmm10,%xmm10+ vpunpcklqdq %xmm2,%xmm0,%xmm3+ vpunpckhqdq %xmm2,%xmm0,%xmm0+ vpaddd 192-256(%rcx),%xmm12,%xmm12+ vpaddd 208-256(%rcx),%xmm13,%xmm13+ vpaddd 224-256(%rcx),%xmm9,%xmm9+ vpaddd 240-256(%rcx),%xmm14,%xmm14++ vpunpckldq %xmm13,%xmm12,%xmm2+ vpunpckldq %xmm14,%xmm9,%xmm15+ vpunpckhdq %xmm13,%xmm12,%xmm12+ vpunpckhdq %xmm14,%xmm9,%xmm9+ vpunpcklqdq %xmm15,%xmm2,%xmm13+ vpunpckhqdq %xmm15,%xmm2,%xmm2+ vpunpcklqdq %xmm9,%xmm12,%xmm14+ vpunpckhqdq %xmm9,%xmm12,%xmm12+ vpaddd 256-256(%rcx),%xmm4,%xmm4+ vpaddd 272-256(%rcx),%xmm5,%xmm5+ vpaddd 288-256(%rcx),%xmm6,%xmm6+ vpaddd 304-256(%rcx),%xmm7,%xmm7++ vpunpckldq %xmm5,%xmm4,%xmm9+ vpunpckldq %xmm7,%xmm6,%xmm15+ vpunpckhdq %xmm5,%xmm4,%xmm4+ vpunpckhdq %xmm7,%xmm6,%xmm6+ vpunpcklqdq %xmm15,%xmm9,%xmm5+ vpunpckhqdq %xmm15,%xmm9,%xmm9+ vpunpcklqdq %xmm6,%xmm4,%xmm7+ vpunpckhqdq %xmm6,%xmm4,%xmm4+ vmovdqa 0(%rsp),%xmm6+ vmovdqa 16(%rsp),%xmm15++ cmpq $256,%rdx+ jb L$tail4xop++ vpxor 0(%rsi),%xmm6,%xmm6+ vpxor 16(%rsi),%xmm1,%xmm1+ vpxor 32(%rsi),%xmm13,%xmm13+ vpxor 48(%rsi),%xmm5,%xmm5+ vpxor 64(%rsi),%xmm15,%xmm15+ vpxor 80(%rsi),%xmm10,%xmm10+ vpxor 96(%rsi),%xmm2,%xmm2+ vpxor 112(%rsi),%xmm9,%xmm9+ leaq 128(%rsi),%rsi+ vpxor 0(%rsi),%xmm11,%xmm11+ vpxor 16(%rsi),%xmm3,%xmm3+ vpxor 32(%rsi),%xmm14,%xmm14+ vpxor 48(%rsi),%xmm7,%xmm7+ vpxor 64(%rsi),%xmm8,%xmm8+ vpxor 80(%rsi),%xmm0,%xmm0+ vpxor 96(%rsi),%xmm12,%xmm12+ vpxor 112(%rsi),%xmm4,%xmm4+ leaq 128(%rsi),%rsi++ vmovdqu %xmm6,0(%rdi)+ vmovdqu %xmm1,16(%rdi)+ vmovdqu %xmm13,32(%rdi)+ vmovdqu %xmm5,48(%rdi)+ vmovdqu %xmm15,64(%rdi)+ vmovdqu %xmm10,80(%rdi)+ vmovdqu %xmm2,96(%rdi)+ vmovdqu %xmm9,112(%rdi)+ leaq 128(%rdi),%rdi+ vmovdqu %xmm11,0(%rdi)+ vmovdqu %xmm3,16(%rdi)+ vmovdqu %xmm14,32(%rdi)+ vmovdqu %xmm7,48(%rdi)+ vmovdqu %xmm8,64(%rdi)+ vmovdqu %xmm0,80(%rdi)+ vmovdqu %xmm12,96(%rdi)+ vmovdqu %xmm4,112(%rdi)+ leaq 128(%rdi),%rdi++ subq $256,%rdx+ jnz L$oop_outer4xop++ jmp L$done4xop++.p2align 5+L$tail4xop:+ cmpq $192,%rdx+ jae L$192_or_more4xop+ cmpq $128,%rdx+ jae L$128_or_more4xop+ cmpq $64,%rdx+ jae L$64_or_more4xop++ xorq %r9,%r9+ vmovdqa %xmm6,0(%rsp)+ vmovdqa %xmm1,16(%rsp)+ vmovdqa %xmm13,32(%rsp)+ vmovdqa %xmm5,48(%rsp)+ jmp L$oop_tail4xop++.p2align 5+L$64_or_more4xop:+ vpxor 0(%rsi),%xmm6,%xmm6+ vpxor 16(%rsi),%xmm1,%xmm1+ vpxor 32(%rsi),%xmm13,%xmm13+ vpxor 48(%rsi),%xmm5,%xmm5+ vmovdqu %xmm6,0(%rdi)+ vmovdqu %xmm1,16(%rdi)+ vmovdqu %xmm13,32(%rdi)+ vmovdqu %xmm5,48(%rdi)+ je L$done4xop++ leaq 64(%rsi),%rsi+ vmovdqa %xmm15,0(%rsp)+ xorq %r9,%r9+ vmovdqa %xmm10,16(%rsp)+ leaq 64(%rdi),%rdi+ vmovdqa %xmm2,32(%rsp)+ subq $64,%rdx+ vmovdqa %xmm9,48(%rsp)+ jmp L$oop_tail4xop++.p2align 5+L$128_or_more4xop:+ vpxor 0(%rsi),%xmm6,%xmm6+ vpxor 16(%rsi),%xmm1,%xmm1+ vpxor 32(%rsi),%xmm13,%xmm13+ vpxor 48(%rsi),%xmm5,%xmm5+ vpxor 64(%rsi),%xmm15,%xmm15+ vpxor 80(%rsi),%xmm10,%xmm10+ vpxor 96(%rsi),%xmm2,%xmm2+ vpxor 112(%rsi),%xmm9,%xmm9++ vmovdqu %xmm6,0(%rdi)+ vmovdqu %xmm1,16(%rdi)+ vmovdqu %xmm13,32(%rdi)+ vmovdqu %xmm5,48(%rdi)+ vmovdqu %xmm15,64(%rdi)+ vmovdqu %xmm10,80(%rdi)+ vmovdqu %xmm2,96(%rdi)+ vmovdqu %xmm9,112(%rdi)+ je L$done4xop++ leaq 128(%rsi),%rsi+ vmovdqa %xmm11,0(%rsp)+ xorq %r9,%r9+ vmovdqa %xmm3,16(%rsp)+ leaq 128(%rdi),%rdi+ vmovdqa %xmm14,32(%rsp)+ subq $128,%rdx+ vmovdqa %xmm7,48(%rsp)+ jmp L$oop_tail4xop++.p2align 5+L$192_or_more4xop:+ vpxor 0(%rsi),%xmm6,%xmm6+ vpxor 16(%rsi),%xmm1,%xmm1+ vpxor 32(%rsi),%xmm13,%xmm13+ vpxor 48(%rsi),%xmm5,%xmm5+ vpxor 64(%rsi),%xmm15,%xmm15+ vpxor 80(%rsi),%xmm10,%xmm10+ vpxor 96(%rsi),%xmm2,%xmm2+ vpxor 112(%rsi),%xmm9,%xmm9+ leaq 128(%rsi),%rsi+ vpxor 0(%rsi),%xmm11,%xmm11+ vpxor 16(%rsi),%xmm3,%xmm3+ vpxor 32(%rsi),%xmm14,%xmm14+ vpxor 48(%rsi),%xmm7,%xmm7++ vmovdqu %xmm6,0(%rdi)+ vmovdqu %xmm1,16(%rdi)+ vmovdqu %xmm13,32(%rdi)+ vmovdqu %xmm5,48(%rdi)+ vmovdqu %xmm15,64(%rdi)+ vmovdqu %xmm10,80(%rdi)+ vmovdqu %xmm2,96(%rdi)+ vmovdqu %xmm9,112(%rdi)+ leaq 128(%rdi),%rdi+ vmovdqu %xmm11,0(%rdi)+ vmovdqu %xmm3,16(%rdi)+ vmovdqu %xmm14,32(%rdi)+ vmovdqu %xmm7,48(%rdi)+ je L$done4xop++ leaq 64(%rsi),%rsi+ vmovdqa %xmm8,0(%rsp)+ xorq %r9,%r9+ vmovdqa %xmm0,16(%rsp)+ leaq 64(%rdi),%rdi+ vmovdqa %xmm12,32(%rsp)+ subq $192,%rdx+ vmovdqa %xmm4,48(%rsp)++L$oop_tail4xop:+ movzbl (%rsi,%r9,1),%eax+ movzbl (%rsp,%r9,1),%ecx+ leaq 1(%r9),%r9+ xorl %ecx,%eax+ movb %al,-1(%rdi,%r9,1)+ decq %rdx+ jnz L$oop_tail4xop++L$done4xop:+ vzeroupper+ leaq (%r10),%rsp+.cfi_def_cfa_register %rsp+L$4xop_epilogue:+ .byte 0xf3,0xc3+.cfi_endproc +++.p2align 5+crypton_chacha20_asm_avx2:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa+++L$crypton_chacha20_asm_8x:+ movq %rsp,%r10+.cfi_def_cfa_register %r10+ subq $0x280+8,%rsp+ andq $-32,%rsp+ vzeroupper+++++++++++ vbroadcasti128 L$sigma(%rip),%ymm11+ vbroadcasti128 (%rcx),%ymm3+ vbroadcasti128 16(%rcx),%ymm15+ vbroadcasti128 (%r8),%ymm7+ leaq 256(%rsp),%rcx+ leaq 512(%rsp),%rax+ leaq L$rot16(%rip),%r9+ leaq L$rot24(%rip),%r11++ vpshufd $0x00,%ymm11,%ymm8+ vpshufd $0x55,%ymm11,%ymm9+ vmovdqa %ymm8,128-256(%rcx)+ vpshufd $0xaa,%ymm11,%ymm10+ vmovdqa %ymm9,160-256(%rcx)+ vpshufd $0xff,%ymm11,%ymm11+ vmovdqa %ymm10,192-256(%rcx)+ vmovdqa %ymm11,224-256(%rcx)++ vpshufd $0x00,%ymm3,%ymm0+ vpshufd $0x55,%ymm3,%ymm1+ vmovdqa %ymm0,256-256(%rcx)+ vpshufd $0xaa,%ymm3,%ymm2+ vmovdqa %ymm1,288-256(%rcx)+ vpshufd $0xff,%ymm3,%ymm3+ vmovdqa %ymm2,320-256(%rcx)+ vmovdqa %ymm3,352-256(%rcx)++ vpshufd $0x00,%ymm15,%ymm12+ vpshufd $0x55,%ymm15,%ymm13+ vmovdqa %ymm12,384-512(%rax)+ vpshufd $0xaa,%ymm15,%ymm14+ vmovdqa %ymm13,416-512(%rax)+ vpshufd $0xff,%ymm15,%ymm15+ vmovdqa %ymm14,448-512(%rax)+ vmovdqa %ymm15,480-512(%rax)++ vpshufd $0x00,%ymm7,%ymm4+ vpshufd $0x55,%ymm7,%ymm5+ vpaddd L$incy(%rip),%ymm4,%ymm4+ vpshufd $0xaa,%ymm7,%ymm6+ vmovdqa %ymm5,544-512(%rax)+ vpshufd $0xff,%ymm7,%ymm7+ vmovdqa %ymm6,576-512(%rax)+ vmovdqa %ymm7,608-512(%rax)++ jmp L$oop_enter8x++.p2align 5+L$oop_outer8x:+ vmovdqa 128-256(%rcx),%ymm8+ vmovdqa 160-256(%rcx),%ymm9+ vmovdqa 192-256(%rcx),%ymm10+ vmovdqa 224-256(%rcx),%ymm11+ vmovdqa 256-256(%rcx),%ymm0+ vmovdqa 288-256(%rcx),%ymm1+ vmovdqa 320-256(%rcx),%ymm2+ vmovdqa 352-256(%rcx),%ymm3+ vmovdqa 384-512(%rax),%ymm12+ vmovdqa 416-512(%rax),%ymm13+ vmovdqa 448-512(%rax),%ymm14+ vmovdqa 480-512(%rax),%ymm15+ vmovdqa 512-512(%rax),%ymm4+ vmovdqa 544-512(%rax),%ymm5+ vmovdqa 576-512(%rax),%ymm6+ vmovdqa 608-512(%rax),%ymm7+ vpaddd L$eight(%rip),%ymm4,%ymm4++L$oop_enter8x:+ vmovdqa %ymm14,64(%rsp)+ vmovdqa %ymm15,96(%rsp)+ vbroadcasti128 (%r9),%ymm15+ vmovdqa %ymm4,512-512(%rax)+ movl $10,%eax+ jmp L$oop8x++.p2align 5+L$oop8x:+ vpaddd %ymm0,%ymm8,%ymm8+ vpxor %ymm4,%ymm8,%ymm4+ vpshufb %ymm15,%ymm4,%ymm4+ vpaddd %ymm1,%ymm9,%ymm9+ vpxor %ymm5,%ymm9,%ymm5+ vpshufb %ymm15,%ymm5,%ymm5+ vpaddd %ymm4,%ymm12,%ymm12+ vpxor %ymm0,%ymm12,%ymm0+ vpslld $12,%ymm0,%ymm14+ vpsrld $20,%ymm0,%ymm0+ vpor %ymm0,%ymm14,%ymm0+ vbroadcasti128 (%r11),%ymm14+ vpaddd %ymm5,%ymm13,%ymm13+ vpxor %ymm1,%ymm13,%ymm1+ vpslld $12,%ymm1,%ymm15+ vpsrld $20,%ymm1,%ymm1+ vpor %ymm1,%ymm15,%ymm1+ vpaddd %ymm0,%ymm8,%ymm8+ vpxor %ymm4,%ymm8,%ymm4+ vpshufb %ymm14,%ymm4,%ymm4+ vpaddd %ymm1,%ymm9,%ymm9+ vpxor %ymm5,%ymm9,%ymm5+ vpshufb %ymm14,%ymm5,%ymm5+ vpaddd %ymm4,%ymm12,%ymm12+ vpxor %ymm0,%ymm12,%ymm0+ vpslld $7,%ymm0,%ymm15+ vpsrld $25,%ymm0,%ymm0+ vpor %ymm0,%ymm15,%ymm0+ vbroadcasti128 (%r9),%ymm15+ vpaddd %ymm5,%ymm13,%ymm13+ vpxor %ymm1,%ymm13,%ymm1+ vpslld $7,%ymm1,%ymm14+ vpsrld $25,%ymm1,%ymm1+ vpor %ymm1,%ymm14,%ymm1+ vmovdqa %ymm12,0(%rsp)+ vmovdqa %ymm13,32(%rsp)+ vmovdqa 64(%rsp),%ymm12+ vmovdqa 96(%rsp),%ymm13+ vpaddd %ymm2,%ymm10,%ymm10+ vpxor %ymm6,%ymm10,%ymm6+ vpshufb %ymm15,%ymm6,%ymm6+ vpaddd %ymm3,%ymm11,%ymm11+ vpxor %ymm7,%ymm11,%ymm7+ vpshufb %ymm15,%ymm7,%ymm7+ vpaddd %ymm6,%ymm12,%ymm12+ vpxor %ymm2,%ymm12,%ymm2+ vpslld $12,%ymm2,%ymm14+ vpsrld $20,%ymm2,%ymm2+ vpor %ymm2,%ymm14,%ymm2+ vbroadcasti128 (%r11),%ymm14+ vpaddd %ymm7,%ymm13,%ymm13+ vpxor %ymm3,%ymm13,%ymm3+ vpslld $12,%ymm3,%ymm15+ vpsrld $20,%ymm3,%ymm3+ vpor %ymm3,%ymm15,%ymm3+ vpaddd %ymm2,%ymm10,%ymm10+ vpxor %ymm6,%ymm10,%ymm6+ vpshufb %ymm14,%ymm6,%ymm6+ vpaddd %ymm3,%ymm11,%ymm11+ vpxor %ymm7,%ymm11,%ymm7+ vpshufb %ymm14,%ymm7,%ymm7+ vpaddd %ymm6,%ymm12,%ymm12+ vpxor %ymm2,%ymm12,%ymm2+ vpslld $7,%ymm2,%ymm15+ vpsrld $25,%ymm2,%ymm2+ vpor %ymm2,%ymm15,%ymm2+ vbroadcasti128 (%r9),%ymm15+ vpaddd %ymm7,%ymm13,%ymm13+ vpxor %ymm3,%ymm13,%ymm3+ vpslld $7,%ymm3,%ymm14+ vpsrld $25,%ymm3,%ymm3+ vpor %ymm3,%ymm14,%ymm3+ vpaddd %ymm1,%ymm8,%ymm8+ vpxor %ymm7,%ymm8,%ymm7+ vpshufb %ymm15,%ymm7,%ymm7+ vpaddd %ymm2,%ymm9,%ymm9+ vpxor %ymm4,%ymm9,%ymm4+ vpshufb %ymm15,%ymm4,%ymm4+ vpaddd %ymm7,%ymm12,%ymm12+ vpxor %ymm1,%ymm12,%ymm1+ vpslld $12,%ymm1,%ymm14+ vpsrld $20,%ymm1,%ymm1+ vpor %ymm1,%ymm14,%ymm1+ vbroadcasti128 (%r11),%ymm14+ vpaddd %ymm4,%ymm13,%ymm13+ vpxor %ymm2,%ymm13,%ymm2+ vpslld $12,%ymm2,%ymm15+ vpsrld $20,%ymm2,%ymm2+ vpor %ymm2,%ymm15,%ymm2+ vpaddd %ymm1,%ymm8,%ymm8+ vpxor %ymm7,%ymm8,%ymm7+ vpshufb %ymm14,%ymm7,%ymm7+ vpaddd %ymm2,%ymm9,%ymm9+ vpxor %ymm4,%ymm9,%ymm4+ vpshufb %ymm14,%ymm4,%ymm4+ vpaddd %ymm7,%ymm12,%ymm12+ vpxor %ymm1,%ymm12,%ymm1+ vpslld $7,%ymm1,%ymm15+ vpsrld $25,%ymm1,%ymm1+ vpor %ymm1,%ymm15,%ymm1+ vbroadcasti128 (%r9),%ymm15+ vpaddd %ymm4,%ymm13,%ymm13+ vpxor %ymm2,%ymm13,%ymm2+ vpslld $7,%ymm2,%ymm14+ vpsrld $25,%ymm2,%ymm2+ vpor %ymm2,%ymm14,%ymm2+ vmovdqa %ymm12,64(%rsp)+ vmovdqa %ymm13,96(%rsp)+ vmovdqa 0(%rsp),%ymm12+ vmovdqa 32(%rsp),%ymm13+ vpaddd %ymm3,%ymm10,%ymm10+ vpxor %ymm5,%ymm10,%ymm5+ vpshufb %ymm15,%ymm5,%ymm5+ vpaddd %ymm0,%ymm11,%ymm11+ vpxor %ymm6,%ymm11,%ymm6+ vpshufb %ymm15,%ymm6,%ymm6+ vpaddd %ymm5,%ymm12,%ymm12+ vpxor %ymm3,%ymm12,%ymm3+ vpslld $12,%ymm3,%ymm14+ vpsrld $20,%ymm3,%ymm3+ vpor %ymm3,%ymm14,%ymm3+ vbroadcasti128 (%r11),%ymm14+ vpaddd %ymm6,%ymm13,%ymm13+ vpxor %ymm0,%ymm13,%ymm0+ vpslld $12,%ymm0,%ymm15+ vpsrld $20,%ymm0,%ymm0+ vpor %ymm0,%ymm15,%ymm0+ vpaddd %ymm3,%ymm10,%ymm10+ vpxor %ymm5,%ymm10,%ymm5+ vpshufb %ymm14,%ymm5,%ymm5+ vpaddd %ymm0,%ymm11,%ymm11+ vpxor %ymm6,%ymm11,%ymm6+ vpshufb %ymm14,%ymm6,%ymm6+ vpaddd %ymm5,%ymm12,%ymm12+ vpxor %ymm3,%ymm12,%ymm3+ vpslld $7,%ymm3,%ymm15+ vpsrld $25,%ymm3,%ymm3+ vpor %ymm3,%ymm15,%ymm3+ vbroadcasti128 (%r9),%ymm15+ vpaddd %ymm6,%ymm13,%ymm13+ vpxor %ymm0,%ymm13,%ymm0+ vpslld $7,%ymm0,%ymm14+ vpsrld $25,%ymm0,%ymm0+ vpor %ymm0,%ymm14,%ymm0+ decl %eax+ jnz L$oop8x++ leaq 512(%rsp),%rax+ vpaddd 128-256(%rcx),%ymm8,%ymm8+ vpaddd 160-256(%rcx),%ymm9,%ymm9+ vpaddd 192-256(%rcx),%ymm10,%ymm10+ vpaddd 224-256(%rcx),%ymm11,%ymm11++ vpunpckldq %ymm9,%ymm8,%ymm14+ vpunpckldq %ymm11,%ymm10,%ymm15+ vpunpckhdq %ymm9,%ymm8,%ymm8+ vpunpckhdq %ymm11,%ymm10,%ymm10+ vpunpcklqdq %ymm15,%ymm14,%ymm9+ vpunpckhqdq %ymm15,%ymm14,%ymm14+ vpunpcklqdq %ymm10,%ymm8,%ymm11+ vpunpckhqdq %ymm10,%ymm8,%ymm8+ vpaddd 256-256(%rcx),%ymm0,%ymm0+ vpaddd 288-256(%rcx),%ymm1,%ymm1+ vpaddd 320-256(%rcx),%ymm2,%ymm2+ vpaddd 352-256(%rcx),%ymm3,%ymm3++ vpunpckldq %ymm1,%ymm0,%ymm10+ vpunpckldq %ymm3,%ymm2,%ymm15+ vpunpckhdq %ymm1,%ymm0,%ymm0+ vpunpckhdq %ymm3,%ymm2,%ymm2+ vpunpcklqdq %ymm15,%ymm10,%ymm1+ vpunpckhqdq %ymm15,%ymm10,%ymm10+ vpunpcklqdq %ymm2,%ymm0,%ymm3+ vpunpckhqdq %ymm2,%ymm0,%ymm0+ vperm2i128 $0x20,%ymm1,%ymm9,%ymm15+ vperm2i128 $0x31,%ymm1,%ymm9,%ymm1+ vperm2i128 $0x20,%ymm10,%ymm14,%ymm9+ vperm2i128 $0x31,%ymm10,%ymm14,%ymm10+ vperm2i128 $0x20,%ymm3,%ymm11,%ymm14+ vperm2i128 $0x31,%ymm3,%ymm11,%ymm3+ vperm2i128 $0x20,%ymm0,%ymm8,%ymm11+ vperm2i128 $0x31,%ymm0,%ymm8,%ymm0+ vmovdqa %ymm15,0(%rsp)+ vmovdqa %ymm9,32(%rsp)+ vmovdqa 64(%rsp),%ymm15+ vmovdqa 96(%rsp),%ymm9++ vpaddd 384-512(%rax),%ymm12,%ymm12+ vpaddd 416-512(%rax),%ymm13,%ymm13+ vpaddd 448-512(%rax),%ymm15,%ymm15+ vpaddd 480-512(%rax),%ymm9,%ymm9++ vpunpckldq %ymm13,%ymm12,%ymm2+ vpunpckldq %ymm9,%ymm15,%ymm8+ vpunpckhdq %ymm13,%ymm12,%ymm12+ vpunpckhdq %ymm9,%ymm15,%ymm15+ vpunpcklqdq %ymm8,%ymm2,%ymm13+ vpunpckhqdq %ymm8,%ymm2,%ymm2+ vpunpcklqdq %ymm15,%ymm12,%ymm9+ vpunpckhqdq %ymm15,%ymm12,%ymm12+ vpaddd 512-512(%rax),%ymm4,%ymm4+ vpaddd 544-512(%rax),%ymm5,%ymm5+ vpaddd 576-512(%rax),%ymm6,%ymm6+ vpaddd 608-512(%rax),%ymm7,%ymm7++ vpunpckldq %ymm5,%ymm4,%ymm15+ vpunpckldq %ymm7,%ymm6,%ymm8+ vpunpckhdq %ymm5,%ymm4,%ymm4+ vpunpckhdq %ymm7,%ymm6,%ymm6+ vpunpcklqdq %ymm8,%ymm15,%ymm5+ vpunpckhqdq %ymm8,%ymm15,%ymm15+ vpunpcklqdq %ymm6,%ymm4,%ymm7+ vpunpckhqdq %ymm6,%ymm4,%ymm4+ vperm2i128 $0x20,%ymm5,%ymm13,%ymm8+ vperm2i128 $0x31,%ymm5,%ymm13,%ymm5+ vperm2i128 $0x20,%ymm15,%ymm2,%ymm13+ vperm2i128 $0x31,%ymm15,%ymm2,%ymm15+ vperm2i128 $0x20,%ymm7,%ymm9,%ymm2+ vperm2i128 $0x31,%ymm7,%ymm9,%ymm7+ vperm2i128 $0x20,%ymm4,%ymm12,%ymm9+ vperm2i128 $0x31,%ymm4,%ymm12,%ymm4+ vmovdqa 0(%rsp),%ymm6+ vmovdqa 32(%rsp),%ymm12++ cmpq $512,%rdx+ jb L$tail8x++ vpxor 0(%rsi),%ymm6,%ymm6+ vpxor 32(%rsi),%ymm8,%ymm8+ vpxor 64(%rsi),%ymm1,%ymm1+ vpxor 96(%rsi),%ymm5,%ymm5+ leaq 128(%rsi),%rsi+ vmovdqu %ymm6,0(%rdi)+ vmovdqu %ymm8,32(%rdi)+ vmovdqu %ymm1,64(%rdi)+ vmovdqu %ymm5,96(%rdi)+ leaq 128(%rdi),%rdi++ vpxor 0(%rsi),%ymm12,%ymm12+ vpxor 32(%rsi),%ymm13,%ymm13+ vpxor 64(%rsi),%ymm10,%ymm10+ vpxor 96(%rsi),%ymm15,%ymm15+ leaq 128(%rsi),%rsi+ vmovdqu %ymm12,0(%rdi)+ vmovdqu %ymm13,32(%rdi)+ vmovdqu %ymm10,64(%rdi)+ vmovdqu %ymm15,96(%rdi)+ leaq 128(%rdi),%rdi++ vpxor 0(%rsi),%ymm14,%ymm14+ vpxor 32(%rsi),%ymm2,%ymm2+ vpxor 64(%rsi),%ymm3,%ymm3+ vpxor 96(%rsi),%ymm7,%ymm7+ leaq 128(%rsi),%rsi+ vmovdqu %ymm14,0(%rdi)+ vmovdqu %ymm2,32(%rdi)+ vmovdqu %ymm3,64(%rdi)+ vmovdqu %ymm7,96(%rdi)+ leaq 128(%rdi),%rdi++ vpxor 0(%rsi),%ymm11,%ymm11+ vpxor 32(%rsi),%ymm9,%ymm9+ vpxor 64(%rsi),%ymm0,%ymm0+ vpxor 96(%rsi),%ymm4,%ymm4+ leaq 128(%rsi),%rsi+ vmovdqu %ymm11,0(%rdi)+ vmovdqu %ymm9,32(%rdi)+ vmovdqu %ymm0,64(%rdi)+ vmovdqu %ymm4,96(%rdi)+ leaq 128(%rdi),%rdi++ subq $512,%rdx+ jnz L$oop_outer8x++ jmp L$done8x++L$tail8x:+ cmpq $448,%rdx+ jae L$448_or_more8x+ cmpq $384,%rdx+ jae L$384_or_more8x+ cmpq $320,%rdx+ jae L$320_or_more8x+ cmpq $256,%rdx+ jae L$256_or_more8x+ cmpq $192,%rdx+ jae L$192_or_more8x+ cmpq $128,%rdx+ jae L$128_or_more8x+ cmpq $64,%rdx+ jae L$64_or_more8x++ xorq %r9,%r9+ vmovdqa %ymm6,0(%rsp)+ vmovdqa %ymm8,32(%rsp)+ jmp L$oop_tail8x++.p2align 5+L$64_or_more8x:+ vpxor 0(%rsi),%ymm6,%ymm6+ vpxor 32(%rsi),%ymm8,%ymm8+ vmovdqu %ymm6,0(%rdi)+ vmovdqu %ymm8,32(%rdi)+ je L$done8x++ leaq 64(%rsi),%rsi+ xorq %r9,%r9+ vmovdqa %ymm1,0(%rsp)+ leaq 64(%rdi),%rdi+ subq $64,%rdx+ vmovdqa %ymm5,32(%rsp)+ jmp L$oop_tail8x++.p2align 5+L$128_or_more8x:+ vpxor 0(%rsi),%ymm6,%ymm6+ vpxor 32(%rsi),%ymm8,%ymm8+ vpxor 64(%rsi),%ymm1,%ymm1+ vpxor 96(%rsi),%ymm5,%ymm5+ vmovdqu %ymm6,0(%rdi)+ vmovdqu %ymm8,32(%rdi)+ vmovdqu %ymm1,64(%rdi)+ vmovdqu %ymm5,96(%rdi)+ je L$done8x++ leaq 128(%rsi),%rsi+ xorq %r9,%r9+ vmovdqa %ymm12,0(%rsp)+ leaq 128(%rdi),%rdi+ subq $128,%rdx+ vmovdqa %ymm13,32(%rsp)+ jmp L$oop_tail8x++.p2align 5+L$192_or_more8x:+ vpxor 0(%rsi),%ymm6,%ymm6+ vpxor 32(%rsi),%ymm8,%ymm8+ vpxor 64(%rsi),%ymm1,%ymm1+ vpxor 96(%rsi),%ymm5,%ymm5+ vpxor 128(%rsi),%ymm12,%ymm12+ vpxor 160(%rsi),%ymm13,%ymm13+ vmovdqu %ymm6,0(%rdi)+ vmovdqu %ymm8,32(%rdi)+ vmovdqu %ymm1,64(%rdi)+ vmovdqu %ymm5,96(%rdi)+ vmovdqu %ymm12,128(%rdi)+ vmovdqu %ymm13,160(%rdi)+ je L$done8x++ leaq 192(%rsi),%rsi+ xorq %r9,%r9+ vmovdqa %ymm10,0(%rsp)+ leaq 192(%rdi),%rdi+ subq $192,%rdx+ vmovdqa %ymm15,32(%rsp)+ jmp L$oop_tail8x++.p2align 5+L$256_or_more8x:+ vpxor 0(%rsi),%ymm6,%ymm6+ vpxor 32(%rsi),%ymm8,%ymm8+ vpxor 64(%rsi),%ymm1,%ymm1+ vpxor 96(%rsi),%ymm5,%ymm5+ vpxor 128(%rsi),%ymm12,%ymm12+ vpxor 160(%rsi),%ymm13,%ymm13+ vpxor 192(%rsi),%ymm10,%ymm10+ vpxor 224(%rsi),%ymm15,%ymm15+ vmovdqu %ymm6,0(%rdi)+ vmovdqu %ymm8,32(%rdi)+ vmovdqu %ymm1,64(%rdi)+ vmovdqu %ymm5,96(%rdi)+ vmovdqu %ymm12,128(%rdi)+ vmovdqu %ymm13,160(%rdi)+ vmovdqu %ymm10,192(%rdi)+ vmovdqu %ymm15,224(%rdi)+ je L$done8x++ leaq 256(%rsi),%rsi+ xorq %r9,%r9+ vmovdqa %ymm14,0(%rsp)+ leaq 256(%rdi),%rdi+ subq $256,%rdx+ vmovdqa %ymm2,32(%rsp)+ jmp L$oop_tail8x++.p2align 5+L$320_or_more8x:+ vpxor 0(%rsi),%ymm6,%ymm6+ vpxor 32(%rsi),%ymm8,%ymm8+ vpxor 64(%rsi),%ymm1,%ymm1+ vpxor 96(%rsi),%ymm5,%ymm5+ vpxor 128(%rsi),%ymm12,%ymm12+ vpxor 160(%rsi),%ymm13,%ymm13+ vpxor 192(%rsi),%ymm10,%ymm10+ vpxor 224(%rsi),%ymm15,%ymm15+ vpxor 256(%rsi),%ymm14,%ymm14+ vpxor 288(%rsi),%ymm2,%ymm2+ vmovdqu %ymm6,0(%rdi)+ vmovdqu %ymm8,32(%rdi)+ vmovdqu %ymm1,64(%rdi)+ vmovdqu %ymm5,96(%rdi)+ vmovdqu %ymm12,128(%rdi)+ vmovdqu %ymm13,160(%rdi)+ vmovdqu %ymm10,192(%rdi)+ vmovdqu %ymm15,224(%rdi)+ vmovdqu %ymm14,256(%rdi)+ vmovdqu %ymm2,288(%rdi)+ je L$done8x++ leaq 320(%rsi),%rsi+ xorq %r9,%r9+ vmovdqa %ymm3,0(%rsp)+ leaq 320(%rdi),%rdi+ subq $320,%rdx+ vmovdqa %ymm7,32(%rsp)+ jmp L$oop_tail8x++.p2align 5+L$384_or_more8x:+ vpxor 0(%rsi),%ymm6,%ymm6+ vpxor 32(%rsi),%ymm8,%ymm8+ vpxor 64(%rsi),%ymm1,%ymm1+ vpxor 96(%rsi),%ymm5,%ymm5+ vpxor 128(%rsi),%ymm12,%ymm12+ vpxor 160(%rsi),%ymm13,%ymm13+ vpxor 192(%rsi),%ymm10,%ymm10+ vpxor 224(%rsi),%ymm15,%ymm15+ vpxor 256(%rsi),%ymm14,%ymm14+ vpxor 288(%rsi),%ymm2,%ymm2+ vpxor 320(%rsi),%ymm3,%ymm3+ vpxor 352(%rsi),%ymm7,%ymm7+ vmovdqu %ymm6,0(%rdi)+ vmovdqu %ymm8,32(%rdi)+ vmovdqu %ymm1,64(%rdi)+ vmovdqu %ymm5,96(%rdi)+ vmovdqu %ymm12,128(%rdi)+ vmovdqu %ymm13,160(%rdi)+ vmovdqu %ymm10,192(%rdi)+ vmovdqu %ymm15,224(%rdi)+ vmovdqu %ymm14,256(%rdi)+ vmovdqu %ymm2,288(%rdi)+ vmovdqu %ymm3,320(%rdi)+ vmovdqu %ymm7,352(%rdi)+ je L$done8x++ leaq 384(%rsi),%rsi+ xorq %r9,%r9+ vmovdqa %ymm11,0(%rsp)+ leaq 384(%rdi),%rdi+ subq $384,%rdx+ vmovdqa %ymm9,32(%rsp)+ jmp L$oop_tail8x++.p2align 5+L$448_or_more8x:+ vpxor 0(%rsi),%ymm6,%ymm6+ vpxor 32(%rsi),%ymm8,%ymm8+ vpxor 64(%rsi),%ymm1,%ymm1+ vpxor 96(%rsi),%ymm5,%ymm5+ vpxor 128(%rsi),%ymm12,%ymm12+ vpxor 160(%rsi),%ymm13,%ymm13+ vpxor 192(%rsi),%ymm10,%ymm10+ vpxor 224(%rsi),%ymm15,%ymm15+ vpxor 256(%rsi),%ymm14,%ymm14+ vpxor 288(%rsi),%ymm2,%ymm2+ vpxor 320(%rsi),%ymm3,%ymm3+ vpxor 352(%rsi),%ymm7,%ymm7+ vpxor 384(%rsi),%ymm11,%ymm11+ vpxor 416(%rsi),%ymm9,%ymm9+ vmovdqu %ymm6,0(%rdi)+ vmovdqu %ymm8,32(%rdi)+ vmovdqu %ymm1,64(%rdi)+ vmovdqu %ymm5,96(%rdi)+ vmovdqu %ymm12,128(%rdi)+ vmovdqu %ymm13,160(%rdi)+ vmovdqu %ymm10,192(%rdi)+ vmovdqu %ymm15,224(%rdi)+ vmovdqu %ymm14,256(%rdi)+ vmovdqu %ymm2,288(%rdi)+ vmovdqu %ymm3,320(%rdi)+ vmovdqu %ymm7,352(%rdi)+ vmovdqu %ymm11,384(%rdi)+ vmovdqu %ymm9,416(%rdi)+ je L$done8x++ leaq 448(%rsi),%rsi+ xorq %r9,%r9+ vmovdqa %ymm0,0(%rsp)+ leaq 448(%rdi),%rdi+ subq $448,%rdx+ vmovdqa %ymm4,32(%rsp)++L$oop_tail8x:+ movzbl (%rsi,%r9,1),%eax+ movzbl (%rsp,%r9,1),%ecx+ leaq 1(%r9),%r9+ xorl %ecx,%eax+ movb %al,-1(%rdi,%r9,1)+ decq %rdx+ jnz L$oop_tail8x++L$done8x:+ vzeroall+ leaq (%r10),%rsp+.cfi_def_cfa_register %rsp+L$avx2_epilogue:+ .byte 0xf3,0xc3+.cfi_endproc +
@@ -0,0 +1,2556 @@+.text ++++.p2align 6+.Lzero:+.long 0,0,0,0+.Lone:+.long 1,0,0,0+.Linc:+.long 0,1,2,3+.Lfour:+.long 4,4,4,4+.Lincy:+.long 0,2,4,6,1,3,5,7+.Leight:+.long 8,8,8,8,8,8,8,8+.Lrot16:+.byte 0x2,0x3,0x0,0x1, 0x6,0x7,0x4,0x5, 0xa,0xb,0x8,0x9, 0xe,0xf,0xc,0xd+.Lrot24:+.byte 0x3,0x0,0x1,0x2, 0x7,0x4,0x5,0x6, 0xb,0x8,0x9,0xa, 0xf,0xc,0xd,0xe+.Ltwoy:+.long 2,0,0,0, 2,0,0,0+.p2align 6+.Lzeroz:+.long 0,0,0,0, 1,0,0,0, 2,0,0,0, 3,0,0,0+.Lfourz:+.long 4,0,0,0, 4,0,0,0, 4,0,0,0, 4,0,0,0+.Lincz:+.long 0,1,2,3,4,5,6,7,8,9,10,11,12,13,14,15+.Lsixteen:+.long 16,16,16,16,16,16,16,16,16,16,16,16,16,16,16,16+.Lsigma:+.byte 101,120,112,97,110,100,32,51,50,45,98,121,116,101,32,107,0+.byte 67,104,97,67,104,97,50,48,32,102,111,114,32,120,56,54,95,54,52,44,32,67,82,89,80,84,79,71,65,77,83,32,98,121,32,64,100,111,116,45,97,115,109,0+.globl crypton_chacha20_asm_ctr32+.def crypton_chacha20_asm_ctr32; .scl 2; .type 32; .endef+.p2align 6+crypton_chacha20_asm_ctr32:+ .byte 0xf3,0x0f,0x1e,0xfa+ movq %rdi,8(%rsp)+ movq %rsi,16(%rsp)+ movq %rsp,%rax+.LSEH_begin_crypton_chacha20_asm_ctr32:+++ movq %rcx,%rdi+ movq %rdx,%rsi+ movq %r8,%rdx+ movq %r9,%rcx+ movq 40(%rsp),%r8+ cmpq $0,%rdx+ je .Lno_data+ movq crypton_ia32cap_P+4(%rip),%r9+ testl $512,%r9d+ jnz .Lcrypton_chacha20_asm_ssse3+ pushq %rbx++ pushq %rbp++ pushq %r12++ pushq %r13++ pushq %r14++ pushq %r15++ subq $64+24,%rsp++.Lctr32_body:++ movq %rdx,%rbp++ movq 0(%rcx),%r12+ movq 8(%rcx),%r13+ movq 16(%rcx),%r14+ movq 24(%rcx),%r15+ movq 0(%r8),%rax+ movq 8(%r8),%rdx+ movq %r12,16(%rsp)+ movq %r13,24(%rsp)+ movq %r14,0(%rsp)+ movq %r15,8(%rsp)+ movq %rax,48(%rsp)+ movq %rdx,56(%rsp)+ jmp .Loop_outer++.p2align 5+.Loop_outer:+ movl $0x61707865,%eax+ movl $0x3320646e,%ebx+ movl $0x79622d32,%ecx+ movl $0x6b206574,%edx+ movl 16(%rsp),%r8d+ movl 20(%rsp),%r9d+ movl 24(%rsp),%r10d+ movl 28(%rsp),%r11d+ movl 48(%rsp),%r12d+ movl 52(%rsp),%r13d+ movl 56(%rsp),%r14d+ movq %r15,40(%rsp)+ movl 60(%rsp),%r15d++ movq %rbp,64+0(%rsp)+ movq %rsi,64+8(%rsp)+ movl 0(%rsp),%esi+ movq %rdi,64+16(%rsp)+ movl 4(%rsp),%edi+ movl $10,%ebp+ jmp .Loop++.p2align 5+.Loop:+ addl %r8d,%eax+ xorl %eax,%r12d+ roll $16,%r12d+ addl %r9d,%ebx+ xorl %ebx,%r13d+ roll $16,%r13d+ addl %r12d,%esi+ xorl %esi,%r8d+ roll $12,%r8d+ addl %r13d,%edi+ xorl %edi,%r9d+ roll $12,%r9d+ addl %r8d,%eax+ xorl %eax,%r12d+ roll $8,%r12d+ addl %r9d,%ebx+ xorl %ebx,%r13d+ roll $8,%r13d+ addl %r12d,%esi+ xorl %esi,%r8d+ roll $7,%r8d+ addl %r13d,%edi+ xorl %edi,%r9d+ roll $7,%r9d+ movl %esi,32(%rsp)+ movl %edi,36(%rsp)+ movl 40(%rsp),%esi+ movl 44(%rsp),%edi+ addl %r10d,%ecx+ xorl %ecx,%r14d+ roll $16,%r14d+ addl %r11d,%edx+ xorl %edx,%r15d+ roll $16,%r15d+ addl %r14d,%esi+ xorl %esi,%r10d+ roll $12,%r10d+ addl %r15d,%edi+ xorl %edi,%r11d+ roll $12,%r11d+ addl %r10d,%ecx+ xorl %ecx,%r14d+ roll $8,%r14d+ addl %r11d,%edx+ xorl %edx,%r15d+ roll $8,%r15d+ addl %r14d,%esi+ xorl %esi,%r10d+ roll $7,%r10d+ addl %r15d,%edi+ xorl %edi,%r11d+ roll $7,%r11d+ addl %r9d,%eax+ xorl %eax,%r15d+ roll $16,%r15d+ addl %r10d,%ebx+ xorl %ebx,%r12d+ roll $16,%r12d+ addl %r15d,%esi+ xorl %esi,%r9d+ roll $12,%r9d+ addl %r12d,%edi+ xorl %edi,%r10d+ roll $12,%r10d+ addl %r9d,%eax+ xorl %eax,%r15d+ roll $8,%r15d+ addl %r10d,%ebx+ xorl %ebx,%r12d+ roll $8,%r12d+ addl %r15d,%esi+ xorl %esi,%r9d+ roll $7,%r9d+ addl %r12d,%edi+ xorl %edi,%r10d+ roll $7,%r10d+ movl %esi,40(%rsp)+ movl %edi,44(%rsp)+ movl 32(%rsp),%esi+ movl 36(%rsp),%edi+ addl %r11d,%ecx+ xorl %ecx,%r13d+ roll $16,%r13d+ addl %r8d,%edx+ xorl %edx,%r14d+ roll $16,%r14d+ addl %r13d,%esi+ xorl %esi,%r11d+ roll $12,%r11d+ addl %r14d,%edi+ xorl %edi,%r8d+ roll $12,%r8d+ addl %r11d,%ecx+ xorl %ecx,%r13d+ roll $8,%r13d+ addl %r8d,%edx+ xorl %edx,%r14d+ roll $8,%r14d+ addl %r13d,%esi+ xorl %esi,%r11d+ roll $7,%r11d+ addl %r14d,%edi+ xorl %edi,%r8d+ roll $7,%r8d+ decl %ebp+ jnz .Loop+ addl 0(%rsp),%esi+ addl 4(%rsp),%edi+ movq 64(%rsp),%rbp+ movl %esi,32(%rsp)+ movq 64+8(%rsp),%rsi+ movl %edi,36(%rsp)+ movq 64+16(%rsp),%rdi++ addl $0x61707865,%eax+ addl $0x3320646e,%ebx+ addl $0x79622d32,%ecx+ addl $0x6b206574,%edx+ addl 16(%rsp),%r8d+ addl 20(%rsp),%r9d+ addl 24(%rsp),%r10d+ addl 28(%rsp),%r11d+ addl 48(%rsp),%r12d+ addl 52(%rsp),%r13d+ addl 56(%rsp),%r14d+ addl 60(%rsp),%r15d++ cmpq $64,%rbp+ jb .Ltail++ xorl 0(%rsi),%eax+ xorl 4(%rsi),%ebx+ xorl 8(%rsi),%ecx+ xorl 12(%rsi),%edx+ movl %eax,0(%rdi)+ movl 32(%rsp),%eax+ movl %ebx,4(%rdi)+ movl 36(%rsp),%ebx+ movl %ecx,8(%rdi)+ movl 40(%rsp),%ecx+ movl %edx,12(%rdi)+ movl 44(%rsp),%edx+ xorl 16(%rsi),%r8d+ addl 8(%rsp),%ecx+ xorl 20(%rsi),%r9d+ addl 12(%rsp),%edx+ xorl 24(%rsi),%r10d+ xorl 28(%rsi),%r11d+ xorl 32(%rsi),%eax+ xorl 36(%rsi),%ebx+ xorl 40(%rsi),%ecx+ xorl 44(%rsi),%edx+ xorl 48(%rsi),%r12d+ xorl 52(%rsi),%r13d+ xorl 56(%rsi),%r14d+ xorl 60(%rsi),%r15d+ leaq 64(%rsi),%rsi++ addl $1,48(%rsp)++ movl %r8d,16(%rdi)+ movl %r9d,20(%rdi)+ movl %r10d,24(%rdi)+ movl %r11d,28(%rdi)+ movl %eax,32(%rdi)+ movl %ebx,36(%rdi)+ movl %ecx,40(%rdi)+ movl %edx,44(%rdi)+ movl %r12d,48(%rdi)+ movl %r13d,52(%rdi)+ movl %r14d,56(%rdi)+ movl %r15d,60(%rdi)+ leaq 64(%rdi),%rdi+ movq 8(%rsp),%r15++ subq $64,%rbp+ jnz .Loop_outer++ jmp .Ldone++.p2align 4+.Ltail:+ movl %eax,0(%rsp)+ movl 8(%rsp),%eax+ movl %ebx,4(%rsp)+ movl 12(%rsp),%ebx+ movl %ecx,8(%rsp)+ addl 40(%rsp),%eax+ movl %edx,12(%rsp)+ addl 44(%rsp),%ebx+ movl %r8d,16(%rsp)+ movl %r9d,20(%rsp)+ movl %r10d,24(%rsp)+ movl %r11d,28(%rsp)+ movl %eax,40(%rsp)+ movl %ebx,44(%rsp)+ xorq %rbx,%rbx+ movl %r12d,48(%rsp)+ movl %r13d,52(%rsp)+ movl %r14d,56(%rsp)+ movl %r15d,60(%rsp)++.Loop_tail:+ movzbl (%rsi,%rbx,1),%eax+ movzbl (%rsp,%rbx,1),%edx+ leaq 1(%rbx),%rbx+ xorl %edx,%eax+ movb %al,-1(%rdi,%rbx,1)+ decq %rbp+ jnz .Loop_tail++.Ldone:+ leaq 64+24+48(%rsp),%rsi++ movq -48(%rsi),%r15++ movq -40(%rsi),%r14++ movq -32(%rsi),%r13++ movq -24(%rsi),%r12++ movq -16(%rsi),%rbp++ movq -8(%rsi),%rbx++ leaq (%rsi),%rsp++.Lno_data:+ movq 8(%rsp),%rdi+ movq 16(%rsp),%rsi+ .byte 0xf3,0xc3++.LSEH_end_crypton_chacha20_asm_ctr32:+.def crypton_chacha20_asm_ssse3; .scl 3; .type 32; .endef+.p2align 5+crypton_chacha20_asm_ssse3:+ .byte 0xf3,0x0f,0x1e,0xfa+ movq %rdi,8(%rsp)+ movq %rsi,16(%rsp)+ movq %rsp,%rax+.LSEH_begin_crypton_chacha20_asm_ssse3:+++ movq %rcx,%rdi+ movq %rdx,%rsi+ movq %r8,%rdx+ movq %r9,%rcx+ movq 40(%rsp),%r8+.Lcrypton_chacha20_asm_ssse3:+ movq %rsp,%r10++ testl $2048,%r9d+ jnz .Lcrypton_chacha20_asm_4xop+ cmpq $128,%rdx+ je .Lcrypton_chacha20_asm_128+ ja .Lcrypton_chacha20_asm_4x++.Ldo_sse3_after_all:+ subq $64+40,%rsp+ andq $-16,%rsp+ movaps %xmm6,-40(%r10)+ movaps %xmm7,-24(%r10)+.Lssse3_body:+ movdqa .Lsigma(%rip),%xmm0+ movdqu (%rcx),%xmm1+ movdqu 16(%rcx),%xmm2+ movdqu (%r8),%xmm3+ movdqa .Lrot16(%rip),%xmm6+ movdqa .Lrot24(%rip),%xmm7++ movdqa %xmm0,0(%rsp)+ movdqa %xmm1,16(%rsp)+ movdqa %xmm2,32(%rsp)+ movdqa %xmm3,48(%rsp)+ movq $10,%r8+ jmp .Loop_ssse3++.p2align 5+.Loop_outer_ssse3:+ movdqa .Lone(%rip),%xmm3+ movdqa 0(%rsp),%xmm0+ movdqa 16(%rsp),%xmm1+ movdqa 32(%rsp),%xmm2+ paddd 48(%rsp),%xmm3+ movq $10,%r8+ movdqa %xmm3,48(%rsp)+ jmp .Loop_ssse3++.p2align 5+.Loop_ssse3:+ paddd %xmm1,%xmm0+ pxor %xmm0,%xmm3+.byte 102,15,56,0,222+ paddd %xmm3,%xmm2+ pxor %xmm2,%xmm1+ movdqa %xmm1,%xmm4+ psrld $20,%xmm1+ pslld $12,%xmm4+ por %xmm4,%xmm1+ paddd %xmm1,%xmm0+ pxor %xmm0,%xmm3+.byte 102,15,56,0,223+ paddd %xmm3,%xmm2+ pxor %xmm2,%xmm1+ movdqa %xmm1,%xmm4+ psrld $25,%xmm1+ pslld $7,%xmm4+ por %xmm4,%xmm1+ pshufd $78,%xmm2,%xmm2+ pshufd $57,%xmm1,%xmm1+ pshufd $147,%xmm3,%xmm3+ nop+ paddd %xmm1,%xmm0+ pxor %xmm0,%xmm3+.byte 102,15,56,0,222+ paddd %xmm3,%xmm2+ pxor %xmm2,%xmm1+ movdqa %xmm1,%xmm4+ psrld $20,%xmm1+ pslld $12,%xmm4+ por %xmm4,%xmm1+ paddd %xmm1,%xmm0+ pxor %xmm0,%xmm3+.byte 102,15,56,0,223+ paddd %xmm3,%xmm2+ pxor %xmm2,%xmm1+ movdqa %xmm1,%xmm4+ psrld $25,%xmm1+ pslld $7,%xmm4+ por %xmm4,%xmm1+ pshufd $78,%xmm2,%xmm2+ pshufd $147,%xmm1,%xmm1+ pshufd $57,%xmm3,%xmm3+ decq %r8+ jnz .Loop_ssse3+ paddd 0(%rsp),%xmm0+ paddd 16(%rsp),%xmm1+ paddd 32(%rsp),%xmm2+ paddd 48(%rsp),%xmm3++ cmpq $64,%rdx+ jb .Ltail_ssse3++ movdqu 0(%rsi),%xmm4+ movdqu 16(%rsi),%xmm5+ pxor %xmm4,%xmm0+ movdqu 32(%rsi),%xmm4+ pxor %xmm5,%xmm1+ movdqu 48(%rsi),%xmm5+ leaq 64(%rsi),%rsi+ pxor %xmm4,%xmm2+ pxor %xmm5,%xmm3++ movdqu %xmm0,0(%rdi)+ movdqu %xmm1,16(%rdi)+ movdqu %xmm2,32(%rdi)+ movdqu %xmm3,48(%rdi)+ leaq 64(%rdi),%rdi++ subq $64,%rdx+ jnz .Loop_outer_ssse3++ jmp .Ldone_ssse3++.p2align 4+.Ltail_ssse3:+ movdqa %xmm0,0(%rsp)+ movdqa %xmm1,16(%rsp)+ movdqa %xmm2,32(%rsp)+ movdqa %xmm3,48(%rsp)+ xorq %r8,%r8++.Loop_tail_ssse3:+ movzbl (%rsi,%r8,1),%eax+ movzbl (%rsp,%r8,1),%ecx+ leaq 1(%r8),%r8+ xorl %ecx,%eax+ movb %al,-1(%rdi,%r8,1)+ decq %rdx+ jnz .Loop_tail_ssse3++.Ldone_ssse3:+ movaps -40(%r10),%xmm6+ movaps -24(%r10),%xmm7+ leaq (%r10),%rsp++.Lssse3_epilogue:+ movq 8(%rsp),%rdi+ movq 16(%rsp),%rsi+ .byte 0xf3,0xc3++.LSEH_end_crypton_chacha20_asm_ssse3:+.def crypton_chacha20_asm_128; .scl 3; .type 32; .endef+.p2align 5+crypton_chacha20_asm_128:+ .byte 0xf3,0x0f,0x1e,0xfa+ movq %rdi,8(%rsp)+ movq %rsi,16(%rsp)+ movq %rsp,%rax+.LSEH_begin_crypton_chacha20_asm_128:+++ movq %rcx,%rdi+ movq %rdx,%rsi+ movq %r8,%rdx+ movq %r9,%rcx+ movq 40(%rsp),%r8+.Lcrypton_chacha20_asm_128:+ movq %rsp,%r10++ subq $64+104,%rsp+ andq $-16,%rsp+ movaps %xmm6,-104(%r10)+ movaps %xmm7,-88(%r10)+ movaps %xmm8,-72(%r10)+ movaps %xmm9,-56(%r10)+ movaps %xmm10,-40(%r10)+ movaps %xmm11,-24(%r10)+.L128_body:+ movdqa .Lsigma(%rip),%xmm8+ movdqu (%rcx),%xmm9+ movdqu 16(%rcx),%xmm2+ movdqu (%r8),%xmm3+ movdqa .Lone(%rip),%xmm1+ movdqa .Lrot16(%rip),%xmm6+ movdqa .Lrot24(%rip),%xmm7++ movdqa %xmm8,%xmm10+ movdqa %xmm8,0(%rsp)+ movdqa %xmm9,%xmm11+ movdqa %xmm9,16(%rsp)+ movdqa %xmm2,%xmm0+ movdqa %xmm2,32(%rsp)+ paddd %xmm3,%xmm1+ movdqa %xmm3,48(%rsp)+ movq $10,%r8+ jmp .Loop_128++.p2align 5+.Loop_128:+ paddd %xmm9,%xmm8+ pxor %xmm8,%xmm3+ paddd %xmm11,%xmm10+ pxor %xmm10,%xmm1+.byte 102,15,56,0,222+.byte 102,15,56,0,206+ paddd %xmm3,%xmm2+ paddd %xmm1,%xmm0+ pxor %xmm2,%xmm9+ pxor %xmm0,%xmm11+ movdqa %xmm9,%xmm4+ psrld $20,%xmm9+ movdqa %xmm11,%xmm5+ pslld $12,%xmm4+ psrld $20,%xmm11+ por %xmm4,%xmm9+ pslld $12,%xmm5+ por %xmm5,%xmm11+ paddd %xmm9,%xmm8+ pxor %xmm8,%xmm3+ paddd %xmm11,%xmm10+ pxor %xmm10,%xmm1+.byte 102,15,56,0,223+.byte 102,15,56,0,207+ paddd %xmm3,%xmm2+ paddd %xmm1,%xmm0+ pxor %xmm2,%xmm9+ pxor %xmm0,%xmm11+ movdqa %xmm9,%xmm4+ psrld $25,%xmm9+ movdqa %xmm11,%xmm5+ pslld $7,%xmm4+ psrld $25,%xmm11+ por %xmm4,%xmm9+ pslld $7,%xmm5+ por %xmm5,%xmm11+ pshufd $78,%xmm2,%xmm2+ pshufd $57,%xmm9,%xmm9+ pshufd $147,%xmm3,%xmm3+ pshufd $78,%xmm0,%xmm0+ pshufd $57,%xmm11,%xmm11+ pshufd $147,%xmm1,%xmm1+ paddd %xmm9,%xmm8+ pxor %xmm8,%xmm3+ paddd %xmm11,%xmm10+ pxor %xmm10,%xmm1+.byte 102,15,56,0,222+.byte 102,15,56,0,206+ paddd %xmm3,%xmm2+ paddd %xmm1,%xmm0+ pxor %xmm2,%xmm9+ pxor %xmm0,%xmm11+ movdqa %xmm9,%xmm4+ psrld $20,%xmm9+ movdqa %xmm11,%xmm5+ pslld $12,%xmm4+ psrld $20,%xmm11+ por %xmm4,%xmm9+ pslld $12,%xmm5+ por %xmm5,%xmm11+ paddd %xmm9,%xmm8+ pxor %xmm8,%xmm3+ paddd %xmm11,%xmm10+ pxor %xmm10,%xmm1+.byte 102,15,56,0,223+.byte 102,15,56,0,207+ paddd %xmm3,%xmm2+ paddd %xmm1,%xmm0+ pxor %xmm2,%xmm9+ pxor %xmm0,%xmm11+ movdqa %xmm9,%xmm4+ psrld $25,%xmm9+ movdqa %xmm11,%xmm5+ pslld $7,%xmm4+ psrld $25,%xmm11+ por %xmm4,%xmm9+ pslld $7,%xmm5+ por %xmm5,%xmm11+ pshufd $78,%xmm2,%xmm2+ pshufd $147,%xmm9,%xmm9+ pshufd $57,%xmm3,%xmm3+ pshufd $78,%xmm0,%xmm0+ pshufd $147,%xmm11,%xmm11+ pshufd $57,%xmm1,%xmm1+ decq %r8+ jnz .Loop_128+ paddd 0(%rsp),%xmm8+ paddd 16(%rsp),%xmm9+ paddd 32(%rsp),%xmm2+ paddd 48(%rsp),%xmm3+ paddd .Lone(%rip),%xmm1+ paddd 0(%rsp),%xmm10+ paddd 16(%rsp),%xmm11+ paddd 32(%rsp),%xmm0+ paddd 48(%rsp),%xmm1++ movdqu 0(%rsi),%xmm4+ movdqu 16(%rsi),%xmm5+ pxor %xmm4,%xmm8+ movdqu 32(%rsi),%xmm4+ pxor %xmm5,%xmm9+ movdqu 48(%rsi),%xmm5+ pxor %xmm4,%xmm2+ movdqu 64(%rsi),%xmm4+ pxor %xmm5,%xmm3+ movdqu 80(%rsi),%xmm5+ pxor %xmm4,%xmm10+ movdqu 96(%rsi),%xmm4+ pxor %xmm5,%xmm11+ movdqu 112(%rsi),%xmm5+ pxor %xmm4,%xmm0+ pxor %xmm5,%xmm1++ movdqu %xmm8,0(%rdi)+ movdqu %xmm9,16(%rdi)+ movdqu %xmm2,32(%rdi)+ movdqu %xmm3,48(%rdi)+ movdqu %xmm10,64(%rdi)+ movdqu %xmm11,80(%rdi)+ movdqu %xmm0,96(%rdi)+ movdqu %xmm1,112(%rdi)+ movaps -104(%r10),%xmm6+ movaps -88(%r10),%xmm7+ movaps -72(%r10),%xmm8+ movaps -56(%r10),%xmm9+ movaps -40(%r10),%xmm10+ movaps -24(%r10),%xmm11+ leaq (%r10),%rsp++.L128_epilogue:+ movq 8(%rsp),%rdi+ movq 16(%rsp),%rsi+ .byte 0xf3,0xc3++.LSEH_end_crypton_chacha20_asm_128:+.def crypton_chacha20_asm_4x; .scl 3; .type 32; .endef+.p2align 5+crypton_chacha20_asm_4x:+ .byte 0xf3,0x0f,0x1e,0xfa+ movq %rdi,8(%rsp)+ movq %rsi,16(%rsp)+ movq %rsp,%rax+.LSEH_begin_crypton_chacha20_asm_4x:+++ movq %rcx,%rdi+ movq %rdx,%rsi+ movq %r8,%rdx+ movq %r9,%rcx+ movq 40(%rsp),%r8+.Lcrypton_chacha20_asm_4x:+ movq %rsp,%r10++ movq %r9,%r11+ shrq $32,%r9+ testq $32,%r9+ jnz .Lcrypton_chacha20_asm_8x+ cmpq $192,%rdx+ ja .Lproceed4x++ andq $71303168,%r11+ cmpq $4194304,%r11+ je .Ldo_sse3_after_all++.Lproceed4x:+ subq $0x140+168,%rsp+ andq $-16,%rsp+ movaps %xmm6,-168(%r10)+ movaps %xmm7,-152(%r10)+ movaps %xmm8,-136(%r10)+ movaps %xmm9,-120(%r10)+ movaps %xmm10,-104(%r10)+ movaps %xmm11,-88(%r10)+ movaps %xmm12,-72(%r10)+ movaps %xmm13,-56(%r10)+ movaps %xmm14,-40(%r10)+ movaps %xmm15,-24(%r10)+.L4x_body:+ movdqa .Lsigma(%rip),%xmm11+ movdqu (%rcx),%xmm15+ movdqu 16(%rcx),%xmm7+ movdqu (%r8),%xmm3+ leaq 256(%rsp),%rcx+ leaq .Lrot16(%rip),%r9+ leaq .Lrot24(%rip),%r11++ pshufd $0x00,%xmm11,%xmm8+ pshufd $0x55,%xmm11,%xmm9+ movdqa %xmm8,64(%rsp)+ pshufd $0xaa,%xmm11,%xmm10+ movdqa %xmm9,80(%rsp)+ pshufd $0xff,%xmm11,%xmm11+ movdqa %xmm10,96(%rsp)+ movdqa %xmm11,112(%rsp)++ pshufd $0x00,%xmm15,%xmm12+ pshufd $0x55,%xmm15,%xmm13+ movdqa %xmm12,128-256(%rcx)+ pshufd $0xaa,%xmm15,%xmm14+ movdqa %xmm13,144-256(%rcx)+ pshufd $0xff,%xmm15,%xmm15+ movdqa %xmm14,160-256(%rcx)+ movdqa %xmm15,176-256(%rcx)++ pshufd $0x00,%xmm7,%xmm4+ pshufd $0x55,%xmm7,%xmm5+ movdqa %xmm4,192-256(%rcx)+ pshufd $0xaa,%xmm7,%xmm6+ movdqa %xmm5,208-256(%rcx)+ pshufd $0xff,%xmm7,%xmm7+ movdqa %xmm6,224-256(%rcx)+ movdqa %xmm7,240-256(%rcx)++ pshufd $0x00,%xmm3,%xmm0+ pshufd $0x55,%xmm3,%xmm1+ paddd .Linc(%rip),%xmm0+ pshufd $0xaa,%xmm3,%xmm2+ movdqa %xmm1,272-256(%rcx)+ pshufd $0xff,%xmm3,%xmm3+ movdqa %xmm2,288-256(%rcx)+ movdqa %xmm3,304-256(%rcx)++ jmp .Loop_enter4x++.p2align 5+.Loop_outer4x:+ movdqa 64(%rsp),%xmm8+ movdqa 80(%rsp),%xmm9+ movdqa 96(%rsp),%xmm10+ movdqa 112(%rsp),%xmm11+ movdqa 128-256(%rcx),%xmm12+ movdqa 144-256(%rcx),%xmm13+ movdqa 160-256(%rcx),%xmm14+ movdqa 176-256(%rcx),%xmm15+ movdqa 192-256(%rcx),%xmm4+ movdqa 208-256(%rcx),%xmm5+ movdqa 224-256(%rcx),%xmm6+ movdqa 240-256(%rcx),%xmm7+ movdqa 256-256(%rcx),%xmm0+ movdqa 272-256(%rcx),%xmm1+ movdqa 288-256(%rcx),%xmm2+ movdqa 304-256(%rcx),%xmm3+ paddd .Lfour(%rip),%xmm0++.Loop_enter4x:+ movdqa %xmm6,32(%rsp)+ movdqa %xmm7,48(%rsp)+ movdqa (%r9),%xmm7+ movl $10,%eax+ movdqa %xmm0,256-256(%rcx)+ jmp .Loop4x++.p2align 5+.Loop4x:+ paddd %xmm12,%xmm8+ paddd %xmm13,%xmm9+ pxor %xmm8,%xmm0+ pxor %xmm9,%xmm1+.byte 102,15,56,0,199+.byte 102,15,56,0,207+ paddd %xmm0,%xmm4+ paddd %xmm1,%xmm5+ pxor %xmm4,%xmm12+ pxor %xmm5,%xmm13+ movdqa %xmm12,%xmm6+ pslld $12,%xmm12+ psrld $20,%xmm6+ movdqa %xmm13,%xmm7+ pslld $12,%xmm13+ por %xmm6,%xmm12+ psrld $20,%xmm7+ movdqa (%r11),%xmm6+ por %xmm7,%xmm13+ paddd %xmm12,%xmm8+ paddd %xmm13,%xmm9+ pxor %xmm8,%xmm0+ pxor %xmm9,%xmm1+.byte 102,15,56,0,198+.byte 102,15,56,0,206+ paddd %xmm0,%xmm4+ paddd %xmm1,%xmm5+ pxor %xmm4,%xmm12+ pxor %xmm5,%xmm13+ movdqa %xmm12,%xmm7+ pslld $7,%xmm12+ psrld $25,%xmm7+ movdqa %xmm13,%xmm6+ pslld $7,%xmm13+ por %xmm7,%xmm12+ psrld $25,%xmm6+ movdqa (%r9),%xmm7+ por %xmm6,%xmm13+ movdqa %xmm4,0(%rsp)+ movdqa %xmm5,16(%rsp)+ movdqa 32(%rsp),%xmm4+ movdqa 48(%rsp),%xmm5+ paddd %xmm14,%xmm10+ paddd %xmm15,%xmm11+ pxor %xmm10,%xmm2+ pxor %xmm11,%xmm3+.byte 102,15,56,0,215+.byte 102,15,56,0,223+ paddd %xmm2,%xmm4+ paddd %xmm3,%xmm5+ pxor %xmm4,%xmm14+ pxor %xmm5,%xmm15+ movdqa %xmm14,%xmm6+ pslld $12,%xmm14+ psrld $20,%xmm6+ movdqa %xmm15,%xmm7+ pslld $12,%xmm15+ por %xmm6,%xmm14+ psrld $20,%xmm7+ movdqa (%r11),%xmm6+ por %xmm7,%xmm15+ paddd %xmm14,%xmm10+ paddd %xmm15,%xmm11+ pxor %xmm10,%xmm2+ pxor %xmm11,%xmm3+.byte 102,15,56,0,214+.byte 102,15,56,0,222+ paddd %xmm2,%xmm4+ paddd %xmm3,%xmm5+ pxor %xmm4,%xmm14+ pxor %xmm5,%xmm15+ movdqa %xmm14,%xmm7+ pslld $7,%xmm14+ psrld $25,%xmm7+ movdqa %xmm15,%xmm6+ pslld $7,%xmm15+ por %xmm7,%xmm14+ psrld $25,%xmm6+ movdqa (%r9),%xmm7+ por %xmm6,%xmm15+ paddd %xmm13,%xmm8+ paddd %xmm14,%xmm9+ pxor %xmm8,%xmm3+ pxor %xmm9,%xmm0+.byte 102,15,56,0,223+.byte 102,15,56,0,199+ paddd %xmm3,%xmm4+ paddd %xmm0,%xmm5+ pxor %xmm4,%xmm13+ pxor %xmm5,%xmm14+ movdqa %xmm13,%xmm6+ pslld $12,%xmm13+ psrld $20,%xmm6+ movdqa %xmm14,%xmm7+ pslld $12,%xmm14+ por %xmm6,%xmm13+ psrld $20,%xmm7+ movdqa (%r11),%xmm6+ por %xmm7,%xmm14+ paddd %xmm13,%xmm8+ paddd %xmm14,%xmm9+ pxor %xmm8,%xmm3+ pxor %xmm9,%xmm0+.byte 102,15,56,0,222+.byte 102,15,56,0,198+ paddd %xmm3,%xmm4+ paddd %xmm0,%xmm5+ pxor %xmm4,%xmm13+ pxor %xmm5,%xmm14+ movdqa %xmm13,%xmm7+ pslld $7,%xmm13+ psrld $25,%xmm7+ movdqa %xmm14,%xmm6+ pslld $7,%xmm14+ por %xmm7,%xmm13+ psrld $25,%xmm6+ movdqa (%r9),%xmm7+ por %xmm6,%xmm14+ movdqa %xmm4,32(%rsp)+ movdqa %xmm5,48(%rsp)+ movdqa 0(%rsp),%xmm4+ movdqa 16(%rsp),%xmm5+ paddd %xmm15,%xmm10+ paddd %xmm12,%xmm11+ pxor %xmm10,%xmm1+ pxor %xmm11,%xmm2+.byte 102,15,56,0,207+.byte 102,15,56,0,215+ paddd %xmm1,%xmm4+ paddd %xmm2,%xmm5+ pxor %xmm4,%xmm15+ pxor %xmm5,%xmm12+ movdqa %xmm15,%xmm6+ pslld $12,%xmm15+ psrld $20,%xmm6+ movdqa %xmm12,%xmm7+ pslld $12,%xmm12+ por %xmm6,%xmm15+ psrld $20,%xmm7+ movdqa (%r11),%xmm6+ por %xmm7,%xmm12+ paddd %xmm15,%xmm10+ paddd %xmm12,%xmm11+ pxor %xmm10,%xmm1+ pxor %xmm11,%xmm2+.byte 102,15,56,0,206+.byte 102,15,56,0,214+ paddd %xmm1,%xmm4+ paddd %xmm2,%xmm5+ pxor %xmm4,%xmm15+ pxor %xmm5,%xmm12+ movdqa %xmm15,%xmm7+ pslld $7,%xmm15+ psrld $25,%xmm7+ movdqa %xmm12,%xmm6+ pslld $7,%xmm12+ por %xmm7,%xmm15+ psrld $25,%xmm6+ movdqa (%r9),%xmm7+ por %xmm6,%xmm12+ decl %eax+ jnz .Loop4x++ paddd 64(%rsp),%xmm8+ paddd 80(%rsp),%xmm9+ paddd 96(%rsp),%xmm10+ paddd 112(%rsp),%xmm11++ movdqa %xmm8,%xmm6+ punpckldq %xmm9,%xmm8+ movdqa %xmm10,%xmm7+ punpckldq %xmm11,%xmm10+ punpckhdq %xmm9,%xmm6+ punpckhdq %xmm11,%xmm7+ movdqa %xmm8,%xmm9+ punpcklqdq %xmm10,%xmm8+ movdqa %xmm6,%xmm11+ punpcklqdq %xmm7,%xmm6+ punpckhqdq %xmm10,%xmm9+ punpckhqdq %xmm7,%xmm11+ paddd 128-256(%rcx),%xmm12+ paddd 144-256(%rcx),%xmm13+ paddd 160-256(%rcx),%xmm14+ paddd 176-256(%rcx),%xmm15++ movdqa %xmm8,0(%rsp)+ movdqa %xmm9,16(%rsp)+ movdqa 32(%rsp),%xmm8+ movdqa 48(%rsp),%xmm9++ movdqa %xmm12,%xmm10+ punpckldq %xmm13,%xmm12+ movdqa %xmm14,%xmm7+ punpckldq %xmm15,%xmm14+ punpckhdq %xmm13,%xmm10+ punpckhdq %xmm15,%xmm7+ movdqa %xmm12,%xmm13+ punpcklqdq %xmm14,%xmm12+ movdqa %xmm10,%xmm15+ punpcklqdq %xmm7,%xmm10+ punpckhqdq %xmm14,%xmm13+ punpckhqdq %xmm7,%xmm15+ paddd 192-256(%rcx),%xmm4+ paddd 208-256(%rcx),%xmm5+ paddd 224-256(%rcx),%xmm8+ paddd 240-256(%rcx),%xmm9++ movdqa %xmm6,32(%rsp)+ movdqa %xmm11,48(%rsp)++ movdqa %xmm4,%xmm14+ punpckldq %xmm5,%xmm4+ movdqa %xmm8,%xmm7+ punpckldq %xmm9,%xmm8+ punpckhdq %xmm5,%xmm14+ punpckhdq %xmm9,%xmm7+ movdqa %xmm4,%xmm5+ punpcklqdq %xmm8,%xmm4+ movdqa %xmm14,%xmm9+ punpcklqdq %xmm7,%xmm14+ punpckhqdq %xmm8,%xmm5+ punpckhqdq %xmm7,%xmm9+ paddd 256-256(%rcx),%xmm0+ paddd 272-256(%rcx),%xmm1+ paddd 288-256(%rcx),%xmm2+ paddd 304-256(%rcx),%xmm3++ movdqa %xmm0,%xmm8+ punpckldq %xmm1,%xmm0+ movdqa %xmm2,%xmm7+ punpckldq %xmm3,%xmm2+ punpckhdq %xmm1,%xmm8+ punpckhdq %xmm3,%xmm7+ movdqa %xmm0,%xmm1+ punpcklqdq %xmm2,%xmm0+ movdqa %xmm8,%xmm3+ punpcklqdq %xmm7,%xmm8+ punpckhqdq %xmm2,%xmm1+ punpckhqdq %xmm7,%xmm3+ cmpq $256,%rdx+ jb .Ltail4x++ movdqu 0(%rsi),%xmm6+ movdqu 16(%rsi),%xmm11+ movdqu 32(%rsi),%xmm2+ movdqu 48(%rsi),%xmm7+ pxor 0(%rsp),%xmm6+ pxor %xmm12,%xmm11+ pxor %xmm4,%xmm2+ pxor %xmm0,%xmm7++ movdqu %xmm6,0(%rdi)+ movdqu 64(%rsi),%xmm6+ movdqu %xmm11,16(%rdi)+ movdqu 80(%rsi),%xmm11+ movdqu %xmm2,32(%rdi)+ movdqu 96(%rsi),%xmm2+ movdqu %xmm7,48(%rdi)+ movdqu 112(%rsi),%xmm7+ leaq 128(%rsi),%rsi+ pxor 16(%rsp),%xmm6+ pxor %xmm13,%xmm11+ pxor %xmm5,%xmm2+ pxor %xmm1,%xmm7++ movdqu %xmm6,64(%rdi)+ movdqu 0(%rsi),%xmm6+ movdqu %xmm11,80(%rdi)+ movdqu 16(%rsi),%xmm11+ movdqu %xmm2,96(%rdi)+ movdqu 32(%rsi),%xmm2+ movdqu %xmm7,112(%rdi)+ leaq 128(%rdi),%rdi+ movdqu 48(%rsi),%xmm7+ pxor 32(%rsp),%xmm6+ pxor %xmm10,%xmm11+ pxor %xmm14,%xmm2+ pxor %xmm8,%xmm7++ movdqu %xmm6,0(%rdi)+ movdqu 64(%rsi),%xmm6+ movdqu %xmm11,16(%rdi)+ movdqu 80(%rsi),%xmm11+ movdqu %xmm2,32(%rdi)+ movdqu 96(%rsi),%xmm2+ movdqu %xmm7,48(%rdi)+ movdqu 112(%rsi),%xmm7+ leaq 128(%rsi),%rsi+ pxor 48(%rsp),%xmm6+ pxor %xmm15,%xmm11+ pxor %xmm9,%xmm2+ pxor %xmm3,%xmm7+ movdqu %xmm6,64(%rdi)+ movdqu %xmm11,80(%rdi)+ movdqu %xmm2,96(%rdi)+ movdqu %xmm7,112(%rdi)+ leaq 128(%rdi),%rdi++ subq $256,%rdx+ jnz .Loop_outer4x++ jmp .Ldone4x++.Ltail4x:+ cmpq $192,%rdx+ jae .L192_or_more4x+ cmpq $128,%rdx+ jae .L128_or_more4x+ cmpq $64,%rdx+ jae .L64_or_more4x+++ xorq %r9,%r9++ movdqa %xmm12,16(%rsp)+ movdqa %xmm4,32(%rsp)+ movdqa %xmm0,48(%rsp)+ jmp .Loop_tail4x++.p2align 5+.L64_or_more4x:+ movdqu 0(%rsi),%xmm6+ movdqu 16(%rsi),%xmm11+ movdqu 32(%rsi),%xmm2+ movdqu 48(%rsi),%xmm7+ pxor 0(%rsp),%xmm6+ pxor %xmm12,%xmm11+ pxor %xmm4,%xmm2+ pxor %xmm0,%xmm7+ movdqu %xmm6,0(%rdi)+ movdqu %xmm11,16(%rdi)+ movdqu %xmm2,32(%rdi)+ movdqu %xmm7,48(%rdi)+ je .Ldone4x++ movdqa 16(%rsp),%xmm6+ leaq 64(%rsi),%rsi+ xorq %r9,%r9+ movdqa %xmm6,0(%rsp)+ movdqa %xmm13,16(%rsp)+ leaq 64(%rdi),%rdi+ movdqa %xmm5,32(%rsp)+ subq $64,%rdx+ movdqa %xmm1,48(%rsp)+ jmp .Loop_tail4x++.p2align 5+.L128_or_more4x:+ movdqu 0(%rsi),%xmm6+ movdqu 16(%rsi),%xmm11+ movdqu 32(%rsi),%xmm2+ movdqu 48(%rsi),%xmm7+ pxor 0(%rsp),%xmm6+ pxor %xmm12,%xmm11+ pxor %xmm4,%xmm2+ pxor %xmm0,%xmm7++ movdqu %xmm6,0(%rdi)+ movdqu 64(%rsi),%xmm6+ movdqu %xmm11,16(%rdi)+ movdqu 80(%rsi),%xmm11+ movdqu %xmm2,32(%rdi)+ movdqu 96(%rsi),%xmm2+ movdqu %xmm7,48(%rdi)+ movdqu 112(%rsi),%xmm7+ pxor 16(%rsp),%xmm6+ pxor %xmm13,%xmm11+ pxor %xmm5,%xmm2+ pxor %xmm1,%xmm7+ movdqu %xmm6,64(%rdi)+ movdqu %xmm11,80(%rdi)+ movdqu %xmm2,96(%rdi)+ movdqu %xmm7,112(%rdi)+ je .Ldone4x++ movdqa 32(%rsp),%xmm6+ leaq 128(%rsi),%rsi+ xorq %r9,%r9+ movdqa %xmm6,0(%rsp)+ movdqa %xmm10,16(%rsp)+ leaq 128(%rdi),%rdi+ movdqa %xmm14,32(%rsp)+ subq $128,%rdx+ movdqa %xmm8,48(%rsp)+ jmp .Loop_tail4x++.p2align 5+.L192_or_more4x:+ movdqu 0(%rsi),%xmm6+ movdqu 16(%rsi),%xmm11+ movdqu 32(%rsi),%xmm2+ movdqu 48(%rsi),%xmm7+ pxor 0(%rsp),%xmm6+ pxor %xmm12,%xmm11+ pxor %xmm4,%xmm2+ pxor %xmm0,%xmm7++ movdqu %xmm6,0(%rdi)+ movdqu 64(%rsi),%xmm6+ movdqu %xmm11,16(%rdi)+ movdqu 80(%rsi),%xmm11+ movdqu %xmm2,32(%rdi)+ movdqu 96(%rsi),%xmm2+ movdqu %xmm7,48(%rdi)+ movdqu 112(%rsi),%xmm7+ leaq 128(%rsi),%rsi+ pxor 16(%rsp),%xmm6+ pxor %xmm13,%xmm11+ pxor %xmm5,%xmm2+ pxor %xmm1,%xmm7++ movdqu %xmm6,64(%rdi)+ movdqu 0(%rsi),%xmm6+ movdqu %xmm11,80(%rdi)+ movdqu 16(%rsi),%xmm11+ movdqu %xmm2,96(%rdi)+ movdqu 32(%rsi),%xmm2+ movdqu %xmm7,112(%rdi)+ leaq 128(%rdi),%rdi+ movdqu 48(%rsi),%xmm7+ pxor 32(%rsp),%xmm6+ pxor %xmm10,%xmm11+ pxor %xmm14,%xmm2+ pxor %xmm8,%xmm7+ movdqu %xmm6,0(%rdi)+ movdqu %xmm11,16(%rdi)+ movdqu %xmm2,32(%rdi)+ movdqu %xmm7,48(%rdi)+ je .Ldone4x++ movdqa 48(%rsp),%xmm6+ leaq 64(%rsi),%rsi+ xorq %r9,%r9+ movdqa %xmm6,0(%rsp)+ movdqa %xmm15,16(%rsp)+ leaq 64(%rdi),%rdi+ movdqa %xmm9,32(%rsp)+ subq $192,%rdx+ movdqa %xmm3,48(%rsp)++.Loop_tail4x:+ movzbl (%rsi,%r9,1),%eax+ movzbl (%rsp,%r9,1),%ecx+ leaq 1(%r9),%r9+ xorl %ecx,%eax+ movb %al,-1(%rdi,%r9,1)+ decq %rdx+ jnz .Loop_tail4x++.Ldone4x:+ movaps -168(%r10),%xmm6+ movaps -152(%r10),%xmm7+ movaps -136(%r10),%xmm8+ movaps -120(%r10),%xmm9+ movaps -104(%r10),%xmm10+ movaps -88(%r10),%xmm11+ movaps -72(%r10),%xmm12+ movaps -56(%r10),%xmm13+ movaps -40(%r10),%xmm14+ movaps -24(%r10),%xmm15+ leaq (%r10),%rsp++.L4x_epilogue:+ movq 8(%rsp),%rdi+ movq 16(%rsp),%rsi+ .byte 0xf3,0xc3++.LSEH_end_crypton_chacha20_asm_4x:+.def crypton_chacha20_asm_4xop; .scl 3; .type 32; .endef+.p2align 5+crypton_chacha20_asm_4xop:+ .byte 0xf3,0x0f,0x1e,0xfa+ movq %rdi,8(%rsp)+ movq %rsi,16(%rsp)+ movq %rsp,%rax+.LSEH_begin_crypton_chacha20_asm_4xop:+++ movq %rcx,%rdi+ movq %rdx,%rsi+ movq %r8,%rdx+ movq %r9,%rcx+ movq 40(%rsp),%r8+.Lcrypton_chacha20_asm_4xop:+ movq %rsp,%r10++ subq $0x140+168,%rsp+ andq $-16,%rsp+ movaps %xmm6,-168(%r10)+ movaps %xmm7,-152(%r10)+ movaps %xmm8,-136(%r10)+ movaps %xmm9,-120(%r10)+ movaps %xmm10,-104(%r10)+ movaps %xmm11,-88(%r10)+ movaps %xmm12,-72(%r10)+ movaps %xmm13,-56(%r10)+ movaps %xmm14,-40(%r10)+ movaps %xmm15,-24(%r10)+.L4xop_body:+ vzeroupper++ vmovdqa .Lsigma(%rip),%xmm11+ vmovdqu (%rcx),%xmm3+ vmovdqu 16(%rcx),%xmm15+ vmovdqu (%r8),%xmm7+ leaq 256(%rsp),%rcx++ vpshufd $0x00,%xmm11,%xmm8+ vpshufd $0x55,%xmm11,%xmm9+ vmovdqa %xmm8,64(%rsp)+ vpshufd $0xaa,%xmm11,%xmm10+ vmovdqa %xmm9,80(%rsp)+ vpshufd $0xff,%xmm11,%xmm11+ vmovdqa %xmm10,96(%rsp)+ vmovdqa %xmm11,112(%rsp)++ vpshufd $0x00,%xmm3,%xmm0+ vpshufd $0x55,%xmm3,%xmm1+ vmovdqa %xmm0,128-256(%rcx)+ vpshufd $0xaa,%xmm3,%xmm2+ vmovdqa %xmm1,144-256(%rcx)+ vpshufd $0xff,%xmm3,%xmm3+ vmovdqa %xmm2,160-256(%rcx)+ vmovdqa %xmm3,176-256(%rcx)++ vpshufd $0x00,%xmm15,%xmm12+ vpshufd $0x55,%xmm15,%xmm13+ vmovdqa %xmm12,192-256(%rcx)+ vpshufd $0xaa,%xmm15,%xmm14+ vmovdqa %xmm13,208-256(%rcx)+ vpshufd $0xff,%xmm15,%xmm15+ vmovdqa %xmm14,224-256(%rcx)+ vmovdqa %xmm15,240-256(%rcx)++ vpshufd $0x00,%xmm7,%xmm4+ vpshufd $0x55,%xmm7,%xmm5+ vpaddd .Linc(%rip),%xmm4,%xmm4+ vpshufd $0xaa,%xmm7,%xmm6+ vmovdqa %xmm5,272-256(%rcx)+ vpshufd $0xff,%xmm7,%xmm7+ vmovdqa %xmm6,288-256(%rcx)+ vmovdqa %xmm7,304-256(%rcx)++ jmp .Loop_enter4xop++.p2align 5+.Loop_outer4xop:+ vmovdqa 64(%rsp),%xmm8+ vmovdqa 80(%rsp),%xmm9+ vmovdqa 96(%rsp),%xmm10+ vmovdqa 112(%rsp),%xmm11+ vmovdqa 128-256(%rcx),%xmm0+ vmovdqa 144-256(%rcx),%xmm1+ vmovdqa 160-256(%rcx),%xmm2+ vmovdqa 176-256(%rcx),%xmm3+ vmovdqa 192-256(%rcx),%xmm12+ vmovdqa 208-256(%rcx),%xmm13+ vmovdqa 224-256(%rcx),%xmm14+ vmovdqa 240-256(%rcx),%xmm15+ vmovdqa 256-256(%rcx),%xmm4+ vmovdqa 272-256(%rcx),%xmm5+ vmovdqa 288-256(%rcx),%xmm6+ vmovdqa 304-256(%rcx),%xmm7+ vpaddd .Lfour(%rip),%xmm4,%xmm4++.Loop_enter4xop:+ movl $10,%eax+ vmovdqa %xmm4,256-256(%rcx)+ jmp .Loop4xop++.p2align 5+.Loop4xop:+ vpaddd %xmm0,%xmm8,%xmm8+ vpaddd %xmm1,%xmm9,%xmm9+ vpaddd %xmm2,%xmm10,%xmm10+ vpaddd %xmm3,%xmm11,%xmm11+ vpxor %xmm4,%xmm8,%xmm4+ vpxor %xmm5,%xmm9,%xmm5+ vpxor %xmm6,%xmm10,%xmm6+ vpxor %xmm7,%xmm11,%xmm7+.byte 143,232,120,194,228,16+.byte 143,232,120,194,237,16+.byte 143,232,120,194,246,16+.byte 143,232,120,194,255,16+ vpaddd %xmm4,%xmm12,%xmm12+ vpaddd %xmm5,%xmm13,%xmm13+ vpaddd %xmm6,%xmm14,%xmm14+ vpaddd %xmm7,%xmm15,%xmm15+ vpxor %xmm0,%xmm12,%xmm0+ vpxor %xmm1,%xmm13,%xmm1+ vpxor %xmm14,%xmm2,%xmm2+ vpxor %xmm15,%xmm3,%xmm3+.byte 143,232,120,194,192,12+.byte 143,232,120,194,201,12+.byte 143,232,120,194,210,12+.byte 143,232,120,194,219,12+ vpaddd %xmm8,%xmm0,%xmm8+ vpaddd %xmm9,%xmm1,%xmm9+ vpaddd %xmm2,%xmm10,%xmm10+ vpaddd %xmm3,%xmm11,%xmm11+ vpxor %xmm4,%xmm8,%xmm4+ vpxor %xmm5,%xmm9,%xmm5+ vpxor %xmm6,%xmm10,%xmm6+ vpxor %xmm7,%xmm11,%xmm7+.byte 143,232,120,194,228,8+.byte 143,232,120,194,237,8+.byte 143,232,120,194,246,8+.byte 143,232,120,194,255,8+ vpaddd %xmm4,%xmm12,%xmm12+ vpaddd %xmm5,%xmm13,%xmm13+ vpaddd %xmm6,%xmm14,%xmm14+ vpaddd %xmm7,%xmm15,%xmm15+ vpxor %xmm0,%xmm12,%xmm0+ vpxor %xmm1,%xmm13,%xmm1+ vpxor %xmm14,%xmm2,%xmm2+ vpxor %xmm15,%xmm3,%xmm3+.byte 143,232,120,194,192,7+.byte 143,232,120,194,201,7+.byte 143,232,120,194,210,7+.byte 143,232,120,194,219,7+ vpaddd %xmm1,%xmm8,%xmm8+ vpaddd %xmm2,%xmm9,%xmm9+ vpaddd %xmm3,%xmm10,%xmm10+ vpaddd %xmm0,%xmm11,%xmm11+ vpxor %xmm7,%xmm8,%xmm7+ vpxor %xmm4,%xmm9,%xmm4+ vpxor %xmm5,%xmm10,%xmm5+ vpxor %xmm6,%xmm11,%xmm6+.byte 143,232,120,194,255,16+.byte 143,232,120,194,228,16+.byte 143,232,120,194,237,16+.byte 143,232,120,194,246,16+ vpaddd %xmm7,%xmm14,%xmm14+ vpaddd %xmm4,%xmm15,%xmm15+ vpaddd %xmm5,%xmm12,%xmm12+ vpaddd %xmm6,%xmm13,%xmm13+ vpxor %xmm1,%xmm14,%xmm1+ vpxor %xmm2,%xmm15,%xmm2+ vpxor %xmm12,%xmm3,%xmm3+ vpxor %xmm13,%xmm0,%xmm0+.byte 143,232,120,194,201,12+.byte 143,232,120,194,210,12+.byte 143,232,120,194,219,12+.byte 143,232,120,194,192,12+ vpaddd %xmm8,%xmm1,%xmm8+ vpaddd %xmm9,%xmm2,%xmm9+ vpaddd %xmm3,%xmm10,%xmm10+ vpaddd %xmm0,%xmm11,%xmm11+ vpxor %xmm7,%xmm8,%xmm7+ vpxor %xmm4,%xmm9,%xmm4+ vpxor %xmm5,%xmm10,%xmm5+ vpxor %xmm6,%xmm11,%xmm6+.byte 143,232,120,194,255,8+.byte 143,232,120,194,228,8+.byte 143,232,120,194,237,8+.byte 143,232,120,194,246,8+ vpaddd %xmm7,%xmm14,%xmm14+ vpaddd %xmm4,%xmm15,%xmm15+ vpaddd %xmm5,%xmm12,%xmm12+ vpaddd %xmm6,%xmm13,%xmm13+ vpxor %xmm1,%xmm14,%xmm1+ vpxor %xmm2,%xmm15,%xmm2+ vpxor %xmm12,%xmm3,%xmm3+ vpxor %xmm13,%xmm0,%xmm0+.byte 143,232,120,194,201,7+.byte 143,232,120,194,210,7+.byte 143,232,120,194,219,7+.byte 143,232,120,194,192,7+ decl %eax+ jnz .Loop4xop++ vpaddd 64(%rsp),%xmm8,%xmm8+ vpaddd 80(%rsp),%xmm9,%xmm9+ vpaddd 96(%rsp),%xmm10,%xmm10+ vpaddd 112(%rsp),%xmm11,%xmm11++ vmovdqa %xmm14,32(%rsp)+ vmovdqa %xmm15,48(%rsp)++ vpunpckldq %xmm9,%xmm8,%xmm14+ vpunpckldq %xmm11,%xmm10,%xmm15+ vpunpckhdq %xmm9,%xmm8,%xmm8+ vpunpckhdq %xmm11,%xmm10,%xmm10+ vpunpcklqdq %xmm15,%xmm14,%xmm9+ vpunpckhqdq %xmm15,%xmm14,%xmm14+ vpunpcklqdq %xmm10,%xmm8,%xmm11+ vpunpckhqdq %xmm10,%xmm8,%xmm8+ vpaddd 128-256(%rcx),%xmm0,%xmm0+ vpaddd 144-256(%rcx),%xmm1,%xmm1+ vpaddd 160-256(%rcx),%xmm2,%xmm2+ vpaddd 176-256(%rcx),%xmm3,%xmm3++ vmovdqa %xmm9,0(%rsp)+ vmovdqa %xmm14,16(%rsp)+ vmovdqa 32(%rsp),%xmm9+ vmovdqa 48(%rsp),%xmm14++ vpunpckldq %xmm1,%xmm0,%xmm10+ vpunpckldq %xmm3,%xmm2,%xmm15+ vpunpckhdq %xmm1,%xmm0,%xmm0+ vpunpckhdq %xmm3,%xmm2,%xmm2+ vpunpcklqdq %xmm15,%xmm10,%xmm1+ vpunpckhqdq %xmm15,%xmm10,%xmm10+ vpunpcklqdq %xmm2,%xmm0,%xmm3+ vpunpckhqdq %xmm2,%xmm0,%xmm0+ vpaddd 192-256(%rcx),%xmm12,%xmm12+ vpaddd 208-256(%rcx),%xmm13,%xmm13+ vpaddd 224-256(%rcx),%xmm9,%xmm9+ vpaddd 240-256(%rcx),%xmm14,%xmm14++ vpunpckldq %xmm13,%xmm12,%xmm2+ vpunpckldq %xmm14,%xmm9,%xmm15+ vpunpckhdq %xmm13,%xmm12,%xmm12+ vpunpckhdq %xmm14,%xmm9,%xmm9+ vpunpcklqdq %xmm15,%xmm2,%xmm13+ vpunpckhqdq %xmm15,%xmm2,%xmm2+ vpunpcklqdq %xmm9,%xmm12,%xmm14+ vpunpckhqdq %xmm9,%xmm12,%xmm12+ vpaddd 256-256(%rcx),%xmm4,%xmm4+ vpaddd 272-256(%rcx),%xmm5,%xmm5+ vpaddd 288-256(%rcx),%xmm6,%xmm6+ vpaddd 304-256(%rcx),%xmm7,%xmm7++ vpunpckldq %xmm5,%xmm4,%xmm9+ vpunpckldq %xmm7,%xmm6,%xmm15+ vpunpckhdq %xmm5,%xmm4,%xmm4+ vpunpckhdq %xmm7,%xmm6,%xmm6+ vpunpcklqdq %xmm15,%xmm9,%xmm5+ vpunpckhqdq %xmm15,%xmm9,%xmm9+ vpunpcklqdq %xmm6,%xmm4,%xmm7+ vpunpckhqdq %xmm6,%xmm4,%xmm4+ vmovdqa 0(%rsp),%xmm6+ vmovdqa 16(%rsp),%xmm15++ cmpq $256,%rdx+ jb .Ltail4xop++ vpxor 0(%rsi),%xmm6,%xmm6+ vpxor 16(%rsi),%xmm1,%xmm1+ vpxor 32(%rsi),%xmm13,%xmm13+ vpxor 48(%rsi),%xmm5,%xmm5+ vpxor 64(%rsi),%xmm15,%xmm15+ vpxor 80(%rsi),%xmm10,%xmm10+ vpxor 96(%rsi),%xmm2,%xmm2+ vpxor 112(%rsi),%xmm9,%xmm9+ leaq 128(%rsi),%rsi+ vpxor 0(%rsi),%xmm11,%xmm11+ vpxor 16(%rsi),%xmm3,%xmm3+ vpxor 32(%rsi),%xmm14,%xmm14+ vpxor 48(%rsi),%xmm7,%xmm7+ vpxor 64(%rsi),%xmm8,%xmm8+ vpxor 80(%rsi),%xmm0,%xmm0+ vpxor 96(%rsi),%xmm12,%xmm12+ vpxor 112(%rsi),%xmm4,%xmm4+ leaq 128(%rsi),%rsi++ vmovdqu %xmm6,0(%rdi)+ vmovdqu %xmm1,16(%rdi)+ vmovdqu %xmm13,32(%rdi)+ vmovdqu %xmm5,48(%rdi)+ vmovdqu %xmm15,64(%rdi)+ vmovdqu %xmm10,80(%rdi)+ vmovdqu %xmm2,96(%rdi)+ vmovdqu %xmm9,112(%rdi)+ leaq 128(%rdi),%rdi+ vmovdqu %xmm11,0(%rdi)+ vmovdqu %xmm3,16(%rdi)+ vmovdqu %xmm14,32(%rdi)+ vmovdqu %xmm7,48(%rdi)+ vmovdqu %xmm8,64(%rdi)+ vmovdqu %xmm0,80(%rdi)+ vmovdqu %xmm12,96(%rdi)+ vmovdqu %xmm4,112(%rdi)+ leaq 128(%rdi),%rdi++ subq $256,%rdx+ jnz .Loop_outer4xop++ jmp .Ldone4xop++.p2align 5+.Ltail4xop:+ cmpq $192,%rdx+ jae .L192_or_more4xop+ cmpq $128,%rdx+ jae .L128_or_more4xop+ cmpq $64,%rdx+ jae .L64_or_more4xop++ xorq %r9,%r9+ vmovdqa %xmm6,0(%rsp)+ vmovdqa %xmm1,16(%rsp)+ vmovdqa %xmm13,32(%rsp)+ vmovdqa %xmm5,48(%rsp)+ jmp .Loop_tail4xop++.p2align 5+.L64_or_more4xop:+ vpxor 0(%rsi),%xmm6,%xmm6+ vpxor 16(%rsi),%xmm1,%xmm1+ vpxor 32(%rsi),%xmm13,%xmm13+ vpxor 48(%rsi),%xmm5,%xmm5+ vmovdqu %xmm6,0(%rdi)+ vmovdqu %xmm1,16(%rdi)+ vmovdqu %xmm13,32(%rdi)+ vmovdqu %xmm5,48(%rdi)+ je .Ldone4xop++ leaq 64(%rsi),%rsi+ vmovdqa %xmm15,0(%rsp)+ xorq %r9,%r9+ vmovdqa %xmm10,16(%rsp)+ leaq 64(%rdi),%rdi+ vmovdqa %xmm2,32(%rsp)+ subq $64,%rdx+ vmovdqa %xmm9,48(%rsp)+ jmp .Loop_tail4xop++.p2align 5+.L128_or_more4xop:+ vpxor 0(%rsi),%xmm6,%xmm6+ vpxor 16(%rsi),%xmm1,%xmm1+ vpxor 32(%rsi),%xmm13,%xmm13+ vpxor 48(%rsi),%xmm5,%xmm5+ vpxor 64(%rsi),%xmm15,%xmm15+ vpxor 80(%rsi),%xmm10,%xmm10+ vpxor 96(%rsi),%xmm2,%xmm2+ vpxor 112(%rsi),%xmm9,%xmm9++ vmovdqu %xmm6,0(%rdi)+ vmovdqu %xmm1,16(%rdi)+ vmovdqu %xmm13,32(%rdi)+ vmovdqu %xmm5,48(%rdi)+ vmovdqu %xmm15,64(%rdi)+ vmovdqu %xmm10,80(%rdi)+ vmovdqu %xmm2,96(%rdi)+ vmovdqu %xmm9,112(%rdi)+ je .Ldone4xop++ leaq 128(%rsi),%rsi+ vmovdqa %xmm11,0(%rsp)+ xorq %r9,%r9+ vmovdqa %xmm3,16(%rsp)+ leaq 128(%rdi),%rdi+ vmovdqa %xmm14,32(%rsp)+ subq $128,%rdx+ vmovdqa %xmm7,48(%rsp)+ jmp .Loop_tail4xop++.p2align 5+.L192_or_more4xop:+ vpxor 0(%rsi),%xmm6,%xmm6+ vpxor 16(%rsi),%xmm1,%xmm1+ vpxor 32(%rsi),%xmm13,%xmm13+ vpxor 48(%rsi),%xmm5,%xmm5+ vpxor 64(%rsi),%xmm15,%xmm15+ vpxor 80(%rsi),%xmm10,%xmm10+ vpxor 96(%rsi),%xmm2,%xmm2+ vpxor 112(%rsi),%xmm9,%xmm9+ leaq 128(%rsi),%rsi+ vpxor 0(%rsi),%xmm11,%xmm11+ vpxor 16(%rsi),%xmm3,%xmm3+ vpxor 32(%rsi),%xmm14,%xmm14+ vpxor 48(%rsi),%xmm7,%xmm7++ vmovdqu %xmm6,0(%rdi)+ vmovdqu %xmm1,16(%rdi)+ vmovdqu %xmm13,32(%rdi)+ vmovdqu %xmm5,48(%rdi)+ vmovdqu %xmm15,64(%rdi)+ vmovdqu %xmm10,80(%rdi)+ vmovdqu %xmm2,96(%rdi)+ vmovdqu %xmm9,112(%rdi)+ leaq 128(%rdi),%rdi+ vmovdqu %xmm11,0(%rdi)+ vmovdqu %xmm3,16(%rdi)+ vmovdqu %xmm14,32(%rdi)+ vmovdqu %xmm7,48(%rdi)+ je .Ldone4xop++ leaq 64(%rsi),%rsi+ vmovdqa %xmm8,0(%rsp)+ xorq %r9,%r9+ vmovdqa %xmm0,16(%rsp)+ leaq 64(%rdi),%rdi+ vmovdqa %xmm12,32(%rsp)+ subq $192,%rdx+ vmovdqa %xmm4,48(%rsp)++.Loop_tail4xop:+ movzbl (%rsi,%r9,1),%eax+ movzbl (%rsp,%r9,1),%ecx+ leaq 1(%r9),%r9+ xorl %ecx,%eax+ movb %al,-1(%rdi,%r9,1)+ decq %rdx+ jnz .Loop_tail4xop++.Ldone4xop:+ vzeroupper+ movaps -168(%r10),%xmm6+ movaps -152(%r10),%xmm7+ movaps -136(%r10),%xmm8+ movaps -120(%r10),%xmm9+ movaps -104(%r10),%xmm10+ movaps -88(%r10),%xmm11+ movaps -72(%r10),%xmm12+ movaps -56(%r10),%xmm13+ movaps -40(%r10),%xmm14+ movaps -24(%r10),%xmm15+ leaq (%r10),%rsp++.L4xop_epilogue:+ movq 8(%rsp),%rdi+ movq 16(%rsp),%rsi+ .byte 0xf3,0xc3++.LSEH_end_crypton_chacha20_asm_4xop:+.def crypton_chacha20_asm_avx2; .scl 3; .type 32; .endef+.p2align 5+crypton_chacha20_asm_avx2:+ .byte 0xf3,0x0f,0x1e,0xfa+ movq %rdi,8(%rsp)+ movq %rsi,16(%rsp)+ movq %rsp,%rax+.LSEH_begin_crypton_chacha20_asm_avx2:+++ movq %rcx,%rdi+ movq %rdx,%rsi+ movq %r8,%rdx+ movq %r9,%rcx+ movq 40(%rsp),%r8+.Lcrypton_chacha20_asm_8x:+ movq %rsp,%r10++ subq $0x280+168,%rsp+ andq $-32,%rsp+ movaps %xmm6,-168(%r10)+ movaps %xmm7,-152(%r10)+ movaps %xmm8,-136(%r10)+ movaps %xmm9,-120(%r10)+ movaps %xmm10,-104(%r10)+ movaps %xmm11,-88(%r10)+ movaps %xmm12,-72(%r10)+ movaps %xmm13,-56(%r10)+ movaps %xmm14,-40(%r10)+ movaps %xmm15,-24(%r10)+.Lavx2_body:+ vzeroupper+++++++++++ vbroadcasti128 .Lsigma(%rip),%ymm11+ vbroadcasti128 (%rcx),%ymm3+ vbroadcasti128 16(%rcx),%ymm15+ vbroadcasti128 (%r8),%ymm7+ leaq 256(%rsp),%rcx+ leaq 512(%rsp),%rax+ leaq .Lrot16(%rip),%r9+ leaq .Lrot24(%rip),%r11++ vpshufd $0x00,%ymm11,%ymm8+ vpshufd $0x55,%ymm11,%ymm9+ vmovdqa %ymm8,128-256(%rcx)+ vpshufd $0xaa,%ymm11,%ymm10+ vmovdqa %ymm9,160-256(%rcx)+ vpshufd $0xff,%ymm11,%ymm11+ vmovdqa %ymm10,192-256(%rcx)+ vmovdqa %ymm11,224-256(%rcx)++ vpshufd $0x00,%ymm3,%ymm0+ vpshufd $0x55,%ymm3,%ymm1+ vmovdqa %ymm0,256-256(%rcx)+ vpshufd $0xaa,%ymm3,%ymm2+ vmovdqa %ymm1,288-256(%rcx)+ vpshufd $0xff,%ymm3,%ymm3+ vmovdqa %ymm2,320-256(%rcx)+ vmovdqa %ymm3,352-256(%rcx)++ vpshufd $0x00,%ymm15,%ymm12+ vpshufd $0x55,%ymm15,%ymm13+ vmovdqa %ymm12,384-512(%rax)+ vpshufd $0xaa,%ymm15,%ymm14+ vmovdqa %ymm13,416-512(%rax)+ vpshufd $0xff,%ymm15,%ymm15+ vmovdqa %ymm14,448-512(%rax)+ vmovdqa %ymm15,480-512(%rax)++ vpshufd $0x00,%ymm7,%ymm4+ vpshufd $0x55,%ymm7,%ymm5+ vpaddd .Lincy(%rip),%ymm4,%ymm4+ vpshufd $0xaa,%ymm7,%ymm6+ vmovdqa %ymm5,544-512(%rax)+ vpshufd $0xff,%ymm7,%ymm7+ vmovdqa %ymm6,576-512(%rax)+ vmovdqa %ymm7,608-512(%rax)++ jmp .Loop_enter8x++.p2align 5+.Loop_outer8x:+ vmovdqa 128-256(%rcx),%ymm8+ vmovdqa 160-256(%rcx),%ymm9+ vmovdqa 192-256(%rcx),%ymm10+ vmovdqa 224-256(%rcx),%ymm11+ vmovdqa 256-256(%rcx),%ymm0+ vmovdqa 288-256(%rcx),%ymm1+ vmovdqa 320-256(%rcx),%ymm2+ vmovdqa 352-256(%rcx),%ymm3+ vmovdqa 384-512(%rax),%ymm12+ vmovdqa 416-512(%rax),%ymm13+ vmovdqa 448-512(%rax),%ymm14+ vmovdqa 480-512(%rax),%ymm15+ vmovdqa 512-512(%rax),%ymm4+ vmovdqa 544-512(%rax),%ymm5+ vmovdqa 576-512(%rax),%ymm6+ vmovdqa 608-512(%rax),%ymm7+ vpaddd .Leight(%rip),%ymm4,%ymm4++.Loop_enter8x:+ vmovdqa %ymm14,64(%rsp)+ vmovdqa %ymm15,96(%rsp)+ vbroadcasti128 (%r9),%ymm15+ vmovdqa %ymm4,512-512(%rax)+ movl $10,%eax+ jmp .Loop8x++.p2align 5+.Loop8x:+ vpaddd %ymm0,%ymm8,%ymm8+ vpxor %ymm4,%ymm8,%ymm4+ vpshufb %ymm15,%ymm4,%ymm4+ vpaddd %ymm1,%ymm9,%ymm9+ vpxor %ymm5,%ymm9,%ymm5+ vpshufb %ymm15,%ymm5,%ymm5+ vpaddd %ymm4,%ymm12,%ymm12+ vpxor %ymm0,%ymm12,%ymm0+ vpslld $12,%ymm0,%ymm14+ vpsrld $20,%ymm0,%ymm0+ vpor %ymm0,%ymm14,%ymm0+ vbroadcasti128 (%r11),%ymm14+ vpaddd %ymm5,%ymm13,%ymm13+ vpxor %ymm1,%ymm13,%ymm1+ vpslld $12,%ymm1,%ymm15+ vpsrld $20,%ymm1,%ymm1+ vpor %ymm1,%ymm15,%ymm1+ vpaddd %ymm0,%ymm8,%ymm8+ vpxor %ymm4,%ymm8,%ymm4+ vpshufb %ymm14,%ymm4,%ymm4+ vpaddd %ymm1,%ymm9,%ymm9+ vpxor %ymm5,%ymm9,%ymm5+ vpshufb %ymm14,%ymm5,%ymm5+ vpaddd %ymm4,%ymm12,%ymm12+ vpxor %ymm0,%ymm12,%ymm0+ vpslld $7,%ymm0,%ymm15+ vpsrld $25,%ymm0,%ymm0+ vpor %ymm0,%ymm15,%ymm0+ vbroadcasti128 (%r9),%ymm15+ vpaddd %ymm5,%ymm13,%ymm13+ vpxor %ymm1,%ymm13,%ymm1+ vpslld $7,%ymm1,%ymm14+ vpsrld $25,%ymm1,%ymm1+ vpor %ymm1,%ymm14,%ymm1+ vmovdqa %ymm12,0(%rsp)+ vmovdqa %ymm13,32(%rsp)+ vmovdqa 64(%rsp),%ymm12+ vmovdqa 96(%rsp),%ymm13+ vpaddd %ymm2,%ymm10,%ymm10+ vpxor %ymm6,%ymm10,%ymm6+ vpshufb %ymm15,%ymm6,%ymm6+ vpaddd %ymm3,%ymm11,%ymm11+ vpxor %ymm7,%ymm11,%ymm7+ vpshufb %ymm15,%ymm7,%ymm7+ vpaddd %ymm6,%ymm12,%ymm12+ vpxor %ymm2,%ymm12,%ymm2+ vpslld $12,%ymm2,%ymm14+ vpsrld $20,%ymm2,%ymm2+ vpor %ymm2,%ymm14,%ymm2+ vbroadcasti128 (%r11),%ymm14+ vpaddd %ymm7,%ymm13,%ymm13+ vpxor %ymm3,%ymm13,%ymm3+ vpslld $12,%ymm3,%ymm15+ vpsrld $20,%ymm3,%ymm3+ vpor %ymm3,%ymm15,%ymm3+ vpaddd %ymm2,%ymm10,%ymm10+ vpxor %ymm6,%ymm10,%ymm6+ vpshufb %ymm14,%ymm6,%ymm6+ vpaddd %ymm3,%ymm11,%ymm11+ vpxor %ymm7,%ymm11,%ymm7+ vpshufb %ymm14,%ymm7,%ymm7+ vpaddd %ymm6,%ymm12,%ymm12+ vpxor %ymm2,%ymm12,%ymm2+ vpslld $7,%ymm2,%ymm15+ vpsrld $25,%ymm2,%ymm2+ vpor %ymm2,%ymm15,%ymm2+ vbroadcasti128 (%r9),%ymm15+ vpaddd %ymm7,%ymm13,%ymm13+ vpxor %ymm3,%ymm13,%ymm3+ vpslld $7,%ymm3,%ymm14+ vpsrld $25,%ymm3,%ymm3+ vpor %ymm3,%ymm14,%ymm3+ vpaddd %ymm1,%ymm8,%ymm8+ vpxor %ymm7,%ymm8,%ymm7+ vpshufb %ymm15,%ymm7,%ymm7+ vpaddd %ymm2,%ymm9,%ymm9+ vpxor %ymm4,%ymm9,%ymm4+ vpshufb %ymm15,%ymm4,%ymm4+ vpaddd %ymm7,%ymm12,%ymm12+ vpxor %ymm1,%ymm12,%ymm1+ vpslld $12,%ymm1,%ymm14+ vpsrld $20,%ymm1,%ymm1+ vpor %ymm1,%ymm14,%ymm1+ vbroadcasti128 (%r11),%ymm14+ vpaddd %ymm4,%ymm13,%ymm13+ vpxor %ymm2,%ymm13,%ymm2+ vpslld $12,%ymm2,%ymm15+ vpsrld $20,%ymm2,%ymm2+ vpor %ymm2,%ymm15,%ymm2+ vpaddd %ymm1,%ymm8,%ymm8+ vpxor %ymm7,%ymm8,%ymm7+ vpshufb %ymm14,%ymm7,%ymm7+ vpaddd %ymm2,%ymm9,%ymm9+ vpxor %ymm4,%ymm9,%ymm4+ vpshufb %ymm14,%ymm4,%ymm4+ vpaddd %ymm7,%ymm12,%ymm12+ vpxor %ymm1,%ymm12,%ymm1+ vpslld $7,%ymm1,%ymm15+ vpsrld $25,%ymm1,%ymm1+ vpor %ymm1,%ymm15,%ymm1+ vbroadcasti128 (%r9),%ymm15+ vpaddd %ymm4,%ymm13,%ymm13+ vpxor %ymm2,%ymm13,%ymm2+ vpslld $7,%ymm2,%ymm14+ vpsrld $25,%ymm2,%ymm2+ vpor %ymm2,%ymm14,%ymm2+ vmovdqa %ymm12,64(%rsp)+ vmovdqa %ymm13,96(%rsp)+ vmovdqa 0(%rsp),%ymm12+ vmovdqa 32(%rsp),%ymm13+ vpaddd %ymm3,%ymm10,%ymm10+ vpxor %ymm5,%ymm10,%ymm5+ vpshufb %ymm15,%ymm5,%ymm5+ vpaddd %ymm0,%ymm11,%ymm11+ vpxor %ymm6,%ymm11,%ymm6+ vpshufb %ymm15,%ymm6,%ymm6+ vpaddd %ymm5,%ymm12,%ymm12+ vpxor %ymm3,%ymm12,%ymm3+ vpslld $12,%ymm3,%ymm14+ vpsrld $20,%ymm3,%ymm3+ vpor %ymm3,%ymm14,%ymm3+ vbroadcasti128 (%r11),%ymm14+ vpaddd %ymm6,%ymm13,%ymm13+ vpxor %ymm0,%ymm13,%ymm0+ vpslld $12,%ymm0,%ymm15+ vpsrld $20,%ymm0,%ymm0+ vpor %ymm0,%ymm15,%ymm0+ vpaddd %ymm3,%ymm10,%ymm10+ vpxor %ymm5,%ymm10,%ymm5+ vpshufb %ymm14,%ymm5,%ymm5+ vpaddd %ymm0,%ymm11,%ymm11+ vpxor %ymm6,%ymm11,%ymm6+ vpshufb %ymm14,%ymm6,%ymm6+ vpaddd %ymm5,%ymm12,%ymm12+ vpxor %ymm3,%ymm12,%ymm3+ vpslld $7,%ymm3,%ymm15+ vpsrld $25,%ymm3,%ymm3+ vpor %ymm3,%ymm15,%ymm3+ vbroadcasti128 (%r9),%ymm15+ vpaddd %ymm6,%ymm13,%ymm13+ vpxor %ymm0,%ymm13,%ymm0+ vpslld $7,%ymm0,%ymm14+ vpsrld $25,%ymm0,%ymm0+ vpor %ymm0,%ymm14,%ymm0+ decl %eax+ jnz .Loop8x++ leaq 512(%rsp),%rax+ vpaddd 128-256(%rcx),%ymm8,%ymm8+ vpaddd 160-256(%rcx),%ymm9,%ymm9+ vpaddd 192-256(%rcx),%ymm10,%ymm10+ vpaddd 224-256(%rcx),%ymm11,%ymm11++ vpunpckldq %ymm9,%ymm8,%ymm14+ vpunpckldq %ymm11,%ymm10,%ymm15+ vpunpckhdq %ymm9,%ymm8,%ymm8+ vpunpckhdq %ymm11,%ymm10,%ymm10+ vpunpcklqdq %ymm15,%ymm14,%ymm9+ vpunpckhqdq %ymm15,%ymm14,%ymm14+ vpunpcklqdq %ymm10,%ymm8,%ymm11+ vpunpckhqdq %ymm10,%ymm8,%ymm8+ vpaddd 256-256(%rcx),%ymm0,%ymm0+ vpaddd 288-256(%rcx),%ymm1,%ymm1+ vpaddd 320-256(%rcx),%ymm2,%ymm2+ vpaddd 352-256(%rcx),%ymm3,%ymm3++ vpunpckldq %ymm1,%ymm0,%ymm10+ vpunpckldq %ymm3,%ymm2,%ymm15+ vpunpckhdq %ymm1,%ymm0,%ymm0+ vpunpckhdq %ymm3,%ymm2,%ymm2+ vpunpcklqdq %ymm15,%ymm10,%ymm1+ vpunpckhqdq %ymm15,%ymm10,%ymm10+ vpunpcklqdq %ymm2,%ymm0,%ymm3+ vpunpckhqdq %ymm2,%ymm0,%ymm0+ vperm2i128 $0x20,%ymm1,%ymm9,%ymm15+ vperm2i128 $0x31,%ymm1,%ymm9,%ymm1+ vperm2i128 $0x20,%ymm10,%ymm14,%ymm9+ vperm2i128 $0x31,%ymm10,%ymm14,%ymm10+ vperm2i128 $0x20,%ymm3,%ymm11,%ymm14+ vperm2i128 $0x31,%ymm3,%ymm11,%ymm3+ vperm2i128 $0x20,%ymm0,%ymm8,%ymm11+ vperm2i128 $0x31,%ymm0,%ymm8,%ymm0+ vmovdqa %ymm15,0(%rsp)+ vmovdqa %ymm9,32(%rsp)+ vmovdqa 64(%rsp),%ymm15+ vmovdqa 96(%rsp),%ymm9++ vpaddd 384-512(%rax),%ymm12,%ymm12+ vpaddd 416-512(%rax),%ymm13,%ymm13+ vpaddd 448-512(%rax),%ymm15,%ymm15+ vpaddd 480-512(%rax),%ymm9,%ymm9++ vpunpckldq %ymm13,%ymm12,%ymm2+ vpunpckldq %ymm9,%ymm15,%ymm8+ vpunpckhdq %ymm13,%ymm12,%ymm12+ vpunpckhdq %ymm9,%ymm15,%ymm15+ vpunpcklqdq %ymm8,%ymm2,%ymm13+ vpunpckhqdq %ymm8,%ymm2,%ymm2+ vpunpcklqdq %ymm15,%ymm12,%ymm9+ vpunpckhqdq %ymm15,%ymm12,%ymm12+ vpaddd 512-512(%rax),%ymm4,%ymm4+ vpaddd 544-512(%rax),%ymm5,%ymm5+ vpaddd 576-512(%rax),%ymm6,%ymm6+ vpaddd 608-512(%rax),%ymm7,%ymm7++ vpunpckldq %ymm5,%ymm4,%ymm15+ vpunpckldq %ymm7,%ymm6,%ymm8+ vpunpckhdq %ymm5,%ymm4,%ymm4+ vpunpckhdq %ymm7,%ymm6,%ymm6+ vpunpcklqdq %ymm8,%ymm15,%ymm5+ vpunpckhqdq %ymm8,%ymm15,%ymm15+ vpunpcklqdq %ymm6,%ymm4,%ymm7+ vpunpckhqdq %ymm6,%ymm4,%ymm4+ vperm2i128 $0x20,%ymm5,%ymm13,%ymm8+ vperm2i128 $0x31,%ymm5,%ymm13,%ymm5+ vperm2i128 $0x20,%ymm15,%ymm2,%ymm13+ vperm2i128 $0x31,%ymm15,%ymm2,%ymm15+ vperm2i128 $0x20,%ymm7,%ymm9,%ymm2+ vperm2i128 $0x31,%ymm7,%ymm9,%ymm7+ vperm2i128 $0x20,%ymm4,%ymm12,%ymm9+ vperm2i128 $0x31,%ymm4,%ymm12,%ymm4+ vmovdqa 0(%rsp),%ymm6+ vmovdqa 32(%rsp),%ymm12++ cmpq $512,%rdx+ jb .Ltail8x++ vpxor 0(%rsi),%ymm6,%ymm6+ vpxor 32(%rsi),%ymm8,%ymm8+ vpxor 64(%rsi),%ymm1,%ymm1+ vpxor 96(%rsi),%ymm5,%ymm5+ leaq 128(%rsi),%rsi+ vmovdqu %ymm6,0(%rdi)+ vmovdqu %ymm8,32(%rdi)+ vmovdqu %ymm1,64(%rdi)+ vmovdqu %ymm5,96(%rdi)+ leaq 128(%rdi),%rdi++ vpxor 0(%rsi),%ymm12,%ymm12+ vpxor 32(%rsi),%ymm13,%ymm13+ vpxor 64(%rsi),%ymm10,%ymm10+ vpxor 96(%rsi),%ymm15,%ymm15+ leaq 128(%rsi),%rsi+ vmovdqu %ymm12,0(%rdi)+ vmovdqu %ymm13,32(%rdi)+ vmovdqu %ymm10,64(%rdi)+ vmovdqu %ymm15,96(%rdi)+ leaq 128(%rdi),%rdi++ vpxor 0(%rsi),%ymm14,%ymm14+ vpxor 32(%rsi),%ymm2,%ymm2+ vpxor 64(%rsi),%ymm3,%ymm3+ vpxor 96(%rsi),%ymm7,%ymm7+ leaq 128(%rsi),%rsi+ vmovdqu %ymm14,0(%rdi)+ vmovdqu %ymm2,32(%rdi)+ vmovdqu %ymm3,64(%rdi)+ vmovdqu %ymm7,96(%rdi)+ leaq 128(%rdi),%rdi++ vpxor 0(%rsi),%ymm11,%ymm11+ vpxor 32(%rsi),%ymm9,%ymm9+ vpxor 64(%rsi),%ymm0,%ymm0+ vpxor 96(%rsi),%ymm4,%ymm4+ leaq 128(%rsi),%rsi+ vmovdqu %ymm11,0(%rdi)+ vmovdqu %ymm9,32(%rdi)+ vmovdqu %ymm0,64(%rdi)+ vmovdqu %ymm4,96(%rdi)+ leaq 128(%rdi),%rdi++ subq $512,%rdx+ jnz .Loop_outer8x++ jmp .Ldone8x++.Ltail8x:+ cmpq $448,%rdx+ jae .L448_or_more8x+ cmpq $384,%rdx+ jae .L384_or_more8x+ cmpq $320,%rdx+ jae .L320_or_more8x+ cmpq $256,%rdx+ jae .L256_or_more8x+ cmpq $192,%rdx+ jae .L192_or_more8x+ cmpq $128,%rdx+ jae .L128_or_more8x+ cmpq $64,%rdx+ jae .L64_or_more8x++ xorq %r9,%r9+ vmovdqa %ymm6,0(%rsp)+ vmovdqa %ymm8,32(%rsp)+ jmp .Loop_tail8x++.p2align 5+.L64_or_more8x:+ vpxor 0(%rsi),%ymm6,%ymm6+ vpxor 32(%rsi),%ymm8,%ymm8+ vmovdqu %ymm6,0(%rdi)+ vmovdqu %ymm8,32(%rdi)+ je .Ldone8x++ leaq 64(%rsi),%rsi+ xorq %r9,%r9+ vmovdqa %ymm1,0(%rsp)+ leaq 64(%rdi),%rdi+ subq $64,%rdx+ vmovdqa %ymm5,32(%rsp)+ jmp .Loop_tail8x++.p2align 5+.L128_or_more8x:+ vpxor 0(%rsi),%ymm6,%ymm6+ vpxor 32(%rsi),%ymm8,%ymm8+ vpxor 64(%rsi),%ymm1,%ymm1+ vpxor 96(%rsi),%ymm5,%ymm5+ vmovdqu %ymm6,0(%rdi)+ vmovdqu %ymm8,32(%rdi)+ vmovdqu %ymm1,64(%rdi)+ vmovdqu %ymm5,96(%rdi)+ je .Ldone8x++ leaq 128(%rsi),%rsi+ xorq %r9,%r9+ vmovdqa %ymm12,0(%rsp)+ leaq 128(%rdi),%rdi+ subq $128,%rdx+ vmovdqa %ymm13,32(%rsp)+ jmp .Loop_tail8x++.p2align 5+.L192_or_more8x:+ vpxor 0(%rsi),%ymm6,%ymm6+ vpxor 32(%rsi),%ymm8,%ymm8+ vpxor 64(%rsi),%ymm1,%ymm1+ vpxor 96(%rsi),%ymm5,%ymm5+ vpxor 128(%rsi),%ymm12,%ymm12+ vpxor 160(%rsi),%ymm13,%ymm13+ vmovdqu %ymm6,0(%rdi)+ vmovdqu %ymm8,32(%rdi)+ vmovdqu %ymm1,64(%rdi)+ vmovdqu %ymm5,96(%rdi)+ vmovdqu %ymm12,128(%rdi)+ vmovdqu %ymm13,160(%rdi)+ je .Ldone8x++ leaq 192(%rsi),%rsi+ xorq %r9,%r9+ vmovdqa %ymm10,0(%rsp)+ leaq 192(%rdi),%rdi+ subq $192,%rdx+ vmovdqa %ymm15,32(%rsp)+ jmp .Loop_tail8x++.p2align 5+.L256_or_more8x:+ vpxor 0(%rsi),%ymm6,%ymm6+ vpxor 32(%rsi),%ymm8,%ymm8+ vpxor 64(%rsi),%ymm1,%ymm1+ vpxor 96(%rsi),%ymm5,%ymm5+ vpxor 128(%rsi),%ymm12,%ymm12+ vpxor 160(%rsi),%ymm13,%ymm13+ vpxor 192(%rsi),%ymm10,%ymm10+ vpxor 224(%rsi),%ymm15,%ymm15+ vmovdqu %ymm6,0(%rdi)+ vmovdqu %ymm8,32(%rdi)+ vmovdqu %ymm1,64(%rdi)+ vmovdqu %ymm5,96(%rdi)+ vmovdqu %ymm12,128(%rdi)+ vmovdqu %ymm13,160(%rdi)+ vmovdqu %ymm10,192(%rdi)+ vmovdqu %ymm15,224(%rdi)+ je .Ldone8x++ leaq 256(%rsi),%rsi+ xorq %r9,%r9+ vmovdqa %ymm14,0(%rsp)+ leaq 256(%rdi),%rdi+ subq $256,%rdx+ vmovdqa %ymm2,32(%rsp)+ jmp .Loop_tail8x++.p2align 5+.L320_or_more8x:+ vpxor 0(%rsi),%ymm6,%ymm6+ vpxor 32(%rsi),%ymm8,%ymm8+ vpxor 64(%rsi),%ymm1,%ymm1+ vpxor 96(%rsi),%ymm5,%ymm5+ vpxor 128(%rsi),%ymm12,%ymm12+ vpxor 160(%rsi),%ymm13,%ymm13+ vpxor 192(%rsi),%ymm10,%ymm10+ vpxor 224(%rsi),%ymm15,%ymm15+ vpxor 256(%rsi),%ymm14,%ymm14+ vpxor 288(%rsi),%ymm2,%ymm2+ vmovdqu %ymm6,0(%rdi)+ vmovdqu %ymm8,32(%rdi)+ vmovdqu %ymm1,64(%rdi)+ vmovdqu %ymm5,96(%rdi)+ vmovdqu %ymm12,128(%rdi)+ vmovdqu %ymm13,160(%rdi)+ vmovdqu %ymm10,192(%rdi)+ vmovdqu %ymm15,224(%rdi)+ vmovdqu %ymm14,256(%rdi)+ vmovdqu %ymm2,288(%rdi)+ je .Ldone8x++ leaq 320(%rsi),%rsi+ xorq %r9,%r9+ vmovdqa %ymm3,0(%rsp)+ leaq 320(%rdi),%rdi+ subq $320,%rdx+ vmovdqa %ymm7,32(%rsp)+ jmp .Loop_tail8x++.p2align 5+.L384_or_more8x:+ vpxor 0(%rsi),%ymm6,%ymm6+ vpxor 32(%rsi),%ymm8,%ymm8+ vpxor 64(%rsi),%ymm1,%ymm1+ vpxor 96(%rsi),%ymm5,%ymm5+ vpxor 128(%rsi),%ymm12,%ymm12+ vpxor 160(%rsi),%ymm13,%ymm13+ vpxor 192(%rsi),%ymm10,%ymm10+ vpxor 224(%rsi),%ymm15,%ymm15+ vpxor 256(%rsi),%ymm14,%ymm14+ vpxor 288(%rsi),%ymm2,%ymm2+ vpxor 320(%rsi),%ymm3,%ymm3+ vpxor 352(%rsi),%ymm7,%ymm7+ vmovdqu %ymm6,0(%rdi)+ vmovdqu %ymm8,32(%rdi)+ vmovdqu %ymm1,64(%rdi)+ vmovdqu %ymm5,96(%rdi)+ vmovdqu %ymm12,128(%rdi)+ vmovdqu %ymm13,160(%rdi)+ vmovdqu %ymm10,192(%rdi)+ vmovdqu %ymm15,224(%rdi)+ vmovdqu %ymm14,256(%rdi)+ vmovdqu %ymm2,288(%rdi)+ vmovdqu %ymm3,320(%rdi)+ vmovdqu %ymm7,352(%rdi)+ je .Ldone8x++ leaq 384(%rsi),%rsi+ xorq %r9,%r9+ vmovdqa %ymm11,0(%rsp)+ leaq 384(%rdi),%rdi+ subq $384,%rdx+ vmovdqa %ymm9,32(%rsp)+ jmp .Loop_tail8x++.p2align 5+.L448_or_more8x:+ vpxor 0(%rsi),%ymm6,%ymm6+ vpxor 32(%rsi),%ymm8,%ymm8+ vpxor 64(%rsi),%ymm1,%ymm1+ vpxor 96(%rsi),%ymm5,%ymm5+ vpxor 128(%rsi),%ymm12,%ymm12+ vpxor 160(%rsi),%ymm13,%ymm13+ vpxor 192(%rsi),%ymm10,%ymm10+ vpxor 224(%rsi),%ymm15,%ymm15+ vpxor 256(%rsi),%ymm14,%ymm14+ vpxor 288(%rsi),%ymm2,%ymm2+ vpxor 320(%rsi),%ymm3,%ymm3+ vpxor 352(%rsi),%ymm7,%ymm7+ vpxor 384(%rsi),%ymm11,%ymm11+ vpxor 416(%rsi),%ymm9,%ymm9+ vmovdqu %ymm6,0(%rdi)+ vmovdqu %ymm8,32(%rdi)+ vmovdqu %ymm1,64(%rdi)+ vmovdqu %ymm5,96(%rdi)+ vmovdqu %ymm12,128(%rdi)+ vmovdqu %ymm13,160(%rdi)+ vmovdqu %ymm10,192(%rdi)+ vmovdqu %ymm15,224(%rdi)+ vmovdqu %ymm14,256(%rdi)+ vmovdqu %ymm2,288(%rdi)+ vmovdqu %ymm3,320(%rdi)+ vmovdqu %ymm7,352(%rdi)+ vmovdqu %ymm11,384(%rdi)+ vmovdqu %ymm9,416(%rdi)+ je .Ldone8x++ leaq 448(%rsi),%rsi+ xorq %r9,%r9+ vmovdqa %ymm0,0(%rsp)+ leaq 448(%rdi),%rdi+ subq $448,%rdx+ vmovdqa %ymm4,32(%rsp)++.Loop_tail8x:+ movzbl (%rsi,%r9,1),%eax+ movzbl (%rsp,%r9,1),%ecx+ leaq 1(%r9),%r9+ xorl %ecx,%eax+ movb %al,-1(%rdi,%r9,1)+ decq %rdx+ jnz .Loop_tail8x++.Ldone8x:+ vzeroall+ movaps -168(%r10),%xmm6+ movaps -152(%r10),%xmm7+ movaps -136(%r10),%xmm8+ movaps -120(%r10),%xmm9+ movaps -104(%r10),%xmm10+ movaps -88(%r10),%xmm11+ movaps -72(%r10),%xmm12+ movaps -56(%r10),%xmm13+ movaps -40(%r10),%xmm14+ movaps -24(%r10),%xmm15+ leaq (%r10),%rsp++.Lavx2_epilogue:+ movq 8(%rsp),%rdi+ movq 16(%rsp),%rsi+ .byte 0xf3,0xc3++.LSEH_end_crypton_chacha20_asm_avx2:++.def se_handler; .scl 3; .type 32; .endef+.p2align 4+se_handler:+ .byte 0xf3,0x0f,0x1e,0xfa++ pushq %rsi+ pushq %rdi+ pushq %rbx+ pushq %rbp+ pushq %r12+ pushq %r13+ pushq %r14+ pushq %r15+ pushfq+ subq $64,%rsp++ movq 120(%r8),%rax+ movq 248(%r8),%rbx++ movq 8(%r9),%rsi+ movq 56(%r9),%r11++ leaq .Lctr32_body(%rip),%r10+ cmpq %r10,%rbx+ jb .Lcommon_seh_tail++ movq 152(%r8),%rax++ leaq .Lno_data(%rip),%r10+ cmpq %r10,%rbx+ jae .Lcommon_seh_tail++ leaq 64+24+48(%rax),%rax++ movq -8(%rax),%rbx+ movq -16(%rax),%rbp+ movq -24(%rax),%r12+ movq -32(%rax),%r13+ movq -40(%rax),%r14+ movq -48(%rax),%r15+ movq %rbx,144(%r8)+ movq %rbp,160(%r8)+ movq %r12,216(%r8)+ movq %r13,224(%r8)+ movq %r14,232(%r8)+ movq %r15,240(%r8)++.Lcommon_seh_tail:+ movq 8(%rax),%rdi+ movq 16(%rax),%rsi+ movq %rax,152(%r8)+ movq %rsi,168(%r8)+ movq %rdi,176(%r8)++ movq 40(%r9),%rdi+ movq %r8,%rsi+ movl $154,%ecx+.long 0xa548f3fc++ movq %r9,%rsi+ xorq %rcx,%rcx+ movq 8(%rsi),%rdx+ movq 0(%rsi),%r8+ movq 16(%rsi),%r9+ movq 40(%rsi),%r10+ leaq 56(%rsi),%r11+ leaq 24(%rsi),%r12+ movq %r10,32(%rsp)+ movq %r11,40(%rsp)+ movq %r12,48(%rsp)+ movq %rcx,56(%rsp)+ call *__imp_RtlVirtualUnwind(%rip)++ movl $1,%eax+ addq $64,%rsp+ popfq+ popq %r15+ popq %r14+ popq %r13+ popq %r12+ popq %rbp+ popq %rbx+ popq %rdi+ popq %rsi+ .byte 0xf3,0xc3+++.def simd_handler; .scl 3; .type 32; .endef+.p2align 4+simd_handler:+ .byte 0xf3,0x0f,0x1e,0xfa++ pushq %rsi+ pushq %rdi+ pushq %rbx+ pushq %rbp+ pushq %r12+ pushq %r13+ pushq %r14+ pushq %r15+ pushfq+ subq $64,%rsp++ movq 120(%r8),%rax+ movq 248(%r8),%rbx++ movq 8(%r9),%rsi+ movq 56(%r9),%r11++ movl 0(%r11),%r10d+ leaq (%rsi,%r10,1),%r10+ cmpq %r10,%rbx+ jb .Lcommon_seh_tail++ movq 200(%r8),%rax++ movl 4(%r11),%r10d+ movl 8(%r11),%ecx+ leaq (%rsi,%r10,1),%r10+ cmpq %r10,%rbx+ jae .Lcommon_seh_tail++ negq %rcx+ leaq -8(%rax,%rcx,1),%rsi+ leaq 512(%r8),%rdi+ negl %ecx+ shrl $3,%ecx+.long 0xa548f3fc++ jmp .Lcommon_seh_tail+++.section .pdata+.p2align 2+.rva .LSEH_begin_crypton_chacha20_asm_ctr32+.rva .LSEH_end_crypton_chacha20_asm_ctr32+.rva .LSEH_info_crypton_chacha20_asm_ctr32++.rva .LSEH_begin_crypton_chacha20_asm_ssse3+.rva .LSEH_end_crypton_chacha20_asm_ssse3+.rva .LSEH_info_crypton_chacha20_asm_ssse3++.rva .LSEH_begin_crypton_chacha20_asm_128+.rva .LSEH_end_crypton_chacha20_asm_128+.rva .LSEH_info_crypton_chacha20_asm_128++.rva .LSEH_begin_crypton_chacha20_asm_4x+.rva .LSEH_end_crypton_chacha20_asm_4x+.rva .LSEH_info_crypton_chacha20_asm_4x+.rva .LSEH_begin_crypton_chacha20_asm_4xop+.rva .LSEH_end_crypton_chacha20_asm_4xop+.rva .LSEH_info_crypton_chacha20_asm_4xop+.rva .LSEH_begin_crypton_chacha20_asm_avx2+.rva .LSEH_end_crypton_chacha20_asm_avx2+.rva .LSEH_info_crypton_chacha20_asm_avx2+.section .xdata+.p2align 3+.LSEH_info_crypton_chacha20_asm_ctr32:+.byte 9,0,0,0+.rva se_handler++.LSEH_info_crypton_chacha20_asm_ssse3:+.byte 9,0,0,0+.rva simd_handler+.rva .Lssse3_body,.Lssse3_epilogue+.long 0x20,0++.LSEH_info_crypton_chacha20_asm_128:+.byte 9,0,0,0+.rva simd_handler+.rva .L128_body,.L128_epilogue+.long 0x60,0++.LSEH_info_crypton_chacha20_asm_4x:+.byte 9,0,0,0+.rva simd_handler+.rva .L4x_body,.L4x_epilogue+.long 0xa0,0+.LSEH_info_crypton_chacha20_asm_4xop:+.byte 9,0,0,0+.rva simd_handler+.rva .L4xop_body,.L4xop_epilogue+.long 0xa0,0+.LSEH_info_crypton_chacha20_asm_avx2:+.byte 9,0,0,0+.rva simd_handler+.rva .Lavx2_body,.Lavx2_epilogue+.long 0xa0,0
@@ -0,0 +1,4044 @@+#!/usr/bin/env perl+#+# ====================================================================+# Written by Andy Polyakov, @dot-asm, initially for the OpenSSL+# project.+# ====================================================================+#+# November 2014+#+# ChaCha20 for x86_64.+#+# December 2016+#+# Add AVX512F code path.+#+# December 2017+#+# Add AVX512VL code path.+#+# Performance in cycles per byte out of large buffer.+#+# IALU/gcc 4.8(i) 1x/2xSSSE3(ii) 4xSSSE3 NxAVX(v)+#+# P4 9.48/+99% - -+# Core2 7.83/+55% 7.90/5.76 4.35+# Westmere 7.19/+50% 5.60/4.50 3.00+# Sandy Bridge 8.31/+42% 5.45/4.00 2.72+# Ivy Bridge 6.71/+46% 5.40/? 2.41+# Haswell 5.92/+43% 5.20/3.45 2.42 1.23+# Skylake[-X] 5.87/+39% 4.70/3.22 2.31 1.19[0.80(vi)]+# Cannon Lake 5.87/+39% 4.60/3.20 2.26 0.80(vi)+# Rocket Lake 5.86/+39% ? 2.30 0.58+# Silvermont 12.0/+33% 7.75/6.90 7.03(iii)+# Knights L 11.7/- ? 9.60(iii) 0.80+# Goldmont 10.6/+17% 5.10/3.52 3.28+# Sledgehammer 7.28/+52% - -+# Bulldozer 9.66/+28% 9.85/5.35(iv) 3.06(iv)+# Ryzen 5.96/+50% 5.19/3.00 2.40 2.09+# VIA Nano 10.5/+46% 6.72/6.88 6.05+#+# (i) compared to older gcc 3.x one can observe >2x improvement on+# most platforms;+# (ii) 2xSSSE3 is code path optimized specifically for 128 bytes used+# by chacha20_poly1305_tls_cipher, results are EVP-free;+# (iii) this is not optimal result for Atom because of MSROM+# limitations, SSE2 can do better, but gain is considered too+# low to justify the [maintenance] effort;+# (iv) Bulldozer actually executes 4xXOP code path that delivers 2.20+# and 4.85 for 128-byte inputs;+# (v) 8xAVX2, 8xAVX512VL or 16xAVX512F, whichever best applicable;+# (vi) even though Skylake-X can execute AVX512F code and deliver 0.57+# cpb in single thread, the corresponding capability is suppressed;++$flavour = shift;+$output = shift;+if ($flavour =~ /\./) { $output = $flavour; undef $flavour; }++$win64=0; $win64=1 if ($flavour =~ /[nm]asm|mingw64/ || $output =~ /\.asm$/);++$0 =~ m/(.*[\/\\])[^\/\\]+$/; $dir=$1;+( $xlate="${dir}x86_64-xlate.pl" and -f $xlate ) or+( $xlate="${dir}../../perlasm/x86_64-xlate.pl" and -f $xlate) or+die "can't locate x86_64-xlate.pl";++$avx=undef;++if (!defined($avx) && $win64 && ($flavour =~ /nasm/ || $ENV{ASM} =~ /nasm/) &&+ ($ENV{ASM} //= "nasm") &&+ `"$ENV{ASM}" -v 2>&1` =~ /NASM version ([0-9]+\.[0-9]+)(?:\.([0-9]+))?/) {+ $avx = ($1>=2.09) + ($1>=2.10) + ($1>=2.12);+ $avx += 1 if ($1==2.11 && $2>=8);+}++if (!defined($avx) && $win64 && ($flavour =~ /masm/ || $ENV{ASM} =~ /ml64/) &&+ ($ENV{ASM} //= "ml64") &&+ `"$ENV{ASM}" 2>&1` =~ /Version ([0-9]+)\./) {+ $avx = ($1>=10) + ($1>=11) + ($1>=14);+}++$ENV{CC} //= "cc";+if (!defined($avx) && `$ENV{CC} -Wa,-v -c -o /dev/zero -x assembler /dev/null 2>&1`+ =~ /GNU assembler version ([0-9]+)\.([0-9]+)/) {+ my $ver = $1 + $2/100.0; # 3.1->3.01, 3.10->3.10+ $avx = ($ver>=2.19) + ($ver>=2.22) + ($ver>=2.25);+}++if (!defined($avx) && `$ENV{CC} -v 2>&1`+ =~ /((?:^clang|LLVM) version|.*based on LLVM) ([0-9]+)\.([0-9]+)/) {+ my $ver = $2 + $3/100.0; # 3.1->3.01, 3.10->3.10+ $avx = ($ver>=3.0) + ($ver>3.0);+ $avx += ($ver>=7.0) if ($1 =~ /^clang/);+}++open OUT,"| \"$^X\" \"$xlate\" $flavour \"$output\"";+*STDOUT=*OUT;++# input parameter block+($out,$inp,$len,$key,$counter)=("%rdi","%rsi","%rdx","%rcx","%r8");++$code.=<<___;+.text++.extern OPENSSL_ia32cap_P++.align 64+.Lzero:+.long 0,0,0,0+.Lone:+.long 1,0,0,0+.Linc:+.long 0,1,2,3+.Lfour:+.long 4,4,4,4+.Lincy:+.long 0,2,4,6,1,3,5,7+.Leight:+.long 8,8,8,8,8,8,8,8+.Lrot16:+.byte 0x2,0x3,0x0,0x1, 0x6,0x7,0x4,0x5, 0xa,0xb,0x8,0x9, 0xe,0xf,0xc,0xd+.Lrot24:+.byte 0x3,0x0,0x1,0x2, 0x7,0x4,0x5,0x6, 0xb,0x8,0x9,0xa, 0xf,0xc,0xd,0xe+.Ltwoy:+.long 2,0,0,0, 2,0,0,0+.align 64+.Lzeroz:+.long 0,0,0,0, 1,0,0,0, 2,0,0,0, 3,0,0,0+.Lfourz:+.long 4,0,0,0, 4,0,0,0, 4,0,0,0, 4,0,0,0+.Lincz:+.long 0,1,2,3,4,5,6,7,8,9,10,11,12,13,14,15+.Lsixteen:+.long 16,16,16,16,16,16,16,16,16,16,16,16,16,16,16,16+.Lsigma:+.asciz "expand 32-byte k"+.asciz "ChaCha20 for x86_64, CRYPTOGAMS by \@dot-asm"+___++sub AUTOLOAD() # thunk [simplified] 32-bit style perlasm+{ my $opcode = $AUTOLOAD; $opcode =~ s/.*:://;+ my $arg = pop;+ $arg = "\$$arg" if ($arg*1 eq $arg);+ $code .= "\t$opcode\t".join(',',$arg,reverse @_)."\n";+}++@x=("%eax","%ebx","%ecx","%edx",map("%r${_}d",(8..11)),+ "%nox","%nox","%nox","%nox",map("%r${_}d",(12..15)));+@t=("%esi","%edi");++sub ROUND { # critical path is 24 cycles per round+my ($a0,$b0,$c0,$d0)=@_;+my ($a1,$b1,$c1,$d1)=map(($_&~3)+(($_+1)&3),($a0,$b0,$c0,$d0));+my ($a2,$b2,$c2,$d2)=map(($_&~3)+(($_+1)&3),($a1,$b1,$c1,$d1));+my ($a3,$b3,$c3,$d3)=map(($_&~3)+(($_+1)&3),($a2,$b2,$c2,$d2));+my ($xc,$xc_)=map("\"$_\"",@t);+my @x=map("\"$_\"",@x);++ # Consider order in which variables are addressed by their+ # index:+ #+ # a b c d+ #+ # 0 4 8 12 < even round+ # 1 5 9 13+ # 2 6 10 14+ # 3 7 11 15+ # 0 5 10 15 < odd round+ # 1 6 11 12+ # 2 7 8 13+ # 3 4 9 14+ #+ # 'a', 'b' and 'd's are permanently allocated in registers,+ # @x[0..7,12..15], while 'c's are maintained in memory. If+ # you observe 'c' column, you'll notice that pair of 'c's is+ # invariant between rounds. This means that we have to reload+ # them once per round, in the middle. This is why you'll see+ # bunch of 'c' stores and loads in the middle, but none in+ # the beginning or end.++ # Normally instructions would be interleaved to favour in-order+ # execution. Generally out-of-order cores manage it gracefully,+ # but not this time for some reason. As in-order execution+ # cores are dying breed, old Atom is the only one around,+ # instructions are left uninterleaved. Besides, Atom is better+ # off executing 1xSSSE3 code anyway...++ (+ "&add (@x[$a0],@x[$b0])", # Q1+ "&xor (@x[$d0],@x[$a0])",+ "&rol (@x[$d0],16)",+ "&add (@x[$a1],@x[$b1])", # Q2+ "&xor (@x[$d1],@x[$a1])",+ "&rol (@x[$d1],16)",++ "&add ($xc,@x[$d0])",+ "&xor (@x[$b0],$xc)",+ "&rol (@x[$b0],12)",+ "&add ($xc_,@x[$d1])",+ "&xor (@x[$b1],$xc_)",+ "&rol (@x[$b1],12)",++ "&add (@x[$a0],@x[$b0])",+ "&xor (@x[$d0],@x[$a0])",+ "&rol (@x[$d0],8)",+ "&add (@x[$a1],@x[$b1])",+ "&xor (@x[$d1],@x[$a1])",+ "&rol (@x[$d1],8)",++ "&add ($xc,@x[$d0])",+ "&xor (@x[$b0],$xc)",+ "&rol (@x[$b0],7)",+ "&add ($xc_,@x[$d1])",+ "&xor (@x[$b1],$xc_)",+ "&rol (@x[$b1],7)",++ "&mov (\"4*$c0(%rsp)\",$xc)", # reload pair of 'c's+ "&mov (\"4*$c1(%rsp)\",$xc_)",+ "&mov ($xc,\"4*$c2(%rsp)\")",+ "&mov ($xc_,\"4*$c3(%rsp)\")",++ "&add (@x[$a2],@x[$b2])", # Q3+ "&xor (@x[$d2],@x[$a2])",+ "&rol (@x[$d2],16)",+ "&add (@x[$a3],@x[$b3])", # Q4+ "&xor (@x[$d3],@x[$a3])",+ "&rol (@x[$d3],16)",++ "&add ($xc,@x[$d2])",+ "&xor (@x[$b2],$xc)",+ "&rol (@x[$b2],12)",+ "&add ($xc_,@x[$d3])",+ "&xor (@x[$b3],$xc_)",+ "&rol (@x[$b3],12)",++ "&add (@x[$a2],@x[$b2])",+ "&xor (@x[$d2],@x[$a2])",+ "&rol (@x[$d2],8)",+ "&add (@x[$a3],@x[$b3])",+ "&xor (@x[$d3],@x[$a3])",+ "&rol (@x[$d3],8)",++ "&add ($xc,@x[$d2])",+ "&xor (@x[$b2],$xc)",+ "&rol (@x[$b2],7)",+ "&add ($xc_,@x[$d3])",+ "&xor (@x[$b3],$xc_)",+ "&rol (@x[$b3],7)"+ );+}++########################################################################+# Generic code path that handles all lengths on pre-SSSE3 processors.+$code.=<<___;+.globl ChaCha20_ctr32+.type ChaCha20_ctr32,\@function,5+.align 64+ChaCha20_ctr32:+.cfi_startproc+ cmp \$0,$len+ je .Lno_data+___+ if ($flavour !~ /kernel/) {+$code.=<<___;+ mov OPENSSL_ia32cap_P+4(%rip),%r9+___+$code.=<<___ if ($avx>2);+ bt \$48,%r9 # check for AVX512F+ jc .LChaCha20_avx512+ test %r9,%r9 # check for AVX512VL+ js .LChaCha20_avx512vl+___+$code.=<<___;+ test \$`1<<(41-32)`,%r9d+ jnz .LChaCha20_ssse3+___+ }+$code.=<<___;+ push %rbx+.cfi_push %rbx+ push %rbp+.cfi_push %rbp+ push %r12+.cfi_push %r12+ push %r13+.cfi_push %r13+ push %r14+.cfi_push %r14+ push %r15+.cfi_push %r15+ sub \$64+24,%rsp+.cfi_adjust_cfa_offset 64+24+.Lctr32_body:++ mov $len,%rbp # reassign $len++ mov 0($key),%r12 # copy key and counter to stack+ mov 8($key),%r13+ mov 16($key),%r14+ mov 24($key),%r15+ mov 0($counter),%rax+ mov 8($counter),%rdx+ mov %r12,4*4(%rsp)+ mov %r13,4*6(%rsp)+ mov %r14,4*0(%rsp)+ mov %r15,4*2(%rsp)+ mov %rax,4*12(%rsp)+ mov %rdx,4*14(%rsp)+ jmp .Loop_outer++.align 32+.Loop_outer:+ mov \$0x61707865,@x[0] # 'expa'+ mov \$0x3320646e,@x[1] # 'nd 3'+ mov \$0x79622d32,@x[2] # '2-by'+ mov \$0x6b206574,@x[3] # 'te k'+ mov 4*4(%rsp),@x[4]+ mov 4*5(%rsp),@x[5]+ mov 4*6(%rsp),@x[6]+ mov 4*7(%rsp),@x[7]+ mov 4*12(%rsp),@x[12]+ mov 4*13(%rsp),@x[13]+ mov 4*14(%rsp),@x[14]+ mov %r15,4*10(%rsp) # "@x[10]:@x[11]"+ mov 4*15(%rsp),@x[15]++ mov %rbp,64+0(%rsp) # save len+ mov $inp,64+8(%rsp) # save inp+ mov 0(%rsp),@t[0] # "@x[8]"+ mov $out,64+16(%rsp) # save out+ mov 4(%rsp),@t[1] # "@x[9]"+ mov \$10,%ebp+ jmp .Loop++.align 32+.Loop:+___+ foreach (&ROUND (0, 4, 8,12)) { eval; }+ foreach (&ROUND (0, 5,10,15)) { eval; }+ &dec ("%ebp");+ &jnz (".Loop");++$code.=<<___;+ add 4*0(%rsp),@t[0] # modulo-scheduled+ add 4*1(%rsp),@t[1]+ mov 64(%rsp),%rbp # load len+ mov @t[0],4*8(%rsp)+ mov 64+8(%rsp),$inp # load inp+ mov @t[1],4*9(%rsp)+ mov 64+16(%rsp),$out # load out++ add \$0x61707865,@x[0] # 'expa'+ add \$0x3320646e,@x[1] # 'nd 3'+ add \$0x79622d32,@x[2] # '2-by'+ add \$0x6b206574,@x[3] # 'te k'+ add 4*4(%rsp),@x[4]+ add 4*5(%rsp),@x[5]+ add 4*6(%rsp),@x[6]+ add 4*7(%rsp),@x[7]+ add 4*12(%rsp),@x[12]+ add 4*13(%rsp),@x[13]+ add 4*14(%rsp),@x[14]+ add 4*15(%rsp),@x[15]++ cmp \$64,%rbp+ jb .Ltail++ xor 4*0($inp),@x[0] # xor with input+ xor 4*1($inp),@x[1]+ xor 4*2($inp),@x[2]+ xor 4*3($inp),@x[3]+ mov @x[0],4*0($out) # write output+ mov 4*8(%rsp),@x[0] # load @x[8]-@x[11]+ mov @x[1],4*1($out)+ mov 4*9(%rsp),@x[1]+ mov @x[2],4*2($out)+ mov 4*10(%rsp),@x[2]+ mov @x[3],4*3($out)+ mov 4*11(%rsp),@x[3]+ xor 4*4($inp),@x[4]+ add 4*2(%rsp),@x[2]+ xor 4*5($inp),@x[5]+ add 4*3(%rsp),@x[3]+ xor 4*6($inp),@x[6]+ xor 4*7($inp),@x[7]+ xor 4*8($inp),@x[0]+ xor 4*9($inp),@x[1]+ xor 4*10($inp),@x[2]+ xor 4*11($inp),@x[3]+ xor 4*12($inp),@x[12]+ xor 4*13($inp),@x[13]+ xor 4*14($inp),@x[14]+ xor 4*15($inp),@x[15]+ lea 4*16($inp),$inp # inp+=64++ addl \$1,4*12(%rsp) # increment counter++ mov @x[4],4*4($out)+ mov @x[5],4*5($out)+ mov @x[6],4*6($out)+ mov @x[7],4*7($out)+ mov @x[0],4*8($out)+ mov @x[1],4*9($out)+ mov @x[2],4*10($out)+ mov @x[3],4*11($out)+ mov @x[12],4*12($out)+ mov @x[13],4*13($out)+ mov @x[14],4*14($out)+ mov @x[15],4*15($out)+ lea 4*16($out),$out # out+=64+ mov 4*2(%rsp),%r15++ sub \$64,%rbp+ jnz .Loop_outer++ jmp .Ldone++.align 16+.Ltail:+ mov @x[0],4*0(%rsp)+ mov 4*2(%rsp),@x[0]+ mov @x[1],4*1(%rsp)+ mov 4*3(%rsp),@x[1]+ mov @x[2],4*2(%rsp)+ add 4*10(%rsp),@x[0]+ mov @x[3],4*3(%rsp)+ add 4*11(%rsp),@x[1]+ mov @x[4],4*4(%rsp)+ mov @x[5],4*5(%rsp)+ mov @x[6],4*6(%rsp)+ mov @x[7],4*7(%rsp)+ mov @x[0],4*10(%rsp)+ mov @x[1],4*11(%rsp)+ xor %rbx,%rbx+ mov @x[12],4*12(%rsp)+ mov @x[13],4*13(%rsp)+ mov @x[14],4*14(%rsp)+ mov @x[15],4*15(%rsp)++.Loop_tail:+ movzb ($inp,%rbx),%eax+ movzb (%rsp,%rbx),%edx+ lea 1(%rbx),%rbx+ xor %edx,%eax+ mov %al,-1($out,%rbx)+ dec %rbp+ jnz .Loop_tail++.Ldone:+ lea 64+24+48(%rsp),%rsi+.cfi_def_cfa %rsi,8+ mov -48(%rsi),%r15+.cfi_restore %r15+ mov -40(%rsi),%r14+.cfi_restore %r14+ mov -32(%rsi),%r13+.cfi_restore %r13+ mov -24(%rsi),%r12+.cfi_restore %r12+ mov -16(%rsi),%rbp+.cfi_restore %rbp+ mov -8(%rsi),%rbx+.cfi_restore %rbx+ lea (%rsi),%rsp+.cfi_def_cfa_register %rsp+.Lno_data:+ ret+.cfi_endproc+.size ChaCha20_ctr32,.-ChaCha20_ctr32+___++########################################################################+# SSSE3 code path that handles shorter lengths+{+my ($a,$b,$c,$d,$t,$t1,$rot16,$rot24)=map("%xmm$_",(0..7));++sub SSSE3ROUND { # critical path is 20 "SIMD ticks" per round+ &paddd ($a,$b);+ &pxor ($d,$a);+ &pshufb ($d,$rot16);++ &paddd ($c,$d);+ &pxor ($b,$c);+ &movdqa ($t,$b);+ &psrld ($b,20);+ &pslld ($t,12);+ &por ($b,$t);++ &paddd ($a,$b);+ &pxor ($d,$a);+ &pshufb ($d,$rot24);++ &paddd ($c,$d);+ &pxor ($b,$c);+ &movdqa ($t,$b);+ &psrld ($b,25);+ &pslld ($t,7);+ &por ($b,$t);+}++my $xframe = $win64 ? 32+8 : 8;++$code.=<<___ if ($flavour =~ /kernel/);+.globl ChaCha20_ssse3+___+$code.=<<___;+.type ChaCha20_ssse3,\@function,5+.align 32+ChaCha20_ssse3:+.cfi_startproc+.LChaCha20_ssse3:+ mov %rsp,%r10 # frame pointer+.cfi_def_cfa_register %r10+___+$code.=<<___ if ($avx && $flavour !~ /kernel/);+ test \$`1<<(43-32)`,%r9d+ jnz .LChaCha20_4xop # XOP is fastest even if we use 1/4+___+$code.=<<___;+ cmp \$128,$len # we might throw away some data,+ je .LChaCha20_128+ ja .LChaCha20_4x # but overall it won't be slower++.Ldo_sse3_after_all:+ sub \$64+$xframe,%rsp+ and \$-16,%rsp+___+$code.=<<___ if ($win64);+ movaps %xmm6,-0x28(%r10)+ movaps %xmm7,-0x18(%r10)+.Lssse3_body:+___+$code.=<<___;+ movdqa .Lsigma(%rip),$a+ movdqu ($key),$b+ movdqu 16($key),$c+ movdqu ($counter),$d+ movdqa .Lrot16(%rip),$rot16+ movdqa .Lrot24(%rip),$rot24++ movdqa $a,0x00(%rsp)+ movdqa $b,0x10(%rsp)+ movdqa $c,0x20(%rsp)+ movdqa $d,0x30(%rsp)+ mov \$10,$counter # reuse $counter+ jmp .Loop_ssse3++.align 32+.Loop_outer_ssse3:+ movdqa .Lone(%rip),$d+ movdqa 0x00(%rsp),$a+ movdqa 0x10(%rsp),$b+ movdqa 0x20(%rsp),$c+ paddd 0x30(%rsp),$d+ mov \$10,$counter+ movdqa $d,0x30(%rsp)+ jmp .Loop_ssse3++.align 32+.Loop_ssse3:+___+ &SSSE3ROUND();+ &pshufd ($c,$c,0b01001110);+ &pshufd ($b,$b,0b00111001);+ &pshufd ($d,$d,0b10010011);+ &nop ();++ &SSSE3ROUND();+ &pshufd ($c,$c,0b01001110);+ &pshufd ($b,$b,0b10010011);+ &pshufd ($d,$d,0b00111001);++ &dec ($counter);+ &jnz (".Loop_ssse3");++$code.=<<___;+ paddd 0x00(%rsp),$a+ paddd 0x10(%rsp),$b+ paddd 0x20(%rsp),$c+ paddd 0x30(%rsp),$d++ cmp \$64,$len+ jb .Ltail_ssse3++ movdqu 0x00($inp),$t+ movdqu 0x10($inp),$t1+ pxor $t,$a # xor with input+ movdqu 0x20($inp),$t+ pxor $t1,$b+ movdqu 0x30($inp),$t1+ lea 0x40($inp),$inp # inp+=64+ pxor $t,$c+ pxor $t1,$d++ movdqu $a,0x00($out) # write output+ movdqu $b,0x10($out)+ movdqu $c,0x20($out)+ movdqu $d,0x30($out)+ lea 0x40($out),$out # out+=64++ sub \$64,$len+ jnz .Loop_outer_ssse3++ jmp .Ldone_ssse3++.align 16+.Ltail_ssse3:+ movdqa $a,0x00(%rsp)+ movdqa $b,0x10(%rsp)+ movdqa $c,0x20(%rsp)+ movdqa $d,0x30(%rsp)+ xor $counter,$counter++.Loop_tail_ssse3:+ movzb ($inp,$counter),%eax+ movzb (%rsp,$counter),%ecx+ lea 1($counter),$counter+ xor %ecx,%eax+ mov %al,-1($out,$counter)+ dec $len+ jnz .Loop_tail_ssse3++.Ldone_ssse3:+___+$code.=<<___ if ($win64);+ movaps -0x28(%r10),%xmm6+ movaps -0x18(%r10),%xmm7+___+$code.=<<___;+ lea (%r10),%rsp+.cfi_def_cfa_register %rsp+.Lssse3_epilogue:+ ret+.cfi_endproc+.size ChaCha20_ssse3,.-ChaCha20_ssse3+___+}++########################################################################+# SSSE3 code path that handles 128-byte inputs+{+my ($a,$b,$c,$d,$t,$t1,$rot16,$rot24)=map("%xmm$_",(8,9,2..7));+my ($a1,$b1,$c1,$d1)=map("%xmm$_",(10,11,0,1));++sub SSSE3ROUND_2x {+ &paddd ($a,$b);+ &pxor ($d,$a);+ &paddd ($a1,$b1);+ &pxor ($d1,$a1);+ &pshufb ($d,$rot16);+ &pshufb($d1,$rot16);++ &paddd ($c,$d);+ &paddd ($c1,$d1);+ &pxor ($b,$c);+ &pxor ($b1,$c1);+ &movdqa ($t,$b);+ &psrld ($b,20);+ &movdqa($t1,$b1);+ &pslld ($t,12);+ &psrld ($b1,20);+ &por ($b,$t);+ &pslld ($t1,12);+ &por ($b1,$t1);++ &paddd ($a,$b);+ &pxor ($d,$a);+ &paddd ($a1,$b1);+ &pxor ($d1,$a1);+ &pshufb ($d,$rot24);+ &pshufb($d1,$rot24);++ &paddd ($c,$d);+ &paddd ($c1,$d1);+ &pxor ($b,$c);+ &pxor ($b1,$c1);+ &movdqa ($t,$b);+ &psrld ($b,25);+ &movdqa($t1,$b1);+ &pslld ($t,7);+ &psrld ($b1,25);+ &por ($b,$t);+ &pslld ($t1,7);+ &por ($b1,$t1);+}++my $xframe = $win64 ? 0x68 : 8;++$code.=<<___;+.type ChaCha20_128,\@function,5+.align 32+ChaCha20_128:+.cfi_startproc+.LChaCha20_128:+ mov %rsp,%r10 # frame pointer+.cfi_def_cfa_register %r10+ sub \$64+$xframe,%rsp+ and \$-16,%rsp+___+$code.=<<___ if ($win64);+ movaps %xmm6,-0x68(%r10)+ movaps %xmm7,-0x58(%r10)+ movaps %xmm8,-0x48(%r10)+ movaps %xmm9,-0x38(%r10)+ movaps %xmm10,-0x28(%r10)+ movaps %xmm11,-0x18(%r10)+.L128_body:+___+$code.=<<___;+ movdqa .Lsigma(%rip),$a+ movdqu ($key),$b+ movdqu 16($key),$c+ movdqu ($counter),$d+ movdqa .Lone(%rip),$d1+ movdqa .Lrot16(%rip),$rot16+ movdqa .Lrot24(%rip),$rot24++ movdqa $a,$a1+ movdqa $a,0x00(%rsp)+ movdqa $b,$b1+ movdqa $b,0x10(%rsp)+ movdqa $c,$c1+ movdqa $c,0x20(%rsp)+ paddd $d,$d1+ movdqa $d,0x30(%rsp)+ mov \$10,$counter # reuse $counter+ jmp .Loop_128++.align 32+.Loop_128:+___+ &SSSE3ROUND_2x();+ &pshufd ($c,$c,0b01001110);+ &pshufd ($b,$b,0b00111001);+ &pshufd ($d,$d,0b10010011);+ &pshufd ($c1,$c1,0b01001110);+ &pshufd ($b1,$b1,0b00111001);+ &pshufd ($d1,$d1,0b10010011);++ &SSSE3ROUND_2x();+ &pshufd ($c,$c,0b01001110);+ &pshufd ($b,$b,0b10010011);+ &pshufd ($d,$d,0b00111001);+ &pshufd ($c1,$c1,0b01001110);+ &pshufd ($b1,$b1,0b10010011);+ &pshufd ($d1,$d1,0b00111001);++ &dec ($counter);+ &jnz (".Loop_128");++$code.=<<___;+ paddd 0x00(%rsp),$a+ paddd 0x10(%rsp),$b+ paddd 0x20(%rsp),$c+ paddd 0x30(%rsp),$d+ paddd .Lone(%rip),$d1+ paddd 0x00(%rsp),$a1+ paddd 0x10(%rsp),$b1+ paddd 0x20(%rsp),$c1+ paddd 0x30(%rsp),$d1++ movdqu 0x00($inp),$t+ movdqu 0x10($inp),$t1+ pxor $t,$a # xor with input+ movdqu 0x20($inp),$t+ pxor $t1,$b+ movdqu 0x30($inp),$t1+ pxor $t,$c+ movdqu 0x40($inp),$t+ pxor $t1,$d+ movdqu 0x50($inp),$t1+ pxor $t,$a1+ movdqu 0x60($inp),$t+ pxor $t1,$b1+ movdqu 0x70($inp),$t1+ pxor $t,$c1+ pxor $t1,$d1++ movdqu $a,0x00($out) # write output+ movdqu $b,0x10($out)+ movdqu $c,0x20($out)+ movdqu $d,0x30($out)+ movdqu $a1,0x40($out)+ movdqu $b1,0x50($out)+ movdqu $c1,0x60($out)+ movdqu $d1,0x70($out)+___+$code.=<<___ if ($win64);+ movaps -0x68(%r10),%xmm6+ movaps -0x58(%r10),%xmm7+ movaps -0x48(%r10),%xmm8+ movaps -0x38(%r10),%xmm9+ movaps -0x28(%r10),%xmm10+ movaps -0x18(%r10),%xmm11+___+$code.=<<___;+ lea (%r10),%rsp+.cfi_def_cfa_register %rsp+.L128_epilogue:+ ret+.cfi_endproc+.size ChaCha20_128,.-ChaCha20_128+___+}++########################################################################+# SSSE3 code path that handles longer messages.+{+# assign variables to favor Atom front-end+my ($xd0,$xd1,$xd2,$xd3, $xt0,$xt1,$xt2,$xt3,+ $xa0,$xa1,$xa2,$xa3, $xb0,$xb1,$xb2,$xb3)=map("%xmm$_",(0..15));+my @xx=($xa0,$xa1,$xa2,$xa3, $xb0,$xb1,$xb2,$xb3,+ "%nox","%nox","%nox","%nox", $xd0,$xd1,$xd2,$xd3);++sub SSSE3_lane_ROUND {+my ($a0,$b0,$c0,$d0)=@_;+my ($a1,$b1,$c1,$d1)=map(($_&~3)+(($_+1)&3),($a0,$b0,$c0,$d0));+my ($a2,$b2,$c2,$d2)=map(($_&~3)+(($_+1)&3),($a1,$b1,$c1,$d1));+my ($a3,$b3,$c3,$d3)=map(($_&~3)+(($_+1)&3),($a2,$b2,$c2,$d2));+my ($xc,$xc_,$t0,$t1)=map("\"$_\"",$xt0,$xt1,$xt2,$xt3);+my @x=map("\"$_\"",@xx);++ # Consider order in which variables are addressed by their+ # index:+ #+ # a b c d+ #+ # 0 4 8 12 < even round+ # 1 5 9 13+ # 2 6 10 14+ # 3 7 11 15+ # 0 5 10 15 < odd round+ # 1 6 11 12+ # 2 7 8 13+ # 3 4 9 14+ #+ # 'a', 'b' and 'd's are permanently allocated in registers,+ # @x[0..7,12..15], while 'c's are maintained in memory. If+ # you observe 'c' column, you'll notice that pair of 'c's is+ # invariant between rounds. This means that we have to reload+ # them once per round, in the middle. This is why you'll see+ # bunch of 'c' stores and loads in the middle, but none in+ # the beginning or end.++ (+ "&paddd (@x[$a0],@x[$b0])", # Q1+ "&paddd (@x[$a1],@x[$b1])", # Q2+ "&pxor (@x[$d0],@x[$a0])",+ "&pxor (@x[$d1],@x[$a1])",+ "&pshufb (@x[$d0],$t1)",+ "&pshufb (@x[$d1],$t1)",++ "&paddd ($xc,@x[$d0])",+ "&paddd ($xc_,@x[$d1])",+ "&pxor (@x[$b0],$xc)",+ "&pxor (@x[$b1],$xc_)",+ "&movdqa ($t0,@x[$b0])",+ "&pslld (@x[$b0],12)",+ "&psrld ($t0,20)",+ "&movdqa ($t1,@x[$b1])",+ "&pslld (@x[$b1],12)",+ "&por (@x[$b0],$t0)",+ "&psrld ($t1,20)",+ "&movdqa ($t0,'(%r11)')", # .Lrot24(%rip)+ "&por (@x[$b1],$t1)",++ "&paddd (@x[$a0],@x[$b0])",+ "&paddd (@x[$a1],@x[$b1])",+ "&pxor (@x[$d0],@x[$a0])",+ "&pxor (@x[$d1],@x[$a1])",+ "&pshufb (@x[$d0],$t0)",+ "&pshufb (@x[$d1],$t0)",++ "&paddd ($xc,@x[$d0])",+ "&paddd ($xc_,@x[$d1])",+ "&pxor (@x[$b0],$xc)",+ "&pxor (@x[$b1],$xc_)",+ "&movdqa ($t1,@x[$b0])",+ "&pslld (@x[$b0],7)",+ "&psrld ($t1,25)",+ "&movdqa ($t0,@x[$b1])",+ "&pslld (@x[$b1],7)",+ "&por (@x[$b0],$t1)",+ "&psrld ($t0,25)",+ "&movdqa ($t1,'(%r9)')", # .Lrot16(%rip)+ "&por (@x[$b1],$t0)",++ "&movdqa (\"`16*($c0-8)`(%rsp)\",$xc)", # reload pair of 'c's+ "&movdqa (\"`16*($c1-8)`(%rsp)\",$xc_)",+ "&movdqa ($xc,\"`16*($c2-8)`(%rsp)\")",+ "&movdqa ($xc_,\"`16*($c3-8)`(%rsp)\")",++ "&paddd (@x[$a2],@x[$b2])", # Q3+ "&paddd (@x[$a3],@x[$b3])", # Q4+ "&pxor (@x[$d2],@x[$a2])",+ "&pxor (@x[$d3],@x[$a3])",+ "&pshufb (@x[$d2],$t1)",+ "&pshufb (@x[$d3],$t1)",++ "&paddd ($xc,@x[$d2])",+ "&paddd ($xc_,@x[$d3])",+ "&pxor (@x[$b2],$xc)",+ "&pxor (@x[$b3],$xc_)",+ "&movdqa ($t0,@x[$b2])",+ "&pslld (@x[$b2],12)",+ "&psrld ($t0,20)",+ "&movdqa ($t1,@x[$b3])",+ "&pslld (@x[$b3],12)",+ "&por (@x[$b2],$t0)",+ "&psrld ($t1,20)",+ "&movdqa ($t0,'(%r11)')", # .Lrot24(%rip)+ "&por (@x[$b3],$t1)",++ "&paddd (@x[$a2],@x[$b2])",+ "&paddd (@x[$a3],@x[$b3])",+ "&pxor (@x[$d2],@x[$a2])",+ "&pxor (@x[$d3],@x[$a3])",+ "&pshufb (@x[$d2],$t0)",+ "&pshufb (@x[$d3],$t0)",++ "&paddd ($xc,@x[$d2])",+ "&paddd ($xc_,@x[$d3])",+ "&pxor (@x[$b2],$xc)",+ "&pxor (@x[$b3],$xc_)",+ "&movdqa ($t1,@x[$b2])",+ "&pslld (@x[$b2],7)",+ "&psrld ($t1,25)",+ "&movdqa ($t0,@x[$b3])",+ "&pslld (@x[$b3],7)",+ "&por (@x[$b2],$t1)",+ "&psrld ($t0,25)",+ "&movdqa ($t1,'(%r9)')", # .Lrot16(%rip)+ "&por (@x[$b3],$t0)"+ );+}++my $xframe = $win64 ? 0xa8 : 8;++$code.=<<___;+.type ChaCha20_4x,\@function,5+.align 32+ChaCha20_4x:+.cfi_startproc+.LChaCha20_4x:+ mov %rsp,%r10 # frame pointer+.cfi_def_cfa_register %r10+ mov %r9,%r11+___+$code.=<<___ if ($avx>1 && $flavour !~ /kernel/);+ shr \$32,%r9 # OPENSSL_ia32cap_P+8+ test \$`1<<5`,%r9 # test AVX2+ jnz .LChaCha20_8x+___+$code.=<<___;+ cmp \$192,$len+ ja .Lproceed4x++ and \$`1<<26|1<<22`,%r11 # isolate XSAVE+MOVBE+ cmp \$`1<<22`,%r11 # check for MOVBE without XSAVE+ je .Ldo_sse3_after_all # to detect Atom++.Lproceed4x:+ sub \$0x140+$xframe,%rsp+ and \$-16,%rsp+___+ ################ stack layout+ # +0x00 SIMD equivalent of @x[8-12]+ # ...+ # +0x40 constant copy of key[0-2] smashed by lanes+ # ...+ # +0x100 SIMD counters (with nonce smashed by lanes)+ # ...+ # +0x140+$code.=<<___ if ($win64);+ movaps %xmm6,-0xa8(%r10)+ movaps %xmm7,-0x98(%r10)+ movaps %xmm8,-0x88(%r10)+ movaps %xmm9,-0x78(%r10)+ movaps %xmm10,-0x68(%r10)+ movaps %xmm11,-0x58(%r10)+ movaps %xmm12,-0x48(%r10)+ movaps %xmm13,-0x38(%r10)+ movaps %xmm14,-0x28(%r10)+ movaps %xmm15,-0x18(%r10)+.L4x_body:+___+$code.=<<___;+ movdqa .Lsigma(%rip),$xa3 # key[0]+ movdqu ($key),$xb3 # key[1]+ movdqu 16($key),$xt3 # key[2]+ movdqu ($counter),$xd3 # key[3]+ lea 0x100(%rsp),%rcx # size optimization+ lea .Lrot16(%rip),%r9+ lea .Lrot24(%rip),%r11++ pshufd \$0x00,$xa3,$xa0 # smash key by lanes...+ pshufd \$0x55,$xa3,$xa1+ movdqa $xa0,0x40(%rsp) # ... and offload+ pshufd \$0xaa,$xa3,$xa2+ movdqa $xa1,0x50(%rsp)+ pshufd \$0xff,$xa3,$xa3+ movdqa $xa2,0x60(%rsp)+ movdqa $xa3,0x70(%rsp)++ pshufd \$0x00,$xb3,$xb0+ pshufd \$0x55,$xb3,$xb1+ movdqa $xb0,0x80-0x100(%rcx)+ pshufd \$0xaa,$xb3,$xb2+ movdqa $xb1,0x90-0x100(%rcx)+ pshufd \$0xff,$xb3,$xb3+ movdqa $xb2,0xa0-0x100(%rcx)+ movdqa $xb3,0xb0-0x100(%rcx)++ pshufd \$0x00,$xt3,$xt0 # "$xc0"+ pshufd \$0x55,$xt3,$xt1 # "$xc1"+ movdqa $xt0,0xc0-0x100(%rcx)+ pshufd \$0xaa,$xt3,$xt2 # "$xc2"+ movdqa $xt1,0xd0-0x100(%rcx)+ pshufd \$0xff,$xt3,$xt3 # "$xc3"+ movdqa $xt2,0xe0-0x100(%rcx)+ movdqa $xt3,0xf0-0x100(%rcx)++ pshufd \$0x00,$xd3,$xd0+ pshufd \$0x55,$xd3,$xd1+ paddd .Linc(%rip),$xd0 # don't save counters yet+ pshufd \$0xaa,$xd3,$xd2+ movdqa $xd1,0x110-0x100(%rcx)+ pshufd \$0xff,$xd3,$xd3+ movdqa $xd2,0x120-0x100(%rcx)+ movdqa $xd3,0x130-0x100(%rcx)++ jmp .Loop_enter4x++.align 32+.Loop_outer4x:+ movdqa 0x40(%rsp),$xa0 # re-load smashed key+ movdqa 0x50(%rsp),$xa1+ movdqa 0x60(%rsp),$xa2+ movdqa 0x70(%rsp),$xa3+ movdqa 0x80-0x100(%rcx),$xb0+ movdqa 0x90-0x100(%rcx),$xb1+ movdqa 0xa0-0x100(%rcx),$xb2+ movdqa 0xb0-0x100(%rcx),$xb3+ movdqa 0xc0-0x100(%rcx),$xt0 # "$xc0"+ movdqa 0xd0-0x100(%rcx),$xt1 # "$xc1"+ movdqa 0xe0-0x100(%rcx),$xt2 # "$xc2"+ movdqa 0xf0-0x100(%rcx),$xt3 # "$xc3"+ movdqa 0x100-0x100(%rcx),$xd0+ movdqa 0x110-0x100(%rcx),$xd1+ movdqa 0x120-0x100(%rcx),$xd2+ movdqa 0x130-0x100(%rcx),$xd3+ paddd .Lfour(%rip),$xd0 # next SIMD counters++.Loop_enter4x:+ movdqa $xt2,0x20(%rsp) # SIMD equivalent of "@x[10]"+ movdqa $xt3,0x30(%rsp) # SIMD equivalent of "@x[11]"+ movdqa (%r9),$xt3 # .Lrot16(%rip)+ mov \$10,%eax+ movdqa $xd0,0x100-0x100(%rcx) # save SIMD counters+ jmp .Loop4x++.align 32+.Loop4x:+___+ foreach (&SSSE3_lane_ROUND(0, 4, 8,12)) { eval; }+ foreach (&SSSE3_lane_ROUND(0, 5,10,15)) { eval; }+$code.=<<___;+ dec %eax+ jnz .Loop4x++ paddd 0x40(%rsp),$xa0 # accumulate key material+ paddd 0x50(%rsp),$xa1+ paddd 0x60(%rsp),$xa2+ paddd 0x70(%rsp),$xa3++ movdqa $xa0,$xt2 # "de-interlace" data+ punpckldq $xa1,$xa0+ movdqa $xa2,$xt3+ punpckldq $xa3,$xa2+ punpckhdq $xa1,$xt2+ punpckhdq $xa3,$xt3+ movdqa $xa0,$xa1+ punpcklqdq $xa2,$xa0 # "a0"+ movdqa $xt2,$xa3+ punpcklqdq $xt3,$xt2 # "a2"+ punpckhqdq $xa2,$xa1 # "a1"+ punpckhqdq $xt3,$xa3 # "a3"+___+ ($xa2,$xt2)=($xt2,$xa2);+$code.=<<___;+ paddd 0x80-0x100(%rcx),$xb0+ paddd 0x90-0x100(%rcx),$xb1+ paddd 0xa0-0x100(%rcx),$xb2+ paddd 0xb0-0x100(%rcx),$xb3++ movdqa $xa0,0x00(%rsp) # offload $xaN+ movdqa $xa1,0x10(%rsp)+ movdqa 0x20(%rsp),$xa0 # "xc2"+ movdqa 0x30(%rsp),$xa1 # "xc3"++ movdqa $xb0,$xt2+ punpckldq $xb1,$xb0+ movdqa $xb2,$xt3+ punpckldq $xb3,$xb2+ punpckhdq $xb1,$xt2+ punpckhdq $xb3,$xt3+ movdqa $xb0,$xb1+ punpcklqdq $xb2,$xb0 # "b0"+ movdqa $xt2,$xb3+ punpcklqdq $xt3,$xt2 # "b2"+ punpckhqdq $xb2,$xb1 # "b1"+ punpckhqdq $xt3,$xb3 # "b3"+___+ ($xb2,$xt2)=($xt2,$xb2);+ my ($xc0,$xc1,$xc2,$xc3)=($xt0,$xt1,$xa0,$xa1);+$code.=<<___;+ paddd 0xc0-0x100(%rcx),$xc0+ paddd 0xd0-0x100(%rcx),$xc1+ paddd 0xe0-0x100(%rcx),$xc2+ paddd 0xf0-0x100(%rcx),$xc3++ movdqa $xa2,0x20(%rsp) # keep offloading $xaN+ movdqa $xa3,0x30(%rsp)++ movdqa $xc0,$xt2+ punpckldq $xc1,$xc0+ movdqa $xc2,$xt3+ punpckldq $xc3,$xc2+ punpckhdq $xc1,$xt2+ punpckhdq $xc3,$xt3+ movdqa $xc0,$xc1+ punpcklqdq $xc2,$xc0 # "c0"+ movdqa $xt2,$xc3+ punpcklqdq $xt3,$xt2 # "c2"+ punpckhqdq $xc2,$xc1 # "c1"+ punpckhqdq $xt3,$xc3 # "c3"+___+ ($xc2,$xt2)=($xt2,$xc2);+ ($xt0,$xt1)=($xa2,$xa3); # use $xaN as temporary+$code.=<<___;+ paddd 0x100-0x100(%rcx),$xd0+ paddd 0x110-0x100(%rcx),$xd1+ paddd 0x120-0x100(%rcx),$xd2+ paddd 0x130-0x100(%rcx),$xd3++ movdqa $xd0,$xt2+ punpckldq $xd1,$xd0+ movdqa $xd2,$xt3+ punpckldq $xd3,$xd2+ punpckhdq $xd1,$xt2+ punpckhdq $xd3,$xt3+ movdqa $xd0,$xd1+ punpcklqdq $xd2,$xd0 # "d0"+ movdqa $xt2,$xd3+ punpcklqdq $xt3,$xt2 # "d2"+ punpckhqdq $xd2,$xd1 # "d1"+ punpckhqdq $xt3,$xd3 # "d3"+___+ ($xd2,$xt2)=($xt2,$xd2);+$code.=<<___;+ cmp \$64*4,$len+ jb .Ltail4x++ movdqu 0x00($inp),$xt0 # xor with input+ movdqu 0x10($inp),$xt1+ movdqu 0x20($inp),$xt2+ movdqu 0x30($inp),$xt3+ pxor 0x00(%rsp),$xt0 # $xaN is offloaded, remember?+ pxor $xb0,$xt1+ pxor $xc0,$xt2+ pxor $xd0,$xt3++ movdqu $xt0,0x00($out)+ movdqu 0x40($inp),$xt0+ movdqu $xt1,0x10($out)+ movdqu 0x50($inp),$xt1+ movdqu $xt2,0x20($out)+ movdqu 0x60($inp),$xt2+ movdqu $xt3,0x30($out)+ movdqu 0x70($inp),$xt3+ lea 0x80($inp),$inp # size optimization+ pxor 0x10(%rsp),$xt0+ pxor $xb1,$xt1+ pxor $xc1,$xt2+ pxor $xd1,$xt3++ movdqu $xt0,0x40($out)+ movdqu 0x00($inp),$xt0+ movdqu $xt1,0x50($out)+ movdqu 0x10($inp),$xt1+ movdqu $xt2,0x60($out)+ movdqu 0x20($inp),$xt2+ movdqu $xt3,0x70($out)+ lea 0x80($out),$out # size optimization+ movdqu 0x30($inp),$xt3+ pxor 0x20(%rsp),$xt0+ pxor $xb2,$xt1+ pxor $xc2,$xt2+ pxor $xd2,$xt3++ movdqu $xt0,0x00($out)+ movdqu 0x40($inp),$xt0+ movdqu $xt1,0x10($out)+ movdqu 0x50($inp),$xt1+ movdqu $xt2,0x20($out)+ movdqu 0x60($inp),$xt2+ movdqu $xt3,0x30($out)+ movdqu 0x70($inp),$xt3+ lea 0x80($inp),$inp # inp+=64*4+ pxor 0x30(%rsp),$xt0+ pxor $xb3,$xt1+ pxor $xc3,$xt2+ pxor $xd3,$xt3+ movdqu $xt0,0x40($out)+ movdqu $xt1,0x50($out)+ movdqu $xt2,0x60($out)+ movdqu $xt3,0x70($out)+ lea 0x80($out),$out # out+=64*4++ sub \$64*4,$len+ jnz .Loop_outer4x++ jmp .Ldone4x++.Ltail4x:+ cmp \$192,$len+ jae .L192_or_more4x+ cmp \$128,$len+ jae .L128_or_more4x+ cmp \$64,$len+ jae .L64_or_more4x++ #movdqa 0x00(%rsp),$xt0 # $xaN is offloaded, remember?+ xor %r9,%r9+ #movdqa $xt0,0x00(%rsp)+ movdqa $xb0,0x10(%rsp)+ movdqa $xc0,0x20(%rsp)+ movdqa $xd0,0x30(%rsp)+ jmp .Loop_tail4x++.align 32+.L64_or_more4x:+ movdqu 0x00($inp),$xt0 # xor with input+ movdqu 0x10($inp),$xt1+ movdqu 0x20($inp),$xt2+ movdqu 0x30($inp),$xt3+ pxor 0x00(%rsp),$xt0 # $xaxN is offloaded, remember?+ pxor $xb0,$xt1+ pxor $xc0,$xt2+ pxor $xd0,$xt3+ movdqu $xt0,0x00($out)+ movdqu $xt1,0x10($out)+ movdqu $xt2,0x20($out)+ movdqu $xt3,0x30($out)+ je .Ldone4x++ movdqa 0x10(%rsp),$xt0 # $xaN is offloaded, remember?+ lea 0x40($inp),$inp # inp+=64*1+ xor %r9,%r9+ movdqa $xt0,0x00(%rsp)+ movdqa $xb1,0x10(%rsp)+ lea 0x40($out),$out # out+=64*1+ movdqa $xc1,0x20(%rsp)+ sub \$64,$len # len-=64*1+ movdqa $xd1,0x30(%rsp)+ jmp .Loop_tail4x++.align 32+.L128_or_more4x:+ movdqu 0x00($inp),$xt0 # xor with input+ movdqu 0x10($inp),$xt1+ movdqu 0x20($inp),$xt2+ movdqu 0x30($inp),$xt3+ pxor 0x00(%rsp),$xt0 # $xaN is offloaded, remember?+ pxor $xb0,$xt1+ pxor $xc0,$xt2+ pxor $xd0,$xt3++ movdqu $xt0,0x00($out)+ movdqu 0x40($inp),$xt0+ movdqu $xt1,0x10($out)+ movdqu 0x50($inp),$xt1+ movdqu $xt2,0x20($out)+ movdqu 0x60($inp),$xt2+ movdqu $xt3,0x30($out)+ movdqu 0x70($inp),$xt3+ pxor 0x10(%rsp),$xt0+ pxor $xb1,$xt1+ pxor $xc1,$xt2+ pxor $xd1,$xt3+ movdqu $xt0,0x40($out)+ movdqu $xt1,0x50($out)+ movdqu $xt2,0x60($out)+ movdqu $xt3,0x70($out)+ je .Ldone4x++ movdqa 0x20(%rsp),$xt0 # $xaN is offloaded, remember?+ lea 0x80($inp),$inp # inp+=64*2+ xor %r9,%r9+ movdqa $xt0,0x00(%rsp)+ movdqa $xb2,0x10(%rsp)+ lea 0x80($out),$out # out+=64*2+ movdqa $xc2,0x20(%rsp)+ sub \$128,$len # len-=64*2+ movdqa $xd2,0x30(%rsp)+ jmp .Loop_tail4x++.align 32+.L192_or_more4x:+ movdqu 0x00($inp),$xt0 # xor with input+ movdqu 0x10($inp),$xt1+ movdqu 0x20($inp),$xt2+ movdqu 0x30($inp),$xt3+ pxor 0x00(%rsp),$xt0 # $xaN is offloaded, remember?+ pxor $xb0,$xt1+ pxor $xc0,$xt2+ pxor $xd0,$xt3++ movdqu $xt0,0x00($out)+ movdqu 0x40($inp),$xt0+ movdqu $xt1,0x10($out)+ movdqu 0x50($inp),$xt1+ movdqu $xt2,0x20($out)+ movdqu 0x60($inp),$xt2+ movdqu $xt3,0x30($out)+ movdqu 0x70($inp),$xt3+ lea 0x80($inp),$inp # size optimization+ pxor 0x10(%rsp),$xt0+ pxor $xb1,$xt1+ pxor $xc1,$xt2+ pxor $xd1,$xt3++ movdqu $xt0,0x40($out)+ movdqu 0x00($inp),$xt0+ movdqu $xt1,0x50($out)+ movdqu 0x10($inp),$xt1+ movdqu $xt2,0x60($out)+ movdqu 0x20($inp),$xt2+ movdqu $xt3,0x70($out)+ lea 0x80($out),$out # size optimization+ movdqu 0x30($inp),$xt3+ pxor 0x20(%rsp),$xt0+ pxor $xb2,$xt1+ pxor $xc2,$xt2+ pxor $xd2,$xt3+ movdqu $xt0,0x00($out)+ movdqu $xt1,0x10($out)+ movdqu $xt2,0x20($out)+ movdqu $xt3,0x30($out)+ je .Ldone4x++ movdqa 0x30(%rsp),$xt0 # $xaN is offloaded, remember?+ lea 0x40($inp),$inp # inp+=64*3+ xor %r9,%r9+ movdqa $xt0,0x00(%rsp)+ movdqa $xb3,0x10(%rsp)+ lea 0x40($out),$out # out+=64*3+ movdqa $xc3,0x20(%rsp)+ sub \$192,$len # len-=64*3+ movdqa $xd3,0x30(%rsp)++.Loop_tail4x:+ movzb ($inp,%r9),%eax+ movzb (%rsp,%r9),%ecx+ lea 1(%r9),%r9+ xor %ecx,%eax+ mov %al,-1($out,%r9)+ dec $len+ jnz .Loop_tail4x++.Ldone4x:+___+$code.=<<___ if ($win64);+ movaps -0xa8(%r10),%xmm6+ movaps -0x98(%r10),%xmm7+ movaps -0x88(%r10),%xmm8+ movaps -0x78(%r10),%xmm9+ movaps -0x68(%r10),%xmm10+ movaps -0x58(%r10),%xmm11+ movaps -0x48(%r10),%xmm12+ movaps -0x38(%r10),%xmm13+ movaps -0x28(%r10),%xmm14+ movaps -0x18(%r10),%xmm15+___+$code.=<<___;+ lea (%r10),%rsp+.cfi_def_cfa_register %rsp+.L4x_epilogue:+ ret+.cfi_endproc+.size ChaCha20_4x,.-ChaCha20_4x+___+}++########################################################################+# XOP code path that handles all lengths.+if ($avx) {+# There is some "anomaly" observed depending on instructions' size or+# alignment. If you look closely at below code you'll notice that+# sometimes argument order varies. The order affects instruction+# encoding by making it larger, and such fiddling gives 5% performance+# improvement. This is on FX-4100...++my ($xb0,$xb1,$xb2,$xb3, $xd0,$xd1,$xd2,$xd3,+ $xa0,$xa1,$xa2,$xa3, $xt0,$xt1,$xt2,$xt3)=map("%xmm$_",(0..15));+my @xx=($xa0,$xa1,$xa2,$xa3, $xb0,$xb1,$xb2,$xb3,+ $xt0,$xt1,$xt2,$xt3, $xd0,$xd1,$xd2,$xd3);++sub XOP_lane_ROUND {+my ($a0,$b0,$c0,$d0)=@_;+my ($a1,$b1,$c1,$d1)=map(($_&~3)+(($_+1)&3),($a0,$b0,$c0,$d0));+my ($a2,$b2,$c2,$d2)=map(($_&~3)+(($_+1)&3),($a1,$b1,$c1,$d1));+my ($a3,$b3,$c3,$d3)=map(($_&~3)+(($_+1)&3),($a2,$b2,$c2,$d2));+my @x=map("\"$_\"",@xx);++ (+ "&vpaddd (@x[$a0],@x[$a0],@x[$b0])", # Q1+ "&vpaddd (@x[$a1],@x[$a1],@x[$b1])", # Q2+ "&vpaddd (@x[$a2],@x[$a2],@x[$b2])", # Q3+ "&vpaddd (@x[$a3],@x[$a3],@x[$b3])", # Q4+ "&vpxor (@x[$d0],@x[$a0],@x[$d0])",+ "&vpxor (@x[$d1],@x[$a1],@x[$d1])",+ "&vpxor (@x[$d2],@x[$a2],@x[$d2])",+ "&vpxor (@x[$d3],@x[$a3],@x[$d3])",+ "&vprotd (@x[$d0],@x[$d0],16)",+ "&vprotd (@x[$d1],@x[$d1],16)",+ "&vprotd (@x[$d2],@x[$d2],16)",+ "&vprotd (@x[$d3],@x[$d3],16)",++ "&vpaddd (@x[$c0],@x[$c0],@x[$d0])",+ "&vpaddd (@x[$c1],@x[$c1],@x[$d1])",+ "&vpaddd (@x[$c2],@x[$c2],@x[$d2])",+ "&vpaddd (@x[$c3],@x[$c3],@x[$d3])",+ "&vpxor (@x[$b0],@x[$c0],@x[$b0])",+ "&vpxor (@x[$b1],@x[$c1],@x[$b1])",+ "&vpxor (@x[$b2],@x[$b2],@x[$c2])", # flip+ "&vpxor (@x[$b3],@x[$b3],@x[$c3])", # flip+ "&vprotd (@x[$b0],@x[$b0],12)",+ "&vprotd (@x[$b1],@x[$b1],12)",+ "&vprotd (@x[$b2],@x[$b2],12)",+ "&vprotd (@x[$b3],@x[$b3],12)",++ "&vpaddd (@x[$a0],@x[$b0],@x[$a0])", # flip+ "&vpaddd (@x[$a1],@x[$b1],@x[$a1])", # flip+ "&vpaddd (@x[$a2],@x[$a2],@x[$b2])",+ "&vpaddd (@x[$a3],@x[$a3],@x[$b3])",+ "&vpxor (@x[$d0],@x[$a0],@x[$d0])",+ "&vpxor (@x[$d1],@x[$a1],@x[$d1])",+ "&vpxor (@x[$d2],@x[$a2],@x[$d2])",+ "&vpxor (@x[$d3],@x[$a3],@x[$d3])",+ "&vprotd (@x[$d0],@x[$d0],8)",+ "&vprotd (@x[$d1],@x[$d1],8)",+ "&vprotd (@x[$d2],@x[$d2],8)",+ "&vprotd (@x[$d3],@x[$d3],8)",++ "&vpaddd (@x[$c0],@x[$c0],@x[$d0])",+ "&vpaddd (@x[$c1],@x[$c1],@x[$d1])",+ "&vpaddd (@x[$c2],@x[$c2],@x[$d2])",+ "&vpaddd (@x[$c3],@x[$c3],@x[$d3])",+ "&vpxor (@x[$b0],@x[$c0],@x[$b0])",+ "&vpxor (@x[$b1],@x[$c1],@x[$b1])",+ "&vpxor (@x[$b2],@x[$b2],@x[$c2])", # flip+ "&vpxor (@x[$b3],@x[$b3],@x[$c3])", # flip+ "&vprotd (@x[$b0],@x[$b0],7)",+ "&vprotd (@x[$b1],@x[$b1],7)",+ "&vprotd (@x[$b2],@x[$b2],7)",+ "&vprotd (@x[$b3],@x[$b3],7)"+ );+}++my $xframe = $win64 ? 0xa8 : 8;++$code.=<<___ if ($flavour =~ /kernel/);+.globl ChaCha20_4xop+___+$code.=<<___;+.type ChaCha20_4xop,\@function,5+.align 32+ChaCha20_4xop:+.cfi_startproc+.LChaCha20_4xop:+ mov %rsp,%r10 # frame pointer+.cfi_def_cfa_register %r10+ sub \$0x140+$xframe,%rsp+ and \$-16,%rsp+___+ ################ stack layout+ # +0x00 SIMD equivalent of @x[8-12]+ # ...+ # +0x40 constant copy of key[0-2] smashed by lanes+ # ...+ # +0x100 SIMD counters (with nonce smashed by lanes)+ # ...+ # +0x140+$code.=<<___ if ($win64);+ movaps %xmm6,-0xa8(%r10)+ movaps %xmm7,-0x98(%r10)+ movaps %xmm8,-0x88(%r10)+ movaps %xmm9,-0x78(%r10)+ movaps %xmm10,-0x68(%r10)+ movaps %xmm11,-0x58(%r10)+ movaps %xmm12,-0x48(%r10)+ movaps %xmm13,-0x38(%r10)+ movaps %xmm14,-0x28(%r10)+ movaps %xmm15,-0x18(%r10)+.L4xop_body:+___+$code.=<<___;+ vzeroupper++ vmovdqa .Lsigma(%rip),$xa3 # key[0]+ vmovdqu ($key),$xb3 # key[1]+ vmovdqu 16($key),$xt3 # key[2]+ vmovdqu ($counter),$xd3 # key[3]+ lea 0x100(%rsp),%rcx # size optimization++ vpshufd \$0x00,$xa3,$xa0 # smash key by lanes...+ vpshufd \$0x55,$xa3,$xa1+ vmovdqa $xa0,0x40(%rsp) # ... and offload+ vpshufd \$0xaa,$xa3,$xa2+ vmovdqa $xa1,0x50(%rsp)+ vpshufd \$0xff,$xa3,$xa3+ vmovdqa $xa2,0x60(%rsp)+ vmovdqa $xa3,0x70(%rsp)++ vpshufd \$0x00,$xb3,$xb0+ vpshufd \$0x55,$xb3,$xb1+ vmovdqa $xb0,0x80-0x100(%rcx)+ vpshufd \$0xaa,$xb3,$xb2+ vmovdqa $xb1,0x90-0x100(%rcx)+ vpshufd \$0xff,$xb3,$xb3+ vmovdqa $xb2,0xa0-0x100(%rcx)+ vmovdqa $xb3,0xb0-0x100(%rcx)++ vpshufd \$0x00,$xt3,$xt0 # "$xc0"+ vpshufd \$0x55,$xt3,$xt1 # "$xc1"+ vmovdqa $xt0,0xc0-0x100(%rcx)+ vpshufd \$0xaa,$xt3,$xt2 # "$xc2"+ vmovdqa $xt1,0xd0-0x100(%rcx)+ vpshufd \$0xff,$xt3,$xt3 # "$xc3"+ vmovdqa $xt2,0xe0-0x100(%rcx)+ vmovdqa $xt3,0xf0-0x100(%rcx)++ vpshufd \$0x00,$xd3,$xd0+ vpshufd \$0x55,$xd3,$xd1+ vpaddd .Linc(%rip),$xd0,$xd0 # don't save counters yet+ vpshufd \$0xaa,$xd3,$xd2+ vmovdqa $xd1,0x110-0x100(%rcx)+ vpshufd \$0xff,$xd3,$xd3+ vmovdqa $xd2,0x120-0x100(%rcx)+ vmovdqa $xd3,0x130-0x100(%rcx)++ jmp .Loop_enter4xop++.align 32+.Loop_outer4xop:+ vmovdqa 0x40(%rsp),$xa0 # re-load smashed key+ vmovdqa 0x50(%rsp),$xa1+ vmovdqa 0x60(%rsp),$xa2+ vmovdqa 0x70(%rsp),$xa3+ vmovdqa 0x80-0x100(%rcx),$xb0+ vmovdqa 0x90-0x100(%rcx),$xb1+ vmovdqa 0xa0-0x100(%rcx),$xb2+ vmovdqa 0xb0-0x100(%rcx),$xb3+ vmovdqa 0xc0-0x100(%rcx),$xt0 # "$xc0"+ vmovdqa 0xd0-0x100(%rcx),$xt1 # "$xc1"+ vmovdqa 0xe0-0x100(%rcx),$xt2 # "$xc2"+ vmovdqa 0xf0-0x100(%rcx),$xt3 # "$xc3"+ vmovdqa 0x100-0x100(%rcx),$xd0+ vmovdqa 0x110-0x100(%rcx),$xd1+ vmovdqa 0x120-0x100(%rcx),$xd2+ vmovdqa 0x130-0x100(%rcx),$xd3+ vpaddd .Lfour(%rip),$xd0,$xd0 # next SIMD counters++.Loop_enter4xop:+ mov \$10,%eax+ vmovdqa $xd0,0x100-0x100(%rcx) # save SIMD counters+ jmp .Loop4xop++.align 32+.Loop4xop:+___+ foreach (&XOP_lane_ROUND(0, 4, 8,12)) { eval; }+ foreach (&XOP_lane_ROUND(0, 5,10,15)) { eval; }+$code.=<<___;+ dec %eax+ jnz .Loop4xop++ vpaddd 0x40(%rsp),$xa0,$xa0 # accumulate key material+ vpaddd 0x50(%rsp),$xa1,$xa1+ vpaddd 0x60(%rsp),$xa2,$xa2+ vpaddd 0x70(%rsp),$xa3,$xa3++ vmovdqa $xt2,0x20(%rsp) # offload $xc2,3+ vmovdqa $xt3,0x30(%rsp)++ vpunpckldq $xa1,$xa0,$xt2 # "de-interlace" data+ vpunpckldq $xa3,$xa2,$xt3+ vpunpckhdq $xa1,$xa0,$xa0+ vpunpckhdq $xa3,$xa2,$xa2+ vpunpcklqdq $xt3,$xt2,$xa1 # "a0"+ vpunpckhqdq $xt3,$xt2,$xt2 # "a1"+ vpunpcklqdq $xa2,$xa0,$xa3 # "a2"+ vpunpckhqdq $xa2,$xa0,$xa0 # "a3"+___+ ($xa0,$xa1,$xa2,$xa3,$xt2)=($xa1,$xt2,$xa3,$xa0,$xa2);+$code.=<<___;+ vpaddd 0x80-0x100(%rcx),$xb0,$xb0+ vpaddd 0x90-0x100(%rcx),$xb1,$xb1+ vpaddd 0xa0-0x100(%rcx),$xb2,$xb2+ vpaddd 0xb0-0x100(%rcx),$xb3,$xb3++ vmovdqa $xa0,0x00(%rsp) # offload $xa0,1+ vmovdqa $xa1,0x10(%rsp)+ vmovdqa 0x20(%rsp),$xa0 # "xc2"+ vmovdqa 0x30(%rsp),$xa1 # "xc3"++ vpunpckldq $xb1,$xb0,$xt2+ vpunpckldq $xb3,$xb2,$xt3+ vpunpckhdq $xb1,$xb0,$xb0+ vpunpckhdq $xb3,$xb2,$xb2+ vpunpcklqdq $xt3,$xt2,$xb1 # "b0"+ vpunpckhqdq $xt3,$xt2,$xt2 # "b1"+ vpunpcklqdq $xb2,$xb0,$xb3 # "b2"+ vpunpckhqdq $xb2,$xb0,$xb0 # "b3"+___+ ($xb0,$xb1,$xb2,$xb3,$xt2)=($xb1,$xt2,$xb3,$xb0,$xb2);+ my ($xc0,$xc1,$xc2,$xc3)=($xt0,$xt1,$xa0,$xa1);+$code.=<<___;+ vpaddd 0xc0-0x100(%rcx),$xc0,$xc0+ vpaddd 0xd0-0x100(%rcx),$xc1,$xc1+ vpaddd 0xe0-0x100(%rcx),$xc2,$xc2+ vpaddd 0xf0-0x100(%rcx),$xc3,$xc3++ vpunpckldq $xc1,$xc0,$xt2+ vpunpckldq $xc3,$xc2,$xt3+ vpunpckhdq $xc1,$xc0,$xc0+ vpunpckhdq $xc3,$xc2,$xc2+ vpunpcklqdq $xt3,$xt2,$xc1 # "c0"+ vpunpckhqdq $xt3,$xt2,$xt2 # "c1"+ vpunpcklqdq $xc2,$xc0,$xc3 # "c2"+ vpunpckhqdq $xc2,$xc0,$xc0 # "c3"+___+ ($xc0,$xc1,$xc2,$xc3,$xt2)=($xc1,$xt2,$xc3,$xc0,$xc2);+$code.=<<___;+ vpaddd 0x100-0x100(%rcx),$xd0,$xd0+ vpaddd 0x110-0x100(%rcx),$xd1,$xd1+ vpaddd 0x120-0x100(%rcx),$xd2,$xd2+ vpaddd 0x130-0x100(%rcx),$xd3,$xd3++ vpunpckldq $xd1,$xd0,$xt2+ vpunpckldq $xd3,$xd2,$xt3+ vpunpckhdq $xd1,$xd0,$xd0+ vpunpckhdq $xd3,$xd2,$xd2+ vpunpcklqdq $xt3,$xt2,$xd1 # "d0"+ vpunpckhqdq $xt3,$xt2,$xt2 # "d1"+ vpunpcklqdq $xd2,$xd0,$xd3 # "d2"+ vpunpckhqdq $xd2,$xd0,$xd0 # "d3"+___+ ($xd0,$xd1,$xd2,$xd3,$xt2)=($xd1,$xt2,$xd3,$xd0,$xd2);+ ($xa0,$xa1)=($xt2,$xt3);+$code.=<<___;+ vmovdqa 0x00(%rsp),$xa0 # restore $xa0,1+ vmovdqa 0x10(%rsp),$xa1++ cmp \$64*4,$len+ jb .Ltail4xop++ vpxor 0x00($inp),$xa0,$xa0 # xor with input+ vpxor 0x10($inp),$xb0,$xb0+ vpxor 0x20($inp),$xc0,$xc0+ vpxor 0x30($inp),$xd0,$xd0+ vpxor 0x40($inp),$xa1,$xa1+ vpxor 0x50($inp),$xb1,$xb1+ vpxor 0x60($inp),$xc1,$xc1+ vpxor 0x70($inp),$xd1,$xd1+ lea 0x80($inp),$inp # size optimization+ vpxor 0x00($inp),$xa2,$xa2+ vpxor 0x10($inp),$xb2,$xb2+ vpxor 0x20($inp),$xc2,$xc2+ vpxor 0x30($inp),$xd2,$xd2+ vpxor 0x40($inp),$xa3,$xa3+ vpxor 0x50($inp),$xb3,$xb3+ vpxor 0x60($inp),$xc3,$xc3+ vpxor 0x70($inp),$xd3,$xd3+ lea 0x80($inp),$inp # inp+=64*4++ vmovdqu $xa0,0x00($out)+ vmovdqu $xb0,0x10($out)+ vmovdqu $xc0,0x20($out)+ vmovdqu $xd0,0x30($out)+ vmovdqu $xa1,0x40($out)+ vmovdqu $xb1,0x50($out)+ vmovdqu $xc1,0x60($out)+ vmovdqu $xd1,0x70($out)+ lea 0x80($out),$out # size optimization+ vmovdqu $xa2,0x00($out)+ vmovdqu $xb2,0x10($out)+ vmovdqu $xc2,0x20($out)+ vmovdqu $xd2,0x30($out)+ vmovdqu $xa3,0x40($out)+ vmovdqu $xb3,0x50($out)+ vmovdqu $xc3,0x60($out)+ vmovdqu $xd3,0x70($out)+ lea 0x80($out),$out # out+=64*4++ sub \$64*4,$len+ jnz .Loop_outer4xop++ jmp .Ldone4xop++.align 32+.Ltail4xop:+ cmp \$192,$len+ jae .L192_or_more4xop+ cmp \$128,$len+ jae .L128_or_more4xop+ cmp \$64,$len+ jae .L64_or_more4xop++ xor %r9,%r9+ vmovdqa $xa0,0x00(%rsp)+ vmovdqa $xb0,0x10(%rsp)+ vmovdqa $xc0,0x20(%rsp)+ vmovdqa $xd0,0x30(%rsp)+ jmp .Loop_tail4xop++.align 32+.L64_or_more4xop:+ vpxor 0x00($inp),$xa0,$xa0 # xor with input+ vpxor 0x10($inp),$xb0,$xb0+ vpxor 0x20($inp),$xc0,$xc0+ vpxor 0x30($inp),$xd0,$xd0+ vmovdqu $xa0,0x00($out)+ vmovdqu $xb0,0x10($out)+ vmovdqu $xc0,0x20($out)+ vmovdqu $xd0,0x30($out)+ je .Ldone4xop++ lea 0x40($inp),$inp # inp+=64*1+ vmovdqa $xa1,0x00(%rsp)+ xor %r9,%r9+ vmovdqa $xb1,0x10(%rsp)+ lea 0x40($out),$out # out+=64*1+ vmovdqa $xc1,0x20(%rsp)+ sub \$64,$len # len-=64*1+ vmovdqa $xd1,0x30(%rsp)+ jmp .Loop_tail4xop++.align 32+.L128_or_more4xop:+ vpxor 0x00($inp),$xa0,$xa0 # xor with input+ vpxor 0x10($inp),$xb0,$xb0+ vpxor 0x20($inp),$xc0,$xc0+ vpxor 0x30($inp),$xd0,$xd0+ vpxor 0x40($inp),$xa1,$xa1+ vpxor 0x50($inp),$xb1,$xb1+ vpxor 0x60($inp),$xc1,$xc1+ vpxor 0x70($inp),$xd1,$xd1++ vmovdqu $xa0,0x00($out)+ vmovdqu $xb0,0x10($out)+ vmovdqu $xc0,0x20($out)+ vmovdqu $xd0,0x30($out)+ vmovdqu $xa1,0x40($out)+ vmovdqu $xb1,0x50($out)+ vmovdqu $xc1,0x60($out)+ vmovdqu $xd1,0x70($out)+ je .Ldone4xop++ lea 0x80($inp),$inp # inp+=64*2+ vmovdqa $xa2,0x00(%rsp)+ xor %r9,%r9+ vmovdqa $xb2,0x10(%rsp)+ lea 0x80($out),$out # out+=64*2+ vmovdqa $xc2,0x20(%rsp)+ sub \$128,$len # len-=64*2+ vmovdqa $xd2,0x30(%rsp)+ jmp .Loop_tail4xop++.align 32+.L192_or_more4xop:+ vpxor 0x00($inp),$xa0,$xa0 # xor with input+ vpxor 0x10($inp),$xb0,$xb0+ vpxor 0x20($inp),$xc0,$xc0+ vpxor 0x30($inp),$xd0,$xd0+ vpxor 0x40($inp),$xa1,$xa1+ vpxor 0x50($inp),$xb1,$xb1+ vpxor 0x60($inp),$xc1,$xc1+ vpxor 0x70($inp),$xd1,$xd1+ lea 0x80($inp),$inp # size optimization+ vpxor 0x00($inp),$xa2,$xa2+ vpxor 0x10($inp),$xb2,$xb2+ vpxor 0x20($inp),$xc2,$xc2+ vpxor 0x30($inp),$xd2,$xd2++ vmovdqu $xa0,0x00($out)+ vmovdqu $xb0,0x10($out)+ vmovdqu $xc0,0x20($out)+ vmovdqu $xd0,0x30($out)+ vmovdqu $xa1,0x40($out)+ vmovdqu $xb1,0x50($out)+ vmovdqu $xc1,0x60($out)+ vmovdqu $xd1,0x70($out)+ lea 0x80($out),$out # size optimization+ vmovdqu $xa2,0x00($out)+ vmovdqu $xb2,0x10($out)+ vmovdqu $xc2,0x20($out)+ vmovdqu $xd2,0x30($out)+ je .Ldone4xop++ lea 0x40($inp),$inp # inp+=64*3+ vmovdqa $xa3,0x00(%rsp)+ xor %r9,%r9+ vmovdqa $xb3,0x10(%rsp)+ lea 0x40($out),$out # out+=64*3+ vmovdqa $xc3,0x20(%rsp)+ sub \$192,$len # len-=64*3+ vmovdqa $xd3,0x30(%rsp)++.Loop_tail4xop:+ movzb ($inp,%r9),%eax+ movzb (%rsp,%r9),%ecx+ lea 1(%r9),%r9+ xor %ecx,%eax+ mov %al,-1($out,%r9)+ dec $len+ jnz .Loop_tail4xop++.Ldone4xop:+ vzeroupper+___+$code.=<<___ if ($win64);+ movaps -0xa8(%r10),%xmm6+ movaps -0x98(%r10),%xmm7+ movaps -0x88(%r10),%xmm8+ movaps -0x78(%r10),%xmm9+ movaps -0x68(%r10),%xmm10+ movaps -0x58(%r10),%xmm11+ movaps -0x48(%r10),%xmm12+ movaps -0x38(%r10),%xmm13+ movaps -0x28(%r10),%xmm14+ movaps -0x18(%r10),%xmm15+___+$code.=<<___;+ lea (%r10),%rsp+.cfi_def_cfa_register %rsp+.L4xop_epilogue:+ ret+.cfi_endproc+.size ChaCha20_4xop,.-ChaCha20_4xop+___+}++########################################################################+# AVX2 code path+if ($avx>1) {+my ($xb0,$xb1,$xb2,$xb3, $xd0,$xd1,$xd2,$xd3,+ $xa0,$xa1,$xa2,$xa3, $xt0,$xt1,$xt2,$xt3)=map("%ymm$_",(0..15));+my @xx=($xa0,$xa1,$xa2,$xa3, $xb0,$xb1,$xb2,$xb3,+ "%nox","%nox","%nox","%nox", $xd0,$xd1,$xd2,$xd3);++sub AVX2_lane_ROUND {+my ($a0,$b0,$c0,$d0)=@_;+my ($a1,$b1,$c1,$d1)=map(($_&~3)+(($_+1)&3),($a0,$b0,$c0,$d0));+my ($a2,$b2,$c2,$d2)=map(($_&~3)+(($_+1)&3),($a1,$b1,$c1,$d1));+my ($a3,$b3,$c3,$d3)=map(($_&~3)+(($_+1)&3),($a2,$b2,$c2,$d2));+my ($xc,$xc_,$t0,$t1)=map("\"$_\"",$xt0,$xt1,$xt2,$xt3);+my @x=map("\"$_\"",@xx);++ # Consider order in which variables are addressed by their+ # index:+ #+ # a b c d+ #+ # 0 4 8 12 < even round+ # 1 5 9 13+ # 2 6 10 14+ # 3 7 11 15+ # 0 5 10 15 < odd round+ # 1 6 11 12+ # 2 7 8 13+ # 3 4 9 14+ #+ # 'a', 'b' and 'd's are permanently allocated in registers,+ # @x[0..7,12..15], while 'c's are maintained in memory. If+ # you observe 'c' column, you'll notice that pair of 'c's is+ # invariant between rounds. This means that we have to reload+ # them once per round, in the middle. This is why you'll see+ # bunch of 'c' stores and loads in the middle, but none in+ # the beginning or end.++ (+ "&vpaddd (@x[$a0],@x[$a0],@x[$b0])", # Q1+ "&vpxor (@x[$d0],@x[$a0],@x[$d0])",+ "&vpshufb (@x[$d0],@x[$d0],$t1)",+ "&vpaddd (@x[$a1],@x[$a1],@x[$b1])", # Q2+ "&vpxor (@x[$d1],@x[$a1],@x[$d1])",+ "&vpshufb (@x[$d1],@x[$d1],$t1)",++ "&vpaddd ($xc,$xc,@x[$d0])",+ "&vpxor (@x[$b0],$xc,@x[$b0])",+ "&vpslld ($t0,@x[$b0],12)",+ "&vpsrld (@x[$b0],@x[$b0],20)",+ "&vpor (@x[$b0],$t0,@x[$b0])",+ "&vbroadcasti128($t0,'(%r11)')", # .Lrot24(%rip)+ "&vpaddd ($xc_,$xc_,@x[$d1])",+ "&vpxor (@x[$b1],$xc_,@x[$b1])",+ "&vpslld ($t1,@x[$b1],12)",+ "&vpsrld (@x[$b1],@x[$b1],20)",+ "&vpor (@x[$b1],$t1,@x[$b1])",++ "&vpaddd (@x[$a0],@x[$a0],@x[$b0])",+ "&vpxor (@x[$d0],@x[$a0],@x[$d0])",+ "&vpshufb (@x[$d0],@x[$d0],$t0)",+ "&vpaddd (@x[$a1],@x[$a1],@x[$b1])",+ "&vpxor (@x[$d1],@x[$a1],@x[$d1])",+ "&vpshufb (@x[$d1],@x[$d1],$t0)",++ "&vpaddd ($xc,$xc,@x[$d0])",+ "&vpxor (@x[$b0],$xc,@x[$b0])",+ "&vpslld ($t1,@x[$b0],7)",+ "&vpsrld (@x[$b0],@x[$b0],25)",+ "&vpor (@x[$b0],$t1,@x[$b0])",+ "&vbroadcasti128($t1,'(%r9)')", # .Lrot16(%rip)+ "&vpaddd ($xc_,$xc_,@x[$d1])",+ "&vpxor (@x[$b1],$xc_,@x[$b1])",+ "&vpslld ($t0,@x[$b1],7)",+ "&vpsrld (@x[$b1],@x[$b1],25)",+ "&vpor (@x[$b1],$t0,@x[$b1])",++ "&vmovdqa (\"`32*($c0-8)`(%rsp)\",$xc)", # reload pair of 'c's+ "&vmovdqa (\"`32*($c1-8)`(%rsp)\",$xc_)",+ "&vmovdqa ($xc,\"`32*($c2-8)`(%rsp)\")",+ "&vmovdqa ($xc_,\"`32*($c3-8)`(%rsp)\")",++ "&vpaddd (@x[$a2],@x[$a2],@x[$b2])", # Q3+ "&vpxor (@x[$d2],@x[$a2],@x[$d2])",+ "&vpshufb (@x[$d2],@x[$d2],$t1)",+ "&vpaddd (@x[$a3],@x[$a3],@x[$b3])", # Q4+ "&vpxor (@x[$d3],@x[$a3],@x[$d3])",+ "&vpshufb (@x[$d3],@x[$d3],$t1)",++ "&vpaddd ($xc,$xc,@x[$d2])",+ "&vpxor (@x[$b2],$xc,@x[$b2])",+ "&vpslld ($t0,@x[$b2],12)",+ "&vpsrld (@x[$b2],@x[$b2],20)",+ "&vpor (@x[$b2],$t0,@x[$b2])",+ "&vbroadcasti128($t0,'(%r11)')", # .Lrot24(%rip)+ "&vpaddd ($xc_,$xc_,@x[$d3])",+ "&vpxor (@x[$b3],$xc_,@x[$b3])",+ "&vpslld ($t1,@x[$b3],12)",+ "&vpsrld (@x[$b3],@x[$b3],20)",+ "&vpor (@x[$b3],$t1,@x[$b3])",++ "&vpaddd (@x[$a2],@x[$a2],@x[$b2])",+ "&vpxor (@x[$d2],@x[$a2],@x[$d2])",+ "&vpshufb (@x[$d2],@x[$d2],$t0)",+ "&vpaddd (@x[$a3],@x[$a3],@x[$b3])",+ "&vpxor (@x[$d3],@x[$a3],@x[$d3])",+ "&vpshufb (@x[$d3],@x[$d3],$t0)",++ "&vpaddd ($xc,$xc,@x[$d2])",+ "&vpxor (@x[$b2],$xc,@x[$b2])",+ "&vpslld ($t1,@x[$b2],7)",+ "&vpsrld (@x[$b2],@x[$b2],25)",+ "&vpor (@x[$b2],$t1,@x[$b2])",+ "&vbroadcasti128($t1,'(%r9)')", # .Lrot16(%rip)+ "&vpaddd ($xc_,$xc_,@x[$d3])",+ "&vpxor (@x[$b3],$xc_,@x[$b3])",+ "&vpslld ($t0,@x[$b3],7)",+ "&vpsrld (@x[$b3],@x[$b3],25)",+ "&vpor (@x[$b3],$t0,@x[$b3])"+ );+}++my $xframe = $win64 ? 0xa8 : 8;++$code.=<<___ if ($flavour =~ /kernel/);+.globl ChaCha20_avx2+___+$code.=<<___;+.type ChaCha20_avx2,\@function,5+.align 32+ChaCha20_avx2:+.cfi_startproc+.LChaCha20_8x:+ mov %rsp,%r10 # frame register+.cfi_def_cfa_register %r10+ sub \$0x280+$xframe,%rsp+ and \$-32,%rsp+___+$code.=<<___ if ($win64);+ movaps %xmm6,-0xa8(%r10)+ movaps %xmm7,-0x98(%r10)+ movaps %xmm8,-0x88(%r10)+ movaps %xmm9,-0x78(%r10)+ movaps %xmm10,-0x68(%r10)+ movaps %xmm11,-0x58(%r10)+ movaps %xmm12,-0x48(%r10)+ movaps %xmm13,-0x38(%r10)+ movaps %xmm14,-0x28(%r10)+ movaps %xmm15,-0x18(%r10)+.Lavx2_body:+___+$code.=<<___;+ vzeroupper++ ################ stack layout+ # +0x00 SIMD equivalent of @x[8-12]+ # ...+ # +0x80 constant copy of key[0-2] smashed by lanes+ # ...+ # +0x200 SIMD counters (with nonce smashed by lanes)+ # ...+ # +0x280++ vbroadcasti128 .Lsigma(%rip),$xa3 # key[0]+ vbroadcasti128 ($key),$xb3 # key[1]+ vbroadcasti128 16($key),$xt3 # key[2]+ vbroadcasti128 ($counter),$xd3 # key[3]+ lea 0x100(%rsp),%rcx # size optimization+ lea 0x200(%rsp),%rax # size optimization+ lea .Lrot16(%rip),%r9+ lea .Lrot24(%rip),%r11++ vpshufd \$0x00,$xa3,$xa0 # smash key by lanes...+ vpshufd \$0x55,$xa3,$xa1+ vmovdqa $xa0,0x80-0x100(%rcx) # ... and offload+ vpshufd \$0xaa,$xa3,$xa2+ vmovdqa $xa1,0xa0-0x100(%rcx)+ vpshufd \$0xff,$xa3,$xa3+ vmovdqa $xa2,0xc0-0x100(%rcx)+ vmovdqa $xa3,0xe0-0x100(%rcx)++ vpshufd \$0x00,$xb3,$xb0+ vpshufd \$0x55,$xb3,$xb1+ vmovdqa $xb0,0x100-0x100(%rcx)+ vpshufd \$0xaa,$xb3,$xb2+ vmovdqa $xb1,0x120-0x100(%rcx)+ vpshufd \$0xff,$xb3,$xb3+ vmovdqa $xb2,0x140-0x100(%rcx)+ vmovdqa $xb3,0x160-0x100(%rcx)++ vpshufd \$0x00,$xt3,$xt0 # "xc0"+ vpshufd \$0x55,$xt3,$xt1 # "xc1"+ vmovdqa $xt0,0x180-0x200(%rax)+ vpshufd \$0xaa,$xt3,$xt2 # "xc2"+ vmovdqa $xt1,0x1a0-0x200(%rax)+ vpshufd \$0xff,$xt3,$xt3 # "xc3"+ vmovdqa $xt2,0x1c0-0x200(%rax)+ vmovdqa $xt3,0x1e0-0x200(%rax)++ vpshufd \$0x00,$xd3,$xd0+ vpshufd \$0x55,$xd3,$xd1+ vpaddd .Lincy(%rip),$xd0,$xd0 # don't save counters yet+ vpshufd \$0xaa,$xd3,$xd2+ vmovdqa $xd1,0x220-0x200(%rax)+ vpshufd \$0xff,$xd3,$xd3+ vmovdqa $xd2,0x240-0x200(%rax)+ vmovdqa $xd3,0x260-0x200(%rax)++ jmp .Loop_enter8x++.align 32+.Loop_outer8x:+ vmovdqa 0x80-0x100(%rcx),$xa0 # re-load smashed key+ vmovdqa 0xa0-0x100(%rcx),$xa1+ vmovdqa 0xc0-0x100(%rcx),$xa2+ vmovdqa 0xe0-0x100(%rcx),$xa3+ vmovdqa 0x100-0x100(%rcx),$xb0+ vmovdqa 0x120-0x100(%rcx),$xb1+ vmovdqa 0x140-0x100(%rcx),$xb2+ vmovdqa 0x160-0x100(%rcx),$xb3+ vmovdqa 0x180-0x200(%rax),$xt0 # "xc0"+ vmovdqa 0x1a0-0x200(%rax),$xt1 # "xc1"+ vmovdqa 0x1c0-0x200(%rax),$xt2 # "xc2"+ vmovdqa 0x1e0-0x200(%rax),$xt3 # "xc3"+ vmovdqa 0x200-0x200(%rax),$xd0+ vmovdqa 0x220-0x200(%rax),$xd1+ vmovdqa 0x240-0x200(%rax),$xd2+ vmovdqa 0x260-0x200(%rax),$xd3+ vpaddd .Leight(%rip),$xd0,$xd0 # next SIMD counters++.Loop_enter8x:+ vmovdqa $xt2,0x40(%rsp) # SIMD equivalent of "@x[10]"+ vmovdqa $xt3,0x60(%rsp) # SIMD equivalent of "@x[11]"+ vbroadcasti128 (%r9),$xt3+ vmovdqa $xd0,0x200-0x200(%rax) # save SIMD counters+ mov \$10,%eax+ jmp .Loop8x++.align 32+.Loop8x:+___+ foreach (&AVX2_lane_ROUND(0, 4, 8,12)) { eval; }+ foreach (&AVX2_lane_ROUND(0, 5,10,15)) { eval; }+$code.=<<___;+ dec %eax+ jnz .Loop8x++ lea 0x200(%rsp),%rax # size optimization+ vpaddd 0x80-0x100(%rcx),$xa0,$xa0 # accumulate key+ vpaddd 0xa0-0x100(%rcx),$xa1,$xa1+ vpaddd 0xc0-0x100(%rcx),$xa2,$xa2+ vpaddd 0xe0-0x100(%rcx),$xa3,$xa3++ vpunpckldq $xa1,$xa0,$xt2 # "de-interlace" data+ vpunpckldq $xa3,$xa2,$xt3+ vpunpckhdq $xa1,$xa0,$xa0+ vpunpckhdq $xa3,$xa2,$xa2+ vpunpcklqdq $xt3,$xt2,$xa1 # "a0"+ vpunpckhqdq $xt3,$xt2,$xt2 # "a1"+ vpunpcklqdq $xa2,$xa0,$xa3 # "a2"+ vpunpckhqdq $xa2,$xa0,$xa0 # "a3"+___+ ($xa0,$xa1,$xa2,$xa3,$xt2)=($xa1,$xt2,$xa3,$xa0,$xa2);+$code.=<<___;+ vpaddd 0x100-0x100(%rcx),$xb0,$xb0+ vpaddd 0x120-0x100(%rcx),$xb1,$xb1+ vpaddd 0x140-0x100(%rcx),$xb2,$xb2+ vpaddd 0x160-0x100(%rcx),$xb3,$xb3++ vpunpckldq $xb1,$xb0,$xt2+ vpunpckldq $xb3,$xb2,$xt3+ vpunpckhdq $xb1,$xb0,$xb0+ vpunpckhdq $xb3,$xb2,$xb2+ vpunpcklqdq $xt3,$xt2,$xb1 # "b0"+ vpunpckhqdq $xt3,$xt2,$xt2 # "b1"+ vpunpcklqdq $xb2,$xb0,$xb3 # "b2"+ vpunpckhqdq $xb2,$xb0,$xb0 # "b3"+___+ ($xb0,$xb1,$xb2,$xb3,$xt2)=($xb1,$xt2,$xb3,$xb0,$xb2);+$code.=<<___;+ vperm2i128 \$0x20,$xb0,$xa0,$xt3 # "de-interlace" further+ vperm2i128 \$0x31,$xb0,$xa0,$xb0+ vperm2i128 \$0x20,$xb1,$xa1,$xa0+ vperm2i128 \$0x31,$xb1,$xa1,$xb1+ vperm2i128 \$0x20,$xb2,$xa2,$xa1+ vperm2i128 \$0x31,$xb2,$xa2,$xb2+ vperm2i128 \$0x20,$xb3,$xa3,$xa2+ vperm2i128 \$0x31,$xb3,$xa3,$xb3+___+ ($xa0,$xa1,$xa2,$xa3,$xt3)=($xt3,$xa0,$xa1,$xa2,$xa3);+ my ($xc0,$xc1,$xc2,$xc3)=($xt0,$xt1,$xa0,$xa1);+$code.=<<___;+ vmovdqa $xa0,0x00(%rsp) # offload $xaN+ vmovdqa $xa1,0x20(%rsp)+ vmovdqa 0x40(%rsp),$xc2 # $xa0+ vmovdqa 0x60(%rsp),$xc3 # $xa1++ vpaddd 0x180-0x200(%rax),$xc0,$xc0+ vpaddd 0x1a0-0x200(%rax),$xc1,$xc1+ vpaddd 0x1c0-0x200(%rax),$xc2,$xc2+ vpaddd 0x1e0-0x200(%rax),$xc3,$xc3++ vpunpckldq $xc1,$xc0,$xt2+ vpunpckldq $xc3,$xc2,$xt3+ vpunpckhdq $xc1,$xc0,$xc0+ vpunpckhdq $xc3,$xc2,$xc2+ vpunpcklqdq $xt3,$xt2,$xc1 # "c0"+ vpunpckhqdq $xt3,$xt2,$xt2 # "c1"+ vpunpcklqdq $xc2,$xc0,$xc3 # "c2"+ vpunpckhqdq $xc2,$xc0,$xc0 # "c3"+___+ ($xc0,$xc1,$xc2,$xc3,$xt2)=($xc1,$xt2,$xc3,$xc0,$xc2);+$code.=<<___;+ vpaddd 0x200-0x200(%rax),$xd0,$xd0+ vpaddd 0x220-0x200(%rax),$xd1,$xd1+ vpaddd 0x240-0x200(%rax),$xd2,$xd2+ vpaddd 0x260-0x200(%rax),$xd3,$xd3++ vpunpckldq $xd1,$xd0,$xt2+ vpunpckldq $xd3,$xd2,$xt3+ vpunpckhdq $xd1,$xd0,$xd0+ vpunpckhdq $xd3,$xd2,$xd2+ vpunpcklqdq $xt3,$xt2,$xd1 # "d0"+ vpunpckhqdq $xt3,$xt2,$xt2 # "d1"+ vpunpcklqdq $xd2,$xd0,$xd3 # "d2"+ vpunpckhqdq $xd2,$xd0,$xd0 # "d3"+___+ ($xd0,$xd1,$xd2,$xd3,$xt2)=($xd1,$xt2,$xd3,$xd0,$xd2);+$code.=<<___;+ vperm2i128 \$0x20,$xd0,$xc0,$xt3 # "de-interlace" further+ vperm2i128 \$0x31,$xd0,$xc0,$xd0+ vperm2i128 \$0x20,$xd1,$xc1,$xc0+ vperm2i128 \$0x31,$xd1,$xc1,$xd1+ vperm2i128 \$0x20,$xd2,$xc2,$xc1+ vperm2i128 \$0x31,$xd2,$xc2,$xd2+ vperm2i128 \$0x20,$xd3,$xc3,$xc2+ vperm2i128 \$0x31,$xd3,$xc3,$xd3+___+ ($xc0,$xc1,$xc2,$xc3,$xt3)=($xt3,$xc0,$xc1,$xc2,$xc3);+ ($xb0,$xb1,$xb2,$xb3,$xc0,$xc1,$xc2,$xc3)=+ ($xc0,$xc1,$xc2,$xc3,$xb0,$xb1,$xb2,$xb3);+ ($xa0,$xa1)=($xt2,$xt3);+$code.=<<___;+ vmovdqa 0x00(%rsp),$xa0 # $xaN was offloaded, remember?+ vmovdqa 0x20(%rsp),$xa1++ cmp \$64*8,$len+ jb .Ltail8x++ vpxor 0x00($inp),$xa0,$xa0 # xor with input+ vpxor 0x20($inp),$xb0,$xb0+ vpxor 0x40($inp),$xc0,$xc0+ vpxor 0x60($inp),$xd0,$xd0+ lea 0x80($inp),$inp # size optimization+ vmovdqu $xa0,0x00($out)+ vmovdqu $xb0,0x20($out)+ vmovdqu $xc0,0x40($out)+ vmovdqu $xd0,0x60($out)+ lea 0x80($out),$out # size optimization++ vpxor 0x00($inp),$xa1,$xa1+ vpxor 0x20($inp),$xb1,$xb1+ vpxor 0x40($inp),$xc1,$xc1+ vpxor 0x60($inp),$xd1,$xd1+ lea 0x80($inp),$inp # size optimization+ vmovdqu $xa1,0x00($out)+ vmovdqu $xb1,0x20($out)+ vmovdqu $xc1,0x40($out)+ vmovdqu $xd1,0x60($out)+ lea 0x80($out),$out # size optimization++ vpxor 0x00($inp),$xa2,$xa2+ vpxor 0x20($inp),$xb2,$xb2+ vpxor 0x40($inp),$xc2,$xc2+ vpxor 0x60($inp),$xd2,$xd2+ lea 0x80($inp),$inp # size optimization+ vmovdqu $xa2,0x00($out)+ vmovdqu $xb2,0x20($out)+ vmovdqu $xc2,0x40($out)+ vmovdqu $xd2,0x60($out)+ lea 0x80($out),$out # size optimization++ vpxor 0x00($inp),$xa3,$xa3+ vpxor 0x20($inp),$xb3,$xb3+ vpxor 0x40($inp),$xc3,$xc3+ vpxor 0x60($inp),$xd3,$xd3+ lea 0x80($inp),$inp # size optimization+ vmovdqu $xa3,0x00($out)+ vmovdqu $xb3,0x20($out)+ vmovdqu $xc3,0x40($out)+ vmovdqu $xd3,0x60($out)+ lea 0x80($out),$out # size optimization++ sub \$64*8,$len+ jnz .Loop_outer8x++ jmp .Ldone8x++.Ltail8x:+ cmp \$448,$len+ jae .L448_or_more8x+ cmp \$384,$len+ jae .L384_or_more8x+ cmp \$320,$len+ jae .L320_or_more8x+ cmp \$256,$len+ jae .L256_or_more8x+ cmp \$192,$len+ jae .L192_or_more8x+ cmp \$128,$len+ jae .L128_or_more8x+ cmp \$64,$len+ jae .L64_or_more8x++ xor %r9,%r9+ vmovdqa $xa0,0x00(%rsp)+ vmovdqa $xb0,0x20(%rsp)+ jmp .Loop_tail8x++.align 32+.L64_or_more8x:+ vpxor 0x00($inp),$xa0,$xa0 # xor with input+ vpxor 0x20($inp),$xb0,$xb0+ vmovdqu $xa0,0x00($out)+ vmovdqu $xb0,0x20($out)+ je .Ldone8x++ lea 0x40($inp),$inp # inp+=64*1+ xor %r9,%r9+ vmovdqa $xc0,0x00(%rsp)+ lea 0x40($out),$out # out+=64*1+ sub \$64,$len # len-=64*1+ vmovdqa $xd0,0x20(%rsp)+ jmp .Loop_tail8x++.align 32+.L128_or_more8x:+ vpxor 0x00($inp),$xa0,$xa0 # xor with input+ vpxor 0x20($inp),$xb0,$xb0+ vpxor 0x40($inp),$xc0,$xc0+ vpxor 0x60($inp),$xd0,$xd0+ vmovdqu $xa0,0x00($out)+ vmovdqu $xb0,0x20($out)+ vmovdqu $xc0,0x40($out)+ vmovdqu $xd0,0x60($out)+ je .Ldone8x++ lea 0x80($inp),$inp # inp+=64*2+ xor %r9,%r9+ vmovdqa $xa1,0x00(%rsp)+ lea 0x80($out),$out # out+=64*2+ sub \$128,$len # len-=64*2+ vmovdqa $xb1,0x20(%rsp)+ jmp .Loop_tail8x++.align 32+.L192_or_more8x:+ vpxor 0x00($inp),$xa0,$xa0 # xor with input+ vpxor 0x20($inp),$xb0,$xb0+ vpxor 0x40($inp),$xc0,$xc0+ vpxor 0x60($inp),$xd0,$xd0+ vpxor 0x80($inp),$xa1,$xa1+ vpxor 0xa0($inp),$xb1,$xb1+ vmovdqu $xa0,0x00($out)+ vmovdqu $xb0,0x20($out)+ vmovdqu $xc0,0x40($out)+ vmovdqu $xd0,0x60($out)+ vmovdqu $xa1,0x80($out)+ vmovdqu $xb1,0xa0($out)+ je .Ldone8x++ lea 0xc0($inp),$inp # inp+=64*3+ xor %r9,%r9+ vmovdqa $xc1,0x00(%rsp)+ lea 0xc0($out),$out # out+=64*3+ sub \$192,$len # len-=64*3+ vmovdqa $xd1,0x20(%rsp)+ jmp .Loop_tail8x++.align 32+.L256_or_more8x:+ vpxor 0x00($inp),$xa0,$xa0 # xor with input+ vpxor 0x20($inp),$xb0,$xb0+ vpxor 0x40($inp),$xc0,$xc0+ vpxor 0x60($inp),$xd0,$xd0+ vpxor 0x80($inp),$xa1,$xa1+ vpxor 0xa0($inp),$xb1,$xb1+ vpxor 0xc0($inp),$xc1,$xc1+ vpxor 0xe0($inp),$xd1,$xd1+ vmovdqu $xa0,0x00($out)+ vmovdqu $xb0,0x20($out)+ vmovdqu $xc0,0x40($out)+ vmovdqu $xd0,0x60($out)+ vmovdqu $xa1,0x80($out)+ vmovdqu $xb1,0xa0($out)+ vmovdqu $xc1,0xc0($out)+ vmovdqu $xd1,0xe0($out)+ je .Ldone8x++ lea 0x100($inp),$inp # inp+=64*4+ xor %r9,%r9+ vmovdqa $xa2,0x00(%rsp)+ lea 0x100($out),$out # out+=64*4+ sub \$256,$len # len-=64*4+ vmovdqa $xb2,0x20(%rsp)+ jmp .Loop_tail8x++.align 32+.L320_or_more8x:+ vpxor 0x00($inp),$xa0,$xa0 # xor with input+ vpxor 0x20($inp),$xb0,$xb0+ vpxor 0x40($inp),$xc0,$xc0+ vpxor 0x60($inp),$xd0,$xd0+ vpxor 0x80($inp),$xa1,$xa1+ vpxor 0xa0($inp),$xb1,$xb1+ vpxor 0xc0($inp),$xc1,$xc1+ vpxor 0xe0($inp),$xd1,$xd1+ vpxor 0x100($inp),$xa2,$xa2+ vpxor 0x120($inp),$xb2,$xb2+ vmovdqu $xa0,0x00($out)+ vmovdqu $xb0,0x20($out)+ vmovdqu $xc0,0x40($out)+ vmovdqu $xd0,0x60($out)+ vmovdqu $xa1,0x80($out)+ vmovdqu $xb1,0xa0($out)+ vmovdqu $xc1,0xc0($out)+ vmovdqu $xd1,0xe0($out)+ vmovdqu $xa2,0x100($out)+ vmovdqu $xb2,0x120($out)+ je .Ldone8x++ lea 0x140($inp),$inp # inp+=64*5+ xor %r9,%r9+ vmovdqa $xc2,0x00(%rsp)+ lea 0x140($out),$out # out+=64*5+ sub \$320,$len # len-=64*5+ vmovdqa $xd2,0x20(%rsp)+ jmp .Loop_tail8x++.align 32+.L384_or_more8x:+ vpxor 0x00($inp),$xa0,$xa0 # xor with input+ vpxor 0x20($inp),$xb0,$xb0+ vpxor 0x40($inp),$xc0,$xc0+ vpxor 0x60($inp),$xd0,$xd0+ vpxor 0x80($inp),$xa1,$xa1+ vpxor 0xa0($inp),$xb1,$xb1+ vpxor 0xc0($inp),$xc1,$xc1+ vpxor 0xe0($inp),$xd1,$xd1+ vpxor 0x100($inp),$xa2,$xa2+ vpxor 0x120($inp),$xb2,$xb2+ vpxor 0x140($inp),$xc2,$xc2+ vpxor 0x160($inp),$xd2,$xd2+ vmovdqu $xa0,0x00($out)+ vmovdqu $xb0,0x20($out)+ vmovdqu $xc0,0x40($out)+ vmovdqu $xd0,0x60($out)+ vmovdqu $xa1,0x80($out)+ vmovdqu $xb1,0xa0($out)+ vmovdqu $xc1,0xc0($out)+ vmovdqu $xd1,0xe0($out)+ vmovdqu $xa2,0x100($out)+ vmovdqu $xb2,0x120($out)+ vmovdqu $xc2,0x140($out)+ vmovdqu $xd2,0x160($out)+ je .Ldone8x++ lea 0x180($inp),$inp # inp+=64*6+ xor %r9,%r9+ vmovdqa $xa3,0x00(%rsp)+ lea 0x180($out),$out # out+=64*6+ sub \$384,$len # len-=64*6+ vmovdqa $xb3,0x20(%rsp)+ jmp .Loop_tail8x++.align 32+.L448_or_more8x:+ vpxor 0x00($inp),$xa0,$xa0 # xor with input+ vpxor 0x20($inp),$xb0,$xb0+ vpxor 0x40($inp),$xc0,$xc0+ vpxor 0x60($inp),$xd0,$xd0+ vpxor 0x80($inp),$xa1,$xa1+ vpxor 0xa0($inp),$xb1,$xb1+ vpxor 0xc0($inp),$xc1,$xc1+ vpxor 0xe0($inp),$xd1,$xd1+ vpxor 0x100($inp),$xa2,$xa2+ vpxor 0x120($inp),$xb2,$xb2+ vpxor 0x140($inp),$xc2,$xc2+ vpxor 0x160($inp),$xd2,$xd2+ vpxor 0x180($inp),$xa3,$xa3+ vpxor 0x1a0($inp),$xb3,$xb3+ vmovdqu $xa0,0x00($out)+ vmovdqu $xb0,0x20($out)+ vmovdqu $xc0,0x40($out)+ vmovdqu $xd0,0x60($out)+ vmovdqu $xa1,0x80($out)+ vmovdqu $xb1,0xa0($out)+ vmovdqu $xc1,0xc0($out)+ vmovdqu $xd1,0xe0($out)+ vmovdqu $xa2,0x100($out)+ vmovdqu $xb2,0x120($out)+ vmovdqu $xc2,0x140($out)+ vmovdqu $xd2,0x160($out)+ vmovdqu $xa3,0x180($out)+ vmovdqu $xb3,0x1a0($out)+ je .Ldone8x++ lea 0x1c0($inp),$inp # inp+=64*7+ xor %r9,%r9+ vmovdqa $xc3,0x00(%rsp)+ lea 0x1c0($out),$out # out+=64*7+ sub \$448,$len # len-=64*7+ vmovdqa $xd3,0x20(%rsp)++.Loop_tail8x:+ movzb ($inp,%r9),%eax+ movzb (%rsp,%r9),%ecx+ lea 1(%r9),%r9+ xor %ecx,%eax+ mov %al,-1($out,%r9)+ dec $len+ jnz .Loop_tail8x++.Ldone8x:+ vzeroall+___+$code.=<<___ if ($win64);+ movaps -0xa8(%r10),%xmm6+ movaps -0x98(%r10),%xmm7+ movaps -0x88(%r10),%xmm8+ movaps -0x78(%r10),%xmm9+ movaps -0x68(%r10),%xmm10+ movaps -0x58(%r10),%xmm11+ movaps -0x48(%r10),%xmm12+ movaps -0x38(%r10),%xmm13+ movaps -0x28(%r10),%xmm14+ movaps -0x18(%r10),%xmm15+___+$code.=<<___;+ lea (%r10),%rsp+.cfi_def_cfa_register %rsp+.Lavx2_epilogue:+ ret+.cfi_endproc+.size ChaCha20_avx2,.-ChaCha20_avx2+___+}++########################################################################+# AVX512 code paths+if ($avx>2) {+# This one handles shorter inputs...++my ($a,$b,$c,$d, $a_,$b_,$c_,$d_,$fourz) = map("%zmm$_",(0..3,16..20));+my ($t0,$t1,$t2,$t3) = map("%xmm$_",(4..7));++sub vpxord() # size optimization+{ my $opcode = "vpxor"; # adhere to vpxor when possible++ foreach (@_) {+ if (/%([zy])mm([0-9]+)/ && ($1 eq "z" || $2>=16)) {+ $opcode = "vpxord";+ last;+ }+ }++ $code .= "\t$opcode\t".join(',',reverse @_)."\n";+}++sub AVX512ROUND { # critical path is 14 "SIMD ticks" per round+ &vpaddd ($a,$a,$b);+ &vpxord ($d,$d,$a);+ &vprold ($d,$d,16);++ &vpaddd ($c,$c,$d);+ &vpxord ($b,$b,$c);+ &vprold ($b,$b,12);++ &vpaddd ($a,$a,$b);+ &vpxord ($d,$d,$a);+ &vprold ($d,$d,8);++ &vpaddd ($c,$c,$d);+ &vpxord ($b,$b,$c);+ &vprold ($b,$b,7);+}++my $xframe = $win64 ? 32+8 : 8;++$code.=<<___ if ($flavour =~ /kernel/);+.globl ChaCha20_avx512+___+$code.=<<___;+.type ChaCha20_avx512,\@function,5+.align 32+ChaCha20_avx512:+.cfi_startproc+.LChaCha20_avx512:+ mov %rsp,%r10 # frame pointer+.cfi_def_cfa_register %r10+ cmp \$512,$len+ ja .LChaCha20_16x++ sub \$64+$xframe,%rsp+ and \$-16,%rsp+___+$code.=<<___ if ($win64);+ movaps %xmm6,-0x28(%r10)+ movaps %xmm7,-0x18(%r10)+.Lavx512_body:+___+$code.=<<___;+ vbroadcasti32x4 .Lsigma(%rip),$a+ vbroadcasti32x4 ($key),$b_+ vbroadcasti32x4 16($key),$c_+ vbroadcasti32x4 ($counter),$d_++ vmovdqa32 $a,$a_+ vmovdqa32 .Lfourz(%rip),$fourz+ vpaddd .Lzeroz(%rip),$d_,$d+ jmp .Loop_outer_avx512++.align 32+.Loop_outer_avx512:+ vmovdqa32 $b_,$b+ vmovdqa32 $c_,$c+ vmovdqa32 $d,$d_+ mov \$10,$counter # reuse $counter+ jmp .Loop_avx512++.align 32+.Loop_avx512:+___+ &AVX512ROUND();+ &vpshufd ($c,$c,0b01001110);+ &vpshufd ($b,$b,0b00111001);+ &vpshufd ($d,$d,0b10010011);++ &AVX512ROUND();+ &vpshufd ($c,$c,0b01001110);+ &vpshufd ($b,$b,0b10010011);+ &vpshufd ($d,$d,0b00111001);++ &dec ($counter);+ &jnz (".Loop_avx512");++$code.=<<___;+ vpaddd $a_,$a,$a+ vpaddd $b_,$b,$b+ vpaddd $c_,$c,$c+ vpaddd $d_,$d,$d++ sub \$64,$len+ jb .Ltail64_avx512++ vpxor 0x00($inp),%x#$a,$t0 # xor with input+ vpxor 0x10($inp),%x#$b,$t1+ vpxor 0x20($inp),%x#$c,$t2+ vpxor 0x30($inp),%x#$d,$t3+ lea 0x40($inp),$inp # inp+=64++ vmovdqu $t0,0x00($out) # write output+ vmovdqu $t1,0x10($out)+ vmovdqu $t2,0x20($out)+ vmovdqu $t3,0x30($out)+ lea 0x40($out),$out # out+=64++ jz .Ldone_avx512++ vextracti32x4 \$1,$a,$t0+ vextracti32x4 \$1,$b,$t1+ vextracti32x4 \$1,$c,$t2+ vextracti32x4 \$1,$d,$t3++ sub \$64,$len+ jb .Ltail_avx512++ vpxor 0x00($inp),$t0,$t0 # xor with input+ vpxor 0x10($inp),$t1,$t1+ vpxor 0x20($inp),$t2,$t2+ vpxor 0x30($inp),$t3,$t3+ lea 0x40($inp),$inp # inp+=64++ vmovdqu $t0,0x00($out) # write output+ vmovdqu $t1,0x10($out)+ vmovdqu $t2,0x20($out)+ vmovdqu $t3,0x30($out)+ lea 0x40($out),$out # out+=64++ jz .Ldone_avx512++ vextracti32x4 \$2,$a,$t0+ vextracti32x4 \$2,$b,$t1+ vextracti32x4 \$2,$c,$t2+ vextracti32x4 \$2,$d,$t3++ sub \$64,$len+ jb .Ltail_avx512++ vpxor 0x00($inp),$t0,$t0 # xor with input+ vpxor 0x10($inp),$t1,$t1+ vpxor 0x20($inp),$t2,$t2+ vpxor 0x30($inp),$t3,$t3+ lea 0x40($inp),$inp # inp+=64++ vmovdqu $t0,0x00($out) # write output+ vmovdqu $t1,0x10($out)+ vmovdqu $t2,0x20($out)+ vmovdqu $t3,0x30($out)+ lea 0x40($out),$out # out+=64++ jz .Ldone_avx512++ vextracti32x4 \$3,$a,$t0+ vextracti32x4 \$3,$b,$t1+ vextracti32x4 \$3,$c,$t2+ vextracti32x4 \$3,$d,$t3++ sub \$64,$len+ jb .Ltail_avx512++ vmovdqa32 $a_,$a+ vpaddd $fourz,$d_,$d++ vpxor 0x00($inp),$t0,$t0 # xor with input+ vpxor 0x10($inp),$t1,$t1+ vpxor 0x20($inp),$t2,$t2+ vpxor 0x30($inp),$t3,$t3+ lea 0x40($inp),$inp # inp+=64++ vmovdqu $t0,0x00($out) # write output+ vmovdqu $t1,0x10($out)+ vmovdqu $t2,0x20($out)+ vmovdqu $t3,0x30($out)+ lea 0x40($out),$out # out+=64++ jnz .Loop_outer_avx512++ jmp .Ldone_avx512++.align 16+.Ltail64_avx512:+ vmovdqa %x#$a,0x00(%rsp)+ vmovdqa %x#$b,0x10(%rsp)+ vmovdqa %x#$c,0x20(%rsp)+ vmovdqa %x#$d,0x30(%rsp)+ add \$64,$len+ jmp .Loop_tail_avx512++.align 16+.Ltail_avx512:+ vmovdqa $t0,0x00(%rsp)+ vmovdqa $t1,0x10(%rsp)+ vmovdqa $t2,0x20(%rsp)+ vmovdqa $t3,0x30(%rsp)+ add \$64,$len++.Loop_tail_avx512:+ movzb ($inp,$counter),%eax+ movzb (%rsp,$counter),%ecx+ lea 1($counter),$counter+ xor %ecx,%eax+ mov %al,-1($out,$counter)+ dec $len+ jnz .Loop_tail_avx512++ vmovdqu32 $a_,0x00(%rsp)++.Ldone_avx512:+ vzeroall+___+$code.=<<___ if ($win64);+ movaps -0x28(%r10),%xmm6+ movaps -0x18(%r10),%xmm7+___+$code.=<<___;+ lea (%r10),%rsp+.cfi_def_cfa_register %rsp+.Lavx512_epilogue:+ ret+.cfi_endproc+.size ChaCha20_avx512,.-ChaCha20_avx512+___++map(s/%z/%y/, $a,$b,$c,$d, $a_,$b_,$c_,$d_,$fourz);++$code.=<<___ if ($flavour =~ /kernel/);+.globl ChaCha20_avx512vl+___+$code.=<<___;+.type ChaCha20_avx512vl,\@function,5+.align 32+ChaCha20_avx512vl:+.cfi_startproc+.LChaCha20_avx512vl:+ mov %rsp,%r10 # frame pointer+.cfi_def_cfa_register %r10+ cmp \$128,$len+ ja .LChaCha20_8xvl++ sub \$64+$xframe,%rsp+___+$code.=<<___ if ($win64);+ movaps %xmm6,-0x28(%r10)+ movaps %xmm7,-0x18(%r10)+.Lavx512vl_body:+___+$code.=<<___;+ vbroadcasti32x4 .Lsigma(%rip),$a+ vbroadcasti32x4 ($key),$b_+ vbroadcasti32x4 16($key),$c_+ vbroadcasti32x4 ($counter),$d_++ vmovdqa32 $a,$a_+ vmovdqa32 .Ltwoy(%rip),$fourz+ vpaddd .Lzeroz(%rip),$d_,$d+ jmp .Loop_outer_avx512vl++.align 32+.Loop_outer_avx512vl:+ vmovdqa32 $b_,$b+ vmovdqa32 $c_,$c+ vmovdqa32 $d,$d_+ mov \$10,$counter # reuse $counter+ jmp .Loop_avx512vl++.align 32+.Loop_avx512vl:+___+ &AVX512ROUND();+ &vpshufd ($c,$c,0b01001110);+ &vpshufd ($b,$b,0b00111001);+ &vpshufd ($d,$d,0b10010011);++ &AVX512ROUND();+ &vpshufd ($c,$c,0b01001110);+ &vpshufd ($b,$b,0b10010011);+ &vpshufd ($d,$d,0b00111001);++ &sub ($counter,1);+ &jnz (".Loop_avx512vl");++$code.=<<___;+ vpaddd $a_,$a,$a+ vpaddd $b_,$b,$b+ vpaddd $c_,$c,$c+ vpaddd $d_,$d,$d++ sub \$64,$len+ jb .Ltail64_avx512vl++ vpxor 0x00($inp),%x#$a,$t0 # xor with input+ vpxor 0x10($inp),%x#$b,$t1+ vpxor 0x20($inp),%x#$c,$t2+ vpxor 0x30($inp),%x#$d,$t3+ lea 0x40($inp),$inp # inp+=64++ vmovdqu $t0,0x00($out) # write output+ vmovdqu $t1,0x10($out)+ vmovdqu $t2,0x20($out)+ vmovdqu $t3,0x30($out)+ lea 0x40($out),$out # out+=64++ jz .Ldone_avx512vl++ vextracti128 \$1,$a,$t0+ vextracti128 \$1,$b,$t1+ vextracti128 \$1,$c,$t2+ vextracti128 \$1,$d,$t3++ sub \$64,$len+ jb .Ltail_avx512vl++ vmovdqa32 $a_,$a+ vpaddd $fourz,$d_,$d++ vpxor 0x00($inp),$t0,$t0 # xor with input+ vpxor 0x10($inp),$t1,$t1+ vpxor 0x20($inp),$t2,$t2+ vpxor 0x30($inp),$t3,$t3+ lea 0x40($inp),$inp # inp+=64++ vmovdqu $t0,0x00($out) # write output+ vmovdqu $t1,0x10($out)+ vmovdqu $t2,0x20($out)+ vmovdqu $t3,0x30($out)+ lea 0x40($out),$out # out+=64++ jnz .Loop_outer_avx512vl++ jmp .Ldone_avx512vl++.align 16+.Ltail64_avx512vl:+ vmovdqa %x#$a,0x00(%rsp)+ vmovdqa %x#$b,0x10(%rsp)+ vmovdqa %x#$c,0x20(%rsp)+ vmovdqa %x#$d,0x30(%rsp)+ add \$64,$len+ jmp .Loop_tail_avx512vl++.align 16+.Ltail_avx512vl:+ vmovdqa $t0,0x00(%rsp)+ vmovdqa $t1,0x10(%rsp)+ vmovdqa $t2,0x20(%rsp)+ vmovdqa $t3,0x30(%rsp)+ add \$64,$len++.Loop_tail_avx512vl:+ movzb ($inp,$counter),%eax+ movzb (%rsp,$counter),%ecx+ lea 1($counter),$counter+ xor %ecx,%eax+ mov %al,-1($out,$counter)+ dec $len+ jnz .Loop_tail_avx512vl++ vmovdqu32 $a_,0x00(%rsp)+ vmovdqu32 $a_,0x20(%rsp)++.Ldone_avx512vl:+ vzeroall+___+$code.=<<___ if ($win64);+ movaps -0x28(%r10),%xmm6+ movaps -0x18(%r10),%xmm7+___+$code.=<<___;+ lea (%r10),%rsp+.cfi_def_cfa_register %rsp+.Lavx512vl_epilogue:+ ret+.cfi_endproc+.size ChaCha20_avx512vl,.-ChaCha20_avx512vl+___+}+if ($avx>2) {+# This one handles longer inputs...++my ($xa0,$xa1,$xa2,$xa3, $xb0,$xb1,$xb2,$xb3,+ $xc0,$xc1,$xc2,$xc3, $xd0,$xd1,$xd2,$xd3)=map("%zmm$_",(0..15));+my @xx=($xa0,$xa1,$xa2,$xa3, $xb0,$xb1,$xb2,$xb3,+ $xc0,$xc1,$xc2,$xc3, $xd0,$xd1,$xd2,$xd3);+my @key=map("%zmm$_",(16..31));+my ($xt0,$xt1,$xt2,$xt3)=@key[0..3];++sub AVX512_lane_ROUND {+my ($a0,$b0,$c0,$d0)=@_;+my ($a1,$b1,$c1,$d1)=map(($_&~3)+(($_+1)&3),($a0,$b0,$c0,$d0));+my ($a2,$b2,$c2,$d2)=map(($_&~3)+(($_+1)&3),($a1,$b1,$c1,$d1));+my ($a3,$b3,$c3,$d3)=map(($_&~3)+(($_+1)&3),($a2,$b2,$c2,$d2));+my @x=map("\"$_\"",@xx);++ (+ "&vpaddd (@x[$a0],@x[$a0],@x[$b0])", # Q1+ "&vpaddd (@x[$a1],@x[$a1],@x[$b1])", # Q2+ "&vpaddd (@x[$a2],@x[$a2],@x[$b2])", # Q3+ "&vpaddd (@x[$a3],@x[$a3],@x[$b3])", # Q4+ "&vpxord (@x[$d0],@x[$d0],@x[$a0])",+ "&vpxord (@x[$d1],@x[$d1],@x[$a1])",+ "&vpxord (@x[$d2],@x[$d2],@x[$a2])",+ "&vpxord (@x[$d3],@x[$d3],@x[$a3])",+ "&vprold (@x[$d0],@x[$d0],16)",+ "&vprold (@x[$d1],@x[$d1],16)",+ "&vprold (@x[$d2],@x[$d2],16)",+ "&vprold (@x[$d3],@x[$d3],16)",++ "&vpaddd (@x[$c0],@x[$c0],@x[$d0])",+ "&vpaddd (@x[$c1],@x[$c1],@x[$d1])",+ "&vpaddd (@x[$c2],@x[$c2],@x[$d2])",+ "&vpaddd (@x[$c3],@x[$c3],@x[$d3])",+ "&vpxord (@x[$b0],@x[$b0],@x[$c0])",+ "&vpxord (@x[$b1],@x[$b1],@x[$c1])",+ "&vpxord (@x[$b2],@x[$b2],@x[$c2])",+ "&vpxord (@x[$b3],@x[$b3],@x[$c3])",+ "&vprold (@x[$b0],@x[$b0],12)",+ "&vprold (@x[$b1],@x[$b1],12)",+ "&vprold (@x[$b2],@x[$b2],12)",+ "&vprold (@x[$b3],@x[$b3],12)",++ "&vpaddd (@x[$a0],@x[$a0],@x[$b0])",+ "&vpaddd (@x[$a1],@x[$a1],@x[$b1])",+ "&vpaddd (@x[$a2],@x[$a2],@x[$b2])",+ "&vpaddd (@x[$a3],@x[$a3],@x[$b3])",+ "&vpxord (@x[$d0],@x[$d0],@x[$a0])",+ "&vpxord (@x[$d1],@x[$d1],@x[$a1])",+ "&vpxord (@x[$d2],@x[$d2],@x[$a2])",+ "&vpxord (@x[$d3],@x[$d3],@x[$a3])",+ "&vprold (@x[$d0],@x[$d0],8)",+ "&vprold (@x[$d1],@x[$d1],8)",+ "&vprold (@x[$d2],@x[$d2],8)",+ "&vprold (@x[$d3],@x[$d3],8)",++ "&vpaddd (@x[$c0],@x[$c0],@x[$d0])",+ "&vpaddd (@x[$c1],@x[$c1],@x[$d1])",+ "&vpaddd (@x[$c2],@x[$c2],@x[$d2])",+ "&vpaddd (@x[$c3],@x[$c3],@x[$d3])",+ "&vpxord (@x[$b0],@x[$b0],@x[$c0])",+ "&vpxord (@x[$b1],@x[$b1],@x[$c1])",+ "&vpxord (@x[$b2],@x[$b2],@x[$c2])",+ "&vpxord (@x[$b3],@x[$b3],@x[$c3])",+ "&vprold (@x[$b0],@x[$b0],7)",+ "&vprold (@x[$b1],@x[$b1],7)",+ "&vprold (@x[$b2],@x[$b2],7)",+ "&vprold (@x[$b3],@x[$b3],7)"+ );+}++my $xframe = $win64 ? 0xa8 : 8;++$code.=<<___;+.type ChaCha20_16x,\@function,5+.align 32+ChaCha20_16x:+.cfi_startproc+.LChaCha20_16x:+ mov %rsp,%r10 # frame register+.cfi_def_cfa_register %r10+ sub \$64+$xframe,%rsp+ and \$-64,%rsp+___+$code.=<<___ if ($win64);+ movaps %xmm6,-0xa8(%r10)+ movaps %xmm7,-0x98(%r10)+ movaps %xmm8,-0x88(%r10)+ movaps %xmm9,-0x78(%r10)+ movaps %xmm10,-0x68(%r10)+ movaps %xmm11,-0x58(%r10)+ movaps %xmm12,-0x48(%r10)+ movaps %xmm13,-0x38(%r10)+ movaps %xmm14,-0x28(%r10)+ movaps %xmm15,-0x18(%r10)+.L16x_body:+___+$code.=<<___;+ vzeroupper++ lea .Lsigma(%rip),%r9+ vbroadcasti32x4 (%r9),$xa3 # key[0]+ vbroadcasti32x4 ($key),$xb3 # key[1]+ vbroadcasti32x4 16($key),$xc3 # key[2]+ vbroadcasti32x4 ($counter),$xd3 # key[3]++ vpshufd \$0x00,$xa3,$xa0 # smash key by lanes...+ vpshufd \$0x55,$xa3,$xa1+ vpshufd \$0xaa,$xa3,$xa2+ vpshufd \$0xff,$xa3,$xa3+ vmovdqa64 $xa0,@key[0]+ vmovdqa64 $xa1,@key[1]+ vmovdqa64 $xa2,@key[2]+ vmovdqa64 $xa3,@key[3]++ vpshufd \$0x00,$xb3,$xb0+ vpshufd \$0x55,$xb3,$xb1+ vpshufd \$0xaa,$xb3,$xb2+ vpshufd \$0xff,$xb3,$xb3+ vmovdqa64 $xb0,@key[4]+ vmovdqa64 $xb1,@key[5]+ vmovdqa64 $xb2,@key[6]+ vmovdqa64 $xb3,@key[7]++ vpshufd \$0x00,$xc3,$xc0+ vpshufd \$0x55,$xc3,$xc1+ vpshufd \$0xaa,$xc3,$xc2+ vpshufd \$0xff,$xc3,$xc3+ vmovdqa64 $xc0,@key[8]+ vmovdqa64 $xc1,@key[9]+ vmovdqa64 $xc2,@key[10]+ vmovdqa64 $xc3,@key[11]++ vpshufd \$0x00,$xd3,$xd0+ vpshufd \$0x55,$xd3,$xd1+ vpshufd \$0xaa,$xd3,$xd2+ vpshufd \$0xff,$xd3,$xd3+ vpaddd .Lincz(%rip),$xd0,$xd0 # don't save counters yet+ vmovdqa64 $xd0,@key[12]+ vmovdqa64 $xd1,@key[13]+ vmovdqa64 $xd2,@key[14]+ vmovdqa64 $xd3,@key[15]++ mov \$10,%eax+ jmp .Loop16x++.align 32+.Loop_outer16x:+ vpbroadcastd 0(%r9),$xa0 # reload key+ vpbroadcastd 4(%r9),$xa1+ vpbroadcastd 8(%r9),$xa2+ vpbroadcastd 12(%r9),$xa3+ vpaddd .Lsixteen(%rip),@key[12],@key[12] # next SIMD counters+ vmovdqa64 @key[4],$xb0+ vmovdqa64 @key[5],$xb1+ vmovdqa64 @key[6],$xb2+ vmovdqa64 @key[7],$xb3+ vmovdqa64 @key[8],$xc0+ vmovdqa64 @key[9],$xc1+ vmovdqa64 @key[10],$xc2+ vmovdqa64 @key[11],$xc3+ vmovdqa64 @key[12],$xd0+ vmovdqa64 @key[13],$xd1+ vmovdqa64 @key[14],$xd2+ vmovdqa64 @key[15],$xd3++ vmovdqa64 $xa0,@key[0]+ vmovdqa64 $xa1,@key[1]+ vmovdqa64 $xa2,@key[2]+ vmovdqa64 $xa3,@key[3]++ mov \$10,%eax+ jmp .Loop16x++.align 32+.Loop16x:+___+ foreach (&AVX512_lane_ROUND(0, 4, 8,12)) { eval; }+ foreach (&AVX512_lane_ROUND(0, 5,10,15)) { eval; }+$code.=<<___;+ dec %eax+ jnz .Loop16x++ vpaddd @key[0],$xa0,$xa0 # accumulate key+ vpaddd @key[1],$xa1,$xa1+ vpaddd @key[2],$xa2,$xa2+ vpaddd @key[3],$xa3,$xa3++ vpunpckldq $xa1,$xa0,$xt2 # "de-interlace" data+ vpunpckldq $xa3,$xa2,$xt3+ vpunpckhdq $xa1,$xa0,$xa0+ vpunpckhdq $xa3,$xa2,$xa2+ vpunpcklqdq $xt3,$xt2,$xa1 # "a0"+ vpunpckhqdq $xt3,$xt2,$xt2 # "a1"+ vpunpcklqdq $xa2,$xa0,$xa3 # "a2"+ vpunpckhqdq $xa2,$xa0,$xa0 # "a3"+___+ ($xa0,$xa1,$xa2,$xa3,$xt2)=($xa1,$xt2,$xa3,$xa0,$xa2);+$code.=<<___;+ vpaddd @key[4],$xb0,$xb0+ vpaddd @key[5],$xb1,$xb1+ vpaddd @key[6],$xb2,$xb2+ vpaddd @key[7],$xb3,$xb3++ vpunpckldq $xb1,$xb0,$xt2+ vpunpckldq $xb3,$xb2,$xt3+ vpunpckhdq $xb1,$xb0,$xb0+ vpunpckhdq $xb3,$xb2,$xb2+ vpunpcklqdq $xt3,$xt2,$xb1 # "b0"+ vpunpckhqdq $xt3,$xt2,$xt2 # "b1"+ vpunpcklqdq $xb2,$xb0,$xb3 # "b2"+ vpunpckhqdq $xb2,$xb0,$xb0 # "b3"+___+ ($xb0,$xb1,$xb2,$xb3,$xt2)=($xb1,$xt2,$xb3,$xb0,$xb2);+$code.=<<___;+ vshufi32x4 \$0x44,$xb0,$xa0,$xt3 # "de-interlace" further+ vshufi32x4 \$0xee,$xb0,$xa0,$xb0+ vshufi32x4 \$0x44,$xb1,$xa1,$xa0+ vshufi32x4 \$0xee,$xb1,$xa1,$xb1+ vshufi32x4 \$0x44,$xb2,$xa2,$xa1+ vshufi32x4 \$0xee,$xb2,$xa2,$xb2+ vshufi32x4 \$0x44,$xb3,$xa3,$xa2+ vshufi32x4 \$0xee,$xb3,$xa3,$xb3+___+ ($xa0,$xa1,$xa2,$xa3,$xt3)=($xt3,$xa0,$xa1,$xa2,$xa3);+$code.=<<___;+ vpaddd @key[8],$xc0,$xc0+ vpaddd @key[9],$xc1,$xc1+ vpaddd @key[10],$xc2,$xc2+ vpaddd @key[11],$xc3,$xc3++ vpunpckldq $xc1,$xc0,$xt2+ vpunpckldq $xc3,$xc2,$xt3+ vpunpckhdq $xc1,$xc0,$xc0+ vpunpckhdq $xc3,$xc2,$xc2+ vpunpcklqdq $xt3,$xt2,$xc1 # "c0"+ vpunpckhqdq $xt3,$xt2,$xt2 # "c1"+ vpunpcklqdq $xc2,$xc0,$xc3 # "c2"+ vpunpckhqdq $xc2,$xc0,$xc0 # "c3"+___+ ($xc0,$xc1,$xc2,$xc3,$xt2)=($xc1,$xt2,$xc3,$xc0,$xc2);+$code.=<<___;+ vpaddd @key[12],$xd0,$xd0+ vpaddd @key[13],$xd1,$xd1+ vpaddd @key[14],$xd2,$xd2+ vpaddd @key[15],$xd3,$xd3++ vpunpckldq $xd1,$xd0,$xt2+ vpunpckldq $xd3,$xd2,$xt3+ vpunpckhdq $xd1,$xd0,$xd0+ vpunpckhdq $xd3,$xd2,$xd2+ vpunpcklqdq $xt3,$xt2,$xd1 # "d0"+ vpunpckhqdq $xt3,$xt2,$xt2 # "d1"+ vpunpcklqdq $xd2,$xd0,$xd3 # "d2"+ vpunpckhqdq $xd2,$xd0,$xd0 # "d3"+___+ ($xd0,$xd1,$xd2,$xd3,$xt2)=($xd1,$xt2,$xd3,$xd0,$xd2);+$code.=<<___;+ vshufi32x4 \$0x44,$xd0,$xc0,$xt3 # "de-interlace" further+ vshufi32x4 \$0xee,$xd0,$xc0,$xd0+ vshufi32x4 \$0x44,$xd1,$xc1,$xc0+ vshufi32x4 \$0xee,$xd1,$xc1,$xd1+ vshufi32x4 \$0x44,$xd2,$xc2,$xc1+ vshufi32x4 \$0xee,$xd2,$xc2,$xd2+ vshufi32x4 \$0x44,$xd3,$xc3,$xc2+ vshufi32x4 \$0xee,$xd3,$xc3,$xd3+___+ ($xc0,$xc1,$xc2,$xc3,$xt3)=($xt3,$xc0,$xc1,$xc2,$xc3);+$code.=<<___;+ vshufi32x4 \$0x88,$xc0,$xa0,$xt0 # "de-interlace" further+ vshufi32x4 \$0xdd,$xc0,$xa0,$xa0+ vshufi32x4 \$0x88,$xd0,$xb0,$xc0+ vshufi32x4 \$0xdd,$xd0,$xb0,$xd0+ vshufi32x4 \$0x88,$xc1,$xa1,$xt1+ vshufi32x4 \$0xdd,$xc1,$xa1,$xa1+ vshufi32x4 \$0x88,$xd1,$xb1,$xc1+ vshufi32x4 \$0xdd,$xd1,$xb1,$xd1+ vshufi32x4 \$0x88,$xc2,$xa2,$xt2+ vshufi32x4 \$0xdd,$xc2,$xa2,$xa2+ vshufi32x4 \$0x88,$xd2,$xb2,$xc2+ vshufi32x4 \$0xdd,$xd2,$xb2,$xd2+ vshufi32x4 \$0x88,$xc3,$xa3,$xt3+ vshufi32x4 \$0xdd,$xc3,$xa3,$xa3+ vshufi32x4 \$0x88,$xd3,$xb3,$xc3+ vshufi32x4 \$0xdd,$xd3,$xb3,$xd3+___+ ($xa0,$xa1,$xa2,$xa3,$xb0,$xb1,$xb2,$xb3)=+ ($xt0,$xt1,$xt2,$xt3,$xa0,$xa1,$xa2,$xa3);++ ($xa0,$xb0,$xc0,$xd0, $xa1,$xb1,$xc1,$xd1,+ $xa2,$xb2,$xc2,$xd2, $xa3,$xb3,$xc3,$xd3) =+ ($xa0,$xa1,$xa2,$xa3, $xb0,$xb1,$xb2,$xb3,+ $xc0,$xc1,$xc2,$xc3, $xd0,$xd1,$xd2,$xd3);+$code.=<<___;+ cmp \$64*16,$len+ jb .Ltail16x++ vpxord 0x00($inp),$xa0,$xa0 # xor with input+ vpxord 0x40($inp),$xb0,$xb0+ vpxord 0x80($inp),$xc0,$xc0+ vpxord 0xc0($inp),$xd0,$xd0+ vmovdqu32 $xa0,0x00($out)+ vmovdqu32 $xb0,0x40($out)+ vmovdqu32 $xc0,0x80($out)+ vmovdqu32 $xd0,0xc0($out)++ vpxord 0x100($inp),$xa1,$xa1+ vpxord 0x140($inp),$xb1,$xb1+ vpxord 0x180($inp),$xc1,$xc1+ vpxord 0x1c0($inp),$xd1,$xd1+ vmovdqu32 $xa1,0x100($out)+ vmovdqu32 $xb1,0x140($out)+ vmovdqu32 $xc1,0x180($out)+ vmovdqu32 $xd1,0x1c0($out)++ vpxord 0x200($inp),$xa2,$xa2+ vpxord 0x240($inp),$xb2,$xb2+ vpxord 0x280($inp),$xc2,$xc2+ vpxord 0x2c0($inp),$xd2,$xd2+ vmovdqu32 $xa2,0x200($out)+ vmovdqu32 $xb2,0x240($out)+ vmovdqu32 $xc2,0x280($out)+ vmovdqu32 $xd2,0x2c0($out)++ vpxord 0x300($inp),$xa3,$xa3+ vpxord 0x340($inp),$xb3,$xb3+ vpxord 0x380($inp),$xc3,$xc3+ vpxord 0x3c0($inp),$xd3,$xd3+ lea 0x400($inp),$inp+ vmovdqu32 $xa3,0x300($out)+ vmovdqu32 $xb3,0x340($out)+ vmovdqu32 $xc3,0x380($out)+ vmovdqu32 $xd3,0x3c0($out)+ lea 0x400($out),$out++ sub \$64*16,$len+ jnz .Loop_outer16x++ jmp .Ldone16x++.align 32+.Ltail16x:+ xor %r9,%r9+ sub $inp,$out+ cmp \$64*1,$len+ jb .Less_than_64_16x+ vpxord ($inp),$xa0,$xa0 # xor with input+ vmovdqu32 $xa0,($out,$inp)+ je .Ldone16x+ vmovdqa32 $xb0,$xa0+ lea 64($inp),$inp++ cmp \$64*2,$len+ jb .Less_than_64_16x+ vpxord ($inp),$xb0,$xb0+ vmovdqu32 $xb0,($out,$inp)+ je .Ldone16x+ vmovdqa32 $xc0,$xa0+ lea 64($inp),$inp++ cmp \$64*3,$len+ jb .Less_than_64_16x+ vpxord ($inp),$xc0,$xc0+ vmovdqu32 $xc0,($out,$inp)+ je .Ldone16x+ vmovdqa32 $xd0,$xa0+ lea 64($inp),$inp++ cmp \$64*4,$len+ jb .Less_than_64_16x+ vpxord ($inp),$xd0,$xd0+ vmovdqu32 $xd0,($out,$inp)+ je .Ldone16x+ vmovdqa32 $xa1,$xa0+ lea 64($inp),$inp++ cmp \$64*5,$len+ jb .Less_than_64_16x+ vpxord ($inp),$xa1,$xa1+ vmovdqu32 $xa1,($out,$inp)+ je .Ldone16x+ vmovdqa32 $xb1,$xa0+ lea 64($inp),$inp++ cmp \$64*6,$len+ jb .Less_than_64_16x+ vpxord ($inp),$xb1,$xb1+ vmovdqu32 $xb1,($out,$inp)+ je .Ldone16x+ vmovdqa32 $xc1,$xa0+ lea 64($inp),$inp++ cmp \$64*7,$len+ jb .Less_than_64_16x+ vpxord ($inp),$xc1,$xc1+ vmovdqu32 $xc1,($out,$inp)+ je .Ldone16x+ vmovdqa32 $xd1,$xa0+ lea 64($inp),$inp++ cmp \$64*8,$len+ jb .Less_than_64_16x+ vpxord ($inp),$xd1,$xd1+ vmovdqu32 $xd1,($out,$inp)+ je .Ldone16x+ vmovdqa32 $xa2,$xa0+ lea 64($inp),$inp++ cmp \$64*9,$len+ jb .Less_than_64_16x+ vpxord ($inp),$xa2,$xa2+ vmovdqu32 $xa2,($out,$inp)+ je .Ldone16x+ vmovdqa32 $xb2,$xa0+ lea 64($inp),$inp++ cmp \$64*10,$len+ jb .Less_than_64_16x+ vpxord ($inp),$xb2,$xb2+ vmovdqu32 $xb2,($out,$inp)+ je .Ldone16x+ vmovdqa32 $xc2,$xa0+ lea 64($inp),$inp++ cmp \$64*11,$len+ jb .Less_than_64_16x+ vpxord ($inp),$xc2,$xc2+ vmovdqu32 $xc2,($out,$inp)+ je .Ldone16x+ vmovdqa32 $xd2,$xa0+ lea 64($inp),$inp++ cmp \$64*12,$len+ jb .Less_than_64_16x+ vpxord ($inp),$xd2,$xd2+ vmovdqu32 $xd2,($out,$inp)+ je .Ldone16x+ vmovdqa32 $xa3,$xa0+ lea 64($inp),$inp++ cmp \$64*13,$len+ jb .Less_than_64_16x+ vpxord ($inp),$xa3,$xa3+ vmovdqu32 $xa3,($out,$inp)+ je .Ldone16x+ vmovdqa32 $xb3,$xa0+ lea 64($inp),$inp++ cmp \$64*14,$len+ jb .Less_than_64_16x+ vpxord ($inp),$xb3,$xb3+ vmovdqu32 $xb3,($out,$inp)+ je .Ldone16x+ vmovdqa32 $xc3,$xa0+ lea 64($inp),$inp++ cmp \$64*15,$len+ jb .Less_than_64_16x+ vpxord ($inp),$xc3,$xc3+ vmovdqu32 $xc3,($out,$inp)+ je .Ldone16x+ vmovdqa32 $xd3,$xa0+ lea 64($inp),$inp++.Less_than_64_16x:+ vmovdqa32 $xa0,0x00(%rsp)+ lea ($out,$inp),$out+ and \$63,$len++.Loop_tail16x:+ movzb ($inp,%r9),%eax+ movzb (%rsp,%r9),%ecx+ lea 1(%r9),%r9+ xor %ecx,%eax+ mov %al,-1($out,%r9)+ dec $len+ jnz .Loop_tail16x++ vpxord $xa0,$xa0,$xa0+ vmovdqa32 $xa0,0(%rsp)++.Ldone16x:+ vzeroall+___+$code.=<<___ if ($win64);+ movaps -0xa8(%r10),%xmm6+ movaps -0x98(%r10),%xmm7+ movaps -0x88(%r10),%xmm8+ movaps -0x78(%r10),%xmm9+ movaps -0x68(%r10),%xmm10+ movaps -0x58(%r10),%xmm11+ movaps -0x48(%r10),%xmm12+ movaps -0x38(%r10),%xmm13+ movaps -0x28(%r10),%xmm14+ movaps -0x18(%r10),%xmm15+___+$code.=<<___;+ lea (%r10),%rsp+.cfi_def_cfa_register %rsp+.L16x_epilogue:+ ret+.cfi_endproc+.size ChaCha20_16x,.-ChaCha20_16x+___++# switch to %ymm domain+($xa0,$xa1,$xa2,$xa3, $xb0,$xb1,$xb2,$xb3,+ $xc0,$xc1,$xc2,$xc3, $xd0,$xd1,$xd2,$xd3)=map("%ymm$_",(0..15));+@xx=($xa0,$xa1,$xa2,$xa3, $xb0,$xb1,$xb2,$xb3,+ $xc0,$xc1,$xc2,$xc3, $xd0,$xd1,$xd2,$xd3);+@key=map("%ymm$_",(16..31));+($xt0,$xt1,$xt2,$xt3)=@key[0..3];++$code.=<<___;+.type ChaCha20_8xvl,\@function,5+.align 32+ChaCha20_8xvl:+.cfi_startproc+.LChaCha20_8xvl:+ mov %rsp,%r10 # frame register+.cfi_def_cfa_register %r10+ sub \$64+$xframe,%rsp+ and \$-64,%rsp+___+$code.=<<___ if ($win64);+ movaps %xmm6,-0xa8(%r10)+ movaps %xmm7,-0x98(%r10)+ movaps %xmm8,-0x88(%r10)+ movaps %xmm9,-0x78(%r10)+ movaps %xmm10,-0x68(%r10)+ movaps %xmm11,-0x58(%r10)+ movaps %xmm12,-0x48(%r10)+ movaps %xmm13,-0x38(%r10)+ movaps %xmm14,-0x28(%r10)+ movaps %xmm15,-0x18(%r10)+.L8xvl_body:+___+$code.=<<___;+ vzeroupper++ lea .Lsigma(%rip),%r9+ vbroadcasti128 (%r9),$xa3 # key[0]+ vbroadcasti128 ($key),$xb3 # key[1]+ vbroadcasti128 16($key),$xc3 # key[2]+ vbroadcasti128 ($counter),$xd3 # key[3]++ vpshufd \$0x00,$xa3,$xa0 # smash key by lanes...+ vpshufd \$0x55,$xa3,$xa1+ vpshufd \$0xaa,$xa3,$xa2+ vpshufd \$0xff,$xa3,$xa3+ vmovdqa64 $xa0,@key[0]+ vmovdqa64 $xa1,@key[1]+ vmovdqa64 $xa2,@key[2]+ vmovdqa64 $xa3,@key[3]++ vpshufd \$0x00,$xb3,$xb0+ vpshufd \$0x55,$xb3,$xb1+ vpshufd \$0xaa,$xb3,$xb2+ vpshufd \$0xff,$xb3,$xb3+ vmovdqa64 $xb0,@key[4]+ vmovdqa64 $xb1,@key[5]+ vmovdqa64 $xb2,@key[6]+ vmovdqa64 $xb3,@key[7]++ vpshufd \$0x00,$xc3,$xc0+ vpshufd \$0x55,$xc3,$xc1+ vpshufd \$0xaa,$xc3,$xc2+ vpshufd \$0xff,$xc3,$xc3+ vmovdqa64 $xc0,@key[8]+ vmovdqa64 $xc1,@key[9]+ vmovdqa64 $xc2,@key[10]+ vmovdqa64 $xc3,@key[11]++ vpshufd \$0x00,$xd3,$xd0+ vpshufd \$0x55,$xd3,$xd1+ vpshufd \$0xaa,$xd3,$xd2+ vpshufd \$0xff,$xd3,$xd3+ vpaddd .Lincy(%rip),$xd0,$xd0 # don't save counters yet+ vmovdqa64 $xd0,@key[12]+ vmovdqa64 $xd1,@key[13]+ vmovdqa64 $xd2,@key[14]+ vmovdqa64 $xd3,@key[15]++ mov \$10,%eax+ jmp .Loop8xvl++.align 32+.Loop_outer8xvl:+ #vpbroadcastd 0(%r9),$xa0 # reload key+ #vpbroadcastd 4(%r9),$xa1+ vpbroadcastd 8(%r9),$xa2+ vpbroadcastd 12(%r9),$xa3+ vpaddd .Leight(%rip),@key[12],@key[12] # next SIMD counters+ vmovdqa64 @key[4],$xb0+ vmovdqa64 @key[5],$xb1+ vmovdqa64 @key[6],$xb2+ vmovdqa64 @key[7],$xb3+ vmovdqa64 @key[8],$xc0+ vmovdqa64 @key[9],$xc1+ vmovdqa64 @key[10],$xc2+ vmovdqa64 @key[11],$xc3+ vmovdqa64 @key[12],$xd0+ vmovdqa64 @key[13],$xd1+ vmovdqa64 @key[14],$xd2+ vmovdqa64 @key[15],$xd3++ vmovdqa64 $xa0,@key[0]+ vmovdqa64 $xa1,@key[1]+ vmovdqa64 $xa2,@key[2]+ vmovdqa64 $xa3,@key[3]++ mov \$10,%eax+ jmp .Loop8xvl++.align 32+.Loop8xvl:+___+ foreach (&AVX512_lane_ROUND(0, 4, 8,12)) { eval; }+ foreach (&AVX512_lane_ROUND(0, 5,10,15)) { eval; }+$code.=<<___;+ dec %eax+ jnz .Loop8xvl++ vpaddd @key[0],$xa0,$xa0 # accumulate key+ vpaddd @key[1],$xa1,$xa1+ vpaddd @key[2],$xa2,$xa2+ vpaddd @key[3],$xa3,$xa3++ vpunpckldq $xa1,$xa0,$xt2 # "de-interlace" data+ vpunpckldq $xa3,$xa2,$xt3+ vpunpckhdq $xa1,$xa0,$xa0+ vpunpckhdq $xa3,$xa2,$xa2+ vpunpcklqdq $xt3,$xt2,$xa1 # "a0"+ vpunpckhqdq $xt3,$xt2,$xt2 # "a1"+ vpunpcklqdq $xa2,$xa0,$xa3 # "a2"+ vpunpckhqdq $xa2,$xa0,$xa0 # "a3"+___+ ($xa0,$xa1,$xa2,$xa3,$xt2)=($xa1,$xt2,$xa3,$xa0,$xa2);+$code.=<<___;+ vpaddd @key[4],$xb0,$xb0+ vpaddd @key[5],$xb1,$xb1+ vpaddd @key[6],$xb2,$xb2+ vpaddd @key[7],$xb3,$xb3++ vpunpckldq $xb1,$xb0,$xt2+ vpunpckldq $xb3,$xb2,$xt3+ vpunpckhdq $xb1,$xb0,$xb0+ vpunpckhdq $xb3,$xb2,$xb2+ vpunpcklqdq $xt3,$xt2,$xb1 # "b0"+ vpunpckhqdq $xt3,$xt2,$xt2 # "b1"+ vpunpcklqdq $xb2,$xb0,$xb3 # "b2"+ vpunpckhqdq $xb2,$xb0,$xb0 # "b3"+___+ ($xb0,$xb1,$xb2,$xb3,$xt2)=($xb1,$xt2,$xb3,$xb0,$xb2);+$code.=<<___;+ vshufi32x4 \$0,$xb0,$xa0,$xt3 # "de-interlace" further+ vshufi32x4 \$3,$xb0,$xa0,$xb0+ vshufi32x4 \$0,$xb1,$xa1,$xa0+ vshufi32x4 \$3,$xb1,$xa1,$xb1+ vshufi32x4 \$0,$xb2,$xa2,$xa1+ vshufi32x4 \$3,$xb2,$xa2,$xb2+ vshufi32x4 \$0,$xb3,$xa3,$xa2+ vshufi32x4 \$3,$xb3,$xa3,$xb3+___+ ($xa0,$xa1,$xa2,$xa3,$xt3)=($xt3,$xa0,$xa1,$xa2,$xa3);+$code.=<<___;+ vpaddd @key[8],$xc0,$xc0+ vpaddd @key[9],$xc1,$xc1+ vpaddd @key[10],$xc2,$xc2+ vpaddd @key[11],$xc3,$xc3++ vpunpckldq $xc1,$xc0,$xt2+ vpunpckldq $xc3,$xc2,$xt3+ vpunpckhdq $xc1,$xc0,$xc0+ vpunpckhdq $xc3,$xc2,$xc2+ vpunpcklqdq $xt3,$xt2,$xc1 # "c0"+ vpunpckhqdq $xt3,$xt2,$xt2 # "c1"+ vpunpcklqdq $xc2,$xc0,$xc3 # "c2"+ vpunpckhqdq $xc2,$xc0,$xc0 # "c3"+___+ ($xc0,$xc1,$xc2,$xc3,$xt2)=($xc1,$xt2,$xc3,$xc0,$xc2);+$code.=<<___;+ vpaddd @key[12],$xd0,$xd0+ vpaddd @key[13],$xd1,$xd1+ vpaddd @key[14],$xd2,$xd2+ vpaddd @key[15],$xd3,$xd3++ vpunpckldq $xd1,$xd0,$xt2+ vpunpckldq $xd3,$xd2,$xt3+ vpunpckhdq $xd1,$xd0,$xd0+ vpunpckhdq $xd3,$xd2,$xd2+ vpunpcklqdq $xt3,$xt2,$xd1 # "d0"+ vpunpckhqdq $xt3,$xt2,$xt2 # "d1"+ vpunpcklqdq $xd2,$xd0,$xd3 # "d2"+ vpunpckhqdq $xd2,$xd0,$xd0 # "d3"+___+ ($xd0,$xd1,$xd2,$xd3,$xt2)=($xd1,$xt2,$xd3,$xd0,$xd2);+$code.=<<___;+ vperm2i128 \$0x20,$xd0,$xc0,$xt3 # "de-interlace" further+ vperm2i128 \$0x31,$xd0,$xc0,$xd0+ vperm2i128 \$0x20,$xd1,$xc1,$xc0+ vperm2i128 \$0x31,$xd1,$xc1,$xd1+ vperm2i128 \$0x20,$xd2,$xc2,$xc1+ vperm2i128 \$0x31,$xd2,$xc2,$xd2+ vperm2i128 \$0x20,$xd3,$xc3,$xc2+ vperm2i128 \$0x31,$xd3,$xc3,$xd3+___+ ($xc0,$xc1,$xc2,$xc3,$xt3)=($xt3,$xc0,$xc1,$xc2,$xc3);+ ($xb0,$xb1,$xb2,$xb3,$xc0,$xc1,$xc2,$xc3)=+ ($xc0,$xc1,$xc2,$xc3,$xb0,$xb1,$xb2,$xb3);+$code.=<<___;+ cmp \$64*8,$len+ jb .Ltail8xvl++ mov \$0x80,%eax # size optimization+ vpxord 0x00($inp),$xa0,$xa0 # xor with input+ vpxor 0x20($inp),$xb0,$xb0+ vpxor 0x40($inp),$xc0,$xc0+ vpxor 0x60($inp),$xd0,$xd0+ lea ($inp,%rax),$inp # size optimization+ vmovdqu32 $xa0,0x00($out)+ vmovdqu $xb0,0x20($out)+ vmovdqu $xc0,0x40($out)+ vmovdqu $xd0,0x60($out)+ lea ($out,%rax),$out # size optimization++ vpxor 0x00($inp),$xa1,$xa1+ vpxor 0x20($inp),$xb1,$xb1+ vpxor 0x40($inp),$xc1,$xc1+ vpxor 0x60($inp),$xd1,$xd1+ lea ($inp,%rax),$inp # size optimization+ vmovdqu $xa1,0x00($out)+ vmovdqu $xb1,0x20($out)+ vmovdqu $xc1,0x40($out)+ vmovdqu $xd1,0x60($out)+ lea ($out,%rax),$out # size optimization++ vpxord 0x00($inp),$xa2,$xa2+ vpxor 0x20($inp),$xb2,$xb2+ vpxor 0x40($inp),$xc2,$xc2+ vpxor 0x60($inp),$xd2,$xd2+ lea ($inp,%rax),$inp # size optimization+ vmovdqu32 $xa2,0x00($out)+ vmovdqu $xb2,0x20($out)+ vmovdqu $xc2,0x40($out)+ vmovdqu $xd2,0x60($out)+ lea ($out,%rax),$out # size optimization++ vpxor 0x00($inp),$xa3,$xa3+ vpxor 0x20($inp),$xb3,$xb3+ vpxor 0x40($inp),$xc3,$xc3+ vpxor 0x60($inp),$xd3,$xd3+ lea ($inp,%rax),$inp # size optimization+ vmovdqu $xa3,0x00($out)+ vmovdqu $xb3,0x20($out)+ vmovdqu $xc3,0x40($out)+ vmovdqu $xd3,0x60($out)+ lea ($out,%rax),$out # size optimization++ vpbroadcastd 0(%r9),%ymm0 # reload key+ vpbroadcastd 4(%r9),%ymm1++ sub \$64*8,$len+ jnz .Loop_outer8xvl++ jmp .Ldone8xvl++.align 32+.Ltail8xvl:+ vmovdqa64 $xa0,%ymm8 # size optimization+___+$xa0 = "%ymm8";+$code.=<<___;+ xor %r9,%r9+ sub $inp,$out+ cmp \$64*1,$len+ jb .Less_than_64_8xvl+ vpxor 0x00($inp),$xa0,$xa0 # xor with input+ vpxor 0x20($inp),$xb0,$xb0+ vmovdqu $xa0,0x00($out,$inp)+ vmovdqu $xb0,0x20($out,$inp)+ je .Ldone8xvl+ vmovdqa $xc0,$xa0+ vmovdqa $xd0,$xb0+ lea 64($inp),$inp++ cmp \$64*2,$len+ jb .Less_than_64_8xvl+ vpxor 0x00($inp),$xc0,$xc0+ vpxor 0x20($inp),$xd0,$xd0+ vmovdqu $xc0,0x00($out,$inp)+ vmovdqu $xd0,0x20($out,$inp)+ je .Ldone8xvl+ vmovdqa $xa1,$xa0+ vmovdqa $xb1,$xb0+ lea 64($inp),$inp++ cmp \$64*3,$len+ jb .Less_than_64_8xvl+ vpxor 0x00($inp),$xa1,$xa1+ vpxor 0x20($inp),$xb1,$xb1+ vmovdqu $xa1,0x00($out,$inp)+ vmovdqu $xb1,0x20($out,$inp)+ je .Ldone8xvl+ vmovdqa $xc1,$xa0+ vmovdqa $xd1,$xb0+ lea 64($inp),$inp++ cmp \$64*4,$len+ jb .Less_than_64_8xvl+ vpxor 0x00($inp),$xc1,$xc1+ vpxor 0x20($inp),$xd1,$xd1+ vmovdqu $xc1,0x00($out,$inp)+ vmovdqu $xd1,0x20($out,$inp)+ je .Ldone8xvl+ vmovdqa32 $xa2,$xa0+ vmovdqa $xb2,$xb0+ lea 64($inp),$inp++ cmp \$64*5,$len+ jb .Less_than_64_8xvl+ vpxord 0x00($inp),$xa2,$xa2+ vpxor 0x20($inp),$xb2,$xb2+ vmovdqu32 $xa2,0x00($out,$inp)+ vmovdqu $xb2,0x20($out,$inp)+ je .Ldone8xvl+ vmovdqa $xc2,$xa0+ vmovdqa $xd2,$xb0+ lea 64($inp),$inp++ cmp \$64*6,$len+ jb .Less_than_64_8xvl+ vpxor 0x00($inp),$xc2,$xc2+ vpxor 0x20($inp),$xd2,$xd2+ vmovdqu $xc2,0x00($out,$inp)+ vmovdqu $xd2,0x20($out,$inp)+ je .Ldone8xvl+ vmovdqa $xa3,$xa0+ vmovdqa $xb3,$xb0+ lea 64($inp),$inp++ cmp \$64*7,$len+ jb .Less_than_64_8xvl+ vpxor 0x00($inp),$xa3,$xa3+ vpxor 0x20($inp),$xb3,$xb3+ vmovdqu $xa3,0x00($out,$inp)+ vmovdqu $xb3,0x20($out,$inp)+ je .Ldone8xvl+ vmovdqa $xc3,$xa0+ vmovdqa $xd3,$xb0+ lea 64($inp),$inp++.Less_than_64_8xvl:+ vmovdqa $xa0,0x00(%rsp)+ vmovdqa $xb0,0x20(%rsp)+ lea ($out,$inp),$out+ and \$63,$len++.Loop_tail8xvl:+ movzb ($inp,%r9),%eax+ movzb (%rsp,%r9),%ecx+ lea 1(%r9),%r9+ xor %ecx,%eax+ mov %al,-1($out,%r9)+ dec $len+ jnz .Loop_tail8xvl++ vpxor $xa0,$xa0,$xa0+ vmovdqa $xa0,0x00(%rsp)+ vmovdqa $xa0,0x20(%rsp)++.Ldone8xvl:+ vzeroall+___+$code.=<<___ if ($win64);+ movaps -0xa8(%r10),%xmm6+ movaps -0x98(%r10),%xmm7+ movaps -0x88(%r10),%xmm8+ movaps -0x78(%r10),%xmm9+ movaps -0x68(%r10),%xmm10+ movaps -0x58(%r10),%xmm11+ movaps -0x48(%r10),%xmm12+ movaps -0x38(%r10),%xmm13+ movaps -0x28(%r10),%xmm14+ movaps -0x18(%r10),%xmm15+___+$code.=<<___;+ lea (%r10),%rsp+.cfi_def_cfa_register %rsp+.L8xvl_epilogue:+ ret+.cfi_endproc+.size ChaCha20_8xvl,.-ChaCha20_8xvl+___+}++# EXCEPTION_DISPOSITION handler (EXCEPTION_RECORD *rec,ULONG64 frame,+# CONTEXT *context,DISPATCHER_CONTEXT *disp)+if ($win64) {+$rec="%rcx";+$frame="%rdx";+$context="%r8";+$disp="%r9";++$code.=<<___;+.extern __imp_RtlVirtualUnwind+.type se_handler,\@abi-omnipotent+.align 16+se_handler:+ push %rsi+ push %rdi+ push %rbx+ push %rbp+ push %r12+ push %r13+ push %r14+ push %r15+ pushfq+ sub \$64,%rsp++ mov 120($context),%rax # pull context->Rax+ mov 248($context),%rbx # pull context->Rip++ mov 8($disp),%rsi # disp->ImageBase+ mov 56($disp),%r11 # disp->HandlerData++ lea .Lctr32_body(%rip),%r10+ cmp %r10,%rbx # context->Rip<.Lprologue+ jb .Lcommon_seh_tail++ mov 152($context),%rax # pull context->Rsp++ lea .Lno_data(%rip),%r10 # epilogue label+ cmp %r10,%rbx # context->Rip>=.Lepilogue+ jae .Lcommon_seh_tail++ lea 64+24+48(%rax),%rax++ mov -8(%rax),%rbx+ mov -16(%rax),%rbp+ mov -24(%rax),%r12+ mov -32(%rax),%r13+ mov -40(%rax),%r14+ mov -48(%rax),%r15+ mov %rbx,144($context) # restore context->Rbx+ mov %rbp,160($context) # restore context->Rbp+ mov %r12,216($context) # restore context->R12+ mov %r13,224($context) # restore context->R13+ mov %r14,232($context) # restore context->R14+ mov %r15,240($context) # restore context->R14++.Lcommon_seh_tail:+ mov 8(%rax),%rdi+ mov 16(%rax),%rsi+ mov %rax,152($context) # restore context->Rsp+ mov %rsi,168($context) # restore context->Rsi+ mov %rdi,176($context) # restore context->Rdi++ mov 40($disp),%rdi # disp->ContextRecord+ mov $context,%rsi # context+ mov \$154,%ecx # sizeof(CONTEXT)+ .long 0xa548f3fc # cld; rep movsq++ mov $disp,%rsi+ xor %rcx,%rcx # arg1, UNW_FLAG_NHANDLER+ mov 8(%rsi),%rdx # arg2, disp->ImageBase+ mov 0(%rsi),%r8 # arg3, disp->ControlPc+ mov 16(%rsi),%r9 # arg4, disp->FunctionEntry+ mov 40(%rsi),%r10 # disp->ContextRecord+ lea 56(%rsi),%r11 # &disp->HandlerData+ lea 24(%rsi),%r12 # &disp->EstablisherFrame+ mov %r10,32(%rsp) # arg5+ mov %r11,40(%rsp) # arg6+ mov %r12,48(%rsp) # arg7+ mov %rcx,56(%rsp) # arg8, (NULL)+ call *__imp_RtlVirtualUnwind(%rip)++ mov \$1,%eax # ExceptionContinueSearch+ add \$64,%rsp+ popfq+ pop %r15+ pop %r14+ pop %r13+ pop %r12+ pop %rbp+ pop %rbx+ pop %rdi+ pop %rsi+ ret+.size se_handler,.-se_handler++.type simd_handler,\@abi-omnipotent+.align 16+simd_handler:+ push %rsi+ push %rdi+ push %rbx+ push %rbp+ push %r12+ push %r13+ push %r14+ push %r15+ pushfq+ sub \$64,%rsp++ mov 120($context),%rax # pull context->Rax+ mov 248($context),%rbx # pull context->Rip++ mov 8($disp),%rsi # disp->ImageBase+ mov 56($disp),%r11 # disp->HandlerData++ mov 0(%r11),%r10d # HandlerData[0]+ lea (%rsi,%r10),%r10 # prologue label+ cmp %r10,%rbx # context->Rip<prologue label+ jb .Lcommon_seh_tail++ mov 200($context),%rax # pull context->R10++ mov 4(%r11),%r10d # HandlerData[1]+ mov 8(%r11),%ecx # HandlerData[2]+ lea (%rsi,%r10),%r10 # epilogue label+ cmp %r10,%rbx # context->Rip>=epilogue label+ jae .Lcommon_seh_tail++ neg %rcx+ lea -8(%rax,%rcx),%rsi+ lea 512($context),%rdi # &context.Xmm6+ neg %ecx+ shr \$3,%ecx+ .long 0xa548f3fc # cld; rep movsq++ jmp .Lcommon_seh_tail+.size simd_handler,.-simd_handler++.section .pdata+.align 4+ .rva .LSEH_begin_ChaCha20_ctr32+ .rva .LSEH_end_ChaCha20_ctr32+ .rva .LSEH_info_ChaCha20_ctr32++ .rva .LSEH_begin_ChaCha20_ssse3+ .rva .LSEH_end_ChaCha20_ssse3+ .rva .LSEH_info_ChaCha20_ssse3++ .rva .LSEH_begin_ChaCha20_128+ .rva .LSEH_end_ChaCha20_128+ .rva .LSEH_info_ChaCha20_128++ .rva .LSEH_begin_ChaCha20_4x+ .rva .LSEH_end_ChaCha20_4x+ .rva .LSEH_info_ChaCha20_4x+___+$code.=<<___ if ($avx);+ .rva .LSEH_begin_ChaCha20_4xop+ .rva .LSEH_end_ChaCha20_4xop+ .rva .LSEH_info_ChaCha20_4xop+___+$code.=<<___ if ($avx>1);+ .rva .LSEH_begin_ChaCha20_avx2+ .rva .LSEH_end_ChaCha20_avx2+ .rva .LSEH_info_ChaCha20_avx2+___+$code.=<<___ if ($avx>2);+ .rva .LSEH_begin_ChaCha20_avx512+ .rva .LSEH_end_ChaCha20_avx512+ .rva .LSEH_info_ChaCha20_avx512++ .rva .LSEH_begin_ChaCha20_avx512vl+ .rva .LSEH_end_ChaCha20_avx512vl+ .rva .LSEH_info_ChaCha20_avx512vl++ .rva .LSEH_begin_ChaCha20_16x+ .rva .LSEH_end_ChaCha20_16x+ .rva .LSEH_info_ChaCha20_16x++ .rva .LSEH_begin_ChaCha20_8xvl+ .rva .LSEH_end_ChaCha20_8xvl+ .rva .LSEH_info_ChaCha20_8xvl+___+$code.=<<___;+.section .xdata+.align 8+.LSEH_info_ChaCha20_ctr32:+ .byte 9,0,0,0+ .rva se_handler++.LSEH_info_ChaCha20_ssse3:+ .byte 9,0,0,0+ .rva simd_handler+ .rva .Lssse3_body,.Lssse3_epilogue+ .long 0x20,0++.LSEH_info_ChaCha20_128:+ .byte 9,0,0,0+ .rva simd_handler+ .rva .L128_body,.L128_epilogue+ .long 0x60,0++.LSEH_info_ChaCha20_4x:+ .byte 9,0,0,0+ .rva simd_handler+ .rva .L4x_body,.L4x_epilogue+ .long 0xa0,0+___+$code.=<<___ if ($avx);+.LSEH_info_ChaCha20_4xop:+ .byte 9,0,0,0+ .rva simd_handler+ .rva .L4xop_body,.L4xop_epilogue # HandlerData[]+ .long 0xa0,0+___+$code.=<<___ if ($avx>1);+.LSEH_info_ChaCha20_avx2:+ .byte 9,0,0,0+ .rva simd_handler+ .rva .Lavx2_body,.Lavx2_epilogue # HandlerData[]+ .long 0xa0,0+___+$code.=<<___ if ($avx>2);+.LSEH_info_ChaCha20_avx512:+ .byte 9,0,0,0+ .rva simd_handler+ .rva .Lavx512_body,.Lavx512_epilogue # HandlerData[]+ .long 0x20,0++.LSEH_info_ChaCha20_avx512vl:+ .byte 9,0,0,0+ .rva simd_handler+ .rva .Lavx512vl_body,.Lavx512vl_epilogue # HandlerData[]+ .long 0x20,0++.LSEH_info_ChaCha20_16x:+ .byte 9,0,0,0+ .rva simd_handler+ .rva .L16x_body,.L16x_epilogue # HandlerData[]+ .long 0xa0,0++.LSEH_info_ChaCha20_8xvl:+ .byte 9,0,0,0+ .rva simd_handler+ .rva .L8xvl_body,.L8xvl_epilogue # HandlerData[]+ .long 0xa0,0+___+}++foreach (split("\n",$code)) {+ s/\`([^\`]*)\`/eval $1/ge;++ s/%x#%[yz]/%x/g; # "down-shift"++ print $_,"\n";+}++close STDOUT;
@@ -0,0 +1,153 @@+#!/bin/sh+#+# Regenerate the assembly checked in beside this script.+#+# The .pl files come from the CRYPTOGAMS distribution, unmodified:+#+# https://github.com/dot-asm/cryptogams+# x86_64/aesni-gcm-x86_64.pl x86_64/chacha-x86_64.pl+# x86_64/poly1305-x86_64.pl x86_64/sha512-x86_64.pl+# x86_64/keccak1600-x86_64.pl x86_64/x86_64-xlate.pl+# arm/chacha-armv8.pl arm/poly1305-armv8.pl+# arm/sha1-armv8.pl arm/sha512-armv8.pl+# arm/keccak1600-armv8.pl arm/arm-xlate.pl+# arm/arm_arch.h+#+# The .pl files are the generator, not the product: each one emits+# assembly for a given "flavour", which is the calling convention and the+# object format together. The output is checked in so that building+# crypton needs no perl.+#+# Two things are done to the output here. The entry points are renamed:+# a program that links both crypton and OpenSSL would otherwise have two+# definitions of, say, aesni_gcm_encrypt, and the linker is entitled to+# refuse that. The same goes for OPENSSL_armcap_P, which the ChaCha+# module reads to find out whether the processor has NEON, and which+# crypton defines for itself in cbits/crypton_chacha.c. And the ELF+# output of the x86-64 module is given the note that says the code does+# not want an executable stack, which the generator leaves to the caller.+#+# Run this on a GNU/Linux host. The mingw64 flavour asks the compiler+# what __USER_LABEL_PREFIX__ is for its target, and a compiler for a+# platform that decorates symbols -- Apple's, for one -- answers for+# itself rather than for Windows, which would leave every entry point in+# that file with a leading underscore that nothing looks for.+#+# Usage: cd cbits/asm && ./generate.sh++set -e++# The x86-64 generators choose what to emit from the version of the+# assembler they are told about, so they are told one, rather than left to+# ask whatever compiler happens to be here: the checked-in files should not+# depend on the host that produced them. 2.24 predates AVX-512, which is+# the point -- the Poly1305 module has paths for it, and this does not take+# them, no machine here being able to run them, and a path nothing has+# executed not being worth the few per cent it might be worth. It leaves+# both modules with everything through AVX2.+cat > tmp-cc <<'SHIM'+#!/bin/sh+case "$*" in+*-Wa,-v*) echo "GNU assembler version 2.24" ;;+esac+exit 0+SHIM+chmod +x tmp-cc+CC=./tmp-cc+export CC++for flavour in elf macosx mingw64; do+ perl aesni-gcm-x86_64.pl $flavour tmp-$flavour.S+ sed -e 's/aesni_gcm_/crypton_gcm_asm_/g' \+ -e 's/aesni_ctr32_/crypton_gcm_asm_ctr32_/g' \+ tmp-$flavour.S > aesni-gcm-x86_64-$flavour.S++ perl poly1305-x86_64.pl $flavour tmp-$flavour.S+ sed -e 's/poly1305_/crypton_poly1305_asm_/g' \+ -e 's/xor128_/crypton_xor128_/g' \+ -e 's/OPENSSL_ia32cap_P/crypton_ia32cap_P/g' \+ tmp-$flavour.S > poly1305-x86_64-$flavour.S++ perl chacha-x86_64.pl $flavour tmp-$flavour.S+ sed -e 's/ChaCha20_/crypton_chacha20_asm_/g' \+ -e 's/OPENSSL_ia32cap_P/crypton_ia32cap_P/g' \+ tmp-$flavour.S > chacha-x86_64-$flavour.S++ # as on AArch64, this generator emits SHA-512 or SHA-256 according+ # to the name it is given, and both are wanted here+ perl sha512-x86_64.pl $flavour tmp-$flavour.S+ sed -e 's/sha256_block_/crypton_sha256_asm_block_/g' \+ -e 's/OPENSSL_ia32cap_P/crypton_ia32cap_P/g' \+ tmp-$flavour.S > sha256-x86_64-$flavour.S++ perl keccak1600-x86_64.pl $flavour tmp-k-$flavour.S+ sed -e 's/SHA3_absorb/crypton_keccak_asm_absorb/g' \+ -e 's/SHA3_squeeze/crypton_keccak_asm_squeeze/g' \+ -e 's/KeccakF1600/crypton_keccak_asm_f1600/g' \+ tmp-k-$flavour.S > keccak1600-x86_64-$flavour.S++ perl sha512-x86_64.pl $flavour tmp-512-$flavour.S+ sed -e 's/sha512_block_/crypton_sha512_asm_block_/g' \+ -e 's/OPENSSL_ia32cap_P/crypton_ia32cap_P/g' \+ tmp-512-$flavour.S > sha512-x86_64-$flavour.S+ rm -f tmp-$flavour.S tmp-512-$flavour.S tmp-k-$flavour.S+done++for f in aesni-gcm-x86_64-elf.S poly1305-x86_64-elf.S chacha-x86_64-elf.S \+ sha256-x86_64-elf.S sha512-x86_64-elf.S keccak1600-x86_64-elf.S; do+ cat >> $f <<-NOTE++ .section .note.GNU-stack,"",@progbits+ NOTE+done++unset CC+rm -f tmp-cc++for flavour in linux64 ios64; do+ perl chacha-armv8.pl $flavour tmp-$flavour.S+ sed -e 's/ChaCha20_/crypton_chacha20_asm_/g' \+ -e 's/OPENSSL_armcap_P/crypton_armcap_P/g' \+ tmp-$flavour.S > chacha-armv8-$flavour.S++ perl poly1305-armv8.pl $flavour tmp-$flavour.S+ sed -e 's/poly1305_/crypton_poly1305_asm_/g' \+ -e 's/OPENSSL_armcap_P/crypton_armcap_P/g' \+ tmp-$flavour.S > poly1305-armv8-$flavour.S++ # the same generator emits SHA-512 or SHA-256 according to the name+ # it is given, and only the SHA-256 one is wanted here+ perl sha512-armv8.pl $flavour tmp-$flavour.S+ sed -e 's/sha256_block_/crypton_sha256_asm_block_/g' \+ -e 's/OPENSSL_armcap_P/crypton_armcap_P/g' \+ tmp-$flavour.S > sha256-armv8-$flavour.S++ perl sha1-armv8.pl $flavour tmp-$flavour.S+ sed -e 's/sha1_block_/crypton_sha1_asm_block_/g' \+ -e 's/OPENSSL_armcap_P/crypton_armcap_P/g' \+ tmp-$flavour.S > sha1-armv8-$flavour.S++ perl keccak1600-armv8.pl $flavour tmp-$flavour.S+ sed -e 's/SHA3_absorb/crypton_keccak_asm_absorb/g' \+ -e 's/SHA3_squeeze/crypton_keccak_asm_squeeze/g' \+ tmp-$flavour.S > keccak1600-armv8-$flavour.S+ rm -f tmp-$flavour.S+done++cat >> chacha-armv8-linux64.S <<'NOTE'++.section .note.GNU-stack,"",%progbits+NOTE++cat >> poly1305-armv8-linux64.S <<'NOTE'++.section .note.GNU-stack,"",%progbits+NOTE++for f in sha1-armv8-linux64.S sha256-armv8-linux64.S \+ keccak1600-armv8-linux64.S; do+ cat >> $f <<-NOTE++ .section .note.GNU-stack,"",%progbits+ NOTE+done
@@ -0,0 +1,841 @@+.text++.align 8 // strategic alignment and padding that allows to use+ // address value as loop termination condition...+.quad 0,0,0,0,0,0,0,0++iotas:+.quad 0x0000000000000001+.quad 0x0000000000008082+.quad 0x800000000000808a+.quad 0x8000000080008000+.quad 0x000000000000808b+.quad 0x0000000080000001+.quad 0x8000000080008081+.quad 0x8000000000008009+.quad 0x000000000000008a+.quad 0x0000000000000088+.quad 0x0000000080008009+.quad 0x000000008000000a+Liotas12:+.quad 0x000000008000808b+.quad 0x800000000000008b+.quad 0x8000000000008089+.quad 0x8000000000008003+.quad 0x8000000000008002+.quad 0x8000000000000080+.quad 0x000000000000800a+.quad 0x800000008000000a+.quad 0x8000000080008081+.quad 0x8000000000008080+.quad 0x0000000080000001+.quad 0x8000000080008008+++.align 5+KeccakF1600_int:+.long 0xd503233f // paciasp+ stp x28,x30,[sp,#16] // stack is pre-allocated+ b Loop+.align 4+Loop:+ ////////////////////////////////////////// Theta+ eor x26,x0,x5+ stp x4,x9,[sp,#0] // offload pair...+ eor x27,x1,x6+ eor x28,x2,x7+ eor x30,x3,x8+ eor x4,x4,x9+ eor x26,x26,x10+ eor x27,x27,x11+ eor x28,x28,x12+ eor x30,x30,x13+ eor x4,x4,x14+ eor x26,x26,x15+ eor x27,x27,x16+ eor x28,x28,x17+ eor x30,x30,x25+ eor x4,x4,x19+ eor x26,x26,x20+ eor x28,x28,x22+ eor x27,x27,x21+ eor x30,x30,x23+ eor x4,x4,x24++ eor x9,x26,x28,ror#63++ eor x1,x1,x9+ eor x6,x6,x9+ eor x11,x11,x9+ eor x16,x16,x9+ eor x21,x21,x9++ eor x9,x27,x30,ror#63+ eor x28,x28,x4,ror#63+ eor x30,x30,x26,ror#63+ eor x4,x4,x27,ror#63++ eor x27, x2,x9 // mov x27,x2+ eor x7,x7,x9+ eor x12,x12,x9+ eor x17,x17,x9+ eor x22,x22,x9++ eor x0,x0,x4+ eor x5,x5,x4+ eor x10,x10,x4+ eor x15,x15,x4+ eor x20,x20,x4+ ldp x4,x9,[sp,#0] // re-load offloaded data+ eor x26, x3,x28 // mov x26,x3+ eor x8,x8,x28+ eor x13,x13,x28+ eor x25,x25,x28+ eor x23,x23,x28++ eor x28, x4,x30 // mov x28,x4+ eor x9,x9,x30+ eor x14,x14,x30+ eor x19,x19,x30+ eor x24,x24,x30++ ////////////////////////////////////////// Rho+Pi+ mov x30,x1+ ror x1,x6,#64-44+ //mov x27,x2+ ror x2,x12,#64-43+ //mov x26,x3+ ror x3,x25,#64-21 // ?+ //mov x28,x4+ ror x4,x24,#64-14 // ?++ ror x6,x9,#64-20 // ?+ ror x12,x13,#64-25 // ?+ ror x25,x17,#64-15+ ror x24,x21,#64-2 // ?++ ror x9,x22,#64-61+ ror x13,x19,#64-8+ ror x17,x11,#64-10+ ror x21,x8,#64-55++ ror x22,x14,#64-39+ ror x19,x23,#64-56+ ror x11,x7,#64-6 // ?+ ror x8,x16,#64-45++ ror x14,x20,#64-18+ ror x23,x15,#64-41+ ror x7,x10,#64-3+ ror x16,x5,#64-36 // ?++ ror x5,x26,#64-28 // ?+ ror x10,x30,#64-1+ ror x15,x28,#64-27 // ?+ ror x20,x27,#64-62 // ?++ ////////////////////////////////////////// Chi+Iota+ bic x26,x2,x1+ bic x27,x3,x2+ bic x28,x0,x4+ bic x30,x1,x0+ eor x0,x0,x26+ bic x26,x4,x3+ eor x1,x1,x27+ ldr x27,[sp,#16]+ eor x3,x3,x28+ eor x4,x4,x30+ eor x2,x2,x26+ ldr x30,[x27],#8 // Iota[i++]++ bic x26,x7,x6+ tst x27,#255 // are we done?+ str x27,[sp,#16]+ bic x27,x8,x7+ bic x28,x5,x9+ eor x0,x0,x30 // A[0][0] ^= Iota+ bic x30,x6,x5+ eor x5,x5,x26+ bic x26,x9,x8+ eor x6,x6,x27+ eor x8,x8,x28+ eor x9,x9,x30+ eor x7,x7,x26++ bic x26,x12,x11+ bic x27,x13,x12+ bic x28,x10,x14+ bic x30,x11,x10+ eor x10,x10,x26+ bic x26,x14,x13+ eor x11,x11,x27+ eor x13,x13,x28+ eor x14,x14,x30+ eor x12,x12,x26++ bic x26,x17,x16+ bic x27,x25,x17+ bic x28,x15,x19+ bic x30,x16,x15+ eor x15,x15,x26+ bic x26,x19,x25+ eor x16,x16,x27+ eor x25,x25,x28+ eor x19,x19,x30+ eor x17,x17,x26++ bic x26,x22,x21+ bic x27,x23,x22+ bic x28,x20,x24+ bic x30,x21,x20+ eor x20,x20,x26+ bic x26,x24,x23+ eor x21,x21,x27+ eor x23,x23,x28+ eor x24,x24,x30+ eor x22,x22,x26++ bne Loop++ ldr x30,[sp,#16+__SIZEOF_POINTER__]+.long 0xd50323bf // autiasp+ ret++++.align 5+KeccakF1600:+.long 0xd503233f // paciasp+ stp x29,x30,[sp,#-16*__SIZEOF_POINTER__]!+ add x29,sp,#0+ stp x19,x20,[sp,#2*__SIZEOF_POINTER__]+ stp x21,x22,[sp,#4*__SIZEOF_POINTER__]+ stp x23,x24,[sp,#6*__SIZEOF_POINTER__]+ stp x25,x26,[sp,#8*__SIZEOF_POINTER__]+ stp x27,x28,[sp,#10*__SIZEOF_POINTER__]+ sub sp,sp,#16+4*__SIZEOF_POINTER__++ str x0,[sp,#16+2*__SIZEOF_POINTER__] // offload argument+ mov x26,x0+ ldp x0,x1,[x0,#16*0]+ ldp x2,x3,[x26,#16*1]+ ldp x4,x5,[x26,#16*2]+ ldp x6,x7,[x26,#16*3]+ ldp x8,x9,[x26,#16*4]+ ldp x10,x11,[x26,#16*5]+ ldp x12,x13,[x26,#16*6]+ ldp x14,x15,[x26,#16*7]+ ldp x16,x17,[x26,#16*8]+ ldp x25,x19,[x26,#16*9]+ ldp x20,x21,[x26,#16*10]+ ldp x22,x23,[x26,#16*11]+ ldr x24,[x26,#16*12]++ adr x28,iotas+ bl KeccakF1600_int++ ldr x26,[sp,#16+2*__SIZEOF_POINTER__]+ stp x0,x1,[x26,#16*0]+ stp x2,x3,[x26,#16*1]+ stp x4,x5,[x26,#16*2]+ stp x6,x7,[x26,#16*3]+ stp x8,x9,[x26,#16*4]+ stp x10,x11,[x26,#16*5]+ stp x12,x13,[x26,#16*6]+ stp x14,x15,[x26,#16*7]+ stp x16,x17,[x26,#16*8]+ stp x25,x19,[x26,#16*9]+ stp x20,x21,[x26,#16*10]+ stp x22,x23,[x26,#16*11]+ str x24,[x26,#16*12]++ ldp x19,x20,[x29,#2*__SIZEOF_POINTER__]+ add sp,sp,#16+4*__SIZEOF_POINTER__+ ldp x21,x22,[x29,#4*__SIZEOF_POINTER__]+ ldp x23,x24,[x29,#6*__SIZEOF_POINTER__]+ ldp x25,x26,[x29,#8*__SIZEOF_POINTER__]+ ldp x27,x28,[x29,#10*__SIZEOF_POINTER__]+ ldp x29,x30,[sp],#16*__SIZEOF_POINTER__+.long 0xd50323bf // autiasp+ ret+++.globl _crypton_keccak_asm_absorb++.align 5+_crypton_keccak_asm_absorb:+.long 0xd503233f // paciasp+ stp x29,x30,[sp,#-16*__SIZEOF_POINTER__]!+ add x29,sp,#0+ stp x19,x20,[sp,#2*__SIZEOF_POINTER__]+ stp x21,x22,[sp,#4*__SIZEOF_POINTER__]+ stp x23,x24,[sp,#6*__SIZEOF_POINTER__]+ stp x25,x26,[sp,#8*__SIZEOF_POINTER__]+ stp x27,x28,[sp,#10*__SIZEOF_POINTER__]+ sub sp,sp,#16+4*__SIZEOF_POINTER__+16++ stp x0,x1,[sp,#16+2*__SIZEOF_POINTER__] // offload arguments+ stp x2,x3,[sp,#16+4*__SIZEOF_POINTER__]++ mov x26,x0 // uint64_t A[5][5]+ mov x27,x1 // const void *inp+ mov x28,x2 // size_t len+ mov x30,x3 // size_t bsz+ ldp x0,x1,[x26,#16*0]+ ldp x2,x3,[x26,#16*1]+ ldp x4,x5,[x26,#16*2]+ ldp x6,x7,[x26,#16*3]+ ldp x8,x9,[x26,#16*4]+ ldp x10,x11,[x26,#16*5]+ ldp x12,x13,[x26,#16*6]+ ldp x14,x15,[x26,#16*7]+ ldp x16,x17,[x26,#16*8]+ ldp x25,x19,[x26,#16*9]+ ldp x20,x21,[x26,#16*10]+ ldp x22,x23,[x26,#16*11]+ ldr x24,[x26,#16*12]+ b Loop_absorb++.align 4+Loop_absorb:+ subs x26,x28,x30 // len - bsz+ blo Labsorbed++ str x26,[sp,#16+4*__SIZEOF_POINTER__] // save len - bsz+ cmp x30,#104+ ldr x26,[x27,#0] // A[0][0] ^= *inp+++#ifdef __AARCH64EB__+ rev x26,x26+#endif+ eor x0,x0,x26+ ldr x26,[x27,#8] // A[0][1] ^= *inp+++#ifdef __AARCH64EB__+ rev x26,x26+#endif+ eor x1,x1,x26+ ldr x26,[x27,#16] // A[0][2] ^= *inp+++#ifdef __AARCH64EB__+ rev x26,x26+#endif+ eor x2,x2,x26+ ldr x26,[x27,#24] // A[0][3] ^= *inp+++#ifdef __AARCH64EB__+ rev x26,x26+#endif+ eor x3,x3,x26+ ldr x26,[x27,#32] // A[0][4] ^= *inp+++#ifdef __AARCH64EB__+ rev x26,x26+#endif+ eor x4,x4,x26+ ldr x26,[x27,#40] // A[1][0] ^= *inp+++#ifdef __AARCH64EB__+ rev x26,x26+#endif+ eor x5,x5,x26+ ldr x26,[x27,#48] // A[1][1] ^= *inp+++#ifdef __AARCH64EB__+ rev x26,x26+#endif+ eor x6,x6,x26+ ldr x26,[x27,#56] // A[1][2] ^= *inp+++#ifdef __AARCH64EB__+ rev x26,x26+#endif+ eor x7,x7,x26+ ldr x26,[x27,#64] // A[1][3] ^= *inp+++#ifdef __AARCH64EB__+ rev x26,x26+#endif+ eor x8,x8,x26+ blo Lprocess_block++ ldr x26,[x27,#72] // A[1][4] ^= *inp+++#ifdef __AARCH64EB__+ rev x26,x26+#endif+ eor x9,x9,x26+ ldr x26,[x27,#80] // A[2][0] ^= *inp+++#ifdef __AARCH64EB__+ rev x26,x26+#endif+ eor x10,x10,x26+ ldr x26,[x27,#88] // A[2][1] ^= *inp+++#ifdef __AARCH64EB__+ rev x26,x26+#endif+ eor x11,x11,x26+ ldr x26,[x27,#96] // A[2][2] ^= *inp+++#ifdef __AARCH64EB__+ rev x26,x26+#endif+ eor x12,x12,x26+ beq Lprocess_block++ cmp x30,#144+ ldr x26,[x27,#104] // A[2][3] ^= *inp+++#ifdef __AARCH64EB__+ rev x26,x26+#endif+ eor x13,x13,x26+ ldr x26,[x27,#112] // A[2][4] ^= *inp+++#ifdef __AARCH64EB__+ rev x26,x26+#endif+ eor x14,x14,x26+ ldr x26,[x27,#120] // A[3][0] ^= *inp+++#ifdef __AARCH64EB__+ rev x26,x26+#endif+ eor x15,x15,x26+ ldr x26,[x27,#128] // A[3][1] ^= *inp+++#ifdef __AARCH64EB__+ rev x26,x26+#endif+ eor x16,x16,x26+ blo Lprocess_block++ ldr x26,[x27,#136] // A[3][2] ^= *inp+++#ifdef __AARCH64EB__+ rev x26,x26+#endif+ eor x17,x17,x26+ beq Lprocess_block++ ldr x26,[x27,#144] // A[3][3] ^= *inp+++#ifdef __AARCH64EB__+ rev x26,x26+#endif+ eor x25,x25,x26+ ldr x26,[x27,#152] // A[3][4] ^= *inp+++#ifdef __AARCH64EB__+ rev x26,x26+#endif+ eor x19,x19,x26+ ldr x26,[x27,#160] // A[4][0] ^= *inp+++#ifdef __AARCH64EB__+ rev x26,x26+#endif+ eor x20,x20,x26++Lprocess_block:+ add x27,x27,x30+ str x27,[sp,#16+3*__SIZEOF_POINTER__] // save inp++ adr x28,iotas+ bl KeccakF1600_int++ ldr x27,[sp,#16+3*__SIZEOF_POINTER__] // restore arguments+ ldp x28,x30,[sp,#16+4*__SIZEOF_POINTER__]+ b Loop_absorb++.align 4+Labsorbed:+ ldr x27,[sp,#16+2*__SIZEOF_POINTER__]+ stp x0,x1,[x27,#16*0]+ stp x2,x3,[x27,#16*1]+ stp x4,x5,[x27,#16*2]+ stp x6,x7,[x27,#16*3]+ stp x8,x9,[x27,#16*4]+ stp x10,x11,[x27,#16*5]+ stp x12,x13,[x27,#16*6]+ stp x14,x15,[x27,#16*7]+ stp x16,x17,[x27,#16*8]+ stp x25,x19,[x27,#16*9]+ stp x20,x21,[x27,#16*10]+ stp x22,x23,[x27,#16*11]+ str x24,[x27,#16*12]++ mov x0,x28 // return value+ ldp x19,x20,[x29,#2*__SIZEOF_POINTER__]+ add sp,sp,#16+4*__SIZEOF_POINTER__+16+ ldp x21,x22,[x29,#4*__SIZEOF_POINTER__]+ ldp x23,x24,[x29,#6*__SIZEOF_POINTER__]+ ldp x25,x26,[x29,#8*__SIZEOF_POINTER__]+ ldp x27,x28,[x29,#10*__SIZEOF_POINTER__]+ ldp x29,x30,[sp],#16*__SIZEOF_POINTER__+.long 0xd50323bf // autiasp+ ret++.globl _crypton_keccak_asm_squeeze++.align 5+_crypton_keccak_asm_squeeze:+.long 0xd503233f // paciasp+ stp x29,x30,[sp,#-6*__SIZEOF_POINTER__]!+ add x29,sp,#0+ stp x19,x20,[sp,#2*__SIZEOF_POINTER__]+ stp x21,x22,[sp,#4*__SIZEOF_POINTER__]++ mov x19,x0 // put aside arguments+ mov x20,x1+ mov x21,x2+ mov x22,x3++Loop_squeeze:+ ldr x4,[x0],#8+ cmp x21,#8+ blo Lsqueeze_tail+#ifdef __AARCH64EB__+ rev x4,x4+#endif+ str x4,[x20],#8+ subs x21,x21,#8+ beq Lsqueeze_done++ subs x3,x3,#8+ bhi Loop_squeeze++ mov x0,x19+ bl KeccakF1600+ mov x0,x19+ mov x3,x22+ b Loop_squeeze++.align 4+Lsqueeze_tail:+ strb w4,[x20],#1+ lsr x4,x4,#8+ subs x21,x21,#1+ beq Lsqueeze_done+ strb w4,[x20],#1+ lsr x4,x4,#8+ subs x21,x21,#1+ beq Lsqueeze_done+ strb w4,[x20],#1+ lsr x4,x4,#8+ subs x21,x21,#1+ beq Lsqueeze_done+ strb w4,[x20],#1+ lsr x4,x4,#8+ subs x21,x21,#1+ beq Lsqueeze_done+ strb w4,[x20],#1+ lsr x4,x4,#8+ subs x21,x21,#1+ beq Lsqueeze_done+ strb w4,[x20],#1+ lsr x4,x4,#8+ subs x21,x21,#1+ beq Lsqueeze_done+ strb w4,[x20],#1++Lsqueeze_done:+ ldp x19,x20,[sp,#2*__SIZEOF_POINTER__]+ ldp x21,x22,[sp,#4*__SIZEOF_POINTER__]+ ldp x29,x30,[sp],#6*__SIZEOF_POINTER__+.long 0xd50323bf // autiasp+ ret+++.align 5+KeccakF1600_ce:+Loop_ce:+ ////////////////////////////////////////////////// Theta+.long 0xce0f2a99 //eor3 v25.16b,v20.16b,v15.16b,v10.16b+.long 0xce102eba //eor3 v26.16b,v21.16b,v16.16b,v11.16b+.long 0xce1132db //eor3 v27.16b,v22.16b,v17.16b,v12.16b+.long 0xce1236fc //eor3 v28.16b,v23.16b,v18.16b,v13.16b+.long 0xce133b1d //eor3 v29.16b,v24.16b,v19.16b,v14.16b+.long 0xce050339 //eor3 v25.16b,v25.16b, v5.16b,v0.16b+.long 0xce06075a //eor3 v26.16b,v26.16b, v6.16b,v1.16b+.long 0xce070b7b //eor3 v27.16b,v27.16b, v7.16b,v2.16b+.long 0xce080f9c //eor3 v28.16b,v28.16b, v8.16b,v3.16b+.long 0xce0913bd //eor3 v29.16b,v29.16b, v9.16b,v4.16b++.long 0xce7b8f3e //rax1 v30.2d,v25.2d,v27.2d // D[1]+.long 0xce7c8f5f //rax1 v31.2d,v26.2d,v28.2d // D[2]+.long 0xce7d8f7b //rax1 v27.2d,v27.2d,v29.2d // D[3]+.long 0xce798f9c //rax1 v28.2d,v28.2d,v25.2d // D[4]+.long 0xce7a8fbd //rax1 v29.2d,v29.2d,v26.2d // D[0]++ ////////////////////////////////////////////////// Theta+Rho+Pi+.long 0xce9efc39 //xar v25.2d, v1.2d,v30.2d,#64-1 // C[0]=A[2][0]++.long 0xce9e50c1 //xar v1.2d,v6.2d,v30.2d,#64-44+.long 0xce9cb126 //xar v6.2d,v9.2d,v28.2d,#64-20+.long 0xce9f0ec9 //xar v9.2d,v22.2d,v31.2d,#64-61+.long 0xce9c65d6 //xar v22.2d,v14.2d,v28.2d,#64-39+.long 0xce9dba8e //xar v14.2d,v20.2d,v29.2d,#64-18++.long 0xce9f085a //xar v26.2d, v2.2d,v31.2d,#64-62 // C[1]=A[4][0]++.long 0xce9f5582 //xar v2.2d,v12.2d,v31.2d,#64-43+.long 0xce9b9dac //xar v12.2d,v13.2d,v27.2d,#64-25+.long 0xce9ce26d //xar v13.2d,v19.2d,v28.2d,#64-8+.long 0xce9b22f3 //xar v19.2d,v23.2d,v27.2d,#64-56+.long 0xce9d5df7 //xar v23.2d,v15.2d,v29.2d,#64-41++.long 0xce9c948f //xar v15.2d,v4.2d,v28.2d,#64-27++.long 0xce9ccb1c //xar v28.2d, v24.2d,v28.2d,#64-14 // D[4]=A[0][4]+.long 0xce9efab8 //xar v24.2d,v21.2d,v30.2d,#64-2+.long 0xce9b2508 //xar v8.2d,v8.2d,v27.2d,#64-55 // A[1][3]=A[4][1]+.long 0xce9e4e04 //xar v4.2d,v16.2d,v30.2d,#64-45 // A[0][4]=A[1][3]+.long 0xce9d70b0 //xar v16.2d,v5.2d,v29.2d,#64-36++.long 0xce9b9065 //xar v5.2d,v3.2d,v27.2d,#64-28++ eor v0.16b,v0.16b,v29.16b++.long 0xce9bae5b //xar v27.2d, v18.2d,v27.2d,#64-21 // D[3]=A[0][3]+.long 0xce9fc623 //xar v3.2d,v17.2d,v31.2d,#64-15 // A[0][3]=A[3][3]+.long 0xce9ed97e //xar v30.2d, v11.2d,v30.2d,#64-10 // D[1]=A[3][2]+.long 0xce9fe8ff //xar v31.2d, v7.2d,v31.2d,#64-6 // D[2]=A[2][1]+.long 0xce9df55d //xar v29.2d, v10.2d,v29.2d,#64-3 // D[0]=A[1][2]++ ////////////////////////////////////////////////// Chi+Iota+.long 0xce362354 //bcax v20.16b,v26.16b, v22.16b,v8.16b // A[1][3]=A[4][1]+.long 0xce375915 //bcax v21.16b,v8.16b,v23.16b,v22.16b // A[1][3]=A[4][1]+.long 0xce385ed6 //bcax v22.16b,v22.16b,v24.16b,v23.16b+.long 0xce3a62f7 //bcax v23.16b,v23.16b,v26.16b, v24.16b+.long 0xce286b18 //bcax v24.16b,v24.16b,v8.16b,v26.16b // A[1][3]=A[4][1]++ ld1r {v26.2d},[x10],#8++.long 0xce330fd1 //bcax v17.16b,v30.16b, v19.16b,v3.16b // A[0][3]=A[3][3]+.long 0xce2f4c72 //bcax v18.16b,v3.16b,v15.16b,v19.16b // A[0][3]=A[3][3]+.long 0xce303e73 //bcax v19.16b,v19.16b,v16.16b,v15.16b+.long 0xce3e41ef //bcax v15.16b,v15.16b,v30.16b, v16.16b+.long 0xce237a10 //bcax v16.16b,v16.16b,v3.16b,v30.16b // A[0][3]=A[3][3]++.long 0xce2c7f2a //bcax v10.16b,v25.16b, v12.16b,v31.16b+.long 0xce2d33eb //bcax v11.16b,v31.16b, v13.16b,v12.16b+.long 0xce2e358c //bcax v12.16b,v12.16b,v14.16b,v13.16b+.long 0xce3939ad //bcax v13.16b,v13.16b,v25.16b, v14.16b+.long 0xce3f65ce //bcax v14.16b,v14.16b,v31.16b, v25.16b++.long 0xce2913a7 //bcax v7.16b,v29.16b, v9.16b,v4.16b // A[0][4]=A[1][3]+.long 0xce252488 //bcax v8.16b,v4.16b,v5.16b,v9.16b // A[0][4]=A[1][3]+.long 0xce261529 //bcax v9.16b,v9.16b,v6.16b,v5.16b+.long 0xce3d18a5 //bcax v5.16b,v5.16b,v29.16b, v6.16b+.long 0xce2474c6 //bcax v6.16b,v6.16b,v4.16b,v29.16b // A[0][4]=A[1][3]++.long 0xce207363 //bcax v3.16b,v27.16b, v0.16b,v28.16b+.long 0xce210384 //bcax v4.16b,v28.16b, v1.16b,v0.16b+.long 0xce220400 //bcax v0.16b,v0.16b,v2.16b,v1.16b+.long 0xce3b0821 //bcax v1.16b,v1.16b,v27.16b, v2.16b+.long 0xce3c6c42 //bcax v2.16b,v2.16b,v28.16b, v27.16b++ eor v0.16b,v0.16b,v26.16b++ tst x10,#255+ bne Loop_ce++ ret++++.align 5+KeccakF1600_cext:+.long 0xd503233f // paciasp+ stp x29,x30,[sp,#-2*__SIZEOF_POINTER__-64]!+ add x29,sp,#0+ stp d8,d9,[sp,#2*__SIZEOF_POINTER__+0] // per ABI requirement+ stp d10,d11,[sp,#2*__SIZEOF_POINTER__+16]+ stp d12,d13,[sp,#2*__SIZEOF_POINTER__+32]+ stp d14,d15,[sp,#2*__SIZEOF_POINTER__+48]+ ldp d0,d1,[x0,#8*0]+ ldp d2,d3,[x0,#8*2]+ ldp d4,d5,[x0,#8*4]+ ldp d6,d7,[x0,#8*6]+ ldp d8,d9,[x0,#8*8]+ ldp d10,d11,[x0,#8*10]+ ldp d12,d13,[x0,#8*12]+ ldp d14,d15,[x0,#8*14]+ ldp d16,d17,[x0,#8*16]+ ldp d18,d19,[x0,#8*18]+ ldp d20,d21,[x0,#8*20]+ ldp d22,d23,[x0,#8*22]+ ldr d24,[x0,#8*24]+ adr x10,iotas+ bl KeccakF1600_ce+ ldr x30,[sp,#__SIZEOF_POINTER__]+ stp d0,d1,[x0,#8*0]+ stp d2,d3,[x0,#8*2]+ stp d4,d5,[x0,#8*4]+ stp d6,d7,[x0,#8*6]+ stp d8,d9,[x0,#8*8]+ stp d10,d11,[x0,#8*10]+ stp d12,d13,[x0,#8*12]+ stp d14,d15,[x0,#8*14]+ stp d16,d17,[x0,#8*16]+ stp d18,d19,[x0,#8*18]+ stp d20,d21,[x0,#8*20]+ stp d22,d23,[x0,#8*22]+ str d24,[x0,#8*24]++ ldp d8,d9,[sp,#2*__SIZEOF_POINTER__+0]+ ldp d10,d11,[sp,#2*__SIZEOF_POINTER__+16]+ ldp d12,d13,[sp,#2*__SIZEOF_POINTER__+32]+ ldp d14,d15,[sp,#2*__SIZEOF_POINTER__+48]+ ldr x29,[sp],#2*__SIZEOF_POINTER__+64+.long 0xd50323bf // autiasp+ ret++.globl _crypton_keccak_asm_absorb_cext++.align 5+_crypton_keccak_asm_absorb_cext:+.long 0xd503233f // paciasp+ stp x29,x30,[sp,#-2*__SIZEOF_POINTER__-64]!+ add x29,sp,#0+ stp d8,d9,[sp,#2*__SIZEOF_POINTER__+0] // per ABI requirement+ stp d10,d11,[sp,#2*__SIZEOF_POINTER__+16]+ stp d12,d13,[sp,#2*__SIZEOF_POINTER__+32]+ stp d14,d15,[sp,#2*__SIZEOF_POINTER__+48]+ ldp d0,d1,[x0,#8*0]+ ldp d2,d3,[x0,#8*2]+ ldp d4,d5,[x0,#8*4]+ ldp d6,d7,[x0,#8*6]+ ldp d8,d9,[x0,#8*8]+ ldp d10,d11,[x0,#8*10]+ ldp d12,d13,[x0,#8*12]+ ldp d14,d15,[x0,#8*14]+ ldp d16,d17,[x0,#8*16]+ ldp d18,d19,[x0,#8*18]+ ldp d20,d21,[x0,#8*20]+ ldp d22,d23,[x0,#8*22]+ ldr d24,[x0,#8*24]+ b Loop_absorb_ce++.align 4+Loop_absorb_ce:+ subs x2,x2,x3 // len - bsz+ blo Labsorbed_ce++ cmp x3,#104+ ld1 {v27.8b,v28.8b,v29.8b,v30.8b},[x1],#32+ eor v0.16b,v0.16b,v27.16b+ eor v1.16b,v1.16b,v28.16b+ eor v2.16b,v2.16b,v29.16b+ eor v3.16b,v3.16b,v30.16b+ ld1 {v27.8b,v28.8b,v29.8b,v30.8b},[x1],#32+ eor v4.16b,v4.16b,v27.16b+ eor v5.16b,v5.16b,v28.16b+ eor v6.16b,v6.16b,v29.16b+ eor v7.16b,v7.16b,v30.16b+ ld1 {v31.8b},[x1],#8 // A[1][4] ^= *inp+++ eor v8.16b,v8.16b,v31.16b+ blo Lprocess_block_ce++ ld1 {v27.8b,v28.8b,v29.8b,v30.8b},[x1],#32+ eor v9.16b,v9.16b,v27.16b+ eor v10.16b,v10.16b,v28.16b+ eor v11.16b,v11.16b,v29.16b+ eor v12.16b,v12.16b,v30.16b+ beq Lprocess_block_ce++ cmp x3,#144+ ld1 {v27.8b,v28.8b,v29.8b,v30.8b},[x1],#32+ eor v13.16b,v13.16b,v27.16b+ eor v14.16b,v14.16b,v28.16b+ eor v15.16b,v15.16b,v29.16b+ eor v16.16b,v16.16b,v30.16b+ blo Lprocess_block_ce++ ld1 {v31.8b},[x1],#8 // A[3][3] ^= *inp+++ eor v17.16b,v17.16b,v31.16b+ beq Lprocess_block_ce++ ld1 {v28.8b,v29.8b,v30.8b},[x1],#24+ eor v18.16b,v18.16b,v28.16b+ eor v19.16b,v19.16b,v29.16b+ eor v20.16b,v20.16b,v30.16b++Lprocess_block_ce:+ adr x10,iotas+ bl KeccakF1600_ce++ b Loop_absorb_ce++.align 4+Labsorbed_ce:+ stp d0,d1,[x0,#8*0]+ stp d2,d3,[x0,#8*2]+ stp d4,d5,[x0,#8*4]+ stp d6,d7,[x0,#8*6]+ stp d8,d9,[x0,#8*8]+ stp d10,d11,[x0,#8*10]+ stp d12,d13,[x0,#8*12]+ stp d14,d15,[x0,#8*14]+ stp d16,d17,[x0,#8*16]+ stp d18,d19,[x0,#8*18]+ stp d20,d21,[x0,#8*20]+ stp d22,d23,[x0,#8*22]+ str d24,[x0,#8*24]+ add x0,x2,x3 // return value++ ldp d8,d9,[sp,#2*__SIZEOF_POINTER__+0]+ ldp d10,d11,[sp,#2*__SIZEOF_POINTER__+16]+ ldp d12,d13,[sp,#2*__SIZEOF_POINTER__+32]+ ldp d14,d15,[sp,#2*__SIZEOF_POINTER__+48]+ ldp x29,x30,[sp],#2*__SIZEOF_POINTER__+64+.long 0xd50323bf // autiasp+ ret++.globl _crypton_keccak_asm_squeeze_cext++.align 5+_crypton_keccak_asm_squeeze_cext:+.long 0xd503233f // paciasp+ stp x29,x30,[sp,#-2*__SIZEOF_POINTER__]!+ add x29,sp,#0+ mov x9,x0+ mov x10,x3++Loop_squeeze_ce:+ ldr x4,[x9],#8+ cmp x2,#8+ blo Lsqueeze_tail_ce+#ifdef __AARCH64EB__+ rev x4,x4+#endif+ str x4,[x1],#8+ beq Lsqueeze_done_ce++ sub x2,x2,#8+ subs x10,x10,#8+ bhi Loop_squeeze_ce++ bl KeccakF1600_cext+ ldr x30,[sp,#__SIZEOF_POINTER__]+ mov x9,x0+ mov x10,x3+ b Loop_squeeze_ce++.align 4+Lsqueeze_tail_ce:+ strb w4,[x1],#1+ lsr x4,x4,#8+ subs x2,x2,#1+ beq Lsqueeze_done_ce+ strb w4,[x1],#1+ lsr x4,x4,#8+ subs x2,x2,#1+ beq Lsqueeze_done_ce+ strb w4,[x1],#1+ lsr x4,x4,#8+ subs x2,x2,#1+ beq Lsqueeze_done_ce+ strb w4,[x1],#1+ lsr x4,x4,#8+ subs x2,x2,#1+ beq Lsqueeze_done_ce+ strb w4,[x1],#1+ lsr x4,x4,#8+ subs x2,x2,#1+ beq Lsqueeze_done_ce+ strb w4,[x1],#1+ lsr x4,x4,#8+ subs x2,x2,#1+ beq Lsqueeze_done_ce+ strb w4,[x1],#1++Lsqueeze_done_ce:+ ldr x29,[sp],#2*__SIZEOF_POINTER__+.long 0xd50323bf // autiasp+ ret++.byte 75,101,99,99,97,107,45,49,54,48,48,32,97,98,115,111,114,98,32,97,110,100,32,115,113,117,101,101,122,101,32,102,111,114,32,65,82,77,118,56,44,32,67,82,89,80,84,79,71,65,77,83,32,98,121,32,64,100,111,116,45,97,115,109,0+.align 2
@@ -0,0 +1,843 @@+.text++.align 8 // strategic alignment and padding that allows to use+ // address value as loop termination condition...+.quad 0,0,0,0,0,0,0,0+.type iotas,%object+iotas:+.quad 0x0000000000000001+.quad 0x0000000000008082+.quad 0x800000000000808a+.quad 0x8000000080008000+.quad 0x000000000000808b+.quad 0x0000000080000001+.quad 0x8000000080008081+.quad 0x8000000000008009+.quad 0x000000000000008a+.quad 0x0000000000000088+.quad 0x0000000080008009+.quad 0x000000008000000a+.Liotas12:+.quad 0x000000008000808b+.quad 0x800000000000008b+.quad 0x8000000000008089+.quad 0x8000000000008003+.quad 0x8000000000008002+.quad 0x8000000000000080+.quad 0x000000000000800a+.quad 0x800000008000000a+.quad 0x8000000080008081+.quad 0x8000000000008080+.quad 0x0000000080000001+.quad 0x8000000080008008+.size iotas,.-iotas+.type KeccakF1600_int,%function+.align 5+KeccakF1600_int:+.inst 0xd503233f // paciasp+ stp x28,x30,[sp,#16] // stack is pre-allocated+ b .Loop+.align 4+.Loop:+ ////////////////////////////////////////// Theta+ eor x26,x0,x5+ stp x4,x9,[sp,#0] // offload pair...+ eor x27,x1,x6+ eor x28,x2,x7+ eor x30,x3,x8+ eor x4,x4,x9+ eor x26,x26,x10+ eor x27,x27,x11+ eor x28,x28,x12+ eor x30,x30,x13+ eor x4,x4,x14+ eor x26,x26,x15+ eor x27,x27,x16+ eor x28,x28,x17+ eor x30,x30,x25+ eor x4,x4,x19+ eor x26,x26,x20+ eor x28,x28,x22+ eor x27,x27,x21+ eor x30,x30,x23+ eor x4,x4,x24++ eor x9,x26,x28,ror#63++ eor x1,x1,x9+ eor x6,x6,x9+ eor x11,x11,x9+ eor x16,x16,x9+ eor x21,x21,x9++ eor x9,x27,x30,ror#63+ eor x28,x28,x4,ror#63+ eor x30,x30,x26,ror#63+ eor x4,x4,x27,ror#63++ eor x27, x2,x9 // mov x27,x2+ eor x7,x7,x9+ eor x12,x12,x9+ eor x17,x17,x9+ eor x22,x22,x9++ eor x0,x0,x4+ eor x5,x5,x4+ eor x10,x10,x4+ eor x15,x15,x4+ eor x20,x20,x4+ ldp x4,x9,[sp,#0] // re-load offloaded data+ eor x26, x3,x28 // mov x26,x3+ eor x8,x8,x28+ eor x13,x13,x28+ eor x25,x25,x28+ eor x23,x23,x28++ eor x28, x4,x30 // mov x28,x4+ eor x9,x9,x30+ eor x14,x14,x30+ eor x19,x19,x30+ eor x24,x24,x30++ ////////////////////////////////////////// Rho+Pi+ mov x30,x1+ ror x1,x6,#64-44+ //mov x27,x2+ ror x2,x12,#64-43+ //mov x26,x3+ ror x3,x25,#64-21 // ?+ //mov x28,x4+ ror x4,x24,#64-14 // ?++ ror x6,x9,#64-20 // ?+ ror x12,x13,#64-25 // ?+ ror x25,x17,#64-15+ ror x24,x21,#64-2 // ?++ ror x9,x22,#64-61+ ror x13,x19,#64-8+ ror x17,x11,#64-10+ ror x21,x8,#64-55++ ror x22,x14,#64-39+ ror x19,x23,#64-56+ ror x11,x7,#64-6 // ?+ ror x8,x16,#64-45++ ror x14,x20,#64-18+ ror x23,x15,#64-41+ ror x7,x10,#64-3+ ror x16,x5,#64-36 // ?++ ror x5,x26,#64-28 // ?+ ror x10,x30,#64-1+ ror x15,x28,#64-27 // ?+ ror x20,x27,#64-62 // ?++ ////////////////////////////////////////// Chi+Iota+ bic x26,x2,x1+ bic x27,x3,x2+ bic x28,x0,x4+ bic x30,x1,x0+ eor x0,x0,x26+ bic x26,x4,x3+ eor x1,x1,x27+ ldr x27,[sp,#16]+ eor x3,x3,x28+ eor x4,x4,x30+ eor x2,x2,x26+ ldr x30,[x27],#8 // Iota[i++]++ bic x26,x7,x6+ tst x27,#255 // are we done?+ str x27,[sp,#16]+ bic x27,x8,x7+ bic x28,x5,x9+ eor x0,x0,x30 // A[0][0] ^= Iota+ bic x30,x6,x5+ eor x5,x5,x26+ bic x26,x9,x8+ eor x6,x6,x27+ eor x8,x8,x28+ eor x9,x9,x30+ eor x7,x7,x26++ bic x26,x12,x11+ bic x27,x13,x12+ bic x28,x10,x14+ bic x30,x11,x10+ eor x10,x10,x26+ bic x26,x14,x13+ eor x11,x11,x27+ eor x13,x13,x28+ eor x14,x14,x30+ eor x12,x12,x26++ bic x26,x17,x16+ bic x27,x25,x17+ bic x28,x15,x19+ bic x30,x16,x15+ eor x15,x15,x26+ bic x26,x19,x25+ eor x16,x16,x27+ eor x25,x25,x28+ eor x19,x19,x30+ eor x17,x17,x26++ bic x26,x22,x21+ bic x27,x23,x22+ bic x28,x20,x24+ bic x30,x21,x20+ eor x20,x20,x26+ bic x26,x24,x23+ eor x21,x21,x27+ eor x23,x23,x28+ eor x24,x24,x30+ eor x22,x22,x26++ bne .Loop++ ldr x30,[sp,#16+__SIZEOF_POINTER__]+.inst 0xd50323bf // autiasp+ ret+.size KeccakF1600_int,.-KeccakF1600_int++.type KeccakF1600,%function+.align 5+KeccakF1600:+.inst 0xd503233f // paciasp+ stp x29,x30,[sp,#-16*__SIZEOF_POINTER__]!+ add x29,sp,#0+ stp x19,x20,[sp,#2*__SIZEOF_POINTER__]+ stp x21,x22,[sp,#4*__SIZEOF_POINTER__]+ stp x23,x24,[sp,#6*__SIZEOF_POINTER__]+ stp x25,x26,[sp,#8*__SIZEOF_POINTER__]+ stp x27,x28,[sp,#10*__SIZEOF_POINTER__]+ sub sp,sp,#16+4*__SIZEOF_POINTER__++ str x0,[sp,#16+2*__SIZEOF_POINTER__] // offload argument+ mov x26,x0+ ldp x0,x1,[x0,#16*0]+ ldp x2,x3,[x26,#16*1]+ ldp x4,x5,[x26,#16*2]+ ldp x6,x7,[x26,#16*3]+ ldp x8,x9,[x26,#16*4]+ ldp x10,x11,[x26,#16*5]+ ldp x12,x13,[x26,#16*6]+ ldp x14,x15,[x26,#16*7]+ ldp x16,x17,[x26,#16*8]+ ldp x25,x19,[x26,#16*9]+ ldp x20,x21,[x26,#16*10]+ ldp x22,x23,[x26,#16*11]+ ldr x24,[x26,#16*12]++ adr x28,iotas+ bl KeccakF1600_int++ ldr x26,[sp,#16+2*__SIZEOF_POINTER__]+ stp x0,x1,[x26,#16*0]+ stp x2,x3,[x26,#16*1]+ stp x4,x5,[x26,#16*2]+ stp x6,x7,[x26,#16*3]+ stp x8,x9,[x26,#16*4]+ stp x10,x11,[x26,#16*5]+ stp x12,x13,[x26,#16*6]+ stp x14,x15,[x26,#16*7]+ stp x16,x17,[x26,#16*8]+ stp x25,x19,[x26,#16*9]+ stp x20,x21,[x26,#16*10]+ stp x22,x23,[x26,#16*11]+ str x24,[x26,#16*12]++ ldp x19,x20,[x29,#2*__SIZEOF_POINTER__]+ add sp,sp,#16+4*__SIZEOF_POINTER__+ ldp x21,x22,[x29,#4*__SIZEOF_POINTER__]+ ldp x23,x24,[x29,#6*__SIZEOF_POINTER__]+ ldp x25,x26,[x29,#8*__SIZEOF_POINTER__]+ ldp x27,x28,[x29,#10*__SIZEOF_POINTER__]+ ldp x29,x30,[sp],#16*__SIZEOF_POINTER__+.inst 0xd50323bf // autiasp+ ret+.size KeccakF1600,.-KeccakF1600++.globl crypton_keccak_asm_absorb+.type crypton_keccak_asm_absorb,%function+.align 5+crypton_keccak_asm_absorb:+.inst 0xd503233f // paciasp+ stp x29,x30,[sp,#-16*__SIZEOF_POINTER__]!+ add x29,sp,#0+ stp x19,x20,[sp,#2*__SIZEOF_POINTER__]+ stp x21,x22,[sp,#4*__SIZEOF_POINTER__]+ stp x23,x24,[sp,#6*__SIZEOF_POINTER__]+ stp x25,x26,[sp,#8*__SIZEOF_POINTER__]+ stp x27,x28,[sp,#10*__SIZEOF_POINTER__]+ sub sp,sp,#16+4*__SIZEOF_POINTER__+16++ stp x0,x1,[sp,#16+2*__SIZEOF_POINTER__] // offload arguments+ stp x2,x3,[sp,#16+4*__SIZEOF_POINTER__]++ mov x26,x0 // uint64_t A[5][5]+ mov x27,x1 // const void *inp+ mov x28,x2 // size_t len+ mov x30,x3 // size_t bsz+ ldp x0,x1,[x26,#16*0]+ ldp x2,x3,[x26,#16*1]+ ldp x4,x5,[x26,#16*2]+ ldp x6,x7,[x26,#16*3]+ ldp x8,x9,[x26,#16*4]+ ldp x10,x11,[x26,#16*5]+ ldp x12,x13,[x26,#16*6]+ ldp x14,x15,[x26,#16*7]+ ldp x16,x17,[x26,#16*8]+ ldp x25,x19,[x26,#16*9]+ ldp x20,x21,[x26,#16*10]+ ldp x22,x23,[x26,#16*11]+ ldr x24,[x26,#16*12]+ b .Loop_absorb++.align 4+.Loop_absorb:+ subs x26,x28,x30 // len - bsz+ blo .Labsorbed++ str x26,[sp,#16+4*__SIZEOF_POINTER__] // save len - bsz+ cmp x30,#104+ ldr x26,[x27,#0] // A[0][0] ^= *inp+++#ifdef __AARCH64EB__+ rev x26,x26+#endif+ eor x0,x0,x26+ ldr x26,[x27,#8] // A[0][1] ^= *inp+++#ifdef __AARCH64EB__+ rev x26,x26+#endif+ eor x1,x1,x26+ ldr x26,[x27,#16] // A[0][2] ^= *inp+++#ifdef __AARCH64EB__+ rev x26,x26+#endif+ eor x2,x2,x26+ ldr x26,[x27,#24] // A[0][3] ^= *inp+++#ifdef __AARCH64EB__+ rev x26,x26+#endif+ eor x3,x3,x26+ ldr x26,[x27,#32] // A[0][4] ^= *inp+++#ifdef __AARCH64EB__+ rev x26,x26+#endif+ eor x4,x4,x26+ ldr x26,[x27,#40] // A[1][0] ^= *inp+++#ifdef __AARCH64EB__+ rev x26,x26+#endif+ eor x5,x5,x26+ ldr x26,[x27,#48] // A[1][1] ^= *inp+++#ifdef __AARCH64EB__+ rev x26,x26+#endif+ eor x6,x6,x26+ ldr x26,[x27,#56] // A[1][2] ^= *inp+++#ifdef __AARCH64EB__+ rev x26,x26+#endif+ eor x7,x7,x26+ ldr x26,[x27,#64] // A[1][3] ^= *inp+++#ifdef __AARCH64EB__+ rev x26,x26+#endif+ eor x8,x8,x26+ blo .Lprocess_block++ ldr x26,[x27,#72] // A[1][4] ^= *inp+++#ifdef __AARCH64EB__+ rev x26,x26+#endif+ eor x9,x9,x26+ ldr x26,[x27,#80] // A[2][0] ^= *inp+++#ifdef __AARCH64EB__+ rev x26,x26+#endif+ eor x10,x10,x26+ ldr x26,[x27,#88] // A[2][1] ^= *inp+++#ifdef __AARCH64EB__+ rev x26,x26+#endif+ eor x11,x11,x26+ ldr x26,[x27,#96] // A[2][2] ^= *inp+++#ifdef __AARCH64EB__+ rev x26,x26+#endif+ eor x12,x12,x26+ beq .Lprocess_block++ cmp x30,#144+ ldr x26,[x27,#104] // A[2][3] ^= *inp+++#ifdef __AARCH64EB__+ rev x26,x26+#endif+ eor x13,x13,x26+ ldr x26,[x27,#112] // A[2][4] ^= *inp+++#ifdef __AARCH64EB__+ rev x26,x26+#endif+ eor x14,x14,x26+ ldr x26,[x27,#120] // A[3][0] ^= *inp+++#ifdef __AARCH64EB__+ rev x26,x26+#endif+ eor x15,x15,x26+ ldr x26,[x27,#128] // A[3][1] ^= *inp+++#ifdef __AARCH64EB__+ rev x26,x26+#endif+ eor x16,x16,x26+ blo .Lprocess_block++ ldr x26,[x27,#136] // A[3][2] ^= *inp+++#ifdef __AARCH64EB__+ rev x26,x26+#endif+ eor x17,x17,x26+ beq .Lprocess_block++ ldr x26,[x27,#144] // A[3][3] ^= *inp+++#ifdef __AARCH64EB__+ rev x26,x26+#endif+ eor x25,x25,x26+ ldr x26,[x27,#152] // A[3][4] ^= *inp+++#ifdef __AARCH64EB__+ rev x26,x26+#endif+ eor x19,x19,x26+ ldr x26,[x27,#160] // A[4][0] ^= *inp+++#ifdef __AARCH64EB__+ rev x26,x26+#endif+ eor x20,x20,x26++.Lprocess_block:+ add x27,x27,x30+ str x27,[sp,#16+3*__SIZEOF_POINTER__] // save inp++ adr x28,iotas+ bl KeccakF1600_int++ ldr x27,[sp,#16+3*__SIZEOF_POINTER__] // restore arguments+ ldp x28,x30,[sp,#16+4*__SIZEOF_POINTER__]+ b .Loop_absorb++.align 4+.Labsorbed:+ ldr x27,[sp,#16+2*__SIZEOF_POINTER__]+ stp x0,x1,[x27,#16*0]+ stp x2,x3,[x27,#16*1]+ stp x4,x5,[x27,#16*2]+ stp x6,x7,[x27,#16*3]+ stp x8,x9,[x27,#16*4]+ stp x10,x11,[x27,#16*5]+ stp x12,x13,[x27,#16*6]+ stp x14,x15,[x27,#16*7]+ stp x16,x17,[x27,#16*8]+ stp x25,x19,[x27,#16*9]+ stp x20,x21,[x27,#16*10]+ stp x22,x23,[x27,#16*11]+ str x24,[x27,#16*12]++ mov x0,x28 // return value+ ldp x19,x20,[x29,#2*__SIZEOF_POINTER__]+ add sp,sp,#16+4*__SIZEOF_POINTER__+16+ ldp x21,x22,[x29,#4*__SIZEOF_POINTER__]+ ldp x23,x24,[x29,#6*__SIZEOF_POINTER__]+ ldp x25,x26,[x29,#8*__SIZEOF_POINTER__]+ ldp x27,x28,[x29,#10*__SIZEOF_POINTER__]+ ldp x29,x30,[sp],#16*__SIZEOF_POINTER__+.inst 0xd50323bf // autiasp+ ret+.size crypton_keccak_asm_absorb,.-crypton_keccak_asm_absorb+.globl crypton_keccak_asm_squeeze+.type crypton_keccak_asm_squeeze,%function+.align 5+crypton_keccak_asm_squeeze:+.inst 0xd503233f // paciasp+ stp x29,x30,[sp,#-6*__SIZEOF_POINTER__]!+ add x29,sp,#0+ stp x19,x20,[sp,#2*__SIZEOF_POINTER__]+ stp x21,x22,[sp,#4*__SIZEOF_POINTER__]++ mov x19,x0 // put aside arguments+ mov x20,x1+ mov x21,x2+ mov x22,x3++.Loop_squeeze:+ ldr x4,[x0],#8+ cmp x21,#8+ blo .Lsqueeze_tail+#ifdef __AARCH64EB__+ rev x4,x4+#endif+ str x4,[x20],#8+ subs x21,x21,#8+ beq .Lsqueeze_done++ subs x3,x3,#8+ bhi .Loop_squeeze++ mov x0,x19+ bl KeccakF1600+ mov x0,x19+ mov x3,x22+ b .Loop_squeeze++.align 4+.Lsqueeze_tail:+ strb w4,[x20],#1+ lsr x4,x4,#8+ subs x21,x21,#1+ beq .Lsqueeze_done+ strb w4,[x20],#1+ lsr x4,x4,#8+ subs x21,x21,#1+ beq .Lsqueeze_done+ strb w4,[x20],#1+ lsr x4,x4,#8+ subs x21,x21,#1+ beq .Lsqueeze_done+ strb w4,[x20],#1+ lsr x4,x4,#8+ subs x21,x21,#1+ beq .Lsqueeze_done+ strb w4,[x20],#1+ lsr x4,x4,#8+ subs x21,x21,#1+ beq .Lsqueeze_done+ strb w4,[x20],#1+ lsr x4,x4,#8+ subs x21,x21,#1+ beq .Lsqueeze_done+ strb w4,[x20],#1++.Lsqueeze_done:+ ldp x19,x20,[sp,#2*__SIZEOF_POINTER__]+ ldp x21,x22,[sp,#4*__SIZEOF_POINTER__]+ ldp x29,x30,[sp],#6*__SIZEOF_POINTER__+.inst 0xd50323bf // autiasp+ ret+.size crypton_keccak_asm_squeeze,.-crypton_keccak_asm_squeeze+.type KeccakF1600_ce,%function+.align 5+KeccakF1600_ce:+.Loop_ce:+ ////////////////////////////////////////////////// Theta+.inst 0xce0f2a99 //eor3 v25.16b,v20.16b,v15.16b,v10.16b+.inst 0xce102eba //eor3 v26.16b,v21.16b,v16.16b,v11.16b+.inst 0xce1132db //eor3 v27.16b,v22.16b,v17.16b,v12.16b+.inst 0xce1236fc //eor3 v28.16b,v23.16b,v18.16b,v13.16b+.inst 0xce133b1d //eor3 v29.16b,v24.16b,v19.16b,v14.16b+.inst 0xce050339 //eor3 v25.16b,v25.16b, v5.16b,v0.16b+.inst 0xce06075a //eor3 v26.16b,v26.16b, v6.16b,v1.16b+.inst 0xce070b7b //eor3 v27.16b,v27.16b, v7.16b,v2.16b+.inst 0xce080f9c //eor3 v28.16b,v28.16b, v8.16b,v3.16b+.inst 0xce0913bd //eor3 v29.16b,v29.16b, v9.16b,v4.16b++.inst 0xce7b8f3e //rax1 v30.2d,v25.2d,v27.2d // D[1]+.inst 0xce7c8f5f //rax1 v31.2d,v26.2d,v28.2d // D[2]+.inst 0xce7d8f7b //rax1 v27.2d,v27.2d,v29.2d // D[3]+.inst 0xce798f9c //rax1 v28.2d,v28.2d,v25.2d // D[4]+.inst 0xce7a8fbd //rax1 v29.2d,v29.2d,v26.2d // D[0]++ ////////////////////////////////////////////////// Theta+Rho+Pi+.inst 0xce9efc39 //xar v25.2d, v1.2d,v30.2d,#64-1 // C[0]=A[2][0]++.inst 0xce9e50c1 //xar v1.2d,v6.2d,v30.2d,#64-44+.inst 0xce9cb126 //xar v6.2d,v9.2d,v28.2d,#64-20+.inst 0xce9f0ec9 //xar v9.2d,v22.2d,v31.2d,#64-61+.inst 0xce9c65d6 //xar v22.2d,v14.2d,v28.2d,#64-39+.inst 0xce9dba8e //xar v14.2d,v20.2d,v29.2d,#64-18++.inst 0xce9f085a //xar v26.2d, v2.2d,v31.2d,#64-62 // C[1]=A[4][0]++.inst 0xce9f5582 //xar v2.2d,v12.2d,v31.2d,#64-43+.inst 0xce9b9dac //xar v12.2d,v13.2d,v27.2d,#64-25+.inst 0xce9ce26d //xar v13.2d,v19.2d,v28.2d,#64-8+.inst 0xce9b22f3 //xar v19.2d,v23.2d,v27.2d,#64-56+.inst 0xce9d5df7 //xar v23.2d,v15.2d,v29.2d,#64-41++.inst 0xce9c948f //xar v15.2d,v4.2d,v28.2d,#64-27++.inst 0xce9ccb1c //xar v28.2d, v24.2d,v28.2d,#64-14 // D[4]=A[0][4]+.inst 0xce9efab8 //xar v24.2d,v21.2d,v30.2d,#64-2+.inst 0xce9b2508 //xar v8.2d,v8.2d,v27.2d,#64-55 // A[1][3]=A[4][1]+.inst 0xce9e4e04 //xar v4.2d,v16.2d,v30.2d,#64-45 // A[0][4]=A[1][3]+.inst 0xce9d70b0 //xar v16.2d,v5.2d,v29.2d,#64-36++.inst 0xce9b9065 //xar v5.2d,v3.2d,v27.2d,#64-28++ eor v0.16b,v0.16b,v29.16b++.inst 0xce9bae5b //xar v27.2d, v18.2d,v27.2d,#64-21 // D[3]=A[0][3]+.inst 0xce9fc623 //xar v3.2d,v17.2d,v31.2d,#64-15 // A[0][3]=A[3][3]+.inst 0xce9ed97e //xar v30.2d, v11.2d,v30.2d,#64-10 // D[1]=A[3][2]+.inst 0xce9fe8ff //xar v31.2d, v7.2d,v31.2d,#64-6 // D[2]=A[2][1]+.inst 0xce9df55d //xar v29.2d, v10.2d,v29.2d,#64-3 // D[0]=A[1][2]++ ////////////////////////////////////////////////// Chi+Iota+.inst 0xce362354 //bcax v20.16b,v26.16b, v22.16b,v8.16b // A[1][3]=A[4][1]+.inst 0xce375915 //bcax v21.16b,v8.16b,v23.16b,v22.16b // A[1][3]=A[4][1]+.inst 0xce385ed6 //bcax v22.16b,v22.16b,v24.16b,v23.16b+.inst 0xce3a62f7 //bcax v23.16b,v23.16b,v26.16b, v24.16b+.inst 0xce286b18 //bcax v24.16b,v24.16b,v8.16b,v26.16b // A[1][3]=A[4][1]++ ld1r {v26.2d},[x10],#8++.inst 0xce330fd1 //bcax v17.16b,v30.16b, v19.16b,v3.16b // A[0][3]=A[3][3]+.inst 0xce2f4c72 //bcax v18.16b,v3.16b,v15.16b,v19.16b // A[0][3]=A[3][3]+.inst 0xce303e73 //bcax v19.16b,v19.16b,v16.16b,v15.16b+.inst 0xce3e41ef //bcax v15.16b,v15.16b,v30.16b, v16.16b+.inst 0xce237a10 //bcax v16.16b,v16.16b,v3.16b,v30.16b // A[0][3]=A[3][3]++.inst 0xce2c7f2a //bcax v10.16b,v25.16b, v12.16b,v31.16b+.inst 0xce2d33eb //bcax v11.16b,v31.16b, v13.16b,v12.16b+.inst 0xce2e358c //bcax v12.16b,v12.16b,v14.16b,v13.16b+.inst 0xce3939ad //bcax v13.16b,v13.16b,v25.16b, v14.16b+.inst 0xce3f65ce //bcax v14.16b,v14.16b,v31.16b, v25.16b++.inst 0xce2913a7 //bcax v7.16b,v29.16b, v9.16b,v4.16b // A[0][4]=A[1][3]+.inst 0xce252488 //bcax v8.16b,v4.16b,v5.16b,v9.16b // A[0][4]=A[1][3]+.inst 0xce261529 //bcax v9.16b,v9.16b,v6.16b,v5.16b+.inst 0xce3d18a5 //bcax v5.16b,v5.16b,v29.16b, v6.16b+.inst 0xce2474c6 //bcax v6.16b,v6.16b,v4.16b,v29.16b // A[0][4]=A[1][3]++.inst 0xce207363 //bcax v3.16b,v27.16b, v0.16b,v28.16b+.inst 0xce210384 //bcax v4.16b,v28.16b, v1.16b,v0.16b+.inst 0xce220400 //bcax v0.16b,v0.16b,v2.16b,v1.16b+.inst 0xce3b0821 //bcax v1.16b,v1.16b,v27.16b, v2.16b+.inst 0xce3c6c42 //bcax v2.16b,v2.16b,v28.16b, v27.16b++ eor v0.16b,v0.16b,v26.16b++ tst x10,#255+ bne .Loop_ce++ ret+.size KeccakF1600_ce,.-KeccakF1600_ce++.type KeccakF1600_cext,%function+.align 5+KeccakF1600_cext:+.inst 0xd503233f // paciasp+ stp x29,x30,[sp,#-2*__SIZEOF_POINTER__-64]!+ add x29,sp,#0+ stp d8,d9,[sp,#2*__SIZEOF_POINTER__+0] // per ABI requirement+ stp d10,d11,[sp,#2*__SIZEOF_POINTER__+16]+ stp d12,d13,[sp,#2*__SIZEOF_POINTER__+32]+ stp d14,d15,[sp,#2*__SIZEOF_POINTER__+48]+ ldp d0,d1,[x0,#8*0]+ ldp d2,d3,[x0,#8*2]+ ldp d4,d5,[x0,#8*4]+ ldp d6,d7,[x0,#8*6]+ ldp d8,d9,[x0,#8*8]+ ldp d10,d11,[x0,#8*10]+ ldp d12,d13,[x0,#8*12]+ ldp d14,d15,[x0,#8*14]+ ldp d16,d17,[x0,#8*16]+ ldp d18,d19,[x0,#8*18]+ ldp d20,d21,[x0,#8*20]+ ldp d22,d23,[x0,#8*22]+ ldr d24,[x0,#8*24]+ adr x10,iotas+ bl KeccakF1600_ce+ ldr x30,[sp,#__SIZEOF_POINTER__]+ stp d0,d1,[x0,#8*0]+ stp d2,d3,[x0,#8*2]+ stp d4,d5,[x0,#8*4]+ stp d6,d7,[x0,#8*6]+ stp d8,d9,[x0,#8*8]+ stp d10,d11,[x0,#8*10]+ stp d12,d13,[x0,#8*12]+ stp d14,d15,[x0,#8*14]+ stp d16,d17,[x0,#8*16]+ stp d18,d19,[x0,#8*18]+ stp d20,d21,[x0,#8*20]+ stp d22,d23,[x0,#8*22]+ str d24,[x0,#8*24]++ ldp d8,d9,[sp,#2*__SIZEOF_POINTER__+0]+ ldp d10,d11,[sp,#2*__SIZEOF_POINTER__+16]+ ldp d12,d13,[sp,#2*__SIZEOF_POINTER__+32]+ ldp d14,d15,[sp,#2*__SIZEOF_POINTER__+48]+ ldr x29,[sp],#2*__SIZEOF_POINTER__+64+.inst 0xd50323bf // autiasp+ ret+.size KeccakF1600_cext,.-KeccakF1600_cext+.globl crypton_keccak_asm_absorb_cext+.type crypton_keccak_asm_absorb_cext,%function+.align 5+crypton_keccak_asm_absorb_cext:+.inst 0xd503233f // paciasp+ stp x29,x30,[sp,#-2*__SIZEOF_POINTER__-64]!+ add x29,sp,#0+ stp d8,d9,[sp,#2*__SIZEOF_POINTER__+0] // per ABI requirement+ stp d10,d11,[sp,#2*__SIZEOF_POINTER__+16]+ stp d12,d13,[sp,#2*__SIZEOF_POINTER__+32]+ stp d14,d15,[sp,#2*__SIZEOF_POINTER__+48]+ ldp d0,d1,[x0,#8*0]+ ldp d2,d3,[x0,#8*2]+ ldp d4,d5,[x0,#8*4]+ ldp d6,d7,[x0,#8*6]+ ldp d8,d9,[x0,#8*8]+ ldp d10,d11,[x0,#8*10]+ ldp d12,d13,[x0,#8*12]+ ldp d14,d15,[x0,#8*14]+ ldp d16,d17,[x0,#8*16]+ ldp d18,d19,[x0,#8*18]+ ldp d20,d21,[x0,#8*20]+ ldp d22,d23,[x0,#8*22]+ ldr d24,[x0,#8*24]+ b .Loop_absorb_ce++.align 4+.Loop_absorb_ce:+ subs x2,x2,x3 // len - bsz+ blo .Labsorbed_ce++ cmp x3,#104+ ld1 {v27.8b,v28.8b,v29.8b,v30.8b},[x1],#32+ eor v0.16b,v0.16b,v27.16b+ eor v1.16b,v1.16b,v28.16b+ eor v2.16b,v2.16b,v29.16b+ eor v3.16b,v3.16b,v30.16b+ ld1 {v27.8b,v28.8b,v29.8b,v30.8b},[x1],#32+ eor v4.16b,v4.16b,v27.16b+ eor v5.16b,v5.16b,v28.16b+ eor v6.16b,v6.16b,v29.16b+ eor v7.16b,v7.16b,v30.16b+ ld1 {v31.8b},[x1],#8 // A[1][4] ^= *inp+++ eor v8.16b,v8.16b,v31.16b+ blo .Lprocess_block_ce++ ld1 {v27.8b,v28.8b,v29.8b,v30.8b},[x1],#32+ eor v9.16b,v9.16b,v27.16b+ eor v10.16b,v10.16b,v28.16b+ eor v11.16b,v11.16b,v29.16b+ eor v12.16b,v12.16b,v30.16b+ beq .Lprocess_block_ce++ cmp x3,#144+ ld1 {v27.8b,v28.8b,v29.8b,v30.8b},[x1],#32+ eor v13.16b,v13.16b,v27.16b+ eor v14.16b,v14.16b,v28.16b+ eor v15.16b,v15.16b,v29.16b+ eor v16.16b,v16.16b,v30.16b+ blo .Lprocess_block_ce++ ld1 {v31.8b},[x1],#8 // A[3][3] ^= *inp+++ eor v17.16b,v17.16b,v31.16b+ beq .Lprocess_block_ce++ ld1 {v28.8b,v29.8b,v30.8b},[x1],#24+ eor v18.16b,v18.16b,v28.16b+ eor v19.16b,v19.16b,v29.16b+ eor v20.16b,v20.16b,v30.16b++.Lprocess_block_ce:+ adr x10,iotas+ bl KeccakF1600_ce++ b .Loop_absorb_ce++.align 4+.Labsorbed_ce:+ stp d0,d1,[x0,#8*0]+ stp d2,d3,[x0,#8*2]+ stp d4,d5,[x0,#8*4]+ stp d6,d7,[x0,#8*6]+ stp d8,d9,[x0,#8*8]+ stp d10,d11,[x0,#8*10]+ stp d12,d13,[x0,#8*12]+ stp d14,d15,[x0,#8*14]+ stp d16,d17,[x0,#8*16]+ stp d18,d19,[x0,#8*18]+ stp d20,d21,[x0,#8*20]+ stp d22,d23,[x0,#8*22]+ str d24,[x0,#8*24]+ add x0,x2,x3 // return value++ ldp d8,d9,[sp,#2*__SIZEOF_POINTER__+0]+ ldp d10,d11,[sp,#2*__SIZEOF_POINTER__+16]+ ldp d12,d13,[sp,#2*__SIZEOF_POINTER__+32]+ ldp d14,d15,[sp,#2*__SIZEOF_POINTER__+48]+ ldp x29,x30,[sp],#2*__SIZEOF_POINTER__+64+.inst 0xd50323bf // autiasp+ ret+.size crypton_keccak_asm_absorb_cext,.-crypton_keccak_asm_absorb_cext+.globl crypton_keccak_asm_squeeze_cext+.type crypton_keccak_asm_squeeze_cext,%function+.align 5+crypton_keccak_asm_squeeze_cext:+.inst 0xd503233f // paciasp+ stp x29,x30,[sp,#-2*__SIZEOF_POINTER__]!+ add x29,sp,#0+ mov x9,x0+ mov x10,x3++.Loop_squeeze_ce:+ ldr x4,[x9],#8+ cmp x2,#8+ blo .Lsqueeze_tail_ce+#ifdef __AARCH64EB__+ rev x4,x4+#endif+ str x4,[x1],#8+ beq .Lsqueeze_done_ce++ sub x2,x2,#8+ subs x10,x10,#8+ bhi .Loop_squeeze_ce++ bl KeccakF1600_cext+ ldr x30,[sp,#__SIZEOF_POINTER__]+ mov x9,x0+ mov x10,x3+ b .Loop_squeeze_ce++.align 4+.Lsqueeze_tail_ce:+ strb w4,[x1],#1+ lsr x4,x4,#8+ subs x2,x2,#1+ beq .Lsqueeze_done_ce+ strb w4,[x1],#1+ lsr x4,x4,#8+ subs x2,x2,#1+ beq .Lsqueeze_done_ce+ strb w4,[x1],#1+ lsr x4,x4,#8+ subs x2,x2,#1+ beq .Lsqueeze_done_ce+ strb w4,[x1],#1+ lsr x4,x4,#8+ subs x2,x2,#1+ beq .Lsqueeze_done_ce+ strb w4,[x1],#1+ lsr x4,x4,#8+ subs x2,x2,#1+ beq .Lsqueeze_done_ce+ strb w4,[x1],#1+ lsr x4,x4,#8+ subs x2,x2,#1+ beq .Lsqueeze_done_ce+ strb w4,[x1],#1++.Lsqueeze_done_ce:+ ldr x29,[sp],#2*__SIZEOF_POINTER__+.inst 0xd50323bf // autiasp+ ret+.size crypton_keccak_asm_squeeze_cext,.-crypton_keccak_asm_squeeze_cext+.byte 75,101,99,99,97,107,45,49,54,48,48,32,97,98,115,111,114,98,32,97,110,100,32,115,113,117,101,101,122,101,32,102,111,114,32,65,82,77,118,56,44,32,67,82,89,80,84,79,71,65,77,83,32,98,121,32,64,100,111,116,45,97,115,109,0+.align 2++.section .note.GNU-stack,"",%progbits
@@ -0,0 +1,932 @@+#!/usr/bin/env perl+#+# ====================================================================+# Written by Andy Polyakov, @dot-asm, initially for the OpenSSL+# project. The module is, however, dual licensed under OpenSSL and+# CRYPTOGAMS licenses depending on where you obtain it. For further+# details see http://www.openssl.org/~appro/cryptogams/.+# ====================================================================+#+# Keccak-1600 for ARMv8.+#+# June 2017.+#+# This is straightforward KECCAK_1X_ALT implementation. It makes no+# sense to attempt SIMD/NEON implementation for following reason.+# 64-bit lanes of vector registers can't be addressed as easily as in+# 32-bit mode. This means that 64-bit NEON is bound to be slower than+# 32-bit NEON, and this implementation is faster than 32-bit NEON on+# same processor. Even though it takes more scalar xor's and andn's,+# it gets compensated by availability of rotate. Not to forget that+# most processors achieve higher issue rate with scalar instructions.+#+# February 2018.+#+# Add hardware-assisted ARMv8.2 implementation. It's KECCAK_1X_ALT+# variant with register permutation/rotation twist that allows to+# eliminate copies to temporary registers. If you look closely you'll+# notice that it uses only one lane of vector registers. The new+# instructions effectively facilitate parallel hashing, which we don't+# support [yet?]. But lowest-level core procedure is prepared for it.+# The inner round is 67 [vector] instructions, so it's not actually+# obvious that it will provide performance improvement [in serial+# hash] as long as vector instructions issue rate is limited to 1 per+# cycle...+#+######################################################################+# Numbers are cycles per processed byte.+#+# r=1088(*)+#+# Cortex-A53 13+# Cortex-A57 12+# Cortex-A76 7.9+# Cortex-X2 6.1 (***)+# Cortex-X925 3.0 (**)+# X-Gene 14+# Mongoose 10+# Kryo 12+# Snapdragon X 3.8 (**)+# Denver 7.8+# Apple A7 7.2+# Apple A10 6.1+# Apple A12 4.4+# Apple A14/M1 3.5 (**)+# ThunderX2 9.7+#+# (*) Corresponds to SHA3-256. No improvement coefficients are listed+# because they vary too much from compiler to compiler. Newer+# compiler does much better and improvement varies from 5% on+# Cortex-A57 to 25% on Cortex-A53. While in comparison to older+# compiler this code is at least 2x faster...+# (**) The result is for hardware-assisted implementation below.+# (***) Hardware-assisted code is significantly slower, 11.3,+# apparently because the processor can issue just one SHA3+# instruction per cycle.++$flavour = shift;+$output = shift;++if ($flavour && $flavour ne "void") {+ $0 =~ m/(.*[\/\\])[^\/\\]+$/; $dir=$1;+ ( $xlate="${dir}arm-xlate.pl" and -f $xlate ) or+ ( $xlate="${dir}../../perlasm/arm-xlate.pl" and -f $xlate) or+ die "can't locate arm-xlate.pl";++ open STDOUT,"| \"$^X\" $xlate $flavour $output";+} else {+ open STDOUT,">$output";+}++my @rhotates = ([ 0, 1, 62, 28, 27 ],+ [ 36, 44, 6, 55, 20 ],+ [ 3, 10, 43, 25, 39 ],+ [ 41, 45, 15, 21, 8 ],+ [ 18, 2, 61, 56, 14 ]);++my $sha3ops = ($flavour =~ /\+sha3/);++$code.=<<___ if ($sha3ops);+.arch armv8.2-a+sha3+___+$code.=<<___;+.text++.align 8 // strategic alignment and padding that allows to use+ // address value as loop termination condition...+ .quad 0,0,0,0,0,0,0,0+.type iotas,%object+iotas:+ .quad 0x0000000000000001+ .quad 0x0000000000008082+ .quad 0x800000000000808a+ .quad 0x8000000080008000+ .quad 0x000000000000808b+ .quad 0x0000000080000001+ .quad 0x8000000080008081+ .quad 0x8000000000008009+ .quad 0x000000000000008a+ .quad 0x0000000000000088+ .quad 0x0000000080008009+ .quad 0x000000008000000a+.Liotas12:+ .quad 0x000000008000808b+ .quad 0x800000000000008b+ .quad 0x8000000000008089+ .quad 0x8000000000008003+ .quad 0x8000000000008002+ .quad 0x8000000000000080+ .quad 0x000000000000800a+ .quad 0x800000008000000a+ .quad 0x8000000080008081+ .quad 0x8000000000008080+ .quad 0x0000000080000001+ .quad 0x8000000080008008+.size iotas,.-iotas+___+ {{{+my @A = map([ "x$_", "x".($_+1), "x".($_+2), "x".($_+3), "x".($_+4) ],+ (0, 5, 10, 15, 20));+ $A[3][3] = "x25"; # x18 is reserved++my @C = map("x$_", (26,27,28,30));++$code.=<<___;+.type KeccakF1600_int,%function+.align 5+KeccakF1600_int:+ .inst 0xd503233f // paciasp+ stp c#$C[2],c30,[csp,#16] // stack is pre-allocated+ b .Loop+.align 4+.Loop:+ ////////////////////////////////////////// Theta+ eor $C[0],$A[0][0],$A[1][0]+ stp $A[0][4],$A[1][4],[sp,#0] // offload pair...+ eor $C[1],$A[0][1],$A[1][1]+ eor $C[2],$A[0][2],$A[1][2]+ eor $C[3],$A[0][3],$A[1][3]+___+ $C[4]=$A[0][4];+ $C[5]=$A[1][4];+$code.=<<___;+ eor $C[4],$A[0][4],$A[1][4]+ eor $C[0],$C[0],$A[2][0]+ eor $C[1],$C[1],$A[2][1]+ eor $C[2],$C[2],$A[2][2]+ eor $C[3],$C[3],$A[2][3]+ eor $C[4],$C[4],$A[2][4]+ eor $C[0],$C[0],$A[3][0]+ eor $C[1],$C[1],$A[3][1]+ eor $C[2],$C[2],$A[3][2]+ eor $C[3],$C[3],$A[3][3]+ eor $C[4],$C[4],$A[3][4]+ eor $C[0],$C[0],$A[4][0]+ eor $C[2],$C[2],$A[4][2]+ eor $C[1],$C[1],$A[4][1]+ eor $C[3],$C[3],$A[4][3]+ eor $C[4],$C[4],$A[4][4]++ eor $C[5],$C[0],$C[2],ror#63++ eor $A[0][1],$A[0][1],$C[5]+ eor $A[1][1],$A[1][1],$C[5]+ eor $A[2][1],$A[2][1],$C[5]+ eor $A[3][1],$A[3][1],$C[5]+ eor $A[4][1],$A[4][1],$C[5]++ eor $C[5],$C[1],$C[3],ror#63+ eor $C[2],$C[2],$C[4],ror#63+ eor $C[3],$C[3],$C[0],ror#63+ eor $C[4],$C[4],$C[1],ror#63++ eor $C[1], $A[0][2],$C[5] // mov $C[1],$A[0][2]+ eor $A[1][2],$A[1][2],$C[5]+ eor $A[2][2],$A[2][2],$C[5]+ eor $A[3][2],$A[3][2],$C[5]+ eor $A[4][2],$A[4][2],$C[5]++ eor $A[0][0],$A[0][0],$C[4]+ eor $A[1][0],$A[1][0],$C[4]+ eor $A[2][0],$A[2][0],$C[4]+ eor $A[3][0],$A[3][0],$C[4]+ eor $A[4][0],$A[4][0],$C[4]+___+ $C[4]=undef;+ $C[5]=undef;+$code.=<<___;+ ldp $A[0][4],$A[1][4],[sp,#0] // re-load offloaded data+ eor $C[0], $A[0][3],$C[2] // mov $C[0],$A[0][3]+ eor $A[1][3],$A[1][3],$C[2]+ eor $A[2][3],$A[2][3],$C[2]+ eor $A[3][3],$A[3][3],$C[2]+ eor $A[4][3],$A[4][3],$C[2]++ eor $C[2], $A[0][4],$C[3] // mov $C[2],$A[0][4]+ eor $A[1][4],$A[1][4],$C[3]+ eor $A[2][4],$A[2][4],$C[3]+ eor $A[3][4],$A[3][4],$C[3]+ eor $A[4][4],$A[4][4],$C[3]++ ////////////////////////////////////////// Rho+Pi+ mov $C[3],$A[0][1]+ ror $A[0][1],$A[1][1],#64-$rhotates[1][1]+ //mov $C[1],$A[0][2]+ ror $A[0][2],$A[2][2],#64-$rhotates[2][2]+ //mov $C[0],$A[0][3]+ ror $A[0][3],$A[3][3],#64-$rhotates[3][3] // ?+ //mov $C[2],$A[0][4]+ ror $A[0][4],$A[4][4],#64-$rhotates[4][4] // ?++ ror $A[1][1],$A[1][4],#64-$rhotates[1][4] // ?+ ror $A[2][2],$A[2][3],#64-$rhotates[2][3] // ?+ ror $A[3][3],$A[3][2],#64-$rhotates[3][2]+ ror $A[4][4],$A[4][1],#64-$rhotates[4][1] // ?++ ror $A[1][4],$A[4][2],#64-$rhotates[4][2]+ ror $A[2][3],$A[3][4],#64-$rhotates[3][4]+ ror $A[3][2],$A[2][1],#64-$rhotates[2][1]+ ror $A[4][1],$A[1][3],#64-$rhotates[1][3]++ ror $A[4][2],$A[2][4],#64-$rhotates[2][4]+ ror $A[3][4],$A[4][3],#64-$rhotates[4][3]+ ror $A[2][1],$A[1][2],#64-$rhotates[1][2] // ?+ ror $A[1][3],$A[3][1],#64-$rhotates[3][1]++ ror $A[2][4],$A[4][0],#64-$rhotates[4][0]+ ror $A[4][3],$A[3][0],#64-$rhotates[3][0]+ ror $A[1][2],$A[2][0],#64-$rhotates[2][0]+ ror $A[3][1],$A[1][0],#64-$rhotates[1][0] // ?++ ror $A[1][0],$C[0],#64-$rhotates[0][3] // ?+ ror $A[2][0],$C[3],#64-$rhotates[0][1]+ ror $A[3][0],$C[2],#64-$rhotates[0][4] // ?+ ror $A[4][0],$C[1],#64-$rhotates[0][2] // ?++ ////////////////////////////////////////// Chi+Iota+ bic $C[0],$A[0][2],$A[0][1]+ bic $C[1],$A[0][3],$A[0][2]+ bic $C[2],$A[0][0],$A[0][4]+ bic $C[3],$A[0][1],$A[0][0]+ eor $A[0][0],$A[0][0],$C[0]+ bic $C[0],$A[0][4],$A[0][3]+ eor $A[0][1],$A[0][1],$C[1]+ ldr c#$C[1],[csp,#16]+ eor $A[0][3],$A[0][3],$C[2]+ eor $A[0][4],$A[0][4],$C[3]+ eor $A[0][2],$A[0][2],$C[0]+ ldr $C[3],[$C[1]],#8 // Iota[i++]++ bic $C[0],$A[1][2],$A[1][1]+ tst $C[1],#255 // are we done?+ str c#$C[1],[csp,#16]+ bic $C[1],$A[1][3],$A[1][2]+ bic $C[2],$A[1][0],$A[1][4]+ eor $A[0][0],$A[0][0],$C[3] // A[0][0] ^= Iota+ bic $C[3],$A[1][1],$A[1][0]+ eor $A[1][0],$A[1][0],$C[0]+ bic $C[0],$A[1][4],$A[1][3]+ eor $A[1][1],$A[1][1],$C[1]+ eor $A[1][3],$A[1][3],$C[2]+ eor $A[1][4],$A[1][4],$C[3]+ eor $A[1][2],$A[1][2],$C[0]++ bic $C[0],$A[2][2],$A[2][1]+ bic $C[1],$A[2][3],$A[2][2]+ bic $C[2],$A[2][0],$A[2][4]+ bic $C[3],$A[2][1],$A[2][0]+ eor $A[2][0],$A[2][0],$C[0]+ bic $C[0],$A[2][4],$A[2][3]+ eor $A[2][1],$A[2][1],$C[1]+ eor $A[2][3],$A[2][3],$C[2]+ eor $A[2][4],$A[2][4],$C[3]+ eor $A[2][2],$A[2][2],$C[0]++ bic $C[0],$A[3][2],$A[3][1]+ bic $C[1],$A[3][3],$A[3][2]+ bic $C[2],$A[3][0],$A[3][4]+ bic $C[3],$A[3][1],$A[3][0]+ eor $A[3][0],$A[3][0],$C[0]+ bic $C[0],$A[3][4],$A[3][3]+ eor $A[3][1],$A[3][1],$C[1]+ eor $A[3][3],$A[3][3],$C[2]+ eor $A[3][4],$A[3][4],$C[3]+ eor $A[3][2],$A[3][2],$C[0]++ bic $C[0],$A[4][2],$A[4][1]+ bic $C[1],$A[4][3],$A[4][2]+ bic $C[2],$A[4][0],$A[4][4]+ bic $C[3],$A[4][1],$A[4][0]+ eor $A[4][0],$A[4][0],$C[0]+ bic $C[0],$A[4][4],$A[4][3]+ eor $A[4][1],$A[4][1],$C[1]+ eor $A[4][3],$A[4][3],$C[2]+ eor $A[4][4],$A[4][4],$C[3]+ eor $A[4][2],$A[4][2],$C[0]++ bne .Loop++ ldr c30,[csp,#16+__SIZEOF_POINTER__]+ .inst 0xd50323bf // autiasp+ ret+.size KeccakF1600_int,.-KeccakF1600_int++.type KeccakF1600,%function+.align 5+KeccakF1600:+ .inst 0xd503233f // paciasp+ stp c29,c30,[csp,#-16*__SIZEOF_POINTER__]!+ add c29,csp,#0+ stp c19,c20,[csp,#2*__SIZEOF_POINTER__]+ stp c21,c22,[csp,#4*__SIZEOF_POINTER__]+ stp c23,c24,[csp,#6*__SIZEOF_POINTER__]+ stp c25,c26,[csp,#8*__SIZEOF_POINTER__]+ stp c27,c28,[csp,#10*__SIZEOF_POINTER__]+ sub csp,csp,#16+4*__SIZEOF_POINTER__++ str c0,[csp,#16+2*__SIZEOF_POINTER__] // offload argument+ mov c#$C[0],c0+ ldp $A[0][0],$A[0][1],[x0,#16*0]+ ldp $A[0][2],$A[0][3],[$C[0],#16*1]+ ldp $A[0][4],$A[1][0],[$C[0],#16*2]+ ldp $A[1][1],$A[1][2],[$C[0],#16*3]+ ldp $A[1][3],$A[1][4],[$C[0],#16*4]+ ldp $A[2][0],$A[2][1],[$C[0],#16*5]+ ldp $A[2][2],$A[2][3],[$C[0],#16*6]+ ldp $A[2][4],$A[3][0],[$C[0],#16*7]+ ldp $A[3][1],$A[3][2],[$C[0],#16*8]+ ldp $A[3][3],$A[3][4],[$C[0],#16*9]+ ldp $A[4][0],$A[4][1],[$C[0],#16*10]+ ldp $A[4][2],$A[4][3],[$C[0],#16*11]+ ldr $A[4][4],[$C[0],#16*12]++ adr $C[2],iotas+ bl KeccakF1600_int++ ldr c#$C[0],[csp,#16+2*__SIZEOF_POINTER__]+ stp $A[0][0],$A[0][1],[$C[0],#16*0]+ stp $A[0][2],$A[0][3],[$C[0],#16*1]+ stp $A[0][4],$A[1][0],[$C[0],#16*2]+ stp $A[1][1],$A[1][2],[$C[0],#16*3]+ stp $A[1][3],$A[1][4],[$C[0],#16*4]+ stp $A[2][0],$A[2][1],[$C[0],#16*5]+ stp $A[2][2],$A[2][3],[$C[0],#16*6]+ stp $A[2][4],$A[3][0],[$C[0],#16*7]+ stp $A[3][1],$A[3][2],[$C[0],#16*8]+ stp $A[3][3],$A[3][4],[$C[0],#16*9]+ stp $A[4][0],$A[4][1],[$C[0],#16*10]+ stp $A[4][2],$A[4][3],[$C[0],#16*11]+ str $A[4][4],[$C[0],#16*12]++ ldp c19,c20,[c29,#2*__SIZEOF_POINTER__]+ add csp,csp,#16+4*__SIZEOF_POINTER__+ ldp c21,c22,[c29,#4*__SIZEOF_POINTER__]+ ldp c23,c24,[c29,#6*__SIZEOF_POINTER__]+ ldp c25,c26,[c29,#8*__SIZEOF_POINTER__]+ ldp c27,c28,[c29,#10*__SIZEOF_POINTER__]+ ldp c29,c30,[csp],#16*__SIZEOF_POINTER__+ .inst 0xd50323bf // autiasp+ ret+.size KeccakF1600,.-KeccakF1600++.globl SHA3_absorb+.type SHA3_absorb,%function+.align 5+SHA3_absorb:+ .inst 0xd503233f // paciasp+ stp c29,c30,[csp,#-16*__SIZEOF_POINTER__]!+ add c29,csp,#0+ stp c19,c20,[csp,#2*__SIZEOF_POINTER__]+ stp c21,c22,[csp,#4*__SIZEOF_POINTER__]+ stp c23,c24,[csp,#6*__SIZEOF_POINTER__]+ stp c25,c26,[csp,#8*__SIZEOF_POINTER__]+ stp c27,c28,[csp,#10*__SIZEOF_POINTER__]+ sub csp,csp,#16+4*__SIZEOF_POINTER__+16++ stp c0,c1,[csp,#16+2*__SIZEOF_POINTER__] // offload arguments+ stp x2,x3,[csp,#16+4*__SIZEOF_POINTER__]++ mov c#$C[0],c0 // uint64_t A[5][5]+ mov c#$C[1],c1 // const void *inp+ mov $C[2],x2 // size_t len+ mov $C[3],x3 // size_t bsz+ ldp $A[0][0],$A[0][1],[$C[0],#16*0]+ ldp $A[0][2],$A[0][3],[$C[0],#16*1]+ ldp $A[0][4],$A[1][0],[$C[0],#16*2]+ ldp $A[1][1],$A[1][2],[$C[0],#16*3]+ ldp $A[1][3],$A[1][4],[$C[0],#16*4]+ ldp $A[2][0],$A[2][1],[$C[0],#16*5]+ ldp $A[2][2],$A[2][3],[$C[0],#16*6]+ ldp $A[2][4],$A[3][0],[$C[0],#16*7]+ ldp $A[3][1],$A[3][2],[$C[0],#16*8]+ ldp $A[3][3],$A[3][4],[$C[0],#16*9]+ ldp $A[4][0],$A[4][1],[$C[0],#16*10]+ ldp $A[4][2],$A[4][3],[$C[0],#16*11]+ ldr $A[4][4],[$C[0],#16*12]+ b .Loop_absorb++.align 4+.Loop_absorb:+ subs $C[0],$C[2],$C[3] // len - bsz+ blo .Labsorbed++ str $C[0],[csp,#16+4*__SIZEOF_POINTER__] // save len - bsz+ cmp $C[3],#104+___+sub load_n_xor {+ my ($from,$to) = @_;++ for (my $i=$from; $i<=$to; $i++) {+$code.=<<___;+ ldr $C[0],[$C[1],#`8*$i`] // A[`$i/5`][`$i%5`] ^= *inp+++#ifdef __AARCH64EB__+ rev $C[0],$C[0]+#endif+ eor $A[$i/5][$i%5],$A[$i/5][$i%5],$C[0]+___+ }+}+load_n_xor(0,8);+$code.=<<___;+ blo .Lprocess_block++___+load_n_xor(9,12);+$code.=<<___;+ beq .Lprocess_block++ cmp $C[3],#144+___+load_n_xor(13,16);+$code.=<<___;+ blo .Lprocess_block++___+load_n_xor(17,17);+$code.=<<___;+ beq .Lprocess_block++___+load_n_xor(18,20);+$code.=<<___;++.Lprocess_block:+ add c#$C[1],c#@C[1],@C[3]+ str c#$C[1],[csp,#16+3*__SIZEOF_POINTER__] // save inp++ adr $C[2],iotas+ bl KeccakF1600_int++ ldr c#$C[1],[csp,#16+3*__SIZEOF_POINTER__] // restore arguments+ ldp $C[2],$C[3],[csp,#16+4*__SIZEOF_POINTER__]+ b .Loop_absorb++.align 4+.Labsorbed:+ ldr c#$C[1],[sp,#16+2*__SIZEOF_POINTER__]+ stp $A[0][0],$A[0][1],[$C[1],#16*0]+ stp $A[0][2],$A[0][3],[$C[1],#16*1]+ stp $A[0][4],$A[1][0],[$C[1],#16*2]+ stp $A[1][1],$A[1][2],[$C[1],#16*3]+ stp $A[1][3],$A[1][4],[$C[1],#16*4]+ stp $A[2][0],$A[2][1],[$C[1],#16*5]+ stp $A[2][2],$A[2][3],[$C[1],#16*6]+ stp $A[2][4],$A[3][0],[$C[1],#16*7]+ stp $A[3][1],$A[3][2],[$C[1],#16*8]+ stp $A[3][3],$A[3][4],[$C[1],#16*9]+ stp $A[4][0],$A[4][1],[$C[1],#16*10]+ stp $A[4][2],$A[4][3],[$C[1],#16*11]+ str $A[4][4],[$C[1],#16*12]++ mov x0,$C[2] // return value+ ldp c19,c20,[c29,#2*__SIZEOF_POINTER__]+ add csp,csp,#16+4*__SIZEOF_POINTER__+16+ ldp c21,c22,[c29,#4*__SIZEOF_POINTER__]+ ldp c23,c24,[c29,#6*__SIZEOF_POINTER__]+ ldp c25,c26,[c29,#8*__SIZEOF_POINTER__]+ ldp c27,c28,[c29,#10*__SIZEOF_POINTER__]+ ldp c29,c30,[csp],#16*__SIZEOF_POINTER__+ .inst 0xd50323bf // autiasp+ ret+.size SHA3_absorb,.-SHA3_absorb+___+{+my ($A_flat,$out,$len,$bsz) = map("x$_",(19..22));+$code.=<<___;+.globl SHA3_squeeze+.type SHA3_squeeze,%function+.align 5+SHA3_squeeze:+ .inst 0xd503233f // paciasp+ stp c29,c30,[csp,#-6*__SIZEOF_POINTER__]!+ add c29,csp,#0+ stp c19,c20,[csp,#2*__SIZEOF_POINTER__]+ stp c21,c22,[csp,#4*__SIZEOF_POINTER__]++ cmov $A_flat,x0 // put aside arguments+ cmov $out,x1+ mov $len,x2+ mov $bsz,x3++.Loop_squeeze:+ ldr x4,[x0],#8+ cmp $len,#8+ blo .Lsqueeze_tail+#ifdef __AARCH64EB__+ rev x4,x4+#endif+ str x4,[$out],#8+ subs $len,$len,#8+ beq .Lsqueeze_done++ subs x3,x3,#8+ bhi .Loop_squeeze++ cmov x0,$A_flat+ bl KeccakF1600+ cmov x0,$A_flat+ mov x3,$bsz+ b .Loop_squeeze++.align 4+.Lsqueeze_tail:+ strb w4,[$out],#1+ lsr x4,x4,#8+ subs $len,$len,#1+ beq .Lsqueeze_done+ strb w4,[$out],#1+ lsr x4,x4,#8+ subs $len,$len,#1+ beq .Lsqueeze_done+ strb w4,[$out],#1+ lsr x4,x4,#8+ subs $len,$len,#1+ beq .Lsqueeze_done+ strb w4,[$out],#1+ lsr x4,x4,#8+ subs $len,$len,#1+ beq .Lsqueeze_done+ strb w4,[$out],#1+ lsr x4,x4,#8+ subs $len,$len,#1+ beq .Lsqueeze_done+ strb w4,[$out],#1+ lsr x4,x4,#8+ subs $len,$len,#1+ beq .Lsqueeze_done+ strb w4,[$out],#1++.Lsqueeze_done:+ ldp c19,c20,[csp,#2*__SIZEOF_POINTER__]+ ldp c21,c22,[csp,#4*__SIZEOF_POINTER__]+ ldp c29,c30,[csp],#6*__SIZEOF_POINTER__+ .inst 0xd50323bf // autiasp+ ret+.size SHA3_squeeze,.-SHA3_squeeze+___+} }}}+ {{{+my @A = map([ "v".$_.".16b", "v".($_+1).".16b", "v".($_+2).".16b",+ "v".($_+3).".16b", "v".($_+4).".16b" ],+ (0, 5, 10, 15, 20));++my @C = map("v$_.16b", (25..31));+my @D = @C[4,5,6,2,3];++$code.=<<___;+.type KeccakF1600_ce,%function+.align 5+KeccakF1600_ce:+.Loop_ce:+ ////////////////////////////////////////////////// Theta+ eor3 $C[0],$A[4][0],$A[3][0],$A[2][0]+ eor3 $C[1],$A[4][1],$A[3][1],$A[2][1]+ eor3 $C[2],$A[4][2],$A[3][2],$A[2][2]+ eor3 $C[3],$A[4][3],$A[3][3],$A[2][3]+ eor3 $C[4],$A[4][4],$A[3][4],$A[2][4]+ eor3 $C[0],$C[0], $A[1][0],$A[0][0]+ eor3 $C[1],$C[1], $A[1][1],$A[0][1]+ eor3 $C[2],$C[2], $A[1][2],$A[0][2]+ eor3 $C[3],$C[3], $A[1][3],$A[0][3]+ eor3 $C[4],$C[4], $A[1][4],$A[0][4]++ rax1 $C[5],$C[0],$C[2] // D[1]+ rax1 $C[6],$C[1],$C[3] // D[2]+ rax1 $C[2],$C[2],$C[4] // D[3]+ rax1 $C[3],$C[3],$C[0] // D[4]+ rax1 $C[4],$C[4],$C[1] // D[0]++ ////////////////////////////////////////////////// Theta+Rho+Pi+ xar $C[0], $A[0][1],$D[1],#64-$rhotates[0][1] // C[0]=A[2][0]++ xar $A[0][1],$A[1][1],$D[1],#64-$rhotates[1][1]+ xar $A[1][1],$A[1][4],$D[4],#64-$rhotates[1][4]+ xar $A[1][4],$A[4][2],$D[2],#64-$rhotates[4][2]+ xar $A[4][2],$A[2][4],$D[4],#64-$rhotates[2][4]+ xar $A[2][4],$A[4][0],$D[0],#64-$rhotates[4][0]++ xar $C[1], $A[0][2],$D[2],#64-$rhotates[0][2] // C[1]=A[4][0]++ xar $A[0][2],$A[2][2],$D[2],#64-$rhotates[2][2]+ xar $A[2][2],$A[2][3],$D[3],#64-$rhotates[2][3]+ xar $A[2][3],$A[3][4],$D[4],#64-$rhotates[3][4]+ xar $A[3][4],$A[4][3],$D[3],#64-$rhotates[4][3]+ xar $A[4][3],$A[3][0],$D[0],#64-$rhotates[3][0]++ xar $A[3][0],$A[0][4],$D[4],#64-$rhotates[0][4]++ xar $D[4], $A[4][4],$D[4],#64-$rhotates[4][4] // D[4]=A[0][4]+ xar $A[4][4],$A[4][1],$D[1],#64-$rhotates[4][1]+ xar $A[1][3],$A[1][3],$D[3],#64-$rhotates[1][3] // A[1][3]=A[4][1]+ xar $A[0][4],$A[3][1],$D[1],#64-$rhotates[3][1] // A[0][4]=A[1][3]+ xar $A[3][1],$A[1][0],$D[0],#64-$rhotates[1][0]++ xar $A[1][0],$A[0][3],$D[3],#64-$rhotates[0][3]++ eor $A[0][0],$A[0][0],$D[0]++ xar $D[3], $A[3][3],$D[3],#64-$rhotates[3][3] // D[3]=A[0][3]+ xar $A[0][3],$A[3][2],$D[2],#64-$rhotates[3][2] // A[0][3]=A[3][3]+ xar $D[1], $A[2][1],$D[1],#64-$rhotates[2][1] // D[1]=A[3][2]+ xar $D[2], $A[1][2],$D[2],#64-$rhotates[1][2] // D[2]=A[2][1]+ xar $D[0], $A[2][0],$D[0],#64-$rhotates[2][0] // D[0]=A[1][2]++ ////////////////////////////////////////////////// Chi+Iota+ bcax $A[4][0],$C[1], $A[4][2],$A[1][3] // A[1][3]=A[4][1]+ bcax $A[4][1],$A[1][3],$A[4][3],$A[4][2] // A[1][3]=A[4][1]+ bcax $A[4][2],$A[4][2],$A[4][4],$A[4][3]+ bcax $A[4][3],$A[4][3],$C[1], $A[4][4]+ bcax $A[4][4],$A[4][4],$A[1][3],$C[1] // A[1][3]=A[4][1]++ ld1r {$C[1]},[x10],#8++ bcax $A[3][2],$D[1], $A[3][4],$A[0][3] // A[0][3]=A[3][3]+ bcax $A[3][3],$A[0][3],$A[3][0],$A[3][4] // A[0][3]=A[3][3]+ bcax $A[3][4],$A[3][4],$A[3][1],$A[3][0]+ bcax $A[3][0],$A[3][0],$D[1], $A[3][1]+ bcax $A[3][1],$A[3][1],$A[0][3],$D[1] // A[0][3]=A[3][3]++ bcax $A[2][0],$C[0], $A[2][2],$D[2]+ bcax $A[2][1],$D[2], $A[2][3],$A[2][2]+ bcax $A[2][2],$A[2][2],$A[2][4],$A[2][3]+ bcax $A[2][3],$A[2][3],$C[0], $A[2][4]+ bcax $A[2][4],$A[2][4],$D[2], $C[0]++ bcax $A[1][2],$D[0], $A[1][4],$A[0][4] // A[0][4]=A[1][3]+ bcax $A[1][3],$A[0][4],$A[1][0],$A[1][4] // A[0][4]=A[1][3]+ bcax $A[1][4],$A[1][4],$A[1][1],$A[1][0]+ bcax $A[1][0],$A[1][0],$D[0], $A[1][1]+ bcax $A[1][1],$A[1][1],$A[0][4],$D[0] // A[0][4]=A[1][3]++ bcax $A[0][3],$D[3], $A[0][0],$D[4]+ bcax $A[0][4],$D[4], $A[0][1],$A[0][0]+ bcax $A[0][0],$A[0][0],$A[0][2],$A[0][1]+ bcax $A[0][1],$A[0][1],$D[3], $A[0][2]+ bcax $A[0][2],$A[0][2],$D[4], $D[3]++ eor $A[0][0],$A[0][0],$C[1]++ tst x10,#255+ bne .Loop_ce++ ret+.size KeccakF1600_ce,.-KeccakF1600_ce++.type KeccakF1600_cext,%function+.align 5+KeccakF1600_cext:+ .inst 0xd503233f // paciasp+ stp c29,c30,[csp,#-2*__SIZEOF_POINTER__-64]!+ add c29,csp,#0+ stp d8,d9,[csp,#2*__SIZEOF_POINTER__+0] // per ABI requirement+ stp d10,d11,[csp,#2*__SIZEOF_POINTER__+16]+ stp d12,d13,[csp,#2*__SIZEOF_POINTER__+32]+ stp d14,d15,[csp,#2*__SIZEOF_POINTER__+48]+___+for($i=0; $i<24; $i+=2) { # load A[5][5]+my $j=$i+1;+$code.=<<___;+ ldp d$i,d$j,[x0,#8*$i]+___+}+$code.=<<___;+ ldr d24,[x0,#8*$i]+ adr x10,iotas+ bl KeccakF1600_ce+ ldr c30,[csp,#__SIZEOF_POINTER__]+___+for($i=0; $i<24; $i+=2) { # store A[5][5]+my $j=$i+1;+$code.=<<___;+ stp d$i,d$j,[x0,#8*$i]+___+}+$code.=<<___;+ str d24,[x0,#8*$i]++ ldp d8,d9,[csp,#2*__SIZEOF_POINTER__+0]+ ldp d10,d11,[csp,#2*__SIZEOF_POINTER__+16]+ ldp d12,d13,[csp,#2*__SIZEOF_POINTER__+32]+ ldp d14,d15,[csp,#2*__SIZEOF_POINTER__+48]+ ldr c29,[csp],#2*__SIZEOF_POINTER__+64+ .inst 0xd50323bf // autiasp+ ret+.size KeccakF1600_cext,.-KeccakF1600_cext+___++{+my ($ctx,$inp,$len,$bsz) = map("x$_",(0..3));++$code.=<<___;+.globl SHA3_absorb_cext+.type SHA3_absorb_cext,%function+.align 5+SHA3_absorb_cext:+ .inst 0xd503233f // paciasp+ stp c29,c30,[csp,#-2*__SIZEOF_POINTER__-64]!+ add c29,csp,#0+ stp d8,d9,[csp,#2*__SIZEOF_POINTER__+0] // per ABI requirement+ stp d10,d11,[csp,#2*__SIZEOF_POINTER__+16]+ stp d12,d13,[csp,#2*__SIZEOF_POINTER__+32]+ stp d14,d15,[csp,#2*__SIZEOF_POINTER__+48]+___+for($i=0; $i<24; $i+=2) { # load A[5][5]+my $j=$i+1;+$code.=<<___;+ ldp d$i,d$j,[x0,#8*$i]+___+}+$code.=<<___;+ ldr d24,[x0,#8*$i]+ b .Loop_absorb_ce++.align 4+.Loop_absorb_ce:+ subs $len,$len,$bsz // len - bsz+ blo .Labsorbed_ce++ cmp $bsz,#104+___+sub load_n_xor_ce {+ my ($from,$to) = @_;+ my $range = $to-$from+1;++ while ($range>=4) {+$code.=<<___;+ ld1 {v27.8b-v30.8b},[$inp],#32+ eor $A[$from/5][$from%5],$A[$from/5][$from++%5],v27.16b+ eor $A[$from/5][$from%5],$A[$from/5][$from++%5],v28.16b+ eor $A[$from/5][$from%5],$A[$from/5][$from++%5],v29.16b+ eor $A[$from/5][$from%5],$A[$from/5][$from++%5],v30.16b+___+ $range-=4;+ }+ while ($range>=3) {+$code.=<<___;+ ld1 {v28.8b-v30.8b},[$inp],#24+ eor $A[$from/5][$from%5],$A[$from/5][$from++%5],v28.16b+ eor $A[$from/5][$from%5],$A[$from/5][$from++%5],v29.16b+ eor $A[$from/5][$from%5],$A[$from/5][$from++%5],v30.16b+___+ $range-=3;+ }+ while ($from<=$to) {+$code.=<<___;+ ld1 {v31.8b},[$inp],#8 // A[`$from/5`][`$from%5`] ^= *inp+++ eor $A[$from/5][$from%5],$A[$from/5][$from++%5],v31.16b+___+ }+}+load_n_xor_ce(0,8);+$code.=<<___;+ blo .Lprocess_block_ce++___+load_n_xor_ce(9,12);+$code.=<<___;+ beq .Lprocess_block_ce++ cmp $bsz,#144+___+load_n_xor_ce(13,16);+$code.=<<___;+ blo .Lprocess_block_ce++___+load_n_xor_ce(17,17);+$code.=<<___;+ beq .Lprocess_block_ce++___+load_n_xor_ce(18,20);+$code.=<<___;++.Lprocess_block_ce:+ adr x10,iotas+ bl KeccakF1600_ce++ b .Loop_absorb_ce++.align 4+.Labsorbed_ce:+___+for($i=0; $i<24; $i+=2) { # store A[5][5]+my $j=$i+1;+$code.=<<___;+ stp d$i,d$j,[x0,#8*$i]+___+}+$code.=<<___;+ str d24,[x0,#8*$i]+ add x0,$len,$bsz // return value++ ldp d8,d9,[csp,#2*__SIZEOF_POINTER__+0]+ ldp d10,d11,[csp,#2*__SIZEOF_POINTER__+16]+ ldp d12,d13,[csp,#2*__SIZEOF_POINTER__+32]+ ldp d14,d15,[csp,#2*__SIZEOF_POINTER__+48]+ ldp c29,c30,[csp],#2*__SIZEOF_POINTER__+64+ .inst 0xd50323bf // autiasp+ ret+.size SHA3_absorb_cext,.-SHA3_absorb_cext+___+}+{+my ($ctx,$out,$len,$bsz) = map("x$_",(0..3));+$code.=<<___;+.globl SHA3_squeeze_cext+.type SHA3_squeeze_cext,%function+.align 5+SHA3_squeeze_cext:+ .inst 0xd503233f // paciasp+ stp c29,c30,[csp,#-2*__SIZEOF_POINTER__]!+ add c29,csp,#0+ cmov x9,$ctx+ mov x10,$bsz++.Loop_squeeze_ce:+ ldr x4,[x9],#8+ cmp $len,#8+ blo .Lsqueeze_tail_ce+#ifdef __AARCH64EB__+ rev x4,x4+#endif+ str x4,[$out],#8+ beq .Lsqueeze_done_ce++ sub $len,$len,#8+ subs x10,x10,#8+ bhi .Loop_squeeze_ce++ bl KeccakF1600_cext+ ldr c30,[csp,#__SIZEOF_POINTER__]+ cmov x9,$ctx+ mov x10,$bsz+ b .Loop_squeeze_ce++.align 4+.Lsqueeze_tail_ce:+ strb w4,[$out],#1+ lsr x4,x4,#8+ subs $len,$len,#1+ beq .Lsqueeze_done_ce+ strb w4,[$out],#1+ lsr x4,x4,#8+ subs $len,$len,#1+ beq .Lsqueeze_done_ce+ strb w4,[$out],#1+ lsr x4,x4,#8+ subs $len,$len,#1+ beq .Lsqueeze_done_ce+ strb w4,[$out],#1+ lsr x4,x4,#8+ subs $len,$len,#1+ beq .Lsqueeze_done_ce+ strb w4,[$out],#1+ lsr x4,x4,#8+ subs $len,$len,#1+ beq .Lsqueeze_done_ce+ strb w4,[$out],#1+ lsr x4,x4,#8+ subs $len,$len,#1+ beq .Lsqueeze_done_ce+ strb w4,[$out],#1++.Lsqueeze_done_ce:+ ldr c29,[csp],#2*__SIZEOF_POINTER__+ .inst 0xd50323bf // autiasp+ ret+.size SHA3_squeeze_cext,.-SHA3_squeeze_cext+___+} }}}+$code.=<<___;+.asciz "Keccak-1600 absorb and squeeze for ARMv8, CRYPTOGAMS by \@dot-asm"+___++{ my %opcode = (+ "rax1" => 0xce608c00, "eor3" => 0xce000000,+ "bcax" => 0xce200000, "xar" => 0xce800000 );++ sub unsha3 {+ my ($mnemonic,$arg)=@_;++ $arg =~ m/[qv]([0-9]+)[^,]*,\s*[qv]([0-9]+)[^,]*(?:,\s*[qv]([0-9]+)[^,]*(?:,\s*[qv#]([0-9\-]+))?)?/+ &&+ sprintf ".inst\t0x%08x\t//%s %s",+ $opcode{$mnemonic}|$1|($2<<5)|($3<<16)|(eval($4)<<10),+ $mnemonic,$arg;+ }+}++foreach(split("\n",$code)) {+ use integer;++ s/\`([^\`]*)\`/eval($1)/ge;++ m/\b(ld1r|rax1|xar)\b/ and s/\.16b/.2d/g;+ $sha3ops or s/\b(eor3|rax1|xar|bcax)\s+(v.*)/unsha3($1,$2)/ge;+ s/([cw])#x([0-9]+)/$1$2/g;++ print $_,"\n";+}++close STDOUT;
@@ -0,0 +1,538 @@+.text ++.type __crypton_keccak_asm_f1600,@function+.align 32+__crypton_keccak_asm_f1600:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa++ movq 60(%rdi),%rax+ movq 68(%rdi),%rbx+ movq 76(%rdi),%rcx+ movq 84(%rdi),%rdx+ movq 92(%rdi),%rbp+ jmp .Loop++.align 32+.Loop:+ movq -100(%rdi),%r8+ movq -52(%rdi),%r9+ movq -4(%rdi),%r10+ movq 44(%rdi),%r11++ xorq -84(%rdi),%rcx+ xorq -76(%rdi),%rdx+ xorq %r8,%rax+ xorq -92(%rdi),%rbx+ xorq -44(%rdi),%rcx+ xorq -60(%rdi),%rax+ movq %rbp,%r12+ xorq -68(%rdi),%rbp++ xorq %r10,%rcx+ xorq -20(%rdi),%rax+ xorq -36(%rdi),%rdx+ xorq %r9,%rbx+ xorq -28(%rdi),%rbp++ xorq 36(%rdi),%rcx+ xorq 20(%rdi),%rax+ xorq 4(%rdi),%rdx+ xorq -12(%rdi),%rbx+ xorq 12(%rdi),%rbp++ movq %rcx,%r13+ rolq $1,%rcx+ xorq %rax,%rcx+ xorq %r11,%rdx++ rolq $1,%rax+ xorq %rdx,%rax+ xorq 28(%rdi),%rbx++ rolq $1,%rdx+ xorq %rbx,%rdx+ xorq 52(%rdi),%rbp++ rolq $1,%rbx+ xorq %rbp,%rbx++ rolq $1,%rbp+ xorq %r13,%rbp+ xorq %rcx,%r9+ xorq %rdx,%r10+ rolq $44,%r9+ xorq %rbp,%r11+ xorq %rax,%r12+ rolq $43,%r10+ xorq %rbx,%r8+ movq %r9,%r13+ rolq $21,%r11+ orq %r10,%r9+ xorq %r8,%r9+ rolq $14,%r12++ xorq (%r15),%r9+ leaq 8(%r15),%r15++ movq %r12,%r14+ andq %r11,%r12+ movq %r9,-100(%rsi)+ xorq %r10,%r12+ notq %r10+ movq %r12,-84(%rsi)++ orq %r11,%r10+ movq 76(%rdi),%r12+ xorq %r13,%r10+ movq %r10,-92(%rsi)++ andq %r8,%r13+ movq -28(%rdi),%r9+ xorq %r14,%r13+ movq -20(%rdi),%r10+ movq %r13,-68(%rsi)++ orq %r8,%r14+ movq -76(%rdi),%r8+ xorq %r11,%r14+ movq 28(%rdi),%r11+ movq %r14,-76(%rsi)+++ xorq %rbp,%r8+ xorq %rdx,%r12+ rolq $28,%r8+ xorq %rcx,%r11+ xorq %rax,%r9+ rolq $61,%r12+ rolq $45,%r11+ xorq %rbx,%r10+ rolq $20,%r9+ movq %r8,%r13+ orq %r12,%r8+ rolq $3,%r10++ xorq %r11,%r8+ movq %r8,-36(%rsi)++ movq %r9,%r14+ andq %r13,%r9+ movq -92(%rdi),%r8+ xorq %r12,%r9+ notq %r12+ movq %r9,-28(%rsi)++ orq %r11,%r12+ movq -44(%rdi),%r9+ xorq %r10,%r12+ movq %r12,-44(%rsi)++ andq %r10,%r11+ movq 60(%rdi),%r12+ xorq %r14,%r11+ movq %r11,-52(%rsi)++ orq %r10,%r14+ movq 4(%rdi),%r10+ xorq %r13,%r14+ movq 52(%rdi),%r11+ movq %r14,-60(%rsi)+++ xorq %rbp,%r10+ xorq %rax,%r11+ rolq $25,%r10+ xorq %rdx,%r9+ rolq $8,%r11+ xorq %rbx,%r12+ rolq $6,%r9+ xorq %rcx,%r8+ rolq $18,%r12+ movq %r10,%r13+ andq %r11,%r10+ rolq $1,%r8++ notq %r11+ xorq %r9,%r10+ movq %r10,-12(%rsi)++ movq %r12,%r14+ andq %r11,%r12+ movq -12(%rdi),%r10+ xorq %r13,%r12+ movq %r12,-4(%rsi)++ orq %r9,%r13+ movq 84(%rdi),%r12+ xorq %r8,%r13+ movq %r13,-20(%rsi)++ andq %r8,%r9+ xorq %r14,%r9+ movq %r9,12(%rsi)++ orq %r8,%r14+ movq -60(%rdi),%r9+ xorq %r11,%r14+ movq 36(%rdi),%r11+ movq %r14,4(%rsi)+++ movq -68(%rdi),%r8++ xorq %rcx,%r10+ xorq %rdx,%r11+ rolq $10,%r10+ xorq %rbx,%r9+ rolq $15,%r11+ xorq %rbp,%r12+ rolq $36,%r9+ xorq %rax,%r8+ rolq $56,%r12+ movq %r10,%r13+ orq %r11,%r10+ rolq $27,%r8++ notq %r11+ xorq %r9,%r10+ movq %r10,28(%rsi)++ movq %r12,%r14+ orq %r11,%r12+ xorq %r13,%r12+ movq %r12,36(%rsi)++ andq %r9,%r13+ xorq %r8,%r13+ movq %r13,20(%rsi)++ orq %r8,%r9+ xorq %r14,%r9+ movq %r9,52(%rsi)++ andq %r14,%r8+ xorq %r11,%r8+ movq %r8,44(%rsi)+++ xorq -84(%rdi),%rdx+ xorq -36(%rdi),%rbp+ rolq $62,%rdx+ xorq 68(%rdi),%rcx+ rolq $55,%rbp+ xorq 12(%rdi),%rax+ rolq $2,%rcx+ xorq 20(%rdi),%rbx+ xchgq %rsi,%rdi+ rolq $39,%rax+ rolq $41,%rbx+ movq %rdx,%r13+ andq %rbp,%rdx+ notq %rbp+ xorq %rcx,%rdx+ movq %rdx,92(%rdi)++ movq %rax,%r14+ andq %rbp,%rax+ xorq %r13,%rax+ movq %rax,60(%rdi)++ orq %rcx,%r13+ xorq %rbx,%r13+ movq %r13,84(%rdi)++ andq %rbx,%rcx+ xorq %r14,%rcx+ movq %rcx,76(%rdi)++ orq %r14,%rbx+ xorq %rbp,%rbx+ movq %rbx,68(%rdi)++ movq %rdx,%rbp+ movq %r13,%rdx++ testq $255,%r15+ jnz .Loop++ leaq -192(%r15),%r15+ .byte 0xf3,0xc3+.cfi_endproc+.size __crypton_keccak_asm_f1600,.-__crypton_keccak_asm_f1600++.globl crypton_keccak_asm_f1600+.type crypton_keccak_asm_f1600,@function+.align 32+crypton_keccak_asm_f1600:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa+++ pushq %rbx+.cfi_adjust_cfa_offset 8+.cfi_offset %rbx,-16+ pushq %rbp+.cfi_adjust_cfa_offset 8+.cfi_offset %rbp,-24+ pushq %r12+.cfi_adjust_cfa_offset 8+.cfi_offset %r12,-32+ pushq %r13+.cfi_adjust_cfa_offset 8+.cfi_offset %r13,-40+ pushq %r14+.cfi_adjust_cfa_offset 8+.cfi_offset %r14,-48+ pushq %r15+.cfi_adjust_cfa_offset 8+.cfi_offset %r15,-56++ leaq 100(%rdi),%rdi+ subq $200,%rsp+.cfi_adjust_cfa_offset 200+++ notq -92(%rdi)+ notq -84(%rdi)+ notq -36(%rdi)+ notq -4(%rdi)+ notq 36(%rdi)+ notq 60(%rdi)++ leaq iotas(%rip),%r15+ leaq 100(%rsp),%rsi++ call __crypton_keccak_asm_f1600++ notq -92(%rdi)+ notq -84(%rdi)+ notq -36(%rdi)+ notq -4(%rdi)+ notq 36(%rdi)+ notq 60(%rdi)+ leaq -100(%rdi),%rdi++ leaq 248(%rsp),%r11+.cfi_def_cfa %r11,8+ movq -48(%r11),%r15+ movq -40(%r11),%r14+ movq -32(%r11),%r13+ movq -24(%r11),%r12+ movq -16(%r11),%rbp+ movq -8(%r11),%rbx+ leaq (%r11),%rsp+.cfi_restore %r12+.cfi_restore %r13+.cfi_restore %r14+.cfi_restore %r15+.cfi_restore %rbp+.cfi_restore %rbx+ .byte 0xf3,0xc3+.cfi_endproc +.size crypton_keccak_asm_f1600,.-crypton_keccak_asm_f1600+.globl crypton_keccak_asm_absorb+.type crypton_keccak_asm_absorb,@function+.align 32+crypton_keccak_asm_absorb:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa+++ pushq %rbx+.cfi_adjust_cfa_offset 8+.cfi_offset %rbx,-16+ pushq %rbp+.cfi_adjust_cfa_offset 8+.cfi_offset %rbp,-24+ pushq %r12+.cfi_adjust_cfa_offset 8+.cfi_offset %r12,-32+ pushq %r13+.cfi_adjust_cfa_offset 8+.cfi_offset %r13,-40+ pushq %r14+.cfi_adjust_cfa_offset 8+.cfi_offset %r14,-48+ pushq %r15+.cfi_adjust_cfa_offset 8+.cfi_offset %r15,-56++ leaq 100(%rdi),%rdi+ subq $232,%rsp+.cfi_adjust_cfa_offset 232+++ movq %rsi,%r9+ leaq 100(%rsp),%rsi++ notq -92(%rdi)+ notq -84(%rdi)+ notq -36(%rdi)+ notq -4(%rdi)+ notq 36(%rdi)+ notq 60(%rdi)+ leaq iotas(%rip),%r15++ movq %rcx,216-100(%rsi)++.Loop_absorb:+ cmpq %rcx,%rdx+ jc .Ldone_absorb++ shrq $3,%rcx+ leaq -100(%rdi),%r8++.Lblock_absorb:+ movq (%r9),%rax+ leaq 8(%r9),%r9+ xorq (%r8),%rax+ leaq 8(%r8),%r8+ subq $8,%rdx+ movq %rax,-8(%r8)+ subq $1,%rcx+ jnz .Lblock_absorb++ movq %r9,200-100(%rsi)+ movq %rdx,208-100(%rsi)+ call __crypton_keccak_asm_f1600+ movq 200-100(%rsi),%r9+ movq 208-100(%rsi),%rdx+ movq 216-100(%rsi),%rcx+ jmp .Loop_absorb++.align 32+.Ldone_absorb:+ movq %rdx,%rax++ notq -92(%rdi)+ notq -84(%rdi)+ notq -36(%rdi)+ notq -4(%rdi)+ notq 36(%rdi)+ notq 60(%rdi)++ leaq 280(%rsp),%r11+.cfi_def_cfa %r11,8+ movq -48(%r11),%r15+ movq -40(%r11),%r14+ movq -32(%r11),%r13+ movq -24(%r11),%r12+ movq -16(%r11),%rbp+ movq -8(%r11),%rbx+ leaq (%r11),%rsp+.cfi_restore %r12+.cfi_restore %r13+.cfi_restore %r14+.cfi_restore %r15+.cfi_restore %rbp+.cfi_restore %rbx+ .byte 0xf3,0xc3+.cfi_endproc +.size crypton_keccak_asm_absorb,.-crypton_keccak_asm_absorb+.globl crypton_keccak_asm_squeeze+.type crypton_keccak_asm_squeeze,@function+.align 32+crypton_keccak_asm_squeeze:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa+++ pushq %r12+.cfi_adjust_cfa_offset 8+.cfi_offset %r12,-16+ pushq %r13+.cfi_adjust_cfa_offset 8+.cfi_offset %r13,-24+ pushq %r14+.cfi_adjust_cfa_offset 8+.cfi_offset %r14,-32+ subq $32,%rsp+.cfi_adjust_cfa_offset 32+++ shrq $3,%rcx+ movq %rdi,%r8+ movq %rsi,%r12+ movq %rdx,%r13+ movq %rcx,%r14+ jmp .Loop_squeeze++.align 32+.Loop_squeeze:+ cmpq $8,%r13+ jb .Ltail_squeeze++ movq (%r8),%rax+ leaq 8(%r8),%r8+ movq %rax,(%r12)+ leaq 8(%r12),%r12+ subq $8,%r13+ jz .Ldone_squeeze++ subq $1,%rcx+ jnz .Loop_squeeze++ movq %rdi,%rcx+ call crypton_keccak_asm_f1600+ movq %rdi,%r8+ movq %r14,%rcx+ jmp .Loop_squeeze++.Ltail_squeeze:+ movq %r8,%rsi+ movq %r12,%rdi+ movq %r13,%rcx+.byte 0xf3,0xa4++.Ldone_squeeze:+ movq 32(%rsp),%r14+ movq 40(%rsp),%r13+ movq 48(%rsp),%r12+ addq $56,%rsp+.cfi_adjust_cfa_offset -56+.cfi_restore %r12+.cfi_restore %r13+.cfi_restore %r14+ .byte 0xf3,0xc3+.cfi_endproc +.size crypton_keccak_asm_squeeze,.-crypton_keccak_asm_squeeze+.align 256+.quad 0,0,0,0,0,0,0,0+.type iotas,@object+iotas:+.quad 0x0000000000000001+.quad 0x0000000000008082+.quad 0x800000000000808a+.quad 0x8000000080008000+.quad 0x000000000000808b+.quad 0x0000000080000001+.quad 0x8000000080008081+.quad 0x8000000000008009+.quad 0x000000000000008a+.quad 0x0000000000000088+.quad 0x0000000080008009+.quad 0x000000008000000a+.quad 0x000000008000808b+.quad 0x800000000000008b+.quad 0x8000000000008089+.quad 0x8000000000008003+.quad 0x8000000000008002+.quad 0x8000000000000080+.quad 0x000000000000800a+.quad 0x800000008000000a+.quad 0x8000000080008081+.quad 0x8000000000008080+.quad 0x0000000080000001+.quad 0x8000000080008008+.size iotas,.-iotas+.byte 75,101,99,99,97,107,45,49,54,48,48,32,97,98,115,111,114,98,32,97,110,100,32,115,113,117,101,101,122,101,32,102,111,114,32,120,56,54,95,54,52,44,32,67,82,89,80,84,79,71,65,77,83,32,98,121,32,60,97,112,112,114,111,64,111,112,101,110,115,115,108,46,111,114,103,62,0++.section .note.gnu.property,"a",@note+ .long 4,2f-1f,5+ .byte 0x47,0x4E,0x55,0+1: .long 0xc0000002,4,3+.align 8+2:++.section .note.GNU-stack,"",@progbits
@@ -0,0 +1,529 @@+.text +++.p2align 5+__crypton_keccak_asm_f1600:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa++ movq 60(%rdi),%rax+ movq 68(%rdi),%rbx+ movq 76(%rdi),%rcx+ movq 84(%rdi),%rdx+ movq 92(%rdi),%rbp+ jmp L$oop++.p2align 5+L$oop:+ movq -100(%rdi),%r8+ movq -52(%rdi),%r9+ movq -4(%rdi),%r10+ movq 44(%rdi),%r11++ xorq -84(%rdi),%rcx+ xorq -76(%rdi),%rdx+ xorq %r8,%rax+ xorq -92(%rdi),%rbx+ xorq -44(%rdi),%rcx+ xorq -60(%rdi),%rax+ movq %rbp,%r12+ xorq -68(%rdi),%rbp++ xorq %r10,%rcx+ xorq -20(%rdi),%rax+ xorq -36(%rdi),%rdx+ xorq %r9,%rbx+ xorq -28(%rdi),%rbp++ xorq 36(%rdi),%rcx+ xorq 20(%rdi),%rax+ xorq 4(%rdi),%rdx+ xorq -12(%rdi),%rbx+ xorq 12(%rdi),%rbp++ movq %rcx,%r13+ rolq $1,%rcx+ xorq %rax,%rcx+ xorq %r11,%rdx++ rolq $1,%rax+ xorq %rdx,%rax+ xorq 28(%rdi),%rbx++ rolq $1,%rdx+ xorq %rbx,%rdx+ xorq 52(%rdi),%rbp++ rolq $1,%rbx+ xorq %rbp,%rbx++ rolq $1,%rbp+ xorq %r13,%rbp+ xorq %rcx,%r9+ xorq %rdx,%r10+ rolq $44,%r9+ xorq %rbp,%r11+ xorq %rax,%r12+ rolq $43,%r10+ xorq %rbx,%r8+ movq %r9,%r13+ rolq $21,%r11+ orq %r10,%r9+ xorq %r8,%r9+ rolq $14,%r12++ xorq (%r15),%r9+ leaq 8(%r15),%r15++ movq %r12,%r14+ andq %r11,%r12+ movq %r9,-100(%rsi)+ xorq %r10,%r12+ notq %r10+ movq %r12,-84(%rsi)++ orq %r11,%r10+ movq 76(%rdi),%r12+ xorq %r13,%r10+ movq %r10,-92(%rsi)++ andq %r8,%r13+ movq -28(%rdi),%r9+ xorq %r14,%r13+ movq -20(%rdi),%r10+ movq %r13,-68(%rsi)++ orq %r8,%r14+ movq -76(%rdi),%r8+ xorq %r11,%r14+ movq 28(%rdi),%r11+ movq %r14,-76(%rsi)+++ xorq %rbp,%r8+ xorq %rdx,%r12+ rolq $28,%r8+ xorq %rcx,%r11+ xorq %rax,%r9+ rolq $61,%r12+ rolq $45,%r11+ xorq %rbx,%r10+ rolq $20,%r9+ movq %r8,%r13+ orq %r12,%r8+ rolq $3,%r10++ xorq %r11,%r8+ movq %r8,-36(%rsi)++ movq %r9,%r14+ andq %r13,%r9+ movq -92(%rdi),%r8+ xorq %r12,%r9+ notq %r12+ movq %r9,-28(%rsi)++ orq %r11,%r12+ movq -44(%rdi),%r9+ xorq %r10,%r12+ movq %r12,-44(%rsi)++ andq %r10,%r11+ movq 60(%rdi),%r12+ xorq %r14,%r11+ movq %r11,-52(%rsi)++ orq %r10,%r14+ movq 4(%rdi),%r10+ xorq %r13,%r14+ movq 52(%rdi),%r11+ movq %r14,-60(%rsi)+++ xorq %rbp,%r10+ xorq %rax,%r11+ rolq $25,%r10+ xorq %rdx,%r9+ rolq $8,%r11+ xorq %rbx,%r12+ rolq $6,%r9+ xorq %rcx,%r8+ rolq $18,%r12+ movq %r10,%r13+ andq %r11,%r10+ rolq $1,%r8++ notq %r11+ xorq %r9,%r10+ movq %r10,-12(%rsi)++ movq %r12,%r14+ andq %r11,%r12+ movq -12(%rdi),%r10+ xorq %r13,%r12+ movq %r12,-4(%rsi)++ orq %r9,%r13+ movq 84(%rdi),%r12+ xorq %r8,%r13+ movq %r13,-20(%rsi)++ andq %r8,%r9+ xorq %r14,%r9+ movq %r9,12(%rsi)++ orq %r8,%r14+ movq -60(%rdi),%r9+ xorq %r11,%r14+ movq 36(%rdi),%r11+ movq %r14,4(%rsi)+++ movq -68(%rdi),%r8++ xorq %rcx,%r10+ xorq %rdx,%r11+ rolq $10,%r10+ xorq %rbx,%r9+ rolq $15,%r11+ xorq %rbp,%r12+ rolq $36,%r9+ xorq %rax,%r8+ rolq $56,%r12+ movq %r10,%r13+ orq %r11,%r10+ rolq $27,%r8++ notq %r11+ xorq %r9,%r10+ movq %r10,28(%rsi)++ movq %r12,%r14+ orq %r11,%r12+ xorq %r13,%r12+ movq %r12,36(%rsi)++ andq %r9,%r13+ xorq %r8,%r13+ movq %r13,20(%rsi)++ orq %r8,%r9+ xorq %r14,%r9+ movq %r9,52(%rsi)++ andq %r14,%r8+ xorq %r11,%r8+ movq %r8,44(%rsi)+++ xorq -84(%rdi),%rdx+ xorq -36(%rdi),%rbp+ rolq $62,%rdx+ xorq 68(%rdi),%rcx+ rolq $55,%rbp+ xorq 12(%rdi),%rax+ rolq $2,%rcx+ xorq 20(%rdi),%rbx+ xchgq %rsi,%rdi+ rolq $39,%rax+ rolq $41,%rbx+ movq %rdx,%r13+ andq %rbp,%rdx+ notq %rbp+ xorq %rcx,%rdx+ movq %rdx,92(%rdi)++ movq %rax,%r14+ andq %rbp,%rax+ xorq %r13,%rax+ movq %rax,60(%rdi)++ orq %rcx,%r13+ xorq %rbx,%r13+ movq %r13,84(%rdi)++ andq %rbx,%rcx+ xorq %r14,%rcx+ movq %rcx,76(%rdi)++ orq %r14,%rbx+ xorq %rbp,%rbx+ movq %rbx,68(%rdi)++ movq %rdx,%rbp+ movq %r13,%rdx++ testq $255,%r15+ jnz L$oop++ leaq -192(%r15),%r15+ .byte 0xf3,0xc3+.cfi_endproc+++.globl _crypton_keccak_asm_f1600++.p2align 5+_crypton_keccak_asm_f1600:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa+++ pushq %rbx+.cfi_adjust_cfa_offset 8+.cfi_offset %rbx,-16+ pushq %rbp+.cfi_adjust_cfa_offset 8+.cfi_offset %rbp,-24+ pushq %r12+.cfi_adjust_cfa_offset 8+.cfi_offset %r12,-32+ pushq %r13+.cfi_adjust_cfa_offset 8+.cfi_offset %r13,-40+ pushq %r14+.cfi_adjust_cfa_offset 8+.cfi_offset %r14,-48+ pushq %r15+.cfi_adjust_cfa_offset 8+.cfi_offset %r15,-56++ leaq 100(%rdi),%rdi+ subq $200,%rsp+.cfi_adjust_cfa_offset 200+++ notq -92(%rdi)+ notq -84(%rdi)+ notq -36(%rdi)+ notq -4(%rdi)+ notq 36(%rdi)+ notq 60(%rdi)++ leaq iotas(%rip),%r15+ leaq 100(%rsp),%rsi++ call __crypton_keccak_asm_f1600++ notq -92(%rdi)+ notq -84(%rdi)+ notq -36(%rdi)+ notq -4(%rdi)+ notq 36(%rdi)+ notq 60(%rdi)+ leaq -100(%rdi),%rdi++ leaq 248(%rsp),%r11+.cfi_def_cfa %r11,8+ movq -48(%r11),%r15+ movq -40(%r11),%r14+ movq -32(%r11),%r13+ movq -24(%r11),%r12+ movq -16(%r11),%rbp+ movq -8(%r11),%rbx+ leaq (%r11),%rsp+.cfi_restore %r12+.cfi_restore %r13+.cfi_restore %r14+.cfi_restore %r15+.cfi_restore %rbp+.cfi_restore %rbx+ .byte 0xf3,0xc3+.cfi_endproc ++.globl _crypton_keccak_asm_absorb++.p2align 5+_crypton_keccak_asm_absorb:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa+++ pushq %rbx+.cfi_adjust_cfa_offset 8+.cfi_offset %rbx,-16+ pushq %rbp+.cfi_adjust_cfa_offset 8+.cfi_offset %rbp,-24+ pushq %r12+.cfi_adjust_cfa_offset 8+.cfi_offset %r12,-32+ pushq %r13+.cfi_adjust_cfa_offset 8+.cfi_offset %r13,-40+ pushq %r14+.cfi_adjust_cfa_offset 8+.cfi_offset %r14,-48+ pushq %r15+.cfi_adjust_cfa_offset 8+.cfi_offset %r15,-56++ leaq 100(%rdi),%rdi+ subq $232,%rsp+.cfi_adjust_cfa_offset 232+++ movq %rsi,%r9+ leaq 100(%rsp),%rsi++ notq -92(%rdi)+ notq -84(%rdi)+ notq -36(%rdi)+ notq -4(%rdi)+ notq 36(%rdi)+ notq 60(%rdi)+ leaq iotas(%rip),%r15++ movq %rcx,216-100(%rsi)++L$oop_absorb:+ cmpq %rcx,%rdx+ jc L$done_absorb++ shrq $3,%rcx+ leaq -100(%rdi),%r8++L$block_absorb:+ movq (%r9),%rax+ leaq 8(%r9),%r9+ xorq (%r8),%rax+ leaq 8(%r8),%r8+ subq $8,%rdx+ movq %rax,-8(%r8)+ subq $1,%rcx+ jnz L$block_absorb++ movq %r9,200-100(%rsi)+ movq %rdx,208-100(%rsi)+ call __crypton_keccak_asm_f1600+ movq 200-100(%rsi),%r9+ movq 208-100(%rsi),%rdx+ movq 216-100(%rsi),%rcx+ jmp L$oop_absorb++.p2align 5+L$done_absorb:+ movq %rdx,%rax++ notq -92(%rdi)+ notq -84(%rdi)+ notq -36(%rdi)+ notq -4(%rdi)+ notq 36(%rdi)+ notq 60(%rdi)++ leaq 280(%rsp),%r11+.cfi_def_cfa %r11,8+ movq -48(%r11),%r15+ movq -40(%r11),%r14+ movq -32(%r11),%r13+ movq -24(%r11),%r12+ movq -16(%r11),%rbp+ movq -8(%r11),%rbx+ leaq (%r11),%rsp+.cfi_restore %r12+.cfi_restore %r13+.cfi_restore %r14+.cfi_restore %r15+.cfi_restore %rbp+.cfi_restore %rbx+ .byte 0xf3,0xc3+.cfi_endproc ++.globl _crypton_keccak_asm_squeeze++.p2align 5+_crypton_keccak_asm_squeeze:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa+++ pushq %r12+.cfi_adjust_cfa_offset 8+.cfi_offset %r12,-16+ pushq %r13+.cfi_adjust_cfa_offset 8+.cfi_offset %r13,-24+ pushq %r14+.cfi_adjust_cfa_offset 8+.cfi_offset %r14,-32+ subq $32,%rsp+.cfi_adjust_cfa_offset 32+++ shrq $3,%rcx+ movq %rdi,%r8+ movq %rsi,%r12+ movq %rdx,%r13+ movq %rcx,%r14+ jmp L$oop_squeeze++.p2align 5+L$oop_squeeze:+ cmpq $8,%r13+ jb L$tail_squeeze++ movq (%r8),%rax+ leaq 8(%r8),%r8+ movq %rax,(%r12)+ leaq 8(%r12),%r12+ subq $8,%r13+ jz L$done_squeeze++ subq $1,%rcx+ jnz L$oop_squeeze++ movq %rdi,%rcx+ call _crypton_keccak_asm_f1600+ movq %rdi,%r8+ movq %r14,%rcx+ jmp L$oop_squeeze++L$tail_squeeze:+ movq %r8,%rsi+ movq %r12,%rdi+ movq %r13,%rcx+.byte 0xf3,0xa4++L$done_squeeze:+ movq 32(%rsp),%r14+ movq 40(%rsp),%r13+ movq 48(%rsp),%r12+ addq $56,%rsp+.cfi_adjust_cfa_offset -56+.cfi_restore %r12+.cfi_restore %r13+.cfi_restore %r14+ .byte 0xf3,0xc3+.cfi_endproc ++.p2align 8+.quad 0,0,0,0,0,0,0,0++iotas:+.quad 0x0000000000000001+.quad 0x0000000000008082+.quad 0x800000000000808a+.quad 0x8000000080008000+.quad 0x000000000000808b+.quad 0x0000000080000001+.quad 0x8000000080008081+.quad 0x8000000000008009+.quad 0x000000000000008a+.quad 0x0000000000000088+.quad 0x0000000080008009+.quad 0x000000008000000a+.quad 0x000000008000808b+.quad 0x800000000000008b+.quad 0x8000000000008089+.quad 0x8000000000008003+.quad 0x8000000000008002+.quad 0x8000000000000080+.quad 0x000000000000800a+.quad 0x800000008000000a+.quad 0x8000000080008081+.quad 0x8000000000008080+.quad 0x0000000080000001+.quad 0x8000000080008008++.byte 75,101,99,99,97,107,45,49,54,48,48,32,97,98,115,111,114,98,32,97,110,100,32,115,113,117,101,101,122,101,32,102,111,114,32,120,56,54,95,54,52,44,32,67,82,89,80,84,79,71,65,77,83,32,98,121,32,60,97,112,112,114,111,64,111,112,101,110,115,115,108,46,111,114,103,62,0
@@ -0,0 +1,648 @@+.text ++.def __crypton_keccak_asm_f1600; .scl 3; .type 32; .endef+.p2align 5+__crypton_keccak_asm_f1600:+ .byte 0xf3,0x0f,0x1e,0xfa++ movq 60(%rdi),%rax+ movq 68(%rdi),%rbx+ movq 76(%rdi),%rcx+ movq 84(%rdi),%rdx+ movq 92(%rdi),%rbp+ jmp .Loop++.p2align 5+.Loop:+ movq -100(%rdi),%r8+ movq -52(%rdi),%r9+ movq -4(%rdi),%r10+ movq 44(%rdi),%r11++ xorq -84(%rdi),%rcx+ xorq -76(%rdi),%rdx+ xorq %r8,%rax+ xorq -92(%rdi),%rbx+ xorq -44(%rdi),%rcx+ xorq -60(%rdi),%rax+ movq %rbp,%r12+ xorq -68(%rdi),%rbp++ xorq %r10,%rcx+ xorq -20(%rdi),%rax+ xorq -36(%rdi),%rdx+ xorq %r9,%rbx+ xorq -28(%rdi),%rbp++ xorq 36(%rdi),%rcx+ xorq 20(%rdi),%rax+ xorq 4(%rdi),%rdx+ xorq -12(%rdi),%rbx+ xorq 12(%rdi),%rbp++ movq %rcx,%r13+ rolq $1,%rcx+ xorq %rax,%rcx+ xorq %r11,%rdx++ rolq $1,%rax+ xorq %rdx,%rax+ xorq 28(%rdi),%rbx++ rolq $1,%rdx+ xorq %rbx,%rdx+ xorq 52(%rdi),%rbp++ rolq $1,%rbx+ xorq %rbp,%rbx++ rolq $1,%rbp+ xorq %r13,%rbp+ xorq %rcx,%r9+ xorq %rdx,%r10+ rolq $44,%r9+ xorq %rbp,%r11+ xorq %rax,%r12+ rolq $43,%r10+ xorq %rbx,%r8+ movq %r9,%r13+ rolq $21,%r11+ orq %r10,%r9+ xorq %r8,%r9+ rolq $14,%r12++ xorq (%r15),%r9+ leaq 8(%r15),%r15++ movq %r12,%r14+ andq %r11,%r12+ movq %r9,-100(%rsi)+ xorq %r10,%r12+ notq %r10+ movq %r12,-84(%rsi)++ orq %r11,%r10+ movq 76(%rdi),%r12+ xorq %r13,%r10+ movq %r10,-92(%rsi)++ andq %r8,%r13+ movq -28(%rdi),%r9+ xorq %r14,%r13+ movq -20(%rdi),%r10+ movq %r13,-68(%rsi)++ orq %r8,%r14+ movq -76(%rdi),%r8+ xorq %r11,%r14+ movq 28(%rdi),%r11+ movq %r14,-76(%rsi)+++ xorq %rbp,%r8+ xorq %rdx,%r12+ rolq $28,%r8+ xorq %rcx,%r11+ xorq %rax,%r9+ rolq $61,%r12+ rolq $45,%r11+ xorq %rbx,%r10+ rolq $20,%r9+ movq %r8,%r13+ orq %r12,%r8+ rolq $3,%r10++ xorq %r11,%r8+ movq %r8,-36(%rsi)++ movq %r9,%r14+ andq %r13,%r9+ movq -92(%rdi),%r8+ xorq %r12,%r9+ notq %r12+ movq %r9,-28(%rsi)++ orq %r11,%r12+ movq -44(%rdi),%r9+ xorq %r10,%r12+ movq %r12,-44(%rsi)++ andq %r10,%r11+ movq 60(%rdi),%r12+ xorq %r14,%r11+ movq %r11,-52(%rsi)++ orq %r10,%r14+ movq 4(%rdi),%r10+ xorq %r13,%r14+ movq 52(%rdi),%r11+ movq %r14,-60(%rsi)+++ xorq %rbp,%r10+ xorq %rax,%r11+ rolq $25,%r10+ xorq %rdx,%r9+ rolq $8,%r11+ xorq %rbx,%r12+ rolq $6,%r9+ xorq %rcx,%r8+ rolq $18,%r12+ movq %r10,%r13+ andq %r11,%r10+ rolq $1,%r8++ notq %r11+ xorq %r9,%r10+ movq %r10,-12(%rsi)++ movq %r12,%r14+ andq %r11,%r12+ movq -12(%rdi),%r10+ xorq %r13,%r12+ movq %r12,-4(%rsi)++ orq %r9,%r13+ movq 84(%rdi),%r12+ xorq %r8,%r13+ movq %r13,-20(%rsi)++ andq %r8,%r9+ xorq %r14,%r9+ movq %r9,12(%rsi)++ orq %r8,%r14+ movq -60(%rdi),%r9+ xorq %r11,%r14+ movq 36(%rdi),%r11+ movq %r14,4(%rsi)+++ movq -68(%rdi),%r8++ xorq %rcx,%r10+ xorq %rdx,%r11+ rolq $10,%r10+ xorq %rbx,%r9+ rolq $15,%r11+ xorq %rbp,%r12+ rolq $36,%r9+ xorq %rax,%r8+ rolq $56,%r12+ movq %r10,%r13+ orq %r11,%r10+ rolq $27,%r8++ notq %r11+ xorq %r9,%r10+ movq %r10,28(%rsi)++ movq %r12,%r14+ orq %r11,%r12+ xorq %r13,%r12+ movq %r12,36(%rsi)++ andq %r9,%r13+ xorq %r8,%r13+ movq %r13,20(%rsi)++ orq %r8,%r9+ xorq %r14,%r9+ movq %r9,52(%rsi)++ andq %r14,%r8+ xorq %r11,%r8+ movq %r8,44(%rsi)+++ xorq -84(%rdi),%rdx+ xorq -36(%rdi),%rbp+ rolq $62,%rdx+ xorq 68(%rdi),%rcx+ rolq $55,%rbp+ xorq 12(%rdi),%rax+ rolq $2,%rcx+ xorq 20(%rdi),%rbx+ xchgq %rsi,%rdi+ rolq $39,%rax+ rolq $41,%rbx+ movq %rdx,%r13+ andq %rbp,%rdx+ notq %rbp+ xorq %rcx,%rdx+ movq %rdx,92(%rdi)++ movq %rax,%r14+ andq %rbp,%rax+ xorq %r13,%rax+ movq %rax,60(%rdi)++ orq %rcx,%r13+ xorq %rbx,%r13+ movq %r13,84(%rdi)++ andq %rbx,%rcx+ xorq %r14,%rcx+ movq %rcx,76(%rdi)++ orq %r14,%rbx+ xorq %rbp,%rbx+ movq %rbx,68(%rdi)++ movq %rdx,%rbp+ movq %r13,%rdx++ testq $255,%r15+ jnz .Loop++ leaq -192(%r15),%r15+ .byte 0xf3,0xc3+++.globl crypton_keccak_asm_f1600+.def crypton_keccak_asm_f1600; .scl 2; .type 32; .endef+.p2align 5+crypton_keccak_asm_f1600:+ .byte 0xf3,0x0f,0x1e,0xfa+ movq %rdi,8(%rsp)+ movq %rsi,16(%rsp)+ movq %rsp,%r11+.LSEH_begin_crypton_keccak_asm_f1600:+++ movq %rcx,%rdi+ pushq %rbx++ pushq %rbp++ pushq %r12++ pushq %r13++ pushq %r14++ pushq %r15+++ leaq 100(%rdi),%rdi+ subq $200,%rsp++.LSEH_body_crypton_keccak_asm_f1600:+++ notq -92(%rdi)+ notq -84(%rdi)+ notq -36(%rdi)+ notq -4(%rdi)+ notq 36(%rdi)+ notq 60(%rdi)++ leaq iotas(%rip),%r15+ leaq 100(%rsp),%rsi++ call __crypton_keccak_asm_f1600++ notq -92(%rdi)+ notq -84(%rdi)+ notq -36(%rdi)+ notq -4(%rdi)+ notq 36(%rdi)+ notq 60(%rdi)+ leaq -100(%rdi),%rdi++ leaq 248(%rsp),%r11++ movq -48(%r11),%r15+ movq -40(%r11),%r14+ movq -32(%r11),%r13+ movq -24(%r11),%r12+ movq -16(%r11),%rbp+ movq -8(%r11),%rbx+ leaq (%r11),%rsp+.LSEH_epilogue_crypton_keccak_asm_f1600:+ mov 8(%r11),%rdi+ mov 16(%r11),%rsi++ .byte 0xf3,0xc3++.LSEH_end_crypton_keccak_asm_f1600:+.globl crypton_keccak_asm_absorb+.def crypton_keccak_asm_absorb; .scl 2; .type 32; .endef+.p2align 5+crypton_keccak_asm_absorb:+ .byte 0xf3,0x0f,0x1e,0xfa+ movq %rdi,8(%rsp)+ movq %rsi,16(%rsp)+ movq %rsp,%r11+.LSEH_begin_crypton_keccak_asm_absorb:+++ movq %rcx,%rdi+ movq %rdx,%rsi+ movq %r8,%rdx+ movq %r9,%rcx+ pushq %rbx++ pushq %rbp++ pushq %r12++ pushq %r13++ pushq %r14++ pushq %r15+++ leaq 100(%rdi),%rdi+ subq $232,%rsp++.LSEH_body_crypton_keccak_asm_absorb:+++ movq %rsi,%r9+ leaq 100(%rsp),%rsi++ notq -92(%rdi)+ notq -84(%rdi)+ notq -36(%rdi)+ notq -4(%rdi)+ notq 36(%rdi)+ notq 60(%rdi)+ leaq iotas(%rip),%r15++ movq %rcx,216-100(%rsi)++.Loop_absorb:+ cmpq %rcx,%rdx+ jc .Ldone_absorb++ shrq $3,%rcx+ leaq -100(%rdi),%r8++.Lblock_absorb:+ movq (%r9),%rax+ leaq 8(%r9),%r9+ xorq (%r8),%rax+ leaq 8(%r8),%r8+ subq $8,%rdx+ movq %rax,-8(%r8)+ subq $1,%rcx+ jnz .Lblock_absorb++ movq %r9,200-100(%rsi)+ movq %rdx,208-100(%rsi)+ call __crypton_keccak_asm_f1600+ movq 200-100(%rsi),%r9+ movq 208-100(%rsi),%rdx+ movq 216-100(%rsi),%rcx+ jmp .Loop_absorb++.p2align 5+.Ldone_absorb:+ movq %rdx,%rax++ notq -92(%rdi)+ notq -84(%rdi)+ notq -36(%rdi)+ notq -4(%rdi)+ notq 36(%rdi)+ notq 60(%rdi)++ leaq 280(%rsp),%r11++ movq -48(%r11),%r15+ movq -40(%r11),%r14+ movq -32(%r11),%r13+ movq -24(%r11),%r12+ movq -16(%r11),%rbp+ movq -8(%r11),%rbx+ leaq (%r11),%rsp+.LSEH_epilogue_crypton_keccak_asm_absorb:+ mov 8(%r11),%rdi+ mov 16(%r11),%rsi++ .byte 0xf3,0xc3++.LSEH_end_crypton_keccak_asm_absorb:+.globl crypton_keccak_asm_squeeze+.def crypton_keccak_asm_squeeze; .scl 2; .type 32; .endef+.p2align 5+crypton_keccak_asm_squeeze:+ .byte 0xf3,0x0f,0x1e,0xfa+ movq %rdi,8(%rsp)+ movq %rsi,16(%rsp)+ movq %rsp,%r11+.LSEH_begin_crypton_keccak_asm_squeeze:+++ movq %rcx,%rdi+ movq %rdx,%rsi+ movq %r8,%rdx+ movq %r9,%rcx+ pushq %r12++ pushq %r13++ pushq %r14++ subq $32,%rsp++.LSEH_body_crypton_keccak_asm_squeeze:+++ shrq $3,%rcx+ movq %rdi,%r8+ movq %rsi,%r12+ movq %rdx,%r13+ movq %rcx,%r14+ jmp .Loop_squeeze++.p2align 5+.Loop_squeeze:+ cmpq $8,%r13+ jb .Ltail_squeeze++ movq (%r8),%rax+ leaq 8(%r8),%r8+ movq %rax,(%r12)+ leaq 8(%r12),%r12+ subq $8,%r13+ jz .Ldone_squeeze++ subq $1,%rcx+ jnz .Loop_squeeze++ movq %rdi,%rcx+ call crypton_keccak_asm_f1600+ movq %rdi,%r8+ movq %r14,%rcx+ jmp .Loop_squeeze++.Ltail_squeeze:+ movq %r8,%rsi+ movq %r12,%rdi+ movq %r13,%rcx+.byte 0xf3,0xa4++.Ldone_squeeze:+ movq 32(%rsp),%r14+ movq 40(%rsp),%r13+ movq 48(%rsp),%r12+ addq $56,%rsp++.LSEH_epilogue_crypton_keccak_asm_squeeze:+ mov 8(%rsp),%rdi+ mov 16(%rsp),%rsi++ .byte 0xf3,0xc3++.LSEH_end_crypton_keccak_asm_squeeze:+.p2align 8+.quad 0,0,0,0,0,0,0,0++iotas:+.quad 0x0000000000000001+.quad 0x0000000000008082+.quad 0x800000000000808a+.quad 0x8000000080008000+.quad 0x000000000000808b+.quad 0x0000000080000001+.quad 0x8000000080008081+.quad 0x8000000000008009+.quad 0x000000000000008a+.quad 0x0000000000000088+.quad 0x0000000080008009+.quad 0x000000008000000a+.quad 0x000000008000808b+.quad 0x800000000000008b+.quad 0x8000000000008089+.quad 0x8000000000008003+.quad 0x8000000000008002+.quad 0x8000000000000080+.quad 0x000000000000800a+.quad 0x800000008000000a+.quad 0x8000000080008081+.quad 0x8000000000008080+.quad 0x0000000080000001+.quad 0x8000000080008008++.byte 75,101,99,99,97,107,45,49,54,48,48,32,97,98,115,111,114,98,32,97,110,100,32,115,113,117,101,101,122,101,32,102,111,114,32,120,56,54,95,54,52,44,32,67,82,89,80,84,79,71,65,77,83,32,98,121,32,60,97,112,112,114,111,64,111,112,101,110,115,115,108,46,111,114,103,62,0+.section .pdata+.p2align 2+.rva .LSEH_begin_crypton_keccak_asm_f1600+.rva .LSEH_body_crypton_keccak_asm_f1600+.rva .LSEH_info_crypton_keccak_asm_f1600_prologue++.rva .LSEH_body_crypton_keccak_asm_f1600+.rva .LSEH_epilogue_crypton_keccak_asm_f1600+.rva .LSEH_info_crypton_keccak_asm_f1600_body++.rva .LSEH_epilogue_crypton_keccak_asm_f1600+.rva .LSEH_end_crypton_keccak_asm_f1600+.rva .LSEH_info_crypton_keccak_asm_f1600_epilogue++.rva .LSEH_begin_crypton_keccak_asm_absorb+.rva .LSEH_body_crypton_keccak_asm_absorb+.rva .LSEH_info_crypton_keccak_asm_absorb_prologue++.rva .LSEH_body_crypton_keccak_asm_absorb+.rva .LSEH_epilogue_crypton_keccak_asm_absorb+.rva .LSEH_info_crypton_keccak_asm_absorb_body++.rva .LSEH_epilogue_crypton_keccak_asm_absorb+.rva .LSEH_end_crypton_keccak_asm_absorb+.rva .LSEH_info_crypton_keccak_asm_absorb_epilogue++.rva .LSEH_begin_crypton_keccak_asm_squeeze+.rva .LSEH_body_crypton_keccak_asm_squeeze+.rva .LSEH_info_crypton_keccak_asm_squeeze_prologue++.rva .LSEH_body_crypton_keccak_asm_squeeze+.rva .LSEH_epilogue_crypton_keccak_asm_squeeze+.rva .LSEH_info_crypton_keccak_asm_squeeze_body++.rva .LSEH_epilogue_crypton_keccak_asm_squeeze+.rva .LSEH_end_crypton_keccak_asm_squeeze+.rva .LSEH_info_crypton_keccak_asm_squeeze_epilogue++.section .xdata+.p2align 3+.LSEH_info_crypton_keccak_asm_f1600_prologue:+.byte 1,0,5,0x0b+.byte 0,0x74,1,0+.byte 0,0x64,2,0+.byte 0,0xb3+.byte 0,0+.long 0,0+.LSEH_info_crypton_keccak_asm_f1600_body:+.byte 1,0,18,0+.byte 0x00,0xf4,0x19,0x00+.byte 0x00,0xe4,0x1a,0x00+.byte 0x00,0xd4,0x1b,0x00+.byte 0x00,0xc4,0x1c,0x00+.byte 0x00,0x54,0x1d,0x00+.byte 0x00,0x34,0x1e,0x00+.byte 0x00,0x74,0x20,0x00+.byte 0x00,0x64,0x21,0x00+.byte 0x00,0x01,0x1f,0x00+.byte 0x00,0x00,0x00,0x00+.byte 0x00,0x00,0x00,0x00+.LSEH_info_crypton_keccak_asm_f1600_epilogue:+.byte 1,0,5,11+.byte 0x00,0x74,0x01,0x00+.byte 0x00,0x64,0x02,0x00+.byte 0x00,0xb3+.byte 0x00,0x00,0x00,0x00,0x00,0x00+.byte 0x00,0x00,0x00,0x00++.LSEH_info_crypton_keccak_asm_absorb_prologue:+.byte 1,0,5,0x0b+.byte 0,0x74,1,0+.byte 0,0x64,2,0+.byte 0,0xb3+.byte 0,0+.long 0,0+.LSEH_info_crypton_keccak_asm_absorb_body:+.byte 1,0,18,0+.byte 0x00,0xf4,0x1d,0x00+.byte 0x00,0xe4,0x1e,0x00+.byte 0x00,0xd4,0x1f,0x00+.byte 0x00,0xc4,0x20,0x00+.byte 0x00,0x54,0x21,0x00+.byte 0x00,0x34,0x22,0x00+.byte 0x00,0x74,0x24,0x00+.byte 0x00,0x64,0x25,0x00+.byte 0x00,0x01,0x23,0x00+.byte 0x00,0x00,0x00,0x00+.byte 0x00,0x00,0x00,0x00+.LSEH_info_crypton_keccak_asm_absorb_epilogue:+.byte 1,0,5,11+.byte 0x00,0x74,0x01,0x00+.byte 0x00,0x64,0x02,0x00+.byte 0x00,0xb3+.byte 0x00,0x00,0x00,0x00,0x00,0x00+.byte 0x00,0x00,0x00,0x00++.LSEH_info_crypton_keccak_asm_squeeze_prologue:+.byte 1,0,5,0x0b+.byte 0,0x74,1,0+.byte 0,0x64,2,0+.byte 0,0xb3+.byte 0,0+.long 0,0+.LSEH_info_crypton_keccak_asm_squeeze_body:+.byte 1,0,11,0+.byte 0x00,0xe4,0x04,0x00+.byte 0x00,0xd4,0x05,0x00+.byte 0x00,0xc4,0x06,0x00+.byte 0x00,0x74,0x08,0x00+.byte 0x00,0x64,0x09,0x00+.byte 0x00,0x62+.byte 0x00,0x00,0x00,0x00,0x00,0x00+.LSEH_info_crypton_keccak_asm_squeeze_epilogue:+.byte 1,0,4,0+.byte 0x00,0x74,0x01,0x00+.byte 0x00,0x64,0x02,0x00+.byte 0x00,0x00,0x00,0x00+
@@ -0,0 +1,601 @@+#!/usr/bin/env perl+#+# ====================================================================+# Written by Andy Polyakov <appro@openssl.org> for the OpenSSL+# project. The module is, however, dual licensed under OpenSSL and+# CRYPTOGAMS licenses depending on where you obtain it. For further+# details see http://www.openssl.org/~appro/cryptogams/.+# ====================================================================+#+# Keccak-1600 for x86_64.+#+# June 2017.+#+# Below code is [lane complementing] KECCAK_2X implementation (see+# sha/keccak1600.c) with C[5] and D[5] held in register bank. Though+# instead of actually unrolling the loop pair-wise I simply flip+# pointers to T[][] and A[][] at the end of round. Since number of+# rounds is even, last round writes to A[][] and everything works out.+# How does it compare to x86_64 assembly module in Keccak Code Package?+# Depending on processor it's either as fast or faster by up to 15%...+#+########################################################################+# Numbers are cycles per processed byte out of large message.+#+# r=1088(*)+#+# P4 25.8+# Core 2 12.9+# Westmere 13.7+# Sandy Bridge 12.9(**)+# Haswell 9.6+# Skylake 9.4+# Ice Lake 8.6+# Silvermont 22.8+# Goldmont 15.8+# VIA Nano 17.3+# Sledgehammer 13.3+# Bulldozer 16.5+# Ryzen 8.8+# Zen 4 7.6+#+# (*) Corresponds to SHA3-256. Improvement over compiler-generate+# varies a lot, most commont coefficient is 15% in comparison to+# gcc-5.x, 50% for gcc-4.x, 90% for gcc-3.x.+# (**) Sandy Bridge has broken rotate instruction. Performance can be+# improved by 14% by replacing rotates with double-precision+# shift with same register as source and destination.++$flavour = shift;+$output = shift;+if ($flavour =~ /\./) { $output = $flavour; undef $flavour; }++$win64=0; $win64=1 if ($flavour =~ /[nm]asm|mingw64/ || $output =~ /\.asm$/);++$0 =~ m/(.*[\/\\])[^\/\\]+$/; $dir=$1;+( $xlate="${dir}x86_64-xlate.pl" and -f $xlate ) or+( $xlate="${dir}../../perlasm/x86_64-xlate.pl" and -f $xlate) or+die "can't locate x86_64-xlate.pl";++open OUT,"| \"$^X\" \"$xlate\" $flavour \"$output\"";+*STDOUT=*OUT;++my @A = map([ 8*$_-100, 8*($_+1)-100, 8*($_+2)-100,+ 8*($_+3)-100, 8*($_+4)-100 ], (0,5,10,15,20));++my @C = ("%rax","%rbx","%rcx","%rdx","%rbp");+my @D = map("%r$_",(8..12));+my @T = map("%r$_",(13..14));+my $iotas = "%r15";++my @rhotates = ([ 0, 1, 62, 28, 27 ],+ [ 36, 44, 6, 55, 20 ],+ [ 3, 10, 43, 25, 39 ],+ [ 41, 45, 15, 21, 8 ],+ [ 18, 2, 61, 56, 14 ]);++$code.=<<___;+.text++.type __KeccakF1600,\@abi-omnipotent+.align 32+__KeccakF1600:+ mov $A[4][0](%rdi),@C[0]+ mov $A[4][1](%rdi),@C[1]+ mov $A[4][2](%rdi),@C[2]+ mov $A[4][3](%rdi),@C[3]+ mov $A[4][4](%rdi),@C[4]+ jmp .Loop++.align 32+.Loop:+ mov $A[0][0](%rdi),@D[0]+ mov $A[1][1](%rdi),@D[1]+ mov $A[2][2](%rdi),@D[2]+ mov $A[3][3](%rdi),@D[3]++ xor $A[0][2](%rdi),@C[2]+ xor $A[0][3](%rdi),@C[3]+ xor @D[0], @C[0]+ xor $A[0][1](%rdi),@C[1]+ xor $A[1][2](%rdi),@C[2]+ xor $A[1][0](%rdi),@C[0]+ mov @C[4],@D[4]+ xor $A[0][4](%rdi),@C[4]++ xor @D[2], @C[2]+ xor $A[2][0](%rdi),@C[0]+ xor $A[1][3](%rdi),@C[3]+ xor @D[1], @C[1]+ xor $A[1][4](%rdi),@C[4]++ xor $A[3][2](%rdi),@C[2]+ xor $A[3][0](%rdi),@C[0]+ xor $A[2][3](%rdi),@C[3]+ xor $A[2][1](%rdi),@C[1]+ xor $A[2][4](%rdi),@C[4]++ mov @C[2],@T[0]+ rol \$1,@C[2]+ xor @C[0],@C[2] # D[1] = ROL64(C[2], 1) ^ C[0]+ xor @D[3], @C[3]++ rol \$1,@C[0]+ xor @C[3],@C[0] # D[4] = ROL64(C[0], 1) ^ C[3]+ xor $A[3][1](%rdi),@C[1]++ rol \$1,@C[3]+ xor @C[1],@C[3] # D[2] = ROL64(C[3], 1) ^ C[1]+ xor $A[3][4](%rdi),@C[4]++ rol \$1,@C[1]+ xor @C[4],@C[1] # D[0] = ROL64(C[1], 1) ^ C[4]++ rol \$1,@C[4]+ xor @T[0],@C[4] # D[3] = ROL64(C[4], 1) ^ C[2]+___+ (@D[0..4], @C) = (@C[1..4,0], @D);+$code.=<<___;+ xor @D[1],@C[1]+ xor @D[2],@C[2]+ rol \$$rhotates[1][1],@C[1]+ xor @D[3],@C[3]+ xor @D[4],@C[4]+ rol \$$rhotates[2][2],@C[2]+ xor @D[0],@C[0]+ mov @C[1],@T[0]+ rol \$$rhotates[3][3],@C[3]+ or @C[2],@C[1]+ xor @C[0],@C[1] # C[0] ^ ( C[1] | C[2])+ rol \$$rhotates[4][4],@C[4]++ xor ($iotas),@C[1]+ lea 8($iotas),$iotas++ mov @C[4],@T[1]+ and @C[3],@C[4]+ mov @C[1],$A[0][0](%rsi) # R[0][0] = C[0] ^ ( C[1] | C[2]) ^ iotas[i]+ xor @C[2],@C[4] # C[2] ^ ( C[4] & C[3])+ not @C[2]+ mov @C[4],$A[0][2](%rsi) # R[0][2] = C[2] ^ ( C[4] & C[3])++ or @C[3],@C[2]+ mov $A[4][2](%rdi),@C[4]+ xor @T[0],@C[2] # C[1] ^ (~C[2] | C[3])+ mov @C[2],$A[0][1](%rsi) # R[0][1] = C[1] ^ (~C[2] | C[3])++ and @C[0],@T[0]+ mov $A[1][4](%rdi),@C[1]+ xor @T[1],@T[0] # C[4] ^ ( C[1] & C[0])+ mov $A[2][0](%rdi),@C[2]+ mov @T[0],$A[0][4](%rsi) # R[0][4] = C[4] ^ ( C[1] & C[0])++ or @C[0],@T[1]+ mov $A[0][3](%rdi),@C[0]+ xor @C[3],@T[1] # C[3] ^ ( C[4] | C[0])+ mov $A[3][1](%rdi),@C[3]+ mov @T[1],$A[0][3](%rsi) # R[0][3] = C[3] ^ ( C[4] | C[0])+++ xor @D[3],@C[0]+ xor @D[2],@C[4]+ rol \$$rhotates[0][3],@C[0]+ xor @D[1],@C[3]+ xor @D[4],@C[1]+ rol \$$rhotates[4][2],@C[4]+ rol \$$rhotates[3][1],@C[3]+ xor @D[0],@C[2]+ rol \$$rhotates[1][4],@C[1]+ mov @C[0],@T[0]+ or @C[4],@C[0]+ rol \$$rhotates[2][0],@C[2]++ xor @C[3],@C[0] # C[3] ^ (C[0] | C[4])+ mov @C[0],$A[1][3](%rsi) # R[1][3] = C[3] ^ (C[0] | C[4])++ mov @C[1],@T[1]+ and @T[0],@C[1]+ mov $A[0][1](%rdi),@C[0]+ xor @C[4],@C[1] # C[4] ^ (C[1] & C[0])+ not @C[4]+ mov @C[1],$A[1][4](%rsi) # R[1][4] = C[4] ^ (C[1] & C[0])++ or @C[3],@C[4]+ mov $A[1][2](%rdi),@C[1]+ xor @C[2],@C[4] # C[2] ^ (~C[4] | C[3])+ mov @C[4],$A[1][2](%rsi) # R[1][2] = C[2] ^ (~C[4] | C[3])++ and @C[2],@C[3]+ mov $A[4][0](%rdi),@C[4]+ xor @T[1],@C[3] # C[1] ^ (C[3] & C[2])+ mov @C[3],$A[1][1](%rsi) # R[1][1] = C[1] ^ (C[3] & C[2])++ or @C[2],@T[1]+ mov $A[2][3](%rdi),@C[2]+ xor @T[0],@T[1] # C[0] ^ (C[1] | C[2])+ mov $A[3][4](%rdi),@C[3]+ mov @T[1],$A[1][0](%rsi) # R[1][0] = C[0] ^ (C[1] | C[2])+++ xor @D[3],@C[2]+ xor @D[4],@C[3]+ rol \$$rhotates[2][3],@C[2]+ xor @D[2],@C[1]+ rol \$$rhotates[3][4],@C[3]+ xor @D[0],@C[4]+ rol \$$rhotates[1][2],@C[1]+ xor @D[1],@C[0]+ rol \$$rhotates[4][0],@C[4]+ mov @C[2],@T[0]+ and @C[3],@C[2]+ rol \$$rhotates[0][1],@C[0]++ not @C[3]+ xor @C[1],@C[2] # C[1] ^ ( C[2] & C[3])+ mov @C[2],$A[2][1](%rsi) # R[2][1] = C[1] ^ ( C[2] & C[3])++ mov @C[4],@T[1]+ and @C[3],@C[4]+ mov $A[2][1](%rdi),@C[2]+ xor @T[0],@C[4] # C[2] ^ ( C[4] & ~C[3])+ mov @C[4],$A[2][2](%rsi) # R[2][2] = C[2] ^ ( C[4] & ~C[3])++ or @C[1],@T[0]+ mov $A[4][3](%rdi),@C[4]+ xor @C[0],@T[0] # C[0] ^ ( C[2] | C[1])+ mov @T[0],$A[2][0](%rsi) # R[2][0] = C[0] ^ ( C[2] | C[1])++ and @C[0],@C[1]+ xor @T[1],@C[1] # C[4] ^ ( C[1] & C[0])+ mov @C[1],$A[2][4](%rsi) # R[2][4] = C[4] ^ ( C[1] & C[0])++ or @C[0],@T[1]+ mov $A[1][0](%rdi),@C[1]+ xor @C[3],@T[1] # ~C[3] ^ ( C[0] | C[4])+ mov $A[3][2](%rdi),@C[3]+ mov @T[1],$A[2][3](%rsi) # R[2][3] = ~C[3] ^ ( C[0] | C[4])+++ mov $A[0][4](%rdi),@C[0]++ xor @D[1],@C[2]+ xor @D[2],@C[3]+ rol \$$rhotates[2][1],@C[2]+ xor @D[0],@C[1]+ rol \$$rhotates[3][2],@C[3]+ xor @D[3],@C[4]+ rol \$$rhotates[1][0],@C[1]+ xor @D[4],@C[0]+ rol \$$rhotates[4][3],@C[4]+ mov @C[2],@T[0]+ or @C[3],@C[2]+ rol \$$rhotates[0][4],@C[0]++ not @C[3]+ xor @C[1],@C[2] # C[1] ^ ( C[2] | C[3])+ mov @C[2],$A[3][1](%rsi) # R[3][1] = C[1] ^ ( C[2] | C[3])++ mov @C[4],@T[1]+ or @C[3],@C[4]+ xor @T[0],@C[4] # C[2] ^ ( C[4] | ~C[3])+ mov @C[4],$A[3][2](%rsi) # R[3][2] = C[2] ^ ( C[4] | ~C[3])++ and @C[1],@T[0]+ xor @C[0],@T[0] # C[0] ^ ( C[2] & C[1])+ mov @T[0],$A[3][0](%rsi) # R[3][0] = C[0] ^ ( C[2] & C[1])++ or @C[0],@C[1]+ xor @T[1],@C[1] # C[4] ^ ( C[1] | C[0])+ mov @C[1],$A[3][4](%rsi) # R[3][4] = C[4] ^ ( C[1] | C[0])++ and @T[1],@C[0]+ xor @C[3],@C[0] # ~C[3] ^ ( C[0] & C[4])+ mov @C[0],$A[3][3](%rsi) # R[3][3] = ~C[3] ^ ( C[0] & C[4])+++ xor $A[0][2](%rdi),@D[2]+ xor $A[1][3](%rdi),@D[3]+ rol \$$rhotates[0][2],@D[2]+ xor $A[4][1](%rdi),@D[1]+ rol \$$rhotates[1][3],@D[3]+ xor $A[2][4](%rdi),@D[4]+ rol \$$rhotates[4][1],@D[1]+ xor $A[3][0](%rdi),@D[0]+ xchg %rsi,%rdi+ rol \$$rhotates[2][4],@D[4]+ rol \$$rhotates[3][0],@D[0]+___+ @C = @D[2..4,0,1];+$code.=<<___;+ mov @C[0],@T[0]+ and @C[1],@C[0]+ not @C[1]+ xor @C[4],@C[0] # C[4] ^ ( C[0] & C[1])+ mov @C[0],$A[4][4](%rdi) # R[4][4] = C[4] ^ ( C[0] & C[1])++ mov @C[2],@T[1]+ and @C[1],@C[2]+ xor @T[0],@C[2] # C[0] ^ ( C[2] & ~C[1])+ mov @C[2],$A[4][0](%rdi) # R[4][0] = C[0] ^ ( C[2] & ~C[1])++ or @C[4],@T[0]+ xor @C[3],@T[0] # C[3] ^ ( C[0] | C[4])+ mov @T[0],$A[4][3](%rdi) # R[4][3] = C[3] ^ ( C[0] | C[4])++ and @C[3],@C[4]+ xor @T[1],@C[4] # C[2] ^ ( C[4] & C[3])+ mov @C[4],$A[4][2](%rdi) # R[4][2] = C[2] ^ ( C[4] & C[3])++ or @T[1],@C[3]+ xor @C[1],@C[3] # ~C[1] ^ ( C[2] | C[3])+ mov @C[3],$A[4][1](%rdi) # R[4][1] = ~C[1] ^ ( C[2] | C[3])++ mov @C[0],@C[1] # harmonize with the loop top+ mov @T[0],@C[0]++ test \$255,$iotas+ jnz .Loop++ lea -192($iotas),$iotas # rewind iotas+ ret+.size __KeccakF1600,.-__KeccakF1600++.globl KeccakF1600+.type KeccakF1600,\@function,1,"unwind"+.align 32+KeccakF1600:+.cfi_startproc+ push %rbx+.cfi_push %rbx+ push %rbp+.cfi_push %rbp+ push %r12+.cfi_push %r12+ push %r13+.cfi_push %r13+ push %r14+.cfi_push %r14+ push %r15+.cfi_push %r15++ lea 100(%rdi),%rdi # size optimization+ sub \$200,%rsp+.cfi_alloca 200+.cfi_end_prologue++ notq $A[0][1](%rdi)+ notq $A[0][2](%rdi)+ notq $A[1][3](%rdi)+ notq $A[2][2](%rdi)+ notq $A[3][2](%rdi)+ notq $A[4][0](%rdi)++ lea iotas(%rip),$iotas+ lea 100(%rsp),%rsi # size optimization++ call __KeccakF1600++ notq $A[0][1](%rdi)+ notq $A[0][2](%rdi)+ notq $A[1][3](%rdi)+ notq $A[2][2](%rdi)+ notq $A[3][2](%rdi)+ notq $A[4][0](%rdi)+ lea -100(%rdi),%rdi # preserve A[][]++ lea 248(%rsp),%r11+.cfi_def_cfa %r11,8+ mov -48(%r11),%r15+ mov -40(%r11),%r14+ mov -32(%r11),%r13+ mov -24(%r11),%r12+ mov -16(%r11),%rbp+ mov -8(%r11),%rbx+ lea (%r11),%rsp+.cfi_epilogue+ ret+.cfi_endproc+.size KeccakF1600,.-KeccakF1600+___++{ my ($A_flat,$inp,$len,$bsz) = ("%rdi","%rsi","%rdx","%rcx");+ ($A_flat,$inp) = ("%r8","%r9");+$code.=<<___;+.globl SHA3_absorb+.type SHA3_absorb,\@function,4,"unwind"+.align 32+SHA3_absorb:+.cfi_startproc+ push %rbx+.cfi_push %rbx+ push %rbp+.cfi_push %rbp+ push %r12+.cfi_push %r12+ push %r13+.cfi_push %r13+ push %r14+.cfi_push %r14+ push %r15+.cfi_push %r15++ lea 100(%rdi),%rdi # size optimization+ sub \$232,%rsp+.cfi_alloca 232+.cfi_end_prologue++ mov %rsi,$inp+ lea 100(%rsp),%rsi # size optimization++ notq $A[0][1](%rdi)+ notq $A[0][2](%rdi)+ notq $A[1][3](%rdi)+ notq $A[2][2](%rdi)+ notq $A[3][2](%rdi)+ notq $A[4][0](%rdi)+ lea iotas(%rip),$iotas++ mov $bsz,216-100(%rsi) # save bsz++.Loop_absorb:+ cmp $bsz,$len+ jc .Ldone_absorb++ shr \$3,$bsz+ lea -100(%rdi),$A_flat++.Lblock_absorb:+ mov ($inp),%rax+ lea 8($inp),$inp+ xor ($A_flat),%rax+ lea 8($A_flat),$A_flat+ sub \$8,$len+ mov %rax,-8($A_flat)+ sub \$1,$bsz+ jnz .Lblock_absorb++ mov $inp,200-100(%rsi) # save inp+ mov $len,208-100(%rsi) # save len+ call __KeccakF1600+ mov 200-100(%rsi),$inp # pull inp+ mov 208-100(%rsi),$len # pull len+ mov 216-100(%rsi),$bsz # pull bsz+ jmp .Loop_absorb++.align 32+.Ldone_absorb:+ mov $len,%rax # return value++ notq $A[0][1](%rdi)+ notq $A[0][2](%rdi)+ notq $A[1][3](%rdi)+ notq $A[2][2](%rdi)+ notq $A[3][2](%rdi)+ notq $A[4][0](%rdi)++ lea 280(%rsp),%r11+.cfi_def_cfa %r11,8+ mov -48(%r11),%r15+ mov -40(%r11),%r14+ mov -32(%r11),%r13+ mov -24(%r11),%r12+ mov -16(%r11),%rbp+ mov -8(%r11),%rbx+ lea (%r11),%rsp+.cfi_epilogue+ ret+.cfi_endproc+.size SHA3_absorb,.-SHA3_absorb+___+}+{ my ($A_flat,$out,$len,$bsz) = ("%rdi","%rsi","%rdx","%rcx");+ ($out,$len,$bsz) = ("%r12","%r13","%r14");++$code.=<<___;+.globl SHA3_squeeze+.type SHA3_squeeze,\@function,4,"unwind"+.align 32+SHA3_squeeze:+.cfi_startproc+ push %r12+.cfi_push %r12+ push %r13+.cfi_push %r13+ push %r14+.cfi_push %r14+ sub \$32,%rsp # Windows thing+.cfi_alloca 32+.cfi_end_prologue++ shr \$3,%rcx+ mov $A_flat,%r8+ mov %rsi,$out+ mov %rdx,$len+ mov %rcx,$bsz+ jmp .Loop_squeeze++.align 32+.Loop_squeeze:+ cmp \$8,$len+ jb .Ltail_squeeze++ mov (%r8),%rax+ lea 8(%r8),%r8+ mov %rax,($out)+ lea 8($out),$out+ sub \$8,$len # len -= 8+ jz .Ldone_squeeze++ sub \$1,%rcx # bsz--+ jnz .Loop_squeeze++ mov %rdi,%rcx # Windows thing+ call KeccakF1600+ mov $A_flat,%r8+ mov $bsz,%rcx+ jmp .Loop_squeeze++.Ltail_squeeze:+ mov %r8, %rsi+ mov $out,%rdi+ mov $len,%rcx+ .byte 0xf3,0xa4 # rep movsb++.Ldone_squeeze:+ mov 32(%rsp),%r14+ mov 40(%rsp),%r13+ mov 48(%rsp),%r12+ add \$56,%rsp+.cfi_alloca -56+.cfi_epilogue+ ret+.cfi_endproc+.size SHA3_squeeze,.-SHA3_squeeze+___+}+$code.=<<___;+.align 256+ .quad 0,0,0,0,0,0,0,0+.type iotas,\@object+iotas:+ .quad 0x0000000000000001+ .quad 0x0000000000008082+ .quad 0x800000000000808a+ .quad 0x8000000080008000+ .quad 0x000000000000808b+ .quad 0x0000000080000001+ .quad 0x8000000080008081+ .quad 0x8000000000008009+ .quad 0x000000000000008a+ .quad 0x0000000000000088+ .quad 0x0000000080008009+ .quad 0x000000008000000a+ .quad 0x000000008000808b+ .quad 0x800000000000008b+ .quad 0x8000000000008089+ .quad 0x8000000000008003+ .quad 0x8000000000008002+ .quad 0x8000000000000080+ .quad 0x000000000000800a+ .quad 0x800000008000000a+ .quad 0x8000000080008081+ .quad 0x8000000000008080+ .quad 0x0000000080000001+ .quad 0x8000000080008008+.size iotas,.-iotas+.asciz "Keccak-1600 absorb and squeeze for x86_64, CRYPTOGAMS by <appro\@openssl.org>"+___++foreach (split("\n",$code)) {+ # Below replacement results in 11.2 on Sandy Bridge, 9.4 on+ # Haswell, but it hurts other processors by up to 2-3-4x...+ #s/rol\s+(\$[0-9]+),(%[a-z][a-z0-9]+)/shld\t$1,$2,$2/;++ # Below replacement results in 9.3 on Haswell [as well as+ # on Ryzen, i.e. it *hurts* Ryzen]...+ #s/rol\s+\$([0-9]+),(%[a-z][a-z0-9]+)/rorx\t\$64-$1,$2,$2/;++ print $_, "\n";+}++close STDOUT;
@@ -0,0 +1,844 @@+#ifndef __KERNEL__+# include "arm_arch.h"++#endif++.text++// forward "declarations" are required for Apple+.globl _crypton_poly1305_asm_blocks+.globl _crypton_poly1305_asm_emit++.globl _crypton_poly1305_asm_init++.align 5+_crypton_poly1305_asm_init:+ cmp x1,xzr+ stp xzr,xzr,[x0] // zero hash value+ stp xzr,xzr,[x0,#16] // [along with is_base2_26]++ csel x0,xzr,x0,eq+ b.eq Lno_key++#ifndef __KERNEL__+ adrp x17,_crypton_armcap_P@PAGE+ ldr w17,[x17,_crypton_armcap_P@PAGEOFF]+#endif++ ldp x7,x8,[x1] // load key+ mov x9,#0xfffffffc0fffffff+ movk x9,#0x0fff,lsl#48+#ifdef __AARCH64EB__+ rev x7,x7 // flip bytes+ rev x8,x8+#endif+ and x7,x7,x9 // &=0ffffffc0fffffff+ and x9,x9,#-4+ and x8,x8,x9 // &=0ffffffc0ffffffc+ mov w9,#-1+ stp x7,x8,[x0,#32] // save key value+ str w9,[x0,#48] // impossible key power value++#ifndef __KERNEL__+ tst w17,#ARMV7_NEON++ adr x13,Lcrypton_poly1305_asm_blocks+ adr x15,Lcrypton_poly1305_asm_blocks_neon+ adr x14,Lcrypton_poly1305_asm_emit++ csel x13,x13,x15,eq+# ifdef __CHERI_PURE_CAPABILITY__+ add x13, x13, #1+ add x14, x14, #1+ seal x13, x13, rb+ seal x14, x14, rb+# endif++# ifdef __ILP32__+ stp w13,w14,[x2]+# else+ stp x13,x14,[x2]+# endif+ mov x0,#1+#else+ mov x0,#0+#endif+Lno_key:+ ret++++.align 5+_crypton_poly1305_asm_blocks:+Lcrypton_poly1305_asm_blocks:+ ands x2,x2,#-16+ b.eq Lno_data++ ldp x4,x5,[x0] // load hash value+ ldp x6,x17,[x0,#16] // [along with is_base2_26]+ ldp x7,x8,[x0,#32] // load key value++#ifdef __AARCH64EB__+ lsr x12,x4,#32+ mov w13,w4+ lsr x14,x5,#32+ mov w15,w5+ lsr x16,x6,#32+#else+ mov w12,w4+ lsr x13,x4,#32+ mov w14,w5+ lsr x15,x5,#32+ mov w16,w6+#endif++ add x12,x12,x13,lsl#26 // base 2^26 -> base 2^64+ lsr x13,x14,#12+ adds x12,x12,x14,lsl#52+ add x13,x13,x15,lsl#14+ adc x13,x13,xzr+ lsr x14,x16,#24+ adds x13,x13,x16,lsl#40+ adc x14,x14,xzr++ cmp x17,#0 // is_base2_26?+ add x9,x8,x8,lsr#2 // s1 = r1 + (r1 >> 2)+ csel x4,x4,x12,eq // choose between radixes+ csel x5,x5,x13,eq+ csel x6,x6,x14,eq++Loop:+ ldp x10,x11,[x1],#16 // load input+ sub x2,x2,#16+#ifdef __AARCH64EB__+ rev x10,x10+ rev x11,x11+#endif+ adds x4,x4,x10 // accumulate input+ adcs x5,x5,x11++ mul x12,x4,x7 // h0*r0+ adc x6,x6,x3+ umulh x13,x4,x7++ mul x10,x5,x9 // h1*5*r1+ umulh x11,x5,x9++ adds x12,x12,x10+ mul x10,x4,x8 // h0*r1+ adc x13,x13,x11+ umulh x14,x4,x8++ adds x13,x13,x10+ mul x10,x5,x7 // h1*r0+ adc x14,x14,xzr+ umulh x11,x5,x7++ adds x13,x13,x10+ mul x10,x6,x9 // h2*5*r1+ adc x14,x14,x11+ mul x11,x6,x7 // h2*r0++ adds x13,x13,x10+ adc x14,x14,x11++ and x10,x14,#-4 // final reduction+ and x6,x14,#3+ add x10,x10,x14,lsr#2+ adds x4,x12,x10+ adcs x5,x13,xzr+ adc x6,x6,xzr++ cbnz x2,Loop++ stp x4,x5,[x0] // store hash value+ stp x6,xzr,[x0,#16] // [and clear is_base2_26]++Lno_data:+ ret++++.align 5+_crypton_poly1305_asm_emit:+Lcrypton_poly1305_asm_emit:+ ldp x4,x5,[x0] // load hash base 2^64+ ldp x6,x7,[x0,#16] // [along with is_base2_26]+ ldp x10,x11,[x2] // load nonce++#ifdef __AARCH64EB__+ lsr x12,x4,#32+ mov w13,w4+ lsr x14,x5,#32+ mov w15,w5+ lsr x16,x6,#32+#else+ mov w12,w4+ lsr x13,x4,#32+ mov w14,w5+ lsr x15,x5,#32+ mov w16,w6+#endif++ add x12,x12,x13,lsl#26 // base 2^26 -> base 2^64+ lsr x13,x14,#12+ adds x12,x12,x14,lsl#52+ add x13,x13,x15,lsl#14+ adc x13,x13,xzr+ lsr x14,x16,#24+ adds x13,x13,x16,lsl#40+ adc x14,x14,xzr++ cmp x7,#0 // is_base2_26?+ csel x4,x4,x12,eq // choose between radixes+ csel x5,x5,x13,eq+ csel x6,x6,x14,eq++ adds x12,x4,#5 // compare to modulus+ adcs x13,x5,xzr+ adc x14,x6,xzr++ tst x14,#-4 // see if it's carried/borrowed++ csel x4,x4,x12,eq+ csel x5,x5,x13,eq++#ifdef __AARCH64EB__+ ror x10,x10,#32 // flip nonce words+ ror x11,x11,#32+#endif+ adds x4,x4,x10 // accumulate nonce+ adc x5,x5,x11+#ifdef __AARCH64EB__+ rev x4,x4 // flip output bytes+ rev x5,x5+#endif+ stp x4,x5,[x1] // write result++ ret+++.align 5+crypton_poly1305_asm_mult:+ mul x12,x4,x7 // h0*r0+ umulh x13,x4,x7++ mul x10,x5,x9 // h1*5*r1+ umulh x11,x5,x9++ adds x12,x12,x10+ mul x10,x4,x8 // h0*r1+ adc x13,x13,x11+ umulh x14,x4,x8++ adds x13,x13,x10+ mul x10,x5,x7 // h1*r0+ adc x14,x14,xzr+ umulh x11,x5,x7++ adds x13,x13,x10+ mul x10,x6,x9 // h2*5*r1+ adc x14,x14,x11+ mul x11,x6,x7 // h2*r0++ adds x13,x13,x10+ adc x14,x14,x11++ and x10,x14,#-4 // final reduction+ and x6,x14,#3+ add x10,x10,x14,lsr#2+ adds x4,x12,x10+ adcs x5,x13,xzr+ adc x6,x6,xzr++ ret++++.align 4+crypton_poly1305_asm_splat:+ and x12,x4,#0x03ffffff // base 2^64 -> base 2^26+ ubfx x13,x4,#26,#26+ extr x14,x5,x4,#52+ and x14,x14,#0x03ffffff+ ubfx x15,x5,#14,#26+ extr x16,x6,x5,#40++ str w12,[x0,#16*0] // r0+ add w12,w13,w13,lsl#2 // r1*5+ str w13,[x0,#16*1] // r1+ add w13,w14,w14,lsl#2 // r2*5+ str w12,[x0,#16*2] // s1+ str w14,[x0,#16*3] // r2+ add w14,w15,w15,lsl#2 // r3*5+ str w13,[x0,#16*4] // s2+ str w15,[x0,#16*5] // r3+ add w15,w16,w16,lsl#2 // r4*5+ str w14,[x0,#16*6] // s3+ str w16,[x0,#16*7] // r4+ str w15,[x0,#16*8] // s4++ ret+++#ifdef __KERNEL__+.globl _crypton_poly1305_asm_blocks_neon+#endif++.align 5+_crypton_poly1305_asm_blocks_neon:+Lcrypton_poly1305_asm_blocks_neon:+ ldr x17,[x0,#24]+ cmp x2,#128+ b.lo Lcrypton_poly1305_asm_blocks++.long 0xd503233f // paciasp+ stp x29,x30,[sp,#-2*__SIZEOF_POINTER__-64]!+ add x29,sp,#0++ stp d8,d9,[sp,#2*__SIZEOF_POINTER__+0] // meet ABI requirements+ stp d10,d11,[sp,#2*__SIZEOF_POINTER__+16]+ stp d12,d13,[sp,#2*__SIZEOF_POINTER__+32]+ stp d14,d15,[sp,#2*__SIZEOF_POINTER__+48]++ cbz x17,Lbase2_64_neon++ ldp w10,w11,[x0] // load hash value base 2^26+ ldp w12,w13,[x0,#8]+ ldr w14,[x0,#16]++ tst x2,#31+ b.eq Leven_neon++ ldp x7,x8,[x0,#32] // load key value++ add x4,x10,x11,lsl#26 // base 2^26 -> base 2^64+ lsr x5,x12,#12+ adds x4,x4,x12,lsl#52+ add x5,x5,x13,lsl#14+ adc x5,x5,xzr+ lsr x6,x14,#24+ adds x5,x5,x14,lsl#40+ adc x14,x6,xzr // can be partially reduced...++ ldp x12,x13,[x1],#16 // load input+ sub x2,x2,#16+ add x9,x8,x8,lsr#2 // s1 = r1 + (r1 >> 2)++#ifdef __AARCH64EB__+ rev x12,x12+ rev x13,x13+#endif+ adds x4,x4,x12 // accumulate input+ adcs x5,x5,x13+ adc x6,x6,x3++ bl crypton_poly1305_asm_mult++ and x10,x4,#0x03ffffff // base 2^64 -> base 2^26+ ubfx x11,x4,#26,#26+ extr x12,x5,x4,#52+ and x12,x12,#0x03ffffff+ ubfx x13,x5,#14,#26+ extr x14,x6,x5,#40++ b Leven_neon++.align 4+Lbase2_64_neon:+ ldp x7,x8,[x0,#32] // load key value++ ldp x4,x5,[x0] // load hash value base 2^64+ ldr x6,[x0,#16]++ tst x2,#31+ b.eq Linit_neon++ ldp x12,x13,[x1],#16 // load input+ sub x2,x2,#16+ add x9,x8,x8,lsr#2 // s1 = r1 + (r1 >> 2)+#ifdef __AARCH64EB__+ rev x12,x12+ rev x13,x13+#endif+ adds x4,x4,x12 // accumulate input+ adcs x5,x5,x13+ adc x6,x6,x3++ bl crypton_poly1305_asm_mult++Linit_neon:+ ldr w17,[x0,#48] // first table element+ and x10,x4,#0x03ffffff // base 2^64 -> base 2^26+ ubfx x11,x4,#26,#26+ extr x12,x5,x4,#52+ and x12,x12,#0x03ffffff+ ubfx x13,x5,#14,#26+ extr x14,x6,x5,#40++ cmp w17,#-1 // is value impossible?+ b.ne Leven_neon++ fmov d24,x10+ fmov d25,x11+ fmov d26,x12+ fmov d27,x13+ fmov d28,x14++ ////////////////////////////////// initialize r^n table+ mov x4,x7 // r^1+ add x9,x8,x8,lsr#2 // s1 = r1 + (r1 >> 2)+ mov x5,x8+ mov x6,xzr+ add x0,x0,#48+12+ bl crypton_poly1305_asm_splat++ bl crypton_poly1305_asm_mult // r^2+ sub x0,x0,#4+ bl crypton_poly1305_asm_splat++ bl crypton_poly1305_asm_mult // r^3+ sub x0,x0,#4+ bl crypton_poly1305_asm_splat++ bl crypton_poly1305_asm_mult // r^4+ sub x0,x0,#4+ bl crypton_poly1305_asm_splat+ sub x0,x0,#48+ b Ldo_neon++.align 4+Leven_neon:+ fmov d24,x10+ fmov d25,x11+ fmov d26,x12+ fmov d27,x13+ fmov d28,x14++Ldo_neon:+ ldp x8,x12,[x1,#32] // inp[2:3]+ subs x2,x2,#64+ ldp x9,x13,[x1,#48]+ add x16,x1,#96+ adr x17,Lzeros++ lsl x3,x3,#24+ add x15,x0,#48++#ifdef __AARCH64EB__+ rev x8,x8+ rev x12,x12+ rev x9,x9+ rev x13,x13+#endif+ and x4,x8,#0x03ffffff // base 2^64 -> base 2^26+ and x5,x9,#0x03ffffff+ ubfx x6,x8,#26,#26+ ubfx x7,x9,#26,#26+ add x4,x4,x5,lsl#32 // bfi x4,x5,#32,#32+ extr x8,x12,x8,#52+ extr x9,x13,x9,#52+ add x6,x6,x7,lsl#32 // bfi x6,x7,#32,#32+ fmov d14,x4+ and x8,x8,#0x03ffffff+ and x9,x9,#0x03ffffff+ ubfx x10,x12,#14,#26+ ubfx x11,x13,#14,#26+ add x12,x3,x12,lsr#40+ add x13,x3,x13,lsr#40+ add x8,x8,x9,lsl#32 // bfi x8,x9,#32,#32+ fmov d15,x6+ add x10,x10,x11,lsl#32 // bfi x10,x11,#32,#32+ add x12,x12,x13,lsl#32 // bfi x12,x13,#32,#32+ fmov d16,x8+ fmov d17,x10+ fmov d18,x12++ ldp x8,x12,[x1],#16 // inp[0:1]+ ldp x9,x13,[x1],#48++ ld1 {v0.4s,v1.4s,v2.4s,v3.4s},[x15],#64+ ld1 {v4.4s,v5.4s,v6.4s,v7.4s},[x15],#64+ ld1 {v8.4s},[x15]++#ifdef __AARCH64EB__+ rev x8,x8+ rev x12,x12+ rev x9,x9+ rev x13,x13+#endif+ and x4,x8,#0x03ffffff // base 2^64 -> base 2^26+ and x5,x9,#0x03ffffff+ ubfx x6,x8,#26,#26+ ubfx x7,x9,#26,#26+ add x4,x4,x5,lsl#32 // bfi x4,x5,#32,#32+ extr x8,x12,x8,#52+ extr x9,x13,x9,#52+ add x6,x6,x7,lsl#32 // bfi x6,x7,#32,#32+ fmov d9,x4+ and x8,x8,#0x03ffffff+ and x9,x9,#0x03ffffff+ ubfx x10,x12,#14,#26+ ubfx x11,x13,#14,#26+ add x12,x3,x12,lsr#40+ add x13,x3,x13,lsr#40+ add x8,x8,x9,lsl#32 // bfi x8,x9,#32,#32+ fmov d10,x6+ add x10,x10,x11,lsl#32 // bfi x10,x11,#32,#32+ add x12,x12,x13,lsl#32 // bfi x12,x13,#32,#32+ movi v31.2d,#-1+ fmov d11,x8+ fmov d12,x10+ fmov d13,x12+ ushr v31.2d,v31.2d,#38++ b.ls Lskip_loop++.align 4+Loop_neon:+ ////////////////////////////////////////////////////////////////+ // ((inp[0]*r^4+inp[2]*r^2+inp[4])*r^4+inp[6]*r^2+ // ((inp[1]*r^4+inp[3]*r^2+inp[5])*r^3+inp[7]*r+ // ___________________/+ // ((inp[0]*r^4+inp[2]*r^2+inp[4])*r^4+inp[6]*r^2+inp[8])*r^2+ // ((inp[1]*r^4+inp[3]*r^2+inp[5])*r^4+inp[7]*r^2+inp[9])*r+ // ___________________/ ____________________/+ //+ // Note that we start with inp[2:3]*r^2. This is because it+ // doesn't depend on reduction in previous iteration.+ ////////////////////////////////////////////////////////////////+ // d4 = h0*r4 + h1*r3 + h2*r2 + h3*r1 + h4*r0+ // d3 = h0*r3 + h1*r2 + h2*r1 + h3*r0 + h4*5*r4+ // d2 = h0*r2 + h1*r1 + h2*r0 + h3*5*r4 + h4*5*r3+ // d1 = h0*r1 + h1*r0 + h2*5*r4 + h3*5*r3 + h4*5*r2+ // d0 = h0*r0 + h1*5*r4 + h2*5*r3 + h3*5*r2 + h4*5*r1++ subs x2,x2,#64+ umull v23.2d,v14.2s,v7.s[2]+ csel x16,x17,x16,lo+ umull v22.2d,v14.2s,v5.s[2]+ umull v21.2d,v14.2s,v3.s[2]+ ldp x8,x12,[x16],#16 // inp[2:3] (or zero)+ umull v20.2d,v14.2s,v1.s[2]+ ldp x9,x13,[x16],#48+ umull v19.2d,v14.2s,v0.s[2]+#ifdef __AARCH64EB__+ rev x8,x8+ rev x12,x12+ rev x9,x9+ rev x13,x13+#endif++ umlal v23.2d,v15.2s,v5.s[2]+ and x4,x8,#0x03ffffff // base 2^64 -> base 2^26+ umlal v22.2d,v15.2s,v3.s[2]+ and x5,x9,#0x03ffffff+ umlal v21.2d,v15.2s,v1.s[2]+ ubfx x6,x8,#26,#26+ umlal v20.2d,v15.2s,v0.s[2]+ ubfx x7,x9,#26,#26+ umlal v19.2d,v15.2s,v8.s[2]+ add x4,x4,x5,lsl#32 // bfi x4,x5,#32,#32++ umlal v23.2d,v16.2s,v3.s[2]+ extr x8,x12,x8,#52+ umlal v22.2d,v16.2s,v1.s[2]+ extr x9,x13,x9,#52+ umlal v21.2d,v16.2s,v0.s[2]+ add x6,x6,x7,lsl#32 // bfi x6,x7,#32,#32+ umlal v20.2d,v16.2s,v8.s[2]+ fmov d14,x4+ umlal v19.2d,v16.2s,v6.s[2]+ and x8,x8,#0x03ffffff++ umlal v23.2d,v17.2s,v1.s[2]+ and x9,x9,#0x03ffffff+ umlal v22.2d,v17.2s,v0.s[2]+ ubfx x10,x12,#14,#26+ umlal v21.2d,v17.2s,v8.s[2]+ ubfx x11,x13,#14,#26+ umlal v20.2d,v17.2s,v6.s[2]+ add x8,x8,x9,lsl#32 // bfi x8,x9,#32,#32+ umlal v19.2d,v17.2s,v4.s[2]+ fmov d15,x6++ add v11.2s,v11.2s,v26.2s+ add x12,x3,x12,lsr#40+ umlal v23.2d,v18.2s,v0.s[2]+ add x13,x3,x13,lsr#40+ umlal v22.2d,v18.2s,v8.s[2]+ add x10,x10,x11,lsl#32 // bfi x10,x11,#32,#32+ umlal v21.2d,v18.2s,v6.s[2]+ add x12,x12,x13,lsl#32 // bfi x12,x13,#32,#32+ umlal v20.2d,v18.2s,v4.s[2]+ fmov d16,x8+ umlal v19.2d,v18.2s,v2.s[2]+ fmov d17,x10++ ////////////////////////////////////////////////////////////////+ // (hash+inp[0:1])*r^4 and accumulate++ add v9.2s,v9.2s,v24.2s+ fmov d18,x12+ umlal v22.2d,v11.2s,v1.s[0]+ ldp x8,x12,[x1],#16 // inp[0:1]+ umlal v19.2d,v11.2s,v6.s[0]+ ldp x9,x13,[x1],#48+ umlal v23.2d,v11.2s,v3.s[0]+ umlal v20.2d,v11.2s,v8.s[0]+ umlal v21.2d,v11.2s,v0.s[0]+#ifdef __AARCH64EB__+ rev x8,x8+ rev x12,x12+ rev x9,x9+ rev x13,x13+#endif++ add v10.2s,v10.2s,v25.2s+ umlal v22.2d,v9.2s,v5.s[0]+ umlal v23.2d,v9.2s,v7.s[0]+ and x4,x8,#0x03ffffff // base 2^64 -> base 2^26+ umlal v21.2d,v9.2s,v3.s[0]+ and x5,x9,#0x03ffffff+ umlal v19.2d,v9.2s,v0.s[0]+ ubfx x6,x8,#26,#26+ umlal v20.2d,v9.2s,v1.s[0]+ ubfx x7,x9,#26,#26++ add v12.2s,v12.2s,v27.2s+ add x4,x4,x5,lsl#32 // bfi x4,x5,#32,#32+ umlal v22.2d,v10.2s,v3.s[0]+ extr x8,x12,x8,#52+ umlal v23.2d,v10.2s,v5.s[0]+ extr x9,x13,x9,#52+ umlal v19.2d,v10.2s,v8.s[0]+ add x6,x6,x7,lsl#32 // bfi x6,x7,#32,#32+ umlal v21.2d,v10.2s,v1.s[0]+ fmov d9,x4+ umlal v20.2d,v10.2s,v0.s[0]+ and x8,x8,#0x03ffffff++ add v13.2s,v13.2s,v28.2s+ and x9,x9,#0x03ffffff+ umlal v22.2d,v12.2s,v0.s[0]+ ubfx x10,x12,#14,#26+ umlal v19.2d,v12.2s,v4.s[0]+ ubfx x11,x13,#14,#26+ umlal v23.2d,v12.2s,v1.s[0]+ add x8,x8,x9,lsl#32 // bfi x8,x9,#32,#32+ umlal v20.2d,v12.2s,v6.s[0]+ fmov d10,x6+ umlal v21.2d,v12.2s,v8.s[0]+ add x12,x3,x12,lsr#40++ umlal v22.2d,v13.2s,v8.s[0]+ add x13,x3,x13,lsr#40+ umlal v19.2d,v13.2s,v2.s[0]+ add x10,x10,x11,lsl#32 // bfi x10,x11,#32,#32+ umlal v23.2d,v13.2s,v0.s[0]+ add x12,x12,x13,lsl#32 // bfi x12,x13,#32,#32+ umlal v20.2d,v13.2s,v4.s[0]+ fmov d11,x8+ umlal v21.2d,v13.2s,v6.s[0]+ fmov d12,x10+ fmov d13,x12++ /////////////////////////////////////////////////////////////////+ // lazy reduction as discussed in "NEON crypto" by D.J. Bernstein+ // and P. Schwabe+ //+ // [see discussion in poly1305-armv4 module]++ ushr v29.2d,v22.2d,#26+ xtn v27.2s,v22.2d+ ushr v30.2d,v19.2d,#26+ and v19.16b,v19.16b,v31.16b+ add v23.2d,v23.2d,v29.2d // h3 -> h4+ bic v27.2s,#0xfc,lsl#24 // &=0x03ffffff+ add v20.2d,v20.2d,v30.2d // h0 -> h1++ ushr v29.2d,v23.2d,#26+ xtn v28.2s,v23.2d+ ushr v30.2d,v20.2d,#26+ xtn v25.2s,v20.2d+ bic v28.2s,#0xfc,lsl#24+ add v21.2d,v21.2d,v30.2d // h1 -> h2++ add v19.2d,v19.2d,v29.2d+ shl v29.2d,v29.2d,#2+ shrn v30.2s,v21.2d,#26+ xtn v26.2s,v21.2d+ add v19.2d,v19.2d,v29.2d // h4 -> h0+ bic v25.2s,#0xfc,lsl#24+ add v27.2s,v27.2s,v30.2s // h2 -> h3+ bic v26.2s,#0xfc,lsl#24++ shrn v29.2s,v19.2d,#26+ xtn v24.2s,v19.2d+ ushr v30.2s,v27.2s,#26+ bic v27.2s,#0xfc,lsl#24+ bic v24.2s,#0xfc,lsl#24+ add v25.2s,v25.2s,v29.2s // h0 -> h1+ add v28.2s,v28.2s,v30.2s // h3 -> h4++ b.hi Loop_neon++Lskip_loop:+ dup v16.2d,v16.d[0]+ add v11.2s,v11.2s,v26.2s++ ////////////////////////////////////////////////////////////////+ // multiply (inp[0:1]+hash) or inp[2:3] by r^2:r^1++ adds x2,x2,#32+ b.ne Long_tail++ dup v16.2d,v11.d[0]+ add v14.2s,v9.2s,v24.2s+ add v17.2s,v12.2s,v27.2s+ add v15.2s,v10.2s,v25.2s+ add v18.2s,v13.2s,v28.2s++Long_tail:+ dup v14.2d,v14.d[0]+ umull2 v19.2d,v16.4s,v6.4s+ umull2 v22.2d,v16.4s,v1.4s+ umull2 v23.2d,v16.4s,v3.4s+ umull2 v21.2d,v16.4s,v0.4s+ umull2 v20.2d,v16.4s,v8.4s++ dup v15.2d,v15.d[0]+ umlal2 v19.2d,v14.4s,v0.4s+ umlal2 v21.2d,v14.4s,v3.4s+ umlal2 v22.2d,v14.4s,v5.4s+ umlal2 v23.2d,v14.4s,v7.4s+ umlal2 v20.2d,v14.4s,v1.4s++ dup v17.2d,v17.d[0]+ umlal2 v19.2d,v15.4s,v8.4s+ umlal2 v22.2d,v15.4s,v3.4s+ umlal2 v21.2d,v15.4s,v1.4s+ umlal2 v23.2d,v15.4s,v5.4s+ umlal2 v20.2d,v15.4s,v0.4s++ dup v18.2d,v18.d[0]+ umlal2 v22.2d,v17.4s,v0.4s+ umlal2 v23.2d,v17.4s,v1.4s+ umlal2 v19.2d,v17.4s,v4.4s+ umlal2 v20.2d,v17.4s,v6.4s+ umlal2 v21.2d,v17.4s,v8.4s++ umlal2 v22.2d,v18.4s,v8.4s+ umlal2 v19.2d,v18.4s,v2.4s+ umlal2 v23.2d,v18.4s,v0.4s+ umlal2 v20.2d,v18.4s,v4.4s+ umlal2 v21.2d,v18.4s,v6.4s++ b.eq Lshort_tail++ ////////////////////////////////////////////////////////////////+ // (hash+inp[0:1])*r^4:r^3 and accumulate++ add v9.2s,v9.2s,v24.2s+ umlal v22.2d,v11.2s,v1.2s+ umlal v19.2d,v11.2s,v6.2s+ umlal v23.2d,v11.2s,v3.2s+ umlal v20.2d,v11.2s,v8.2s+ umlal v21.2d,v11.2s,v0.2s++ add v10.2s,v10.2s,v25.2s+ umlal v22.2d,v9.2s,v5.2s+ umlal v19.2d,v9.2s,v0.2s+ umlal v23.2d,v9.2s,v7.2s+ umlal v20.2d,v9.2s,v1.2s+ umlal v21.2d,v9.2s,v3.2s++ add v12.2s,v12.2s,v27.2s+ umlal v22.2d,v10.2s,v3.2s+ umlal v19.2d,v10.2s,v8.2s+ umlal v23.2d,v10.2s,v5.2s+ umlal v20.2d,v10.2s,v0.2s+ umlal v21.2d,v10.2s,v1.2s++ add v13.2s,v13.2s,v28.2s+ umlal v22.2d,v12.2s,v0.2s+ umlal v19.2d,v12.2s,v4.2s+ umlal v23.2d,v12.2s,v1.2s+ umlal v20.2d,v12.2s,v6.2s+ umlal v21.2d,v12.2s,v8.2s++ umlal v22.2d,v13.2s,v8.2s+ umlal v19.2d,v13.2s,v2.2s+ umlal v23.2d,v13.2s,v0.2s+ umlal v20.2d,v13.2s,v4.2s+ umlal v21.2d,v13.2s,v6.2s++Lshort_tail:+ ////////////////////////////////////////////////////////////////+ // horizontal add++ addp v22.2d,v22.2d,v22.2d+ ldp d8,d9,[sp,#2*__SIZEOF_POINTER__+0] // meet ABI requirements+ addp v19.2d,v19.2d,v19.2d+ ldp d10,d11,[sp,#2*__SIZEOF_POINTER__+16]+ addp v23.2d,v23.2d,v23.2d+ ldp d12,d13,[sp,#2*__SIZEOF_POINTER__+32]+ addp v20.2d,v20.2d,v20.2d+ ldp d14,d15,[sp,#2*__SIZEOF_POINTER__+48]+ addp v21.2d,v21.2d,v21.2d+ ldr x30,[sp,#__SIZEOF_POINTER__]++ ////////////////////////////////////////////////////////////////+ // lazy reduction, but without narrowing++ ushr v29.2d,v22.2d,#26+ and v22.16b,v22.16b,v31.16b+ ushr v30.2d,v19.2d,#26+ and v19.16b,v19.16b,v31.16b++ add v23.2d,v23.2d,v29.2d // h3 -> h4+ add v20.2d,v20.2d,v30.2d // h0 -> h1++ ushr v29.2d,v23.2d,#26+ and v23.16b,v23.16b,v31.16b+ ushr v30.2d,v20.2d,#26+ and v20.16b,v20.16b,v31.16b+ add v21.2d,v21.2d,v30.2d // h1 -> h2++ add v19.2d,v19.2d,v29.2d+ shl v29.2d,v29.2d,#2+ ushr v30.2d,v21.2d,#26+ and v21.16b,v21.16b,v31.16b+ add v19.2d,v19.2d,v29.2d // h4 -> h0+ add v22.2d,v22.2d,v30.2d // h2 -> h3++ ushr v29.2d,v19.2d,#26+ and v19.16b,v19.16b,v31.16b+ ushr v30.2d,v22.2d,#26+ and v22.16b,v22.16b,v31.16b+ add v20.2d,v20.2d,v29.2d // h0 -> h1+ add v23.2d,v23.2d,v30.2d // h3 -> h4++ ////////////////////////////////////////////////////////////////+ // write the result, can be partially reduced++ st4 {v19.s,v20.s,v21.s,v22.s}[0],[x0],#16+ mov x4,#1+ st1 {v23.s}[0],[x0]+ str x4,[x0,#8] // set is_base2_26++ ldr x29,[sp],#2*__SIZEOF_POINTER__+64+.long 0xd50323bf // autiasp+ ret+++.align 5+Lzeros:+.long 0,0,0,0,0,0,0,0+.byte 80,111,108,121,49,51,48,53,32,102,111,114,32,65,82,77,118,56,44,32,67,82,89,80,84,79,71,65,77,83,32,98,121,32,64,100,111,116,45,97,115,109,0+.align 2+.align 2+#if !defined(__KERNEL__) && !defined(_WIN64)+.comm __crypton_armcap_P,4+.private_extern _crypton_armcap_P+#endif
@@ -0,0 +1,846 @@+#ifndef __KERNEL__+# include "arm_arch.h"++#endif++.text++// forward "declarations" are required for Apple+.globl crypton_poly1305_asm_blocks+.globl crypton_poly1305_asm_emit++.globl crypton_poly1305_asm_init+.type crypton_poly1305_asm_init,%function+.align 5+crypton_poly1305_asm_init:+ cmp x1,xzr+ stp xzr,xzr,[x0] // zero hash value+ stp xzr,xzr,[x0,#16] // [along with is_base2_26]++ csel x0,xzr,x0,eq+ b.eq .Lno_key++#ifndef __KERNEL__+ adrp x17,crypton_armcap_P+ ldr w17,[x17,#:lo12:crypton_armcap_P]+#endif++ ldp x7,x8,[x1] // load key+ mov x9,#0xfffffffc0fffffff+ movk x9,#0x0fff,lsl#48+#ifdef __AARCH64EB__+ rev x7,x7 // flip bytes+ rev x8,x8+#endif+ and x7,x7,x9 // &=0ffffffc0fffffff+ and x9,x9,#-4+ and x8,x8,x9 // &=0ffffffc0ffffffc+ mov w9,#-1+ stp x7,x8,[x0,#32] // save key value+ str w9,[x0,#48] // impossible key power value++#ifndef __KERNEL__+ tst w17,#ARMV7_NEON++ adr x13,.Lcrypton_poly1305_asm_blocks+ adr x15,.Lcrypton_poly1305_asm_blocks_neon+ adr x14,.Lcrypton_poly1305_asm_emit++ csel x13,x13,x15,eq+# ifdef __CHERI_PURE_CAPABILITY__+ add x13, x13, #1+ add x14, x14, #1+ seal x13, x13, rb+ seal x14, x14, rb+# endif++# ifdef __ILP32__+ stp w13,w14,[x2]+# else+ stp x13,x14,[x2]+# endif+ mov x0,#1+#else+ mov x0,#0+#endif+.Lno_key:+ ret+.size crypton_poly1305_asm_init,.-crypton_poly1305_asm_init++.type crypton_poly1305_asm_blocks,%function+.align 5+crypton_poly1305_asm_blocks:+.Lcrypton_poly1305_asm_blocks:+ ands x2,x2,#-16+ b.eq .Lno_data++ ldp x4,x5,[x0] // load hash value+ ldp x6,x17,[x0,#16] // [along with is_base2_26]+ ldp x7,x8,[x0,#32] // load key value++#ifdef __AARCH64EB__+ lsr x12,x4,#32+ mov w13,w4+ lsr x14,x5,#32+ mov w15,w5+ lsr x16,x6,#32+#else+ mov w12,w4+ lsr x13,x4,#32+ mov w14,w5+ lsr x15,x5,#32+ mov w16,w6+#endif++ add x12,x12,x13,lsl#26 // base 2^26 -> base 2^64+ lsr x13,x14,#12+ adds x12,x12,x14,lsl#52+ add x13,x13,x15,lsl#14+ adc x13,x13,xzr+ lsr x14,x16,#24+ adds x13,x13,x16,lsl#40+ adc x14,x14,xzr++ cmp x17,#0 // is_base2_26?+ add x9,x8,x8,lsr#2 // s1 = r1 + (r1 >> 2)+ csel x4,x4,x12,eq // choose between radixes+ csel x5,x5,x13,eq+ csel x6,x6,x14,eq++.Loop:+ ldp x10,x11,[x1],#16 // load input+ sub x2,x2,#16+#ifdef __AARCH64EB__+ rev x10,x10+ rev x11,x11+#endif+ adds x4,x4,x10 // accumulate input+ adcs x5,x5,x11++ mul x12,x4,x7 // h0*r0+ adc x6,x6,x3+ umulh x13,x4,x7++ mul x10,x5,x9 // h1*5*r1+ umulh x11,x5,x9++ adds x12,x12,x10+ mul x10,x4,x8 // h0*r1+ adc x13,x13,x11+ umulh x14,x4,x8++ adds x13,x13,x10+ mul x10,x5,x7 // h1*r0+ adc x14,x14,xzr+ umulh x11,x5,x7++ adds x13,x13,x10+ mul x10,x6,x9 // h2*5*r1+ adc x14,x14,x11+ mul x11,x6,x7 // h2*r0++ adds x13,x13,x10+ adc x14,x14,x11++ and x10,x14,#-4 // final reduction+ and x6,x14,#3+ add x10,x10,x14,lsr#2+ adds x4,x12,x10+ adcs x5,x13,xzr+ adc x6,x6,xzr++ cbnz x2,.Loop++ stp x4,x5,[x0] // store hash value+ stp x6,xzr,[x0,#16] // [and clear is_base2_26]++.Lno_data:+ ret+.size crypton_poly1305_asm_blocks,.-crypton_poly1305_asm_blocks++.type crypton_poly1305_asm_emit,%function+.align 5+crypton_poly1305_asm_emit:+.Lcrypton_poly1305_asm_emit:+ ldp x4,x5,[x0] // load hash base 2^64+ ldp x6,x7,[x0,#16] // [along with is_base2_26]+ ldp x10,x11,[x2] // load nonce++#ifdef __AARCH64EB__+ lsr x12,x4,#32+ mov w13,w4+ lsr x14,x5,#32+ mov w15,w5+ lsr x16,x6,#32+#else+ mov w12,w4+ lsr x13,x4,#32+ mov w14,w5+ lsr x15,x5,#32+ mov w16,w6+#endif++ add x12,x12,x13,lsl#26 // base 2^26 -> base 2^64+ lsr x13,x14,#12+ adds x12,x12,x14,lsl#52+ add x13,x13,x15,lsl#14+ adc x13,x13,xzr+ lsr x14,x16,#24+ adds x13,x13,x16,lsl#40+ adc x14,x14,xzr++ cmp x7,#0 // is_base2_26?+ csel x4,x4,x12,eq // choose between radixes+ csel x5,x5,x13,eq+ csel x6,x6,x14,eq++ adds x12,x4,#5 // compare to modulus+ adcs x13,x5,xzr+ adc x14,x6,xzr++ tst x14,#-4 // see if it's carried/borrowed++ csel x4,x4,x12,eq+ csel x5,x5,x13,eq++#ifdef __AARCH64EB__+ ror x10,x10,#32 // flip nonce words+ ror x11,x11,#32+#endif+ adds x4,x4,x10 // accumulate nonce+ adc x5,x5,x11+#ifdef __AARCH64EB__+ rev x4,x4 // flip output bytes+ rev x5,x5+#endif+ stp x4,x5,[x1] // write result++ ret+.size crypton_poly1305_asm_emit,.-crypton_poly1305_asm_emit+.type crypton_poly1305_asm_mult,%function+.align 5+crypton_poly1305_asm_mult:+ mul x12,x4,x7 // h0*r0+ umulh x13,x4,x7++ mul x10,x5,x9 // h1*5*r1+ umulh x11,x5,x9++ adds x12,x12,x10+ mul x10,x4,x8 // h0*r1+ adc x13,x13,x11+ umulh x14,x4,x8++ adds x13,x13,x10+ mul x10,x5,x7 // h1*r0+ adc x14,x14,xzr+ umulh x11,x5,x7++ adds x13,x13,x10+ mul x10,x6,x9 // h2*5*r1+ adc x14,x14,x11+ mul x11,x6,x7 // h2*r0++ adds x13,x13,x10+ adc x14,x14,x11++ and x10,x14,#-4 // final reduction+ and x6,x14,#3+ add x10,x10,x14,lsr#2+ adds x4,x12,x10+ adcs x5,x13,xzr+ adc x6,x6,xzr++ ret+.size crypton_poly1305_asm_mult,.-crypton_poly1305_asm_mult++.type crypton_poly1305_asm_splat,%function+.align 4+crypton_poly1305_asm_splat:+ and x12,x4,#0x03ffffff // base 2^64 -> base 2^26+ ubfx x13,x4,#26,#26+ extr x14,x5,x4,#52+ and x14,x14,#0x03ffffff+ ubfx x15,x5,#14,#26+ extr x16,x6,x5,#40++ str w12,[x0,#16*0] // r0+ add w12,w13,w13,lsl#2 // r1*5+ str w13,[x0,#16*1] // r1+ add w13,w14,w14,lsl#2 // r2*5+ str w12,[x0,#16*2] // s1+ str w14,[x0,#16*3] // r2+ add w14,w15,w15,lsl#2 // r3*5+ str w13,[x0,#16*4] // s2+ str w15,[x0,#16*5] // r3+ add w15,w16,w16,lsl#2 // r4*5+ str w14,[x0,#16*6] // s3+ str w16,[x0,#16*7] // r4+ str w15,[x0,#16*8] // s4++ ret+.size crypton_poly1305_asm_splat,.-crypton_poly1305_asm_splat++#ifdef __KERNEL__+.globl crypton_poly1305_asm_blocks_neon+#endif+.type crypton_poly1305_asm_blocks_neon,%function+.align 5+crypton_poly1305_asm_blocks_neon:+.Lcrypton_poly1305_asm_blocks_neon:+ ldr x17,[x0,#24]+ cmp x2,#128+ b.lo .Lcrypton_poly1305_asm_blocks++.inst 0xd503233f // paciasp+ stp x29,x30,[sp,#-2*__SIZEOF_POINTER__-64]!+ add x29,sp,#0++ stp d8,d9,[sp,#2*__SIZEOF_POINTER__+0] // meet ABI requirements+ stp d10,d11,[sp,#2*__SIZEOF_POINTER__+16]+ stp d12,d13,[sp,#2*__SIZEOF_POINTER__+32]+ stp d14,d15,[sp,#2*__SIZEOF_POINTER__+48]++ cbz x17,.Lbase2_64_neon++ ldp w10,w11,[x0] // load hash value base 2^26+ ldp w12,w13,[x0,#8]+ ldr w14,[x0,#16]++ tst x2,#31+ b.eq .Leven_neon++ ldp x7,x8,[x0,#32] // load key value++ add x4,x10,x11,lsl#26 // base 2^26 -> base 2^64+ lsr x5,x12,#12+ adds x4,x4,x12,lsl#52+ add x5,x5,x13,lsl#14+ adc x5,x5,xzr+ lsr x6,x14,#24+ adds x5,x5,x14,lsl#40+ adc x14,x6,xzr // can be partially reduced...++ ldp x12,x13,[x1],#16 // load input+ sub x2,x2,#16+ add x9,x8,x8,lsr#2 // s1 = r1 + (r1 >> 2)++#ifdef __AARCH64EB__+ rev x12,x12+ rev x13,x13+#endif+ adds x4,x4,x12 // accumulate input+ adcs x5,x5,x13+ adc x6,x6,x3++ bl crypton_poly1305_asm_mult++ and x10,x4,#0x03ffffff // base 2^64 -> base 2^26+ ubfx x11,x4,#26,#26+ extr x12,x5,x4,#52+ and x12,x12,#0x03ffffff+ ubfx x13,x5,#14,#26+ extr x14,x6,x5,#40++ b .Leven_neon++.align 4+.Lbase2_64_neon:+ ldp x7,x8,[x0,#32] // load key value++ ldp x4,x5,[x0] // load hash value base 2^64+ ldr x6,[x0,#16]++ tst x2,#31+ b.eq .Linit_neon++ ldp x12,x13,[x1],#16 // load input+ sub x2,x2,#16+ add x9,x8,x8,lsr#2 // s1 = r1 + (r1 >> 2)+#ifdef __AARCH64EB__+ rev x12,x12+ rev x13,x13+#endif+ adds x4,x4,x12 // accumulate input+ adcs x5,x5,x13+ adc x6,x6,x3++ bl crypton_poly1305_asm_mult++.Linit_neon:+ ldr w17,[x0,#48] // first table element+ and x10,x4,#0x03ffffff // base 2^64 -> base 2^26+ ubfx x11,x4,#26,#26+ extr x12,x5,x4,#52+ and x12,x12,#0x03ffffff+ ubfx x13,x5,#14,#26+ extr x14,x6,x5,#40++ cmp w17,#-1 // is value impossible?+ b.ne .Leven_neon++ fmov d24,x10+ fmov d25,x11+ fmov d26,x12+ fmov d27,x13+ fmov d28,x14++ ////////////////////////////////// initialize r^n table+ mov x4,x7 // r^1+ add x9,x8,x8,lsr#2 // s1 = r1 + (r1 >> 2)+ mov x5,x8+ mov x6,xzr+ add x0,x0,#48+12+ bl crypton_poly1305_asm_splat++ bl crypton_poly1305_asm_mult // r^2+ sub x0,x0,#4+ bl crypton_poly1305_asm_splat++ bl crypton_poly1305_asm_mult // r^3+ sub x0,x0,#4+ bl crypton_poly1305_asm_splat++ bl crypton_poly1305_asm_mult // r^4+ sub x0,x0,#4+ bl crypton_poly1305_asm_splat+ sub x0,x0,#48+ b .Ldo_neon++.align 4+.Leven_neon:+ fmov d24,x10+ fmov d25,x11+ fmov d26,x12+ fmov d27,x13+ fmov d28,x14++.Ldo_neon:+ ldp x8,x12,[x1,#32] // inp[2:3]+ subs x2,x2,#64+ ldp x9,x13,[x1,#48]+ add x16,x1,#96+ adr x17,.Lzeros++ lsl x3,x3,#24+ add x15,x0,#48++#ifdef __AARCH64EB__+ rev x8,x8+ rev x12,x12+ rev x9,x9+ rev x13,x13+#endif+ and x4,x8,#0x03ffffff // base 2^64 -> base 2^26+ and x5,x9,#0x03ffffff+ ubfx x6,x8,#26,#26+ ubfx x7,x9,#26,#26+ add x4,x4,x5,lsl#32 // bfi x4,x5,#32,#32+ extr x8,x12,x8,#52+ extr x9,x13,x9,#52+ add x6,x6,x7,lsl#32 // bfi x6,x7,#32,#32+ fmov d14,x4+ and x8,x8,#0x03ffffff+ and x9,x9,#0x03ffffff+ ubfx x10,x12,#14,#26+ ubfx x11,x13,#14,#26+ add x12,x3,x12,lsr#40+ add x13,x3,x13,lsr#40+ add x8,x8,x9,lsl#32 // bfi x8,x9,#32,#32+ fmov d15,x6+ add x10,x10,x11,lsl#32 // bfi x10,x11,#32,#32+ add x12,x12,x13,lsl#32 // bfi x12,x13,#32,#32+ fmov d16,x8+ fmov d17,x10+ fmov d18,x12++ ldp x8,x12,[x1],#16 // inp[0:1]+ ldp x9,x13,[x1],#48++ ld1 {v0.4s,v1.4s,v2.4s,v3.4s},[x15],#64+ ld1 {v4.4s,v5.4s,v6.4s,v7.4s},[x15],#64+ ld1 {v8.4s},[x15]++#ifdef __AARCH64EB__+ rev x8,x8+ rev x12,x12+ rev x9,x9+ rev x13,x13+#endif+ and x4,x8,#0x03ffffff // base 2^64 -> base 2^26+ and x5,x9,#0x03ffffff+ ubfx x6,x8,#26,#26+ ubfx x7,x9,#26,#26+ add x4,x4,x5,lsl#32 // bfi x4,x5,#32,#32+ extr x8,x12,x8,#52+ extr x9,x13,x9,#52+ add x6,x6,x7,lsl#32 // bfi x6,x7,#32,#32+ fmov d9,x4+ and x8,x8,#0x03ffffff+ and x9,x9,#0x03ffffff+ ubfx x10,x12,#14,#26+ ubfx x11,x13,#14,#26+ add x12,x3,x12,lsr#40+ add x13,x3,x13,lsr#40+ add x8,x8,x9,lsl#32 // bfi x8,x9,#32,#32+ fmov d10,x6+ add x10,x10,x11,lsl#32 // bfi x10,x11,#32,#32+ add x12,x12,x13,lsl#32 // bfi x12,x13,#32,#32+ movi v31.2d,#-1+ fmov d11,x8+ fmov d12,x10+ fmov d13,x12+ ushr v31.2d,v31.2d,#38++ b.ls .Lskip_loop++.align 4+.Loop_neon:+ ////////////////////////////////////////////////////////////////+ // ((inp[0]*r^4+inp[2]*r^2+inp[4])*r^4+inp[6]*r^2+ // ((inp[1]*r^4+inp[3]*r^2+inp[5])*r^3+inp[7]*r+ // ___________________/+ // ((inp[0]*r^4+inp[2]*r^2+inp[4])*r^4+inp[6]*r^2+inp[8])*r^2+ // ((inp[1]*r^4+inp[3]*r^2+inp[5])*r^4+inp[7]*r^2+inp[9])*r+ // ___________________/ ____________________/+ //+ // Note that we start with inp[2:3]*r^2. This is because it+ // doesn't depend on reduction in previous iteration.+ ////////////////////////////////////////////////////////////////+ // d4 = h0*r4 + h1*r3 + h2*r2 + h3*r1 + h4*r0+ // d3 = h0*r3 + h1*r2 + h2*r1 + h3*r0 + h4*5*r4+ // d2 = h0*r2 + h1*r1 + h2*r0 + h3*5*r4 + h4*5*r3+ // d1 = h0*r1 + h1*r0 + h2*5*r4 + h3*5*r3 + h4*5*r2+ // d0 = h0*r0 + h1*5*r4 + h2*5*r3 + h3*5*r2 + h4*5*r1++ subs x2,x2,#64+ umull v23.2d,v14.2s,v7.s[2]+ csel x16,x17,x16,lo+ umull v22.2d,v14.2s,v5.s[2]+ umull v21.2d,v14.2s,v3.s[2]+ ldp x8,x12,[x16],#16 // inp[2:3] (or zero)+ umull v20.2d,v14.2s,v1.s[2]+ ldp x9,x13,[x16],#48+ umull v19.2d,v14.2s,v0.s[2]+#ifdef __AARCH64EB__+ rev x8,x8+ rev x12,x12+ rev x9,x9+ rev x13,x13+#endif++ umlal v23.2d,v15.2s,v5.s[2]+ and x4,x8,#0x03ffffff // base 2^64 -> base 2^26+ umlal v22.2d,v15.2s,v3.s[2]+ and x5,x9,#0x03ffffff+ umlal v21.2d,v15.2s,v1.s[2]+ ubfx x6,x8,#26,#26+ umlal v20.2d,v15.2s,v0.s[2]+ ubfx x7,x9,#26,#26+ umlal v19.2d,v15.2s,v8.s[2]+ add x4,x4,x5,lsl#32 // bfi x4,x5,#32,#32++ umlal v23.2d,v16.2s,v3.s[2]+ extr x8,x12,x8,#52+ umlal v22.2d,v16.2s,v1.s[2]+ extr x9,x13,x9,#52+ umlal v21.2d,v16.2s,v0.s[2]+ add x6,x6,x7,lsl#32 // bfi x6,x7,#32,#32+ umlal v20.2d,v16.2s,v8.s[2]+ fmov d14,x4+ umlal v19.2d,v16.2s,v6.s[2]+ and x8,x8,#0x03ffffff++ umlal v23.2d,v17.2s,v1.s[2]+ and x9,x9,#0x03ffffff+ umlal v22.2d,v17.2s,v0.s[2]+ ubfx x10,x12,#14,#26+ umlal v21.2d,v17.2s,v8.s[2]+ ubfx x11,x13,#14,#26+ umlal v20.2d,v17.2s,v6.s[2]+ add x8,x8,x9,lsl#32 // bfi x8,x9,#32,#32+ umlal v19.2d,v17.2s,v4.s[2]+ fmov d15,x6++ add v11.2s,v11.2s,v26.2s+ add x12,x3,x12,lsr#40+ umlal v23.2d,v18.2s,v0.s[2]+ add x13,x3,x13,lsr#40+ umlal v22.2d,v18.2s,v8.s[2]+ add x10,x10,x11,lsl#32 // bfi x10,x11,#32,#32+ umlal v21.2d,v18.2s,v6.s[2]+ add x12,x12,x13,lsl#32 // bfi x12,x13,#32,#32+ umlal v20.2d,v18.2s,v4.s[2]+ fmov d16,x8+ umlal v19.2d,v18.2s,v2.s[2]+ fmov d17,x10++ ////////////////////////////////////////////////////////////////+ // (hash+inp[0:1])*r^4 and accumulate++ add v9.2s,v9.2s,v24.2s+ fmov d18,x12+ umlal v22.2d,v11.2s,v1.s[0]+ ldp x8,x12,[x1],#16 // inp[0:1]+ umlal v19.2d,v11.2s,v6.s[0]+ ldp x9,x13,[x1],#48+ umlal v23.2d,v11.2s,v3.s[0]+ umlal v20.2d,v11.2s,v8.s[0]+ umlal v21.2d,v11.2s,v0.s[0]+#ifdef __AARCH64EB__+ rev x8,x8+ rev x12,x12+ rev x9,x9+ rev x13,x13+#endif++ add v10.2s,v10.2s,v25.2s+ umlal v22.2d,v9.2s,v5.s[0]+ umlal v23.2d,v9.2s,v7.s[0]+ and x4,x8,#0x03ffffff // base 2^64 -> base 2^26+ umlal v21.2d,v9.2s,v3.s[0]+ and x5,x9,#0x03ffffff+ umlal v19.2d,v9.2s,v0.s[0]+ ubfx x6,x8,#26,#26+ umlal v20.2d,v9.2s,v1.s[0]+ ubfx x7,x9,#26,#26++ add v12.2s,v12.2s,v27.2s+ add x4,x4,x5,lsl#32 // bfi x4,x5,#32,#32+ umlal v22.2d,v10.2s,v3.s[0]+ extr x8,x12,x8,#52+ umlal v23.2d,v10.2s,v5.s[0]+ extr x9,x13,x9,#52+ umlal v19.2d,v10.2s,v8.s[0]+ add x6,x6,x7,lsl#32 // bfi x6,x7,#32,#32+ umlal v21.2d,v10.2s,v1.s[0]+ fmov d9,x4+ umlal v20.2d,v10.2s,v0.s[0]+ and x8,x8,#0x03ffffff++ add v13.2s,v13.2s,v28.2s+ and x9,x9,#0x03ffffff+ umlal v22.2d,v12.2s,v0.s[0]+ ubfx x10,x12,#14,#26+ umlal v19.2d,v12.2s,v4.s[0]+ ubfx x11,x13,#14,#26+ umlal v23.2d,v12.2s,v1.s[0]+ add x8,x8,x9,lsl#32 // bfi x8,x9,#32,#32+ umlal v20.2d,v12.2s,v6.s[0]+ fmov d10,x6+ umlal v21.2d,v12.2s,v8.s[0]+ add x12,x3,x12,lsr#40++ umlal v22.2d,v13.2s,v8.s[0]+ add x13,x3,x13,lsr#40+ umlal v19.2d,v13.2s,v2.s[0]+ add x10,x10,x11,lsl#32 // bfi x10,x11,#32,#32+ umlal v23.2d,v13.2s,v0.s[0]+ add x12,x12,x13,lsl#32 // bfi x12,x13,#32,#32+ umlal v20.2d,v13.2s,v4.s[0]+ fmov d11,x8+ umlal v21.2d,v13.2s,v6.s[0]+ fmov d12,x10+ fmov d13,x12++ /////////////////////////////////////////////////////////////////+ // lazy reduction as discussed in "NEON crypto" by D.J. Bernstein+ // and P. Schwabe+ //+ // [see discussion in poly1305-armv4 module]++ ushr v29.2d,v22.2d,#26+ xtn v27.2s,v22.2d+ ushr v30.2d,v19.2d,#26+ and v19.16b,v19.16b,v31.16b+ add v23.2d,v23.2d,v29.2d // h3 -> h4+ bic v27.2s,#0xfc,lsl#24 // &=0x03ffffff+ add v20.2d,v20.2d,v30.2d // h0 -> h1++ ushr v29.2d,v23.2d,#26+ xtn v28.2s,v23.2d+ ushr v30.2d,v20.2d,#26+ xtn v25.2s,v20.2d+ bic v28.2s,#0xfc,lsl#24+ add v21.2d,v21.2d,v30.2d // h1 -> h2++ add v19.2d,v19.2d,v29.2d+ shl v29.2d,v29.2d,#2+ shrn v30.2s,v21.2d,#26+ xtn v26.2s,v21.2d+ add v19.2d,v19.2d,v29.2d // h4 -> h0+ bic v25.2s,#0xfc,lsl#24+ add v27.2s,v27.2s,v30.2s // h2 -> h3+ bic v26.2s,#0xfc,lsl#24++ shrn v29.2s,v19.2d,#26+ xtn v24.2s,v19.2d+ ushr v30.2s,v27.2s,#26+ bic v27.2s,#0xfc,lsl#24+ bic v24.2s,#0xfc,lsl#24+ add v25.2s,v25.2s,v29.2s // h0 -> h1+ add v28.2s,v28.2s,v30.2s // h3 -> h4++ b.hi .Loop_neon++.Lskip_loop:+ dup v16.2d,v16.d[0]+ add v11.2s,v11.2s,v26.2s++ ////////////////////////////////////////////////////////////////+ // multiply (inp[0:1]+hash) or inp[2:3] by r^2:r^1++ adds x2,x2,#32+ b.ne .Long_tail++ dup v16.2d,v11.d[0]+ add v14.2s,v9.2s,v24.2s+ add v17.2s,v12.2s,v27.2s+ add v15.2s,v10.2s,v25.2s+ add v18.2s,v13.2s,v28.2s++.Long_tail:+ dup v14.2d,v14.d[0]+ umull2 v19.2d,v16.4s,v6.4s+ umull2 v22.2d,v16.4s,v1.4s+ umull2 v23.2d,v16.4s,v3.4s+ umull2 v21.2d,v16.4s,v0.4s+ umull2 v20.2d,v16.4s,v8.4s++ dup v15.2d,v15.d[0]+ umlal2 v19.2d,v14.4s,v0.4s+ umlal2 v21.2d,v14.4s,v3.4s+ umlal2 v22.2d,v14.4s,v5.4s+ umlal2 v23.2d,v14.4s,v7.4s+ umlal2 v20.2d,v14.4s,v1.4s++ dup v17.2d,v17.d[0]+ umlal2 v19.2d,v15.4s,v8.4s+ umlal2 v22.2d,v15.4s,v3.4s+ umlal2 v21.2d,v15.4s,v1.4s+ umlal2 v23.2d,v15.4s,v5.4s+ umlal2 v20.2d,v15.4s,v0.4s++ dup v18.2d,v18.d[0]+ umlal2 v22.2d,v17.4s,v0.4s+ umlal2 v23.2d,v17.4s,v1.4s+ umlal2 v19.2d,v17.4s,v4.4s+ umlal2 v20.2d,v17.4s,v6.4s+ umlal2 v21.2d,v17.4s,v8.4s++ umlal2 v22.2d,v18.4s,v8.4s+ umlal2 v19.2d,v18.4s,v2.4s+ umlal2 v23.2d,v18.4s,v0.4s+ umlal2 v20.2d,v18.4s,v4.4s+ umlal2 v21.2d,v18.4s,v6.4s++ b.eq .Lshort_tail++ ////////////////////////////////////////////////////////////////+ // (hash+inp[0:1])*r^4:r^3 and accumulate++ add v9.2s,v9.2s,v24.2s+ umlal v22.2d,v11.2s,v1.2s+ umlal v19.2d,v11.2s,v6.2s+ umlal v23.2d,v11.2s,v3.2s+ umlal v20.2d,v11.2s,v8.2s+ umlal v21.2d,v11.2s,v0.2s++ add v10.2s,v10.2s,v25.2s+ umlal v22.2d,v9.2s,v5.2s+ umlal v19.2d,v9.2s,v0.2s+ umlal v23.2d,v9.2s,v7.2s+ umlal v20.2d,v9.2s,v1.2s+ umlal v21.2d,v9.2s,v3.2s++ add v12.2s,v12.2s,v27.2s+ umlal v22.2d,v10.2s,v3.2s+ umlal v19.2d,v10.2s,v8.2s+ umlal v23.2d,v10.2s,v5.2s+ umlal v20.2d,v10.2s,v0.2s+ umlal v21.2d,v10.2s,v1.2s++ add v13.2s,v13.2s,v28.2s+ umlal v22.2d,v12.2s,v0.2s+ umlal v19.2d,v12.2s,v4.2s+ umlal v23.2d,v12.2s,v1.2s+ umlal v20.2d,v12.2s,v6.2s+ umlal v21.2d,v12.2s,v8.2s++ umlal v22.2d,v13.2s,v8.2s+ umlal v19.2d,v13.2s,v2.2s+ umlal v23.2d,v13.2s,v0.2s+ umlal v20.2d,v13.2s,v4.2s+ umlal v21.2d,v13.2s,v6.2s++.Lshort_tail:+ ////////////////////////////////////////////////////////////////+ // horizontal add++ addp v22.2d,v22.2d,v22.2d+ ldp d8,d9,[sp,#2*__SIZEOF_POINTER__+0] // meet ABI requirements+ addp v19.2d,v19.2d,v19.2d+ ldp d10,d11,[sp,#2*__SIZEOF_POINTER__+16]+ addp v23.2d,v23.2d,v23.2d+ ldp d12,d13,[sp,#2*__SIZEOF_POINTER__+32]+ addp v20.2d,v20.2d,v20.2d+ ldp d14,d15,[sp,#2*__SIZEOF_POINTER__+48]+ addp v21.2d,v21.2d,v21.2d+ ldr x30,[sp,#__SIZEOF_POINTER__]++ ////////////////////////////////////////////////////////////////+ // lazy reduction, but without narrowing++ ushr v29.2d,v22.2d,#26+ and v22.16b,v22.16b,v31.16b+ ushr v30.2d,v19.2d,#26+ and v19.16b,v19.16b,v31.16b++ add v23.2d,v23.2d,v29.2d // h3 -> h4+ add v20.2d,v20.2d,v30.2d // h0 -> h1++ ushr v29.2d,v23.2d,#26+ and v23.16b,v23.16b,v31.16b+ ushr v30.2d,v20.2d,#26+ and v20.16b,v20.16b,v31.16b+ add v21.2d,v21.2d,v30.2d // h1 -> h2++ add v19.2d,v19.2d,v29.2d+ shl v29.2d,v29.2d,#2+ ushr v30.2d,v21.2d,#26+ and v21.16b,v21.16b,v31.16b+ add v19.2d,v19.2d,v29.2d // h4 -> h0+ add v22.2d,v22.2d,v30.2d // h2 -> h3++ ushr v29.2d,v19.2d,#26+ and v19.16b,v19.16b,v31.16b+ ushr v30.2d,v22.2d,#26+ and v22.16b,v22.16b,v31.16b+ add v20.2d,v20.2d,v29.2d // h0 -> h1+ add v23.2d,v23.2d,v30.2d // h3 -> h4++ ////////////////////////////////////////////////////////////////+ // write the result, can be partially reduced++ st4 {v19.s,v20.s,v21.s,v22.s}[0],[x0],#16+ mov x4,#1+ st1 {v23.s}[0],[x0]+ str x4,[x0,#8] // set is_base2_26++ ldr x29,[sp],#2*__SIZEOF_POINTER__+64+.inst 0xd50323bf // autiasp+ ret+.size crypton_poly1305_asm_blocks_neon,.-crypton_poly1305_asm_blocks_neon++.align 5+.Lzeros:+.long 0,0,0,0,0,0,0,0+.byte 80,111,108,121,49,51,48,53,32,102,111,114,32,65,82,77,118,56,44,32,67,82,89,80,84,79,71,65,77,83,32,98,121,32,64,100,111,116,45,97,115,109,0+.align 2+.align 2+#if !defined(__KERNEL__) && !defined(_WIN64)+.comm crypton_armcap_P,4,4+.hidden crypton_armcap_P+#endif++.section .note.GNU-stack,"",%progbits
@@ -0,0 +1,927 @@+#!/usr/bin/env perl+# SPDX-License-Identifier: GPL-1.0+ OR BSD-3-Clause+#+# ====================================================================+# Written by Andy Polyakov, @dot-asm, initially for the OpenSSL+# project.+# ====================================================================+#+# This module implements Poly1305 hash for ARMv8.+#+# June 2015+#+# Numbers are cycles per processed byte with poly1305_blocks alone.+#+# IALU/gcc-4.9 NEON+#+# Apple A7 1.86/+5% 0.72+# Apple A10 0.71+# Apple A14/M1 0.97/+63% 0.48+# Cortex-A53 2.69/+58% 1.47+# Cortex-A57 2.70/+7% 1.14+# Cortex-A76 2.60 1.00+# Cortex-X2 1.00 0.66+# Cortex-X925 1.00 0.53+# Denver 1.64/+50% 1.18(*)+# X-Gene 2.13/+68% 2.27+# Mongoose 1.77/+75% 1.12+# Kryo 2.70/+55% 1.13+# ThunderX2 1.17/+95% 1.36+# Snapdragon X 0.95 0.48+#+# (*) estimate based on resources availability is less than 1.0,+# i.e. measured result is worse than expected, presumably binary+# translator is not almighty;++$flavour=shift;+$output=shift;++if ($flavour && $flavour ne "void") {+ $0 =~ m/(.*[\/\\])[^\/\\]+$/; $dir=$1;+ ( $xlate="${dir}arm-xlate.pl" and -f $xlate ) or+ ( $xlate="${dir}../../perlasm/arm-xlate.pl" and -f $xlate) or+ die "can't locate arm-xlate.pl";++ open STDOUT,"| \"$^X\" $xlate $flavour $output";+} else {+ open STDOUT,">$output";+}++my ($ctx,$inp,$len,$padbit) = map("x$_",(0..3));+my ($mac,$nonce)=($inp,$len);++my ($h0,$h1,$h2,$r0,$r1,$s1,$t0,$t1,$d0,$d1,$d2) = map("x$_",(4..14));++$code.=<<___;+#ifndef __KERNEL__+# include "arm_arch.h"+.extern OPENSSL_armcap_P+#endif++.text++// forward "declarations" are required for Apple+.globl poly1305_blocks+.globl poly1305_emit++.globl poly1305_init+.type poly1305_init,%function+.align 5+poly1305_init:+ cmp $inp,xzr+ stp xzr,xzr,[$ctx] // zero hash value+ stp xzr,xzr,[$ctx,#16] // [along with is_base2_26]++ csel c0,czr,c0,eq+ b.eq .Lno_key++#ifndef __KERNEL__+ adrp c17,OPENSSL_armcap_P+ ldr w17,[c17,#:lo12:OPENSSL_armcap_P]+#endif++ ldp $r0,$r1,[$inp] // load key+ mov $s1,#0xfffffffc0fffffff+ movk $s1,#0x0fff,lsl#48+#ifdef __AARCH64EB__+ rev $r0,$r0 // flip bytes+ rev $r1,$r1+#endif+ and $r0,$r0,$s1 // &=0ffffffc0fffffff+ and $s1,$s1,#-4+ and $r1,$r1,$s1 // &=0ffffffc0ffffffc+ mov w#$s1,#-1+ stp $r0,$r1,[$ctx,#32] // save key value+ str w#$s1,[$ctx,#48] // impossible key power value++#ifndef __KERNEL__+ tst w17,#ARMV7_NEON++ adr c13,.Lpoly1305_blocks+ adr c15,.Lpoly1305_blocks_neon+ adr c14,.Lpoly1305_emit++ csel c13,c13,c15,eq+# ifdef __CHERI_PURE_CAPABILITY__+ add c13, c13, #1+ add c14, c14, #1+ seal c13, c13, rb+ seal c14, c14, rb+# endif++# ifdef __ILP32__+ stp w13,w14,[$len]+# else+ stp c13,c14,[$len]+# endif+ mov x0,#1+#else+ mov x0,#0+#endif+.Lno_key:+ ret+.size poly1305_init,.-poly1305_init++.type poly1305_blocks,%function+.align 5+poly1305_blocks:+.Lpoly1305_blocks:+ ands $len,$len,#-16+ b.eq .Lno_data++ ldp $h0,$h1,[$ctx] // load hash value+ ldp $h2,x17,[$ctx,#16] // [along with is_base2_26]+ ldp $r0,$r1,[$ctx,#32] // load key value++#ifdef __AARCH64EB__+ lsr $d0,$h0,#32+ mov w#$d1,w#$h0+ lsr $d2,$h1,#32+ mov w15,w#$h1+ lsr x16,$h2,#32+#else+ mov w#$d0,w#$h0+ lsr $d1,$h0,#32+ mov w#$d2,w#$h1+ lsr x15,$h1,#32+ mov w16,w#$h2+#endif++ add $d0,$d0,$d1,lsl#26 // base 2^26 -> base 2^64+ lsr $d1,$d2,#12+ adds $d0,$d0,$d2,lsl#52+ add $d1,$d1,x15,lsl#14+ adc $d1,$d1,xzr+ lsr $d2,x16,#24+ adds $d1,$d1,x16,lsl#40+ adc $d2,$d2,xzr++ cmp x17,#0 // is_base2_26?+ add $s1,$r1,$r1,lsr#2 // s1 = r1 + (r1 >> 2)+ csel $h0,$h0,$d0,eq // choose between radixes+ csel $h1,$h1,$d1,eq+ csel $h2,$h2,$d2,eq++.Loop:+ ldp $t0,$t1,[$inp],#16 // load input+ sub $len,$len,#16+#ifdef __AARCH64EB__+ rev $t0,$t0+ rev $t1,$t1+#endif+ adds $h0,$h0,$t0 // accumulate input+ adcs $h1,$h1,$t1++ mul $d0,$h0,$r0 // h0*r0+ adc $h2,$h2,$padbit+ umulh $d1,$h0,$r0++ mul $t0,$h1,$s1 // h1*5*r1+ umulh $t1,$h1,$s1++ adds $d0,$d0,$t0+ mul $t0,$h0,$r1 // h0*r1+ adc $d1,$d1,$t1+ umulh $d2,$h0,$r1++ adds $d1,$d1,$t0+ mul $t0,$h1,$r0 // h1*r0+ adc $d2,$d2,xzr+ umulh $t1,$h1,$r0++ adds $d1,$d1,$t0+ mul $t0,$h2,$s1 // h2*5*r1+ adc $d2,$d2,$t1+ mul $t1,$h2,$r0 // h2*r0++ adds $d1,$d1,$t0+ adc $d2,$d2,$t1++ and $t0,$d2,#-4 // final reduction+ and $h2,$d2,#3+ add $t0,$t0,$d2,lsr#2+ adds $h0,$d0,$t0+ adcs $h1,$d1,xzr+ adc $h2,$h2,xzr++ cbnz $len,.Loop++ stp $h0,$h1,[$ctx] // store hash value+ stp $h2,xzr,[$ctx,#16] // [and clear is_base2_26]++.Lno_data:+ ret+.size poly1305_blocks,.-poly1305_blocks++.type poly1305_emit,%function+.align 5+poly1305_emit:+.Lpoly1305_emit:+ ldp $h0,$h1,[$ctx] // load hash base 2^64+ ldp $h2,$r0,[$ctx,#16] // [along with is_base2_26]+ ldp $t0,$t1,[$nonce] // load nonce++#ifdef __AARCH64EB__+ lsr $d0,$h0,#32+ mov w#$d1,w#$h0+ lsr $d2,$h1,#32+ mov w15,w#$h1+ lsr x16,$h2,#32+#else+ mov w#$d0,w#$h0+ lsr $d1,$h0,#32+ mov w#$d2,w#$h1+ lsr x15,$h1,#32+ mov w16,w#$h2+#endif++ add $d0,$d0,$d1,lsl#26 // base 2^26 -> base 2^64+ lsr $d1,$d2,#12+ adds $d0,$d0,$d2,lsl#52+ add $d1,$d1,x15,lsl#14+ adc $d1,$d1,xzr+ lsr $d2,x16,#24+ adds $d1,$d1,x16,lsl#40+ adc $d2,$d2,xzr++ cmp $r0,#0 // is_base2_26?+ csel $h0,$h0,$d0,eq // choose between radixes+ csel $h1,$h1,$d1,eq+ csel $h2,$h2,$d2,eq++ adds $d0,$h0,#5 // compare to modulus+ adcs $d1,$h1,xzr+ adc $d2,$h2,xzr++ tst $d2,#-4 // see if it's carried/borrowed++ csel $h0,$h0,$d0,eq+ csel $h1,$h1,$d1,eq++#ifdef __AARCH64EB__+ ror $t0,$t0,#32 // flip nonce words+ ror $t1,$t1,#32+#endif+ adds $h0,$h0,$t0 // accumulate nonce+ adc $h1,$h1,$t1+#ifdef __AARCH64EB__+ rev $h0,$h0 // flip output bytes+ rev $h1,$h1+#endif+ stp $h0,$h1,[$mac] // write result++ ret+.size poly1305_emit,.-poly1305_emit+___+my ($R0,$R1,$S1,$R2,$S2,$R3,$S3,$R4,$S4) = map("v$_.4s",(0..8));+my ($IN01_0,$IN01_1,$IN01_2,$IN01_3,$IN01_4) = map("v$_.2s",(9..13));+my ($IN23_0,$IN23_1,$IN23_2,$IN23_3,$IN23_4) = map("v$_.2s",(14..18));+my ($ACC0,$ACC1,$ACC2,$ACC3,$ACC4) = map("v$_.2d",(19..23));+my ($H0,$H1,$H2,$H3,$H4) = map("v$_.2s",(24..28));+my ($T0,$T1,$MASK) = map("v$_",(29..31));++my ($in2,$zeros)=("x16","x17");+my $is_base2_26 = $zeros; # borrow++$code.=<<___;+.type poly1305_mult,%function+.align 5+poly1305_mult:+ mul $d0,$h0,$r0 // h0*r0+ umulh $d1,$h0,$r0++ mul $t0,$h1,$s1 // h1*5*r1+ umulh $t1,$h1,$s1++ adds $d0,$d0,$t0+ mul $t0,$h0,$r1 // h0*r1+ adc $d1,$d1,$t1+ umulh $d2,$h0,$r1++ adds $d1,$d1,$t0+ mul $t0,$h1,$r0 // h1*r0+ adc $d2,$d2,xzr+ umulh $t1,$h1,$r0++ adds $d1,$d1,$t0+ mul $t0,$h2,$s1 // h2*5*r1+ adc $d2,$d2,$t1+ mul $t1,$h2,$r0 // h2*r0++ adds $d1,$d1,$t0+ adc $d2,$d2,$t1++ and $t0,$d2,#-4 // final reduction+ and $h2,$d2,#3+ add $t0,$t0,$d2,lsr#2+ adds $h0,$d0,$t0+ adcs $h1,$d1,xzr+ adc $h2,$h2,xzr++ ret+.size poly1305_mult,.-poly1305_mult++.type poly1305_splat,%function+.align 4+poly1305_splat:+ and x12,$h0,#0x03ffffff // base 2^64 -> base 2^26+ ubfx x13,$h0,#26,#26+ extr x14,$h1,$h0,#52+ and x14,x14,#0x03ffffff+ ubfx x15,$h1,#14,#26+ extr x16,$h2,$h1,#40++ str w12,[$ctx,#16*0] // r0+ add w12,w13,w13,lsl#2 // r1*5+ str w13,[$ctx,#16*1] // r1+ add w13,w14,w14,lsl#2 // r2*5+ str w12,[$ctx,#16*2] // s1+ str w14,[$ctx,#16*3] // r2+ add w14,w15,w15,lsl#2 // r3*5+ str w13,[$ctx,#16*4] // s2+ str w15,[$ctx,#16*5] // r3+ add w15,w16,w16,lsl#2 // r4*5+ str w14,[$ctx,#16*6] // s3+ str w16,[$ctx,#16*7] // r4+ str w15,[$ctx,#16*8] // s4++ ret+.size poly1305_splat,.-poly1305_splat++#ifdef __KERNEL__+.globl poly1305_blocks_neon+#endif+.type poly1305_blocks_neon,%function+.align 5+poly1305_blocks_neon:+.Lpoly1305_blocks_neon:+ ldr $is_base2_26,[$ctx,#24]+ cmp $len,#128+ b.lo .Lpoly1305_blocks++ .inst 0xd503233f // paciasp+ stp c29,c30,[csp,#-2*__SIZEOF_POINTER__-64]!+ add c29,csp,#0++ stp d8,d9,[csp,#2*__SIZEOF_POINTER__+0] // meet ABI requirements+ stp d10,d11,[csp,#2*__SIZEOF_POINTER__+16]+ stp d12,d13,[csp,#2*__SIZEOF_POINTER__+32]+ stp d14,d15,[csp,#2*__SIZEOF_POINTER__+48]++ cbz $is_base2_26,.Lbase2_64_neon++ ldp w10,w11,[$ctx] // load hash value base 2^26+ ldp w12,w13,[$ctx,#8]+ ldr w14,[$ctx,#16]++ tst $len,#31+ b.eq .Leven_neon++ ldp $r0,$r1,[$ctx,#32] // load key value++ add $h0,x10,x11,lsl#26 // base 2^26 -> base 2^64+ lsr $h1,x12,#12+ adds $h0,$h0,x12,lsl#52+ add $h1,$h1,x13,lsl#14+ adc $h1,$h1,xzr+ lsr $h2,x14,#24+ adds $h1,$h1,x14,lsl#40+ adc $d2,$h2,xzr // can be partially reduced...++ ldp $d0,$d1,[$inp],#16 // load input+ sub $len,$len,#16+ add $s1,$r1,$r1,lsr#2 // s1 = r1 + (r1 >> 2)++#ifdef __AARCH64EB__+ rev $d0,$d0+ rev $d1,$d1+#endif+ adds $h0,$h0,$d0 // accumulate input+ adcs $h1,$h1,$d1+ adc $h2,$h2,$padbit++ bl poly1305_mult++ and x10,$h0,#0x03ffffff // base 2^64 -> base 2^26+ ubfx x11,$h0,#26,#26+ extr x12,$h1,$h0,#52+ and x12,x12,#0x03ffffff+ ubfx x13,$h1,#14,#26+ extr x14,$h2,$h1,#40++ b .Leven_neon++.align 4+.Lbase2_64_neon:+ ldp $r0,$r1,[$ctx,#32] // load key value++ ldp $h0,$h1,[$ctx] // load hash value base 2^64+ ldr $h2,[$ctx,#16]++ tst $len,#31+ b.eq .Linit_neon++ ldp $d0,$d1,[$inp],#16 // load input+ sub $len,$len,#16+ add $s1,$r1,$r1,lsr#2 // s1 = r1 + (r1 >> 2)+#ifdef __AARCH64EB__+ rev $d0,$d0+ rev $d1,$d1+#endif+ adds $h0,$h0,$d0 // accumulate input+ adcs $h1,$h1,$d1+ adc $h2,$h2,$padbit++ bl poly1305_mult++.Linit_neon:+ ldr w17,[$ctx,#48] // first table element+ and x10,$h0,#0x03ffffff // base 2^64 -> base 2^26+ ubfx x11,$h0,#26,#26+ extr x12,$h1,$h0,#52+ and x12,x12,#0x03ffffff+ ubfx x13,$h1,#14,#26+ extr x14,$h2,$h1,#40++ cmp w17,#-1 // is value impossible?+ b.ne .Leven_neon++ fmov ${H0},x10+ fmov ${H1},x11+ fmov ${H2},x12+ fmov ${H3},x13+ fmov ${H4},x14++ ////////////////////////////////// initialize r^n table+ mov $h0,$r0 // r^1+ add $s1,$r1,$r1,lsr#2 // s1 = r1 + (r1 >> 2)+ mov $h1,$r1+ mov $h2,xzr+ cadd $ctx,$ctx,#48+12+ bl poly1305_splat++ bl poly1305_mult // r^2+ csub $ctx,$ctx,#4+ bl poly1305_splat++ bl poly1305_mult // r^3+ csub $ctx,$ctx,#4+ bl poly1305_splat++ bl poly1305_mult // r^4+ csub $ctx,$ctx,#4+ bl poly1305_splat+ csub $ctx,$ctx,#48 // restore original $ctx+ b .Ldo_neon++.align 4+.Leven_neon:+ fmov ${H0},x10+ fmov ${H1},x11+ fmov ${H2},x12+ fmov ${H3},x13+ fmov ${H4},x14++.Ldo_neon:+ ldp x8,x12,[$inp,#32] // inp[2:3]+ subs $len,$len,#64+ ldp x9,x13,[$inp,#48]+ cadd $in2,$inp,#96+ adr $zeros,.Lzeros++ lsl $padbit,$padbit,#24+ cadd x15,$ctx,#48++#ifdef __AARCH64EB__+ rev x8,x8+ rev x12,x12+ rev x9,x9+ rev x13,x13+#endif+ and x4,x8,#0x03ffffff // base 2^64 -> base 2^26+ and x5,x9,#0x03ffffff+ ubfx x6,x8,#26,#26+ ubfx x7,x9,#26,#26+ add x4,x4,x5,lsl#32 // bfi x4,x5,#32,#32+ extr x8,x12,x8,#52+ extr x9,x13,x9,#52+ add x6,x6,x7,lsl#32 // bfi x6,x7,#32,#32+ fmov $IN23_0,x4+ and x8,x8,#0x03ffffff+ and x9,x9,#0x03ffffff+ ubfx x10,x12,#14,#26+ ubfx x11,x13,#14,#26+ add x12,$padbit,x12,lsr#40+ add x13,$padbit,x13,lsr#40+ add x8,x8,x9,lsl#32 // bfi x8,x9,#32,#32+ fmov $IN23_1,x6+ add x10,x10,x11,lsl#32 // bfi x10,x11,#32,#32+ add x12,x12,x13,lsl#32 // bfi x12,x13,#32,#32+ fmov $IN23_2,x8+ fmov $IN23_3,x10+ fmov $IN23_4,x12++ ldp x8,x12,[$inp],#16 // inp[0:1]+ ldp x9,x13,[$inp],#48++ ld1 {$R0,$R1,$S1,$R2},[x15],#64+ ld1 {$S2,$R3,$S3,$R4},[x15],#64+ ld1 {$S4},[x15]++#ifdef __AARCH64EB__+ rev x8,x8+ rev x12,x12+ rev x9,x9+ rev x13,x13+#endif+ and x4,x8,#0x03ffffff // base 2^64 -> base 2^26+ and x5,x9,#0x03ffffff+ ubfx x6,x8,#26,#26+ ubfx x7,x9,#26,#26+ add x4,x4,x5,lsl#32 // bfi x4,x5,#32,#32+ extr x8,x12,x8,#52+ extr x9,x13,x9,#52+ add x6,x6,x7,lsl#32 // bfi x6,x7,#32,#32+ fmov $IN01_0,x4+ and x8,x8,#0x03ffffff+ and x9,x9,#0x03ffffff+ ubfx x10,x12,#14,#26+ ubfx x11,x13,#14,#26+ add x12,$padbit,x12,lsr#40+ add x13,$padbit,x13,lsr#40+ add x8,x8,x9,lsl#32 // bfi x8,x9,#32,#32+ fmov $IN01_1,x6+ add x10,x10,x11,lsl#32 // bfi x10,x11,#32,#32+ add x12,x12,x13,lsl#32 // bfi x12,x13,#32,#32+ movi $MASK.2d,#-1+ fmov $IN01_2,x8+ fmov $IN01_3,x10+ fmov $IN01_4,x12+ ushr $MASK.2d,$MASK.2d,#38++ b.ls .Lskip_loop++.align 4+.Loop_neon:+ ////////////////////////////////////////////////////////////////+ // ((inp[0]*r^4+inp[2]*r^2+inp[4])*r^4+inp[6]*r^2+ // ((inp[1]*r^4+inp[3]*r^2+inp[5])*r^3+inp[7]*r+ // \___________________/+ // ((inp[0]*r^4+inp[2]*r^2+inp[4])*r^4+inp[6]*r^2+inp[8])*r^2+ // ((inp[1]*r^4+inp[3]*r^2+inp[5])*r^4+inp[7]*r^2+inp[9])*r+ // \___________________/ \____________________/+ //+ // Note that we start with inp[2:3]*r^2. This is because it+ // doesn't depend on reduction in previous iteration.+ ////////////////////////////////////////////////////////////////+ // d4 = h0*r4 + h1*r3 + h2*r2 + h3*r1 + h4*r0+ // d3 = h0*r3 + h1*r2 + h2*r1 + h3*r0 + h4*5*r4+ // d2 = h0*r2 + h1*r1 + h2*r0 + h3*5*r4 + h4*5*r3+ // d1 = h0*r1 + h1*r0 + h2*5*r4 + h3*5*r3 + h4*5*r2+ // d0 = h0*r0 + h1*5*r4 + h2*5*r3 + h3*5*r2 + h4*5*r1++ subs $len,$len,#64+ umull $ACC4,$IN23_0,${R4}[2]+ csel c#$in2,c#$zeros,c#$in2,lo+ umull $ACC3,$IN23_0,${R3}[2]+ umull $ACC2,$IN23_0,${R2}[2]+ ldp x8,x12,[$in2],#16 // inp[2:3] (or zero)+ umull $ACC1,$IN23_0,${R1}[2]+ ldp x9,x13,[$in2],#48+ umull $ACC0,$IN23_0,${R0}[2]+#ifdef __AARCH64EB__+ rev x8,x8+ rev x12,x12+ rev x9,x9+ rev x13,x13+#endif++ umlal $ACC4,$IN23_1,${R3}[2]+ and x4,x8,#0x03ffffff // base 2^64 -> base 2^26+ umlal $ACC3,$IN23_1,${R2}[2]+ and x5,x9,#0x03ffffff+ umlal $ACC2,$IN23_1,${R1}[2]+ ubfx x6,x8,#26,#26+ umlal $ACC1,$IN23_1,${R0}[2]+ ubfx x7,x9,#26,#26+ umlal $ACC0,$IN23_1,${S4}[2]+ add x4,x4,x5,lsl#32 // bfi x4,x5,#32,#32++ umlal $ACC4,$IN23_2,${R2}[2]+ extr x8,x12,x8,#52+ umlal $ACC3,$IN23_2,${R1}[2]+ extr x9,x13,x9,#52+ umlal $ACC2,$IN23_2,${R0}[2]+ add x6,x6,x7,lsl#32 // bfi x6,x7,#32,#32+ umlal $ACC1,$IN23_2,${S4}[2]+ fmov $IN23_0,x4+ umlal $ACC0,$IN23_2,${S3}[2]+ and x8,x8,#0x03ffffff++ umlal $ACC4,$IN23_3,${R1}[2]+ and x9,x9,#0x03ffffff+ umlal $ACC3,$IN23_3,${R0}[2]+ ubfx x10,x12,#14,#26+ umlal $ACC2,$IN23_3,${S4}[2]+ ubfx x11,x13,#14,#26+ umlal $ACC1,$IN23_3,${S3}[2]+ add x8,x8,x9,lsl#32 // bfi x8,x9,#32,#32+ umlal $ACC0,$IN23_3,${S2}[2]+ fmov $IN23_1,x6++ add $IN01_2,$IN01_2,$H2+ add x12,$padbit,x12,lsr#40+ umlal $ACC4,$IN23_4,${R0}[2]+ add x13,$padbit,x13,lsr#40+ umlal $ACC3,$IN23_4,${S4}[2]+ add x10,x10,x11,lsl#32 // bfi x10,x11,#32,#32+ umlal $ACC2,$IN23_4,${S3}[2]+ add x12,x12,x13,lsl#32 // bfi x12,x13,#32,#32+ umlal $ACC1,$IN23_4,${S2}[2]+ fmov $IN23_2,x8+ umlal $ACC0,$IN23_4,${S1}[2]+ fmov $IN23_3,x10++ ////////////////////////////////////////////////////////////////+ // (hash+inp[0:1])*r^4 and accumulate++ add $IN01_0,$IN01_0,$H0+ fmov $IN23_4,x12+ umlal $ACC3,$IN01_2,${R1}[0]+ ldp x8,x12,[$inp],#16 // inp[0:1]+ umlal $ACC0,$IN01_2,${S3}[0]+ ldp x9,x13,[$inp],#48+ umlal $ACC4,$IN01_2,${R2}[0]+ umlal $ACC1,$IN01_2,${S4}[0]+ umlal $ACC2,$IN01_2,${R0}[0]+#ifdef __AARCH64EB__+ rev x8,x8+ rev x12,x12+ rev x9,x9+ rev x13,x13+#endif++ add $IN01_1,$IN01_1,$H1+ umlal $ACC3,$IN01_0,${R3}[0]+ umlal $ACC4,$IN01_0,${R4}[0]+ and x4,x8,#0x03ffffff // base 2^64 -> base 2^26+ umlal $ACC2,$IN01_0,${R2}[0]+ and x5,x9,#0x03ffffff+ umlal $ACC0,$IN01_0,${R0}[0]+ ubfx x6,x8,#26,#26+ umlal $ACC1,$IN01_0,${R1}[0]+ ubfx x7,x9,#26,#26++ add $IN01_3,$IN01_3,$H3+ add x4,x4,x5,lsl#32 // bfi x4,x5,#32,#32+ umlal $ACC3,$IN01_1,${R2}[0]+ extr x8,x12,x8,#52+ umlal $ACC4,$IN01_1,${R3}[0]+ extr x9,x13,x9,#52+ umlal $ACC0,$IN01_1,${S4}[0]+ add x6,x6,x7,lsl#32 // bfi x6,x7,#32,#32+ umlal $ACC2,$IN01_1,${R1}[0]+ fmov $IN01_0,x4+ umlal $ACC1,$IN01_1,${R0}[0]+ and x8,x8,#0x03ffffff++ add $IN01_4,$IN01_4,$H4+ and x9,x9,#0x03ffffff+ umlal $ACC3,$IN01_3,${R0}[0]+ ubfx x10,x12,#14,#26+ umlal $ACC0,$IN01_3,${S2}[0]+ ubfx x11,x13,#14,#26+ umlal $ACC4,$IN01_3,${R1}[0]+ add x8,x8,x9,lsl#32 // bfi x8,x9,#32,#32+ umlal $ACC1,$IN01_3,${S3}[0]+ fmov $IN01_1,x6+ umlal $ACC2,$IN01_3,${S4}[0]+ add x12,$padbit,x12,lsr#40++ umlal $ACC3,$IN01_4,${S4}[0]+ add x13,$padbit,x13,lsr#40+ umlal $ACC0,$IN01_4,${S1}[0]+ add x10,x10,x11,lsl#32 // bfi x10,x11,#32,#32+ umlal $ACC4,$IN01_4,${R0}[0]+ add x12,x12,x13,lsl#32 // bfi x12,x13,#32,#32+ umlal $ACC1,$IN01_4,${S2}[0]+ fmov $IN01_2,x8+ umlal $ACC2,$IN01_4,${S3}[0]+ fmov $IN01_3,x10+ fmov $IN01_4,x12++ /////////////////////////////////////////////////////////////////+ // lazy reduction as discussed in "NEON crypto" by D.J. Bernstein+ // and P. Schwabe+ //+ // [see discussion in poly1305-armv4 module]++ ushr $T0.2d,$ACC3,#26+ xtn $H3,$ACC3+ ushr $T1.2d,$ACC0,#26+ and $ACC0,$ACC0,$MASK.2d+ add $ACC4,$ACC4,$T0.2d // h3 -> h4+ bic $H3,#0xfc,lsl#24 // &=0x03ffffff+ add $ACC1,$ACC1,$T1.2d // h0 -> h1++ ushr $T0.2d,$ACC4,#26+ xtn $H4,$ACC4+ ushr $T1.2d,$ACC1,#26+ xtn $H1,$ACC1+ bic $H4,#0xfc,lsl#24+ add $ACC2,$ACC2,$T1.2d // h1 -> h2++ add $ACC0,$ACC0,$T0.2d+ shl $T0.2d,$T0.2d,#2+ shrn $T1.2s,$ACC2,#26+ xtn $H2,$ACC2+ add $ACC0,$ACC0,$T0.2d // h4 -> h0+ bic $H1,#0xfc,lsl#24+ add $H3,$H3,$T1.2s // h2 -> h3+ bic $H2,#0xfc,lsl#24++ shrn $T0.2s,$ACC0,#26+ xtn $H0,$ACC0+ ushr $T1.2s,$H3,#26+ bic $H3,#0xfc,lsl#24+ bic $H0,#0xfc,lsl#24+ add $H1,$H1,$T0.2s // h0 -> h1+ add $H4,$H4,$T1.2s // h3 -> h4++ b.hi .Loop_neon++.Lskip_loop:+ dup $IN23_2,${IN23_2}[0]+ add $IN01_2,$IN01_2,$H2++ ////////////////////////////////////////////////////////////////+ // multiply (inp[0:1]+hash) or inp[2:3] by r^2:r^1++ adds $len,$len,#32+ b.ne .Long_tail++ dup $IN23_2,${IN01_2}[0]+ add $IN23_0,$IN01_0,$H0+ add $IN23_3,$IN01_3,$H3+ add $IN23_1,$IN01_1,$H1+ add $IN23_4,$IN01_4,$H4++.Long_tail:+ dup $IN23_0,${IN23_0}[0]+ umull2 $ACC0,$IN23_2,${S3}+ umull2 $ACC3,$IN23_2,${R1}+ umull2 $ACC4,$IN23_2,${R2}+ umull2 $ACC2,$IN23_2,${R0}+ umull2 $ACC1,$IN23_2,${S4}++ dup $IN23_1,${IN23_1}[0]+ umlal2 $ACC0,$IN23_0,${R0}+ umlal2 $ACC2,$IN23_0,${R2}+ umlal2 $ACC3,$IN23_0,${R3}+ umlal2 $ACC4,$IN23_0,${R4}+ umlal2 $ACC1,$IN23_0,${R1}++ dup $IN23_3,${IN23_3}[0]+ umlal2 $ACC0,$IN23_1,${S4}+ umlal2 $ACC3,$IN23_1,${R2}+ umlal2 $ACC2,$IN23_1,${R1}+ umlal2 $ACC4,$IN23_1,${R3}+ umlal2 $ACC1,$IN23_1,${R0}++ dup $IN23_4,${IN23_4}[0]+ umlal2 $ACC3,$IN23_3,${R0}+ umlal2 $ACC4,$IN23_3,${R1}+ umlal2 $ACC0,$IN23_3,${S2}+ umlal2 $ACC1,$IN23_3,${S3}+ umlal2 $ACC2,$IN23_3,${S4}++ umlal2 $ACC3,$IN23_4,${S4}+ umlal2 $ACC0,$IN23_4,${S1}+ umlal2 $ACC4,$IN23_4,${R0}+ umlal2 $ACC1,$IN23_4,${S2}+ umlal2 $ACC2,$IN23_4,${S3}++ b.eq .Lshort_tail++ ////////////////////////////////////////////////////////////////+ // (hash+inp[0:1])*r^4:r^3 and accumulate++ add $IN01_0,$IN01_0,$H0+ umlal $ACC3,$IN01_2,${R1}+ umlal $ACC0,$IN01_2,${S3}+ umlal $ACC4,$IN01_2,${R2}+ umlal $ACC1,$IN01_2,${S4}+ umlal $ACC2,$IN01_2,${R0}++ add $IN01_1,$IN01_1,$H1+ umlal $ACC3,$IN01_0,${R3}+ umlal $ACC0,$IN01_0,${R0}+ umlal $ACC4,$IN01_0,${R4}+ umlal $ACC1,$IN01_0,${R1}+ umlal $ACC2,$IN01_0,${R2}++ add $IN01_3,$IN01_3,$H3+ umlal $ACC3,$IN01_1,${R2}+ umlal $ACC0,$IN01_1,${S4}+ umlal $ACC4,$IN01_1,${R3}+ umlal $ACC1,$IN01_1,${R0}+ umlal $ACC2,$IN01_1,${R1}++ add $IN01_4,$IN01_4,$H4+ umlal $ACC3,$IN01_3,${R0}+ umlal $ACC0,$IN01_3,${S2}+ umlal $ACC4,$IN01_3,${R1}+ umlal $ACC1,$IN01_3,${S3}+ umlal $ACC2,$IN01_3,${S4}++ umlal $ACC3,$IN01_4,${S4}+ umlal $ACC0,$IN01_4,${S1}+ umlal $ACC4,$IN01_4,${R0}+ umlal $ACC1,$IN01_4,${S2}+ umlal $ACC2,$IN01_4,${S3}++.Lshort_tail:+ ////////////////////////////////////////////////////////////////+ // horizontal add++ addp $ACC3,$ACC3,$ACC3+ ldp d8,d9,[sp,#2*__SIZEOF_POINTER__+0] // meet ABI requirements+ addp $ACC0,$ACC0,$ACC0+ ldp d10,d11,[sp,#2*__SIZEOF_POINTER__+16]+ addp $ACC4,$ACC4,$ACC4+ ldp d12,d13,[sp,#2*__SIZEOF_POINTER__+32]+ addp $ACC1,$ACC1,$ACC1+ ldp d14,d15,[sp,#2*__SIZEOF_POINTER__+48]+ addp $ACC2,$ACC2,$ACC2+ ldr c30,[csp,#__SIZEOF_POINTER__]++ ////////////////////////////////////////////////////////////////+ // lazy reduction, but without narrowing++ ushr $T0.2d,$ACC3,#26+ and $ACC3,$ACC3,$MASK.2d+ ushr $T1.2d,$ACC0,#26+ and $ACC0,$ACC0,$MASK.2d++ add $ACC4,$ACC4,$T0.2d // h3 -> h4+ add $ACC1,$ACC1,$T1.2d // h0 -> h1++ ushr $T0.2d,$ACC4,#26+ and $ACC4,$ACC4,$MASK.2d+ ushr $T1.2d,$ACC1,#26+ and $ACC1,$ACC1,$MASK.2d+ add $ACC2,$ACC2,$T1.2d // h1 -> h2++ add $ACC0,$ACC0,$T0.2d+ shl $T0.2d,$T0.2d,#2+ ushr $T1.2d,$ACC2,#26+ and $ACC2,$ACC2,$MASK.2d+ add $ACC0,$ACC0,$T0.2d // h4 -> h0+ add $ACC3,$ACC3,$T1.2d // h2 -> h3++ ushr $T0.2d,$ACC0,#26+ and $ACC0,$ACC0,$MASK.2d+ ushr $T1.2d,$ACC3,#26+ and $ACC3,$ACC3,$MASK.2d+ add $ACC1,$ACC1,$T0.2d // h0 -> h1+ add $ACC4,$ACC4,$T1.2d // h3 -> h4++ ////////////////////////////////////////////////////////////////+ // write the result, can be partially reduced++ st4 {$ACC0,$ACC1,$ACC2,$ACC3}[0],[$ctx],#16+ mov x4,#1+ st1 {$ACC4}[0],[$ctx]+ str x4,[$ctx,#8] // set is_base2_26++ ldr c29,[csp],#2*__SIZEOF_POINTER__+64+ .inst 0xd50323bf // autiasp+ ret+.size poly1305_blocks_neon,.-poly1305_blocks_neon++.align 5+.Lzeros:+.long 0,0,0,0,0,0,0,0+.asciz "Poly1305 for ARMv8, CRYPTOGAMS by \@dot-asm"+.align 2+#if !defined(__KERNEL__) && !defined(_WIN64)+.comm OPENSSL_armcap_P,4,4+.hidden OPENSSL_armcap_P+#endif+___++foreach (split("\n",$code)) {+ s/\b(shrn\s+v[0-9]+)\.[24]d/$1.2s/ or+ s/\b(fmov\s+)v([0-9]+)[^,]*,\s*x([0-9]+)/$1d$2,x$3/ or+ (m/\bdup\b/ and (s/\.[24]s/.2d/g or 1)) or+ (m/\b(eor|and)/ and (s/\.[248][sdh]/.16b/g or 1)) or+ (m/\bum(ul|la)l\b/ and (s/\.4s/.2s/g or 1)) or+ (m/\bum(ul|la)l2\b/ and (s/\.2s/.4s/g or 1)) or+ (m/\bst[1-4]\s+{[^}]+}\[/ and (s/\.[24]d/.s/g or 1));++ s/\.[124]([sd])\[/.$1\[/;+ s/([cw])#x([0-9]+)/$1$2/g;++ print $_,"\n";+}+close STDOUT;
@@ -0,0 +1,2033 @@+.text ++++.globl crypton_poly1305_asm_init+.hidden crypton_poly1305_asm_init+.globl crypton_poly1305_asm_blocks+.hidden crypton_poly1305_asm_blocks+.globl crypton_poly1305_asm_emit+.hidden crypton_poly1305_asm_emit++.type crypton_poly1305_asm_init,@function+.align 32+crypton_poly1305_asm_init:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa++ xorq %rax,%rax+ movq %rax,0(%rdi)+ movq %rax,8(%rdi)+ movq %rax,16(%rdi)++ cmpq $0,%rsi+ je .Lno_key++ movq $0x0ffffffc0fffffff,%rax+ leaq -3(%rax),%rcx+ andq 0(%rsi),%rax+ andq 8(%rsi),%rcx+ movq %rax,24(%rdi)+ movq %rcx,32(%rdi)+ movl $-1,48(%rdi)+ leaq crypton_poly1305_asm_blocks(%rip),%r10+ leaq crypton_poly1305_asm_emit(%rip),%r11+ movq crypton_ia32cap_P+4(%rip),%r9+ leaq crypton_poly1305_asm_blocks_avx(%rip),%rax+ btq $28,%r9+ cmovcq %rax,%r10+ leaq crypton_poly1305_asm_blocks_avx2(%rip),%rax+ btq $37,%r9+ cmovcq %rax,%r10+ movq %r10,0(%rdx)+ movq %r11,8(%rdx)+ movl $1,%eax+.Lno_key:+ .byte 0xf3,0xc3+.cfi_endproc+.size crypton_poly1305_asm_init,.-crypton_poly1305_asm_init++.type crypton_poly1305_asm_blocks,@function+.align 32+crypton_poly1305_asm_blocks:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa+++.Lblocks:+ shrq $4,%rdx+ jz .Lno_data++ pushq %rbx+.cfi_adjust_cfa_offset 8+.cfi_offset %rbx,-16+ pushq %rbp+.cfi_adjust_cfa_offset 8+.cfi_offset %rbp,-24+ pushq %r12+.cfi_adjust_cfa_offset 8+.cfi_offset %r12,-32+ pushq %r13+.cfi_adjust_cfa_offset 8+.cfi_offset %r13,-40+ pushq %r14+.cfi_adjust_cfa_offset 8+.cfi_offset %r14,-48+ pushq %r15+.cfi_adjust_cfa_offset 8+.cfi_offset %r15,-56+ leaq -8(%rsp),%rsp+.cfi_adjust_cfa_offset 8+.Lblocks_body:++ movq %rdx,%r15++ movq 24(%rdi),%r11+ movq 32(%rdi),%r13++ movq 0(%rdi),%r14+ movq 8(%rdi),%rbx+ movq 16(%rdi),%rbp++ movl %r14d,%eax+ movl 4(%rdi),%edx+ movl %ebx,%r8d+ movl 12(%rdi),%r10d+ movl %ebp,%r12d++ shlq $26,%rdx+ movq %r8,%r9+ shlq $52,%r8+ addq %rdx,%rax+ shrq $12,%r9+ addq %rax,%r8+ adcq $0,%r9++ shlq $14,%r10+ movq %r12,%rax+ shrq $24,%r12+ addq %r10,%r9+ shlq $40,%rax+ addq %rax,%r9+ adcq $0,%r12++ cmpq $4,%rbp++ cmovaq %r8,%r14+ cmovaq %r9,%rbx+ cmovaq %r12,%rbp++ movq %r13,%r12+ shrq $2,%r13+ movq %r12,%rax+ addq %r12,%r13+ jmp .Loop++.align 32+.Loop:+ addq 0(%rsi),%r14+ adcq 8(%rsi),%rbx+ leaq 16(%rsi),%rsi+ adcq %rcx,%rbp+ mulq %r14+ movq %rax,%r9+ movq %r11,%rax+ movq %rdx,%r10++ mulq %r14+ movq %rax,%r14+ movq %r11,%rax+ movq %rdx,%r8++ mulq %rbx+ addq %rax,%r9+ movq %r13,%rax+ adcq %rdx,%r10++ mulq %rbx+ movq %rbp,%rbx+ addq %rax,%r14+ adcq %rdx,%r8++ imulq %r13,%rbx+ addq %rbx,%r9+ movq %r8,%rbx+ adcq $0,%r10++ imulq %r11,%rbp+ addq %r9,%rbx+ movq $-4,%rax+ adcq %rbp,%r10++ andq %r10,%rax+ movq %r10,%rbp+ shrq $2,%r10+ andq $3,%rbp+ addq %r10,%rax+ addq %rax,%r14+ adcq $0,%rbx+ adcq $0,%rbp+ movq %r12,%rax+ decq %r15+ jnz .Loop++ movq %r14,0(%rdi)+ movq %rbx,8(%rdi)+ movq %rbp,16(%rdi)++ movq 8(%rsp),%r15+.cfi_restore %r15+ movq 16(%rsp),%r14+.cfi_restore %r14+ movq 24(%rsp),%r13+.cfi_restore %r13+ movq 32(%rsp),%r12+.cfi_restore %r12+ movq 40(%rsp),%rbp+.cfi_restore %rbp+ movq 48(%rsp),%rbx+.cfi_restore %rbx+ leaq 56(%rsp),%rsp+.cfi_adjust_cfa_offset -56+.Lno_data:+.Lblocks_epilogue:+ .byte 0xf3,0xc3+.cfi_endproc +.size crypton_poly1305_asm_blocks,.-crypton_poly1305_asm_blocks++.type crypton_poly1305_asm_emit,@function+.align 32+crypton_poly1305_asm_emit:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa++ movl 0(%rdi),%eax+ movl 4(%rdi),%ecx+ movl 8(%rdi),%r8d+ movl 12(%rdi),%r11d+ movl 16(%rdi),%r10d++ shlq $26,%rcx+ movq %r8,%r9+ shlq $52,%r8+ addq %rcx,%rax+ shrq $12,%r9+ addq %rax,%r8+ adcq $0,%r9++ shlq $14,%r11+ movq %r10,%rax+ shrq $24,%r10+ addq %r11,%r9+ movq 0(%rdi),%rcx+ shlq $40,%rax+ movq 8(%rdi),%r11+ addq %rax,%r9+ movq 16(%rdi),%rax+ adcq $0,%r10++ cmpq $4,%rax++ cmovbeq %rcx,%r8+ cmovbeq %r11,%r9+ cmovbeq %rax,%r10++ movq %r8,%rax+ addq $5,%r8+ movq %r9,%rcx+ adcq $0,%r9+ adcq $0,%r10+ shrq $2,%r10+ cmovnzq %r8,%rax+ cmovnzq %r9,%rcx++ addq 0(%rdx),%rax+ adcq 8(%rdx),%rcx+ movq %rax,0(%rsi)+ movq %rcx,8(%rsi)++ .byte 0xf3,0xc3+.cfi_endproc+.size crypton_poly1305_asm_emit,.-crypton_poly1305_asm_emit+.type __crypton_poly1305_asm_block,@function+.align 32+__crypton_poly1305_asm_block:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa++ mulq %r14+ movq %rax,%r9+ movq %r11,%rax+ movq %rdx,%r10++ mulq %r14+ movq %rax,%r14+ movq %r11,%rax+ movq %rdx,%r8++ mulq %rbx+ addq %rax,%r9+ movq %r13,%rax+ adcq %rdx,%r10++ mulq %rbx+ movq %rbp,%rbx+ addq %rax,%r14+ adcq %rdx,%r8++ imulq %r13,%rbx+ addq %rbx,%r9+ movq %r8,%rbx+ adcq $0,%r10++ imulq %r11,%rbp+ addq %r9,%rbx+ movq $-4,%rax+ adcq %rbp,%r10++ andq %r10,%rax+ movq %r10,%rbp+ shrq $2,%r10+ andq $3,%rbp+ addq %r10,%rax+ addq %rax,%r14+ adcq $0,%rbx+ adcq $0,%rbp+ .byte 0xf3,0xc3+.cfi_endproc+.size __crypton_poly1305_asm_block,.-__crypton_poly1305_asm_block++.type __crypton_poly1305_asm_init_avx,@function+.align 32+__crypton_poly1305_asm_init_avx:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa++ cmpl $-1,48(%rdi)+ jne .Ldone_init_avx++ movq %r11,%r14+ movq %r12,%rbx+ xorq %rbp,%rbp++ leaq 48+64(%rdi),%rdi++ movq %r12,%rax+ call __crypton_poly1305_asm_block++ movl $0x3ffffff,%eax+ movl $0x3ffffff,%edx+ movq %r14,%r8+ andl %r14d,%eax+ movq %r11,%r9+ andl %r11d,%edx+ movl %eax,-64(%rdi)+ shrq $26,%r8+ movl %edx,-60(%rdi)+ shrq $26,%r9++ movl $0x3ffffff,%eax+ movl $0x3ffffff,%edx+ andl %r8d,%eax+ andl %r9d,%edx+ movl %eax,-48(%rdi)+ leal (%rax,%rax,4),%eax+ movl %edx,-44(%rdi)+ leal (%rdx,%rdx,4),%edx+ movl %eax,-32(%rdi)+ shrq $26,%r8+ movl %edx,-28(%rdi)+ shrq $26,%r9++ movq %rbx,%rax+ movq %r12,%rdx+ shlq $12,%rax+ shlq $12,%rdx+ orq %r8,%rax+ orq %r9,%rdx+ andl $0x3ffffff,%eax+ andl $0x3ffffff,%edx+ movl %eax,-16(%rdi)+ leal (%rax,%rax,4),%eax+ movl %edx,-12(%rdi)+ leal (%rdx,%rdx,4),%edx+ movl %eax,0(%rdi)+ movq %rbx,%r8+ movl %edx,4(%rdi)+ movq %r12,%r9++ movl $0x3ffffff,%eax+ movl $0x3ffffff,%edx+ shrq $14,%r8+ shrq $14,%r9+ andl %r8d,%eax+ andl %r9d,%edx+ movl %eax,16(%rdi)+ leal (%rax,%rax,4),%eax+ movl %edx,20(%rdi)+ leal (%rdx,%rdx,4),%edx+ movl %eax,32(%rdi)+ shrq $26,%r8+ movl %edx,36(%rdi)+ shrq $26,%r9++ movq %rbp,%rax+ shlq $24,%rax+ orq %rax,%r8+ movl %r8d,48(%rdi)+ leaq (%r8,%r8,4),%r8+ movl %r9d,52(%rdi)+ leaq (%r9,%r9,4),%r9+ movl %r8d,64(%rdi)+ movl %r9d,68(%rdi)++ movq %r12,%rax+ call __crypton_poly1305_asm_block++ movl $0x3ffffff,%eax+ movq %r14,%r8+ andl %r14d,%eax+ shrq $26,%r8+ movl %eax,-52(%rdi)++ movl $0x3ffffff,%edx+ andl %r8d,%edx+ movl %edx,-36(%rdi)+ leal (%rdx,%rdx,4),%edx+ shrq $26,%r8+ movl %edx,-20(%rdi)++ movq %rbx,%rax+ shlq $12,%rax+ orq %r8,%rax+ andl $0x3ffffff,%eax+ movl %eax,-4(%rdi)+ leal (%rax,%rax,4),%eax+ movq %rbx,%r8+ movl %eax,12(%rdi)++ movl $0x3ffffff,%edx+ shrq $14,%r8+ andl %r8d,%edx+ movl %edx,28(%rdi)+ leal (%rdx,%rdx,4),%edx+ shrq $26,%r8+ movl %edx,44(%rdi)++ movq %rbp,%rax+ shlq $24,%rax+ orq %rax,%r8+ movl %r8d,60(%rdi)+ leaq (%r8,%r8,4),%r8+ movl %r8d,76(%rdi)++ movq %r12,%rax+ call __crypton_poly1305_asm_block++ movl $0x3ffffff,%eax+ movq %r14,%r8+ andl %r14d,%eax+ shrq $26,%r8+ movl %eax,-56(%rdi)++ movl $0x3ffffff,%edx+ andl %r8d,%edx+ movl %edx,-40(%rdi)+ leal (%rdx,%rdx,4),%edx+ shrq $26,%r8+ movl %edx,-24(%rdi)++ movq %rbx,%rax+ shlq $12,%rax+ orq %r8,%rax+ andl $0x3ffffff,%eax+ movl %eax,-8(%rdi)+ leal (%rax,%rax,4),%eax+ movq %rbx,%r8+ movl %eax,8(%rdi)++ movl $0x3ffffff,%edx+ shrq $14,%r8+ andl %r8d,%edx+ movl %edx,24(%rdi)+ leal (%rdx,%rdx,4),%edx+ shrq $26,%r8+ movl %edx,40(%rdi)++ movq %rbp,%rax+ shlq $24,%rax+ orq %rax,%r8+ movl %r8d,56(%rdi)+ leaq (%r8,%r8,4),%r8+ movl %r8d,72(%rdi)++ leaq -48-64(%rdi),%rdi+.Ldone_init_avx:+ .byte 0xf3,0xc3+.cfi_endproc+.size __crypton_poly1305_asm_init_avx,.-__crypton_poly1305_asm_init_avx++.type crypton_poly1305_asm_blocks_avx,@function+.align 32+crypton_poly1305_asm_blocks_avx:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa+++ movl 20(%rdi),%r8d+ cmpq $128,%rdx+ jb .Lblocks++ andq $-16,%rdx++ vzeroupper++ testl %r8d,%r8d+ jz .Lbase2_64_avx++ testq $31,%rdx+ jz .Leven_avx++ pushq %rbx+.cfi_adjust_cfa_offset 8+.cfi_offset %rbx,-16+ pushq %rbp+.cfi_adjust_cfa_offset 8+.cfi_offset %rbp,-24+ pushq %r12+.cfi_adjust_cfa_offset 8+.cfi_offset %r12,-32+ pushq %r13+.cfi_adjust_cfa_offset 8+.cfi_offset %r13,-40+ pushq %r14+.cfi_adjust_cfa_offset 8+.cfi_offset %r14,-48+ pushq %r15+.cfi_adjust_cfa_offset 8+.cfi_offset %r15,-56+ leaq -8(%rsp),%rsp+.cfi_adjust_cfa_offset 8+.Lblocks_avx_body:++ movq %rdx,%r15++ movq 0(%rdi),%r8+ movq 8(%rdi),%r9+ movl 16(%rdi),%ebp++ movq 24(%rdi),%r11+ movq 32(%rdi),%r13+++ movl %r8d,%r14d+ andq $-2147483648,%r8+ movq %r9,%r12+ movl %r9d,%ebx+ andq $-2147483648,%r9++ shrq $6,%r8+ shlq $52,%r12+ addq %r8,%r14+ shrq $12,%rbx+ shrq $18,%r9+ addq %r12,%r14+ adcq %r9,%rbx++ movq %rbp,%r8+ shlq $40,%r8+ shrq $24,%rbp+ addq %r8,%rbx+ adcq $0,%rbp++ movq %r13,%r12+ movq %r13,%rax+ shrq $2,%r13+ addq %r12,%r13++ addq 0(%rsi),%r14+ adcq 8(%rsi),%rbx+ leaq 16(%rsi),%rsi+ adcq %rcx,%rbp++ call __crypton_poly1305_asm_block+++ movq %r14,%rax+ movq %r14,%rdx+ shrq $52,%r14+ movq %rbx,%r11+ movq %rbx,%r12+ shrq $26,%rdx+ andq $0x3ffffff,%rax+ shlq $12,%r11+ andq $0x3ffffff,%rdx+ shrq $14,%rbx+ orq %r11,%r14+ shlq $24,%rbp+ andq $0x3ffffff,%r14+ shrq $40,%r12+ andq $0x3ffffff,%rbx+ orq %r12,%rbp++ vmovd %eax,%xmm0+ vmovd %edx,%xmm1+ vmovd %r14d,%xmm2+ vmovd %ebx,%xmm3+ vmovd %ebp,%xmm4++ leaq -16(%r15),%rdx++ movq 8(%rsp),%r15+.cfi_restore %r15+ movq 16(%rsp),%r14+.cfi_restore %r14+ movq 24(%rsp),%r13+.cfi_restore %r13+ movq 32(%rsp),%r12+.cfi_restore %r12+ movq 40(%rsp),%rbp+.cfi_restore %rbp+ movq 48(%rsp),%rbx+.cfi_restore %rbx+ leaq 56(%rsp),%rax+ leaq 56(%rsp),%rsp+.cfi_adjust_cfa_offset -56+.Lblocks_avx_epilogue:+ jmp .Ldo_avx+.cfi_endproc ++.align 32+.Lbase2_64_avx:+.cfi_startproc + pushq %rbx+.cfi_adjust_cfa_offset 8+.cfi_offset %rbx,-16+ pushq %rbp+.cfi_adjust_cfa_offset 8+.cfi_offset %rbp,-24+ pushq %r12+.cfi_adjust_cfa_offset 8+.cfi_offset %r12,-32+ pushq %r13+.cfi_adjust_cfa_offset 8+.cfi_offset %r13,-40+ pushq %r14+.cfi_adjust_cfa_offset 8+.cfi_offset %r14,-48+ pushq %r15+.cfi_adjust_cfa_offset 8+.cfi_offset %r15,-56+ leaq -8(%rsp),%rsp+.cfi_adjust_cfa_offset 8+.Lbase2_64_avx_body:++ movq %rdx,%r15++ movq 24(%rdi),%r11+ movq 32(%rdi),%r13++ movq 0(%rdi),%r14+ movq 8(%rdi),%rbx+ movl 16(%rdi),%ebp++ movq %r13,%r12+ movq %r13,%rax+ shrq $2,%r13+ addq %r12,%r13++ testq $31,%rdx+ jz .Linit_avx++ addq 0(%rsi),%r14+ adcq 8(%rsi),%rbx+ leaq 16(%rsi),%rsi+ adcq %rcx,%rbp+ subq $16,%r15++ call __crypton_poly1305_asm_block++.Linit_avx:++ movq %r14,%rax+ movq %r14,%rdx+ shrq $52,%r14+ movq %rbx,%r8+ movq %rbx,%r9+ shrq $26,%rdx+ andq $0x3ffffff,%rax+ shlq $12,%r8+ andq $0x3ffffff,%rdx+ shrq $14,%rbx+ orq %r8,%r14+ shlq $24,%rbp+ andq $0x3ffffff,%r14+ shrq $40,%r9+ andq $0x3ffffff,%rbx+ orq %r9,%rbp++ vmovd %eax,%xmm0+ vmovd %edx,%xmm1+ vmovd %r14d,%xmm2+ vmovd %ebx,%xmm3+ vmovd %ebp,%xmm4+ movl $1,20(%rdi)++ call __crypton_poly1305_asm_init_avx++ movq %r15,%rdx++ movq 8(%rsp),%r15+.cfi_restore %r15+ movq 16(%rsp),%r14+.cfi_restore %r14+ movq 24(%rsp),%r13+.cfi_restore %r13+ movq 32(%rsp),%r12+.cfi_restore %r12+ movq 40(%rsp),%rbp+.cfi_restore %rbp+ movq 48(%rsp),%rbx+.cfi_restore %rbx+ leaq 56(%rsp),%rax+ leaq 56(%rsp),%rsp+.cfi_adjust_cfa_offset -56+.Lbase2_64_avx_epilogue:+ jmp .Ldo_avx+.cfi_endproc ++.align 32+.Leven_avx:+.cfi_startproc + vmovd 0(%rdi),%xmm0+ vmovd 4(%rdi),%xmm1+ vmovd 8(%rdi),%xmm2+ vmovd 12(%rdi),%xmm3+ vmovd 16(%rdi),%xmm4++.Ldo_avx:+ leaq -88(%rsp),%r11+.cfi_def_cfa %r11,0x60+ subq $0x178,%rsp+ subq $64,%rdx+ leaq -32(%rsi),%rax+ cmovcq %rax,%rsi++ vmovdqu 48(%rdi),%xmm14+ leaq 112(%rdi),%rdi+ leaq .Lconst(%rip),%rcx++++ vmovdqu 32(%rsi),%xmm5+ vmovdqu 48(%rsi),%xmm6+ vmovdqa 64(%rcx),%xmm15++ vpsrldq $6,%xmm5,%xmm7+ vpsrldq $6,%xmm6,%xmm8+ vpunpckhqdq %xmm6,%xmm5,%xmm9+ vpunpcklqdq %xmm6,%xmm5,%xmm5+ vpunpcklqdq %xmm8,%xmm7,%xmm8++ vpsrlq $40,%xmm9,%xmm9+ vpsrlq $26,%xmm5,%xmm6+ vpand %xmm15,%xmm5,%xmm5+ vpsrlq $4,%xmm8,%xmm7+ vpand %xmm15,%xmm6,%xmm6+ vpsrlq $30,%xmm8,%xmm8+ vpand %xmm15,%xmm7,%xmm7+ vpand %xmm15,%xmm8,%xmm8+ vpor 32(%rcx),%xmm9,%xmm9++ jbe .Lskip_loop_avx+++ vmovdqu -48(%rdi),%xmm11+ vmovdqu -32(%rdi),%xmm12+ vpshufd $0xEE,%xmm14,%xmm13+ vpshufd $0x44,%xmm14,%xmm10+ vmovdqa %xmm13,-144(%r11)+ vmovdqa %xmm10,0(%rsp)+ vpshufd $0xEE,%xmm11,%xmm14+ vmovdqu -16(%rdi),%xmm10+ vpshufd $0x44,%xmm11,%xmm11+ vmovdqa %xmm14,-128(%r11)+ vmovdqa %xmm11,16(%rsp)+ vpshufd $0xEE,%xmm12,%xmm13+ vmovdqu 0(%rdi),%xmm11+ vpshufd $0x44,%xmm12,%xmm12+ vmovdqa %xmm13,-112(%r11)+ vmovdqa %xmm12,32(%rsp)+ vpshufd $0xEE,%xmm10,%xmm14+ vmovdqu 16(%rdi),%xmm12+ vpshufd $0x44,%xmm10,%xmm10+ vmovdqa %xmm14,-96(%r11)+ vmovdqa %xmm10,48(%rsp)+ vpshufd $0xEE,%xmm11,%xmm13+ vmovdqu 32(%rdi),%xmm10+ vpshufd $0x44,%xmm11,%xmm11+ vmovdqa %xmm13,-80(%r11)+ vmovdqa %xmm11,64(%rsp)+ vpshufd $0xEE,%xmm12,%xmm14+ vmovdqu 48(%rdi),%xmm11+ vpshufd $0x44,%xmm12,%xmm12+ vmovdqa %xmm14,-64(%r11)+ vmovdqa %xmm12,80(%rsp)+ vpshufd $0xEE,%xmm10,%xmm13+ vmovdqu 64(%rdi),%xmm12+ vpshufd $0x44,%xmm10,%xmm10+ vmovdqa %xmm13,-48(%r11)+ vmovdqa %xmm10,96(%rsp)+ vpshufd $0xEE,%xmm11,%xmm14+ vpshufd $0x44,%xmm11,%xmm11+ vmovdqa %xmm14,-32(%r11)+ vmovdqa %xmm11,112(%rsp)+ vpshufd $0xEE,%xmm12,%xmm13+ vmovdqa 0(%rsp),%xmm14+ vpshufd $0x44,%xmm12,%xmm12+ vmovdqa %xmm13,-16(%r11)+ vmovdqa %xmm12,128(%rsp)++ jmp .Loop_avx++.align 32+.Loop_avx:+++++++++++++++++++++ vpmuludq %xmm5,%xmm14,%xmm10+ vpmuludq %xmm6,%xmm14,%xmm11+ vmovdqa %xmm2,32(%r11)+ vpmuludq %xmm7,%xmm14,%xmm12+ vmovdqa 16(%rsp),%xmm2+ vpmuludq %xmm8,%xmm14,%xmm13+ vpmuludq %xmm9,%xmm14,%xmm14++ vmovdqa %xmm0,0(%r11)+ vpmuludq 32(%rsp),%xmm9,%xmm0+ vmovdqa %xmm1,16(%r11)+ vpmuludq %xmm8,%xmm2,%xmm1+ vpaddq %xmm0,%xmm10,%xmm10+ vpaddq %xmm1,%xmm14,%xmm14+ vmovdqa %xmm3,48(%r11)+ vpmuludq %xmm7,%xmm2,%xmm0+ vpmuludq %xmm6,%xmm2,%xmm1+ vpaddq %xmm0,%xmm13,%xmm13+ vmovdqa 48(%rsp),%xmm3+ vpaddq %xmm1,%xmm12,%xmm12+ vmovdqa %xmm4,64(%r11)+ vpmuludq %xmm5,%xmm2,%xmm2+ vpmuludq %xmm7,%xmm3,%xmm0+ vpaddq %xmm2,%xmm11,%xmm11++ vmovdqa 64(%rsp),%xmm4+ vpaddq %xmm0,%xmm14,%xmm14+ vpmuludq %xmm6,%xmm3,%xmm1+ vpmuludq %xmm5,%xmm3,%xmm3+ vpaddq %xmm1,%xmm13,%xmm13+ vmovdqa 80(%rsp),%xmm2+ vpaddq %xmm3,%xmm12,%xmm12+ vpmuludq %xmm9,%xmm4,%xmm0+ vpmuludq %xmm8,%xmm4,%xmm4+ vpaddq %xmm0,%xmm11,%xmm11+ vmovdqa 96(%rsp),%xmm3+ vpaddq %xmm4,%xmm10,%xmm10++ vmovdqa 128(%rsp),%xmm4+ vpmuludq %xmm6,%xmm2,%xmm1+ vpmuludq %xmm5,%xmm2,%xmm2+ vpaddq %xmm1,%xmm14,%xmm14+ vpaddq %xmm2,%xmm13,%xmm13+ vpmuludq %xmm9,%xmm3,%xmm0+ vpmuludq %xmm8,%xmm3,%xmm1+ vpaddq %xmm0,%xmm12,%xmm12+ vmovdqu 0(%rsi),%xmm0+ vpaddq %xmm1,%xmm11,%xmm11+ vpmuludq %xmm7,%xmm3,%xmm3+ vpmuludq %xmm7,%xmm4,%xmm7+ vpaddq %xmm3,%xmm10,%xmm10++ vmovdqu 16(%rsi),%xmm1+ vpaddq %xmm7,%xmm11,%xmm11+ vpmuludq %xmm8,%xmm4,%xmm8+ vpmuludq %xmm9,%xmm4,%xmm9+ vpsrldq $6,%xmm0,%xmm2+ vpaddq %xmm8,%xmm12,%xmm12+ vpaddq %xmm9,%xmm13,%xmm13+ vpsrldq $6,%xmm1,%xmm3+ vpmuludq 112(%rsp),%xmm5,%xmm9+ vpmuludq %xmm6,%xmm4,%xmm5+ vpunpckhqdq %xmm1,%xmm0,%xmm4+ vpaddq %xmm9,%xmm14,%xmm14+ vmovdqa -144(%r11),%xmm9+ vpaddq %xmm5,%xmm10,%xmm10++ vpunpcklqdq %xmm1,%xmm0,%xmm0+ vpunpcklqdq %xmm3,%xmm2,%xmm3+++ vpsrldq $5,%xmm4,%xmm4+ vpsrlq $26,%xmm0,%xmm1+ vpand %xmm15,%xmm0,%xmm0+ vpsrlq $4,%xmm3,%xmm2+ vpand %xmm15,%xmm1,%xmm1+ vpand 0(%rcx),%xmm4,%xmm4+ vpsrlq $30,%xmm3,%xmm3+ vpand %xmm15,%xmm2,%xmm2+ vpand %xmm15,%xmm3,%xmm3+ vpor 32(%rcx),%xmm4,%xmm4++ vpaddq 0(%r11),%xmm0,%xmm0+ vpaddq 16(%r11),%xmm1,%xmm1+ vpaddq 32(%r11),%xmm2,%xmm2+ vpaddq 48(%r11),%xmm3,%xmm3+ vpaddq 64(%r11),%xmm4,%xmm4++ leaq 32(%rsi),%rax+ leaq 64(%rsi),%rsi+ subq $64,%rdx+ cmovcq %rax,%rsi+++++++++++ vpmuludq %xmm0,%xmm9,%xmm5+ vpmuludq %xmm1,%xmm9,%xmm6+ vpaddq %xmm5,%xmm10,%xmm10+ vpaddq %xmm6,%xmm11,%xmm11+ vmovdqa -128(%r11),%xmm7+ vpmuludq %xmm2,%xmm9,%xmm5+ vpmuludq %xmm3,%xmm9,%xmm6+ vpaddq %xmm5,%xmm12,%xmm12+ vpaddq %xmm6,%xmm13,%xmm13+ vpmuludq %xmm4,%xmm9,%xmm9+ vpmuludq -112(%r11),%xmm4,%xmm5+ vpaddq %xmm9,%xmm14,%xmm14++ vpaddq %xmm5,%xmm10,%xmm10+ vpmuludq %xmm2,%xmm7,%xmm6+ vpmuludq %xmm3,%xmm7,%xmm5+ vpaddq %xmm6,%xmm13,%xmm13+ vmovdqa -96(%r11),%xmm8+ vpaddq %xmm5,%xmm14,%xmm14+ vpmuludq %xmm1,%xmm7,%xmm6+ vpmuludq %xmm0,%xmm7,%xmm7+ vpaddq %xmm6,%xmm12,%xmm12+ vpaddq %xmm7,%xmm11,%xmm11++ vmovdqa -80(%r11),%xmm9+ vpmuludq %xmm2,%xmm8,%xmm5+ vpmuludq %xmm1,%xmm8,%xmm6+ vpaddq %xmm5,%xmm14,%xmm14+ vpaddq %xmm6,%xmm13,%xmm13+ vmovdqa -64(%r11),%xmm7+ vpmuludq %xmm0,%xmm8,%xmm8+ vpmuludq %xmm4,%xmm9,%xmm5+ vpaddq %xmm8,%xmm12,%xmm12+ vpaddq %xmm5,%xmm11,%xmm11+ vmovdqa -48(%r11),%xmm8+ vpmuludq %xmm3,%xmm9,%xmm9+ vpmuludq %xmm1,%xmm7,%xmm6+ vpaddq %xmm9,%xmm10,%xmm10++ vmovdqa -16(%r11),%xmm9+ vpaddq %xmm6,%xmm14,%xmm14+ vpmuludq %xmm0,%xmm7,%xmm7+ vpmuludq %xmm4,%xmm8,%xmm5+ vpaddq %xmm7,%xmm13,%xmm13+ vpaddq %xmm5,%xmm12,%xmm12+ vmovdqu 32(%rsi),%xmm5+ vpmuludq %xmm3,%xmm8,%xmm7+ vpmuludq %xmm2,%xmm8,%xmm8+ vpaddq %xmm7,%xmm11,%xmm11+ vmovdqu 48(%rsi),%xmm6+ vpaddq %xmm8,%xmm10,%xmm10++ vpmuludq %xmm2,%xmm9,%xmm2+ vpmuludq %xmm3,%xmm9,%xmm3+ vpsrldq $6,%xmm5,%xmm7+ vpaddq %xmm2,%xmm11,%xmm11+ vpmuludq %xmm4,%xmm9,%xmm4+ vpsrldq $6,%xmm6,%xmm8+ vpaddq %xmm3,%xmm12,%xmm2+ vpaddq %xmm4,%xmm13,%xmm3+ vpmuludq -32(%r11),%xmm0,%xmm4+ vpmuludq %xmm1,%xmm9,%xmm0+ vpunpckhqdq %xmm6,%xmm5,%xmm9+ vpaddq %xmm4,%xmm14,%xmm4+ vpaddq %xmm0,%xmm10,%xmm0++ vpunpcklqdq %xmm6,%xmm5,%xmm5+ vpunpcklqdq %xmm8,%xmm7,%xmm8+++ vpsrldq $5,%xmm9,%xmm9+ vpsrlq $26,%xmm5,%xmm6+ vmovdqa 0(%rsp),%xmm14+ vpand %xmm15,%xmm5,%xmm5+ vpsrlq $4,%xmm8,%xmm7+ vpand %xmm15,%xmm6,%xmm6+ vpand 0(%rcx),%xmm9,%xmm9+ vpsrlq $30,%xmm8,%xmm8+ vpand %xmm15,%xmm7,%xmm7+ vpand %xmm15,%xmm8,%xmm8+ vpor 32(%rcx),%xmm9,%xmm9++++++ vpsrlq $26,%xmm3,%xmm13+ vpand %xmm15,%xmm3,%xmm3+ vpaddq %xmm13,%xmm4,%xmm4++ vpsrlq $26,%xmm0,%xmm10+ vpand %xmm15,%xmm0,%xmm0+ vpaddq %xmm10,%xmm11,%xmm1++ vpsrlq $26,%xmm4,%xmm10+ vpand %xmm15,%xmm4,%xmm4++ vpsrlq $26,%xmm1,%xmm11+ vpand %xmm15,%xmm1,%xmm1+ vpaddq %xmm11,%xmm2,%xmm2++ vpaddq %xmm10,%xmm0,%xmm0+ vpsllq $2,%xmm10,%xmm10+ vpaddq %xmm10,%xmm0,%xmm0++ vpsrlq $26,%xmm2,%xmm12+ vpand %xmm15,%xmm2,%xmm2+ vpaddq %xmm12,%xmm3,%xmm3++ vpsrlq $26,%xmm0,%xmm10+ vpand %xmm15,%xmm0,%xmm0+ vpaddq %xmm10,%xmm1,%xmm1++ vpsrlq $26,%xmm3,%xmm13+ vpand %xmm15,%xmm3,%xmm3+ vpaddq %xmm13,%xmm4,%xmm4++ ja .Loop_avx++.Lskip_loop_avx:++++ vpshufd $0x10,%xmm14,%xmm14+ addq $32,%rdx+ jnz .Long_tail_avx++ vpaddq %xmm2,%xmm7,%xmm7+ vpaddq %xmm0,%xmm5,%xmm5+ vpaddq %xmm1,%xmm6,%xmm6+ vpaddq %xmm3,%xmm8,%xmm8+ vpaddq %xmm4,%xmm9,%xmm9++.Long_tail_avx:+ vmovdqa %xmm2,32(%r11)+ vmovdqa %xmm0,0(%r11)+ vmovdqa %xmm1,16(%r11)+ vmovdqa %xmm3,48(%r11)+ vmovdqa %xmm4,64(%r11)++++++++ vpmuludq %xmm7,%xmm14,%xmm12+ vpmuludq %xmm5,%xmm14,%xmm10+ vpshufd $0x10,-48(%rdi),%xmm2+ vpmuludq %xmm6,%xmm14,%xmm11+ vpmuludq %xmm8,%xmm14,%xmm13+ vpmuludq %xmm9,%xmm14,%xmm14++ vpmuludq %xmm8,%xmm2,%xmm0+ vpaddq %xmm0,%xmm14,%xmm14+ vpshufd $0x10,-32(%rdi),%xmm3+ vpmuludq %xmm7,%xmm2,%xmm1+ vpaddq %xmm1,%xmm13,%xmm13+ vpshufd $0x10,-16(%rdi),%xmm4+ vpmuludq %xmm6,%xmm2,%xmm0+ vpaddq %xmm0,%xmm12,%xmm12+ vpmuludq %xmm5,%xmm2,%xmm2+ vpaddq %xmm2,%xmm11,%xmm11+ vpmuludq %xmm9,%xmm3,%xmm3+ vpaddq %xmm3,%xmm10,%xmm10++ vpshufd $0x10,0(%rdi),%xmm2+ vpmuludq %xmm7,%xmm4,%xmm1+ vpaddq %xmm1,%xmm14,%xmm14+ vpmuludq %xmm6,%xmm4,%xmm0+ vpaddq %xmm0,%xmm13,%xmm13+ vpshufd $0x10,16(%rdi),%xmm3+ vpmuludq %xmm5,%xmm4,%xmm4+ vpaddq %xmm4,%xmm12,%xmm12+ vpmuludq %xmm9,%xmm2,%xmm1+ vpaddq %xmm1,%xmm11,%xmm11+ vpshufd $0x10,32(%rdi),%xmm4+ vpmuludq %xmm8,%xmm2,%xmm2+ vpaddq %xmm2,%xmm10,%xmm10++ vpmuludq %xmm6,%xmm3,%xmm0+ vpaddq %xmm0,%xmm14,%xmm14+ vpmuludq %xmm5,%xmm3,%xmm3+ vpaddq %xmm3,%xmm13,%xmm13+ vpshufd $0x10,48(%rdi),%xmm2+ vpmuludq %xmm9,%xmm4,%xmm1+ vpaddq %xmm1,%xmm12,%xmm12+ vpshufd $0x10,64(%rdi),%xmm3+ vpmuludq %xmm8,%xmm4,%xmm0+ vpaddq %xmm0,%xmm11,%xmm11+ vpmuludq %xmm7,%xmm4,%xmm4+ vpaddq %xmm4,%xmm10,%xmm10++ vpmuludq %xmm5,%xmm2,%xmm2+ vpaddq %xmm2,%xmm14,%xmm14+ vpmuludq %xmm9,%xmm3,%xmm1+ vpaddq %xmm1,%xmm13,%xmm13+ vpmuludq %xmm8,%xmm3,%xmm0+ vpaddq %xmm0,%xmm12,%xmm12+ vpmuludq %xmm7,%xmm3,%xmm1+ vpaddq %xmm1,%xmm11,%xmm11+ vpmuludq %xmm6,%xmm3,%xmm3+ vpaddq %xmm3,%xmm10,%xmm10++ jz .Lshort_tail_avx++ vmovdqu 0(%rsi),%xmm0+ vmovdqu 16(%rsi),%xmm1++ vpsrldq $6,%xmm0,%xmm2+ vpsrldq $6,%xmm1,%xmm3+ vpunpckhqdq %xmm1,%xmm0,%xmm4+ vpunpcklqdq %xmm1,%xmm0,%xmm0+ vpunpcklqdq %xmm3,%xmm2,%xmm3++ vpsrlq $40,%xmm4,%xmm4+ vpsrlq $26,%xmm0,%xmm1+ vpand %xmm15,%xmm0,%xmm0+ vpsrlq $4,%xmm3,%xmm2+ vpand %xmm15,%xmm1,%xmm1+ vpsrlq $30,%xmm3,%xmm3+ vpand %xmm15,%xmm2,%xmm2+ vpand %xmm15,%xmm3,%xmm3+ vpor 32(%rcx),%xmm4,%xmm4++ vpshufd $0x32,-64(%rdi),%xmm9+ vpaddq 0(%r11),%xmm0,%xmm0+ vpaddq 16(%r11),%xmm1,%xmm1+ vpaddq 32(%r11),%xmm2,%xmm2+ vpaddq 48(%r11),%xmm3,%xmm3+ vpaddq 64(%r11),%xmm4,%xmm4+++++ vpmuludq %xmm0,%xmm9,%xmm5+ vpaddq %xmm5,%xmm10,%xmm10+ vpmuludq %xmm1,%xmm9,%xmm6+ vpaddq %xmm6,%xmm11,%xmm11+ vpmuludq %xmm2,%xmm9,%xmm5+ vpaddq %xmm5,%xmm12,%xmm12+ vpshufd $0x32,-48(%rdi),%xmm7+ vpmuludq %xmm3,%xmm9,%xmm6+ vpaddq %xmm6,%xmm13,%xmm13+ vpmuludq %xmm4,%xmm9,%xmm9+ vpaddq %xmm9,%xmm14,%xmm14++ vpmuludq %xmm3,%xmm7,%xmm5+ vpaddq %xmm5,%xmm14,%xmm14+ vpshufd $0x32,-32(%rdi),%xmm8+ vpmuludq %xmm2,%xmm7,%xmm6+ vpaddq %xmm6,%xmm13,%xmm13+ vpshufd $0x32,-16(%rdi),%xmm9+ vpmuludq %xmm1,%xmm7,%xmm5+ vpaddq %xmm5,%xmm12,%xmm12+ vpmuludq %xmm0,%xmm7,%xmm7+ vpaddq %xmm7,%xmm11,%xmm11+ vpmuludq %xmm4,%xmm8,%xmm8+ vpaddq %xmm8,%xmm10,%xmm10++ vpshufd $0x32,0(%rdi),%xmm7+ vpmuludq %xmm2,%xmm9,%xmm6+ vpaddq %xmm6,%xmm14,%xmm14+ vpmuludq %xmm1,%xmm9,%xmm5+ vpaddq %xmm5,%xmm13,%xmm13+ vpshufd $0x32,16(%rdi),%xmm8+ vpmuludq %xmm0,%xmm9,%xmm9+ vpaddq %xmm9,%xmm12,%xmm12+ vpmuludq %xmm4,%xmm7,%xmm6+ vpaddq %xmm6,%xmm11,%xmm11+ vpshufd $0x32,32(%rdi),%xmm9+ vpmuludq %xmm3,%xmm7,%xmm7+ vpaddq %xmm7,%xmm10,%xmm10++ vpmuludq %xmm1,%xmm8,%xmm5+ vpaddq %xmm5,%xmm14,%xmm14+ vpmuludq %xmm0,%xmm8,%xmm8+ vpaddq %xmm8,%xmm13,%xmm13+ vpshufd $0x32,48(%rdi),%xmm7+ vpmuludq %xmm4,%xmm9,%xmm6+ vpaddq %xmm6,%xmm12,%xmm12+ vpshufd $0x32,64(%rdi),%xmm8+ vpmuludq %xmm3,%xmm9,%xmm5+ vpaddq %xmm5,%xmm11,%xmm11+ vpmuludq %xmm2,%xmm9,%xmm9+ vpaddq %xmm9,%xmm10,%xmm10++ vpmuludq %xmm0,%xmm7,%xmm7+ vpaddq %xmm7,%xmm14,%xmm14+ vpmuludq %xmm4,%xmm8,%xmm6+ vpaddq %xmm6,%xmm13,%xmm13+ vpmuludq %xmm3,%xmm8,%xmm5+ vpaddq %xmm5,%xmm12,%xmm12+ vpmuludq %xmm2,%xmm8,%xmm6+ vpaddq %xmm6,%xmm11,%xmm11+ vpmuludq %xmm1,%xmm8,%xmm8+ vpaddq %xmm8,%xmm10,%xmm10++.Lshort_tail_avx:++++ vpsrldq $8,%xmm14,%xmm9+ vpsrldq $8,%xmm13,%xmm8+ vpsrldq $8,%xmm11,%xmm6+ vpsrldq $8,%xmm10,%xmm5+ vpsrldq $8,%xmm12,%xmm7+ vpaddq %xmm8,%xmm13,%xmm13+ vpaddq %xmm9,%xmm14,%xmm14+ vpaddq %xmm5,%xmm10,%xmm10+ vpaddq %xmm6,%xmm11,%xmm11+ vpaddq %xmm7,%xmm12,%xmm12+++++ vpsrlq $26,%xmm13,%xmm3+ vpand %xmm15,%xmm13,%xmm13+ vpaddq %xmm3,%xmm14,%xmm14++ vpsrlq $26,%xmm10,%xmm0+ vpand %xmm15,%xmm10,%xmm10+ vpaddq %xmm0,%xmm11,%xmm11++ vpsrlq $26,%xmm14,%xmm4+ vpand %xmm15,%xmm14,%xmm14++ vpsrlq $26,%xmm11,%xmm1+ vpand %xmm15,%xmm11,%xmm11+ vpaddq %xmm1,%xmm12,%xmm12++ vpaddq %xmm4,%xmm10,%xmm10+ vpsllq $2,%xmm4,%xmm4+ vpaddq %xmm4,%xmm10,%xmm10++ vpsrlq $26,%xmm12,%xmm2+ vpand %xmm15,%xmm12,%xmm12+ vpaddq %xmm2,%xmm13,%xmm13++ vpsrlq $26,%xmm10,%xmm0+ vpand %xmm15,%xmm10,%xmm10+ vpaddq %xmm0,%xmm11,%xmm11++ vpsrlq $26,%xmm13,%xmm3+ vpand %xmm15,%xmm13,%xmm13+ vpaddq %xmm3,%xmm14,%xmm14++ vmovd %xmm10,-112(%rdi)+ vmovd %xmm11,-108(%rdi)+ vmovd %xmm12,-104(%rdi)+ vmovd %xmm13,-100(%rdi)+ vmovd %xmm14,-96(%rdi)+ leaq 88(%r11),%rsp+.cfi_def_cfa %rsp,8+ vzeroupper+ .byte 0xf3,0xc3+.cfi_endproc +.size crypton_poly1305_asm_blocks_avx,.-crypton_poly1305_asm_blocks_avx+.type crypton_poly1305_asm_blocks_avx2,@function+.align 32+crypton_poly1305_asm_blocks_avx2:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa+++ movl 20(%rdi),%r8d+ cmpq $128,%rdx+ jb .Lblocks++ andq $-16,%rdx++ vzeroupper++ testl %r8d,%r8d+ jz .Lbase2_64_avx2++ testq $63,%rdx+ jz .Leven_avx2++ pushq %rbx+.cfi_adjust_cfa_offset 8+.cfi_offset %rbx,-16+ pushq %rbp+.cfi_adjust_cfa_offset 8+.cfi_offset %rbp,-24+ pushq %r12+.cfi_adjust_cfa_offset 8+.cfi_offset %r12,-32+ pushq %r13+.cfi_adjust_cfa_offset 8+.cfi_offset %r13,-40+ pushq %r14+.cfi_adjust_cfa_offset 8+.cfi_offset %r14,-48+ pushq %r15+.cfi_adjust_cfa_offset 8+.cfi_offset %r15,-56+ leaq -8(%rsp),%rsp+.cfi_adjust_cfa_offset 8+.Lblocks_avx2_body:++ movq %rdx,%r15++ movq 0(%rdi),%r8+ movq 8(%rdi),%r9+ movl 16(%rdi),%ebp++ movq 24(%rdi),%r11+ movq 32(%rdi),%r13+++ movl %r8d,%r14d+ andq $-2147483648,%r8+ movq %r9,%r12+ movl %r9d,%ebx+ andq $-2147483648,%r9++ shrq $6,%r8+ shlq $52,%r12+ addq %r8,%r14+ shrq $12,%rbx+ shrq $18,%r9+ addq %r12,%r14+ adcq %r9,%rbx++ movq %rbp,%r8+ shlq $40,%r8+ shrq $24,%rbp+ addq %r8,%rbx+ adcq $0,%rbp++ movq %r13,%r12+ movq %r13,%rax+ shrq $2,%r13+ addq %r12,%r13++.Lbase2_26_pre_avx2:+ addq 0(%rsi),%r14+ adcq 8(%rsi),%rbx+ leaq 16(%rsi),%rsi+ adcq %rcx,%rbp+ subq $16,%r15++ call __crypton_poly1305_asm_block+ movq %r12,%rax++ testq $63,%r15+ jnz .Lbase2_26_pre_avx2+++ movq %r14,%rax+ movq %r14,%rdx+ shrq $52,%r14+ movq %rbx,%r11+ movq %rbx,%r12+ shrq $26,%rdx+ andq $0x3ffffff,%rax+ shlq $12,%r11+ andq $0x3ffffff,%rdx+ shrq $14,%rbx+ orq %r11,%r14+ shlq $24,%rbp+ andq $0x3ffffff,%r14+ shrq $40,%r12+ andq $0x3ffffff,%rbx+ orq %r12,%rbp++ vmovd %eax,%xmm0+ vmovd %edx,%xmm1+ vmovd %r14d,%xmm2+ vmovd %ebx,%xmm3+ vmovd %ebp,%xmm4++ movq %r15,%rdx++ movq 8(%rsp),%r15+.cfi_restore %r15+ movq 16(%rsp),%r14+.cfi_restore %r14+ movq 24(%rsp),%r13+.cfi_restore %r13+ movq 32(%rsp),%r12+.cfi_restore %r12+ movq 40(%rsp),%rbp+.cfi_restore %rbp+ movq 48(%rsp),%rbx+.cfi_restore %rbx+ leaq 56(%rsp),%rax+ leaq 56(%rsp),%rsp+.cfi_adjust_cfa_offset -56+.Lblocks_avx2_epilogue:+ jmp .Ldo_avx2+.cfi_endproc ++.align 32+.Lbase2_64_avx2:+.cfi_startproc + pushq %rbx+.cfi_adjust_cfa_offset 8+.cfi_offset %rbx,-16+ pushq %rbp+.cfi_adjust_cfa_offset 8+.cfi_offset %rbp,-24+ pushq %r12+.cfi_adjust_cfa_offset 8+.cfi_offset %r12,-32+ pushq %r13+.cfi_adjust_cfa_offset 8+.cfi_offset %r13,-40+ pushq %r14+.cfi_adjust_cfa_offset 8+.cfi_offset %r14,-48+ pushq %r15+.cfi_adjust_cfa_offset 8+.cfi_offset %r15,-56+ leaq -8(%rsp),%rsp+.cfi_adjust_cfa_offset 8+.Lbase2_64_avx2_body:++ movq %rdx,%r15++ movq 24(%rdi),%r11+ movq 32(%rdi),%r13++ movq 0(%rdi),%r14+ movq 8(%rdi),%rbx+ movl 16(%rdi),%ebp++ movq %r13,%r12+ movq %r13,%rax+ shrq $2,%r13+ addq %r12,%r13++ testq $63,%rdx+ jz .Linit_avx2++.Lbase2_64_pre_avx2:+ addq 0(%rsi),%r14+ adcq 8(%rsi),%rbx+ leaq 16(%rsi),%rsi+ adcq %rcx,%rbp+ subq $16,%r15++ call __crypton_poly1305_asm_block+ movq %r12,%rax++ testq $63,%r15+ jnz .Lbase2_64_pre_avx2++.Linit_avx2:++ movq %r14,%rax+ movq %r14,%rdx+ shrq $52,%r14+ movq %rbx,%r8+ movq %rbx,%r9+ shrq $26,%rdx+ andq $0x3ffffff,%rax+ shlq $12,%r8+ andq $0x3ffffff,%rdx+ shrq $14,%rbx+ orq %r8,%r14+ shlq $24,%rbp+ andq $0x3ffffff,%r14+ shrq $40,%r9+ andq $0x3ffffff,%rbx+ orq %r9,%rbp++ vmovd %eax,%xmm0+ vmovd %edx,%xmm1+ vmovd %r14d,%xmm2+ vmovd %ebx,%xmm3+ vmovd %ebp,%xmm4+ movl $1,20(%rdi)++ call __crypton_poly1305_asm_init_avx++ movq %r15,%rdx++ movq 8(%rsp),%r15+.cfi_restore %r15+ movq 16(%rsp),%r14+.cfi_restore %r14+ movq 24(%rsp),%r13+.cfi_restore %r13+ movq 32(%rsp),%r12+.cfi_restore %r12+ movq 40(%rsp),%rbp+.cfi_restore %rbp+ movq 48(%rsp),%rbx+.cfi_restore %rbx+ leaq 56(%rsp),%rax+ leaq 56(%rsp),%rsp+.cfi_adjust_cfa_offset -56+.Lbase2_64_avx2_epilogue:+ jmp .Ldo_avx2+.cfi_endproc ++.align 32+.Leven_avx2:+.cfi_startproc + vmovd 0(%rdi),%xmm0+ vmovd 4(%rdi),%xmm1+ vmovd 8(%rdi),%xmm2+ vmovd 12(%rdi),%xmm3+ vmovd 16(%rdi),%xmm4++.Ldo_avx2:+ leaq -8(%rsp),%r11+.cfi_def_cfa %r11,16+ subq $0x128,%rsp+ leaq .Lconst(%rip),%rcx+ leaq 48+64(%rdi),%rdi+ vmovdqa 96(%rcx),%ymm7+++ vmovdqu -64(%rdi),%xmm9+ andq $-512,%rsp+ vmovdqu -48(%rdi),%xmm10+ vmovdqu -32(%rdi),%xmm6+ vmovdqu -16(%rdi),%xmm11+ vmovdqu 0(%rdi),%xmm12+ vmovdqu 16(%rdi),%xmm13+ leaq 144(%rsp),%rax+ vmovdqu 32(%rdi),%xmm14+ vpermd %ymm9,%ymm7,%ymm9+ vmovdqu 48(%rdi),%xmm15+ vpermd %ymm10,%ymm7,%ymm10+ vmovdqu 64(%rdi),%xmm5+ vpermd %ymm6,%ymm7,%ymm6+ vmovdqa %ymm9,0(%rsp)+ vpermd %ymm11,%ymm7,%ymm11+ vmovdqa %ymm10,32-144(%rax)+ vpermd %ymm12,%ymm7,%ymm12+ vmovdqa %ymm6,64-144(%rax)+ vpermd %ymm13,%ymm7,%ymm13+ vmovdqa %ymm11,96-144(%rax)+ vpermd %ymm14,%ymm7,%ymm14+ vmovdqa %ymm12,128-144(%rax)+ vpermd %ymm15,%ymm7,%ymm15+ vmovdqa %ymm13,160-144(%rax)+ vpermd %ymm5,%ymm7,%ymm5+ vmovdqa %ymm14,192-144(%rax)+ vmovdqa %ymm15,224-144(%rax)+ vmovdqa %ymm5,256-144(%rax)+ vmovdqa 64(%rcx),%ymm5++++ vmovdqu 0(%rsi),%xmm7+ vmovdqu 16(%rsi),%xmm8+ vinserti128 $1,32(%rsi),%ymm7,%ymm7+ vinserti128 $1,48(%rsi),%ymm8,%ymm8+ leaq 64(%rsi),%rsi++ vpsrldq $6,%ymm7,%ymm9+ vpsrldq $6,%ymm8,%ymm10+ vpunpckhqdq %ymm8,%ymm7,%ymm6+ vpunpcklqdq %ymm10,%ymm9,%ymm9+ vpunpcklqdq %ymm8,%ymm7,%ymm7++ vpsrlq $30,%ymm9,%ymm10+ vpsrlq $4,%ymm9,%ymm9+ vpsrlq $26,%ymm7,%ymm8+ vpsrlq $40,%ymm6,%ymm6+ vpand %ymm5,%ymm9,%ymm9+ vpand %ymm5,%ymm7,%ymm7+ vpand %ymm5,%ymm8,%ymm8+ vpand %ymm5,%ymm10,%ymm10+ vpor 32(%rcx),%ymm6,%ymm6++ vpaddq %ymm2,%ymm9,%ymm2+ subq $64,%rdx+ jz .Ltail_avx2+ jmp .Loop_avx2++.align 32+.Loop_avx2:+++++++++ vpaddq %ymm0,%ymm7,%ymm0+ vmovdqa 0(%rsp),%ymm7+ vpaddq %ymm1,%ymm8,%ymm1+ vmovdqa 32(%rsp),%ymm8+ vpaddq %ymm3,%ymm10,%ymm3+ vmovdqa 96(%rsp),%ymm9+ vpaddq %ymm4,%ymm6,%ymm4+ vmovdqa 48(%rax),%ymm10+ vmovdqa 112(%rax),%ymm5+++++++++++++++++ vpmuludq %ymm2,%ymm7,%ymm13+ vpmuludq %ymm2,%ymm8,%ymm14+ vpmuludq %ymm2,%ymm9,%ymm15+ vpmuludq %ymm2,%ymm10,%ymm11+ vpmuludq %ymm2,%ymm5,%ymm12++ vpmuludq %ymm0,%ymm8,%ymm6+ vpmuludq %ymm1,%ymm8,%ymm2+ vpaddq %ymm6,%ymm12,%ymm12+ vpaddq %ymm2,%ymm13,%ymm13+ vpmuludq %ymm3,%ymm8,%ymm6+ vpmuludq 64(%rsp),%ymm4,%ymm2+ vpaddq %ymm6,%ymm15,%ymm15+ vpaddq %ymm2,%ymm11,%ymm11+ vmovdqa -16(%rax),%ymm8++ vpmuludq %ymm0,%ymm7,%ymm6+ vpmuludq %ymm1,%ymm7,%ymm2+ vpaddq %ymm6,%ymm11,%ymm11+ vpaddq %ymm2,%ymm12,%ymm12+ vpmuludq %ymm3,%ymm7,%ymm6+ vpmuludq %ymm4,%ymm7,%ymm2+ vmovdqu 0(%rsi),%xmm7+ vpaddq %ymm6,%ymm14,%ymm14+ vpaddq %ymm2,%ymm15,%ymm15+ vinserti128 $1,32(%rsi),%ymm7,%ymm7++ vpmuludq %ymm3,%ymm8,%ymm6+ vpmuludq %ymm4,%ymm8,%ymm2+ vmovdqu 16(%rsi),%xmm8+ vpaddq %ymm6,%ymm11,%ymm11+ vpaddq %ymm2,%ymm12,%ymm12+ vmovdqa 16(%rax),%ymm2+ vpmuludq %ymm1,%ymm9,%ymm6+ vpmuludq %ymm0,%ymm9,%ymm9+ vpaddq %ymm6,%ymm14,%ymm14+ vpaddq %ymm9,%ymm13,%ymm13+ vinserti128 $1,48(%rsi),%ymm8,%ymm8+ leaq 64(%rsi),%rsi++ vpmuludq %ymm1,%ymm2,%ymm6+ vpmuludq %ymm0,%ymm2,%ymm2+ vpsrldq $6,%ymm7,%ymm9+ vpaddq %ymm6,%ymm15,%ymm15+ vpaddq %ymm2,%ymm14,%ymm14+ vpmuludq %ymm3,%ymm10,%ymm6+ vpmuludq %ymm4,%ymm10,%ymm2+ vpsrldq $6,%ymm8,%ymm10+ vpaddq %ymm6,%ymm12,%ymm12+ vpaddq %ymm2,%ymm13,%ymm13+ vpunpckhqdq %ymm8,%ymm7,%ymm6++ vpmuludq %ymm3,%ymm5,%ymm3+ vpmuludq %ymm4,%ymm5,%ymm4+ vpunpcklqdq %ymm8,%ymm7,%ymm7+ vpaddq %ymm3,%ymm13,%ymm2+ vpaddq %ymm4,%ymm14,%ymm3+ vpunpcklqdq %ymm10,%ymm9,%ymm10+ vpmuludq 80(%rax),%ymm0,%ymm4+ vpmuludq %ymm1,%ymm5,%ymm0+ vmovdqa 64(%rcx),%ymm5+ vpaddq %ymm4,%ymm15,%ymm4+ vpaddq %ymm0,%ymm11,%ymm0+++++ vpsrlq $26,%ymm3,%ymm14+ vpand %ymm5,%ymm3,%ymm3+ vpaddq %ymm14,%ymm4,%ymm4++ vpsrlq $26,%ymm0,%ymm11+ vpand %ymm5,%ymm0,%ymm0+ vpaddq %ymm11,%ymm12,%ymm1++ vpsrlq $26,%ymm4,%ymm15+ vpand %ymm5,%ymm4,%ymm4++ vpsrlq $4,%ymm10,%ymm9++ vpsrlq $26,%ymm1,%ymm12+ vpand %ymm5,%ymm1,%ymm1+ vpaddq %ymm12,%ymm2,%ymm2++ vpaddq %ymm15,%ymm0,%ymm0+ vpsllq $2,%ymm15,%ymm15+ vpaddq %ymm15,%ymm0,%ymm0++ vpand %ymm5,%ymm9,%ymm9+ vpsrlq $26,%ymm7,%ymm8++ vpsrlq $26,%ymm2,%ymm13+ vpand %ymm5,%ymm2,%ymm2+ vpaddq %ymm13,%ymm3,%ymm3++ vpaddq %ymm9,%ymm2,%ymm2+ vpsrlq $30,%ymm10,%ymm10++ vpsrlq $26,%ymm0,%ymm11+ vpand %ymm5,%ymm0,%ymm0+ vpaddq %ymm11,%ymm1,%ymm1++ vpsrlq $40,%ymm6,%ymm6++ vpsrlq $26,%ymm3,%ymm14+ vpand %ymm5,%ymm3,%ymm3+ vpaddq %ymm14,%ymm4,%ymm4++ vpand %ymm5,%ymm7,%ymm7+ vpand %ymm5,%ymm8,%ymm8+ vpand %ymm5,%ymm10,%ymm10+ vpor 32(%rcx),%ymm6,%ymm6++ subq $64,%rdx+ jnz .Loop_avx2++.byte 0x66,0x90+.Ltail_avx2:++++++++ vpaddq %ymm0,%ymm7,%ymm0+ vmovdqu 4(%rsp),%ymm7+ vpaddq %ymm1,%ymm8,%ymm1+ vmovdqu 36(%rsp),%ymm8+ vpaddq %ymm3,%ymm10,%ymm3+ vmovdqu 100(%rsp),%ymm9+ vpaddq %ymm4,%ymm6,%ymm4+ vmovdqu 52(%rax),%ymm10+ vmovdqu 116(%rax),%ymm5++ vpmuludq %ymm2,%ymm7,%ymm13+ vpmuludq %ymm2,%ymm8,%ymm14+ vpmuludq %ymm2,%ymm9,%ymm15+ vpmuludq %ymm2,%ymm10,%ymm11+ vpmuludq %ymm2,%ymm5,%ymm12++ vpmuludq %ymm0,%ymm8,%ymm6+ vpmuludq %ymm1,%ymm8,%ymm2+ vpaddq %ymm6,%ymm12,%ymm12+ vpaddq %ymm2,%ymm13,%ymm13+ vpmuludq %ymm3,%ymm8,%ymm6+ vpmuludq 68(%rsp),%ymm4,%ymm2+ vpaddq %ymm6,%ymm15,%ymm15+ vpaddq %ymm2,%ymm11,%ymm11++ vpmuludq %ymm0,%ymm7,%ymm6+ vpmuludq %ymm1,%ymm7,%ymm2+ vpaddq %ymm6,%ymm11,%ymm11+ vmovdqu -12(%rax),%ymm8+ vpaddq %ymm2,%ymm12,%ymm12+ vpmuludq %ymm3,%ymm7,%ymm6+ vpmuludq %ymm4,%ymm7,%ymm2+ vpaddq %ymm6,%ymm14,%ymm14+ vpaddq %ymm2,%ymm15,%ymm15++ vpmuludq %ymm3,%ymm8,%ymm6+ vpmuludq %ymm4,%ymm8,%ymm2+ vpaddq %ymm6,%ymm11,%ymm11+ vpaddq %ymm2,%ymm12,%ymm12+ vmovdqu 20(%rax),%ymm2+ vpmuludq %ymm1,%ymm9,%ymm6+ vpmuludq %ymm0,%ymm9,%ymm9+ vpaddq %ymm6,%ymm14,%ymm14+ vpaddq %ymm9,%ymm13,%ymm13++ vpmuludq %ymm1,%ymm2,%ymm6+ vpmuludq %ymm0,%ymm2,%ymm2+ vpaddq %ymm6,%ymm15,%ymm15+ vpaddq %ymm2,%ymm14,%ymm14+ vpmuludq %ymm3,%ymm10,%ymm6+ vpmuludq %ymm4,%ymm10,%ymm2+ vpaddq %ymm6,%ymm12,%ymm12+ vpaddq %ymm2,%ymm13,%ymm13++ vpmuludq %ymm3,%ymm5,%ymm3+ vpmuludq %ymm4,%ymm5,%ymm4+ vpaddq %ymm3,%ymm13,%ymm2+ vpaddq %ymm4,%ymm14,%ymm3+ vpmuludq 84(%rax),%ymm0,%ymm4+ vpmuludq %ymm1,%ymm5,%ymm0+ vmovdqa 64(%rcx),%ymm5+ vpaddq %ymm4,%ymm15,%ymm4+ vpaddq %ymm0,%ymm11,%ymm0+++++ vpsrldq $8,%ymm12,%ymm8+ vpsrldq $8,%ymm2,%ymm9+ vpsrldq $8,%ymm3,%ymm10+ vpsrldq $8,%ymm4,%ymm6+ vpsrldq $8,%ymm0,%ymm7+ vpaddq %ymm8,%ymm12,%ymm12+ vpaddq %ymm9,%ymm2,%ymm2+ vpaddq %ymm10,%ymm3,%ymm3+ vpaddq %ymm6,%ymm4,%ymm4+ vpaddq %ymm7,%ymm0,%ymm0++ vpermq $0x2,%ymm3,%ymm10+ vpermq $0x2,%ymm4,%ymm6+ vpermq $0x2,%ymm0,%ymm7+ vpermq $0x2,%ymm12,%ymm8+ vpermq $0x2,%ymm2,%ymm9+ vpaddq %ymm10,%ymm3,%ymm3+ vpaddq %ymm6,%ymm4,%ymm4+ vpaddq %ymm7,%ymm0,%ymm0+ vpaddq %ymm8,%ymm12,%ymm12+ vpaddq %ymm9,%ymm2,%ymm2+++++ vpsrlq $26,%ymm3,%ymm14+ vpand %ymm5,%ymm3,%ymm3+ vpaddq %ymm14,%ymm4,%ymm4++ vpsrlq $26,%ymm0,%ymm11+ vpand %ymm5,%ymm0,%ymm0+ vpaddq %ymm11,%ymm12,%ymm1++ vpsrlq $26,%ymm4,%ymm15+ vpand %ymm5,%ymm4,%ymm4++ vpsrlq $26,%ymm1,%ymm12+ vpand %ymm5,%ymm1,%ymm1+ vpaddq %ymm12,%ymm2,%ymm2++ vpaddq %ymm15,%ymm0,%ymm0+ vpsllq $2,%ymm15,%ymm15+ vpaddq %ymm15,%ymm0,%ymm0++ vpsrlq $26,%ymm2,%ymm13+ vpand %ymm5,%ymm2,%ymm2+ vpaddq %ymm13,%ymm3,%ymm3++ vpsrlq $26,%ymm0,%ymm11+ vpand %ymm5,%ymm0,%ymm0+ vpaddq %ymm11,%ymm1,%ymm1++ vpsrlq $26,%ymm3,%ymm14+ vpand %ymm5,%ymm3,%ymm3+ vpaddq %ymm14,%ymm4,%ymm4++ vmovd %xmm0,-112(%rdi)+ vmovd %xmm1,-108(%rdi)+ vmovd %xmm2,-104(%rdi)+ vmovd %xmm3,-100(%rdi)+ vmovd %xmm4,-96(%rdi)+ leaq 8(%r11),%rsp+.cfi_def_cfa %rsp,8+ vzeroupper+ .byte 0xf3,0xc3+.cfi_endproc +.size crypton_poly1305_asm_blocks_avx2,.-crypton_poly1305_asm_blocks_avx2+.align 64+.Lconst:+.Lmask24:+.long 0x0ffffff,0,0x0ffffff,0,0x0ffffff,0,0x0ffffff,0+.L129:+.long 16777216,0,16777216,0,16777216,0,16777216,0+.Lmask26:+.long 0x3ffffff,0,0x3ffffff,0,0x3ffffff,0,0x3ffffff,0+.Lpermd_avx2:+.long 2,2,2,3,2,0,2,1+.Lpermd_avx512:+.long 0,0,0,1, 0,2,0,3, 0,4,0,5, 0,6,0,7++.L2_44_inp_permd:+.long 0,1,1,2,2,3,7,7+.L2_44_inp_shift:+.quad 0,12,24,64+.L2_44_mask:+.quad 0xfffffffffff,0xfffffffffff,0x3ffffffffff,0xffffffffffffffff+.L2_44_shift_rgt:+.quad 44,44,42,64+.L2_44_shift_lft:+.quad 8,8,10,64++.align 64+.Lx_mask44:+.quad 0xfffffffffff,0xfffffffffff,0xfffffffffff,0xfffffffffff+.quad 0xfffffffffff,0xfffffffffff,0xfffffffffff,0xfffffffffff+.Lx_mask42:+.quad 0x3ffffffffff,0x3ffffffffff,0x3ffffffffff,0x3ffffffffff+.quad 0x3ffffffffff,0x3ffffffffff,0x3ffffffffff,0x3ffffffffff+.byte 80,111,108,121,49,51,48,53,32,102,111,114,32,120,56,54,95,54,52,44,32,67,82,89,80,84,79,71,65,77,83,32,98,121,32,64,100,111,116,45,97,115,109,0+.align 16+.globl crypton_xor128_encrypt_n_pad+.type crypton_xor128_encrypt_n_pad,@function+.align 16+crypton_xor128_encrypt_n_pad:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa++ subq %rdx,%rsi+ subq %rdx,%rdi+ movq %rcx,%r10+ shrq $4,%rcx+ jz .Ltail_enc+ nop+.Loop_enc_xmm:+ movdqu (%rsi,%rdx,1),%xmm0+ pxor (%rdx),%xmm0+ movdqu %xmm0,(%rdi,%rdx,1)+ movdqa %xmm0,(%rdx)+ leaq 16(%rdx),%rdx+ decq %rcx+ jnz .Loop_enc_xmm++ andq $15,%r10+ jz .Ldone_enc++.Ltail_enc:+ movq $16,%rcx+ subq %r10,%rcx+ xorl %eax,%eax+.Loop_enc_byte:+ movb (%rsi,%rdx,1),%al+ xorb (%rdx),%al+ movb %al,(%rdi,%rdx,1)+ movb %al,(%rdx)+ leaq 1(%rdx),%rdx+ decq %r10+ jnz .Loop_enc_byte++ xorl %eax,%eax+.Loop_enc_pad:+ movb %al,(%rdx)+ leaq 1(%rdx),%rdx+ decq %rcx+ jnz .Loop_enc_pad++.Ldone_enc:+ movq %rdx,%rax+ .byte 0xf3,0xc3+.cfi_endproc+.size crypton_xor128_encrypt_n_pad,.-crypton_xor128_encrypt_n_pad++.globl crypton_xor128_decrypt_n_pad+.type crypton_xor128_decrypt_n_pad,@function+.align 16+crypton_xor128_decrypt_n_pad:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa++ subq %rdx,%rsi+ subq %rdx,%rdi+ movq %rcx,%r10+ shrq $4,%rcx+ jz .Ltail_dec+ nop+.Loop_dec_xmm:+ movdqu (%rsi,%rdx,1),%xmm0+ movdqa (%rdx),%xmm1+ pxor %xmm0,%xmm1+ movdqu %xmm1,(%rdi,%rdx,1)+ movdqa %xmm0,(%rdx)+ leaq 16(%rdx),%rdx+ decq %rcx+ jnz .Loop_dec_xmm++ pxor %xmm1,%xmm1+ andq $15,%r10+ jz .Ldone_dec++.Ltail_dec:+ movq $16,%rcx+ subq %r10,%rcx+ xorl %eax,%eax+ xorq %r11,%r11+.Loop_dec_byte:+ movb (%rsi,%rdx,1),%r11b+ movb (%rdx),%al+ xorb %r11b,%al+ movb %al,(%rdi,%rdx,1)+ movb %r11b,(%rdx)+ leaq 1(%rdx),%rdx+ decq %r10+ jnz .Loop_dec_byte++ xorl %eax,%eax+.Loop_dec_pad:+ movb %al,(%rdx)+ leaq 1(%rdx),%rdx+ decq %rcx+ jnz .Loop_dec_pad++.Ldone_dec:+ movq %rdx,%rax+ .byte 0xf3,0xc3+.cfi_endproc+.size crypton_xor128_decrypt_n_pad,.-crypton_xor128_decrypt_n_pad++.section .note.gnu.property,"a",@note+ .long 4,2f-1f,5+ .byte 0x47,0x4E,0x55,0+1: .long 0xc0000002,4,3+.align 8+2:++.section .note.GNU-stack,"",@progbits
@@ -0,0 +1,2024 @@+.text ++++.globl _crypton_poly1305_asm_init+.private_extern _crypton_poly1305_asm_init+.globl _crypton_poly1305_asm_blocks+.private_extern _crypton_poly1305_asm_blocks+.globl _crypton_poly1305_asm_emit+.private_extern _crypton_poly1305_asm_emit+++.p2align 5+_crypton_poly1305_asm_init:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa++ xorq %rax,%rax+ movq %rax,0(%rdi)+ movq %rax,8(%rdi)+ movq %rax,16(%rdi)++ cmpq $0,%rsi+ je L$no_key++ movq $0x0ffffffc0fffffff,%rax+ leaq -3(%rax),%rcx+ andq 0(%rsi),%rax+ andq 8(%rsi),%rcx+ movq %rax,24(%rdi)+ movq %rcx,32(%rdi)+ movl $-1,48(%rdi)+ leaq _crypton_poly1305_asm_blocks(%rip),%r10+ leaq _crypton_poly1305_asm_emit(%rip),%r11+ movq _crypton_ia32cap_P+4(%rip),%r9+ leaq crypton_poly1305_asm_blocks_avx(%rip),%rax+ btq $28,%r9+ cmovcq %rax,%r10+ leaq crypton_poly1305_asm_blocks_avx2(%rip),%rax+ btq $37,%r9+ cmovcq %rax,%r10+ movq %r10,0(%rdx)+ movq %r11,8(%rdx)+ movl $1,%eax+L$no_key:+ .byte 0xf3,0xc3+.cfi_endproc++++.p2align 5+_crypton_poly1305_asm_blocks:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa+++L$blocks:+ shrq $4,%rdx+ jz L$no_data++ pushq %rbx+.cfi_adjust_cfa_offset 8+.cfi_offset %rbx,-16+ pushq %rbp+.cfi_adjust_cfa_offset 8+.cfi_offset %rbp,-24+ pushq %r12+.cfi_adjust_cfa_offset 8+.cfi_offset %r12,-32+ pushq %r13+.cfi_adjust_cfa_offset 8+.cfi_offset %r13,-40+ pushq %r14+.cfi_adjust_cfa_offset 8+.cfi_offset %r14,-48+ pushq %r15+.cfi_adjust_cfa_offset 8+.cfi_offset %r15,-56+ leaq -8(%rsp),%rsp+.cfi_adjust_cfa_offset 8+L$blocks_body:++ movq %rdx,%r15++ movq 24(%rdi),%r11+ movq 32(%rdi),%r13++ movq 0(%rdi),%r14+ movq 8(%rdi),%rbx+ movq 16(%rdi),%rbp++ movl %r14d,%eax+ movl 4(%rdi),%edx+ movl %ebx,%r8d+ movl 12(%rdi),%r10d+ movl %ebp,%r12d++ shlq $26,%rdx+ movq %r8,%r9+ shlq $52,%r8+ addq %rdx,%rax+ shrq $12,%r9+ addq %rax,%r8+ adcq $0,%r9++ shlq $14,%r10+ movq %r12,%rax+ shrq $24,%r12+ addq %r10,%r9+ shlq $40,%rax+ addq %rax,%r9+ adcq $0,%r12++ cmpq $4,%rbp++ cmovaq %r8,%r14+ cmovaq %r9,%rbx+ cmovaq %r12,%rbp++ movq %r13,%r12+ shrq $2,%r13+ movq %r12,%rax+ addq %r12,%r13+ jmp L$oop++.p2align 5+L$oop:+ addq 0(%rsi),%r14+ adcq 8(%rsi),%rbx+ leaq 16(%rsi),%rsi+ adcq %rcx,%rbp+ mulq %r14+ movq %rax,%r9+ movq %r11,%rax+ movq %rdx,%r10++ mulq %r14+ movq %rax,%r14+ movq %r11,%rax+ movq %rdx,%r8++ mulq %rbx+ addq %rax,%r9+ movq %r13,%rax+ adcq %rdx,%r10++ mulq %rbx+ movq %rbp,%rbx+ addq %rax,%r14+ adcq %rdx,%r8++ imulq %r13,%rbx+ addq %rbx,%r9+ movq %r8,%rbx+ adcq $0,%r10++ imulq %r11,%rbp+ addq %r9,%rbx+ movq $-4,%rax+ adcq %rbp,%r10++ andq %r10,%rax+ movq %r10,%rbp+ shrq $2,%r10+ andq $3,%rbp+ addq %r10,%rax+ addq %rax,%r14+ adcq $0,%rbx+ adcq $0,%rbp+ movq %r12,%rax+ decq %r15+ jnz L$oop++ movq %r14,0(%rdi)+ movq %rbx,8(%rdi)+ movq %rbp,16(%rdi)++ movq 8(%rsp),%r15+.cfi_restore %r15+ movq 16(%rsp),%r14+.cfi_restore %r14+ movq 24(%rsp),%r13+.cfi_restore %r13+ movq 32(%rsp),%r12+.cfi_restore %r12+ movq 40(%rsp),%rbp+.cfi_restore %rbp+ movq 48(%rsp),%rbx+.cfi_restore %rbx+ leaq 56(%rsp),%rsp+.cfi_adjust_cfa_offset -56+L$no_data:+L$blocks_epilogue:+ .byte 0xf3,0xc3+.cfi_endproc ++++.p2align 5+_crypton_poly1305_asm_emit:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa++ movl 0(%rdi),%eax+ movl 4(%rdi),%ecx+ movl 8(%rdi),%r8d+ movl 12(%rdi),%r11d+ movl 16(%rdi),%r10d++ shlq $26,%rcx+ movq %r8,%r9+ shlq $52,%r8+ addq %rcx,%rax+ shrq $12,%r9+ addq %rax,%r8+ adcq $0,%r9++ shlq $14,%r11+ movq %r10,%rax+ shrq $24,%r10+ addq %r11,%r9+ movq 0(%rdi),%rcx+ shlq $40,%rax+ movq 8(%rdi),%r11+ addq %rax,%r9+ movq 16(%rdi),%rax+ adcq $0,%r10++ cmpq $4,%rax++ cmovbeq %rcx,%r8+ cmovbeq %r11,%r9+ cmovbeq %rax,%r10++ movq %r8,%rax+ addq $5,%r8+ movq %r9,%rcx+ adcq $0,%r9+ adcq $0,%r10+ shrq $2,%r10+ cmovnzq %r8,%rax+ cmovnzq %r9,%rcx++ addq 0(%rdx),%rax+ adcq 8(%rdx),%rcx+ movq %rax,0(%rsi)+ movq %rcx,8(%rsi)++ .byte 0xf3,0xc3+.cfi_endproc+++.p2align 5+__crypton_poly1305_asm_block:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa++ mulq %r14+ movq %rax,%r9+ movq %r11,%rax+ movq %rdx,%r10++ mulq %r14+ movq %rax,%r14+ movq %r11,%rax+ movq %rdx,%r8++ mulq %rbx+ addq %rax,%r9+ movq %r13,%rax+ adcq %rdx,%r10++ mulq %rbx+ movq %rbp,%rbx+ addq %rax,%r14+ adcq %rdx,%r8++ imulq %r13,%rbx+ addq %rbx,%r9+ movq %r8,%rbx+ adcq $0,%r10++ imulq %r11,%rbp+ addq %r9,%rbx+ movq $-4,%rax+ adcq %rbp,%r10++ andq %r10,%rax+ movq %r10,%rbp+ shrq $2,%r10+ andq $3,%rbp+ addq %r10,%rax+ addq %rax,%r14+ adcq $0,%rbx+ adcq $0,%rbp+ .byte 0xf3,0xc3+.cfi_endproc++++.p2align 5+__crypton_poly1305_asm_init_avx:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa++ cmpl $-1,48(%rdi)+ jne L$done_init_avx++ movq %r11,%r14+ movq %r12,%rbx+ xorq %rbp,%rbp++ leaq 48+64(%rdi),%rdi++ movq %r12,%rax+ call __crypton_poly1305_asm_block++ movl $0x3ffffff,%eax+ movl $0x3ffffff,%edx+ movq %r14,%r8+ andl %r14d,%eax+ movq %r11,%r9+ andl %r11d,%edx+ movl %eax,-64(%rdi)+ shrq $26,%r8+ movl %edx,-60(%rdi)+ shrq $26,%r9++ movl $0x3ffffff,%eax+ movl $0x3ffffff,%edx+ andl %r8d,%eax+ andl %r9d,%edx+ movl %eax,-48(%rdi)+ leal (%rax,%rax,4),%eax+ movl %edx,-44(%rdi)+ leal (%rdx,%rdx,4),%edx+ movl %eax,-32(%rdi)+ shrq $26,%r8+ movl %edx,-28(%rdi)+ shrq $26,%r9++ movq %rbx,%rax+ movq %r12,%rdx+ shlq $12,%rax+ shlq $12,%rdx+ orq %r8,%rax+ orq %r9,%rdx+ andl $0x3ffffff,%eax+ andl $0x3ffffff,%edx+ movl %eax,-16(%rdi)+ leal (%rax,%rax,4),%eax+ movl %edx,-12(%rdi)+ leal (%rdx,%rdx,4),%edx+ movl %eax,0(%rdi)+ movq %rbx,%r8+ movl %edx,4(%rdi)+ movq %r12,%r9++ movl $0x3ffffff,%eax+ movl $0x3ffffff,%edx+ shrq $14,%r8+ shrq $14,%r9+ andl %r8d,%eax+ andl %r9d,%edx+ movl %eax,16(%rdi)+ leal (%rax,%rax,4),%eax+ movl %edx,20(%rdi)+ leal (%rdx,%rdx,4),%edx+ movl %eax,32(%rdi)+ shrq $26,%r8+ movl %edx,36(%rdi)+ shrq $26,%r9++ movq %rbp,%rax+ shlq $24,%rax+ orq %rax,%r8+ movl %r8d,48(%rdi)+ leaq (%r8,%r8,4),%r8+ movl %r9d,52(%rdi)+ leaq (%r9,%r9,4),%r9+ movl %r8d,64(%rdi)+ movl %r9d,68(%rdi)++ movq %r12,%rax+ call __crypton_poly1305_asm_block++ movl $0x3ffffff,%eax+ movq %r14,%r8+ andl %r14d,%eax+ shrq $26,%r8+ movl %eax,-52(%rdi)++ movl $0x3ffffff,%edx+ andl %r8d,%edx+ movl %edx,-36(%rdi)+ leal (%rdx,%rdx,4),%edx+ shrq $26,%r8+ movl %edx,-20(%rdi)++ movq %rbx,%rax+ shlq $12,%rax+ orq %r8,%rax+ andl $0x3ffffff,%eax+ movl %eax,-4(%rdi)+ leal (%rax,%rax,4),%eax+ movq %rbx,%r8+ movl %eax,12(%rdi)++ movl $0x3ffffff,%edx+ shrq $14,%r8+ andl %r8d,%edx+ movl %edx,28(%rdi)+ leal (%rdx,%rdx,4),%edx+ shrq $26,%r8+ movl %edx,44(%rdi)++ movq %rbp,%rax+ shlq $24,%rax+ orq %rax,%r8+ movl %r8d,60(%rdi)+ leaq (%r8,%r8,4),%r8+ movl %r8d,76(%rdi)++ movq %r12,%rax+ call __crypton_poly1305_asm_block++ movl $0x3ffffff,%eax+ movq %r14,%r8+ andl %r14d,%eax+ shrq $26,%r8+ movl %eax,-56(%rdi)++ movl $0x3ffffff,%edx+ andl %r8d,%edx+ movl %edx,-40(%rdi)+ leal (%rdx,%rdx,4),%edx+ shrq $26,%r8+ movl %edx,-24(%rdi)++ movq %rbx,%rax+ shlq $12,%rax+ orq %r8,%rax+ andl $0x3ffffff,%eax+ movl %eax,-8(%rdi)+ leal (%rax,%rax,4),%eax+ movq %rbx,%r8+ movl %eax,8(%rdi)++ movl $0x3ffffff,%edx+ shrq $14,%r8+ andl %r8d,%edx+ movl %edx,24(%rdi)+ leal (%rdx,%rdx,4),%edx+ shrq $26,%r8+ movl %edx,40(%rdi)++ movq %rbp,%rax+ shlq $24,%rax+ orq %rax,%r8+ movl %r8d,56(%rdi)+ leaq (%r8,%r8,4),%r8+ movl %r8d,72(%rdi)++ leaq -48-64(%rdi),%rdi+L$done_init_avx:+ .byte 0xf3,0xc3+.cfi_endproc++++.p2align 5+crypton_poly1305_asm_blocks_avx:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa+++ movl 20(%rdi),%r8d+ cmpq $128,%rdx+ jb L$blocks++ andq $-16,%rdx++ vzeroupper++ testl %r8d,%r8d+ jz L$base2_64_avx++ testq $31,%rdx+ jz L$even_avx++ pushq %rbx+.cfi_adjust_cfa_offset 8+.cfi_offset %rbx,-16+ pushq %rbp+.cfi_adjust_cfa_offset 8+.cfi_offset %rbp,-24+ pushq %r12+.cfi_adjust_cfa_offset 8+.cfi_offset %r12,-32+ pushq %r13+.cfi_adjust_cfa_offset 8+.cfi_offset %r13,-40+ pushq %r14+.cfi_adjust_cfa_offset 8+.cfi_offset %r14,-48+ pushq %r15+.cfi_adjust_cfa_offset 8+.cfi_offset %r15,-56+ leaq -8(%rsp),%rsp+.cfi_adjust_cfa_offset 8+L$blocks_avx_body:++ movq %rdx,%r15++ movq 0(%rdi),%r8+ movq 8(%rdi),%r9+ movl 16(%rdi),%ebp++ movq 24(%rdi),%r11+ movq 32(%rdi),%r13+++ movl %r8d,%r14d+ andq $-2147483648,%r8+ movq %r9,%r12+ movl %r9d,%ebx+ andq $-2147483648,%r9++ shrq $6,%r8+ shlq $52,%r12+ addq %r8,%r14+ shrq $12,%rbx+ shrq $18,%r9+ addq %r12,%r14+ adcq %r9,%rbx++ movq %rbp,%r8+ shlq $40,%r8+ shrq $24,%rbp+ addq %r8,%rbx+ adcq $0,%rbp++ movq %r13,%r12+ movq %r13,%rax+ shrq $2,%r13+ addq %r12,%r13++ addq 0(%rsi),%r14+ adcq 8(%rsi),%rbx+ leaq 16(%rsi),%rsi+ adcq %rcx,%rbp++ call __crypton_poly1305_asm_block+++ movq %r14,%rax+ movq %r14,%rdx+ shrq $52,%r14+ movq %rbx,%r11+ movq %rbx,%r12+ shrq $26,%rdx+ andq $0x3ffffff,%rax+ shlq $12,%r11+ andq $0x3ffffff,%rdx+ shrq $14,%rbx+ orq %r11,%r14+ shlq $24,%rbp+ andq $0x3ffffff,%r14+ shrq $40,%r12+ andq $0x3ffffff,%rbx+ orq %r12,%rbp++ vmovd %eax,%xmm0+ vmovd %edx,%xmm1+ vmovd %r14d,%xmm2+ vmovd %ebx,%xmm3+ vmovd %ebp,%xmm4++ leaq -16(%r15),%rdx++ movq 8(%rsp),%r15+.cfi_restore %r15+ movq 16(%rsp),%r14+.cfi_restore %r14+ movq 24(%rsp),%r13+.cfi_restore %r13+ movq 32(%rsp),%r12+.cfi_restore %r12+ movq 40(%rsp),%rbp+.cfi_restore %rbp+ movq 48(%rsp),%rbx+.cfi_restore %rbx+ leaq 56(%rsp),%rax+ leaq 56(%rsp),%rsp+.cfi_adjust_cfa_offset -56+L$blocks_avx_epilogue:+ jmp L$do_avx+.cfi_endproc ++.p2align 5+L$base2_64_avx:+.cfi_startproc + pushq %rbx+.cfi_adjust_cfa_offset 8+.cfi_offset %rbx,-16+ pushq %rbp+.cfi_adjust_cfa_offset 8+.cfi_offset %rbp,-24+ pushq %r12+.cfi_adjust_cfa_offset 8+.cfi_offset %r12,-32+ pushq %r13+.cfi_adjust_cfa_offset 8+.cfi_offset %r13,-40+ pushq %r14+.cfi_adjust_cfa_offset 8+.cfi_offset %r14,-48+ pushq %r15+.cfi_adjust_cfa_offset 8+.cfi_offset %r15,-56+ leaq -8(%rsp),%rsp+.cfi_adjust_cfa_offset 8+L$base2_64_avx_body:++ movq %rdx,%r15++ movq 24(%rdi),%r11+ movq 32(%rdi),%r13++ movq 0(%rdi),%r14+ movq 8(%rdi),%rbx+ movl 16(%rdi),%ebp++ movq %r13,%r12+ movq %r13,%rax+ shrq $2,%r13+ addq %r12,%r13++ testq $31,%rdx+ jz L$init_avx++ addq 0(%rsi),%r14+ adcq 8(%rsi),%rbx+ leaq 16(%rsi),%rsi+ adcq %rcx,%rbp+ subq $16,%r15++ call __crypton_poly1305_asm_block++L$init_avx:++ movq %r14,%rax+ movq %r14,%rdx+ shrq $52,%r14+ movq %rbx,%r8+ movq %rbx,%r9+ shrq $26,%rdx+ andq $0x3ffffff,%rax+ shlq $12,%r8+ andq $0x3ffffff,%rdx+ shrq $14,%rbx+ orq %r8,%r14+ shlq $24,%rbp+ andq $0x3ffffff,%r14+ shrq $40,%r9+ andq $0x3ffffff,%rbx+ orq %r9,%rbp++ vmovd %eax,%xmm0+ vmovd %edx,%xmm1+ vmovd %r14d,%xmm2+ vmovd %ebx,%xmm3+ vmovd %ebp,%xmm4+ movl $1,20(%rdi)++ call __crypton_poly1305_asm_init_avx++ movq %r15,%rdx++ movq 8(%rsp),%r15+.cfi_restore %r15+ movq 16(%rsp),%r14+.cfi_restore %r14+ movq 24(%rsp),%r13+.cfi_restore %r13+ movq 32(%rsp),%r12+.cfi_restore %r12+ movq 40(%rsp),%rbp+.cfi_restore %rbp+ movq 48(%rsp),%rbx+.cfi_restore %rbx+ leaq 56(%rsp),%rax+ leaq 56(%rsp),%rsp+.cfi_adjust_cfa_offset -56+L$base2_64_avx_epilogue:+ jmp L$do_avx+.cfi_endproc ++.p2align 5+L$even_avx:+.cfi_startproc + vmovd 0(%rdi),%xmm0+ vmovd 4(%rdi),%xmm1+ vmovd 8(%rdi),%xmm2+ vmovd 12(%rdi),%xmm3+ vmovd 16(%rdi),%xmm4++L$do_avx:+ leaq -88(%rsp),%r11+.cfi_def_cfa %r11,0x60+ subq $0x178,%rsp+ subq $64,%rdx+ leaq -32(%rsi),%rax+ cmovcq %rax,%rsi++ vmovdqu 48(%rdi),%xmm14+ leaq 112(%rdi),%rdi+ leaq L$const(%rip),%rcx++++ vmovdqu 32(%rsi),%xmm5+ vmovdqu 48(%rsi),%xmm6+ vmovdqa 64(%rcx),%xmm15++ vpsrldq $6,%xmm5,%xmm7+ vpsrldq $6,%xmm6,%xmm8+ vpunpckhqdq %xmm6,%xmm5,%xmm9+ vpunpcklqdq %xmm6,%xmm5,%xmm5+ vpunpcklqdq %xmm8,%xmm7,%xmm8++ vpsrlq $40,%xmm9,%xmm9+ vpsrlq $26,%xmm5,%xmm6+ vpand %xmm15,%xmm5,%xmm5+ vpsrlq $4,%xmm8,%xmm7+ vpand %xmm15,%xmm6,%xmm6+ vpsrlq $30,%xmm8,%xmm8+ vpand %xmm15,%xmm7,%xmm7+ vpand %xmm15,%xmm8,%xmm8+ vpor 32(%rcx),%xmm9,%xmm9++ jbe L$skip_loop_avx+++ vmovdqu -48(%rdi),%xmm11+ vmovdqu -32(%rdi),%xmm12+ vpshufd $0xEE,%xmm14,%xmm13+ vpshufd $0x44,%xmm14,%xmm10+ vmovdqa %xmm13,-144(%r11)+ vmovdqa %xmm10,0(%rsp)+ vpshufd $0xEE,%xmm11,%xmm14+ vmovdqu -16(%rdi),%xmm10+ vpshufd $0x44,%xmm11,%xmm11+ vmovdqa %xmm14,-128(%r11)+ vmovdqa %xmm11,16(%rsp)+ vpshufd $0xEE,%xmm12,%xmm13+ vmovdqu 0(%rdi),%xmm11+ vpshufd $0x44,%xmm12,%xmm12+ vmovdqa %xmm13,-112(%r11)+ vmovdqa %xmm12,32(%rsp)+ vpshufd $0xEE,%xmm10,%xmm14+ vmovdqu 16(%rdi),%xmm12+ vpshufd $0x44,%xmm10,%xmm10+ vmovdqa %xmm14,-96(%r11)+ vmovdqa %xmm10,48(%rsp)+ vpshufd $0xEE,%xmm11,%xmm13+ vmovdqu 32(%rdi),%xmm10+ vpshufd $0x44,%xmm11,%xmm11+ vmovdqa %xmm13,-80(%r11)+ vmovdqa %xmm11,64(%rsp)+ vpshufd $0xEE,%xmm12,%xmm14+ vmovdqu 48(%rdi),%xmm11+ vpshufd $0x44,%xmm12,%xmm12+ vmovdqa %xmm14,-64(%r11)+ vmovdqa %xmm12,80(%rsp)+ vpshufd $0xEE,%xmm10,%xmm13+ vmovdqu 64(%rdi),%xmm12+ vpshufd $0x44,%xmm10,%xmm10+ vmovdqa %xmm13,-48(%r11)+ vmovdqa %xmm10,96(%rsp)+ vpshufd $0xEE,%xmm11,%xmm14+ vpshufd $0x44,%xmm11,%xmm11+ vmovdqa %xmm14,-32(%r11)+ vmovdqa %xmm11,112(%rsp)+ vpshufd $0xEE,%xmm12,%xmm13+ vmovdqa 0(%rsp),%xmm14+ vpshufd $0x44,%xmm12,%xmm12+ vmovdqa %xmm13,-16(%r11)+ vmovdqa %xmm12,128(%rsp)++ jmp L$oop_avx++.p2align 5+L$oop_avx:+++++++++++++++++++++ vpmuludq %xmm5,%xmm14,%xmm10+ vpmuludq %xmm6,%xmm14,%xmm11+ vmovdqa %xmm2,32(%r11)+ vpmuludq %xmm7,%xmm14,%xmm12+ vmovdqa 16(%rsp),%xmm2+ vpmuludq %xmm8,%xmm14,%xmm13+ vpmuludq %xmm9,%xmm14,%xmm14++ vmovdqa %xmm0,0(%r11)+ vpmuludq 32(%rsp),%xmm9,%xmm0+ vmovdqa %xmm1,16(%r11)+ vpmuludq %xmm8,%xmm2,%xmm1+ vpaddq %xmm0,%xmm10,%xmm10+ vpaddq %xmm1,%xmm14,%xmm14+ vmovdqa %xmm3,48(%r11)+ vpmuludq %xmm7,%xmm2,%xmm0+ vpmuludq %xmm6,%xmm2,%xmm1+ vpaddq %xmm0,%xmm13,%xmm13+ vmovdqa 48(%rsp),%xmm3+ vpaddq %xmm1,%xmm12,%xmm12+ vmovdqa %xmm4,64(%r11)+ vpmuludq %xmm5,%xmm2,%xmm2+ vpmuludq %xmm7,%xmm3,%xmm0+ vpaddq %xmm2,%xmm11,%xmm11++ vmovdqa 64(%rsp),%xmm4+ vpaddq %xmm0,%xmm14,%xmm14+ vpmuludq %xmm6,%xmm3,%xmm1+ vpmuludq %xmm5,%xmm3,%xmm3+ vpaddq %xmm1,%xmm13,%xmm13+ vmovdqa 80(%rsp),%xmm2+ vpaddq %xmm3,%xmm12,%xmm12+ vpmuludq %xmm9,%xmm4,%xmm0+ vpmuludq %xmm8,%xmm4,%xmm4+ vpaddq %xmm0,%xmm11,%xmm11+ vmovdqa 96(%rsp),%xmm3+ vpaddq %xmm4,%xmm10,%xmm10++ vmovdqa 128(%rsp),%xmm4+ vpmuludq %xmm6,%xmm2,%xmm1+ vpmuludq %xmm5,%xmm2,%xmm2+ vpaddq %xmm1,%xmm14,%xmm14+ vpaddq %xmm2,%xmm13,%xmm13+ vpmuludq %xmm9,%xmm3,%xmm0+ vpmuludq %xmm8,%xmm3,%xmm1+ vpaddq %xmm0,%xmm12,%xmm12+ vmovdqu 0(%rsi),%xmm0+ vpaddq %xmm1,%xmm11,%xmm11+ vpmuludq %xmm7,%xmm3,%xmm3+ vpmuludq %xmm7,%xmm4,%xmm7+ vpaddq %xmm3,%xmm10,%xmm10++ vmovdqu 16(%rsi),%xmm1+ vpaddq %xmm7,%xmm11,%xmm11+ vpmuludq %xmm8,%xmm4,%xmm8+ vpmuludq %xmm9,%xmm4,%xmm9+ vpsrldq $6,%xmm0,%xmm2+ vpaddq %xmm8,%xmm12,%xmm12+ vpaddq %xmm9,%xmm13,%xmm13+ vpsrldq $6,%xmm1,%xmm3+ vpmuludq 112(%rsp),%xmm5,%xmm9+ vpmuludq %xmm6,%xmm4,%xmm5+ vpunpckhqdq %xmm1,%xmm0,%xmm4+ vpaddq %xmm9,%xmm14,%xmm14+ vmovdqa -144(%r11),%xmm9+ vpaddq %xmm5,%xmm10,%xmm10++ vpunpcklqdq %xmm1,%xmm0,%xmm0+ vpunpcklqdq %xmm3,%xmm2,%xmm3+++ vpsrldq $5,%xmm4,%xmm4+ vpsrlq $26,%xmm0,%xmm1+ vpand %xmm15,%xmm0,%xmm0+ vpsrlq $4,%xmm3,%xmm2+ vpand %xmm15,%xmm1,%xmm1+ vpand 0(%rcx),%xmm4,%xmm4+ vpsrlq $30,%xmm3,%xmm3+ vpand %xmm15,%xmm2,%xmm2+ vpand %xmm15,%xmm3,%xmm3+ vpor 32(%rcx),%xmm4,%xmm4++ vpaddq 0(%r11),%xmm0,%xmm0+ vpaddq 16(%r11),%xmm1,%xmm1+ vpaddq 32(%r11),%xmm2,%xmm2+ vpaddq 48(%r11),%xmm3,%xmm3+ vpaddq 64(%r11),%xmm4,%xmm4++ leaq 32(%rsi),%rax+ leaq 64(%rsi),%rsi+ subq $64,%rdx+ cmovcq %rax,%rsi+++++++++++ vpmuludq %xmm0,%xmm9,%xmm5+ vpmuludq %xmm1,%xmm9,%xmm6+ vpaddq %xmm5,%xmm10,%xmm10+ vpaddq %xmm6,%xmm11,%xmm11+ vmovdqa -128(%r11),%xmm7+ vpmuludq %xmm2,%xmm9,%xmm5+ vpmuludq %xmm3,%xmm9,%xmm6+ vpaddq %xmm5,%xmm12,%xmm12+ vpaddq %xmm6,%xmm13,%xmm13+ vpmuludq %xmm4,%xmm9,%xmm9+ vpmuludq -112(%r11),%xmm4,%xmm5+ vpaddq %xmm9,%xmm14,%xmm14++ vpaddq %xmm5,%xmm10,%xmm10+ vpmuludq %xmm2,%xmm7,%xmm6+ vpmuludq %xmm3,%xmm7,%xmm5+ vpaddq %xmm6,%xmm13,%xmm13+ vmovdqa -96(%r11),%xmm8+ vpaddq %xmm5,%xmm14,%xmm14+ vpmuludq %xmm1,%xmm7,%xmm6+ vpmuludq %xmm0,%xmm7,%xmm7+ vpaddq %xmm6,%xmm12,%xmm12+ vpaddq %xmm7,%xmm11,%xmm11++ vmovdqa -80(%r11),%xmm9+ vpmuludq %xmm2,%xmm8,%xmm5+ vpmuludq %xmm1,%xmm8,%xmm6+ vpaddq %xmm5,%xmm14,%xmm14+ vpaddq %xmm6,%xmm13,%xmm13+ vmovdqa -64(%r11),%xmm7+ vpmuludq %xmm0,%xmm8,%xmm8+ vpmuludq %xmm4,%xmm9,%xmm5+ vpaddq %xmm8,%xmm12,%xmm12+ vpaddq %xmm5,%xmm11,%xmm11+ vmovdqa -48(%r11),%xmm8+ vpmuludq %xmm3,%xmm9,%xmm9+ vpmuludq %xmm1,%xmm7,%xmm6+ vpaddq %xmm9,%xmm10,%xmm10++ vmovdqa -16(%r11),%xmm9+ vpaddq %xmm6,%xmm14,%xmm14+ vpmuludq %xmm0,%xmm7,%xmm7+ vpmuludq %xmm4,%xmm8,%xmm5+ vpaddq %xmm7,%xmm13,%xmm13+ vpaddq %xmm5,%xmm12,%xmm12+ vmovdqu 32(%rsi),%xmm5+ vpmuludq %xmm3,%xmm8,%xmm7+ vpmuludq %xmm2,%xmm8,%xmm8+ vpaddq %xmm7,%xmm11,%xmm11+ vmovdqu 48(%rsi),%xmm6+ vpaddq %xmm8,%xmm10,%xmm10++ vpmuludq %xmm2,%xmm9,%xmm2+ vpmuludq %xmm3,%xmm9,%xmm3+ vpsrldq $6,%xmm5,%xmm7+ vpaddq %xmm2,%xmm11,%xmm11+ vpmuludq %xmm4,%xmm9,%xmm4+ vpsrldq $6,%xmm6,%xmm8+ vpaddq %xmm3,%xmm12,%xmm2+ vpaddq %xmm4,%xmm13,%xmm3+ vpmuludq -32(%r11),%xmm0,%xmm4+ vpmuludq %xmm1,%xmm9,%xmm0+ vpunpckhqdq %xmm6,%xmm5,%xmm9+ vpaddq %xmm4,%xmm14,%xmm4+ vpaddq %xmm0,%xmm10,%xmm0++ vpunpcklqdq %xmm6,%xmm5,%xmm5+ vpunpcklqdq %xmm8,%xmm7,%xmm8+++ vpsrldq $5,%xmm9,%xmm9+ vpsrlq $26,%xmm5,%xmm6+ vmovdqa 0(%rsp),%xmm14+ vpand %xmm15,%xmm5,%xmm5+ vpsrlq $4,%xmm8,%xmm7+ vpand %xmm15,%xmm6,%xmm6+ vpand 0(%rcx),%xmm9,%xmm9+ vpsrlq $30,%xmm8,%xmm8+ vpand %xmm15,%xmm7,%xmm7+ vpand %xmm15,%xmm8,%xmm8+ vpor 32(%rcx),%xmm9,%xmm9++++++ vpsrlq $26,%xmm3,%xmm13+ vpand %xmm15,%xmm3,%xmm3+ vpaddq %xmm13,%xmm4,%xmm4++ vpsrlq $26,%xmm0,%xmm10+ vpand %xmm15,%xmm0,%xmm0+ vpaddq %xmm10,%xmm11,%xmm1++ vpsrlq $26,%xmm4,%xmm10+ vpand %xmm15,%xmm4,%xmm4++ vpsrlq $26,%xmm1,%xmm11+ vpand %xmm15,%xmm1,%xmm1+ vpaddq %xmm11,%xmm2,%xmm2++ vpaddq %xmm10,%xmm0,%xmm0+ vpsllq $2,%xmm10,%xmm10+ vpaddq %xmm10,%xmm0,%xmm0++ vpsrlq $26,%xmm2,%xmm12+ vpand %xmm15,%xmm2,%xmm2+ vpaddq %xmm12,%xmm3,%xmm3++ vpsrlq $26,%xmm0,%xmm10+ vpand %xmm15,%xmm0,%xmm0+ vpaddq %xmm10,%xmm1,%xmm1++ vpsrlq $26,%xmm3,%xmm13+ vpand %xmm15,%xmm3,%xmm3+ vpaddq %xmm13,%xmm4,%xmm4++ ja L$oop_avx++L$skip_loop_avx:++++ vpshufd $0x10,%xmm14,%xmm14+ addq $32,%rdx+ jnz L$ong_tail_avx++ vpaddq %xmm2,%xmm7,%xmm7+ vpaddq %xmm0,%xmm5,%xmm5+ vpaddq %xmm1,%xmm6,%xmm6+ vpaddq %xmm3,%xmm8,%xmm8+ vpaddq %xmm4,%xmm9,%xmm9++L$ong_tail_avx:+ vmovdqa %xmm2,32(%r11)+ vmovdqa %xmm0,0(%r11)+ vmovdqa %xmm1,16(%r11)+ vmovdqa %xmm3,48(%r11)+ vmovdqa %xmm4,64(%r11)++++++++ vpmuludq %xmm7,%xmm14,%xmm12+ vpmuludq %xmm5,%xmm14,%xmm10+ vpshufd $0x10,-48(%rdi),%xmm2+ vpmuludq %xmm6,%xmm14,%xmm11+ vpmuludq %xmm8,%xmm14,%xmm13+ vpmuludq %xmm9,%xmm14,%xmm14++ vpmuludq %xmm8,%xmm2,%xmm0+ vpaddq %xmm0,%xmm14,%xmm14+ vpshufd $0x10,-32(%rdi),%xmm3+ vpmuludq %xmm7,%xmm2,%xmm1+ vpaddq %xmm1,%xmm13,%xmm13+ vpshufd $0x10,-16(%rdi),%xmm4+ vpmuludq %xmm6,%xmm2,%xmm0+ vpaddq %xmm0,%xmm12,%xmm12+ vpmuludq %xmm5,%xmm2,%xmm2+ vpaddq %xmm2,%xmm11,%xmm11+ vpmuludq %xmm9,%xmm3,%xmm3+ vpaddq %xmm3,%xmm10,%xmm10++ vpshufd $0x10,0(%rdi),%xmm2+ vpmuludq %xmm7,%xmm4,%xmm1+ vpaddq %xmm1,%xmm14,%xmm14+ vpmuludq %xmm6,%xmm4,%xmm0+ vpaddq %xmm0,%xmm13,%xmm13+ vpshufd $0x10,16(%rdi),%xmm3+ vpmuludq %xmm5,%xmm4,%xmm4+ vpaddq %xmm4,%xmm12,%xmm12+ vpmuludq %xmm9,%xmm2,%xmm1+ vpaddq %xmm1,%xmm11,%xmm11+ vpshufd $0x10,32(%rdi),%xmm4+ vpmuludq %xmm8,%xmm2,%xmm2+ vpaddq %xmm2,%xmm10,%xmm10++ vpmuludq %xmm6,%xmm3,%xmm0+ vpaddq %xmm0,%xmm14,%xmm14+ vpmuludq %xmm5,%xmm3,%xmm3+ vpaddq %xmm3,%xmm13,%xmm13+ vpshufd $0x10,48(%rdi),%xmm2+ vpmuludq %xmm9,%xmm4,%xmm1+ vpaddq %xmm1,%xmm12,%xmm12+ vpshufd $0x10,64(%rdi),%xmm3+ vpmuludq %xmm8,%xmm4,%xmm0+ vpaddq %xmm0,%xmm11,%xmm11+ vpmuludq %xmm7,%xmm4,%xmm4+ vpaddq %xmm4,%xmm10,%xmm10++ vpmuludq %xmm5,%xmm2,%xmm2+ vpaddq %xmm2,%xmm14,%xmm14+ vpmuludq %xmm9,%xmm3,%xmm1+ vpaddq %xmm1,%xmm13,%xmm13+ vpmuludq %xmm8,%xmm3,%xmm0+ vpaddq %xmm0,%xmm12,%xmm12+ vpmuludq %xmm7,%xmm3,%xmm1+ vpaddq %xmm1,%xmm11,%xmm11+ vpmuludq %xmm6,%xmm3,%xmm3+ vpaddq %xmm3,%xmm10,%xmm10++ jz L$short_tail_avx++ vmovdqu 0(%rsi),%xmm0+ vmovdqu 16(%rsi),%xmm1++ vpsrldq $6,%xmm0,%xmm2+ vpsrldq $6,%xmm1,%xmm3+ vpunpckhqdq %xmm1,%xmm0,%xmm4+ vpunpcklqdq %xmm1,%xmm0,%xmm0+ vpunpcklqdq %xmm3,%xmm2,%xmm3++ vpsrlq $40,%xmm4,%xmm4+ vpsrlq $26,%xmm0,%xmm1+ vpand %xmm15,%xmm0,%xmm0+ vpsrlq $4,%xmm3,%xmm2+ vpand %xmm15,%xmm1,%xmm1+ vpsrlq $30,%xmm3,%xmm3+ vpand %xmm15,%xmm2,%xmm2+ vpand %xmm15,%xmm3,%xmm3+ vpor 32(%rcx),%xmm4,%xmm4++ vpshufd $0x32,-64(%rdi),%xmm9+ vpaddq 0(%r11),%xmm0,%xmm0+ vpaddq 16(%r11),%xmm1,%xmm1+ vpaddq 32(%r11),%xmm2,%xmm2+ vpaddq 48(%r11),%xmm3,%xmm3+ vpaddq 64(%r11),%xmm4,%xmm4+++++ vpmuludq %xmm0,%xmm9,%xmm5+ vpaddq %xmm5,%xmm10,%xmm10+ vpmuludq %xmm1,%xmm9,%xmm6+ vpaddq %xmm6,%xmm11,%xmm11+ vpmuludq %xmm2,%xmm9,%xmm5+ vpaddq %xmm5,%xmm12,%xmm12+ vpshufd $0x32,-48(%rdi),%xmm7+ vpmuludq %xmm3,%xmm9,%xmm6+ vpaddq %xmm6,%xmm13,%xmm13+ vpmuludq %xmm4,%xmm9,%xmm9+ vpaddq %xmm9,%xmm14,%xmm14++ vpmuludq %xmm3,%xmm7,%xmm5+ vpaddq %xmm5,%xmm14,%xmm14+ vpshufd $0x32,-32(%rdi),%xmm8+ vpmuludq %xmm2,%xmm7,%xmm6+ vpaddq %xmm6,%xmm13,%xmm13+ vpshufd $0x32,-16(%rdi),%xmm9+ vpmuludq %xmm1,%xmm7,%xmm5+ vpaddq %xmm5,%xmm12,%xmm12+ vpmuludq %xmm0,%xmm7,%xmm7+ vpaddq %xmm7,%xmm11,%xmm11+ vpmuludq %xmm4,%xmm8,%xmm8+ vpaddq %xmm8,%xmm10,%xmm10++ vpshufd $0x32,0(%rdi),%xmm7+ vpmuludq %xmm2,%xmm9,%xmm6+ vpaddq %xmm6,%xmm14,%xmm14+ vpmuludq %xmm1,%xmm9,%xmm5+ vpaddq %xmm5,%xmm13,%xmm13+ vpshufd $0x32,16(%rdi),%xmm8+ vpmuludq %xmm0,%xmm9,%xmm9+ vpaddq %xmm9,%xmm12,%xmm12+ vpmuludq %xmm4,%xmm7,%xmm6+ vpaddq %xmm6,%xmm11,%xmm11+ vpshufd $0x32,32(%rdi),%xmm9+ vpmuludq %xmm3,%xmm7,%xmm7+ vpaddq %xmm7,%xmm10,%xmm10++ vpmuludq %xmm1,%xmm8,%xmm5+ vpaddq %xmm5,%xmm14,%xmm14+ vpmuludq %xmm0,%xmm8,%xmm8+ vpaddq %xmm8,%xmm13,%xmm13+ vpshufd $0x32,48(%rdi),%xmm7+ vpmuludq %xmm4,%xmm9,%xmm6+ vpaddq %xmm6,%xmm12,%xmm12+ vpshufd $0x32,64(%rdi),%xmm8+ vpmuludq %xmm3,%xmm9,%xmm5+ vpaddq %xmm5,%xmm11,%xmm11+ vpmuludq %xmm2,%xmm9,%xmm9+ vpaddq %xmm9,%xmm10,%xmm10++ vpmuludq %xmm0,%xmm7,%xmm7+ vpaddq %xmm7,%xmm14,%xmm14+ vpmuludq %xmm4,%xmm8,%xmm6+ vpaddq %xmm6,%xmm13,%xmm13+ vpmuludq %xmm3,%xmm8,%xmm5+ vpaddq %xmm5,%xmm12,%xmm12+ vpmuludq %xmm2,%xmm8,%xmm6+ vpaddq %xmm6,%xmm11,%xmm11+ vpmuludq %xmm1,%xmm8,%xmm8+ vpaddq %xmm8,%xmm10,%xmm10++L$short_tail_avx:++++ vpsrldq $8,%xmm14,%xmm9+ vpsrldq $8,%xmm13,%xmm8+ vpsrldq $8,%xmm11,%xmm6+ vpsrldq $8,%xmm10,%xmm5+ vpsrldq $8,%xmm12,%xmm7+ vpaddq %xmm8,%xmm13,%xmm13+ vpaddq %xmm9,%xmm14,%xmm14+ vpaddq %xmm5,%xmm10,%xmm10+ vpaddq %xmm6,%xmm11,%xmm11+ vpaddq %xmm7,%xmm12,%xmm12+++++ vpsrlq $26,%xmm13,%xmm3+ vpand %xmm15,%xmm13,%xmm13+ vpaddq %xmm3,%xmm14,%xmm14++ vpsrlq $26,%xmm10,%xmm0+ vpand %xmm15,%xmm10,%xmm10+ vpaddq %xmm0,%xmm11,%xmm11++ vpsrlq $26,%xmm14,%xmm4+ vpand %xmm15,%xmm14,%xmm14++ vpsrlq $26,%xmm11,%xmm1+ vpand %xmm15,%xmm11,%xmm11+ vpaddq %xmm1,%xmm12,%xmm12++ vpaddq %xmm4,%xmm10,%xmm10+ vpsllq $2,%xmm4,%xmm4+ vpaddq %xmm4,%xmm10,%xmm10++ vpsrlq $26,%xmm12,%xmm2+ vpand %xmm15,%xmm12,%xmm12+ vpaddq %xmm2,%xmm13,%xmm13++ vpsrlq $26,%xmm10,%xmm0+ vpand %xmm15,%xmm10,%xmm10+ vpaddq %xmm0,%xmm11,%xmm11++ vpsrlq $26,%xmm13,%xmm3+ vpand %xmm15,%xmm13,%xmm13+ vpaddq %xmm3,%xmm14,%xmm14++ vmovd %xmm10,-112(%rdi)+ vmovd %xmm11,-108(%rdi)+ vmovd %xmm12,-104(%rdi)+ vmovd %xmm13,-100(%rdi)+ vmovd %xmm14,-96(%rdi)+ leaq 88(%r11),%rsp+.cfi_def_cfa %rsp,8+ vzeroupper+ .byte 0xf3,0xc3+.cfi_endproc +++.p2align 5+crypton_poly1305_asm_blocks_avx2:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa+++ movl 20(%rdi),%r8d+ cmpq $128,%rdx+ jb L$blocks++ andq $-16,%rdx++ vzeroupper++ testl %r8d,%r8d+ jz L$base2_64_avx2++ testq $63,%rdx+ jz L$even_avx2++ pushq %rbx+.cfi_adjust_cfa_offset 8+.cfi_offset %rbx,-16+ pushq %rbp+.cfi_adjust_cfa_offset 8+.cfi_offset %rbp,-24+ pushq %r12+.cfi_adjust_cfa_offset 8+.cfi_offset %r12,-32+ pushq %r13+.cfi_adjust_cfa_offset 8+.cfi_offset %r13,-40+ pushq %r14+.cfi_adjust_cfa_offset 8+.cfi_offset %r14,-48+ pushq %r15+.cfi_adjust_cfa_offset 8+.cfi_offset %r15,-56+ leaq -8(%rsp),%rsp+.cfi_adjust_cfa_offset 8+L$blocks_avx2_body:++ movq %rdx,%r15++ movq 0(%rdi),%r8+ movq 8(%rdi),%r9+ movl 16(%rdi),%ebp++ movq 24(%rdi),%r11+ movq 32(%rdi),%r13+++ movl %r8d,%r14d+ andq $-2147483648,%r8+ movq %r9,%r12+ movl %r9d,%ebx+ andq $-2147483648,%r9++ shrq $6,%r8+ shlq $52,%r12+ addq %r8,%r14+ shrq $12,%rbx+ shrq $18,%r9+ addq %r12,%r14+ adcq %r9,%rbx++ movq %rbp,%r8+ shlq $40,%r8+ shrq $24,%rbp+ addq %r8,%rbx+ adcq $0,%rbp++ movq %r13,%r12+ movq %r13,%rax+ shrq $2,%r13+ addq %r12,%r13++L$base2_26_pre_avx2:+ addq 0(%rsi),%r14+ adcq 8(%rsi),%rbx+ leaq 16(%rsi),%rsi+ adcq %rcx,%rbp+ subq $16,%r15++ call __crypton_poly1305_asm_block+ movq %r12,%rax++ testq $63,%r15+ jnz L$base2_26_pre_avx2+++ movq %r14,%rax+ movq %r14,%rdx+ shrq $52,%r14+ movq %rbx,%r11+ movq %rbx,%r12+ shrq $26,%rdx+ andq $0x3ffffff,%rax+ shlq $12,%r11+ andq $0x3ffffff,%rdx+ shrq $14,%rbx+ orq %r11,%r14+ shlq $24,%rbp+ andq $0x3ffffff,%r14+ shrq $40,%r12+ andq $0x3ffffff,%rbx+ orq %r12,%rbp++ vmovd %eax,%xmm0+ vmovd %edx,%xmm1+ vmovd %r14d,%xmm2+ vmovd %ebx,%xmm3+ vmovd %ebp,%xmm4++ movq %r15,%rdx++ movq 8(%rsp),%r15+.cfi_restore %r15+ movq 16(%rsp),%r14+.cfi_restore %r14+ movq 24(%rsp),%r13+.cfi_restore %r13+ movq 32(%rsp),%r12+.cfi_restore %r12+ movq 40(%rsp),%rbp+.cfi_restore %rbp+ movq 48(%rsp),%rbx+.cfi_restore %rbx+ leaq 56(%rsp),%rax+ leaq 56(%rsp),%rsp+.cfi_adjust_cfa_offset -56+L$blocks_avx2_epilogue:+ jmp L$do_avx2+.cfi_endproc ++.p2align 5+L$base2_64_avx2:+.cfi_startproc + pushq %rbx+.cfi_adjust_cfa_offset 8+.cfi_offset %rbx,-16+ pushq %rbp+.cfi_adjust_cfa_offset 8+.cfi_offset %rbp,-24+ pushq %r12+.cfi_adjust_cfa_offset 8+.cfi_offset %r12,-32+ pushq %r13+.cfi_adjust_cfa_offset 8+.cfi_offset %r13,-40+ pushq %r14+.cfi_adjust_cfa_offset 8+.cfi_offset %r14,-48+ pushq %r15+.cfi_adjust_cfa_offset 8+.cfi_offset %r15,-56+ leaq -8(%rsp),%rsp+.cfi_adjust_cfa_offset 8+L$base2_64_avx2_body:++ movq %rdx,%r15++ movq 24(%rdi),%r11+ movq 32(%rdi),%r13++ movq 0(%rdi),%r14+ movq 8(%rdi),%rbx+ movl 16(%rdi),%ebp++ movq %r13,%r12+ movq %r13,%rax+ shrq $2,%r13+ addq %r12,%r13++ testq $63,%rdx+ jz L$init_avx2++L$base2_64_pre_avx2:+ addq 0(%rsi),%r14+ adcq 8(%rsi),%rbx+ leaq 16(%rsi),%rsi+ adcq %rcx,%rbp+ subq $16,%r15++ call __crypton_poly1305_asm_block+ movq %r12,%rax++ testq $63,%r15+ jnz L$base2_64_pre_avx2++L$init_avx2:++ movq %r14,%rax+ movq %r14,%rdx+ shrq $52,%r14+ movq %rbx,%r8+ movq %rbx,%r9+ shrq $26,%rdx+ andq $0x3ffffff,%rax+ shlq $12,%r8+ andq $0x3ffffff,%rdx+ shrq $14,%rbx+ orq %r8,%r14+ shlq $24,%rbp+ andq $0x3ffffff,%r14+ shrq $40,%r9+ andq $0x3ffffff,%rbx+ orq %r9,%rbp++ vmovd %eax,%xmm0+ vmovd %edx,%xmm1+ vmovd %r14d,%xmm2+ vmovd %ebx,%xmm3+ vmovd %ebp,%xmm4+ movl $1,20(%rdi)++ call __crypton_poly1305_asm_init_avx++ movq %r15,%rdx++ movq 8(%rsp),%r15+.cfi_restore %r15+ movq 16(%rsp),%r14+.cfi_restore %r14+ movq 24(%rsp),%r13+.cfi_restore %r13+ movq 32(%rsp),%r12+.cfi_restore %r12+ movq 40(%rsp),%rbp+.cfi_restore %rbp+ movq 48(%rsp),%rbx+.cfi_restore %rbx+ leaq 56(%rsp),%rax+ leaq 56(%rsp),%rsp+.cfi_adjust_cfa_offset -56+L$base2_64_avx2_epilogue:+ jmp L$do_avx2+.cfi_endproc ++.p2align 5+L$even_avx2:+.cfi_startproc + vmovd 0(%rdi),%xmm0+ vmovd 4(%rdi),%xmm1+ vmovd 8(%rdi),%xmm2+ vmovd 12(%rdi),%xmm3+ vmovd 16(%rdi),%xmm4++L$do_avx2:+ leaq -8(%rsp),%r11+.cfi_def_cfa %r11,16+ subq $0x128,%rsp+ leaq L$const(%rip),%rcx+ leaq 48+64(%rdi),%rdi+ vmovdqa 96(%rcx),%ymm7+++ vmovdqu -64(%rdi),%xmm9+ andq $-512,%rsp+ vmovdqu -48(%rdi),%xmm10+ vmovdqu -32(%rdi),%xmm6+ vmovdqu -16(%rdi),%xmm11+ vmovdqu 0(%rdi),%xmm12+ vmovdqu 16(%rdi),%xmm13+ leaq 144(%rsp),%rax+ vmovdqu 32(%rdi),%xmm14+ vpermd %ymm9,%ymm7,%ymm9+ vmovdqu 48(%rdi),%xmm15+ vpermd %ymm10,%ymm7,%ymm10+ vmovdqu 64(%rdi),%xmm5+ vpermd %ymm6,%ymm7,%ymm6+ vmovdqa %ymm9,0(%rsp)+ vpermd %ymm11,%ymm7,%ymm11+ vmovdqa %ymm10,32-144(%rax)+ vpermd %ymm12,%ymm7,%ymm12+ vmovdqa %ymm6,64-144(%rax)+ vpermd %ymm13,%ymm7,%ymm13+ vmovdqa %ymm11,96-144(%rax)+ vpermd %ymm14,%ymm7,%ymm14+ vmovdqa %ymm12,128-144(%rax)+ vpermd %ymm15,%ymm7,%ymm15+ vmovdqa %ymm13,160-144(%rax)+ vpermd %ymm5,%ymm7,%ymm5+ vmovdqa %ymm14,192-144(%rax)+ vmovdqa %ymm15,224-144(%rax)+ vmovdqa %ymm5,256-144(%rax)+ vmovdqa 64(%rcx),%ymm5++++ vmovdqu 0(%rsi),%xmm7+ vmovdqu 16(%rsi),%xmm8+ vinserti128 $1,32(%rsi),%ymm7,%ymm7+ vinserti128 $1,48(%rsi),%ymm8,%ymm8+ leaq 64(%rsi),%rsi++ vpsrldq $6,%ymm7,%ymm9+ vpsrldq $6,%ymm8,%ymm10+ vpunpckhqdq %ymm8,%ymm7,%ymm6+ vpunpcklqdq %ymm10,%ymm9,%ymm9+ vpunpcklqdq %ymm8,%ymm7,%ymm7++ vpsrlq $30,%ymm9,%ymm10+ vpsrlq $4,%ymm9,%ymm9+ vpsrlq $26,%ymm7,%ymm8+ vpsrlq $40,%ymm6,%ymm6+ vpand %ymm5,%ymm9,%ymm9+ vpand %ymm5,%ymm7,%ymm7+ vpand %ymm5,%ymm8,%ymm8+ vpand %ymm5,%ymm10,%ymm10+ vpor 32(%rcx),%ymm6,%ymm6++ vpaddq %ymm2,%ymm9,%ymm2+ subq $64,%rdx+ jz L$tail_avx2+ jmp L$oop_avx2++.p2align 5+L$oop_avx2:+++++++++ vpaddq %ymm0,%ymm7,%ymm0+ vmovdqa 0(%rsp),%ymm7+ vpaddq %ymm1,%ymm8,%ymm1+ vmovdqa 32(%rsp),%ymm8+ vpaddq %ymm3,%ymm10,%ymm3+ vmovdqa 96(%rsp),%ymm9+ vpaddq %ymm4,%ymm6,%ymm4+ vmovdqa 48(%rax),%ymm10+ vmovdqa 112(%rax),%ymm5+++++++++++++++++ vpmuludq %ymm2,%ymm7,%ymm13+ vpmuludq %ymm2,%ymm8,%ymm14+ vpmuludq %ymm2,%ymm9,%ymm15+ vpmuludq %ymm2,%ymm10,%ymm11+ vpmuludq %ymm2,%ymm5,%ymm12++ vpmuludq %ymm0,%ymm8,%ymm6+ vpmuludq %ymm1,%ymm8,%ymm2+ vpaddq %ymm6,%ymm12,%ymm12+ vpaddq %ymm2,%ymm13,%ymm13+ vpmuludq %ymm3,%ymm8,%ymm6+ vpmuludq 64(%rsp),%ymm4,%ymm2+ vpaddq %ymm6,%ymm15,%ymm15+ vpaddq %ymm2,%ymm11,%ymm11+ vmovdqa -16(%rax),%ymm8++ vpmuludq %ymm0,%ymm7,%ymm6+ vpmuludq %ymm1,%ymm7,%ymm2+ vpaddq %ymm6,%ymm11,%ymm11+ vpaddq %ymm2,%ymm12,%ymm12+ vpmuludq %ymm3,%ymm7,%ymm6+ vpmuludq %ymm4,%ymm7,%ymm2+ vmovdqu 0(%rsi),%xmm7+ vpaddq %ymm6,%ymm14,%ymm14+ vpaddq %ymm2,%ymm15,%ymm15+ vinserti128 $1,32(%rsi),%ymm7,%ymm7++ vpmuludq %ymm3,%ymm8,%ymm6+ vpmuludq %ymm4,%ymm8,%ymm2+ vmovdqu 16(%rsi),%xmm8+ vpaddq %ymm6,%ymm11,%ymm11+ vpaddq %ymm2,%ymm12,%ymm12+ vmovdqa 16(%rax),%ymm2+ vpmuludq %ymm1,%ymm9,%ymm6+ vpmuludq %ymm0,%ymm9,%ymm9+ vpaddq %ymm6,%ymm14,%ymm14+ vpaddq %ymm9,%ymm13,%ymm13+ vinserti128 $1,48(%rsi),%ymm8,%ymm8+ leaq 64(%rsi),%rsi++ vpmuludq %ymm1,%ymm2,%ymm6+ vpmuludq %ymm0,%ymm2,%ymm2+ vpsrldq $6,%ymm7,%ymm9+ vpaddq %ymm6,%ymm15,%ymm15+ vpaddq %ymm2,%ymm14,%ymm14+ vpmuludq %ymm3,%ymm10,%ymm6+ vpmuludq %ymm4,%ymm10,%ymm2+ vpsrldq $6,%ymm8,%ymm10+ vpaddq %ymm6,%ymm12,%ymm12+ vpaddq %ymm2,%ymm13,%ymm13+ vpunpckhqdq %ymm8,%ymm7,%ymm6++ vpmuludq %ymm3,%ymm5,%ymm3+ vpmuludq %ymm4,%ymm5,%ymm4+ vpunpcklqdq %ymm8,%ymm7,%ymm7+ vpaddq %ymm3,%ymm13,%ymm2+ vpaddq %ymm4,%ymm14,%ymm3+ vpunpcklqdq %ymm10,%ymm9,%ymm10+ vpmuludq 80(%rax),%ymm0,%ymm4+ vpmuludq %ymm1,%ymm5,%ymm0+ vmovdqa 64(%rcx),%ymm5+ vpaddq %ymm4,%ymm15,%ymm4+ vpaddq %ymm0,%ymm11,%ymm0+++++ vpsrlq $26,%ymm3,%ymm14+ vpand %ymm5,%ymm3,%ymm3+ vpaddq %ymm14,%ymm4,%ymm4++ vpsrlq $26,%ymm0,%ymm11+ vpand %ymm5,%ymm0,%ymm0+ vpaddq %ymm11,%ymm12,%ymm1++ vpsrlq $26,%ymm4,%ymm15+ vpand %ymm5,%ymm4,%ymm4++ vpsrlq $4,%ymm10,%ymm9++ vpsrlq $26,%ymm1,%ymm12+ vpand %ymm5,%ymm1,%ymm1+ vpaddq %ymm12,%ymm2,%ymm2++ vpaddq %ymm15,%ymm0,%ymm0+ vpsllq $2,%ymm15,%ymm15+ vpaddq %ymm15,%ymm0,%ymm0++ vpand %ymm5,%ymm9,%ymm9+ vpsrlq $26,%ymm7,%ymm8++ vpsrlq $26,%ymm2,%ymm13+ vpand %ymm5,%ymm2,%ymm2+ vpaddq %ymm13,%ymm3,%ymm3++ vpaddq %ymm9,%ymm2,%ymm2+ vpsrlq $30,%ymm10,%ymm10++ vpsrlq $26,%ymm0,%ymm11+ vpand %ymm5,%ymm0,%ymm0+ vpaddq %ymm11,%ymm1,%ymm1++ vpsrlq $40,%ymm6,%ymm6++ vpsrlq $26,%ymm3,%ymm14+ vpand %ymm5,%ymm3,%ymm3+ vpaddq %ymm14,%ymm4,%ymm4++ vpand %ymm5,%ymm7,%ymm7+ vpand %ymm5,%ymm8,%ymm8+ vpand %ymm5,%ymm10,%ymm10+ vpor 32(%rcx),%ymm6,%ymm6++ subq $64,%rdx+ jnz L$oop_avx2++.byte 0x66,0x90+L$tail_avx2:++++++++ vpaddq %ymm0,%ymm7,%ymm0+ vmovdqu 4(%rsp),%ymm7+ vpaddq %ymm1,%ymm8,%ymm1+ vmovdqu 36(%rsp),%ymm8+ vpaddq %ymm3,%ymm10,%ymm3+ vmovdqu 100(%rsp),%ymm9+ vpaddq %ymm4,%ymm6,%ymm4+ vmovdqu 52(%rax),%ymm10+ vmovdqu 116(%rax),%ymm5++ vpmuludq %ymm2,%ymm7,%ymm13+ vpmuludq %ymm2,%ymm8,%ymm14+ vpmuludq %ymm2,%ymm9,%ymm15+ vpmuludq %ymm2,%ymm10,%ymm11+ vpmuludq %ymm2,%ymm5,%ymm12++ vpmuludq %ymm0,%ymm8,%ymm6+ vpmuludq %ymm1,%ymm8,%ymm2+ vpaddq %ymm6,%ymm12,%ymm12+ vpaddq %ymm2,%ymm13,%ymm13+ vpmuludq %ymm3,%ymm8,%ymm6+ vpmuludq 68(%rsp),%ymm4,%ymm2+ vpaddq %ymm6,%ymm15,%ymm15+ vpaddq %ymm2,%ymm11,%ymm11++ vpmuludq %ymm0,%ymm7,%ymm6+ vpmuludq %ymm1,%ymm7,%ymm2+ vpaddq %ymm6,%ymm11,%ymm11+ vmovdqu -12(%rax),%ymm8+ vpaddq %ymm2,%ymm12,%ymm12+ vpmuludq %ymm3,%ymm7,%ymm6+ vpmuludq %ymm4,%ymm7,%ymm2+ vpaddq %ymm6,%ymm14,%ymm14+ vpaddq %ymm2,%ymm15,%ymm15++ vpmuludq %ymm3,%ymm8,%ymm6+ vpmuludq %ymm4,%ymm8,%ymm2+ vpaddq %ymm6,%ymm11,%ymm11+ vpaddq %ymm2,%ymm12,%ymm12+ vmovdqu 20(%rax),%ymm2+ vpmuludq %ymm1,%ymm9,%ymm6+ vpmuludq %ymm0,%ymm9,%ymm9+ vpaddq %ymm6,%ymm14,%ymm14+ vpaddq %ymm9,%ymm13,%ymm13++ vpmuludq %ymm1,%ymm2,%ymm6+ vpmuludq %ymm0,%ymm2,%ymm2+ vpaddq %ymm6,%ymm15,%ymm15+ vpaddq %ymm2,%ymm14,%ymm14+ vpmuludq %ymm3,%ymm10,%ymm6+ vpmuludq %ymm4,%ymm10,%ymm2+ vpaddq %ymm6,%ymm12,%ymm12+ vpaddq %ymm2,%ymm13,%ymm13++ vpmuludq %ymm3,%ymm5,%ymm3+ vpmuludq %ymm4,%ymm5,%ymm4+ vpaddq %ymm3,%ymm13,%ymm2+ vpaddq %ymm4,%ymm14,%ymm3+ vpmuludq 84(%rax),%ymm0,%ymm4+ vpmuludq %ymm1,%ymm5,%ymm0+ vmovdqa 64(%rcx),%ymm5+ vpaddq %ymm4,%ymm15,%ymm4+ vpaddq %ymm0,%ymm11,%ymm0+++++ vpsrldq $8,%ymm12,%ymm8+ vpsrldq $8,%ymm2,%ymm9+ vpsrldq $8,%ymm3,%ymm10+ vpsrldq $8,%ymm4,%ymm6+ vpsrldq $8,%ymm0,%ymm7+ vpaddq %ymm8,%ymm12,%ymm12+ vpaddq %ymm9,%ymm2,%ymm2+ vpaddq %ymm10,%ymm3,%ymm3+ vpaddq %ymm6,%ymm4,%ymm4+ vpaddq %ymm7,%ymm0,%ymm0++ vpermq $0x2,%ymm3,%ymm10+ vpermq $0x2,%ymm4,%ymm6+ vpermq $0x2,%ymm0,%ymm7+ vpermq $0x2,%ymm12,%ymm8+ vpermq $0x2,%ymm2,%ymm9+ vpaddq %ymm10,%ymm3,%ymm3+ vpaddq %ymm6,%ymm4,%ymm4+ vpaddq %ymm7,%ymm0,%ymm0+ vpaddq %ymm8,%ymm12,%ymm12+ vpaddq %ymm9,%ymm2,%ymm2+++++ vpsrlq $26,%ymm3,%ymm14+ vpand %ymm5,%ymm3,%ymm3+ vpaddq %ymm14,%ymm4,%ymm4++ vpsrlq $26,%ymm0,%ymm11+ vpand %ymm5,%ymm0,%ymm0+ vpaddq %ymm11,%ymm12,%ymm1++ vpsrlq $26,%ymm4,%ymm15+ vpand %ymm5,%ymm4,%ymm4++ vpsrlq $26,%ymm1,%ymm12+ vpand %ymm5,%ymm1,%ymm1+ vpaddq %ymm12,%ymm2,%ymm2++ vpaddq %ymm15,%ymm0,%ymm0+ vpsllq $2,%ymm15,%ymm15+ vpaddq %ymm15,%ymm0,%ymm0++ vpsrlq $26,%ymm2,%ymm13+ vpand %ymm5,%ymm2,%ymm2+ vpaddq %ymm13,%ymm3,%ymm3++ vpsrlq $26,%ymm0,%ymm11+ vpand %ymm5,%ymm0,%ymm0+ vpaddq %ymm11,%ymm1,%ymm1++ vpsrlq $26,%ymm3,%ymm14+ vpand %ymm5,%ymm3,%ymm3+ vpaddq %ymm14,%ymm4,%ymm4++ vmovd %xmm0,-112(%rdi)+ vmovd %xmm1,-108(%rdi)+ vmovd %xmm2,-104(%rdi)+ vmovd %xmm3,-100(%rdi)+ vmovd %xmm4,-96(%rdi)+ leaq 8(%r11),%rsp+.cfi_def_cfa %rsp,8+ vzeroupper+ .byte 0xf3,0xc3+.cfi_endproc ++.p2align 6+L$const:+L$mask24:+.long 0x0ffffff,0,0x0ffffff,0,0x0ffffff,0,0x0ffffff,0+L$129:+.long 16777216,0,16777216,0,16777216,0,16777216,0+L$mask26:+.long 0x3ffffff,0,0x3ffffff,0,0x3ffffff,0,0x3ffffff,0+L$permd_avx2:+.long 2,2,2,3,2,0,2,1+L$permd_avx512:+.long 0,0,0,1, 0,2,0,3, 0,4,0,5, 0,6,0,7++L$2_44_inp_permd:+.long 0,1,1,2,2,3,7,7+L$2_44_inp_shift:+.quad 0,12,24,64+L$2_44_mask:+.quad 0xfffffffffff,0xfffffffffff,0x3ffffffffff,0xffffffffffffffff+L$2_44_shift_rgt:+.quad 44,44,42,64+L$2_44_shift_lft:+.quad 8,8,10,64++.p2align 6+L$x_mask44:+.quad 0xfffffffffff,0xfffffffffff,0xfffffffffff,0xfffffffffff+.quad 0xfffffffffff,0xfffffffffff,0xfffffffffff,0xfffffffffff+L$x_mask42:+.quad 0x3ffffffffff,0x3ffffffffff,0x3ffffffffff,0x3ffffffffff+.quad 0x3ffffffffff,0x3ffffffffff,0x3ffffffffff,0x3ffffffffff+.byte 80,111,108,121,49,51,48,53,32,102,111,114,32,120,56,54,95,54,52,44,32,67,82,89,80,84,79,71,65,77,83,32,98,121,32,64,100,111,116,45,97,115,109,0+.p2align 4+.globl _crypton_xor128_encrypt_n_pad++.p2align 4+_crypton_xor128_encrypt_n_pad:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa++ subq %rdx,%rsi+ subq %rdx,%rdi+ movq %rcx,%r10+ shrq $4,%rcx+ jz L$tail_enc+ nop+L$oop_enc_xmm:+ movdqu (%rsi,%rdx,1),%xmm0+ pxor (%rdx),%xmm0+ movdqu %xmm0,(%rdi,%rdx,1)+ movdqa %xmm0,(%rdx)+ leaq 16(%rdx),%rdx+ decq %rcx+ jnz L$oop_enc_xmm++ andq $15,%r10+ jz L$done_enc++L$tail_enc:+ movq $16,%rcx+ subq %r10,%rcx+ xorl %eax,%eax+L$oop_enc_byte:+ movb (%rsi,%rdx,1),%al+ xorb (%rdx),%al+ movb %al,(%rdi,%rdx,1)+ movb %al,(%rdx)+ leaq 1(%rdx),%rdx+ decq %r10+ jnz L$oop_enc_byte++ xorl %eax,%eax+L$oop_enc_pad:+ movb %al,(%rdx)+ leaq 1(%rdx),%rdx+ decq %rcx+ jnz L$oop_enc_pad++L$done_enc:+ movq %rdx,%rax+ .byte 0xf3,0xc3+.cfi_endproc+++.globl _crypton_xor128_decrypt_n_pad++.p2align 4+_crypton_xor128_decrypt_n_pad:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa++ subq %rdx,%rsi+ subq %rdx,%rdi+ movq %rcx,%r10+ shrq $4,%rcx+ jz L$tail_dec+ nop+L$oop_dec_xmm:+ movdqu (%rsi,%rdx,1),%xmm0+ movdqa (%rdx),%xmm1+ pxor %xmm0,%xmm1+ movdqu %xmm1,(%rdi,%rdx,1)+ movdqa %xmm0,(%rdx)+ leaq 16(%rdx),%rdx+ decq %rcx+ jnz L$oop_dec_xmm++ pxor %xmm1,%xmm1+ andq $15,%r10+ jz L$done_dec++L$tail_dec:+ movq $16,%rcx+ subq %r10,%rcx+ xorl %eax,%eax+ xorq %r11,%r11+L$oop_dec_byte:+ movb (%rsi,%rdx,1),%r11b+ movb (%rdx),%al+ xorb %r11b,%al+ movb %al,(%rdi,%rdx,1)+ movb %r11b,(%rdx)+ leaq 1(%rdx),%rdx+ decq %r10+ jnz L$oop_dec_byte++ xorl %eax,%eax+L$oop_dec_pad:+ movb %al,(%rdx)+ leaq 1(%rdx),%rdx+ decq %rcx+ jnz L$oop_dec_pad++L$done_dec:+ movq %rdx,%rax+ .byte 0xf3,0xc3+.cfi_endproc+
@@ -0,0 +1,2281 @@+.text ++++.globl crypton_poly1305_asm_init++.globl crypton_poly1305_asm_blocks++.globl crypton_poly1305_asm_emit+++.def crypton_poly1305_asm_init; .scl 2; .type 32; .endef+.p2align 5+crypton_poly1305_asm_init:+ .byte 0xf3,0x0f,0x1e,0xfa+ movq %rdi,8(%rsp)+ movq %rsi,16(%rsp)+ movq %rsp,%rax+.LSEH_begin_crypton_poly1305_asm_init:++ movq %rcx,%rdi+ movq %rdx,%rsi+ movq %r8,%rdx+ xorq %rax,%rax+ movq %rax,0(%rdi)+ movq %rax,8(%rdi)+ movq %rax,16(%rdi)++ cmpq $0,%rsi+ je .Lno_key++ movq $0x0ffffffc0fffffff,%rax+ leaq -3(%rax),%rcx+ andq 0(%rsi),%rax+ andq 8(%rsi),%rcx+ movq %rax,24(%rdi)+ movq %rcx,32(%rdi)+ movl $-1,48(%rdi)+ leaq crypton_poly1305_asm_blocks(%rip),%r10+ leaq crypton_poly1305_asm_emit(%rip),%r11+ movq crypton_ia32cap_P+4(%rip),%r9+ leaq crypton_poly1305_asm_blocks_avx(%rip),%rax+ btq $28,%r9+ cmovcq %rax,%r10+ leaq crypton_poly1305_asm_blocks_avx2(%rip),%rax+ btq $37,%r9+ cmovcq %rax,%r10+ movq %r10,0(%rdx)+ movq %r11,8(%rdx)+ movl $1,%eax+.Lno_key:+ movq 8(%rsp),%rdi+ movq 16(%rsp),%rsi+ .byte 0xf3,0xc3+.LSEH_end_crypton_poly1305_asm_init:++.def crypton_poly1305_asm_blocks; .scl 2; .type 32; .endef+.p2align 5+crypton_poly1305_asm_blocks:+ .byte 0xf3,0x0f,0x1e,0xfa+ movq %rdi,8(%rsp)+ movq %rsi,16(%rsp)+ movq %rsp,%rax+.LSEH_begin_crypton_poly1305_asm_blocks:+++ movq %rcx,%rdi+ movq %rdx,%rsi+ movq %r8,%rdx+ movq %r9,%rcx+.Lblocks:+ shrq $4,%rdx+ jz .Lno_data++ pushq %rbx++ pushq %rbp++ pushq %r12++ pushq %r13++ pushq %r14++ pushq %r15++ leaq -8(%rsp),%rsp++.Lblocks_body:++ movq %rdx,%r15++ movq 24(%rdi),%r11+ movq 32(%rdi),%r13++ movq 0(%rdi),%r14+ movq 8(%rdi),%rbx+ movq 16(%rdi),%rbp++ movl %r14d,%eax+ movl 4(%rdi),%edx+ movl %ebx,%r8d+ movl 12(%rdi),%r10d+ movl %ebp,%r12d++ shlq $26,%rdx+ movq %r8,%r9+ shlq $52,%r8+ addq %rdx,%rax+ shrq $12,%r9+ addq %rax,%r8+ adcq $0,%r9++ shlq $14,%r10+ movq %r12,%rax+ shrq $24,%r12+ addq %r10,%r9+ shlq $40,%rax+ addq %rax,%r9+ adcq $0,%r12++ cmpq $4,%rbp++ cmovaq %r8,%r14+ cmovaq %r9,%rbx+ cmovaq %r12,%rbp++ movq %r13,%r12+ shrq $2,%r13+ movq %r12,%rax+ addq %r12,%r13+ jmp .Loop++.p2align 5+.Loop:+ addq 0(%rsi),%r14+ adcq 8(%rsi),%rbx+ leaq 16(%rsi),%rsi+ adcq %rcx,%rbp+ mulq %r14+ movq %rax,%r9+ movq %r11,%rax+ movq %rdx,%r10++ mulq %r14+ movq %rax,%r14+ movq %r11,%rax+ movq %rdx,%r8++ mulq %rbx+ addq %rax,%r9+ movq %r13,%rax+ adcq %rdx,%r10++ mulq %rbx+ movq %rbp,%rbx+ addq %rax,%r14+ adcq %rdx,%r8++ imulq %r13,%rbx+ addq %rbx,%r9+ movq %r8,%rbx+ adcq $0,%r10++ imulq %r11,%rbp+ addq %r9,%rbx+ movq $-4,%rax+ adcq %rbp,%r10++ andq %r10,%rax+ movq %r10,%rbp+ shrq $2,%r10+ andq $3,%rbp+ addq %r10,%rax+ addq %rax,%r14+ adcq $0,%rbx+ adcq $0,%rbp+ movq %r12,%rax+ decq %r15+ jnz .Loop++ movq %r14,0(%rdi)+ movq %rbx,8(%rdi)+ movq %rbp,16(%rdi)++ movq 8(%rsp),%r15++ movq 16(%rsp),%r14++ movq 24(%rsp),%r13++ movq 32(%rsp),%r12++ movq 40(%rsp),%rbp++ movq 48(%rsp),%rbx++ leaq 56(%rsp),%rsp++.Lno_data:+.Lblocks_epilogue:+ movq 8(%rsp),%rdi+ movq 16(%rsp),%rsi+ .byte 0xf3,0xc3++.LSEH_end_crypton_poly1305_asm_blocks:++.def crypton_poly1305_asm_emit; .scl 2; .type 32; .endef+.p2align 5+crypton_poly1305_asm_emit:+ .byte 0xf3,0x0f,0x1e,0xfa+ movq %rdi,8(%rsp)+ movq %rsi,16(%rsp)+ movq %rsp,%rax+.LSEH_begin_crypton_poly1305_asm_emit:++ movq %rcx,%rdi+ movq %rdx,%rsi+ movq %r8,%rdx+ movl 0(%rdi),%eax+ movl 4(%rdi),%ecx+ movl 8(%rdi),%r8d+ movl 12(%rdi),%r11d+ movl 16(%rdi),%r10d++ shlq $26,%rcx+ movq %r8,%r9+ shlq $52,%r8+ addq %rcx,%rax+ shrq $12,%r9+ addq %rax,%r8+ adcq $0,%r9++ shlq $14,%r11+ movq %r10,%rax+ shrq $24,%r10+ addq %r11,%r9+ movq 0(%rdi),%rcx+ shlq $40,%rax+ movq 8(%rdi),%r11+ addq %rax,%r9+ movq 16(%rdi),%rax+ adcq $0,%r10++ cmpq $4,%rax++ cmovbeq %rcx,%r8+ cmovbeq %r11,%r9+ cmovbeq %rax,%r10++ movq %r8,%rax+ addq $5,%r8+ movq %r9,%rcx+ adcq $0,%r9+ adcq $0,%r10+ shrq $2,%r10+ cmovnzq %r8,%rax+ cmovnzq %r9,%rcx++ addq 0(%rdx),%rax+ adcq 8(%rdx),%rcx+ movq %rax,0(%rsi)+ movq %rcx,8(%rsi)++ movq 8(%rsp),%rdi+ movq 16(%rsp),%rsi+ .byte 0xf3,0xc3+.LSEH_end_crypton_poly1305_asm_emit:+.def __crypton_poly1305_asm_block; .scl 3; .type 32; .endef+.p2align 5+__crypton_poly1305_asm_block:+ .byte 0xf3,0x0f,0x1e,0xfa++ mulq %r14+ movq %rax,%r9+ movq %r11,%rax+ movq %rdx,%r10++ mulq %r14+ movq %rax,%r14+ movq %r11,%rax+ movq %rdx,%r8++ mulq %rbx+ addq %rax,%r9+ movq %r13,%rax+ adcq %rdx,%r10++ mulq %rbx+ movq %rbp,%rbx+ addq %rax,%r14+ adcq %rdx,%r8++ imulq %r13,%rbx+ addq %rbx,%r9+ movq %r8,%rbx+ adcq $0,%r10++ imulq %r11,%rbp+ addq %r9,%rbx+ movq $-4,%rax+ adcq %rbp,%r10++ andq %r10,%rax+ movq %r10,%rbp+ shrq $2,%r10+ andq $3,%rbp+ addq %r10,%rax+ addq %rax,%r14+ adcq $0,%rbx+ adcq $0,%rbp+ .byte 0xf3,0xc3+++.def __crypton_poly1305_asm_init_avx; .scl 3; .type 32; .endef+.p2align 5+__crypton_poly1305_asm_init_avx:+ .byte 0xf3,0x0f,0x1e,0xfa++ cmpl $-1,48(%rdi)+ jne .Ldone_init_avx++ movq %r11,%r14+ movq %r12,%rbx+ xorq %rbp,%rbp++ leaq 48+64(%rdi),%rdi++ movq %r12,%rax+ call __crypton_poly1305_asm_block++ movl $0x3ffffff,%eax+ movl $0x3ffffff,%edx+ movq %r14,%r8+ andl %r14d,%eax+ movq %r11,%r9+ andl %r11d,%edx+ movl %eax,-64(%rdi)+ shrq $26,%r8+ movl %edx,-60(%rdi)+ shrq $26,%r9++ movl $0x3ffffff,%eax+ movl $0x3ffffff,%edx+ andl %r8d,%eax+ andl %r9d,%edx+ movl %eax,-48(%rdi)+ leal (%rax,%rax,4),%eax+ movl %edx,-44(%rdi)+ leal (%rdx,%rdx,4),%edx+ movl %eax,-32(%rdi)+ shrq $26,%r8+ movl %edx,-28(%rdi)+ shrq $26,%r9++ movq %rbx,%rax+ movq %r12,%rdx+ shlq $12,%rax+ shlq $12,%rdx+ orq %r8,%rax+ orq %r9,%rdx+ andl $0x3ffffff,%eax+ andl $0x3ffffff,%edx+ movl %eax,-16(%rdi)+ leal (%rax,%rax,4),%eax+ movl %edx,-12(%rdi)+ leal (%rdx,%rdx,4),%edx+ movl %eax,0(%rdi)+ movq %rbx,%r8+ movl %edx,4(%rdi)+ movq %r12,%r9++ movl $0x3ffffff,%eax+ movl $0x3ffffff,%edx+ shrq $14,%r8+ shrq $14,%r9+ andl %r8d,%eax+ andl %r9d,%edx+ movl %eax,16(%rdi)+ leal (%rax,%rax,4),%eax+ movl %edx,20(%rdi)+ leal (%rdx,%rdx,4),%edx+ movl %eax,32(%rdi)+ shrq $26,%r8+ movl %edx,36(%rdi)+ shrq $26,%r9++ movq %rbp,%rax+ shlq $24,%rax+ orq %rax,%r8+ movl %r8d,48(%rdi)+ leaq (%r8,%r8,4),%r8+ movl %r9d,52(%rdi)+ leaq (%r9,%r9,4),%r9+ movl %r8d,64(%rdi)+ movl %r9d,68(%rdi)++ movq %r12,%rax+ call __crypton_poly1305_asm_block++ movl $0x3ffffff,%eax+ movq %r14,%r8+ andl %r14d,%eax+ shrq $26,%r8+ movl %eax,-52(%rdi)++ movl $0x3ffffff,%edx+ andl %r8d,%edx+ movl %edx,-36(%rdi)+ leal (%rdx,%rdx,4),%edx+ shrq $26,%r8+ movl %edx,-20(%rdi)++ movq %rbx,%rax+ shlq $12,%rax+ orq %r8,%rax+ andl $0x3ffffff,%eax+ movl %eax,-4(%rdi)+ leal (%rax,%rax,4),%eax+ movq %rbx,%r8+ movl %eax,12(%rdi)++ movl $0x3ffffff,%edx+ shrq $14,%r8+ andl %r8d,%edx+ movl %edx,28(%rdi)+ leal (%rdx,%rdx,4),%edx+ shrq $26,%r8+ movl %edx,44(%rdi)++ movq %rbp,%rax+ shlq $24,%rax+ orq %rax,%r8+ movl %r8d,60(%rdi)+ leaq (%r8,%r8,4),%r8+ movl %r8d,76(%rdi)++ movq %r12,%rax+ call __crypton_poly1305_asm_block++ movl $0x3ffffff,%eax+ movq %r14,%r8+ andl %r14d,%eax+ shrq $26,%r8+ movl %eax,-56(%rdi)++ movl $0x3ffffff,%edx+ andl %r8d,%edx+ movl %edx,-40(%rdi)+ leal (%rdx,%rdx,4),%edx+ shrq $26,%r8+ movl %edx,-24(%rdi)++ movq %rbx,%rax+ shlq $12,%rax+ orq %r8,%rax+ andl $0x3ffffff,%eax+ movl %eax,-8(%rdi)+ leal (%rax,%rax,4),%eax+ movq %rbx,%r8+ movl %eax,8(%rdi)++ movl $0x3ffffff,%edx+ shrq $14,%r8+ andl %r8d,%edx+ movl %edx,24(%rdi)+ leal (%rdx,%rdx,4),%edx+ shrq $26,%r8+ movl %edx,40(%rdi)++ movq %rbp,%rax+ shlq $24,%rax+ orq %rax,%r8+ movl %r8d,56(%rdi)+ leaq (%r8,%r8,4),%r8+ movl %r8d,72(%rdi)++ leaq -48-64(%rdi),%rdi+.Ldone_init_avx:+ .byte 0xf3,0xc3+++.def crypton_poly1305_asm_blocks_avx; .scl 3; .type 32; .endef+.p2align 5+crypton_poly1305_asm_blocks_avx:+ .byte 0xf3,0x0f,0x1e,0xfa+ movq %rdi,8(%rsp)+ movq %rsi,16(%rsp)+ movq %rsp,%rax+.LSEH_begin_crypton_poly1305_asm_blocks_avx:+++ movq %rcx,%rdi+ movq %rdx,%rsi+ movq %r8,%rdx+ movq %r9,%rcx+ movl 20(%rdi),%r8d+ cmpq $128,%rdx+ jb .Lblocks++ andq $-16,%rdx++ vzeroupper++ testl %r8d,%r8d+ jz .Lbase2_64_avx++ testq $31,%rdx+ jz .Leven_avx++ pushq %rbx++ pushq %rbp++ pushq %r12++ pushq %r13++ pushq %r14++ pushq %r15++ leaq -8(%rsp),%rsp++.Lblocks_avx_body:++ movq %rdx,%r15++ movq 0(%rdi),%r8+ movq 8(%rdi),%r9+ movl 16(%rdi),%ebp++ movq 24(%rdi),%r11+ movq 32(%rdi),%r13+++ movl %r8d,%r14d+ andq $-2147483648,%r8+ movq %r9,%r12+ movl %r9d,%ebx+ andq $-2147483648,%r9++ shrq $6,%r8+ shlq $52,%r12+ addq %r8,%r14+ shrq $12,%rbx+ shrq $18,%r9+ addq %r12,%r14+ adcq %r9,%rbx++ movq %rbp,%r8+ shlq $40,%r8+ shrq $24,%rbp+ addq %r8,%rbx+ adcq $0,%rbp++ movq %r13,%r12+ movq %r13,%rax+ shrq $2,%r13+ addq %r12,%r13++ addq 0(%rsi),%r14+ adcq 8(%rsi),%rbx+ leaq 16(%rsi),%rsi+ adcq %rcx,%rbp++ call __crypton_poly1305_asm_block+++ movq %r14,%rax+ movq %r14,%rdx+ shrq $52,%r14+ movq %rbx,%r11+ movq %rbx,%r12+ shrq $26,%rdx+ andq $0x3ffffff,%rax+ shlq $12,%r11+ andq $0x3ffffff,%rdx+ shrq $14,%rbx+ orq %r11,%r14+ shlq $24,%rbp+ andq $0x3ffffff,%r14+ shrq $40,%r12+ andq $0x3ffffff,%rbx+ orq %r12,%rbp++ vmovd %eax,%xmm0+ vmovd %edx,%xmm1+ vmovd %r14d,%xmm2+ vmovd %ebx,%xmm3+ vmovd %ebp,%xmm4++ leaq -16(%r15),%rdx++ movq 8(%rsp),%r15++ movq 16(%rsp),%r14++ movq 24(%rsp),%r13++ movq 32(%rsp),%r12++ movq 40(%rsp),%rbp++ movq 48(%rsp),%rbx++ leaq 56(%rsp),%rax+ leaq 56(%rsp),%rsp++.Lblocks_avx_epilogue:+ jmp .Ldo_avx+++.p2align 5+.Lbase2_64_avx:++ pushq %rbx++ pushq %rbp++ pushq %r12++ pushq %r13++ pushq %r14++ pushq %r15++ leaq -8(%rsp),%rsp++.Lbase2_64_avx_body:++ movq %rdx,%r15++ movq 24(%rdi),%r11+ movq 32(%rdi),%r13++ movq 0(%rdi),%r14+ movq 8(%rdi),%rbx+ movl 16(%rdi),%ebp++ movq %r13,%r12+ movq %r13,%rax+ shrq $2,%r13+ addq %r12,%r13++ testq $31,%rdx+ jz .Linit_avx++ addq 0(%rsi),%r14+ adcq 8(%rsi),%rbx+ leaq 16(%rsi),%rsi+ adcq %rcx,%rbp+ subq $16,%r15++ call __crypton_poly1305_asm_block++.Linit_avx:++ movq %r14,%rax+ movq %r14,%rdx+ shrq $52,%r14+ movq %rbx,%r8+ movq %rbx,%r9+ shrq $26,%rdx+ andq $0x3ffffff,%rax+ shlq $12,%r8+ andq $0x3ffffff,%rdx+ shrq $14,%rbx+ orq %r8,%r14+ shlq $24,%rbp+ andq $0x3ffffff,%r14+ shrq $40,%r9+ andq $0x3ffffff,%rbx+ orq %r9,%rbp++ vmovd %eax,%xmm0+ vmovd %edx,%xmm1+ vmovd %r14d,%xmm2+ vmovd %ebx,%xmm3+ vmovd %ebp,%xmm4+ movl $1,20(%rdi)++ call __crypton_poly1305_asm_init_avx++ movq %r15,%rdx++ movq 8(%rsp),%r15++ movq 16(%rsp),%r14++ movq 24(%rsp),%r13++ movq 32(%rsp),%r12++ movq 40(%rsp),%rbp++ movq 48(%rsp),%rbx++ leaq 56(%rsp),%rax+ leaq 56(%rsp),%rsp++.Lbase2_64_avx_epilogue:+ jmp .Ldo_avx+++.p2align 5+.Leven_avx:++ vmovd 0(%rdi),%xmm0+ vmovd 4(%rdi),%xmm1+ vmovd 8(%rdi),%xmm2+ vmovd 12(%rdi),%xmm3+ vmovd 16(%rdi),%xmm4++.Ldo_avx:+ leaq -248(%rsp),%r11+ subq $0x218,%rsp+ vmovdqa %xmm6,80(%r11)+ vmovdqa %xmm7,96(%r11)+ vmovdqa %xmm8,112(%r11)+ vmovdqa %xmm9,128(%r11)+ vmovdqa %xmm10,144(%r11)+ vmovdqa %xmm11,160(%r11)+ vmovdqa %xmm12,176(%r11)+ vmovdqa %xmm13,192(%r11)+ vmovdqa %xmm14,208(%r11)+ vmovdqa %xmm15,224(%r11)+.Ldo_avx_body:+ subq $64,%rdx+ leaq -32(%rsi),%rax+ cmovcq %rax,%rsi++ vmovdqu 48(%rdi),%xmm14+ leaq 112(%rdi),%rdi+ leaq .Lconst(%rip),%rcx++++ vmovdqu 32(%rsi),%xmm5+ vmovdqu 48(%rsi),%xmm6+ vmovdqa 64(%rcx),%xmm15++ vpsrldq $6,%xmm5,%xmm7+ vpsrldq $6,%xmm6,%xmm8+ vpunpckhqdq %xmm6,%xmm5,%xmm9+ vpunpcklqdq %xmm6,%xmm5,%xmm5+ vpunpcklqdq %xmm8,%xmm7,%xmm8++ vpsrlq $40,%xmm9,%xmm9+ vpsrlq $26,%xmm5,%xmm6+ vpand %xmm15,%xmm5,%xmm5+ vpsrlq $4,%xmm8,%xmm7+ vpand %xmm15,%xmm6,%xmm6+ vpsrlq $30,%xmm8,%xmm8+ vpand %xmm15,%xmm7,%xmm7+ vpand %xmm15,%xmm8,%xmm8+ vpor 32(%rcx),%xmm9,%xmm9++ jbe .Lskip_loop_avx+++ vmovdqu -48(%rdi),%xmm11+ vmovdqu -32(%rdi),%xmm12+ vpshufd $0xEE,%xmm14,%xmm13+ vpshufd $0x44,%xmm14,%xmm10+ vmovdqa %xmm13,-144(%r11)+ vmovdqa %xmm10,0(%rsp)+ vpshufd $0xEE,%xmm11,%xmm14+ vmovdqu -16(%rdi),%xmm10+ vpshufd $0x44,%xmm11,%xmm11+ vmovdqa %xmm14,-128(%r11)+ vmovdqa %xmm11,16(%rsp)+ vpshufd $0xEE,%xmm12,%xmm13+ vmovdqu 0(%rdi),%xmm11+ vpshufd $0x44,%xmm12,%xmm12+ vmovdqa %xmm13,-112(%r11)+ vmovdqa %xmm12,32(%rsp)+ vpshufd $0xEE,%xmm10,%xmm14+ vmovdqu 16(%rdi),%xmm12+ vpshufd $0x44,%xmm10,%xmm10+ vmovdqa %xmm14,-96(%r11)+ vmovdqa %xmm10,48(%rsp)+ vpshufd $0xEE,%xmm11,%xmm13+ vmovdqu 32(%rdi),%xmm10+ vpshufd $0x44,%xmm11,%xmm11+ vmovdqa %xmm13,-80(%r11)+ vmovdqa %xmm11,64(%rsp)+ vpshufd $0xEE,%xmm12,%xmm14+ vmovdqu 48(%rdi),%xmm11+ vpshufd $0x44,%xmm12,%xmm12+ vmovdqa %xmm14,-64(%r11)+ vmovdqa %xmm12,80(%rsp)+ vpshufd $0xEE,%xmm10,%xmm13+ vmovdqu 64(%rdi),%xmm12+ vpshufd $0x44,%xmm10,%xmm10+ vmovdqa %xmm13,-48(%r11)+ vmovdqa %xmm10,96(%rsp)+ vpshufd $0xEE,%xmm11,%xmm14+ vpshufd $0x44,%xmm11,%xmm11+ vmovdqa %xmm14,-32(%r11)+ vmovdqa %xmm11,112(%rsp)+ vpshufd $0xEE,%xmm12,%xmm13+ vmovdqa 0(%rsp),%xmm14+ vpshufd $0x44,%xmm12,%xmm12+ vmovdqa %xmm13,-16(%r11)+ vmovdqa %xmm12,128(%rsp)++ jmp .Loop_avx++.p2align 5+.Loop_avx:+++++++++++++++++++++ vpmuludq %xmm5,%xmm14,%xmm10+ vpmuludq %xmm6,%xmm14,%xmm11+ vmovdqa %xmm2,32(%r11)+ vpmuludq %xmm7,%xmm14,%xmm12+ vmovdqa 16(%rsp),%xmm2+ vpmuludq %xmm8,%xmm14,%xmm13+ vpmuludq %xmm9,%xmm14,%xmm14++ vmovdqa %xmm0,0(%r11)+ vpmuludq 32(%rsp),%xmm9,%xmm0+ vmovdqa %xmm1,16(%r11)+ vpmuludq %xmm8,%xmm2,%xmm1+ vpaddq %xmm0,%xmm10,%xmm10+ vpaddq %xmm1,%xmm14,%xmm14+ vmovdqa %xmm3,48(%r11)+ vpmuludq %xmm7,%xmm2,%xmm0+ vpmuludq %xmm6,%xmm2,%xmm1+ vpaddq %xmm0,%xmm13,%xmm13+ vmovdqa 48(%rsp),%xmm3+ vpaddq %xmm1,%xmm12,%xmm12+ vmovdqa %xmm4,64(%r11)+ vpmuludq %xmm5,%xmm2,%xmm2+ vpmuludq %xmm7,%xmm3,%xmm0+ vpaddq %xmm2,%xmm11,%xmm11++ vmovdqa 64(%rsp),%xmm4+ vpaddq %xmm0,%xmm14,%xmm14+ vpmuludq %xmm6,%xmm3,%xmm1+ vpmuludq %xmm5,%xmm3,%xmm3+ vpaddq %xmm1,%xmm13,%xmm13+ vmovdqa 80(%rsp),%xmm2+ vpaddq %xmm3,%xmm12,%xmm12+ vpmuludq %xmm9,%xmm4,%xmm0+ vpmuludq %xmm8,%xmm4,%xmm4+ vpaddq %xmm0,%xmm11,%xmm11+ vmovdqa 96(%rsp),%xmm3+ vpaddq %xmm4,%xmm10,%xmm10++ vmovdqa 128(%rsp),%xmm4+ vpmuludq %xmm6,%xmm2,%xmm1+ vpmuludq %xmm5,%xmm2,%xmm2+ vpaddq %xmm1,%xmm14,%xmm14+ vpaddq %xmm2,%xmm13,%xmm13+ vpmuludq %xmm9,%xmm3,%xmm0+ vpmuludq %xmm8,%xmm3,%xmm1+ vpaddq %xmm0,%xmm12,%xmm12+ vmovdqu 0(%rsi),%xmm0+ vpaddq %xmm1,%xmm11,%xmm11+ vpmuludq %xmm7,%xmm3,%xmm3+ vpmuludq %xmm7,%xmm4,%xmm7+ vpaddq %xmm3,%xmm10,%xmm10++ vmovdqu 16(%rsi),%xmm1+ vpaddq %xmm7,%xmm11,%xmm11+ vpmuludq %xmm8,%xmm4,%xmm8+ vpmuludq %xmm9,%xmm4,%xmm9+ vpsrldq $6,%xmm0,%xmm2+ vpaddq %xmm8,%xmm12,%xmm12+ vpaddq %xmm9,%xmm13,%xmm13+ vpsrldq $6,%xmm1,%xmm3+ vpmuludq 112(%rsp),%xmm5,%xmm9+ vpmuludq %xmm6,%xmm4,%xmm5+ vpunpckhqdq %xmm1,%xmm0,%xmm4+ vpaddq %xmm9,%xmm14,%xmm14+ vmovdqa -144(%r11),%xmm9+ vpaddq %xmm5,%xmm10,%xmm10++ vpunpcklqdq %xmm1,%xmm0,%xmm0+ vpunpcklqdq %xmm3,%xmm2,%xmm3+++ vpsrldq $5,%xmm4,%xmm4+ vpsrlq $26,%xmm0,%xmm1+ vpand %xmm15,%xmm0,%xmm0+ vpsrlq $4,%xmm3,%xmm2+ vpand %xmm15,%xmm1,%xmm1+ vpand 0(%rcx),%xmm4,%xmm4+ vpsrlq $30,%xmm3,%xmm3+ vpand %xmm15,%xmm2,%xmm2+ vpand %xmm15,%xmm3,%xmm3+ vpor 32(%rcx),%xmm4,%xmm4++ vpaddq 0(%r11),%xmm0,%xmm0+ vpaddq 16(%r11),%xmm1,%xmm1+ vpaddq 32(%r11),%xmm2,%xmm2+ vpaddq 48(%r11),%xmm3,%xmm3+ vpaddq 64(%r11),%xmm4,%xmm4++ leaq 32(%rsi),%rax+ leaq 64(%rsi),%rsi+ subq $64,%rdx+ cmovcq %rax,%rsi+++++++++++ vpmuludq %xmm0,%xmm9,%xmm5+ vpmuludq %xmm1,%xmm9,%xmm6+ vpaddq %xmm5,%xmm10,%xmm10+ vpaddq %xmm6,%xmm11,%xmm11+ vmovdqa -128(%r11),%xmm7+ vpmuludq %xmm2,%xmm9,%xmm5+ vpmuludq %xmm3,%xmm9,%xmm6+ vpaddq %xmm5,%xmm12,%xmm12+ vpaddq %xmm6,%xmm13,%xmm13+ vpmuludq %xmm4,%xmm9,%xmm9+ vpmuludq -112(%r11),%xmm4,%xmm5+ vpaddq %xmm9,%xmm14,%xmm14++ vpaddq %xmm5,%xmm10,%xmm10+ vpmuludq %xmm2,%xmm7,%xmm6+ vpmuludq %xmm3,%xmm7,%xmm5+ vpaddq %xmm6,%xmm13,%xmm13+ vmovdqa -96(%r11),%xmm8+ vpaddq %xmm5,%xmm14,%xmm14+ vpmuludq %xmm1,%xmm7,%xmm6+ vpmuludq %xmm0,%xmm7,%xmm7+ vpaddq %xmm6,%xmm12,%xmm12+ vpaddq %xmm7,%xmm11,%xmm11++ vmovdqa -80(%r11),%xmm9+ vpmuludq %xmm2,%xmm8,%xmm5+ vpmuludq %xmm1,%xmm8,%xmm6+ vpaddq %xmm5,%xmm14,%xmm14+ vpaddq %xmm6,%xmm13,%xmm13+ vmovdqa -64(%r11),%xmm7+ vpmuludq %xmm0,%xmm8,%xmm8+ vpmuludq %xmm4,%xmm9,%xmm5+ vpaddq %xmm8,%xmm12,%xmm12+ vpaddq %xmm5,%xmm11,%xmm11+ vmovdqa -48(%r11),%xmm8+ vpmuludq %xmm3,%xmm9,%xmm9+ vpmuludq %xmm1,%xmm7,%xmm6+ vpaddq %xmm9,%xmm10,%xmm10++ vmovdqa -16(%r11),%xmm9+ vpaddq %xmm6,%xmm14,%xmm14+ vpmuludq %xmm0,%xmm7,%xmm7+ vpmuludq %xmm4,%xmm8,%xmm5+ vpaddq %xmm7,%xmm13,%xmm13+ vpaddq %xmm5,%xmm12,%xmm12+ vmovdqu 32(%rsi),%xmm5+ vpmuludq %xmm3,%xmm8,%xmm7+ vpmuludq %xmm2,%xmm8,%xmm8+ vpaddq %xmm7,%xmm11,%xmm11+ vmovdqu 48(%rsi),%xmm6+ vpaddq %xmm8,%xmm10,%xmm10++ vpmuludq %xmm2,%xmm9,%xmm2+ vpmuludq %xmm3,%xmm9,%xmm3+ vpsrldq $6,%xmm5,%xmm7+ vpaddq %xmm2,%xmm11,%xmm11+ vpmuludq %xmm4,%xmm9,%xmm4+ vpsrldq $6,%xmm6,%xmm8+ vpaddq %xmm3,%xmm12,%xmm2+ vpaddq %xmm4,%xmm13,%xmm3+ vpmuludq -32(%r11),%xmm0,%xmm4+ vpmuludq %xmm1,%xmm9,%xmm0+ vpunpckhqdq %xmm6,%xmm5,%xmm9+ vpaddq %xmm4,%xmm14,%xmm4+ vpaddq %xmm0,%xmm10,%xmm0++ vpunpcklqdq %xmm6,%xmm5,%xmm5+ vpunpcklqdq %xmm8,%xmm7,%xmm8+++ vpsrldq $5,%xmm9,%xmm9+ vpsrlq $26,%xmm5,%xmm6+ vmovdqa 0(%rsp),%xmm14+ vpand %xmm15,%xmm5,%xmm5+ vpsrlq $4,%xmm8,%xmm7+ vpand %xmm15,%xmm6,%xmm6+ vpand 0(%rcx),%xmm9,%xmm9+ vpsrlq $30,%xmm8,%xmm8+ vpand %xmm15,%xmm7,%xmm7+ vpand %xmm15,%xmm8,%xmm8+ vpor 32(%rcx),%xmm9,%xmm9++++++ vpsrlq $26,%xmm3,%xmm13+ vpand %xmm15,%xmm3,%xmm3+ vpaddq %xmm13,%xmm4,%xmm4++ vpsrlq $26,%xmm0,%xmm10+ vpand %xmm15,%xmm0,%xmm0+ vpaddq %xmm10,%xmm11,%xmm1++ vpsrlq $26,%xmm4,%xmm10+ vpand %xmm15,%xmm4,%xmm4++ vpsrlq $26,%xmm1,%xmm11+ vpand %xmm15,%xmm1,%xmm1+ vpaddq %xmm11,%xmm2,%xmm2++ vpaddq %xmm10,%xmm0,%xmm0+ vpsllq $2,%xmm10,%xmm10+ vpaddq %xmm10,%xmm0,%xmm0++ vpsrlq $26,%xmm2,%xmm12+ vpand %xmm15,%xmm2,%xmm2+ vpaddq %xmm12,%xmm3,%xmm3++ vpsrlq $26,%xmm0,%xmm10+ vpand %xmm15,%xmm0,%xmm0+ vpaddq %xmm10,%xmm1,%xmm1++ vpsrlq $26,%xmm3,%xmm13+ vpand %xmm15,%xmm3,%xmm3+ vpaddq %xmm13,%xmm4,%xmm4++ ja .Loop_avx++.Lskip_loop_avx:++++ vpshufd $0x10,%xmm14,%xmm14+ addq $32,%rdx+ jnz .Long_tail_avx++ vpaddq %xmm2,%xmm7,%xmm7+ vpaddq %xmm0,%xmm5,%xmm5+ vpaddq %xmm1,%xmm6,%xmm6+ vpaddq %xmm3,%xmm8,%xmm8+ vpaddq %xmm4,%xmm9,%xmm9++.Long_tail_avx:+ vmovdqa %xmm2,32(%r11)+ vmovdqa %xmm0,0(%r11)+ vmovdqa %xmm1,16(%r11)+ vmovdqa %xmm3,48(%r11)+ vmovdqa %xmm4,64(%r11)++++++++ vpmuludq %xmm7,%xmm14,%xmm12+ vpmuludq %xmm5,%xmm14,%xmm10+ vpshufd $0x10,-48(%rdi),%xmm2+ vpmuludq %xmm6,%xmm14,%xmm11+ vpmuludq %xmm8,%xmm14,%xmm13+ vpmuludq %xmm9,%xmm14,%xmm14++ vpmuludq %xmm8,%xmm2,%xmm0+ vpaddq %xmm0,%xmm14,%xmm14+ vpshufd $0x10,-32(%rdi),%xmm3+ vpmuludq %xmm7,%xmm2,%xmm1+ vpaddq %xmm1,%xmm13,%xmm13+ vpshufd $0x10,-16(%rdi),%xmm4+ vpmuludq %xmm6,%xmm2,%xmm0+ vpaddq %xmm0,%xmm12,%xmm12+ vpmuludq %xmm5,%xmm2,%xmm2+ vpaddq %xmm2,%xmm11,%xmm11+ vpmuludq %xmm9,%xmm3,%xmm3+ vpaddq %xmm3,%xmm10,%xmm10++ vpshufd $0x10,0(%rdi),%xmm2+ vpmuludq %xmm7,%xmm4,%xmm1+ vpaddq %xmm1,%xmm14,%xmm14+ vpmuludq %xmm6,%xmm4,%xmm0+ vpaddq %xmm0,%xmm13,%xmm13+ vpshufd $0x10,16(%rdi),%xmm3+ vpmuludq %xmm5,%xmm4,%xmm4+ vpaddq %xmm4,%xmm12,%xmm12+ vpmuludq %xmm9,%xmm2,%xmm1+ vpaddq %xmm1,%xmm11,%xmm11+ vpshufd $0x10,32(%rdi),%xmm4+ vpmuludq %xmm8,%xmm2,%xmm2+ vpaddq %xmm2,%xmm10,%xmm10++ vpmuludq %xmm6,%xmm3,%xmm0+ vpaddq %xmm0,%xmm14,%xmm14+ vpmuludq %xmm5,%xmm3,%xmm3+ vpaddq %xmm3,%xmm13,%xmm13+ vpshufd $0x10,48(%rdi),%xmm2+ vpmuludq %xmm9,%xmm4,%xmm1+ vpaddq %xmm1,%xmm12,%xmm12+ vpshufd $0x10,64(%rdi),%xmm3+ vpmuludq %xmm8,%xmm4,%xmm0+ vpaddq %xmm0,%xmm11,%xmm11+ vpmuludq %xmm7,%xmm4,%xmm4+ vpaddq %xmm4,%xmm10,%xmm10++ vpmuludq %xmm5,%xmm2,%xmm2+ vpaddq %xmm2,%xmm14,%xmm14+ vpmuludq %xmm9,%xmm3,%xmm1+ vpaddq %xmm1,%xmm13,%xmm13+ vpmuludq %xmm8,%xmm3,%xmm0+ vpaddq %xmm0,%xmm12,%xmm12+ vpmuludq %xmm7,%xmm3,%xmm1+ vpaddq %xmm1,%xmm11,%xmm11+ vpmuludq %xmm6,%xmm3,%xmm3+ vpaddq %xmm3,%xmm10,%xmm10++ jz .Lshort_tail_avx++ vmovdqu 0(%rsi),%xmm0+ vmovdqu 16(%rsi),%xmm1++ vpsrldq $6,%xmm0,%xmm2+ vpsrldq $6,%xmm1,%xmm3+ vpunpckhqdq %xmm1,%xmm0,%xmm4+ vpunpcklqdq %xmm1,%xmm0,%xmm0+ vpunpcklqdq %xmm3,%xmm2,%xmm3++ vpsrlq $40,%xmm4,%xmm4+ vpsrlq $26,%xmm0,%xmm1+ vpand %xmm15,%xmm0,%xmm0+ vpsrlq $4,%xmm3,%xmm2+ vpand %xmm15,%xmm1,%xmm1+ vpsrlq $30,%xmm3,%xmm3+ vpand %xmm15,%xmm2,%xmm2+ vpand %xmm15,%xmm3,%xmm3+ vpor 32(%rcx),%xmm4,%xmm4++ vpshufd $0x32,-64(%rdi),%xmm9+ vpaddq 0(%r11),%xmm0,%xmm0+ vpaddq 16(%r11),%xmm1,%xmm1+ vpaddq 32(%r11),%xmm2,%xmm2+ vpaddq 48(%r11),%xmm3,%xmm3+ vpaddq 64(%r11),%xmm4,%xmm4+++++ vpmuludq %xmm0,%xmm9,%xmm5+ vpaddq %xmm5,%xmm10,%xmm10+ vpmuludq %xmm1,%xmm9,%xmm6+ vpaddq %xmm6,%xmm11,%xmm11+ vpmuludq %xmm2,%xmm9,%xmm5+ vpaddq %xmm5,%xmm12,%xmm12+ vpshufd $0x32,-48(%rdi),%xmm7+ vpmuludq %xmm3,%xmm9,%xmm6+ vpaddq %xmm6,%xmm13,%xmm13+ vpmuludq %xmm4,%xmm9,%xmm9+ vpaddq %xmm9,%xmm14,%xmm14++ vpmuludq %xmm3,%xmm7,%xmm5+ vpaddq %xmm5,%xmm14,%xmm14+ vpshufd $0x32,-32(%rdi),%xmm8+ vpmuludq %xmm2,%xmm7,%xmm6+ vpaddq %xmm6,%xmm13,%xmm13+ vpshufd $0x32,-16(%rdi),%xmm9+ vpmuludq %xmm1,%xmm7,%xmm5+ vpaddq %xmm5,%xmm12,%xmm12+ vpmuludq %xmm0,%xmm7,%xmm7+ vpaddq %xmm7,%xmm11,%xmm11+ vpmuludq %xmm4,%xmm8,%xmm8+ vpaddq %xmm8,%xmm10,%xmm10++ vpshufd $0x32,0(%rdi),%xmm7+ vpmuludq %xmm2,%xmm9,%xmm6+ vpaddq %xmm6,%xmm14,%xmm14+ vpmuludq %xmm1,%xmm9,%xmm5+ vpaddq %xmm5,%xmm13,%xmm13+ vpshufd $0x32,16(%rdi),%xmm8+ vpmuludq %xmm0,%xmm9,%xmm9+ vpaddq %xmm9,%xmm12,%xmm12+ vpmuludq %xmm4,%xmm7,%xmm6+ vpaddq %xmm6,%xmm11,%xmm11+ vpshufd $0x32,32(%rdi),%xmm9+ vpmuludq %xmm3,%xmm7,%xmm7+ vpaddq %xmm7,%xmm10,%xmm10++ vpmuludq %xmm1,%xmm8,%xmm5+ vpaddq %xmm5,%xmm14,%xmm14+ vpmuludq %xmm0,%xmm8,%xmm8+ vpaddq %xmm8,%xmm13,%xmm13+ vpshufd $0x32,48(%rdi),%xmm7+ vpmuludq %xmm4,%xmm9,%xmm6+ vpaddq %xmm6,%xmm12,%xmm12+ vpshufd $0x32,64(%rdi),%xmm8+ vpmuludq %xmm3,%xmm9,%xmm5+ vpaddq %xmm5,%xmm11,%xmm11+ vpmuludq %xmm2,%xmm9,%xmm9+ vpaddq %xmm9,%xmm10,%xmm10++ vpmuludq %xmm0,%xmm7,%xmm7+ vpaddq %xmm7,%xmm14,%xmm14+ vpmuludq %xmm4,%xmm8,%xmm6+ vpaddq %xmm6,%xmm13,%xmm13+ vpmuludq %xmm3,%xmm8,%xmm5+ vpaddq %xmm5,%xmm12,%xmm12+ vpmuludq %xmm2,%xmm8,%xmm6+ vpaddq %xmm6,%xmm11,%xmm11+ vpmuludq %xmm1,%xmm8,%xmm8+ vpaddq %xmm8,%xmm10,%xmm10++.Lshort_tail_avx:++++ vpsrldq $8,%xmm14,%xmm9+ vpsrldq $8,%xmm13,%xmm8+ vpsrldq $8,%xmm11,%xmm6+ vpsrldq $8,%xmm10,%xmm5+ vpsrldq $8,%xmm12,%xmm7+ vpaddq %xmm8,%xmm13,%xmm13+ vpaddq %xmm9,%xmm14,%xmm14+ vpaddq %xmm5,%xmm10,%xmm10+ vpaddq %xmm6,%xmm11,%xmm11+ vpaddq %xmm7,%xmm12,%xmm12+++++ vpsrlq $26,%xmm13,%xmm3+ vpand %xmm15,%xmm13,%xmm13+ vpaddq %xmm3,%xmm14,%xmm14++ vpsrlq $26,%xmm10,%xmm0+ vpand %xmm15,%xmm10,%xmm10+ vpaddq %xmm0,%xmm11,%xmm11++ vpsrlq $26,%xmm14,%xmm4+ vpand %xmm15,%xmm14,%xmm14++ vpsrlq $26,%xmm11,%xmm1+ vpand %xmm15,%xmm11,%xmm11+ vpaddq %xmm1,%xmm12,%xmm12++ vpaddq %xmm4,%xmm10,%xmm10+ vpsllq $2,%xmm4,%xmm4+ vpaddq %xmm4,%xmm10,%xmm10++ vpsrlq $26,%xmm12,%xmm2+ vpand %xmm15,%xmm12,%xmm12+ vpaddq %xmm2,%xmm13,%xmm13++ vpsrlq $26,%xmm10,%xmm0+ vpand %xmm15,%xmm10,%xmm10+ vpaddq %xmm0,%xmm11,%xmm11++ vpsrlq $26,%xmm13,%xmm3+ vpand %xmm15,%xmm13,%xmm13+ vpaddq %xmm3,%xmm14,%xmm14++ vmovd %xmm10,-112(%rdi)+ vmovd %xmm11,-108(%rdi)+ vmovd %xmm12,-104(%rdi)+ vmovd %xmm13,-100(%rdi)+ vmovd %xmm14,-96(%rdi)+ vmovdqa 80(%r11),%xmm6+ vmovdqa 96(%r11),%xmm7+ vmovdqa 112(%r11),%xmm8+ vmovdqa 128(%r11),%xmm9+ vmovdqa 144(%r11),%xmm10+ vmovdqa 160(%r11),%xmm11+ vmovdqa 176(%r11),%xmm12+ vmovdqa 192(%r11),%xmm13+ vmovdqa 208(%r11),%xmm14+ vmovdqa 224(%r11),%xmm15+ leaq 248(%r11),%rsp+.Ldo_avx_epilogue:+ vzeroupper+ movq 8(%rsp),%rdi+ movq 16(%rsp),%rsi+ .byte 0xf3,0xc3++.LSEH_end_crypton_poly1305_asm_blocks_avx:+.def crypton_poly1305_asm_blocks_avx2; .scl 3; .type 32; .endef+.p2align 5+crypton_poly1305_asm_blocks_avx2:+ .byte 0xf3,0x0f,0x1e,0xfa+ movq %rdi,8(%rsp)+ movq %rsi,16(%rsp)+ movq %rsp,%rax+.LSEH_begin_crypton_poly1305_asm_blocks_avx2:+++ movq %rcx,%rdi+ movq %rdx,%rsi+ movq %r8,%rdx+ movq %r9,%rcx+ movl 20(%rdi),%r8d+ cmpq $128,%rdx+ jb .Lblocks++ andq $-16,%rdx++ vzeroupper++ testl %r8d,%r8d+ jz .Lbase2_64_avx2++ testq $63,%rdx+ jz .Leven_avx2++ pushq %rbx++ pushq %rbp++ pushq %r12++ pushq %r13++ pushq %r14++ pushq %r15++ leaq -8(%rsp),%rsp++.Lblocks_avx2_body:++ movq %rdx,%r15++ movq 0(%rdi),%r8+ movq 8(%rdi),%r9+ movl 16(%rdi),%ebp++ movq 24(%rdi),%r11+ movq 32(%rdi),%r13+++ movl %r8d,%r14d+ andq $-2147483648,%r8+ movq %r9,%r12+ movl %r9d,%ebx+ andq $-2147483648,%r9++ shrq $6,%r8+ shlq $52,%r12+ addq %r8,%r14+ shrq $12,%rbx+ shrq $18,%r9+ addq %r12,%r14+ adcq %r9,%rbx++ movq %rbp,%r8+ shlq $40,%r8+ shrq $24,%rbp+ addq %r8,%rbx+ adcq $0,%rbp++ movq %r13,%r12+ movq %r13,%rax+ shrq $2,%r13+ addq %r12,%r13++.Lbase2_26_pre_avx2:+ addq 0(%rsi),%r14+ adcq 8(%rsi),%rbx+ leaq 16(%rsi),%rsi+ adcq %rcx,%rbp+ subq $16,%r15++ call __crypton_poly1305_asm_block+ movq %r12,%rax++ testq $63,%r15+ jnz .Lbase2_26_pre_avx2+++ movq %r14,%rax+ movq %r14,%rdx+ shrq $52,%r14+ movq %rbx,%r11+ movq %rbx,%r12+ shrq $26,%rdx+ andq $0x3ffffff,%rax+ shlq $12,%r11+ andq $0x3ffffff,%rdx+ shrq $14,%rbx+ orq %r11,%r14+ shlq $24,%rbp+ andq $0x3ffffff,%r14+ shrq $40,%r12+ andq $0x3ffffff,%rbx+ orq %r12,%rbp++ vmovd %eax,%xmm0+ vmovd %edx,%xmm1+ vmovd %r14d,%xmm2+ vmovd %ebx,%xmm3+ vmovd %ebp,%xmm4++ movq %r15,%rdx++ movq 8(%rsp),%r15++ movq 16(%rsp),%r14++ movq 24(%rsp),%r13++ movq 32(%rsp),%r12++ movq 40(%rsp),%rbp++ movq 48(%rsp),%rbx++ leaq 56(%rsp),%rax+ leaq 56(%rsp),%rsp++.Lblocks_avx2_epilogue:+ jmp .Ldo_avx2+++.p2align 5+.Lbase2_64_avx2:++ pushq %rbx++ pushq %rbp++ pushq %r12++ pushq %r13++ pushq %r14++ pushq %r15++ leaq -8(%rsp),%rsp++.Lbase2_64_avx2_body:++ movq %rdx,%r15++ movq 24(%rdi),%r11+ movq 32(%rdi),%r13++ movq 0(%rdi),%r14+ movq 8(%rdi),%rbx+ movl 16(%rdi),%ebp++ movq %r13,%r12+ movq %r13,%rax+ shrq $2,%r13+ addq %r12,%r13++ testq $63,%rdx+ jz .Linit_avx2++.Lbase2_64_pre_avx2:+ addq 0(%rsi),%r14+ adcq 8(%rsi),%rbx+ leaq 16(%rsi),%rsi+ adcq %rcx,%rbp+ subq $16,%r15++ call __crypton_poly1305_asm_block+ movq %r12,%rax++ testq $63,%r15+ jnz .Lbase2_64_pre_avx2++.Linit_avx2:++ movq %r14,%rax+ movq %r14,%rdx+ shrq $52,%r14+ movq %rbx,%r8+ movq %rbx,%r9+ shrq $26,%rdx+ andq $0x3ffffff,%rax+ shlq $12,%r8+ andq $0x3ffffff,%rdx+ shrq $14,%rbx+ orq %r8,%r14+ shlq $24,%rbp+ andq $0x3ffffff,%r14+ shrq $40,%r9+ andq $0x3ffffff,%rbx+ orq %r9,%rbp++ vmovd %eax,%xmm0+ vmovd %edx,%xmm1+ vmovd %r14d,%xmm2+ vmovd %ebx,%xmm3+ vmovd %ebp,%xmm4+ movl $1,20(%rdi)++ call __crypton_poly1305_asm_init_avx++ movq %r15,%rdx++ movq 8(%rsp),%r15++ movq 16(%rsp),%r14++ movq 24(%rsp),%r13++ movq 32(%rsp),%r12++ movq 40(%rsp),%rbp++ movq 48(%rsp),%rbx++ leaq 56(%rsp),%rax+ leaq 56(%rsp),%rsp++.Lbase2_64_avx2_epilogue:+ jmp .Ldo_avx2+++.p2align 5+.Leven_avx2:++ vmovd 0(%rdi),%xmm0+ vmovd 4(%rdi),%xmm1+ vmovd 8(%rdi),%xmm2+ vmovd 12(%rdi),%xmm3+ vmovd 16(%rdi),%xmm4++.Ldo_avx2:+ leaq -248(%rsp),%r11+ subq $0x1c8,%rsp+ vmovdqa %xmm6,80(%r11)+ vmovdqa %xmm7,96(%r11)+ vmovdqa %xmm8,112(%r11)+ vmovdqa %xmm9,128(%r11)+ vmovdqa %xmm10,144(%r11)+ vmovdqa %xmm11,160(%r11)+ vmovdqa %xmm12,176(%r11)+ vmovdqa %xmm13,192(%r11)+ vmovdqa %xmm14,208(%r11)+ vmovdqa %xmm15,224(%r11)+.Ldo_avx2_body:+ leaq .Lconst(%rip),%rcx+ leaq 48+64(%rdi),%rdi+ vmovdqa 96(%rcx),%ymm7+++ vmovdqu -64(%rdi),%xmm9+ andq $-512,%rsp+ vmovdqu -48(%rdi),%xmm10+ vmovdqu -32(%rdi),%xmm6+ vmovdqu -16(%rdi),%xmm11+ vmovdqu 0(%rdi),%xmm12+ vmovdqu 16(%rdi),%xmm13+ leaq 144(%rsp),%rax+ vmovdqu 32(%rdi),%xmm14+ vpermd %ymm9,%ymm7,%ymm9+ vmovdqu 48(%rdi),%xmm15+ vpermd %ymm10,%ymm7,%ymm10+ vmovdqu 64(%rdi),%xmm5+ vpermd %ymm6,%ymm7,%ymm6+ vmovdqa %ymm9,0(%rsp)+ vpermd %ymm11,%ymm7,%ymm11+ vmovdqa %ymm10,32-144(%rax)+ vpermd %ymm12,%ymm7,%ymm12+ vmovdqa %ymm6,64-144(%rax)+ vpermd %ymm13,%ymm7,%ymm13+ vmovdqa %ymm11,96-144(%rax)+ vpermd %ymm14,%ymm7,%ymm14+ vmovdqa %ymm12,128-144(%rax)+ vpermd %ymm15,%ymm7,%ymm15+ vmovdqa %ymm13,160-144(%rax)+ vpermd %ymm5,%ymm7,%ymm5+ vmovdqa %ymm14,192-144(%rax)+ vmovdqa %ymm15,224-144(%rax)+ vmovdqa %ymm5,256-144(%rax)+ vmovdqa 64(%rcx),%ymm5++++ vmovdqu 0(%rsi),%xmm7+ vmovdqu 16(%rsi),%xmm8+ vinserti128 $1,32(%rsi),%ymm7,%ymm7+ vinserti128 $1,48(%rsi),%ymm8,%ymm8+ leaq 64(%rsi),%rsi++ vpsrldq $6,%ymm7,%ymm9+ vpsrldq $6,%ymm8,%ymm10+ vpunpckhqdq %ymm8,%ymm7,%ymm6+ vpunpcklqdq %ymm10,%ymm9,%ymm9+ vpunpcklqdq %ymm8,%ymm7,%ymm7++ vpsrlq $30,%ymm9,%ymm10+ vpsrlq $4,%ymm9,%ymm9+ vpsrlq $26,%ymm7,%ymm8+ vpsrlq $40,%ymm6,%ymm6+ vpand %ymm5,%ymm9,%ymm9+ vpand %ymm5,%ymm7,%ymm7+ vpand %ymm5,%ymm8,%ymm8+ vpand %ymm5,%ymm10,%ymm10+ vpor 32(%rcx),%ymm6,%ymm6++ vpaddq %ymm2,%ymm9,%ymm2+ subq $64,%rdx+ jz .Ltail_avx2+ jmp .Loop_avx2++.p2align 5+.Loop_avx2:+++++++++ vpaddq %ymm0,%ymm7,%ymm0+ vmovdqa 0(%rsp),%ymm7+ vpaddq %ymm1,%ymm8,%ymm1+ vmovdqa 32(%rsp),%ymm8+ vpaddq %ymm3,%ymm10,%ymm3+ vmovdqa 96(%rsp),%ymm9+ vpaddq %ymm4,%ymm6,%ymm4+ vmovdqa 48(%rax),%ymm10+ vmovdqa 112(%rax),%ymm5+++++++++++++++++ vpmuludq %ymm2,%ymm7,%ymm13+ vpmuludq %ymm2,%ymm8,%ymm14+ vpmuludq %ymm2,%ymm9,%ymm15+ vpmuludq %ymm2,%ymm10,%ymm11+ vpmuludq %ymm2,%ymm5,%ymm12++ vpmuludq %ymm0,%ymm8,%ymm6+ vpmuludq %ymm1,%ymm8,%ymm2+ vpaddq %ymm6,%ymm12,%ymm12+ vpaddq %ymm2,%ymm13,%ymm13+ vpmuludq %ymm3,%ymm8,%ymm6+ vpmuludq 64(%rsp),%ymm4,%ymm2+ vpaddq %ymm6,%ymm15,%ymm15+ vpaddq %ymm2,%ymm11,%ymm11+ vmovdqa -16(%rax),%ymm8++ vpmuludq %ymm0,%ymm7,%ymm6+ vpmuludq %ymm1,%ymm7,%ymm2+ vpaddq %ymm6,%ymm11,%ymm11+ vpaddq %ymm2,%ymm12,%ymm12+ vpmuludq %ymm3,%ymm7,%ymm6+ vpmuludq %ymm4,%ymm7,%ymm2+ vmovdqu 0(%rsi),%xmm7+ vpaddq %ymm6,%ymm14,%ymm14+ vpaddq %ymm2,%ymm15,%ymm15+ vinserti128 $1,32(%rsi),%ymm7,%ymm7++ vpmuludq %ymm3,%ymm8,%ymm6+ vpmuludq %ymm4,%ymm8,%ymm2+ vmovdqu 16(%rsi),%xmm8+ vpaddq %ymm6,%ymm11,%ymm11+ vpaddq %ymm2,%ymm12,%ymm12+ vmovdqa 16(%rax),%ymm2+ vpmuludq %ymm1,%ymm9,%ymm6+ vpmuludq %ymm0,%ymm9,%ymm9+ vpaddq %ymm6,%ymm14,%ymm14+ vpaddq %ymm9,%ymm13,%ymm13+ vinserti128 $1,48(%rsi),%ymm8,%ymm8+ leaq 64(%rsi),%rsi++ vpmuludq %ymm1,%ymm2,%ymm6+ vpmuludq %ymm0,%ymm2,%ymm2+ vpsrldq $6,%ymm7,%ymm9+ vpaddq %ymm6,%ymm15,%ymm15+ vpaddq %ymm2,%ymm14,%ymm14+ vpmuludq %ymm3,%ymm10,%ymm6+ vpmuludq %ymm4,%ymm10,%ymm2+ vpsrldq $6,%ymm8,%ymm10+ vpaddq %ymm6,%ymm12,%ymm12+ vpaddq %ymm2,%ymm13,%ymm13+ vpunpckhqdq %ymm8,%ymm7,%ymm6++ vpmuludq %ymm3,%ymm5,%ymm3+ vpmuludq %ymm4,%ymm5,%ymm4+ vpunpcklqdq %ymm8,%ymm7,%ymm7+ vpaddq %ymm3,%ymm13,%ymm2+ vpaddq %ymm4,%ymm14,%ymm3+ vpunpcklqdq %ymm10,%ymm9,%ymm10+ vpmuludq 80(%rax),%ymm0,%ymm4+ vpmuludq %ymm1,%ymm5,%ymm0+ vmovdqa 64(%rcx),%ymm5+ vpaddq %ymm4,%ymm15,%ymm4+ vpaddq %ymm0,%ymm11,%ymm0+++++ vpsrlq $26,%ymm3,%ymm14+ vpand %ymm5,%ymm3,%ymm3+ vpaddq %ymm14,%ymm4,%ymm4++ vpsrlq $26,%ymm0,%ymm11+ vpand %ymm5,%ymm0,%ymm0+ vpaddq %ymm11,%ymm12,%ymm1++ vpsrlq $26,%ymm4,%ymm15+ vpand %ymm5,%ymm4,%ymm4++ vpsrlq $4,%ymm10,%ymm9++ vpsrlq $26,%ymm1,%ymm12+ vpand %ymm5,%ymm1,%ymm1+ vpaddq %ymm12,%ymm2,%ymm2++ vpaddq %ymm15,%ymm0,%ymm0+ vpsllq $2,%ymm15,%ymm15+ vpaddq %ymm15,%ymm0,%ymm0++ vpand %ymm5,%ymm9,%ymm9+ vpsrlq $26,%ymm7,%ymm8++ vpsrlq $26,%ymm2,%ymm13+ vpand %ymm5,%ymm2,%ymm2+ vpaddq %ymm13,%ymm3,%ymm3++ vpaddq %ymm9,%ymm2,%ymm2+ vpsrlq $30,%ymm10,%ymm10++ vpsrlq $26,%ymm0,%ymm11+ vpand %ymm5,%ymm0,%ymm0+ vpaddq %ymm11,%ymm1,%ymm1++ vpsrlq $40,%ymm6,%ymm6++ vpsrlq $26,%ymm3,%ymm14+ vpand %ymm5,%ymm3,%ymm3+ vpaddq %ymm14,%ymm4,%ymm4++ vpand %ymm5,%ymm7,%ymm7+ vpand %ymm5,%ymm8,%ymm8+ vpand %ymm5,%ymm10,%ymm10+ vpor 32(%rcx),%ymm6,%ymm6++ subq $64,%rdx+ jnz .Loop_avx2++.byte 0x66,0x90+.Ltail_avx2:++++++++ vpaddq %ymm0,%ymm7,%ymm0+ vmovdqu 4(%rsp),%ymm7+ vpaddq %ymm1,%ymm8,%ymm1+ vmovdqu 36(%rsp),%ymm8+ vpaddq %ymm3,%ymm10,%ymm3+ vmovdqu 100(%rsp),%ymm9+ vpaddq %ymm4,%ymm6,%ymm4+ vmovdqu 52(%rax),%ymm10+ vmovdqu 116(%rax),%ymm5++ vpmuludq %ymm2,%ymm7,%ymm13+ vpmuludq %ymm2,%ymm8,%ymm14+ vpmuludq %ymm2,%ymm9,%ymm15+ vpmuludq %ymm2,%ymm10,%ymm11+ vpmuludq %ymm2,%ymm5,%ymm12++ vpmuludq %ymm0,%ymm8,%ymm6+ vpmuludq %ymm1,%ymm8,%ymm2+ vpaddq %ymm6,%ymm12,%ymm12+ vpaddq %ymm2,%ymm13,%ymm13+ vpmuludq %ymm3,%ymm8,%ymm6+ vpmuludq 68(%rsp),%ymm4,%ymm2+ vpaddq %ymm6,%ymm15,%ymm15+ vpaddq %ymm2,%ymm11,%ymm11++ vpmuludq %ymm0,%ymm7,%ymm6+ vpmuludq %ymm1,%ymm7,%ymm2+ vpaddq %ymm6,%ymm11,%ymm11+ vmovdqu -12(%rax),%ymm8+ vpaddq %ymm2,%ymm12,%ymm12+ vpmuludq %ymm3,%ymm7,%ymm6+ vpmuludq %ymm4,%ymm7,%ymm2+ vpaddq %ymm6,%ymm14,%ymm14+ vpaddq %ymm2,%ymm15,%ymm15++ vpmuludq %ymm3,%ymm8,%ymm6+ vpmuludq %ymm4,%ymm8,%ymm2+ vpaddq %ymm6,%ymm11,%ymm11+ vpaddq %ymm2,%ymm12,%ymm12+ vmovdqu 20(%rax),%ymm2+ vpmuludq %ymm1,%ymm9,%ymm6+ vpmuludq %ymm0,%ymm9,%ymm9+ vpaddq %ymm6,%ymm14,%ymm14+ vpaddq %ymm9,%ymm13,%ymm13++ vpmuludq %ymm1,%ymm2,%ymm6+ vpmuludq %ymm0,%ymm2,%ymm2+ vpaddq %ymm6,%ymm15,%ymm15+ vpaddq %ymm2,%ymm14,%ymm14+ vpmuludq %ymm3,%ymm10,%ymm6+ vpmuludq %ymm4,%ymm10,%ymm2+ vpaddq %ymm6,%ymm12,%ymm12+ vpaddq %ymm2,%ymm13,%ymm13++ vpmuludq %ymm3,%ymm5,%ymm3+ vpmuludq %ymm4,%ymm5,%ymm4+ vpaddq %ymm3,%ymm13,%ymm2+ vpaddq %ymm4,%ymm14,%ymm3+ vpmuludq 84(%rax),%ymm0,%ymm4+ vpmuludq %ymm1,%ymm5,%ymm0+ vmovdqa 64(%rcx),%ymm5+ vpaddq %ymm4,%ymm15,%ymm4+ vpaddq %ymm0,%ymm11,%ymm0+++++ vpsrldq $8,%ymm12,%ymm8+ vpsrldq $8,%ymm2,%ymm9+ vpsrldq $8,%ymm3,%ymm10+ vpsrldq $8,%ymm4,%ymm6+ vpsrldq $8,%ymm0,%ymm7+ vpaddq %ymm8,%ymm12,%ymm12+ vpaddq %ymm9,%ymm2,%ymm2+ vpaddq %ymm10,%ymm3,%ymm3+ vpaddq %ymm6,%ymm4,%ymm4+ vpaddq %ymm7,%ymm0,%ymm0++ vpermq $0x2,%ymm3,%ymm10+ vpermq $0x2,%ymm4,%ymm6+ vpermq $0x2,%ymm0,%ymm7+ vpermq $0x2,%ymm12,%ymm8+ vpermq $0x2,%ymm2,%ymm9+ vpaddq %ymm10,%ymm3,%ymm3+ vpaddq %ymm6,%ymm4,%ymm4+ vpaddq %ymm7,%ymm0,%ymm0+ vpaddq %ymm8,%ymm12,%ymm12+ vpaddq %ymm9,%ymm2,%ymm2+++++ vpsrlq $26,%ymm3,%ymm14+ vpand %ymm5,%ymm3,%ymm3+ vpaddq %ymm14,%ymm4,%ymm4++ vpsrlq $26,%ymm0,%ymm11+ vpand %ymm5,%ymm0,%ymm0+ vpaddq %ymm11,%ymm12,%ymm1++ vpsrlq $26,%ymm4,%ymm15+ vpand %ymm5,%ymm4,%ymm4++ vpsrlq $26,%ymm1,%ymm12+ vpand %ymm5,%ymm1,%ymm1+ vpaddq %ymm12,%ymm2,%ymm2++ vpaddq %ymm15,%ymm0,%ymm0+ vpsllq $2,%ymm15,%ymm15+ vpaddq %ymm15,%ymm0,%ymm0++ vpsrlq $26,%ymm2,%ymm13+ vpand %ymm5,%ymm2,%ymm2+ vpaddq %ymm13,%ymm3,%ymm3++ vpsrlq $26,%ymm0,%ymm11+ vpand %ymm5,%ymm0,%ymm0+ vpaddq %ymm11,%ymm1,%ymm1++ vpsrlq $26,%ymm3,%ymm14+ vpand %ymm5,%ymm3,%ymm3+ vpaddq %ymm14,%ymm4,%ymm4++ vmovd %xmm0,-112(%rdi)+ vmovd %xmm1,-108(%rdi)+ vmovd %xmm2,-104(%rdi)+ vmovd %xmm3,-100(%rdi)+ vmovd %xmm4,-96(%rdi)+ vmovdqa 80(%r11),%xmm6+ vmovdqa 96(%r11),%xmm7+ vmovdqa 112(%r11),%xmm8+ vmovdqa 128(%r11),%xmm9+ vmovdqa 144(%r11),%xmm10+ vmovdqa 160(%r11),%xmm11+ vmovdqa 176(%r11),%xmm12+ vmovdqa 192(%r11),%xmm13+ vmovdqa 208(%r11),%xmm14+ vmovdqa 224(%r11),%xmm15+ leaq 248(%r11),%rsp+.Ldo_avx2_epilogue:+ vzeroupper+ movq 8(%rsp),%rdi+ movq 16(%rsp),%rsi+ .byte 0xf3,0xc3++.LSEH_end_crypton_poly1305_asm_blocks_avx2:+.p2align 6+.Lconst:+.Lmask24:+.long 0x0ffffff,0,0x0ffffff,0,0x0ffffff,0,0x0ffffff,0+.L129:+.long 16777216,0,16777216,0,16777216,0,16777216,0+.Lmask26:+.long 0x3ffffff,0,0x3ffffff,0,0x3ffffff,0,0x3ffffff,0+.Lpermd_avx2:+.long 2,2,2,3,2,0,2,1+.Lpermd_avx512:+.long 0,0,0,1, 0,2,0,3, 0,4,0,5, 0,6,0,7++.L2_44_inp_permd:+.long 0,1,1,2,2,3,7,7+.L2_44_inp_shift:+.quad 0,12,24,64+.L2_44_mask:+.quad 0xfffffffffff,0xfffffffffff,0x3ffffffffff,0xffffffffffffffff+.L2_44_shift_rgt:+.quad 44,44,42,64+.L2_44_shift_lft:+.quad 8,8,10,64++.p2align 6+.Lx_mask44:+.quad 0xfffffffffff,0xfffffffffff,0xfffffffffff,0xfffffffffff+.quad 0xfffffffffff,0xfffffffffff,0xfffffffffff,0xfffffffffff+.Lx_mask42:+.quad 0x3ffffffffff,0x3ffffffffff,0x3ffffffffff,0x3ffffffffff+.quad 0x3ffffffffff,0x3ffffffffff,0x3ffffffffff,0x3ffffffffff+.byte 80,111,108,121,49,51,48,53,32,102,111,114,32,120,56,54,95,54,52,44,32,67,82,89,80,84,79,71,65,77,83,32,98,121,32,64,100,111,116,45,97,115,109,0+.p2align 4+.globl crypton_xor128_encrypt_n_pad+.def crypton_xor128_encrypt_n_pad; .scl 2; .type 32; .endef+.p2align 4+crypton_xor128_encrypt_n_pad:+ .byte 0xf3,0x0f,0x1e,0xfa++ subq %r8,%rdx+ subq %r8,%rcx+ movq %r9,%r10+ shrq $4,%r9+ jz .Ltail_enc+ nop+.Loop_enc_xmm:+ movdqu (%rdx,%r8,1),%xmm0+ pxor (%r8),%xmm0+ movdqu %xmm0,(%rcx,%r8,1)+ movdqa %xmm0,(%r8)+ leaq 16(%r8),%r8+ decq %r9+ jnz .Loop_enc_xmm++ andq $15,%r10+ jz .Ldone_enc++.Ltail_enc:+ movq $16,%r9+ subq %r10,%r9+ xorl %eax,%eax+.Loop_enc_byte:+ movb (%rdx,%r8,1),%al+ xorb (%r8),%al+ movb %al,(%rcx,%r8,1)+ movb %al,(%r8)+ leaq 1(%r8),%r8+ decq %r10+ jnz .Loop_enc_byte++ xorl %eax,%eax+.Loop_enc_pad:+ movb %al,(%r8)+ leaq 1(%r8),%r8+ decq %r9+ jnz .Loop_enc_pad++.Ldone_enc:+ movq %r8,%rax+ .byte 0xf3,0xc3+++.globl crypton_xor128_decrypt_n_pad+.def crypton_xor128_decrypt_n_pad; .scl 2; .type 32; .endef+.p2align 4+crypton_xor128_decrypt_n_pad:+ .byte 0xf3,0x0f,0x1e,0xfa++ subq %r8,%rdx+ subq %r8,%rcx+ movq %r9,%r10+ shrq $4,%r9+ jz .Ltail_dec+ nop+.Loop_dec_xmm:+ movdqu (%rdx,%r8,1),%xmm0+ movdqa (%r8),%xmm1+ pxor %xmm0,%xmm1+ movdqu %xmm1,(%rcx,%r8,1)+ movdqa %xmm0,(%r8)+ leaq 16(%r8),%r8+ decq %r9+ jnz .Loop_dec_xmm++ pxor %xmm1,%xmm1+ andq $15,%r10+ jz .Ldone_dec++.Ltail_dec:+ movq $16,%r9+ subq %r10,%r9+ xorl %eax,%eax+ xorq %r11,%r11+.Loop_dec_byte:+ movb (%rdx,%r8,1),%r11b+ movb (%r8),%al+ xorb %r11b,%al+ movb %al,(%rcx,%r8,1)+ movb %r11b,(%r8)+ leaq 1(%r8),%r8+ decq %r10+ jnz .Loop_dec_byte++ xorl %eax,%eax+.Loop_dec_pad:+ movb %al,(%r8)+ leaq 1(%r8),%r8+ decq %r9+ jnz .Loop_dec_pad++.Ldone_dec:+ movq %r8,%rax+ .byte 0xf3,0xc3+++.def se_handler; .scl 3; .type 32; .endef+.p2align 4+se_handler:+ .byte 0xf3,0x0f,0x1e,0xfa++ pushq %rsi+ pushq %rdi+ pushq %rbx+ pushq %rbp+ pushq %r12+ pushq %r13+ pushq %r14+ pushq %r15+ pushfq+ subq $64,%rsp++ movq 120(%r8),%rax+ movq 248(%r8),%rbx++ movq 8(%r9),%rsi+ movq 56(%r9),%r11++ movl 0(%r11),%r10d+ leaq (%rsi,%r10,1),%r10+ cmpq %r10,%rbx+ jb .Lcommon_seh_tail++ movq 152(%r8),%rax++ movl 4(%r11),%r10d+ leaq (%rsi,%r10,1),%r10+ cmpq %r10,%rbx+ jae .Lcommon_seh_tail++ leaq 56(%rax),%rax++ movq -8(%rax),%rbx+ movq -16(%rax),%rbp+ movq -24(%rax),%r12+ movq -32(%rax),%r13+ movq -40(%rax),%r14+ movq -48(%rax),%r15+ movq %rbx,144(%r8)+ movq %rbp,160(%r8)+ movq %r12,216(%r8)+ movq %r13,224(%r8)+ movq %r14,232(%r8)+ movq %r15,240(%r8)++ jmp .Lcommon_seh_tail+++.def avx_handler; .scl 3; .type 32; .endef+.p2align 4+avx_handler:+ .byte 0xf3,0x0f,0x1e,0xfa++ pushq %rsi+ pushq %rdi+ pushq %rbx+ pushq %rbp+ pushq %r12+ pushq %r13+ pushq %r14+ pushq %r15+ pushfq+ subq $64,%rsp++ movq 120(%r8),%rax+ movq 248(%r8),%rbx++ movq 8(%r9),%rsi+ movq 56(%r9),%r11++ movl 0(%r11),%r10d+ leaq (%rsi,%r10,1),%r10+ cmpq %r10,%rbx+ jb .Lcommon_seh_tail++ movq 152(%r8),%rax++ movl 4(%r11),%r10d+ leaq (%rsi,%r10,1),%r10+ cmpq %r10,%rbx+ jae .Lcommon_seh_tail++ movq 208(%r8),%rax++ leaq 80(%rax),%rsi+ leaq 248(%rax),%rax+ leaq 512(%r8),%rdi+ movl $20,%ecx+.long 0xa548f3fc++.Lcommon_seh_tail:+ movq 8(%rax),%rdi+ movq 16(%rax),%rsi+ movq %rax,152(%r8)+ movq %rsi,168(%r8)+ movq %rdi,176(%r8)++ movq 40(%r9),%rdi+ movq %r8,%rsi+ movl $154,%ecx+.long 0xa548f3fc++ movq %r9,%rsi+ xorq %rcx,%rcx+ movq 8(%rsi),%rdx+ movq 0(%rsi),%r8+ movq 16(%rsi),%r9+ movq 40(%rsi),%r10+ leaq 56(%rsi),%r11+ leaq 24(%rsi),%r12+ movq %r10,32(%rsp)+ movq %r11,40(%rsp)+ movq %r12,48(%rsp)+ movq %rcx,56(%rsp)+ call *__imp_RtlVirtualUnwind(%rip)++ movl $1,%eax+ addq $64,%rsp+ popfq+ popq %r15+ popq %r14+ popq %r13+ popq %r12+ popq %rbp+ popq %rbx+ popq %rdi+ popq %rsi+ .byte 0xf3,0xc3+++.section .pdata+.p2align 2+.rva .LSEH_begin_crypton_poly1305_asm_init+.rva .LSEH_end_crypton_poly1305_asm_init+.rva .LSEH_info_crypton_poly1305_asm_init++.rva .LSEH_begin_crypton_poly1305_asm_blocks+.rva .LSEH_end_crypton_poly1305_asm_blocks+.rva .LSEH_info_crypton_poly1305_asm_blocks++.rva .LSEH_begin_crypton_poly1305_asm_emit+.rva .LSEH_end_crypton_poly1305_asm_emit+.rva .LSEH_info_crypton_poly1305_asm_emit+.rva .LSEH_begin_crypton_poly1305_asm_blocks_avx+.rva .Lbase2_64_avx+.rva .LSEH_info_crypton_poly1305_asm_blocks_avx_1++.rva .Lbase2_64_avx+.rva .Leven_avx+.rva .LSEH_info_crypton_poly1305_asm_blocks_avx_2++.rva .Leven_avx+.rva .LSEH_end_crypton_poly1305_asm_blocks_avx+.rva .LSEH_info_crypton_poly1305_asm_blocks_avx_3+.rva .LSEH_begin_crypton_poly1305_asm_blocks_avx2+.rva .Lbase2_64_avx2+.rva .LSEH_info_crypton_poly1305_asm_blocks_avx2_1++.rva .Lbase2_64_avx2+.rva .Leven_avx2+.rva .LSEH_info_crypton_poly1305_asm_blocks_avx2_2++.rva .Leven_avx2+.rva .LSEH_end_crypton_poly1305_asm_blocks_avx2+.rva .LSEH_info_crypton_poly1305_asm_blocks_avx2_3+.section .xdata+.p2align 3+.LSEH_info_crypton_poly1305_asm_init:+.byte 9,0,0,0+.rva se_handler+.long 0,0++.LSEH_info_crypton_poly1305_asm_blocks:+.byte 9,0,0,0+.rva se_handler+.rva .Lblocks_body,.Lblocks_epilogue++.LSEH_info_crypton_poly1305_asm_emit:+.byte 9,0,0,0+.rva se_handler+.long 0,0+.LSEH_info_crypton_poly1305_asm_blocks_avx_1:+.byte 9,0,0,0+.rva se_handler+.rva .Lblocks_avx_body,.Lblocks_avx_epilogue++.LSEH_info_crypton_poly1305_asm_blocks_avx_2:+.byte 9,0,0,0+.rva se_handler+.rva .Lbase2_64_avx_body,.Lbase2_64_avx_epilogue++.LSEH_info_crypton_poly1305_asm_blocks_avx_3:+.byte 9,0,0,0+.rva avx_handler+.rva .Ldo_avx_body,.Ldo_avx_epilogue+.LSEH_info_crypton_poly1305_asm_blocks_avx2_1:+.byte 9,0,0,0+.rva se_handler+.rva .Lblocks_avx2_body,.Lblocks_avx2_epilogue++.LSEH_info_crypton_poly1305_asm_blocks_avx2_2:+.byte 9,0,0,0+.rva se_handler+.rva .Lbase2_64_avx2_body,.Lbase2_64_avx2_epilogue++.LSEH_info_crypton_poly1305_asm_blocks_avx2_3:+.byte 9,0,0,0+.rva avx_handler+.rva .Ldo_avx2_body,.Ldo_avx2_epilogue
@@ -0,0 +1,4333 @@+#!/usr/bin/env perl+#+# ====================================================================+# Written by Andy Polyakov, @dot-asm, initially for the OpenSSL+# project.+# ====================================================================+#+# This module implements Poly1305 hash for x86_64.+#+# March 2015+#+# Initial release.+#+# December 2016+#+# Add AVX512F+VL+BW code path.+#+# November 2017+#+# Convert AVX512F+VL+BW code path to pure AVX512F, so that it can be+# executed even on Knights Landing. Trigger for modification was+# observation that AVX512 code paths can negatively affect overall+# Skylake-X system performance. Since we are likely to suppress+# AVX512F capability flag [at least on Skylake-X], conversion serves+# as kind of "investment protection". Note that next *lake processor,+# Cannonlake, has AVX512IFMA code path to execute...+#+# Numbers are cycles per processed byte with poly1305_blocks alone,+# most are measured with rdtsc at fixed clock frequency.+#+# IALU/gcc-4.8(i) AVX(ii) AVX2 AVX-512+# P4 4.46/+120% -+# Core 2 2.41/+90% -+# Westmere 1.88/+120% -+# Sandy Bridge 1.39/+140% 1.10+# Haswell 1.14/+175% 1.11 0.65+# Skylake[-X] 1.13/+120% 0.96 0.51 [0.35]+# Cannon Lake 1.13/+120% 0.93 0.38(iv)0.24(iv)+# Rocket Lake 1.13/+120% 0.84 0.43(iv)0.24(iv)+# Silvermont 2.83/+95% -+# Knights L 3.60/? 1.65 1.10 0.41(iii)+# Goldmont 1.70/+180% -+# VIA Nano 1.82/+150% -+# Sledgehammer 1.38/+160% -+# Bulldozer 2.30/+130% 0.97+# Ryzen 1.15/+200% 1.08 1.18+#+# (i) improvement coefficients relative to clang are more modest and+# are ~50% on most processors, in both cases we are comparing to+# __int128 code;+# (ii) SSE2 implementation was attempted, but among non-AVX processors+# it was faster than integer-only code only on older Intel P4 and+# Core processors, 50-30%, less newer processor is, but slower on+# contemporary ones, for example almost 2x slower on Atom, and as+# former are naturally disappearing, SSE2 is deemed unnecessary;+# (iii) strangely enough performance seems to vary from core to core,+# listed result is best case;+# (iv) these are IFMA results, which in addition means that first IALU+# column does not reflect short-input performance;++$flavour = shift;+$output = shift;+if ($flavour =~ /\./) { $output = $flavour; undef $flavour; }++$win64=0; $win64=1 if ($flavour =~ /[nm]asm|mingw64/ || $output =~ /\.asm$/);++$0 =~ m/(.*[\/\\])[^\/\\]+$/; $dir=$1;+( $xlate="${dir}x86_64-xlate.pl" and -f $xlate ) or+( $xlate="${dir}../../perlasm/x86_64-xlate.pl" and -f $xlate) or+die "can't locate x86_64-xlate.pl";++$avx=undef;++if (!defined($avx) && $win64 && ($flavour =~ /nasm/ || $ENV{ASM} =~ /nasm/) &&+ ($ENV{ASM} //= "nasm") &&+ `"$ENV{ASM}" -v 2>&1` =~ /NASM version ([0-9]+\.[0-9]+)(?:\.([0-9]+))?/) {+ $avx = ($1>=2.09) + ($1>=2.10) + 2 * ($1>=2.12);+ $avx += 2 if ($1==2.11 && $2>=8);+}++if (!defined($avx) && $win64 && ($flavour =~ /masm/ || $ENV{ASM} =~ /ml64/) &&+ ($ENV{ASM} //= "ml64") &&+ `"$ENV{ASM}" 2>&1` =~ /Version ([0-9]+)\./) {+ $avx = ($1>=10) + ($1>=12) + 2 * ($1>=14);+}++$ENV{CC} //= "cc";+if (!defined($avx) && `$ENV{CC} -Wa,-v -c -o /dev/zero -x assembler /dev/null 2>&1`+ =~ /GNU assembler version ([0-9]+)\.([0-9]+)/) {+ my $ver = $1 + $2/100.0; # 3.1->3.01, 3.10->3.10+ $avx = ($ver>=2.19) + ($ver>=2.22) + ($ver>=2.25) + ($ver>=2.26);+}++if (!defined($avx) && `$ENV{CC} -v 2>&1`+ =~ /((?:^clang|LLVM) version|.*based on LLVM) ([0-9]+)\.([0-9]+)/) {+ my $ver = $2 + $3/100.0; # 3.1->3.01, 3.10->3.10+ $avx = ($ver>=3.0) + ($ver>3.0);+ $avx += 2*($ver>=7.0) if ($1 =~ /^clang/);+}++open OUT,"| \"$^X\" \"$xlate\" $flavour \"$output\"";+*STDOUT=*OUT;++my ($ctx,$inp,$len,$padbit)=("%rdi","%rsi","%rdx","%rcx");+my ($mac,$nonce)=($inp,$len); # *_emit arguments+my ($d1,$d2,$d3, $r0,$r1,$s1)=map("%r$_",(8..13));+my ($h0,$h1,$h2)=("%r14","%rbx","%rbp");++sub poly1305_iteration {+# input: copy of $r1 in %rax, $h0-$h2, $r0-$r1+# output: $h0-$h2 *= $r0-$r1+$code.=<<___;+ mulq $h0 # h0*r1+ mov %rax,$d2+ mov $r0,%rax+ mov %rdx,$d3++ mulq $h0 # h0*r0+ mov %rax,$h0 # future $h0+ mov $r0,%rax+ mov %rdx,$d1++ mulq $h1 # h1*r0+ add %rax,$d2+ mov $s1,%rax+ adc %rdx,$d3++ mulq $h1 # h1*s1+ mov $h2,$h1 # borrow $h1+ add %rax,$h0+ adc %rdx,$d1++ imulq $s1,$h1 # h2*s1+ add $h1,$d2+ mov $d1,$h1+ adc \$0,$d3++ imulq $r0,$h2 # h2*r0+ add $d2,$h1+ mov \$-4,%rax # mask value+ adc $h2,$d3++ and $d3,%rax # last reduction step+ mov $d3,$h2+ shr \$2,$d3+ and \$3,$h2+ add $d3,%rax+ add %rax,$h0+ adc \$0,$h1+ adc \$0,$h2+___+}++########################################################################+# Layout of opaque area is following.+#+# unsigned __int64 h[3]; # current hash value base 2^64+# unsigned __int64 r[2]; # key value base 2^64++if ($flavour =~ /kernel/) {+$code.=<<___ if ($avx);+.globl poly1305_blocks_avx+___+$code.=<<___ if ($avx>1);+.globl poly1305_blocks_avx2+___+$code.=<<___ if ($avx>3);+.globl poly1305_init_base2_44+.globl poly1305_blocks_base2_44+.globl poly1305_emit_base2_44+.globl poly1305_blocks_vpmadd52+___+}+$code.=<<___;+.text++.extern OPENSSL_ia32cap_P++.globl poly1305_init+.hidden poly1305_init+.globl poly1305_blocks+.hidden poly1305_blocks+.globl poly1305_emit+.hidden poly1305_emit++.type poly1305_init,\@function,3+.align 32+poly1305_init:+ xor %rax,%rax+ mov %rax,0($ctx) # initialize hash value+ mov %rax,8($ctx)+ mov %rax,16($ctx) # [along with is_base2_26]++ cmp \$0,$inp+ je .Lno_key++ mov \$0x0ffffffc0fffffff,%rax+ lea -3(%rax),%rcx # $0x0ffffffc0ffffffc+ and 0($inp),%rax+ and 8($inp),%rcx+ mov %rax,24($ctx)+ mov %rcx,32($ctx)+___+$code.=<<___ if ($avx);+ movl \$-1,48($ctx) # write impossible value+___+ if ($flavour !~ /kernel/) {+$code.=<<___;+ lea poly1305_blocks(%rip),%r10+ lea poly1305_emit(%rip),%r11+___+$code.=<<___ if ($avx);+ mov OPENSSL_ia32cap_P+4(%rip),%r9+ lea poly1305_blocks_avx(%rip),%rax+ bt \$`60-32`,%r9 # AVX?+ cmovc %rax,%r10+___+$code.=<<___ if ($avx>1);+ lea poly1305_blocks_avx2(%rip),%rax+ bt \$`5+32`,%r9 # AVX2?+ cmovc %rax,%r10+___+$code.=<<___ if ($avx>3);+ mov \$`(1<<31|1<<21)`,%rax # AVX512VL|AVX512IFMA+ shr \$32,%r9+ and %rax,%r9+ cmp %rax,%r9+ je .Linit_base2_44+___+$code.=<<___ if ($flavour !~ /elf32/);+ mov %r10,0(%rdx)+ mov %r11,8(%rdx)+___+$code.=<<___ if ($flavour =~ /elf32/);+ mov %r10d,0(%rdx)+ mov %r11d,4(%rdx)+___+ }+$code.=<<___;+ mov \$1,%eax+.Lno_key:+ ret+.size poly1305_init,.-poly1305_init++.type poly1305_blocks,\@function,4+.align 32+poly1305_blocks:+.cfi_startproc+.Lblocks:+ shr \$4,$len+ jz .Lno_data # too short++ push %rbx+.cfi_push %rbx+ push %rbp+.cfi_push %rbp+ push %r12+.cfi_push %r12+ push %r13+.cfi_push %r13+ push %r14+.cfi_push %r14+ push %r15+.cfi_push %r15+ lea -8(%rsp),%rsp+.cfi_adjust_cfa_offset 8+.Lblocks_body:++ mov $len,%r15 # reassign $len++ mov 24($ctx),$r0 # load r+ mov 32($ctx),$s1++ mov 0($ctx),$h0 # load hash value base 2^64+ mov 8($ctx),$h1+ mov 16($ctx),$h2 # [along with is_base2_26]++ mov $h0#d,%eax # load hash value base 2^26+ mov 4($ctx),%edx+ mov $h1#d,%r8d+ mov 12($ctx),%r10d+ mov $h2#d,%r12d++ shl \$26,%rdx # base 2^26 -> base 2^64+ mov %r8,%r9+ shl \$52,%r8+ add %rdx,%rax+ shr \$12,%r9+ add %rax,%r8 # h0+ adc \$0,%r9++ shl \$14,%r10+ mov %r12,%rax+ shr \$24,%r12+ add %r10,%r9+ shl \$40,%rax+ add %rax,%r9 # h1+ adc \$0,%r12 # h2++ cmp \$4,$h2 # is_base2_26? [4 is as good as 2^32-1]++ cmova %r8,$h0 # choose between radixes+ cmova %r9,$h1+ cmova %r12,$h2++ mov $s1,$r1+ shr \$2,$s1+ mov $r1,%rax+ add $r1,$s1 # s1 = r1 + (r1 >> 2)+ jmp .Loop++.align 32+.Loop:+ add 0($inp),$h0 # accumulate input+ adc 8($inp),$h1+ lea 16($inp),$inp+ adc $padbit,$h2+___+ &poly1305_iteration();+$code.=<<___;+ mov $r1,%rax+ dec %r15 # len-=16+ jnz .Loop++ mov $h0,0($ctx) # store hash value+ mov $h1,8($ctx)+ mov $h2,16($ctx)++ mov 8(%rsp),%r15+.cfi_restore %r15+ mov 16(%rsp),%r14+.cfi_restore %r14+ mov 24(%rsp),%r13+.cfi_restore %r13+ mov 32(%rsp),%r12+.cfi_restore %r12+ mov 40(%rsp),%rbp+.cfi_restore %rbp+ mov 48(%rsp),%rbx+.cfi_restore %rbx+ lea 56(%rsp),%rsp+.cfi_adjust_cfa_offset -56+.Lno_data:+.Lblocks_epilogue:+ ret+.cfi_endproc+.size poly1305_blocks,.-poly1305_blocks++.type poly1305_emit,\@function,3+.align 32+poly1305_emit:+ mov 0($ctx),%eax # load hash value base 2^26+ mov 4($ctx),%ecx+ mov 8($ctx),%r8d+ mov 12($ctx),%r11d+ mov 16($ctx),%r10d++ shl \$26,%rcx # base 2^26 -> base 2^64+ mov %r8,%r9+ shl \$52,%r8+ add %rcx,%rax+ shr \$12,%r9+ add %rax,%r8 # h0+ adc \$0,%r9++ shl \$14,%r11+ mov %r10,%rax+ shr \$24,%r10+ add %r11,%r9+ mov 0($ctx),%rcx # load hash value base 2^64+ shl \$40,%rax+ mov 8($ctx),%r11+ add %rax,%r9 # h1+ mov 16($ctx),%rax # [along with is_base2_26]+ adc \$0,%r10 # h2++ cmp \$4,%rax # is_base2_26? [4 is as good as 2^32-1]++ cmovbe %rcx,%r8 # choose between radixes+ cmovbe %r11,%r9+ cmovbe %rax,%r10++ mov %r8,%rax+ add \$5,%r8 # compare to modulus+ mov %r9,%rcx+ adc \$0,%r9+ adc \$0,%r10+ shr \$2,%r10 # did 130-bit value overflow?+ cmovnz %r8,%rax+ cmovnz %r9,%rcx++ add 0($nonce),%rax # accumulate nonce+ adc 8($nonce),%rcx+ mov %rax,0($mac) # write result+ mov %rcx,8($mac)++ ret+.size poly1305_emit,.-poly1305_emit+___+if ($avx) {++########################################################################+# Layout of opaque area is following.+#+# unsigned __int32 h[5]; # current hash value base 2^26+# unsigned __int32 is_base2_26;+# unsigned __int64 r[2]; # key value base 2^64+# unsigned __int64 pad;+# struct { unsigned __int32 r^2, r^1, r^4, r^3; } r[9];+#+# where r^n are base 2^26 digits of degrees of multiplier key. There are+# 5 digits, but last four are interleaved with multiples of 5, totalling+# in 9 elements: r0, r1, 5*r1, r2, 5*r2, r3, 5*r3, r4, 5*r4.++my ($H0,$H1,$H2,$H3,$H4, $T0,$T1,$T2,$T3,$T4, $D0,$D1,$D2,$D3,$D4, $MASK) =+ map("%xmm$_",(0..15));++$code.=<<___;+.type __poly1305_block,\@abi-omnipotent+.align 32+__poly1305_block:+___+ &poly1305_iteration();+$code.=<<___;+ ret+.size __poly1305_block,.-__poly1305_block++.type __poly1305_init_avx,\@abi-omnipotent+.align 32+__poly1305_init_avx:+ cmpl \$-1,48($ctx)+ jne .Ldone_init_avx++ mov $r0,$h0+ mov $r1,$h1+ xor $h2,$h2++ lea 48+64($ctx),$ctx # size optimization++ mov $r1,%rax+ call __poly1305_block # r^2++ mov \$0x3ffffff,%eax # save interleaved r^2 and r base 2^26+ mov \$0x3ffffff,%edx+ mov $h0,$d1+ and $h0#d,%eax+ mov $r0,$d2+ and $r0#d,%edx+ mov %eax,`16*0+0-64`($ctx)+ shr \$26,$d1+ mov %edx,`16*0+4-64`($ctx)+ shr \$26,$d2++ mov \$0x3ffffff,%eax+ mov \$0x3ffffff,%edx+ and $d1#d,%eax+ and $d2#d,%edx+ mov %eax,`16*1+0-64`($ctx)+ lea (%rax,%rax,4),%eax # *5+ mov %edx,`16*1+4-64`($ctx)+ lea (%rdx,%rdx,4),%edx # *5+ mov %eax,`16*2+0-64`($ctx)+ shr \$26,$d1+ mov %edx,`16*2+4-64`($ctx)+ shr \$26,$d2++ mov $h1,%rax+ mov $r1,%rdx+ shl \$12,%rax+ shl \$12,%rdx+ or $d1,%rax+ or $d2,%rdx+ and \$0x3ffffff,%eax+ and \$0x3ffffff,%edx+ mov %eax,`16*3+0-64`($ctx)+ lea (%rax,%rax,4),%eax # *5+ mov %edx,`16*3+4-64`($ctx)+ lea (%rdx,%rdx,4),%edx # *5+ mov %eax,`16*4+0-64`($ctx)+ mov $h1,$d1+ mov %edx,`16*4+4-64`($ctx)+ mov $r1,$d2++ mov \$0x3ffffff,%eax+ mov \$0x3ffffff,%edx+ shr \$14,$d1+ shr \$14,$d2+ and $d1#d,%eax+ and $d2#d,%edx+ mov %eax,`16*5+0-64`($ctx)+ lea (%rax,%rax,4),%eax # *5+ mov %edx,`16*5+4-64`($ctx)+ lea (%rdx,%rdx,4),%edx # *5+ mov %eax,`16*6+0-64`($ctx)+ shr \$26,$d1+ mov %edx,`16*6+4-64`($ctx)+ shr \$26,$d2++ mov $h2,%rax+ shl \$24,%rax+ or %rax,$d1+ mov $d1#d,`16*7+0-64`($ctx)+ lea ($d1,$d1,4),$d1 # *5+ mov $d2#d,`16*7+4-64`($ctx)+ lea ($d2,$d2,4),$d2 # *5+ mov $d1#d,`16*8+0-64`($ctx)+ mov $d2#d,`16*8+4-64`($ctx)++ mov $r1,%rax+ call __poly1305_block # r^3++ mov \$0x3ffffff,%eax # save r^3 base 2^26+ mov $h0,$d1+ and $h0#d,%eax+ shr \$26,$d1+ mov %eax,`16*0+12-64`($ctx)++ mov \$0x3ffffff,%edx+ and $d1#d,%edx+ mov %edx,`16*1+12-64`($ctx)+ lea (%rdx,%rdx,4),%edx # *5+ shr \$26,$d1+ mov %edx,`16*2+12-64`($ctx)++ mov $h1,%rax+ shl \$12,%rax+ or $d1,%rax+ and \$0x3ffffff,%eax+ mov %eax,`16*3+12-64`($ctx)+ lea (%rax,%rax,4),%eax # *5+ mov $h1,$d1+ mov %eax,`16*4+12-64`($ctx)++ mov \$0x3ffffff,%edx+ shr \$14,$d1+ and $d1#d,%edx+ mov %edx,`16*5+12-64`($ctx)+ lea (%rdx,%rdx,4),%edx # *5+ shr \$26,$d1+ mov %edx,`16*6+12-64`($ctx)++ mov $h2,%rax+ shl \$24,%rax+ or %rax,$d1+ mov $d1#d,`16*7+12-64`($ctx)+ lea ($d1,$d1,4),$d1 # *5+ mov $d1#d,`16*8+12-64`($ctx)++ mov $r1,%rax+ call __poly1305_block # r^4++ mov \$0x3ffffff,%eax # save r^4 base 2^26+ mov $h0,$d1+ and $h0#d,%eax+ shr \$26,$d1+ mov %eax,`16*0+8-64`($ctx)++ mov \$0x3ffffff,%edx+ and $d1#d,%edx+ mov %edx,`16*1+8-64`($ctx)+ lea (%rdx,%rdx,4),%edx # *5+ shr \$26,$d1+ mov %edx,`16*2+8-64`($ctx)++ mov $h1,%rax+ shl \$12,%rax+ or $d1,%rax+ and \$0x3ffffff,%eax+ mov %eax,`16*3+8-64`($ctx)+ lea (%rax,%rax,4),%eax # *5+ mov $h1,$d1+ mov %eax,`16*4+8-64`($ctx)++ mov \$0x3ffffff,%edx+ shr \$14,$d1+ and $d1#d,%edx+ mov %edx,`16*5+8-64`($ctx)+ lea (%rdx,%rdx,4),%edx # *5+ shr \$26,$d1+ mov %edx,`16*6+8-64`($ctx)++ mov $h2,%rax+ shl \$24,%rax+ or %rax,$d1+ mov $d1#d,`16*7+8-64`($ctx)+ lea ($d1,$d1,4),$d1 # *5+ mov $d1#d,`16*8+8-64`($ctx)++ lea -48-64($ctx),$ctx # size [de-]optimization+.Ldone_init_avx:+ ret+.size __poly1305_init_avx,.-__poly1305_init_avx++.type poly1305_blocks_avx,\@function,4+.align 32+poly1305_blocks_avx:+.cfi_startproc+ mov 20($ctx),%r8d # load is_base2_26+ cmp \$128,$len+ jb .Lblocks++ and \$-16,$len++ vzeroupper++ test %r8d,%r8d # is_base2_26?+ jz .Lbase2_64_avx++ test \$31,$len+ jz .Leven_avx++ push %rbx+.cfi_push %rbx+ push %rbp+.cfi_push %rbp+ push %r12+.cfi_push %r12+ push %r13+.cfi_push %r13+ push %r14+.cfi_push %r14+ push %r15+.cfi_push %r15+ lea -8(%rsp),%rsp+.cfi_adjust_cfa_offset 8+.Lblocks_avx_body:++ mov $len,%r15 # reassign $len++ mov 0($ctx),$d1 # load hash value+ mov 8($ctx),$d2+ mov 16($ctx),$h2#d++ mov 24($ctx),$r0 # load r+ mov 32($ctx),$s1++ ################################# base 2^26 -> base 2^64+ mov $d1#d,$h0#d+ and \$`-1*(1<<31)`,$d1+ mov $d2,$r1 # borrow $r1+ mov $d2#d,$h1#d+ and \$`-1*(1<<31)`,$d2++ shr \$6,$d1+ shl \$52,$r1+ add $d1,$h0+ shr \$12,$h1+ shr \$18,$d2+ add $r1,$h0+ adc $d2,$h1++ mov $h2,$d1+ shl \$40,$d1+ shr \$24,$h2+ add $d1,$h1+ adc \$0,$h2 # can be partially reduced...++ mov $s1,$r1+ mov $s1,%rax+ shr \$2,$s1+ add $r1,$s1 # s1 = r1 + (r1 >> 2)++ add 0($inp),$h0 # accumulate input+ adc 8($inp),$h1+ lea 16($inp),$inp+ adc $padbit,$h2++ call __poly1305_block++ ################################# base 2^64 -> base 2^26+ mov $h0,%rax+ mov $h0,%rdx+ shr \$52,$h0+ mov $h1,$r0+ mov $h1,$r1+ shr \$26,%rdx+ and \$0x3ffffff,%rax # h[0]+ shl \$12,$r0+ and \$0x3ffffff,%rdx # h[1]+ shr \$14,$h1+ or $r0,$h0+ shl \$24,$h2+ and \$0x3ffffff,$h0 # h[2]+ shr \$40,$r1+ and \$0x3ffffff,$h1 # h[3]+ or $r1,$h2 # h[4]++ vmovd %rax#d,$H0+ vmovd %rdx#d,$H1+ vmovd $h0#d,$H2+ vmovd $h1#d,$H3+ vmovd $h2#d,$H4++ lea -16(%r15),$len++ mov 8(%rsp),%r15+.cfi_restore %r15+ mov 16(%rsp),%r14+.cfi_restore %r14+ mov 24(%rsp),%r13+.cfi_restore %r13+ mov 32(%rsp),%r12+.cfi_restore %r12+ mov 40(%rsp),%rbp+.cfi_restore %rbp+ mov 48(%rsp),%rbx+.cfi_restore %rbx+ lea 56(%rsp),%rax # for win64+ lea 56(%rsp),%rsp+.cfi_adjust_cfa_offset -56+.Lblocks_avx_epilogue:+ jmp .Ldo_avx+.cfi_endproc++.align 32+.Lbase2_64_avx:+.cfi_startproc+ push %rbx+.cfi_push %rbx+ push %rbp+.cfi_push %rbp+ push %r12+.cfi_push %r12+ push %r13+.cfi_push %r13+ push %r14+.cfi_push %r14+ push %r15+.cfi_push %r15+ lea -8(%rsp),%rsp+.cfi_adjust_cfa_offset 8+.Lbase2_64_avx_body:++ mov $len,%r15 # reassign $len++ mov 24($ctx),$r0 # load r+ mov 32($ctx),$s1++ mov 0($ctx),$h0 # load hash value+ mov 8($ctx),$h1+ mov 16($ctx),$h2#d++ mov $s1,$r1+ mov $s1,%rax+ shr \$2,$s1+ add $r1,$s1 # s1 = r1 + (r1 >> 2)++ test \$31,$len+ jz .Linit_avx++ add 0($inp),$h0 # accumulate input+ adc 8($inp),$h1+ lea 16($inp),$inp+ adc $padbit,$h2+ sub \$16,%r15++ call __poly1305_block++.Linit_avx:+ ################################# base 2^64 -> base 2^26+ mov $h0,%rax+ mov $h0,%rdx+ shr \$52,$h0+ mov $h1,$d1+ mov $h1,$d2+ shr \$26,%rdx+ and \$0x3ffffff,%rax # h[0]+ shl \$12,$d1+ and \$0x3ffffff,%rdx # h[1]+ shr \$14,$h1+ or $d1,$h0+ shl \$24,$h2+ and \$0x3ffffff,$h0 # h[2]+ shr \$40,$d2+ and \$0x3ffffff,$h1 # h[3]+ or $d2,$h2 # h[4]++ vmovd %rax#d,$H0+ vmovd %rdx#d,$H1+ vmovd $h0#d,$H2+ vmovd $h1#d,$H3+ vmovd $h2#d,$H4+ movl \$1,20($ctx) # set is_base2_26++ call __poly1305_init_avx++ mov %r15,$len++ mov 8(%rsp),%r15+.cfi_restore %r15+ mov 16(%rsp),%r14+.cfi_restore %r14+ mov 24(%rsp),%r13+.cfi_restore %r13+ mov 32(%rsp),%r12+.cfi_restore %r12+ mov 40(%rsp),%rbp+.cfi_restore %rbp+ mov 48(%rsp),%rbx+.cfi_restore %rbx+ lea 56(%rsp),%rax # for win64+ lea 56(%rsp),%rsp+.cfi_adjust_cfa_offset -56+.Lbase2_64_avx_epilogue:+ jmp .Ldo_avx+.cfi_endproc++.align 32+.Leven_avx:+.cfi_startproc+ vmovd 4*0($ctx),$H0 # load hash value+ vmovd 4*1($ctx),$H1+ vmovd 4*2($ctx),$H2+ vmovd 4*3($ctx),$H3+ vmovd 4*4($ctx),$H4++.Ldo_avx:+___+$code.=<<___ if (!$win64);+ lea -0x58(%rsp),%r11+.cfi_def_cfa %r11,0x60+ sub \$0x178,%rsp+___+$code.=<<___ if ($win64);+ lea -0xf8(%rsp),%r11+ sub \$0x218,%rsp+ vmovdqa %xmm6,0x50(%r11)+ vmovdqa %xmm7,0x60(%r11)+ vmovdqa %xmm8,0x70(%r11)+ vmovdqa %xmm9,0x80(%r11)+ vmovdqa %xmm10,0x90(%r11)+ vmovdqa %xmm11,0xa0(%r11)+ vmovdqa %xmm12,0xb0(%r11)+ vmovdqa %xmm13,0xc0(%r11)+ vmovdqa %xmm14,0xd0(%r11)+ vmovdqa %xmm15,0xe0(%r11)+.Ldo_avx_body:+___+$code.=<<___;+ sub \$64,$len+ lea -32($inp),%rax+ cmovc %rax,$inp++ vmovdqu `16*3`($ctx),$D4 # preload r0^2+ lea `16*3+64`($ctx),$ctx # size optimization+ lea .Lconst(%rip),%rcx++ ################################################################+ # load input+ vmovdqu 16*2($inp),$T0+ vmovdqu 16*3($inp),$T1+ vmovdqa 64(%rcx),$MASK # .Lmask26++ vpsrldq \$6,$T0,$T2 # splat input+ vpsrldq \$6,$T1,$T3+ vpunpckhqdq $T1,$T0,$T4 # 4+ vpunpcklqdq $T1,$T0,$T0 # 0:1+ vpunpcklqdq $T3,$T2,$T3 # 2:3++ vpsrlq \$40,$T4,$T4 # 4+ vpsrlq \$26,$T0,$T1+ vpand $MASK,$T0,$T0 # 0+ vpsrlq \$4,$T3,$T2+ vpand $MASK,$T1,$T1 # 1+ vpsrlq \$30,$T3,$T3+ vpand $MASK,$T2,$T2 # 2+ vpand $MASK,$T3,$T3 # 3+ vpor 32(%rcx),$T4,$T4 # padbit, yes, always++ jbe .Lskip_loop_avx++ # expand and copy pre-calculated table to stack+ vmovdqu `16*1-64`($ctx),$D1+ vmovdqu `16*2-64`($ctx),$D2+ vpshufd \$0xEE,$D4,$D3 # 34xx -> 3434+ vpshufd \$0x44,$D4,$D0 # xx12 -> 1212+ vmovdqa $D3,-0x90(%r11)+ vmovdqa $D0,0x00(%rsp)+ vpshufd \$0xEE,$D1,$D4+ vmovdqu `16*3-64`($ctx),$D0+ vpshufd \$0x44,$D1,$D1+ vmovdqa $D4,-0x80(%r11)+ vmovdqa $D1,0x10(%rsp)+ vpshufd \$0xEE,$D2,$D3+ vmovdqu `16*4-64`($ctx),$D1+ vpshufd \$0x44,$D2,$D2+ vmovdqa $D3,-0x70(%r11)+ vmovdqa $D2,0x20(%rsp)+ vpshufd \$0xEE,$D0,$D4+ vmovdqu `16*5-64`($ctx),$D2+ vpshufd \$0x44,$D0,$D0+ vmovdqa $D4,-0x60(%r11)+ vmovdqa $D0,0x30(%rsp)+ vpshufd \$0xEE,$D1,$D3+ vmovdqu `16*6-64`($ctx),$D0+ vpshufd \$0x44,$D1,$D1+ vmovdqa $D3,-0x50(%r11)+ vmovdqa $D1,0x40(%rsp)+ vpshufd \$0xEE,$D2,$D4+ vmovdqu `16*7-64`($ctx),$D1+ vpshufd \$0x44,$D2,$D2+ vmovdqa $D4,-0x40(%r11)+ vmovdqa $D2,0x50(%rsp)+ vpshufd \$0xEE,$D0,$D3+ vmovdqu `16*8-64`($ctx),$D2+ vpshufd \$0x44,$D0,$D0+ vmovdqa $D3,-0x30(%r11)+ vmovdqa $D0,0x60(%rsp)+ vpshufd \$0xEE,$D1,$D4+ vpshufd \$0x44,$D1,$D1+ vmovdqa $D4,-0x20(%r11)+ vmovdqa $D1,0x70(%rsp)+ vpshufd \$0xEE,$D2,$D3+ vmovdqa 0x00(%rsp),$D4 # preload r0^2+ vpshufd \$0x44,$D2,$D2+ vmovdqa $D3,-0x10(%r11)+ vmovdqa $D2,0x80(%rsp)++ jmp .Loop_avx++.align 32+.Loop_avx:+ ################################################################+ # ((inp[0]*r^4+inp[2]*r^2+inp[4])*r^4+inp[6]*r^2+ # ((inp[1]*r^4+inp[3]*r^2+inp[5])*r^3+inp[7]*r+ # \___________________/+ # ((inp[0]*r^4+inp[2]*r^2+inp[4])*r^4+inp[6]*r^2+inp[8])*r^2+ # ((inp[1]*r^4+inp[3]*r^2+inp[5])*r^4+inp[7]*r^2+inp[9])*r+ # \___________________/ \____________________/+ #+ # Note that we start with inp[2:3]*r^2. This is because it+ # doesn't depend on reduction in previous iteration.+ ################################################################+ # d4 = h4*r0 + h3*r1 + h2*r2 + h1*r3 + h0*r4+ # d3 = h3*r0 + h2*r1 + h1*r2 + h0*r3 + h4*5*r4+ # d2 = h2*r0 + h1*r1 + h0*r2 + h4*5*r3 + h3*5*r4+ # d1 = h1*r0 + h0*r1 + h4*5*r2 + h3*5*r3 + h2*5*r4+ # d0 = h0*r0 + h4*5*r1 + h3*5*r2 + h2*5*r3 + h1*5*r4+ #+ # though note that $Tx and $Hx are "reversed" in this section,+ # and $D4 is preloaded with r0^2...++ vpmuludq $T0,$D4,$D0 # d0 = h0*r0+ vpmuludq $T1,$D4,$D1 # d1 = h1*r0+ vmovdqa $H2,0x20(%r11) # offload hash+ vpmuludq $T2,$D4,$D2 # d3 = h2*r0+ vmovdqa 0x10(%rsp),$H2 # r1^2+ vpmuludq $T3,$D4,$D3 # d3 = h3*r0+ vpmuludq $T4,$D4,$D4 # d4 = h4*r0++ vmovdqa $H0,0x00(%r11) #+ vpmuludq 0x20(%rsp),$T4,$H0 # h4*s1+ vmovdqa $H1,0x10(%r11) #+ vpmuludq $T3,$H2,$H1 # h3*r1+ vpaddq $H0,$D0,$D0 # d0 += h4*s1+ vpaddq $H1,$D4,$D4 # d4 += h3*r1+ vmovdqa $H3,0x30(%r11) #+ vpmuludq $T2,$H2,$H0 # h2*r1+ vpmuludq $T1,$H2,$H1 # h1*r1+ vpaddq $H0,$D3,$D3 # d3 += h2*r1+ vmovdqa 0x30(%rsp),$H3 # r2^2+ vpaddq $H1,$D2,$D2 # d2 += h1*r1+ vmovdqa $H4,0x40(%r11) #+ vpmuludq $T0,$H2,$H2 # h0*r1+ vpmuludq $T2,$H3,$H0 # h2*r2+ vpaddq $H2,$D1,$D1 # d1 += h0*r1++ vmovdqa 0x40(%rsp),$H4 # s2^2+ vpaddq $H0,$D4,$D4 # d4 += h2*r2+ vpmuludq $T1,$H3,$H1 # h1*r2+ vpmuludq $T0,$H3,$H3 # h0*r2+ vpaddq $H1,$D3,$D3 # d3 += h1*r2+ vmovdqa 0x50(%rsp),$H2 # r3^2+ vpaddq $H3,$D2,$D2 # d2 += h0*r2+ vpmuludq $T4,$H4,$H0 # h4*s2+ vpmuludq $T3,$H4,$H4 # h3*s2+ vpaddq $H0,$D1,$D1 # d1 += h4*s2+ vmovdqa 0x60(%rsp),$H3 # s3^2+ vpaddq $H4,$D0,$D0 # d0 += h3*s2++ vmovdqa 0x80(%rsp),$H4 # s4^2+ vpmuludq $T1,$H2,$H1 # h1*r3+ vpmuludq $T0,$H2,$H2 # h0*r3+ vpaddq $H1,$D4,$D4 # d4 += h1*r3+ vpaddq $H2,$D3,$D3 # d3 += h0*r3+ vpmuludq $T4,$H3,$H0 # h4*s3+ vpmuludq $T3,$H3,$H1 # h3*s3+ vpaddq $H0,$D2,$D2 # d2 += h4*s3+ vmovdqu 16*0($inp),$H0 # load input+ vpaddq $H1,$D1,$D1 # d1 += h3*s3+ vpmuludq $T2,$H3,$H3 # h2*s3+ vpmuludq $T2,$H4,$T2 # h2*s4+ vpaddq $H3,$D0,$D0 # d0 += h2*s3++ vmovdqu 16*1($inp),$H1 #+ vpaddq $T2,$D1,$D1 # d1 += h2*s4+ vpmuludq $T3,$H4,$T3 # h3*s4+ vpmuludq $T4,$H4,$T4 # h4*s4+ vpsrldq \$6,$H0,$H2 # splat input+ vpaddq $T3,$D2,$D2 # d2 += h3*s4+ vpaddq $T4,$D3,$D3 # d3 += h4*s4+ vpsrldq \$6,$H1,$H3 #+ vpmuludq 0x70(%rsp),$T0,$T4 # h0*r4+ vpmuludq $T1,$H4,$T0 # h1*s4+ vpunpckhqdq $H1,$H0,$H4 # 4+ vpaddq $T4,$D4,$D4 # d4 += h0*r4+ vmovdqa -0x90(%r11),$T4 # r0^4+ vpaddq $T0,$D0,$D0 # d0 += h1*s4++ vpunpcklqdq $H1,$H0,$H0 # 0:1+ vpunpcklqdq $H3,$H2,$H3 # 2:3++ #vpsrlq \$40,$H4,$H4 # 4+ vpsrldq \$`40/8`,$H4,$H4 # 4+ vpsrlq \$26,$H0,$H1+ vpand $MASK,$H0,$H0 # 0+ vpsrlq \$4,$H3,$H2+ vpand $MASK,$H1,$H1 # 1+ vpand 0(%rcx),$H4,$H4 # .Lmask24+ vpsrlq \$30,$H3,$H3+ vpand $MASK,$H2,$H2 # 2+ vpand $MASK,$H3,$H3 # 3+ vpor 32(%rcx),$H4,$H4 # padbit, yes, always++ vpaddq 0x00(%r11),$H0,$H0 # add hash value+ vpaddq 0x10(%r11),$H1,$H1+ vpaddq 0x20(%r11),$H2,$H2+ vpaddq 0x30(%r11),$H3,$H3+ vpaddq 0x40(%r11),$H4,$H4++ lea 16*2($inp),%rax+ lea 16*4($inp),$inp+ sub \$64,$len+ cmovc %rax,$inp++ ################################################################+ # Now we accumulate (inp[0:1]+hash)*r^4+ ################################################################+ # d4 = h4*r0 + h3*r1 + h2*r2 + h1*r3 + h0*r4+ # d3 = h3*r0 + h2*r1 + h1*r2 + h0*r3 + h4*5*r4+ # d2 = h2*r0 + h1*r1 + h0*r2 + h4*5*r3 + h3*5*r4+ # d1 = h1*r0 + h0*r1 + h4*5*r2 + h3*5*r3 + h2*5*r4+ # d0 = h0*r0 + h4*5*r1 + h3*5*r2 + h2*5*r3 + h1*5*r4++ vpmuludq $H0,$T4,$T0 # h0*r0+ vpmuludq $H1,$T4,$T1 # h1*r0+ vpaddq $T0,$D0,$D0+ vpaddq $T1,$D1,$D1+ vmovdqa -0x80(%r11),$T2 # r1^4+ vpmuludq $H2,$T4,$T0 # h2*r0+ vpmuludq $H3,$T4,$T1 # h3*r0+ vpaddq $T0,$D2,$D2+ vpaddq $T1,$D3,$D3+ vpmuludq $H4,$T4,$T4 # h4*r0+ vpmuludq -0x70(%r11),$H4,$T0 # h4*s1+ vpaddq $T4,$D4,$D4++ vpaddq $T0,$D0,$D0 # d0 += h4*s1+ vpmuludq $H2,$T2,$T1 # h2*r1+ vpmuludq $H3,$T2,$T0 # h3*r1+ vpaddq $T1,$D3,$D3 # d3 += h2*r1+ vmovdqa -0x60(%r11),$T3 # r2^4+ vpaddq $T0,$D4,$D4 # d4 += h3*r1+ vpmuludq $H1,$T2,$T1 # h1*r1+ vpmuludq $H0,$T2,$T2 # h0*r1+ vpaddq $T1,$D2,$D2 # d2 += h1*r1+ vpaddq $T2,$D1,$D1 # d1 += h0*r1++ vmovdqa -0x50(%r11),$T4 # s2^4+ vpmuludq $H2,$T3,$T0 # h2*r2+ vpmuludq $H1,$T3,$T1 # h1*r2+ vpaddq $T0,$D4,$D4 # d4 += h2*r2+ vpaddq $T1,$D3,$D3 # d3 += h1*r2+ vmovdqa -0x40(%r11),$T2 # r3^4+ vpmuludq $H0,$T3,$T3 # h0*r2+ vpmuludq $H4,$T4,$T0 # h4*s2+ vpaddq $T3,$D2,$D2 # d2 += h0*r2+ vpaddq $T0,$D1,$D1 # d1 += h4*s2+ vmovdqa -0x30(%r11),$T3 # s3^4+ vpmuludq $H3,$T4,$T4 # h3*s2+ vpmuludq $H1,$T2,$T1 # h1*r3+ vpaddq $T4,$D0,$D0 # d0 += h3*s2++ vmovdqa -0x10(%r11),$T4 # s4^4+ vpaddq $T1,$D4,$D4 # d4 += h1*r3+ vpmuludq $H0,$T2,$T2 # h0*r3+ vpmuludq $H4,$T3,$T0 # h4*s3+ vpaddq $T2,$D3,$D3 # d3 += h0*r3+ vpaddq $T0,$D2,$D2 # d2 += h4*s3+ vmovdqu 16*2($inp),$T0 # load input+ vpmuludq $H3,$T3,$T2 # h3*s3+ vpmuludq $H2,$T3,$T3 # h2*s3+ vpaddq $T2,$D1,$D1 # d1 += h3*s3+ vmovdqu 16*3($inp),$T1 #+ vpaddq $T3,$D0,$D0 # d0 += h2*s3++ vpmuludq $H2,$T4,$H2 # h2*s4+ vpmuludq $H3,$T4,$H3 # h3*s4+ vpsrldq \$6,$T0,$T2 # splat input+ vpaddq $H2,$D1,$D1 # d1 += h2*s4+ vpmuludq $H4,$T4,$H4 # h4*s4+ vpsrldq \$6,$T1,$T3 #+ vpaddq $H3,$D2,$H2 # h2 = d2 + h3*s4+ vpaddq $H4,$D3,$H3 # h3 = d3 + h4*s4+ vpmuludq -0x20(%r11),$H0,$H4 # h0*r4+ vpmuludq $H1,$T4,$H0+ vpunpckhqdq $T1,$T0,$T4 # 4+ vpaddq $H4,$D4,$H4 # h4 = d4 + h0*r4+ vpaddq $H0,$D0,$H0 # h0 = d0 + h1*s4++ vpunpcklqdq $T1,$T0,$T0 # 0:1+ vpunpcklqdq $T3,$T2,$T3 # 2:3++ #vpsrlq \$40,$T4,$T4 # 4+ vpsrldq \$`40/8`,$T4,$T4 # 4+ vpsrlq \$26,$T0,$T1+ vmovdqa 0x00(%rsp),$D4 # preload r0^2+ vpand $MASK,$T0,$T0 # 0+ vpsrlq \$4,$T3,$T2+ vpand $MASK,$T1,$T1 # 1+ vpand 0(%rcx),$T4,$T4 # .Lmask24+ vpsrlq \$30,$T3,$T3+ vpand $MASK,$T2,$T2 # 2+ vpand $MASK,$T3,$T3 # 3+ vpor 32(%rcx),$T4,$T4 # padbit, yes, always++ ################################################################+ # lazy reduction as discussed in "NEON crypto" by D.J. Bernstein+ # and P. Schwabe++ vpsrlq \$26,$H3,$D3+ vpand $MASK,$H3,$H3+ vpaddq $D3,$H4,$H4 # h3 -> h4++ vpsrlq \$26,$H0,$D0+ vpand $MASK,$H0,$H0+ vpaddq $D0,$D1,$H1 # h0 -> h1++ vpsrlq \$26,$H4,$D0+ vpand $MASK,$H4,$H4++ vpsrlq \$26,$H1,$D1+ vpand $MASK,$H1,$H1+ vpaddq $D1,$H2,$H2 # h1 -> h2++ vpaddq $D0,$H0,$H0+ vpsllq \$2,$D0,$D0+ vpaddq $D0,$H0,$H0 # h4 -> h0++ vpsrlq \$26,$H2,$D2+ vpand $MASK,$H2,$H2+ vpaddq $D2,$H3,$H3 # h2 -> h3++ vpsrlq \$26,$H0,$D0+ vpand $MASK,$H0,$H0+ vpaddq $D0,$H1,$H1 # h0 -> h1++ vpsrlq \$26,$H3,$D3+ vpand $MASK,$H3,$H3+ vpaddq $D3,$H4,$H4 # h3 -> h4++ ja .Loop_avx++.Lskip_loop_avx:+ ################################################################+ # multiply (inp[0:1]+hash) or inp[2:3] by r^2:r^1++ vpshufd \$0x10,$D4,$D4 # r0^n, xx12 -> x1x2+ add \$32,$len+ jnz .Long_tail_avx++ vpaddq $H2,$T2,$T2+ vpaddq $H0,$T0,$T0+ vpaddq $H1,$T1,$T1+ vpaddq $H3,$T3,$T3+ vpaddq $H4,$T4,$T4++.Long_tail_avx:+ vmovdqa $H2,0x20(%r11)+ vmovdqa $H0,0x00(%r11)+ vmovdqa $H1,0x10(%r11)+ vmovdqa $H3,0x30(%r11)+ vmovdqa $H4,0x40(%r11)++ # d4 = h4*r0 + h3*r1 + h2*r2 + h1*r3 + h0*r4+ # d3 = h3*r0 + h2*r1 + h1*r2 + h0*r3 + h4*5*r4+ # d2 = h2*r0 + h1*r1 + h0*r2 + h4*5*r3 + h3*5*r4+ # d1 = h1*r0 + h0*r1 + h4*5*r2 + h3*5*r3 + h2*5*r4+ # d0 = h0*r0 + h4*5*r1 + h3*5*r2 + h2*5*r3 + h1*5*r4++ vpmuludq $T2,$D4,$D2 # d2 = h2*r0+ vpmuludq $T0,$D4,$D0 # d0 = h0*r0+ vpshufd \$0x10,`16*1-64`($ctx),$H2 # r1^n+ vpmuludq $T1,$D4,$D1 # d1 = h1*r0+ vpmuludq $T3,$D4,$D3 # d3 = h3*r0+ vpmuludq $T4,$D4,$D4 # d4 = h4*r0++ vpmuludq $T3,$H2,$H0 # h3*r1+ vpaddq $H0,$D4,$D4 # d4 += h3*r1+ vpshufd \$0x10,`16*2-64`($ctx),$H3 # s1^n+ vpmuludq $T2,$H2,$H1 # h2*r1+ vpaddq $H1,$D3,$D3 # d3 += h2*r1+ vpshufd \$0x10,`16*3-64`($ctx),$H4 # r2^n+ vpmuludq $T1,$H2,$H0 # h1*r1+ vpaddq $H0,$D2,$D2 # d2 += h1*r1+ vpmuludq $T0,$H2,$H2 # h0*r1+ vpaddq $H2,$D1,$D1 # d1 += h0*r1+ vpmuludq $T4,$H3,$H3 # h4*s1+ vpaddq $H3,$D0,$D0 # d0 += h4*s1++ vpshufd \$0x10,`16*4-64`($ctx),$H2 # s2^n+ vpmuludq $T2,$H4,$H1 # h2*r2+ vpaddq $H1,$D4,$D4 # d4 += h2*r2+ vpmuludq $T1,$H4,$H0 # h1*r2+ vpaddq $H0,$D3,$D3 # d3 += h1*r2+ vpshufd \$0x10,`16*5-64`($ctx),$H3 # r3^n+ vpmuludq $T0,$H4,$H4 # h0*r2+ vpaddq $H4,$D2,$D2 # d2 += h0*r2+ vpmuludq $T4,$H2,$H1 # h4*s2+ vpaddq $H1,$D1,$D1 # d1 += h4*s2+ vpshufd \$0x10,`16*6-64`($ctx),$H4 # s3^n+ vpmuludq $T3,$H2,$H2 # h3*s2+ vpaddq $H2,$D0,$D0 # d0 += h3*s2++ vpmuludq $T1,$H3,$H0 # h1*r3+ vpaddq $H0,$D4,$D4 # d4 += h1*r3+ vpmuludq $T0,$H3,$H3 # h0*r3+ vpaddq $H3,$D3,$D3 # d3 += h0*r3+ vpshufd \$0x10,`16*7-64`($ctx),$H2 # r4^n+ vpmuludq $T4,$H4,$H1 # h4*s3+ vpaddq $H1,$D2,$D2 # d2 += h4*s3+ vpshufd \$0x10,`16*8-64`($ctx),$H3 # s4^n+ vpmuludq $T3,$H4,$H0 # h3*s3+ vpaddq $H0,$D1,$D1 # d1 += h3*s3+ vpmuludq $T2,$H4,$H4 # h2*s3+ vpaddq $H4,$D0,$D0 # d0 += h2*s3++ vpmuludq $T0,$H2,$H2 # h0*r4+ vpaddq $H2,$D4,$D4 # h4 = d4 + h0*r4+ vpmuludq $T4,$H3,$H1 # h4*s4+ vpaddq $H1,$D3,$D3 # h3 = d3 + h4*s4+ vpmuludq $T3,$H3,$H0 # h3*s4+ vpaddq $H0,$D2,$D2 # h2 = d2 + h3*s4+ vpmuludq $T2,$H3,$H1 # h2*s4+ vpaddq $H1,$D1,$D1 # h1 = d1 + h2*s4+ vpmuludq $T1,$H3,$H3 # h1*s4+ vpaddq $H3,$D0,$D0 # h0 = d0 + h1*s4++ jz .Lshort_tail_avx++ vmovdqu 16*0($inp),$H0 # load input+ vmovdqu 16*1($inp),$H1++ vpsrldq \$6,$H0,$H2 # splat input+ vpsrldq \$6,$H1,$H3+ vpunpckhqdq $H1,$H0,$H4 # 4+ vpunpcklqdq $H1,$H0,$H0 # 0:1+ vpunpcklqdq $H3,$H2,$H3 # 2:3++ vpsrlq \$40,$H4,$H4 # 4+ vpsrlq \$26,$H0,$H1+ vpand $MASK,$H0,$H0 # 0+ vpsrlq \$4,$H3,$H2+ vpand $MASK,$H1,$H1 # 1+ vpsrlq \$30,$H3,$H3+ vpand $MASK,$H2,$H2 # 2+ vpand $MASK,$H3,$H3 # 3+ vpor 32(%rcx),$H4,$H4 # padbit, yes, always++ vpshufd \$0x32,`16*0-64`($ctx),$T4 # r0^n, 34xx -> x3x4+ vpaddq 0x00(%r11),$H0,$H0+ vpaddq 0x10(%r11),$H1,$H1+ vpaddq 0x20(%r11),$H2,$H2+ vpaddq 0x30(%r11),$H3,$H3+ vpaddq 0x40(%r11),$H4,$H4++ ################################################################+ # multiply (inp[0:1]+hash) by r^4:r^3 and accumulate++ vpmuludq $H0,$T4,$T0 # h0*r0+ vpaddq $T0,$D0,$D0 # d0 += h0*r0+ vpmuludq $H1,$T4,$T1 # h1*r0+ vpaddq $T1,$D1,$D1 # d1 += h1*r0+ vpmuludq $H2,$T4,$T0 # h2*r0+ vpaddq $T0,$D2,$D2 # d2 += h2*r0+ vpshufd \$0x32,`16*1-64`($ctx),$T2 # r1^n+ vpmuludq $H3,$T4,$T1 # h3*r0+ vpaddq $T1,$D3,$D3 # d3 += h3*r0+ vpmuludq $H4,$T4,$T4 # h4*r0+ vpaddq $T4,$D4,$D4 # d4 += h4*r0++ vpmuludq $H3,$T2,$T0 # h3*r1+ vpaddq $T0,$D4,$D4 # d4 += h3*r1+ vpshufd \$0x32,`16*2-64`($ctx),$T3 # s1+ vpmuludq $H2,$T2,$T1 # h2*r1+ vpaddq $T1,$D3,$D3 # d3 += h2*r1+ vpshufd \$0x32,`16*3-64`($ctx),$T4 # r2+ vpmuludq $H1,$T2,$T0 # h1*r1+ vpaddq $T0,$D2,$D2 # d2 += h1*r1+ vpmuludq $H0,$T2,$T2 # h0*r1+ vpaddq $T2,$D1,$D1 # d1 += h0*r1+ vpmuludq $H4,$T3,$T3 # h4*s1+ vpaddq $T3,$D0,$D0 # d0 += h4*s1++ vpshufd \$0x32,`16*4-64`($ctx),$T2 # s2+ vpmuludq $H2,$T4,$T1 # h2*r2+ vpaddq $T1,$D4,$D4 # d4 += h2*r2+ vpmuludq $H1,$T4,$T0 # h1*r2+ vpaddq $T0,$D3,$D3 # d3 += h1*r2+ vpshufd \$0x32,`16*5-64`($ctx),$T3 # r3+ vpmuludq $H0,$T4,$T4 # h0*r2+ vpaddq $T4,$D2,$D2 # d2 += h0*r2+ vpmuludq $H4,$T2,$T1 # h4*s2+ vpaddq $T1,$D1,$D1 # d1 += h4*s2+ vpshufd \$0x32,`16*6-64`($ctx),$T4 # s3+ vpmuludq $H3,$T2,$T2 # h3*s2+ vpaddq $T2,$D0,$D0 # d0 += h3*s2++ vpmuludq $H1,$T3,$T0 # h1*r3+ vpaddq $T0,$D4,$D4 # d4 += h1*r3+ vpmuludq $H0,$T3,$T3 # h0*r3+ vpaddq $T3,$D3,$D3 # d3 += h0*r3+ vpshufd \$0x32,`16*7-64`($ctx),$T2 # r4+ vpmuludq $H4,$T4,$T1 # h4*s3+ vpaddq $T1,$D2,$D2 # d2 += h4*s3+ vpshufd \$0x32,`16*8-64`($ctx),$T3 # s4+ vpmuludq $H3,$T4,$T0 # h3*s3+ vpaddq $T0,$D1,$D1 # d1 += h3*s3+ vpmuludq $H2,$T4,$T4 # h2*s3+ vpaddq $T4,$D0,$D0 # d0 += h2*s3++ vpmuludq $H0,$T2,$T2 # h0*r4+ vpaddq $T2,$D4,$D4 # d4 += h0*r4+ vpmuludq $H4,$T3,$T1 # h4*s4+ vpaddq $T1,$D3,$D3 # d3 += h4*s4+ vpmuludq $H3,$T3,$T0 # h3*s4+ vpaddq $T0,$D2,$D2 # d2 += h3*s4+ vpmuludq $H2,$T3,$T1 # h2*s4+ vpaddq $T1,$D1,$D1 # d1 += h2*s4+ vpmuludq $H1,$T3,$T3 # h1*s4+ vpaddq $T3,$D0,$D0 # d0 += h1*s4++.Lshort_tail_avx:+ ################################################################+ # horizontal addition++ vpsrldq \$8,$D4,$T4+ vpsrldq \$8,$D3,$T3+ vpsrldq \$8,$D1,$T1+ vpsrldq \$8,$D0,$T0+ vpsrldq \$8,$D2,$T2+ vpaddq $T3,$D3,$D3+ vpaddq $T4,$D4,$D4+ vpaddq $T0,$D0,$D0+ vpaddq $T1,$D1,$D1+ vpaddq $T2,$D2,$D2++ ################################################################+ # lazy reduction++ vpsrlq \$26,$D3,$H3+ vpand $MASK,$D3,$D3+ vpaddq $H3,$D4,$D4 # h3 -> h4++ vpsrlq \$26,$D0,$H0+ vpand $MASK,$D0,$D0+ vpaddq $H0,$D1,$D1 # h0 -> h1++ vpsrlq \$26,$D4,$H4+ vpand $MASK,$D4,$D4++ vpsrlq \$26,$D1,$H1+ vpand $MASK,$D1,$D1+ vpaddq $H1,$D2,$D2 # h1 -> h2++ vpaddq $H4,$D0,$D0+ vpsllq \$2,$H4,$H4+ vpaddq $H4,$D0,$D0 # h4 -> h0++ vpsrlq \$26,$D2,$H2+ vpand $MASK,$D2,$D2+ vpaddq $H2,$D3,$D3 # h2 -> h3++ vpsrlq \$26,$D0,$H0+ vpand $MASK,$D0,$D0+ vpaddq $H0,$D1,$D1 # h0 -> h1++ vpsrlq \$26,$D3,$H3+ vpand $MASK,$D3,$D3+ vpaddq $H3,$D4,$D4 # h3 -> h4++ vmovd $D0,`4*0-48-64`($ctx) # save partially reduced+ vmovd $D1,`4*1-48-64`($ctx)+ vmovd $D2,`4*2-48-64`($ctx)+ vmovd $D3,`4*3-48-64`($ctx)+ vmovd $D4,`4*4-48-64`($ctx)+___+$code.=<<___ if ($win64);+ vmovdqa 0x50(%r11),%xmm6+ vmovdqa 0x60(%r11),%xmm7+ vmovdqa 0x70(%r11),%xmm8+ vmovdqa 0x80(%r11),%xmm9+ vmovdqa 0x90(%r11),%xmm10+ vmovdqa 0xa0(%r11),%xmm11+ vmovdqa 0xb0(%r11),%xmm12+ vmovdqa 0xc0(%r11),%xmm13+ vmovdqa 0xd0(%r11),%xmm14+ vmovdqa 0xe0(%r11),%xmm15+ lea 0xf8(%r11),%rsp+.Ldo_avx_epilogue:+___+$code.=<<___ if (!$win64);+ lea 0x58(%r11),%rsp+.cfi_def_cfa %rsp,8+___+$code.=<<___;+ vzeroupper+ ret+.cfi_endproc+.size poly1305_blocks_avx,.-poly1305_blocks_avx+___++if ($avx>1) {+my ($H0,$H1,$H2,$H3,$H4, $MASK, $T4,$T0,$T1,$T2,$T3, $D0,$D1,$D2,$D3,$D4) =+ map("%ymm$_",(0..15));+my $S4=$MASK;++$code.=<<___;+.type poly1305_blocks_avx2,\@function,4+.align 32+poly1305_blocks_avx2:+.cfi_startproc+ mov 20($ctx),%r8d # load is_base2_26+ cmp \$128,$len+ jb .Lblocks++ and \$-16,$len++ vzeroupper++ test %r8d,%r8d # is_base2_26?+ jz .Lbase2_64_avx2++ test \$63,$len+ jz .Leven_avx2++ push %rbx+.cfi_push %rbx+ push %rbp+.cfi_push %rbp+ push %r12+.cfi_push %r12+ push %r13+.cfi_push %r13+ push %r14+.cfi_push %r14+ push %r15+.cfi_push %r15+ lea -8(%rsp),%rsp+.cfi_adjust_cfa_offset 8+.Lblocks_avx2_body:++ mov $len,%r15 # reassign $len++ mov 0($ctx),$d1 # load hash value+ mov 8($ctx),$d2+ mov 16($ctx),$h2#d++ mov 24($ctx),$r0 # load r+ mov 32($ctx),$s1++ ################################# base 2^26 -> base 2^64+ mov $d1#d,$h0#d+ and \$`-1*(1<<31)`,$d1+ mov $d2,$r1 # borrow $r1+ mov $d2#d,$h1#d+ and \$`-1*(1<<31)`,$d2++ shr \$6,$d1+ shl \$52,$r1+ add $d1,$h0+ shr \$12,$h1+ shr \$18,$d2+ add $r1,$h0+ adc $d2,$h1++ mov $h2,$d1+ shl \$40,$d1+ shr \$24,$h2+ add $d1,$h1+ adc \$0,$h2 # can be partially reduced...++ mov $s1,$r1+ mov $s1,%rax+ shr \$2,$s1+ add $r1,$s1 # s1 = r1 + (r1 >> 2)++.Lbase2_26_pre_avx2:+ add 0($inp),$h0 # accumulate input+ adc 8($inp),$h1+ lea 16($inp),$inp+ adc $padbit,$h2+ sub \$16,%r15++ call __poly1305_block+ mov $r1,%rax++ test \$63,%r15+ jnz .Lbase2_26_pre_avx2++ ################################# base 2^64 -> base 2^26+ mov $h0,%rax+ mov $h0,%rdx+ shr \$52,$h0+ mov $h1,$r0+ mov $h1,$r1+ shr \$26,%rdx+ and \$0x3ffffff,%rax # h[0]+ shl \$12,$r0+ and \$0x3ffffff,%rdx # h[1]+ shr \$14,$h1+ or $r0,$h0+ shl \$24,$h2+ and \$0x3ffffff,$h0 # h[2]+ shr \$40,$r1+ and \$0x3ffffff,$h1 # h[3]+ or $r1,$h2 # h[4]++ vmovd %rax#d,%x#$H0+ vmovd %rdx#d,%x#$H1+ vmovd $h0#d,%x#$H2+ vmovd $h1#d,%x#$H3+ vmovd $h2#d,%x#$H4++ mov %r15,$len # restore $len++ mov 8(%rsp),%r15+.cfi_restore %r15+ mov 16(%rsp),%r14+.cfi_restore %r14+ mov 24(%rsp),%r13+.cfi_restore %r13+ mov 32(%rsp),%r12+.cfi_restore %r12+ mov 40(%rsp),%rbp+.cfi_restore %rbp+ mov 48(%rsp),%rbx+.cfi_restore %rbx+ lea 56(%rsp),%rax # for win64+ lea 56(%rsp),%rsp+.cfi_adjust_cfa_offset -56+.Lblocks_avx2_epilogue:+ jmp .Ldo_avx2+.cfi_endproc++.align 32+.Lbase2_64_avx2:+.cfi_startproc+ push %rbx+.cfi_push %rbx+ push %rbp+.cfi_push %rbp+ push %r12+.cfi_push %r12+ push %r13+.cfi_push %r13+ push %r14+.cfi_push %r14+ push %r15+.cfi_push %r15+ lea -8(%rsp),%rsp+.cfi_adjust_cfa_offset 8+.Lbase2_64_avx2_body:++ mov $len,%r15 # reassign $len++ mov 24($ctx),$r0 # load r+ mov 32($ctx),$s1++ mov 0($ctx),$h0 # load hash value+ mov 8($ctx),$h1+ mov 16($ctx),$h2#d++ mov $s1,$r1+ mov $s1,%rax+ shr \$2,$s1+ add $r1,$s1 # s1 = r1 + (r1 >> 2)++ test \$63,$len+ jz .Linit_avx2++.Lbase2_64_pre_avx2:+ add 0($inp),$h0 # accumulate input+ adc 8($inp),$h1+ lea 16($inp),$inp+ adc $padbit,$h2+ sub \$16,%r15++ call __poly1305_block+ mov $r1,%rax++ test \$63,%r15+ jnz .Lbase2_64_pre_avx2++.Linit_avx2:+ ################################# base 2^64 -> base 2^26+ mov $h0,%rax+ mov $h0,%rdx+ shr \$52,$h0+ mov $h1,$d1+ mov $h1,$d2+ shr \$26,%rdx+ and \$0x3ffffff,%rax # h[0]+ shl \$12,$d1+ and \$0x3ffffff,%rdx # h[1]+ shr \$14,$h1+ or $d1,$h0+ shl \$24,$h2+ and \$0x3ffffff,$h0 # h[2]+ shr \$40,$d2+ and \$0x3ffffff,$h1 # h[3]+ or $d2,$h2 # h[4]++ vmovd %rax#d,%x#$H0+ vmovd %rdx#d,%x#$H1+ vmovd $h0#d,%x#$H2+ vmovd $h1#d,%x#$H3+ vmovd $h2#d,%x#$H4+ movl \$1,20($ctx) # set is_base2_26++ call __poly1305_init_avx++ mov %r15,$len # restore $len++ mov 8(%rsp),%r15+.cfi_restore %r15+ mov 16(%rsp),%r14+.cfi_restore %r14+ mov 24(%rsp),%r13+.cfi_restore %r13+ mov 32(%rsp),%r12+.cfi_restore %r12+ mov 40(%rsp),%rbp+.cfi_restore %rbp+ mov 48(%rsp),%rbx+.cfi_restore %rbx+ lea 56(%rsp),%rax # for inw64+ lea 56(%rsp),%rsp+.cfi_adjust_cfa_offset -56+.Lbase2_64_avx2_epilogue:+ jmp .Ldo_avx2+.cfi_endproc++.align 32+.Leven_avx2:+.cfi_startproc+ vmovd 4*0($ctx),%x#$H0 # load hash value base 2^26+ vmovd 4*1($ctx),%x#$H1+ vmovd 4*2($ctx),%x#$H2+ vmovd 4*3($ctx),%x#$H3+ vmovd 4*4($ctx),%x#$H4++.Ldo_avx2:+___+$code.=<<___ if ($avx>2 && $flavour !~ /kernel/);+ mov OPENSSL_ia32cap_P+8(%rip),%r10d+ cmp \$512,$len+ jb .Lskip_avx512+ test \$`1<<16`,%r10d # check for AVX512F+ jnz .Lblocks_avx512+.Lskip_avx512:+___+$code.=<<___ if (!$win64);+ lea -8(%rsp),%r11+.cfi_def_cfa %r11,16+ sub \$0x128,%rsp+___+$code.=<<___ if ($win64);+ lea -0xf8(%rsp),%r11+ sub \$0x1c8,%rsp+ vmovdqa %xmm6,0x50(%r11)+ vmovdqa %xmm7,0x60(%r11)+ vmovdqa %xmm8,0x70(%r11)+ vmovdqa %xmm9,0x80(%r11)+ vmovdqa %xmm10,0x90(%r11)+ vmovdqa %xmm11,0xa0(%r11)+ vmovdqa %xmm12,0xb0(%r11)+ vmovdqa %xmm13,0xc0(%r11)+ vmovdqa %xmm14,0xd0(%r11)+ vmovdqa %xmm15,0xe0(%r11)+.Ldo_avx2_body:+___+$code.=<<___;+ lea .Lconst(%rip),%rcx+ lea 48+64($ctx),$ctx # size optimization+ vmovdqa 96(%rcx),$T0 # .Lpermd_avx2++ # expand and copy pre-calculated table to stack+ vmovdqu `16*0-64`($ctx),%x#$T2+ and \$-512,%rsp+ vmovdqu `16*1-64`($ctx),%x#$T3+ vmovdqu `16*2-64`($ctx),%x#$T4+ vmovdqu `16*3-64`($ctx),%x#$D0+ vmovdqu `16*4-64`($ctx),%x#$D1+ vmovdqu `16*5-64`($ctx),%x#$D2+ lea 0x90(%rsp),%rax # size optimization+ vmovdqu `16*6-64`($ctx),%x#$D3+ vpermd $T2,$T0,$T2 # 00003412 -> 14243444+ vmovdqu `16*7-64`($ctx),%x#$D4+ vpermd $T3,$T0,$T3+ vmovdqu `16*8-64`($ctx),%x#$MASK+ vpermd $T4,$T0,$T4+ vmovdqa $T2,0x00(%rsp)+ vpermd $D0,$T0,$D0+ vmovdqa $T3,0x20-0x90(%rax)+ vpermd $D1,$T0,$D1+ vmovdqa $T4,0x40-0x90(%rax)+ vpermd $D2,$T0,$D2+ vmovdqa $D0,0x60-0x90(%rax)+ vpermd $D3,$T0,$D3+ vmovdqa $D1,0x80-0x90(%rax)+ vpermd $D4,$T0,$D4+ vmovdqa $D2,0xa0-0x90(%rax)+ vpermd $MASK,$T0,$MASK+ vmovdqa $D3,0xc0-0x90(%rax)+ vmovdqa $D4,0xe0-0x90(%rax)+ vmovdqa $MASK,0x100-0x90(%rax)+ vmovdqa 64(%rcx),$MASK # .Lmask26++ ################################################################+ # load input+ vmovdqu 16*0($inp),%x#$T0+ vmovdqu 16*1($inp),%x#$T1+ vinserti128 \$1,16*2($inp),$T0,$T0+ vinserti128 \$1,16*3($inp),$T1,$T1+ lea 16*4($inp),$inp++ vpsrldq \$6,$T0,$T2 # splat input+ vpsrldq \$6,$T1,$T3+ vpunpckhqdq $T1,$T0,$T4 # 4+ vpunpcklqdq $T3,$T2,$T2 # 2:3+ vpunpcklqdq $T1,$T0,$T0 # 0:1++ vpsrlq \$30,$T2,$T3+ vpsrlq \$4,$T2,$T2+ vpsrlq \$26,$T0,$T1+ vpsrlq \$40,$T4,$T4 # 4+ vpand $MASK,$T2,$T2 # 2+ vpand $MASK,$T0,$T0 # 0+ vpand $MASK,$T1,$T1 # 1+ vpand $MASK,$T3,$T3 # 3+ vpor 32(%rcx),$T4,$T4 # padbit, yes, always++ vpaddq $H2,$T2,$H2 # accumulate input+ sub \$64,$len+ jz .Ltail_avx2+ jmp .Loop_avx2++.align 32+.Loop_avx2:+ ################################################################+ # ((inp[0]*r^4+inp[4])*r^4+inp[ 8])*r^4+ # ((inp[1]*r^4+inp[5])*r^4+inp[ 9])*r^3+ # ((inp[2]*r^4+inp[6])*r^4+inp[10])*r^2+ # ((inp[3]*r^4+inp[7])*r^4+inp[11])*r^1+ # \________/\__________/+ ################################################################+ #vpaddq $H2,$T2,$H2 # accumulate input+ vpaddq $H0,$T0,$H0+ vmovdqa `32*0`(%rsp),$T0 # r0^4+ vpaddq $H1,$T1,$H1+ vmovdqa `32*1`(%rsp),$T1 # r1^4+ vpaddq $H3,$T3,$H3+ vmovdqa `32*3`(%rsp),$T2 # r2^4+ vpaddq $H4,$T4,$H4+ vmovdqa `32*6-0x90`(%rax),$T3 # s3^4+ vmovdqa `32*8-0x90`(%rax),$S4 # s4^4++ # d4 = h4*r0 + h3*r1 + h2*r2 + h1*r3 + h0*r4+ # d3 = h3*r0 + h2*r1 + h1*r2 + h0*r3 + h4*5*r4+ # d2 = h2*r0 + h1*r1 + h0*r2 + h4*5*r3 + h3*5*r4+ # d1 = h1*r0 + h0*r1 + h4*5*r2 + h3*5*r3 + h2*5*r4+ # d0 = h0*r0 + h4*5*r1 + h3*5*r2 + h2*5*r3 + h1*5*r4+ #+ # however, as h2 is "chronologically" first one available pull+ # corresponding operations up, so it's+ #+ # d4 = h2*r2 + h4*r0 + h3*r1 + h1*r3 + h0*r4+ # d3 = h2*r1 + h3*r0 + h1*r2 + h0*r3 + h4*5*r4+ # d2 = h2*r0 + h1*r1 + h0*r2 + h4*5*r3 + h3*5*r4+ # d1 = h2*5*r4 + h1*r0 + h0*r1 + h4*5*r2 + h3*5*r3+ # d0 = h2*5*r3 + h0*r0 + h4*5*r1 + h3*5*r2 + h1*5*r4++ vpmuludq $H2,$T0,$D2 # d2 = h2*r0+ vpmuludq $H2,$T1,$D3 # d3 = h2*r1+ vpmuludq $H2,$T2,$D4 # d4 = h2*r2+ vpmuludq $H2,$T3,$D0 # d0 = h2*s3+ vpmuludq $H2,$S4,$D1 # d1 = h2*s4++ vpmuludq $H0,$T1,$T4 # h0*r1+ vpmuludq $H1,$T1,$H2 # h1*r1, borrow $H2 as temp+ vpaddq $T4,$D1,$D1 # d1 += h0*r1+ vpaddq $H2,$D2,$D2 # d2 += h1*r1+ vpmuludq $H3,$T1,$T4 # h3*r1+ vpmuludq `32*2`(%rsp),$H4,$H2 # h4*s1+ vpaddq $T4,$D4,$D4 # d4 += h3*r1+ vpaddq $H2,$D0,$D0 # d0 += h4*s1+ vmovdqa `32*4-0x90`(%rax),$T1 # s2++ vpmuludq $H0,$T0,$T4 # h0*r0+ vpmuludq $H1,$T0,$H2 # h1*r0+ vpaddq $T4,$D0,$D0 # d0 += h0*r0+ vpaddq $H2,$D1,$D1 # d1 += h1*r0+ vpmuludq $H3,$T0,$T4 # h3*r0+ vpmuludq $H4,$T0,$H2 # h4*r0+ vmovdqu 16*0($inp),%x#$T0 # load input+ vpaddq $T4,$D3,$D3 # d3 += h3*r0+ vpaddq $H2,$D4,$D4 # d4 += h4*r0+ vinserti128 \$1,16*2($inp),$T0,$T0++ vpmuludq $H3,$T1,$T4 # h3*s2+ vpmuludq $H4,$T1,$H2 # h4*s2+ vmovdqu 16*1($inp),%x#$T1+ vpaddq $T4,$D0,$D0 # d0 += h3*s2+ vpaddq $H2,$D1,$D1 # d1 += h4*s2+ vmovdqa `32*5-0x90`(%rax),$H2 # r3+ vpmuludq $H1,$T2,$T4 # h1*r2+ vpmuludq $H0,$T2,$T2 # h0*r2+ vpaddq $T4,$D3,$D3 # d3 += h1*r2+ vpaddq $T2,$D2,$D2 # d2 += h0*r2+ vinserti128 \$1,16*3($inp),$T1,$T1+ lea 16*4($inp),$inp++ vpmuludq $H1,$H2,$T4 # h1*r3+ vpmuludq $H0,$H2,$H2 # h0*r3+ vpsrldq \$6,$T0,$T2 # splat input+ vpaddq $T4,$D4,$D4 # d4 += h1*r3+ vpaddq $H2,$D3,$D3 # d3 += h0*r3+ vpmuludq $H3,$T3,$T4 # h3*s3+ vpmuludq $H4,$T3,$H2 # h4*s3+ vpsrldq \$6,$T1,$T3+ vpaddq $T4,$D1,$D1 # d1 += h3*s3+ vpaddq $H2,$D2,$D2 # d2 += h4*s3+ vpunpckhqdq $T1,$T0,$T4 # 4++ vpmuludq $H3,$S4,$H3 # h3*s4+ vpmuludq $H4,$S4,$H4 # h4*s4+ vpunpcklqdq $T1,$T0,$T0 # 0:1+ vpaddq $H3,$D2,$H2 # h2 = d2 + h3*r4+ vpaddq $H4,$D3,$H3 # h3 = d3 + h4*r4+ vpunpcklqdq $T3,$T2,$T3 # 2:3+ vpmuludq `32*7-0x90`(%rax),$H0,$H4 # h0*r4+ vpmuludq $H1,$S4,$H0 # h1*s4+ vmovdqa 64(%rcx),$MASK # .Lmask26+ vpaddq $H4,$D4,$H4 # h4 = d4 + h0*r4+ vpaddq $H0,$D0,$H0 # h0 = d0 + h1*s4++ ################################################################+ # lazy reduction (interleaved with tail of input splat)++ vpsrlq \$26,$H3,$D3+ vpand $MASK,$H3,$H3+ vpaddq $D3,$H4,$H4 # h3 -> h4++ vpsrlq \$26,$H0,$D0+ vpand $MASK,$H0,$H0+ vpaddq $D0,$D1,$H1 # h0 -> h1++ vpsrlq \$26,$H4,$D4+ vpand $MASK,$H4,$H4++ vpsrlq \$4,$T3,$T2++ vpsrlq \$26,$H1,$D1+ vpand $MASK,$H1,$H1+ vpaddq $D1,$H2,$H2 # h1 -> h2++ vpaddq $D4,$H0,$H0+ vpsllq \$2,$D4,$D4+ vpaddq $D4,$H0,$H0 # h4 -> h0++ vpand $MASK,$T2,$T2 # 2+ vpsrlq \$26,$T0,$T1++ vpsrlq \$26,$H2,$D2+ vpand $MASK,$H2,$H2+ vpaddq $D2,$H3,$H3 # h2 -> h3++ vpaddq $T2,$H2,$H2 # modulo-scheduled+ vpsrlq \$30,$T3,$T3++ vpsrlq \$26,$H0,$D0+ vpand $MASK,$H0,$H0+ vpaddq $D0,$H1,$H1 # h0 -> h1++ vpsrlq \$40,$T4,$T4 # 4++ vpsrlq \$26,$H3,$D3+ vpand $MASK,$H3,$H3+ vpaddq $D3,$H4,$H4 # h3 -> h4++ vpand $MASK,$T0,$T0 # 0+ vpand $MASK,$T1,$T1 # 1+ vpand $MASK,$T3,$T3 # 3+ vpor 32(%rcx),$T4,$T4 # padbit, yes, always++ sub \$64,$len+ jnz .Loop_avx2++ .byte 0x66,0x90+.Ltail_avx2:+ ################################################################+ # while above multiplications were by r^4 in all lanes, in last+ # iteration we multiply least significant lane by r^4 and most+ # significant one by r, so copy of above except that references+ # to the precomputed table are displaced by 4...++ #vpaddq $H2,$T2,$H2 # accumulate input+ vpaddq $H0,$T0,$H0+ vmovdqu `32*0+4`(%rsp),$T0 # r0^4+ vpaddq $H1,$T1,$H1+ vmovdqu `32*1+4`(%rsp),$T1 # r1^4+ vpaddq $H3,$T3,$H3+ vmovdqu `32*3+4`(%rsp),$T2 # r2^4+ vpaddq $H4,$T4,$H4+ vmovdqu `32*6+4-0x90`(%rax),$T3 # s3^4+ vmovdqu `32*8+4-0x90`(%rax),$S4 # s4^4++ vpmuludq $H2,$T0,$D2 # d2 = h2*r0+ vpmuludq $H2,$T1,$D3 # d3 = h2*r1+ vpmuludq $H2,$T2,$D4 # d4 = h2*r2+ vpmuludq $H2,$T3,$D0 # d0 = h2*s3+ vpmuludq $H2,$S4,$D1 # d1 = h2*s4++ vpmuludq $H0,$T1,$T4 # h0*r1+ vpmuludq $H1,$T1,$H2 # h1*r1+ vpaddq $T4,$D1,$D1 # d1 += h0*r1+ vpaddq $H2,$D2,$D2 # d2 += h1*r1+ vpmuludq $H3,$T1,$T4 # h3*r1+ vpmuludq `32*2+4`(%rsp),$H4,$H2 # h4*s1+ vpaddq $T4,$D4,$D4 # d4 += h3*r1+ vpaddq $H2,$D0,$D0 # d0 += h4*s1++ vpmuludq $H0,$T0,$T4 # h0*r0+ vpmuludq $H1,$T0,$H2 # h1*r0+ vpaddq $T4,$D0,$D0 # d0 += h0*r0+ vmovdqu `32*4+4-0x90`(%rax),$T1 # s2+ vpaddq $H2,$D1,$D1 # d1 += h1*r0+ vpmuludq $H3,$T0,$T4 # h3*r0+ vpmuludq $H4,$T0,$H2 # h4*r0+ vpaddq $T4,$D3,$D3 # d3 += h3*r0+ vpaddq $H2,$D4,$D4 # d4 += h4*r0++ vpmuludq $H3,$T1,$T4 # h3*s2+ vpmuludq $H4,$T1,$H2 # h4*s2+ vpaddq $T4,$D0,$D0 # d0 += h3*s2+ vpaddq $H2,$D1,$D1 # d1 += h4*s2+ vmovdqu `32*5+4-0x90`(%rax),$H2 # r3+ vpmuludq $H1,$T2,$T4 # h1*r2+ vpmuludq $H0,$T2,$T2 # h0*r2+ vpaddq $T4,$D3,$D3 # d3 += h1*r2+ vpaddq $T2,$D2,$D2 # d2 += h0*r2++ vpmuludq $H1,$H2,$T4 # h1*r3+ vpmuludq $H0,$H2,$H2 # h0*r3+ vpaddq $T4,$D4,$D4 # d4 += h1*r3+ vpaddq $H2,$D3,$D3 # d3 += h0*r3+ vpmuludq $H3,$T3,$T4 # h3*s3+ vpmuludq $H4,$T3,$H2 # h4*s3+ vpaddq $T4,$D1,$D1 # d1 += h3*s3+ vpaddq $H2,$D2,$D2 # d2 += h4*s3++ vpmuludq $H3,$S4,$H3 # h3*s4+ vpmuludq $H4,$S4,$H4 # h4*s4+ vpaddq $H3,$D2,$H2 # h2 = d2 + h3*r4+ vpaddq $H4,$D3,$H3 # h3 = d3 + h4*r4+ vpmuludq `32*7+4-0x90`(%rax),$H0,$H4 # h0*r4+ vpmuludq $H1,$S4,$H0 # h1*s4+ vmovdqa 64(%rcx),$MASK # .Lmask26+ vpaddq $H4,$D4,$H4 # h4 = d4 + h0*r4+ vpaddq $H0,$D0,$H0 # h0 = d0 + h1*s4++ ################################################################+ # horizontal addition++ vpsrldq \$8,$D1,$T1+ vpsrldq \$8,$H2,$T2+ vpsrldq \$8,$H3,$T3+ vpsrldq \$8,$H4,$T4+ vpsrldq \$8,$H0,$T0+ vpaddq $T1,$D1,$D1+ vpaddq $T2,$H2,$H2+ vpaddq $T3,$H3,$H3+ vpaddq $T4,$H4,$H4+ vpaddq $T0,$H0,$H0++ vpermq \$0x2,$H3,$T3+ vpermq \$0x2,$H4,$T4+ vpermq \$0x2,$H0,$T0+ vpermq \$0x2,$D1,$T1+ vpermq \$0x2,$H2,$T2+ vpaddq $T3,$H3,$H3+ vpaddq $T4,$H4,$H4+ vpaddq $T0,$H0,$H0+ vpaddq $T1,$D1,$D1+ vpaddq $T2,$H2,$H2++ ################################################################+ # lazy reduction++ vpsrlq \$26,$H3,$D3+ vpand $MASK,$H3,$H3+ vpaddq $D3,$H4,$H4 # h3 -> h4++ vpsrlq \$26,$H0,$D0+ vpand $MASK,$H0,$H0+ vpaddq $D0,$D1,$H1 # h0 -> h1++ vpsrlq \$26,$H4,$D4+ vpand $MASK,$H4,$H4++ vpsrlq \$26,$H1,$D1+ vpand $MASK,$H1,$H1+ vpaddq $D1,$H2,$H2 # h1 -> h2++ vpaddq $D4,$H0,$H0+ vpsllq \$2,$D4,$D4+ vpaddq $D4,$H0,$H0 # h4 -> h0++ vpsrlq \$26,$H2,$D2+ vpand $MASK,$H2,$H2+ vpaddq $D2,$H3,$H3 # h2 -> h3++ vpsrlq \$26,$H0,$D0+ vpand $MASK,$H0,$H0+ vpaddq $D0,$H1,$H1 # h0 -> h1++ vpsrlq \$26,$H3,$D3+ vpand $MASK,$H3,$H3+ vpaddq $D3,$H4,$H4 # h3 -> h4++ vmovd %x#$H0,`4*0-48-64`($ctx)# save partially reduced+ vmovd %x#$H1,`4*1-48-64`($ctx)+ vmovd %x#$H2,`4*2-48-64`($ctx)+ vmovd %x#$H3,`4*3-48-64`($ctx)+ vmovd %x#$H4,`4*4-48-64`($ctx)+___+$code.=<<___ if ($win64);+ vmovdqa 0x50(%r11),%xmm6+ vmovdqa 0x60(%r11),%xmm7+ vmovdqa 0x70(%r11),%xmm8+ vmovdqa 0x80(%r11),%xmm9+ vmovdqa 0x90(%r11),%xmm10+ vmovdqa 0xa0(%r11),%xmm11+ vmovdqa 0xb0(%r11),%xmm12+ vmovdqa 0xc0(%r11),%xmm13+ vmovdqa 0xd0(%r11),%xmm14+ vmovdqa 0xe0(%r11),%xmm15+ lea 0xf8(%r11),%rsp+.Ldo_avx2_epilogue:+___+$code.=<<___ if (!$win64);+ lea 8(%r11),%rsp+.cfi_def_cfa %rsp,8+___+$code.=<<___;+ vzeroupper+ ret+.cfi_endproc+.size poly1305_blocks_avx2,.-poly1305_blocks_avx2+___+#######################################################################+if ($avx>2 && $flavour !~ /kernel/) {+# On entry we have input length divisible by 64. But since inner loop+# processes 128 bytes per iteration, cases when length is not divisible+# by 128 are handled by passing tail 64 bytes to .Ltail_avx2. For this+# reason stack layout is kept identical to poly1305_blocks_avx2. If not+# for this tail, we wouldn't have to even allocate stack frame...++my ($R0,$R1,$R2,$R3,$R4, $S1,$S2,$S3,$S4) = map("%zmm$_",(16..24));+my ($M0,$M1,$M2,$M3,$M4) = map("%zmm$_",(25..29));+my $PADBIT="%zmm30";++map(s/%y/%z/,($T4,$T0,$T1,$T2,$T3)); # switch to %zmm domain+map(s/%y/%z/,($D0,$D1,$D2,$D3,$D4));+map(s/%y/%z/,($H0,$H1,$H2,$H3,$H4));+map(s/%y/%z/,($MASK));++$code.=<<___;+.type poly1305_blocks_avx512,\@function,4+.align 32+poly1305_blocks_avx512:+.cfi_startproc+.Lblocks_avx512:+ mov \$15,%eax+ kmovw %eax,%k2+___+$code.=<<___ if (!$win64);+ lea -8(%rsp),%r11+.cfi_def_cfa %r11,16+ sub \$0x128,%rsp+___+$code.=<<___ if ($win64);+ lea -0xf8(%rsp),%r11+ sub \$0x1c8,%rsp+ vmovdqa %xmm6,0x50(%r11)+ vmovdqa %xmm7,0x60(%r11)+ vmovdqa %xmm8,0x70(%r11)+ vmovdqa %xmm9,0x80(%r11)+ vmovdqa %xmm10,0x90(%r11)+ vmovdqa %xmm11,0xa0(%r11)+ vmovdqa %xmm12,0xb0(%r11)+ vmovdqa %xmm13,0xc0(%r11)+ vmovdqa %xmm14,0xd0(%r11)+ vmovdqa %xmm15,0xe0(%r11)+.Ldo_avx512_body:+___+$code.=<<___;+ lea .Lconst(%rip),%rcx+ lea 48+64($ctx),$ctx # size optimization+ vmovdqa 96(%rcx),%y#$T2 # .Lpermd_avx2++ # expand pre-calculated table+ vmovdqu `16*0-64`($ctx),%x#$D0 # will become expanded ${R0}+ and \$-512,%rsp+ vmovdqu `16*1-64`($ctx),%x#$D1 # will become ... ${R1}+ mov \$0x20,%rax+ vmovdqu `16*2-64`($ctx),%x#$T0 # ... ${S1}+ vmovdqu `16*3-64`($ctx),%x#$D2 # ... ${R2}+ vmovdqu `16*4-64`($ctx),%x#$T1 # ... ${S2}+ vmovdqu `16*5-64`($ctx),%x#$D3 # ... ${R3}+ vmovdqu `16*6-64`($ctx),%x#$T3 # ... ${S3}+ vmovdqu `16*7-64`($ctx),%x#$D4 # ... ${R4}+ vmovdqu `16*8-64`($ctx),%x#$T4 # ... ${S4}+ vpermd $D0,$T2,$R0 # 00003412 -> 14243444+ vpbroadcastq 64(%rcx),$MASK # .Lmask26+ vpermd $D1,$T2,$R1+ vpermd $T0,$T2,$S1+ vpermd $D2,$T2,$R2+ vmovdqa64 $R0,0x00(%rsp){%k2} # save in case $len%128 != 0+ vpsrlq \$32,$R0,$T0 # 14243444 -> 01020304+ vpermd $T1,$T2,$S2+ vmovdqu64 $R1,0x00(%rsp,%rax){%k2}+ vpsrlq \$32,$R1,$T1+ vpermd $D3,$T2,$R3+ vmovdqa64 $S1,0x40(%rsp){%k2}+ vpermd $T3,$T2,$S3+ vpermd $D4,$T2,$R4+ vmovdqu64 $R2,0x40(%rsp,%rax){%k2}+ vpermd $T4,$T2,$S4+ vmovdqa64 $S2,0x80(%rsp){%k2}+ vmovdqu64 $R3,0x80(%rsp,%rax){%k2}+ vmovdqa64 $S3,0xc0(%rsp){%k2}+ vmovdqu64 $R4,0xc0(%rsp,%rax){%k2}+ vmovdqa64 $S4,0x100(%rsp){%k2}++ ################################################################+ # calculate 5th through 8th powers of the key+ #+ # d0 = r0'*r0 + r1'*5*r4 + r2'*5*r3 + r3'*5*r2 + r4'*5*r1+ # d1 = r0'*r1 + r1'*r0 + r2'*5*r4 + r3'*5*r3 + r4'*5*r2+ # d2 = r0'*r2 + r1'*r1 + r2'*r0 + r3'*5*r4 + r4'*5*r3+ # d3 = r0'*r3 + r1'*r2 + r2'*r1 + r3'*r0 + r4'*5*r4+ # d4 = r0'*r4 + r1'*r3 + r2'*r2 + r3'*r1 + r4'*r0++ vpmuludq $T0,$R0,$D0 # d0 = r0'*r0+ vpmuludq $T0,$R1,$D1 # d1 = r0'*r1+ vpmuludq $T0,$R2,$D2 # d2 = r0'*r2+ vpmuludq $T0,$R3,$D3 # d3 = r0'*r3+ vpmuludq $T0,$R4,$D4 # d4 = r0'*r4+ vpsrlq \$32,$R2,$T2++ vpmuludq $T1,$S4,$M0+ vpmuludq $T1,$R0,$M1+ vpmuludq $T1,$R1,$M2+ vpmuludq $T1,$R2,$M3+ vpmuludq $T1,$R3,$M4+ vpsrlq \$32,$R3,$T3+ vpaddq $M0,$D0,$D0 # d0 += r1'*5*r4+ vpaddq $M1,$D1,$D1 # d1 += r1'*r0+ vpaddq $M2,$D2,$D2 # d2 += r1'*r1+ vpaddq $M3,$D3,$D3 # d3 += r1'*r2+ vpaddq $M4,$D4,$D4 # d4 += r1'*r3++ vpmuludq $T2,$S3,$M0+ vpmuludq $T2,$S4,$M1+ vpmuludq $T2,$R1,$M3+ vpmuludq $T2,$R2,$M4+ vpmuludq $T2,$R0,$M2+ vpsrlq \$32,$R4,$T4+ vpaddq $M0,$D0,$D0 # d0 += r2'*5*r3+ vpaddq $M1,$D1,$D1 # d1 += r2'*5*r4+ vpaddq $M3,$D3,$D3 # d3 += r2'*r1+ vpaddq $M4,$D4,$D4 # d4 += r2'*r2+ vpaddq $M2,$D2,$D2 # d2 += r2'*r0++ vpmuludq $T3,$S2,$M0+ vpmuludq $T3,$R0,$M3+ vpmuludq $T3,$R1,$M4+ vpmuludq $T3,$S3,$M1+ vpmuludq $T3,$S4,$M2+ vpaddq $M0,$D0,$D0 # d0 += r3'*5*r2+ vpaddq $M3,$D3,$D3 # d3 += r3'*r0+ vpaddq $M4,$D4,$D4 # d4 += r3'*r1+ vpaddq $M1,$D1,$D1 # d1 += r3'*5*r3+ vpaddq $M2,$D2,$D2 # d2 += r3'*5*r4++ vpmuludq $T4,$S4,$M3+ vpmuludq $T4,$R0,$M4+ vpmuludq $T4,$S1,$M0+ vpmuludq $T4,$S2,$M1+ vpmuludq $T4,$S3,$M2+ vpaddq $M3,$D3,$D3 # d3 += r2'*5*r4+ vpaddq $M4,$D4,$D4 # d4 += r2'*r0+ vpaddq $M0,$D0,$D0 # d0 += r2'*5*r1+ vpaddq $M1,$D1,$D1 # d1 += r2'*5*r2+ vpaddq $M2,$D2,$D2 # d2 += r2'*5*r3++ ################################################################+ # load input+ vmovdqu64 16*0($inp),%z#$T3+ vmovdqu64 16*4($inp),%z#$T4+ lea 16*8($inp),$inp++ ################################################################+ # lazy reduction++ vpsrlq \$26,$D3,$M3+ vpandq $MASK,$D3,$D3+ vpaddq $M3,$D4,$D4 # d3 -> d4++ vpsrlq \$26,$D0,$M0+ vpandq $MASK,$D0,$D0+ vpaddq $M0,$D1,$D1 # d0 -> d1++ vpsrlq \$26,$D4,$M4+ vpandq $MASK,$D4,$D4++ vpsrlq \$26,$D1,$M1+ vpandq $MASK,$D1,$D1+ vpaddq $M1,$D2,$D2 # d1 -> d2++ vpaddq $M4,$D0,$D0+ vpsllq \$2,$M4,$M4+ vpaddq $M4,$D0,$D0 # d4 -> d0++ vpsrlq \$26,$D2,$M2+ vpandq $MASK,$D2,$D2+ vpaddq $M2,$D3,$D3 # d2 -> d3++ vpsrlq \$26,$D0,$M0+ vpandq $MASK,$D0,$D0+ vpaddq $M0,$D1,$D1 # d0 -> d1++ vpsrlq \$26,$D3,$M3+ vpandq $MASK,$D3,$D3+ vpaddq $M3,$D4,$D4 # d3 -> d4++ ################################################################+ # at this point we have 14243444 in $R0-$S4 and 05060708 in+ # $D0-$D4, ...++ vpunpcklqdq $T4,$T3,$T0 # transpose input+ vpunpckhqdq $T4,$T3,$T4++ # ... since input 64-bit lanes are ordered as 73625140, we could+ # "vperm" it to 76543210 (here and in each loop iteration), *or*+ # we could just flow along, hence the goal for $R0-$S4 is+ # 1858286838784888 ...++ vmovdqa32 128(%rcx),$M0 # .Lpermd_avx512:+ mov \$0x7777,%eax+ kmovw %eax,%k1++ vpermd $R0,$M0,$R0 # 14243444 -> 1---2---3---4---+ vpermd $R1,$M0,$R1+ vpermd $R2,$M0,$R2+ vpermd $R3,$M0,$R3+ vpermd $R4,$M0,$R4++ vpermd $D0,$M0,${R0}{%k1} # 05060708 -> 1858286838784888+ vpermd $D1,$M0,${R1}{%k1}+ vpermd $D2,$M0,${R2}{%k1}+ vpermd $D3,$M0,${R3}{%k1}+ vpermd $D4,$M0,${R4}{%k1}++ vpslld \$2,$R1,$S1 # *5+ vpslld \$2,$R2,$S2+ vpslld \$2,$R3,$S3+ vpslld \$2,$R4,$S4+ vpaddd $R1,$S1,$S1+ vpaddd $R2,$S2,$S2+ vpaddd $R3,$S3,$S3+ vpaddd $R4,$S4,$S4++ vpbroadcastq 32(%rcx),$PADBIT # .L129++ vpsrlq \$52,$T0,$T2 # splat input+ vpsllq \$12,$T4,$T3+ vporq $T3,$T2,$T2+ vpsrlq \$26,$T0,$T1+ vpsrlq \$14,$T4,$T3+ vpsrlq \$40,$T4,$T4 # 4+ vpandq $MASK,$T2,$T2 # 2+ vpandq $MASK,$T0,$T0 # 0+ #vpandq $MASK,$T1,$T1 # 1+ #vpandq $MASK,$T3,$T3 # 3+ #vporq $PADBIT,$T4,$T4 # padbit, yes, always++ vpaddq $H2,$T2,$H2 # accumulate input+ sub \$192,$len+ jbe .Ltail_avx512+ jmp .Loop_avx512++.align 32+.Loop_avx512:+ ################################################################+ # ((inp[0]*r^8+inp[ 8])*r^8+inp[16])*r^8+ # ((inp[1]*r^8+inp[ 9])*r^8+inp[17])*r^7+ # ((inp[2]*r^8+inp[10])*r^8+inp[18])*r^6+ # ((inp[3]*r^8+inp[11])*r^8+inp[19])*r^5+ # ((inp[4]*r^8+inp[12])*r^8+inp[20])*r^4+ # ((inp[5]*r^8+inp[13])*r^8+inp[21])*r^3+ # ((inp[6]*r^8+inp[14])*r^8+inp[22])*r^2+ # ((inp[7]*r^8+inp[15])*r^8+inp[23])*r^1+ # \________/\___________/+ ################################################################+ #vpaddq $H2,$T2,$H2 # accumulate input++ # d4 = h4*r0 + h3*r1 + h2*r2 + h1*r3 + h0*r4+ # d3 = h3*r0 + h2*r1 + h1*r2 + h0*r3 + h4*5*r4+ # d2 = h2*r0 + h1*r1 + h0*r2 + h4*5*r3 + h3*5*r4+ # d1 = h1*r0 + h0*r1 + h4*5*r2 + h3*5*r3 + h2*5*r4+ # d0 = h0*r0 + h4*5*r1 + h3*5*r2 + h2*5*r3 + h1*5*r4+ #+ # however, as h2 is "chronologically" first one available pull+ # corresponding operations up, so it's+ #+ # d3 = h2*r1 + h0*r3 + h1*r2 + h3*r0 + h4*5*r4+ # d4 = h2*r2 + h0*r4 + h1*r3 + h3*r1 + h4*r0+ # d0 = h2*5*r3 + h0*r0 + h1*5*r4 + h3*5*r2 + h4*5*r1+ # d1 = h2*5*r4 + h0*r1 + h1*r0 + h3*5*r3 + h4*5*r2+ # d2 = h2*r0 + h0*r2 + h1*r1 + h3*5*r4 + h4*5*r3++ vpmuludq $H2,$R1,$D3 # d3 = h2*r1+ vpaddq $H0,$T0,$H0+ vpmuludq $H2,$R2,$D4 # d4 = h2*r2+ vpandq $MASK,$T1,$T1 # 1+ vpmuludq $H2,$S3,$D0 # d0 = h2*s3+ vpandq $MASK,$T3,$T3 # 3+ vpmuludq $H2,$S4,$D1 # d1 = h2*s4+ vporq $PADBIT,$T4,$T4 # padbit, yes, always+ vpmuludq $H2,$R0,$D2 # d2 = h2*r0+ vpaddq $H1,$T1,$H1 # accumulate input+ vpaddq $H3,$T3,$H3+ vpaddq $H4,$T4,$H4++ vmovdqu64 16*0($inp),$T3 # load input+ vmovdqu64 16*4($inp),$T4+ lea 16*8($inp),$inp+ vpmuludq $H0,$R3,$M3+ vpmuludq $H0,$R4,$M4+ vpmuludq $H0,$R0,$M0+ vpmuludq $H0,$R1,$M1+ vpaddq $M3,$D3,$D3 # d3 += h0*r3+ vpaddq $M4,$D4,$D4 # d4 += h0*r4+ vpaddq $M0,$D0,$D0 # d0 += h0*r0+ vpaddq $M1,$D1,$D1 # d1 += h0*r1++ vpmuludq $H1,$R2,$M3+ vpmuludq $H1,$R3,$M4+ vpmuludq $H1,$S4,$M0+ vpmuludq $H0,$R2,$M2+ vpaddq $M3,$D3,$D3 # d3 += h1*r2+ vpaddq $M4,$D4,$D4 # d4 += h1*r3+ vpaddq $M0,$D0,$D0 # d0 += h1*s4+ vpaddq $M2,$D2,$D2 # d2 += h0*r2++ vpunpcklqdq $T4,$T3,$T0 # transpose input+ vpunpckhqdq $T4,$T3,$T4++ vpmuludq $H3,$R0,$M3+ vpmuludq $H3,$R1,$M4+ vpmuludq $H1,$R0,$M1+ vpmuludq $H1,$R1,$M2+ vpaddq $M3,$D3,$D3 # d3 += h3*r0+ vpaddq $M4,$D4,$D4 # d4 += h3*r1+ vpaddq $M1,$D1,$D1 # d1 += h1*r0+ vpaddq $M2,$D2,$D2 # d2 += h1*r1++ vpmuludq $H4,$S4,$M3+ vpmuludq $H4,$R0,$M4+ vpmuludq $H3,$S2,$M0+ vpmuludq $H3,$S3,$M1+ vpaddq $M3,$D3,$D3 # d3 += h4*s4+ vpmuludq $H3,$S4,$M2+ vpaddq $M4,$D4,$D4 # d4 += h4*r0+ vpaddq $M0,$D0,$D0 # d0 += h3*s2+ vpaddq $M1,$D1,$D1 # d1 += h3*s3+ vpaddq $M2,$D2,$D2 # d2 += h3*s4++ vpmuludq $H4,$S1,$M0+ vpmuludq $H4,$S2,$M1+ vpmuludq $H4,$S3,$M2+ vpaddq $M0,$D0,$H0 # h0 = d0 + h4*s1+ vpaddq $M1,$D1,$H1 # h1 = d2 + h4*s2+ vpaddq $M2,$D2,$H2 # h2 = d3 + h4*s3++ ################################################################+ # lazy reduction (interleaved with input splat)++ vpsrlq \$52,$T0,$T2 # splat input+ vpsllq \$12,$T4,$T3++ vpsrlq \$26,$D3,$H3+ vpandq $MASK,$D3,$D3+ vpaddq $H3,$D4,$H4 # h3 -> h4++ vporq $T3,$T2,$T2++ vpsrlq \$26,$H0,$D0+ vpandq $MASK,$H0,$H0+ vpaddq $D0,$H1,$H1 # h0 -> h1++ vpandq $MASK,$T2,$T2 # 2++ vpsrlq \$26,$H4,$D4+ vpandq $MASK,$H4,$H4++ vpsrlq \$26,$H1,$D1+ vpandq $MASK,$H1,$H1+ vpaddq $D1,$H2,$H2 # h1 -> h2++ vpaddq $D4,$H0,$H0+ vpsllq \$2,$D4,$D4+ vpaddq $D4,$H0,$H0 # h4 -> h0++ vpaddq $T2,$H2,$H2 # modulo-scheduled+ vpsrlq \$26,$T0,$T1++ vpsrlq \$26,$H2,$D2+ vpandq $MASK,$H2,$H2+ vpaddq $D2,$D3,$H3 # h2 -> h3++ vpsrlq \$14,$T4,$T3++ vpsrlq \$26,$H0,$D0+ vpandq $MASK,$H0,$H0+ vpaddq $D0,$H1,$H1 # h0 -> h1++ vpsrlq \$40,$T4,$T4 # 4++ vpsrlq \$26,$H3,$D3+ vpandq $MASK,$H3,$H3+ vpaddq $D3,$H4,$H4 # h3 -> h4++ vpandq $MASK,$T0,$T0 # 0+ #vpandq $MASK,$T1,$T1 # 1+ #vpandq $MASK,$T3,$T3 # 3+ #vporq $PADBIT,$T4,$T4 # padbit, yes, always++ sub \$128,$len+ ja .Loop_avx512++.Ltail_avx512:+ ################################################################+ # while above multiplications were by r^8 in all lanes, in last+ # iteration we multiply least significant lane by r^8 and most+ # significant one by r, that's why table gets shifted...++ vpsrlq \$32,$R0,$R0 # 0105020603070408+ vpsrlq \$32,$R1,$R1+ vpsrlq \$32,$R2,$R2+ vpsrlq \$32,$S3,$S3+ vpsrlq \$32,$S4,$S4+ vpsrlq \$32,$R3,$R3+ vpsrlq \$32,$R4,$R4+ vpsrlq \$32,$S1,$S1+ vpsrlq \$32,$S2,$S2++ ################################################################+ # load either next or last 64 byte of input+ lea ($inp,$len),$inp++ #vpaddq $H2,$T2,$H2 # accumulate input+ vpaddq $H0,$T0,$H0++ vpmuludq $H2,$R1,$D3 # d3 = h2*r1+ vpmuludq $H2,$R2,$D4 # d4 = h2*r2+ vpmuludq $H2,$S3,$D0 # d0 = h2*s3+ vpandq $MASK,$T1,$T1 # 1+ vpmuludq $H2,$S4,$D1 # d1 = h2*s4+ vpandq $MASK,$T3,$T3 # 3+ vpmuludq $H2,$R0,$D2 # d2 = h2*r0+ vporq $PADBIT,$T4,$T4 # padbit, yes, always+ vpaddq $H1,$T1,$H1 # accumulate input+ vpaddq $H3,$T3,$H3+ vpaddq $H4,$T4,$H4++ vmovdqu 16*0($inp),%x#$T0+ vpmuludq $H0,$R3,$M3+ vpmuludq $H0,$R4,$M4+ vpmuludq $H0,$R0,$M0+ vpmuludq $H0,$R1,$M1+ vpaddq $M3,$D3,$D3 # d3 += h0*r3+ vpaddq $M4,$D4,$D4 # d4 += h0*r4+ vpaddq $M0,$D0,$D0 # d0 += h0*r0+ vpaddq $M1,$D1,$D1 # d1 += h0*r1++ vmovdqu 16*1($inp),%x#$T1+ vpmuludq $H1,$R2,$M3+ vpmuludq $H1,$R3,$M4+ vpmuludq $H1,$S4,$M0+ vpmuludq $H0,$R2,$M2+ vpaddq $M3,$D3,$D3 # d3 += h1*r2+ vpaddq $M4,$D4,$D4 # d4 += h1*r3+ vpaddq $M0,$D0,$D0 # d0 += h1*s4+ vpaddq $M2,$D2,$D2 # d2 += h0*r2++ vinserti128 \$1,16*2($inp),%y#$T0,%y#$T0+ vpmuludq $H3,$R0,$M3+ vpmuludq $H3,$R1,$M4+ vpmuludq $H1,$R0,$M1+ vpmuludq $H1,$R1,$M2+ vpaddq $M3,$D3,$D3 # d3 += h3*r0+ vpaddq $M4,$D4,$D4 # d4 += h3*r1+ vpaddq $M1,$D1,$D1 # d1 += h1*r0+ vpaddq $M2,$D2,$D2 # d2 += h1*r1++ vinserti128 \$1,16*3($inp),%y#$T1,%y#$T1+ vpmuludq $H4,$S4,$M3+ vpmuludq $H4,$R0,$M4+ vpmuludq $H3,$S2,$M0+ vpmuludq $H3,$S3,$M1+ vpmuludq $H3,$S4,$M2+ vpaddq $M3,$D3,$H3 # h3 = d3 + h4*s4+ vpaddq $M4,$D4,$D4 # d4 += h4*r0+ vpaddq $M0,$D0,$D0 # d0 += h3*s2+ vpaddq $M1,$D1,$D1 # d1 += h3*s3+ vpaddq $M2,$D2,$D2 # d2 += h3*s4++ vpmuludq $H4,$S1,$M0+ vpmuludq $H4,$S2,$M1+ vpmuludq $H4,$S3,$M2+ vpaddq $M0,$D0,$H0 # h0 = d0 + h4*s1+ vpaddq $M1,$D1,$H1 # h1 = d2 + h4*s2+ vpaddq $M2,$D2,$H2 # h2 = d3 + h4*s3++ ################################################################+ # horizontal addition++ mov \$1,%eax+ vpermq \$0xb1,$H3,$D3+ vpermq \$0xb1,$D4,$H4+ vpermq \$0xb1,$H0,$D0+ vpermq \$0xb1,$H1,$D1+ vpermq \$0xb1,$H2,$D2+ vpaddq $D3,$H3,$H3+ vpaddq $D4,$H4,$H4+ vpaddq $D0,$H0,$H0+ vpaddq $D1,$H1,$H1+ vpaddq $D2,$H2,$H2++ kmovw %eax,%k3+ vpermq \$0x2,$H3,$D3+ vpermq \$0x2,$H4,$D4+ vpermq \$0x2,$H0,$D0+ vpermq \$0x2,$H1,$D1+ vpermq \$0x2,$H2,$D2+ vpaddq $D3,$H3,$H3+ vpaddq $D4,$H4,$H4+ vpaddq $D0,$H0,$H0+ vpaddq $D1,$H1,$H1+ vpaddq $D2,$H2,$H2++ vextracti64x4 \$0x1,$H3,%y#$D3+ vextracti64x4 \$0x1,$H4,%y#$D4+ vextracti64x4 \$0x1,$H0,%y#$D0+ vextracti64x4 \$0x1,$H1,%y#$D1+ vextracti64x4 \$0x1,$H2,%y#$D2+ vpaddq $D3,$H3,${H3}{%k3}{z} # keep single qword in case+ vpaddq $D4,$H4,${H4}{%k3}{z} # it's passed to .Ltail_avx2+ vpaddq $D0,$H0,${H0}{%k3}{z}+ vpaddq $D1,$H1,${H1}{%k3}{z}+ vpaddq $D2,$H2,${H2}{%k3}{z}+___+map(s/%z/%y/,($T0,$T1,$T2,$T3,$T4, $PADBIT));+map(s/%z/%y/,($H0,$H1,$H2,$H3,$H4, $D0,$D1,$D2,$D3,$D4, $MASK));+$code.=<<___;+ ################################################################+ # lazy reduction (interleaved with input splat)++ vpsrlq \$26,$H3,$D3+ vpand $MASK,$H3,$H3+ vpsrldq \$6,$T0,$T2 # splat input+ vpsrldq \$6,$T1,$T3+ vpunpckhqdq $T1,$T0,$T4 # 4+ vpaddq $D3,$H4,$H4 # h3 -> h4++ vpsrlq \$26,$H0,$D0+ vpand $MASK,$H0,$H0+ vpunpcklqdq $T3,$T2,$T2 # 2:3+ vpunpcklqdq $T1,$T0,$T0 # 0:1+ vpaddq $D0,$H1,$H1 # h0 -> h1++ vpsrlq \$26,$H4,$D4+ vpand $MASK,$H4,$H4++ vpsrlq \$26,$H1,$D1+ vpand $MASK,$H1,$H1+ vpsrlq \$30,$T2,$T3+ vpsrlq \$4,$T2,$T2+ vpaddq $D1,$H2,$H2 # h1 -> h2++ vpaddq $D4,$H0,$H0+ vpsllq \$2,$D4,$D4+ vpsrlq \$26,$T0,$T1+ vpsrlq \$40,$T4,$T4 # 4+ vpaddq $D4,$H0,$H0 # h4 -> h0++ vpsrlq \$26,$H2,$D2+ vpand $MASK,$H2,$H2+ vpand $MASK,$T2,$T2 # 2+ vpand $MASK,$T0,$T0 # 0+ vpaddq $D2,$H3,$H3 # h2 -> h3++ vpsrlq \$26,$H0,$D0+ vpand $MASK,$H0,$H0+ vpaddq $H2,$T2,$H2 # accumulate input for .Ltail_avx2+ vpand $MASK,$T1,$T1 # 1+ vpaddq $D0,$H1,$H1 # h0 -> h1++ vpsrlq \$26,$H3,$D3+ vpand $MASK,$H3,$H3+ vpand $MASK,$T3,$T3 # 3+ vpor 32(%rcx),$T4,$T4 # padbit, yes, always+ vpaddq $D3,$H4,$H4 # h3 -> h4++ lea 0x90(%rsp),%rax # size optimization for .Ltail_avx2+ add \$64,$len+ jnz .Ltail_avx2++ vpsubq $T2,$H2,$H2 # undo input accumulation+ vmovd %x#$H0,`4*0-48-64`($ctx)# save partially reduced+ vmovd %x#$H1,`4*1-48-64`($ctx)+ vmovd %x#$H2,`4*2-48-64`($ctx)+ vmovd %x#$H3,`4*3-48-64`($ctx)+ vmovd %x#$H4,`4*4-48-64`($ctx)+ vzeroall+___+$code.=<<___ if ($win64);+ movdqa 0x50(%r11),%xmm6+ movdqa 0x60(%r11),%xmm7+ movdqa 0x70(%r11),%xmm8+ movdqa 0x80(%r11),%xmm9+ movdqa 0x90(%r11),%xmm10+ movdqa 0xa0(%r11),%xmm11+ movdqa 0xb0(%r11),%xmm12+ movdqa 0xc0(%r11),%xmm13+ movdqa 0xd0(%r11),%xmm14+ movdqa 0xe0(%r11),%xmm15+ lea 0xf8(%r11),%rsp+.Ldo_avx512_epilogue:+___+$code.=<<___ if (!$win64);+ lea 8(%r11),%rsp+.cfi_def_cfa %rsp,8+___+$code.=<<___;+ ret+.cfi_endproc+.size poly1305_blocks_avx512,.-poly1305_blocks_avx512+___+}+if ($avx>3) {+########################################################################+# VPMADD52 version using 2^44 radix.+#+# One can argue that base 2^52 would be more natural. Well, even though+# some operations would be more natural, one has to recognize couple of+# things. Base 2^52 doesn't provide advantage over base 2^44 if you look+# at amount of multiply-n-accumulate operations. Secondly, it makes it+# impossible to pre-compute multiples of 5 [referred to as s[]/sN in+# reference implementations], which means that more such operations+# would have to be performed in inner loop, which in turn makes critical+# path longer. In other words, even though base 2^44 reduction might+# look less elegant, overall critical path is actually shorter...++########################################################################+# Layout of opaque area is following.+#+# unsigned __int64 h[3]; # current hash value base 2^44+# unsigned __int64 s[2]; # key value*20 base 2^44+# unsigned __int64 r[3]; # key value base 2^44+# struct { unsigned __int64 r^1, r^3, r^2, r^4; } R[4];+# # r^n positions reflect+# # placement in register, not+# # memory, R[3] is R[1]*20++$code.=<<___;+.type poly1305_init_base2_44,\@function,3+.align 32+poly1305_init_base2_44:+ xor %rax,%rax+ mov %rax,0($ctx) # initialize hash value+ mov %rax,8($ctx)+ mov %rax,16($ctx)++ cmp \$0,$inp+ je .Lno_key_base2_44++.Linit_base2_44:+ mov \$0x0ffffffc0fffffff,%rax+ mov \$0x0ffffffc0ffffffc,%rcx+ and 0($inp),%rax+ mov \$0x00000fffffffffff,%r8+ and 8($inp),%rcx+ mov \$0x00000fffffffffff,%r9+ and %rax,%r8 # base 2^64 -> base 2^44+ shrd \$44,%rcx,%rax+ mov %r8,40($ctx) # r0+ and %r9,%rax+ shr \$24,%rcx+ mov %rax,48($ctx) # r1+ lea (%rax,%rax,4),%rax # *5+ mov %rcx,56($ctx) # r2+ shl \$2,%rax # magic <<2+ lea (%rcx,%rcx,4),%rcx # *5+ shl \$2,%rcx # magic <<2+ mov %rax,24($ctx) # s1+ mov %rcx,32($ctx) # s2+ movq \$-1,64($ctx) # write impossible value+___+ if ($flavour !~ /kernel/) {+$code.=<<___;+ lea poly1305_blocks_vpmadd52(%rip),%r10+ lea poly1305_emit_base2_44(%rip),%r11+___+$code.=<<___ if ($flavour !~ /elf32/);+ mov %r10,0(%rdx)+ mov %r11,8(%rdx)+___+$code.=<<___ if ($flavour =~ /elf32/);+ mov %r10d,0(%rdx)+ mov %r11d,4(%rdx)+___+ }+$code.=<<___;+ mov \$1,%eax+.Lno_key_base2_44:+ ret+.size poly1305_init_base2_44,.-poly1305_init_base2_44+___+{+my ($h0,$h1,$h2, $d1,$d2,$d3, $r0,$r1,$s2) = map("%r$_",("dx",8..15));+$code.=<<___;+.type poly1305_blocks_base2_44,\@function,4+.align 32+poly1305_blocks_base2_44:+.cfi_startproc+.Lblocks_base2_44:+ push %rbx+.cfi_push %rbx+ push %rbp+.cfi_push %rbp+ push %r12+.cfi_push %r12+ push %r13+.cfi_push %r13+ push %r14+.cfi_push %r14+ push %r15+.cfi_push %r15++ and \$-16,$len+ add $inp,$len # end of buffer+ shl \$40,$padbit+ push $len+.cfi_adjust_cfa_offset 8+.Lblocks_base2_44_body:++ mov 0($ctx),$h0 # load hash value+ mov 8($ctx),$h1+ mov 16($ctx),$h2++ mov 40($ctx),$r0 # load key+ mov 48($ctx),$r1+ mov 32($ctx),$s2+ mov \$0xfffff00000000000,%rax+ jmp .Loop_base2_44+ ud2++.align 32+.Loop_base2_44:+ mov 0($inp),$d2 # load input+ mov 8($inp),$d3+ lea 16($inp),$inp++ andn $d2,%rax,$d1 # base 2^64 -> base 2^44+ shrd \$44,$d3,$d2+ add $d1,$h0 # accumulate input+ shr \$24,$d3+ andn $d2,%rax,$d2+ add $padbit,$h2++ add $d2,$h1+ add $d3,$h2++ #mov $h0,%rdx # h0 is %rdx+ mulx $r0,$d1,%rbx # h0*r0+ mulx $r1,$d2,%rcx # h0*r1+ mulx 56($ctx),$d3,%rbp # h0*r2++ mov $h1,%rdx+ mulx $s2,%rax,$h1 # h1*s2+ add %rax,$d1+ adc %rbx,$h1+ mulx $r0,%rax,%rbx # h1*r0+ add %rax,$d2+ adc %rbx,%rcx+ mulx $r1,%rax,%rbx # h1*r1+ mov $h2,%rdx+ add %rax,$d3+ adc %rbx,%rbp++ mulx 24($ctx),%rax,%rbx # h2*s1+ add %rax,$d1+ adc %rbx,$h1+ mulx $s2,%rax,$h2 # h2*s2+ add %rax,$d2+ adc %rcx,$h2+ mulx $r0,%rax,%rbx # h2*r0+ add %rax,$d3+ adc %rbx,%rbp++ mov \$0xfffff00000000000,%rax+ andn $d1,%rax,$h0+ shrd \$44,$h1,$d1+ add $d1,$d2+ adc \$0,$h2+ andn $d2,%rax,$h1+ shrd \$44,$h2,$d2+ mov \$0x03ffffffffff,$h2+ add $d2,$d3+ adc \$0,%rbp+ and $d3,$h2+ shrd \$42,%rbp,$d3++ mov \$0x10000000000,$padbit+ lea ($d3,$d3,4),$d3 # *=5+ add $d3,$h0++ cmp 0(%rsp),$inp+ jb .Loop_base2_44++ mov $h0,0($ctx) # store hash value+ mov $h1,8($ctx)+ mov $h2,16($ctx)++ mov 8(%rsp),%r15+.cfi_restore %r15+ mov 16(%rsp),%r14+.cfi_restore %r14+ mov 24(%rsp),%r13+.cfi_restore %r13+ mov 32(%rsp),%r12+.cfi_restore %r12+ mov 40(%rsp),%rbp+.cfi_restore %rbp+ mov 48(%rsp),%rbx+.cfi_restore %rbx+ lea 56(%rsp),%rsp+.cfi_adjust_cfa_offset -56+.Lblocks_base2_44_epilogue:+ ret+.cfi_endproc+.size poly1305_blocks_base2_44,.-poly1305_blocks_base2_44+___+}+{+my ($H0,$H1,$H2,$r2r1r0,$r1r0s2,$r0s2s1,$Dlo,$Dhi) = map("%ymm$_",(0..5,16,17));+my ($T0,$inp_permd,$inp_shift,$PAD) = map("%ymm$_",(18..21));+my ($reduc_mask,$reduc_rght,$reduc_left) = map("%ymm$_",(22..25));+my ($T1,$T2,$T3) = map("%ymm$_",(26..28));++$code.=<<___;+.type poly1305_blocks_vpmadd52,\@function,4+.align 32+poly1305_blocks_vpmadd52:+ and \$-16,$len+ jz .Lno_data_vpmadd52 # too short++ mov 64($ctx),%r8 # peek on power of the key++ # if powers of the key are not calculated yet, process up to 3+ # blocks with scalar single-block subroutine above, otherwise+ # ensure that input length is divisible by 2 blocks and pass+ # the rest down to next subroutine...++ mov \$0x30,%r9+ mov \$0x10,%r10+ cmp \$0x40,$len # is input long+ cmovae %r10,%r9+ test %r8,%r8 # is power value impossible?+ cmovns %r10,%r9++ and $len,%r9 # is input of favourable length?+ jz .Lblocks_vpmadd52_4x++ sub %r9,$len+ cmovz %r9,$len+ jz .Lblocks_base2_44++ #########################################+ mov \$7,%r10d+ mov \$1,%r11d+ shl \$40,$padbit+ kmovw %r10d,%k7+ lea .L2_44_inp_permd(%rip),%r10+ kmovw %r11d,%k1++ vmovq $padbit,%x#$PAD+ shr \$40,$padbit # restore original value+ vmovdqa64 0(%r10),$inp_permd # .L2_44_inp_permd+ vmovdqa64 32(%r10),$inp_shift # .L2_44_inp_shift+ vpermq \$0xcf,$PAD,$PAD+ vmovdqa64 64(%r10),$reduc_mask # .L2_44_mask++ vmovdqu64 0($ctx),${Dlo}{%k7}{z} # load hash value+ vmovdqu64 40($ctx),${r2r1r0}{%k7}{z} # load keys+ vmovdqu64 32($ctx),${r1r0s2}{%k7}{z}+ vmovdqu64 24($ctx),${r0s2s1}{%k7}{z}++ vmovdqa64 96(%r10),$reduc_rght # .L2_44_shift_rgt+ vmovdqa64 128(%r10),$reduc_left # .L2_44_shift_lft++ vmovdqu32 0($inp),%x#$T0 # load input as ----3210+ lea 16($inp),$inp++ vpermd $T0,$inp_permd,$T0 # ----3210 -> --322110+ vpsrlvq $inp_shift,$T0,$T0+ vpandq $reduc_mask,$T0,$T0+ vporq $PAD,$T0,$T0++ vpaddq $T0,$Dlo,$Dlo # accumulate input+ vpxord $T2,$T2,$T2+ vpxord $T3,$T3,$T3++ vpermq \$0,$Dlo,${H0}{%k7}{z} # smash hash value+ vpermq \$0b01010101,$Dlo,${H1}{%k7}{z}+ vpermq \$0b10101010,$Dlo,${H2}{%k7}{z}++ vpxord $T0,$T0,$T0+ vpxord $T1,$T1,$T1+ vpmadd52luq $r2r1r0,$H0,$T2+ vpmadd52huq $r2r1r0,$H0,$T3++ vpxord $Dlo,$Dlo,$Dlo+ vpxord $Dhi,$Dhi,$Dhi+ vpmadd52luq $r1r0s2,$H1,$T0+ vpmadd52huq $r1r0s2,$H1,$T1++ vpmadd52luq $r0s2s1,$H2,$Dlo+ vpmadd52huq $r0s2s1,$H2,$Dhi++ vpaddq $T0,$T2,$T2+ vpaddq $T1,$T3,$T3+ vpaddq $T2,$Dlo,$Dlo+ vpaddq $T3,$Dhi,$Dhi++ vpsrlvq $reduc_rght,$Dlo,$T0 # 0 in topmost qword+ vpsllvq $reduc_left,$Dhi,$Dhi # 0 in topmost qword+ vpandq $reduc_mask,$Dlo,$Dlo++ vpaddq $T0,$Dhi,$Dhi++ vpermq \$0b10010011,$Dhi,$Dhi # 0 in lowest qword++ vpaddq $Dhi,$Dlo,$Dlo # note topmost qword :-)++ vpsrlvq $reduc_rght,$Dlo,$T0 # 0 in topmost word+ vpandq $reduc_mask,$Dlo,$Dlo++ vpermq \$0b10010011,$T0,$T0++ vpaddq $T0,$Dlo,$Dlo++ vpermq \$0b10010011,$Dlo,${T0}{%k1}{z}++ vpaddq $T0,$Dlo,$Dlo+ vpsllq \$2,$T0,$T0++ vpaddq $T0,$Dlo,$Dlo++ vmovdqu64 $Dlo,0($ctx){%k7} # store hash value++ jmp .Lblocks_vpmadd52_4x++.Lno_data_vpmadd52:+ ret+.size poly1305_blocks_vpmadd52,.-poly1305_blocks_vpmadd52+___+}+{+########################################################################+# As implied by its name 4x subroutine processes 4 blocks in parallel+# (but handles even 4*n+2 blocks lengths). It takes up to 4th key power+# and is handled in 256-bit %ymm registers.++my ($H0,$H1,$H2,$R0,$R1,$R2,$S1,$S2) = map("%ymm$_",(0..5,16,17));+my ($D0lo,$D0hi,$D1lo,$D1hi,$D2lo,$D2hi) = map("%ymm$_",(18..23));+my ($T0,$T1,$T2,$T3,$T4,$tmp,$mask44,$PAD) = map("%ymm$_",(24..31));++$code.=<<___;+.type poly1305_blocks_vpmadd52_4x,\@function,4+.align 32+poly1305_blocks_vpmadd52_4x:+ and \$-16,$len+ jz .Lno_data_vpmadd52_4x # too short++ mov 64($ctx),%r8 # peek on power of the key++.Lblocks_vpmadd52_4x:+ shl \$40,$padbit+ shr \$4,$len+ vpbroadcastq $padbit,$PAD++ vmovdqa64 .Lx_mask44(%rip),$mask44+ mov \$5,%eax+ kmovw %eax,%k1 # used in 2x path++ test %r8,%r8 # is power value impossible?+ js .Linit_vpmadd52 # if it is, then init R[4]++ vmovq 0($ctx),%x#$H0 # load current hash value+ vmovq 8($ctx),%x#$H1+ vmovq 16($ctx),%x#$H2++ test \$3,$len # is length 4*n+2?+ jnz .Lblocks_vpmadd52_2x_do++.Lblocks_vpmadd52_4x_do:+ vpbroadcastq 64($ctx),$R0 # load 4th power of the key+ vpbroadcastq 96($ctx),$R1+ vpbroadcastq 128($ctx),$R2+ vpbroadcastq 160($ctx),$S1++.Lblocks_vpmadd52_4x_key_loaded:+ vpsllq \$2,$R2,$S2 # S2 = R2*5*4+ vpaddq $R2,$S2,$S2+ vpsllq \$2,$S2,$S2++ #test \$7,$len # is len 8*n?+ #jz .Lblocks_vpmadd52_8x++ vmovdqu64 16*0($inp),$T2 # load data+ vmovdqu64 16*2($inp),$T3+ lea 16*4($inp),$inp++ vpunpcklqdq $T3,$T2,$T1 # transpose data+ vpunpckhqdq $T3,$T2,$T3++ # at this point 64-bit lanes are ordered as 3-1-2-0++ vpsrlq \$24,$T3,$T2 # splat the data+ vporq $PAD,$T2,$T2+ vpaddq $T2,$H2,$H2 # accumulate input+ vpandq $mask44,$T1,$T0+ vpsrlq \$44,$T1,$T1+ vpsllq \$20,$T3,$T3+ vporq $T3,$T1,$T1+ vpandq $mask44,$T1,$T1++ sub \$4,$len+ jz .Ltail_vpmadd52_4x+ jmp .Loop_vpmadd52_4x+ ud2++.align 32+.Linit_vpmadd52:+ vmovq 24($ctx),%x#$S1 # load key+ vmovq 56($ctx),%x#$H2+ vmovq 32($ctx),%x#$S2+ vmovq 40($ctx),%x#$R0+ vmovq 48($ctx),%x#$R1++ vmovdqa $R0,$H0+ vmovdqa $R1,$H1+ vmovdqa $H2,$R2++ mov \$2,%eax++.Lmul_init_vpmadd52:+ vpxorq $D0lo,$D0lo,$D0lo+ vpxorq $D0hi,$D0hi,$D0hi+ vpxorq $D1lo,$D1lo,$D1lo+ vpxorq $D1hi,$D1hi,$D1hi+ vpxorq $D2lo,$D2lo,$D2lo+ vpxorq $D2hi,$D2hi,$D2hi+ vpmadd52luq $H2,$S1,$D0lo+ vpxorq $T0,$T0,$T0+ vpxorq $T1,$T1,$T1+ vpmadd52huq $H2,$S1,$D0hi+ vpxorq $T2,$T2,$T2+ vpxorq $T3,$T3,$T3+ vpmadd52luq $H2,$S2,$D1lo+ vpxorq $T4,$T4,$T4+ vpxorq $tmp,$tmp,$tmp+ vpmadd52huq $H2,$S2,$D1hi+ vpmadd52luq $H2,$R0,$D2lo+ vpmadd52huq $H2,$R0,$D2hi++ vpmadd52luq $H0,$R0,$T0+ vpmadd52huq $H0,$R0,$T1+ vpmadd52luq $H0,$R1,$T2+ vpmadd52huq $H0,$R1,$T3+ vpmadd52luq $H0,$R2,$T4+ vpmadd52huq $H0,$R2,$tmp++ vpmadd52luq $H1,$S2,$D0lo+ vpmadd52huq $H1,$S2,$D0hi+ vpmadd52luq $H1,$R0,$D1lo+ vpmadd52huq $H1,$R0,$D1hi+ vpaddq $T0,$D0lo,$D0lo+ vpaddq $T1,$D0hi,$D0hi+ vpmadd52luq $H1,$R1,$D2lo+ vpaddq $T2,$D1lo,$D1lo+ vpaddq $T3,$D1hi,$D1hi+ vpmadd52huq $H1,$R1,$D2hi+ vpaddq $T4,$D2lo,$D2lo+ vpaddq $tmp,$D2hi,$D2hi++ ################################################################+ # partial reduction+ vpsrlq \$44,$D0lo,$tmp+ vpsllq \$8,$D0hi,$D0hi+ vpandq $mask44,$D0lo,$H0+ vpaddq $tmp,$D0hi,$D0hi++ vpaddq $D0hi,$D1lo,$D1lo++ vpsrlq \$44,$D1lo,$tmp+ vpsllq \$8,$D1hi,$D1hi+ vpandq $mask44,$D1lo,$H1+ vpaddq $tmp,$D1hi,$D1hi++ vpaddq $D1hi,$D2lo,$D2lo++ vpsrlq \$42,$D2lo,$tmp+ vpsllq \$10,$D2hi,$D2hi+ vpandq .Lx_mask42(%rip),$D2lo,$H2+ vpaddq $tmp,$D2hi,$D2hi++ vpaddq $D2hi,$H0,$H0+ vpsllq \$2,$D2hi,$D2hi++ vpaddq $D2hi,$H0,$H0++ vpsrlq \$44,$H0,$tmp # additional step+ vpandq $mask44,$H0,$H0++ vpaddq $tmp,$H1,$H1++ dec %eax+ jz .Ldone_init_vpmadd52++ vpunpcklqdq $R1,$H1,$R1 # 1,2+ vpbroadcastq %x#$H1,%x#$H1 # 2,2+ vpunpcklqdq $R2,$H2,$R2+ vpbroadcastq %x#$H2,%x#$H2+ vpunpcklqdq $R0,$H0,$R0+ vpbroadcastq %x#$H0,%x#$H0++ vpsllq \$2,$R1,$S1 # S1 = R1*5*4+ vpsllq \$2,$R2,$S2 # S2 = R2*5*4+ vpaddq $R1,$S1,$S1+ vpaddq $R2,$S2,$S2+ vpsllq \$2,$S1,$S1+ vpsllq \$2,$S2,$S2++ jmp .Lmul_init_vpmadd52+ ud2++.align 32+.Ldone_init_vpmadd52:+ vinserti128 \$1,%x#$R1,$H1,$R1 # 1,2,3,4+ vinserti128 \$1,%x#$R2,$H2,$R2+ vinserti128 \$1,%x#$R0,$H0,$R0++ vpermq \$0b11011000,$R1,$R1 # 1,3,2,4+ vpermq \$0b11011000,$R2,$R2+ vpermq \$0b11011000,$R0,$R0++ vpsllq \$2,$R1,$S1 # S1 = R1*5*4+ vpaddq $R1,$S1,$S1+ vpsllq \$2,$S1,$S1++ vmovq 0($ctx),%x#$H0 # load current hash value+ vmovq 8($ctx),%x#$H1+ vmovq 16($ctx),%x#$H2++ test \$3,$len # is length 4*n+2?+ jnz .Ldone_init_vpmadd52_2x++ vmovdqu64 $R0,64($ctx) # save key powers+ vpbroadcastq %x#$R0,$R0 # broadcast 4th power+ vmovdqu64 $R1,96($ctx)+ vpbroadcastq %x#$R1,$R1+ vmovdqu64 $R2,128($ctx)+ vpbroadcastq %x#$R2,$R2+ vmovdqu64 $S1,160($ctx)+ vpbroadcastq %x#$S1,$S1++ jmp .Lblocks_vpmadd52_4x_key_loaded+ ud2++.align 32+.Ldone_init_vpmadd52_2x:+ vmovdqu64 $R0,64($ctx) # save key powers+ vpsrldq \$8,$R0,$R0 # 0-1-0-2+ vmovdqu64 $R1,96($ctx)+ vpsrldq \$8,$R1,$R1+ vmovdqu64 $R2,128($ctx)+ vpsrldq \$8,$R2,$R2+ vmovdqu64 $S1,160($ctx)+ vpsrldq \$8,$S1,$S1+ jmp .Lblocks_vpmadd52_2x_key_loaded+ ud2++.align 32+.Lblocks_vpmadd52_2x_do:+ vmovdqu64 128+8($ctx),${R2}{%k1}{z}# load 2nd and 1st key powers+ vmovdqu64 160+8($ctx),${S1}{%k1}{z}+ vmovdqu64 64+8($ctx),${R0}{%k1}{z}+ vmovdqu64 96+8($ctx),${R1}{%k1}{z}++.Lblocks_vpmadd52_2x_key_loaded:+ vmovdqu64 16*0($inp),$T2 # load data+ vpxorq $T3,$T3,$T3+ lea 16*2($inp),$inp++ vpunpcklqdq $T3,$T2,$T1 # transpose data+ vpunpckhqdq $T3,$T2,$T3++ # at this point 64-bit lanes are ordered as x-1-x-0++ vpsrlq \$24,$T3,$T2 # splat the data+ vporq $PAD,$T2,$T2+ vpaddq $T2,$H2,$H2 # accumulate input+ vpandq $mask44,$T1,$T0+ vpsrlq \$44,$T1,$T1+ vpsllq \$20,$T3,$T3+ vporq $T3,$T1,$T1+ vpandq $mask44,$T1,$T1++ jmp .Ltail_vpmadd52_2x+ ud2++.align 32+.Loop_vpmadd52_4x:+ #vpaddq $T2,$H2,$H2 # accumulate input+ vpaddq $T0,$H0,$H0+ vpaddq $T1,$H1,$H1++ vpxorq $D0lo,$D0lo,$D0lo+ vpxorq $D0hi,$D0hi,$D0hi+ vpxorq $D1lo,$D1lo,$D1lo+ vpxorq $D1hi,$D1hi,$D1hi+ vpxorq $D2lo,$D2lo,$D2lo+ vpxorq $D2hi,$D2hi,$D2hi+ vpmadd52luq $H2,$S1,$D0lo+ vpxorq $T0,$T0,$T0+ vpxorq $T1,$T1,$T1+ vpmadd52huq $H2,$S1,$D0hi+ vpxorq $T2,$T2,$T2+ vpxorq $T3,$T3,$T3+ vpmadd52luq $H2,$S2,$D1lo+ vpxorq $T4,$T4,$T4+ vpxorq $tmp,$tmp,$tmp+ vpmadd52huq $H2,$S2,$D1hi+ vpmadd52luq $H2,$R0,$D2lo+ vpmadd52huq $H2,$R0,$D2hi++ vpmadd52luq $H0,$R0,$T0+ vpmadd52huq $H0,$R0,$T1+ vpmadd52luq $H0,$R1,$T2+ vpmadd52huq $H0,$R1,$T3+ vpmadd52luq $H0,$R2,$T4+ vpmadd52huq $H0,$R2,$tmp++ vpmadd52luq $H1,$S2,$D0lo+ vpmadd52huq $H1,$S2,$D0hi+ vpmadd52luq $H1,$R0,$D1lo+ vpmadd52huq $H1,$R0,$D1hi+ vpaddq $T0,$D0lo,$D0lo+ vpaddq $T1,$D0hi,$D0hi+ vpmadd52luq $H1,$R1,$D2lo+ vpaddq $T2,$D1lo,$D1lo+ vpaddq $T3,$D1hi,$D1hi+ vpmadd52huq $H1,$R1,$D2hi+ vpaddq $T4,$D2lo,$D2lo+ vpaddq $tmp,$D2hi,$D2hi++ vmovdqu64 16*0($inp),$T2 # load data+ vmovdqu64 16*2($inp),$T3+ lea 16*4($inp),$inp+ vpunpcklqdq $T3,$T2,$T1 # transpose data+ vpunpckhqdq $T3,$T2,$T3++ ################################################################+ # partial reduction (interleaved with data splat)+ vpsrlq \$44,$D0lo,$tmp+ vpsllq \$8,$D0hi,$D0hi+ vpandq $mask44,$D0lo,$H0+ vpaddq $tmp,$D0hi,$D0hi++ vpsrlq \$24,$T3,$T2+ vporq $PAD,$T2,$T2+ vpaddq $D0hi,$D1lo,$D1lo++ vpsrlq \$44,$D1lo,$tmp+ vpsllq \$8,$D1hi,$D1hi+ vpandq $mask44,$D1lo,$H1+ vpaddq $tmp,$D1hi,$D1hi++ vpandq $mask44,$T1,$T0+ vpsrlq \$44,$T1,$T1+ vpsllq \$20,$T3,$T3+ vpaddq $D1hi,$D2lo,$D2lo++ vpsrlq \$42,$D2lo,$tmp+ vpsllq \$10,$D2hi,$D2hi+ vpandq .Lx_mask42(%rip),$D2lo,$H2+ vpaddq $tmp,$D2hi,$D2hi++ vpaddq $T2,$H2,$H2 # accumulate input+ vpaddq $D2hi,$H0,$H0+ vpsllq \$2,$D2hi,$D2hi++ vpaddq $D2hi,$H0,$H0+ vporq $T3,$T1,$T1+ vpandq $mask44,$T1,$T1++ vpsrlq \$44,$H0,$tmp # additional step+ vpandq $mask44,$H0,$H0++ vpaddq $tmp,$H1,$H1++ sub \$4,$len # len-=64+ jnz .Loop_vpmadd52_4x++.Ltail_vpmadd52_4x:+ vmovdqu64 128($ctx),$R2 # load all key powers+ vmovdqu64 160($ctx),$S1+ vmovdqu64 64($ctx),$R0+ vmovdqu64 96($ctx),$R1++.Ltail_vpmadd52_2x:+ vpsllq \$2,$R2,$S2 # S2 = R2*5*4+ vpaddq $R2,$S2,$S2+ vpsllq \$2,$S2,$S2++ #vpaddq $T2,$H2,$H2 # accumulate input+ vpaddq $T0,$H0,$H0+ vpaddq $T1,$H1,$H1++ vpxorq $D0lo,$D0lo,$D0lo+ vpxorq $D0hi,$D0hi,$D0hi+ vpxorq $D1lo,$D1lo,$D1lo+ vpxorq $D1hi,$D1hi,$D1hi+ vpxorq $D2lo,$D2lo,$D2lo+ vpxorq $D2hi,$D2hi,$D2hi+ vpmadd52luq $H2,$S1,$D0lo+ vpxorq $T0,$T0,$T0+ vpxorq $T1,$T1,$T1+ vpmadd52huq $H2,$S1,$D0hi+ vpxorq $T2,$T2,$T2+ vpxorq $T3,$T3,$T3+ vpmadd52luq $H2,$S2,$D1lo+ vpxorq $T4,$T4,$T4+ vpxorq $tmp,$tmp,$tmp+ vpmadd52huq $H2,$S2,$D1hi+ vpmadd52luq $H2,$R0,$D2lo+ vpmadd52huq $H2,$R0,$D2hi++ vpmadd52luq $H0,$R0,$T0+ vpmadd52huq $H0,$R0,$T1+ vpmadd52luq $H0,$R1,$T2+ vpmadd52huq $H0,$R1,$T3+ vpmadd52luq $H0,$R2,$T4+ vpmadd52huq $H0,$R2,$tmp++ vpmadd52luq $H1,$S2,$D0lo+ vpmadd52huq $H1,$S2,$D0hi+ vpmadd52luq $H1,$R0,$D1lo+ vpmadd52huq $H1,$R0,$D1hi+ vpaddq $T0,$D0lo,$D0lo+ vpaddq $T1,$D0hi,$D0hi+ vpmadd52luq $H1,$R1,$D2lo+ vpaddq $T2,$D1lo,$D1lo+ vpaddq $T3,$D1hi,$D1hi+ vpmadd52huq $H1,$R1,$D2hi+ vpaddq $T4,$D2lo,$D2lo+ vpaddq $tmp,$D2hi,$D2hi++ ################################################################+ # horizontal addition++ mov \$1,%eax+ kmovw %eax,%k1+ vpsrldq \$8,$D0lo,$T0+ vpsrldq \$8,$D0hi,$H0+ vpsrldq \$8,$D1lo,$T1+ vpsrldq \$8,$D1hi,$H1+ vpaddq $T0,$D0lo,$D0lo+ vpaddq $H0,$D0hi,$D0hi+ vpsrldq \$8,$D2lo,$T2+ vpsrldq \$8,$D2hi,$H2+ vpaddq $T1,$D1lo,$D1lo+ vpaddq $H1,$D1hi,$D1hi+ vpermq \$0x2,$D0lo,$T0+ vpermq \$0x2,$D0hi,$H0+ vpaddq $T2,$D2lo,$D2lo+ vpaddq $H2,$D2hi,$D2hi++ vpermq \$0x2,$D1lo,$T1+ vpermq \$0x2,$D1hi,$H1+ vpaddq $T0,$D0lo,${D0lo}{%k1}{z}+ vpaddq $H0,$D0hi,${D0hi}{%k1}{z}+ vpermq \$0x2,$D2lo,$T2+ vpermq \$0x2,$D2hi,$H2+ vpaddq $T1,$D1lo,${D1lo}{%k1}{z}+ vpaddq $H1,$D1hi,${D1hi}{%k1}{z}+ vpaddq $T2,$D2lo,${D2lo}{%k1}{z}+ vpaddq $H2,$D2hi,${D2hi}{%k1}{z}++ ################################################################+ # partial reduction+ vpsrlq \$44,$D0lo,$tmp+ vpsllq \$8,$D0hi,$D0hi+ vpandq $mask44,$D0lo,$H0+ vpaddq $tmp,$D0hi,$D0hi++ vpaddq $D0hi,$D1lo,$D1lo++ vpsrlq \$44,$D1lo,$tmp+ vpsllq \$8,$D1hi,$D1hi+ vpandq $mask44,$D1lo,$H1+ vpaddq $tmp,$D1hi,$D1hi++ vpaddq $D1hi,$D2lo,$D2lo++ vpsrlq \$42,$D2lo,$tmp+ vpsllq \$10,$D2hi,$D2hi+ vpandq .Lx_mask42(%rip),$D2lo,$H2+ vpaddq $tmp,$D2hi,$D2hi++ vpaddq $D2hi,$H0,$H0+ vpsllq \$2,$D2hi,$D2hi++ vpaddq $D2hi,$H0,$H0++ vpsrlq \$44,$H0,$tmp # additional step+ vpandq $mask44,$H0,$H0++ vpaddq $tmp,$H1,$H1+ # at this point $len is+ # either 4*n+2 or 0...+ sub \$2,$len # len-=32+ ja .Lblocks_vpmadd52_4x_do++ vmovq %x#$H0,0($ctx)+ vmovq %x#$H1,8($ctx)+ vmovq %x#$H2,16($ctx)+ vzeroall++.Lno_data_vpmadd52_4x:+ ret+.size poly1305_blocks_vpmadd52_4x,.-poly1305_blocks_vpmadd52_4x+___+}+if (0) {+########################################################################+# As implied by its name 8x subroutine processes 8 blocks in parallel...+# This is intermediate version, as it's used only in cases when input+# length is either 8*n, 8*n+1 or 8*n+2...++my ($H0,$H1,$H2,$R0,$R1,$R2,$S1,$S2) = map("%ymm$_",(0..5,16,17));+my ($D0lo,$D0hi,$D1lo,$D1hi,$D2lo,$D2hi) = map("%ymm$_",(18..23));+my ($T0,$T1,$T2,$T3,$mask44,$mask42,$tmp,$PAD) = map("%ymm$_",(24..31));+my ($RR0,$RR1,$RR2,$SS1,$SS2) = map("%ymm$_",(6..10));++$code.=<<___;+.type poly1305_blocks_vpmadd52_8x,\@function,4+.align 32+poly1305_blocks_vpmadd52_8x:+ shr \$4,$len+ jz .Lno_data_vpmadd52_8x # too short++ shl \$40,$padbit+ mov 64($ctx),%r8 # peek on power of the key++ vmovdqa64 .Lx_mask44(%rip),$mask44+ vmovdqa64 .Lx_mask42(%rip),$mask42++ test %r8,%r8 # is power value impossible?+ js .Linit_vpmadd52 # if it is, then init R[4]++ vmovq 0($ctx),%x#$H0 # load current hash value+ vmovq 8($ctx),%x#$H1+ vmovq 16($ctx),%x#$H2++.Lblocks_vpmadd52_8x:+ ################################################################+ # fist we calculate more key powers++ vmovdqu64 128($ctx),$R2 # load 1-3-2-4 powers+ vmovdqu64 160($ctx),$S1+ vmovdqu64 64($ctx),$R0+ vmovdqu64 96($ctx),$R1++ vpsllq \$2,$R2,$S2 # S2 = R2*5*4+ vpaddq $R2,$S2,$S2+ vpsllq \$2,$S2,$S2++ vpbroadcastq %x#$R2,$RR2 # broadcast 4th power+ vpbroadcastq %x#$R0,$RR0+ vpbroadcastq %x#$R1,$RR1++ vpxorq $D0lo,$D0lo,$D0lo+ vpmadd52luq $RR2,$S1,$D0lo+ vpxorq $D0hi,$D0hi,$D0hi+ vpmadd52huq $RR2,$S1,$D0hi+ vpxorq $D1lo,$D1lo,$D1lo+ vpmadd52luq $RR2,$S2,$D1lo+ vpxorq $D1hi,$D1hi,$D1hi+ vpmadd52huq $RR2,$S2,$D1hi+ vpxorq $D2lo,$D2lo,$D2lo+ vpmadd52luq $RR2,$R0,$D2lo+ vpxorq $D2hi,$D2hi,$D2hi+ vpmadd52huq $RR2,$R0,$D2hi++ vpmadd52luq $RR0,$R0,$D0lo+ vpmadd52huq $RR0,$R0,$D0hi+ vpmadd52luq $RR0,$R1,$D1lo+ vpmadd52huq $RR0,$R1,$D1hi+ vpmadd52luq $RR0,$R2,$D2lo+ vpmadd52huq $RR0,$R2,$D2hi++ vpmadd52luq $RR1,$S2,$D0lo+ vpmadd52huq $RR1,$S2,$D0hi+ vpmadd52luq $RR1,$R0,$D1lo+ vpmadd52huq $RR1,$R0,$D1hi+ vpmadd52luq $RR1,$R1,$D2lo+ vpmadd52huq $RR1,$R1,$D2hi++ ################################################################+ # partial reduction+ vpsrlq \$44,$D0lo,$tmp+ vpsllq \$8,$D0hi,$D0hi+ vpandq $mask44,$D0lo,$RR0+ vpaddq $tmp,$D0hi,$D0hi++ vpaddq $D0hi,$D1lo,$D1lo++ vpsrlq \$44,$D1lo,$tmp+ vpsllq \$8,$D1hi,$D1hi+ vpandq $mask44,$D1lo,$RR1+ vpaddq $tmp,$D1hi,$D1hi++ vpaddq $D1hi,$D2lo,$D2lo++ vpsrlq \$42,$D2lo,$tmp+ vpsllq \$10,$D2hi,$D2hi+ vpandq $mask42,$D2lo,$RR2+ vpaddq $tmp,$D2hi,$D2hi++ vpaddq $D2hi,$RR0,$RR0+ vpsllq \$2,$D2hi,$D2hi++ vpaddq $D2hi,$RR0,$RR0++ vpsrlq \$44,$RR0,$tmp # additional step+ vpandq $mask44,$RR0,$RR0++ vpaddq $tmp,$RR1,$RR1++ ################################################################+ # At this point Rx holds 1324 powers, RRx - 5768, and the goal+ # is 15263748, which reflects how data is loaded...++ vpunpcklqdq $R2,$RR2,$T2 # 3748+ vpunpckhqdq $R2,$RR2,$R2 # 1526+ vpunpcklqdq $R0,$RR0,$T0+ vpunpckhqdq $R0,$RR0,$R0+ vpunpcklqdq $R1,$RR1,$T1+ vpunpckhqdq $R1,$RR1,$R1+___+######## switch to %zmm+map(s/%y/%z/, $H0,$H1,$H2,$R0,$R1,$R2,$S1,$S2);+map(s/%y/%z/, $D0lo,$D0hi,$D1lo,$D1hi,$D2lo,$D2hi);+map(s/%y/%z/, $T0,$T1,$T2,$T3,$mask44,$mask42,$tmp,$PAD);+map(s/%y/%z/, $RR0,$RR1,$RR2,$SS1,$SS2);++$code.=<<___;+ vshufi64x2 \$0x44,$R2,$T2,$RR2 # 15263748+ vshufi64x2 \$0x44,$R0,$T0,$RR0+ vshufi64x2 \$0x44,$R1,$T1,$RR1++ vmovdqu64 16*0($inp),$T2 # load data+ vmovdqu64 16*4($inp),$T3+ lea 16*8($inp),$inp++ vpsllq \$2,$RR2,$SS2 # S2 = R2*5*4+ vpsllq \$2,$RR1,$SS1 # S1 = R1*5*4+ vpaddq $RR2,$SS2,$SS2+ vpaddq $RR1,$SS1,$SS1+ vpsllq \$2,$SS2,$SS2+ vpsllq \$2,$SS1,$SS1++ vpbroadcastq $padbit,$PAD+ vpbroadcastq %x#$mask44,$mask44+ vpbroadcastq %x#$mask42,$mask42++ vpbroadcastq %x#$SS1,$S1 # broadcast 8th power+ vpbroadcastq %x#$SS2,$S2+ vpbroadcastq %x#$RR0,$R0+ vpbroadcastq %x#$RR1,$R1+ vpbroadcastq %x#$RR2,$R2++ vpunpcklqdq $T3,$T2,$T1 # transpose data+ vpunpckhqdq $T3,$T2,$T3++ # at this point 64-bit lanes are ordered as 73625140++ vpsrlq \$24,$T3,$T2 # splat the data+ vporq $PAD,$T2,$T2+ vpaddq $T2,$H2,$H2 # accumulate input+ vpandq $mask44,$T1,$T0+ vpsrlq \$44,$T1,$T1+ vpsllq \$20,$T3,$T3+ vporq $T3,$T1,$T1+ vpandq $mask44,$T1,$T1++ sub \$8,$len+ jz .Ltail_vpmadd52_8x+ jmp .Loop_vpmadd52_8x++.align 32+.Loop_vpmadd52_8x:+ #vpaddq $T2,$H2,$H2 # accumulate input+ vpaddq $T0,$H0,$H0+ vpaddq $T1,$H1,$H1++ vpxorq $D0lo,$D0lo,$D0lo+ vpmadd52luq $H2,$S1,$D0lo+ vpxorq $D0hi,$D0hi,$D0hi+ vpmadd52huq $H2,$S1,$D0hi+ vpxorq $D1lo,$D1lo,$D1lo+ vpmadd52luq $H2,$S2,$D1lo+ vpxorq $D1hi,$D1hi,$D1hi+ vpmadd52huq $H2,$S2,$D1hi+ vpxorq $D2lo,$D2lo,$D2lo+ vpmadd52luq $H2,$R0,$D2lo+ vpxorq $D2hi,$D2hi,$D2hi+ vpmadd52huq $H2,$R0,$D2hi++ vmovdqu64 16*0($inp),$T2 # load data+ vmovdqu64 16*4($inp),$T3+ lea 16*8($inp),$inp+ vpmadd52luq $H0,$R0,$D0lo+ vpmadd52huq $H0,$R0,$D0hi+ vpmadd52luq $H0,$R1,$D1lo+ vpmadd52huq $H0,$R1,$D1hi+ vpmadd52luq $H0,$R2,$D2lo+ vpmadd52huq $H0,$R2,$D2hi++ vpunpcklqdq $T3,$T2,$T1 # transpose data+ vpunpckhqdq $T3,$T2,$T3+ vpmadd52luq $H1,$S2,$D0lo+ vpmadd52huq $H1,$S2,$D0hi+ vpmadd52luq $H1,$R0,$D1lo+ vpmadd52huq $H1,$R0,$D1hi+ vpmadd52luq $H1,$R1,$D2lo+ vpmadd52huq $H1,$R1,$D2hi++ ################################################################+ # partial reduction (interleaved with data splat)+ vpsrlq \$44,$D0lo,$tmp+ vpsllq \$8,$D0hi,$D0hi+ vpandq $mask44,$D0lo,$H0+ vpaddq $tmp,$D0hi,$D0hi++ vpsrlq \$24,$T3,$T2+ vporq $PAD,$T2,$T2+ vpaddq $D0hi,$D1lo,$D1lo++ vpsrlq \$44,$D1lo,$tmp+ vpsllq \$8,$D1hi,$D1hi+ vpandq $mask44,$D1lo,$H1+ vpaddq $tmp,$D1hi,$D1hi++ vpandq $mask44,$T1,$T0+ vpsrlq \$44,$T1,$T1+ vpsllq \$20,$T3,$T3+ vpaddq $D1hi,$D2lo,$D2lo++ vpsrlq \$42,$D2lo,$tmp+ vpsllq \$10,$D2hi,$D2hi+ vpandq $mask42,$D2lo,$H2+ vpaddq $tmp,$D2hi,$D2hi++ vpaddq $T2,$H2,$H2 # accumulate input+ vpaddq $D2hi,$H0,$H0+ vpsllq \$2,$D2hi,$D2hi++ vpaddq $D2hi,$H0,$H0+ vporq $T3,$T1,$T1+ vpandq $mask44,$T1,$T1++ vpsrlq \$44,$H0,$tmp # additional step+ vpandq $mask44,$H0,$H0++ vpaddq $tmp,$H1,$H1++ sub \$8,$len # len-=128+ jnz .Loop_vpmadd52_8x++.Ltail_vpmadd52_8x:+ #vpaddq $T2,$H2,$H2 # accumulate input+ vpaddq $T0,$H0,$H0+ vpaddq $T1,$H1,$H1++ vpxorq $D0lo,$D0lo,$D0lo+ vpmadd52luq $H2,$SS1,$D0lo+ vpxorq $D0hi,$D0hi,$D0hi+ vpmadd52huq $H2,$SS1,$D0hi+ vpxorq $D1lo,$D1lo,$D1lo+ vpmadd52luq $H2,$SS2,$D1lo+ vpxorq $D1hi,$D1hi,$D1hi+ vpmadd52huq $H2,$SS2,$D1hi+ vpxorq $D2lo,$D2lo,$D2lo+ vpmadd52luq $H2,$RR0,$D2lo+ vpxorq $D2hi,$D2hi,$D2hi+ vpmadd52huq $H2,$RR0,$D2hi++ vpmadd52luq $H0,$RR0,$D0lo+ vpmadd52huq $H0,$RR0,$D0hi+ vpmadd52luq $H0,$RR1,$D1lo+ vpmadd52huq $H0,$RR1,$D1hi+ vpmadd52luq $H0,$RR2,$D2lo+ vpmadd52huq $H0,$RR2,$D2hi++ vpmadd52luq $H1,$SS2,$D0lo+ vpmadd52huq $H1,$SS2,$D0hi+ vpmadd52luq $H1,$RR0,$D1lo+ vpmadd52huq $H1,$RR0,$D1hi+ vpmadd52luq $H1,$RR1,$D2lo+ vpmadd52huq $H1,$RR1,$D2hi++ ################################################################+ # horizontal addition++ mov \$1,%eax+ kmovw %eax,%k1+ vpsrldq \$8,$D0lo,$T0+ vpsrldq \$8,$D0hi,$H0+ vpsrldq \$8,$D1lo,$T1+ vpsrldq \$8,$D1hi,$H1+ vpaddq $T0,$D0lo,$D0lo+ vpaddq $H0,$D0hi,$D0hi+ vpsrldq \$8,$D2lo,$T2+ vpsrldq \$8,$D2hi,$H2+ vpaddq $T1,$D1lo,$D1lo+ vpaddq $H1,$D1hi,$D1hi+ vpermq \$0x2,$D0lo,$T0+ vpermq \$0x2,$D0hi,$H0+ vpaddq $T2,$D2lo,$D2lo+ vpaddq $H2,$D2hi,$D2hi++ vpermq \$0x2,$D1lo,$T1+ vpermq \$0x2,$D1hi,$H1+ vpaddq $T0,$D0lo,$D0lo+ vpaddq $H0,$D0hi,$D0hi+ vpermq \$0x2,$D2lo,$T2+ vpermq \$0x2,$D2hi,$H2+ vpaddq $T1,$D1lo,$D1lo+ vpaddq $H1,$D1hi,$D1hi+ vextracti64x4 \$1,$D0lo,%y#$T0+ vextracti64x4 \$1,$D0hi,%y#$H0+ vpaddq $T2,$D2lo,$D2lo+ vpaddq $H2,$D2hi,$D2hi++ vextracti64x4 \$1,$D1lo,%y#$T1+ vextracti64x4 \$1,$D1hi,%y#$H1+ vextracti64x4 \$1,$D2lo,%y#$T2+ vextracti64x4 \$1,$D2hi,%y#$H2+___+######## switch back to %ymm+map(s/%z/%y/, $H0,$H1,$H2,$R0,$R1,$R2,$S1,$S2);+map(s/%z/%y/, $D0lo,$D0hi,$D1lo,$D1hi,$D2lo,$D2hi);+map(s/%z/%y/, $T0,$T1,$T2,$T3,$mask44,$mask42,$tmp,$PAD);++$code.=<<___;+ vpaddq $T0,$D0lo,${D0lo}{%k1}{z}+ vpaddq $H0,$D0hi,${D0hi}{%k1}{z}+ vpaddq $T1,$D1lo,${D1lo}{%k1}{z}+ vpaddq $H1,$D1hi,${D1hi}{%k1}{z}+ vpaddq $T2,$D2lo,${D2lo}{%k1}{z}+ vpaddq $H2,$D2hi,${D2hi}{%k1}{z}++ ################################################################+ # partial reduction+ vpsrlq \$44,$D0lo,$tmp+ vpsllq \$8,$D0hi,$D0hi+ vpandq $mask44,$D0lo,$H0+ vpaddq $tmp,$D0hi,$D0hi++ vpaddq $D0hi,$D1lo,$D1lo++ vpsrlq \$44,$D1lo,$tmp+ vpsllq \$8,$D1hi,$D1hi+ vpandq $mask44,$D1lo,$H1+ vpaddq $tmp,$D1hi,$D1hi++ vpaddq $D1hi,$D2lo,$D2lo++ vpsrlq \$42,$D2lo,$tmp+ vpsllq \$10,$D2hi,$D2hi+ vpandq $mask42,$D2lo,$H2+ vpaddq $tmp,$D2hi,$D2hi++ vpaddq $D2hi,$H0,$H0+ vpsllq \$2,$D2hi,$D2hi++ vpaddq $D2hi,$H0,$H0++ vpsrlq \$44,$H0,$tmp # additional step+ vpandq $mask44,$H0,$H0++ vpaddq $tmp,$H1,$H1++ ################################################################++ vmovq %x#$H0,0($ctx)+ vmovq %x#$H1,8($ctx)+ vmovq %x#$H2,16($ctx)+ vzeroall++.Lno_data_vpmadd52_8x:+ ret+.size poly1305_blocks_vpmadd52_8x,.-poly1305_blocks_vpmadd52_8x+___+}+$code.=<<___;+.type poly1305_emit_base2_44,\@function,3+.align 32+poly1305_emit_base2_44:+ mov 0($ctx),%r8 # load hash value+ mov 8($ctx),%r9+ mov 16($ctx),%r10++ mov %r9,%rax # base 2^44 -> base 2^64+ shr \$20,%r9+ shl \$44,%rax+ mov %r10,%rcx+ shr \$40,%r10+ shl \$24,%rcx++ add %rax,%r8+ adc %rcx,%r9+ adc \$0,%r10++ mov %r8,%rax+ add \$5,%r8 # compare to modulus+ mov %r9,%rcx+ adc \$0,%r9+ adc \$0,%r10+ shr \$2,%r10 # did 130-bit value overflow?+ cmovnz %r8,%rax+ cmovnz %r9,%rcx++ add 0($nonce),%rax # accumulate nonce+ adc 8($nonce),%rcx+ mov %rax,0($mac) # write result+ mov %rcx,8($mac)++ ret+.size poly1305_emit_base2_44,.-poly1305_emit_base2_44+___+} }+$code.=<<___;+.align 64+.Lconst:+.Lmask24:+.long 0x0ffffff,0,0x0ffffff,0,0x0ffffff,0,0x0ffffff,0+.L129:+.long `1<<24`,0,`1<<24`,0,`1<<24`,0,`1<<24`,0+.Lmask26:+.long 0x3ffffff,0,0x3ffffff,0,0x3ffffff,0,0x3ffffff,0+.Lpermd_avx2:+.long 2,2,2,3,2,0,2,1+.Lpermd_avx512:+.long 0,0,0,1, 0,2,0,3, 0,4,0,5, 0,6,0,7++.L2_44_inp_permd:+.long 0,1,1,2,2,3,7,7+.L2_44_inp_shift:+.quad 0,12,24,64+.L2_44_mask:+.quad 0xfffffffffff,0xfffffffffff,0x3ffffffffff,0xffffffffffffffff+.L2_44_shift_rgt:+.quad 44,44,42,64+.L2_44_shift_lft:+.quad 8,8,10,64++.align 64+.Lx_mask44:+.quad 0xfffffffffff,0xfffffffffff,0xfffffffffff,0xfffffffffff+.quad 0xfffffffffff,0xfffffffffff,0xfffffffffff,0xfffffffffff+.Lx_mask42:+.quad 0x3ffffffffff,0x3ffffffffff,0x3ffffffffff,0x3ffffffffff+.quad 0x3ffffffffff,0x3ffffffffff,0x3ffffffffff,0x3ffffffffff+___+}+$code.=<<___;+.asciz "Poly1305 for x86_64, CRYPTOGAMS by \@dot-asm"+.align 16+___++{ # chacha20-poly1305 helpers+my ($out,$inp,$otp,$len)=$win64 ? ("%rcx","%rdx","%r8", "%r9") : # Win64 order+ ("%rdi","%rsi","%rdx","%rcx"); # Unix order+$code.=<<___;+.globl xor128_encrypt_n_pad+.type xor128_encrypt_n_pad,\@abi-omnipotent+.align 16+xor128_encrypt_n_pad:+ sub $otp,$inp+ sub $otp,$out+ mov $len,%r10 # put len aside+ shr \$4,$len # len / 16+ jz .Ltail_enc+ nop+.Loop_enc_xmm:+ movdqu ($inp,$otp),%xmm0+ pxor ($otp),%xmm0+ movdqu %xmm0,($out,$otp)+ movdqa %xmm0,($otp)+ lea 16($otp),$otp+ dec $len+ jnz .Loop_enc_xmm++ and \$15,%r10 # len % 16+ jz .Ldone_enc++.Ltail_enc:+ mov \$16,$len+ sub %r10,$len+ xor %eax,%eax+.Loop_enc_byte:+ mov ($inp,$otp),%al+ xor ($otp),%al+ mov %al,($out,$otp)+ mov %al,($otp)+ lea 1($otp),$otp+ dec %r10+ jnz .Loop_enc_byte++ xor %eax,%eax+.Loop_enc_pad:+ mov %al,($otp)+ lea 1($otp),$otp+ dec $len+ jnz .Loop_enc_pad++.Ldone_enc:+ mov $otp,%rax+ ret+.size xor128_encrypt_n_pad,.-xor128_encrypt_n_pad++.globl xor128_decrypt_n_pad+.type xor128_decrypt_n_pad,\@abi-omnipotent+.align 16+xor128_decrypt_n_pad:+ sub $otp,$inp+ sub $otp,$out+ mov $len,%r10 # put len aside+ shr \$4,$len # len / 16+ jz .Ltail_dec+ nop+.Loop_dec_xmm:+ movdqu ($inp,$otp),%xmm0+ movdqa ($otp),%xmm1+ pxor %xmm0,%xmm1+ movdqu %xmm1,($out,$otp)+ movdqa %xmm0,($otp)+ lea 16($otp),$otp+ dec $len+ jnz .Loop_dec_xmm++ pxor %xmm1,%xmm1+ and \$15,%r10 # len % 16+ jz .Ldone_dec++.Ltail_dec:+ mov \$16,$len+ sub %r10,$len+ xor %eax,%eax+ xor %r11,%r11+.Loop_dec_byte:+ mov ($inp,$otp),%r11b+ mov ($otp),%al+ xor %r11b,%al+ mov %al,($out,$otp)+ mov %r11b,($otp)+ lea 1($otp),$otp+ dec %r10+ jnz .Loop_dec_byte++ xor %eax,%eax+.Loop_dec_pad:+ mov %al,($otp)+ lea 1($otp),$otp+ dec $len+ jnz .Loop_dec_pad++.Ldone_dec:+ mov $otp,%rax+ ret+.size xor128_decrypt_n_pad,.-xor128_decrypt_n_pad+___+}++# EXCEPTION_DISPOSITION handler (EXCEPTION_RECORD *rec,ULONG64 frame,+# CONTEXT *context,DISPATCHER_CONTEXT *disp)+if ($win64) {+$rec="%rcx";+$frame="%rdx";+$context="%r8";+$disp="%r9";++$code.=<<___;+.extern __imp_RtlVirtualUnwind+.type se_handler,\@abi-omnipotent+.align 16+se_handler:+ push %rsi+ push %rdi+ push %rbx+ push %rbp+ push %r12+ push %r13+ push %r14+ push %r15+ pushfq+ sub \$64,%rsp++ mov 120($context),%rax # pull context->Rax+ mov 248($context),%rbx # pull context->Rip++ mov 8($disp),%rsi # disp->ImageBase+ mov 56($disp),%r11 # disp->HandlerData++ mov 0(%r11),%r10d # HandlerData[0]+ lea (%rsi,%r10),%r10 # prologue label+ cmp %r10,%rbx # context->Rip<.Lprologue+ jb .Lcommon_seh_tail++ mov 152($context),%rax # pull context->Rsp++ mov 4(%r11),%r10d # HandlerData[1]+ lea (%rsi,%r10),%r10 # epilogue label+ cmp %r10,%rbx # context->Rip>=.Lepilogue+ jae .Lcommon_seh_tail++ lea 56(%rax),%rax++ mov -8(%rax),%rbx+ mov -16(%rax),%rbp+ mov -24(%rax),%r12+ mov -32(%rax),%r13+ mov -40(%rax),%r14+ mov -48(%rax),%r15+ mov %rbx,144($context) # restore context->Rbx+ mov %rbp,160($context) # restore context->Rbp+ mov %r12,216($context) # restore context->R12+ mov %r13,224($context) # restore context->R13+ mov %r14,232($context) # restore context->R14+ mov %r15,240($context) # restore context->R14++ jmp .Lcommon_seh_tail+.size se_handler,.-se_handler++.type avx_handler,\@abi-omnipotent+.align 16+avx_handler:+ push %rsi+ push %rdi+ push %rbx+ push %rbp+ push %r12+ push %r13+ push %r14+ push %r15+ pushfq+ sub \$64,%rsp++ mov 120($context),%rax # pull context->Rax+ mov 248($context),%rbx # pull context->Rip++ mov 8($disp),%rsi # disp->ImageBase+ mov 56($disp),%r11 # disp->HandlerData++ mov 0(%r11),%r10d # HandlerData[0]+ lea (%rsi,%r10),%r10 # prologue label+ cmp %r10,%rbx # context->Rip<prologue label+ jb .Lcommon_seh_tail++ mov 152($context),%rax # pull context->Rsp++ mov 4(%r11),%r10d # HandlerData[1]+ lea (%rsi,%r10),%r10 # epilogue label+ cmp %r10,%rbx # context->Rip>=epilogue label+ jae .Lcommon_seh_tail++ mov 208($context),%rax # pull context->R11++ lea 0x50(%rax),%rsi+ lea 0xf8(%rax),%rax+ lea 512($context),%rdi # &context.Xmm6+ mov \$20,%ecx+ .long 0xa548f3fc # cld; rep movsq++.Lcommon_seh_tail:+ mov 8(%rax),%rdi+ mov 16(%rax),%rsi+ mov %rax,152($context) # restore context->Rsp+ mov %rsi,168($context) # restore context->Rsi+ mov %rdi,176($context) # restore context->Rdi++ mov 40($disp),%rdi # disp->ContextRecord+ mov $context,%rsi # context+ mov \$154,%ecx # sizeof(CONTEXT)+ .long 0xa548f3fc # cld; rep movsq++ mov $disp,%rsi+ xor %rcx,%rcx # arg1, UNW_FLAG_NHANDLER+ mov 8(%rsi),%rdx # arg2, disp->ImageBase+ mov 0(%rsi),%r8 # arg3, disp->ControlPc+ mov 16(%rsi),%r9 # arg4, disp->FunctionEntry+ mov 40(%rsi),%r10 # disp->ContextRecord+ lea 56(%rsi),%r11 # &disp->HandlerData+ lea 24(%rsi),%r12 # &disp->EstablisherFrame+ mov %r10,32(%rsp) # arg5+ mov %r11,40(%rsp) # arg6+ mov %r12,48(%rsp) # arg7+ mov %rcx,56(%rsp) # arg8, (NULL)+ call *__imp_RtlVirtualUnwind(%rip)++ mov \$1,%eax # ExceptionContinueSearch+ add \$64,%rsp+ popfq+ pop %r15+ pop %r14+ pop %r13+ pop %r12+ pop %rbp+ pop %rbx+ pop %rdi+ pop %rsi+ ret+.size avx_handler,.-avx_handler++.section .pdata+.align 4+ .rva .LSEH_begin_poly1305_init+ .rva .LSEH_end_poly1305_init+ .rva .LSEH_info_poly1305_init++ .rva .LSEH_begin_poly1305_blocks+ .rva .LSEH_end_poly1305_blocks+ .rva .LSEH_info_poly1305_blocks++ .rva .LSEH_begin_poly1305_emit+ .rva .LSEH_end_poly1305_emit+ .rva .LSEH_info_poly1305_emit+___+$code.=<<___ if ($avx);+ .rva .LSEH_begin_poly1305_blocks_avx+ .rva .Lbase2_64_avx+ .rva .LSEH_info_poly1305_blocks_avx_1++ .rva .Lbase2_64_avx+ .rva .Leven_avx+ .rva .LSEH_info_poly1305_blocks_avx_2++ .rva .Leven_avx+ .rva .LSEH_end_poly1305_blocks_avx+ .rva .LSEH_info_poly1305_blocks_avx_3+___+$code.=<<___ if ($avx>1);+ .rva .LSEH_begin_poly1305_blocks_avx2+ .rva .Lbase2_64_avx2+ .rva .LSEH_info_poly1305_blocks_avx2_1++ .rva .Lbase2_64_avx2+ .rva .Leven_avx2+ .rva .LSEH_info_poly1305_blocks_avx2_2++ .rva .Leven_avx2+ .rva .LSEH_end_poly1305_blocks_avx2+ .rva .LSEH_info_poly1305_blocks_avx2_3+___+$code.=<<___ if ($avx>2);+ .rva .LSEH_begin_poly1305_blocks_avx512+ .rva .LSEH_end_poly1305_blocks_avx512+ .rva .LSEH_info_poly1305_blocks_avx512+___+$code.=<<___ if ($avx>3);+ .rva .LSEH_begin_poly1305_init_base2_44+ .rva .LSEH_end_poly1305_init_base2_44+ .rva .LSEH_info_poly1305_init_base2_44++ .rva .LSEH_begin_poly1305_blocks_base2_44+ .rva .LSEH_end_poly1305_blocks_base2_44+ .rva .LSEH_info_poly1305_blocks_base2_44++ .rva .LSEH_begin_poly1305_blocks_vpmadd52+ .rva .LSEH_end_poly1305_blocks_vpmadd52+ .rva .LSEH_info_poly1305_blocks_vpmadd52++ .rva .LSEH_begin_poly1305_blocks_vpmadd52_4x+ .rva .LSEH_end_poly1305_blocks_vpmadd52_4x+ .rva .LSEH_info_poly1305_blocks_vpmadd52_4x++ .rva .LSEH_begin_poly1305_emit_base2_44+ .rva .LSEH_end_poly1305_emit_base2_44+ .rva .LSEH_info_poly1305_emit_base2_44+___+$code.=<<___;+.section .xdata+.align 8+.LSEH_info_poly1305_init:+ .byte 9,0,0,0+ .rva se_handler+ .long 0,0 # 0,0 means "no stack frame allocated"++.LSEH_info_poly1305_blocks:+ .byte 9,0,0,0+ .rva se_handler+ .rva .Lblocks_body,.Lblocks_epilogue++.LSEH_info_poly1305_emit:+ .byte 9,0,0,0+ .rva se_handler+ .long 0,0+___+$code.=<<___ if ($avx);+.LSEH_info_poly1305_blocks_avx_1:+ .byte 9,0,0,0+ .rva se_handler+ .rva .Lblocks_avx_body,.Lblocks_avx_epilogue # HandlerData[]++.LSEH_info_poly1305_blocks_avx_2:+ .byte 9,0,0,0+ .rva se_handler+ .rva .Lbase2_64_avx_body,.Lbase2_64_avx_epilogue # HandlerData[]++.LSEH_info_poly1305_blocks_avx_3:+ .byte 9,0,0,0+ .rva avx_handler+ .rva .Ldo_avx_body,.Ldo_avx_epilogue # HandlerData[]+___+$code.=<<___ if ($avx>1);+.LSEH_info_poly1305_blocks_avx2_1:+ .byte 9,0,0,0+ .rva se_handler+ .rva .Lblocks_avx2_body,.Lblocks_avx2_epilogue # HandlerData[]++.LSEH_info_poly1305_blocks_avx2_2:+ .byte 9,0,0,0+ .rva se_handler+ .rva .Lbase2_64_avx2_body,.Lbase2_64_avx2_epilogue # HandlerData[]++.LSEH_info_poly1305_blocks_avx2_3:+ .byte 9,0,0,0+ .rva avx_handler+ .rva .Ldo_avx2_body,.Ldo_avx2_epilogue # HandlerData[]+___+$code.=<<___ if ($avx>2);+.LSEH_info_poly1305_blocks_avx512:+ .byte 9,0,0,0+ .rva avx_handler+ .rva .Ldo_avx512_body,.Ldo_avx512_epilogue # HandlerData[]+___+$code.=<<___ if ($avx>3);+.LSEH_info_poly1305_init_base2_44:+ .byte 9,0,0,0+ .rva se_handler+ .long 0,0++.LSEH_info_poly1305_blocks_base2_44:+ .byte 9,0,0,0+ .rva se_handler+ .rva .Lblocks_base2_44_body,.Lblocks_base2_44_epilogue++.LSEH_info_poly1305_blocks_vpmadd52:+ .byte 9,0,0,0+ .rva se_handler+ .long 0,0++.LSEH_info_poly1305_blocks_vpmadd52_4x:+ .byte 9,0,0,0+ .rva se_handler+ .long 0,0++.LSEH_info_poly1305_emit_base2_44:+ .byte 9,0,0,0+ .rva se_handler+ .long 0,0+___+}++foreach (split('\n',$code)) {+ s/\`([^\`]*)\`/eval($1)/ge;+ s/%r([a-z]+)#d/%e$1/g;+ s/%r([0-9]+)#d/%r$1d/g;+ s/%x#%[yz]/%x/g or s/%y#%z/%y/g or s/%z#%[yz]/%z/g;++ print $_,"\n";+}+close STDOUT;
@@ -0,0 +1,1216 @@+#ifndef __KERNEL__+# include "arm_arch.h"++#else+.globl _crypton_sha1_asm_block_armv8+#endif++.text++.globl _crypton_sha1_asm_block_data_order++.align 6+_crypton_sha1_asm_block_data_order:+ adrp x16,_crypton_armcap_P@PAGE+ ldr w16,[x16,_crypton_armcap_P@PAGEOFF]+ tst w16,#ARMV8_SHA1+ b.ne Lv8_entry++ stp x29,x30,[sp,#-12*__SIZEOF_POINTER__]!+ add x29,sp,#0+ stp x19,x20,[sp,#2*__SIZEOF_POINTER__]+ stp x21,x22,[sp,#4*__SIZEOF_POINTER__]+ stp x23,x24,[sp,#6*__SIZEOF_POINTER__]+ stp x25,x26,[sp,#8*__SIZEOF_POINTER__]+ stp x27,x28,[sp,#10*__SIZEOF_POINTER__]++ ldp w20,w21,[x0]+ ldp w22,w23,[x0,#8]+ ldr w24,[x0,#16]++Loop:+ ldr x3,[x1],#64+ movz w28,#0x7999+ sub x2,x2,#1+ movk w28,#0x5a82,lsl#16+#ifdef __AARCH64EB__+ ror x3,x3,#32+#else+ rev32 x3,x3+#endif+ add w24,w24,w28 // warm it up+ add w24,w24,w3+ lsr x4,x3,#32+ ldur x5,[x1,#-56]+ bic w25,w23,w21+ and w26,w22,w21+ ror w27,w20,#27+ add w23,w23,w28 // future e+=K+ orr w25,w25,w26+ add w24,w24,w27 // e+=rot(a,5)+ ror w21,w21,#2+ add w23,w23,w4 // future e+=X[i]+ add w24,w24,w25 // e+=F(b,c,d)+#ifdef __AARCH64EB__+ ror x5,x5,#32+#else+ rev32 x5,x5+#endif+ bic w25,w22,w20+ and w26,w21,w20+ ror w27,w24,#27+ add w22,w22,w28 // future e+=K+ orr w25,w25,w26+ add w23,w23,w27 // e+=rot(a,5)+ ror w20,w20,#2+ add w22,w22,w5 // future e+=X[i]+ add w23,w23,w25 // e+=F(b,c,d)+ lsr x6,x5,#32+ ldur x7,[x1,#-48]+ bic w25,w21,w24+ and w26,w20,w24+ ror w27,w23,#27+ add w21,w21,w28 // future e+=K+ orr w25,w25,w26+ add w22,w22,w27 // e+=rot(a,5)+ ror w24,w24,#2+ add w21,w21,w6 // future e+=X[i]+ add w22,w22,w25 // e+=F(b,c,d)+#ifdef __AARCH64EB__+ ror x7,x7,#32+#else+ rev32 x7,x7+#endif+ bic w25,w20,w23+ and w26,w24,w23+ ror w27,w22,#27+ add w20,w20,w28 // future e+=K+ orr w25,w25,w26+ add w21,w21,w27 // e+=rot(a,5)+ ror w23,w23,#2+ add w20,w20,w7 // future e+=X[i]+ add w21,w21,w25 // e+=F(b,c,d)+ lsr x8,x7,#32+ ldur x9,[x1,#-40]+ bic w25,w24,w22+ and w26,w23,w22+ ror w27,w21,#27+ add w24,w24,w28 // future e+=K+ orr w25,w25,w26+ add w20,w20,w27 // e+=rot(a,5)+ ror w22,w22,#2+ add w24,w24,w8 // future e+=X[i]+ add w20,w20,w25 // e+=F(b,c,d)+#ifdef __AARCH64EB__+ ror x9,x9,#32+#else+ rev32 x9,x9+#endif+ bic w25,w23,w21+ and w26,w22,w21+ ror w27,w20,#27+ add w23,w23,w28 // future e+=K+ orr w25,w25,w26+ add w24,w24,w27 // e+=rot(a,5)+ ror w21,w21,#2+ add w23,w23,w9 // future e+=X[i]+ add w24,w24,w25 // e+=F(b,c,d)+ lsr x10,x9,#32+ ldur x11,[x1,#-32]+ bic w25,w22,w20+ and w26,w21,w20+ ror w27,w24,#27+ add w22,w22,w28 // future e+=K+ orr w25,w25,w26+ add w23,w23,w27 // e+=rot(a,5)+ ror w20,w20,#2+ add w22,w22,w10 // future e+=X[i]+ add w23,w23,w25 // e+=F(b,c,d)+#ifdef __AARCH64EB__+ ror x11,x11,#32+#else+ rev32 x11,x11+#endif+ bic w25,w21,w24+ and w26,w20,w24+ ror w27,w23,#27+ add w21,w21,w28 // future e+=K+ orr w25,w25,w26+ add w22,w22,w27 // e+=rot(a,5)+ ror w24,w24,#2+ add w21,w21,w11 // future e+=X[i]+ add w22,w22,w25 // e+=F(b,c,d)+ lsr x12,x11,#32+ ldur x13,[x1,#-24]+ bic w25,w20,w23+ and w26,w24,w23+ ror w27,w22,#27+ add w20,w20,w28 // future e+=K+ orr w25,w25,w26+ add w21,w21,w27 // e+=rot(a,5)+ ror w23,w23,#2+ add w20,w20,w12 // future e+=X[i]+ add w21,w21,w25 // e+=F(b,c,d)+#ifdef __AARCH64EB__+ ror x13,x13,#32+#else+ rev32 x13,x13+#endif+ bic w25,w24,w22+ and w26,w23,w22+ ror w27,w21,#27+ add w24,w24,w28 // future e+=K+ orr w25,w25,w26+ add w20,w20,w27 // e+=rot(a,5)+ ror w22,w22,#2+ add w24,w24,w13 // future e+=X[i]+ add w20,w20,w25 // e+=F(b,c,d)+ lsr x14,x13,#32+ ldur x15,[x1,#-16]+ bic w25,w23,w21+ and w26,w22,w21+ ror w27,w20,#27+ add w23,w23,w28 // future e+=K+ orr w25,w25,w26+ add w24,w24,w27 // e+=rot(a,5)+ ror w21,w21,#2+ add w23,w23,w14 // future e+=X[i]+ add w24,w24,w25 // e+=F(b,c,d)+#ifdef __AARCH64EB__+ ror x15,x15,#32+#else+ rev32 x15,x15+#endif+ bic w25,w22,w20+ and w26,w21,w20+ ror w27,w24,#27+ add w22,w22,w28 // future e+=K+ orr w25,w25,w26+ add w23,w23,w27 // e+=rot(a,5)+ ror w20,w20,#2+ add w22,w22,w15 // future e+=X[i]+ add w23,w23,w25 // e+=F(b,c,d)+ lsr x16,x15,#32+ ldur x17,[x1,#-8]+ bic w25,w21,w24+ and w26,w20,w24+ ror w27,w23,#27+ add w21,w21,w28 // future e+=K+ orr w25,w25,w26+ add w22,w22,w27 // e+=rot(a,5)+ ror w24,w24,#2+ add w21,w21,w16 // future e+=X[i]+ add w22,w22,w25 // e+=F(b,c,d)+#ifdef __AARCH64EB__+ ror x17,x17,#32+#else+ rev32 x17,x17+#endif+ bic w25,w20,w23+ and w26,w24,w23+ ror w27,w22,#27+ add w20,w20,w28 // future e+=K+ orr w25,w25,w26+ add w21,w21,w27 // e+=rot(a,5)+ ror w23,w23,#2+ add w20,w20,w17 // future e+=X[i]+ add w21,w21,w25 // e+=F(b,c,d)+ lsr x19,x17,#32+ eor w3,w3,w5+ bic w25,w24,w22+ and w26,w23,w22+ ror w27,w21,#27+ eor w3,w3,w11+ add w24,w24,w28 // future e+=K+ orr w25,w25,w26+ add w20,w20,w27 // e+=rot(a,5)+ eor w3,w3,w16+ ror w22,w22,#2+ add w24,w24,w19 // future e+=X[i]+ add w20,w20,w25 // e+=F(b,c,d)+ ror w3,w3,#31+ eor w4,w4,w6+ bic w25,w23,w21+ and w26,w22,w21+ ror w27,w20,#27+ eor w4,w4,w12+ add w23,w23,w28 // future e+=K+ orr w25,w25,w26+ add w24,w24,w27 // e+=rot(a,5)+ eor w4,w4,w17+ ror w21,w21,#2+ add w23,w23,w3 // future e+=X[i]+ add w24,w24,w25 // e+=F(b,c,d)+ ror w4,w4,#31+ eor w5,w5,w7+ bic w25,w22,w20+ and w26,w21,w20+ ror w27,w24,#27+ eor w5,w5,w13+ add w22,w22,w28 // future e+=K+ orr w25,w25,w26+ add w23,w23,w27 // e+=rot(a,5)+ eor w5,w5,w19+ ror w20,w20,#2+ add w22,w22,w4 // future e+=X[i]+ add w23,w23,w25 // e+=F(b,c,d)+ ror w5,w5,#31+ eor w6,w6,w8+ bic w25,w21,w24+ and w26,w20,w24+ ror w27,w23,#27+ eor w6,w6,w14+ add w21,w21,w28 // future e+=K+ orr w25,w25,w26+ add w22,w22,w27 // e+=rot(a,5)+ eor w6,w6,w3+ ror w24,w24,#2+ add w21,w21,w5 // future e+=X[i]+ add w22,w22,w25 // e+=F(b,c,d)+ ror w6,w6,#31+ eor w7,w7,w9+ bic w25,w20,w23+ and w26,w24,w23+ ror w27,w22,#27+ eor w7,w7,w15+ add w20,w20,w28 // future e+=K+ orr w25,w25,w26+ add w21,w21,w27 // e+=rot(a,5)+ eor w7,w7,w4+ ror w23,w23,#2+ add w20,w20,w6 // future e+=X[i]+ add w21,w21,w25 // e+=F(b,c,d)+ ror w7,w7,#31+ movz w28,#0xeba1+ movk w28,#0x6ed9,lsl#16+ eor w8,w8,w10+ bic w25,w24,w22+ and w26,w23,w22+ ror w27,w21,#27+ eor w8,w8,w16+ add w24,w24,w28 // future e+=K+ orr w25,w25,w26+ add w20,w20,w27 // e+=rot(a,5)+ eor w8,w8,w5+ ror w22,w22,#2+ add w24,w24,w7 // future e+=X[i]+ add w20,w20,w25 // e+=F(b,c,d)+ ror w8,w8,#31+ eor w9,w9,w11+ eor w25,w23,w21+ ror w27,w20,#27+ add w23,w23,w28 // future e+=K+ eor w9,w9,w17+ eor w25,w25,w22+ add w24,w24,w27 // e+=rot(a,5)+ ror w21,w21,#2+ eor w9,w9,w6+ add w23,w23,w8 // future e+=X[i]+ add w24,w24,w25 // e+=F(b,c,d)+ ror w9,w9,#31+ eor w10,w10,w12+ eor w25,w22,w20+ ror w27,w24,#27+ add w22,w22,w28 // future e+=K+ eor w10,w10,w19+ eor w25,w25,w21+ add w23,w23,w27 // e+=rot(a,5)+ ror w20,w20,#2+ eor w10,w10,w7+ add w22,w22,w9 // future e+=X[i]+ add w23,w23,w25 // e+=F(b,c,d)+ ror w10,w10,#31+ eor w11,w11,w13+ eor w25,w21,w24+ ror w27,w23,#27+ add w21,w21,w28 // future e+=K+ eor w11,w11,w3+ eor w25,w25,w20+ add w22,w22,w27 // e+=rot(a,5)+ ror w24,w24,#2+ eor w11,w11,w8+ add w21,w21,w10 // future e+=X[i]+ add w22,w22,w25 // e+=F(b,c,d)+ ror w11,w11,#31+ eor w12,w12,w14+ eor w25,w20,w23+ ror w27,w22,#27+ add w20,w20,w28 // future e+=K+ eor w12,w12,w4+ eor w25,w25,w24+ add w21,w21,w27 // e+=rot(a,5)+ ror w23,w23,#2+ eor w12,w12,w9+ add w20,w20,w11 // future e+=X[i]+ add w21,w21,w25 // e+=F(b,c,d)+ ror w12,w12,#31+ eor w13,w13,w15+ eor w25,w24,w22+ ror w27,w21,#27+ add w24,w24,w28 // future e+=K+ eor w13,w13,w5+ eor w25,w25,w23+ add w20,w20,w27 // e+=rot(a,5)+ ror w22,w22,#2+ eor w13,w13,w10+ add w24,w24,w12 // future e+=X[i]+ add w20,w20,w25 // e+=F(b,c,d)+ ror w13,w13,#31+ eor w14,w14,w16+ eor w25,w23,w21+ ror w27,w20,#27+ add w23,w23,w28 // future e+=K+ eor w14,w14,w6+ eor w25,w25,w22+ add w24,w24,w27 // e+=rot(a,5)+ ror w21,w21,#2+ eor w14,w14,w11+ add w23,w23,w13 // future e+=X[i]+ add w24,w24,w25 // e+=F(b,c,d)+ ror w14,w14,#31+ eor w15,w15,w17+ eor w25,w22,w20+ ror w27,w24,#27+ add w22,w22,w28 // future e+=K+ eor w15,w15,w7+ eor w25,w25,w21+ add w23,w23,w27 // e+=rot(a,5)+ ror w20,w20,#2+ eor w15,w15,w12+ add w22,w22,w14 // future e+=X[i]+ add w23,w23,w25 // e+=F(b,c,d)+ ror w15,w15,#31+ eor w16,w16,w19+ eor w25,w21,w24+ ror w27,w23,#27+ add w21,w21,w28 // future e+=K+ eor w16,w16,w8+ eor w25,w25,w20+ add w22,w22,w27 // e+=rot(a,5)+ ror w24,w24,#2+ eor w16,w16,w13+ add w21,w21,w15 // future e+=X[i]+ add w22,w22,w25 // e+=F(b,c,d)+ ror w16,w16,#31+ eor w17,w17,w3+ eor w25,w20,w23+ ror w27,w22,#27+ add w20,w20,w28 // future e+=K+ eor w17,w17,w9+ eor w25,w25,w24+ add w21,w21,w27 // e+=rot(a,5)+ ror w23,w23,#2+ eor w17,w17,w14+ add w20,w20,w16 // future e+=X[i]+ add w21,w21,w25 // e+=F(b,c,d)+ ror w17,w17,#31+ eor w19,w19,w4+ eor w25,w24,w22+ ror w27,w21,#27+ add w24,w24,w28 // future e+=K+ eor w19,w19,w10+ eor w25,w25,w23+ add w20,w20,w27 // e+=rot(a,5)+ ror w22,w22,#2+ eor w19,w19,w15+ add w24,w24,w17 // future e+=X[i]+ add w20,w20,w25 // e+=F(b,c,d)+ ror w19,w19,#31+ eor w3,w3,w5+ eor w25,w23,w21+ ror w27,w20,#27+ add w23,w23,w28 // future e+=K+ eor w3,w3,w11+ eor w25,w25,w22+ add w24,w24,w27 // e+=rot(a,5)+ ror w21,w21,#2+ eor w3,w3,w16+ add w23,w23,w19 // future e+=X[i]+ add w24,w24,w25 // e+=F(b,c,d)+ ror w3,w3,#31+ eor w4,w4,w6+ eor w25,w22,w20+ ror w27,w24,#27+ add w22,w22,w28 // future e+=K+ eor w4,w4,w12+ eor w25,w25,w21+ add w23,w23,w27 // e+=rot(a,5)+ ror w20,w20,#2+ eor w4,w4,w17+ add w22,w22,w3 // future e+=X[i]+ add w23,w23,w25 // e+=F(b,c,d)+ ror w4,w4,#31+ eor w5,w5,w7+ eor w25,w21,w24+ ror w27,w23,#27+ add w21,w21,w28 // future e+=K+ eor w5,w5,w13+ eor w25,w25,w20+ add w22,w22,w27 // e+=rot(a,5)+ ror w24,w24,#2+ eor w5,w5,w19+ add w21,w21,w4 // future e+=X[i]+ add w22,w22,w25 // e+=F(b,c,d)+ ror w5,w5,#31+ eor w6,w6,w8+ eor w25,w20,w23+ ror w27,w22,#27+ add w20,w20,w28 // future e+=K+ eor w6,w6,w14+ eor w25,w25,w24+ add w21,w21,w27 // e+=rot(a,5)+ ror w23,w23,#2+ eor w6,w6,w3+ add w20,w20,w5 // future e+=X[i]+ add w21,w21,w25 // e+=F(b,c,d)+ ror w6,w6,#31+ eor w7,w7,w9+ eor w25,w24,w22+ ror w27,w21,#27+ add w24,w24,w28 // future e+=K+ eor w7,w7,w15+ eor w25,w25,w23+ add w20,w20,w27 // e+=rot(a,5)+ ror w22,w22,#2+ eor w7,w7,w4+ add w24,w24,w6 // future e+=X[i]+ add w20,w20,w25 // e+=F(b,c,d)+ ror w7,w7,#31+ eor w8,w8,w10+ eor w25,w23,w21+ ror w27,w20,#27+ add w23,w23,w28 // future e+=K+ eor w8,w8,w16+ eor w25,w25,w22+ add w24,w24,w27 // e+=rot(a,5)+ ror w21,w21,#2+ eor w8,w8,w5+ add w23,w23,w7 // future e+=X[i]+ add w24,w24,w25 // e+=F(b,c,d)+ ror w8,w8,#31+ eor w9,w9,w11+ eor w25,w22,w20+ ror w27,w24,#27+ add w22,w22,w28 // future e+=K+ eor w9,w9,w17+ eor w25,w25,w21+ add w23,w23,w27 // e+=rot(a,5)+ ror w20,w20,#2+ eor w9,w9,w6+ add w22,w22,w8 // future e+=X[i]+ add w23,w23,w25 // e+=F(b,c,d)+ ror w9,w9,#31+ eor w10,w10,w12+ eor w25,w21,w24+ ror w27,w23,#27+ add w21,w21,w28 // future e+=K+ eor w10,w10,w19+ eor w25,w25,w20+ add w22,w22,w27 // e+=rot(a,5)+ ror w24,w24,#2+ eor w10,w10,w7+ add w21,w21,w9 // future e+=X[i]+ add w22,w22,w25 // e+=F(b,c,d)+ ror w10,w10,#31+ eor w11,w11,w13+ eor w25,w20,w23+ ror w27,w22,#27+ add w20,w20,w28 // future e+=K+ eor w11,w11,w3+ eor w25,w25,w24+ add w21,w21,w27 // e+=rot(a,5)+ ror w23,w23,#2+ eor w11,w11,w8+ add w20,w20,w10 // future e+=X[i]+ add w21,w21,w25 // e+=F(b,c,d)+ ror w11,w11,#31+ movz w28,#0xbcdc+ movk w28,#0x8f1b,lsl#16+ eor w12,w12,w14+ eor w25,w24,w22+ ror w27,w21,#27+ add w24,w24,w28 // future e+=K+ eor w12,w12,w4+ eor w25,w25,w23+ add w20,w20,w27 // e+=rot(a,5)+ ror w22,w22,#2+ eor w12,w12,w9+ add w24,w24,w11 // future e+=X[i]+ add w20,w20,w25 // e+=F(b,c,d)+ ror w12,w12,#31+ orr w25,w21,w22+ and w26,w21,w22+ eor w13,w13,w15+ ror w27,w20,#27+ and w25,w25,w23+ add w23,w23,w28 // future e+=K+ eor w13,w13,w5+ add w24,w24,w27 // e+=rot(a,5)+ orr w25,w25,w26+ ror w21,w21,#2+ eor w13,w13,w10+ add w23,w23,w12 // future e+=X[i]+ add w24,w24,w25 // e+=F(b,c,d)+ ror w13,w13,#31+ orr w25,w20,w21+ and w26,w20,w21+ eor w14,w14,w16+ ror w27,w24,#27+ and w25,w25,w22+ add w22,w22,w28 // future e+=K+ eor w14,w14,w6+ add w23,w23,w27 // e+=rot(a,5)+ orr w25,w25,w26+ ror w20,w20,#2+ eor w14,w14,w11+ add w22,w22,w13 // future e+=X[i]+ add w23,w23,w25 // e+=F(b,c,d)+ ror w14,w14,#31+ orr w25,w24,w20+ and w26,w24,w20+ eor w15,w15,w17+ ror w27,w23,#27+ and w25,w25,w21+ add w21,w21,w28 // future e+=K+ eor w15,w15,w7+ add w22,w22,w27 // e+=rot(a,5)+ orr w25,w25,w26+ ror w24,w24,#2+ eor w15,w15,w12+ add w21,w21,w14 // future e+=X[i]+ add w22,w22,w25 // e+=F(b,c,d)+ ror w15,w15,#31+ orr w25,w23,w24+ and w26,w23,w24+ eor w16,w16,w19+ ror w27,w22,#27+ and w25,w25,w20+ add w20,w20,w28 // future e+=K+ eor w16,w16,w8+ add w21,w21,w27 // e+=rot(a,5)+ orr w25,w25,w26+ ror w23,w23,#2+ eor w16,w16,w13+ add w20,w20,w15 // future e+=X[i]+ add w21,w21,w25 // e+=F(b,c,d)+ ror w16,w16,#31+ orr w25,w22,w23+ and w26,w22,w23+ eor w17,w17,w3+ ror w27,w21,#27+ and w25,w25,w24+ add w24,w24,w28 // future e+=K+ eor w17,w17,w9+ add w20,w20,w27 // e+=rot(a,5)+ orr w25,w25,w26+ ror w22,w22,#2+ eor w17,w17,w14+ add w24,w24,w16 // future e+=X[i]+ add w20,w20,w25 // e+=F(b,c,d)+ ror w17,w17,#31+ orr w25,w21,w22+ and w26,w21,w22+ eor w19,w19,w4+ ror w27,w20,#27+ and w25,w25,w23+ add w23,w23,w28 // future e+=K+ eor w19,w19,w10+ add w24,w24,w27 // e+=rot(a,5)+ orr w25,w25,w26+ ror w21,w21,#2+ eor w19,w19,w15+ add w23,w23,w17 // future e+=X[i]+ add w24,w24,w25 // e+=F(b,c,d)+ ror w19,w19,#31+ orr w25,w20,w21+ and w26,w20,w21+ eor w3,w3,w5+ ror w27,w24,#27+ and w25,w25,w22+ add w22,w22,w28 // future e+=K+ eor w3,w3,w11+ add w23,w23,w27 // e+=rot(a,5)+ orr w25,w25,w26+ ror w20,w20,#2+ eor w3,w3,w16+ add w22,w22,w19 // future e+=X[i]+ add w23,w23,w25 // e+=F(b,c,d)+ ror w3,w3,#31+ orr w25,w24,w20+ and w26,w24,w20+ eor w4,w4,w6+ ror w27,w23,#27+ and w25,w25,w21+ add w21,w21,w28 // future e+=K+ eor w4,w4,w12+ add w22,w22,w27 // e+=rot(a,5)+ orr w25,w25,w26+ ror w24,w24,#2+ eor w4,w4,w17+ add w21,w21,w3 // future e+=X[i]+ add w22,w22,w25 // e+=F(b,c,d)+ ror w4,w4,#31+ orr w25,w23,w24+ and w26,w23,w24+ eor w5,w5,w7+ ror w27,w22,#27+ and w25,w25,w20+ add w20,w20,w28 // future e+=K+ eor w5,w5,w13+ add w21,w21,w27 // e+=rot(a,5)+ orr w25,w25,w26+ ror w23,w23,#2+ eor w5,w5,w19+ add w20,w20,w4 // future e+=X[i]+ add w21,w21,w25 // e+=F(b,c,d)+ ror w5,w5,#31+ orr w25,w22,w23+ and w26,w22,w23+ eor w6,w6,w8+ ror w27,w21,#27+ and w25,w25,w24+ add w24,w24,w28 // future e+=K+ eor w6,w6,w14+ add w20,w20,w27 // e+=rot(a,5)+ orr w25,w25,w26+ ror w22,w22,#2+ eor w6,w6,w3+ add w24,w24,w5 // future e+=X[i]+ add w20,w20,w25 // e+=F(b,c,d)+ ror w6,w6,#31+ orr w25,w21,w22+ and w26,w21,w22+ eor w7,w7,w9+ ror w27,w20,#27+ and w25,w25,w23+ add w23,w23,w28 // future e+=K+ eor w7,w7,w15+ add w24,w24,w27 // e+=rot(a,5)+ orr w25,w25,w26+ ror w21,w21,#2+ eor w7,w7,w4+ add w23,w23,w6 // future e+=X[i]+ add w24,w24,w25 // e+=F(b,c,d)+ ror w7,w7,#31+ orr w25,w20,w21+ and w26,w20,w21+ eor w8,w8,w10+ ror w27,w24,#27+ and w25,w25,w22+ add w22,w22,w28 // future e+=K+ eor w8,w8,w16+ add w23,w23,w27 // e+=rot(a,5)+ orr w25,w25,w26+ ror w20,w20,#2+ eor w8,w8,w5+ add w22,w22,w7 // future e+=X[i]+ add w23,w23,w25 // e+=F(b,c,d)+ ror w8,w8,#31+ orr w25,w24,w20+ and w26,w24,w20+ eor w9,w9,w11+ ror w27,w23,#27+ and w25,w25,w21+ add w21,w21,w28 // future e+=K+ eor w9,w9,w17+ add w22,w22,w27 // e+=rot(a,5)+ orr w25,w25,w26+ ror w24,w24,#2+ eor w9,w9,w6+ add w21,w21,w8 // future e+=X[i]+ add w22,w22,w25 // e+=F(b,c,d)+ ror w9,w9,#31+ orr w25,w23,w24+ and w26,w23,w24+ eor w10,w10,w12+ ror w27,w22,#27+ and w25,w25,w20+ add w20,w20,w28 // future e+=K+ eor w10,w10,w19+ add w21,w21,w27 // e+=rot(a,5)+ orr w25,w25,w26+ ror w23,w23,#2+ eor w10,w10,w7+ add w20,w20,w9 // future e+=X[i]+ add w21,w21,w25 // e+=F(b,c,d)+ ror w10,w10,#31+ orr w25,w22,w23+ and w26,w22,w23+ eor w11,w11,w13+ ror w27,w21,#27+ and w25,w25,w24+ add w24,w24,w28 // future e+=K+ eor w11,w11,w3+ add w20,w20,w27 // e+=rot(a,5)+ orr w25,w25,w26+ ror w22,w22,#2+ eor w11,w11,w8+ add w24,w24,w10 // future e+=X[i]+ add w20,w20,w25 // e+=F(b,c,d)+ ror w11,w11,#31+ orr w25,w21,w22+ and w26,w21,w22+ eor w12,w12,w14+ ror w27,w20,#27+ and w25,w25,w23+ add w23,w23,w28 // future e+=K+ eor w12,w12,w4+ add w24,w24,w27 // e+=rot(a,5)+ orr w25,w25,w26+ ror w21,w21,#2+ eor w12,w12,w9+ add w23,w23,w11 // future e+=X[i]+ add w24,w24,w25 // e+=F(b,c,d)+ ror w12,w12,#31+ orr w25,w20,w21+ and w26,w20,w21+ eor w13,w13,w15+ ror w27,w24,#27+ and w25,w25,w22+ add w22,w22,w28 // future e+=K+ eor w13,w13,w5+ add w23,w23,w27 // e+=rot(a,5)+ orr w25,w25,w26+ ror w20,w20,#2+ eor w13,w13,w10+ add w22,w22,w12 // future e+=X[i]+ add w23,w23,w25 // e+=F(b,c,d)+ ror w13,w13,#31+ orr w25,w24,w20+ and w26,w24,w20+ eor w14,w14,w16+ ror w27,w23,#27+ and w25,w25,w21+ add w21,w21,w28 // future e+=K+ eor w14,w14,w6+ add w22,w22,w27 // e+=rot(a,5)+ orr w25,w25,w26+ ror w24,w24,#2+ eor w14,w14,w11+ add w21,w21,w13 // future e+=X[i]+ add w22,w22,w25 // e+=F(b,c,d)+ ror w14,w14,#31+ orr w25,w23,w24+ and w26,w23,w24+ eor w15,w15,w17+ ror w27,w22,#27+ and w25,w25,w20+ add w20,w20,w28 // future e+=K+ eor w15,w15,w7+ add w21,w21,w27 // e+=rot(a,5)+ orr w25,w25,w26+ ror w23,w23,#2+ eor w15,w15,w12+ add w20,w20,w14 // future e+=X[i]+ add w21,w21,w25 // e+=F(b,c,d)+ ror w15,w15,#31+ movz w28,#0xc1d6+ movk w28,#0xca62,lsl#16+ orr w25,w22,w23+ and w26,w22,w23+ eor w16,w16,w19+ ror w27,w21,#27+ and w25,w25,w24+ add w24,w24,w28 // future e+=K+ eor w16,w16,w8+ add w20,w20,w27 // e+=rot(a,5)+ orr w25,w25,w26+ ror w22,w22,#2+ eor w16,w16,w13+ add w24,w24,w15 // future e+=X[i]+ add w20,w20,w25 // e+=F(b,c,d)+ ror w16,w16,#31+ eor w17,w17,w3+ eor w25,w23,w21+ ror w27,w20,#27+ add w23,w23,w28 // future e+=K+ eor w17,w17,w9+ eor w25,w25,w22+ add w24,w24,w27 // e+=rot(a,5)+ ror w21,w21,#2+ eor w17,w17,w14+ add w23,w23,w16 // future e+=X[i]+ add w24,w24,w25 // e+=F(b,c,d)+ ror w17,w17,#31+ eor w19,w19,w4+ eor w25,w22,w20+ ror w27,w24,#27+ add w22,w22,w28 // future e+=K+ eor w19,w19,w10+ eor w25,w25,w21+ add w23,w23,w27 // e+=rot(a,5)+ ror w20,w20,#2+ eor w19,w19,w15+ add w22,w22,w17 // future e+=X[i]+ add w23,w23,w25 // e+=F(b,c,d)+ ror w19,w19,#31+ eor w3,w3,w5+ eor w25,w21,w24+ ror w27,w23,#27+ add w21,w21,w28 // future e+=K+ eor w3,w3,w11+ eor w25,w25,w20+ add w22,w22,w27 // e+=rot(a,5)+ ror w24,w24,#2+ eor w3,w3,w16+ add w21,w21,w19 // future e+=X[i]+ add w22,w22,w25 // e+=F(b,c,d)+ ror w3,w3,#31+ eor w4,w4,w6+ eor w25,w20,w23+ ror w27,w22,#27+ add w20,w20,w28 // future e+=K+ eor w4,w4,w12+ eor w25,w25,w24+ add w21,w21,w27 // e+=rot(a,5)+ ror w23,w23,#2+ eor w4,w4,w17+ add w20,w20,w3 // future e+=X[i]+ add w21,w21,w25 // e+=F(b,c,d)+ ror w4,w4,#31+ eor w5,w5,w7+ eor w25,w24,w22+ ror w27,w21,#27+ add w24,w24,w28 // future e+=K+ eor w5,w5,w13+ eor w25,w25,w23+ add w20,w20,w27 // e+=rot(a,5)+ ror w22,w22,#2+ eor w5,w5,w19+ add w24,w24,w4 // future e+=X[i]+ add w20,w20,w25 // e+=F(b,c,d)+ ror w5,w5,#31+ eor w6,w6,w8+ eor w25,w23,w21+ ror w27,w20,#27+ add w23,w23,w28 // future e+=K+ eor w6,w6,w14+ eor w25,w25,w22+ add w24,w24,w27 // e+=rot(a,5)+ ror w21,w21,#2+ eor w6,w6,w3+ add w23,w23,w5 // future e+=X[i]+ add w24,w24,w25 // e+=F(b,c,d)+ ror w6,w6,#31+ eor w7,w7,w9+ eor w25,w22,w20+ ror w27,w24,#27+ add w22,w22,w28 // future e+=K+ eor w7,w7,w15+ eor w25,w25,w21+ add w23,w23,w27 // e+=rot(a,5)+ ror w20,w20,#2+ eor w7,w7,w4+ add w22,w22,w6 // future e+=X[i]+ add w23,w23,w25 // e+=F(b,c,d)+ ror w7,w7,#31+ eor w8,w8,w10+ eor w25,w21,w24+ ror w27,w23,#27+ add w21,w21,w28 // future e+=K+ eor w8,w8,w16+ eor w25,w25,w20+ add w22,w22,w27 // e+=rot(a,5)+ ror w24,w24,#2+ eor w8,w8,w5+ add w21,w21,w7 // future e+=X[i]+ add w22,w22,w25 // e+=F(b,c,d)+ ror w8,w8,#31+ eor w9,w9,w11+ eor w25,w20,w23+ ror w27,w22,#27+ add w20,w20,w28 // future e+=K+ eor w9,w9,w17+ eor w25,w25,w24+ add w21,w21,w27 // e+=rot(a,5)+ ror w23,w23,#2+ eor w9,w9,w6+ add w20,w20,w8 // future e+=X[i]+ add w21,w21,w25 // e+=F(b,c,d)+ ror w9,w9,#31+ eor w10,w10,w12+ eor w25,w24,w22+ ror w27,w21,#27+ add w24,w24,w28 // future e+=K+ eor w10,w10,w19+ eor w25,w25,w23+ add w20,w20,w27 // e+=rot(a,5)+ ror w22,w22,#2+ eor w10,w10,w7+ add w24,w24,w9 // future e+=X[i]+ add w20,w20,w25 // e+=F(b,c,d)+ ror w10,w10,#31+ eor w11,w11,w13+ eor w25,w23,w21+ ror w27,w20,#27+ add w23,w23,w28 // future e+=K+ eor w11,w11,w3+ eor w25,w25,w22+ add w24,w24,w27 // e+=rot(a,5)+ ror w21,w21,#2+ eor w11,w11,w8+ add w23,w23,w10 // future e+=X[i]+ add w24,w24,w25 // e+=F(b,c,d)+ ror w11,w11,#31+ eor w12,w12,w14+ eor w25,w22,w20+ ror w27,w24,#27+ add w22,w22,w28 // future e+=K+ eor w12,w12,w4+ eor w25,w25,w21+ add w23,w23,w27 // e+=rot(a,5)+ ror w20,w20,#2+ eor w12,w12,w9+ add w22,w22,w11 // future e+=X[i]+ add w23,w23,w25 // e+=F(b,c,d)+ ror w12,w12,#31+ eor w13,w13,w15+ eor w25,w21,w24+ ror w27,w23,#27+ add w21,w21,w28 // future e+=K+ eor w13,w13,w5+ eor w25,w25,w20+ add w22,w22,w27 // e+=rot(a,5)+ ror w24,w24,#2+ eor w13,w13,w10+ add w21,w21,w12 // future e+=X[i]+ add w22,w22,w25 // e+=F(b,c,d)+ ror w13,w13,#31+ eor w14,w14,w16+ eor w25,w20,w23+ ror w27,w22,#27+ add w20,w20,w28 // future e+=K+ eor w14,w14,w6+ eor w25,w25,w24+ add w21,w21,w27 // e+=rot(a,5)+ ror w23,w23,#2+ eor w14,w14,w11+ add w20,w20,w13 // future e+=X[i]+ add w21,w21,w25 // e+=F(b,c,d)+ ror w14,w14,#31+ eor w15,w15,w17+ eor w25,w24,w22+ ror w27,w21,#27+ add w24,w24,w28 // future e+=K+ eor w15,w15,w7+ eor w25,w25,w23+ add w20,w20,w27 // e+=rot(a,5)+ ror w22,w22,#2+ eor w15,w15,w12+ add w24,w24,w14 // future e+=X[i]+ add w20,w20,w25 // e+=F(b,c,d)+ ror w15,w15,#31+ eor w16,w16,w19+ eor w25,w23,w21+ ror w27,w20,#27+ add w23,w23,w28 // future e+=K+ eor w16,w16,w8+ eor w25,w25,w22+ add w24,w24,w27 // e+=rot(a,5)+ ror w21,w21,#2+ eor w16,w16,w13+ add w23,w23,w15 // future e+=X[i]+ add w24,w24,w25 // e+=F(b,c,d)+ ror w16,w16,#31+ eor w17,w17,w3+ eor w25,w22,w20+ ror w27,w24,#27+ add w22,w22,w28 // future e+=K+ eor w17,w17,w9+ eor w25,w25,w21+ add w23,w23,w27 // e+=rot(a,5)+ ror w20,w20,#2+ eor w17,w17,w14+ add w22,w22,w16 // future e+=X[i]+ add w23,w23,w25 // e+=F(b,c,d)+ ror w17,w17,#31+ eor w19,w19,w4+ eor w25,w21,w24+ ror w27,w23,#27+ add w21,w21,w28 // future e+=K+ eor w19,w19,w10+ eor w25,w25,w20+ add w22,w22,w27 // e+=rot(a,5)+ ror w24,w24,#2+ eor w19,w19,w15+ add w21,w21,w17 // future e+=X[i]+ add w22,w22,w25 // e+=F(b,c,d)+ ror w19,w19,#31+ ldp w4,w5,[x0]+ eor w25,w20,w23+ ror w27,w22,#27+ add w20,w20,w28 // future e+=K+ eor w25,w25,w24+ add w21,w21,w27 // e+=rot(a,5)+ ror w23,w23,#2+ add w20,w20,w19 // future e+=X[i]+ add w21,w21,w25 // e+=F(b,c,d)+ ldp w6,w7,[x0,#8]+ eor w25,w24,w22+ ror w27,w21,#27+ eor w25,w25,w23+ add w20,w20,w27 // e+=rot(a,5)+ ror w22,w22,#2+ ldr w8,[x0,#16]+ add w20,w20,w25 // e+=F(b,c,d)+ add w21,w21,w5+ add w22,w22,w6+ add w20,w20,w4+ add w23,w23,w7+ add w24,w24,w8+ stp w20,w21,[x0]+ stp w22,w23,[x0,#8]+ str w24,[x0,#16]+ cbnz x2,Loop++ ldp x19,x20,[sp,#2*__SIZEOF_POINTER__]+ ldp x21,x22,[sp,#4*__SIZEOF_POINTER__]+ ldp x23,x24,[sp,#6*__SIZEOF_POINTER__]+ ldp x25,x26,[sp,#8*__SIZEOF_POINTER__]+ ldp x27,x28,[sp,#10*__SIZEOF_POINTER__]+ ldr x29,[sp],#12*__SIZEOF_POINTER__+ ret+++.align 6+_crypton_sha1_asm_block_armv8:+Lv8_entry:+ stp x29,x30,[sp,#-16]!+ add x29,sp,#0++ adr x4,Lconst+ eor v1.16b,v1.16b,v1.16b+ ld1 {v0.4s},[x0],#16+ ld1 {v1.s}[0],[x0]+ sub x0,x0,#16+ ld1 {v16.4s,v17.4s,v18.4s,v19.4s},[x4]++Loop_hw:+ ld1 {v4.16b,v5.16b,v6.16b,v7.16b},[x1],#64+ sub x2,x2,#1+ rev32 v4.16b,v4.16b+ rev32 v5.16b,v5.16b++ add v20.4s,v16.4s,v4.4s+ rev32 v6.16b,v6.16b+ orr v22.16b,v0.16b,v0.16b // offload++ add v21.4s,v16.4s,v5.4s+ rev32 v7.16b,v7.16b+.long 0x5e280803 //sha1h v3.16b,v0.16b+.long 0x5e140020 //sha1c v0.16b,v1.16b,v20.4s // 0+ add v20.4s,v16.4s,v6.4s+.long 0x5e0630a4 //sha1su0 v4.16b,v5.16b,v6.16b+.long 0x5e280802 //sha1h v2.16b,v0.16b // 1+.long 0x5e150060 //sha1c v0.16b,v3.16b,v21.4s+ add v21.4s,v16.4s,v7.4s+.long 0x5e2818e4 //sha1su1 v4.16b,v7.16b+.long 0x5e0730c5 //sha1su0 v5.16b,v6.16b,v7.16b+.long 0x5e280803 //sha1h v3.16b,v0.16b // 2+.long 0x5e140040 //sha1c v0.16b,v2.16b,v20.4s+ add v20.4s,v16.4s,v4.4s+.long 0x5e281885 //sha1su1 v5.16b,v4.16b+.long 0x5e0430e6 //sha1su0 v6.16b,v7.16b,v4.16b+.long 0x5e280802 //sha1h v2.16b,v0.16b // 3+.long 0x5e150060 //sha1c v0.16b,v3.16b,v21.4s+ add v21.4s,v17.4s,v5.4s+.long 0x5e2818a6 //sha1su1 v6.16b,v5.16b+.long 0x5e053087 //sha1su0 v7.16b,v4.16b,v5.16b+.long 0x5e280803 //sha1h v3.16b,v0.16b // 4+.long 0x5e140040 //sha1c v0.16b,v2.16b,v20.4s+ add v20.4s,v17.4s,v6.4s+.long 0x5e2818c7 //sha1su1 v7.16b,v6.16b+.long 0x5e0630a4 //sha1su0 v4.16b,v5.16b,v6.16b+.long 0x5e280802 //sha1h v2.16b,v0.16b // 5+.long 0x5e151060 //sha1p v0.16b,v3.16b,v21.4s+ add v21.4s,v17.4s,v7.4s+.long 0x5e2818e4 //sha1su1 v4.16b,v7.16b+.long 0x5e0730c5 //sha1su0 v5.16b,v6.16b,v7.16b+.long 0x5e280803 //sha1h v3.16b,v0.16b // 6+.long 0x5e141040 //sha1p v0.16b,v2.16b,v20.4s+ add v20.4s,v17.4s,v4.4s+.long 0x5e281885 //sha1su1 v5.16b,v4.16b+.long 0x5e0430e6 //sha1su0 v6.16b,v7.16b,v4.16b+.long 0x5e280802 //sha1h v2.16b,v0.16b // 7+.long 0x5e151060 //sha1p v0.16b,v3.16b,v21.4s+ add v21.4s,v17.4s,v5.4s+.long 0x5e2818a6 //sha1su1 v6.16b,v5.16b+.long 0x5e053087 //sha1su0 v7.16b,v4.16b,v5.16b+.long 0x5e280803 //sha1h v3.16b,v0.16b // 8+.long 0x5e141040 //sha1p v0.16b,v2.16b,v20.4s+ add v20.4s,v18.4s,v6.4s+.long 0x5e2818c7 //sha1su1 v7.16b,v6.16b+.long 0x5e0630a4 //sha1su0 v4.16b,v5.16b,v6.16b+.long 0x5e280802 //sha1h v2.16b,v0.16b // 9+.long 0x5e151060 //sha1p v0.16b,v3.16b,v21.4s+ add v21.4s,v18.4s,v7.4s+.long 0x5e2818e4 //sha1su1 v4.16b,v7.16b+.long 0x5e0730c5 //sha1su0 v5.16b,v6.16b,v7.16b+.long 0x5e280803 //sha1h v3.16b,v0.16b // 10+.long 0x5e142040 //sha1m v0.16b,v2.16b,v20.4s+ add v20.4s,v18.4s,v4.4s+.long 0x5e281885 //sha1su1 v5.16b,v4.16b+.long 0x5e0430e6 //sha1su0 v6.16b,v7.16b,v4.16b+.long 0x5e280802 //sha1h v2.16b,v0.16b // 11+.long 0x5e152060 //sha1m v0.16b,v3.16b,v21.4s+ add v21.4s,v18.4s,v5.4s+.long 0x5e2818a6 //sha1su1 v6.16b,v5.16b+.long 0x5e053087 //sha1su0 v7.16b,v4.16b,v5.16b+.long 0x5e280803 //sha1h v3.16b,v0.16b // 12+.long 0x5e142040 //sha1m v0.16b,v2.16b,v20.4s+ add v20.4s,v18.4s,v6.4s+.long 0x5e2818c7 //sha1su1 v7.16b,v6.16b+.long 0x5e0630a4 //sha1su0 v4.16b,v5.16b,v6.16b+.long 0x5e280802 //sha1h v2.16b,v0.16b // 13+.long 0x5e152060 //sha1m v0.16b,v3.16b,v21.4s+ add v21.4s,v19.4s,v7.4s+.long 0x5e2818e4 //sha1su1 v4.16b,v7.16b+.long 0x5e0730c5 //sha1su0 v5.16b,v6.16b,v7.16b+.long 0x5e280803 //sha1h v3.16b,v0.16b // 14+.long 0x5e142040 //sha1m v0.16b,v2.16b,v20.4s+ add v20.4s,v19.4s,v4.4s+.long 0x5e281885 //sha1su1 v5.16b,v4.16b+.long 0x5e0430e6 //sha1su0 v6.16b,v7.16b,v4.16b+.long 0x5e280802 //sha1h v2.16b,v0.16b // 15+.long 0x5e151060 //sha1p v0.16b,v3.16b,v21.4s+ add v21.4s,v19.4s,v5.4s+.long 0x5e2818a6 //sha1su1 v6.16b,v5.16b+.long 0x5e053087 //sha1su0 v7.16b,v4.16b,v5.16b+.long 0x5e280803 //sha1h v3.16b,v0.16b // 16+.long 0x5e141040 //sha1p v0.16b,v2.16b,v20.4s+ add v20.4s,v19.4s,v6.4s+.long 0x5e2818c7 //sha1su1 v7.16b,v6.16b+.long 0x5e280802 //sha1h v2.16b,v0.16b // 17+.long 0x5e151060 //sha1p v0.16b,v3.16b,v21.4s+ add v21.4s,v19.4s,v7.4s++.long 0x5e280803 //sha1h v3.16b,v0.16b // 18+.long 0x5e141040 //sha1p v0.16b,v2.16b,v20.4s++.long 0x5e280802 //sha1h v2.16b,v0.16b // 19+.long 0x5e151060 //sha1p v0.16b,v3.16b,v21.4s++ add v1.4s,v1.4s,v2.4s+ add v0.4s,v0.4s,v22.4s++ cbnz x2,Loop_hw++ st1 {v0.4s},[x0],#16+ st1 {v1.s}[0],[x0]++ ldr x29,[sp],#16+ ret++.align 6+Lconst:+.long 0x5a827999,0x5a827999,0x5a827999,0x5a827999 //K_00_19+.long 0x6ed9eba1,0x6ed9eba1,0x6ed9eba1,0x6ed9eba1 //K_20_39+.long 0x8f1bbcdc,0x8f1bbcdc,0x8f1bbcdc,0x8f1bbcdc //K_40_59+.long 0xca62c1d6,0xca62c1d6,0xca62c1d6,0xca62c1d6 //K_60_79+.byte 83,72,65,49,32,98,108,111,99,107,32,116,114,97,110,115,102,111,114,109,32,102,111,114,32,65,82,77,118,56,44,32,67,82,89,80,84,79,71,65,77,83,32,98,121,32,64,100,111,116,45,97,115,109,0+.align 2+.align 2+#if !defined(__KERNELL__) && !defined(_WIN64)+.comm __crypton_armcap_P,4+.private_extern _crypton_armcap_P+#endif
@@ -0,0 +1,1218 @@+#ifndef __KERNEL__+# include "arm_arch.h"++#else+.globl crypton_sha1_asm_block_armv8+#endif++.text++.globl crypton_sha1_asm_block_data_order+.type crypton_sha1_asm_block_data_order,%function+.align 6+crypton_sha1_asm_block_data_order:+ adrp x16,crypton_armcap_P+ ldr w16,[x16,#:lo12:crypton_armcap_P]+ tst w16,#ARMV8_SHA1+ b.ne .Lv8_entry++ stp x29,x30,[sp,#-12*__SIZEOF_POINTER__]!+ add x29,sp,#0+ stp x19,x20,[sp,#2*__SIZEOF_POINTER__]+ stp x21,x22,[sp,#4*__SIZEOF_POINTER__]+ stp x23,x24,[sp,#6*__SIZEOF_POINTER__]+ stp x25,x26,[sp,#8*__SIZEOF_POINTER__]+ stp x27,x28,[sp,#10*__SIZEOF_POINTER__]++ ldp w20,w21,[x0]+ ldp w22,w23,[x0,#8]+ ldr w24,[x0,#16]++.Loop:+ ldr x3,[x1],#64+ movz w28,#0x7999+ sub x2,x2,#1+ movk w28,#0x5a82,lsl#16+#ifdef __AARCH64EB__+ ror x3,x3,#32+#else+ rev32 x3,x3+#endif+ add w24,w24,w28 // warm it up+ add w24,w24,w3+ lsr x4,x3,#32+ ldur x5,[x1,#-56]+ bic w25,w23,w21+ and w26,w22,w21+ ror w27,w20,#27+ add w23,w23,w28 // future e+=K+ orr w25,w25,w26+ add w24,w24,w27 // e+=rot(a,5)+ ror w21,w21,#2+ add w23,w23,w4 // future e+=X[i]+ add w24,w24,w25 // e+=F(b,c,d)+#ifdef __AARCH64EB__+ ror x5,x5,#32+#else+ rev32 x5,x5+#endif+ bic w25,w22,w20+ and w26,w21,w20+ ror w27,w24,#27+ add w22,w22,w28 // future e+=K+ orr w25,w25,w26+ add w23,w23,w27 // e+=rot(a,5)+ ror w20,w20,#2+ add w22,w22,w5 // future e+=X[i]+ add w23,w23,w25 // e+=F(b,c,d)+ lsr x6,x5,#32+ ldur x7,[x1,#-48]+ bic w25,w21,w24+ and w26,w20,w24+ ror w27,w23,#27+ add w21,w21,w28 // future e+=K+ orr w25,w25,w26+ add w22,w22,w27 // e+=rot(a,5)+ ror w24,w24,#2+ add w21,w21,w6 // future e+=X[i]+ add w22,w22,w25 // e+=F(b,c,d)+#ifdef __AARCH64EB__+ ror x7,x7,#32+#else+ rev32 x7,x7+#endif+ bic w25,w20,w23+ and w26,w24,w23+ ror w27,w22,#27+ add w20,w20,w28 // future e+=K+ orr w25,w25,w26+ add w21,w21,w27 // e+=rot(a,5)+ ror w23,w23,#2+ add w20,w20,w7 // future e+=X[i]+ add w21,w21,w25 // e+=F(b,c,d)+ lsr x8,x7,#32+ ldur x9,[x1,#-40]+ bic w25,w24,w22+ and w26,w23,w22+ ror w27,w21,#27+ add w24,w24,w28 // future e+=K+ orr w25,w25,w26+ add w20,w20,w27 // e+=rot(a,5)+ ror w22,w22,#2+ add w24,w24,w8 // future e+=X[i]+ add w20,w20,w25 // e+=F(b,c,d)+#ifdef __AARCH64EB__+ ror x9,x9,#32+#else+ rev32 x9,x9+#endif+ bic w25,w23,w21+ and w26,w22,w21+ ror w27,w20,#27+ add w23,w23,w28 // future e+=K+ orr w25,w25,w26+ add w24,w24,w27 // e+=rot(a,5)+ ror w21,w21,#2+ add w23,w23,w9 // future e+=X[i]+ add w24,w24,w25 // e+=F(b,c,d)+ lsr x10,x9,#32+ ldur x11,[x1,#-32]+ bic w25,w22,w20+ and w26,w21,w20+ ror w27,w24,#27+ add w22,w22,w28 // future e+=K+ orr w25,w25,w26+ add w23,w23,w27 // e+=rot(a,5)+ ror w20,w20,#2+ add w22,w22,w10 // future e+=X[i]+ add w23,w23,w25 // e+=F(b,c,d)+#ifdef __AARCH64EB__+ ror x11,x11,#32+#else+ rev32 x11,x11+#endif+ bic w25,w21,w24+ and w26,w20,w24+ ror w27,w23,#27+ add w21,w21,w28 // future e+=K+ orr w25,w25,w26+ add w22,w22,w27 // e+=rot(a,5)+ ror w24,w24,#2+ add w21,w21,w11 // future e+=X[i]+ add w22,w22,w25 // e+=F(b,c,d)+ lsr x12,x11,#32+ ldur x13,[x1,#-24]+ bic w25,w20,w23+ and w26,w24,w23+ ror w27,w22,#27+ add w20,w20,w28 // future e+=K+ orr w25,w25,w26+ add w21,w21,w27 // e+=rot(a,5)+ ror w23,w23,#2+ add w20,w20,w12 // future e+=X[i]+ add w21,w21,w25 // e+=F(b,c,d)+#ifdef __AARCH64EB__+ ror x13,x13,#32+#else+ rev32 x13,x13+#endif+ bic w25,w24,w22+ and w26,w23,w22+ ror w27,w21,#27+ add w24,w24,w28 // future e+=K+ orr w25,w25,w26+ add w20,w20,w27 // e+=rot(a,5)+ ror w22,w22,#2+ add w24,w24,w13 // future e+=X[i]+ add w20,w20,w25 // e+=F(b,c,d)+ lsr x14,x13,#32+ ldur x15,[x1,#-16]+ bic w25,w23,w21+ and w26,w22,w21+ ror w27,w20,#27+ add w23,w23,w28 // future e+=K+ orr w25,w25,w26+ add w24,w24,w27 // e+=rot(a,5)+ ror w21,w21,#2+ add w23,w23,w14 // future e+=X[i]+ add w24,w24,w25 // e+=F(b,c,d)+#ifdef __AARCH64EB__+ ror x15,x15,#32+#else+ rev32 x15,x15+#endif+ bic w25,w22,w20+ and w26,w21,w20+ ror w27,w24,#27+ add w22,w22,w28 // future e+=K+ orr w25,w25,w26+ add w23,w23,w27 // e+=rot(a,5)+ ror w20,w20,#2+ add w22,w22,w15 // future e+=X[i]+ add w23,w23,w25 // e+=F(b,c,d)+ lsr x16,x15,#32+ ldur x17,[x1,#-8]+ bic w25,w21,w24+ and w26,w20,w24+ ror w27,w23,#27+ add w21,w21,w28 // future e+=K+ orr w25,w25,w26+ add w22,w22,w27 // e+=rot(a,5)+ ror w24,w24,#2+ add w21,w21,w16 // future e+=X[i]+ add w22,w22,w25 // e+=F(b,c,d)+#ifdef __AARCH64EB__+ ror x17,x17,#32+#else+ rev32 x17,x17+#endif+ bic w25,w20,w23+ and w26,w24,w23+ ror w27,w22,#27+ add w20,w20,w28 // future e+=K+ orr w25,w25,w26+ add w21,w21,w27 // e+=rot(a,5)+ ror w23,w23,#2+ add w20,w20,w17 // future e+=X[i]+ add w21,w21,w25 // e+=F(b,c,d)+ lsr x19,x17,#32+ eor w3,w3,w5+ bic w25,w24,w22+ and w26,w23,w22+ ror w27,w21,#27+ eor w3,w3,w11+ add w24,w24,w28 // future e+=K+ orr w25,w25,w26+ add w20,w20,w27 // e+=rot(a,5)+ eor w3,w3,w16+ ror w22,w22,#2+ add w24,w24,w19 // future e+=X[i]+ add w20,w20,w25 // e+=F(b,c,d)+ ror w3,w3,#31+ eor w4,w4,w6+ bic w25,w23,w21+ and w26,w22,w21+ ror w27,w20,#27+ eor w4,w4,w12+ add w23,w23,w28 // future e+=K+ orr w25,w25,w26+ add w24,w24,w27 // e+=rot(a,5)+ eor w4,w4,w17+ ror w21,w21,#2+ add w23,w23,w3 // future e+=X[i]+ add w24,w24,w25 // e+=F(b,c,d)+ ror w4,w4,#31+ eor w5,w5,w7+ bic w25,w22,w20+ and w26,w21,w20+ ror w27,w24,#27+ eor w5,w5,w13+ add w22,w22,w28 // future e+=K+ orr w25,w25,w26+ add w23,w23,w27 // e+=rot(a,5)+ eor w5,w5,w19+ ror w20,w20,#2+ add w22,w22,w4 // future e+=X[i]+ add w23,w23,w25 // e+=F(b,c,d)+ ror w5,w5,#31+ eor w6,w6,w8+ bic w25,w21,w24+ and w26,w20,w24+ ror w27,w23,#27+ eor w6,w6,w14+ add w21,w21,w28 // future e+=K+ orr w25,w25,w26+ add w22,w22,w27 // e+=rot(a,5)+ eor w6,w6,w3+ ror w24,w24,#2+ add w21,w21,w5 // future e+=X[i]+ add w22,w22,w25 // e+=F(b,c,d)+ ror w6,w6,#31+ eor w7,w7,w9+ bic w25,w20,w23+ and w26,w24,w23+ ror w27,w22,#27+ eor w7,w7,w15+ add w20,w20,w28 // future e+=K+ orr w25,w25,w26+ add w21,w21,w27 // e+=rot(a,5)+ eor w7,w7,w4+ ror w23,w23,#2+ add w20,w20,w6 // future e+=X[i]+ add w21,w21,w25 // e+=F(b,c,d)+ ror w7,w7,#31+ movz w28,#0xeba1+ movk w28,#0x6ed9,lsl#16+ eor w8,w8,w10+ bic w25,w24,w22+ and w26,w23,w22+ ror w27,w21,#27+ eor w8,w8,w16+ add w24,w24,w28 // future e+=K+ orr w25,w25,w26+ add w20,w20,w27 // e+=rot(a,5)+ eor w8,w8,w5+ ror w22,w22,#2+ add w24,w24,w7 // future e+=X[i]+ add w20,w20,w25 // e+=F(b,c,d)+ ror w8,w8,#31+ eor w9,w9,w11+ eor w25,w23,w21+ ror w27,w20,#27+ add w23,w23,w28 // future e+=K+ eor w9,w9,w17+ eor w25,w25,w22+ add w24,w24,w27 // e+=rot(a,5)+ ror w21,w21,#2+ eor w9,w9,w6+ add w23,w23,w8 // future e+=X[i]+ add w24,w24,w25 // e+=F(b,c,d)+ ror w9,w9,#31+ eor w10,w10,w12+ eor w25,w22,w20+ ror w27,w24,#27+ add w22,w22,w28 // future e+=K+ eor w10,w10,w19+ eor w25,w25,w21+ add w23,w23,w27 // e+=rot(a,5)+ ror w20,w20,#2+ eor w10,w10,w7+ add w22,w22,w9 // future e+=X[i]+ add w23,w23,w25 // e+=F(b,c,d)+ ror w10,w10,#31+ eor w11,w11,w13+ eor w25,w21,w24+ ror w27,w23,#27+ add w21,w21,w28 // future e+=K+ eor w11,w11,w3+ eor w25,w25,w20+ add w22,w22,w27 // e+=rot(a,5)+ ror w24,w24,#2+ eor w11,w11,w8+ add w21,w21,w10 // future e+=X[i]+ add w22,w22,w25 // e+=F(b,c,d)+ ror w11,w11,#31+ eor w12,w12,w14+ eor w25,w20,w23+ ror w27,w22,#27+ add w20,w20,w28 // future e+=K+ eor w12,w12,w4+ eor w25,w25,w24+ add w21,w21,w27 // e+=rot(a,5)+ ror w23,w23,#2+ eor w12,w12,w9+ add w20,w20,w11 // future e+=X[i]+ add w21,w21,w25 // e+=F(b,c,d)+ ror w12,w12,#31+ eor w13,w13,w15+ eor w25,w24,w22+ ror w27,w21,#27+ add w24,w24,w28 // future e+=K+ eor w13,w13,w5+ eor w25,w25,w23+ add w20,w20,w27 // e+=rot(a,5)+ ror w22,w22,#2+ eor w13,w13,w10+ add w24,w24,w12 // future e+=X[i]+ add w20,w20,w25 // e+=F(b,c,d)+ ror w13,w13,#31+ eor w14,w14,w16+ eor w25,w23,w21+ ror w27,w20,#27+ add w23,w23,w28 // future e+=K+ eor w14,w14,w6+ eor w25,w25,w22+ add w24,w24,w27 // e+=rot(a,5)+ ror w21,w21,#2+ eor w14,w14,w11+ add w23,w23,w13 // future e+=X[i]+ add w24,w24,w25 // e+=F(b,c,d)+ ror w14,w14,#31+ eor w15,w15,w17+ eor w25,w22,w20+ ror w27,w24,#27+ add w22,w22,w28 // future e+=K+ eor w15,w15,w7+ eor w25,w25,w21+ add w23,w23,w27 // e+=rot(a,5)+ ror w20,w20,#2+ eor w15,w15,w12+ add w22,w22,w14 // future e+=X[i]+ add w23,w23,w25 // e+=F(b,c,d)+ ror w15,w15,#31+ eor w16,w16,w19+ eor w25,w21,w24+ ror w27,w23,#27+ add w21,w21,w28 // future e+=K+ eor w16,w16,w8+ eor w25,w25,w20+ add w22,w22,w27 // e+=rot(a,5)+ ror w24,w24,#2+ eor w16,w16,w13+ add w21,w21,w15 // future e+=X[i]+ add w22,w22,w25 // e+=F(b,c,d)+ ror w16,w16,#31+ eor w17,w17,w3+ eor w25,w20,w23+ ror w27,w22,#27+ add w20,w20,w28 // future e+=K+ eor w17,w17,w9+ eor w25,w25,w24+ add w21,w21,w27 // e+=rot(a,5)+ ror w23,w23,#2+ eor w17,w17,w14+ add w20,w20,w16 // future e+=X[i]+ add w21,w21,w25 // e+=F(b,c,d)+ ror w17,w17,#31+ eor w19,w19,w4+ eor w25,w24,w22+ ror w27,w21,#27+ add w24,w24,w28 // future e+=K+ eor w19,w19,w10+ eor w25,w25,w23+ add w20,w20,w27 // e+=rot(a,5)+ ror w22,w22,#2+ eor w19,w19,w15+ add w24,w24,w17 // future e+=X[i]+ add w20,w20,w25 // e+=F(b,c,d)+ ror w19,w19,#31+ eor w3,w3,w5+ eor w25,w23,w21+ ror w27,w20,#27+ add w23,w23,w28 // future e+=K+ eor w3,w3,w11+ eor w25,w25,w22+ add w24,w24,w27 // e+=rot(a,5)+ ror w21,w21,#2+ eor w3,w3,w16+ add w23,w23,w19 // future e+=X[i]+ add w24,w24,w25 // e+=F(b,c,d)+ ror w3,w3,#31+ eor w4,w4,w6+ eor w25,w22,w20+ ror w27,w24,#27+ add w22,w22,w28 // future e+=K+ eor w4,w4,w12+ eor w25,w25,w21+ add w23,w23,w27 // e+=rot(a,5)+ ror w20,w20,#2+ eor w4,w4,w17+ add w22,w22,w3 // future e+=X[i]+ add w23,w23,w25 // e+=F(b,c,d)+ ror w4,w4,#31+ eor w5,w5,w7+ eor w25,w21,w24+ ror w27,w23,#27+ add w21,w21,w28 // future e+=K+ eor w5,w5,w13+ eor w25,w25,w20+ add w22,w22,w27 // e+=rot(a,5)+ ror w24,w24,#2+ eor w5,w5,w19+ add w21,w21,w4 // future e+=X[i]+ add w22,w22,w25 // e+=F(b,c,d)+ ror w5,w5,#31+ eor w6,w6,w8+ eor w25,w20,w23+ ror w27,w22,#27+ add w20,w20,w28 // future e+=K+ eor w6,w6,w14+ eor w25,w25,w24+ add w21,w21,w27 // e+=rot(a,5)+ ror w23,w23,#2+ eor w6,w6,w3+ add w20,w20,w5 // future e+=X[i]+ add w21,w21,w25 // e+=F(b,c,d)+ ror w6,w6,#31+ eor w7,w7,w9+ eor w25,w24,w22+ ror w27,w21,#27+ add w24,w24,w28 // future e+=K+ eor w7,w7,w15+ eor w25,w25,w23+ add w20,w20,w27 // e+=rot(a,5)+ ror w22,w22,#2+ eor w7,w7,w4+ add w24,w24,w6 // future e+=X[i]+ add w20,w20,w25 // e+=F(b,c,d)+ ror w7,w7,#31+ eor w8,w8,w10+ eor w25,w23,w21+ ror w27,w20,#27+ add w23,w23,w28 // future e+=K+ eor w8,w8,w16+ eor w25,w25,w22+ add w24,w24,w27 // e+=rot(a,5)+ ror w21,w21,#2+ eor w8,w8,w5+ add w23,w23,w7 // future e+=X[i]+ add w24,w24,w25 // e+=F(b,c,d)+ ror w8,w8,#31+ eor w9,w9,w11+ eor w25,w22,w20+ ror w27,w24,#27+ add w22,w22,w28 // future e+=K+ eor w9,w9,w17+ eor w25,w25,w21+ add w23,w23,w27 // e+=rot(a,5)+ ror w20,w20,#2+ eor w9,w9,w6+ add w22,w22,w8 // future e+=X[i]+ add w23,w23,w25 // e+=F(b,c,d)+ ror w9,w9,#31+ eor w10,w10,w12+ eor w25,w21,w24+ ror w27,w23,#27+ add w21,w21,w28 // future e+=K+ eor w10,w10,w19+ eor w25,w25,w20+ add w22,w22,w27 // e+=rot(a,5)+ ror w24,w24,#2+ eor w10,w10,w7+ add w21,w21,w9 // future e+=X[i]+ add w22,w22,w25 // e+=F(b,c,d)+ ror w10,w10,#31+ eor w11,w11,w13+ eor w25,w20,w23+ ror w27,w22,#27+ add w20,w20,w28 // future e+=K+ eor w11,w11,w3+ eor w25,w25,w24+ add w21,w21,w27 // e+=rot(a,5)+ ror w23,w23,#2+ eor w11,w11,w8+ add w20,w20,w10 // future e+=X[i]+ add w21,w21,w25 // e+=F(b,c,d)+ ror w11,w11,#31+ movz w28,#0xbcdc+ movk w28,#0x8f1b,lsl#16+ eor w12,w12,w14+ eor w25,w24,w22+ ror w27,w21,#27+ add w24,w24,w28 // future e+=K+ eor w12,w12,w4+ eor w25,w25,w23+ add w20,w20,w27 // e+=rot(a,5)+ ror w22,w22,#2+ eor w12,w12,w9+ add w24,w24,w11 // future e+=X[i]+ add w20,w20,w25 // e+=F(b,c,d)+ ror w12,w12,#31+ orr w25,w21,w22+ and w26,w21,w22+ eor w13,w13,w15+ ror w27,w20,#27+ and w25,w25,w23+ add w23,w23,w28 // future e+=K+ eor w13,w13,w5+ add w24,w24,w27 // e+=rot(a,5)+ orr w25,w25,w26+ ror w21,w21,#2+ eor w13,w13,w10+ add w23,w23,w12 // future e+=X[i]+ add w24,w24,w25 // e+=F(b,c,d)+ ror w13,w13,#31+ orr w25,w20,w21+ and w26,w20,w21+ eor w14,w14,w16+ ror w27,w24,#27+ and w25,w25,w22+ add w22,w22,w28 // future e+=K+ eor w14,w14,w6+ add w23,w23,w27 // e+=rot(a,5)+ orr w25,w25,w26+ ror w20,w20,#2+ eor w14,w14,w11+ add w22,w22,w13 // future e+=X[i]+ add w23,w23,w25 // e+=F(b,c,d)+ ror w14,w14,#31+ orr w25,w24,w20+ and w26,w24,w20+ eor w15,w15,w17+ ror w27,w23,#27+ and w25,w25,w21+ add w21,w21,w28 // future e+=K+ eor w15,w15,w7+ add w22,w22,w27 // e+=rot(a,5)+ orr w25,w25,w26+ ror w24,w24,#2+ eor w15,w15,w12+ add w21,w21,w14 // future e+=X[i]+ add w22,w22,w25 // e+=F(b,c,d)+ ror w15,w15,#31+ orr w25,w23,w24+ and w26,w23,w24+ eor w16,w16,w19+ ror w27,w22,#27+ and w25,w25,w20+ add w20,w20,w28 // future e+=K+ eor w16,w16,w8+ add w21,w21,w27 // e+=rot(a,5)+ orr w25,w25,w26+ ror w23,w23,#2+ eor w16,w16,w13+ add w20,w20,w15 // future e+=X[i]+ add w21,w21,w25 // e+=F(b,c,d)+ ror w16,w16,#31+ orr w25,w22,w23+ and w26,w22,w23+ eor w17,w17,w3+ ror w27,w21,#27+ and w25,w25,w24+ add w24,w24,w28 // future e+=K+ eor w17,w17,w9+ add w20,w20,w27 // e+=rot(a,5)+ orr w25,w25,w26+ ror w22,w22,#2+ eor w17,w17,w14+ add w24,w24,w16 // future e+=X[i]+ add w20,w20,w25 // e+=F(b,c,d)+ ror w17,w17,#31+ orr w25,w21,w22+ and w26,w21,w22+ eor w19,w19,w4+ ror w27,w20,#27+ and w25,w25,w23+ add w23,w23,w28 // future e+=K+ eor w19,w19,w10+ add w24,w24,w27 // e+=rot(a,5)+ orr w25,w25,w26+ ror w21,w21,#2+ eor w19,w19,w15+ add w23,w23,w17 // future e+=X[i]+ add w24,w24,w25 // e+=F(b,c,d)+ ror w19,w19,#31+ orr w25,w20,w21+ and w26,w20,w21+ eor w3,w3,w5+ ror w27,w24,#27+ and w25,w25,w22+ add w22,w22,w28 // future e+=K+ eor w3,w3,w11+ add w23,w23,w27 // e+=rot(a,5)+ orr w25,w25,w26+ ror w20,w20,#2+ eor w3,w3,w16+ add w22,w22,w19 // future e+=X[i]+ add w23,w23,w25 // e+=F(b,c,d)+ ror w3,w3,#31+ orr w25,w24,w20+ and w26,w24,w20+ eor w4,w4,w6+ ror w27,w23,#27+ and w25,w25,w21+ add w21,w21,w28 // future e+=K+ eor w4,w4,w12+ add w22,w22,w27 // e+=rot(a,5)+ orr w25,w25,w26+ ror w24,w24,#2+ eor w4,w4,w17+ add w21,w21,w3 // future e+=X[i]+ add w22,w22,w25 // e+=F(b,c,d)+ ror w4,w4,#31+ orr w25,w23,w24+ and w26,w23,w24+ eor w5,w5,w7+ ror w27,w22,#27+ and w25,w25,w20+ add w20,w20,w28 // future e+=K+ eor w5,w5,w13+ add w21,w21,w27 // e+=rot(a,5)+ orr w25,w25,w26+ ror w23,w23,#2+ eor w5,w5,w19+ add w20,w20,w4 // future e+=X[i]+ add w21,w21,w25 // e+=F(b,c,d)+ ror w5,w5,#31+ orr w25,w22,w23+ and w26,w22,w23+ eor w6,w6,w8+ ror w27,w21,#27+ and w25,w25,w24+ add w24,w24,w28 // future e+=K+ eor w6,w6,w14+ add w20,w20,w27 // e+=rot(a,5)+ orr w25,w25,w26+ ror w22,w22,#2+ eor w6,w6,w3+ add w24,w24,w5 // future e+=X[i]+ add w20,w20,w25 // e+=F(b,c,d)+ ror w6,w6,#31+ orr w25,w21,w22+ and w26,w21,w22+ eor w7,w7,w9+ ror w27,w20,#27+ and w25,w25,w23+ add w23,w23,w28 // future e+=K+ eor w7,w7,w15+ add w24,w24,w27 // e+=rot(a,5)+ orr w25,w25,w26+ ror w21,w21,#2+ eor w7,w7,w4+ add w23,w23,w6 // future e+=X[i]+ add w24,w24,w25 // e+=F(b,c,d)+ ror w7,w7,#31+ orr w25,w20,w21+ and w26,w20,w21+ eor w8,w8,w10+ ror w27,w24,#27+ and w25,w25,w22+ add w22,w22,w28 // future e+=K+ eor w8,w8,w16+ add w23,w23,w27 // e+=rot(a,5)+ orr w25,w25,w26+ ror w20,w20,#2+ eor w8,w8,w5+ add w22,w22,w7 // future e+=X[i]+ add w23,w23,w25 // e+=F(b,c,d)+ ror w8,w8,#31+ orr w25,w24,w20+ and w26,w24,w20+ eor w9,w9,w11+ ror w27,w23,#27+ and w25,w25,w21+ add w21,w21,w28 // future e+=K+ eor w9,w9,w17+ add w22,w22,w27 // e+=rot(a,5)+ orr w25,w25,w26+ ror w24,w24,#2+ eor w9,w9,w6+ add w21,w21,w8 // future e+=X[i]+ add w22,w22,w25 // e+=F(b,c,d)+ ror w9,w9,#31+ orr w25,w23,w24+ and w26,w23,w24+ eor w10,w10,w12+ ror w27,w22,#27+ and w25,w25,w20+ add w20,w20,w28 // future e+=K+ eor w10,w10,w19+ add w21,w21,w27 // e+=rot(a,5)+ orr w25,w25,w26+ ror w23,w23,#2+ eor w10,w10,w7+ add w20,w20,w9 // future e+=X[i]+ add w21,w21,w25 // e+=F(b,c,d)+ ror w10,w10,#31+ orr w25,w22,w23+ and w26,w22,w23+ eor w11,w11,w13+ ror w27,w21,#27+ and w25,w25,w24+ add w24,w24,w28 // future e+=K+ eor w11,w11,w3+ add w20,w20,w27 // e+=rot(a,5)+ orr w25,w25,w26+ ror w22,w22,#2+ eor w11,w11,w8+ add w24,w24,w10 // future e+=X[i]+ add w20,w20,w25 // e+=F(b,c,d)+ ror w11,w11,#31+ orr w25,w21,w22+ and w26,w21,w22+ eor w12,w12,w14+ ror w27,w20,#27+ and w25,w25,w23+ add w23,w23,w28 // future e+=K+ eor w12,w12,w4+ add w24,w24,w27 // e+=rot(a,5)+ orr w25,w25,w26+ ror w21,w21,#2+ eor w12,w12,w9+ add w23,w23,w11 // future e+=X[i]+ add w24,w24,w25 // e+=F(b,c,d)+ ror w12,w12,#31+ orr w25,w20,w21+ and w26,w20,w21+ eor w13,w13,w15+ ror w27,w24,#27+ and w25,w25,w22+ add w22,w22,w28 // future e+=K+ eor w13,w13,w5+ add w23,w23,w27 // e+=rot(a,5)+ orr w25,w25,w26+ ror w20,w20,#2+ eor w13,w13,w10+ add w22,w22,w12 // future e+=X[i]+ add w23,w23,w25 // e+=F(b,c,d)+ ror w13,w13,#31+ orr w25,w24,w20+ and w26,w24,w20+ eor w14,w14,w16+ ror w27,w23,#27+ and w25,w25,w21+ add w21,w21,w28 // future e+=K+ eor w14,w14,w6+ add w22,w22,w27 // e+=rot(a,5)+ orr w25,w25,w26+ ror w24,w24,#2+ eor w14,w14,w11+ add w21,w21,w13 // future e+=X[i]+ add w22,w22,w25 // e+=F(b,c,d)+ ror w14,w14,#31+ orr w25,w23,w24+ and w26,w23,w24+ eor w15,w15,w17+ ror w27,w22,#27+ and w25,w25,w20+ add w20,w20,w28 // future e+=K+ eor w15,w15,w7+ add w21,w21,w27 // e+=rot(a,5)+ orr w25,w25,w26+ ror w23,w23,#2+ eor w15,w15,w12+ add w20,w20,w14 // future e+=X[i]+ add w21,w21,w25 // e+=F(b,c,d)+ ror w15,w15,#31+ movz w28,#0xc1d6+ movk w28,#0xca62,lsl#16+ orr w25,w22,w23+ and w26,w22,w23+ eor w16,w16,w19+ ror w27,w21,#27+ and w25,w25,w24+ add w24,w24,w28 // future e+=K+ eor w16,w16,w8+ add w20,w20,w27 // e+=rot(a,5)+ orr w25,w25,w26+ ror w22,w22,#2+ eor w16,w16,w13+ add w24,w24,w15 // future e+=X[i]+ add w20,w20,w25 // e+=F(b,c,d)+ ror w16,w16,#31+ eor w17,w17,w3+ eor w25,w23,w21+ ror w27,w20,#27+ add w23,w23,w28 // future e+=K+ eor w17,w17,w9+ eor w25,w25,w22+ add w24,w24,w27 // e+=rot(a,5)+ ror w21,w21,#2+ eor w17,w17,w14+ add w23,w23,w16 // future e+=X[i]+ add w24,w24,w25 // e+=F(b,c,d)+ ror w17,w17,#31+ eor w19,w19,w4+ eor w25,w22,w20+ ror w27,w24,#27+ add w22,w22,w28 // future e+=K+ eor w19,w19,w10+ eor w25,w25,w21+ add w23,w23,w27 // e+=rot(a,5)+ ror w20,w20,#2+ eor w19,w19,w15+ add w22,w22,w17 // future e+=X[i]+ add w23,w23,w25 // e+=F(b,c,d)+ ror w19,w19,#31+ eor w3,w3,w5+ eor w25,w21,w24+ ror w27,w23,#27+ add w21,w21,w28 // future e+=K+ eor w3,w3,w11+ eor w25,w25,w20+ add w22,w22,w27 // e+=rot(a,5)+ ror w24,w24,#2+ eor w3,w3,w16+ add w21,w21,w19 // future e+=X[i]+ add w22,w22,w25 // e+=F(b,c,d)+ ror w3,w3,#31+ eor w4,w4,w6+ eor w25,w20,w23+ ror w27,w22,#27+ add w20,w20,w28 // future e+=K+ eor w4,w4,w12+ eor w25,w25,w24+ add w21,w21,w27 // e+=rot(a,5)+ ror w23,w23,#2+ eor w4,w4,w17+ add w20,w20,w3 // future e+=X[i]+ add w21,w21,w25 // e+=F(b,c,d)+ ror w4,w4,#31+ eor w5,w5,w7+ eor w25,w24,w22+ ror w27,w21,#27+ add w24,w24,w28 // future e+=K+ eor w5,w5,w13+ eor w25,w25,w23+ add w20,w20,w27 // e+=rot(a,5)+ ror w22,w22,#2+ eor w5,w5,w19+ add w24,w24,w4 // future e+=X[i]+ add w20,w20,w25 // e+=F(b,c,d)+ ror w5,w5,#31+ eor w6,w6,w8+ eor w25,w23,w21+ ror w27,w20,#27+ add w23,w23,w28 // future e+=K+ eor w6,w6,w14+ eor w25,w25,w22+ add w24,w24,w27 // e+=rot(a,5)+ ror w21,w21,#2+ eor w6,w6,w3+ add w23,w23,w5 // future e+=X[i]+ add w24,w24,w25 // e+=F(b,c,d)+ ror w6,w6,#31+ eor w7,w7,w9+ eor w25,w22,w20+ ror w27,w24,#27+ add w22,w22,w28 // future e+=K+ eor w7,w7,w15+ eor w25,w25,w21+ add w23,w23,w27 // e+=rot(a,5)+ ror w20,w20,#2+ eor w7,w7,w4+ add w22,w22,w6 // future e+=X[i]+ add w23,w23,w25 // e+=F(b,c,d)+ ror w7,w7,#31+ eor w8,w8,w10+ eor w25,w21,w24+ ror w27,w23,#27+ add w21,w21,w28 // future e+=K+ eor w8,w8,w16+ eor w25,w25,w20+ add w22,w22,w27 // e+=rot(a,5)+ ror w24,w24,#2+ eor w8,w8,w5+ add w21,w21,w7 // future e+=X[i]+ add w22,w22,w25 // e+=F(b,c,d)+ ror w8,w8,#31+ eor w9,w9,w11+ eor w25,w20,w23+ ror w27,w22,#27+ add w20,w20,w28 // future e+=K+ eor w9,w9,w17+ eor w25,w25,w24+ add w21,w21,w27 // e+=rot(a,5)+ ror w23,w23,#2+ eor w9,w9,w6+ add w20,w20,w8 // future e+=X[i]+ add w21,w21,w25 // e+=F(b,c,d)+ ror w9,w9,#31+ eor w10,w10,w12+ eor w25,w24,w22+ ror w27,w21,#27+ add w24,w24,w28 // future e+=K+ eor w10,w10,w19+ eor w25,w25,w23+ add w20,w20,w27 // e+=rot(a,5)+ ror w22,w22,#2+ eor w10,w10,w7+ add w24,w24,w9 // future e+=X[i]+ add w20,w20,w25 // e+=F(b,c,d)+ ror w10,w10,#31+ eor w11,w11,w13+ eor w25,w23,w21+ ror w27,w20,#27+ add w23,w23,w28 // future e+=K+ eor w11,w11,w3+ eor w25,w25,w22+ add w24,w24,w27 // e+=rot(a,5)+ ror w21,w21,#2+ eor w11,w11,w8+ add w23,w23,w10 // future e+=X[i]+ add w24,w24,w25 // e+=F(b,c,d)+ ror w11,w11,#31+ eor w12,w12,w14+ eor w25,w22,w20+ ror w27,w24,#27+ add w22,w22,w28 // future e+=K+ eor w12,w12,w4+ eor w25,w25,w21+ add w23,w23,w27 // e+=rot(a,5)+ ror w20,w20,#2+ eor w12,w12,w9+ add w22,w22,w11 // future e+=X[i]+ add w23,w23,w25 // e+=F(b,c,d)+ ror w12,w12,#31+ eor w13,w13,w15+ eor w25,w21,w24+ ror w27,w23,#27+ add w21,w21,w28 // future e+=K+ eor w13,w13,w5+ eor w25,w25,w20+ add w22,w22,w27 // e+=rot(a,5)+ ror w24,w24,#2+ eor w13,w13,w10+ add w21,w21,w12 // future e+=X[i]+ add w22,w22,w25 // e+=F(b,c,d)+ ror w13,w13,#31+ eor w14,w14,w16+ eor w25,w20,w23+ ror w27,w22,#27+ add w20,w20,w28 // future e+=K+ eor w14,w14,w6+ eor w25,w25,w24+ add w21,w21,w27 // e+=rot(a,5)+ ror w23,w23,#2+ eor w14,w14,w11+ add w20,w20,w13 // future e+=X[i]+ add w21,w21,w25 // e+=F(b,c,d)+ ror w14,w14,#31+ eor w15,w15,w17+ eor w25,w24,w22+ ror w27,w21,#27+ add w24,w24,w28 // future e+=K+ eor w15,w15,w7+ eor w25,w25,w23+ add w20,w20,w27 // e+=rot(a,5)+ ror w22,w22,#2+ eor w15,w15,w12+ add w24,w24,w14 // future e+=X[i]+ add w20,w20,w25 // e+=F(b,c,d)+ ror w15,w15,#31+ eor w16,w16,w19+ eor w25,w23,w21+ ror w27,w20,#27+ add w23,w23,w28 // future e+=K+ eor w16,w16,w8+ eor w25,w25,w22+ add w24,w24,w27 // e+=rot(a,5)+ ror w21,w21,#2+ eor w16,w16,w13+ add w23,w23,w15 // future e+=X[i]+ add w24,w24,w25 // e+=F(b,c,d)+ ror w16,w16,#31+ eor w17,w17,w3+ eor w25,w22,w20+ ror w27,w24,#27+ add w22,w22,w28 // future e+=K+ eor w17,w17,w9+ eor w25,w25,w21+ add w23,w23,w27 // e+=rot(a,5)+ ror w20,w20,#2+ eor w17,w17,w14+ add w22,w22,w16 // future e+=X[i]+ add w23,w23,w25 // e+=F(b,c,d)+ ror w17,w17,#31+ eor w19,w19,w4+ eor w25,w21,w24+ ror w27,w23,#27+ add w21,w21,w28 // future e+=K+ eor w19,w19,w10+ eor w25,w25,w20+ add w22,w22,w27 // e+=rot(a,5)+ ror w24,w24,#2+ eor w19,w19,w15+ add w21,w21,w17 // future e+=X[i]+ add w22,w22,w25 // e+=F(b,c,d)+ ror w19,w19,#31+ ldp w4,w5,[x0]+ eor w25,w20,w23+ ror w27,w22,#27+ add w20,w20,w28 // future e+=K+ eor w25,w25,w24+ add w21,w21,w27 // e+=rot(a,5)+ ror w23,w23,#2+ add w20,w20,w19 // future e+=X[i]+ add w21,w21,w25 // e+=F(b,c,d)+ ldp w6,w7,[x0,#8]+ eor w25,w24,w22+ ror w27,w21,#27+ eor w25,w25,w23+ add w20,w20,w27 // e+=rot(a,5)+ ror w22,w22,#2+ ldr w8,[x0,#16]+ add w20,w20,w25 // e+=F(b,c,d)+ add w21,w21,w5+ add w22,w22,w6+ add w20,w20,w4+ add w23,w23,w7+ add w24,w24,w8+ stp w20,w21,[x0]+ stp w22,w23,[x0,#8]+ str w24,[x0,#16]+ cbnz x2,.Loop++ ldp x19,x20,[sp,#2*__SIZEOF_POINTER__]+ ldp x21,x22,[sp,#4*__SIZEOF_POINTER__]+ ldp x23,x24,[sp,#6*__SIZEOF_POINTER__]+ ldp x25,x26,[sp,#8*__SIZEOF_POINTER__]+ ldp x27,x28,[sp,#10*__SIZEOF_POINTER__]+ ldr x29,[sp],#12*__SIZEOF_POINTER__+ ret+.size crypton_sha1_asm_block_data_order,.-crypton_sha1_asm_block_data_order+.type crypton_sha1_asm_block_armv8,%function+.align 6+crypton_sha1_asm_block_armv8:+.Lv8_entry:+ stp x29,x30,[sp,#-16]!+ add x29,sp,#0++ adr x4,.Lconst+ eor v1.16b,v1.16b,v1.16b+ ld1 {v0.4s},[x0],#16+ ld1 {v1.s}[0],[x0]+ sub x0,x0,#16+ ld1 {v16.4s,v17.4s,v18.4s,v19.4s},[x4]++.Loop_hw:+ ld1 {v4.16b,v5.16b,v6.16b,v7.16b},[x1],#64+ sub x2,x2,#1+ rev32 v4.16b,v4.16b+ rev32 v5.16b,v5.16b++ add v20.4s,v16.4s,v4.4s+ rev32 v6.16b,v6.16b+ orr v22.16b,v0.16b,v0.16b // offload++ add v21.4s,v16.4s,v5.4s+ rev32 v7.16b,v7.16b+.inst 0x5e280803 //sha1h v3.16b,v0.16b+.inst 0x5e140020 //sha1c v0.16b,v1.16b,v20.4s // 0+ add v20.4s,v16.4s,v6.4s+.inst 0x5e0630a4 //sha1su0 v4.16b,v5.16b,v6.16b+.inst 0x5e280802 //sha1h v2.16b,v0.16b // 1+.inst 0x5e150060 //sha1c v0.16b,v3.16b,v21.4s+ add v21.4s,v16.4s,v7.4s+.inst 0x5e2818e4 //sha1su1 v4.16b,v7.16b+.inst 0x5e0730c5 //sha1su0 v5.16b,v6.16b,v7.16b+.inst 0x5e280803 //sha1h v3.16b,v0.16b // 2+.inst 0x5e140040 //sha1c v0.16b,v2.16b,v20.4s+ add v20.4s,v16.4s,v4.4s+.inst 0x5e281885 //sha1su1 v5.16b,v4.16b+.inst 0x5e0430e6 //sha1su0 v6.16b,v7.16b,v4.16b+.inst 0x5e280802 //sha1h v2.16b,v0.16b // 3+.inst 0x5e150060 //sha1c v0.16b,v3.16b,v21.4s+ add v21.4s,v17.4s,v5.4s+.inst 0x5e2818a6 //sha1su1 v6.16b,v5.16b+.inst 0x5e053087 //sha1su0 v7.16b,v4.16b,v5.16b+.inst 0x5e280803 //sha1h v3.16b,v0.16b // 4+.inst 0x5e140040 //sha1c v0.16b,v2.16b,v20.4s+ add v20.4s,v17.4s,v6.4s+.inst 0x5e2818c7 //sha1su1 v7.16b,v6.16b+.inst 0x5e0630a4 //sha1su0 v4.16b,v5.16b,v6.16b+.inst 0x5e280802 //sha1h v2.16b,v0.16b // 5+.inst 0x5e151060 //sha1p v0.16b,v3.16b,v21.4s+ add v21.4s,v17.4s,v7.4s+.inst 0x5e2818e4 //sha1su1 v4.16b,v7.16b+.inst 0x5e0730c5 //sha1su0 v5.16b,v6.16b,v7.16b+.inst 0x5e280803 //sha1h v3.16b,v0.16b // 6+.inst 0x5e141040 //sha1p v0.16b,v2.16b,v20.4s+ add v20.4s,v17.4s,v4.4s+.inst 0x5e281885 //sha1su1 v5.16b,v4.16b+.inst 0x5e0430e6 //sha1su0 v6.16b,v7.16b,v4.16b+.inst 0x5e280802 //sha1h v2.16b,v0.16b // 7+.inst 0x5e151060 //sha1p v0.16b,v3.16b,v21.4s+ add v21.4s,v17.4s,v5.4s+.inst 0x5e2818a6 //sha1su1 v6.16b,v5.16b+.inst 0x5e053087 //sha1su0 v7.16b,v4.16b,v5.16b+.inst 0x5e280803 //sha1h v3.16b,v0.16b // 8+.inst 0x5e141040 //sha1p v0.16b,v2.16b,v20.4s+ add v20.4s,v18.4s,v6.4s+.inst 0x5e2818c7 //sha1su1 v7.16b,v6.16b+.inst 0x5e0630a4 //sha1su0 v4.16b,v5.16b,v6.16b+.inst 0x5e280802 //sha1h v2.16b,v0.16b // 9+.inst 0x5e151060 //sha1p v0.16b,v3.16b,v21.4s+ add v21.4s,v18.4s,v7.4s+.inst 0x5e2818e4 //sha1su1 v4.16b,v7.16b+.inst 0x5e0730c5 //sha1su0 v5.16b,v6.16b,v7.16b+.inst 0x5e280803 //sha1h v3.16b,v0.16b // 10+.inst 0x5e142040 //sha1m v0.16b,v2.16b,v20.4s+ add v20.4s,v18.4s,v4.4s+.inst 0x5e281885 //sha1su1 v5.16b,v4.16b+.inst 0x5e0430e6 //sha1su0 v6.16b,v7.16b,v4.16b+.inst 0x5e280802 //sha1h v2.16b,v0.16b // 11+.inst 0x5e152060 //sha1m v0.16b,v3.16b,v21.4s+ add v21.4s,v18.4s,v5.4s+.inst 0x5e2818a6 //sha1su1 v6.16b,v5.16b+.inst 0x5e053087 //sha1su0 v7.16b,v4.16b,v5.16b+.inst 0x5e280803 //sha1h v3.16b,v0.16b // 12+.inst 0x5e142040 //sha1m v0.16b,v2.16b,v20.4s+ add v20.4s,v18.4s,v6.4s+.inst 0x5e2818c7 //sha1su1 v7.16b,v6.16b+.inst 0x5e0630a4 //sha1su0 v4.16b,v5.16b,v6.16b+.inst 0x5e280802 //sha1h v2.16b,v0.16b // 13+.inst 0x5e152060 //sha1m v0.16b,v3.16b,v21.4s+ add v21.4s,v19.4s,v7.4s+.inst 0x5e2818e4 //sha1su1 v4.16b,v7.16b+.inst 0x5e0730c5 //sha1su0 v5.16b,v6.16b,v7.16b+.inst 0x5e280803 //sha1h v3.16b,v0.16b // 14+.inst 0x5e142040 //sha1m v0.16b,v2.16b,v20.4s+ add v20.4s,v19.4s,v4.4s+.inst 0x5e281885 //sha1su1 v5.16b,v4.16b+.inst 0x5e0430e6 //sha1su0 v6.16b,v7.16b,v4.16b+.inst 0x5e280802 //sha1h v2.16b,v0.16b // 15+.inst 0x5e151060 //sha1p v0.16b,v3.16b,v21.4s+ add v21.4s,v19.4s,v5.4s+.inst 0x5e2818a6 //sha1su1 v6.16b,v5.16b+.inst 0x5e053087 //sha1su0 v7.16b,v4.16b,v5.16b+.inst 0x5e280803 //sha1h v3.16b,v0.16b // 16+.inst 0x5e141040 //sha1p v0.16b,v2.16b,v20.4s+ add v20.4s,v19.4s,v6.4s+.inst 0x5e2818c7 //sha1su1 v7.16b,v6.16b+.inst 0x5e280802 //sha1h v2.16b,v0.16b // 17+.inst 0x5e151060 //sha1p v0.16b,v3.16b,v21.4s+ add v21.4s,v19.4s,v7.4s++.inst 0x5e280803 //sha1h v3.16b,v0.16b // 18+.inst 0x5e141040 //sha1p v0.16b,v2.16b,v20.4s++.inst 0x5e280802 //sha1h v2.16b,v0.16b // 19+.inst 0x5e151060 //sha1p v0.16b,v3.16b,v21.4s++ add v1.4s,v1.4s,v2.4s+ add v0.4s,v0.4s,v22.4s++ cbnz x2,.Loop_hw++ st1 {v0.4s},[x0],#16+ st1 {v1.s}[0],[x0]++ ldr x29,[sp],#16+ ret+.size crypton_sha1_asm_block_armv8,.-crypton_sha1_asm_block_armv8+.align 6+.Lconst:+.long 0x5a827999,0x5a827999,0x5a827999,0x5a827999 //K_00_19+.long 0x6ed9eba1,0x6ed9eba1,0x6ed9eba1,0x6ed9eba1 //K_20_39+.long 0x8f1bbcdc,0x8f1bbcdc,0x8f1bbcdc,0x8f1bbcdc //K_40_59+.long 0xca62c1d6,0xca62c1d6,0xca62c1d6,0xca62c1d6 //K_60_79+.byte 83,72,65,49,32,98,108,111,99,107,32,116,114,97,110,115,102,111,114,109,32,102,111,114,32,65,82,77,118,56,44,32,67,82,89,80,84,79,71,65,77,83,32,98,121,32,64,100,111,116,45,97,115,109,0+.align 2+.align 2+#if !defined(__KERNELL__) && !defined(_WIN64)+.comm crypton_armcap_P,4,4+.hidden crypton_armcap_P+#endif++.section .note.GNU-stack,"",%progbits
@@ -0,0 +1,362 @@+#!/usr/bin/env perl+#+# ====================================================================+# Written by Andy Polyakov, @dot-asm, initially for the OpenSSL+# project.+# ====================================================================+#+# SHA1 for ARMv8.+#+# Performance in cycles per processed byte and improvement coefficient+# over code generated with "default" compiler:+#+# hardware-assisted software(*)+# Apple A7 2.31 4.13 (+14%)+# Apple A10 1.61+# Apple A14/M1 1.32 3.82 (-13%)(***)+# Cortex-A53 2.24 8.03 (+97%)+# Cortex-A57 2.35 7.88 (+74%)+# Cortex-A76 1.64 5.20+# Cortex-X2 1.63 4.07+# Cortex-X925 1.64 3.81+# Denver 2.13 3.97 (+0%)(**)+# X-Gene 8.80 (+200%)+# Mongoose 2.05 6.50 (+160%)+# Kryo 1.88 8.00 (+90%)+# ThunderX2 2.64 6.36 (+150%)+# Snapdraon X 1.48 3.82+#+# (*) Software results are presented mostly for reference purposes.+# (**) Keep in mind that Denver relies on binary translation, which+# optimizes compiler output at run-time.+# (***) There is some room for improvement on "extra-wide" processor+# such as A14/M1. Nothing is done, because it's not used anyway.++$flavour = shift;+$output = shift;++if ($flavour && $flavour ne "void") {+ $0 =~ m/(.*[\/\\])[^\/\\]+$/; $dir=$1;+ ( $xlate="${dir}arm-xlate.pl" and -f $xlate ) or+ ( $xlate="${dir}../../perlasm/arm-xlate.pl" and -f $xlate) or+ die "can't locate arm-xlate.pl";++ open STDOUT,"| \"$^X\" $xlate $flavour $output";+} else {+ open STDOUT,">$output";+}++($ctx,$inp,$num)=("x0","x1","x2");+@Xw=map("w$_",(3..17,19));+@Xx=map("x$_",(3..17,19));+@V=($A,$B,$C,$D,$E)=map("w$_",(20..24));+($t0,$t1,$t2,$K)=map("w$_",(25..28));+++sub BODY_00_19 {+my ($i,$a,$b,$c,$d,$e)=@_;+my $j=($i+2)&15;++$code.=<<___ if ($i<15 && !($i&1));+ lsr @Xx[$i+1],@Xx[$i],#32+___+$code.=<<___ if ($i<14 && !($i&1));+ ldur @Xx[$i+2],[$inp,#`($i+2)*4-64`]+___+$code.=<<___ if ($i<14 && ($i&1));+#ifdef __AARCH64EB__+ ror @Xx[$i+1],@Xx[$i+1],#32+#else+ rev32 @Xx[$i+1],@Xx[$i+1]+#endif+___+$code.=<<___ if ($i<14);+ bic $t0,$d,$b+ and $t1,$c,$b+ ror $t2,$a,#27+ add $d,$d,$K // future e+=K+ orr $t0,$t0,$t1+ add $e,$e,$t2 // e+=rot(a,5)+ ror $b,$b,#2+ add $d,$d,@Xw[($i+1)&15] // future e+=X[i]+ add $e,$e,$t0 // e+=F(b,c,d)+___+$code.=<<___ if ($i==19);+ movz $K,#0xeba1+ movk $K,#0x6ed9,lsl#16+___+$code.=<<___ if ($i>=14);+ eor @Xw[$j],@Xw[$j],@Xw[($j+2)&15]+ bic $t0,$d,$b+ and $t1,$c,$b+ ror $t2,$a,#27+ eor @Xw[$j],@Xw[$j],@Xw[($j+8)&15]+ add $d,$d,$K // future e+=K+ orr $t0,$t0,$t1+ add $e,$e,$t2 // e+=rot(a,5)+ eor @Xw[$j],@Xw[$j],@Xw[($j+13)&15]+ ror $b,$b,#2+ add $d,$d,@Xw[($i+1)&15] // future e+=X[i]+ add $e,$e,$t0 // e+=F(b,c,d)+ ror @Xw[$j],@Xw[$j],#31+___+}++sub BODY_40_59 {+my ($i,$a,$b,$c,$d,$e)=@_;+my $j=($i+2)&15;++$code.=<<___ if ($i==59);+ movz $K,#0xc1d6+ movk $K,#0xca62,lsl#16+___+$code.=<<___;+ orr $t0,$b,$c+ and $t1,$b,$c+ eor @Xw[$j],@Xw[$j],@Xw[($j+2)&15]+ ror $t2,$a,#27+ and $t0,$t0,$d+ add $d,$d,$K // future e+=K+ eor @Xw[$j],@Xw[$j],@Xw[($j+8)&15]+ add $e,$e,$t2 // e+=rot(a,5)+ orr $t0,$t0,$t1+ ror $b,$b,#2+ eor @Xw[$j],@Xw[$j],@Xw[($j+13)&15]+ add $d,$d,@Xw[($i+1)&15] // future e+=X[i]+ add $e,$e,$t0 // e+=F(b,c,d)+ ror @Xw[$j],@Xw[$j],#31+___+}++sub BODY_20_39 {+my ($i,$a,$b,$c,$d,$e)=@_;+my $j=($i+2)&15;++$code.=<<___ if ($i==39);+ movz $K,#0xbcdc+ movk $K,#0x8f1b,lsl#16+___+$code.=<<___ if ($i<78);+ eor @Xw[$j],@Xw[$j],@Xw[($j+2)&15]+ eor $t0,$d,$b+ ror $t2,$a,#27+ add $d,$d,$K // future e+=K+ eor @Xw[$j],@Xw[$j],@Xw[($j+8)&15]+ eor $t0,$t0,$c+ add $e,$e,$t2 // e+=rot(a,5)+ ror $b,$b,#2+ eor @Xw[$j],@Xw[$j],@Xw[($j+13)&15]+ add $d,$d,@Xw[($i+1)&15] // future e+=X[i]+ add $e,$e,$t0 // e+=F(b,c,d)+ ror @Xw[$j],@Xw[$j],#31+___+$code.=<<___ if ($i==78);+ ldp @Xw[1],@Xw[2],[$ctx]+ eor $t0,$d,$b+ ror $t2,$a,#27+ add $d,$d,$K // future e+=K+ eor $t0,$t0,$c+ add $e,$e,$t2 // e+=rot(a,5)+ ror $b,$b,#2+ add $d,$d,@Xw[($i+1)&15] // future e+=X[i]+ add $e,$e,$t0 // e+=F(b,c,d)+___+$code.=<<___ if ($i==79);+ ldp @Xw[3],@Xw[4],[$ctx,#8]+ eor $t0,$d,$b+ ror $t2,$a,#27+ eor $t0,$t0,$c+ add $e,$e,$t2 // e+=rot(a,5)+ ror $b,$b,#2+ ldr @Xw[5],[$ctx,#16]+ add $e,$e,$t0 // e+=F(b,c,d)+___+}++$code.=<<___;+#ifndef __KERNEL__+# include "arm_arch.h"+.extern OPENSSL_armcap_P+#else+.globl sha1_block_armv8+#endif++.text++.globl sha1_block_data_order+.type sha1_block_data_order,%function+.align 6+sha1_block_data_order:+ adrp c16,OPENSSL_armcap_P+ ldr w16,[c16,#:lo12:OPENSSL_armcap_P]+ tst w16,#ARMV8_SHA1+ b.ne .Lv8_entry++ stp c29,c30,[csp,#-12*__SIZEOF_POINTER__]!+ add c29,csp,#0+ stp c19,c20,[csp,#2*__SIZEOF_POINTER__]+ stp c21,c22,[csp,#4*__SIZEOF_POINTER__]+ stp c23,c24,[csp,#6*__SIZEOF_POINTER__]+ stp c25,c26,[csp,#8*__SIZEOF_POINTER__]+ stp c27,c28,[csp,#10*__SIZEOF_POINTER__]++ ldp $A,$B,[$ctx]+ ldp $C,$D,[$ctx,#8]+ ldr $E,[$ctx,#16]++.Loop:+ ldr @Xx[0],[$inp],#64+ movz $K,#0x7999+ sub $num,$num,#1+ movk $K,#0x5a82,lsl#16+#ifdef __AARCH64EB__+ ror $Xx[0],@Xx[0],#32+#else+ rev32 @Xx[0],@Xx[0]+#endif+ add $E,$E,$K // warm it up+ add $E,$E,@Xw[0]+___+for($i=0;$i<20;$i++) { &BODY_00_19($i,@V); unshift(@V,pop(@V)); }+for(;$i<40;$i++) { &BODY_20_39($i,@V); unshift(@V,pop(@V)); }+for(;$i<60;$i++) { &BODY_40_59($i,@V); unshift(@V,pop(@V)); }+for(;$i<80;$i++) { &BODY_20_39($i,@V); unshift(@V,pop(@V)); }+$code.=<<___;+ add $B,$B,@Xw[2]+ add $C,$C,@Xw[3]+ add $A,$A,@Xw[1]+ add $D,$D,@Xw[4]+ add $E,$E,@Xw[5]+ stp $A,$B,[$ctx]+ stp $C,$D,[$ctx,#8]+ str $E,[$ctx,#16]+ cbnz $num,.Loop++ ldp c19,c20,[csp,#2*__SIZEOF_POINTER__]+ ldp c21,c22,[csp,#4*__SIZEOF_POINTER__]+ ldp c23,c24,[csp,#6*__SIZEOF_POINTER__]+ ldp c25,c26,[csp,#8*__SIZEOF_POINTER__]+ ldp c27,c28,[csp,#10*__SIZEOF_POINTER__]+ ldr c29,[csp],#12*__SIZEOF_POINTER__+ ret+.size sha1_block_data_order,.-sha1_block_data_order+___+{{{+my ($ABCD,$E,$E0,$E1)=map("v$_.16b",(0..3));+my @MSG=map("v$_.16b",(4..7));+my @Kxx=map("v$_.4s",(16..19));+my ($W0,$W1)=("v20.4s","v21.4s");+my $ABCD_SAVE="v22.16b";++$code.=<<___;+.type sha1_block_armv8,%function+.align 6+sha1_block_armv8:+.Lv8_entry:+ stp x29,x30,[sp,#-16]!+ add x29,sp,#0++ adr x4,.Lconst+ eor $E,$E,$E+ ld1.32 {$ABCD},[$ctx],#16+ ld1.32 {$E}[0],[$ctx]+ csub $ctx,$ctx,#16+ ld1.32 {@Kxx[0]-@Kxx[3]},[x4]++.Loop_hw:+ ld1 {@MSG[0]-@MSG[3]},[$inp],#64+ sub $num,$num,#1+ rev32 @MSG[0],@MSG[0]+ rev32 @MSG[1],@MSG[1]++ add.i32 $W0,@Kxx[0],@MSG[0]+ rev32 @MSG[2],@MSG[2]+ orr $ABCD_SAVE,$ABCD,$ABCD // offload++ add.i32 $W1,@Kxx[0],@MSG[1]+ rev32 @MSG[3],@MSG[3]+ sha1h $E1,$ABCD+ sha1c $ABCD,$E,$W0 // 0+ add.i32 $W0,@Kxx[$j],@MSG[2]+ sha1su0 @MSG[0],@MSG[1],@MSG[2]+___+for ($j=0,$i=1;$i<20-3;$i++) {+my $f=("c","p","m","p")[$i/5];+$code.=<<___;+ sha1h $E0,$ABCD // $i+ sha1$f $ABCD,$E1,$W1+ add.i32 $W1,@Kxx[$j],@MSG[3]+ sha1su1 @MSG[0],@MSG[3]+___+$code.=<<___ if ($i<20-4);+ sha1su0 @MSG[1],@MSG[2],@MSG[3]+___+ ($E0,$E1)=($E1,$E0); ($W0,$W1)=($W1,$W0);+ push(@MSG,shift(@MSG)); $j++ if ((($i+3)%5)==0);+}+$code.=<<___;+ sha1h $E0,$ABCD // $i+ sha1p $ABCD,$E1,$W1+ add.i32 $W1,@Kxx[$j],@MSG[3]++ sha1h $E1,$ABCD // 18+ sha1p $ABCD,$E0,$W0++ sha1h $E0,$ABCD // 19+ sha1p $ABCD,$E1,$W1++ add.i32 $E,$E,$E0+ add.i32 $ABCD,$ABCD,$ABCD_SAVE++ cbnz $num,.Loop_hw++ st1.32 {$ABCD},[$ctx],#16+ st1.32 {$E}[0],[$ctx]++ ldr x29,[sp],#16+ ret+.size sha1_block_armv8,.-sha1_block_armv8+.align 6+.Lconst:+.long 0x5a827999,0x5a827999,0x5a827999,0x5a827999 //K_00_19+.long 0x6ed9eba1,0x6ed9eba1,0x6ed9eba1,0x6ed9eba1 //K_20_39+.long 0x8f1bbcdc,0x8f1bbcdc,0x8f1bbcdc,0x8f1bbcdc //K_40_59+.long 0xca62c1d6,0xca62c1d6,0xca62c1d6,0xca62c1d6 //K_60_79+.asciz "SHA1 block transform for ARMv8, CRYPTOGAMS by \@dot-asm"+.align 2+#if !defined(__KERNELL__) && !defined(_WIN64)+.comm OPENSSL_armcap_P,4,4+.hidden OPENSSL_armcap_P+#endif+___+}}}++{ my %opcode = (+ "sha1c" => 0x5e000000, "sha1p" => 0x5e001000,+ "sha1m" => 0x5e002000, "sha1su0" => 0x5e003000,+ "sha1h" => 0x5e280800, "sha1su1" => 0x5e281800 );++ sub unsha1 {+ my ($mnemonic,$arg)=@_;++ $arg =~ m/[qv]([0-9]+)[^,]*,\s*[qv]([0-9]+)[^,]*(?:,\s*[qv]([0-9]+))?/o+ &&+ sprintf ".inst\t0x%08x\t//%s %s",+ $opcode{$mnemonic}|$1|($2<<5)|($3<<16),+ $mnemonic,$arg;+ }+}++foreach(split("\n",$code)) {++ s/\`([^\`]*)\`/eval($1)/geo;++ s/\b(sha1\w+)\s+([qv].*)/unsha1($1,$2)/geo;++ s/\.\w?32\b//o and s/\.16b/\.4s/go;+ m/(ld|st)1[^\[]+\[0\]/o and s/\.4s/\.s/go;++ print $_,"\n";+}++close STDOUT;
@@ -0,0 +1,2051 @@+// SPDX-License-Identifier: GPL-1.0+ OR BSD-3-Clause+//+// ====================================================================+// Written by Andy Polyakov, @dot-asm, initially for the OpenSSL+// project.+// ====================================================================+//+// SHA256/512 for ARMv8.+//+// Performance in cycles per processed byte and improvement coefficient+// over code generated with "default" compiler:+//+// SHA256-hw SHA256(*) SHA512+// Apple A7 1.97 10.5 (+33%) 6.73 (-1%(**))+// Apple A10 1.30 5.81+// Apple A12 1.31 5.06+// Apple A14/M1 1.30 8.19 (+14%) 2.24 (hw)+// Cortex-A53 2.38 15.5 (+115%) 10.0 (+150%(***))+// Cortex-A57 2.31 11.6 (+86%) 7.51 (+260%(***))+// Cortex-A76 1.60 9.5 6.05+// Cortex-X2 1.60 7.3 2.60 (hw)+// Cortex-X925 1.57 5.97 2.55 (hw)+// Denver 2.01 10.5 (+26%) 6.70 (+8%)+// X-Gene 20.0 (+100%) 12.8 (+300%(***))+// Mongoose 2.36 13.0 (+50%) 8.36 (+33%)+// Kryo 1.92 17.4 (+30%) 11.2 (+8%)+// ThunderX2 2.54 13.2 (+40%) 8.40 (+18%)+// Shapdragon X 1.40 7.43 2.23 (hw)+//+// (*) Software SHA256 results are of lesser relevance, presented+// mostly for informational purposes.+// (**) The result is a trade-off: it's possible to improve it by+// 10% (or by 1 cycle per round), but at the cost of 20% loss+// on Cortex-A53 (or by 4 cycles per round).+// (***) Super-impressive coefficients over gcc-generated code are+// indication of some compiler "pathology", most notably code+// generated with -mgeneral-regs-only is significantly faster+// and the gap is only 40-90%.+//+// October 2016.+//+// Originally it was reckoned that it makes no sense to implement NEON+// version of SHA256 for 64-bit processors. This is because performance+// improvement on most wide-spread Cortex-A5x processors was observed+// to be marginal, same on Cortex-A53 and ~10% on A57. But then it was+// observed that 32-bit NEON SHA256 performs significantly better than+// 64-bit scalar version on *some* of the more recent processors. As+// result 64-bit NEON version of SHA256 was added to provide best+// all-round performance. For example it executes ~30% faster on X-Gene+// and Mongoose. [For reference, NEON version of SHA512 is bound to+// deliver much less improvement, likely *negative* on Cortex-A5x.+// Which is why NEON support is limited to SHA256.]++#ifndef __KERNEL__+# include "arm_arch.h"++#endif++.text++.globl _crypton_sha256_asm_block_data_order++.align 6+_crypton_sha256_asm_block_data_order:+#ifndef __KERNEL__+ adrp x16,_crypton_armcap_P@PAGE+ ldr w16,[x16,_crypton_armcap_P@PAGEOFF]+ tst w16,#ARMV8_SHA256+ b.ne Lv8_entry+ tst w16,#ARMV7_NEON+ b.ne Lneon_entry+#endif+.long 0xd503233f // paciasp+ stp x29,x30,[sp,#-16*__SIZEOF_POINTER__]!+ add x29,sp,#0++ stp x19,x20,[sp,#2*__SIZEOF_POINTER__]+ stp x21,x22,[sp,#4*__SIZEOF_POINTER__]+ stp x23,x24,[sp,#6*__SIZEOF_POINTER__]+ stp x25,x26,[sp,#8*__SIZEOF_POINTER__]+ stp x27,x28,[sp,#10*__SIZEOF_POINTER__]+ sub sp,sp,#4*4++ ldp w20,w21,[x0] // load context+ ldp w22,w23,[x0,#2*4]+ lsl x2,x2,#6+ ldp w24,w25,[x0,#4*4]+ add x2,x1,x2+ ldp w26,w27,[x0,#6*4]+ adr x30,LK256+ stp x0,x2,[x29,#12*__SIZEOF_POINTER__]++Loop:+ ldp w3,w4,[x1],#2*4+ ldr w19,[x30],#4 // *K+++ eor w28,w21,w22 // magic seed+ str x1,[x29,#14*__SIZEOF_POINTER__]+#ifndef __AARCH64EB__+ rev w3,w3 // 0+#endif+ ror w16,w24,#6+ add w27,w27,w19 // h+=K[i]+ eor w6,w24,w24,ror#14+ and w17,w25,w24+ bic w19,w26,w24+ add w27,w27,w3 // h+=X[i]+ orr w17,w17,w19 // Ch(e,f,g)+ eor w19,w20,w21 // a^b, b^c in next round+ eor w16,w16,w6,ror#11 // Sigma1(e)+ ror w6,w20,#2+ add w27,w27,w17 // h+=Ch(e,f,g)+ eor w17,w20,w20,ror#9+ add w27,w27,w16 // h+=Sigma1(e)+ and w28,w28,w19 // (b^c)&=(a^b)+ add w23,w23,w27 // d+=h+ eor w28,w28,w21 // Maj(a,b,c)+ eor w17,w6,w17,ror#13 // Sigma0(a)+ add w27,w27,w28 // h+=Maj(a,b,c)+ ldr w28,[x30],#4 // *K++, w19 in next round+ //add w27,w27,w17 // h+=Sigma0(a)+#ifndef __AARCH64EB__+ rev w4,w4 // 1+#endif+ ldp w5,w6,[x1],#2*4+ add w27,w27,w17 // h+=Sigma0(a)+ ror w16,w23,#6+ add w26,w26,w28 // h+=K[i]+ eor w7,w23,w23,ror#14+ and w17,w24,w23+ bic w28,w25,w23+ add w26,w26,w4 // h+=X[i]+ orr w17,w17,w28 // Ch(e,f,g)+ eor w28,w27,w20 // a^b, b^c in next round+ eor w16,w16,w7,ror#11 // Sigma1(e)+ ror w7,w27,#2+ add w26,w26,w17 // h+=Ch(e,f,g)+ eor w17,w27,w27,ror#9+ add w26,w26,w16 // h+=Sigma1(e)+ and w19,w19,w28 // (b^c)&=(a^b)+ add w22,w22,w26 // d+=h+ eor w19,w19,w20 // Maj(a,b,c)+ eor w17,w7,w17,ror#13 // Sigma0(a)+ add w26,w26,w19 // h+=Maj(a,b,c)+ ldr w19,[x30],#4 // *K++, w28 in next round+ //add w26,w26,w17 // h+=Sigma0(a)+#ifndef __AARCH64EB__+ rev w5,w5 // 2+#endif+ add w26,w26,w17 // h+=Sigma0(a)+ ror w16,w22,#6+ add w25,w25,w19 // h+=K[i]+ eor w8,w22,w22,ror#14+ and w17,w23,w22+ bic w19,w24,w22+ add w25,w25,w5 // h+=X[i]+ orr w17,w17,w19 // Ch(e,f,g)+ eor w19,w26,w27 // a^b, b^c in next round+ eor w16,w16,w8,ror#11 // Sigma1(e)+ ror w8,w26,#2+ add w25,w25,w17 // h+=Ch(e,f,g)+ eor w17,w26,w26,ror#9+ add w25,w25,w16 // h+=Sigma1(e)+ and w28,w28,w19 // (b^c)&=(a^b)+ add w21,w21,w25 // d+=h+ eor w28,w28,w27 // Maj(a,b,c)+ eor w17,w8,w17,ror#13 // Sigma0(a)+ add w25,w25,w28 // h+=Maj(a,b,c)+ ldr w28,[x30],#4 // *K++, w19 in next round+ //add w25,w25,w17 // h+=Sigma0(a)+#ifndef __AARCH64EB__+ rev w6,w6 // 3+#endif+ ldp w7,w8,[x1],#2*4+ add w25,w25,w17 // h+=Sigma0(a)+ ror w16,w21,#6+ add w24,w24,w28 // h+=K[i]+ eor w9,w21,w21,ror#14+ and w17,w22,w21+ bic w28,w23,w21+ add w24,w24,w6 // h+=X[i]+ orr w17,w17,w28 // Ch(e,f,g)+ eor w28,w25,w26 // a^b, b^c in next round+ eor w16,w16,w9,ror#11 // Sigma1(e)+ ror w9,w25,#2+ add w24,w24,w17 // h+=Ch(e,f,g)+ eor w17,w25,w25,ror#9+ add w24,w24,w16 // h+=Sigma1(e)+ and w19,w19,w28 // (b^c)&=(a^b)+ add w20,w20,w24 // d+=h+ eor w19,w19,w26 // Maj(a,b,c)+ eor w17,w9,w17,ror#13 // Sigma0(a)+ add w24,w24,w19 // h+=Maj(a,b,c)+ ldr w19,[x30],#4 // *K++, w28 in next round+ //add w24,w24,w17 // h+=Sigma0(a)+#ifndef __AARCH64EB__+ rev w7,w7 // 4+#endif+ add w24,w24,w17 // h+=Sigma0(a)+ ror w16,w20,#6+ add w23,w23,w19 // h+=K[i]+ eor w10,w20,w20,ror#14+ and w17,w21,w20+ bic w19,w22,w20+ add w23,w23,w7 // h+=X[i]+ orr w17,w17,w19 // Ch(e,f,g)+ eor w19,w24,w25 // a^b, b^c in next round+ eor w16,w16,w10,ror#11 // Sigma1(e)+ ror w10,w24,#2+ add w23,w23,w17 // h+=Ch(e,f,g)+ eor w17,w24,w24,ror#9+ add w23,w23,w16 // h+=Sigma1(e)+ and w28,w28,w19 // (b^c)&=(a^b)+ add w27,w27,w23 // d+=h+ eor w28,w28,w25 // Maj(a,b,c)+ eor w17,w10,w17,ror#13 // Sigma0(a)+ add w23,w23,w28 // h+=Maj(a,b,c)+ ldr w28,[x30],#4 // *K++, w19 in next round+ //add w23,w23,w17 // h+=Sigma0(a)+#ifndef __AARCH64EB__+ rev w8,w8 // 5+#endif+ ldp w9,w10,[x1],#2*4+ add w23,w23,w17 // h+=Sigma0(a)+ ror w16,w27,#6+ add w22,w22,w28 // h+=K[i]+ eor w11,w27,w27,ror#14+ and w17,w20,w27+ bic w28,w21,w27+ add w22,w22,w8 // h+=X[i]+ orr w17,w17,w28 // Ch(e,f,g)+ eor w28,w23,w24 // a^b, b^c in next round+ eor w16,w16,w11,ror#11 // Sigma1(e)+ ror w11,w23,#2+ add w22,w22,w17 // h+=Ch(e,f,g)+ eor w17,w23,w23,ror#9+ add w22,w22,w16 // h+=Sigma1(e)+ and w19,w19,w28 // (b^c)&=(a^b)+ add w26,w26,w22 // d+=h+ eor w19,w19,w24 // Maj(a,b,c)+ eor w17,w11,w17,ror#13 // Sigma0(a)+ add w22,w22,w19 // h+=Maj(a,b,c)+ ldr w19,[x30],#4 // *K++, w28 in next round+ //add w22,w22,w17 // h+=Sigma0(a)+#ifndef __AARCH64EB__+ rev w9,w9 // 6+#endif+ add w22,w22,w17 // h+=Sigma0(a)+ ror w16,w26,#6+ add w21,w21,w19 // h+=K[i]+ eor w12,w26,w26,ror#14+ and w17,w27,w26+ bic w19,w20,w26+ add w21,w21,w9 // h+=X[i]+ orr w17,w17,w19 // Ch(e,f,g)+ eor w19,w22,w23 // a^b, b^c in next round+ eor w16,w16,w12,ror#11 // Sigma1(e)+ ror w12,w22,#2+ add w21,w21,w17 // h+=Ch(e,f,g)+ eor w17,w22,w22,ror#9+ add w21,w21,w16 // h+=Sigma1(e)+ and w28,w28,w19 // (b^c)&=(a^b)+ add w25,w25,w21 // d+=h+ eor w28,w28,w23 // Maj(a,b,c)+ eor w17,w12,w17,ror#13 // Sigma0(a)+ add w21,w21,w28 // h+=Maj(a,b,c)+ ldr w28,[x30],#4 // *K++, w19 in next round+ //add w21,w21,w17 // h+=Sigma0(a)+#ifndef __AARCH64EB__+ rev w10,w10 // 7+#endif+ ldp w11,w12,[x1],#2*4+ add w21,w21,w17 // h+=Sigma0(a)+ ror w16,w25,#6+ add w20,w20,w28 // h+=K[i]+ eor w13,w25,w25,ror#14+ and w17,w26,w25+ bic w28,w27,w25+ add w20,w20,w10 // h+=X[i]+ orr w17,w17,w28 // Ch(e,f,g)+ eor w28,w21,w22 // a^b, b^c in next round+ eor w16,w16,w13,ror#11 // Sigma1(e)+ ror w13,w21,#2+ add w20,w20,w17 // h+=Ch(e,f,g)+ eor w17,w21,w21,ror#9+ add w20,w20,w16 // h+=Sigma1(e)+ and w19,w19,w28 // (b^c)&=(a^b)+ add w24,w24,w20 // d+=h+ eor w19,w19,w22 // Maj(a,b,c)+ eor w17,w13,w17,ror#13 // Sigma0(a)+ add w20,w20,w19 // h+=Maj(a,b,c)+ ldr w19,[x30],#4 // *K++, w28 in next round+ //add w20,w20,w17 // h+=Sigma0(a)+#ifndef __AARCH64EB__+ rev w11,w11 // 8+#endif+ add w20,w20,w17 // h+=Sigma0(a)+ ror w16,w24,#6+ add w27,w27,w19 // h+=K[i]+ eor w14,w24,w24,ror#14+ and w17,w25,w24+ bic w19,w26,w24+ add w27,w27,w11 // h+=X[i]+ orr w17,w17,w19 // Ch(e,f,g)+ eor w19,w20,w21 // a^b, b^c in next round+ eor w16,w16,w14,ror#11 // Sigma1(e)+ ror w14,w20,#2+ add w27,w27,w17 // h+=Ch(e,f,g)+ eor w17,w20,w20,ror#9+ add w27,w27,w16 // h+=Sigma1(e)+ and w28,w28,w19 // (b^c)&=(a^b)+ add w23,w23,w27 // d+=h+ eor w28,w28,w21 // Maj(a,b,c)+ eor w17,w14,w17,ror#13 // Sigma0(a)+ add w27,w27,w28 // h+=Maj(a,b,c)+ ldr w28,[x30],#4 // *K++, w19 in next round+ //add w27,w27,w17 // h+=Sigma0(a)+#ifndef __AARCH64EB__+ rev w12,w12 // 9+#endif+ ldp w13,w14,[x1],#2*4+ add w27,w27,w17 // h+=Sigma0(a)+ ror w16,w23,#6+ add w26,w26,w28 // h+=K[i]+ eor w15,w23,w23,ror#14+ and w17,w24,w23+ bic w28,w25,w23+ add w26,w26,w12 // h+=X[i]+ orr w17,w17,w28 // Ch(e,f,g)+ eor w28,w27,w20 // a^b, b^c in next round+ eor w16,w16,w15,ror#11 // Sigma1(e)+ ror w15,w27,#2+ add w26,w26,w17 // h+=Ch(e,f,g)+ eor w17,w27,w27,ror#9+ add w26,w26,w16 // h+=Sigma1(e)+ and w19,w19,w28 // (b^c)&=(a^b)+ add w22,w22,w26 // d+=h+ eor w19,w19,w20 // Maj(a,b,c)+ eor w17,w15,w17,ror#13 // Sigma0(a)+ add w26,w26,w19 // h+=Maj(a,b,c)+ ldr w19,[x30],#4 // *K++, w28 in next round+ //add w26,w26,w17 // h+=Sigma0(a)+#ifndef __AARCH64EB__+ rev w13,w13 // 10+#endif+ add w26,w26,w17 // h+=Sigma0(a)+ ror w16,w22,#6+ add w25,w25,w19 // h+=K[i]+ eor w0,w22,w22,ror#14+ and w17,w23,w22+ bic w19,w24,w22+ add w25,w25,w13 // h+=X[i]+ orr w17,w17,w19 // Ch(e,f,g)+ eor w19,w26,w27 // a^b, b^c in next round+ eor w16,w16,w0,ror#11 // Sigma1(e)+ ror w0,w26,#2+ add w25,w25,w17 // h+=Ch(e,f,g)+ eor w17,w26,w26,ror#9+ add w25,w25,w16 // h+=Sigma1(e)+ and w28,w28,w19 // (b^c)&=(a^b)+ add w21,w21,w25 // d+=h+ eor w28,w28,w27 // Maj(a,b,c)+ eor w17,w0,w17,ror#13 // Sigma0(a)+ add w25,w25,w28 // h+=Maj(a,b,c)+ ldr w28,[x30],#4 // *K++, w19 in next round+ //add w25,w25,w17 // h+=Sigma0(a)+#ifndef __AARCH64EB__+ rev w14,w14 // 11+#endif+ ldp w15,w0,[x1],#2*4+ add w25,w25,w17 // h+=Sigma0(a)+ str w6,[sp,#12]+ ror w16,w21,#6+ add w24,w24,w28 // h+=K[i]+ eor w6,w21,w21,ror#14+ and w17,w22,w21+ bic w28,w23,w21+ add w24,w24,w14 // h+=X[i]+ orr w17,w17,w28 // Ch(e,f,g)+ eor w28,w25,w26 // a^b, b^c in next round+ eor w16,w16,w6,ror#11 // Sigma1(e)+ ror w6,w25,#2+ add w24,w24,w17 // h+=Ch(e,f,g)+ eor w17,w25,w25,ror#9+ add w24,w24,w16 // h+=Sigma1(e)+ and w19,w19,w28 // (b^c)&=(a^b)+ add w20,w20,w24 // d+=h+ eor w19,w19,w26 // Maj(a,b,c)+ eor w17,w6,w17,ror#13 // Sigma0(a)+ add w24,w24,w19 // h+=Maj(a,b,c)+ ldr w19,[x30],#4 // *K++, w28 in next round+ //add w24,w24,w17 // h+=Sigma0(a)+#ifndef __AARCH64EB__+ rev w15,w15 // 12+#endif+ add w24,w24,w17 // h+=Sigma0(a)+ str w7,[sp,#0]+ ror w16,w20,#6+ add w23,w23,w19 // h+=K[i]+ eor w7,w20,w20,ror#14+ and w17,w21,w20+ bic w19,w22,w20+ add w23,w23,w15 // h+=X[i]+ orr w17,w17,w19 // Ch(e,f,g)+ eor w19,w24,w25 // a^b, b^c in next round+ eor w16,w16,w7,ror#11 // Sigma1(e)+ ror w7,w24,#2+ add w23,w23,w17 // h+=Ch(e,f,g)+ eor w17,w24,w24,ror#9+ add w23,w23,w16 // h+=Sigma1(e)+ and w28,w28,w19 // (b^c)&=(a^b)+ add w27,w27,w23 // d+=h+ eor w28,w28,w25 // Maj(a,b,c)+ eor w17,w7,w17,ror#13 // Sigma0(a)+ add w23,w23,w28 // h+=Maj(a,b,c)+ ldr w28,[x30],#4 // *K++, w19 in next round+ //add w23,w23,w17 // h+=Sigma0(a)+#ifndef __AARCH64EB__+ rev w0,w0 // 13+#endif+ ldp w1,w2,[x1]+ add w23,w23,w17 // h+=Sigma0(a)+ str w8,[sp,#4]+ ror w16,w27,#6+ add w22,w22,w28 // h+=K[i]+ eor w8,w27,w27,ror#14+ and w17,w20,w27+ bic w28,w21,w27+ add w22,w22,w0 // h+=X[i]+ orr w17,w17,w28 // Ch(e,f,g)+ eor w28,w23,w24 // a^b, b^c in next round+ eor w16,w16,w8,ror#11 // Sigma1(e)+ ror w8,w23,#2+ add w22,w22,w17 // h+=Ch(e,f,g)+ eor w17,w23,w23,ror#9+ add w22,w22,w16 // h+=Sigma1(e)+ and w19,w19,w28 // (b^c)&=(a^b)+ add w26,w26,w22 // d+=h+ eor w19,w19,w24 // Maj(a,b,c)+ eor w17,w8,w17,ror#13 // Sigma0(a)+ add w22,w22,w19 // h+=Maj(a,b,c)+ ldr w19,[x30],#4 // *K++, w28 in next round+ //add w22,w22,w17 // h+=Sigma0(a)+#ifndef __AARCH64EB__+ rev w1,w1 // 14+#endif+ ldr w6,[sp,#12]+ add w22,w22,w17 // h+=Sigma0(a)+ str w9,[sp,#8]+ ror w16,w26,#6+ add w21,w21,w19 // h+=K[i]+ eor w9,w26,w26,ror#14+ and w17,w27,w26+ bic w19,w20,w26+ add w21,w21,w1 // h+=X[i]+ orr w17,w17,w19 // Ch(e,f,g)+ eor w19,w22,w23 // a^b, b^c in next round+ eor w16,w16,w9,ror#11 // Sigma1(e)+ ror w9,w22,#2+ add w21,w21,w17 // h+=Ch(e,f,g)+ eor w17,w22,w22,ror#9+ add w21,w21,w16 // h+=Sigma1(e)+ and w28,w28,w19 // (b^c)&=(a^b)+ add w25,w25,w21 // d+=h+ eor w28,w28,w23 // Maj(a,b,c)+ eor w17,w9,w17,ror#13 // Sigma0(a)+ add w21,w21,w28 // h+=Maj(a,b,c)+ ldr w28,[x30],#4 // *K++, w19 in next round+ //add w21,w21,w17 // h+=Sigma0(a)+#ifndef __AARCH64EB__+ rev w2,w2 // 15+#endif+ ldr w7,[sp,#0]+ add w21,w21,w17 // h+=Sigma0(a)+ str w10,[sp,#12]+ ror w16,w25,#6+ add w20,w20,w28 // h+=K[i]+ ror w9,w4,#7+ and w17,w26,w25+ ror w8,w1,#17+ bic w28,w27,w25+ ror w10,w21,#2+ add w20,w20,w2 // h+=X[i]+ eor w16,w16,w25,ror#11+ eor w9,w9,w4,ror#18+ orr w17,w17,w28 // Ch(e,f,g)+ eor w28,w21,w22 // a^b, b^c in next round+ eor w16,w16,w25,ror#25 // Sigma1(e)+ eor w10,w10,w21,ror#13+ add w20,w20,w17 // h+=Ch(e,f,g)+ and w19,w19,w28 // (b^c)&=(a^b)+ eor w8,w8,w1,ror#19+ eor w9,w9,w4,lsr#3 // sigma0(X[i+1])+ add w20,w20,w16 // h+=Sigma1(e)+ eor w19,w19,w22 // Maj(a,b,c)+ eor w17,w10,w21,ror#22 // Sigma0(a)+ eor w8,w8,w1,lsr#10 // sigma1(X[i+14])+ add w3,w3,w12+ add w24,w24,w20 // d+=h+ add w20,w20,w19 // h+=Maj(a,b,c)+ ldr w19,[x30],#4 // *K++, w28 in next round+ add w3,w3,w9+ add w20,w20,w17 // h+=Sigma0(a)+ add w3,w3,w8+Loop_16_xx:+ ldr w8,[sp,#4]+ str w11,[sp,#0]+ ror w16,w24,#6+ add w27,w27,w19 // h+=K[i]+ ror w10,w5,#7+ and w17,w25,w24+ ror w9,w2,#17+ bic w19,w26,w24+ ror w11,w20,#2+ add w27,w27,w3 // h+=X[i]+ eor w16,w16,w24,ror#11+ eor w10,w10,w5,ror#18+ orr w17,w17,w19 // Ch(e,f,g)+ eor w19,w20,w21 // a^b, b^c in next round+ eor w16,w16,w24,ror#25 // Sigma1(e)+ eor w11,w11,w20,ror#13+ add w27,w27,w17 // h+=Ch(e,f,g)+ and w28,w28,w19 // (b^c)&=(a^b)+ eor w9,w9,w2,ror#19+ eor w10,w10,w5,lsr#3 // sigma0(X[i+1])+ add w27,w27,w16 // h+=Sigma1(e)+ eor w28,w28,w21 // Maj(a,b,c)+ eor w17,w11,w20,ror#22 // Sigma0(a)+ eor w9,w9,w2,lsr#10 // sigma1(X[i+14])+ add w4,w4,w13+ add w23,w23,w27 // d+=h+ add w27,w27,w28 // h+=Maj(a,b,c)+ ldr w28,[x30],#4 // *K++, w19 in next round+ add w4,w4,w10+ add w27,w27,w17 // h+=Sigma0(a)+ add w4,w4,w9+ ldr w9,[sp,#8]+ str w12,[sp,#4]+ ror w16,w23,#6+ add w26,w26,w28 // h+=K[i]+ ror w11,w6,#7+ and w17,w24,w23+ ror w10,w3,#17+ bic w28,w25,w23+ ror w12,w27,#2+ add w26,w26,w4 // h+=X[i]+ eor w16,w16,w23,ror#11+ eor w11,w11,w6,ror#18+ orr w17,w17,w28 // Ch(e,f,g)+ eor w28,w27,w20 // a^b, b^c in next round+ eor w16,w16,w23,ror#25 // Sigma1(e)+ eor w12,w12,w27,ror#13+ add w26,w26,w17 // h+=Ch(e,f,g)+ and w19,w19,w28 // (b^c)&=(a^b)+ eor w10,w10,w3,ror#19+ eor w11,w11,w6,lsr#3 // sigma0(X[i+1])+ add w26,w26,w16 // h+=Sigma1(e)+ eor w19,w19,w20 // Maj(a,b,c)+ eor w17,w12,w27,ror#22 // Sigma0(a)+ eor w10,w10,w3,lsr#10 // sigma1(X[i+14])+ add w5,w5,w14+ add w22,w22,w26 // d+=h+ add w26,w26,w19 // h+=Maj(a,b,c)+ ldr w19,[x30],#4 // *K++, w28 in next round+ add w5,w5,w11+ add w26,w26,w17 // h+=Sigma0(a)+ add w5,w5,w10+ ldr w10,[sp,#12]+ str w13,[sp,#8]+ ror w16,w22,#6+ add w25,w25,w19 // h+=K[i]+ ror w12,w7,#7+ and w17,w23,w22+ ror w11,w4,#17+ bic w19,w24,w22+ ror w13,w26,#2+ add w25,w25,w5 // h+=X[i]+ eor w16,w16,w22,ror#11+ eor w12,w12,w7,ror#18+ orr w17,w17,w19 // Ch(e,f,g)+ eor w19,w26,w27 // a^b, b^c in next round+ eor w16,w16,w22,ror#25 // Sigma1(e)+ eor w13,w13,w26,ror#13+ add w25,w25,w17 // h+=Ch(e,f,g)+ and w28,w28,w19 // (b^c)&=(a^b)+ eor w11,w11,w4,ror#19+ eor w12,w12,w7,lsr#3 // sigma0(X[i+1])+ add w25,w25,w16 // h+=Sigma1(e)+ eor w28,w28,w27 // Maj(a,b,c)+ eor w17,w13,w26,ror#22 // Sigma0(a)+ eor w11,w11,w4,lsr#10 // sigma1(X[i+14])+ add w6,w6,w15+ add w21,w21,w25 // d+=h+ add w25,w25,w28 // h+=Maj(a,b,c)+ ldr w28,[x30],#4 // *K++, w19 in next round+ add w6,w6,w12+ add w25,w25,w17 // h+=Sigma0(a)+ add w6,w6,w11+ ldr w11,[sp,#0]+ str w14,[sp,#12]+ ror w16,w21,#6+ add w24,w24,w28 // h+=K[i]+ ror w13,w8,#7+ and w17,w22,w21+ ror w12,w5,#17+ bic w28,w23,w21+ ror w14,w25,#2+ add w24,w24,w6 // h+=X[i]+ eor w16,w16,w21,ror#11+ eor w13,w13,w8,ror#18+ orr w17,w17,w28 // Ch(e,f,g)+ eor w28,w25,w26 // a^b, b^c in next round+ eor w16,w16,w21,ror#25 // Sigma1(e)+ eor w14,w14,w25,ror#13+ add w24,w24,w17 // h+=Ch(e,f,g)+ and w19,w19,w28 // (b^c)&=(a^b)+ eor w12,w12,w5,ror#19+ eor w13,w13,w8,lsr#3 // sigma0(X[i+1])+ add w24,w24,w16 // h+=Sigma1(e)+ eor w19,w19,w26 // Maj(a,b,c)+ eor w17,w14,w25,ror#22 // Sigma0(a)+ eor w12,w12,w5,lsr#10 // sigma1(X[i+14])+ add w7,w7,w0+ add w20,w20,w24 // d+=h+ add w24,w24,w19 // h+=Maj(a,b,c)+ ldr w19,[x30],#4 // *K++, w28 in next round+ add w7,w7,w13+ add w24,w24,w17 // h+=Sigma0(a)+ add w7,w7,w12+ ldr w12,[sp,#4]+ str w15,[sp,#0]+ ror w16,w20,#6+ add w23,w23,w19 // h+=K[i]+ ror w14,w9,#7+ and w17,w21,w20+ ror w13,w6,#17+ bic w19,w22,w20+ ror w15,w24,#2+ add w23,w23,w7 // h+=X[i]+ eor w16,w16,w20,ror#11+ eor w14,w14,w9,ror#18+ orr w17,w17,w19 // Ch(e,f,g)+ eor w19,w24,w25 // a^b, b^c in next round+ eor w16,w16,w20,ror#25 // Sigma1(e)+ eor w15,w15,w24,ror#13+ add w23,w23,w17 // h+=Ch(e,f,g)+ and w28,w28,w19 // (b^c)&=(a^b)+ eor w13,w13,w6,ror#19+ eor w14,w14,w9,lsr#3 // sigma0(X[i+1])+ add w23,w23,w16 // h+=Sigma1(e)+ eor w28,w28,w25 // Maj(a,b,c)+ eor w17,w15,w24,ror#22 // Sigma0(a)+ eor w13,w13,w6,lsr#10 // sigma1(X[i+14])+ add w8,w8,w1+ add w27,w27,w23 // d+=h+ add w23,w23,w28 // h+=Maj(a,b,c)+ ldr w28,[x30],#4 // *K++, w19 in next round+ add w8,w8,w14+ add w23,w23,w17 // h+=Sigma0(a)+ add w8,w8,w13+ ldr w13,[sp,#8]+ str w0,[sp,#4]+ ror w16,w27,#6+ add w22,w22,w28 // h+=K[i]+ ror w15,w10,#7+ and w17,w20,w27+ ror w14,w7,#17+ bic w28,w21,w27+ ror w0,w23,#2+ add w22,w22,w8 // h+=X[i]+ eor w16,w16,w27,ror#11+ eor w15,w15,w10,ror#18+ orr w17,w17,w28 // Ch(e,f,g)+ eor w28,w23,w24 // a^b, b^c in next round+ eor w16,w16,w27,ror#25 // Sigma1(e)+ eor w0,w0,w23,ror#13+ add w22,w22,w17 // h+=Ch(e,f,g)+ and w19,w19,w28 // (b^c)&=(a^b)+ eor w14,w14,w7,ror#19+ eor w15,w15,w10,lsr#3 // sigma0(X[i+1])+ add w22,w22,w16 // h+=Sigma1(e)+ eor w19,w19,w24 // Maj(a,b,c)+ eor w17,w0,w23,ror#22 // Sigma0(a)+ eor w14,w14,w7,lsr#10 // sigma1(X[i+14])+ add w9,w9,w2+ add w26,w26,w22 // d+=h+ add w22,w22,w19 // h+=Maj(a,b,c)+ ldr w19,[x30],#4 // *K++, w28 in next round+ add w9,w9,w15+ add w22,w22,w17 // h+=Sigma0(a)+ add w9,w9,w14+ ldr w14,[sp,#12]+ str w1,[sp,#8]+ ror w16,w26,#6+ add w21,w21,w19 // h+=K[i]+ ror w0,w11,#7+ and w17,w27,w26+ ror w15,w8,#17+ bic w19,w20,w26+ ror w1,w22,#2+ add w21,w21,w9 // h+=X[i]+ eor w16,w16,w26,ror#11+ eor w0,w0,w11,ror#18+ orr w17,w17,w19 // Ch(e,f,g)+ eor w19,w22,w23 // a^b, b^c in next round+ eor w16,w16,w26,ror#25 // Sigma1(e)+ eor w1,w1,w22,ror#13+ add w21,w21,w17 // h+=Ch(e,f,g)+ and w28,w28,w19 // (b^c)&=(a^b)+ eor w15,w15,w8,ror#19+ eor w0,w0,w11,lsr#3 // sigma0(X[i+1])+ add w21,w21,w16 // h+=Sigma1(e)+ eor w28,w28,w23 // Maj(a,b,c)+ eor w17,w1,w22,ror#22 // Sigma0(a)+ eor w15,w15,w8,lsr#10 // sigma1(X[i+14])+ add w10,w10,w3+ add w25,w25,w21 // d+=h+ add w21,w21,w28 // h+=Maj(a,b,c)+ ldr w28,[x30],#4 // *K++, w19 in next round+ add w10,w10,w0+ add w21,w21,w17 // h+=Sigma0(a)+ add w10,w10,w15+ ldr w15,[sp,#0]+ str w2,[sp,#12]+ ror w16,w25,#6+ add w20,w20,w28 // h+=K[i]+ ror w1,w12,#7+ and w17,w26,w25+ ror w0,w9,#17+ bic w28,w27,w25+ ror w2,w21,#2+ add w20,w20,w10 // h+=X[i]+ eor w16,w16,w25,ror#11+ eor w1,w1,w12,ror#18+ orr w17,w17,w28 // Ch(e,f,g)+ eor w28,w21,w22 // a^b, b^c in next round+ eor w16,w16,w25,ror#25 // Sigma1(e)+ eor w2,w2,w21,ror#13+ add w20,w20,w17 // h+=Ch(e,f,g)+ and w19,w19,w28 // (b^c)&=(a^b)+ eor w0,w0,w9,ror#19+ eor w1,w1,w12,lsr#3 // sigma0(X[i+1])+ add w20,w20,w16 // h+=Sigma1(e)+ eor w19,w19,w22 // Maj(a,b,c)+ eor w17,w2,w21,ror#22 // Sigma0(a)+ eor w0,w0,w9,lsr#10 // sigma1(X[i+14])+ add w11,w11,w4+ add w24,w24,w20 // d+=h+ add w20,w20,w19 // h+=Maj(a,b,c)+ ldr w19,[x30],#4 // *K++, w28 in next round+ add w11,w11,w1+ add w20,w20,w17 // h+=Sigma0(a)+ add w11,w11,w0+ ldr w0,[sp,#4]+ str w3,[sp,#0]+ ror w16,w24,#6+ add w27,w27,w19 // h+=K[i]+ ror w2,w13,#7+ and w17,w25,w24+ ror w1,w10,#17+ bic w19,w26,w24+ ror w3,w20,#2+ add w27,w27,w11 // h+=X[i]+ eor w16,w16,w24,ror#11+ eor w2,w2,w13,ror#18+ orr w17,w17,w19 // Ch(e,f,g)+ eor w19,w20,w21 // a^b, b^c in next round+ eor w16,w16,w24,ror#25 // Sigma1(e)+ eor w3,w3,w20,ror#13+ add w27,w27,w17 // h+=Ch(e,f,g)+ and w28,w28,w19 // (b^c)&=(a^b)+ eor w1,w1,w10,ror#19+ eor w2,w2,w13,lsr#3 // sigma0(X[i+1])+ add w27,w27,w16 // h+=Sigma1(e)+ eor w28,w28,w21 // Maj(a,b,c)+ eor w17,w3,w20,ror#22 // Sigma0(a)+ eor w1,w1,w10,lsr#10 // sigma1(X[i+14])+ add w12,w12,w5+ add w23,w23,w27 // d+=h+ add w27,w27,w28 // h+=Maj(a,b,c)+ ldr w28,[x30],#4 // *K++, w19 in next round+ add w12,w12,w2+ add w27,w27,w17 // h+=Sigma0(a)+ add w12,w12,w1+ ldr w1,[sp,#8]+ str w4,[sp,#4]+ ror w16,w23,#6+ add w26,w26,w28 // h+=K[i]+ ror w3,w14,#7+ and w17,w24,w23+ ror w2,w11,#17+ bic w28,w25,w23+ ror w4,w27,#2+ add w26,w26,w12 // h+=X[i]+ eor w16,w16,w23,ror#11+ eor w3,w3,w14,ror#18+ orr w17,w17,w28 // Ch(e,f,g)+ eor w28,w27,w20 // a^b, b^c in next round+ eor w16,w16,w23,ror#25 // Sigma1(e)+ eor w4,w4,w27,ror#13+ add w26,w26,w17 // h+=Ch(e,f,g)+ and w19,w19,w28 // (b^c)&=(a^b)+ eor w2,w2,w11,ror#19+ eor w3,w3,w14,lsr#3 // sigma0(X[i+1])+ add w26,w26,w16 // h+=Sigma1(e)+ eor w19,w19,w20 // Maj(a,b,c)+ eor w17,w4,w27,ror#22 // Sigma0(a)+ eor w2,w2,w11,lsr#10 // sigma1(X[i+14])+ add w13,w13,w6+ add w22,w22,w26 // d+=h+ add w26,w26,w19 // h+=Maj(a,b,c)+ ldr w19,[x30],#4 // *K++, w28 in next round+ add w13,w13,w3+ add w26,w26,w17 // h+=Sigma0(a)+ add w13,w13,w2+ ldr w2,[sp,#12]+ str w5,[sp,#8]+ ror w16,w22,#6+ add w25,w25,w19 // h+=K[i]+ ror w4,w15,#7+ and w17,w23,w22+ ror w3,w12,#17+ bic w19,w24,w22+ ror w5,w26,#2+ add w25,w25,w13 // h+=X[i]+ eor w16,w16,w22,ror#11+ eor w4,w4,w15,ror#18+ orr w17,w17,w19 // Ch(e,f,g)+ eor w19,w26,w27 // a^b, b^c in next round+ eor w16,w16,w22,ror#25 // Sigma1(e)+ eor w5,w5,w26,ror#13+ add w25,w25,w17 // h+=Ch(e,f,g)+ and w28,w28,w19 // (b^c)&=(a^b)+ eor w3,w3,w12,ror#19+ eor w4,w4,w15,lsr#3 // sigma0(X[i+1])+ add w25,w25,w16 // h+=Sigma1(e)+ eor w28,w28,w27 // Maj(a,b,c)+ eor w17,w5,w26,ror#22 // Sigma0(a)+ eor w3,w3,w12,lsr#10 // sigma1(X[i+14])+ add w14,w14,w7+ add w21,w21,w25 // d+=h+ add w25,w25,w28 // h+=Maj(a,b,c)+ ldr w28,[x30],#4 // *K++, w19 in next round+ add w14,w14,w4+ add w25,w25,w17 // h+=Sigma0(a)+ add w14,w14,w3+ ldr w3,[sp,#0]+ str w6,[sp,#12]+ ror w16,w21,#6+ add w24,w24,w28 // h+=K[i]+ ror w5,w0,#7+ and w17,w22,w21+ ror w4,w13,#17+ bic w28,w23,w21+ ror w6,w25,#2+ add w24,w24,w14 // h+=X[i]+ eor w16,w16,w21,ror#11+ eor w5,w5,w0,ror#18+ orr w17,w17,w28 // Ch(e,f,g)+ eor w28,w25,w26 // a^b, b^c in next round+ eor w16,w16,w21,ror#25 // Sigma1(e)+ eor w6,w6,w25,ror#13+ add w24,w24,w17 // h+=Ch(e,f,g)+ and w19,w19,w28 // (b^c)&=(a^b)+ eor w4,w4,w13,ror#19+ eor w5,w5,w0,lsr#3 // sigma0(X[i+1])+ add w24,w24,w16 // h+=Sigma1(e)+ eor w19,w19,w26 // Maj(a,b,c)+ eor w17,w6,w25,ror#22 // Sigma0(a)+ eor w4,w4,w13,lsr#10 // sigma1(X[i+14])+ add w15,w15,w8+ add w20,w20,w24 // d+=h+ add w24,w24,w19 // h+=Maj(a,b,c)+ ldr w19,[x30],#4 // *K++, w28 in next round+ add w15,w15,w5+ add w24,w24,w17 // h+=Sigma0(a)+ add w15,w15,w4+ ldr w4,[sp,#4]+ str w7,[sp,#0]+ ror w16,w20,#6+ add w23,w23,w19 // h+=K[i]+ ror w6,w1,#7+ and w17,w21,w20+ ror w5,w14,#17+ bic w19,w22,w20+ ror w7,w24,#2+ add w23,w23,w15 // h+=X[i]+ eor w16,w16,w20,ror#11+ eor w6,w6,w1,ror#18+ orr w17,w17,w19 // Ch(e,f,g)+ eor w19,w24,w25 // a^b, b^c in next round+ eor w16,w16,w20,ror#25 // Sigma1(e)+ eor w7,w7,w24,ror#13+ add w23,w23,w17 // h+=Ch(e,f,g)+ and w28,w28,w19 // (b^c)&=(a^b)+ eor w5,w5,w14,ror#19+ eor w6,w6,w1,lsr#3 // sigma0(X[i+1])+ add w23,w23,w16 // h+=Sigma1(e)+ eor w28,w28,w25 // Maj(a,b,c)+ eor w17,w7,w24,ror#22 // Sigma0(a)+ eor w5,w5,w14,lsr#10 // sigma1(X[i+14])+ add w0,w0,w9+ add w27,w27,w23 // d+=h+ add w23,w23,w28 // h+=Maj(a,b,c)+ ldr w28,[x30],#4 // *K++, w19 in next round+ add w0,w0,w6+ add w23,w23,w17 // h+=Sigma0(a)+ add w0,w0,w5+ ldr w5,[sp,#8]+ str w8,[sp,#4]+ ror w16,w27,#6+ add w22,w22,w28 // h+=K[i]+ ror w7,w2,#7+ and w17,w20,w27+ ror w6,w15,#17+ bic w28,w21,w27+ ror w8,w23,#2+ add w22,w22,w0 // h+=X[i]+ eor w16,w16,w27,ror#11+ eor w7,w7,w2,ror#18+ orr w17,w17,w28 // Ch(e,f,g)+ eor w28,w23,w24 // a^b, b^c in next round+ eor w16,w16,w27,ror#25 // Sigma1(e)+ eor w8,w8,w23,ror#13+ add w22,w22,w17 // h+=Ch(e,f,g)+ and w19,w19,w28 // (b^c)&=(a^b)+ eor w6,w6,w15,ror#19+ eor w7,w7,w2,lsr#3 // sigma0(X[i+1])+ add w22,w22,w16 // h+=Sigma1(e)+ eor w19,w19,w24 // Maj(a,b,c)+ eor w17,w8,w23,ror#22 // Sigma0(a)+ eor w6,w6,w15,lsr#10 // sigma1(X[i+14])+ add w1,w1,w10+ add w26,w26,w22 // d+=h+ add w22,w22,w19 // h+=Maj(a,b,c)+ ldr w19,[x30],#4 // *K++, w28 in next round+ add w1,w1,w7+ add w22,w22,w17 // h+=Sigma0(a)+ add w1,w1,w6+ ldr w6,[sp,#12]+ str w9,[sp,#8]+ ror w16,w26,#6+ add w21,w21,w19 // h+=K[i]+ ror w8,w3,#7+ and w17,w27,w26+ ror w7,w0,#17+ bic w19,w20,w26+ ror w9,w22,#2+ add w21,w21,w1 // h+=X[i]+ eor w16,w16,w26,ror#11+ eor w8,w8,w3,ror#18+ orr w17,w17,w19 // Ch(e,f,g)+ eor w19,w22,w23 // a^b, b^c in next round+ eor w16,w16,w26,ror#25 // Sigma1(e)+ eor w9,w9,w22,ror#13+ add w21,w21,w17 // h+=Ch(e,f,g)+ and w28,w28,w19 // (b^c)&=(a^b)+ eor w7,w7,w0,ror#19+ eor w8,w8,w3,lsr#3 // sigma0(X[i+1])+ add w21,w21,w16 // h+=Sigma1(e)+ eor w28,w28,w23 // Maj(a,b,c)+ eor w17,w9,w22,ror#22 // Sigma0(a)+ eor w7,w7,w0,lsr#10 // sigma1(X[i+14])+ add w2,w2,w11+ add w25,w25,w21 // d+=h+ add w21,w21,w28 // h+=Maj(a,b,c)+ ldr w28,[x30],#4 // *K++, w19 in next round+ add w2,w2,w8+ add w21,w21,w17 // h+=Sigma0(a)+ add w2,w2,w7+ ldr w7,[sp,#0]+ str w10,[sp,#12]+ ror w16,w25,#6+ add w20,w20,w28 // h+=K[i]+ ror w9,w4,#7+ and w17,w26,w25+ ror w8,w1,#17+ bic w28,w27,w25+ ror w10,w21,#2+ add w20,w20,w2 // h+=X[i]+ eor w16,w16,w25,ror#11+ eor w9,w9,w4,ror#18+ orr w17,w17,w28 // Ch(e,f,g)+ eor w28,w21,w22 // a^b, b^c in next round+ eor w16,w16,w25,ror#25 // Sigma1(e)+ eor w10,w10,w21,ror#13+ add w20,w20,w17 // h+=Ch(e,f,g)+ and w19,w19,w28 // (b^c)&=(a^b)+ eor w8,w8,w1,ror#19+ eor w9,w9,w4,lsr#3 // sigma0(X[i+1])+ add w20,w20,w16 // h+=Sigma1(e)+ eor w19,w19,w22 // Maj(a,b,c)+ eor w17,w10,w21,ror#22 // Sigma0(a)+ eor w8,w8,w1,lsr#10 // sigma1(X[i+14])+ add w3,w3,w12+ add w24,w24,w20 // d+=h+ add w20,w20,w19 // h+=Maj(a,b,c)+ ldr w19,[x30],#4 // *K++, w28 in next round+ add w3,w3,w9+ add w20,w20,w17 // h+=Sigma0(a)+ add w3,w3,w8+ cbnz w19,Loop_16_xx++ ldp x0,x2,[x29,#12*__SIZEOF_POINTER__]+ ldr x1,[x29,#14*__SIZEOF_POINTER__]+ sub x30,x30,#260++ ldp w3,w4,[x0]+ ldp w5,w6,[x0,#2*4]+ add x1,x1,#14*4+ ldp w7,w8,[x0,#4*4]+ add w20,w20,w3+ ldp w9,w10,[x0,#6*4]+ add w21,w21,w4+ add w22,w22,w5+ add w23,w23,w6+ stp w20,w21,[x0]+ add w24,w24,w7+ add w25,w25,w8+ stp w22,w23,[x0,#2*4]+ add w26,w26,w9+ add w27,w27,w10+ cmp x1,x2+ stp w24,w25,[x0,#4*4]+ stp w26,w27,[x0,#6*4]+ b.ne Loop++ ldp x19,x20,[x29,#2*__SIZEOF_POINTER__]+ add sp,sp,#4*4+ ldp x21,x22,[x29,#4*__SIZEOF_POINTER__]+ ldp x23,x24,[x29,#6*__SIZEOF_POINTER__]+ ldp x25,x26,[x29,#8*__SIZEOF_POINTER__]+ ldp x27,x28,[x29,#10*__SIZEOF_POINTER__]+ ldp x29,x30,[sp],#16*__SIZEOF_POINTER__+.long 0xd50323bf // autiasp+ ret+++.align 6++LK256:+.long 0x428a2f98,0x71374491,0xb5c0fbcf,0xe9b5dba5+.long 0x3956c25b,0x59f111f1,0x923f82a4,0xab1c5ed5+.long 0xd807aa98,0x12835b01,0x243185be,0x550c7dc3+.long 0x72be5d74,0x80deb1fe,0x9bdc06a7,0xc19bf174+.long 0xe49b69c1,0xefbe4786,0x0fc19dc6,0x240ca1cc+.long 0x2de92c6f,0x4a7484aa,0x5cb0a9dc,0x76f988da+.long 0x983e5152,0xa831c66d,0xb00327c8,0xbf597fc7+.long 0xc6e00bf3,0xd5a79147,0x06ca6351,0x14292967+.long 0x27b70a85,0x2e1b2138,0x4d2c6dfc,0x53380d13+.long 0x650a7354,0x766a0abb,0x81c2c92e,0x92722c85+.long 0xa2bfe8a1,0xa81a664b,0xc24b8b70,0xc76c51a3+.long 0xd192e819,0xd6990624,0xf40e3585,0x106aa070+.long 0x19a4c116,0x1e376c08,0x2748774c,0x34b0bcb5+.long 0x391c0cb3,0x4ed8aa4a,0x5b9cca4f,0x682e6ff3+.long 0x748f82ee,0x78a5636f,0x84c87814,0x8cc70208+.long 0x90befffa,0xa4506ceb,0xbef9a3f7,0xc67178f2+.long 0 //terminator++.byte 83,72,65,50,53,54,32,98,108,111,99,107,32,116,114,97,110,115,102,111,114,109,32,102,111,114,32,65,82,77,118,56,44,32,67,82,89,80,84,79,71,65,77,83,32,98,121,32,64,100,111,116,45,97,115,109,0+.align 2+.align 2+#ifndef __KERNEL__++.align 6+crypton_sha256_asm_block_armv8:+Lv8_entry:+ stp x29,x30,[sp,#-2*__SIZEOF_POINTER__]!+ add x29,sp,#0++ ld1 {v0.4s,v1.4s},[x0]+ adr x3,LK256++Loop_hw:+ ld1 {v4.16b,v5.16b,v6.16b,v7.16b},[x1],#64+ sub x2,x2,#1+ ld1 {v16.4s},[x3],#16+ rev32 v4.16b,v4.16b+ rev32 v5.16b,v5.16b+ rev32 v6.16b,v6.16b+ rev32 v7.16b,v7.16b+ orr v18.16b,v0.16b,v0.16b // offload+ orr v19.16b,v1.16b,v1.16b+ ld1 {v17.4s},[x3],#16+ add v16.4s,v16.4s,v4.4s+.long 0x5e2828a4 //sha256su0 v4.16b,v5.16b+ orr v2.16b,v0.16b,v0.16b+.long 0x5e104020 //sha256h v0.16b,v1.16b,v16.4s+.long 0x5e105041 //sha256h2 v1.16b,v2.16b,v16.4s+.long 0x5e0760c4 //sha256su1 v4.16b,v6.16b,v7.16b+ ld1 {v16.4s},[x3],#16+ add v17.4s,v17.4s,v5.4s+.long 0x5e2828c5 //sha256su0 v5.16b,v6.16b+ orr v2.16b,v0.16b,v0.16b+.long 0x5e114020 //sha256h v0.16b,v1.16b,v17.4s+.long 0x5e115041 //sha256h2 v1.16b,v2.16b,v17.4s+.long 0x5e0460e5 //sha256su1 v5.16b,v7.16b,v4.16b+ ld1 {v17.4s},[x3],#16+ add v16.4s,v16.4s,v6.4s+.long 0x5e2828e6 //sha256su0 v6.16b,v7.16b+ orr v2.16b,v0.16b,v0.16b+.long 0x5e104020 //sha256h v0.16b,v1.16b,v16.4s+.long 0x5e105041 //sha256h2 v1.16b,v2.16b,v16.4s+.long 0x5e056086 //sha256su1 v6.16b,v4.16b,v5.16b+ ld1 {v16.4s},[x3],#16+ add v17.4s,v17.4s,v7.4s+.long 0x5e282887 //sha256su0 v7.16b,v4.16b+ orr v2.16b,v0.16b,v0.16b+.long 0x5e114020 //sha256h v0.16b,v1.16b,v17.4s+.long 0x5e115041 //sha256h2 v1.16b,v2.16b,v17.4s+.long 0x5e0660a7 //sha256su1 v7.16b,v5.16b,v6.16b+ ld1 {v17.4s},[x3],#16+ add v16.4s,v16.4s,v4.4s+.long 0x5e2828a4 //sha256su0 v4.16b,v5.16b+ orr v2.16b,v0.16b,v0.16b+.long 0x5e104020 //sha256h v0.16b,v1.16b,v16.4s+.long 0x5e105041 //sha256h2 v1.16b,v2.16b,v16.4s+.long 0x5e0760c4 //sha256su1 v4.16b,v6.16b,v7.16b+ ld1 {v16.4s},[x3],#16+ add v17.4s,v17.4s,v5.4s+.long 0x5e2828c5 //sha256su0 v5.16b,v6.16b+ orr v2.16b,v0.16b,v0.16b+.long 0x5e114020 //sha256h v0.16b,v1.16b,v17.4s+.long 0x5e115041 //sha256h2 v1.16b,v2.16b,v17.4s+.long 0x5e0460e5 //sha256su1 v5.16b,v7.16b,v4.16b+ ld1 {v17.4s},[x3],#16+ add v16.4s,v16.4s,v6.4s+.long 0x5e2828e6 //sha256su0 v6.16b,v7.16b+ orr v2.16b,v0.16b,v0.16b+.long 0x5e104020 //sha256h v0.16b,v1.16b,v16.4s+.long 0x5e105041 //sha256h2 v1.16b,v2.16b,v16.4s+.long 0x5e056086 //sha256su1 v6.16b,v4.16b,v5.16b+ ld1 {v16.4s},[x3],#16+ add v17.4s,v17.4s,v7.4s+.long 0x5e282887 //sha256su0 v7.16b,v4.16b+ orr v2.16b,v0.16b,v0.16b+.long 0x5e114020 //sha256h v0.16b,v1.16b,v17.4s+.long 0x5e115041 //sha256h2 v1.16b,v2.16b,v17.4s+.long 0x5e0660a7 //sha256su1 v7.16b,v5.16b,v6.16b+ ld1 {v17.4s},[x3],#16+ add v16.4s,v16.4s,v4.4s+.long 0x5e2828a4 //sha256su0 v4.16b,v5.16b+ orr v2.16b,v0.16b,v0.16b+.long 0x5e104020 //sha256h v0.16b,v1.16b,v16.4s+.long 0x5e105041 //sha256h2 v1.16b,v2.16b,v16.4s+.long 0x5e0760c4 //sha256su1 v4.16b,v6.16b,v7.16b+ ld1 {v16.4s},[x3],#16+ add v17.4s,v17.4s,v5.4s+.long 0x5e2828c5 //sha256su0 v5.16b,v6.16b+ orr v2.16b,v0.16b,v0.16b+.long 0x5e114020 //sha256h v0.16b,v1.16b,v17.4s+.long 0x5e115041 //sha256h2 v1.16b,v2.16b,v17.4s+.long 0x5e0460e5 //sha256su1 v5.16b,v7.16b,v4.16b+ ld1 {v17.4s},[x3],#16+ add v16.4s,v16.4s,v6.4s+.long 0x5e2828e6 //sha256su0 v6.16b,v7.16b+ orr v2.16b,v0.16b,v0.16b+.long 0x5e104020 //sha256h v0.16b,v1.16b,v16.4s+.long 0x5e105041 //sha256h2 v1.16b,v2.16b,v16.4s+.long 0x5e056086 //sha256su1 v6.16b,v4.16b,v5.16b+ ld1 {v16.4s},[x3],#16+ add v17.4s,v17.4s,v7.4s+.long 0x5e282887 //sha256su0 v7.16b,v4.16b+ orr v2.16b,v0.16b,v0.16b+.long 0x5e114020 //sha256h v0.16b,v1.16b,v17.4s+.long 0x5e115041 //sha256h2 v1.16b,v2.16b,v17.4s+.long 0x5e0660a7 //sha256su1 v7.16b,v5.16b,v6.16b+ ld1 {v17.4s},[x3],#16+ add v16.4s,v16.4s,v4.4s+ orr v2.16b,v0.16b,v0.16b+.long 0x5e104020 //sha256h v0.16b,v1.16b,v16.4s+.long 0x5e105041 //sha256h2 v1.16b,v2.16b,v16.4s++ ld1 {v16.4s},[x3],#16+ add v17.4s,v17.4s,v5.4s+ orr v2.16b,v0.16b,v0.16b+.long 0x5e114020 //sha256h v0.16b,v1.16b,v17.4s+.long 0x5e115041 //sha256h2 v1.16b,v2.16b,v17.4s++ ld1 {v17.4s},[x3]+ add v16.4s,v16.4s,v6.4s+ sub x3,x3,#64*4-16+ orr v2.16b,v0.16b,v0.16b+.long 0x5e104020 //sha256h v0.16b,v1.16b,v16.4s+.long 0x5e105041 //sha256h2 v1.16b,v2.16b,v16.4s++ add v17.4s,v17.4s,v7.4s+ orr v2.16b,v0.16b,v0.16b+.long 0x5e114020 //sha256h v0.16b,v1.16b,v17.4s+.long 0x5e115041 //sha256h2 v1.16b,v2.16b,v17.4s++ add v0.4s,v0.4s,v18.4s+ add v1.4s,v1.4s,v19.4s++ cbnz x2,Loop_hw++ st1 {v0.4s,v1.4s},[x0]++ ldr x29,[sp],#2*__SIZEOF_POINTER__+ ret++#endif+#ifdef __KERNEL__+.globl _crypton_sha256_asm_block_neon+#endif++.align 4+_crypton_sha256_asm_block_neon:+Lneon_entry:+ stp x29, x30, [sp, #-2*__SIZEOF_POINTER__]!+ mov x29, sp+ sub sp,sp,#16*4++ adr x16,LK256+ add x2,x1,x2,lsl#6 // len to point at the end of inp++ ld1 {v0.16b},[x1], #16+ ld1 {v1.16b},[x1], #16+ ld1 {v2.16b},[x1], #16+ ld1 {v3.16b},[x1], #16+ ld1 {v4.4s},[x16], #16+ ld1 {v5.4s},[x16], #16+ ld1 {v6.4s},[x16], #16+ ld1 {v7.4s},[x16], #16+ rev32 v0.16b,v0.16b // yes, even on+ rev32 v1.16b,v1.16b // big-endian+ rev32 v2.16b,v2.16b+ rev32 v3.16b,v3.16b+ mov x17,sp+ add v4.4s,v4.4s,v0.4s+ add v5.4s,v5.4s,v1.4s+ add v6.4s,v6.4s,v2.4s+ st1 {v4.4s,v5.4s},[x17], #32+ add v7.4s,v7.4s,v3.4s+ st1 {v6.4s,v7.4s},[x17]+ sub x17,x17,#32++ ldp w3,w4,[x0]+ ldp w5,w6,[x0,#8]+ ldp w7,w8,[x0,#16]+ ldp w9,w10,[x0,#24]+ ldr w12,[sp,#0]+ mov w13,wzr+ eor w14,w4,w5+ mov w15,wzr+ b L_00_48++.align 4+L_00_48:+ ext v4.16b,v0.16b,v1.16b,#4+ add w10,w10,w12+ add w3,w3,w15+ and w12,w8,w7+ bic w15,w9,w7+ ext v7.16b,v2.16b,v3.16b,#4+ eor w11,w7,w7,ror#5+ add w3,w3,w13+ mov d19,v3.d[1]+ orr w12,w12,w15+ eor w11,w11,w7,ror#19+ ushr v6.4s,v4.4s,#7+ eor w15,w3,w3,ror#11+ ushr v5.4s,v4.4s,#3+ add w10,w10,w12+ add v0.4s,v0.4s,v7.4s+ ror w11,w11,#6+ sli v6.4s,v4.4s,#25+ eor w13,w3,w4+ eor w15,w15,w3,ror#20+ ushr v7.4s,v4.4s,#18+ add w10,w10,w11+ ldr w12,[sp,#4]+ and w14,w14,w13+ eor v5.16b,v5.16b,v6.16b+ ror w15,w15,#2+ add w6,w6,w10+ sli v7.4s,v4.4s,#14+ eor w14,w14,w4+ ushr v16.4s,v19.4s,#17+ add w9,w9,w12+ add w10,w10,w15+ and w12,w7,w6+ eor v5.16b,v5.16b,v7.16b+ bic w15,w8,w6+ eor w11,w6,w6,ror#5+ sli v16.4s,v19.4s,#15+ add w10,w10,w14+ orr w12,w12,w15+ ushr v17.4s,v19.4s,#10+ eor w11,w11,w6,ror#19+ eor w15,w10,w10,ror#11+ ushr v7.4s,v19.4s,#19+ add w9,w9,w12+ ror w11,w11,#6+ add v0.4s,v0.4s,v5.4s+ eor w14,w10,w3+ eor w15,w15,w10,ror#20+ sli v7.4s,v19.4s,#13+ add w9,w9,w11+ ldr w12,[sp,#8]+ and w13,w13,w14+ eor v17.16b,v17.16b,v16.16b+ ror w15,w15,#2+ add w5,w5,w9+ eor w13,w13,w3+ eor v17.16b,v17.16b,v7.16b+ add w8,w8,w12+ add w9,w9,w15+ and w12,w6,w5+ add v0.4s,v0.4s,v17.4s+ bic w15,w7,w5+ eor w11,w5,w5,ror#5+ add w9,w9,w13+ ushr v18.4s,v0.4s,#17+ orr w12,w12,w15+ ushr v19.4s,v0.4s,#10+ eor w11,w11,w5,ror#19+ eor w15,w9,w9,ror#11+ sli v18.4s,v0.4s,#15+ add w8,w8,w12+ ushr v17.4s,v0.4s,#19+ ror w11,w11,#6+ eor w13,w9,w10+ eor v19.16b,v19.16b,v18.16b+ eor w15,w15,w9,ror#20+ add w8,w8,w11+ sli v17.4s,v0.4s,#13+ ldr w12,[sp,#12]+ and w14,w14,w13+ ror w15,w15,#2+ ld1 {v4.4s},[x16], #16+ add w4,w4,w8+ eor v19.16b,v19.16b,v17.16b+ eor w14,w14,w10+ eor v17.16b,v17.16b,v17.16b+ add w7,w7,w12+ add w8,w8,w15+ and w12,w5,w4+ mov v17.d[1],v19.d[0]+ bic w15,w6,w4+ eor w11,w4,w4,ror#5+ add w8,w8,w14+ add v0.4s,v0.4s,v17.4s+ orr w12,w12,w15+ eor w11,w11,w4,ror#19+ eor w15,w8,w8,ror#11+ add v4.4s,v4.4s,v0.4s+ add w7,w7,w12+ ror w11,w11,#6+ eor w14,w8,w9+ eor w15,w15,w8,ror#20+ add w7,w7,w11+ ldr w12,[sp,#16]+ and w13,w13,w14+ ror w15,w15,#2+ add w3,w3,w7+ eor w13,w13,w9+ st1 {v4.4s},[x17], #16+ ext v4.16b,v1.16b,v2.16b,#4+ add w6,w6,w12+ add w7,w7,w15+ and w12,w4,w3+ bic w15,w5,w3+ ext v7.16b,v3.16b,v0.16b,#4+ eor w11,w3,w3,ror#5+ add w7,w7,w13+ mov d19,v0.d[1]+ orr w12,w12,w15+ eor w11,w11,w3,ror#19+ ushr v6.4s,v4.4s,#7+ eor w15,w7,w7,ror#11+ ushr v5.4s,v4.4s,#3+ add w6,w6,w12+ add v1.4s,v1.4s,v7.4s+ ror w11,w11,#6+ sli v6.4s,v4.4s,#25+ eor w13,w7,w8+ eor w15,w15,w7,ror#20+ ushr v7.4s,v4.4s,#18+ add w6,w6,w11+ ldr w12,[sp,#20]+ and w14,w14,w13+ eor v5.16b,v5.16b,v6.16b+ ror w15,w15,#2+ add w10,w10,w6+ sli v7.4s,v4.4s,#14+ eor w14,w14,w8+ ushr v16.4s,v19.4s,#17+ add w5,w5,w12+ add w6,w6,w15+ and w12,w3,w10+ eor v5.16b,v5.16b,v7.16b+ bic w15,w4,w10+ eor w11,w10,w10,ror#5+ sli v16.4s,v19.4s,#15+ add w6,w6,w14+ orr w12,w12,w15+ ushr v17.4s,v19.4s,#10+ eor w11,w11,w10,ror#19+ eor w15,w6,w6,ror#11+ ushr v7.4s,v19.4s,#19+ add w5,w5,w12+ ror w11,w11,#6+ add v1.4s,v1.4s,v5.4s+ eor w14,w6,w7+ eor w15,w15,w6,ror#20+ sli v7.4s,v19.4s,#13+ add w5,w5,w11+ ldr w12,[sp,#24]+ and w13,w13,w14+ eor v17.16b,v17.16b,v16.16b+ ror w15,w15,#2+ add w9,w9,w5+ eor w13,w13,w7+ eor v17.16b,v17.16b,v7.16b+ add w4,w4,w12+ add w5,w5,w15+ and w12,w10,w9+ add v1.4s,v1.4s,v17.4s+ bic w15,w3,w9+ eor w11,w9,w9,ror#5+ add w5,w5,w13+ ushr v18.4s,v1.4s,#17+ orr w12,w12,w15+ ushr v19.4s,v1.4s,#10+ eor w11,w11,w9,ror#19+ eor w15,w5,w5,ror#11+ sli v18.4s,v1.4s,#15+ add w4,w4,w12+ ushr v17.4s,v1.4s,#19+ ror w11,w11,#6+ eor w13,w5,w6+ eor v19.16b,v19.16b,v18.16b+ eor w15,w15,w5,ror#20+ add w4,w4,w11+ sli v17.4s,v1.4s,#13+ ldr w12,[sp,#28]+ and w14,w14,w13+ ror w15,w15,#2+ ld1 {v4.4s},[x16], #16+ add w8,w8,w4+ eor v19.16b,v19.16b,v17.16b+ eor w14,w14,w6+ eor v17.16b,v17.16b,v17.16b+ add w3,w3,w12+ add w4,w4,w15+ and w12,w9,w8+ mov v17.d[1],v19.d[0]+ bic w15,w10,w8+ eor w11,w8,w8,ror#5+ add w4,w4,w14+ add v1.4s,v1.4s,v17.4s+ orr w12,w12,w15+ eor w11,w11,w8,ror#19+ eor w15,w4,w4,ror#11+ add v4.4s,v4.4s,v1.4s+ add w3,w3,w12+ ror w11,w11,#6+ eor w14,w4,w5+ eor w15,w15,w4,ror#20+ add w3,w3,w11+ ldr w12,[sp,#32]+ and w13,w13,w14+ ror w15,w15,#2+ add w7,w7,w3+ eor w13,w13,w5+ st1 {v4.4s},[x17], #16+ ext v4.16b,v2.16b,v3.16b,#4+ add w10,w10,w12+ add w3,w3,w15+ and w12,w8,w7+ bic w15,w9,w7+ ext v7.16b,v0.16b,v1.16b,#4+ eor w11,w7,w7,ror#5+ add w3,w3,w13+ mov d19,v1.d[1]+ orr w12,w12,w15+ eor w11,w11,w7,ror#19+ ushr v6.4s,v4.4s,#7+ eor w15,w3,w3,ror#11+ ushr v5.4s,v4.4s,#3+ add w10,w10,w12+ add v2.4s,v2.4s,v7.4s+ ror w11,w11,#6+ sli v6.4s,v4.4s,#25+ eor w13,w3,w4+ eor w15,w15,w3,ror#20+ ushr v7.4s,v4.4s,#18+ add w10,w10,w11+ ldr w12,[sp,#36]+ and w14,w14,w13+ eor v5.16b,v5.16b,v6.16b+ ror w15,w15,#2+ add w6,w6,w10+ sli v7.4s,v4.4s,#14+ eor w14,w14,w4+ ushr v16.4s,v19.4s,#17+ add w9,w9,w12+ add w10,w10,w15+ and w12,w7,w6+ eor v5.16b,v5.16b,v7.16b+ bic w15,w8,w6+ eor w11,w6,w6,ror#5+ sli v16.4s,v19.4s,#15+ add w10,w10,w14+ orr w12,w12,w15+ ushr v17.4s,v19.4s,#10+ eor w11,w11,w6,ror#19+ eor w15,w10,w10,ror#11+ ushr v7.4s,v19.4s,#19+ add w9,w9,w12+ ror w11,w11,#6+ add v2.4s,v2.4s,v5.4s+ eor w14,w10,w3+ eor w15,w15,w10,ror#20+ sli v7.4s,v19.4s,#13+ add w9,w9,w11+ ldr w12,[sp,#40]+ and w13,w13,w14+ eor v17.16b,v17.16b,v16.16b+ ror w15,w15,#2+ add w5,w5,w9+ eor w13,w13,w3+ eor v17.16b,v17.16b,v7.16b+ add w8,w8,w12+ add w9,w9,w15+ and w12,w6,w5+ add v2.4s,v2.4s,v17.4s+ bic w15,w7,w5+ eor w11,w5,w5,ror#5+ add w9,w9,w13+ ushr v18.4s,v2.4s,#17+ orr w12,w12,w15+ ushr v19.4s,v2.4s,#10+ eor w11,w11,w5,ror#19+ eor w15,w9,w9,ror#11+ sli v18.4s,v2.4s,#15+ add w8,w8,w12+ ushr v17.4s,v2.4s,#19+ ror w11,w11,#6+ eor w13,w9,w10+ eor v19.16b,v19.16b,v18.16b+ eor w15,w15,w9,ror#20+ add w8,w8,w11+ sli v17.4s,v2.4s,#13+ ldr w12,[sp,#44]+ and w14,w14,w13+ ror w15,w15,#2+ ld1 {v4.4s},[x16], #16+ add w4,w4,w8+ eor v19.16b,v19.16b,v17.16b+ eor w14,w14,w10+ eor v17.16b,v17.16b,v17.16b+ add w7,w7,w12+ add w8,w8,w15+ and w12,w5,w4+ mov v17.d[1],v19.d[0]+ bic w15,w6,w4+ eor w11,w4,w4,ror#5+ add w8,w8,w14+ add v2.4s,v2.4s,v17.4s+ orr w12,w12,w15+ eor w11,w11,w4,ror#19+ eor w15,w8,w8,ror#11+ add v4.4s,v4.4s,v2.4s+ add w7,w7,w12+ ror w11,w11,#6+ eor w14,w8,w9+ eor w15,w15,w8,ror#20+ add w7,w7,w11+ ldr w12,[sp,#48]+ and w13,w13,w14+ ror w15,w15,#2+ add w3,w3,w7+ eor w13,w13,w9+ st1 {v4.4s},[x17], #16+ ext v4.16b,v3.16b,v0.16b,#4+ add w6,w6,w12+ add w7,w7,w15+ and w12,w4,w3+ bic w15,w5,w3+ ext v7.16b,v1.16b,v2.16b,#4+ eor w11,w3,w3,ror#5+ add w7,w7,w13+ mov d19,v2.d[1]+ orr w12,w12,w15+ eor w11,w11,w3,ror#19+ ushr v6.4s,v4.4s,#7+ eor w15,w7,w7,ror#11+ ushr v5.4s,v4.4s,#3+ add w6,w6,w12+ add v3.4s,v3.4s,v7.4s+ ror w11,w11,#6+ sli v6.4s,v4.4s,#25+ eor w13,w7,w8+ eor w15,w15,w7,ror#20+ ushr v7.4s,v4.4s,#18+ add w6,w6,w11+ ldr w12,[sp,#52]+ and w14,w14,w13+ eor v5.16b,v5.16b,v6.16b+ ror w15,w15,#2+ add w10,w10,w6+ sli v7.4s,v4.4s,#14+ eor w14,w14,w8+ ushr v16.4s,v19.4s,#17+ add w5,w5,w12+ add w6,w6,w15+ and w12,w3,w10+ eor v5.16b,v5.16b,v7.16b+ bic w15,w4,w10+ eor w11,w10,w10,ror#5+ sli v16.4s,v19.4s,#15+ add w6,w6,w14+ orr w12,w12,w15+ ushr v17.4s,v19.4s,#10+ eor w11,w11,w10,ror#19+ eor w15,w6,w6,ror#11+ ushr v7.4s,v19.4s,#19+ add w5,w5,w12+ ror w11,w11,#6+ add v3.4s,v3.4s,v5.4s+ eor w14,w6,w7+ eor w15,w15,w6,ror#20+ sli v7.4s,v19.4s,#13+ add w5,w5,w11+ ldr w12,[sp,#56]+ and w13,w13,w14+ eor v17.16b,v17.16b,v16.16b+ ror w15,w15,#2+ add w9,w9,w5+ eor w13,w13,w7+ eor v17.16b,v17.16b,v7.16b+ add w4,w4,w12+ add w5,w5,w15+ and w12,w10,w9+ add v3.4s,v3.4s,v17.4s+ bic w15,w3,w9+ eor w11,w9,w9,ror#5+ add w5,w5,w13+ ushr v18.4s,v3.4s,#17+ orr w12,w12,w15+ ushr v19.4s,v3.4s,#10+ eor w11,w11,w9,ror#19+ eor w15,w5,w5,ror#11+ sli v18.4s,v3.4s,#15+ add w4,w4,w12+ ushr v17.4s,v3.4s,#19+ ror w11,w11,#6+ eor w13,w5,w6+ eor v19.16b,v19.16b,v18.16b+ eor w15,w15,w5,ror#20+ add w4,w4,w11+ sli v17.4s,v3.4s,#13+ ldr w12,[sp,#60]+ and w14,w14,w13+ ror w15,w15,#2+ ld1 {v4.4s},[x16], #16+ add w8,w8,w4+ eor v19.16b,v19.16b,v17.16b+ eor w14,w14,w6+ eor v17.16b,v17.16b,v17.16b+ add w3,w3,w12+ add w4,w4,w15+ and w12,w9,w8+ mov v17.d[1],v19.d[0]+ bic w15,w10,w8+ eor w11,w8,w8,ror#5+ add w4,w4,w14+ add v3.4s,v3.4s,v17.4s+ orr w12,w12,w15+ eor w11,w11,w8,ror#19+ eor w15,w4,w4,ror#11+ add v4.4s,v4.4s,v3.4s+ add w3,w3,w12+ ror w11,w11,#6+ eor w14,w4,w5+ eor w15,w15,w4,ror#20+ add w3,w3,w11+ ldr w12,[x16]+ and w13,w13,w14+ ror w15,w15,#2+ add w7,w7,w3+ eor w13,w13,w5+ st1 {v4.4s},[x17], #16+ cmp w12,#0 // check for K256 terminator+ ldr w12,[sp,#0]+ sub x17,x17,#64+ bne L_00_48++ sub x16,x16,#256+ cmp x1,x2+ mov x17, #-64+ csel x17, x17, xzr, eq+ add x1,x1,x17+ mov x17,sp+ add w10,w10,w12+ add w3,w3,w15+ and w12,w8,w7+ ld1 {v0.16b},[x1],#16+ bic w15,w9,w7+ eor w11,w7,w7,ror#5+ ld1 {v4.4s},[x16],#16+ add w3,w3,w13+ orr w12,w12,w15+ eor w11,w11,w7,ror#19+ eor w15,w3,w3,ror#11+ rev32 v0.16b,v0.16b+ add w10,w10,w12+ ror w11,w11,#6+ eor w13,w3,w4+ eor w15,w15,w3,ror#20+ add v4.4s,v4.4s,v0.4s+ add w10,w10,w11+ ldr w12,[sp,#4]+ and w14,w14,w13+ ror w15,w15,#2+ add w6,w6,w10+ eor w14,w14,w4+ add w9,w9,w12+ add w10,w10,w15+ and w12,w7,w6+ bic w15,w8,w6+ eor w11,w6,w6,ror#5+ add w10,w10,w14+ orr w12,w12,w15+ eor w11,w11,w6,ror#19+ eor w15,w10,w10,ror#11+ add w9,w9,w12+ ror w11,w11,#6+ eor w14,w10,w3+ eor w15,w15,w10,ror#20+ add w9,w9,w11+ ldr w12,[sp,#8]+ and w13,w13,w14+ ror w15,w15,#2+ add w5,w5,w9+ eor w13,w13,w3+ add w8,w8,w12+ add w9,w9,w15+ and w12,w6,w5+ bic w15,w7,w5+ eor w11,w5,w5,ror#5+ add w9,w9,w13+ orr w12,w12,w15+ eor w11,w11,w5,ror#19+ eor w15,w9,w9,ror#11+ add w8,w8,w12+ ror w11,w11,#6+ eor w13,w9,w10+ eor w15,w15,w9,ror#20+ add w8,w8,w11+ ldr w12,[sp,#12]+ and w14,w14,w13+ ror w15,w15,#2+ add w4,w4,w8+ eor w14,w14,w10+ add w7,w7,w12+ add w8,w8,w15+ and w12,w5,w4+ bic w15,w6,w4+ eor w11,w4,w4,ror#5+ add w8,w8,w14+ orr w12,w12,w15+ eor w11,w11,w4,ror#19+ eor w15,w8,w8,ror#11+ add w7,w7,w12+ ror w11,w11,#6+ eor w14,w8,w9+ eor w15,w15,w8,ror#20+ add w7,w7,w11+ ldr w12,[sp,#16]+ and w13,w13,w14+ ror w15,w15,#2+ add w3,w3,w7+ eor w13,w13,w9+ st1 {v4.4s},[x17], #16+ add w6,w6,w12+ add w7,w7,w15+ and w12,w4,w3+ ld1 {v1.16b},[x1],#16+ bic w15,w5,w3+ eor w11,w3,w3,ror#5+ ld1 {v4.4s},[x16],#16+ add w7,w7,w13+ orr w12,w12,w15+ eor w11,w11,w3,ror#19+ eor w15,w7,w7,ror#11+ rev32 v1.16b,v1.16b+ add w6,w6,w12+ ror w11,w11,#6+ eor w13,w7,w8+ eor w15,w15,w7,ror#20+ add v4.4s,v4.4s,v1.4s+ add w6,w6,w11+ ldr w12,[sp,#20]+ and w14,w14,w13+ ror w15,w15,#2+ add w10,w10,w6+ eor w14,w14,w8+ add w5,w5,w12+ add w6,w6,w15+ and w12,w3,w10+ bic w15,w4,w10+ eor w11,w10,w10,ror#5+ add w6,w6,w14+ orr w12,w12,w15+ eor w11,w11,w10,ror#19+ eor w15,w6,w6,ror#11+ add w5,w5,w12+ ror w11,w11,#6+ eor w14,w6,w7+ eor w15,w15,w6,ror#20+ add w5,w5,w11+ ldr w12,[sp,#24]+ and w13,w13,w14+ ror w15,w15,#2+ add w9,w9,w5+ eor w13,w13,w7+ add w4,w4,w12+ add w5,w5,w15+ and w12,w10,w9+ bic w15,w3,w9+ eor w11,w9,w9,ror#5+ add w5,w5,w13+ orr w12,w12,w15+ eor w11,w11,w9,ror#19+ eor w15,w5,w5,ror#11+ add w4,w4,w12+ ror w11,w11,#6+ eor w13,w5,w6+ eor w15,w15,w5,ror#20+ add w4,w4,w11+ ldr w12,[sp,#28]+ and w14,w14,w13+ ror w15,w15,#2+ add w8,w8,w4+ eor w14,w14,w6+ add w3,w3,w12+ add w4,w4,w15+ and w12,w9,w8+ bic w15,w10,w8+ eor w11,w8,w8,ror#5+ add w4,w4,w14+ orr w12,w12,w15+ eor w11,w11,w8,ror#19+ eor w15,w4,w4,ror#11+ add w3,w3,w12+ ror w11,w11,#6+ eor w14,w4,w5+ eor w15,w15,w4,ror#20+ add w3,w3,w11+ ldr w12,[sp,#32]+ and w13,w13,w14+ ror w15,w15,#2+ add w7,w7,w3+ eor w13,w13,w5+ st1 {v4.4s},[x17], #16+ add w10,w10,w12+ add w3,w3,w15+ and w12,w8,w7+ ld1 {v2.16b},[x1],#16+ bic w15,w9,w7+ eor w11,w7,w7,ror#5+ ld1 {v4.4s},[x16],#16+ add w3,w3,w13+ orr w12,w12,w15+ eor w11,w11,w7,ror#19+ eor w15,w3,w3,ror#11+ rev32 v2.16b,v2.16b+ add w10,w10,w12+ ror w11,w11,#6+ eor w13,w3,w4+ eor w15,w15,w3,ror#20+ add v4.4s,v4.4s,v2.4s+ add w10,w10,w11+ ldr w12,[sp,#36]+ and w14,w14,w13+ ror w15,w15,#2+ add w6,w6,w10+ eor w14,w14,w4+ add w9,w9,w12+ add w10,w10,w15+ and w12,w7,w6+ bic w15,w8,w6+ eor w11,w6,w6,ror#5+ add w10,w10,w14+ orr w12,w12,w15+ eor w11,w11,w6,ror#19+ eor w15,w10,w10,ror#11+ add w9,w9,w12+ ror w11,w11,#6+ eor w14,w10,w3+ eor w15,w15,w10,ror#20+ add w9,w9,w11+ ldr w12,[sp,#40]+ and w13,w13,w14+ ror w15,w15,#2+ add w5,w5,w9+ eor w13,w13,w3+ add w8,w8,w12+ add w9,w9,w15+ and w12,w6,w5+ bic w15,w7,w5+ eor w11,w5,w5,ror#5+ add w9,w9,w13+ orr w12,w12,w15+ eor w11,w11,w5,ror#19+ eor w15,w9,w9,ror#11+ add w8,w8,w12+ ror w11,w11,#6+ eor w13,w9,w10+ eor w15,w15,w9,ror#20+ add w8,w8,w11+ ldr w12,[sp,#44]+ and w14,w14,w13+ ror w15,w15,#2+ add w4,w4,w8+ eor w14,w14,w10+ add w7,w7,w12+ add w8,w8,w15+ and w12,w5,w4+ bic w15,w6,w4+ eor w11,w4,w4,ror#5+ add w8,w8,w14+ orr w12,w12,w15+ eor w11,w11,w4,ror#19+ eor w15,w8,w8,ror#11+ add w7,w7,w12+ ror w11,w11,#6+ eor w14,w8,w9+ eor w15,w15,w8,ror#20+ add w7,w7,w11+ ldr w12,[sp,#48]+ and w13,w13,w14+ ror w15,w15,#2+ add w3,w3,w7+ eor w13,w13,w9+ st1 {v4.4s},[x17], #16+ add w6,w6,w12+ add w7,w7,w15+ and w12,w4,w3+ ld1 {v3.16b},[x1],#16+ bic w15,w5,w3+ eor w11,w3,w3,ror#5+ ld1 {v4.4s},[x16],#16+ add w7,w7,w13+ orr w12,w12,w15+ eor w11,w11,w3,ror#19+ eor w15,w7,w7,ror#11+ rev32 v3.16b,v3.16b+ add w6,w6,w12+ ror w11,w11,#6+ eor w13,w7,w8+ eor w15,w15,w7,ror#20+ add v4.4s,v4.4s,v3.4s+ add w6,w6,w11+ ldr w12,[sp,#52]+ and w14,w14,w13+ ror w15,w15,#2+ add w10,w10,w6+ eor w14,w14,w8+ add w5,w5,w12+ add w6,w6,w15+ and w12,w3,w10+ bic w15,w4,w10+ eor w11,w10,w10,ror#5+ add w6,w6,w14+ orr w12,w12,w15+ eor w11,w11,w10,ror#19+ eor w15,w6,w6,ror#11+ add w5,w5,w12+ ror w11,w11,#6+ eor w14,w6,w7+ eor w15,w15,w6,ror#20+ add w5,w5,w11+ ldr w12,[sp,#56]+ and w13,w13,w14+ ror w15,w15,#2+ add w9,w9,w5+ eor w13,w13,w7+ add w4,w4,w12+ add w5,w5,w15+ and w12,w10,w9+ bic w15,w3,w9+ eor w11,w9,w9,ror#5+ add w5,w5,w13+ orr w12,w12,w15+ eor w11,w11,w9,ror#19+ eor w15,w5,w5,ror#11+ add w4,w4,w12+ ror w11,w11,#6+ eor w13,w5,w6+ eor w15,w15,w5,ror#20+ add w4,w4,w11+ ldr w12,[sp,#60]+ and w14,w14,w13+ ror w15,w15,#2+ add w8,w8,w4+ eor w14,w14,w6+ add w3,w3,w12+ add w4,w4,w15+ and w12,w9,w8+ bic w15,w10,w8+ eor w11,w8,w8,ror#5+ add w4,w4,w14+ orr w12,w12,w15+ eor w11,w11,w8,ror#19+ eor w15,w4,w4,ror#11+ add w3,w3,w12+ ror w11,w11,#6+ eor w14,w4,w5+ eor w15,w15,w4,ror#20+ add w3,w3,w11+ and w13,w13,w14+ ror w15,w15,#2+ add w7,w7,w3+ eor w13,w13,w5+ st1 {v4.4s},[x17], #16+ add w3,w3,w15 // h+=Sigma0(a) from the past+ ldp w11,w12,[x0,#0]+ add w3,w3,w13 // h+=Maj(a,b,c) from the past+ ldp w13,w14,[x0,#8]+ add w3,w3,w11 // accumulate+ add w4,w4,w12+ ldp w11,w12,[x0,#16]+ add w5,w5,w13+ add w6,w6,w14+ ldp w13,w14,[x0,#24]+ add w7,w7,w11+ add w8,w8,w12+ ldr w12,[sp,#0]+ stp w3,w4,[x0,#0]+ add w9,w9,w13+ mov w13,wzr+ stp w5,w6,[x0,#8]+ add w10,w10,w14+ stp w7,w8,[x0,#16]+ eor w14,w4,w5+ stp w9,w10,[x0,#24]+ mov w15,wzr+ mov x17,sp+ b.ne L_00_48++ ldr x29,[x29]+ add sp,sp,#16*4+2*__SIZEOF_POINTER__+ ret++#if !defined(__KERNEL__) && !defined(_WIN64)+.comm __crypton_armcap_P,4+.private_extern _crypton_armcap_P+#endif
@@ -0,0 +1,2053 @@+// SPDX-License-Identifier: GPL-1.0+ OR BSD-3-Clause+//+// ====================================================================+// Written by Andy Polyakov, @dot-asm, initially for the OpenSSL+// project.+// ====================================================================+//+// SHA256/512 for ARMv8.+//+// Performance in cycles per processed byte and improvement coefficient+// over code generated with "default" compiler:+//+// SHA256-hw SHA256(*) SHA512+// Apple A7 1.97 10.5 (+33%) 6.73 (-1%(**))+// Apple A10 1.30 5.81+// Apple A12 1.31 5.06+// Apple A14/M1 1.30 8.19 (+14%) 2.24 (hw)+// Cortex-A53 2.38 15.5 (+115%) 10.0 (+150%(***))+// Cortex-A57 2.31 11.6 (+86%) 7.51 (+260%(***))+// Cortex-A76 1.60 9.5 6.05+// Cortex-X2 1.60 7.3 2.60 (hw)+// Cortex-X925 1.57 5.97 2.55 (hw)+// Denver 2.01 10.5 (+26%) 6.70 (+8%)+// X-Gene 20.0 (+100%) 12.8 (+300%(***))+// Mongoose 2.36 13.0 (+50%) 8.36 (+33%)+// Kryo 1.92 17.4 (+30%) 11.2 (+8%)+// ThunderX2 2.54 13.2 (+40%) 8.40 (+18%)+// Shapdragon X 1.40 7.43 2.23 (hw)+//+// (*) Software SHA256 results are of lesser relevance, presented+// mostly for informational purposes.+// (**) The result is a trade-off: it's possible to improve it by+// 10% (or by 1 cycle per round), but at the cost of 20% loss+// on Cortex-A53 (or by 4 cycles per round).+// (***) Super-impressive coefficients over gcc-generated code are+// indication of some compiler "pathology", most notably code+// generated with -mgeneral-regs-only is significantly faster+// and the gap is only 40-90%.+//+// October 2016.+//+// Originally it was reckoned that it makes no sense to implement NEON+// version of SHA256 for 64-bit processors. This is because performance+// improvement on most wide-spread Cortex-A5x processors was observed+// to be marginal, same on Cortex-A53 and ~10% on A57. But then it was+// observed that 32-bit NEON SHA256 performs significantly better than+// 64-bit scalar version on *some* of the more recent processors. As+// result 64-bit NEON version of SHA256 was added to provide best+// all-round performance. For example it executes ~30% faster on X-Gene+// and Mongoose. [For reference, NEON version of SHA512 is bound to+// deliver much less improvement, likely *negative* on Cortex-A5x.+// Which is why NEON support is limited to SHA256.]++#ifndef __KERNEL__+# include "arm_arch.h"++#endif++.text++.globl crypton_sha256_asm_block_data_order+.type crypton_sha256_asm_block_data_order,%function+.align 6+crypton_sha256_asm_block_data_order:+#ifndef __KERNEL__+ adrp x16,crypton_armcap_P+ ldr w16,[x16,#:lo12:crypton_armcap_P]+ tst w16,#ARMV8_SHA256+ b.ne .Lv8_entry+ tst w16,#ARMV7_NEON+ b.ne .Lneon_entry+#endif+.inst 0xd503233f // paciasp+ stp x29,x30,[sp,#-16*__SIZEOF_POINTER__]!+ add x29,sp,#0++ stp x19,x20,[sp,#2*__SIZEOF_POINTER__]+ stp x21,x22,[sp,#4*__SIZEOF_POINTER__]+ stp x23,x24,[sp,#6*__SIZEOF_POINTER__]+ stp x25,x26,[sp,#8*__SIZEOF_POINTER__]+ stp x27,x28,[sp,#10*__SIZEOF_POINTER__]+ sub sp,sp,#4*4++ ldp w20,w21,[x0] // load context+ ldp w22,w23,[x0,#2*4]+ lsl x2,x2,#6+ ldp w24,w25,[x0,#4*4]+ add x2,x1,x2+ ldp w26,w27,[x0,#6*4]+ adr x30,.LK256+ stp x0,x2,[x29,#12*__SIZEOF_POINTER__]++.Loop:+ ldp w3,w4,[x1],#2*4+ ldr w19,[x30],#4 // *K+++ eor w28,w21,w22 // magic seed+ str x1,[x29,#14*__SIZEOF_POINTER__]+#ifndef __AARCH64EB__+ rev w3,w3 // 0+#endif+ ror w16,w24,#6+ add w27,w27,w19 // h+=K[i]+ eor w6,w24,w24,ror#14+ and w17,w25,w24+ bic w19,w26,w24+ add w27,w27,w3 // h+=X[i]+ orr w17,w17,w19 // Ch(e,f,g)+ eor w19,w20,w21 // a^b, b^c in next round+ eor w16,w16,w6,ror#11 // Sigma1(e)+ ror w6,w20,#2+ add w27,w27,w17 // h+=Ch(e,f,g)+ eor w17,w20,w20,ror#9+ add w27,w27,w16 // h+=Sigma1(e)+ and w28,w28,w19 // (b^c)&=(a^b)+ add w23,w23,w27 // d+=h+ eor w28,w28,w21 // Maj(a,b,c)+ eor w17,w6,w17,ror#13 // Sigma0(a)+ add w27,w27,w28 // h+=Maj(a,b,c)+ ldr w28,[x30],#4 // *K++, w19 in next round+ //add w27,w27,w17 // h+=Sigma0(a)+#ifndef __AARCH64EB__+ rev w4,w4 // 1+#endif+ ldp w5,w6,[x1],#2*4+ add w27,w27,w17 // h+=Sigma0(a)+ ror w16,w23,#6+ add w26,w26,w28 // h+=K[i]+ eor w7,w23,w23,ror#14+ and w17,w24,w23+ bic w28,w25,w23+ add w26,w26,w4 // h+=X[i]+ orr w17,w17,w28 // Ch(e,f,g)+ eor w28,w27,w20 // a^b, b^c in next round+ eor w16,w16,w7,ror#11 // Sigma1(e)+ ror w7,w27,#2+ add w26,w26,w17 // h+=Ch(e,f,g)+ eor w17,w27,w27,ror#9+ add w26,w26,w16 // h+=Sigma1(e)+ and w19,w19,w28 // (b^c)&=(a^b)+ add w22,w22,w26 // d+=h+ eor w19,w19,w20 // Maj(a,b,c)+ eor w17,w7,w17,ror#13 // Sigma0(a)+ add w26,w26,w19 // h+=Maj(a,b,c)+ ldr w19,[x30],#4 // *K++, w28 in next round+ //add w26,w26,w17 // h+=Sigma0(a)+#ifndef __AARCH64EB__+ rev w5,w5 // 2+#endif+ add w26,w26,w17 // h+=Sigma0(a)+ ror w16,w22,#6+ add w25,w25,w19 // h+=K[i]+ eor w8,w22,w22,ror#14+ and w17,w23,w22+ bic w19,w24,w22+ add w25,w25,w5 // h+=X[i]+ orr w17,w17,w19 // Ch(e,f,g)+ eor w19,w26,w27 // a^b, b^c in next round+ eor w16,w16,w8,ror#11 // Sigma1(e)+ ror w8,w26,#2+ add w25,w25,w17 // h+=Ch(e,f,g)+ eor w17,w26,w26,ror#9+ add w25,w25,w16 // h+=Sigma1(e)+ and w28,w28,w19 // (b^c)&=(a^b)+ add w21,w21,w25 // d+=h+ eor w28,w28,w27 // Maj(a,b,c)+ eor w17,w8,w17,ror#13 // Sigma0(a)+ add w25,w25,w28 // h+=Maj(a,b,c)+ ldr w28,[x30],#4 // *K++, w19 in next round+ //add w25,w25,w17 // h+=Sigma0(a)+#ifndef __AARCH64EB__+ rev w6,w6 // 3+#endif+ ldp w7,w8,[x1],#2*4+ add w25,w25,w17 // h+=Sigma0(a)+ ror w16,w21,#6+ add w24,w24,w28 // h+=K[i]+ eor w9,w21,w21,ror#14+ and w17,w22,w21+ bic w28,w23,w21+ add w24,w24,w6 // h+=X[i]+ orr w17,w17,w28 // Ch(e,f,g)+ eor w28,w25,w26 // a^b, b^c in next round+ eor w16,w16,w9,ror#11 // Sigma1(e)+ ror w9,w25,#2+ add w24,w24,w17 // h+=Ch(e,f,g)+ eor w17,w25,w25,ror#9+ add w24,w24,w16 // h+=Sigma1(e)+ and w19,w19,w28 // (b^c)&=(a^b)+ add w20,w20,w24 // d+=h+ eor w19,w19,w26 // Maj(a,b,c)+ eor w17,w9,w17,ror#13 // Sigma0(a)+ add w24,w24,w19 // h+=Maj(a,b,c)+ ldr w19,[x30],#4 // *K++, w28 in next round+ //add w24,w24,w17 // h+=Sigma0(a)+#ifndef __AARCH64EB__+ rev w7,w7 // 4+#endif+ add w24,w24,w17 // h+=Sigma0(a)+ ror w16,w20,#6+ add w23,w23,w19 // h+=K[i]+ eor w10,w20,w20,ror#14+ and w17,w21,w20+ bic w19,w22,w20+ add w23,w23,w7 // h+=X[i]+ orr w17,w17,w19 // Ch(e,f,g)+ eor w19,w24,w25 // a^b, b^c in next round+ eor w16,w16,w10,ror#11 // Sigma1(e)+ ror w10,w24,#2+ add w23,w23,w17 // h+=Ch(e,f,g)+ eor w17,w24,w24,ror#9+ add w23,w23,w16 // h+=Sigma1(e)+ and w28,w28,w19 // (b^c)&=(a^b)+ add w27,w27,w23 // d+=h+ eor w28,w28,w25 // Maj(a,b,c)+ eor w17,w10,w17,ror#13 // Sigma0(a)+ add w23,w23,w28 // h+=Maj(a,b,c)+ ldr w28,[x30],#4 // *K++, w19 in next round+ //add w23,w23,w17 // h+=Sigma0(a)+#ifndef __AARCH64EB__+ rev w8,w8 // 5+#endif+ ldp w9,w10,[x1],#2*4+ add w23,w23,w17 // h+=Sigma0(a)+ ror w16,w27,#6+ add w22,w22,w28 // h+=K[i]+ eor w11,w27,w27,ror#14+ and w17,w20,w27+ bic w28,w21,w27+ add w22,w22,w8 // h+=X[i]+ orr w17,w17,w28 // Ch(e,f,g)+ eor w28,w23,w24 // a^b, b^c in next round+ eor w16,w16,w11,ror#11 // Sigma1(e)+ ror w11,w23,#2+ add w22,w22,w17 // h+=Ch(e,f,g)+ eor w17,w23,w23,ror#9+ add w22,w22,w16 // h+=Sigma1(e)+ and w19,w19,w28 // (b^c)&=(a^b)+ add w26,w26,w22 // d+=h+ eor w19,w19,w24 // Maj(a,b,c)+ eor w17,w11,w17,ror#13 // Sigma0(a)+ add w22,w22,w19 // h+=Maj(a,b,c)+ ldr w19,[x30],#4 // *K++, w28 in next round+ //add w22,w22,w17 // h+=Sigma0(a)+#ifndef __AARCH64EB__+ rev w9,w9 // 6+#endif+ add w22,w22,w17 // h+=Sigma0(a)+ ror w16,w26,#6+ add w21,w21,w19 // h+=K[i]+ eor w12,w26,w26,ror#14+ and w17,w27,w26+ bic w19,w20,w26+ add w21,w21,w9 // h+=X[i]+ orr w17,w17,w19 // Ch(e,f,g)+ eor w19,w22,w23 // a^b, b^c in next round+ eor w16,w16,w12,ror#11 // Sigma1(e)+ ror w12,w22,#2+ add w21,w21,w17 // h+=Ch(e,f,g)+ eor w17,w22,w22,ror#9+ add w21,w21,w16 // h+=Sigma1(e)+ and w28,w28,w19 // (b^c)&=(a^b)+ add w25,w25,w21 // d+=h+ eor w28,w28,w23 // Maj(a,b,c)+ eor w17,w12,w17,ror#13 // Sigma0(a)+ add w21,w21,w28 // h+=Maj(a,b,c)+ ldr w28,[x30],#4 // *K++, w19 in next round+ //add w21,w21,w17 // h+=Sigma0(a)+#ifndef __AARCH64EB__+ rev w10,w10 // 7+#endif+ ldp w11,w12,[x1],#2*4+ add w21,w21,w17 // h+=Sigma0(a)+ ror w16,w25,#6+ add w20,w20,w28 // h+=K[i]+ eor w13,w25,w25,ror#14+ and w17,w26,w25+ bic w28,w27,w25+ add w20,w20,w10 // h+=X[i]+ orr w17,w17,w28 // Ch(e,f,g)+ eor w28,w21,w22 // a^b, b^c in next round+ eor w16,w16,w13,ror#11 // Sigma1(e)+ ror w13,w21,#2+ add w20,w20,w17 // h+=Ch(e,f,g)+ eor w17,w21,w21,ror#9+ add w20,w20,w16 // h+=Sigma1(e)+ and w19,w19,w28 // (b^c)&=(a^b)+ add w24,w24,w20 // d+=h+ eor w19,w19,w22 // Maj(a,b,c)+ eor w17,w13,w17,ror#13 // Sigma0(a)+ add w20,w20,w19 // h+=Maj(a,b,c)+ ldr w19,[x30],#4 // *K++, w28 in next round+ //add w20,w20,w17 // h+=Sigma0(a)+#ifndef __AARCH64EB__+ rev w11,w11 // 8+#endif+ add w20,w20,w17 // h+=Sigma0(a)+ ror w16,w24,#6+ add w27,w27,w19 // h+=K[i]+ eor w14,w24,w24,ror#14+ and w17,w25,w24+ bic w19,w26,w24+ add w27,w27,w11 // h+=X[i]+ orr w17,w17,w19 // Ch(e,f,g)+ eor w19,w20,w21 // a^b, b^c in next round+ eor w16,w16,w14,ror#11 // Sigma1(e)+ ror w14,w20,#2+ add w27,w27,w17 // h+=Ch(e,f,g)+ eor w17,w20,w20,ror#9+ add w27,w27,w16 // h+=Sigma1(e)+ and w28,w28,w19 // (b^c)&=(a^b)+ add w23,w23,w27 // d+=h+ eor w28,w28,w21 // Maj(a,b,c)+ eor w17,w14,w17,ror#13 // Sigma0(a)+ add w27,w27,w28 // h+=Maj(a,b,c)+ ldr w28,[x30],#4 // *K++, w19 in next round+ //add w27,w27,w17 // h+=Sigma0(a)+#ifndef __AARCH64EB__+ rev w12,w12 // 9+#endif+ ldp w13,w14,[x1],#2*4+ add w27,w27,w17 // h+=Sigma0(a)+ ror w16,w23,#6+ add w26,w26,w28 // h+=K[i]+ eor w15,w23,w23,ror#14+ and w17,w24,w23+ bic w28,w25,w23+ add w26,w26,w12 // h+=X[i]+ orr w17,w17,w28 // Ch(e,f,g)+ eor w28,w27,w20 // a^b, b^c in next round+ eor w16,w16,w15,ror#11 // Sigma1(e)+ ror w15,w27,#2+ add w26,w26,w17 // h+=Ch(e,f,g)+ eor w17,w27,w27,ror#9+ add w26,w26,w16 // h+=Sigma1(e)+ and w19,w19,w28 // (b^c)&=(a^b)+ add w22,w22,w26 // d+=h+ eor w19,w19,w20 // Maj(a,b,c)+ eor w17,w15,w17,ror#13 // Sigma0(a)+ add w26,w26,w19 // h+=Maj(a,b,c)+ ldr w19,[x30],#4 // *K++, w28 in next round+ //add w26,w26,w17 // h+=Sigma0(a)+#ifndef __AARCH64EB__+ rev w13,w13 // 10+#endif+ add w26,w26,w17 // h+=Sigma0(a)+ ror w16,w22,#6+ add w25,w25,w19 // h+=K[i]+ eor w0,w22,w22,ror#14+ and w17,w23,w22+ bic w19,w24,w22+ add w25,w25,w13 // h+=X[i]+ orr w17,w17,w19 // Ch(e,f,g)+ eor w19,w26,w27 // a^b, b^c in next round+ eor w16,w16,w0,ror#11 // Sigma1(e)+ ror w0,w26,#2+ add w25,w25,w17 // h+=Ch(e,f,g)+ eor w17,w26,w26,ror#9+ add w25,w25,w16 // h+=Sigma1(e)+ and w28,w28,w19 // (b^c)&=(a^b)+ add w21,w21,w25 // d+=h+ eor w28,w28,w27 // Maj(a,b,c)+ eor w17,w0,w17,ror#13 // Sigma0(a)+ add w25,w25,w28 // h+=Maj(a,b,c)+ ldr w28,[x30],#4 // *K++, w19 in next round+ //add w25,w25,w17 // h+=Sigma0(a)+#ifndef __AARCH64EB__+ rev w14,w14 // 11+#endif+ ldp w15,w0,[x1],#2*4+ add w25,w25,w17 // h+=Sigma0(a)+ str w6,[sp,#12]+ ror w16,w21,#6+ add w24,w24,w28 // h+=K[i]+ eor w6,w21,w21,ror#14+ and w17,w22,w21+ bic w28,w23,w21+ add w24,w24,w14 // h+=X[i]+ orr w17,w17,w28 // Ch(e,f,g)+ eor w28,w25,w26 // a^b, b^c in next round+ eor w16,w16,w6,ror#11 // Sigma1(e)+ ror w6,w25,#2+ add w24,w24,w17 // h+=Ch(e,f,g)+ eor w17,w25,w25,ror#9+ add w24,w24,w16 // h+=Sigma1(e)+ and w19,w19,w28 // (b^c)&=(a^b)+ add w20,w20,w24 // d+=h+ eor w19,w19,w26 // Maj(a,b,c)+ eor w17,w6,w17,ror#13 // Sigma0(a)+ add w24,w24,w19 // h+=Maj(a,b,c)+ ldr w19,[x30],#4 // *K++, w28 in next round+ //add w24,w24,w17 // h+=Sigma0(a)+#ifndef __AARCH64EB__+ rev w15,w15 // 12+#endif+ add w24,w24,w17 // h+=Sigma0(a)+ str w7,[sp,#0]+ ror w16,w20,#6+ add w23,w23,w19 // h+=K[i]+ eor w7,w20,w20,ror#14+ and w17,w21,w20+ bic w19,w22,w20+ add w23,w23,w15 // h+=X[i]+ orr w17,w17,w19 // Ch(e,f,g)+ eor w19,w24,w25 // a^b, b^c in next round+ eor w16,w16,w7,ror#11 // Sigma1(e)+ ror w7,w24,#2+ add w23,w23,w17 // h+=Ch(e,f,g)+ eor w17,w24,w24,ror#9+ add w23,w23,w16 // h+=Sigma1(e)+ and w28,w28,w19 // (b^c)&=(a^b)+ add w27,w27,w23 // d+=h+ eor w28,w28,w25 // Maj(a,b,c)+ eor w17,w7,w17,ror#13 // Sigma0(a)+ add w23,w23,w28 // h+=Maj(a,b,c)+ ldr w28,[x30],#4 // *K++, w19 in next round+ //add w23,w23,w17 // h+=Sigma0(a)+#ifndef __AARCH64EB__+ rev w0,w0 // 13+#endif+ ldp w1,w2,[x1]+ add w23,w23,w17 // h+=Sigma0(a)+ str w8,[sp,#4]+ ror w16,w27,#6+ add w22,w22,w28 // h+=K[i]+ eor w8,w27,w27,ror#14+ and w17,w20,w27+ bic w28,w21,w27+ add w22,w22,w0 // h+=X[i]+ orr w17,w17,w28 // Ch(e,f,g)+ eor w28,w23,w24 // a^b, b^c in next round+ eor w16,w16,w8,ror#11 // Sigma1(e)+ ror w8,w23,#2+ add w22,w22,w17 // h+=Ch(e,f,g)+ eor w17,w23,w23,ror#9+ add w22,w22,w16 // h+=Sigma1(e)+ and w19,w19,w28 // (b^c)&=(a^b)+ add w26,w26,w22 // d+=h+ eor w19,w19,w24 // Maj(a,b,c)+ eor w17,w8,w17,ror#13 // Sigma0(a)+ add w22,w22,w19 // h+=Maj(a,b,c)+ ldr w19,[x30],#4 // *K++, w28 in next round+ //add w22,w22,w17 // h+=Sigma0(a)+#ifndef __AARCH64EB__+ rev w1,w1 // 14+#endif+ ldr w6,[sp,#12]+ add w22,w22,w17 // h+=Sigma0(a)+ str w9,[sp,#8]+ ror w16,w26,#6+ add w21,w21,w19 // h+=K[i]+ eor w9,w26,w26,ror#14+ and w17,w27,w26+ bic w19,w20,w26+ add w21,w21,w1 // h+=X[i]+ orr w17,w17,w19 // Ch(e,f,g)+ eor w19,w22,w23 // a^b, b^c in next round+ eor w16,w16,w9,ror#11 // Sigma1(e)+ ror w9,w22,#2+ add w21,w21,w17 // h+=Ch(e,f,g)+ eor w17,w22,w22,ror#9+ add w21,w21,w16 // h+=Sigma1(e)+ and w28,w28,w19 // (b^c)&=(a^b)+ add w25,w25,w21 // d+=h+ eor w28,w28,w23 // Maj(a,b,c)+ eor w17,w9,w17,ror#13 // Sigma0(a)+ add w21,w21,w28 // h+=Maj(a,b,c)+ ldr w28,[x30],#4 // *K++, w19 in next round+ //add w21,w21,w17 // h+=Sigma0(a)+#ifndef __AARCH64EB__+ rev w2,w2 // 15+#endif+ ldr w7,[sp,#0]+ add w21,w21,w17 // h+=Sigma0(a)+ str w10,[sp,#12]+ ror w16,w25,#6+ add w20,w20,w28 // h+=K[i]+ ror w9,w4,#7+ and w17,w26,w25+ ror w8,w1,#17+ bic w28,w27,w25+ ror w10,w21,#2+ add w20,w20,w2 // h+=X[i]+ eor w16,w16,w25,ror#11+ eor w9,w9,w4,ror#18+ orr w17,w17,w28 // Ch(e,f,g)+ eor w28,w21,w22 // a^b, b^c in next round+ eor w16,w16,w25,ror#25 // Sigma1(e)+ eor w10,w10,w21,ror#13+ add w20,w20,w17 // h+=Ch(e,f,g)+ and w19,w19,w28 // (b^c)&=(a^b)+ eor w8,w8,w1,ror#19+ eor w9,w9,w4,lsr#3 // sigma0(X[i+1])+ add w20,w20,w16 // h+=Sigma1(e)+ eor w19,w19,w22 // Maj(a,b,c)+ eor w17,w10,w21,ror#22 // Sigma0(a)+ eor w8,w8,w1,lsr#10 // sigma1(X[i+14])+ add w3,w3,w12+ add w24,w24,w20 // d+=h+ add w20,w20,w19 // h+=Maj(a,b,c)+ ldr w19,[x30],#4 // *K++, w28 in next round+ add w3,w3,w9+ add w20,w20,w17 // h+=Sigma0(a)+ add w3,w3,w8+.Loop_16_xx:+ ldr w8,[sp,#4]+ str w11,[sp,#0]+ ror w16,w24,#6+ add w27,w27,w19 // h+=K[i]+ ror w10,w5,#7+ and w17,w25,w24+ ror w9,w2,#17+ bic w19,w26,w24+ ror w11,w20,#2+ add w27,w27,w3 // h+=X[i]+ eor w16,w16,w24,ror#11+ eor w10,w10,w5,ror#18+ orr w17,w17,w19 // Ch(e,f,g)+ eor w19,w20,w21 // a^b, b^c in next round+ eor w16,w16,w24,ror#25 // Sigma1(e)+ eor w11,w11,w20,ror#13+ add w27,w27,w17 // h+=Ch(e,f,g)+ and w28,w28,w19 // (b^c)&=(a^b)+ eor w9,w9,w2,ror#19+ eor w10,w10,w5,lsr#3 // sigma0(X[i+1])+ add w27,w27,w16 // h+=Sigma1(e)+ eor w28,w28,w21 // Maj(a,b,c)+ eor w17,w11,w20,ror#22 // Sigma0(a)+ eor w9,w9,w2,lsr#10 // sigma1(X[i+14])+ add w4,w4,w13+ add w23,w23,w27 // d+=h+ add w27,w27,w28 // h+=Maj(a,b,c)+ ldr w28,[x30],#4 // *K++, w19 in next round+ add w4,w4,w10+ add w27,w27,w17 // h+=Sigma0(a)+ add w4,w4,w9+ ldr w9,[sp,#8]+ str w12,[sp,#4]+ ror w16,w23,#6+ add w26,w26,w28 // h+=K[i]+ ror w11,w6,#7+ and w17,w24,w23+ ror w10,w3,#17+ bic w28,w25,w23+ ror w12,w27,#2+ add w26,w26,w4 // h+=X[i]+ eor w16,w16,w23,ror#11+ eor w11,w11,w6,ror#18+ orr w17,w17,w28 // Ch(e,f,g)+ eor w28,w27,w20 // a^b, b^c in next round+ eor w16,w16,w23,ror#25 // Sigma1(e)+ eor w12,w12,w27,ror#13+ add w26,w26,w17 // h+=Ch(e,f,g)+ and w19,w19,w28 // (b^c)&=(a^b)+ eor w10,w10,w3,ror#19+ eor w11,w11,w6,lsr#3 // sigma0(X[i+1])+ add w26,w26,w16 // h+=Sigma1(e)+ eor w19,w19,w20 // Maj(a,b,c)+ eor w17,w12,w27,ror#22 // Sigma0(a)+ eor w10,w10,w3,lsr#10 // sigma1(X[i+14])+ add w5,w5,w14+ add w22,w22,w26 // d+=h+ add w26,w26,w19 // h+=Maj(a,b,c)+ ldr w19,[x30],#4 // *K++, w28 in next round+ add w5,w5,w11+ add w26,w26,w17 // h+=Sigma0(a)+ add w5,w5,w10+ ldr w10,[sp,#12]+ str w13,[sp,#8]+ ror w16,w22,#6+ add w25,w25,w19 // h+=K[i]+ ror w12,w7,#7+ and w17,w23,w22+ ror w11,w4,#17+ bic w19,w24,w22+ ror w13,w26,#2+ add w25,w25,w5 // h+=X[i]+ eor w16,w16,w22,ror#11+ eor w12,w12,w7,ror#18+ orr w17,w17,w19 // Ch(e,f,g)+ eor w19,w26,w27 // a^b, b^c in next round+ eor w16,w16,w22,ror#25 // Sigma1(e)+ eor w13,w13,w26,ror#13+ add w25,w25,w17 // h+=Ch(e,f,g)+ and w28,w28,w19 // (b^c)&=(a^b)+ eor w11,w11,w4,ror#19+ eor w12,w12,w7,lsr#3 // sigma0(X[i+1])+ add w25,w25,w16 // h+=Sigma1(e)+ eor w28,w28,w27 // Maj(a,b,c)+ eor w17,w13,w26,ror#22 // Sigma0(a)+ eor w11,w11,w4,lsr#10 // sigma1(X[i+14])+ add w6,w6,w15+ add w21,w21,w25 // d+=h+ add w25,w25,w28 // h+=Maj(a,b,c)+ ldr w28,[x30],#4 // *K++, w19 in next round+ add w6,w6,w12+ add w25,w25,w17 // h+=Sigma0(a)+ add w6,w6,w11+ ldr w11,[sp,#0]+ str w14,[sp,#12]+ ror w16,w21,#6+ add w24,w24,w28 // h+=K[i]+ ror w13,w8,#7+ and w17,w22,w21+ ror w12,w5,#17+ bic w28,w23,w21+ ror w14,w25,#2+ add w24,w24,w6 // h+=X[i]+ eor w16,w16,w21,ror#11+ eor w13,w13,w8,ror#18+ orr w17,w17,w28 // Ch(e,f,g)+ eor w28,w25,w26 // a^b, b^c in next round+ eor w16,w16,w21,ror#25 // Sigma1(e)+ eor w14,w14,w25,ror#13+ add w24,w24,w17 // h+=Ch(e,f,g)+ and w19,w19,w28 // (b^c)&=(a^b)+ eor w12,w12,w5,ror#19+ eor w13,w13,w8,lsr#3 // sigma0(X[i+1])+ add w24,w24,w16 // h+=Sigma1(e)+ eor w19,w19,w26 // Maj(a,b,c)+ eor w17,w14,w25,ror#22 // Sigma0(a)+ eor w12,w12,w5,lsr#10 // sigma1(X[i+14])+ add w7,w7,w0+ add w20,w20,w24 // d+=h+ add w24,w24,w19 // h+=Maj(a,b,c)+ ldr w19,[x30],#4 // *K++, w28 in next round+ add w7,w7,w13+ add w24,w24,w17 // h+=Sigma0(a)+ add w7,w7,w12+ ldr w12,[sp,#4]+ str w15,[sp,#0]+ ror w16,w20,#6+ add w23,w23,w19 // h+=K[i]+ ror w14,w9,#7+ and w17,w21,w20+ ror w13,w6,#17+ bic w19,w22,w20+ ror w15,w24,#2+ add w23,w23,w7 // h+=X[i]+ eor w16,w16,w20,ror#11+ eor w14,w14,w9,ror#18+ orr w17,w17,w19 // Ch(e,f,g)+ eor w19,w24,w25 // a^b, b^c in next round+ eor w16,w16,w20,ror#25 // Sigma1(e)+ eor w15,w15,w24,ror#13+ add w23,w23,w17 // h+=Ch(e,f,g)+ and w28,w28,w19 // (b^c)&=(a^b)+ eor w13,w13,w6,ror#19+ eor w14,w14,w9,lsr#3 // sigma0(X[i+1])+ add w23,w23,w16 // h+=Sigma1(e)+ eor w28,w28,w25 // Maj(a,b,c)+ eor w17,w15,w24,ror#22 // Sigma0(a)+ eor w13,w13,w6,lsr#10 // sigma1(X[i+14])+ add w8,w8,w1+ add w27,w27,w23 // d+=h+ add w23,w23,w28 // h+=Maj(a,b,c)+ ldr w28,[x30],#4 // *K++, w19 in next round+ add w8,w8,w14+ add w23,w23,w17 // h+=Sigma0(a)+ add w8,w8,w13+ ldr w13,[sp,#8]+ str w0,[sp,#4]+ ror w16,w27,#6+ add w22,w22,w28 // h+=K[i]+ ror w15,w10,#7+ and w17,w20,w27+ ror w14,w7,#17+ bic w28,w21,w27+ ror w0,w23,#2+ add w22,w22,w8 // h+=X[i]+ eor w16,w16,w27,ror#11+ eor w15,w15,w10,ror#18+ orr w17,w17,w28 // Ch(e,f,g)+ eor w28,w23,w24 // a^b, b^c in next round+ eor w16,w16,w27,ror#25 // Sigma1(e)+ eor w0,w0,w23,ror#13+ add w22,w22,w17 // h+=Ch(e,f,g)+ and w19,w19,w28 // (b^c)&=(a^b)+ eor w14,w14,w7,ror#19+ eor w15,w15,w10,lsr#3 // sigma0(X[i+1])+ add w22,w22,w16 // h+=Sigma1(e)+ eor w19,w19,w24 // Maj(a,b,c)+ eor w17,w0,w23,ror#22 // Sigma0(a)+ eor w14,w14,w7,lsr#10 // sigma1(X[i+14])+ add w9,w9,w2+ add w26,w26,w22 // d+=h+ add w22,w22,w19 // h+=Maj(a,b,c)+ ldr w19,[x30],#4 // *K++, w28 in next round+ add w9,w9,w15+ add w22,w22,w17 // h+=Sigma0(a)+ add w9,w9,w14+ ldr w14,[sp,#12]+ str w1,[sp,#8]+ ror w16,w26,#6+ add w21,w21,w19 // h+=K[i]+ ror w0,w11,#7+ and w17,w27,w26+ ror w15,w8,#17+ bic w19,w20,w26+ ror w1,w22,#2+ add w21,w21,w9 // h+=X[i]+ eor w16,w16,w26,ror#11+ eor w0,w0,w11,ror#18+ orr w17,w17,w19 // Ch(e,f,g)+ eor w19,w22,w23 // a^b, b^c in next round+ eor w16,w16,w26,ror#25 // Sigma1(e)+ eor w1,w1,w22,ror#13+ add w21,w21,w17 // h+=Ch(e,f,g)+ and w28,w28,w19 // (b^c)&=(a^b)+ eor w15,w15,w8,ror#19+ eor w0,w0,w11,lsr#3 // sigma0(X[i+1])+ add w21,w21,w16 // h+=Sigma1(e)+ eor w28,w28,w23 // Maj(a,b,c)+ eor w17,w1,w22,ror#22 // Sigma0(a)+ eor w15,w15,w8,lsr#10 // sigma1(X[i+14])+ add w10,w10,w3+ add w25,w25,w21 // d+=h+ add w21,w21,w28 // h+=Maj(a,b,c)+ ldr w28,[x30],#4 // *K++, w19 in next round+ add w10,w10,w0+ add w21,w21,w17 // h+=Sigma0(a)+ add w10,w10,w15+ ldr w15,[sp,#0]+ str w2,[sp,#12]+ ror w16,w25,#6+ add w20,w20,w28 // h+=K[i]+ ror w1,w12,#7+ and w17,w26,w25+ ror w0,w9,#17+ bic w28,w27,w25+ ror w2,w21,#2+ add w20,w20,w10 // h+=X[i]+ eor w16,w16,w25,ror#11+ eor w1,w1,w12,ror#18+ orr w17,w17,w28 // Ch(e,f,g)+ eor w28,w21,w22 // a^b, b^c in next round+ eor w16,w16,w25,ror#25 // Sigma1(e)+ eor w2,w2,w21,ror#13+ add w20,w20,w17 // h+=Ch(e,f,g)+ and w19,w19,w28 // (b^c)&=(a^b)+ eor w0,w0,w9,ror#19+ eor w1,w1,w12,lsr#3 // sigma0(X[i+1])+ add w20,w20,w16 // h+=Sigma1(e)+ eor w19,w19,w22 // Maj(a,b,c)+ eor w17,w2,w21,ror#22 // Sigma0(a)+ eor w0,w0,w9,lsr#10 // sigma1(X[i+14])+ add w11,w11,w4+ add w24,w24,w20 // d+=h+ add w20,w20,w19 // h+=Maj(a,b,c)+ ldr w19,[x30],#4 // *K++, w28 in next round+ add w11,w11,w1+ add w20,w20,w17 // h+=Sigma0(a)+ add w11,w11,w0+ ldr w0,[sp,#4]+ str w3,[sp,#0]+ ror w16,w24,#6+ add w27,w27,w19 // h+=K[i]+ ror w2,w13,#7+ and w17,w25,w24+ ror w1,w10,#17+ bic w19,w26,w24+ ror w3,w20,#2+ add w27,w27,w11 // h+=X[i]+ eor w16,w16,w24,ror#11+ eor w2,w2,w13,ror#18+ orr w17,w17,w19 // Ch(e,f,g)+ eor w19,w20,w21 // a^b, b^c in next round+ eor w16,w16,w24,ror#25 // Sigma1(e)+ eor w3,w3,w20,ror#13+ add w27,w27,w17 // h+=Ch(e,f,g)+ and w28,w28,w19 // (b^c)&=(a^b)+ eor w1,w1,w10,ror#19+ eor w2,w2,w13,lsr#3 // sigma0(X[i+1])+ add w27,w27,w16 // h+=Sigma1(e)+ eor w28,w28,w21 // Maj(a,b,c)+ eor w17,w3,w20,ror#22 // Sigma0(a)+ eor w1,w1,w10,lsr#10 // sigma1(X[i+14])+ add w12,w12,w5+ add w23,w23,w27 // d+=h+ add w27,w27,w28 // h+=Maj(a,b,c)+ ldr w28,[x30],#4 // *K++, w19 in next round+ add w12,w12,w2+ add w27,w27,w17 // h+=Sigma0(a)+ add w12,w12,w1+ ldr w1,[sp,#8]+ str w4,[sp,#4]+ ror w16,w23,#6+ add w26,w26,w28 // h+=K[i]+ ror w3,w14,#7+ and w17,w24,w23+ ror w2,w11,#17+ bic w28,w25,w23+ ror w4,w27,#2+ add w26,w26,w12 // h+=X[i]+ eor w16,w16,w23,ror#11+ eor w3,w3,w14,ror#18+ orr w17,w17,w28 // Ch(e,f,g)+ eor w28,w27,w20 // a^b, b^c in next round+ eor w16,w16,w23,ror#25 // Sigma1(e)+ eor w4,w4,w27,ror#13+ add w26,w26,w17 // h+=Ch(e,f,g)+ and w19,w19,w28 // (b^c)&=(a^b)+ eor w2,w2,w11,ror#19+ eor w3,w3,w14,lsr#3 // sigma0(X[i+1])+ add w26,w26,w16 // h+=Sigma1(e)+ eor w19,w19,w20 // Maj(a,b,c)+ eor w17,w4,w27,ror#22 // Sigma0(a)+ eor w2,w2,w11,lsr#10 // sigma1(X[i+14])+ add w13,w13,w6+ add w22,w22,w26 // d+=h+ add w26,w26,w19 // h+=Maj(a,b,c)+ ldr w19,[x30],#4 // *K++, w28 in next round+ add w13,w13,w3+ add w26,w26,w17 // h+=Sigma0(a)+ add w13,w13,w2+ ldr w2,[sp,#12]+ str w5,[sp,#8]+ ror w16,w22,#6+ add w25,w25,w19 // h+=K[i]+ ror w4,w15,#7+ and w17,w23,w22+ ror w3,w12,#17+ bic w19,w24,w22+ ror w5,w26,#2+ add w25,w25,w13 // h+=X[i]+ eor w16,w16,w22,ror#11+ eor w4,w4,w15,ror#18+ orr w17,w17,w19 // Ch(e,f,g)+ eor w19,w26,w27 // a^b, b^c in next round+ eor w16,w16,w22,ror#25 // Sigma1(e)+ eor w5,w5,w26,ror#13+ add w25,w25,w17 // h+=Ch(e,f,g)+ and w28,w28,w19 // (b^c)&=(a^b)+ eor w3,w3,w12,ror#19+ eor w4,w4,w15,lsr#3 // sigma0(X[i+1])+ add w25,w25,w16 // h+=Sigma1(e)+ eor w28,w28,w27 // Maj(a,b,c)+ eor w17,w5,w26,ror#22 // Sigma0(a)+ eor w3,w3,w12,lsr#10 // sigma1(X[i+14])+ add w14,w14,w7+ add w21,w21,w25 // d+=h+ add w25,w25,w28 // h+=Maj(a,b,c)+ ldr w28,[x30],#4 // *K++, w19 in next round+ add w14,w14,w4+ add w25,w25,w17 // h+=Sigma0(a)+ add w14,w14,w3+ ldr w3,[sp,#0]+ str w6,[sp,#12]+ ror w16,w21,#6+ add w24,w24,w28 // h+=K[i]+ ror w5,w0,#7+ and w17,w22,w21+ ror w4,w13,#17+ bic w28,w23,w21+ ror w6,w25,#2+ add w24,w24,w14 // h+=X[i]+ eor w16,w16,w21,ror#11+ eor w5,w5,w0,ror#18+ orr w17,w17,w28 // Ch(e,f,g)+ eor w28,w25,w26 // a^b, b^c in next round+ eor w16,w16,w21,ror#25 // Sigma1(e)+ eor w6,w6,w25,ror#13+ add w24,w24,w17 // h+=Ch(e,f,g)+ and w19,w19,w28 // (b^c)&=(a^b)+ eor w4,w4,w13,ror#19+ eor w5,w5,w0,lsr#3 // sigma0(X[i+1])+ add w24,w24,w16 // h+=Sigma1(e)+ eor w19,w19,w26 // Maj(a,b,c)+ eor w17,w6,w25,ror#22 // Sigma0(a)+ eor w4,w4,w13,lsr#10 // sigma1(X[i+14])+ add w15,w15,w8+ add w20,w20,w24 // d+=h+ add w24,w24,w19 // h+=Maj(a,b,c)+ ldr w19,[x30],#4 // *K++, w28 in next round+ add w15,w15,w5+ add w24,w24,w17 // h+=Sigma0(a)+ add w15,w15,w4+ ldr w4,[sp,#4]+ str w7,[sp,#0]+ ror w16,w20,#6+ add w23,w23,w19 // h+=K[i]+ ror w6,w1,#7+ and w17,w21,w20+ ror w5,w14,#17+ bic w19,w22,w20+ ror w7,w24,#2+ add w23,w23,w15 // h+=X[i]+ eor w16,w16,w20,ror#11+ eor w6,w6,w1,ror#18+ orr w17,w17,w19 // Ch(e,f,g)+ eor w19,w24,w25 // a^b, b^c in next round+ eor w16,w16,w20,ror#25 // Sigma1(e)+ eor w7,w7,w24,ror#13+ add w23,w23,w17 // h+=Ch(e,f,g)+ and w28,w28,w19 // (b^c)&=(a^b)+ eor w5,w5,w14,ror#19+ eor w6,w6,w1,lsr#3 // sigma0(X[i+1])+ add w23,w23,w16 // h+=Sigma1(e)+ eor w28,w28,w25 // Maj(a,b,c)+ eor w17,w7,w24,ror#22 // Sigma0(a)+ eor w5,w5,w14,lsr#10 // sigma1(X[i+14])+ add w0,w0,w9+ add w27,w27,w23 // d+=h+ add w23,w23,w28 // h+=Maj(a,b,c)+ ldr w28,[x30],#4 // *K++, w19 in next round+ add w0,w0,w6+ add w23,w23,w17 // h+=Sigma0(a)+ add w0,w0,w5+ ldr w5,[sp,#8]+ str w8,[sp,#4]+ ror w16,w27,#6+ add w22,w22,w28 // h+=K[i]+ ror w7,w2,#7+ and w17,w20,w27+ ror w6,w15,#17+ bic w28,w21,w27+ ror w8,w23,#2+ add w22,w22,w0 // h+=X[i]+ eor w16,w16,w27,ror#11+ eor w7,w7,w2,ror#18+ orr w17,w17,w28 // Ch(e,f,g)+ eor w28,w23,w24 // a^b, b^c in next round+ eor w16,w16,w27,ror#25 // Sigma1(e)+ eor w8,w8,w23,ror#13+ add w22,w22,w17 // h+=Ch(e,f,g)+ and w19,w19,w28 // (b^c)&=(a^b)+ eor w6,w6,w15,ror#19+ eor w7,w7,w2,lsr#3 // sigma0(X[i+1])+ add w22,w22,w16 // h+=Sigma1(e)+ eor w19,w19,w24 // Maj(a,b,c)+ eor w17,w8,w23,ror#22 // Sigma0(a)+ eor w6,w6,w15,lsr#10 // sigma1(X[i+14])+ add w1,w1,w10+ add w26,w26,w22 // d+=h+ add w22,w22,w19 // h+=Maj(a,b,c)+ ldr w19,[x30],#4 // *K++, w28 in next round+ add w1,w1,w7+ add w22,w22,w17 // h+=Sigma0(a)+ add w1,w1,w6+ ldr w6,[sp,#12]+ str w9,[sp,#8]+ ror w16,w26,#6+ add w21,w21,w19 // h+=K[i]+ ror w8,w3,#7+ and w17,w27,w26+ ror w7,w0,#17+ bic w19,w20,w26+ ror w9,w22,#2+ add w21,w21,w1 // h+=X[i]+ eor w16,w16,w26,ror#11+ eor w8,w8,w3,ror#18+ orr w17,w17,w19 // Ch(e,f,g)+ eor w19,w22,w23 // a^b, b^c in next round+ eor w16,w16,w26,ror#25 // Sigma1(e)+ eor w9,w9,w22,ror#13+ add w21,w21,w17 // h+=Ch(e,f,g)+ and w28,w28,w19 // (b^c)&=(a^b)+ eor w7,w7,w0,ror#19+ eor w8,w8,w3,lsr#3 // sigma0(X[i+1])+ add w21,w21,w16 // h+=Sigma1(e)+ eor w28,w28,w23 // Maj(a,b,c)+ eor w17,w9,w22,ror#22 // Sigma0(a)+ eor w7,w7,w0,lsr#10 // sigma1(X[i+14])+ add w2,w2,w11+ add w25,w25,w21 // d+=h+ add w21,w21,w28 // h+=Maj(a,b,c)+ ldr w28,[x30],#4 // *K++, w19 in next round+ add w2,w2,w8+ add w21,w21,w17 // h+=Sigma0(a)+ add w2,w2,w7+ ldr w7,[sp,#0]+ str w10,[sp,#12]+ ror w16,w25,#6+ add w20,w20,w28 // h+=K[i]+ ror w9,w4,#7+ and w17,w26,w25+ ror w8,w1,#17+ bic w28,w27,w25+ ror w10,w21,#2+ add w20,w20,w2 // h+=X[i]+ eor w16,w16,w25,ror#11+ eor w9,w9,w4,ror#18+ orr w17,w17,w28 // Ch(e,f,g)+ eor w28,w21,w22 // a^b, b^c in next round+ eor w16,w16,w25,ror#25 // Sigma1(e)+ eor w10,w10,w21,ror#13+ add w20,w20,w17 // h+=Ch(e,f,g)+ and w19,w19,w28 // (b^c)&=(a^b)+ eor w8,w8,w1,ror#19+ eor w9,w9,w4,lsr#3 // sigma0(X[i+1])+ add w20,w20,w16 // h+=Sigma1(e)+ eor w19,w19,w22 // Maj(a,b,c)+ eor w17,w10,w21,ror#22 // Sigma0(a)+ eor w8,w8,w1,lsr#10 // sigma1(X[i+14])+ add w3,w3,w12+ add w24,w24,w20 // d+=h+ add w20,w20,w19 // h+=Maj(a,b,c)+ ldr w19,[x30],#4 // *K++, w28 in next round+ add w3,w3,w9+ add w20,w20,w17 // h+=Sigma0(a)+ add w3,w3,w8+ cbnz w19,.Loop_16_xx++ ldp x0,x2,[x29,#12*__SIZEOF_POINTER__]+ ldr x1,[x29,#14*__SIZEOF_POINTER__]+ sub x30,x30,#260++ ldp w3,w4,[x0]+ ldp w5,w6,[x0,#2*4]+ add x1,x1,#14*4+ ldp w7,w8,[x0,#4*4]+ add w20,w20,w3+ ldp w9,w10,[x0,#6*4]+ add w21,w21,w4+ add w22,w22,w5+ add w23,w23,w6+ stp w20,w21,[x0]+ add w24,w24,w7+ add w25,w25,w8+ stp w22,w23,[x0,#2*4]+ add w26,w26,w9+ add w27,w27,w10+ cmp x1,x2+ stp w24,w25,[x0,#4*4]+ stp w26,w27,[x0,#6*4]+ b.ne .Loop++ ldp x19,x20,[x29,#2*__SIZEOF_POINTER__]+ add sp,sp,#4*4+ ldp x21,x22,[x29,#4*__SIZEOF_POINTER__]+ ldp x23,x24,[x29,#6*__SIZEOF_POINTER__]+ ldp x25,x26,[x29,#8*__SIZEOF_POINTER__]+ ldp x27,x28,[x29,#10*__SIZEOF_POINTER__]+ ldp x29,x30,[sp],#16*__SIZEOF_POINTER__+.inst 0xd50323bf // autiasp+ ret+.size crypton_sha256_asm_block_data_order,.-crypton_sha256_asm_block_data_order++.align 6+.type .LK256,%object+.LK256:+.long 0x428a2f98,0x71374491,0xb5c0fbcf,0xe9b5dba5+.long 0x3956c25b,0x59f111f1,0x923f82a4,0xab1c5ed5+.long 0xd807aa98,0x12835b01,0x243185be,0x550c7dc3+.long 0x72be5d74,0x80deb1fe,0x9bdc06a7,0xc19bf174+.long 0xe49b69c1,0xefbe4786,0x0fc19dc6,0x240ca1cc+.long 0x2de92c6f,0x4a7484aa,0x5cb0a9dc,0x76f988da+.long 0x983e5152,0xa831c66d,0xb00327c8,0xbf597fc7+.long 0xc6e00bf3,0xd5a79147,0x06ca6351,0x14292967+.long 0x27b70a85,0x2e1b2138,0x4d2c6dfc,0x53380d13+.long 0x650a7354,0x766a0abb,0x81c2c92e,0x92722c85+.long 0xa2bfe8a1,0xa81a664b,0xc24b8b70,0xc76c51a3+.long 0xd192e819,0xd6990624,0xf40e3585,0x106aa070+.long 0x19a4c116,0x1e376c08,0x2748774c,0x34b0bcb5+.long 0x391c0cb3,0x4ed8aa4a,0x5b9cca4f,0x682e6ff3+.long 0x748f82ee,0x78a5636f,0x84c87814,0x8cc70208+.long 0x90befffa,0xa4506ceb,0xbef9a3f7,0xc67178f2+.long 0 //terminator+.size .LK256,.-.LK256+.byte 83,72,65,50,53,54,32,98,108,111,99,107,32,116,114,97,110,115,102,111,114,109,32,102,111,114,32,65,82,77,118,56,44,32,67,82,89,80,84,79,71,65,77,83,32,98,121,32,64,100,111,116,45,97,115,109,0+.align 2+.align 2+#ifndef __KERNEL__+.type crypton_sha256_asm_block_armv8,%function+.align 6+crypton_sha256_asm_block_armv8:+.Lv8_entry:+ stp x29,x30,[sp,#-2*__SIZEOF_POINTER__]!+ add x29,sp,#0++ ld1 {v0.4s,v1.4s},[x0]+ adr x3,.LK256++.Loop_hw:+ ld1 {v4.16b,v5.16b,v6.16b,v7.16b},[x1],#64+ sub x2,x2,#1+ ld1 {v16.4s},[x3],#16+ rev32 v4.16b,v4.16b+ rev32 v5.16b,v5.16b+ rev32 v6.16b,v6.16b+ rev32 v7.16b,v7.16b+ orr v18.16b,v0.16b,v0.16b // offload+ orr v19.16b,v1.16b,v1.16b+ ld1 {v17.4s},[x3],#16+ add v16.4s,v16.4s,v4.4s+.inst 0x5e2828a4 //sha256su0 v4.16b,v5.16b+ orr v2.16b,v0.16b,v0.16b+.inst 0x5e104020 //sha256h v0.16b,v1.16b,v16.4s+.inst 0x5e105041 //sha256h2 v1.16b,v2.16b,v16.4s+.inst 0x5e0760c4 //sha256su1 v4.16b,v6.16b,v7.16b+ ld1 {v16.4s},[x3],#16+ add v17.4s,v17.4s,v5.4s+.inst 0x5e2828c5 //sha256su0 v5.16b,v6.16b+ orr v2.16b,v0.16b,v0.16b+.inst 0x5e114020 //sha256h v0.16b,v1.16b,v17.4s+.inst 0x5e115041 //sha256h2 v1.16b,v2.16b,v17.4s+.inst 0x5e0460e5 //sha256su1 v5.16b,v7.16b,v4.16b+ ld1 {v17.4s},[x3],#16+ add v16.4s,v16.4s,v6.4s+.inst 0x5e2828e6 //sha256su0 v6.16b,v7.16b+ orr v2.16b,v0.16b,v0.16b+.inst 0x5e104020 //sha256h v0.16b,v1.16b,v16.4s+.inst 0x5e105041 //sha256h2 v1.16b,v2.16b,v16.4s+.inst 0x5e056086 //sha256su1 v6.16b,v4.16b,v5.16b+ ld1 {v16.4s},[x3],#16+ add v17.4s,v17.4s,v7.4s+.inst 0x5e282887 //sha256su0 v7.16b,v4.16b+ orr v2.16b,v0.16b,v0.16b+.inst 0x5e114020 //sha256h v0.16b,v1.16b,v17.4s+.inst 0x5e115041 //sha256h2 v1.16b,v2.16b,v17.4s+.inst 0x5e0660a7 //sha256su1 v7.16b,v5.16b,v6.16b+ ld1 {v17.4s},[x3],#16+ add v16.4s,v16.4s,v4.4s+.inst 0x5e2828a4 //sha256su0 v4.16b,v5.16b+ orr v2.16b,v0.16b,v0.16b+.inst 0x5e104020 //sha256h v0.16b,v1.16b,v16.4s+.inst 0x5e105041 //sha256h2 v1.16b,v2.16b,v16.4s+.inst 0x5e0760c4 //sha256su1 v4.16b,v6.16b,v7.16b+ ld1 {v16.4s},[x3],#16+ add v17.4s,v17.4s,v5.4s+.inst 0x5e2828c5 //sha256su0 v5.16b,v6.16b+ orr v2.16b,v0.16b,v0.16b+.inst 0x5e114020 //sha256h v0.16b,v1.16b,v17.4s+.inst 0x5e115041 //sha256h2 v1.16b,v2.16b,v17.4s+.inst 0x5e0460e5 //sha256su1 v5.16b,v7.16b,v4.16b+ ld1 {v17.4s},[x3],#16+ add v16.4s,v16.4s,v6.4s+.inst 0x5e2828e6 //sha256su0 v6.16b,v7.16b+ orr v2.16b,v0.16b,v0.16b+.inst 0x5e104020 //sha256h v0.16b,v1.16b,v16.4s+.inst 0x5e105041 //sha256h2 v1.16b,v2.16b,v16.4s+.inst 0x5e056086 //sha256su1 v6.16b,v4.16b,v5.16b+ ld1 {v16.4s},[x3],#16+ add v17.4s,v17.4s,v7.4s+.inst 0x5e282887 //sha256su0 v7.16b,v4.16b+ orr v2.16b,v0.16b,v0.16b+.inst 0x5e114020 //sha256h v0.16b,v1.16b,v17.4s+.inst 0x5e115041 //sha256h2 v1.16b,v2.16b,v17.4s+.inst 0x5e0660a7 //sha256su1 v7.16b,v5.16b,v6.16b+ ld1 {v17.4s},[x3],#16+ add v16.4s,v16.4s,v4.4s+.inst 0x5e2828a4 //sha256su0 v4.16b,v5.16b+ orr v2.16b,v0.16b,v0.16b+.inst 0x5e104020 //sha256h v0.16b,v1.16b,v16.4s+.inst 0x5e105041 //sha256h2 v1.16b,v2.16b,v16.4s+.inst 0x5e0760c4 //sha256su1 v4.16b,v6.16b,v7.16b+ ld1 {v16.4s},[x3],#16+ add v17.4s,v17.4s,v5.4s+.inst 0x5e2828c5 //sha256su0 v5.16b,v6.16b+ orr v2.16b,v0.16b,v0.16b+.inst 0x5e114020 //sha256h v0.16b,v1.16b,v17.4s+.inst 0x5e115041 //sha256h2 v1.16b,v2.16b,v17.4s+.inst 0x5e0460e5 //sha256su1 v5.16b,v7.16b,v4.16b+ ld1 {v17.4s},[x3],#16+ add v16.4s,v16.4s,v6.4s+.inst 0x5e2828e6 //sha256su0 v6.16b,v7.16b+ orr v2.16b,v0.16b,v0.16b+.inst 0x5e104020 //sha256h v0.16b,v1.16b,v16.4s+.inst 0x5e105041 //sha256h2 v1.16b,v2.16b,v16.4s+.inst 0x5e056086 //sha256su1 v6.16b,v4.16b,v5.16b+ ld1 {v16.4s},[x3],#16+ add v17.4s,v17.4s,v7.4s+.inst 0x5e282887 //sha256su0 v7.16b,v4.16b+ orr v2.16b,v0.16b,v0.16b+.inst 0x5e114020 //sha256h v0.16b,v1.16b,v17.4s+.inst 0x5e115041 //sha256h2 v1.16b,v2.16b,v17.4s+.inst 0x5e0660a7 //sha256su1 v7.16b,v5.16b,v6.16b+ ld1 {v17.4s},[x3],#16+ add v16.4s,v16.4s,v4.4s+ orr v2.16b,v0.16b,v0.16b+.inst 0x5e104020 //sha256h v0.16b,v1.16b,v16.4s+.inst 0x5e105041 //sha256h2 v1.16b,v2.16b,v16.4s++ ld1 {v16.4s},[x3],#16+ add v17.4s,v17.4s,v5.4s+ orr v2.16b,v0.16b,v0.16b+.inst 0x5e114020 //sha256h v0.16b,v1.16b,v17.4s+.inst 0x5e115041 //sha256h2 v1.16b,v2.16b,v17.4s++ ld1 {v17.4s},[x3]+ add v16.4s,v16.4s,v6.4s+ sub x3,x3,#64*4-16+ orr v2.16b,v0.16b,v0.16b+.inst 0x5e104020 //sha256h v0.16b,v1.16b,v16.4s+.inst 0x5e105041 //sha256h2 v1.16b,v2.16b,v16.4s++ add v17.4s,v17.4s,v7.4s+ orr v2.16b,v0.16b,v0.16b+.inst 0x5e114020 //sha256h v0.16b,v1.16b,v17.4s+.inst 0x5e115041 //sha256h2 v1.16b,v2.16b,v17.4s++ add v0.4s,v0.4s,v18.4s+ add v1.4s,v1.4s,v19.4s++ cbnz x2,.Loop_hw++ st1 {v0.4s,v1.4s},[x0]++ ldr x29,[sp],#2*__SIZEOF_POINTER__+ ret+.size crypton_sha256_asm_block_armv8,.-crypton_sha256_asm_block_armv8+#endif+#ifdef __KERNEL__+.globl crypton_sha256_asm_block_neon+#endif+.type crypton_sha256_asm_block_neon,%function+.align 4+crypton_sha256_asm_block_neon:+.Lneon_entry:+ stp x29, x30, [sp, #-2*__SIZEOF_POINTER__]!+ mov x29, sp+ sub sp,sp,#16*4++ adr x16,.LK256+ add x2,x1,x2,lsl#6 // len to point at the end of inp++ ld1 {v0.16b},[x1], #16+ ld1 {v1.16b},[x1], #16+ ld1 {v2.16b},[x1], #16+ ld1 {v3.16b},[x1], #16+ ld1 {v4.4s},[x16], #16+ ld1 {v5.4s},[x16], #16+ ld1 {v6.4s},[x16], #16+ ld1 {v7.4s},[x16], #16+ rev32 v0.16b,v0.16b // yes, even on+ rev32 v1.16b,v1.16b // big-endian+ rev32 v2.16b,v2.16b+ rev32 v3.16b,v3.16b+ mov x17,sp+ add v4.4s,v4.4s,v0.4s+ add v5.4s,v5.4s,v1.4s+ add v6.4s,v6.4s,v2.4s+ st1 {v4.4s,v5.4s},[x17], #32+ add v7.4s,v7.4s,v3.4s+ st1 {v6.4s,v7.4s},[x17]+ sub x17,x17,#32++ ldp w3,w4,[x0]+ ldp w5,w6,[x0,#8]+ ldp w7,w8,[x0,#16]+ ldp w9,w10,[x0,#24]+ ldr w12,[sp,#0]+ mov w13,wzr+ eor w14,w4,w5+ mov w15,wzr+ b .L_00_48++.align 4+.L_00_48:+ ext v4.16b,v0.16b,v1.16b,#4+ add w10,w10,w12+ add w3,w3,w15+ and w12,w8,w7+ bic w15,w9,w7+ ext v7.16b,v2.16b,v3.16b,#4+ eor w11,w7,w7,ror#5+ add w3,w3,w13+ mov d19,v3.d[1]+ orr w12,w12,w15+ eor w11,w11,w7,ror#19+ ushr v6.4s,v4.4s,#7+ eor w15,w3,w3,ror#11+ ushr v5.4s,v4.4s,#3+ add w10,w10,w12+ add v0.4s,v0.4s,v7.4s+ ror w11,w11,#6+ sli v6.4s,v4.4s,#25+ eor w13,w3,w4+ eor w15,w15,w3,ror#20+ ushr v7.4s,v4.4s,#18+ add w10,w10,w11+ ldr w12,[sp,#4]+ and w14,w14,w13+ eor v5.16b,v5.16b,v6.16b+ ror w15,w15,#2+ add w6,w6,w10+ sli v7.4s,v4.4s,#14+ eor w14,w14,w4+ ushr v16.4s,v19.4s,#17+ add w9,w9,w12+ add w10,w10,w15+ and w12,w7,w6+ eor v5.16b,v5.16b,v7.16b+ bic w15,w8,w6+ eor w11,w6,w6,ror#5+ sli v16.4s,v19.4s,#15+ add w10,w10,w14+ orr w12,w12,w15+ ushr v17.4s,v19.4s,#10+ eor w11,w11,w6,ror#19+ eor w15,w10,w10,ror#11+ ushr v7.4s,v19.4s,#19+ add w9,w9,w12+ ror w11,w11,#6+ add v0.4s,v0.4s,v5.4s+ eor w14,w10,w3+ eor w15,w15,w10,ror#20+ sli v7.4s,v19.4s,#13+ add w9,w9,w11+ ldr w12,[sp,#8]+ and w13,w13,w14+ eor v17.16b,v17.16b,v16.16b+ ror w15,w15,#2+ add w5,w5,w9+ eor w13,w13,w3+ eor v17.16b,v17.16b,v7.16b+ add w8,w8,w12+ add w9,w9,w15+ and w12,w6,w5+ add v0.4s,v0.4s,v17.4s+ bic w15,w7,w5+ eor w11,w5,w5,ror#5+ add w9,w9,w13+ ushr v18.4s,v0.4s,#17+ orr w12,w12,w15+ ushr v19.4s,v0.4s,#10+ eor w11,w11,w5,ror#19+ eor w15,w9,w9,ror#11+ sli v18.4s,v0.4s,#15+ add w8,w8,w12+ ushr v17.4s,v0.4s,#19+ ror w11,w11,#6+ eor w13,w9,w10+ eor v19.16b,v19.16b,v18.16b+ eor w15,w15,w9,ror#20+ add w8,w8,w11+ sli v17.4s,v0.4s,#13+ ldr w12,[sp,#12]+ and w14,w14,w13+ ror w15,w15,#2+ ld1 {v4.4s},[x16], #16+ add w4,w4,w8+ eor v19.16b,v19.16b,v17.16b+ eor w14,w14,w10+ eor v17.16b,v17.16b,v17.16b+ add w7,w7,w12+ add w8,w8,w15+ and w12,w5,w4+ mov v17.d[1],v19.d[0]+ bic w15,w6,w4+ eor w11,w4,w4,ror#5+ add w8,w8,w14+ add v0.4s,v0.4s,v17.4s+ orr w12,w12,w15+ eor w11,w11,w4,ror#19+ eor w15,w8,w8,ror#11+ add v4.4s,v4.4s,v0.4s+ add w7,w7,w12+ ror w11,w11,#6+ eor w14,w8,w9+ eor w15,w15,w8,ror#20+ add w7,w7,w11+ ldr w12,[sp,#16]+ and w13,w13,w14+ ror w15,w15,#2+ add w3,w3,w7+ eor w13,w13,w9+ st1 {v4.4s},[x17], #16+ ext v4.16b,v1.16b,v2.16b,#4+ add w6,w6,w12+ add w7,w7,w15+ and w12,w4,w3+ bic w15,w5,w3+ ext v7.16b,v3.16b,v0.16b,#4+ eor w11,w3,w3,ror#5+ add w7,w7,w13+ mov d19,v0.d[1]+ orr w12,w12,w15+ eor w11,w11,w3,ror#19+ ushr v6.4s,v4.4s,#7+ eor w15,w7,w7,ror#11+ ushr v5.4s,v4.4s,#3+ add w6,w6,w12+ add v1.4s,v1.4s,v7.4s+ ror w11,w11,#6+ sli v6.4s,v4.4s,#25+ eor w13,w7,w8+ eor w15,w15,w7,ror#20+ ushr v7.4s,v4.4s,#18+ add w6,w6,w11+ ldr w12,[sp,#20]+ and w14,w14,w13+ eor v5.16b,v5.16b,v6.16b+ ror w15,w15,#2+ add w10,w10,w6+ sli v7.4s,v4.4s,#14+ eor w14,w14,w8+ ushr v16.4s,v19.4s,#17+ add w5,w5,w12+ add w6,w6,w15+ and w12,w3,w10+ eor v5.16b,v5.16b,v7.16b+ bic w15,w4,w10+ eor w11,w10,w10,ror#5+ sli v16.4s,v19.4s,#15+ add w6,w6,w14+ orr w12,w12,w15+ ushr v17.4s,v19.4s,#10+ eor w11,w11,w10,ror#19+ eor w15,w6,w6,ror#11+ ushr v7.4s,v19.4s,#19+ add w5,w5,w12+ ror w11,w11,#6+ add v1.4s,v1.4s,v5.4s+ eor w14,w6,w7+ eor w15,w15,w6,ror#20+ sli v7.4s,v19.4s,#13+ add w5,w5,w11+ ldr w12,[sp,#24]+ and w13,w13,w14+ eor v17.16b,v17.16b,v16.16b+ ror w15,w15,#2+ add w9,w9,w5+ eor w13,w13,w7+ eor v17.16b,v17.16b,v7.16b+ add w4,w4,w12+ add w5,w5,w15+ and w12,w10,w9+ add v1.4s,v1.4s,v17.4s+ bic w15,w3,w9+ eor w11,w9,w9,ror#5+ add w5,w5,w13+ ushr v18.4s,v1.4s,#17+ orr w12,w12,w15+ ushr v19.4s,v1.4s,#10+ eor w11,w11,w9,ror#19+ eor w15,w5,w5,ror#11+ sli v18.4s,v1.4s,#15+ add w4,w4,w12+ ushr v17.4s,v1.4s,#19+ ror w11,w11,#6+ eor w13,w5,w6+ eor v19.16b,v19.16b,v18.16b+ eor w15,w15,w5,ror#20+ add w4,w4,w11+ sli v17.4s,v1.4s,#13+ ldr w12,[sp,#28]+ and w14,w14,w13+ ror w15,w15,#2+ ld1 {v4.4s},[x16], #16+ add w8,w8,w4+ eor v19.16b,v19.16b,v17.16b+ eor w14,w14,w6+ eor v17.16b,v17.16b,v17.16b+ add w3,w3,w12+ add w4,w4,w15+ and w12,w9,w8+ mov v17.d[1],v19.d[0]+ bic w15,w10,w8+ eor w11,w8,w8,ror#5+ add w4,w4,w14+ add v1.4s,v1.4s,v17.4s+ orr w12,w12,w15+ eor w11,w11,w8,ror#19+ eor w15,w4,w4,ror#11+ add v4.4s,v4.4s,v1.4s+ add w3,w3,w12+ ror w11,w11,#6+ eor w14,w4,w5+ eor w15,w15,w4,ror#20+ add w3,w3,w11+ ldr w12,[sp,#32]+ and w13,w13,w14+ ror w15,w15,#2+ add w7,w7,w3+ eor w13,w13,w5+ st1 {v4.4s},[x17], #16+ ext v4.16b,v2.16b,v3.16b,#4+ add w10,w10,w12+ add w3,w3,w15+ and w12,w8,w7+ bic w15,w9,w7+ ext v7.16b,v0.16b,v1.16b,#4+ eor w11,w7,w7,ror#5+ add w3,w3,w13+ mov d19,v1.d[1]+ orr w12,w12,w15+ eor w11,w11,w7,ror#19+ ushr v6.4s,v4.4s,#7+ eor w15,w3,w3,ror#11+ ushr v5.4s,v4.4s,#3+ add w10,w10,w12+ add v2.4s,v2.4s,v7.4s+ ror w11,w11,#6+ sli v6.4s,v4.4s,#25+ eor w13,w3,w4+ eor w15,w15,w3,ror#20+ ushr v7.4s,v4.4s,#18+ add w10,w10,w11+ ldr w12,[sp,#36]+ and w14,w14,w13+ eor v5.16b,v5.16b,v6.16b+ ror w15,w15,#2+ add w6,w6,w10+ sli v7.4s,v4.4s,#14+ eor w14,w14,w4+ ushr v16.4s,v19.4s,#17+ add w9,w9,w12+ add w10,w10,w15+ and w12,w7,w6+ eor v5.16b,v5.16b,v7.16b+ bic w15,w8,w6+ eor w11,w6,w6,ror#5+ sli v16.4s,v19.4s,#15+ add w10,w10,w14+ orr w12,w12,w15+ ushr v17.4s,v19.4s,#10+ eor w11,w11,w6,ror#19+ eor w15,w10,w10,ror#11+ ushr v7.4s,v19.4s,#19+ add w9,w9,w12+ ror w11,w11,#6+ add v2.4s,v2.4s,v5.4s+ eor w14,w10,w3+ eor w15,w15,w10,ror#20+ sli v7.4s,v19.4s,#13+ add w9,w9,w11+ ldr w12,[sp,#40]+ and w13,w13,w14+ eor v17.16b,v17.16b,v16.16b+ ror w15,w15,#2+ add w5,w5,w9+ eor w13,w13,w3+ eor v17.16b,v17.16b,v7.16b+ add w8,w8,w12+ add w9,w9,w15+ and w12,w6,w5+ add v2.4s,v2.4s,v17.4s+ bic w15,w7,w5+ eor w11,w5,w5,ror#5+ add w9,w9,w13+ ushr v18.4s,v2.4s,#17+ orr w12,w12,w15+ ushr v19.4s,v2.4s,#10+ eor w11,w11,w5,ror#19+ eor w15,w9,w9,ror#11+ sli v18.4s,v2.4s,#15+ add w8,w8,w12+ ushr v17.4s,v2.4s,#19+ ror w11,w11,#6+ eor w13,w9,w10+ eor v19.16b,v19.16b,v18.16b+ eor w15,w15,w9,ror#20+ add w8,w8,w11+ sli v17.4s,v2.4s,#13+ ldr w12,[sp,#44]+ and w14,w14,w13+ ror w15,w15,#2+ ld1 {v4.4s},[x16], #16+ add w4,w4,w8+ eor v19.16b,v19.16b,v17.16b+ eor w14,w14,w10+ eor v17.16b,v17.16b,v17.16b+ add w7,w7,w12+ add w8,w8,w15+ and w12,w5,w4+ mov v17.d[1],v19.d[0]+ bic w15,w6,w4+ eor w11,w4,w4,ror#5+ add w8,w8,w14+ add v2.4s,v2.4s,v17.4s+ orr w12,w12,w15+ eor w11,w11,w4,ror#19+ eor w15,w8,w8,ror#11+ add v4.4s,v4.4s,v2.4s+ add w7,w7,w12+ ror w11,w11,#6+ eor w14,w8,w9+ eor w15,w15,w8,ror#20+ add w7,w7,w11+ ldr w12,[sp,#48]+ and w13,w13,w14+ ror w15,w15,#2+ add w3,w3,w7+ eor w13,w13,w9+ st1 {v4.4s},[x17], #16+ ext v4.16b,v3.16b,v0.16b,#4+ add w6,w6,w12+ add w7,w7,w15+ and w12,w4,w3+ bic w15,w5,w3+ ext v7.16b,v1.16b,v2.16b,#4+ eor w11,w3,w3,ror#5+ add w7,w7,w13+ mov d19,v2.d[1]+ orr w12,w12,w15+ eor w11,w11,w3,ror#19+ ushr v6.4s,v4.4s,#7+ eor w15,w7,w7,ror#11+ ushr v5.4s,v4.4s,#3+ add w6,w6,w12+ add v3.4s,v3.4s,v7.4s+ ror w11,w11,#6+ sli v6.4s,v4.4s,#25+ eor w13,w7,w8+ eor w15,w15,w7,ror#20+ ushr v7.4s,v4.4s,#18+ add w6,w6,w11+ ldr w12,[sp,#52]+ and w14,w14,w13+ eor v5.16b,v5.16b,v6.16b+ ror w15,w15,#2+ add w10,w10,w6+ sli v7.4s,v4.4s,#14+ eor w14,w14,w8+ ushr v16.4s,v19.4s,#17+ add w5,w5,w12+ add w6,w6,w15+ and w12,w3,w10+ eor v5.16b,v5.16b,v7.16b+ bic w15,w4,w10+ eor w11,w10,w10,ror#5+ sli v16.4s,v19.4s,#15+ add w6,w6,w14+ orr w12,w12,w15+ ushr v17.4s,v19.4s,#10+ eor w11,w11,w10,ror#19+ eor w15,w6,w6,ror#11+ ushr v7.4s,v19.4s,#19+ add w5,w5,w12+ ror w11,w11,#6+ add v3.4s,v3.4s,v5.4s+ eor w14,w6,w7+ eor w15,w15,w6,ror#20+ sli v7.4s,v19.4s,#13+ add w5,w5,w11+ ldr w12,[sp,#56]+ and w13,w13,w14+ eor v17.16b,v17.16b,v16.16b+ ror w15,w15,#2+ add w9,w9,w5+ eor w13,w13,w7+ eor v17.16b,v17.16b,v7.16b+ add w4,w4,w12+ add w5,w5,w15+ and w12,w10,w9+ add v3.4s,v3.4s,v17.4s+ bic w15,w3,w9+ eor w11,w9,w9,ror#5+ add w5,w5,w13+ ushr v18.4s,v3.4s,#17+ orr w12,w12,w15+ ushr v19.4s,v3.4s,#10+ eor w11,w11,w9,ror#19+ eor w15,w5,w5,ror#11+ sli v18.4s,v3.4s,#15+ add w4,w4,w12+ ushr v17.4s,v3.4s,#19+ ror w11,w11,#6+ eor w13,w5,w6+ eor v19.16b,v19.16b,v18.16b+ eor w15,w15,w5,ror#20+ add w4,w4,w11+ sli v17.4s,v3.4s,#13+ ldr w12,[sp,#60]+ and w14,w14,w13+ ror w15,w15,#2+ ld1 {v4.4s},[x16], #16+ add w8,w8,w4+ eor v19.16b,v19.16b,v17.16b+ eor w14,w14,w6+ eor v17.16b,v17.16b,v17.16b+ add w3,w3,w12+ add w4,w4,w15+ and w12,w9,w8+ mov v17.d[1],v19.d[0]+ bic w15,w10,w8+ eor w11,w8,w8,ror#5+ add w4,w4,w14+ add v3.4s,v3.4s,v17.4s+ orr w12,w12,w15+ eor w11,w11,w8,ror#19+ eor w15,w4,w4,ror#11+ add v4.4s,v4.4s,v3.4s+ add w3,w3,w12+ ror w11,w11,#6+ eor w14,w4,w5+ eor w15,w15,w4,ror#20+ add w3,w3,w11+ ldr w12,[x16]+ and w13,w13,w14+ ror w15,w15,#2+ add w7,w7,w3+ eor w13,w13,w5+ st1 {v4.4s},[x17], #16+ cmp w12,#0 // check for K256 terminator+ ldr w12,[sp,#0]+ sub x17,x17,#64+ bne .L_00_48++ sub x16,x16,#256+ cmp x1,x2+ mov x17, #-64+ csel x17, x17, xzr, eq+ add x1,x1,x17+ mov x17,sp+ add w10,w10,w12+ add w3,w3,w15+ and w12,w8,w7+ ld1 {v0.16b},[x1],#16+ bic w15,w9,w7+ eor w11,w7,w7,ror#5+ ld1 {v4.4s},[x16],#16+ add w3,w3,w13+ orr w12,w12,w15+ eor w11,w11,w7,ror#19+ eor w15,w3,w3,ror#11+ rev32 v0.16b,v0.16b+ add w10,w10,w12+ ror w11,w11,#6+ eor w13,w3,w4+ eor w15,w15,w3,ror#20+ add v4.4s,v4.4s,v0.4s+ add w10,w10,w11+ ldr w12,[sp,#4]+ and w14,w14,w13+ ror w15,w15,#2+ add w6,w6,w10+ eor w14,w14,w4+ add w9,w9,w12+ add w10,w10,w15+ and w12,w7,w6+ bic w15,w8,w6+ eor w11,w6,w6,ror#5+ add w10,w10,w14+ orr w12,w12,w15+ eor w11,w11,w6,ror#19+ eor w15,w10,w10,ror#11+ add w9,w9,w12+ ror w11,w11,#6+ eor w14,w10,w3+ eor w15,w15,w10,ror#20+ add w9,w9,w11+ ldr w12,[sp,#8]+ and w13,w13,w14+ ror w15,w15,#2+ add w5,w5,w9+ eor w13,w13,w3+ add w8,w8,w12+ add w9,w9,w15+ and w12,w6,w5+ bic w15,w7,w5+ eor w11,w5,w5,ror#5+ add w9,w9,w13+ orr w12,w12,w15+ eor w11,w11,w5,ror#19+ eor w15,w9,w9,ror#11+ add w8,w8,w12+ ror w11,w11,#6+ eor w13,w9,w10+ eor w15,w15,w9,ror#20+ add w8,w8,w11+ ldr w12,[sp,#12]+ and w14,w14,w13+ ror w15,w15,#2+ add w4,w4,w8+ eor w14,w14,w10+ add w7,w7,w12+ add w8,w8,w15+ and w12,w5,w4+ bic w15,w6,w4+ eor w11,w4,w4,ror#5+ add w8,w8,w14+ orr w12,w12,w15+ eor w11,w11,w4,ror#19+ eor w15,w8,w8,ror#11+ add w7,w7,w12+ ror w11,w11,#6+ eor w14,w8,w9+ eor w15,w15,w8,ror#20+ add w7,w7,w11+ ldr w12,[sp,#16]+ and w13,w13,w14+ ror w15,w15,#2+ add w3,w3,w7+ eor w13,w13,w9+ st1 {v4.4s},[x17], #16+ add w6,w6,w12+ add w7,w7,w15+ and w12,w4,w3+ ld1 {v1.16b},[x1],#16+ bic w15,w5,w3+ eor w11,w3,w3,ror#5+ ld1 {v4.4s},[x16],#16+ add w7,w7,w13+ orr w12,w12,w15+ eor w11,w11,w3,ror#19+ eor w15,w7,w7,ror#11+ rev32 v1.16b,v1.16b+ add w6,w6,w12+ ror w11,w11,#6+ eor w13,w7,w8+ eor w15,w15,w7,ror#20+ add v4.4s,v4.4s,v1.4s+ add w6,w6,w11+ ldr w12,[sp,#20]+ and w14,w14,w13+ ror w15,w15,#2+ add w10,w10,w6+ eor w14,w14,w8+ add w5,w5,w12+ add w6,w6,w15+ and w12,w3,w10+ bic w15,w4,w10+ eor w11,w10,w10,ror#5+ add w6,w6,w14+ orr w12,w12,w15+ eor w11,w11,w10,ror#19+ eor w15,w6,w6,ror#11+ add w5,w5,w12+ ror w11,w11,#6+ eor w14,w6,w7+ eor w15,w15,w6,ror#20+ add w5,w5,w11+ ldr w12,[sp,#24]+ and w13,w13,w14+ ror w15,w15,#2+ add w9,w9,w5+ eor w13,w13,w7+ add w4,w4,w12+ add w5,w5,w15+ and w12,w10,w9+ bic w15,w3,w9+ eor w11,w9,w9,ror#5+ add w5,w5,w13+ orr w12,w12,w15+ eor w11,w11,w9,ror#19+ eor w15,w5,w5,ror#11+ add w4,w4,w12+ ror w11,w11,#6+ eor w13,w5,w6+ eor w15,w15,w5,ror#20+ add w4,w4,w11+ ldr w12,[sp,#28]+ and w14,w14,w13+ ror w15,w15,#2+ add w8,w8,w4+ eor w14,w14,w6+ add w3,w3,w12+ add w4,w4,w15+ and w12,w9,w8+ bic w15,w10,w8+ eor w11,w8,w8,ror#5+ add w4,w4,w14+ orr w12,w12,w15+ eor w11,w11,w8,ror#19+ eor w15,w4,w4,ror#11+ add w3,w3,w12+ ror w11,w11,#6+ eor w14,w4,w5+ eor w15,w15,w4,ror#20+ add w3,w3,w11+ ldr w12,[sp,#32]+ and w13,w13,w14+ ror w15,w15,#2+ add w7,w7,w3+ eor w13,w13,w5+ st1 {v4.4s},[x17], #16+ add w10,w10,w12+ add w3,w3,w15+ and w12,w8,w7+ ld1 {v2.16b},[x1],#16+ bic w15,w9,w7+ eor w11,w7,w7,ror#5+ ld1 {v4.4s},[x16],#16+ add w3,w3,w13+ orr w12,w12,w15+ eor w11,w11,w7,ror#19+ eor w15,w3,w3,ror#11+ rev32 v2.16b,v2.16b+ add w10,w10,w12+ ror w11,w11,#6+ eor w13,w3,w4+ eor w15,w15,w3,ror#20+ add v4.4s,v4.4s,v2.4s+ add w10,w10,w11+ ldr w12,[sp,#36]+ and w14,w14,w13+ ror w15,w15,#2+ add w6,w6,w10+ eor w14,w14,w4+ add w9,w9,w12+ add w10,w10,w15+ and w12,w7,w6+ bic w15,w8,w6+ eor w11,w6,w6,ror#5+ add w10,w10,w14+ orr w12,w12,w15+ eor w11,w11,w6,ror#19+ eor w15,w10,w10,ror#11+ add w9,w9,w12+ ror w11,w11,#6+ eor w14,w10,w3+ eor w15,w15,w10,ror#20+ add w9,w9,w11+ ldr w12,[sp,#40]+ and w13,w13,w14+ ror w15,w15,#2+ add w5,w5,w9+ eor w13,w13,w3+ add w8,w8,w12+ add w9,w9,w15+ and w12,w6,w5+ bic w15,w7,w5+ eor w11,w5,w5,ror#5+ add w9,w9,w13+ orr w12,w12,w15+ eor w11,w11,w5,ror#19+ eor w15,w9,w9,ror#11+ add w8,w8,w12+ ror w11,w11,#6+ eor w13,w9,w10+ eor w15,w15,w9,ror#20+ add w8,w8,w11+ ldr w12,[sp,#44]+ and w14,w14,w13+ ror w15,w15,#2+ add w4,w4,w8+ eor w14,w14,w10+ add w7,w7,w12+ add w8,w8,w15+ and w12,w5,w4+ bic w15,w6,w4+ eor w11,w4,w4,ror#5+ add w8,w8,w14+ orr w12,w12,w15+ eor w11,w11,w4,ror#19+ eor w15,w8,w8,ror#11+ add w7,w7,w12+ ror w11,w11,#6+ eor w14,w8,w9+ eor w15,w15,w8,ror#20+ add w7,w7,w11+ ldr w12,[sp,#48]+ and w13,w13,w14+ ror w15,w15,#2+ add w3,w3,w7+ eor w13,w13,w9+ st1 {v4.4s},[x17], #16+ add w6,w6,w12+ add w7,w7,w15+ and w12,w4,w3+ ld1 {v3.16b},[x1],#16+ bic w15,w5,w3+ eor w11,w3,w3,ror#5+ ld1 {v4.4s},[x16],#16+ add w7,w7,w13+ orr w12,w12,w15+ eor w11,w11,w3,ror#19+ eor w15,w7,w7,ror#11+ rev32 v3.16b,v3.16b+ add w6,w6,w12+ ror w11,w11,#6+ eor w13,w7,w8+ eor w15,w15,w7,ror#20+ add v4.4s,v4.4s,v3.4s+ add w6,w6,w11+ ldr w12,[sp,#52]+ and w14,w14,w13+ ror w15,w15,#2+ add w10,w10,w6+ eor w14,w14,w8+ add w5,w5,w12+ add w6,w6,w15+ and w12,w3,w10+ bic w15,w4,w10+ eor w11,w10,w10,ror#5+ add w6,w6,w14+ orr w12,w12,w15+ eor w11,w11,w10,ror#19+ eor w15,w6,w6,ror#11+ add w5,w5,w12+ ror w11,w11,#6+ eor w14,w6,w7+ eor w15,w15,w6,ror#20+ add w5,w5,w11+ ldr w12,[sp,#56]+ and w13,w13,w14+ ror w15,w15,#2+ add w9,w9,w5+ eor w13,w13,w7+ add w4,w4,w12+ add w5,w5,w15+ and w12,w10,w9+ bic w15,w3,w9+ eor w11,w9,w9,ror#5+ add w5,w5,w13+ orr w12,w12,w15+ eor w11,w11,w9,ror#19+ eor w15,w5,w5,ror#11+ add w4,w4,w12+ ror w11,w11,#6+ eor w13,w5,w6+ eor w15,w15,w5,ror#20+ add w4,w4,w11+ ldr w12,[sp,#60]+ and w14,w14,w13+ ror w15,w15,#2+ add w8,w8,w4+ eor w14,w14,w6+ add w3,w3,w12+ add w4,w4,w15+ and w12,w9,w8+ bic w15,w10,w8+ eor w11,w8,w8,ror#5+ add w4,w4,w14+ orr w12,w12,w15+ eor w11,w11,w8,ror#19+ eor w15,w4,w4,ror#11+ add w3,w3,w12+ ror w11,w11,#6+ eor w14,w4,w5+ eor w15,w15,w4,ror#20+ add w3,w3,w11+ and w13,w13,w14+ ror w15,w15,#2+ add w7,w7,w3+ eor w13,w13,w5+ st1 {v4.4s},[x17], #16+ add w3,w3,w15 // h+=Sigma0(a) from the past+ ldp w11,w12,[x0,#0]+ add w3,w3,w13 // h+=Maj(a,b,c) from the past+ ldp w13,w14,[x0,#8]+ add w3,w3,w11 // accumulate+ add w4,w4,w12+ ldp w11,w12,[x0,#16]+ add w5,w5,w13+ add w6,w6,w14+ ldp w13,w14,[x0,#24]+ add w7,w7,w11+ add w8,w8,w12+ ldr w12,[sp,#0]+ stp w3,w4,[x0,#0]+ add w9,w9,w13+ mov w13,wzr+ stp w5,w6,[x0,#8]+ add w10,w10,w14+ stp w7,w8,[x0,#16]+ eor w14,w4,w5+ stp w9,w10,[x0,#24]+ mov w15,wzr+ mov x17,sp+ b.ne .L_00_48++ ldr x29,[x29]+ add sp,sp,#16*4+2*__SIZEOF_POINTER__+ ret+.size crypton_sha256_asm_block_neon,.-crypton_sha256_asm_block_neon+#if !defined(__KERNEL__) && !defined(_WIN64)+.comm crypton_armcap_P,4,4+.hidden crypton_armcap_P+#endif++.section .note.GNU-stack,"",%progbits
@@ -0,0 +1,5463 @@+.text +++.globl crypton_sha256_asm_block_data_order+.type crypton_sha256_asm_block_data_order,@function+.align 16+crypton_sha256_asm_block_data_order:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa+++ pushq %rbp+.cfi_adjust_cfa_offset 8+.cfi_offset %rbp,-16+ movq %rsp,%rbp+.cfi_def_cfa_register %rbp+ leaq crypton_ia32cap_P(%rip),%rax+ movl 0(%rax),%r9d+ movl 4(%rax),%r10d+ movl 8(%rax),%eax+ testl $536870912,%eax+ jnz .Lshaext_shortcut+ andl $296,%eax+ cmpl $296,%eax+ je .Lavx2_shortcut+ andl $1073741824,%r9d+ andl $268435968,%r10d+ orl %r9d,%r10d+ cmpl $1342177792,%r10d+ je .Lavx_shortcut+ testl $512,%r10d+ jnz .Lssse3_shortcut+ pushq %rbx+.cfi_offset %rbx,-24+ pushq %r12+.cfi_offset %r12,-32+ pushq %r13+.cfi_offset %r13,-40+ pushq %r14+.cfi_offset %r14,-48+ pushq %r15+.cfi_offset %r15,-56+ shlq $4,%rdx+ subq $64+24,%rsp++.cfi_def_cfa %rsp,144++ leaq (%rsi,%rdx,4),%rdx+ movq %rdi,64+0(%rsp)+ movq %rsi,64+8(%rsp)+ movq %rdx,64+16(%rsp)++ movl 0(%rdi),%eax+ movl 4(%rdi),%ebx+ movl 8(%rdi),%ecx+ movl 12(%rdi),%edx+ movl 16(%rdi),%r8d+ movl 20(%rdi),%r9d+ movl 24(%rdi),%r10d+ movl 28(%rdi),%r11d+ jmp .Lloop++.align 16+.Lloop:+ movl %ebx,%edi+ leaq K256(%rip),%rbp+ xorl %ecx,%edi+ movl 0(%rsi),%r12d+ movl %r8d,%r13d+ movl %eax,%r14d+ bswapl %r12d+ rorl $14,%r13d+ movl %r9d,%r15d++ xorl %r8d,%r13d+ rorl $9,%r14d+ xorl %r10d,%r15d++ movl %r12d,0(%rsp)+ xorl %eax,%r14d+ andl %r8d,%r15d++ rorl $5,%r13d+ addl %r11d,%r12d+ xorl %r10d,%r15d++ rorl $11,%r14d+ xorl %r8d,%r13d+ addl %r15d,%r12d++ movl %eax,%r15d+ addl (%rbp),%r12d+ xorl %eax,%r14d++ xorl %ebx,%r15d+ rorl $6,%r13d+ movl %ebx,%r11d++ andl %r15d,%edi+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %edi,%r11d+ addl %r12d,%edx+ addl %r12d,%r11d++ leaq 4(%rbp),%rbp+ addl %r14d,%r11d+ movl 4(%rsi),%r12d+ movl %edx,%r13d+ movl %r11d,%r14d+ bswapl %r12d+ rorl $14,%r13d+ movl %r8d,%edi++ xorl %edx,%r13d+ rorl $9,%r14d+ xorl %r9d,%edi++ movl %r12d,4(%rsp)+ xorl %r11d,%r14d+ andl %edx,%edi++ rorl $5,%r13d+ addl %r10d,%r12d+ xorl %r9d,%edi++ rorl $11,%r14d+ xorl %edx,%r13d+ addl %edi,%r12d++ movl %r11d,%edi+ addl (%rbp),%r12d+ xorl %r11d,%r14d++ xorl %eax,%edi+ rorl $6,%r13d+ movl %eax,%r10d++ andl %edi,%r15d+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %r15d,%r10d+ addl %r12d,%ecx+ addl %r12d,%r10d++ leaq 4(%rbp),%rbp+ addl %r14d,%r10d+ movl 8(%rsi),%r12d+ movl %ecx,%r13d+ movl %r10d,%r14d+ bswapl %r12d+ rorl $14,%r13d+ movl %edx,%r15d++ xorl %ecx,%r13d+ rorl $9,%r14d+ xorl %r8d,%r15d++ movl %r12d,8(%rsp)+ xorl %r10d,%r14d+ andl %ecx,%r15d++ rorl $5,%r13d+ addl %r9d,%r12d+ xorl %r8d,%r15d++ rorl $11,%r14d+ xorl %ecx,%r13d+ addl %r15d,%r12d++ movl %r10d,%r15d+ addl (%rbp),%r12d+ xorl %r10d,%r14d++ xorl %r11d,%r15d+ rorl $6,%r13d+ movl %r11d,%r9d++ andl %r15d,%edi+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %edi,%r9d+ addl %r12d,%ebx+ addl %r12d,%r9d++ leaq 4(%rbp),%rbp+ addl %r14d,%r9d+ movl 12(%rsi),%r12d+ movl %ebx,%r13d+ movl %r9d,%r14d+ bswapl %r12d+ rorl $14,%r13d+ movl %ecx,%edi++ xorl %ebx,%r13d+ rorl $9,%r14d+ xorl %edx,%edi++ movl %r12d,12(%rsp)+ xorl %r9d,%r14d+ andl %ebx,%edi++ rorl $5,%r13d+ addl %r8d,%r12d+ xorl %edx,%edi++ rorl $11,%r14d+ xorl %ebx,%r13d+ addl %edi,%r12d++ movl %r9d,%edi+ addl (%rbp),%r12d+ xorl %r9d,%r14d++ xorl %r10d,%edi+ rorl $6,%r13d+ movl %r10d,%r8d++ andl %edi,%r15d+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %r15d,%r8d+ addl %r12d,%eax+ addl %r12d,%r8d++ leaq 20(%rbp),%rbp+ addl %r14d,%r8d+ movl 16(%rsi),%r12d+ movl %eax,%r13d+ movl %r8d,%r14d+ bswapl %r12d+ rorl $14,%r13d+ movl %ebx,%r15d++ xorl %eax,%r13d+ rorl $9,%r14d+ xorl %ecx,%r15d++ movl %r12d,16(%rsp)+ xorl %r8d,%r14d+ andl %eax,%r15d++ rorl $5,%r13d+ addl %edx,%r12d+ xorl %ecx,%r15d++ rorl $11,%r14d+ xorl %eax,%r13d+ addl %r15d,%r12d++ movl %r8d,%r15d+ addl (%rbp),%r12d+ xorl %r8d,%r14d++ xorl %r9d,%r15d+ rorl $6,%r13d+ movl %r9d,%edx++ andl %r15d,%edi+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %edi,%edx+ addl %r12d,%r11d+ addl %r12d,%edx++ leaq 4(%rbp),%rbp+ addl %r14d,%edx+ movl 20(%rsi),%r12d+ movl %r11d,%r13d+ movl %edx,%r14d+ bswapl %r12d+ rorl $14,%r13d+ movl %eax,%edi++ xorl %r11d,%r13d+ rorl $9,%r14d+ xorl %ebx,%edi++ movl %r12d,20(%rsp)+ xorl %edx,%r14d+ andl %r11d,%edi++ rorl $5,%r13d+ addl %ecx,%r12d+ xorl %ebx,%edi++ rorl $11,%r14d+ xorl %r11d,%r13d+ addl %edi,%r12d++ movl %edx,%edi+ addl (%rbp),%r12d+ xorl %edx,%r14d++ xorl %r8d,%edi+ rorl $6,%r13d+ movl %r8d,%ecx++ andl %edi,%r15d+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %r15d,%ecx+ addl %r12d,%r10d+ addl %r12d,%ecx++ leaq 4(%rbp),%rbp+ addl %r14d,%ecx+ movl 24(%rsi),%r12d+ movl %r10d,%r13d+ movl %ecx,%r14d+ bswapl %r12d+ rorl $14,%r13d+ movl %r11d,%r15d++ xorl %r10d,%r13d+ rorl $9,%r14d+ xorl %eax,%r15d++ movl %r12d,24(%rsp)+ xorl %ecx,%r14d+ andl %r10d,%r15d++ rorl $5,%r13d+ addl %ebx,%r12d+ xorl %eax,%r15d++ rorl $11,%r14d+ xorl %r10d,%r13d+ addl %r15d,%r12d++ movl %ecx,%r15d+ addl (%rbp),%r12d+ xorl %ecx,%r14d++ xorl %edx,%r15d+ rorl $6,%r13d+ movl %edx,%ebx++ andl %r15d,%edi+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %edi,%ebx+ addl %r12d,%r9d+ addl %r12d,%ebx++ leaq 4(%rbp),%rbp+ addl %r14d,%ebx+ movl 28(%rsi),%r12d+ movl %r9d,%r13d+ movl %ebx,%r14d+ bswapl %r12d+ rorl $14,%r13d+ movl %r10d,%edi++ xorl %r9d,%r13d+ rorl $9,%r14d+ xorl %r11d,%edi++ movl %r12d,28(%rsp)+ xorl %ebx,%r14d+ andl %r9d,%edi++ rorl $5,%r13d+ addl %eax,%r12d+ xorl %r11d,%edi++ rorl $11,%r14d+ xorl %r9d,%r13d+ addl %edi,%r12d++ movl %ebx,%edi+ addl (%rbp),%r12d+ xorl %ebx,%r14d++ xorl %ecx,%edi+ rorl $6,%r13d+ movl %ecx,%eax++ andl %edi,%r15d+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %r15d,%eax+ addl %r12d,%r8d+ addl %r12d,%eax++ leaq 20(%rbp),%rbp+ addl %r14d,%eax+ movl 32(%rsi),%r12d+ movl %r8d,%r13d+ movl %eax,%r14d+ bswapl %r12d+ rorl $14,%r13d+ movl %r9d,%r15d++ xorl %r8d,%r13d+ rorl $9,%r14d+ xorl %r10d,%r15d++ movl %r12d,32(%rsp)+ xorl %eax,%r14d+ andl %r8d,%r15d++ rorl $5,%r13d+ addl %r11d,%r12d+ xorl %r10d,%r15d++ rorl $11,%r14d+ xorl %r8d,%r13d+ addl %r15d,%r12d++ movl %eax,%r15d+ addl (%rbp),%r12d+ xorl %eax,%r14d++ xorl %ebx,%r15d+ rorl $6,%r13d+ movl %ebx,%r11d++ andl %r15d,%edi+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %edi,%r11d+ addl %r12d,%edx+ addl %r12d,%r11d++ leaq 4(%rbp),%rbp+ addl %r14d,%r11d+ movl 36(%rsi),%r12d+ movl %edx,%r13d+ movl %r11d,%r14d+ bswapl %r12d+ rorl $14,%r13d+ movl %r8d,%edi++ xorl %edx,%r13d+ rorl $9,%r14d+ xorl %r9d,%edi++ movl %r12d,36(%rsp)+ xorl %r11d,%r14d+ andl %edx,%edi++ rorl $5,%r13d+ addl %r10d,%r12d+ xorl %r9d,%edi++ rorl $11,%r14d+ xorl %edx,%r13d+ addl %edi,%r12d++ movl %r11d,%edi+ addl (%rbp),%r12d+ xorl %r11d,%r14d++ xorl %eax,%edi+ rorl $6,%r13d+ movl %eax,%r10d++ andl %edi,%r15d+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %r15d,%r10d+ addl %r12d,%ecx+ addl %r12d,%r10d++ leaq 4(%rbp),%rbp+ addl %r14d,%r10d+ movl 40(%rsi),%r12d+ movl %ecx,%r13d+ movl %r10d,%r14d+ bswapl %r12d+ rorl $14,%r13d+ movl %edx,%r15d++ xorl %ecx,%r13d+ rorl $9,%r14d+ xorl %r8d,%r15d++ movl %r12d,40(%rsp)+ xorl %r10d,%r14d+ andl %ecx,%r15d++ rorl $5,%r13d+ addl %r9d,%r12d+ xorl %r8d,%r15d++ rorl $11,%r14d+ xorl %ecx,%r13d+ addl %r15d,%r12d++ movl %r10d,%r15d+ addl (%rbp),%r12d+ xorl %r10d,%r14d++ xorl %r11d,%r15d+ rorl $6,%r13d+ movl %r11d,%r9d++ andl %r15d,%edi+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %edi,%r9d+ addl %r12d,%ebx+ addl %r12d,%r9d++ leaq 4(%rbp),%rbp+ addl %r14d,%r9d+ movl 44(%rsi),%r12d+ movl %ebx,%r13d+ movl %r9d,%r14d+ bswapl %r12d+ rorl $14,%r13d+ movl %ecx,%edi++ xorl %ebx,%r13d+ rorl $9,%r14d+ xorl %edx,%edi++ movl %r12d,44(%rsp)+ xorl %r9d,%r14d+ andl %ebx,%edi++ rorl $5,%r13d+ addl %r8d,%r12d+ xorl %edx,%edi++ rorl $11,%r14d+ xorl %ebx,%r13d+ addl %edi,%r12d++ movl %r9d,%edi+ addl (%rbp),%r12d+ xorl %r9d,%r14d++ xorl %r10d,%edi+ rorl $6,%r13d+ movl %r10d,%r8d++ andl %edi,%r15d+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %r15d,%r8d+ addl %r12d,%eax+ addl %r12d,%r8d++ leaq 20(%rbp),%rbp+ addl %r14d,%r8d+ movl 48(%rsi),%r12d+ movl %eax,%r13d+ movl %r8d,%r14d+ bswapl %r12d+ rorl $14,%r13d+ movl %ebx,%r15d++ xorl %eax,%r13d+ rorl $9,%r14d+ xorl %ecx,%r15d++ movl %r12d,48(%rsp)+ xorl %r8d,%r14d+ andl %eax,%r15d++ rorl $5,%r13d+ addl %edx,%r12d+ xorl %ecx,%r15d++ rorl $11,%r14d+ xorl %eax,%r13d+ addl %r15d,%r12d++ movl %r8d,%r15d+ addl (%rbp),%r12d+ xorl %r8d,%r14d++ xorl %r9d,%r15d+ rorl $6,%r13d+ movl %r9d,%edx++ andl %r15d,%edi+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %edi,%edx+ addl %r12d,%r11d+ addl %r12d,%edx++ leaq 4(%rbp),%rbp+ addl %r14d,%edx+ movl 52(%rsi),%r12d+ movl %r11d,%r13d+ movl %edx,%r14d+ bswapl %r12d+ rorl $14,%r13d+ movl %eax,%edi++ xorl %r11d,%r13d+ rorl $9,%r14d+ xorl %ebx,%edi++ movl %r12d,52(%rsp)+ xorl %edx,%r14d+ andl %r11d,%edi++ rorl $5,%r13d+ addl %ecx,%r12d+ xorl %ebx,%edi++ rorl $11,%r14d+ xorl %r11d,%r13d+ addl %edi,%r12d++ movl %edx,%edi+ addl (%rbp),%r12d+ xorl %edx,%r14d++ xorl %r8d,%edi+ rorl $6,%r13d+ movl %r8d,%ecx++ andl %edi,%r15d+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %r15d,%ecx+ addl %r12d,%r10d+ addl %r12d,%ecx++ leaq 4(%rbp),%rbp+ addl %r14d,%ecx+ movl 56(%rsi),%r12d+ movl %r10d,%r13d+ movl %ecx,%r14d+ bswapl %r12d+ rorl $14,%r13d+ movl %r11d,%r15d++ xorl %r10d,%r13d+ rorl $9,%r14d+ xorl %eax,%r15d++ movl %r12d,56(%rsp)+ xorl %ecx,%r14d+ andl %r10d,%r15d++ rorl $5,%r13d+ addl %ebx,%r12d+ xorl %eax,%r15d++ rorl $11,%r14d+ xorl %r10d,%r13d+ addl %r15d,%r12d++ movl %ecx,%r15d+ addl (%rbp),%r12d+ xorl %ecx,%r14d++ xorl %edx,%r15d+ rorl $6,%r13d+ movl %edx,%ebx++ andl %r15d,%edi+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %edi,%ebx+ addl %r12d,%r9d+ addl %r12d,%ebx++ leaq 4(%rbp),%rbp+ addl %r14d,%ebx+ movl 60(%rsi),%r12d+ movl %r9d,%r13d+ movl %ebx,%r14d+ bswapl %r12d+ rorl $14,%r13d+ movl %r10d,%edi++ xorl %r9d,%r13d+ rorl $9,%r14d+ xorl %r11d,%edi++ movl %r12d,60(%rsp)+ xorl %ebx,%r14d+ andl %r9d,%edi++ rorl $5,%r13d+ addl %eax,%r12d+ xorl %r11d,%edi++ rorl $11,%r14d+ xorl %r9d,%r13d+ addl %edi,%r12d++ movl %ebx,%edi+ addl (%rbp),%r12d+ xorl %ebx,%r14d++ xorl %ecx,%edi+ rorl $6,%r13d+ movl %ecx,%eax++ andl %edi,%r15d+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %r15d,%eax+ addl %r12d,%r8d+ addl %r12d,%eax++ leaq 20(%rbp),%rbp+ jmp .Lrounds_16_xx+.align 16+.Lrounds_16_xx:+ movl 4(%rsp),%r13d+ movl 56(%rsp),%r15d++ movl %r13d,%r12d+ rorl $11,%r13d+ addl %r14d,%eax+ movl %r15d,%r14d+ rorl $2,%r15d++ xorl %r12d,%r13d+ shrl $3,%r12d+ rorl $7,%r13d+ xorl %r14d,%r15d+ shrl $10,%r14d++ rorl $17,%r15d+ xorl %r13d,%r12d+ xorl %r14d,%r15d+ addl 36(%rsp),%r12d++ addl 0(%rsp),%r12d+ movl %r8d,%r13d+ addl %r15d,%r12d+ movl %eax,%r14d+ rorl $14,%r13d+ movl %r9d,%r15d++ xorl %r8d,%r13d+ rorl $9,%r14d+ xorl %r10d,%r15d++ movl %r12d,0(%rsp)+ xorl %eax,%r14d+ andl %r8d,%r15d++ rorl $5,%r13d+ addl %r11d,%r12d+ xorl %r10d,%r15d++ rorl $11,%r14d+ xorl %r8d,%r13d+ addl %r15d,%r12d++ movl %eax,%r15d+ addl (%rbp),%r12d+ xorl %eax,%r14d++ xorl %ebx,%r15d+ rorl $6,%r13d+ movl %ebx,%r11d++ andl %r15d,%edi+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %edi,%r11d+ addl %r12d,%edx+ addl %r12d,%r11d++ leaq 4(%rbp),%rbp+ movl 8(%rsp),%r13d+ movl 60(%rsp),%edi++ movl %r13d,%r12d+ rorl $11,%r13d+ addl %r14d,%r11d+ movl %edi,%r14d+ rorl $2,%edi++ xorl %r12d,%r13d+ shrl $3,%r12d+ rorl $7,%r13d+ xorl %r14d,%edi+ shrl $10,%r14d++ rorl $17,%edi+ xorl %r13d,%r12d+ xorl %r14d,%edi+ addl 40(%rsp),%r12d++ addl 4(%rsp),%r12d+ movl %edx,%r13d+ addl %edi,%r12d+ movl %r11d,%r14d+ rorl $14,%r13d+ movl %r8d,%edi++ xorl %edx,%r13d+ rorl $9,%r14d+ xorl %r9d,%edi++ movl %r12d,4(%rsp)+ xorl %r11d,%r14d+ andl %edx,%edi++ rorl $5,%r13d+ addl %r10d,%r12d+ xorl %r9d,%edi++ rorl $11,%r14d+ xorl %edx,%r13d+ addl %edi,%r12d++ movl %r11d,%edi+ addl (%rbp),%r12d+ xorl %r11d,%r14d++ xorl %eax,%edi+ rorl $6,%r13d+ movl %eax,%r10d++ andl %edi,%r15d+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %r15d,%r10d+ addl %r12d,%ecx+ addl %r12d,%r10d++ leaq 4(%rbp),%rbp+ movl 12(%rsp),%r13d+ movl 0(%rsp),%r15d++ movl %r13d,%r12d+ rorl $11,%r13d+ addl %r14d,%r10d+ movl %r15d,%r14d+ rorl $2,%r15d++ xorl %r12d,%r13d+ shrl $3,%r12d+ rorl $7,%r13d+ xorl %r14d,%r15d+ shrl $10,%r14d++ rorl $17,%r15d+ xorl %r13d,%r12d+ xorl %r14d,%r15d+ addl 44(%rsp),%r12d++ addl 8(%rsp),%r12d+ movl %ecx,%r13d+ addl %r15d,%r12d+ movl %r10d,%r14d+ rorl $14,%r13d+ movl %edx,%r15d++ xorl %ecx,%r13d+ rorl $9,%r14d+ xorl %r8d,%r15d++ movl %r12d,8(%rsp)+ xorl %r10d,%r14d+ andl %ecx,%r15d++ rorl $5,%r13d+ addl %r9d,%r12d+ xorl %r8d,%r15d++ rorl $11,%r14d+ xorl %ecx,%r13d+ addl %r15d,%r12d++ movl %r10d,%r15d+ addl (%rbp),%r12d+ xorl %r10d,%r14d++ xorl %r11d,%r15d+ rorl $6,%r13d+ movl %r11d,%r9d++ andl %r15d,%edi+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %edi,%r9d+ addl %r12d,%ebx+ addl %r12d,%r9d++ leaq 4(%rbp),%rbp+ movl 16(%rsp),%r13d+ movl 4(%rsp),%edi++ movl %r13d,%r12d+ rorl $11,%r13d+ addl %r14d,%r9d+ movl %edi,%r14d+ rorl $2,%edi++ xorl %r12d,%r13d+ shrl $3,%r12d+ rorl $7,%r13d+ xorl %r14d,%edi+ shrl $10,%r14d++ rorl $17,%edi+ xorl %r13d,%r12d+ xorl %r14d,%edi+ addl 48(%rsp),%r12d++ addl 12(%rsp),%r12d+ movl %ebx,%r13d+ addl %edi,%r12d+ movl %r9d,%r14d+ rorl $14,%r13d+ movl %ecx,%edi++ xorl %ebx,%r13d+ rorl $9,%r14d+ xorl %edx,%edi++ movl %r12d,12(%rsp)+ xorl %r9d,%r14d+ andl %ebx,%edi++ rorl $5,%r13d+ addl %r8d,%r12d+ xorl %edx,%edi++ rorl $11,%r14d+ xorl %ebx,%r13d+ addl %edi,%r12d++ movl %r9d,%edi+ addl (%rbp),%r12d+ xorl %r9d,%r14d++ xorl %r10d,%edi+ rorl $6,%r13d+ movl %r10d,%r8d++ andl %edi,%r15d+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %r15d,%r8d+ addl %r12d,%eax+ addl %r12d,%r8d++ leaq 20(%rbp),%rbp+ movl 20(%rsp),%r13d+ movl 8(%rsp),%r15d++ movl %r13d,%r12d+ rorl $11,%r13d+ addl %r14d,%r8d+ movl %r15d,%r14d+ rorl $2,%r15d++ xorl %r12d,%r13d+ shrl $3,%r12d+ rorl $7,%r13d+ xorl %r14d,%r15d+ shrl $10,%r14d++ rorl $17,%r15d+ xorl %r13d,%r12d+ xorl %r14d,%r15d+ addl 52(%rsp),%r12d++ addl 16(%rsp),%r12d+ movl %eax,%r13d+ addl %r15d,%r12d+ movl %r8d,%r14d+ rorl $14,%r13d+ movl %ebx,%r15d++ xorl %eax,%r13d+ rorl $9,%r14d+ xorl %ecx,%r15d++ movl %r12d,16(%rsp)+ xorl %r8d,%r14d+ andl %eax,%r15d++ rorl $5,%r13d+ addl %edx,%r12d+ xorl %ecx,%r15d++ rorl $11,%r14d+ xorl %eax,%r13d+ addl %r15d,%r12d++ movl %r8d,%r15d+ addl (%rbp),%r12d+ xorl %r8d,%r14d++ xorl %r9d,%r15d+ rorl $6,%r13d+ movl %r9d,%edx++ andl %r15d,%edi+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %edi,%edx+ addl %r12d,%r11d+ addl %r12d,%edx++ leaq 4(%rbp),%rbp+ movl 24(%rsp),%r13d+ movl 12(%rsp),%edi++ movl %r13d,%r12d+ rorl $11,%r13d+ addl %r14d,%edx+ movl %edi,%r14d+ rorl $2,%edi++ xorl %r12d,%r13d+ shrl $3,%r12d+ rorl $7,%r13d+ xorl %r14d,%edi+ shrl $10,%r14d++ rorl $17,%edi+ xorl %r13d,%r12d+ xorl %r14d,%edi+ addl 56(%rsp),%r12d++ addl 20(%rsp),%r12d+ movl %r11d,%r13d+ addl %edi,%r12d+ movl %edx,%r14d+ rorl $14,%r13d+ movl %eax,%edi++ xorl %r11d,%r13d+ rorl $9,%r14d+ xorl %ebx,%edi++ movl %r12d,20(%rsp)+ xorl %edx,%r14d+ andl %r11d,%edi++ rorl $5,%r13d+ addl %ecx,%r12d+ xorl %ebx,%edi++ rorl $11,%r14d+ xorl %r11d,%r13d+ addl %edi,%r12d++ movl %edx,%edi+ addl (%rbp),%r12d+ xorl %edx,%r14d++ xorl %r8d,%edi+ rorl $6,%r13d+ movl %r8d,%ecx++ andl %edi,%r15d+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %r15d,%ecx+ addl %r12d,%r10d+ addl %r12d,%ecx++ leaq 4(%rbp),%rbp+ movl 28(%rsp),%r13d+ movl 16(%rsp),%r15d++ movl %r13d,%r12d+ rorl $11,%r13d+ addl %r14d,%ecx+ movl %r15d,%r14d+ rorl $2,%r15d++ xorl %r12d,%r13d+ shrl $3,%r12d+ rorl $7,%r13d+ xorl %r14d,%r15d+ shrl $10,%r14d++ rorl $17,%r15d+ xorl %r13d,%r12d+ xorl %r14d,%r15d+ addl 60(%rsp),%r12d++ addl 24(%rsp),%r12d+ movl %r10d,%r13d+ addl %r15d,%r12d+ movl %ecx,%r14d+ rorl $14,%r13d+ movl %r11d,%r15d++ xorl %r10d,%r13d+ rorl $9,%r14d+ xorl %eax,%r15d++ movl %r12d,24(%rsp)+ xorl %ecx,%r14d+ andl %r10d,%r15d++ rorl $5,%r13d+ addl %ebx,%r12d+ xorl %eax,%r15d++ rorl $11,%r14d+ xorl %r10d,%r13d+ addl %r15d,%r12d++ movl %ecx,%r15d+ addl (%rbp),%r12d+ xorl %ecx,%r14d++ xorl %edx,%r15d+ rorl $6,%r13d+ movl %edx,%ebx++ andl %r15d,%edi+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %edi,%ebx+ addl %r12d,%r9d+ addl %r12d,%ebx++ leaq 4(%rbp),%rbp+ movl 32(%rsp),%r13d+ movl 20(%rsp),%edi++ movl %r13d,%r12d+ rorl $11,%r13d+ addl %r14d,%ebx+ movl %edi,%r14d+ rorl $2,%edi++ xorl %r12d,%r13d+ shrl $3,%r12d+ rorl $7,%r13d+ xorl %r14d,%edi+ shrl $10,%r14d++ rorl $17,%edi+ xorl %r13d,%r12d+ xorl %r14d,%edi+ addl 0(%rsp),%r12d++ addl 28(%rsp),%r12d+ movl %r9d,%r13d+ addl %edi,%r12d+ movl %ebx,%r14d+ rorl $14,%r13d+ movl %r10d,%edi++ xorl %r9d,%r13d+ rorl $9,%r14d+ xorl %r11d,%edi++ movl %r12d,28(%rsp)+ xorl %ebx,%r14d+ andl %r9d,%edi++ rorl $5,%r13d+ addl %eax,%r12d+ xorl %r11d,%edi++ rorl $11,%r14d+ xorl %r9d,%r13d+ addl %edi,%r12d++ movl %ebx,%edi+ addl (%rbp),%r12d+ xorl %ebx,%r14d++ xorl %ecx,%edi+ rorl $6,%r13d+ movl %ecx,%eax++ andl %edi,%r15d+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %r15d,%eax+ addl %r12d,%r8d+ addl %r12d,%eax++ leaq 20(%rbp),%rbp+ movl 36(%rsp),%r13d+ movl 24(%rsp),%r15d++ movl %r13d,%r12d+ rorl $11,%r13d+ addl %r14d,%eax+ movl %r15d,%r14d+ rorl $2,%r15d++ xorl %r12d,%r13d+ shrl $3,%r12d+ rorl $7,%r13d+ xorl %r14d,%r15d+ shrl $10,%r14d++ rorl $17,%r15d+ xorl %r13d,%r12d+ xorl %r14d,%r15d+ addl 4(%rsp),%r12d++ addl 32(%rsp),%r12d+ movl %r8d,%r13d+ addl %r15d,%r12d+ movl %eax,%r14d+ rorl $14,%r13d+ movl %r9d,%r15d++ xorl %r8d,%r13d+ rorl $9,%r14d+ xorl %r10d,%r15d++ movl %r12d,32(%rsp)+ xorl %eax,%r14d+ andl %r8d,%r15d++ rorl $5,%r13d+ addl %r11d,%r12d+ xorl %r10d,%r15d++ rorl $11,%r14d+ xorl %r8d,%r13d+ addl %r15d,%r12d++ movl %eax,%r15d+ addl (%rbp),%r12d+ xorl %eax,%r14d++ xorl %ebx,%r15d+ rorl $6,%r13d+ movl %ebx,%r11d++ andl %r15d,%edi+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %edi,%r11d+ addl %r12d,%edx+ addl %r12d,%r11d++ leaq 4(%rbp),%rbp+ movl 40(%rsp),%r13d+ movl 28(%rsp),%edi++ movl %r13d,%r12d+ rorl $11,%r13d+ addl %r14d,%r11d+ movl %edi,%r14d+ rorl $2,%edi++ xorl %r12d,%r13d+ shrl $3,%r12d+ rorl $7,%r13d+ xorl %r14d,%edi+ shrl $10,%r14d++ rorl $17,%edi+ xorl %r13d,%r12d+ xorl %r14d,%edi+ addl 8(%rsp),%r12d++ addl 36(%rsp),%r12d+ movl %edx,%r13d+ addl %edi,%r12d+ movl %r11d,%r14d+ rorl $14,%r13d+ movl %r8d,%edi++ xorl %edx,%r13d+ rorl $9,%r14d+ xorl %r9d,%edi++ movl %r12d,36(%rsp)+ xorl %r11d,%r14d+ andl %edx,%edi++ rorl $5,%r13d+ addl %r10d,%r12d+ xorl %r9d,%edi++ rorl $11,%r14d+ xorl %edx,%r13d+ addl %edi,%r12d++ movl %r11d,%edi+ addl (%rbp),%r12d+ xorl %r11d,%r14d++ xorl %eax,%edi+ rorl $6,%r13d+ movl %eax,%r10d++ andl %edi,%r15d+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %r15d,%r10d+ addl %r12d,%ecx+ addl %r12d,%r10d++ leaq 4(%rbp),%rbp+ movl 44(%rsp),%r13d+ movl 32(%rsp),%r15d++ movl %r13d,%r12d+ rorl $11,%r13d+ addl %r14d,%r10d+ movl %r15d,%r14d+ rorl $2,%r15d++ xorl %r12d,%r13d+ shrl $3,%r12d+ rorl $7,%r13d+ xorl %r14d,%r15d+ shrl $10,%r14d++ rorl $17,%r15d+ xorl %r13d,%r12d+ xorl %r14d,%r15d+ addl 12(%rsp),%r12d++ addl 40(%rsp),%r12d+ movl %ecx,%r13d+ addl %r15d,%r12d+ movl %r10d,%r14d+ rorl $14,%r13d+ movl %edx,%r15d++ xorl %ecx,%r13d+ rorl $9,%r14d+ xorl %r8d,%r15d++ movl %r12d,40(%rsp)+ xorl %r10d,%r14d+ andl %ecx,%r15d++ rorl $5,%r13d+ addl %r9d,%r12d+ xorl %r8d,%r15d++ rorl $11,%r14d+ xorl %ecx,%r13d+ addl %r15d,%r12d++ movl %r10d,%r15d+ addl (%rbp),%r12d+ xorl %r10d,%r14d++ xorl %r11d,%r15d+ rorl $6,%r13d+ movl %r11d,%r9d++ andl %r15d,%edi+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %edi,%r9d+ addl %r12d,%ebx+ addl %r12d,%r9d++ leaq 4(%rbp),%rbp+ movl 48(%rsp),%r13d+ movl 36(%rsp),%edi++ movl %r13d,%r12d+ rorl $11,%r13d+ addl %r14d,%r9d+ movl %edi,%r14d+ rorl $2,%edi++ xorl %r12d,%r13d+ shrl $3,%r12d+ rorl $7,%r13d+ xorl %r14d,%edi+ shrl $10,%r14d++ rorl $17,%edi+ xorl %r13d,%r12d+ xorl %r14d,%edi+ addl 16(%rsp),%r12d++ addl 44(%rsp),%r12d+ movl %ebx,%r13d+ addl %edi,%r12d+ movl %r9d,%r14d+ rorl $14,%r13d+ movl %ecx,%edi++ xorl %ebx,%r13d+ rorl $9,%r14d+ xorl %edx,%edi++ movl %r12d,44(%rsp)+ xorl %r9d,%r14d+ andl %ebx,%edi++ rorl $5,%r13d+ addl %r8d,%r12d+ xorl %edx,%edi++ rorl $11,%r14d+ xorl %ebx,%r13d+ addl %edi,%r12d++ movl %r9d,%edi+ addl (%rbp),%r12d+ xorl %r9d,%r14d++ xorl %r10d,%edi+ rorl $6,%r13d+ movl %r10d,%r8d++ andl %edi,%r15d+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %r15d,%r8d+ addl %r12d,%eax+ addl %r12d,%r8d++ leaq 20(%rbp),%rbp+ movl 52(%rsp),%r13d+ movl 40(%rsp),%r15d++ movl %r13d,%r12d+ rorl $11,%r13d+ addl %r14d,%r8d+ movl %r15d,%r14d+ rorl $2,%r15d++ xorl %r12d,%r13d+ shrl $3,%r12d+ rorl $7,%r13d+ xorl %r14d,%r15d+ shrl $10,%r14d++ rorl $17,%r15d+ xorl %r13d,%r12d+ xorl %r14d,%r15d+ addl 20(%rsp),%r12d++ addl 48(%rsp),%r12d+ movl %eax,%r13d+ addl %r15d,%r12d+ movl %r8d,%r14d+ rorl $14,%r13d+ movl %ebx,%r15d++ xorl %eax,%r13d+ rorl $9,%r14d+ xorl %ecx,%r15d++ movl %r12d,48(%rsp)+ xorl %r8d,%r14d+ andl %eax,%r15d++ rorl $5,%r13d+ addl %edx,%r12d+ xorl %ecx,%r15d++ rorl $11,%r14d+ xorl %eax,%r13d+ addl %r15d,%r12d++ movl %r8d,%r15d+ addl (%rbp),%r12d+ xorl %r8d,%r14d++ xorl %r9d,%r15d+ rorl $6,%r13d+ movl %r9d,%edx++ andl %r15d,%edi+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %edi,%edx+ addl %r12d,%r11d+ addl %r12d,%edx++ leaq 4(%rbp),%rbp+ movl 56(%rsp),%r13d+ movl 44(%rsp),%edi++ movl %r13d,%r12d+ rorl $11,%r13d+ addl %r14d,%edx+ movl %edi,%r14d+ rorl $2,%edi++ xorl %r12d,%r13d+ shrl $3,%r12d+ rorl $7,%r13d+ xorl %r14d,%edi+ shrl $10,%r14d++ rorl $17,%edi+ xorl %r13d,%r12d+ xorl %r14d,%edi+ addl 24(%rsp),%r12d++ addl 52(%rsp),%r12d+ movl %r11d,%r13d+ addl %edi,%r12d+ movl %edx,%r14d+ rorl $14,%r13d+ movl %eax,%edi++ xorl %r11d,%r13d+ rorl $9,%r14d+ xorl %ebx,%edi++ movl %r12d,52(%rsp)+ xorl %edx,%r14d+ andl %r11d,%edi++ rorl $5,%r13d+ addl %ecx,%r12d+ xorl %ebx,%edi++ rorl $11,%r14d+ xorl %r11d,%r13d+ addl %edi,%r12d++ movl %edx,%edi+ addl (%rbp),%r12d+ xorl %edx,%r14d++ xorl %r8d,%edi+ rorl $6,%r13d+ movl %r8d,%ecx++ andl %edi,%r15d+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %r15d,%ecx+ addl %r12d,%r10d+ addl %r12d,%ecx++ leaq 4(%rbp),%rbp+ movl 60(%rsp),%r13d+ movl 48(%rsp),%r15d++ movl %r13d,%r12d+ rorl $11,%r13d+ addl %r14d,%ecx+ movl %r15d,%r14d+ rorl $2,%r15d++ xorl %r12d,%r13d+ shrl $3,%r12d+ rorl $7,%r13d+ xorl %r14d,%r15d+ shrl $10,%r14d++ rorl $17,%r15d+ xorl %r13d,%r12d+ xorl %r14d,%r15d+ addl 28(%rsp),%r12d++ addl 56(%rsp),%r12d+ movl %r10d,%r13d+ addl %r15d,%r12d+ movl %ecx,%r14d+ rorl $14,%r13d+ movl %r11d,%r15d++ xorl %r10d,%r13d+ rorl $9,%r14d+ xorl %eax,%r15d++ movl %r12d,56(%rsp)+ xorl %ecx,%r14d+ andl %r10d,%r15d++ rorl $5,%r13d+ addl %ebx,%r12d+ xorl %eax,%r15d++ rorl $11,%r14d+ xorl %r10d,%r13d+ addl %r15d,%r12d++ movl %ecx,%r15d+ addl (%rbp),%r12d+ xorl %ecx,%r14d++ xorl %edx,%r15d+ rorl $6,%r13d+ movl %edx,%ebx++ andl %r15d,%edi+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %edi,%ebx+ addl %r12d,%r9d+ addl %r12d,%ebx++ leaq 4(%rbp),%rbp+ movl 0(%rsp),%r13d+ movl 52(%rsp),%edi++ movl %r13d,%r12d+ rorl $11,%r13d+ addl %r14d,%ebx+ movl %edi,%r14d+ rorl $2,%edi++ xorl %r12d,%r13d+ shrl $3,%r12d+ rorl $7,%r13d+ xorl %r14d,%edi+ shrl $10,%r14d++ rorl $17,%edi+ xorl %r13d,%r12d+ xorl %r14d,%edi+ addl 32(%rsp),%r12d++ addl 60(%rsp),%r12d+ movl %r9d,%r13d+ addl %edi,%r12d+ movl %ebx,%r14d+ rorl $14,%r13d+ movl %r10d,%edi++ xorl %r9d,%r13d+ rorl $9,%r14d+ xorl %r11d,%edi++ movl %r12d,60(%rsp)+ xorl %ebx,%r14d+ andl %r9d,%edi++ rorl $5,%r13d+ addl %eax,%r12d+ xorl %r11d,%edi++ rorl $11,%r14d+ xorl %r9d,%r13d+ addl %edi,%r12d++ movl %ebx,%edi+ addl (%rbp),%r12d+ xorl %ebx,%r14d++ xorl %ecx,%edi+ rorl $6,%r13d+ movl %ecx,%eax++ andl %edi,%r15d+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %r15d,%eax+ addl %r12d,%r8d+ addl %r12d,%eax++ leaq 20(%rbp),%rbp+ cmpb $0,3(%rbp)+ jnz .Lrounds_16_xx++ movq 64+0(%rsp),%rdi+ addl %r14d,%eax+ leaq 64(%rsi),%rsi++ addl 0(%rdi),%eax+ addl 4(%rdi),%ebx+ addl 8(%rdi),%ecx+ addl 12(%rdi),%edx+ addl 16(%rdi),%r8d+ addl 20(%rdi),%r9d+ addl 24(%rdi),%r10d+ addl 28(%rdi),%r11d++ cmpq 64+16(%rsp),%rsi++ movl %eax,0(%rdi)+ movl %ebx,4(%rdi)+ movl %ecx,8(%rdi)+ movl %edx,12(%rdi)+ movl %r8d,16(%rdi)+ movl %r9d,20(%rdi)+ movl %r10d,24(%rdi)+ movl %r11d,28(%rdi)+ jb .Lloop++ leaq 64+24+48(%rsp),%r11+.cfi_def_cfa %r11,8+ movq 64+24(%rsp),%r15+ movq -40(%r11),%r14+ movq -32(%r11),%r13+ movq -24(%r11),%r12+ movq -16(%r11),%rbx+ movq -8(%r11),%rbp+.cfi_restore %r12+.cfi_restore %r13+.cfi_restore %r14+.cfi_restore %r15+.cfi_restore %rbp+.cfi_restore %rbx+ leaq (%r11),%rsp+ .byte 0xf3,0xc3+.cfi_endproc +.size crypton_sha256_asm_block_data_order,.-crypton_sha256_asm_block_data_order+.align 64+.type K256,@object+K256:+.long 0x428a2f98,0x71374491,0xb5c0fbcf,0xe9b5dba5+.long 0x428a2f98,0x71374491,0xb5c0fbcf,0xe9b5dba5+.long 0x3956c25b,0x59f111f1,0x923f82a4,0xab1c5ed5+.long 0x3956c25b,0x59f111f1,0x923f82a4,0xab1c5ed5+.long 0xd807aa98,0x12835b01,0x243185be,0x550c7dc3+.long 0xd807aa98,0x12835b01,0x243185be,0x550c7dc3+.long 0x72be5d74,0x80deb1fe,0x9bdc06a7,0xc19bf174+.long 0x72be5d74,0x80deb1fe,0x9bdc06a7,0xc19bf174+.long 0xe49b69c1,0xefbe4786,0x0fc19dc6,0x240ca1cc+.long 0xe49b69c1,0xefbe4786,0x0fc19dc6,0x240ca1cc+.long 0x2de92c6f,0x4a7484aa,0x5cb0a9dc,0x76f988da+.long 0x2de92c6f,0x4a7484aa,0x5cb0a9dc,0x76f988da+.long 0x983e5152,0xa831c66d,0xb00327c8,0xbf597fc7+.long 0x983e5152,0xa831c66d,0xb00327c8,0xbf597fc7+.long 0xc6e00bf3,0xd5a79147,0x06ca6351,0x14292967+.long 0xc6e00bf3,0xd5a79147,0x06ca6351,0x14292967+.long 0x27b70a85,0x2e1b2138,0x4d2c6dfc,0x53380d13+.long 0x27b70a85,0x2e1b2138,0x4d2c6dfc,0x53380d13+.long 0x650a7354,0x766a0abb,0x81c2c92e,0x92722c85+.long 0x650a7354,0x766a0abb,0x81c2c92e,0x92722c85+.long 0xa2bfe8a1,0xa81a664b,0xc24b8b70,0xc76c51a3+.long 0xa2bfe8a1,0xa81a664b,0xc24b8b70,0xc76c51a3+.long 0xd192e819,0xd6990624,0xf40e3585,0x106aa070+.long 0xd192e819,0xd6990624,0xf40e3585,0x106aa070+.long 0x19a4c116,0x1e376c08,0x2748774c,0x34b0bcb5+.long 0x19a4c116,0x1e376c08,0x2748774c,0x34b0bcb5+.long 0x391c0cb3,0x4ed8aa4a,0x5b9cca4f,0x682e6ff3+.long 0x391c0cb3,0x4ed8aa4a,0x5b9cca4f,0x682e6ff3+.long 0x748f82ee,0x78a5636f,0x84c87814,0x8cc70208+.long 0x748f82ee,0x78a5636f,0x84c87814,0x8cc70208+.long 0x90befffa,0xa4506ceb,0xbef9a3f7,0xc67178f2+.long 0x90befffa,0xa4506ceb,0xbef9a3f7,0xc67178f2++.long 0x00010203,0x04050607,0x08090a0b,0x0c0d0e0f+.long 0x00010203,0x04050607,0x08090a0b,0x0c0d0e0f+.long 0x03020100,0x0b0a0908,0xffffffff,0xffffffff+.long 0x03020100,0x0b0a0908,0xffffffff,0xffffffff+.long 0xffffffff,0xffffffff,0x03020100,0x0b0a0908+.long 0xffffffff,0xffffffff,0x03020100,0x0b0a0908+.byte 83,72,65,50,53,54,32,98,108,111,99,107,32,116,114,97,110,115,102,111,114,109,32,102,111,114,32,120,56,54,95,54,52,44,32,67,82,89,80,84,79,71,65,77,83,32,98,121,32,64,100,111,116,45,97,115,109,0+.type crypton_sha256_asm_block_data_order_shaext,@function+.align 64+crypton_sha256_asm_block_data_order_shaext:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa+++ pushq %rbp+.cfi_adjust_cfa_offset 8+.cfi_offset %rbp,-16+ movq %rsp,%rbp+.cfi_def_cfa_register %rbp+.Lshaext_shortcut:++ leaq K256+128(%rip),%rcx+ movdqu (%rdi),%xmm1+ movdqu 16(%rdi),%xmm2+ movdqa 512-128(%rcx),%xmm7++ pshufd $0x1b,%xmm1,%xmm0+ pshufd $0xb1,%xmm1,%xmm1+ pshufd $0x1b,%xmm2,%xmm2+ movdqa %xmm7,%xmm8+.byte 102,15,58,15,202,8+ punpcklqdq %xmm0,%xmm2+ jmp .Loop_shaext++.align 16+.Loop_shaext:+ movdqu (%rsi),%xmm3+ movdqu 16(%rsi),%xmm4+ movdqu 32(%rsi),%xmm5+.byte 102,15,56,0,223+ movdqu 48(%rsi),%xmm6++ movdqa 0-128(%rcx),%xmm0+ paddd %xmm3,%xmm0+.byte 102,15,56,0,231+ movdqa %xmm2,%xmm10+.byte 15,56,203,209+ pshufd $0x0e,%xmm0,%xmm0+ nop+ movdqa %xmm1,%xmm9+.byte 15,56,203,202++ movdqa 32-128(%rcx),%xmm0+ paddd %xmm4,%xmm0+.byte 102,15,56,0,239+.byte 15,56,203,209+ pshufd $0x0e,%xmm0,%xmm0+ leaq 64(%rsi),%rsi+.byte 15,56,204,220+.byte 15,56,203,202++ movdqa 64-128(%rcx),%xmm0+ paddd %xmm5,%xmm0+.byte 102,15,56,0,247+.byte 15,56,203,209+ pshufd $0x0e,%xmm0,%xmm0+ movdqa %xmm6,%xmm7+.byte 102,15,58,15,253,4+ nop+ paddd %xmm7,%xmm3+.byte 15,56,204,229+.byte 15,56,203,202++ movdqa 96-128(%rcx),%xmm0+ paddd %xmm6,%xmm0+.byte 15,56,205,222+.byte 15,56,203,209+ pshufd $0x0e,%xmm0,%xmm0+ movdqa %xmm3,%xmm7+.byte 102,15,58,15,254,4+ nop+ paddd %xmm7,%xmm4+.byte 15,56,204,238+.byte 15,56,203,202+ movdqa 128-128(%rcx),%xmm0+ paddd %xmm3,%xmm0+.byte 15,56,205,227+.byte 15,56,203,209+ pshufd $0x0e,%xmm0,%xmm0+ movdqa %xmm4,%xmm7+.byte 102,15,58,15,251,4+ nop+ paddd %xmm7,%xmm5+.byte 15,56,204,243+.byte 15,56,203,202+ movdqa 160-128(%rcx),%xmm0+ paddd %xmm4,%xmm0+.byte 15,56,205,236+.byte 15,56,203,209+ pshufd $0x0e,%xmm0,%xmm0+ movdqa %xmm5,%xmm7+.byte 102,15,58,15,252,4+ nop+ paddd %xmm7,%xmm6+.byte 15,56,204,220+.byte 15,56,203,202+ movdqa 192-128(%rcx),%xmm0+ paddd %xmm5,%xmm0+.byte 15,56,205,245+.byte 15,56,203,209+ pshufd $0x0e,%xmm0,%xmm0+ movdqa %xmm6,%xmm7+.byte 102,15,58,15,253,4+ nop+ paddd %xmm7,%xmm3+.byte 15,56,204,229+.byte 15,56,203,202+ movdqa 224-128(%rcx),%xmm0+ paddd %xmm6,%xmm0+.byte 15,56,205,222+.byte 15,56,203,209+ pshufd $0x0e,%xmm0,%xmm0+ movdqa %xmm3,%xmm7+.byte 102,15,58,15,254,4+ nop+ paddd %xmm7,%xmm4+.byte 15,56,204,238+.byte 15,56,203,202+ movdqa 256-128(%rcx),%xmm0+ paddd %xmm3,%xmm0+.byte 15,56,205,227+.byte 15,56,203,209+ pshufd $0x0e,%xmm0,%xmm0+ movdqa %xmm4,%xmm7+.byte 102,15,58,15,251,4+ nop+ paddd %xmm7,%xmm5+.byte 15,56,204,243+.byte 15,56,203,202+ movdqa 288-128(%rcx),%xmm0+ paddd %xmm4,%xmm0+.byte 15,56,205,236+.byte 15,56,203,209+ pshufd $0x0e,%xmm0,%xmm0+ movdqa %xmm5,%xmm7+.byte 102,15,58,15,252,4+ nop+ paddd %xmm7,%xmm6+.byte 15,56,204,220+.byte 15,56,203,202+ movdqa 320-128(%rcx),%xmm0+ paddd %xmm5,%xmm0+.byte 15,56,205,245+.byte 15,56,203,209+ pshufd $0x0e,%xmm0,%xmm0+ movdqa %xmm6,%xmm7+.byte 102,15,58,15,253,4+ nop+ paddd %xmm7,%xmm3+.byte 15,56,204,229+.byte 15,56,203,202+ movdqa 352-128(%rcx),%xmm0+ paddd %xmm6,%xmm0+.byte 15,56,205,222+.byte 15,56,203,209+ pshufd $0x0e,%xmm0,%xmm0+ movdqa %xmm3,%xmm7+.byte 102,15,58,15,254,4+ nop+ paddd %xmm7,%xmm4+.byte 15,56,204,238+.byte 15,56,203,202+ movdqa 384-128(%rcx),%xmm0+ paddd %xmm3,%xmm0+.byte 15,56,205,227+.byte 15,56,203,209+ pshufd $0x0e,%xmm0,%xmm0+ movdqa %xmm4,%xmm7+.byte 102,15,58,15,251,4+ nop+ paddd %xmm7,%xmm5+.byte 15,56,204,243+.byte 15,56,203,202+ movdqa 416-128(%rcx),%xmm0+ paddd %xmm4,%xmm0+.byte 15,56,205,236+.byte 15,56,203,209+ pshufd $0x0e,%xmm0,%xmm0+ movdqa %xmm5,%xmm7+.byte 102,15,58,15,252,4+.byte 15,56,203,202+ paddd %xmm7,%xmm6++ movdqa 448-128(%rcx),%xmm0+ paddd %xmm5,%xmm0+.byte 15,56,203,209+ pshufd $0x0e,%xmm0,%xmm0+.byte 15,56,205,245+ movdqa %xmm8,%xmm7+.byte 15,56,203,202++ movdqa 480-128(%rcx),%xmm0+ paddd %xmm6,%xmm0+ nop+.byte 15,56,203,209+ pshufd $0x0e,%xmm0,%xmm0+ decq %rdx+ nop+.byte 15,56,203,202++ paddd %xmm10,%xmm2+ paddd %xmm9,%xmm1+ jnz .Loop_shaext++ pshufd $0xb1,%xmm2,%xmm2+ pshufd $0x1b,%xmm1,%xmm7+ pshufd $0xb1,%xmm1,%xmm1+ punpckhqdq %xmm2,%xmm1+.byte 102,15,58,15,215,8++ movdqu %xmm1,(%rdi)+ movdqu %xmm2,16(%rdi)+.cfi_def_cfa_register %rsp+ popq %rbp+.cfi_adjust_cfa_offset -8+.cfi_restore %rbp++ .byte 0xf3,0xc3+.cfi_endproc +.size crypton_sha256_asm_block_data_order_shaext,.-crypton_sha256_asm_block_data_order_shaext+.type crypton_sha256_asm_block_data_order_ssse3,@function+.align 64+crypton_sha256_asm_block_data_order_ssse3:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa+++ pushq %rbp+.cfi_adjust_cfa_offset 8+.cfi_offset %rbp,-16+ movq %rsp,%rbp+.cfi_def_cfa_register %rbp+.Lssse3_shortcut:+ pushq %rbx+.cfi_offset %rbx,-24+ pushq %r12+.cfi_offset %r12,-32+ pushq %r13+.cfi_offset %r13,-40+ pushq %r14+.cfi_offset %r14,-48+ pushq %r15+.cfi_offset %r15,-56+ shlq $4,%rdx+ subq $24,%rsp++ leaq (%rsi,%rdx,4),%rdx+ movq %rdi,-64(%rbp)++ movq %rdx,-48(%rbp)+++ leaq -64(%rsp),%rsp+ movl 0(%rdi),%eax+ andq $-64,%rsp+ movl 4(%rdi),%ebx+ movl 8(%rdi),%ecx+ movl 12(%rdi),%edx+ movl 16(%rdi),%r8d+ movl 20(%rdi),%r9d+ movl 24(%rdi),%r10d+ movl 28(%rdi),%r11d+++ jmp .Lloop_ssse3+.align 16+.Lloop_ssse3:+ movdqa K256+512(%rip),%xmm7+ movq %rsi,-56(%rbp)+ movdqu 0(%rsi),%xmm0+ movdqu 16(%rsi),%xmm1+ movdqu 32(%rsi),%xmm2+.byte 102,15,56,0,199+ movdqu 48(%rsi),%xmm3+ leaq K256(%rip),%rsi+.byte 102,15,56,0,207+ movdqa 0(%rsi),%xmm4+ movdqa 32(%rsi),%xmm5+.byte 102,15,56,0,215+ paddd %xmm0,%xmm4+ movdqa 64(%rsi),%xmm6+.byte 102,15,56,0,223+ movdqa 96(%rsi),%xmm7+ paddd %xmm1,%xmm5+ paddd %xmm2,%xmm6+ paddd %xmm3,%xmm7+ movdqa %xmm4,0(%rsp)+ movl %eax,%r14d+ movdqa %xmm5,16(%rsp)+ movl %ebx,%edi+ movdqa %xmm6,32(%rsp)+ xorl %ecx,%edi+ movdqa %xmm7,48(%rsp)+ movl %r8d,%r13d+ jmp .Lssse3_00_47++.align 16+.Lssse3_00_47:+ subq $-128,%rsi+ rorl $14,%r13d+ movdqa %xmm1,%xmm4+ movl %r14d,%eax+ movl %r9d,%r12d+ movdqa %xmm3,%xmm7+ rorl $9,%r14d+ xorl %r8d,%r13d+ xorl %r10d,%r12d+ rorl $5,%r13d+ xorl %eax,%r14d+.byte 102,15,58,15,224,4+ andl %r8d,%r12d+ xorl %r8d,%r13d+.byte 102,15,58,15,250,4+ addl 0(%rsp),%r11d+ movl %eax,%r15d+ xorl %r10d,%r12d+ rorl $11,%r14d+ movdqa %xmm4,%xmm5+ xorl %ebx,%r15d+ addl %r12d,%r11d+ movdqa %xmm4,%xmm6+ rorl $6,%r13d+ andl %r15d,%edi+ psrld $3,%xmm4+ xorl %eax,%r14d+ addl %r13d,%r11d+ xorl %ebx,%edi+ paddd %xmm7,%xmm0+ rorl $2,%r14d+ addl %r11d,%edx+ psrld $7,%xmm6+ addl %edi,%r11d+ movl %edx,%r13d+ pshufd $250,%xmm3,%xmm7+ addl %r11d,%r14d+ rorl $14,%r13d+ pslld $14,%xmm5+ movl %r14d,%r11d+ movl %r8d,%r12d+ pxor %xmm6,%xmm4+ rorl $9,%r14d+ xorl %edx,%r13d+ xorl %r9d,%r12d+ rorl $5,%r13d+ psrld $11,%xmm6+ xorl %r11d,%r14d+ pxor %xmm5,%xmm4+ andl %edx,%r12d+ xorl %edx,%r13d+ pslld $11,%xmm5+ addl 4(%rsp),%r10d+ movl %r11d,%edi+ pxor %xmm6,%xmm4+ xorl %r9d,%r12d+ rorl $11,%r14d+ movdqa %xmm7,%xmm6+ xorl %eax,%edi+ addl %r12d,%r10d+ pxor %xmm5,%xmm4+ rorl $6,%r13d+ andl %edi,%r15d+ xorl %r11d,%r14d+ psrld $10,%xmm7+ addl %r13d,%r10d+ xorl %eax,%r15d+ paddd %xmm4,%xmm0+ rorl $2,%r14d+ addl %r10d,%ecx+ psrlq $17,%xmm6+ addl %r15d,%r10d+ movl %ecx,%r13d+ addl %r10d,%r14d+ pxor %xmm6,%xmm7+ rorl $14,%r13d+ movl %r14d,%r10d+ movl %edx,%r12d+ rorl $9,%r14d+ psrlq $2,%xmm6+ xorl %ecx,%r13d+ xorl %r8d,%r12d+ pxor %xmm6,%xmm7+ rorl $5,%r13d+ xorl %r10d,%r14d+ andl %ecx,%r12d+ pshufd $128,%xmm7,%xmm7+ xorl %ecx,%r13d+ addl 8(%rsp),%r9d+ movl %r10d,%r15d+ psrldq $8,%xmm7+ xorl %r8d,%r12d+ rorl $11,%r14d+ xorl %r11d,%r15d+ addl %r12d,%r9d+ rorl $6,%r13d+ paddd %xmm7,%xmm0+ andl %r15d,%edi+ xorl %r10d,%r14d+ addl %r13d,%r9d+ pshufd $80,%xmm0,%xmm7+ xorl %r11d,%edi+ rorl $2,%r14d+ addl %r9d,%ebx+ movdqa %xmm7,%xmm6+ addl %edi,%r9d+ movl %ebx,%r13d+ psrld $10,%xmm7+ addl %r9d,%r14d+ rorl $14,%r13d+ psrlq $17,%xmm6+ movl %r14d,%r9d+ movl %ecx,%r12d+ pxor %xmm6,%xmm7+ rorl $9,%r14d+ xorl %ebx,%r13d+ xorl %edx,%r12d+ rorl $5,%r13d+ xorl %r9d,%r14d+ psrlq $2,%xmm6+ andl %ebx,%r12d+ xorl %ebx,%r13d+ addl 12(%rsp),%r8d+ pxor %xmm6,%xmm7+ movl %r9d,%edi+ xorl %edx,%r12d+ rorl $11,%r14d+ pshufd $8,%xmm7,%xmm7+ xorl %r10d,%edi+ addl %r12d,%r8d+ movdqa 0(%rsi),%xmm6+ rorl $6,%r13d+ andl %edi,%r15d+ pslldq $8,%xmm7+ xorl %r9d,%r14d+ addl %r13d,%r8d+ xorl %r10d,%r15d+ paddd %xmm7,%xmm0+ rorl $2,%r14d+ addl %r8d,%eax+ addl %r15d,%r8d+ paddd %xmm0,%xmm6+ movl %eax,%r13d+ addl %r8d,%r14d+ movdqa %xmm6,0(%rsp)+ rorl $14,%r13d+ movdqa %xmm2,%xmm4+ movl %r14d,%r8d+ movl %ebx,%r12d+ movdqa %xmm0,%xmm7+ rorl $9,%r14d+ xorl %eax,%r13d+ xorl %ecx,%r12d+ rorl $5,%r13d+ xorl %r8d,%r14d+.byte 102,15,58,15,225,4+ andl %eax,%r12d+ xorl %eax,%r13d+.byte 102,15,58,15,251,4+ addl 16(%rsp),%edx+ movl %r8d,%r15d+ xorl %ecx,%r12d+ rorl $11,%r14d+ movdqa %xmm4,%xmm5+ xorl %r9d,%r15d+ addl %r12d,%edx+ movdqa %xmm4,%xmm6+ rorl $6,%r13d+ andl %r15d,%edi+ psrld $3,%xmm4+ xorl %r8d,%r14d+ addl %r13d,%edx+ xorl %r9d,%edi+ paddd %xmm7,%xmm1+ rorl $2,%r14d+ addl %edx,%r11d+ psrld $7,%xmm6+ addl %edi,%edx+ movl %r11d,%r13d+ pshufd $250,%xmm0,%xmm7+ addl %edx,%r14d+ rorl $14,%r13d+ pslld $14,%xmm5+ movl %r14d,%edx+ movl %eax,%r12d+ pxor %xmm6,%xmm4+ rorl $9,%r14d+ xorl %r11d,%r13d+ xorl %ebx,%r12d+ rorl $5,%r13d+ psrld $11,%xmm6+ xorl %edx,%r14d+ pxor %xmm5,%xmm4+ andl %r11d,%r12d+ xorl %r11d,%r13d+ pslld $11,%xmm5+ addl 20(%rsp),%ecx+ movl %edx,%edi+ pxor %xmm6,%xmm4+ xorl %ebx,%r12d+ rorl $11,%r14d+ movdqa %xmm7,%xmm6+ xorl %r8d,%edi+ addl %r12d,%ecx+ pxor %xmm5,%xmm4+ rorl $6,%r13d+ andl %edi,%r15d+ xorl %edx,%r14d+ psrld $10,%xmm7+ addl %r13d,%ecx+ xorl %r8d,%r15d+ paddd %xmm4,%xmm1+ rorl $2,%r14d+ addl %ecx,%r10d+ psrlq $17,%xmm6+ addl %r15d,%ecx+ movl %r10d,%r13d+ addl %ecx,%r14d+ pxor %xmm6,%xmm7+ rorl $14,%r13d+ movl %r14d,%ecx+ movl %r11d,%r12d+ rorl $9,%r14d+ psrlq $2,%xmm6+ xorl %r10d,%r13d+ xorl %eax,%r12d+ pxor %xmm6,%xmm7+ rorl $5,%r13d+ xorl %ecx,%r14d+ andl %r10d,%r12d+ pshufd $128,%xmm7,%xmm7+ xorl %r10d,%r13d+ addl 24(%rsp),%ebx+ movl %ecx,%r15d+ psrldq $8,%xmm7+ xorl %eax,%r12d+ rorl $11,%r14d+ xorl %edx,%r15d+ addl %r12d,%ebx+ rorl $6,%r13d+ paddd %xmm7,%xmm1+ andl %r15d,%edi+ xorl %ecx,%r14d+ addl %r13d,%ebx+ pshufd $80,%xmm1,%xmm7+ xorl %edx,%edi+ rorl $2,%r14d+ addl %ebx,%r9d+ movdqa %xmm7,%xmm6+ addl %edi,%ebx+ movl %r9d,%r13d+ psrld $10,%xmm7+ addl %ebx,%r14d+ rorl $14,%r13d+ psrlq $17,%xmm6+ movl %r14d,%ebx+ movl %r10d,%r12d+ pxor %xmm6,%xmm7+ rorl $9,%r14d+ xorl %r9d,%r13d+ xorl %r11d,%r12d+ rorl $5,%r13d+ xorl %ebx,%r14d+ psrlq $2,%xmm6+ andl %r9d,%r12d+ xorl %r9d,%r13d+ addl 28(%rsp),%eax+ pxor %xmm6,%xmm7+ movl %ebx,%edi+ xorl %r11d,%r12d+ rorl $11,%r14d+ pshufd $8,%xmm7,%xmm7+ xorl %ecx,%edi+ addl %r12d,%eax+ movdqa 32(%rsi),%xmm6+ rorl $6,%r13d+ andl %edi,%r15d+ pslldq $8,%xmm7+ xorl %ebx,%r14d+ addl %r13d,%eax+ xorl %ecx,%r15d+ paddd %xmm7,%xmm1+ rorl $2,%r14d+ addl %eax,%r8d+ addl %r15d,%eax+ paddd %xmm1,%xmm6+ movl %r8d,%r13d+ addl %eax,%r14d+ movdqa %xmm6,16(%rsp)+ rorl $14,%r13d+ movdqa %xmm3,%xmm4+ movl %r14d,%eax+ movl %r9d,%r12d+ movdqa %xmm1,%xmm7+ rorl $9,%r14d+ xorl %r8d,%r13d+ xorl %r10d,%r12d+ rorl $5,%r13d+ xorl %eax,%r14d+.byte 102,15,58,15,226,4+ andl %r8d,%r12d+ xorl %r8d,%r13d+.byte 102,15,58,15,248,4+ addl 32(%rsp),%r11d+ movl %eax,%r15d+ xorl %r10d,%r12d+ rorl $11,%r14d+ movdqa %xmm4,%xmm5+ xorl %ebx,%r15d+ addl %r12d,%r11d+ movdqa %xmm4,%xmm6+ rorl $6,%r13d+ andl %r15d,%edi+ psrld $3,%xmm4+ xorl %eax,%r14d+ addl %r13d,%r11d+ xorl %ebx,%edi+ paddd %xmm7,%xmm2+ rorl $2,%r14d+ addl %r11d,%edx+ psrld $7,%xmm6+ addl %edi,%r11d+ movl %edx,%r13d+ pshufd $250,%xmm1,%xmm7+ addl %r11d,%r14d+ rorl $14,%r13d+ pslld $14,%xmm5+ movl %r14d,%r11d+ movl %r8d,%r12d+ pxor %xmm6,%xmm4+ rorl $9,%r14d+ xorl %edx,%r13d+ xorl %r9d,%r12d+ rorl $5,%r13d+ psrld $11,%xmm6+ xorl %r11d,%r14d+ pxor %xmm5,%xmm4+ andl %edx,%r12d+ xorl %edx,%r13d+ pslld $11,%xmm5+ addl 36(%rsp),%r10d+ movl %r11d,%edi+ pxor %xmm6,%xmm4+ xorl %r9d,%r12d+ rorl $11,%r14d+ movdqa %xmm7,%xmm6+ xorl %eax,%edi+ addl %r12d,%r10d+ pxor %xmm5,%xmm4+ rorl $6,%r13d+ andl %edi,%r15d+ xorl %r11d,%r14d+ psrld $10,%xmm7+ addl %r13d,%r10d+ xorl %eax,%r15d+ paddd %xmm4,%xmm2+ rorl $2,%r14d+ addl %r10d,%ecx+ psrlq $17,%xmm6+ addl %r15d,%r10d+ movl %ecx,%r13d+ addl %r10d,%r14d+ pxor %xmm6,%xmm7+ rorl $14,%r13d+ movl %r14d,%r10d+ movl %edx,%r12d+ rorl $9,%r14d+ psrlq $2,%xmm6+ xorl %ecx,%r13d+ xorl %r8d,%r12d+ pxor %xmm6,%xmm7+ rorl $5,%r13d+ xorl %r10d,%r14d+ andl %ecx,%r12d+ pshufd $128,%xmm7,%xmm7+ xorl %ecx,%r13d+ addl 40(%rsp),%r9d+ movl %r10d,%r15d+ psrldq $8,%xmm7+ xorl %r8d,%r12d+ rorl $11,%r14d+ xorl %r11d,%r15d+ addl %r12d,%r9d+ rorl $6,%r13d+ paddd %xmm7,%xmm2+ andl %r15d,%edi+ xorl %r10d,%r14d+ addl %r13d,%r9d+ pshufd $80,%xmm2,%xmm7+ xorl %r11d,%edi+ rorl $2,%r14d+ addl %r9d,%ebx+ movdqa %xmm7,%xmm6+ addl %edi,%r9d+ movl %ebx,%r13d+ psrld $10,%xmm7+ addl %r9d,%r14d+ rorl $14,%r13d+ psrlq $17,%xmm6+ movl %r14d,%r9d+ movl %ecx,%r12d+ pxor %xmm6,%xmm7+ rorl $9,%r14d+ xorl %ebx,%r13d+ xorl %edx,%r12d+ rorl $5,%r13d+ xorl %r9d,%r14d+ psrlq $2,%xmm6+ andl %ebx,%r12d+ xorl %ebx,%r13d+ addl 44(%rsp),%r8d+ pxor %xmm6,%xmm7+ movl %r9d,%edi+ xorl %edx,%r12d+ rorl $11,%r14d+ pshufd $8,%xmm7,%xmm7+ xorl %r10d,%edi+ addl %r12d,%r8d+ movdqa 64(%rsi),%xmm6+ rorl $6,%r13d+ andl %edi,%r15d+ pslldq $8,%xmm7+ xorl %r9d,%r14d+ addl %r13d,%r8d+ xorl %r10d,%r15d+ paddd %xmm7,%xmm2+ rorl $2,%r14d+ addl %r8d,%eax+ addl %r15d,%r8d+ paddd %xmm2,%xmm6+ movl %eax,%r13d+ addl %r8d,%r14d+ movdqa %xmm6,32(%rsp)+ rorl $14,%r13d+ movdqa %xmm0,%xmm4+ movl %r14d,%r8d+ movl %ebx,%r12d+ movdqa %xmm2,%xmm7+ rorl $9,%r14d+ xorl %eax,%r13d+ xorl %ecx,%r12d+ rorl $5,%r13d+ xorl %r8d,%r14d+.byte 102,15,58,15,227,4+ andl %eax,%r12d+ xorl %eax,%r13d+.byte 102,15,58,15,249,4+ addl 48(%rsp),%edx+ movl %r8d,%r15d+ xorl %ecx,%r12d+ rorl $11,%r14d+ movdqa %xmm4,%xmm5+ xorl %r9d,%r15d+ addl %r12d,%edx+ movdqa %xmm4,%xmm6+ rorl $6,%r13d+ andl %r15d,%edi+ psrld $3,%xmm4+ xorl %r8d,%r14d+ addl %r13d,%edx+ xorl %r9d,%edi+ paddd %xmm7,%xmm3+ rorl $2,%r14d+ addl %edx,%r11d+ psrld $7,%xmm6+ addl %edi,%edx+ movl %r11d,%r13d+ pshufd $250,%xmm2,%xmm7+ addl %edx,%r14d+ rorl $14,%r13d+ pslld $14,%xmm5+ movl %r14d,%edx+ movl %eax,%r12d+ pxor %xmm6,%xmm4+ rorl $9,%r14d+ xorl %r11d,%r13d+ xorl %ebx,%r12d+ rorl $5,%r13d+ psrld $11,%xmm6+ xorl %edx,%r14d+ pxor %xmm5,%xmm4+ andl %r11d,%r12d+ xorl %r11d,%r13d+ pslld $11,%xmm5+ addl 52(%rsp),%ecx+ movl %edx,%edi+ pxor %xmm6,%xmm4+ xorl %ebx,%r12d+ rorl $11,%r14d+ movdqa %xmm7,%xmm6+ xorl %r8d,%edi+ addl %r12d,%ecx+ pxor %xmm5,%xmm4+ rorl $6,%r13d+ andl %edi,%r15d+ xorl %edx,%r14d+ psrld $10,%xmm7+ addl %r13d,%ecx+ xorl %r8d,%r15d+ paddd %xmm4,%xmm3+ rorl $2,%r14d+ addl %ecx,%r10d+ psrlq $17,%xmm6+ addl %r15d,%ecx+ movl %r10d,%r13d+ addl %ecx,%r14d+ pxor %xmm6,%xmm7+ rorl $14,%r13d+ movl %r14d,%ecx+ movl %r11d,%r12d+ rorl $9,%r14d+ psrlq $2,%xmm6+ xorl %r10d,%r13d+ xorl %eax,%r12d+ pxor %xmm6,%xmm7+ rorl $5,%r13d+ xorl %ecx,%r14d+ andl %r10d,%r12d+ pshufd $128,%xmm7,%xmm7+ xorl %r10d,%r13d+ addl 56(%rsp),%ebx+ movl %ecx,%r15d+ psrldq $8,%xmm7+ xorl %eax,%r12d+ rorl $11,%r14d+ xorl %edx,%r15d+ addl %r12d,%ebx+ rorl $6,%r13d+ paddd %xmm7,%xmm3+ andl %r15d,%edi+ xorl %ecx,%r14d+ addl %r13d,%ebx+ pshufd $80,%xmm3,%xmm7+ xorl %edx,%edi+ rorl $2,%r14d+ addl %ebx,%r9d+ movdqa %xmm7,%xmm6+ addl %edi,%ebx+ movl %r9d,%r13d+ psrld $10,%xmm7+ addl %ebx,%r14d+ rorl $14,%r13d+ psrlq $17,%xmm6+ movl %r14d,%ebx+ movl %r10d,%r12d+ pxor %xmm6,%xmm7+ rorl $9,%r14d+ xorl %r9d,%r13d+ xorl %r11d,%r12d+ rorl $5,%r13d+ xorl %ebx,%r14d+ psrlq $2,%xmm6+ andl %r9d,%r12d+ xorl %r9d,%r13d+ addl 60(%rsp),%eax+ pxor %xmm6,%xmm7+ movl %ebx,%edi+ xorl %r11d,%r12d+ rorl $11,%r14d+ pshufd $8,%xmm7,%xmm7+ xorl %ecx,%edi+ addl %r12d,%eax+ movdqa 96(%rsi),%xmm6+ rorl $6,%r13d+ andl %edi,%r15d+ pslldq $8,%xmm7+ xorl %ebx,%r14d+ addl %r13d,%eax+ xorl %ecx,%r15d+ paddd %xmm7,%xmm3+ rorl $2,%r14d+ addl %eax,%r8d+ addl %r15d,%eax+ paddd %xmm3,%xmm6+ movl %r8d,%r13d+ addl %eax,%r14d+ movdqa %xmm6,48(%rsp)+ cmpb $0,131(%rsi)+ jne .Lssse3_00_47+ rorl $14,%r13d+ movl %r14d,%eax+ movl %r9d,%r12d+ rorl $9,%r14d+ xorl %r8d,%r13d+ xorl %r10d,%r12d+ rorl $5,%r13d+ xorl %eax,%r14d+ andl %r8d,%r12d+ xorl %r8d,%r13d+ addl 0(%rsp),%r11d+ movl %eax,%r15d+ xorl %r10d,%r12d+ rorl $11,%r14d+ xorl %ebx,%r15d+ addl %r12d,%r11d+ rorl $6,%r13d+ andl %r15d,%edi+ xorl %eax,%r14d+ addl %r13d,%r11d+ xorl %ebx,%edi+ rorl $2,%r14d+ addl %r11d,%edx+ addl %edi,%r11d+ movl %edx,%r13d+ addl %r11d,%r14d+ rorl $14,%r13d+ movl %r14d,%r11d+ movl %r8d,%r12d+ rorl $9,%r14d+ xorl %edx,%r13d+ xorl %r9d,%r12d+ rorl $5,%r13d+ xorl %r11d,%r14d+ andl %edx,%r12d+ xorl %edx,%r13d+ addl 4(%rsp),%r10d+ movl %r11d,%edi+ xorl %r9d,%r12d+ rorl $11,%r14d+ xorl %eax,%edi+ addl %r12d,%r10d+ rorl $6,%r13d+ andl %edi,%r15d+ xorl %r11d,%r14d+ addl %r13d,%r10d+ xorl %eax,%r15d+ rorl $2,%r14d+ addl %r10d,%ecx+ addl %r15d,%r10d+ movl %ecx,%r13d+ addl %r10d,%r14d+ rorl $14,%r13d+ movl %r14d,%r10d+ movl %edx,%r12d+ rorl $9,%r14d+ xorl %ecx,%r13d+ xorl %r8d,%r12d+ rorl $5,%r13d+ xorl %r10d,%r14d+ andl %ecx,%r12d+ xorl %ecx,%r13d+ addl 8(%rsp),%r9d+ movl %r10d,%r15d+ xorl %r8d,%r12d+ rorl $11,%r14d+ xorl %r11d,%r15d+ addl %r12d,%r9d+ rorl $6,%r13d+ andl %r15d,%edi+ xorl %r10d,%r14d+ addl %r13d,%r9d+ xorl %r11d,%edi+ rorl $2,%r14d+ addl %r9d,%ebx+ addl %edi,%r9d+ movl %ebx,%r13d+ addl %r9d,%r14d+ rorl $14,%r13d+ movl %r14d,%r9d+ movl %ecx,%r12d+ rorl $9,%r14d+ xorl %ebx,%r13d+ xorl %edx,%r12d+ rorl $5,%r13d+ xorl %r9d,%r14d+ andl %ebx,%r12d+ xorl %ebx,%r13d+ addl 12(%rsp),%r8d+ movl %r9d,%edi+ xorl %edx,%r12d+ rorl $11,%r14d+ xorl %r10d,%edi+ addl %r12d,%r8d+ rorl $6,%r13d+ andl %edi,%r15d+ xorl %r9d,%r14d+ addl %r13d,%r8d+ xorl %r10d,%r15d+ rorl $2,%r14d+ addl %r8d,%eax+ addl %r15d,%r8d+ movl %eax,%r13d+ addl %r8d,%r14d+ rorl $14,%r13d+ movl %r14d,%r8d+ movl %ebx,%r12d+ rorl $9,%r14d+ xorl %eax,%r13d+ xorl %ecx,%r12d+ rorl $5,%r13d+ xorl %r8d,%r14d+ andl %eax,%r12d+ xorl %eax,%r13d+ addl 16(%rsp),%edx+ movl %r8d,%r15d+ xorl %ecx,%r12d+ rorl $11,%r14d+ xorl %r9d,%r15d+ addl %r12d,%edx+ rorl $6,%r13d+ andl %r15d,%edi+ xorl %r8d,%r14d+ addl %r13d,%edx+ xorl %r9d,%edi+ rorl $2,%r14d+ addl %edx,%r11d+ addl %edi,%edx+ movl %r11d,%r13d+ addl %edx,%r14d+ rorl $14,%r13d+ movl %r14d,%edx+ movl %eax,%r12d+ rorl $9,%r14d+ xorl %r11d,%r13d+ xorl %ebx,%r12d+ rorl $5,%r13d+ xorl %edx,%r14d+ andl %r11d,%r12d+ xorl %r11d,%r13d+ addl 20(%rsp),%ecx+ movl %edx,%edi+ xorl %ebx,%r12d+ rorl $11,%r14d+ xorl %r8d,%edi+ addl %r12d,%ecx+ rorl $6,%r13d+ andl %edi,%r15d+ xorl %edx,%r14d+ addl %r13d,%ecx+ xorl %r8d,%r15d+ rorl $2,%r14d+ addl %ecx,%r10d+ addl %r15d,%ecx+ movl %r10d,%r13d+ addl %ecx,%r14d+ rorl $14,%r13d+ movl %r14d,%ecx+ movl %r11d,%r12d+ rorl $9,%r14d+ xorl %r10d,%r13d+ xorl %eax,%r12d+ rorl $5,%r13d+ xorl %ecx,%r14d+ andl %r10d,%r12d+ xorl %r10d,%r13d+ addl 24(%rsp),%ebx+ movl %ecx,%r15d+ xorl %eax,%r12d+ rorl $11,%r14d+ xorl %edx,%r15d+ addl %r12d,%ebx+ rorl $6,%r13d+ andl %r15d,%edi+ xorl %ecx,%r14d+ addl %r13d,%ebx+ xorl %edx,%edi+ rorl $2,%r14d+ addl %ebx,%r9d+ addl %edi,%ebx+ movl %r9d,%r13d+ addl %ebx,%r14d+ rorl $14,%r13d+ movl %r14d,%ebx+ movl %r10d,%r12d+ rorl $9,%r14d+ xorl %r9d,%r13d+ xorl %r11d,%r12d+ rorl $5,%r13d+ xorl %ebx,%r14d+ andl %r9d,%r12d+ xorl %r9d,%r13d+ addl 28(%rsp),%eax+ movl %ebx,%edi+ xorl %r11d,%r12d+ rorl $11,%r14d+ xorl %ecx,%edi+ addl %r12d,%eax+ rorl $6,%r13d+ andl %edi,%r15d+ xorl %ebx,%r14d+ addl %r13d,%eax+ xorl %ecx,%r15d+ rorl $2,%r14d+ addl %eax,%r8d+ addl %r15d,%eax+ movl %r8d,%r13d+ addl %eax,%r14d+ rorl $14,%r13d+ movl %r14d,%eax+ movl %r9d,%r12d+ rorl $9,%r14d+ xorl %r8d,%r13d+ xorl %r10d,%r12d+ rorl $5,%r13d+ xorl %eax,%r14d+ andl %r8d,%r12d+ xorl %r8d,%r13d+ addl 32(%rsp),%r11d+ movl %eax,%r15d+ xorl %r10d,%r12d+ rorl $11,%r14d+ xorl %ebx,%r15d+ addl %r12d,%r11d+ rorl $6,%r13d+ andl %r15d,%edi+ xorl %eax,%r14d+ addl %r13d,%r11d+ xorl %ebx,%edi+ rorl $2,%r14d+ addl %r11d,%edx+ addl %edi,%r11d+ movl %edx,%r13d+ addl %r11d,%r14d+ rorl $14,%r13d+ movl %r14d,%r11d+ movl %r8d,%r12d+ rorl $9,%r14d+ xorl %edx,%r13d+ xorl %r9d,%r12d+ rorl $5,%r13d+ xorl %r11d,%r14d+ andl %edx,%r12d+ xorl %edx,%r13d+ addl 36(%rsp),%r10d+ movl %r11d,%edi+ xorl %r9d,%r12d+ rorl $11,%r14d+ xorl %eax,%edi+ addl %r12d,%r10d+ rorl $6,%r13d+ andl %edi,%r15d+ xorl %r11d,%r14d+ addl %r13d,%r10d+ xorl %eax,%r15d+ rorl $2,%r14d+ addl %r10d,%ecx+ addl %r15d,%r10d+ movl %ecx,%r13d+ addl %r10d,%r14d+ rorl $14,%r13d+ movl %r14d,%r10d+ movl %edx,%r12d+ rorl $9,%r14d+ xorl %ecx,%r13d+ xorl %r8d,%r12d+ rorl $5,%r13d+ xorl %r10d,%r14d+ andl %ecx,%r12d+ xorl %ecx,%r13d+ addl 40(%rsp),%r9d+ movl %r10d,%r15d+ xorl %r8d,%r12d+ rorl $11,%r14d+ xorl %r11d,%r15d+ addl %r12d,%r9d+ rorl $6,%r13d+ andl %r15d,%edi+ xorl %r10d,%r14d+ addl %r13d,%r9d+ xorl %r11d,%edi+ rorl $2,%r14d+ addl %r9d,%ebx+ addl %edi,%r9d+ movl %ebx,%r13d+ addl %r9d,%r14d+ rorl $14,%r13d+ movl %r14d,%r9d+ movl %ecx,%r12d+ rorl $9,%r14d+ xorl %ebx,%r13d+ xorl %edx,%r12d+ rorl $5,%r13d+ xorl %r9d,%r14d+ andl %ebx,%r12d+ xorl %ebx,%r13d+ addl 44(%rsp),%r8d+ movl %r9d,%edi+ xorl %edx,%r12d+ rorl $11,%r14d+ xorl %r10d,%edi+ addl %r12d,%r8d+ rorl $6,%r13d+ andl %edi,%r15d+ xorl %r9d,%r14d+ addl %r13d,%r8d+ xorl %r10d,%r15d+ rorl $2,%r14d+ addl %r8d,%eax+ addl %r15d,%r8d+ movl %eax,%r13d+ addl %r8d,%r14d+ rorl $14,%r13d+ movl %r14d,%r8d+ movl %ebx,%r12d+ rorl $9,%r14d+ xorl %eax,%r13d+ xorl %ecx,%r12d+ rorl $5,%r13d+ xorl %r8d,%r14d+ andl %eax,%r12d+ xorl %eax,%r13d+ addl 48(%rsp),%edx+ movl %r8d,%r15d+ xorl %ecx,%r12d+ rorl $11,%r14d+ xorl %r9d,%r15d+ addl %r12d,%edx+ rorl $6,%r13d+ andl %r15d,%edi+ xorl %r8d,%r14d+ addl %r13d,%edx+ xorl %r9d,%edi+ rorl $2,%r14d+ addl %edx,%r11d+ addl %edi,%edx+ movl %r11d,%r13d+ addl %edx,%r14d+ rorl $14,%r13d+ movl %r14d,%edx+ movl %eax,%r12d+ rorl $9,%r14d+ xorl %r11d,%r13d+ xorl %ebx,%r12d+ rorl $5,%r13d+ xorl %edx,%r14d+ andl %r11d,%r12d+ xorl %r11d,%r13d+ addl 52(%rsp),%ecx+ movl %edx,%edi+ xorl %ebx,%r12d+ rorl $11,%r14d+ xorl %r8d,%edi+ addl %r12d,%ecx+ rorl $6,%r13d+ andl %edi,%r15d+ xorl %edx,%r14d+ addl %r13d,%ecx+ xorl %r8d,%r15d+ rorl $2,%r14d+ addl %ecx,%r10d+ addl %r15d,%ecx+ movl %r10d,%r13d+ addl %ecx,%r14d+ rorl $14,%r13d+ movl %r14d,%ecx+ movl %r11d,%r12d+ rorl $9,%r14d+ xorl %r10d,%r13d+ xorl %eax,%r12d+ rorl $5,%r13d+ xorl %ecx,%r14d+ andl %r10d,%r12d+ xorl %r10d,%r13d+ addl 56(%rsp),%ebx+ movl %ecx,%r15d+ xorl %eax,%r12d+ rorl $11,%r14d+ xorl %edx,%r15d+ addl %r12d,%ebx+ rorl $6,%r13d+ andl %r15d,%edi+ xorl %ecx,%r14d+ addl %r13d,%ebx+ xorl %edx,%edi+ rorl $2,%r14d+ addl %ebx,%r9d+ addl %edi,%ebx+ movl %r9d,%r13d+ addl %ebx,%r14d+ rorl $14,%r13d+ movl %r14d,%ebx+ movl %r10d,%r12d+ rorl $9,%r14d+ xorl %r9d,%r13d+ xorl %r11d,%r12d+ rorl $5,%r13d+ xorl %ebx,%r14d+ andl %r9d,%r12d+ xorl %r9d,%r13d+ addl 60(%rsp),%eax+ movl %ebx,%edi+ xorl %r11d,%r12d+ rorl $11,%r14d+ xorl %ecx,%edi+ addl %r12d,%eax+ rorl $6,%r13d+ andl %edi,%r15d+ xorl %ebx,%r14d+ addl %r13d,%eax+ xorl %ecx,%r15d+ rorl $2,%r14d+ addl %eax,%r8d+ addl %r15d,%eax+ movl %r8d,%r13d+ addl %eax,%r14d+ movq -64(%rbp),%rdi+ movl %r14d,%eax+ movq -56(%rbp),%rsi++ addl 0(%rdi),%eax+ addl 4(%rdi),%ebx+ addl 8(%rdi),%ecx+ addl 12(%rdi),%edx+ addl 16(%rdi),%r8d+ addl 20(%rdi),%r9d+ addl 24(%rdi),%r10d+ addl 28(%rdi),%r11d++ leaq 64(%rsi),%rsi+ cmpq -48(%rbp),%rsi++ movl %eax,0(%rdi)+ movl %ebx,4(%rdi)+ movl %ecx,8(%rdi)+ movl %edx,12(%rdi)+ movl %r8d,16(%rdi)+ movl %r9d,20(%rdi)+ movl %r10d,24(%rdi)+ movl %r11d,28(%rdi)+ jb .Lloop_ssse3++ movq -40(%rbp),%r15+ movq -32(%rbp),%r14+ movq -24(%rbp),%r13+ movq -16(%rbp),%r12+ movq -8(%rbp),%rbx+ movq %rbp,%rsp+.cfi_def_cfa_register %rsp+ popq %rbp+.cfi_adjust_cfa_offset -8+.cfi_restore %rbp+.cfi_restore %r12+.cfi_restore %r13+.cfi_restore %r14+.cfi_restore %r15+.cfi_restore %rbx+ .byte 0xf3,0xc3+.cfi_endproc +.size crypton_sha256_asm_block_data_order_ssse3,.-crypton_sha256_asm_block_data_order_ssse3+.type crypton_sha256_asm_block_data_order_avx,@function+.align 64+crypton_sha256_asm_block_data_order_avx:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa+++ pushq %rbp+.cfi_adjust_cfa_offset 8+.cfi_offset %rbp,-16+ movq %rsp,%rbp+.cfi_def_cfa_register %rbp+.Lavx_shortcut:+ pushq %rbx+.cfi_offset %rbx,-24+ pushq %r12+.cfi_offset %r12,-32+ pushq %r13+.cfi_offset %r13,-40+ pushq %r14+.cfi_offset %r14,-48+ pushq %r15+.cfi_offset %r15,-56+ shlq $4,%rdx+ subq $24,%rsp++ leaq (%rsi,%rdx,4),%rdx+ movq %rdi,-64(%rbp)++ movq %rdx,-48(%rbp)+++ leaq -64(%rsp),%rsp+ vzeroupper+ andq $-64,%rsp+ movl 0(%rdi),%eax+ movl 4(%rdi),%ebx+ movl 8(%rdi),%ecx+ movl 12(%rdi),%edx+ movl 16(%rdi),%r8d+ movl 20(%rdi),%r9d+ movl 24(%rdi),%r10d+ movl 28(%rdi),%r11d+ vmovdqa K256+512+32(%rip),%xmm8+ vmovdqa K256+512+64(%rip),%xmm9+ jmp .Lloop_avx+.align 16+.Lloop_avx:+ vmovdqa K256+512(%rip),%xmm7+ movq %rsi,-56(%rbp)+ vmovdqu 0(%rsi),%xmm0+ vmovdqu 16(%rsi),%xmm1+ vmovdqu 32(%rsi),%xmm2+ vmovdqu 48(%rsi),%xmm3+ vpshufb %xmm7,%xmm0,%xmm0+ leaq K256(%rip),%rsi+ vpshufb %xmm7,%xmm1,%xmm1+ vpshufb %xmm7,%xmm2,%xmm2+ vpaddd 0(%rsi),%xmm0,%xmm4+ vpshufb %xmm7,%xmm3,%xmm3+ vpaddd 32(%rsi),%xmm1,%xmm5+ vpaddd 64(%rsi),%xmm2,%xmm6+ vpaddd 96(%rsi),%xmm3,%xmm7+ vmovdqa %xmm4,0(%rsp)+ movl %eax,%r14d+ vmovdqa %xmm5,16(%rsp)+ movl %ebx,%edi+ vmovdqa %xmm6,32(%rsp)+ xorl %ecx,%edi+ vmovdqa %xmm7,48(%rsp)+ movl %r8d,%r13d+ jmp .Lavx_00_47++.align 16+.Lavx_00_47:+ subq $-128,%rsi+ vpalignr $4,%xmm0,%xmm1,%xmm4+ shrdl $14,%r13d,%r13d+ movl %r14d,%eax+ movl %r9d,%r12d+ vpalignr $4,%xmm2,%xmm3,%xmm7+ shrdl $9,%r14d,%r14d+ xorl %r8d,%r13d+ xorl %r10d,%r12d+ vpsrld $7,%xmm4,%xmm6+ shrdl $5,%r13d,%r13d+ xorl %eax,%r14d+ andl %r8d,%r12d+ vpaddd %xmm7,%xmm0,%xmm0+ xorl %r8d,%r13d+ addl 0(%rsp),%r11d+ movl %eax,%r15d+ vpsrld $3,%xmm4,%xmm7+ xorl %r10d,%r12d+ shrdl $11,%r14d,%r14d+ xorl %ebx,%r15d+ vpslld $14,%xmm4,%xmm5+ addl %r12d,%r11d+ shrdl $6,%r13d,%r13d+ andl %r15d,%edi+ vpxor %xmm6,%xmm7,%xmm4+ xorl %eax,%r14d+ addl %r13d,%r11d+ xorl %ebx,%edi+ vpshufd $250,%xmm3,%xmm7+ shrdl $2,%r14d,%r14d+ addl %r11d,%edx+ addl %edi,%r11d+ vpsrld $11,%xmm6,%xmm6+ movl %edx,%r13d+ addl %r11d,%r14d+ shrdl $14,%r13d,%r13d+ vpxor %xmm5,%xmm4,%xmm4+ movl %r14d,%r11d+ movl %r8d,%r12d+ shrdl $9,%r14d,%r14d+ vpslld $11,%xmm5,%xmm5+ xorl %edx,%r13d+ xorl %r9d,%r12d+ shrdl $5,%r13d,%r13d+ vpxor %xmm6,%xmm4,%xmm4+ xorl %r11d,%r14d+ andl %edx,%r12d+ xorl %edx,%r13d+ vpsrld $10,%xmm7,%xmm6+ addl 4(%rsp),%r10d+ movl %r11d,%edi+ xorl %r9d,%r12d+ vpxor %xmm5,%xmm4,%xmm4+ shrdl $11,%r14d,%r14d+ xorl %eax,%edi+ addl %r12d,%r10d+ vpsrlq $17,%xmm7,%xmm7+ shrdl $6,%r13d,%r13d+ andl %edi,%r15d+ xorl %r11d,%r14d+ vpaddd %xmm4,%xmm0,%xmm0+ addl %r13d,%r10d+ xorl %eax,%r15d+ shrdl $2,%r14d,%r14d+ vpxor %xmm7,%xmm6,%xmm6+ addl %r10d,%ecx+ addl %r15d,%r10d+ movl %ecx,%r13d+ vpsrlq $2,%xmm7,%xmm7+ addl %r10d,%r14d+ shrdl $14,%r13d,%r13d+ movl %r14d,%r10d+ vpxor %xmm7,%xmm6,%xmm6+ movl %edx,%r12d+ shrdl $9,%r14d,%r14d+ xorl %ecx,%r13d+ vpshufb %xmm8,%xmm6,%xmm6+ xorl %r8d,%r12d+ shrdl $5,%r13d,%r13d+ xorl %r10d,%r14d+ vpaddd %xmm6,%xmm0,%xmm0+ andl %ecx,%r12d+ xorl %ecx,%r13d+ addl 8(%rsp),%r9d+ vpshufd $80,%xmm0,%xmm7+ movl %r10d,%r15d+ xorl %r8d,%r12d+ shrdl $11,%r14d,%r14d+ vpsrld $10,%xmm7,%xmm6+ xorl %r11d,%r15d+ addl %r12d,%r9d+ shrdl $6,%r13d,%r13d+ vpsrlq $17,%xmm7,%xmm7+ andl %r15d,%edi+ xorl %r10d,%r14d+ addl %r13d,%r9d+ vpxor %xmm7,%xmm6,%xmm6+ xorl %r11d,%edi+ shrdl $2,%r14d,%r14d+ addl %r9d,%ebx+ vpsrlq $2,%xmm7,%xmm7+ addl %edi,%r9d+ movl %ebx,%r13d+ addl %r9d,%r14d+ vpxor %xmm7,%xmm6,%xmm6+ shrdl $14,%r13d,%r13d+ movl %r14d,%r9d+ movl %ecx,%r12d+ vpshufb %xmm9,%xmm6,%xmm6+ shrdl $9,%r14d,%r14d+ xorl %ebx,%r13d+ xorl %edx,%r12d+ vpaddd %xmm6,%xmm0,%xmm0+ shrdl $5,%r13d,%r13d+ xorl %r9d,%r14d+ andl %ebx,%r12d+ vpaddd 0(%rsi),%xmm0,%xmm6+ xorl %ebx,%r13d+ addl 12(%rsp),%r8d+ movl %r9d,%edi+ xorl %edx,%r12d+ shrdl $11,%r14d,%r14d+ xorl %r10d,%edi+ addl %r12d,%r8d+ shrdl $6,%r13d,%r13d+ andl %edi,%r15d+ xorl %r9d,%r14d+ addl %r13d,%r8d+ xorl %r10d,%r15d+ shrdl $2,%r14d,%r14d+ addl %r8d,%eax+ addl %r15d,%r8d+ movl %eax,%r13d+ addl %r8d,%r14d+ vmovdqa %xmm6,0(%rsp)+ vpalignr $4,%xmm1,%xmm2,%xmm4+ shrdl $14,%r13d,%r13d+ movl %r14d,%r8d+ movl %ebx,%r12d+ vpalignr $4,%xmm3,%xmm0,%xmm7+ shrdl $9,%r14d,%r14d+ xorl %eax,%r13d+ xorl %ecx,%r12d+ vpsrld $7,%xmm4,%xmm6+ shrdl $5,%r13d,%r13d+ xorl %r8d,%r14d+ andl %eax,%r12d+ vpaddd %xmm7,%xmm1,%xmm1+ xorl %eax,%r13d+ addl 16(%rsp),%edx+ movl %r8d,%r15d+ vpsrld $3,%xmm4,%xmm7+ xorl %ecx,%r12d+ shrdl $11,%r14d,%r14d+ xorl %r9d,%r15d+ vpslld $14,%xmm4,%xmm5+ addl %r12d,%edx+ shrdl $6,%r13d,%r13d+ andl %r15d,%edi+ vpxor %xmm6,%xmm7,%xmm4+ xorl %r8d,%r14d+ addl %r13d,%edx+ xorl %r9d,%edi+ vpshufd $250,%xmm0,%xmm7+ shrdl $2,%r14d,%r14d+ addl %edx,%r11d+ addl %edi,%edx+ vpsrld $11,%xmm6,%xmm6+ movl %r11d,%r13d+ addl %edx,%r14d+ shrdl $14,%r13d,%r13d+ vpxor %xmm5,%xmm4,%xmm4+ movl %r14d,%edx+ movl %eax,%r12d+ shrdl $9,%r14d,%r14d+ vpslld $11,%xmm5,%xmm5+ xorl %r11d,%r13d+ xorl %ebx,%r12d+ shrdl $5,%r13d,%r13d+ vpxor %xmm6,%xmm4,%xmm4+ xorl %edx,%r14d+ andl %r11d,%r12d+ xorl %r11d,%r13d+ vpsrld $10,%xmm7,%xmm6+ addl 20(%rsp),%ecx+ movl %edx,%edi+ xorl %ebx,%r12d+ vpxor %xmm5,%xmm4,%xmm4+ shrdl $11,%r14d,%r14d+ xorl %r8d,%edi+ addl %r12d,%ecx+ vpsrlq $17,%xmm7,%xmm7+ shrdl $6,%r13d,%r13d+ andl %edi,%r15d+ xorl %edx,%r14d+ vpaddd %xmm4,%xmm1,%xmm1+ addl %r13d,%ecx+ xorl %r8d,%r15d+ shrdl $2,%r14d,%r14d+ vpxor %xmm7,%xmm6,%xmm6+ addl %ecx,%r10d+ addl %r15d,%ecx+ movl %r10d,%r13d+ vpsrlq $2,%xmm7,%xmm7+ addl %ecx,%r14d+ shrdl $14,%r13d,%r13d+ movl %r14d,%ecx+ vpxor %xmm7,%xmm6,%xmm6+ movl %r11d,%r12d+ shrdl $9,%r14d,%r14d+ xorl %r10d,%r13d+ vpshufb %xmm8,%xmm6,%xmm6+ xorl %eax,%r12d+ shrdl $5,%r13d,%r13d+ xorl %ecx,%r14d+ vpaddd %xmm6,%xmm1,%xmm1+ andl %r10d,%r12d+ xorl %r10d,%r13d+ addl 24(%rsp),%ebx+ vpshufd $80,%xmm1,%xmm7+ movl %ecx,%r15d+ xorl %eax,%r12d+ shrdl $11,%r14d,%r14d+ vpsrld $10,%xmm7,%xmm6+ xorl %edx,%r15d+ addl %r12d,%ebx+ shrdl $6,%r13d,%r13d+ vpsrlq $17,%xmm7,%xmm7+ andl %r15d,%edi+ xorl %ecx,%r14d+ addl %r13d,%ebx+ vpxor %xmm7,%xmm6,%xmm6+ xorl %edx,%edi+ shrdl $2,%r14d,%r14d+ addl %ebx,%r9d+ vpsrlq $2,%xmm7,%xmm7+ addl %edi,%ebx+ movl %r9d,%r13d+ addl %ebx,%r14d+ vpxor %xmm7,%xmm6,%xmm6+ shrdl $14,%r13d,%r13d+ movl %r14d,%ebx+ movl %r10d,%r12d+ vpshufb %xmm9,%xmm6,%xmm6+ shrdl $9,%r14d,%r14d+ xorl %r9d,%r13d+ xorl %r11d,%r12d+ vpaddd %xmm6,%xmm1,%xmm1+ shrdl $5,%r13d,%r13d+ xorl %ebx,%r14d+ andl %r9d,%r12d+ vpaddd 32(%rsi),%xmm1,%xmm6+ xorl %r9d,%r13d+ addl 28(%rsp),%eax+ movl %ebx,%edi+ xorl %r11d,%r12d+ shrdl $11,%r14d,%r14d+ xorl %ecx,%edi+ addl %r12d,%eax+ shrdl $6,%r13d,%r13d+ andl %edi,%r15d+ xorl %ebx,%r14d+ addl %r13d,%eax+ xorl %ecx,%r15d+ shrdl $2,%r14d,%r14d+ addl %eax,%r8d+ addl %r15d,%eax+ movl %r8d,%r13d+ addl %eax,%r14d+ vmovdqa %xmm6,16(%rsp)+ vpalignr $4,%xmm2,%xmm3,%xmm4+ shrdl $14,%r13d,%r13d+ movl %r14d,%eax+ movl %r9d,%r12d+ vpalignr $4,%xmm0,%xmm1,%xmm7+ shrdl $9,%r14d,%r14d+ xorl %r8d,%r13d+ xorl %r10d,%r12d+ vpsrld $7,%xmm4,%xmm6+ shrdl $5,%r13d,%r13d+ xorl %eax,%r14d+ andl %r8d,%r12d+ vpaddd %xmm7,%xmm2,%xmm2+ xorl %r8d,%r13d+ addl 32(%rsp),%r11d+ movl %eax,%r15d+ vpsrld $3,%xmm4,%xmm7+ xorl %r10d,%r12d+ shrdl $11,%r14d,%r14d+ xorl %ebx,%r15d+ vpslld $14,%xmm4,%xmm5+ addl %r12d,%r11d+ shrdl $6,%r13d,%r13d+ andl %r15d,%edi+ vpxor %xmm6,%xmm7,%xmm4+ xorl %eax,%r14d+ addl %r13d,%r11d+ xorl %ebx,%edi+ vpshufd $250,%xmm1,%xmm7+ shrdl $2,%r14d,%r14d+ addl %r11d,%edx+ addl %edi,%r11d+ vpsrld $11,%xmm6,%xmm6+ movl %edx,%r13d+ addl %r11d,%r14d+ shrdl $14,%r13d,%r13d+ vpxor %xmm5,%xmm4,%xmm4+ movl %r14d,%r11d+ movl %r8d,%r12d+ shrdl $9,%r14d,%r14d+ vpslld $11,%xmm5,%xmm5+ xorl %edx,%r13d+ xorl %r9d,%r12d+ shrdl $5,%r13d,%r13d+ vpxor %xmm6,%xmm4,%xmm4+ xorl %r11d,%r14d+ andl %edx,%r12d+ xorl %edx,%r13d+ vpsrld $10,%xmm7,%xmm6+ addl 36(%rsp),%r10d+ movl %r11d,%edi+ xorl %r9d,%r12d+ vpxor %xmm5,%xmm4,%xmm4+ shrdl $11,%r14d,%r14d+ xorl %eax,%edi+ addl %r12d,%r10d+ vpsrlq $17,%xmm7,%xmm7+ shrdl $6,%r13d,%r13d+ andl %edi,%r15d+ xorl %r11d,%r14d+ vpaddd %xmm4,%xmm2,%xmm2+ addl %r13d,%r10d+ xorl %eax,%r15d+ shrdl $2,%r14d,%r14d+ vpxor %xmm7,%xmm6,%xmm6+ addl %r10d,%ecx+ addl %r15d,%r10d+ movl %ecx,%r13d+ vpsrlq $2,%xmm7,%xmm7+ addl %r10d,%r14d+ shrdl $14,%r13d,%r13d+ movl %r14d,%r10d+ vpxor %xmm7,%xmm6,%xmm6+ movl %edx,%r12d+ shrdl $9,%r14d,%r14d+ xorl %ecx,%r13d+ vpshufb %xmm8,%xmm6,%xmm6+ xorl %r8d,%r12d+ shrdl $5,%r13d,%r13d+ xorl %r10d,%r14d+ vpaddd %xmm6,%xmm2,%xmm2+ andl %ecx,%r12d+ xorl %ecx,%r13d+ addl 40(%rsp),%r9d+ vpshufd $80,%xmm2,%xmm7+ movl %r10d,%r15d+ xorl %r8d,%r12d+ shrdl $11,%r14d,%r14d+ vpsrld $10,%xmm7,%xmm6+ xorl %r11d,%r15d+ addl %r12d,%r9d+ shrdl $6,%r13d,%r13d+ vpsrlq $17,%xmm7,%xmm7+ andl %r15d,%edi+ xorl %r10d,%r14d+ addl %r13d,%r9d+ vpxor %xmm7,%xmm6,%xmm6+ xorl %r11d,%edi+ shrdl $2,%r14d,%r14d+ addl %r9d,%ebx+ vpsrlq $2,%xmm7,%xmm7+ addl %edi,%r9d+ movl %ebx,%r13d+ addl %r9d,%r14d+ vpxor %xmm7,%xmm6,%xmm6+ shrdl $14,%r13d,%r13d+ movl %r14d,%r9d+ movl %ecx,%r12d+ vpshufb %xmm9,%xmm6,%xmm6+ shrdl $9,%r14d,%r14d+ xorl %ebx,%r13d+ xorl %edx,%r12d+ vpaddd %xmm6,%xmm2,%xmm2+ shrdl $5,%r13d,%r13d+ xorl %r9d,%r14d+ andl %ebx,%r12d+ vpaddd 64(%rsi),%xmm2,%xmm6+ xorl %ebx,%r13d+ addl 44(%rsp),%r8d+ movl %r9d,%edi+ xorl %edx,%r12d+ shrdl $11,%r14d,%r14d+ xorl %r10d,%edi+ addl %r12d,%r8d+ shrdl $6,%r13d,%r13d+ andl %edi,%r15d+ xorl %r9d,%r14d+ addl %r13d,%r8d+ xorl %r10d,%r15d+ shrdl $2,%r14d,%r14d+ addl %r8d,%eax+ addl %r15d,%r8d+ movl %eax,%r13d+ addl %r8d,%r14d+ vmovdqa %xmm6,32(%rsp)+ vpalignr $4,%xmm3,%xmm0,%xmm4+ shrdl $14,%r13d,%r13d+ movl %r14d,%r8d+ movl %ebx,%r12d+ vpalignr $4,%xmm1,%xmm2,%xmm7+ shrdl $9,%r14d,%r14d+ xorl %eax,%r13d+ xorl %ecx,%r12d+ vpsrld $7,%xmm4,%xmm6+ shrdl $5,%r13d,%r13d+ xorl %r8d,%r14d+ andl %eax,%r12d+ vpaddd %xmm7,%xmm3,%xmm3+ xorl %eax,%r13d+ addl 48(%rsp),%edx+ movl %r8d,%r15d+ vpsrld $3,%xmm4,%xmm7+ xorl %ecx,%r12d+ shrdl $11,%r14d,%r14d+ xorl %r9d,%r15d+ vpslld $14,%xmm4,%xmm5+ addl %r12d,%edx+ shrdl $6,%r13d,%r13d+ andl %r15d,%edi+ vpxor %xmm6,%xmm7,%xmm4+ xorl %r8d,%r14d+ addl %r13d,%edx+ xorl %r9d,%edi+ vpshufd $250,%xmm2,%xmm7+ shrdl $2,%r14d,%r14d+ addl %edx,%r11d+ addl %edi,%edx+ vpsrld $11,%xmm6,%xmm6+ movl %r11d,%r13d+ addl %edx,%r14d+ shrdl $14,%r13d,%r13d+ vpxor %xmm5,%xmm4,%xmm4+ movl %r14d,%edx+ movl %eax,%r12d+ shrdl $9,%r14d,%r14d+ vpslld $11,%xmm5,%xmm5+ xorl %r11d,%r13d+ xorl %ebx,%r12d+ shrdl $5,%r13d,%r13d+ vpxor %xmm6,%xmm4,%xmm4+ xorl %edx,%r14d+ andl %r11d,%r12d+ xorl %r11d,%r13d+ vpsrld $10,%xmm7,%xmm6+ addl 52(%rsp),%ecx+ movl %edx,%edi+ xorl %ebx,%r12d+ vpxor %xmm5,%xmm4,%xmm4+ shrdl $11,%r14d,%r14d+ xorl %r8d,%edi+ addl %r12d,%ecx+ vpsrlq $17,%xmm7,%xmm7+ shrdl $6,%r13d,%r13d+ andl %edi,%r15d+ xorl %edx,%r14d+ vpaddd %xmm4,%xmm3,%xmm3+ addl %r13d,%ecx+ xorl %r8d,%r15d+ shrdl $2,%r14d,%r14d+ vpxor %xmm7,%xmm6,%xmm6+ addl %ecx,%r10d+ addl %r15d,%ecx+ movl %r10d,%r13d+ vpsrlq $2,%xmm7,%xmm7+ addl %ecx,%r14d+ shrdl $14,%r13d,%r13d+ movl %r14d,%ecx+ vpxor %xmm7,%xmm6,%xmm6+ movl %r11d,%r12d+ shrdl $9,%r14d,%r14d+ xorl %r10d,%r13d+ vpshufb %xmm8,%xmm6,%xmm6+ xorl %eax,%r12d+ shrdl $5,%r13d,%r13d+ xorl %ecx,%r14d+ vpaddd %xmm6,%xmm3,%xmm3+ andl %r10d,%r12d+ xorl %r10d,%r13d+ addl 56(%rsp),%ebx+ vpshufd $80,%xmm3,%xmm7+ movl %ecx,%r15d+ xorl %eax,%r12d+ shrdl $11,%r14d,%r14d+ vpsrld $10,%xmm7,%xmm6+ xorl %edx,%r15d+ addl %r12d,%ebx+ shrdl $6,%r13d,%r13d+ vpsrlq $17,%xmm7,%xmm7+ andl %r15d,%edi+ xorl %ecx,%r14d+ addl %r13d,%ebx+ vpxor %xmm7,%xmm6,%xmm6+ xorl %edx,%edi+ shrdl $2,%r14d,%r14d+ addl %ebx,%r9d+ vpsrlq $2,%xmm7,%xmm7+ addl %edi,%ebx+ movl %r9d,%r13d+ addl %ebx,%r14d+ vpxor %xmm7,%xmm6,%xmm6+ shrdl $14,%r13d,%r13d+ movl %r14d,%ebx+ movl %r10d,%r12d+ vpshufb %xmm9,%xmm6,%xmm6+ shrdl $9,%r14d,%r14d+ xorl %r9d,%r13d+ xorl %r11d,%r12d+ vpaddd %xmm6,%xmm3,%xmm3+ shrdl $5,%r13d,%r13d+ xorl %ebx,%r14d+ andl %r9d,%r12d+ vpaddd 96(%rsi),%xmm3,%xmm6+ xorl %r9d,%r13d+ addl 60(%rsp),%eax+ movl %ebx,%edi+ xorl %r11d,%r12d+ shrdl $11,%r14d,%r14d+ xorl %ecx,%edi+ addl %r12d,%eax+ shrdl $6,%r13d,%r13d+ andl %edi,%r15d+ xorl %ebx,%r14d+ addl %r13d,%eax+ xorl %ecx,%r15d+ shrdl $2,%r14d,%r14d+ addl %eax,%r8d+ addl %r15d,%eax+ movl %r8d,%r13d+ addl %eax,%r14d+ vmovdqa %xmm6,48(%rsp)+ cmpb $0,131(%rsi)+ jne .Lavx_00_47+ shrdl $14,%r13d,%r13d+ movl %r14d,%eax+ movl %r9d,%r12d+ shrdl $9,%r14d,%r14d+ xorl %r8d,%r13d+ xorl %r10d,%r12d+ shrdl $5,%r13d,%r13d+ xorl %eax,%r14d+ andl %r8d,%r12d+ xorl %r8d,%r13d+ addl 0(%rsp),%r11d+ movl %eax,%r15d+ xorl %r10d,%r12d+ shrdl $11,%r14d,%r14d+ xorl %ebx,%r15d+ addl %r12d,%r11d+ shrdl $6,%r13d,%r13d+ andl %r15d,%edi+ xorl %eax,%r14d+ addl %r13d,%r11d+ xorl %ebx,%edi+ shrdl $2,%r14d,%r14d+ addl %r11d,%edx+ addl %edi,%r11d+ movl %edx,%r13d+ addl %r11d,%r14d+ shrdl $14,%r13d,%r13d+ movl %r14d,%r11d+ movl %r8d,%r12d+ shrdl $9,%r14d,%r14d+ xorl %edx,%r13d+ xorl %r9d,%r12d+ shrdl $5,%r13d,%r13d+ xorl %r11d,%r14d+ andl %edx,%r12d+ xorl %edx,%r13d+ addl 4(%rsp),%r10d+ movl %r11d,%edi+ xorl %r9d,%r12d+ shrdl $11,%r14d,%r14d+ xorl %eax,%edi+ addl %r12d,%r10d+ shrdl $6,%r13d,%r13d+ andl %edi,%r15d+ xorl %r11d,%r14d+ addl %r13d,%r10d+ xorl %eax,%r15d+ shrdl $2,%r14d,%r14d+ addl %r10d,%ecx+ addl %r15d,%r10d+ movl %ecx,%r13d+ addl %r10d,%r14d+ shrdl $14,%r13d,%r13d+ movl %r14d,%r10d+ movl %edx,%r12d+ shrdl $9,%r14d,%r14d+ xorl %ecx,%r13d+ xorl %r8d,%r12d+ shrdl $5,%r13d,%r13d+ xorl %r10d,%r14d+ andl %ecx,%r12d+ xorl %ecx,%r13d+ addl 8(%rsp),%r9d+ movl %r10d,%r15d+ xorl %r8d,%r12d+ shrdl $11,%r14d,%r14d+ xorl %r11d,%r15d+ addl %r12d,%r9d+ shrdl $6,%r13d,%r13d+ andl %r15d,%edi+ xorl %r10d,%r14d+ addl %r13d,%r9d+ xorl %r11d,%edi+ shrdl $2,%r14d,%r14d+ addl %r9d,%ebx+ addl %edi,%r9d+ movl %ebx,%r13d+ addl %r9d,%r14d+ shrdl $14,%r13d,%r13d+ movl %r14d,%r9d+ movl %ecx,%r12d+ shrdl $9,%r14d,%r14d+ xorl %ebx,%r13d+ xorl %edx,%r12d+ shrdl $5,%r13d,%r13d+ xorl %r9d,%r14d+ andl %ebx,%r12d+ xorl %ebx,%r13d+ addl 12(%rsp),%r8d+ movl %r9d,%edi+ xorl %edx,%r12d+ shrdl $11,%r14d,%r14d+ xorl %r10d,%edi+ addl %r12d,%r8d+ shrdl $6,%r13d,%r13d+ andl %edi,%r15d+ xorl %r9d,%r14d+ addl %r13d,%r8d+ xorl %r10d,%r15d+ shrdl $2,%r14d,%r14d+ addl %r8d,%eax+ addl %r15d,%r8d+ movl %eax,%r13d+ addl %r8d,%r14d+ shrdl $14,%r13d,%r13d+ movl %r14d,%r8d+ movl %ebx,%r12d+ shrdl $9,%r14d,%r14d+ xorl %eax,%r13d+ xorl %ecx,%r12d+ shrdl $5,%r13d,%r13d+ xorl %r8d,%r14d+ andl %eax,%r12d+ xorl %eax,%r13d+ addl 16(%rsp),%edx+ movl %r8d,%r15d+ xorl %ecx,%r12d+ shrdl $11,%r14d,%r14d+ xorl %r9d,%r15d+ addl %r12d,%edx+ shrdl $6,%r13d,%r13d+ andl %r15d,%edi+ xorl %r8d,%r14d+ addl %r13d,%edx+ xorl %r9d,%edi+ shrdl $2,%r14d,%r14d+ addl %edx,%r11d+ addl %edi,%edx+ movl %r11d,%r13d+ addl %edx,%r14d+ shrdl $14,%r13d,%r13d+ movl %r14d,%edx+ movl %eax,%r12d+ shrdl $9,%r14d,%r14d+ xorl %r11d,%r13d+ xorl %ebx,%r12d+ shrdl $5,%r13d,%r13d+ xorl %edx,%r14d+ andl %r11d,%r12d+ xorl %r11d,%r13d+ addl 20(%rsp),%ecx+ movl %edx,%edi+ xorl %ebx,%r12d+ shrdl $11,%r14d,%r14d+ xorl %r8d,%edi+ addl %r12d,%ecx+ shrdl $6,%r13d,%r13d+ andl %edi,%r15d+ xorl %edx,%r14d+ addl %r13d,%ecx+ xorl %r8d,%r15d+ shrdl $2,%r14d,%r14d+ addl %ecx,%r10d+ addl %r15d,%ecx+ movl %r10d,%r13d+ addl %ecx,%r14d+ shrdl $14,%r13d,%r13d+ movl %r14d,%ecx+ movl %r11d,%r12d+ shrdl $9,%r14d,%r14d+ xorl %r10d,%r13d+ xorl %eax,%r12d+ shrdl $5,%r13d,%r13d+ xorl %ecx,%r14d+ andl %r10d,%r12d+ xorl %r10d,%r13d+ addl 24(%rsp),%ebx+ movl %ecx,%r15d+ xorl %eax,%r12d+ shrdl $11,%r14d,%r14d+ xorl %edx,%r15d+ addl %r12d,%ebx+ shrdl $6,%r13d,%r13d+ andl %r15d,%edi+ xorl %ecx,%r14d+ addl %r13d,%ebx+ xorl %edx,%edi+ shrdl $2,%r14d,%r14d+ addl %ebx,%r9d+ addl %edi,%ebx+ movl %r9d,%r13d+ addl %ebx,%r14d+ shrdl $14,%r13d,%r13d+ movl %r14d,%ebx+ movl %r10d,%r12d+ shrdl $9,%r14d,%r14d+ xorl %r9d,%r13d+ xorl %r11d,%r12d+ shrdl $5,%r13d,%r13d+ xorl %ebx,%r14d+ andl %r9d,%r12d+ xorl %r9d,%r13d+ addl 28(%rsp),%eax+ movl %ebx,%edi+ xorl %r11d,%r12d+ shrdl $11,%r14d,%r14d+ xorl %ecx,%edi+ addl %r12d,%eax+ shrdl $6,%r13d,%r13d+ andl %edi,%r15d+ xorl %ebx,%r14d+ addl %r13d,%eax+ xorl %ecx,%r15d+ shrdl $2,%r14d,%r14d+ addl %eax,%r8d+ addl %r15d,%eax+ movl %r8d,%r13d+ addl %eax,%r14d+ shrdl $14,%r13d,%r13d+ movl %r14d,%eax+ movl %r9d,%r12d+ shrdl $9,%r14d,%r14d+ xorl %r8d,%r13d+ xorl %r10d,%r12d+ shrdl $5,%r13d,%r13d+ xorl %eax,%r14d+ andl %r8d,%r12d+ xorl %r8d,%r13d+ addl 32(%rsp),%r11d+ movl %eax,%r15d+ xorl %r10d,%r12d+ shrdl $11,%r14d,%r14d+ xorl %ebx,%r15d+ addl %r12d,%r11d+ shrdl $6,%r13d,%r13d+ andl %r15d,%edi+ xorl %eax,%r14d+ addl %r13d,%r11d+ xorl %ebx,%edi+ shrdl $2,%r14d,%r14d+ addl %r11d,%edx+ addl %edi,%r11d+ movl %edx,%r13d+ addl %r11d,%r14d+ shrdl $14,%r13d,%r13d+ movl %r14d,%r11d+ movl %r8d,%r12d+ shrdl $9,%r14d,%r14d+ xorl %edx,%r13d+ xorl %r9d,%r12d+ shrdl $5,%r13d,%r13d+ xorl %r11d,%r14d+ andl %edx,%r12d+ xorl %edx,%r13d+ addl 36(%rsp),%r10d+ movl %r11d,%edi+ xorl %r9d,%r12d+ shrdl $11,%r14d,%r14d+ xorl %eax,%edi+ addl %r12d,%r10d+ shrdl $6,%r13d,%r13d+ andl %edi,%r15d+ xorl %r11d,%r14d+ addl %r13d,%r10d+ xorl %eax,%r15d+ shrdl $2,%r14d,%r14d+ addl %r10d,%ecx+ addl %r15d,%r10d+ movl %ecx,%r13d+ addl %r10d,%r14d+ shrdl $14,%r13d,%r13d+ movl %r14d,%r10d+ movl %edx,%r12d+ shrdl $9,%r14d,%r14d+ xorl %ecx,%r13d+ xorl %r8d,%r12d+ shrdl $5,%r13d,%r13d+ xorl %r10d,%r14d+ andl %ecx,%r12d+ xorl %ecx,%r13d+ addl 40(%rsp),%r9d+ movl %r10d,%r15d+ xorl %r8d,%r12d+ shrdl $11,%r14d,%r14d+ xorl %r11d,%r15d+ addl %r12d,%r9d+ shrdl $6,%r13d,%r13d+ andl %r15d,%edi+ xorl %r10d,%r14d+ addl %r13d,%r9d+ xorl %r11d,%edi+ shrdl $2,%r14d,%r14d+ addl %r9d,%ebx+ addl %edi,%r9d+ movl %ebx,%r13d+ addl %r9d,%r14d+ shrdl $14,%r13d,%r13d+ movl %r14d,%r9d+ movl %ecx,%r12d+ shrdl $9,%r14d,%r14d+ xorl %ebx,%r13d+ xorl %edx,%r12d+ shrdl $5,%r13d,%r13d+ xorl %r9d,%r14d+ andl %ebx,%r12d+ xorl %ebx,%r13d+ addl 44(%rsp),%r8d+ movl %r9d,%edi+ xorl %edx,%r12d+ shrdl $11,%r14d,%r14d+ xorl %r10d,%edi+ addl %r12d,%r8d+ shrdl $6,%r13d,%r13d+ andl %edi,%r15d+ xorl %r9d,%r14d+ addl %r13d,%r8d+ xorl %r10d,%r15d+ shrdl $2,%r14d,%r14d+ addl %r8d,%eax+ addl %r15d,%r8d+ movl %eax,%r13d+ addl %r8d,%r14d+ shrdl $14,%r13d,%r13d+ movl %r14d,%r8d+ movl %ebx,%r12d+ shrdl $9,%r14d,%r14d+ xorl %eax,%r13d+ xorl %ecx,%r12d+ shrdl $5,%r13d,%r13d+ xorl %r8d,%r14d+ andl %eax,%r12d+ xorl %eax,%r13d+ addl 48(%rsp),%edx+ movl %r8d,%r15d+ xorl %ecx,%r12d+ shrdl $11,%r14d,%r14d+ xorl %r9d,%r15d+ addl %r12d,%edx+ shrdl $6,%r13d,%r13d+ andl %r15d,%edi+ xorl %r8d,%r14d+ addl %r13d,%edx+ xorl %r9d,%edi+ shrdl $2,%r14d,%r14d+ addl %edx,%r11d+ addl %edi,%edx+ movl %r11d,%r13d+ addl %edx,%r14d+ shrdl $14,%r13d,%r13d+ movl %r14d,%edx+ movl %eax,%r12d+ shrdl $9,%r14d,%r14d+ xorl %r11d,%r13d+ xorl %ebx,%r12d+ shrdl $5,%r13d,%r13d+ xorl %edx,%r14d+ andl %r11d,%r12d+ xorl %r11d,%r13d+ addl 52(%rsp),%ecx+ movl %edx,%edi+ xorl %ebx,%r12d+ shrdl $11,%r14d,%r14d+ xorl %r8d,%edi+ addl %r12d,%ecx+ shrdl $6,%r13d,%r13d+ andl %edi,%r15d+ xorl %edx,%r14d+ addl %r13d,%ecx+ xorl %r8d,%r15d+ shrdl $2,%r14d,%r14d+ addl %ecx,%r10d+ addl %r15d,%ecx+ movl %r10d,%r13d+ addl %ecx,%r14d+ shrdl $14,%r13d,%r13d+ movl %r14d,%ecx+ movl %r11d,%r12d+ shrdl $9,%r14d,%r14d+ xorl %r10d,%r13d+ xorl %eax,%r12d+ shrdl $5,%r13d,%r13d+ xorl %ecx,%r14d+ andl %r10d,%r12d+ xorl %r10d,%r13d+ addl 56(%rsp),%ebx+ movl %ecx,%r15d+ xorl %eax,%r12d+ shrdl $11,%r14d,%r14d+ xorl %edx,%r15d+ addl %r12d,%ebx+ shrdl $6,%r13d,%r13d+ andl %r15d,%edi+ xorl %ecx,%r14d+ addl %r13d,%ebx+ xorl %edx,%edi+ shrdl $2,%r14d,%r14d+ addl %ebx,%r9d+ addl %edi,%ebx+ movl %r9d,%r13d+ addl %ebx,%r14d+ shrdl $14,%r13d,%r13d+ movl %r14d,%ebx+ movl %r10d,%r12d+ shrdl $9,%r14d,%r14d+ xorl %r9d,%r13d+ xorl %r11d,%r12d+ shrdl $5,%r13d,%r13d+ xorl %ebx,%r14d+ andl %r9d,%r12d+ xorl %r9d,%r13d+ addl 60(%rsp),%eax+ movl %ebx,%edi+ xorl %r11d,%r12d+ shrdl $11,%r14d,%r14d+ xorl %ecx,%edi+ addl %r12d,%eax+ shrdl $6,%r13d,%r13d+ andl %edi,%r15d+ xorl %ebx,%r14d+ addl %r13d,%eax+ xorl %ecx,%r15d+ shrdl $2,%r14d,%r14d+ addl %eax,%r8d+ addl %r15d,%eax+ movl %r8d,%r13d+ addl %eax,%r14d+ movq -64(%rbp),%rdi+ movl %r14d,%eax+ movq -56(%rbp),%rsi++ addl 0(%rdi),%eax+ addl 4(%rdi),%ebx+ addl 8(%rdi),%ecx+ addl 12(%rdi),%edx+ addl 16(%rdi),%r8d+ addl 20(%rdi),%r9d+ addl 24(%rdi),%r10d+ addl 28(%rdi),%r11d++ leaq 64(%rsi),%rsi+ cmpq -48(%rbp),%rsi++ movl %eax,0(%rdi)+ movl %ebx,4(%rdi)+ movl %ecx,8(%rdi)+ movl %edx,12(%rdi)+ movl %r8d,16(%rdi)+ movl %r9d,20(%rdi)+ movl %r10d,24(%rdi)+ movl %r11d,28(%rdi)+ jb .Lloop_avx++ vzeroupper+ movq -40(%rbp),%r15+ movq -32(%rbp),%r14+ movq -24(%rbp),%r13+ movq -16(%rbp),%r12+ movq -8(%rbp),%rbx+ movq %rbp,%rsp+.cfi_def_cfa_register %rsp+ popq %rbp+.cfi_adjust_cfa_offset -8+.cfi_restore %rbp+.cfi_restore %r12+.cfi_restore %r13+.cfi_restore %r14+.cfi_restore %r15+.cfi_restore %rbx+ .byte 0xf3,0xc3+.cfi_endproc +.size crypton_sha256_asm_block_data_order_avx,.-crypton_sha256_asm_block_data_order_avx+.type crypton_sha256_asm_block_data_order_avx2,@function+.align 64+crypton_sha256_asm_block_data_order_avx2:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa+++ pushq %rbp+.cfi_adjust_cfa_offset 8+.cfi_offset %rbp,-16+ movq %rsp,%rbp+.cfi_def_cfa_register %rbp+.Lavx2_shortcut:+ pushq %rbx+.cfi_offset %rbx,-24+ pushq %r12+.cfi_offset %r12,-32+ pushq %r13+.cfi_offset %r13,-40+ pushq %r14+.cfi_offset %r14,-48+ pushq %r15+.cfi_offset %r15,-56+ shlq $4,%rdx+ subq $24,%rsp++ leaq (%rsi,%rdx,4),%rdx+ movq %rdi,-64(%rbp)++ movq %rdx,-48(%rbp)+++ leaq -64(%rsp),%rsp+ vzeroupper+ andq $-64,%rsp+ subq $-64,%rsi+ movl 0(%rdi),%eax+ movq %rsi,%r12+ movl 4(%rdi),%ebx+ cmpq %rdx,%rsi+ movl 8(%rdi),%ecx+ cmoveq %rsp,%r12+ movl 12(%rdi),%edx+ movl 16(%rdi),%r8d+ movl 20(%rdi),%r9d+ movl 24(%rdi),%r10d+ movl 28(%rdi),%r11d+ vmovdqa K256+512+32(%rip),%ymm8+ vmovdqa K256+512+64(%rip),%ymm9+ jmp .Loop_avx2+.align 16+.Loop_avx2:+ vmovdqa K256+512(%rip),%ymm7+ movq %rsi,-56(%rbp)+ vmovdqu -64+0(%rsi),%xmm0+ vmovdqu -64+16(%rsi),%xmm1+ vmovdqu -64+32(%rsi),%xmm2+ vmovdqu -64+48(%rsi),%xmm3+ leaq K256(%rip),%rsi+ vinserti128 $1,(%r12),%ymm0,%ymm0+ vinserti128 $1,16(%r12),%ymm1,%ymm1+ vpshufb %ymm7,%ymm0,%ymm0+ vinserti128 $1,32(%r12),%ymm2,%ymm2+ vpshufb %ymm7,%ymm1,%ymm1+ vinserti128 $1,48(%r12),%ymm3,%ymm3++ vpshufb %ymm7,%ymm2,%ymm2+ vpaddd 0(%rsi),%ymm0,%ymm4+ vpshufb %ymm7,%ymm3,%ymm3+ vpaddd 32(%rsi),%ymm1,%ymm5+ vpaddd 64(%rsi),%ymm2,%ymm6+ vpaddd 96(%rsi),%ymm3,%ymm7+ vmovdqa %ymm4,0(%rsp)+ xorl %r14d,%r14d+ vmovdqa %ymm5,32(%rsp)+ leaq -64(%rsp),%rsp+ movl %ebx,%edi+ vmovdqa %ymm6,0(%rsp)+ xorl %ecx,%edi+ vmovdqa %ymm7,32(%rsp)+ movl %r9d,%r12d+ subq $-32*4,%rsi+ jmp .Lavx2_00_47++.align 16+.Lavx2_00_47:+ leaq -64(%rsp),%rsp+ vpalignr $4,%ymm0,%ymm1,%ymm4+ addl 0+128(%rsp),%r11d+ andl %r8d,%r12d+ rorxl $25,%r8d,%r13d+ vpalignr $4,%ymm2,%ymm3,%ymm7+ rorxl $11,%r8d,%r15d+ leal (%rax,%r14,1),%eax+ leal (%r11,%r12,1),%r11d+ vpsrld $7,%ymm4,%ymm6+ andnl %r10d,%r8d,%r12d+ xorl %r15d,%r13d+ rorxl $6,%r8d,%r14d+ vpaddd %ymm7,%ymm0,%ymm0+ leal (%r11,%r12,1),%r11d+ xorl %r14d,%r13d+ movl %eax,%r15d+ vpsrld $3,%ymm4,%ymm7+ rorxl $22,%eax,%r12d+ leal (%r11,%r13,1),%r11d+ xorl %ebx,%r15d+ vpslld $14,%ymm4,%ymm5+ rorxl $13,%eax,%r14d+ rorxl $2,%eax,%r13d+ leal (%rdx,%r11,1),%edx+ vpxor %ymm6,%ymm7,%ymm4+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %ebx,%edi+ vpshufd $250,%ymm3,%ymm7+ xorl %r13d,%r14d+ leal (%r11,%rdi,1),%r11d+ movl %r8d,%r12d+ vpsrld $11,%ymm6,%ymm6+ addl 4+128(%rsp),%r10d+ andl %edx,%r12d+ rorxl $25,%edx,%r13d+ vpxor %ymm5,%ymm4,%ymm4+ rorxl $11,%edx,%edi+ leal (%r11,%r14,1),%r11d+ leal (%r10,%r12,1),%r10d+ vpslld $11,%ymm5,%ymm5+ andnl %r9d,%edx,%r12d+ xorl %edi,%r13d+ rorxl $6,%edx,%r14d+ vpxor %ymm6,%ymm4,%ymm4+ leal (%r10,%r12,1),%r10d+ xorl %r14d,%r13d+ movl %r11d,%edi+ vpsrld $10,%ymm7,%ymm6+ rorxl $22,%r11d,%r12d+ leal (%r10,%r13,1),%r10d+ xorl %eax,%edi+ vpxor %ymm5,%ymm4,%ymm4+ rorxl $13,%r11d,%r14d+ rorxl $2,%r11d,%r13d+ leal (%rcx,%r10,1),%ecx+ vpsrlq $17,%ymm7,%ymm7+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %eax,%r15d+ vpaddd %ymm4,%ymm0,%ymm0+ xorl %r13d,%r14d+ leal (%r10,%r15,1),%r10d+ movl %edx,%r12d+ vpxor %ymm7,%ymm6,%ymm6+ addl 8+128(%rsp),%r9d+ andl %ecx,%r12d+ rorxl $25,%ecx,%r13d+ vpsrlq $2,%ymm7,%ymm7+ rorxl $11,%ecx,%r15d+ leal (%r10,%r14,1),%r10d+ leal (%r9,%r12,1),%r9d+ vpxor %ymm7,%ymm6,%ymm6+ andnl %r8d,%ecx,%r12d+ xorl %r15d,%r13d+ rorxl $6,%ecx,%r14d+ vpshufb %ymm8,%ymm6,%ymm6+ leal (%r9,%r12,1),%r9d+ xorl %r14d,%r13d+ movl %r10d,%r15d+ vpaddd %ymm6,%ymm0,%ymm0+ rorxl $22,%r10d,%r12d+ leal (%r9,%r13,1),%r9d+ xorl %r11d,%r15d+ vpshufd $80,%ymm0,%ymm7+ rorxl $13,%r10d,%r14d+ rorxl $2,%r10d,%r13d+ leal (%rbx,%r9,1),%ebx+ vpsrld $10,%ymm7,%ymm6+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %r11d,%edi+ vpsrlq $17,%ymm7,%ymm7+ xorl %r13d,%r14d+ leal (%r9,%rdi,1),%r9d+ movl %ecx,%r12d+ vpxor %ymm7,%ymm6,%ymm6+ addl 12+128(%rsp),%r8d+ andl %ebx,%r12d+ rorxl $25,%ebx,%r13d+ vpsrlq $2,%ymm7,%ymm7+ rorxl $11,%ebx,%edi+ leal (%r9,%r14,1),%r9d+ leal (%r8,%r12,1),%r8d+ vpxor %ymm7,%ymm6,%ymm6+ andnl %edx,%ebx,%r12d+ xorl %edi,%r13d+ rorxl $6,%ebx,%r14d+ vpshufb %ymm9,%ymm6,%ymm6+ leal (%r8,%r12,1),%r8d+ xorl %r14d,%r13d+ movl %r9d,%edi+ vpaddd %ymm6,%ymm0,%ymm0+ rorxl $22,%r9d,%r12d+ leal (%r8,%r13,1),%r8d+ xorl %r10d,%edi+ vpaddd 0(%rsi),%ymm0,%ymm6+ rorxl $13,%r9d,%r14d+ rorxl $2,%r9d,%r13d+ leal (%rax,%r8,1),%eax+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %r10d,%r15d+ xorl %r13d,%r14d+ leal (%r8,%r15,1),%r8d+ movl %ebx,%r12d+ vmovdqa %ymm6,0(%rsp)+ vpalignr $4,%ymm1,%ymm2,%ymm4+ addl 32+128(%rsp),%edx+ andl %eax,%r12d+ rorxl $25,%eax,%r13d+ vpalignr $4,%ymm3,%ymm0,%ymm7+ rorxl $11,%eax,%r15d+ leal (%r8,%r14,1),%r8d+ leal (%rdx,%r12,1),%edx+ vpsrld $7,%ymm4,%ymm6+ andnl %ecx,%eax,%r12d+ xorl %r15d,%r13d+ rorxl $6,%eax,%r14d+ vpaddd %ymm7,%ymm1,%ymm1+ leal (%rdx,%r12,1),%edx+ xorl %r14d,%r13d+ movl %r8d,%r15d+ vpsrld $3,%ymm4,%ymm7+ rorxl $22,%r8d,%r12d+ leal (%rdx,%r13,1),%edx+ xorl %r9d,%r15d+ vpslld $14,%ymm4,%ymm5+ rorxl $13,%r8d,%r14d+ rorxl $2,%r8d,%r13d+ leal (%r11,%rdx,1),%r11d+ vpxor %ymm6,%ymm7,%ymm4+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %r9d,%edi+ vpshufd $250,%ymm0,%ymm7+ xorl %r13d,%r14d+ leal (%rdx,%rdi,1),%edx+ movl %eax,%r12d+ vpsrld $11,%ymm6,%ymm6+ addl 36+128(%rsp),%ecx+ andl %r11d,%r12d+ rorxl $25,%r11d,%r13d+ vpxor %ymm5,%ymm4,%ymm4+ rorxl $11,%r11d,%edi+ leal (%rdx,%r14,1),%edx+ leal (%rcx,%r12,1),%ecx+ vpslld $11,%ymm5,%ymm5+ andnl %ebx,%r11d,%r12d+ xorl %edi,%r13d+ rorxl $6,%r11d,%r14d+ vpxor %ymm6,%ymm4,%ymm4+ leal (%rcx,%r12,1),%ecx+ xorl %r14d,%r13d+ movl %edx,%edi+ vpsrld $10,%ymm7,%ymm6+ rorxl $22,%edx,%r12d+ leal (%rcx,%r13,1),%ecx+ xorl %r8d,%edi+ vpxor %ymm5,%ymm4,%ymm4+ rorxl $13,%edx,%r14d+ rorxl $2,%edx,%r13d+ leal (%r10,%rcx,1),%r10d+ vpsrlq $17,%ymm7,%ymm7+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %r8d,%r15d+ vpaddd %ymm4,%ymm1,%ymm1+ xorl %r13d,%r14d+ leal (%rcx,%r15,1),%ecx+ movl %r11d,%r12d+ vpxor %ymm7,%ymm6,%ymm6+ addl 40+128(%rsp),%ebx+ andl %r10d,%r12d+ rorxl $25,%r10d,%r13d+ vpsrlq $2,%ymm7,%ymm7+ rorxl $11,%r10d,%r15d+ leal (%rcx,%r14,1),%ecx+ leal (%rbx,%r12,1),%ebx+ vpxor %ymm7,%ymm6,%ymm6+ andnl %eax,%r10d,%r12d+ xorl %r15d,%r13d+ rorxl $6,%r10d,%r14d+ vpshufb %ymm8,%ymm6,%ymm6+ leal (%rbx,%r12,1),%ebx+ xorl %r14d,%r13d+ movl %ecx,%r15d+ vpaddd %ymm6,%ymm1,%ymm1+ rorxl $22,%ecx,%r12d+ leal (%rbx,%r13,1),%ebx+ xorl %edx,%r15d+ vpshufd $80,%ymm1,%ymm7+ rorxl $13,%ecx,%r14d+ rorxl $2,%ecx,%r13d+ leal (%r9,%rbx,1),%r9d+ vpsrld $10,%ymm7,%ymm6+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %edx,%edi+ vpsrlq $17,%ymm7,%ymm7+ xorl %r13d,%r14d+ leal (%rbx,%rdi,1),%ebx+ movl %r10d,%r12d+ vpxor %ymm7,%ymm6,%ymm6+ addl 44+128(%rsp),%eax+ andl %r9d,%r12d+ rorxl $25,%r9d,%r13d+ vpsrlq $2,%ymm7,%ymm7+ rorxl $11,%r9d,%edi+ leal (%rbx,%r14,1),%ebx+ leal (%rax,%r12,1),%eax+ vpxor %ymm7,%ymm6,%ymm6+ andnl %r11d,%r9d,%r12d+ xorl %edi,%r13d+ rorxl $6,%r9d,%r14d+ vpshufb %ymm9,%ymm6,%ymm6+ leal (%rax,%r12,1),%eax+ xorl %r14d,%r13d+ movl %ebx,%edi+ vpaddd %ymm6,%ymm1,%ymm1+ rorxl $22,%ebx,%r12d+ leal (%rax,%r13,1),%eax+ xorl %ecx,%edi+ vpaddd 32(%rsi),%ymm1,%ymm6+ rorxl $13,%ebx,%r14d+ rorxl $2,%ebx,%r13d+ leal (%r8,%rax,1),%r8d+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %ecx,%r15d+ xorl %r13d,%r14d+ leal (%rax,%r15,1),%eax+ movl %r9d,%r12d+ vmovdqa %ymm6,32(%rsp)+ leaq -64(%rsp),%rsp+ vpalignr $4,%ymm2,%ymm3,%ymm4+ addl 0+128(%rsp),%r11d+ andl %r8d,%r12d+ rorxl $25,%r8d,%r13d+ vpalignr $4,%ymm0,%ymm1,%ymm7+ rorxl $11,%r8d,%r15d+ leal (%rax,%r14,1),%eax+ leal (%r11,%r12,1),%r11d+ vpsrld $7,%ymm4,%ymm6+ andnl %r10d,%r8d,%r12d+ xorl %r15d,%r13d+ rorxl $6,%r8d,%r14d+ vpaddd %ymm7,%ymm2,%ymm2+ leal (%r11,%r12,1),%r11d+ xorl %r14d,%r13d+ movl %eax,%r15d+ vpsrld $3,%ymm4,%ymm7+ rorxl $22,%eax,%r12d+ leal (%r11,%r13,1),%r11d+ xorl %ebx,%r15d+ vpslld $14,%ymm4,%ymm5+ rorxl $13,%eax,%r14d+ rorxl $2,%eax,%r13d+ leal (%rdx,%r11,1),%edx+ vpxor %ymm6,%ymm7,%ymm4+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %ebx,%edi+ vpshufd $250,%ymm1,%ymm7+ xorl %r13d,%r14d+ leal (%r11,%rdi,1),%r11d+ movl %r8d,%r12d+ vpsrld $11,%ymm6,%ymm6+ addl 4+128(%rsp),%r10d+ andl %edx,%r12d+ rorxl $25,%edx,%r13d+ vpxor %ymm5,%ymm4,%ymm4+ rorxl $11,%edx,%edi+ leal (%r11,%r14,1),%r11d+ leal (%r10,%r12,1),%r10d+ vpslld $11,%ymm5,%ymm5+ andnl %r9d,%edx,%r12d+ xorl %edi,%r13d+ rorxl $6,%edx,%r14d+ vpxor %ymm6,%ymm4,%ymm4+ leal (%r10,%r12,1),%r10d+ xorl %r14d,%r13d+ movl %r11d,%edi+ vpsrld $10,%ymm7,%ymm6+ rorxl $22,%r11d,%r12d+ leal (%r10,%r13,1),%r10d+ xorl %eax,%edi+ vpxor %ymm5,%ymm4,%ymm4+ rorxl $13,%r11d,%r14d+ rorxl $2,%r11d,%r13d+ leal (%rcx,%r10,1),%ecx+ vpsrlq $17,%ymm7,%ymm7+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %eax,%r15d+ vpaddd %ymm4,%ymm2,%ymm2+ xorl %r13d,%r14d+ leal (%r10,%r15,1),%r10d+ movl %edx,%r12d+ vpxor %ymm7,%ymm6,%ymm6+ addl 8+128(%rsp),%r9d+ andl %ecx,%r12d+ rorxl $25,%ecx,%r13d+ vpsrlq $2,%ymm7,%ymm7+ rorxl $11,%ecx,%r15d+ leal (%r10,%r14,1),%r10d+ leal (%r9,%r12,1),%r9d+ vpxor %ymm7,%ymm6,%ymm6+ andnl %r8d,%ecx,%r12d+ xorl %r15d,%r13d+ rorxl $6,%ecx,%r14d+ vpshufb %ymm8,%ymm6,%ymm6+ leal (%r9,%r12,1),%r9d+ xorl %r14d,%r13d+ movl %r10d,%r15d+ vpaddd %ymm6,%ymm2,%ymm2+ rorxl $22,%r10d,%r12d+ leal (%r9,%r13,1),%r9d+ xorl %r11d,%r15d+ vpshufd $80,%ymm2,%ymm7+ rorxl $13,%r10d,%r14d+ rorxl $2,%r10d,%r13d+ leal (%rbx,%r9,1),%ebx+ vpsrld $10,%ymm7,%ymm6+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %r11d,%edi+ vpsrlq $17,%ymm7,%ymm7+ xorl %r13d,%r14d+ leal (%r9,%rdi,1),%r9d+ movl %ecx,%r12d+ vpxor %ymm7,%ymm6,%ymm6+ addl 12+128(%rsp),%r8d+ andl %ebx,%r12d+ rorxl $25,%ebx,%r13d+ vpsrlq $2,%ymm7,%ymm7+ rorxl $11,%ebx,%edi+ leal (%r9,%r14,1),%r9d+ leal (%r8,%r12,1),%r8d+ vpxor %ymm7,%ymm6,%ymm6+ andnl %edx,%ebx,%r12d+ xorl %edi,%r13d+ rorxl $6,%ebx,%r14d+ vpshufb %ymm9,%ymm6,%ymm6+ leal (%r8,%r12,1),%r8d+ xorl %r14d,%r13d+ movl %r9d,%edi+ vpaddd %ymm6,%ymm2,%ymm2+ rorxl $22,%r9d,%r12d+ leal (%r8,%r13,1),%r8d+ xorl %r10d,%edi+ vpaddd 64(%rsi),%ymm2,%ymm6+ rorxl $13,%r9d,%r14d+ rorxl $2,%r9d,%r13d+ leal (%rax,%r8,1),%eax+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %r10d,%r15d+ xorl %r13d,%r14d+ leal (%r8,%r15,1),%r8d+ movl %ebx,%r12d+ vmovdqa %ymm6,0(%rsp)+ vpalignr $4,%ymm3,%ymm0,%ymm4+ addl 32+128(%rsp),%edx+ andl %eax,%r12d+ rorxl $25,%eax,%r13d+ vpalignr $4,%ymm1,%ymm2,%ymm7+ rorxl $11,%eax,%r15d+ leal (%r8,%r14,1),%r8d+ leal (%rdx,%r12,1),%edx+ vpsrld $7,%ymm4,%ymm6+ andnl %ecx,%eax,%r12d+ xorl %r15d,%r13d+ rorxl $6,%eax,%r14d+ vpaddd %ymm7,%ymm3,%ymm3+ leal (%rdx,%r12,1),%edx+ xorl %r14d,%r13d+ movl %r8d,%r15d+ vpsrld $3,%ymm4,%ymm7+ rorxl $22,%r8d,%r12d+ leal (%rdx,%r13,1),%edx+ xorl %r9d,%r15d+ vpslld $14,%ymm4,%ymm5+ rorxl $13,%r8d,%r14d+ rorxl $2,%r8d,%r13d+ leal (%r11,%rdx,1),%r11d+ vpxor %ymm6,%ymm7,%ymm4+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %r9d,%edi+ vpshufd $250,%ymm2,%ymm7+ xorl %r13d,%r14d+ leal (%rdx,%rdi,1),%edx+ movl %eax,%r12d+ vpsrld $11,%ymm6,%ymm6+ addl 36+128(%rsp),%ecx+ andl %r11d,%r12d+ rorxl $25,%r11d,%r13d+ vpxor %ymm5,%ymm4,%ymm4+ rorxl $11,%r11d,%edi+ leal (%rdx,%r14,1),%edx+ leal (%rcx,%r12,1),%ecx+ vpslld $11,%ymm5,%ymm5+ andnl %ebx,%r11d,%r12d+ xorl %edi,%r13d+ rorxl $6,%r11d,%r14d+ vpxor %ymm6,%ymm4,%ymm4+ leal (%rcx,%r12,1),%ecx+ xorl %r14d,%r13d+ movl %edx,%edi+ vpsrld $10,%ymm7,%ymm6+ rorxl $22,%edx,%r12d+ leal (%rcx,%r13,1),%ecx+ xorl %r8d,%edi+ vpxor %ymm5,%ymm4,%ymm4+ rorxl $13,%edx,%r14d+ rorxl $2,%edx,%r13d+ leal (%r10,%rcx,1),%r10d+ vpsrlq $17,%ymm7,%ymm7+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %r8d,%r15d+ vpaddd %ymm4,%ymm3,%ymm3+ xorl %r13d,%r14d+ leal (%rcx,%r15,1),%ecx+ movl %r11d,%r12d+ vpxor %ymm7,%ymm6,%ymm6+ addl 40+128(%rsp),%ebx+ andl %r10d,%r12d+ rorxl $25,%r10d,%r13d+ vpsrlq $2,%ymm7,%ymm7+ rorxl $11,%r10d,%r15d+ leal (%rcx,%r14,1),%ecx+ leal (%rbx,%r12,1),%ebx+ vpxor %ymm7,%ymm6,%ymm6+ andnl %eax,%r10d,%r12d+ xorl %r15d,%r13d+ rorxl $6,%r10d,%r14d+ vpshufb %ymm8,%ymm6,%ymm6+ leal (%rbx,%r12,1),%ebx+ xorl %r14d,%r13d+ movl %ecx,%r15d+ vpaddd %ymm6,%ymm3,%ymm3+ rorxl $22,%ecx,%r12d+ leal (%rbx,%r13,1),%ebx+ xorl %edx,%r15d+ vpshufd $80,%ymm3,%ymm7+ rorxl $13,%ecx,%r14d+ rorxl $2,%ecx,%r13d+ leal (%r9,%rbx,1),%r9d+ vpsrld $10,%ymm7,%ymm6+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %edx,%edi+ vpsrlq $17,%ymm7,%ymm7+ xorl %r13d,%r14d+ leal (%rbx,%rdi,1),%ebx+ movl %r10d,%r12d+ vpxor %ymm7,%ymm6,%ymm6+ addl 44+128(%rsp),%eax+ andl %r9d,%r12d+ rorxl $25,%r9d,%r13d+ vpsrlq $2,%ymm7,%ymm7+ rorxl $11,%r9d,%edi+ leal (%rbx,%r14,1),%ebx+ leal (%rax,%r12,1),%eax+ vpxor %ymm7,%ymm6,%ymm6+ andnl %r11d,%r9d,%r12d+ xorl %edi,%r13d+ rorxl $6,%r9d,%r14d+ vpshufb %ymm9,%ymm6,%ymm6+ leal (%rax,%r12,1),%eax+ xorl %r14d,%r13d+ movl %ebx,%edi+ vpaddd %ymm6,%ymm3,%ymm3+ rorxl $22,%ebx,%r12d+ leal (%rax,%r13,1),%eax+ xorl %ecx,%edi+ vpaddd 96(%rsi),%ymm3,%ymm6+ rorxl $13,%ebx,%r14d+ rorxl $2,%ebx,%r13d+ leal (%r8,%rax,1),%r8d+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %ecx,%r15d+ xorl %r13d,%r14d+ leal (%rax,%r15,1),%eax+ movl %r9d,%r12d+ vmovdqa %ymm6,32(%rsp)+ leaq 128(%rsi),%rsi+ cmpb $0,3(%rsi)+ jne .Lavx2_00_47+ addl 0+64(%rsp),%r11d+ andl %r8d,%r12d+ rorxl $25,%r8d,%r13d+ rorxl $11,%r8d,%r15d+ leal (%rax,%r14,1),%eax+ leal (%r11,%r12,1),%r11d+ andnl %r10d,%r8d,%r12d+ xorl %r15d,%r13d+ rorxl $6,%r8d,%r14d+ leal (%r11,%r12,1),%r11d+ xorl %r14d,%r13d+ movl %eax,%r15d+ rorxl $22,%eax,%r12d+ leal (%r11,%r13,1),%r11d+ xorl %ebx,%r15d+ rorxl $13,%eax,%r14d+ rorxl $2,%eax,%r13d+ leal (%rdx,%r11,1),%edx+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %ebx,%edi+ xorl %r13d,%r14d+ leal (%r11,%rdi,1),%r11d+ movl %r8d,%r12d+ addl 4+64(%rsp),%r10d+ andl %edx,%r12d+ rorxl $25,%edx,%r13d+ rorxl $11,%edx,%edi+ leal (%r11,%r14,1),%r11d+ leal (%r10,%r12,1),%r10d+ andnl %r9d,%edx,%r12d+ xorl %edi,%r13d+ rorxl $6,%edx,%r14d+ leal (%r10,%r12,1),%r10d+ xorl %r14d,%r13d+ movl %r11d,%edi+ rorxl $22,%r11d,%r12d+ leal (%r10,%r13,1),%r10d+ xorl %eax,%edi+ rorxl $13,%r11d,%r14d+ rorxl $2,%r11d,%r13d+ leal (%rcx,%r10,1),%ecx+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %eax,%r15d+ xorl %r13d,%r14d+ leal (%r10,%r15,1),%r10d+ movl %edx,%r12d+ addl 8+64(%rsp),%r9d+ andl %ecx,%r12d+ rorxl $25,%ecx,%r13d+ rorxl $11,%ecx,%r15d+ leal (%r10,%r14,1),%r10d+ leal (%r9,%r12,1),%r9d+ andnl %r8d,%ecx,%r12d+ xorl %r15d,%r13d+ rorxl $6,%ecx,%r14d+ leal (%r9,%r12,1),%r9d+ xorl %r14d,%r13d+ movl %r10d,%r15d+ rorxl $22,%r10d,%r12d+ leal (%r9,%r13,1),%r9d+ xorl %r11d,%r15d+ rorxl $13,%r10d,%r14d+ rorxl $2,%r10d,%r13d+ leal (%rbx,%r9,1),%ebx+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %r11d,%edi+ xorl %r13d,%r14d+ leal (%r9,%rdi,1),%r9d+ movl %ecx,%r12d+ addl 12+64(%rsp),%r8d+ andl %ebx,%r12d+ rorxl $25,%ebx,%r13d+ rorxl $11,%ebx,%edi+ leal (%r9,%r14,1),%r9d+ leal (%r8,%r12,1),%r8d+ andnl %edx,%ebx,%r12d+ xorl %edi,%r13d+ rorxl $6,%ebx,%r14d+ leal (%r8,%r12,1),%r8d+ xorl %r14d,%r13d+ movl %r9d,%edi+ rorxl $22,%r9d,%r12d+ leal (%r8,%r13,1),%r8d+ xorl %r10d,%edi+ rorxl $13,%r9d,%r14d+ rorxl $2,%r9d,%r13d+ leal (%rax,%r8,1),%eax+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %r10d,%r15d+ xorl %r13d,%r14d+ leal (%r8,%r15,1),%r8d+ movl %ebx,%r12d+ addl 32+64(%rsp),%edx+ andl %eax,%r12d+ rorxl $25,%eax,%r13d+ rorxl $11,%eax,%r15d+ leal (%r8,%r14,1),%r8d+ leal (%rdx,%r12,1),%edx+ andnl %ecx,%eax,%r12d+ xorl %r15d,%r13d+ rorxl $6,%eax,%r14d+ leal (%rdx,%r12,1),%edx+ xorl %r14d,%r13d+ movl %r8d,%r15d+ rorxl $22,%r8d,%r12d+ leal (%rdx,%r13,1),%edx+ xorl %r9d,%r15d+ rorxl $13,%r8d,%r14d+ rorxl $2,%r8d,%r13d+ leal (%r11,%rdx,1),%r11d+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %r9d,%edi+ xorl %r13d,%r14d+ leal (%rdx,%rdi,1),%edx+ movl %eax,%r12d+ addl 36+64(%rsp),%ecx+ andl %r11d,%r12d+ rorxl $25,%r11d,%r13d+ rorxl $11,%r11d,%edi+ leal (%rdx,%r14,1),%edx+ leal (%rcx,%r12,1),%ecx+ andnl %ebx,%r11d,%r12d+ xorl %edi,%r13d+ rorxl $6,%r11d,%r14d+ leal (%rcx,%r12,1),%ecx+ xorl %r14d,%r13d+ movl %edx,%edi+ rorxl $22,%edx,%r12d+ leal (%rcx,%r13,1),%ecx+ xorl %r8d,%edi+ rorxl $13,%edx,%r14d+ rorxl $2,%edx,%r13d+ leal (%r10,%rcx,1),%r10d+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %r8d,%r15d+ xorl %r13d,%r14d+ leal (%rcx,%r15,1),%ecx+ movl %r11d,%r12d+ addl 40+64(%rsp),%ebx+ andl %r10d,%r12d+ rorxl $25,%r10d,%r13d+ rorxl $11,%r10d,%r15d+ leal (%rcx,%r14,1),%ecx+ leal (%rbx,%r12,1),%ebx+ andnl %eax,%r10d,%r12d+ xorl %r15d,%r13d+ rorxl $6,%r10d,%r14d+ leal (%rbx,%r12,1),%ebx+ xorl %r14d,%r13d+ movl %ecx,%r15d+ rorxl $22,%ecx,%r12d+ leal (%rbx,%r13,1),%ebx+ xorl %edx,%r15d+ rorxl $13,%ecx,%r14d+ rorxl $2,%ecx,%r13d+ leal (%r9,%rbx,1),%r9d+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %edx,%edi+ xorl %r13d,%r14d+ leal (%rbx,%rdi,1),%ebx+ movl %r10d,%r12d+ addl 44+64(%rsp),%eax+ andl %r9d,%r12d+ rorxl $25,%r9d,%r13d+ rorxl $11,%r9d,%edi+ leal (%rbx,%r14,1),%ebx+ leal (%rax,%r12,1),%eax+ andnl %r11d,%r9d,%r12d+ xorl %edi,%r13d+ rorxl $6,%r9d,%r14d+ leal (%rax,%r12,1),%eax+ xorl %r14d,%r13d+ movl %ebx,%edi+ rorxl $22,%ebx,%r12d+ leal (%rax,%r13,1),%eax+ xorl %ecx,%edi+ rorxl $13,%ebx,%r14d+ rorxl $2,%ebx,%r13d+ leal (%r8,%rax,1),%r8d+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %ecx,%r15d+ xorl %r13d,%r14d+ leal (%rax,%r15,1),%eax+ movl %r9d,%r12d+ addl 0(%rsp),%r11d+ andl %r8d,%r12d+ rorxl $25,%r8d,%r13d+ rorxl $11,%r8d,%r15d+ leal (%rax,%r14,1),%eax+ leal (%r11,%r12,1),%r11d+ andnl %r10d,%r8d,%r12d+ xorl %r15d,%r13d+ rorxl $6,%r8d,%r14d+ leal (%r11,%r12,1),%r11d+ xorl %r14d,%r13d+ movl %eax,%r15d+ rorxl $22,%eax,%r12d+ leal (%r11,%r13,1),%r11d+ xorl %ebx,%r15d+ rorxl $13,%eax,%r14d+ rorxl $2,%eax,%r13d+ leal (%rdx,%r11,1),%edx+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %ebx,%edi+ xorl %r13d,%r14d+ leal (%r11,%rdi,1),%r11d+ movl %r8d,%r12d+ addl 4(%rsp),%r10d+ andl %edx,%r12d+ rorxl $25,%edx,%r13d+ rorxl $11,%edx,%edi+ leal (%r11,%r14,1),%r11d+ leal (%r10,%r12,1),%r10d+ andnl %r9d,%edx,%r12d+ xorl %edi,%r13d+ rorxl $6,%edx,%r14d+ leal (%r10,%r12,1),%r10d+ xorl %r14d,%r13d+ movl %r11d,%edi+ rorxl $22,%r11d,%r12d+ leal (%r10,%r13,1),%r10d+ xorl %eax,%edi+ rorxl $13,%r11d,%r14d+ rorxl $2,%r11d,%r13d+ leal (%rcx,%r10,1),%ecx+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %eax,%r15d+ xorl %r13d,%r14d+ leal (%r10,%r15,1),%r10d+ movl %edx,%r12d+ addl 8(%rsp),%r9d+ andl %ecx,%r12d+ rorxl $25,%ecx,%r13d+ rorxl $11,%ecx,%r15d+ leal (%r10,%r14,1),%r10d+ leal (%r9,%r12,1),%r9d+ andnl %r8d,%ecx,%r12d+ xorl %r15d,%r13d+ rorxl $6,%ecx,%r14d+ leal (%r9,%r12,1),%r9d+ xorl %r14d,%r13d+ movl %r10d,%r15d+ rorxl $22,%r10d,%r12d+ leal (%r9,%r13,1),%r9d+ xorl %r11d,%r15d+ rorxl $13,%r10d,%r14d+ rorxl $2,%r10d,%r13d+ leal (%rbx,%r9,1),%ebx+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %r11d,%edi+ xorl %r13d,%r14d+ leal (%r9,%rdi,1),%r9d+ movl %ecx,%r12d+ addl 12(%rsp),%r8d+ andl %ebx,%r12d+ rorxl $25,%ebx,%r13d+ rorxl $11,%ebx,%edi+ leal (%r9,%r14,1),%r9d+ leal (%r8,%r12,1),%r8d+ andnl %edx,%ebx,%r12d+ xorl %edi,%r13d+ rorxl $6,%ebx,%r14d+ leal (%r8,%r12,1),%r8d+ xorl %r14d,%r13d+ movl %r9d,%edi+ rorxl $22,%r9d,%r12d+ leal (%r8,%r13,1),%r8d+ xorl %r10d,%edi+ rorxl $13,%r9d,%r14d+ rorxl $2,%r9d,%r13d+ leal (%rax,%r8,1),%eax+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %r10d,%r15d+ xorl %r13d,%r14d+ leal (%r8,%r15,1),%r8d+ movl %ebx,%r12d+ addl 32(%rsp),%edx+ andl %eax,%r12d+ rorxl $25,%eax,%r13d+ rorxl $11,%eax,%r15d+ leal (%r8,%r14,1),%r8d+ leal (%rdx,%r12,1),%edx+ andnl %ecx,%eax,%r12d+ xorl %r15d,%r13d+ rorxl $6,%eax,%r14d+ leal (%rdx,%r12,1),%edx+ xorl %r14d,%r13d+ movl %r8d,%r15d+ rorxl $22,%r8d,%r12d+ leal (%rdx,%r13,1),%edx+ xorl %r9d,%r15d+ rorxl $13,%r8d,%r14d+ rorxl $2,%r8d,%r13d+ leal (%r11,%rdx,1),%r11d+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %r9d,%edi+ xorl %r13d,%r14d+ leal (%rdx,%rdi,1),%edx+ movl %eax,%r12d+ addl 36(%rsp),%ecx+ andl %r11d,%r12d+ rorxl $25,%r11d,%r13d+ rorxl $11,%r11d,%edi+ leal (%rdx,%r14,1),%edx+ leal (%rcx,%r12,1),%ecx+ andnl %ebx,%r11d,%r12d+ xorl %edi,%r13d+ rorxl $6,%r11d,%r14d+ leal (%rcx,%r12,1),%ecx+ xorl %r14d,%r13d+ movl %edx,%edi+ rorxl $22,%edx,%r12d+ leal (%rcx,%r13,1),%ecx+ xorl %r8d,%edi+ rorxl $13,%edx,%r14d+ rorxl $2,%edx,%r13d+ leal (%r10,%rcx,1),%r10d+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %r8d,%r15d+ xorl %r13d,%r14d+ leal (%rcx,%r15,1),%ecx+ movl %r11d,%r12d+ addl 40(%rsp),%ebx+ andl %r10d,%r12d+ rorxl $25,%r10d,%r13d+ rorxl $11,%r10d,%r15d+ leal (%rcx,%r14,1),%ecx+ leal (%rbx,%r12,1),%ebx+ andnl %eax,%r10d,%r12d+ xorl %r15d,%r13d+ rorxl $6,%r10d,%r14d+ leal (%rbx,%r12,1),%ebx+ xorl %r14d,%r13d+ movl %ecx,%r15d+ rorxl $22,%ecx,%r12d+ leal (%rbx,%r13,1),%ebx+ xorl %edx,%r15d+ rorxl $13,%ecx,%r14d+ rorxl $2,%ecx,%r13d+ leal (%r9,%rbx,1),%r9d+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %edx,%edi+ xorl %r13d,%r14d+ leal (%rbx,%rdi,1),%ebx+ movl %r10d,%r12d+ addl 44(%rsp),%eax+ andl %r9d,%r12d+ rorxl $25,%r9d,%r13d+ rorxl $11,%r9d,%edi+ leal (%rbx,%r14,1),%ebx+ leal (%rax,%r12,1),%eax+ andnl %r11d,%r9d,%r12d+ xorl %edi,%r13d+ rorxl $6,%r9d,%r14d+ leal (%rax,%r12,1),%eax+ xorl %r14d,%r13d+ movl %ebx,%edi+ rorxl $22,%ebx,%r12d+ leal (%rax,%r13,1),%eax+ xorl %ecx,%edi+ rorxl $13,%ebx,%r14d+ rorxl $2,%ebx,%r13d+ leal (%r8,%rax,1),%r8d+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %ecx,%r15d+ xorl %r13d,%r14d+ leal (%rax,%r15,1),%eax+ movl %r9d,%r12d+ movq -64(%rbp),%rdi+ addl %r14d,%eax+ movl -56(%rbp),%r12d++ addl 0(%rdi),%eax+ addl 4(%rdi),%ebx+ addl 8(%rdi),%ecx+ addl 12(%rdi),%edx+ addl 16(%rdi),%r8d+ addl 20(%rdi),%r9d+ addl 24(%rdi),%r10d+ addl 28(%rdi),%r11d++ movl %eax,0(%rdi)+ movl %ebx,4(%rdi)+ movl %ecx,8(%rdi)+ movl %edx,12(%rdi)+ movl %r8d,16(%rdi)+ movl %r9d,20(%rdi)+ movl %r10d,24(%rdi)+ movl %r11d,28(%rdi)++ cmpl -48(%rbp),%r12d+ je .Ldone_avx2++ leaq 448(%rsp),%rsi+ xorl %r14d,%r14d+ movl %ebx,%edi+ xorl %ecx,%edi+ movl %r9d,%r12d+ jmp .Lower_avx2+.align 16+.Lower_avx2:+ addl 0+16(%rsi),%r11d+ andl %r8d,%r12d+ rorxl $25,%r8d,%r13d+ rorxl $11,%r8d,%r15d+ leal (%rax,%r14,1),%eax+ leal (%r11,%r12,1),%r11d+ andnl %r10d,%r8d,%r12d+ xorl %r15d,%r13d+ rorxl $6,%r8d,%r14d+ leal (%r11,%r12,1),%r11d+ xorl %r14d,%r13d+ movl %eax,%r15d+ rorxl $22,%eax,%r12d+ leal (%r11,%r13,1),%r11d+ xorl %ebx,%r15d+ rorxl $13,%eax,%r14d+ rorxl $2,%eax,%r13d+ leal (%rdx,%r11,1),%edx+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %ebx,%edi+ xorl %r13d,%r14d+ leal (%r11,%rdi,1),%r11d+ movl %r8d,%r12d+ addl 4+16(%rsi),%r10d+ andl %edx,%r12d+ rorxl $25,%edx,%r13d+ rorxl $11,%edx,%edi+ leal (%r11,%r14,1),%r11d+ leal (%r10,%r12,1),%r10d+ andnl %r9d,%edx,%r12d+ xorl %edi,%r13d+ rorxl $6,%edx,%r14d+ leal (%r10,%r12,1),%r10d+ xorl %r14d,%r13d+ movl %r11d,%edi+ rorxl $22,%r11d,%r12d+ leal (%r10,%r13,1),%r10d+ xorl %eax,%edi+ rorxl $13,%r11d,%r14d+ rorxl $2,%r11d,%r13d+ leal (%rcx,%r10,1),%ecx+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %eax,%r15d+ xorl %r13d,%r14d+ leal (%r10,%r15,1),%r10d+ movl %edx,%r12d+ addl 8+16(%rsi),%r9d+ andl %ecx,%r12d+ rorxl $25,%ecx,%r13d+ rorxl $11,%ecx,%r15d+ leal (%r10,%r14,1),%r10d+ leal (%r9,%r12,1),%r9d+ andnl %r8d,%ecx,%r12d+ xorl %r15d,%r13d+ rorxl $6,%ecx,%r14d+ leal (%r9,%r12,1),%r9d+ xorl %r14d,%r13d+ movl %r10d,%r15d+ rorxl $22,%r10d,%r12d+ leal (%r9,%r13,1),%r9d+ xorl %r11d,%r15d+ rorxl $13,%r10d,%r14d+ rorxl $2,%r10d,%r13d+ leal (%rbx,%r9,1),%ebx+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %r11d,%edi+ xorl %r13d,%r14d+ leal (%r9,%rdi,1),%r9d+ movl %ecx,%r12d+ addl 12+16(%rsi),%r8d+ andl %ebx,%r12d+ rorxl $25,%ebx,%r13d+ rorxl $11,%ebx,%edi+ leal (%r9,%r14,1),%r9d+ leal (%r8,%r12,1),%r8d+ andnl %edx,%ebx,%r12d+ xorl %edi,%r13d+ rorxl $6,%ebx,%r14d+ leal (%r8,%r12,1),%r8d+ xorl %r14d,%r13d+ movl %r9d,%edi+ rorxl $22,%r9d,%r12d+ leal (%r8,%r13,1),%r8d+ xorl %r10d,%edi+ rorxl $13,%r9d,%r14d+ rorxl $2,%r9d,%r13d+ leal (%rax,%r8,1),%eax+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %r10d,%r15d+ xorl %r13d,%r14d+ leal (%r8,%r15,1),%r8d+ movl %ebx,%r12d+ addl 32+16(%rsi),%edx+ andl %eax,%r12d+ rorxl $25,%eax,%r13d+ rorxl $11,%eax,%r15d+ leal (%r8,%r14,1),%r8d+ leal (%rdx,%r12,1),%edx+ andnl %ecx,%eax,%r12d+ xorl %r15d,%r13d+ rorxl $6,%eax,%r14d+ leal (%rdx,%r12,1),%edx+ xorl %r14d,%r13d+ movl %r8d,%r15d+ rorxl $22,%r8d,%r12d+ leal (%rdx,%r13,1),%edx+ xorl %r9d,%r15d+ rorxl $13,%r8d,%r14d+ rorxl $2,%r8d,%r13d+ leal (%r11,%rdx,1),%r11d+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %r9d,%edi+ xorl %r13d,%r14d+ leal (%rdx,%rdi,1),%edx+ movl %eax,%r12d+ addl 36+16(%rsi),%ecx+ andl %r11d,%r12d+ rorxl $25,%r11d,%r13d+ rorxl $11,%r11d,%edi+ leal (%rdx,%r14,1),%edx+ leal (%rcx,%r12,1),%ecx+ andnl %ebx,%r11d,%r12d+ xorl %edi,%r13d+ rorxl $6,%r11d,%r14d+ leal (%rcx,%r12,1),%ecx+ xorl %r14d,%r13d+ movl %edx,%edi+ rorxl $22,%edx,%r12d+ leal (%rcx,%r13,1),%ecx+ xorl %r8d,%edi+ rorxl $13,%edx,%r14d+ rorxl $2,%edx,%r13d+ leal (%r10,%rcx,1),%r10d+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %r8d,%r15d+ xorl %r13d,%r14d+ leal (%rcx,%r15,1),%ecx+ movl %r11d,%r12d+ addl 40+16(%rsi),%ebx+ andl %r10d,%r12d+ rorxl $25,%r10d,%r13d+ rorxl $11,%r10d,%r15d+ leal (%rcx,%r14,1),%ecx+ leal (%rbx,%r12,1),%ebx+ andnl %eax,%r10d,%r12d+ xorl %r15d,%r13d+ rorxl $6,%r10d,%r14d+ leal (%rbx,%r12,1),%ebx+ xorl %r14d,%r13d+ movl %ecx,%r15d+ rorxl $22,%ecx,%r12d+ leal (%rbx,%r13,1),%ebx+ xorl %edx,%r15d+ rorxl $13,%ecx,%r14d+ rorxl $2,%ecx,%r13d+ leal (%r9,%rbx,1),%r9d+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %edx,%edi+ xorl %r13d,%r14d+ leal (%rbx,%rdi,1),%ebx+ movl %r10d,%r12d+ addl 44+16(%rsi),%eax+ andl %r9d,%r12d+ rorxl $25,%r9d,%r13d+ rorxl $11,%r9d,%edi+ leal (%rbx,%r14,1),%ebx+ leal (%rax,%r12,1),%eax+ andnl %r11d,%r9d,%r12d+ xorl %edi,%r13d+ rorxl $6,%r9d,%r14d+ leal (%rax,%r12,1),%eax+ xorl %r14d,%r13d+ movl %ebx,%edi+ rorxl $22,%ebx,%r12d+ leal (%rax,%r13,1),%eax+ xorl %ecx,%edi+ rorxl $13,%ebx,%r14d+ rorxl $2,%ebx,%r13d+ leal (%r8,%rax,1),%r8d+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %ecx,%r15d+ xorl %r13d,%r14d+ leal (%rax,%r15,1),%eax+ movl %r9d,%r12d+ leaq -64(%rsi),%rsi+ cmpq %rsp,%rsi+ jae .Lower_avx2++ movq -64(%rbp),%rdi+ addl %r14d,%eax+ movq -56(%rbp),%rsi+ leaq 448(%rsp),%rsp++ addl 0(%rdi),%eax+ addl 4(%rdi),%ebx+ addl 8(%rdi),%ecx+ addl 12(%rdi),%edx+ addl 16(%rdi),%r8d+ addl 20(%rdi),%r9d+ leaq 128(%rsi),%rsi+ addl 24(%rdi),%r10d+ movq %rsi,%r12+ addl 28(%rdi),%r11d+ cmpq -48(%rbp),%rsi++ movl %eax,0(%rdi)+ cmoveq %rsp,%r12+ movl %ebx,4(%rdi)+ movl %ecx,8(%rdi)+ movl %edx,12(%rdi)+ movl %r8d,16(%rdi)+ movl %r9d,20(%rdi)+ movl %r10d,24(%rdi)+ movl %r11d,28(%rdi)++ jbe .Loop_avx2++.Ldone_avx2:+ vzeroupper+ movq -40(%rbp),%r15+ movq -32(%rbp),%r14+ movq -24(%rbp),%r13+ movq -16(%rbp),%r12+ movq -8(%rbp),%rbx+ movq %rbp,%rsp+.cfi_def_cfa_register %rsp+ popq %rbp+.cfi_adjust_cfa_offset -8+.cfi_restore %rbp+.cfi_restore %r12+.cfi_restore %r13+.cfi_restore %r14+.cfi_restore %r15+.cfi_restore %rbx+ .byte 0xf3,0xc3+.cfi_endproc +.size crypton_sha256_asm_block_data_order_avx2,.-crypton_sha256_asm_block_data_order_avx2++.section .note.gnu.property,"a",@note+ .long 4,2f-1f,5+ .byte 0x47,0x4E,0x55,0+1: .long 0xc0000002,4,3+.align 8+2:++.section .note.GNU-stack,"",@progbits
@@ -0,0 +1,5454 @@+.text +++.globl _crypton_sha256_asm_block_data_order++.p2align 4+_crypton_sha256_asm_block_data_order:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa+++ pushq %rbp+.cfi_adjust_cfa_offset 8+.cfi_offset %rbp,-16+ movq %rsp,%rbp+.cfi_def_cfa_register %rbp+ leaq _crypton_ia32cap_P(%rip),%rax+ movl 0(%rax),%r9d+ movl 4(%rax),%r10d+ movl 8(%rax),%eax+ testl $536870912,%eax+ jnz L$shaext_shortcut+ andl $296,%eax+ cmpl $296,%eax+ je L$avx2_shortcut+ andl $1073741824,%r9d+ andl $268435968,%r10d+ orl %r9d,%r10d+ cmpl $1342177792,%r10d+ je L$avx_shortcut+ testl $512,%r10d+ jnz L$ssse3_shortcut+ pushq %rbx+.cfi_offset %rbx,-24+ pushq %r12+.cfi_offset %r12,-32+ pushq %r13+.cfi_offset %r13,-40+ pushq %r14+.cfi_offset %r14,-48+ pushq %r15+.cfi_offset %r15,-56+ shlq $4,%rdx+ subq $64+24,%rsp++.cfi_def_cfa %rsp,144++ leaq (%rsi,%rdx,4),%rdx+ movq %rdi,64+0(%rsp)+ movq %rsi,64+8(%rsp)+ movq %rdx,64+16(%rsp)++ movl 0(%rdi),%eax+ movl 4(%rdi),%ebx+ movl 8(%rdi),%ecx+ movl 12(%rdi),%edx+ movl 16(%rdi),%r8d+ movl 20(%rdi),%r9d+ movl 24(%rdi),%r10d+ movl 28(%rdi),%r11d+ jmp L$loop++.p2align 4+L$loop:+ movl %ebx,%edi+ leaq K256(%rip),%rbp+ xorl %ecx,%edi+ movl 0(%rsi),%r12d+ movl %r8d,%r13d+ movl %eax,%r14d+ bswapl %r12d+ rorl $14,%r13d+ movl %r9d,%r15d++ xorl %r8d,%r13d+ rorl $9,%r14d+ xorl %r10d,%r15d++ movl %r12d,0(%rsp)+ xorl %eax,%r14d+ andl %r8d,%r15d++ rorl $5,%r13d+ addl %r11d,%r12d+ xorl %r10d,%r15d++ rorl $11,%r14d+ xorl %r8d,%r13d+ addl %r15d,%r12d++ movl %eax,%r15d+ addl (%rbp),%r12d+ xorl %eax,%r14d++ xorl %ebx,%r15d+ rorl $6,%r13d+ movl %ebx,%r11d++ andl %r15d,%edi+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %edi,%r11d+ addl %r12d,%edx+ addl %r12d,%r11d++ leaq 4(%rbp),%rbp+ addl %r14d,%r11d+ movl 4(%rsi),%r12d+ movl %edx,%r13d+ movl %r11d,%r14d+ bswapl %r12d+ rorl $14,%r13d+ movl %r8d,%edi++ xorl %edx,%r13d+ rorl $9,%r14d+ xorl %r9d,%edi++ movl %r12d,4(%rsp)+ xorl %r11d,%r14d+ andl %edx,%edi++ rorl $5,%r13d+ addl %r10d,%r12d+ xorl %r9d,%edi++ rorl $11,%r14d+ xorl %edx,%r13d+ addl %edi,%r12d++ movl %r11d,%edi+ addl (%rbp),%r12d+ xorl %r11d,%r14d++ xorl %eax,%edi+ rorl $6,%r13d+ movl %eax,%r10d++ andl %edi,%r15d+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %r15d,%r10d+ addl %r12d,%ecx+ addl %r12d,%r10d++ leaq 4(%rbp),%rbp+ addl %r14d,%r10d+ movl 8(%rsi),%r12d+ movl %ecx,%r13d+ movl %r10d,%r14d+ bswapl %r12d+ rorl $14,%r13d+ movl %edx,%r15d++ xorl %ecx,%r13d+ rorl $9,%r14d+ xorl %r8d,%r15d++ movl %r12d,8(%rsp)+ xorl %r10d,%r14d+ andl %ecx,%r15d++ rorl $5,%r13d+ addl %r9d,%r12d+ xorl %r8d,%r15d++ rorl $11,%r14d+ xorl %ecx,%r13d+ addl %r15d,%r12d++ movl %r10d,%r15d+ addl (%rbp),%r12d+ xorl %r10d,%r14d++ xorl %r11d,%r15d+ rorl $6,%r13d+ movl %r11d,%r9d++ andl %r15d,%edi+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %edi,%r9d+ addl %r12d,%ebx+ addl %r12d,%r9d++ leaq 4(%rbp),%rbp+ addl %r14d,%r9d+ movl 12(%rsi),%r12d+ movl %ebx,%r13d+ movl %r9d,%r14d+ bswapl %r12d+ rorl $14,%r13d+ movl %ecx,%edi++ xorl %ebx,%r13d+ rorl $9,%r14d+ xorl %edx,%edi++ movl %r12d,12(%rsp)+ xorl %r9d,%r14d+ andl %ebx,%edi++ rorl $5,%r13d+ addl %r8d,%r12d+ xorl %edx,%edi++ rorl $11,%r14d+ xorl %ebx,%r13d+ addl %edi,%r12d++ movl %r9d,%edi+ addl (%rbp),%r12d+ xorl %r9d,%r14d++ xorl %r10d,%edi+ rorl $6,%r13d+ movl %r10d,%r8d++ andl %edi,%r15d+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %r15d,%r8d+ addl %r12d,%eax+ addl %r12d,%r8d++ leaq 20(%rbp),%rbp+ addl %r14d,%r8d+ movl 16(%rsi),%r12d+ movl %eax,%r13d+ movl %r8d,%r14d+ bswapl %r12d+ rorl $14,%r13d+ movl %ebx,%r15d++ xorl %eax,%r13d+ rorl $9,%r14d+ xorl %ecx,%r15d++ movl %r12d,16(%rsp)+ xorl %r8d,%r14d+ andl %eax,%r15d++ rorl $5,%r13d+ addl %edx,%r12d+ xorl %ecx,%r15d++ rorl $11,%r14d+ xorl %eax,%r13d+ addl %r15d,%r12d++ movl %r8d,%r15d+ addl (%rbp),%r12d+ xorl %r8d,%r14d++ xorl %r9d,%r15d+ rorl $6,%r13d+ movl %r9d,%edx++ andl %r15d,%edi+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %edi,%edx+ addl %r12d,%r11d+ addl %r12d,%edx++ leaq 4(%rbp),%rbp+ addl %r14d,%edx+ movl 20(%rsi),%r12d+ movl %r11d,%r13d+ movl %edx,%r14d+ bswapl %r12d+ rorl $14,%r13d+ movl %eax,%edi++ xorl %r11d,%r13d+ rorl $9,%r14d+ xorl %ebx,%edi++ movl %r12d,20(%rsp)+ xorl %edx,%r14d+ andl %r11d,%edi++ rorl $5,%r13d+ addl %ecx,%r12d+ xorl %ebx,%edi++ rorl $11,%r14d+ xorl %r11d,%r13d+ addl %edi,%r12d++ movl %edx,%edi+ addl (%rbp),%r12d+ xorl %edx,%r14d++ xorl %r8d,%edi+ rorl $6,%r13d+ movl %r8d,%ecx++ andl %edi,%r15d+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %r15d,%ecx+ addl %r12d,%r10d+ addl %r12d,%ecx++ leaq 4(%rbp),%rbp+ addl %r14d,%ecx+ movl 24(%rsi),%r12d+ movl %r10d,%r13d+ movl %ecx,%r14d+ bswapl %r12d+ rorl $14,%r13d+ movl %r11d,%r15d++ xorl %r10d,%r13d+ rorl $9,%r14d+ xorl %eax,%r15d++ movl %r12d,24(%rsp)+ xorl %ecx,%r14d+ andl %r10d,%r15d++ rorl $5,%r13d+ addl %ebx,%r12d+ xorl %eax,%r15d++ rorl $11,%r14d+ xorl %r10d,%r13d+ addl %r15d,%r12d++ movl %ecx,%r15d+ addl (%rbp),%r12d+ xorl %ecx,%r14d++ xorl %edx,%r15d+ rorl $6,%r13d+ movl %edx,%ebx++ andl %r15d,%edi+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %edi,%ebx+ addl %r12d,%r9d+ addl %r12d,%ebx++ leaq 4(%rbp),%rbp+ addl %r14d,%ebx+ movl 28(%rsi),%r12d+ movl %r9d,%r13d+ movl %ebx,%r14d+ bswapl %r12d+ rorl $14,%r13d+ movl %r10d,%edi++ xorl %r9d,%r13d+ rorl $9,%r14d+ xorl %r11d,%edi++ movl %r12d,28(%rsp)+ xorl %ebx,%r14d+ andl %r9d,%edi++ rorl $5,%r13d+ addl %eax,%r12d+ xorl %r11d,%edi++ rorl $11,%r14d+ xorl %r9d,%r13d+ addl %edi,%r12d++ movl %ebx,%edi+ addl (%rbp),%r12d+ xorl %ebx,%r14d++ xorl %ecx,%edi+ rorl $6,%r13d+ movl %ecx,%eax++ andl %edi,%r15d+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %r15d,%eax+ addl %r12d,%r8d+ addl %r12d,%eax++ leaq 20(%rbp),%rbp+ addl %r14d,%eax+ movl 32(%rsi),%r12d+ movl %r8d,%r13d+ movl %eax,%r14d+ bswapl %r12d+ rorl $14,%r13d+ movl %r9d,%r15d++ xorl %r8d,%r13d+ rorl $9,%r14d+ xorl %r10d,%r15d++ movl %r12d,32(%rsp)+ xorl %eax,%r14d+ andl %r8d,%r15d++ rorl $5,%r13d+ addl %r11d,%r12d+ xorl %r10d,%r15d++ rorl $11,%r14d+ xorl %r8d,%r13d+ addl %r15d,%r12d++ movl %eax,%r15d+ addl (%rbp),%r12d+ xorl %eax,%r14d++ xorl %ebx,%r15d+ rorl $6,%r13d+ movl %ebx,%r11d++ andl %r15d,%edi+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %edi,%r11d+ addl %r12d,%edx+ addl %r12d,%r11d++ leaq 4(%rbp),%rbp+ addl %r14d,%r11d+ movl 36(%rsi),%r12d+ movl %edx,%r13d+ movl %r11d,%r14d+ bswapl %r12d+ rorl $14,%r13d+ movl %r8d,%edi++ xorl %edx,%r13d+ rorl $9,%r14d+ xorl %r9d,%edi++ movl %r12d,36(%rsp)+ xorl %r11d,%r14d+ andl %edx,%edi++ rorl $5,%r13d+ addl %r10d,%r12d+ xorl %r9d,%edi++ rorl $11,%r14d+ xorl %edx,%r13d+ addl %edi,%r12d++ movl %r11d,%edi+ addl (%rbp),%r12d+ xorl %r11d,%r14d++ xorl %eax,%edi+ rorl $6,%r13d+ movl %eax,%r10d++ andl %edi,%r15d+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %r15d,%r10d+ addl %r12d,%ecx+ addl %r12d,%r10d++ leaq 4(%rbp),%rbp+ addl %r14d,%r10d+ movl 40(%rsi),%r12d+ movl %ecx,%r13d+ movl %r10d,%r14d+ bswapl %r12d+ rorl $14,%r13d+ movl %edx,%r15d++ xorl %ecx,%r13d+ rorl $9,%r14d+ xorl %r8d,%r15d++ movl %r12d,40(%rsp)+ xorl %r10d,%r14d+ andl %ecx,%r15d++ rorl $5,%r13d+ addl %r9d,%r12d+ xorl %r8d,%r15d++ rorl $11,%r14d+ xorl %ecx,%r13d+ addl %r15d,%r12d++ movl %r10d,%r15d+ addl (%rbp),%r12d+ xorl %r10d,%r14d++ xorl %r11d,%r15d+ rorl $6,%r13d+ movl %r11d,%r9d++ andl %r15d,%edi+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %edi,%r9d+ addl %r12d,%ebx+ addl %r12d,%r9d++ leaq 4(%rbp),%rbp+ addl %r14d,%r9d+ movl 44(%rsi),%r12d+ movl %ebx,%r13d+ movl %r9d,%r14d+ bswapl %r12d+ rorl $14,%r13d+ movl %ecx,%edi++ xorl %ebx,%r13d+ rorl $9,%r14d+ xorl %edx,%edi++ movl %r12d,44(%rsp)+ xorl %r9d,%r14d+ andl %ebx,%edi++ rorl $5,%r13d+ addl %r8d,%r12d+ xorl %edx,%edi++ rorl $11,%r14d+ xorl %ebx,%r13d+ addl %edi,%r12d++ movl %r9d,%edi+ addl (%rbp),%r12d+ xorl %r9d,%r14d++ xorl %r10d,%edi+ rorl $6,%r13d+ movl %r10d,%r8d++ andl %edi,%r15d+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %r15d,%r8d+ addl %r12d,%eax+ addl %r12d,%r8d++ leaq 20(%rbp),%rbp+ addl %r14d,%r8d+ movl 48(%rsi),%r12d+ movl %eax,%r13d+ movl %r8d,%r14d+ bswapl %r12d+ rorl $14,%r13d+ movl %ebx,%r15d++ xorl %eax,%r13d+ rorl $9,%r14d+ xorl %ecx,%r15d++ movl %r12d,48(%rsp)+ xorl %r8d,%r14d+ andl %eax,%r15d++ rorl $5,%r13d+ addl %edx,%r12d+ xorl %ecx,%r15d++ rorl $11,%r14d+ xorl %eax,%r13d+ addl %r15d,%r12d++ movl %r8d,%r15d+ addl (%rbp),%r12d+ xorl %r8d,%r14d++ xorl %r9d,%r15d+ rorl $6,%r13d+ movl %r9d,%edx++ andl %r15d,%edi+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %edi,%edx+ addl %r12d,%r11d+ addl %r12d,%edx++ leaq 4(%rbp),%rbp+ addl %r14d,%edx+ movl 52(%rsi),%r12d+ movl %r11d,%r13d+ movl %edx,%r14d+ bswapl %r12d+ rorl $14,%r13d+ movl %eax,%edi++ xorl %r11d,%r13d+ rorl $9,%r14d+ xorl %ebx,%edi++ movl %r12d,52(%rsp)+ xorl %edx,%r14d+ andl %r11d,%edi++ rorl $5,%r13d+ addl %ecx,%r12d+ xorl %ebx,%edi++ rorl $11,%r14d+ xorl %r11d,%r13d+ addl %edi,%r12d++ movl %edx,%edi+ addl (%rbp),%r12d+ xorl %edx,%r14d++ xorl %r8d,%edi+ rorl $6,%r13d+ movl %r8d,%ecx++ andl %edi,%r15d+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %r15d,%ecx+ addl %r12d,%r10d+ addl %r12d,%ecx++ leaq 4(%rbp),%rbp+ addl %r14d,%ecx+ movl 56(%rsi),%r12d+ movl %r10d,%r13d+ movl %ecx,%r14d+ bswapl %r12d+ rorl $14,%r13d+ movl %r11d,%r15d++ xorl %r10d,%r13d+ rorl $9,%r14d+ xorl %eax,%r15d++ movl %r12d,56(%rsp)+ xorl %ecx,%r14d+ andl %r10d,%r15d++ rorl $5,%r13d+ addl %ebx,%r12d+ xorl %eax,%r15d++ rorl $11,%r14d+ xorl %r10d,%r13d+ addl %r15d,%r12d++ movl %ecx,%r15d+ addl (%rbp),%r12d+ xorl %ecx,%r14d++ xorl %edx,%r15d+ rorl $6,%r13d+ movl %edx,%ebx++ andl %r15d,%edi+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %edi,%ebx+ addl %r12d,%r9d+ addl %r12d,%ebx++ leaq 4(%rbp),%rbp+ addl %r14d,%ebx+ movl 60(%rsi),%r12d+ movl %r9d,%r13d+ movl %ebx,%r14d+ bswapl %r12d+ rorl $14,%r13d+ movl %r10d,%edi++ xorl %r9d,%r13d+ rorl $9,%r14d+ xorl %r11d,%edi++ movl %r12d,60(%rsp)+ xorl %ebx,%r14d+ andl %r9d,%edi++ rorl $5,%r13d+ addl %eax,%r12d+ xorl %r11d,%edi++ rorl $11,%r14d+ xorl %r9d,%r13d+ addl %edi,%r12d++ movl %ebx,%edi+ addl (%rbp),%r12d+ xorl %ebx,%r14d++ xorl %ecx,%edi+ rorl $6,%r13d+ movl %ecx,%eax++ andl %edi,%r15d+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %r15d,%eax+ addl %r12d,%r8d+ addl %r12d,%eax++ leaq 20(%rbp),%rbp+ jmp L$rounds_16_xx+.p2align 4+L$rounds_16_xx:+ movl 4(%rsp),%r13d+ movl 56(%rsp),%r15d++ movl %r13d,%r12d+ rorl $11,%r13d+ addl %r14d,%eax+ movl %r15d,%r14d+ rorl $2,%r15d++ xorl %r12d,%r13d+ shrl $3,%r12d+ rorl $7,%r13d+ xorl %r14d,%r15d+ shrl $10,%r14d++ rorl $17,%r15d+ xorl %r13d,%r12d+ xorl %r14d,%r15d+ addl 36(%rsp),%r12d++ addl 0(%rsp),%r12d+ movl %r8d,%r13d+ addl %r15d,%r12d+ movl %eax,%r14d+ rorl $14,%r13d+ movl %r9d,%r15d++ xorl %r8d,%r13d+ rorl $9,%r14d+ xorl %r10d,%r15d++ movl %r12d,0(%rsp)+ xorl %eax,%r14d+ andl %r8d,%r15d++ rorl $5,%r13d+ addl %r11d,%r12d+ xorl %r10d,%r15d++ rorl $11,%r14d+ xorl %r8d,%r13d+ addl %r15d,%r12d++ movl %eax,%r15d+ addl (%rbp),%r12d+ xorl %eax,%r14d++ xorl %ebx,%r15d+ rorl $6,%r13d+ movl %ebx,%r11d++ andl %r15d,%edi+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %edi,%r11d+ addl %r12d,%edx+ addl %r12d,%r11d++ leaq 4(%rbp),%rbp+ movl 8(%rsp),%r13d+ movl 60(%rsp),%edi++ movl %r13d,%r12d+ rorl $11,%r13d+ addl %r14d,%r11d+ movl %edi,%r14d+ rorl $2,%edi++ xorl %r12d,%r13d+ shrl $3,%r12d+ rorl $7,%r13d+ xorl %r14d,%edi+ shrl $10,%r14d++ rorl $17,%edi+ xorl %r13d,%r12d+ xorl %r14d,%edi+ addl 40(%rsp),%r12d++ addl 4(%rsp),%r12d+ movl %edx,%r13d+ addl %edi,%r12d+ movl %r11d,%r14d+ rorl $14,%r13d+ movl %r8d,%edi++ xorl %edx,%r13d+ rorl $9,%r14d+ xorl %r9d,%edi++ movl %r12d,4(%rsp)+ xorl %r11d,%r14d+ andl %edx,%edi++ rorl $5,%r13d+ addl %r10d,%r12d+ xorl %r9d,%edi++ rorl $11,%r14d+ xorl %edx,%r13d+ addl %edi,%r12d++ movl %r11d,%edi+ addl (%rbp),%r12d+ xorl %r11d,%r14d++ xorl %eax,%edi+ rorl $6,%r13d+ movl %eax,%r10d++ andl %edi,%r15d+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %r15d,%r10d+ addl %r12d,%ecx+ addl %r12d,%r10d++ leaq 4(%rbp),%rbp+ movl 12(%rsp),%r13d+ movl 0(%rsp),%r15d++ movl %r13d,%r12d+ rorl $11,%r13d+ addl %r14d,%r10d+ movl %r15d,%r14d+ rorl $2,%r15d++ xorl %r12d,%r13d+ shrl $3,%r12d+ rorl $7,%r13d+ xorl %r14d,%r15d+ shrl $10,%r14d++ rorl $17,%r15d+ xorl %r13d,%r12d+ xorl %r14d,%r15d+ addl 44(%rsp),%r12d++ addl 8(%rsp),%r12d+ movl %ecx,%r13d+ addl %r15d,%r12d+ movl %r10d,%r14d+ rorl $14,%r13d+ movl %edx,%r15d++ xorl %ecx,%r13d+ rorl $9,%r14d+ xorl %r8d,%r15d++ movl %r12d,8(%rsp)+ xorl %r10d,%r14d+ andl %ecx,%r15d++ rorl $5,%r13d+ addl %r9d,%r12d+ xorl %r8d,%r15d++ rorl $11,%r14d+ xorl %ecx,%r13d+ addl %r15d,%r12d++ movl %r10d,%r15d+ addl (%rbp),%r12d+ xorl %r10d,%r14d++ xorl %r11d,%r15d+ rorl $6,%r13d+ movl %r11d,%r9d++ andl %r15d,%edi+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %edi,%r9d+ addl %r12d,%ebx+ addl %r12d,%r9d++ leaq 4(%rbp),%rbp+ movl 16(%rsp),%r13d+ movl 4(%rsp),%edi++ movl %r13d,%r12d+ rorl $11,%r13d+ addl %r14d,%r9d+ movl %edi,%r14d+ rorl $2,%edi++ xorl %r12d,%r13d+ shrl $3,%r12d+ rorl $7,%r13d+ xorl %r14d,%edi+ shrl $10,%r14d++ rorl $17,%edi+ xorl %r13d,%r12d+ xorl %r14d,%edi+ addl 48(%rsp),%r12d++ addl 12(%rsp),%r12d+ movl %ebx,%r13d+ addl %edi,%r12d+ movl %r9d,%r14d+ rorl $14,%r13d+ movl %ecx,%edi++ xorl %ebx,%r13d+ rorl $9,%r14d+ xorl %edx,%edi++ movl %r12d,12(%rsp)+ xorl %r9d,%r14d+ andl %ebx,%edi++ rorl $5,%r13d+ addl %r8d,%r12d+ xorl %edx,%edi++ rorl $11,%r14d+ xorl %ebx,%r13d+ addl %edi,%r12d++ movl %r9d,%edi+ addl (%rbp),%r12d+ xorl %r9d,%r14d++ xorl %r10d,%edi+ rorl $6,%r13d+ movl %r10d,%r8d++ andl %edi,%r15d+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %r15d,%r8d+ addl %r12d,%eax+ addl %r12d,%r8d++ leaq 20(%rbp),%rbp+ movl 20(%rsp),%r13d+ movl 8(%rsp),%r15d++ movl %r13d,%r12d+ rorl $11,%r13d+ addl %r14d,%r8d+ movl %r15d,%r14d+ rorl $2,%r15d++ xorl %r12d,%r13d+ shrl $3,%r12d+ rorl $7,%r13d+ xorl %r14d,%r15d+ shrl $10,%r14d++ rorl $17,%r15d+ xorl %r13d,%r12d+ xorl %r14d,%r15d+ addl 52(%rsp),%r12d++ addl 16(%rsp),%r12d+ movl %eax,%r13d+ addl %r15d,%r12d+ movl %r8d,%r14d+ rorl $14,%r13d+ movl %ebx,%r15d++ xorl %eax,%r13d+ rorl $9,%r14d+ xorl %ecx,%r15d++ movl %r12d,16(%rsp)+ xorl %r8d,%r14d+ andl %eax,%r15d++ rorl $5,%r13d+ addl %edx,%r12d+ xorl %ecx,%r15d++ rorl $11,%r14d+ xorl %eax,%r13d+ addl %r15d,%r12d++ movl %r8d,%r15d+ addl (%rbp),%r12d+ xorl %r8d,%r14d++ xorl %r9d,%r15d+ rorl $6,%r13d+ movl %r9d,%edx++ andl %r15d,%edi+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %edi,%edx+ addl %r12d,%r11d+ addl %r12d,%edx++ leaq 4(%rbp),%rbp+ movl 24(%rsp),%r13d+ movl 12(%rsp),%edi++ movl %r13d,%r12d+ rorl $11,%r13d+ addl %r14d,%edx+ movl %edi,%r14d+ rorl $2,%edi++ xorl %r12d,%r13d+ shrl $3,%r12d+ rorl $7,%r13d+ xorl %r14d,%edi+ shrl $10,%r14d++ rorl $17,%edi+ xorl %r13d,%r12d+ xorl %r14d,%edi+ addl 56(%rsp),%r12d++ addl 20(%rsp),%r12d+ movl %r11d,%r13d+ addl %edi,%r12d+ movl %edx,%r14d+ rorl $14,%r13d+ movl %eax,%edi++ xorl %r11d,%r13d+ rorl $9,%r14d+ xorl %ebx,%edi++ movl %r12d,20(%rsp)+ xorl %edx,%r14d+ andl %r11d,%edi++ rorl $5,%r13d+ addl %ecx,%r12d+ xorl %ebx,%edi++ rorl $11,%r14d+ xorl %r11d,%r13d+ addl %edi,%r12d++ movl %edx,%edi+ addl (%rbp),%r12d+ xorl %edx,%r14d++ xorl %r8d,%edi+ rorl $6,%r13d+ movl %r8d,%ecx++ andl %edi,%r15d+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %r15d,%ecx+ addl %r12d,%r10d+ addl %r12d,%ecx++ leaq 4(%rbp),%rbp+ movl 28(%rsp),%r13d+ movl 16(%rsp),%r15d++ movl %r13d,%r12d+ rorl $11,%r13d+ addl %r14d,%ecx+ movl %r15d,%r14d+ rorl $2,%r15d++ xorl %r12d,%r13d+ shrl $3,%r12d+ rorl $7,%r13d+ xorl %r14d,%r15d+ shrl $10,%r14d++ rorl $17,%r15d+ xorl %r13d,%r12d+ xorl %r14d,%r15d+ addl 60(%rsp),%r12d++ addl 24(%rsp),%r12d+ movl %r10d,%r13d+ addl %r15d,%r12d+ movl %ecx,%r14d+ rorl $14,%r13d+ movl %r11d,%r15d++ xorl %r10d,%r13d+ rorl $9,%r14d+ xorl %eax,%r15d++ movl %r12d,24(%rsp)+ xorl %ecx,%r14d+ andl %r10d,%r15d++ rorl $5,%r13d+ addl %ebx,%r12d+ xorl %eax,%r15d++ rorl $11,%r14d+ xorl %r10d,%r13d+ addl %r15d,%r12d++ movl %ecx,%r15d+ addl (%rbp),%r12d+ xorl %ecx,%r14d++ xorl %edx,%r15d+ rorl $6,%r13d+ movl %edx,%ebx++ andl %r15d,%edi+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %edi,%ebx+ addl %r12d,%r9d+ addl %r12d,%ebx++ leaq 4(%rbp),%rbp+ movl 32(%rsp),%r13d+ movl 20(%rsp),%edi++ movl %r13d,%r12d+ rorl $11,%r13d+ addl %r14d,%ebx+ movl %edi,%r14d+ rorl $2,%edi++ xorl %r12d,%r13d+ shrl $3,%r12d+ rorl $7,%r13d+ xorl %r14d,%edi+ shrl $10,%r14d++ rorl $17,%edi+ xorl %r13d,%r12d+ xorl %r14d,%edi+ addl 0(%rsp),%r12d++ addl 28(%rsp),%r12d+ movl %r9d,%r13d+ addl %edi,%r12d+ movl %ebx,%r14d+ rorl $14,%r13d+ movl %r10d,%edi++ xorl %r9d,%r13d+ rorl $9,%r14d+ xorl %r11d,%edi++ movl %r12d,28(%rsp)+ xorl %ebx,%r14d+ andl %r9d,%edi++ rorl $5,%r13d+ addl %eax,%r12d+ xorl %r11d,%edi++ rorl $11,%r14d+ xorl %r9d,%r13d+ addl %edi,%r12d++ movl %ebx,%edi+ addl (%rbp),%r12d+ xorl %ebx,%r14d++ xorl %ecx,%edi+ rorl $6,%r13d+ movl %ecx,%eax++ andl %edi,%r15d+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %r15d,%eax+ addl %r12d,%r8d+ addl %r12d,%eax++ leaq 20(%rbp),%rbp+ movl 36(%rsp),%r13d+ movl 24(%rsp),%r15d++ movl %r13d,%r12d+ rorl $11,%r13d+ addl %r14d,%eax+ movl %r15d,%r14d+ rorl $2,%r15d++ xorl %r12d,%r13d+ shrl $3,%r12d+ rorl $7,%r13d+ xorl %r14d,%r15d+ shrl $10,%r14d++ rorl $17,%r15d+ xorl %r13d,%r12d+ xorl %r14d,%r15d+ addl 4(%rsp),%r12d++ addl 32(%rsp),%r12d+ movl %r8d,%r13d+ addl %r15d,%r12d+ movl %eax,%r14d+ rorl $14,%r13d+ movl %r9d,%r15d++ xorl %r8d,%r13d+ rorl $9,%r14d+ xorl %r10d,%r15d++ movl %r12d,32(%rsp)+ xorl %eax,%r14d+ andl %r8d,%r15d++ rorl $5,%r13d+ addl %r11d,%r12d+ xorl %r10d,%r15d++ rorl $11,%r14d+ xorl %r8d,%r13d+ addl %r15d,%r12d++ movl %eax,%r15d+ addl (%rbp),%r12d+ xorl %eax,%r14d++ xorl %ebx,%r15d+ rorl $6,%r13d+ movl %ebx,%r11d++ andl %r15d,%edi+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %edi,%r11d+ addl %r12d,%edx+ addl %r12d,%r11d++ leaq 4(%rbp),%rbp+ movl 40(%rsp),%r13d+ movl 28(%rsp),%edi++ movl %r13d,%r12d+ rorl $11,%r13d+ addl %r14d,%r11d+ movl %edi,%r14d+ rorl $2,%edi++ xorl %r12d,%r13d+ shrl $3,%r12d+ rorl $7,%r13d+ xorl %r14d,%edi+ shrl $10,%r14d++ rorl $17,%edi+ xorl %r13d,%r12d+ xorl %r14d,%edi+ addl 8(%rsp),%r12d++ addl 36(%rsp),%r12d+ movl %edx,%r13d+ addl %edi,%r12d+ movl %r11d,%r14d+ rorl $14,%r13d+ movl %r8d,%edi++ xorl %edx,%r13d+ rorl $9,%r14d+ xorl %r9d,%edi++ movl %r12d,36(%rsp)+ xorl %r11d,%r14d+ andl %edx,%edi++ rorl $5,%r13d+ addl %r10d,%r12d+ xorl %r9d,%edi++ rorl $11,%r14d+ xorl %edx,%r13d+ addl %edi,%r12d++ movl %r11d,%edi+ addl (%rbp),%r12d+ xorl %r11d,%r14d++ xorl %eax,%edi+ rorl $6,%r13d+ movl %eax,%r10d++ andl %edi,%r15d+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %r15d,%r10d+ addl %r12d,%ecx+ addl %r12d,%r10d++ leaq 4(%rbp),%rbp+ movl 44(%rsp),%r13d+ movl 32(%rsp),%r15d++ movl %r13d,%r12d+ rorl $11,%r13d+ addl %r14d,%r10d+ movl %r15d,%r14d+ rorl $2,%r15d++ xorl %r12d,%r13d+ shrl $3,%r12d+ rorl $7,%r13d+ xorl %r14d,%r15d+ shrl $10,%r14d++ rorl $17,%r15d+ xorl %r13d,%r12d+ xorl %r14d,%r15d+ addl 12(%rsp),%r12d++ addl 40(%rsp),%r12d+ movl %ecx,%r13d+ addl %r15d,%r12d+ movl %r10d,%r14d+ rorl $14,%r13d+ movl %edx,%r15d++ xorl %ecx,%r13d+ rorl $9,%r14d+ xorl %r8d,%r15d++ movl %r12d,40(%rsp)+ xorl %r10d,%r14d+ andl %ecx,%r15d++ rorl $5,%r13d+ addl %r9d,%r12d+ xorl %r8d,%r15d++ rorl $11,%r14d+ xorl %ecx,%r13d+ addl %r15d,%r12d++ movl %r10d,%r15d+ addl (%rbp),%r12d+ xorl %r10d,%r14d++ xorl %r11d,%r15d+ rorl $6,%r13d+ movl %r11d,%r9d++ andl %r15d,%edi+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %edi,%r9d+ addl %r12d,%ebx+ addl %r12d,%r9d++ leaq 4(%rbp),%rbp+ movl 48(%rsp),%r13d+ movl 36(%rsp),%edi++ movl %r13d,%r12d+ rorl $11,%r13d+ addl %r14d,%r9d+ movl %edi,%r14d+ rorl $2,%edi++ xorl %r12d,%r13d+ shrl $3,%r12d+ rorl $7,%r13d+ xorl %r14d,%edi+ shrl $10,%r14d++ rorl $17,%edi+ xorl %r13d,%r12d+ xorl %r14d,%edi+ addl 16(%rsp),%r12d++ addl 44(%rsp),%r12d+ movl %ebx,%r13d+ addl %edi,%r12d+ movl %r9d,%r14d+ rorl $14,%r13d+ movl %ecx,%edi++ xorl %ebx,%r13d+ rorl $9,%r14d+ xorl %edx,%edi++ movl %r12d,44(%rsp)+ xorl %r9d,%r14d+ andl %ebx,%edi++ rorl $5,%r13d+ addl %r8d,%r12d+ xorl %edx,%edi++ rorl $11,%r14d+ xorl %ebx,%r13d+ addl %edi,%r12d++ movl %r9d,%edi+ addl (%rbp),%r12d+ xorl %r9d,%r14d++ xorl %r10d,%edi+ rorl $6,%r13d+ movl %r10d,%r8d++ andl %edi,%r15d+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %r15d,%r8d+ addl %r12d,%eax+ addl %r12d,%r8d++ leaq 20(%rbp),%rbp+ movl 52(%rsp),%r13d+ movl 40(%rsp),%r15d++ movl %r13d,%r12d+ rorl $11,%r13d+ addl %r14d,%r8d+ movl %r15d,%r14d+ rorl $2,%r15d++ xorl %r12d,%r13d+ shrl $3,%r12d+ rorl $7,%r13d+ xorl %r14d,%r15d+ shrl $10,%r14d++ rorl $17,%r15d+ xorl %r13d,%r12d+ xorl %r14d,%r15d+ addl 20(%rsp),%r12d++ addl 48(%rsp),%r12d+ movl %eax,%r13d+ addl %r15d,%r12d+ movl %r8d,%r14d+ rorl $14,%r13d+ movl %ebx,%r15d++ xorl %eax,%r13d+ rorl $9,%r14d+ xorl %ecx,%r15d++ movl %r12d,48(%rsp)+ xorl %r8d,%r14d+ andl %eax,%r15d++ rorl $5,%r13d+ addl %edx,%r12d+ xorl %ecx,%r15d++ rorl $11,%r14d+ xorl %eax,%r13d+ addl %r15d,%r12d++ movl %r8d,%r15d+ addl (%rbp),%r12d+ xorl %r8d,%r14d++ xorl %r9d,%r15d+ rorl $6,%r13d+ movl %r9d,%edx++ andl %r15d,%edi+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %edi,%edx+ addl %r12d,%r11d+ addl %r12d,%edx++ leaq 4(%rbp),%rbp+ movl 56(%rsp),%r13d+ movl 44(%rsp),%edi++ movl %r13d,%r12d+ rorl $11,%r13d+ addl %r14d,%edx+ movl %edi,%r14d+ rorl $2,%edi++ xorl %r12d,%r13d+ shrl $3,%r12d+ rorl $7,%r13d+ xorl %r14d,%edi+ shrl $10,%r14d++ rorl $17,%edi+ xorl %r13d,%r12d+ xorl %r14d,%edi+ addl 24(%rsp),%r12d++ addl 52(%rsp),%r12d+ movl %r11d,%r13d+ addl %edi,%r12d+ movl %edx,%r14d+ rorl $14,%r13d+ movl %eax,%edi++ xorl %r11d,%r13d+ rorl $9,%r14d+ xorl %ebx,%edi++ movl %r12d,52(%rsp)+ xorl %edx,%r14d+ andl %r11d,%edi++ rorl $5,%r13d+ addl %ecx,%r12d+ xorl %ebx,%edi++ rorl $11,%r14d+ xorl %r11d,%r13d+ addl %edi,%r12d++ movl %edx,%edi+ addl (%rbp),%r12d+ xorl %edx,%r14d++ xorl %r8d,%edi+ rorl $6,%r13d+ movl %r8d,%ecx++ andl %edi,%r15d+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %r15d,%ecx+ addl %r12d,%r10d+ addl %r12d,%ecx++ leaq 4(%rbp),%rbp+ movl 60(%rsp),%r13d+ movl 48(%rsp),%r15d++ movl %r13d,%r12d+ rorl $11,%r13d+ addl %r14d,%ecx+ movl %r15d,%r14d+ rorl $2,%r15d++ xorl %r12d,%r13d+ shrl $3,%r12d+ rorl $7,%r13d+ xorl %r14d,%r15d+ shrl $10,%r14d++ rorl $17,%r15d+ xorl %r13d,%r12d+ xorl %r14d,%r15d+ addl 28(%rsp),%r12d++ addl 56(%rsp),%r12d+ movl %r10d,%r13d+ addl %r15d,%r12d+ movl %ecx,%r14d+ rorl $14,%r13d+ movl %r11d,%r15d++ xorl %r10d,%r13d+ rorl $9,%r14d+ xorl %eax,%r15d++ movl %r12d,56(%rsp)+ xorl %ecx,%r14d+ andl %r10d,%r15d++ rorl $5,%r13d+ addl %ebx,%r12d+ xorl %eax,%r15d++ rorl $11,%r14d+ xorl %r10d,%r13d+ addl %r15d,%r12d++ movl %ecx,%r15d+ addl (%rbp),%r12d+ xorl %ecx,%r14d++ xorl %edx,%r15d+ rorl $6,%r13d+ movl %edx,%ebx++ andl %r15d,%edi+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %edi,%ebx+ addl %r12d,%r9d+ addl %r12d,%ebx++ leaq 4(%rbp),%rbp+ movl 0(%rsp),%r13d+ movl 52(%rsp),%edi++ movl %r13d,%r12d+ rorl $11,%r13d+ addl %r14d,%ebx+ movl %edi,%r14d+ rorl $2,%edi++ xorl %r12d,%r13d+ shrl $3,%r12d+ rorl $7,%r13d+ xorl %r14d,%edi+ shrl $10,%r14d++ rorl $17,%edi+ xorl %r13d,%r12d+ xorl %r14d,%edi+ addl 32(%rsp),%r12d++ addl 60(%rsp),%r12d+ movl %r9d,%r13d+ addl %edi,%r12d+ movl %ebx,%r14d+ rorl $14,%r13d+ movl %r10d,%edi++ xorl %r9d,%r13d+ rorl $9,%r14d+ xorl %r11d,%edi++ movl %r12d,60(%rsp)+ xorl %ebx,%r14d+ andl %r9d,%edi++ rorl $5,%r13d+ addl %eax,%r12d+ xorl %r11d,%edi++ rorl $11,%r14d+ xorl %r9d,%r13d+ addl %edi,%r12d++ movl %ebx,%edi+ addl (%rbp),%r12d+ xorl %ebx,%r14d++ xorl %ecx,%edi+ rorl $6,%r13d+ movl %ecx,%eax++ andl %edi,%r15d+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %r15d,%eax+ addl %r12d,%r8d+ addl %r12d,%eax++ leaq 20(%rbp),%rbp+ cmpb $0,3(%rbp)+ jnz L$rounds_16_xx++ movq 64+0(%rsp),%rdi+ addl %r14d,%eax+ leaq 64(%rsi),%rsi++ addl 0(%rdi),%eax+ addl 4(%rdi),%ebx+ addl 8(%rdi),%ecx+ addl 12(%rdi),%edx+ addl 16(%rdi),%r8d+ addl 20(%rdi),%r9d+ addl 24(%rdi),%r10d+ addl 28(%rdi),%r11d++ cmpq 64+16(%rsp),%rsi++ movl %eax,0(%rdi)+ movl %ebx,4(%rdi)+ movl %ecx,8(%rdi)+ movl %edx,12(%rdi)+ movl %r8d,16(%rdi)+ movl %r9d,20(%rdi)+ movl %r10d,24(%rdi)+ movl %r11d,28(%rdi)+ jb L$loop++ leaq 64+24+48(%rsp),%r11+.cfi_def_cfa %r11,8+ movq 64+24(%rsp),%r15+ movq -40(%r11),%r14+ movq -32(%r11),%r13+ movq -24(%r11),%r12+ movq -16(%r11),%rbx+ movq -8(%r11),%rbp+.cfi_restore %r12+.cfi_restore %r13+.cfi_restore %r14+.cfi_restore %r15+.cfi_restore %rbp+.cfi_restore %rbx+ leaq (%r11),%rsp+ .byte 0xf3,0xc3+.cfi_endproc ++.p2align 6++K256:+.long 0x428a2f98,0x71374491,0xb5c0fbcf,0xe9b5dba5+.long 0x428a2f98,0x71374491,0xb5c0fbcf,0xe9b5dba5+.long 0x3956c25b,0x59f111f1,0x923f82a4,0xab1c5ed5+.long 0x3956c25b,0x59f111f1,0x923f82a4,0xab1c5ed5+.long 0xd807aa98,0x12835b01,0x243185be,0x550c7dc3+.long 0xd807aa98,0x12835b01,0x243185be,0x550c7dc3+.long 0x72be5d74,0x80deb1fe,0x9bdc06a7,0xc19bf174+.long 0x72be5d74,0x80deb1fe,0x9bdc06a7,0xc19bf174+.long 0xe49b69c1,0xefbe4786,0x0fc19dc6,0x240ca1cc+.long 0xe49b69c1,0xefbe4786,0x0fc19dc6,0x240ca1cc+.long 0x2de92c6f,0x4a7484aa,0x5cb0a9dc,0x76f988da+.long 0x2de92c6f,0x4a7484aa,0x5cb0a9dc,0x76f988da+.long 0x983e5152,0xa831c66d,0xb00327c8,0xbf597fc7+.long 0x983e5152,0xa831c66d,0xb00327c8,0xbf597fc7+.long 0xc6e00bf3,0xd5a79147,0x06ca6351,0x14292967+.long 0xc6e00bf3,0xd5a79147,0x06ca6351,0x14292967+.long 0x27b70a85,0x2e1b2138,0x4d2c6dfc,0x53380d13+.long 0x27b70a85,0x2e1b2138,0x4d2c6dfc,0x53380d13+.long 0x650a7354,0x766a0abb,0x81c2c92e,0x92722c85+.long 0x650a7354,0x766a0abb,0x81c2c92e,0x92722c85+.long 0xa2bfe8a1,0xa81a664b,0xc24b8b70,0xc76c51a3+.long 0xa2bfe8a1,0xa81a664b,0xc24b8b70,0xc76c51a3+.long 0xd192e819,0xd6990624,0xf40e3585,0x106aa070+.long 0xd192e819,0xd6990624,0xf40e3585,0x106aa070+.long 0x19a4c116,0x1e376c08,0x2748774c,0x34b0bcb5+.long 0x19a4c116,0x1e376c08,0x2748774c,0x34b0bcb5+.long 0x391c0cb3,0x4ed8aa4a,0x5b9cca4f,0x682e6ff3+.long 0x391c0cb3,0x4ed8aa4a,0x5b9cca4f,0x682e6ff3+.long 0x748f82ee,0x78a5636f,0x84c87814,0x8cc70208+.long 0x748f82ee,0x78a5636f,0x84c87814,0x8cc70208+.long 0x90befffa,0xa4506ceb,0xbef9a3f7,0xc67178f2+.long 0x90befffa,0xa4506ceb,0xbef9a3f7,0xc67178f2++.long 0x00010203,0x04050607,0x08090a0b,0x0c0d0e0f+.long 0x00010203,0x04050607,0x08090a0b,0x0c0d0e0f+.long 0x03020100,0x0b0a0908,0xffffffff,0xffffffff+.long 0x03020100,0x0b0a0908,0xffffffff,0xffffffff+.long 0xffffffff,0xffffffff,0x03020100,0x0b0a0908+.long 0xffffffff,0xffffffff,0x03020100,0x0b0a0908+.byte 83,72,65,50,53,54,32,98,108,111,99,107,32,116,114,97,110,115,102,111,114,109,32,102,111,114,32,120,56,54,95,54,52,44,32,67,82,89,80,84,79,71,65,77,83,32,98,121,32,64,100,111,116,45,97,115,109,0++.p2align 6+crypton_sha256_asm_block_data_order_shaext:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa+++ pushq %rbp+.cfi_adjust_cfa_offset 8+.cfi_offset %rbp,-16+ movq %rsp,%rbp+.cfi_def_cfa_register %rbp+L$shaext_shortcut:++ leaq K256+128(%rip),%rcx+ movdqu (%rdi),%xmm1+ movdqu 16(%rdi),%xmm2+ movdqa 512-128(%rcx),%xmm7++ pshufd $0x1b,%xmm1,%xmm0+ pshufd $0xb1,%xmm1,%xmm1+ pshufd $0x1b,%xmm2,%xmm2+ movdqa %xmm7,%xmm8+.byte 102,15,58,15,202,8+ punpcklqdq %xmm0,%xmm2+ jmp L$oop_shaext++.p2align 4+L$oop_shaext:+ movdqu (%rsi),%xmm3+ movdqu 16(%rsi),%xmm4+ movdqu 32(%rsi),%xmm5+.byte 102,15,56,0,223+ movdqu 48(%rsi),%xmm6++ movdqa 0-128(%rcx),%xmm0+ paddd %xmm3,%xmm0+.byte 102,15,56,0,231+ movdqa %xmm2,%xmm10+.byte 15,56,203,209+ pshufd $0x0e,%xmm0,%xmm0+ nop+ movdqa %xmm1,%xmm9+.byte 15,56,203,202++ movdqa 32-128(%rcx),%xmm0+ paddd %xmm4,%xmm0+.byte 102,15,56,0,239+.byte 15,56,203,209+ pshufd $0x0e,%xmm0,%xmm0+ leaq 64(%rsi),%rsi+.byte 15,56,204,220+.byte 15,56,203,202++ movdqa 64-128(%rcx),%xmm0+ paddd %xmm5,%xmm0+.byte 102,15,56,0,247+.byte 15,56,203,209+ pshufd $0x0e,%xmm0,%xmm0+ movdqa %xmm6,%xmm7+.byte 102,15,58,15,253,4+ nop+ paddd %xmm7,%xmm3+.byte 15,56,204,229+.byte 15,56,203,202++ movdqa 96-128(%rcx),%xmm0+ paddd %xmm6,%xmm0+.byte 15,56,205,222+.byte 15,56,203,209+ pshufd $0x0e,%xmm0,%xmm0+ movdqa %xmm3,%xmm7+.byte 102,15,58,15,254,4+ nop+ paddd %xmm7,%xmm4+.byte 15,56,204,238+.byte 15,56,203,202+ movdqa 128-128(%rcx),%xmm0+ paddd %xmm3,%xmm0+.byte 15,56,205,227+.byte 15,56,203,209+ pshufd $0x0e,%xmm0,%xmm0+ movdqa %xmm4,%xmm7+.byte 102,15,58,15,251,4+ nop+ paddd %xmm7,%xmm5+.byte 15,56,204,243+.byte 15,56,203,202+ movdqa 160-128(%rcx),%xmm0+ paddd %xmm4,%xmm0+.byte 15,56,205,236+.byte 15,56,203,209+ pshufd $0x0e,%xmm0,%xmm0+ movdqa %xmm5,%xmm7+.byte 102,15,58,15,252,4+ nop+ paddd %xmm7,%xmm6+.byte 15,56,204,220+.byte 15,56,203,202+ movdqa 192-128(%rcx),%xmm0+ paddd %xmm5,%xmm0+.byte 15,56,205,245+.byte 15,56,203,209+ pshufd $0x0e,%xmm0,%xmm0+ movdqa %xmm6,%xmm7+.byte 102,15,58,15,253,4+ nop+ paddd %xmm7,%xmm3+.byte 15,56,204,229+.byte 15,56,203,202+ movdqa 224-128(%rcx),%xmm0+ paddd %xmm6,%xmm0+.byte 15,56,205,222+.byte 15,56,203,209+ pshufd $0x0e,%xmm0,%xmm0+ movdqa %xmm3,%xmm7+.byte 102,15,58,15,254,4+ nop+ paddd %xmm7,%xmm4+.byte 15,56,204,238+.byte 15,56,203,202+ movdqa 256-128(%rcx),%xmm0+ paddd %xmm3,%xmm0+.byte 15,56,205,227+.byte 15,56,203,209+ pshufd $0x0e,%xmm0,%xmm0+ movdqa %xmm4,%xmm7+.byte 102,15,58,15,251,4+ nop+ paddd %xmm7,%xmm5+.byte 15,56,204,243+.byte 15,56,203,202+ movdqa 288-128(%rcx),%xmm0+ paddd %xmm4,%xmm0+.byte 15,56,205,236+.byte 15,56,203,209+ pshufd $0x0e,%xmm0,%xmm0+ movdqa %xmm5,%xmm7+.byte 102,15,58,15,252,4+ nop+ paddd %xmm7,%xmm6+.byte 15,56,204,220+.byte 15,56,203,202+ movdqa 320-128(%rcx),%xmm0+ paddd %xmm5,%xmm0+.byte 15,56,205,245+.byte 15,56,203,209+ pshufd $0x0e,%xmm0,%xmm0+ movdqa %xmm6,%xmm7+.byte 102,15,58,15,253,4+ nop+ paddd %xmm7,%xmm3+.byte 15,56,204,229+.byte 15,56,203,202+ movdqa 352-128(%rcx),%xmm0+ paddd %xmm6,%xmm0+.byte 15,56,205,222+.byte 15,56,203,209+ pshufd $0x0e,%xmm0,%xmm0+ movdqa %xmm3,%xmm7+.byte 102,15,58,15,254,4+ nop+ paddd %xmm7,%xmm4+.byte 15,56,204,238+.byte 15,56,203,202+ movdqa 384-128(%rcx),%xmm0+ paddd %xmm3,%xmm0+.byte 15,56,205,227+.byte 15,56,203,209+ pshufd $0x0e,%xmm0,%xmm0+ movdqa %xmm4,%xmm7+.byte 102,15,58,15,251,4+ nop+ paddd %xmm7,%xmm5+.byte 15,56,204,243+.byte 15,56,203,202+ movdqa 416-128(%rcx),%xmm0+ paddd %xmm4,%xmm0+.byte 15,56,205,236+.byte 15,56,203,209+ pshufd $0x0e,%xmm0,%xmm0+ movdqa %xmm5,%xmm7+.byte 102,15,58,15,252,4+.byte 15,56,203,202+ paddd %xmm7,%xmm6++ movdqa 448-128(%rcx),%xmm0+ paddd %xmm5,%xmm0+.byte 15,56,203,209+ pshufd $0x0e,%xmm0,%xmm0+.byte 15,56,205,245+ movdqa %xmm8,%xmm7+.byte 15,56,203,202++ movdqa 480-128(%rcx),%xmm0+ paddd %xmm6,%xmm0+ nop+.byte 15,56,203,209+ pshufd $0x0e,%xmm0,%xmm0+ decq %rdx+ nop+.byte 15,56,203,202++ paddd %xmm10,%xmm2+ paddd %xmm9,%xmm1+ jnz L$oop_shaext++ pshufd $0xb1,%xmm2,%xmm2+ pshufd $0x1b,%xmm1,%xmm7+ pshufd $0xb1,%xmm1,%xmm1+ punpckhqdq %xmm2,%xmm1+.byte 102,15,58,15,215,8++ movdqu %xmm1,(%rdi)+ movdqu %xmm2,16(%rdi)+.cfi_def_cfa_register %rsp+ popq %rbp+.cfi_adjust_cfa_offset -8+.cfi_restore %rbp++ .byte 0xf3,0xc3+.cfi_endproc +++.p2align 6+crypton_sha256_asm_block_data_order_ssse3:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa+++ pushq %rbp+.cfi_adjust_cfa_offset 8+.cfi_offset %rbp,-16+ movq %rsp,%rbp+.cfi_def_cfa_register %rbp+L$ssse3_shortcut:+ pushq %rbx+.cfi_offset %rbx,-24+ pushq %r12+.cfi_offset %r12,-32+ pushq %r13+.cfi_offset %r13,-40+ pushq %r14+.cfi_offset %r14,-48+ pushq %r15+.cfi_offset %r15,-56+ shlq $4,%rdx+ subq $24,%rsp++ leaq (%rsi,%rdx,4),%rdx+ movq %rdi,-64(%rbp)++ movq %rdx,-48(%rbp)+++ leaq -64(%rsp),%rsp+ movl 0(%rdi),%eax+ andq $-64,%rsp+ movl 4(%rdi),%ebx+ movl 8(%rdi),%ecx+ movl 12(%rdi),%edx+ movl 16(%rdi),%r8d+ movl 20(%rdi),%r9d+ movl 24(%rdi),%r10d+ movl 28(%rdi),%r11d+++ jmp L$loop_ssse3+.p2align 4+L$loop_ssse3:+ movdqa K256+512(%rip),%xmm7+ movq %rsi,-56(%rbp)+ movdqu 0(%rsi),%xmm0+ movdqu 16(%rsi),%xmm1+ movdqu 32(%rsi),%xmm2+.byte 102,15,56,0,199+ movdqu 48(%rsi),%xmm3+ leaq K256(%rip),%rsi+.byte 102,15,56,0,207+ movdqa 0(%rsi),%xmm4+ movdqa 32(%rsi),%xmm5+.byte 102,15,56,0,215+ paddd %xmm0,%xmm4+ movdqa 64(%rsi),%xmm6+.byte 102,15,56,0,223+ movdqa 96(%rsi),%xmm7+ paddd %xmm1,%xmm5+ paddd %xmm2,%xmm6+ paddd %xmm3,%xmm7+ movdqa %xmm4,0(%rsp)+ movl %eax,%r14d+ movdqa %xmm5,16(%rsp)+ movl %ebx,%edi+ movdqa %xmm6,32(%rsp)+ xorl %ecx,%edi+ movdqa %xmm7,48(%rsp)+ movl %r8d,%r13d+ jmp L$ssse3_00_47++.p2align 4+L$ssse3_00_47:+ subq $-128,%rsi+ rorl $14,%r13d+ movdqa %xmm1,%xmm4+ movl %r14d,%eax+ movl %r9d,%r12d+ movdqa %xmm3,%xmm7+ rorl $9,%r14d+ xorl %r8d,%r13d+ xorl %r10d,%r12d+ rorl $5,%r13d+ xorl %eax,%r14d+.byte 102,15,58,15,224,4+ andl %r8d,%r12d+ xorl %r8d,%r13d+.byte 102,15,58,15,250,4+ addl 0(%rsp),%r11d+ movl %eax,%r15d+ xorl %r10d,%r12d+ rorl $11,%r14d+ movdqa %xmm4,%xmm5+ xorl %ebx,%r15d+ addl %r12d,%r11d+ movdqa %xmm4,%xmm6+ rorl $6,%r13d+ andl %r15d,%edi+ psrld $3,%xmm4+ xorl %eax,%r14d+ addl %r13d,%r11d+ xorl %ebx,%edi+ paddd %xmm7,%xmm0+ rorl $2,%r14d+ addl %r11d,%edx+ psrld $7,%xmm6+ addl %edi,%r11d+ movl %edx,%r13d+ pshufd $250,%xmm3,%xmm7+ addl %r11d,%r14d+ rorl $14,%r13d+ pslld $14,%xmm5+ movl %r14d,%r11d+ movl %r8d,%r12d+ pxor %xmm6,%xmm4+ rorl $9,%r14d+ xorl %edx,%r13d+ xorl %r9d,%r12d+ rorl $5,%r13d+ psrld $11,%xmm6+ xorl %r11d,%r14d+ pxor %xmm5,%xmm4+ andl %edx,%r12d+ xorl %edx,%r13d+ pslld $11,%xmm5+ addl 4(%rsp),%r10d+ movl %r11d,%edi+ pxor %xmm6,%xmm4+ xorl %r9d,%r12d+ rorl $11,%r14d+ movdqa %xmm7,%xmm6+ xorl %eax,%edi+ addl %r12d,%r10d+ pxor %xmm5,%xmm4+ rorl $6,%r13d+ andl %edi,%r15d+ xorl %r11d,%r14d+ psrld $10,%xmm7+ addl %r13d,%r10d+ xorl %eax,%r15d+ paddd %xmm4,%xmm0+ rorl $2,%r14d+ addl %r10d,%ecx+ psrlq $17,%xmm6+ addl %r15d,%r10d+ movl %ecx,%r13d+ addl %r10d,%r14d+ pxor %xmm6,%xmm7+ rorl $14,%r13d+ movl %r14d,%r10d+ movl %edx,%r12d+ rorl $9,%r14d+ psrlq $2,%xmm6+ xorl %ecx,%r13d+ xorl %r8d,%r12d+ pxor %xmm6,%xmm7+ rorl $5,%r13d+ xorl %r10d,%r14d+ andl %ecx,%r12d+ pshufd $128,%xmm7,%xmm7+ xorl %ecx,%r13d+ addl 8(%rsp),%r9d+ movl %r10d,%r15d+ psrldq $8,%xmm7+ xorl %r8d,%r12d+ rorl $11,%r14d+ xorl %r11d,%r15d+ addl %r12d,%r9d+ rorl $6,%r13d+ paddd %xmm7,%xmm0+ andl %r15d,%edi+ xorl %r10d,%r14d+ addl %r13d,%r9d+ pshufd $80,%xmm0,%xmm7+ xorl %r11d,%edi+ rorl $2,%r14d+ addl %r9d,%ebx+ movdqa %xmm7,%xmm6+ addl %edi,%r9d+ movl %ebx,%r13d+ psrld $10,%xmm7+ addl %r9d,%r14d+ rorl $14,%r13d+ psrlq $17,%xmm6+ movl %r14d,%r9d+ movl %ecx,%r12d+ pxor %xmm6,%xmm7+ rorl $9,%r14d+ xorl %ebx,%r13d+ xorl %edx,%r12d+ rorl $5,%r13d+ xorl %r9d,%r14d+ psrlq $2,%xmm6+ andl %ebx,%r12d+ xorl %ebx,%r13d+ addl 12(%rsp),%r8d+ pxor %xmm6,%xmm7+ movl %r9d,%edi+ xorl %edx,%r12d+ rorl $11,%r14d+ pshufd $8,%xmm7,%xmm7+ xorl %r10d,%edi+ addl %r12d,%r8d+ movdqa 0(%rsi),%xmm6+ rorl $6,%r13d+ andl %edi,%r15d+ pslldq $8,%xmm7+ xorl %r9d,%r14d+ addl %r13d,%r8d+ xorl %r10d,%r15d+ paddd %xmm7,%xmm0+ rorl $2,%r14d+ addl %r8d,%eax+ addl %r15d,%r8d+ paddd %xmm0,%xmm6+ movl %eax,%r13d+ addl %r8d,%r14d+ movdqa %xmm6,0(%rsp)+ rorl $14,%r13d+ movdqa %xmm2,%xmm4+ movl %r14d,%r8d+ movl %ebx,%r12d+ movdqa %xmm0,%xmm7+ rorl $9,%r14d+ xorl %eax,%r13d+ xorl %ecx,%r12d+ rorl $5,%r13d+ xorl %r8d,%r14d+.byte 102,15,58,15,225,4+ andl %eax,%r12d+ xorl %eax,%r13d+.byte 102,15,58,15,251,4+ addl 16(%rsp),%edx+ movl %r8d,%r15d+ xorl %ecx,%r12d+ rorl $11,%r14d+ movdqa %xmm4,%xmm5+ xorl %r9d,%r15d+ addl %r12d,%edx+ movdqa %xmm4,%xmm6+ rorl $6,%r13d+ andl %r15d,%edi+ psrld $3,%xmm4+ xorl %r8d,%r14d+ addl %r13d,%edx+ xorl %r9d,%edi+ paddd %xmm7,%xmm1+ rorl $2,%r14d+ addl %edx,%r11d+ psrld $7,%xmm6+ addl %edi,%edx+ movl %r11d,%r13d+ pshufd $250,%xmm0,%xmm7+ addl %edx,%r14d+ rorl $14,%r13d+ pslld $14,%xmm5+ movl %r14d,%edx+ movl %eax,%r12d+ pxor %xmm6,%xmm4+ rorl $9,%r14d+ xorl %r11d,%r13d+ xorl %ebx,%r12d+ rorl $5,%r13d+ psrld $11,%xmm6+ xorl %edx,%r14d+ pxor %xmm5,%xmm4+ andl %r11d,%r12d+ xorl %r11d,%r13d+ pslld $11,%xmm5+ addl 20(%rsp),%ecx+ movl %edx,%edi+ pxor %xmm6,%xmm4+ xorl %ebx,%r12d+ rorl $11,%r14d+ movdqa %xmm7,%xmm6+ xorl %r8d,%edi+ addl %r12d,%ecx+ pxor %xmm5,%xmm4+ rorl $6,%r13d+ andl %edi,%r15d+ xorl %edx,%r14d+ psrld $10,%xmm7+ addl %r13d,%ecx+ xorl %r8d,%r15d+ paddd %xmm4,%xmm1+ rorl $2,%r14d+ addl %ecx,%r10d+ psrlq $17,%xmm6+ addl %r15d,%ecx+ movl %r10d,%r13d+ addl %ecx,%r14d+ pxor %xmm6,%xmm7+ rorl $14,%r13d+ movl %r14d,%ecx+ movl %r11d,%r12d+ rorl $9,%r14d+ psrlq $2,%xmm6+ xorl %r10d,%r13d+ xorl %eax,%r12d+ pxor %xmm6,%xmm7+ rorl $5,%r13d+ xorl %ecx,%r14d+ andl %r10d,%r12d+ pshufd $128,%xmm7,%xmm7+ xorl %r10d,%r13d+ addl 24(%rsp),%ebx+ movl %ecx,%r15d+ psrldq $8,%xmm7+ xorl %eax,%r12d+ rorl $11,%r14d+ xorl %edx,%r15d+ addl %r12d,%ebx+ rorl $6,%r13d+ paddd %xmm7,%xmm1+ andl %r15d,%edi+ xorl %ecx,%r14d+ addl %r13d,%ebx+ pshufd $80,%xmm1,%xmm7+ xorl %edx,%edi+ rorl $2,%r14d+ addl %ebx,%r9d+ movdqa %xmm7,%xmm6+ addl %edi,%ebx+ movl %r9d,%r13d+ psrld $10,%xmm7+ addl %ebx,%r14d+ rorl $14,%r13d+ psrlq $17,%xmm6+ movl %r14d,%ebx+ movl %r10d,%r12d+ pxor %xmm6,%xmm7+ rorl $9,%r14d+ xorl %r9d,%r13d+ xorl %r11d,%r12d+ rorl $5,%r13d+ xorl %ebx,%r14d+ psrlq $2,%xmm6+ andl %r9d,%r12d+ xorl %r9d,%r13d+ addl 28(%rsp),%eax+ pxor %xmm6,%xmm7+ movl %ebx,%edi+ xorl %r11d,%r12d+ rorl $11,%r14d+ pshufd $8,%xmm7,%xmm7+ xorl %ecx,%edi+ addl %r12d,%eax+ movdqa 32(%rsi),%xmm6+ rorl $6,%r13d+ andl %edi,%r15d+ pslldq $8,%xmm7+ xorl %ebx,%r14d+ addl %r13d,%eax+ xorl %ecx,%r15d+ paddd %xmm7,%xmm1+ rorl $2,%r14d+ addl %eax,%r8d+ addl %r15d,%eax+ paddd %xmm1,%xmm6+ movl %r8d,%r13d+ addl %eax,%r14d+ movdqa %xmm6,16(%rsp)+ rorl $14,%r13d+ movdqa %xmm3,%xmm4+ movl %r14d,%eax+ movl %r9d,%r12d+ movdqa %xmm1,%xmm7+ rorl $9,%r14d+ xorl %r8d,%r13d+ xorl %r10d,%r12d+ rorl $5,%r13d+ xorl %eax,%r14d+.byte 102,15,58,15,226,4+ andl %r8d,%r12d+ xorl %r8d,%r13d+.byte 102,15,58,15,248,4+ addl 32(%rsp),%r11d+ movl %eax,%r15d+ xorl %r10d,%r12d+ rorl $11,%r14d+ movdqa %xmm4,%xmm5+ xorl %ebx,%r15d+ addl %r12d,%r11d+ movdqa %xmm4,%xmm6+ rorl $6,%r13d+ andl %r15d,%edi+ psrld $3,%xmm4+ xorl %eax,%r14d+ addl %r13d,%r11d+ xorl %ebx,%edi+ paddd %xmm7,%xmm2+ rorl $2,%r14d+ addl %r11d,%edx+ psrld $7,%xmm6+ addl %edi,%r11d+ movl %edx,%r13d+ pshufd $250,%xmm1,%xmm7+ addl %r11d,%r14d+ rorl $14,%r13d+ pslld $14,%xmm5+ movl %r14d,%r11d+ movl %r8d,%r12d+ pxor %xmm6,%xmm4+ rorl $9,%r14d+ xorl %edx,%r13d+ xorl %r9d,%r12d+ rorl $5,%r13d+ psrld $11,%xmm6+ xorl %r11d,%r14d+ pxor %xmm5,%xmm4+ andl %edx,%r12d+ xorl %edx,%r13d+ pslld $11,%xmm5+ addl 36(%rsp),%r10d+ movl %r11d,%edi+ pxor %xmm6,%xmm4+ xorl %r9d,%r12d+ rorl $11,%r14d+ movdqa %xmm7,%xmm6+ xorl %eax,%edi+ addl %r12d,%r10d+ pxor %xmm5,%xmm4+ rorl $6,%r13d+ andl %edi,%r15d+ xorl %r11d,%r14d+ psrld $10,%xmm7+ addl %r13d,%r10d+ xorl %eax,%r15d+ paddd %xmm4,%xmm2+ rorl $2,%r14d+ addl %r10d,%ecx+ psrlq $17,%xmm6+ addl %r15d,%r10d+ movl %ecx,%r13d+ addl %r10d,%r14d+ pxor %xmm6,%xmm7+ rorl $14,%r13d+ movl %r14d,%r10d+ movl %edx,%r12d+ rorl $9,%r14d+ psrlq $2,%xmm6+ xorl %ecx,%r13d+ xorl %r8d,%r12d+ pxor %xmm6,%xmm7+ rorl $5,%r13d+ xorl %r10d,%r14d+ andl %ecx,%r12d+ pshufd $128,%xmm7,%xmm7+ xorl %ecx,%r13d+ addl 40(%rsp),%r9d+ movl %r10d,%r15d+ psrldq $8,%xmm7+ xorl %r8d,%r12d+ rorl $11,%r14d+ xorl %r11d,%r15d+ addl %r12d,%r9d+ rorl $6,%r13d+ paddd %xmm7,%xmm2+ andl %r15d,%edi+ xorl %r10d,%r14d+ addl %r13d,%r9d+ pshufd $80,%xmm2,%xmm7+ xorl %r11d,%edi+ rorl $2,%r14d+ addl %r9d,%ebx+ movdqa %xmm7,%xmm6+ addl %edi,%r9d+ movl %ebx,%r13d+ psrld $10,%xmm7+ addl %r9d,%r14d+ rorl $14,%r13d+ psrlq $17,%xmm6+ movl %r14d,%r9d+ movl %ecx,%r12d+ pxor %xmm6,%xmm7+ rorl $9,%r14d+ xorl %ebx,%r13d+ xorl %edx,%r12d+ rorl $5,%r13d+ xorl %r9d,%r14d+ psrlq $2,%xmm6+ andl %ebx,%r12d+ xorl %ebx,%r13d+ addl 44(%rsp),%r8d+ pxor %xmm6,%xmm7+ movl %r9d,%edi+ xorl %edx,%r12d+ rorl $11,%r14d+ pshufd $8,%xmm7,%xmm7+ xorl %r10d,%edi+ addl %r12d,%r8d+ movdqa 64(%rsi),%xmm6+ rorl $6,%r13d+ andl %edi,%r15d+ pslldq $8,%xmm7+ xorl %r9d,%r14d+ addl %r13d,%r8d+ xorl %r10d,%r15d+ paddd %xmm7,%xmm2+ rorl $2,%r14d+ addl %r8d,%eax+ addl %r15d,%r8d+ paddd %xmm2,%xmm6+ movl %eax,%r13d+ addl %r8d,%r14d+ movdqa %xmm6,32(%rsp)+ rorl $14,%r13d+ movdqa %xmm0,%xmm4+ movl %r14d,%r8d+ movl %ebx,%r12d+ movdqa %xmm2,%xmm7+ rorl $9,%r14d+ xorl %eax,%r13d+ xorl %ecx,%r12d+ rorl $5,%r13d+ xorl %r8d,%r14d+.byte 102,15,58,15,227,4+ andl %eax,%r12d+ xorl %eax,%r13d+.byte 102,15,58,15,249,4+ addl 48(%rsp),%edx+ movl %r8d,%r15d+ xorl %ecx,%r12d+ rorl $11,%r14d+ movdqa %xmm4,%xmm5+ xorl %r9d,%r15d+ addl %r12d,%edx+ movdqa %xmm4,%xmm6+ rorl $6,%r13d+ andl %r15d,%edi+ psrld $3,%xmm4+ xorl %r8d,%r14d+ addl %r13d,%edx+ xorl %r9d,%edi+ paddd %xmm7,%xmm3+ rorl $2,%r14d+ addl %edx,%r11d+ psrld $7,%xmm6+ addl %edi,%edx+ movl %r11d,%r13d+ pshufd $250,%xmm2,%xmm7+ addl %edx,%r14d+ rorl $14,%r13d+ pslld $14,%xmm5+ movl %r14d,%edx+ movl %eax,%r12d+ pxor %xmm6,%xmm4+ rorl $9,%r14d+ xorl %r11d,%r13d+ xorl %ebx,%r12d+ rorl $5,%r13d+ psrld $11,%xmm6+ xorl %edx,%r14d+ pxor %xmm5,%xmm4+ andl %r11d,%r12d+ xorl %r11d,%r13d+ pslld $11,%xmm5+ addl 52(%rsp),%ecx+ movl %edx,%edi+ pxor %xmm6,%xmm4+ xorl %ebx,%r12d+ rorl $11,%r14d+ movdqa %xmm7,%xmm6+ xorl %r8d,%edi+ addl %r12d,%ecx+ pxor %xmm5,%xmm4+ rorl $6,%r13d+ andl %edi,%r15d+ xorl %edx,%r14d+ psrld $10,%xmm7+ addl %r13d,%ecx+ xorl %r8d,%r15d+ paddd %xmm4,%xmm3+ rorl $2,%r14d+ addl %ecx,%r10d+ psrlq $17,%xmm6+ addl %r15d,%ecx+ movl %r10d,%r13d+ addl %ecx,%r14d+ pxor %xmm6,%xmm7+ rorl $14,%r13d+ movl %r14d,%ecx+ movl %r11d,%r12d+ rorl $9,%r14d+ psrlq $2,%xmm6+ xorl %r10d,%r13d+ xorl %eax,%r12d+ pxor %xmm6,%xmm7+ rorl $5,%r13d+ xorl %ecx,%r14d+ andl %r10d,%r12d+ pshufd $128,%xmm7,%xmm7+ xorl %r10d,%r13d+ addl 56(%rsp),%ebx+ movl %ecx,%r15d+ psrldq $8,%xmm7+ xorl %eax,%r12d+ rorl $11,%r14d+ xorl %edx,%r15d+ addl %r12d,%ebx+ rorl $6,%r13d+ paddd %xmm7,%xmm3+ andl %r15d,%edi+ xorl %ecx,%r14d+ addl %r13d,%ebx+ pshufd $80,%xmm3,%xmm7+ xorl %edx,%edi+ rorl $2,%r14d+ addl %ebx,%r9d+ movdqa %xmm7,%xmm6+ addl %edi,%ebx+ movl %r9d,%r13d+ psrld $10,%xmm7+ addl %ebx,%r14d+ rorl $14,%r13d+ psrlq $17,%xmm6+ movl %r14d,%ebx+ movl %r10d,%r12d+ pxor %xmm6,%xmm7+ rorl $9,%r14d+ xorl %r9d,%r13d+ xorl %r11d,%r12d+ rorl $5,%r13d+ xorl %ebx,%r14d+ psrlq $2,%xmm6+ andl %r9d,%r12d+ xorl %r9d,%r13d+ addl 60(%rsp),%eax+ pxor %xmm6,%xmm7+ movl %ebx,%edi+ xorl %r11d,%r12d+ rorl $11,%r14d+ pshufd $8,%xmm7,%xmm7+ xorl %ecx,%edi+ addl %r12d,%eax+ movdqa 96(%rsi),%xmm6+ rorl $6,%r13d+ andl %edi,%r15d+ pslldq $8,%xmm7+ xorl %ebx,%r14d+ addl %r13d,%eax+ xorl %ecx,%r15d+ paddd %xmm7,%xmm3+ rorl $2,%r14d+ addl %eax,%r8d+ addl %r15d,%eax+ paddd %xmm3,%xmm6+ movl %r8d,%r13d+ addl %eax,%r14d+ movdqa %xmm6,48(%rsp)+ cmpb $0,131(%rsi)+ jne L$ssse3_00_47+ rorl $14,%r13d+ movl %r14d,%eax+ movl %r9d,%r12d+ rorl $9,%r14d+ xorl %r8d,%r13d+ xorl %r10d,%r12d+ rorl $5,%r13d+ xorl %eax,%r14d+ andl %r8d,%r12d+ xorl %r8d,%r13d+ addl 0(%rsp),%r11d+ movl %eax,%r15d+ xorl %r10d,%r12d+ rorl $11,%r14d+ xorl %ebx,%r15d+ addl %r12d,%r11d+ rorl $6,%r13d+ andl %r15d,%edi+ xorl %eax,%r14d+ addl %r13d,%r11d+ xorl %ebx,%edi+ rorl $2,%r14d+ addl %r11d,%edx+ addl %edi,%r11d+ movl %edx,%r13d+ addl %r11d,%r14d+ rorl $14,%r13d+ movl %r14d,%r11d+ movl %r8d,%r12d+ rorl $9,%r14d+ xorl %edx,%r13d+ xorl %r9d,%r12d+ rorl $5,%r13d+ xorl %r11d,%r14d+ andl %edx,%r12d+ xorl %edx,%r13d+ addl 4(%rsp),%r10d+ movl %r11d,%edi+ xorl %r9d,%r12d+ rorl $11,%r14d+ xorl %eax,%edi+ addl %r12d,%r10d+ rorl $6,%r13d+ andl %edi,%r15d+ xorl %r11d,%r14d+ addl %r13d,%r10d+ xorl %eax,%r15d+ rorl $2,%r14d+ addl %r10d,%ecx+ addl %r15d,%r10d+ movl %ecx,%r13d+ addl %r10d,%r14d+ rorl $14,%r13d+ movl %r14d,%r10d+ movl %edx,%r12d+ rorl $9,%r14d+ xorl %ecx,%r13d+ xorl %r8d,%r12d+ rorl $5,%r13d+ xorl %r10d,%r14d+ andl %ecx,%r12d+ xorl %ecx,%r13d+ addl 8(%rsp),%r9d+ movl %r10d,%r15d+ xorl %r8d,%r12d+ rorl $11,%r14d+ xorl %r11d,%r15d+ addl %r12d,%r9d+ rorl $6,%r13d+ andl %r15d,%edi+ xorl %r10d,%r14d+ addl %r13d,%r9d+ xorl %r11d,%edi+ rorl $2,%r14d+ addl %r9d,%ebx+ addl %edi,%r9d+ movl %ebx,%r13d+ addl %r9d,%r14d+ rorl $14,%r13d+ movl %r14d,%r9d+ movl %ecx,%r12d+ rorl $9,%r14d+ xorl %ebx,%r13d+ xorl %edx,%r12d+ rorl $5,%r13d+ xorl %r9d,%r14d+ andl %ebx,%r12d+ xorl %ebx,%r13d+ addl 12(%rsp),%r8d+ movl %r9d,%edi+ xorl %edx,%r12d+ rorl $11,%r14d+ xorl %r10d,%edi+ addl %r12d,%r8d+ rorl $6,%r13d+ andl %edi,%r15d+ xorl %r9d,%r14d+ addl %r13d,%r8d+ xorl %r10d,%r15d+ rorl $2,%r14d+ addl %r8d,%eax+ addl %r15d,%r8d+ movl %eax,%r13d+ addl %r8d,%r14d+ rorl $14,%r13d+ movl %r14d,%r8d+ movl %ebx,%r12d+ rorl $9,%r14d+ xorl %eax,%r13d+ xorl %ecx,%r12d+ rorl $5,%r13d+ xorl %r8d,%r14d+ andl %eax,%r12d+ xorl %eax,%r13d+ addl 16(%rsp),%edx+ movl %r8d,%r15d+ xorl %ecx,%r12d+ rorl $11,%r14d+ xorl %r9d,%r15d+ addl %r12d,%edx+ rorl $6,%r13d+ andl %r15d,%edi+ xorl %r8d,%r14d+ addl %r13d,%edx+ xorl %r9d,%edi+ rorl $2,%r14d+ addl %edx,%r11d+ addl %edi,%edx+ movl %r11d,%r13d+ addl %edx,%r14d+ rorl $14,%r13d+ movl %r14d,%edx+ movl %eax,%r12d+ rorl $9,%r14d+ xorl %r11d,%r13d+ xorl %ebx,%r12d+ rorl $5,%r13d+ xorl %edx,%r14d+ andl %r11d,%r12d+ xorl %r11d,%r13d+ addl 20(%rsp),%ecx+ movl %edx,%edi+ xorl %ebx,%r12d+ rorl $11,%r14d+ xorl %r8d,%edi+ addl %r12d,%ecx+ rorl $6,%r13d+ andl %edi,%r15d+ xorl %edx,%r14d+ addl %r13d,%ecx+ xorl %r8d,%r15d+ rorl $2,%r14d+ addl %ecx,%r10d+ addl %r15d,%ecx+ movl %r10d,%r13d+ addl %ecx,%r14d+ rorl $14,%r13d+ movl %r14d,%ecx+ movl %r11d,%r12d+ rorl $9,%r14d+ xorl %r10d,%r13d+ xorl %eax,%r12d+ rorl $5,%r13d+ xorl %ecx,%r14d+ andl %r10d,%r12d+ xorl %r10d,%r13d+ addl 24(%rsp),%ebx+ movl %ecx,%r15d+ xorl %eax,%r12d+ rorl $11,%r14d+ xorl %edx,%r15d+ addl %r12d,%ebx+ rorl $6,%r13d+ andl %r15d,%edi+ xorl %ecx,%r14d+ addl %r13d,%ebx+ xorl %edx,%edi+ rorl $2,%r14d+ addl %ebx,%r9d+ addl %edi,%ebx+ movl %r9d,%r13d+ addl %ebx,%r14d+ rorl $14,%r13d+ movl %r14d,%ebx+ movl %r10d,%r12d+ rorl $9,%r14d+ xorl %r9d,%r13d+ xorl %r11d,%r12d+ rorl $5,%r13d+ xorl %ebx,%r14d+ andl %r9d,%r12d+ xorl %r9d,%r13d+ addl 28(%rsp),%eax+ movl %ebx,%edi+ xorl %r11d,%r12d+ rorl $11,%r14d+ xorl %ecx,%edi+ addl %r12d,%eax+ rorl $6,%r13d+ andl %edi,%r15d+ xorl %ebx,%r14d+ addl %r13d,%eax+ xorl %ecx,%r15d+ rorl $2,%r14d+ addl %eax,%r8d+ addl %r15d,%eax+ movl %r8d,%r13d+ addl %eax,%r14d+ rorl $14,%r13d+ movl %r14d,%eax+ movl %r9d,%r12d+ rorl $9,%r14d+ xorl %r8d,%r13d+ xorl %r10d,%r12d+ rorl $5,%r13d+ xorl %eax,%r14d+ andl %r8d,%r12d+ xorl %r8d,%r13d+ addl 32(%rsp),%r11d+ movl %eax,%r15d+ xorl %r10d,%r12d+ rorl $11,%r14d+ xorl %ebx,%r15d+ addl %r12d,%r11d+ rorl $6,%r13d+ andl %r15d,%edi+ xorl %eax,%r14d+ addl %r13d,%r11d+ xorl %ebx,%edi+ rorl $2,%r14d+ addl %r11d,%edx+ addl %edi,%r11d+ movl %edx,%r13d+ addl %r11d,%r14d+ rorl $14,%r13d+ movl %r14d,%r11d+ movl %r8d,%r12d+ rorl $9,%r14d+ xorl %edx,%r13d+ xorl %r9d,%r12d+ rorl $5,%r13d+ xorl %r11d,%r14d+ andl %edx,%r12d+ xorl %edx,%r13d+ addl 36(%rsp),%r10d+ movl %r11d,%edi+ xorl %r9d,%r12d+ rorl $11,%r14d+ xorl %eax,%edi+ addl %r12d,%r10d+ rorl $6,%r13d+ andl %edi,%r15d+ xorl %r11d,%r14d+ addl %r13d,%r10d+ xorl %eax,%r15d+ rorl $2,%r14d+ addl %r10d,%ecx+ addl %r15d,%r10d+ movl %ecx,%r13d+ addl %r10d,%r14d+ rorl $14,%r13d+ movl %r14d,%r10d+ movl %edx,%r12d+ rorl $9,%r14d+ xorl %ecx,%r13d+ xorl %r8d,%r12d+ rorl $5,%r13d+ xorl %r10d,%r14d+ andl %ecx,%r12d+ xorl %ecx,%r13d+ addl 40(%rsp),%r9d+ movl %r10d,%r15d+ xorl %r8d,%r12d+ rorl $11,%r14d+ xorl %r11d,%r15d+ addl %r12d,%r9d+ rorl $6,%r13d+ andl %r15d,%edi+ xorl %r10d,%r14d+ addl %r13d,%r9d+ xorl %r11d,%edi+ rorl $2,%r14d+ addl %r9d,%ebx+ addl %edi,%r9d+ movl %ebx,%r13d+ addl %r9d,%r14d+ rorl $14,%r13d+ movl %r14d,%r9d+ movl %ecx,%r12d+ rorl $9,%r14d+ xorl %ebx,%r13d+ xorl %edx,%r12d+ rorl $5,%r13d+ xorl %r9d,%r14d+ andl %ebx,%r12d+ xorl %ebx,%r13d+ addl 44(%rsp),%r8d+ movl %r9d,%edi+ xorl %edx,%r12d+ rorl $11,%r14d+ xorl %r10d,%edi+ addl %r12d,%r8d+ rorl $6,%r13d+ andl %edi,%r15d+ xorl %r9d,%r14d+ addl %r13d,%r8d+ xorl %r10d,%r15d+ rorl $2,%r14d+ addl %r8d,%eax+ addl %r15d,%r8d+ movl %eax,%r13d+ addl %r8d,%r14d+ rorl $14,%r13d+ movl %r14d,%r8d+ movl %ebx,%r12d+ rorl $9,%r14d+ xorl %eax,%r13d+ xorl %ecx,%r12d+ rorl $5,%r13d+ xorl %r8d,%r14d+ andl %eax,%r12d+ xorl %eax,%r13d+ addl 48(%rsp),%edx+ movl %r8d,%r15d+ xorl %ecx,%r12d+ rorl $11,%r14d+ xorl %r9d,%r15d+ addl %r12d,%edx+ rorl $6,%r13d+ andl %r15d,%edi+ xorl %r8d,%r14d+ addl %r13d,%edx+ xorl %r9d,%edi+ rorl $2,%r14d+ addl %edx,%r11d+ addl %edi,%edx+ movl %r11d,%r13d+ addl %edx,%r14d+ rorl $14,%r13d+ movl %r14d,%edx+ movl %eax,%r12d+ rorl $9,%r14d+ xorl %r11d,%r13d+ xorl %ebx,%r12d+ rorl $5,%r13d+ xorl %edx,%r14d+ andl %r11d,%r12d+ xorl %r11d,%r13d+ addl 52(%rsp),%ecx+ movl %edx,%edi+ xorl %ebx,%r12d+ rorl $11,%r14d+ xorl %r8d,%edi+ addl %r12d,%ecx+ rorl $6,%r13d+ andl %edi,%r15d+ xorl %edx,%r14d+ addl %r13d,%ecx+ xorl %r8d,%r15d+ rorl $2,%r14d+ addl %ecx,%r10d+ addl %r15d,%ecx+ movl %r10d,%r13d+ addl %ecx,%r14d+ rorl $14,%r13d+ movl %r14d,%ecx+ movl %r11d,%r12d+ rorl $9,%r14d+ xorl %r10d,%r13d+ xorl %eax,%r12d+ rorl $5,%r13d+ xorl %ecx,%r14d+ andl %r10d,%r12d+ xorl %r10d,%r13d+ addl 56(%rsp),%ebx+ movl %ecx,%r15d+ xorl %eax,%r12d+ rorl $11,%r14d+ xorl %edx,%r15d+ addl %r12d,%ebx+ rorl $6,%r13d+ andl %r15d,%edi+ xorl %ecx,%r14d+ addl %r13d,%ebx+ xorl %edx,%edi+ rorl $2,%r14d+ addl %ebx,%r9d+ addl %edi,%ebx+ movl %r9d,%r13d+ addl %ebx,%r14d+ rorl $14,%r13d+ movl %r14d,%ebx+ movl %r10d,%r12d+ rorl $9,%r14d+ xorl %r9d,%r13d+ xorl %r11d,%r12d+ rorl $5,%r13d+ xorl %ebx,%r14d+ andl %r9d,%r12d+ xorl %r9d,%r13d+ addl 60(%rsp),%eax+ movl %ebx,%edi+ xorl %r11d,%r12d+ rorl $11,%r14d+ xorl %ecx,%edi+ addl %r12d,%eax+ rorl $6,%r13d+ andl %edi,%r15d+ xorl %ebx,%r14d+ addl %r13d,%eax+ xorl %ecx,%r15d+ rorl $2,%r14d+ addl %eax,%r8d+ addl %r15d,%eax+ movl %r8d,%r13d+ addl %eax,%r14d+ movq -64(%rbp),%rdi+ movl %r14d,%eax+ movq -56(%rbp),%rsi++ addl 0(%rdi),%eax+ addl 4(%rdi),%ebx+ addl 8(%rdi),%ecx+ addl 12(%rdi),%edx+ addl 16(%rdi),%r8d+ addl 20(%rdi),%r9d+ addl 24(%rdi),%r10d+ addl 28(%rdi),%r11d++ leaq 64(%rsi),%rsi+ cmpq -48(%rbp),%rsi++ movl %eax,0(%rdi)+ movl %ebx,4(%rdi)+ movl %ecx,8(%rdi)+ movl %edx,12(%rdi)+ movl %r8d,16(%rdi)+ movl %r9d,20(%rdi)+ movl %r10d,24(%rdi)+ movl %r11d,28(%rdi)+ jb L$loop_ssse3++ movq -40(%rbp),%r15+ movq -32(%rbp),%r14+ movq -24(%rbp),%r13+ movq -16(%rbp),%r12+ movq -8(%rbp),%rbx+ movq %rbp,%rsp+.cfi_def_cfa_register %rsp+ popq %rbp+.cfi_adjust_cfa_offset -8+.cfi_restore %rbp+.cfi_restore %r12+.cfi_restore %r13+.cfi_restore %r14+.cfi_restore %r15+.cfi_restore %rbx+ .byte 0xf3,0xc3+.cfi_endproc +++.p2align 6+crypton_sha256_asm_block_data_order_avx:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa+++ pushq %rbp+.cfi_adjust_cfa_offset 8+.cfi_offset %rbp,-16+ movq %rsp,%rbp+.cfi_def_cfa_register %rbp+L$avx_shortcut:+ pushq %rbx+.cfi_offset %rbx,-24+ pushq %r12+.cfi_offset %r12,-32+ pushq %r13+.cfi_offset %r13,-40+ pushq %r14+.cfi_offset %r14,-48+ pushq %r15+.cfi_offset %r15,-56+ shlq $4,%rdx+ subq $24,%rsp++ leaq (%rsi,%rdx,4),%rdx+ movq %rdi,-64(%rbp)++ movq %rdx,-48(%rbp)+++ leaq -64(%rsp),%rsp+ vzeroupper+ andq $-64,%rsp+ movl 0(%rdi),%eax+ movl 4(%rdi),%ebx+ movl 8(%rdi),%ecx+ movl 12(%rdi),%edx+ movl 16(%rdi),%r8d+ movl 20(%rdi),%r9d+ movl 24(%rdi),%r10d+ movl 28(%rdi),%r11d+ vmovdqa K256+512+32(%rip),%xmm8+ vmovdqa K256+512+64(%rip),%xmm9+ jmp L$loop_avx+.p2align 4+L$loop_avx:+ vmovdqa K256+512(%rip),%xmm7+ movq %rsi,-56(%rbp)+ vmovdqu 0(%rsi),%xmm0+ vmovdqu 16(%rsi),%xmm1+ vmovdqu 32(%rsi),%xmm2+ vmovdqu 48(%rsi),%xmm3+ vpshufb %xmm7,%xmm0,%xmm0+ leaq K256(%rip),%rsi+ vpshufb %xmm7,%xmm1,%xmm1+ vpshufb %xmm7,%xmm2,%xmm2+ vpaddd 0(%rsi),%xmm0,%xmm4+ vpshufb %xmm7,%xmm3,%xmm3+ vpaddd 32(%rsi),%xmm1,%xmm5+ vpaddd 64(%rsi),%xmm2,%xmm6+ vpaddd 96(%rsi),%xmm3,%xmm7+ vmovdqa %xmm4,0(%rsp)+ movl %eax,%r14d+ vmovdqa %xmm5,16(%rsp)+ movl %ebx,%edi+ vmovdqa %xmm6,32(%rsp)+ xorl %ecx,%edi+ vmovdqa %xmm7,48(%rsp)+ movl %r8d,%r13d+ jmp L$avx_00_47++.p2align 4+L$avx_00_47:+ subq $-128,%rsi+ vpalignr $4,%xmm0,%xmm1,%xmm4+ shrdl $14,%r13d,%r13d+ movl %r14d,%eax+ movl %r9d,%r12d+ vpalignr $4,%xmm2,%xmm3,%xmm7+ shrdl $9,%r14d,%r14d+ xorl %r8d,%r13d+ xorl %r10d,%r12d+ vpsrld $7,%xmm4,%xmm6+ shrdl $5,%r13d,%r13d+ xorl %eax,%r14d+ andl %r8d,%r12d+ vpaddd %xmm7,%xmm0,%xmm0+ xorl %r8d,%r13d+ addl 0(%rsp),%r11d+ movl %eax,%r15d+ vpsrld $3,%xmm4,%xmm7+ xorl %r10d,%r12d+ shrdl $11,%r14d,%r14d+ xorl %ebx,%r15d+ vpslld $14,%xmm4,%xmm5+ addl %r12d,%r11d+ shrdl $6,%r13d,%r13d+ andl %r15d,%edi+ vpxor %xmm6,%xmm7,%xmm4+ xorl %eax,%r14d+ addl %r13d,%r11d+ xorl %ebx,%edi+ vpshufd $250,%xmm3,%xmm7+ shrdl $2,%r14d,%r14d+ addl %r11d,%edx+ addl %edi,%r11d+ vpsrld $11,%xmm6,%xmm6+ movl %edx,%r13d+ addl %r11d,%r14d+ shrdl $14,%r13d,%r13d+ vpxor %xmm5,%xmm4,%xmm4+ movl %r14d,%r11d+ movl %r8d,%r12d+ shrdl $9,%r14d,%r14d+ vpslld $11,%xmm5,%xmm5+ xorl %edx,%r13d+ xorl %r9d,%r12d+ shrdl $5,%r13d,%r13d+ vpxor %xmm6,%xmm4,%xmm4+ xorl %r11d,%r14d+ andl %edx,%r12d+ xorl %edx,%r13d+ vpsrld $10,%xmm7,%xmm6+ addl 4(%rsp),%r10d+ movl %r11d,%edi+ xorl %r9d,%r12d+ vpxor %xmm5,%xmm4,%xmm4+ shrdl $11,%r14d,%r14d+ xorl %eax,%edi+ addl %r12d,%r10d+ vpsrlq $17,%xmm7,%xmm7+ shrdl $6,%r13d,%r13d+ andl %edi,%r15d+ xorl %r11d,%r14d+ vpaddd %xmm4,%xmm0,%xmm0+ addl %r13d,%r10d+ xorl %eax,%r15d+ shrdl $2,%r14d,%r14d+ vpxor %xmm7,%xmm6,%xmm6+ addl %r10d,%ecx+ addl %r15d,%r10d+ movl %ecx,%r13d+ vpsrlq $2,%xmm7,%xmm7+ addl %r10d,%r14d+ shrdl $14,%r13d,%r13d+ movl %r14d,%r10d+ vpxor %xmm7,%xmm6,%xmm6+ movl %edx,%r12d+ shrdl $9,%r14d,%r14d+ xorl %ecx,%r13d+ vpshufb %xmm8,%xmm6,%xmm6+ xorl %r8d,%r12d+ shrdl $5,%r13d,%r13d+ xorl %r10d,%r14d+ vpaddd %xmm6,%xmm0,%xmm0+ andl %ecx,%r12d+ xorl %ecx,%r13d+ addl 8(%rsp),%r9d+ vpshufd $80,%xmm0,%xmm7+ movl %r10d,%r15d+ xorl %r8d,%r12d+ shrdl $11,%r14d,%r14d+ vpsrld $10,%xmm7,%xmm6+ xorl %r11d,%r15d+ addl %r12d,%r9d+ shrdl $6,%r13d,%r13d+ vpsrlq $17,%xmm7,%xmm7+ andl %r15d,%edi+ xorl %r10d,%r14d+ addl %r13d,%r9d+ vpxor %xmm7,%xmm6,%xmm6+ xorl %r11d,%edi+ shrdl $2,%r14d,%r14d+ addl %r9d,%ebx+ vpsrlq $2,%xmm7,%xmm7+ addl %edi,%r9d+ movl %ebx,%r13d+ addl %r9d,%r14d+ vpxor %xmm7,%xmm6,%xmm6+ shrdl $14,%r13d,%r13d+ movl %r14d,%r9d+ movl %ecx,%r12d+ vpshufb %xmm9,%xmm6,%xmm6+ shrdl $9,%r14d,%r14d+ xorl %ebx,%r13d+ xorl %edx,%r12d+ vpaddd %xmm6,%xmm0,%xmm0+ shrdl $5,%r13d,%r13d+ xorl %r9d,%r14d+ andl %ebx,%r12d+ vpaddd 0(%rsi),%xmm0,%xmm6+ xorl %ebx,%r13d+ addl 12(%rsp),%r8d+ movl %r9d,%edi+ xorl %edx,%r12d+ shrdl $11,%r14d,%r14d+ xorl %r10d,%edi+ addl %r12d,%r8d+ shrdl $6,%r13d,%r13d+ andl %edi,%r15d+ xorl %r9d,%r14d+ addl %r13d,%r8d+ xorl %r10d,%r15d+ shrdl $2,%r14d,%r14d+ addl %r8d,%eax+ addl %r15d,%r8d+ movl %eax,%r13d+ addl %r8d,%r14d+ vmovdqa %xmm6,0(%rsp)+ vpalignr $4,%xmm1,%xmm2,%xmm4+ shrdl $14,%r13d,%r13d+ movl %r14d,%r8d+ movl %ebx,%r12d+ vpalignr $4,%xmm3,%xmm0,%xmm7+ shrdl $9,%r14d,%r14d+ xorl %eax,%r13d+ xorl %ecx,%r12d+ vpsrld $7,%xmm4,%xmm6+ shrdl $5,%r13d,%r13d+ xorl %r8d,%r14d+ andl %eax,%r12d+ vpaddd %xmm7,%xmm1,%xmm1+ xorl %eax,%r13d+ addl 16(%rsp),%edx+ movl %r8d,%r15d+ vpsrld $3,%xmm4,%xmm7+ xorl %ecx,%r12d+ shrdl $11,%r14d,%r14d+ xorl %r9d,%r15d+ vpslld $14,%xmm4,%xmm5+ addl %r12d,%edx+ shrdl $6,%r13d,%r13d+ andl %r15d,%edi+ vpxor %xmm6,%xmm7,%xmm4+ xorl %r8d,%r14d+ addl %r13d,%edx+ xorl %r9d,%edi+ vpshufd $250,%xmm0,%xmm7+ shrdl $2,%r14d,%r14d+ addl %edx,%r11d+ addl %edi,%edx+ vpsrld $11,%xmm6,%xmm6+ movl %r11d,%r13d+ addl %edx,%r14d+ shrdl $14,%r13d,%r13d+ vpxor %xmm5,%xmm4,%xmm4+ movl %r14d,%edx+ movl %eax,%r12d+ shrdl $9,%r14d,%r14d+ vpslld $11,%xmm5,%xmm5+ xorl %r11d,%r13d+ xorl %ebx,%r12d+ shrdl $5,%r13d,%r13d+ vpxor %xmm6,%xmm4,%xmm4+ xorl %edx,%r14d+ andl %r11d,%r12d+ xorl %r11d,%r13d+ vpsrld $10,%xmm7,%xmm6+ addl 20(%rsp),%ecx+ movl %edx,%edi+ xorl %ebx,%r12d+ vpxor %xmm5,%xmm4,%xmm4+ shrdl $11,%r14d,%r14d+ xorl %r8d,%edi+ addl %r12d,%ecx+ vpsrlq $17,%xmm7,%xmm7+ shrdl $6,%r13d,%r13d+ andl %edi,%r15d+ xorl %edx,%r14d+ vpaddd %xmm4,%xmm1,%xmm1+ addl %r13d,%ecx+ xorl %r8d,%r15d+ shrdl $2,%r14d,%r14d+ vpxor %xmm7,%xmm6,%xmm6+ addl %ecx,%r10d+ addl %r15d,%ecx+ movl %r10d,%r13d+ vpsrlq $2,%xmm7,%xmm7+ addl %ecx,%r14d+ shrdl $14,%r13d,%r13d+ movl %r14d,%ecx+ vpxor %xmm7,%xmm6,%xmm6+ movl %r11d,%r12d+ shrdl $9,%r14d,%r14d+ xorl %r10d,%r13d+ vpshufb %xmm8,%xmm6,%xmm6+ xorl %eax,%r12d+ shrdl $5,%r13d,%r13d+ xorl %ecx,%r14d+ vpaddd %xmm6,%xmm1,%xmm1+ andl %r10d,%r12d+ xorl %r10d,%r13d+ addl 24(%rsp),%ebx+ vpshufd $80,%xmm1,%xmm7+ movl %ecx,%r15d+ xorl %eax,%r12d+ shrdl $11,%r14d,%r14d+ vpsrld $10,%xmm7,%xmm6+ xorl %edx,%r15d+ addl %r12d,%ebx+ shrdl $6,%r13d,%r13d+ vpsrlq $17,%xmm7,%xmm7+ andl %r15d,%edi+ xorl %ecx,%r14d+ addl %r13d,%ebx+ vpxor %xmm7,%xmm6,%xmm6+ xorl %edx,%edi+ shrdl $2,%r14d,%r14d+ addl %ebx,%r9d+ vpsrlq $2,%xmm7,%xmm7+ addl %edi,%ebx+ movl %r9d,%r13d+ addl %ebx,%r14d+ vpxor %xmm7,%xmm6,%xmm6+ shrdl $14,%r13d,%r13d+ movl %r14d,%ebx+ movl %r10d,%r12d+ vpshufb %xmm9,%xmm6,%xmm6+ shrdl $9,%r14d,%r14d+ xorl %r9d,%r13d+ xorl %r11d,%r12d+ vpaddd %xmm6,%xmm1,%xmm1+ shrdl $5,%r13d,%r13d+ xorl %ebx,%r14d+ andl %r9d,%r12d+ vpaddd 32(%rsi),%xmm1,%xmm6+ xorl %r9d,%r13d+ addl 28(%rsp),%eax+ movl %ebx,%edi+ xorl %r11d,%r12d+ shrdl $11,%r14d,%r14d+ xorl %ecx,%edi+ addl %r12d,%eax+ shrdl $6,%r13d,%r13d+ andl %edi,%r15d+ xorl %ebx,%r14d+ addl %r13d,%eax+ xorl %ecx,%r15d+ shrdl $2,%r14d,%r14d+ addl %eax,%r8d+ addl %r15d,%eax+ movl %r8d,%r13d+ addl %eax,%r14d+ vmovdqa %xmm6,16(%rsp)+ vpalignr $4,%xmm2,%xmm3,%xmm4+ shrdl $14,%r13d,%r13d+ movl %r14d,%eax+ movl %r9d,%r12d+ vpalignr $4,%xmm0,%xmm1,%xmm7+ shrdl $9,%r14d,%r14d+ xorl %r8d,%r13d+ xorl %r10d,%r12d+ vpsrld $7,%xmm4,%xmm6+ shrdl $5,%r13d,%r13d+ xorl %eax,%r14d+ andl %r8d,%r12d+ vpaddd %xmm7,%xmm2,%xmm2+ xorl %r8d,%r13d+ addl 32(%rsp),%r11d+ movl %eax,%r15d+ vpsrld $3,%xmm4,%xmm7+ xorl %r10d,%r12d+ shrdl $11,%r14d,%r14d+ xorl %ebx,%r15d+ vpslld $14,%xmm4,%xmm5+ addl %r12d,%r11d+ shrdl $6,%r13d,%r13d+ andl %r15d,%edi+ vpxor %xmm6,%xmm7,%xmm4+ xorl %eax,%r14d+ addl %r13d,%r11d+ xorl %ebx,%edi+ vpshufd $250,%xmm1,%xmm7+ shrdl $2,%r14d,%r14d+ addl %r11d,%edx+ addl %edi,%r11d+ vpsrld $11,%xmm6,%xmm6+ movl %edx,%r13d+ addl %r11d,%r14d+ shrdl $14,%r13d,%r13d+ vpxor %xmm5,%xmm4,%xmm4+ movl %r14d,%r11d+ movl %r8d,%r12d+ shrdl $9,%r14d,%r14d+ vpslld $11,%xmm5,%xmm5+ xorl %edx,%r13d+ xorl %r9d,%r12d+ shrdl $5,%r13d,%r13d+ vpxor %xmm6,%xmm4,%xmm4+ xorl %r11d,%r14d+ andl %edx,%r12d+ xorl %edx,%r13d+ vpsrld $10,%xmm7,%xmm6+ addl 36(%rsp),%r10d+ movl %r11d,%edi+ xorl %r9d,%r12d+ vpxor %xmm5,%xmm4,%xmm4+ shrdl $11,%r14d,%r14d+ xorl %eax,%edi+ addl %r12d,%r10d+ vpsrlq $17,%xmm7,%xmm7+ shrdl $6,%r13d,%r13d+ andl %edi,%r15d+ xorl %r11d,%r14d+ vpaddd %xmm4,%xmm2,%xmm2+ addl %r13d,%r10d+ xorl %eax,%r15d+ shrdl $2,%r14d,%r14d+ vpxor %xmm7,%xmm6,%xmm6+ addl %r10d,%ecx+ addl %r15d,%r10d+ movl %ecx,%r13d+ vpsrlq $2,%xmm7,%xmm7+ addl %r10d,%r14d+ shrdl $14,%r13d,%r13d+ movl %r14d,%r10d+ vpxor %xmm7,%xmm6,%xmm6+ movl %edx,%r12d+ shrdl $9,%r14d,%r14d+ xorl %ecx,%r13d+ vpshufb %xmm8,%xmm6,%xmm6+ xorl %r8d,%r12d+ shrdl $5,%r13d,%r13d+ xorl %r10d,%r14d+ vpaddd %xmm6,%xmm2,%xmm2+ andl %ecx,%r12d+ xorl %ecx,%r13d+ addl 40(%rsp),%r9d+ vpshufd $80,%xmm2,%xmm7+ movl %r10d,%r15d+ xorl %r8d,%r12d+ shrdl $11,%r14d,%r14d+ vpsrld $10,%xmm7,%xmm6+ xorl %r11d,%r15d+ addl %r12d,%r9d+ shrdl $6,%r13d,%r13d+ vpsrlq $17,%xmm7,%xmm7+ andl %r15d,%edi+ xorl %r10d,%r14d+ addl %r13d,%r9d+ vpxor %xmm7,%xmm6,%xmm6+ xorl %r11d,%edi+ shrdl $2,%r14d,%r14d+ addl %r9d,%ebx+ vpsrlq $2,%xmm7,%xmm7+ addl %edi,%r9d+ movl %ebx,%r13d+ addl %r9d,%r14d+ vpxor %xmm7,%xmm6,%xmm6+ shrdl $14,%r13d,%r13d+ movl %r14d,%r9d+ movl %ecx,%r12d+ vpshufb %xmm9,%xmm6,%xmm6+ shrdl $9,%r14d,%r14d+ xorl %ebx,%r13d+ xorl %edx,%r12d+ vpaddd %xmm6,%xmm2,%xmm2+ shrdl $5,%r13d,%r13d+ xorl %r9d,%r14d+ andl %ebx,%r12d+ vpaddd 64(%rsi),%xmm2,%xmm6+ xorl %ebx,%r13d+ addl 44(%rsp),%r8d+ movl %r9d,%edi+ xorl %edx,%r12d+ shrdl $11,%r14d,%r14d+ xorl %r10d,%edi+ addl %r12d,%r8d+ shrdl $6,%r13d,%r13d+ andl %edi,%r15d+ xorl %r9d,%r14d+ addl %r13d,%r8d+ xorl %r10d,%r15d+ shrdl $2,%r14d,%r14d+ addl %r8d,%eax+ addl %r15d,%r8d+ movl %eax,%r13d+ addl %r8d,%r14d+ vmovdqa %xmm6,32(%rsp)+ vpalignr $4,%xmm3,%xmm0,%xmm4+ shrdl $14,%r13d,%r13d+ movl %r14d,%r8d+ movl %ebx,%r12d+ vpalignr $4,%xmm1,%xmm2,%xmm7+ shrdl $9,%r14d,%r14d+ xorl %eax,%r13d+ xorl %ecx,%r12d+ vpsrld $7,%xmm4,%xmm6+ shrdl $5,%r13d,%r13d+ xorl %r8d,%r14d+ andl %eax,%r12d+ vpaddd %xmm7,%xmm3,%xmm3+ xorl %eax,%r13d+ addl 48(%rsp),%edx+ movl %r8d,%r15d+ vpsrld $3,%xmm4,%xmm7+ xorl %ecx,%r12d+ shrdl $11,%r14d,%r14d+ xorl %r9d,%r15d+ vpslld $14,%xmm4,%xmm5+ addl %r12d,%edx+ shrdl $6,%r13d,%r13d+ andl %r15d,%edi+ vpxor %xmm6,%xmm7,%xmm4+ xorl %r8d,%r14d+ addl %r13d,%edx+ xorl %r9d,%edi+ vpshufd $250,%xmm2,%xmm7+ shrdl $2,%r14d,%r14d+ addl %edx,%r11d+ addl %edi,%edx+ vpsrld $11,%xmm6,%xmm6+ movl %r11d,%r13d+ addl %edx,%r14d+ shrdl $14,%r13d,%r13d+ vpxor %xmm5,%xmm4,%xmm4+ movl %r14d,%edx+ movl %eax,%r12d+ shrdl $9,%r14d,%r14d+ vpslld $11,%xmm5,%xmm5+ xorl %r11d,%r13d+ xorl %ebx,%r12d+ shrdl $5,%r13d,%r13d+ vpxor %xmm6,%xmm4,%xmm4+ xorl %edx,%r14d+ andl %r11d,%r12d+ xorl %r11d,%r13d+ vpsrld $10,%xmm7,%xmm6+ addl 52(%rsp),%ecx+ movl %edx,%edi+ xorl %ebx,%r12d+ vpxor %xmm5,%xmm4,%xmm4+ shrdl $11,%r14d,%r14d+ xorl %r8d,%edi+ addl %r12d,%ecx+ vpsrlq $17,%xmm7,%xmm7+ shrdl $6,%r13d,%r13d+ andl %edi,%r15d+ xorl %edx,%r14d+ vpaddd %xmm4,%xmm3,%xmm3+ addl %r13d,%ecx+ xorl %r8d,%r15d+ shrdl $2,%r14d,%r14d+ vpxor %xmm7,%xmm6,%xmm6+ addl %ecx,%r10d+ addl %r15d,%ecx+ movl %r10d,%r13d+ vpsrlq $2,%xmm7,%xmm7+ addl %ecx,%r14d+ shrdl $14,%r13d,%r13d+ movl %r14d,%ecx+ vpxor %xmm7,%xmm6,%xmm6+ movl %r11d,%r12d+ shrdl $9,%r14d,%r14d+ xorl %r10d,%r13d+ vpshufb %xmm8,%xmm6,%xmm6+ xorl %eax,%r12d+ shrdl $5,%r13d,%r13d+ xorl %ecx,%r14d+ vpaddd %xmm6,%xmm3,%xmm3+ andl %r10d,%r12d+ xorl %r10d,%r13d+ addl 56(%rsp),%ebx+ vpshufd $80,%xmm3,%xmm7+ movl %ecx,%r15d+ xorl %eax,%r12d+ shrdl $11,%r14d,%r14d+ vpsrld $10,%xmm7,%xmm6+ xorl %edx,%r15d+ addl %r12d,%ebx+ shrdl $6,%r13d,%r13d+ vpsrlq $17,%xmm7,%xmm7+ andl %r15d,%edi+ xorl %ecx,%r14d+ addl %r13d,%ebx+ vpxor %xmm7,%xmm6,%xmm6+ xorl %edx,%edi+ shrdl $2,%r14d,%r14d+ addl %ebx,%r9d+ vpsrlq $2,%xmm7,%xmm7+ addl %edi,%ebx+ movl %r9d,%r13d+ addl %ebx,%r14d+ vpxor %xmm7,%xmm6,%xmm6+ shrdl $14,%r13d,%r13d+ movl %r14d,%ebx+ movl %r10d,%r12d+ vpshufb %xmm9,%xmm6,%xmm6+ shrdl $9,%r14d,%r14d+ xorl %r9d,%r13d+ xorl %r11d,%r12d+ vpaddd %xmm6,%xmm3,%xmm3+ shrdl $5,%r13d,%r13d+ xorl %ebx,%r14d+ andl %r9d,%r12d+ vpaddd 96(%rsi),%xmm3,%xmm6+ xorl %r9d,%r13d+ addl 60(%rsp),%eax+ movl %ebx,%edi+ xorl %r11d,%r12d+ shrdl $11,%r14d,%r14d+ xorl %ecx,%edi+ addl %r12d,%eax+ shrdl $6,%r13d,%r13d+ andl %edi,%r15d+ xorl %ebx,%r14d+ addl %r13d,%eax+ xorl %ecx,%r15d+ shrdl $2,%r14d,%r14d+ addl %eax,%r8d+ addl %r15d,%eax+ movl %r8d,%r13d+ addl %eax,%r14d+ vmovdqa %xmm6,48(%rsp)+ cmpb $0,131(%rsi)+ jne L$avx_00_47+ shrdl $14,%r13d,%r13d+ movl %r14d,%eax+ movl %r9d,%r12d+ shrdl $9,%r14d,%r14d+ xorl %r8d,%r13d+ xorl %r10d,%r12d+ shrdl $5,%r13d,%r13d+ xorl %eax,%r14d+ andl %r8d,%r12d+ xorl %r8d,%r13d+ addl 0(%rsp),%r11d+ movl %eax,%r15d+ xorl %r10d,%r12d+ shrdl $11,%r14d,%r14d+ xorl %ebx,%r15d+ addl %r12d,%r11d+ shrdl $6,%r13d,%r13d+ andl %r15d,%edi+ xorl %eax,%r14d+ addl %r13d,%r11d+ xorl %ebx,%edi+ shrdl $2,%r14d,%r14d+ addl %r11d,%edx+ addl %edi,%r11d+ movl %edx,%r13d+ addl %r11d,%r14d+ shrdl $14,%r13d,%r13d+ movl %r14d,%r11d+ movl %r8d,%r12d+ shrdl $9,%r14d,%r14d+ xorl %edx,%r13d+ xorl %r9d,%r12d+ shrdl $5,%r13d,%r13d+ xorl %r11d,%r14d+ andl %edx,%r12d+ xorl %edx,%r13d+ addl 4(%rsp),%r10d+ movl %r11d,%edi+ xorl %r9d,%r12d+ shrdl $11,%r14d,%r14d+ xorl %eax,%edi+ addl %r12d,%r10d+ shrdl $6,%r13d,%r13d+ andl %edi,%r15d+ xorl %r11d,%r14d+ addl %r13d,%r10d+ xorl %eax,%r15d+ shrdl $2,%r14d,%r14d+ addl %r10d,%ecx+ addl %r15d,%r10d+ movl %ecx,%r13d+ addl %r10d,%r14d+ shrdl $14,%r13d,%r13d+ movl %r14d,%r10d+ movl %edx,%r12d+ shrdl $9,%r14d,%r14d+ xorl %ecx,%r13d+ xorl %r8d,%r12d+ shrdl $5,%r13d,%r13d+ xorl %r10d,%r14d+ andl %ecx,%r12d+ xorl %ecx,%r13d+ addl 8(%rsp),%r9d+ movl %r10d,%r15d+ xorl %r8d,%r12d+ shrdl $11,%r14d,%r14d+ xorl %r11d,%r15d+ addl %r12d,%r9d+ shrdl $6,%r13d,%r13d+ andl %r15d,%edi+ xorl %r10d,%r14d+ addl %r13d,%r9d+ xorl %r11d,%edi+ shrdl $2,%r14d,%r14d+ addl %r9d,%ebx+ addl %edi,%r9d+ movl %ebx,%r13d+ addl %r9d,%r14d+ shrdl $14,%r13d,%r13d+ movl %r14d,%r9d+ movl %ecx,%r12d+ shrdl $9,%r14d,%r14d+ xorl %ebx,%r13d+ xorl %edx,%r12d+ shrdl $5,%r13d,%r13d+ xorl %r9d,%r14d+ andl %ebx,%r12d+ xorl %ebx,%r13d+ addl 12(%rsp),%r8d+ movl %r9d,%edi+ xorl %edx,%r12d+ shrdl $11,%r14d,%r14d+ xorl %r10d,%edi+ addl %r12d,%r8d+ shrdl $6,%r13d,%r13d+ andl %edi,%r15d+ xorl %r9d,%r14d+ addl %r13d,%r8d+ xorl %r10d,%r15d+ shrdl $2,%r14d,%r14d+ addl %r8d,%eax+ addl %r15d,%r8d+ movl %eax,%r13d+ addl %r8d,%r14d+ shrdl $14,%r13d,%r13d+ movl %r14d,%r8d+ movl %ebx,%r12d+ shrdl $9,%r14d,%r14d+ xorl %eax,%r13d+ xorl %ecx,%r12d+ shrdl $5,%r13d,%r13d+ xorl %r8d,%r14d+ andl %eax,%r12d+ xorl %eax,%r13d+ addl 16(%rsp),%edx+ movl %r8d,%r15d+ xorl %ecx,%r12d+ shrdl $11,%r14d,%r14d+ xorl %r9d,%r15d+ addl %r12d,%edx+ shrdl $6,%r13d,%r13d+ andl %r15d,%edi+ xorl %r8d,%r14d+ addl %r13d,%edx+ xorl %r9d,%edi+ shrdl $2,%r14d,%r14d+ addl %edx,%r11d+ addl %edi,%edx+ movl %r11d,%r13d+ addl %edx,%r14d+ shrdl $14,%r13d,%r13d+ movl %r14d,%edx+ movl %eax,%r12d+ shrdl $9,%r14d,%r14d+ xorl %r11d,%r13d+ xorl %ebx,%r12d+ shrdl $5,%r13d,%r13d+ xorl %edx,%r14d+ andl %r11d,%r12d+ xorl %r11d,%r13d+ addl 20(%rsp),%ecx+ movl %edx,%edi+ xorl %ebx,%r12d+ shrdl $11,%r14d,%r14d+ xorl %r8d,%edi+ addl %r12d,%ecx+ shrdl $6,%r13d,%r13d+ andl %edi,%r15d+ xorl %edx,%r14d+ addl %r13d,%ecx+ xorl %r8d,%r15d+ shrdl $2,%r14d,%r14d+ addl %ecx,%r10d+ addl %r15d,%ecx+ movl %r10d,%r13d+ addl %ecx,%r14d+ shrdl $14,%r13d,%r13d+ movl %r14d,%ecx+ movl %r11d,%r12d+ shrdl $9,%r14d,%r14d+ xorl %r10d,%r13d+ xorl %eax,%r12d+ shrdl $5,%r13d,%r13d+ xorl %ecx,%r14d+ andl %r10d,%r12d+ xorl %r10d,%r13d+ addl 24(%rsp),%ebx+ movl %ecx,%r15d+ xorl %eax,%r12d+ shrdl $11,%r14d,%r14d+ xorl %edx,%r15d+ addl %r12d,%ebx+ shrdl $6,%r13d,%r13d+ andl %r15d,%edi+ xorl %ecx,%r14d+ addl %r13d,%ebx+ xorl %edx,%edi+ shrdl $2,%r14d,%r14d+ addl %ebx,%r9d+ addl %edi,%ebx+ movl %r9d,%r13d+ addl %ebx,%r14d+ shrdl $14,%r13d,%r13d+ movl %r14d,%ebx+ movl %r10d,%r12d+ shrdl $9,%r14d,%r14d+ xorl %r9d,%r13d+ xorl %r11d,%r12d+ shrdl $5,%r13d,%r13d+ xorl %ebx,%r14d+ andl %r9d,%r12d+ xorl %r9d,%r13d+ addl 28(%rsp),%eax+ movl %ebx,%edi+ xorl %r11d,%r12d+ shrdl $11,%r14d,%r14d+ xorl %ecx,%edi+ addl %r12d,%eax+ shrdl $6,%r13d,%r13d+ andl %edi,%r15d+ xorl %ebx,%r14d+ addl %r13d,%eax+ xorl %ecx,%r15d+ shrdl $2,%r14d,%r14d+ addl %eax,%r8d+ addl %r15d,%eax+ movl %r8d,%r13d+ addl %eax,%r14d+ shrdl $14,%r13d,%r13d+ movl %r14d,%eax+ movl %r9d,%r12d+ shrdl $9,%r14d,%r14d+ xorl %r8d,%r13d+ xorl %r10d,%r12d+ shrdl $5,%r13d,%r13d+ xorl %eax,%r14d+ andl %r8d,%r12d+ xorl %r8d,%r13d+ addl 32(%rsp),%r11d+ movl %eax,%r15d+ xorl %r10d,%r12d+ shrdl $11,%r14d,%r14d+ xorl %ebx,%r15d+ addl %r12d,%r11d+ shrdl $6,%r13d,%r13d+ andl %r15d,%edi+ xorl %eax,%r14d+ addl %r13d,%r11d+ xorl %ebx,%edi+ shrdl $2,%r14d,%r14d+ addl %r11d,%edx+ addl %edi,%r11d+ movl %edx,%r13d+ addl %r11d,%r14d+ shrdl $14,%r13d,%r13d+ movl %r14d,%r11d+ movl %r8d,%r12d+ shrdl $9,%r14d,%r14d+ xorl %edx,%r13d+ xorl %r9d,%r12d+ shrdl $5,%r13d,%r13d+ xorl %r11d,%r14d+ andl %edx,%r12d+ xorl %edx,%r13d+ addl 36(%rsp),%r10d+ movl %r11d,%edi+ xorl %r9d,%r12d+ shrdl $11,%r14d,%r14d+ xorl %eax,%edi+ addl %r12d,%r10d+ shrdl $6,%r13d,%r13d+ andl %edi,%r15d+ xorl %r11d,%r14d+ addl %r13d,%r10d+ xorl %eax,%r15d+ shrdl $2,%r14d,%r14d+ addl %r10d,%ecx+ addl %r15d,%r10d+ movl %ecx,%r13d+ addl %r10d,%r14d+ shrdl $14,%r13d,%r13d+ movl %r14d,%r10d+ movl %edx,%r12d+ shrdl $9,%r14d,%r14d+ xorl %ecx,%r13d+ xorl %r8d,%r12d+ shrdl $5,%r13d,%r13d+ xorl %r10d,%r14d+ andl %ecx,%r12d+ xorl %ecx,%r13d+ addl 40(%rsp),%r9d+ movl %r10d,%r15d+ xorl %r8d,%r12d+ shrdl $11,%r14d,%r14d+ xorl %r11d,%r15d+ addl %r12d,%r9d+ shrdl $6,%r13d,%r13d+ andl %r15d,%edi+ xorl %r10d,%r14d+ addl %r13d,%r9d+ xorl %r11d,%edi+ shrdl $2,%r14d,%r14d+ addl %r9d,%ebx+ addl %edi,%r9d+ movl %ebx,%r13d+ addl %r9d,%r14d+ shrdl $14,%r13d,%r13d+ movl %r14d,%r9d+ movl %ecx,%r12d+ shrdl $9,%r14d,%r14d+ xorl %ebx,%r13d+ xorl %edx,%r12d+ shrdl $5,%r13d,%r13d+ xorl %r9d,%r14d+ andl %ebx,%r12d+ xorl %ebx,%r13d+ addl 44(%rsp),%r8d+ movl %r9d,%edi+ xorl %edx,%r12d+ shrdl $11,%r14d,%r14d+ xorl %r10d,%edi+ addl %r12d,%r8d+ shrdl $6,%r13d,%r13d+ andl %edi,%r15d+ xorl %r9d,%r14d+ addl %r13d,%r8d+ xorl %r10d,%r15d+ shrdl $2,%r14d,%r14d+ addl %r8d,%eax+ addl %r15d,%r8d+ movl %eax,%r13d+ addl %r8d,%r14d+ shrdl $14,%r13d,%r13d+ movl %r14d,%r8d+ movl %ebx,%r12d+ shrdl $9,%r14d,%r14d+ xorl %eax,%r13d+ xorl %ecx,%r12d+ shrdl $5,%r13d,%r13d+ xorl %r8d,%r14d+ andl %eax,%r12d+ xorl %eax,%r13d+ addl 48(%rsp),%edx+ movl %r8d,%r15d+ xorl %ecx,%r12d+ shrdl $11,%r14d,%r14d+ xorl %r9d,%r15d+ addl %r12d,%edx+ shrdl $6,%r13d,%r13d+ andl %r15d,%edi+ xorl %r8d,%r14d+ addl %r13d,%edx+ xorl %r9d,%edi+ shrdl $2,%r14d,%r14d+ addl %edx,%r11d+ addl %edi,%edx+ movl %r11d,%r13d+ addl %edx,%r14d+ shrdl $14,%r13d,%r13d+ movl %r14d,%edx+ movl %eax,%r12d+ shrdl $9,%r14d,%r14d+ xorl %r11d,%r13d+ xorl %ebx,%r12d+ shrdl $5,%r13d,%r13d+ xorl %edx,%r14d+ andl %r11d,%r12d+ xorl %r11d,%r13d+ addl 52(%rsp),%ecx+ movl %edx,%edi+ xorl %ebx,%r12d+ shrdl $11,%r14d,%r14d+ xorl %r8d,%edi+ addl %r12d,%ecx+ shrdl $6,%r13d,%r13d+ andl %edi,%r15d+ xorl %edx,%r14d+ addl %r13d,%ecx+ xorl %r8d,%r15d+ shrdl $2,%r14d,%r14d+ addl %ecx,%r10d+ addl %r15d,%ecx+ movl %r10d,%r13d+ addl %ecx,%r14d+ shrdl $14,%r13d,%r13d+ movl %r14d,%ecx+ movl %r11d,%r12d+ shrdl $9,%r14d,%r14d+ xorl %r10d,%r13d+ xorl %eax,%r12d+ shrdl $5,%r13d,%r13d+ xorl %ecx,%r14d+ andl %r10d,%r12d+ xorl %r10d,%r13d+ addl 56(%rsp),%ebx+ movl %ecx,%r15d+ xorl %eax,%r12d+ shrdl $11,%r14d,%r14d+ xorl %edx,%r15d+ addl %r12d,%ebx+ shrdl $6,%r13d,%r13d+ andl %r15d,%edi+ xorl %ecx,%r14d+ addl %r13d,%ebx+ xorl %edx,%edi+ shrdl $2,%r14d,%r14d+ addl %ebx,%r9d+ addl %edi,%ebx+ movl %r9d,%r13d+ addl %ebx,%r14d+ shrdl $14,%r13d,%r13d+ movl %r14d,%ebx+ movl %r10d,%r12d+ shrdl $9,%r14d,%r14d+ xorl %r9d,%r13d+ xorl %r11d,%r12d+ shrdl $5,%r13d,%r13d+ xorl %ebx,%r14d+ andl %r9d,%r12d+ xorl %r9d,%r13d+ addl 60(%rsp),%eax+ movl %ebx,%edi+ xorl %r11d,%r12d+ shrdl $11,%r14d,%r14d+ xorl %ecx,%edi+ addl %r12d,%eax+ shrdl $6,%r13d,%r13d+ andl %edi,%r15d+ xorl %ebx,%r14d+ addl %r13d,%eax+ xorl %ecx,%r15d+ shrdl $2,%r14d,%r14d+ addl %eax,%r8d+ addl %r15d,%eax+ movl %r8d,%r13d+ addl %eax,%r14d+ movq -64(%rbp),%rdi+ movl %r14d,%eax+ movq -56(%rbp),%rsi++ addl 0(%rdi),%eax+ addl 4(%rdi),%ebx+ addl 8(%rdi),%ecx+ addl 12(%rdi),%edx+ addl 16(%rdi),%r8d+ addl 20(%rdi),%r9d+ addl 24(%rdi),%r10d+ addl 28(%rdi),%r11d++ leaq 64(%rsi),%rsi+ cmpq -48(%rbp),%rsi++ movl %eax,0(%rdi)+ movl %ebx,4(%rdi)+ movl %ecx,8(%rdi)+ movl %edx,12(%rdi)+ movl %r8d,16(%rdi)+ movl %r9d,20(%rdi)+ movl %r10d,24(%rdi)+ movl %r11d,28(%rdi)+ jb L$loop_avx++ vzeroupper+ movq -40(%rbp),%r15+ movq -32(%rbp),%r14+ movq -24(%rbp),%r13+ movq -16(%rbp),%r12+ movq -8(%rbp),%rbx+ movq %rbp,%rsp+.cfi_def_cfa_register %rsp+ popq %rbp+.cfi_adjust_cfa_offset -8+.cfi_restore %rbp+.cfi_restore %r12+.cfi_restore %r13+.cfi_restore %r14+.cfi_restore %r15+.cfi_restore %rbx+ .byte 0xf3,0xc3+.cfi_endproc +++.p2align 6+crypton_sha256_asm_block_data_order_avx2:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa+++ pushq %rbp+.cfi_adjust_cfa_offset 8+.cfi_offset %rbp,-16+ movq %rsp,%rbp+.cfi_def_cfa_register %rbp+L$avx2_shortcut:+ pushq %rbx+.cfi_offset %rbx,-24+ pushq %r12+.cfi_offset %r12,-32+ pushq %r13+.cfi_offset %r13,-40+ pushq %r14+.cfi_offset %r14,-48+ pushq %r15+.cfi_offset %r15,-56+ shlq $4,%rdx+ subq $24,%rsp++ leaq (%rsi,%rdx,4),%rdx+ movq %rdi,-64(%rbp)++ movq %rdx,-48(%rbp)+++ leaq -64(%rsp),%rsp+ vzeroupper+ andq $-64,%rsp+ subq $-64,%rsi+ movl 0(%rdi),%eax+ movq %rsi,%r12+ movl 4(%rdi),%ebx+ cmpq %rdx,%rsi+ movl 8(%rdi),%ecx+ cmoveq %rsp,%r12+ movl 12(%rdi),%edx+ movl 16(%rdi),%r8d+ movl 20(%rdi),%r9d+ movl 24(%rdi),%r10d+ movl 28(%rdi),%r11d+ vmovdqa K256+512+32(%rip),%ymm8+ vmovdqa K256+512+64(%rip),%ymm9+ jmp L$oop_avx2+.p2align 4+L$oop_avx2:+ vmovdqa K256+512(%rip),%ymm7+ movq %rsi,-56(%rbp)+ vmovdqu -64+0(%rsi),%xmm0+ vmovdqu -64+16(%rsi),%xmm1+ vmovdqu -64+32(%rsi),%xmm2+ vmovdqu -64+48(%rsi),%xmm3+ leaq K256(%rip),%rsi+ vinserti128 $1,(%r12),%ymm0,%ymm0+ vinserti128 $1,16(%r12),%ymm1,%ymm1+ vpshufb %ymm7,%ymm0,%ymm0+ vinserti128 $1,32(%r12),%ymm2,%ymm2+ vpshufb %ymm7,%ymm1,%ymm1+ vinserti128 $1,48(%r12),%ymm3,%ymm3++ vpshufb %ymm7,%ymm2,%ymm2+ vpaddd 0(%rsi),%ymm0,%ymm4+ vpshufb %ymm7,%ymm3,%ymm3+ vpaddd 32(%rsi),%ymm1,%ymm5+ vpaddd 64(%rsi),%ymm2,%ymm6+ vpaddd 96(%rsi),%ymm3,%ymm7+ vmovdqa %ymm4,0(%rsp)+ xorl %r14d,%r14d+ vmovdqa %ymm5,32(%rsp)+ leaq -64(%rsp),%rsp+ movl %ebx,%edi+ vmovdqa %ymm6,0(%rsp)+ xorl %ecx,%edi+ vmovdqa %ymm7,32(%rsp)+ movl %r9d,%r12d+ subq $-32*4,%rsi+ jmp L$avx2_00_47++.p2align 4+L$avx2_00_47:+ leaq -64(%rsp),%rsp+ vpalignr $4,%ymm0,%ymm1,%ymm4+ addl 0+128(%rsp),%r11d+ andl %r8d,%r12d+ rorxl $25,%r8d,%r13d+ vpalignr $4,%ymm2,%ymm3,%ymm7+ rorxl $11,%r8d,%r15d+ leal (%rax,%r14,1),%eax+ leal (%r11,%r12,1),%r11d+ vpsrld $7,%ymm4,%ymm6+ andnl %r10d,%r8d,%r12d+ xorl %r15d,%r13d+ rorxl $6,%r8d,%r14d+ vpaddd %ymm7,%ymm0,%ymm0+ leal (%r11,%r12,1),%r11d+ xorl %r14d,%r13d+ movl %eax,%r15d+ vpsrld $3,%ymm4,%ymm7+ rorxl $22,%eax,%r12d+ leal (%r11,%r13,1),%r11d+ xorl %ebx,%r15d+ vpslld $14,%ymm4,%ymm5+ rorxl $13,%eax,%r14d+ rorxl $2,%eax,%r13d+ leal (%rdx,%r11,1),%edx+ vpxor %ymm6,%ymm7,%ymm4+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %ebx,%edi+ vpshufd $250,%ymm3,%ymm7+ xorl %r13d,%r14d+ leal (%r11,%rdi,1),%r11d+ movl %r8d,%r12d+ vpsrld $11,%ymm6,%ymm6+ addl 4+128(%rsp),%r10d+ andl %edx,%r12d+ rorxl $25,%edx,%r13d+ vpxor %ymm5,%ymm4,%ymm4+ rorxl $11,%edx,%edi+ leal (%r11,%r14,1),%r11d+ leal (%r10,%r12,1),%r10d+ vpslld $11,%ymm5,%ymm5+ andnl %r9d,%edx,%r12d+ xorl %edi,%r13d+ rorxl $6,%edx,%r14d+ vpxor %ymm6,%ymm4,%ymm4+ leal (%r10,%r12,1),%r10d+ xorl %r14d,%r13d+ movl %r11d,%edi+ vpsrld $10,%ymm7,%ymm6+ rorxl $22,%r11d,%r12d+ leal (%r10,%r13,1),%r10d+ xorl %eax,%edi+ vpxor %ymm5,%ymm4,%ymm4+ rorxl $13,%r11d,%r14d+ rorxl $2,%r11d,%r13d+ leal (%rcx,%r10,1),%ecx+ vpsrlq $17,%ymm7,%ymm7+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %eax,%r15d+ vpaddd %ymm4,%ymm0,%ymm0+ xorl %r13d,%r14d+ leal (%r10,%r15,1),%r10d+ movl %edx,%r12d+ vpxor %ymm7,%ymm6,%ymm6+ addl 8+128(%rsp),%r9d+ andl %ecx,%r12d+ rorxl $25,%ecx,%r13d+ vpsrlq $2,%ymm7,%ymm7+ rorxl $11,%ecx,%r15d+ leal (%r10,%r14,1),%r10d+ leal (%r9,%r12,1),%r9d+ vpxor %ymm7,%ymm6,%ymm6+ andnl %r8d,%ecx,%r12d+ xorl %r15d,%r13d+ rorxl $6,%ecx,%r14d+ vpshufb %ymm8,%ymm6,%ymm6+ leal (%r9,%r12,1),%r9d+ xorl %r14d,%r13d+ movl %r10d,%r15d+ vpaddd %ymm6,%ymm0,%ymm0+ rorxl $22,%r10d,%r12d+ leal (%r9,%r13,1),%r9d+ xorl %r11d,%r15d+ vpshufd $80,%ymm0,%ymm7+ rorxl $13,%r10d,%r14d+ rorxl $2,%r10d,%r13d+ leal (%rbx,%r9,1),%ebx+ vpsrld $10,%ymm7,%ymm6+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %r11d,%edi+ vpsrlq $17,%ymm7,%ymm7+ xorl %r13d,%r14d+ leal (%r9,%rdi,1),%r9d+ movl %ecx,%r12d+ vpxor %ymm7,%ymm6,%ymm6+ addl 12+128(%rsp),%r8d+ andl %ebx,%r12d+ rorxl $25,%ebx,%r13d+ vpsrlq $2,%ymm7,%ymm7+ rorxl $11,%ebx,%edi+ leal (%r9,%r14,1),%r9d+ leal (%r8,%r12,1),%r8d+ vpxor %ymm7,%ymm6,%ymm6+ andnl %edx,%ebx,%r12d+ xorl %edi,%r13d+ rorxl $6,%ebx,%r14d+ vpshufb %ymm9,%ymm6,%ymm6+ leal (%r8,%r12,1),%r8d+ xorl %r14d,%r13d+ movl %r9d,%edi+ vpaddd %ymm6,%ymm0,%ymm0+ rorxl $22,%r9d,%r12d+ leal (%r8,%r13,1),%r8d+ xorl %r10d,%edi+ vpaddd 0(%rsi),%ymm0,%ymm6+ rorxl $13,%r9d,%r14d+ rorxl $2,%r9d,%r13d+ leal (%rax,%r8,1),%eax+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %r10d,%r15d+ xorl %r13d,%r14d+ leal (%r8,%r15,1),%r8d+ movl %ebx,%r12d+ vmovdqa %ymm6,0(%rsp)+ vpalignr $4,%ymm1,%ymm2,%ymm4+ addl 32+128(%rsp),%edx+ andl %eax,%r12d+ rorxl $25,%eax,%r13d+ vpalignr $4,%ymm3,%ymm0,%ymm7+ rorxl $11,%eax,%r15d+ leal (%r8,%r14,1),%r8d+ leal (%rdx,%r12,1),%edx+ vpsrld $7,%ymm4,%ymm6+ andnl %ecx,%eax,%r12d+ xorl %r15d,%r13d+ rorxl $6,%eax,%r14d+ vpaddd %ymm7,%ymm1,%ymm1+ leal (%rdx,%r12,1),%edx+ xorl %r14d,%r13d+ movl %r8d,%r15d+ vpsrld $3,%ymm4,%ymm7+ rorxl $22,%r8d,%r12d+ leal (%rdx,%r13,1),%edx+ xorl %r9d,%r15d+ vpslld $14,%ymm4,%ymm5+ rorxl $13,%r8d,%r14d+ rorxl $2,%r8d,%r13d+ leal (%r11,%rdx,1),%r11d+ vpxor %ymm6,%ymm7,%ymm4+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %r9d,%edi+ vpshufd $250,%ymm0,%ymm7+ xorl %r13d,%r14d+ leal (%rdx,%rdi,1),%edx+ movl %eax,%r12d+ vpsrld $11,%ymm6,%ymm6+ addl 36+128(%rsp),%ecx+ andl %r11d,%r12d+ rorxl $25,%r11d,%r13d+ vpxor %ymm5,%ymm4,%ymm4+ rorxl $11,%r11d,%edi+ leal (%rdx,%r14,1),%edx+ leal (%rcx,%r12,1),%ecx+ vpslld $11,%ymm5,%ymm5+ andnl %ebx,%r11d,%r12d+ xorl %edi,%r13d+ rorxl $6,%r11d,%r14d+ vpxor %ymm6,%ymm4,%ymm4+ leal (%rcx,%r12,1),%ecx+ xorl %r14d,%r13d+ movl %edx,%edi+ vpsrld $10,%ymm7,%ymm6+ rorxl $22,%edx,%r12d+ leal (%rcx,%r13,1),%ecx+ xorl %r8d,%edi+ vpxor %ymm5,%ymm4,%ymm4+ rorxl $13,%edx,%r14d+ rorxl $2,%edx,%r13d+ leal (%r10,%rcx,1),%r10d+ vpsrlq $17,%ymm7,%ymm7+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %r8d,%r15d+ vpaddd %ymm4,%ymm1,%ymm1+ xorl %r13d,%r14d+ leal (%rcx,%r15,1),%ecx+ movl %r11d,%r12d+ vpxor %ymm7,%ymm6,%ymm6+ addl 40+128(%rsp),%ebx+ andl %r10d,%r12d+ rorxl $25,%r10d,%r13d+ vpsrlq $2,%ymm7,%ymm7+ rorxl $11,%r10d,%r15d+ leal (%rcx,%r14,1),%ecx+ leal (%rbx,%r12,1),%ebx+ vpxor %ymm7,%ymm6,%ymm6+ andnl %eax,%r10d,%r12d+ xorl %r15d,%r13d+ rorxl $6,%r10d,%r14d+ vpshufb %ymm8,%ymm6,%ymm6+ leal (%rbx,%r12,1),%ebx+ xorl %r14d,%r13d+ movl %ecx,%r15d+ vpaddd %ymm6,%ymm1,%ymm1+ rorxl $22,%ecx,%r12d+ leal (%rbx,%r13,1),%ebx+ xorl %edx,%r15d+ vpshufd $80,%ymm1,%ymm7+ rorxl $13,%ecx,%r14d+ rorxl $2,%ecx,%r13d+ leal (%r9,%rbx,1),%r9d+ vpsrld $10,%ymm7,%ymm6+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %edx,%edi+ vpsrlq $17,%ymm7,%ymm7+ xorl %r13d,%r14d+ leal (%rbx,%rdi,1),%ebx+ movl %r10d,%r12d+ vpxor %ymm7,%ymm6,%ymm6+ addl 44+128(%rsp),%eax+ andl %r9d,%r12d+ rorxl $25,%r9d,%r13d+ vpsrlq $2,%ymm7,%ymm7+ rorxl $11,%r9d,%edi+ leal (%rbx,%r14,1),%ebx+ leal (%rax,%r12,1),%eax+ vpxor %ymm7,%ymm6,%ymm6+ andnl %r11d,%r9d,%r12d+ xorl %edi,%r13d+ rorxl $6,%r9d,%r14d+ vpshufb %ymm9,%ymm6,%ymm6+ leal (%rax,%r12,1),%eax+ xorl %r14d,%r13d+ movl %ebx,%edi+ vpaddd %ymm6,%ymm1,%ymm1+ rorxl $22,%ebx,%r12d+ leal (%rax,%r13,1),%eax+ xorl %ecx,%edi+ vpaddd 32(%rsi),%ymm1,%ymm6+ rorxl $13,%ebx,%r14d+ rorxl $2,%ebx,%r13d+ leal (%r8,%rax,1),%r8d+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %ecx,%r15d+ xorl %r13d,%r14d+ leal (%rax,%r15,1),%eax+ movl %r9d,%r12d+ vmovdqa %ymm6,32(%rsp)+ leaq -64(%rsp),%rsp+ vpalignr $4,%ymm2,%ymm3,%ymm4+ addl 0+128(%rsp),%r11d+ andl %r8d,%r12d+ rorxl $25,%r8d,%r13d+ vpalignr $4,%ymm0,%ymm1,%ymm7+ rorxl $11,%r8d,%r15d+ leal (%rax,%r14,1),%eax+ leal (%r11,%r12,1),%r11d+ vpsrld $7,%ymm4,%ymm6+ andnl %r10d,%r8d,%r12d+ xorl %r15d,%r13d+ rorxl $6,%r8d,%r14d+ vpaddd %ymm7,%ymm2,%ymm2+ leal (%r11,%r12,1),%r11d+ xorl %r14d,%r13d+ movl %eax,%r15d+ vpsrld $3,%ymm4,%ymm7+ rorxl $22,%eax,%r12d+ leal (%r11,%r13,1),%r11d+ xorl %ebx,%r15d+ vpslld $14,%ymm4,%ymm5+ rorxl $13,%eax,%r14d+ rorxl $2,%eax,%r13d+ leal (%rdx,%r11,1),%edx+ vpxor %ymm6,%ymm7,%ymm4+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %ebx,%edi+ vpshufd $250,%ymm1,%ymm7+ xorl %r13d,%r14d+ leal (%r11,%rdi,1),%r11d+ movl %r8d,%r12d+ vpsrld $11,%ymm6,%ymm6+ addl 4+128(%rsp),%r10d+ andl %edx,%r12d+ rorxl $25,%edx,%r13d+ vpxor %ymm5,%ymm4,%ymm4+ rorxl $11,%edx,%edi+ leal (%r11,%r14,1),%r11d+ leal (%r10,%r12,1),%r10d+ vpslld $11,%ymm5,%ymm5+ andnl %r9d,%edx,%r12d+ xorl %edi,%r13d+ rorxl $6,%edx,%r14d+ vpxor %ymm6,%ymm4,%ymm4+ leal (%r10,%r12,1),%r10d+ xorl %r14d,%r13d+ movl %r11d,%edi+ vpsrld $10,%ymm7,%ymm6+ rorxl $22,%r11d,%r12d+ leal (%r10,%r13,1),%r10d+ xorl %eax,%edi+ vpxor %ymm5,%ymm4,%ymm4+ rorxl $13,%r11d,%r14d+ rorxl $2,%r11d,%r13d+ leal (%rcx,%r10,1),%ecx+ vpsrlq $17,%ymm7,%ymm7+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %eax,%r15d+ vpaddd %ymm4,%ymm2,%ymm2+ xorl %r13d,%r14d+ leal (%r10,%r15,1),%r10d+ movl %edx,%r12d+ vpxor %ymm7,%ymm6,%ymm6+ addl 8+128(%rsp),%r9d+ andl %ecx,%r12d+ rorxl $25,%ecx,%r13d+ vpsrlq $2,%ymm7,%ymm7+ rorxl $11,%ecx,%r15d+ leal (%r10,%r14,1),%r10d+ leal (%r9,%r12,1),%r9d+ vpxor %ymm7,%ymm6,%ymm6+ andnl %r8d,%ecx,%r12d+ xorl %r15d,%r13d+ rorxl $6,%ecx,%r14d+ vpshufb %ymm8,%ymm6,%ymm6+ leal (%r9,%r12,1),%r9d+ xorl %r14d,%r13d+ movl %r10d,%r15d+ vpaddd %ymm6,%ymm2,%ymm2+ rorxl $22,%r10d,%r12d+ leal (%r9,%r13,1),%r9d+ xorl %r11d,%r15d+ vpshufd $80,%ymm2,%ymm7+ rorxl $13,%r10d,%r14d+ rorxl $2,%r10d,%r13d+ leal (%rbx,%r9,1),%ebx+ vpsrld $10,%ymm7,%ymm6+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %r11d,%edi+ vpsrlq $17,%ymm7,%ymm7+ xorl %r13d,%r14d+ leal (%r9,%rdi,1),%r9d+ movl %ecx,%r12d+ vpxor %ymm7,%ymm6,%ymm6+ addl 12+128(%rsp),%r8d+ andl %ebx,%r12d+ rorxl $25,%ebx,%r13d+ vpsrlq $2,%ymm7,%ymm7+ rorxl $11,%ebx,%edi+ leal (%r9,%r14,1),%r9d+ leal (%r8,%r12,1),%r8d+ vpxor %ymm7,%ymm6,%ymm6+ andnl %edx,%ebx,%r12d+ xorl %edi,%r13d+ rorxl $6,%ebx,%r14d+ vpshufb %ymm9,%ymm6,%ymm6+ leal (%r8,%r12,1),%r8d+ xorl %r14d,%r13d+ movl %r9d,%edi+ vpaddd %ymm6,%ymm2,%ymm2+ rorxl $22,%r9d,%r12d+ leal (%r8,%r13,1),%r8d+ xorl %r10d,%edi+ vpaddd 64(%rsi),%ymm2,%ymm6+ rorxl $13,%r9d,%r14d+ rorxl $2,%r9d,%r13d+ leal (%rax,%r8,1),%eax+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %r10d,%r15d+ xorl %r13d,%r14d+ leal (%r8,%r15,1),%r8d+ movl %ebx,%r12d+ vmovdqa %ymm6,0(%rsp)+ vpalignr $4,%ymm3,%ymm0,%ymm4+ addl 32+128(%rsp),%edx+ andl %eax,%r12d+ rorxl $25,%eax,%r13d+ vpalignr $4,%ymm1,%ymm2,%ymm7+ rorxl $11,%eax,%r15d+ leal (%r8,%r14,1),%r8d+ leal (%rdx,%r12,1),%edx+ vpsrld $7,%ymm4,%ymm6+ andnl %ecx,%eax,%r12d+ xorl %r15d,%r13d+ rorxl $6,%eax,%r14d+ vpaddd %ymm7,%ymm3,%ymm3+ leal (%rdx,%r12,1),%edx+ xorl %r14d,%r13d+ movl %r8d,%r15d+ vpsrld $3,%ymm4,%ymm7+ rorxl $22,%r8d,%r12d+ leal (%rdx,%r13,1),%edx+ xorl %r9d,%r15d+ vpslld $14,%ymm4,%ymm5+ rorxl $13,%r8d,%r14d+ rorxl $2,%r8d,%r13d+ leal (%r11,%rdx,1),%r11d+ vpxor %ymm6,%ymm7,%ymm4+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %r9d,%edi+ vpshufd $250,%ymm2,%ymm7+ xorl %r13d,%r14d+ leal (%rdx,%rdi,1),%edx+ movl %eax,%r12d+ vpsrld $11,%ymm6,%ymm6+ addl 36+128(%rsp),%ecx+ andl %r11d,%r12d+ rorxl $25,%r11d,%r13d+ vpxor %ymm5,%ymm4,%ymm4+ rorxl $11,%r11d,%edi+ leal (%rdx,%r14,1),%edx+ leal (%rcx,%r12,1),%ecx+ vpslld $11,%ymm5,%ymm5+ andnl %ebx,%r11d,%r12d+ xorl %edi,%r13d+ rorxl $6,%r11d,%r14d+ vpxor %ymm6,%ymm4,%ymm4+ leal (%rcx,%r12,1),%ecx+ xorl %r14d,%r13d+ movl %edx,%edi+ vpsrld $10,%ymm7,%ymm6+ rorxl $22,%edx,%r12d+ leal (%rcx,%r13,1),%ecx+ xorl %r8d,%edi+ vpxor %ymm5,%ymm4,%ymm4+ rorxl $13,%edx,%r14d+ rorxl $2,%edx,%r13d+ leal (%r10,%rcx,1),%r10d+ vpsrlq $17,%ymm7,%ymm7+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %r8d,%r15d+ vpaddd %ymm4,%ymm3,%ymm3+ xorl %r13d,%r14d+ leal (%rcx,%r15,1),%ecx+ movl %r11d,%r12d+ vpxor %ymm7,%ymm6,%ymm6+ addl 40+128(%rsp),%ebx+ andl %r10d,%r12d+ rorxl $25,%r10d,%r13d+ vpsrlq $2,%ymm7,%ymm7+ rorxl $11,%r10d,%r15d+ leal (%rcx,%r14,1),%ecx+ leal (%rbx,%r12,1),%ebx+ vpxor %ymm7,%ymm6,%ymm6+ andnl %eax,%r10d,%r12d+ xorl %r15d,%r13d+ rorxl $6,%r10d,%r14d+ vpshufb %ymm8,%ymm6,%ymm6+ leal (%rbx,%r12,1),%ebx+ xorl %r14d,%r13d+ movl %ecx,%r15d+ vpaddd %ymm6,%ymm3,%ymm3+ rorxl $22,%ecx,%r12d+ leal (%rbx,%r13,1),%ebx+ xorl %edx,%r15d+ vpshufd $80,%ymm3,%ymm7+ rorxl $13,%ecx,%r14d+ rorxl $2,%ecx,%r13d+ leal (%r9,%rbx,1),%r9d+ vpsrld $10,%ymm7,%ymm6+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %edx,%edi+ vpsrlq $17,%ymm7,%ymm7+ xorl %r13d,%r14d+ leal (%rbx,%rdi,1),%ebx+ movl %r10d,%r12d+ vpxor %ymm7,%ymm6,%ymm6+ addl 44+128(%rsp),%eax+ andl %r9d,%r12d+ rorxl $25,%r9d,%r13d+ vpsrlq $2,%ymm7,%ymm7+ rorxl $11,%r9d,%edi+ leal (%rbx,%r14,1),%ebx+ leal (%rax,%r12,1),%eax+ vpxor %ymm7,%ymm6,%ymm6+ andnl %r11d,%r9d,%r12d+ xorl %edi,%r13d+ rorxl $6,%r9d,%r14d+ vpshufb %ymm9,%ymm6,%ymm6+ leal (%rax,%r12,1),%eax+ xorl %r14d,%r13d+ movl %ebx,%edi+ vpaddd %ymm6,%ymm3,%ymm3+ rorxl $22,%ebx,%r12d+ leal (%rax,%r13,1),%eax+ xorl %ecx,%edi+ vpaddd 96(%rsi),%ymm3,%ymm6+ rorxl $13,%ebx,%r14d+ rorxl $2,%ebx,%r13d+ leal (%r8,%rax,1),%r8d+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %ecx,%r15d+ xorl %r13d,%r14d+ leal (%rax,%r15,1),%eax+ movl %r9d,%r12d+ vmovdqa %ymm6,32(%rsp)+ leaq 128(%rsi),%rsi+ cmpb $0,3(%rsi)+ jne L$avx2_00_47+ addl 0+64(%rsp),%r11d+ andl %r8d,%r12d+ rorxl $25,%r8d,%r13d+ rorxl $11,%r8d,%r15d+ leal (%rax,%r14,1),%eax+ leal (%r11,%r12,1),%r11d+ andnl %r10d,%r8d,%r12d+ xorl %r15d,%r13d+ rorxl $6,%r8d,%r14d+ leal (%r11,%r12,1),%r11d+ xorl %r14d,%r13d+ movl %eax,%r15d+ rorxl $22,%eax,%r12d+ leal (%r11,%r13,1),%r11d+ xorl %ebx,%r15d+ rorxl $13,%eax,%r14d+ rorxl $2,%eax,%r13d+ leal (%rdx,%r11,1),%edx+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %ebx,%edi+ xorl %r13d,%r14d+ leal (%r11,%rdi,1),%r11d+ movl %r8d,%r12d+ addl 4+64(%rsp),%r10d+ andl %edx,%r12d+ rorxl $25,%edx,%r13d+ rorxl $11,%edx,%edi+ leal (%r11,%r14,1),%r11d+ leal (%r10,%r12,1),%r10d+ andnl %r9d,%edx,%r12d+ xorl %edi,%r13d+ rorxl $6,%edx,%r14d+ leal (%r10,%r12,1),%r10d+ xorl %r14d,%r13d+ movl %r11d,%edi+ rorxl $22,%r11d,%r12d+ leal (%r10,%r13,1),%r10d+ xorl %eax,%edi+ rorxl $13,%r11d,%r14d+ rorxl $2,%r11d,%r13d+ leal (%rcx,%r10,1),%ecx+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %eax,%r15d+ xorl %r13d,%r14d+ leal (%r10,%r15,1),%r10d+ movl %edx,%r12d+ addl 8+64(%rsp),%r9d+ andl %ecx,%r12d+ rorxl $25,%ecx,%r13d+ rorxl $11,%ecx,%r15d+ leal (%r10,%r14,1),%r10d+ leal (%r9,%r12,1),%r9d+ andnl %r8d,%ecx,%r12d+ xorl %r15d,%r13d+ rorxl $6,%ecx,%r14d+ leal (%r9,%r12,1),%r9d+ xorl %r14d,%r13d+ movl %r10d,%r15d+ rorxl $22,%r10d,%r12d+ leal (%r9,%r13,1),%r9d+ xorl %r11d,%r15d+ rorxl $13,%r10d,%r14d+ rorxl $2,%r10d,%r13d+ leal (%rbx,%r9,1),%ebx+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %r11d,%edi+ xorl %r13d,%r14d+ leal (%r9,%rdi,1),%r9d+ movl %ecx,%r12d+ addl 12+64(%rsp),%r8d+ andl %ebx,%r12d+ rorxl $25,%ebx,%r13d+ rorxl $11,%ebx,%edi+ leal (%r9,%r14,1),%r9d+ leal (%r8,%r12,1),%r8d+ andnl %edx,%ebx,%r12d+ xorl %edi,%r13d+ rorxl $6,%ebx,%r14d+ leal (%r8,%r12,1),%r8d+ xorl %r14d,%r13d+ movl %r9d,%edi+ rorxl $22,%r9d,%r12d+ leal (%r8,%r13,1),%r8d+ xorl %r10d,%edi+ rorxl $13,%r9d,%r14d+ rorxl $2,%r9d,%r13d+ leal (%rax,%r8,1),%eax+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %r10d,%r15d+ xorl %r13d,%r14d+ leal (%r8,%r15,1),%r8d+ movl %ebx,%r12d+ addl 32+64(%rsp),%edx+ andl %eax,%r12d+ rorxl $25,%eax,%r13d+ rorxl $11,%eax,%r15d+ leal (%r8,%r14,1),%r8d+ leal (%rdx,%r12,1),%edx+ andnl %ecx,%eax,%r12d+ xorl %r15d,%r13d+ rorxl $6,%eax,%r14d+ leal (%rdx,%r12,1),%edx+ xorl %r14d,%r13d+ movl %r8d,%r15d+ rorxl $22,%r8d,%r12d+ leal (%rdx,%r13,1),%edx+ xorl %r9d,%r15d+ rorxl $13,%r8d,%r14d+ rorxl $2,%r8d,%r13d+ leal (%r11,%rdx,1),%r11d+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %r9d,%edi+ xorl %r13d,%r14d+ leal (%rdx,%rdi,1),%edx+ movl %eax,%r12d+ addl 36+64(%rsp),%ecx+ andl %r11d,%r12d+ rorxl $25,%r11d,%r13d+ rorxl $11,%r11d,%edi+ leal (%rdx,%r14,1),%edx+ leal (%rcx,%r12,1),%ecx+ andnl %ebx,%r11d,%r12d+ xorl %edi,%r13d+ rorxl $6,%r11d,%r14d+ leal (%rcx,%r12,1),%ecx+ xorl %r14d,%r13d+ movl %edx,%edi+ rorxl $22,%edx,%r12d+ leal (%rcx,%r13,1),%ecx+ xorl %r8d,%edi+ rorxl $13,%edx,%r14d+ rorxl $2,%edx,%r13d+ leal (%r10,%rcx,1),%r10d+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %r8d,%r15d+ xorl %r13d,%r14d+ leal (%rcx,%r15,1),%ecx+ movl %r11d,%r12d+ addl 40+64(%rsp),%ebx+ andl %r10d,%r12d+ rorxl $25,%r10d,%r13d+ rorxl $11,%r10d,%r15d+ leal (%rcx,%r14,1),%ecx+ leal (%rbx,%r12,1),%ebx+ andnl %eax,%r10d,%r12d+ xorl %r15d,%r13d+ rorxl $6,%r10d,%r14d+ leal (%rbx,%r12,1),%ebx+ xorl %r14d,%r13d+ movl %ecx,%r15d+ rorxl $22,%ecx,%r12d+ leal (%rbx,%r13,1),%ebx+ xorl %edx,%r15d+ rorxl $13,%ecx,%r14d+ rorxl $2,%ecx,%r13d+ leal (%r9,%rbx,1),%r9d+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %edx,%edi+ xorl %r13d,%r14d+ leal (%rbx,%rdi,1),%ebx+ movl %r10d,%r12d+ addl 44+64(%rsp),%eax+ andl %r9d,%r12d+ rorxl $25,%r9d,%r13d+ rorxl $11,%r9d,%edi+ leal (%rbx,%r14,1),%ebx+ leal (%rax,%r12,1),%eax+ andnl %r11d,%r9d,%r12d+ xorl %edi,%r13d+ rorxl $6,%r9d,%r14d+ leal (%rax,%r12,1),%eax+ xorl %r14d,%r13d+ movl %ebx,%edi+ rorxl $22,%ebx,%r12d+ leal (%rax,%r13,1),%eax+ xorl %ecx,%edi+ rorxl $13,%ebx,%r14d+ rorxl $2,%ebx,%r13d+ leal (%r8,%rax,1),%r8d+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %ecx,%r15d+ xorl %r13d,%r14d+ leal (%rax,%r15,1),%eax+ movl %r9d,%r12d+ addl 0(%rsp),%r11d+ andl %r8d,%r12d+ rorxl $25,%r8d,%r13d+ rorxl $11,%r8d,%r15d+ leal (%rax,%r14,1),%eax+ leal (%r11,%r12,1),%r11d+ andnl %r10d,%r8d,%r12d+ xorl %r15d,%r13d+ rorxl $6,%r8d,%r14d+ leal (%r11,%r12,1),%r11d+ xorl %r14d,%r13d+ movl %eax,%r15d+ rorxl $22,%eax,%r12d+ leal (%r11,%r13,1),%r11d+ xorl %ebx,%r15d+ rorxl $13,%eax,%r14d+ rorxl $2,%eax,%r13d+ leal (%rdx,%r11,1),%edx+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %ebx,%edi+ xorl %r13d,%r14d+ leal (%r11,%rdi,1),%r11d+ movl %r8d,%r12d+ addl 4(%rsp),%r10d+ andl %edx,%r12d+ rorxl $25,%edx,%r13d+ rorxl $11,%edx,%edi+ leal (%r11,%r14,1),%r11d+ leal (%r10,%r12,1),%r10d+ andnl %r9d,%edx,%r12d+ xorl %edi,%r13d+ rorxl $6,%edx,%r14d+ leal (%r10,%r12,1),%r10d+ xorl %r14d,%r13d+ movl %r11d,%edi+ rorxl $22,%r11d,%r12d+ leal (%r10,%r13,1),%r10d+ xorl %eax,%edi+ rorxl $13,%r11d,%r14d+ rorxl $2,%r11d,%r13d+ leal (%rcx,%r10,1),%ecx+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %eax,%r15d+ xorl %r13d,%r14d+ leal (%r10,%r15,1),%r10d+ movl %edx,%r12d+ addl 8(%rsp),%r9d+ andl %ecx,%r12d+ rorxl $25,%ecx,%r13d+ rorxl $11,%ecx,%r15d+ leal (%r10,%r14,1),%r10d+ leal (%r9,%r12,1),%r9d+ andnl %r8d,%ecx,%r12d+ xorl %r15d,%r13d+ rorxl $6,%ecx,%r14d+ leal (%r9,%r12,1),%r9d+ xorl %r14d,%r13d+ movl %r10d,%r15d+ rorxl $22,%r10d,%r12d+ leal (%r9,%r13,1),%r9d+ xorl %r11d,%r15d+ rorxl $13,%r10d,%r14d+ rorxl $2,%r10d,%r13d+ leal (%rbx,%r9,1),%ebx+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %r11d,%edi+ xorl %r13d,%r14d+ leal (%r9,%rdi,1),%r9d+ movl %ecx,%r12d+ addl 12(%rsp),%r8d+ andl %ebx,%r12d+ rorxl $25,%ebx,%r13d+ rorxl $11,%ebx,%edi+ leal (%r9,%r14,1),%r9d+ leal (%r8,%r12,1),%r8d+ andnl %edx,%ebx,%r12d+ xorl %edi,%r13d+ rorxl $6,%ebx,%r14d+ leal (%r8,%r12,1),%r8d+ xorl %r14d,%r13d+ movl %r9d,%edi+ rorxl $22,%r9d,%r12d+ leal (%r8,%r13,1),%r8d+ xorl %r10d,%edi+ rorxl $13,%r9d,%r14d+ rorxl $2,%r9d,%r13d+ leal (%rax,%r8,1),%eax+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %r10d,%r15d+ xorl %r13d,%r14d+ leal (%r8,%r15,1),%r8d+ movl %ebx,%r12d+ addl 32(%rsp),%edx+ andl %eax,%r12d+ rorxl $25,%eax,%r13d+ rorxl $11,%eax,%r15d+ leal (%r8,%r14,1),%r8d+ leal (%rdx,%r12,1),%edx+ andnl %ecx,%eax,%r12d+ xorl %r15d,%r13d+ rorxl $6,%eax,%r14d+ leal (%rdx,%r12,1),%edx+ xorl %r14d,%r13d+ movl %r8d,%r15d+ rorxl $22,%r8d,%r12d+ leal (%rdx,%r13,1),%edx+ xorl %r9d,%r15d+ rorxl $13,%r8d,%r14d+ rorxl $2,%r8d,%r13d+ leal (%r11,%rdx,1),%r11d+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %r9d,%edi+ xorl %r13d,%r14d+ leal (%rdx,%rdi,1),%edx+ movl %eax,%r12d+ addl 36(%rsp),%ecx+ andl %r11d,%r12d+ rorxl $25,%r11d,%r13d+ rorxl $11,%r11d,%edi+ leal (%rdx,%r14,1),%edx+ leal (%rcx,%r12,1),%ecx+ andnl %ebx,%r11d,%r12d+ xorl %edi,%r13d+ rorxl $6,%r11d,%r14d+ leal (%rcx,%r12,1),%ecx+ xorl %r14d,%r13d+ movl %edx,%edi+ rorxl $22,%edx,%r12d+ leal (%rcx,%r13,1),%ecx+ xorl %r8d,%edi+ rorxl $13,%edx,%r14d+ rorxl $2,%edx,%r13d+ leal (%r10,%rcx,1),%r10d+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %r8d,%r15d+ xorl %r13d,%r14d+ leal (%rcx,%r15,1),%ecx+ movl %r11d,%r12d+ addl 40(%rsp),%ebx+ andl %r10d,%r12d+ rorxl $25,%r10d,%r13d+ rorxl $11,%r10d,%r15d+ leal (%rcx,%r14,1),%ecx+ leal (%rbx,%r12,1),%ebx+ andnl %eax,%r10d,%r12d+ xorl %r15d,%r13d+ rorxl $6,%r10d,%r14d+ leal (%rbx,%r12,1),%ebx+ xorl %r14d,%r13d+ movl %ecx,%r15d+ rorxl $22,%ecx,%r12d+ leal (%rbx,%r13,1),%ebx+ xorl %edx,%r15d+ rorxl $13,%ecx,%r14d+ rorxl $2,%ecx,%r13d+ leal (%r9,%rbx,1),%r9d+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %edx,%edi+ xorl %r13d,%r14d+ leal (%rbx,%rdi,1),%ebx+ movl %r10d,%r12d+ addl 44(%rsp),%eax+ andl %r9d,%r12d+ rorxl $25,%r9d,%r13d+ rorxl $11,%r9d,%edi+ leal (%rbx,%r14,1),%ebx+ leal (%rax,%r12,1),%eax+ andnl %r11d,%r9d,%r12d+ xorl %edi,%r13d+ rorxl $6,%r9d,%r14d+ leal (%rax,%r12,1),%eax+ xorl %r14d,%r13d+ movl %ebx,%edi+ rorxl $22,%ebx,%r12d+ leal (%rax,%r13,1),%eax+ xorl %ecx,%edi+ rorxl $13,%ebx,%r14d+ rorxl $2,%ebx,%r13d+ leal (%r8,%rax,1),%r8d+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %ecx,%r15d+ xorl %r13d,%r14d+ leal (%rax,%r15,1),%eax+ movl %r9d,%r12d+ movq -64(%rbp),%rdi+ addl %r14d,%eax+ movl -56(%rbp),%r12d++ addl 0(%rdi),%eax+ addl 4(%rdi),%ebx+ addl 8(%rdi),%ecx+ addl 12(%rdi),%edx+ addl 16(%rdi),%r8d+ addl 20(%rdi),%r9d+ addl 24(%rdi),%r10d+ addl 28(%rdi),%r11d++ movl %eax,0(%rdi)+ movl %ebx,4(%rdi)+ movl %ecx,8(%rdi)+ movl %edx,12(%rdi)+ movl %r8d,16(%rdi)+ movl %r9d,20(%rdi)+ movl %r10d,24(%rdi)+ movl %r11d,28(%rdi)++ cmpl -48(%rbp),%r12d+ je L$done_avx2++ leaq 448(%rsp),%rsi+ xorl %r14d,%r14d+ movl %ebx,%edi+ xorl %ecx,%edi+ movl %r9d,%r12d+ jmp L$ower_avx2+.p2align 4+L$ower_avx2:+ addl 0+16(%rsi),%r11d+ andl %r8d,%r12d+ rorxl $25,%r8d,%r13d+ rorxl $11,%r8d,%r15d+ leal (%rax,%r14,1),%eax+ leal (%r11,%r12,1),%r11d+ andnl %r10d,%r8d,%r12d+ xorl %r15d,%r13d+ rorxl $6,%r8d,%r14d+ leal (%r11,%r12,1),%r11d+ xorl %r14d,%r13d+ movl %eax,%r15d+ rorxl $22,%eax,%r12d+ leal (%r11,%r13,1),%r11d+ xorl %ebx,%r15d+ rorxl $13,%eax,%r14d+ rorxl $2,%eax,%r13d+ leal (%rdx,%r11,1),%edx+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %ebx,%edi+ xorl %r13d,%r14d+ leal (%r11,%rdi,1),%r11d+ movl %r8d,%r12d+ addl 4+16(%rsi),%r10d+ andl %edx,%r12d+ rorxl $25,%edx,%r13d+ rorxl $11,%edx,%edi+ leal (%r11,%r14,1),%r11d+ leal (%r10,%r12,1),%r10d+ andnl %r9d,%edx,%r12d+ xorl %edi,%r13d+ rorxl $6,%edx,%r14d+ leal (%r10,%r12,1),%r10d+ xorl %r14d,%r13d+ movl %r11d,%edi+ rorxl $22,%r11d,%r12d+ leal (%r10,%r13,1),%r10d+ xorl %eax,%edi+ rorxl $13,%r11d,%r14d+ rorxl $2,%r11d,%r13d+ leal (%rcx,%r10,1),%ecx+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %eax,%r15d+ xorl %r13d,%r14d+ leal (%r10,%r15,1),%r10d+ movl %edx,%r12d+ addl 8+16(%rsi),%r9d+ andl %ecx,%r12d+ rorxl $25,%ecx,%r13d+ rorxl $11,%ecx,%r15d+ leal (%r10,%r14,1),%r10d+ leal (%r9,%r12,1),%r9d+ andnl %r8d,%ecx,%r12d+ xorl %r15d,%r13d+ rorxl $6,%ecx,%r14d+ leal (%r9,%r12,1),%r9d+ xorl %r14d,%r13d+ movl %r10d,%r15d+ rorxl $22,%r10d,%r12d+ leal (%r9,%r13,1),%r9d+ xorl %r11d,%r15d+ rorxl $13,%r10d,%r14d+ rorxl $2,%r10d,%r13d+ leal (%rbx,%r9,1),%ebx+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %r11d,%edi+ xorl %r13d,%r14d+ leal (%r9,%rdi,1),%r9d+ movl %ecx,%r12d+ addl 12+16(%rsi),%r8d+ andl %ebx,%r12d+ rorxl $25,%ebx,%r13d+ rorxl $11,%ebx,%edi+ leal (%r9,%r14,1),%r9d+ leal (%r8,%r12,1),%r8d+ andnl %edx,%ebx,%r12d+ xorl %edi,%r13d+ rorxl $6,%ebx,%r14d+ leal (%r8,%r12,1),%r8d+ xorl %r14d,%r13d+ movl %r9d,%edi+ rorxl $22,%r9d,%r12d+ leal (%r8,%r13,1),%r8d+ xorl %r10d,%edi+ rorxl $13,%r9d,%r14d+ rorxl $2,%r9d,%r13d+ leal (%rax,%r8,1),%eax+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %r10d,%r15d+ xorl %r13d,%r14d+ leal (%r8,%r15,1),%r8d+ movl %ebx,%r12d+ addl 32+16(%rsi),%edx+ andl %eax,%r12d+ rorxl $25,%eax,%r13d+ rorxl $11,%eax,%r15d+ leal (%r8,%r14,1),%r8d+ leal (%rdx,%r12,1),%edx+ andnl %ecx,%eax,%r12d+ xorl %r15d,%r13d+ rorxl $6,%eax,%r14d+ leal (%rdx,%r12,1),%edx+ xorl %r14d,%r13d+ movl %r8d,%r15d+ rorxl $22,%r8d,%r12d+ leal (%rdx,%r13,1),%edx+ xorl %r9d,%r15d+ rorxl $13,%r8d,%r14d+ rorxl $2,%r8d,%r13d+ leal (%r11,%rdx,1),%r11d+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %r9d,%edi+ xorl %r13d,%r14d+ leal (%rdx,%rdi,1),%edx+ movl %eax,%r12d+ addl 36+16(%rsi),%ecx+ andl %r11d,%r12d+ rorxl $25,%r11d,%r13d+ rorxl $11,%r11d,%edi+ leal (%rdx,%r14,1),%edx+ leal (%rcx,%r12,1),%ecx+ andnl %ebx,%r11d,%r12d+ xorl %edi,%r13d+ rorxl $6,%r11d,%r14d+ leal (%rcx,%r12,1),%ecx+ xorl %r14d,%r13d+ movl %edx,%edi+ rorxl $22,%edx,%r12d+ leal (%rcx,%r13,1),%ecx+ xorl %r8d,%edi+ rorxl $13,%edx,%r14d+ rorxl $2,%edx,%r13d+ leal (%r10,%rcx,1),%r10d+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %r8d,%r15d+ xorl %r13d,%r14d+ leal (%rcx,%r15,1),%ecx+ movl %r11d,%r12d+ addl 40+16(%rsi),%ebx+ andl %r10d,%r12d+ rorxl $25,%r10d,%r13d+ rorxl $11,%r10d,%r15d+ leal (%rcx,%r14,1),%ecx+ leal (%rbx,%r12,1),%ebx+ andnl %eax,%r10d,%r12d+ xorl %r15d,%r13d+ rorxl $6,%r10d,%r14d+ leal (%rbx,%r12,1),%ebx+ xorl %r14d,%r13d+ movl %ecx,%r15d+ rorxl $22,%ecx,%r12d+ leal (%rbx,%r13,1),%ebx+ xorl %edx,%r15d+ rorxl $13,%ecx,%r14d+ rorxl $2,%ecx,%r13d+ leal (%r9,%rbx,1),%r9d+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %edx,%edi+ xorl %r13d,%r14d+ leal (%rbx,%rdi,1),%ebx+ movl %r10d,%r12d+ addl 44+16(%rsi),%eax+ andl %r9d,%r12d+ rorxl $25,%r9d,%r13d+ rorxl $11,%r9d,%edi+ leal (%rbx,%r14,1),%ebx+ leal (%rax,%r12,1),%eax+ andnl %r11d,%r9d,%r12d+ xorl %edi,%r13d+ rorxl $6,%r9d,%r14d+ leal (%rax,%r12,1),%eax+ xorl %r14d,%r13d+ movl %ebx,%edi+ rorxl $22,%ebx,%r12d+ leal (%rax,%r13,1),%eax+ xorl %ecx,%edi+ rorxl $13,%ebx,%r14d+ rorxl $2,%ebx,%r13d+ leal (%r8,%rax,1),%r8d+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %ecx,%r15d+ xorl %r13d,%r14d+ leal (%rax,%r15,1),%eax+ movl %r9d,%r12d+ leaq -64(%rsi),%rsi+ cmpq %rsp,%rsi+ jae L$ower_avx2++ movq -64(%rbp),%rdi+ addl %r14d,%eax+ movq -56(%rbp),%rsi+ leaq 448(%rsp),%rsp++ addl 0(%rdi),%eax+ addl 4(%rdi),%ebx+ addl 8(%rdi),%ecx+ addl 12(%rdi),%edx+ addl 16(%rdi),%r8d+ addl 20(%rdi),%r9d+ leaq 128(%rsi),%rsi+ addl 24(%rdi),%r10d+ movq %rsi,%r12+ addl 28(%rdi),%r11d+ cmpq -48(%rbp),%rsi++ movl %eax,0(%rdi)+ cmoveq %rsp,%r12+ movl %ebx,4(%rdi)+ movl %ecx,8(%rdi)+ movl %edx,12(%rdi)+ movl %r8d,16(%rdi)+ movl %r9d,20(%rdi)+ movl %r10d,24(%rdi)+ movl %r11d,28(%rdi)++ jbe L$oop_avx2++L$done_avx2:+ vzeroupper+ movq -40(%rbp),%r15+ movq -32(%rbp),%r14+ movq -24(%rbp),%r13+ movq -16(%rbp),%r12+ movq -8(%rbp),%rbx+ movq %rbp,%rsp+.cfi_def_cfa_register %rsp+ popq %rbp+.cfi_adjust_cfa_offset -8+.cfi_restore %rbp+.cfi_restore %r12+.cfi_restore %r13+.cfi_restore %r14+.cfi_restore %r15+.cfi_restore %rbx+ .byte 0xf3,0xc3+.cfi_endproc +
@@ -0,0 +1,5731 @@+.text +++.globl crypton_sha256_asm_block_data_order+.def crypton_sha256_asm_block_data_order; .scl 2; .type 32; .endef+.p2align 4+crypton_sha256_asm_block_data_order:+ .byte 0xf3,0x0f,0x1e,0xfa+ movq %rdi,8(%rsp)+ movq %rsi,16(%rsp)+ movq %rsp,%r11+.LSEH_begin_crypton_sha256_asm_block_data_order:+++ pushq %rbp++ movq %rsp,%rbp++ movq %rcx,%rdi+ movq %rdx,%rsi+ movq %r8,%rdx+ leaq crypton_ia32cap_P(%rip),%rax+ movl 0(%rax),%r9d+ movl 4(%rax),%r10d+ movl 8(%rax),%eax+ testl $536870912,%eax+ jnz .Lshaext_shortcut+ andl $296,%eax+ cmpl $296,%eax+ je .Lavx2_shortcut+ andl $1073741824,%r9d+ andl $268435968,%r10d+ orl %r9d,%r10d+ cmpl $1342177792,%r10d+ je .Lavx_shortcut+ testl $512,%r10d+ jnz .Lssse3_shortcut+ pushq %rbx++ pushq %r12++ pushq %r13++ pushq %r14++ pushq %r15++ shlq $4,%rdx+ subq $64+24,%rsp+++.LSEH_body_crypton_sha256_asm_block_data_order:++ leaq (%rsi,%rdx,4),%rdx+ movq %rdi,64+0(%rsp)+ movq %rsi,64+8(%rsp)+ movq %rdx,64+16(%rsp)++ movl 0(%rdi),%eax+ movl 4(%rdi),%ebx+ movl 8(%rdi),%ecx+ movl 12(%rdi),%edx+ movl 16(%rdi),%r8d+ movl 20(%rdi),%r9d+ movl 24(%rdi),%r10d+ movl 28(%rdi),%r11d+ jmp .Lloop++.p2align 4+.Lloop:+ movl %ebx,%edi+ leaq K256(%rip),%rbp+ xorl %ecx,%edi+ movl 0(%rsi),%r12d+ movl %r8d,%r13d+ movl %eax,%r14d+ bswapl %r12d+ rorl $14,%r13d+ movl %r9d,%r15d++ xorl %r8d,%r13d+ rorl $9,%r14d+ xorl %r10d,%r15d++ movl %r12d,0(%rsp)+ xorl %eax,%r14d+ andl %r8d,%r15d++ rorl $5,%r13d+ addl %r11d,%r12d+ xorl %r10d,%r15d++ rorl $11,%r14d+ xorl %r8d,%r13d+ addl %r15d,%r12d++ movl %eax,%r15d+ addl (%rbp),%r12d+ xorl %eax,%r14d++ xorl %ebx,%r15d+ rorl $6,%r13d+ movl %ebx,%r11d++ andl %r15d,%edi+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %edi,%r11d+ addl %r12d,%edx+ addl %r12d,%r11d++ leaq 4(%rbp),%rbp+ addl %r14d,%r11d+ movl 4(%rsi),%r12d+ movl %edx,%r13d+ movl %r11d,%r14d+ bswapl %r12d+ rorl $14,%r13d+ movl %r8d,%edi++ xorl %edx,%r13d+ rorl $9,%r14d+ xorl %r9d,%edi++ movl %r12d,4(%rsp)+ xorl %r11d,%r14d+ andl %edx,%edi++ rorl $5,%r13d+ addl %r10d,%r12d+ xorl %r9d,%edi++ rorl $11,%r14d+ xorl %edx,%r13d+ addl %edi,%r12d++ movl %r11d,%edi+ addl (%rbp),%r12d+ xorl %r11d,%r14d++ xorl %eax,%edi+ rorl $6,%r13d+ movl %eax,%r10d++ andl %edi,%r15d+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %r15d,%r10d+ addl %r12d,%ecx+ addl %r12d,%r10d++ leaq 4(%rbp),%rbp+ addl %r14d,%r10d+ movl 8(%rsi),%r12d+ movl %ecx,%r13d+ movl %r10d,%r14d+ bswapl %r12d+ rorl $14,%r13d+ movl %edx,%r15d++ xorl %ecx,%r13d+ rorl $9,%r14d+ xorl %r8d,%r15d++ movl %r12d,8(%rsp)+ xorl %r10d,%r14d+ andl %ecx,%r15d++ rorl $5,%r13d+ addl %r9d,%r12d+ xorl %r8d,%r15d++ rorl $11,%r14d+ xorl %ecx,%r13d+ addl %r15d,%r12d++ movl %r10d,%r15d+ addl (%rbp),%r12d+ xorl %r10d,%r14d++ xorl %r11d,%r15d+ rorl $6,%r13d+ movl %r11d,%r9d++ andl %r15d,%edi+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %edi,%r9d+ addl %r12d,%ebx+ addl %r12d,%r9d++ leaq 4(%rbp),%rbp+ addl %r14d,%r9d+ movl 12(%rsi),%r12d+ movl %ebx,%r13d+ movl %r9d,%r14d+ bswapl %r12d+ rorl $14,%r13d+ movl %ecx,%edi++ xorl %ebx,%r13d+ rorl $9,%r14d+ xorl %edx,%edi++ movl %r12d,12(%rsp)+ xorl %r9d,%r14d+ andl %ebx,%edi++ rorl $5,%r13d+ addl %r8d,%r12d+ xorl %edx,%edi++ rorl $11,%r14d+ xorl %ebx,%r13d+ addl %edi,%r12d++ movl %r9d,%edi+ addl (%rbp),%r12d+ xorl %r9d,%r14d++ xorl %r10d,%edi+ rorl $6,%r13d+ movl %r10d,%r8d++ andl %edi,%r15d+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %r15d,%r8d+ addl %r12d,%eax+ addl %r12d,%r8d++ leaq 20(%rbp),%rbp+ addl %r14d,%r8d+ movl 16(%rsi),%r12d+ movl %eax,%r13d+ movl %r8d,%r14d+ bswapl %r12d+ rorl $14,%r13d+ movl %ebx,%r15d++ xorl %eax,%r13d+ rorl $9,%r14d+ xorl %ecx,%r15d++ movl %r12d,16(%rsp)+ xorl %r8d,%r14d+ andl %eax,%r15d++ rorl $5,%r13d+ addl %edx,%r12d+ xorl %ecx,%r15d++ rorl $11,%r14d+ xorl %eax,%r13d+ addl %r15d,%r12d++ movl %r8d,%r15d+ addl (%rbp),%r12d+ xorl %r8d,%r14d++ xorl %r9d,%r15d+ rorl $6,%r13d+ movl %r9d,%edx++ andl %r15d,%edi+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %edi,%edx+ addl %r12d,%r11d+ addl %r12d,%edx++ leaq 4(%rbp),%rbp+ addl %r14d,%edx+ movl 20(%rsi),%r12d+ movl %r11d,%r13d+ movl %edx,%r14d+ bswapl %r12d+ rorl $14,%r13d+ movl %eax,%edi++ xorl %r11d,%r13d+ rorl $9,%r14d+ xorl %ebx,%edi++ movl %r12d,20(%rsp)+ xorl %edx,%r14d+ andl %r11d,%edi++ rorl $5,%r13d+ addl %ecx,%r12d+ xorl %ebx,%edi++ rorl $11,%r14d+ xorl %r11d,%r13d+ addl %edi,%r12d++ movl %edx,%edi+ addl (%rbp),%r12d+ xorl %edx,%r14d++ xorl %r8d,%edi+ rorl $6,%r13d+ movl %r8d,%ecx++ andl %edi,%r15d+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %r15d,%ecx+ addl %r12d,%r10d+ addl %r12d,%ecx++ leaq 4(%rbp),%rbp+ addl %r14d,%ecx+ movl 24(%rsi),%r12d+ movl %r10d,%r13d+ movl %ecx,%r14d+ bswapl %r12d+ rorl $14,%r13d+ movl %r11d,%r15d++ xorl %r10d,%r13d+ rorl $9,%r14d+ xorl %eax,%r15d++ movl %r12d,24(%rsp)+ xorl %ecx,%r14d+ andl %r10d,%r15d++ rorl $5,%r13d+ addl %ebx,%r12d+ xorl %eax,%r15d++ rorl $11,%r14d+ xorl %r10d,%r13d+ addl %r15d,%r12d++ movl %ecx,%r15d+ addl (%rbp),%r12d+ xorl %ecx,%r14d++ xorl %edx,%r15d+ rorl $6,%r13d+ movl %edx,%ebx++ andl %r15d,%edi+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %edi,%ebx+ addl %r12d,%r9d+ addl %r12d,%ebx++ leaq 4(%rbp),%rbp+ addl %r14d,%ebx+ movl 28(%rsi),%r12d+ movl %r9d,%r13d+ movl %ebx,%r14d+ bswapl %r12d+ rorl $14,%r13d+ movl %r10d,%edi++ xorl %r9d,%r13d+ rorl $9,%r14d+ xorl %r11d,%edi++ movl %r12d,28(%rsp)+ xorl %ebx,%r14d+ andl %r9d,%edi++ rorl $5,%r13d+ addl %eax,%r12d+ xorl %r11d,%edi++ rorl $11,%r14d+ xorl %r9d,%r13d+ addl %edi,%r12d++ movl %ebx,%edi+ addl (%rbp),%r12d+ xorl %ebx,%r14d++ xorl %ecx,%edi+ rorl $6,%r13d+ movl %ecx,%eax++ andl %edi,%r15d+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %r15d,%eax+ addl %r12d,%r8d+ addl %r12d,%eax++ leaq 20(%rbp),%rbp+ addl %r14d,%eax+ movl 32(%rsi),%r12d+ movl %r8d,%r13d+ movl %eax,%r14d+ bswapl %r12d+ rorl $14,%r13d+ movl %r9d,%r15d++ xorl %r8d,%r13d+ rorl $9,%r14d+ xorl %r10d,%r15d++ movl %r12d,32(%rsp)+ xorl %eax,%r14d+ andl %r8d,%r15d++ rorl $5,%r13d+ addl %r11d,%r12d+ xorl %r10d,%r15d++ rorl $11,%r14d+ xorl %r8d,%r13d+ addl %r15d,%r12d++ movl %eax,%r15d+ addl (%rbp),%r12d+ xorl %eax,%r14d++ xorl %ebx,%r15d+ rorl $6,%r13d+ movl %ebx,%r11d++ andl %r15d,%edi+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %edi,%r11d+ addl %r12d,%edx+ addl %r12d,%r11d++ leaq 4(%rbp),%rbp+ addl %r14d,%r11d+ movl 36(%rsi),%r12d+ movl %edx,%r13d+ movl %r11d,%r14d+ bswapl %r12d+ rorl $14,%r13d+ movl %r8d,%edi++ xorl %edx,%r13d+ rorl $9,%r14d+ xorl %r9d,%edi++ movl %r12d,36(%rsp)+ xorl %r11d,%r14d+ andl %edx,%edi++ rorl $5,%r13d+ addl %r10d,%r12d+ xorl %r9d,%edi++ rorl $11,%r14d+ xorl %edx,%r13d+ addl %edi,%r12d++ movl %r11d,%edi+ addl (%rbp),%r12d+ xorl %r11d,%r14d++ xorl %eax,%edi+ rorl $6,%r13d+ movl %eax,%r10d++ andl %edi,%r15d+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %r15d,%r10d+ addl %r12d,%ecx+ addl %r12d,%r10d++ leaq 4(%rbp),%rbp+ addl %r14d,%r10d+ movl 40(%rsi),%r12d+ movl %ecx,%r13d+ movl %r10d,%r14d+ bswapl %r12d+ rorl $14,%r13d+ movl %edx,%r15d++ xorl %ecx,%r13d+ rorl $9,%r14d+ xorl %r8d,%r15d++ movl %r12d,40(%rsp)+ xorl %r10d,%r14d+ andl %ecx,%r15d++ rorl $5,%r13d+ addl %r9d,%r12d+ xorl %r8d,%r15d++ rorl $11,%r14d+ xorl %ecx,%r13d+ addl %r15d,%r12d++ movl %r10d,%r15d+ addl (%rbp),%r12d+ xorl %r10d,%r14d++ xorl %r11d,%r15d+ rorl $6,%r13d+ movl %r11d,%r9d++ andl %r15d,%edi+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %edi,%r9d+ addl %r12d,%ebx+ addl %r12d,%r9d++ leaq 4(%rbp),%rbp+ addl %r14d,%r9d+ movl 44(%rsi),%r12d+ movl %ebx,%r13d+ movl %r9d,%r14d+ bswapl %r12d+ rorl $14,%r13d+ movl %ecx,%edi++ xorl %ebx,%r13d+ rorl $9,%r14d+ xorl %edx,%edi++ movl %r12d,44(%rsp)+ xorl %r9d,%r14d+ andl %ebx,%edi++ rorl $5,%r13d+ addl %r8d,%r12d+ xorl %edx,%edi++ rorl $11,%r14d+ xorl %ebx,%r13d+ addl %edi,%r12d++ movl %r9d,%edi+ addl (%rbp),%r12d+ xorl %r9d,%r14d++ xorl %r10d,%edi+ rorl $6,%r13d+ movl %r10d,%r8d++ andl %edi,%r15d+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %r15d,%r8d+ addl %r12d,%eax+ addl %r12d,%r8d++ leaq 20(%rbp),%rbp+ addl %r14d,%r8d+ movl 48(%rsi),%r12d+ movl %eax,%r13d+ movl %r8d,%r14d+ bswapl %r12d+ rorl $14,%r13d+ movl %ebx,%r15d++ xorl %eax,%r13d+ rorl $9,%r14d+ xorl %ecx,%r15d++ movl %r12d,48(%rsp)+ xorl %r8d,%r14d+ andl %eax,%r15d++ rorl $5,%r13d+ addl %edx,%r12d+ xorl %ecx,%r15d++ rorl $11,%r14d+ xorl %eax,%r13d+ addl %r15d,%r12d++ movl %r8d,%r15d+ addl (%rbp),%r12d+ xorl %r8d,%r14d++ xorl %r9d,%r15d+ rorl $6,%r13d+ movl %r9d,%edx++ andl %r15d,%edi+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %edi,%edx+ addl %r12d,%r11d+ addl %r12d,%edx++ leaq 4(%rbp),%rbp+ addl %r14d,%edx+ movl 52(%rsi),%r12d+ movl %r11d,%r13d+ movl %edx,%r14d+ bswapl %r12d+ rorl $14,%r13d+ movl %eax,%edi++ xorl %r11d,%r13d+ rorl $9,%r14d+ xorl %ebx,%edi++ movl %r12d,52(%rsp)+ xorl %edx,%r14d+ andl %r11d,%edi++ rorl $5,%r13d+ addl %ecx,%r12d+ xorl %ebx,%edi++ rorl $11,%r14d+ xorl %r11d,%r13d+ addl %edi,%r12d++ movl %edx,%edi+ addl (%rbp),%r12d+ xorl %edx,%r14d++ xorl %r8d,%edi+ rorl $6,%r13d+ movl %r8d,%ecx++ andl %edi,%r15d+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %r15d,%ecx+ addl %r12d,%r10d+ addl %r12d,%ecx++ leaq 4(%rbp),%rbp+ addl %r14d,%ecx+ movl 56(%rsi),%r12d+ movl %r10d,%r13d+ movl %ecx,%r14d+ bswapl %r12d+ rorl $14,%r13d+ movl %r11d,%r15d++ xorl %r10d,%r13d+ rorl $9,%r14d+ xorl %eax,%r15d++ movl %r12d,56(%rsp)+ xorl %ecx,%r14d+ andl %r10d,%r15d++ rorl $5,%r13d+ addl %ebx,%r12d+ xorl %eax,%r15d++ rorl $11,%r14d+ xorl %r10d,%r13d+ addl %r15d,%r12d++ movl %ecx,%r15d+ addl (%rbp),%r12d+ xorl %ecx,%r14d++ xorl %edx,%r15d+ rorl $6,%r13d+ movl %edx,%ebx++ andl %r15d,%edi+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %edi,%ebx+ addl %r12d,%r9d+ addl %r12d,%ebx++ leaq 4(%rbp),%rbp+ addl %r14d,%ebx+ movl 60(%rsi),%r12d+ movl %r9d,%r13d+ movl %ebx,%r14d+ bswapl %r12d+ rorl $14,%r13d+ movl %r10d,%edi++ xorl %r9d,%r13d+ rorl $9,%r14d+ xorl %r11d,%edi++ movl %r12d,60(%rsp)+ xorl %ebx,%r14d+ andl %r9d,%edi++ rorl $5,%r13d+ addl %eax,%r12d+ xorl %r11d,%edi++ rorl $11,%r14d+ xorl %r9d,%r13d+ addl %edi,%r12d++ movl %ebx,%edi+ addl (%rbp),%r12d+ xorl %ebx,%r14d++ xorl %ecx,%edi+ rorl $6,%r13d+ movl %ecx,%eax++ andl %edi,%r15d+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %r15d,%eax+ addl %r12d,%r8d+ addl %r12d,%eax++ leaq 20(%rbp),%rbp+ jmp .Lrounds_16_xx+.p2align 4+.Lrounds_16_xx:+ movl 4(%rsp),%r13d+ movl 56(%rsp),%r15d++ movl %r13d,%r12d+ rorl $11,%r13d+ addl %r14d,%eax+ movl %r15d,%r14d+ rorl $2,%r15d++ xorl %r12d,%r13d+ shrl $3,%r12d+ rorl $7,%r13d+ xorl %r14d,%r15d+ shrl $10,%r14d++ rorl $17,%r15d+ xorl %r13d,%r12d+ xorl %r14d,%r15d+ addl 36(%rsp),%r12d++ addl 0(%rsp),%r12d+ movl %r8d,%r13d+ addl %r15d,%r12d+ movl %eax,%r14d+ rorl $14,%r13d+ movl %r9d,%r15d++ xorl %r8d,%r13d+ rorl $9,%r14d+ xorl %r10d,%r15d++ movl %r12d,0(%rsp)+ xorl %eax,%r14d+ andl %r8d,%r15d++ rorl $5,%r13d+ addl %r11d,%r12d+ xorl %r10d,%r15d++ rorl $11,%r14d+ xorl %r8d,%r13d+ addl %r15d,%r12d++ movl %eax,%r15d+ addl (%rbp),%r12d+ xorl %eax,%r14d++ xorl %ebx,%r15d+ rorl $6,%r13d+ movl %ebx,%r11d++ andl %r15d,%edi+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %edi,%r11d+ addl %r12d,%edx+ addl %r12d,%r11d++ leaq 4(%rbp),%rbp+ movl 8(%rsp),%r13d+ movl 60(%rsp),%edi++ movl %r13d,%r12d+ rorl $11,%r13d+ addl %r14d,%r11d+ movl %edi,%r14d+ rorl $2,%edi++ xorl %r12d,%r13d+ shrl $3,%r12d+ rorl $7,%r13d+ xorl %r14d,%edi+ shrl $10,%r14d++ rorl $17,%edi+ xorl %r13d,%r12d+ xorl %r14d,%edi+ addl 40(%rsp),%r12d++ addl 4(%rsp),%r12d+ movl %edx,%r13d+ addl %edi,%r12d+ movl %r11d,%r14d+ rorl $14,%r13d+ movl %r8d,%edi++ xorl %edx,%r13d+ rorl $9,%r14d+ xorl %r9d,%edi++ movl %r12d,4(%rsp)+ xorl %r11d,%r14d+ andl %edx,%edi++ rorl $5,%r13d+ addl %r10d,%r12d+ xorl %r9d,%edi++ rorl $11,%r14d+ xorl %edx,%r13d+ addl %edi,%r12d++ movl %r11d,%edi+ addl (%rbp),%r12d+ xorl %r11d,%r14d++ xorl %eax,%edi+ rorl $6,%r13d+ movl %eax,%r10d++ andl %edi,%r15d+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %r15d,%r10d+ addl %r12d,%ecx+ addl %r12d,%r10d++ leaq 4(%rbp),%rbp+ movl 12(%rsp),%r13d+ movl 0(%rsp),%r15d++ movl %r13d,%r12d+ rorl $11,%r13d+ addl %r14d,%r10d+ movl %r15d,%r14d+ rorl $2,%r15d++ xorl %r12d,%r13d+ shrl $3,%r12d+ rorl $7,%r13d+ xorl %r14d,%r15d+ shrl $10,%r14d++ rorl $17,%r15d+ xorl %r13d,%r12d+ xorl %r14d,%r15d+ addl 44(%rsp),%r12d++ addl 8(%rsp),%r12d+ movl %ecx,%r13d+ addl %r15d,%r12d+ movl %r10d,%r14d+ rorl $14,%r13d+ movl %edx,%r15d++ xorl %ecx,%r13d+ rorl $9,%r14d+ xorl %r8d,%r15d++ movl %r12d,8(%rsp)+ xorl %r10d,%r14d+ andl %ecx,%r15d++ rorl $5,%r13d+ addl %r9d,%r12d+ xorl %r8d,%r15d++ rorl $11,%r14d+ xorl %ecx,%r13d+ addl %r15d,%r12d++ movl %r10d,%r15d+ addl (%rbp),%r12d+ xorl %r10d,%r14d++ xorl %r11d,%r15d+ rorl $6,%r13d+ movl %r11d,%r9d++ andl %r15d,%edi+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %edi,%r9d+ addl %r12d,%ebx+ addl %r12d,%r9d++ leaq 4(%rbp),%rbp+ movl 16(%rsp),%r13d+ movl 4(%rsp),%edi++ movl %r13d,%r12d+ rorl $11,%r13d+ addl %r14d,%r9d+ movl %edi,%r14d+ rorl $2,%edi++ xorl %r12d,%r13d+ shrl $3,%r12d+ rorl $7,%r13d+ xorl %r14d,%edi+ shrl $10,%r14d++ rorl $17,%edi+ xorl %r13d,%r12d+ xorl %r14d,%edi+ addl 48(%rsp),%r12d++ addl 12(%rsp),%r12d+ movl %ebx,%r13d+ addl %edi,%r12d+ movl %r9d,%r14d+ rorl $14,%r13d+ movl %ecx,%edi++ xorl %ebx,%r13d+ rorl $9,%r14d+ xorl %edx,%edi++ movl %r12d,12(%rsp)+ xorl %r9d,%r14d+ andl %ebx,%edi++ rorl $5,%r13d+ addl %r8d,%r12d+ xorl %edx,%edi++ rorl $11,%r14d+ xorl %ebx,%r13d+ addl %edi,%r12d++ movl %r9d,%edi+ addl (%rbp),%r12d+ xorl %r9d,%r14d++ xorl %r10d,%edi+ rorl $6,%r13d+ movl %r10d,%r8d++ andl %edi,%r15d+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %r15d,%r8d+ addl %r12d,%eax+ addl %r12d,%r8d++ leaq 20(%rbp),%rbp+ movl 20(%rsp),%r13d+ movl 8(%rsp),%r15d++ movl %r13d,%r12d+ rorl $11,%r13d+ addl %r14d,%r8d+ movl %r15d,%r14d+ rorl $2,%r15d++ xorl %r12d,%r13d+ shrl $3,%r12d+ rorl $7,%r13d+ xorl %r14d,%r15d+ shrl $10,%r14d++ rorl $17,%r15d+ xorl %r13d,%r12d+ xorl %r14d,%r15d+ addl 52(%rsp),%r12d++ addl 16(%rsp),%r12d+ movl %eax,%r13d+ addl %r15d,%r12d+ movl %r8d,%r14d+ rorl $14,%r13d+ movl %ebx,%r15d++ xorl %eax,%r13d+ rorl $9,%r14d+ xorl %ecx,%r15d++ movl %r12d,16(%rsp)+ xorl %r8d,%r14d+ andl %eax,%r15d++ rorl $5,%r13d+ addl %edx,%r12d+ xorl %ecx,%r15d++ rorl $11,%r14d+ xorl %eax,%r13d+ addl %r15d,%r12d++ movl %r8d,%r15d+ addl (%rbp),%r12d+ xorl %r8d,%r14d++ xorl %r9d,%r15d+ rorl $6,%r13d+ movl %r9d,%edx++ andl %r15d,%edi+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %edi,%edx+ addl %r12d,%r11d+ addl %r12d,%edx++ leaq 4(%rbp),%rbp+ movl 24(%rsp),%r13d+ movl 12(%rsp),%edi++ movl %r13d,%r12d+ rorl $11,%r13d+ addl %r14d,%edx+ movl %edi,%r14d+ rorl $2,%edi++ xorl %r12d,%r13d+ shrl $3,%r12d+ rorl $7,%r13d+ xorl %r14d,%edi+ shrl $10,%r14d++ rorl $17,%edi+ xorl %r13d,%r12d+ xorl %r14d,%edi+ addl 56(%rsp),%r12d++ addl 20(%rsp),%r12d+ movl %r11d,%r13d+ addl %edi,%r12d+ movl %edx,%r14d+ rorl $14,%r13d+ movl %eax,%edi++ xorl %r11d,%r13d+ rorl $9,%r14d+ xorl %ebx,%edi++ movl %r12d,20(%rsp)+ xorl %edx,%r14d+ andl %r11d,%edi++ rorl $5,%r13d+ addl %ecx,%r12d+ xorl %ebx,%edi++ rorl $11,%r14d+ xorl %r11d,%r13d+ addl %edi,%r12d++ movl %edx,%edi+ addl (%rbp),%r12d+ xorl %edx,%r14d++ xorl %r8d,%edi+ rorl $6,%r13d+ movl %r8d,%ecx++ andl %edi,%r15d+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %r15d,%ecx+ addl %r12d,%r10d+ addl %r12d,%ecx++ leaq 4(%rbp),%rbp+ movl 28(%rsp),%r13d+ movl 16(%rsp),%r15d++ movl %r13d,%r12d+ rorl $11,%r13d+ addl %r14d,%ecx+ movl %r15d,%r14d+ rorl $2,%r15d++ xorl %r12d,%r13d+ shrl $3,%r12d+ rorl $7,%r13d+ xorl %r14d,%r15d+ shrl $10,%r14d++ rorl $17,%r15d+ xorl %r13d,%r12d+ xorl %r14d,%r15d+ addl 60(%rsp),%r12d++ addl 24(%rsp),%r12d+ movl %r10d,%r13d+ addl %r15d,%r12d+ movl %ecx,%r14d+ rorl $14,%r13d+ movl %r11d,%r15d++ xorl %r10d,%r13d+ rorl $9,%r14d+ xorl %eax,%r15d++ movl %r12d,24(%rsp)+ xorl %ecx,%r14d+ andl %r10d,%r15d++ rorl $5,%r13d+ addl %ebx,%r12d+ xorl %eax,%r15d++ rorl $11,%r14d+ xorl %r10d,%r13d+ addl %r15d,%r12d++ movl %ecx,%r15d+ addl (%rbp),%r12d+ xorl %ecx,%r14d++ xorl %edx,%r15d+ rorl $6,%r13d+ movl %edx,%ebx++ andl %r15d,%edi+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %edi,%ebx+ addl %r12d,%r9d+ addl %r12d,%ebx++ leaq 4(%rbp),%rbp+ movl 32(%rsp),%r13d+ movl 20(%rsp),%edi++ movl %r13d,%r12d+ rorl $11,%r13d+ addl %r14d,%ebx+ movl %edi,%r14d+ rorl $2,%edi++ xorl %r12d,%r13d+ shrl $3,%r12d+ rorl $7,%r13d+ xorl %r14d,%edi+ shrl $10,%r14d++ rorl $17,%edi+ xorl %r13d,%r12d+ xorl %r14d,%edi+ addl 0(%rsp),%r12d++ addl 28(%rsp),%r12d+ movl %r9d,%r13d+ addl %edi,%r12d+ movl %ebx,%r14d+ rorl $14,%r13d+ movl %r10d,%edi++ xorl %r9d,%r13d+ rorl $9,%r14d+ xorl %r11d,%edi++ movl %r12d,28(%rsp)+ xorl %ebx,%r14d+ andl %r9d,%edi++ rorl $5,%r13d+ addl %eax,%r12d+ xorl %r11d,%edi++ rorl $11,%r14d+ xorl %r9d,%r13d+ addl %edi,%r12d++ movl %ebx,%edi+ addl (%rbp),%r12d+ xorl %ebx,%r14d++ xorl %ecx,%edi+ rorl $6,%r13d+ movl %ecx,%eax++ andl %edi,%r15d+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %r15d,%eax+ addl %r12d,%r8d+ addl %r12d,%eax++ leaq 20(%rbp),%rbp+ movl 36(%rsp),%r13d+ movl 24(%rsp),%r15d++ movl %r13d,%r12d+ rorl $11,%r13d+ addl %r14d,%eax+ movl %r15d,%r14d+ rorl $2,%r15d++ xorl %r12d,%r13d+ shrl $3,%r12d+ rorl $7,%r13d+ xorl %r14d,%r15d+ shrl $10,%r14d++ rorl $17,%r15d+ xorl %r13d,%r12d+ xorl %r14d,%r15d+ addl 4(%rsp),%r12d++ addl 32(%rsp),%r12d+ movl %r8d,%r13d+ addl %r15d,%r12d+ movl %eax,%r14d+ rorl $14,%r13d+ movl %r9d,%r15d++ xorl %r8d,%r13d+ rorl $9,%r14d+ xorl %r10d,%r15d++ movl %r12d,32(%rsp)+ xorl %eax,%r14d+ andl %r8d,%r15d++ rorl $5,%r13d+ addl %r11d,%r12d+ xorl %r10d,%r15d++ rorl $11,%r14d+ xorl %r8d,%r13d+ addl %r15d,%r12d++ movl %eax,%r15d+ addl (%rbp),%r12d+ xorl %eax,%r14d++ xorl %ebx,%r15d+ rorl $6,%r13d+ movl %ebx,%r11d++ andl %r15d,%edi+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %edi,%r11d+ addl %r12d,%edx+ addl %r12d,%r11d++ leaq 4(%rbp),%rbp+ movl 40(%rsp),%r13d+ movl 28(%rsp),%edi++ movl %r13d,%r12d+ rorl $11,%r13d+ addl %r14d,%r11d+ movl %edi,%r14d+ rorl $2,%edi++ xorl %r12d,%r13d+ shrl $3,%r12d+ rorl $7,%r13d+ xorl %r14d,%edi+ shrl $10,%r14d++ rorl $17,%edi+ xorl %r13d,%r12d+ xorl %r14d,%edi+ addl 8(%rsp),%r12d++ addl 36(%rsp),%r12d+ movl %edx,%r13d+ addl %edi,%r12d+ movl %r11d,%r14d+ rorl $14,%r13d+ movl %r8d,%edi++ xorl %edx,%r13d+ rorl $9,%r14d+ xorl %r9d,%edi++ movl %r12d,36(%rsp)+ xorl %r11d,%r14d+ andl %edx,%edi++ rorl $5,%r13d+ addl %r10d,%r12d+ xorl %r9d,%edi++ rorl $11,%r14d+ xorl %edx,%r13d+ addl %edi,%r12d++ movl %r11d,%edi+ addl (%rbp),%r12d+ xorl %r11d,%r14d++ xorl %eax,%edi+ rorl $6,%r13d+ movl %eax,%r10d++ andl %edi,%r15d+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %r15d,%r10d+ addl %r12d,%ecx+ addl %r12d,%r10d++ leaq 4(%rbp),%rbp+ movl 44(%rsp),%r13d+ movl 32(%rsp),%r15d++ movl %r13d,%r12d+ rorl $11,%r13d+ addl %r14d,%r10d+ movl %r15d,%r14d+ rorl $2,%r15d++ xorl %r12d,%r13d+ shrl $3,%r12d+ rorl $7,%r13d+ xorl %r14d,%r15d+ shrl $10,%r14d++ rorl $17,%r15d+ xorl %r13d,%r12d+ xorl %r14d,%r15d+ addl 12(%rsp),%r12d++ addl 40(%rsp),%r12d+ movl %ecx,%r13d+ addl %r15d,%r12d+ movl %r10d,%r14d+ rorl $14,%r13d+ movl %edx,%r15d++ xorl %ecx,%r13d+ rorl $9,%r14d+ xorl %r8d,%r15d++ movl %r12d,40(%rsp)+ xorl %r10d,%r14d+ andl %ecx,%r15d++ rorl $5,%r13d+ addl %r9d,%r12d+ xorl %r8d,%r15d++ rorl $11,%r14d+ xorl %ecx,%r13d+ addl %r15d,%r12d++ movl %r10d,%r15d+ addl (%rbp),%r12d+ xorl %r10d,%r14d++ xorl %r11d,%r15d+ rorl $6,%r13d+ movl %r11d,%r9d++ andl %r15d,%edi+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %edi,%r9d+ addl %r12d,%ebx+ addl %r12d,%r9d++ leaq 4(%rbp),%rbp+ movl 48(%rsp),%r13d+ movl 36(%rsp),%edi++ movl %r13d,%r12d+ rorl $11,%r13d+ addl %r14d,%r9d+ movl %edi,%r14d+ rorl $2,%edi++ xorl %r12d,%r13d+ shrl $3,%r12d+ rorl $7,%r13d+ xorl %r14d,%edi+ shrl $10,%r14d++ rorl $17,%edi+ xorl %r13d,%r12d+ xorl %r14d,%edi+ addl 16(%rsp),%r12d++ addl 44(%rsp),%r12d+ movl %ebx,%r13d+ addl %edi,%r12d+ movl %r9d,%r14d+ rorl $14,%r13d+ movl %ecx,%edi++ xorl %ebx,%r13d+ rorl $9,%r14d+ xorl %edx,%edi++ movl %r12d,44(%rsp)+ xorl %r9d,%r14d+ andl %ebx,%edi++ rorl $5,%r13d+ addl %r8d,%r12d+ xorl %edx,%edi++ rorl $11,%r14d+ xorl %ebx,%r13d+ addl %edi,%r12d++ movl %r9d,%edi+ addl (%rbp),%r12d+ xorl %r9d,%r14d++ xorl %r10d,%edi+ rorl $6,%r13d+ movl %r10d,%r8d++ andl %edi,%r15d+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %r15d,%r8d+ addl %r12d,%eax+ addl %r12d,%r8d++ leaq 20(%rbp),%rbp+ movl 52(%rsp),%r13d+ movl 40(%rsp),%r15d++ movl %r13d,%r12d+ rorl $11,%r13d+ addl %r14d,%r8d+ movl %r15d,%r14d+ rorl $2,%r15d++ xorl %r12d,%r13d+ shrl $3,%r12d+ rorl $7,%r13d+ xorl %r14d,%r15d+ shrl $10,%r14d++ rorl $17,%r15d+ xorl %r13d,%r12d+ xorl %r14d,%r15d+ addl 20(%rsp),%r12d++ addl 48(%rsp),%r12d+ movl %eax,%r13d+ addl %r15d,%r12d+ movl %r8d,%r14d+ rorl $14,%r13d+ movl %ebx,%r15d++ xorl %eax,%r13d+ rorl $9,%r14d+ xorl %ecx,%r15d++ movl %r12d,48(%rsp)+ xorl %r8d,%r14d+ andl %eax,%r15d++ rorl $5,%r13d+ addl %edx,%r12d+ xorl %ecx,%r15d++ rorl $11,%r14d+ xorl %eax,%r13d+ addl %r15d,%r12d++ movl %r8d,%r15d+ addl (%rbp),%r12d+ xorl %r8d,%r14d++ xorl %r9d,%r15d+ rorl $6,%r13d+ movl %r9d,%edx++ andl %r15d,%edi+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %edi,%edx+ addl %r12d,%r11d+ addl %r12d,%edx++ leaq 4(%rbp),%rbp+ movl 56(%rsp),%r13d+ movl 44(%rsp),%edi++ movl %r13d,%r12d+ rorl $11,%r13d+ addl %r14d,%edx+ movl %edi,%r14d+ rorl $2,%edi++ xorl %r12d,%r13d+ shrl $3,%r12d+ rorl $7,%r13d+ xorl %r14d,%edi+ shrl $10,%r14d++ rorl $17,%edi+ xorl %r13d,%r12d+ xorl %r14d,%edi+ addl 24(%rsp),%r12d++ addl 52(%rsp),%r12d+ movl %r11d,%r13d+ addl %edi,%r12d+ movl %edx,%r14d+ rorl $14,%r13d+ movl %eax,%edi++ xorl %r11d,%r13d+ rorl $9,%r14d+ xorl %ebx,%edi++ movl %r12d,52(%rsp)+ xorl %edx,%r14d+ andl %r11d,%edi++ rorl $5,%r13d+ addl %ecx,%r12d+ xorl %ebx,%edi++ rorl $11,%r14d+ xorl %r11d,%r13d+ addl %edi,%r12d++ movl %edx,%edi+ addl (%rbp),%r12d+ xorl %edx,%r14d++ xorl %r8d,%edi+ rorl $6,%r13d+ movl %r8d,%ecx++ andl %edi,%r15d+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %r15d,%ecx+ addl %r12d,%r10d+ addl %r12d,%ecx++ leaq 4(%rbp),%rbp+ movl 60(%rsp),%r13d+ movl 48(%rsp),%r15d++ movl %r13d,%r12d+ rorl $11,%r13d+ addl %r14d,%ecx+ movl %r15d,%r14d+ rorl $2,%r15d++ xorl %r12d,%r13d+ shrl $3,%r12d+ rorl $7,%r13d+ xorl %r14d,%r15d+ shrl $10,%r14d++ rorl $17,%r15d+ xorl %r13d,%r12d+ xorl %r14d,%r15d+ addl 28(%rsp),%r12d++ addl 56(%rsp),%r12d+ movl %r10d,%r13d+ addl %r15d,%r12d+ movl %ecx,%r14d+ rorl $14,%r13d+ movl %r11d,%r15d++ xorl %r10d,%r13d+ rorl $9,%r14d+ xorl %eax,%r15d++ movl %r12d,56(%rsp)+ xorl %ecx,%r14d+ andl %r10d,%r15d++ rorl $5,%r13d+ addl %ebx,%r12d+ xorl %eax,%r15d++ rorl $11,%r14d+ xorl %r10d,%r13d+ addl %r15d,%r12d++ movl %ecx,%r15d+ addl (%rbp),%r12d+ xorl %ecx,%r14d++ xorl %edx,%r15d+ rorl $6,%r13d+ movl %edx,%ebx++ andl %r15d,%edi+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %edi,%ebx+ addl %r12d,%r9d+ addl %r12d,%ebx++ leaq 4(%rbp),%rbp+ movl 0(%rsp),%r13d+ movl 52(%rsp),%edi++ movl %r13d,%r12d+ rorl $11,%r13d+ addl %r14d,%ebx+ movl %edi,%r14d+ rorl $2,%edi++ xorl %r12d,%r13d+ shrl $3,%r12d+ rorl $7,%r13d+ xorl %r14d,%edi+ shrl $10,%r14d++ rorl $17,%edi+ xorl %r13d,%r12d+ xorl %r14d,%edi+ addl 32(%rsp),%r12d++ addl 60(%rsp),%r12d+ movl %r9d,%r13d+ addl %edi,%r12d+ movl %ebx,%r14d+ rorl $14,%r13d+ movl %r10d,%edi++ xorl %r9d,%r13d+ rorl $9,%r14d+ xorl %r11d,%edi++ movl %r12d,60(%rsp)+ xorl %ebx,%r14d+ andl %r9d,%edi++ rorl $5,%r13d+ addl %eax,%r12d+ xorl %r11d,%edi++ rorl $11,%r14d+ xorl %r9d,%r13d+ addl %edi,%r12d++ movl %ebx,%edi+ addl (%rbp),%r12d+ xorl %ebx,%r14d++ xorl %ecx,%edi+ rorl $6,%r13d+ movl %ecx,%eax++ andl %edi,%r15d+ rorl $2,%r14d+ addl %r13d,%r12d++ xorl %r15d,%eax+ addl %r12d,%r8d+ addl %r12d,%eax++ leaq 20(%rbp),%rbp+ cmpb $0,3(%rbp)+ jnz .Lrounds_16_xx++ movq 64+0(%rsp),%rdi+ addl %r14d,%eax+ leaq 64(%rsi),%rsi++ addl 0(%rdi),%eax+ addl 4(%rdi),%ebx+ addl 8(%rdi),%ecx+ addl 12(%rdi),%edx+ addl 16(%rdi),%r8d+ addl 20(%rdi),%r9d+ addl 24(%rdi),%r10d+ addl 28(%rdi),%r11d++ cmpq 64+16(%rsp),%rsi++ movl %eax,0(%rdi)+ movl %ebx,4(%rdi)+ movl %ecx,8(%rdi)+ movl %edx,12(%rdi)+ movl %r8d,16(%rdi)+ movl %r9d,20(%rdi)+ movl %r10d,24(%rdi)+ movl %r11d,28(%rdi)+ jb .Lloop++ leaq 64+24+48(%rsp),%r11++ movq 64+24(%rsp),%r15+ movq -40(%r11),%r14+ movq -32(%r11),%r13+ movq -24(%r11),%r12+ movq -16(%r11),%rbx+ movq -8(%r11),%rbp+.LSEH_epilogue_crypton_sha256_asm_block_data_order:+ mov 8(%r11),%rdi+ mov 16(%r11),%rsi++ leaq (%r11),%rsp+ .byte 0xf3,0xc3++.LSEH_end_crypton_sha256_asm_block_data_order:+.p2align 6++K256:+.long 0x428a2f98,0x71374491,0xb5c0fbcf,0xe9b5dba5+.long 0x428a2f98,0x71374491,0xb5c0fbcf,0xe9b5dba5+.long 0x3956c25b,0x59f111f1,0x923f82a4,0xab1c5ed5+.long 0x3956c25b,0x59f111f1,0x923f82a4,0xab1c5ed5+.long 0xd807aa98,0x12835b01,0x243185be,0x550c7dc3+.long 0xd807aa98,0x12835b01,0x243185be,0x550c7dc3+.long 0x72be5d74,0x80deb1fe,0x9bdc06a7,0xc19bf174+.long 0x72be5d74,0x80deb1fe,0x9bdc06a7,0xc19bf174+.long 0xe49b69c1,0xefbe4786,0x0fc19dc6,0x240ca1cc+.long 0xe49b69c1,0xefbe4786,0x0fc19dc6,0x240ca1cc+.long 0x2de92c6f,0x4a7484aa,0x5cb0a9dc,0x76f988da+.long 0x2de92c6f,0x4a7484aa,0x5cb0a9dc,0x76f988da+.long 0x983e5152,0xa831c66d,0xb00327c8,0xbf597fc7+.long 0x983e5152,0xa831c66d,0xb00327c8,0xbf597fc7+.long 0xc6e00bf3,0xd5a79147,0x06ca6351,0x14292967+.long 0xc6e00bf3,0xd5a79147,0x06ca6351,0x14292967+.long 0x27b70a85,0x2e1b2138,0x4d2c6dfc,0x53380d13+.long 0x27b70a85,0x2e1b2138,0x4d2c6dfc,0x53380d13+.long 0x650a7354,0x766a0abb,0x81c2c92e,0x92722c85+.long 0x650a7354,0x766a0abb,0x81c2c92e,0x92722c85+.long 0xa2bfe8a1,0xa81a664b,0xc24b8b70,0xc76c51a3+.long 0xa2bfe8a1,0xa81a664b,0xc24b8b70,0xc76c51a3+.long 0xd192e819,0xd6990624,0xf40e3585,0x106aa070+.long 0xd192e819,0xd6990624,0xf40e3585,0x106aa070+.long 0x19a4c116,0x1e376c08,0x2748774c,0x34b0bcb5+.long 0x19a4c116,0x1e376c08,0x2748774c,0x34b0bcb5+.long 0x391c0cb3,0x4ed8aa4a,0x5b9cca4f,0x682e6ff3+.long 0x391c0cb3,0x4ed8aa4a,0x5b9cca4f,0x682e6ff3+.long 0x748f82ee,0x78a5636f,0x84c87814,0x8cc70208+.long 0x748f82ee,0x78a5636f,0x84c87814,0x8cc70208+.long 0x90befffa,0xa4506ceb,0xbef9a3f7,0xc67178f2+.long 0x90befffa,0xa4506ceb,0xbef9a3f7,0xc67178f2++.long 0x00010203,0x04050607,0x08090a0b,0x0c0d0e0f+.long 0x00010203,0x04050607,0x08090a0b,0x0c0d0e0f+.long 0x03020100,0x0b0a0908,0xffffffff,0xffffffff+.long 0x03020100,0x0b0a0908,0xffffffff,0xffffffff+.long 0xffffffff,0xffffffff,0x03020100,0x0b0a0908+.long 0xffffffff,0xffffffff,0x03020100,0x0b0a0908+.byte 83,72,65,50,53,54,32,98,108,111,99,107,32,116,114,97,110,115,102,111,114,109,32,102,111,114,32,120,56,54,95,54,52,44,32,67,82,89,80,84,79,71,65,77,83,32,98,121,32,64,100,111,116,45,97,115,109,0+.def crypton_sha256_asm_block_data_order_shaext; .scl 3; .type 32; .endef+.p2align 6+crypton_sha256_asm_block_data_order_shaext:+ .byte 0xf3,0x0f,0x1e,0xfa+ movq %rdi,8(%rsp)+ movq %rsi,16(%rsp)+ movq %rsp,%r11+.LSEH_begin_crypton_sha256_asm_block_data_order_shaext:+++ pushq %rbp++ movq %rsp,%rbp++ movq %rcx,%rdi+ movq %rdx,%rsi+ movq %r8,%rdx+.Lshaext_shortcut:+ subq $0x50,%rsp++ movaps %xmm6,-80(%rbp)+ movaps %xmm7,-64(%rbp)+ movaps %xmm8,-48(%rbp)+ movaps %xmm9,-32(%rbp)+ movaps %xmm10,-16(%rbp)++.LSEH_body_crypton_sha256_asm_block_data_order_shaext:++ leaq K256+128(%rip),%rcx+ movdqu (%rdi),%xmm1+ movdqu 16(%rdi),%xmm2+ movdqa 512-128(%rcx),%xmm7++ pshufd $0x1b,%xmm1,%xmm0+ pshufd $0xb1,%xmm1,%xmm1+ pshufd $0x1b,%xmm2,%xmm2+ movdqa %xmm7,%xmm8+.byte 102,15,58,15,202,8+ punpcklqdq %xmm0,%xmm2+ jmp .Loop_shaext++.p2align 4+.Loop_shaext:+ movdqu (%rsi),%xmm3+ movdqu 16(%rsi),%xmm4+ movdqu 32(%rsi),%xmm5+.byte 102,15,56,0,223+ movdqu 48(%rsi),%xmm6++ movdqa 0-128(%rcx),%xmm0+ paddd %xmm3,%xmm0+.byte 102,15,56,0,231+ movdqa %xmm2,%xmm10+.byte 15,56,203,209+ pshufd $0x0e,%xmm0,%xmm0+ nop+ movdqa %xmm1,%xmm9+.byte 15,56,203,202++ movdqa 32-128(%rcx),%xmm0+ paddd %xmm4,%xmm0+.byte 102,15,56,0,239+.byte 15,56,203,209+ pshufd $0x0e,%xmm0,%xmm0+ leaq 64(%rsi),%rsi+.byte 15,56,204,220+.byte 15,56,203,202++ movdqa 64-128(%rcx),%xmm0+ paddd %xmm5,%xmm0+.byte 102,15,56,0,247+.byte 15,56,203,209+ pshufd $0x0e,%xmm0,%xmm0+ movdqa %xmm6,%xmm7+.byte 102,15,58,15,253,4+ nop+ paddd %xmm7,%xmm3+.byte 15,56,204,229+.byte 15,56,203,202++ movdqa 96-128(%rcx),%xmm0+ paddd %xmm6,%xmm0+.byte 15,56,205,222+.byte 15,56,203,209+ pshufd $0x0e,%xmm0,%xmm0+ movdqa %xmm3,%xmm7+.byte 102,15,58,15,254,4+ nop+ paddd %xmm7,%xmm4+.byte 15,56,204,238+.byte 15,56,203,202+ movdqa 128-128(%rcx),%xmm0+ paddd %xmm3,%xmm0+.byte 15,56,205,227+.byte 15,56,203,209+ pshufd $0x0e,%xmm0,%xmm0+ movdqa %xmm4,%xmm7+.byte 102,15,58,15,251,4+ nop+ paddd %xmm7,%xmm5+.byte 15,56,204,243+.byte 15,56,203,202+ movdqa 160-128(%rcx),%xmm0+ paddd %xmm4,%xmm0+.byte 15,56,205,236+.byte 15,56,203,209+ pshufd $0x0e,%xmm0,%xmm0+ movdqa %xmm5,%xmm7+.byte 102,15,58,15,252,4+ nop+ paddd %xmm7,%xmm6+.byte 15,56,204,220+.byte 15,56,203,202+ movdqa 192-128(%rcx),%xmm0+ paddd %xmm5,%xmm0+.byte 15,56,205,245+.byte 15,56,203,209+ pshufd $0x0e,%xmm0,%xmm0+ movdqa %xmm6,%xmm7+.byte 102,15,58,15,253,4+ nop+ paddd %xmm7,%xmm3+.byte 15,56,204,229+.byte 15,56,203,202+ movdqa 224-128(%rcx),%xmm0+ paddd %xmm6,%xmm0+.byte 15,56,205,222+.byte 15,56,203,209+ pshufd $0x0e,%xmm0,%xmm0+ movdqa %xmm3,%xmm7+.byte 102,15,58,15,254,4+ nop+ paddd %xmm7,%xmm4+.byte 15,56,204,238+.byte 15,56,203,202+ movdqa 256-128(%rcx),%xmm0+ paddd %xmm3,%xmm0+.byte 15,56,205,227+.byte 15,56,203,209+ pshufd $0x0e,%xmm0,%xmm0+ movdqa %xmm4,%xmm7+.byte 102,15,58,15,251,4+ nop+ paddd %xmm7,%xmm5+.byte 15,56,204,243+.byte 15,56,203,202+ movdqa 288-128(%rcx),%xmm0+ paddd %xmm4,%xmm0+.byte 15,56,205,236+.byte 15,56,203,209+ pshufd $0x0e,%xmm0,%xmm0+ movdqa %xmm5,%xmm7+.byte 102,15,58,15,252,4+ nop+ paddd %xmm7,%xmm6+.byte 15,56,204,220+.byte 15,56,203,202+ movdqa 320-128(%rcx),%xmm0+ paddd %xmm5,%xmm0+.byte 15,56,205,245+.byte 15,56,203,209+ pshufd $0x0e,%xmm0,%xmm0+ movdqa %xmm6,%xmm7+.byte 102,15,58,15,253,4+ nop+ paddd %xmm7,%xmm3+.byte 15,56,204,229+.byte 15,56,203,202+ movdqa 352-128(%rcx),%xmm0+ paddd %xmm6,%xmm0+.byte 15,56,205,222+.byte 15,56,203,209+ pshufd $0x0e,%xmm0,%xmm0+ movdqa %xmm3,%xmm7+.byte 102,15,58,15,254,4+ nop+ paddd %xmm7,%xmm4+.byte 15,56,204,238+.byte 15,56,203,202+ movdqa 384-128(%rcx),%xmm0+ paddd %xmm3,%xmm0+.byte 15,56,205,227+.byte 15,56,203,209+ pshufd $0x0e,%xmm0,%xmm0+ movdqa %xmm4,%xmm7+.byte 102,15,58,15,251,4+ nop+ paddd %xmm7,%xmm5+.byte 15,56,204,243+.byte 15,56,203,202+ movdqa 416-128(%rcx),%xmm0+ paddd %xmm4,%xmm0+.byte 15,56,205,236+.byte 15,56,203,209+ pshufd $0x0e,%xmm0,%xmm0+ movdqa %xmm5,%xmm7+.byte 102,15,58,15,252,4+.byte 15,56,203,202+ paddd %xmm7,%xmm6++ movdqa 448-128(%rcx),%xmm0+ paddd %xmm5,%xmm0+.byte 15,56,203,209+ pshufd $0x0e,%xmm0,%xmm0+.byte 15,56,205,245+ movdqa %xmm8,%xmm7+.byte 15,56,203,202++ movdqa 480-128(%rcx),%xmm0+ paddd %xmm6,%xmm0+ nop+.byte 15,56,203,209+ pshufd $0x0e,%xmm0,%xmm0+ decq %rdx+ nop+.byte 15,56,203,202++ paddd %xmm10,%xmm2+ paddd %xmm9,%xmm1+ jnz .Loop_shaext++ pshufd $0xb1,%xmm2,%xmm2+ pshufd $0x1b,%xmm1,%xmm7+ pshufd $0xb1,%xmm1,%xmm1+ punpckhqdq %xmm2,%xmm1+.byte 102,15,58,15,215,8++ movdqu %xmm1,(%rdi)+ movdqu %xmm2,16(%rdi)+ movaps -80(%rbp),%xmm6+ movaps -64(%rbp),%xmm7+ movaps -48(%rbp),%xmm8+ movaps -32(%rbp),%xmm9+ movaps -16(%rbp),%xmm10+ movq %rbp,%rsp++ popq %rbp++.LSEH_epilogue_crypton_sha256_asm_block_data_order_shaext:+ mov 8(%rsp),%rdi+ mov 16(%rsp),%rsi++ .byte 0xf3,0xc3++.LSEH_end_crypton_sha256_asm_block_data_order_shaext:+.def crypton_sha256_asm_block_data_order_ssse3; .scl 3; .type 32; .endef+.p2align 6+crypton_sha256_asm_block_data_order_ssse3:+ .byte 0xf3,0x0f,0x1e,0xfa+ movq %rdi,8(%rsp)+ movq %rsi,16(%rsp)+ movq %rsp,%r11+.LSEH_begin_crypton_sha256_asm_block_data_order_ssse3:+++ pushq %rbp++ movq %rsp,%rbp++ movq %rcx,%rdi+ movq %rdx,%rsi+ movq %r8,%rdx+.Lssse3_shortcut:+ pushq %rbx++ pushq %r12++ pushq %r13++ pushq %r14++ pushq %r15++ shlq $4,%rdx+ subq $88,%rsp++ leaq (%rsi,%rdx,4),%rdx+ movq %rdi,-64(%rbp)++ movq %rdx,-48(%rbp)+ movaps %xmm6,-128(%rbp)+ movaps %xmm7,-112(%rbp)+ movaps %xmm8,-96(%rbp)+ movaps %xmm9,-80(%rbp)++.LSEH_body_crypton_sha256_asm_block_data_order_ssse3:+++ leaq -64(%rsp),%rsp+ movl 0(%rdi),%eax+ andq $-64,%rsp+ movl 4(%rdi),%ebx+ movl 8(%rdi),%ecx+ movl 12(%rdi),%edx+ movl 16(%rdi),%r8d+ movl 20(%rdi),%r9d+ movl 24(%rdi),%r10d+ movl 28(%rdi),%r11d+++ jmp .Lloop_ssse3+.p2align 4+.Lloop_ssse3:+ movdqa K256+512(%rip),%xmm7+ movq %rsi,-56(%rbp)+ movdqu 0(%rsi),%xmm0+ movdqu 16(%rsi),%xmm1+ movdqu 32(%rsi),%xmm2+.byte 102,15,56,0,199+ movdqu 48(%rsi),%xmm3+ leaq K256(%rip),%rsi+.byte 102,15,56,0,207+ movdqa 0(%rsi),%xmm4+ movdqa 32(%rsi),%xmm5+.byte 102,15,56,0,215+ paddd %xmm0,%xmm4+ movdqa 64(%rsi),%xmm6+.byte 102,15,56,0,223+ movdqa 96(%rsi),%xmm7+ paddd %xmm1,%xmm5+ paddd %xmm2,%xmm6+ paddd %xmm3,%xmm7+ movdqa %xmm4,0(%rsp)+ movl %eax,%r14d+ movdqa %xmm5,16(%rsp)+ movl %ebx,%edi+ movdqa %xmm6,32(%rsp)+ xorl %ecx,%edi+ movdqa %xmm7,48(%rsp)+ movl %r8d,%r13d+ jmp .Lssse3_00_47++.p2align 4+.Lssse3_00_47:+ subq $-128,%rsi+ rorl $14,%r13d+ movdqa %xmm1,%xmm4+ movl %r14d,%eax+ movl %r9d,%r12d+ movdqa %xmm3,%xmm7+ rorl $9,%r14d+ xorl %r8d,%r13d+ xorl %r10d,%r12d+ rorl $5,%r13d+ xorl %eax,%r14d+.byte 102,15,58,15,224,4+ andl %r8d,%r12d+ xorl %r8d,%r13d+.byte 102,15,58,15,250,4+ addl 0(%rsp),%r11d+ movl %eax,%r15d+ xorl %r10d,%r12d+ rorl $11,%r14d+ movdqa %xmm4,%xmm5+ xorl %ebx,%r15d+ addl %r12d,%r11d+ movdqa %xmm4,%xmm6+ rorl $6,%r13d+ andl %r15d,%edi+ psrld $3,%xmm4+ xorl %eax,%r14d+ addl %r13d,%r11d+ xorl %ebx,%edi+ paddd %xmm7,%xmm0+ rorl $2,%r14d+ addl %r11d,%edx+ psrld $7,%xmm6+ addl %edi,%r11d+ movl %edx,%r13d+ pshufd $250,%xmm3,%xmm7+ addl %r11d,%r14d+ rorl $14,%r13d+ pslld $14,%xmm5+ movl %r14d,%r11d+ movl %r8d,%r12d+ pxor %xmm6,%xmm4+ rorl $9,%r14d+ xorl %edx,%r13d+ xorl %r9d,%r12d+ rorl $5,%r13d+ psrld $11,%xmm6+ xorl %r11d,%r14d+ pxor %xmm5,%xmm4+ andl %edx,%r12d+ xorl %edx,%r13d+ pslld $11,%xmm5+ addl 4(%rsp),%r10d+ movl %r11d,%edi+ pxor %xmm6,%xmm4+ xorl %r9d,%r12d+ rorl $11,%r14d+ movdqa %xmm7,%xmm6+ xorl %eax,%edi+ addl %r12d,%r10d+ pxor %xmm5,%xmm4+ rorl $6,%r13d+ andl %edi,%r15d+ xorl %r11d,%r14d+ psrld $10,%xmm7+ addl %r13d,%r10d+ xorl %eax,%r15d+ paddd %xmm4,%xmm0+ rorl $2,%r14d+ addl %r10d,%ecx+ psrlq $17,%xmm6+ addl %r15d,%r10d+ movl %ecx,%r13d+ addl %r10d,%r14d+ pxor %xmm6,%xmm7+ rorl $14,%r13d+ movl %r14d,%r10d+ movl %edx,%r12d+ rorl $9,%r14d+ psrlq $2,%xmm6+ xorl %ecx,%r13d+ xorl %r8d,%r12d+ pxor %xmm6,%xmm7+ rorl $5,%r13d+ xorl %r10d,%r14d+ andl %ecx,%r12d+ pshufd $128,%xmm7,%xmm7+ xorl %ecx,%r13d+ addl 8(%rsp),%r9d+ movl %r10d,%r15d+ psrldq $8,%xmm7+ xorl %r8d,%r12d+ rorl $11,%r14d+ xorl %r11d,%r15d+ addl %r12d,%r9d+ rorl $6,%r13d+ paddd %xmm7,%xmm0+ andl %r15d,%edi+ xorl %r10d,%r14d+ addl %r13d,%r9d+ pshufd $80,%xmm0,%xmm7+ xorl %r11d,%edi+ rorl $2,%r14d+ addl %r9d,%ebx+ movdqa %xmm7,%xmm6+ addl %edi,%r9d+ movl %ebx,%r13d+ psrld $10,%xmm7+ addl %r9d,%r14d+ rorl $14,%r13d+ psrlq $17,%xmm6+ movl %r14d,%r9d+ movl %ecx,%r12d+ pxor %xmm6,%xmm7+ rorl $9,%r14d+ xorl %ebx,%r13d+ xorl %edx,%r12d+ rorl $5,%r13d+ xorl %r9d,%r14d+ psrlq $2,%xmm6+ andl %ebx,%r12d+ xorl %ebx,%r13d+ addl 12(%rsp),%r8d+ pxor %xmm6,%xmm7+ movl %r9d,%edi+ xorl %edx,%r12d+ rorl $11,%r14d+ pshufd $8,%xmm7,%xmm7+ xorl %r10d,%edi+ addl %r12d,%r8d+ movdqa 0(%rsi),%xmm6+ rorl $6,%r13d+ andl %edi,%r15d+ pslldq $8,%xmm7+ xorl %r9d,%r14d+ addl %r13d,%r8d+ xorl %r10d,%r15d+ paddd %xmm7,%xmm0+ rorl $2,%r14d+ addl %r8d,%eax+ addl %r15d,%r8d+ paddd %xmm0,%xmm6+ movl %eax,%r13d+ addl %r8d,%r14d+ movdqa %xmm6,0(%rsp)+ rorl $14,%r13d+ movdqa %xmm2,%xmm4+ movl %r14d,%r8d+ movl %ebx,%r12d+ movdqa %xmm0,%xmm7+ rorl $9,%r14d+ xorl %eax,%r13d+ xorl %ecx,%r12d+ rorl $5,%r13d+ xorl %r8d,%r14d+.byte 102,15,58,15,225,4+ andl %eax,%r12d+ xorl %eax,%r13d+.byte 102,15,58,15,251,4+ addl 16(%rsp),%edx+ movl %r8d,%r15d+ xorl %ecx,%r12d+ rorl $11,%r14d+ movdqa %xmm4,%xmm5+ xorl %r9d,%r15d+ addl %r12d,%edx+ movdqa %xmm4,%xmm6+ rorl $6,%r13d+ andl %r15d,%edi+ psrld $3,%xmm4+ xorl %r8d,%r14d+ addl %r13d,%edx+ xorl %r9d,%edi+ paddd %xmm7,%xmm1+ rorl $2,%r14d+ addl %edx,%r11d+ psrld $7,%xmm6+ addl %edi,%edx+ movl %r11d,%r13d+ pshufd $250,%xmm0,%xmm7+ addl %edx,%r14d+ rorl $14,%r13d+ pslld $14,%xmm5+ movl %r14d,%edx+ movl %eax,%r12d+ pxor %xmm6,%xmm4+ rorl $9,%r14d+ xorl %r11d,%r13d+ xorl %ebx,%r12d+ rorl $5,%r13d+ psrld $11,%xmm6+ xorl %edx,%r14d+ pxor %xmm5,%xmm4+ andl %r11d,%r12d+ xorl %r11d,%r13d+ pslld $11,%xmm5+ addl 20(%rsp),%ecx+ movl %edx,%edi+ pxor %xmm6,%xmm4+ xorl %ebx,%r12d+ rorl $11,%r14d+ movdqa %xmm7,%xmm6+ xorl %r8d,%edi+ addl %r12d,%ecx+ pxor %xmm5,%xmm4+ rorl $6,%r13d+ andl %edi,%r15d+ xorl %edx,%r14d+ psrld $10,%xmm7+ addl %r13d,%ecx+ xorl %r8d,%r15d+ paddd %xmm4,%xmm1+ rorl $2,%r14d+ addl %ecx,%r10d+ psrlq $17,%xmm6+ addl %r15d,%ecx+ movl %r10d,%r13d+ addl %ecx,%r14d+ pxor %xmm6,%xmm7+ rorl $14,%r13d+ movl %r14d,%ecx+ movl %r11d,%r12d+ rorl $9,%r14d+ psrlq $2,%xmm6+ xorl %r10d,%r13d+ xorl %eax,%r12d+ pxor %xmm6,%xmm7+ rorl $5,%r13d+ xorl %ecx,%r14d+ andl %r10d,%r12d+ pshufd $128,%xmm7,%xmm7+ xorl %r10d,%r13d+ addl 24(%rsp),%ebx+ movl %ecx,%r15d+ psrldq $8,%xmm7+ xorl %eax,%r12d+ rorl $11,%r14d+ xorl %edx,%r15d+ addl %r12d,%ebx+ rorl $6,%r13d+ paddd %xmm7,%xmm1+ andl %r15d,%edi+ xorl %ecx,%r14d+ addl %r13d,%ebx+ pshufd $80,%xmm1,%xmm7+ xorl %edx,%edi+ rorl $2,%r14d+ addl %ebx,%r9d+ movdqa %xmm7,%xmm6+ addl %edi,%ebx+ movl %r9d,%r13d+ psrld $10,%xmm7+ addl %ebx,%r14d+ rorl $14,%r13d+ psrlq $17,%xmm6+ movl %r14d,%ebx+ movl %r10d,%r12d+ pxor %xmm6,%xmm7+ rorl $9,%r14d+ xorl %r9d,%r13d+ xorl %r11d,%r12d+ rorl $5,%r13d+ xorl %ebx,%r14d+ psrlq $2,%xmm6+ andl %r9d,%r12d+ xorl %r9d,%r13d+ addl 28(%rsp),%eax+ pxor %xmm6,%xmm7+ movl %ebx,%edi+ xorl %r11d,%r12d+ rorl $11,%r14d+ pshufd $8,%xmm7,%xmm7+ xorl %ecx,%edi+ addl %r12d,%eax+ movdqa 32(%rsi),%xmm6+ rorl $6,%r13d+ andl %edi,%r15d+ pslldq $8,%xmm7+ xorl %ebx,%r14d+ addl %r13d,%eax+ xorl %ecx,%r15d+ paddd %xmm7,%xmm1+ rorl $2,%r14d+ addl %eax,%r8d+ addl %r15d,%eax+ paddd %xmm1,%xmm6+ movl %r8d,%r13d+ addl %eax,%r14d+ movdqa %xmm6,16(%rsp)+ rorl $14,%r13d+ movdqa %xmm3,%xmm4+ movl %r14d,%eax+ movl %r9d,%r12d+ movdqa %xmm1,%xmm7+ rorl $9,%r14d+ xorl %r8d,%r13d+ xorl %r10d,%r12d+ rorl $5,%r13d+ xorl %eax,%r14d+.byte 102,15,58,15,226,4+ andl %r8d,%r12d+ xorl %r8d,%r13d+.byte 102,15,58,15,248,4+ addl 32(%rsp),%r11d+ movl %eax,%r15d+ xorl %r10d,%r12d+ rorl $11,%r14d+ movdqa %xmm4,%xmm5+ xorl %ebx,%r15d+ addl %r12d,%r11d+ movdqa %xmm4,%xmm6+ rorl $6,%r13d+ andl %r15d,%edi+ psrld $3,%xmm4+ xorl %eax,%r14d+ addl %r13d,%r11d+ xorl %ebx,%edi+ paddd %xmm7,%xmm2+ rorl $2,%r14d+ addl %r11d,%edx+ psrld $7,%xmm6+ addl %edi,%r11d+ movl %edx,%r13d+ pshufd $250,%xmm1,%xmm7+ addl %r11d,%r14d+ rorl $14,%r13d+ pslld $14,%xmm5+ movl %r14d,%r11d+ movl %r8d,%r12d+ pxor %xmm6,%xmm4+ rorl $9,%r14d+ xorl %edx,%r13d+ xorl %r9d,%r12d+ rorl $5,%r13d+ psrld $11,%xmm6+ xorl %r11d,%r14d+ pxor %xmm5,%xmm4+ andl %edx,%r12d+ xorl %edx,%r13d+ pslld $11,%xmm5+ addl 36(%rsp),%r10d+ movl %r11d,%edi+ pxor %xmm6,%xmm4+ xorl %r9d,%r12d+ rorl $11,%r14d+ movdqa %xmm7,%xmm6+ xorl %eax,%edi+ addl %r12d,%r10d+ pxor %xmm5,%xmm4+ rorl $6,%r13d+ andl %edi,%r15d+ xorl %r11d,%r14d+ psrld $10,%xmm7+ addl %r13d,%r10d+ xorl %eax,%r15d+ paddd %xmm4,%xmm2+ rorl $2,%r14d+ addl %r10d,%ecx+ psrlq $17,%xmm6+ addl %r15d,%r10d+ movl %ecx,%r13d+ addl %r10d,%r14d+ pxor %xmm6,%xmm7+ rorl $14,%r13d+ movl %r14d,%r10d+ movl %edx,%r12d+ rorl $9,%r14d+ psrlq $2,%xmm6+ xorl %ecx,%r13d+ xorl %r8d,%r12d+ pxor %xmm6,%xmm7+ rorl $5,%r13d+ xorl %r10d,%r14d+ andl %ecx,%r12d+ pshufd $128,%xmm7,%xmm7+ xorl %ecx,%r13d+ addl 40(%rsp),%r9d+ movl %r10d,%r15d+ psrldq $8,%xmm7+ xorl %r8d,%r12d+ rorl $11,%r14d+ xorl %r11d,%r15d+ addl %r12d,%r9d+ rorl $6,%r13d+ paddd %xmm7,%xmm2+ andl %r15d,%edi+ xorl %r10d,%r14d+ addl %r13d,%r9d+ pshufd $80,%xmm2,%xmm7+ xorl %r11d,%edi+ rorl $2,%r14d+ addl %r9d,%ebx+ movdqa %xmm7,%xmm6+ addl %edi,%r9d+ movl %ebx,%r13d+ psrld $10,%xmm7+ addl %r9d,%r14d+ rorl $14,%r13d+ psrlq $17,%xmm6+ movl %r14d,%r9d+ movl %ecx,%r12d+ pxor %xmm6,%xmm7+ rorl $9,%r14d+ xorl %ebx,%r13d+ xorl %edx,%r12d+ rorl $5,%r13d+ xorl %r9d,%r14d+ psrlq $2,%xmm6+ andl %ebx,%r12d+ xorl %ebx,%r13d+ addl 44(%rsp),%r8d+ pxor %xmm6,%xmm7+ movl %r9d,%edi+ xorl %edx,%r12d+ rorl $11,%r14d+ pshufd $8,%xmm7,%xmm7+ xorl %r10d,%edi+ addl %r12d,%r8d+ movdqa 64(%rsi),%xmm6+ rorl $6,%r13d+ andl %edi,%r15d+ pslldq $8,%xmm7+ xorl %r9d,%r14d+ addl %r13d,%r8d+ xorl %r10d,%r15d+ paddd %xmm7,%xmm2+ rorl $2,%r14d+ addl %r8d,%eax+ addl %r15d,%r8d+ paddd %xmm2,%xmm6+ movl %eax,%r13d+ addl %r8d,%r14d+ movdqa %xmm6,32(%rsp)+ rorl $14,%r13d+ movdqa %xmm0,%xmm4+ movl %r14d,%r8d+ movl %ebx,%r12d+ movdqa %xmm2,%xmm7+ rorl $9,%r14d+ xorl %eax,%r13d+ xorl %ecx,%r12d+ rorl $5,%r13d+ xorl %r8d,%r14d+.byte 102,15,58,15,227,4+ andl %eax,%r12d+ xorl %eax,%r13d+.byte 102,15,58,15,249,4+ addl 48(%rsp),%edx+ movl %r8d,%r15d+ xorl %ecx,%r12d+ rorl $11,%r14d+ movdqa %xmm4,%xmm5+ xorl %r9d,%r15d+ addl %r12d,%edx+ movdqa %xmm4,%xmm6+ rorl $6,%r13d+ andl %r15d,%edi+ psrld $3,%xmm4+ xorl %r8d,%r14d+ addl %r13d,%edx+ xorl %r9d,%edi+ paddd %xmm7,%xmm3+ rorl $2,%r14d+ addl %edx,%r11d+ psrld $7,%xmm6+ addl %edi,%edx+ movl %r11d,%r13d+ pshufd $250,%xmm2,%xmm7+ addl %edx,%r14d+ rorl $14,%r13d+ pslld $14,%xmm5+ movl %r14d,%edx+ movl %eax,%r12d+ pxor %xmm6,%xmm4+ rorl $9,%r14d+ xorl %r11d,%r13d+ xorl %ebx,%r12d+ rorl $5,%r13d+ psrld $11,%xmm6+ xorl %edx,%r14d+ pxor %xmm5,%xmm4+ andl %r11d,%r12d+ xorl %r11d,%r13d+ pslld $11,%xmm5+ addl 52(%rsp),%ecx+ movl %edx,%edi+ pxor %xmm6,%xmm4+ xorl %ebx,%r12d+ rorl $11,%r14d+ movdqa %xmm7,%xmm6+ xorl %r8d,%edi+ addl %r12d,%ecx+ pxor %xmm5,%xmm4+ rorl $6,%r13d+ andl %edi,%r15d+ xorl %edx,%r14d+ psrld $10,%xmm7+ addl %r13d,%ecx+ xorl %r8d,%r15d+ paddd %xmm4,%xmm3+ rorl $2,%r14d+ addl %ecx,%r10d+ psrlq $17,%xmm6+ addl %r15d,%ecx+ movl %r10d,%r13d+ addl %ecx,%r14d+ pxor %xmm6,%xmm7+ rorl $14,%r13d+ movl %r14d,%ecx+ movl %r11d,%r12d+ rorl $9,%r14d+ psrlq $2,%xmm6+ xorl %r10d,%r13d+ xorl %eax,%r12d+ pxor %xmm6,%xmm7+ rorl $5,%r13d+ xorl %ecx,%r14d+ andl %r10d,%r12d+ pshufd $128,%xmm7,%xmm7+ xorl %r10d,%r13d+ addl 56(%rsp),%ebx+ movl %ecx,%r15d+ psrldq $8,%xmm7+ xorl %eax,%r12d+ rorl $11,%r14d+ xorl %edx,%r15d+ addl %r12d,%ebx+ rorl $6,%r13d+ paddd %xmm7,%xmm3+ andl %r15d,%edi+ xorl %ecx,%r14d+ addl %r13d,%ebx+ pshufd $80,%xmm3,%xmm7+ xorl %edx,%edi+ rorl $2,%r14d+ addl %ebx,%r9d+ movdqa %xmm7,%xmm6+ addl %edi,%ebx+ movl %r9d,%r13d+ psrld $10,%xmm7+ addl %ebx,%r14d+ rorl $14,%r13d+ psrlq $17,%xmm6+ movl %r14d,%ebx+ movl %r10d,%r12d+ pxor %xmm6,%xmm7+ rorl $9,%r14d+ xorl %r9d,%r13d+ xorl %r11d,%r12d+ rorl $5,%r13d+ xorl %ebx,%r14d+ psrlq $2,%xmm6+ andl %r9d,%r12d+ xorl %r9d,%r13d+ addl 60(%rsp),%eax+ pxor %xmm6,%xmm7+ movl %ebx,%edi+ xorl %r11d,%r12d+ rorl $11,%r14d+ pshufd $8,%xmm7,%xmm7+ xorl %ecx,%edi+ addl %r12d,%eax+ movdqa 96(%rsi),%xmm6+ rorl $6,%r13d+ andl %edi,%r15d+ pslldq $8,%xmm7+ xorl %ebx,%r14d+ addl %r13d,%eax+ xorl %ecx,%r15d+ paddd %xmm7,%xmm3+ rorl $2,%r14d+ addl %eax,%r8d+ addl %r15d,%eax+ paddd %xmm3,%xmm6+ movl %r8d,%r13d+ addl %eax,%r14d+ movdqa %xmm6,48(%rsp)+ cmpb $0,131(%rsi)+ jne .Lssse3_00_47+ rorl $14,%r13d+ movl %r14d,%eax+ movl %r9d,%r12d+ rorl $9,%r14d+ xorl %r8d,%r13d+ xorl %r10d,%r12d+ rorl $5,%r13d+ xorl %eax,%r14d+ andl %r8d,%r12d+ xorl %r8d,%r13d+ addl 0(%rsp),%r11d+ movl %eax,%r15d+ xorl %r10d,%r12d+ rorl $11,%r14d+ xorl %ebx,%r15d+ addl %r12d,%r11d+ rorl $6,%r13d+ andl %r15d,%edi+ xorl %eax,%r14d+ addl %r13d,%r11d+ xorl %ebx,%edi+ rorl $2,%r14d+ addl %r11d,%edx+ addl %edi,%r11d+ movl %edx,%r13d+ addl %r11d,%r14d+ rorl $14,%r13d+ movl %r14d,%r11d+ movl %r8d,%r12d+ rorl $9,%r14d+ xorl %edx,%r13d+ xorl %r9d,%r12d+ rorl $5,%r13d+ xorl %r11d,%r14d+ andl %edx,%r12d+ xorl %edx,%r13d+ addl 4(%rsp),%r10d+ movl %r11d,%edi+ xorl %r9d,%r12d+ rorl $11,%r14d+ xorl %eax,%edi+ addl %r12d,%r10d+ rorl $6,%r13d+ andl %edi,%r15d+ xorl %r11d,%r14d+ addl %r13d,%r10d+ xorl %eax,%r15d+ rorl $2,%r14d+ addl %r10d,%ecx+ addl %r15d,%r10d+ movl %ecx,%r13d+ addl %r10d,%r14d+ rorl $14,%r13d+ movl %r14d,%r10d+ movl %edx,%r12d+ rorl $9,%r14d+ xorl %ecx,%r13d+ xorl %r8d,%r12d+ rorl $5,%r13d+ xorl %r10d,%r14d+ andl %ecx,%r12d+ xorl %ecx,%r13d+ addl 8(%rsp),%r9d+ movl %r10d,%r15d+ xorl %r8d,%r12d+ rorl $11,%r14d+ xorl %r11d,%r15d+ addl %r12d,%r9d+ rorl $6,%r13d+ andl %r15d,%edi+ xorl %r10d,%r14d+ addl %r13d,%r9d+ xorl %r11d,%edi+ rorl $2,%r14d+ addl %r9d,%ebx+ addl %edi,%r9d+ movl %ebx,%r13d+ addl %r9d,%r14d+ rorl $14,%r13d+ movl %r14d,%r9d+ movl %ecx,%r12d+ rorl $9,%r14d+ xorl %ebx,%r13d+ xorl %edx,%r12d+ rorl $5,%r13d+ xorl %r9d,%r14d+ andl %ebx,%r12d+ xorl %ebx,%r13d+ addl 12(%rsp),%r8d+ movl %r9d,%edi+ xorl %edx,%r12d+ rorl $11,%r14d+ xorl %r10d,%edi+ addl %r12d,%r8d+ rorl $6,%r13d+ andl %edi,%r15d+ xorl %r9d,%r14d+ addl %r13d,%r8d+ xorl %r10d,%r15d+ rorl $2,%r14d+ addl %r8d,%eax+ addl %r15d,%r8d+ movl %eax,%r13d+ addl %r8d,%r14d+ rorl $14,%r13d+ movl %r14d,%r8d+ movl %ebx,%r12d+ rorl $9,%r14d+ xorl %eax,%r13d+ xorl %ecx,%r12d+ rorl $5,%r13d+ xorl %r8d,%r14d+ andl %eax,%r12d+ xorl %eax,%r13d+ addl 16(%rsp),%edx+ movl %r8d,%r15d+ xorl %ecx,%r12d+ rorl $11,%r14d+ xorl %r9d,%r15d+ addl %r12d,%edx+ rorl $6,%r13d+ andl %r15d,%edi+ xorl %r8d,%r14d+ addl %r13d,%edx+ xorl %r9d,%edi+ rorl $2,%r14d+ addl %edx,%r11d+ addl %edi,%edx+ movl %r11d,%r13d+ addl %edx,%r14d+ rorl $14,%r13d+ movl %r14d,%edx+ movl %eax,%r12d+ rorl $9,%r14d+ xorl %r11d,%r13d+ xorl %ebx,%r12d+ rorl $5,%r13d+ xorl %edx,%r14d+ andl %r11d,%r12d+ xorl %r11d,%r13d+ addl 20(%rsp),%ecx+ movl %edx,%edi+ xorl %ebx,%r12d+ rorl $11,%r14d+ xorl %r8d,%edi+ addl %r12d,%ecx+ rorl $6,%r13d+ andl %edi,%r15d+ xorl %edx,%r14d+ addl %r13d,%ecx+ xorl %r8d,%r15d+ rorl $2,%r14d+ addl %ecx,%r10d+ addl %r15d,%ecx+ movl %r10d,%r13d+ addl %ecx,%r14d+ rorl $14,%r13d+ movl %r14d,%ecx+ movl %r11d,%r12d+ rorl $9,%r14d+ xorl %r10d,%r13d+ xorl %eax,%r12d+ rorl $5,%r13d+ xorl %ecx,%r14d+ andl %r10d,%r12d+ xorl %r10d,%r13d+ addl 24(%rsp),%ebx+ movl %ecx,%r15d+ xorl %eax,%r12d+ rorl $11,%r14d+ xorl %edx,%r15d+ addl %r12d,%ebx+ rorl $6,%r13d+ andl %r15d,%edi+ xorl %ecx,%r14d+ addl %r13d,%ebx+ xorl %edx,%edi+ rorl $2,%r14d+ addl %ebx,%r9d+ addl %edi,%ebx+ movl %r9d,%r13d+ addl %ebx,%r14d+ rorl $14,%r13d+ movl %r14d,%ebx+ movl %r10d,%r12d+ rorl $9,%r14d+ xorl %r9d,%r13d+ xorl %r11d,%r12d+ rorl $5,%r13d+ xorl %ebx,%r14d+ andl %r9d,%r12d+ xorl %r9d,%r13d+ addl 28(%rsp),%eax+ movl %ebx,%edi+ xorl %r11d,%r12d+ rorl $11,%r14d+ xorl %ecx,%edi+ addl %r12d,%eax+ rorl $6,%r13d+ andl %edi,%r15d+ xorl %ebx,%r14d+ addl %r13d,%eax+ xorl %ecx,%r15d+ rorl $2,%r14d+ addl %eax,%r8d+ addl %r15d,%eax+ movl %r8d,%r13d+ addl %eax,%r14d+ rorl $14,%r13d+ movl %r14d,%eax+ movl %r9d,%r12d+ rorl $9,%r14d+ xorl %r8d,%r13d+ xorl %r10d,%r12d+ rorl $5,%r13d+ xorl %eax,%r14d+ andl %r8d,%r12d+ xorl %r8d,%r13d+ addl 32(%rsp),%r11d+ movl %eax,%r15d+ xorl %r10d,%r12d+ rorl $11,%r14d+ xorl %ebx,%r15d+ addl %r12d,%r11d+ rorl $6,%r13d+ andl %r15d,%edi+ xorl %eax,%r14d+ addl %r13d,%r11d+ xorl %ebx,%edi+ rorl $2,%r14d+ addl %r11d,%edx+ addl %edi,%r11d+ movl %edx,%r13d+ addl %r11d,%r14d+ rorl $14,%r13d+ movl %r14d,%r11d+ movl %r8d,%r12d+ rorl $9,%r14d+ xorl %edx,%r13d+ xorl %r9d,%r12d+ rorl $5,%r13d+ xorl %r11d,%r14d+ andl %edx,%r12d+ xorl %edx,%r13d+ addl 36(%rsp),%r10d+ movl %r11d,%edi+ xorl %r9d,%r12d+ rorl $11,%r14d+ xorl %eax,%edi+ addl %r12d,%r10d+ rorl $6,%r13d+ andl %edi,%r15d+ xorl %r11d,%r14d+ addl %r13d,%r10d+ xorl %eax,%r15d+ rorl $2,%r14d+ addl %r10d,%ecx+ addl %r15d,%r10d+ movl %ecx,%r13d+ addl %r10d,%r14d+ rorl $14,%r13d+ movl %r14d,%r10d+ movl %edx,%r12d+ rorl $9,%r14d+ xorl %ecx,%r13d+ xorl %r8d,%r12d+ rorl $5,%r13d+ xorl %r10d,%r14d+ andl %ecx,%r12d+ xorl %ecx,%r13d+ addl 40(%rsp),%r9d+ movl %r10d,%r15d+ xorl %r8d,%r12d+ rorl $11,%r14d+ xorl %r11d,%r15d+ addl %r12d,%r9d+ rorl $6,%r13d+ andl %r15d,%edi+ xorl %r10d,%r14d+ addl %r13d,%r9d+ xorl %r11d,%edi+ rorl $2,%r14d+ addl %r9d,%ebx+ addl %edi,%r9d+ movl %ebx,%r13d+ addl %r9d,%r14d+ rorl $14,%r13d+ movl %r14d,%r9d+ movl %ecx,%r12d+ rorl $9,%r14d+ xorl %ebx,%r13d+ xorl %edx,%r12d+ rorl $5,%r13d+ xorl %r9d,%r14d+ andl %ebx,%r12d+ xorl %ebx,%r13d+ addl 44(%rsp),%r8d+ movl %r9d,%edi+ xorl %edx,%r12d+ rorl $11,%r14d+ xorl %r10d,%edi+ addl %r12d,%r8d+ rorl $6,%r13d+ andl %edi,%r15d+ xorl %r9d,%r14d+ addl %r13d,%r8d+ xorl %r10d,%r15d+ rorl $2,%r14d+ addl %r8d,%eax+ addl %r15d,%r8d+ movl %eax,%r13d+ addl %r8d,%r14d+ rorl $14,%r13d+ movl %r14d,%r8d+ movl %ebx,%r12d+ rorl $9,%r14d+ xorl %eax,%r13d+ xorl %ecx,%r12d+ rorl $5,%r13d+ xorl %r8d,%r14d+ andl %eax,%r12d+ xorl %eax,%r13d+ addl 48(%rsp),%edx+ movl %r8d,%r15d+ xorl %ecx,%r12d+ rorl $11,%r14d+ xorl %r9d,%r15d+ addl %r12d,%edx+ rorl $6,%r13d+ andl %r15d,%edi+ xorl %r8d,%r14d+ addl %r13d,%edx+ xorl %r9d,%edi+ rorl $2,%r14d+ addl %edx,%r11d+ addl %edi,%edx+ movl %r11d,%r13d+ addl %edx,%r14d+ rorl $14,%r13d+ movl %r14d,%edx+ movl %eax,%r12d+ rorl $9,%r14d+ xorl %r11d,%r13d+ xorl %ebx,%r12d+ rorl $5,%r13d+ xorl %edx,%r14d+ andl %r11d,%r12d+ xorl %r11d,%r13d+ addl 52(%rsp),%ecx+ movl %edx,%edi+ xorl %ebx,%r12d+ rorl $11,%r14d+ xorl %r8d,%edi+ addl %r12d,%ecx+ rorl $6,%r13d+ andl %edi,%r15d+ xorl %edx,%r14d+ addl %r13d,%ecx+ xorl %r8d,%r15d+ rorl $2,%r14d+ addl %ecx,%r10d+ addl %r15d,%ecx+ movl %r10d,%r13d+ addl %ecx,%r14d+ rorl $14,%r13d+ movl %r14d,%ecx+ movl %r11d,%r12d+ rorl $9,%r14d+ xorl %r10d,%r13d+ xorl %eax,%r12d+ rorl $5,%r13d+ xorl %ecx,%r14d+ andl %r10d,%r12d+ xorl %r10d,%r13d+ addl 56(%rsp),%ebx+ movl %ecx,%r15d+ xorl %eax,%r12d+ rorl $11,%r14d+ xorl %edx,%r15d+ addl %r12d,%ebx+ rorl $6,%r13d+ andl %r15d,%edi+ xorl %ecx,%r14d+ addl %r13d,%ebx+ xorl %edx,%edi+ rorl $2,%r14d+ addl %ebx,%r9d+ addl %edi,%ebx+ movl %r9d,%r13d+ addl %ebx,%r14d+ rorl $14,%r13d+ movl %r14d,%ebx+ movl %r10d,%r12d+ rorl $9,%r14d+ xorl %r9d,%r13d+ xorl %r11d,%r12d+ rorl $5,%r13d+ xorl %ebx,%r14d+ andl %r9d,%r12d+ xorl %r9d,%r13d+ addl 60(%rsp),%eax+ movl %ebx,%edi+ xorl %r11d,%r12d+ rorl $11,%r14d+ xorl %ecx,%edi+ addl %r12d,%eax+ rorl $6,%r13d+ andl %edi,%r15d+ xorl %ebx,%r14d+ addl %r13d,%eax+ xorl %ecx,%r15d+ rorl $2,%r14d+ addl %eax,%r8d+ addl %r15d,%eax+ movl %r8d,%r13d+ addl %eax,%r14d+ movq -64(%rbp),%rdi+ movl %r14d,%eax+ movq -56(%rbp),%rsi++ addl 0(%rdi),%eax+ addl 4(%rdi),%ebx+ addl 8(%rdi),%ecx+ addl 12(%rdi),%edx+ addl 16(%rdi),%r8d+ addl 20(%rdi),%r9d+ addl 24(%rdi),%r10d+ addl 28(%rdi),%r11d++ leaq 64(%rsi),%rsi+ cmpq -48(%rbp),%rsi++ movl %eax,0(%rdi)+ movl %ebx,4(%rdi)+ movl %ecx,8(%rdi)+ movl %edx,12(%rdi)+ movl %r8d,16(%rdi)+ movl %r9d,20(%rdi)+ movl %r10d,24(%rdi)+ movl %r11d,28(%rdi)+ jb .Lloop_ssse3++ movaps -128(%rbp),%xmm6+ movaps -112(%rbp),%xmm7+ movaps -96(%rbp),%xmm8+ movaps -80(%rbp),%xmm9+ movq -40(%rbp),%r15+ movq -32(%rbp),%r14+ movq -24(%rbp),%r13+ movq -16(%rbp),%r12+ movq -8(%rbp),%rbx+ movq %rbp,%rsp++ popq %rbp++.LSEH_epilogue_crypton_sha256_asm_block_data_order_ssse3:+ mov 8(%rsp),%rdi+ mov 16(%rsp),%rsi++ .byte 0xf3,0xc3++.LSEH_end_crypton_sha256_asm_block_data_order_ssse3:+.def crypton_sha256_asm_block_data_order_avx; .scl 3; .type 32; .endef+.p2align 6+crypton_sha256_asm_block_data_order_avx:+ .byte 0xf3,0x0f,0x1e,0xfa+ movq %rdi,8(%rsp)+ movq %rsi,16(%rsp)+ movq %rsp,%r11+.LSEH_begin_crypton_sha256_asm_block_data_order_avx:+++ pushq %rbp++ movq %rsp,%rbp++ movq %rcx,%rdi+ movq %rdx,%rsi+ movq %r8,%rdx+.Lavx_shortcut:+ pushq %rbx++ pushq %r12++ pushq %r13++ pushq %r14++ pushq %r15++ shlq $4,%rdx+ subq $120,%rsp++ leaq (%rsi,%rdx,4),%rdx+ movq %rdi,-64(%rbp)++ movq %rdx,-48(%rbp)+ movaps %xmm6,-160(%rbp)+ movaps %xmm7,-144(%rbp)+ movaps %xmm8,-128(%rbp)+ movaps %xmm9,-112(%rbp)++.LSEH_body_crypton_sha256_asm_block_data_order_avx:+++ leaq -64(%rsp),%rsp+ vzeroupper+ andq $-64,%rsp+ movl 0(%rdi),%eax+ movl 4(%rdi),%ebx+ movl 8(%rdi),%ecx+ movl 12(%rdi),%edx+ movl 16(%rdi),%r8d+ movl 20(%rdi),%r9d+ movl 24(%rdi),%r10d+ movl 28(%rdi),%r11d+ vmovdqa K256+512+32(%rip),%xmm8+ vmovdqa K256+512+64(%rip),%xmm9+ jmp .Lloop_avx+.p2align 4+.Lloop_avx:+ vmovdqa K256+512(%rip),%xmm7+ movq %rsi,-56(%rbp)+ vmovdqu 0(%rsi),%xmm0+ vmovdqu 16(%rsi),%xmm1+ vmovdqu 32(%rsi),%xmm2+ vmovdqu 48(%rsi),%xmm3+ vpshufb %xmm7,%xmm0,%xmm0+ leaq K256(%rip),%rsi+ vpshufb %xmm7,%xmm1,%xmm1+ vpshufb %xmm7,%xmm2,%xmm2+ vpaddd 0(%rsi),%xmm0,%xmm4+ vpshufb %xmm7,%xmm3,%xmm3+ vpaddd 32(%rsi),%xmm1,%xmm5+ vpaddd 64(%rsi),%xmm2,%xmm6+ vpaddd 96(%rsi),%xmm3,%xmm7+ vmovdqa %xmm4,0(%rsp)+ movl %eax,%r14d+ vmovdqa %xmm5,16(%rsp)+ movl %ebx,%edi+ vmovdqa %xmm6,32(%rsp)+ xorl %ecx,%edi+ vmovdqa %xmm7,48(%rsp)+ movl %r8d,%r13d+ jmp .Lavx_00_47++.p2align 4+.Lavx_00_47:+ subq $-128,%rsi+ vpalignr $4,%xmm0,%xmm1,%xmm4+ shrdl $14,%r13d,%r13d+ movl %r14d,%eax+ movl %r9d,%r12d+ vpalignr $4,%xmm2,%xmm3,%xmm7+ shrdl $9,%r14d,%r14d+ xorl %r8d,%r13d+ xorl %r10d,%r12d+ vpsrld $7,%xmm4,%xmm6+ shrdl $5,%r13d,%r13d+ xorl %eax,%r14d+ andl %r8d,%r12d+ vpaddd %xmm7,%xmm0,%xmm0+ xorl %r8d,%r13d+ addl 0(%rsp),%r11d+ movl %eax,%r15d+ vpsrld $3,%xmm4,%xmm7+ xorl %r10d,%r12d+ shrdl $11,%r14d,%r14d+ xorl %ebx,%r15d+ vpslld $14,%xmm4,%xmm5+ addl %r12d,%r11d+ shrdl $6,%r13d,%r13d+ andl %r15d,%edi+ vpxor %xmm6,%xmm7,%xmm4+ xorl %eax,%r14d+ addl %r13d,%r11d+ xorl %ebx,%edi+ vpshufd $250,%xmm3,%xmm7+ shrdl $2,%r14d,%r14d+ addl %r11d,%edx+ addl %edi,%r11d+ vpsrld $11,%xmm6,%xmm6+ movl %edx,%r13d+ addl %r11d,%r14d+ shrdl $14,%r13d,%r13d+ vpxor %xmm5,%xmm4,%xmm4+ movl %r14d,%r11d+ movl %r8d,%r12d+ shrdl $9,%r14d,%r14d+ vpslld $11,%xmm5,%xmm5+ xorl %edx,%r13d+ xorl %r9d,%r12d+ shrdl $5,%r13d,%r13d+ vpxor %xmm6,%xmm4,%xmm4+ xorl %r11d,%r14d+ andl %edx,%r12d+ xorl %edx,%r13d+ vpsrld $10,%xmm7,%xmm6+ addl 4(%rsp),%r10d+ movl %r11d,%edi+ xorl %r9d,%r12d+ vpxor %xmm5,%xmm4,%xmm4+ shrdl $11,%r14d,%r14d+ xorl %eax,%edi+ addl %r12d,%r10d+ vpsrlq $17,%xmm7,%xmm7+ shrdl $6,%r13d,%r13d+ andl %edi,%r15d+ xorl %r11d,%r14d+ vpaddd %xmm4,%xmm0,%xmm0+ addl %r13d,%r10d+ xorl %eax,%r15d+ shrdl $2,%r14d,%r14d+ vpxor %xmm7,%xmm6,%xmm6+ addl %r10d,%ecx+ addl %r15d,%r10d+ movl %ecx,%r13d+ vpsrlq $2,%xmm7,%xmm7+ addl %r10d,%r14d+ shrdl $14,%r13d,%r13d+ movl %r14d,%r10d+ vpxor %xmm7,%xmm6,%xmm6+ movl %edx,%r12d+ shrdl $9,%r14d,%r14d+ xorl %ecx,%r13d+ vpshufb %xmm8,%xmm6,%xmm6+ xorl %r8d,%r12d+ shrdl $5,%r13d,%r13d+ xorl %r10d,%r14d+ vpaddd %xmm6,%xmm0,%xmm0+ andl %ecx,%r12d+ xorl %ecx,%r13d+ addl 8(%rsp),%r9d+ vpshufd $80,%xmm0,%xmm7+ movl %r10d,%r15d+ xorl %r8d,%r12d+ shrdl $11,%r14d,%r14d+ vpsrld $10,%xmm7,%xmm6+ xorl %r11d,%r15d+ addl %r12d,%r9d+ shrdl $6,%r13d,%r13d+ vpsrlq $17,%xmm7,%xmm7+ andl %r15d,%edi+ xorl %r10d,%r14d+ addl %r13d,%r9d+ vpxor %xmm7,%xmm6,%xmm6+ xorl %r11d,%edi+ shrdl $2,%r14d,%r14d+ addl %r9d,%ebx+ vpsrlq $2,%xmm7,%xmm7+ addl %edi,%r9d+ movl %ebx,%r13d+ addl %r9d,%r14d+ vpxor %xmm7,%xmm6,%xmm6+ shrdl $14,%r13d,%r13d+ movl %r14d,%r9d+ movl %ecx,%r12d+ vpshufb %xmm9,%xmm6,%xmm6+ shrdl $9,%r14d,%r14d+ xorl %ebx,%r13d+ xorl %edx,%r12d+ vpaddd %xmm6,%xmm0,%xmm0+ shrdl $5,%r13d,%r13d+ xorl %r9d,%r14d+ andl %ebx,%r12d+ vpaddd 0(%rsi),%xmm0,%xmm6+ xorl %ebx,%r13d+ addl 12(%rsp),%r8d+ movl %r9d,%edi+ xorl %edx,%r12d+ shrdl $11,%r14d,%r14d+ xorl %r10d,%edi+ addl %r12d,%r8d+ shrdl $6,%r13d,%r13d+ andl %edi,%r15d+ xorl %r9d,%r14d+ addl %r13d,%r8d+ xorl %r10d,%r15d+ shrdl $2,%r14d,%r14d+ addl %r8d,%eax+ addl %r15d,%r8d+ movl %eax,%r13d+ addl %r8d,%r14d+ vmovdqa %xmm6,0(%rsp)+ vpalignr $4,%xmm1,%xmm2,%xmm4+ shrdl $14,%r13d,%r13d+ movl %r14d,%r8d+ movl %ebx,%r12d+ vpalignr $4,%xmm3,%xmm0,%xmm7+ shrdl $9,%r14d,%r14d+ xorl %eax,%r13d+ xorl %ecx,%r12d+ vpsrld $7,%xmm4,%xmm6+ shrdl $5,%r13d,%r13d+ xorl %r8d,%r14d+ andl %eax,%r12d+ vpaddd %xmm7,%xmm1,%xmm1+ xorl %eax,%r13d+ addl 16(%rsp),%edx+ movl %r8d,%r15d+ vpsrld $3,%xmm4,%xmm7+ xorl %ecx,%r12d+ shrdl $11,%r14d,%r14d+ xorl %r9d,%r15d+ vpslld $14,%xmm4,%xmm5+ addl %r12d,%edx+ shrdl $6,%r13d,%r13d+ andl %r15d,%edi+ vpxor %xmm6,%xmm7,%xmm4+ xorl %r8d,%r14d+ addl %r13d,%edx+ xorl %r9d,%edi+ vpshufd $250,%xmm0,%xmm7+ shrdl $2,%r14d,%r14d+ addl %edx,%r11d+ addl %edi,%edx+ vpsrld $11,%xmm6,%xmm6+ movl %r11d,%r13d+ addl %edx,%r14d+ shrdl $14,%r13d,%r13d+ vpxor %xmm5,%xmm4,%xmm4+ movl %r14d,%edx+ movl %eax,%r12d+ shrdl $9,%r14d,%r14d+ vpslld $11,%xmm5,%xmm5+ xorl %r11d,%r13d+ xorl %ebx,%r12d+ shrdl $5,%r13d,%r13d+ vpxor %xmm6,%xmm4,%xmm4+ xorl %edx,%r14d+ andl %r11d,%r12d+ xorl %r11d,%r13d+ vpsrld $10,%xmm7,%xmm6+ addl 20(%rsp),%ecx+ movl %edx,%edi+ xorl %ebx,%r12d+ vpxor %xmm5,%xmm4,%xmm4+ shrdl $11,%r14d,%r14d+ xorl %r8d,%edi+ addl %r12d,%ecx+ vpsrlq $17,%xmm7,%xmm7+ shrdl $6,%r13d,%r13d+ andl %edi,%r15d+ xorl %edx,%r14d+ vpaddd %xmm4,%xmm1,%xmm1+ addl %r13d,%ecx+ xorl %r8d,%r15d+ shrdl $2,%r14d,%r14d+ vpxor %xmm7,%xmm6,%xmm6+ addl %ecx,%r10d+ addl %r15d,%ecx+ movl %r10d,%r13d+ vpsrlq $2,%xmm7,%xmm7+ addl %ecx,%r14d+ shrdl $14,%r13d,%r13d+ movl %r14d,%ecx+ vpxor %xmm7,%xmm6,%xmm6+ movl %r11d,%r12d+ shrdl $9,%r14d,%r14d+ xorl %r10d,%r13d+ vpshufb %xmm8,%xmm6,%xmm6+ xorl %eax,%r12d+ shrdl $5,%r13d,%r13d+ xorl %ecx,%r14d+ vpaddd %xmm6,%xmm1,%xmm1+ andl %r10d,%r12d+ xorl %r10d,%r13d+ addl 24(%rsp),%ebx+ vpshufd $80,%xmm1,%xmm7+ movl %ecx,%r15d+ xorl %eax,%r12d+ shrdl $11,%r14d,%r14d+ vpsrld $10,%xmm7,%xmm6+ xorl %edx,%r15d+ addl %r12d,%ebx+ shrdl $6,%r13d,%r13d+ vpsrlq $17,%xmm7,%xmm7+ andl %r15d,%edi+ xorl %ecx,%r14d+ addl %r13d,%ebx+ vpxor %xmm7,%xmm6,%xmm6+ xorl %edx,%edi+ shrdl $2,%r14d,%r14d+ addl %ebx,%r9d+ vpsrlq $2,%xmm7,%xmm7+ addl %edi,%ebx+ movl %r9d,%r13d+ addl %ebx,%r14d+ vpxor %xmm7,%xmm6,%xmm6+ shrdl $14,%r13d,%r13d+ movl %r14d,%ebx+ movl %r10d,%r12d+ vpshufb %xmm9,%xmm6,%xmm6+ shrdl $9,%r14d,%r14d+ xorl %r9d,%r13d+ xorl %r11d,%r12d+ vpaddd %xmm6,%xmm1,%xmm1+ shrdl $5,%r13d,%r13d+ xorl %ebx,%r14d+ andl %r9d,%r12d+ vpaddd 32(%rsi),%xmm1,%xmm6+ xorl %r9d,%r13d+ addl 28(%rsp),%eax+ movl %ebx,%edi+ xorl %r11d,%r12d+ shrdl $11,%r14d,%r14d+ xorl %ecx,%edi+ addl %r12d,%eax+ shrdl $6,%r13d,%r13d+ andl %edi,%r15d+ xorl %ebx,%r14d+ addl %r13d,%eax+ xorl %ecx,%r15d+ shrdl $2,%r14d,%r14d+ addl %eax,%r8d+ addl %r15d,%eax+ movl %r8d,%r13d+ addl %eax,%r14d+ vmovdqa %xmm6,16(%rsp)+ vpalignr $4,%xmm2,%xmm3,%xmm4+ shrdl $14,%r13d,%r13d+ movl %r14d,%eax+ movl %r9d,%r12d+ vpalignr $4,%xmm0,%xmm1,%xmm7+ shrdl $9,%r14d,%r14d+ xorl %r8d,%r13d+ xorl %r10d,%r12d+ vpsrld $7,%xmm4,%xmm6+ shrdl $5,%r13d,%r13d+ xorl %eax,%r14d+ andl %r8d,%r12d+ vpaddd %xmm7,%xmm2,%xmm2+ xorl %r8d,%r13d+ addl 32(%rsp),%r11d+ movl %eax,%r15d+ vpsrld $3,%xmm4,%xmm7+ xorl %r10d,%r12d+ shrdl $11,%r14d,%r14d+ xorl %ebx,%r15d+ vpslld $14,%xmm4,%xmm5+ addl %r12d,%r11d+ shrdl $6,%r13d,%r13d+ andl %r15d,%edi+ vpxor %xmm6,%xmm7,%xmm4+ xorl %eax,%r14d+ addl %r13d,%r11d+ xorl %ebx,%edi+ vpshufd $250,%xmm1,%xmm7+ shrdl $2,%r14d,%r14d+ addl %r11d,%edx+ addl %edi,%r11d+ vpsrld $11,%xmm6,%xmm6+ movl %edx,%r13d+ addl %r11d,%r14d+ shrdl $14,%r13d,%r13d+ vpxor %xmm5,%xmm4,%xmm4+ movl %r14d,%r11d+ movl %r8d,%r12d+ shrdl $9,%r14d,%r14d+ vpslld $11,%xmm5,%xmm5+ xorl %edx,%r13d+ xorl %r9d,%r12d+ shrdl $5,%r13d,%r13d+ vpxor %xmm6,%xmm4,%xmm4+ xorl %r11d,%r14d+ andl %edx,%r12d+ xorl %edx,%r13d+ vpsrld $10,%xmm7,%xmm6+ addl 36(%rsp),%r10d+ movl %r11d,%edi+ xorl %r9d,%r12d+ vpxor %xmm5,%xmm4,%xmm4+ shrdl $11,%r14d,%r14d+ xorl %eax,%edi+ addl %r12d,%r10d+ vpsrlq $17,%xmm7,%xmm7+ shrdl $6,%r13d,%r13d+ andl %edi,%r15d+ xorl %r11d,%r14d+ vpaddd %xmm4,%xmm2,%xmm2+ addl %r13d,%r10d+ xorl %eax,%r15d+ shrdl $2,%r14d,%r14d+ vpxor %xmm7,%xmm6,%xmm6+ addl %r10d,%ecx+ addl %r15d,%r10d+ movl %ecx,%r13d+ vpsrlq $2,%xmm7,%xmm7+ addl %r10d,%r14d+ shrdl $14,%r13d,%r13d+ movl %r14d,%r10d+ vpxor %xmm7,%xmm6,%xmm6+ movl %edx,%r12d+ shrdl $9,%r14d,%r14d+ xorl %ecx,%r13d+ vpshufb %xmm8,%xmm6,%xmm6+ xorl %r8d,%r12d+ shrdl $5,%r13d,%r13d+ xorl %r10d,%r14d+ vpaddd %xmm6,%xmm2,%xmm2+ andl %ecx,%r12d+ xorl %ecx,%r13d+ addl 40(%rsp),%r9d+ vpshufd $80,%xmm2,%xmm7+ movl %r10d,%r15d+ xorl %r8d,%r12d+ shrdl $11,%r14d,%r14d+ vpsrld $10,%xmm7,%xmm6+ xorl %r11d,%r15d+ addl %r12d,%r9d+ shrdl $6,%r13d,%r13d+ vpsrlq $17,%xmm7,%xmm7+ andl %r15d,%edi+ xorl %r10d,%r14d+ addl %r13d,%r9d+ vpxor %xmm7,%xmm6,%xmm6+ xorl %r11d,%edi+ shrdl $2,%r14d,%r14d+ addl %r9d,%ebx+ vpsrlq $2,%xmm7,%xmm7+ addl %edi,%r9d+ movl %ebx,%r13d+ addl %r9d,%r14d+ vpxor %xmm7,%xmm6,%xmm6+ shrdl $14,%r13d,%r13d+ movl %r14d,%r9d+ movl %ecx,%r12d+ vpshufb %xmm9,%xmm6,%xmm6+ shrdl $9,%r14d,%r14d+ xorl %ebx,%r13d+ xorl %edx,%r12d+ vpaddd %xmm6,%xmm2,%xmm2+ shrdl $5,%r13d,%r13d+ xorl %r9d,%r14d+ andl %ebx,%r12d+ vpaddd 64(%rsi),%xmm2,%xmm6+ xorl %ebx,%r13d+ addl 44(%rsp),%r8d+ movl %r9d,%edi+ xorl %edx,%r12d+ shrdl $11,%r14d,%r14d+ xorl %r10d,%edi+ addl %r12d,%r8d+ shrdl $6,%r13d,%r13d+ andl %edi,%r15d+ xorl %r9d,%r14d+ addl %r13d,%r8d+ xorl %r10d,%r15d+ shrdl $2,%r14d,%r14d+ addl %r8d,%eax+ addl %r15d,%r8d+ movl %eax,%r13d+ addl %r8d,%r14d+ vmovdqa %xmm6,32(%rsp)+ vpalignr $4,%xmm3,%xmm0,%xmm4+ shrdl $14,%r13d,%r13d+ movl %r14d,%r8d+ movl %ebx,%r12d+ vpalignr $4,%xmm1,%xmm2,%xmm7+ shrdl $9,%r14d,%r14d+ xorl %eax,%r13d+ xorl %ecx,%r12d+ vpsrld $7,%xmm4,%xmm6+ shrdl $5,%r13d,%r13d+ xorl %r8d,%r14d+ andl %eax,%r12d+ vpaddd %xmm7,%xmm3,%xmm3+ xorl %eax,%r13d+ addl 48(%rsp),%edx+ movl %r8d,%r15d+ vpsrld $3,%xmm4,%xmm7+ xorl %ecx,%r12d+ shrdl $11,%r14d,%r14d+ xorl %r9d,%r15d+ vpslld $14,%xmm4,%xmm5+ addl %r12d,%edx+ shrdl $6,%r13d,%r13d+ andl %r15d,%edi+ vpxor %xmm6,%xmm7,%xmm4+ xorl %r8d,%r14d+ addl %r13d,%edx+ xorl %r9d,%edi+ vpshufd $250,%xmm2,%xmm7+ shrdl $2,%r14d,%r14d+ addl %edx,%r11d+ addl %edi,%edx+ vpsrld $11,%xmm6,%xmm6+ movl %r11d,%r13d+ addl %edx,%r14d+ shrdl $14,%r13d,%r13d+ vpxor %xmm5,%xmm4,%xmm4+ movl %r14d,%edx+ movl %eax,%r12d+ shrdl $9,%r14d,%r14d+ vpslld $11,%xmm5,%xmm5+ xorl %r11d,%r13d+ xorl %ebx,%r12d+ shrdl $5,%r13d,%r13d+ vpxor %xmm6,%xmm4,%xmm4+ xorl %edx,%r14d+ andl %r11d,%r12d+ xorl %r11d,%r13d+ vpsrld $10,%xmm7,%xmm6+ addl 52(%rsp),%ecx+ movl %edx,%edi+ xorl %ebx,%r12d+ vpxor %xmm5,%xmm4,%xmm4+ shrdl $11,%r14d,%r14d+ xorl %r8d,%edi+ addl %r12d,%ecx+ vpsrlq $17,%xmm7,%xmm7+ shrdl $6,%r13d,%r13d+ andl %edi,%r15d+ xorl %edx,%r14d+ vpaddd %xmm4,%xmm3,%xmm3+ addl %r13d,%ecx+ xorl %r8d,%r15d+ shrdl $2,%r14d,%r14d+ vpxor %xmm7,%xmm6,%xmm6+ addl %ecx,%r10d+ addl %r15d,%ecx+ movl %r10d,%r13d+ vpsrlq $2,%xmm7,%xmm7+ addl %ecx,%r14d+ shrdl $14,%r13d,%r13d+ movl %r14d,%ecx+ vpxor %xmm7,%xmm6,%xmm6+ movl %r11d,%r12d+ shrdl $9,%r14d,%r14d+ xorl %r10d,%r13d+ vpshufb %xmm8,%xmm6,%xmm6+ xorl %eax,%r12d+ shrdl $5,%r13d,%r13d+ xorl %ecx,%r14d+ vpaddd %xmm6,%xmm3,%xmm3+ andl %r10d,%r12d+ xorl %r10d,%r13d+ addl 56(%rsp),%ebx+ vpshufd $80,%xmm3,%xmm7+ movl %ecx,%r15d+ xorl %eax,%r12d+ shrdl $11,%r14d,%r14d+ vpsrld $10,%xmm7,%xmm6+ xorl %edx,%r15d+ addl %r12d,%ebx+ shrdl $6,%r13d,%r13d+ vpsrlq $17,%xmm7,%xmm7+ andl %r15d,%edi+ xorl %ecx,%r14d+ addl %r13d,%ebx+ vpxor %xmm7,%xmm6,%xmm6+ xorl %edx,%edi+ shrdl $2,%r14d,%r14d+ addl %ebx,%r9d+ vpsrlq $2,%xmm7,%xmm7+ addl %edi,%ebx+ movl %r9d,%r13d+ addl %ebx,%r14d+ vpxor %xmm7,%xmm6,%xmm6+ shrdl $14,%r13d,%r13d+ movl %r14d,%ebx+ movl %r10d,%r12d+ vpshufb %xmm9,%xmm6,%xmm6+ shrdl $9,%r14d,%r14d+ xorl %r9d,%r13d+ xorl %r11d,%r12d+ vpaddd %xmm6,%xmm3,%xmm3+ shrdl $5,%r13d,%r13d+ xorl %ebx,%r14d+ andl %r9d,%r12d+ vpaddd 96(%rsi),%xmm3,%xmm6+ xorl %r9d,%r13d+ addl 60(%rsp),%eax+ movl %ebx,%edi+ xorl %r11d,%r12d+ shrdl $11,%r14d,%r14d+ xorl %ecx,%edi+ addl %r12d,%eax+ shrdl $6,%r13d,%r13d+ andl %edi,%r15d+ xorl %ebx,%r14d+ addl %r13d,%eax+ xorl %ecx,%r15d+ shrdl $2,%r14d,%r14d+ addl %eax,%r8d+ addl %r15d,%eax+ movl %r8d,%r13d+ addl %eax,%r14d+ vmovdqa %xmm6,48(%rsp)+ cmpb $0,131(%rsi)+ jne .Lavx_00_47+ shrdl $14,%r13d,%r13d+ movl %r14d,%eax+ movl %r9d,%r12d+ shrdl $9,%r14d,%r14d+ xorl %r8d,%r13d+ xorl %r10d,%r12d+ shrdl $5,%r13d,%r13d+ xorl %eax,%r14d+ andl %r8d,%r12d+ xorl %r8d,%r13d+ addl 0(%rsp),%r11d+ movl %eax,%r15d+ xorl %r10d,%r12d+ shrdl $11,%r14d,%r14d+ xorl %ebx,%r15d+ addl %r12d,%r11d+ shrdl $6,%r13d,%r13d+ andl %r15d,%edi+ xorl %eax,%r14d+ addl %r13d,%r11d+ xorl %ebx,%edi+ shrdl $2,%r14d,%r14d+ addl %r11d,%edx+ addl %edi,%r11d+ movl %edx,%r13d+ addl %r11d,%r14d+ shrdl $14,%r13d,%r13d+ movl %r14d,%r11d+ movl %r8d,%r12d+ shrdl $9,%r14d,%r14d+ xorl %edx,%r13d+ xorl %r9d,%r12d+ shrdl $5,%r13d,%r13d+ xorl %r11d,%r14d+ andl %edx,%r12d+ xorl %edx,%r13d+ addl 4(%rsp),%r10d+ movl %r11d,%edi+ xorl %r9d,%r12d+ shrdl $11,%r14d,%r14d+ xorl %eax,%edi+ addl %r12d,%r10d+ shrdl $6,%r13d,%r13d+ andl %edi,%r15d+ xorl %r11d,%r14d+ addl %r13d,%r10d+ xorl %eax,%r15d+ shrdl $2,%r14d,%r14d+ addl %r10d,%ecx+ addl %r15d,%r10d+ movl %ecx,%r13d+ addl %r10d,%r14d+ shrdl $14,%r13d,%r13d+ movl %r14d,%r10d+ movl %edx,%r12d+ shrdl $9,%r14d,%r14d+ xorl %ecx,%r13d+ xorl %r8d,%r12d+ shrdl $5,%r13d,%r13d+ xorl %r10d,%r14d+ andl %ecx,%r12d+ xorl %ecx,%r13d+ addl 8(%rsp),%r9d+ movl %r10d,%r15d+ xorl %r8d,%r12d+ shrdl $11,%r14d,%r14d+ xorl %r11d,%r15d+ addl %r12d,%r9d+ shrdl $6,%r13d,%r13d+ andl %r15d,%edi+ xorl %r10d,%r14d+ addl %r13d,%r9d+ xorl %r11d,%edi+ shrdl $2,%r14d,%r14d+ addl %r9d,%ebx+ addl %edi,%r9d+ movl %ebx,%r13d+ addl %r9d,%r14d+ shrdl $14,%r13d,%r13d+ movl %r14d,%r9d+ movl %ecx,%r12d+ shrdl $9,%r14d,%r14d+ xorl %ebx,%r13d+ xorl %edx,%r12d+ shrdl $5,%r13d,%r13d+ xorl %r9d,%r14d+ andl %ebx,%r12d+ xorl %ebx,%r13d+ addl 12(%rsp),%r8d+ movl %r9d,%edi+ xorl %edx,%r12d+ shrdl $11,%r14d,%r14d+ xorl %r10d,%edi+ addl %r12d,%r8d+ shrdl $6,%r13d,%r13d+ andl %edi,%r15d+ xorl %r9d,%r14d+ addl %r13d,%r8d+ xorl %r10d,%r15d+ shrdl $2,%r14d,%r14d+ addl %r8d,%eax+ addl %r15d,%r8d+ movl %eax,%r13d+ addl %r8d,%r14d+ shrdl $14,%r13d,%r13d+ movl %r14d,%r8d+ movl %ebx,%r12d+ shrdl $9,%r14d,%r14d+ xorl %eax,%r13d+ xorl %ecx,%r12d+ shrdl $5,%r13d,%r13d+ xorl %r8d,%r14d+ andl %eax,%r12d+ xorl %eax,%r13d+ addl 16(%rsp),%edx+ movl %r8d,%r15d+ xorl %ecx,%r12d+ shrdl $11,%r14d,%r14d+ xorl %r9d,%r15d+ addl %r12d,%edx+ shrdl $6,%r13d,%r13d+ andl %r15d,%edi+ xorl %r8d,%r14d+ addl %r13d,%edx+ xorl %r9d,%edi+ shrdl $2,%r14d,%r14d+ addl %edx,%r11d+ addl %edi,%edx+ movl %r11d,%r13d+ addl %edx,%r14d+ shrdl $14,%r13d,%r13d+ movl %r14d,%edx+ movl %eax,%r12d+ shrdl $9,%r14d,%r14d+ xorl %r11d,%r13d+ xorl %ebx,%r12d+ shrdl $5,%r13d,%r13d+ xorl %edx,%r14d+ andl %r11d,%r12d+ xorl %r11d,%r13d+ addl 20(%rsp),%ecx+ movl %edx,%edi+ xorl %ebx,%r12d+ shrdl $11,%r14d,%r14d+ xorl %r8d,%edi+ addl %r12d,%ecx+ shrdl $6,%r13d,%r13d+ andl %edi,%r15d+ xorl %edx,%r14d+ addl %r13d,%ecx+ xorl %r8d,%r15d+ shrdl $2,%r14d,%r14d+ addl %ecx,%r10d+ addl %r15d,%ecx+ movl %r10d,%r13d+ addl %ecx,%r14d+ shrdl $14,%r13d,%r13d+ movl %r14d,%ecx+ movl %r11d,%r12d+ shrdl $9,%r14d,%r14d+ xorl %r10d,%r13d+ xorl %eax,%r12d+ shrdl $5,%r13d,%r13d+ xorl %ecx,%r14d+ andl %r10d,%r12d+ xorl %r10d,%r13d+ addl 24(%rsp),%ebx+ movl %ecx,%r15d+ xorl %eax,%r12d+ shrdl $11,%r14d,%r14d+ xorl %edx,%r15d+ addl %r12d,%ebx+ shrdl $6,%r13d,%r13d+ andl %r15d,%edi+ xorl %ecx,%r14d+ addl %r13d,%ebx+ xorl %edx,%edi+ shrdl $2,%r14d,%r14d+ addl %ebx,%r9d+ addl %edi,%ebx+ movl %r9d,%r13d+ addl %ebx,%r14d+ shrdl $14,%r13d,%r13d+ movl %r14d,%ebx+ movl %r10d,%r12d+ shrdl $9,%r14d,%r14d+ xorl %r9d,%r13d+ xorl %r11d,%r12d+ shrdl $5,%r13d,%r13d+ xorl %ebx,%r14d+ andl %r9d,%r12d+ xorl %r9d,%r13d+ addl 28(%rsp),%eax+ movl %ebx,%edi+ xorl %r11d,%r12d+ shrdl $11,%r14d,%r14d+ xorl %ecx,%edi+ addl %r12d,%eax+ shrdl $6,%r13d,%r13d+ andl %edi,%r15d+ xorl %ebx,%r14d+ addl %r13d,%eax+ xorl %ecx,%r15d+ shrdl $2,%r14d,%r14d+ addl %eax,%r8d+ addl %r15d,%eax+ movl %r8d,%r13d+ addl %eax,%r14d+ shrdl $14,%r13d,%r13d+ movl %r14d,%eax+ movl %r9d,%r12d+ shrdl $9,%r14d,%r14d+ xorl %r8d,%r13d+ xorl %r10d,%r12d+ shrdl $5,%r13d,%r13d+ xorl %eax,%r14d+ andl %r8d,%r12d+ xorl %r8d,%r13d+ addl 32(%rsp),%r11d+ movl %eax,%r15d+ xorl %r10d,%r12d+ shrdl $11,%r14d,%r14d+ xorl %ebx,%r15d+ addl %r12d,%r11d+ shrdl $6,%r13d,%r13d+ andl %r15d,%edi+ xorl %eax,%r14d+ addl %r13d,%r11d+ xorl %ebx,%edi+ shrdl $2,%r14d,%r14d+ addl %r11d,%edx+ addl %edi,%r11d+ movl %edx,%r13d+ addl %r11d,%r14d+ shrdl $14,%r13d,%r13d+ movl %r14d,%r11d+ movl %r8d,%r12d+ shrdl $9,%r14d,%r14d+ xorl %edx,%r13d+ xorl %r9d,%r12d+ shrdl $5,%r13d,%r13d+ xorl %r11d,%r14d+ andl %edx,%r12d+ xorl %edx,%r13d+ addl 36(%rsp),%r10d+ movl %r11d,%edi+ xorl %r9d,%r12d+ shrdl $11,%r14d,%r14d+ xorl %eax,%edi+ addl %r12d,%r10d+ shrdl $6,%r13d,%r13d+ andl %edi,%r15d+ xorl %r11d,%r14d+ addl %r13d,%r10d+ xorl %eax,%r15d+ shrdl $2,%r14d,%r14d+ addl %r10d,%ecx+ addl %r15d,%r10d+ movl %ecx,%r13d+ addl %r10d,%r14d+ shrdl $14,%r13d,%r13d+ movl %r14d,%r10d+ movl %edx,%r12d+ shrdl $9,%r14d,%r14d+ xorl %ecx,%r13d+ xorl %r8d,%r12d+ shrdl $5,%r13d,%r13d+ xorl %r10d,%r14d+ andl %ecx,%r12d+ xorl %ecx,%r13d+ addl 40(%rsp),%r9d+ movl %r10d,%r15d+ xorl %r8d,%r12d+ shrdl $11,%r14d,%r14d+ xorl %r11d,%r15d+ addl %r12d,%r9d+ shrdl $6,%r13d,%r13d+ andl %r15d,%edi+ xorl %r10d,%r14d+ addl %r13d,%r9d+ xorl %r11d,%edi+ shrdl $2,%r14d,%r14d+ addl %r9d,%ebx+ addl %edi,%r9d+ movl %ebx,%r13d+ addl %r9d,%r14d+ shrdl $14,%r13d,%r13d+ movl %r14d,%r9d+ movl %ecx,%r12d+ shrdl $9,%r14d,%r14d+ xorl %ebx,%r13d+ xorl %edx,%r12d+ shrdl $5,%r13d,%r13d+ xorl %r9d,%r14d+ andl %ebx,%r12d+ xorl %ebx,%r13d+ addl 44(%rsp),%r8d+ movl %r9d,%edi+ xorl %edx,%r12d+ shrdl $11,%r14d,%r14d+ xorl %r10d,%edi+ addl %r12d,%r8d+ shrdl $6,%r13d,%r13d+ andl %edi,%r15d+ xorl %r9d,%r14d+ addl %r13d,%r8d+ xorl %r10d,%r15d+ shrdl $2,%r14d,%r14d+ addl %r8d,%eax+ addl %r15d,%r8d+ movl %eax,%r13d+ addl %r8d,%r14d+ shrdl $14,%r13d,%r13d+ movl %r14d,%r8d+ movl %ebx,%r12d+ shrdl $9,%r14d,%r14d+ xorl %eax,%r13d+ xorl %ecx,%r12d+ shrdl $5,%r13d,%r13d+ xorl %r8d,%r14d+ andl %eax,%r12d+ xorl %eax,%r13d+ addl 48(%rsp),%edx+ movl %r8d,%r15d+ xorl %ecx,%r12d+ shrdl $11,%r14d,%r14d+ xorl %r9d,%r15d+ addl %r12d,%edx+ shrdl $6,%r13d,%r13d+ andl %r15d,%edi+ xorl %r8d,%r14d+ addl %r13d,%edx+ xorl %r9d,%edi+ shrdl $2,%r14d,%r14d+ addl %edx,%r11d+ addl %edi,%edx+ movl %r11d,%r13d+ addl %edx,%r14d+ shrdl $14,%r13d,%r13d+ movl %r14d,%edx+ movl %eax,%r12d+ shrdl $9,%r14d,%r14d+ xorl %r11d,%r13d+ xorl %ebx,%r12d+ shrdl $5,%r13d,%r13d+ xorl %edx,%r14d+ andl %r11d,%r12d+ xorl %r11d,%r13d+ addl 52(%rsp),%ecx+ movl %edx,%edi+ xorl %ebx,%r12d+ shrdl $11,%r14d,%r14d+ xorl %r8d,%edi+ addl %r12d,%ecx+ shrdl $6,%r13d,%r13d+ andl %edi,%r15d+ xorl %edx,%r14d+ addl %r13d,%ecx+ xorl %r8d,%r15d+ shrdl $2,%r14d,%r14d+ addl %ecx,%r10d+ addl %r15d,%ecx+ movl %r10d,%r13d+ addl %ecx,%r14d+ shrdl $14,%r13d,%r13d+ movl %r14d,%ecx+ movl %r11d,%r12d+ shrdl $9,%r14d,%r14d+ xorl %r10d,%r13d+ xorl %eax,%r12d+ shrdl $5,%r13d,%r13d+ xorl %ecx,%r14d+ andl %r10d,%r12d+ xorl %r10d,%r13d+ addl 56(%rsp),%ebx+ movl %ecx,%r15d+ xorl %eax,%r12d+ shrdl $11,%r14d,%r14d+ xorl %edx,%r15d+ addl %r12d,%ebx+ shrdl $6,%r13d,%r13d+ andl %r15d,%edi+ xorl %ecx,%r14d+ addl %r13d,%ebx+ xorl %edx,%edi+ shrdl $2,%r14d,%r14d+ addl %ebx,%r9d+ addl %edi,%ebx+ movl %r9d,%r13d+ addl %ebx,%r14d+ shrdl $14,%r13d,%r13d+ movl %r14d,%ebx+ movl %r10d,%r12d+ shrdl $9,%r14d,%r14d+ xorl %r9d,%r13d+ xorl %r11d,%r12d+ shrdl $5,%r13d,%r13d+ xorl %ebx,%r14d+ andl %r9d,%r12d+ xorl %r9d,%r13d+ addl 60(%rsp),%eax+ movl %ebx,%edi+ xorl %r11d,%r12d+ shrdl $11,%r14d,%r14d+ xorl %ecx,%edi+ addl %r12d,%eax+ shrdl $6,%r13d,%r13d+ andl %edi,%r15d+ xorl %ebx,%r14d+ addl %r13d,%eax+ xorl %ecx,%r15d+ shrdl $2,%r14d,%r14d+ addl %eax,%r8d+ addl %r15d,%eax+ movl %r8d,%r13d+ addl %eax,%r14d+ movq -64(%rbp),%rdi+ movl %r14d,%eax+ movq -56(%rbp),%rsi++ addl 0(%rdi),%eax+ addl 4(%rdi),%ebx+ addl 8(%rdi),%ecx+ addl 12(%rdi),%edx+ addl 16(%rdi),%r8d+ addl 20(%rdi),%r9d+ addl 24(%rdi),%r10d+ addl 28(%rdi),%r11d++ leaq 64(%rsi),%rsi+ cmpq -48(%rbp),%rsi++ movl %eax,0(%rdi)+ movl %ebx,4(%rdi)+ movl %ecx,8(%rdi)+ movl %edx,12(%rdi)+ movl %r8d,16(%rdi)+ movl %r9d,20(%rdi)+ movl %r10d,24(%rdi)+ movl %r11d,28(%rdi)+ jb .Lloop_avx++ vzeroupper+ movaps -160(%rbp),%xmm6+ movaps -144(%rbp),%xmm7+ movaps -128(%rbp),%xmm8+ movaps -112(%rbp),%xmm9+ movq -40(%rbp),%r15+ movq -32(%rbp),%r14+ movq -24(%rbp),%r13+ movq -16(%rbp),%r12+ movq -8(%rbp),%rbx+ movq %rbp,%rsp++ popq %rbp++.LSEH_epilogue_crypton_sha256_asm_block_data_order_avx:+ mov 8(%rsp),%rdi+ mov 16(%rsp),%rsi++ .byte 0xf3,0xc3++.LSEH_end_crypton_sha256_asm_block_data_order_avx:+.def crypton_sha256_asm_block_data_order_avx2; .scl 3; .type 32; .endef+.p2align 6+crypton_sha256_asm_block_data_order_avx2:+ .byte 0xf3,0x0f,0x1e,0xfa+ movq %rdi,8(%rsp)+ movq %rsi,16(%rsp)+ movq %rsp,%r11+.LSEH_begin_crypton_sha256_asm_block_data_order_avx2:+++ pushq %rbp++ movq %rsp,%rbp++ movq %rcx,%rdi+ movq %rdx,%rsi+ movq %r8,%rdx+.Lavx2_shortcut:+ pushq %rbx++ pushq %r12++ pushq %r13++ pushq %r14++ pushq %r15++ shlq $4,%rdx+ subq $120,%rsp++ leaq (%rsi,%rdx,4),%rdx+ movq %rdi,-64(%rbp)++ movq %rdx,-48(%rbp)+ movaps %xmm6,-160(%rbp)+ movaps %xmm7,-144(%rbp)+ movaps %xmm8,-128(%rbp)+ movaps %xmm9,-112(%rbp)++.LSEH_body_crypton_sha256_asm_block_data_order_avx2:+++ leaq -64(%rsp),%rsp+ vzeroupper+ andq $-64,%rsp+ subq $-64,%rsi+ movl 0(%rdi),%eax+ movq %rsi,%r12+ movl 4(%rdi),%ebx+ cmpq %rdx,%rsi+ movl 8(%rdi),%ecx+ cmoveq %rsp,%r12+ movl 12(%rdi),%edx+ movl 16(%rdi),%r8d+ movl 20(%rdi),%r9d+ movl 24(%rdi),%r10d+ movl 28(%rdi),%r11d+ vmovdqa K256+512+32(%rip),%ymm8+ vmovdqa K256+512+64(%rip),%ymm9+ jmp .Loop_avx2+.p2align 4+.Loop_avx2:+ vmovdqa K256+512(%rip),%ymm7+ movq %rsi,-56(%rbp)+ vmovdqu -64+0(%rsi),%xmm0+ vmovdqu -64+16(%rsi),%xmm1+ vmovdqu -64+32(%rsi),%xmm2+ vmovdqu -64+48(%rsi),%xmm3+ leaq K256(%rip),%rsi+ vinserti128 $1,(%r12),%ymm0,%ymm0+ vinserti128 $1,16(%r12),%ymm1,%ymm1+ vpshufb %ymm7,%ymm0,%ymm0+ vinserti128 $1,32(%r12),%ymm2,%ymm2+ vpshufb %ymm7,%ymm1,%ymm1+ vinserti128 $1,48(%r12),%ymm3,%ymm3++ vpshufb %ymm7,%ymm2,%ymm2+ vpaddd 0(%rsi),%ymm0,%ymm4+ vpshufb %ymm7,%ymm3,%ymm3+ vpaddd 32(%rsi),%ymm1,%ymm5+ vpaddd 64(%rsi),%ymm2,%ymm6+ vpaddd 96(%rsi),%ymm3,%ymm7+ vmovdqa %ymm4,0(%rsp)+ xorl %r14d,%r14d+ vmovdqa %ymm5,32(%rsp)+ leaq -64(%rsp),%rsp+ movl %ebx,%edi+ vmovdqa %ymm6,0(%rsp)+ xorl %ecx,%edi+ vmovdqa %ymm7,32(%rsp)+ movl %r9d,%r12d+ subq $-32*4,%rsi+ jmp .Lavx2_00_47++.p2align 4+.Lavx2_00_47:+ leaq -64(%rsp),%rsp+ vpalignr $4,%ymm0,%ymm1,%ymm4+ addl 0+128(%rsp),%r11d+ andl %r8d,%r12d+ rorxl $25,%r8d,%r13d+ vpalignr $4,%ymm2,%ymm3,%ymm7+ rorxl $11,%r8d,%r15d+ leal (%rax,%r14,1),%eax+ leal (%r11,%r12,1),%r11d+ vpsrld $7,%ymm4,%ymm6+ andnl %r10d,%r8d,%r12d+ xorl %r15d,%r13d+ rorxl $6,%r8d,%r14d+ vpaddd %ymm7,%ymm0,%ymm0+ leal (%r11,%r12,1),%r11d+ xorl %r14d,%r13d+ movl %eax,%r15d+ vpsrld $3,%ymm4,%ymm7+ rorxl $22,%eax,%r12d+ leal (%r11,%r13,1),%r11d+ xorl %ebx,%r15d+ vpslld $14,%ymm4,%ymm5+ rorxl $13,%eax,%r14d+ rorxl $2,%eax,%r13d+ leal (%rdx,%r11,1),%edx+ vpxor %ymm6,%ymm7,%ymm4+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %ebx,%edi+ vpshufd $250,%ymm3,%ymm7+ xorl %r13d,%r14d+ leal (%r11,%rdi,1),%r11d+ movl %r8d,%r12d+ vpsrld $11,%ymm6,%ymm6+ addl 4+128(%rsp),%r10d+ andl %edx,%r12d+ rorxl $25,%edx,%r13d+ vpxor %ymm5,%ymm4,%ymm4+ rorxl $11,%edx,%edi+ leal (%r11,%r14,1),%r11d+ leal (%r10,%r12,1),%r10d+ vpslld $11,%ymm5,%ymm5+ andnl %r9d,%edx,%r12d+ xorl %edi,%r13d+ rorxl $6,%edx,%r14d+ vpxor %ymm6,%ymm4,%ymm4+ leal (%r10,%r12,1),%r10d+ xorl %r14d,%r13d+ movl %r11d,%edi+ vpsrld $10,%ymm7,%ymm6+ rorxl $22,%r11d,%r12d+ leal (%r10,%r13,1),%r10d+ xorl %eax,%edi+ vpxor %ymm5,%ymm4,%ymm4+ rorxl $13,%r11d,%r14d+ rorxl $2,%r11d,%r13d+ leal (%rcx,%r10,1),%ecx+ vpsrlq $17,%ymm7,%ymm7+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %eax,%r15d+ vpaddd %ymm4,%ymm0,%ymm0+ xorl %r13d,%r14d+ leal (%r10,%r15,1),%r10d+ movl %edx,%r12d+ vpxor %ymm7,%ymm6,%ymm6+ addl 8+128(%rsp),%r9d+ andl %ecx,%r12d+ rorxl $25,%ecx,%r13d+ vpsrlq $2,%ymm7,%ymm7+ rorxl $11,%ecx,%r15d+ leal (%r10,%r14,1),%r10d+ leal (%r9,%r12,1),%r9d+ vpxor %ymm7,%ymm6,%ymm6+ andnl %r8d,%ecx,%r12d+ xorl %r15d,%r13d+ rorxl $6,%ecx,%r14d+ vpshufb %ymm8,%ymm6,%ymm6+ leal (%r9,%r12,1),%r9d+ xorl %r14d,%r13d+ movl %r10d,%r15d+ vpaddd %ymm6,%ymm0,%ymm0+ rorxl $22,%r10d,%r12d+ leal (%r9,%r13,1),%r9d+ xorl %r11d,%r15d+ vpshufd $80,%ymm0,%ymm7+ rorxl $13,%r10d,%r14d+ rorxl $2,%r10d,%r13d+ leal (%rbx,%r9,1),%ebx+ vpsrld $10,%ymm7,%ymm6+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %r11d,%edi+ vpsrlq $17,%ymm7,%ymm7+ xorl %r13d,%r14d+ leal (%r9,%rdi,1),%r9d+ movl %ecx,%r12d+ vpxor %ymm7,%ymm6,%ymm6+ addl 12+128(%rsp),%r8d+ andl %ebx,%r12d+ rorxl $25,%ebx,%r13d+ vpsrlq $2,%ymm7,%ymm7+ rorxl $11,%ebx,%edi+ leal (%r9,%r14,1),%r9d+ leal (%r8,%r12,1),%r8d+ vpxor %ymm7,%ymm6,%ymm6+ andnl %edx,%ebx,%r12d+ xorl %edi,%r13d+ rorxl $6,%ebx,%r14d+ vpshufb %ymm9,%ymm6,%ymm6+ leal (%r8,%r12,1),%r8d+ xorl %r14d,%r13d+ movl %r9d,%edi+ vpaddd %ymm6,%ymm0,%ymm0+ rorxl $22,%r9d,%r12d+ leal (%r8,%r13,1),%r8d+ xorl %r10d,%edi+ vpaddd 0(%rsi),%ymm0,%ymm6+ rorxl $13,%r9d,%r14d+ rorxl $2,%r9d,%r13d+ leal (%rax,%r8,1),%eax+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %r10d,%r15d+ xorl %r13d,%r14d+ leal (%r8,%r15,1),%r8d+ movl %ebx,%r12d+ vmovdqa %ymm6,0(%rsp)+ vpalignr $4,%ymm1,%ymm2,%ymm4+ addl 32+128(%rsp),%edx+ andl %eax,%r12d+ rorxl $25,%eax,%r13d+ vpalignr $4,%ymm3,%ymm0,%ymm7+ rorxl $11,%eax,%r15d+ leal (%r8,%r14,1),%r8d+ leal (%rdx,%r12,1),%edx+ vpsrld $7,%ymm4,%ymm6+ andnl %ecx,%eax,%r12d+ xorl %r15d,%r13d+ rorxl $6,%eax,%r14d+ vpaddd %ymm7,%ymm1,%ymm1+ leal (%rdx,%r12,1),%edx+ xorl %r14d,%r13d+ movl %r8d,%r15d+ vpsrld $3,%ymm4,%ymm7+ rorxl $22,%r8d,%r12d+ leal (%rdx,%r13,1),%edx+ xorl %r9d,%r15d+ vpslld $14,%ymm4,%ymm5+ rorxl $13,%r8d,%r14d+ rorxl $2,%r8d,%r13d+ leal (%r11,%rdx,1),%r11d+ vpxor %ymm6,%ymm7,%ymm4+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %r9d,%edi+ vpshufd $250,%ymm0,%ymm7+ xorl %r13d,%r14d+ leal (%rdx,%rdi,1),%edx+ movl %eax,%r12d+ vpsrld $11,%ymm6,%ymm6+ addl 36+128(%rsp),%ecx+ andl %r11d,%r12d+ rorxl $25,%r11d,%r13d+ vpxor %ymm5,%ymm4,%ymm4+ rorxl $11,%r11d,%edi+ leal (%rdx,%r14,1),%edx+ leal (%rcx,%r12,1),%ecx+ vpslld $11,%ymm5,%ymm5+ andnl %ebx,%r11d,%r12d+ xorl %edi,%r13d+ rorxl $6,%r11d,%r14d+ vpxor %ymm6,%ymm4,%ymm4+ leal (%rcx,%r12,1),%ecx+ xorl %r14d,%r13d+ movl %edx,%edi+ vpsrld $10,%ymm7,%ymm6+ rorxl $22,%edx,%r12d+ leal (%rcx,%r13,1),%ecx+ xorl %r8d,%edi+ vpxor %ymm5,%ymm4,%ymm4+ rorxl $13,%edx,%r14d+ rorxl $2,%edx,%r13d+ leal (%r10,%rcx,1),%r10d+ vpsrlq $17,%ymm7,%ymm7+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %r8d,%r15d+ vpaddd %ymm4,%ymm1,%ymm1+ xorl %r13d,%r14d+ leal (%rcx,%r15,1),%ecx+ movl %r11d,%r12d+ vpxor %ymm7,%ymm6,%ymm6+ addl 40+128(%rsp),%ebx+ andl %r10d,%r12d+ rorxl $25,%r10d,%r13d+ vpsrlq $2,%ymm7,%ymm7+ rorxl $11,%r10d,%r15d+ leal (%rcx,%r14,1),%ecx+ leal (%rbx,%r12,1),%ebx+ vpxor %ymm7,%ymm6,%ymm6+ andnl %eax,%r10d,%r12d+ xorl %r15d,%r13d+ rorxl $6,%r10d,%r14d+ vpshufb %ymm8,%ymm6,%ymm6+ leal (%rbx,%r12,1),%ebx+ xorl %r14d,%r13d+ movl %ecx,%r15d+ vpaddd %ymm6,%ymm1,%ymm1+ rorxl $22,%ecx,%r12d+ leal (%rbx,%r13,1),%ebx+ xorl %edx,%r15d+ vpshufd $80,%ymm1,%ymm7+ rorxl $13,%ecx,%r14d+ rorxl $2,%ecx,%r13d+ leal (%r9,%rbx,1),%r9d+ vpsrld $10,%ymm7,%ymm6+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %edx,%edi+ vpsrlq $17,%ymm7,%ymm7+ xorl %r13d,%r14d+ leal (%rbx,%rdi,1),%ebx+ movl %r10d,%r12d+ vpxor %ymm7,%ymm6,%ymm6+ addl 44+128(%rsp),%eax+ andl %r9d,%r12d+ rorxl $25,%r9d,%r13d+ vpsrlq $2,%ymm7,%ymm7+ rorxl $11,%r9d,%edi+ leal (%rbx,%r14,1),%ebx+ leal (%rax,%r12,1),%eax+ vpxor %ymm7,%ymm6,%ymm6+ andnl %r11d,%r9d,%r12d+ xorl %edi,%r13d+ rorxl $6,%r9d,%r14d+ vpshufb %ymm9,%ymm6,%ymm6+ leal (%rax,%r12,1),%eax+ xorl %r14d,%r13d+ movl %ebx,%edi+ vpaddd %ymm6,%ymm1,%ymm1+ rorxl $22,%ebx,%r12d+ leal (%rax,%r13,1),%eax+ xorl %ecx,%edi+ vpaddd 32(%rsi),%ymm1,%ymm6+ rorxl $13,%ebx,%r14d+ rorxl $2,%ebx,%r13d+ leal (%r8,%rax,1),%r8d+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %ecx,%r15d+ xorl %r13d,%r14d+ leal (%rax,%r15,1),%eax+ movl %r9d,%r12d+ vmovdqa %ymm6,32(%rsp)+ leaq -64(%rsp),%rsp+ vpalignr $4,%ymm2,%ymm3,%ymm4+ addl 0+128(%rsp),%r11d+ andl %r8d,%r12d+ rorxl $25,%r8d,%r13d+ vpalignr $4,%ymm0,%ymm1,%ymm7+ rorxl $11,%r8d,%r15d+ leal (%rax,%r14,1),%eax+ leal (%r11,%r12,1),%r11d+ vpsrld $7,%ymm4,%ymm6+ andnl %r10d,%r8d,%r12d+ xorl %r15d,%r13d+ rorxl $6,%r8d,%r14d+ vpaddd %ymm7,%ymm2,%ymm2+ leal (%r11,%r12,1),%r11d+ xorl %r14d,%r13d+ movl %eax,%r15d+ vpsrld $3,%ymm4,%ymm7+ rorxl $22,%eax,%r12d+ leal (%r11,%r13,1),%r11d+ xorl %ebx,%r15d+ vpslld $14,%ymm4,%ymm5+ rorxl $13,%eax,%r14d+ rorxl $2,%eax,%r13d+ leal (%rdx,%r11,1),%edx+ vpxor %ymm6,%ymm7,%ymm4+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %ebx,%edi+ vpshufd $250,%ymm1,%ymm7+ xorl %r13d,%r14d+ leal (%r11,%rdi,1),%r11d+ movl %r8d,%r12d+ vpsrld $11,%ymm6,%ymm6+ addl 4+128(%rsp),%r10d+ andl %edx,%r12d+ rorxl $25,%edx,%r13d+ vpxor %ymm5,%ymm4,%ymm4+ rorxl $11,%edx,%edi+ leal (%r11,%r14,1),%r11d+ leal (%r10,%r12,1),%r10d+ vpslld $11,%ymm5,%ymm5+ andnl %r9d,%edx,%r12d+ xorl %edi,%r13d+ rorxl $6,%edx,%r14d+ vpxor %ymm6,%ymm4,%ymm4+ leal (%r10,%r12,1),%r10d+ xorl %r14d,%r13d+ movl %r11d,%edi+ vpsrld $10,%ymm7,%ymm6+ rorxl $22,%r11d,%r12d+ leal (%r10,%r13,1),%r10d+ xorl %eax,%edi+ vpxor %ymm5,%ymm4,%ymm4+ rorxl $13,%r11d,%r14d+ rorxl $2,%r11d,%r13d+ leal (%rcx,%r10,1),%ecx+ vpsrlq $17,%ymm7,%ymm7+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %eax,%r15d+ vpaddd %ymm4,%ymm2,%ymm2+ xorl %r13d,%r14d+ leal (%r10,%r15,1),%r10d+ movl %edx,%r12d+ vpxor %ymm7,%ymm6,%ymm6+ addl 8+128(%rsp),%r9d+ andl %ecx,%r12d+ rorxl $25,%ecx,%r13d+ vpsrlq $2,%ymm7,%ymm7+ rorxl $11,%ecx,%r15d+ leal (%r10,%r14,1),%r10d+ leal (%r9,%r12,1),%r9d+ vpxor %ymm7,%ymm6,%ymm6+ andnl %r8d,%ecx,%r12d+ xorl %r15d,%r13d+ rorxl $6,%ecx,%r14d+ vpshufb %ymm8,%ymm6,%ymm6+ leal (%r9,%r12,1),%r9d+ xorl %r14d,%r13d+ movl %r10d,%r15d+ vpaddd %ymm6,%ymm2,%ymm2+ rorxl $22,%r10d,%r12d+ leal (%r9,%r13,1),%r9d+ xorl %r11d,%r15d+ vpshufd $80,%ymm2,%ymm7+ rorxl $13,%r10d,%r14d+ rorxl $2,%r10d,%r13d+ leal (%rbx,%r9,1),%ebx+ vpsrld $10,%ymm7,%ymm6+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %r11d,%edi+ vpsrlq $17,%ymm7,%ymm7+ xorl %r13d,%r14d+ leal (%r9,%rdi,1),%r9d+ movl %ecx,%r12d+ vpxor %ymm7,%ymm6,%ymm6+ addl 12+128(%rsp),%r8d+ andl %ebx,%r12d+ rorxl $25,%ebx,%r13d+ vpsrlq $2,%ymm7,%ymm7+ rorxl $11,%ebx,%edi+ leal (%r9,%r14,1),%r9d+ leal (%r8,%r12,1),%r8d+ vpxor %ymm7,%ymm6,%ymm6+ andnl %edx,%ebx,%r12d+ xorl %edi,%r13d+ rorxl $6,%ebx,%r14d+ vpshufb %ymm9,%ymm6,%ymm6+ leal (%r8,%r12,1),%r8d+ xorl %r14d,%r13d+ movl %r9d,%edi+ vpaddd %ymm6,%ymm2,%ymm2+ rorxl $22,%r9d,%r12d+ leal (%r8,%r13,1),%r8d+ xorl %r10d,%edi+ vpaddd 64(%rsi),%ymm2,%ymm6+ rorxl $13,%r9d,%r14d+ rorxl $2,%r9d,%r13d+ leal (%rax,%r8,1),%eax+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %r10d,%r15d+ xorl %r13d,%r14d+ leal (%r8,%r15,1),%r8d+ movl %ebx,%r12d+ vmovdqa %ymm6,0(%rsp)+ vpalignr $4,%ymm3,%ymm0,%ymm4+ addl 32+128(%rsp),%edx+ andl %eax,%r12d+ rorxl $25,%eax,%r13d+ vpalignr $4,%ymm1,%ymm2,%ymm7+ rorxl $11,%eax,%r15d+ leal (%r8,%r14,1),%r8d+ leal (%rdx,%r12,1),%edx+ vpsrld $7,%ymm4,%ymm6+ andnl %ecx,%eax,%r12d+ xorl %r15d,%r13d+ rorxl $6,%eax,%r14d+ vpaddd %ymm7,%ymm3,%ymm3+ leal (%rdx,%r12,1),%edx+ xorl %r14d,%r13d+ movl %r8d,%r15d+ vpsrld $3,%ymm4,%ymm7+ rorxl $22,%r8d,%r12d+ leal (%rdx,%r13,1),%edx+ xorl %r9d,%r15d+ vpslld $14,%ymm4,%ymm5+ rorxl $13,%r8d,%r14d+ rorxl $2,%r8d,%r13d+ leal (%r11,%rdx,1),%r11d+ vpxor %ymm6,%ymm7,%ymm4+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %r9d,%edi+ vpshufd $250,%ymm2,%ymm7+ xorl %r13d,%r14d+ leal (%rdx,%rdi,1),%edx+ movl %eax,%r12d+ vpsrld $11,%ymm6,%ymm6+ addl 36+128(%rsp),%ecx+ andl %r11d,%r12d+ rorxl $25,%r11d,%r13d+ vpxor %ymm5,%ymm4,%ymm4+ rorxl $11,%r11d,%edi+ leal (%rdx,%r14,1),%edx+ leal (%rcx,%r12,1),%ecx+ vpslld $11,%ymm5,%ymm5+ andnl %ebx,%r11d,%r12d+ xorl %edi,%r13d+ rorxl $6,%r11d,%r14d+ vpxor %ymm6,%ymm4,%ymm4+ leal (%rcx,%r12,1),%ecx+ xorl %r14d,%r13d+ movl %edx,%edi+ vpsrld $10,%ymm7,%ymm6+ rorxl $22,%edx,%r12d+ leal (%rcx,%r13,1),%ecx+ xorl %r8d,%edi+ vpxor %ymm5,%ymm4,%ymm4+ rorxl $13,%edx,%r14d+ rorxl $2,%edx,%r13d+ leal (%r10,%rcx,1),%r10d+ vpsrlq $17,%ymm7,%ymm7+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %r8d,%r15d+ vpaddd %ymm4,%ymm3,%ymm3+ xorl %r13d,%r14d+ leal (%rcx,%r15,1),%ecx+ movl %r11d,%r12d+ vpxor %ymm7,%ymm6,%ymm6+ addl 40+128(%rsp),%ebx+ andl %r10d,%r12d+ rorxl $25,%r10d,%r13d+ vpsrlq $2,%ymm7,%ymm7+ rorxl $11,%r10d,%r15d+ leal (%rcx,%r14,1),%ecx+ leal (%rbx,%r12,1),%ebx+ vpxor %ymm7,%ymm6,%ymm6+ andnl %eax,%r10d,%r12d+ xorl %r15d,%r13d+ rorxl $6,%r10d,%r14d+ vpshufb %ymm8,%ymm6,%ymm6+ leal (%rbx,%r12,1),%ebx+ xorl %r14d,%r13d+ movl %ecx,%r15d+ vpaddd %ymm6,%ymm3,%ymm3+ rorxl $22,%ecx,%r12d+ leal (%rbx,%r13,1),%ebx+ xorl %edx,%r15d+ vpshufd $80,%ymm3,%ymm7+ rorxl $13,%ecx,%r14d+ rorxl $2,%ecx,%r13d+ leal (%r9,%rbx,1),%r9d+ vpsrld $10,%ymm7,%ymm6+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %edx,%edi+ vpsrlq $17,%ymm7,%ymm7+ xorl %r13d,%r14d+ leal (%rbx,%rdi,1),%ebx+ movl %r10d,%r12d+ vpxor %ymm7,%ymm6,%ymm6+ addl 44+128(%rsp),%eax+ andl %r9d,%r12d+ rorxl $25,%r9d,%r13d+ vpsrlq $2,%ymm7,%ymm7+ rorxl $11,%r9d,%edi+ leal (%rbx,%r14,1),%ebx+ leal (%rax,%r12,1),%eax+ vpxor %ymm7,%ymm6,%ymm6+ andnl %r11d,%r9d,%r12d+ xorl %edi,%r13d+ rorxl $6,%r9d,%r14d+ vpshufb %ymm9,%ymm6,%ymm6+ leal (%rax,%r12,1),%eax+ xorl %r14d,%r13d+ movl %ebx,%edi+ vpaddd %ymm6,%ymm3,%ymm3+ rorxl $22,%ebx,%r12d+ leal (%rax,%r13,1),%eax+ xorl %ecx,%edi+ vpaddd 96(%rsi),%ymm3,%ymm6+ rorxl $13,%ebx,%r14d+ rorxl $2,%ebx,%r13d+ leal (%r8,%rax,1),%r8d+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %ecx,%r15d+ xorl %r13d,%r14d+ leal (%rax,%r15,1),%eax+ movl %r9d,%r12d+ vmovdqa %ymm6,32(%rsp)+ leaq 128(%rsi),%rsi+ cmpb $0,3(%rsi)+ jne .Lavx2_00_47+ addl 0+64(%rsp),%r11d+ andl %r8d,%r12d+ rorxl $25,%r8d,%r13d+ rorxl $11,%r8d,%r15d+ leal (%rax,%r14,1),%eax+ leal (%r11,%r12,1),%r11d+ andnl %r10d,%r8d,%r12d+ xorl %r15d,%r13d+ rorxl $6,%r8d,%r14d+ leal (%r11,%r12,1),%r11d+ xorl %r14d,%r13d+ movl %eax,%r15d+ rorxl $22,%eax,%r12d+ leal (%r11,%r13,1),%r11d+ xorl %ebx,%r15d+ rorxl $13,%eax,%r14d+ rorxl $2,%eax,%r13d+ leal (%rdx,%r11,1),%edx+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %ebx,%edi+ xorl %r13d,%r14d+ leal (%r11,%rdi,1),%r11d+ movl %r8d,%r12d+ addl 4+64(%rsp),%r10d+ andl %edx,%r12d+ rorxl $25,%edx,%r13d+ rorxl $11,%edx,%edi+ leal (%r11,%r14,1),%r11d+ leal (%r10,%r12,1),%r10d+ andnl %r9d,%edx,%r12d+ xorl %edi,%r13d+ rorxl $6,%edx,%r14d+ leal (%r10,%r12,1),%r10d+ xorl %r14d,%r13d+ movl %r11d,%edi+ rorxl $22,%r11d,%r12d+ leal (%r10,%r13,1),%r10d+ xorl %eax,%edi+ rorxl $13,%r11d,%r14d+ rorxl $2,%r11d,%r13d+ leal (%rcx,%r10,1),%ecx+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %eax,%r15d+ xorl %r13d,%r14d+ leal (%r10,%r15,1),%r10d+ movl %edx,%r12d+ addl 8+64(%rsp),%r9d+ andl %ecx,%r12d+ rorxl $25,%ecx,%r13d+ rorxl $11,%ecx,%r15d+ leal (%r10,%r14,1),%r10d+ leal (%r9,%r12,1),%r9d+ andnl %r8d,%ecx,%r12d+ xorl %r15d,%r13d+ rorxl $6,%ecx,%r14d+ leal (%r9,%r12,1),%r9d+ xorl %r14d,%r13d+ movl %r10d,%r15d+ rorxl $22,%r10d,%r12d+ leal (%r9,%r13,1),%r9d+ xorl %r11d,%r15d+ rorxl $13,%r10d,%r14d+ rorxl $2,%r10d,%r13d+ leal (%rbx,%r9,1),%ebx+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %r11d,%edi+ xorl %r13d,%r14d+ leal (%r9,%rdi,1),%r9d+ movl %ecx,%r12d+ addl 12+64(%rsp),%r8d+ andl %ebx,%r12d+ rorxl $25,%ebx,%r13d+ rorxl $11,%ebx,%edi+ leal (%r9,%r14,1),%r9d+ leal (%r8,%r12,1),%r8d+ andnl %edx,%ebx,%r12d+ xorl %edi,%r13d+ rorxl $6,%ebx,%r14d+ leal (%r8,%r12,1),%r8d+ xorl %r14d,%r13d+ movl %r9d,%edi+ rorxl $22,%r9d,%r12d+ leal (%r8,%r13,1),%r8d+ xorl %r10d,%edi+ rorxl $13,%r9d,%r14d+ rorxl $2,%r9d,%r13d+ leal (%rax,%r8,1),%eax+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %r10d,%r15d+ xorl %r13d,%r14d+ leal (%r8,%r15,1),%r8d+ movl %ebx,%r12d+ addl 32+64(%rsp),%edx+ andl %eax,%r12d+ rorxl $25,%eax,%r13d+ rorxl $11,%eax,%r15d+ leal (%r8,%r14,1),%r8d+ leal (%rdx,%r12,1),%edx+ andnl %ecx,%eax,%r12d+ xorl %r15d,%r13d+ rorxl $6,%eax,%r14d+ leal (%rdx,%r12,1),%edx+ xorl %r14d,%r13d+ movl %r8d,%r15d+ rorxl $22,%r8d,%r12d+ leal (%rdx,%r13,1),%edx+ xorl %r9d,%r15d+ rorxl $13,%r8d,%r14d+ rorxl $2,%r8d,%r13d+ leal (%r11,%rdx,1),%r11d+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %r9d,%edi+ xorl %r13d,%r14d+ leal (%rdx,%rdi,1),%edx+ movl %eax,%r12d+ addl 36+64(%rsp),%ecx+ andl %r11d,%r12d+ rorxl $25,%r11d,%r13d+ rorxl $11,%r11d,%edi+ leal (%rdx,%r14,1),%edx+ leal (%rcx,%r12,1),%ecx+ andnl %ebx,%r11d,%r12d+ xorl %edi,%r13d+ rorxl $6,%r11d,%r14d+ leal (%rcx,%r12,1),%ecx+ xorl %r14d,%r13d+ movl %edx,%edi+ rorxl $22,%edx,%r12d+ leal (%rcx,%r13,1),%ecx+ xorl %r8d,%edi+ rorxl $13,%edx,%r14d+ rorxl $2,%edx,%r13d+ leal (%r10,%rcx,1),%r10d+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %r8d,%r15d+ xorl %r13d,%r14d+ leal (%rcx,%r15,1),%ecx+ movl %r11d,%r12d+ addl 40+64(%rsp),%ebx+ andl %r10d,%r12d+ rorxl $25,%r10d,%r13d+ rorxl $11,%r10d,%r15d+ leal (%rcx,%r14,1),%ecx+ leal (%rbx,%r12,1),%ebx+ andnl %eax,%r10d,%r12d+ xorl %r15d,%r13d+ rorxl $6,%r10d,%r14d+ leal (%rbx,%r12,1),%ebx+ xorl %r14d,%r13d+ movl %ecx,%r15d+ rorxl $22,%ecx,%r12d+ leal (%rbx,%r13,1),%ebx+ xorl %edx,%r15d+ rorxl $13,%ecx,%r14d+ rorxl $2,%ecx,%r13d+ leal (%r9,%rbx,1),%r9d+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %edx,%edi+ xorl %r13d,%r14d+ leal (%rbx,%rdi,1),%ebx+ movl %r10d,%r12d+ addl 44+64(%rsp),%eax+ andl %r9d,%r12d+ rorxl $25,%r9d,%r13d+ rorxl $11,%r9d,%edi+ leal (%rbx,%r14,1),%ebx+ leal (%rax,%r12,1),%eax+ andnl %r11d,%r9d,%r12d+ xorl %edi,%r13d+ rorxl $6,%r9d,%r14d+ leal (%rax,%r12,1),%eax+ xorl %r14d,%r13d+ movl %ebx,%edi+ rorxl $22,%ebx,%r12d+ leal (%rax,%r13,1),%eax+ xorl %ecx,%edi+ rorxl $13,%ebx,%r14d+ rorxl $2,%ebx,%r13d+ leal (%r8,%rax,1),%r8d+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %ecx,%r15d+ xorl %r13d,%r14d+ leal (%rax,%r15,1),%eax+ movl %r9d,%r12d+ addl 0(%rsp),%r11d+ andl %r8d,%r12d+ rorxl $25,%r8d,%r13d+ rorxl $11,%r8d,%r15d+ leal (%rax,%r14,1),%eax+ leal (%r11,%r12,1),%r11d+ andnl %r10d,%r8d,%r12d+ xorl %r15d,%r13d+ rorxl $6,%r8d,%r14d+ leal (%r11,%r12,1),%r11d+ xorl %r14d,%r13d+ movl %eax,%r15d+ rorxl $22,%eax,%r12d+ leal (%r11,%r13,1),%r11d+ xorl %ebx,%r15d+ rorxl $13,%eax,%r14d+ rorxl $2,%eax,%r13d+ leal (%rdx,%r11,1),%edx+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %ebx,%edi+ xorl %r13d,%r14d+ leal (%r11,%rdi,1),%r11d+ movl %r8d,%r12d+ addl 4(%rsp),%r10d+ andl %edx,%r12d+ rorxl $25,%edx,%r13d+ rorxl $11,%edx,%edi+ leal (%r11,%r14,1),%r11d+ leal (%r10,%r12,1),%r10d+ andnl %r9d,%edx,%r12d+ xorl %edi,%r13d+ rorxl $6,%edx,%r14d+ leal (%r10,%r12,1),%r10d+ xorl %r14d,%r13d+ movl %r11d,%edi+ rorxl $22,%r11d,%r12d+ leal (%r10,%r13,1),%r10d+ xorl %eax,%edi+ rorxl $13,%r11d,%r14d+ rorxl $2,%r11d,%r13d+ leal (%rcx,%r10,1),%ecx+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %eax,%r15d+ xorl %r13d,%r14d+ leal (%r10,%r15,1),%r10d+ movl %edx,%r12d+ addl 8(%rsp),%r9d+ andl %ecx,%r12d+ rorxl $25,%ecx,%r13d+ rorxl $11,%ecx,%r15d+ leal (%r10,%r14,1),%r10d+ leal (%r9,%r12,1),%r9d+ andnl %r8d,%ecx,%r12d+ xorl %r15d,%r13d+ rorxl $6,%ecx,%r14d+ leal (%r9,%r12,1),%r9d+ xorl %r14d,%r13d+ movl %r10d,%r15d+ rorxl $22,%r10d,%r12d+ leal (%r9,%r13,1),%r9d+ xorl %r11d,%r15d+ rorxl $13,%r10d,%r14d+ rorxl $2,%r10d,%r13d+ leal (%rbx,%r9,1),%ebx+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %r11d,%edi+ xorl %r13d,%r14d+ leal (%r9,%rdi,1),%r9d+ movl %ecx,%r12d+ addl 12(%rsp),%r8d+ andl %ebx,%r12d+ rorxl $25,%ebx,%r13d+ rorxl $11,%ebx,%edi+ leal (%r9,%r14,1),%r9d+ leal (%r8,%r12,1),%r8d+ andnl %edx,%ebx,%r12d+ xorl %edi,%r13d+ rorxl $6,%ebx,%r14d+ leal (%r8,%r12,1),%r8d+ xorl %r14d,%r13d+ movl %r9d,%edi+ rorxl $22,%r9d,%r12d+ leal (%r8,%r13,1),%r8d+ xorl %r10d,%edi+ rorxl $13,%r9d,%r14d+ rorxl $2,%r9d,%r13d+ leal (%rax,%r8,1),%eax+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %r10d,%r15d+ xorl %r13d,%r14d+ leal (%r8,%r15,1),%r8d+ movl %ebx,%r12d+ addl 32(%rsp),%edx+ andl %eax,%r12d+ rorxl $25,%eax,%r13d+ rorxl $11,%eax,%r15d+ leal (%r8,%r14,1),%r8d+ leal (%rdx,%r12,1),%edx+ andnl %ecx,%eax,%r12d+ xorl %r15d,%r13d+ rorxl $6,%eax,%r14d+ leal (%rdx,%r12,1),%edx+ xorl %r14d,%r13d+ movl %r8d,%r15d+ rorxl $22,%r8d,%r12d+ leal (%rdx,%r13,1),%edx+ xorl %r9d,%r15d+ rorxl $13,%r8d,%r14d+ rorxl $2,%r8d,%r13d+ leal (%r11,%rdx,1),%r11d+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %r9d,%edi+ xorl %r13d,%r14d+ leal (%rdx,%rdi,1),%edx+ movl %eax,%r12d+ addl 36(%rsp),%ecx+ andl %r11d,%r12d+ rorxl $25,%r11d,%r13d+ rorxl $11,%r11d,%edi+ leal (%rdx,%r14,1),%edx+ leal (%rcx,%r12,1),%ecx+ andnl %ebx,%r11d,%r12d+ xorl %edi,%r13d+ rorxl $6,%r11d,%r14d+ leal (%rcx,%r12,1),%ecx+ xorl %r14d,%r13d+ movl %edx,%edi+ rorxl $22,%edx,%r12d+ leal (%rcx,%r13,1),%ecx+ xorl %r8d,%edi+ rorxl $13,%edx,%r14d+ rorxl $2,%edx,%r13d+ leal (%r10,%rcx,1),%r10d+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %r8d,%r15d+ xorl %r13d,%r14d+ leal (%rcx,%r15,1),%ecx+ movl %r11d,%r12d+ addl 40(%rsp),%ebx+ andl %r10d,%r12d+ rorxl $25,%r10d,%r13d+ rorxl $11,%r10d,%r15d+ leal (%rcx,%r14,1),%ecx+ leal (%rbx,%r12,1),%ebx+ andnl %eax,%r10d,%r12d+ xorl %r15d,%r13d+ rorxl $6,%r10d,%r14d+ leal (%rbx,%r12,1),%ebx+ xorl %r14d,%r13d+ movl %ecx,%r15d+ rorxl $22,%ecx,%r12d+ leal (%rbx,%r13,1),%ebx+ xorl %edx,%r15d+ rorxl $13,%ecx,%r14d+ rorxl $2,%ecx,%r13d+ leal (%r9,%rbx,1),%r9d+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %edx,%edi+ xorl %r13d,%r14d+ leal (%rbx,%rdi,1),%ebx+ movl %r10d,%r12d+ addl 44(%rsp),%eax+ andl %r9d,%r12d+ rorxl $25,%r9d,%r13d+ rorxl $11,%r9d,%edi+ leal (%rbx,%r14,1),%ebx+ leal (%rax,%r12,1),%eax+ andnl %r11d,%r9d,%r12d+ xorl %edi,%r13d+ rorxl $6,%r9d,%r14d+ leal (%rax,%r12,1),%eax+ xorl %r14d,%r13d+ movl %ebx,%edi+ rorxl $22,%ebx,%r12d+ leal (%rax,%r13,1),%eax+ xorl %ecx,%edi+ rorxl $13,%ebx,%r14d+ rorxl $2,%ebx,%r13d+ leal (%r8,%rax,1),%r8d+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %ecx,%r15d+ xorl %r13d,%r14d+ leal (%rax,%r15,1),%eax+ movl %r9d,%r12d+ movq -64(%rbp),%rdi+ addl %r14d,%eax+ movl -56(%rbp),%r12d++ addl 0(%rdi),%eax+ addl 4(%rdi),%ebx+ addl 8(%rdi),%ecx+ addl 12(%rdi),%edx+ addl 16(%rdi),%r8d+ addl 20(%rdi),%r9d+ addl 24(%rdi),%r10d+ addl 28(%rdi),%r11d++ movl %eax,0(%rdi)+ movl %ebx,4(%rdi)+ movl %ecx,8(%rdi)+ movl %edx,12(%rdi)+ movl %r8d,16(%rdi)+ movl %r9d,20(%rdi)+ movl %r10d,24(%rdi)+ movl %r11d,28(%rdi)++ cmpl -48(%rbp),%r12d+ je .Ldone_avx2++ leaq 448(%rsp),%rsi+ xorl %r14d,%r14d+ movl %ebx,%edi+ xorl %ecx,%edi+ movl %r9d,%r12d+ jmp .Lower_avx2+.p2align 4+.Lower_avx2:+ addl 0+16(%rsi),%r11d+ andl %r8d,%r12d+ rorxl $25,%r8d,%r13d+ rorxl $11,%r8d,%r15d+ leal (%rax,%r14,1),%eax+ leal (%r11,%r12,1),%r11d+ andnl %r10d,%r8d,%r12d+ xorl %r15d,%r13d+ rorxl $6,%r8d,%r14d+ leal (%r11,%r12,1),%r11d+ xorl %r14d,%r13d+ movl %eax,%r15d+ rorxl $22,%eax,%r12d+ leal (%r11,%r13,1),%r11d+ xorl %ebx,%r15d+ rorxl $13,%eax,%r14d+ rorxl $2,%eax,%r13d+ leal (%rdx,%r11,1),%edx+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %ebx,%edi+ xorl %r13d,%r14d+ leal (%r11,%rdi,1),%r11d+ movl %r8d,%r12d+ addl 4+16(%rsi),%r10d+ andl %edx,%r12d+ rorxl $25,%edx,%r13d+ rorxl $11,%edx,%edi+ leal (%r11,%r14,1),%r11d+ leal (%r10,%r12,1),%r10d+ andnl %r9d,%edx,%r12d+ xorl %edi,%r13d+ rorxl $6,%edx,%r14d+ leal (%r10,%r12,1),%r10d+ xorl %r14d,%r13d+ movl %r11d,%edi+ rorxl $22,%r11d,%r12d+ leal (%r10,%r13,1),%r10d+ xorl %eax,%edi+ rorxl $13,%r11d,%r14d+ rorxl $2,%r11d,%r13d+ leal (%rcx,%r10,1),%ecx+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %eax,%r15d+ xorl %r13d,%r14d+ leal (%r10,%r15,1),%r10d+ movl %edx,%r12d+ addl 8+16(%rsi),%r9d+ andl %ecx,%r12d+ rorxl $25,%ecx,%r13d+ rorxl $11,%ecx,%r15d+ leal (%r10,%r14,1),%r10d+ leal (%r9,%r12,1),%r9d+ andnl %r8d,%ecx,%r12d+ xorl %r15d,%r13d+ rorxl $6,%ecx,%r14d+ leal (%r9,%r12,1),%r9d+ xorl %r14d,%r13d+ movl %r10d,%r15d+ rorxl $22,%r10d,%r12d+ leal (%r9,%r13,1),%r9d+ xorl %r11d,%r15d+ rorxl $13,%r10d,%r14d+ rorxl $2,%r10d,%r13d+ leal (%rbx,%r9,1),%ebx+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %r11d,%edi+ xorl %r13d,%r14d+ leal (%r9,%rdi,1),%r9d+ movl %ecx,%r12d+ addl 12+16(%rsi),%r8d+ andl %ebx,%r12d+ rorxl $25,%ebx,%r13d+ rorxl $11,%ebx,%edi+ leal (%r9,%r14,1),%r9d+ leal (%r8,%r12,1),%r8d+ andnl %edx,%ebx,%r12d+ xorl %edi,%r13d+ rorxl $6,%ebx,%r14d+ leal (%r8,%r12,1),%r8d+ xorl %r14d,%r13d+ movl %r9d,%edi+ rorxl $22,%r9d,%r12d+ leal (%r8,%r13,1),%r8d+ xorl %r10d,%edi+ rorxl $13,%r9d,%r14d+ rorxl $2,%r9d,%r13d+ leal (%rax,%r8,1),%eax+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %r10d,%r15d+ xorl %r13d,%r14d+ leal (%r8,%r15,1),%r8d+ movl %ebx,%r12d+ addl 32+16(%rsi),%edx+ andl %eax,%r12d+ rorxl $25,%eax,%r13d+ rorxl $11,%eax,%r15d+ leal (%r8,%r14,1),%r8d+ leal (%rdx,%r12,1),%edx+ andnl %ecx,%eax,%r12d+ xorl %r15d,%r13d+ rorxl $6,%eax,%r14d+ leal (%rdx,%r12,1),%edx+ xorl %r14d,%r13d+ movl %r8d,%r15d+ rorxl $22,%r8d,%r12d+ leal (%rdx,%r13,1),%edx+ xorl %r9d,%r15d+ rorxl $13,%r8d,%r14d+ rorxl $2,%r8d,%r13d+ leal (%r11,%rdx,1),%r11d+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %r9d,%edi+ xorl %r13d,%r14d+ leal (%rdx,%rdi,1),%edx+ movl %eax,%r12d+ addl 36+16(%rsi),%ecx+ andl %r11d,%r12d+ rorxl $25,%r11d,%r13d+ rorxl $11,%r11d,%edi+ leal (%rdx,%r14,1),%edx+ leal (%rcx,%r12,1),%ecx+ andnl %ebx,%r11d,%r12d+ xorl %edi,%r13d+ rorxl $6,%r11d,%r14d+ leal (%rcx,%r12,1),%ecx+ xorl %r14d,%r13d+ movl %edx,%edi+ rorxl $22,%edx,%r12d+ leal (%rcx,%r13,1),%ecx+ xorl %r8d,%edi+ rorxl $13,%edx,%r14d+ rorxl $2,%edx,%r13d+ leal (%r10,%rcx,1),%r10d+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %r8d,%r15d+ xorl %r13d,%r14d+ leal (%rcx,%r15,1),%ecx+ movl %r11d,%r12d+ addl 40+16(%rsi),%ebx+ andl %r10d,%r12d+ rorxl $25,%r10d,%r13d+ rorxl $11,%r10d,%r15d+ leal (%rcx,%r14,1),%ecx+ leal (%rbx,%r12,1),%ebx+ andnl %eax,%r10d,%r12d+ xorl %r15d,%r13d+ rorxl $6,%r10d,%r14d+ leal (%rbx,%r12,1),%ebx+ xorl %r14d,%r13d+ movl %ecx,%r15d+ rorxl $22,%ecx,%r12d+ leal (%rbx,%r13,1),%ebx+ xorl %edx,%r15d+ rorxl $13,%ecx,%r14d+ rorxl $2,%ecx,%r13d+ leal (%r9,%rbx,1),%r9d+ andl %r15d,%edi+ xorl %r12d,%r14d+ xorl %edx,%edi+ xorl %r13d,%r14d+ leal (%rbx,%rdi,1),%ebx+ movl %r10d,%r12d+ addl 44+16(%rsi),%eax+ andl %r9d,%r12d+ rorxl $25,%r9d,%r13d+ rorxl $11,%r9d,%edi+ leal (%rbx,%r14,1),%ebx+ leal (%rax,%r12,1),%eax+ andnl %r11d,%r9d,%r12d+ xorl %edi,%r13d+ rorxl $6,%r9d,%r14d+ leal (%rax,%r12,1),%eax+ xorl %r14d,%r13d+ movl %ebx,%edi+ rorxl $22,%ebx,%r12d+ leal (%rax,%r13,1),%eax+ xorl %ecx,%edi+ rorxl $13,%ebx,%r14d+ rorxl $2,%ebx,%r13d+ leal (%r8,%rax,1),%r8d+ andl %edi,%r15d+ xorl %r12d,%r14d+ xorl %ecx,%r15d+ xorl %r13d,%r14d+ leal (%rax,%r15,1),%eax+ movl %r9d,%r12d+ leaq -64(%rsi),%rsi+ cmpq %rsp,%rsi+ jae .Lower_avx2++ movq -64(%rbp),%rdi+ addl %r14d,%eax+ movq -56(%rbp),%rsi+ leaq 448(%rsp),%rsp++ addl 0(%rdi),%eax+ addl 4(%rdi),%ebx+ addl 8(%rdi),%ecx+ addl 12(%rdi),%edx+ addl 16(%rdi),%r8d+ addl 20(%rdi),%r9d+ leaq 128(%rsi),%rsi+ addl 24(%rdi),%r10d+ movq %rsi,%r12+ addl 28(%rdi),%r11d+ cmpq -48(%rbp),%rsi++ movl %eax,0(%rdi)+ cmoveq %rsp,%r12+ movl %ebx,4(%rdi)+ movl %ecx,8(%rdi)+ movl %edx,12(%rdi)+ movl %r8d,16(%rdi)+ movl %r9d,20(%rdi)+ movl %r10d,24(%rdi)+ movl %r11d,28(%rdi)++ jbe .Loop_avx2++.Ldone_avx2:+ vzeroupper+ movaps -160(%rbp),%xmm6+ movaps -144(%rbp),%xmm7+ movaps -128(%rbp),%xmm8+ movaps -112(%rbp),%xmm9+ movq -40(%rbp),%r15+ movq -32(%rbp),%r14+ movq -24(%rbp),%r13+ movq -16(%rbp),%r12+ movq -8(%rbp),%rbx+ movq %rbp,%rsp++ popq %rbp++.LSEH_epilogue_crypton_sha256_asm_block_data_order_avx2:+ mov 8(%rsp),%rdi+ mov 16(%rsp),%rsi++ .byte 0xf3,0xc3++.LSEH_end_crypton_sha256_asm_block_data_order_avx2:+.section .pdata+.p2align 2+.rva .LSEH_begin_crypton_sha256_asm_block_data_order+.rva .LSEH_body_crypton_sha256_asm_block_data_order+.rva .LSEH_info_crypton_sha256_asm_block_data_order_prologue++.rva .LSEH_body_crypton_sha256_asm_block_data_order+.rva .LSEH_epilogue_crypton_sha256_asm_block_data_order+.rva .LSEH_info_crypton_sha256_asm_block_data_order_body++.rva .LSEH_epilogue_crypton_sha256_asm_block_data_order+.rva .LSEH_end_crypton_sha256_asm_block_data_order+.rva .LSEH_info_crypton_sha256_asm_block_data_order_epilogue++.rva .LSEH_begin_crypton_sha256_asm_block_data_order_shaext+.rva .LSEH_body_crypton_sha256_asm_block_data_order_shaext+.rva .LSEH_info_crypton_sha256_asm_block_data_order_shaext_prologue++.rva .LSEH_body_crypton_sha256_asm_block_data_order_shaext+.rva .LSEH_epilogue_crypton_sha256_asm_block_data_order_shaext+.rva .LSEH_info_crypton_sha256_asm_block_data_order_shaext_body++.rva .LSEH_epilogue_crypton_sha256_asm_block_data_order_shaext+.rva .LSEH_end_crypton_sha256_asm_block_data_order_shaext+.rva .LSEH_info_crypton_sha256_asm_block_data_order_shaext_epilogue++.rva .LSEH_begin_crypton_sha256_asm_block_data_order_ssse3+.rva .LSEH_body_crypton_sha256_asm_block_data_order_ssse3+.rva .LSEH_info_crypton_sha256_asm_block_data_order_ssse3_prologue++.rva .LSEH_body_crypton_sha256_asm_block_data_order_ssse3+.rva .LSEH_epilogue_crypton_sha256_asm_block_data_order_ssse3+.rva .LSEH_info_crypton_sha256_asm_block_data_order_ssse3_body++.rva .LSEH_epilogue_crypton_sha256_asm_block_data_order_ssse3+.rva .LSEH_end_crypton_sha256_asm_block_data_order_ssse3+.rva .LSEH_info_crypton_sha256_asm_block_data_order_ssse3_epilogue++.rva .LSEH_begin_crypton_sha256_asm_block_data_order_avx+.rva .LSEH_body_crypton_sha256_asm_block_data_order_avx+.rva .LSEH_info_crypton_sha256_asm_block_data_order_avx_prologue++.rva .LSEH_body_crypton_sha256_asm_block_data_order_avx+.rva .LSEH_epilogue_crypton_sha256_asm_block_data_order_avx+.rva .LSEH_info_crypton_sha256_asm_block_data_order_avx_body++.rva .LSEH_epilogue_crypton_sha256_asm_block_data_order_avx+.rva .LSEH_end_crypton_sha256_asm_block_data_order_avx+.rva .LSEH_info_crypton_sha256_asm_block_data_order_avx_epilogue++.rva .LSEH_begin_crypton_sha256_asm_block_data_order_avx2+.rva .LSEH_body_crypton_sha256_asm_block_data_order_avx2+.rva .LSEH_info_crypton_sha256_asm_block_data_order_avx2_prologue++.rva .LSEH_body_crypton_sha256_asm_block_data_order_avx2+.rva .LSEH_epilogue_crypton_sha256_asm_block_data_order_avx2+.rva .LSEH_info_crypton_sha256_asm_block_data_order_avx2_body++.rva .LSEH_epilogue_crypton_sha256_asm_block_data_order_avx2+.rva .LSEH_end_crypton_sha256_asm_block_data_order_avx2+.rva .LSEH_info_crypton_sha256_asm_block_data_order_avx2_epilogue++.section .xdata+.p2align 3+.LSEH_info_crypton_sha256_asm_block_data_order_prologue:+.byte 1,4,6,0x05+.byte 4,0x74,2,0+.byte 4,0x64,3,0+.byte 4,0x53+.byte 1,0x50+.long 0,0+.LSEH_info_crypton_sha256_asm_block_data_order_body:+.byte 1,0,18,0+.byte 0x00,0xf4,0x0b,0x00+.byte 0x00,0xe4,0x0c,0x00+.byte 0x00,0xd4,0x0d,0x00+.byte 0x00,0xc4,0x0e,0x00+.byte 0x00,0x34,0x0f,0x00+.byte 0x00,0x54,0x10,0x00+.byte 0x00,0x74,0x12,0x00+.byte 0x00,0x64,0x13,0x00+.byte 0x00,0x01,0x11,0x00+.byte 0x00,0x00,0x00,0x00+.byte 0x00,0x00,0x00,0x00+.LSEH_info_crypton_sha256_asm_block_data_order_epilogue:+.byte 1,0,5,11+.byte 0x00,0x74,0x01,0x00+.byte 0x00,0x64,0x02,0x00+.byte 0x00,0xb3+.byte 0x00,0x00,0x00,0x00,0x00,0x00+.byte 0x00,0x00,0x00,0x00++.LSEH_info_crypton_sha256_asm_block_data_order_shaext_prologue:+.byte 1,4,6,0x05+.byte 4,0x74,2,0+.byte 4,0x64,3,0+.byte 4,0x53+.byte 1,0x50+.long 0,0+.LSEH_info_crypton_sha256_asm_block_data_order_shaext_body:+.byte 1,0,17,85+.byte 0x00,0x68,0x00,0x00+.byte 0x00,0x78,0x01,0x00+.byte 0x00,0x88,0x02,0x00+.byte 0x00,0x98,0x03,0x00+.byte 0x00,0xa8,0x04,0x00+.byte 0x00,0x74,0x0c,0x00+.byte 0x00,0x64,0x0d,0x00+.byte 0x00,0x53+.byte 0x00,0x92+.byte 0x00,0x50+.byte 0x00,0x00,0x00,0x00,0x00,0x00+.byte 0x00,0x00,0x00,0x00+.LSEH_info_crypton_sha256_asm_block_data_order_shaext_epilogue:+.byte 1,0,4,0+.byte 0x00,0x74,0x01,0x00+.byte 0x00,0x64,0x02,0x00+.byte 0x00,0x00,0x00,0x00++.LSEH_info_crypton_sha256_asm_block_data_order_ssse3_prologue:+.byte 1,4,6,0x05+.byte 4,0x74,2,0+.byte 4,0x64,3,0+.byte 4,0x53+.byte 1,0x50+.long 0,0+.LSEH_info_crypton_sha256_asm_block_data_order_ssse3_body:+.byte 1,0,25,133+.byte 0x00,0x68,0x00,0x00+.byte 0x00,0x78,0x01,0x00+.byte 0x00,0x88,0x02,0x00+.byte 0x00,0x98,0x03,0x00+.byte 0x00,0xf4,0x0b,0x00+.byte 0x00,0xe4,0x0c,0x00+.byte 0x00,0xd4,0x0d,0x00+.byte 0x00,0xc4,0x0e,0x00+.byte 0x00,0x34,0x0f,0x00+.byte 0x00,0x74,0x12,0x00+.byte 0x00,0x64,0x13,0x00+.byte 0x00,0x53+.byte 0x00,0xf2+.byte 0x00,0x50+.byte 0x00,0x00,0x00,0x00,0x00,0x00+.byte 0x00,0x00,0x00,0x00+.LSEH_info_crypton_sha256_asm_block_data_order_ssse3_epilogue:+.byte 1,0,4,0+.byte 0x00,0x74,0x01,0x00+.byte 0x00,0x64,0x02,0x00+.byte 0x00,0x00,0x00,0x00++.LSEH_info_crypton_sha256_asm_block_data_order_avx_prologue:+.byte 1,4,6,0x05+.byte 4,0x74,2,0+.byte 4,0x64,3,0+.byte 4,0x53+.byte 1,0x50+.long 0,0+.LSEH_info_crypton_sha256_asm_block_data_order_avx_body:+.byte 1,0,26,165+.byte 0x00,0x68,0x00,0x00+.byte 0x00,0x78,0x01,0x00+.byte 0x00,0x88,0x02,0x00+.byte 0x00,0x98,0x03,0x00+.byte 0x00,0xf4,0x0f,0x00+.byte 0x00,0xe4,0x10,0x00+.byte 0x00,0xd4,0x11,0x00+.byte 0x00,0xc4,0x12,0x00+.byte 0x00,0x34,0x13,0x00+.byte 0x00,0x74,0x16,0x00+.byte 0x00,0x64,0x17,0x00+.byte 0x00,0x53+.byte 0x00,0x01,0x14,0x00+.byte 0x00,0x50+.byte 0x00,0x00,0x00,0x00+.byte 0x00,0x00,0x00,0x00+.LSEH_info_crypton_sha256_asm_block_data_order_avx_epilogue:+.byte 1,0,4,0+.byte 0x00,0x74,0x01,0x00+.byte 0x00,0x64,0x02,0x00+.byte 0x00,0x00,0x00,0x00++.LSEH_info_crypton_sha256_asm_block_data_order_avx2_prologue:+.byte 1,4,6,0x05+.byte 4,0x74,2,0+.byte 4,0x64,3,0+.byte 4,0x53+.byte 1,0x50+.long 0,0+.LSEH_info_crypton_sha256_asm_block_data_order_avx2_body:+.byte 1,0,26,165+.byte 0x00,0x68,0x00,0x00+.byte 0x00,0x78,0x01,0x00+.byte 0x00,0x88,0x02,0x00+.byte 0x00,0x98,0x03,0x00+.byte 0x00,0xf4,0x0f,0x00+.byte 0x00,0xe4,0x10,0x00+.byte 0x00,0xd4,0x11,0x00+.byte 0x00,0xc4,0x12,0x00+.byte 0x00,0x34,0x13,0x00+.byte 0x00,0x74,0x16,0x00+.byte 0x00,0x64,0x17,0x00+.byte 0x00,0x53+.byte 0x00,0x01,0x14,0x00+.byte 0x00,0x50+.byte 0x00,0x00,0x00,0x00+.byte 0x00,0x00,0x00,0x00+.LSEH_info_crypton_sha256_asm_block_data_order_avx2_epilogue:+.byte 1,0,4,0+.byte 0x00,0x74,0x01,0x00+.byte 0x00,0x64,0x02,0x00+.byte 0x00,0x00,0x00,0x00+
@@ -0,0 +1,892 @@+#!/usr/bin/env perl+# SPDX-License-Identifier: GPL-1.0+ OR BSD-3-Clause+#+# ====================================================================+# Written by Andy Polyakov, @dot-asm, initially for the OpenSSL+# project.+# ====================================================================+#+# SHA256/512 for ARMv8.+#+# Performance in cycles per processed byte and improvement coefficient+# over code generated with "default" compiler:+#+# SHA256-hw SHA256(*) SHA512+# Apple A7 1.97 10.5 (+33%) 6.73 (-1%(**))+# Apple A10 1.30 5.81+# Apple A12 1.31 5.06+# Apple A14/M1 1.30 8.19 (+14%) 2.24 (hw)+# Cortex-A53 2.38 15.5 (+115%) 10.0 (+150%(***))+# Cortex-A57 2.31 11.6 (+86%) 7.51 (+260%(***))+# Cortex-A76 1.60 9.5 6.05+# Cortex-X2 1.60 7.3 2.60 (hw)+# Cortex-X925 1.57 5.97 2.55 (hw)+# Denver 2.01 10.5 (+26%) 6.70 (+8%)+# X-Gene 20.0 (+100%) 12.8 (+300%(***))+# Mongoose 2.36 13.0 (+50%) 8.36 (+33%)+# Kryo 1.92 17.4 (+30%) 11.2 (+8%)+# ThunderX2 2.54 13.2 (+40%) 8.40 (+18%)+# Shapdragon X 1.40 7.43 2.23 (hw)+#+# (*) Software SHA256 results are of lesser relevance, presented+# mostly for informational purposes.+# (**) The result is a trade-off: it's possible to improve it by+# 10% (or by 1 cycle per round), but at the cost of 20% loss+# on Cortex-A53 (or by 4 cycles per round).+# (***) Super-impressive coefficients over gcc-generated code are+# indication of some compiler "pathology", most notably code+# generated with -mgeneral-regs-only is significantly faster+# and the gap is only 40-90%.+#+# October 2016.+#+# Originally it was reckoned that it makes no sense to implement NEON+# version of SHA256 for 64-bit processors. This is because performance+# improvement on most wide-spread Cortex-A5x processors was observed+# to be marginal, same on Cortex-A53 and ~10% on A57. But then it was+# observed that 32-bit NEON SHA256 performs significantly better than+# 64-bit scalar version on *some* of the more recent processors. As+# result 64-bit NEON version of SHA256 was added to provide best+# all-round performance. For example it executes ~30% faster on X-Gene+# and Mongoose. [For reference, NEON version of SHA512 is bound to+# deliver much less improvement, likely *negative* on Cortex-A5x.+# Which is why NEON support is limited to SHA256.]++$flavour = shift;+$output = shift;++if ($flavour && $flavour ne "void") {+ $0 =~ m/(.*[\/\\])[^\/\\]+$/; $dir=$1;+ ( $xlate="${dir}arm-xlate.pl" and -f $xlate ) or+ ( $xlate="${dir}../../perlasm/arm-xlate.pl" and -f $xlate) or+ die "can't locate arm-xlate.pl";++ open STDOUT,"| \"$^X\" $xlate $flavour $output";+} else {+ open STDOUT,">$output";+}++if ($output =~ /512/) {+ $BITS=512;+ $SZ=8;+ @Sigma0=(28,34,39);+ @Sigma1=(14,18,41);+ @sigma0=(1, 8, 7);+ @sigma1=(19,61, 6);+ $rounds=80;+ $reg_t="x";+} else {+ $BITS=256;+ $SZ=4;+ @Sigma0=( 2,13,22);+ @Sigma1=( 6,11,25);+ @sigma0=( 7,18, 3);+ @sigma1=(17,19,10);+ $rounds=64;+ $reg_t="w";+}++$func="sha${BITS}_block_data_order";++($ctx,$inp,$num,$Ktbl)=map("x$_",(0..2,30));++@X=map("$reg_t$_",(3..15,0..2));+@V=($A,$B,$C,$D,$E,$F,$G,$H)=map("$reg_t$_",(20..27));+($t0,$t1,$t2,$t3)=map("$reg_t$_",(16,17,19,28));++sub BODY_00_xx {+my ($i,$a,$b,$c,$d,$e,$f,$g,$h)=@_;+my $j=($i+1)&15;+my ($T0,$T1,$T2)=(@X[($i-8)&15],@X[($i-9)&15],@X[($i-10)&15]);+ $T0=@X[$i+3] if ($i<11);++$code.=<<___ if ($i<16);+#ifndef __AARCH64EB__+ rev @X[$i],@X[$i] // $i+#endif+___+$code.=<<___ if ($i<13 && ($i&1));+ ldp @X[$i+1],@X[$i+2],[$inp],#2*$SZ+___+$code.=<<___ if ($i==13);+ ldp @X[14],@X[15],[$inp]+___+$code.=<<___ if ($i>=14);+ ldr @X[($i-11)&15],[sp,#`$SZ*(($i-11)%4)`]+___+$code.=<<___ if ($i>0 && $i<16);+ add $a,$a,$t1 // h+=Sigma0(a)+___+$code.=<<___ if ($i>=11);+ str @X[($i-8)&15],[sp,#`$SZ*(($i-8)%4)`]+___+# While ARMv8 specifies merged rotate-n-logical operation such as+# 'eor x,y,z,ror#n', it was found to negatively affect performance+# on Apple A7. The reason seems to be that it requires even 'y' to+# be available earlier. This means that such merged instruction is+# not necessarily best choice on critical path... On the other hand+# Cortex-A5x handles merged instructions much better than disjoint+# rotate and logical... See (**) footnote above.+$code.=<<___ if ($i<15);+ ror $t0,$e,#$Sigma1[0]+ add $h,$h,$t2 // h+=K[i]+ eor $T0,$e,$e,ror#`$Sigma1[2]-$Sigma1[1]`+ and $t1,$f,$e+ bic $t2,$g,$e+ add $h,$h,@X[$i&15] // h+=X[i]+ orr $t1,$t1,$t2 // Ch(e,f,g)+ eor $t2,$a,$b // a^b, b^c in next round+ eor $t0,$t0,$T0,ror#$Sigma1[1] // Sigma1(e)+ ror $T0,$a,#$Sigma0[0]+ add $h,$h,$t1 // h+=Ch(e,f,g)+ eor $t1,$a,$a,ror#`$Sigma0[2]-$Sigma0[1]`+ add $h,$h,$t0 // h+=Sigma1(e)+ and $t3,$t3,$t2 // (b^c)&=(a^b)+ add $d,$d,$h // d+=h+ eor $t3,$t3,$b // Maj(a,b,c)+ eor $t1,$T0,$t1,ror#$Sigma0[1] // Sigma0(a)+ add $h,$h,$t3 // h+=Maj(a,b,c)+ ldr $t3,[$Ktbl],#$SZ // *K++, $t2 in next round+ //add $h,$h,$t1 // h+=Sigma0(a)+___+$code.=<<___ if ($i>=15);+ ror $t0,$e,#$Sigma1[0]+ add $h,$h,$t2 // h+=K[i]+ ror $T1,@X[($j+1)&15],#$sigma0[0]+ and $t1,$f,$e+ ror $T2,@X[($j+14)&15],#$sigma1[0]+ bic $t2,$g,$e+ ror $T0,$a,#$Sigma0[0]+ add $h,$h,@X[$i&15] // h+=X[i]+ eor $t0,$t0,$e,ror#$Sigma1[1]+ eor $T1,$T1,@X[($j+1)&15],ror#$sigma0[1]+ orr $t1,$t1,$t2 // Ch(e,f,g)+ eor $t2,$a,$b // a^b, b^c in next round+ eor $t0,$t0,$e,ror#$Sigma1[2] // Sigma1(e)+ eor $T0,$T0,$a,ror#$Sigma0[1]+ add $h,$h,$t1 // h+=Ch(e,f,g)+ and $t3,$t3,$t2 // (b^c)&=(a^b)+ eor $T2,$T2,@X[($j+14)&15],ror#$sigma1[1]+ eor $T1,$T1,@X[($j+1)&15],lsr#$sigma0[2] // sigma0(X[i+1])+ add $h,$h,$t0 // h+=Sigma1(e)+ eor $t3,$t3,$b // Maj(a,b,c)+ eor $t1,$T0,$a,ror#$Sigma0[2] // Sigma0(a)+ eor $T2,$T2,@X[($j+14)&15],lsr#$sigma1[2] // sigma1(X[i+14])+ add @X[$j],@X[$j],@X[($j+9)&15]+ add $d,$d,$h // d+=h+ add $h,$h,$t3 // h+=Maj(a,b,c)+ ldr $t3,[$Ktbl],#$SZ // *K++, $t2 in next round+ add @X[$j],@X[$j],$T1+ add $h,$h,$t1 // h+=Sigma0(a)+ add @X[$j],@X[$j],$T2+___+ ($t2,$t3)=($t3,$t2);+}++$code.=<<___;+#ifndef __KERNEL__+# include "arm_arch.h"+.extern OPENSSL_armcap_P+#endif++.text++.globl $func+.type $func,%function+.align 6+$func:+#ifndef __KERNEL__+ adrp c16,OPENSSL_armcap_P+ ldr w16,[c16,#:lo12:OPENSSL_armcap_P]+___+$code.=<<___ if ($SZ==4);+ tst w16,#ARMV8_SHA256+ b.ne .Lv8_entry+ tst w16,#ARMV7_NEON+ b.ne .Lneon_entry+___+$code.=<<___ if ($SZ==8);+ tst w16,#ARMV8_SHA512+ b.ne .Lv8_entry+___+$code.=<<___;+#endif+ .inst 0xd503233f // paciasp+ stp c29,c30,[csp,#-16*__SIZEOF_POINTER__]!+ add c29,csp,#0++ stp c19,c20,[csp,#2*__SIZEOF_POINTER__]+ stp c21,c22,[csp,#4*__SIZEOF_POINTER__]+ stp c23,c24,[csp,#6*__SIZEOF_POINTER__]+ stp c25,c26,[csp,#8*__SIZEOF_POINTER__]+ stp c27,c28,[csp,#10*__SIZEOF_POINTER__]+ sub csp,csp,#4*$SZ++ ldp $A,$B,[$ctx] // load context+ ldp $C,$D,[$ctx,#2*$SZ]+ lsl $num,$num,#`log(16*$SZ)/log(2)`+ ldp $E,$F,[$ctx,#4*$SZ]+ cadd $num,$inp,$num // end of input+ ldp $G,$H,[$ctx,#6*$SZ]+ adr $Ktbl,.LK$BITS+ stp c#$ctx,c#$num,[c29,#12*__SIZEOF_POINTER__]++.Loop:+ ldp @X[0],@X[1],[$inp],#2*$SZ+ ldr $t2,[$Ktbl],#$SZ // *K+++ eor $t3,$B,$C // magic seed+ str c#$inp,[c29,#14*__SIZEOF_POINTER__]+___+for ($i=0;$i<16;$i++) { &BODY_00_xx($i,@V); unshift(@V,pop(@V)); }+$code.=".Loop_16_xx:\n";+for (;$i<32;$i++) { &BODY_00_xx($i,@V); unshift(@V,pop(@V)); }+$code.=<<___;+ cbnz $t2,.Loop_16_xx++ ldp c#$ctx,c#$num,[c29,#12*__SIZEOF_POINTER__]+ ldr c#$inp,[c29,#14*__SIZEOF_POINTER__]+ csub $Ktbl,$Ktbl,#`$SZ*($rounds+1)` // rewind++ ldp @X[0],@X[1],[$ctx]+ ldp @X[2],@X[3],[$ctx,#2*$SZ]+ cadd $inp,$inp,#14*$SZ // advance input pointer+ ldp @X[4],@X[5],[$ctx,#4*$SZ]+ add $A,$A,@X[0]+ ldp @X[6],@X[7],[$ctx,#6*$SZ]+ add $B,$B,@X[1]+ add $C,$C,@X[2]+ add $D,$D,@X[3]+ stp $A,$B,[$ctx]+ add $E,$E,@X[4]+ add $F,$F,@X[5]+ stp $C,$D,[$ctx,#2*$SZ]+ add $G,$G,@X[6]+ add $H,$H,@X[7]+ cmp $inp,$num+ stp $E,$F,[$ctx,#4*$SZ]+ stp $G,$H,[$ctx,#6*$SZ]+ b.ne .Loop++ ldp c19,c20,[c29,#2*__SIZEOF_POINTER__]+ add csp,csp,#4*$SZ+ ldp c21,c22,[c29,#4*__SIZEOF_POINTER__]+ ldp c23,c24,[c29,#6*__SIZEOF_POINTER__]+ ldp c25,c26,[c29,#8*__SIZEOF_POINTER__]+ ldp c27,c28,[c29,#10*__SIZEOF_POINTER__]+ ldp c29,c30,[csp],#16*__SIZEOF_POINTER__+ .inst 0xd50323bf // autiasp+ ret+.size $func,.-$func++.align 6+.type .LK$BITS,%object+.LK$BITS:+___+$code.=<<___ if ($SZ==8);+ .quad 0x428a2f98d728ae22,0x7137449123ef65cd+ .quad 0xb5c0fbcfec4d3b2f,0xe9b5dba58189dbbc+ .quad 0x3956c25bf348b538,0x59f111f1b605d019+ .quad 0x923f82a4af194f9b,0xab1c5ed5da6d8118+ .quad 0xd807aa98a3030242,0x12835b0145706fbe+ .quad 0x243185be4ee4b28c,0x550c7dc3d5ffb4e2+ .quad 0x72be5d74f27b896f,0x80deb1fe3b1696b1+ .quad 0x9bdc06a725c71235,0xc19bf174cf692694+ .quad 0xe49b69c19ef14ad2,0xefbe4786384f25e3+ .quad 0x0fc19dc68b8cd5b5,0x240ca1cc77ac9c65+ .quad 0x2de92c6f592b0275,0x4a7484aa6ea6e483+ .quad 0x5cb0a9dcbd41fbd4,0x76f988da831153b5+ .quad 0x983e5152ee66dfab,0xa831c66d2db43210+ .quad 0xb00327c898fb213f,0xbf597fc7beef0ee4+ .quad 0xc6e00bf33da88fc2,0xd5a79147930aa725+ .quad 0x06ca6351e003826f,0x142929670a0e6e70+ .quad 0x27b70a8546d22ffc,0x2e1b21385c26c926+ .quad 0x4d2c6dfc5ac42aed,0x53380d139d95b3df+ .quad 0x650a73548baf63de,0x766a0abb3c77b2a8+ .quad 0x81c2c92e47edaee6,0x92722c851482353b+ .quad 0xa2bfe8a14cf10364,0xa81a664bbc423001+ .quad 0xc24b8b70d0f89791,0xc76c51a30654be30+ .quad 0xd192e819d6ef5218,0xd69906245565a910+ .quad 0xf40e35855771202a,0x106aa07032bbd1b8+ .quad 0x19a4c116b8d2d0c8,0x1e376c085141ab53+ .quad 0x2748774cdf8eeb99,0x34b0bcb5e19b48a8+ .quad 0x391c0cb3c5c95a63,0x4ed8aa4ae3418acb+ .quad 0x5b9cca4f7763e373,0x682e6ff3d6b2b8a3+ .quad 0x748f82ee5defb2fc,0x78a5636f43172f60+ .quad 0x84c87814a1f0ab72,0x8cc702081a6439ec+ .quad 0x90befffa23631e28,0xa4506cebde82bde9+ .quad 0xbef9a3f7b2c67915,0xc67178f2e372532b+ .quad 0xca273eceea26619c,0xd186b8c721c0c207+ .quad 0xeada7dd6cde0eb1e,0xf57d4f7fee6ed178+ .quad 0x06f067aa72176fba,0x0a637dc5a2c898a6+ .quad 0x113f9804bef90dae,0x1b710b35131c471b+ .quad 0x28db77f523047d84,0x32caab7b40c72493+ .quad 0x3c9ebe0a15c9bebc,0x431d67c49c100d4c+ .quad 0x4cc5d4becb3e42b6,0x597f299cfc657e2a+ .quad 0x5fcb6fab3ad6faec,0x6c44198c4a475817+ .quad 0 // terminator+___+$code.=<<___ if ($SZ==4);+ .long 0x428a2f98,0x71374491,0xb5c0fbcf,0xe9b5dba5+ .long 0x3956c25b,0x59f111f1,0x923f82a4,0xab1c5ed5+ .long 0xd807aa98,0x12835b01,0x243185be,0x550c7dc3+ .long 0x72be5d74,0x80deb1fe,0x9bdc06a7,0xc19bf174+ .long 0xe49b69c1,0xefbe4786,0x0fc19dc6,0x240ca1cc+ .long 0x2de92c6f,0x4a7484aa,0x5cb0a9dc,0x76f988da+ .long 0x983e5152,0xa831c66d,0xb00327c8,0xbf597fc7+ .long 0xc6e00bf3,0xd5a79147,0x06ca6351,0x14292967+ .long 0x27b70a85,0x2e1b2138,0x4d2c6dfc,0x53380d13+ .long 0x650a7354,0x766a0abb,0x81c2c92e,0x92722c85+ .long 0xa2bfe8a1,0xa81a664b,0xc24b8b70,0xc76c51a3+ .long 0xd192e819,0xd6990624,0xf40e3585,0x106aa070+ .long 0x19a4c116,0x1e376c08,0x2748774c,0x34b0bcb5+ .long 0x391c0cb3,0x4ed8aa4a,0x5b9cca4f,0x682e6ff3+ .long 0x748f82ee,0x78a5636f,0x84c87814,0x8cc70208+ .long 0x90befffa,0xa4506ceb,0xbef9a3f7,0xc67178f2+ .long 0 //terminator+___+$code.=<<___;+.size .LK$BITS,.-.LK$BITS+.asciz "SHA$BITS block transform for ARMv8, CRYPTOGAMS by \@dot-asm"+.align 2+___++if ($SZ==4) {+my $Ktbl="x3";++my ($ABCD,$EFGH,$abcd)=map("v$_.16b",(0..2));+my @MSG=map("v$_.16b",(4..7));+my ($W0,$W1)=("v16.4s","v17.4s");+my ($ABCD_SAVE,$EFGH_SAVE)=("v18.16b","v19.16b");++$code.=<<___;+#ifndef __KERNEL__+.type sha256_block_armv8,%function+.align 6+sha256_block_armv8:+.Lv8_entry:+ stp c29,c30,[csp,#-2*__SIZEOF_POINTER__]!+ add c29,csp,#0++ ld1.32 {$ABCD,$EFGH},[$ctx]+ adr $Ktbl,.LK256++.Loop_hw:+ ld1 {@MSG[0]-@MSG[3]},[$inp],#64+ sub $num,$num,#1+ ld1.32 {$W0},[$Ktbl],#16+ rev32 @MSG[0],@MSG[0]+ rev32 @MSG[1],@MSG[1]+ rev32 @MSG[2],@MSG[2]+ rev32 @MSG[3],@MSG[3]+ orr $ABCD_SAVE,$ABCD,$ABCD // offload+ orr $EFGH_SAVE,$EFGH,$EFGH+___+for($i=0;$i<12;$i++) {+$code.=<<___;+ ld1.32 {$W1},[$Ktbl],#16+ add.i32 $W0,$W0,@MSG[0]+ sha256su0 @MSG[0],@MSG[1]+ orr $abcd,$ABCD,$ABCD+ sha256h $ABCD,$EFGH,$W0+ sha256h2 $EFGH,$abcd,$W0+ sha256su1 @MSG[0],@MSG[2],@MSG[3]+___+ ($W0,$W1)=($W1,$W0); push(@MSG,shift(@MSG));+}+$code.=<<___;+ ld1.32 {$W1},[$Ktbl],#16+ add.i32 $W0,$W0,@MSG[0]+ orr $abcd,$ABCD,$ABCD+ sha256h $ABCD,$EFGH,$W0+ sha256h2 $EFGH,$abcd,$W0++ ld1.32 {$W0},[$Ktbl],#16+ add.i32 $W1,$W1,@MSG[1]+ orr $abcd,$ABCD,$ABCD+ sha256h $ABCD,$EFGH,$W1+ sha256h2 $EFGH,$abcd,$W1++ ld1.32 {$W1},[$Ktbl]+ add.i32 $W0,$W0,@MSG[2]+ csub $Ktbl,$Ktbl,#$rounds*$SZ-16 // rewind+ orr $abcd,$ABCD,$ABCD+ sha256h $ABCD,$EFGH,$W0+ sha256h2 $EFGH,$abcd,$W0++ add.i32 $W1,$W1,@MSG[3]+ orr $abcd,$ABCD,$ABCD+ sha256h $ABCD,$EFGH,$W1+ sha256h2 $EFGH,$abcd,$W1++ add.i32 $ABCD,$ABCD,$ABCD_SAVE+ add.i32 $EFGH,$EFGH,$EFGH_SAVE++ cbnz $num,.Loop_hw++ st1.32 {$ABCD,$EFGH},[$ctx]++ ldr c29,[csp],#2*__SIZEOF_POINTER__+ ret+.size sha256_block_armv8,.-sha256_block_armv8+#endif+___+}++if ($SZ==4) { ######################################### NEON stuff #+# You'll surely note a lot of similarities with sha256-armv4 module,+# and of course it's not a coincidence. sha256-armv4 was used as+# initial template, but was adapted for ARMv8 instruction set and+# extensively re-tuned for all-round performance.++my @V = ($A,$B,$C,$D,$E,$F,$G,$H) = map("w$_",(3..10));+my ($t0,$t1,$t2,$t3,$t4) = map("w$_",(11..15));+my $Ktbl="x16";+my $Xfer="x17";+my @X = map("q$_",(0..3));+my ($T0,$T1,$T2,$T3,$T4,$T5,$T6,$T7) = map("q$_",(4..7,16..19));+my $j=0;++sub AUTOLOAD() # thunk [simplified] x86-style perlasm+{ my $opcode = $AUTOLOAD; $opcode =~ s/.*:://; $opcode =~ s/_/\./;+ my $arg = pop;+ $arg = "#$arg" if ($arg*1 eq $arg);+ $code .= "\t$opcode\t".join(',',@_,$arg)."\n";+}++sub Dscalar { shift =~ m|[qv]([0-9]+)|?"d$1":""; }+sub Dlo { shift =~ m|[qv]([0-9]+)|?"v$1.d[0]":""; }+sub Dhi { shift =~ m|[qv]([0-9]+)|?"v$1.d[1]":""; }++sub Xupdate()+{ use integer;+ my $body = shift;+ my @insns = (&$body,&$body,&$body,&$body);+ my ($a,$b,$c,$d,$e,$f,$g,$h);++ &ext_8 ($T0,@X[0],@X[1],4); # X[1..4]+ eval(shift(@insns));+ eval(shift(@insns));+ eval(shift(@insns));+ &ext_8 ($T3,@X[2],@X[3],4); # X[9..12]+ eval(shift(@insns));+ eval(shift(@insns));+ &mov (&Dscalar($T7),&Dhi(@X[3])); # X[14..15]+ eval(shift(@insns));+ eval(shift(@insns));+ &ushr_32 ($T2,$T0,$sigma0[0]);+ eval(shift(@insns));+ &ushr_32 ($T1,$T0,$sigma0[2]);+ eval(shift(@insns));+ &add_32 (@X[0],@X[0],$T3); # X[0..3] += X[9..12]+ eval(shift(@insns));+ &sli_32 ($T2,$T0,32-$sigma0[0]);+ eval(shift(@insns));+ eval(shift(@insns));+ &ushr_32 ($T3,$T0,$sigma0[1]);+ eval(shift(@insns));+ eval(shift(@insns));+ &eor_8 ($T1,$T1,$T2);+ eval(shift(@insns));+ eval(shift(@insns));+ &sli_32 ($T3,$T0,32-$sigma0[1]);+ eval(shift(@insns));+ eval(shift(@insns));+ &ushr_32 ($T4,$T7,$sigma1[0]);+ eval(shift(@insns));+ eval(shift(@insns));+ &eor_8 ($T1,$T1,$T3); # sigma0(X[1..4])+ eval(shift(@insns));+ eval(shift(@insns));+ &sli_32 ($T4,$T7,32-$sigma1[0]);+ eval(shift(@insns));+ eval(shift(@insns));+ &ushr_32 ($T5,$T7,$sigma1[2]);+ eval(shift(@insns));+ eval(shift(@insns));+ &ushr_32 ($T3,$T7,$sigma1[1]);+ eval(shift(@insns));+ eval(shift(@insns));+ &add_32 (@X[0],@X[0],$T1); # X[0..3] += sigma0(X[1..4])+ eval(shift(@insns));+ eval(shift(@insns));+ &sli_u32 ($T3,$T7,32-$sigma1[1]);+ eval(shift(@insns));+ eval(shift(@insns));+ &eor_8 ($T5,$T5,$T4);+ eval(shift(@insns));+ eval(shift(@insns));+ eval(shift(@insns));+ &eor_8 ($T5,$T5,$T3); # sigma1(X[14..15])+ eval(shift(@insns));+ eval(shift(@insns));+ eval(shift(@insns));+ &add_32 (@X[0],@X[0],$T5); # X[0..1] += sigma1(X[14..15])+ eval(shift(@insns));+ eval(shift(@insns));+ eval(shift(@insns));+ &ushr_32 ($T6,@X[0],$sigma1[0]);+ eval(shift(@insns));+ &ushr_32 ($T7,@X[0],$sigma1[2]);+ eval(shift(@insns));+ eval(shift(@insns));+ &sli_32 ($T6,@X[0],32-$sigma1[0]);+ eval(shift(@insns));+ &ushr_32 ($T5,@X[0],$sigma1[1]);+ eval(shift(@insns));+ eval(shift(@insns));+ &eor_8 ($T7,$T7,$T6);+ eval(shift(@insns));+ eval(shift(@insns));+ &sli_32 ($T5,@X[0],32-$sigma1[1]);+ eval(shift(@insns));+ eval(shift(@insns));+ &ld1_32 ("{$T0}","[$Ktbl], #16");+ eval(shift(@insns));+ &eor_8 ($T7,$T7,$T5); # sigma1(X[16..17])+ eval(shift(@insns));+ eval(shift(@insns));+ &eor_8 ($T5,$T5,$T5);+ eval(shift(@insns));+ eval(shift(@insns));+ &mov (&Dhi($T5), &Dlo($T7));+ eval(shift(@insns));+ eval(shift(@insns));+ eval(shift(@insns));+ &add_32 (@X[0],@X[0],$T5); # X[2..3] += sigma1(X[16..17])+ eval(shift(@insns));+ eval(shift(@insns));+ eval(shift(@insns));+ &add_32 ($T0,$T0,@X[0]);+ while($#insns>=1) { eval(shift(@insns)); }+ &st1_32 ("{$T0}","[$Xfer], #16");+ eval(shift(@insns));++ push(@X,shift(@X)); # "rotate" X[]+}++sub Xpreload()+{ use integer;+ my $body = shift;+ my @insns = (&$body,&$body,&$body,&$body);+ my ($a,$b,$c,$d,$e,$f,$g,$h);++ eval(shift(@insns));+ eval(shift(@insns));+ &ld1_8 ("{@X[0]}","[$inp],#16");+ eval(shift(@insns));+ eval(shift(@insns));+ &ld1_32 ("{$T0}","[$Ktbl],#16");+ eval(shift(@insns));+ eval(shift(@insns));+ eval(shift(@insns));+ eval(shift(@insns));+ &rev32 (@X[0],@X[0]);+ eval(shift(@insns));+ eval(shift(@insns));+ eval(shift(@insns));+ eval(shift(@insns));+ &add_32 ($T0,$T0,@X[0]);+ foreach (@insns) { eval; } # remaining instructions+ &st1_32 ("{$T0}","[$Xfer], #16");++ push(@X,shift(@X)); # "rotate" X[]+}++sub body_00_15 () {+ (+ '($a,$b,$c,$d,$e,$f,$g,$h)=@V;'.+ '&add ($h,$h,$t1)', # h+=X[i]+K[i]+ '&add ($a,$a,$t4);'. # h+=Sigma0(a) from the past+ '&and ($t1,$f,$e)',+ '&bic ($t4,$g,$e)',+ '&eor ($t0,$e,$e,"ror#".($Sigma1[1]-$Sigma1[0]))',+ '&add ($a,$a,$t2)', # h+=Maj(a,b,c) from the past+ '&orr ($t1,$t1,$t4)', # Ch(e,f,g)+ '&eor ($t0,$t0,$e,"ror#".($Sigma1[2]-$Sigma1[0]))', # Sigma1(e)+ '&eor ($t4,$a,$a,"ror#".($Sigma0[1]-$Sigma0[0]))',+ '&add ($h,$h,$t1)', # h+=Ch(e,f,g)+ '&ror ($t0,$t0,"#$Sigma1[0]")',+ '&eor ($t2,$a,$b)', # a^b, b^c in next round+ '&eor ($t4,$t4,$a,"ror#".($Sigma0[2]-$Sigma0[0]))', # Sigma0(a)+ '&add ($h,$h,$t0)', # h+=Sigma1(e)+ '&ldr ($t1,sprintf "[sp,#%d]",4*(($j+1)&15)) if (($j&15)!=15);'.+ '&ldr ($t1,"[$Ktbl]") if ($j==15);'.+ '&and ($t3,$t3,$t2)', # (b^c)&=(a^b)+ '&ror ($t4,$t4,"#$Sigma0[0]")',+ '&add ($d,$d,$h)', # d+=h+ '&eor ($t3,$t3,$b)', # Maj(a,b,c)+ '$j++; unshift(@V,pop(@V)); ($t2,$t3)=($t3,$t2);'+ )+}++$code.=<<___;+#ifdef __KERNEL__+.globl sha256_block_neon+#endif+.type sha256_block_neon,%function+.align 4+sha256_block_neon:+.Lneon_entry:+ stp c29, c30, [csp, #-2*__SIZEOF_POINTER__]!+ mov c29, csp+ sub csp,csp,#16*4++ adr $Ktbl,.LK256+ add $num,$inp,$num,lsl#6 // len to point at the end of inp++ ld1.8 {@X[0]},[$inp], #16+ ld1.8 {@X[1]},[$inp], #16+ ld1.8 {@X[2]},[$inp], #16+ ld1.8 {@X[3]},[$inp], #16+ ld1.32 {$T0},[$Ktbl], #16+ ld1.32 {$T1},[$Ktbl], #16+ ld1.32 {$T2},[$Ktbl], #16+ ld1.32 {$T3},[$Ktbl], #16+ rev32 @X[0],@X[0] // yes, even on+ rev32 @X[1],@X[1] // big-endian+ rev32 @X[2],@X[2]+ rev32 @X[3],@X[3]+ cmov $Xfer,sp+ add.32 $T0,$T0,@X[0]+ add.32 $T1,$T1,@X[1]+ add.32 $T2,$T2,@X[2]+ st1.32 {$T0-$T1},[$Xfer], #32+ add.32 $T3,$T3,@X[3]+ st1.32 {$T2-$T3},[$Xfer]+ csub $Xfer,$Xfer,#32++ ldp $A,$B,[$ctx]+ ldp $C,$D,[$ctx,#8]+ ldp $E,$F,[$ctx,#16]+ ldp $G,$H,[$ctx,#24]+ ldr $t1,[sp,#0]+ mov $t2,wzr+ eor $t3,$B,$C+ mov $t4,wzr+ b .L_00_48++.align 4+.L_00_48:+___+ &Xupdate(\&body_00_15);+ &Xupdate(\&body_00_15);+ &Xupdate(\&body_00_15);+ &Xupdate(\&body_00_15);+$code.=<<___;+ cmp $t1,#0 // check for K256 terminator+ ldr $t1,[sp,#0]+ csub $Xfer,$Xfer,#64+ bne .L_00_48++ csub $Ktbl,$Ktbl,#256 // rewind $Ktbl+ cmp $inp,$num+ mov $Xfer, #-64+ csel $Xfer, $Xfer, xzr, eq+ cadd $inp,$inp,$Xfer // avoid SEGV+ cmov $Xfer,sp+___+ &Xpreload(\&body_00_15);+ &Xpreload(\&body_00_15);+ &Xpreload(\&body_00_15);+ &Xpreload(\&body_00_15);+$code.=<<___;+ add $A,$A,$t4 // h+=Sigma0(a) from the past+ ldp $t0,$t1,[$ctx,#0]+ add $A,$A,$t2 // h+=Maj(a,b,c) from the past+ ldp $t2,$t3,[$ctx,#8]+ add $A,$A,$t0 // accumulate+ add $B,$B,$t1+ ldp $t0,$t1,[$ctx,#16]+ add $C,$C,$t2+ add $D,$D,$t3+ ldp $t2,$t3,[$ctx,#24]+ add $E,$E,$t0+ add $F,$F,$t1+ ldr $t1,[sp,#0]+ stp $A,$B,[$ctx,#0]+ add $G,$G,$t2+ mov $t2,wzr+ stp $C,$D,[$ctx,#8]+ add $H,$H,$t3+ stp $E,$F,[$ctx,#16]+ eor $t3,$B,$C+ stp $G,$H,[$ctx,#24]+ mov $t4,wzr+ cmov $Xfer,sp+ b.ne .L_00_48++ ldr c29,[c29]+ add csp,csp,#16*4+2*__SIZEOF_POINTER__+ ret+.size sha256_block_neon,.-sha256_block_neon+___+}++if ($SZ==8) {+my $Ktbl="x3";++my @H = map("v$_.16b",(0..4));+my ($fg,$de,$m9_10)=map("v$_.16b",(5..7));+my @MSG=map("v$_.16b",(16..23));+my ($W0,$W1)=("v24.2d","v25.2d");+my ($AB,$CD,$EF,$GH)=map("v$_.16b",(26..29));++$code.=<<___;+#ifndef __KERNEL__+.type sha512_block_armv8,%function+.align 6+sha512_block_armv8:+.Lv8_entry:+ stp c29,c30,[csp,#-2*__SIZEOF_POINTER__]!+ add c29,csp,#0++ ld1 {@MSG[0]-@MSG[3]},[$inp],#64 // load input+ ld1 {@MSG[4]-@MSG[7]},[$inp],#64++ ld1.64 {@H[0]-@H[3]},[$ctx] // load context+ adr $Ktbl,.LK512++ rev64 @MSG[0],@MSG[0]+ rev64 @MSG[1],@MSG[1]+ rev64 @MSG[2],@MSG[2]+ rev64 @MSG[3],@MSG[3]+ rev64 @MSG[4],@MSG[4]+ rev64 @MSG[5],@MSG[5]+ rev64 @MSG[6],@MSG[6]+ rev64 @MSG[7],@MSG[7]+ b .Loop_hw++.align 4+.Loop_hw:+ ld1.64 {$W0},[$Ktbl],#16+ subs $num,$num,#1+ sub c4,c#$inp,#128+ orr $AB,@H[0],@H[0] // offload+ orr $CD,@H[1],@H[1]+ orr $EF,@H[2],@H[2]+ orr $GH,@H[3],@H[3]+ csel c#$inp,c#$inp,c4,ne // conditional rewind+___+for($i=0;$i<32;$i++) {+$code.=<<___;+ add.i64 $W0,$W0,@MSG[0]+ ld1.64 {$W1},[$Ktbl],#16+ ext $W0,$W0,$W0,#8+ ext $fg,@H[2],@H[3],#8+ ext $de,@H[1],@H[2],#8+ add.i64 @H[3],@H[3],$W0 // "T1 + H + K512[i]"+ sha512su0 @MSG[0],@MSG[1]+ ext $m9_10,@MSG[4],@MSG[5],#8+ sha512h @H[3],$fg,$de+ sha512su1 @MSG[0],@MSG[7],$m9_10+ add.i64 @H[4],@H[1],@H[3] // "D + T1"+ sha512h2 @H[3],$H[1],@H[0]+___+ ($W0,$W1)=($W1,$W0); push(@MSG,shift(@MSG));+ @H = (@H[3],@H[0],@H[4],@H[2],@H[1]);+}+for(;$i<40;$i++) {+$code.=<<___ if ($i<39);+ ld1.64 {$W1},[$Ktbl],#16+___+$code.=<<___ if ($i==39);+ csub $Ktbl,$Ktbl,#$rounds*$SZ // rewind+___+$code.=<<___;+ add.i64 $W0,$W0,@MSG[0]+ ld1 {@MSG[0]},[$inp],#16 // load next input+ ext $W0,$W0,$W0,#8+ ext $fg,@H[2],@H[3],#8+ ext $de,@H[1],@H[2],#8+ add.i64 @H[3],@H[3],$W0 // "T1 + H + K512[i]"+ sha512h @H[3],$fg,$de+ rev64 @MSG[0],@MSG[0]+ add.i64 @H[4],@H[1],@H[3] // "D + T1"+ sha512h2 @H[3],$H[1],@H[0]+___+ ($W0,$W1)=($W1,$W0); push(@MSG,shift(@MSG));+ @H = (@H[3],@H[0],@H[4],@H[2],@H[1]);+}+$code.=<<___;+ add.i64 @H[0],@H[0],$AB // accumulate+ add.i64 @H[1],@H[1],$CD+ add.i64 @H[2],@H[2],$EF+ add.i64 @H[3],@H[3],$GH++ cbnz $num,.Loop_hw++ st1.64 {@H[0]-@H[3]},[$ctx] // store context++ ldr c29,[csp],#2*__SIZEOF_POINTER__+ ret+.size sha512_block_armv8,.-sha512_block_armv8+#endif+___+}++$code.=<<___;+#if !defined(__KERNEL__) && !defined(_WIN64)+.comm OPENSSL_armcap_P,4,4+.hidden OPENSSL_armcap_P+#endif+___++{ my %opcode = (+ "sha256h" => 0x5e004000, "sha256h2" => 0x5e005000,+ "sha256su0" => 0x5e282800, "sha256su1" => 0x5e006000 );++ sub unsha256 {+ my ($mnemonic,$arg)=@_;++ $arg =~ m/[qv]([0-9]+)[^,]*,\s*[qv]([0-9]+)[^,]*(?:,\s*[qv]([0-9]+))?/o+ &&+ sprintf ".inst\t0x%08x\t//%s %s",+ $opcode{$mnemonic}|$1|($2<<5)|($3<<16),+ $mnemonic,$arg;+ }+}++{ my %opcode = (+ "sha512h" => 0xce608000, "sha512h2" => 0xce608400,+ "sha512su0" => 0xcec08000, "sha512su1" => 0xce608800 );++ sub unsha512 {+ my ($mnemonic,$arg)=@_;++ $arg =~ m/[qv]([0-9]+)[^,]*,\s*[qv]([0-9]+)[^,]*(?:,\s*[qv]([0-9]+))?/o+ &&+ sprintf ".inst\t0x%08x\t//%s %s",+ $opcode{$mnemonic}|$1|($2<<5)|($3<<16),+ $mnemonic,$arg;+ }+}++open SELF,$0;+while(<SELF>) {+ next if (/^#!/);+ last if (!s/^#/\/\// and !/^$/);+ print;+}+close SELF;++foreach(split("\n",$code)) {++ s/\`([^\`]*)\`/eval($1)/ge;++ s/\b(sha512\w+)\s+([qv].*)/unsha512($1,$2)/ge or+ s/\b(sha256\w+)\s+([qv].*)/unsha256($1,$2)/ge;++ s/\bq([0-9]+)\b/v$1.16b/g; # old->new registers++ s/\.[ui]?8(\s)/$1/;+ s/\.\w?64\b// and s/\.16b/\.2d/g or+ s/\.\w?32\b// and s/\.16b/\.4s/g;+ m/\bext\b/ and s/\.2d/\.16b/g or+ m/(ld|st)1[^\[]+\[0\]/ and s/\.4s/\.s/g;++ s/([cw])#x([0-9]+)/$1$2/g;++ print $_,"\n";+}++close STDOUT;
@@ -0,0 +1,5727 @@+.text +++.globl crypton_sha512_asm_block_data_order+.type crypton_sha512_asm_block_data_order,@function+.align 16+crypton_sha512_asm_block_data_order:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa+++ pushq %rbp+.cfi_adjust_cfa_offset 8+.cfi_offset %rbp,-16+ movq %rsp,%rbp+.cfi_def_cfa_register %rbp+ leaq crypton_ia32cap_P(%rip),%rax+ movl 0(%rax),%r9d+ movl 4(%rax),%r10d+ movl 8(%rax),%eax+ testl $2048,%r10d+ jnz .Lxop_shortcut+ andl $296,%eax+ cmpl $296,%eax+ je .Lavx2_shortcut+ andl $1073741824,%r9d+ andl $268435968,%r10d+ orl %r9d,%r10d+ cmpl $1342177792,%r10d+ je .Lavx_shortcut+ pushq %rbx+.cfi_offset %rbx,-24+ pushq %r12+.cfi_offset %r12,-32+ pushq %r13+.cfi_offset %r13,-40+ pushq %r14+.cfi_offset %r14,-48+ pushq %r15+.cfi_offset %r15,-56+ shlq $4,%rdx+ subq $128+24,%rsp++.cfi_def_cfa %rsp,208++ leaq (%rsi,%rdx,8),%rdx+ movq %rdi,128+0(%rsp)+ movq %rsi,128+8(%rsp)+ movq %rdx,128+16(%rsp)++ movq 0(%rdi),%rax+ movq 8(%rdi),%rbx+ movq 16(%rdi),%rcx+ movq 24(%rdi),%rdx+ movq 32(%rdi),%r8+ movq 40(%rdi),%r9+ movq 48(%rdi),%r10+ movq 56(%rdi),%r11+ jmp .Lloop++.align 16+.Lloop:+ movq %rbx,%rdi+ leaq K512(%rip),%rbp+ xorq %rcx,%rdi+ movq 0(%rsi),%r12+ movq %r8,%r13+ movq %rax,%r14+ bswapq %r12+ rorq $23,%r13+ movq %r9,%r15++ xorq %r8,%r13+ rorq $5,%r14+ xorq %r10,%r15++ movq %r12,0(%rsp)+ xorq %rax,%r14+ andq %r8,%r15++ rorq $4,%r13+ addq %r11,%r12+ xorq %r10,%r15++ rorq $6,%r14+ xorq %r8,%r13+ addq %r15,%r12++ movq %rax,%r15+ addq (%rbp),%r12+ xorq %rax,%r14++ xorq %rbx,%r15+ rorq $14,%r13+ movq %rbx,%r11++ andq %r15,%rdi+ rorq $28,%r14+ addq %r13,%r12++ xorq %rdi,%r11+ addq %r12,%rdx+ addq %r12,%r11++ leaq 8(%rbp),%rbp+ addq %r14,%r11+ movq 8(%rsi),%r12+ movq %rdx,%r13+ movq %r11,%r14+ bswapq %r12+ rorq $23,%r13+ movq %r8,%rdi++ xorq %rdx,%r13+ rorq $5,%r14+ xorq %r9,%rdi++ movq %r12,8(%rsp)+ xorq %r11,%r14+ andq %rdx,%rdi++ rorq $4,%r13+ addq %r10,%r12+ xorq %r9,%rdi++ rorq $6,%r14+ xorq %rdx,%r13+ addq %rdi,%r12++ movq %r11,%rdi+ addq (%rbp),%r12+ xorq %r11,%r14++ xorq %rax,%rdi+ rorq $14,%r13+ movq %rax,%r10++ andq %rdi,%r15+ rorq $28,%r14+ addq %r13,%r12++ xorq %r15,%r10+ addq %r12,%rcx+ addq %r12,%r10++ leaq 24(%rbp),%rbp+ addq %r14,%r10+ movq 16(%rsi),%r12+ movq %rcx,%r13+ movq %r10,%r14+ bswapq %r12+ rorq $23,%r13+ movq %rdx,%r15++ xorq %rcx,%r13+ rorq $5,%r14+ xorq %r8,%r15++ movq %r12,16(%rsp)+ xorq %r10,%r14+ andq %rcx,%r15++ rorq $4,%r13+ addq %r9,%r12+ xorq %r8,%r15++ rorq $6,%r14+ xorq %rcx,%r13+ addq %r15,%r12++ movq %r10,%r15+ addq (%rbp),%r12+ xorq %r10,%r14++ xorq %r11,%r15+ rorq $14,%r13+ movq %r11,%r9++ andq %r15,%rdi+ rorq $28,%r14+ addq %r13,%r12++ xorq %rdi,%r9+ addq %r12,%rbx+ addq %r12,%r9++ leaq 8(%rbp),%rbp+ addq %r14,%r9+ movq 24(%rsi),%r12+ movq %rbx,%r13+ movq %r9,%r14+ bswapq %r12+ rorq $23,%r13+ movq %rcx,%rdi++ xorq %rbx,%r13+ rorq $5,%r14+ xorq %rdx,%rdi++ movq %r12,24(%rsp)+ xorq %r9,%r14+ andq %rbx,%rdi++ rorq $4,%r13+ addq %r8,%r12+ xorq %rdx,%rdi++ rorq $6,%r14+ xorq %rbx,%r13+ addq %rdi,%r12++ movq %r9,%rdi+ addq (%rbp),%r12+ xorq %r9,%r14++ xorq %r10,%rdi+ rorq $14,%r13+ movq %r10,%r8++ andq %rdi,%r15+ rorq $28,%r14+ addq %r13,%r12++ xorq %r15,%r8+ addq %r12,%rax+ addq %r12,%r8++ leaq 24(%rbp),%rbp+ addq %r14,%r8+ movq 32(%rsi),%r12+ movq %rax,%r13+ movq %r8,%r14+ bswapq %r12+ rorq $23,%r13+ movq %rbx,%r15++ xorq %rax,%r13+ rorq $5,%r14+ xorq %rcx,%r15++ movq %r12,32(%rsp)+ xorq %r8,%r14+ andq %rax,%r15++ rorq $4,%r13+ addq %rdx,%r12+ xorq %rcx,%r15++ rorq $6,%r14+ xorq %rax,%r13+ addq %r15,%r12++ movq %r8,%r15+ addq (%rbp),%r12+ xorq %r8,%r14++ xorq %r9,%r15+ rorq $14,%r13+ movq %r9,%rdx++ andq %r15,%rdi+ rorq $28,%r14+ addq %r13,%r12++ xorq %rdi,%rdx+ addq %r12,%r11+ addq %r12,%rdx++ leaq 8(%rbp),%rbp+ addq %r14,%rdx+ movq 40(%rsi),%r12+ movq %r11,%r13+ movq %rdx,%r14+ bswapq %r12+ rorq $23,%r13+ movq %rax,%rdi++ xorq %r11,%r13+ rorq $5,%r14+ xorq %rbx,%rdi++ movq %r12,40(%rsp)+ xorq %rdx,%r14+ andq %r11,%rdi++ rorq $4,%r13+ addq %rcx,%r12+ xorq %rbx,%rdi++ rorq $6,%r14+ xorq %r11,%r13+ addq %rdi,%r12++ movq %rdx,%rdi+ addq (%rbp),%r12+ xorq %rdx,%r14++ xorq %r8,%rdi+ rorq $14,%r13+ movq %r8,%rcx++ andq %rdi,%r15+ rorq $28,%r14+ addq %r13,%r12++ xorq %r15,%rcx+ addq %r12,%r10+ addq %r12,%rcx++ leaq 24(%rbp),%rbp+ addq %r14,%rcx+ movq 48(%rsi),%r12+ movq %r10,%r13+ movq %rcx,%r14+ bswapq %r12+ rorq $23,%r13+ movq %r11,%r15++ xorq %r10,%r13+ rorq $5,%r14+ xorq %rax,%r15++ movq %r12,48(%rsp)+ xorq %rcx,%r14+ andq %r10,%r15++ rorq $4,%r13+ addq %rbx,%r12+ xorq %rax,%r15++ rorq $6,%r14+ xorq %r10,%r13+ addq %r15,%r12++ movq %rcx,%r15+ addq (%rbp),%r12+ xorq %rcx,%r14++ xorq %rdx,%r15+ rorq $14,%r13+ movq %rdx,%rbx++ andq %r15,%rdi+ rorq $28,%r14+ addq %r13,%r12++ xorq %rdi,%rbx+ addq %r12,%r9+ addq %r12,%rbx++ leaq 8(%rbp),%rbp+ addq %r14,%rbx+ movq 56(%rsi),%r12+ movq %r9,%r13+ movq %rbx,%r14+ bswapq %r12+ rorq $23,%r13+ movq %r10,%rdi++ xorq %r9,%r13+ rorq $5,%r14+ xorq %r11,%rdi++ movq %r12,56(%rsp)+ xorq %rbx,%r14+ andq %r9,%rdi++ rorq $4,%r13+ addq %rax,%r12+ xorq %r11,%rdi++ rorq $6,%r14+ xorq %r9,%r13+ addq %rdi,%r12++ movq %rbx,%rdi+ addq (%rbp),%r12+ xorq %rbx,%r14++ xorq %rcx,%rdi+ rorq $14,%r13+ movq %rcx,%rax++ andq %rdi,%r15+ rorq $28,%r14+ addq %r13,%r12++ xorq %r15,%rax+ addq %r12,%r8+ addq %r12,%rax++ leaq 24(%rbp),%rbp+ addq %r14,%rax+ movq 64(%rsi),%r12+ movq %r8,%r13+ movq %rax,%r14+ bswapq %r12+ rorq $23,%r13+ movq %r9,%r15++ xorq %r8,%r13+ rorq $5,%r14+ xorq %r10,%r15++ movq %r12,64(%rsp)+ xorq %rax,%r14+ andq %r8,%r15++ rorq $4,%r13+ addq %r11,%r12+ xorq %r10,%r15++ rorq $6,%r14+ xorq %r8,%r13+ addq %r15,%r12++ movq %rax,%r15+ addq (%rbp),%r12+ xorq %rax,%r14++ xorq %rbx,%r15+ rorq $14,%r13+ movq %rbx,%r11++ andq %r15,%rdi+ rorq $28,%r14+ addq %r13,%r12++ xorq %rdi,%r11+ addq %r12,%rdx+ addq %r12,%r11++ leaq 8(%rbp),%rbp+ addq %r14,%r11+ movq 72(%rsi),%r12+ movq %rdx,%r13+ movq %r11,%r14+ bswapq %r12+ rorq $23,%r13+ movq %r8,%rdi++ xorq %rdx,%r13+ rorq $5,%r14+ xorq %r9,%rdi++ movq %r12,72(%rsp)+ xorq %r11,%r14+ andq %rdx,%rdi++ rorq $4,%r13+ addq %r10,%r12+ xorq %r9,%rdi++ rorq $6,%r14+ xorq %rdx,%r13+ addq %rdi,%r12++ movq %r11,%rdi+ addq (%rbp),%r12+ xorq %r11,%r14++ xorq %rax,%rdi+ rorq $14,%r13+ movq %rax,%r10++ andq %rdi,%r15+ rorq $28,%r14+ addq %r13,%r12++ xorq %r15,%r10+ addq %r12,%rcx+ addq %r12,%r10++ leaq 24(%rbp),%rbp+ addq %r14,%r10+ movq 80(%rsi),%r12+ movq %rcx,%r13+ movq %r10,%r14+ bswapq %r12+ rorq $23,%r13+ movq %rdx,%r15++ xorq %rcx,%r13+ rorq $5,%r14+ xorq %r8,%r15++ movq %r12,80(%rsp)+ xorq %r10,%r14+ andq %rcx,%r15++ rorq $4,%r13+ addq %r9,%r12+ xorq %r8,%r15++ rorq $6,%r14+ xorq %rcx,%r13+ addq %r15,%r12++ movq %r10,%r15+ addq (%rbp),%r12+ xorq %r10,%r14++ xorq %r11,%r15+ rorq $14,%r13+ movq %r11,%r9++ andq %r15,%rdi+ rorq $28,%r14+ addq %r13,%r12++ xorq %rdi,%r9+ addq %r12,%rbx+ addq %r12,%r9++ leaq 8(%rbp),%rbp+ addq %r14,%r9+ movq 88(%rsi),%r12+ movq %rbx,%r13+ movq %r9,%r14+ bswapq %r12+ rorq $23,%r13+ movq %rcx,%rdi++ xorq %rbx,%r13+ rorq $5,%r14+ xorq %rdx,%rdi++ movq %r12,88(%rsp)+ xorq %r9,%r14+ andq %rbx,%rdi++ rorq $4,%r13+ addq %r8,%r12+ xorq %rdx,%rdi++ rorq $6,%r14+ xorq %rbx,%r13+ addq %rdi,%r12++ movq %r9,%rdi+ addq (%rbp),%r12+ xorq %r9,%r14++ xorq %r10,%rdi+ rorq $14,%r13+ movq %r10,%r8++ andq %rdi,%r15+ rorq $28,%r14+ addq %r13,%r12++ xorq %r15,%r8+ addq %r12,%rax+ addq %r12,%r8++ leaq 24(%rbp),%rbp+ addq %r14,%r8+ movq 96(%rsi),%r12+ movq %rax,%r13+ movq %r8,%r14+ bswapq %r12+ rorq $23,%r13+ movq %rbx,%r15++ xorq %rax,%r13+ rorq $5,%r14+ xorq %rcx,%r15++ movq %r12,96(%rsp)+ xorq %r8,%r14+ andq %rax,%r15++ rorq $4,%r13+ addq %rdx,%r12+ xorq %rcx,%r15++ rorq $6,%r14+ xorq %rax,%r13+ addq %r15,%r12++ movq %r8,%r15+ addq (%rbp),%r12+ xorq %r8,%r14++ xorq %r9,%r15+ rorq $14,%r13+ movq %r9,%rdx++ andq %r15,%rdi+ rorq $28,%r14+ addq %r13,%r12++ xorq %rdi,%rdx+ addq %r12,%r11+ addq %r12,%rdx++ leaq 8(%rbp),%rbp+ addq %r14,%rdx+ movq 104(%rsi),%r12+ movq %r11,%r13+ movq %rdx,%r14+ bswapq %r12+ rorq $23,%r13+ movq %rax,%rdi++ xorq %r11,%r13+ rorq $5,%r14+ xorq %rbx,%rdi++ movq %r12,104(%rsp)+ xorq %rdx,%r14+ andq %r11,%rdi++ rorq $4,%r13+ addq %rcx,%r12+ xorq %rbx,%rdi++ rorq $6,%r14+ xorq %r11,%r13+ addq %rdi,%r12++ movq %rdx,%rdi+ addq (%rbp),%r12+ xorq %rdx,%r14++ xorq %r8,%rdi+ rorq $14,%r13+ movq %r8,%rcx++ andq %rdi,%r15+ rorq $28,%r14+ addq %r13,%r12++ xorq %r15,%rcx+ addq %r12,%r10+ addq %r12,%rcx++ leaq 24(%rbp),%rbp+ addq %r14,%rcx+ movq 112(%rsi),%r12+ movq %r10,%r13+ movq %rcx,%r14+ bswapq %r12+ rorq $23,%r13+ movq %r11,%r15++ xorq %r10,%r13+ rorq $5,%r14+ xorq %rax,%r15++ movq %r12,112(%rsp)+ xorq %rcx,%r14+ andq %r10,%r15++ rorq $4,%r13+ addq %rbx,%r12+ xorq %rax,%r15++ rorq $6,%r14+ xorq %r10,%r13+ addq %r15,%r12++ movq %rcx,%r15+ addq (%rbp),%r12+ xorq %rcx,%r14++ xorq %rdx,%r15+ rorq $14,%r13+ movq %rdx,%rbx++ andq %r15,%rdi+ rorq $28,%r14+ addq %r13,%r12++ xorq %rdi,%rbx+ addq %r12,%r9+ addq %r12,%rbx++ leaq 8(%rbp),%rbp+ addq %r14,%rbx+ movq 120(%rsi),%r12+ movq %r9,%r13+ movq %rbx,%r14+ bswapq %r12+ rorq $23,%r13+ movq %r10,%rdi++ xorq %r9,%r13+ rorq $5,%r14+ xorq %r11,%rdi++ movq %r12,120(%rsp)+ xorq %rbx,%r14+ andq %r9,%rdi++ rorq $4,%r13+ addq %rax,%r12+ xorq %r11,%rdi++ rorq $6,%r14+ xorq %r9,%r13+ addq %rdi,%r12++ movq %rbx,%rdi+ addq (%rbp),%r12+ xorq %rbx,%r14++ xorq %rcx,%rdi+ rorq $14,%r13+ movq %rcx,%rax++ andq %rdi,%r15+ rorq $28,%r14+ addq %r13,%r12++ xorq %r15,%rax+ addq %r12,%r8+ addq %r12,%rax++ leaq 24(%rbp),%rbp+ jmp .Lrounds_16_xx+.align 16+.Lrounds_16_xx:+ movq 8(%rsp),%r13+ movq 112(%rsp),%r15++ movq %r13,%r12+ rorq $7,%r13+ addq %r14,%rax+ movq %r15,%r14+ rorq $42,%r15++ xorq %r12,%r13+ shrq $7,%r12+ rorq $1,%r13+ xorq %r14,%r15+ shrq $6,%r14++ rorq $19,%r15+ xorq %r13,%r12+ xorq %r14,%r15+ addq 72(%rsp),%r12++ addq 0(%rsp),%r12+ movq %r8,%r13+ addq %r15,%r12+ movq %rax,%r14+ rorq $23,%r13+ movq %r9,%r15++ xorq %r8,%r13+ rorq $5,%r14+ xorq %r10,%r15++ movq %r12,0(%rsp)+ xorq %rax,%r14+ andq %r8,%r15++ rorq $4,%r13+ addq %r11,%r12+ xorq %r10,%r15++ rorq $6,%r14+ xorq %r8,%r13+ addq %r15,%r12++ movq %rax,%r15+ addq (%rbp),%r12+ xorq %rax,%r14++ xorq %rbx,%r15+ rorq $14,%r13+ movq %rbx,%r11++ andq %r15,%rdi+ rorq $28,%r14+ addq %r13,%r12++ xorq %rdi,%r11+ addq %r12,%rdx+ addq %r12,%r11++ leaq 8(%rbp),%rbp+ movq 16(%rsp),%r13+ movq 120(%rsp),%rdi++ movq %r13,%r12+ rorq $7,%r13+ addq %r14,%r11+ movq %rdi,%r14+ rorq $42,%rdi++ xorq %r12,%r13+ shrq $7,%r12+ rorq $1,%r13+ xorq %r14,%rdi+ shrq $6,%r14++ rorq $19,%rdi+ xorq %r13,%r12+ xorq %r14,%rdi+ addq 80(%rsp),%r12++ addq 8(%rsp),%r12+ movq %rdx,%r13+ addq %rdi,%r12+ movq %r11,%r14+ rorq $23,%r13+ movq %r8,%rdi++ xorq %rdx,%r13+ rorq $5,%r14+ xorq %r9,%rdi++ movq %r12,8(%rsp)+ xorq %r11,%r14+ andq %rdx,%rdi++ rorq $4,%r13+ addq %r10,%r12+ xorq %r9,%rdi++ rorq $6,%r14+ xorq %rdx,%r13+ addq %rdi,%r12++ movq %r11,%rdi+ addq (%rbp),%r12+ xorq %r11,%r14++ xorq %rax,%rdi+ rorq $14,%r13+ movq %rax,%r10++ andq %rdi,%r15+ rorq $28,%r14+ addq %r13,%r12++ xorq %r15,%r10+ addq %r12,%rcx+ addq %r12,%r10++ leaq 24(%rbp),%rbp+ movq 24(%rsp),%r13+ movq 0(%rsp),%r15++ movq %r13,%r12+ rorq $7,%r13+ addq %r14,%r10+ movq %r15,%r14+ rorq $42,%r15++ xorq %r12,%r13+ shrq $7,%r12+ rorq $1,%r13+ xorq %r14,%r15+ shrq $6,%r14++ rorq $19,%r15+ xorq %r13,%r12+ xorq %r14,%r15+ addq 88(%rsp),%r12++ addq 16(%rsp),%r12+ movq %rcx,%r13+ addq %r15,%r12+ movq %r10,%r14+ rorq $23,%r13+ movq %rdx,%r15++ xorq %rcx,%r13+ rorq $5,%r14+ xorq %r8,%r15++ movq %r12,16(%rsp)+ xorq %r10,%r14+ andq %rcx,%r15++ rorq $4,%r13+ addq %r9,%r12+ xorq %r8,%r15++ rorq $6,%r14+ xorq %rcx,%r13+ addq %r15,%r12++ movq %r10,%r15+ addq (%rbp),%r12+ xorq %r10,%r14++ xorq %r11,%r15+ rorq $14,%r13+ movq %r11,%r9++ andq %r15,%rdi+ rorq $28,%r14+ addq %r13,%r12++ xorq %rdi,%r9+ addq %r12,%rbx+ addq %r12,%r9++ leaq 8(%rbp),%rbp+ movq 32(%rsp),%r13+ movq 8(%rsp),%rdi++ movq %r13,%r12+ rorq $7,%r13+ addq %r14,%r9+ movq %rdi,%r14+ rorq $42,%rdi++ xorq %r12,%r13+ shrq $7,%r12+ rorq $1,%r13+ xorq %r14,%rdi+ shrq $6,%r14++ rorq $19,%rdi+ xorq %r13,%r12+ xorq %r14,%rdi+ addq 96(%rsp),%r12++ addq 24(%rsp),%r12+ movq %rbx,%r13+ addq %rdi,%r12+ movq %r9,%r14+ rorq $23,%r13+ movq %rcx,%rdi++ xorq %rbx,%r13+ rorq $5,%r14+ xorq %rdx,%rdi++ movq %r12,24(%rsp)+ xorq %r9,%r14+ andq %rbx,%rdi++ rorq $4,%r13+ addq %r8,%r12+ xorq %rdx,%rdi++ rorq $6,%r14+ xorq %rbx,%r13+ addq %rdi,%r12++ movq %r9,%rdi+ addq (%rbp),%r12+ xorq %r9,%r14++ xorq %r10,%rdi+ rorq $14,%r13+ movq %r10,%r8++ andq %rdi,%r15+ rorq $28,%r14+ addq %r13,%r12++ xorq %r15,%r8+ addq %r12,%rax+ addq %r12,%r8++ leaq 24(%rbp),%rbp+ movq 40(%rsp),%r13+ movq 16(%rsp),%r15++ movq %r13,%r12+ rorq $7,%r13+ addq %r14,%r8+ movq %r15,%r14+ rorq $42,%r15++ xorq %r12,%r13+ shrq $7,%r12+ rorq $1,%r13+ xorq %r14,%r15+ shrq $6,%r14++ rorq $19,%r15+ xorq %r13,%r12+ xorq %r14,%r15+ addq 104(%rsp),%r12++ addq 32(%rsp),%r12+ movq %rax,%r13+ addq %r15,%r12+ movq %r8,%r14+ rorq $23,%r13+ movq %rbx,%r15++ xorq %rax,%r13+ rorq $5,%r14+ xorq %rcx,%r15++ movq %r12,32(%rsp)+ xorq %r8,%r14+ andq %rax,%r15++ rorq $4,%r13+ addq %rdx,%r12+ xorq %rcx,%r15++ rorq $6,%r14+ xorq %rax,%r13+ addq %r15,%r12++ movq %r8,%r15+ addq (%rbp),%r12+ xorq %r8,%r14++ xorq %r9,%r15+ rorq $14,%r13+ movq %r9,%rdx++ andq %r15,%rdi+ rorq $28,%r14+ addq %r13,%r12++ xorq %rdi,%rdx+ addq %r12,%r11+ addq %r12,%rdx++ leaq 8(%rbp),%rbp+ movq 48(%rsp),%r13+ movq 24(%rsp),%rdi++ movq %r13,%r12+ rorq $7,%r13+ addq %r14,%rdx+ movq %rdi,%r14+ rorq $42,%rdi++ xorq %r12,%r13+ shrq $7,%r12+ rorq $1,%r13+ xorq %r14,%rdi+ shrq $6,%r14++ rorq $19,%rdi+ xorq %r13,%r12+ xorq %r14,%rdi+ addq 112(%rsp),%r12++ addq 40(%rsp),%r12+ movq %r11,%r13+ addq %rdi,%r12+ movq %rdx,%r14+ rorq $23,%r13+ movq %rax,%rdi++ xorq %r11,%r13+ rorq $5,%r14+ xorq %rbx,%rdi++ movq %r12,40(%rsp)+ xorq %rdx,%r14+ andq %r11,%rdi++ rorq $4,%r13+ addq %rcx,%r12+ xorq %rbx,%rdi++ rorq $6,%r14+ xorq %r11,%r13+ addq %rdi,%r12++ movq %rdx,%rdi+ addq (%rbp),%r12+ xorq %rdx,%r14++ xorq %r8,%rdi+ rorq $14,%r13+ movq %r8,%rcx++ andq %rdi,%r15+ rorq $28,%r14+ addq %r13,%r12++ xorq %r15,%rcx+ addq %r12,%r10+ addq %r12,%rcx++ leaq 24(%rbp),%rbp+ movq 56(%rsp),%r13+ movq 32(%rsp),%r15++ movq %r13,%r12+ rorq $7,%r13+ addq %r14,%rcx+ movq %r15,%r14+ rorq $42,%r15++ xorq %r12,%r13+ shrq $7,%r12+ rorq $1,%r13+ xorq %r14,%r15+ shrq $6,%r14++ rorq $19,%r15+ xorq %r13,%r12+ xorq %r14,%r15+ addq 120(%rsp),%r12++ addq 48(%rsp),%r12+ movq %r10,%r13+ addq %r15,%r12+ movq %rcx,%r14+ rorq $23,%r13+ movq %r11,%r15++ xorq %r10,%r13+ rorq $5,%r14+ xorq %rax,%r15++ movq %r12,48(%rsp)+ xorq %rcx,%r14+ andq %r10,%r15++ rorq $4,%r13+ addq %rbx,%r12+ xorq %rax,%r15++ rorq $6,%r14+ xorq %r10,%r13+ addq %r15,%r12++ movq %rcx,%r15+ addq (%rbp),%r12+ xorq %rcx,%r14++ xorq %rdx,%r15+ rorq $14,%r13+ movq %rdx,%rbx++ andq %r15,%rdi+ rorq $28,%r14+ addq %r13,%r12++ xorq %rdi,%rbx+ addq %r12,%r9+ addq %r12,%rbx++ leaq 8(%rbp),%rbp+ movq 64(%rsp),%r13+ movq 40(%rsp),%rdi++ movq %r13,%r12+ rorq $7,%r13+ addq %r14,%rbx+ movq %rdi,%r14+ rorq $42,%rdi++ xorq %r12,%r13+ shrq $7,%r12+ rorq $1,%r13+ xorq %r14,%rdi+ shrq $6,%r14++ rorq $19,%rdi+ xorq %r13,%r12+ xorq %r14,%rdi+ addq 0(%rsp),%r12++ addq 56(%rsp),%r12+ movq %r9,%r13+ addq %rdi,%r12+ movq %rbx,%r14+ rorq $23,%r13+ movq %r10,%rdi++ xorq %r9,%r13+ rorq $5,%r14+ xorq %r11,%rdi++ movq %r12,56(%rsp)+ xorq %rbx,%r14+ andq %r9,%rdi++ rorq $4,%r13+ addq %rax,%r12+ xorq %r11,%rdi++ rorq $6,%r14+ xorq %r9,%r13+ addq %rdi,%r12++ movq %rbx,%rdi+ addq (%rbp),%r12+ xorq %rbx,%r14++ xorq %rcx,%rdi+ rorq $14,%r13+ movq %rcx,%rax++ andq %rdi,%r15+ rorq $28,%r14+ addq %r13,%r12++ xorq %r15,%rax+ addq %r12,%r8+ addq %r12,%rax++ leaq 24(%rbp),%rbp+ movq 72(%rsp),%r13+ movq 48(%rsp),%r15++ movq %r13,%r12+ rorq $7,%r13+ addq %r14,%rax+ movq %r15,%r14+ rorq $42,%r15++ xorq %r12,%r13+ shrq $7,%r12+ rorq $1,%r13+ xorq %r14,%r15+ shrq $6,%r14++ rorq $19,%r15+ xorq %r13,%r12+ xorq %r14,%r15+ addq 8(%rsp),%r12++ addq 64(%rsp),%r12+ movq %r8,%r13+ addq %r15,%r12+ movq %rax,%r14+ rorq $23,%r13+ movq %r9,%r15++ xorq %r8,%r13+ rorq $5,%r14+ xorq %r10,%r15++ movq %r12,64(%rsp)+ xorq %rax,%r14+ andq %r8,%r15++ rorq $4,%r13+ addq %r11,%r12+ xorq %r10,%r15++ rorq $6,%r14+ xorq %r8,%r13+ addq %r15,%r12++ movq %rax,%r15+ addq (%rbp),%r12+ xorq %rax,%r14++ xorq %rbx,%r15+ rorq $14,%r13+ movq %rbx,%r11++ andq %r15,%rdi+ rorq $28,%r14+ addq %r13,%r12++ xorq %rdi,%r11+ addq %r12,%rdx+ addq %r12,%r11++ leaq 8(%rbp),%rbp+ movq 80(%rsp),%r13+ movq 56(%rsp),%rdi++ movq %r13,%r12+ rorq $7,%r13+ addq %r14,%r11+ movq %rdi,%r14+ rorq $42,%rdi++ xorq %r12,%r13+ shrq $7,%r12+ rorq $1,%r13+ xorq %r14,%rdi+ shrq $6,%r14++ rorq $19,%rdi+ xorq %r13,%r12+ xorq %r14,%rdi+ addq 16(%rsp),%r12++ addq 72(%rsp),%r12+ movq %rdx,%r13+ addq %rdi,%r12+ movq %r11,%r14+ rorq $23,%r13+ movq %r8,%rdi++ xorq %rdx,%r13+ rorq $5,%r14+ xorq %r9,%rdi++ movq %r12,72(%rsp)+ xorq %r11,%r14+ andq %rdx,%rdi++ rorq $4,%r13+ addq %r10,%r12+ xorq %r9,%rdi++ rorq $6,%r14+ xorq %rdx,%r13+ addq %rdi,%r12++ movq %r11,%rdi+ addq (%rbp),%r12+ xorq %r11,%r14++ xorq %rax,%rdi+ rorq $14,%r13+ movq %rax,%r10++ andq %rdi,%r15+ rorq $28,%r14+ addq %r13,%r12++ xorq %r15,%r10+ addq %r12,%rcx+ addq %r12,%r10++ leaq 24(%rbp),%rbp+ movq 88(%rsp),%r13+ movq 64(%rsp),%r15++ movq %r13,%r12+ rorq $7,%r13+ addq %r14,%r10+ movq %r15,%r14+ rorq $42,%r15++ xorq %r12,%r13+ shrq $7,%r12+ rorq $1,%r13+ xorq %r14,%r15+ shrq $6,%r14++ rorq $19,%r15+ xorq %r13,%r12+ xorq %r14,%r15+ addq 24(%rsp),%r12++ addq 80(%rsp),%r12+ movq %rcx,%r13+ addq %r15,%r12+ movq %r10,%r14+ rorq $23,%r13+ movq %rdx,%r15++ xorq %rcx,%r13+ rorq $5,%r14+ xorq %r8,%r15++ movq %r12,80(%rsp)+ xorq %r10,%r14+ andq %rcx,%r15++ rorq $4,%r13+ addq %r9,%r12+ xorq %r8,%r15++ rorq $6,%r14+ xorq %rcx,%r13+ addq %r15,%r12++ movq %r10,%r15+ addq (%rbp),%r12+ xorq %r10,%r14++ xorq %r11,%r15+ rorq $14,%r13+ movq %r11,%r9++ andq %r15,%rdi+ rorq $28,%r14+ addq %r13,%r12++ xorq %rdi,%r9+ addq %r12,%rbx+ addq %r12,%r9++ leaq 8(%rbp),%rbp+ movq 96(%rsp),%r13+ movq 72(%rsp),%rdi++ movq %r13,%r12+ rorq $7,%r13+ addq %r14,%r9+ movq %rdi,%r14+ rorq $42,%rdi++ xorq %r12,%r13+ shrq $7,%r12+ rorq $1,%r13+ xorq %r14,%rdi+ shrq $6,%r14++ rorq $19,%rdi+ xorq %r13,%r12+ xorq %r14,%rdi+ addq 32(%rsp),%r12++ addq 88(%rsp),%r12+ movq %rbx,%r13+ addq %rdi,%r12+ movq %r9,%r14+ rorq $23,%r13+ movq %rcx,%rdi++ xorq %rbx,%r13+ rorq $5,%r14+ xorq %rdx,%rdi++ movq %r12,88(%rsp)+ xorq %r9,%r14+ andq %rbx,%rdi++ rorq $4,%r13+ addq %r8,%r12+ xorq %rdx,%rdi++ rorq $6,%r14+ xorq %rbx,%r13+ addq %rdi,%r12++ movq %r9,%rdi+ addq (%rbp),%r12+ xorq %r9,%r14++ xorq %r10,%rdi+ rorq $14,%r13+ movq %r10,%r8++ andq %rdi,%r15+ rorq $28,%r14+ addq %r13,%r12++ xorq %r15,%r8+ addq %r12,%rax+ addq %r12,%r8++ leaq 24(%rbp),%rbp+ movq 104(%rsp),%r13+ movq 80(%rsp),%r15++ movq %r13,%r12+ rorq $7,%r13+ addq %r14,%r8+ movq %r15,%r14+ rorq $42,%r15++ xorq %r12,%r13+ shrq $7,%r12+ rorq $1,%r13+ xorq %r14,%r15+ shrq $6,%r14++ rorq $19,%r15+ xorq %r13,%r12+ xorq %r14,%r15+ addq 40(%rsp),%r12++ addq 96(%rsp),%r12+ movq %rax,%r13+ addq %r15,%r12+ movq %r8,%r14+ rorq $23,%r13+ movq %rbx,%r15++ xorq %rax,%r13+ rorq $5,%r14+ xorq %rcx,%r15++ movq %r12,96(%rsp)+ xorq %r8,%r14+ andq %rax,%r15++ rorq $4,%r13+ addq %rdx,%r12+ xorq %rcx,%r15++ rorq $6,%r14+ xorq %rax,%r13+ addq %r15,%r12++ movq %r8,%r15+ addq (%rbp),%r12+ xorq %r8,%r14++ xorq %r9,%r15+ rorq $14,%r13+ movq %r9,%rdx++ andq %r15,%rdi+ rorq $28,%r14+ addq %r13,%r12++ xorq %rdi,%rdx+ addq %r12,%r11+ addq %r12,%rdx++ leaq 8(%rbp),%rbp+ movq 112(%rsp),%r13+ movq 88(%rsp),%rdi++ movq %r13,%r12+ rorq $7,%r13+ addq %r14,%rdx+ movq %rdi,%r14+ rorq $42,%rdi++ xorq %r12,%r13+ shrq $7,%r12+ rorq $1,%r13+ xorq %r14,%rdi+ shrq $6,%r14++ rorq $19,%rdi+ xorq %r13,%r12+ xorq %r14,%rdi+ addq 48(%rsp),%r12++ addq 104(%rsp),%r12+ movq %r11,%r13+ addq %rdi,%r12+ movq %rdx,%r14+ rorq $23,%r13+ movq %rax,%rdi++ xorq %r11,%r13+ rorq $5,%r14+ xorq %rbx,%rdi++ movq %r12,104(%rsp)+ xorq %rdx,%r14+ andq %r11,%rdi++ rorq $4,%r13+ addq %rcx,%r12+ xorq %rbx,%rdi++ rorq $6,%r14+ xorq %r11,%r13+ addq %rdi,%r12++ movq %rdx,%rdi+ addq (%rbp),%r12+ xorq %rdx,%r14++ xorq %r8,%rdi+ rorq $14,%r13+ movq %r8,%rcx++ andq %rdi,%r15+ rorq $28,%r14+ addq %r13,%r12++ xorq %r15,%rcx+ addq %r12,%r10+ addq %r12,%rcx++ leaq 24(%rbp),%rbp+ movq 120(%rsp),%r13+ movq 96(%rsp),%r15++ movq %r13,%r12+ rorq $7,%r13+ addq %r14,%rcx+ movq %r15,%r14+ rorq $42,%r15++ xorq %r12,%r13+ shrq $7,%r12+ rorq $1,%r13+ xorq %r14,%r15+ shrq $6,%r14++ rorq $19,%r15+ xorq %r13,%r12+ xorq %r14,%r15+ addq 56(%rsp),%r12++ addq 112(%rsp),%r12+ movq %r10,%r13+ addq %r15,%r12+ movq %rcx,%r14+ rorq $23,%r13+ movq %r11,%r15++ xorq %r10,%r13+ rorq $5,%r14+ xorq %rax,%r15++ movq %r12,112(%rsp)+ xorq %rcx,%r14+ andq %r10,%r15++ rorq $4,%r13+ addq %rbx,%r12+ xorq %rax,%r15++ rorq $6,%r14+ xorq %r10,%r13+ addq %r15,%r12++ movq %rcx,%r15+ addq (%rbp),%r12+ xorq %rcx,%r14++ xorq %rdx,%r15+ rorq $14,%r13+ movq %rdx,%rbx++ andq %r15,%rdi+ rorq $28,%r14+ addq %r13,%r12++ xorq %rdi,%rbx+ addq %r12,%r9+ addq %r12,%rbx++ leaq 8(%rbp),%rbp+ movq 0(%rsp),%r13+ movq 104(%rsp),%rdi++ movq %r13,%r12+ rorq $7,%r13+ addq %r14,%rbx+ movq %rdi,%r14+ rorq $42,%rdi++ xorq %r12,%r13+ shrq $7,%r12+ rorq $1,%r13+ xorq %r14,%rdi+ shrq $6,%r14++ rorq $19,%rdi+ xorq %r13,%r12+ xorq %r14,%rdi+ addq 64(%rsp),%r12++ addq 120(%rsp),%r12+ movq %r9,%r13+ addq %rdi,%r12+ movq %rbx,%r14+ rorq $23,%r13+ movq %r10,%rdi++ xorq %r9,%r13+ rorq $5,%r14+ xorq %r11,%rdi++ movq %r12,120(%rsp)+ xorq %rbx,%r14+ andq %r9,%rdi++ rorq $4,%r13+ addq %rax,%r12+ xorq %r11,%rdi++ rorq $6,%r14+ xorq %r9,%r13+ addq %rdi,%r12++ movq %rbx,%rdi+ addq (%rbp),%r12+ xorq %rbx,%r14++ xorq %rcx,%rdi+ rorq $14,%r13+ movq %rcx,%rax++ andq %rdi,%r15+ rorq $28,%r14+ addq %r13,%r12++ xorq %r15,%rax+ addq %r12,%r8+ addq %r12,%rax++ leaq 24(%rbp),%rbp+ cmpb $0,7(%rbp)+ jnz .Lrounds_16_xx++ movq 128+0(%rsp),%rdi+ addq %r14,%rax+ leaq 128(%rsi),%rsi++ addq 0(%rdi),%rax+ addq 8(%rdi),%rbx+ addq 16(%rdi),%rcx+ addq 24(%rdi),%rdx+ addq 32(%rdi),%r8+ addq 40(%rdi),%r9+ addq 48(%rdi),%r10+ addq 56(%rdi),%r11++ cmpq 128+16(%rsp),%rsi++ movq %rax,0(%rdi)+ movq %rbx,8(%rdi)+ movq %rcx,16(%rdi)+ movq %rdx,24(%rdi)+ movq %r8,32(%rdi)+ movq %r9,40(%rdi)+ movq %r10,48(%rdi)+ movq %r11,56(%rdi)+ jb .Lloop++ leaq 128+24+48(%rsp),%r11+.cfi_def_cfa %r11,8+ movq 128+24(%rsp),%r15+ movq -40(%r11),%r14+ movq -32(%r11),%r13+ movq -24(%r11),%r12+ movq -16(%r11),%rbx+ movq -8(%r11),%rbp+.cfi_restore %r12+.cfi_restore %r13+.cfi_restore %r14+.cfi_restore %r15+.cfi_restore %rbp+.cfi_restore %rbx+ leaq (%r11),%rsp+ .byte 0xf3,0xc3+.cfi_endproc +.size crypton_sha512_asm_block_data_order,.-crypton_sha512_asm_block_data_order+.align 64+.type K512,@object+K512:+.quad 0x428a2f98d728ae22,0x7137449123ef65cd+.quad 0x428a2f98d728ae22,0x7137449123ef65cd+.quad 0xb5c0fbcfec4d3b2f,0xe9b5dba58189dbbc+.quad 0xb5c0fbcfec4d3b2f,0xe9b5dba58189dbbc+.quad 0x3956c25bf348b538,0x59f111f1b605d019+.quad 0x3956c25bf348b538,0x59f111f1b605d019+.quad 0x923f82a4af194f9b,0xab1c5ed5da6d8118+.quad 0x923f82a4af194f9b,0xab1c5ed5da6d8118+.quad 0xd807aa98a3030242,0x12835b0145706fbe+.quad 0xd807aa98a3030242,0x12835b0145706fbe+.quad 0x243185be4ee4b28c,0x550c7dc3d5ffb4e2+.quad 0x243185be4ee4b28c,0x550c7dc3d5ffb4e2+.quad 0x72be5d74f27b896f,0x80deb1fe3b1696b1+.quad 0x72be5d74f27b896f,0x80deb1fe3b1696b1+.quad 0x9bdc06a725c71235,0xc19bf174cf692694+.quad 0x9bdc06a725c71235,0xc19bf174cf692694+.quad 0xe49b69c19ef14ad2,0xefbe4786384f25e3+.quad 0xe49b69c19ef14ad2,0xefbe4786384f25e3+.quad 0x0fc19dc68b8cd5b5,0x240ca1cc77ac9c65+.quad 0x0fc19dc68b8cd5b5,0x240ca1cc77ac9c65+.quad 0x2de92c6f592b0275,0x4a7484aa6ea6e483+.quad 0x2de92c6f592b0275,0x4a7484aa6ea6e483+.quad 0x5cb0a9dcbd41fbd4,0x76f988da831153b5+.quad 0x5cb0a9dcbd41fbd4,0x76f988da831153b5+.quad 0x983e5152ee66dfab,0xa831c66d2db43210+.quad 0x983e5152ee66dfab,0xa831c66d2db43210+.quad 0xb00327c898fb213f,0xbf597fc7beef0ee4+.quad 0xb00327c898fb213f,0xbf597fc7beef0ee4+.quad 0xc6e00bf33da88fc2,0xd5a79147930aa725+.quad 0xc6e00bf33da88fc2,0xd5a79147930aa725+.quad 0x06ca6351e003826f,0x142929670a0e6e70+.quad 0x06ca6351e003826f,0x142929670a0e6e70+.quad 0x27b70a8546d22ffc,0x2e1b21385c26c926+.quad 0x27b70a8546d22ffc,0x2e1b21385c26c926+.quad 0x4d2c6dfc5ac42aed,0x53380d139d95b3df+.quad 0x4d2c6dfc5ac42aed,0x53380d139d95b3df+.quad 0x650a73548baf63de,0x766a0abb3c77b2a8+.quad 0x650a73548baf63de,0x766a0abb3c77b2a8+.quad 0x81c2c92e47edaee6,0x92722c851482353b+.quad 0x81c2c92e47edaee6,0x92722c851482353b+.quad 0xa2bfe8a14cf10364,0xa81a664bbc423001+.quad 0xa2bfe8a14cf10364,0xa81a664bbc423001+.quad 0xc24b8b70d0f89791,0xc76c51a30654be30+.quad 0xc24b8b70d0f89791,0xc76c51a30654be30+.quad 0xd192e819d6ef5218,0xd69906245565a910+.quad 0xd192e819d6ef5218,0xd69906245565a910+.quad 0xf40e35855771202a,0x106aa07032bbd1b8+.quad 0xf40e35855771202a,0x106aa07032bbd1b8+.quad 0x19a4c116b8d2d0c8,0x1e376c085141ab53+.quad 0x19a4c116b8d2d0c8,0x1e376c085141ab53+.quad 0x2748774cdf8eeb99,0x34b0bcb5e19b48a8+.quad 0x2748774cdf8eeb99,0x34b0bcb5e19b48a8+.quad 0x391c0cb3c5c95a63,0x4ed8aa4ae3418acb+.quad 0x391c0cb3c5c95a63,0x4ed8aa4ae3418acb+.quad 0x5b9cca4f7763e373,0x682e6ff3d6b2b8a3+.quad 0x5b9cca4f7763e373,0x682e6ff3d6b2b8a3+.quad 0x748f82ee5defb2fc,0x78a5636f43172f60+.quad 0x748f82ee5defb2fc,0x78a5636f43172f60+.quad 0x84c87814a1f0ab72,0x8cc702081a6439ec+.quad 0x84c87814a1f0ab72,0x8cc702081a6439ec+.quad 0x90befffa23631e28,0xa4506cebde82bde9+.quad 0x90befffa23631e28,0xa4506cebde82bde9+.quad 0xbef9a3f7b2c67915,0xc67178f2e372532b+.quad 0xbef9a3f7b2c67915,0xc67178f2e372532b+.quad 0xca273eceea26619c,0xd186b8c721c0c207+.quad 0xca273eceea26619c,0xd186b8c721c0c207+.quad 0xeada7dd6cde0eb1e,0xf57d4f7fee6ed178+.quad 0xeada7dd6cde0eb1e,0xf57d4f7fee6ed178+.quad 0x06f067aa72176fba,0x0a637dc5a2c898a6+.quad 0x06f067aa72176fba,0x0a637dc5a2c898a6+.quad 0x113f9804bef90dae,0x1b710b35131c471b+.quad 0x113f9804bef90dae,0x1b710b35131c471b+.quad 0x28db77f523047d84,0x32caab7b40c72493+.quad 0x28db77f523047d84,0x32caab7b40c72493+.quad 0x3c9ebe0a15c9bebc,0x431d67c49c100d4c+.quad 0x3c9ebe0a15c9bebc,0x431d67c49c100d4c+.quad 0x4cc5d4becb3e42b6,0x597f299cfc657e2a+.quad 0x4cc5d4becb3e42b6,0x597f299cfc657e2a+.quad 0x5fcb6fab3ad6faec,0x6c44198c4a475817+.quad 0x5fcb6fab3ad6faec,0x6c44198c4a475817++.quad 0x0001020304050607,0x08090a0b0c0d0e0f+.quad 0x0001020304050607,0x08090a0b0c0d0e0f++K512_nodup:+.quad 0x428a2f98d728ae22,0x7137449123ef65cd+.quad 0xb5c0fbcfec4d3b2f,0xe9b5dba58189dbbc+.quad 0x3956c25bf348b538,0x59f111f1b605d019+.quad 0x923f82a4af194f9b,0xab1c5ed5da6d8118+.quad 0xd807aa98a3030242,0x12835b0145706fbe+.quad 0x243185be4ee4b28c,0x550c7dc3d5ffb4e2+.quad 0x72be5d74f27b896f,0x80deb1fe3b1696b1+.quad 0x9bdc06a725c71235,0xc19bf174cf692694+.quad 0xe49b69c19ef14ad2,0xefbe4786384f25e3+.quad 0x0fc19dc68b8cd5b5,0x240ca1cc77ac9c65+.quad 0x2de92c6f592b0275,0x4a7484aa6ea6e483+.quad 0x5cb0a9dcbd41fbd4,0x76f988da831153b5+.quad 0x983e5152ee66dfab,0xa831c66d2db43210+.quad 0xb00327c898fb213f,0xbf597fc7beef0ee4+.quad 0xc6e00bf33da88fc2,0xd5a79147930aa725+.quad 0x06ca6351e003826f,0x142929670a0e6e70+.quad 0x27b70a8546d22ffc,0x2e1b21385c26c926+.quad 0x4d2c6dfc5ac42aed,0x53380d139d95b3df+.quad 0x650a73548baf63de,0x766a0abb3c77b2a8+.quad 0x81c2c92e47edaee6,0x92722c851482353b+.quad 0xa2bfe8a14cf10364,0xa81a664bbc423001+.quad 0xc24b8b70d0f89791,0xc76c51a30654be30+.quad 0xd192e819d6ef5218,0xd69906245565a910+.quad 0xf40e35855771202a,0x106aa07032bbd1b8+.quad 0x19a4c116b8d2d0c8,0x1e376c085141ab53+.quad 0x2748774cdf8eeb99,0x34b0bcb5e19b48a8+.quad 0x391c0cb3c5c95a63,0x4ed8aa4ae3418acb+.quad 0x5b9cca4f7763e373,0x682e6ff3d6b2b8a3+.quad 0x748f82ee5defb2fc,0x78a5636f43172f60+.quad 0x84c87814a1f0ab72,0x8cc702081a6439ec+.quad 0x90befffa23631e28,0xa4506cebde82bde9+.quad 0xbef9a3f7b2c67915,0xc67178f2e372532b+.quad 0xca273eceea26619c,0xd186b8c721c0c207+.quad 0xeada7dd6cde0eb1e,0xf57d4f7fee6ed178+.quad 0x06f067aa72176fba,0x0a637dc5a2c898a6+.quad 0x113f9804bef90dae,0x1b710b35131c471b+.quad 0x28db77f523047d84,0x32caab7b40c72493+.quad 0x3c9ebe0a15c9bebc,0x431d67c49c100d4c+.quad 0x4cc5d4becb3e42b6,0x597f299cfc657e2a+.quad 0x5fcb6fab3ad6faec,0x6c44198c4a475817++.byte 83,72,65,53,49,50,32,98,108,111,99,107,32,116,114,97,110,115,102,111,114,109,32,102,111,114,32,120,56,54,95,54,52,44,32,67,82,89,80,84,79,71,65,77,83,32,98,121,32,64,100,111,116,45,97,115,109,0+.globl crypton_sha512_asm_block_data_order_shaext+.type crypton_sha512_asm_block_data_order_shaext,@function+.align 64+crypton_sha512_asm_block_data_order_shaext:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa+++ pushq %rbp+.cfi_adjust_cfa_offset 8+.cfi_offset %rbp,-16+ movq %rsp,%rbp+.cfi_def_cfa_register %rbp+.Lshaext_shortcut:++ leaq K512_nodup+128(%rip),%rcx+ vmovdqu (%rdi),%ymm0+ vmovdqu 32(%rdi),%ymm1+ vmovdqa -160(%rcx),%ymm8++ vpermq $27,%ymm0,%ymm0+ vpblendd $15,%ymm1,%ymm0,%ymm5+ vpblendd $15,%ymm0,%ymm1,%ymm6+ vpermq $225,%ymm5,%ymm5+ vpermq $75,%ymm6,%ymm6+ jmp .Loop_shaext++.align 16+.Loop_shaext:+ vmovdqu (%rsi),%ymm0+ vmovdqu 32(%rsi),%ymm1+ vmovdqu 64(%rsi),%ymm2+ vpshufb %ymm8,%ymm0,%ymm0+ vmovdqu 96(%rsi),%ymm3++ vpaddq 0-128(%rcx),%ymm0,%ymm4+ vpshufb %ymm8,%ymm1,%ymm1+ vmovdqa %ymm6,%ymm10+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+ vmovdqa %ymm5,%ymm9+.byte 196,226,79,203,236++ vpaddq 32-128(%rcx),%ymm1,%ymm4+ vpshufb %ymm8,%ymm2,%ymm2+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+ leaq 128(%rsi),%rsi+.byte 196,226,127,204,193+.byte 196,226,79,203,236++ vpaddq 64-128(%rcx),%ymm2,%ymm4+ vpshufb %ymm8,%ymm3,%ymm3+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+ vpblendd $0x03,%ymm3,%ymm2,%ymm7+ vpermq $0x39,%ymm7,%ymm7+ vpaddq %ymm7,%ymm0,%ymm0+.byte 196,226,127,204,202+.byte 196,226,79,203,236++ vpaddq 96-128(%rcx),%ymm3,%ymm4+.byte 196,226,127,205,195+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+ vpblendd $0x03,%ymm0,%ymm3,%ymm7+ vpermq $0x39,%ymm7,%ymm7+ vpaddq %ymm7,%ymm1,%ymm1+.byte 196,226,127,204,211+.byte 196,226,79,203,236+ vpaddq 128-128(%rcx),%ymm0,%ymm4+.byte 196,226,127,205,200+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+ vpblendd $0x03,%ymm1,%ymm0,%ymm7+ vpermq $0x39,%ymm7,%ymm7+ vpaddq %ymm7,%ymm2,%ymm2+.byte 196,226,127,204,216+.byte 196,226,79,203,236+ vpaddq 160-128(%rcx),%ymm1,%ymm4+.byte 196,226,127,205,209+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+ vpblendd $0x03,%ymm2,%ymm1,%ymm7+ vpermq $0x39,%ymm7,%ymm7+ vpaddq %ymm7,%ymm3,%ymm3+.byte 196,226,127,204,193+.byte 196,226,79,203,236+ vpaddq 192-128(%rcx),%ymm2,%ymm4+.byte 196,226,127,205,218+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+ vpblendd $0x03,%ymm3,%ymm2,%ymm7+ vpermq $0x39,%ymm7,%ymm7+ vpaddq %ymm7,%ymm0,%ymm0+.byte 196,226,127,204,202+.byte 196,226,79,203,236+ vpaddq 224-128(%rcx),%ymm3,%ymm4+.byte 196,226,127,205,195+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+ vpblendd $0x03,%ymm0,%ymm3,%ymm7+ vpermq $0x39,%ymm7,%ymm7+ vpaddq %ymm7,%ymm1,%ymm1+.byte 196,226,127,204,211+.byte 196,226,79,203,236+ vpaddq 256-128(%rcx),%ymm0,%ymm4+.byte 196,226,127,205,200+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+ vpblendd $0x03,%ymm1,%ymm0,%ymm7+ vpermq $0x39,%ymm7,%ymm7+ vpaddq %ymm7,%ymm2,%ymm2+.byte 196,226,127,204,216+.byte 196,226,79,203,236+ vpaddq 288-128(%rcx),%ymm1,%ymm4+.byte 196,226,127,205,209+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+ vpblendd $0x03,%ymm2,%ymm1,%ymm7+ vpermq $0x39,%ymm7,%ymm7+ vpaddq %ymm7,%ymm3,%ymm3+.byte 196,226,127,204,193+.byte 196,226,79,203,236+ vpaddq 320-128(%rcx),%ymm2,%ymm4+.byte 196,226,127,205,218+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+ vpblendd $0x03,%ymm3,%ymm2,%ymm7+ vpermq $0x39,%ymm7,%ymm7+ vpaddq %ymm7,%ymm0,%ymm0+.byte 196,226,127,204,202+.byte 196,226,79,203,236+ vpaddq 352-128(%rcx),%ymm3,%ymm4+.byte 196,226,127,205,195+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+ vpblendd $0x03,%ymm0,%ymm3,%ymm7+ vpermq $0x39,%ymm7,%ymm7+ vpaddq %ymm7,%ymm1,%ymm1+.byte 196,226,127,204,211+.byte 196,226,79,203,236+ vpaddq 384-128(%rcx),%ymm0,%ymm4+.byte 196,226,127,205,200+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+ vpblendd $0x03,%ymm1,%ymm0,%ymm7+ vpermq $0x39,%ymm7,%ymm7+ vpaddq %ymm7,%ymm2,%ymm2+.byte 196,226,127,204,216+.byte 196,226,79,203,236+ vpaddq 416-128(%rcx),%ymm1,%ymm4+.byte 196,226,127,205,209+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+ vpblendd $0x03,%ymm2,%ymm1,%ymm7+ vpermq $0x39,%ymm7,%ymm7+ vpaddq %ymm7,%ymm3,%ymm3+.byte 196,226,127,204,193+.byte 196,226,79,203,236+ vpaddq 448-128(%rcx),%ymm2,%ymm4+.byte 196,226,127,205,218+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+ vpblendd $0x03,%ymm3,%ymm2,%ymm7+ vpermq $0x39,%ymm7,%ymm7+ vpaddq %ymm7,%ymm0,%ymm0+.byte 196,226,127,204,202+.byte 196,226,79,203,236+ vpaddq 480-128(%rcx),%ymm3,%ymm4+.byte 196,226,127,205,195+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+ vpblendd $0x03,%ymm0,%ymm3,%ymm7+ vpermq $0x39,%ymm7,%ymm7+ vpaddq %ymm7,%ymm1,%ymm1+.byte 196,226,127,204,211+.byte 196,226,79,203,236+ vpaddq 512-128(%rcx),%ymm0,%ymm4+.byte 196,226,127,205,200+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+ vpblendd $0x03,%ymm1,%ymm0,%ymm7+ vpermq $0x39,%ymm7,%ymm7+ vpaddq %ymm7,%ymm2,%ymm2+.byte 196,226,127,204,216+.byte 196,226,79,203,236+ vpaddq 544-128(%rcx),%ymm1,%ymm4+.byte 196,226,127,205,209+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+ vpblendd $0x03,%ymm2,%ymm1,%ymm7+ vpermq $0x39,%ymm7,%ymm7+.byte 196,226,79,203,236+ vpaddq %ymm7,%ymm3,%ymm3++ vpaddq 576-128(%rcx),%ymm2,%ymm4+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+.byte 196,226,127,205,218+.byte 196,226,79,203,236++ vpaddq 608-128(%rcx),%ymm3,%ymm4+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+ decq %rdx+.byte 196,226,79,203,236++ vpaddq %ymm10,%ymm6,%ymm6+ vpaddq %ymm9,%ymm5,%ymm5+ jnz .Loop_shaext++ vpermq $75,%ymm5,%ymm5+ vpblendd $240,%ymm6,%ymm5,%ymm1+ vpblendd $240,%ymm5,%ymm6,%ymm2+ vpermq $180,%ymm1,%ymm1+ vpermq $27,%ymm2,%ymm2++ vmovdqu %ymm1,(%rdi)+ vmovdqu %ymm2,32(%rdi)++ vzeroupper+.cfi_def_cfa_register %rsp+ popq %rbp+.cfi_adjust_cfa_offset -8+.cfi_restore %rbp++ .byte 0xf3,0xc3+.cfi_endproc +.size crypton_sha512_asm_block_data_order_shaext,.-crypton_sha512_asm_block_data_order_shaext+.type crypton_sha512_asm_block_data_order_xop,@function+.align 64+crypton_sha512_asm_block_data_order_xop:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa+++ pushq %rbp+.cfi_adjust_cfa_offset 8+.cfi_offset %rbp,-16+ movq %rsp,%rbp+.cfi_def_cfa_register %rbp+.Lxop_shortcut:+ pushq %rbx+.cfi_offset %rbx,-24+ pushq %r12+.cfi_offset %r12,-32+ pushq %r13+.cfi_offset %r13,-40+ pushq %r14+.cfi_offset %r14,-48+ pushq %r15+.cfi_offset %r15,-56+ shlq $4,%rdx+ subq $24,%rsp++ leaq (%rsi,%rdx,8),%rdx+ movq %rdi,-64(%rbp)++ movq %rdx,-48(%rbp)+++ leaq -128(%rsp),%rsp+ vzeroupper+ andq $-64,%rsp+ movq 0(%rdi),%rax+ movq 8(%rdi),%rbx+ movq 16(%rdi),%rcx+ movq 24(%rdi),%rdx+ movq 32(%rdi),%r8+ movq 40(%rdi),%r9+ movq 48(%rdi),%r10+ movq 56(%rdi),%r11+ jmp .Lloop_xop+.align 16+.Lloop_xop:+ vmovdqa K512+1280(%rip),%xmm11+ movq %rsi,-56(%rbp)+ vmovdqu 0(%rsi),%xmm0+ vmovdqu 16(%rsi),%xmm1+ vmovdqu 32(%rsi),%xmm2+ vpshufb %xmm11,%xmm0,%xmm0+ vmovdqu 48(%rsi),%xmm3+ vpshufb %xmm11,%xmm1,%xmm1+ vmovdqu 64(%rsi),%xmm4+ vpshufb %xmm11,%xmm2,%xmm2+ vmovdqu 80(%rsi),%xmm5+ vpshufb %xmm11,%xmm3,%xmm3+ vmovdqu 96(%rsi),%xmm6+ vpshufb %xmm11,%xmm4,%xmm4+ vmovdqu 112(%rsi),%xmm7+ leaq K512+128(%rip),%rsi+ vpshufb %xmm11,%xmm5,%xmm5+ vpaddq -128(%rsi),%xmm0,%xmm8+ vpshufb %xmm11,%xmm6,%xmm6+ vpaddq -96(%rsi),%xmm1,%xmm9+ vpshufb %xmm11,%xmm7,%xmm7+ vpaddq -64(%rsi),%xmm2,%xmm10+ vpaddq -32(%rsi),%xmm3,%xmm11+ vmovdqa %xmm8,0(%rsp)+ vpaddq 0(%rsi),%xmm4,%xmm8+ vmovdqa %xmm9,16(%rsp)+ vpaddq 32(%rsi),%xmm5,%xmm9+ vmovdqa %xmm10,32(%rsp)+ vpaddq 64(%rsi),%xmm6,%xmm10+ vmovdqa %xmm11,48(%rsp)+ vpaddq 96(%rsi),%xmm7,%xmm11+ vmovdqa %xmm8,64(%rsp)+ movq %rax,%r14+ vmovdqa %xmm9,80(%rsp)+ movq %rbx,%rdi+ vmovdqa %xmm10,96(%rsp)+ xorq %rcx,%rdi+ vmovdqa %xmm11,112(%rsp)+ movq %r8,%r13+ jmp .Lxop_00_47++.align 16+.Lxop_00_47:+ addq $256,%rsi+ vpalignr $8,%xmm0,%xmm1,%xmm8+ rorq $23,%r13+ movq %r14,%rax+ vpalignr $8,%xmm4,%xmm5,%xmm11+ movq %r9,%r12+ rorq $5,%r14+.byte 143,72,120,195,200,56+ xorq %r8,%r13+ xorq %r10,%r12+ vpsrlq $7,%xmm8,%xmm8+ rorq $4,%r13+ xorq %rax,%r14+ vpaddq %xmm11,%xmm0,%xmm0+ andq %r8,%r12+ xorq %r8,%r13+ addq 0(%rsp),%r11+ movq %rax,%r15+.byte 143,72,120,195,209,7+ xorq %r10,%r12+ rorq $6,%r14+ vpxor %xmm9,%xmm8,%xmm8+ xorq %rbx,%r15+ addq %r12,%r11+ rorq $14,%r13+ andq %r15,%rdi+.byte 143,104,120,195,223,3+ xorq %rax,%r14+ addq %r13,%r11+ vpxor %xmm10,%xmm8,%xmm8+ xorq %rbx,%rdi+ rorq $28,%r14+ vpsrlq $6,%xmm7,%xmm10+ addq %r11,%rdx+ addq %rdi,%r11+ vpaddq %xmm8,%xmm0,%xmm0+ movq %rdx,%r13+ addq %r11,%r14+.byte 143,72,120,195,203,42+ rorq $23,%r13+ movq %r14,%r11+ vpxor %xmm10,%xmm11,%xmm11+ movq %r8,%r12+ rorq $5,%r14+ xorq %rdx,%r13+ xorq %r9,%r12+ vpxor %xmm9,%xmm11,%xmm11+ rorq $4,%r13+ xorq %r11,%r14+ andq %rdx,%r12+ xorq %rdx,%r13+ vpaddq %xmm11,%xmm0,%xmm0+ addq 8(%rsp),%r10+ movq %r11,%rdi+ xorq %r9,%r12+ rorq $6,%r14+ vpaddq -128(%rsi),%xmm0,%xmm10+ xorq %rax,%rdi+ addq %r12,%r10+ rorq $14,%r13+ andq %rdi,%r15+ xorq %r11,%r14+ addq %r13,%r10+ xorq %rax,%r15+ rorq $28,%r14+ addq %r10,%rcx+ addq %r15,%r10+ movq %rcx,%r13+ addq %r10,%r14+ vmovdqa %xmm10,0(%rsp)+ vpalignr $8,%xmm1,%xmm2,%xmm8+ rorq $23,%r13+ movq %r14,%r10+ vpalignr $8,%xmm5,%xmm6,%xmm11+ movq %rdx,%r12+ rorq $5,%r14+.byte 143,72,120,195,200,56+ xorq %rcx,%r13+ xorq %r8,%r12+ vpsrlq $7,%xmm8,%xmm8+ rorq $4,%r13+ xorq %r10,%r14+ vpaddq %xmm11,%xmm1,%xmm1+ andq %rcx,%r12+ xorq %rcx,%r13+ addq 16(%rsp),%r9+ movq %r10,%r15+.byte 143,72,120,195,209,7+ xorq %r8,%r12+ rorq $6,%r14+ vpxor %xmm9,%xmm8,%xmm8+ xorq %r11,%r15+ addq %r12,%r9+ rorq $14,%r13+ andq %r15,%rdi+.byte 143,104,120,195,216,3+ xorq %r10,%r14+ addq %r13,%r9+ vpxor %xmm10,%xmm8,%xmm8+ xorq %r11,%rdi+ rorq $28,%r14+ vpsrlq $6,%xmm0,%xmm10+ addq %r9,%rbx+ addq %rdi,%r9+ vpaddq %xmm8,%xmm1,%xmm1+ movq %rbx,%r13+ addq %r9,%r14+.byte 143,72,120,195,203,42+ rorq $23,%r13+ movq %r14,%r9+ vpxor %xmm10,%xmm11,%xmm11+ movq %rcx,%r12+ rorq $5,%r14+ xorq %rbx,%r13+ xorq %rdx,%r12+ vpxor %xmm9,%xmm11,%xmm11+ rorq $4,%r13+ xorq %r9,%r14+ andq %rbx,%r12+ xorq %rbx,%r13+ vpaddq %xmm11,%xmm1,%xmm1+ addq 24(%rsp),%r8+ movq %r9,%rdi+ xorq %rdx,%r12+ rorq $6,%r14+ vpaddq -96(%rsi),%xmm1,%xmm10+ xorq %r10,%rdi+ addq %r12,%r8+ rorq $14,%r13+ andq %rdi,%r15+ xorq %r9,%r14+ addq %r13,%r8+ xorq %r10,%r15+ rorq $28,%r14+ addq %r8,%rax+ addq %r15,%r8+ movq %rax,%r13+ addq %r8,%r14+ vmovdqa %xmm10,16(%rsp)+ vpalignr $8,%xmm2,%xmm3,%xmm8+ rorq $23,%r13+ movq %r14,%r8+ vpalignr $8,%xmm6,%xmm7,%xmm11+ movq %rbx,%r12+ rorq $5,%r14+.byte 143,72,120,195,200,56+ xorq %rax,%r13+ xorq %rcx,%r12+ vpsrlq $7,%xmm8,%xmm8+ rorq $4,%r13+ xorq %r8,%r14+ vpaddq %xmm11,%xmm2,%xmm2+ andq %rax,%r12+ xorq %rax,%r13+ addq 32(%rsp),%rdx+ movq %r8,%r15+.byte 143,72,120,195,209,7+ xorq %rcx,%r12+ rorq $6,%r14+ vpxor %xmm9,%xmm8,%xmm8+ xorq %r9,%r15+ addq %r12,%rdx+ rorq $14,%r13+ andq %r15,%rdi+.byte 143,104,120,195,217,3+ xorq %r8,%r14+ addq %r13,%rdx+ vpxor %xmm10,%xmm8,%xmm8+ xorq %r9,%rdi+ rorq $28,%r14+ vpsrlq $6,%xmm1,%xmm10+ addq %rdx,%r11+ addq %rdi,%rdx+ vpaddq %xmm8,%xmm2,%xmm2+ movq %r11,%r13+ addq %rdx,%r14+.byte 143,72,120,195,203,42+ rorq $23,%r13+ movq %r14,%rdx+ vpxor %xmm10,%xmm11,%xmm11+ movq %rax,%r12+ rorq $5,%r14+ xorq %r11,%r13+ xorq %rbx,%r12+ vpxor %xmm9,%xmm11,%xmm11+ rorq $4,%r13+ xorq %rdx,%r14+ andq %r11,%r12+ xorq %r11,%r13+ vpaddq %xmm11,%xmm2,%xmm2+ addq 40(%rsp),%rcx+ movq %rdx,%rdi+ xorq %rbx,%r12+ rorq $6,%r14+ vpaddq -64(%rsi),%xmm2,%xmm10+ xorq %r8,%rdi+ addq %r12,%rcx+ rorq $14,%r13+ andq %rdi,%r15+ xorq %rdx,%r14+ addq %r13,%rcx+ xorq %r8,%r15+ rorq $28,%r14+ addq %rcx,%r10+ addq %r15,%rcx+ movq %r10,%r13+ addq %rcx,%r14+ vmovdqa %xmm10,32(%rsp)+ vpalignr $8,%xmm3,%xmm4,%xmm8+ rorq $23,%r13+ movq %r14,%rcx+ vpalignr $8,%xmm7,%xmm0,%xmm11+ movq %r11,%r12+ rorq $5,%r14+.byte 143,72,120,195,200,56+ xorq %r10,%r13+ xorq %rax,%r12+ vpsrlq $7,%xmm8,%xmm8+ rorq $4,%r13+ xorq %rcx,%r14+ vpaddq %xmm11,%xmm3,%xmm3+ andq %r10,%r12+ xorq %r10,%r13+ addq 48(%rsp),%rbx+ movq %rcx,%r15+.byte 143,72,120,195,209,7+ xorq %rax,%r12+ rorq $6,%r14+ vpxor %xmm9,%xmm8,%xmm8+ xorq %rdx,%r15+ addq %r12,%rbx+ rorq $14,%r13+ andq %r15,%rdi+.byte 143,104,120,195,218,3+ xorq %rcx,%r14+ addq %r13,%rbx+ vpxor %xmm10,%xmm8,%xmm8+ xorq %rdx,%rdi+ rorq $28,%r14+ vpsrlq $6,%xmm2,%xmm10+ addq %rbx,%r9+ addq %rdi,%rbx+ vpaddq %xmm8,%xmm3,%xmm3+ movq %r9,%r13+ addq %rbx,%r14+.byte 143,72,120,195,203,42+ rorq $23,%r13+ movq %r14,%rbx+ vpxor %xmm10,%xmm11,%xmm11+ movq %r10,%r12+ rorq $5,%r14+ xorq %r9,%r13+ xorq %r11,%r12+ vpxor %xmm9,%xmm11,%xmm11+ rorq $4,%r13+ xorq %rbx,%r14+ andq %r9,%r12+ xorq %r9,%r13+ vpaddq %xmm11,%xmm3,%xmm3+ addq 56(%rsp),%rax+ movq %rbx,%rdi+ xorq %r11,%r12+ rorq $6,%r14+ vpaddq -32(%rsi),%xmm3,%xmm10+ xorq %rcx,%rdi+ addq %r12,%rax+ rorq $14,%r13+ andq %rdi,%r15+ xorq %rbx,%r14+ addq %r13,%rax+ xorq %rcx,%r15+ rorq $28,%r14+ addq %rax,%r8+ addq %r15,%rax+ movq %r8,%r13+ addq %rax,%r14+ vmovdqa %xmm10,48(%rsp)+ vpalignr $8,%xmm4,%xmm5,%xmm8+ rorq $23,%r13+ movq %r14,%rax+ vpalignr $8,%xmm0,%xmm1,%xmm11+ movq %r9,%r12+ rorq $5,%r14+.byte 143,72,120,195,200,56+ xorq %r8,%r13+ xorq %r10,%r12+ vpsrlq $7,%xmm8,%xmm8+ rorq $4,%r13+ xorq %rax,%r14+ vpaddq %xmm11,%xmm4,%xmm4+ andq %r8,%r12+ xorq %r8,%r13+ addq 64(%rsp),%r11+ movq %rax,%r15+.byte 143,72,120,195,209,7+ xorq %r10,%r12+ rorq $6,%r14+ vpxor %xmm9,%xmm8,%xmm8+ xorq %rbx,%r15+ addq %r12,%r11+ rorq $14,%r13+ andq %r15,%rdi+.byte 143,104,120,195,219,3+ xorq %rax,%r14+ addq %r13,%r11+ vpxor %xmm10,%xmm8,%xmm8+ xorq %rbx,%rdi+ rorq $28,%r14+ vpsrlq $6,%xmm3,%xmm10+ addq %r11,%rdx+ addq %rdi,%r11+ vpaddq %xmm8,%xmm4,%xmm4+ movq %rdx,%r13+ addq %r11,%r14+.byte 143,72,120,195,203,42+ rorq $23,%r13+ movq %r14,%r11+ vpxor %xmm10,%xmm11,%xmm11+ movq %r8,%r12+ rorq $5,%r14+ xorq %rdx,%r13+ xorq %r9,%r12+ vpxor %xmm9,%xmm11,%xmm11+ rorq $4,%r13+ xorq %r11,%r14+ andq %rdx,%r12+ xorq %rdx,%r13+ vpaddq %xmm11,%xmm4,%xmm4+ addq 72(%rsp),%r10+ movq %r11,%rdi+ xorq %r9,%r12+ rorq $6,%r14+ vpaddq 0(%rsi),%xmm4,%xmm10+ xorq %rax,%rdi+ addq %r12,%r10+ rorq $14,%r13+ andq %rdi,%r15+ xorq %r11,%r14+ addq %r13,%r10+ xorq %rax,%r15+ rorq $28,%r14+ addq %r10,%rcx+ addq %r15,%r10+ movq %rcx,%r13+ addq %r10,%r14+ vmovdqa %xmm10,64(%rsp)+ vpalignr $8,%xmm5,%xmm6,%xmm8+ rorq $23,%r13+ movq %r14,%r10+ vpalignr $8,%xmm1,%xmm2,%xmm11+ movq %rdx,%r12+ rorq $5,%r14+.byte 143,72,120,195,200,56+ xorq %rcx,%r13+ xorq %r8,%r12+ vpsrlq $7,%xmm8,%xmm8+ rorq $4,%r13+ xorq %r10,%r14+ vpaddq %xmm11,%xmm5,%xmm5+ andq %rcx,%r12+ xorq %rcx,%r13+ addq 80(%rsp),%r9+ movq %r10,%r15+.byte 143,72,120,195,209,7+ xorq %r8,%r12+ rorq $6,%r14+ vpxor %xmm9,%xmm8,%xmm8+ xorq %r11,%r15+ addq %r12,%r9+ rorq $14,%r13+ andq %r15,%rdi+.byte 143,104,120,195,220,3+ xorq %r10,%r14+ addq %r13,%r9+ vpxor %xmm10,%xmm8,%xmm8+ xorq %r11,%rdi+ rorq $28,%r14+ vpsrlq $6,%xmm4,%xmm10+ addq %r9,%rbx+ addq %rdi,%r9+ vpaddq %xmm8,%xmm5,%xmm5+ movq %rbx,%r13+ addq %r9,%r14+.byte 143,72,120,195,203,42+ rorq $23,%r13+ movq %r14,%r9+ vpxor %xmm10,%xmm11,%xmm11+ movq %rcx,%r12+ rorq $5,%r14+ xorq %rbx,%r13+ xorq %rdx,%r12+ vpxor %xmm9,%xmm11,%xmm11+ rorq $4,%r13+ xorq %r9,%r14+ andq %rbx,%r12+ xorq %rbx,%r13+ vpaddq %xmm11,%xmm5,%xmm5+ addq 88(%rsp),%r8+ movq %r9,%rdi+ xorq %rdx,%r12+ rorq $6,%r14+ vpaddq 32(%rsi),%xmm5,%xmm10+ xorq %r10,%rdi+ addq %r12,%r8+ rorq $14,%r13+ andq %rdi,%r15+ xorq %r9,%r14+ addq %r13,%r8+ xorq %r10,%r15+ rorq $28,%r14+ addq %r8,%rax+ addq %r15,%r8+ movq %rax,%r13+ addq %r8,%r14+ vmovdqa %xmm10,80(%rsp)+ vpalignr $8,%xmm6,%xmm7,%xmm8+ rorq $23,%r13+ movq %r14,%r8+ vpalignr $8,%xmm2,%xmm3,%xmm11+ movq %rbx,%r12+ rorq $5,%r14+.byte 143,72,120,195,200,56+ xorq %rax,%r13+ xorq %rcx,%r12+ vpsrlq $7,%xmm8,%xmm8+ rorq $4,%r13+ xorq %r8,%r14+ vpaddq %xmm11,%xmm6,%xmm6+ andq %rax,%r12+ xorq %rax,%r13+ addq 96(%rsp),%rdx+ movq %r8,%r15+.byte 143,72,120,195,209,7+ xorq %rcx,%r12+ rorq $6,%r14+ vpxor %xmm9,%xmm8,%xmm8+ xorq %r9,%r15+ addq %r12,%rdx+ rorq $14,%r13+ andq %r15,%rdi+.byte 143,104,120,195,221,3+ xorq %r8,%r14+ addq %r13,%rdx+ vpxor %xmm10,%xmm8,%xmm8+ xorq %r9,%rdi+ rorq $28,%r14+ vpsrlq $6,%xmm5,%xmm10+ addq %rdx,%r11+ addq %rdi,%rdx+ vpaddq %xmm8,%xmm6,%xmm6+ movq %r11,%r13+ addq %rdx,%r14+.byte 143,72,120,195,203,42+ rorq $23,%r13+ movq %r14,%rdx+ vpxor %xmm10,%xmm11,%xmm11+ movq %rax,%r12+ rorq $5,%r14+ xorq %r11,%r13+ xorq %rbx,%r12+ vpxor %xmm9,%xmm11,%xmm11+ rorq $4,%r13+ xorq %rdx,%r14+ andq %r11,%r12+ xorq %r11,%r13+ vpaddq %xmm11,%xmm6,%xmm6+ addq 104(%rsp),%rcx+ movq %rdx,%rdi+ xorq %rbx,%r12+ rorq $6,%r14+ vpaddq 64(%rsi),%xmm6,%xmm10+ xorq %r8,%rdi+ addq %r12,%rcx+ rorq $14,%r13+ andq %rdi,%r15+ xorq %rdx,%r14+ addq %r13,%rcx+ xorq %r8,%r15+ rorq $28,%r14+ addq %rcx,%r10+ addq %r15,%rcx+ movq %r10,%r13+ addq %rcx,%r14+ vmovdqa %xmm10,96(%rsp)+ vpalignr $8,%xmm7,%xmm0,%xmm8+ rorq $23,%r13+ movq %r14,%rcx+ vpalignr $8,%xmm3,%xmm4,%xmm11+ movq %r11,%r12+ rorq $5,%r14+.byte 143,72,120,195,200,56+ xorq %r10,%r13+ xorq %rax,%r12+ vpsrlq $7,%xmm8,%xmm8+ rorq $4,%r13+ xorq %rcx,%r14+ vpaddq %xmm11,%xmm7,%xmm7+ andq %r10,%r12+ xorq %r10,%r13+ addq 112(%rsp),%rbx+ movq %rcx,%r15+.byte 143,72,120,195,209,7+ xorq %rax,%r12+ rorq $6,%r14+ vpxor %xmm9,%xmm8,%xmm8+ xorq %rdx,%r15+ addq %r12,%rbx+ rorq $14,%r13+ andq %r15,%rdi+.byte 143,104,120,195,222,3+ xorq %rcx,%r14+ addq %r13,%rbx+ vpxor %xmm10,%xmm8,%xmm8+ xorq %rdx,%rdi+ rorq $28,%r14+ vpsrlq $6,%xmm6,%xmm10+ addq %rbx,%r9+ addq %rdi,%rbx+ vpaddq %xmm8,%xmm7,%xmm7+ movq %r9,%r13+ addq %rbx,%r14+.byte 143,72,120,195,203,42+ rorq $23,%r13+ movq %r14,%rbx+ vpxor %xmm10,%xmm11,%xmm11+ movq %r10,%r12+ rorq $5,%r14+ xorq %r9,%r13+ xorq %r11,%r12+ vpxor %xmm9,%xmm11,%xmm11+ rorq $4,%r13+ xorq %rbx,%r14+ andq %r9,%r12+ xorq %r9,%r13+ vpaddq %xmm11,%xmm7,%xmm7+ addq 120(%rsp),%rax+ movq %rbx,%rdi+ xorq %r11,%r12+ rorq $6,%r14+ vpaddq 96(%rsi),%xmm7,%xmm10+ xorq %rcx,%rdi+ addq %r12,%rax+ rorq $14,%r13+ andq %rdi,%r15+ xorq %rbx,%r14+ addq %r13,%rax+ xorq %rcx,%r15+ rorq $28,%r14+ addq %rax,%r8+ addq %r15,%rax+ movq %r8,%r13+ addq %rax,%r14+ vmovdqa %xmm10,112(%rsp)+ cmpb $0,135(%rsi)+ jne .Lxop_00_47+ rorq $23,%r13+ movq %r14,%rax+ movq %r9,%r12+ rorq $5,%r14+ xorq %r8,%r13+ xorq %r10,%r12+ rorq $4,%r13+ xorq %rax,%r14+ andq %r8,%r12+ xorq %r8,%r13+ addq 0(%rsp),%r11+ movq %rax,%r15+ xorq %r10,%r12+ rorq $6,%r14+ xorq %rbx,%r15+ addq %r12,%r11+ rorq $14,%r13+ andq %r15,%rdi+ xorq %rax,%r14+ addq %r13,%r11+ xorq %rbx,%rdi+ rorq $28,%r14+ addq %r11,%rdx+ addq %rdi,%r11+ movq %rdx,%r13+ addq %r11,%r14+ rorq $23,%r13+ movq %r14,%r11+ movq %r8,%r12+ rorq $5,%r14+ xorq %rdx,%r13+ xorq %r9,%r12+ rorq $4,%r13+ xorq %r11,%r14+ andq %rdx,%r12+ xorq %rdx,%r13+ addq 8(%rsp),%r10+ movq %r11,%rdi+ xorq %r9,%r12+ rorq $6,%r14+ xorq %rax,%rdi+ addq %r12,%r10+ rorq $14,%r13+ andq %rdi,%r15+ xorq %r11,%r14+ addq %r13,%r10+ xorq %rax,%r15+ rorq $28,%r14+ addq %r10,%rcx+ addq %r15,%r10+ movq %rcx,%r13+ addq %r10,%r14+ rorq $23,%r13+ movq %r14,%r10+ movq %rdx,%r12+ rorq $5,%r14+ xorq %rcx,%r13+ xorq %r8,%r12+ rorq $4,%r13+ xorq %r10,%r14+ andq %rcx,%r12+ xorq %rcx,%r13+ addq 16(%rsp),%r9+ movq %r10,%r15+ xorq %r8,%r12+ rorq $6,%r14+ xorq %r11,%r15+ addq %r12,%r9+ rorq $14,%r13+ andq %r15,%rdi+ xorq %r10,%r14+ addq %r13,%r9+ xorq %r11,%rdi+ rorq $28,%r14+ addq %r9,%rbx+ addq %rdi,%r9+ movq %rbx,%r13+ addq %r9,%r14+ rorq $23,%r13+ movq %r14,%r9+ movq %rcx,%r12+ rorq $5,%r14+ xorq %rbx,%r13+ xorq %rdx,%r12+ rorq $4,%r13+ xorq %r9,%r14+ andq %rbx,%r12+ xorq %rbx,%r13+ addq 24(%rsp),%r8+ movq %r9,%rdi+ xorq %rdx,%r12+ rorq $6,%r14+ xorq %r10,%rdi+ addq %r12,%r8+ rorq $14,%r13+ andq %rdi,%r15+ xorq %r9,%r14+ addq %r13,%r8+ xorq %r10,%r15+ rorq $28,%r14+ addq %r8,%rax+ addq %r15,%r8+ movq %rax,%r13+ addq %r8,%r14+ rorq $23,%r13+ movq %r14,%r8+ movq %rbx,%r12+ rorq $5,%r14+ xorq %rax,%r13+ xorq %rcx,%r12+ rorq $4,%r13+ xorq %r8,%r14+ andq %rax,%r12+ xorq %rax,%r13+ addq 32(%rsp),%rdx+ movq %r8,%r15+ xorq %rcx,%r12+ rorq $6,%r14+ xorq %r9,%r15+ addq %r12,%rdx+ rorq $14,%r13+ andq %r15,%rdi+ xorq %r8,%r14+ addq %r13,%rdx+ xorq %r9,%rdi+ rorq $28,%r14+ addq %rdx,%r11+ addq %rdi,%rdx+ movq %r11,%r13+ addq %rdx,%r14+ rorq $23,%r13+ movq %r14,%rdx+ movq %rax,%r12+ rorq $5,%r14+ xorq %r11,%r13+ xorq %rbx,%r12+ rorq $4,%r13+ xorq %rdx,%r14+ andq %r11,%r12+ xorq %r11,%r13+ addq 40(%rsp),%rcx+ movq %rdx,%rdi+ xorq %rbx,%r12+ rorq $6,%r14+ xorq %r8,%rdi+ addq %r12,%rcx+ rorq $14,%r13+ andq %rdi,%r15+ xorq %rdx,%r14+ addq %r13,%rcx+ xorq %r8,%r15+ rorq $28,%r14+ addq %rcx,%r10+ addq %r15,%rcx+ movq %r10,%r13+ addq %rcx,%r14+ rorq $23,%r13+ movq %r14,%rcx+ movq %r11,%r12+ rorq $5,%r14+ xorq %r10,%r13+ xorq %rax,%r12+ rorq $4,%r13+ xorq %rcx,%r14+ andq %r10,%r12+ xorq %r10,%r13+ addq 48(%rsp),%rbx+ movq %rcx,%r15+ xorq %rax,%r12+ rorq $6,%r14+ xorq %rdx,%r15+ addq %r12,%rbx+ rorq $14,%r13+ andq %r15,%rdi+ xorq %rcx,%r14+ addq %r13,%rbx+ xorq %rdx,%rdi+ rorq $28,%r14+ addq %rbx,%r9+ addq %rdi,%rbx+ movq %r9,%r13+ addq %rbx,%r14+ rorq $23,%r13+ movq %r14,%rbx+ movq %r10,%r12+ rorq $5,%r14+ xorq %r9,%r13+ xorq %r11,%r12+ rorq $4,%r13+ xorq %rbx,%r14+ andq %r9,%r12+ xorq %r9,%r13+ addq 56(%rsp),%rax+ movq %rbx,%rdi+ xorq %r11,%r12+ rorq $6,%r14+ xorq %rcx,%rdi+ addq %r12,%rax+ rorq $14,%r13+ andq %rdi,%r15+ xorq %rbx,%r14+ addq %r13,%rax+ xorq %rcx,%r15+ rorq $28,%r14+ addq %rax,%r8+ addq %r15,%rax+ movq %r8,%r13+ addq %rax,%r14+ rorq $23,%r13+ movq %r14,%rax+ movq %r9,%r12+ rorq $5,%r14+ xorq %r8,%r13+ xorq %r10,%r12+ rorq $4,%r13+ xorq %rax,%r14+ andq %r8,%r12+ xorq %r8,%r13+ addq 64(%rsp),%r11+ movq %rax,%r15+ xorq %r10,%r12+ rorq $6,%r14+ xorq %rbx,%r15+ addq %r12,%r11+ rorq $14,%r13+ andq %r15,%rdi+ xorq %rax,%r14+ addq %r13,%r11+ xorq %rbx,%rdi+ rorq $28,%r14+ addq %r11,%rdx+ addq %rdi,%r11+ movq %rdx,%r13+ addq %r11,%r14+ rorq $23,%r13+ movq %r14,%r11+ movq %r8,%r12+ rorq $5,%r14+ xorq %rdx,%r13+ xorq %r9,%r12+ rorq $4,%r13+ xorq %r11,%r14+ andq %rdx,%r12+ xorq %rdx,%r13+ addq 72(%rsp),%r10+ movq %r11,%rdi+ xorq %r9,%r12+ rorq $6,%r14+ xorq %rax,%rdi+ addq %r12,%r10+ rorq $14,%r13+ andq %rdi,%r15+ xorq %r11,%r14+ addq %r13,%r10+ xorq %rax,%r15+ rorq $28,%r14+ addq %r10,%rcx+ addq %r15,%r10+ movq %rcx,%r13+ addq %r10,%r14+ rorq $23,%r13+ movq %r14,%r10+ movq %rdx,%r12+ rorq $5,%r14+ xorq %rcx,%r13+ xorq %r8,%r12+ rorq $4,%r13+ xorq %r10,%r14+ andq %rcx,%r12+ xorq %rcx,%r13+ addq 80(%rsp),%r9+ movq %r10,%r15+ xorq %r8,%r12+ rorq $6,%r14+ xorq %r11,%r15+ addq %r12,%r9+ rorq $14,%r13+ andq %r15,%rdi+ xorq %r10,%r14+ addq %r13,%r9+ xorq %r11,%rdi+ rorq $28,%r14+ addq %r9,%rbx+ addq %rdi,%r9+ movq %rbx,%r13+ addq %r9,%r14+ rorq $23,%r13+ movq %r14,%r9+ movq %rcx,%r12+ rorq $5,%r14+ xorq %rbx,%r13+ xorq %rdx,%r12+ rorq $4,%r13+ xorq %r9,%r14+ andq %rbx,%r12+ xorq %rbx,%r13+ addq 88(%rsp),%r8+ movq %r9,%rdi+ xorq %rdx,%r12+ rorq $6,%r14+ xorq %r10,%rdi+ addq %r12,%r8+ rorq $14,%r13+ andq %rdi,%r15+ xorq %r9,%r14+ addq %r13,%r8+ xorq %r10,%r15+ rorq $28,%r14+ addq %r8,%rax+ addq %r15,%r8+ movq %rax,%r13+ addq %r8,%r14+ rorq $23,%r13+ movq %r14,%r8+ movq %rbx,%r12+ rorq $5,%r14+ xorq %rax,%r13+ xorq %rcx,%r12+ rorq $4,%r13+ xorq %r8,%r14+ andq %rax,%r12+ xorq %rax,%r13+ addq 96(%rsp),%rdx+ movq %r8,%r15+ xorq %rcx,%r12+ rorq $6,%r14+ xorq %r9,%r15+ addq %r12,%rdx+ rorq $14,%r13+ andq %r15,%rdi+ xorq %r8,%r14+ addq %r13,%rdx+ xorq %r9,%rdi+ rorq $28,%r14+ addq %rdx,%r11+ addq %rdi,%rdx+ movq %r11,%r13+ addq %rdx,%r14+ rorq $23,%r13+ movq %r14,%rdx+ movq %rax,%r12+ rorq $5,%r14+ xorq %r11,%r13+ xorq %rbx,%r12+ rorq $4,%r13+ xorq %rdx,%r14+ andq %r11,%r12+ xorq %r11,%r13+ addq 104(%rsp),%rcx+ movq %rdx,%rdi+ xorq %rbx,%r12+ rorq $6,%r14+ xorq %r8,%rdi+ addq %r12,%rcx+ rorq $14,%r13+ andq %rdi,%r15+ xorq %rdx,%r14+ addq %r13,%rcx+ xorq %r8,%r15+ rorq $28,%r14+ addq %rcx,%r10+ addq %r15,%rcx+ movq %r10,%r13+ addq %rcx,%r14+ rorq $23,%r13+ movq %r14,%rcx+ movq %r11,%r12+ rorq $5,%r14+ xorq %r10,%r13+ xorq %rax,%r12+ rorq $4,%r13+ xorq %rcx,%r14+ andq %r10,%r12+ xorq %r10,%r13+ addq 112(%rsp),%rbx+ movq %rcx,%r15+ xorq %rax,%r12+ rorq $6,%r14+ xorq %rdx,%r15+ addq %r12,%rbx+ rorq $14,%r13+ andq %r15,%rdi+ xorq %rcx,%r14+ addq %r13,%rbx+ xorq %rdx,%rdi+ rorq $28,%r14+ addq %rbx,%r9+ addq %rdi,%rbx+ movq %r9,%r13+ addq %rbx,%r14+ rorq $23,%r13+ movq %r14,%rbx+ movq %r10,%r12+ rorq $5,%r14+ xorq %r9,%r13+ xorq %r11,%r12+ rorq $4,%r13+ xorq %rbx,%r14+ andq %r9,%r12+ xorq %r9,%r13+ addq 120(%rsp),%rax+ movq %rbx,%rdi+ xorq %r11,%r12+ rorq $6,%r14+ xorq %rcx,%rdi+ addq %r12,%rax+ rorq $14,%r13+ andq %rdi,%r15+ xorq %rbx,%r14+ addq %r13,%rax+ xorq %rcx,%r15+ rorq $28,%r14+ addq %rax,%r8+ addq %r15,%rax+ movq %r8,%r13+ addq %rax,%r14+ movq -64(%rbp),%rdi+ movq %r14,%rax+ movq -56(%rbp),%rsi++ addq 0(%rdi),%rax+ addq 8(%rdi),%rbx+ addq 16(%rdi),%rcx+ addq 24(%rdi),%rdx+ addq 32(%rdi),%r8+ addq 40(%rdi),%r9+ addq 48(%rdi),%r10+ addq 56(%rdi),%r11++ leaq 128(%rsi),%rsi+ cmpq -48(%rbp),%rsi++ movq %rax,0(%rdi)+ movq %rbx,8(%rdi)+ movq %rcx,16(%rdi)+ movq %rdx,24(%rdi)+ movq %r8,32(%rdi)+ movq %r9,40(%rdi)+ movq %r10,48(%rdi)+ movq %r11,56(%rdi)+ jb .Lloop_xop++ vzeroupper+ movq -40(%rbp),%r15+ movq -32(%rbp),%r14+ movq -24(%rbp),%r13+ movq -16(%rbp),%r12+ movq -8(%rbp),%rbx+ movq %rbp,%rsp+.cfi_def_cfa_register %rsp+ popq %rbp+.cfi_adjust_cfa_offset -8+.cfi_restore %rbp+.cfi_restore %r12+.cfi_restore %r13+.cfi_restore %r14+.cfi_restore %r15+.cfi_restore %rbx+ .byte 0xf3,0xc3+.cfi_endproc +.size crypton_sha512_asm_block_data_order_xop,.-crypton_sha512_asm_block_data_order_xop+.type crypton_sha512_asm_block_data_order_avx,@function+.align 64+crypton_sha512_asm_block_data_order_avx:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa+++ pushq %rbp+.cfi_adjust_cfa_offset 8+.cfi_offset %rbp,-16+ movq %rsp,%rbp+.cfi_def_cfa_register %rbp+.Lavx_shortcut:+ pushq %rbx+.cfi_offset %rbx,-24+ pushq %r12+.cfi_offset %r12,-32+ pushq %r13+.cfi_offset %r13,-40+ pushq %r14+.cfi_offset %r14,-48+ pushq %r15+.cfi_offset %r15,-56+ shlq $4,%rdx+ subq $24,%rsp++ leaq (%rsi,%rdx,8),%rdx+ movq %rdi,-64(%rbp)++ movq %rdx,-48(%rbp)+++ leaq -128(%rsp),%rsp+ vzeroupper+ andq $-64,%rsp+ movq 0(%rdi),%rax+ movq 8(%rdi),%rbx+ movq 16(%rdi),%rcx+ movq 24(%rdi),%rdx+ movq 32(%rdi),%r8+ movq 40(%rdi),%r9+ movq 48(%rdi),%r10+ movq 56(%rdi),%r11+ jmp .Lloop_avx+.align 16+.Lloop_avx:+ vmovdqa K512+1280(%rip),%xmm11+ movq %rsi,-56(%rbp)+ vmovdqu 0(%rsi),%xmm0+ vmovdqu 16(%rsi),%xmm1+ vmovdqu 32(%rsi),%xmm2+ vpshufb %xmm11,%xmm0,%xmm0+ vmovdqu 48(%rsi),%xmm3+ vpshufb %xmm11,%xmm1,%xmm1+ vmovdqu 64(%rsi),%xmm4+ vpshufb %xmm11,%xmm2,%xmm2+ vmovdqu 80(%rsi),%xmm5+ vpshufb %xmm11,%xmm3,%xmm3+ vmovdqu 96(%rsi),%xmm6+ vpshufb %xmm11,%xmm4,%xmm4+ vmovdqu 112(%rsi),%xmm7+ leaq K512+128(%rip),%rsi+ vpshufb %xmm11,%xmm5,%xmm5+ vpaddq -128(%rsi),%xmm0,%xmm8+ vpshufb %xmm11,%xmm6,%xmm6+ vpaddq -96(%rsi),%xmm1,%xmm9+ vpshufb %xmm11,%xmm7,%xmm7+ vpaddq -64(%rsi),%xmm2,%xmm10+ vpaddq -32(%rsi),%xmm3,%xmm11+ vmovdqa %xmm8,0(%rsp)+ vpaddq 0(%rsi),%xmm4,%xmm8+ vmovdqa %xmm9,16(%rsp)+ vpaddq 32(%rsi),%xmm5,%xmm9+ vmovdqa %xmm10,32(%rsp)+ vpaddq 64(%rsi),%xmm6,%xmm10+ vmovdqa %xmm11,48(%rsp)+ vpaddq 96(%rsi),%xmm7,%xmm11+ vmovdqa %xmm8,64(%rsp)+ movq %rax,%r14+ vmovdqa %xmm9,80(%rsp)+ movq %rbx,%rdi+ vmovdqa %xmm10,96(%rsp)+ xorq %rcx,%rdi+ vmovdqa %xmm11,112(%rsp)+ movq %r8,%r13+ jmp .Lavx_00_47++.align 16+.Lavx_00_47:+ addq $256,%rsi+ vpalignr $8,%xmm0,%xmm1,%xmm8+ shrdq $23,%r13,%r13+ movq %r14,%rax+ vpalignr $8,%xmm4,%xmm5,%xmm11+ movq %r9,%r12+ shrdq $5,%r14,%r14+ vpsrlq $1,%xmm8,%xmm10+ xorq %r8,%r13+ xorq %r10,%r12+ vpaddq %xmm11,%xmm0,%xmm0+ shrdq $4,%r13,%r13+ xorq %rax,%r14+ vpsrlq $7,%xmm8,%xmm11+ andq %r8,%r12+ xorq %r8,%r13+ vpsllq $56,%xmm8,%xmm9+ addq 0(%rsp),%r11+ movq %rax,%r15+ vpxor %xmm10,%xmm11,%xmm8+ xorq %r10,%r12+ shrdq $6,%r14,%r14+ vpsrlq $7,%xmm10,%xmm10+ xorq %rbx,%r15+ addq %r12,%r11+ vpxor %xmm9,%xmm8,%xmm8+ shrdq $14,%r13,%r13+ andq %r15,%rdi+ vpsllq $7,%xmm9,%xmm9+ xorq %rax,%r14+ addq %r13,%r11+ vpxor %xmm10,%xmm8,%xmm8+ xorq %rbx,%rdi+ shrdq $28,%r14,%r14+ vpsrlq $6,%xmm7,%xmm11+ addq %r11,%rdx+ addq %rdi,%r11+ vpxor %xmm9,%xmm8,%xmm8+ movq %rdx,%r13+ addq %r11,%r14+ vpsllq $3,%xmm7,%xmm10+ shrdq $23,%r13,%r13+ movq %r14,%r11+ vpaddq %xmm8,%xmm0,%xmm0+ movq %r8,%r12+ shrdq $5,%r14,%r14+ vpsrlq $19,%xmm7,%xmm9+ xorq %rdx,%r13+ xorq %r9,%r12+ vpxor %xmm10,%xmm11,%xmm11+ shrdq $4,%r13,%r13+ xorq %r11,%r14+ vpsllq $42,%xmm10,%xmm10+ andq %rdx,%r12+ xorq %rdx,%r13+ vpxor %xmm9,%xmm11,%xmm11+ addq 8(%rsp),%r10+ movq %r11,%rdi+ vpsrlq $42,%xmm9,%xmm9+ xorq %r9,%r12+ shrdq $6,%r14,%r14+ vpxor %xmm10,%xmm11,%xmm11+ xorq %rax,%rdi+ addq %r12,%r10+ vpxor %xmm9,%xmm11,%xmm11+ shrdq $14,%r13,%r13+ andq %rdi,%r15+ vpaddq %xmm11,%xmm0,%xmm0+ xorq %r11,%r14+ addq %r13,%r10+ vpaddq -128(%rsi),%xmm0,%xmm10+ xorq %rax,%r15+ shrdq $28,%r14,%r14+ addq %r10,%rcx+ addq %r15,%r10+ movq %rcx,%r13+ addq %r10,%r14+ vmovdqa %xmm10,0(%rsp)+ vpalignr $8,%xmm1,%xmm2,%xmm8+ shrdq $23,%r13,%r13+ movq %r14,%r10+ vpalignr $8,%xmm5,%xmm6,%xmm11+ movq %rdx,%r12+ shrdq $5,%r14,%r14+ vpsrlq $1,%xmm8,%xmm10+ xorq %rcx,%r13+ xorq %r8,%r12+ vpaddq %xmm11,%xmm1,%xmm1+ shrdq $4,%r13,%r13+ xorq %r10,%r14+ vpsrlq $7,%xmm8,%xmm11+ andq %rcx,%r12+ xorq %rcx,%r13+ vpsllq $56,%xmm8,%xmm9+ addq 16(%rsp),%r9+ movq %r10,%r15+ vpxor %xmm10,%xmm11,%xmm8+ xorq %r8,%r12+ shrdq $6,%r14,%r14+ vpsrlq $7,%xmm10,%xmm10+ xorq %r11,%r15+ addq %r12,%r9+ vpxor %xmm9,%xmm8,%xmm8+ shrdq $14,%r13,%r13+ andq %r15,%rdi+ vpsllq $7,%xmm9,%xmm9+ xorq %r10,%r14+ addq %r13,%r9+ vpxor %xmm10,%xmm8,%xmm8+ xorq %r11,%rdi+ shrdq $28,%r14,%r14+ vpsrlq $6,%xmm0,%xmm11+ addq %r9,%rbx+ addq %rdi,%r9+ vpxor %xmm9,%xmm8,%xmm8+ movq %rbx,%r13+ addq %r9,%r14+ vpsllq $3,%xmm0,%xmm10+ shrdq $23,%r13,%r13+ movq %r14,%r9+ vpaddq %xmm8,%xmm1,%xmm1+ movq %rcx,%r12+ shrdq $5,%r14,%r14+ vpsrlq $19,%xmm0,%xmm9+ xorq %rbx,%r13+ xorq %rdx,%r12+ vpxor %xmm10,%xmm11,%xmm11+ shrdq $4,%r13,%r13+ xorq %r9,%r14+ vpsllq $42,%xmm10,%xmm10+ andq %rbx,%r12+ xorq %rbx,%r13+ vpxor %xmm9,%xmm11,%xmm11+ addq 24(%rsp),%r8+ movq %r9,%rdi+ vpsrlq $42,%xmm9,%xmm9+ xorq %rdx,%r12+ shrdq $6,%r14,%r14+ vpxor %xmm10,%xmm11,%xmm11+ xorq %r10,%rdi+ addq %r12,%r8+ vpxor %xmm9,%xmm11,%xmm11+ shrdq $14,%r13,%r13+ andq %rdi,%r15+ vpaddq %xmm11,%xmm1,%xmm1+ xorq %r9,%r14+ addq %r13,%r8+ vpaddq -96(%rsi),%xmm1,%xmm10+ xorq %r10,%r15+ shrdq $28,%r14,%r14+ addq %r8,%rax+ addq %r15,%r8+ movq %rax,%r13+ addq %r8,%r14+ vmovdqa %xmm10,16(%rsp)+ vpalignr $8,%xmm2,%xmm3,%xmm8+ shrdq $23,%r13,%r13+ movq %r14,%r8+ vpalignr $8,%xmm6,%xmm7,%xmm11+ movq %rbx,%r12+ shrdq $5,%r14,%r14+ vpsrlq $1,%xmm8,%xmm10+ xorq %rax,%r13+ xorq %rcx,%r12+ vpaddq %xmm11,%xmm2,%xmm2+ shrdq $4,%r13,%r13+ xorq %r8,%r14+ vpsrlq $7,%xmm8,%xmm11+ andq %rax,%r12+ xorq %rax,%r13+ vpsllq $56,%xmm8,%xmm9+ addq 32(%rsp),%rdx+ movq %r8,%r15+ vpxor %xmm10,%xmm11,%xmm8+ xorq %rcx,%r12+ shrdq $6,%r14,%r14+ vpsrlq $7,%xmm10,%xmm10+ xorq %r9,%r15+ addq %r12,%rdx+ vpxor %xmm9,%xmm8,%xmm8+ shrdq $14,%r13,%r13+ andq %r15,%rdi+ vpsllq $7,%xmm9,%xmm9+ xorq %r8,%r14+ addq %r13,%rdx+ vpxor %xmm10,%xmm8,%xmm8+ xorq %r9,%rdi+ shrdq $28,%r14,%r14+ vpsrlq $6,%xmm1,%xmm11+ addq %rdx,%r11+ addq %rdi,%rdx+ vpxor %xmm9,%xmm8,%xmm8+ movq %r11,%r13+ addq %rdx,%r14+ vpsllq $3,%xmm1,%xmm10+ shrdq $23,%r13,%r13+ movq %r14,%rdx+ vpaddq %xmm8,%xmm2,%xmm2+ movq %rax,%r12+ shrdq $5,%r14,%r14+ vpsrlq $19,%xmm1,%xmm9+ xorq %r11,%r13+ xorq %rbx,%r12+ vpxor %xmm10,%xmm11,%xmm11+ shrdq $4,%r13,%r13+ xorq %rdx,%r14+ vpsllq $42,%xmm10,%xmm10+ andq %r11,%r12+ xorq %r11,%r13+ vpxor %xmm9,%xmm11,%xmm11+ addq 40(%rsp),%rcx+ movq %rdx,%rdi+ vpsrlq $42,%xmm9,%xmm9+ xorq %rbx,%r12+ shrdq $6,%r14,%r14+ vpxor %xmm10,%xmm11,%xmm11+ xorq %r8,%rdi+ addq %r12,%rcx+ vpxor %xmm9,%xmm11,%xmm11+ shrdq $14,%r13,%r13+ andq %rdi,%r15+ vpaddq %xmm11,%xmm2,%xmm2+ xorq %rdx,%r14+ addq %r13,%rcx+ vpaddq -64(%rsi),%xmm2,%xmm10+ xorq %r8,%r15+ shrdq $28,%r14,%r14+ addq %rcx,%r10+ addq %r15,%rcx+ movq %r10,%r13+ addq %rcx,%r14+ vmovdqa %xmm10,32(%rsp)+ vpalignr $8,%xmm3,%xmm4,%xmm8+ shrdq $23,%r13,%r13+ movq %r14,%rcx+ vpalignr $8,%xmm7,%xmm0,%xmm11+ movq %r11,%r12+ shrdq $5,%r14,%r14+ vpsrlq $1,%xmm8,%xmm10+ xorq %r10,%r13+ xorq %rax,%r12+ vpaddq %xmm11,%xmm3,%xmm3+ shrdq $4,%r13,%r13+ xorq %rcx,%r14+ vpsrlq $7,%xmm8,%xmm11+ andq %r10,%r12+ xorq %r10,%r13+ vpsllq $56,%xmm8,%xmm9+ addq 48(%rsp),%rbx+ movq %rcx,%r15+ vpxor %xmm10,%xmm11,%xmm8+ xorq %rax,%r12+ shrdq $6,%r14,%r14+ vpsrlq $7,%xmm10,%xmm10+ xorq %rdx,%r15+ addq %r12,%rbx+ vpxor %xmm9,%xmm8,%xmm8+ shrdq $14,%r13,%r13+ andq %r15,%rdi+ vpsllq $7,%xmm9,%xmm9+ xorq %rcx,%r14+ addq %r13,%rbx+ vpxor %xmm10,%xmm8,%xmm8+ xorq %rdx,%rdi+ shrdq $28,%r14,%r14+ vpsrlq $6,%xmm2,%xmm11+ addq %rbx,%r9+ addq %rdi,%rbx+ vpxor %xmm9,%xmm8,%xmm8+ movq %r9,%r13+ addq %rbx,%r14+ vpsllq $3,%xmm2,%xmm10+ shrdq $23,%r13,%r13+ movq %r14,%rbx+ vpaddq %xmm8,%xmm3,%xmm3+ movq %r10,%r12+ shrdq $5,%r14,%r14+ vpsrlq $19,%xmm2,%xmm9+ xorq %r9,%r13+ xorq %r11,%r12+ vpxor %xmm10,%xmm11,%xmm11+ shrdq $4,%r13,%r13+ xorq %rbx,%r14+ vpsllq $42,%xmm10,%xmm10+ andq %r9,%r12+ xorq %r9,%r13+ vpxor %xmm9,%xmm11,%xmm11+ addq 56(%rsp),%rax+ movq %rbx,%rdi+ vpsrlq $42,%xmm9,%xmm9+ xorq %r11,%r12+ shrdq $6,%r14,%r14+ vpxor %xmm10,%xmm11,%xmm11+ xorq %rcx,%rdi+ addq %r12,%rax+ vpxor %xmm9,%xmm11,%xmm11+ shrdq $14,%r13,%r13+ andq %rdi,%r15+ vpaddq %xmm11,%xmm3,%xmm3+ xorq %rbx,%r14+ addq %r13,%rax+ vpaddq -32(%rsi),%xmm3,%xmm10+ xorq %rcx,%r15+ shrdq $28,%r14,%r14+ addq %rax,%r8+ addq %r15,%rax+ movq %r8,%r13+ addq %rax,%r14+ vmovdqa %xmm10,48(%rsp)+ vpalignr $8,%xmm4,%xmm5,%xmm8+ shrdq $23,%r13,%r13+ movq %r14,%rax+ vpalignr $8,%xmm0,%xmm1,%xmm11+ movq %r9,%r12+ shrdq $5,%r14,%r14+ vpsrlq $1,%xmm8,%xmm10+ xorq %r8,%r13+ xorq %r10,%r12+ vpaddq %xmm11,%xmm4,%xmm4+ shrdq $4,%r13,%r13+ xorq %rax,%r14+ vpsrlq $7,%xmm8,%xmm11+ andq %r8,%r12+ xorq %r8,%r13+ vpsllq $56,%xmm8,%xmm9+ addq 64(%rsp),%r11+ movq %rax,%r15+ vpxor %xmm10,%xmm11,%xmm8+ xorq %r10,%r12+ shrdq $6,%r14,%r14+ vpsrlq $7,%xmm10,%xmm10+ xorq %rbx,%r15+ addq %r12,%r11+ vpxor %xmm9,%xmm8,%xmm8+ shrdq $14,%r13,%r13+ andq %r15,%rdi+ vpsllq $7,%xmm9,%xmm9+ xorq %rax,%r14+ addq %r13,%r11+ vpxor %xmm10,%xmm8,%xmm8+ xorq %rbx,%rdi+ shrdq $28,%r14,%r14+ vpsrlq $6,%xmm3,%xmm11+ addq %r11,%rdx+ addq %rdi,%r11+ vpxor %xmm9,%xmm8,%xmm8+ movq %rdx,%r13+ addq %r11,%r14+ vpsllq $3,%xmm3,%xmm10+ shrdq $23,%r13,%r13+ movq %r14,%r11+ vpaddq %xmm8,%xmm4,%xmm4+ movq %r8,%r12+ shrdq $5,%r14,%r14+ vpsrlq $19,%xmm3,%xmm9+ xorq %rdx,%r13+ xorq %r9,%r12+ vpxor %xmm10,%xmm11,%xmm11+ shrdq $4,%r13,%r13+ xorq %r11,%r14+ vpsllq $42,%xmm10,%xmm10+ andq %rdx,%r12+ xorq %rdx,%r13+ vpxor %xmm9,%xmm11,%xmm11+ addq 72(%rsp),%r10+ movq %r11,%rdi+ vpsrlq $42,%xmm9,%xmm9+ xorq %r9,%r12+ shrdq $6,%r14,%r14+ vpxor %xmm10,%xmm11,%xmm11+ xorq %rax,%rdi+ addq %r12,%r10+ vpxor %xmm9,%xmm11,%xmm11+ shrdq $14,%r13,%r13+ andq %rdi,%r15+ vpaddq %xmm11,%xmm4,%xmm4+ xorq %r11,%r14+ addq %r13,%r10+ vpaddq 0(%rsi),%xmm4,%xmm10+ xorq %rax,%r15+ shrdq $28,%r14,%r14+ addq %r10,%rcx+ addq %r15,%r10+ movq %rcx,%r13+ addq %r10,%r14+ vmovdqa %xmm10,64(%rsp)+ vpalignr $8,%xmm5,%xmm6,%xmm8+ shrdq $23,%r13,%r13+ movq %r14,%r10+ vpalignr $8,%xmm1,%xmm2,%xmm11+ movq %rdx,%r12+ shrdq $5,%r14,%r14+ vpsrlq $1,%xmm8,%xmm10+ xorq %rcx,%r13+ xorq %r8,%r12+ vpaddq %xmm11,%xmm5,%xmm5+ shrdq $4,%r13,%r13+ xorq %r10,%r14+ vpsrlq $7,%xmm8,%xmm11+ andq %rcx,%r12+ xorq %rcx,%r13+ vpsllq $56,%xmm8,%xmm9+ addq 80(%rsp),%r9+ movq %r10,%r15+ vpxor %xmm10,%xmm11,%xmm8+ xorq %r8,%r12+ shrdq $6,%r14,%r14+ vpsrlq $7,%xmm10,%xmm10+ xorq %r11,%r15+ addq %r12,%r9+ vpxor %xmm9,%xmm8,%xmm8+ shrdq $14,%r13,%r13+ andq %r15,%rdi+ vpsllq $7,%xmm9,%xmm9+ xorq %r10,%r14+ addq %r13,%r9+ vpxor %xmm10,%xmm8,%xmm8+ xorq %r11,%rdi+ shrdq $28,%r14,%r14+ vpsrlq $6,%xmm4,%xmm11+ addq %r9,%rbx+ addq %rdi,%r9+ vpxor %xmm9,%xmm8,%xmm8+ movq %rbx,%r13+ addq %r9,%r14+ vpsllq $3,%xmm4,%xmm10+ shrdq $23,%r13,%r13+ movq %r14,%r9+ vpaddq %xmm8,%xmm5,%xmm5+ movq %rcx,%r12+ shrdq $5,%r14,%r14+ vpsrlq $19,%xmm4,%xmm9+ xorq %rbx,%r13+ xorq %rdx,%r12+ vpxor %xmm10,%xmm11,%xmm11+ shrdq $4,%r13,%r13+ xorq %r9,%r14+ vpsllq $42,%xmm10,%xmm10+ andq %rbx,%r12+ xorq %rbx,%r13+ vpxor %xmm9,%xmm11,%xmm11+ addq 88(%rsp),%r8+ movq %r9,%rdi+ vpsrlq $42,%xmm9,%xmm9+ xorq %rdx,%r12+ shrdq $6,%r14,%r14+ vpxor %xmm10,%xmm11,%xmm11+ xorq %r10,%rdi+ addq %r12,%r8+ vpxor %xmm9,%xmm11,%xmm11+ shrdq $14,%r13,%r13+ andq %rdi,%r15+ vpaddq %xmm11,%xmm5,%xmm5+ xorq %r9,%r14+ addq %r13,%r8+ vpaddq 32(%rsi),%xmm5,%xmm10+ xorq %r10,%r15+ shrdq $28,%r14,%r14+ addq %r8,%rax+ addq %r15,%r8+ movq %rax,%r13+ addq %r8,%r14+ vmovdqa %xmm10,80(%rsp)+ vpalignr $8,%xmm6,%xmm7,%xmm8+ shrdq $23,%r13,%r13+ movq %r14,%r8+ vpalignr $8,%xmm2,%xmm3,%xmm11+ movq %rbx,%r12+ shrdq $5,%r14,%r14+ vpsrlq $1,%xmm8,%xmm10+ xorq %rax,%r13+ xorq %rcx,%r12+ vpaddq %xmm11,%xmm6,%xmm6+ shrdq $4,%r13,%r13+ xorq %r8,%r14+ vpsrlq $7,%xmm8,%xmm11+ andq %rax,%r12+ xorq %rax,%r13+ vpsllq $56,%xmm8,%xmm9+ addq 96(%rsp),%rdx+ movq %r8,%r15+ vpxor %xmm10,%xmm11,%xmm8+ xorq %rcx,%r12+ shrdq $6,%r14,%r14+ vpsrlq $7,%xmm10,%xmm10+ xorq %r9,%r15+ addq %r12,%rdx+ vpxor %xmm9,%xmm8,%xmm8+ shrdq $14,%r13,%r13+ andq %r15,%rdi+ vpsllq $7,%xmm9,%xmm9+ xorq %r8,%r14+ addq %r13,%rdx+ vpxor %xmm10,%xmm8,%xmm8+ xorq %r9,%rdi+ shrdq $28,%r14,%r14+ vpsrlq $6,%xmm5,%xmm11+ addq %rdx,%r11+ addq %rdi,%rdx+ vpxor %xmm9,%xmm8,%xmm8+ movq %r11,%r13+ addq %rdx,%r14+ vpsllq $3,%xmm5,%xmm10+ shrdq $23,%r13,%r13+ movq %r14,%rdx+ vpaddq %xmm8,%xmm6,%xmm6+ movq %rax,%r12+ shrdq $5,%r14,%r14+ vpsrlq $19,%xmm5,%xmm9+ xorq %r11,%r13+ xorq %rbx,%r12+ vpxor %xmm10,%xmm11,%xmm11+ shrdq $4,%r13,%r13+ xorq %rdx,%r14+ vpsllq $42,%xmm10,%xmm10+ andq %r11,%r12+ xorq %r11,%r13+ vpxor %xmm9,%xmm11,%xmm11+ addq 104(%rsp),%rcx+ movq %rdx,%rdi+ vpsrlq $42,%xmm9,%xmm9+ xorq %rbx,%r12+ shrdq $6,%r14,%r14+ vpxor %xmm10,%xmm11,%xmm11+ xorq %r8,%rdi+ addq %r12,%rcx+ vpxor %xmm9,%xmm11,%xmm11+ shrdq $14,%r13,%r13+ andq %rdi,%r15+ vpaddq %xmm11,%xmm6,%xmm6+ xorq %rdx,%r14+ addq %r13,%rcx+ vpaddq 64(%rsi),%xmm6,%xmm10+ xorq %r8,%r15+ shrdq $28,%r14,%r14+ addq %rcx,%r10+ addq %r15,%rcx+ movq %r10,%r13+ addq %rcx,%r14+ vmovdqa %xmm10,96(%rsp)+ vpalignr $8,%xmm7,%xmm0,%xmm8+ shrdq $23,%r13,%r13+ movq %r14,%rcx+ vpalignr $8,%xmm3,%xmm4,%xmm11+ movq %r11,%r12+ shrdq $5,%r14,%r14+ vpsrlq $1,%xmm8,%xmm10+ xorq %r10,%r13+ xorq %rax,%r12+ vpaddq %xmm11,%xmm7,%xmm7+ shrdq $4,%r13,%r13+ xorq %rcx,%r14+ vpsrlq $7,%xmm8,%xmm11+ andq %r10,%r12+ xorq %r10,%r13+ vpsllq $56,%xmm8,%xmm9+ addq 112(%rsp),%rbx+ movq %rcx,%r15+ vpxor %xmm10,%xmm11,%xmm8+ xorq %rax,%r12+ shrdq $6,%r14,%r14+ vpsrlq $7,%xmm10,%xmm10+ xorq %rdx,%r15+ addq %r12,%rbx+ vpxor %xmm9,%xmm8,%xmm8+ shrdq $14,%r13,%r13+ andq %r15,%rdi+ vpsllq $7,%xmm9,%xmm9+ xorq %rcx,%r14+ addq %r13,%rbx+ vpxor %xmm10,%xmm8,%xmm8+ xorq %rdx,%rdi+ shrdq $28,%r14,%r14+ vpsrlq $6,%xmm6,%xmm11+ addq %rbx,%r9+ addq %rdi,%rbx+ vpxor %xmm9,%xmm8,%xmm8+ movq %r9,%r13+ addq %rbx,%r14+ vpsllq $3,%xmm6,%xmm10+ shrdq $23,%r13,%r13+ movq %r14,%rbx+ vpaddq %xmm8,%xmm7,%xmm7+ movq %r10,%r12+ shrdq $5,%r14,%r14+ vpsrlq $19,%xmm6,%xmm9+ xorq %r9,%r13+ xorq %r11,%r12+ vpxor %xmm10,%xmm11,%xmm11+ shrdq $4,%r13,%r13+ xorq %rbx,%r14+ vpsllq $42,%xmm10,%xmm10+ andq %r9,%r12+ xorq %r9,%r13+ vpxor %xmm9,%xmm11,%xmm11+ addq 120(%rsp),%rax+ movq %rbx,%rdi+ vpsrlq $42,%xmm9,%xmm9+ xorq %r11,%r12+ shrdq $6,%r14,%r14+ vpxor %xmm10,%xmm11,%xmm11+ xorq %rcx,%rdi+ addq %r12,%rax+ vpxor %xmm9,%xmm11,%xmm11+ shrdq $14,%r13,%r13+ andq %rdi,%r15+ vpaddq %xmm11,%xmm7,%xmm7+ xorq %rbx,%r14+ addq %r13,%rax+ vpaddq 96(%rsi),%xmm7,%xmm10+ xorq %rcx,%r15+ shrdq $28,%r14,%r14+ addq %rax,%r8+ addq %r15,%rax+ movq %r8,%r13+ addq %rax,%r14+ vmovdqa %xmm10,112(%rsp)+ cmpb $0,135(%rsi)+ jne .Lavx_00_47+ shrdq $23,%r13,%r13+ movq %r14,%rax+ movq %r9,%r12+ shrdq $5,%r14,%r14+ xorq %r8,%r13+ xorq %r10,%r12+ shrdq $4,%r13,%r13+ xorq %rax,%r14+ andq %r8,%r12+ xorq %r8,%r13+ addq 0(%rsp),%r11+ movq %rax,%r15+ xorq %r10,%r12+ shrdq $6,%r14,%r14+ xorq %rbx,%r15+ addq %r12,%r11+ shrdq $14,%r13,%r13+ andq %r15,%rdi+ xorq %rax,%r14+ addq %r13,%r11+ xorq %rbx,%rdi+ shrdq $28,%r14,%r14+ addq %r11,%rdx+ addq %rdi,%r11+ movq %rdx,%r13+ addq %r11,%r14+ shrdq $23,%r13,%r13+ movq %r14,%r11+ movq %r8,%r12+ shrdq $5,%r14,%r14+ xorq %rdx,%r13+ xorq %r9,%r12+ shrdq $4,%r13,%r13+ xorq %r11,%r14+ andq %rdx,%r12+ xorq %rdx,%r13+ addq 8(%rsp),%r10+ movq %r11,%rdi+ xorq %r9,%r12+ shrdq $6,%r14,%r14+ xorq %rax,%rdi+ addq %r12,%r10+ shrdq $14,%r13,%r13+ andq %rdi,%r15+ xorq %r11,%r14+ addq %r13,%r10+ xorq %rax,%r15+ shrdq $28,%r14,%r14+ addq %r10,%rcx+ addq %r15,%r10+ movq %rcx,%r13+ addq %r10,%r14+ shrdq $23,%r13,%r13+ movq %r14,%r10+ movq %rdx,%r12+ shrdq $5,%r14,%r14+ xorq %rcx,%r13+ xorq %r8,%r12+ shrdq $4,%r13,%r13+ xorq %r10,%r14+ andq %rcx,%r12+ xorq %rcx,%r13+ addq 16(%rsp),%r9+ movq %r10,%r15+ xorq %r8,%r12+ shrdq $6,%r14,%r14+ xorq %r11,%r15+ addq %r12,%r9+ shrdq $14,%r13,%r13+ andq %r15,%rdi+ xorq %r10,%r14+ addq %r13,%r9+ xorq %r11,%rdi+ shrdq $28,%r14,%r14+ addq %r9,%rbx+ addq %rdi,%r9+ movq %rbx,%r13+ addq %r9,%r14+ shrdq $23,%r13,%r13+ movq %r14,%r9+ movq %rcx,%r12+ shrdq $5,%r14,%r14+ xorq %rbx,%r13+ xorq %rdx,%r12+ shrdq $4,%r13,%r13+ xorq %r9,%r14+ andq %rbx,%r12+ xorq %rbx,%r13+ addq 24(%rsp),%r8+ movq %r9,%rdi+ xorq %rdx,%r12+ shrdq $6,%r14,%r14+ xorq %r10,%rdi+ addq %r12,%r8+ shrdq $14,%r13,%r13+ andq %rdi,%r15+ xorq %r9,%r14+ addq %r13,%r8+ xorq %r10,%r15+ shrdq $28,%r14,%r14+ addq %r8,%rax+ addq %r15,%r8+ movq %rax,%r13+ addq %r8,%r14+ shrdq $23,%r13,%r13+ movq %r14,%r8+ movq %rbx,%r12+ shrdq $5,%r14,%r14+ xorq %rax,%r13+ xorq %rcx,%r12+ shrdq $4,%r13,%r13+ xorq %r8,%r14+ andq %rax,%r12+ xorq %rax,%r13+ addq 32(%rsp),%rdx+ movq %r8,%r15+ xorq %rcx,%r12+ shrdq $6,%r14,%r14+ xorq %r9,%r15+ addq %r12,%rdx+ shrdq $14,%r13,%r13+ andq %r15,%rdi+ xorq %r8,%r14+ addq %r13,%rdx+ xorq %r9,%rdi+ shrdq $28,%r14,%r14+ addq %rdx,%r11+ addq %rdi,%rdx+ movq %r11,%r13+ addq %rdx,%r14+ shrdq $23,%r13,%r13+ movq %r14,%rdx+ movq %rax,%r12+ shrdq $5,%r14,%r14+ xorq %r11,%r13+ xorq %rbx,%r12+ shrdq $4,%r13,%r13+ xorq %rdx,%r14+ andq %r11,%r12+ xorq %r11,%r13+ addq 40(%rsp),%rcx+ movq %rdx,%rdi+ xorq %rbx,%r12+ shrdq $6,%r14,%r14+ xorq %r8,%rdi+ addq %r12,%rcx+ shrdq $14,%r13,%r13+ andq %rdi,%r15+ xorq %rdx,%r14+ addq %r13,%rcx+ xorq %r8,%r15+ shrdq $28,%r14,%r14+ addq %rcx,%r10+ addq %r15,%rcx+ movq %r10,%r13+ addq %rcx,%r14+ shrdq $23,%r13,%r13+ movq %r14,%rcx+ movq %r11,%r12+ shrdq $5,%r14,%r14+ xorq %r10,%r13+ xorq %rax,%r12+ shrdq $4,%r13,%r13+ xorq %rcx,%r14+ andq %r10,%r12+ xorq %r10,%r13+ addq 48(%rsp),%rbx+ movq %rcx,%r15+ xorq %rax,%r12+ shrdq $6,%r14,%r14+ xorq %rdx,%r15+ addq %r12,%rbx+ shrdq $14,%r13,%r13+ andq %r15,%rdi+ xorq %rcx,%r14+ addq %r13,%rbx+ xorq %rdx,%rdi+ shrdq $28,%r14,%r14+ addq %rbx,%r9+ addq %rdi,%rbx+ movq %r9,%r13+ addq %rbx,%r14+ shrdq $23,%r13,%r13+ movq %r14,%rbx+ movq %r10,%r12+ shrdq $5,%r14,%r14+ xorq %r9,%r13+ xorq %r11,%r12+ shrdq $4,%r13,%r13+ xorq %rbx,%r14+ andq %r9,%r12+ xorq %r9,%r13+ addq 56(%rsp),%rax+ movq %rbx,%rdi+ xorq %r11,%r12+ shrdq $6,%r14,%r14+ xorq %rcx,%rdi+ addq %r12,%rax+ shrdq $14,%r13,%r13+ andq %rdi,%r15+ xorq %rbx,%r14+ addq %r13,%rax+ xorq %rcx,%r15+ shrdq $28,%r14,%r14+ addq %rax,%r8+ addq %r15,%rax+ movq %r8,%r13+ addq %rax,%r14+ shrdq $23,%r13,%r13+ movq %r14,%rax+ movq %r9,%r12+ shrdq $5,%r14,%r14+ xorq %r8,%r13+ xorq %r10,%r12+ shrdq $4,%r13,%r13+ xorq %rax,%r14+ andq %r8,%r12+ xorq %r8,%r13+ addq 64(%rsp),%r11+ movq %rax,%r15+ xorq %r10,%r12+ shrdq $6,%r14,%r14+ xorq %rbx,%r15+ addq %r12,%r11+ shrdq $14,%r13,%r13+ andq %r15,%rdi+ xorq %rax,%r14+ addq %r13,%r11+ xorq %rbx,%rdi+ shrdq $28,%r14,%r14+ addq %r11,%rdx+ addq %rdi,%r11+ movq %rdx,%r13+ addq %r11,%r14+ shrdq $23,%r13,%r13+ movq %r14,%r11+ movq %r8,%r12+ shrdq $5,%r14,%r14+ xorq %rdx,%r13+ xorq %r9,%r12+ shrdq $4,%r13,%r13+ xorq %r11,%r14+ andq %rdx,%r12+ xorq %rdx,%r13+ addq 72(%rsp),%r10+ movq %r11,%rdi+ xorq %r9,%r12+ shrdq $6,%r14,%r14+ xorq %rax,%rdi+ addq %r12,%r10+ shrdq $14,%r13,%r13+ andq %rdi,%r15+ xorq %r11,%r14+ addq %r13,%r10+ xorq %rax,%r15+ shrdq $28,%r14,%r14+ addq %r10,%rcx+ addq %r15,%r10+ movq %rcx,%r13+ addq %r10,%r14+ shrdq $23,%r13,%r13+ movq %r14,%r10+ movq %rdx,%r12+ shrdq $5,%r14,%r14+ xorq %rcx,%r13+ xorq %r8,%r12+ shrdq $4,%r13,%r13+ xorq %r10,%r14+ andq %rcx,%r12+ xorq %rcx,%r13+ addq 80(%rsp),%r9+ movq %r10,%r15+ xorq %r8,%r12+ shrdq $6,%r14,%r14+ xorq %r11,%r15+ addq %r12,%r9+ shrdq $14,%r13,%r13+ andq %r15,%rdi+ xorq %r10,%r14+ addq %r13,%r9+ xorq %r11,%rdi+ shrdq $28,%r14,%r14+ addq %r9,%rbx+ addq %rdi,%r9+ movq %rbx,%r13+ addq %r9,%r14+ shrdq $23,%r13,%r13+ movq %r14,%r9+ movq %rcx,%r12+ shrdq $5,%r14,%r14+ xorq %rbx,%r13+ xorq %rdx,%r12+ shrdq $4,%r13,%r13+ xorq %r9,%r14+ andq %rbx,%r12+ xorq %rbx,%r13+ addq 88(%rsp),%r8+ movq %r9,%rdi+ xorq %rdx,%r12+ shrdq $6,%r14,%r14+ xorq %r10,%rdi+ addq %r12,%r8+ shrdq $14,%r13,%r13+ andq %rdi,%r15+ xorq %r9,%r14+ addq %r13,%r8+ xorq %r10,%r15+ shrdq $28,%r14,%r14+ addq %r8,%rax+ addq %r15,%r8+ movq %rax,%r13+ addq %r8,%r14+ shrdq $23,%r13,%r13+ movq %r14,%r8+ movq %rbx,%r12+ shrdq $5,%r14,%r14+ xorq %rax,%r13+ xorq %rcx,%r12+ shrdq $4,%r13,%r13+ xorq %r8,%r14+ andq %rax,%r12+ xorq %rax,%r13+ addq 96(%rsp),%rdx+ movq %r8,%r15+ xorq %rcx,%r12+ shrdq $6,%r14,%r14+ xorq %r9,%r15+ addq %r12,%rdx+ shrdq $14,%r13,%r13+ andq %r15,%rdi+ xorq %r8,%r14+ addq %r13,%rdx+ xorq %r9,%rdi+ shrdq $28,%r14,%r14+ addq %rdx,%r11+ addq %rdi,%rdx+ movq %r11,%r13+ addq %rdx,%r14+ shrdq $23,%r13,%r13+ movq %r14,%rdx+ movq %rax,%r12+ shrdq $5,%r14,%r14+ xorq %r11,%r13+ xorq %rbx,%r12+ shrdq $4,%r13,%r13+ xorq %rdx,%r14+ andq %r11,%r12+ xorq %r11,%r13+ addq 104(%rsp),%rcx+ movq %rdx,%rdi+ xorq %rbx,%r12+ shrdq $6,%r14,%r14+ xorq %r8,%rdi+ addq %r12,%rcx+ shrdq $14,%r13,%r13+ andq %rdi,%r15+ xorq %rdx,%r14+ addq %r13,%rcx+ xorq %r8,%r15+ shrdq $28,%r14,%r14+ addq %rcx,%r10+ addq %r15,%rcx+ movq %r10,%r13+ addq %rcx,%r14+ shrdq $23,%r13,%r13+ movq %r14,%rcx+ movq %r11,%r12+ shrdq $5,%r14,%r14+ xorq %r10,%r13+ xorq %rax,%r12+ shrdq $4,%r13,%r13+ xorq %rcx,%r14+ andq %r10,%r12+ xorq %r10,%r13+ addq 112(%rsp),%rbx+ movq %rcx,%r15+ xorq %rax,%r12+ shrdq $6,%r14,%r14+ xorq %rdx,%r15+ addq %r12,%rbx+ shrdq $14,%r13,%r13+ andq %r15,%rdi+ xorq %rcx,%r14+ addq %r13,%rbx+ xorq %rdx,%rdi+ shrdq $28,%r14,%r14+ addq %rbx,%r9+ addq %rdi,%rbx+ movq %r9,%r13+ addq %rbx,%r14+ shrdq $23,%r13,%r13+ movq %r14,%rbx+ movq %r10,%r12+ shrdq $5,%r14,%r14+ xorq %r9,%r13+ xorq %r11,%r12+ shrdq $4,%r13,%r13+ xorq %rbx,%r14+ andq %r9,%r12+ xorq %r9,%r13+ addq 120(%rsp),%rax+ movq %rbx,%rdi+ xorq %r11,%r12+ shrdq $6,%r14,%r14+ xorq %rcx,%rdi+ addq %r12,%rax+ shrdq $14,%r13,%r13+ andq %rdi,%r15+ xorq %rbx,%r14+ addq %r13,%rax+ xorq %rcx,%r15+ shrdq $28,%r14,%r14+ addq %rax,%r8+ addq %r15,%rax+ movq %r8,%r13+ addq %rax,%r14+ movq -64(%rbp),%rdi+ movq %r14,%rax+ movq -56(%rbp),%rsi++ addq 0(%rdi),%rax+ addq 8(%rdi),%rbx+ addq 16(%rdi),%rcx+ addq 24(%rdi),%rdx+ addq 32(%rdi),%r8+ addq 40(%rdi),%r9+ addq 48(%rdi),%r10+ addq 56(%rdi),%r11++ leaq 128(%rsi),%rsi+ cmpq -48(%rbp),%rsi++ movq %rax,0(%rdi)+ movq %rbx,8(%rdi)+ movq %rcx,16(%rdi)+ movq %rdx,24(%rdi)+ movq %r8,32(%rdi)+ movq %r9,40(%rdi)+ movq %r10,48(%rdi)+ movq %r11,56(%rdi)+ jb .Lloop_avx++ vzeroupper+ movq -40(%rbp),%r15+ movq -32(%rbp),%r14+ movq -24(%rbp),%r13+ movq -16(%rbp),%r12+ movq -8(%rbp),%rbx+ movq %rbp,%rsp+.cfi_def_cfa_register %rsp+ popq %rbp+.cfi_adjust_cfa_offset -8+.cfi_restore %rbp+.cfi_restore %r12+.cfi_restore %r13+.cfi_restore %r14+.cfi_restore %r15+.cfi_restore %rbx+ .byte 0xf3,0xc3+.cfi_endproc +.size crypton_sha512_asm_block_data_order_avx,.-crypton_sha512_asm_block_data_order_avx+.type crypton_sha512_asm_block_data_order_avx2,@function+.align 64+crypton_sha512_asm_block_data_order_avx2:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa+++ pushq %rbp+.cfi_adjust_cfa_offset 8+.cfi_offset %rbp,-16+ movq %rsp,%rbp+.cfi_def_cfa_register %rbp+.Lavx2_shortcut:+ pushq %rbx+.cfi_offset %rbx,-24+ pushq %r12+.cfi_offset %r12,-32+ pushq %r13+.cfi_offset %r13,-40+ pushq %r14+.cfi_offset %r14,-48+ pushq %r15+.cfi_offset %r15,-56+ shlq $4,%rdx+ subq $24,%rsp++ leaq (%rsi,%rdx,8),%rdx+ movq %rdi,-64(%rbp)++ movq %rdx,-48(%rbp)+++ leaq -128(%rsp),%rsp+ vzeroupper+ andq $-128,%rsp+ subq $-128,%rsi+ movq 0(%rdi),%rax+ movq %rsi,%r12+ movq 8(%rdi),%rbx+ cmpq %rdx,%rsi+ movq 16(%rdi),%rcx+ cmoveq %rsp,%r12+ movq 24(%rdi),%rdx+ movq 32(%rdi),%r8+ movq 40(%rdi),%r9+ movq 48(%rdi),%r10+ movq 56(%rdi),%r11+ jmp .Loop_avx2+.align 16+.Loop_avx2:+ vmovdqa K512+1280(%rip),%ymm10+ movq %rsi,-56(%rbp)+ vmovdqu -128(%rsi),%xmm0+ vmovdqu -128+16(%rsi),%xmm1+ vmovdqu -128+32(%rsi),%xmm2+ vmovdqu -128+48(%rsi),%xmm3+ vmovdqu -128+64(%rsi),%xmm4+ vmovdqu -128+80(%rsi),%xmm5+ vmovdqu -128+96(%rsi),%xmm6+ vmovdqu -128+112(%rsi),%xmm7+ leaq K512+128(%rip),%rsi+ vinserti128 $1,(%r12),%ymm0,%ymm0+ vinserti128 $1,16(%r12),%ymm1,%ymm1+ vpshufb %ymm10,%ymm0,%ymm0+ vinserti128 $1,32(%r12),%ymm2,%ymm2+ vpshufb %ymm10,%ymm1,%ymm1+ vinserti128 $1,48(%r12),%ymm3,%ymm3+ vpshufb %ymm10,%ymm2,%ymm2+ vinserti128 $1,64(%r12),%ymm4,%ymm4+ vpshufb %ymm10,%ymm3,%ymm3+ vinserti128 $1,80(%r12),%ymm5,%ymm5+ vpshufb %ymm10,%ymm4,%ymm4+ vinserti128 $1,96(%r12),%ymm6,%ymm6+ vpshufb %ymm10,%ymm5,%ymm5+ vinserti128 $1,112(%r12),%ymm7,%ymm7++ vpaddq -128(%rsi),%ymm0,%ymm8+ vpshufb %ymm10,%ymm6,%ymm6+ vpaddq -96(%rsi),%ymm1,%ymm9+ vpshufb %ymm10,%ymm7,%ymm7+ vpaddq -64(%rsi),%ymm2,%ymm10+ vpaddq -32(%rsi),%ymm3,%ymm11+ vmovdqa %ymm8,0(%rsp)+ vpaddq 0(%rsi),%ymm4,%ymm8+ vmovdqa %ymm9,32(%rsp)+ vpaddq 32(%rsi),%ymm5,%ymm9+ vmovdqa %ymm10,64(%rsp)+ vpaddq 64(%rsi),%ymm6,%ymm10+ vmovdqa %ymm11,96(%rsp)+ leaq -128(%rsp),%rsp+ vpaddq 96(%rsi),%ymm7,%ymm11+ vmovdqa %ymm8,0(%rsp)+ xorq %r14,%r14+ vmovdqa %ymm9,32(%rsp)+ movq %rbx,%rdi+ vmovdqa %ymm10,64(%rsp)+ xorq %rcx,%rdi+ vmovdqa %ymm11,96(%rsp)+ movq %r9,%r12+ addq $32*8,%rsi+ jmp .Lavx2_00_47++.align 16+.Lavx2_00_47:+ leaq -128(%rsp),%rsp+ vpalignr $8,%ymm0,%ymm1,%ymm8+ addq 0+256(%rsp),%r11+ andq %r8,%r12+ rorxq $41,%r8,%r13+ vpalignr $8,%ymm4,%ymm5,%ymm11+ rorxq $18,%r8,%r15+ leaq (%rax,%r14,1),%rax+ leaq (%r11,%r12,1),%r11+ vpsrlq $1,%ymm8,%ymm10+ andnq %r10,%r8,%r12+ xorq %r15,%r13+ rorxq $14,%r8,%r14+ vpaddq %ymm11,%ymm0,%ymm0+ vpsrlq $7,%ymm8,%ymm11+ leaq (%r11,%r12,1),%r11+ xorq %r14,%r13+ movq %rax,%r15+ vpsllq $56,%ymm8,%ymm9+ vpxor %ymm10,%ymm11,%ymm8+ rorxq $39,%rax,%r12+ leaq (%r11,%r13,1),%r11+ xorq %rbx,%r15+ vpsrlq $7,%ymm10,%ymm10+ vpxor %ymm9,%ymm8,%ymm8+ rorxq $34,%rax,%r14+ rorxq $28,%rax,%r13+ leaq (%rdx,%r11,1),%rdx+ vpsllq $7,%ymm9,%ymm9+ vpxor %ymm10,%ymm8,%ymm8+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %rbx,%rdi+ vpsrlq $6,%ymm7,%ymm11+ vpxor %ymm9,%ymm8,%ymm8+ xorq %r13,%r14+ leaq (%r11,%rdi,1),%r11+ movq %r8,%r12+ vpsllq $3,%ymm7,%ymm10+ vpaddq %ymm8,%ymm0,%ymm0+ addq 8+256(%rsp),%r10+ andq %rdx,%r12+ rorxq $41,%rdx,%r13+ vpsrlq $19,%ymm7,%ymm9+ vpxor %ymm10,%ymm11,%ymm11+ rorxq $18,%rdx,%rdi+ leaq (%r11,%r14,1),%r11+ leaq (%r10,%r12,1),%r10+ vpsllq $42,%ymm10,%ymm10+ vpxor %ymm9,%ymm11,%ymm11+ andnq %r9,%rdx,%r12+ xorq %rdi,%r13+ rorxq $14,%rdx,%r14+ vpsrlq $42,%ymm9,%ymm9+ vpxor %ymm10,%ymm11,%ymm11+ leaq (%r10,%r12,1),%r10+ xorq %r14,%r13+ movq %r11,%rdi+ vpxor %ymm9,%ymm11,%ymm11+ rorxq $39,%r11,%r12+ leaq (%r10,%r13,1),%r10+ xorq %rax,%rdi+ vpaddq %ymm11,%ymm0,%ymm0+ rorxq $34,%r11,%r14+ rorxq $28,%r11,%r13+ leaq (%rcx,%r10,1),%rcx+ vpaddq -128(%rsi),%ymm0,%ymm10+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %rax,%r15+ xorq %r13,%r14+ leaq (%r10,%r15,1),%r10+ movq %rdx,%r12+ vmovdqa %ymm10,0(%rsp)+ vpalignr $8,%ymm1,%ymm2,%ymm8+ addq 32+256(%rsp),%r9+ andq %rcx,%r12+ rorxq $41,%rcx,%r13+ vpalignr $8,%ymm5,%ymm6,%ymm11+ rorxq $18,%rcx,%r15+ leaq (%r10,%r14,1),%r10+ leaq (%r9,%r12,1),%r9+ vpsrlq $1,%ymm8,%ymm10+ andnq %r8,%rcx,%r12+ xorq %r15,%r13+ rorxq $14,%rcx,%r14+ vpaddq %ymm11,%ymm1,%ymm1+ vpsrlq $7,%ymm8,%ymm11+ leaq (%r9,%r12,1),%r9+ xorq %r14,%r13+ movq %r10,%r15+ vpsllq $56,%ymm8,%ymm9+ vpxor %ymm10,%ymm11,%ymm8+ rorxq $39,%r10,%r12+ leaq (%r9,%r13,1),%r9+ xorq %r11,%r15+ vpsrlq $7,%ymm10,%ymm10+ vpxor %ymm9,%ymm8,%ymm8+ rorxq $34,%r10,%r14+ rorxq $28,%r10,%r13+ leaq (%rbx,%r9,1),%rbx+ vpsllq $7,%ymm9,%ymm9+ vpxor %ymm10,%ymm8,%ymm8+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %r11,%rdi+ vpsrlq $6,%ymm0,%ymm11+ vpxor %ymm9,%ymm8,%ymm8+ xorq %r13,%r14+ leaq (%r9,%rdi,1),%r9+ movq %rcx,%r12+ vpsllq $3,%ymm0,%ymm10+ vpaddq %ymm8,%ymm1,%ymm1+ addq 40+256(%rsp),%r8+ andq %rbx,%r12+ rorxq $41,%rbx,%r13+ vpsrlq $19,%ymm0,%ymm9+ vpxor %ymm10,%ymm11,%ymm11+ rorxq $18,%rbx,%rdi+ leaq (%r9,%r14,1),%r9+ leaq (%r8,%r12,1),%r8+ vpsllq $42,%ymm10,%ymm10+ vpxor %ymm9,%ymm11,%ymm11+ andnq %rdx,%rbx,%r12+ xorq %rdi,%r13+ rorxq $14,%rbx,%r14+ vpsrlq $42,%ymm9,%ymm9+ vpxor %ymm10,%ymm11,%ymm11+ leaq (%r8,%r12,1),%r8+ xorq %r14,%r13+ movq %r9,%rdi+ vpxor %ymm9,%ymm11,%ymm11+ rorxq $39,%r9,%r12+ leaq (%r8,%r13,1),%r8+ xorq %r10,%rdi+ vpaddq %ymm11,%ymm1,%ymm1+ rorxq $34,%r9,%r14+ rorxq $28,%r9,%r13+ leaq (%rax,%r8,1),%rax+ vpaddq -96(%rsi),%ymm1,%ymm10+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %r10,%r15+ xorq %r13,%r14+ leaq (%r8,%r15,1),%r8+ movq %rbx,%r12+ vmovdqa %ymm10,32(%rsp)+ vpalignr $8,%ymm2,%ymm3,%ymm8+ addq 64+256(%rsp),%rdx+ andq %rax,%r12+ rorxq $41,%rax,%r13+ vpalignr $8,%ymm6,%ymm7,%ymm11+ rorxq $18,%rax,%r15+ leaq (%r8,%r14,1),%r8+ leaq (%rdx,%r12,1),%rdx+ vpsrlq $1,%ymm8,%ymm10+ andnq %rcx,%rax,%r12+ xorq %r15,%r13+ rorxq $14,%rax,%r14+ vpaddq %ymm11,%ymm2,%ymm2+ vpsrlq $7,%ymm8,%ymm11+ leaq (%rdx,%r12,1),%rdx+ xorq %r14,%r13+ movq %r8,%r15+ vpsllq $56,%ymm8,%ymm9+ vpxor %ymm10,%ymm11,%ymm8+ rorxq $39,%r8,%r12+ leaq (%rdx,%r13,1),%rdx+ xorq %r9,%r15+ vpsrlq $7,%ymm10,%ymm10+ vpxor %ymm9,%ymm8,%ymm8+ rorxq $34,%r8,%r14+ rorxq $28,%r8,%r13+ leaq (%r11,%rdx,1),%r11+ vpsllq $7,%ymm9,%ymm9+ vpxor %ymm10,%ymm8,%ymm8+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %r9,%rdi+ vpsrlq $6,%ymm1,%ymm11+ vpxor %ymm9,%ymm8,%ymm8+ xorq %r13,%r14+ leaq (%rdx,%rdi,1),%rdx+ movq %rax,%r12+ vpsllq $3,%ymm1,%ymm10+ vpaddq %ymm8,%ymm2,%ymm2+ addq 72+256(%rsp),%rcx+ andq %r11,%r12+ rorxq $41,%r11,%r13+ vpsrlq $19,%ymm1,%ymm9+ vpxor %ymm10,%ymm11,%ymm11+ rorxq $18,%r11,%rdi+ leaq (%rdx,%r14,1),%rdx+ leaq (%rcx,%r12,1),%rcx+ vpsllq $42,%ymm10,%ymm10+ vpxor %ymm9,%ymm11,%ymm11+ andnq %rbx,%r11,%r12+ xorq %rdi,%r13+ rorxq $14,%r11,%r14+ vpsrlq $42,%ymm9,%ymm9+ vpxor %ymm10,%ymm11,%ymm11+ leaq (%rcx,%r12,1),%rcx+ xorq %r14,%r13+ movq %rdx,%rdi+ vpxor %ymm9,%ymm11,%ymm11+ rorxq $39,%rdx,%r12+ leaq (%rcx,%r13,1),%rcx+ xorq %r8,%rdi+ vpaddq %ymm11,%ymm2,%ymm2+ rorxq $34,%rdx,%r14+ rorxq $28,%rdx,%r13+ leaq (%r10,%rcx,1),%r10+ vpaddq -64(%rsi),%ymm2,%ymm10+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %r8,%r15+ xorq %r13,%r14+ leaq (%rcx,%r15,1),%rcx+ movq %r11,%r12+ vmovdqa %ymm10,64(%rsp)+ vpalignr $8,%ymm3,%ymm4,%ymm8+ addq 96+256(%rsp),%rbx+ andq %r10,%r12+ rorxq $41,%r10,%r13+ vpalignr $8,%ymm7,%ymm0,%ymm11+ rorxq $18,%r10,%r15+ leaq (%rcx,%r14,1),%rcx+ leaq (%rbx,%r12,1),%rbx+ vpsrlq $1,%ymm8,%ymm10+ andnq %rax,%r10,%r12+ xorq %r15,%r13+ rorxq $14,%r10,%r14+ vpaddq %ymm11,%ymm3,%ymm3+ vpsrlq $7,%ymm8,%ymm11+ leaq (%rbx,%r12,1),%rbx+ xorq %r14,%r13+ movq %rcx,%r15+ vpsllq $56,%ymm8,%ymm9+ vpxor %ymm10,%ymm11,%ymm8+ rorxq $39,%rcx,%r12+ leaq (%rbx,%r13,1),%rbx+ xorq %rdx,%r15+ vpsrlq $7,%ymm10,%ymm10+ vpxor %ymm9,%ymm8,%ymm8+ rorxq $34,%rcx,%r14+ rorxq $28,%rcx,%r13+ leaq (%r9,%rbx,1),%r9+ vpsllq $7,%ymm9,%ymm9+ vpxor %ymm10,%ymm8,%ymm8+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %rdx,%rdi+ vpsrlq $6,%ymm2,%ymm11+ vpxor %ymm9,%ymm8,%ymm8+ xorq %r13,%r14+ leaq (%rbx,%rdi,1),%rbx+ movq %r10,%r12+ vpsllq $3,%ymm2,%ymm10+ vpaddq %ymm8,%ymm3,%ymm3+ addq 104+256(%rsp),%rax+ andq %r9,%r12+ rorxq $41,%r9,%r13+ vpsrlq $19,%ymm2,%ymm9+ vpxor %ymm10,%ymm11,%ymm11+ rorxq $18,%r9,%rdi+ leaq (%rbx,%r14,1),%rbx+ leaq (%rax,%r12,1),%rax+ vpsllq $42,%ymm10,%ymm10+ vpxor %ymm9,%ymm11,%ymm11+ andnq %r11,%r9,%r12+ xorq %rdi,%r13+ rorxq $14,%r9,%r14+ vpsrlq $42,%ymm9,%ymm9+ vpxor %ymm10,%ymm11,%ymm11+ leaq (%rax,%r12,1),%rax+ xorq %r14,%r13+ movq %rbx,%rdi+ vpxor %ymm9,%ymm11,%ymm11+ rorxq $39,%rbx,%r12+ leaq (%rax,%r13,1),%rax+ xorq %rcx,%rdi+ vpaddq %ymm11,%ymm3,%ymm3+ rorxq $34,%rbx,%r14+ rorxq $28,%rbx,%r13+ leaq (%r8,%rax,1),%r8+ vpaddq -32(%rsi),%ymm3,%ymm10+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %rcx,%r15+ xorq %r13,%r14+ leaq (%rax,%r15,1),%rax+ movq %r9,%r12+ vmovdqa %ymm10,96(%rsp)+ leaq -128(%rsp),%rsp+ vpalignr $8,%ymm4,%ymm5,%ymm8+ addq 0+256(%rsp),%r11+ andq %r8,%r12+ rorxq $41,%r8,%r13+ vpalignr $8,%ymm0,%ymm1,%ymm11+ rorxq $18,%r8,%r15+ leaq (%rax,%r14,1),%rax+ leaq (%r11,%r12,1),%r11+ vpsrlq $1,%ymm8,%ymm10+ andnq %r10,%r8,%r12+ xorq %r15,%r13+ rorxq $14,%r8,%r14+ vpaddq %ymm11,%ymm4,%ymm4+ vpsrlq $7,%ymm8,%ymm11+ leaq (%r11,%r12,1),%r11+ xorq %r14,%r13+ movq %rax,%r15+ vpsllq $56,%ymm8,%ymm9+ vpxor %ymm10,%ymm11,%ymm8+ rorxq $39,%rax,%r12+ leaq (%r11,%r13,1),%r11+ xorq %rbx,%r15+ vpsrlq $7,%ymm10,%ymm10+ vpxor %ymm9,%ymm8,%ymm8+ rorxq $34,%rax,%r14+ rorxq $28,%rax,%r13+ leaq (%rdx,%r11,1),%rdx+ vpsllq $7,%ymm9,%ymm9+ vpxor %ymm10,%ymm8,%ymm8+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %rbx,%rdi+ vpsrlq $6,%ymm3,%ymm11+ vpxor %ymm9,%ymm8,%ymm8+ xorq %r13,%r14+ leaq (%r11,%rdi,1),%r11+ movq %r8,%r12+ vpsllq $3,%ymm3,%ymm10+ vpaddq %ymm8,%ymm4,%ymm4+ addq 8+256(%rsp),%r10+ andq %rdx,%r12+ rorxq $41,%rdx,%r13+ vpsrlq $19,%ymm3,%ymm9+ vpxor %ymm10,%ymm11,%ymm11+ rorxq $18,%rdx,%rdi+ leaq (%r11,%r14,1),%r11+ leaq (%r10,%r12,1),%r10+ vpsllq $42,%ymm10,%ymm10+ vpxor %ymm9,%ymm11,%ymm11+ andnq %r9,%rdx,%r12+ xorq %rdi,%r13+ rorxq $14,%rdx,%r14+ vpsrlq $42,%ymm9,%ymm9+ vpxor %ymm10,%ymm11,%ymm11+ leaq (%r10,%r12,1),%r10+ xorq %r14,%r13+ movq %r11,%rdi+ vpxor %ymm9,%ymm11,%ymm11+ rorxq $39,%r11,%r12+ leaq (%r10,%r13,1),%r10+ xorq %rax,%rdi+ vpaddq %ymm11,%ymm4,%ymm4+ rorxq $34,%r11,%r14+ rorxq $28,%r11,%r13+ leaq (%rcx,%r10,1),%rcx+ vpaddq 0(%rsi),%ymm4,%ymm10+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %rax,%r15+ xorq %r13,%r14+ leaq (%r10,%r15,1),%r10+ movq %rdx,%r12+ vmovdqa %ymm10,0(%rsp)+ vpalignr $8,%ymm5,%ymm6,%ymm8+ addq 32+256(%rsp),%r9+ andq %rcx,%r12+ rorxq $41,%rcx,%r13+ vpalignr $8,%ymm1,%ymm2,%ymm11+ rorxq $18,%rcx,%r15+ leaq (%r10,%r14,1),%r10+ leaq (%r9,%r12,1),%r9+ vpsrlq $1,%ymm8,%ymm10+ andnq %r8,%rcx,%r12+ xorq %r15,%r13+ rorxq $14,%rcx,%r14+ vpaddq %ymm11,%ymm5,%ymm5+ vpsrlq $7,%ymm8,%ymm11+ leaq (%r9,%r12,1),%r9+ xorq %r14,%r13+ movq %r10,%r15+ vpsllq $56,%ymm8,%ymm9+ vpxor %ymm10,%ymm11,%ymm8+ rorxq $39,%r10,%r12+ leaq (%r9,%r13,1),%r9+ xorq %r11,%r15+ vpsrlq $7,%ymm10,%ymm10+ vpxor %ymm9,%ymm8,%ymm8+ rorxq $34,%r10,%r14+ rorxq $28,%r10,%r13+ leaq (%rbx,%r9,1),%rbx+ vpsllq $7,%ymm9,%ymm9+ vpxor %ymm10,%ymm8,%ymm8+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %r11,%rdi+ vpsrlq $6,%ymm4,%ymm11+ vpxor %ymm9,%ymm8,%ymm8+ xorq %r13,%r14+ leaq (%r9,%rdi,1),%r9+ movq %rcx,%r12+ vpsllq $3,%ymm4,%ymm10+ vpaddq %ymm8,%ymm5,%ymm5+ addq 40+256(%rsp),%r8+ andq %rbx,%r12+ rorxq $41,%rbx,%r13+ vpsrlq $19,%ymm4,%ymm9+ vpxor %ymm10,%ymm11,%ymm11+ rorxq $18,%rbx,%rdi+ leaq (%r9,%r14,1),%r9+ leaq (%r8,%r12,1),%r8+ vpsllq $42,%ymm10,%ymm10+ vpxor %ymm9,%ymm11,%ymm11+ andnq %rdx,%rbx,%r12+ xorq %rdi,%r13+ rorxq $14,%rbx,%r14+ vpsrlq $42,%ymm9,%ymm9+ vpxor %ymm10,%ymm11,%ymm11+ leaq (%r8,%r12,1),%r8+ xorq %r14,%r13+ movq %r9,%rdi+ vpxor %ymm9,%ymm11,%ymm11+ rorxq $39,%r9,%r12+ leaq (%r8,%r13,1),%r8+ xorq %r10,%rdi+ vpaddq %ymm11,%ymm5,%ymm5+ rorxq $34,%r9,%r14+ rorxq $28,%r9,%r13+ leaq (%rax,%r8,1),%rax+ vpaddq 32(%rsi),%ymm5,%ymm10+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %r10,%r15+ xorq %r13,%r14+ leaq (%r8,%r15,1),%r8+ movq %rbx,%r12+ vmovdqa %ymm10,32(%rsp)+ vpalignr $8,%ymm6,%ymm7,%ymm8+ addq 64+256(%rsp),%rdx+ andq %rax,%r12+ rorxq $41,%rax,%r13+ vpalignr $8,%ymm2,%ymm3,%ymm11+ rorxq $18,%rax,%r15+ leaq (%r8,%r14,1),%r8+ leaq (%rdx,%r12,1),%rdx+ vpsrlq $1,%ymm8,%ymm10+ andnq %rcx,%rax,%r12+ xorq %r15,%r13+ rorxq $14,%rax,%r14+ vpaddq %ymm11,%ymm6,%ymm6+ vpsrlq $7,%ymm8,%ymm11+ leaq (%rdx,%r12,1),%rdx+ xorq %r14,%r13+ movq %r8,%r15+ vpsllq $56,%ymm8,%ymm9+ vpxor %ymm10,%ymm11,%ymm8+ rorxq $39,%r8,%r12+ leaq (%rdx,%r13,1),%rdx+ xorq %r9,%r15+ vpsrlq $7,%ymm10,%ymm10+ vpxor %ymm9,%ymm8,%ymm8+ rorxq $34,%r8,%r14+ rorxq $28,%r8,%r13+ leaq (%r11,%rdx,1),%r11+ vpsllq $7,%ymm9,%ymm9+ vpxor %ymm10,%ymm8,%ymm8+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %r9,%rdi+ vpsrlq $6,%ymm5,%ymm11+ vpxor %ymm9,%ymm8,%ymm8+ xorq %r13,%r14+ leaq (%rdx,%rdi,1),%rdx+ movq %rax,%r12+ vpsllq $3,%ymm5,%ymm10+ vpaddq %ymm8,%ymm6,%ymm6+ addq 72+256(%rsp),%rcx+ andq %r11,%r12+ rorxq $41,%r11,%r13+ vpsrlq $19,%ymm5,%ymm9+ vpxor %ymm10,%ymm11,%ymm11+ rorxq $18,%r11,%rdi+ leaq (%rdx,%r14,1),%rdx+ leaq (%rcx,%r12,1),%rcx+ vpsllq $42,%ymm10,%ymm10+ vpxor %ymm9,%ymm11,%ymm11+ andnq %rbx,%r11,%r12+ xorq %rdi,%r13+ rorxq $14,%r11,%r14+ vpsrlq $42,%ymm9,%ymm9+ vpxor %ymm10,%ymm11,%ymm11+ leaq (%rcx,%r12,1),%rcx+ xorq %r14,%r13+ movq %rdx,%rdi+ vpxor %ymm9,%ymm11,%ymm11+ rorxq $39,%rdx,%r12+ leaq (%rcx,%r13,1),%rcx+ xorq %r8,%rdi+ vpaddq %ymm11,%ymm6,%ymm6+ rorxq $34,%rdx,%r14+ rorxq $28,%rdx,%r13+ leaq (%r10,%rcx,1),%r10+ vpaddq 64(%rsi),%ymm6,%ymm10+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %r8,%r15+ xorq %r13,%r14+ leaq (%rcx,%r15,1),%rcx+ movq %r11,%r12+ vmovdqa %ymm10,64(%rsp)+ vpalignr $8,%ymm7,%ymm0,%ymm8+ addq 96+256(%rsp),%rbx+ andq %r10,%r12+ rorxq $41,%r10,%r13+ vpalignr $8,%ymm3,%ymm4,%ymm11+ rorxq $18,%r10,%r15+ leaq (%rcx,%r14,1),%rcx+ leaq (%rbx,%r12,1),%rbx+ vpsrlq $1,%ymm8,%ymm10+ andnq %rax,%r10,%r12+ xorq %r15,%r13+ rorxq $14,%r10,%r14+ vpaddq %ymm11,%ymm7,%ymm7+ vpsrlq $7,%ymm8,%ymm11+ leaq (%rbx,%r12,1),%rbx+ xorq %r14,%r13+ movq %rcx,%r15+ vpsllq $56,%ymm8,%ymm9+ vpxor %ymm10,%ymm11,%ymm8+ rorxq $39,%rcx,%r12+ leaq (%rbx,%r13,1),%rbx+ xorq %rdx,%r15+ vpsrlq $7,%ymm10,%ymm10+ vpxor %ymm9,%ymm8,%ymm8+ rorxq $34,%rcx,%r14+ rorxq $28,%rcx,%r13+ leaq (%r9,%rbx,1),%r9+ vpsllq $7,%ymm9,%ymm9+ vpxor %ymm10,%ymm8,%ymm8+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %rdx,%rdi+ vpsrlq $6,%ymm6,%ymm11+ vpxor %ymm9,%ymm8,%ymm8+ xorq %r13,%r14+ leaq (%rbx,%rdi,1),%rbx+ movq %r10,%r12+ vpsllq $3,%ymm6,%ymm10+ vpaddq %ymm8,%ymm7,%ymm7+ addq 104+256(%rsp),%rax+ andq %r9,%r12+ rorxq $41,%r9,%r13+ vpsrlq $19,%ymm6,%ymm9+ vpxor %ymm10,%ymm11,%ymm11+ rorxq $18,%r9,%rdi+ leaq (%rbx,%r14,1),%rbx+ leaq (%rax,%r12,1),%rax+ vpsllq $42,%ymm10,%ymm10+ vpxor %ymm9,%ymm11,%ymm11+ andnq %r11,%r9,%r12+ xorq %rdi,%r13+ rorxq $14,%r9,%r14+ vpsrlq $42,%ymm9,%ymm9+ vpxor %ymm10,%ymm11,%ymm11+ leaq (%rax,%r12,1),%rax+ xorq %r14,%r13+ movq %rbx,%rdi+ vpxor %ymm9,%ymm11,%ymm11+ rorxq $39,%rbx,%r12+ leaq (%rax,%r13,1),%rax+ xorq %rcx,%rdi+ vpaddq %ymm11,%ymm7,%ymm7+ rorxq $34,%rbx,%r14+ rorxq $28,%rbx,%r13+ leaq (%r8,%rax,1),%r8+ vpaddq 96(%rsi),%ymm7,%ymm10+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %rcx,%r15+ xorq %r13,%r14+ leaq (%rax,%r15,1),%rax+ movq %r9,%r12+ vmovdqa %ymm10,96(%rsp)+ leaq 256(%rsi),%rsi+ cmpb $0,-121(%rsi)+ jne .Lavx2_00_47+ addq 0+128(%rsp),%r11+ andq %r8,%r12+ rorxq $41,%r8,%r13+ rorxq $18,%r8,%r15+ leaq (%rax,%r14,1),%rax+ leaq (%r11,%r12,1),%r11+ andnq %r10,%r8,%r12+ xorq %r15,%r13+ rorxq $14,%r8,%r14+ leaq (%r11,%r12,1),%r11+ xorq %r14,%r13+ movq %rax,%r15+ rorxq $39,%rax,%r12+ leaq (%r11,%r13,1),%r11+ xorq %rbx,%r15+ rorxq $34,%rax,%r14+ rorxq $28,%rax,%r13+ leaq (%rdx,%r11,1),%rdx+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %rbx,%rdi+ xorq %r13,%r14+ leaq (%r11,%rdi,1),%r11+ movq %r8,%r12+ addq 8+128(%rsp),%r10+ andq %rdx,%r12+ rorxq $41,%rdx,%r13+ rorxq $18,%rdx,%rdi+ leaq (%r11,%r14,1),%r11+ leaq (%r10,%r12,1),%r10+ andnq %r9,%rdx,%r12+ xorq %rdi,%r13+ rorxq $14,%rdx,%r14+ leaq (%r10,%r12,1),%r10+ xorq %r14,%r13+ movq %r11,%rdi+ rorxq $39,%r11,%r12+ leaq (%r10,%r13,1),%r10+ xorq %rax,%rdi+ rorxq $34,%r11,%r14+ rorxq $28,%r11,%r13+ leaq (%rcx,%r10,1),%rcx+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %rax,%r15+ xorq %r13,%r14+ leaq (%r10,%r15,1),%r10+ movq %rdx,%r12+ addq 32+128(%rsp),%r9+ andq %rcx,%r12+ rorxq $41,%rcx,%r13+ rorxq $18,%rcx,%r15+ leaq (%r10,%r14,1),%r10+ leaq (%r9,%r12,1),%r9+ andnq %r8,%rcx,%r12+ xorq %r15,%r13+ rorxq $14,%rcx,%r14+ leaq (%r9,%r12,1),%r9+ xorq %r14,%r13+ movq %r10,%r15+ rorxq $39,%r10,%r12+ leaq (%r9,%r13,1),%r9+ xorq %r11,%r15+ rorxq $34,%r10,%r14+ rorxq $28,%r10,%r13+ leaq (%rbx,%r9,1),%rbx+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %r11,%rdi+ xorq %r13,%r14+ leaq (%r9,%rdi,1),%r9+ movq %rcx,%r12+ addq 40+128(%rsp),%r8+ andq %rbx,%r12+ rorxq $41,%rbx,%r13+ rorxq $18,%rbx,%rdi+ leaq (%r9,%r14,1),%r9+ leaq (%r8,%r12,1),%r8+ andnq %rdx,%rbx,%r12+ xorq %rdi,%r13+ rorxq $14,%rbx,%r14+ leaq (%r8,%r12,1),%r8+ xorq %r14,%r13+ movq %r9,%rdi+ rorxq $39,%r9,%r12+ leaq (%r8,%r13,1),%r8+ xorq %r10,%rdi+ rorxq $34,%r9,%r14+ rorxq $28,%r9,%r13+ leaq (%rax,%r8,1),%rax+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %r10,%r15+ xorq %r13,%r14+ leaq (%r8,%r15,1),%r8+ movq %rbx,%r12+ addq 64+128(%rsp),%rdx+ andq %rax,%r12+ rorxq $41,%rax,%r13+ rorxq $18,%rax,%r15+ leaq (%r8,%r14,1),%r8+ leaq (%rdx,%r12,1),%rdx+ andnq %rcx,%rax,%r12+ xorq %r15,%r13+ rorxq $14,%rax,%r14+ leaq (%rdx,%r12,1),%rdx+ xorq %r14,%r13+ movq %r8,%r15+ rorxq $39,%r8,%r12+ leaq (%rdx,%r13,1),%rdx+ xorq %r9,%r15+ rorxq $34,%r8,%r14+ rorxq $28,%r8,%r13+ leaq (%r11,%rdx,1),%r11+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %r9,%rdi+ xorq %r13,%r14+ leaq (%rdx,%rdi,1),%rdx+ movq %rax,%r12+ addq 72+128(%rsp),%rcx+ andq %r11,%r12+ rorxq $41,%r11,%r13+ rorxq $18,%r11,%rdi+ leaq (%rdx,%r14,1),%rdx+ leaq (%rcx,%r12,1),%rcx+ andnq %rbx,%r11,%r12+ xorq %rdi,%r13+ rorxq $14,%r11,%r14+ leaq (%rcx,%r12,1),%rcx+ xorq %r14,%r13+ movq %rdx,%rdi+ rorxq $39,%rdx,%r12+ leaq (%rcx,%r13,1),%rcx+ xorq %r8,%rdi+ rorxq $34,%rdx,%r14+ rorxq $28,%rdx,%r13+ leaq (%r10,%rcx,1),%r10+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %r8,%r15+ xorq %r13,%r14+ leaq (%rcx,%r15,1),%rcx+ movq %r11,%r12+ addq 96+128(%rsp),%rbx+ andq %r10,%r12+ rorxq $41,%r10,%r13+ rorxq $18,%r10,%r15+ leaq (%rcx,%r14,1),%rcx+ leaq (%rbx,%r12,1),%rbx+ andnq %rax,%r10,%r12+ xorq %r15,%r13+ rorxq $14,%r10,%r14+ leaq (%rbx,%r12,1),%rbx+ xorq %r14,%r13+ movq %rcx,%r15+ rorxq $39,%rcx,%r12+ leaq (%rbx,%r13,1),%rbx+ xorq %rdx,%r15+ rorxq $34,%rcx,%r14+ rorxq $28,%rcx,%r13+ leaq (%r9,%rbx,1),%r9+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %rdx,%rdi+ xorq %r13,%r14+ leaq (%rbx,%rdi,1),%rbx+ movq %r10,%r12+ addq 104+128(%rsp),%rax+ andq %r9,%r12+ rorxq $41,%r9,%r13+ rorxq $18,%r9,%rdi+ leaq (%rbx,%r14,1),%rbx+ leaq (%rax,%r12,1),%rax+ andnq %r11,%r9,%r12+ xorq %rdi,%r13+ rorxq $14,%r9,%r14+ leaq (%rax,%r12,1),%rax+ xorq %r14,%r13+ movq %rbx,%rdi+ rorxq $39,%rbx,%r12+ leaq (%rax,%r13,1),%rax+ xorq %rcx,%rdi+ rorxq $34,%rbx,%r14+ rorxq $28,%rbx,%r13+ leaq (%r8,%rax,1),%r8+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %rcx,%r15+ xorq %r13,%r14+ leaq (%rax,%r15,1),%rax+ movq %r9,%r12+ addq 0(%rsp),%r11+ andq %r8,%r12+ rorxq $41,%r8,%r13+ rorxq $18,%r8,%r15+ leaq (%rax,%r14,1),%rax+ leaq (%r11,%r12,1),%r11+ andnq %r10,%r8,%r12+ xorq %r15,%r13+ rorxq $14,%r8,%r14+ leaq (%r11,%r12,1),%r11+ xorq %r14,%r13+ movq %rax,%r15+ rorxq $39,%rax,%r12+ leaq (%r11,%r13,1),%r11+ xorq %rbx,%r15+ rorxq $34,%rax,%r14+ rorxq $28,%rax,%r13+ leaq (%rdx,%r11,1),%rdx+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %rbx,%rdi+ xorq %r13,%r14+ leaq (%r11,%rdi,1),%r11+ movq %r8,%r12+ addq 8(%rsp),%r10+ andq %rdx,%r12+ rorxq $41,%rdx,%r13+ rorxq $18,%rdx,%rdi+ leaq (%r11,%r14,1),%r11+ leaq (%r10,%r12,1),%r10+ andnq %r9,%rdx,%r12+ xorq %rdi,%r13+ rorxq $14,%rdx,%r14+ leaq (%r10,%r12,1),%r10+ xorq %r14,%r13+ movq %r11,%rdi+ rorxq $39,%r11,%r12+ leaq (%r10,%r13,1),%r10+ xorq %rax,%rdi+ rorxq $34,%r11,%r14+ rorxq $28,%r11,%r13+ leaq (%rcx,%r10,1),%rcx+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %rax,%r15+ xorq %r13,%r14+ leaq (%r10,%r15,1),%r10+ movq %rdx,%r12+ addq 32(%rsp),%r9+ andq %rcx,%r12+ rorxq $41,%rcx,%r13+ rorxq $18,%rcx,%r15+ leaq (%r10,%r14,1),%r10+ leaq (%r9,%r12,1),%r9+ andnq %r8,%rcx,%r12+ xorq %r15,%r13+ rorxq $14,%rcx,%r14+ leaq (%r9,%r12,1),%r9+ xorq %r14,%r13+ movq %r10,%r15+ rorxq $39,%r10,%r12+ leaq (%r9,%r13,1),%r9+ xorq %r11,%r15+ rorxq $34,%r10,%r14+ rorxq $28,%r10,%r13+ leaq (%rbx,%r9,1),%rbx+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %r11,%rdi+ xorq %r13,%r14+ leaq (%r9,%rdi,1),%r9+ movq %rcx,%r12+ addq 40(%rsp),%r8+ andq %rbx,%r12+ rorxq $41,%rbx,%r13+ rorxq $18,%rbx,%rdi+ leaq (%r9,%r14,1),%r9+ leaq (%r8,%r12,1),%r8+ andnq %rdx,%rbx,%r12+ xorq %rdi,%r13+ rorxq $14,%rbx,%r14+ leaq (%r8,%r12,1),%r8+ xorq %r14,%r13+ movq %r9,%rdi+ rorxq $39,%r9,%r12+ leaq (%r8,%r13,1),%r8+ xorq %r10,%rdi+ rorxq $34,%r9,%r14+ rorxq $28,%r9,%r13+ leaq (%rax,%r8,1),%rax+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %r10,%r15+ xorq %r13,%r14+ leaq (%r8,%r15,1),%r8+ movq %rbx,%r12+ addq 64(%rsp),%rdx+ andq %rax,%r12+ rorxq $41,%rax,%r13+ rorxq $18,%rax,%r15+ leaq (%r8,%r14,1),%r8+ leaq (%rdx,%r12,1),%rdx+ andnq %rcx,%rax,%r12+ xorq %r15,%r13+ rorxq $14,%rax,%r14+ leaq (%rdx,%r12,1),%rdx+ xorq %r14,%r13+ movq %r8,%r15+ rorxq $39,%r8,%r12+ leaq (%rdx,%r13,1),%rdx+ xorq %r9,%r15+ rorxq $34,%r8,%r14+ rorxq $28,%r8,%r13+ leaq (%r11,%rdx,1),%r11+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %r9,%rdi+ xorq %r13,%r14+ leaq (%rdx,%rdi,1),%rdx+ movq %rax,%r12+ addq 72(%rsp),%rcx+ andq %r11,%r12+ rorxq $41,%r11,%r13+ rorxq $18,%r11,%rdi+ leaq (%rdx,%r14,1),%rdx+ leaq (%rcx,%r12,1),%rcx+ andnq %rbx,%r11,%r12+ xorq %rdi,%r13+ rorxq $14,%r11,%r14+ leaq (%rcx,%r12,1),%rcx+ xorq %r14,%r13+ movq %rdx,%rdi+ rorxq $39,%rdx,%r12+ leaq (%rcx,%r13,1),%rcx+ xorq %r8,%rdi+ rorxq $34,%rdx,%r14+ rorxq $28,%rdx,%r13+ leaq (%r10,%rcx,1),%r10+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %r8,%r15+ xorq %r13,%r14+ leaq (%rcx,%r15,1),%rcx+ movq %r11,%r12+ addq 96(%rsp),%rbx+ andq %r10,%r12+ rorxq $41,%r10,%r13+ rorxq $18,%r10,%r15+ leaq (%rcx,%r14,1),%rcx+ leaq (%rbx,%r12,1),%rbx+ andnq %rax,%r10,%r12+ xorq %r15,%r13+ rorxq $14,%r10,%r14+ leaq (%rbx,%r12,1),%rbx+ xorq %r14,%r13+ movq %rcx,%r15+ rorxq $39,%rcx,%r12+ leaq (%rbx,%r13,1),%rbx+ xorq %rdx,%r15+ rorxq $34,%rcx,%r14+ rorxq $28,%rcx,%r13+ leaq (%r9,%rbx,1),%r9+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %rdx,%rdi+ xorq %r13,%r14+ leaq (%rbx,%rdi,1),%rbx+ movq %r10,%r12+ addq 104(%rsp),%rax+ andq %r9,%r12+ rorxq $41,%r9,%r13+ rorxq $18,%r9,%rdi+ leaq (%rbx,%r14,1),%rbx+ leaq (%rax,%r12,1),%rax+ andnq %r11,%r9,%r12+ xorq %rdi,%r13+ rorxq $14,%r9,%r14+ leaq (%rax,%r12,1),%rax+ xorq %r14,%r13+ movq %rbx,%rdi+ rorxq $39,%rbx,%r12+ leaq (%rax,%r13,1),%rax+ xorq %rcx,%rdi+ rorxq $34,%rbx,%r14+ rorxq $28,%rbx,%r13+ leaq (%r8,%rax,1),%r8+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %rcx,%r15+ xorq %r13,%r14+ leaq (%rax,%r15,1),%rax+ movq %r9,%r12+ movq -64(%rbp),%rdi+ addq %r14,%rax+ movq -56(%rbp),%r12++ addq 0(%rdi),%rax+ addq 8(%rdi),%rbx+ addq 16(%rdi),%rcx+ addq 24(%rdi),%rdx+ addq 32(%rdi),%r8+ addq 40(%rdi),%r9+ addq 48(%rdi),%r10+ addq 56(%rdi),%r11++ movq %rax,0(%rdi)+ movq %rbx,8(%rdi)+ movq %rcx,16(%rdi)+ movq %rdx,24(%rdi)+ movq %r8,32(%rdi)+ movq %r9,40(%rdi)+ movq %r10,48(%rdi)+ movq %r11,56(%rdi)++ cmpq -48(%rbp),%r12+ je .Ldone_avx2++ leaq 1152(%rsp),%rsi+ xorq %r14,%r14+ movq %rbx,%rdi+ xorq %rcx,%rdi+ movq %r9,%r12+ jmp .Lower_avx2+.align 16+.Lower_avx2:+ addq 0+16(%rsi),%r11+ andq %r8,%r12+ rorxq $41,%r8,%r13+ rorxq $18,%r8,%r15+ leaq (%rax,%r14,1),%rax+ leaq (%r11,%r12,1),%r11+ andnq %r10,%r8,%r12+ xorq %r15,%r13+ rorxq $14,%r8,%r14+ leaq (%r11,%r12,1),%r11+ xorq %r14,%r13+ movq %rax,%r15+ rorxq $39,%rax,%r12+ leaq (%r11,%r13,1),%r11+ xorq %rbx,%r15+ rorxq $34,%rax,%r14+ rorxq $28,%rax,%r13+ leaq (%rdx,%r11,1),%rdx+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %rbx,%rdi+ xorq %r13,%r14+ leaq (%r11,%rdi,1),%r11+ movq %r8,%r12+ addq 8+16(%rsi),%r10+ andq %rdx,%r12+ rorxq $41,%rdx,%r13+ rorxq $18,%rdx,%rdi+ leaq (%r11,%r14,1),%r11+ leaq (%r10,%r12,1),%r10+ andnq %r9,%rdx,%r12+ xorq %rdi,%r13+ rorxq $14,%rdx,%r14+ leaq (%r10,%r12,1),%r10+ xorq %r14,%r13+ movq %r11,%rdi+ rorxq $39,%r11,%r12+ leaq (%r10,%r13,1),%r10+ xorq %rax,%rdi+ rorxq $34,%r11,%r14+ rorxq $28,%r11,%r13+ leaq (%rcx,%r10,1),%rcx+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %rax,%r15+ xorq %r13,%r14+ leaq (%r10,%r15,1),%r10+ movq %rdx,%r12+ addq 32+16(%rsi),%r9+ andq %rcx,%r12+ rorxq $41,%rcx,%r13+ rorxq $18,%rcx,%r15+ leaq (%r10,%r14,1),%r10+ leaq (%r9,%r12,1),%r9+ andnq %r8,%rcx,%r12+ xorq %r15,%r13+ rorxq $14,%rcx,%r14+ leaq (%r9,%r12,1),%r9+ xorq %r14,%r13+ movq %r10,%r15+ rorxq $39,%r10,%r12+ leaq (%r9,%r13,1),%r9+ xorq %r11,%r15+ rorxq $34,%r10,%r14+ rorxq $28,%r10,%r13+ leaq (%rbx,%r9,1),%rbx+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %r11,%rdi+ xorq %r13,%r14+ leaq (%r9,%rdi,1),%r9+ movq %rcx,%r12+ addq 40+16(%rsi),%r8+ andq %rbx,%r12+ rorxq $41,%rbx,%r13+ rorxq $18,%rbx,%rdi+ leaq (%r9,%r14,1),%r9+ leaq (%r8,%r12,1),%r8+ andnq %rdx,%rbx,%r12+ xorq %rdi,%r13+ rorxq $14,%rbx,%r14+ leaq (%r8,%r12,1),%r8+ xorq %r14,%r13+ movq %r9,%rdi+ rorxq $39,%r9,%r12+ leaq (%r8,%r13,1),%r8+ xorq %r10,%rdi+ rorxq $34,%r9,%r14+ rorxq $28,%r9,%r13+ leaq (%rax,%r8,1),%rax+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %r10,%r15+ xorq %r13,%r14+ leaq (%r8,%r15,1),%r8+ movq %rbx,%r12+ addq 64+16(%rsi),%rdx+ andq %rax,%r12+ rorxq $41,%rax,%r13+ rorxq $18,%rax,%r15+ leaq (%r8,%r14,1),%r8+ leaq (%rdx,%r12,1),%rdx+ andnq %rcx,%rax,%r12+ xorq %r15,%r13+ rorxq $14,%rax,%r14+ leaq (%rdx,%r12,1),%rdx+ xorq %r14,%r13+ movq %r8,%r15+ rorxq $39,%r8,%r12+ leaq (%rdx,%r13,1),%rdx+ xorq %r9,%r15+ rorxq $34,%r8,%r14+ rorxq $28,%r8,%r13+ leaq (%r11,%rdx,1),%r11+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %r9,%rdi+ xorq %r13,%r14+ leaq (%rdx,%rdi,1),%rdx+ movq %rax,%r12+ addq 72+16(%rsi),%rcx+ andq %r11,%r12+ rorxq $41,%r11,%r13+ rorxq $18,%r11,%rdi+ leaq (%rdx,%r14,1),%rdx+ leaq (%rcx,%r12,1),%rcx+ andnq %rbx,%r11,%r12+ xorq %rdi,%r13+ rorxq $14,%r11,%r14+ leaq (%rcx,%r12,1),%rcx+ xorq %r14,%r13+ movq %rdx,%rdi+ rorxq $39,%rdx,%r12+ leaq (%rcx,%r13,1),%rcx+ xorq %r8,%rdi+ rorxq $34,%rdx,%r14+ rorxq $28,%rdx,%r13+ leaq (%r10,%rcx,1),%r10+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %r8,%r15+ xorq %r13,%r14+ leaq (%rcx,%r15,1),%rcx+ movq %r11,%r12+ addq 96+16(%rsi),%rbx+ andq %r10,%r12+ rorxq $41,%r10,%r13+ rorxq $18,%r10,%r15+ leaq (%rcx,%r14,1),%rcx+ leaq (%rbx,%r12,1),%rbx+ andnq %rax,%r10,%r12+ xorq %r15,%r13+ rorxq $14,%r10,%r14+ leaq (%rbx,%r12,1),%rbx+ xorq %r14,%r13+ movq %rcx,%r15+ rorxq $39,%rcx,%r12+ leaq (%rbx,%r13,1),%rbx+ xorq %rdx,%r15+ rorxq $34,%rcx,%r14+ rorxq $28,%rcx,%r13+ leaq (%r9,%rbx,1),%r9+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %rdx,%rdi+ xorq %r13,%r14+ leaq (%rbx,%rdi,1),%rbx+ movq %r10,%r12+ addq 104+16(%rsi),%rax+ andq %r9,%r12+ rorxq $41,%r9,%r13+ rorxq $18,%r9,%rdi+ leaq (%rbx,%r14,1),%rbx+ leaq (%rax,%r12,1),%rax+ andnq %r11,%r9,%r12+ xorq %rdi,%r13+ rorxq $14,%r9,%r14+ leaq (%rax,%r12,1),%rax+ xorq %r14,%r13+ movq %rbx,%rdi+ rorxq $39,%rbx,%r12+ leaq (%rax,%r13,1),%rax+ xorq %rcx,%rdi+ rorxq $34,%rbx,%r14+ rorxq $28,%rbx,%r13+ leaq (%r8,%rax,1),%r8+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %rcx,%r15+ xorq %r13,%r14+ leaq (%rax,%r15,1),%rax+ movq %r9,%r12+ leaq -128(%rsi),%rsi+ cmpq %rsp,%rsi+ jae .Lower_avx2++ movq -64(%rbp),%rdi+ addq %r14,%rax+ movq -56(%rbp),%rsi+ leaq 1152(%rsp),%rsp++ addq 0(%rdi),%rax+ addq 8(%rdi),%rbx+ addq 16(%rdi),%rcx+ addq 24(%rdi),%rdx+ addq 32(%rdi),%r8+ addq 40(%rdi),%r9+ leaq 256(%rsi),%rsi+ addq 48(%rdi),%r10+ movq %rsi,%r12+ addq 56(%rdi),%r11+ cmpq -48(%rbp),%rsi++ movq %rax,0(%rdi)+ cmoveq %rsp,%r12+ movq %rbx,8(%rdi)+ movq %rcx,16(%rdi)+ movq %rdx,24(%rdi)+ movq %r8,32(%rdi)+ movq %r9,40(%rdi)+ movq %r10,48(%rdi)+ movq %r11,56(%rdi)++ jbe .Loop_avx2++.Ldone_avx2:+ vzeroupper+ movq -40(%rbp),%r15+ movq -32(%rbp),%r14+ movq -24(%rbp),%r13+ movq -16(%rbp),%r12+ movq -8(%rbp),%rbx+ movq %rbp,%rsp+.cfi_def_cfa_register %rsp+ popq %rbp+.cfi_adjust_cfa_offset -8+.cfi_restore %rbp+.cfi_restore %r12+.cfi_restore %r13+.cfi_restore %r14+.cfi_restore %r15+.cfi_restore %rbx+ .byte 0xf3,0xc3+.cfi_endproc +.size crypton_sha512_asm_block_data_order_avx2,.-crypton_sha512_asm_block_data_order_avx2++.section .note.gnu.property,"a",@note+ .long 4,2f-1f,5+ .byte 0x47,0x4E,0x55,0+1: .long 0xc0000002,4,3+.align 8+2:++.section .note.GNU-stack,"",@progbits
@@ -0,0 +1,5718 @@+.text +++.globl _crypton_sha512_asm_block_data_order++.p2align 4+_crypton_sha512_asm_block_data_order:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa+++ pushq %rbp+.cfi_adjust_cfa_offset 8+.cfi_offset %rbp,-16+ movq %rsp,%rbp+.cfi_def_cfa_register %rbp+ leaq _crypton_ia32cap_P(%rip),%rax+ movl 0(%rax),%r9d+ movl 4(%rax),%r10d+ movl 8(%rax),%eax+ testl $2048,%r10d+ jnz L$xop_shortcut+ andl $296,%eax+ cmpl $296,%eax+ je L$avx2_shortcut+ andl $1073741824,%r9d+ andl $268435968,%r10d+ orl %r9d,%r10d+ cmpl $1342177792,%r10d+ je L$avx_shortcut+ pushq %rbx+.cfi_offset %rbx,-24+ pushq %r12+.cfi_offset %r12,-32+ pushq %r13+.cfi_offset %r13,-40+ pushq %r14+.cfi_offset %r14,-48+ pushq %r15+.cfi_offset %r15,-56+ shlq $4,%rdx+ subq $128+24,%rsp++.cfi_def_cfa %rsp,208++ leaq (%rsi,%rdx,8),%rdx+ movq %rdi,128+0(%rsp)+ movq %rsi,128+8(%rsp)+ movq %rdx,128+16(%rsp)++ movq 0(%rdi),%rax+ movq 8(%rdi),%rbx+ movq 16(%rdi),%rcx+ movq 24(%rdi),%rdx+ movq 32(%rdi),%r8+ movq 40(%rdi),%r9+ movq 48(%rdi),%r10+ movq 56(%rdi),%r11+ jmp L$loop++.p2align 4+L$loop:+ movq %rbx,%rdi+ leaq K512(%rip),%rbp+ xorq %rcx,%rdi+ movq 0(%rsi),%r12+ movq %r8,%r13+ movq %rax,%r14+ bswapq %r12+ rorq $23,%r13+ movq %r9,%r15++ xorq %r8,%r13+ rorq $5,%r14+ xorq %r10,%r15++ movq %r12,0(%rsp)+ xorq %rax,%r14+ andq %r8,%r15++ rorq $4,%r13+ addq %r11,%r12+ xorq %r10,%r15++ rorq $6,%r14+ xorq %r8,%r13+ addq %r15,%r12++ movq %rax,%r15+ addq (%rbp),%r12+ xorq %rax,%r14++ xorq %rbx,%r15+ rorq $14,%r13+ movq %rbx,%r11++ andq %r15,%rdi+ rorq $28,%r14+ addq %r13,%r12++ xorq %rdi,%r11+ addq %r12,%rdx+ addq %r12,%r11++ leaq 8(%rbp),%rbp+ addq %r14,%r11+ movq 8(%rsi),%r12+ movq %rdx,%r13+ movq %r11,%r14+ bswapq %r12+ rorq $23,%r13+ movq %r8,%rdi++ xorq %rdx,%r13+ rorq $5,%r14+ xorq %r9,%rdi++ movq %r12,8(%rsp)+ xorq %r11,%r14+ andq %rdx,%rdi++ rorq $4,%r13+ addq %r10,%r12+ xorq %r9,%rdi++ rorq $6,%r14+ xorq %rdx,%r13+ addq %rdi,%r12++ movq %r11,%rdi+ addq (%rbp),%r12+ xorq %r11,%r14++ xorq %rax,%rdi+ rorq $14,%r13+ movq %rax,%r10++ andq %rdi,%r15+ rorq $28,%r14+ addq %r13,%r12++ xorq %r15,%r10+ addq %r12,%rcx+ addq %r12,%r10++ leaq 24(%rbp),%rbp+ addq %r14,%r10+ movq 16(%rsi),%r12+ movq %rcx,%r13+ movq %r10,%r14+ bswapq %r12+ rorq $23,%r13+ movq %rdx,%r15++ xorq %rcx,%r13+ rorq $5,%r14+ xorq %r8,%r15++ movq %r12,16(%rsp)+ xorq %r10,%r14+ andq %rcx,%r15++ rorq $4,%r13+ addq %r9,%r12+ xorq %r8,%r15++ rorq $6,%r14+ xorq %rcx,%r13+ addq %r15,%r12++ movq %r10,%r15+ addq (%rbp),%r12+ xorq %r10,%r14++ xorq %r11,%r15+ rorq $14,%r13+ movq %r11,%r9++ andq %r15,%rdi+ rorq $28,%r14+ addq %r13,%r12++ xorq %rdi,%r9+ addq %r12,%rbx+ addq %r12,%r9++ leaq 8(%rbp),%rbp+ addq %r14,%r9+ movq 24(%rsi),%r12+ movq %rbx,%r13+ movq %r9,%r14+ bswapq %r12+ rorq $23,%r13+ movq %rcx,%rdi++ xorq %rbx,%r13+ rorq $5,%r14+ xorq %rdx,%rdi++ movq %r12,24(%rsp)+ xorq %r9,%r14+ andq %rbx,%rdi++ rorq $4,%r13+ addq %r8,%r12+ xorq %rdx,%rdi++ rorq $6,%r14+ xorq %rbx,%r13+ addq %rdi,%r12++ movq %r9,%rdi+ addq (%rbp),%r12+ xorq %r9,%r14++ xorq %r10,%rdi+ rorq $14,%r13+ movq %r10,%r8++ andq %rdi,%r15+ rorq $28,%r14+ addq %r13,%r12++ xorq %r15,%r8+ addq %r12,%rax+ addq %r12,%r8++ leaq 24(%rbp),%rbp+ addq %r14,%r8+ movq 32(%rsi),%r12+ movq %rax,%r13+ movq %r8,%r14+ bswapq %r12+ rorq $23,%r13+ movq %rbx,%r15++ xorq %rax,%r13+ rorq $5,%r14+ xorq %rcx,%r15++ movq %r12,32(%rsp)+ xorq %r8,%r14+ andq %rax,%r15++ rorq $4,%r13+ addq %rdx,%r12+ xorq %rcx,%r15++ rorq $6,%r14+ xorq %rax,%r13+ addq %r15,%r12++ movq %r8,%r15+ addq (%rbp),%r12+ xorq %r8,%r14++ xorq %r9,%r15+ rorq $14,%r13+ movq %r9,%rdx++ andq %r15,%rdi+ rorq $28,%r14+ addq %r13,%r12++ xorq %rdi,%rdx+ addq %r12,%r11+ addq %r12,%rdx++ leaq 8(%rbp),%rbp+ addq %r14,%rdx+ movq 40(%rsi),%r12+ movq %r11,%r13+ movq %rdx,%r14+ bswapq %r12+ rorq $23,%r13+ movq %rax,%rdi++ xorq %r11,%r13+ rorq $5,%r14+ xorq %rbx,%rdi++ movq %r12,40(%rsp)+ xorq %rdx,%r14+ andq %r11,%rdi++ rorq $4,%r13+ addq %rcx,%r12+ xorq %rbx,%rdi++ rorq $6,%r14+ xorq %r11,%r13+ addq %rdi,%r12++ movq %rdx,%rdi+ addq (%rbp),%r12+ xorq %rdx,%r14++ xorq %r8,%rdi+ rorq $14,%r13+ movq %r8,%rcx++ andq %rdi,%r15+ rorq $28,%r14+ addq %r13,%r12++ xorq %r15,%rcx+ addq %r12,%r10+ addq %r12,%rcx++ leaq 24(%rbp),%rbp+ addq %r14,%rcx+ movq 48(%rsi),%r12+ movq %r10,%r13+ movq %rcx,%r14+ bswapq %r12+ rorq $23,%r13+ movq %r11,%r15++ xorq %r10,%r13+ rorq $5,%r14+ xorq %rax,%r15++ movq %r12,48(%rsp)+ xorq %rcx,%r14+ andq %r10,%r15++ rorq $4,%r13+ addq %rbx,%r12+ xorq %rax,%r15++ rorq $6,%r14+ xorq %r10,%r13+ addq %r15,%r12++ movq %rcx,%r15+ addq (%rbp),%r12+ xorq %rcx,%r14++ xorq %rdx,%r15+ rorq $14,%r13+ movq %rdx,%rbx++ andq %r15,%rdi+ rorq $28,%r14+ addq %r13,%r12++ xorq %rdi,%rbx+ addq %r12,%r9+ addq %r12,%rbx++ leaq 8(%rbp),%rbp+ addq %r14,%rbx+ movq 56(%rsi),%r12+ movq %r9,%r13+ movq %rbx,%r14+ bswapq %r12+ rorq $23,%r13+ movq %r10,%rdi++ xorq %r9,%r13+ rorq $5,%r14+ xorq %r11,%rdi++ movq %r12,56(%rsp)+ xorq %rbx,%r14+ andq %r9,%rdi++ rorq $4,%r13+ addq %rax,%r12+ xorq %r11,%rdi++ rorq $6,%r14+ xorq %r9,%r13+ addq %rdi,%r12++ movq %rbx,%rdi+ addq (%rbp),%r12+ xorq %rbx,%r14++ xorq %rcx,%rdi+ rorq $14,%r13+ movq %rcx,%rax++ andq %rdi,%r15+ rorq $28,%r14+ addq %r13,%r12++ xorq %r15,%rax+ addq %r12,%r8+ addq %r12,%rax++ leaq 24(%rbp),%rbp+ addq %r14,%rax+ movq 64(%rsi),%r12+ movq %r8,%r13+ movq %rax,%r14+ bswapq %r12+ rorq $23,%r13+ movq %r9,%r15++ xorq %r8,%r13+ rorq $5,%r14+ xorq %r10,%r15++ movq %r12,64(%rsp)+ xorq %rax,%r14+ andq %r8,%r15++ rorq $4,%r13+ addq %r11,%r12+ xorq %r10,%r15++ rorq $6,%r14+ xorq %r8,%r13+ addq %r15,%r12++ movq %rax,%r15+ addq (%rbp),%r12+ xorq %rax,%r14++ xorq %rbx,%r15+ rorq $14,%r13+ movq %rbx,%r11++ andq %r15,%rdi+ rorq $28,%r14+ addq %r13,%r12++ xorq %rdi,%r11+ addq %r12,%rdx+ addq %r12,%r11++ leaq 8(%rbp),%rbp+ addq %r14,%r11+ movq 72(%rsi),%r12+ movq %rdx,%r13+ movq %r11,%r14+ bswapq %r12+ rorq $23,%r13+ movq %r8,%rdi++ xorq %rdx,%r13+ rorq $5,%r14+ xorq %r9,%rdi++ movq %r12,72(%rsp)+ xorq %r11,%r14+ andq %rdx,%rdi++ rorq $4,%r13+ addq %r10,%r12+ xorq %r9,%rdi++ rorq $6,%r14+ xorq %rdx,%r13+ addq %rdi,%r12++ movq %r11,%rdi+ addq (%rbp),%r12+ xorq %r11,%r14++ xorq %rax,%rdi+ rorq $14,%r13+ movq %rax,%r10++ andq %rdi,%r15+ rorq $28,%r14+ addq %r13,%r12++ xorq %r15,%r10+ addq %r12,%rcx+ addq %r12,%r10++ leaq 24(%rbp),%rbp+ addq %r14,%r10+ movq 80(%rsi),%r12+ movq %rcx,%r13+ movq %r10,%r14+ bswapq %r12+ rorq $23,%r13+ movq %rdx,%r15++ xorq %rcx,%r13+ rorq $5,%r14+ xorq %r8,%r15++ movq %r12,80(%rsp)+ xorq %r10,%r14+ andq %rcx,%r15++ rorq $4,%r13+ addq %r9,%r12+ xorq %r8,%r15++ rorq $6,%r14+ xorq %rcx,%r13+ addq %r15,%r12++ movq %r10,%r15+ addq (%rbp),%r12+ xorq %r10,%r14++ xorq %r11,%r15+ rorq $14,%r13+ movq %r11,%r9++ andq %r15,%rdi+ rorq $28,%r14+ addq %r13,%r12++ xorq %rdi,%r9+ addq %r12,%rbx+ addq %r12,%r9++ leaq 8(%rbp),%rbp+ addq %r14,%r9+ movq 88(%rsi),%r12+ movq %rbx,%r13+ movq %r9,%r14+ bswapq %r12+ rorq $23,%r13+ movq %rcx,%rdi++ xorq %rbx,%r13+ rorq $5,%r14+ xorq %rdx,%rdi++ movq %r12,88(%rsp)+ xorq %r9,%r14+ andq %rbx,%rdi++ rorq $4,%r13+ addq %r8,%r12+ xorq %rdx,%rdi++ rorq $6,%r14+ xorq %rbx,%r13+ addq %rdi,%r12++ movq %r9,%rdi+ addq (%rbp),%r12+ xorq %r9,%r14++ xorq %r10,%rdi+ rorq $14,%r13+ movq %r10,%r8++ andq %rdi,%r15+ rorq $28,%r14+ addq %r13,%r12++ xorq %r15,%r8+ addq %r12,%rax+ addq %r12,%r8++ leaq 24(%rbp),%rbp+ addq %r14,%r8+ movq 96(%rsi),%r12+ movq %rax,%r13+ movq %r8,%r14+ bswapq %r12+ rorq $23,%r13+ movq %rbx,%r15++ xorq %rax,%r13+ rorq $5,%r14+ xorq %rcx,%r15++ movq %r12,96(%rsp)+ xorq %r8,%r14+ andq %rax,%r15++ rorq $4,%r13+ addq %rdx,%r12+ xorq %rcx,%r15++ rorq $6,%r14+ xorq %rax,%r13+ addq %r15,%r12++ movq %r8,%r15+ addq (%rbp),%r12+ xorq %r8,%r14++ xorq %r9,%r15+ rorq $14,%r13+ movq %r9,%rdx++ andq %r15,%rdi+ rorq $28,%r14+ addq %r13,%r12++ xorq %rdi,%rdx+ addq %r12,%r11+ addq %r12,%rdx++ leaq 8(%rbp),%rbp+ addq %r14,%rdx+ movq 104(%rsi),%r12+ movq %r11,%r13+ movq %rdx,%r14+ bswapq %r12+ rorq $23,%r13+ movq %rax,%rdi++ xorq %r11,%r13+ rorq $5,%r14+ xorq %rbx,%rdi++ movq %r12,104(%rsp)+ xorq %rdx,%r14+ andq %r11,%rdi++ rorq $4,%r13+ addq %rcx,%r12+ xorq %rbx,%rdi++ rorq $6,%r14+ xorq %r11,%r13+ addq %rdi,%r12++ movq %rdx,%rdi+ addq (%rbp),%r12+ xorq %rdx,%r14++ xorq %r8,%rdi+ rorq $14,%r13+ movq %r8,%rcx++ andq %rdi,%r15+ rorq $28,%r14+ addq %r13,%r12++ xorq %r15,%rcx+ addq %r12,%r10+ addq %r12,%rcx++ leaq 24(%rbp),%rbp+ addq %r14,%rcx+ movq 112(%rsi),%r12+ movq %r10,%r13+ movq %rcx,%r14+ bswapq %r12+ rorq $23,%r13+ movq %r11,%r15++ xorq %r10,%r13+ rorq $5,%r14+ xorq %rax,%r15++ movq %r12,112(%rsp)+ xorq %rcx,%r14+ andq %r10,%r15++ rorq $4,%r13+ addq %rbx,%r12+ xorq %rax,%r15++ rorq $6,%r14+ xorq %r10,%r13+ addq %r15,%r12++ movq %rcx,%r15+ addq (%rbp),%r12+ xorq %rcx,%r14++ xorq %rdx,%r15+ rorq $14,%r13+ movq %rdx,%rbx++ andq %r15,%rdi+ rorq $28,%r14+ addq %r13,%r12++ xorq %rdi,%rbx+ addq %r12,%r9+ addq %r12,%rbx++ leaq 8(%rbp),%rbp+ addq %r14,%rbx+ movq 120(%rsi),%r12+ movq %r9,%r13+ movq %rbx,%r14+ bswapq %r12+ rorq $23,%r13+ movq %r10,%rdi++ xorq %r9,%r13+ rorq $5,%r14+ xorq %r11,%rdi++ movq %r12,120(%rsp)+ xorq %rbx,%r14+ andq %r9,%rdi++ rorq $4,%r13+ addq %rax,%r12+ xorq %r11,%rdi++ rorq $6,%r14+ xorq %r9,%r13+ addq %rdi,%r12++ movq %rbx,%rdi+ addq (%rbp),%r12+ xorq %rbx,%r14++ xorq %rcx,%rdi+ rorq $14,%r13+ movq %rcx,%rax++ andq %rdi,%r15+ rorq $28,%r14+ addq %r13,%r12++ xorq %r15,%rax+ addq %r12,%r8+ addq %r12,%rax++ leaq 24(%rbp),%rbp+ jmp L$rounds_16_xx+.p2align 4+L$rounds_16_xx:+ movq 8(%rsp),%r13+ movq 112(%rsp),%r15++ movq %r13,%r12+ rorq $7,%r13+ addq %r14,%rax+ movq %r15,%r14+ rorq $42,%r15++ xorq %r12,%r13+ shrq $7,%r12+ rorq $1,%r13+ xorq %r14,%r15+ shrq $6,%r14++ rorq $19,%r15+ xorq %r13,%r12+ xorq %r14,%r15+ addq 72(%rsp),%r12++ addq 0(%rsp),%r12+ movq %r8,%r13+ addq %r15,%r12+ movq %rax,%r14+ rorq $23,%r13+ movq %r9,%r15++ xorq %r8,%r13+ rorq $5,%r14+ xorq %r10,%r15++ movq %r12,0(%rsp)+ xorq %rax,%r14+ andq %r8,%r15++ rorq $4,%r13+ addq %r11,%r12+ xorq %r10,%r15++ rorq $6,%r14+ xorq %r8,%r13+ addq %r15,%r12++ movq %rax,%r15+ addq (%rbp),%r12+ xorq %rax,%r14++ xorq %rbx,%r15+ rorq $14,%r13+ movq %rbx,%r11++ andq %r15,%rdi+ rorq $28,%r14+ addq %r13,%r12++ xorq %rdi,%r11+ addq %r12,%rdx+ addq %r12,%r11++ leaq 8(%rbp),%rbp+ movq 16(%rsp),%r13+ movq 120(%rsp),%rdi++ movq %r13,%r12+ rorq $7,%r13+ addq %r14,%r11+ movq %rdi,%r14+ rorq $42,%rdi++ xorq %r12,%r13+ shrq $7,%r12+ rorq $1,%r13+ xorq %r14,%rdi+ shrq $6,%r14++ rorq $19,%rdi+ xorq %r13,%r12+ xorq %r14,%rdi+ addq 80(%rsp),%r12++ addq 8(%rsp),%r12+ movq %rdx,%r13+ addq %rdi,%r12+ movq %r11,%r14+ rorq $23,%r13+ movq %r8,%rdi++ xorq %rdx,%r13+ rorq $5,%r14+ xorq %r9,%rdi++ movq %r12,8(%rsp)+ xorq %r11,%r14+ andq %rdx,%rdi++ rorq $4,%r13+ addq %r10,%r12+ xorq %r9,%rdi++ rorq $6,%r14+ xorq %rdx,%r13+ addq %rdi,%r12++ movq %r11,%rdi+ addq (%rbp),%r12+ xorq %r11,%r14++ xorq %rax,%rdi+ rorq $14,%r13+ movq %rax,%r10++ andq %rdi,%r15+ rorq $28,%r14+ addq %r13,%r12++ xorq %r15,%r10+ addq %r12,%rcx+ addq %r12,%r10++ leaq 24(%rbp),%rbp+ movq 24(%rsp),%r13+ movq 0(%rsp),%r15++ movq %r13,%r12+ rorq $7,%r13+ addq %r14,%r10+ movq %r15,%r14+ rorq $42,%r15++ xorq %r12,%r13+ shrq $7,%r12+ rorq $1,%r13+ xorq %r14,%r15+ shrq $6,%r14++ rorq $19,%r15+ xorq %r13,%r12+ xorq %r14,%r15+ addq 88(%rsp),%r12++ addq 16(%rsp),%r12+ movq %rcx,%r13+ addq %r15,%r12+ movq %r10,%r14+ rorq $23,%r13+ movq %rdx,%r15++ xorq %rcx,%r13+ rorq $5,%r14+ xorq %r8,%r15++ movq %r12,16(%rsp)+ xorq %r10,%r14+ andq %rcx,%r15++ rorq $4,%r13+ addq %r9,%r12+ xorq %r8,%r15++ rorq $6,%r14+ xorq %rcx,%r13+ addq %r15,%r12++ movq %r10,%r15+ addq (%rbp),%r12+ xorq %r10,%r14++ xorq %r11,%r15+ rorq $14,%r13+ movq %r11,%r9++ andq %r15,%rdi+ rorq $28,%r14+ addq %r13,%r12++ xorq %rdi,%r9+ addq %r12,%rbx+ addq %r12,%r9++ leaq 8(%rbp),%rbp+ movq 32(%rsp),%r13+ movq 8(%rsp),%rdi++ movq %r13,%r12+ rorq $7,%r13+ addq %r14,%r9+ movq %rdi,%r14+ rorq $42,%rdi++ xorq %r12,%r13+ shrq $7,%r12+ rorq $1,%r13+ xorq %r14,%rdi+ shrq $6,%r14++ rorq $19,%rdi+ xorq %r13,%r12+ xorq %r14,%rdi+ addq 96(%rsp),%r12++ addq 24(%rsp),%r12+ movq %rbx,%r13+ addq %rdi,%r12+ movq %r9,%r14+ rorq $23,%r13+ movq %rcx,%rdi++ xorq %rbx,%r13+ rorq $5,%r14+ xorq %rdx,%rdi++ movq %r12,24(%rsp)+ xorq %r9,%r14+ andq %rbx,%rdi++ rorq $4,%r13+ addq %r8,%r12+ xorq %rdx,%rdi++ rorq $6,%r14+ xorq %rbx,%r13+ addq %rdi,%r12++ movq %r9,%rdi+ addq (%rbp),%r12+ xorq %r9,%r14++ xorq %r10,%rdi+ rorq $14,%r13+ movq %r10,%r8++ andq %rdi,%r15+ rorq $28,%r14+ addq %r13,%r12++ xorq %r15,%r8+ addq %r12,%rax+ addq %r12,%r8++ leaq 24(%rbp),%rbp+ movq 40(%rsp),%r13+ movq 16(%rsp),%r15++ movq %r13,%r12+ rorq $7,%r13+ addq %r14,%r8+ movq %r15,%r14+ rorq $42,%r15++ xorq %r12,%r13+ shrq $7,%r12+ rorq $1,%r13+ xorq %r14,%r15+ shrq $6,%r14++ rorq $19,%r15+ xorq %r13,%r12+ xorq %r14,%r15+ addq 104(%rsp),%r12++ addq 32(%rsp),%r12+ movq %rax,%r13+ addq %r15,%r12+ movq %r8,%r14+ rorq $23,%r13+ movq %rbx,%r15++ xorq %rax,%r13+ rorq $5,%r14+ xorq %rcx,%r15++ movq %r12,32(%rsp)+ xorq %r8,%r14+ andq %rax,%r15++ rorq $4,%r13+ addq %rdx,%r12+ xorq %rcx,%r15++ rorq $6,%r14+ xorq %rax,%r13+ addq %r15,%r12++ movq %r8,%r15+ addq (%rbp),%r12+ xorq %r8,%r14++ xorq %r9,%r15+ rorq $14,%r13+ movq %r9,%rdx++ andq %r15,%rdi+ rorq $28,%r14+ addq %r13,%r12++ xorq %rdi,%rdx+ addq %r12,%r11+ addq %r12,%rdx++ leaq 8(%rbp),%rbp+ movq 48(%rsp),%r13+ movq 24(%rsp),%rdi++ movq %r13,%r12+ rorq $7,%r13+ addq %r14,%rdx+ movq %rdi,%r14+ rorq $42,%rdi++ xorq %r12,%r13+ shrq $7,%r12+ rorq $1,%r13+ xorq %r14,%rdi+ shrq $6,%r14++ rorq $19,%rdi+ xorq %r13,%r12+ xorq %r14,%rdi+ addq 112(%rsp),%r12++ addq 40(%rsp),%r12+ movq %r11,%r13+ addq %rdi,%r12+ movq %rdx,%r14+ rorq $23,%r13+ movq %rax,%rdi++ xorq %r11,%r13+ rorq $5,%r14+ xorq %rbx,%rdi++ movq %r12,40(%rsp)+ xorq %rdx,%r14+ andq %r11,%rdi++ rorq $4,%r13+ addq %rcx,%r12+ xorq %rbx,%rdi++ rorq $6,%r14+ xorq %r11,%r13+ addq %rdi,%r12++ movq %rdx,%rdi+ addq (%rbp),%r12+ xorq %rdx,%r14++ xorq %r8,%rdi+ rorq $14,%r13+ movq %r8,%rcx++ andq %rdi,%r15+ rorq $28,%r14+ addq %r13,%r12++ xorq %r15,%rcx+ addq %r12,%r10+ addq %r12,%rcx++ leaq 24(%rbp),%rbp+ movq 56(%rsp),%r13+ movq 32(%rsp),%r15++ movq %r13,%r12+ rorq $7,%r13+ addq %r14,%rcx+ movq %r15,%r14+ rorq $42,%r15++ xorq %r12,%r13+ shrq $7,%r12+ rorq $1,%r13+ xorq %r14,%r15+ shrq $6,%r14++ rorq $19,%r15+ xorq %r13,%r12+ xorq %r14,%r15+ addq 120(%rsp),%r12++ addq 48(%rsp),%r12+ movq %r10,%r13+ addq %r15,%r12+ movq %rcx,%r14+ rorq $23,%r13+ movq %r11,%r15++ xorq %r10,%r13+ rorq $5,%r14+ xorq %rax,%r15++ movq %r12,48(%rsp)+ xorq %rcx,%r14+ andq %r10,%r15++ rorq $4,%r13+ addq %rbx,%r12+ xorq %rax,%r15++ rorq $6,%r14+ xorq %r10,%r13+ addq %r15,%r12++ movq %rcx,%r15+ addq (%rbp),%r12+ xorq %rcx,%r14++ xorq %rdx,%r15+ rorq $14,%r13+ movq %rdx,%rbx++ andq %r15,%rdi+ rorq $28,%r14+ addq %r13,%r12++ xorq %rdi,%rbx+ addq %r12,%r9+ addq %r12,%rbx++ leaq 8(%rbp),%rbp+ movq 64(%rsp),%r13+ movq 40(%rsp),%rdi++ movq %r13,%r12+ rorq $7,%r13+ addq %r14,%rbx+ movq %rdi,%r14+ rorq $42,%rdi++ xorq %r12,%r13+ shrq $7,%r12+ rorq $1,%r13+ xorq %r14,%rdi+ shrq $6,%r14++ rorq $19,%rdi+ xorq %r13,%r12+ xorq %r14,%rdi+ addq 0(%rsp),%r12++ addq 56(%rsp),%r12+ movq %r9,%r13+ addq %rdi,%r12+ movq %rbx,%r14+ rorq $23,%r13+ movq %r10,%rdi++ xorq %r9,%r13+ rorq $5,%r14+ xorq %r11,%rdi++ movq %r12,56(%rsp)+ xorq %rbx,%r14+ andq %r9,%rdi++ rorq $4,%r13+ addq %rax,%r12+ xorq %r11,%rdi++ rorq $6,%r14+ xorq %r9,%r13+ addq %rdi,%r12++ movq %rbx,%rdi+ addq (%rbp),%r12+ xorq %rbx,%r14++ xorq %rcx,%rdi+ rorq $14,%r13+ movq %rcx,%rax++ andq %rdi,%r15+ rorq $28,%r14+ addq %r13,%r12++ xorq %r15,%rax+ addq %r12,%r8+ addq %r12,%rax++ leaq 24(%rbp),%rbp+ movq 72(%rsp),%r13+ movq 48(%rsp),%r15++ movq %r13,%r12+ rorq $7,%r13+ addq %r14,%rax+ movq %r15,%r14+ rorq $42,%r15++ xorq %r12,%r13+ shrq $7,%r12+ rorq $1,%r13+ xorq %r14,%r15+ shrq $6,%r14++ rorq $19,%r15+ xorq %r13,%r12+ xorq %r14,%r15+ addq 8(%rsp),%r12++ addq 64(%rsp),%r12+ movq %r8,%r13+ addq %r15,%r12+ movq %rax,%r14+ rorq $23,%r13+ movq %r9,%r15++ xorq %r8,%r13+ rorq $5,%r14+ xorq %r10,%r15++ movq %r12,64(%rsp)+ xorq %rax,%r14+ andq %r8,%r15++ rorq $4,%r13+ addq %r11,%r12+ xorq %r10,%r15++ rorq $6,%r14+ xorq %r8,%r13+ addq %r15,%r12++ movq %rax,%r15+ addq (%rbp),%r12+ xorq %rax,%r14++ xorq %rbx,%r15+ rorq $14,%r13+ movq %rbx,%r11++ andq %r15,%rdi+ rorq $28,%r14+ addq %r13,%r12++ xorq %rdi,%r11+ addq %r12,%rdx+ addq %r12,%r11++ leaq 8(%rbp),%rbp+ movq 80(%rsp),%r13+ movq 56(%rsp),%rdi++ movq %r13,%r12+ rorq $7,%r13+ addq %r14,%r11+ movq %rdi,%r14+ rorq $42,%rdi++ xorq %r12,%r13+ shrq $7,%r12+ rorq $1,%r13+ xorq %r14,%rdi+ shrq $6,%r14++ rorq $19,%rdi+ xorq %r13,%r12+ xorq %r14,%rdi+ addq 16(%rsp),%r12++ addq 72(%rsp),%r12+ movq %rdx,%r13+ addq %rdi,%r12+ movq %r11,%r14+ rorq $23,%r13+ movq %r8,%rdi++ xorq %rdx,%r13+ rorq $5,%r14+ xorq %r9,%rdi++ movq %r12,72(%rsp)+ xorq %r11,%r14+ andq %rdx,%rdi++ rorq $4,%r13+ addq %r10,%r12+ xorq %r9,%rdi++ rorq $6,%r14+ xorq %rdx,%r13+ addq %rdi,%r12++ movq %r11,%rdi+ addq (%rbp),%r12+ xorq %r11,%r14++ xorq %rax,%rdi+ rorq $14,%r13+ movq %rax,%r10++ andq %rdi,%r15+ rorq $28,%r14+ addq %r13,%r12++ xorq %r15,%r10+ addq %r12,%rcx+ addq %r12,%r10++ leaq 24(%rbp),%rbp+ movq 88(%rsp),%r13+ movq 64(%rsp),%r15++ movq %r13,%r12+ rorq $7,%r13+ addq %r14,%r10+ movq %r15,%r14+ rorq $42,%r15++ xorq %r12,%r13+ shrq $7,%r12+ rorq $1,%r13+ xorq %r14,%r15+ shrq $6,%r14++ rorq $19,%r15+ xorq %r13,%r12+ xorq %r14,%r15+ addq 24(%rsp),%r12++ addq 80(%rsp),%r12+ movq %rcx,%r13+ addq %r15,%r12+ movq %r10,%r14+ rorq $23,%r13+ movq %rdx,%r15++ xorq %rcx,%r13+ rorq $5,%r14+ xorq %r8,%r15++ movq %r12,80(%rsp)+ xorq %r10,%r14+ andq %rcx,%r15++ rorq $4,%r13+ addq %r9,%r12+ xorq %r8,%r15++ rorq $6,%r14+ xorq %rcx,%r13+ addq %r15,%r12++ movq %r10,%r15+ addq (%rbp),%r12+ xorq %r10,%r14++ xorq %r11,%r15+ rorq $14,%r13+ movq %r11,%r9++ andq %r15,%rdi+ rorq $28,%r14+ addq %r13,%r12++ xorq %rdi,%r9+ addq %r12,%rbx+ addq %r12,%r9++ leaq 8(%rbp),%rbp+ movq 96(%rsp),%r13+ movq 72(%rsp),%rdi++ movq %r13,%r12+ rorq $7,%r13+ addq %r14,%r9+ movq %rdi,%r14+ rorq $42,%rdi++ xorq %r12,%r13+ shrq $7,%r12+ rorq $1,%r13+ xorq %r14,%rdi+ shrq $6,%r14++ rorq $19,%rdi+ xorq %r13,%r12+ xorq %r14,%rdi+ addq 32(%rsp),%r12++ addq 88(%rsp),%r12+ movq %rbx,%r13+ addq %rdi,%r12+ movq %r9,%r14+ rorq $23,%r13+ movq %rcx,%rdi++ xorq %rbx,%r13+ rorq $5,%r14+ xorq %rdx,%rdi++ movq %r12,88(%rsp)+ xorq %r9,%r14+ andq %rbx,%rdi++ rorq $4,%r13+ addq %r8,%r12+ xorq %rdx,%rdi++ rorq $6,%r14+ xorq %rbx,%r13+ addq %rdi,%r12++ movq %r9,%rdi+ addq (%rbp),%r12+ xorq %r9,%r14++ xorq %r10,%rdi+ rorq $14,%r13+ movq %r10,%r8++ andq %rdi,%r15+ rorq $28,%r14+ addq %r13,%r12++ xorq %r15,%r8+ addq %r12,%rax+ addq %r12,%r8++ leaq 24(%rbp),%rbp+ movq 104(%rsp),%r13+ movq 80(%rsp),%r15++ movq %r13,%r12+ rorq $7,%r13+ addq %r14,%r8+ movq %r15,%r14+ rorq $42,%r15++ xorq %r12,%r13+ shrq $7,%r12+ rorq $1,%r13+ xorq %r14,%r15+ shrq $6,%r14++ rorq $19,%r15+ xorq %r13,%r12+ xorq %r14,%r15+ addq 40(%rsp),%r12++ addq 96(%rsp),%r12+ movq %rax,%r13+ addq %r15,%r12+ movq %r8,%r14+ rorq $23,%r13+ movq %rbx,%r15++ xorq %rax,%r13+ rorq $5,%r14+ xorq %rcx,%r15++ movq %r12,96(%rsp)+ xorq %r8,%r14+ andq %rax,%r15++ rorq $4,%r13+ addq %rdx,%r12+ xorq %rcx,%r15++ rorq $6,%r14+ xorq %rax,%r13+ addq %r15,%r12++ movq %r8,%r15+ addq (%rbp),%r12+ xorq %r8,%r14++ xorq %r9,%r15+ rorq $14,%r13+ movq %r9,%rdx++ andq %r15,%rdi+ rorq $28,%r14+ addq %r13,%r12++ xorq %rdi,%rdx+ addq %r12,%r11+ addq %r12,%rdx++ leaq 8(%rbp),%rbp+ movq 112(%rsp),%r13+ movq 88(%rsp),%rdi++ movq %r13,%r12+ rorq $7,%r13+ addq %r14,%rdx+ movq %rdi,%r14+ rorq $42,%rdi++ xorq %r12,%r13+ shrq $7,%r12+ rorq $1,%r13+ xorq %r14,%rdi+ shrq $6,%r14++ rorq $19,%rdi+ xorq %r13,%r12+ xorq %r14,%rdi+ addq 48(%rsp),%r12++ addq 104(%rsp),%r12+ movq %r11,%r13+ addq %rdi,%r12+ movq %rdx,%r14+ rorq $23,%r13+ movq %rax,%rdi++ xorq %r11,%r13+ rorq $5,%r14+ xorq %rbx,%rdi++ movq %r12,104(%rsp)+ xorq %rdx,%r14+ andq %r11,%rdi++ rorq $4,%r13+ addq %rcx,%r12+ xorq %rbx,%rdi++ rorq $6,%r14+ xorq %r11,%r13+ addq %rdi,%r12++ movq %rdx,%rdi+ addq (%rbp),%r12+ xorq %rdx,%r14++ xorq %r8,%rdi+ rorq $14,%r13+ movq %r8,%rcx++ andq %rdi,%r15+ rorq $28,%r14+ addq %r13,%r12++ xorq %r15,%rcx+ addq %r12,%r10+ addq %r12,%rcx++ leaq 24(%rbp),%rbp+ movq 120(%rsp),%r13+ movq 96(%rsp),%r15++ movq %r13,%r12+ rorq $7,%r13+ addq %r14,%rcx+ movq %r15,%r14+ rorq $42,%r15++ xorq %r12,%r13+ shrq $7,%r12+ rorq $1,%r13+ xorq %r14,%r15+ shrq $6,%r14++ rorq $19,%r15+ xorq %r13,%r12+ xorq %r14,%r15+ addq 56(%rsp),%r12++ addq 112(%rsp),%r12+ movq %r10,%r13+ addq %r15,%r12+ movq %rcx,%r14+ rorq $23,%r13+ movq %r11,%r15++ xorq %r10,%r13+ rorq $5,%r14+ xorq %rax,%r15++ movq %r12,112(%rsp)+ xorq %rcx,%r14+ andq %r10,%r15++ rorq $4,%r13+ addq %rbx,%r12+ xorq %rax,%r15++ rorq $6,%r14+ xorq %r10,%r13+ addq %r15,%r12++ movq %rcx,%r15+ addq (%rbp),%r12+ xorq %rcx,%r14++ xorq %rdx,%r15+ rorq $14,%r13+ movq %rdx,%rbx++ andq %r15,%rdi+ rorq $28,%r14+ addq %r13,%r12++ xorq %rdi,%rbx+ addq %r12,%r9+ addq %r12,%rbx++ leaq 8(%rbp),%rbp+ movq 0(%rsp),%r13+ movq 104(%rsp),%rdi++ movq %r13,%r12+ rorq $7,%r13+ addq %r14,%rbx+ movq %rdi,%r14+ rorq $42,%rdi++ xorq %r12,%r13+ shrq $7,%r12+ rorq $1,%r13+ xorq %r14,%rdi+ shrq $6,%r14++ rorq $19,%rdi+ xorq %r13,%r12+ xorq %r14,%rdi+ addq 64(%rsp),%r12++ addq 120(%rsp),%r12+ movq %r9,%r13+ addq %rdi,%r12+ movq %rbx,%r14+ rorq $23,%r13+ movq %r10,%rdi++ xorq %r9,%r13+ rorq $5,%r14+ xorq %r11,%rdi++ movq %r12,120(%rsp)+ xorq %rbx,%r14+ andq %r9,%rdi++ rorq $4,%r13+ addq %rax,%r12+ xorq %r11,%rdi++ rorq $6,%r14+ xorq %r9,%r13+ addq %rdi,%r12++ movq %rbx,%rdi+ addq (%rbp),%r12+ xorq %rbx,%r14++ xorq %rcx,%rdi+ rorq $14,%r13+ movq %rcx,%rax++ andq %rdi,%r15+ rorq $28,%r14+ addq %r13,%r12++ xorq %r15,%rax+ addq %r12,%r8+ addq %r12,%rax++ leaq 24(%rbp),%rbp+ cmpb $0,7(%rbp)+ jnz L$rounds_16_xx++ movq 128+0(%rsp),%rdi+ addq %r14,%rax+ leaq 128(%rsi),%rsi++ addq 0(%rdi),%rax+ addq 8(%rdi),%rbx+ addq 16(%rdi),%rcx+ addq 24(%rdi),%rdx+ addq 32(%rdi),%r8+ addq 40(%rdi),%r9+ addq 48(%rdi),%r10+ addq 56(%rdi),%r11++ cmpq 128+16(%rsp),%rsi++ movq %rax,0(%rdi)+ movq %rbx,8(%rdi)+ movq %rcx,16(%rdi)+ movq %rdx,24(%rdi)+ movq %r8,32(%rdi)+ movq %r9,40(%rdi)+ movq %r10,48(%rdi)+ movq %r11,56(%rdi)+ jb L$loop++ leaq 128+24+48(%rsp),%r11+.cfi_def_cfa %r11,8+ movq 128+24(%rsp),%r15+ movq -40(%r11),%r14+ movq -32(%r11),%r13+ movq -24(%r11),%r12+ movq -16(%r11),%rbx+ movq -8(%r11),%rbp+.cfi_restore %r12+.cfi_restore %r13+.cfi_restore %r14+.cfi_restore %r15+.cfi_restore %rbp+.cfi_restore %rbx+ leaq (%r11),%rsp+ .byte 0xf3,0xc3+.cfi_endproc ++.p2align 6++K512:+.quad 0x428a2f98d728ae22,0x7137449123ef65cd+.quad 0x428a2f98d728ae22,0x7137449123ef65cd+.quad 0xb5c0fbcfec4d3b2f,0xe9b5dba58189dbbc+.quad 0xb5c0fbcfec4d3b2f,0xe9b5dba58189dbbc+.quad 0x3956c25bf348b538,0x59f111f1b605d019+.quad 0x3956c25bf348b538,0x59f111f1b605d019+.quad 0x923f82a4af194f9b,0xab1c5ed5da6d8118+.quad 0x923f82a4af194f9b,0xab1c5ed5da6d8118+.quad 0xd807aa98a3030242,0x12835b0145706fbe+.quad 0xd807aa98a3030242,0x12835b0145706fbe+.quad 0x243185be4ee4b28c,0x550c7dc3d5ffb4e2+.quad 0x243185be4ee4b28c,0x550c7dc3d5ffb4e2+.quad 0x72be5d74f27b896f,0x80deb1fe3b1696b1+.quad 0x72be5d74f27b896f,0x80deb1fe3b1696b1+.quad 0x9bdc06a725c71235,0xc19bf174cf692694+.quad 0x9bdc06a725c71235,0xc19bf174cf692694+.quad 0xe49b69c19ef14ad2,0xefbe4786384f25e3+.quad 0xe49b69c19ef14ad2,0xefbe4786384f25e3+.quad 0x0fc19dc68b8cd5b5,0x240ca1cc77ac9c65+.quad 0x0fc19dc68b8cd5b5,0x240ca1cc77ac9c65+.quad 0x2de92c6f592b0275,0x4a7484aa6ea6e483+.quad 0x2de92c6f592b0275,0x4a7484aa6ea6e483+.quad 0x5cb0a9dcbd41fbd4,0x76f988da831153b5+.quad 0x5cb0a9dcbd41fbd4,0x76f988da831153b5+.quad 0x983e5152ee66dfab,0xa831c66d2db43210+.quad 0x983e5152ee66dfab,0xa831c66d2db43210+.quad 0xb00327c898fb213f,0xbf597fc7beef0ee4+.quad 0xb00327c898fb213f,0xbf597fc7beef0ee4+.quad 0xc6e00bf33da88fc2,0xd5a79147930aa725+.quad 0xc6e00bf33da88fc2,0xd5a79147930aa725+.quad 0x06ca6351e003826f,0x142929670a0e6e70+.quad 0x06ca6351e003826f,0x142929670a0e6e70+.quad 0x27b70a8546d22ffc,0x2e1b21385c26c926+.quad 0x27b70a8546d22ffc,0x2e1b21385c26c926+.quad 0x4d2c6dfc5ac42aed,0x53380d139d95b3df+.quad 0x4d2c6dfc5ac42aed,0x53380d139d95b3df+.quad 0x650a73548baf63de,0x766a0abb3c77b2a8+.quad 0x650a73548baf63de,0x766a0abb3c77b2a8+.quad 0x81c2c92e47edaee6,0x92722c851482353b+.quad 0x81c2c92e47edaee6,0x92722c851482353b+.quad 0xa2bfe8a14cf10364,0xa81a664bbc423001+.quad 0xa2bfe8a14cf10364,0xa81a664bbc423001+.quad 0xc24b8b70d0f89791,0xc76c51a30654be30+.quad 0xc24b8b70d0f89791,0xc76c51a30654be30+.quad 0xd192e819d6ef5218,0xd69906245565a910+.quad 0xd192e819d6ef5218,0xd69906245565a910+.quad 0xf40e35855771202a,0x106aa07032bbd1b8+.quad 0xf40e35855771202a,0x106aa07032bbd1b8+.quad 0x19a4c116b8d2d0c8,0x1e376c085141ab53+.quad 0x19a4c116b8d2d0c8,0x1e376c085141ab53+.quad 0x2748774cdf8eeb99,0x34b0bcb5e19b48a8+.quad 0x2748774cdf8eeb99,0x34b0bcb5e19b48a8+.quad 0x391c0cb3c5c95a63,0x4ed8aa4ae3418acb+.quad 0x391c0cb3c5c95a63,0x4ed8aa4ae3418acb+.quad 0x5b9cca4f7763e373,0x682e6ff3d6b2b8a3+.quad 0x5b9cca4f7763e373,0x682e6ff3d6b2b8a3+.quad 0x748f82ee5defb2fc,0x78a5636f43172f60+.quad 0x748f82ee5defb2fc,0x78a5636f43172f60+.quad 0x84c87814a1f0ab72,0x8cc702081a6439ec+.quad 0x84c87814a1f0ab72,0x8cc702081a6439ec+.quad 0x90befffa23631e28,0xa4506cebde82bde9+.quad 0x90befffa23631e28,0xa4506cebde82bde9+.quad 0xbef9a3f7b2c67915,0xc67178f2e372532b+.quad 0xbef9a3f7b2c67915,0xc67178f2e372532b+.quad 0xca273eceea26619c,0xd186b8c721c0c207+.quad 0xca273eceea26619c,0xd186b8c721c0c207+.quad 0xeada7dd6cde0eb1e,0xf57d4f7fee6ed178+.quad 0xeada7dd6cde0eb1e,0xf57d4f7fee6ed178+.quad 0x06f067aa72176fba,0x0a637dc5a2c898a6+.quad 0x06f067aa72176fba,0x0a637dc5a2c898a6+.quad 0x113f9804bef90dae,0x1b710b35131c471b+.quad 0x113f9804bef90dae,0x1b710b35131c471b+.quad 0x28db77f523047d84,0x32caab7b40c72493+.quad 0x28db77f523047d84,0x32caab7b40c72493+.quad 0x3c9ebe0a15c9bebc,0x431d67c49c100d4c+.quad 0x3c9ebe0a15c9bebc,0x431d67c49c100d4c+.quad 0x4cc5d4becb3e42b6,0x597f299cfc657e2a+.quad 0x4cc5d4becb3e42b6,0x597f299cfc657e2a+.quad 0x5fcb6fab3ad6faec,0x6c44198c4a475817+.quad 0x5fcb6fab3ad6faec,0x6c44198c4a475817++.quad 0x0001020304050607,0x08090a0b0c0d0e0f+.quad 0x0001020304050607,0x08090a0b0c0d0e0f++K512_nodup:+.quad 0x428a2f98d728ae22,0x7137449123ef65cd+.quad 0xb5c0fbcfec4d3b2f,0xe9b5dba58189dbbc+.quad 0x3956c25bf348b538,0x59f111f1b605d019+.quad 0x923f82a4af194f9b,0xab1c5ed5da6d8118+.quad 0xd807aa98a3030242,0x12835b0145706fbe+.quad 0x243185be4ee4b28c,0x550c7dc3d5ffb4e2+.quad 0x72be5d74f27b896f,0x80deb1fe3b1696b1+.quad 0x9bdc06a725c71235,0xc19bf174cf692694+.quad 0xe49b69c19ef14ad2,0xefbe4786384f25e3+.quad 0x0fc19dc68b8cd5b5,0x240ca1cc77ac9c65+.quad 0x2de92c6f592b0275,0x4a7484aa6ea6e483+.quad 0x5cb0a9dcbd41fbd4,0x76f988da831153b5+.quad 0x983e5152ee66dfab,0xa831c66d2db43210+.quad 0xb00327c898fb213f,0xbf597fc7beef0ee4+.quad 0xc6e00bf33da88fc2,0xd5a79147930aa725+.quad 0x06ca6351e003826f,0x142929670a0e6e70+.quad 0x27b70a8546d22ffc,0x2e1b21385c26c926+.quad 0x4d2c6dfc5ac42aed,0x53380d139d95b3df+.quad 0x650a73548baf63de,0x766a0abb3c77b2a8+.quad 0x81c2c92e47edaee6,0x92722c851482353b+.quad 0xa2bfe8a14cf10364,0xa81a664bbc423001+.quad 0xc24b8b70d0f89791,0xc76c51a30654be30+.quad 0xd192e819d6ef5218,0xd69906245565a910+.quad 0xf40e35855771202a,0x106aa07032bbd1b8+.quad 0x19a4c116b8d2d0c8,0x1e376c085141ab53+.quad 0x2748774cdf8eeb99,0x34b0bcb5e19b48a8+.quad 0x391c0cb3c5c95a63,0x4ed8aa4ae3418acb+.quad 0x5b9cca4f7763e373,0x682e6ff3d6b2b8a3+.quad 0x748f82ee5defb2fc,0x78a5636f43172f60+.quad 0x84c87814a1f0ab72,0x8cc702081a6439ec+.quad 0x90befffa23631e28,0xa4506cebde82bde9+.quad 0xbef9a3f7b2c67915,0xc67178f2e372532b+.quad 0xca273eceea26619c,0xd186b8c721c0c207+.quad 0xeada7dd6cde0eb1e,0xf57d4f7fee6ed178+.quad 0x06f067aa72176fba,0x0a637dc5a2c898a6+.quad 0x113f9804bef90dae,0x1b710b35131c471b+.quad 0x28db77f523047d84,0x32caab7b40c72493+.quad 0x3c9ebe0a15c9bebc,0x431d67c49c100d4c+.quad 0x4cc5d4becb3e42b6,0x597f299cfc657e2a+.quad 0x5fcb6fab3ad6faec,0x6c44198c4a475817++.byte 83,72,65,53,49,50,32,98,108,111,99,107,32,116,114,97,110,115,102,111,114,109,32,102,111,114,32,120,56,54,95,54,52,44,32,67,82,89,80,84,79,71,65,77,83,32,98,121,32,64,100,111,116,45,97,115,109,0+.globl _crypton_sha512_asm_block_data_order_shaext++.p2align 6+_crypton_sha512_asm_block_data_order_shaext:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa+++ pushq %rbp+.cfi_adjust_cfa_offset 8+.cfi_offset %rbp,-16+ movq %rsp,%rbp+.cfi_def_cfa_register %rbp+L$shaext_shortcut:++ leaq K512_nodup+128(%rip),%rcx+ vmovdqu (%rdi),%ymm0+ vmovdqu 32(%rdi),%ymm1+ vmovdqa -160(%rcx),%ymm8++ vpermq $27,%ymm0,%ymm0+ vpblendd $15,%ymm1,%ymm0,%ymm5+ vpblendd $15,%ymm0,%ymm1,%ymm6+ vpermq $225,%ymm5,%ymm5+ vpermq $75,%ymm6,%ymm6+ jmp L$oop_shaext++.p2align 4+L$oop_shaext:+ vmovdqu (%rsi),%ymm0+ vmovdqu 32(%rsi),%ymm1+ vmovdqu 64(%rsi),%ymm2+ vpshufb %ymm8,%ymm0,%ymm0+ vmovdqu 96(%rsi),%ymm3++ vpaddq 0-128(%rcx),%ymm0,%ymm4+ vpshufb %ymm8,%ymm1,%ymm1+ vmovdqa %ymm6,%ymm10+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+ vmovdqa %ymm5,%ymm9+.byte 196,226,79,203,236++ vpaddq 32-128(%rcx),%ymm1,%ymm4+ vpshufb %ymm8,%ymm2,%ymm2+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+ leaq 128(%rsi),%rsi+.byte 196,226,127,204,193+.byte 196,226,79,203,236++ vpaddq 64-128(%rcx),%ymm2,%ymm4+ vpshufb %ymm8,%ymm3,%ymm3+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+ vpblendd $0x03,%ymm3,%ymm2,%ymm7+ vpermq $0x39,%ymm7,%ymm7+ vpaddq %ymm7,%ymm0,%ymm0+.byte 196,226,127,204,202+.byte 196,226,79,203,236++ vpaddq 96-128(%rcx),%ymm3,%ymm4+.byte 196,226,127,205,195+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+ vpblendd $0x03,%ymm0,%ymm3,%ymm7+ vpermq $0x39,%ymm7,%ymm7+ vpaddq %ymm7,%ymm1,%ymm1+.byte 196,226,127,204,211+.byte 196,226,79,203,236+ vpaddq 128-128(%rcx),%ymm0,%ymm4+.byte 196,226,127,205,200+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+ vpblendd $0x03,%ymm1,%ymm0,%ymm7+ vpermq $0x39,%ymm7,%ymm7+ vpaddq %ymm7,%ymm2,%ymm2+.byte 196,226,127,204,216+.byte 196,226,79,203,236+ vpaddq 160-128(%rcx),%ymm1,%ymm4+.byte 196,226,127,205,209+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+ vpblendd $0x03,%ymm2,%ymm1,%ymm7+ vpermq $0x39,%ymm7,%ymm7+ vpaddq %ymm7,%ymm3,%ymm3+.byte 196,226,127,204,193+.byte 196,226,79,203,236+ vpaddq 192-128(%rcx),%ymm2,%ymm4+.byte 196,226,127,205,218+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+ vpblendd $0x03,%ymm3,%ymm2,%ymm7+ vpermq $0x39,%ymm7,%ymm7+ vpaddq %ymm7,%ymm0,%ymm0+.byte 196,226,127,204,202+.byte 196,226,79,203,236+ vpaddq 224-128(%rcx),%ymm3,%ymm4+.byte 196,226,127,205,195+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+ vpblendd $0x03,%ymm0,%ymm3,%ymm7+ vpermq $0x39,%ymm7,%ymm7+ vpaddq %ymm7,%ymm1,%ymm1+.byte 196,226,127,204,211+.byte 196,226,79,203,236+ vpaddq 256-128(%rcx),%ymm0,%ymm4+.byte 196,226,127,205,200+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+ vpblendd $0x03,%ymm1,%ymm0,%ymm7+ vpermq $0x39,%ymm7,%ymm7+ vpaddq %ymm7,%ymm2,%ymm2+.byte 196,226,127,204,216+.byte 196,226,79,203,236+ vpaddq 288-128(%rcx),%ymm1,%ymm4+.byte 196,226,127,205,209+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+ vpblendd $0x03,%ymm2,%ymm1,%ymm7+ vpermq $0x39,%ymm7,%ymm7+ vpaddq %ymm7,%ymm3,%ymm3+.byte 196,226,127,204,193+.byte 196,226,79,203,236+ vpaddq 320-128(%rcx),%ymm2,%ymm4+.byte 196,226,127,205,218+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+ vpblendd $0x03,%ymm3,%ymm2,%ymm7+ vpermq $0x39,%ymm7,%ymm7+ vpaddq %ymm7,%ymm0,%ymm0+.byte 196,226,127,204,202+.byte 196,226,79,203,236+ vpaddq 352-128(%rcx),%ymm3,%ymm4+.byte 196,226,127,205,195+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+ vpblendd $0x03,%ymm0,%ymm3,%ymm7+ vpermq $0x39,%ymm7,%ymm7+ vpaddq %ymm7,%ymm1,%ymm1+.byte 196,226,127,204,211+.byte 196,226,79,203,236+ vpaddq 384-128(%rcx),%ymm0,%ymm4+.byte 196,226,127,205,200+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+ vpblendd $0x03,%ymm1,%ymm0,%ymm7+ vpermq $0x39,%ymm7,%ymm7+ vpaddq %ymm7,%ymm2,%ymm2+.byte 196,226,127,204,216+.byte 196,226,79,203,236+ vpaddq 416-128(%rcx),%ymm1,%ymm4+.byte 196,226,127,205,209+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+ vpblendd $0x03,%ymm2,%ymm1,%ymm7+ vpermq $0x39,%ymm7,%ymm7+ vpaddq %ymm7,%ymm3,%ymm3+.byte 196,226,127,204,193+.byte 196,226,79,203,236+ vpaddq 448-128(%rcx),%ymm2,%ymm4+.byte 196,226,127,205,218+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+ vpblendd $0x03,%ymm3,%ymm2,%ymm7+ vpermq $0x39,%ymm7,%ymm7+ vpaddq %ymm7,%ymm0,%ymm0+.byte 196,226,127,204,202+.byte 196,226,79,203,236+ vpaddq 480-128(%rcx),%ymm3,%ymm4+.byte 196,226,127,205,195+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+ vpblendd $0x03,%ymm0,%ymm3,%ymm7+ vpermq $0x39,%ymm7,%ymm7+ vpaddq %ymm7,%ymm1,%ymm1+.byte 196,226,127,204,211+.byte 196,226,79,203,236+ vpaddq 512-128(%rcx),%ymm0,%ymm4+.byte 196,226,127,205,200+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+ vpblendd $0x03,%ymm1,%ymm0,%ymm7+ vpermq $0x39,%ymm7,%ymm7+ vpaddq %ymm7,%ymm2,%ymm2+.byte 196,226,127,204,216+.byte 196,226,79,203,236+ vpaddq 544-128(%rcx),%ymm1,%ymm4+.byte 196,226,127,205,209+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+ vpblendd $0x03,%ymm2,%ymm1,%ymm7+ vpermq $0x39,%ymm7,%ymm7+.byte 196,226,79,203,236+ vpaddq %ymm7,%ymm3,%ymm3++ vpaddq 576-128(%rcx),%ymm2,%ymm4+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+.byte 196,226,127,205,218+.byte 196,226,79,203,236++ vpaddq 608-128(%rcx),%ymm3,%ymm4+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+ decq %rdx+.byte 196,226,79,203,236++ vpaddq %ymm10,%ymm6,%ymm6+ vpaddq %ymm9,%ymm5,%ymm5+ jnz L$oop_shaext++ vpermq $75,%ymm5,%ymm5+ vpblendd $240,%ymm6,%ymm5,%ymm1+ vpblendd $240,%ymm5,%ymm6,%ymm2+ vpermq $180,%ymm1,%ymm1+ vpermq $27,%ymm2,%ymm2++ vmovdqu %ymm1,(%rdi)+ vmovdqu %ymm2,32(%rdi)++ vzeroupper+.cfi_def_cfa_register %rsp+ popq %rbp+.cfi_adjust_cfa_offset -8+.cfi_restore %rbp++ .byte 0xf3,0xc3+.cfi_endproc +++.p2align 6+crypton_sha512_asm_block_data_order_xop:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa+++ pushq %rbp+.cfi_adjust_cfa_offset 8+.cfi_offset %rbp,-16+ movq %rsp,%rbp+.cfi_def_cfa_register %rbp+L$xop_shortcut:+ pushq %rbx+.cfi_offset %rbx,-24+ pushq %r12+.cfi_offset %r12,-32+ pushq %r13+.cfi_offset %r13,-40+ pushq %r14+.cfi_offset %r14,-48+ pushq %r15+.cfi_offset %r15,-56+ shlq $4,%rdx+ subq $24,%rsp++ leaq (%rsi,%rdx,8),%rdx+ movq %rdi,-64(%rbp)++ movq %rdx,-48(%rbp)+++ leaq -128(%rsp),%rsp+ vzeroupper+ andq $-64,%rsp+ movq 0(%rdi),%rax+ movq 8(%rdi),%rbx+ movq 16(%rdi),%rcx+ movq 24(%rdi),%rdx+ movq 32(%rdi),%r8+ movq 40(%rdi),%r9+ movq 48(%rdi),%r10+ movq 56(%rdi),%r11+ jmp L$loop_xop+.p2align 4+L$loop_xop:+ vmovdqa K512+1280(%rip),%xmm11+ movq %rsi,-56(%rbp)+ vmovdqu 0(%rsi),%xmm0+ vmovdqu 16(%rsi),%xmm1+ vmovdqu 32(%rsi),%xmm2+ vpshufb %xmm11,%xmm0,%xmm0+ vmovdqu 48(%rsi),%xmm3+ vpshufb %xmm11,%xmm1,%xmm1+ vmovdqu 64(%rsi),%xmm4+ vpshufb %xmm11,%xmm2,%xmm2+ vmovdqu 80(%rsi),%xmm5+ vpshufb %xmm11,%xmm3,%xmm3+ vmovdqu 96(%rsi),%xmm6+ vpshufb %xmm11,%xmm4,%xmm4+ vmovdqu 112(%rsi),%xmm7+ leaq K512+128(%rip),%rsi+ vpshufb %xmm11,%xmm5,%xmm5+ vpaddq -128(%rsi),%xmm0,%xmm8+ vpshufb %xmm11,%xmm6,%xmm6+ vpaddq -96(%rsi),%xmm1,%xmm9+ vpshufb %xmm11,%xmm7,%xmm7+ vpaddq -64(%rsi),%xmm2,%xmm10+ vpaddq -32(%rsi),%xmm3,%xmm11+ vmovdqa %xmm8,0(%rsp)+ vpaddq 0(%rsi),%xmm4,%xmm8+ vmovdqa %xmm9,16(%rsp)+ vpaddq 32(%rsi),%xmm5,%xmm9+ vmovdqa %xmm10,32(%rsp)+ vpaddq 64(%rsi),%xmm6,%xmm10+ vmovdqa %xmm11,48(%rsp)+ vpaddq 96(%rsi),%xmm7,%xmm11+ vmovdqa %xmm8,64(%rsp)+ movq %rax,%r14+ vmovdqa %xmm9,80(%rsp)+ movq %rbx,%rdi+ vmovdqa %xmm10,96(%rsp)+ xorq %rcx,%rdi+ vmovdqa %xmm11,112(%rsp)+ movq %r8,%r13+ jmp L$xop_00_47++.p2align 4+L$xop_00_47:+ addq $256,%rsi+ vpalignr $8,%xmm0,%xmm1,%xmm8+ rorq $23,%r13+ movq %r14,%rax+ vpalignr $8,%xmm4,%xmm5,%xmm11+ movq %r9,%r12+ rorq $5,%r14+.byte 143,72,120,195,200,56+ xorq %r8,%r13+ xorq %r10,%r12+ vpsrlq $7,%xmm8,%xmm8+ rorq $4,%r13+ xorq %rax,%r14+ vpaddq %xmm11,%xmm0,%xmm0+ andq %r8,%r12+ xorq %r8,%r13+ addq 0(%rsp),%r11+ movq %rax,%r15+.byte 143,72,120,195,209,7+ xorq %r10,%r12+ rorq $6,%r14+ vpxor %xmm9,%xmm8,%xmm8+ xorq %rbx,%r15+ addq %r12,%r11+ rorq $14,%r13+ andq %r15,%rdi+.byte 143,104,120,195,223,3+ xorq %rax,%r14+ addq %r13,%r11+ vpxor %xmm10,%xmm8,%xmm8+ xorq %rbx,%rdi+ rorq $28,%r14+ vpsrlq $6,%xmm7,%xmm10+ addq %r11,%rdx+ addq %rdi,%r11+ vpaddq %xmm8,%xmm0,%xmm0+ movq %rdx,%r13+ addq %r11,%r14+.byte 143,72,120,195,203,42+ rorq $23,%r13+ movq %r14,%r11+ vpxor %xmm10,%xmm11,%xmm11+ movq %r8,%r12+ rorq $5,%r14+ xorq %rdx,%r13+ xorq %r9,%r12+ vpxor %xmm9,%xmm11,%xmm11+ rorq $4,%r13+ xorq %r11,%r14+ andq %rdx,%r12+ xorq %rdx,%r13+ vpaddq %xmm11,%xmm0,%xmm0+ addq 8(%rsp),%r10+ movq %r11,%rdi+ xorq %r9,%r12+ rorq $6,%r14+ vpaddq -128(%rsi),%xmm0,%xmm10+ xorq %rax,%rdi+ addq %r12,%r10+ rorq $14,%r13+ andq %rdi,%r15+ xorq %r11,%r14+ addq %r13,%r10+ xorq %rax,%r15+ rorq $28,%r14+ addq %r10,%rcx+ addq %r15,%r10+ movq %rcx,%r13+ addq %r10,%r14+ vmovdqa %xmm10,0(%rsp)+ vpalignr $8,%xmm1,%xmm2,%xmm8+ rorq $23,%r13+ movq %r14,%r10+ vpalignr $8,%xmm5,%xmm6,%xmm11+ movq %rdx,%r12+ rorq $5,%r14+.byte 143,72,120,195,200,56+ xorq %rcx,%r13+ xorq %r8,%r12+ vpsrlq $7,%xmm8,%xmm8+ rorq $4,%r13+ xorq %r10,%r14+ vpaddq %xmm11,%xmm1,%xmm1+ andq %rcx,%r12+ xorq %rcx,%r13+ addq 16(%rsp),%r9+ movq %r10,%r15+.byte 143,72,120,195,209,7+ xorq %r8,%r12+ rorq $6,%r14+ vpxor %xmm9,%xmm8,%xmm8+ xorq %r11,%r15+ addq %r12,%r9+ rorq $14,%r13+ andq %r15,%rdi+.byte 143,104,120,195,216,3+ xorq %r10,%r14+ addq %r13,%r9+ vpxor %xmm10,%xmm8,%xmm8+ xorq %r11,%rdi+ rorq $28,%r14+ vpsrlq $6,%xmm0,%xmm10+ addq %r9,%rbx+ addq %rdi,%r9+ vpaddq %xmm8,%xmm1,%xmm1+ movq %rbx,%r13+ addq %r9,%r14+.byte 143,72,120,195,203,42+ rorq $23,%r13+ movq %r14,%r9+ vpxor %xmm10,%xmm11,%xmm11+ movq %rcx,%r12+ rorq $5,%r14+ xorq %rbx,%r13+ xorq %rdx,%r12+ vpxor %xmm9,%xmm11,%xmm11+ rorq $4,%r13+ xorq %r9,%r14+ andq %rbx,%r12+ xorq %rbx,%r13+ vpaddq %xmm11,%xmm1,%xmm1+ addq 24(%rsp),%r8+ movq %r9,%rdi+ xorq %rdx,%r12+ rorq $6,%r14+ vpaddq -96(%rsi),%xmm1,%xmm10+ xorq %r10,%rdi+ addq %r12,%r8+ rorq $14,%r13+ andq %rdi,%r15+ xorq %r9,%r14+ addq %r13,%r8+ xorq %r10,%r15+ rorq $28,%r14+ addq %r8,%rax+ addq %r15,%r8+ movq %rax,%r13+ addq %r8,%r14+ vmovdqa %xmm10,16(%rsp)+ vpalignr $8,%xmm2,%xmm3,%xmm8+ rorq $23,%r13+ movq %r14,%r8+ vpalignr $8,%xmm6,%xmm7,%xmm11+ movq %rbx,%r12+ rorq $5,%r14+.byte 143,72,120,195,200,56+ xorq %rax,%r13+ xorq %rcx,%r12+ vpsrlq $7,%xmm8,%xmm8+ rorq $4,%r13+ xorq %r8,%r14+ vpaddq %xmm11,%xmm2,%xmm2+ andq %rax,%r12+ xorq %rax,%r13+ addq 32(%rsp),%rdx+ movq %r8,%r15+.byte 143,72,120,195,209,7+ xorq %rcx,%r12+ rorq $6,%r14+ vpxor %xmm9,%xmm8,%xmm8+ xorq %r9,%r15+ addq %r12,%rdx+ rorq $14,%r13+ andq %r15,%rdi+.byte 143,104,120,195,217,3+ xorq %r8,%r14+ addq %r13,%rdx+ vpxor %xmm10,%xmm8,%xmm8+ xorq %r9,%rdi+ rorq $28,%r14+ vpsrlq $6,%xmm1,%xmm10+ addq %rdx,%r11+ addq %rdi,%rdx+ vpaddq %xmm8,%xmm2,%xmm2+ movq %r11,%r13+ addq %rdx,%r14+.byte 143,72,120,195,203,42+ rorq $23,%r13+ movq %r14,%rdx+ vpxor %xmm10,%xmm11,%xmm11+ movq %rax,%r12+ rorq $5,%r14+ xorq %r11,%r13+ xorq %rbx,%r12+ vpxor %xmm9,%xmm11,%xmm11+ rorq $4,%r13+ xorq %rdx,%r14+ andq %r11,%r12+ xorq %r11,%r13+ vpaddq %xmm11,%xmm2,%xmm2+ addq 40(%rsp),%rcx+ movq %rdx,%rdi+ xorq %rbx,%r12+ rorq $6,%r14+ vpaddq -64(%rsi),%xmm2,%xmm10+ xorq %r8,%rdi+ addq %r12,%rcx+ rorq $14,%r13+ andq %rdi,%r15+ xorq %rdx,%r14+ addq %r13,%rcx+ xorq %r8,%r15+ rorq $28,%r14+ addq %rcx,%r10+ addq %r15,%rcx+ movq %r10,%r13+ addq %rcx,%r14+ vmovdqa %xmm10,32(%rsp)+ vpalignr $8,%xmm3,%xmm4,%xmm8+ rorq $23,%r13+ movq %r14,%rcx+ vpalignr $8,%xmm7,%xmm0,%xmm11+ movq %r11,%r12+ rorq $5,%r14+.byte 143,72,120,195,200,56+ xorq %r10,%r13+ xorq %rax,%r12+ vpsrlq $7,%xmm8,%xmm8+ rorq $4,%r13+ xorq %rcx,%r14+ vpaddq %xmm11,%xmm3,%xmm3+ andq %r10,%r12+ xorq %r10,%r13+ addq 48(%rsp),%rbx+ movq %rcx,%r15+.byte 143,72,120,195,209,7+ xorq %rax,%r12+ rorq $6,%r14+ vpxor %xmm9,%xmm8,%xmm8+ xorq %rdx,%r15+ addq %r12,%rbx+ rorq $14,%r13+ andq %r15,%rdi+.byte 143,104,120,195,218,3+ xorq %rcx,%r14+ addq %r13,%rbx+ vpxor %xmm10,%xmm8,%xmm8+ xorq %rdx,%rdi+ rorq $28,%r14+ vpsrlq $6,%xmm2,%xmm10+ addq %rbx,%r9+ addq %rdi,%rbx+ vpaddq %xmm8,%xmm3,%xmm3+ movq %r9,%r13+ addq %rbx,%r14+.byte 143,72,120,195,203,42+ rorq $23,%r13+ movq %r14,%rbx+ vpxor %xmm10,%xmm11,%xmm11+ movq %r10,%r12+ rorq $5,%r14+ xorq %r9,%r13+ xorq %r11,%r12+ vpxor %xmm9,%xmm11,%xmm11+ rorq $4,%r13+ xorq %rbx,%r14+ andq %r9,%r12+ xorq %r9,%r13+ vpaddq %xmm11,%xmm3,%xmm3+ addq 56(%rsp),%rax+ movq %rbx,%rdi+ xorq %r11,%r12+ rorq $6,%r14+ vpaddq -32(%rsi),%xmm3,%xmm10+ xorq %rcx,%rdi+ addq %r12,%rax+ rorq $14,%r13+ andq %rdi,%r15+ xorq %rbx,%r14+ addq %r13,%rax+ xorq %rcx,%r15+ rorq $28,%r14+ addq %rax,%r8+ addq %r15,%rax+ movq %r8,%r13+ addq %rax,%r14+ vmovdqa %xmm10,48(%rsp)+ vpalignr $8,%xmm4,%xmm5,%xmm8+ rorq $23,%r13+ movq %r14,%rax+ vpalignr $8,%xmm0,%xmm1,%xmm11+ movq %r9,%r12+ rorq $5,%r14+.byte 143,72,120,195,200,56+ xorq %r8,%r13+ xorq %r10,%r12+ vpsrlq $7,%xmm8,%xmm8+ rorq $4,%r13+ xorq %rax,%r14+ vpaddq %xmm11,%xmm4,%xmm4+ andq %r8,%r12+ xorq %r8,%r13+ addq 64(%rsp),%r11+ movq %rax,%r15+.byte 143,72,120,195,209,7+ xorq %r10,%r12+ rorq $6,%r14+ vpxor %xmm9,%xmm8,%xmm8+ xorq %rbx,%r15+ addq %r12,%r11+ rorq $14,%r13+ andq %r15,%rdi+.byte 143,104,120,195,219,3+ xorq %rax,%r14+ addq %r13,%r11+ vpxor %xmm10,%xmm8,%xmm8+ xorq %rbx,%rdi+ rorq $28,%r14+ vpsrlq $6,%xmm3,%xmm10+ addq %r11,%rdx+ addq %rdi,%r11+ vpaddq %xmm8,%xmm4,%xmm4+ movq %rdx,%r13+ addq %r11,%r14+.byte 143,72,120,195,203,42+ rorq $23,%r13+ movq %r14,%r11+ vpxor %xmm10,%xmm11,%xmm11+ movq %r8,%r12+ rorq $5,%r14+ xorq %rdx,%r13+ xorq %r9,%r12+ vpxor %xmm9,%xmm11,%xmm11+ rorq $4,%r13+ xorq %r11,%r14+ andq %rdx,%r12+ xorq %rdx,%r13+ vpaddq %xmm11,%xmm4,%xmm4+ addq 72(%rsp),%r10+ movq %r11,%rdi+ xorq %r9,%r12+ rorq $6,%r14+ vpaddq 0(%rsi),%xmm4,%xmm10+ xorq %rax,%rdi+ addq %r12,%r10+ rorq $14,%r13+ andq %rdi,%r15+ xorq %r11,%r14+ addq %r13,%r10+ xorq %rax,%r15+ rorq $28,%r14+ addq %r10,%rcx+ addq %r15,%r10+ movq %rcx,%r13+ addq %r10,%r14+ vmovdqa %xmm10,64(%rsp)+ vpalignr $8,%xmm5,%xmm6,%xmm8+ rorq $23,%r13+ movq %r14,%r10+ vpalignr $8,%xmm1,%xmm2,%xmm11+ movq %rdx,%r12+ rorq $5,%r14+.byte 143,72,120,195,200,56+ xorq %rcx,%r13+ xorq %r8,%r12+ vpsrlq $7,%xmm8,%xmm8+ rorq $4,%r13+ xorq %r10,%r14+ vpaddq %xmm11,%xmm5,%xmm5+ andq %rcx,%r12+ xorq %rcx,%r13+ addq 80(%rsp),%r9+ movq %r10,%r15+.byte 143,72,120,195,209,7+ xorq %r8,%r12+ rorq $6,%r14+ vpxor %xmm9,%xmm8,%xmm8+ xorq %r11,%r15+ addq %r12,%r9+ rorq $14,%r13+ andq %r15,%rdi+.byte 143,104,120,195,220,3+ xorq %r10,%r14+ addq %r13,%r9+ vpxor %xmm10,%xmm8,%xmm8+ xorq %r11,%rdi+ rorq $28,%r14+ vpsrlq $6,%xmm4,%xmm10+ addq %r9,%rbx+ addq %rdi,%r9+ vpaddq %xmm8,%xmm5,%xmm5+ movq %rbx,%r13+ addq %r9,%r14+.byte 143,72,120,195,203,42+ rorq $23,%r13+ movq %r14,%r9+ vpxor %xmm10,%xmm11,%xmm11+ movq %rcx,%r12+ rorq $5,%r14+ xorq %rbx,%r13+ xorq %rdx,%r12+ vpxor %xmm9,%xmm11,%xmm11+ rorq $4,%r13+ xorq %r9,%r14+ andq %rbx,%r12+ xorq %rbx,%r13+ vpaddq %xmm11,%xmm5,%xmm5+ addq 88(%rsp),%r8+ movq %r9,%rdi+ xorq %rdx,%r12+ rorq $6,%r14+ vpaddq 32(%rsi),%xmm5,%xmm10+ xorq %r10,%rdi+ addq %r12,%r8+ rorq $14,%r13+ andq %rdi,%r15+ xorq %r9,%r14+ addq %r13,%r8+ xorq %r10,%r15+ rorq $28,%r14+ addq %r8,%rax+ addq %r15,%r8+ movq %rax,%r13+ addq %r8,%r14+ vmovdqa %xmm10,80(%rsp)+ vpalignr $8,%xmm6,%xmm7,%xmm8+ rorq $23,%r13+ movq %r14,%r8+ vpalignr $8,%xmm2,%xmm3,%xmm11+ movq %rbx,%r12+ rorq $5,%r14+.byte 143,72,120,195,200,56+ xorq %rax,%r13+ xorq %rcx,%r12+ vpsrlq $7,%xmm8,%xmm8+ rorq $4,%r13+ xorq %r8,%r14+ vpaddq %xmm11,%xmm6,%xmm6+ andq %rax,%r12+ xorq %rax,%r13+ addq 96(%rsp),%rdx+ movq %r8,%r15+.byte 143,72,120,195,209,7+ xorq %rcx,%r12+ rorq $6,%r14+ vpxor %xmm9,%xmm8,%xmm8+ xorq %r9,%r15+ addq %r12,%rdx+ rorq $14,%r13+ andq %r15,%rdi+.byte 143,104,120,195,221,3+ xorq %r8,%r14+ addq %r13,%rdx+ vpxor %xmm10,%xmm8,%xmm8+ xorq %r9,%rdi+ rorq $28,%r14+ vpsrlq $6,%xmm5,%xmm10+ addq %rdx,%r11+ addq %rdi,%rdx+ vpaddq %xmm8,%xmm6,%xmm6+ movq %r11,%r13+ addq %rdx,%r14+.byte 143,72,120,195,203,42+ rorq $23,%r13+ movq %r14,%rdx+ vpxor %xmm10,%xmm11,%xmm11+ movq %rax,%r12+ rorq $5,%r14+ xorq %r11,%r13+ xorq %rbx,%r12+ vpxor %xmm9,%xmm11,%xmm11+ rorq $4,%r13+ xorq %rdx,%r14+ andq %r11,%r12+ xorq %r11,%r13+ vpaddq %xmm11,%xmm6,%xmm6+ addq 104(%rsp),%rcx+ movq %rdx,%rdi+ xorq %rbx,%r12+ rorq $6,%r14+ vpaddq 64(%rsi),%xmm6,%xmm10+ xorq %r8,%rdi+ addq %r12,%rcx+ rorq $14,%r13+ andq %rdi,%r15+ xorq %rdx,%r14+ addq %r13,%rcx+ xorq %r8,%r15+ rorq $28,%r14+ addq %rcx,%r10+ addq %r15,%rcx+ movq %r10,%r13+ addq %rcx,%r14+ vmovdqa %xmm10,96(%rsp)+ vpalignr $8,%xmm7,%xmm0,%xmm8+ rorq $23,%r13+ movq %r14,%rcx+ vpalignr $8,%xmm3,%xmm4,%xmm11+ movq %r11,%r12+ rorq $5,%r14+.byte 143,72,120,195,200,56+ xorq %r10,%r13+ xorq %rax,%r12+ vpsrlq $7,%xmm8,%xmm8+ rorq $4,%r13+ xorq %rcx,%r14+ vpaddq %xmm11,%xmm7,%xmm7+ andq %r10,%r12+ xorq %r10,%r13+ addq 112(%rsp),%rbx+ movq %rcx,%r15+.byte 143,72,120,195,209,7+ xorq %rax,%r12+ rorq $6,%r14+ vpxor %xmm9,%xmm8,%xmm8+ xorq %rdx,%r15+ addq %r12,%rbx+ rorq $14,%r13+ andq %r15,%rdi+.byte 143,104,120,195,222,3+ xorq %rcx,%r14+ addq %r13,%rbx+ vpxor %xmm10,%xmm8,%xmm8+ xorq %rdx,%rdi+ rorq $28,%r14+ vpsrlq $6,%xmm6,%xmm10+ addq %rbx,%r9+ addq %rdi,%rbx+ vpaddq %xmm8,%xmm7,%xmm7+ movq %r9,%r13+ addq %rbx,%r14+.byte 143,72,120,195,203,42+ rorq $23,%r13+ movq %r14,%rbx+ vpxor %xmm10,%xmm11,%xmm11+ movq %r10,%r12+ rorq $5,%r14+ xorq %r9,%r13+ xorq %r11,%r12+ vpxor %xmm9,%xmm11,%xmm11+ rorq $4,%r13+ xorq %rbx,%r14+ andq %r9,%r12+ xorq %r9,%r13+ vpaddq %xmm11,%xmm7,%xmm7+ addq 120(%rsp),%rax+ movq %rbx,%rdi+ xorq %r11,%r12+ rorq $6,%r14+ vpaddq 96(%rsi),%xmm7,%xmm10+ xorq %rcx,%rdi+ addq %r12,%rax+ rorq $14,%r13+ andq %rdi,%r15+ xorq %rbx,%r14+ addq %r13,%rax+ xorq %rcx,%r15+ rorq $28,%r14+ addq %rax,%r8+ addq %r15,%rax+ movq %r8,%r13+ addq %rax,%r14+ vmovdqa %xmm10,112(%rsp)+ cmpb $0,135(%rsi)+ jne L$xop_00_47+ rorq $23,%r13+ movq %r14,%rax+ movq %r9,%r12+ rorq $5,%r14+ xorq %r8,%r13+ xorq %r10,%r12+ rorq $4,%r13+ xorq %rax,%r14+ andq %r8,%r12+ xorq %r8,%r13+ addq 0(%rsp),%r11+ movq %rax,%r15+ xorq %r10,%r12+ rorq $6,%r14+ xorq %rbx,%r15+ addq %r12,%r11+ rorq $14,%r13+ andq %r15,%rdi+ xorq %rax,%r14+ addq %r13,%r11+ xorq %rbx,%rdi+ rorq $28,%r14+ addq %r11,%rdx+ addq %rdi,%r11+ movq %rdx,%r13+ addq %r11,%r14+ rorq $23,%r13+ movq %r14,%r11+ movq %r8,%r12+ rorq $5,%r14+ xorq %rdx,%r13+ xorq %r9,%r12+ rorq $4,%r13+ xorq %r11,%r14+ andq %rdx,%r12+ xorq %rdx,%r13+ addq 8(%rsp),%r10+ movq %r11,%rdi+ xorq %r9,%r12+ rorq $6,%r14+ xorq %rax,%rdi+ addq %r12,%r10+ rorq $14,%r13+ andq %rdi,%r15+ xorq %r11,%r14+ addq %r13,%r10+ xorq %rax,%r15+ rorq $28,%r14+ addq %r10,%rcx+ addq %r15,%r10+ movq %rcx,%r13+ addq %r10,%r14+ rorq $23,%r13+ movq %r14,%r10+ movq %rdx,%r12+ rorq $5,%r14+ xorq %rcx,%r13+ xorq %r8,%r12+ rorq $4,%r13+ xorq %r10,%r14+ andq %rcx,%r12+ xorq %rcx,%r13+ addq 16(%rsp),%r9+ movq %r10,%r15+ xorq %r8,%r12+ rorq $6,%r14+ xorq %r11,%r15+ addq %r12,%r9+ rorq $14,%r13+ andq %r15,%rdi+ xorq %r10,%r14+ addq %r13,%r9+ xorq %r11,%rdi+ rorq $28,%r14+ addq %r9,%rbx+ addq %rdi,%r9+ movq %rbx,%r13+ addq %r9,%r14+ rorq $23,%r13+ movq %r14,%r9+ movq %rcx,%r12+ rorq $5,%r14+ xorq %rbx,%r13+ xorq %rdx,%r12+ rorq $4,%r13+ xorq %r9,%r14+ andq %rbx,%r12+ xorq %rbx,%r13+ addq 24(%rsp),%r8+ movq %r9,%rdi+ xorq %rdx,%r12+ rorq $6,%r14+ xorq %r10,%rdi+ addq %r12,%r8+ rorq $14,%r13+ andq %rdi,%r15+ xorq %r9,%r14+ addq %r13,%r8+ xorq %r10,%r15+ rorq $28,%r14+ addq %r8,%rax+ addq %r15,%r8+ movq %rax,%r13+ addq %r8,%r14+ rorq $23,%r13+ movq %r14,%r8+ movq %rbx,%r12+ rorq $5,%r14+ xorq %rax,%r13+ xorq %rcx,%r12+ rorq $4,%r13+ xorq %r8,%r14+ andq %rax,%r12+ xorq %rax,%r13+ addq 32(%rsp),%rdx+ movq %r8,%r15+ xorq %rcx,%r12+ rorq $6,%r14+ xorq %r9,%r15+ addq %r12,%rdx+ rorq $14,%r13+ andq %r15,%rdi+ xorq %r8,%r14+ addq %r13,%rdx+ xorq %r9,%rdi+ rorq $28,%r14+ addq %rdx,%r11+ addq %rdi,%rdx+ movq %r11,%r13+ addq %rdx,%r14+ rorq $23,%r13+ movq %r14,%rdx+ movq %rax,%r12+ rorq $5,%r14+ xorq %r11,%r13+ xorq %rbx,%r12+ rorq $4,%r13+ xorq %rdx,%r14+ andq %r11,%r12+ xorq %r11,%r13+ addq 40(%rsp),%rcx+ movq %rdx,%rdi+ xorq %rbx,%r12+ rorq $6,%r14+ xorq %r8,%rdi+ addq %r12,%rcx+ rorq $14,%r13+ andq %rdi,%r15+ xorq %rdx,%r14+ addq %r13,%rcx+ xorq %r8,%r15+ rorq $28,%r14+ addq %rcx,%r10+ addq %r15,%rcx+ movq %r10,%r13+ addq %rcx,%r14+ rorq $23,%r13+ movq %r14,%rcx+ movq %r11,%r12+ rorq $5,%r14+ xorq %r10,%r13+ xorq %rax,%r12+ rorq $4,%r13+ xorq %rcx,%r14+ andq %r10,%r12+ xorq %r10,%r13+ addq 48(%rsp),%rbx+ movq %rcx,%r15+ xorq %rax,%r12+ rorq $6,%r14+ xorq %rdx,%r15+ addq %r12,%rbx+ rorq $14,%r13+ andq %r15,%rdi+ xorq %rcx,%r14+ addq %r13,%rbx+ xorq %rdx,%rdi+ rorq $28,%r14+ addq %rbx,%r9+ addq %rdi,%rbx+ movq %r9,%r13+ addq %rbx,%r14+ rorq $23,%r13+ movq %r14,%rbx+ movq %r10,%r12+ rorq $5,%r14+ xorq %r9,%r13+ xorq %r11,%r12+ rorq $4,%r13+ xorq %rbx,%r14+ andq %r9,%r12+ xorq %r9,%r13+ addq 56(%rsp),%rax+ movq %rbx,%rdi+ xorq %r11,%r12+ rorq $6,%r14+ xorq %rcx,%rdi+ addq %r12,%rax+ rorq $14,%r13+ andq %rdi,%r15+ xorq %rbx,%r14+ addq %r13,%rax+ xorq %rcx,%r15+ rorq $28,%r14+ addq %rax,%r8+ addq %r15,%rax+ movq %r8,%r13+ addq %rax,%r14+ rorq $23,%r13+ movq %r14,%rax+ movq %r9,%r12+ rorq $5,%r14+ xorq %r8,%r13+ xorq %r10,%r12+ rorq $4,%r13+ xorq %rax,%r14+ andq %r8,%r12+ xorq %r8,%r13+ addq 64(%rsp),%r11+ movq %rax,%r15+ xorq %r10,%r12+ rorq $6,%r14+ xorq %rbx,%r15+ addq %r12,%r11+ rorq $14,%r13+ andq %r15,%rdi+ xorq %rax,%r14+ addq %r13,%r11+ xorq %rbx,%rdi+ rorq $28,%r14+ addq %r11,%rdx+ addq %rdi,%r11+ movq %rdx,%r13+ addq %r11,%r14+ rorq $23,%r13+ movq %r14,%r11+ movq %r8,%r12+ rorq $5,%r14+ xorq %rdx,%r13+ xorq %r9,%r12+ rorq $4,%r13+ xorq %r11,%r14+ andq %rdx,%r12+ xorq %rdx,%r13+ addq 72(%rsp),%r10+ movq %r11,%rdi+ xorq %r9,%r12+ rorq $6,%r14+ xorq %rax,%rdi+ addq %r12,%r10+ rorq $14,%r13+ andq %rdi,%r15+ xorq %r11,%r14+ addq %r13,%r10+ xorq %rax,%r15+ rorq $28,%r14+ addq %r10,%rcx+ addq %r15,%r10+ movq %rcx,%r13+ addq %r10,%r14+ rorq $23,%r13+ movq %r14,%r10+ movq %rdx,%r12+ rorq $5,%r14+ xorq %rcx,%r13+ xorq %r8,%r12+ rorq $4,%r13+ xorq %r10,%r14+ andq %rcx,%r12+ xorq %rcx,%r13+ addq 80(%rsp),%r9+ movq %r10,%r15+ xorq %r8,%r12+ rorq $6,%r14+ xorq %r11,%r15+ addq %r12,%r9+ rorq $14,%r13+ andq %r15,%rdi+ xorq %r10,%r14+ addq %r13,%r9+ xorq %r11,%rdi+ rorq $28,%r14+ addq %r9,%rbx+ addq %rdi,%r9+ movq %rbx,%r13+ addq %r9,%r14+ rorq $23,%r13+ movq %r14,%r9+ movq %rcx,%r12+ rorq $5,%r14+ xorq %rbx,%r13+ xorq %rdx,%r12+ rorq $4,%r13+ xorq %r9,%r14+ andq %rbx,%r12+ xorq %rbx,%r13+ addq 88(%rsp),%r8+ movq %r9,%rdi+ xorq %rdx,%r12+ rorq $6,%r14+ xorq %r10,%rdi+ addq %r12,%r8+ rorq $14,%r13+ andq %rdi,%r15+ xorq %r9,%r14+ addq %r13,%r8+ xorq %r10,%r15+ rorq $28,%r14+ addq %r8,%rax+ addq %r15,%r8+ movq %rax,%r13+ addq %r8,%r14+ rorq $23,%r13+ movq %r14,%r8+ movq %rbx,%r12+ rorq $5,%r14+ xorq %rax,%r13+ xorq %rcx,%r12+ rorq $4,%r13+ xorq %r8,%r14+ andq %rax,%r12+ xorq %rax,%r13+ addq 96(%rsp),%rdx+ movq %r8,%r15+ xorq %rcx,%r12+ rorq $6,%r14+ xorq %r9,%r15+ addq %r12,%rdx+ rorq $14,%r13+ andq %r15,%rdi+ xorq %r8,%r14+ addq %r13,%rdx+ xorq %r9,%rdi+ rorq $28,%r14+ addq %rdx,%r11+ addq %rdi,%rdx+ movq %r11,%r13+ addq %rdx,%r14+ rorq $23,%r13+ movq %r14,%rdx+ movq %rax,%r12+ rorq $5,%r14+ xorq %r11,%r13+ xorq %rbx,%r12+ rorq $4,%r13+ xorq %rdx,%r14+ andq %r11,%r12+ xorq %r11,%r13+ addq 104(%rsp),%rcx+ movq %rdx,%rdi+ xorq %rbx,%r12+ rorq $6,%r14+ xorq %r8,%rdi+ addq %r12,%rcx+ rorq $14,%r13+ andq %rdi,%r15+ xorq %rdx,%r14+ addq %r13,%rcx+ xorq %r8,%r15+ rorq $28,%r14+ addq %rcx,%r10+ addq %r15,%rcx+ movq %r10,%r13+ addq %rcx,%r14+ rorq $23,%r13+ movq %r14,%rcx+ movq %r11,%r12+ rorq $5,%r14+ xorq %r10,%r13+ xorq %rax,%r12+ rorq $4,%r13+ xorq %rcx,%r14+ andq %r10,%r12+ xorq %r10,%r13+ addq 112(%rsp),%rbx+ movq %rcx,%r15+ xorq %rax,%r12+ rorq $6,%r14+ xorq %rdx,%r15+ addq %r12,%rbx+ rorq $14,%r13+ andq %r15,%rdi+ xorq %rcx,%r14+ addq %r13,%rbx+ xorq %rdx,%rdi+ rorq $28,%r14+ addq %rbx,%r9+ addq %rdi,%rbx+ movq %r9,%r13+ addq %rbx,%r14+ rorq $23,%r13+ movq %r14,%rbx+ movq %r10,%r12+ rorq $5,%r14+ xorq %r9,%r13+ xorq %r11,%r12+ rorq $4,%r13+ xorq %rbx,%r14+ andq %r9,%r12+ xorq %r9,%r13+ addq 120(%rsp),%rax+ movq %rbx,%rdi+ xorq %r11,%r12+ rorq $6,%r14+ xorq %rcx,%rdi+ addq %r12,%rax+ rorq $14,%r13+ andq %rdi,%r15+ xorq %rbx,%r14+ addq %r13,%rax+ xorq %rcx,%r15+ rorq $28,%r14+ addq %rax,%r8+ addq %r15,%rax+ movq %r8,%r13+ addq %rax,%r14+ movq -64(%rbp),%rdi+ movq %r14,%rax+ movq -56(%rbp),%rsi++ addq 0(%rdi),%rax+ addq 8(%rdi),%rbx+ addq 16(%rdi),%rcx+ addq 24(%rdi),%rdx+ addq 32(%rdi),%r8+ addq 40(%rdi),%r9+ addq 48(%rdi),%r10+ addq 56(%rdi),%r11++ leaq 128(%rsi),%rsi+ cmpq -48(%rbp),%rsi++ movq %rax,0(%rdi)+ movq %rbx,8(%rdi)+ movq %rcx,16(%rdi)+ movq %rdx,24(%rdi)+ movq %r8,32(%rdi)+ movq %r9,40(%rdi)+ movq %r10,48(%rdi)+ movq %r11,56(%rdi)+ jb L$loop_xop++ vzeroupper+ movq -40(%rbp),%r15+ movq -32(%rbp),%r14+ movq -24(%rbp),%r13+ movq -16(%rbp),%r12+ movq -8(%rbp),%rbx+ movq %rbp,%rsp+.cfi_def_cfa_register %rsp+ popq %rbp+.cfi_adjust_cfa_offset -8+.cfi_restore %rbp+.cfi_restore %r12+.cfi_restore %r13+.cfi_restore %r14+.cfi_restore %r15+.cfi_restore %rbx+ .byte 0xf3,0xc3+.cfi_endproc +++.p2align 6+crypton_sha512_asm_block_data_order_avx:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa+++ pushq %rbp+.cfi_adjust_cfa_offset 8+.cfi_offset %rbp,-16+ movq %rsp,%rbp+.cfi_def_cfa_register %rbp+L$avx_shortcut:+ pushq %rbx+.cfi_offset %rbx,-24+ pushq %r12+.cfi_offset %r12,-32+ pushq %r13+.cfi_offset %r13,-40+ pushq %r14+.cfi_offset %r14,-48+ pushq %r15+.cfi_offset %r15,-56+ shlq $4,%rdx+ subq $24,%rsp++ leaq (%rsi,%rdx,8),%rdx+ movq %rdi,-64(%rbp)++ movq %rdx,-48(%rbp)+++ leaq -128(%rsp),%rsp+ vzeroupper+ andq $-64,%rsp+ movq 0(%rdi),%rax+ movq 8(%rdi),%rbx+ movq 16(%rdi),%rcx+ movq 24(%rdi),%rdx+ movq 32(%rdi),%r8+ movq 40(%rdi),%r9+ movq 48(%rdi),%r10+ movq 56(%rdi),%r11+ jmp L$loop_avx+.p2align 4+L$loop_avx:+ vmovdqa K512+1280(%rip),%xmm11+ movq %rsi,-56(%rbp)+ vmovdqu 0(%rsi),%xmm0+ vmovdqu 16(%rsi),%xmm1+ vmovdqu 32(%rsi),%xmm2+ vpshufb %xmm11,%xmm0,%xmm0+ vmovdqu 48(%rsi),%xmm3+ vpshufb %xmm11,%xmm1,%xmm1+ vmovdqu 64(%rsi),%xmm4+ vpshufb %xmm11,%xmm2,%xmm2+ vmovdqu 80(%rsi),%xmm5+ vpshufb %xmm11,%xmm3,%xmm3+ vmovdqu 96(%rsi),%xmm6+ vpshufb %xmm11,%xmm4,%xmm4+ vmovdqu 112(%rsi),%xmm7+ leaq K512+128(%rip),%rsi+ vpshufb %xmm11,%xmm5,%xmm5+ vpaddq -128(%rsi),%xmm0,%xmm8+ vpshufb %xmm11,%xmm6,%xmm6+ vpaddq -96(%rsi),%xmm1,%xmm9+ vpshufb %xmm11,%xmm7,%xmm7+ vpaddq -64(%rsi),%xmm2,%xmm10+ vpaddq -32(%rsi),%xmm3,%xmm11+ vmovdqa %xmm8,0(%rsp)+ vpaddq 0(%rsi),%xmm4,%xmm8+ vmovdqa %xmm9,16(%rsp)+ vpaddq 32(%rsi),%xmm5,%xmm9+ vmovdqa %xmm10,32(%rsp)+ vpaddq 64(%rsi),%xmm6,%xmm10+ vmovdqa %xmm11,48(%rsp)+ vpaddq 96(%rsi),%xmm7,%xmm11+ vmovdqa %xmm8,64(%rsp)+ movq %rax,%r14+ vmovdqa %xmm9,80(%rsp)+ movq %rbx,%rdi+ vmovdqa %xmm10,96(%rsp)+ xorq %rcx,%rdi+ vmovdqa %xmm11,112(%rsp)+ movq %r8,%r13+ jmp L$avx_00_47++.p2align 4+L$avx_00_47:+ addq $256,%rsi+ vpalignr $8,%xmm0,%xmm1,%xmm8+ shrdq $23,%r13,%r13+ movq %r14,%rax+ vpalignr $8,%xmm4,%xmm5,%xmm11+ movq %r9,%r12+ shrdq $5,%r14,%r14+ vpsrlq $1,%xmm8,%xmm10+ xorq %r8,%r13+ xorq %r10,%r12+ vpaddq %xmm11,%xmm0,%xmm0+ shrdq $4,%r13,%r13+ xorq %rax,%r14+ vpsrlq $7,%xmm8,%xmm11+ andq %r8,%r12+ xorq %r8,%r13+ vpsllq $56,%xmm8,%xmm9+ addq 0(%rsp),%r11+ movq %rax,%r15+ vpxor %xmm10,%xmm11,%xmm8+ xorq %r10,%r12+ shrdq $6,%r14,%r14+ vpsrlq $7,%xmm10,%xmm10+ xorq %rbx,%r15+ addq %r12,%r11+ vpxor %xmm9,%xmm8,%xmm8+ shrdq $14,%r13,%r13+ andq %r15,%rdi+ vpsllq $7,%xmm9,%xmm9+ xorq %rax,%r14+ addq %r13,%r11+ vpxor %xmm10,%xmm8,%xmm8+ xorq %rbx,%rdi+ shrdq $28,%r14,%r14+ vpsrlq $6,%xmm7,%xmm11+ addq %r11,%rdx+ addq %rdi,%r11+ vpxor %xmm9,%xmm8,%xmm8+ movq %rdx,%r13+ addq %r11,%r14+ vpsllq $3,%xmm7,%xmm10+ shrdq $23,%r13,%r13+ movq %r14,%r11+ vpaddq %xmm8,%xmm0,%xmm0+ movq %r8,%r12+ shrdq $5,%r14,%r14+ vpsrlq $19,%xmm7,%xmm9+ xorq %rdx,%r13+ xorq %r9,%r12+ vpxor %xmm10,%xmm11,%xmm11+ shrdq $4,%r13,%r13+ xorq %r11,%r14+ vpsllq $42,%xmm10,%xmm10+ andq %rdx,%r12+ xorq %rdx,%r13+ vpxor %xmm9,%xmm11,%xmm11+ addq 8(%rsp),%r10+ movq %r11,%rdi+ vpsrlq $42,%xmm9,%xmm9+ xorq %r9,%r12+ shrdq $6,%r14,%r14+ vpxor %xmm10,%xmm11,%xmm11+ xorq %rax,%rdi+ addq %r12,%r10+ vpxor %xmm9,%xmm11,%xmm11+ shrdq $14,%r13,%r13+ andq %rdi,%r15+ vpaddq %xmm11,%xmm0,%xmm0+ xorq %r11,%r14+ addq %r13,%r10+ vpaddq -128(%rsi),%xmm0,%xmm10+ xorq %rax,%r15+ shrdq $28,%r14,%r14+ addq %r10,%rcx+ addq %r15,%r10+ movq %rcx,%r13+ addq %r10,%r14+ vmovdqa %xmm10,0(%rsp)+ vpalignr $8,%xmm1,%xmm2,%xmm8+ shrdq $23,%r13,%r13+ movq %r14,%r10+ vpalignr $8,%xmm5,%xmm6,%xmm11+ movq %rdx,%r12+ shrdq $5,%r14,%r14+ vpsrlq $1,%xmm8,%xmm10+ xorq %rcx,%r13+ xorq %r8,%r12+ vpaddq %xmm11,%xmm1,%xmm1+ shrdq $4,%r13,%r13+ xorq %r10,%r14+ vpsrlq $7,%xmm8,%xmm11+ andq %rcx,%r12+ xorq %rcx,%r13+ vpsllq $56,%xmm8,%xmm9+ addq 16(%rsp),%r9+ movq %r10,%r15+ vpxor %xmm10,%xmm11,%xmm8+ xorq %r8,%r12+ shrdq $6,%r14,%r14+ vpsrlq $7,%xmm10,%xmm10+ xorq %r11,%r15+ addq %r12,%r9+ vpxor %xmm9,%xmm8,%xmm8+ shrdq $14,%r13,%r13+ andq %r15,%rdi+ vpsllq $7,%xmm9,%xmm9+ xorq %r10,%r14+ addq %r13,%r9+ vpxor %xmm10,%xmm8,%xmm8+ xorq %r11,%rdi+ shrdq $28,%r14,%r14+ vpsrlq $6,%xmm0,%xmm11+ addq %r9,%rbx+ addq %rdi,%r9+ vpxor %xmm9,%xmm8,%xmm8+ movq %rbx,%r13+ addq %r9,%r14+ vpsllq $3,%xmm0,%xmm10+ shrdq $23,%r13,%r13+ movq %r14,%r9+ vpaddq %xmm8,%xmm1,%xmm1+ movq %rcx,%r12+ shrdq $5,%r14,%r14+ vpsrlq $19,%xmm0,%xmm9+ xorq %rbx,%r13+ xorq %rdx,%r12+ vpxor %xmm10,%xmm11,%xmm11+ shrdq $4,%r13,%r13+ xorq %r9,%r14+ vpsllq $42,%xmm10,%xmm10+ andq %rbx,%r12+ xorq %rbx,%r13+ vpxor %xmm9,%xmm11,%xmm11+ addq 24(%rsp),%r8+ movq %r9,%rdi+ vpsrlq $42,%xmm9,%xmm9+ xorq %rdx,%r12+ shrdq $6,%r14,%r14+ vpxor %xmm10,%xmm11,%xmm11+ xorq %r10,%rdi+ addq %r12,%r8+ vpxor %xmm9,%xmm11,%xmm11+ shrdq $14,%r13,%r13+ andq %rdi,%r15+ vpaddq %xmm11,%xmm1,%xmm1+ xorq %r9,%r14+ addq %r13,%r8+ vpaddq -96(%rsi),%xmm1,%xmm10+ xorq %r10,%r15+ shrdq $28,%r14,%r14+ addq %r8,%rax+ addq %r15,%r8+ movq %rax,%r13+ addq %r8,%r14+ vmovdqa %xmm10,16(%rsp)+ vpalignr $8,%xmm2,%xmm3,%xmm8+ shrdq $23,%r13,%r13+ movq %r14,%r8+ vpalignr $8,%xmm6,%xmm7,%xmm11+ movq %rbx,%r12+ shrdq $5,%r14,%r14+ vpsrlq $1,%xmm8,%xmm10+ xorq %rax,%r13+ xorq %rcx,%r12+ vpaddq %xmm11,%xmm2,%xmm2+ shrdq $4,%r13,%r13+ xorq %r8,%r14+ vpsrlq $7,%xmm8,%xmm11+ andq %rax,%r12+ xorq %rax,%r13+ vpsllq $56,%xmm8,%xmm9+ addq 32(%rsp),%rdx+ movq %r8,%r15+ vpxor %xmm10,%xmm11,%xmm8+ xorq %rcx,%r12+ shrdq $6,%r14,%r14+ vpsrlq $7,%xmm10,%xmm10+ xorq %r9,%r15+ addq %r12,%rdx+ vpxor %xmm9,%xmm8,%xmm8+ shrdq $14,%r13,%r13+ andq %r15,%rdi+ vpsllq $7,%xmm9,%xmm9+ xorq %r8,%r14+ addq %r13,%rdx+ vpxor %xmm10,%xmm8,%xmm8+ xorq %r9,%rdi+ shrdq $28,%r14,%r14+ vpsrlq $6,%xmm1,%xmm11+ addq %rdx,%r11+ addq %rdi,%rdx+ vpxor %xmm9,%xmm8,%xmm8+ movq %r11,%r13+ addq %rdx,%r14+ vpsllq $3,%xmm1,%xmm10+ shrdq $23,%r13,%r13+ movq %r14,%rdx+ vpaddq %xmm8,%xmm2,%xmm2+ movq %rax,%r12+ shrdq $5,%r14,%r14+ vpsrlq $19,%xmm1,%xmm9+ xorq %r11,%r13+ xorq %rbx,%r12+ vpxor %xmm10,%xmm11,%xmm11+ shrdq $4,%r13,%r13+ xorq %rdx,%r14+ vpsllq $42,%xmm10,%xmm10+ andq %r11,%r12+ xorq %r11,%r13+ vpxor %xmm9,%xmm11,%xmm11+ addq 40(%rsp),%rcx+ movq %rdx,%rdi+ vpsrlq $42,%xmm9,%xmm9+ xorq %rbx,%r12+ shrdq $6,%r14,%r14+ vpxor %xmm10,%xmm11,%xmm11+ xorq %r8,%rdi+ addq %r12,%rcx+ vpxor %xmm9,%xmm11,%xmm11+ shrdq $14,%r13,%r13+ andq %rdi,%r15+ vpaddq %xmm11,%xmm2,%xmm2+ xorq %rdx,%r14+ addq %r13,%rcx+ vpaddq -64(%rsi),%xmm2,%xmm10+ xorq %r8,%r15+ shrdq $28,%r14,%r14+ addq %rcx,%r10+ addq %r15,%rcx+ movq %r10,%r13+ addq %rcx,%r14+ vmovdqa %xmm10,32(%rsp)+ vpalignr $8,%xmm3,%xmm4,%xmm8+ shrdq $23,%r13,%r13+ movq %r14,%rcx+ vpalignr $8,%xmm7,%xmm0,%xmm11+ movq %r11,%r12+ shrdq $5,%r14,%r14+ vpsrlq $1,%xmm8,%xmm10+ xorq %r10,%r13+ xorq %rax,%r12+ vpaddq %xmm11,%xmm3,%xmm3+ shrdq $4,%r13,%r13+ xorq %rcx,%r14+ vpsrlq $7,%xmm8,%xmm11+ andq %r10,%r12+ xorq %r10,%r13+ vpsllq $56,%xmm8,%xmm9+ addq 48(%rsp),%rbx+ movq %rcx,%r15+ vpxor %xmm10,%xmm11,%xmm8+ xorq %rax,%r12+ shrdq $6,%r14,%r14+ vpsrlq $7,%xmm10,%xmm10+ xorq %rdx,%r15+ addq %r12,%rbx+ vpxor %xmm9,%xmm8,%xmm8+ shrdq $14,%r13,%r13+ andq %r15,%rdi+ vpsllq $7,%xmm9,%xmm9+ xorq %rcx,%r14+ addq %r13,%rbx+ vpxor %xmm10,%xmm8,%xmm8+ xorq %rdx,%rdi+ shrdq $28,%r14,%r14+ vpsrlq $6,%xmm2,%xmm11+ addq %rbx,%r9+ addq %rdi,%rbx+ vpxor %xmm9,%xmm8,%xmm8+ movq %r9,%r13+ addq %rbx,%r14+ vpsllq $3,%xmm2,%xmm10+ shrdq $23,%r13,%r13+ movq %r14,%rbx+ vpaddq %xmm8,%xmm3,%xmm3+ movq %r10,%r12+ shrdq $5,%r14,%r14+ vpsrlq $19,%xmm2,%xmm9+ xorq %r9,%r13+ xorq %r11,%r12+ vpxor %xmm10,%xmm11,%xmm11+ shrdq $4,%r13,%r13+ xorq %rbx,%r14+ vpsllq $42,%xmm10,%xmm10+ andq %r9,%r12+ xorq %r9,%r13+ vpxor %xmm9,%xmm11,%xmm11+ addq 56(%rsp),%rax+ movq %rbx,%rdi+ vpsrlq $42,%xmm9,%xmm9+ xorq %r11,%r12+ shrdq $6,%r14,%r14+ vpxor %xmm10,%xmm11,%xmm11+ xorq %rcx,%rdi+ addq %r12,%rax+ vpxor %xmm9,%xmm11,%xmm11+ shrdq $14,%r13,%r13+ andq %rdi,%r15+ vpaddq %xmm11,%xmm3,%xmm3+ xorq %rbx,%r14+ addq %r13,%rax+ vpaddq -32(%rsi),%xmm3,%xmm10+ xorq %rcx,%r15+ shrdq $28,%r14,%r14+ addq %rax,%r8+ addq %r15,%rax+ movq %r8,%r13+ addq %rax,%r14+ vmovdqa %xmm10,48(%rsp)+ vpalignr $8,%xmm4,%xmm5,%xmm8+ shrdq $23,%r13,%r13+ movq %r14,%rax+ vpalignr $8,%xmm0,%xmm1,%xmm11+ movq %r9,%r12+ shrdq $5,%r14,%r14+ vpsrlq $1,%xmm8,%xmm10+ xorq %r8,%r13+ xorq %r10,%r12+ vpaddq %xmm11,%xmm4,%xmm4+ shrdq $4,%r13,%r13+ xorq %rax,%r14+ vpsrlq $7,%xmm8,%xmm11+ andq %r8,%r12+ xorq %r8,%r13+ vpsllq $56,%xmm8,%xmm9+ addq 64(%rsp),%r11+ movq %rax,%r15+ vpxor %xmm10,%xmm11,%xmm8+ xorq %r10,%r12+ shrdq $6,%r14,%r14+ vpsrlq $7,%xmm10,%xmm10+ xorq %rbx,%r15+ addq %r12,%r11+ vpxor %xmm9,%xmm8,%xmm8+ shrdq $14,%r13,%r13+ andq %r15,%rdi+ vpsllq $7,%xmm9,%xmm9+ xorq %rax,%r14+ addq %r13,%r11+ vpxor %xmm10,%xmm8,%xmm8+ xorq %rbx,%rdi+ shrdq $28,%r14,%r14+ vpsrlq $6,%xmm3,%xmm11+ addq %r11,%rdx+ addq %rdi,%r11+ vpxor %xmm9,%xmm8,%xmm8+ movq %rdx,%r13+ addq %r11,%r14+ vpsllq $3,%xmm3,%xmm10+ shrdq $23,%r13,%r13+ movq %r14,%r11+ vpaddq %xmm8,%xmm4,%xmm4+ movq %r8,%r12+ shrdq $5,%r14,%r14+ vpsrlq $19,%xmm3,%xmm9+ xorq %rdx,%r13+ xorq %r9,%r12+ vpxor %xmm10,%xmm11,%xmm11+ shrdq $4,%r13,%r13+ xorq %r11,%r14+ vpsllq $42,%xmm10,%xmm10+ andq %rdx,%r12+ xorq %rdx,%r13+ vpxor %xmm9,%xmm11,%xmm11+ addq 72(%rsp),%r10+ movq %r11,%rdi+ vpsrlq $42,%xmm9,%xmm9+ xorq %r9,%r12+ shrdq $6,%r14,%r14+ vpxor %xmm10,%xmm11,%xmm11+ xorq %rax,%rdi+ addq %r12,%r10+ vpxor %xmm9,%xmm11,%xmm11+ shrdq $14,%r13,%r13+ andq %rdi,%r15+ vpaddq %xmm11,%xmm4,%xmm4+ xorq %r11,%r14+ addq %r13,%r10+ vpaddq 0(%rsi),%xmm4,%xmm10+ xorq %rax,%r15+ shrdq $28,%r14,%r14+ addq %r10,%rcx+ addq %r15,%r10+ movq %rcx,%r13+ addq %r10,%r14+ vmovdqa %xmm10,64(%rsp)+ vpalignr $8,%xmm5,%xmm6,%xmm8+ shrdq $23,%r13,%r13+ movq %r14,%r10+ vpalignr $8,%xmm1,%xmm2,%xmm11+ movq %rdx,%r12+ shrdq $5,%r14,%r14+ vpsrlq $1,%xmm8,%xmm10+ xorq %rcx,%r13+ xorq %r8,%r12+ vpaddq %xmm11,%xmm5,%xmm5+ shrdq $4,%r13,%r13+ xorq %r10,%r14+ vpsrlq $7,%xmm8,%xmm11+ andq %rcx,%r12+ xorq %rcx,%r13+ vpsllq $56,%xmm8,%xmm9+ addq 80(%rsp),%r9+ movq %r10,%r15+ vpxor %xmm10,%xmm11,%xmm8+ xorq %r8,%r12+ shrdq $6,%r14,%r14+ vpsrlq $7,%xmm10,%xmm10+ xorq %r11,%r15+ addq %r12,%r9+ vpxor %xmm9,%xmm8,%xmm8+ shrdq $14,%r13,%r13+ andq %r15,%rdi+ vpsllq $7,%xmm9,%xmm9+ xorq %r10,%r14+ addq %r13,%r9+ vpxor %xmm10,%xmm8,%xmm8+ xorq %r11,%rdi+ shrdq $28,%r14,%r14+ vpsrlq $6,%xmm4,%xmm11+ addq %r9,%rbx+ addq %rdi,%r9+ vpxor %xmm9,%xmm8,%xmm8+ movq %rbx,%r13+ addq %r9,%r14+ vpsllq $3,%xmm4,%xmm10+ shrdq $23,%r13,%r13+ movq %r14,%r9+ vpaddq %xmm8,%xmm5,%xmm5+ movq %rcx,%r12+ shrdq $5,%r14,%r14+ vpsrlq $19,%xmm4,%xmm9+ xorq %rbx,%r13+ xorq %rdx,%r12+ vpxor %xmm10,%xmm11,%xmm11+ shrdq $4,%r13,%r13+ xorq %r9,%r14+ vpsllq $42,%xmm10,%xmm10+ andq %rbx,%r12+ xorq %rbx,%r13+ vpxor %xmm9,%xmm11,%xmm11+ addq 88(%rsp),%r8+ movq %r9,%rdi+ vpsrlq $42,%xmm9,%xmm9+ xorq %rdx,%r12+ shrdq $6,%r14,%r14+ vpxor %xmm10,%xmm11,%xmm11+ xorq %r10,%rdi+ addq %r12,%r8+ vpxor %xmm9,%xmm11,%xmm11+ shrdq $14,%r13,%r13+ andq %rdi,%r15+ vpaddq %xmm11,%xmm5,%xmm5+ xorq %r9,%r14+ addq %r13,%r8+ vpaddq 32(%rsi),%xmm5,%xmm10+ xorq %r10,%r15+ shrdq $28,%r14,%r14+ addq %r8,%rax+ addq %r15,%r8+ movq %rax,%r13+ addq %r8,%r14+ vmovdqa %xmm10,80(%rsp)+ vpalignr $8,%xmm6,%xmm7,%xmm8+ shrdq $23,%r13,%r13+ movq %r14,%r8+ vpalignr $8,%xmm2,%xmm3,%xmm11+ movq %rbx,%r12+ shrdq $5,%r14,%r14+ vpsrlq $1,%xmm8,%xmm10+ xorq %rax,%r13+ xorq %rcx,%r12+ vpaddq %xmm11,%xmm6,%xmm6+ shrdq $4,%r13,%r13+ xorq %r8,%r14+ vpsrlq $7,%xmm8,%xmm11+ andq %rax,%r12+ xorq %rax,%r13+ vpsllq $56,%xmm8,%xmm9+ addq 96(%rsp),%rdx+ movq %r8,%r15+ vpxor %xmm10,%xmm11,%xmm8+ xorq %rcx,%r12+ shrdq $6,%r14,%r14+ vpsrlq $7,%xmm10,%xmm10+ xorq %r9,%r15+ addq %r12,%rdx+ vpxor %xmm9,%xmm8,%xmm8+ shrdq $14,%r13,%r13+ andq %r15,%rdi+ vpsllq $7,%xmm9,%xmm9+ xorq %r8,%r14+ addq %r13,%rdx+ vpxor %xmm10,%xmm8,%xmm8+ xorq %r9,%rdi+ shrdq $28,%r14,%r14+ vpsrlq $6,%xmm5,%xmm11+ addq %rdx,%r11+ addq %rdi,%rdx+ vpxor %xmm9,%xmm8,%xmm8+ movq %r11,%r13+ addq %rdx,%r14+ vpsllq $3,%xmm5,%xmm10+ shrdq $23,%r13,%r13+ movq %r14,%rdx+ vpaddq %xmm8,%xmm6,%xmm6+ movq %rax,%r12+ shrdq $5,%r14,%r14+ vpsrlq $19,%xmm5,%xmm9+ xorq %r11,%r13+ xorq %rbx,%r12+ vpxor %xmm10,%xmm11,%xmm11+ shrdq $4,%r13,%r13+ xorq %rdx,%r14+ vpsllq $42,%xmm10,%xmm10+ andq %r11,%r12+ xorq %r11,%r13+ vpxor %xmm9,%xmm11,%xmm11+ addq 104(%rsp),%rcx+ movq %rdx,%rdi+ vpsrlq $42,%xmm9,%xmm9+ xorq %rbx,%r12+ shrdq $6,%r14,%r14+ vpxor %xmm10,%xmm11,%xmm11+ xorq %r8,%rdi+ addq %r12,%rcx+ vpxor %xmm9,%xmm11,%xmm11+ shrdq $14,%r13,%r13+ andq %rdi,%r15+ vpaddq %xmm11,%xmm6,%xmm6+ xorq %rdx,%r14+ addq %r13,%rcx+ vpaddq 64(%rsi),%xmm6,%xmm10+ xorq %r8,%r15+ shrdq $28,%r14,%r14+ addq %rcx,%r10+ addq %r15,%rcx+ movq %r10,%r13+ addq %rcx,%r14+ vmovdqa %xmm10,96(%rsp)+ vpalignr $8,%xmm7,%xmm0,%xmm8+ shrdq $23,%r13,%r13+ movq %r14,%rcx+ vpalignr $8,%xmm3,%xmm4,%xmm11+ movq %r11,%r12+ shrdq $5,%r14,%r14+ vpsrlq $1,%xmm8,%xmm10+ xorq %r10,%r13+ xorq %rax,%r12+ vpaddq %xmm11,%xmm7,%xmm7+ shrdq $4,%r13,%r13+ xorq %rcx,%r14+ vpsrlq $7,%xmm8,%xmm11+ andq %r10,%r12+ xorq %r10,%r13+ vpsllq $56,%xmm8,%xmm9+ addq 112(%rsp),%rbx+ movq %rcx,%r15+ vpxor %xmm10,%xmm11,%xmm8+ xorq %rax,%r12+ shrdq $6,%r14,%r14+ vpsrlq $7,%xmm10,%xmm10+ xorq %rdx,%r15+ addq %r12,%rbx+ vpxor %xmm9,%xmm8,%xmm8+ shrdq $14,%r13,%r13+ andq %r15,%rdi+ vpsllq $7,%xmm9,%xmm9+ xorq %rcx,%r14+ addq %r13,%rbx+ vpxor %xmm10,%xmm8,%xmm8+ xorq %rdx,%rdi+ shrdq $28,%r14,%r14+ vpsrlq $6,%xmm6,%xmm11+ addq %rbx,%r9+ addq %rdi,%rbx+ vpxor %xmm9,%xmm8,%xmm8+ movq %r9,%r13+ addq %rbx,%r14+ vpsllq $3,%xmm6,%xmm10+ shrdq $23,%r13,%r13+ movq %r14,%rbx+ vpaddq %xmm8,%xmm7,%xmm7+ movq %r10,%r12+ shrdq $5,%r14,%r14+ vpsrlq $19,%xmm6,%xmm9+ xorq %r9,%r13+ xorq %r11,%r12+ vpxor %xmm10,%xmm11,%xmm11+ shrdq $4,%r13,%r13+ xorq %rbx,%r14+ vpsllq $42,%xmm10,%xmm10+ andq %r9,%r12+ xorq %r9,%r13+ vpxor %xmm9,%xmm11,%xmm11+ addq 120(%rsp),%rax+ movq %rbx,%rdi+ vpsrlq $42,%xmm9,%xmm9+ xorq %r11,%r12+ shrdq $6,%r14,%r14+ vpxor %xmm10,%xmm11,%xmm11+ xorq %rcx,%rdi+ addq %r12,%rax+ vpxor %xmm9,%xmm11,%xmm11+ shrdq $14,%r13,%r13+ andq %rdi,%r15+ vpaddq %xmm11,%xmm7,%xmm7+ xorq %rbx,%r14+ addq %r13,%rax+ vpaddq 96(%rsi),%xmm7,%xmm10+ xorq %rcx,%r15+ shrdq $28,%r14,%r14+ addq %rax,%r8+ addq %r15,%rax+ movq %r8,%r13+ addq %rax,%r14+ vmovdqa %xmm10,112(%rsp)+ cmpb $0,135(%rsi)+ jne L$avx_00_47+ shrdq $23,%r13,%r13+ movq %r14,%rax+ movq %r9,%r12+ shrdq $5,%r14,%r14+ xorq %r8,%r13+ xorq %r10,%r12+ shrdq $4,%r13,%r13+ xorq %rax,%r14+ andq %r8,%r12+ xorq %r8,%r13+ addq 0(%rsp),%r11+ movq %rax,%r15+ xorq %r10,%r12+ shrdq $6,%r14,%r14+ xorq %rbx,%r15+ addq %r12,%r11+ shrdq $14,%r13,%r13+ andq %r15,%rdi+ xorq %rax,%r14+ addq %r13,%r11+ xorq %rbx,%rdi+ shrdq $28,%r14,%r14+ addq %r11,%rdx+ addq %rdi,%r11+ movq %rdx,%r13+ addq %r11,%r14+ shrdq $23,%r13,%r13+ movq %r14,%r11+ movq %r8,%r12+ shrdq $5,%r14,%r14+ xorq %rdx,%r13+ xorq %r9,%r12+ shrdq $4,%r13,%r13+ xorq %r11,%r14+ andq %rdx,%r12+ xorq %rdx,%r13+ addq 8(%rsp),%r10+ movq %r11,%rdi+ xorq %r9,%r12+ shrdq $6,%r14,%r14+ xorq %rax,%rdi+ addq %r12,%r10+ shrdq $14,%r13,%r13+ andq %rdi,%r15+ xorq %r11,%r14+ addq %r13,%r10+ xorq %rax,%r15+ shrdq $28,%r14,%r14+ addq %r10,%rcx+ addq %r15,%r10+ movq %rcx,%r13+ addq %r10,%r14+ shrdq $23,%r13,%r13+ movq %r14,%r10+ movq %rdx,%r12+ shrdq $5,%r14,%r14+ xorq %rcx,%r13+ xorq %r8,%r12+ shrdq $4,%r13,%r13+ xorq %r10,%r14+ andq %rcx,%r12+ xorq %rcx,%r13+ addq 16(%rsp),%r9+ movq %r10,%r15+ xorq %r8,%r12+ shrdq $6,%r14,%r14+ xorq %r11,%r15+ addq %r12,%r9+ shrdq $14,%r13,%r13+ andq %r15,%rdi+ xorq %r10,%r14+ addq %r13,%r9+ xorq %r11,%rdi+ shrdq $28,%r14,%r14+ addq %r9,%rbx+ addq %rdi,%r9+ movq %rbx,%r13+ addq %r9,%r14+ shrdq $23,%r13,%r13+ movq %r14,%r9+ movq %rcx,%r12+ shrdq $5,%r14,%r14+ xorq %rbx,%r13+ xorq %rdx,%r12+ shrdq $4,%r13,%r13+ xorq %r9,%r14+ andq %rbx,%r12+ xorq %rbx,%r13+ addq 24(%rsp),%r8+ movq %r9,%rdi+ xorq %rdx,%r12+ shrdq $6,%r14,%r14+ xorq %r10,%rdi+ addq %r12,%r8+ shrdq $14,%r13,%r13+ andq %rdi,%r15+ xorq %r9,%r14+ addq %r13,%r8+ xorq %r10,%r15+ shrdq $28,%r14,%r14+ addq %r8,%rax+ addq %r15,%r8+ movq %rax,%r13+ addq %r8,%r14+ shrdq $23,%r13,%r13+ movq %r14,%r8+ movq %rbx,%r12+ shrdq $5,%r14,%r14+ xorq %rax,%r13+ xorq %rcx,%r12+ shrdq $4,%r13,%r13+ xorq %r8,%r14+ andq %rax,%r12+ xorq %rax,%r13+ addq 32(%rsp),%rdx+ movq %r8,%r15+ xorq %rcx,%r12+ shrdq $6,%r14,%r14+ xorq %r9,%r15+ addq %r12,%rdx+ shrdq $14,%r13,%r13+ andq %r15,%rdi+ xorq %r8,%r14+ addq %r13,%rdx+ xorq %r9,%rdi+ shrdq $28,%r14,%r14+ addq %rdx,%r11+ addq %rdi,%rdx+ movq %r11,%r13+ addq %rdx,%r14+ shrdq $23,%r13,%r13+ movq %r14,%rdx+ movq %rax,%r12+ shrdq $5,%r14,%r14+ xorq %r11,%r13+ xorq %rbx,%r12+ shrdq $4,%r13,%r13+ xorq %rdx,%r14+ andq %r11,%r12+ xorq %r11,%r13+ addq 40(%rsp),%rcx+ movq %rdx,%rdi+ xorq %rbx,%r12+ shrdq $6,%r14,%r14+ xorq %r8,%rdi+ addq %r12,%rcx+ shrdq $14,%r13,%r13+ andq %rdi,%r15+ xorq %rdx,%r14+ addq %r13,%rcx+ xorq %r8,%r15+ shrdq $28,%r14,%r14+ addq %rcx,%r10+ addq %r15,%rcx+ movq %r10,%r13+ addq %rcx,%r14+ shrdq $23,%r13,%r13+ movq %r14,%rcx+ movq %r11,%r12+ shrdq $5,%r14,%r14+ xorq %r10,%r13+ xorq %rax,%r12+ shrdq $4,%r13,%r13+ xorq %rcx,%r14+ andq %r10,%r12+ xorq %r10,%r13+ addq 48(%rsp),%rbx+ movq %rcx,%r15+ xorq %rax,%r12+ shrdq $6,%r14,%r14+ xorq %rdx,%r15+ addq %r12,%rbx+ shrdq $14,%r13,%r13+ andq %r15,%rdi+ xorq %rcx,%r14+ addq %r13,%rbx+ xorq %rdx,%rdi+ shrdq $28,%r14,%r14+ addq %rbx,%r9+ addq %rdi,%rbx+ movq %r9,%r13+ addq %rbx,%r14+ shrdq $23,%r13,%r13+ movq %r14,%rbx+ movq %r10,%r12+ shrdq $5,%r14,%r14+ xorq %r9,%r13+ xorq %r11,%r12+ shrdq $4,%r13,%r13+ xorq %rbx,%r14+ andq %r9,%r12+ xorq %r9,%r13+ addq 56(%rsp),%rax+ movq %rbx,%rdi+ xorq %r11,%r12+ shrdq $6,%r14,%r14+ xorq %rcx,%rdi+ addq %r12,%rax+ shrdq $14,%r13,%r13+ andq %rdi,%r15+ xorq %rbx,%r14+ addq %r13,%rax+ xorq %rcx,%r15+ shrdq $28,%r14,%r14+ addq %rax,%r8+ addq %r15,%rax+ movq %r8,%r13+ addq %rax,%r14+ shrdq $23,%r13,%r13+ movq %r14,%rax+ movq %r9,%r12+ shrdq $5,%r14,%r14+ xorq %r8,%r13+ xorq %r10,%r12+ shrdq $4,%r13,%r13+ xorq %rax,%r14+ andq %r8,%r12+ xorq %r8,%r13+ addq 64(%rsp),%r11+ movq %rax,%r15+ xorq %r10,%r12+ shrdq $6,%r14,%r14+ xorq %rbx,%r15+ addq %r12,%r11+ shrdq $14,%r13,%r13+ andq %r15,%rdi+ xorq %rax,%r14+ addq %r13,%r11+ xorq %rbx,%rdi+ shrdq $28,%r14,%r14+ addq %r11,%rdx+ addq %rdi,%r11+ movq %rdx,%r13+ addq %r11,%r14+ shrdq $23,%r13,%r13+ movq %r14,%r11+ movq %r8,%r12+ shrdq $5,%r14,%r14+ xorq %rdx,%r13+ xorq %r9,%r12+ shrdq $4,%r13,%r13+ xorq %r11,%r14+ andq %rdx,%r12+ xorq %rdx,%r13+ addq 72(%rsp),%r10+ movq %r11,%rdi+ xorq %r9,%r12+ shrdq $6,%r14,%r14+ xorq %rax,%rdi+ addq %r12,%r10+ shrdq $14,%r13,%r13+ andq %rdi,%r15+ xorq %r11,%r14+ addq %r13,%r10+ xorq %rax,%r15+ shrdq $28,%r14,%r14+ addq %r10,%rcx+ addq %r15,%r10+ movq %rcx,%r13+ addq %r10,%r14+ shrdq $23,%r13,%r13+ movq %r14,%r10+ movq %rdx,%r12+ shrdq $5,%r14,%r14+ xorq %rcx,%r13+ xorq %r8,%r12+ shrdq $4,%r13,%r13+ xorq %r10,%r14+ andq %rcx,%r12+ xorq %rcx,%r13+ addq 80(%rsp),%r9+ movq %r10,%r15+ xorq %r8,%r12+ shrdq $6,%r14,%r14+ xorq %r11,%r15+ addq %r12,%r9+ shrdq $14,%r13,%r13+ andq %r15,%rdi+ xorq %r10,%r14+ addq %r13,%r9+ xorq %r11,%rdi+ shrdq $28,%r14,%r14+ addq %r9,%rbx+ addq %rdi,%r9+ movq %rbx,%r13+ addq %r9,%r14+ shrdq $23,%r13,%r13+ movq %r14,%r9+ movq %rcx,%r12+ shrdq $5,%r14,%r14+ xorq %rbx,%r13+ xorq %rdx,%r12+ shrdq $4,%r13,%r13+ xorq %r9,%r14+ andq %rbx,%r12+ xorq %rbx,%r13+ addq 88(%rsp),%r8+ movq %r9,%rdi+ xorq %rdx,%r12+ shrdq $6,%r14,%r14+ xorq %r10,%rdi+ addq %r12,%r8+ shrdq $14,%r13,%r13+ andq %rdi,%r15+ xorq %r9,%r14+ addq %r13,%r8+ xorq %r10,%r15+ shrdq $28,%r14,%r14+ addq %r8,%rax+ addq %r15,%r8+ movq %rax,%r13+ addq %r8,%r14+ shrdq $23,%r13,%r13+ movq %r14,%r8+ movq %rbx,%r12+ shrdq $5,%r14,%r14+ xorq %rax,%r13+ xorq %rcx,%r12+ shrdq $4,%r13,%r13+ xorq %r8,%r14+ andq %rax,%r12+ xorq %rax,%r13+ addq 96(%rsp),%rdx+ movq %r8,%r15+ xorq %rcx,%r12+ shrdq $6,%r14,%r14+ xorq %r9,%r15+ addq %r12,%rdx+ shrdq $14,%r13,%r13+ andq %r15,%rdi+ xorq %r8,%r14+ addq %r13,%rdx+ xorq %r9,%rdi+ shrdq $28,%r14,%r14+ addq %rdx,%r11+ addq %rdi,%rdx+ movq %r11,%r13+ addq %rdx,%r14+ shrdq $23,%r13,%r13+ movq %r14,%rdx+ movq %rax,%r12+ shrdq $5,%r14,%r14+ xorq %r11,%r13+ xorq %rbx,%r12+ shrdq $4,%r13,%r13+ xorq %rdx,%r14+ andq %r11,%r12+ xorq %r11,%r13+ addq 104(%rsp),%rcx+ movq %rdx,%rdi+ xorq %rbx,%r12+ shrdq $6,%r14,%r14+ xorq %r8,%rdi+ addq %r12,%rcx+ shrdq $14,%r13,%r13+ andq %rdi,%r15+ xorq %rdx,%r14+ addq %r13,%rcx+ xorq %r8,%r15+ shrdq $28,%r14,%r14+ addq %rcx,%r10+ addq %r15,%rcx+ movq %r10,%r13+ addq %rcx,%r14+ shrdq $23,%r13,%r13+ movq %r14,%rcx+ movq %r11,%r12+ shrdq $5,%r14,%r14+ xorq %r10,%r13+ xorq %rax,%r12+ shrdq $4,%r13,%r13+ xorq %rcx,%r14+ andq %r10,%r12+ xorq %r10,%r13+ addq 112(%rsp),%rbx+ movq %rcx,%r15+ xorq %rax,%r12+ shrdq $6,%r14,%r14+ xorq %rdx,%r15+ addq %r12,%rbx+ shrdq $14,%r13,%r13+ andq %r15,%rdi+ xorq %rcx,%r14+ addq %r13,%rbx+ xorq %rdx,%rdi+ shrdq $28,%r14,%r14+ addq %rbx,%r9+ addq %rdi,%rbx+ movq %r9,%r13+ addq %rbx,%r14+ shrdq $23,%r13,%r13+ movq %r14,%rbx+ movq %r10,%r12+ shrdq $5,%r14,%r14+ xorq %r9,%r13+ xorq %r11,%r12+ shrdq $4,%r13,%r13+ xorq %rbx,%r14+ andq %r9,%r12+ xorq %r9,%r13+ addq 120(%rsp),%rax+ movq %rbx,%rdi+ xorq %r11,%r12+ shrdq $6,%r14,%r14+ xorq %rcx,%rdi+ addq %r12,%rax+ shrdq $14,%r13,%r13+ andq %rdi,%r15+ xorq %rbx,%r14+ addq %r13,%rax+ xorq %rcx,%r15+ shrdq $28,%r14,%r14+ addq %rax,%r8+ addq %r15,%rax+ movq %r8,%r13+ addq %rax,%r14+ movq -64(%rbp),%rdi+ movq %r14,%rax+ movq -56(%rbp),%rsi++ addq 0(%rdi),%rax+ addq 8(%rdi),%rbx+ addq 16(%rdi),%rcx+ addq 24(%rdi),%rdx+ addq 32(%rdi),%r8+ addq 40(%rdi),%r9+ addq 48(%rdi),%r10+ addq 56(%rdi),%r11++ leaq 128(%rsi),%rsi+ cmpq -48(%rbp),%rsi++ movq %rax,0(%rdi)+ movq %rbx,8(%rdi)+ movq %rcx,16(%rdi)+ movq %rdx,24(%rdi)+ movq %r8,32(%rdi)+ movq %r9,40(%rdi)+ movq %r10,48(%rdi)+ movq %r11,56(%rdi)+ jb L$loop_avx++ vzeroupper+ movq -40(%rbp),%r15+ movq -32(%rbp),%r14+ movq -24(%rbp),%r13+ movq -16(%rbp),%r12+ movq -8(%rbp),%rbx+ movq %rbp,%rsp+.cfi_def_cfa_register %rsp+ popq %rbp+.cfi_adjust_cfa_offset -8+.cfi_restore %rbp+.cfi_restore %r12+.cfi_restore %r13+.cfi_restore %r14+.cfi_restore %r15+.cfi_restore %rbx+ .byte 0xf3,0xc3+.cfi_endproc +++.p2align 6+crypton_sha512_asm_block_data_order_avx2:+.cfi_startproc+ .byte 0xf3,0x0f,0x1e,0xfa+++ pushq %rbp+.cfi_adjust_cfa_offset 8+.cfi_offset %rbp,-16+ movq %rsp,%rbp+.cfi_def_cfa_register %rbp+L$avx2_shortcut:+ pushq %rbx+.cfi_offset %rbx,-24+ pushq %r12+.cfi_offset %r12,-32+ pushq %r13+.cfi_offset %r13,-40+ pushq %r14+.cfi_offset %r14,-48+ pushq %r15+.cfi_offset %r15,-56+ shlq $4,%rdx+ subq $24,%rsp++ leaq (%rsi,%rdx,8),%rdx+ movq %rdi,-64(%rbp)++ movq %rdx,-48(%rbp)+++ leaq -128(%rsp),%rsp+ vzeroupper+ andq $-128,%rsp+ subq $-128,%rsi+ movq 0(%rdi),%rax+ movq %rsi,%r12+ movq 8(%rdi),%rbx+ cmpq %rdx,%rsi+ movq 16(%rdi),%rcx+ cmoveq %rsp,%r12+ movq 24(%rdi),%rdx+ movq 32(%rdi),%r8+ movq 40(%rdi),%r9+ movq 48(%rdi),%r10+ movq 56(%rdi),%r11+ jmp L$oop_avx2+.p2align 4+L$oop_avx2:+ vmovdqa K512+1280(%rip),%ymm10+ movq %rsi,-56(%rbp)+ vmovdqu -128(%rsi),%xmm0+ vmovdqu -128+16(%rsi),%xmm1+ vmovdqu -128+32(%rsi),%xmm2+ vmovdqu -128+48(%rsi),%xmm3+ vmovdqu -128+64(%rsi),%xmm4+ vmovdqu -128+80(%rsi),%xmm5+ vmovdqu -128+96(%rsi),%xmm6+ vmovdqu -128+112(%rsi),%xmm7+ leaq K512+128(%rip),%rsi+ vinserti128 $1,(%r12),%ymm0,%ymm0+ vinserti128 $1,16(%r12),%ymm1,%ymm1+ vpshufb %ymm10,%ymm0,%ymm0+ vinserti128 $1,32(%r12),%ymm2,%ymm2+ vpshufb %ymm10,%ymm1,%ymm1+ vinserti128 $1,48(%r12),%ymm3,%ymm3+ vpshufb %ymm10,%ymm2,%ymm2+ vinserti128 $1,64(%r12),%ymm4,%ymm4+ vpshufb %ymm10,%ymm3,%ymm3+ vinserti128 $1,80(%r12),%ymm5,%ymm5+ vpshufb %ymm10,%ymm4,%ymm4+ vinserti128 $1,96(%r12),%ymm6,%ymm6+ vpshufb %ymm10,%ymm5,%ymm5+ vinserti128 $1,112(%r12),%ymm7,%ymm7++ vpaddq -128(%rsi),%ymm0,%ymm8+ vpshufb %ymm10,%ymm6,%ymm6+ vpaddq -96(%rsi),%ymm1,%ymm9+ vpshufb %ymm10,%ymm7,%ymm7+ vpaddq -64(%rsi),%ymm2,%ymm10+ vpaddq -32(%rsi),%ymm3,%ymm11+ vmovdqa %ymm8,0(%rsp)+ vpaddq 0(%rsi),%ymm4,%ymm8+ vmovdqa %ymm9,32(%rsp)+ vpaddq 32(%rsi),%ymm5,%ymm9+ vmovdqa %ymm10,64(%rsp)+ vpaddq 64(%rsi),%ymm6,%ymm10+ vmovdqa %ymm11,96(%rsp)+ leaq -128(%rsp),%rsp+ vpaddq 96(%rsi),%ymm7,%ymm11+ vmovdqa %ymm8,0(%rsp)+ xorq %r14,%r14+ vmovdqa %ymm9,32(%rsp)+ movq %rbx,%rdi+ vmovdqa %ymm10,64(%rsp)+ xorq %rcx,%rdi+ vmovdqa %ymm11,96(%rsp)+ movq %r9,%r12+ addq $32*8,%rsi+ jmp L$avx2_00_47++.p2align 4+L$avx2_00_47:+ leaq -128(%rsp),%rsp+ vpalignr $8,%ymm0,%ymm1,%ymm8+ addq 0+256(%rsp),%r11+ andq %r8,%r12+ rorxq $41,%r8,%r13+ vpalignr $8,%ymm4,%ymm5,%ymm11+ rorxq $18,%r8,%r15+ leaq (%rax,%r14,1),%rax+ leaq (%r11,%r12,1),%r11+ vpsrlq $1,%ymm8,%ymm10+ andnq %r10,%r8,%r12+ xorq %r15,%r13+ rorxq $14,%r8,%r14+ vpaddq %ymm11,%ymm0,%ymm0+ vpsrlq $7,%ymm8,%ymm11+ leaq (%r11,%r12,1),%r11+ xorq %r14,%r13+ movq %rax,%r15+ vpsllq $56,%ymm8,%ymm9+ vpxor %ymm10,%ymm11,%ymm8+ rorxq $39,%rax,%r12+ leaq (%r11,%r13,1),%r11+ xorq %rbx,%r15+ vpsrlq $7,%ymm10,%ymm10+ vpxor %ymm9,%ymm8,%ymm8+ rorxq $34,%rax,%r14+ rorxq $28,%rax,%r13+ leaq (%rdx,%r11,1),%rdx+ vpsllq $7,%ymm9,%ymm9+ vpxor %ymm10,%ymm8,%ymm8+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %rbx,%rdi+ vpsrlq $6,%ymm7,%ymm11+ vpxor %ymm9,%ymm8,%ymm8+ xorq %r13,%r14+ leaq (%r11,%rdi,1),%r11+ movq %r8,%r12+ vpsllq $3,%ymm7,%ymm10+ vpaddq %ymm8,%ymm0,%ymm0+ addq 8+256(%rsp),%r10+ andq %rdx,%r12+ rorxq $41,%rdx,%r13+ vpsrlq $19,%ymm7,%ymm9+ vpxor %ymm10,%ymm11,%ymm11+ rorxq $18,%rdx,%rdi+ leaq (%r11,%r14,1),%r11+ leaq (%r10,%r12,1),%r10+ vpsllq $42,%ymm10,%ymm10+ vpxor %ymm9,%ymm11,%ymm11+ andnq %r9,%rdx,%r12+ xorq %rdi,%r13+ rorxq $14,%rdx,%r14+ vpsrlq $42,%ymm9,%ymm9+ vpxor %ymm10,%ymm11,%ymm11+ leaq (%r10,%r12,1),%r10+ xorq %r14,%r13+ movq %r11,%rdi+ vpxor %ymm9,%ymm11,%ymm11+ rorxq $39,%r11,%r12+ leaq (%r10,%r13,1),%r10+ xorq %rax,%rdi+ vpaddq %ymm11,%ymm0,%ymm0+ rorxq $34,%r11,%r14+ rorxq $28,%r11,%r13+ leaq (%rcx,%r10,1),%rcx+ vpaddq -128(%rsi),%ymm0,%ymm10+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %rax,%r15+ xorq %r13,%r14+ leaq (%r10,%r15,1),%r10+ movq %rdx,%r12+ vmovdqa %ymm10,0(%rsp)+ vpalignr $8,%ymm1,%ymm2,%ymm8+ addq 32+256(%rsp),%r9+ andq %rcx,%r12+ rorxq $41,%rcx,%r13+ vpalignr $8,%ymm5,%ymm6,%ymm11+ rorxq $18,%rcx,%r15+ leaq (%r10,%r14,1),%r10+ leaq (%r9,%r12,1),%r9+ vpsrlq $1,%ymm8,%ymm10+ andnq %r8,%rcx,%r12+ xorq %r15,%r13+ rorxq $14,%rcx,%r14+ vpaddq %ymm11,%ymm1,%ymm1+ vpsrlq $7,%ymm8,%ymm11+ leaq (%r9,%r12,1),%r9+ xorq %r14,%r13+ movq %r10,%r15+ vpsllq $56,%ymm8,%ymm9+ vpxor %ymm10,%ymm11,%ymm8+ rorxq $39,%r10,%r12+ leaq (%r9,%r13,1),%r9+ xorq %r11,%r15+ vpsrlq $7,%ymm10,%ymm10+ vpxor %ymm9,%ymm8,%ymm8+ rorxq $34,%r10,%r14+ rorxq $28,%r10,%r13+ leaq (%rbx,%r9,1),%rbx+ vpsllq $7,%ymm9,%ymm9+ vpxor %ymm10,%ymm8,%ymm8+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %r11,%rdi+ vpsrlq $6,%ymm0,%ymm11+ vpxor %ymm9,%ymm8,%ymm8+ xorq %r13,%r14+ leaq (%r9,%rdi,1),%r9+ movq %rcx,%r12+ vpsllq $3,%ymm0,%ymm10+ vpaddq %ymm8,%ymm1,%ymm1+ addq 40+256(%rsp),%r8+ andq %rbx,%r12+ rorxq $41,%rbx,%r13+ vpsrlq $19,%ymm0,%ymm9+ vpxor %ymm10,%ymm11,%ymm11+ rorxq $18,%rbx,%rdi+ leaq (%r9,%r14,1),%r9+ leaq (%r8,%r12,1),%r8+ vpsllq $42,%ymm10,%ymm10+ vpxor %ymm9,%ymm11,%ymm11+ andnq %rdx,%rbx,%r12+ xorq %rdi,%r13+ rorxq $14,%rbx,%r14+ vpsrlq $42,%ymm9,%ymm9+ vpxor %ymm10,%ymm11,%ymm11+ leaq (%r8,%r12,1),%r8+ xorq %r14,%r13+ movq %r9,%rdi+ vpxor %ymm9,%ymm11,%ymm11+ rorxq $39,%r9,%r12+ leaq (%r8,%r13,1),%r8+ xorq %r10,%rdi+ vpaddq %ymm11,%ymm1,%ymm1+ rorxq $34,%r9,%r14+ rorxq $28,%r9,%r13+ leaq (%rax,%r8,1),%rax+ vpaddq -96(%rsi),%ymm1,%ymm10+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %r10,%r15+ xorq %r13,%r14+ leaq (%r8,%r15,1),%r8+ movq %rbx,%r12+ vmovdqa %ymm10,32(%rsp)+ vpalignr $8,%ymm2,%ymm3,%ymm8+ addq 64+256(%rsp),%rdx+ andq %rax,%r12+ rorxq $41,%rax,%r13+ vpalignr $8,%ymm6,%ymm7,%ymm11+ rorxq $18,%rax,%r15+ leaq (%r8,%r14,1),%r8+ leaq (%rdx,%r12,1),%rdx+ vpsrlq $1,%ymm8,%ymm10+ andnq %rcx,%rax,%r12+ xorq %r15,%r13+ rorxq $14,%rax,%r14+ vpaddq %ymm11,%ymm2,%ymm2+ vpsrlq $7,%ymm8,%ymm11+ leaq (%rdx,%r12,1),%rdx+ xorq %r14,%r13+ movq %r8,%r15+ vpsllq $56,%ymm8,%ymm9+ vpxor %ymm10,%ymm11,%ymm8+ rorxq $39,%r8,%r12+ leaq (%rdx,%r13,1),%rdx+ xorq %r9,%r15+ vpsrlq $7,%ymm10,%ymm10+ vpxor %ymm9,%ymm8,%ymm8+ rorxq $34,%r8,%r14+ rorxq $28,%r8,%r13+ leaq (%r11,%rdx,1),%r11+ vpsllq $7,%ymm9,%ymm9+ vpxor %ymm10,%ymm8,%ymm8+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %r9,%rdi+ vpsrlq $6,%ymm1,%ymm11+ vpxor %ymm9,%ymm8,%ymm8+ xorq %r13,%r14+ leaq (%rdx,%rdi,1),%rdx+ movq %rax,%r12+ vpsllq $3,%ymm1,%ymm10+ vpaddq %ymm8,%ymm2,%ymm2+ addq 72+256(%rsp),%rcx+ andq %r11,%r12+ rorxq $41,%r11,%r13+ vpsrlq $19,%ymm1,%ymm9+ vpxor %ymm10,%ymm11,%ymm11+ rorxq $18,%r11,%rdi+ leaq (%rdx,%r14,1),%rdx+ leaq (%rcx,%r12,1),%rcx+ vpsllq $42,%ymm10,%ymm10+ vpxor %ymm9,%ymm11,%ymm11+ andnq %rbx,%r11,%r12+ xorq %rdi,%r13+ rorxq $14,%r11,%r14+ vpsrlq $42,%ymm9,%ymm9+ vpxor %ymm10,%ymm11,%ymm11+ leaq (%rcx,%r12,1),%rcx+ xorq %r14,%r13+ movq %rdx,%rdi+ vpxor %ymm9,%ymm11,%ymm11+ rorxq $39,%rdx,%r12+ leaq (%rcx,%r13,1),%rcx+ xorq %r8,%rdi+ vpaddq %ymm11,%ymm2,%ymm2+ rorxq $34,%rdx,%r14+ rorxq $28,%rdx,%r13+ leaq (%r10,%rcx,1),%r10+ vpaddq -64(%rsi),%ymm2,%ymm10+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %r8,%r15+ xorq %r13,%r14+ leaq (%rcx,%r15,1),%rcx+ movq %r11,%r12+ vmovdqa %ymm10,64(%rsp)+ vpalignr $8,%ymm3,%ymm4,%ymm8+ addq 96+256(%rsp),%rbx+ andq %r10,%r12+ rorxq $41,%r10,%r13+ vpalignr $8,%ymm7,%ymm0,%ymm11+ rorxq $18,%r10,%r15+ leaq (%rcx,%r14,1),%rcx+ leaq (%rbx,%r12,1),%rbx+ vpsrlq $1,%ymm8,%ymm10+ andnq %rax,%r10,%r12+ xorq %r15,%r13+ rorxq $14,%r10,%r14+ vpaddq %ymm11,%ymm3,%ymm3+ vpsrlq $7,%ymm8,%ymm11+ leaq (%rbx,%r12,1),%rbx+ xorq %r14,%r13+ movq %rcx,%r15+ vpsllq $56,%ymm8,%ymm9+ vpxor %ymm10,%ymm11,%ymm8+ rorxq $39,%rcx,%r12+ leaq (%rbx,%r13,1),%rbx+ xorq %rdx,%r15+ vpsrlq $7,%ymm10,%ymm10+ vpxor %ymm9,%ymm8,%ymm8+ rorxq $34,%rcx,%r14+ rorxq $28,%rcx,%r13+ leaq (%r9,%rbx,1),%r9+ vpsllq $7,%ymm9,%ymm9+ vpxor %ymm10,%ymm8,%ymm8+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %rdx,%rdi+ vpsrlq $6,%ymm2,%ymm11+ vpxor %ymm9,%ymm8,%ymm8+ xorq %r13,%r14+ leaq (%rbx,%rdi,1),%rbx+ movq %r10,%r12+ vpsllq $3,%ymm2,%ymm10+ vpaddq %ymm8,%ymm3,%ymm3+ addq 104+256(%rsp),%rax+ andq %r9,%r12+ rorxq $41,%r9,%r13+ vpsrlq $19,%ymm2,%ymm9+ vpxor %ymm10,%ymm11,%ymm11+ rorxq $18,%r9,%rdi+ leaq (%rbx,%r14,1),%rbx+ leaq (%rax,%r12,1),%rax+ vpsllq $42,%ymm10,%ymm10+ vpxor %ymm9,%ymm11,%ymm11+ andnq %r11,%r9,%r12+ xorq %rdi,%r13+ rorxq $14,%r9,%r14+ vpsrlq $42,%ymm9,%ymm9+ vpxor %ymm10,%ymm11,%ymm11+ leaq (%rax,%r12,1),%rax+ xorq %r14,%r13+ movq %rbx,%rdi+ vpxor %ymm9,%ymm11,%ymm11+ rorxq $39,%rbx,%r12+ leaq (%rax,%r13,1),%rax+ xorq %rcx,%rdi+ vpaddq %ymm11,%ymm3,%ymm3+ rorxq $34,%rbx,%r14+ rorxq $28,%rbx,%r13+ leaq (%r8,%rax,1),%r8+ vpaddq -32(%rsi),%ymm3,%ymm10+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %rcx,%r15+ xorq %r13,%r14+ leaq (%rax,%r15,1),%rax+ movq %r9,%r12+ vmovdqa %ymm10,96(%rsp)+ leaq -128(%rsp),%rsp+ vpalignr $8,%ymm4,%ymm5,%ymm8+ addq 0+256(%rsp),%r11+ andq %r8,%r12+ rorxq $41,%r8,%r13+ vpalignr $8,%ymm0,%ymm1,%ymm11+ rorxq $18,%r8,%r15+ leaq (%rax,%r14,1),%rax+ leaq (%r11,%r12,1),%r11+ vpsrlq $1,%ymm8,%ymm10+ andnq %r10,%r8,%r12+ xorq %r15,%r13+ rorxq $14,%r8,%r14+ vpaddq %ymm11,%ymm4,%ymm4+ vpsrlq $7,%ymm8,%ymm11+ leaq (%r11,%r12,1),%r11+ xorq %r14,%r13+ movq %rax,%r15+ vpsllq $56,%ymm8,%ymm9+ vpxor %ymm10,%ymm11,%ymm8+ rorxq $39,%rax,%r12+ leaq (%r11,%r13,1),%r11+ xorq %rbx,%r15+ vpsrlq $7,%ymm10,%ymm10+ vpxor %ymm9,%ymm8,%ymm8+ rorxq $34,%rax,%r14+ rorxq $28,%rax,%r13+ leaq (%rdx,%r11,1),%rdx+ vpsllq $7,%ymm9,%ymm9+ vpxor %ymm10,%ymm8,%ymm8+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %rbx,%rdi+ vpsrlq $6,%ymm3,%ymm11+ vpxor %ymm9,%ymm8,%ymm8+ xorq %r13,%r14+ leaq (%r11,%rdi,1),%r11+ movq %r8,%r12+ vpsllq $3,%ymm3,%ymm10+ vpaddq %ymm8,%ymm4,%ymm4+ addq 8+256(%rsp),%r10+ andq %rdx,%r12+ rorxq $41,%rdx,%r13+ vpsrlq $19,%ymm3,%ymm9+ vpxor %ymm10,%ymm11,%ymm11+ rorxq $18,%rdx,%rdi+ leaq (%r11,%r14,1),%r11+ leaq (%r10,%r12,1),%r10+ vpsllq $42,%ymm10,%ymm10+ vpxor %ymm9,%ymm11,%ymm11+ andnq %r9,%rdx,%r12+ xorq %rdi,%r13+ rorxq $14,%rdx,%r14+ vpsrlq $42,%ymm9,%ymm9+ vpxor %ymm10,%ymm11,%ymm11+ leaq (%r10,%r12,1),%r10+ xorq %r14,%r13+ movq %r11,%rdi+ vpxor %ymm9,%ymm11,%ymm11+ rorxq $39,%r11,%r12+ leaq (%r10,%r13,1),%r10+ xorq %rax,%rdi+ vpaddq %ymm11,%ymm4,%ymm4+ rorxq $34,%r11,%r14+ rorxq $28,%r11,%r13+ leaq (%rcx,%r10,1),%rcx+ vpaddq 0(%rsi),%ymm4,%ymm10+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %rax,%r15+ xorq %r13,%r14+ leaq (%r10,%r15,1),%r10+ movq %rdx,%r12+ vmovdqa %ymm10,0(%rsp)+ vpalignr $8,%ymm5,%ymm6,%ymm8+ addq 32+256(%rsp),%r9+ andq %rcx,%r12+ rorxq $41,%rcx,%r13+ vpalignr $8,%ymm1,%ymm2,%ymm11+ rorxq $18,%rcx,%r15+ leaq (%r10,%r14,1),%r10+ leaq (%r9,%r12,1),%r9+ vpsrlq $1,%ymm8,%ymm10+ andnq %r8,%rcx,%r12+ xorq %r15,%r13+ rorxq $14,%rcx,%r14+ vpaddq %ymm11,%ymm5,%ymm5+ vpsrlq $7,%ymm8,%ymm11+ leaq (%r9,%r12,1),%r9+ xorq %r14,%r13+ movq %r10,%r15+ vpsllq $56,%ymm8,%ymm9+ vpxor %ymm10,%ymm11,%ymm8+ rorxq $39,%r10,%r12+ leaq (%r9,%r13,1),%r9+ xorq %r11,%r15+ vpsrlq $7,%ymm10,%ymm10+ vpxor %ymm9,%ymm8,%ymm8+ rorxq $34,%r10,%r14+ rorxq $28,%r10,%r13+ leaq (%rbx,%r9,1),%rbx+ vpsllq $7,%ymm9,%ymm9+ vpxor %ymm10,%ymm8,%ymm8+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %r11,%rdi+ vpsrlq $6,%ymm4,%ymm11+ vpxor %ymm9,%ymm8,%ymm8+ xorq %r13,%r14+ leaq (%r9,%rdi,1),%r9+ movq %rcx,%r12+ vpsllq $3,%ymm4,%ymm10+ vpaddq %ymm8,%ymm5,%ymm5+ addq 40+256(%rsp),%r8+ andq %rbx,%r12+ rorxq $41,%rbx,%r13+ vpsrlq $19,%ymm4,%ymm9+ vpxor %ymm10,%ymm11,%ymm11+ rorxq $18,%rbx,%rdi+ leaq (%r9,%r14,1),%r9+ leaq (%r8,%r12,1),%r8+ vpsllq $42,%ymm10,%ymm10+ vpxor %ymm9,%ymm11,%ymm11+ andnq %rdx,%rbx,%r12+ xorq %rdi,%r13+ rorxq $14,%rbx,%r14+ vpsrlq $42,%ymm9,%ymm9+ vpxor %ymm10,%ymm11,%ymm11+ leaq (%r8,%r12,1),%r8+ xorq %r14,%r13+ movq %r9,%rdi+ vpxor %ymm9,%ymm11,%ymm11+ rorxq $39,%r9,%r12+ leaq (%r8,%r13,1),%r8+ xorq %r10,%rdi+ vpaddq %ymm11,%ymm5,%ymm5+ rorxq $34,%r9,%r14+ rorxq $28,%r9,%r13+ leaq (%rax,%r8,1),%rax+ vpaddq 32(%rsi),%ymm5,%ymm10+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %r10,%r15+ xorq %r13,%r14+ leaq (%r8,%r15,1),%r8+ movq %rbx,%r12+ vmovdqa %ymm10,32(%rsp)+ vpalignr $8,%ymm6,%ymm7,%ymm8+ addq 64+256(%rsp),%rdx+ andq %rax,%r12+ rorxq $41,%rax,%r13+ vpalignr $8,%ymm2,%ymm3,%ymm11+ rorxq $18,%rax,%r15+ leaq (%r8,%r14,1),%r8+ leaq (%rdx,%r12,1),%rdx+ vpsrlq $1,%ymm8,%ymm10+ andnq %rcx,%rax,%r12+ xorq %r15,%r13+ rorxq $14,%rax,%r14+ vpaddq %ymm11,%ymm6,%ymm6+ vpsrlq $7,%ymm8,%ymm11+ leaq (%rdx,%r12,1),%rdx+ xorq %r14,%r13+ movq %r8,%r15+ vpsllq $56,%ymm8,%ymm9+ vpxor %ymm10,%ymm11,%ymm8+ rorxq $39,%r8,%r12+ leaq (%rdx,%r13,1),%rdx+ xorq %r9,%r15+ vpsrlq $7,%ymm10,%ymm10+ vpxor %ymm9,%ymm8,%ymm8+ rorxq $34,%r8,%r14+ rorxq $28,%r8,%r13+ leaq (%r11,%rdx,1),%r11+ vpsllq $7,%ymm9,%ymm9+ vpxor %ymm10,%ymm8,%ymm8+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %r9,%rdi+ vpsrlq $6,%ymm5,%ymm11+ vpxor %ymm9,%ymm8,%ymm8+ xorq %r13,%r14+ leaq (%rdx,%rdi,1),%rdx+ movq %rax,%r12+ vpsllq $3,%ymm5,%ymm10+ vpaddq %ymm8,%ymm6,%ymm6+ addq 72+256(%rsp),%rcx+ andq %r11,%r12+ rorxq $41,%r11,%r13+ vpsrlq $19,%ymm5,%ymm9+ vpxor %ymm10,%ymm11,%ymm11+ rorxq $18,%r11,%rdi+ leaq (%rdx,%r14,1),%rdx+ leaq (%rcx,%r12,1),%rcx+ vpsllq $42,%ymm10,%ymm10+ vpxor %ymm9,%ymm11,%ymm11+ andnq %rbx,%r11,%r12+ xorq %rdi,%r13+ rorxq $14,%r11,%r14+ vpsrlq $42,%ymm9,%ymm9+ vpxor %ymm10,%ymm11,%ymm11+ leaq (%rcx,%r12,1),%rcx+ xorq %r14,%r13+ movq %rdx,%rdi+ vpxor %ymm9,%ymm11,%ymm11+ rorxq $39,%rdx,%r12+ leaq (%rcx,%r13,1),%rcx+ xorq %r8,%rdi+ vpaddq %ymm11,%ymm6,%ymm6+ rorxq $34,%rdx,%r14+ rorxq $28,%rdx,%r13+ leaq (%r10,%rcx,1),%r10+ vpaddq 64(%rsi),%ymm6,%ymm10+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %r8,%r15+ xorq %r13,%r14+ leaq (%rcx,%r15,1),%rcx+ movq %r11,%r12+ vmovdqa %ymm10,64(%rsp)+ vpalignr $8,%ymm7,%ymm0,%ymm8+ addq 96+256(%rsp),%rbx+ andq %r10,%r12+ rorxq $41,%r10,%r13+ vpalignr $8,%ymm3,%ymm4,%ymm11+ rorxq $18,%r10,%r15+ leaq (%rcx,%r14,1),%rcx+ leaq (%rbx,%r12,1),%rbx+ vpsrlq $1,%ymm8,%ymm10+ andnq %rax,%r10,%r12+ xorq %r15,%r13+ rorxq $14,%r10,%r14+ vpaddq %ymm11,%ymm7,%ymm7+ vpsrlq $7,%ymm8,%ymm11+ leaq (%rbx,%r12,1),%rbx+ xorq %r14,%r13+ movq %rcx,%r15+ vpsllq $56,%ymm8,%ymm9+ vpxor %ymm10,%ymm11,%ymm8+ rorxq $39,%rcx,%r12+ leaq (%rbx,%r13,1),%rbx+ xorq %rdx,%r15+ vpsrlq $7,%ymm10,%ymm10+ vpxor %ymm9,%ymm8,%ymm8+ rorxq $34,%rcx,%r14+ rorxq $28,%rcx,%r13+ leaq (%r9,%rbx,1),%r9+ vpsllq $7,%ymm9,%ymm9+ vpxor %ymm10,%ymm8,%ymm8+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %rdx,%rdi+ vpsrlq $6,%ymm6,%ymm11+ vpxor %ymm9,%ymm8,%ymm8+ xorq %r13,%r14+ leaq (%rbx,%rdi,1),%rbx+ movq %r10,%r12+ vpsllq $3,%ymm6,%ymm10+ vpaddq %ymm8,%ymm7,%ymm7+ addq 104+256(%rsp),%rax+ andq %r9,%r12+ rorxq $41,%r9,%r13+ vpsrlq $19,%ymm6,%ymm9+ vpxor %ymm10,%ymm11,%ymm11+ rorxq $18,%r9,%rdi+ leaq (%rbx,%r14,1),%rbx+ leaq (%rax,%r12,1),%rax+ vpsllq $42,%ymm10,%ymm10+ vpxor %ymm9,%ymm11,%ymm11+ andnq %r11,%r9,%r12+ xorq %rdi,%r13+ rorxq $14,%r9,%r14+ vpsrlq $42,%ymm9,%ymm9+ vpxor %ymm10,%ymm11,%ymm11+ leaq (%rax,%r12,1),%rax+ xorq %r14,%r13+ movq %rbx,%rdi+ vpxor %ymm9,%ymm11,%ymm11+ rorxq $39,%rbx,%r12+ leaq (%rax,%r13,1),%rax+ xorq %rcx,%rdi+ vpaddq %ymm11,%ymm7,%ymm7+ rorxq $34,%rbx,%r14+ rorxq $28,%rbx,%r13+ leaq (%r8,%rax,1),%r8+ vpaddq 96(%rsi),%ymm7,%ymm10+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %rcx,%r15+ xorq %r13,%r14+ leaq (%rax,%r15,1),%rax+ movq %r9,%r12+ vmovdqa %ymm10,96(%rsp)+ leaq 256(%rsi),%rsi+ cmpb $0,-121(%rsi)+ jne L$avx2_00_47+ addq 0+128(%rsp),%r11+ andq %r8,%r12+ rorxq $41,%r8,%r13+ rorxq $18,%r8,%r15+ leaq (%rax,%r14,1),%rax+ leaq (%r11,%r12,1),%r11+ andnq %r10,%r8,%r12+ xorq %r15,%r13+ rorxq $14,%r8,%r14+ leaq (%r11,%r12,1),%r11+ xorq %r14,%r13+ movq %rax,%r15+ rorxq $39,%rax,%r12+ leaq (%r11,%r13,1),%r11+ xorq %rbx,%r15+ rorxq $34,%rax,%r14+ rorxq $28,%rax,%r13+ leaq (%rdx,%r11,1),%rdx+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %rbx,%rdi+ xorq %r13,%r14+ leaq (%r11,%rdi,1),%r11+ movq %r8,%r12+ addq 8+128(%rsp),%r10+ andq %rdx,%r12+ rorxq $41,%rdx,%r13+ rorxq $18,%rdx,%rdi+ leaq (%r11,%r14,1),%r11+ leaq (%r10,%r12,1),%r10+ andnq %r9,%rdx,%r12+ xorq %rdi,%r13+ rorxq $14,%rdx,%r14+ leaq (%r10,%r12,1),%r10+ xorq %r14,%r13+ movq %r11,%rdi+ rorxq $39,%r11,%r12+ leaq (%r10,%r13,1),%r10+ xorq %rax,%rdi+ rorxq $34,%r11,%r14+ rorxq $28,%r11,%r13+ leaq (%rcx,%r10,1),%rcx+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %rax,%r15+ xorq %r13,%r14+ leaq (%r10,%r15,1),%r10+ movq %rdx,%r12+ addq 32+128(%rsp),%r9+ andq %rcx,%r12+ rorxq $41,%rcx,%r13+ rorxq $18,%rcx,%r15+ leaq (%r10,%r14,1),%r10+ leaq (%r9,%r12,1),%r9+ andnq %r8,%rcx,%r12+ xorq %r15,%r13+ rorxq $14,%rcx,%r14+ leaq (%r9,%r12,1),%r9+ xorq %r14,%r13+ movq %r10,%r15+ rorxq $39,%r10,%r12+ leaq (%r9,%r13,1),%r9+ xorq %r11,%r15+ rorxq $34,%r10,%r14+ rorxq $28,%r10,%r13+ leaq (%rbx,%r9,1),%rbx+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %r11,%rdi+ xorq %r13,%r14+ leaq (%r9,%rdi,1),%r9+ movq %rcx,%r12+ addq 40+128(%rsp),%r8+ andq %rbx,%r12+ rorxq $41,%rbx,%r13+ rorxq $18,%rbx,%rdi+ leaq (%r9,%r14,1),%r9+ leaq (%r8,%r12,1),%r8+ andnq %rdx,%rbx,%r12+ xorq %rdi,%r13+ rorxq $14,%rbx,%r14+ leaq (%r8,%r12,1),%r8+ xorq %r14,%r13+ movq %r9,%rdi+ rorxq $39,%r9,%r12+ leaq (%r8,%r13,1),%r8+ xorq %r10,%rdi+ rorxq $34,%r9,%r14+ rorxq $28,%r9,%r13+ leaq (%rax,%r8,1),%rax+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %r10,%r15+ xorq %r13,%r14+ leaq (%r8,%r15,1),%r8+ movq %rbx,%r12+ addq 64+128(%rsp),%rdx+ andq %rax,%r12+ rorxq $41,%rax,%r13+ rorxq $18,%rax,%r15+ leaq (%r8,%r14,1),%r8+ leaq (%rdx,%r12,1),%rdx+ andnq %rcx,%rax,%r12+ xorq %r15,%r13+ rorxq $14,%rax,%r14+ leaq (%rdx,%r12,1),%rdx+ xorq %r14,%r13+ movq %r8,%r15+ rorxq $39,%r8,%r12+ leaq (%rdx,%r13,1),%rdx+ xorq %r9,%r15+ rorxq $34,%r8,%r14+ rorxq $28,%r8,%r13+ leaq (%r11,%rdx,1),%r11+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %r9,%rdi+ xorq %r13,%r14+ leaq (%rdx,%rdi,1),%rdx+ movq %rax,%r12+ addq 72+128(%rsp),%rcx+ andq %r11,%r12+ rorxq $41,%r11,%r13+ rorxq $18,%r11,%rdi+ leaq (%rdx,%r14,1),%rdx+ leaq (%rcx,%r12,1),%rcx+ andnq %rbx,%r11,%r12+ xorq %rdi,%r13+ rorxq $14,%r11,%r14+ leaq (%rcx,%r12,1),%rcx+ xorq %r14,%r13+ movq %rdx,%rdi+ rorxq $39,%rdx,%r12+ leaq (%rcx,%r13,1),%rcx+ xorq %r8,%rdi+ rorxq $34,%rdx,%r14+ rorxq $28,%rdx,%r13+ leaq (%r10,%rcx,1),%r10+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %r8,%r15+ xorq %r13,%r14+ leaq (%rcx,%r15,1),%rcx+ movq %r11,%r12+ addq 96+128(%rsp),%rbx+ andq %r10,%r12+ rorxq $41,%r10,%r13+ rorxq $18,%r10,%r15+ leaq (%rcx,%r14,1),%rcx+ leaq (%rbx,%r12,1),%rbx+ andnq %rax,%r10,%r12+ xorq %r15,%r13+ rorxq $14,%r10,%r14+ leaq (%rbx,%r12,1),%rbx+ xorq %r14,%r13+ movq %rcx,%r15+ rorxq $39,%rcx,%r12+ leaq (%rbx,%r13,1),%rbx+ xorq %rdx,%r15+ rorxq $34,%rcx,%r14+ rorxq $28,%rcx,%r13+ leaq (%r9,%rbx,1),%r9+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %rdx,%rdi+ xorq %r13,%r14+ leaq (%rbx,%rdi,1),%rbx+ movq %r10,%r12+ addq 104+128(%rsp),%rax+ andq %r9,%r12+ rorxq $41,%r9,%r13+ rorxq $18,%r9,%rdi+ leaq (%rbx,%r14,1),%rbx+ leaq (%rax,%r12,1),%rax+ andnq %r11,%r9,%r12+ xorq %rdi,%r13+ rorxq $14,%r9,%r14+ leaq (%rax,%r12,1),%rax+ xorq %r14,%r13+ movq %rbx,%rdi+ rorxq $39,%rbx,%r12+ leaq (%rax,%r13,1),%rax+ xorq %rcx,%rdi+ rorxq $34,%rbx,%r14+ rorxq $28,%rbx,%r13+ leaq (%r8,%rax,1),%r8+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %rcx,%r15+ xorq %r13,%r14+ leaq (%rax,%r15,1),%rax+ movq %r9,%r12+ addq 0(%rsp),%r11+ andq %r8,%r12+ rorxq $41,%r8,%r13+ rorxq $18,%r8,%r15+ leaq (%rax,%r14,1),%rax+ leaq (%r11,%r12,1),%r11+ andnq %r10,%r8,%r12+ xorq %r15,%r13+ rorxq $14,%r8,%r14+ leaq (%r11,%r12,1),%r11+ xorq %r14,%r13+ movq %rax,%r15+ rorxq $39,%rax,%r12+ leaq (%r11,%r13,1),%r11+ xorq %rbx,%r15+ rorxq $34,%rax,%r14+ rorxq $28,%rax,%r13+ leaq (%rdx,%r11,1),%rdx+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %rbx,%rdi+ xorq %r13,%r14+ leaq (%r11,%rdi,1),%r11+ movq %r8,%r12+ addq 8(%rsp),%r10+ andq %rdx,%r12+ rorxq $41,%rdx,%r13+ rorxq $18,%rdx,%rdi+ leaq (%r11,%r14,1),%r11+ leaq (%r10,%r12,1),%r10+ andnq %r9,%rdx,%r12+ xorq %rdi,%r13+ rorxq $14,%rdx,%r14+ leaq (%r10,%r12,1),%r10+ xorq %r14,%r13+ movq %r11,%rdi+ rorxq $39,%r11,%r12+ leaq (%r10,%r13,1),%r10+ xorq %rax,%rdi+ rorxq $34,%r11,%r14+ rorxq $28,%r11,%r13+ leaq (%rcx,%r10,1),%rcx+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %rax,%r15+ xorq %r13,%r14+ leaq (%r10,%r15,1),%r10+ movq %rdx,%r12+ addq 32(%rsp),%r9+ andq %rcx,%r12+ rorxq $41,%rcx,%r13+ rorxq $18,%rcx,%r15+ leaq (%r10,%r14,1),%r10+ leaq (%r9,%r12,1),%r9+ andnq %r8,%rcx,%r12+ xorq %r15,%r13+ rorxq $14,%rcx,%r14+ leaq (%r9,%r12,1),%r9+ xorq %r14,%r13+ movq %r10,%r15+ rorxq $39,%r10,%r12+ leaq (%r9,%r13,1),%r9+ xorq %r11,%r15+ rorxq $34,%r10,%r14+ rorxq $28,%r10,%r13+ leaq (%rbx,%r9,1),%rbx+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %r11,%rdi+ xorq %r13,%r14+ leaq (%r9,%rdi,1),%r9+ movq %rcx,%r12+ addq 40(%rsp),%r8+ andq %rbx,%r12+ rorxq $41,%rbx,%r13+ rorxq $18,%rbx,%rdi+ leaq (%r9,%r14,1),%r9+ leaq (%r8,%r12,1),%r8+ andnq %rdx,%rbx,%r12+ xorq %rdi,%r13+ rorxq $14,%rbx,%r14+ leaq (%r8,%r12,1),%r8+ xorq %r14,%r13+ movq %r9,%rdi+ rorxq $39,%r9,%r12+ leaq (%r8,%r13,1),%r8+ xorq %r10,%rdi+ rorxq $34,%r9,%r14+ rorxq $28,%r9,%r13+ leaq (%rax,%r8,1),%rax+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %r10,%r15+ xorq %r13,%r14+ leaq (%r8,%r15,1),%r8+ movq %rbx,%r12+ addq 64(%rsp),%rdx+ andq %rax,%r12+ rorxq $41,%rax,%r13+ rorxq $18,%rax,%r15+ leaq (%r8,%r14,1),%r8+ leaq (%rdx,%r12,1),%rdx+ andnq %rcx,%rax,%r12+ xorq %r15,%r13+ rorxq $14,%rax,%r14+ leaq (%rdx,%r12,1),%rdx+ xorq %r14,%r13+ movq %r8,%r15+ rorxq $39,%r8,%r12+ leaq (%rdx,%r13,1),%rdx+ xorq %r9,%r15+ rorxq $34,%r8,%r14+ rorxq $28,%r8,%r13+ leaq (%r11,%rdx,1),%r11+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %r9,%rdi+ xorq %r13,%r14+ leaq (%rdx,%rdi,1),%rdx+ movq %rax,%r12+ addq 72(%rsp),%rcx+ andq %r11,%r12+ rorxq $41,%r11,%r13+ rorxq $18,%r11,%rdi+ leaq (%rdx,%r14,1),%rdx+ leaq (%rcx,%r12,1),%rcx+ andnq %rbx,%r11,%r12+ xorq %rdi,%r13+ rorxq $14,%r11,%r14+ leaq (%rcx,%r12,1),%rcx+ xorq %r14,%r13+ movq %rdx,%rdi+ rorxq $39,%rdx,%r12+ leaq (%rcx,%r13,1),%rcx+ xorq %r8,%rdi+ rorxq $34,%rdx,%r14+ rorxq $28,%rdx,%r13+ leaq (%r10,%rcx,1),%r10+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %r8,%r15+ xorq %r13,%r14+ leaq (%rcx,%r15,1),%rcx+ movq %r11,%r12+ addq 96(%rsp),%rbx+ andq %r10,%r12+ rorxq $41,%r10,%r13+ rorxq $18,%r10,%r15+ leaq (%rcx,%r14,1),%rcx+ leaq (%rbx,%r12,1),%rbx+ andnq %rax,%r10,%r12+ xorq %r15,%r13+ rorxq $14,%r10,%r14+ leaq (%rbx,%r12,1),%rbx+ xorq %r14,%r13+ movq %rcx,%r15+ rorxq $39,%rcx,%r12+ leaq (%rbx,%r13,1),%rbx+ xorq %rdx,%r15+ rorxq $34,%rcx,%r14+ rorxq $28,%rcx,%r13+ leaq (%r9,%rbx,1),%r9+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %rdx,%rdi+ xorq %r13,%r14+ leaq (%rbx,%rdi,1),%rbx+ movq %r10,%r12+ addq 104(%rsp),%rax+ andq %r9,%r12+ rorxq $41,%r9,%r13+ rorxq $18,%r9,%rdi+ leaq (%rbx,%r14,1),%rbx+ leaq (%rax,%r12,1),%rax+ andnq %r11,%r9,%r12+ xorq %rdi,%r13+ rorxq $14,%r9,%r14+ leaq (%rax,%r12,1),%rax+ xorq %r14,%r13+ movq %rbx,%rdi+ rorxq $39,%rbx,%r12+ leaq (%rax,%r13,1),%rax+ xorq %rcx,%rdi+ rorxq $34,%rbx,%r14+ rorxq $28,%rbx,%r13+ leaq (%r8,%rax,1),%r8+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %rcx,%r15+ xorq %r13,%r14+ leaq (%rax,%r15,1),%rax+ movq %r9,%r12+ movq -64(%rbp),%rdi+ addq %r14,%rax+ movq -56(%rbp),%r12++ addq 0(%rdi),%rax+ addq 8(%rdi),%rbx+ addq 16(%rdi),%rcx+ addq 24(%rdi),%rdx+ addq 32(%rdi),%r8+ addq 40(%rdi),%r9+ addq 48(%rdi),%r10+ addq 56(%rdi),%r11++ movq %rax,0(%rdi)+ movq %rbx,8(%rdi)+ movq %rcx,16(%rdi)+ movq %rdx,24(%rdi)+ movq %r8,32(%rdi)+ movq %r9,40(%rdi)+ movq %r10,48(%rdi)+ movq %r11,56(%rdi)++ cmpq -48(%rbp),%r12+ je L$done_avx2++ leaq 1152(%rsp),%rsi+ xorq %r14,%r14+ movq %rbx,%rdi+ xorq %rcx,%rdi+ movq %r9,%r12+ jmp L$ower_avx2+.p2align 4+L$ower_avx2:+ addq 0+16(%rsi),%r11+ andq %r8,%r12+ rorxq $41,%r8,%r13+ rorxq $18,%r8,%r15+ leaq (%rax,%r14,1),%rax+ leaq (%r11,%r12,1),%r11+ andnq %r10,%r8,%r12+ xorq %r15,%r13+ rorxq $14,%r8,%r14+ leaq (%r11,%r12,1),%r11+ xorq %r14,%r13+ movq %rax,%r15+ rorxq $39,%rax,%r12+ leaq (%r11,%r13,1),%r11+ xorq %rbx,%r15+ rorxq $34,%rax,%r14+ rorxq $28,%rax,%r13+ leaq (%rdx,%r11,1),%rdx+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %rbx,%rdi+ xorq %r13,%r14+ leaq (%r11,%rdi,1),%r11+ movq %r8,%r12+ addq 8+16(%rsi),%r10+ andq %rdx,%r12+ rorxq $41,%rdx,%r13+ rorxq $18,%rdx,%rdi+ leaq (%r11,%r14,1),%r11+ leaq (%r10,%r12,1),%r10+ andnq %r9,%rdx,%r12+ xorq %rdi,%r13+ rorxq $14,%rdx,%r14+ leaq (%r10,%r12,1),%r10+ xorq %r14,%r13+ movq %r11,%rdi+ rorxq $39,%r11,%r12+ leaq (%r10,%r13,1),%r10+ xorq %rax,%rdi+ rorxq $34,%r11,%r14+ rorxq $28,%r11,%r13+ leaq (%rcx,%r10,1),%rcx+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %rax,%r15+ xorq %r13,%r14+ leaq (%r10,%r15,1),%r10+ movq %rdx,%r12+ addq 32+16(%rsi),%r9+ andq %rcx,%r12+ rorxq $41,%rcx,%r13+ rorxq $18,%rcx,%r15+ leaq (%r10,%r14,1),%r10+ leaq (%r9,%r12,1),%r9+ andnq %r8,%rcx,%r12+ xorq %r15,%r13+ rorxq $14,%rcx,%r14+ leaq (%r9,%r12,1),%r9+ xorq %r14,%r13+ movq %r10,%r15+ rorxq $39,%r10,%r12+ leaq (%r9,%r13,1),%r9+ xorq %r11,%r15+ rorxq $34,%r10,%r14+ rorxq $28,%r10,%r13+ leaq (%rbx,%r9,1),%rbx+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %r11,%rdi+ xorq %r13,%r14+ leaq (%r9,%rdi,1),%r9+ movq %rcx,%r12+ addq 40+16(%rsi),%r8+ andq %rbx,%r12+ rorxq $41,%rbx,%r13+ rorxq $18,%rbx,%rdi+ leaq (%r9,%r14,1),%r9+ leaq (%r8,%r12,1),%r8+ andnq %rdx,%rbx,%r12+ xorq %rdi,%r13+ rorxq $14,%rbx,%r14+ leaq (%r8,%r12,1),%r8+ xorq %r14,%r13+ movq %r9,%rdi+ rorxq $39,%r9,%r12+ leaq (%r8,%r13,1),%r8+ xorq %r10,%rdi+ rorxq $34,%r9,%r14+ rorxq $28,%r9,%r13+ leaq (%rax,%r8,1),%rax+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %r10,%r15+ xorq %r13,%r14+ leaq (%r8,%r15,1),%r8+ movq %rbx,%r12+ addq 64+16(%rsi),%rdx+ andq %rax,%r12+ rorxq $41,%rax,%r13+ rorxq $18,%rax,%r15+ leaq (%r8,%r14,1),%r8+ leaq (%rdx,%r12,1),%rdx+ andnq %rcx,%rax,%r12+ xorq %r15,%r13+ rorxq $14,%rax,%r14+ leaq (%rdx,%r12,1),%rdx+ xorq %r14,%r13+ movq %r8,%r15+ rorxq $39,%r8,%r12+ leaq (%rdx,%r13,1),%rdx+ xorq %r9,%r15+ rorxq $34,%r8,%r14+ rorxq $28,%r8,%r13+ leaq (%r11,%rdx,1),%r11+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %r9,%rdi+ xorq %r13,%r14+ leaq (%rdx,%rdi,1),%rdx+ movq %rax,%r12+ addq 72+16(%rsi),%rcx+ andq %r11,%r12+ rorxq $41,%r11,%r13+ rorxq $18,%r11,%rdi+ leaq (%rdx,%r14,1),%rdx+ leaq (%rcx,%r12,1),%rcx+ andnq %rbx,%r11,%r12+ xorq %rdi,%r13+ rorxq $14,%r11,%r14+ leaq (%rcx,%r12,1),%rcx+ xorq %r14,%r13+ movq %rdx,%rdi+ rorxq $39,%rdx,%r12+ leaq (%rcx,%r13,1),%rcx+ xorq %r8,%rdi+ rorxq $34,%rdx,%r14+ rorxq $28,%rdx,%r13+ leaq (%r10,%rcx,1),%r10+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %r8,%r15+ xorq %r13,%r14+ leaq (%rcx,%r15,1),%rcx+ movq %r11,%r12+ addq 96+16(%rsi),%rbx+ andq %r10,%r12+ rorxq $41,%r10,%r13+ rorxq $18,%r10,%r15+ leaq (%rcx,%r14,1),%rcx+ leaq (%rbx,%r12,1),%rbx+ andnq %rax,%r10,%r12+ xorq %r15,%r13+ rorxq $14,%r10,%r14+ leaq (%rbx,%r12,1),%rbx+ xorq %r14,%r13+ movq %rcx,%r15+ rorxq $39,%rcx,%r12+ leaq (%rbx,%r13,1),%rbx+ xorq %rdx,%r15+ rorxq $34,%rcx,%r14+ rorxq $28,%rcx,%r13+ leaq (%r9,%rbx,1),%r9+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %rdx,%rdi+ xorq %r13,%r14+ leaq (%rbx,%rdi,1),%rbx+ movq %r10,%r12+ addq 104+16(%rsi),%rax+ andq %r9,%r12+ rorxq $41,%r9,%r13+ rorxq $18,%r9,%rdi+ leaq (%rbx,%r14,1),%rbx+ leaq (%rax,%r12,1),%rax+ andnq %r11,%r9,%r12+ xorq %rdi,%r13+ rorxq $14,%r9,%r14+ leaq (%rax,%r12,1),%rax+ xorq %r14,%r13+ movq %rbx,%rdi+ rorxq $39,%rbx,%r12+ leaq (%rax,%r13,1),%rax+ xorq %rcx,%rdi+ rorxq $34,%rbx,%r14+ rorxq $28,%rbx,%r13+ leaq (%r8,%rax,1),%r8+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %rcx,%r15+ xorq %r13,%r14+ leaq (%rax,%r15,1),%rax+ movq %r9,%r12+ leaq -128(%rsi),%rsi+ cmpq %rsp,%rsi+ jae L$ower_avx2++ movq -64(%rbp),%rdi+ addq %r14,%rax+ movq -56(%rbp),%rsi+ leaq 1152(%rsp),%rsp++ addq 0(%rdi),%rax+ addq 8(%rdi),%rbx+ addq 16(%rdi),%rcx+ addq 24(%rdi),%rdx+ addq 32(%rdi),%r8+ addq 40(%rdi),%r9+ leaq 256(%rsi),%rsi+ addq 48(%rdi),%r10+ movq %rsi,%r12+ addq 56(%rdi),%r11+ cmpq -48(%rbp),%rsi++ movq %rax,0(%rdi)+ cmoveq %rsp,%r12+ movq %rbx,8(%rdi)+ movq %rcx,16(%rdi)+ movq %rdx,24(%rdi)+ movq %r8,32(%rdi)+ movq %r9,40(%rdi)+ movq %r10,48(%rdi)+ movq %r11,56(%rdi)++ jbe L$oop_avx2++L$done_avx2:+ vzeroupper+ movq -40(%rbp),%r15+ movq -32(%rbp),%r14+ movq -24(%rbp),%r13+ movq -16(%rbp),%r12+ movq -8(%rbp),%rbx+ movq %rbp,%rsp+.cfi_def_cfa_register %rsp+ popq %rbp+.cfi_adjust_cfa_offset -8+.cfi_restore %rbp+.cfi_restore %r12+.cfi_restore %r13+.cfi_restore %r14+.cfi_restore %r15+.cfi_restore %rbx+ .byte 0xf3,0xc3+.cfi_endproc +
@@ -0,0 +1,6016 @@+.text +++.globl crypton_sha512_asm_block_data_order+.def crypton_sha512_asm_block_data_order; .scl 2; .type 32; .endef+.p2align 4+crypton_sha512_asm_block_data_order:+ .byte 0xf3,0x0f,0x1e,0xfa+ movq %rdi,8(%rsp)+ movq %rsi,16(%rsp)+ movq %rsp,%r11+.LSEH_begin_crypton_sha512_asm_block_data_order:+++ pushq %rbp++ movq %rsp,%rbp++ movq %rcx,%rdi+ movq %rdx,%rsi+ movq %r8,%rdx+ leaq crypton_ia32cap_P(%rip),%rax+ movl 0(%rax),%r9d+ movl 4(%rax),%r10d+ movl 8(%rax),%eax+ testl $2048,%r10d+ jnz .Lxop_shortcut+ andl $296,%eax+ cmpl $296,%eax+ je .Lavx2_shortcut+ andl $1073741824,%r9d+ andl $268435968,%r10d+ orl %r9d,%r10d+ cmpl $1342177792,%r10d+ je .Lavx_shortcut+ pushq %rbx++ pushq %r12++ pushq %r13++ pushq %r14++ pushq %r15++ shlq $4,%rdx+ subq $128+24,%rsp+++.LSEH_body_crypton_sha512_asm_block_data_order:++ leaq (%rsi,%rdx,8),%rdx+ movq %rdi,128+0(%rsp)+ movq %rsi,128+8(%rsp)+ movq %rdx,128+16(%rsp)++ movq 0(%rdi),%rax+ movq 8(%rdi),%rbx+ movq 16(%rdi),%rcx+ movq 24(%rdi),%rdx+ movq 32(%rdi),%r8+ movq 40(%rdi),%r9+ movq 48(%rdi),%r10+ movq 56(%rdi),%r11+ jmp .Lloop++.p2align 4+.Lloop:+ movq %rbx,%rdi+ leaq K512(%rip),%rbp+ xorq %rcx,%rdi+ movq 0(%rsi),%r12+ movq %r8,%r13+ movq %rax,%r14+ bswapq %r12+ rorq $23,%r13+ movq %r9,%r15++ xorq %r8,%r13+ rorq $5,%r14+ xorq %r10,%r15++ movq %r12,0(%rsp)+ xorq %rax,%r14+ andq %r8,%r15++ rorq $4,%r13+ addq %r11,%r12+ xorq %r10,%r15++ rorq $6,%r14+ xorq %r8,%r13+ addq %r15,%r12++ movq %rax,%r15+ addq (%rbp),%r12+ xorq %rax,%r14++ xorq %rbx,%r15+ rorq $14,%r13+ movq %rbx,%r11++ andq %r15,%rdi+ rorq $28,%r14+ addq %r13,%r12++ xorq %rdi,%r11+ addq %r12,%rdx+ addq %r12,%r11++ leaq 8(%rbp),%rbp+ addq %r14,%r11+ movq 8(%rsi),%r12+ movq %rdx,%r13+ movq %r11,%r14+ bswapq %r12+ rorq $23,%r13+ movq %r8,%rdi++ xorq %rdx,%r13+ rorq $5,%r14+ xorq %r9,%rdi++ movq %r12,8(%rsp)+ xorq %r11,%r14+ andq %rdx,%rdi++ rorq $4,%r13+ addq %r10,%r12+ xorq %r9,%rdi++ rorq $6,%r14+ xorq %rdx,%r13+ addq %rdi,%r12++ movq %r11,%rdi+ addq (%rbp),%r12+ xorq %r11,%r14++ xorq %rax,%rdi+ rorq $14,%r13+ movq %rax,%r10++ andq %rdi,%r15+ rorq $28,%r14+ addq %r13,%r12++ xorq %r15,%r10+ addq %r12,%rcx+ addq %r12,%r10++ leaq 24(%rbp),%rbp+ addq %r14,%r10+ movq 16(%rsi),%r12+ movq %rcx,%r13+ movq %r10,%r14+ bswapq %r12+ rorq $23,%r13+ movq %rdx,%r15++ xorq %rcx,%r13+ rorq $5,%r14+ xorq %r8,%r15++ movq %r12,16(%rsp)+ xorq %r10,%r14+ andq %rcx,%r15++ rorq $4,%r13+ addq %r9,%r12+ xorq %r8,%r15++ rorq $6,%r14+ xorq %rcx,%r13+ addq %r15,%r12++ movq %r10,%r15+ addq (%rbp),%r12+ xorq %r10,%r14++ xorq %r11,%r15+ rorq $14,%r13+ movq %r11,%r9++ andq %r15,%rdi+ rorq $28,%r14+ addq %r13,%r12++ xorq %rdi,%r9+ addq %r12,%rbx+ addq %r12,%r9++ leaq 8(%rbp),%rbp+ addq %r14,%r9+ movq 24(%rsi),%r12+ movq %rbx,%r13+ movq %r9,%r14+ bswapq %r12+ rorq $23,%r13+ movq %rcx,%rdi++ xorq %rbx,%r13+ rorq $5,%r14+ xorq %rdx,%rdi++ movq %r12,24(%rsp)+ xorq %r9,%r14+ andq %rbx,%rdi++ rorq $4,%r13+ addq %r8,%r12+ xorq %rdx,%rdi++ rorq $6,%r14+ xorq %rbx,%r13+ addq %rdi,%r12++ movq %r9,%rdi+ addq (%rbp),%r12+ xorq %r9,%r14++ xorq %r10,%rdi+ rorq $14,%r13+ movq %r10,%r8++ andq %rdi,%r15+ rorq $28,%r14+ addq %r13,%r12++ xorq %r15,%r8+ addq %r12,%rax+ addq %r12,%r8++ leaq 24(%rbp),%rbp+ addq %r14,%r8+ movq 32(%rsi),%r12+ movq %rax,%r13+ movq %r8,%r14+ bswapq %r12+ rorq $23,%r13+ movq %rbx,%r15++ xorq %rax,%r13+ rorq $5,%r14+ xorq %rcx,%r15++ movq %r12,32(%rsp)+ xorq %r8,%r14+ andq %rax,%r15++ rorq $4,%r13+ addq %rdx,%r12+ xorq %rcx,%r15++ rorq $6,%r14+ xorq %rax,%r13+ addq %r15,%r12++ movq %r8,%r15+ addq (%rbp),%r12+ xorq %r8,%r14++ xorq %r9,%r15+ rorq $14,%r13+ movq %r9,%rdx++ andq %r15,%rdi+ rorq $28,%r14+ addq %r13,%r12++ xorq %rdi,%rdx+ addq %r12,%r11+ addq %r12,%rdx++ leaq 8(%rbp),%rbp+ addq %r14,%rdx+ movq 40(%rsi),%r12+ movq %r11,%r13+ movq %rdx,%r14+ bswapq %r12+ rorq $23,%r13+ movq %rax,%rdi++ xorq %r11,%r13+ rorq $5,%r14+ xorq %rbx,%rdi++ movq %r12,40(%rsp)+ xorq %rdx,%r14+ andq %r11,%rdi++ rorq $4,%r13+ addq %rcx,%r12+ xorq %rbx,%rdi++ rorq $6,%r14+ xorq %r11,%r13+ addq %rdi,%r12++ movq %rdx,%rdi+ addq (%rbp),%r12+ xorq %rdx,%r14++ xorq %r8,%rdi+ rorq $14,%r13+ movq %r8,%rcx++ andq %rdi,%r15+ rorq $28,%r14+ addq %r13,%r12++ xorq %r15,%rcx+ addq %r12,%r10+ addq %r12,%rcx++ leaq 24(%rbp),%rbp+ addq %r14,%rcx+ movq 48(%rsi),%r12+ movq %r10,%r13+ movq %rcx,%r14+ bswapq %r12+ rorq $23,%r13+ movq %r11,%r15++ xorq %r10,%r13+ rorq $5,%r14+ xorq %rax,%r15++ movq %r12,48(%rsp)+ xorq %rcx,%r14+ andq %r10,%r15++ rorq $4,%r13+ addq %rbx,%r12+ xorq %rax,%r15++ rorq $6,%r14+ xorq %r10,%r13+ addq %r15,%r12++ movq %rcx,%r15+ addq (%rbp),%r12+ xorq %rcx,%r14++ xorq %rdx,%r15+ rorq $14,%r13+ movq %rdx,%rbx++ andq %r15,%rdi+ rorq $28,%r14+ addq %r13,%r12++ xorq %rdi,%rbx+ addq %r12,%r9+ addq %r12,%rbx++ leaq 8(%rbp),%rbp+ addq %r14,%rbx+ movq 56(%rsi),%r12+ movq %r9,%r13+ movq %rbx,%r14+ bswapq %r12+ rorq $23,%r13+ movq %r10,%rdi++ xorq %r9,%r13+ rorq $5,%r14+ xorq %r11,%rdi++ movq %r12,56(%rsp)+ xorq %rbx,%r14+ andq %r9,%rdi++ rorq $4,%r13+ addq %rax,%r12+ xorq %r11,%rdi++ rorq $6,%r14+ xorq %r9,%r13+ addq %rdi,%r12++ movq %rbx,%rdi+ addq (%rbp),%r12+ xorq %rbx,%r14++ xorq %rcx,%rdi+ rorq $14,%r13+ movq %rcx,%rax++ andq %rdi,%r15+ rorq $28,%r14+ addq %r13,%r12++ xorq %r15,%rax+ addq %r12,%r8+ addq %r12,%rax++ leaq 24(%rbp),%rbp+ addq %r14,%rax+ movq 64(%rsi),%r12+ movq %r8,%r13+ movq %rax,%r14+ bswapq %r12+ rorq $23,%r13+ movq %r9,%r15++ xorq %r8,%r13+ rorq $5,%r14+ xorq %r10,%r15++ movq %r12,64(%rsp)+ xorq %rax,%r14+ andq %r8,%r15++ rorq $4,%r13+ addq %r11,%r12+ xorq %r10,%r15++ rorq $6,%r14+ xorq %r8,%r13+ addq %r15,%r12++ movq %rax,%r15+ addq (%rbp),%r12+ xorq %rax,%r14++ xorq %rbx,%r15+ rorq $14,%r13+ movq %rbx,%r11++ andq %r15,%rdi+ rorq $28,%r14+ addq %r13,%r12++ xorq %rdi,%r11+ addq %r12,%rdx+ addq %r12,%r11++ leaq 8(%rbp),%rbp+ addq %r14,%r11+ movq 72(%rsi),%r12+ movq %rdx,%r13+ movq %r11,%r14+ bswapq %r12+ rorq $23,%r13+ movq %r8,%rdi++ xorq %rdx,%r13+ rorq $5,%r14+ xorq %r9,%rdi++ movq %r12,72(%rsp)+ xorq %r11,%r14+ andq %rdx,%rdi++ rorq $4,%r13+ addq %r10,%r12+ xorq %r9,%rdi++ rorq $6,%r14+ xorq %rdx,%r13+ addq %rdi,%r12++ movq %r11,%rdi+ addq (%rbp),%r12+ xorq %r11,%r14++ xorq %rax,%rdi+ rorq $14,%r13+ movq %rax,%r10++ andq %rdi,%r15+ rorq $28,%r14+ addq %r13,%r12++ xorq %r15,%r10+ addq %r12,%rcx+ addq %r12,%r10++ leaq 24(%rbp),%rbp+ addq %r14,%r10+ movq 80(%rsi),%r12+ movq %rcx,%r13+ movq %r10,%r14+ bswapq %r12+ rorq $23,%r13+ movq %rdx,%r15++ xorq %rcx,%r13+ rorq $5,%r14+ xorq %r8,%r15++ movq %r12,80(%rsp)+ xorq %r10,%r14+ andq %rcx,%r15++ rorq $4,%r13+ addq %r9,%r12+ xorq %r8,%r15++ rorq $6,%r14+ xorq %rcx,%r13+ addq %r15,%r12++ movq %r10,%r15+ addq (%rbp),%r12+ xorq %r10,%r14++ xorq %r11,%r15+ rorq $14,%r13+ movq %r11,%r9++ andq %r15,%rdi+ rorq $28,%r14+ addq %r13,%r12++ xorq %rdi,%r9+ addq %r12,%rbx+ addq %r12,%r9++ leaq 8(%rbp),%rbp+ addq %r14,%r9+ movq 88(%rsi),%r12+ movq %rbx,%r13+ movq %r9,%r14+ bswapq %r12+ rorq $23,%r13+ movq %rcx,%rdi++ xorq %rbx,%r13+ rorq $5,%r14+ xorq %rdx,%rdi++ movq %r12,88(%rsp)+ xorq %r9,%r14+ andq %rbx,%rdi++ rorq $4,%r13+ addq %r8,%r12+ xorq %rdx,%rdi++ rorq $6,%r14+ xorq %rbx,%r13+ addq %rdi,%r12++ movq %r9,%rdi+ addq (%rbp),%r12+ xorq %r9,%r14++ xorq %r10,%rdi+ rorq $14,%r13+ movq %r10,%r8++ andq %rdi,%r15+ rorq $28,%r14+ addq %r13,%r12++ xorq %r15,%r8+ addq %r12,%rax+ addq %r12,%r8++ leaq 24(%rbp),%rbp+ addq %r14,%r8+ movq 96(%rsi),%r12+ movq %rax,%r13+ movq %r8,%r14+ bswapq %r12+ rorq $23,%r13+ movq %rbx,%r15++ xorq %rax,%r13+ rorq $5,%r14+ xorq %rcx,%r15++ movq %r12,96(%rsp)+ xorq %r8,%r14+ andq %rax,%r15++ rorq $4,%r13+ addq %rdx,%r12+ xorq %rcx,%r15++ rorq $6,%r14+ xorq %rax,%r13+ addq %r15,%r12++ movq %r8,%r15+ addq (%rbp),%r12+ xorq %r8,%r14++ xorq %r9,%r15+ rorq $14,%r13+ movq %r9,%rdx++ andq %r15,%rdi+ rorq $28,%r14+ addq %r13,%r12++ xorq %rdi,%rdx+ addq %r12,%r11+ addq %r12,%rdx++ leaq 8(%rbp),%rbp+ addq %r14,%rdx+ movq 104(%rsi),%r12+ movq %r11,%r13+ movq %rdx,%r14+ bswapq %r12+ rorq $23,%r13+ movq %rax,%rdi++ xorq %r11,%r13+ rorq $5,%r14+ xorq %rbx,%rdi++ movq %r12,104(%rsp)+ xorq %rdx,%r14+ andq %r11,%rdi++ rorq $4,%r13+ addq %rcx,%r12+ xorq %rbx,%rdi++ rorq $6,%r14+ xorq %r11,%r13+ addq %rdi,%r12++ movq %rdx,%rdi+ addq (%rbp),%r12+ xorq %rdx,%r14++ xorq %r8,%rdi+ rorq $14,%r13+ movq %r8,%rcx++ andq %rdi,%r15+ rorq $28,%r14+ addq %r13,%r12++ xorq %r15,%rcx+ addq %r12,%r10+ addq %r12,%rcx++ leaq 24(%rbp),%rbp+ addq %r14,%rcx+ movq 112(%rsi),%r12+ movq %r10,%r13+ movq %rcx,%r14+ bswapq %r12+ rorq $23,%r13+ movq %r11,%r15++ xorq %r10,%r13+ rorq $5,%r14+ xorq %rax,%r15++ movq %r12,112(%rsp)+ xorq %rcx,%r14+ andq %r10,%r15++ rorq $4,%r13+ addq %rbx,%r12+ xorq %rax,%r15++ rorq $6,%r14+ xorq %r10,%r13+ addq %r15,%r12++ movq %rcx,%r15+ addq (%rbp),%r12+ xorq %rcx,%r14++ xorq %rdx,%r15+ rorq $14,%r13+ movq %rdx,%rbx++ andq %r15,%rdi+ rorq $28,%r14+ addq %r13,%r12++ xorq %rdi,%rbx+ addq %r12,%r9+ addq %r12,%rbx++ leaq 8(%rbp),%rbp+ addq %r14,%rbx+ movq 120(%rsi),%r12+ movq %r9,%r13+ movq %rbx,%r14+ bswapq %r12+ rorq $23,%r13+ movq %r10,%rdi++ xorq %r9,%r13+ rorq $5,%r14+ xorq %r11,%rdi++ movq %r12,120(%rsp)+ xorq %rbx,%r14+ andq %r9,%rdi++ rorq $4,%r13+ addq %rax,%r12+ xorq %r11,%rdi++ rorq $6,%r14+ xorq %r9,%r13+ addq %rdi,%r12++ movq %rbx,%rdi+ addq (%rbp),%r12+ xorq %rbx,%r14++ xorq %rcx,%rdi+ rorq $14,%r13+ movq %rcx,%rax++ andq %rdi,%r15+ rorq $28,%r14+ addq %r13,%r12++ xorq %r15,%rax+ addq %r12,%r8+ addq %r12,%rax++ leaq 24(%rbp),%rbp+ jmp .Lrounds_16_xx+.p2align 4+.Lrounds_16_xx:+ movq 8(%rsp),%r13+ movq 112(%rsp),%r15++ movq %r13,%r12+ rorq $7,%r13+ addq %r14,%rax+ movq %r15,%r14+ rorq $42,%r15++ xorq %r12,%r13+ shrq $7,%r12+ rorq $1,%r13+ xorq %r14,%r15+ shrq $6,%r14++ rorq $19,%r15+ xorq %r13,%r12+ xorq %r14,%r15+ addq 72(%rsp),%r12++ addq 0(%rsp),%r12+ movq %r8,%r13+ addq %r15,%r12+ movq %rax,%r14+ rorq $23,%r13+ movq %r9,%r15++ xorq %r8,%r13+ rorq $5,%r14+ xorq %r10,%r15++ movq %r12,0(%rsp)+ xorq %rax,%r14+ andq %r8,%r15++ rorq $4,%r13+ addq %r11,%r12+ xorq %r10,%r15++ rorq $6,%r14+ xorq %r8,%r13+ addq %r15,%r12++ movq %rax,%r15+ addq (%rbp),%r12+ xorq %rax,%r14++ xorq %rbx,%r15+ rorq $14,%r13+ movq %rbx,%r11++ andq %r15,%rdi+ rorq $28,%r14+ addq %r13,%r12++ xorq %rdi,%r11+ addq %r12,%rdx+ addq %r12,%r11++ leaq 8(%rbp),%rbp+ movq 16(%rsp),%r13+ movq 120(%rsp),%rdi++ movq %r13,%r12+ rorq $7,%r13+ addq %r14,%r11+ movq %rdi,%r14+ rorq $42,%rdi++ xorq %r12,%r13+ shrq $7,%r12+ rorq $1,%r13+ xorq %r14,%rdi+ shrq $6,%r14++ rorq $19,%rdi+ xorq %r13,%r12+ xorq %r14,%rdi+ addq 80(%rsp),%r12++ addq 8(%rsp),%r12+ movq %rdx,%r13+ addq %rdi,%r12+ movq %r11,%r14+ rorq $23,%r13+ movq %r8,%rdi++ xorq %rdx,%r13+ rorq $5,%r14+ xorq %r9,%rdi++ movq %r12,8(%rsp)+ xorq %r11,%r14+ andq %rdx,%rdi++ rorq $4,%r13+ addq %r10,%r12+ xorq %r9,%rdi++ rorq $6,%r14+ xorq %rdx,%r13+ addq %rdi,%r12++ movq %r11,%rdi+ addq (%rbp),%r12+ xorq %r11,%r14++ xorq %rax,%rdi+ rorq $14,%r13+ movq %rax,%r10++ andq %rdi,%r15+ rorq $28,%r14+ addq %r13,%r12++ xorq %r15,%r10+ addq %r12,%rcx+ addq %r12,%r10++ leaq 24(%rbp),%rbp+ movq 24(%rsp),%r13+ movq 0(%rsp),%r15++ movq %r13,%r12+ rorq $7,%r13+ addq %r14,%r10+ movq %r15,%r14+ rorq $42,%r15++ xorq %r12,%r13+ shrq $7,%r12+ rorq $1,%r13+ xorq %r14,%r15+ shrq $6,%r14++ rorq $19,%r15+ xorq %r13,%r12+ xorq %r14,%r15+ addq 88(%rsp),%r12++ addq 16(%rsp),%r12+ movq %rcx,%r13+ addq %r15,%r12+ movq %r10,%r14+ rorq $23,%r13+ movq %rdx,%r15++ xorq %rcx,%r13+ rorq $5,%r14+ xorq %r8,%r15++ movq %r12,16(%rsp)+ xorq %r10,%r14+ andq %rcx,%r15++ rorq $4,%r13+ addq %r9,%r12+ xorq %r8,%r15++ rorq $6,%r14+ xorq %rcx,%r13+ addq %r15,%r12++ movq %r10,%r15+ addq (%rbp),%r12+ xorq %r10,%r14++ xorq %r11,%r15+ rorq $14,%r13+ movq %r11,%r9++ andq %r15,%rdi+ rorq $28,%r14+ addq %r13,%r12++ xorq %rdi,%r9+ addq %r12,%rbx+ addq %r12,%r9++ leaq 8(%rbp),%rbp+ movq 32(%rsp),%r13+ movq 8(%rsp),%rdi++ movq %r13,%r12+ rorq $7,%r13+ addq %r14,%r9+ movq %rdi,%r14+ rorq $42,%rdi++ xorq %r12,%r13+ shrq $7,%r12+ rorq $1,%r13+ xorq %r14,%rdi+ shrq $6,%r14++ rorq $19,%rdi+ xorq %r13,%r12+ xorq %r14,%rdi+ addq 96(%rsp),%r12++ addq 24(%rsp),%r12+ movq %rbx,%r13+ addq %rdi,%r12+ movq %r9,%r14+ rorq $23,%r13+ movq %rcx,%rdi++ xorq %rbx,%r13+ rorq $5,%r14+ xorq %rdx,%rdi++ movq %r12,24(%rsp)+ xorq %r9,%r14+ andq %rbx,%rdi++ rorq $4,%r13+ addq %r8,%r12+ xorq %rdx,%rdi++ rorq $6,%r14+ xorq %rbx,%r13+ addq %rdi,%r12++ movq %r9,%rdi+ addq (%rbp),%r12+ xorq %r9,%r14++ xorq %r10,%rdi+ rorq $14,%r13+ movq %r10,%r8++ andq %rdi,%r15+ rorq $28,%r14+ addq %r13,%r12++ xorq %r15,%r8+ addq %r12,%rax+ addq %r12,%r8++ leaq 24(%rbp),%rbp+ movq 40(%rsp),%r13+ movq 16(%rsp),%r15++ movq %r13,%r12+ rorq $7,%r13+ addq %r14,%r8+ movq %r15,%r14+ rorq $42,%r15++ xorq %r12,%r13+ shrq $7,%r12+ rorq $1,%r13+ xorq %r14,%r15+ shrq $6,%r14++ rorq $19,%r15+ xorq %r13,%r12+ xorq %r14,%r15+ addq 104(%rsp),%r12++ addq 32(%rsp),%r12+ movq %rax,%r13+ addq %r15,%r12+ movq %r8,%r14+ rorq $23,%r13+ movq %rbx,%r15++ xorq %rax,%r13+ rorq $5,%r14+ xorq %rcx,%r15++ movq %r12,32(%rsp)+ xorq %r8,%r14+ andq %rax,%r15++ rorq $4,%r13+ addq %rdx,%r12+ xorq %rcx,%r15++ rorq $6,%r14+ xorq %rax,%r13+ addq %r15,%r12++ movq %r8,%r15+ addq (%rbp),%r12+ xorq %r8,%r14++ xorq %r9,%r15+ rorq $14,%r13+ movq %r9,%rdx++ andq %r15,%rdi+ rorq $28,%r14+ addq %r13,%r12++ xorq %rdi,%rdx+ addq %r12,%r11+ addq %r12,%rdx++ leaq 8(%rbp),%rbp+ movq 48(%rsp),%r13+ movq 24(%rsp),%rdi++ movq %r13,%r12+ rorq $7,%r13+ addq %r14,%rdx+ movq %rdi,%r14+ rorq $42,%rdi++ xorq %r12,%r13+ shrq $7,%r12+ rorq $1,%r13+ xorq %r14,%rdi+ shrq $6,%r14++ rorq $19,%rdi+ xorq %r13,%r12+ xorq %r14,%rdi+ addq 112(%rsp),%r12++ addq 40(%rsp),%r12+ movq %r11,%r13+ addq %rdi,%r12+ movq %rdx,%r14+ rorq $23,%r13+ movq %rax,%rdi++ xorq %r11,%r13+ rorq $5,%r14+ xorq %rbx,%rdi++ movq %r12,40(%rsp)+ xorq %rdx,%r14+ andq %r11,%rdi++ rorq $4,%r13+ addq %rcx,%r12+ xorq %rbx,%rdi++ rorq $6,%r14+ xorq %r11,%r13+ addq %rdi,%r12++ movq %rdx,%rdi+ addq (%rbp),%r12+ xorq %rdx,%r14++ xorq %r8,%rdi+ rorq $14,%r13+ movq %r8,%rcx++ andq %rdi,%r15+ rorq $28,%r14+ addq %r13,%r12++ xorq %r15,%rcx+ addq %r12,%r10+ addq %r12,%rcx++ leaq 24(%rbp),%rbp+ movq 56(%rsp),%r13+ movq 32(%rsp),%r15++ movq %r13,%r12+ rorq $7,%r13+ addq %r14,%rcx+ movq %r15,%r14+ rorq $42,%r15++ xorq %r12,%r13+ shrq $7,%r12+ rorq $1,%r13+ xorq %r14,%r15+ shrq $6,%r14++ rorq $19,%r15+ xorq %r13,%r12+ xorq %r14,%r15+ addq 120(%rsp),%r12++ addq 48(%rsp),%r12+ movq %r10,%r13+ addq %r15,%r12+ movq %rcx,%r14+ rorq $23,%r13+ movq %r11,%r15++ xorq %r10,%r13+ rorq $5,%r14+ xorq %rax,%r15++ movq %r12,48(%rsp)+ xorq %rcx,%r14+ andq %r10,%r15++ rorq $4,%r13+ addq %rbx,%r12+ xorq %rax,%r15++ rorq $6,%r14+ xorq %r10,%r13+ addq %r15,%r12++ movq %rcx,%r15+ addq (%rbp),%r12+ xorq %rcx,%r14++ xorq %rdx,%r15+ rorq $14,%r13+ movq %rdx,%rbx++ andq %r15,%rdi+ rorq $28,%r14+ addq %r13,%r12++ xorq %rdi,%rbx+ addq %r12,%r9+ addq %r12,%rbx++ leaq 8(%rbp),%rbp+ movq 64(%rsp),%r13+ movq 40(%rsp),%rdi++ movq %r13,%r12+ rorq $7,%r13+ addq %r14,%rbx+ movq %rdi,%r14+ rorq $42,%rdi++ xorq %r12,%r13+ shrq $7,%r12+ rorq $1,%r13+ xorq %r14,%rdi+ shrq $6,%r14++ rorq $19,%rdi+ xorq %r13,%r12+ xorq %r14,%rdi+ addq 0(%rsp),%r12++ addq 56(%rsp),%r12+ movq %r9,%r13+ addq %rdi,%r12+ movq %rbx,%r14+ rorq $23,%r13+ movq %r10,%rdi++ xorq %r9,%r13+ rorq $5,%r14+ xorq %r11,%rdi++ movq %r12,56(%rsp)+ xorq %rbx,%r14+ andq %r9,%rdi++ rorq $4,%r13+ addq %rax,%r12+ xorq %r11,%rdi++ rorq $6,%r14+ xorq %r9,%r13+ addq %rdi,%r12++ movq %rbx,%rdi+ addq (%rbp),%r12+ xorq %rbx,%r14++ xorq %rcx,%rdi+ rorq $14,%r13+ movq %rcx,%rax++ andq %rdi,%r15+ rorq $28,%r14+ addq %r13,%r12++ xorq %r15,%rax+ addq %r12,%r8+ addq %r12,%rax++ leaq 24(%rbp),%rbp+ movq 72(%rsp),%r13+ movq 48(%rsp),%r15++ movq %r13,%r12+ rorq $7,%r13+ addq %r14,%rax+ movq %r15,%r14+ rorq $42,%r15++ xorq %r12,%r13+ shrq $7,%r12+ rorq $1,%r13+ xorq %r14,%r15+ shrq $6,%r14++ rorq $19,%r15+ xorq %r13,%r12+ xorq %r14,%r15+ addq 8(%rsp),%r12++ addq 64(%rsp),%r12+ movq %r8,%r13+ addq %r15,%r12+ movq %rax,%r14+ rorq $23,%r13+ movq %r9,%r15++ xorq %r8,%r13+ rorq $5,%r14+ xorq %r10,%r15++ movq %r12,64(%rsp)+ xorq %rax,%r14+ andq %r8,%r15++ rorq $4,%r13+ addq %r11,%r12+ xorq %r10,%r15++ rorq $6,%r14+ xorq %r8,%r13+ addq %r15,%r12++ movq %rax,%r15+ addq (%rbp),%r12+ xorq %rax,%r14++ xorq %rbx,%r15+ rorq $14,%r13+ movq %rbx,%r11++ andq %r15,%rdi+ rorq $28,%r14+ addq %r13,%r12++ xorq %rdi,%r11+ addq %r12,%rdx+ addq %r12,%r11++ leaq 8(%rbp),%rbp+ movq 80(%rsp),%r13+ movq 56(%rsp),%rdi++ movq %r13,%r12+ rorq $7,%r13+ addq %r14,%r11+ movq %rdi,%r14+ rorq $42,%rdi++ xorq %r12,%r13+ shrq $7,%r12+ rorq $1,%r13+ xorq %r14,%rdi+ shrq $6,%r14++ rorq $19,%rdi+ xorq %r13,%r12+ xorq %r14,%rdi+ addq 16(%rsp),%r12++ addq 72(%rsp),%r12+ movq %rdx,%r13+ addq %rdi,%r12+ movq %r11,%r14+ rorq $23,%r13+ movq %r8,%rdi++ xorq %rdx,%r13+ rorq $5,%r14+ xorq %r9,%rdi++ movq %r12,72(%rsp)+ xorq %r11,%r14+ andq %rdx,%rdi++ rorq $4,%r13+ addq %r10,%r12+ xorq %r9,%rdi++ rorq $6,%r14+ xorq %rdx,%r13+ addq %rdi,%r12++ movq %r11,%rdi+ addq (%rbp),%r12+ xorq %r11,%r14++ xorq %rax,%rdi+ rorq $14,%r13+ movq %rax,%r10++ andq %rdi,%r15+ rorq $28,%r14+ addq %r13,%r12++ xorq %r15,%r10+ addq %r12,%rcx+ addq %r12,%r10++ leaq 24(%rbp),%rbp+ movq 88(%rsp),%r13+ movq 64(%rsp),%r15++ movq %r13,%r12+ rorq $7,%r13+ addq %r14,%r10+ movq %r15,%r14+ rorq $42,%r15++ xorq %r12,%r13+ shrq $7,%r12+ rorq $1,%r13+ xorq %r14,%r15+ shrq $6,%r14++ rorq $19,%r15+ xorq %r13,%r12+ xorq %r14,%r15+ addq 24(%rsp),%r12++ addq 80(%rsp),%r12+ movq %rcx,%r13+ addq %r15,%r12+ movq %r10,%r14+ rorq $23,%r13+ movq %rdx,%r15++ xorq %rcx,%r13+ rorq $5,%r14+ xorq %r8,%r15++ movq %r12,80(%rsp)+ xorq %r10,%r14+ andq %rcx,%r15++ rorq $4,%r13+ addq %r9,%r12+ xorq %r8,%r15++ rorq $6,%r14+ xorq %rcx,%r13+ addq %r15,%r12++ movq %r10,%r15+ addq (%rbp),%r12+ xorq %r10,%r14++ xorq %r11,%r15+ rorq $14,%r13+ movq %r11,%r9++ andq %r15,%rdi+ rorq $28,%r14+ addq %r13,%r12++ xorq %rdi,%r9+ addq %r12,%rbx+ addq %r12,%r9++ leaq 8(%rbp),%rbp+ movq 96(%rsp),%r13+ movq 72(%rsp),%rdi++ movq %r13,%r12+ rorq $7,%r13+ addq %r14,%r9+ movq %rdi,%r14+ rorq $42,%rdi++ xorq %r12,%r13+ shrq $7,%r12+ rorq $1,%r13+ xorq %r14,%rdi+ shrq $6,%r14++ rorq $19,%rdi+ xorq %r13,%r12+ xorq %r14,%rdi+ addq 32(%rsp),%r12++ addq 88(%rsp),%r12+ movq %rbx,%r13+ addq %rdi,%r12+ movq %r9,%r14+ rorq $23,%r13+ movq %rcx,%rdi++ xorq %rbx,%r13+ rorq $5,%r14+ xorq %rdx,%rdi++ movq %r12,88(%rsp)+ xorq %r9,%r14+ andq %rbx,%rdi++ rorq $4,%r13+ addq %r8,%r12+ xorq %rdx,%rdi++ rorq $6,%r14+ xorq %rbx,%r13+ addq %rdi,%r12++ movq %r9,%rdi+ addq (%rbp),%r12+ xorq %r9,%r14++ xorq %r10,%rdi+ rorq $14,%r13+ movq %r10,%r8++ andq %rdi,%r15+ rorq $28,%r14+ addq %r13,%r12++ xorq %r15,%r8+ addq %r12,%rax+ addq %r12,%r8++ leaq 24(%rbp),%rbp+ movq 104(%rsp),%r13+ movq 80(%rsp),%r15++ movq %r13,%r12+ rorq $7,%r13+ addq %r14,%r8+ movq %r15,%r14+ rorq $42,%r15++ xorq %r12,%r13+ shrq $7,%r12+ rorq $1,%r13+ xorq %r14,%r15+ shrq $6,%r14++ rorq $19,%r15+ xorq %r13,%r12+ xorq %r14,%r15+ addq 40(%rsp),%r12++ addq 96(%rsp),%r12+ movq %rax,%r13+ addq %r15,%r12+ movq %r8,%r14+ rorq $23,%r13+ movq %rbx,%r15++ xorq %rax,%r13+ rorq $5,%r14+ xorq %rcx,%r15++ movq %r12,96(%rsp)+ xorq %r8,%r14+ andq %rax,%r15++ rorq $4,%r13+ addq %rdx,%r12+ xorq %rcx,%r15++ rorq $6,%r14+ xorq %rax,%r13+ addq %r15,%r12++ movq %r8,%r15+ addq (%rbp),%r12+ xorq %r8,%r14++ xorq %r9,%r15+ rorq $14,%r13+ movq %r9,%rdx++ andq %r15,%rdi+ rorq $28,%r14+ addq %r13,%r12++ xorq %rdi,%rdx+ addq %r12,%r11+ addq %r12,%rdx++ leaq 8(%rbp),%rbp+ movq 112(%rsp),%r13+ movq 88(%rsp),%rdi++ movq %r13,%r12+ rorq $7,%r13+ addq %r14,%rdx+ movq %rdi,%r14+ rorq $42,%rdi++ xorq %r12,%r13+ shrq $7,%r12+ rorq $1,%r13+ xorq %r14,%rdi+ shrq $6,%r14++ rorq $19,%rdi+ xorq %r13,%r12+ xorq %r14,%rdi+ addq 48(%rsp),%r12++ addq 104(%rsp),%r12+ movq %r11,%r13+ addq %rdi,%r12+ movq %rdx,%r14+ rorq $23,%r13+ movq %rax,%rdi++ xorq %r11,%r13+ rorq $5,%r14+ xorq %rbx,%rdi++ movq %r12,104(%rsp)+ xorq %rdx,%r14+ andq %r11,%rdi++ rorq $4,%r13+ addq %rcx,%r12+ xorq %rbx,%rdi++ rorq $6,%r14+ xorq %r11,%r13+ addq %rdi,%r12++ movq %rdx,%rdi+ addq (%rbp),%r12+ xorq %rdx,%r14++ xorq %r8,%rdi+ rorq $14,%r13+ movq %r8,%rcx++ andq %rdi,%r15+ rorq $28,%r14+ addq %r13,%r12++ xorq %r15,%rcx+ addq %r12,%r10+ addq %r12,%rcx++ leaq 24(%rbp),%rbp+ movq 120(%rsp),%r13+ movq 96(%rsp),%r15++ movq %r13,%r12+ rorq $7,%r13+ addq %r14,%rcx+ movq %r15,%r14+ rorq $42,%r15++ xorq %r12,%r13+ shrq $7,%r12+ rorq $1,%r13+ xorq %r14,%r15+ shrq $6,%r14++ rorq $19,%r15+ xorq %r13,%r12+ xorq %r14,%r15+ addq 56(%rsp),%r12++ addq 112(%rsp),%r12+ movq %r10,%r13+ addq %r15,%r12+ movq %rcx,%r14+ rorq $23,%r13+ movq %r11,%r15++ xorq %r10,%r13+ rorq $5,%r14+ xorq %rax,%r15++ movq %r12,112(%rsp)+ xorq %rcx,%r14+ andq %r10,%r15++ rorq $4,%r13+ addq %rbx,%r12+ xorq %rax,%r15++ rorq $6,%r14+ xorq %r10,%r13+ addq %r15,%r12++ movq %rcx,%r15+ addq (%rbp),%r12+ xorq %rcx,%r14++ xorq %rdx,%r15+ rorq $14,%r13+ movq %rdx,%rbx++ andq %r15,%rdi+ rorq $28,%r14+ addq %r13,%r12++ xorq %rdi,%rbx+ addq %r12,%r9+ addq %r12,%rbx++ leaq 8(%rbp),%rbp+ movq 0(%rsp),%r13+ movq 104(%rsp),%rdi++ movq %r13,%r12+ rorq $7,%r13+ addq %r14,%rbx+ movq %rdi,%r14+ rorq $42,%rdi++ xorq %r12,%r13+ shrq $7,%r12+ rorq $1,%r13+ xorq %r14,%rdi+ shrq $6,%r14++ rorq $19,%rdi+ xorq %r13,%r12+ xorq %r14,%rdi+ addq 64(%rsp),%r12++ addq 120(%rsp),%r12+ movq %r9,%r13+ addq %rdi,%r12+ movq %rbx,%r14+ rorq $23,%r13+ movq %r10,%rdi++ xorq %r9,%r13+ rorq $5,%r14+ xorq %r11,%rdi++ movq %r12,120(%rsp)+ xorq %rbx,%r14+ andq %r9,%rdi++ rorq $4,%r13+ addq %rax,%r12+ xorq %r11,%rdi++ rorq $6,%r14+ xorq %r9,%r13+ addq %rdi,%r12++ movq %rbx,%rdi+ addq (%rbp),%r12+ xorq %rbx,%r14++ xorq %rcx,%rdi+ rorq $14,%r13+ movq %rcx,%rax++ andq %rdi,%r15+ rorq $28,%r14+ addq %r13,%r12++ xorq %r15,%rax+ addq %r12,%r8+ addq %r12,%rax++ leaq 24(%rbp),%rbp+ cmpb $0,7(%rbp)+ jnz .Lrounds_16_xx++ movq 128+0(%rsp),%rdi+ addq %r14,%rax+ leaq 128(%rsi),%rsi++ addq 0(%rdi),%rax+ addq 8(%rdi),%rbx+ addq 16(%rdi),%rcx+ addq 24(%rdi),%rdx+ addq 32(%rdi),%r8+ addq 40(%rdi),%r9+ addq 48(%rdi),%r10+ addq 56(%rdi),%r11++ cmpq 128+16(%rsp),%rsi++ movq %rax,0(%rdi)+ movq %rbx,8(%rdi)+ movq %rcx,16(%rdi)+ movq %rdx,24(%rdi)+ movq %r8,32(%rdi)+ movq %r9,40(%rdi)+ movq %r10,48(%rdi)+ movq %r11,56(%rdi)+ jb .Lloop++ leaq 128+24+48(%rsp),%r11++ movq 128+24(%rsp),%r15+ movq -40(%r11),%r14+ movq -32(%r11),%r13+ movq -24(%r11),%r12+ movq -16(%r11),%rbx+ movq -8(%r11),%rbp+.LSEH_epilogue_crypton_sha512_asm_block_data_order:+ mov 8(%r11),%rdi+ mov 16(%r11),%rsi++ leaq (%r11),%rsp+ .byte 0xf3,0xc3++.LSEH_end_crypton_sha512_asm_block_data_order:+.p2align 6++K512:+.quad 0x428a2f98d728ae22,0x7137449123ef65cd+.quad 0x428a2f98d728ae22,0x7137449123ef65cd+.quad 0xb5c0fbcfec4d3b2f,0xe9b5dba58189dbbc+.quad 0xb5c0fbcfec4d3b2f,0xe9b5dba58189dbbc+.quad 0x3956c25bf348b538,0x59f111f1b605d019+.quad 0x3956c25bf348b538,0x59f111f1b605d019+.quad 0x923f82a4af194f9b,0xab1c5ed5da6d8118+.quad 0x923f82a4af194f9b,0xab1c5ed5da6d8118+.quad 0xd807aa98a3030242,0x12835b0145706fbe+.quad 0xd807aa98a3030242,0x12835b0145706fbe+.quad 0x243185be4ee4b28c,0x550c7dc3d5ffb4e2+.quad 0x243185be4ee4b28c,0x550c7dc3d5ffb4e2+.quad 0x72be5d74f27b896f,0x80deb1fe3b1696b1+.quad 0x72be5d74f27b896f,0x80deb1fe3b1696b1+.quad 0x9bdc06a725c71235,0xc19bf174cf692694+.quad 0x9bdc06a725c71235,0xc19bf174cf692694+.quad 0xe49b69c19ef14ad2,0xefbe4786384f25e3+.quad 0xe49b69c19ef14ad2,0xefbe4786384f25e3+.quad 0x0fc19dc68b8cd5b5,0x240ca1cc77ac9c65+.quad 0x0fc19dc68b8cd5b5,0x240ca1cc77ac9c65+.quad 0x2de92c6f592b0275,0x4a7484aa6ea6e483+.quad 0x2de92c6f592b0275,0x4a7484aa6ea6e483+.quad 0x5cb0a9dcbd41fbd4,0x76f988da831153b5+.quad 0x5cb0a9dcbd41fbd4,0x76f988da831153b5+.quad 0x983e5152ee66dfab,0xa831c66d2db43210+.quad 0x983e5152ee66dfab,0xa831c66d2db43210+.quad 0xb00327c898fb213f,0xbf597fc7beef0ee4+.quad 0xb00327c898fb213f,0xbf597fc7beef0ee4+.quad 0xc6e00bf33da88fc2,0xd5a79147930aa725+.quad 0xc6e00bf33da88fc2,0xd5a79147930aa725+.quad 0x06ca6351e003826f,0x142929670a0e6e70+.quad 0x06ca6351e003826f,0x142929670a0e6e70+.quad 0x27b70a8546d22ffc,0x2e1b21385c26c926+.quad 0x27b70a8546d22ffc,0x2e1b21385c26c926+.quad 0x4d2c6dfc5ac42aed,0x53380d139d95b3df+.quad 0x4d2c6dfc5ac42aed,0x53380d139d95b3df+.quad 0x650a73548baf63de,0x766a0abb3c77b2a8+.quad 0x650a73548baf63de,0x766a0abb3c77b2a8+.quad 0x81c2c92e47edaee6,0x92722c851482353b+.quad 0x81c2c92e47edaee6,0x92722c851482353b+.quad 0xa2bfe8a14cf10364,0xa81a664bbc423001+.quad 0xa2bfe8a14cf10364,0xa81a664bbc423001+.quad 0xc24b8b70d0f89791,0xc76c51a30654be30+.quad 0xc24b8b70d0f89791,0xc76c51a30654be30+.quad 0xd192e819d6ef5218,0xd69906245565a910+.quad 0xd192e819d6ef5218,0xd69906245565a910+.quad 0xf40e35855771202a,0x106aa07032bbd1b8+.quad 0xf40e35855771202a,0x106aa07032bbd1b8+.quad 0x19a4c116b8d2d0c8,0x1e376c085141ab53+.quad 0x19a4c116b8d2d0c8,0x1e376c085141ab53+.quad 0x2748774cdf8eeb99,0x34b0bcb5e19b48a8+.quad 0x2748774cdf8eeb99,0x34b0bcb5e19b48a8+.quad 0x391c0cb3c5c95a63,0x4ed8aa4ae3418acb+.quad 0x391c0cb3c5c95a63,0x4ed8aa4ae3418acb+.quad 0x5b9cca4f7763e373,0x682e6ff3d6b2b8a3+.quad 0x5b9cca4f7763e373,0x682e6ff3d6b2b8a3+.quad 0x748f82ee5defb2fc,0x78a5636f43172f60+.quad 0x748f82ee5defb2fc,0x78a5636f43172f60+.quad 0x84c87814a1f0ab72,0x8cc702081a6439ec+.quad 0x84c87814a1f0ab72,0x8cc702081a6439ec+.quad 0x90befffa23631e28,0xa4506cebde82bde9+.quad 0x90befffa23631e28,0xa4506cebde82bde9+.quad 0xbef9a3f7b2c67915,0xc67178f2e372532b+.quad 0xbef9a3f7b2c67915,0xc67178f2e372532b+.quad 0xca273eceea26619c,0xd186b8c721c0c207+.quad 0xca273eceea26619c,0xd186b8c721c0c207+.quad 0xeada7dd6cde0eb1e,0xf57d4f7fee6ed178+.quad 0xeada7dd6cde0eb1e,0xf57d4f7fee6ed178+.quad 0x06f067aa72176fba,0x0a637dc5a2c898a6+.quad 0x06f067aa72176fba,0x0a637dc5a2c898a6+.quad 0x113f9804bef90dae,0x1b710b35131c471b+.quad 0x113f9804bef90dae,0x1b710b35131c471b+.quad 0x28db77f523047d84,0x32caab7b40c72493+.quad 0x28db77f523047d84,0x32caab7b40c72493+.quad 0x3c9ebe0a15c9bebc,0x431d67c49c100d4c+.quad 0x3c9ebe0a15c9bebc,0x431d67c49c100d4c+.quad 0x4cc5d4becb3e42b6,0x597f299cfc657e2a+.quad 0x4cc5d4becb3e42b6,0x597f299cfc657e2a+.quad 0x5fcb6fab3ad6faec,0x6c44198c4a475817+.quad 0x5fcb6fab3ad6faec,0x6c44198c4a475817++.quad 0x0001020304050607,0x08090a0b0c0d0e0f+.quad 0x0001020304050607,0x08090a0b0c0d0e0f++K512_nodup:+.quad 0x428a2f98d728ae22,0x7137449123ef65cd+.quad 0xb5c0fbcfec4d3b2f,0xe9b5dba58189dbbc+.quad 0x3956c25bf348b538,0x59f111f1b605d019+.quad 0x923f82a4af194f9b,0xab1c5ed5da6d8118+.quad 0xd807aa98a3030242,0x12835b0145706fbe+.quad 0x243185be4ee4b28c,0x550c7dc3d5ffb4e2+.quad 0x72be5d74f27b896f,0x80deb1fe3b1696b1+.quad 0x9bdc06a725c71235,0xc19bf174cf692694+.quad 0xe49b69c19ef14ad2,0xefbe4786384f25e3+.quad 0x0fc19dc68b8cd5b5,0x240ca1cc77ac9c65+.quad 0x2de92c6f592b0275,0x4a7484aa6ea6e483+.quad 0x5cb0a9dcbd41fbd4,0x76f988da831153b5+.quad 0x983e5152ee66dfab,0xa831c66d2db43210+.quad 0xb00327c898fb213f,0xbf597fc7beef0ee4+.quad 0xc6e00bf33da88fc2,0xd5a79147930aa725+.quad 0x06ca6351e003826f,0x142929670a0e6e70+.quad 0x27b70a8546d22ffc,0x2e1b21385c26c926+.quad 0x4d2c6dfc5ac42aed,0x53380d139d95b3df+.quad 0x650a73548baf63de,0x766a0abb3c77b2a8+.quad 0x81c2c92e47edaee6,0x92722c851482353b+.quad 0xa2bfe8a14cf10364,0xa81a664bbc423001+.quad 0xc24b8b70d0f89791,0xc76c51a30654be30+.quad 0xd192e819d6ef5218,0xd69906245565a910+.quad 0xf40e35855771202a,0x106aa07032bbd1b8+.quad 0x19a4c116b8d2d0c8,0x1e376c085141ab53+.quad 0x2748774cdf8eeb99,0x34b0bcb5e19b48a8+.quad 0x391c0cb3c5c95a63,0x4ed8aa4ae3418acb+.quad 0x5b9cca4f7763e373,0x682e6ff3d6b2b8a3+.quad 0x748f82ee5defb2fc,0x78a5636f43172f60+.quad 0x84c87814a1f0ab72,0x8cc702081a6439ec+.quad 0x90befffa23631e28,0xa4506cebde82bde9+.quad 0xbef9a3f7b2c67915,0xc67178f2e372532b+.quad 0xca273eceea26619c,0xd186b8c721c0c207+.quad 0xeada7dd6cde0eb1e,0xf57d4f7fee6ed178+.quad 0x06f067aa72176fba,0x0a637dc5a2c898a6+.quad 0x113f9804bef90dae,0x1b710b35131c471b+.quad 0x28db77f523047d84,0x32caab7b40c72493+.quad 0x3c9ebe0a15c9bebc,0x431d67c49c100d4c+.quad 0x4cc5d4becb3e42b6,0x597f299cfc657e2a+.quad 0x5fcb6fab3ad6faec,0x6c44198c4a475817++.byte 83,72,65,53,49,50,32,98,108,111,99,107,32,116,114,97,110,115,102,111,114,109,32,102,111,114,32,120,56,54,95,54,52,44,32,67,82,89,80,84,79,71,65,77,83,32,98,121,32,64,100,111,116,45,97,115,109,0+.globl crypton_sha512_asm_block_data_order_shaext+.def crypton_sha512_asm_block_data_order_shaext; .scl 2; .type 32; .endef+.p2align 6+crypton_sha512_asm_block_data_order_shaext:+ .byte 0xf3,0x0f,0x1e,0xfa+ movq %rdi,8(%rsp)+ movq %rsi,16(%rsp)+ movq %rsp,%r11+.LSEH_begin_crypton_sha512_asm_block_data_order_shaext:+++ pushq %rbp++ movq %rsp,%rbp++ movq %rcx,%rdi+ movq %rdx,%rsi+ movq %r8,%rdx+.Lshaext_shortcut:+ subq $0x50,%rsp++ movaps %xmm6,-80(%rbp)+ movaps %xmm7,-64(%rbp)+ movaps %xmm8,-48(%rbp)+ movaps %xmm9,-32(%rbp)+ movaps %xmm10,-16(%rbp)++.LSEH_body_crypton_sha512_asm_block_data_order_shaext:++ leaq K512_nodup+128(%rip),%rcx+ vmovdqu (%rdi),%ymm0+ vmovdqu 32(%rdi),%ymm1+ vmovdqa -160(%rcx),%ymm8++ vpermq $27,%ymm0,%ymm0+ vpblendd $15,%ymm1,%ymm0,%ymm5+ vpblendd $15,%ymm0,%ymm1,%ymm6+ vpermq $225,%ymm5,%ymm5+ vpermq $75,%ymm6,%ymm6+ jmp .Loop_shaext++.p2align 4+.Loop_shaext:+ vmovdqu (%rsi),%ymm0+ vmovdqu 32(%rsi),%ymm1+ vmovdqu 64(%rsi),%ymm2+ vpshufb %ymm8,%ymm0,%ymm0+ vmovdqu 96(%rsi),%ymm3++ vpaddq 0-128(%rcx),%ymm0,%ymm4+ vpshufb %ymm8,%ymm1,%ymm1+ vmovdqa %ymm6,%ymm10+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+ vmovdqa %ymm5,%ymm9+.byte 196,226,79,203,236++ vpaddq 32-128(%rcx),%ymm1,%ymm4+ vpshufb %ymm8,%ymm2,%ymm2+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+ leaq 128(%rsi),%rsi+.byte 196,226,127,204,193+.byte 196,226,79,203,236++ vpaddq 64-128(%rcx),%ymm2,%ymm4+ vpshufb %ymm8,%ymm3,%ymm3+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+ vpblendd $0x03,%ymm3,%ymm2,%ymm7+ vpermq $0x39,%ymm7,%ymm7+ vpaddq %ymm7,%ymm0,%ymm0+.byte 196,226,127,204,202+.byte 196,226,79,203,236++ vpaddq 96-128(%rcx),%ymm3,%ymm4+.byte 196,226,127,205,195+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+ vpblendd $0x03,%ymm0,%ymm3,%ymm7+ vpermq $0x39,%ymm7,%ymm7+ vpaddq %ymm7,%ymm1,%ymm1+.byte 196,226,127,204,211+.byte 196,226,79,203,236+ vpaddq 128-128(%rcx),%ymm0,%ymm4+.byte 196,226,127,205,200+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+ vpblendd $0x03,%ymm1,%ymm0,%ymm7+ vpermq $0x39,%ymm7,%ymm7+ vpaddq %ymm7,%ymm2,%ymm2+.byte 196,226,127,204,216+.byte 196,226,79,203,236+ vpaddq 160-128(%rcx),%ymm1,%ymm4+.byte 196,226,127,205,209+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+ vpblendd $0x03,%ymm2,%ymm1,%ymm7+ vpermq $0x39,%ymm7,%ymm7+ vpaddq %ymm7,%ymm3,%ymm3+.byte 196,226,127,204,193+.byte 196,226,79,203,236+ vpaddq 192-128(%rcx),%ymm2,%ymm4+.byte 196,226,127,205,218+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+ vpblendd $0x03,%ymm3,%ymm2,%ymm7+ vpermq $0x39,%ymm7,%ymm7+ vpaddq %ymm7,%ymm0,%ymm0+.byte 196,226,127,204,202+.byte 196,226,79,203,236+ vpaddq 224-128(%rcx),%ymm3,%ymm4+.byte 196,226,127,205,195+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+ vpblendd $0x03,%ymm0,%ymm3,%ymm7+ vpermq $0x39,%ymm7,%ymm7+ vpaddq %ymm7,%ymm1,%ymm1+.byte 196,226,127,204,211+.byte 196,226,79,203,236+ vpaddq 256-128(%rcx),%ymm0,%ymm4+.byte 196,226,127,205,200+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+ vpblendd $0x03,%ymm1,%ymm0,%ymm7+ vpermq $0x39,%ymm7,%ymm7+ vpaddq %ymm7,%ymm2,%ymm2+.byte 196,226,127,204,216+.byte 196,226,79,203,236+ vpaddq 288-128(%rcx),%ymm1,%ymm4+.byte 196,226,127,205,209+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+ vpblendd $0x03,%ymm2,%ymm1,%ymm7+ vpermq $0x39,%ymm7,%ymm7+ vpaddq %ymm7,%ymm3,%ymm3+.byte 196,226,127,204,193+.byte 196,226,79,203,236+ vpaddq 320-128(%rcx),%ymm2,%ymm4+.byte 196,226,127,205,218+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+ vpblendd $0x03,%ymm3,%ymm2,%ymm7+ vpermq $0x39,%ymm7,%ymm7+ vpaddq %ymm7,%ymm0,%ymm0+.byte 196,226,127,204,202+.byte 196,226,79,203,236+ vpaddq 352-128(%rcx),%ymm3,%ymm4+.byte 196,226,127,205,195+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+ vpblendd $0x03,%ymm0,%ymm3,%ymm7+ vpermq $0x39,%ymm7,%ymm7+ vpaddq %ymm7,%ymm1,%ymm1+.byte 196,226,127,204,211+.byte 196,226,79,203,236+ vpaddq 384-128(%rcx),%ymm0,%ymm4+.byte 196,226,127,205,200+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+ vpblendd $0x03,%ymm1,%ymm0,%ymm7+ vpermq $0x39,%ymm7,%ymm7+ vpaddq %ymm7,%ymm2,%ymm2+.byte 196,226,127,204,216+.byte 196,226,79,203,236+ vpaddq 416-128(%rcx),%ymm1,%ymm4+.byte 196,226,127,205,209+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+ vpblendd $0x03,%ymm2,%ymm1,%ymm7+ vpermq $0x39,%ymm7,%ymm7+ vpaddq %ymm7,%ymm3,%ymm3+.byte 196,226,127,204,193+.byte 196,226,79,203,236+ vpaddq 448-128(%rcx),%ymm2,%ymm4+.byte 196,226,127,205,218+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+ vpblendd $0x03,%ymm3,%ymm2,%ymm7+ vpermq $0x39,%ymm7,%ymm7+ vpaddq %ymm7,%ymm0,%ymm0+.byte 196,226,127,204,202+.byte 196,226,79,203,236+ vpaddq 480-128(%rcx),%ymm3,%ymm4+.byte 196,226,127,205,195+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+ vpblendd $0x03,%ymm0,%ymm3,%ymm7+ vpermq $0x39,%ymm7,%ymm7+ vpaddq %ymm7,%ymm1,%ymm1+.byte 196,226,127,204,211+.byte 196,226,79,203,236+ vpaddq 512-128(%rcx),%ymm0,%ymm4+.byte 196,226,127,205,200+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+ vpblendd $0x03,%ymm1,%ymm0,%ymm7+ vpermq $0x39,%ymm7,%ymm7+ vpaddq %ymm7,%ymm2,%ymm2+.byte 196,226,127,204,216+.byte 196,226,79,203,236+ vpaddq 544-128(%rcx),%ymm1,%ymm4+.byte 196,226,127,205,209+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+ vpblendd $0x03,%ymm2,%ymm1,%ymm7+ vpermq $0x39,%ymm7,%ymm7+.byte 196,226,79,203,236+ vpaddq %ymm7,%ymm3,%ymm3++ vpaddq 576-128(%rcx),%ymm2,%ymm4+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+.byte 196,226,127,205,218+.byte 196,226,79,203,236++ vpaddq 608-128(%rcx),%ymm3,%ymm4+.byte 196,226,87,203,244+ vextracti128 $1,%ymm4,%xmm4+ decq %rdx+.byte 196,226,79,203,236++ vpaddq %ymm10,%ymm6,%ymm6+ vpaddq %ymm9,%ymm5,%ymm5+ jnz .Loop_shaext++ vpermq $75,%ymm5,%ymm5+ vpblendd $240,%ymm6,%ymm5,%ymm1+ vpblendd $240,%ymm5,%ymm6,%ymm2+ vpermq $180,%ymm1,%ymm1+ vpermq $27,%ymm2,%ymm2++ vmovdqu %ymm1,(%rdi)+ vmovdqu %ymm2,32(%rdi)++ vzeroupper+ movaps -80(%rbp),%xmm6+ movaps -64(%rbp),%xmm7+ movaps -48(%rbp),%xmm8+ movaps -32(%rbp),%xmm9+ movaps -16(%rbp),%xmm10+ movq %rbp,%rsp++ popq %rbp++.LSEH_epilogue_crypton_sha512_asm_block_data_order_shaext:+ mov 8(%rsp),%rdi+ mov 16(%rsp),%rsi++ .byte 0xf3,0xc3++.LSEH_end_crypton_sha512_asm_block_data_order_shaext:+.def crypton_sha512_asm_block_data_order_xop; .scl 3; .type 32; .endef+.p2align 6+crypton_sha512_asm_block_data_order_xop:+ .byte 0xf3,0x0f,0x1e,0xfa+ movq %rdi,8(%rsp)+ movq %rsi,16(%rsp)+ movq %rsp,%r11+.LSEH_begin_crypton_sha512_asm_block_data_order_xop:+++ pushq %rbp++ movq %rsp,%rbp++ movq %rcx,%rdi+ movq %rdx,%rsi+ movq %r8,%rdx+.Lxop_shortcut:+ pushq %rbx++ pushq %r12++ pushq %r13++ pushq %r14++ pushq %r15++ shlq $4,%rdx+ subq $120,%rsp++ leaq (%rsi,%rdx,8),%rdx+ movq %rdi,-64(%rbp)++ movq %rdx,-48(%rbp)+ movaps %xmm6,-160(%rbp)+ movaps %xmm7,-144(%rbp)+ movaps %xmm8,-128(%rbp)+ movaps %xmm9,-112(%rbp)++ movaps %xmm10,-96(%rbp)+ movaps %xmm11,-80(%rbp)++.LSEH_body_crypton_sha512_asm_block_data_order_xop:+++ leaq -128(%rsp),%rsp+ vzeroupper+ andq $-64,%rsp+ movq 0(%rdi),%rax+ movq 8(%rdi),%rbx+ movq 16(%rdi),%rcx+ movq 24(%rdi),%rdx+ movq 32(%rdi),%r8+ movq 40(%rdi),%r9+ movq 48(%rdi),%r10+ movq 56(%rdi),%r11+ jmp .Lloop_xop+.p2align 4+.Lloop_xop:+ vmovdqa K512+1280(%rip),%xmm11+ movq %rsi,-56(%rbp)+ vmovdqu 0(%rsi),%xmm0+ vmovdqu 16(%rsi),%xmm1+ vmovdqu 32(%rsi),%xmm2+ vpshufb %xmm11,%xmm0,%xmm0+ vmovdqu 48(%rsi),%xmm3+ vpshufb %xmm11,%xmm1,%xmm1+ vmovdqu 64(%rsi),%xmm4+ vpshufb %xmm11,%xmm2,%xmm2+ vmovdqu 80(%rsi),%xmm5+ vpshufb %xmm11,%xmm3,%xmm3+ vmovdqu 96(%rsi),%xmm6+ vpshufb %xmm11,%xmm4,%xmm4+ vmovdqu 112(%rsi),%xmm7+ leaq K512+128(%rip),%rsi+ vpshufb %xmm11,%xmm5,%xmm5+ vpaddq -128(%rsi),%xmm0,%xmm8+ vpshufb %xmm11,%xmm6,%xmm6+ vpaddq -96(%rsi),%xmm1,%xmm9+ vpshufb %xmm11,%xmm7,%xmm7+ vpaddq -64(%rsi),%xmm2,%xmm10+ vpaddq -32(%rsi),%xmm3,%xmm11+ vmovdqa %xmm8,0(%rsp)+ vpaddq 0(%rsi),%xmm4,%xmm8+ vmovdqa %xmm9,16(%rsp)+ vpaddq 32(%rsi),%xmm5,%xmm9+ vmovdqa %xmm10,32(%rsp)+ vpaddq 64(%rsi),%xmm6,%xmm10+ vmovdqa %xmm11,48(%rsp)+ vpaddq 96(%rsi),%xmm7,%xmm11+ vmovdqa %xmm8,64(%rsp)+ movq %rax,%r14+ vmovdqa %xmm9,80(%rsp)+ movq %rbx,%rdi+ vmovdqa %xmm10,96(%rsp)+ xorq %rcx,%rdi+ vmovdqa %xmm11,112(%rsp)+ movq %r8,%r13+ jmp .Lxop_00_47++.p2align 4+.Lxop_00_47:+ addq $256,%rsi+ vpalignr $8,%xmm0,%xmm1,%xmm8+ rorq $23,%r13+ movq %r14,%rax+ vpalignr $8,%xmm4,%xmm5,%xmm11+ movq %r9,%r12+ rorq $5,%r14+.byte 143,72,120,195,200,56+ xorq %r8,%r13+ xorq %r10,%r12+ vpsrlq $7,%xmm8,%xmm8+ rorq $4,%r13+ xorq %rax,%r14+ vpaddq %xmm11,%xmm0,%xmm0+ andq %r8,%r12+ xorq %r8,%r13+ addq 0(%rsp),%r11+ movq %rax,%r15+.byte 143,72,120,195,209,7+ xorq %r10,%r12+ rorq $6,%r14+ vpxor %xmm9,%xmm8,%xmm8+ xorq %rbx,%r15+ addq %r12,%r11+ rorq $14,%r13+ andq %r15,%rdi+.byte 143,104,120,195,223,3+ xorq %rax,%r14+ addq %r13,%r11+ vpxor %xmm10,%xmm8,%xmm8+ xorq %rbx,%rdi+ rorq $28,%r14+ vpsrlq $6,%xmm7,%xmm10+ addq %r11,%rdx+ addq %rdi,%r11+ vpaddq %xmm8,%xmm0,%xmm0+ movq %rdx,%r13+ addq %r11,%r14+.byte 143,72,120,195,203,42+ rorq $23,%r13+ movq %r14,%r11+ vpxor %xmm10,%xmm11,%xmm11+ movq %r8,%r12+ rorq $5,%r14+ xorq %rdx,%r13+ xorq %r9,%r12+ vpxor %xmm9,%xmm11,%xmm11+ rorq $4,%r13+ xorq %r11,%r14+ andq %rdx,%r12+ xorq %rdx,%r13+ vpaddq %xmm11,%xmm0,%xmm0+ addq 8(%rsp),%r10+ movq %r11,%rdi+ xorq %r9,%r12+ rorq $6,%r14+ vpaddq -128(%rsi),%xmm0,%xmm10+ xorq %rax,%rdi+ addq %r12,%r10+ rorq $14,%r13+ andq %rdi,%r15+ xorq %r11,%r14+ addq %r13,%r10+ xorq %rax,%r15+ rorq $28,%r14+ addq %r10,%rcx+ addq %r15,%r10+ movq %rcx,%r13+ addq %r10,%r14+ vmovdqa %xmm10,0(%rsp)+ vpalignr $8,%xmm1,%xmm2,%xmm8+ rorq $23,%r13+ movq %r14,%r10+ vpalignr $8,%xmm5,%xmm6,%xmm11+ movq %rdx,%r12+ rorq $5,%r14+.byte 143,72,120,195,200,56+ xorq %rcx,%r13+ xorq %r8,%r12+ vpsrlq $7,%xmm8,%xmm8+ rorq $4,%r13+ xorq %r10,%r14+ vpaddq %xmm11,%xmm1,%xmm1+ andq %rcx,%r12+ xorq %rcx,%r13+ addq 16(%rsp),%r9+ movq %r10,%r15+.byte 143,72,120,195,209,7+ xorq %r8,%r12+ rorq $6,%r14+ vpxor %xmm9,%xmm8,%xmm8+ xorq %r11,%r15+ addq %r12,%r9+ rorq $14,%r13+ andq %r15,%rdi+.byte 143,104,120,195,216,3+ xorq %r10,%r14+ addq %r13,%r9+ vpxor %xmm10,%xmm8,%xmm8+ xorq %r11,%rdi+ rorq $28,%r14+ vpsrlq $6,%xmm0,%xmm10+ addq %r9,%rbx+ addq %rdi,%r9+ vpaddq %xmm8,%xmm1,%xmm1+ movq %rbx,%r13+ addq %r9,%r14+.byte 143,72,120,195,203,42+ rorq $23,%r13+ movq %r14,%r9+ vpxor %xmm10,%xmm11,%xmm11+ movq %rcx,%r12+ rorq $5,%r14+ xorq %rbx,%r13+ xorq %rdx,%r12+ vpxor %xmm9,%xmm11,%xmm11+ rorq $4,%r13+ xorq %r9,%r14+ andq %rbx,%r12+ xorq %rbx,%r13+ vpaddq %xmm11,%xmm1,%xmm1+ addq 24(%rsp),%r8+ movq %r9,%rdi+ xorq %rdx,%r12+ rorq $6,%r14+ vpaddq -96(%rsi),%xmm1,%xmm10+ xorq %r10,%rdi+ addq %r12,%r8+ rorq $14,%r13+ andq %rdi,%r15+ xorq %r9,%r14+ addq %r13,%r8+ xorq %r10,%r15+ rorq $28,%r14+ addq %r8,%rax+ addq %r15,%r8+ movq %rax,%r13+ addq %r8,%r14+ vmovdqa %xmm10,16(%rsp)+ vpalignr $8,%xmm2,%xmm3,%xmm8+ rorq $23,%r13+ movq %r14,%r8+ vpalignr $8,%xmm6,%xmm7,%xmm11+ movq %rbx,%r12+ rorq $5,%r14+.byte 143,72,120,195,200,56+ xorq %rax,%r13+ xorq %rcx,%r12+ vpsrlq $7,%xmm8,%xmm8+ rorq $4,%r13+ xorq %r8,%r14+ vpaddq %xmm11,%xmm2,%xmm2+ andq %rax,%r12+ xorq %rax,%r13+ addq 32(%rsp),%rdx+ movq %r8,%r15+.byte 143,72,120,195,209,7+ xorq %rcx,%r12+ rorq $6,%r14+ vpxor %xmm9,%xmm8,%xmm8+ xorq %r9,%r15+ addq %r12,%rdx+ rorq $14,%r13+ andq %r15,%rdi+.byte 143,104,120,195,217,3+ xorq %r8,%r14+ addq %r13,%rdx+ vpxor %xmm10,%xmm8,%xmm8+ xorq %r9,%rdi+ rorq $28,%r14+ vpsrlq $6,%xmm1,%xmm10+ addq %rdx,%r11+ addq %rdi,%rdx+ vpaddq %xmm8,%xmm2,%xmm2+ movq %r11,%r13+ addq %rdx,%r14+.byte 143,72,120,195,203,42+ rorq $23,%r13+ movq %r14,%rdx+ vpxor %xmm10,%xmm11,%xmm11+ movq %rax,%r12+ rorq $5,%r14+ xorq %r11,%r13+ xorq %rbx,%r12+ vpxor %xmm9,%xmm11,%xmm11+ rorq $4,%r13+ xorq %rdx,%r14+ andq %r11,%r12+ xorq %r11,%r13+ vpaddq %xmm11,%xmm2,%xmm2+ addq 40(%rsp),%rcx+ movq %rdx,%rdi+ xorq %rbx,%r12+ rorq $6,%r14+ vpaddq -64(%rsi),%xmm2,%xmm10+ xorq %r8,%rdi+ addq %r12,%rcx+ rorq $14,%r13+ andq %rdi,%r15+ xorq %rdx,%r14+ addq %r13,%rcx+ xorq %r8,%r15+ rorq $28,%r14+ addq %rcx,%r10+ addq %r15,%rcx+ movq %r10,%r13+ addq %rcx,%r14+ vmovdqa %xmm10,32(%rsp)+ vpalignr $8,%xmm3,%xmm4,%xmm8+ rorq $23,%r13+ movq %r14,%rcx+ vpalignr $8,%xmm7,%xmm0,%xmm11+ movq %r11,%r12+ rorq $5,%r14+.byte 143,72,120,195,200,56+ xorq %r10,%r13+ xorq %rax,%r12+ vpsrlq $7,%xmm8,%xmm8+ rorq $4,%r13+ xorq %rcx,%r14+ vpaddq %xmm11,%xmm3,%xmm3+ andq %r10,%r12+ xorq %r10,%r13+ addq 48(%rsp),%rbx+ movq %rcx,%r15+.byte 143,72,120,195,209,7+ xorq %rax,%r12+ rorq $6,%r14+ vpxor %xmm9,%xmm8,%xmm8+ xorq %rdx,%r15+ addq %r12,%rbx+ rorq $14,%r13+ andq %r15,%rdi+.byte 143,104,120,195,218,3+ xorq %rcx,%r14+ addq %r13,%rbx+ vpxor %xmm10,%xmm8,%xmm8+ xorq %rdx,%rdi+ rorq $28,%r14+ vpsrlq $6,%xmm2,%xmm10+ addq %rbx,%r9+ addq %rdi,%rbx+ vpaddq %xmm8,%xmm3,%xmm3+ movq %r9,%r13+ addq %rbx,%r14+.byte 143,72,120,195,203,42+ rorq $23,%r13+ movq %r14,%rbx+ vpxor %xmm10,%xmm11,%xmm11+ movq %r10,%r12+ rorq $5,%r14+ xorq %r9,%r13+ xorq %r11,%r12+ vpxor %xmm9,%xmm11,%xmm11+ rorq $4,%r13+ xorq %rbx,%r14+ andq %r9,%r12+ xorq %r9,%r13+ vpaddq %xmm11,%xmm3,%xmm3+ addq 56(%rsp),%rax+ movq %rbx,%rdi+ xorq %r11,%r12+ rorq $6,%r14+ vpaddq -32(%rsi),%xmm3,%xmm10+ xorq %rcx,%rdi+ addq %r12,%rax+ rorq $14,%r13+ andq %rdi,%r15+ xorq %rbx,%r14+ addq %r13,%rax+ xorq %rcx,%r15+ rorq $28,%r14+ addq %rax,%r8+ addq %r15,%rax+ movq %r8,%r13+ addq %rax,%r14+ vmovdqa %xmm10,48(%rsp)+ vpalignr $8,%xmm4,%xmm5,%xmm8+ rorq $23,%r13+ movq %r14,%rax+ vpalignr $8,%xmm0,%xmm1,%xmm11+ movq %r9,%r12+ rorq $5,%r14+.byte 143,72,120,195,200,56+ xorq %r8,%r13+ xorq %r10,%r12+ vpsrlq $7,%xmm8,%xmm8+ rorq $4,%r13+ xorq %rax,%r14+ vpaddq %xmm11,%xmm4,%xmm4+ andq %r8,%r12+ xorq %r8,%r13+ addq 64(%rsp),%r11+ movq %rax,%r15+.byte 143,72,120,195,209,7+ xorq %r10,%r12+ rorq $6,%r14+ vpxor %xmm9,%xmm8,%xmm8+ xorq %rbx,%r15+ addq %r12,%r11+ rorq $14,%r13+ andq %r15,%rdi+.byte 143,104,120,195,219,3+ xorq %rax,%r14+ addq %r13,%r11+ vpxor %xmm10,%xmm8,%xmm8+ xorq %rbx,%rdi+ rorq $28,%r14+ vpsrlq $6,%xmm3,%xmm10+ addq %r11,%rdx+ addq %rdi,%r11+ vpaddq %xmm8,%xmm4,%xmm4+ movq %rdx,%r13+ addq %r11,%r14+.byte 143,72,120,195,203,42+ rorq $23,%r13+ movq %r14,%r11+ vpxor %xmm10,%xmm11,%xmm11+ movq %r8,%r12+ rorq $5,%r14+ xorq %rdx,%r13+ xorq %r9,%r12+ vpxor %xmm9,%xmm11,%xmm11+ rorq $4,%r13+ xorq %r11,%r14+ andq %rdx,%r12+ xorq %rdx,%r13+ vpaddq %xmm11,%xmm4,%xmm4+ addq 72(%rsp),%r10+ movq %r11,%rdi+ xorq %r9,%r12+ rorq $6,%r14+ vpaddq 0(%rsi),%xmm4,%xmm10+ xorq %rax,%rdi+ addq %r12,%r10+ rorq $14,%r13+ andq %rdi,%r15+ xorq %r11,%r14+ addq %r13,%r10+ xorq %rax,%r15+ rorq $28,%r14+ addq %r10,%rcx+ addq %r15,%r10+ movq %rcx,%r13+ addq %r10,%r14+ vmovdqa %xmm10,64(%rsp)+ vpalignr $8,%xmm5,%xmm6,%xmm8+ rorq $23,%r13+ movq %r14,%r10+ vpalignr $8,%xmm1,%xmm2,%xmm11+ movq %rdx,%r12+ rorq $5,%r14+.byte 143,72,120,195,200,56+ xorq %rcx,%r13+ xorq %r8,%r12+ vpsrlq $7,%xmm8,%xmm8+ rorq $4,%r13+ xorq %r10,%r14+ vpaddq %xmm11,%xmm5,%xmm5+ andq %rcx,%r12+ xorq %rcx,%r13+ addq 80(%rsp),%r9+ movq %r10,%r15+.byte 143,72,120,195,209,7+ xorq %r8,%r12+ rorq $6,%r14+ vpxor %xmm9,%xmm8,%xmm8+ xorq %r11,%r15+ addq %r12,%r9+ rorq $14,%r13+ andq %r15,%rdi+.byte 143,104,120,195,220,3+ xorq %r10,%r14+ addq %r13,%r9+ vpxor %xmm10,%xmm8,%xmm8+ xorq %r11,%rdi+ rorq $28,%r14+ vpsrlq $6,%xmm4,%xmm10+ addq %r9,%rbx+ addq %rdi,%r9+ vpaddq %xmm8,%xmm5,%xmm5+ movq %rbx,%r13+ addq %r9,%r14+.byte 143,72,120,195,203,42+ rorq $23,%r13+ movq %r14,%r9+ vpxor %xmm10,%xmm11,%xmm11+ movq %rcx,%r12+ rorq $5,%r14+ xorq %rbx,%r13+ xorq %rdx,%r12+ vpxor %xmm9,%xmm11,%xmm11+ rorq $4,%r13+ xorq %r9,%r14+ andq %rbx,%r12+ xorq %rbx,%r13+ vpaddq %xmm11,%xmm5,%xmm5+ addq 88(%rsp),%r8+ movq %r9,%rdi+ xorq %rdx,%r12+ rorq $6,%r14+ vpaddq 32(%rsi),%xmm5,%xmm10+ xorq %r10,%rdi+ addq %r12,%r8+ rorq $14,%r13+ andq %rdi,%r15+ xorq %r9,%r14+ addq %r13,%r8+ xorq %r10,%r15+ rorq $28,%r14+ addq %r8,%rax+ addq %r15,%r8+ movq %rax,%r13+ addq %r8,%r14+ vmovdqa %xmm10,80(%rsp)+ vpalignr $8,%xmm6,%xmm7,%xmm8+ rorq $23,%r13+ movq %r14,%r8+ vpalignr $8,%xmm2,%xmm3,%xmm11+ movq %rbx,%r12+ rorq $5,%r14+.byte 143,72,120,195,200,56+ xorq %rax,%r13+ xorq %rcx,%r12+ vpsrlq $7,%xmm8,%xmm8+ rorq $4,%r13+ xorq %r8,%r14+ vpaddq %xmm11,%xmm6,%xmm6+ andq %rax,%r12+ xorq %rax,%r13+ addq 96(%rsp),%rdx+ movq %r8,%r15+.byte 143,72,120,195,209,7+ xorq %rcx,%r12+ rorq $6,%r14+ vpxor %xmm9,%xmm8,%xmm8+ xorq %r9,%r15+ addq %r12,%rdx+ rorq $14,%r13+ andq %r15,%rdi+.byte 143,104,120,195,221,3+ xorq %r8,%r14+ addq %r13,%rdx+ vpxor %xmm10,%xmm8,%xmm8+ xorq %r9,%rdi+ rorq $28,%r14+ vpsrlq $6,%xmm5,%xmm10+ addq %rdx,%r11+ addq %rdi,%rdx+ vpaddq %xmm8,%xmm6,%xmm6+ movq %r11,%r13+ addq %rdx,%r14+.byte 143,72,120,195,203,42+ rorq $23,%r13+ movq %r14,%rdx+ vpxor %xmm10,%xmm11,%xmm11+ movq %rax,%r12+ rorq $5,%r14+ xorq %r11,%r13+ xorq %rbx,%r12+ vpxor %xmm9,%xmm11,%xmm11+ rorq $4,%r13+ xorq %rdx,%r14+ andq %r11,%r12+ xorq %r11,%r13+ vpaddq %xmm11,%xmm6,%xmm6+ addq 104(%rsp),%rcx+ movq %rdx,%rdi+ xorq %rbx,%r12+ rorq $6,%r14+ vpaddq 64(%rsi),%xmm6,%xmm10+ xorq %r8,%rdi+ addq %r12,%rcx+ rorq $14,%r13+ andq %rdi,%r15+ xorq %rdx,%r14+ addq %r13,%rcx+ xorq %r8,%r15+ rorq $28,%r14+ addq %rcx,%r10+ addq %r15,%rcx+ movq %r10,%r13+ addq %rcx,%r14+ vmovdqa %xmm10,96(%rsp)+ vpalignr $8,%xmm7,%xmm0,%xmm8+ rorq $23,%r13+ movq %r14,%rcx+ vpalignr $8,%xmm3,%xmm4,%xmm11+ movq %r11,%r12+ rorq $5,%r14+.byte 143,72,120,195,200,56+ xorq %r10,%r13+ xorq %rax,%r12+ vpsrlq $7,%xmm8,%xmm8+ rorq $4,%r13+ xorq %rcx,%r14+ vpaddq %xmm11,%xmm7,%xmm7+ andq %r10,%r12+ xorq %r10,%r13+ addq 112(%rsp),%rbx+ movq %rcx,%r15+.byte 143,72,120,195,209,7+ xorq %rax,%r12+ rorq $6,%r14+ vpxor %xmm9,%xmm8,%xmm8+ xorq %rdx,%r15+ addq %r12,%rbx+ rorq $14,%r13+ andq %r15,%rdi+.byte 143,104,120,195,222,3+ xorq %rcx,%r14+ addq %r13,%rbx+ vpxor %xmm10,%xmm8,%xmm8+ xorq %rdx,%rdi+ rorq $28,%r14+ vpsrlq $6,%xmm6,%xmm10+ addq %rbx,%r9+ addq %rdi,%rbx+ vpaddq %xmm8,%xmm7,%xmm7+ movq %r9,%r13+ addq %rbx,%r14+.byte 143,72,120,195,203,42+ rorq $23,%r13+ movq %r14,%rbx+ vpxor %xmm10,%xmm11,%xmm11+ movq %r10,%r12+ rorq $5,%r14+ xorq %r9,%r13+ xorq %r11,%r12+ vpxor %xmm9,%xmm11,%xmm11+ rorq $4,%r13+ xorq %rbx,%r14+ andq %r9,%r12+ xorq %r9,%r13+ vpaddq %xmm11,%xmm7,%xmm7+ addq 120(%rsp),%rax+ movq %rbx,%rdi+ xorq %r11,%r12+ rorq $6,%r14+ vpaddq 96(%rsi),%xmm7,%xmm10+ xorq %rcx,%rdi+ addq %r12,%rax+ rorq $14,%r13+ andq %rdi,%r15+ xorq %rbx,%r14+ addq %r13,%rax+ xorq %rcx,%r15+ rorq $28,%r14+ addq %rax,%r8+ addq %r15,%rax+ movq %r8,%r13+ addq %rax,%r14+ vmovdqa %xmm10,112(%rsp)+ cmpb $0,135(%rsi)+ jne .Lxop_00_47+ rorq $23,%r13+ movq %r14,%rax+ movq %r9,%r12+ rorq $5,%r14+ xorq %r8,%r13+ xorq %r10,%r12+ rorq $4,%r13+ xorq %rax,%r14+ andq %r8,%r12+ xorq %r8,%r13+ addq 0(%rsp),%r11+ movq %rax,%r15+ xorq %r10,%r12+ rorq $6,%r14+ xorq %rbx,%r15+ addq %r12,%r11+ rorq $14,%r13+ andq %r15,%rdi+ xorq %rax,%r14+ addq %r13,%r11+ xorq %rbx,%rdi+ rorq $28,%r14+ addq %r11,%rdx+ addq %rdi,%r11+ movq %rdx,%r13+ addq %r11,%r14+ rorq $23,%r13+ movq %r14,%r11+ movq %r8,%r12+ rorq $5,%r14+ xorq %rdx,%r13+ xorq %r9,%r12+ rorq $4,%r13+ xorq %r11,%r14+ andq %rdx,%r12+ xorq %rdx,%r13+ addq 8(%rsp),%r10+ movq %r11,%rdi+ xorq %r9,%r12+ rorq $6,%r14+ xorq %rax,%rdi+ addq %r12,%r10+ rorq $14,%r13+ andq %rdi,%r15+ xorq %r11,%r14+ addq %r13,%r10+ xorq %rax,%r15+ rorq $28,%r14+ addq %r10,%rcx+ addq %r15,%r10+ movq %rcx,%r13+ addq %r10,%r14+ rorq $23,%r13+ movq %r14,%r10+ movq %rdx,%r12+ rorq $5,%r14+ xorq %rcx,%r13+ xorq %r8,%r12+ rorq $4,%r13+ xorq %r10,%r14+ andq %rcx,%r12+ xorq %rcx,%r13+ addq 16(%rsp),%r9+ movq %r10,%r15+ xorq %r8,%r12+ rorq $6,%r14+ xorq %r11,%r15+ addq %r12,%r9+ rorq $14,%r13+ andq %r15,%rdi+ xorq %r10,%r14+ addq %r13,%r9+ xorq %r11,%rdi+ rorq $28,%r14+ addq %r9,%rbx+ addq %rdi,%r9+ movq %rbx,%r13+ addq %r9,%r14+ rorq $23,%r13+ movq %r14,%r9+ movq %rcx,%r12+ rorq $5,%r14+ xorq %rbx,%r13+ xorq %rdx,%r12+ rorq $4,%r13+ xorq %r9,%r14+ andq %rbx,%r12+ xorq %rbx,%r13+ addq 24(%rsp),%r8+ movq %r9,%rdi+ xorq %rdx,%r12+ rorq $6,%r14+ xorq %r10,%rdi+ addq %r12,%r8+ rorq $14,%r13+ andq %rdi,%r15+ xorq %r9,%r14+ addq %r13,%r8+ xorq %r10,%r15+ rorq $28,%r14+ addq %r8,%rax+ addq %r15,%r8+ movq %rax,%r13+ addq %r8,%r14+ rorq $23,%r13+ movq %r14,%r8+ movq %rbx,%r12+ rorq $5,%r14+ xorq %rax,%r13+ xorq %rcx,%r12+ rorq $4,%r13+ xorq %r8,%r14+ andq %rax,%r12+ xorq %rax,%r13+ addq 32(%rsp),%rdx+ movq %r8,%r15+ xorq %rcx,%r12+ rorq $6,%r14+ xorq %r9,%r15+ addq %r12,%rdx+ rorq $14,%r13+ andq %r15,%rdi+ xorq %r8,%r14+ addq %r13,%rdx+ xorq %r9,%rdi+ rorq $28,%r14+ addq %rdx,%r11+ addq %rdi,%rdx+ movq %r11,%r13+ addq %rdx,%r14+ rorq $23,%r13+ movq %r14,%rdx+ movq %rax,%r12+ rorq $5,%r14+ xorq %r11,%r13+ xorq %rbx,%r12+ rorq $4,%r13+ xorq %rdx,%r14+ andq %r11,%r12+ xorq %r11,%r13+ addq 40(%rsp),%rcx+ movq %rdx,%rdi+ xorq %rbx,%r12+ rorq $6,%r14+ xorq %r8,%rdi+ addq %r12,%rcx+ rorq $14,%r13+ andq %rdi,%r15+ xorq %rdx,%r14+ addq %r13,%rcx+ xorq %r8,%r15+ rorq $28,%r14+ addq %rcx,%r10+ addq %r15,%rcx+ movq %r10,%r13+ addq %rcx,%r14+ rorq $23,%r13+ movq %r14,%rcx+ movq %r11,%r12+ rorq $5,%r14+ xorq %r10,%r13+ xorq %rax,%r12+ rorq $4,%r13+ xorq %rcx,%r14+ andq %r10,%r12+ xorq %r10,%r13+ addq 48(%rsp),%rbx+ movq %rcx,%r15+ xorq %rax,%r12+ rorq $6,%r14+ xorq %rdx,%r15+ addq %r12,%rbx+ rorq $14,%r13+ andq %r15,%rdi+ xorq %rcx,%r14+ addq %r13,%rbx+ xorq %rdx,%rdi+ rorq $28,%r14+ addq %rbx,%r9+ addq %rdi,%rbx+ movq %r9,%r13+ addq %rbx,%r14+ rorq $23,%r13+ movq %r14,%rbx+ movq %r10,%r12+ rorq $5,%r14+ xorq %r9,%r13+ xorq %r11,%r12+ rorq $4,%r13+ xorq %rbx,%r14+ andq %r9,%r12+ xorq %r9,%r13+ addq 56(%rsp),%rax+ movq %rbx,%rdi+ xorq %r11,%r12+ rorq $6,%r14+ xorq %rcx,%rdi+ addq %r12,%rax+ rorq $14,%r13+ andq %rdi,%r15+ xorq %rbx,%r14+ addq %r13,%rax+ xorq %rcx,%r15+ rorq $28,%r14+ addq %rax,%r8+ addq %r15,%rax+ movq %r8,%r13+ addq %rax,%r14+ rorq $23,%r13+ movq %r14,%rax+ movq %r9,%r12+ rorq $5,%r14+ xorq %r8,%r13+ xorq %r10,%r12+ rorq $4,%r13+ xorq %rax,%r14+ andq %r8,%r12+ xorq %r8,%r13+ addq 64(%rsp),%r11+ movq %rax,%r15+ xorq %r10,%r12+ rorq $6,%r14+ xorq %rbx,%r15+ addq %r12,%r11+ rorq $14,%r13+ andq %r15,%rdi+ xorq %rax,%r14+ addq %r13,%r11+ xorq %rbx,%rdi+ rorq $28,%r14+ addq %r11,%rdx+ addq %rdi,%r11+ movq %rdx,%r13+ addq %r11,%r14+ rorq $23,%r13+ movq %r14,%r11+ movq %r8,%r12+ rorq $5,%r14+ xorq %rdx,%r13+ xorq %r9,%r12+ rorq $4,%r13+ xorq %r11,%r14+ andq %rdx,%r12+ xorq %rdx,%r13+ addq 72(%rsp),%r10+ movq %r11,%rdi+ xorq %r9,%r12+ rorq $6,%r14+ xorq %rax,%rdi+ addq %r12,%r10+ rorq $14,%r13+ andq %rdi,%r15+ xorq %r11,%r14+ addq %r13,%r10+ xorq %rax,%r15+ rorq $28,%r14+ addq %r10,%rcx+ addq %r15,%r10+ movq %rcx,%r13+ addq %r10,%r14+ rorq $23,%r13+ movq %r14,%r10+ movq %rdx,%r12+ rorq $5,%r14+ xorq %rcx,%r13+ xorq %r8,%r12+ rorq $4,%r13+ xorq %r10,%r14+ andq %rcx,%r12+ xorq %rcx,%r13+ addq 80(%rsp),%r9+ movq %r10,%r15+ xorq %r8,%r12+ rorq $6,%r14+ xorq %r11,%r15+ addq %r12,%r9+ rorq $14,%r13+ andq %r15,%rdi+ xorq %r10,%r14+ addq %r13,%r9+ xorq %r11,%rdi+ rorq $28,%r14+ addq %r9,%rbx+ addq %rdi,%r9+ movq %rbx,%r13+ addq %r9,%r14+ rorq $23,%r13+ movq %r14,%r9+ movq %rcx,%r12+ rorq $5,%r14+ xorq %rbx,%r13+ xorq %rdx,%r12+ rorq $4,%r13+ xorq %r9,%r14+ andq %rbx,%r12+ xorq %rbx,%r13+ addq 88(%rsp),%r8+ movq %r9,%rdi+ xorq %rdx,%r12+ rorq $6,%r14+ xorq %r10,%rdi+ addq %r12,%r8+ rorq $14,%r13+ andq %rdi,%r15+ xorq %r9,%r14+ addq %r13,%r8+ xorq %r10,%r15+ rorq $28,%r14+ addq %r8,%rax+ addq %r15,%r8+ movq %rax,%r13+ addq %r8,%r14+ rorq $23,%r13+ movq %r14,%r8+ movq %rbx,%r12+ rorq $5,%r14+ xorq %rax,%r13+ xorq %rcx,%r12+ rorq $4,%r13+ xorq %r8,%r14+ andq %rax,%r12+ xorq %rax,%r13+ addq 96(%rsp),%rdx+ movq %r8,%r15+ xorq %rcx,%r12+ rorq $6,%r14+ xorq %r9,%r15+ addq %r12,%rdx+ rorq $14,%r13+ andq %r15,%rdi+ xorq %r8,%r14+ addq %r13,%rdx+ xorq %r9,%rdi+ rorq $28,%r14+ addq %rdx,%r11+ addq %rdi,%rdx+ movq %r11,%r13+ addq %rdx,%r14+ rorq $23,%r13+ movq %r14,%rdx+ movq %rax,%r12+ rorq $5,%r14+ xorq %r11,%r13+ xorq %rbx,%r12+ rorq $4,%r13+ xorq %rdx,%r14+ andq %r11,%r12+ xorq %r11,%r13+ addq 104(%rsp),%rcx+ movq %rdx,%rdi+ xorq %rbx,%r12+ rorq $6,%r14+ xorq %r8,%rdi+ addq %r12,%rcx+ rorq $14,%r13+ andq %rdi,%r15+ xorq %rdx,%r14+ addq %r13,%rcx+ xorq %r8,%r15+ rorq $28,%r14+ addq %rcx,%r10+ addq %r15,%rcx+ movq %r10,%r13+ addq %rcx,%r14+ rorq $23,%r13+ movq %r14,%rcx+ movq %r11,%r12+ rorq $5,%r14+ xorq %r10,%r13+ xorq %rax,%r12+ rorq $4,%r13+ xorq %rcx,%r14+ andq %r10,%r12+ xorq %r10,%r13+ addq 112(%rsp),%rbx+ movq %rcx,%r15+ xorq %rax,%r12+ rorq $6,%r14+ xorq %rdx,%r15+ addq %r12,%rbx+ rorq $14,%r13+ andq %r15,%rdi+ xorq %rcx,%r14+ addq %r13,%rbx+ xorq %rdx,%rdi+ rorq $28,%r14+ addq %rbx,%r9+ addq %rdi,%rbx+ movq %r9,%r13+ addq %rbx,%r14+ rorq $23,%r13+ movq %r14,%rbx+ movq %r10,%r12+ rorq $5,%r14+ xorq %r9,%r13+ xorq %r11,%r12+ rorq $4,%r13+ xorq %rbx,%r14+ andq %r9,%r12+ xorq %r9,%r13+ addq 120(%rsp),%rax+ movq %rbx,%rdi+ xorq %r11,%r12+ rorq $6,%r14+ xorq %rcx,%rdi+ addq %r12,%rax+ rorq $14,%r13+ andq %rdi,%r15+ xorq %rbx,%r14+ addq %r13,%rax+ xorq %rcx,%r15+ rorq $28,%r14+ addq %rax,%r8+ addq %r15,%rax+ movq %r8,%r13+ addq %rax,%r14+ movq -64(%rbp),%rdi+ movq %r14,%rax+ movq -56(%rbp),%rsi++ addq 0(%rdi),%rax+ addq 8(%rdi),%rbx+ addq 16(%rdi),%rcx+ addq 24(%rdi),%rdx+ addq 32(%rdi),%r8+ addq 40(%rdi),%r9+ addq 48(%rdi),%r10+ addq 56(%rdi),%r11++ leaq 128(%rsi),%rsi+ cmpq -48(%rbp),%rsi++ movq %rax,0(%rdi)+ movq %rbx,8(%rdi)+ movq %rcx,16(%rdi)+ movq %rdx,24(%rdi)+ movq %r8,32(%rdi)+ movq %r9,40(%rdi)+ movq %r10,48(%rdi)+ movq %r11,56(%rdi)+ jb .Lloop_xop++ vzeroupper+ movaps -160(%rbp),%xmm6+ movaps -144(%rbp),%xmm7+ movaps -128(%rbp),%xmm8+ movaps -112(%rbp),%xmm9+ movaps -96(%rbp),%xmm10+ movaps -80(%rbp),%xmm11+ movq -40(%rbp),%r15+ movq -32(%rbp),%r14+ movq -24(%rbp),%r13+ movq -16(%rbp),%r12+ movq -8(%rbp),%rbx+ movq %rbp,%rsp++ popq %rbp++.LSEH_epilogue_crypton_sha512_asm_block_data_order_xop:+ mov 8(%rsp),%rdi+ mov 16(%rsp),%rsi++ .byte 0xf3,0xc3++.LSEH_end_crypton_sha512_asm_block_data_order_xop:+.def crypton_sha512_asm_block_data_order_avx; .scl 3; .type 32; .endef+.p2align 6+crypton_sha512_asm_block_data_order_avx:+ .byte 0xf3,0x0f,0x1e,0xfa+ movq %rdi,8(%rsp)+ movq %rsi,16(%rsp)+ movq %rsp,%r11+.LSEH_begin_crypton_sha512_asm_block_data_order_avx:+++ pushq %rbp++ movq %rsp,%rbp++ movq %rcx,%rdi+ movq %rdx,%rsi+ movq %r8,%rdx+.Lavx_shortcut:+ pushq %rbx++ pushq %r12++ pushq %r13++ pushq %r14++ pushq %r15++ shlq $4,%rdx+ subq $120,%rsp++ leaq (%rsi,%rdx,8),%rdx+ movq %rdi,-64(%rbp)++ movq %rdx,-48(%rbp)+ movaps %xmm6,-160(%rbp)+ movaps %xmm7,-144(%rbp)+ movaps %xmm8,-128(%rbp)+ movaps %xmm9,-112(%rbp)++ movaps %xmm10,-96(%rbp)+ movaps %xmm11,-80(%rbp)++.LSEH_body_crypton_sha512_asm_block_data_order_avx:+++ leaq -128(%rsp),%rsp+ vzeroupper+ andq $-64,%rsp+ movq 0(%rdi),%rax+ movq 8(%rdi),%rbx+ movq 16(%rdi),%rcx+ movq 24(%rdi),%rdx+ movq 32(%rdi),%r8+ movq 40(%rdi),%r9+ movq 48(%rdi),%r10+ movq 56(%rdi),%r11+ jmp .Lloop_avx+.p2align 4+.Lloop_avx:+ vmovdqa K512+1280(%rip),%xmm11+ movq %rsi,-56(%rbp)+ vmovdqu 0(%rsi),%xmm0+ vmovdqu 16(%rsi),%xmm1+ vmovdqu 32(%rsi),%xmm2+ vpshufb %xmm11,%xmm0,%xmm0+ vmovdqu 48(%rsi),%xmm3+ vpshufb %xmm11,%xmm1,%xmm1+ vmovdqu 64(%rsi),%xmm4+ vpshufb %xmm11,%xmm2,%xmm2+ vmovdqu 80(%rsi),%xmm5+ vpshufb %xmm11,%xmm3,%xmm3+ vmovdqu 96(%rsi),%xmm6+ vpshufb %xmm11,%xmm4,%xmm4+ vmovdqu 112(%rsi),%xmm7+ leaq K512+128(%rip),%rsi+ vpshufb %xmm11,%xmm5,%xmm5+ vpaddq -128(%rsi),%xmm0,%xmm8+ vpshufb %xmm11,%xmm6,%xmm6+ vpaddq -96(%rsi),%xmm1,%xmm9+ vpshufb %xmm11,%xmm7,%xmm7+ vpaddq -64(%rsi),%xmm2,%xmm10+ vpaddq -32(%rsi),%xmm3,%xmm11+ vmovdqa %xmm8,0(%rsp)+ vpaddq 0(%rsi),%xmm4,%xmm8+ vmovdqa %xmm9,16(%rsp)+ vpaddq 32(%rsi),%xmm5,%xmm9+ vmovdqa %xmm10,32(%rsp)+ vpaddq 64(%rsi),%xmm6,%xmm10+ vmovdqa %xmm11,48(%rsp)+ vpaddq 96(%rsi),%xmm7,%xmm11+ vmovdqa %xmm8,64(%rsp)+ movq %rax,%r14+ vmovdqa %xmm9,80(%rsp)+ movq %rbx,%rdi+ vmovdqa %xmm10,96(%rsp)+ xorq %rcx,%rdi+ vmovdqa %xmm11,112(%rsp)+ movq %r8,%r13+ jmp .Lavx_00_47++.p2align 4+.Lavx_00_47:+ addq $256,%rsi+ vpalignr $8,%xmm0,%xmm1,%xmm8+ shrdq $23,%r13,%r13+ movq %r14,%rax+ vpalignr $8,%xmm4,%xmm5,%xmm11+ movq %r9,%r12+ shrdq $5,%r14,%r14+ vpsrlq $1,%xmm8,%xmm10+ xorq %r8,%r13+ xorq %r10,%r12+ vpaddq %xmm11,%xmm0,%xmm0+ shrdq $4,%r13,%r13+ xorq %rax,%r14+ vpsrlq $7,%xmm8,%xmm11+ andq %r8,%r12+ xorq %r8,%r13+ vpsllq $56,%xmm8,%xmm9+ addq 0(%rsp),%r11+ movq %rax,%r15+ vpxor %xmm10,%xmm11,%xmm8+ xorq %r10,%r12+ shrdq $6,%r14,%r14+ vpsrlq $7,%xmm10,%xmm10+ xorq %rbx,%r15+ addq %r12,%r11+ vpxor %xmm9,%xmm8,%xmm8+ shrdq $14,%r13,%r13+ andq %r15,%rdi+ vpsllq $7,%xmm9,%xmm9+ xorq %rax,%r14+ addq %r13,%r11+ vpxor %xmm10,%xmm8,%xmm8+ xorq %rbx,%rdi+ shrdq $28,%r14,%r14+ vpsrlq $6,%xmm7,%xmm11+ addq %r11,%rdx+ addq %rdi,%r11+ vpxor %xmm9,%xmm8,%xmm8+ movq %rdx,%r13+ addq %r11,%r14+ vpsllq $3,%xmm7,%xmm10+ shrdq $23,%r13,%r13+ movq %r14,%r11+ vpaddq %xmm8,%xmm0,%xmm0+ movq %r8,%r12+ shrdq $5,%r14,%r14+ vpsrlq $19,%xmm7,%xmm9+ xorq %rdx,%r13+ xorq %r9,%r12+ vpxor %xmm10,%xmm11,%xmm11+ shrdq $4,%r13,%r13+ xorq %r11,%r14+ vpsllq $42,%xmm10,%xmm10+ andq %rdx,%r12+ xorq %rdx,%r13+ vpxor %xmm9,%xmm11,%xmm11+ addq 8(%rsp),%r10+ movq %r11,%rdi+ vpsrlq $42,%xmm9,%xmm9+ xorq %r9,%r12+ shrdq $6,%r14,%r14+ vpxor %xmm10,%xmm11,%xmm11+ xorq %rax,%rdi+ addq %r12,%r10+ vpxor %xmm9,%xmm11,%xmm11+ shrdq $14,%r13,%r13+ andq %rdi,%r15+ vpaddq %xmm11,%xmm0,%xmm0+ xorq %r11,%r14+ addq %r13,%r10+ vpaddq -128(%rsi),%xmm0,%xmm10+ xorq %rax,%r15+ shrdq $28,%r14,%r14+ addq %r10,%rcx+ addq %r15,%r10+ movq %rcx,%r13+ addq %r10,%r14+ vmovdqa %xmm10,0(%rsp)+ vpalignr $8,%xmm1,%xmm2,%xmm8+ shrdq $23,%r13,%r13+ movq %r14,%r10+ vpalignr $8,%xmm5,%xmm6,%xmm11+ movq %rdx,%r12+ shrdq $5,%r14,%r14+ vpsrlq $1,%xmm8,%xmm10+ xorq %rcx,%r13+ xorq %r8,%r12+ vpaddq %xmm11,%xmm1,%xmm1+ shrdq $4,%r13,%r13+ xorq %r10,%r14+ vpsrlq $7,%xmm8,%xmm11+ andq %rcx,%r12+ xorq %rcx,%r13+ vpsllq $56,%xmm8,%xmm9+ addq 16(%rsp),%r9+ movq %r10,%r15+ vpxor %xmm10,%xmm11,%xmm8+ xorq %r8,%r12+ shrdq $6,%r14,%r14+ vpsrlq $7,%xmm10,%xmm10+ xorq %r11,%r15+ addq %r12,%r9+ vpxor %xmm9,%xmm8,%xmm8+ shrdq $14,%r13,%r13+ andq %r15,%rdi+ vpsllq $7,%xmm9,%xmm9+ xorq %r10,%r14+ addq %r13,%r9+ vpxor %xmm10,%xmm8,%xmm8+ xorq %r11,%rdi+ shrdq $28,%r14,%r14+ vpsrlq $6,%xmm0,%xmm11+ addq %r9,%rbx+ addq %rdi,%r9+ vpxor %xmm9,%xmm8,%xmm8+ movq %rbx,%r13+ addq %r9,%r14+ vpsllq $3,%xmm0,%xmm10+ shrdq $23,%r13,%r13+ movq %r14,%r9+ vpaddq %xmm8,%xmm1,%xmm1+ movq %rcx,%r12+ shrdq $5,%r14,%r14+ vpsrlq $19,%xmm0,%xmm9+ xorq %rbx,%r13+ xorq %rdx,%r12+ vpxor %xmm10,%xmm11,%xmm11+ shrdq $4,%r13,%r13+ xorq %r9,%r14+ vpsllq $42,%xmm10,%xmm10+ andq %rbx,%r12+ xorq %rbx,%r13+ vpxor %xmm9,%xmm11,%xmm11+ addq 24(%rsp),%r8+ movq %r9,%rdi+ vpsrlq $42,%xmm9,%xmm9+ xorq %rdx,%r12+ shrdq $6,%r14,%r14+ vpxor %xmm10,%xmm11,%xmm11+ xorq %r10,%rdi+ addq %r12,%r8+ vpxor %xmm9,%xmm11,%xmm11+ shrdq $14,%r13,%r13+ andq %rdi,%r15+ vpaddq %xmm11,%xmm1,%xmm1+ xorq %r9,%r14+ addq %r13,%r8+ vpaddq -96(%rsi),%xmm1,%xmm10+ xorq %r10,%r15+ shrdq $28,%r14,%r14+ addq %r8,%rax+ addq %r15,%r8+ movq %rax,%r13+ addq %r8,%r14+ vmovdqa %xmm10,16(%rsp)+ vpalignr $8,%xmm2,%xmm3,%xmm8+ shrdq $23,%r13,%r13+ movq %r14,%r8+ vpalignr $8,%xmm6,%xmm7,%xmm11+ movq %rbx,%r12+ shrdq $5,%r14,%r14+ vpsrlq $1,%xmm8,%xmm10+ xorq %rax,%r13+ xorq %rcx,%r12+ vpaddq %xmm11,%xmm2,%xmm2+ shrdq $4,%r13,%r13+ xorq %r8,%r14+ vpsrlq $7,%xmm8,%xmm11+ andq %rax,%r12+ xorq %rax,%r13+ vpsllq $56,%xmm8,%xmm9+ addq 32(%rsp),%rdx+ movq %r8,%r15+ vpxor %xmm10,%xmm11,%xmm8+ xorq %rcx,%r12+ shrdq $6,%r14,%r14+ vpsrlq $7,%xmm10,%xmm10+ xorq %r9,%r15+ addq %r12,%rdx+ vpxor %xmm9,%xmm8,%xmm8+ shrdq $14,%r13,%r13+ andq %r15,%rdi+ vpsllq $7,%xmm9,%xmm9+ xorq %r8,%r14+ addq %r13,%rdx+ vpxor %xmm10,%xmm8,%xmm8+ xorq %r9,%rdi+ shrdq $28,%r14,%r14+ vpsrlq $6,%xmm1,%xmm11+ addq %rdx,%r11+ addq %rdi,%rdx+ vpxor %xmm9,%xmm8,%xmm8+ movq %r11,%r13+ addq %rdx,%r14+ vpsllq $3,%xmm1,%xmm10+ shrdq $23,%r13,%r13+ movq %r14,%rdx+ vpaddq %xmm8,%xmm2,%xmm2+ movq %rax,%r12+ shrdq $5,%r14,%r14+ vpsrlq $19,%xmm1,%xmm9+ xorq %r11,%r13+ xorq %rbx,%r12+ vpxor %xmm10,%xmm11,%xmm11+ shrdq $4,%r13,%r13+ xorq %rdx,%r14+ vpsllq $42,%xmm10,%xmm10+ andq %r11,%r12+ xorq %r11,%r13+ vpxor %xmm9,%xmm11,%xmm11+ addq 40(%rsp),%rcx+ movq %rdx,%rdi+ vpsrlq $42,%xmm9,%xmm9+ xorq %rbx,%r12+ shrdq $6,%r14,%r14+ vpxor %xmm10,%xmm11,%xmm11+ xorq %r8,%rdi+ addq %r12,%rcx+ vpxor %xmm9,%xmm11,%xmm11+ shrdq $14,%r13,%r13+ andq %rdi,%r15+ vpaddq %xmm11,%xmm2,%xmm2+ xorq %rdx,%r14+ addq %r13,%rcx+ vpaddq -64(%rsi),%xmm2,%xmm10+ xorq %r8,%r15+ shrdq $28,%r14,%r14+ addq %rcx,%r10+ addq %r15,%rcx+ movq %r10,%r13+ addq %rcx,%r14+ vmovdqa %xmm10,32(%rsp)+ vpalignr $8,%xmm3,%xmm4,%xmm8+ shrdq $23,%r13,%r13+ movq %r14,%rcx+ vpalignr $8,%xmm7,%xmm0,%xmm11+ movq %r11,%r12+ shrdq $5,%r14,%r14+ vpsrlq $1,%xmm8,%xmm10+ xorq %r10,%r13+ xorq %rax,%r12+ vpaddq %xmm11,%xmm3,%xmm3+ shrdq $4,%r13,%r13+ xorq %rcx,%r14+ vpsrlq $7,%xmm8,%xmm11+ andq %r10,%r12+ xorq %r10,%r13+ vpsllq $56,%xmm8,%xmm9+ addq 48(%rsp),%rbx+ movq %rcx,%r15+ vpxor %xmm10,%xmm11,%xmm8+ xorq %rax,%r12+ shrdq $6,%r14,%r14+ vpsrlq $7,%xmm10,%xmm10+ xorq %rdx,%r15+ addq %r12,%rbx+ vpxor %xmm9,%xmm8,%xmm8+ shrdq $14,%r13,%r13+ andq %r15,%rdi+ vpsllq $7,%xmm9,%xmm9+ xorq %rcx,%r14+ addq %r13,%rbx+ vpxor %xmm10,%xmm8,%xmm8+ xorq %rdx,%rdi+ shrdq $28,%r14,%r14+ vpsrlq $6,%xmm2,%xmm11+ addq %rbx,%r9+ addq %rdi,%rbx+ vpxor %xmm9,%xmm8,%xmm8+ movq %r9,%r13+ addq %rbx,%r14+ vpsllq $3,%xmm2,%xmm10+ shrdq $23,%r13,%r13+ movq %r14,%rbx+ vpaddq %xmm8,%xmm3,%xmm3+ movq %r10,%r12+ shrdq $5,%r14,%r14+ vpsrlq $19,%xmm2,%xmm9+ xorq %r9,%r13+ xorq %r11,%r12+ vpxor %xmm10,%xmm11,%xmm11+ shrdq $4,%r13,%r13+ xorq %rbx,%r14+ vpsllq $42,%xmm10,%xmm10+ andq %r9,%r12+ xorq %r9,%r13+ vpxor %xmm9,%xmm11,%xmm11+ addq 56(%rsp),%rax+ movq %rbx,%rdi+ vpsrlq $42,%xmm9,%xmm9+ xorq %r11,%r12+ shrdq $6,%r14,%r14+ vpxor %xmm10,%xmm11,%xmm11+ xorq %rcx,%rdi+ addq %r12,%rax+ vpxor %xmm9,%xmm11,%xmm11+ shrdq $14,%r13,%r13+ andq %rdi,%r15+ vpaddq %xmm11,%xmm3,%xmm3+ xorq %rbx,%r14+ addq %r13,%rax+ vpaddq -32(%rsi),%xmm3,%xmm10+ xorq %rcx,%r15+ shrdq $28,%r14,%r14+ addq %rax,%r8+ addq %r15,%rax+ movq %r8,%r13+ addq %rax,%r14+ vmovdqa %xmm10,48(%rsp)+ vpalignr $8,%xmm4,%xmm5,%xmm8+ shrdq $23,%r13,%r13+ movq %r14,%rax+ vpalignr $8,%xmm0,%xmm1,%xmm11+ movq %r9,%r12+ shrdq $5,%r14,%r14+ vpsrlq $1,%xmm8,%xmm10+ xorq %r8,%r13+ xorq %r10,%r12+ vpaddq %xmm11,%xmm4,%xmm4+ shrdq $4,%r13,%r13+ xorq %rax,%r14+ vpsrlq $7,%xmm8,%xmm11+ andq %r8,%r12+ xorq %r8,%r13+ vpsllq $56,%xmm8,%xmm9+ addq 64(%rsp),%r11+ movq %rax,%r15+ vpxor %xmm10,%xmm11,%xmm8+ xorq %r10,%r12+ shrdq $6,%r14,%r14+ vpsrlq $7,%xmm10,%xmm10+ xorq %rbx,%r15+ addq %r12,%r11+ vpxor %xmm9,%xmm8,%xmm8+ shrdq $14,%r13,%r13+ andq %r15,%rdi+ vpsllq $7,%xmm9,%xmm9+ xorq %rax,%r14+ addq %r13,%r11+ vpxor %xmm10,%xmm8,%xmm8+ xorq %rbx,%rdi+ shrdq $28,%r14,%r14+ vpsrlq $6,%xmm3,%xmm11+ addq %r11,%rdx+ addq %rdi,%r11+ vpxor %xmm9,%xmm8,%xmm8+ movq %rdx,%r13+ addq %r11,%r14+ vpsllq $3,%xmm3,%xmm10+ shrdq $23,%r13,%r13+ movq %r14,%r11+ vpaddq %xmm8,%xmm4,%xmm4+ movq %r8,%r12+ shrdq $5,%r14,%r14+ vpsrlq $19,%xmm3,%xmm9+ xorq %rdx,%r13+ xorq %r9,%r12+ vpxor %xmm10,%xmm11,%xmm11+ shrdq $4,%r13,%r13+ xorq %r11,%r14+ vpsllq $42,%xmm10,%xmm10+ andq %rdx,%r12+ xorq %rdx,%r13+ vpxor %xmm9,%xmm11,%xmm11+ addq 72(%rsp),%r10+ movq %r11,%rdi+ vpsrlq $42,%xmm9,%xmm9+ xorq %r9,%r12+ shrdq $6,%r14,%r14+ vpxor %xmm10,%xmm11,%xmm11+ xorq %rax,%rdi+ addq %r12,%r10+ vpxor %xmm9,%xmm11,%xmm11+ shrdq $14,%r13,%r13+ andq %rdi,%r15+ vpaddq %xmm11,%xmm4,%xmm4+ xorq %r11,%r14+ addq %r13,%r10+ vpaddq 0(%rsi),%xmm4,%xmm10+ xorq %rax,%r15+ shrdq $28,%r14,%r14+ addq %r10,%rcx+ addq %r15,%r10+ movq %rcx,%r13+ addq %r10,%r14+ vmovdqa %xmm10,64(%rsp)+ vpalignr $8,%xmm5,%xmm6,%xmm8+ shrdq $23,%r13,%r13+ movq %r14,%r10+ vpalignr $8,%xmm1,%xmm2,%xmm11+ movq %rdx,%r12+ shrdq $5,%r14,%r14+ vpsrlq $1,%xmm8,%xmm10+ xorq %rcx,%r13+ xorq %r8,%r12+ vpaddq %xmm11,%xmm5,%xmm5+ shrdq $4,%r13,%r13+ xorq %r10,%r14+ vpsrlq $7,%xmm8,%xmm11+ andq %rcx,%r12+ xorq %rcx,%r13+ vpsllq $56,%xmm8,%xmm9+ addq 80(%rsp),%r9+ movq %r10,%r15+ vpxor %xmm10,%xmm11,%xmm8+ xorq %r8,%r12+ shrdq $6,%r14,%r14+ vpsrlq $7,%xmm10,%xmm10+ xorq %r11,%r15+ addq %r12,%r9+ vpxor %xmm9,%xmm8,%xmm8+ shrdq $14,%r13,%r13+ andq %r15,%rdi+ vpsllq $7,%xmm9,%xmm9+ xorq %r10,%r14+ addq %r13,%r9+ vpxor %xmm10,%xmm8,%xmm8+ xorq %r11,%rdi+ shrdq $28,%r14,%r14+ vpsrlq $6,%xmm4,%xmm11+ addq %r9,%rbx+ addq %rdi,%r9+ vpxor %xmm9,%xmm8,%xmm8+ movq %rbx,%r13+ addq %r9,%r14+ vpsllq $3,%xmm4,%xmm10+ shrdq $23,%r13,%r13+ movq %r14,%r9+ vpaddq %xmm8,%xmm5,%xmm5+ movq %rcx,%r12+ shrdq $5,%r14,%r14+ vpsrlq $19,%xmm4,%xmm9+ xorq %rbx,%r13+ xorq %rdx,%r12+ vpxor %xmm10,%xmm11,%xmm11+ shrdq $4,%r13,%r13+ xorq %r9,%r14+ vpsllq $42,%xmm10,%xmm10+ andq %rbx,%r12+ xorq %rbx,%r13+ vpxor %xmm9,%xmm11,%xmm11+ addq 88(%rsp),%r8+ movq %r9,%rdi+ vpsrlq $42,%xmm9,%xmm9+ xorq %rdx,%r12+ shrdq $6,%r14,%r14+ vpxor %xmm10,%xmm11,%xmm11+ xorq %r10,%rdi+ addq %r12,%r8+ vpxor %xmm9,%xmm11,%xmm11+ shrdq $14,%r13,%r13+ andq %rdi,%r15+ vpaddq %xmm11,%xmm5,%xmm5+ xorq %r9,%r14+ addq %r13,%r8+ vpaddq 32(%rsi),%xmm5,%xmm10+ xorq %r10,%r15+ shrdq $28,%r14,%r14+ addq %r8,%rax+ addq %r15,%r8+ movq %rax,%r13+ addq %r8,%r14+ vmovdqa %xmm10,80(%rsp)+ vpalignr $8,%xmm6,%xmm7,%xmm8+ shrdq $23,%r13,%r13+ movq %r14,%r8+ vpalignr $8,%xmm2,%xmm3,%xmm11+ movq %rbx,%r12+ shrdq $5,%r14,%r14+ vpsrlq $1,%xmm8,%xmm10+ xorq %rax,%r13+ xorq %rcx,%r12+ vpaddq %xmm11,%xmm6,%xmm6+ shrdq $4,%r13,%r13+ xorq %r8,%r14+ vpsrlq $7,%xmm8,%xmm11+ andq %rax,%r12+ xorq %rax,%r13+ vpsllq $56,%xmm8,%xmm9+ addq 96(%rsp),%rdx+ movq %r8,%r15+ vpxor %xmm10,%xmm11,%xmm8+ xorq %rcx,%r12+ shrdq $6,%r14,%r14+ vpsrlq $7,%xmm10,%xmm10+ xorq %r9,%r15+ addq %r12,%rdx+ vpxor %xmm9,%xmm8,%xmm8+ shrdq $14,%r13,%r13+ andq %r15,%rdi+ vpsllq $7,%xmm9,%xmm9+ xorq %r8,%r14+ addq %r13,%rdx+ vpxor %xmm10,%xmm8,%xmm8+ xorq %r9,%rdi+ shrdq $28,%r14,%r14+ vpsrlq $6,%xmm5,%xmm11+ addq %rdx,%r11+ addq %rdi,%rdx+ vpxor %xmm9,%xmm8,%xmm8+ movq %r11,%r13+ addq %rdx,%r14+ vpsllq $3,%xmm5,%xmm10+ shrdq $23,%r13,%r13+ movq %r14,%rdx+ vpaddq %xmm8,%xmm6,%xmm6+ movq %rax,%r12+ shrdq $5,%r14,%r14+ vpsrlq $19,%xmm5,%xmm9+ xorq %r11,%r13+ xorq %rbx,%r12+ vpxor %xmm10,%xmm11,%xmm11+ shrdq $4,%r13,%r13+ xorq %rdx,%r14+ vpsllq $42,%xmm10,%xmm10+ andq %r11,%r12+ xorq %r11,%r13+ vpxor %xmm9,%xmm11,%xmm11+ addq 104(%rsp),%rcx+ movq %rdx,%rdi+ vpsrlq $42,%xmm9,%xmm9+ xorq %rbx,%r12+ shrdq $6,%r14,%r14+ vpxor %xmm10,%xmm11,%xmm11+ xorq %r8,%rdi+ addq %r12,%rcx+ vpxor %xmm9,%xmm11,%xmm11+ shrdq $14,%r13,%r13+ andq %rdi,%r15+ vpaddq %xmm11,%xmm6,%xmm6+ xorq %rdx,%r14+ addq %r13,%rcx+ vpaddq 64(%rsi),%xmm6,%xmm10+ xorq %r8,%r15+ shrdq $28,%r14,%r14+ addq %rcx,%r10+ addq %r15,%rcx+ movq %r10,%r13+ addq %rcx,%r14+ vmovdqa %xmm10,96(%rsp)+ vpalignr $8,%xmm7,%xmm0,%xmm8+ shrdq $23,%r13,%r13+ movq %r14,%rcx+ vpalignr $8,%xmm3,%xmm4,%xmm11+ movq %r11,%r12+ shrdq $5,%r14,%r14+ vpsrlq $1,%xmm8,%xmm10+ xorq %r10,%r13+ xorq %rax,%r12+ vpaddq %xmm11,%xmm7,%xmm7+ shrdq $4,%r13,%r13+ xorq %rcx,%r14+ vpsrlq $7,%xmm8,%xmm11+ andq %r10,%r12+ xorq %r10,%r13+ vpsllq $56,%xmm8,%xmm9+ addq 112(%rsp),%rbx+ movq %rcx,%r15+ vpxor %xmm10,%xmm11,%xmm8+ xorq %rax,%r12+ shrdq $6,%r14,%r14+ vpsrlq $7,%xmm10,%xmm10+ xorq %rdx,%r15+ addq %r12,%rbx+ vpxor %xmm9,%xmm8,%xmm8+ shrdq $14,%r13,%r13+ andq %r15,%rdi+ vpsllq $7,%xmm9,%xmm9+ xorq %rcx,%r14+ addq %r13,%rbx+ vpxor %xmm10,%xmm8,%xmm8+ xorq %rdx,%rdi+ shrdq $28,%r14,%r14+ vpsrlq $6,%xmm6,%xmm11+ addq %rbx,%r9+ addq %rdi,%rbx+ vpxor %xmm9,%xmm8,%xmm8+ movq %r9,%r13+ addq %rbx,%r14+ vpsllq $3,%xmm6,%xmm10+ shrdq $23,%r13,%r13+ movq %r14,%rbx+ vpaddq %xmm8,%xmm7,%xmm7+ movq %r10,%r12+ shrdq $5,%r14,%r14+ vpsrlq $19,%xmm6,%xmm9+ xorq %r9,%r13+ xorq %r11,%r12+ vpxor %xmm10,%xmm11,%xmm11+ shrdq $4,%r13,%r13+ xorq %rbx,%r14+ vpsllq $42,%xmm10,%xmm10+ andq %r9,%r12+ xorq %r9,%r13+ vpxor %xmm9,%xmm11,%xmm11+ addq 120(%rsp),%rax+ movq %rbx,%rdi+ vpsrlq $42,%xmm9,%xmm9+ xorq %r11,%r12+ shrdq $6,%r14,%r14+ vpxor %xmm10,%xmm11,%xmm11+ xorq %rcx,%rdi+ addq %r12,%rax+ vpxor %xmm9,%xmm11,%xmm11+ shrdq $14,%r13,%r13+ andq %rdi,%r15+ vpaddq %xmm11,%xmm7,%xmm7+ xorq %rbx,%r14+ addq %r13,%rax+ vpaddq 96(%rsi),%xmm7,%xmm10+ xorq %rcx,%r15+ shrdq $28,%r14,%r14+ addq %rax,%r8+ addq %r15,%rax+ movq %r8,%r13+ addq %rax,%r14+ vmovdqa %xmm10,112(%rsp)+ cmpb $0,135(%rsi)+ jne .Lavx_00_47+ shrdq $23,%r13,%r13+ movq %r14,%rax+ movq %r9,%r12+ shrdq $5,%r14,%r14+ xorq %r8,%r13+ xorq %r10,%r12+ shrdq $4,%r13,%r13+ xorq %rax,%r14+ andq %r8,%r12+ xorq %r8,%r13+ addq 0(%rsp),%r11+ movq %rax,%r15+ xorq %r10,%r12+ shrdq $6,%r14,%r14+ xorq %rbx,%r15+ addq %r12,%r11+ shrdq $14,%r13,%r13+ andq %r15,%rdi+ xorq %rax,%r14+ addq %r13,%r11+ xorq %rbx,%rdi+ shrdq $28,%r14,%r14+ addq %r11,%rdx+ addq %rdi,%r11+ movq %rdx,%r13+ addq %r11,%r14+ shrdq $23,%r13,%r13+ movq %r14,%r11+ movq %r8,%r12+ shrdq $5,%r14,%r14+ xorq %rdx,%r13+ xorq %r9,%r12+ shrdq $4,%r13,%r13+ xorq %r11,%r14+ andq %rdx,%r12+ xorq %rdx,%r13+ addq 8(%rsp),%r10+ movq %r11,%rdi+ xorq %r9,%r12+ shrdq $6,%r14,%r14+ xorq %rax,%rdi+ addq %r12,%r10+ shrdq $14,%r13,%r13+ andq %rdi,%r15+ xorq %r11,%r14+ addq %r13,%r10+ xorq %rax,%r15+ shrdq $28,%r14,%r14+ addq %r10,%rcx+ addq %r15,%r10+ movq %rcx,%r13+ addq %r10,%r14+ shrdq $23,%r13,%r13+ movq %r14,%r10+ movq %rdx,%r12+ shrdq $5,%r14,%r14+ xorq %rcx,%r13+ xorq %r8,%r12+ shrdq $4,%r13,%r13+ xorq %r10,%r14+ andq %rcx,%r12+ xorq %rcx,%r13+ addq 16(%rsp),%r9+ movq %r10,%r15+ xorq %r8,%r12+ shrdq $6,%r14,%r14+ xorq %r11,%r15+ addq %r12,%r9+ shrdq $14,%r13,%r13+ andq %r15,%rdi+ xorq %r10,%r14+ addq %r13,%r9+ xorq %r11,%rdi+ shrdq $28,%r14,%r14+ addq %r9,%rbx+ addq %rdi,%r9+ movq %rbx,%r13+ addq %r9,%r14+ shrdq $23,%r13,%r13+ movq %r14,%r9+ movq %rcx,%r12+ shrdq $5,%r14,%r14+ xorq %rbx,%r13+ xorq %rdx,%r12+ shrdq $4,%r13,%r13+ xorq %r9,%r14+ andq %rbx,%r12+ xorq %rbx,%r13+ addq 24(%rsp),%r8+ movq %r9,%rdi+ xorq %rdx,%r12+ shrdq $6,%r14,%r14+ xorq %r10,%rdi+ addq %r12,%r8+ shrdq $14,%r13,%r13+ andq %rdi,%r15+ xorq %r9,%r14+ addq %r13,%r8+ xorq %r10,%r15+ shrdq $28,%r14,%r14+ addq %r8,%rax+ addq %r15,%r8+ movq %rax,%r13+ addq %r8,%r14+ shrdq $23,%r13,%r13+ movq %r14,%r8+ movq %rbx,%r12+ shrdq $5,%r14,%r14+ xorq %rax,%r13+ xorq %rcx,%r12+ shrdq $4,%r13,%r13+ xorq %r8,%r14+ andq %rax,%r12+ xorq %rax,%r13+ addq 32(%rsp),%rdx+ movq %r8,%r15+ xorq %rcx,%r12+ shrdq $6,%r14,%r14+ xorq %r9,%r15+ addq %r12,%rdx+ shrdq $14,%r13,%r13+ andq %r15,%rdi+ xorq %r8,%r14+ addq %r13,%rdx+ xorq %r9,%rdi+ shrdq $28,%r14,%r14+ addq %rdx,%r11+ addq %rdi,%rdx+ movq %r11,%r13+ addq %rdx,%r14+ shrdq $23,%r13,%r13+ movq %r14,%rdx+ movq %rax,%r12+ shrdq $5,%r14,%r14+ xorq %r11,%r13+ xorq %rbx,%r12+ shrdq $4,%r13,%r13+ xorq %rdx,%r14+ andq %r11,%r12+ xorq %r11,%r13+ addq 40(%rsp),%rcx+ movq %rdx,%rdi+ xorq %rbx,%r12+ shrdq $6,%r14,%r14+ xorq %r8,%rdi+ addq %r12,%rcx+ shrdq $14,%r13,%r13+ andq %rdi,%r15+ xorq %rdx,%r14+ addq %r13,%rcx+ xorq %r8,%r15+ shrdq $28,%r14,%r14+ addq %rcx,%r10+ addq %r15,%rcx+ movq %r10,%r13+ addq %rcx,%r14+ shrdq $23,%r13,%r13+ movq %r14,%rcx+ movq %r11,%r12+ shrdq $5,%r14,%r14+ xorq %r10,%r13+ xorq %rax,%r12+ shrdq $4,%r13,%r13+ xorq %rcx,%r14+ andq %r10,%r12+ xorq %r10,%r13+ addq 48(%rsp),%rbx+ movq %rcx,%r15+ xorq %rax,%r12+ shrdq $6,%r14,%r14+ xorq %rdx,%r15+ addq %r12,%rbx+ shrdq $14,%r13,%r13+ andq %r15,%rdi+ xorq %rcx,%r14+ addq %r13,%rbx+ xorq %rdx,%rdi+ shrdq $28,%r14,%r14+ addq %rbx,%r9+ addq %rdi,%rbx+ movq %r9,%r13+ addq %rbx,%r14+ shrdq $23,%r13,%r13+ movq %r14,%rbx+ movq %r10,%r12+ shrdq $5,%r14,%r14+ xorq %r9,%r13+ xorq %r11,%r12+ shrdq $4,%r13,%r13+ xorq %rbx,%r14+ andq %r9,%r12+ xorq %r9,%r13+ addq 56(%rsp),%rax+ movq %rbx,%rdi+ xorq %r11,%r12+ shrdq $6,%r14,%r14+ xorq %rcx,%rdi+ addq %r12,%rax+ shrdq $14,%r13,%r13+ andq %rdi,%r15+ xorq %rbx,%r14+ addq %r13,%rax+ xorq %rcx,%r15+ shrdq $28,%r14,%r14+ addq %rax,%r8+ addq %r15,%rax+ movq %r8,%r13+ addq %rax,%r14+ shrdq $23,%r13,%r13+ movq %r14,%rax+ movq %r9,%r12+ shrdq $5,%r14,%r14+ xorq %r8,%r13+ xorq %r10,%r12+ shrdq $4,%r13,%r13+ xorq %rax,%r14+ andq %r8,%r12+ xorq %r8,%r13+ addq 64(%rsp),%r11+ movq %rax,%r15+ xorq %r10,%r12+ shrdq $6,%r14,%r14+ xorq %rbx,%r15+ addq %r12,%r11+ shrdq $14,%r13,%r13+ andq %r15,%rdi+ xorq %rax,%r14+ addq %r13,%r11+ xorq %rbx,%rdi+ shrdq $28,%r14,%r14+ addq %r11,%rdx+ addq %rdi,%r11+ movq %rdx,%r13+ addq %r11,%r14+ shrdq $23,%r13,%r13+ movq %r14,%r11+ movq %r8,%r12+ shrdq $5,%r14,%r14+ xorq %rdx,%r13+ xorq %r9,%r12+ shrdq $4,%r13,%r13+ xorq %r11,%r14+ andq %rdx,%r12+ xorq %rdx,%r13+ addq 72(%rsp),%r10+ movq %r11,%rdi+ xorq %r9,%r12+ shrdq $6,%r14,%r14+ xorq %rax,%rdi+ addq %r12,%r10+ shrdq $14,%r13,%r13+ andq %rdi,%r15+ xorq %r11,%r14+ addq %r13,%r10+ xorq %rax,%r15+ shrdq $28,%r14,%r14+ addq %r10,%rcx+ addq %r15,%r10+ movq %rcx,%r13+ addq %r10,%r14+ shrdq $23,%r13,%r13+ movq %r14,%r10+ movq %rdx,%r12+ shrdq $5,%r14,%r14+ xorq %rcx,%r13+ xorq %r8,%r12+ shrdq $4,%r13,%r13+ xorq %r10,%r14+ andq %rcx,%r12+ xorq %rcx,%r13+ addq 80(%rsp),%r9+ movq %r10,%r15+ xorq %r8,%r12+ shrdq $6,%r14,%r14+ xorq %r11,%r15+ addq %r12,%r9+ shrdq $14,%r13,%r13+ andq %r15,%rdi+ xorq %r10,%r14+ addq %r13,%r9+ xorq %r11,%rdi+ shrdq $28,%r14,%r14+ addq %r9,%rbx+ addq %rdi,%r9+ movq %rbx,%r13+ addq %r9,%r14+ shrdq $23,%r13,%r13+ movq %r14,%r9+ movq %rcx,%r12+ shrdq $5,%r14,%r14+ xorq %rbx,%r13+ xorq %rdx,%r12+ shrdq $4,%r13,%r13+ xorq %r9,%r14+ andq %rbx,%r12+ xorq %rbx,%r13+ addq 88(%rsp),%r8+ movq %r9,%rdi+ xorq %rdx,%r12+ shrdq $6,%r14,%r14+ xorq %r10,%rdi+ addq %r12,%r8+ shrdq $14,%r13,%r13+ andq %rdi,%r15+ xorq %r9,%r14+ addq %r13,%r8+ xorq %r10,%r15+ shrdq $28,%r14,%r14+ addq %r8,%rax+ addq %r15,%r8+ movq %rax,%r13+ addq %r8,%r14+ shrdq $23,%r13,%r13+ movq %r14,%r8+ movq %rbx,%r12+ shrdq $5,%r14,%r14+ xorq %rax,%r13+ xorq %rcx,%r12+ shrdq $4,%r13,%r13+ xorq %r8,%r14+ andq %rax,%r12+ xorq %rax,%r13+ addq 96(%rsp),%rdx+ movq %r8,%r15+ xorq %rcx,%r12+ shrdq $6,%r14,%r14+ xorq %r9,%r15+ addq %r12,%rdx+ shrdq $14,%r13,%r13+ andq %r15,%rdi+ xorq %r8,%r14+ addq %r13,%rdx+ xorq %r9,%rdi+ shrdq $28,%r14,%r14+ addq %rdx,%r11+ addq %rdi,%rdx+ movq %r11,%r13+ addq %rdx,%r14+ shrdq $23,%r13,%r13+ movq %r14,%rdx+ movq %rax,%r12+ shrdq $5,%r14,%r14+ xorq %r11,%r13+ xorq %rbx,%r12+ shrdq $4,%r13,%r13+ xorq %rdx,%r14+ andq %r11,%r12+ xorq %r11,%r13+ addq 104(%rsp),%rcx+ movq %rdx,%rdi+ xorq %rbx,%r12+ shrdq $6,%r14,%r14+ xorq %r8,%rdi+ addq %r12,%rcx+ shrdq $14,%r13,%r13+ andq %rdi,%r15+ xorq %rdx,%r14+ addq %r13,%rcx+ xorq %r8,%r15+ shrdq $28,%r14,%r14+ addq %rcx,%r10+ addq %r15,%rcx+ movq %r10,%r13+ addq %rcx,%r14+ shrdq $23,%r13,%r13+ movq %r14,%rcx+ movq %r11,%r12+ shrdq $5,%r14,%r14+ xorq %r10,%r13+ xorq %rax,%r12+ shrdq $4,%r13,%r13+ xorq %rcx,%r14+ andq %r10,%r12+ xorq %r10,%r13+ addq 112(%rsp),%rbx+ movq %rcx,%r15+ xorq %rax,%r12+ shrdq $6,%r14,%r14+ xorq %rdx,%r15+ addq %r12,%rbx+ shrdq $14,%r13,%r13+ andq %r15,%rdi+ xorq %rcx,%r14+ addq %r13,%rbx+ xorq %rdx,%rdi+ shrdq $28,%r14,%r14+ addq %rbx,%r9+ addq %rdi,%rbx+ movq %r9,%r13+ addq %rbx,%r14+ shrdq $23,%r13,%r13+ movq %r14,%rbx+ movq %r10,%r12+ shrdq $5,%r14,%r14+ xorq %r9,%r13+ xorq %r11,%r12+ shrdq $4,%r13,%r13+ xorq %rbx,%r14+ andq %r9,%r12+ xorq %r9,%r13+ addq 120(%rsp),%rax+ movq %rbx,%rdi+ xorq %r11,%r12+ shrdq $6,%r14,%r14+ xorq %rcx,%rdi+ addq %r12,%rax+ shrdq $14,%r13,%r13+ andq %rdi,%r15+ xorq %rbx,%r14+ addq %r13,%rax+ xorq %rcx,%r15+ shrdq $28,%r14,%r14+ addq %rax,%r8+ addq %r15,%rax+ movq %r8,%r13+ addq %rax,%r14+ movq -64(%rbp),%rdi+ movq %r14,%rax+ movq -56(%rbp),%rsi++ addq 0(%rdi),%rax+ addq 8(%rdi),%rbx+ addq 16(%rdi),%rcx+ addq 24(%rdi),%rdx+ addq 32(%rdi),%r8+ addq 40(%rdi),%r9+ addq 48(%rdi),%r10+ addq 56(%rdi),%r11++ leaq 128(%rsi),%rsi+ cmpq -48(%rbp),%rsi++ movq %rax,0(%rdi)+ movq %rbx,8(%rdi)+ movq %rcx,16(%rdi)+ movq %rdx,24(%rdi)+ movq %r8,32(%rdi)+ movq %r9,40(%rdi)+ movq %r10,48(%rdi)+ movq %r11,56(%rdi)+ jb .Lloop_avx++ vzeroupper+ movaps -160(%rbp),%xmm6+ movaps -144(%rbp),%xmm7+ movaps -128(%rbp),%xmm8+ movaps -112(%rbp),%xmm9+ movaps -96(%rbp),%xmm10+ movaps -80(%rbp),%xmm11+ movq -40(%rbp),%r15+ movq -32(%rbp),%r14+ movq -24(%rbp),%r13+ movq -16(%rbp),%r12+ movq -8(%rbp),%rbx+ movq %rbp,%rsp++ popq %rbp++.LSEH_epilogue_crypton_sha512_asm_block_data_order_avx:+ mov 8(%rsp),%rdi+ mov 16(%rsp),%rsi++ .byte 0xf3,0xc3++.LSEH_end_crypton_sha512_asm_block_data_order_avx:+.def crypton_sha512_asm_block_data_order_avx2; .scl 3; .type 32; .endef+.p2align 6+crypton_sha512_asm_block_data_order_avx2:+ .byte 0xf3,0x0f,0x1e,0xfa+ movq %rdi,8(%rsp)+ movq %rsi,16(%rsp)+ movq %rsp,%r11+.LSEH_begin_crypton_sha512_asm_block_data_order_avx2:+++ pushq %rbp++ movq %rsp,%rbp++ movq %rcx,%rdi+ movq %rdx,%rsi+ movq %r8,%rdx+.Lavx2_shortcut:+ pushq %rbx++ pushq %r12++ pushq %r13++ pushq %r14++ pushq %r15++ shlq $4,%rdx+ subq $120,%rsp++ leaq (%rsi,%rdx,8),%rdx+ movq %rdi,-64(%rbp)++ movq %rdx,-48(%rbp)+ movaps %xmm6,-160(%rbp)+ movaps %xmm7,-144(%rbp)+ movaps %xmm8,-128(%rbp)+ movaps %xmm9,-112(%rbp)++ movaps %xmm10,-96(%rbp)+ movaps %xmm11,-80(%rbp)++.LSEH_body_crypton_sha512_asm_block_data_order_avx2:+++ leaq -128(%rsp),%rsp+ vzeroupper+ andq $-128,%rsp+ subq $-128,%rsi+ movq 0(%rdi),%rax+ movq %rsi,%r12+ movq 8(%rdi),%rbx+ cmpq %rdx,%rsi+ movq 16(%rdi),%rcx+ cmoveq %rsp,%r12+ movq 24(%rdi),%rdx+ movq 32(%rdi),%r8+ movq 40(%rdi),%r9+ movq 48(%rdi),%r10+ movq 56(%rdi),%r11+ jmp .Loop_avx2+.p2align 4+.Loop_avx2:+ vmovdqa K512+1280(%rip),%ymm10+ movq %rsi,-56(%rbp)+ vmovdqu -128(%rsi),%xmm0+ vmovdqu -128+16(%rsi),%xmm1+ vmovdqu -128+32(%rsi),%xmm2+ vmovdqu -128+48(%rsi),%xmm3+ vmovdqu -128+64(%rsi),%xmm4+ vmovdqu -128+80(%rsi),%xmm5+ vmovdqu -128+96(%rsi),%xmm6+ vmovdqu -128+112(%rsi),%xmm7+ leaq K512+128(%rip),%rsi+ vinserti128 $1,(%r12),%ymm0,%ymm0+ vinserti128 $1,16(%r12),%ymm1,%ymm1+ vpshufb %ymm10,%ymm0,%ymm0+ vinserti128 $1,32(%r12),%ymm2,%ymm2+ vpshufb %ymm10,%ymm1,%ymm1+ vinserti128 $1,48(%r12),%ymm3,%ymm3+ vpshufb %ymm10,%ymm2,%ymm2+ vinserti128 $1,64(%r12),%ymm4,%ymm4+ vpshufb %ymm10,%ymm3,%ymm3+ vinserti128 $1,80(%r12),%ymm5,%ymm5+ vpshufb %ymm10,%ymm4,%ymm4+ vinserti128 $1,96(%r12),%ymm6,%ymm6+ vpshufb %ymm10,%ymm5,%ymm5+ vinserti128 $1,112(%r12),%ymm7,%ymm7++ vpaddq -128(%rsi),%ymm0,%ymm8+ vpshufb %ymm10,%ymm6,%ymm6+ vpaddq -96(%rsi),%ymm1,%ymm9+ vpshufb %ymm10,%ymm7,%ymm7+ vpaddq -64(%rsi),%ymm2,%ymm10+ vpaddq -32(%rsi),%ymm3,%ymm11+ vmovdqa %ymm8,0(%rsp)+ vpaddq 0(%rsi),%ymm4,%ymm8+ vmovdqa %ymm9,32(%rsp)+ vpaddq 32(%rsi),%ymm5,%ymm9+ vmovdqa %ymm10,64(%rsp)+ vpaddq 64(%rsi),%ymm6,%ymm10+ vmovdqa %ymm11,96(%rsp)+ leaq -128(%rsp),%rsp+ vpaddq 96(%rsi),%ymm7,%ymm11+ vmovdqa %ymm8,0(%rsp)+ xorq %r14,%r14+ vmovdqa %ymm9,32(%rsp)+ movq %rbx,%rdi+ vmovdqa %ymm10,64(%rsp)+ xorq %rcx,%rdi+ vmovdqa %ymm11,96(%rsp)+ movq %r9,%r12+ addq $32*8,%rsi+ jmp .Lavx2_00_47++.p2align 4+.Lavx2_00_47:+ leaq -128(%rsp),%rsp+ vpalignr $8,%ymm0,%ymm1,%ymm8+ addq 0+256(%rsp),%r11+ andq %r8,%r12+ rorxq $41,%r8,%r13+ vpalignr $8,%ymm4,%ymm5,%ymm11+ rorxq $18,%r8,%r15+ leaq (%rax,%r14,1),%rax+ leaq (%r11,%r12,1),%r11+ vpsrlq $1,%ymm8,%ymm10+ andnq %r10,%r8,%r12+ xorq %r15,%r13+ rorxq $14,%r8,%r14+ vpaddq %ymm11,%ymm0,%ymm0+ vpsrlq $7,%ymm8,%ymm11+ leaq (%r11,%r12,1),%r11+ xorq %r14,%r13+ movq %rax,%r15+ vpsllq $56,%ymm8,%ymm9+ vpxor %ymm10,%ymm11,%ymm8+ rorxq $39,%rax,%r12+ leaq (%r11,%r13,1),%r11+ xorq %rbx,%r15+ vpsrlq $7,%ymm10,%ymm10+ vpxor %ymm9,%ymm8,%ymm8+ rorxq $34,%rax,%r14+ rorxq $28,%rax,%r13+ leaq (%rdx,%r11,1),%rdx+ vpsllq $7,%ymm9,%ymm9+ vpxor %ymm10,%ymm8,%ymm8+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %rbx,%rdi+ vpsrlq $6,%ymm7,%ymm11+ vpxor %ymm9,%ymm8,%ymm8+ xorq %r13,%r14+ leaq (%r11,%rdi,1),%r11+ movq %r8,%r12+ vpsllq $3,%ymm7,%ymm10+ vpaddq %ymm8,%ymm0,%ymm0+ addq 8+256(%rsp),%r10+ andq %rdx,%r12+ rorxq $41,%rdx,%r13+ vpsrlq $19,%ymm7,%ymm9+ vpxor %ymm10,%ymm11,%ymm11+ rorxq $18,%rdx,%rdi+ leaq (%r11,%r14,1),%r11+ leaq (%r10,%r12,1),%r10+ vpsllq $42,%ymm10,%ymm10+ vpxor %ymm9,%ymm11,%ymm11+ andnq %r9,%rdx,%r12+ xorq %rdi,%r13+ rorxq $14,%rdx,%r14+ vpsrlq $42,%ymm9,%ymm9+ vpxor %ymm10,%ymm11,%ymm11+ leaq (%r10,%r12,1),%r10+ xorq %r14,%r13+ movq %r11,%rdi+ vpxor %ymm9,%ymm11,%ymm11+ rorxq $39,%r11,%r12+ leaq (%r10,%r13,1),%r10+ xorq %rax,%rdi+ vpaddq %ymm11,%ymm0,%ymm0+ rorxq $34,%r11,%r14+ rorxq $28,%r11,%r13+ leaq (%rcx,%r10,1),%rcx+ vpaddq -128(%rsi),%ymm0,%ymm10+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %rax,%r15+ xorq %r13,%r14+ leaq (%r10,%r15,1),%r10+ movq %rdx,%r12+ vmovdqa %ymm10,0(%rsp)+ vpalignr $8,%ymm1,%ymm2,%ymm8+ addq 32+256(%rsp),%r9+ andq %rcx,%r12+ rorxq $41,%rcx,%r13+ vpalignr $8,%ymm5,%ymm6,%ymm11+ rorxq $18,%rcx,%r15+ leaq (%r10,%r14,1),%r10+ leaq (%r9,%r12,1),%r9+ vpsrlq $1,%ymm8,%ymm10+ andnq %r8,%rcx,%r12+ xorq %r15,%r13+ rorxq $14,%rcx,%r14+ vpaddq %ymm11,%ymm1,%ymm1+ vpsrlq $7,%ymm8,%ymm11+ leaq (%r9,%r12,1),%r9+ xorq %r14,%r13+ movq %r10,%r15+ vpsllq $56,%ymm8,%ymm9+ vpxor %ymm10,%ymm11,%ymm8+ rorxq $39,%r10,%r12+ leaq (%r9,%r13,1),%r9+ xorq %r11,%r15+ vpsrlq $7,%ymm10,%ymm10+ vpxor %ymm9,%ymm8,%ymm8+ rorxq $34,%r10,%r14+ rorxq $28,%r10,%r13+ leaq (%rbx,%r9,1),%rbx+ vpsllq $7,%ymm9,%ymm9+ vpxor %ymm10,%ymm8,%ymm8+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %r11,%rdi+ vpsrlq $6,%ymm0,%ymm11+ vpxor %ymm9,%ymm8,%ymm8+ xorq %r13,%r14+ leaq (%r9,%rdi,1),%r9+ movq %rcx,%r12+ vpsllq $3,%ymm0,%ymm10+ vpaddq %ymm8,%ymm1,%ymm1+ addq 40+256(%rsp),%r8+ andq %rbx,%r12+ rorxq $41,%rbx,%r13+ vpsrlq $19,%ymm0,%ymm9+ vpxor %ymm10,%ymm11,%ymm11+ rorxq $18,%rbx,%rdi+ leaq (%r9,%r14,1),%r9+ leaq (%r8,%r12,1),%r8+ vpsllq $42,%ymm10,%ymm10+ vpxor %ymm9,%ymm11,%ymm11+ andnq %rdx,%rbx,%r12+ xorq %rdi,%r13+ rorxq $14,%rbx,%r14+ vpsrlq $42,%ymm9,%ymm9+ vpxor %ymm10,%ymm11,%ymm11+ leaq (%r8,%r12,1),%r8+ xorq %r14,%r13+ movq %r9,%rdi+ vpxor %ymm9,%ymm11,%ymm11+ rorxq $39,%r9,%r12+ leaq (%r8,%r13,1),%r8+ xorq %r10,%rdi+ vpaddq %ymm11,%ymm1,%ymm1+ rorxq $34,%r9,%r14+ rorxq $28,%r9,%r13+ leaq (%rax,%r8,1),%rax+ vpaddq -96(%rsi),%ymm1,%ymm10+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %r10,%r15+ xorq %r13,%r14+ leaq (%r8,%r15,1),%r8+ movq %rbx,%r12+ vmovdqa %ymm10,32(%rsp)+ vpalignr $8,%ymm2,%ymm3,%ymm8+ addq 64+256(%rsp),%rdx+ andq %rax,%r12+ rorxq $41,%rax,%r13+ vpalignr $8,%ymm6,%ymm7,%ymm11+ rorxq $18,%rax,%r15+ leaq (%r8,%r14,1),%r8+ leaq (%rdx,%r12,1),%rdx+ vpsrlq $1,%ymm8,%ymm10+ andnq %rcx,%rax,%r12+ xorq %r15,%r13+ rorxq $14,%rax,%r14+ vpaddq %ymm11,%ymm2,%ymm2+ vpsrlq $7,%ymm8,%ymm11+ leaq (%rdx,%r12,1),%rdx+ xorq %r14,%r13+ movq %r8,%r15+ vpsllq $56,%ymm8,%ymm9+ vpxor %ymm10,%ymm11,%ymm8+ rorxq $39,%r8,%r12+ leaq (%rdx,%r13,1),%rdx+ xorq %r9,%r15+ vpsrlq $7,%ymm10,%ymm10+ vpxor %ymm9,%ymm8,%ymm8+ rorxq $34,%r8,%r14+ rorxq $28,%r8,%r13+ leaq (%r11,%rdx,1),%r11+ vpsllq $7,%ymm9,%ymm9+ vpxor %ymm10,%ymm8,%ymm8+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %r9,%rdi+ vpsrlq $6,%ymm1,%ymm11+ vpxor %ymm9,%ymm8,%ymm8+ xorq %r13,%r14+ leaq (%rdx,%rdi,1),%rdx+ movq %rax,%r12+ vpsllq $3,%ymm1,%ymm10+ vpaddq %ymm8,%ymm2,%ymm2+ addq 72+256(%rsp),%rcx+ andq %r11,%r12+ rorxq $41,%r11,%r13+ vpsrlq $19,%ymm1,%ymm9+ vpxor %ymm10,%ymm11,%ymm11+ rorxq $18,%r11,%rdi+ leaq (%rdx,%r14,1),%rdx+ leaq (%rcx,%r12,1),%rcx+ vpsllq $42,%ymm10,%ymm10+ vpxor %ymm9,%ymm11,%ymm11+ andnq %rbx,%r11,%r12+ xorq %rdi,%r13+ rorxq $14,%r11,%r14+ vpsrlq $42,%ymm9,%ymm9+ vpxor %ymm10,%ymm11,%ymm11+ leaq (%rcx,%r12,1),%rcx+ xorq %r14,%r13+ movq %rdx,%rdi+ vpxor %ymm9,%ymm11,%ymm11+ rorxq $39,%rdx,%r12+ leaq (%rcx,%r13,1),%rcx+ xorq %r8,%rdi+ vpaddq %ymm11,%ymm2,%ymm2+ rorxq $34,%rdx,%r14+ rorxq $28,%rdx,%r13+ leaq (%r10,%rcx,1),%r10+ vpaddq -64(%rsi),%ymm2,%ymm10+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %r8,%r15+ xorq %r13,%r14+ leaq (%rcx,%r15,1),%rcx+ movq %r11,%r12+ vmovdqa %ymm10,64(%rsp)+ vpalignr $8,%ymm3,%ymm4,%ymm8+ addq 96+256(%rsp),%rbx+ andq %r10,%r12+ rorxq $41,%r10,%r13+ vpalignr $8,%ymm7,%ymm0,%ymm11+ rorxq $18,%r10,%r15+ leaq (%rcx,%r14,1),%rcx+ leaq (%rbx,%r12,1),%rbx+ vpsrlq $1,%ymm8,%ymm10+ andnq %rax,%r10,%r12+ xorq %r15,%r13+ rorxq $14,%r10,%r14+ vpaddq %ymm11,%ymm3,%ymm3+ vpsrlq $7,%ymm8,%ymm11+ leaq (%rbx,%r12,1),%rbx+ xorq %r14,%r13+ movq %rcx,%r15+ vpsllq $56,%ymm8,%ymm9+ vpxor %ymm10,%ymm11,%ymm8+ rorxq $39,%rcx,%r12+ leaq (%rbx,%r13,1),%rbx+ xorq %rdx,%r15+ vpsrlq $7,%ymm10,%ymm10+ vpxor %ymm9,%ymm8,%ymm8+ rorxq $34,%rcx,%r14+ rorxq $28,%rcx,%r13+ leaq (%r9,%rbx,1),%r9+ vpsllq $7,%ymm9,%ymm9+ vpxor %ymm10,%ymm8,%ymm8+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %rdx,%rdi+ vpsrlq $6,%ymm2,%ymm11+ vpxor %ymm9,%ymm8,%ymm8+ xorq %r13,%r14+ leaq (%rbx,%rdi,1),%rbx+ movq %r10,%r12+ vpsllq $3,%ymm2,%ymm10+ vpaddq %ymm8,%ymm3,%ymm3+ addq 104+256(%rsp),%rax+ andq %r9,%r12+ rorxq $41,%r9,%r13+ vpsrlq $19,%ymm2,%ymm9+ vpxor %ymm10,%ymm11,%ymm11+ rorxq $18,%r9,%rdi+ leaq (%rbx,%r14,1),%rbx+ leaq (%rax,%r12,1),%rax+ vpsllq $42,%ymm10,%ymm10+ vpxor %ymm9,%ymm11,%ymm11+ andnq %r11,%r9,%r12+ xorq %rdi,%r13+ rorxq $14,%r9,%r14+ vpsrlq $42,%ymm9,%ymm9+ vpxor %ymm10,%ymm11,%ymm11+ leaq (%rax,%r12,1),%rax+ xorq %r14,%r13+ movq %rbx,%rdi+ vpxor %ymm9,%ymm11,%ymm11+ rorxq $39,%rbx,%r12+ leaq (%rax,%r13,1),%rax+ xorq %rcx,%rdi+ vpaddq %ymm11,%ymm3,%ymm3+ rorxq $34,%rbx,%r14+ rorxq $28,%rbx,%r13+ leaq (%r8,%rax,1),%r8+ vpaddq -32(%rsi),%ymm3,%ymm10+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %rcx,%r15+ xorq %r13,%r14+ leaq (%rax,%r15,1),%rax+ movq %r9,%r12+ vmovdqa %ymm10,96(%rsp)+ leaq -128(%rsp),%rsp+ vpalignr $8,%ymm4,%ymm5,%ymm8+ addq 0+256(%rsp),%r11+ andq %r8,%r12+ rorxq $41,%r8,%r13+ vpalignr $8,%ymm0,%ymm1,%ymm11+ rorxq $18,%r8,%r15+ leaq (%rax,%r14,1),%rax+ leaq (%r11,%r12,1),%r11+ vpsrlq $1,%ymm8,%ymm10+ andnq %r10,%r8,%r12+ xorq %r15,%r13+ rorxq $14,%r8,%r14+ vpaddq %ymm11,%ymm4,%ymm4+ vpsrlq $7,%ymm8,%ymm11+ leaq (%r11,%r12,1),%r11+ xorq %r14,%r13+ movq %rax,%r15+ vpsllq $56,%ymm8,%ymm9+ vpxor %ymm10,%ymm11,%ymm8+ rorxq $39,%rax,%r12+ leaq (%r11,%r13,1),%r11+ xorq %rbx,%r15+ vpsrlq $7,%ymm10,%ymm10+ vpxor %ymm9,%ymm8,%ymm8+ rorxq $34,%rax,%r14+ rorxq $28,%rax,%r13+ leaq (%rdx,%r11,1),%rdx+ vpsllq $7,%ymm9,%ymm9+ vpxor %ymm10,%ymm8,%ymm8+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %rbx,%rdi+ vpsrlq $6,%ymm3,%ymm11+ vpxor %ymm9,%ymm8,%ymm8+ xorq %r13,%r14+ leaq (%r11,%rdi,1),%r11+ movq %r8,%r12+ vpsllq $3,%ymm3,%ymm10+ vpaddq %ymm8,%ymm4,%ymm4+ addq 8+256(%rsp),%r10+ andq %rdx,%r12+ rorxq $41,%rdx,%r13+ vpsrlq $19,%ymm3,%ymm9+ vpxor %ymm10,%ymm11,%ymm11+ rorxq $18,%rdx,%rdi+ leaq (%r11,%r14,1),%r11+ leaq (%r10,%r12,1),%r10+ vpsllq $42,%ymm10,%ymm10+ vpxor %ymm9,%ymm11,%ymm11+ andnq %r9,%rdx,%r12+ xorq %rdi,%r13+ rorxq $14,%rdx,%r14+ vpsrlq $42,%ymm9,%ymm9+ vpxor %ymm10,%ymm11,%ymm11+ leaq (%r10,%r12,1),%r10+ xorq %r14,%r13+ movq %r11,%rdi+ vpxor %ymm9,%ymm11,%ymm11+ rorxq $39,%r11,%r12+ leaq (%r10,%r13,1),%r10+ xorq %rax,%rdi+ vpaddq %ymm11,%ymm4,%ymm4+ rorxq $34,%r11,%r14+ rorxq $28,%r11,%r13+ leaq (%rcx,%r10,1),%rcx+ vpaddq 0(%rsi),%ymm4,%ymm10+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %rax,%r15+ xorq %r13,%r14+ leaq (%r10,%r15,1),%r10+ movq %rdx,%r12+ vmovdqa %ymm10,0(%rsp)+ vpalignr $8,%ymm5,%ymm6,%ymm8+ addq 32+256(%rsp),%r9+ andq %rcx,%r12+ rorxq $41,%rcx,%r13+ vpalignr $8,%ymm1,%ymm2,%ymm11+ rorxq $18,%rcx,%r15+ leaq (%r10,%r14,1),%r10+ leaq (%r9,%r12,1),%r9+ vpsrlq $1,%ymm8,%ymm10+ andnq %r8,%rcx,%r12+ xorq %r15,%r13+ rorxq $14,%rcx,%r14+ vpaddq %ymm11,%ymm5,%ymm5+ vpsrlq $7,%ymm8,%ymm11+ leaq (%r9,%r12,1),%r9+ xorq %r14,%r13+ movq %r10,%r15+ vpsllq $56,%ymm8,%ymm9+ vpxor %ymm10,%ymm11,%ymm8+ rorxq $39,%r10,%r12+ leaq (%r9,%r13,1),%r9+ xorq %r11,%r15+ vpsrlq $7,%ymm10,%ymm10+ vpxor %ymm9,%ymm8,%ymm8+ rorxq $34,%r10,%r14+ rorxq $28,%r10,%r13+ leaq (%rbx,%r9,1),%rbx+ vpsllq $7,%ymm9,%ymm9+ vpxor %ymm10,%ymm8,%ymm8+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %r11,%rdi+ vpsrlq $6,%ymm4,%ymm11+ vpxor %ymm9,%ymm8,%ymm8+ xorq %r13,%r14+ leaq (%r9,%rdi,1),%r9+ movq %rcx,%r12+ vpsllq $3,%ymm4,%ymm10+ vpaddq %ymm8,%ymm5,%ymm5+ addq 40+256(%rsp),%r8+ andq %rbx,%r12+ rorxq $41,%rbx,%r13+ vpsrlq $19,%ymm4,%ymm9+ vpxor %ymm10,%ymm11,%ymm11+ rorxq $18,%rbx,%rdi+ leaq (%r9,%r14,1),%r9+ leaq (%r8,%r12,1),%r8+ vpsllq $42,%ymm10,%ymm10+ vpxor %ymm9,%ymm11,%ymm11+ andnq %rdx,%rbx,%r12+ xorq %rdi,%r13+ rorxq $14,%rbx,%r14+ vpsrlq $42,%ymm9,%ymm9+ vpxor %ymm10,%ymm11,%ymm11+ leaq (%r8,%r12,1),%r8+ xorq %r14,%r13+ movq %r9,%rdi+ vpxor %ymm9,%ymm11,%ymm11+ rorxq $39,%r9,%r12+ leaq (%r8,%r13,1),%r8+ xorq %r10,%rdi+ vpaddq %ymm11,%ymm5,%ymm5+ rorxq $34,%r9,%r14+ rorxq $28,%r9,%r13+ leaq (%rax,%r8,1),%rax+ vpaddq 32(%rsi),%ymm5,%ymm10+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %r10,%r15+ xorq %r13,%r14+ leaq (%r8,%r15,1),%r8+ movq %rbx,%r12+ vmovdqa %ymm10,32(%rsp)+ vpalignr $8,%ymm6,%ymm7,%ymm8+ addq 64+256(%rsp),%rdx+ andq %rax,%r12+ rorxq $41,%rax,%r13+ vpalignr $8,%ymm2,%ymm3,%ymm11+ rorxq $18,%rax,%r15+ leaq (%r8,%r14,1),%r8+ leaq (%rdx,%r12,1),%rdx+ vpsrlq $1,%ymm8,%ymm10+ andnq %rcx,%rax,%r12+ xorq %r15,%r13+ rorxq $14,%rax,%r14+ vpaddq %ymm11,%ymm6,%ymm6+ vpsrlq $7,%ymm8,%ymm11+ leaq (%rdx,%r12,1),%rdx+ xorq %r14,%r13+ movq %r8,%r15+ vpsllq $56,%ymm8,%ymm9+ vpxor %ymm10,%ymm11,%ymm8+ rorxq $39,%r8,%r12+ leaq (%rdx,%r13,1),%rdx+ xorq %r9,%r15+ vpsrlq $7,%ymm10,%ymm10+ vpxor %ymm9,%ymm8,%ymm8+ rorxq $34,%r8,%r14+ rorxq $28,%r8,%r13+ leaq (%r11,%rdx,1),%r11+ vpsllq $7,%ymm9,%ymm9+ vpxor %ymm10,%ymm8,%ymm8+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %r9,%rdi+ vpsrlq $6,%ymm5,%ymm11+ vpxor %ymm9,%ymm8,%ymm8+ xorq %r13,%r14+ leaq (%rdx,%rdi,1),%rdx+ movq %rax,%r12+ vpsllq $3,%ymm5,%ymm10+ vpaddq %ymm8,%ymm6,%ymm6+ addq 72+256(%rsp),%rcx+ andq %r11,%r12+ rorxq $41,%r11,%r13+ vpsrlq $19,%ymm5,%ymm9+ vpxor %ymm10,%ymm11,%ymm11+ rorxq $18,%r11,%rdi+ leaq (%rdx,%r14,1),%rdx+ leaq (%rcx,%r12,1),%rcx+ vpsllq $42,%ymm10,%ymm10+ vpxor %ymm9,%ymm11,%ymm11+ andnq %rbx,%r11,%r12+ xorq %rdi,%r13+ rorxq $14,%r11,%r14+ vpsrlq $42,%ymm9,%ymm9+ vpxor %ymm10,%ymm11,%ymm11+ leaq (%rcx,%r12,1),%rcx+ xorq %r14,%r13+ movq %rdx,%rdi+ vpxor %ymm9,%ymm11,%ymm11+ rorxq $39,%rdx,%r12+ leaq (%rcx,%r13,1),%rcx+ xorq %r8,%rdi+ vpaddq %ymm11,%ymm6,%ymm6+ rorxq $34,%rdx,%r14+ rorxq $28,%rdx,%r13+ leaq (%r10,%rcx,1),%r10+ vpaddq 64(%rsi),%ymm6,%ymm10+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %r8,%r15+ xorq %r13,%r14+ leaq (%rcx,%r15,1),%rcx+ movq %r11,%r12+ vmovdqa %ymm10,64(%rsp)+ vpalignr $8,%ymm7,%ymm0,%ymm8+ addq 96+256(%rsp),%rbx+ andq %r10,%r12+ rorxq $41,%r10,%r13+ vpalignr $8,%ymm3,%ymm4,%ymm11+ rorxq $18,%r10,%r15+ leaq (%rcx,%r14,1),%rcx+ leaq (%rbx,%r12,1),%rbx+ vpsrlq $1,%ymm8,%ymm10+ andnq %rax,%r10,%r12+ xorq %r15,%r13+ rorxq $14,%r10,%r14+ vpaddq %ymm11,%ymm7,%ymm7+ vpsrlq $7,%ymm8,%ymm11+ leaq (%rbx,%r12,1),%rbx+ xorq %r14,%r13+ movq %rcx,%r15+ vpsllq $56,%ymm8,%ymm9+ vpxor %ymm10,%ymm11,%ymm8+ rorxq $39,%rcx,%r12+ leaq (%rbx,%r13,1),%rbx+ xorq %rdx,%r15+ vpsrlq $7,%ymm10,%ymm10+ vpxor %ymm9,%ymm8,%ymm8+ rorxq $34,%rcx,%r14+ rorxq $28,%rcx,%r13+ leaq (%r9,%rbx,1),%r9+ vpsllq $7,%ymm9,%ymm9+ vpxor %ymm10,%ymm8,%ymm8+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %rdx,%rdi+ vpsrlq $6,%ymm6,%ymm11+ vpxor %ymm9,%ymm8,%ymm8+ xorq %r13,%r14+ leaq (%rbx,%rdi,1),%rbx+ movq %r10,%r12+ vpsllq $3,%ymm6,%ymm10+ vpaddq %ymm8,%ymm7,%ymm7+ addq 104+256(%rsp),%rax+ andq %r9,%r12+ rorxq $41,%r9,%r13+ vpsrlq $19,%ymm6,%ymm9+ vpxor %ymm10,%ymm11,%ymm11+ rorxq $18,%r9,%rdi+ leaq (%rbx,%r14,1),%rbx+ leaq (%rax,%r12,1),%rax+ vpsllq $42,%ymm10,%ymm10+ vpxor %ymm9,%ymm11,%ymm11+ andnq %r11,%r9,%r12+ xorq %rdi,%r13+ rorxq $14,%r9,%r14+ vpsrlq $42,%ymm9,%ymm9+ vpxor %ymm10,%ymm11,%ymm11+ leaq (%rax,%r12,1),%rax+ xorq %r14,%r13+ movq %rbx,%rdi+ vpxor %ymm9,%ymm11,%ymm11+ rorxq $39,%rbx,%r12+ leaq (%rax,%r13,1),%rax+ xorq %rcx,%rdi+ vpaddq %ymm11,%ymm7,%ymm7+ rorxq $34,%rbx,%r14+ rorxq $28,%rbx,%r13+ leaq (%r8,%rax,1),%r8+ vpaddq 96(%rsi),%ymm7,%ymm10+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %rcx,%r15+ xorq %r13,%r14+ leaq (%rax,%r15,1),%rax+ movq %r9,%r12+ vmovdqa %ymm10,96(%rsp)+ leaq 256(%rsi),%rsi+ cmpb $0,-121(%rsi)+ jne .Lavx2_00_47+ addq 0+128(%rsp),%r11+ andq %r8,%r12+ rorxq $41,%r8,%r13+ rorxq $18,%r8,%r15+ leaq (%rax,%r14,1),%rax+ leaq (%r11,%r12,1),%r11+ andnq %r10,%r8,%r12+ xorq %r15,%r13+ rorxq $14,%r8,%r14+ leaq (%r11,%r12,1),%r11+ xorq %r14,%r13+ movq %rax,%r15+ rorxq $39,%rax,%r12+ leaq (%r11,%r13,1),%r11+ xorq %rbx,%r15+ rorxq $34,%rax,%r14+ rorxq $28,%rax,%r13+ leaq (%rdx,%r11,1),%rdx+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %rbx,%rdi+ xorq %r13,%r14+ leaq (%r11,%rdi,1),%r11+ movq %r8,%r12+ addq 8+128(%rsp),%r10+ andq %rdx,%r12+ rorxq $41,%rdx,%r13+ rorxq $18,%rdx,%rdi+ leaq (%r11,%r14,1),%r11+ leaq (%r10,%r12,1),%r10+ andnq %r9,%rdx,%r12+ xorq %rdi,%r13+ rorxq $14,%rdx,%r14+ leaq (%r10,%r12,1),%r10+ xorq %r14,%r13+ movq %r11,%rdi+ rorxq $39,%r11,%r12+ leaq (%r10,%r13,1),%r10+ xorq %rax,%rdi+ rorxq $34,%r11,%r14+ rorxq $28,%r11,%r13+ leaq (%rcx,%r10,1),%rcx+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %rax,%r15+ xorq %r13,%r14+ leaq (%r10,%r15,1),%r10+ movq %rdx,%r12+ addq 32+128(%rsp),%r9+ andq %rcx,%r12+ rorxq $41,%rcx,%r13+ rorxq $18,%rcx,%r15+ leaq (%r10,%r14,1),%r10+ leaq (%r9,%r12,1),%r9+ andnq %r8,%rcx,%r12+ xorq %r15,%r13+ rorxq $14,%rcx,%r14+ leaq (%r9,%r12,1),%r9+ xorq %r14,%r13+ movq %r10,%r15+ rorxq $39,%r10,%r12+ leaq (%r9,%r13,1),%r9+ xorq %r11,%r15+ rorxq $34,%r10,%r14+ rorxq $28,%r10,%r13+ leaq (%rbx,%r9,1),%rbx+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %r11,%rdi+ xorq %r13,%r14+ leaq (%r9,%rdi,1),%r9+ movq %rcx,%r12+ addq 40+128(%rsp),%r8+ andq %rbx,%r12+ rorxq $41,%rbx,%r13+ rorxq $18,%rbx,%rdi+ leaq (%r9,%r14,1),%r9+ leaq (%r8,%r12,1),%r8+ andnq %rdx,%rbx,%r12+ xorq %rdi,%r13+ rorxq $14,%rbx,%r14+ leaq (%r8,%r12,1),%r8+ xorq %r14,%r13+ movq %r9,%rdi+ rorxq $39,%r9,%r12+ leaq (%r8,%r13,1),%r8+ xorq %r10,%rdi+ rorxq $34,%r9,%r14+ rorxq $28,%r9,%r13+ leaq (%rax,%r8,1),%rax+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %r10,%r15+ xorq %r13,%r14+ leaq (%r8,%r15,1),%r8+ movq %rbx,%r12+ addq 64+128(%rsp),%rdx+ andq %rax,%r12+ rorxq $41,%rax,%r13+ rorxq $18,%rax,%r15+ leaq (%r8,%r14,1),%r8+ leaq (%rdx,%r12,1),%rdx+ andnq %rcx,%rax,%r12+ xorq %r15,%r13+ rorxq $14,%rax,%r14+ leaq (%rdx,%r12,1),%rdx+ xorq %r14,%r13+ movq %r8,%r15+ rorxq $39,%r8,%r12+ leaq (%rdx,%r13,1),%rdx+ xorq %r9,%r15+ rorxq $34,%r8,%r14+ rorxq $28,%r8,%r13+ leaq (%r11,%rdx,1),%r11+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %r9,%rdi+ xorq %r13,%r14+ leaq (%rdx,%rdi,1),%rdx+ movq %rax,%r12+ addq 72+128(%rsp),%rcx+ andq %r11,%r12+ rorxq $41,%r11,%r13+ rorxq $18,%r11,%rdi+ leaq (%rdx,%r14,1),%rdx+ leaq (%rcx,%r12,1),%rcx+ andnq %rbx,%r11,%r12+ xorq %rdi,%r13+ rorxq $14,%r11,%r14+ leaq (%rcx,%r12,1),%rcx+ xorq %r14,%r13+ movq %rdx,%rdi+ rorxq $39,%rdx,%r12+ leaq (%rcx,%r13,1),%rcx+ xorq %r8,%rdi+ rorxq $34,%rdx,%r14+ rorxq $28,%rdx,%r13+ leaq (%r10,%rcx,1),%r10+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %r8,%r15+ xorq %r13,%r14+ leaq (%rcx,%r15,1),%rcx+ movq %r11,%r12+ addq 96+128(%rsp),%rbx+ andq %r10,%r12+ rorxq $41,%r10,%r13+ rorxq $18,%r10,%r15+ leaq (%rcx,%r14,1),%rcx+ leaq (%rbx,%r12,1),%rbx+ andnq %rax,%r10,%r12+ xorq %r15,%r13+ rorxq $14,%r10,%r14+ leaq (%rbx,%r12,1),%rbx+ xorq %r14,%r13+ movq %rcx,%r15+ rorxq $39,%rcx,%r12+ leaq (%rbx,%r13,1),%rbx+ xorq %rdx,%r15+ rorxq $34,%rcx,%r14+ rorxq $28,%rcx,%r13+ leaq (%r9,%rbx,1),%r9+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %rdx,%rdi+ xorq %r13,%r14+ leaq (%rbx,%rdi,1),%rbx+ movq %r10,%r12+ addq 104+128(%rsp),%rax+ andq %r9,%r12+ rorxq $41,%r9,%r13+ rorxq $18,%r9,%rdi+ leaq (%rbx,%r14,1),%rbx+ leaq (%rax,%r12,1),%rax+ andnq %r11,%r9,%r12+ xorq %rdi,%r13+ rorxq $14,%r9,%r14+ leaq (%rax,%r12,1),%rax+ xorq %r14,%r13+ movq %rbx,%rdi+ rorxq $39,%rbx,%r12+ leaq (%rax,%r13,1),%rax+ xorq %rcx,%rdi+ rorxq $34,%rbx,%r14+ rorxq $28,%rbx,%r13+ leaq (%r8,%rax,1),%r8+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %rcx,%r15+ xorq %r13,%r14+ leaq (%rax,%r15,1),%rax+ movq %r9,%r12+ addq 0(%rsp),%r11+ andq %r8,%r12+ rorxq $41,%r8,%r13+ rorxq $18,%r8,%r15+ leaq (%rax,%r14,1),%rax+ leaq (%r11,%r12,1),%r11+ andnq %r10,%r8,%r12+ xorq %r15,%r13+ rorxq $14,%r8,%r14+ leaq (%r11,%r12,1),%r11+ xorq %r14,%r13+ movq %rax,%r15+ rorxq $39,%rax,%r12+ leaq (%r11,%r13,1),%r11+ xorq %rbx,%r15+ rorxq $34,%rax,%r14+ rorxq $28,%rax,%r13+ leaq (%rdx,%r11,1),%rdx+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %rbx,%rdi+ xorq %r13,%r14+ leaq (%r11,%rdi,1),%r11+ movq %r8,%r12+ addq 8(%rsp),%r10+ andq %rdx,%r12+ rorxq $41,%rdx,%r13+ rorxq $18,%rdx,%rdi+ leaq (%r11,%r14,1),%r11+ leaq (%r10,%r12,1),%r10+ andnq %r9,%rdx,%r12+ xorq %rdi,%r13+ rorxq $14,%rdx,%r14+ leaq (%r10,%r12,1),%r10+ xorq %r14,%r13+ movq %r11,%rdi+ rorxq $39,%r11,%r12+ leaq (%r10,%r13,1),%r10+ xorq %rax,%rdi+ rorxq $34,%r11,%r14+ rorxq $28,%r11,%r13+ leaq (%rcx,%r10,1),%rcx+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %rax,%r15+ xorq %r13,%r14+ leaq (%r10,%r15,1),%r10+ movq %rdx,%r12+ addq 32(%rsp),%r9+ andq %rcx,%r12+ rorxq $41,%rcx,%r13+ rorxq $18,%rcx,%r15+ leaq (%r10,%r14,1),%r10+ leaq (%r9,%r12,1),%r9+ andnq %r8,%rcx,%r12+ xorq %r15,%r13+ rorxq $14,%rcx,%r14+ leaq (%r9,%r12,1),%r9+ xorq %r14,%r13+ movq %r10,%r15+ rorxq $39,%r10,%r12+ leaq (%r9,%r13,1),%r9+ xorq %r11,%r15+ rorxq $34,%r10,%r14+ rorxq $28,%r10,%r13+ leaq (%rbx,%r9,1),%rbx+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %r11,%rdi+ xorq %r13,%r14+ leaq (%r9,%rdi,1),%r9+ movq %rcx,%r12+ addq 40(%rsp),%r8+ andq %rbx,%r12+ rorxq $41,%rbx,%r13+ rorxq $18,%rbx,%rdi+ leaq (%r9,%r14,1),%r9+ leaq (%r8,%r12,1),%r8+ andnq %rdx,%rbx,%r12+ xorq %rdi,%r13+ rorxq $14,%rbx,%r14+ leaq (%r8,%r12,1),%r8+ xorq %r14,%r13+ movq %r9,%rdi+ rorxq $39,%r9,%r12+ leaq (%r8,%r13,1),%r8+ xorq %r10,%rdi+ rorxq $34,%r9,%r14+ rorxq $28,%r9,%r13+ leaq (%rax,%r8,1),%rax+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %r10,%r15+ xorq %r13,%r14+ leaq (%r8,%r15,1),%r8+ movq %rbx,%r12+ addq 64(%rsp),%rdx+ andq %rax,%r12+ rorxq $41,%rax,%r13+ rorxq $18,%rax,%r15+ leaq (%r8,%r14,1),%r8+ leaq (%rdx,%r12,1),%rdx+ andnq %rcx,%rax,%r12+ xorq %r15,%r13+ rorxq $14,%rax,%r14+ leaq (%rdx,%r12,1),%rdx+ xorq %r14,%r13+ movq %r8,%r15+ rorxq $39,%r8,%r12+ leaq (%rdx,%r13,1),%rdx+ xorq %r9,%r15+ rorxq $34,%r8,%r14+ rorxq $28,%r8,%r13+ leaq (%r11,%rdx,1),%r11+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %r9,%rdi+ xorq %r13,%r14+ leaq (%rdx,%rdi,1),%rdx+ movq %rax,%r12+ addq 72(%rsp),%rcx+ andq %r11,%r12+ rorxq $41,%r11,%r13+ rorxq $18,%r11,%rdi+ leaq (%rdx,%r14,1),%rdx+ leaq (%rcx,%r12,1),%rcx+ andnq %rbx,%r11,%r12+ xorq %rdi,%r13+ rorxq $14,%r11,%r14+ leaq (%rcx,%r12,1),%rcx+ xorq %r14,%r13+ movq %rdx,%rdi+ rorxq $39,%rdx,%r12+ leaq (%rcx,%r13,1),%rcx+ xorq %r8,%rdi+ rorxq $34,%rdx,%r14+ rorxq $28,%rdx,%r13+ leaq (%r10,%rcx,1),%r10+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %r8,%r15+ xorq %r13,%r14+ leaq (%rcx,%r15,1),%rcx+ movq %r11,%r12+ addq 96(%rsp),%rbx+ andq %r10,%r12+ rorxq $41,%r10,%r13+ rorxq $18,%r10,%r15+ leaq (%rcx,%r14,1),%rcx+ leaq (%rbx,%r12,1),%rbx+ andnq %rax,%r10,%r12+ xorq %r15,%r13+ rorxq $14,%r10,%r14+ leaq (%rbx,%r12,1),%rbx+ xorq %r14,%r13+ movq %rcx,%r15+ rorxq $39,%rcx,%r12+ leaq (%rbx,%r13,1),%rbx+ xorq %rdx,%r15+ rorxq $34,%rcx,%r14+ rorxq $28,%rcx,%r13+ leaq (%r9,%rbx,1),%r9+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %rdx,%rdi+ xorq %r13,%r14+ leaq (%rbx,%rdi,1),%rbx+ movq %r10,%r12+ addq 104(%rsp),%rax+ andq %r9,%r12+ rorxq $41,%r9,%r13+ rorxq $18,%r9,%rdi+ leaq (%rbx,%r14,1),%rbx+ leaq (%rax,%r12,1),%rax+ andnq %r11,%r9,%r12+ xorq %rdi,%r13+ rorxq $14,%r9,%r14+ leaq (%rax,%r12,1),%rax+ xorq %r14,%r13+ movq %rbx,%rdi+ rorxq $39,%rbx,%r12+ leaq (%rax,%r13,1),%rax+ xorq %rcx,%rdi+ rorxq $34,%rbx,%r14+ rorxq $28,%rbx,%r13+ leaq (%r8,%rax,1),%r8+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %rcx,%r15+ xorq %r13,%r14+ leaq (%rax,%r15,1),%rax+ movq %r9,%r12+ movq -64(%rbp),%rdi+ addq %r14,%rax+ movq -56(%rbp),%r12++ addq 0(%rdi),%rax+ addq 8(%rdi),%rbx+ addq 16(%rdi),%rcx+ addq 24(%rdi),%rdx+ addq 32(%rdi),%r8+ addq 40(%rdi),%r9+ addq 48(%rdi),%r10+ addq 56(%rdi),%r11++ movq %rax,0(%rdi)+ movq %rbx,8(%rdi)+ movq %rcx,16(%rdi)+ movq %rdx,24(%rdi)+ movq %r8,32(%rdi)+ movq %r9,40(%rdi)+ movq %r10,48(%rdi)+ movq %r11,56(%rdi)++ cmpq -48(%rbp),%r12+ je .Ldone_avx2++ leaq 1152(%rsp),%rsi+ xorq %r14,%r14+ movq %rbx,%rdi+ xorq %rcx,%rdi+ movq %r9,%r12+ jmp .Lower_avx2+.p2align 4+.Lower_avx2:+ addq 0+16(%rsi),%r11+ andq %r8,%r12+ rorxq $41,%r8,%r13+ rorxq $18,%r8,%r15+ leaq (%rax,%r14,1),%rax+ leaq (%r11,%r12,1),%r11+ andnq %r10,%r8,%r12+ xorq %r15,%r13+ rorxq $14,%r8,%r14+ leaq (%r11,%r12,1),%r11+ xorq %r14,%r13+ movq %rax,%r15+ rorxq $39,%rax,%r12+ leaq (%r11,%r13,1),%r11+ xorq %rbx,%r15+ rorxq $34,%rax,%r14+ rorxq $28,%rax,%r13+ leaq (%rdx,%r11,1),%rdx+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %rbx,%rdi+ xorq %r13,%r14+ leaq (%r11,%rdi,1),%r11+ movq %r8,%r12+ addq 8+16(%rsi),%r10+ andq %rdx,%r12+ rorxq $41,%rdx,%r13+ rorxq $18,%rdx,%rdi+ leaq (%r11,%r14,1),%r11+ leaq (%r10,%r12,1),%r10+ andnq %r9,%rdx,%r12+ xorq %rdi,%r13+ rorxq $14,%rdx,%r14+ leaq (%r10,%r12,1),%r10+ xorq %r14,%r13+ movq %r11,%rdi+ rorxq $39,%r11,%r12+ leaq (%r10,%r13,1),%r10+ xorq %rax,%rdi+ rorxq $34,%r11,%r14+ rorxq $28,%r11,%r13+ leaq (%rcx,%r10,1),%rcx+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %rax,%r15+ xorq %r13,%r14+ leaq (%r10,%r15,1),%r10+ movq %rdx,%r12+ addq 32+16(%rsi),%r9+ andq %rcx,%r12+ rorxq $41,%rcx,%r13+ rorxq $18,%rcx,%r15+ leaq (%r10,%r14,1),%r10+ leaq (%r9,%r12,1),%r9+ andnq %r8,%rcx,%r12+ xorq %r15,%r13+ rorxq $14,%rcx,%r14+ leaq (%r9,%r12,1),%r9+ xorq %r14,%r13+ movq %r10,%r15+ rorxq $39,%r10,%r12+ leaq (%r9,%r13,1),%r9+ xorq %r11,%r15+ rorxq $34,%r10,%r14+ rorxq $28,%r10,%r13+ leaq (%rbx,%r9,1),%rbx+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %r11,%rdi+ xorq %r13,%r14+ leaq (%r9,%rdi,1),%r9+ movq %rcx,%r12+ addq 40+16(%rsi),%r8+ andq %rbx,%r12+ rorxq $41,%rbx,%r13+ rorxq $18,%rbx,%rdi+ leaq (%r9,%r14,1),%r9+ leaq (%r8,%r12,1),%r8+ andnq %rdx,%rbx,%r12+ xorq %rdi,%r13+ rorxq $14,%rbx,%r14+ leaq (%r8,%r12,1),%r8+ xorq %r14,%r13+ movq %r9,%rdi+ rorxq $39,%r9,%r12+ leaq (%r8,%r13,1),%r8+ xorq %r10,%rdi+ rorxq $34,%r9,%r14+ rorxq $28,%r9,%r13+ leaq (%rax,%r8,1),%rax+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %r10,%r15+ xorq %r13,%r14+ leaq (%r8,%r15,1),%r8+ movq %rbx,%r12+ addq 64+16(%rsi),%rdx+ andq %rax,%r12+ rorxq $41,%rax,%r13+ rorxq $18,%rax,%r15+ leaq (%r8,%r14,1),%r8+ leaq (%rdx,%r12,1),%rdx+ andnq %rcx,%rax,%r12+ xorq %r15,%r13+ rorxq $14,%rax,%r14+ leaq (%rdx,%r12,1),%rdx+ xorq %r14,%r13+ movq %r8,%r15+ rorxq $39,%r8,%r12+ leaq (%rdx,%r13,1),%rdx+ xorq %r9,%r15+ rorxq $34,%r8,%r14+ rorxq $28,%r8,%r13+ leaq (%r11,%rdx,1),%r11+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %r9,%rdi+ xorq %r13,%r14+ leaq (%rdx,%rdi,1),%rdx+ movq %rax,%r12+ addq 72+16(%rsi),%rcx+ andq %r11,%r12+ rorxq $41,%r11,%r13+ rorxq $18,%r11,%rdi+ leaq (%rdx,%r14,1),%rdx+ leaq (%rcx,%r12,1),%rcx+ andnq %rbx,%r11,%r12+ xorq %rdi,%r13+ rorxq $14,%r11,%r14+ leaq (%rcx,%r12,1),%rcx+ xorq %r14,%r13+ movq %rdx,%rdi+ rorxq $39,%rdx,%r12+ leaq (%rcx,%r13,1),%rcx+ xorq %r8,%rdi+ rorxq $34,%rdx,%r14+ rorxq $28,%rdx,%r13+ leaq (%r10,%rcx,1),%r10+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %r8,%r15+ xorq %r13,%r14+ leaq (%rcx,%r15,1),%rcx+ movq %r11,%r12+ addq 96+16(%rsi),%rbx+ andq %r10,%r12+ rorxq $41,%r10,%r13+ rorxq $18,%r10,%r15+ leaq (%rcx,%r14,1),%rcx+ leaq (%rbx,%r12,1),%rbx+ andnq %rax,%r10,%r12+ xorq %r15,%r13+ rorxq $14,%r10,%r14+ leaq (%rbx,%r12,1),%rbx+ xorq %r14,%r13+ movq %rcx,%r15+ rorxq $39,%rcx,%r12+ leaq (%rbx,%r13,1),%rbx+ xorq %rdx,%r15+ rorxq $34,%rcx,%r14+ rorxq $28,%rcx,%r13+ leaq (%r9,%rbx,1),%r9+ andq %r15,%rdi+ xorq %r12,%r14+ xorq %rdx,%rdi+ xorq %r13,%r14+ leaq (%rbx,%rdi,1),%rbx+ movq %r10,%r12+ addq 104+16(%rsi),%rax+ andq %r9,%r12+ rorxq $41,%r9,%r13+ rorxq $18,%r9,%rdi+ leaq (%rbx,%r14,1),%rbx+ leaq (%rax,%r12,1),%rax+ andnq %r11,%r9,%r12+ xorq %rdi,%r13+ rorxq $14,%r9,%r14+ leaq (%rax,%r12,1),%rax+ xorq %r14,%r13+ movq %rbx,%rdi+ rorxq $39,%rbx,%r12+ leaq (%rax,%r13,1),%rax+ xorq %rcx,%rdi+ rorxq $34,%rbx,%r14+ rorxq $28,%rbx,%r13+ leaq (%r8,%rax,1),%r8+ andq %rdi,%r15+ xorq %r12,%r14+ xorq %rcx,%r15+ xorq %r13,%r14+ leaq (%rax,%r15,1),%rax+ movq %r9,%r12+ leaq -128(%rsi),%rsi+ cmpq %rsp,%rsi+ jae .Lower_avx2++ movq -64(%rbp),%rdi+ addq %r14,%rax+ movq -56(%rbp),%rsi+ leaq 1152(%rsp),%rsp++ addq 0(%rdi),%rax+ addq 8(%rdi),%rbx+ addq 16(%rdi),%rcx+ addq 24(%rdi),%rdx+ addq 32(%rdi),%r8+ addq 40(%rdi),%r9+ leaq 256(%rsi),%rsi+ addq 48(%rdi),%r10+ movq %rsi,%r12+ addq 56(%rdi),%r11+ cmpq -48(%rbp),%rsi++ movq %rax,0(%rdi)+ cmoveq %rsp,%r12+ movq %rbx,8(%rdi)+ movq %rcx,16(%rdi)+ movq %rdx,24(%rdi)+ movq %r8,32(%rdi)+ movq %r9,40(%rdi)+ movq %r10,48(%rdi)+ movq %r11,56(%rdi)++ jbe .Loop_avx2++.Ldone_avx2:+ vzeroupper+ movaps -160(%rbp),%xmm6+ movaps -144(%rbp),%xmm7+ movaps -128(%rbp),%xmm8+ movaps -112(%rbp),%xmm9+ movaps -96(%rbp),%xmm10+ movaps -80(%rbp),%xmm11+ movq -40(%rbp),%r15+ movq -32(%rbp),%r14+ movq -24(%rbp),%r13+ movq -16(%rbp),%r12+ movq -8(%rbp),%rbx+ movq %rbp,%rsp++ popq %rbp++.LSEH_epilogue_crypton_sha512_asm_block_data_order_avx2:+ mov 8(%rsp),%rdi+ mov 16(%rsp),%rsi++ .byte 0xf3,0xc3++.LSEH_end_crypton_sha512_asm_block_data_order_avx2:+.section .pdata+.p2align 2+.rva .LSEH_begin_crypton_sha512_asm_block_data_order+.rva .LSEH_body_crypton_sha512_asm_block_data_order+.rva .LSEH_info_crypton_sha512_asm_block_data_order_prologue++.rva .LSEH_body_crypton_sha512_asm_block_data_order+.rva .LSEH_epilogue_crypton_sha512_asm_block_data_order+.rva .LSEH_info_crypton_sha512_asm_block_data_order_body++.rva .LSEH_epilogue_crypton_sha512_asm_block_data_order+.rva .LSEH_end_crypton_sha512_asm_block_data_order+.rva .LSEH_info_crypton_sha512_asm_block_data_order_epilogue++.rva .LSEH_begin_crypton_sha512_asm_block_data_order_shaext+.rva .LSEH_body_crypton_sha512_asm_block_data_order_shaext+.rva .LSEH_info_crypton_sha512_asm_block_data_order_shaext_prologue++.rva .LSEH_body_crypton_sha512_asm_block_data_order_shaext+.rva .LSEH_epilogue_crypton_sha512_asm_block_data_order_shaext+.rva .LSEH_info_crypton_sha512_asm_block_data_order_shaext_body++.rva .LSEH_epilogue_crypton_sha512_asm_block_data_order_shaext+.rva .LSEH_end_crypton_sha512_asm_block_data_order_shaext+.rva .LSEH_info_crypton_sha512_asm_block_data_order_shaext_epilogue++.rva .LSEH_begin_crypton_sha512_asm_block_data_order_xop+.rva .LSEH_body_crypton_sha512_asm_block_data_order_xop+.rva .LSEH_info_crypton_sha512_asm_block_data_order_xop_prologue++.rva .LSEH_body_crypton_sha512_asm_block_data_order_xop+.rva .LSEH_epilogue_crypton_sha512_asm_block_data_order_xop+.rva .LSEH_info_crypton_sha512_asm_block_data_order_xop_body++.rva .LSEH_epilogue_crypton_sha512_asm_block_data_order_xop+.rva .LSEH_end_crypton_sha512_asm_block_data_order_xop+.rva .LSEH_info_crypton_sha512_asm_block_data_order_xop_epilogue++.rva .LSEH_begin_crypton_sha512_asm_block_data_order_avx+.rva .LSEH_body_crypton_sha512_asm_block_data_order_avx+.rva .LSEH_info_crypton_sha512_asm_block_data_order_avx_prologue++.rva .LSEH_body_crypton_sha512_asm_block_data_order_avx+.rva .LSEH_epilogue_crypton_sha512_asm_block_data_order_avx+.rva .LSEH_info_crypton_sha512_asm_block_data_order_avx_body++.rva .LSEH_epilogue_crypton_sha512_asm_block_data_order_avx+.rva .LSEH_end_crypton_sha512_asm_block_data_order_avx+.rva .LSEH_info_crypton_sha512_asm_block_data_order_avx_epilogue++.rva .LSEH_begin_crypton_sha512_asm_block_data_order_avx2+.rva .LSEH_body_crypton_sha512_asm_block_data_order_avx2+.rva .LSEH_info_crypton_sha512_asm_block_data_order_avx2_prologue++.rva .LSEH_body_crypton_sha512_asm_block_data_order_avx2+.rva .LSEH_epilogue_crypton_sha512_asm_block_data_order_avx2+.rva .LSEH_info_crypton_sha512_asm_block_data_order_avx2_body++.rva .LSEH_epilogue_crypton_sha512_asm_block_data_order_avx2+.rva .LSEH_end_crypton_sha512_asm_block_data_order_avx2+.rva .LSEH_info_crypton_sha512_asm_block_data_order_avx2_epilogue++.section .xdata+.p2align 3+.LSEH_info_crypton_sha512_asm_block_data_order_prologue:+.byte 1,4,6,0x05+.byte 4,0x74,2,0+.byte 4,0x64,3,0+.byte 4,0x53+.byte 1,0x50+.long 0,0+.LSEH_info_crypton_sha512_asm_block_data_order_body:+.byte 1,0,18,0+.byte 0x00,0xf4,0x13,0x00+.byte 0x00,0xe4,0x14,0x00+.byte 0x00,0xd4,0x15,0x00+.byte 0x00,0xc4,0x16,0x00+.byte 0x00,0x34,0x17,0x00+.byte 0x00,0x54,0x18,0x00+.byte 0x00,0x74,0x1a,0x00+.byte 0x00,0x64,0x1b,0x00+.byte 0x00,0x01,0x19,0x00+.byte 0x00,0x00,0x00,0x00+.byte 0x00,0x00,0x00,0x00+.LSEH_info_crypton_sha512_asm_block_data_order_epilogue:+.byte 1,0,5,11+.byte 0x00,0x74,0x01,0x00+.byte 0x00,0x64,0x02,0x00+.byte 0x00,0xb3+.byte 0x00,0x00,0x00,0x00,0x00,0x00+.byte 0x00,0x00,0x00,0x00++.LSEH_info_crypton_sha512_asm_block_data_order_shaext_prologue:+.byte 1,4,6,0x05+.byte 4,0x74,2,0+.byte 4,0x64,3,0+.byte 4,0x53+.byte 1,0x50+.long 0,0+.LSEH_info_crypton_sha512_asm_block_data_order_shaext_body:+.byte 1,0,17,85+.byte 0x00,0x68,0x00,0x00+.byte 0x00,0x78,0x01,0x00+.byte 0x00,0x88,0x02,0x00+.byte 0x00,0x98,0x03,0x00+.byte 0x00,0xa8,0x04,0x00+.byte 0x00,0x74,0x0c,0x00+.byte 0x00,0x64,0x0d,0x00+.byte 0x00,0x53+.byte 0x00,0x92+.byte 0x00,0x50+.byte 0x00,0x00,0x00,0x00,0x00,0x00+.byte 0x00,0x00,0x00,0x00+.LSEH_info_crypton_sha512_asm_block_data_order_shaext_epilogue:+.byte 1,0,4,0+.byte 0x00,0x74,0x01,0x00+.byte 0x00,0x64,0x02,0x00+.byte 0x00,0x00,0x00,0x00++.LSEH_info_crypton_sha512_asm_block_data_order_xop_prologue:+.byte 1,4,6,0x05+.byte 4,0x74,2,0+.byte 4,0x64,3,0+.byte 4,0x53+.byte 1,0x50+.long 0,0+.LSEH_info_crypton_sha512_asm_block_data_order_xop_body:+.byte 1,0,30,165+.byte 0x00,0x68,0x00,0x00+.byte 0x00,0x78,0x01,0x00+.byte 0x00,0x88,0x02,0x00+.byte 0x00,0x98,0x03,0x00+.byte 0x00,0xa8,0x04,0x00+.byte 0x00,0xb8,0x05,0x00+.byte 0x00,0xf4,0x0f,0x00+.byte 0x00,0xe4,0x10,0x00+.byte 0x00,0xd4,0x11,0x00+.byte 0x00,0xc4,0x12,0x00+.byte 0x00,0x34,0x13,0x00+.byte 0x00,0x74,0x16,0x00+.byte 0x00,0x64,0x17,0x00+.byte 0x00,0x53+.byte 0x00,0x01,0x14,0x00+.byte 0x00,0x50+.byte 0x00,0x00,0x00,0x00+.byte 0x00,0x00,0x00,0x00+.LSEH_info_crypton_sha512_asm_block_data_order_xop_epilogue:+.byte 1,0,4,0+.byte 0x00,0x74,0x01,0x00+.byte 0x00,0x64,0x02,0x00+.byte 0x00,0x00,0x00,0x00++.LSEH_info_crypton_sha512_asm_block_data_order_avx_prologue:+.byte 1,4,6,0x05+.byte 4,0x74,2,0+.byte 4,0x64,3,0+.byte 4,0x53+.byte 1,0x50+.long 0,0+.LSEH_info_crypton_sha512_asm_block_data_order_avx_body:+.byte 1,0,30,165+.byte 0x00,0x68,0x00,0x00+.byte 0x00,0x78,0x01,0x00+.byte 0x00,0x88,0x02,0x00+.byte 0x00,0x98,0x03,0x00+.byte 0x00,0xa8,0x04,0x00+.byte 0x00,0xb8,0x05,0x00+.byte 0x00,0xf4,0x0f,0x00+.byte 0x00,0xe4,0x10,0x00+.byte 0x00,0xd4,0x11,0x00+.byte 0x00,0xc4,0x12,0x00+.byte 0x00,0x34,0x13,0x00+.byte 0x00,0x74,0x16,0x00+.byte 0x00,0x64,0x17,0x00+.byte 0x00,0x53+.byte 0x00,0x01,0x14,0x00+.byte 0x00,0x50+.byte 0x00,0x00,0x00,0x00+.byte 0x00,0x00,0x00,0x00+.LSEH_info_crypton_sha512_asm_block_data_order_avx_epilogue:+.byte 1,0,4,0+.byte 0x00,0x74,0x01,0x00+.byte 0x00,0x64,0x02,0x00+.byte 0x00,0x00,0x00,0x00++.LSEH_info_crypton_sha512_asm_block_data_order_avx2_prologue:+.byte 1,4,6,0x05+.byte 4,0x74,2,0+.byte 4,0x64,3,0+.byte 4,0x53+.byte 1,0x50+.long 0,0+.LSEH_info_crypton_sha512_asm_block_data_order_avx2_body:+.byte 1,0,30,165+.byte 0x00,0x68,0x00,0x00+.byte 0x00,0x78,0x01,0x00+.byte 0x00,0x88,0x02,0x00+.byte 0x00,0x98,0x03,0x00+.byte 0x00,0xa8,0x04,0x00+.byte 0x00,0xb8,0x05,0x00+.byte 0x00,0xf4,0x0f,0x00+.byte 0x00,0xe4,0x10,0x00+.byte 0x00,0xd4,0x11,0x00+.byte 0x00,0xc4,0x12,0x00+.byte 0x00,0x34,0x13,0x00+.byte 0x00,0x74,0x16,0x00+.byte 0x00,0x64,0x17,0x00+.byte 0x00,0x53+.byte 0x00,0x01,0x14,0x00+.byte 0x00,0x50+.byte 0x00,0x00,0x00,0x00+.byte 0x00,0x00,0x00,0x00+.LSEH_info_crypton_sha512_asm_block_data_order_avx2_epilogue:+.byte 1,0,4,0+.byte 0x00,0x74,0x01,0x00+.byte 0x00,0x64,0x02,0x00+.byte 0x00,0x00,0x00,0x00+
@@ -0,0 +1,2519 @@+#!/usr/bin/env perl+#+# ====================================================================+# Written by Andy Polyakov, @dot-asm, initially for the OpenSSL+# project.+# ====================================================================+#+# sha256/512_block procedure for x86_64.+#+# 40% improvement over compiler-generated code on Opteron. On EM64T+# sha256 was observed to run >80% faster and sha512 - >40%. No magical+# tricks, just straight implementation... I really wonder why gcc+# [being armed with inline assembler] fails to generate as fast code.+# The only thing which is cool about this module is that it's very+# same instruction sequence used for both SHA-256 and SHA-512. In+# former case the instructions operate on 32-bit operands, while in+# latter - on 64-bit ones. All I had to do is to get one flavor right,+# the other one passed the test right away:-)+#+# sha256_block runs in ~1005 cycles on Opteron, which gives you+# asymptotic performance of 64*1000/1005=63.7MBps times CPU clock+# frequency in GHz. sha512_block runs in ~1275 cycles, which results+# in 128*1000/1275=100MBps per GHz. Is there room for improvement?+# Well, if you compare it to IA-64 implementation, which maintains+# X[16] in register bank[!], tends to 4 instructions per CPU clock+# cycle and runs in 1003 cycles, 1275 is very good result for 3-way+# issue Opteron pipeline and X[16] maintained in memory. So that *if*+# there is a way to improve it, *then* the only way would be to try to+# offload X[16] updates to SSE unit, but that would require "deeper"+# loop unroll, which in turn would naturally cause size blow-up, not+# to mention increased complexity! And once again, only *if* it's+# actually possible to noticeably improve overall ILP, instruction+# level parallelism, on a given CPU implementation in this case.+#+# Special note on Intel EM64T. While Opteron CPU exhibits perfect+# performance ratio of 1.5 between 64- and 32-bit flavors [see above],+# [currently available] EM64T CPUs apparently are far from it. On the+# contrary, 64-bit version, sha512_block, is ~30% *slower* than 32-bit+# sha256_block:-( This is presumably because 64-bit shifts/rotates+# apparently are not atomic instructions, but implemented in microcode.+#+# May 2012.+#+# Optimization including one of Pavel Semjanov's ideas, alternative+# Maj, resulted in >=5% improvement on most CPUs, +20% SHA256 and+# unfortunately -2% SHA512 on P4 [which nobody should care about+# that much].+#+# June 2012.+#+# Add SIMD code paths, see below for improvement coefficients. SSSE3+# code path was not attempted for SHA512, because improvement is not+# estimated to be high enough, noticeably less than 9%, to justify+# the effort, not on pre-AVX processors. [Obviously with exclusion+# for VIA Nano, but it has SHA512 instruction that is faster and+# should be used instead.] For reference, corresponding estimated+# upper limit for improvement for SSSE3 SHA256 is 28%. The fact that+# higher coefficients are observed on VIA Nano and Bulldozer has more+# to do with specifics of their architecture [which is topic for+# separate discussion].+#+# November 2012.+#+# Add AVX2 code path. Two consecutive input blocks are loaded to+# 256-bit %ymm registers, with data from first block to least+# significant 128-bit halves and data from second to most significant.+# The data is then processed with same SIMD instruction sequence as+# for AVX, but with %ymm as operands. Side effect is increased stack+# frame, 448 additional bytes in SHA256 and 1152 in SHA512, and 1.2KB+# code size increase.+#+# March 2014.+#+# Add support for Intel SHA Extensions.+#+# October 2023.+#+# Add support for Intel SHA512 Extension.++######################################################################+# Current performance in cycles per processed byte (less is better):+#+# SHA256 SSSE3 AVX/XOP(*) SHA512 AVX/XOP(*)+#+# AMD K8 14.9 - - 9.57 -+# P4 17.3 - - 30.8 -+# Core 2 15.6 13.8(+13%) - 9.97 -+# Westmere 14.8 12.3(+19%) - 9.58 -+# Sandy Bridge 17.4 14.2(+23%) 11.6(+50%(**)) 11.2 8.10(+38%(**))+# Ivy Bridge 12.6 10.5(+20%) 10.3(+22%) 8.17 7.22(+13%)+# Haswell 12.2 9.28(+31%) 7.80(+56%) 7.66 5.40(+42%)+# Skylake 11.4 9.03(+26%) 7.70(+48%) 7.25 5.20(+40%)+# Cannon Lake 11.4 9.00(+27%) 3.55(+220%) 7.20 5.12(+41%)+# Rocket Lake 10.4 9.13(+14%) 2.43(+330%) 6.66 5.34(+25%)+# Bulldozer 21.1 13.6(+54%) 13.6(+54%(***)) 13.5 8.58(+57%)+# Ryzen 11.0 9.02(+22%) 2.05(+440%) 7.05 5.67(+20%)+# VIA Nano 23.0 16.5(+39%) - 14.7 -+# Atom 23.0 18.9(+22%) - 14.7 -+# Silvermont 27.4 20.6(+33%) - 17.5 -+# Knights L 27.4 21.0(+30%) 19.6(+40%) 17.5 12.8(+37%)+# Goldmont 18.9 14.3(+32%) 4.16(+350%) 12.0 -+#+# (*) whichever best applicable, including SHAEXT;+# (**) switch from ror to shrd stands for fair share of improvement;+# (***) execution time is fully determined by remaining integer-only+# part, body_00_15; reducing the amount of SIMD instructions+# below certain limit makes no difference/sense; to conserve+# space SHA256 XOP code path is therefore omitted;++$flavour = shift;+$output = pop;+if ($flavour =~ /\./) { $output = $flavour; undef $flavour; }++$win64=0; $win64=1 if ($flavour =~ /[nm]asm|mingw64/ || $output =~ /\.asm$/);++$0 =~ m/(.*[\/\\])[^\/\\]+$/; $dir=$1;+( $xlate="${dir}x86_64-xlate.pl" and -f $xlate ) or+( $xlate="${dir}../../perlasm/x86_64-xlate.pl" and -f $xlate) or+die "can't locate x86_64-xlate.pl";++$avx=undef;+$shaext=1; ### set to zero if compiling for 1.0.1++if (!defined($avx) && $win64 && ($flavour =~ /nasm/ || $ENV{ASM} =~ /nasm/) &&+ ($ENV{ASM} //= "nasm") &&+ `"$ENV{ASM}" -v 2>&1` =~ /NASM version ([0-9]+\.[0-9]+)(?:\.([0-9]+))?/) {+ $avx = ($1>=2.09) + ($1>=2.10) + 2 * ($1>=2.12);+ $avx += 2 if ($1==2.11 && $2>=8);+}++if (!defined($avx) && $win64 && ($flavour =~ /masm/ || $ENV{ASM} =~ /ml64/) &&+ ($ENV{ASM} //= "ml64") &&+ `"$ENV{ASM}" 2>&1` =~ /Version ([0-9]+)\./) {+ $avx = ($1>=10) + ($1>=12) + 2 * ($1>=14);+}++$ENV{CC} //= "cc";+if (!defined($avx) && `$ENV{CC} -Wa,-v -c -o /dev/zero -x assembler /dev/null 2>&1`+ =~ /GNU assembler version ([0-9]+)\.([0-9]+)/) {+ my $ver = $1 + $2/100.0; # 3.1->3.01, 3.10->3.10+ $avx = ($ver>=2.19) + ($ver>=2.22) + ($ver>=2.25) + ($ver>=2.26);+}++if (!defined($avx) && `$ENV{CC} -v 2>&1`+ =~ /((?:^clang|LLVM) version|.*based on LLVM) ([0-9]+)\.([0-9]+)/) {+ my $ver = $2 + $3/100.0; # 3.1->3.01, 3.10->3.10+ $avx = ($ver>=3.0) + ($ver>3.0);+ $avx += 2*($ver>=7.0) if ($1 =~ /^clang/);+}++open STDOUT,"| \"$^X\" \"$xlate\" $flavour \"$output\"";++if ($output =~ /512/) {+ $func="sha512_block_data_order";+ $TABLE="K512";+ $SZ=8;+ @ROT=($A,$B,$C,$D,$E,$F,$G,$H)=("%rax","%rbx","%rcx","%rdx",+ "%r8", "%r9", "%r10","%r11");+ ($T1,$a0,$a1,$a2,$a3)=("%r12","%r13","%r14","%r15","%rdi");+ @Sigma0=(28,34,39);+ @Sigma1=(14,18,41);+ @sigma0=(1, 8, 7);+ @sigma1=(19,61, 6);+ $rounds=80;+} else {+ $func="sha256_block_data_order";+ $TABLE="K256";+ $SZ=4;+ @ROT=($A,$B,$C,$D,$E,$F,$G,$H)=("%eax","%ebx","%ecx","%edx",+ "%r8d","%r9d","%r10d","%r11d");+ ($T1,$a0,$a1,$a2,$a3)=("%r12d","%r13d","%r14d","%r15d","%edi");+ @Sigma0=( 2,13,22);+ @Sigma1=( 6,11,25);+ @sigma0=( 7,18, 3);+ @sigma1=(17,19,10);+ $rounds=64;+}++$ctx="%rdi"; # 1st arg, zapped by $a3+$inp="%rsi"; # 2nd arg+$Tbl="%rbp";++$_ctx="16*$SZ+0*8(%rsp)";+$_inp="16*$SZ+1*8(%rsp)";+$_end="16*$SZ+2*8(%rsp)";+$framesz="16*$SZ+3*8";+++sub ROUND_00_15()+{ my ($i,$a,$b,$c,$d,$e,$f,$g,$h) = @_;+ my $STRIDE=$SZ;+ $STRIDE += 16 if ($i%(16/$SZ)==(16/$SZ-1));++$code.=<<___;+ ror \$`$Sigma1[2]-$Sigma1[1]`,$a0+ mov $f,$a2++ xor $e,$a0+ ror \$`$Sigma0[2]-$Sigma0[1]`,$a1+ xor $g,$a2 # f^g++ mov $T1,`$SZ*($i&0xf)`(%rsp)+ xor $a,$a1+ and $e,$a2 # (f^g)&e++ ror \$`$Sigma1[1]-$Sigma1[0]`,$a0+ add $h,$T1 # T1+=h+ xor $g,$a2 # Ch(e,f,g)=((f^g)&e)^g++ ror \$`$Sigma0[1]-$Sigma0[0]`,$a1+ xor $e,$a0+ add $a2,$T1 # T1+=Ch(e,f,g)++ mov $a,$a2+ add ($Tbl),$T1 # T1+=K[round]+ xor $a,$a1++ xor $b,$a2 # a^b, b^c in next round+ ror \$$Sigma1[0],$a0 # Sigma1(e)+ mov $b,$h++ and $a2,$a3+ ror \$$Sigma0[0],$a1 # Sigma0(a)+ add $a0,$T1 # T1+=Sigma1(e)++ xor $a3,$h # h=Maj(a,b,c)=Ch(a^b,c,b)+ add $T1,$d # d+=T1+ add $T1,$h # h+=T1++ lea $STRIDE($Tbl),$Tbl # round+++___+$code.=<<___ if ($i<15);+ add $a1,$h # h+=Sigma0(a)+___+ ($a2,$a3) = ($a3,$a2);+}++sub ROUND_16_XX()+{ my ($i,$a,$b,$c,$d,$e,$f,$g,$h) = @_;++$code.=<<___;+ mov `$SZ*(($i+1)&0xf)`(%rsp),$a0+ mov `$SZ*(($i+14)&0xf)`(%rsp),$a2++ mov $a0,$T1+ ror \$`$sigma0[1]-$sigma0[0]`,$a0+ add $a1,$a # modulo-scheduled h+=Sigma0(a)+ mov $a2,$a1+ ror \$`$sigma1[1]-$sigma1[0]`,$a2++ xor $T1,$a0+ shr \$$sigma0[2],$T1+ ror \$$sigma0[0],$a0+ xor $a1,$a2+ shr \$$sigma1[2],$a1++ ror \$$sigma1[0],$a2+ xor $a0,$T1 # sigma0(X[(i+1)&0xf])+ xor $a1,$a2 # sigma1(X[(i+14)&0xf])+ add `$SZ*(($i+9)&0xf)`(%rsp),$T1++ add `$SZ*($i&0xf)`(%rsp),$T1+ mov $e,$a0+ add $a2,$T1+ mov $a,$a1+___+ &ROUND_00_15(@_);+}++$code=<<___;+.text++.extern OPENSSL_ia32cap_P+.globl $func+.type $func,\@function,3,"unwind"+.align 16+$func:+.cfi_startproc+ push %rbp+.cfi_push %rbp+ mov %rsp,%rbp+.cfi_def_cfa_register %rbp+___+$code.=<<___ if ($SZ==4 || $avx);+ lea OPENSSL_ia32cap_P(%rip),%rax+ mov 0(%rax),%r9d+ mov 4(%rax),%r10d+ mov 8(%rax),%eax+___+$code.=<<___ if ($SZ==4 && $shaext);+ test \$`1<<29`,%eax # check for SHA+ jnz .Lshaext_shortcut+___+$code.=<<___ if ($avx && $SZ==8);+ test \$`1<<11`,%r10d # check for XOP+ jnz .Lxop_shortcut+___+$code.=<<___ if ($avx>1);+ and \$`1<<8|1<<5|1<<3`,%eax # check for BMI2+AVX2+BMI1+ cmp \$`1<<8|1<<5|1<<3`,%eax+ je .Lavx2_shortcut+___+$code.=<<___ if ($avx);+ and \$`1<<30`,%r9d # mask "Intel CPU" bit+ and \$`1<<28|1<<9`,%r10d # mask AVX and SSSE3 bits+ or %r9d,%r10d+ cmp \$`1<<28|1<<9|1<<30`,%r10d+ je .Lavx_shortcut+___+$code.=<<___ if ($SZ==4);+ test \$`1<<9`,%r10d+ jnz .Lssse3_shortcut+___+$code.=<<___;+ push %rbx+.cfi_push %rbx+ push %r12+.cfi_push %r12+ push %r13+.cfi_push %r13+ push %r14+.cfi_push %r14+ push %r15+.cfi_push %r15+ shl \$4,%rdx # num*16+ sub \$$framesz,%rsp+.cfi_alloca $framesz+.cfi_def_cfa %rsp+.cfi_end_prologue+ lea ($inp,%rdx,$SZ),%rdx # inp+num*16*$SZ+ mov $ctx,$_ctx # save ctx, 1st arg+ mov $inp,$_inp # save inp, 2nd arh+ mov %rdx,$_end # save end pointer, "3rd" arg++ mov $SZ*0($ctx),$A+ mov $SZ*1($ctx),$B+ mov $SZ*2($ctx),$C+ mov $SZ*3($ctx),$D+ mov $SZ*4($ctx),$E+ mov $SZ*5($ctx),$F+ mov $SZ*6($ctx),$G+ mov $SZ*7($ctx),$H+ jmp .Lloop++.align 16+.Lloop:+ mov $B,$a3+ lea $TABLE(%rip),$Tbl+ xor $C,$a3 # magic+___+ for($i=0;$i<16;$i++) {+ $code.=" mov $SZ*$i($inp),$T1\n";+ $code.=" mov @ROT[4],$a0\n";+ $code.=" mov @ROT[0],$a1\n";+ $code.=" bswap $T1\n";+ &ROUND_00_15($i,@ROT);+ unshift(@ROT,pop(@ROT));+ }+$code.=<<___;+ jmp .Lrounds_16_xx+.align 16+.Lrounds_16_xx:+___+ for(;$i<32;$i++) {+ &ROUND_16_XX($i,@ROT);+ unshift(@ROT,pop(@ROT));+ }++$code.=<<___;+ cmpb \$0,`$SZ-1`($Tbl)+ jnz .Lrounds_16_xx++ mov $_ctx,$ctx+ add $a1,$A # modulo-scheduled h+=Sigma0(a)+ lea 16*$SZ($inp),$inp++ add $SZ*0($ctx),$A+ add $SZ*1($ctx),$B+ add $SZ*2($ctx),$C+ add $SZ*3($ctx),$D+ add $SZ*4($ctx),$E+ add $SZ*5($ctx),$F+ add $SZ*6($ctx),$G+ add $SZ*7($ctx),$H++ cmp $_end,$inp++ mov $A,$SZ*0($ctx)+ mov $B,$SZ*1($ctx)+ mov $C,$SZ*2($ctx)+ mov $D,$SZ*3($ctx)+ mov $E,$SZ*4($ctx)+ mov $F,$SZ*5($ctx)+ mov $G,$SZ*6($ctx)+ mov $H,$SZ*7($ctx)+ jb .Lloop++ lea $framesz+6*8(%rsp),%r11+.cfi_def_cfa %r11,8+ mov $framesz(%rsp),%r15+ mov -40(%r11),%r14+ mov -32(%r11),%r13+ mov -24(%r11),%r12+ mov -16(%r11),%rbx+ mov -8(%r11),%rbp+.cfi_epilogue+ lea (%r11),%rsp+ ret+.cfi_endproc+.size $func,.-$func+___++if ($SZ==4) {+$code.=<<___;+.align 64+.type $TABLE,\@object+$TABLE:+ .long 0x428a2f98,0x71374491,0xb5c0fbcf,0xe9b5dba5+ .long 0x428a2f98,0x71374491,0xb5c0fbcf,0xe9b5dba5+ .long 0x3956c25b,0x59f111f1,0x923f82a4,0xab1c5ed5+ .long 0x3956c25b,0x59f111f1,0x923f82a4,0xab1c5ed5+ .long 0xd807aa98,0x12835b01,0x243185be,0x550c7dc3+ .long 0xd807aa98,0x12835b01,0x243185be,0x550c7dc3+ .long 0x72be5d74,0x80deb1fe,0x9bdc06a7,0xc19bf174+ .long 0x72be5d74,0x80deb1fe,0x9bdc06a7,0xc19bf174+ .long 0xe49b69c1,0xefbe4786,0x0fc19dc6,0x240ca1cc+ .long 0xe49b69c1,0xefbe4786,0x0fc19dc6,0x240ca1cc+ .long 0x2de92c6f,0x4a7484aa,0x5cb0a9dc,0x76f988da+ .long 0x2de92c6f,0x4a7484aa,0x5cb0a9dc,0x76f988da+ .long 0x983e5152,0xa831c66d,0xb00327c8,0xbf597fc7+ .long 0x983e5152,0xa831c66d,0xb00327c8,0xbf597fc7+ .long 0xc6e00bf3,0xd5a79147,0x06ca6351,0x14292967+ .long 0xc6e00bf3,0xd5a79147,0x06ca6351,0x14292967+ .long 0x27b70a85,0x2e1b2138,0x4d2c6dfc,0x53380d13+ .long 0x27b70a85,0x2e1b2138,0x4d2c6dfc,0x53380d13+ .long 0x650a7354,0x766a0abb,0x81c2c92e,0x92722c85+ .long 0x650a7354,0x766a0abb,0x81c2c92e,0x92722c85+ .long 0xa2bfe8a1,0xa81a664b,0xc24b8b70,0xc76c51a3+ .long 0xa2bfe8a1,0xa81a664b,0xc24b8b70,0xc76c51a3+ .long 0xd192e819,0xd6990624,0xf40e3585,0x106aa070+ .long 0xd192e819,0xd6990624,0xf40e3585,0x106aa070+ .long 0x19a4c116,0x1e376c08,0x2748774c,0x34b0bcb5+ .long 0x19a4c116,0x1e376c08,0x2748774c,0x34b0bcb5+ .long 0x391c0cb3,0x4ed8aa4a,0x5b9cca4f,0x682e6ff3+ .long 0x391c0cb3,0x4ed8aa4a,0x5b9cca4f,0x682e6ff3+ .long 0x748f82ee,0x78a5636f,0x84c87814,0x8cc70208+ .long 0x748f82ee,0x78a5636f,0x84c87814,0x8cc70208+ .long 0x90befffa,0xa4506ceb,0xbef9a3f7,0xc67178f2+ .long 0x90befffa,0xa4506ceb,0xbef9a3f7,0xc67178f2++ .long 0x00010203,0x04050607,0x08090a0b,0x0c0d0e0f+ .long 0x00010203,0x04050607,0x08090a0b,0x0c0d0e0f+ .long 0x03020100,0x0b0a0908,0xffffffff,0xffffffff+ .long 0x03020100,0x0b0a0908,0xffffffff,0xffffffff+ .long 0xffffffff,0xffffffff,0x03020100,0x0b0a0908+ .long 0xffffffff,0xffffffff,0x03020100,0x0b0a0908+ .asciz "SHA256 block transform for x86_64, CRYPTOGAMS by \@dot-asm"+___+} else {+$code.=<<___;+.align 64+.type $TABLE,\@object+$TABLE:+ .quad 0x428a2f98d728ae22,0x7137449123ef65cd+ .quad 0x428a2f98d728ae22,0x7137449123ef65cd+ .quad 0xb5c0fbcfec4d3b2f,0xe9b5dba58189dbbc+ .quad 0xb5c0fbcfec4d3b2f,0xe9b5dba58189dbbc+ .quad 0x3956c25bf348b538,0x59f111f1b605d019+ .quad 0x3956c25bf348b538,0x59f111f1b605d019+ .quad 0x923f82a4af194f9b,0xab1c5ed5da6d8118+ .quad 0x923f82a4af194f9b,0xab1c5ed5da6d8118+ .quad 0xd807aa98a3030242,0x12835b0145706fbe+ .quad 0xd807aa98a3030242,0x12835b0145706fbe+ .quad 0x243185be4ee4b28c,0x550c7dc3d5ffb4e2+ .quad 0x243185be4ee4b28c,0x550c7dc3d5ffb4e2+ .quad 0x72be5d74f27b896f,0x80deb1fe3b1696b1+ .quad 0x72be5d74f27b896f,0x80deb1fe3b1696b1+ .quad 0x9bdc06a725c71235,0xc19bf174cf692694+ .quad 0x9bdc06a725c71235,0xc19bf174cf692694+ .quad 0xe49b69c19ef14ad2,0xefbe4786384f25e3+ .quad 0xe49b69c19ef14ad2,0xefbe4786384f25e3+ .quad 0x0fc19dc68b8cd5b5,0x240ca1cc77ac9c65+ .quad 0x0fc19dc68b8cd5b5,0x240ca1cc77ac9c65+ .quad 0x2de92c6f592b0275,0x4a7484aa6ea6e483+ .quad 0x2de92c6f592b0275,0x4a7484aa6ea6e483+ .quad 0x5cb0a9dcbd41fbd4,0x76f988da831153b5+ .quad 0x5cb0a9dcbd41fbd4,0x76f988da831153b5+ .quad 0x983e5152ee66dfab,0xa831c66d2db43210+ .quad 0x983e5152ee66dfab,0xa831c66d2db43210+ .quad 0xb00327c898fb213f,0xbf597fc7beef0ee4+ .quad 0xb00327c898fb213f,0xbf597fc7beef0ee4+ .quad 0xc6e00bf33da88fc2,0xd5a79147930aa725+ .quad 0xc6e00bf33da88fc2,0xd5a79147930aa725+ .quad 0x06ca6351e003826f,0x142929670a0e6e70+ .quad 0x06ca6351e003826f,0x142929670a0e6e70+ .quad 0x27b70a8546d22ffc,0x2e1b21385c26c926+ .quad 0x27b70a8546d22ffc,0x2e1b21385c26c926+ .quad 0x4d2c6dfc5ac42aed,0x53380d139d95b3df+ .quad 0x4d2c6dfc5ac42aed,0x53380d139d95b3df+ .quad 0x650a73548baf63de,0x766a0abb3c77b2a8+ .quad 0x650a73548baf63de,0x766a0abb3c77b2a8+ .quad 0x81c2c92e47edaee6,0x92722c851482353b+ .quad 0x81c2c92e47edaee6,0x92722c851482353b+ .quad 0xa2bfe8a14cf10364,0xa81a664bbc423001+ .quad 0xa2bfe8a14cf10364,0xa81a664bbc423001+ .quad 0xc24b8b70d0f89791,0xc76c51a30654be30+ .quad 0xc24b8b70d0f89791,0xc76c51a30654be30+ .quad 0xd192e819d6ef5218,0xd69906245565a910+ .quad 0xd192e819d6ef5218,0xd69906245565a910+ .quad 0xf40e35855771202a,0x106aa07032bbd1b8+ .quad 0xf40e35855771202a,0x106aa07032bbd1b8+ .quad 0x19a4c116b8d2d0c8,0x1e376c085141ab53+ .quad 0x19a4c116b8d2d0c8,0x1e376c085141ab53+ .quad 0x2748774cdf8eeb99,0x34b0bcb5e19b48a8+ .quad 0x2748774cdf8eeb99,0x34b0bcb5e19b48a8+ .quad 0x391c0cb3c5c95a63,0x4ed8aa4ae3418acb+ .quad 0x391c0cb3c5c95a63,0x4ed8aa4ae3418acb+ .quad 0x5b9cca4f7763e373,0x682e6ff3d6b2b8a3+ .quad 0x5b9cca4f7763e373,0x682e6ff3d6b2b8a3+ .quad 0x748f82ee5defb2fc,0x78a5636f43172f60+ .quad 0x748f82ee5defb2fc,0x78a5636f43172f60+ .quad 0x84c87814a1f0ab72,0x8cc702081a6439ec+ .quad 0x84c87814a1f0ab72,0x8cc702081a6439ec+ .quad 0x90befffa23631e28,0xa4506cebde82bde9+ .quad 0x90befffa23631e28,0xa4506cebde82bde9+ .quad 0xbef9a3f7b2c67915,0xc67178f2e372532b+ .quad 0xbef9a3f7b2c67915,0xc67178f2e372532b+ .quad 0xca273eceea26619c,0xd186b8c721c0c207+ .quad 0xca273eceea26619c,0xd186b8c721c0c207+ .quad 0xeada7dd6cde0eb1e,0xf57d4f7fee6ed178+ .quad 0xeada7dd6cde0eb1e,0xf57d4f7fee6ed178+ .quad 0x06f067aa72176fba,0x0a637dc5a2c898a6+ .quad 0x06f067aa72176fba,0x0a637dc5a2c898a6+ .quad 0x113f9804bef90dae,0x1b710b35131c471b+ .quad 0x113f9804bef90dae,0x1b710b35131c471b+ .quad 0x28db77f523047d84,0x32caab7b40c72493+ .quad 0x28db77f523047d84,0x32caab7b40c72493+ .quad 0x3c9ebe0a15c9bebc,0x431d67c49c100d4c+ .quad 0x3c9ebe0a15c9bebc,0x431d67c49c100d4c+ .quad 0x4cc5d4becb3e42b6,0x597f299cfc657e2a+ .quad 0x4cc5d4becb3e42b6,0x597f299cfc657e2a+ .quad 0x5fcb6fab3ad6faec,0x6c44198c4a475817+ .quad 0x5fcb6fab3ad6faec,0x6c44198c4a475817++ .quad 0x0001020304050607,0x08090a0b0c0d0e0f+ .quad 0x0001020304050607,0x08090a0b0c0d0e0f++${TABLE}_nodup:+ .quad 0x428a2f98d728ae22,0x7137449123ef65cd+ .quad 0xb5c0fbcfec4d3b2f,0xe9b5dba58189dbbc+ .quad 0x3956c25bf348b538,0x59f111f1b605d019+ .quad 0x923f82a4af194f9b,0xab1c5ed5da6d8118+ .quad 0xd807aa98a3030242,0x12835b0145706fbe+ .quad 0x243185be4ee4b28c,0x550c7dc3d5ffb4e2+ .quad 0x72be5d74f27b896f,0x80deb1fe3b1696b1+ .quad 0x9bdc06a725c71235,0xc19bf174cf692694+ .quad 0xe49b69c19ef14ad2,0xefbe4786384f25e3+ .quad 0x0fc19dc68b8cd5b5,0x240ca1cc77ac9c65+ .quad 0x2de92c6f592b0275,0x4a7484aa6ea6e483+ .quad 0x5cb0a9dcbd41fbd4,0x76f988da831153b5+ .quad 0x983e5152ee66dfab,0xa831c66d2db43210+ .quad 0xb00327c898fb213f,0xbf597fc7beef0ee4+ .quad 0xc6e00bf33da88fc2,0xd5a79147930aa725+ .quad 0x06ca6351e003826f,0x142929670a0e6e70+ .quad 0x27b70a8546d22ffc,0x2e1b21385c26c926+ .quad 0x4d2c6dfc5ac42aed,0x53380d139d95b3df+ .quad 0x650a73548baf63de,0x766a0abb3c77b2a8+ .quad 0x81c2c92e47edaee6,0x92722c851482353b+ .quad 0xa2bfe8a14cf10364,0xa81a664bbc423001+ .quad 0xc24b8b70d0f89791,0xc76c51a30654be30+ .quad 0xd192e819d6ef5218,0xd69906245565a910+ .quad 0xf40e35855771202a,0x106aa07032bbd1b8+ .quad 0x19a4c116b8d2d0c8,0x1e376c085141ab53+ .quad 0x2748774cdf8eeb99,0x34b0bcb5e19b48a8+ .quad 0x391c0cb3c5c95a63,0x4ed8aa4ae3418acb+ .quad 0x5b9cca4f7763e373,0x682e6ff3d6b2b8a3+ .quad 0x748f82ee5defb2fc,0x78a5636f43172f60+ .quad 0x84c87814a1f0ab72,0x8cc702081a6439ec+ .quad 0x90befffa23631e28,0xa4506cebde82bde9+ .quad 0xbef9a3f7b2c67915,0xc67178f2e372532b+ .quad 0xca273eceea26619c,0xd186b8c721c0c207+ .quad 0xeada7dd6cde0eb1e,0xf57d4f7fee6ed178+ .quad 0x06f067aa72176fba,0x0a637dc5a2c898a6+ .quad 0x113f9804bef90dae,0x1b710b35131c471b+ .quad 0x28db77f523047d84,0x32caab7b40c72493+ .quad 0x3c9ebe0a15c9bebc,0x431d67c49c100d4c+ .quad 0x4cc5d4becb3e42b6,0x597f299cfc657e2a+ .quad 0x5fcb6fab3ad6faec,0x6c44198c4a475817++ .asciz "SHA512 block transform for x86_64, CRYPTOGAMS by \@dot-asm"+___+}++######################################################################+# SIMD code paths+#+if ($SZ==4 && $shaext) {{{+######################################################################+# Intel SHA Extensions implementation of SHA256 update function.+#+my ($ctx,$inp,$num,$Tbl)=("%rdi","%rsi","%rdx","%rcx");++my ($Wi,$ABEF,$CDGH,$TMP,$BSWAP,$ABEF_SAVE,$CDGH_SAVE)=map("%xmm$_",(0..2,7..10));+my @MSG=map("%xmm$_",(3..6));++$code.=<<___;+.type sha256_block_data_order_shaext,\@function,3,"unwind"+.align 64+sha256_block_data_order_shaext:+.cfi_startproc+ push %rbp+.cfi_push %rbp+ mov %rsp,%rbp+.cfi_def_cfa_register %rbp+.Lshaext_shortcut:+___+$code.=<<___ if ($win64);+ sub \$0x50,%rsp+.cfi_alloca 0x50+ movaps %xmm6,-0x50(%rbp)+ movaps %xmm7,-0x40(%rbp)+ movaps %xmm8,-0x30(%rbp)+ movaps %xmm9,-0x20(%rbp)+ movaps %xmm10,-0x10(%rbp)+.cfi_offset %xmm6-%xmm10,-0x60+___+$code.=<<___;+.cfi_end_prologue+ lea K256+0x80(%rip),$Tbl+ movdqu ($ctx),$ABEF # DCBA+ movdqu 16($ctx),$CDGH # HGFE+ movdqa 0x200-0x80($Tbl),$TMP # byte swap mask++ pshufd \$0x1b,$ABEF,$Wi # ABCD+ pshufd \$0xb1,$ABEF,$ABEF # CDAB+ pshufd \$0x1b,$CDGH,$CDGH # EFGH+ movdqa $TMP,$BSWAP # offload+ palignr \$8,$CDGH,$ABEF # ABEF+ punpcklqdq $Wi,$CDGH # CDGH+ jmp .Loop_shaext++.align 16+.Loop_shaext:+ movdqu ($inp),@MSG[0]+ movdqu 0x10($inp),@MSG[1]+ movdqu 0x20($inp),@MSG[2]+ pshufb $TMP,@MSG[0]+ movdqu 0x30($inp),@MSG[3]++ movdqa 0*32-0x80($Tbl),$Wi+ paddd @MSG[0],$Wi+ pshufb $TMP,@MSG[1]+ movdqa $CDGH,$CDGH_SAVE # offload+ sha256rnds2 $ABEF,$CDGH # 0-3+ pshufd \$0x0e,$Wi,$Wi+ nop+ movdqa $ABEF,$ABEF_SAVE # offload+ sha256rnds2 $CDGH,$ABEF++ movdqa 1*32-0x80($Tbl),$Wi+ paddd @MSG[1],$Wi+ pshufb $TMP,@MSG[2]+ sha256rnds2 $ABEF,$CDGH # 4-7+ pshufd \$0x0e,$Wi,$Wi+ lea 0x40($inp),$inp+ sha256msg1 @MSG[1],@MSG[0]+ sha256rnds2 $CDGH,$ABEF++ movdqa 2*32-0x80($Tbl),$Wi+ paddd @MSG[2],$Wi+ pshufb $TMP,@MSG[3]+ sha256rnds2 $ABEF,$CDGH # 8-11+ pshufd \$0x0e,$Wi,$Wi+ movdqa @MSG[3],$TMP+ palignr \$4,@MSG[2],$TMP+ nop+ paddd $TMP,@MSG[0]+ sha256msg1 @MSG[2],@MSG[1]+ sha256rnds2 $CDGH,$ABEF++ movdqa 3*32-0x80($Tbl),$Wi+ paddd @MSG[3],$Wi+ sha256msg2 @MSG[3],@MSG[0]+ sha256rnds2 $ABEF,$CDGH # 12-15+ pshufd \$0x0e,$Wi,$Wi+ movdqa @MSG[0],$TMP+ palignr \$4,@MSG[3],$TMP+ nop+ paddd $TMP,@MSG[1]+ sha256msg1 @MSG[3],@MSG[2]+ sha256rnds2 $CDGH,$ABEF+___+for($i=4;$i<16-3;$i++) {+$code.=<<___;+ movdqa $i*32-0x80($Tbl),$Wi+ paddd @MSG[0],$Wi+ sha256msg2 @MSG[0],@MSG[1]+ sha256rnds2 $ABEF,$CDGH # 16-19...+ pshufd \$0x0e,$Wi,$Wi+ movdqa @MSG[1],$TMP+ palignr \$4,@MSG[0],$TMP+ nop+ paddd $TMP,@MSG[2]+ sha256msg1 @MSG[0],@MSG[3]+ sha256rnds2 $CDGH,$ABEF+___+ push(@MSG,shift(@MSG));+}+$code.=<<___;+ movdqa 13*32-0x80($Tbl),$Wi+ paddd @MSG[0],$Wi+ sha256msg2 @MSG[0],@MSG[1]+ sha256rnds2 $ABEF,$CDGH # 52-55+ pshufd \$0x0e,$Wi,$Wi+ movdqa @MSG[1],$TMP+ palignr \$4,@MSG[0],$TMP+ sha256rnds2 $CDGH,$ABEF+ paddd $TMP,@MSG[2]++ movdqa 14*32-0x80($Tbl),$Wi+ paddd @MSG[1],$Wi+ sha256rnds2 $ABEF,$CDGH # 56-59+ pshufd \$0x0e,$Wi,$Wi+ sha256msg2 @MSG[1],@MSG[2]+ movdqa $BSWAP,$TMP+ sha256rnds2 $CDGH,$ABEF++ movdqa 15*32-0x80($Tbl),$Wi+ paddd @MSG[2],$Wi+ nop+ sha256rnds2 $ABEF,$CDGH # 60-63+ pshufd \$0x0e,$Wi,$Wi+ dec $num+ nop+ sha256rnds2 $CDGH,$ABEF++ paddd $CDGH_SAVE,$CDGH+ paddd $ABEF_SAVE,$ABEF+ jnz .Loop_shaext++ pshufd \$0xb1,$CDGH,$CDGH # DCHG+ pshufd \$0x1b,$ABEF,$TMP # FEBA+ pshufd \$0xb1,$ABEF,$ABEF # BAFE+ punpckhqdq $CDGH,$ABEF # DCBA+ palignr \$8,$TMP,$CDGH # HGFE++ movdqu $ABEF,($ctx)+ movdqu $CDGH,16($ctx)+___+$code.=<<___ if ($win64);+ movaps -0x50(%rbp),%xmm6+ movaps -0x40(%rbp),%xmm7+ movaps -0x30(%rbp),%xmm8+ movaps -0x20(%rbp),%xmm9+ movaps -0x10(%rbp),%xmm10+ mov %rbp,%rsp+___+$code.=<<___;+.cfi_def_cfa_register %rsp+ pop %rbp+.cfi_pop %rbp+.cfi_epilogue+ ret+.cfi_endproc+.size sha256_block_data_order_shaext,.-sha256_block_data_order_shaext+___+}}}+if ($SZ==8 && $shaext && $avx>1) {{{+######################################################################+# Intel SHA Extensions implementation of SHA512 update function.+#+my ($ctx,$inp,$num,$Tbl)=("%rdi","%rsi","%rdx","%rcx");++my ($Wi,$ABEF,$CDGH,$TMP,$BSWAP,$ABEF_SAVE,$CDGH_SAVE)=map("%ymm$_",(4..10));+my @MSG=map("%ymm$_",(0..3));++$code.=<<___;+.globl sha512_block_data_order_shaext+.type sha512_block_data_order_shaext,\@function,3,"unwind"+.align 64+sha512_block_data_order_shaext:+.cfi_startproc+ push %rbp+.cfi_push %rbp+ mov %rsp,%rbp+.cfi_def_cfa_register %rbp+.Lshaext_shortcut:+___+$code.=<<___ if ($win64);+ sub \$0x50,%rsp+.cfi_alloca 0x50+ movaps %xmm6,-0x50(%rbp)+ movaps %xmm7,-0x40(%rbp)+ movaps %xmm8,-0x30(%rbp)+ movaps %xmm9,-0x20(%rbp)+ movaps %xmm10,-0x10(%rbp)+.cfi_offset %xmm6-%xmm10,-0x60+___+$code.=<<___;+.cfi_end_prologue+ lea K512_nodup+0x80(%rip),$Tbl+ vmovdqu ($ctx),@MSG[0] # DCBA+ vmovdqu 32($ctx),@MSG[1] # HGFE+ vmovdqa -0xa0($Tbl),$BSWAP++ vpermq \$0b00011011,@MSG[0],@MSG[0] # ABCD+ vpblendd \$0b00001111,@MSG[1],@MSG[0],$ABEF # ABFE+ vpblendd \$0b00001111,@MSG[0],@MSG[1],$CDGH # HGCD+ vpermq \$0b11100001,$ABEF,$ABEF # ABEF+ vpermq \$0b01001011,$CDGH,$CDGH # CDGH+ jmp .Loop_shaext++.align 16+.Loop_shaext:+ vmovdqu ($inp),@MSG[0]+ vmovdqu 0x20($inp),@MSG[1]+ vmovdqu 0x40($inp),@MSG[2]+ vpshufb $BSWAP,@MSG[0],@MSG[0]+ vmovdqu 0x60($inp),@MSG[3]++ vpaddq 0*32-0x80($Tbl),@MSG[0],$Wi+ vpshufb $BSWAP,@MSG[1],@MSG[1]+ vmovdqa $CDGH,$CDGH_SAVE # offload+ vsha512rnds2 %x#$Wi,$ABEF,$CDGH # 0-3+ vextracti128 \$1,$Wi,%x#$Wi+ vmovdqa $ABEF,$ABEF_SAVE # offload+ vsha512rnds2 %x#$Wi,$CDGH,$ABEF++ vpaddq 1*32-0x80($Tbl),@MSG[1],$Wi+ vpshufb $BSWAP,@MSG[2],@MSG[2]+ vsha512rnds2 %x#$Wi,$ABEF,$CDGH # 4-7+ vextracti128 \$1,$Wi,%x#$Wi+ lea 0x80($inp),$inp+ vsha512msg1 @MSG[1],@MSG[0]+ vsha512rnds2 %x#$Wi,$CDGH,$ABEF++ vpaddq 2*32-0x80($Tbl),@MSG[2],$Wi+ vpshufb $BSWAP,@MSG[3],@MSG[3]+ vsha512rnds2 %x#$Wi,$ABEF,$CDGH # 8-11+ vextracti128 \$1,$Wi,%x#$Wi+ vpblendd \$0x03,@MSG[3],@MSG[2],$TMP+ vpermq \$0x39,$TMP,$TMP+ vpaddq $TMP,@MSG[0],@MSG[0]+ vsha512msg1 @MSG[2],@MSG[1]+ vsha512rnds2 %x#$Wi,$CDGH,$ABEF++ vpaddq 3*32-0x80($Tbl),@MSG[3],$Wi+ vsha512msg2 @MSG[3],@MSG[0]+ vsha512rnds2 %x#$Wi,$ABEF,$CDGH # 12-15+ vextracti128 \$1,$Wi,%x#$Wi+ vpblendd \$0x03,@MSG[0],@MSG[3],$TMP+ vpermq \$0x39,$TMP,$TMP+ vpaddq $TMP,@MSG[1],@MSG[1]+ vsha512msg1 @MSG[3],@MSG[2]+ vsha512rnds2 %x#$Wi,$CDGH,$ABEF+___+for($i=4;$i<20-3;$i++) {+$code.=<<___;+ vpaddq $i*32-0x80($Tbl),@MSG[0],$Wi+ vsha512msg2 @MSG[0],@MSG[1]+ vsha512rnds2 %x#$Wi,$ABEF,$CDGH # 16-19...+ vextracti128 \$1,$Wi,%x#$Wi+ vpblendd \$0x03,@MSG[1],@MSG[0],$TMP+ vpermq \$0x39,$TMP,$TMP+ vpaddq $TMP,@MSG[2],@MSG[2]+ vsha512msg1 @MSG[0],@MSG[3]+ vsha512rnds2 %x#$Wi,$CDGH,$ABEF+___+ push(@MSG,shift(@MSG));+}+$code.=<<___;+ vpaddq 17*32-0x80($Tbl),@MSG[0],$Wi+ vsha512msg2 @MSG[0],@MSG[1]+ vsha512rnds2 %x#$Wi,$ABEF,$CDGH # 68-71+ vextracti128 \$1,$Wi,%x#$Wi+ vpblendd \$0x03,@MSG[1],@MSG[0],$TMP+ vpermq \$0x39,$TMP,$TMP+ vsha512rnds2 %x#$Wi,$CDGH,$ABEF+ vpaddq $TMP,@MSG[2],@MSG[2]++ vpaddq 18*32-0x80($Tbl),@MSG[1],$Wi+ vsha512rnds2 %x#$Wi,$ABEF,$CDGH # 72-75+ vextracti128 \$1,$Wi,%x#$Wi+ vsha512msg2 @MSG[1],@MSG[2]+ vsha512rnds2 %x#$Wi,$CDGH,$ABEF++ vpaddq 19*32-0x80($Tbl),@MSG[2],$Wi+ vsha512rnds2 %x#$Wi,$ABEF,$CDGH # 76-79+ vextracti128 \$1,$Wi,%x#$Wi+ dec $num+ vsha512rnds2 %x#$Wi,$CDGH,$ABEF++ vpaddq $CDGH_SAVE,$CDGH,$CDGH+ vpaddq $ABEF_SAVE,$ABEF,$ABEF+ jnz .Loop_shaext++ vpermq \$0b01001011,$ABEF,$ABEF # EFBA+ vpblendd \$0b11110000,$CDGH,$ABEF,@MSG[0] # CDBA+ vpblendd \$0b11110000,$ABEF,$CDGH,@MSG[1] # EFGH+ vpermq \$0b10110100,@MSG[0],@MSG[0] # DCBA+ vpermq \$0b00011011,@MSG[1],@MSG[1] # HGFE++ vmovdqu @MSG[0],($ctx)+ vmovdqu @MSG[1],32($ctx)++ vzeroupper+___+$code.=<<___ if ($win64);+ movaps -0x50(%rbp),%xmm6+ movaps -0x40(%rbp),%xmm7+ movaps -0x30(%rbp),%xmm8+ movaps -0x20(%rbp),%xmm9+ movaps -0x10(%rbp),%xmm10+ mov %rbp,%rsp+___+$code.=<<___;+.cfi_def_cfa_register %rsp+ pop %rbp+.cfi_pop %rbp+.cfi_epilogue+ ret+.cfi_endproc+.size sha512_block_data_order_shaext,.-sha512_block_data_order_shaext+___+}}}+{{{++my $a4=$T1;+my ($a,$b,$c,$d,$e,$f,$g,$h);++sub AUTOLOAD() # thunk [simplified] 32-bit style perlasm+{ my $opcode = $AUTOLOAD; $opcode =~ s/.*:://;+ my $arg = pop;+ $arg = "\$$arg" if ($arg*1 eq $arg);+ $code .= "\t$opcode\t".join(',',$arg,reverse @_)."\n";+}++sub body_00_15 () {+ (+ '($a,$b,$c,$d,$e,$f,$g,$h)=@ROT;'.++ '&ror ($a0,$Sigma1[2]-$Sigma1[1])',+ '&mov ($a,$a1)',+ '&mov ($a4,$f)',++ '&ror ($a1,$Sigma0[2]-$Sigma0[1])',+ '&xor ($a0,$e)',+ '&xor ($a4,$g)', # f^g++ '&ror ($a0,$Sigma1[1]-$Sigma1[0])',+ '&xor ($a1,$a)',+ '&and ($a4,$e)', # (f^g)&e++ '&xor ($a0,$e)',+ '&add ($h,$SZ*($i&15)."(%rsp)")', # h+=X[i]+K[i]+ '&mov ($a2,$a)',++ '&xor ($a4,$g)', # Ch(e,f,g)=((f^g)&e)^g+ '&ror ($a1,$Sigma0[1]-$Sigma0[0])',+ '&xor ($a2,$b)', # a^b, b^c in next round++ '&add ($h,$a4)', # h+=Ch(e,f,g)+ '&ror ($a0,$Sigma1[0])', # Sigma1(e)+ '&and ($a3,$a2)', # (b^c)&(a^b)++ '&xor ($a1,$a)',+ '&add ($h,$a0)', # h+=Sigma1(e)+ '&xor ($a3,$b)', # Maj(a,b,c)=Ch(a^b,c,b)++ '&ror ($a1,$Sigma0[0])', # Sigma0(a)+ '&add ($d,$h)', # d+=h+ '&add ($h,$a3)', # h+=Maj(a,b,c)++ '&mov ($a0,$d)',+ '&add ($a1,$h);'. # h+=Sigma0(a)+ '($a2,$a3) = ($a3,$a2); unshift(@ROT,pop(@ROT)); $i++;'+ );+}++######################################################################+# SSSE3 code path+#+if ($SZ==4) { # SHA256 only+my $Tbl = $inp;+my $_ctx="-64(%rbp)";+my $_inp="-56(%rbp)";+my $_end="-48(%rbp)";+my $framesz=3*8+$win64*16*4;++my @X = map("%xmm$_",(0..3));+my ($t0,$t1,$t2,$t3, $t4,$t5) = map("%xmm$_",(4..9));++$code.=<<___;+.type ${func}_ssse3,\@function,3,"unwind"+.align 64+${func}_ssse3:+.cfi_startproc+ push %rbp+.cfi_push %rbp+ mov %rsp,%rbp+.cfi_def_cfa_register %rbp+.Lssse3_shortcut:+ push %rbx+.cfi_push %rbx+ push %r12+.cfi_push %r12+ push %r13+.cfi_push %r13+ push %r14+.cfi_push %r14+ push %r15+.cfi_push %r15+ shl \$4,%rdx # num*16+ sub \$$framesz,%rsp+.cfi_alloca $framesz+ lea ($inp,%rdx,$SZ),%rdx # inp+num*16*$SZ+ mov $ctx,$_ctx # save ctx, 1st arg+ #mov $inp,$_inp # save inp, 2nd arg+ mov %rdx,$_end # save end pointer, "3rd" arg+___+$code.=<<___ if ($win64);+ movaps %xmm6,-0x80(%rbp)+ movaps %xmm7,-0x70(%rbp)+ movaps %xmm8,-0x60(%rbp)+ movaps %xmm9,-0x50(%rbp)+.cfi_offset %xmm6-%xmm9,-0x90+___+$code.=<<___;+.cfi_end_prologue++ lea -16*$SZ(%rsp),%rsp+ mov $SZ*0($ctx),$A+ and \$-64,%rsp # align stack+ mov $SZ*1($ctx),$B+ mov $SZ*2($ctx),$C+ mov $SZ*3($ctx),$D+ mov $SZ*4($ctx),$E+ mov $SZ*5($ctx),$F+ mov $SZ*6($ctx),$G+ mov $SZ*7($ctx),$H+___++$code.=<<___;+ #movdqa $TABLE+`$SZ*2*$rounds`+32(%rip),$t4+ #movdqa $TABLE+`$SZ*2*$rounds`+64(%rip),$t5+ jmp .Lloop_ssse3+.align 16+.Lloop_ssse3:+ movdqa $TABLE+`$SZ*2*$rounds`(%rip),$t3+ mov $inp,$_inp # offload $inp+ movdqu 0x00($inp),@X[0]+ movdqu 0x10($inp),@X[1]+ movdqu 0x20($inp),@X[2]+ pshufb $t3,@X[0]+ movdqu 0x30($inp),@X[3]+ lea $TABLE(%rip),$Tbl+ pshufb $t3,@X[1]+ movdqa 0x00($Tbl),$t0+ movdqa 0x20($Tbl),$t1+ pshufb $t3,@X[2]+ paddd @X[0],$t0+ movdqa 0x40($Tbl),$t2+ pshufb $t3,@X[3]+ movdqa 0x60($Tbl),$t3+ paddd @X[1],$t1+ paddd @X[2],$t2+ paddd @X[3],$t3+ movdqa $t0,0x00(%rsp)+ mov $A,$a1+ movdqa $t1,0x10(%rsp)+ mov $B,$a3+ movdqa $t2,0x20(%rsp)+ xor $C,$a3 # magic+ movdqa $t3,0x30(%rsp)+ mov $E,$a0+ jmp .Lssse3_00_47++.align 16+.Lssse3_00_47:+ sub \$`-16*2*$SZ`,$Tbl # size optimization+___+sub Xupdate_256_SSSE3 () {+ (+ '&movdqa ($t0,@X[1]);',+ '&movdqa ($t3,@X[3])',+ '&palignr ($t0,@X[0],$SZ)', # X[1..4]+ '&palignr ($t3,@X[2],$SZ);', # X[9..12]+ '&movdqa ($t1,$t0)',+ '&movdqa ($t2,$t0);',+ '&psrld ($t0,$sigma0[2])',+ '&paddd (@X[0],$t3);', # X[0..3] += X[9..12]+ '&psrld ($t2,$sigma0[0])',+ '&pshufd ($t3,@X[3],0b11111010)',# X[14..15]+ '&pslld ($t1,8*$SZ-$sigma0[1]);'.+ '&pxor ($t0,$t2)',+ '&psrld ($t2,$sigma0[1]-$sigma0[0]);'.+ '&pxor ($t0,$t1)',+ '&pslld ($t1,$sigma0[1]-$sigma0[0]);'.+ '&pxor ($t0,$t2);',+ '&movdqa ($t2,$t3)',+ '&pxor ($t0,$t1);', # sigma0(X[1..4])+ '&psrld ($t3,$sigma1[2])',+ '&paddd (@X[0],$t0);', # X[0..3] += sigma0(X[1..4])+ '&psrlq ($t2,$sigma1[0])',+ '&pxor ($t3,$t2);',+ '&psrlq ($t2,$sigma1[1]-$sigma1[0])',+ '&pxor ($t3,$t2)',+ '&pshufb ($t3,$t4)', # sigma1(X[14..15])+ '&paddd (@X[0],$t3)', # X[0..1] += sigma1(X[14..15])+ '&pshufd ($t3,@X[0],0b01010000)',# X[16..17]+ '&movdqa ($t2,$t3);',+ '&psrld ($t3,$sigma1[2])',+ '&psrlq ($t2,$sigma1[0])',+ '&pxor ($t3,$t2);',+ '&psrlq ($t2,$sigma1[1]-$sigma1[0])',+ '&pxor ($t3,$t2);',+ '&movdqa ($t2,16*2*$j."($Tbl)")',+ '&pshufb ($t3,$t5)',+ '&paddd (@X[0],$t3)' # X[2..3] += sigma1(X[16..17])+ );+}++sub SSSE3_256_00_47 () {+my $j = shift;+my $body = shift;+my @X = @_;+my @insns = (&$body,&$body,&$body,&$body); # 104 instructions++ if (0) {+ foreach (Xupdate_256_SSSE3()) { # 36 instructions+ eval;+ eval(shift(@insns));+ eval(shift(@insns));+ eval(shift(@insns));+ }+ } else { # squeeze extra 4% on Westmere and 19% on Atom+ eval(shift(@insns)); #@+ &movdqa ($t0,@X[1]);+ eval(shift(@insns));+ eval(shift(@insns));+ &movdqa ($t3,@X[3]);+ eval(shift(@insns)); #@+ eval(shift(@insns));+ eval(shift(@insns));+ eval(shift(@insns)); #@+ eval(shift(@insns));+ &palignr ($t0,@X[0],$SZ); # X[1..4]+ eval(shift(@insns));+ eval(shift(@insns));+ &palignr ($t3,@X[2],$SZ); # X[9..12]+ eval(shift(@insns));+ eval(shift(@insns));+ eval(shift(@insns));+ eval(shift(@insns)); #@+ &movdqa ($t1,$t0);+ eval(shift(@insns));+ eval(shift(@insns));+ &movdqa ($t2,$t0);+ eval(shift(@insns)); #@+ eval(shift(@insns));+ &psrld ($t0,$sigma0[2]);+ eval(shift(@insns));+ eval(shift(@insns));+ eval(shift(@insns));+ &paddd (@X[0],$t3); # X[0..3] += X[9..12]+ eval(shift(@insns)); #@+ eval(shift(@insns));+ &psrld ($t2,$sigma0[0]);+ eval(shift(@insns));+ eval(shift(@insns));+ &pshufd ($t3,@X[3],0b11111010); # X[4..15]+ eval(shift(@insns));+ eval(shift(@insns)); #@+ &pslld ($t1,8*$SZ-$sigma0[1]);+ eval(shift(@insns));+ eval(shift(@insns));+ &pxor ($t0,$t2);+ eval(shift(@insns)); #@+ eval(shift(@insns));+ eval(shift(@insns));+ eval(shift(@insns)); #@+ &psrld ($t2,$sigma0[1]-$sigma0[0]);+ eval(shift(@insns));+ &pxor ($t0,$t1);+ eval(shift(@insns));+ eval(shift(@insns));+ &pslld ($t1,$sigma0[1]-$sigma0[0]);+ eval(shift(@insns));+ eval(shift(@insns));+ &pxor ($t0,$t2);+ eval(shift(@insns));+ eval(shift(@insns)); #@+ &movdqa ($t2,$t3);+ eval(shift(@insns));+ eval(shift(@insns));+ &pxor ($t0,$t1); # sigma0(X[1..4])+ eval(shift(@insns)); #@+ eval(shift(@insns));+ eval(shift(@insns));+ &psrld ($t3,$sigma1[2]);+ eval(shift(@insns));+ eval(shift(@insns));+ &paddd (@X[0],$t0); # X[0..3] += sigma0(X[1..4])+ eval(shift(@insns)); #@+ eval(shift(@insns));+ &psrlq ($t2,$sigma1[0]);+ eval(shift(@insns));+ eval(shift(@insns));+ eval(shift(@insns));+ &pxor ($t3,$t2);+ eval(shift(@insns)); #@+ eval(shift(@insns));+ eval(shift(@insns));+ eval(shift(@insns)); #@+ &psrlq ($t2,$sigma1[1]-$sigma1[0]);+ eval(shift(@insns));+ eval(shift(@insns));+ &pxor ($t3,$t2);+ eval(shift(@insns)); #@+ eval(shift(@insns));+ eval(shift(@insns));+ #&pshufb ($t3,$t4); # sigma1(X[14..15])+ &pshufd ($t3,$t3,0b10000000);+ eval(shift(@insns));+ eval(shift(@insns));+ eval(shift(@insns));+ &psrldq ($t3,8);+ eval(shift(@insns));+ eval(shift(@insns)); #@+ eval(shift(@insns));+ eval(shift(@insns));+ eval(shift(@insns)); #@+ &paddd (@X[0],$t3); # X[0..1] += sigma1(X[14..15])+ eval(shift(@insns));+ eval(shift(@insns));+ eval(shift(@insns));+ &pshufd ($t3,@X[0],0b01010000); # X[16..17]+ eval(shift(@insns));+ eval(shift(@insns)); #@+ eval(shift(@insns));+ &movdqa ($t2,$t3);+ eval(shift(@insns));+ eval(shift(@insns));+ &psrld ($t3,$sigma1[2]);+ eval(shift(@insns));+ eval(shift(@insns)); #@+ &psrlq ($t2,$sigma1[0]);+ eval(shift(@insns));+ eval(shift(@insns));+ &pxor ($t3,$t2);+ eval(shift(@insns)); #@+ eval(shift(@insns));+ eval(shift(@insns));+ eval(shift(@insns)); #@+ eval(shift(@insns));+ &psrlq ($t2,$sigma1[1]-$sigma1[0]);+ eval(shift(@insns));+ eval(shift(@insns));+ eval(shift(@insns));+ &pxor ($t3,$t2);+ eval(shift(@insns));+ eval(shift(@insns));+ eval(shift(@insns)); #@+ #&pshufb ($t3,$t5);+ &pshufd ($t3,$t3,0b00001000);+ eval(shift(@insns));+ eval(shift(@insns));+ &movdqa ($t2,16*2*$j."($Tbl)");+ eval(shift(@insns)); #@+ eval(shift(@insns));+ &pslldq ($t3,8);+ eval(shift(@insns));+ eval(shift(@insns));+ eval(shift(@insns));+ &paddd (@X[0],$t3); # X[2..3] += sigma1(X[16..17])+ eval(shift(@insns)); #@+ eval(shift(@insns));+ eval(shift(@insns));+ }+ &paddd ($t2,@X[0]);+ foreach (@insns) { eval; } # remaining instructions+ &movdqa (16*$j."(%rsp)",$t2);+}++ for ($i=0,$j=0; $j<4; $j++) {+ &SSSE3_256_00_47($j,\&body_00_15,@X);+ push(@X,shift(@X)); # rotate(@X)+ }+ &cmpb ($SZ-1+16*2*$SZ."($Tbl)",0);+ &jne (".Lssse3_00_47");++ for ($i=0; $i<16; ) {+ foreach(body_00_15()) { eval; }+ }+$code.=<<___;+ mov $_ctx,$ctx+ mov $a1,$A+ mov $_inp,$inp++ add $SZ*0($ctx),$A+ add $SZ*1($ctx),$B+ add $SZ*2($ctx),$C+ add $SZ*3($ctx),$D+ add $SZ*4($ctx),$E+ add $SZ*5($ctx),$F+ add $SZ*6($ctx),$G+ add $SZ*7($ctx),$H++ lea 16*$SZ($inp),$inp+ cmp $_end,$inp++ mov $A,$SZ*0($ctx)+ mov $B,$SZ*1($ctx)+ mov $C,$SZ*2($ctx)+ mov $D,$SZ*3($ctx)+ mov $E,$SZ*4($ctx)+ mov $F,$SZ*5($ctx)+ mov $G,$SZ*6($ctx)+ mov $H,$SZ*7($ctx)+ jb .Lloop_ssse3++___+$code.=<<___ if ($win64);+ movaps -0x80(%rbp),%xmm6+ movaps -0x70(%rbp),%xmm7+ movaps -0x60(%rbp),%xmm8+ movaps -0x50(%rbp),%xmm9+___+$code.=<<___;+ mov -40(%rbp),%r15+ mov -32(%rbp),%r14+ mov -24(%rbp),%r13+ mov -16(%rbp),%r12+ mov -8(%rbp),%rbx+ mov %rbp,%rsp+.cfi_def_cfa_register %rsp+ pop %rbp+.cfi_pop %rbp+.cfi_epilogue+ ret+.cfi_endproc+.size ${func}_ssse3,.-${func}_ssse3+___+}++if ($avx) {{+######################################################################+# XOP code path+#+if ($SZ==8) { # SHA512 only+my $Tbl=$inp;+my $_ctx="-64(%rbp)";+my $_inp="-56(%rbp)";+my $_end="-48(%rbp)";+my $framesz=3*8+$win64*16*6;++$code.=<<___;+.type ${func}_xop,\@function,3,"unwind"+.align 64+${func}_xop:+.cfi_startproc+ push %rbp+.cfi_push %rbp+ mov %rsp,%rbp+.cfi_def_cfa_register %rbp+.Lxop_shortcut:+ push %rbx+.cfi_push %rbx+ push %r12+.cfi_push %r12+ push %r13+.cfi_push %r13+ push %r14+.cfi_push %r14+ push %r15+.cfi_push %r15+ shl \$4,%rdx # num*16+ sub \$$framesz,%rsp+.cfi_alloca $framesz+ lea ($inp,%rdx,$SZ),%rdx # inp+num*16*$SZ+ mov $ctx,$_ctx # save ctx, 1st arg+ #mov $inp,$_inp # save inp, 2nd arg+ mov %rdx,$_end # save end pointer, "3rd" arg+___+$code.=<<___ if ($win64);+ movaps %xmm6,-0xa0(%rbp)+ movaps %xmm7,-0x90(%rbp)+ movaps %xmm8,-0x80(%rbp)+ movaps %xmm9,-0x70(%rbp)+.cfi_offset %xmm6-%xmm9,-0xb0+___+$code.=<<___ if ($win64 && $SZ>4);+ movaps %xmm10,-0x60(%rbp)+ movaps %xmm11,-0x50(%rbp)+.cfi_offset %xmm10-%xmm11,-0x70+___+$code.=<<___;+.cfi_end_prologue++ lea -16*$SZ(%rsp),%rsp+ vzeroupper+ and \$-64,%rsp # align stack+ mov $SZ*0($ctx),$A+ mov $SZ*1($ctx),$B+ mov $SZ*2($ctx),$C+ mov $SZ*3($ctx),$D+ mov $SZ*4($ctx),$E+ mov $SZ*5($ctx),$F+ mov $SZ*6($ctx),$G+ mov $SZ*7($ctx),$H+ jmp .Lloop_xop+___+ if ($SZ==4) { # SHA256+ my @X = map("%xmm$_",(0..3));+ my ($t0,$t1,$t2,$t3) = map("%xmm$_",(4..7));++$code.=<<___;+.align 16+.Lloop_xop:+ vmovdqa $TABLE+`$SZ*2*$rounds`(%rip),$t3+ mov $inp,$_inp # offload $inp+ vmovdqu 0x00($inp),@X[0]+ vmovdqu 0x10($inp),@X[1]+ vmovdqu 0x20($inp),@X[2]+ vmovdqu 0x30($inp),@X[3]+ vpshufb $t3,@X[0],@X[0]+ lea $TABLE(%rip),$Tbl+ vpshufb $t3,@X[1],@X[1]+ vpshufb $t3,@X[2],@X[2]+ vpaddd 0x00($Tbl),@X[0],$t0+ vpshufb $t3,@X[3],@X[3]+ vpaddd 0x20($Tbl),@X[1],$t1+ vpaddd 0x40($Tbl),@X[2],$t2+ vpaddd 0x60($Tbl),@X[3],$t3+ vmovdqa $t0,0x00(%rsp)+ mov $A,$a1+ vmovdqa $t1,0x10(%rsp)+ mov $B,$a3+ vmovdqa $t2,0x20(%rsp)+ xor $C,$a3 # magic+ vmovdqa $t3,0x30(%rsp)+ mov $E,$a0+ jmp .Lxop_00_47++.align 16+.Lxop_00_47:+ sub \$`-16*2*$SZ`,$Tbl # size optimization+___+sub XOP_256_00_47 () {+my $j = shift;+my $body = shift;+my @X = @_;+my @insns = (&$body,&$body,&$body,&$body); # 104 instructions++ &vpalignr ($t0,@X[1],@X[0],$SZ); # X[1..4]+ eval(shift(@insns));+ eval(shift(@insns));+ &vpalignr ($t3,@X[3],@X[2],$SZ); # X[9..12]+ eval(shift(@insns));+ eval(shift(@insns));+ &vprotd ($t1,$t0,8*$SZ-$sigma0[1]);+ eval(shift(@insns));+ eval(shift(@insns));+ &vpsrld ($t0,$t0,$sigma0[2]);+ eval(shift(@insns));+ eval(shift(@insns));+ &vpaddd (@X[0],@X[0],$t3); # X[0..3] += X[9..12]+ eval(shift(@insns));+ eval(shift(@insns));+ eval(shift(@insns));+ eval(shift(@insns));+ &vprotd ($t2,$t1,$sigma0[1]-$sigma0[0]);+ eval(shift(@insns));+ eval(shift(@insns));+ &vpxor ($t0,$t0,$t1);+ eval(shift(@insns));+ eval(shift(@insns));+ eval(shift(@insns));+ eval(shift(@insns));+ &vprotd ($t3,@X[3],8*$SZ-$sigma1[1]);+ eval(shift(@insns));+ eval(shift(@insns));+ &vpxor ($t0,$t0,$t2); # sigma0(X[1..4])+ eval(shift(@insns));+ eval(shift(@insns));+ &vpsrld ($t2,@X[3],$sigma1[2]);+ eval(shift(@insns));+ eval(shift(@insns));+ &vpaddd (@X[0],@X[0],$t0); # X[0..3] += sigma0(X[1..4])+ eval(shift(@insns));+ eval(shift(@insns));+ &vprotd ($t1,$t3,$sigma1[1]-$sigma1[0]);+ eval(shift(@insns));+ eval(shift(@insns));+ &vpxor ($t3,$t3,$t2);+ eval(shift(@insns));+ eval(shift(@insns));+ eval(shift(@insns));+ eval(shift(@insns));+ &vpxor ($t3,$t3,$t1); # sigma1(X[14..15])+ eval(shift(@insns));+ eval(shift(@insns));+ eval(shift(@insns));+ eval(shift(@insns));+ &vpsrldq ($t3,$t3,8);+ eval(shift(@insns));+ eval(shift(@insns));+ eval(shift(@insns));+ eval(shift(@insns));+ &vpaddd (@X[0],@X[0],$t3); # X[0..1] += sigma1(X[14..15])+ eval(shift(@insns));+ eval(shift(@insns));+ eval(shift(@insns));+ eval(shift(@insns));+ &vprotd ($t3,@X[0],8*$SZ-$sigma1[1]);+ eval(shift(@insns));+ eval(shift(@insns));+ &vpsrld ($t2,@X[0],$sigma1[2]);+ eval(shift(@insns));+ eval(shift(@insns));+ &vprotd ($t1,$t3,$sigma1[1]-$sigma1[0]);+ eval(shift(@insns));+ eval(shift(@insns));+ &vpxor ($t3,$t3,$t2);+ eval(shift(@insns));+ eval(shift(@insns));+ eval(shift(@insns));+ eval(shift(@insns));+ &vpxor ($t3,$t3,$t1); # sigma1(X[16..17])+ eval(shift(@insns));+ eval(shift(@insns));+ eval(shift(@insns));+ eval(shift(@insns));+ &vpslldq ($t3,$t3,8); # 22 instructions+ eval(shift(@insns));+ eval(shift(@insns));+ eval(shift(@insns));+ eval(shift(@insns));+ &vpaddd (@X[0],@X[0],$t3); # X[2..3] += sigma1(X[16..17])+ eval(shift(@insns));+ eval(shift(@insns));+ eval(shift(@insns));+ eval(shift(@insns));+ &vpaddd ($t2,@X[0],16*2*$j."($Tbl)");+ foreach (@insns) { eval; } # remaining instructions+ &vmovdqa (16*$j."(%rsp)",$t2);+}++ for ($i=0,$j=0; $j<4; $j++) {+ &XOP_256_00_47($j,\&body_00_15,@X);+ push(@X,shift(@X)); # rotate(@X)+ }+ &cmpb ($SZ-1+16*2*$SZ."($Tbl)",0);+ &jne (".Lxop_00_47");++ for ($i=0; $i<16; ) {+ foreach(body_00_15()) { eval; }+ }++ } else { # SHA512+ my @X = map("%xmm$_",(0..7));+ my ($t0,$t1,$t2,$t3) = map("%xmm$_",(8..11));++$code.=<<___;+.align 16+.Lloop_xop:+ vmovdqa $TABLE+`$SZ*2*$rounds`(%rip),$t3+ mov $inp,$_inp # offload $inp+ vmovdqu 0x00($inp),@X[0]+ vmovdqu 0x10($inp),@X[1]+ vmovdqu 0x20($inp),@X[2]+ vpshufb $t3,@X[0],@X[0]+ vmovdqu 0x30($inp),@X[3]+ vpshufb $t3,@X[1],@X[1]+ vmovdqu 0x40($inp),@X[4]+ vpshufb $t3,@X[2],@X[2]+ vmovdqu 0x50($inp),@X[5]+ vpshufb $t3,@X[3],@X[3]+ vmovdqu 0x60($inp),@X[6]+ vpshufb $t3,@X[4],@X[4]+ vmovdqu 0x70($inp),@X[7]+ lea $TABLE+0x80(%rip),$Tbl # size optimization+ vpshufb $t3,@X[5],@X[5]+ vpaddq -0x80($Tbl),@X[0],$t0+ vpshufb $t3,@X[6],@X[6]+ vpaddq -0x60($Tbl),@X[1],$t1+ vpshufb $t3,@X[7],@X[7]+ vpaddq -0x40($Tbl),@X[2],$t2+ vpaddq -0x20($Tbl),@X[3],$t3+ vmovdqa $t0,0x00(%rsp)+ vpaddq 0x00($Tbl),@X[4],$t0+ vmovdqa $t1,0x10(%rsp)+ vpaddq 0x20($Tbl),@X[5],$t1+ vmovdqa $t2,0x20(%rsp)+ vpaddq 0x40($Tbl),@X[6],$t2+ vmovdqa $t3,0x30(%rsp)+ vpaddq 0x60($Tbl),@X[7],$t3+ vmovdqa $t0,0x40(%rsp)+ mov $A,$a1+ vmovdqa $t1,0x50(%rsp)+ mov $B,$a3+ vmovdqa $t2,0x60(%rsp)+ xor $C,$a3 # magic+ vmovdqa $t3,0x70(%rsp)+ mov $E,$a0+ jmp .Lxop_00_47++.align 16+.Lxop_00_47:+ add \$`16*2*$SZ`,$Tbl+___+sub XOP_512_00_47 () {+my $j = shift;+my $body = shift;+my @X = @_;+my @insns = (&$body,&$body); # 52 instructions++ &vpalignr ($t0,@X[1],@X[0],$SZ); # X[1..2]+ eval(shift(@insns));+ eval(shift(@insns));+ &vpalignr ($t3,@X[5],@X[4],$SZ); # X[9..10]+ eval(shift(@insns));+ eval(shift(@insns));+ &vprotq ($t1,$t0,8*$SZ-$sigma0[1]);+ eval(shift(@insns));+ eval(shift(@insns));+ &vpsrlq ($t0,$t0,$sigma0[2]);+ eval(shift(@insns));+ eval(shift(@insns));+ &vpaddq (@X[0],@X[0],$t3); # X[0..1] += X[9..10]+ eval(shift(@insns));+ eval(shift(@insns));+ eval(shift(@insns));+ eval(shift(@insns));+ &vprotq ($t2,$t1,$sigma0[1]-$sigma0[0]);+ eval(shift(@insns));+ eval(shift(@insns));+ &vpxor ($t0,$t0,$t1);+ eval(shift(@insns));+ eval(shift(@insns));+ eval(shift(@insns));+ eval(shift(@insns));+ &vprotq ($t3,@X[7],8*$SZ-$sigma1[1]);+ eval(shift(@insns));+ eval(shift(@insns));+ &vpxor ($t0,$t0,$t2); # sigma0(X[1..2])+ eval(shift(@insns));+ eval(shift(@insns));+ &vpsrlq ($t2,@X[7],$sigma1[2]);+ eval(shift(@insns));+ eval(shift(@insns));+ &vpaddq (@X[0],@X[0],$t0); # X[0..1] += sigma0(X[1..2])+ eval(shift(@insns));+ eval(shift(@insns));+ &vprotq ($t1,$t3,$sigma1[1]-$sigma1[0]);+ eval(shift(@insns));+ eval(shift(@insns));+ &vpxor ($t3,$t3,$t2);+ eval(shift(@insns));+ eval(shift(@insns));+ eval(shift(@insns));+ eval(shift(@insns));+ &vpxor ($t3,$t3,$t1); # sigma1(X[14..15])+ eval(shift(@insns));+ eval(shift(@insns));+ eval(shift(@insns));+ eval(shift(@insns));+ &vpaddq (@X[0],@X[0],$t3); # X[0..1] += sigma1(X[14..15])+ eval(shift(@insns));+ eval(shift(@insns));+ eval(shift(@insns));+ eval(shift(@insns));+ &vpaddq ($t2,@X[0],16*2*$j-0x80."($Tbl)");+ foreach (@insns) { eval; } # remaining instructions+ &vmovdqa (16*$j."(%rsp)",$t2);+}++ for ($i=0,$j=0; $j<8; $j++) {+ &XOP_512_00_47($j,\&body_00_15,@X);+ push(@X,shift(@X)); # rotate(@X)+ }+ &cmpb ($SZ-1+16*2*$SZ-0x80."($Tbl)",0);+ &jne (".Lxop_00_47");++ for ($i=0; $i<16; ) {+ foreach(body_00_15()) { eval; }+ }+}+$code.=<<___;+ mov $_ctx,$ctx+ mov $a1,$A+ mov $_inp,$inp++ add $SZ*0($ctx),$A+ add $SZ*1($ctx),$B+ add $SZ*2($ctx),$C+ add $SZ*3($ctx),$D+ add $SZ*4($ctx),$E+ add $SZ*5($ctx),$F+ add $SZ*6($ctx),$G+ add $SZ*7($ctx),$H++ lea 16*$SZ($inp),$inp+ cmp $_end,$inp++ mov $A,$SZ*0($ctx)+ mov $B,$SZ*1($ctx)+ mov $C,$SZ*2($ctx)+ mov $D,$SZ*3($ctx)+ mov $E,$SZ*4($ctx)+ mov $F,$SZ*5($ctx)+ mov $G,$SZ*6($ctx)+ mov $H,$SZ*7($ctx)+ jb .Lloop_xop++ vzeroupper+___+$code.=<<___ if ($win64);+ movaps -0xa0(%rbp),%xmm6+ movaps -0x90(%rbp),%xmm7+ movaps -0x80(%rbp),%xmm8+ movaps -0x70(%rbp),%xmm9+___+$code.=<<___ if ($win64 && $SZ>4);+ movaps -0x60(%rbp),%xmm10+ movaps -0x50(%rbp),%xmm11+___+$code.=<<___;+ mov -40(%rbp),%r15+ mov -32(%rbp),%r14+ mov -24(%rbp),%r13+ mov -16(%rbp),%r12+ mov -8(%rbp),%rbx+ mov %rbp,%rsp+.cfi_def_cfa_register %rsp+ pop %rbp+.cfi_pop %rbp+.cfi_epilogue+ ret+.cfi_endproc+.size ${func}_xop,.-${func}_xop+___+}+######################################################################+# AVX+shrd code path+#+my $Tbl=$inp;+my $_ctx="-64(%rbp)";+my $_inp="-56(%rbp)";+my $_end="-48(%rbp)";+my $framesz=3*8+$win64*16*6;++local *ror = sub { &shrd(@_[0],@_) };++$code.=<<___;+.type ${func}_avx,\@function,3,"unwind"+.align 64+${func}_avx:+.cfi_startproc+ push %rbp+.cfi_push %rbp+ mov %rsp,%rbp+.cfi_def_cfa_register %rbp+.Lavx_shortcut:+ push %rbx+.cfi_push %rbx+ push %r12+.cfi_push %r12+ push %r13+.cfi_push %r13+ push %r14+.cfi_push %r14+ push %r15+.cfi_push %r15+ shl \$4,%rdx # num*16+ sub \$$framesz,%rsp+.cfi_alloca $framesz+ lea ($inp,%rdx,$SZ),%rdx # inp+num*16*$SZ+ mov $ctx,$_ctx # save ctx, 1st arg+ #mov $inp,$_inp # save inp, 2nd arg+ mov %rdx,$_end # save end pointer, "3rd" arg+___+$code.=<<___ if ($win64);+ movaps %xmm6,-0xa0(%rbp)+ movaps %xmm7,-0x90(%rbp)+ movaps %xmm8,-0x80(%rbp)+ movaps %xmm9,-0x70(%rbp)+.cfi_offset %xmm6-%xmm9,-0xb0+___+$code.=<<___ if ($win64 && $SZ>4);+ movaps %xmm10,-0x60(%rbp)+ movaps %xmm11,-0x50(%rbp)+.cfi_offset %xmm10-%xmm11,-0x70+___+$code.=<<___;+.cfi_end_prologue++ lea -16*$SZ(%rsp),%rsp+ vzeroupper+ and \$-64,%rsp # align stack+ mov $SZ*0($ctx),$A+ mov $SZ*1($ctx),$B+ mov $SZ*2($ctx),$C+ mov $SZ*3($ctx),$D+ mov $SZ*4($ctx),$E+ mov $SZ*5($ctx),$F+ mov $SZ*6($ctx),$G+ mov $SZ*7($ctx),$H+___+ if ($SZ==4) { # SHA256+ my @X = map("%xmm$_",(0..3));+ my ($t0,$t1,$t2,$t3, $t4,$t5) = map("%xmm$_",(4..9));++$code.=<<___;+ vmovdqa $TABLE+`$SZ*2*$rounds`+32(%rip),$t4+ vmovdqa $TABLE+`$SZ*2*$rounds`+64(%rip),$t5+ jmp .Lloop_avx+.align 16+.Lloop_avx:+ vmovdqa $TABLE+`$SZ*2*$rounds`(%rip),$t3+ mov $inp,$_inp # offload $inp+ vmovdqu 0x00($inp),@X[0]+ vmovdqu 0x10($inp),@X[1]+ vmovdqu 0x20($inp),@X[2]+ vmovdqu 0x30($inp),@X[3]+ vpshufb $t3,@X[0],@X[0]+ lea $TABLE(%rip),$Tbl+ vpshufb $t3,@X[1],@X[1]+ vpshufb $t3,@X[2],@X[2]+ vpaddd 0x00($Tbl),@X[0],$t0+ vpshufb $t3,@X[3],@X[3]+ vpaddd 0x20($Tbl),@X[1],$t1+ vpaddd 0x40($Tbl),@X[2],$t2+ vpaddd 0x60($Tbl),@X[3],$t3+ vmovdqa $t0,0x00(%rsp)+ mov $A,$a1+ vmovdqa $t1,0x10(%rsp)+ mov $B,$a3+ vmovdqa $t2,0x20(%rsp)+ xor $C,$a3 # magic+ vmovdqa $t3,0x30(%rsp)+ mov $E,$a0+ jmp .Lavx_00_47++.align 16+.Lavx_00_47:+ sub \$`-16*2*$SZ`,$Tbl # size optimization+___+sub Xupdate_256_AVX () {+ (+ '&vpalignr ($t0,@X[1],@X[0],$SZ)', # X[1..4]+ '&vpalignr ($t3,@X[3],@X[2],$SZ)', # X[9..12]+ '&vpsrld ($t2,$t0,$sigma0[0]);',+ '&vpaddd (@X[0],@X[0],$t3)', # X[0..3] += X[9..12]+ '&vpsrld ($t3,$t0,$sigma0[2])',+ '&vpslld ($t1,$t0,8*$SZ-$sigma0[1]);',+ '&vpxor ($t0,$t3,$t2)',+ '&vpshufd ($t3,@X[3],0b11111010)',# X[14..15]+ '&vpsrld ($t2,$t2,$sigma0[1]-$sigma0[0]);',+ '&vpxor ($t0,$t0,$t1)',+ '&vpslld ($t1,$t1,$sigma0[1]-$sigma0[0]);',+ '&vpxor ($t0,$t0,$t2)',+ '&vpsrld ($t2,$t3,$sigma1[2]);',+ '&vpxor ($t0,$t0,$t1)', # sigma0(X[1..4])+ '&vpsrlq ($t3,$t3,$sigma1[0]);',+ '&vpaddd (@X[0],@X[0],$t0)', # X[0..3] += sigma0(X[1..4])+ '&vpxor ($t2,$t2,$t3);',+ '&vpsrlq ($t3,$t3,$sigma1[1]-$sigma1[0])',+ '&vpxor ($t2,$t2,$t3)',+ '&vpshufb ($t2,$t2,$t4)', # sigma1(X[14..15])+ '&vpaddd (@X[0],@X[0],$t2)', # X[0..1] += sigma1(X[14..15])+ '&vpshufd ($t3,@X[0],0b01010000)',# X[16..17]+ '&vpsrld ($t2,$t3,$sigma1[2])',+ '&vpsrlq ($t3,$t3,$sigma1[0])',+ '&vpxor ($t2,$t2,$t3);',+ '&vpsrlq ($t3,$t3,$sigma1[1]-$sigma1[0])',+ '&vpxor ($t2,$t2,$t3)',+ '&vpshufb ($t2,$t2,$t5)',+ '&vpaddd (@X[0],@X[0],$t2)' # X[2..3] += sigma1(X[16..17])+ );+}++sub AVX_256_00_47 () {+my $j = shift;+my $body = shift;+my @X = @_;+my @insns = (&$body,&$body,&$body,&$body); # 104 instructions++ foreach (Xupdate_256_AVX()) { # 29 instructions+ eval;+ eval(shift(@insns));+ eval(shift(@insns));+ eval(shift(@insns));+ }+ &vpaddd ($t2,@X[0],16*2*$j."($Tbl)");+ foreach (@insns) { eval; } # remaining instructions+ &vmovdqa (16*$j."(%rsp)",$t2);+}++ for ($i=0,$j=0; $j<4; $j++) {+ &AVX_256_00_47($j,\&body_00_15,@X);+ push(@X,shift(@X)); # rotate(@X)+ }+ &cmpb ($SZ-1+16*2*$SZ."($Tbl)",0);+ &jne (".Lavx_00_47");++ for ($i=0; $i<16; ) {+ foreach(body_00_15()) { eval; }+ }++ } else { # SHA512+ my @X = map("%xmm$_",(0..7));+ my ($t0,$t1,$t2,$t3) = map("%xmm$_",(8..11));++$code.=<<___;+ jmp .Lloop_avx+.align 16+.Lloop_avx:+ vmovdqa $TABLE+`$SZ*2*$rounds`(%rip),$t3+ mov $inp,$_inp # offload $inp+ vmovdqu 0x00($inp),@X[0]+ vmovdqu 0x10($inp),@X[1]+ vmovdqu 0x20($inp),@X[2]+ vpshufb $t3,@X[0],@X[0]+ vmovdqu 0x30($inp),@X[3]+ vpshufb $t3,@X[1],@X[1]+ vmovdqu 0x40($inp),@X[4]+ vpshufb $t3,@X[2],@X[2]+ vmovdqu 0x50($inp),@X[5]+ vpshufb $t3,@X[3],@X[3]+ vmovdqu 0x60($inp),@X[6]+ vpshufb $t3,@X[4],@X[4]+ vmovdqu 0x70($inp),@X[7]+ lea $TABLE+0x80(%rip),$Tbl # size optimization+ vpshufb $t3,@X[5],@X[5]+ vpaddq -0x80($Tbl),@X[0],$t0+ vpshufb $t3,@X[6],@X[6]+ vpaddq -0x60($Tbl),@X[1],$t1+ vpshufb $t3,@X[7],@X[7]+ vpaddq -0x40($Tbl),@X[2],$t2+ vpaddq -0x20($Tbl),@X[3],$t3+ vmovdqa $t0,0x00(%rsp)+ vpaddq 0x00($Tbl),@X[4],$t0+ vmovdqa $t1,0x10(%rsp)+ vpaddq 0x20($Tbl),@X[5],$t1+ vmovdqa $t2,0x20(%rsp)+ vpaddq 0x40($Tbl),@X[6],$t2+ vmovdqa $t3,0x30(%rsp)+ vpaddq 0x60($Tbl),@X[7],$t3+ vmovdqa $t0,0x40(%rsp)+ mov $A,$a1+ vmovdqa $t1,0x50(%rsp)+ mov $B,$a3+ vmovdqa $t2,0x60(%rsp)+ xor $C,$a3 # magic+ vmovdqa $t3,0x70(%rsp)+ mov $E,$a0+ jmp .Lavx_00_47++.align 16+.Lavx_00_47:+ add \$`16*2*$SZ`,$Tbl+___+sub Xupdate_512_AVX () {+ (+ '&vpalignr ($t0,@X[1],@X[0],$SZ)', # X[1..2]+ '&vpalignr ($t3,@X[5],@X[4],$SZ)', # X[9..10]+ '&vpsrlq ($t2,$t0,$sigma0[0])',+ '&vpaddq (@X[0],@X[0],$t3);', # X[0..1] += X[9..10]+ '&vpsrlq ($t3,$t0,$sigma0[2])',+ '&vpsllq ($t1,$t0,8*$SZ-$sigma0[1]);',+ '&vpxor ($t0,$t3,$t2)',+ '&vpsrlq ($t2,$t2,$sigma0[1]-$sigma0[0]);',+ '&vpxor ($t0,$t0,$t1)',+ '&vpsllq ($t1,$t1,$sigma0[1]-$sigma0[0]);',+ '&vpxor ($t0,$t0,$t2)',+ '&vpsrlq ($t3,@X[7],$sigma1[2]);',+ '&vpxor ($t0,$t0,$t1)', # sigma0(X[1..2])+ '&vpsllq ($t2,@X[7],8*$SZ-$sigma1[1]);',+ '&vpaddq (@X[0],@X[0],$t0)', # X[0..1] += sigma0(X[1..2])+ '&vpsrlq ($t1,@X[7],$sigma1[0]);',+ '&vpxor ($t3,$t3,$t2)',+ '&vpsllq ($t2,$t2,$sigma1[1]-$sigma1[0]);',+ '&vpxor ($t3,$t3,$t1)',+ '&vpsrlq ($t1,$t1,$sigma1[1]-$sigma1[0]);',+ '&vpxor ($t3,$t3,$t2)',+ '&vpxor ($t3,$t3,$t1)', # sigma1(X[14..15])+ '&vpaddq (@X[0],@X[0],$t3)', # X[0..1] += sigma1(X[14..15])+ );+}++sub AVX_512_00_47 () {+my $j = shift;+my $body = shift;+my @X = @_;+my @insns = (&$body,&$body); # 52 instructions++ foreach (Xupdate_512_AVX()) { # 23 instructions+ eval;+ eval(shift(@insns));+ eval(shift(@insns));+ }+ &vpaddq ($t2,@X[0],16*2*$j-0x80."($Tbl)");+ foreach (@insns) { eval; } # remaining instructions+ &vmovdqa (16*$j."(%rsp)",$t2);+}++ for ($i=0,$j=0; $j<8; $j++) {+ &AVX_512_00_47($j,\&body_00_15,@X);+ push(@X,shift(@X)); # rotate(@X)+ }+ &cmpb ($SZ-1+16*2*$SZ-0x80."($Tbl)",0);+ &jne (".Lavx_00_47");++ for ($i=0; $i<16; ) {+ foreach(body_00_15()) { eval; }+ }+}+$code.=<<___;+ mov $_ctx,$ctx+ mov $a1,$A+ mov $_inp,$inp++ add $SZ*0($ctx),$A+ add $SZ*1($ctx),$B+ add $SZ*2($ctx),$C+ add $SZ*3($ctx),$D+ add $SZ*4($ctx),$E+ add $SZ*5($ctx),$F+ add $SZ*6($ctx),$G+ add $SZ*7($ctx),$H++ lea 16*$SZ($inp),$inp+ cmp $_end,$inp++ mov $A,$SZ*0($ctx)+ mov $B,$SZ*1($ctx)+ mov $C,$SZ*2($ctx)+ mov $D,$SZ*3($ctx)+ mov $E,$SZ*4($ctx)+ mov $F,$SZ*5($ctx)+ mov $G,$SZ*6($ctx)+ mov $H,$SZ*7($ctx)+ jb .Lloop_avx++ vzeroupper+___+$code.=<<___ if ($win64);+ movaps -0xa0(%rbp),%xmm6+ movaps -0x90(%rbp),%xmm7+ movaps -0x80(%rbp),%xmm8+ movaps -0x70(%rbp),%xmm9+___+$code.=<<___ if ($win64 && $SZ>4);+ movaps -0x60(%rbp),%xmm10+ movaps -0x50(%rbp),%xmm11+___+$code.=<<___;+ mov -40(%rbp),%r15+ mov -32(%rbp),%r14+ mov -24(%rbp),%r13+ mov -16(%rbp),%r12+ mov -8(%rbp),%rbx+ mov %rbp,%rsp+.cfi_def_cfa_register %rsp+ pop %rbp+.cfi_pop %rbp+.cfi_epilogue+ ret+.cfi_endproc+.size ${func}_avx,.-${func}_avx+___++if ($avx>1) {{+######################################################################+# AVX2+BMI code path+#+my $Tbl=$inp;+my $_ctx="-64(%rbp)";+my $_inp="-56(%rbp)";+my $_end="-48(%rbp)";+my $framesz=3*8+$win64*16*6;+my $PUSH8=8*2*$SZ;+use integer;++sub bodyx_00_15 () {+ # at start $a1 should be zero, $a3 - $b^$c and $a4 copy of $f+ (+ '($a,$b,$c,$d,$e,$f,$g,$h)=@ROT;'.++ '&add ($h,(32*($i/(16/$SZ))+$SZ*($i%(16/$SZ)))%$PUSH8.$base)', # h+=X[i]+K[i]+ '&and ($a4,$e)', # f&e+ '&rorx ($a0,$e,$Sigma1[2])',+ '&rorx ($a2,$e,$Sigma1[1])',++ '&lea ($a,"($a,$a1)")', # h+=Sigma0(a) from the past+ '&lea ($h,"($h,$a4)")',+ '&andn ($a4,$e,$g)', # ~e&g+ '&xor ($a0,$a2)',++ '&rorx ($a1,$e,$Sigma1[0])',+ '&lea ($h,"($h,$a4)")', # h+=Ch(e,f,g)=(e&f)+(~e&g)+ '&xor ($a0,$a1)', # Sigma1(e)+ '&mov ($a2,$a)',++ '&rorx ($a4,$a,$Sigma0[2])',+ '&lea ($h,"($h,$a0)")', # h+=Sigma1(e)+ '&xor ($a2,$b)', # a^b, b^c in next round+ '&rorx ($a1,$a,$Sigma0[1])',++ '&rorx ($a0,$a,$Sigma0[0])',+ '&lea ($d,"($d,$h)")', # d+=h+ '&and ($a3,$a2)', # (b^c)&(a^b)+ '&xor ($a1,$a4)',++ '&xor ($a3,$b)', # Maj(a,b,c)=Ch(a^b,c,b)+ '&xor ($a1,$a0)', # Sigma0(a)+ '&lea ($h,"($h,$a3)");'. # h+=Maj(a,b,c)+ '&mov ($a4,$e)', # copy of f in future++ '($a2,$a3) = ($a3,$a2); unshift(@ROT,pop(@ROT)); $i++;'+ );+ # and at the finish one has to $a+=$a1+}++$code.=<<___;+.type ${func}_avx2,\@function,3,"unwind"+.align 64+${func}_avx2:+.cfi_startproc+ push %rbp+.cfi_push %rbp+ mov %rsp,%rbp+.cfi_def_cfa_register %rbp+.Lavx2_shortcut:+ push %rbx+.cfi_push %rbx+ push %r12+.cfi_push %r12+ push %r13+.cfi_push %r13+ push %r14+.cfi_push %r14+ push %r15+.cfi_push %r15+ shl \$4,%rdx # num*16+ sub \$$framesz,%rsp+.cfi_alloca $framesz+ lea ($inp,%rdx,$SZ),%rdx # inp+num*16*$SZ+ mov $ctx,$_ctx # save ctx, 1st arg+ #mov $inp,$_inp # save inp, 2nd arg+ mov %rdx,$_end # save end pointer, "3rd" arg+___+$code.=<<___ if ($win64);+ movaps %xmm6,-0xa0(%rbp)+ movaps %xmm7,-0x90(%rbp)+ movaps %xmm8,-0x80(%rbp)+ movaps %xmm9,-0x70(%rbp)+.cfi_offset %xmm6-%xmm9,-0xb0+___+$code.=<<___ if ($win64 && $SZ>4);+ movaps %xmm10,-0x60(%rbp)+ movaps %xmm11,-0x50(%rbp)+.cfi_offset %xmm10-%xmm11,-0x70+___+$code.=<<___;+.cfi_end_prologue++ lea -$PUSH8(%rsp),%rsp+ vzeroupper+ and \$-$PUSH8,%rsp # align stack+ sub \$-16*$SZ,$inp # inp++, size optimization+ mov $SZ*0($ctx),$A+ mov $inp,%r12 # borrow $T1+ mov $SZ*1($ctx),$B+ cmp %rdx,$inp # $_end+ mov $SZ*2($ctx),$C+ cmove %rsp,%r12 # next block or random data+ mov $SZ*3($ctx),$D+ mov $SZ*4($ctx),$E+ mov $SZ*5($ctx),$F+ mov $SZ*6($ctx),$G+ mov $SZ*7($ctx),$H+___+ if ($SZ==4) { # SHA256+ my @X = map("%ymm$_",(0..3));+ my ($t0,$t1,$t2,$t3, $t4,$t5) = map("%ymm$_",(4..9));++$code.=<<___;+ vmovdqa $TABLE+`$SZ*2*$rounds`+32(%rip),$t4+ vmovdqa $TABLE+`$SZ*2*$rounds`+64(%rip),$t5+ jmp .Loop_avx2+.align 16+.Loop_avx2:+ vmovdqa $TABLE+`$SZ*2*$rounds`(%rip),$t3+ mov $inp,$_inp # offload $inp+ vmovdqu -16*$SZ+0($inp),%xmm0+ vmovdqu -16*$SZ+16($inp),%xmm1+ vmovdqu -16*$SZ+32($inp),%xmm2+ vmovdqu -16*$SZ+48($inp),%xmm3+ lea $TABLE(%rip),$Tbl+ vinserti128 \$1,(%r12),@X[0],@X[0]+ vinserti128 \$1,16(%r12),@X[1],@X[1]+ vpshufb $t3,@X[0],@X[0]+ vinserti128 \$1,32(%r12),@X[2],@X[2]+ vpshufb $t3,@X[1],@X[1]+ vinserti128 \$1,48(%r12),@X[3],@X[3]++ vpshufb $t3,@X[2],@X[2]+ vpaddd 0x00($Tbl),@X[0],$t0+ vpshufb $t3,@X[3],@X[3]+ vpaddd 0x20($Tbl),@X[1],$t1+ vpaddd 0x40($Tbl),@X[2],$t2+ vpaddd 0x60($Tbl),@X[3],$t3+ vmovdqa $t0,0x00(%rsp)+ xor $a1,$a1+ vmovdqa $t1,0x20(%rsp)+ lea -$PUSH8(%rsp),%rsp+ mov $B,$a3+ vmovdqa $t2,0x00(%rsp)+ xor $C,$a3 # magic+ vmovdqa $t3,0x20(%rsp)+ mov $F,$a4+ sub \$-16*2*$SZ,$Tbl # size optimization+ jmp .Lavx2_00_47++.align 16+.Lavx2_00_47:+___++sub AVX2_256_00_47 () {+my $j = shift;+my $body = shift;+my @X = @_;+my @insns = (&$body,&$body,&$body,&$body); # 96 instructions+my $base = "+2*$PUSH8(%rsp)";++ &lea ("%rsp","-$PUSH8(%rsp)") if (($j%2)==0);+ foreach (Xupdate_256_AVX()) { # 29 instructions+ eval;+ eval(shift(@insns));+ eval(shift(@insns));+ eval(shift(@insns));+ }+ &vpaddd ($t2,@X[0],16*2*$j."($Tbl)");+ foreach (@insns) { eval; } # remaining instructions+ &vmovdqa ((32*$j)%$PUSH8."(%rsp)",$t2);+}++ for ($i=0,$j=0; $j<4; $j++) {+ &AVX2_256_00_47($j,\&bodyx_00_15,@X);+ push(@X,shift(@X)); # rotate(@X)+ }+ &lea ($Tbl,16*2*$SZ."($Tbl)");+ &cmpb (($SZ-1)."($Tbl)",0);+ &jne (".Lavx2_00_47");++ for ($i=0; $i<16; ) {+ my $base=$i<8?"+$PUSH8(%rsp)":"(%rsp)";+ foreach(bodyx_00_15()) { eval; }+ }+ } else { # SHA512+ my @X = map("%ymm$_",(0..7));+ my ($t0,$t1,$t2,$t3) = map("%ymm$_",(8..11));++$code.=<<___;+ jmp .Loop_avx2+.align 16+.Loop_avx2:+ vmovdqa $TABLE+`$SZ*2*$rounds`(%rip),$t2+ mov $inp,$_inp # offload $inp+ vmovdqu -16*$SZ($inp),%xmm0+ vmovdqu -16*$SZ+16($inp),%xmm1+ vmovdqu -16*$SZ+32($inp),%xmm2+ vmovdqu -16*$SZ+48($inp),%xmm3+ vmovdqu -16*$SZ+64($inp),%xmm4+ vmovdqu -16*$SZ+80($inp),%xmm5+ vmovdqu -16*$SZ+96($inp),%xmm6+ vmovdqu -16*$SZ+112($inp),%xmm7+ lea $TABLE+0x80(%rip),$Tbl # size optimization+ vinserti128 \$1,(%r12),@X[0],@X[0]+ vinserti128 \$1,16(%r12),@X[1],@X[1]+ vpshufb $t2,@X[0],@X[0]+ vinserti128 \$1,32(%r12),@X[2],@X[2]+ vpshufb $t2,@X[1],@X[1]+ vinserti128 \$1,48(%r12),@X[3],@X[3]+ vpshufb $t2,@X[2],@X[2]+ vinserti128 \$1,64(%r12),@X[4],@X[4]+ vpshufb $t2,@X[3],@X[3]+ vinserti128 \$1,80(%r12),@X[5],@X[5]+ vpshufb $t2,@X[4],@X[4]+ vinserti128 \$1,96(%r12),@X[6],@X[6]+ vpshufb $t2,@X[5],@X[5]+ vinserti128 \$1,112(%r12),@X[7],@X[7]++ vpaddq -0x80($Tbl),@X[0],$t0+ vpshufb $t2,@X[6],@X[6]+ vpaddq -0x60($Tbl),@X[1],$t1+ vpshufb $t2,@X[7],@X[7]+ vpaddq -0x40($Tbl),@X[2],$t2+ vpaddq -0x20($Tbl),@X[3],$t3+ vmovdqa $t0,0x00(%rsp)+ vpaddq 0x00($Tbl),@X[4],$t0+ vmovdqa $t1,0x20(%rsp)+ vpaddq 0x20($Tbl),@X[5],$t1+ vmovdqa $t2,0x40(%rsp)+ vpaddq 0x40($Tbl),@X[6],$t2+ vmovdqa $t3,0x60(%rsp)+ lea -$PUSH8(%rsp),%rsp+ vpaddq 0x60($Tbl),@X[7],$t3+ vmovdqa $t0,0x00(%rsp)+ xor $a1,$a1+ vmovdqa $t1,0x20(%rsp)+ mov $B,$a3+ vmovdqa $t2,0x40(%rsp)+ xor $C,$a3 # magic+ vmovdqa $t3,0x60(%rsp)+ mov $F,$a4+ add \$16*2*$SZ,$Tbl+ jmp .Lavx2_00_47++.align 16+.Lavx2_00_47:+___++sub AVX2_512_00_47 () {+my $j = shift;+my $body = shift;+my @X = @_;+my @insns = (&$body,&$body); # 48 instructions+my $base = "+2*$PUSH8(%rsp)";++ &lea ("%rsp","-$PUSH8(%rsp)") if (($j%4)==0);+ foreach (Xupdate_512_AVX()) { # 23 instructions+ eval;+ if ($_ !~ /\;$/) {+ eval(shift(@insns));+ eval(shift(@insns));+ eval(shift(@insns));+ }+ }+ &vpaddq ($t2,@X[0],16*2*$j-0x80."($Tbl)");+ foreach (@insns) { eval; } # remaining instructions+ &vmovdqa ((32*$j)%$PUSH8."(%rsp)",$t2);+}++ for ($i=0,$j=0; $j<8; $j++) {+ &AVX2_512_00_47($j,\&bodyx_00_15,@X);+ push(@X,shift(@X)); # rotate(@X)+ }+ &lea ($Tbl,16*2*$SZ."($Tbl)");+ &cmpb (($SZ-1-0x80)."($Tbl)",0);+ &jne (".Lavx2_00_47");++ for ($i=0; $i<16; ) {+ my $base=$i<8?"+$PUSH8(%rsp)":"(%rsp)";+ foreach(bodyx_00_15()) { eval; }+ }+}+$code.=<<___;+ mov $_ctx,$ctx+ add $a1,$A+ mov $_inp,$a4++ add $SZ*0($ctx),$A+ add $SZ*1($ctx),$B+ add $SZ*2($ctx),$C+ add $SZ*3($ctx),$D+ add $SZ*4($ctx),$E+ add $SZ*5($ctx),$F+ add $SZ*6($ctx),$G+ add $SZ*7($ctx),$H++ mov $A,$SZ*0($ctx)+ mov $B,$SZ*1($ctx)+ mov $C,$SZ*2($ctx)+ mov $D,$SZ*3($ctx)+ mov $E,$SZ*4($ctx)+ mov $F,$SZ*5($ctx)+ mov $G,$SZ*6($ctx)+ mov $H,$SZ*7($ctx)++ cmp $_end,$a4+ je .Ldone_avx2++ lea `2*$SZ*($rounds-8)`(%rsp),$Tbl+ xor $a1,$a1+ mov $B,$a3+ xor $C,$a3 # magic+ mov $F,$a4+ jmp .Lower_avx2+.align 16+.Lower_avx2:+___+ for ($i=0; $i<8; ) {+ my $base="+16($Tbl)";+ foreach(bodyx_00_15()) { eval; }+ }+$code.=<<___;+ lea -$PUSH8($Tbl),$Tbl+ cmp %rsp,$Tbl+ jae .Lower_avx2++ mov $_ctx,$ctx+ add $a1,$A+ mov $_inp,$inp+ lea `2*$SZ*($rounds-8)`(%rsp),%rsp++ add $SZ*0($ctx),$A+ add $SZ*1($ctx),$B+ add $SZ*2($ctx),$C+ add $SZ*3($ctx),$D+ add $SZ*4($ctx),$E+ add $SZ*5($ctx),$F+ lea `2*16*$SZ`($inp),$inp # inp+=2+ add $SZ*6($ctx),$G+ mov $inp,%r12+ add $SZ*7($ctx),$H+ cmp $_end,$inp++ mov $A,$SZ*0($ctx)+ cmove %rsp,%r12 # next block or stale data+ mov $B,$SZ*1($ctx)+ mov $C,$SZ*2($ctx)+ mov $D,$SZ*3($ctx)+ mov $E,$SZ*4($ctx)+ mov $F,$SZ*5($ctx)+ mov $G,$SZ*6($ctx)+ mov $H,$SZ*7($ctx)++ jbe .Loop_avx2++.Ldone_avx2:+ vzeroupper+___+$code.=<<___ if ($win64);+ movaps -0xa0(%rbp),%xmm6+ movaps -0x90(%rbp),%xmm7+ movaps -0x80(%rbp),%xmm8+ movaps -0x70(%rbp),%xmm9+___+$code.=<<___ if ($win64 && $SZ>4);+ movaps -0x60(%rbp),%xmm10+ movaps -0x50(%rbp),%xmm11+___+$code.=<<___;+ mov -40(%rbp),%r15+ mov -32(%rbp),%r14+ mov -24(%rbp),%r13+ mov -16(%rbp),%r12+ mov -8(%rbp),%rbx+ mov %rbp,%rsp+.cfi_def_cfa_register %rsp+ pop %rbp+.cfi_pop %rbp+.cfi_epilogue+ ret+.cfi_endproc+.size ${func}_avx2,.-${func}_avx2+___+}}+}}}}}++sub sha256op38 {+ my $instr = shift;+ my %opcodelet = (+ "sha256rnds2" => 0xcb,+ "sha256msg1" => 0xcc,+ "sha256msg2" => 0xcd );++ if (defined($opcodelet{$instr}) && @_[0] =~ /%xmm([0-7]),\s*%xmm([0-7])/) {+ my @opcode=(0x0f,0x38);+ push @opcode,$opcodelet{$instr};+ push @opcode,0xc0|($1&7)|(($2&7)<<3); # ModR/M+ return ".byte\t".join(',',@opcode);+ } else {+ return $instr."\t".@_[0];+ }+}++sub vsha512rnds2 {+ my $instr = shift;++ if (@_[0] =~ /%xmm([0-9]+),\s*%ymm([0-9]+),\s*%ymm([0-9]+)/) {+ my @opcode=(0xc4,0xe2,0x7f,0xcb);+ @opcode[1] ^= (($1>>3)<<5)|(($3>>3)<<7);+ @opcode[2] ^= $2<<3;+ push @opcode,0xc0|($1&7)|(($3&7)<<3); # ModR/M+ return ".byte\t".join(',',@opcode);+ } else {+ return $instr."\t".@_[0];+ }+}++sub vsha512msg {+ my $instr = shift;+ my $op = shift;++ if (@_[0] =~ /%[xy]mm([0-9]+),\s*%ymm([0-9]+)/) {+ my @opcode=(0xc4,0xe2,0x7f,0xcb+$op);+ @opcode[1] ^= (($1>>3)<<5)|(($2>>3)<<7);+ push @opcode,0xc0|($1&7)|(($2&7)<<3); # ModR/M+ return ".byte\t".join(',',@opcode);+ } else {+ return $instr.$op."\t".@_[0];+ }+}++foreach (split("\n",$code)) {+ s/\`([^\`]*)\`/eval $1/geo;+ s/%x#%[yz]/%x/go;++ s/\b(sha256[^\s]*)\s+(.*)/sha256op38($1,$2)/eo or+ s/\b(vsha512msg)([12])\s+(.*)/vsha512msg($1,$2,$3)/eo or+ s/\b(vsha512rnds2)\s+(.*)/vsha512rnds2($1,$2)/eo;++ print $_,"\n";+}+close STDOUT;
@@ -0,0 +1,1943 @@+#!/usr/bin/env perl++# Ascetic x86_64 AT&T to MASM/NASM assembler translator by @dot-asm.+#+# Why AT&T to MASM and not vice versa? Several reasons. Because AT&T+# format is way easier to parse. Because it's simpler to "gear" from+# Unix ABI to Windows one [see cross-reference "card" at the end of+# file]. Because Linux targets were available first...+#+# In addition the script also "distills" code suitable for GNU+# assembler, so that it can be compiled with more rigid assemblers,+# such as Solaris /usr/ccs/bin/as.+#+# This translator is not designed to convert *arbitrary* assembler+# code from AT&T format to MASM one. It's designed to convert just+# enough to provide for dual-ABI OpenSSL modules development...+# There *are* limitations and you might have to modify your assembler+# code or this script to achieve the desired result...+#+# Currently recognized limitations:+#+# - can't use multiple ops per line;+#+# Dual-ABI styling rules.+#+# 1. Adhere to Unix register and stack layout [see cross-reference+# ABI "card" at the end for explanation].+# 2. Forget about "red zone," stick to more traditional blended+# stack frame allocation. If volatile storage is actually required+# that is. If not, just leave the stack as is.+# 3. Functions tagged with ".type name,@function" get crafted with+# unified Win64 prologue and epilogue automatically. If you want+# to take care of ABI differences yourself, tag functions as+# ".type name,@abi-omnipotent" instead.+# 4. To optimize the Win64 prologue you can specify number of input+# arguments as ".type name,@function,N." Keep in mind that if N is+# larger than 6, then you *have to* write "abi-omnipotent" code,+# because >6 cases can't be addressed with unified prologue.+# 5. Name local labels as .L*, do *not* use dynamic labels such as 1:+# (sorry about latter).+# 6. Don't use [or hand-code with .byte] "rep ret." "ret" mnemonic is+# required to identify the spots, where to inject Win64 epilogue!+# But on the pros, it's then prefixed with rep automatically:-)+# 7. Stick to explicit ip-relative addressing. If you have to use+# GOTPCREL addressing, stick to mov symbol@GOTPCREL(%rip),%r??.+# Both are recognized and translated to proper Win64 addressing+# modes.+#+# 8. In order to provide for structured exception handling unified+# Win64 prologue copies %rsp value to %rax. [Unless function is+# tagged with additional .type tag.] For further details see SEH+# paragraph at the end.+# 9. .init segment is allowed to contain calls to functions only.+# a. If function accepts more than 4 arguments *and* >4th argument+# is declared as non 64-bit value, do clear its upper part.+++use strict;++my $flavour = shift;+my $output = shift;+if ($flavour =~ /\./) { $output = $flavour; undef $flavour; }++open STDOUT,">$output" || die "can't open $output: $!"+ if (defined($output));++my $gas=1; $gas=0 if ($output =~ /\.asm$/);+my $elf=1; $elf=0 if (!$gas);+my $dwarf=$elf;+my $win64=0;+my $prefix="";+my $decor=".L";++my $masmref=8 + 50727*2**-32; # 8.00.50727 shipped with VS2005+my $masm=0;+my $PTR=" PTR";++my $nasmref=2.03;+my $nasm=0;++if ($flavour eq "mingw64") { $gas=1; $elf=0; $win64=1;+ $prefix=`echo __USER_LABEL_PREFIX__ | \${CC:-false} -E -P -`;+ $prefix =~ s|\R$||; # Better chomp+ }+elsif ($flavour eq "macosx") { $gas=1; $elf=0; $prefix="_"; $decor="L\$"; }+elsif ($flavour eq "masm") { $gas=0; $elf=0; $masm=$masmref; $win64=1; $decor="\$L\$"; }+elsif ($flavour eq "nasm") { $gas=0; $elf=0; $nasm=$nasmref; $win64=1; $decor="\$L\$"; $PTR=""; }+elsif (!$gas)+{ if ($ENV{ASM} =~ m/nasm/ && `nasm -v` =~ m/version ([0-9]+)\.([0-9]+)/i)+ { $nasm = $1 + $2*0.01; $PTR=""; }+ elsif (`ml64 2>&1` =~ m/Version ([0-9]+)\.([0-9]+)(\.([0-9]+))?/)+ { $masm = $1 + $2*2**-16 + $4*2**-32; }+ die "no assembler found on %PATH%" if (!($nasm || $masm));+ $win64=1;+ $elf=0;+ $decor="\$L\$";+}+my $colon= $masm ? "::" : ":";++$dwarf=0 if($win64);++my $current_segment;+my $current_function;+my %globals;++{ package opcode; # pick up opcodes+ sub re {+ my ($class, $line) = @_;+ my $self = {};+ my $ret;++ if ($$line =~ /^([a-z][a-z0-9]*)/i) {+ bless $self,$class;+ $self->{op} = $1;+ $ret = $self;+ $$line = substr($$line,@+[0]); $$line =~ s/^\s+//;++ undef $self->{sz};+ if ($self->{op} =~ /^(movz)x?([bw]).*/) { # movz is pain...+ $self->{op} = $1;+ $self->{sz} = $2;+ } elsif ($self->{op} =~ /cmov[n]?[lb]$/) {+ # pass through+ } elsif ($self->{op} =~ /call|jmp/) {+ $self->{sz} = "";+ } elsif ($self->{op} =~ /^p/ && $' !~ /^(ush|op|insrw)/) { # SSEn+ $self->{sz} = "";+ } elsif ($self->{op} =~ /^[vk]/) { # VEX or k* such as kmov+ $self->{sz} = "";+ } elsif ($self->{op} =~ /mov[dq]/ && $$line =~ /%xmm/) {+ $self->{sz} = "";+ } elsif ($self->{op} =~ /([a-z]{3,})([qlwb])$/) {+ $self->{op} = $1;+ $self->{sz} = $2;+ }+ }+ $ret;+ }+ sub size {+ my ($self, $sz) = @_;+ $self->{sz} = $sz if (defined($sz) && !defined($self->{sz}));+ $self->{sz};+ }+ sub out {+ my $self = shift;+ if ($gas) {+ if ($self->{op} eq "movz") { # movz is pain...+ sprintf "%s%s%s",$self->{op},$self->{sz},shift;+ } elsif ($self->{op} =~ /^set/) {+ "$self->{op}";+ } elsif ($self->{op} eq "ret") {+ my $epilogue = "";+ if ($win64 && $current_function->{abi} eq "svr4"+ && !$current_function->{unwind}) {+ $epilogue = "movq 8(%rsp),%rdi\n\t" .+ "movq 16(%rsp),%rsi\n\t";+ }+ $epilogue . ".byte 0xf3,0xc3";+ } elsif ($self->{op} eq "call" && !$elf && $current_segment eq ".init") {+ ".p2align\t3\n\t.quad";+ } else {+ "$self->{op}$self->{sz}";+ }+ } else {+ $self->{op} =~ s/^movz/movzx/;+ if ($self->{op} eq "ret") {+ $self->{op} = "";+ if ($win64 && $current_function->{abi} eq "svr4"+ && !$current_function->{unwind}) {+ $self->{op} = "mov rdi,QWORD$PTR\[8+rsp\]\t;WIN64 epilogue\n\t".+ "mov rsi,QWORD$PTR\[16+rsp\]\n\t";+ }+ $self->{op} .= "DB\t0F3h,0C3h\t\t;repret";+ } elsif ($self->{op} =~ /^(pop|push)f/) {+ $self->{op} .= $self->{sz};+ } elsif ($self->{op} eq "call" && $current_segment eq ".CRT\$XCU") {+ $self->{op} = "\tDQ";+ }+ $self->{op};+ }+ }+ sub mnemonic {+ my ($self, $op) = @_;+ $self->{op}=$op if (defined($op));+ $self->{op};+ }+}+{ package const; # pick up constants, which start with $+ sub re {+ my ($class, $line) = @_;+ my $self = {};+ my $ret;++ if ($$line =~ /^\$([^,]+)/) {+ bless $self, $class;+ $self->{value} = $1;+ $ret = $self;+ $$line = substr($$line,@+[0]); $$line =~ s/^\s+//;+ }+ $ret;+ }+ sub out {+ my $self = shift;++ $self->{value} =~ s/\b(0b[0-1]+)/oct($1)/eig;+ if ($gas) {+ # Solaris /usr/ccs/bin/as can't handle multiplications+ # in $self->{value}+ my $value = $self->{value};+ no warnings; # oct might complain about overflow, ignore here...+ $value =~ s/(?<![\w\$\.])(0x?[0-9a-f]+)/oct($1)/egi;+ if ($value =~ s/([0-9]+\s*[\*\/\%]\s*[0-9]+)/eval($1)/eg) {+ $self->{value} = $value;+ }+ sprintf "\$%s",$self->{value};+ } else {+ my $value = $self->{value};+ $value =~ s/0x([0-9a-f]+)/0$1h/ig if ($masm);+ sprintf "%s",$value;+ }+ }+}+{ package ea; # pick up effective addresses: expr(%reg,%reg,scale)++ my %szmap = ( b=>"BYTE$PTR", w=>"WORD$PTR",+ l=>"DWORD$PTR", d=>"DWORD$PTR",+ q=>"QWORD$PTR", o=>"OWORD$PTR",+ x=>"XMMWORD$PTR", y=>"YMMWORD$PTR",+ z=>"ZMMWORD$PTR" ) if (!$gas);++ my %sifmap = ( ss=>"d", sd=>"q", # broadcast only+ i32x2=>"q", f32x2=>"q",+ i32x4=>"x", i64x2=>"x", i128=>"x",+ f32x4=>"x", f64x2=>"x", f128=>"x",+ i32x8=>"y", i64x4=>"y",+ f32x8=>"y", f64x4=>"y" ) if (!$gas);++ sub re {+ my ($class, $line, $opcode) = @_;+ my $self = {};+ my $ret;++ # optional * ----vvv--- appears in indirect jmp/call+ if ($$line =~ /^(\*?)([^\(,]*)\(([%\w,\s]+)\)((?:{[^}]+})*)/) {+ bless $self, $class;+ $self->{asterisk} = $1;+ $self->{label} = $2;+ ($self->{base},$self->{index},$self->{scale})=split(/(?:,\s*)/,$3);+ $self->{scale} = 1 if (!defined($self->{scale}));+ $self->{opmask} = $4;+ $ret = $self;+ $$line = substr($$line,@+[0]); $$line =~ s/^\s+//;++ if ($win64 && $self->{label} =~ s/\@GOTPCREL//) {+ die if ($opcode->mnemonic() ne "mov");+ $opcode->mnemonic("lea");+ }+ $self->{base} =~ s/^%//;+ $self->{index} =~ s/^%// if (defined($self->{index}));+ $self->{opcode} = $opcode;+ }+ $ret;+ }+ sub size {}+ sub out {+ my ($self, $sz) = @_;++ $self->{label} =~ s/([_a-z][_a-z0-9\$]*)/$globals{$1} or $1/gei;+ $self->{label} =~ s/\.L/$decor/g;++ # Silently convert all EAs to 64-bit. This is required for+ # elder GNU assembler and results in more compact code,+ # *but* most importantly AES module depends on this feature!+ $self->{index} =~ s/^[er](.?[0-9xpi])[d]?$/r\1/;+ $self->{base} =~ s/^[er](.?[0-9xpi])[d]?$/r\1/;++ # Solaris /usr/ccs/bin/as can't handle multiplications+ # in $self->{label}...+ use integer;+ $self->{label} =~ s/(?<![\w\$\.])(0x?[0-9a-f]+)/oct($1)/egi;+ $self->{label} =~ s/\b([0-9]+\s*[\*\/\%]\s*[0-9]+)\b/eval($1)/eg;++ # Some assemblers insist on signed presentation of 32-bit+ # offsets, but sign extension is a tricky business in perl...+ $self->{label} =~ s/\b([0-9]+)\b/unpack("l",pack("L",$1))/eg;++ # if base register is %rbp or %r13, see if it's possible to+ # flip base and index registers [for better performance]+ if (!$self->{label} && $self->{index} && $self->{scale}==1 &&+ $self->{base} =~ /(rbp|r13)/) {+ $self->{base} = $self->{index}; $self->{index} = $1;+ }++ if ($gas) {+ $self->{label} =~ s/^___imp_/__imp__/ if ($flavour eq "mingw64");++ if (defined($self->{index})) {+ sprintf "%s%s(%s,%%%s,%d)%s",+ $self->{asterisk},$self->{label},+ $self->{base}?"%$self->{base}":"",+ $self->{index},$self->{scale},+ $self->{opmask};+ } else {+ sprintf "%s%s(%%%s)%s", $self->{asterisk},$self->{label},+ $self->{base},$self->{opmask};+ }+ } else {+ $self->{label} =~ s/\./\$/g;+ $self->{label} =~ s/(?<![\w\$\.])0x([0-9a-f]+)/0$1h/ig;+ $self->{label} = "($self->{label})" if ($self->{label} =~ /[\*\+\-\/]/);++ my $mnemonic = $self->{opcode}->mnemonic();+ ($self->{asterisk}) && ($sz="q") ||+ ($mnemonic =~ /^v?mov([qd])$/) && ($sz=$1) ||+ ($mnemonic =~ /^v?pinsr([qdwb])$/) && ($sz=$1) ||+ ($mnemonic =~ /^vpbroadcast([qdwb])$/) && ($sz=$1) ||+ ($mnemonic =~ /^v(?:broadcast|extract|insert)([sif]\w+)$/)+ && ($sz=$sifmap{$1});++ $self->{opmask} =~ s/%(k[0-7])/$1/;++ if (defined($self->{index})) {+ sprintf "%s[%s%s*%d%s]%s",$szmap{$sz},+ $self->{label}?"$self->{label}+":"",+ $self->{index},$self->{scale},+ $self->{base}?"+$self->{base}":"",+ $self->{opmask};+ } elsif ($self->{base} eq "rip") {+ sprintf "%s[%s]",$szmap{$sz},$self->{label};+ } else {+ sprintf "%s[%s%s]%s", $szmap{$sz},+ $self->{label}?"$self->{label}+":"",+ $self->{base},$self->{opmask};+ }+ }+ }+}+{ package register; # pick up registers, which start with %.+ sub re {+ my ($class, $line, $opcode) = @_;+ my $self = {};+ my $ret;++ # optional * ----vvv--- appears in indirect jmp/call+ if ($$line =~ /^(\*?)%(\w+)((?:{[^}]+})*)/) {+ bless $self,$class;+ $self->{asterisk} = $1;+ $self->{value} = $2;+ $self->{opmask} = $3;+ $opcode->size($self->size());+ $ret = $self;+ $$line = substr($$line,@+[0]); $$line =~ s/^\s+//;+ }+ $ret;+ }+ sub size {+ my $self = shift;+ my $ret;++ if ($self->{value} =~ /^r[\d]+b$/i) { $ret="b"; }+ elsif ($self->{value} =~ /^r[\d]+w$/i) { $ret="w"; }+ elsif ($self->{value} =~ /^r[\d]+d$/i) { $ret="l"; }+ elsif ($self->{value} =~ /^r[\w]+$/i) { $ret="q"; }+ elsif ($self->{value} =~ /^[a-d][hl]$/i){ $ret="b"; }+ elsif ($self->{value} =~ /^[\w]{2}l$/i) { $ret="b"; }+ elsif ($self->{value} =~ /^[\w]{2}$/i) { $ret="w"; }+ elsif ($self->{value} =~ /^e[a-z]{2}$/i){ $ret="l"; }++ $ret;+ }+ sub out {+ my $self = shift;+ if ($gas) { sprintf "%s%%%s%s", $self->{asterisk},+ $self->{value},+ $self->{opmask}; }+ else { $self->{opmask} =~ s/%(k[0-7])/$1/;+ $self->{value}.$self->{opmask}; }+ }+}+{ package label; # pick up labels, which end with :+ sub re {+ my ($class, $line) = @_;+ my $self = {};+ my $ret;++ if ($$line =~ /(^[\.\w\$]+)\:/) {+ bless $self,$class;+ $self->{value} = $1;+ $ret = $self;+ $$line = substr($$line,@+[0]); $$line =~ s/^\s+//;++ $self->{value} =~ s/^\.L/$decor/;+ }+ $ret;+ }+ sub win64_args {+ my $narg = $current_function->{narg} // 6;+ return undef if ($narg < 0);+ my $arg5 = 4*8 - cfi_directive::cfa_rsp();+ my $arg6 = $arg5 + 8;+ my $args;+ if ($gas) {+ $args .= " movq %rcx,%rdi\n" if ($narg>0);+ $args .= " movq %rdx,%rsi\n" if ($narg>1);+ $args .= " movq %r8,%rdx\n" if ($narg>2);+ $args .= " movq %r9,%rcx\n" if ($narg>3);+ $args .= " movq $arg5(%rsp),%r8\n" if ($narg>4);+ $args .= " movq $arg6(%rsp),%r9\n" if ($narg>5);+ } else {+ $args .= " mov rdi,rcx\n" if ($narg>0);+ $args .= " mov rsi,rdx\n" if ($narg>1);+ $args .= " mov rdx,r8\n" if ($narg>2);+ $args .= " mov rcx,r9\n" if ($narg>3);+ $args .= " mov r8,QWORD$PTR\[$arg5+rsp\]\n" if ($narg>4);+ $args .= " mov r9,QWORD$PTR\[$arg6+rsp\]\n" if ($narg>5);+ }+ $current_function->{narg} = -1;+ $args;+ }+ sub out {+ my $self = shift;++ if ($gas) {+ my $func = ($globals{$self->{value}} or $self->{value}) . ":";+ if ($current_function->{name} eq $self->{value}) {+ $current_function->{pc} = 0;+ $func .= "\n.cfi_".cfi_directive::startproc() if ($dwarf);+ $func .= "\n .byte 0xf3,0x0f,0x1e,0xfa\n"; # endbranch+ if ($win64) {+ if ($current_function->{abi} eq "svr4") {+ my $fp = $current_function->{unwind} ? "%r11" : "%rax";+ $func .= " movq %rdi,8(%rsp)\n";+ $func .= " movq %rsi,16(%rsp)\n";+ $func .= " movq %rsp,$fp\n";+ $func .= "${decor}SEH_begin_$current_function->{name}:\n";+ } elsif ($current_function->{unwind}) {+ $func .= " movq %rsp,%r11\n";+ $func .= "${decor}SEH_begin_$current_function->{name}:\n";+ }+ }+ } elsif ($win64 && $current_function->{abi} eq "svr4"+ && $current_function->{pc} >= 0) {+ $func = win64_args().$func;+ }+ $func;+ } elsif ($self->{value} ne "$current_function->{name}") {+ my $func;+ if ($win64 && $current_function->{abi} eq "svr4"+ && $current_function->{pc} >= 0) {+ $func = win64_args();+ }+ $func .= $self->{value} . $colon;+ $func;+ } else {+ $current_function->{pc} = 0;+ my $func = "$current_function->{name}" .+ ($nasm ? ":" : "\tPROC $current_function->{scope}") .+ "\n";+ $func .= " DB 243,15,30,250\n"; # endbranch+ if ($current_function->{abi} eq "svr4") {+ my $fp = $current_function->{unwind} ? "r11" : "rax";+ $func .= " mov QWORD$PTR\[8+rsp\],rdi\t;WIN64 prologue\n";+ $func .= " mov QWORD$PTR\[16+rsp\],rsi\n";+ $func .= " mov $fp,rsp\n";+ $func .= "${decor}SEH_begin_$current_function->{name}${colon}\n";+ } elsif ($current_function->{unwind}) {+ $func .= " mov r11,rsp\n";+ $func .= "${decor}SEH_begin_$current_function->{name}${colon}\n";+ }+ $func;+ }+ }+}+{ package expr; # pick up expressions+ sub re {+ my ($class, $line, $opcode) = @_;+ my $self = {};+ my $ret;++ if ($$line =~ /(^[^,]+)/) {+ bless $self,$class;+ $self->{value} = $1;+ $ret = $self;+ $$line = substr($$line,@+[0]); $$line =~ s/^\s+//;++ $self->{value} =~ s/\@PLT// if (!$elf);+ $self->{value} =~ s/([_a-z][_a-z0-9\$]*)/$globals{$1} or $1/gei;+ $self->{value} =~ s/\.L/$decor/g;+ $self->{opcode} = $opcode;+ }+ $ret;+ }+ sub out {+ my $self = shift;+ $self->{value};+ }+}++my @xdata_seg = (".section .xdata", ".align 8");+my @pdata_seg = (".section .pdata", ".align 4");++{ package cfi_directive;+ # CFI directives annotate instructions that are significant for+ # stack unwinding procedure compliant with DWARF specification,+ # see http://dwarfstd.org/. Besides naturally expected for this+ # script platform-specific filtering function, this module adds+ # four auxiliary synthetic directives not recognized by [GNU]+ # assembler:+ #+ # - .cfi_push to annotate push instructions in prologue, which+ # translates to .cfi_adjust_cfa_offset (if needed) and+ # .cfi_offset;+ # - .cfi_pop to annotate pop instructions in epilogue, which+ # translates to .cfi_adjust_cfa_offset (if needed) and+ # .cfi_restore;+ # - .cfi_alloca to annotate stack pointer adjustments, which+ # translates to .cfi_adjust_cfa_offset as needed;+ # - [and most notably] .cfi_cfa_expression which encodes+ # DW_CFA_def_cfa_expression and passes it to .cfi_escape as+ # byte vector;+ #+ # CFA expressions were introduced in DWARF specification version+ # 3 and describe how to deduce CFA, Canonical Frame Address. This+ # becomes handy if your stack frame is variable and you can't+ # spare register for [previous] frame pointer. Suggested directive+ # syntax is made-up mix of DWARF operator suffixes [subset of]+ # and references to registers with optional bias. Following example+ # describes offloaded *original* stack pointer at specific offset+ # from *current* stack pointer:+ #+ # .cfi_cfa_expression %rsp+40,deref,+8+ #+ # Final +8 has everything to do with the fact that CFA is defined+ # as reference to top of caller's stack, and on x86_64 call to+ # subroutine pushes 8-byte return address. In other words original+ # stack pointer upon entry to a subroutine is 8 bytes off from CFA.+ #+ # In addition the .cfi directives are re-purposed even for Win64+ # stack unwinding. Two more synthetic directives were added:+ #+ # - .cfi_end_prologue to denote point when all non-volatile+ # registers are saved and stack or [chosen] frame pointer is+ # stable;+ # - .cfi_epilogue to denote point when all non-volatile registers+ # are restored [and it even adds missing .cfi_restore-s];+ #+ # Though it's not universal "miracle cure," it has its limitations.+ # Most notably .cfi_cfa_expression won't start working... For more+ # information see the end of this file.++ # Below constants are taken from "DWARF Expressions" section of the+ # DWARF specification, section is numbered 7.7 in versions 3 and 4.+ my %DW_OP_simple = ( # no-arg operators, mapped directly+ deref => 0x06, dup => 0x12,+ drop => 0x13, over => 0x14,+ pick => 0x15, swap => 0x16,+ rot => 0x17, xderef => 0x18,++ abs => 0x19, and => 0x1a,+ div => 0x1b, minus => 0x1c,+ mod => 0x1d, mul => 0x1e,+ neg => 0x1f, not => 0x20,+ or => 0x21, plus => 0x22,+ shl => 0x24, shr => 0x25,+ shra => 0x26, xor => 0x27,+ );++ my %DW_OP_complex = ( # used in specific subroutines+ constu => 0x10, # uleb128+ consts => 0x11, # sleb128+ plus_uconst => 0x23, # uleb128+ lit0 => 0x30, # add 0-31 to opcode+ reg0 => 0x50, # add 0-31 to opcode+ breg0 => 0x70, # add 0-31 to opcole, sleb128+ regx => 0x90, # uleb28+ fbreg => 0x91, # sleb128+ bregx => 0x92, # uleb128, sleb128+ piece => 0x93, # uleb128+ );++ # Following constants are defined in x86_64 ABI supplement, for+ # example available at https://www.uclibc.org/docs/psABI-x86_64.pdf,+ # see section 3.7 "Stack Unwind Algorithm".+ my %DW_reg_idx = (+ "%rax"=>0, "%rdx"=>1, "%rcx"=>2, "%rbx"=>3,+ "%rsi"=>4, "%rdi"=>5, "%rbp"=>6, "%rsp"=>7,+ "%r8" =>8, "%r9" =>9, "%r10"=>10, "%r11"=>11,+ "%r12"=>12, "%r13"=>13, "%r14"=>14, "%r15"=>15+ );++ my ($cfa_reg, $cfa_off, $cfa_rsp, %saved_regs);+ my @cfa_stack;++ sub cfa_rsp { return $cfa_rsp // -8; }++ # [us]leb128 format is variable-length integer representation base+ # 2^128, with most significant bit of each byte being 0 denoting+ # *last* most significant digit. See "Variable Length Data" in the+ # DWARF specification, numbered 7.6 at least in versions 3 and 4.+ sub sleb128 {+ use integer; # get right shift extend sign++ my $val = shift;+ my $sign = ($val < 0) ? -1 : 0;+ my @ret = ();++ while(1) {+ push @ret, $val&0x7f;++ # see if remaining bits are same and equal to most+ # significant bit of the current digit, if so, it's+ # last digit...+ last if (($val>>6) == $sign);++ @ret[-1] |= 0x80;+ $val >>= 7;+ }++ return @ret;+ }+ sub uleb128 {+ my $val = shift;+ my @ret = ();++ while(1) {+ push @ret, $val&0x7f;++ # see if it's last significant digit...+ last if (($val >>= 7) == 0);++ @ret[-1] |= 0x80;+ }++ return @ret;+ }+ sub const {+ my $val = shift;++ if ($val >= 0 && $val < 32) {+ return ($DW_OP_complex{lit0}+$val);+ }+ return ($DW_OP_complex{consts}, sleb128($val));+ }+ sub reg {+ my $val = shift;++ return if ($val !~ m/^(%r\w+)(?:([\+\-])((?:0x)?[0-9a-f]+))?/);++ my $reg = $DW_reg_idx{$1};+ my $off = eval ("0 $2 $3");++ return (($DW_OP_complex{breg0} + $reg), sleb128($off));+ # Yes, we use DW_OP_bregX+0 to push register value and not+ # DW_OP_regX, because latter would require even DW_OP_piece,+ # which would be a waste under the circumstances. If you have+ # to use DWP_OP_reg, use "regx:N"...+ }+ sub cfa_expression {+ my $line = shift;+ my @ret;++ foreach my $token (split(/,\s*/,$line)) {+ if ($token =~ /^%r/) {+ push @ret,reg($token);+ } elsif ($token =~ /((?:0x)?[0-9a-f]+)\((%r\w+)\)/) {+ push @ret,reg("$2+$1");+ } elsif ($token =~ /(\w+):(\-?(?:0x)?[0-9a-f]+)(U?)/i) {+ my $i = 1*eval($2);+ push @ret,$DW_OP_complex{$1}, ($3 ? uleb128($i) : sleb128($i));+ } elsif (my $i = 1*eval($token) or $token eq "0") {+ if ($token =~ /^\+/) {+ push @ret,$DW_OP_complex{plus_uconst},uleb128($i);+ } else {+ push @ret,const($i);+ }+ } else {+ push @ret,$DW_OP_simple{$token};+ }+ }++ # Finally we return DW_CFA_def_cfa_expression, 15, followed by+ # length of the expression and of course the expression itself.+ return (15,scalar(@ret),@ret);+ }++ # Following constants are defined in "x64 exception handling" at+ # https://docs.microsoft.com/ and match the register sequence in+ # CONTEXT structure defined in winnt.h.+ my %WIN64_reg_idx = (+ "%rax"=>0, "%rcx"=>1, "%rdx"=>2, "%rbx"=>3,+ "%rsp"=>4, "%rbp"=>5, "%rsi"=>6, "%rdi"=>7,+ "%r8" =>8, "%r9" =>9, "%r10"=>10, "%r11"=>11,+ "%r12"=>12, "%r13"=>13, "%r14"=>14, "%r15"=>15+ );+ sub xdata {+ our @dat = ();+ our $len = 0;++ sub savereg {+ my ($key, $offset) = @_;++ if ($key =~ /%xmm([0-9]+)/) {+ if ($offset < 0x100000) {+ push @dat, [0,($1<<4)|8,unpack("C2",pack("v",$offset>>4))];+ } else {+ push @dat, [0,($1<<4)|9,unpack("C4",pack("V",$offset))];+ }+ } else {+ if ($offset < 0x80000) {+ push @dat, [0,(($WIN64_reg_idx{$key})<<4)|4,+ unpack("C2",pack("v",$offset>>3))];+ } else {+ push @dat, [0,(($WIN64_reg_idx{$key})<<4)|5,+ unpack("C4",pack("V",$offset))];+ }+ }+ $len += $#{@dat[-1]}+1;+ }++ my $fp_info = 0;++ # allocate stack frame+ if ($cfa_rsp < -8) {+ my $offset = -8 - $cfa_rsp;+ if ($cfa_reg ne "%rsp" && $saved_regs{$cfa_reg} == -16) {+ $fp_info = $WIN64_reg_idx{$cfa_reg};+ push @dat, [0,$fp_info<<4]; # UWOP_PUSH_NONVOL+ $len += $#{@dat[-1]}+1;+ $offset -= 8;+ }+ if ($offset <= 128) {+ my $alloc = ($offset - 8) >> 3;+ push @dat, [0,$alloc<<4|2]; # UWOP_ALLOC_SMALL+ } elsif ($offset < 0x80000) {+ push @dat, [0,0x01,unpack("C2",pack("v",$offset>>3))];+ } else {+ push @dat, [0,0x11,unpack("C4",pack("V",$offset))];+ }+ $len += $#{@dat[-1]}+1;+ }++ # save frame pointer [if not pushed already]+ if ($cfa_reg ne "%rsp" && $fp_info == 0) {+ $fp_info = $WIN64_reg_idx{$cfa_reg};+ if (defined(my $offset = $saved_regs{$cfa_reg})) {+ $offset -= $cfa_rsp;+ savereg($cfa_reg, $offset);+ }+ }++ # set up frame pointer+ if ($fp_info) {+ push @dat, [0,($fp_info<<4)|3]; # UWOP_SET_FPREG+ $len += $#{@dat[-1]}+1;+ my $fp_off = $cfa_off - $cfa_rsp;+ ($fp_off > 240 or $fp_off&0xf) and die "invalid FP offset $fp_off";+ $fp_info |= $fp_off&-16;+ }++ # save registers+ foreach my $key (sort { $saved_regs{$b} <=> $saved_regs{$a} }+ keys(%saved_regs)) {+ next if ($cfa_reg ne "%rsp" && $cfa_reg eq $key);+ my $offset = $saved_regs{$key} - $cfa_rsp;+ savereg($key, $offset);+ }++ my @ret;+ # generate 4-byte descriptor+ push @ret, ".byte 1,0,".($len/2).",$fp_info";+ $len += 4;+ # keep objdump happy, pad to 4*n and add a 32-bit zero+ unshift @dat, [(0)x(((-$len)&3)+4)];+ $len += $#{@dat[0]}+1;+ # pad to 8*n+ unshift @dat, [(0)x((-$len)&7)] if ($len&7);+ # emit data+ while(defined(my $row = pop @dat)) {+ push @ret, ".byte ". join(",",+ map { sprintf "0x%02x",$_ } @{$row});+ }++ return @ret;+ }+ sub startproc {+ return if ($cfa_rsp == -8);+ ($cfa_reg, $cfa_off, $cfa_rsp) = ("%rsp", -8, -8);+ %saved_regs = ();+ return "startproc";+ }+ sub endproc {+ return if ($cfa_rsp == 0);+ ($cfa_reg, $cfa_off, $cfa_rsp) = ("%rsp", 0, 0);+ %saved_regs = ();+ return "endproc";+ }+ sub re {+ my ($class, $line) = @_;+ my $self = {};+ my $ret;++ if ($$line =~ s/^\s*\.cfi_(\w+)\s*//) {+ bless $self,$class;+ $ret = $self;+ undef $self->{value};+ my $dir = $1;++ SWITCH: for ($dir) {+ # What is $cfa_rsp? Effectively it's difference between %rsp+ # value and current CFA, Canonical Frame Address, which is+ # why it starts with -8. Recall that CFA is top of caller's+ # stack...+ /startproc/ && do { $dir = startproc(); last; };+ /endproc/ && do { $dir = endproc();+ # .cfi_remember_state directives that are not+ # matched with .cfi_restore_state are+ # unnecessary.+ die "unpaired .cfi_remember_state" if (@cfa_stack);+ last;+ };+ /def_cfa_register/+ && do { $cfa_off = $cfa_rsp if ($cfa_reg eq "%rsp");+ $cfa_reg = $$line;+ $cfa_rsp = $cfa_off if ($cfa_reg eq "%rsp");+ last;+ };+ /def_cfa_offset/+ && do { $cfa_off = -1*eval($$line);+ $cfa_rsp = $cfa_off if ($cfa_reg eq "%rsp");+ last;+ };+ /adjust_cfa_offset/+ && do { my $val = 1*eval($$line);+ $cfa_off -= $val;+ if ($cfa_reg eq "%rsp") {+ $cfa_rsp -= $val;+ }+ $$line = "$val";+ last;+ };+ /alloca/ && do { $dir = undef;+ my $val = 1*eval($$line);+ $cfa_rsp -= $val;+ if ($cfa_reg eq "%rsp") {+ $cfa_off -= $val;+ $dir = "adjust_cfa_offset";+ }+ $$line = "$val";+ last;+ };+ /def_cfa/ && do { if ($$line =~ /(%r\w+)\s*(?:,\s*(.+))?/) {+ $cfa_reg = $1;+ if ($cfa_reg eq "%rsp" && !defined($2)) {+ $cfa_off = $cfa_rsp;+ $$line .= ",".(-$cfa_rsp);+ } else {+ $cfa_off = -1*eval($2);+ $cfa_rsp = $cfa_off if ($cfa_reg eq "%rsp");+ }+ }+ last;+ };+ /push/ && do { $dir = undef;+ $cfa_rsp -= 8;+ if ($cfa_reg eq "%rsp") {+ $cfa_off = $cfa_rsp;+ $self->{value} = ".cfi_adjust_cfa_offset\t8\n";+ }+ $saved_regs{$$line} = $cfa_rsp;+ $self->{value} .= ".cfi_offset\t$$line,$cfa_rsp";+ last;+ };+ /pop/ && do { $dir = undef;+ $cfa_rsp += 8;+ if ($cfa_reg eq "%rsp") {+ $cfa_off = $cfa_rsp;+ $self->{value} = ".cfi_adjust_cfa_offset\t-8\n";+ }+ $self->{value} .= ".cfi_restore\t$$line";+ delete $saved_regs{$$line};+ last;+ };+ /cfa_expression/+ && do { $dir = undef;+ $self->{value} = ".cfi_escape\t" .+ join(",", map(sprintf("0x%02x", $_),+ cfa_expression($$line)));+ last;+ };+ /remember_state/+ && do { push @cfa_stack,+ [$cfa_reg,$cfa_off,$cfa_rsp,%saved_regs];+ last;+ };+ /restore_state/+ && do { ($cfa_reg,$cfa_off,$cfa_rsp,%saved_regs)+ = @{pop @cfa_stack};+ last;+ };+ /offset/ && do { if ($$line =~ /(%\w+)(?:-%xmm(\d+))?\s*,\s*(.+)/) {+ my ($reg, $off, $xmmlast) = ($1, 1*eval($3), $2);+ if ($reg !~ /%xmm(\d+)/) {+ $saved_regs{$reg} = $off;+ } else {+ $dir = undef;+ $xmmlast //= $1;+ for (my $i=$1; $i<=$xmmlast; $i++) {+ $saved_regs{"%xmm$i"} = $off;+ $off += 16;+ }+ }+ }+ last;+ };+ /restore/ && do { delete $saved_regs{$$line}; last; };+ /end_prologue/+ && do { $dir = undef;+ $self->{win64} = ".endprolog";+ last;+ };+ /epilogue/ && do { $dir = undef;+ $self->{win64} = ".epilogue";+ $self->{value} = join("\n",+ map { ".cfi_restore\t$_" }+ sort keys(%saved_regs));+ %saved_regs = ();+ last;+ };+ }++ $self->{value} = ".cfi_$dir\t$$line" if ($dir);++ $$line = "";+ }++ return $ret;+ }+ sub out {+ my $self = shift;+ return $self->{value} if ($dwarf);++ if ($win64 and $current_function->{unwind}+ and my $ret = $self->{win64}) {+ my ($reg, $off) = ($cfa_reg =~ /%(?!rsp)/) ? ($', $cfa_off)+ : ("rsp", $cfa_rsp);+ my $fname = $current_function->{name};++ if ($ret eq ".endprolog") {+ $ret = "";+ if ($current_function->{abi} eq "svr4") {+ $ret .= label::win64_args();+ $saved_regs{"%rdi"} = 0; # relative to CFA, remember?+ $saved_regs{"%rsi"} = 8;+ }++ push @pdata_seg,+ ".rva .LSEH_begin_${fname}",+ ".rva .LSEH_body_${fname}",+ ".rva .LSEH_info_${fname}_prologue","";+ push @xdata_seg,+ ".LSEH_info_${fname}_prologue:";+ if ($current_function->{unwind} eq "%rbp") {+ if ($current_function->{abi} eq "svr4") {+ push @xdata_seg,+ ".byte 1,4,6,0x05", # 6 unwind codes, %rbp is FP+ ".byte 4,0x74,2,0", # %rdi at 16(%rsp)+ ".byte 4,0x64,3,0", # %rsi at 24(%rsp)+ ".byte 4,0x53", # mov %rsp, %rbp+ ".byte 1,0x50", # push %rbp+ ".long 0,0" # pad to keep objdump happy+ ;+ } else {+ push @xdata_seg,+ ".byte 1,4,2,0x05", # 2 unwind codes, %rbp is FP+ ".byte 4,0x53", # mov %rsp, %rbp+ ".byte 1,0x50", # push %rbp+ ".long 0,0" # pad to keep objdump happy+ ;+ }+ } else {+ if ($current_function->{abi} eq "svr4") {+ push @xdata_seg,+ ".byte 1,0,5,0x0b", # 5 unwind codes, %r11 is FP+ ".byte 0,0x74,1,0", # %rdi at 8(%rsp)+ ".byte 0,0x64,2,0", # %rsi at 16(%rsp)+ ".byte 0,0xb3", # set frame pointer+ ".byte 0,0", # padding+ ".long 0,0" # pad to keep objdump happy+ ;+ } else {+ push @xdata_seg,+ ".byte 1,0,1,0x0b", # 1 unwind code, %r11 is FP+ ".byte 0,0xb3", # set frame pointer+ ".byte 0,0", # padding+ ".long 0,0" # pad to keep objdump happy+ ;+ }+ }+ push @pdata_seg,+ ".rva .LSEH_body_${fname}",+ ".rva .LSEH_epilogue_${fname}",+ ".rva .LSEH_info_${fname}_body","";+ push @xdata_seg,".LSEH_info_${fname}_body:", xdata();+ $ret .= "${decor}SEH_body_${fname}${colon}\n";+ } elsif ($ret eq ".epilogue") {+ %saved_regs = ();+ $cfa_rsp = $cfa_off;+ $ret = "${decor}SEH_epilogue_${fname}${colon}\n";+ if ($current_function->{abi} eq "svr4") {+ $saved_regs{"%rdi"} = 0; # relative to CFA, remember?+ $saved_regs{"%rsi"} = 8;++ push @pdata_seg,+ ".rva .LSEH_epilogue_${fname}",+ ".rva .LSEH_end_${fname}",+ ".rva .LSEH_info_${fname}_epilogue","";+ push @xdata_seg,".LSEH_info_${fname}_epilogue:", xdata(), "";+ if ($gas) {+ $ret .= " mov ".(0-$off)."(%$reg),%rdi\n";+ $ret .= " mov ".(8-$off)."(%$reg),%rsi\n";+ } else {+ $ret .= " mov rdi,QWORD$PTR\[".(0-$off)."+$reg\]";+ $ret .= " ;WIN64 epilogue\n";+ $ret .= " mov rsi,QWORD$PTR\[".(8-$off)."+$reg\]\n";+ }+ }+ }+ return $ret;+ }+ return;+ }+}+{ package directive; # pick up directives, which start with .+ sub re {+ my ($class, $line) = @_;+ my $self = {};+ my $ret;+ my $dir;++ # chain-call to cfi_directive+ $ret = cfi_directive->re($line) and return $ret;++ if ($$line =~ /^\s*(\.\w+)/) {+ bless $self,$class;+ $dir = $1;+ $ret = $self;+ undef $self->{value};+ $$line = substr($$line,@+[0]); $$line =~ s/^\s+//;++ SWITCH: for ($dir) {+ /\.global|\.globl|\.extern|\.comm/+ && do { $$line =~ s/([_a-z][_a-z0-9\$]*)/$prefix\1/gi;+ $globals{$1} = $prefix.$1 if ($1);+ last;+ };+ /\.type/ && do { my ($sym,$type,$narg,$unwind) = split(',',$$line);+ if ($type eq "\@function") {+ undef $current_function;+ $current_function->{name} = $sym;+ $current_function->{abi} = "svr4";+ $current_function->{narg} = $narg;+ $current_function->{scope} = defined($globals{$sym})?"PUBLIC":"PRIVATE";+ $current_function->{unwind} = $unwind;+ $current_function->{pc} = -1;+ } elsif ($type eq "\@abi-omnipotent") {+ undef $current_function;+ $current_function->{name} = $sym;+ $current_function->{scope} = defined($globals{$sym})?"PUBLIC":"PRIVATE";+ $current_function->{unwind} = $unwind;+ $current_function->{pc} = -1;+ }+ $$line =~ s/\@abi\-omnipotent/\@function/;+ $$line =~ s/\@function.*/\@function/;+ last;+ };+ /\.asciz/ && do { if ($$line =~ /^"(.*)"$/) {+ $dir = ".byte";+ $$line = join(",",unpack("C*",$1),0);+ }+ last;+ };+ /\.rva|\.long|\.quad/+ && do { $$line =~ s/([_a-z][_a-z0-9\$]*)/$globals{$1} or $1/gei;+ $$line =~ s/\.L/$decor/g;+ last;+ };+ }++ if ($gas) {+ $self->{value} = $dir . "\t" . $$line;++ if ($dir =~ /\.extern/) {+ $self->{value} = ""; # swallow extern+ } elsif (!$elf && $dir =~ /\.type/) {+ $self->{value} = "";+ $self->{value} = ".def\t" . ($globals{$1} or $1) . ";\t" .+ (defined($globals{$1})?".scl 2;":".scl 3;") .+ "\t.type 32;\t.endef"+ if ($win64 && $$line =~ /([^,]+),\@function/);+ } elsif ($dir =~ /\.size/) {+ $self->{value} = "" if (!$elf);+ if ($dwarf and my $endproc = cfi_directive::endproc()) {+ $self->{value} = ".cfi_$endproc\n$self->{value}";+ } elsif (!$elf && defined($current_function)) {+ $self->{value} .= "${decor}SEH_end_$current_function->{name}:"+ if ($win64 && $current_function->{abi} eq "svr4");+ undef $current_function;+ }+ } elsif (!$elf && $dir =~ /\.align/) {+ $self->{value} = ".p2align\t" . (log($$line)/log(2));+ } elsif ($dir eq ".section") {+ $current_segment=$$line;+ if (!$elf && $current_segment eq ".init") {+ if ($flavour eq "macosx") { $self->{value} = ".mod_init_func"; }+ elsif ($flavour eq "mingw64") { $self->{value} = ".section\t.ctors"; }+ }+ if (!$elf && $current_segment eq ".rodata") {+ if ($flavour eq "macosx") { $self->{value} = ".section\t__TEXT,__const"; }+ elsif ($flavour eq "mingw64") { $self->{value} = ".section\t.rdata"; }+ }+ } elsif ($dir =~ /\.(text|data)/) {+ $current_segment=".$1";+ } elsif ($dir =~ /\.hidden/) {+ if ($flavour eq "macosx") { $self->{value} = ".private_extern\t$prefix$$line"; }+ elsif ($flavour eq "mingw64") { $self->{value} = ""; }+ } elsif ($dir =~ /\.comm/) {+ $self->{value} = "$dir\t$$line";+ $self->{value} =~ s|,([0-9]+),([0-9]+)$|",$1,".log($2)/log(2)|e if ($flavour eq "macosx");+ }+ $$line = "";+ return $self;+ }++ # non-gas case or nasm/masm+ SWITCH: for ($dir) {+ /\.text/ && do { my $v=undef;+ if ($nasm) {+ $v="section .text code align=64\n";+ } else {+ $v="$current_segment\tENDS\n" if ($current_segment);+ $current_segment = ".text\$";+ $v.="$current_segment\tSEGMENT ";+ $v.=$masm>=$masmref ? "ALIGN(256)" : "PAGE";+ $v.=" 'CODE'";+ }+ $self->{value} = $v;+ last;+ };+ /\.data/ && do { my $v=undef;+ if ($nasm) {+ $v="section .data data align=8\n";+ } else {+ $v="$current_segment\tENDS\n" if ($current_segment);+ $current_segment = "_DATA";+ $v.="$current_segment\tSEGMENT";+ }+ $self->{value} = $v;+ last;+ };+ /\.section/ && do { my $v=undef;+ $$line =~ s/([^,]*).*/$1/;+ $$line = ".CRT\$XCU" if ($$line eq ".init");+ $$line = ".rdata" if ($$line eq ".rodata");+ my %align = ( p=>4, x=>8, r=>256);+ if ($nasm) {+ $v="section $$line";+ if ($$line=~/\.([pxr])data/) {+ $v.=" rdata align=$align{$1}";+ } elsif ($$line=~/\.CRT\$/i) {+ $v.=" rdata align=8";+ }+ } else {+ $v="$current_segment\tENDS\n" if ($current_segment);+ $v.="$$line\tSEGMENT";+ if ($$line=~/\.([pxr])data/) {+ $v.=" READONLY";+ $v.=" ALIGN($align{$1})" if ($masm>=$masmref);+ } elsif ($$line=~/\.CRT\$/i) {+ $v.=" READONLY ";+ $v.=$masm>=$masmref ? "ALIGN(8)" : "DWORD";+ }+ }+ $current_segment = $$line;+ $self->{value} = $v;+ last;+ };+ /\.extern/ && do { $self->{value} = "EXTERN\t".$$line;+ $self->{value} .= ":NEAR" if ($masm);+ last;+ };+ /\.globl|.global/+ && do { $self->{value} = $masm?"PUBLIC":"global";+ $self->{value} .= "\t".$$line;+ last;+ };+ /\.size/ && do { if (defined($current_function)) {+ undef $self->{value};+ if ($current_function->{abi} eq "svr4") {+ $self->{value}="${decor}SEH_end_$current_function->{name}${colon}\n";+ }+ $self->{value}.="$current_function->{name}\tENDP" if($masm && $current_function->{name});+ undef $current_function;+ }+ last;+ };+ /\.align/ && do { my $max = ($masm && $masm>=$masmref) ? 256 : 4096;+ $self->{value} = "ALIGN\t".($$line>$max?$max:$$line);+ last;+ };+ /\.(value|long|rva|quad)/+ && do { my $sz = substr($1,0,1);+ my @arr = split(/,\s*/,$$line);+ my $last = pop(@arr);+ my $conv = sub { my $var=shift;+ $var=~s/^(0b[0-1]+)/oct($1)/eig;+ $var=~s/^0x([0-9a-f]+)/0$1h/ig if ($masm);+ if ($sz eq "D" && ($current_segment=~/.[px]data/ || $dir eq ".rva"))+ { $var=~s/^([_a-z\$\@][_a-z0-9\$\@]*)/$nasm?"$1 wrt ..imagebase":"imagerel $1"/egi; }+ $var;+ };++ $sz =~ tr/bvlrq/BWDDQ/;+ $self->{value} = "\tD$sz\t";+ for (@arr) { $self->{value} .= &$conv($_).","; }+ $self->{value} .= &$conv($last);+ last;+ };+ /\.byte/ && do { my @str=split(/,\s*/,$$line);+ map(s/(0b[0-1]+)/oct($1)/eig,@str);+ map(s/0x([0-9a-f]+)/0$1h/ig,@str) if ($masm);+ while ($#str>15) {+ $self->{value}.="DB\t"+ .join(",",@str[0..15])."\n";+ foreach (0..15) { shift @str; }+ }+ $self->{value}.="DB\t"+ .join(",",@str) if (@str);+ last;+ };+ /\.comm/ && do { my @str=split(/,\s*/,$$line);+ my $v=undef;+ if ($nasm) {+ $v.="common $prefix@str[0] @str[1]";+ } else {+ $v="$current_segment\tENDS\n" if ($current_segment);+ $current_segment = "_DATA";+ $v.="$current_segment\tSEGMENT\n";+ $v.="COMM @str[0]:DWORD:".@str[1]/4;+ }+ $self->{value} = $v;+ last;+ };+ }+ $$line = "";+ }++ $ret;+ }+ sub out {+ my $self = shift;+ $self->{value};+ }+}++# Upon initial x86_64 introduction SSE>2 extensions were not introduced+# yet. In order not to be bothered by tracing exact assembler versions,+# but at the same time to provide a bare security minimum of AES-NI, we+# hard-code some instructions. Extensions past AES-NI on the other hand+# are traced by examining assembler version in individual perlasm+# modules...++my %regrm = ( "%eax"=>0, "%ecx"=>1, "%edx"=>2, "%ebx"=>3,+ "%esp"=>4, "%ebp"=>5, "%esi"=>6, "%edi"=>7 );++sub rex {+ my $opcode=shift;+ my ($dst,$src,$rex)=@_;++ $rex|=0x04 if($dst>=8);+ $rex|=0x01 if($src>=8);+ push @$opcode,($rex|0x40) if ($rex);+}++my $movq = sub { # elderly gas can't handle inter-register movq+ my $arg = shift;+ my @opcode=(0x66);+ if ($arg =~ /%xmm([0-9]+),\s*%r(\w+)/) {+ my ($src,$dst)=($1,$2);+ if ($dst !~ /[0-9]+/) { $dst = $regrm{"%e$dst"}; }+ rex(\@opcode,$src,$dst,0x8);+ push @opcode,0x0f,0x7e;+ push @opcode,0xc0|(($src&7)<<3)|($dst&7); # ModR/M+ @opcode;+ } elsif ($arg =~ /%r(\w+),\s*%xmm([0-9]+)/) {+ my ($src,$dst)=($2,$1);+ if ($dst !~ /[0-9]+/) { $dst = $regrm{"%e$dst"}; }+ rex(\@opcode,$src,$dst,0x8);+ push @opcode,0x0f,0x6e;+ push @opcode,0xc0|(($src&7)<<3)|($dst&7); # ModR/M+ @opcode;+ } else {+ ();+ }+};++my $pextrd = sub {+ if (shift =~ /\$([0-9]+),\s*%xmm([0-9]+),\s*(%\w+)/) {+ my @opcode=(0x66);+ my $imm=$1;+ my $src=$2;+ my $dst=$3;+ if ($dst =~ /%r([0-9]+)d/) { $dst = $1; }+ elsif ($dst =~ /%e/) { $dst = $regrm{$dst}; }+ rex(\@opcode,$src,$dst);+ push @opcode,0x0f,0x3a,0x16;+ push @opcode,0xc0|(($src&7)<<3)|($dst&7); # ModR/M+ push @opcode,$imm;+ @opcode;+ } else {+ ();+ }+};++my $pinsrd = sub {+ if (shift =~ /\$([0-9]+),\s*(%\w+),\s*%xmm([0-9]+)/) {+ my @opcode=(0x66);+ my $imm=$1;+ my $src=$2;+ my $dst=$3;+ if ($src =~ /%r([0-9]+)/) { $src = $1; }+ elsif ($src =~ /%e/) { $src = $regrm{$src}; }+ rex(\@opcode,$dst,$src);+ push @opcode,0x0f,0x3a,0x22;+ push @opcode,0xc0|(($dst&7)<<3)|($src&7); # ModR/M+ push @opcode,$imm;+ @opcode;+ } else {+ ();+ }+};++my $pshufb = sub {+ if (shift =~ /%xmm([0-9]+),\s*%xmm([0-9]+)/) {+ my @opcode=(0x66);+ rex(\@opcode,$2,$1);+ push @opcode,0x0f,0x38,0x00;+ push @opcode,0xc0|($1&7)|(($2&7)<<3); # ModR/M+ @opcode;+ } else {+ ();+ }+};++my $palignr = sub {+ if (shift =~ /\$([0-9]+),\s*%xmm([0-9]+),\s*%xmm([0-9]+)/) {+ my @opcode=(0x66);+ rex(\@opcode,$3,$2);+ push @opcode,0x0f,0x3a,0x0f;+ push @opcode,0xc0|($2&7)|(($3&7)<<3); # ModR/M+ push @opcode,$1;+ @opcode;+ } else {+ ();+ }+};++my $pclmulqdq = sub {+ if (shift =~ /\$([x0-9a-f]+),\s*%xmm([0-9]+),\s*%xmm([0-9]+)/) {+ my @opcode=(0x66);+ rex(\@opcode,$3,$2);+ push @opcode,0x0f,0x3a,0x44;+ push @opcode,0xc0|($2&7)|(($3&7)<<3); # ModR/M+ my $c=$1;+ push @opcode,$c=~/^0/?oct($c):$c;+ @opcode;+ } else {+ ();+ }+};++my $rdrand = sub {+ if (shift =~ /%[er](\w+)/) {+ my @opcode=();+ my $dst=$1;+ if ($dst !~ /[0-9]+/) { $dst = $regrm{"%e$dst"}; }+ rex(\@opcode,0,$dst,8);+ push @opcode,0x0f,0xc7,0xf0|($dst&7);+ @opcode;+ } else {+ ();+ }+};++my $rdseed = sub {+ if (shift =~ /%[er](\w+)/) {+ my @opcode=();+ my $dst=$1;+ if ($dst !~ /[0-9]+/) { $dst = $regrm{"%e$dst"}; }+ rex(\@opcode,0,$dst,8);+ push @opcode,0x0f,0xc7,0xf8|($dst&7);+ @opcode;+ } else {+ ();+ }+};++# Not all AVX-capable assemblers recognize AMD XOP extension. Since we+# are using only two instructions hand-code them in order to be excused+# from chasing assembler versions...++sub rxb {+ my $opcode=shift;+ my ($dst,$src1,$src2,$rxb)=@_;++ $rxb|=0x7<<5;+ $rxb&=~(0x04<<5) if($dst>=8);+ $rxb&=~(0x01<<5) if($src1>=8);+ $rxb&=~(0x02<<5) if($src2>=8);+ push @$opcode,$rxb;+}++my $vprotd = sub {+ if (shift =~ /\$([x0-9a-f]+),\s*%xmm([0-9]+),\s*%xmm([0-9]+)/) {+ my @opcode=(0x8f);+ rxb(\@opcode,$3,$2,-1,0x08);+ push @opcode,0x78,0xc2;+ push @opcode,0xc0|($2&7)|(($3&7)<<3); # ModR/M+ my $c=$1;+ push @opcode,$c=~/^0/?oct($c):$c;+ @opcode;+ } else {+ ();+ }+};++my $vprotq = sub {+ if (shift =~ /\$([x0-9a-f]+),\s*%xmm([0-9]+),\s*%xmm([0-9]+)/) {+ my @opcode=(0x8f);+ rxb(\@opcode,$3,$2,-1,0x08);+ push @opcode,0x78,0xc3;+ push @opcode,0xc0|($2&7)|(($3&7)<<3); # ModR/M+ my $c=$1;+ push @opcode,$c=~/^0/?oct($c):$c;+ @opcode;+ } else {+ ();+ }+};++# Intel Control-flow Enforcement Technology extension. All functions and+# indirect branch targets will have to start with this instruction...+# However, it should not be used in functions' prologues explicitly, as+# it's added automatically [and in the right spot]. Which leaves only+# non-function indirect branch targets, such as in a case-like dispatch+# table, as application area.++my $endbr64 = sub {+ (0xf3,0x0f,0x1e,0xfa);+};++########################################################################++my $preproc_prefix = "#";++if ($nasm) {+ $preproc_prefix = "%";+ print <<___;+default rel+%define XMMWORD+%define YMMWORD+%define ZMMWORD+___+} elsif ($masm) {+ $preproc_prefix = "";+ print <<___;+OPTION DOTNAME+___+}++sub process {+ my $line = shift;++ $line =~ s|\R$||; # Better chomp++ if ($line =~ m/^#\s*(if|elif|else|endif)(.*)/) { # pass through preproc+ if ($win64 && $current_function->{abi} eq "svr4"+ && $current_function->{narg} >= 0) {+ print label::win64_args();+ }+ print $preproc_prefix,$1,$2,"\n";+ next;+ }++ print $1 if ($line =~ s|(\{\w+\})||);++ $line =~ s|[#!].*$||; # get rid of asm-style comments...+ $line =~ s|/\*.*\*/||; # ... and C-style comments...+ $line =~ s|^\s+||; # ... and skip white spaces in beginning+ $line =~ s|\s+$||; # ... and at the end++ if (my $label=label->re(\$line)) { print $label->out(); }++ if (my $directive=directive->re(\$line)) {+ printf "%s",$directive->out();+ } elsif (my $opcode=opcode->re(\$line)) {+ my $asm = eval("\$".$opcode->mnemonic());++ if ((ref($asm) eq 'CODE') && scalar(my @bytes=&$asm($line))) {+ print $gas?".byte\t":"DB\t",join(',',@bytes),"\n";+ next;+ }++ my @args;+ ARGUMENT: while (1) {+ my $arg;++ ($arg=register->re(\$line, $opcode))||+ ($arg=const->re(\$line)) ||+ ($arg=ea->re(\$line, $opcode)) ||+ ($arg=expr->re(\$line, $opcode)) ||+ last ARGUMENT;++ push @args,$arg;++ last ARGUMENT if ($line !~ /^,/);++ $line =~ s/^,\s*//;+ } # ARGUMENT:++ if ($win64 && $current_function->{abi} eq "svr4"+ && $current_function->{narg} >= 0) {+ my $pc = $current_function->{pc};+ my $op = $opcode->{op};+ my $a0 = @args[0]->{value} if ($#args>=0);+ if (!$current_function->{unwind}+ || $pc == 0 && !($op eq "push" && $a0 eq "rbp")+ || $pc == 1 && !($op eq "mov" && $a0 eq "rsp"+ && @args[1]->{value} eq "rbp"+ && ($current_function->{unwind} = "%rbp"))+ || $pc > 1) {+ print label::win64_args();+ }+ }++ if ($#args>=0) {+ my $insn;+ my $sz=$opcode->size();++ if ($gas) {+ $insn = $opcode->out($#args>=1?$args[$#args]->size():$sz);+ @args = map($_->out($sz),@args);+ printf "\t%s\t%s",$insn,join(",",@args);+ } else {+ $insn = $opcode->out();+ foreach (@args) {+ my $arg = $_->out();+ # $insn.=$sz compensates for movq, pinsrw, ...+ if ($arg =~ /^xmm[0-9]+$/) { $insn.=$sz; $sz="x" if(!$sz); last; }+ if ($arg =~ /^ymm[0-9]+$/) { $insn.=$sz; $sz="y" if(!$sz); last; }+ if ($arg =~ /^zmm[0-9]+$/) { $insn.=$sz; $sz="z" if(!$sz); last; }+ if ($arg =~ /^mm[0-9]+$/) { $insn.=$sz; $sz="q" if(!$sz); last; }+ }+ @args = reverse(@args);+ undef $sz if ($nasm && $opcode->mnemonic() eq "lea");+ printf "\t%s\t%s",$insn,join(",",map($_->out($sz),@args));+ }+ } else {+ printf "\t%s",$opcode->out();+ }++ ++$current_function->{pc} if (defined($current_function));+ }++ print $line,"\n";+}++while(<>) { process($_); }++map { process($_) } @pdata_seg if ($win64 && $#pdata_seg>1);+map { process($_) } @xdata_seg if ($win64 && $#xdata_seg>1);++# platform-specific epilogue+if ($masm) {+ print "\n$current_segment\tENDS\n" if ($current_segment);+ print "END\n";+} elsif ($elf) {+ # -fcf-protection segment, snatched from compiler -S output+ my $align = ($flavour =~ /elf32/) ? 4 : 8;+ print <<___;++.section .note.gnu.property,"a",\@note+ .long 4,2f-1f,5+ .byte 0x47,0x4E,0x55,0+1: .long 0xc0000002,4,3+.align $align+2:+___+}++close STDOUT;++#################################################+# Cross-reference x86_64 ABI "card"+#+# Unix Win64+# %rax * *+# %rbx - -+# %rcx #4 #1+# %rdx #3 #2+# %rsi #2 -+# %rdi #1 -+# %rbp - -+# %rsp - -+# %r8 #5 #3+# %r9 #6 #4+# %r10 * *+# %r11 * *+# %r12 - -+# %r13 - -+# %r14 - -+# %r15 - -+#+# (*) volatile register+# (-) preserved by callee+# (#) Nth argument, volatile+#+# In Unix terms top of stack is argument transfer area for arguments+# which could not be accommodated in registers. Or in other words 7th+# [integer] argument resides at 8(%rsp) upon function entry point.+# 128 bytes above %rsp constitute a "red zone" which is not touched+# by signal handlers and can be used as temporal storage without+# allocating a frame.+#+# In Win64 terms N*8 bytes on top of stack is argument transfer area,+# which belongs to/can be overwritten by callee. N is the number of+# arguments passed to callee, *but* not less than 4! This means that+# upon function entry point 5th argument resides at 40(%rsp), as well+# as that 32 bytes from 8(%rsp) can always be used as temporal+# storage [without allocating a frame]. One can actually argue that+# one can assume a "red zone" above stack pointer under Win64 as well.+# Point is that at apparently no occasion Windows kernel would alter+# the area above user stack pointer in true asynchronous manner...+#+# All the above means that if assembler programmer adheres to Unix+# register and stack layout, but disregards the "red zone" existence,+# it's possible to use following prologue and epilogue to "gear" from+# Unix to Win64 ABI in leaf functions with not more than 6 arguments.+#+# omnipotent_function:+# ifdef WIN64+# movq %rdi,8(%rsp)+# movq %rsi,16(%rsp)+# movq %rcx,%rdi ; if 1st argument is actually present+# movq %rdx,%rsi ; if 2nd argument is actually ...+# movq %r8,%rdx ; if 3rd argument is ...+# movq %r9,%rcx ; if 4th argument ...+# movq 40(%rsp),%r8 ; if 5th ...+# movq 48(%rsp),%r9 ; if 6th ...+# endif+# ...+# ifdef WIN64+# movq 8(%rsp),%rdi+# movq 16(%rsp),%rsi+# endif+# ret+#+#################################################+# Win64 SEH, Structured Exception Handling.+#+# Unlike on Unix systems(*) lack of Win64 stack unwinding information+# has undesired side-effect at run-time: if an exception is raised in+# assembler subroutine such as those in question (basically we're+# referring to segmentation violations caused by malformed input+# parameters), the application is briskly terminated without invoking+# any exception handlers, most notably without generating memory dump+# or any user notification whatsoever. This poses a problem. It's+# possible to address it by registering custom language-specific+# handler that would restore processor context to the state at+# subroutine entry point and return "exception is not handled, keep+# unwinding" code. Writing such handler can be a challenge... But it's+# doable, though requires certain coding convention. Consider following+# snippet:+#+# .type function,@function+# function:+# movq %rsp,%rax # copy rsp to volatile register+# pushq %r15 # save non-volatile registers+# pushq %rbx+# pushq %rbp+# movq %rsp,%r11+# subq %rdi,%r11 # prepare [variable] stack frame+# andq $-64,%r11+# movq %rax,0(%r11) # check for exceptions+# movq %r11,%rsp # allocate [variable] stack frame+# movq %rax,0(%rsp) # save original rsp value+# magic_point:+# ...+# movq 0(%rsp),%rcx # pull original rsp value+# movq -24(%rcx),%rbp # restore non-volatile registers+# movq -16(%rcx),%rbx+# movq -8(%rcx),%r15+# movq %rcx,%rsp # restore original rsp+# magic_epilogue:+# ret+# .size function,.-function+#+# The key is that up to magic_point copy of original rsp value remains+# in chosen volatile register and no non-volatile register, except for+# rsp, is modified. While past magic_point rsp remains constant till+# the very end of the function. In this case custom language-specific+# exception handler would look like this:+#+# EXCEPTION_DISPOSITION handler (EXCEPTION_RECORD *rec,ULONG64 frame,+# CONTEXT *context,DISPATCHER_CONTEXT *disp)+# { ULONG64 *rsp = (ULONG64 *)context->Rax;+# ULONG64 rip = context->Rip;+#+# if (rip >= magic_point)+# { rsp = (ULONG64 *)context->Rsp;+# if (rip < magic_epilogue)+# { rsp = (ULONG64 *)rsp[0];+# context->Rbp = rsp[-3];+# context->Rbx = rsp[-2];+# context->R15 = rsp[-1];+# }+# }+# context->Rsp = (ULONG64)rsp;+# context->Rdi = rsp[1];+# context->Rsi = rsp[2];+#+# memcpy (disp->ContextRecord,context,sizeof(CONTEXT));+# RtlVirtualUnwind(UNW_FLAG_NHANDLER,disp->ImageBase,+# dips->ControlPc,disp->FunctionEntry,disp->ContextRecord,+# &disp->HandlerData,&disp->EstablisherFrame,NULL);+# return ExceptionContinueSearch;+# }+#+# It's appropriate to implement this handler in assembler, directly in+# function's module. In order to do that one has to know members'+# offsets in CONTEXT and DISPATCHER_CONTEXT structures and some constant+# values. Here they are:+#+# CONTEXT.Rax 120+# CONTEXT.Rcx 128+# CONTEXT.Rdx 136+# CONTEXT.Rbx 144+# CONTEXT.Rsp 152+# CONTEXT.Rbp 160+# CONTEXT.Rsi 168+# CONTEXT.Rdi 176+# CONTEXT.R8 184+# CONTEXT.R9 192+# CONTEXT.R10 200+# CONTEXT.R11 208+# CONTEXT.R12 216+# CONTEXT.R13 224+# CONTEXT.R14 232+# CONTEXT.R15 240+# CONTEXT.Rip 248+# CONTEXT.Xmm6 512+# sizeof(CONTEXT) 1232+# DISPATCHER_CONTEXT.ControlPc 0+# DISPATCHER_CONTEXT.ImageBase 8+# DISPATCHER_CONTEXT.FunctionEntry 16+# DISPATCHER_CONTEXT.EstablisherFrame 24+# DISPATCHER_CONTEXT.TargetIp 32+# DISPATCHER_CONTEXT.ContextRecord 40+# DISPATCHER_CONTEXT.LanguageHandler 48+# DISPATCHER_CONTEXT.HandlerData 56+# UNW_FLAG_NHANDLER 0+# ExceptionContinueSearch 1+#+# In order to tie the handler to the function one has to compose+# couple of structures: one for .xdata segment and one for .pdata.+#+# UNWIND_INFO structure for .xdata segment would be+#+# function_unwind_info:+# .byte 9,0,0,0+# .rva handler+#+# This structure designates exception handler for a function with+# zero-length prologue, no stack frame or frame register.+#+# To facilitate composing of .pdata structures, auto-generated "gear"+# prologue copies rsp value to rax and denotes next instruction with+# .LSEH_begin_{function_name} label. This essentially defines the SEH+# styling rule mentioned in the beginning. Position of this label is+# chosen in such manner that possible exceptions raised in the "gear"+# prologue would be accounted to caller and unwound from latter's frame.+# End of function is marked with respective .LSEH_end_{function_name}+# label. To summarize, .pdata segment would contain+#+# .rva .LSEH_begin_function+# .rva .LSEH_end_function+# .rva function_unwind_info+#+# Reference to function_unwind_info from .xdata segment is the anchor.+# In case you wonder why references are 32-bit .rvas and not 64-bit+# .quads. References put into these two segments are required to be+# *relative* to the base address of the current binary module, a.k.a.+# image base. No Win64 module, be it .exe or .dll, can be larger than+# 2GB and thus such relative references can be and are accommodated in+# 32 bits.+#+# Having reviewed the example function code, one can argue that "movq+# %rsp,%rax" above is redundant. It is not! Keep in mind that on Unix+# rax would contain an undefined value. If this "offends" you, use+# another register and refrain from modifying rax till magic_point is+# reached, i.e. as if it was a non-volatile register. If more registers+# are required prior [variable] frame setup is completed, note that+# nobody says that you can have only one "magic point." You can+# "liberate" non-volatile registers by denoting last stack off-load+# instruction and reflecting it in finer grade unwind logic in handler.+# After all, isn't it why it's called *language-specific* handler...+#+# SE handlers are also involved in unwinding stack when executable is+# profiled or debugged. Profiling implies additional limitations that+# are too subtle to discuss here. For now it's sufficient to say that+# in order to simplify handlers one should either a) offload original+# %rsp to stack (like discussed above); or b) if you have a register to+# spare for frame pointer, choose volatile one.+#+# (*) Note that we're talking about run-time, not debug-time. Lack of+# unwind information makes debugging hard on both Windows and+# Unix. "Unlike" refers to the fact that on Unix signal handler+# will always be invoked, core dumped and appropriate exit code+# returned to parent (for user notification).+#+########################################################################+# As of May 2020 an alternative approach that works with both exceptions+# and debugging/profiling was implemented by re-purposing DWARF .cfi+# annotations even for Win64 unwind tables' generation. Unfortunately,+# but not really unexpectedly, it imposes additional limitations on+# coding style. Probably the most significant limitation is that the+# frame pointer has to be at 16*n distance from the stack pointer at the+# exit from prologue. But first things first. There are two additional+# synthetic .cfi directives, .cfi_end_prologue and .cfi_epilogue,+# that need to be added to all functions marked with additional .type+# tag (see example below). There are "do's and don'ts" for prologue+# and epilogue. It shouldn't come as a surprise that in prologue one may+# not modify non-volatile registers, but one may not modify %r11 either.+# This is because it's used as a temporary frame pointer(*). There are+# two exceptions to this rule. 1) One can set up a non-volatile register+# or %r11 as a frame pointer, but it must be last instruction in the+# prologue. 2) One can use 'push %rbp' as first instruction immediately+# followed by 'mov %rsp,%rbp' to use %rbp as "legacy" frame pointer.+# Constraints for epilogue, or rather on its boundary, depend on whether+# the frame is fixed- or variable-length. In fixed-frame subroutine+# stack pointer has to be restored in the last instruction prior to the+# .cfi_epilogue directive. If it's a variable-frame subroutine, and a+# non-volatile register was used as a frame pointer, then the last+# instruction prior to the directive has to restore its original value.+# This means that final stack pointer adjustment would have to be+# pushed past the directive. Normally this would render the epilogue+# non-unwindable, so special care has to be taken. To resolve the+# dilemma, copy the frame pointer to a volatile register in advance.+# To give an example:+#+# .type rbp_as_frame_pointer,\@function,3,"unwind" # mind extra tag!+# rbp_as_frame_pointer:+# .cfi_startproc+# push %rbp+# .cfi_push %rbp+# push %rbx+# .cfi_push %rbx+# mov %rsp,%rbp # last instruction in prologue+# .cfi_def_cfa_register %rbp # %rsp-%rbp has to be 16*n, e.g. 16*0+# .cfi_end_prologue+# sub \$40,%rsp+# and \$-64,%rsp+# ...+# mov %rbp,%r11+# .cfi_def_cfa_register %r11 # copy frame pointer to volatile %r11+# mov 0(%rbp),%rbx+# mov 8(%rbp),%rbp # last instruction prior epilogue+# .cfi_epilogue # may not change %r11 in epilogue+# lea 16(%r11),%rsp+# ret+# .cfi_endproc+# .size rbp_as_frame_pointer,.-rbp_as_frame_pointer+#+# An example of "legacy" frame pointer:+#+# .type legacy_frame_pointer,\@function,3,"unwind" # mind extra tag!+# legacy_frame_pointer:+# .cfi_startproc+# push %rbp+# .cfi_push %rbp+# mov %rsp,%rbp+# .cfi_def_cfa_register %rbp+# push %rbx+# .cfi_push %rbx+# sub \$40,%rsp+# .cfi_alloca 40+# .cfi_end_prologue # %rsp-%rbp has to be 16*n+# and \$-64,%rsp+# ...+# mov -8(%rbp),%rbx+# mov %rbp,%rsp+# .cfi_def_cfa_register %rsp+# pop %rbp # recognized by Windows+# .cfi_pop %rbp+# .cfi_epilogue+# ret+# .cfi_endproc+# .size legacy_frame_pointer,.-legacy_frame_pointer+#+# To give an example of fixed-frame subroutine for reference:+#+# .type fixed_frame,\@function,3,"unwind" # mind extra tag!+# fixed_frame:+# .cfi_startproc+# push %rbp+# .cfi_push %rbp+# push %rbx+# .cfi_push %rbx+# sub \$40,%rsp+# .cfi_adjust_cfa_offset 40+# .cfi_end_prologue+# ...+# mov 40(%rsp),%rbx+# mov 48(%rsp),%rbp+# lea 56(%rsp),%rsp+# .cfi_adjust_cfa_offset -56+# .cfi_epilogue+# ret+# .cfi_endproc+# .size fixed_frame,.-fixed_frame+#+# As for epilogue itself, one can only work on non-volatile registers.+# "Non-volatile" in "Windows" sense, i.e. minus %rdi and %rsi.+#+# On a final note, mixing old-style and modernized subroutines in the+# same file takes some trickery. Ones of the new kind have to appear+# after old-style ones. This has everything to do with the fact that+# entries in the .pdata segment have to appear in strictly same order+# as corresponding subroutines, and auto-generated RUNTIME_FUNCTION+# structures get mechanically appended to whatever existing .pdata.+#+# (*) Just in case, why %r11 and not %rax. This has everything to do+# with the way UNWIND_INFO is, one just can't designate %rax as+# frame pointer.
@@ -0,0 +1,146 @@+/*+ * ChaCha with AVX2, eight blocks at a time.+ *+ * The same arrangement as the SSE and NEON versions, twice as wide: word i+ * of eight blocks goes in lane i of one 256-bit register. Eight blocks is+ * where the register file stops being the constraint -- sixteen registers+ * hold the working state either way, so the wider ones are free.+ *+ * AVX2 is not part of any baseline, so this is reached only after+ * crypton_x86_simd_features() has said the CPU has it and the OS saves the+ * wider registers. It is compiled into a translation unit that is+ * otherwise baseline, through a function attribute, so nothing here can be+ * emitted anywhere else.+ */++#include <stdint.h>+#include <immintrin.h>+#include "crypton_chacha.h"++#ifdef WITH_TARGET_ATTRIBUTES++#define TARGET __attribute__((target("avx2")))++/* rotating a 32-bit lane by sixteen or eight is a byte shuffle, which AVX2+ * does within each 128-bit half -- which is all this needs */+static const int8_t rot16_tbl[32] = {+ 2,3,0,1, 6,7,4,5, 10,11,8,9, 14,15,12,13,+ 2,3,0,1, 6,7,4,5, 10,11,8,9, 14,15,12,13,+};+static const int8_t rot8_tbl[32] = {+ 3,0,1,2, 7,4,5,6, 11,8,9,10, 15,12,13,14,+ 3,0,1,2, 7,4,5,6, 11,8,9,10, 15,12,13,14,+};++#define ROL(x, n) \+ ((n) == 16 ? _mm256_shuffle_epi8((x), _mm256_loadu_si256((const __m256i *) rot16_tbl)) \+ : (n) == 8 ? _mm256_shuffle_epi8((x), _mm256_loadu_si256((const __m256i *) rot8_tbl)) \+ : _mm256_or_si256(_mm256_slli_epi32((x), (n)), _mm256_srli_epi32((x), 32 - (n))))++TARGET+static inline void core8(int rounds, const crypton_chacha_state *in,+ const uint8_t *src, uint8_t *dst, int combine)+{+ __m256i v0, v1, v2, v3, v4, v5, v6, v7;+ __m256i v8, v9, v10, v11, v12, v13, v14, v15;+ const uint32_t c = in->d[12];+ const __m256i ctr = _mm256_setr_epi32((int) c, (int) (c + 1), (int) (c + 2),+ (int) (c + 3), (int) (c + 4), (int) (c + 5),+ (int) (c + 6), (int) (c + 7));+ int i;++#define SET(n) v##n = _mm256_set1_epi32((int) in->d[n])+ SET(0); SET(1); SET(2); SET(3);+ SET(4); SET(5); SET(6); SET(7);+ SET(8); SET(9); SET(10); SET(11);+ SET(13); SET(14); SET(15);+#undef SET+ v12 = ctr;++#define QR(a, b, cc, d) \+ a = _mm256_add_epi32(a, b); d = ROL(_mm256_xor_si256(d, a), 16); \+ cc = _mm256_add_epi32(cc, d); b = ROL(_mm256_xor_si256(b, cc), 12); \+ a = _mm256_add_epi32(a, b); d = ROL(_mm256_xor_si256(d, a), 8); \+ cc = _mm256_add_epi32(cc, d); b = ROL(_mm256_xor_si256(b, cc), 7)++ for (i = rounds; i > 0; i -= 2) {+ QR(v0, v4, v8, v12);+ QR(v1, v5, v9, v13);+ QR(v2, v6, v10, v14);+ QR(v3, v7, v11, v15);++ QR(v0, v5, v10, v15);+ QR(v1, v6, v11, v12);+ QR(v2, v7, v8, v13);+ QR(v3, v4, v9, v14);+ }+#undef QR++#define ADD(n) v##n = _mm256_add_epi32(v##n, _mm256_set1_epi32((int) in->d[n]))+ ADD(0); ADD(1); ADD(2); ADD(3);+ ADD(4); ADD(5); ADD(6); ADD(7);+ ADD(8); ADD(9); ADD(10); ADD(11);+ ADD(13); ADD(14); ADD(15);+#undef ADD+ v12 = _mm256_add_epi32(v12, ctr);++ /*+ * The interleave works within each 128-bit half, so four registers+ * holding word w of blocks 0..7 come apart into words w..w+3 of+ * blocks 0..3 in the low halves and of blocks 4..7 in the high ones.+ *+ * Each piece is exclusive-ored with the input and stored where it+ * belongs as it comes out. Writing the keystream to a buffer and+ * reading it back to combine it cost a pass over every byte, which is+ * a tenth of what this loop does.+ */+#define OUT(j, g, v) \+ do { \+ __m128i o_ = (v); \+ if (combine) \+ o_ = _mm_xor_si128(o_, _mm_loadu_si128( \+ (const __m128i *) (src + 64 * (j) + 4 * (g)))); \+ _mm_storeu_si128((__m128i *) (dst + 64 * (j) + 4 * (g)), o_);\+ } while (0)++#define GROUP(g, qa, qb, qc, qd) \+ do { \+ __m256i t0_ = _mm256_unpacklo_epi32(qa, qb); \+ __m256i t1_ = _mm256_unpackhi_epi32(qa, qb); \+ __m256i t2_ = _mm256_unpacklo_epi32(qc, qd); \+ __m256i t3_ = _mm256_unpackhi_epi32(qc, qd); \+ __m256i u0_ = _mm256_unpacklo_epi64(t0_, t2_); \+ __m256i u1_ = _mm256_unpackhi_epi64(t0_, t2_); \+ __m256i u2_ = _mm256_unpacklo_epi64(t1_, t3_); \+ __m256i u3_ = _mm256_unpackhi_epi64(t1_, t3_); \+ OUT(0, (g), _mm256_castsi256_si128(u0_)); \+ OUT(1, (g), _mm256_castsi256_si128(u1_)); \+ OUT(2, (g), _mm256_castsi256_si128(u2_)); \+ OUT(3, (g), _mm256_castsi256_si128(u3_)); \+ OUT(4, (g), _mm256_extracti128_si256(u0_, 1)); \+ OUT(5, (g), _mm256_extracti128_si256(u1_, 1)); \+ OUT(6, (g), _mm256_extracti128_si256(u2_, 1)); \+ OUT(7, (g), _mm256_extracti128_si256(u3_, 1)); \+ } while (0)+ GROUP(0, v0, v1, v2, v3);+ GROUP(4, v4, v5, v6, v7);+ GROUP(8, v8, v9, v10, v11);+ GROUP(12, v12, v13, v14, v15);+#undef GROUP+#undef OUT+}++TARGET+void crypton_chacha_avx2_combine(int rounds, uint8_t *dst, const uint8_t *src,+ const crypton_chacha_state *in)+{+ core8(rounds, in, src, dst, 1);+}++TARGET+void crypton_chacha_avx2_generate(int rounds, uint8_t *dst, const crypton_chacha_state *in)+{+ core8(rounds, in, NULL, dst, 0);+}++#endif /* WITH_TARGET_ATTRIBUTES */
@@ -0,0 +1,145 @@+/*+ * ChaCha with NEON, four blocks at a time.+ *+ * The state is sixteen 32-bit words and the quarter rounds touch four of+ * them at once, so a single block vectorises only by shuffling lanes+ * between the column and diagonal rounds. Four blocks vectorise without+ * any shuffling at all: word i of the four blocks goes in lane i of one+ * register, every quarter round is then the same operation on whole+ * registers, and the blocks are independent because only the counter+ * differs between them.+ *+ * NEON is part of the AArch64 baseline, so unlike the AES, PMULL and SHA+ * work there is nothing to ask about at runtime and no target attribute+ * to attach.+ */++#include <stddef.h>+#include <stdint.h>+#include <arm_neon.h>+#include "crypton_chacha.h"++/* rotate each 32-bit lane left by n */+#define ROL(x, n) vsriq_n_u32(vshlq_n_u32((x), (n)), (x), 32 - (n))+/* by 16 it is a halfword swap, and by 8 a byte shuffle; both beat the pair+ * of shifts */+#define ROL16(x) vreinterpretq_u32_u16(vrev32q_u16(vreinterpretq_u16_u32(x)))+#define ROL8(x) vreinterpretq_u32_u8(vqtbl1q_u8(vreinterpretq_u8_u32(x), rot8))++#define QR(a, b, c, d) \+ a = vaddq_u32(a, b); d = ROL16(veorq_u32(d, a)); \+ c = vaddq_u32(c, d); b = ROL(veorq_u32(b, c), 12); \+ a = vaddq_u32(a, b); d = ROL8(veorq_u32(d, a)); \+ c = vaddq_u32(c, d); b = ROL(veorq_u32(b, c), 7)++/*+ * Turn four registers holding word w of blocks 0..3 into four holding+ * words w..w+3 of one block each, which is the order they are written in.+ */+#define TRANSPOSE(a, b, c, d) \+ do { \+ uint32x4x2_t t0_ = vtrnq_u32((a), (b)); \+ uint32x4x2_t t1_ = vtrnq_u32((c), (d)); \+ (a) = vcombine_u32(vget_low_u32(t0_.val[0]), \+ vget_low_u32(t1_.val[0])); \+ (b) = vcombine_u32(vget_low_u32(t0_.val[1]), \+ vget_low_u32(t1_.val[1])); \+ (c) = vcombine_u32(vget_high_u32(t0_.val[0]), \+ vget_high_u32(t1_.val[0])); \+ (d) = vcombine_u32(vget_high_u32(t0_.val[1]), \+ vget_high_u32(t1_.val[1])); \+ } while (0)++/*+ * Four blocks with counters d[12], d[12]+1, d[12]+2 and d[12]+3. The+ * caller keeps the state's counter, and only calls this when those four+ * do not carry into d[13].+ */+static inline void core4(int rounds, const crypton_chacha_state *in,+ const uint8_t *src, uint8_t *dst, int combine)+{+ static const uint8_t rot8_tbl[16] =+ { 3,0,1,2, 7,4,5,6, 11,8,9,10, 15,12,13,14 };+ const uint8x16_t rot8 = vld1q_u8(rot8_tbl);+ uint32x4_t v0, v1, v2, v3, v4, v5, v6, v7;+ uint32x4_t v8, v9, v10, v11, v12, v13, v14, v15;+ const uint32_t c = in->d[12];+ const uint32_t ctr4[4] = { c, c + 1, c + 2, c + 3 };+ int i;++ /*+ * Only the working state is kept in registers. The initial state has+ * to be added back at the end, but holding a second copy of it would+ * want thirty-two registers for that alone, and the machine has+ * thirty-two in total; read it again instead, from memory that is+ * certainly warm.+ */+#define SET(n) v##n = vdupq_n_u32(in->d[n])+ SET(0); SET(1); SET(2); SET(3);+ SET(4); SET(5); SET(6); SET(7);+ SET(8); SET(9); SET(10); SET(11);+ SET(13); SET(14); SET(15);+#undef SET+ v12 = vld1q_u32(ctr4);++ for (i = rounds; i > 0; i -= 2) {+ QR(v0, v4, v8, v12);+ QR(v1, v5, v9, v13);+ QR(v2, v6, v10, v14);+ QR(v3, v7, v11, v15);++ QR(v0, v5, v10, v15);+ QR(v1, v6, v11, v12);+ QR(v2, v7, v8, v13);+ QR(v3, v4, v9, v14);+ }++#define ADD(n) v##n = vaddq_u32(v##n, vdupq_n_u32(in->d[n]))+ ADD(0); ADD(1); ADD(2); ADD(3);+ ADD(4); ADD(5); ADD(6); ADD(7);+ ADD(8); ADD(9); ADD(10); ADD(11);+ ADD(13); ADD(14); ADD(15);+#undef ADD+ v12 = vaddq_u32(v12, vld1q_u32(ctr4));++ TRANSPOSE(v0, v1, v2, v3);+ TRANSPOSE(v4, v5, v6, v7);+ TRANSPOSE(v8, v9, v10, v11);+ TRANSPOSE(v12, v13, v14, v15);++ /*+ * Each piece is exclusive-ored with the input and stored where it+ * belongs as it comes out. Writing the keystream to a buffer and+ * reading it back to combine it cost a pass over every byte.+ */+#define ST(j, g, v) \+ do { \+ uint8x16_t o_ = vreinterpretq_u8_u32(v); \+ if (combine) \+ o_ = veorq_u8(o_, vld1q_u8(src + 64 * (j) \+ + 4 * (g))); \+ vst1q_u8(dst + 64 * (j) + 4 * (g), o_); \+ } while (0)+ ST(0, 0, v0); ST(1, 0, v1); ST(2, 0, v2); ST(3, 0, v3);+ ST(0, 4, v4); ST(1, 4, v5); ST(2, 4, v6); ST(3, 4, v7);+ ST(0, 8, v8); ST(1, 8, v9); ST(2, 8, v10); ST(3, 8, v11);+ ST(0, 12, v12); ST(1, 12, v13); ST(2, 12, v14); ST(3, 12, v15);+#undef ST+}++void crypton_chacha_simd_combine(int rounds, uint8_t *dst, const uint8_t *src,+ const crypton_chacha_state *in)+{+ core4(rounds, in, src, dst, 1);+}++void crypton_chacha_simd_generate(int rounds, uint8_t *dst, const crypton_chacha_state *in)+{+ core4(rounds, in, NULL, dst, 0);+}++/* NEON has no wider sibling to choose between, so the answer is fixed. */+int crypton_chacha_simd_width(void)+{+ return 4;+}
@@ -0,0 +1,105 @@+/*+ * ChaCha with SSE, four blocks at a time, and the choice of which x86+ * version to run.+ *+ * Word i of four blocks goes in lane i of one register, so every quarter+ * round is one operation on whole registers and no lane moves between the+ * column and the diagonal rounds. Only the counter differs between the+ * four blocks.+ *+ * SSE2 is part of the x86-64 baseline and needs no check. SSSE3 takes the+ * rotates by sixteen and eight in one instruction each, and AVX2 -- in+ * chacha_avx2.c -- carries eight blocks instead of four; both are reached+ * only after crypton_x86_simd_features() says so. Both also need function+ * attributes to sit in a translation unit that is otherwise baseline, so+ * with use_target_attributes turned off only the SSE2 version is built.+ */++#include <stdint.h>+#include <emmintrin.h>+#ifdef WITH_TARGET_ATTRIBUTES+#include <tmmintrin.h>+#endif+#include "crypton_chacha.h"+#include "crypton_cpu.h"++#define SIZED(n) n##_sse2+#define TARGET+#define ROL(x, n) _mm_or_si128(_mm_slli_epi32((x), (n)), _mm_srli_epi32((x), 32 - (n)))+#include <chacha_sse_impl.c>+#undef SIZED+#undef TARGET+#undef ROL++#ifdef WITH_TARGET_ATTRIBUTES++static const int8_t rot16_tbl[16] = { 2,3,0,1, 6,7,4,5, 10,11,8,9, 14,15,12,13 };+static const int8_t rot8_tbl[16] = { 3,0,1,2, 7,4,5,6, 11,8,9,10, 15,12,13,14 };++#define SIZED(n) n##_ssse3+#define TARGET __attribute__((target("ssse3")))+#define ROL(x, n) \+ ((n) == 16 ? _mm_shuffle_epi8((x), _mm_loadu_si128((const __m128i *) rot16_tbl)) \+ : (n) == 8 ? _mm_shuffle_epi8((x), _mm_loadu_si128((const __m128i *) rot8_tbl)) \+ : _mm_or_si128(_mm_slli_epi32((x), (n)), _mm_srli_epi32((x), 32 - (n))))+#include <chacha_sse_impl.c>+#undef SIZED+#undef TARGET+#undef ROL++void crypton_chacha_avx2_combine(int rounds, uint8_t *dst, const uint8_t *src,+ const crypton_chacha_state *in);+void crypton_chacha_avx2_generate(int rounds, uint8_t *dst, const crypton_chacha_state *in);++#endif++/* how many blocks a call covers, and which version does it */+enum { IMPL_UNRESOLVED = 0, IMPL_SSE2, IMPL_SSSE3, IMPL_AVX2 };++static int impl = IMPL_UNRESOLVED;++/* Two threads racing to answer this both write the same value. */+static int resolve(void)+{+#ifdef WITH_TARGET_ATTRIBUTES+ uint32_t f = crypton_x86_simd_features();++ if (f & CRYPTON_X86_AVX2)+ impl = IMPL_AVX2;+ else if (f & CRYPTON_X86_SSSE3)+ impl = IMPL_SSSE3;+ else+#endif+ impl = IMPL_SSE2;+ return impl;+}++int crypton_chacha_simd_width(void)+{+ int i = impl ? impl : resolve();++ return i == IMPL_AVX2 ? 8 : 4;+}++void crypton_chacha_simd_combine(int rounds, uint8_t *dst, const uint8_t *src,+ const crypton_chacha_state *in)+{+ switch (impl ? impl : resolve()) {+#ifdef WITH_TARGET_ATTRIBUTES+ case IMPL_AVX2: crypton_chacha_avx2_combine(rounds, dst, src, in); return;+ case IMPL_SSSE3: combine_ssse3(rounds, dst, src, in); return;+#endif+ default: combine_sse2(rounds, dst, src, in); return;+ }+}++void crypton_chacha_simd_generate(int rounds, uint8_t *dst, const crypton_chacha_state *in)+{+ switch (impl ? impl : resolve()) {+#ifdef WITH_TARGET_ATTRIBUTES+ case IMPL_AVX2: crypton_chacha_avx2_generate(rounds, dst, in); return;+ case IMPL_SSSE3: generate_ssse3(rounds, dst, in); return;+#endif+ default: generate_sse2(rounds, dst, in); return;+ }+}
@@ -0,0 +1,114 @@+/*+ * Included from chacha_sse2.c once per instruction set, with SIZED()+ * naming the functions, ROL() rotating a lane and TARGET saying what the+ * functions may use. The body is identical; only the two rotates by+ * sixteen and eight differ, and only because SSSE3 can do each in one+ * PSHUFB where SSE2 needs a shift, a shift and an or.+ */++/*+ * Four blocks with counters d[12] .. d[12]+3. The caller keeps the+ * state's counter and only calls this when those four do not carry into+ * d[13].+ */+TARGET+static inline void SIZED(core4)(int rounds, const crypton_chacha_state *in,+ const uint8_t *src, uint8_t *dst, int combine)+{+ __m128i v0, v1, v2, v3, v4, v5, v6, v7;+ __m128i v8, v9, v10, v11, v12, v13, v14, v15;+ const uint32_t c = in->d[12];+ int i;++ /*+ * Sixteen registers hold the working state and the machine has+ * sixteen, so the initial state is read again at the end rather than+ * kept in a second set.+ */+#define SET(n) v##n = _mm_set1_epi32((int) in->d[n])+ SET(0); SET(1); SET(2); SET(3);+ SET(4); SET(5); SET(6); SET(7);+ SET(8); SET(9); SET(10); SET(11);+ SET(13); SET(14); SET(15);+#undef SET+ v12 = _mm_setr_epi32((int) c, (int) (c + 1), (int) (c + 2), (int) (c + 3));++#define QR(a, b, cc, d) \+ a = _mm_add_epi32(a, b); d = ROL(_mm_xor_si128(d, a), 16); \+ cc = _mm_add_epi32(cc, d); b = ROL(_mm_xor_si128(b, cc), 12); \+ a = _mm_add_epi32(a, b); d = ROL(_mm_xor_si128(d, a), 8); \+ cc = _mm_add_epi32(cc, d); b = ROL(_mm_xor_si128(b, cc), 7)++ for (i = rounds; i > 0; i -= 2) {+ QR(v0, v4, v8, v12);+ QR(v1, v5, v9, v13);+ QR(v2, v6, v10, v14);+ QR(v3, v7, v11, v15);++ QR(v0, v5, v10, v15);+ QR(v1, v6, v11, v12);+ QR(v2, v7, v8, v13);+ QR(v3, v4, v9, v14);+ }+#undef QR++#define ADD(n) v##n = _mm_add_epi32(v##n, _mm_set1_epi32((int) in->d[n]))+ ADD(0); ADD(1); ADD(2); ADD(3);+ ADD(4); ADD(5); ADD(6); ADD(7);+ ADD(8); ADD(9); ADD(10); ADD(11);+ ADD(13); ADD(14); ADD(15);+#undef ADD+ v12 = _mm_add_epi32(v12, _mm_setr_epi32((int) c, (int) (c + 1),+ (int) (c + 2), (int) (c + 3)));++ /* four registers holding word w of blocks 0..3 become four holding+ * words w..w+3 of one block each, the order they are written in */+#define TRANSPOSE(qa, qb, qc, qd) \+ do { \+ __m128i t0_ = _mm_unpacklo_epi32(qa, qb); \+ __m128i t1_ = _mm_unpackhi_epi32(qa, qb); \+ __m128i t2_ = _mm_unpacklo_epi32(qc, qd); \+ __m128i t3_ = _mm_unpackhi_epi32(qc, qd); \+ qa = _mm_unpacklo_epi64(t0_, t2_); \+ qb = _mm_unpackhi_epi64(t0_, t2_); \+ qc = _mm_unpacklo_epi64(t1_, t3_); \+ qd = _mm_unpackhi_epi64(t1_, t3_); \+ } while (0)+ TRANSPOSE(v0, v1, v2, v3);+ TRANSPOSE(v4, v5, v6, v7);+ TRANSPOSE(v8, v9, v10, v11);+ TRANSPOSE(v12, v13, v14, v15);+#undef TRANSPOSE++ /*+ * Each piece is exclusive-ored with the input and stored where it+ * belongs as it comes out. Writing the keystream to a buffer and+ * reading it back to combine it cost a pass over every byte.+ */+#define ST(j, g, v) \+ do { \+ __m128i o_ = (v); \+ if (combine) \+ o_ = _mm_xor_si128(o_, _mm_loadu_si128( \+ (const __m128i *) (src + 64 * (j) + 4 * (g)))); \+ _mm_storeu_si128((__m128i *) (dst + 64 * (j) + 4 * (g)), o_); \+ } while (0)+ ST(0, 0, v0); ST(1, 0, v1); ST(2, 0, v2); ST(3, 0, v3);+ ST(0, 4, v4); ST(1, 4, v5); ST(2, 4, v6); ST(3, 4, v7);+ ST(0, 8, v8); ST(1, 8, v9); ST(2, 8, v10); ST(3, 8, v11);+ ST(0, 12, v12); ST(1, 12, v13); ST(2, 12, v14); ST(3, 12, v15);+#undef ST+}++TARGET+static void SIZED(combine)(int rounds, uint8_t *dst, const uint8_t *src,+ const crypton_chacha_state *in)+{+ SIZED(core4)(rounds, in, src, dst, 1);+}++TARGET+static void SIZED(generate)(int rounds, uint8_t *dst, const crypton_chacha_state *in)+{+ SIZED(core4)(rounds, in, NULL, dst, 0);+}
@@ -38,6 +38,9 @@ #include <aes/generic.h> #include <aes/gf.h> #include <aes/x86ni.h>+#ifdef WITH_GCM_FUSED+#include <aes/gcm_fused_x86.h>+#endif void crypton_aes_generic_encrypt_ecb(aes_block *output, aes_key *key, aes_block *input, uint32_t nb_blocks); void crypton_aes_generic_decrypt_ecb(aes_block *output, aes_key *key, aes_block *input, uint32_t nb_blocks);@@ -56,6 +59,42 @@ void crypton_aes_generic_ccm_encrypt(uint8_t *output, aes_ccm *ccm, aes_key *key, uint8_t *input, uint32_t length); void crypton_aes_generic_ccm_decrypt(uint8_t *output, aes_ccm *ccm, aes_key *key, uint8_t *input, uint32_t length); +#ifdef WITH_ARMV8_CRYPTO+void crypton_aes_armv8_init(aes_key *key, uint8_t *origkey, uint8_t size);+int crypton_aes_armv8_gcm_fused_dec(uint8_t *out, const block128 *ht,+ aes_key *key, const uint8_t *nonce,+ const uint8_t *aad, uint32_t aadlen,+ const uint8_t *in, uint32_t inlen,+ const uint8_t *tag, uint32_t taglen,+ uint8_t *outtag);+void crypton_aes_armv8_gcm_fused(uint8_t *out, const block128 *ht,+ aes_key *key, const uint8_t *nonce,+ const uint8_t *aad, uint32_t aadlen,+ const uint8_t *in, uint32_t inlen,+ uint32_t taglen, aes_key *hpkey,+ uint32_t sampleoff, uint8_t *mask);+#define ARMV8_DECLS(sz) \+ void crypton_aes_armv8_encrypt_block##sz(aes_block *output, aes_key *key, aes_block *input); \+ void crypton_aes_armv8_decrypt_block##sz(aes_block *output, aes_key *key, aes_block *input); \+ void crypton_aes_armv8_encrypt_ecb##sz(aes_block *output, aes_key *key, aes_block *input, uint32_t nb_blocks); \+ void crypton_aes_armv8_decrypt_ecb##sz(aes_block *output, aes_key *key, aes_block *input, uint32_t nb_blocks); \+ void crypton_aes_armv8_encrypt_cbc##sz(aes_block *output, aes_key *key, aes_block *iv, aes_block *input, uint32_t nb_blocks); \+ void crypton_aes_armv8_decrypt_cbc##sz(aes_block *output, aes_key *key, aes_block *iv, aes_block *input, uint32_t nb_blocks); \+ void crypton_aes_armv8_encrypt_ctr##sz(uint8_t *output, aes_key *key, aes_block *iv, uint8_t *input, uint32_t len); \+ void crypton_aes_armv8_gcm_encrypt##sz(uint8_t *output, aes_gcm *gcm, aes_key *key, uint8_t *input, uint32_t length); \+ void crypton_aes_armv8_gcm_decrypt##sz(uint8_t *output, aes_gcm *gcm, aes_key *key, uint8_t *input, uint32_t length); \+ void crypton_aes_armv8_encrypt_xts##sz(aes_block *output, aes_key *k1, aes_key *k2, aes_block *dataunit, uint32_t spoint, aes_block *input, uint32_t nb_blocks); \+ void crypton_aes_armv8_decrypt_xts##sz(aes_block *output, aes_key *k1, aes_key *k2, aes_block *dataunit, uint32_t spoint, aes_block *input, uint32_t nb_blocks);+ARMV8_DECLS(128)+ARMV8_DECLS(192)+ARMV8_DECLS(256)+int crypton_aes_armv8_available(void);+int crypton_aes_armv8_pmull_available(void);+void crypton_aes_armv8_hinit_pmull(block128 *htable, const block128 *h);+void crypton_aes_armv8_gf_mul_pmull(block128 *a, const block128 *htable);+void crypton_aes_armv8_gf_mul4_pmull(block128 *a, const block128 *blocks, const block128 *htable);+#endif+ enum { /* init */ INIT_128, INIT_192, INIT_256,@@ -85,7 +124,7 @@ ENCRYPT_CCM_128, ENCRYPT_CCM_192, ENCRYPT_CCM_256, DECRYPT_CCM_128, DECRYPT_CCM_192, DECRYPT_CCM_256, /* ghash */- GHASH_HINIT, GHASH_GF_MUL,+ GHASH_HINIT, GHASH_GF_MUL, GHASH_GF_MUL4, }; void *crypton_aes_branch_table[] = {@@ -153,6 +192,7 @@ /* GHASH */ [GHASH_HINIT] = crypton_aes_generic_hinit, [GHASH_GF_MUL] = crypton_aes_generic_gf_mul,+ [GHASH_GF_MUL4] = crypton_aes_generic_gf_mul4, }; typedef void (*init_f)(aes_key *, uint8_t *, uint8_t);@@ -166,8 +206,9 @@ typedef void (*block_f)(aes_block *output, aes_key *key, aes_block *input); typedef void (*hinit_f)(table_4bit htable, const block128 *h); typedef void (*gf_mul_f)(block128 *a, const table_4bit htable);+typedef void (*gf_mul4_f)(block128 *a, const block128 *blocks, const table_4bit htable); -#ifdef WITH_AESNI+#if defined(WITH_AESNI) || defined(WITH_ARMV8_CRYPTO) #define GET_INIT(strength) \ ((init_f) (crypton_aes_branch_table[INIT_128 + strength])) #define GET_ECB_ENCRYPT(strength) \@@ -206,6 +247,8 @@ (((hinit_f) (crypton_aes_branch_table[GHASH_HINIT]))(t,h)) #define crypton_gf_mul(a,t) \ (((gf_mul_f) (crypton_aes_branch_table[GHASH_GF_MUL]))(a,t))+#define crypton_gf_mul4(a,b,t) \+ (((gf_mul4_f) (crypton_aes_branch_table[GHASH_GF_MUL4]))(a,b,t)) #else #define GET_INIT(strenght) crypton_aes_generic_init #define GET_ECB_ENCRYPT(strength) crypton_aes_generic_encrypt_ecb@@ -226,6 +269,7 @@ #define crypton_aes_decrypt_block(o,k,i) crypton_aes_generic_decrypt_block(o,k,i) #define crypton_hinit(t,h) crypton_aes_generic_hinit(t,h) #define crypton_gf_mul(a,t) crypton_aes_generic_gf_mul(a,t)+#define crypton_gf_mul4(a,b,t) crypton_aes_generic_gf_mul4(a,b,t) #endif #define CPU_AESNI 0@@ -242,39 +286,61 @@ crypton_aes_cpu_options[CPU_AESNI] = 1; crypton_aes_branch_table[INIT_128] = crypton_aesni_init;+ crypton_aes_branch_table[INIT_192] = crypton_aesni_init; crypton_aes_branch_table[INIT_256] = crypton_aesni_init; crypton_aes_branch_table[ENCRYPT_BLOCK_128] = crypton_aesni_encrypt_block128; crypton_aes_branch_table[DECRYPT_BLOCK_128] = crypton_aesni_decrypt_block128;+ crypton_aes_branch_table[ENCRYPT_BLOCK_192] = crypton_aesni_encrypt_block192; crypton_aes_branch_table[ENCRYPT_BLOCK_256] = crypton_aesni_encrypt_block256;+ crypton_aes_branch_table[DECRYPT_BLOCK_192] = crypton_aesni_decrypt_block192; crypton_aes_branch_table[DECRYPT_BLOCK_256] = crypton_aesni_decrypt_block256; /* ECB */ crypton_aes_branch_table[ENCRYPT_ECB_128] = crypton_aesni_encrypt_ecb128; crypton_aes_branch_table[DECRYPT_ECB_128] = crypton_aesni_decrypt_ecb128;+ crypton_aes_branch_table[ENCRYPT_ECB_192] = crypton_aesni_encrypt_ecb192; crypton_aes_branch_table[ENCRYPT_ECB_256] = crypton_aesni_encrypt_ecb256;+ crypton_aes_branch_table[DECRYPT_ECB_192] = crypton_aesni_decrypt_ecb192; crypton_aes_branch_table[DECRYPT_ECB_256] = crypton_aesni_decrypt_ecb256; /* CBC */ crypton_aes_branch_table[ENCRYPT_CBC_128] = crypton_aesni_encrypt_cbc128; crypton_aes_branch_table[DECRYPT_CBC_128] = crypton_aesni_decrypt_cbc128;+ crypton_aes_branch_table[ENCRYPT_CBC_192] = crypton_aesni_encrypt_cbc192; crypton_aes_branch_table[ENCRYPT_CBC_256] = crypton_aesni_encrypt_cbc256;+ crypton_aes_branch_table[DECRYPT_CBC_192] = crypton_aesni_decrypt_cbc192; crypton_aes_branch_table[DECRYPT_CBC_256] = crypton_aesni_decrypt_cbc256; /* CTR */ crypton_aes_branch_table[ENCRYPT_CTR_128] = crypton_aesni_encrypt_ctr128;+ crypton_aes_branch_table[ENCRYPT_CTR_192] = crypton_aesni_encrypt_ctr192; crypton_aes_branch_table[ENCRYPT_CTR_256] = crypton_aesni_encrypt_ctr256; /* CTR with 32-bit wrapping */ crypton_aes_branch_table[ENCRYPT_C32_128] = crypton_aesni_encrypt_c32_128;+ crypton_aes_branch_table[ENCRYPT_C32_192] = crypton_aesni_encrypt_c32_192; crypton_aes_branch_table[ENCRYPT_C32_256] = crypton_aesni_encrypt_c32_256; /* XTS */ crypton_aes_branch_table[ENCRYPT_XTS_128] = crypton_aesni_encrypt_xts128;+ crypton_aes_branch_table[ENCRYPT_XTS_192] = crypton_aesni_encrypt_xts192; crypton_aes_branch_table[ENCRYPT_XTS_256] = crypton_aesni_encrypt_xts256;+ crypton_aes_branch_table[DECRYPT_XTS_128] = crypton_aesni_decrypt_xts128;+ crypton_aes_branch_table[DECRYPT_XTS_192] = crypton_aesni_decrypt_xts192;+ crypton_aes_branch_table[DECRYPT_XTS_256] = crypton_aesni_decrypt_xts256; /* GCM */+ /* GCM, where the build has the carry-less multiply, waits below until+ * the processor is known to have it too: the loop calls the multiply+ * rather than reaching it through the branch pointer, so that it can+ * be scheduled against the rounds, and is compiled with the+ * instruction. The AArch64 table waits for PMULL for the same reason.+ */+#ifndef WITH_PCLMUL crypton_aes_branch_table[ENCRYPT_GCM_128] = crypton_aesni_gcm_encrypt128;+ crypton_aes_branch_table[ENCRYPT_GCM_192] = crypton_aesni_gcm_encrypt192; crypton_aes_branch_table[ENCRYPT_GCM_256] = crypton_aesni_gcm_encrypt256;- /* OCB */- /*- crypton_aes_branch_table[ENCRYPT_OCB_128] = crypton_aesni_ocb_encrypt128;- crypton_aes_branch_table[ENCRYPT_OCB_256] = crypton_aesni_ocb_encrypt256;- */+ crypton_aes_branch_table[DECRYPT_GCM_128] = crypton_aesni_gcm_decrypt128;+ crypton_aes_branch_table[DECRYPT_GCM_192] = crypton_aesni_gcm_decrypt192;+ crypton_aes_branch_table[DECRYPT_GCM_256] = crypton_aesni_gcm_decrypt256;+#endif+ /* OCB drives the ECB paths above a group at a time, so it has no+ * entries of its own */ #ifdef WITH_PCLMUL if (!pclmul) return;@@ -283,16 +349,115 @@ /* GHASH */ crypton_aes_branch_table[GHASH_HINIT] = crypton_aesni_hinit_pclmul, crypton_aes_branch_table[GHASH_GF_MUL] = crypton_aesni_gf_mul_pclmul,+ crypton_aes_branch_table[GHASH_GF_MUL4] = crypton_aesni_gf_mul4_pclmul, crypton_aesni_init_pclmul();++ /* and GCM, which needs both halves */+ crypton_aes_branch_table[ENCRYPT_GCM_128] = crypton_aesni_gcm_encrypt128;+ crypton_aes_branch_table[ENCRYPT_GCM_192] = crypton_aesni_gcm_encrypt192;+ crypton_aes_branch_table[ENCRYPT_GCM_256] = crypton_aesni_gcm_encrypt256;+ crypton_aes_branch_table[DECRYPT_GCM_128] = crypton_aesni_gcm_decrypt128;+ crypton_aes_branch_table[DECRYPT_GCM_192] = crypton_aesni_gcm_decrypt192;+ crypton_aes_branch_table[DECRYPT_GCM_256] = crypton_aesni_gcm_decrypt256; #endif } #endif -uint8_t *crypton_aes_cpu_init(void)+#ifdef WITH_ARMV8_CRYPTO+static void initialize_table_armv8(void) {+ if (!crypton_aes_armv8_available())+ return;+ crypton_aes_cpu_options[CPU_AESNI] = 1;++ crypton_aes_branch_table[INIT_128] = crypton_aes_armv8_init;+ crypton_aes_branch_table[INIT_192] = crypton_aes_armv8_init;+ crypton_aes_branch_table[INIT_256] = crypton_aes_armv8_init;++ crypton_aes_branch_table[ENCRYPT_BLOCK_128] = crypton_aes_armv8_encrypt_block128;+ crypton_aes_branch_table[DECRYPT_BLOCK_128] = crypton_aes_armv8_decrypt_block128;+ crypton_aes_branch_table[ENCRYPT_BLOCK_192] = crypton_aes_armv8_encrypt_block192;+ crypton_aes_branch_table[ENCRYPT_BLOCK_256] = crypton_aes_armv8_encrypt_block256;+ crypton_aes_branch_table[DECRYPT_BLOCK_192] = crypton_aes_armv8_decrypt_block192;+ crypton_aes_branch_table[DECRYPT_BLOCK_256] = crypton_aes_armv8_decrypt_block256;+ /* ECB */+ crypton_aes_branch_table[ENCRYPT_ECB_128] = crypton_aes_armv8_encrypt_ecb128;+ crypton_aes_branch_table[DECRYPT_ECB_128] = crypton_aes_armv8_decrypt_ecb128;+ crypton_aes_branch_table[ENCRYPT_ECB_192] = crypton_aes_armv8_encrypt_ecb192;+ crypton_aes_branch_table[ENCRYPT_ECB_256] = crypton_aes_armv8_encrypt_ecb256;+ crypton_aes_branch_table[DECRYPT_ECB_192] = crypton_aes_armv8_decrypt_ecb192;+ crypton_aes_branch_table[DECRYPT_ECB_256] = crypton_aes_armv8_decrypt_ecb256;+ /* CBC */+ crypton_aes_branch_table[ENCRYPT_CBC_128] = crypton_aes_armv8_encrypt_cbc128;+ crypton_aes_branch_table[DECRYPT_CBC_128] = crypton_aes_armv8_decrypt_cbc128;+ crypton_aes_branch_table[ENCRYPT_CBC_192] = crypton_aes_armv8_encrypt_cbc192;+ crypton_aes_branch_table[ENCRYPT_CBC_256] = crypton_aes_armv8_encrypt_cbc256;+ crypton_aes_branch_table[DECRYPT_CBC_192] = crypton_aes_armv8_decrypt_cbc192;+ crypton_aes_branch_table[DECRYPT_CBC_256] = crypton_aes_armv8_decrypt_cbc256;+ /* CTR, which the generic loop would otherwise drive one block at a time */+ crypton_aes_branch_table[ENCRYPT_CTR_128] = crypton_aes_armv8_encrypt_ctr128;+ crypton_aes_branch_table[ENCRYPT_CTR_192] = crypton_aes_armv8_encrypt_ctr192;+ crypton_aes_branch_table[ENCRYPT_CTR_256] = crypton_aes_armv8_encrypt_ctr256;+ /* XTS, likewise, in both directions */+ crypton_aes_branch_table[ENCRYPT_XTS_128] = crypton_aes_armv8_encrypt_xts128;+ crypton_aes_branch_table[DECRYPT_XTS_128] = crypton_aes_armv8_decrypt_xts128;+ crypton_aes_branch_table[ENCRYPT_XTS_192] = crypton_aes_armv8_encrypt_xts192;+ crypton_aes_branch_table[ENCRYPT_XTS_256] = crypton_aes_armv8_encrypt_xts256;+ crypton_aes_branch_table[DECRYPT_XTS_192] = crypton_aes_armv8_decrypt_xts192;+ crypton_aes_branch_table[DECRYPT_XTS_256] = crypton_aes_armv8_decrypt_xts256;++ /* GHASH, which GCM spends its time in once AES itself is fast */+ if (!crypton_aes_armv8_pmull_available())+ return;+ crypton_aes_cpu_options[CPU_PCLMUL] = 1;+ crypton_aes_branch_table[GHASH_HINIT] = crypton_aes_armv8_hinit_pmull;+ crypton_aes_branch_table[GHASH_GF_MUL] = crypton_aes_armv8_gf_mul_pmull;+ crypton_aes_branch_table[GHASH_GF_MUL4] = crypton_aes_armv8_gf_mul4_pmull;++ /* GCM, which needs both halves and so waits until PMULL is known to+ * be there; the generic loop stands in otherwise */+ crypton_aes_branch_table[ENCRYPT_GCM_128] = crypton_aes_armv8_gcm_encrypt128;+ crypton_aes_branch_table[DECRYPT_GCM_128] = crypton_aes_armv8_gcm_decrypt128;+ crypton_aes_branch_table[ENCRYPT_GCM_192] = crypton_aes_armv8_gcm_encrypt192;+ crypton_aes_branch_table[ENCRYPT_GCM_256] = crypton_aes_armv8_gcm_encrypt256;+ crypton_aes_branch_table[DECRYPT_GCM_192] = crypton_aes_armv8_gcm_decrypt192;+ crypton_aes_branch_table[DECRYPT_GCM_256] = crypton_aes_armv8_gcm_decrypt256;+}+#endif++/* Which implementation each entry of the branch table names is decided once,+ * before anything else runs.+ *+ * It used to be decided again on every crypton_aes_initkey, which meant two+ * threads taking a key at the same time were writing the whole table at the+ * same time -- a data race for as long as the program used AES, not just at+ * the start. ThreadSanitizer reports forty-two of them for eight threads+ * doing nothing but taking keys. The values written are the same ones every+ * time and the table starts out holding valid generic implementations, so+ * nothing has ever come of it; it is a race the standard gives no meaning to+ * all the same.+ *+ * A constructor runs while there is one thread, which is the cheapest way to+ * have no race at all: no flag to test, no lock to take, and one less thing+ * for crypton_aes_initkey to do per key. */+static void crypton_aes_cpu_setup(void)+{ #if defined(ARCH_X86) && defined(WITH_AESNI) crypton_aesni_initialize_hw(initialize_table_ni); #endif+#ifdef WITH_ARMV8_CRYPTO+ initialize_table_armv8();+#endif+}++__attribute__((constructor))+static void crypton_aes_cpu_ctor(void)+{+ crypton_aes_cpu_setup();+}++uint8_t *crypton_aes_cpu_init(void)+{ return crypton_aes_cpu_options; } @@ -303,7 +468,6 @@ case 24: key->nbr = 12; key->strength = 1; break; case 32: key->nbr = 14; key->strength = 2; break; }- crypton_aes_cpu_init(); init_f _init = GET_INIT(key->strength); _init(key, origkey, size); }@@ -332,33 +496,6 @@ d(output, key, iv, input, nb_blocks); } -void crypton_aes_gen_ctr(aes_block *output, aes_key *key, const aes_block *iv, uint32_t nb_blocks)-{- aes_block block;-- /* preload IV in block */- block128_copy(&block, iv);-- for ( ; nb_blocks-- > 0; output++, block128_inc_be(&block)) {- crypton_aes_encrypt_block(output, key, &block);- }-}--void crypton_aes_gen_ctr_cont(aes_block *output, aes_key *key, aes_block *iv, uint32_t nb_blocks)-{- aes_block block;-- /* preload IV in block */- block128_copy(&block, iv);-- for ( ; nb_blocks-- > 0; output++, block128_inc_be(&block)) {- crypton_aes_encrypt_block(output, key, &block);- }-- /* copy back the IV */- block128_copy(iv, &block);-}- void crypton_aes_encrypt_ctr(uint8_t *output, aes_key *key, aes_block *iv, uint8_t *input, uint32_t len) { ctr_f e = GET_CTR_ENCRYPT(key->strength);@@ -381,7 +518,8 @@ void crypton_aes_decrypt_xts(aes_block *output, aes_key *k1, aes_key *k2, aes_block *dataunit, uint32_t spoint, aes_block *input, uint32_t nb_blocks) {- crypton_aes_generic_decrypt_xts(output, k1, k2, dataunit, spoint, input, nb_blocks);+ xts_f d = GET_XTS_DECRYPT(k1->strength);+ d(output, k1, k2, dataunit, spoint, input, nb_blocks); } void crypton_aes_gcm_encrypt(uint8_t *output, aes_gcm *gcm, aes_key *key, uint8_t *input, uint32_t length)@@ -426,20 +564,40 @@ crypton_gf_mul(&gcm->tag, gcm->htable); } -void crypton_aes_gcm_init(aes_gcm *gcm, aes_key *key, uint8_t *iv, uint32_t len)+/* Same, for four consecutive blocks. Where the multiply is a carry-less+ * instruction this costs one reduction instead of four. */+static void gcm_ghash_add4(aes_gcm *gcm, const block128 *b) {+ crypton_gf_mul4(&gcm->tag, b, gcm->htable);+}++/* The part of the state that depends on the key alone: H = encrypt_K(0^128)+ * and the table of its multiples. It is 256 of the 320 bytes, and a caller+ * that keeps a key can compute it once instead of once per message. */+void crypton_aes_gcm_key_init(aes_gcm_key *gk, aes_key *key)+{ block128 h;++ block128_zero(&h);+ crypton_aes_encrypt_block(&h, key, &h);+ crypton_hinit(gk->gcm.htable, &h);+#ifdef WITH_GCM_FUSED+ if (crypton_aes_cpu_options[CPU_AESNI]+ && crypton_aes_cpu_options[CPU_PCLMUL])+ crypton_gcm_fused_key_init(&gk->fused, key);+#endif+}++/* Everything else: what the nonce and the message determine. Leaves htable+ * alone, so it runs on a state whose key part is already there. */+static void gcm_message_init(aes_gcm *gcm, uint8_t *iv, uint32_t len)+{ gcm->length_aad = 0; gcm->length_input = 0; - block128_zero(&h); block128_zero(&gcm->tag); block128_zero(&gcm->iv); - /* prepare H : encrypt_K(0^128) */- crypton_aes_encrypt_block(&h, key, &h);- crypton_hinit(gcm->htable, &h);- if (len == 12) { block128_copy_bytes(&gcm->iv, iv, 12); gcm->iv.b[15] = 0x01;@@ -462,9 +620,22 @@ block128_copy_aligned(&gcm->civ, &gcm->iv); } +void crypton_aes_gcm_init(aes_gcm *gcm, aes_key *key, uint8_t *iv, uint32_t len)+{+ block128 h;++ block128_zero(&h);+ crypton_aes_encrypt_block(&h, key, &h);+ crypton_hinit(gcm->htable, &h);+ gcm_message_init(gcm, iv, len);+}+ void crypton_aes_gcm_aad(aes_gcm *gcm, uint8_t *input, uint32_t length) { gcm->length_aad += length;+ for (; length >= 64; input += 64, length -= 64) {+ gcm_ghash_add4(gcm, (const block128 *) input);+ } for (; length >= 16; input += 16, length -= 16) { gcm_ghash_add(gcm, (block128 *) input); }@@ -495,6 +666,195 @@ } } +/* One message, one call. The key part of the state comes in already built,+ * the rest is set up on the stack, and the additional data, the encryption+ * and the tag all happen before returning, so nothing crosses a language+ * boundary between them and no intermediate state is copied out. The output+ * buffer takes the ciphertext and then the tag, so it wants length + taglen+ * bytes. */+void crypton_aes_gcm_full_encrypt(uint8_t *output, const aes_gcm_key *gcmkey, aes_key *key,+ uint8_t *iv, uint32_t ivlen,+ uint8_t *aad, uint32_t aadlen,+ uint8_t *input, uint32_t length, uint32_t taglen)+{+ aes_gcm gcm;+ uint8_t tag[16];++#ifdef WITH_GCM_FUSED+ /* Short messages go the other way: the assembly below will not start+ * on anything under 288 bytes, and under about 1.5 KB the fused path+ * is ahead of it even where it does. */+ if (ivlen == 12 && length <= CRYPTON_GCM_FUSED_MAX_MESSAGE+ && crypton_aes_cpu_options[CPU_AESNI]+ && crypton_aes_cpu_options[CPU_PCLMUL]) {+ crypton_gcm_fused_encrypt(output, &gcmkey->fused, key, iv,+ aad, aadlen, input, length, taglen,+ NULL, 0, NULL);+ return;+ }+#endif+#ifdef WITH_ARMV8_CRYPTO+ /* The same on AArch64, where the framing is what costs: composing the+ * additional data, the encryption and the tag reaches each through the+ * branch table, so the running state goes back to memory between them+ * and a one-block header pays a reduction of its own. Measured on an+ * Apple M4, a 100-byte packet is 3.0x faster taken in one call.+ *+ * No length limit, unlike x86: there is no vendored assembly on this+ * side for a long message to be handed to instead, and measured+ * against the path this replaces it is never slower -- 1.25x at 1440+ * bytes, level from about 6 KB up. */+ if (ivlen == 12+ && crypton_aes_cpu_options[CPU_AESNI]+ && crypton_aes_cpu_options[CPU_PCLMUL]) {+ crypton_aes_armv8_gcm_fused(output, gcmkey->gcm.htable, key, iv,+ aad, aadlen, input, length, taglen,+ NULL, 0, NULL);+ return;+ }+#endif+ memcpy(gcm.htable, gcmkey->gcm.htable, sizeof(gcm.htable));+ gcm_message_init(&gcm, iv, ivlen);+ if (aadlen)+ crypton_aes_gcm_aad(&gcm, aad, aadlen);+ if (length)+ crypton_aes_gcm_encrypt(output, &gcm, key, input, length);+ crypton_aes_gcm_finish(tag, &gcm, key);+ memcpy(output + length, tag, taglen);+}++/* The same, and then the header protection mask. QUIC takes its sample from+ * the ciphertext, so the mask cannot be had before the encryption -- but it+ * can be had before returning, which saves a second crossing for one AES+ * block. The block itself is about a nanosecond; what it saves is the call.+ * sampleoff is where the sixteen bytes of sample start in the output. */+void crypton_aes_gcm_full_encrypt_mask(uint8_t *output, const aes_gcm_key *gcmkey, aes_key *key,+ uint8_t *iv, uint32_t ivlen,+ uint8_t *aad, uint32_t aadlen,+ uint8_t *input, uint32_t length, uint32_t taglen,+ aes_key *hpkey, uint32_t sampleoff, uint8_t *mask)+{+ block128 sample, m;++#ifdef WITH_GCM_FUSED+ /* Here the mask rides in a lane of the AES pipeline that the message+ * length leaves idle, so it costs very nearly nothing on top of the+ * encryption rather than a block of its own. */+ if (ivlen == 12 && length <= CRYPTON_GCM_FUSED_MAX_MESSAGE+ && crypton_aes_cpu_options[CPU_AESNI]+ && crypton_aes_cpu_options[CPU_PCLMUL]) {+ crypton_gcm_fused_encrypt(output, &gcmkey->fused, key, iv,+ aad, aadlen, input, length, taglen,+ hpkey, sampleoff, mask);+ return;+ }+#endif+#ifdef WITH_ARMV8_CRYPTO+ /* The same on AArch64, where the framing is what costs: composing the+ * additional data, the encryption and the tag reaches each through the+ * branch table, so the running state goes back to memory between them+ * and a one-block header pays a reduction of its own. Measured on an+ * Apple M4, a 100-byte packet is 3.3x faster taken in one call. */+ if (ivlen == 12+ && crypton_aes_cpu_options[CPU_AESNI]+ && crypton_aes_cpu_options[CPU_PCLMUL]) {+ crypton_aes_armv8_gcm_fused(output, gcmkey->gcm.htable, key, iv,+ aad, aadlen, input, length, taglen,+ hpkey, sampleoff, mask);+ return;+ }+#endif+ crypton_aes_gcm_full_encrypt(output, gcmkey, key, iv, ivlen, aad, aadlen,+ input, length, taglen);+ /* copied rather than cast: the sample lands wherever the header put it+ * and a block128 is read as 64-bit words */+ memcpy(&sample, output + sampleoff, 16);+ crypton_aes_encrypt_block(&m, hpkey, &sample);+ memcpy(mask, &m, 16);+}++/* The same the other way, with the tag checked here rather than by the+ * caller: returns 1 when it matches and 0 when it does not, comparing every+ * byte either way. The plaintext is written whatever the answer, so a caller+ * that gets 0 must not use it. */+/* Decrypt, and either compare the tag or hand it back.+ *+ * With outtag NULL this is the verifying form: the tag is compared here, a+ * byte at a time over its whole length whichever way the answer goes, and the+ * answer is the return value. With outtag not NULL the computed tag is+ * written there instead and the return value is 1 -- for a caller that holds+ * the expected tag in a form of its own and will compare it itself. */+static int gcm_full_decrypt(uint8_t *output, const aes_gcm_key *gcmkey, aes_key *key,+ uint8_t *iv, uint32_t ivlen,+ uint8_t *aad, uint32_t aadlen,+ uint8_t *input, uint32_t length,+ const uint8_t *tag, uint32_t taglen,+ uint8_t *outtag)+{+ aes_gcm gcm;+ uint8_t expected[16];+ uint32_t i;+ uint8_t diff = 0;++#ifdef WITH_GCM_FUSED+ /* The same as the encryption side, and simpler: what GHASH absorbs+ * here is the ciphertext, which is the input, so the multiplies need+ * not wait for anything. Measured on an Intel Haswell, a 100-byte+ * packet was three times the cost of encrypting one before this. */+ if (ivlen == 12 && length <= CRYPTON_GCM_FUSED_MAX_MESSAGE+ && crypton_aes_cpu_options[CPU_AESNI]+ && crypton_aes_cpu_options[CPU_PCLMUL])+ return crypton_gcm_fused_decrypt(output, &gcmkey->fused, key,+ iv, aad, aadlen, input,+ length, tag, taglen, outtag);+#endif+#ifdef WITH_ARMV8_CRYPTO+ if (ivlen == 12+ && crypton_aes_cpu_options[CPU_AESNI]+ && crypton_aes_cpu_options[CPU_PCLMUL])+ return crypton_aes_armv8_gcm_fused_dec(output, gcmkey->gcm.htable,+ key, iv, aad, aadlen,+ input, length, tag, taglen,+ outtag);+#endif+ memcpy(gcm.htable, gcmkey->gcm.htable, sizeof(gcm.htable));+ gcm_message_init(&gcm, iv, ivlen);+ if (aadlen)+ crypton_aes_gcm_aad(&gcm, aad, aadlen);+ if (length)+ crypton_aes_gcm_decrypt(output, &gcm, key, input, length);+ crypton_aes_gcm_finish(expected, &gcm, key);++ if (outtag) {+ memcpy(outtag, expected, taglen);+ return 1;+ }+ for (i = 0; i < taglen; i++)+ diff |= (uint8_t) (expected[i] ^ tag[i]);+ return diff == 0;+}++int crypton_aes_gcm_full_decrypt(uint8_t *output, const aes_gcm_key *gcmkey, aes_key *key,+ uint8_t *iv, uint32_t ivlen,+ uint8_t *aad, uint32_t aadlen,+ uint8_t *input, uint32_t length,+ const uint8_t *tag, uint32_t taglen)+{+ return gcm_full_decrypt(output, gcmkey, key, iv, ivlen, aad, aadlen,+ input, length, tag, taglen, NULL);+}++void crypton_aes_gcm_full_decrypt_tag(uint8_t *output, uint8_t *outtag,+ const aes_gcm_key *gcmkey, aes_key *key,+ uint8_t *iv, uint32_t ivlen,+ uint8_t *aad, uint32_t aadlen,+ uint8_t *input, uint32_t length,+ uint32_t taglen)+{+ (void) gcm_full_decrypt(output, gcmkey, key, iv, ivlen, aad, aadlen,+ input, length, NULL, taglen, outtag);+}+ static inline uint8_t ccm_b0_flags(uint32_t has_adata, uint32_t m, uint32_t l) { return 8*m + l + (has_adata? 64: 0);@@ -655,6 +1015,21 @@ #undef L_CACHED } +/*+ * OCB's offsets are a running exclusive-or, so they have to be worked out in+ * order, but the block cipher calls under them do not depend on each other:+ * every block is offset, encrypted, and offset again. So the offsets are+ * computed a group at a time and the group goes through ECB together, which+ * is where the code written for the AES instructions interleaves eight blocks+ * and covers the latency of AESENC. One block at a time left that idle and+ * cost four times what GCM costs on the same machine, for a mode that does+ * less work than GCM.+ *+ * Eight is what the ECB paths interleave; a group beyond that gains nothing+ * and only makes the buffers larger.+ */+#define OCB_WAY 8+ void crypton_aes_ocb_init(aes_ocb *ocb, aes_key *key, uint8_t *iv, uint32_t len, uint32_t taglen) { block128 tmp, nonce, ktop;@@ -714,9 +1089,23 @@ void crypton_aes_ocb_aad(aes_ocb *ocb, aes_key *key, uint8_t *input, uint32_t length) { block128 tmp;- unsigned int i;+ block128 buf[OCB_WAY];+ uint32_t blocks = length / 16;+ unsigned int i = 1, j; - for (i=1; i<= length/16; i++, input=input+16) {+ for (; blocks >= OCB_WAY; blocks -= OCB_WAY, input += 16 * OCB_WAY) {+ for (j = 0; j < OCB_WAY; j++, i++) {+ ocb_get_L_i(&tmp, ocb->li, i);+ block128_xor_aligned(&ocb->offset_aad, &tmp);+ block128_vxor(&buf[j], &ocb->offset_aad,+ (block128 *) (input + 16 * j));+ }+ crypton_aes_encrypt_ecb(buf, key, buf, OCB_WAY);+ for (j = 0; j < OCB_WAY; j++)+ block128_xor_aligned(&ocb->sum_aad, &buf[j]);+ }++ for (; blocks > 0; blocks--, i++, input += 16) { ocb_get_L_i(&tmp, ocb->li, i); block128_xor_aligned(&ocb->offset_aad, &tmp); @@ -882,6 +1271,20 @@ aes_block out; gcm->length_input += length;+ /* four blocks at a time, so GHASH can fold them into one reduction */+ for (; length >= 64; input += 64, output += 64, length -= 64) {+ aes_block buf[4];+ int i;++ for (i = 0; i < 4; i++) {+ block128_inc32_be(&gcm->civ);+ crypton_aes_encrypt_block(&buf[i], key, &gcm->civ);+ block128_xor(&buf[i], (block128 *) (input + 16 * i));+ }+ gcm_ghash_add4(gcm, buf);+ for (i = 0; i < 4; i++)+ block128_copy((block128 *) (output + 16 * i), &buf[i]);+ } for (; length >= 16; input += 16, output += 16, length -= 16) { block128_inc32_be(&gcm->civ); @@ -915,6 +1318,19 @@ aes_block out; gcm->length_input += length;+ /* GHASH all four ciphertext blocks before writing any plaintext, since+ * output may be input */+ for (; length >= 64; input += 64, output += 64, length -= 64) {+ int i;++ gcm_ghash_add4(gcm, (const block128 *) input);+ for (i = 0; i < 4; i++) {+ block128_inc32_be(&gcm->civ);+ crypton_aes_encrypt_block(&out, key, &gcm->civ);+ block128_xor(&out, (block128 *) (input + 16 * i));+ block128_copy((block128 *) (output + 16 * i), &out);+ }+ } for (; length >= 16; input += 16, output += 16, length -= 16) { block128_inc32_be(&gcm->civ); @@ -946,10 +1362,37 @@ uint8_t *input, uint32_t length, int encrypt) { block128 tmp, pad;- unsigned int i;+ block128 offsets[OCB_WAY], buf[OCB_WAY];+ uint32_t blocks = length / 16;+ unsigned int i = 1, j; - for (i = 1; i <= length/16; i++, input += 16, output += 16) {- /* Offset_i = Offset_{i-1} xor L_{ntz(i)} */+ for (; blocks >= OCB_WAY;+ blocks -= OCB_WAY, input += 16 * OCB_WAY, output += 16 * OCB_WAY) {+ for (j = 0; j < OCB_WAY; j++, i++) {+ /* Offset_i = Offset_{i-1} xor L_{ntz(i)} */+ ocb_get_L_i(&tmp, ocb->li, i);+ block128_xor_aligned(&ocb->offset_enc, &tmp);+ block128_copy_aligned(&offsets[j], &ocb->offset_enc);+ block128_vxor(&buf[j], &ocb->offset_enc,+ (block128 *) (input + 16 * j));+ }++ if (encrypt)+ crypton_aes_encrypt_ecb(buf, key, buf, OCB_WAY);+ else+ crypton_aes_decrypt_ecb(buf, key, buf, OCB_WAY);++ for (j = 0; j < OCB_WAY; j++) {+ block128_vxor((block128 *) (output + 16 * j),+ &offsets[j], &buf[j]);+ block128_xor(&ocb->sum_enc,+ (block128 *) ((encrypt ? input : output)+ + 16 * j));+ }+ }++ /* and what is left of the message, a block at a time */+ for (; blocks > 0; blocks--, i++, input += 16, output += 16) { ocb_get_L_i(&tmp, ocb->li, i); block128_xor_aligned(&ocb->offset_enc, &tmp);
@@ -55,6 +55,55 @@ uint64_t length_input; } aes_gcm; +/*+ * How many powers of H a key keeps for the fused path in+ * cbits/aes/gcm_fused_x86.c. A power for every block of the message would+ * fold its whole GHASH into one reduction, which is what picotls does, but+ * then the state grows with the longest message a caller might send and a+ * server holding many keys pays it for each. A fixed count costs one+ * reduction per this many blocks and keeps the state one size. Sixteen was+ * measured against 6, 8, 32, 64, 96 and 256: above eight the choice is worth+ * about two per cent, since only messages short enough to take this path at+ * all reach a second batch. Six is worth avoiding -- at 1440 bytes it is+ * slower than not taking the path.+ */+#define CRYPTON_GCM_FUSED_POWERS 16++/*+ * Beyond this many bytes the stitched assembly in cbits/asm is faster than+ * the fused path, so longer messages go there instead. Measured on an Intel+ * Haswell: even at 1440 bytes, the assembly ahead by 12 per cent at 3 KB and+ * 20 per cent at 16 KB, and the fused path ahead by 1.9x at 100 bytes and+ * 1.16x at 1200. QUIC packets fall below this; TLS records do not.+ */+#define CRYPTON_GCM_FUSED_MAX_MESSAGE 1536++/*+ * The powers themselves, each shifted up by one bit, and beside each the+ * halves of it added together for the Karatsuba term. The two are kept+ * adjacent rather than in two arrays: a multiply wants both, and two arrays+ * put them 256 bytes apart, which is two cache lines where this is one.+ *+ * Defined on every platform so that the key state below is one size+ * everywhere; filled only where the fused path is compiled in.+ */+typedef struct {+ struct {+ aes_block h;+ aes_block r;+ } p[CRYPTON_GCM_FUSED_POWERS];+} aes_gcm_fused;++/*+ * Everything a key determines, built once by crypton_aes_gcm_key_init and+ * read by every message sent under that key: the key half of a GCM state,+ * and the powers of H the fused path reads. 832 bytes.+ */+typedef struct {+ aes_gcm gcm;+ aes_gcm_fused fused;+} aes_gcm_key;+ /* size = 4*16+4*4= 80 */ typedef struct { aes_block xi;@@ -95,8 +144,8 @@ void crypton_aes_encrypt_cbc(aes_block *output, aes_key *key, aes_block *iv, aes_block *input, uint32_t nb_blocks); void crypton_aes_decrypt_cbc(aes_block *output, aes_key *key, aes_block *iv, aes_block *input, uint32_t nb_blocks); -void crypton_aes_gen_ctr(aes_block *output, aes_key *key, const aes_block *iv, uint32_t nb_blocks);-void crypton_aes_gen_ctr_cont(aes_block *output, aes_key *key, aes_block *iv, uint32_t nb_blocks);+void crypton_aes_encrypt_ctr(uint8_t *output, aes_key *key, aes_block *iv, uint8_t *input, uint32_t len);+void crypton_aes_encrypt_c32(uint8_t *output, aes_key *key, aes_block *iv, uint8_t *input, uint32_t len); void crypton_aes_encrypt_xts(aes_block *output, aes_key *key, aes_key *key2, aes_block *sector, uint32_t spoint, aes_block *input, uint32_t nb_blocks);@@ -104,6 +153,27 @@ uint32_t spoint, aes_block *input, uint32_t nb_blocks); void crypton_aes_gcm_init(aes_gcm *gcm, aes_key *key, uint8_t *iv, uint32_t len);+void crypton_aes_gcm_key_init(aes_gcm_key *gk, aes_key *key);+void crypton_aes_gcm_full_encrypt(uint8_t *output, const aes_gcm_key *gcmkey, aes_key *key,+ uint8_t *iv, uint32_t ivlen,+ uint8_t *aad, uint32_t aadlen,+ uint8_t *input, uint32_t length, uint32_t taglen);+void crypton_aes_gcm_full_encrypt_mask(uint8_t *output, const aes_gcm_key *gcmkey, aes_key *key,+ uint8_t *iv, uint32_t ivlen,+ uint8_t *aad, uint32_t aadlen,+ uint8_t *input, uint32_t length, uint32_t taglen,+ aes_key *hpkey, uint32_t sampleoff, uint8_t *mask);+int crypton_aes_gcm_full_decrypt(uint8_t *output, const aes_gcm_key *gcmkey, aes_key *key,+ uint8_t *iv, uint32_t ivlen,+ uint8_t *aad, uint32_t aadlen,+ uint8_t *input, uint32_t length,+ const uint8_t *tag, uint32_t taglen);+void crypton_aes_gcm_full_decrypt_tag(uint8_t *output, uint8_t *outtag,+ const aes_gcm_key *gcmkey, aes_key *key,+ uint8_t *iv, uint32_t ivlen,+ uint8_t *aad, uint32_t aadlen,+ uint8_t *input, uint32_t length,+ uint32_t taglen); void crypton_aes_gcm_aad(aes_gcm *gcm, uint8_t *input, uint32_t length); void crypton_aes_gcm_encrypt(uint8_t *output, aes_gcm *gcm, aes_key *key, uint8_t *input, uint32_t length); void crypton_aes_gcm_decrypt(uint8_t *output, aes_gcm *gcm, aes_key *key, uint8_t *input, uint32_t length);
@@ -3,6 +3,8 @@ #include "crypton_bitfn.h" +#include <string.h>+ #if (defined(__i386__)) # define UNALIGNED_ACCESS_OK #elif defined(__x86_64__)@@ -34,107 +36,119 @@ #define need_alignment(p,n) IS_ALIGNED(p,n) #endif -static inline uint32_t load_le32_aligned(const uint8_t *p)-{- return le32_to_cpu(*((uint32_t *) p)); -}+/*+ * Reading and writing a 32- or 64-bit word at a byte pointer.+ *+ * Through memcpy, not a cast to uint32_t * or uint64_t *. A cast is two+ * things the standard does not allow -- a read of the value through the+ * wrong type, and a read at an address that type is not aligned for -- and+ * this file used to do both wherever UNALIGNED_ACCESS_OK is defined, which+ * is i386 and x86-64. UndefinedBehaviorSanitizer reported seventy-eight+ * lines of it.+ *+ * Every compiler crypton is built with turns a memcpy of four or eight bytes+ * into the one load or store the cast used to be, so this is the same code+ * with none of the licence. Where the target cannot do an unaligned load,+ * the compiler is the one that knows, and it emits what the target needs --+ * which is what the byte-at-a-time versions this replaces were for.+ *+ * The _aligned names stay because nineteen files use them. They no longer+ * ask anything of the pointer.+ */ -static inline void store_le32_aligned(uint8_t *dst, const uint32_t v)+static inline uint32_t load_le32(const uint8_t *p) {- *((uint32_t *) dst) = cpu_to_le32(v);-}+ uint32_t v; -static inline void xor_le32_aligned(uint8_t *dst, const uint32_t v)-{- *((uint32_t *) dst) ^= cpu_to_le32(v);+ memcpy(&v, p, sizeof(v));+ return le32_to_cpu(v); } -static inline void store_be32_aligned(uint8_t *dst, const uint32_t v)+static inline uint64_t load_le64(const uint8_t *p) {- *((uint32_t *) dst) = cpu_to_be32(v);-}+ uint64_t v; -static inline void xor_be32_aligned(uint8_t *dst, const uint32_t v)-{- *((uint32_t *) dst) ^= cpu_to_be32(v);+ memcpy(&v, p, sizeof(v));+ return le64_to_cpu(v); } -static inline void store_le64_aligned(uint8_t *dst, const uint64_t v)+static inline uint32_t load_be32(const uint8_t *p) {- *((uint64_t *) dst) = cpu_to_le64(v);-}+ uint32_t v; -static inline void store_be64_aligned(uint8_t *dst, const uint64_t v)-{- *((uint64_t *) dst) = cpu_to_be64(v);+ memcpy(&v, p, sizeof(v));+ return be32_to_cpu(v); } -static inline void xor_be64_aligned(uint8_t *dst, const uint64_t v)+static inline uint64_t load_be64(const uint8_t *p) {- *((uint64_t *) dst) ^= cpu_to_be64(v);-}+ uint64_t v; -#ifdef UNALIGNED_ACCESS_OK-#define load_le32(a) load_le32_aligned(a)-#else-static inline uint32_t load_le32(const uint8_t *p)-{- return ((uint32_t)p[0]) | ((uint32_t)p[1] << 8) | ((uint32_t)p[2] << 16) | ((uint32_t)p[3] << 24);+ memcpy(&v, p, sizeof(v));+ return be64_to_cpu(v); }-#endif -#ifdef UNALIGNED_ACCESS_OK-#define store_le32(a, b) store_le32_aligned(a, b)-#define xor_le32(a, b) xor_le32_aligned(a, b)-#else static inline void store_le32(uint8_t *dst, const uint32_t v) {- dst[0] = v; dst[1] = v >> 8; dst[2] = v >> 16; dst[3] = v >> 24;+ uint32_t w = cpu_to_le32(v);++ memcpy(dst, &w, sizeof(w)); }+ static inline void xor_le32(uint8_t *dst, const uint32_t v) {- dst[0] ^= v; dst[1] ^= v >> 8; dst[2] ^= v >> 16; dst[3] ^= v >> 24;+ store_le32(dst, le32_to_cpu(load_le32(dst)) ^ v); }-#endif -#ifdef UNALIGNED_ACCESS_OK-#define store_be32(a, b) store_be32_aligned(a, b)-#define xor_be32(a, b) xor_be32_aligned(a, b)-#else static inline void store_be32(uint8_t *dst, const uint32_t v) {- dst[3] = v; dst[2] = v >> 8; dst[1] = v >> 16; dst[0] = v >> 24;+ uint32_t w = cpu_to_be32(v);++ memcpy(dst, &w, sizeof(w)); }+ static inline void xor_be32(uint8_t *dst, const uint32_t v) {- dst[3] ^= v; dst[2] ^= v >> 8; dst[1] ^= v >> 16; dst[0] ^= v >> 24;+ uint32_t w;++ memcpy(&w, dst, sizeof(w));+ w ^= cpu_to_be32(v);+ memcpy(dst, &w, sizeof(w)); }-#endif -#ifdef UNALIGNED_ACCESS_OK-#define store_le64(a, b) store_le64_aligned(a, b)-#else static inline void store_le64(uint8_t *dst, const uint64_t v) {- dst[0] = v ; dst[1] = v >> 8 ; dst[2] = v >> 16; dst[3] = v >> 24;- dst[4] = v >> 32; dst[5] = v >> 40; dst[6] = v >> 48; dst[7] = v >> 56;+ uint64_t w = cpu_to_le64(v);++ memcpy(dst, &w, sizeof(w)); }-#endif -#ifdef UNALIGNED_ACCESS_OK-#define store_be64(a, b) store_be64_aligned(a, b)-#define xor_be64(a, b) xor_be64_aligned(a, b)-#else static inline void store_be64(uint8_t *dst, const uint64_t v) {- dst[7] = v ; dst[6] = v >> 8 ; dst[5] = v >> 16; dst[4] = v >> 24;- dst[3] = v >> 32; dst[2] = v >> 40; dst[1] = v >> 48; dst[0] = v >> 56;+ uint64_t w = cpu_to_be64(v);++ memcpy(dst, &w, sizeof(w)); }+ static inline void xor_be64(uint8_t *dst, const uint64_t v) {- dst[7] ^= v ; dst[6] ^= v >> 8 ; dst[5] ^= v >> 16; dst[4] ^= v >> 24;- dst[3] ^= v >> 32; dst[2] ^= v >> 40; dst[1] ^= v >> 48; dst[0] ^= v >> 56;+ uint64_t w;++ memcpy(&w, dst, sizeof(w));+ w ^= cpu_to_be64(v);+ memcpy(dst, &w, sizeof(w)); }-#endif++#define load_le32_aligned(p) load_le32(p)+#define load_le64_aligned(p) load_le64(p)+#define load_be32_aligned(p) load_be32(p)+#define load_be64_aligned(p) load_be64(p)+#define store_le32_aligned(d, v) store_le32(d, v)+#define xor_le32_aligned(d, v) xor_le32(d, v)+#define store_be32_aligned(d, v) store_be32(d, v)+#define xor_be32_aligned(d, v) xor_be32(d, v)+#define store_le64_aligned(d, v) store_le64(d, v)+#define store_be64_aligned(d, v) store_be64(d, v)+#define xor_be64_aligned(d, v) xor_be64(d, v) #endif
@@ -0,0 +1,40 @@+/*+ * Asking for an AArch64 extension on one function.+ *+ * The instructions these files use are extensions, so a translation unit+ * compiled for the baseline may not emit them, and the function that does has+ * to say which extension it needs. The two compilers spell that differently+ * and did not always:+ *+ * clang target("+crypto") for as long as it matters+ * GCC 13 and up target("+crypto") as well+ * GCC before 13 target("arch=armv8-a+crypto") -- the bare "+feature" form+ * is not understood, and the extension never reaches the+ * function, so an always_inline intrinsic that needs it+ * fails to inline and the build stops+ *+ * That last case is #273: gcc 12.2 on an aarch64 Linux could not build+ * cbits/sha3_armv8.c at all. Naming the architecture as well as the+ * extension is understood by every version of both compilers, so GCC is given+ * that spelling and clang keeps the shorter one, which leaves whatever+ * baseline the caller chose alone.+ */+#ifndef CRYPTON_ARMV8_TARGET_H+#define CRYPTON_ARMV8_TARGET_H++#ifdef WITH_TARGET_ATTRIBUTES+#if defined(__clang__)+#define CRYPTON_TARGET_ARMV8_CRYPTO __attribute__((target("+crypto")))+#define CRYPTON_TARGET_ARMV8_SHA3 __attribute__((target("+sha3")))+#else+#define CRYPTON_TARGET_ARMV8_CRYPTO \+ __attribute__((target("arch=armv8-a+crypto")))+#define CRYPTON_TARGET_ARMV8_SHA3 \+ __attribute__((target("arch=armv8.2-a+sha3")))+#endif+#else+#define CRYPTON_TARGET_ARMV8_CRYPTO+#define CRYPTON_TARGET_ARMV8_SHA3+#endif++#endif
@@ -0,0 +1,512 @@+/*+ * Arithmetic on numbers held as arrays of limbs, least significant first.+ *+ * The modular multiplication is Montgomery's, and every choice it makes --+ * which of two numbers to keep after the final subtraction, which entry of a+ * table to take -- is made with a mask rather than a branch, so that the+ * values being worked on do not steer the work. The exponentiation in+ * crypton_powm.c and the curve arithmetic in crypton_ecc.c are both built on+ * this.+ *+ * Everything here is static inline: each file that includes it gets its own+ * copy, which the compiler can specialise to the sizes it uses.+ */+#ifndef CRYPTON_BIGNUM_H+#define CRYPTON_BIGNUM_H++#include <stdint.h>+#include <string.h>++#if defined(__SIZEOF_INT128__)+typedef uint64_t limb_t;+typedef unsigned __int128 dlimb_t;+#define LIMB_BITS 64+#else+typedef uint32_t limb_t;+typedef uint64_t dlimb_t;+#define LIMB_BITS 32+#endif++#define LIMB_BYTES (LIMB_BITS / 8)++/* four bits of exponent per window, so a table of sixteen and no leftover+ * bits: a byte holds exactly two windows */+#define WINDOW_BITS 4+#define TABLE_SIZE (1 << WINDOW_BITS)++/* r = a - b, returning the borrow out of the top */+static inline limb_t sub_n(limb_t *r, const limb_t *a, const limb_t *b, uint32_t n)+{+ limb_t borrow = 0;+ uint32_t i;++ for (i = 0; i < n; i++) {+ limb_t ai = a[i], bi = b[i];+ limb_t d = ai - bi - borrow;+ /* borrow out, without branching */+ borrow = ((~ai & bi) | (~(ai ^ bi) & d)) >> (LIMB_BITS - 1);+ r[i] = d;+ }+ return borrow;+}++/* r = a + b, returning the carry out of the top */+static inline limb_t add_n(limb_t *r, const limb_t *a, const limb_t *b,+ uint32_t n)+{+ limb_t carry = 0;+ uint32_t i;++ for (i = 0; i < n; i++) {+ dlimb_t s = (dlimb_t) a[i] + b[i] + carry;++ r[i] = (limb_t) s;+ carry = (limb_t) (s >> LIMB_BITS);+ }+ return carry;+}++/* a = 2a, returning the bit shifted out of the top */+static inline limb_t shl1(limb_t *a, uint32_t n)+{+ limb_t carry = 0;+ uint32_t i;++ for (i = 0; i < n; i++) {+ limb_t next = a[i] >> (LIMB_BITS - 1);+ a[i] = (a[i] << 1) | carry;+ carry = next;+ }+ return carry;+}++/* r = take ? a : b */+static inline void select_n(limb_t *r, const limb_t *a, const limb_t *b, limb_t take,+ uint32_t n)+{+ limb_t mask = (limb_t) 0 - take;+ uint32_t i;++ for (i = 0; i < n; i++)+ r[i] = (a[i] & mask) | (b[i] & ~mask);+}++/* all ones when a and b are equal, zero otherwise */+static inline limb_t eq_mask(limb_t a, limb_t b)+{+ limb_t d = a ^ b;+ limb_t nz = d | ((limb_t) 0 - d); /* top bit set unless d is zero */++ return (limb_t) 0 - ((nz >> (LIMB_BITS - 1)) ^ 1);+}++/* -m^-1 mod 2^LIMB_BITS, for odd m */+static inline limb_t mont_n0(limb_t m0)+{+ limb_t inv = 1;+ int i;++ /* Newton's iteration doubles the number of correct bits each time */+ for (i = 0; i < 6; i++)+ inv *= (limb_t) 2 - m0 * inv;+ return (limb_t) 0 - inv;+}++/*+ * t += a * b over n limbs, returning the carry. This is where nearly all of+ * the time goes.+ *+ * The C below takes the limbs eight at a time; what is left over at the end+ * is taken one at a time. On AArch64 the compiler writes each limb as+ * `mul`, `umulh`, `adds`, `cset`, `adds`, `adc`: the carry out of one+ * 128-bit addition leaves the flags for a general register and is added back+ * in the next, because in C each addition is a statement of its own. Two of+ * those instructions are that round trip.+ *+ * Three attempts to take them back measured worse than the C, and are+ * written down here so that they are not tried again -- one RSA-2048 CRT+ * private operation on an Apple M4, best of many:+ *+ * this loop in inline assembly, a carry chain per limb 654.7 us+ * the same with the loads hoisted out of the chain 644.5+ * the multiply interleaved with its reduction (CIOS) 604.1+ * the C below 590.1+ * what the AArch64 block does, four limbs and two chains 514.9+ * the same, with the ragged end of the row written out too 503.4+ *+ * The first two lose because a chain per limb serialises what the spare+ * `cset` lets overlap: `adds`, `adc`, `adds`, `adc` is four dependent steps+ * per limb, and a wide out-of-order core would rather have the extra+ * instruction than the dependency. The third loses because shifting the+ * accumulator down a limb each round costs more than the round trip through+ * 2n limbs that it saves. What works is neither: four limbs to an+ * iteration with the flags carrying through two long chains, one for the low+ * halves of the products and one for the high halves a place up.+ *+ * There is more still there. OpenSSL's armv8-mont.pl runs a 16-limb+ * Montgomery multiplication at about 0.98 multiply-accumulates per cycle;+ * this file was at 0.55 and the block below brings it to 0.64, where 1.0 is+ * the ceiling -- a multiply-accumulate is two instructions and the machine+ * issues two multiplies a cycle. That code cannot be borrowed: it is in+ * OpenSSL's tree only, under Apache-2.0, and CRYPTOGAMS, which this library+ * does vendor from, publishes no Montgomery generator at all. BearSSL's+ * only ARM assembly is 32-bit Thumb for Cortex-M0 to M3, with fifteen-bit+ * limbs for cores that have no fast multiplier, and Botan's AArch64 inline+ * assembly is the `mul`/`umulh`/`adds`/`adc` primitive the compiler already+ * emits.+ */+#if defined(__aarch64__) && LIMB_BITS == 64 \+ && (defined(__GNUC__) || defined(__clang__))+/*+ * Four limbs to an iteration, accumulated in two chains rather than one per+ * limb: the low halves of the four products, with the carry coming in, are+ * one run of `adds` and `adcs`, and the high halves shifted up a place are+ * another. The flags carry the whole way through each, which is what the C+ * above cannot say and what it pays for in `cset` and an extra add.+ *+ * The arrangement is the one in Go's crypto/internal/fips140/bigmod+ * (nat_arm64.s, addMulVVWx), which is BSD-3-Clause like this library --+ * cbits/LICENSE.go carries its notice. Written out here in the assembler+ * this file's compiler speaks.+ */+static inline limb_t addmul_1(limb_t *t, const limb_t *a, uint32_t n, limb_t b)+{+ uint64_t carry = 0;+ uint64_t x0, x1, x2, x3, z0, z1, z2, z3;+ uint64_t l0, l1, l2, l3, h0, h1, h2, h3;+ uint64_t blocks = n / 4, left = n % 4;++ if (blocks) {+ __asm__ volatile(+ "1:\n\t"+ "ldp %[x0], %[x1], [%[a]], #16\n\t"+ "ldp %[x2], %[x3], [%[a]], #16\n\t"+ "ldp %[z0], %[z1], [%[t]]\n\t"+ /* the low halves, one place up from the second chain, with+ * the carry that came in */+ "adds %[z0], %[z0], %[c]\n\t"+ "mul %[l1], %[x1], %[b]\n\t"+ "adcs %[z1], %[z1], %[l1]\n\t"+ "mul %[l2], %[x2], %[b]\n\t"+ "ldp %[z2], %[z3], [%[t], #16]\n\t"+ "adcs %[z2], %[z2], %[l2]\n\t"+ "mul %[l3], %[x3], %[b]\n\t"+ "adcs %[z3], %[z3], %[l3]\n\t"+ "umulh %[h3], %[x3], %[b]\n\t"+ "adc %[h3], %[h3], xzr\n\t"+ /* and the high halves, which is where this block's own carry+ * ends up */+ "mul %[l0], %[x0], %[b]\n\t"+ "adds %[z0], %[z0], %[l0]\n\t"+ "umulh %[h0], %[x0], %[b]\n\t"+ "adcs %[z1], %[z1], %[h0]\n\t"+ "umulh %[h1], %[x1], %[b]\n\t"+ "stp %[z0], %[z1], [%[t]], #16\n\t"+ "adcs %[z2], %[z2], %[h1]\n\t"+ "umulh %[h2], %[x2], %[b]\n\t"+ "adcs %[z3], %[z3], %[h2]\n\t"+ "stp %[z2], %[z3], [%[t]], #16\n\t"+ "adc %[c], %[h3], xzr\n\t"+ "subs %[k], %[k], #1\n\t"+ "b.ne 1b\n\t"+ : [a] "+r"(a), [t] "+r"(t), [c] "+r"(carry), [k] "+r"(blocks),+ [x0] "=&r"(x0), [x1] "=&r"(x1), [x2] "=&r"(x2),+ [x3] "=&r"(x3), [z0] "=&r"(z0), [z1] "=&r"(z1),+ [z2] "=&r"(z2), [z3] "=&r"(z3), [l0] "=&r"(l0),+ [l1] "=&r"(l1), [l2] "=&r"(l2), [l3] "=&r"(l3),+ [h0] "=&r"(h0), [h1] "=&r"(h1), [h2] "=&r"(h2),+ [h3] "=&r"(h3)+ : [b] "r"(b)+ : "cc", "memory");+ }+ /* What is left of the row: three limbs, two, or one, each the same two+ * chains cut short. This is not a rare case to be handed back to C --+ * mont_sqr asks for every length from n-1 down to 1, so three rows in+ * four end ragged. Only 4.7% of the limbs in a 1024-bit exponentiation+ * arrive here, but they were the dearer ones, and writing them out is+ * worth the last two per cent in the table above. The three lengths+ * are spelled out rather than run as 2+1, because chaining two short+ * blocks makes the second wait on the first: that costs two thirds of+ * the gain.+ */+ if (left == 3) {+ __asm__ volatile(+ "ldp %[x0], %[x1], [%[a]]\n\t"+ "ldr %[x2], [%[a], #16]\n\t"+ "add %[a], %[a], #24\n\t"+ "ldp %[z0], %[z1], [%[t]]\n\t"+ "ldr %[z2], [%[t], #16]\n\t"+ "adds %[z0], %[z0], %[c]\n\t"+ "mul %[l1], %[x1], %[b]\n\t"+ "adcs %[z1], %[z1], %[l1]\n\t"+ "mul %[l2], %[x2], %[b]\n\t"+ "adcs %[z2], %[z2], %[l2]\n\t"+ "umulh %[h2], %[x2], %[b]\n\t"+ "adc %[h2], %[h2], xzr\n\t"+ "mul %[l0], %[x0], %[b]\n\t"+ "adds %[z0], %[z0], %[l0]\n\t"+ "umulh %[h0], %[x0], %[b]\n\t"+ "adcs %[z1], %[z1], %[h0]\n\t"+ "umulh %[h1], %[x1], %[b]\n\t"+ "stp %[z0], %[z1], [%[t]], #16\n\t"+ "adcs %[z2], %[z2], %[h1]\n\t"+ "str %[z2], [%[t]], #8\n\t"+ "adc %[c], %[h2], xzr\n\t"+ : [a] "+r"(a), [t] "+r"(t), [c] "+r"(carry),+ [x0] "=&r"(x0), [x1] "=&r"(x1), [x2] "=&r"(x2),+ [z0] "=&r"(z0), [z1] "=&r"(z1), [z2] "=&r"(z2),+ [l0] "=&r"(l0), [l1] "=&r"(l1), [l2] "=&r"(l2),+ [h0] "=&r"(h0), [h1] "=&r"(h1), [h2] "=&r"(h2)+ : [b] "r"(b)+ : "cc", "memory");+ } else if (left == 2) {+ __asm__ volatile(+ "ldp %[x0], %[x1], [%[a]], #16\n\t"+ "ldp %[z0], %[z1], [%[t]]\n\t"+ "adds %[z0], %[z0], %[c]\n\t"+ "mul %[l1], %[x1], %[b]\n\t"+ "adcs %[z1], %[z1], %[l1]\n\t"+ "umulh %[h1], %[x1], %[b]\n\t"+ "adc %[h1], %[h1], xzr\n\t"+ "mul %[l0], %[x0], %[b]\n\t"+ "adds %[z0], %[z0], %[l0]\n\t"+ "umulh %[h0], %[x0], %[b]\n\t"+ "adcs %[z1], %[z1], %[h0]\n\t"+ "stp %[z0], %[z1], [%[t]], #16\n\t"+ "adc %[c], %[h1], xzr\n\t"+ : [a] "+r"(a), [t] "+r"(t), [c] "+r"(carry),+ [x0] "=&r"(x0), [x1] "=&r"(x1), [z0] "=&r"(z0),+ [z1] "=&r"(z1), [l0] "=&r"(l0), [l1] "=&r"(l1),+ [h0] "=&r"(h0), [h1] "=&r"(h1)+ : [b] "r"(b)+ : "cc", "memory");+ } else if (left == 1) {+ __asm__ volatile(+ "ldr %[x0], [%[a]], #8\n\t"+ "ldr %[z0], [%[t]]\n\t"+ "mul %[l0], %[x0], %[b]\n\t"+ "adds %[z0], %[z0], %[c]\n\t"+ "umulh %[h0], %[x0], %[b]\n\t"+ "adc %[h0], %[h0], xzr\n\t"+ "adds %[z0], %[z0], %[l0]\n\t"+ "str %[z0], [%[t]], #8\n\t"+ "adc %[c], %[h0], xzr\n\t"+ : [a] "+r"(a), [t] "+r"(t), [c] "+r"(carry),+ [x0] "=&r"(x0), [z0] "=&r"(z0), [l0] "=&r"(l0),+ [h0] "=&r"(h0)+ : [b] "r"(b)+ : "cc", "memory");+ }+ return carry;+}+#else+/* one limb of it, so that the loops below can say how many they take at a+ * time without saying the rest of it four times over */+#define ADDMUL_STEP(k) \+ p = (dlimb_t) a[i + (k)] * b + t[i + (k)] + carry; \+ t[i + (k)] = (limb_t) p; \+ carry = (limb_t) (p >> LIMB_BITS);++static inline limb_t addmul_1(limb_t *t, const limb_t *a, uint32_t n, limb_t b)+{+ limb_t carry = 0;+ uint32_t i = 0;+ dlimb_t p;++ for (; i + 8 <= n; i += 8) {+ ADDMUL_STEP(0) ADDMUL_STEP(1) ADDMUL_STEP(2) ADDMUL_STEP(3)+ ADDMUL_STEP(4) ADDMUL_STEP(5) ADDMUL_STEP(6) ADDMUL_STEP(7)+ }+ for (; i + 4 <= n; i += 4) {+ ADDMUL_STEP(0) ADDMUL_STEP(1) ADDMUL_STEP(2) ADDMUL_STEP(3)+ }+ for (; i + 2 <= n; i += 2) {+ ADDMUL_STEP(0) ADDMUL_STEP(1)+ }+ for (; i < n; i++) {+ ADDMUL_STEP(0)+ }+ return carry;+}+#endif++/* r = t * R^-1 mod m, with t of 2n limbs and destroyed on the way */+static inline void mont_reduce(limb_t *r, limb_t *t, const limb_t *m, limb_t n0,+ uint32_t n)+{+ limb_t borrow, take, carry = 0;+ uint32_t i;++ for (i = 0; i < n; i++) {+ limb_t u = t[i] * n0;+ limb_t c = addmul_1(t + i, m, n, u);+ dlimb_t s = (dlimb_t) t[n + i] + c + carry;++ t[n + i] = (limb_t) s;+ carry = (limb_t) (s >> LIMB_BITS);+ }++ /* what is left is under 2m, so at most one subtraction; which of the two+ * to keep is a mask */+ borrow = sub_n(r, t + n, m, n);+ take = carry | (borrow ^ 1);+ select_n(r, r, t + n, take & 1, n);+}++/* t = a * b, the low n limbs, returning the limb above them: addmul_1 with+ * nothing to add to, for the row of a product that lands on empty space. */+static inline limb_t mul_1(limb_t *t, const limb_t *a, uint32_t n, limb_t b)+{+ limb_t carry = 0;+ uint32_t i;++ for (i = 0; i < n; i++) {+ dlimb_t p = (dlimb_t) a[i] * b + carry;++ t[i] = (limb_t) p;+ carry = (limb_t) (p >> LIMB_BITS);+ }+ return carry;+}++/* r = a * b * R^-1 mod m, with t of 2n limbs */+static inline void mont_mul(limb_t *r, const limb_t *a, const limb_t *b,+ const limb_t *m, limb_t n0, uint32_t n, limb_t *t)+{+ uint32_t i;++ /* Nothing has to be cleared first. The first row lands on empty space+ * and writes t[0 .. n], and every row after it reads t[i .. i+n-1],+ * whose top limb is the one the row before it wrote. */+ t[n] = mul_1(t, a, n, b[0]);+ for (i = 1; i < n; i++)+ t[n + i] = addmul_1(t + i, a, n, b[i]);+ mont_reduce(r, t, m, n0, n);+}++/* r = a * a * R^-1 mod m, with t of 2n limbs. A square is its own mirror+ * image, so each product off the diagonal is worth two and only half of them+ * are worked out: their sum is doubled, and then the diagonal is added in. */+static inline void mont_sqr(limb_t *r, const limb_t *a, const limb_t *m, limb_t n0,+ uint32_t n, limb_t *t)+{+ limb_t carry = 0;+ uint32_t i;++ /* The first row lands on empty space here too, on t[1 .. n], and the+ * rows between them write t[1 .. 2n-2] before any of it is read. The+ * diagonal below is the only reader of the two ends. */+ t[0] = 0;+ t[2 * n - 1] = 0;+ if (n > 1)+ t[n] = mul_1(t + 1, a + 1, n - 1, a[0]);+ for (i = 1; i + 1 < n; i++)+ t[n + i] = addmul_1(t + i + i + 1, a + i + 1, n - 1 - i, a[i]);+ shl1(t, 2 * n); /* their sum is under half of what 2n limbs hold */+ for (i = 0; i < n; i++) {+ dlimb_t p = (dlimb_t) a[i] * a[i] + t[i + i] + carry;++ t[i + i] = (limb_t) p;+ p = (dlimb_t) t[i + i + 1] + (limb_t) (p >> LIMB_BITS);+ t[i + i + 1] = (limb_t) p;+ carry = (limb_t) (p >> LIMB_BITS);+ }+ mont_reduce(r, t, m, n0, n);+}++/* a = 2a mod m, for an a already under m */+static inline void dbl_mod(limb_t *a, const limb_t *m, uint32_t n, limb_t *tmp)+{+ limb_t carry = shl1(a, n);+ limb_t borrow = sub_n(tmp, a, m, n);++ select_n(a, tmp, a, (carry | (borrow ^ 1)) & 1, n);+}++/* r2 = R^2 mod m, where R is 2^(n * LIMB_BITS)+ *+ * Doubling the whole way there is 2n * LIMB_BITS steps, and at RSA sizes+ * that is a tenth of the exponentiation it is setting up for. Only the+ * first half of it has to be done a bit at a time.+ *+ * Write a value as 2^(lgR + d) mod m. A Montgomery squaring divides by R,+ * so it takes that to 2(lgR + d) - lgR = lgR + 2d: it doubles d. So double+ * up to R mod m, where d is zero, take one more step to make d one, and then+ * climb to d = lgR by the binary expansion of lgR -- a squaring for each bit+ * and one more doubling where the bit is set. For a 1024-bit modulus that+ * is ten squarings in place of a thousand and twenty-five doublings, and on+ * an Apple M4 that is 2.1 microseconds against 24.7.+ *+ * Doubling starts at the highest power of two under the modulus rather than+ * at one, since everything below that power is where doubling would go+ * anyway: for a modulus that fills its limbs, that first half is one step.+ *+ * What the trip counts depend on is the modulus' length, which is what the+ * doubling loop showed as well, and nothing else about it.+ */+static inline void mont_r2(limb_t *r2, const limb_t *m, limb_t n0, uint32_t n,+ limb_t *t)+{+ uint32_t lgr = n * LIMB_BITS;+ uint32_t i, k = 0, msb = 0;++ for (i = n; i > 0 && k == 0; i--)+ if (m[i - 1] != 0) {+ limb_t top = m[i - 1];++ k = (i - 1) * LIMB_BITS;+ while (top != 0) {+ k++;+ top >>= 1;+ }+ }+ memset(r2, 0, n * sizeof(limb_t));+ if (k == 0)+ return; /* a modulus of nothing, which the caller rules out */+ r2[(k - 1) / LIMB_BITS] = (limb_t) 1 << ((k - 1) % LIMB_BITS);+ for (i = k - 1; i < lgr; i++)+ dbl_mod(r2, m, n, t);+ /* r2 is R mod m, so d is zero and squaring would leave it there */+ dbl_mod(r2, m, n, t);+ while ((lgr >> (msb + 1)) != 0)+ msb++;+ for (i = msb; i > 0; i--) {+ mont_sqr(r2, r2, m, n0, n, t);+ if ((lgr >> (i - 1)) & 1)+ dbl_mod(r2, m, n, t);+ }+}++/* big-endian bytes into limbs, least significant limb first; anything above+ * n limbs has to be zero, which is what the contract on the base asks for */+static inline int from_be(limb_t *r, uint32_t n, const uint8_t *src, uint32_t len)+{+ uint32_t i;++ memset(r, 0, n * sizeof(limb_t));+ for (i = 0; i < len; i++) {+ uint8_t byte = src[len - 1 - i];++ if (i / LIMB_BYTES >= n) {+ if (byte != 0)+ return 1;+ continue;+ }+ r[i / LIMB_BYTES] |= (limb_t) byte << (8 * (i % LIMB_BYTES));+ }+ return 0;+}++static inline void to_be(uint8_t *dst, uint32_t len, const limb_t *a, uint32_t n)+{+ uint32_t i;++ for (i = 0; i < len; i++) {+ uint32_t pos = len - 1 - i;+ uint32_t li = i / LIMB_BYTES;++ dst[pos] = li < n ? (uint8_t) (a[li] >> (8 * (i % LIMB_BYTES))) : 0;+ }+}++#endif
@@ -0,0 +1,456 @@+/*+ * Blowfish, and the key setup bcrypt wraps around it.+ *+ * The cipher is the plain one: sixteen Feistel rounds over a schedule of+ * eighteen P words and four S boxes of 256, all of which the key is stirred+ * into. What makes bcrypt out of it is doing that stirring twice for every+ * count the cost asks for, with the salt in the mix, so that the work is+ * whatever the cost says and cannot be skipped.+ *+ * None of this is constant time, and it is not meant to be: what it is given+ * is a password, and what it leaks by timing is how long the password is,+ * which the format says out loud anyway. What matters here is that a round+ * costs what it costs, since that is the whole point of the cost parameter.+ */+#include <stdint.h>+#include <string.h>+#include <crypton_blowfish.h>++/* The P array and the four S boxes, which are the digits of pi. */+static const uint32_t initial_p[18] = {+ 0x243f6a88U, 0x85a308d3U, 0x13198a2eU, 0x03707344U, 0xa4093822U, 0x299f31d0U,+ 0x082efa98U, 0xec4e6c89U, 0x452821e6U, 0x38d01377U, 0xbe5466cfU, 0x34e90c6cU,+ 0xc0ac29b7U, 0xc97c50ddU, 0x3f84d5b5U, 0xb5470917U, 0x9216d5d9U, 0x8979fb1bU+};++static const uint32_t initial_s[4][256] = {+ {+ 0xd1310ba6U, 0x98dfb5acU, 0x2ffd72dbU, 0xd01adfb7U, 0xb8e1afedU, 0x6a267e96U,+ 0xba7c9045U, 0xf12c7f99U, 0x24a19947U, 0xb3916cf7U, 0x0801f2e2U, 0x858efc16U,+ 0x636920d8U, 0x71574e69U, 0xa458fea3U, 0xf4933d7eU, 0x0d95748fU, 0x728eb658U,+ 0x718bcd58U, 0x82154aeeU, 0x7b54a41dU, 0xc25a59b5U, 0x9c30d539U, 0x2af26013U,+ 0xc5d1b023U, 0x286085f0U, 0xca417918U, 0xb8db38efU, 0x8e79dcb0U, 0x603a180eU,+ 0x6c9e0e8bU, 0xb01e8a3eU, 0xd71577c1U, 0xbd314b27U, 0x78af2fdaU, 0x55605c60U,+ 0xe65525f3U, 0xaa55ab94U, 0x57489862U, 0x63e81440U, 0x55ca396aU, 0x2aab10b6U,+ 0xb4cc5c34U, 0x1141e8ceU, 0xa15486afU, 0x7c72e993U, 0xb3ee1411U, 0x636fbc2aU,+ 0x2ba9c55dU, 0x741831f6U, 0xce5c3e16U, 0x9b87931eU, 0xafd6ba33U, 0x6c24cf5cU,+ 0x7a325381U, 0x28958677U, 0x3b8f4898U, 0x6b4bb9afU, 0xc4bfe81bU, 0x66282193U,+ 0x61d809ccU, 0xfb21a991U, 0x487cac60U, 0x5dec8032U, 0xef845d5dU, 0xe98575b1U,+ 0xdc262302U, 0xeb651b88U, 0x23893e81U, 0xd396acc5U, 0x0f6d6ff3U, 0x83f44239U,+ 0x2e0b4482U, 0xa4842004U, 0x69c8f04aU, 0x9e1f9b5eU, 0x21c66842U, 0xf6e96c9aU,+ 0x670c9c61U, 0xabd388f0U, 0x6a51a0d2U, 0xd8542f68U, 0x960fa728U, 0xab5133a3U,+ 0x6eef0b6cU, 0x137a3be4U, 0xba3bf050U, 0x7efb2a98U, 0xa1f1651dU, 0x39af0176U,+ 0x66ca593eU, 0x82430e88U, 0x8cee8619U, 0x456f9fb4U, 0x7d84a5c3U, 0x3b8b5ebeU,+ 0xe06f75d8U, 0x85c12073U, 0x401a449fU, 0x56c16aa6U, 0x4ed3aa62U, 0x363f7706U,+ 0x1bfedf72U, 0x429b023dU, 0x37d0d724U, 0xd00a1248U, 0xdb0fead3U, 0x49f1c09bU,+ 0x075372c9U, 0x80991b7bU, 0x25d479d8U, 0xf6e8def7U, 0xe3fe501aU, 0xb6794c3bU,+ 0x976ce0bdU, 0x04c006baU, 0xc1a94fb6U, 0x409f60c4U, 0x5e5c9ec2U, 0x196a2463U,+ 0x68fb6fafU, 0x3e6c53b5U, 0x1339b2ebU, 0x3b52ec6fU, 0x6dfc511fU, 0x9b30952cU,+ 0xcc814544U, 0xaf5ebd09U, 0xbee3d004U, 0xde334afdU, 0x660f2807U, 0x192e4bb3U,+ 0xc0cba857U, 0x45c8740fU, 0xd20b5f39U, 0xb9d3fbdbU, 0x5579c0bdU, 0x1a60320aU,+ 0xd6a100c6U, 0x402c7279U, 0x679f25feU, 0xfb1fa3ccU, 0x8ea5e9f8U, 0xdb3222f8U,+ 0x3c7516dfU, 0xfd616b15U, 0x2f501ec8U, 0xad0552abU, 0x323db5faU, 0xfd238760U,+ 0x53317b48U, 0x3e00df82U, 0x9e5c57bbU, 0xca6f8ca0U, 0x1a87562eU, 0xdf1769dbU,+ 0xd542a8f6U, 0x287effc3U, 0xac6732c6U, 0x8c4f5573U, 0x695b27b0U, 0xbbca58c8U,+ 0xe1ffa35dU, 0xb8f011a0U, 0x10fa3d98U, 0xfd2183b8U, 0x4afcb56cU, 0x2dd1d35bU,+ 0x9a53e479U, 0xb6f84565U, 0xd28e49bcU, 0x4bfb9790U, 0xe1ddf2daU, 0xa4cb7e33U,+ 0x62fb1341U, 0xcee4c6e8U, 0xef20cadaU, 0x36774c01U, 0xd07e9efeU, 0x2bf11fb4U,+ 0x95dbda4dU, 0xae909198U, 0xeaad8e71U, 0x6b93d5a0U, 0xd08ed1d0U, 0xafc725e0U,+ 0x8e3c5b2fU, 0x8e7594b7U, 0x8ff6e2fbU, 0xf2122b64U, 0x8888b812U, 0x900df01cU,+ 0x4fad5ea0U, 0x688fc31cU, 0xd1cff191U, 0xb3a8c1adU, 0x2f2f2218U, 0xbe0e1777U,+ 0xea752dfeU, 0x8b021fa1U, 0xe5a0cc0fU, 0xb56f74e8U, 0x18acf3d6U, 0xce89e299U,+ 0xb4a84fe0U, 0xfd13e0b7U, 0x7cc43b81U, 0xd2ada8d9U, 0x165fa266U, 0x80957705U,+ 0x93cc7314U, 0x211a1477U, 0xe6ad2065U, 0x77b5fa86U, 0xc75442f5U, 0xfb9d35cfU,+ 0xebcdaf0cU, 0x7b3e89a0U, 0xd6411bd3U, 0xae1e7e49U, 0x00250e2dU, 0x2071b35eU,+ 0x226800bbU, 0x57b8e0afU, 0x2464369bU, 0xf009b91eU, 0x5563911dU, 0x59dfa6aaU,+ 0x78c14389U, 0xd95a537fU, 0x207d5ba2U, 0x02e5b9c5U, 0x83260376U, 0x6295cfa9U,+ 0x11c81968U, 0x4e734a41U, 0xb3472dcaU, 0x7b14a94aU, 0x1b510052U, 0x9a532915U,+ 0xd60f573fU, 0xbc9bc6e4U, 0x2b60a476U, 0x81e67400U, 0x08ba6fb5U, 0x571be91fU,+ 0xf296ec6bU, 0x2a0dd915U, 0xb6636521U, 0xe7b9f9b6U, 0xff34052eU, 0xc5855664U,+ 0x53b02d5dU, 0xa99f8fa1U, 0x08ba4799U, 0x6e85076aU+ },+ {+ 0x4b7a70e9U, 0xb5b32944U, 0xdb75092eU, 0xc4192623U, 0xad6ea6b0U, 0x49a7df7dU,+ 0x9cee60b8U, 0x8fedb266U, 0xecaa8c71U, 0x699a17ffU, 0x5664526cU, 0xc2b19ee1U,+ 0x193602a5U, 0x75094c29U, 0xa0591340U, 0xe4183a3eU, 0x3f54989aU, 0x5b429d65U,+ 0x6b8fe4d6U, 0x99f73fd6U, 0xa1d29c07U, 0xefe830f5U, 0x4d2d38e6U, 0xf0255dc1U,+ 0x4cdd2086U, 0x8470eb26U, 0x6382e9c6U, 0x021ecc5eU, 0x09686b3fU, 0x3ebaefc9U,+ 0x3c971814U, 0x6b6a70a1U, 0x687f3584U, 0x52a0e286U, 0xb79c5305U, 0xaa500737U,+ 0x3e07841cU, 0x7fdeae5cU, 0x8e7d44ecU, 0x5716f2b8U, 0xb03ada37U, 0xf0500c0dU,+ 0xf01c1f04U, 0x0200b3ffU, 0xae0cf51aU, 0x3cb574b2U, 0x25837a58U, 0xdc0921bdU,+ 0xd19113f9U, 0x7ca92ff6U, 0x94324773U, 0x22f54701U, 0x3ae5e581U, 0x37c2dadcU,+ 0xc8b57634U, 0x9af3dda7U, 0xa9446146U, 0x0fd0030eU, 0xecc8c73eU, 0xa4751e41U,+ 0xe238cd99U, 0x3bea0e2fU, 0x3280bba1U, 0x183eb331U, 0x4e548b38U, 0x4f6db908U,+ 0x6f420d03U, 0xf60a04bfU, 0x2cb81290U, 0x24977c79U, 0x5679b072U, 0xbcaf89afU,+ 0xde9a771fU, 0xd9930810U, 0xb38bae12U, 0xdccf3f2eU, 0x5512721fU, 0x2e6b7124U,+ 0x501adde6U, 0x9f84cd87U, 0x7a584718U, 0x7408da17U, 0xbc9f9abcU, 0xe94b7d8cU,+ 0xec7aec3aU, 0xdb851dfaU, 0x63094366U, 0xc464c3d2U, 0xef1c1847U, 0x3215d908U,+ 0xdd433b37U, 0x24c2ba16U, 0x12a14d43U, 0x2a65c451U, 0x50940002U, 0x133ae4ddU,+ 0x71dff89eU, 0x10314e55U, 0x81ac77d6U, 0x5f11199bU, 0x043556f1U, 0xd7a3c76bU,+ 0x3c11183bU, 0x5924a509U, 0xf28fe6edU, 0x97f1fbfaU, 0x9ebabf2cU, 0x1e153c6eU,+ 0x86e34570U, 0xeae96fb1U, 0x860e5e0aU, 0x5a3e2ab3U, 0x771fe71cU, 0x4e3d06faU,+ 0x2965dcb9U, 0x99e71d0fU, 0x803e89d6U, 0x5266c825U, 0x2e4cc978U, 0x9c10b36aU,+ 0xc6150ebaU, 0x94e2ea78U, 0xa5fc3c53U, 0x1e0a2df4U, 0xf2f74ea7U, 0x361d2b3dU,+ 0x1939260fU, 0x19c27960U, 0x5223a708U, 0xf71312b6U, 0xebadfe6eU, 0xeac31f66U,+ 0xe3bc4595U, 0xa67bc883U, 0xb17f37d1U, 0x018cff28U, 0xc332ddefU, 0xbe6c5aa5U,+ 0x65582185U, 0x68ab9802U, 0xeecea50fU, 0xdb2f953bU, 0x2aef7dadU, 0x5b6e2f84U,+ 0x1521b628U, 0x29076170U, 0xecdd4775U, 0x619f1510U, 0x13cca830U, 0xeb61bd96U,+ 0x0334fe1eU, 0xaa0363cfU, 0xb5735c90U, 0x4c70a239U, 0xd59e9e0bU, 0xcbaade14U,+ 0xeecc86bcU, 0x60622ca7U, 0x9cab5cabU, 0xb2f3846eU, 0x648b1eafU, 0x19bdf0caU,+ 0xa02369b9U, 0x655abb50U, 0x40685a32U, 0x3c2ab4b3U, 0x319ee9d5U, 0xc021b8f7U,+ 0x9b540b19U, 0x875fa099U, 0x95f7997eU, 0x623d7da8U, 0xf837889aU, 0x97e32d77U,+ 0x11ed935fU, 0x16681281U, 0x0e358829U, 0xc7e61fd6U, 0x96dedfa1U, 0x7858ba99U,+ 0x57f584a5U, 0x1b227263U, 0x9b83c3ffU, 0x1ac24696U, 0xcdb30aebU, 0x532e3054U,+ 0x8fd948e4U, 0x6dbc3128U, 0x58ebf2efU, 0x34c6ffeaU, 0xfe28ed61U, 0xee7c3c73U,+ 0x5d4a14d9U, 0xe864b7e3U, 0x42105d14U, 0x203e13e0U, 0x45eee2b6U, 0xa3aaabeaU,+ 0xdb6c4f15U, 0xfacb4fd0U, 0xc742f442U, 0xef6abbb5U, 0x654f3b1dU, 0x41cd2105U,+ 0xd81e799eU, 0x86854dc7U, 0xe44b476aU, 0x3d816250U, 0xcf62a1f2U, 0x5b8d2646U,+ 0xfc8883a0U, 0xc1c7b6a3U, 0x7f1524c3U, 0x69cb7492U, 0x47848a0bU, 0x5692b285U,+ 0x095bbf00U, 0xad19489dU, 0x1462b174U, 0x23820e00U, 0x58428d2aU, 0x0c55f5eaU,+ 0x1dadf43eU, 0x233f7061U, 0x3372f092U, 0x8d937e41U, 0xd65fecf1U, 0x6c223bdbU,+ 0x7cde3759U, 0xcbee7460U, 0x4085f2a7U, 0xce77326eU, 0xa6078084U, 0x19f8509eU,+ 0xe8efd855U, 0x61d99735U, 0xa969a7aaU, 0xc50c06c2U, 0x5a04abfcU, 0x800bcadcU,+ 0x9e447a2eU, 0xc3453484U, 0xfdd56705U, 0x0e1e9ec9U, 0xdb73dbd3U, 0x105588cdU,+ 0x675fda79U, 0xe3674340U, 0xc5c43465U, 0x713e38d8U, 0x3d28f89eU, 0xf16dff20U,+ 0x153e21e7U, 0x8fb03d4aU, 0xe6e39f2bU, 0xdb83adf7U+ },+ {+ 0xe93d5a68U, 0x948140f7U, 0xf64c261cU, 0x94692934U, 0x411520f7U, 0x7602d4f7U,+ 0xbcf46b2eU, 0xd4a20068U, 0xd4082471U, 0x3320f46aU, 0x43b7d4b7U, 0x500061afU,+ 0x1e39f62eU, 0x97244546U, 0x14214f74U, 0xbf8b8840U, 0x4d95fc1dU, 0x96b591afU,+ 0x70f4ddd3U, 0x66a02f45U, 0xbfbc09ecU, 0x03bd9785U, 0x7fac6dd0U, 0x31cb8504U,+ 0x96eb27b3U, 0x55fd3941U, 0xda2547e6U, 0xabca0a9aU, 0x28507825U, 0x530429f4U,+ 0x0a2c86daU, 0xe9b66dfbU, 0x68dc1462U, 0xd7486900U, 0x680ec0a4U, 0x27a18deeU,+ 0x4f3ffea2U, 0xe887ad8cU, 0xb58ce006U, 0x7af4d6b6U, 0xaace1e7cU, 0xd3375fecU,+ 0xce78a399U, 0x406b2a42U, 0x20fe9e35U, 0xd9f385b9U, 0xee39d7abU, 0x3b124e8bU,+ 0x1dc9faf7U, 0x4b6d1856U, 0x26a36631U, 0xeae397b2U, 0x3a6efa74U, 0xdd5b4332U,+ 0x6841e7f7U, 0xca7820fbU, 0xfb0af54eU, 0xd8feb397U, 0x454056acU, 0xba489527U,+ 0x55533a3aU, 0x20838d87U, 0xfe6ba9b7U, 0xd096954bU, 0x55a867bcU, 0xa1159a58U,+ 0xcca92963U, 0x99e1db33U, 0xa62a4a56U, 0x3f3125f9U, 0x5ef47e1cU, 0x9029317cU,+ 0xfdf8e802U, 0x04272f70U, 0x80bb155cU, 0x05282ce3U, 0x95c11548U, 0xe4c66d22U,+ 0x48c1133fU, 0xc70f86dcU, 0x07f9c9eeU, 0x41041f0fU, 0x404779a4U, 0x5d886e17U,+ 0x325f51ebU, 0xd59bc0d1U, 0xf2bcc18fU, 0x41113564U, 0x257b7834U, 0x602a9c60U,+ 0xdff8e8a3U, 0x1f636c1bU, 0x0e12b4c2U, 0x02e1329eU, 0xaf664fd1U, 0xcad18115U,+ 0x6b2395e0U, 0x333e92e1U, 0x3b240b62U, 0xeebeb922U, 0x85b2a20eU, 0xe6ba0d99U,+ 0xde720c8cU, 0x2da2f728U, 0xd0127845U, 0x95b794fdU, 0x647d0862U, 0xe7ccf5f0U,+ 0x5449a36fU, 0x877d48faU, 0xc39dfd27U, 0xf33e8d1eU, 0x0a476341U, 0x992eff74U,+ 0x3a6f6eabU, 0xf4f8fd37U, 0xa812dc60U, 0xa1ebddf8U, 0x991be14cU, 0xdb6e6b0dU,+ 0xc67b5510U, 0x6d672c37U, 0x2765d43bU, 0xdcd0e804U, 0xf1290dc7U, 0xcc00ffa3U,+ 0xb5390f92U, 0x690fed0bU, 0x667b9ffbU, 0xcedb7d9cU, 0xa091cf0bU, 0xd9155ea3U,+ 0xbb132f88U, 0x515bad24U, 0x7b9479bfU, 0x763bd6ebU, 0x37392eb3U, 0xcc115979U,+ 0x8026e297U, 0xf42e312dU, 0x6842ada7U, 0xc66a2b3bU, 0x12754cccU, 0x782ef11cU,+ 0x6a124237U, 0xb79251e7U, 0x06a1bbe6U, 0x4bfb6350U, 0x1a6b1018U, 0x11caedfaU,+ 0x3d25bdd8U, 0xe2e1c3c9U, 0x44421659U, 0x0a121386U, 0xd90cec6eU, 0xd5abea2aU,+ 0x64af674eU, 0xda86a85fU, 0xbebfe988U, 0x64e4c3feU, 0x9dbc8057U, 0xf0f7c086U,+ 0x60787bf8U, 0x6003604dU, 0xd1fd8346U, 0xf6381fb0U, 0x7745ae04U, 0xd736fcccU,+ 0x83426b33U, 0xf01eab71U, 0xb0804187U, 0x3c005e5fU, 0x77a057beU, 0xbde8ae24U,+ 0x55464299U, 0xbf582e61U, 0x4e58f48fU, 0xf2ddfda2U, 0xf474ef38U, 0x8789bdc2U,+ 0x5366f9c3U, 0xc8b38e74U, 0xb475f255U, 0x46fcd9b9U, 0x7aeb2661U, 0x8b1ddf84U,+ 0x846a0e79U, 0x915f95e2U, 0x466e598eU, 0x20b45770U, 0x8cd55591U, 0xc902de4cU,+ 0xb90bace1U, 0xbb8205d0U, 0x11a86248U, 0x7574a99eU, 0xb77f19b6U, 0xe0a9dc09U,+ 0x662d09a1U, 0xc4324633U, 0xe85a1f02U, 0x09f0be8cU, 0x4a99a025U, 0x1d6efe10U,+ 0x1ab93d1dU, 0x0ba5a4dfU, 0xa186f20fU, 0x2868f169U, 0xdcb7da83U, 0x573906feU,+ 0xa1e2ce9bU, 0x4fcd7f52U, 0x50115e01U, 0xa70683faU, 0xa002b5c4U, 0x0de6d027U,+ 0x9af88c27U, 0x773f8641U, 0xc3604c06U, 0x61a806b5U, 0xf0177a28U, 0xc0f586e0U,+ 0x006058aaU, 0x30dc7d62U, 0x11e69ed7U, 0x2338ea63U, 0x53c2dd94U, 0xc2c21634U,+ 0xbbcbee56U, 0x90bcb6deU, 0xebfc7da1U, 0xce591d76U, 0x6f05e409U, 0x4b7c0188U,+ 0x39720a3dU, 0x7c927c24U, 0x86e3725fU, 0x724d9db9U, 0x1ac15bb4U, 0xd39eb8fcU,+ 0xed545578U, 0x08fca5b5U, 0xd83d7cd3U, 0x4dad0fc4U, 0x1e50ef5eU, 0xb161e6f8U,+ 0xa28514d9U, 0x6c51133cU, 0x6fd5c7e7U, 0x56e14ec4U, 0x362abfceU, 0xddc6c837U,+ 0xd79a3234U, 0x92638212U, 0x670efa8eU, 0x406000e0U+ },+ {+ 0x3a39ce37U, 0xd3faf5cfU, 0xabc27737U, 0x5ac52d1bU, 0x5cb0679eU, 0x4fa33742U,+ 0xd3822740U, 0x99bc9bbeU, 0xd5118e9dU, 0xbf0f7315U, 0xd62d1c7eU, 0xc700c47bU,+ 0xb78c1b6bU, 0x21a19045U, 0xb26eb1beU, 0x6a366eb4U, 0x5748ab2fU, 0xbc946e79U,+ 0xc6a376d2U, 0x6549c2c8U, 0x530ff8eeU, 0x468dde7dU, 0xd5730a1dU, 0x4cd04dc6U,+ 0x2939bbdbU, 0xa9ba4650U, 0xac9526e8U, 0xbe5ee304U, 0xa1fad5f0U, 0x6a2d519aU,+ 0x63ef8ce2U, 0x9a86ee22U, 0xc089c2b8U, 0x43242ef6U, 0xa51e03aaU, 0x9cf2d0a4U,+ 0x83c061baU, 0x9be96a4dU, 0x8fe51550U, 0xba645bd6U, 0x2826a2f9U, 0xa73a3ae1U,+ 0x4ba99586U, 0xef5562e9U, 0xc72fefd3U, 0xf752f7daU, 0x3f046f69U, 0x77fa0a59U,+ 0x80e4a915U, 0x87b08601U, 0x9b09e6adU, 0x3b3ee593U, 0xe990fd5aU, 0x9e34d797U,+ 0x2cf0b7d9U, 0x022b8b51U, 0x96d5ac3aU, 0x017da67dU, 0xd1cf3ed6U, 0x7c7d2d28U,+ 0x1f9f25cfU, 0xadf2b89bU, 0x5ad6b472U, 0x5a88f54cU, 0xe029ac71U, 0xe019a5e6U,+ 0x47b0acfdU, 0xed93fa9bU, 0xe8d3c48dU, 0x283b57ccU, 0xf8d56629U, 0x79132e28U,+ 0x785f0191U, 0xed756055U, 0xf7960e44U, 0xe3d35e8cU, 0x15056dd4U, 0x88f46dbaU,+ 0x03a16125U, 0x0564f0bdU, 0xc3eb9e15U, 0x3c9057a2U, 0x97271aecU, 0xa93a072aU,+ 0x1b3f6d9bU, 0x1e6321f5U, 0xf59c66fbU, 0x26dcf319U, 0x7533d928U, 0xb155fdf5U,+ 0x03563482U, 0x8aba3cbbU, 0x28517711U, 0xc20ad9f8U, 0xabcc5167U, 0xccad925fU,+ 0x4de81751U, 0x3830dc8eU, 0x379d5862U, 0x9320f991U, 0xea7a90c2U, 0xfb3e7bceU,+ 0x5121ce64U, 0x774fbe32U, 0xa8b6e37eU, 0xc3293d46U, 0x48de5369U, 0x6413e680U,+ 0xa2ae0810U, 0xdd6db224U, 0x69852dfdU, 0x09072166U, 0xb39a460aU, 0x6445c0ddU,+ 0x586cdecfU, 0x1c20c8aeU, 0x5bbef7ddU, 0x1b588d40U, 0xccd2017fU, 0x6bb4e3bbU,+ 0xdda26a7eU, 0x3a59ff45U, 0x3e350a44U, 0xbcb4cdd5U, 0x72eacea8U, 0xfa6484bbU,+ 0x8d6612aeU, 0xbf3c6f47U, 0xd29be463U, 0x542f5d9eU, 0xaec2771bU, 0xf64e6370U,+ 0x740e0d8dU, 0xe75b1357U, 0xf8721671U, 0xaf537d5dU, 0x4040cb08U, 0x4eb4e2ccU,+ 0x34d2466aU, 0x0115af84U, 0xe1b00428U, 0x95983a1dU, 0x06b89fb4U, 0xce6ea048U,+ 0x6f3f3b82U, 0x3520ab82U, 0x011a1d4bU, 0x277227f8U, 0x611560b1U, 0xe7933fdcU,+ 0xbb3a792bU, 0x344525bdU, 0xa08839e1U, 0x51ce794bU, 0x2f32c9b7U, 0xa01fbac9U,+ 0xe01cc87eU, 0xbcc7d1f6U, 0xcf0111c3U, 0xa1e8aac7U, 0x1a908749U, 0xd44fbd9aU,+ 0xd0dadecbU, 0xd50ada38U, 0x0339c32aU, 0xc6913667U, 0x8df9317cU, 0xe0b12b4fU,+ 0xf79e59b7U, 0x43f5bb3aU, 0xf2d519ffU, 0x27d9459cU, 0xbf97222cU, 0x15e6fc2aU,+ 0x0f91fc71U, 0x9b941525U, 0xfae59361U, 0xceb69cebU, 0xc2a86459U, 0x12baa8d1U,+ 0xb6c1075eU, 0xe3056a0cU, 0x10d25065U, 0xcb03a442U, 0xe0ec6e0eU, 0x1698db3bU,+ 0x4c98a0beU, 0x3278e964U, 0x9f1f9532U, 0xe0d392dfU, 0xd3a0342bU, 0x8971f21eU,+ 0x1b0a7441U, 0x4ba3348cU, 0xc5be7120U, 0xc37632d8U, 0xdf359f8dU, 0x9b992f2eU,+ 0xe60b6f47U, 0x0fe3f11dU, 0xe54cda54U, 0x1edad891U, 0xce6279cfU, 0xcd3e7e6fU,+ 0x1618b166U, 0xfd2c1d05U, 0x848fd2c5U, 0xf6fb2299U, 0xf523f357U, 0xa6327623U,+ 0x93a83531U, 0x56cccd02U, 0xacf08162U, 0x5a75ebb5U, 0x6e163697U, 0x88d273ccU,+ 0xde966292U, 0x81b949d0U, 0x4c50901bU, 0x71c65614U, 0xe6c6c7bdU, 0x327a140aU,+ 0x45e1d006U, 0xc3f27b9aU, 0xc9aa53fdU, 0x62a80f00U, 0xbb25bfe2U, 0x35bdd2f6U,+ 0x71126905U, 0xb2040222U, 0xb6cbcf7cU, 0xcd769c2bU, 0x53113ec0U, 0x1640e3d3U,+ 0x38abbd60U, 0x2547adf0U, 0xba38209cU, 0xf746ce76U, 0x77afa1c5U, 0x20756060U,+ 0x85cbfe4eU, 0x8ae88dd8U, 0x7aaaf9b0U, 0x4cf9aa7eU, 0x1948c25cU, 0x02fb8a8cU,+ 0x01c36ae4U, 0xd6ebe1f9U, 0x90d4f869U, 0xa65cdea0U, 0x3f09252dU, 0xc208e69fU,+ 0xb74e6132U, 0xce77e25bU, 0x578fdfe3U, 0x3ac372e6U+ }+};++#define F(ctx, x) \+ ((((ctx)->s[0][(x) >> 24] + (ctx)->s[1][((x) >> 16) & 0xff]) \+ ^ (ctx)->s[2][((x) >> 8) & 0xff]) \+ + (ctx)->s[3][(x) & 0xff])++static void block_encrypt(const crypton_blowfish_ctx *ctx, uint32_t *xl,+ uint32_t *xr)+{+ uint32_t l = *xl, r = *xr;+ int i;++ for (i = 0; i < 16; i += 2) {+ l ^= ctx->p[i];+ r ^= F(ctx, l);+ r ^= ctx->p[i + 1];+ l ^= F(ctx, r);+ }+ l ^= ctx->p[16];+ r ^= ctx->p[17];+ *xl = r;+ *xr = l;+}++static void block_decrypt(const crypton_blowfish_ctx *ctx, uint32_t *xl,+ uint32_t *xr)+{+ uint32_t l = *xl, r = *xr;+ int i;++ for (i = 16; i > 0; i -= 2) {+ l ^= ctx->p[i + 1];+ r ^= F(ctx, l);+ r ^= ctx->p[i];+ l ^= F(ctx, r);+ }+ l ^= ctx->p[1];+ r ^= ctx->p[0];+ *xl = r;+ *xr = l;+}++/* the next four bytes of the key, taken round and round */+static uint32_t key_word(const uint8_t *key, uint32_t keylen, uint32_t *pos)+{+ uint32_t w = 0, i;++ for (i = 0; i < 4; i++) {+ w = (w << 8) | key[*pos];+ *pos = (*pos + 1) % keylen;+ }+ return w;+}++/* the key into the P array, and then the whole schedule rewritten by+ * encrypting its way through itself */+static void expand_key(crypton_blowfish_ctx *ctx, const uint8_t *key,+ uint32_t keylen)+{+ uint32_t pos = 0, l = 0, r = 0;+ int i, j;++ if (keylen > 0)+ for (i = 0; i < 18; i++)+ ctx->p[i] ^= key_word(key, keylen, &pos);+ for (i = 0; i < 18; i += 2) {+ block_encrypt(ctx, &l, &r);+ ctx->p[i] = l;+ ctx->p[i + 1] = r;+ }+ for (i = 0; i < 4; i++)+ for (j = 0; j < 256; j += 2) {+ block_encrypt(ctx, &l, &r);+ ctx->s[i][j] = l;+ ctx->s[i][j + 1] = r;+ }+}++/* the same, with the salt exclusive-ored into what is encrypted at every+ * step, which is what makes the schedule depend on it. The salt is taken+ * round and round as the key is, so a salt of any length will do: bcrypt+ * hands it sixteen bytes and bcrypt_pbkdf hands it sixty-four. */+static void expand_key_with_salt(crypton_blowfish_ctx *ctx, const uint8_t *key,+ uint32_t keylen, const uint8_t *salt,+ uint32_t saltlen)+{+ uint32_t kpos = 0, spos = 0, l = 0, r = 0;+ int i, j;++ if (keylen > 0)+ for (i = 0; i < 18; i++)+ ctx->p[i] ^= key_word(key, keylen, &kpos);+ for (i = 0; i < 18; i += 2) {+ l ^= key_word(salt, saltlen, &spos);+ r ^= key_word(salt, saltlen, &spos);+ block_encrypt(ctx, &l, &r);+ ctx->p[i] = l;+ ctx->p[i + 1] = r;+ }+ for (i = 0; i < 4; i++)+ for (j = 0; j < 256; j += 2) {+ l ^= key_word(salt, saltlen, &spos);+ r ^= key_word(salt, saltlen, &spos);+ block_encrypt(ctx, &l, &r);+ ctx->s[i][j] = l;+ ctx->s[i][j + 1] = r;+ }+}++void crypton_blowfish_init(crypton_blowfish_ctx *ctx, const uint8_t *key,+ uint32_t keylen)+{+ memcpy(ctx->p, initial_p, sizeof(ctx->p));+ memcpy(ctx->s, initial_s, sizeof(ctx->s));+ expand_key(ctx, key, keylen);+}++void crypton_blowfish_encrypt(const crypton_blowfish_ctx *ctx, uint8_t *out,+ const uint8_t *in, uint32_t len)+{+ uint32_t i;++ for (i = 0; i + 8 <= len; i += 8) {+ uint32_t l = ((uint32_t) in[i] << 24) | ((uint32_t) in[i + 1] << 16)+ | ((uint32_t) in[i + 2] << 8) | (uint32_t) in[i + 3];+ uint32_t r = ((uint32_t) in[i + 4] << 24)+ | ((uint32_t) in[i + 5] << 16)+ | ((uint32_t) in[i + 6] << 8) | (uint32_t) in[i + 7];++ block_encrypt(ctx, &l, &r);+ out[i] = (uint8_t) (l >> 24);+ out[i + 1] = (uint8_t) (l >> 16);+ out[i + 2] = (uint8_t) (l >> 8);+ out[i + 3] = (uint8_t) l;+ out[i + 4] = (uint8_t) (r >> 24);+ out[i + 5] = (uint8_t) (r >> 16);+ out[i + 6] = (uint8_t) (r >> 8);+ out[i + 7] = (uint8_t) r;+ }+}++void crypton_blowfish_decrypt(const crypton_blowfish_ctx *ctx, uint8_t *out,+ const uint8_t *in, uint32_t len)+{+ uint32_t i;++ for (i = 0; i + 8 <= len; i += 8) {+ uint32_t l = ((uint32_t) in[i] << 24) | ((uint32_t) in[i + 1] << 16)+ | ((uint32_t) in[i + 2] << 8) | (uint32_t) in[i + 3];+ uint32_t r = ((uint32_t) in[i + 4] << 24)+ | ((uint32_t) in[i + 5] << 16)+ | ((uint32_t) in[i + 6] << 8) | (uint32_t) in[i + 7];++ block_decrypt(ctx, &l, &r);+ out[i] = (uint8_t) (l >> 24);+ out[i + 1] = (uint8_t) (l >> 16);+ out[i + 2] = (uint8_t) (l >> 8);+ out[i + 3] = (uint8_t) l;+ out[i + 4] = (uint8_t) (r >> 24);+ out[i + 5] = (uint8_t) (r >> 16);+ out[i + 6] = (uint8_t) (r >> 8);+ out[i + 7] = (uint8_t) r;+ }+}++int crypton_bcrypt(uint8_t out[24], uint32_t cost, const uint8_t salt[16],+ const uint8_t *key, uint32_t keylen)+{+ /* "OrpheanBeholderScryDoubt", which is what bcrypt encrypts */+ static const uint8_t magic[24] = {+ 0x4f, 0x72, 0x70, 0x68, 0x65, 0x61, 0x6e, 0x42,+ 0x65, 0x68, 0x6f, 0x6c, 0x64, 0x65, 0x72, 0x53,+ 0x63, 0x72, 0x79, 0x44, 0x6f, 0x75, 0x62, 0x74+ };+ crypton_blowfish_ctx ctx;+ uint32_t rounds, i;++ if (cost < 4 || cost > 31 || keylen == 0 || keylen > 73)+ return -1;++ memcpy(ctx.p, initial_p, sizeof(ctx.p));+ memcpy(ctx.s, initial_s, sizeof(ctx.s));+ expand_key_with_salt(&ctx, key, keylen, salt, 16);+ rounds = (uint32_t) 1 << cost;+ for (i = 0; i < rounds; i++) {+ expand_key(&ctx, key, keylen);+ expand_key(&ctx, salt, 16);+ }++ memcpy(out, magic, sizeof(magic));+ for (i = 0; i < 64; i++)+ crypton_blowfish_encrypt(&ctx, out, out, 24);++ memset(&ctx, 0, sizeof(ctx));+ return 0;+}++int crypton_bcrypt_pbkdf_hash(uint8_t out[32], const uint8_t *pass,+ uint32_t passlen, const uint8_t *salt,+ uint32_t saltlen)+{+ /* "OxychromaticBlowfishSwatDynamite", which is what this one encrypts */+ static const uint8_t magic[32] = {+ 0x4f, 0x78, 0x79, 0x63, 0x68, 0x72, 0x6f, 0x6d,+ 0x61, 0x74, 0x69, 0x63, 0x42, 0x6c, 0x6f, 0x77,+ 0x66, 0x69, 0x73, 0x68, 0x53, 0x77, 0x61, 0x74,+ 0x44, 0x79, 0x6e, 0x61, 0x6d, 0x69, 0x74, 0x65+ };+ crypton_blowfish_ctx ctx;+ uint32_t i, j;++ if (passlen == 0 || saltlen == 0)+ return -1;++ memcpy(ctx.p, initial_p, sizeof(ctx.p));+ memcpy(ctx.s, initial_s, sizeof(ctx.s));+ expand_key_with_salt(&ctx, pass, passlen, salt, saltlen);+ for (i = 0; i < 64; i++) {+ expand_key(&ctx, salt, saltlen);+ expand_key(&ctx, pass, passlen);+ }++ /* each block encrypted sixty-four times, and stored with each half the+ * way round that the original implementation stores it */+ for (i = 0; i < 4; i++) {+ uint32_t l = ((uint32_t) magic[8 * i] << 24)+ | ((uint32_t) magic[8 * i + 1] << 16)+ | ((uint32_t) magic[8 * i + 2] << 8)+ | (uint32_t) magic[8 * i + 3];+ uint32_t r = ((uint32_t) magic[8 * i + 4] << 24)+ | ((uint32_t) magic[8 * i + 5] << 16)+ | ((uint32_t) magic[8 * i + 6] << 8)+ | (uint32_t) magic[8 * i + 7];++ for (j = 0; j < 64; j++)+ block_encrypt(&ctx, &l, &r);+ out[8 * i] = (uint8_t) l;+ out[8 * i + 1] = (uint8_t) (l >> 8);+ out[8 * i + 2] = (uint8_t) (l >> 16);+ out[8 * i + 3] = (uint8_t) (l >> 24);+ out[8 * i + 4] = (uint8_t) r;+ out[8 * i + 5] = (uint8_t) (r >> 8);+ out[8 * i + 6] = (uint8_t) (r >> 16);+ out[8 * i + 7] = (uint8_t) (r >> 24);+ }++ memset(&ctx, 0, sizeof(ctx));+ return 0;+}
@@ -0,0 +1,44 @@+#ifndef CRYPTON_BLOWFISH_H+#define CRYPTON_BLOWFISH_H++#include <stdint.h>++/* The key schedule: the P array and the four S boxes, which is all the state+ * Blowfish has. The caller keeps it; nothing here allocates. */+typedef struct {+ uint32_t p[18];+ uint32_t s[4][256];+} crypton_blowfish_ctx;++/* Set a schedule up from a key of keylen bytes, which has to be 1 to 56. */+void crypton_blowfish_init(crypton_blowfish_ctx *ctx, const uint8_t *key,+ uint32_t keylen);++/* Encrypt or decrypt whole blocks: len has to be a multiple of eight, and out+ * may be in. */+void crypton_blowfish_encrypt(const crypton_blowfish_ctx *ctx, uint8_t *out,+ const uint8_t *in, uint32_t len);+void crypton_blowfish_decrypt(const crypton_blowfish_ctx *ctx, uint8_t *out,+ const uint8_t *in, uint32_t len);++/* The whole of what bcrypt does with Blowfish: the key setup that costs what+ * the cost says, and then the sixty-four encryptions. Writes 24 bytes, of+ * which bcrypt keeps 23. The salt is 16 bytes and the key is the password+ * with its terminating zero, 1 to 72 bytes of it.+ *+ * Returns 0, or -1 for a cost or a length it will not take.+ */+int crypton_bcrypt(uint8_t out[24], uint32_t cost, const uint8_t salt[16],+ const uint8_t *key, uint32_t keylen);++/* What bcrypt_pbkdf does with Blowfish: the same key setup, sixty-four times+ * over, and then the four blocks of its own magic. Writes 32 bytes. The two+ * hashes it is given are 64 bytes each in the only caller there is.+ *+ * Returns 0, or -1 for a length it will not take.+ */+int crypton_bcrypt_pbkdf_hash(uint8_t out[32], const uint8_t *pass,+ uint32_t passlen, const uint8_t *salt,+ uint32_t saltlen);++#endif
@@ -0,0 +1,27 @@+/*+ * Erasing memory that the compiler is entitled to decide nobody reads.+ *+ * memset on an object that is about to die -- freed, or a local going out of+ * scope -- is a store to memory nothing can observe, and an optimizer may+ * drop it. That is the whole reason explicit_bzero and memset_s exist.+ * Neither is everywhere, so this writes through a volatile pointer, which the+ * standard says cannot be elided.+ *+ * Use it wherever key material stops being needed. Plain memset is still+ * right for memory that is about to be read again.+ */+#ifndef CRYPTON_BZERO_H+#define CRYPTON_BZERO_H++#include <stddef.h>+#include <stdint.h>++static inline void crypton_bzero(void *p, size_t n)+{+ volatile uint8_t *q = (volatile uint8_t *)p;++ while (n--)+ *q++ = 0;+}++#endif
@@ -0,0 +1,697 @@+/*+ * Camellia with a 128-bit key, as RFC 3713 defines it.+ *+ * SP[i] is generated from that standard: the S-box byte i goes through, spread+ * into the positions the P layer sends it to. P is an exclusive-or of bytes,+ * so the round function is the exclusive-or of eight lookups.+ */+#include <stdint.h>+#include <crypton_camellia.h>++static const uint64_t SP[8][256] = {+{+ 0x7070700070000070ULL, 0x8282820082000082ULL, 0x2c2c2c002c00002cULL, 0xececec00ec0000ecULL,+ 0xb3b3b300b30000b3ULL, 0x2727270027000027ULL, 0xc0c0c000c00000c0ULL, 0xe5e5e500e50000e5ULL,+ 0xe4e4e400e40000e4ULL, 0x8585850085000085ULL, 0x5757570057000057ULL, 0x3535350035000035ULL,+ 0xeaeaea00ea0000eaULL, 0x0c0c0c000c00000cULL, 0xaeaeae00ae0000aeULL, 0x4141410041000041ULL,+ 0x2323230023000023ULL, 0xefefef00ef0000efULL, 0x6b6b6b006b00006bULL, 0x9393930093000093ULL,+ 0x4545450045000045ULL, 0x1919190019000019ULL, 0xa5a5a500a50000a5ULL, 0x2121210021000021ULL,+ 0xededed00ed0000edULL, 0x0e0e0e000e00000eULL, 0x4f4f4f004f00004fULL, 0x4e4e4e004e00004eULL,+ 0x1d1d1d001d00001dULL, 0x6565650065000065ULL, 0x9292920092000092ULL, 0xbdbdbd00bd0000bdULL,+ 0x8686860086000086ULL, 0xb8b8b800b80000b8ULL, 0xafafaf00af0000afULL, 0x8f8f8f008f00008fULL,+ 0x7c7c7c007c00007cULL, 0xebebeb00eb0000ebULL, 0x1f1f1f001f00001fULL, 0xcecece00ce0000ceULL,+ 0x3e3e3e003e00003eULL, 0x3030300030000030ULL, 0xdcdcdc00dc0000dcULL, 0x5f5f5f005f00005fULL,+ 0x5e5e5e005e00005eULL, 0xc5c5c500c50000c5ULL, 0x0b0b0b000b00000bULL, 0x1a1a1a001a00001aULL,+ 0xa6a6a600a60000a6ULL, 0xe1e1e100e10000e1ULL, 0x3939390039000039ULL, 0xcacaca00ca0000caULL,+ 0xd5d5d500d50000d5ULL, 0x4747470047000047ULL, 0x5d5d5d005d00005dULL, 0x3d3d3d003d00003dULL,+ 0xd9d9d900d90000d9ULL, 0x0101010001000001ULL, 0x5a5a5a005a00005aULL, 0xd6d6d600d60000d6ULL,+ 0x5151510051000051ULL, 0x5656560056000056ULL, 0x6c6c6c006c00006cULL, 0x4d4d4d004d00004dULL,+ 0x8b8b8b008b00008bULL, 0x0d0d0d000d00000dULL, 0x9a9a9a009a00009aULL, 0x6666660066000066ULL,+ 0xfbfbfb00fb0000fbULL, 0xcccccc00cc0000ccULL, 0xb0b0b000b00000b0ULL, 0x2d2d2d002d00002dULL,+ 0x7474740074000074ULL, 0x1212120012000012ULL, 0x2b2b2b002b00002bULL, 0x2020200020000020ULL,+ 0xf0f0f000f00000f0ULL, 0xb1b1b100b10000b1ULL, 0x8484840084000084ULL, 0x9999990099000099ULL,+ 0xdfdfdf00df0000dfULL, 0x4c4c4c004c00004cULL, 0xcbcbcb00cb0000cbULL, 0xc2c2c200c20000c2ULL,+ 0x3434340034000034ULL, 0x7e7e7e007e00007eULL, 0x7676760076000076ULL, 0x0505050005000005ULL,+ 0x6d6d6d006d00006dULL, 0xb7b7b700b70000b7ULL, 0xa9a9a900a90000a9ULL, 0x3131310031000031ULL,+ 0xd1d1d100d10000d1ULL, 0x1717170017000017ULL, 0x0404040004000004ULL, 0xd7d7d700d70000d7ULL,+ 0x1414140014000014ULL, 0x5858580058000058ULL, 0x3a3a3a003a00003aULL, 0x6161610061000061ULL,+ 0xdedede00de0000deULL, 0x1b1b1b001b00001bULL, 0x1111110011000011ULL, 0x1c1c1c001c00001cULL,+ 0x3232320032000032ULL, 0x0f0f0f000f00000fULL, 0x9c9c9c009c00009cULL, 0x1616160016000016ULL,+ 0x5353530053000053ULL, 0x1818180018000018ULL, 0xf2f2f200f20000f2ULL, 0x2222220022000022ULL,+ 0xfefefe00fe0000feULL, 0x4444440044000044ULL, 0xcfcfcf00cf0000cfULL, 0xb2b2b200b20000b2ULL,+ 0xc3c3c300c30000c3ULL, 0xb5b5b500b50000b5ULL, 0x7a7a7a007a00007aULL, 0x9191910091000091ULL,+ 0x2424240024000024ULL, 0x0808080008000008ULL, 0xe8e8e800e80000e8ULL, 0xa8a8a800a80000a8ULL,+ 0x6060600060000060ULL, 0xfcfcfc00fc0000fcULL, 0x6969690069000069ULL, 0x5050500050000050ULL,+ 0xaaaaaa00aa0000aaULL, 0xd0d0d000d00000d0ULL, 0xa0a0a000a00000a0ULL, 0x7d7d7d007d00007dULL,+ 0xa1a1a100a10000a1ULL, 0x8989890089000089ULL, 0x6262620062000062ULL, 0x9797970097000097ULL,+ 0x5454540054000054ULL, 0x5b5b5b005b00005bULL, 0x1e1e1e001e00001eULL, 0x9595950095000095ULL,+ 0xe0e0e000e00000e0ULL, 0xffffff00ff0000ffULL, 0x6464640064000064ULL, 0xd2d2d200d20000d2ULL,+ 0x1010100010000010ULL, 0xc4c4c400c40000c4ULL, 0x0000000000000000ULL, 0x4848480048000048ULL,+ 0xa3a3a300a30000a3ULL, 0xf7f7f700f70000f7ULL, 0x7575750075000075ULL, 0xdbdbdb00db0000dbULL,+ 0x8a8a8a008a00008aULL, 0x0303030003000003ULL, 0xe6e6e600e60000e6ULL, 0xdadada00da0000daULL,+ 0x0909090009000009ULL, 0x3f3f3f003f00003fULL, 0xdddddd00dd0000ddULL, 0x9494940094000094ULL,+ 0x8787870087000087ULL, 0x5c5c5c005c00005cULL, 0x8383830083000083ULL, 0x0202020002000002ULL,+ 0xcdcdcd00cd0000cdULL, 0x4a4a4a004a00004aULL, 0x9090900090000090ULL, 0x3333330033000033ULL,+ 0x7373730073000073ULL, 0x6767670067000067ULL, 0xf6f6f600f60000f6ULL, 0xf3f3f300f30000f3ULL,+ 0x9d9d9d009d00009dULL, 0x7f7f7f007f00007fULL, 0xbfbfbf00bf0000bfULL, 0xe2e2e200e20000e2ULL,+ 0x5252520052000052ULL, 0x9b9b9b009b00009bULL, 0xd8d8d800d80000d8ULL, 0x2626260026000026ULL,+ 0xc8c8c800c80000c8ULL, 0x3737370037000037ULL, 0xc6c6c600c60000c6ULL, 0x3b3b3b003b00003bULL,+ 0x8181810081000081ULL, 0x9696960096000096ULL, 0x6f6f6f006f00006fULL, 0x4b4b4b004b00004bULL,+ 0x1313130013000013ULL, 0xbebebe00be0000beULL, 0x6363630063000063ULL, 0x2e2e2e002e00002eULL,+ 0xe9e9e900e90000e9ULL, 0x7979790079000079ULL, 0xa7a7a700a70000a7ULL, 0x8c8c8c008c00008cULL,+ 0x9f9f9f009f00009fULL, 0x6e6e6e006e00006eULL, 0xbcbcbc00bc0000bcULL, 0x8e8e8e008e00008eULL,+ 0x2929290029000029ULL, 0xf5f5f500f50000f5ULL, 0xf9f9f900f90000f9ULL, 0xb6b6b600b60000b6ULL,+ 0x2f2f2f002f00002fULL, 0xfdfdfd00fd0000fdULL, 0xb4b4b400b40000b4ULL, 0x5959590059000059ULL,+ 0x7878780078000078ULL, 0x9898980098000098ULL, 0x0606060006000006ULL, 0x6a6a6a006a00006aULL,+ 0xe7e7e700e70000e7ULL, 0x4646460046000046ULL, 0x7171710071000071ULL, 0xbababa00ba0000baULL,+ 0xd4d4d400d40000d4ULL, 0x2525250025000025ULL, 0xababab00ab0000abULL, 0x4242420042000042ULL,+ 0x8888880088000088ULL, 0xa2a2a200a20000a2ULL, 0x8d8d8d008d00008dULL, 0xfafafa00fa0000faULL,+ 0x7272720072000072ULL, 0x0707070007000007ULL, 0xb9b9b900b90000b9ULL, 0x5555550055000055ULL,+ 0xf8f8f800f80000f8ULL, 0xeeeeee00ee0000eeULL, 0xacacac00ac0000acULL, 0x0a0a0a000a00000aULL,+ 0x3636360036000036ULL, 0x4949490049000049ULL, 0x2a2a2a002a00002aULL, 0x6868680068000068ULL,+ 0x3c3c3c003c00003cULL, 0x3838380038000038ULL, 0xf1f1f100f10000f1ULL, 0xa4a4a400a40000a4ULL,+ 0x4040400040000040ULL, 0x2828280028000028ULL, 0xd3d3d300d30000d3ULL, 0x7b7b7b007b00007bULL,+ 0xbbbbbb00bb0000bbULL, 0xc9c9c900c90000c9ULL, 0x4343430043000043ULL, 0xc1c1c100c10000c1ULL,+ 0x1515150015000015ULL, 0xe3e3e300e30000e3ULL, 0xadadad00ad0000adULL, 0xf4f4f400f40000f4ULL,+ 0x7777770077000077ULL, 0xc7c7c700c70000c7ULL, 0x8080800080000080ULL, 0x9e9e9e009e00009eULL,+},+{+ 0x00e0e0e0e0e00000ULL, 0x0005050505050000ULL, 0x0058585858580000ULL, 0x00d9d9d9d9d90000ULL,+ 0x0067676767670000ULL, 0x004e4e4e4e4e0000ULL, 0x0081818181810000ULL, 0x00cbcbcbcbcb0000ULL,+ 0x00c9c9c9c9c90000ULL, 0x000b0b0b0b0b0000ULL, 0x00aeaeaeaeae0000ULL, 0x006a6a6a6a6a0000ULL,+ 0x00d5d5d5d5d50000ULL, 0x0018181818180000ULL, 0x005d5d5d5d5d0000ULL, 0x0082828282820000ULL,+ 0x0046464646460000ULL, 0x00dfdfdfdfdf0000ULL, 0x00d6d6d6d6d60000ULL, 0x0027272727270000ULL,+ 0x008a8a8a8a8a0000ULL, 0x0032323232320000ULL, 0x004b4b4b4b4b0000ULL, 0x0042424242420000ULL,+ 0x00dbdbdbdbdb0000ULL, 0x001c1c1c1c1c0000ULL, 0x009e9e9e9e9e0000ULL, 0x009c9c9c9c9c0000ULL,+ 0x003a3a3a3a3a0000ULL, 0x00cacacacaca0000ULL, 0x0025252525250000ULL, 0x007b7b7b7b7b0000ULL,+ 0x000d0d0d0d0d0000ULL, 0x0071717171710000ULL, 0x005f5f5f5f5f0000ULL, 0x001f1f1f1f1f0000ULL,+ 0x00f8f8f8f8f80000ULL, 0x00d7d7d7d7d70000ULL, 0x003e3e3e3e3e0000ULL, 0x009d9d9d9d9d0000ULL,+ 0x007c7c7c7c7c0000ULL, 0x0060606060600000ULL, 0x00b9b9b9b9b90000ULL, 0x00bebebebebe0000ULL,+ 0x00bcbcbcbcbc0000ULL, 0x008b8b8b8b8b0000ULL, 0x0016161616160000ULL, 0x0034343434340000ULL,+ 0x004d4d4d4d4d0000ULL, 0x00c3c3c3c3c30000ULL, 0x0072727272720000ULL, 0x0095959595950000ULL,+ 0x00ababababab0000ULL, 0x008e8e8e8e8e0000ULL, 0x00bababababa0000ULL, 0x007a7a7a7a7a0000ULL,+ 0x00b3b3b3b3b30000ULL, 0x0002020202020000ULL, 0x00b4b4b4b4b40000ULL, 0x00adadadadad0000ULL,+ 0x00a2a2a2a2a20000ULL, 0x00acacacacac0000ULL, 0x00d8d8d8d8d80000ULL, 0x009a9a9a9a9a0000ULL,+ 0x0017171717170000ULL, 0x001a1a1a1a1a0000ULL, 0x0035353535350000ULL, 0x00cccccccccc0000ULL,+ 0x00f7f7f7f7f70000ULL, 0x0099999999990000ULL, 0x0061616161610000ULL, 0x005a5a5a5a5a0000ULL,+ 0x00e8e8e8e8e80000ULL, 0x0024242424240000ULL, 0x0056565656560000ULL, 0x0040404040400000ULL,+ 0x00e1e1e1e1e10000ULL, 0x0063636363630000ULL, 0x0009090909090000ULL, 0x0033333333330000ULL,+ 0x00bfbfbfbfbf0000ULL, 0x0098989898980000ULL, 0x0097979797970000ULL, 0x0085858585850000ULL,+ 0x0068686868680000ULL, 0x00fcfcfcfcfc0000ULL, 0x00ececececec0000ULL, 0x000a0a0a0a0a0000ULL,+ 0x00dadadadada0000ULL, 0x006f6f6f6f6f0000ULL, 0x0053535353530000ULL, 0x0062626262620000ULL,+ 0x00a3a3a3a3a30000ULL, 0x002e2e2e2e2e0000ULL, 0x0008080808080000ULL, 0x00afafafafaf0000ULL,+ 0x0028282828280000ULL, 0x00b0b0b0b0b00000ULL, 0x0074747474740000ULL, 0x00c2c2c2c2c20000ULL,+ 0x00bdbdbdbdbd0000ULL, 0x0036363636360000ULL, 0x0022222222220000ULL, 0x0038383838380000ULL,+ 0x0064646464640000ULL, 0x001e1e1e1e1e0000ULL, 0x0039393939390000ULL, 0x002c2c2c2c2c0000ULL,+ 0x00a6a6a6a6a60000ULL, 0x0030303030300000ULL, 0x00e5e5e5e5e50000ULL, 0x0044444444440000ULL,+ 0x00fdfdfdfdfd0000ULL, 0x0088888888880000ULL, 0x009f9f9f9f9f0000ULL, 0x0065656565650000ULL,+ 0x0087878787870000ULL, 0x006b6b6b6b6b0000ULL, 0x00f4f4f4f4f40000ULL, 0x0023232323230000ULL,+ 0x0048484848480000ULL, 0x0010101010100000ULL, 0x00d1d1d1d1d10000ULL, 0x0051515151510000ULL,+ 0x00c0c0c0c0c00000ULL, 0x00f9f9f9f9f90000ULL, 0x00d2d2d2d2d20000ULL, 0x00a0a0a0a0a00000ULL,+ 0x0055555555550000ULL, 0x00a1a1a1a1a10000ULL, 0x0041414141410000ULL, 0x00fafafafafa0000ULL,+ 0x0043434343430000ULL, 0x0013131313130000ULL, 0x00c4c4c4c4c40000ULL, 0x002f2f2f2f2f0000ULL,+ 0x00a8a8a8a8a80000ULL, 0x00b6b6b6b6b60000ULL, 0x003c3c3c3c3c0000ULL, 0x002b2b2b2b2b0000ULL,+ 0x00c1c1c1c1c10000ULL, 0x00ffffffffff0000ULL, 0x00c8c8c8c8c80000ULL, 0x00a5a5a5a5a50000ULL,+ 0x0020202020200000ULL, 0x0089898989890000ULL, 0x0000000000000000ULL, 0x0090909090900000ULL,+ 0x0047474747470000ULL, 0x00efefefefef0000ULL, 0x00eaeaeaeaea0000ULL, 0x00b7b7b7b7b70000ULL,+ 0x0015151515150000ULL, 0x0006060606060000ULL, 0x00cdcdcdcdcd0000ULL, 0x00b5b5b5b5b50000ULL,+ 0x0012121212120000ULL, 0x007e7e7e7e7e0000ULL, 0x00bbbbbbbbbb0000ULL, 0x0029292929290000ULL,+ 0x000f0f0f0f0f0000ULL, 0x00b8b8b8b8b80000ULL, 0x0007070707070000ULL, 0x0004040404040000ULL,+ 0x009b9b9b9b9b0000ULL, 0x0094949494940000ULL, 0x0021212121210000ULL, 0x0066666666660000ULL,+ 0x00e6e6e6e6e60000ULL, 0x00cecececece0000ULL, 0x00ededededed0000ULL, 0x00e7e7e7e7e70000ULL,+ 0x003b3b3b3b3b0000ULL, 0x00fefefefefe0000ULL, 0x007f7f7f7f7f0000ULL, 0x00c5c5c5c5c50000ULL,+ 0x00a4a4a4a4a40000ULL, 0x0037373737370000ULL, 0x00b1b1b1b1b10000ULL, 0x004c4c4c4c4c0000ULL,+ 0x0091919191910000ULL, 0x006e6e6e6e6e0000ULL, 0x008d8d8d8d8d0000ULL, 0x0076767676760000ULL,+ 0x0003030303030000ULL, 0x002d2d2d2d2d0000ULL, 0x00dedededede0000ULL, 0x0096969696960000ULL,+ 0x0026262626260000ULL, 0x007d7d7d7d7d0000ULL, 0x00c6c6c6c6c60000ULL, 0x005c5c5c5c5c0000ULL,+ 0x00d3d3d3d3d30000ULL, 0x00f2f2f2f2f20000ULL, 0x004f4f4f4f4f0000ULL, 0x0019191919190000ULL,+ 0x003f3f3f3f3f0000ULL, 0x00dcdcdcdcdc0000ULL, 0x0079797979790000ULL, 0x001d1d1d1d1d0000ULL,+ 0x0052525252520000ULL, 0x00ebebebebeb0000ULL, 0x00f3f3f3f3f30000ULL, 0x006d6d6d6d6d0000ULL,+ 0x005e5e5e5e5e0000ULL, 0x00fbfbfbfbfb0000ULL, 0x0069696969690000ULL, 0x00b2b2b2b2b20000ULL,+ 0x00f0f0f0f0f00000ULL, 0x0031313131310000ULL, 0x000c0c0c0c0c0000ULL, 0x00d4d4d4d4d40000ULL,+ 0x00cfcfcfcfcf0000ULL, 0x008c8c8c8c8c0000ULL, 0x00e2e2e2e2e20000ULL, 0x0075757575750000ULL,+ 0x00a9a9a9a9a90000ULL, 0x004a4a4a4a4a0000ULL, 0x0057575757570000ULL, 0x0084848484840000ULL,+ 0x0011111111110000ULL, 0x0045454545450000ULL, 0x001b1b1b1b1b0000ULL, 0x00f5f5f5f5f50000ULL,+ 0x00e4e4e4e4e40000ULL, 0x000e0e0e0e0e0000ULL, 0x0073737373730000ULL, 0x00aaaaaaaaaa0000ULL,+ 0x00f1f1f1f1f10000ULL, 0x00dddddddddd0000ULL, 0x0059595959590000ULL, 0x0014141414140000ULL,+ 0x006c6c6c6c6c0000ULL, 0x0092929292920000ULL, 0x0054545454540000ULL, 0x00d0d0d0d0d00000ULL,+ 0x0078787878780000ULL, 0x0070707070700000ULL, 0x00e3e3e3e3e30000ULL, 0x0049494949490000ULL,+ 0x0080808080800000ULL, 0x0050505050500000ULL, 0x00a7a7a7a7a70000ULL, 0x00f6f6f6f6f60000ULL,+ 0x0077777777770000ULL, 0x0093939393930000ULL, 0x0086868686860000ULL, 0x0083838383830000ULL,+ 0x002a2a2a2a2a0000ULL, 0x00c7c7c7c7c70000ULL, 0x005b5b5b5b5b0000ULL, 0x00e9e9e9e9e90000ULL,+ 0x00eeeeeeeeee0000ULL, 0x008f8f8f8f8f0000ULL, 0x0001010101010000ULL, 0x003d3d3d3d3d0000ULL,+},+{+ 0x3800383800383800ULL, 0x4100414100414100ULL, 0x1600161600161600ULL, 0x7600767600767600ULL,+ 0xd900d9d900d9d900ULL, 0x9300939300939300ULL, 0x6000606000606000ULL, 0xf200f2f200f2f200ULL,+ 0x7200727200727200ULL, 0xc200c2c200c2c200ULL, 0xab00abab00abab00ULL, 0x9a009a9a009a9a00ULL,+ 0x7500757500757500ULL, 0x0600060600060600ULL, 0x5700575700575700ULL, 0xa000a0a000a0a000ULL,+ 0x9100919100919100ULL, 0xf700f7f700f7f700ULL, 0xb500b5b500b5b500ULL, 0xc900c9c900c9c900ULL,+ 0xa200a2a200a2a200ULL, 0x8c008c8c008c8c00ULL, 0xd200d2d200d2d200ULL, 0x9000909000909000ULL,+ 0xf600f6f600f6f600ULL, 0x0700070700070700ULL, 0xa700a7a700a7a700ULL, 0x2700272700272700ULL,+ 0x8e008e8e008e8e00ULL, 0xb200b2b200b2b200ULL, 0x4900494900494900ULL, 0xde00dede00dede00ULL,+ 0x4300434300434300ULL, 0x5c005c5c005c5c00ULL, 0xd700d7d700d7d700ULL, 0xc700c7c700c7c700ULL,+ 0x3e003e3e003e3e00ULL, 0xf500f5f500f5f500ULL, 0x8f008f8f008f8f00ULL, 0x6700676700676700ULL,+ 0x1f001f1f001f1f00ULL, 0x1800181800181800ULL, 0x6e006e6e006e6e00ULL, 0xaf00afaf00afaf00ULL,+ 0x2f002f2f002f2f00ULL, 0xe200e2e200e2e200ULL, 0x8500858500858500ULL, 0x0d000d0d000d0d00ULL,+ 0x5300535300535300ULL, 0xf000f0f000f0f000ULL, 0x9c009c9c009c9c00ULL, 0x6500656500656500ULL,+ 0xea00eaea00eaea00ULL, 0xa300a3a300a3a300ULL, 0xae00aeae00aeae00ULL, 0x9e009e9e009e9e00ULL,+ 0xec00ecec00ecec00ULL, 0x8000808000808000ULL, 0x2d002d2d002d2d00ULL, 0x6b006b6b006b6b00ULL,+ 0xa800a8a800a8a800ULL, 0x2b002b2b002b2b00ULL, 0x3600363600363600ULL, 0xa600a6a600a6a600ULL,+ 0xc500c5c500c5c500ULL, 0x8600868600868600ULL, 0x4d004d4d004d4d00ULL, 0x3300333300333300ULL,+ 0xfd00fdfd00fdfd00ULL, 0x6600666600666600ULL, 0x5800585800585800ULL, 0x9600969600969600ULL,+ 0x3a003a3a003a3a00ULL, 0x0900090900090900ULL, 0x9500959500959500ULL, 0x1000101000101000ULL,+ 0x7800787800787800ULL, 0xd800d8d800d8d800ULL, 0x4200424200424200ULL, 0xcc00cccc00cccc00ULL,+ 0xef00efef00efef00ULL, 0x2600262600262600ULL, 0xe500e5e500e5e500ULL, 0x6100616100616100ULL,+ 0x1a001a1a001a1a00ULL, 0x3f003f3f003f3f00ULL, 0x3b003b3b003b3b00ULL, 0x8200828200828200ULL,+ 0xb600b6b600b6b600ULL, 0xdb00dbdb00dbdb00ULL, 0xd400d4d400d4d400ULL, 0x9800989800989800ULL,+ 0xe800e8e800e8e800ULL, 0x8b008b8b008b8b00ULL, 0x0200020200020200ULL, 0xeb00ebeb00ebeb00ULL,+ 0x0a000a0a000a0a00ULL, 0x2c002c2c002c2c00ULL, 0x1d001d1d001d1d00ULL, 0xb000b0b000b0b000ULL,+ 0x6f006f6f006f6f00ULL, 0x8d008d8d008d8d00ULL, 0x8800888800888800ULL, 0x0e000e0e000e0e00ULL,+ 0x1900191900191900ULL, 0x8700878700878700ULL, 0x4e004e4e004e4e00ULL, 0x0b000b0b000b0b00ULL,+ 0xa900a9a900a9a900ULL, 0x0c000c0c000c0c00ULL, 0x7900797900797900ULL, 0x1100111100111100ULL,+ 0x7f007f7f007f7f00ULL, 0x2200222200222200ULL, 0xe700e7e700e7e700ULL, 0x5900595900595900ULL,+ 0xe100e1e100e1e100ULL, 0xda00dada00dada00ULL, 0x3d003d3d003d3d00ULL, 0xc800c8c800c8c800ULL,+ 0x1200121200121200ULL, 0x0400040400040400ULL, 0x7400747400747400ULL, 0x5400545400545400ULL,+ 0x3000303000303000ULL, 0x7e007e7e007e7e00ULL, 0xb400b4b400b4b400ULL, 0x2800282800282800ULL,+ 0x5500555500555500ULL, 0x6800686800686800ULL, 0x5000505000505000ULL, 0xbe00bebe00bebe00ULL,+ 0xd000d0d000d0d000ULL, 0xc400c4c400c4c400ULL, 0x3100313100313100ULL, 0xcb00cbcb00cbcb00ULL,+ 0x2a002a2a002a2a00ULL, 0xad00adad00adad00ULL, 0x0f000f0f000f0f00ULL, 0xca00caca00caca00ULL,+ 0x7000707000707000ULL, 0xff00ffff00ffff00ULL, 0x3200323200323200ULL, 0x6900696900696900ULL,+ 0x0800080800080800ULL, 0x6200626200626200ULL, 0x0000000000000000ULL, 0x2400242400242400ULL,+ 0xd100d1d100d1d100ULL, 0xfb00fbfb00fbfb00ULL, 0xba00baba00baba00ULL, 0xed00eded00eded00ULL,+ 0x4500454500454500ULL, 0x8100818100818100ULL, 0x7300737300737300ULL, 0x6d006d6d006d6d00ULL,+ 0x8400848400848400ULL, 0x9f009f9f009f9f00ULL, 0xee00eeee00eeee00ULL, 0x4a004a4a004a4a00ULL,+ 0xc300c3c300c3c300ULL, 0x2e002e2e002e2e00ULL, 0xc100c1c100c1c100ULL, 0x0100010100010100ULL,+ 0xe600e6e600e6e600ULL, 0x2500252500252500ULL, 0x4800484800484800ULL, 0x9900999900999900ULL,+ 0xb900b9b900b9b900ULL, 0xb300b3b300b3b300ULL, 0x7b007b7b007b7b00ULL, 0xf900f9f900f9f900ULL,+ 0xce00cece00cece00ULL, 0xbf00bfbf00bfbf00ULL, 0xdf00dfdf00dfdf00ULL, 0x7100717100717100ULL,+ 0x2900292900292900ULL, 0xcd00cdcd00cdcd00ULL, 0x6c006c6c006c6c00ULL, 0x1300131300131300ULL,+ 0x6400646400646400ULL, 0x9b009b9b009b9b00ULL, 0x6300636300636300ULL, 0x9d009d9d009d9d00ULL,+ 0xc000c0c000c0c000ULL, 0x4b004b4b004b4b00ULL, 0xb700b7b700b7b700ULL, 0xa500a5a500a5a500ULL,+ 0x8900898900898900ULL, 0x5f005f5f005f5f00ULL, 0xb100b1b100b1b100ULL, 0x1700171700171700ULL,+ 0xf400f4f400f4f400ULL, 0xbc00bcbc00bcbc00ULL, 0xd300d3d300d3d300ULL, 0x4600464600464600ULL,+ 0xcf00cfcf00cfcf00ULL, 0x3700373700373700ULL, 0x5e005e5e005e5e00ULL, 0x4700474700474700ULL,+ 0x9400949400949400ULL, 0xfa00fafa00fafa00ULL, 0xfc00fcfc00fcfc00ULL, 0x5b005b5b005b5b00ULL,+ 0x9700979700979700ULL, 0xfe00fefe00fefe00ULL, 0x5a005a5a005a5a00ULL, 0xac00acac00acac00ULL,+ 0x3c003c3c003c3c00ULL, 0x4c004c4c004c4c00ULL, 0x0300030300030300ULL, 0x3500353500353500ULL,+ 0xf300f3f300f3f300ULL, 0x2300232300232300ULL, 0xb800b8b800b8b800ULL, 0x5d005d5d005d5d00ULL,+ 0x6a006a6a006a6a00ULL, 0x9200929200929200ULL, 0xd500d5d500d5d500ULL, 0x2100212100212100ULL,+ 0x4400444400444400ULL, 0x5100515100515100ULL, 0xc600c6c600c6c600ULL, 0x7d007d7d007d7d00ULL,+ 0x3900393900393900ULL, 0x8300838300838300ULL, 0xdc00dcdc00dcdc00ULL, 0xaa00aaaa00aaaa00ULL,+ 0x7c007c7c007c7c00ULL, 0x7700777700777700ULL, 0x5600565600565600ULL, 0x0500050500050500ULL,+ 0x1b001b1b001b1b00ULL, 0xa400a4a400a4a400ULL, 0x1500151500151500ULL, 0x3400343400343400ULL,+ 0x1e001e1e001e1e00ULL, 0x1c001c1c001c1c00ULL, 0xf800f8f800f8f800ULL, 0x5200525200525200ULL,+ 0x2000202000202000ULL, 0x1400141400141400ULL, 0xe900e9e900e9e900ULL, 0xbd00bdbd00bdbd00ULL,+ 0xdd00dddd00dddd00ULL, 0xe400e4e400e4e400ULL, 0xa100a1a100a1a100ULL, 0xe000e0e000e0e000ULL,+ 0x8a008a8a008a8a00ULL, 0xf100f1f100f1f100ULL, 0xd600d6d600d6d600ULL, 0x7a007a7a007a7a00ULL,+ 0xbb00bbbb00bbbb00ULL, 0xe300e3e300e3e300ULL, 0x4000404000404000ULL, 0x4f004f4f004f4f00ULL,+},+{+ 0x7070007000007070ULL, 0x2c2c002c00002c2cULL, 0xb3b300b30000b3b3ULL, 0xc0c000c00000c0c0ULL,+ 0xe4e400e40000e4e4ULL, 0x5757005700005757ULL, 0xeaea00ea0000eaeaULL, 0xaeae00ae0000aeaeULL,+ 0x2323002300002323ULL, 0x6b6b006b00006b6bULL, 0x4545004500004545ULL, 0xa5a500a50000a5a5ULL,+ 0xeded00ed0000ededULL, 0x4f4f004f00004f4fULL, 0x1d1d001d00001d1dULL, 0x9292009200009292ULL,+ 0x8686008600008686ULL, 0xafaf00af0000afafULL, 0x7c7c007c00007c7cULL, 0x1f1f001f00001f1fULL,+ 0x3e3e003e00003e3eULL, 0xdcdc00dc0000dcdcULL, 0x5e5e005e00005e5eULL, 0x0b0b000b00000b0bULL,+ 0xa6a600a60000a6a6ULL, 0x3939003900003939ULL, 0xd5d500d50000d5d5ULL, 0x5d5d005d00005d5dULL,+ 0xd9d900d90000d9d9ULL, 0x5a5a005a00005a5aULL, 0x5151005100005151ULL, 0x6c6c006c00006c6cULL,+ 0x8b8b008b00008b8bULL, 0x9a9a009a00009a9aULL, 0xfbfb00fb0000fbfbULL, 0xb0b000b00000b0b0ULL,+ 0x7474007400007474ULL, 0x2b2b002b00002b2bULL, 0xf0f000f00000f0f0ULL, 0x8484008400008484ULL,+ 0xdfdf00df0000dfdfULL, 0xcbcb00cb0000cbcbULL, 0x3434003400003434ULL, 0x7676007600007676ULL,+ 0x6d6d006d00006d6dULL, 0xa9a900a90000a9a9ULL, 0xd1d100d10000d1d1ULL, 0x0404000400000404ULL,+ 0x1414001400001414ULL, 0x3a3a003a00003a3aULL, 0xdede00de0000dedeULL, 0x1111001100001111ULL,+ 0x3232003200003232ULL, 0x9c9c009c00009c9cULL, 0x5353005300005353ULL, 0xf2f200f20000f2f2ULL,+ 0xfefe00fe0000fefeULL, 0xcfcf00cf0000cfcfULL, 0xc3c300c30000c3c3ULL, 0x7a7a007a00007a7aULL,+ 0x2424002400002424ULL, 0xe8e800e80000e8e8ULL, 0x6060006000006060ULL, 0x6969006900006969ULL,+ 0xaaaa00aa0000aaaaULL, 0xa0a000a00000a0a0ULL, 0xa1a100a10000a1a1ULL, 0x6262006200006262ULL,+ 0x5454005400005454ULL, 0x1e1e001e00001e1eULL, 0xe0e000e00000e0e0ULL, 0x6464006400006464ULL,+ 0x1010001000001010ULL, 0x0000000000000000ULL, 0xa3a300a30000a3a3ULL, 0x7575007500007575ULL,+ 0x8a8a008a00008a8aULL, 0xe6e600e60000e6e6ULL, 0x0909000900000909ULL, 0xdddd00dd0000ddddULL,+ 0x8787008700008787ULL, 0x8383008300008383ULL, 0xcdcd00cd0000cdcdULL, 0x9090009000009090ULL,+ 0x7373007300007373ULL, 0xf6f600f60000f6f6ULL, 0x9d9d009d00009d9dULL, 0xbfbf00bf0000bfbfULL,+ 0x5252005200005252ULL, 0xd8d800d80000d8d8ULL, 0xc8c800c80000c8c8ULL, 0xc6c600c60000c6c6ULL,+ 0x8181008100008181ULL, 0x6f6f006f00006f6fULL, 0x1313001300001313ULL, 0x6363006300006363ULL,+ 0xe9e900e90000e9e9ULL, 0xa7a700a70000a7a7ULL, 0x9f9f009f00009f9fULL, 0xbcbc00bc0000bcbcULL,+ 0x2929002900002929ULL, 0xf9f900f90000f9f9ULL, 0x2f2f002f00002f2fULL, 0xb4b400b40000b4b4ULL,+ 0x7878007800007878ULL, 0x0606000600000606ULL, 0xe7e700e70000e7e7ULL, 0x7171007100007171ULL,+ 0xd4d400d40000d4d4ULL, 0xabab00ab0000ababULL, 0x8888008800008888ULL, 0x8d8d008d00008d8dULL,+ 0x7272007200007272ULL, 0xb9b900b90000b9b9ULL, 0xf8f800f80000f8f8ULL, 0xacac00ac0000acacULL,+ 0x3636003600003636ULL, 0x2a2a002a00002a2aULL, 0x3c3c003c00003c3cULL, 0xf1f100f10000f1f1ULL,+ 0x4040004000004040ULL, 0xd3d300d30000d3d3ULL, 0xbbbb00bb0000bbbbULL, 0x4343004300004343ULL,+ 0x1515001500001515ULL, 0xadad00ad0000adadULL, 0x7777007700007777ULL, 0x8080008000008080ULL,+ 0x8282008200008282ULL, 0xecec00ec0000ececULL, 0x2727002700002727ULL, 0xe5e500e50000e5e5ULL,+ 0x8585008500008585ULL, 0x3535003500003535ULL, 0x0c0c000c00000c0cULL, 0x4141004100004141ULL,+ 0xefef00ef0000efefULL, 0x9393009300009393ULL, 0x1919001900001919ULL, 0x2121002100002121ULL,+ 0x0e0e000e00000e0eULL, 0x4e4e004e00004e4eULL, 0x6565006500006565ULL, 0xbdbd00bd0000bdbdULL,+ 0xb8b800b80000b8b8ULL, 0x8f8f008f00008f8fULL, 0xebeb00eb0000ebebULL, 0xcece00ce0000ceceULL,+ 0x3030003000003030ULL, 0x5f5f005f00005f5fULL, 0xc5c500c50000c5c5ULL, 0x1a1a001a00001a1aULL,+ 0xe1e100e10000e1e1ULL, 0xcaca00ca0000cacaULL, 0x4747004700004747ULL, 0x3d3d003d00003d3dULL,+ 0x0101000100000101ULL, 0xd6d600d60000d6d6ULL, 0x5656005600005656ULL, 0x4d4d004d00004d4dULL,+ 0x0d0d000d00000d0dULL, 0x6666006600006666ULL, 0xcccc00cc0000ccccULL, 0x2d2d002d00002d2dULL,+ 0x1212001200001212ULL, 0x2020002000002020ULL, 0xb1b100b10000b1b1ULL, 0x9999009900009999ULL,+ 0x4c4c004c00004c4cULL, 0xc2c200c20000c2c2ULL, 0x7e7e007e00007e7eULL, 0x0505000500000505ULL,+ 0xb7b700b70000b7b7ULL, 0x3131003100003131ULL, 0x1717001700001717ULL, 0xd7d700d70000d7d7ULL,+ 0x5858005800005858ULL, 0x6161006100006161ULL, 0x1b1b001b00001b1bULL, 0x1c1c001c00001c1cULL,+ 0x0f0f000f00000f0fULL, 0x1616001600001616ULL, 0x1818001800001818ULL, 0x2222002200002222ULL,+ 0x4444004400004444ULL, 0xb2b200b20000b2b2ULL, 0xb5b500b50000b5b5ULL, 0x9191009100009191ULL,+ 0x0808000800000808ULL, 0xa8a800a80000a8a8ULL, 0xfcfc00fc0000fcfcULL, 0x5050005000005050ULL,+ 0xd0d000d00000d0d0ULL, 0x7d7d007d00007d7dULL, 0x8989008900008989ULL, 0x9797009700009797ULL,+ 0x5b5b005b00005b5bULL, 0x9595009500009595ULL, 0xffff00ff0000ffffULL, 0xd2d200d20000d2d2ULL,+ 0xc4c400c40000c4c4ULL, 0x4848004800004848ULL, 0xf7f700f70000f7f7ULL, 0xdbdb00db0000dbdbULL,+ 0x0303000300000303ULL, 0xdada00da0000dadaULL, 0x3f3f003f00003f3fULL, 0x9494009400009494ULL,+ 0x5c5c005c00005c5cULL, 0x0202000200000202ULL, 0x4a4a004a00004a4aULL, 0x3333003300003333ULL,+ 0x6767006700006767ULL, 0xf3f300f30000f3f3ULL, 0x7f7f007f00007f7fULL, 0xe2e200e20000e2e2ULL,+ 0x9b9b009b00009b9bULL, 0x2626002600002626ULL, 0x3737003700003737ULL, 0x3b3b003b00003b3bULL,+ 0x9696009600009696ULL, 0x4b4b004b00004b4bULL, 0xbebe00be0000bebeULL, 0x2e2e002e00002e2eULL,+ 0x7979007900007979ULL, 0x8c8c008c00008c8cULL, 0x6e6e006e00006e6eULL, 0x8e8e008e00008e8eULL,+ 0xf5f500f50000f5f5ULL, 0xb6b600b60000b6b6ULL, 0xfdfd00fd0000fdfdULL, 0x5959005900005959ULL,+ 0x9898009800009898ULL, 0x6a6a006a00006a6aULL, 0x4646004600004646ULL, 0xbaba00ba0000babaULL,+ 0x2525002500002525ULL, 0x4242004200004242ULL, 0xa2a200a20000a2a2ULL, 0xfafa00fa0000fafaULL,+ 0x0707000700000707ULL, 0x5555005500005555ULL, 0xeeee00ee0000eeeeULL, 0x0a0a000a00000a0aULL,+ 0x4949004900004949ULL, 0x6868006800006868ULL, 0x3838003800003838ULL, 0xa4a400a40000a4a4ULL,+ 0x2828002800002828ULL, 0x7b7b007b00007b7bULL, 0xc9c900c90000c9c9ULL, 0xc1c100c10000c1c1ULL,+ 0xe3e300e30000e3e3ULL, 0xf4f400f40000f4f4ULL, 0xc7c700c70000c7c7ULL, 0x9e9e009e00009e9eULL,+},+{+ 0x00e0e0e000e0e0e0ULL, 0x0005050500050505ULL, 0x0058585800585858ULL, 0x00d9d9d900d9d9d9ULL,+ 0x0067676700676767ULL, 0x004e4e4e004e4e4eULL, 0x0081818100818181ULL, 0x00cbcbcb00cbcbcbULL,+ 0x00c9c9c900c9c9c9ULL, 0x000b0b0b000b0b0bULL, 0x00aeaeae00aeaeaeULL, 0x006a6a6a006a6a6aULL,+ 0x00d5d5d500d5d5d5ULL, 0x0018181800181818ULL, 0x005d5d5d005d5d5dULL, 0x0082828200828282ULL,+ 0x0046464600464646ULL, 0x00dfdfdf00dfdfdfULL, 0x00d6d6d600d6d6d6ULL, 0x0027272700272727ULL,+ 0x008a8a8a008a8a8aULL, 0x0032323200323232ULL, 0x004b4b4b004b4b4bULL, 0x0042424200424242ULL,+ 0x00dbdbdb00dbdbdbULL, 0x001c1c1c001c1c1cULL, 0x009e9e9e009e9e9eULL, 0x009c9c9c009c9c9cULL,+ 0x003a3a3a003a3a3aULL, 0x00cacaca00cacacaULL, 0x0025252500252525ULL, 0x007b7b7b007b7b7bULL,+ 0x000d0d0d000d0d0dULL, 0x0071717100717171ULL, 0x005f5f5f005f5f5fULL, 0x001f1f1f001f1f1fULL,+ 0x00f8f8f800f8f8f8ULL, 0x00d7d7d700d7d7d7ULL, 0x003e3e3e003e3e3eULL, 0x009d9d9d009d9d9dULL,+ 0x007c7c7c007c7c7cULL, 0x0060606000606060ULL, 0x00b9b9b900b9b9b9ULL, 0x00bebebe00bebebeULL,+ 0x00bcbcbc00bcbcbcULL, 0x008b8b8b008b8b8bULL, 0x0016161600161616ULL, 0x0034343400343434ULL,+ 0x004d4d4d004d4d4dULL, 0x00c3c3c300c3c3c3ULL, 0x0072727200727272ULL, 0x0095959500959595ULL,+ 0x00ababab00abababULL, 0x008e8e8e008e8e8eULL, 0x00bababa00bababaULL, 0x007a7a7a007a7a7aULL,+ 0x00b3b3b300b3b3b3ULL, 0x0002020200020202ULL, 0x00b4b4b400b4b4b4ULL, 0x00adadad00adadadULL,+ 0x00a2a2a200a2a2a2ULL, 0x00acacac00acacacULL, 0x00d8d8d800d8d8d8ULL, 0x009a9a9a009a9a9aULL,+ 0x0017171700171717ULL, 0x001a1a1a001a1a1aULL, 0x0035353500353535ULL, 0x00cccccc00ccccccULL,+ 0x00f7f7f700f7f7f7ULL, 0x0099999900999999ULL, 0x0061616100616161ULL, 0x005a5a5a005a5a5aULL,+ 0x00e8e8e800e8e8e8ULL, 0x0024242400242424ULL, 0x0056565600565656ULL, 0x0040404000404040ULL,+ 0x00e1e1e100e1e1e1ULL, 0x0063636300636363ULL, 0x0009090900090909ULL, 0x0033333300333333ULL,+ 0x00bfbfbf00bfbfbfULL, 0x0098989800989898ULL, 0x0097979700979797ULL, 0x0085858500858585ULL,+ 0x0068686800686868ULL, 0x00fcfcfc00fcfcfcULL, 0x00ececec00ecececULL, 0x000a0a0a000a0a0aULL,+ 0x00dadada00dadadaULL, 0x006f6f6f006f6f6fULL, 0x0053535300535353ULL, 0x0062626200626262ULL,+ 0x00a3a3a300a3a3a3ULL, 0x002e2e2e002e2e2eULL, 0x0008080800080808ULL, 0x00afafaf00afafafULL,+ 0x0028282800282828ULL, 0x00b0b0b000b0b0b0ULL, 0x0074747400747474ULL, 0x00c2c2c200c2c2c2ULL,+ 0x00bdbdbd00bdbdbdULL, 0x0036363600363636ULL, 0x0022222200222222ULL, 0x0038383800383838ULL,+ 0x0064646400646464ULL, 0x001e1e1e001e1e1eULL, 0x0039393900393939ULL, 0x002c2c2c002c2c2cULL,+ 0x00a6a6a600a6a6a6ULL, 0x0030303000303030ULL, 0x00e5e5e500e5e5e5ULL, 0x0044444400444444ULL,+ 0x00fdfdfd00fdfdfdULL, 0x0088888800888888ULL, 0x009f9f9f009f9f9fULL, 0x0065656500656565ULL,+ 0x0087878700878787ULL, 0x006b6b6b006b6b6bULL, 0x00f4f4f400f4f4f4ULL, 0x0023232300232323ULL,+ 0x0048484800484848ULL, 0x0010101000101010ULL, 0x00d1d1d100d1d1d1ULL, 0x0051515100515151ULL,+ 0x00c0c0c000c0c0c0ULL, 0x00f9f9f900f9f9f9ULL, 0x00d2d2d200d2d2d2ULL, 0x00a0a0a000a0a0a0ULL,+ 0x0055555500555555ULL, 0x00a1a1a100a1a1a1ULL, 0x0041414100414141ULL, 0x00fafafa00fafafaULL,+ 0x0043434300434343ULL, 0x0013131300131313ULL, 0x00c4c4c400c4c4c4ULL, 0x002f2f2f002f2f2fULL,+ 0x00a8a8a800a8a8a8ULL, 0x00b6b6b600b6b6b6ULL, 0x003c3c3c003c3c3cULL, 0x002b2b2b002b2b2bULL,+ 0x00c1c1c100c1c1c1ULL, 0x00ffffff00ffffffULL, 0x00c8c8c800c8c8c8ULL, 0x00a5a5a500a5a5a5ULL,+ 0x0020202000202020ULL, 0x0089898900898989ULL, 0x0000000000000000ULL, 0x0090909000909090ULL,+ 0x0047474700474747ULL, 0x00efefef00efefefULL, 0x00eaeaea00eaeaeaULL, 0x00b7b7b700b7b7b7ULL,+ 0x0015151500151515ULL, 0x0006060600060606ULL, 0x00cdcdcd00cdcdcdULL, 0x00b5b5b500b5b5b5ULL,+ 0x0012121200121212ULL, 0x007e7e7e007e7e7eULL, 0x00bbbbbb00bbbbbbULL, 0x0029292900292929ULL,+ 0x000f0f0f000f0f0fULL, 0x00b8b8b800b8b8b8ULL, 0x0007070700070707ULL, 0x0004040400040404ULL,+ 0x009b9b9b009b9b9bULL, 0x0094949400949494ULL, 0x0021212100212121ULL, 0x0066666600666666ULL,+ 0x00e6e6e600e6e6e6ULL, 0x00cecece00cececeULL, 0x00ededed00edededULL, 0x00e7e7e700e7e7e7ULL,+ 0x003b3b3b003b3b3bULL, 0x00fefefe00fefefeULL, 0x007f7f7f007f7f7fULL, 0x00c5c5c500c5c5c5ULL,+ 0x00a4a4a400a4a4a4ULL, 0x0037373700373737ULL, 0x00b1b1b100b1b1b1ULL, 0x004c4c4c004c4c4cULL,+ 0x0091919100919191ULL, 0x006e6e6e006e6e6eULL, 0x008d8d8d008d8d8dULL, 0x0076767600767676ULL,+ 0x0003030300030303ULL, 0x002d2d2d002d2d2dULL, 0x00dedede00dededeULL, 0x0096969600969696ULL,+ 0x0026262600262626ULL, 0x007d7d7d007d7d7dULL, 0x00c6c6c600c6c6c6ULL, 0x005c5c5c005c5c5cULL,+ 0x00d3d3d300d3d3d3ULL, 0x00f2f2f200f2f2f2ULL, 0x004f4f4f004f4f4fULL, 0x0019191900191919ULL,+ 0x003f3f3f003f3f3fULL, 0x00dcdcdc00dcdcdcULL, 0x0079797900797979ULL, 0x001d1d1d001d1d1dULL,+ 0x0052525200525252ULL, 0x00ebebeb00ebebebULL, 0x00f3f3f300f3f3f3ULL, 0x006d6d6d006d6d6dULL,+ 0x005e5e5e005e5e5eULL, 0x00fbfbfb00fbfbfbULL, 0x0069696900696969ULL, 0x00b2b2b200b2b2b2ULL,+ 0x00f0f0f000f0f0f0ULL, 0x0031313100313131ULL, 0x000c0c0c000c0c0cULL, 0x00d4d4d400d4d4d4ULL,+ 0x00cfcfcf00cfcfcfULL, 0x008c8c8c008c8c8cULL, 0x00e2e2e200e2e2e2ULL, 0x0075757500757575ULL,+ 0x00a9a9a900a9a9a9ULL, 0x004a4a4a004a4a4aULL, 0x0057575700575757ULL, 0x0084848400848484ULL,+ 0x0011111100111111ULL, 0x0045454500454545ULL, 0x001b1b1b001b1b1bULL, 0x00f5f5f500f5f5f5ULL,+ 0x00e4e4e400e4e4e4ULL, 0x000e0e0e000e0e0eULL, 0x0073737300737373ULL, 0x00aaaaaa00aaaaaaULL,+ 0x00f1f1f100f1f1f1ULL, 0x00dddddd00ddddddULL, 0x0059595900595959ULL, 0x0014141400141414ULL,+ 0x006c6c6c006c6c6cULL, 0x0092929200929292ULL, 0x0054545400545454ULL, 0x00d0d0d000d0d0d0ULL,+ 0x0078787800787878ULL, 0x0070707000707070ULL, 0x00e3e3e300e3e3e3ULL, 0x0049494900494949ULL,+ 0x0080808000808080ULL, 0x0050505000505050ULL, 0x00a7a7a700a7a7a7ULL, 0x00f6f6f600f6f6f6ULL,+ 0x0077777700777777ULL, 0x0093939300939393ULL, 0x0086868600868686ULL, 0x0083838300838383ULL,+ 0x002a2a2a002a2a2aULL, 0x00c7c7c700c7c7c7ULL, 0x005b5b5b005b5b5bULL, 0x00e9e9e900e9e9e9ULL,+ 0x00eeeeee00eeeeeeULL, 0x008f8f8f008f8f8fULL, 0x0001010100010101ULL, 0x003d3d3d003d3d3dULL,+},+{+ 0x3800383838003838ULL, 0x4100414141004141ULL, 0x1600161616001616ULL, 0x7600767676007676ULL,+ 0xd900d9d9d900d9d9ULL, 0x9300939393009393ULL, 0x6000606060006060ULL, 0xf200f2f2f200f2f2ULL,+ 0x7200727272007272ULL, 0xc200c2c2c200c2c2ULL, 0xab00ababab00ababULL, 0x9a009a9a9a009a9aULL,+ 0x7500757575007575ULL, 0x0600060606000606ULL, 0x5700575757005757ULL, 0xa000a0a0a000a0a0ULL,+ 0x9100919191009191ULL, 0xf700f7f7f700f7f7ULL, 0xb500b5b5b500b5b5ULL, 0xc900c9c9c900c9c9ULL,+ 0xa200a2a2a200a2a2ULL, 0x8c008c8c8c008c8cULL, 0xd200d2d2d200d2d2ULL, 0x9000909090009090ULL,+ 0xf600f6f6f600f6f6ULL, 0x0700070707000707ULL, 0xa700a7a7a700a7a7ULL, 0x2700272727002727ULL,+ 0x8e008e8e8e008e8eULL, 0xb200b2b2b200b2b2ULL, 0x4900494949004949ULL, 0xde00dedede00dedeULL,+ 0x4300434343004343ULL, 0x5c005c5c5c005c5cULL, 0xd700d7d7d700d7d7ULL, 0xc700c7c7c700c7c7ULL,+ 0x3e003e3e3e003e3eULL, 0xf500f5f5f500f5f5ULL, 0x8f008f8f8f008f8fULL, 0x6700676767006767ULL,+ 0x1f001f1f1f001f1fULL, 0x1800181818001818ULL, 0x6e006e6e6e006e6eULL, 0xaf00afafaf00afafULL,+ 0x2f002f2f2f002f2fULL, 0xe200e2e2e200e2e2ULL, 0x8500858585008585ULL, 0x0d000d0d0d000d0dULL,+ 0x5300535353005353ULL, 0xf000f0f0f000f0f0ULL, 0x9c009c9c9c009c9cULL, 0x6500656565006565ULL,+ 0xea00eaeaea00eaeaULL, 0xa300a3a3a300a3a3ULL, 0xae00aeaeae00aeaeULL, 0x9e009e9e9e009e9eULL,+ 0xec00ececec00ececULL, 0x8000808080008080ULL, 0x2d002d2d2d002d2dULL, 0x6b006b6b6b006b6bULL,+ 0xa800a8a8a800a8a8ULL, 0x2b002b2b2b002b2bULL, 0x3600363636003636ULL, 0xa600a6a6a600a6a6ULL,+ 0xc500c5c5c500c5c5ULL, 0x8600868686008686ULL, 0x4d004d4d4d004d4dULL, 0x3300333333003333ULL,+ 0xfd00fdfdfd00fdfdULL, 0x6600666666006666ULL, 0x5800585858005858ULL, 0x9600969696009696ULL,+ 0x3a003a3a3a003a3aULL, 0x0900090909000909ULL, 0x9500959595009595ULL, 0x1000101010001010ULL,+ 0x7800787878007878ULL, 0xd800d8d8d800d8d8ULL, 0x4200424242004242ULL, 0xcc00cccccc00ccccULL,+ 0xef00efefef00efefULL, 0x2600262626002626ULL, 0xe500e5e5e500e5e5ULL, 0x6100616161006161ULL,+ 0x1a001a1a1a001a1aULL, 0x3f003f3f3f003f3fULL, 0x3b003b3b3b003b3bULL, 0x8200828282008282ULL,+ 0xb600b6b6b600b6b6ULL, 0xdb00dbdbdb00dbdbULL, 0xd400d4d4d400d4d4ULL, 0x9800989898009898ULL,+ 0xe800e8e8e800e8e8ULL, 0x8b008b8b8b008b8bULL, 0x0200020202000202ULL, 0xeb00ebebeb00ebebULL,+ 0x0a000a0a0a000a0aULL, 0x2c002c2c2c002c2cULL, 0x1d001d1d1d001d1dULL, 0xb000b0b0b000b0b0ULL,+ 0x6f006f6f6f006f6fULL, 0x8d008d8d8d008d8dULL, 0x8800888888008888ULL, 0x0e000e0e0e000e0eULL,+ 0x1900191919001919ULL, 0x8700878787008787ULL, 0x4e004e4e4e004e4eULL, 0x0b000b0b0b000b0bULL,+ 0xa900a9a9a900a9a9ULL, 0x0c000c0c0c000c0cULL, 0x7900797979007979ULL, 0x1100111111001111ULL,+ 0x7f007f7f7f007f7fULL, 0x2200222222002222ULL, 0xe700e7e7e700e7e7ULL, 0x5900595959005959ULL,+ 0xe100e1e1e100e1e1ULL, 0xda00dadada00dadaULL, 0x3d003d3d3d003d3dULL, 0xc800c8c8c800c8c8ULL,+ 0x1200121212001212ULL, 0x0400040404000404ULL, 0x7400747474007474ULL, 0x5400545454005454ULL,+ 0x3000303030003030ULL, 0x7e007e7e7e007e7eULL, 0xb400b4b4b400b4b4ULL, 0x2800282828002828ULL,+ 0x5500555555005555ULL, 0x6800686868006868ULL, 0x5000505050005050ULL, 0xbe00bebebe00bebeULL,+ 0xd000d0d0d000d0d0ULL, 0xc400c4c4c400c4c4ULL, 0x3100313131003131ULL, 0xcb00cbcbcb00cbcbULL,+ 0x2a002a2a2a002a2aULL, 0xad00adadad00adadULL, 0x0f000f0f0f000f0fULL, 0xca00cacaca00cacaULL,+ 0x7000707070007070ULL, 0xff00ffffff00ffffULL, 0x3200323232003232ULL, 0x6900696969006969ULL,+ 0x0800080808000808ULL, 0x6200626262006262ULL, 0x0000000000000000ULL, 0x2400242424002424ULL,+ 0xd100d1d1d100d1d1ULL, 0xfb00fbfbfb00fbfbULL, 0xba00bababa00babaULL, 0xed00ededed00ededULL,+ 0x4500454545004545ULL, 0x8100818181008181ULL, 0x7300737373007373ULL, 0x6d006d6d6d006d6dULL,+ 0x8400848484008484ULL, 0x9f009f9f9f009f9fULL, 0xee00eeeeee00eeeeULL, 0x4a004a4a4a004a4aULL,+ 0xc300c3c3c300c3c3ULL, 0x2e002e2e2e002e2eULL, 0xc100c1c1c100c1c1ULL, 0x0100010101000101ULL,+ 0xe600e6e6e600e6e6ULL, 0x2500252525002525ULL, 0x4800484848004848ULL, 0x9900999999009999ULL,+ 0xb900b9b9b900b9b9ULL, 0xb300b3b3b300b3b3ULL, 0x7b007b7b7b007b7bULL, 0xf900f9f9f900f9f9ULL,+ 0xce00cecece00ceceULL, 0xbf00bfbfbf00bfbfULL, 0xdf00dfdfdf00dfdfULL, 0x7100717171007171ULL,+ 0x2900292929002929ULL, 0xcd00cdcdcd00cdcdULL, 0x6c006c6c6c006c6cULL, 0x1300131313001313ULL,+ 0x6400646464006464ULL, 0x9b009b9b9b009b9bULL, 0x6300636363006363ULL, 0x9d009d9d9d009d9dULL,+ 0xc000c0c0c000c0c0ULL, 0x4b004b4b4b004b4bULL, 0xb700b7b7b700b7b7ULL, 0xa500a5a5a500a5a5ULL,+ 0x8900898989008989ULL, 0x5f005f5f5f005f5fULL, 0xb100b1b1b100b1b1ULL, 0x1700171717001717ULL,+ 0xf400f4f4f400f4f4ULL, 0xbc00bcbcbc00bcbcULL, 0xd300d3d3d300d3d3ULL, 0x4600464646004646ULL,+ 0xcf00cfcfcf00cfcfULL, 0x3700373737003737ULL, 0x5e005e5e5e005e5eULL, 0x4700474747004747ULL,+ 0x9400949494009494ULL, 0xfa00fafafa00fafaULL, 0xfc00fcfcfc00fcfcULL, 0x5b005b5b5b005b5bULL,+ 0x9700979797009797ULL, 0xfe00fefefe00fefeULL, 0x5a005a5a5a005a5aULL, 0xac00acacac00acacULL,+ 0x3c003c3c3c003c3cULL, 0x4c004c4c4c004c4cULL, 0x0300030303000303ULL, 0x3500353535003535ULL,+ 0xf300f3f3f300f3f3ULL, 0x2300232323002323ULL, 0xb800b8b8b800b8b8ULL, 0x5d005d5d5d005d5dULL,+ 0x6a006a6a6a006a6aULL, 0x9200929292009292ULL, 0xd500d5d5d500d5d5ULL, 0x2100212121002121ULL,+ 0x4400444444004444ULL, 0x5100515151005151ULL, 0xc600c6c6c600c6c6ULL, 0x7d007d7d7d007d7dULL,+ 0x3900393939003939ULL, 0x8300838383008383ULL, 0xdc00dcdcdc00dcdcULL, 0xaa00aaaaaa00aaaaULL,+ 0x7c007c7c7c007c7cULL, 0x7700777777007777ULL, 0x5600565656005656ULL, 0x0500050505000505ULL,+ 0x1b001b1b1b001b1bULL, 0xa400a4a4a400a4a4ULL, 0x1500151515001515ULL, 0x3400343434003434ULL,+ 0x1e001e1e1e001e1eULL, 0x1c001c1c1c001c1cULL, 0xf800f8f8f800f8f8ULL, 0x5200525252005252ULL,+ 0x2000202020002020ULL, 0x1400141414001414ULL, 0xe900e9e9e900e9e9ULL, 0xbd00bdbdbd00bdbdULL,+ 0xdd00dddddd00ddddULL, 0xe400e4e4e400e4e4ULL, 0xa100a1a1a100a1a1ULL, 0xe000e0e0e000e0e0ULL,+ 0x8a008a8a8a008a8aULL, 0xf100f1f1f100f1f1ULL, 0xd600d6d6d600d6d6ULL, 0x7a007a7a7a007a7aULL,+ 0xbb00bbbbbb00bbbbULL, 0xe300e3e3e300e3e3ULL, 0x4000404040004040ULL, 0x4f004f4f4f004f4fULL,+},+{+ 0x7070007070700070ULL, 0x2c2c002c2c2c002cULL, 0xb3b300b3b3b300b3ULL, 0xc0c000c0c0c000c0ULL,+ 0xe4e400e4e4e400e4ULL, 0x5757005757570057ULL, 0xeaea00eaeaea00eaULL, 0xaeae00aeaeae00aeULL,+ 0x2323002323230023ULL, 0x6b6b006b6b6b006bULL, 0x4545004545450045ULL, 0xa5a500a5a5a500a5ULL,+ 0xeded00ededed00edULL, 0x4f4f004f4f4f004fULL, 0x1d1d001d1d1d001dULL, 0x9292009292920092ULL,+ 0x8686008686860086ULL, 0xafaf00afafaf00afULL, 0x7c7c007c7c7c007cULL, 0x1f1f001f1f1f001fULL,+ 0x3e3e003e3e3e003eULL, 0xdcdc00dcdcdc00dcULL, 0x5e5e005e5e5e005eULL, 0x0b0b000b0b0b000bULL,+ 0xa6a600a6a6a600a6ULL, 0x3939003939390039ULL, 0xd5d500d5d5d500d5ULL, 0x5d5d005d5d5d005dULL,+ 0xd9d900d9d9d900d9ULL, 0x5a5a005a5a5a005aULL, 0x5151005151510051ULL, 0x6c6c006c6c6c006cULL,+ 0x8b8b008b8b8b008bULL, 0x9a9a009a9a9a009aULL, 0xfbfb00fbfbfb00fbULL, 0xb0b000b0b0b000b0ULL,+ 0x7474007474740074ULL, 0x2b2b002b2b2b002bULL, 0xf0f000f0f0f000f0ULL, 0x8484008484840084ULL,+ 0xdfdf00dfdfdf00dfULL, 0xcbcb00cbcbcb00cbULL, 0x3434003434340034ULL, 0x7676007676760076ULL,+ 0x6d6d006d6d6d006dULL, 0xa9a900a9a9a900a9ULL, 0xd1d100d1d1d100d1ULL, 0x0404000404040004ULL,+ 0x1414001414140014ULL, 0x3a3a003a3a3a003aULL, 0xdede00dedede00deULL, 0x1111001111110011ULL,+ 0x3232003232320032ULL, 0x9c9c009c9c9c009cULL, 0x5353005353530053ULL, 0xf2f200f2f2f200f2ULL,+ 0xfefe00fefefe00feULL, 0xcfcf00cfcfcf00cfULL, 0xc3c300c3c3c300c3ULL, 0x7a7a007a7a7a007aULL,+ 0x2424002424240024ULL, 0xe8e800e8e8e800e8ULL, 0x6060006060600060ULL, 0x6969006969690069ULL,+ 0xaaaa00aaaaaa00aaULL, 0xa0a000a0a0a000a0ULL, 0xa1a100a1a1a100a1ULL, 0x6262006262620062ULL,+ 0x5454005454540054ULL, 0x1e1e001e1e1e001eULL, 0xe0e000e0e0e000e0ULL, 0x6464006464640064ULL,+ 0x1010001010100010ULL, 0x0000000000000000ULL, 0xa3a300a3a3a300a3ULL, 0x7575007575750075ULL,+ 0x8a8a008a8a8a008aULL, 0xe6e600e6e6e600e6ULL, 0x0909000909090009ULL, 0xdddd00dddddd00ddULL,+ 0x8787008787870087ULL, 0x8383008383830083ULL, 0xcdcd00cdcdcd00cdULL, 0x9090009090900090ULL,+ 0x7373007373730073ULL, 0xf6f600f6f6f600f6ULL, 0x9d9d009d9d9d009dULL, 0xbfbf00bfbfbf00bfULL,+ 0x5252005252520052ULL, 0xd8d800d8d8d800d8ULL, 0xc8c800c8c8c800c8ULL, 0xc6c600c6c6c600c6ULL,+ 0x8181008181810081ULL, 0x6f6f006f6f6f006fULL, 0x1313001313130013ULL, 0x6363006363630063ULL,+ 0xe9e900e9e9e900e9ULL, 0xa7a700a7a7a700a7ULL, 0x9f9f009f9f9f009fULL, 0xbcbc00bcbcbc00bcULL,+ 0x2929002929290029ULL, 0xf9f900f9f9f900f9ULL, 0x2f2f002f2f2f002fULL, 0xb4b400b4b4b400b4ULL,+ 0x7878007878780078ULL, 0x0606000606060006ULL, 0xe7e700e7e7e700e7ULL, 0x7171007171710071ULL,+ 0xd4d400d4d4d400d4ULL, 0xabab00ababab00abULL, 0x8888008888880088ULL, 0x8d8d008d8d8d008dULL,+ 0x7272007272720072ULL, 0xb9b900b9b9b900b9ULL, 0xf8f800f8f8f800f8ULL, 0xacac00acacac00acULL,+ 0x3636003636360036ULL, 0x2a2a002a2a2a002aULL, 0x3c3c003c3c3c003cULL, 0xf1f100f1f1f100f1ULL,+ 0x4040004040400040ULL, 0xd3d300d3d3d300d3ULL, 0xbbbb00bbbbbb00bbULL, 0x4343004343430043ULL,+ 0x1515001515150015ULL, 0xadad00adadad00adULL, 0x7777007777770077ULL, 0x8080008080800080ULL,+ 0x8282008282820082ULL, 0xecec00ececec00ecULL, 0x2727002727270027ULL, 0xe5e500e5e5e500e5ULL,+ 0x8585008585850085ULL, 0x3535003535350035ULL, 0x0c0c000c0c0c000cULL, 0x4141004141410041ULL,+ 0xefef00efefef00efULL, 0x9393009393930093ULL, 0x1919001919190019ULL, 0x2121002121210021ULL,+ 0x0e0e000e0e0e000eULL, 0x4e4e004e4e4e004eULL, 0x6565006565650065ULL, 0xbdbd00bdbdbd00bdULL,+ 0xb8b800b8b8b800b8ULL, 0x8f8f008f8f8f008fULL, 0xebeb00ebebeb00ebULL, 0xcece00cecece00ceULL,+ 0x3030003030300030ULL, 0x5f5f005f5f5f005fULL, 0xc5c500c5c5c500c5ULL, 0x1a1a001a1a1a001aULL,+ 0xe1e100e1e1e100e1ULL, 0xcaca00cacaca00caULL, 0x4747004747470047ULL, 0x3d3d003d3d3d003dULL,+ 0x0101000101010001ULL, 0xd6d600d6d6d600d6ULL, 0x5656005656560056ULL, 0x4d4d004d4d4d004dULL,+ 0x0d0d000d0d0d000dULL, 0x6666006666660066ULL, 0xcccc00cccccc00ccULL, 0x2d2d002d2d2d002dULL,+ 0x1212001212120012ULL, 0x2020002020200020ULL, 0xb1b100b1b1b100b1ULL, 0x9999009999990099ULL,+ 0x4c4c004c4c4c004cULL, 0xc2c200c2c2c200c2ULL, 0x7e7e007e7e7e007eULL, 0x0505000505050005ULL,+ 0xb7b700b7b7b700b7ULL, 0x3131003131310031ULL, 0x1717001717170017ULL, 0xd7d700d7d7d700d7ULL,+ 0x5858005858580058ULL, 0x6161006161610061ULL, 0x1b1b001b1b1b001bULL, 0x1c1c001c1c1c001cULL,+ 0x0f0f000f0f0f000fULL, 0x1616001616160016ULL, 0x1818001818180018ULL, 0x2222002222220022ULL,+ 0x4444004444440044ULL, 0xb2b200b2b2b200b2ULL, 0xb5b500b5b5b500b5ULL, 0x9191009191910091ULL,+ 0x0808000808080008ULL, 0xa8a800a8a8a800a8ULL, 0xfcfc00fcfcfc00fcULL, 0x5050005050500050ULL,+ 0xd0d000d0d0d000d0ULL, 0x7d7d007d7d7d007dULL, 0x8989008989890089ULL, 0x9797009797970097ULL,+ 0x5b5b005b5b5b005bULL, 0x9595009595950095ULL, 0xffff00ffffff00ffULL, 0xd2d200d2d2d200d2ULL,+ 0xc4c400c4c4c400c4ULL, 0x4848004848480048ULL, 0xf7f700f7f7f700f7ULL, 0xdbdb00dbdbdb00dbULL,+ 0x0303000303030003ULL, 0xdada00dadada00daULL, 0x3f3f003f3f3f003fULL, 0x9494009494940094ULL,+ 0x5c5c005c5c5c005cULL, 0x0202000202020002ULL, 0x4a4a004a4a4a004aULL, 0x3333003333330033ULL,+ 0x6767006767670067ULL, 0xf3f300f3f3f300f3ULL, 0x7f7f007f7f7f007fULL, 0xe2e200e2e2e200e2ULL,+ 0x9b9b009b9b9b009bULL, 0x2626002626260026ULL, 0x3737003737370037ULL, 0x3b3b003b3b3b003bULL,+ 0x9696009696960096ULL, 0x4b4b004b4b4b004bULL, 0xbebe00bebebe00beULL, 0x2e2e002e2e2e002eULL,+ 0x7979007979790079ULL, 0x8c8c008c8c8c008cULL, 0x6e6e006e6e6e006eULL, 0x8e8e008e8e8e008eULL,+ 0xf5f500f5f5f500f5ULL, 0xb6b600b6b6b600b6ULL, 0xfdfd00fdfdfd00fdULL, 0x5959005959590059ULL,+ 0x9898009898980098ULL, 0x6a6a006a6a6a006aULL, 0x4646004646460046ULL, 0xbaba00bababa00baULL,+ 0x2525002525250025ULL, 0x4242004242420042ULL, 0xa2a200a2a2a200a2ULL, 0xfafa00fafafa00faULL,+ 0x0707000707070007ULL, 0x5555005555550055ULL, 0xeeee00eeeeee00eeULL, 0x0a0a000a0a0a000aULL,+ 0x4949004949490049ULL, 0x6868006868680068ULL, 0x3838003838380038ULL, 0xa4a400a4a4a400a4ULL,+ 0x2828002828280028ULL, 0x7b7b007b7b7b007bULL, 0xc9c900c9c9c900c9ULL, 0xc1c100c1c1c100c1ULL,+ 0xe3e300e3e3e300e3ULL, 0xf4f400f4f4f400f4ULL, 0xc7c700c7c7c700c7ULL, 0x9e9e009e9e9e009eULL,+},+{+ 0x7070700070707000ULL, 0x8282820082828200ULL, 0x2c2c2c002c2c2c00ULL, 0xececec00ececec00ULL,+ 0xb3b3b300b3b3b300ULL, 0x2727270027272700ULL, 0xc0c0c000c0c0c000ULL, 0xe5e5e500e5e5e500ULL,+ 0xe4e4e400e4e4e400ULL, 0x8585850085858500ULL, 0x5757570057575700ULL, 0x3535350035353500ULL,+ 0xeaeaea00eaeaea00ULL, 0x0c0c0c000c0c0c00ULL, 0xaeaeae00aeaeae00ULL, 0x4141410041414100ULL,+ 0x2323230023232300ULL, 0xefefef00efefef00ULL, 0x6b6b6b006b6b6b00ULL, 0x9393930093939300ULL,+ 0x4545450045454500ULL, 0x1919190019191900ULL, 0xa5a5a500a5a5a500ULL, 0x2121210021212100ULL,+ 0xededed00ededed00ULL, 0x0e0e0e000e0e0e00ULL, 0x4f4f4f004f4f4f00ULL, 0x4e4e4e004e4e4e00ULL,+ 0x1d1d1d001d1d1d00ULL, 0x6565650065656500ULL, 0x9292920092929200ULL, 0xbdbdbd00bdbdbd00ULL,+ 0x8686860086868600ULL, 0xb8b8b800b8b8b800ULL, 0xafafaf00afafaf00ULL, 0x8f8f8f008f8f8f00ULL,+ 0x7c7c7c007c7c7c00ULL, 0xebebeb00ebebeb00ULL, 0x1f1f1f001f1f1f00ULL, 0xcecece00cecece00ULL,+ 0x3e3e3e003e3e3e00ULL, 0x3030300030303000ULL, 0xdcdcdc00dcdcdc00ULL, 0x5f5f5f005f5f5f00ULL,+ 0x5e5e5e005e5e5e00ULL, 0xc5c5c500c5c5c500ULL, 0x0b0b0b000b0b0b00ULL, 0x1a1a1a001a1a1a00ULL,+ 0xa6a6a600a6a6a600ULL, 0xe1e1e100e1e1e100ULL, 0x3939390039393900ULL, 0xcacaca00cacaca00ULL,+ 0xd5d5d500d5d5d500ULL, 0x4747470047474700ULL, 0x5d5d5d005d5d5d00ULL, 0x3d3d3d003d3d3d00ULL,+ 0xd9d9d900d9d9d900ULL, 0x0101010001010100ULL, 0x5a5a5a005a5a5a00ULL, 0xd6d6d600d6d6d600ULL,+ 0x5151510051515100ULL, 0x5656560056565600ULL, 0x6c6c6c006c6c6c00ULL, 0x4d4d4d004d4d4d00ULL,+ 0x8b8b8b008b8b8b00ULL, 0x0d0d0d000d0d0d00ULL, 0x9a9a9a009a9a9a00ULL, 0x6666660066666600ULL,+ 0xfbfbfb00fbfbfb00ULL, 0xcccccc00cccccc00ULL, 0xb0b0b000b0b0b000ULL, 0x2d2d2d002d2d2d00ULL,+ 0x7474740074747400ULL, 0x1212120012121200ULL, 0x2b2b2b002b2b2b00ULL, 0x2020200020202000ULL,+ 0xf0f0f000f0f0f000ULL, 0xb1b1b100b1b1b100ULL, 0x8484840084848400ULL, 0x9999990099999900ULL,+ 0xdfdfdf00dfdfdf00ULL, 0x4c4c4c004c4c4c00ULL, 0xcbcbcb00cbcbcb00ULL, 0xc2c2c200c2c2c200ULL,+ 0x3434340034343400ULL, 0x7e7e7e007e7e7e00ULL, 0x7676760076767600ULL, 0x0505050005050500ULL,+ 0x6d6d6d006d6d6d00ULL, 0xb7b7b700b7b7b700ULL, 0xa9a9a900a9a9a900ULL, 0x3131310031313100ULL,+ 0xd1d1d100d1d1d100ULL, 0x1717170017171700ULL, 0x0404040004040400ULL, 0xd7d7d700d7d7d700ULL,+ 0x1414140014141400ULL, 0x5858580058585800ULL, 0x3a3a3a003a3a3a00ULL, 0x6161610061616100ULL,+ 0xdedede00dedede00ULL, 0x1b1b1b001b1b1b00ULL, 0x1111110011111100ULL, 0x1c1c1c001c1c1c00ULL,+ 0x3232320032323200ULL, 0x0f0f0f000f0f0f00ULL, 0x9c9c9c009c9c9c00ULL, 0x1616160016161600ULL,+ 0x5353530053535300ULL, 0x1818180018181800ULL, 0xf2f2f200f2f2f200ULL, 0x2222220022222200ULL,+ 0xfefefe00fefefe00ULL, 0x4444440044444400ULL, 0xcfcfcf00cfcfcf00ULL, 0xb2b2b200b2b2b200ULL,+ 0xc3c3c300c3c3c300ULL, 0xb5b5b500b5b5b500ULL, 0x7a7a7a007a7a7a00ULL, 0x9191910091919100ULL,+ 0x2424240024242400ULL, 0x0808080008080800ULL, 0xe8e8e800e8e8e800ULL, 0xa8a8a800a8a8a800ULL,+ 0x6060600060606000ULL, 0xfcfcfc00fcfcfc00ULL, 0x6969690069696900ULL, 0x5050500050505000ULL,+ 0xaaaaaa00aaaaaa00ULL, 0xd0d0d000d0d0d000ULL, 0xa0a0a000a0a0a000ULL, 0x7d7d7d007d7d7d00ULL,+ 0xa1a1a100a1a1a100ULL, 0x8989890089898900ULL, 0x6262620062626200ULL, 0x9797970097979700ULL,+ 0x5454540054545400ULL, 0x5b5b5b005b5b5b00ULL, 0x1e1e1e001e1e1e00ULL, 0x9595950095959500ULL,+ 0xe0e0e000e0e0e000ULL, 0xffffff00ffffff00ULL, 0x6464640064646400ULL, 0xd2d2d200d2d2d200ULL,+ 0x1010100010101000ULL, 0xc4c4c400c4c4c400ULL, 0x0000000000000000ULL, 0x4848480048484800ULL,+ 0xa3a3a300a3a3a300ULL, 0xf7f7f700f7f7f700ULL, 0x7575750075757500ULL, 0xdbdbdb00dbdbdb00ULL,+ 0x8a8a8a008a8a8a00ULL, 0x0303030003030300ULL, 0xe6e6e600e6e6e600ULL, 0xdadada00dadada00ULL,+ 0x0909090009090900ULL, 0x3f3f3f003f3f3f00ULL, 0xdddddd00dddddd00ULL, 0x9494940094949400ULL,+ 0x8787870087878700ULL, 0x5c5c5c005c5c5c00ULL, 0x8383830083838300ULL, 0x0202020002020200ULL,+ 0xcdcdcd00cdcdcd00ULL, 0x4a4a4a004a4a4a00ULL, 0x9090900090909000ULL, 0x3333330033333300ULL,+ 0x7373730073737300ULL, 0x6767670067676700ULL, 0xf6f6f600f6f6f600ULL, 0xf3f3f300f3f3f300ULL,+ 0x9d9d9d009d9d9d00ULL, 0x7f7f7f007f7f7f00ULL, 0xbfbfbf00bfbfbf00ULL, 0xe2e2e200e2e2e200ULL,+ 0x5252520052525200ULL, 0x9b9b9b009b9b9b00ULL, 0xd8d8d800d8d8d800ULL, 0x2626260026262600ULL,+ 0xc8c8c800c8c8c800ULL, 0x3737370037373700ULL, 0xc6c6c600c6c6c600ULL, 0x3b3b3b003b3b3b00ULL,+ 0x8181810081818100ULL, 0x9696960096969600ULL, 0x6f6f6f006f6f6f00ULL, 0x4b4b4b004b4b4b00ULL,+ 0x1313130013131300ULL, 0xbebebe00bebebe00ULL, 0x6363630063636300ULL, 0x2e2e2e002e2e2e00ULL,+ 0xe9e9e900e9e9e900ULL, 0x7979790079797900ULL, 0xa7a7a700a7a7a700ULL, 0x8c8c8c008c8c8c00ULL,+ 0x9f9f9f009f9f9f00ULL, 0x6e6e6e006e6e6e00ULL, 0xbcbcbc00bcbcbc00ULL, 0x8e8e8e008e8e8e00ULL,+ 0x2929290029292900ULL, 0xf5f5f500f5f5f500ULL, 0xf9f9f900f9f9f900ULL, 0xb6b6b600b6b6b600ULL,+ 0x2f2f2f002f2f2f00ULL, 0xfdfdfd00fdfdfd00ULL, 0xb4b4b400b4b4b400ULL, 0x5959590059595900ULL,+ 0x7878780078787800ULL, 0x9898980098989800ULL, 0x0606060006060600ULL, 0x6a6a6a006a6a6a00ULL,+ 0xe7e7e700e7e7e700ULL, 0x4646460046464600ULL, 0x7171710071717100ULL, 0xbababa00bababa00ULL,+ 0xd4d4d400d4d4d400ULL, 0x2525250025252500ULL, 0xababab00ababab00ULL, 0x4242420042424200ULL,+ 0x8888880088888800ULL, 0xa2a2a200a2a2a200ULL, 0x8d8d8d008d8d8d00ULL, 0xfafafa00fafafa00ULL,+ 0x7272720072727200ULL, 0x0707070007070700ULL, 0xb9b9b900b9b9b900ULL, 0x5555550055555500ULL,+ 0xf8f8f800f8f8f800ULL, 0xeeeeee00eeeeee00ULL, 0xacacac00acacac00ULL, 0x0a0a0a000a0a0a00ULL,+ 0x3636360036363600ULL, 0x4949490049494900ULL, 0x2a2a2a002a2a2a00ULL, 0x6868680068686800ULL,+ 0x3c3c3c003c3c3c00ULL, 0x3838380038383800ULL, 0xf1f1f100f1f1f100ULL, 0xa4a4a400a4a4a400ULL,+ 0x4040400040404000ULL, 0x2828280028282800ULL, 0xd3d3d300d3d3d300ULL, 0x7b7b7b007b7b7b00ULL,+ 0xbbbbbb00bbbbbb00ULL, 0xc9c9c900c9c9c900ULL, 0x4343430043434300ULL, 0xc1c1c100c1c1c100ULL,+ 0x1515150015151500ULL, 0xe3e3e300e3e3e300ULL, 0xadadad00adadad00ULL, 0xf4f4f400f4f4f400ULL,+ 0x7777770077777700ULL, 0xc7c7c700c7c7c700ULL, 0x8080800080808000ULL, 0x9e9e9e009e9e9e00ULL,+},+};++static const uint64_t SIGMA[6] = {+ 0xA09E667F3BCC908BULL, 0xB67AE8584CAA73B2ULL, 0xC6EF372FE94F82BEULL,+ 0x54FF53A5F1D36F1CULL, 0x10E527FADE682D1DULL, 0xB05688C2B3E6C1FDULL+};++static inline uint64_t load_be64(const uint8_t *p)+{+ return ((uint64_t) p[0] << 56) | ((uint64_t) p[1] << 48)+ | ((uint64_t) p[2] << 40) | ((uint64_t) p[3] << 32)+ | ((uint64_t) p[4] << 24) | ((uint64_t) p[5] << 16)+ | ((uint64_t) p[6] << 8) | ((uint64_t) p[7]);+}++static inline void store_be64(uint8_t *p, uint64_t v)+{+ p[0] = (uint8_t) (v >> 56); p[1] = (uint8_t) (v >> 48);+ p[2] = (uint8_t) (v >> 40); p[3] = (uint8_t) (v >> 32);+ p[4] = (uint8_t) (v >> 24); p[5] = (uint8_t) (v >> 16);+ p[6] = (uint8_t) (v >> 8); p[7] = (uint8_t) v;+}++static inline uint64_t camellia_f(uint64_t fin, uint64_t ke)+{+ uint64_t x = fin ^ ke;+ return SP[0][(x >> 56) & 0xff] ^ SP[1][(x >> 48) & 0xff]+ ^ SP[2][(x >> 40) & 0xff] ^ SP[3][(x >> 32) & 0xff]+ ^ SP[4][(x >> 24) & 0xff] ^ SP[5][(x >> 16) & 0xff]+ ^ SP[6][(x >> 8) & 0xff] ^ SP[7][ x & 0xff];+}++static inline uint32_t rotl32(uint32_t v, int n)+{+ return (v << n) | (v >> (32 - n));+}++static inline uint64_t camellia_fl(uint64_t fin, uint64_t ke)+{+ uint32_t x1 = (uint32_t) (fin >> 32), x2 = (uint32_t) fin;+ uint32_t k1 = (uint32_t) (ke >> 32), k2 = (uint32_t) ke;++ x2 ^= rotl32(x1 & k1, 1);+ x1 ^= (x2 | k2);+ return ((uint64_t) x1 << 32) | x2;+}++static inline uint64_t camellia_flinv(uint64_t fin, uint64_t ke)+{+ uint32_t y1 = (uint32_t) (fin >> 32), y2 = (uint32_t) fin;+ uint32_t k1 = (uint32_t) (ke >> 32), k2 = (uint32_t) ke;++ y1 ^= (y2 | k2);+ y2 ^= rotl32(y1 & k1, 1);+ return ((uint64_t) y1 << 32) | y2;+}++/* the halves of a 128-bit value rotated left by n, 0 < n < 128 */+static void rotl128(uint64_t hi, uint64_t lo, int n, uint64_t *rhi, uint64_t *rlo)+{+ if (n >= 64) {+ uint64_t t = hi;+ hi = lo;+ lo = t;+ n -= 64;+ }+ if (n == 0) {+ *rhi = hi;+ *rlo = lo;+ } else {+ *rhi = (hi << n) | (lo >> (64 - n));+ *rlo = (lo << n) | (hi >> (64 - n));+ }+}++void crypton_camellia_init(crypton_camellia_key *ks, const uint8_t *key)+{+ uint64_t klhi = load_be64(key), kllo = load_be64(key + 8);+ uint64_t d1 = klhi, d2 = kllo, kahi, kalo, hi, lo;++ d2 ^= camellia_f(d1, SIGMA[0]);+ d1 ^= camellia_f(d2, SIGMA[1]);+ d1 ^= klhi;+ d2 ^= kllo;+ d2 ^= camellia_f(d1, SIGMA[2]);+ d1 ^= camellia_f(d2, SIGMA[3]);+ kahi = d1;+ kalo = d2;++ ks->kw[0] = klhi;+ ks->kw[1] = kllo;+ ks->k[0] = kahi;+ ks->k[1] = kalo;+ rotl128(klhi, kllo, 15, &hi, &lo); ks->k[2] = hi; ks->k[3] = lo;+ rotl128(kahi, kalo, 15, &hi, &lo); ks->k[4] = hi; ks->k[5] = lo;+ rotl128(kahi, kalo, 30, &hi, &lo); ks->ke[0] = hi; ks->ke[1] = lo;+ rotl128(klhi, kllo, 45, &hi, &lo); ks->k[6] = hi; ks->k[7] = lo;+ rotl128(kahi, kalo, 45, &hi, &lo); ks->k[8] = hi;+ rotl128(klhi, kllo, 60, &hi, &lo); ks->k[9] = lo;+ rotl128(kahi, kalo, 60, &hi, &lo); ks->k[10] = hi; ks->k[11] = lo;+ rotl128(klhi, kllo, 77, &hi, &lo); ks->ke[2] = hi; ks->ke[3] = lo;+ rotl128(klhi, kllo, 94, &hi, &lo); ks->k[12] = hi; ks->k[13] = lo;+ rotl128(kahi, kalo, 94, &hi, &lo); ks->k[14] = hi; ks->k[15] = lo;+ rotl128(klhi, kllo, 111, &hi, &lo); ks->k[16] = hi; ks->k[17] = lo;+ rotl128(kahi, kalo, 111, &hi, &lo); ks->kw[2] = hi; ks->kw[3] = lo;+}++static void camellia_crypt(uint8_t *out, const uint64_t kw[4], const uint64_t k[18],+ const uint64_t ke[4], const uint8_t *in, uint32_t nblocks)+{+ uint32_t i;++ for (i = 0; i < nblocks; i++) {+ uint64_t d1 = load_be64(in + 16 * i) ^ kw[0];+ uint64_t d2 = load_be64(in + 16 * i + 8) ^ kw[1];+ int base;++ for (base = 0; base <= 12; base += 6) {+ d2 ^= camellia_f(d1, k[base + 0]);+ d1 ^= camellia_f(d2, k[base + 1]);+ d2 ^= camellia_f(d1, k[base + 2]);+ d1 ^= camellia_f(d2, k[base + 3]);+ d2 ^= camellia_f(d1, k[base + 4]);+ d1 ^= camellia_f(d2, k[base + 5]);+ if (base == 0) {+ d1 = camellia_fl(d1, ke[0]);+ d2 = camellia_flinv(d2, ke[1]);+ } else if (base == 6) {+ d1 = camellia_fl(d1, ke[2]);+ d2 = camellia_flinv(d2, ke[3]);+ }+ }++ store_be64(out + 16 * i, d2 ^ kw[2]);+ store_be64(out + 16 * i + 8, d1 ^ kw[3]);+ }+}++void crypton_camellia_encrypt(uint8_t *out, const crypton_camellia_key *ks,+ const uint8_t *in, uint32_t nblocks)+{+ camellia_crypt(out, ks->kw, ks->k, ks->ke, in, nblocks);+}++/* Decryption is the same rounds with the subkeys the other way round. */+void crypton_camellia_decrypt(uint8_t *out, const crypton_camellia_key *ks,+ const uint8_t *in, uint32_t nblocks)+{+ uint64_t kw[4], k[18], ke[4];+ int i;++ kw[0] = ks->kw[2]; kw[1] = ks->kw[3]; kw[2] = ks->kw[0]; kw[3] = ks->kw[1];+ for (i = 0; i < 18; i++)+ k[i] = ks->k[17 - i];+ for (i = 0; i < 4; i++)+ ke[i] = ks->ke[3 - i];+ camellia_crypt(out, kw, k, ke, in, nblocks);+}
@@ -0,0 +1,21 @@+#ifndef CRYPTON_CAMELLIA_H+#define CRYPTON_CAMELLIA_H++#include <stdint.h>++/* the subkeys of RFC 3713 section 2.2, for a 128-bit key */+typedef struct {+ uint64_t kw[4];+ uint64_t k[18];+ uint64_t ke[4];+} crypton_camellia_key;++void crypton_camellia_init(crypton_camellia_key *ks, const uint8_t *key);++void crypton_camellia_encrypt(uint8_t *out, const crypton_camellia_key *ks,+ const uint8_t *in, uint32_t nblocks);++void crypton_camellia_decrypt(uint8_t *out, const crypton_camellia_key *ks,+ const uint8_t *in, uint32_t nblocks);++#endif
@@ -35,6 +35,71 @@ #include "crypton_align.h" #include <stdio.h> +/*+ * Four blocks at a time with whichever vector unit the target has: NEON in+ * chacha_neon.c, SSE2 in chacha_sse2.c. Both present the same two entry+ * points, so there is one path here.+ *+ * The state words are held little-endian -- the core below reads them+ * without converting -- so the vector versions, which also do not convert,+ * are left out on a big-endian machine.+ */+#if (defined(WITH_ARMV8_NEON) && !defined(__AARCH64EB__)) || defined(WITH_X86_SSE2)+#define CHACHA_SIMD 1+int crypton_chacha_simd_width(void);+void crypton_chacha_simd_combine(int rounds, uint8_t *dst, const uint8_t *src,+ const crypton_chacha_state *in);+void crypton_chacha_simd_generate(int rounds, uint8_t *dst,+ const crypton_chacha_state *in);+/* The counters in a group must not carry into d[13], which the crypton_chacha_block loop+ * below handles and the vector one does not; that is one run in 2^29. */+#define CHACHA_SIMD_OK(st, n) ((st)->d[12] <= 0xffffffffU - (uint32_t) (n))+#endif++/*+ * ChaCha20 from CRYPTOGAMS, in cbits/asm/chacha-armv8-*.S. It keeps four+ * vector blocks and a fifth in the general registers in flight at once, or+ * six and two above 512 bytes, which is more than the intrinsics above can+ * be made to do: the vector registers hold four states and there is no room+ * for another, so the extra parallelism has to come from the integer side,+ * and that means saying which register holds what.+ *+ * Twenty rounds and the 256-bit constants are built into it, and it takes+ * the counter as 32 bits wide, so it is given only the states it fits.+ */+#if (defined(WITH_ARMV8_CHACHA_ASM) && !defined(__AARCH64EB__)) \+ || defined(WITH_X86_CHACHA_ASM)+#define CHACHA_ASM 1+#include "crypton_cpu.h"+void crypton_chacha20_asm_ctr32(uint8_t *out, const uint8_t *in, size_t len,+ const uint32_t key[8], const uint32_t counter[4]);++/* crypton_cpu.c defines the crypton_armcap_P that the assembly reads to+ * find out whether the processor has NEON. */++/* The four words at the head of the state are the constants that go with a+ * 256-bit key, and the assembly has only those. */+static int chacha_asm_state(const crypton_chacha_state *st)+{+ return st->d[0] == 0x61707865 && st->d[1] == 0x3320646e+ && st->d[2] == 0x79622d32 && st->d[3] == 0x6b206574;+}++/*+ * How much is worth handing over. On AArch64 the module's vector path+ * starts at three blocks and below that its scalar path measures level with+ * the C here, so there is nothing to gain; on x86-64 it is ahead from one+ * crypton_chacha_block, the C there having no vector path until eight.+ */+#ifndef CHACHA_ASM_MIN_BLOCKS+#ifdef WITH_X86_CHACHA_ASM+#define CHACHA_ASM_MIN_BLOCKS 1+#else+#define CHACHA_ASM_MIN_BLOCKS 3+#endif+#endif+#endif+ #define QR(a,b,c,d) \ a += b; d = rol32(d ^ a,16); \ c += d; b = rol32(b ^ c,12); \@@ -47,7 +112,7 @@ static const uint8_t sigma[16] = "expand 32-byte k"; static const uint8_t tau[16] = "expand 16-byte k"; -static void chacha_core(int rounds, block *out, const crypton_chacha_state *in)+static void chacha_core(int rounds, crypton_chacha_block *out, const crypton_chacha_state *in) { uint32_t x0, x1, x2, x3, x4, x5, x6, x7, x8, x9, x10, x11, x12, x13, x14, x15; int i;@@ -231,7 +296,7 @@ void crypton_chacha_combine(uint8_t *dst, crypton_chacha_context *ctx, const uint8_t *src, uint32_t bytes) {- block out;+ crypton_chacha_block out; crypton_chacha_state *st; int i; @@ -256,6 +321,43 @@ st = &ctx->st; +#ifdef CHACHA_ASM+ if (ctx->nb_rounds == 20 && chacha_asm_state(st)) {+ uint32_t blocks = bytes / 64;++ /* the counter is the caller's to advance, and the assembly+ * carries it no further than its own 32 bits */+ if (blocks > 0xffffffffU - st->d[12])+ blocks = 0xffffffffU - st->d[12];+ if (blocks >= CHACHA_ASM_MIN_BLOCKS) {+ const uint32_t done = blocks * 64;++#ifdef CRYPTON_X86_ASM+ /* what the module dispatches on, which it reads+ * directly; resolved once */+ crypton_x86_ia32cap_resolve();+#endif+ crypton_chacha20_asm_ctr32(dst, src, done, &st->d[4],+ &st->d[12]);+ st->d[12] += blocks;+ bytes -= done; src += done; dst += done;+ }+ }+#endif++#ifdef CHACHA_SIMD+ {+ const uint32_t nb = (uint32_t) crypton_chacha_simd_width();+ const uint32_t step = 64 * nb;++ while (bytes >= step && CHACHA_SIMD_OK(st, nb)) {+ crypton_chacha_simd_combine(ctx->nb_rounds, dst, src, st);+ st->d[12] += nb;+ bytes -= step; src += step; dst += step;+ }+ }+#endif+ /* xor new 64-bytes chunks and store the left over if any */ for (; bytes >= 64; bytes -= 64, src += 64, dst += 64) { /* generate new chunk and update state */@@ -332,7 +434,7 @@ void crypton_chacha_generate(uint8_t *dst, crypton_chacha_context *ctx, uint32_t bytes) { crypton_chacha_state *st;- block out;+ crypton_chacha_block out; int i; if (!bytes)@@ -355,11 +457,24 @@ st = &ctx->st; +#ifdef CHACHA_SIMD+ {+ const uint32_t nb = (uint32_t) crypton_chacha_simd_width();+ const uint32_t step = 64 * nb;++ while (bytes >= step && CHACHA_SIMD_OK(st, nb)) {+ crypton_chacha_simd_generate(ctx->nb_rounds, dst, st);+ st->d[12] += nb;+ bytes -= step; dst += step;+ }+ }+#endif+ if (ALIGNED64(dst)) { /* xor new 64-bytes chunks and store the left over if any */ for (; bytes >= 64; bytes -= 64, dst += 64) { /* generate new chunk and update state */- chacha_core(ctx->nb_rounds, (block *) dst, st);+ chacha_core(ctx->nb_rounds, (crypton_chacha_block *) dst, st); uint32_t t0 = le32_to_cpu(st->d[12]); st->d[12] = cpu_to_le32(t0 + 1); if (st->d[12] == 0) {@@ -409,9 +524,9 @@ void crypton_chacha_generate_simple_block(uint8_t *dst, crypton_chacha_state *st, uint8_t rounds) { if (ALIGNED64(dst)) {- chacha_core(rounds, (block *) dst, st);+ chacha_core(rounds, (crypton_chacha_block *) dst, st); } else {- block out;+ crypton_chacha_block out; int i; chacha_core(rounds, &out, st); for (i = 0; i < 64; ++i) {@@ -429,7 +544,7 @@ void crypton_chacha_random(uint32_t rounds, uint8_t *dst, crypton_chacha_state *st, uint32_t bytes) {- block out;+ crypton_chacha_block out; if (!bytes) return;
@@ -34,9 +34,9 @@ uint64_t q[8]; uint32_t d[16]; uint8_t b[64];-} block;+} crypton_chacha_block; -typedef block crypton_chacha_state;+typedef crypton_chacha_block crypton_chacha_state; typedef struct { crypton_chacha_state st;
@@ -0,0 +1,156 @@+/*+ * Copyright (c) 2026 Kazu Yamamoto+ *+ * Redistribution and use in source and binary forms, with or without+ * modification, are permitted provided that the following conditions+ * are met:+ * 1. Redistributions of source code must retain the above copyright+ * notice, this list of conditions and the following disclaimer.+ * 2. Redistributions in binary form must reproduce the above copyright+ * notice, this list of conditions and the following disclaimer in the+ * documentation and/or other materials provided with the distribution.+ *+ * THIS SOFTWARE IS PROVIDED BY THE AUTHORS AND CONTRIBUTORS ``AS IS'' AND+ * ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE+ * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR+ * PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE AUTHORS OR CONTRIBUTORS+ * BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR+ * CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF+ * SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS+ * INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN+ * CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)+ * ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE+ * POSSIBILITY OF SUCH DAMAGE.+ */++#include <stdint.h>+#include <string.h>++#include "crypton_chacha.h"+#include "crypton_chachapoly.h"+#include "crypton_poly1305.h"++/* RFC 8439. The one-time Poly1305 key is the first 32 bytes of the ChaCha20+ * keystream at counter 0; a whole 64-byte block is generated so the counter+ * lands on 1, which is where the message starts. */+static void chachapoly_start(crypton_chacha_context *cctx, poly1305_ctx *pctx,+ const uint8_t *key,+ const uint8_t *nonce, uint32_t noncelen)+{+ uint8_t block[64];++ crypton_chacha_init(cctx, 20, 32, key, noncelen, nonce);+ crypton_chacha_generate(block, cctx, sizeof(block));+ crypton_poly1305_init(pctx, (poly1305_key *) block);+ memset(block, 0, sizeof(block));+}++/* Poly1305 over an associated or encrypted part, then zeros up to the next+ * multiple of sixteen. */+static void absorb_padded(poly1305_ctx *pctx, const uint8_t *p, uint32_t len)+{+ static const uint8_t zeros[16] = {0};+ uint32_t rem;++ if (len)+ crypton_poly1305_update(pctx, (uint8_t *) p, len);+ rem = len % 16;+ if (rem)+ crypton_poly1305_update(pctx, (uint8_t *) zeros, 16 - rem);+}++/* The two lengths, little endian, eight bytes each, which is what the tag+ * ends on. */+static void absorb_lengths(poly1305_ctx *pctx, uint32_t aadlen, uint32_t inlen)+{+ uint8_t lens[16];+ int i;++ for (i = 0; i < 8; i++)+ lens[i] = (uint8_t) (((uint64_t) aadlen) >> (8 * i));+ for (i = 0; i < 8; i++)+ lens[8 + i] = (uint8_t) (((uint64_t) inlen) >> (8 * i));+ crypton_poly1305_update(pctx, lens, sizeof(lens));+}++void crypton_chachapoly_encrypt(uint8_t *out, uint8_t *tag, uint32_t taglen,+ const uint8_t *key,+ const uint8_t *nonce, uint32_t noncelen,+ const uint8_t *aad, uint32_t aadlen,+ const uint8_t *input, uint32_t inlen)+{+ crypton_chacha_context cctx;+ poly1305_ctx pctx;+ poly1305_mac mac;++ chachapoly_start(&cctx, &pctx, key, nonce, noncelen);+ absorb_padded(&pctx, aad, aadlen);+ if (inlen)+ crypton_chacha_combine(out, &cctx, input, inlen);+ /* what the tag covers is the ciphertext, which is now in out */+ absorb_padded(&pctx, out, inlen);+ absorb_lengths(&pctx, aadlen, inlen);+ crypton_poly1305_finalize(mac, &pctx);+ memcpy(tag, mac, taglen);++ memset(&cctx, 0, sizeof(cctx));+ memset(&pctx, 0, sizeof(pctx));+}++/* Shared by the two decrypting entry points: with outtag NULL the tag is+ * compared here and the answer returned, otherwise it is written there. */+static int chachapoly_decrypt(uint8_t *out, const uint8_t *tag, uint32_t taglen,+ uint8_t *outtag, const uint8_t *key,+ const uint8_t *nonce, uint32_t noncelen,+ const uint8_t *aad, uint32_t aadlen,+ const uint8_t *input, uint32_t inlen)+{+ crypton_chacha_context cctx;+ poly1305_ctx pctx;+ poly1305_mac mac;+ uint8_t diff = 0;+ uint32_t i;++ chachapoly_start(&cctx, &pctx, key, nonce, noncelen);+ absorb_padded(&pctx, aad, aadlen);+ /* here the ciphertext is the input, so the tag can be taken before the+ * plaintext is written and out may alias input */+ absorb_padded(&pctx, input, inlen);+ absorb_lengths(&pctx, aadlen, inlen);+ crypton_poly1305_finalize(mac, &pctx);++ if (inlen)+ crypton_chacha_combine(out, &cctx, input, inlen);++ memset(&cctx, 0, sizeof(cctx));+ memset(&pctx, 0, sizeof(pctx));++ if (outtag) {+ memcpy(outtag, mac, taglen);+ return 1;+ }+ for (i = 0; i < taglen; i++)+ diff |= (uint8_t) (mac[i] ^ tag[i]);+ return diff == 0;+}++int crypton_chachapoly_decrypt(uint8_t *out,+ const uint8_t *tag, uint32_t taglen,+ const uint8_t *key,+ const uint8_t *nonce, uint32_t noncelen,+ const uint8_t *aad, uint32_t aadlen,+ const uint8_t *input, uint32_t inlen)+{+ return chachapoly_decrypt(out, tag, taglen, NULL, key, nonce, noncelen,+ aad, aadlen, input, inlen);+}++void crypton_chachapoly_decrypt_tag(uint8_t *out, uint8_t *outtag,+ uint32_t taglen, const uint8_t *key,+ const uint8_t *nonce, uint32_t noncelen,+ const uint8_t *aad, uint32_t aadlen,+ const uint8_t *input, uint32_t inlen)+{+ (void) chachapoly_decrypt(out, NULL, taglen, outtag, key, nonce,+ noncelen, aad, aadlen, input, inlen);+}
@@ -0,0 +1,62 @@+/*+ * Copyright (c) 2026 Kazu Yamamoto+ *+ * Redistribution and use in source and binary forms, with or without+ * modification, are permitted provided that the following conditions+ * are met:+ * 1. Redistributions of source code must retain the above copyright+ * notice, this list of conditions and the following disclaimer.+ * 2. Redistributions in binary form must reproduce the above copyright+ * notice, this list of conditions and the following disclaimer in the+ * documentation and/or other materials provided with the distribution.+ *+ * THIS SOFTWARE IS PROVIDED BY THE AUTHORS AND CONTRIBUTORS ``AS IS'' AND+ * ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE+ * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR+ * PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE AUTHORS OR CONTRIBUTORS+ * BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR+ * CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF+ * SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS+ * INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN+ * CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)+ * ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE+ * POSSIBILITY OF SUCH DAMAGE.+ */++#ifndef CRYPTON_CHACHAPOLY_H+#define CRYPTON_CHACHAPOLY_H++#include <stdint.h>++/* ChaCha20-Poly1305 (RFC 8439) as one call.+ *+ * The pieces are the ChaCha20 and Poly1305 already here; what these do is+ * hold them together, which the Haskell above used to do at the cost of eight+ * foreign calls and the allocations between them.+ *+ * The nonce is the twelve bytes RFC 8439 defines. taglen is at most 16.+ */++void crypton_chachapoly_encrypt(uint8_t *out, uint8_t *tag, uint32_t taglen,+ const uint8_t *key,+ const uint8_t *nonce, uint32_t noncelen,+ const uint8_t *aad, uint32_t aadlen,+ const uint8_t *input, uint32_t inlen);++/* Decrypt and compare, a byte at a time over the whole tag whichever way the+ * answer goes. Returns non-zero when the tag matched. */+int crypton_chachapoly_decrypt(uint8_t *out,+ const uint8_t *tag, uint32_t taglen,+ const uint8_t *key,+ const uint8_t *nonce, uint32_t noncelen,+ const uint8_t *aad, uint32_t aadlen,+ const uint8_t *input, uint32_t inlen);++/* Decrypt and hand the computed tag back rather than comparing it. */+void crypton_chachapoly_decrypt_tag(uint8_t *out, uint8_t *outtag,+ uint32_t taglen, const uint8_t *key,+ const uint8_t *nonce, uint32_t noncelen,+ const uint8_t *aad, uint32_t aadlen,+ const uint8_t *input, uint32_t inlen);++#endif
@@ -31,6 +31,35 @@ #include "crypton_cpu.h" #include <stdint.h> +/*+ * PE has no way to say "hidden": every symbol in an object is local to the+ * image unless something exports it, which is what hidden asks for+ * elsewhere, so nothing is lost by dropping the attribute here. Saying it+ * anyway is not harmless -- the gcc that GHC 9.2 ships for Windows parses+ * the attribute, discards it and warns, and it is the only warning crypton's+ * own code produces anywhere in the CI matrix. Measured on mingw gcc 13.2.0+ * and clang 14.0.6 (the compiler GHC 9.4 and later ship): gcc warns for+ * "hidden" and is silent for "default", clang is silent for both, which is+ * why the vendored decaf and argon2 headers ask for "default" unnoticed.+ */+#if defined(_WIN32) || defined(__CYGWIN__)+#define CRYPTON_HIDDEN+#else+#define CRYPTON_HIDDEN __attribute__((visibility("hidden")))+#endif++/*+ * The word the assembly reads; crypton_cpu.h says what is in it. Hidden,+ * so that the reference to it from the assembly resolves at link time in a+ * shared object as well as a static one. The SHA-256 bit is set by+ * cbits/crypton_sha256.c once it has asked whether the processor has those+ * instructions.+ */+#ifdef CRYPTON_ARM_ASM+CRYPTON_HIDDEN unsigned int crypton_armcap_P =+ CRYPTON_ARMCAP_NEON;+#endif+ #ifdef ARCH_X86 static void cpuid(uint32_t info, uint32_t *eax, uint32_t *ebx, uint32_t *ecx, uint32_t *edx) {@@ -51,6 +80,217 @@ #endif :"+a" (*eax), "=S" (*ebx), "=c" (*ecx), "=d" (*edx) : :"edi");+}++/*+ * What the machine will let us use beyond the x86-64 baseline. XGETBV is+ * spelled out in bytes because it predates some assemblers that are still+ * in use.+ */+static void cpuid_count(uint32_t info, uint32_t sub, uint32_t *eax, uint32_t *ebx, uint32_t *ecx, uint32_t *edx)+{+ *eax = info;+ *ecx = sub;+ __asm__ volatile+ (+#ifdef __x86_64__+ "mov %%rbx, %%rdi;"+#else+ "mov %%ebx, %%edi;"+#endif+ "cpuid;"+ "mov %%ebx, %%esi;"+#ifdef __x86_64__+ "mov %%rdi, %%rbx;"+#else+ "mov %%edi, %%ebx;"+#endif+ :"+a" (*eax), "=S" (*ebx), "+c" (*ecx), "=d" (*edx)+ : :"edi");+}++static uint64_t xcr0(void)+{+ uint32_t lo, hi;++ __asm__ volatile(".byte 0x0f, 0x01, 0xd0" : "=a" (lo), "=d" (hi) : "c" (0));+ return ((uint64_t) hi << 32) | lo;+}++#ifdef CRYPTON_X86_ASM+CRYPTON_HIDDEN unsigned int crypton_ia32cap_P[4];++/*+ * The AVX-512 bits of leaf 7 EBX -- F, DQ, IFMA, PF, ER, CD, BW and VL,+ * which is every bit from 16 up except 21's neighbours and 29, the SHA+ * extensions, which are not AVX-512 and are wanted. They are cleared+ * whatever the processor says: the code they would select in the vendored+ * assembly cannot be run, let alone measured, on any machine here, and+ * shipping a path nothing has executed is not worth the few per cent it+ * might be worth. Turning them on is a one-line change for whoever has+ * the hardware.+ */+#define IA32CAP_AVX512 \+ ((1u << 16) | (1u << 17) | (1u << 21) | (1u << 26) | (1u << 27) \+ | (1u << 28) | (1u << 30) | (1u << 31))++/*+ * cpuid as the assembly reads it, with the two bits it dispatches on -- AVX+ * in leaf 1 and AVX2 in leaf 7 -- left set only where the answer already+ * agreed that the operating system saves the registers. Two threads racing+ * here write the same values.+ */+void crypton_x86_ia32cap_resolve(void)+{+ static int resolved = 0;++ if (!resolved) {+ uint32_t eax, ebx, ecx, edx, maxleaf;+ uint32_t f = crypton_x86_simd_features();+ uint32_t leaf1_ecx, leaf7_ebx = 0;+ int intel;++ cpuid(0, &eax, &ebx, &ecx, &edx);+ maxleaf = eax;+ /* "GenuineIntel", which OpenSSL records in a bit of leaf 1+ * EDX that cpuid leaves reserved: some of the assembly asks,+ * having found a path worth taking on one make and not the+ * other */+ intel = (ebx == 0x756e6547 && edx == 0x49656e69+ && ecx == 0x6c65746e);++ cpuid(1, &eax, &ebx, &ecx, &edx);+ crypton_ia32cap_P[0] = intel ? (edx | (1u << 30)) : edx;+ leaf1_ecx = ecx;+ if (!(f & CRYPTON_X86_AVX))+ leaf1_ecx &= ~(1u << 28);+ /*+ * Bit 11 is not leaf 1's to give. The assembly reads it as+ * AMD's XOP, which lives in leaf 0x80000001, and OpenSSL+ * clears whatever leaf 1 put there before merging the real+ * flag into the place -- on Intel that is SDBG, the silicon+ * debug interface, reported since Broadwell, and reading it+ * as XOP sends SHA-512 and ChaCha20 into a vprotq and a+ * SIGILL. It is cleared and left clear: nothing here can run+ * XOP to test it, and no processor still in service has it,+ * AMD having carried it from Bulldozer to Excavator and Zen+ * having dropped it. That is the reason the AVX-512 bits+ * above are cleared too.+ */+ leaf1_ecx &= ~(1u << 11);+ crypton_ia32cap_P[1] = leaf1_ecx;++ if (maxleaf >= 7) {+ cpuid_count(7, 0, &eax, &ebx, &ecx, &edx);+ leaf7_ebx = ebx;+ }+ if (!(f & CRYPTON_X86_AVX2))+ leaf7_ebx &= ~(1u << 5);+ crypton_ia32cap_P[2] = leaf7_ebx & ~IA32CAP_AVX512;++ resolved = 1;+ }+}+#endif++uint32_t crypton_x86_simd_features(void)+{+ static int resolved = 0;+ static uint32_t features = 0;++ if (!resolved) {+ uint32_t eax, ebx, ecx, edx, leaf1, maxleaf, family, f = 0;+ int amd;++ cpuid(0, &eax, &ebx, &ecx, &edx);+ maxleaf = eax;+ /* "AuthenticAMD" arrives as EBX, EDX, ECX in that order */+ amd = (ebx == 0x68747541 && edx == 0x69746e65+ && ecx == 0x444d4163);++ cpuid(1, &eax, &ebx, &ecx, &edx);+ leaf1 = ecx;+ /* the family is the base one, and the extended field is+ * added to it only when the base reads 0xf, which is how+ * every AMD Zen part reports */+ family = (eax >> 8) & 0xf;+ if (family == 0xf)+ family += (eax >> 20) & 0xff;+ if (leaf1 & (1 << 9))+ f |= CRYPTON_X86_SSSE3;+ if (leaf1 & (1 << 1))+ f |= CRYPTON_X86_PCLMUL;+ if (leaf1 & (1 << 22))+ f |= CRYPTON_X86_MOVBE;+ /* AVX asks the same three things as AVX2 below: the+ * processor has it, OSXSAVE is on, and the operating system+ * says it saves the registers */+ if ((leaf1 & (1 << 28)) && (leaf1 & (1 << 27))+ && ((xcr0() & 6) == 6))+ f |= CRYPTON_X86_AVX;++ /* leaf 7 answers for both of the rest, and a processor that+ * does not have it answers for the highest leaf it does have+ * instead, so ask what that is first */+ if (maxleaf >= 7) {+ cpuid_count(7, 0, &eax, &ebx, &ecx, &edx);+ /* the SHA extensions work in registers the SSE state+ * already covers, so they need nothing of the+ * operating system. The code that uses them also+ * wants SSSE3 and SSE4.1, which every processor that+ * has them has, but ask rather than assume */+ if ((ebx & (1 << 29)) && (leaf1 & (1 << 9))+ && (leaf1 & (1 << 19)))+ f |= CRYPTON_X86_SHA_NI;+ /* AVX2 has the wider registers, which takes three+ * things agreeing: the CPU has it, OSXSAVE is on, and+ * XCR0 says the operating system saves them --+ * without that last one the upper halves are lost+ * across a context switch */+ if ((ebx & (1 << 5)) && (leaf1 & (1 << 27))+ && (leaf1 & (1 << 28)) && ((xcr0() & 6) == 6))+ f |= CRYPTON_X86_AVX2;+ /* BMI2 for MULX and ADX for ADCX/ADOX. Both are+ * wanted together and neither touches vector state,+ * so there is nothing to ask the operating system */+ if ((ebx & (1 << 8)) && (ebx & (1 << 19)))+ f |= CRYPTON_X86_ADX;+ /* VAES and VPCLMULQDQ, leaf 7 ECX bits 9 and 10.+ * They are wanted together -- one without the other+ * leaves half of AES-GCM narrow -- and they need the+ * wide registers, so AVX2 has to have answered first,+ * which settles the operating system's part. */+ if ((ecx & (1 << 9)) && (ecx & (1 << 10))+ && (f & CRYPTON_X86_AVX2))+ f |= CRYPTON_X86_VAES;+ /* The same two instructions in their 512-bit form,+ * which wants AVX-512 F, BW and VL as well -- and+ * three more bits of XCR0, for the mask registers and+ * the two upper halves of the vector state. A+ * machine can report the instructions and still fault+ * on them when the operating system has not said it+ * saves that state, which is what those bits are. */+ if ((f & CRYPTON_X86_VAES)+ && (ebx & (1 << 16)) && (ebx & (1u << 30))+ && (ebx & (1u << 31))+ && ((xcr0() & 0xe6) == 0xe6)+ /* Not on Zen 4, which is AMD family 19h with+ * AVX-512: there the 512-bit instructions are two+ * 256-bit passes through a 256-bit datapath, so+ * they carry the wider encoding for none of the+ * throughput, and AES-GCM measures 0.6 to 3+ * per cent slower than the 256-bit path. Zen 5+ * is family 1Ah and does have the wide datapath,+ * where the same code is half as fast again;+ * Zen 3, the other family 19h part, has no+ * AVX-512 at all and never reaches here. */+ && !(amd && family == 0x19))+ f |= CRYPTON_X86_VAES512;+ }+ features = f;+ resolved = 1;+ }+ return features; } #ifdef USE_AESNI
@@ -31,9 +31,71 @@ #ifndef CPU_H #define CPU_H +#include <stdint.h>+ #if defined(__i386__) || defined(__x86_64__) #define ARCH_X86 #define USE_AESNI+#endif++/* vector extensions beyond the x86-64 baseline, as cpuid reports them and+ * the OS allows them */+#define CRYPTON_X86_SSSE3 1+#define CRYPTON_X86_AVX2 2+#define CRYPTON_X86_PCLMUL 4+/* the SHA extensions, and the SSSE3 and SSE4.1 the code around them uses */+#define CRYPTON_X86_SHA_NI 8+/* the 128-bit half of AVX, which is what the vendored assembly is written+ * in, and the byte-swapping load it reads the message with */+#define CRYPTON_X86_AVX 16+#define CRYPTON_X86_MOVBE 32+/* MULX, ADCX and ADOX together: the two independent carry chains the+ * vendored s2n-bignum assembly wants. They are general-purpose register+ * instructions, so unlike the vector ones above they ask nothing of the+ * operating system. */+#define CRYPTON_X86_ADX 64+/* The AES and carry-less multiply instructions in their 256-bit form, which+ * do two blocks where the 128-bit ones do one. They are VEX-encoded and use+ * the vector registers AVX2 already needs the operating system to save, so+ * they ask nothing further of it -- but AVX2 itself is asked about, since+ * without it there is nowhere to put them. */+#define CRYPTON_X86_VAES 128+/* The same pair in their 512-bit form, four blocks to an instruction. These+ * are EVEX-encoded and need the AVX-512 state as well, which is three more+ * bits of XCR0 than AVX2 wants: the mask registers and the two upper halves+ * of the vector registers. */+#define CRYPTON_X86_VAES512 256+#ifdef ARCH_X86+uint32_t crypton_x86_simd_features(void);+#endif++/*+ * What the vendored AArch64 assembly asks about the processor, in the way+ * OpenSSL asks it and with OpenSSL's bit numbering. NEON is not optional+ * on AArch64 and is set from the start; the SHA-256 instructions are, so+ * the bit for them is set once the runtime check has answered. See+ * cbits/crypton_cpu.c and cbits/asm/README.md.+ */+/*+ * And what the vendored x86-64 assembly asks, which is cpuid's own words in+ * the order OpenSSL keeps them: [0] is leaf 1 EDX, [1] leaf 1 ECX and [2]+ * leaf 7 EBX, with the bits for what the operating system will not preserve+ * cleared. Filled on first use; see cbits/crypton_cpu.c.+ */+#if defined(WITH_X86_POLY1305_ASM) || defined(WITH_X86_CHACHA_ASM) \+ || defined(WITH_X86_SHA256_ASM) || defined(WITH_X86_SHA512_ASM)+#define CRYPTON_X86_ASM 1+extern unsigned int crypton_ia32cap_P[4];+void crypton_x86_ia32cap_resolve(void);+#endif++#if defined(WITH_ARMV8_CHACHA_ASM) || defined(WITH_ARMV8_POLY1305_ASM) \+ || defined(WITH_ARMV8_SHA1_ASM) || defined(WITH_ARMV8_SHA256_ASM)+#define CRYPTON_ARM_ASM 1+#define CRYPTON_ARMCAP_NEON 1+#define CRYPTON_ARMCAP_SHA1 (1 << 3)+#define CRYPTON_ARMCAP_SHA256 (1 << 4)+extern unsigned int crypton_armcap_P; #endif #ifdef USE_AESNI
@@ -0,0 +1,1325 @@+/*+ * DES, as FIPS 46-3 defines it.+ *+ * The tables below are generated from the permutations and S-boxes of that+ * standard: SP[i] combines S-box i with the P permutation, IPL/IPR and FPH/FPL+ * apply the initial and final permutations one input byte at a time, and the+ * 48-bit round key is kept as eight six-bit values so that the E expansion is+ * a rotate and a shift rather than a table.+ *+ * DES is here because callers still meet it, not because it should be chosen:+ * its 56-bit key is exhaustible, and this implementation indexes tables with+ * key-dependent values, so it is not constant time.+ */+#include <stdint.h>+#include <string.h>+#include <crypton_des.h>++static const uint32_t SP[8][64] = {+{+ 0x00808200U, 0x00000000U, 0x00008000U, 0x00808202U, 0x00808002U, 0x00008202U, 0x00000002U, 0x00008000U,+ 0x00000200U, 0x00808200U, 0x00808202U, 0x00000200U, 0x00800202U, 0x00808002U, 0x00800000U, 0x00000002U,+ 0x00000202U, 0x00800200U, 0x00800200U, 0x00008200U, 0x00008200U, 0x00808000U, 0x00808000U, 0x00800202U,+ 0x00008002U, 0x00800002U, 0x00800002U, 0x00008002U, 0x00000000U, 0x00000202U, 0x00008202U, 0x00800000U,+ 0x00008000U, 0x00808202U, 0x00000002U, 0x00808000U, 0x00808200U, 0x00800000U, 0x00800000U, 0x00000200U,+ 0x00808002U, 0x00008000U, 0x00008200U, 0x00800002U, 0x00000200U, 0x00000002U, 0x00800202U, 0x00008202U,+ 0x00808202U, 0x00008002U, 0x00808000U, 0x00800202U, 0x00800002U, 0x00000202U, 0x00008202U, 0x00808200U,+ 0x00000202U, 0x00800200U, 0x00800200U, 0x00000000U, 0x00008002U, 0x00008200U, 0x00000000U, 0x00808002U,+},+{+ 0x40084010U, 0x40004000U, 0x00004000U, 0x00084010U, 0x00080000U, 0x00000010U, 0x40080010U, 0x40004010U,+ 0x40000010U, 0x40084010U, 0x40084000U, 0x40000000U, 0x40004000U, 0x00080000U, 0x00000010U, 0x40080010U,+ 0x00084000U, 0x00080010U, 0x40004010U, 0x00000000U, 0x40000000U, 0x00004000U, 0x00084010U, 0x40080000U,+ 0x00080010U, 0x40000010U, 0x00000000U, 0x00084000U, 0x00004010U, 0x40084000U, 0x40080000U, 0x00004010U,+ 0x00000000U, 0x00084010U, 0x40080010U, 0x00080000U, 0x40004010U, 0x40080000U, 0x40084000U, 0x00004000U,+ 0x40080000U, 0x40004000U, 0x00000010U, 0x40084010U, 0x00084010U, 0x00000010U, 0x00004000U, 0x40000000U,+ 0x00004010U, 0x40084000U, 0x00080000U, 0x40000010U, 0x00080010U, 0x40004010U, 0x40000010U, 0x00080010U,+ 0x00084000U, 0x00000000U, 0x40004000U, 0x00004010U, 0x40000000U, 0x40080010U, 0x40084010U, 0x00084000U,+},+{+ 0x00000104U, 0x04010100U, 0x00000000U, 0x04010004U, 0x04000100U, 0x00000000U, 0x00010104U, 0x04000100U,+ 0x00010004U, 0x04000004U, 0x04000004U, 0x00010000U, 0x04010104U, 0x00010004U, 0x04010000U, 0x00000104U,+ 0x04000000U, 0x00000004U, 0x04010100U, 0x00000100U, 0x00010100U, 0x04010000U, 0x04010004U, 0x00010104U,+ 0x04000104U, 0x00010100U, 0x00010000U, 0x04000104U, 0x00000004U, 0x04010104U, 0x00000100U, 0x04000000U,+ 0x04010100U, 0x04000000U, 0x00010004U, 0x00000104U, 0x00010000U, 0x04010100U, 0x04000100U, 0x00000000U,+ 0x00000100U, 0x00010004U, 0x04010104U, 0x04000100U, 0x04000004U, 0x00000100U, 0x00000000U, 0x04010004U,+ 0x04000104U, 0x00010000U, 0x04000000U, 0x04010104U, 0x00000004U, 0x00010104U, 0x00010100U, 0x04000004U,+ 0x04010000U, 0x04000104U, 0x00000104U, 0x04010000U, 0x00010104U, 0x00000004U, 0x04010004U, 0x00010100U,+},+{+ 0x80401000U, 0x80001040U, 0x80001040U, 0x00000040U, 0x00401040U, 0x80400040U, 0x80400000U, 0x80001000U,+ 0x00000000U, 0x00401000U, 0x00401000U, 0x80401040U, 0x80000040U, 0x00000000U, 0x00400040U, 0x80400000U,+ 0x80000000U, 0x00001000U, 0x00400000U, 0x80401000U, 0x00000040U, 0x00400000U, 0x80001000U, 0x00001040U,+ 0x80400040U, 0x80000000U, 0x00001040U, 0x00400040U, 0x00001000U, 0x00401040U, 0x80401040U, 0x80000040U,+ 0x00400040U, 0x80400000U, 0x00401000U, 0x80401040U, 0x80000040U, 0x00000000U, 0x00000000U, 0x00401000U,+ 0x00001040U, 0x00400040U, 0x80400040U, 0x80000000U, 0x80401000U, 0x80001040U, 0x80001040U, 0x00000040U,+ 0x80401040U, 0x80000040U, 0x80000000U, 0x00001000U, 0x80400000U, 0x80001000U, 0x00401040U, 0x80400040U,+ 0x80001000U, 0x00001040U, 0x00400000U, 0x80401000U, 0x00000040U, 0x00400000U, 0x00001000U, 0x00401040U,+},+{+ 0x00000080U, 0x01040080U, 0x01040000U, 0x21000080U, 0x00040000U, 0x00000080U, 0x20000000U, 0x01040000U,+ 0x20040080U, 0x00040000U, 0x01000080U, 0x20040080U, 0x21000080U, 0x21040000U, 0x00040080U, 0x20000000U,+ 0x01000000U, 0x20040000U, 0x20040000U, 0x00000000U, 0x20000080U, 0x21040080U, 0x21040080U, 0x01000080U,+ 0x21040000U, 0x20000080U, 0x00000000U, 0x21000000U, 0x01040080U, 0x01000000U, 0x21000000U, 0x00040080U,+ 0x00040000U, 0x21000080U, 0x00000080U, 0x01000000U, 0x20000000U, 0x01040000U, 0x21000080U, 0x20040080U,+ 0x01000080U, 0x20000000U, 0x21040000U, 0x01040080U, 0x20040080U, 0x00000080U, 0x01000000U, 0x21040000U,+ 0x21040080U, 0x00040080U, 0x21000000U, 0x21040080U, 0x01040000U, 0x00000000U, 0x20040000U, 0x21000000U,+ 0x00040080U, 0x01000080U, 0x20000080U, 0x00040000U, 0x00000000U, 0x20040000U, 0x01040080U, 0x20000080U,+},+{+ 0x10000008U, 0x10200000U, 0x00002000U, 0x10202008U, 0x10200000U, 0x00000008U, 0x10202008U, 0x00200000U,+ 0x10002000U, 0x00202008U, 0x00200000U, 0x10000008U, 0x00200008U, 0x10002000U, 0x10000000U, 0x00002008U,+ 0x00000000U, 0x00200008U, 0x10002008U, 0x00002000U, 0x00202000U, 0x10002008U, 0x00000008U, 0x10200008U,+ 0x10200008U, 0x00000000U, 0x00202008U, 0x10202000U, 0x00002008U, 0x00202000U, 0x10202000U, 0x10000000U,+ 0x10002000U, 0x00000008U, 0x10200008U, 0x00202000U, 0x10202008U, 0x00200000U, 0x00002008U, 0x10000008U,+ 0x00200000U, 0x10002000U, 0x10000000U, 0x00002008U, 0x10000008U, 0x10202008U, 0x00202000U, 0x10200000U,+ 0x00202008U, 0x10202000U, 0x00000000U, 0x10200008U, 0x00000008U, 0x00002000U, 0x10200000U, 0x00202008U,+ 0x00002000U, 0x00200008U, 0x10002008U, 0x00000000U, 0x10202000U, 0x10000000U, 0x00200008U, 0x10002008U,+},+{+ 0x00100000U, 0x02100001U, 0x02000401U, 0x00000000U, 0x00000400U, 0x02000401U, 0x00100401U, 0x02100400U,+ 0x02100401U, 0x00100000U, 0x00000000U, 0x02000001U, 0x00000001U, 0x02000000U, 0x02100001U, 0x00000401U,+ 0x02000400U, 0x00100401U, 0x00100001U, 0x02000400U, 0x02000001U, 0x02100000U, 0x02100400U, 0x00100001U,+ 0x02100000U, 0x00000400U, 0x00000401U, 0x02100401U, 0x00100400U, 0x00000001U, 0x02000000U, 0x00100400U,+ 0x02000000U, 0x00100400U, 0x00100000U, 0x02000401U, 0x02000401U, 0x02100001U, 0x02100001U, 0x00000001U,+ 0x00100001U, 0x02000000U, 0x02000400U, 0x00100000U, 0x02100400U, 0x00000401U, 0x00100401U, 0x02100400U,+ 0x00000401U, 0x02000001U, 0x02100401U, 0x02100000U, 0x00100400U, 0x00000000U, 0x00000001U, 0x02100401U,+ 0x00000000U, 0x00100401U, 0x02100000U, 0x00000400U, 0x02000001U, 0x02000400U, 0x00000400U, 0x00100001U,+},+{+ 0x08000820U, 0x00000800U, 0x00020000U, 0x08020820U, 0x08000000U, 0x08000820U, 0x00000020U, 0x08000000U,+ 0x00020020U, 0x08020000U, 0x08020820U, 0x00020800U, 0x08020800U, 0x00020820U, 0x00000800U, 0x00000020U,+ 0x08020000U, 0x08000020U, 0x08000800U, 0x00000820U, 0x00020800U, 0x00020020U, 0x08020020U, 0x08020800U,+ 0x00000820U, 0x00000000U, 0x00000000U, 0x08020020U, 0x08000020U, 0x08000800U, 0x00020820U, 0x00020000U,+ 0x00020820U, 0x00020000U, 0x08020800U, 0x00000800U, 0x00000020U, 0x08020020U, 0x00000800U, 0x00020820U,+ 0x08000800U, 0x00000020U, 0x08000020U, 0x08020000U, 0x08020020U, 0x08000000U, 0x00020000U, 0x08000820U,+ 0x00000000U, 0x08020820U, 0x00020020U, 0x08000020U, 0x08020000U, 0x08000800U, 0x08000820U, 0x00000000U,+ 0x08020820U, 0x00020800U, 0x00020800U, 0x00000820U, 0x00000820U, 0x00020020U, 0x08000000U, 0x08020800U,+},+};++static const uint32_t IPL[8][256] = {+{+ 0x00000000U, 0x00000001U, 0x00000000U, 0x00000001U, 0x00000100U, 0x00000101U, 0x00000100U, 0x00000101U,+ 0x00000000U, 0x00000001U, 0x00000000U, 0x00000001U, 0x00000100U, 0x00000101U, 0x00000100U, 0x00000101U,+ 0x00010000U, 0x00010001U, 0x00010000U, 0x00010001U, 0x00010100U, 0x00010101U, 0x00010100U, 0x00010101U,+ 0x00010000U, 0x00010001U, 0x00010000U, 0x00010001U, 0x00010100U, 0x00010101U, 0x00010100U, 0x00010101U,+ 0x00000000U, 0x00000001U, 0x00000000U, 0x00000001U, 0x00000100U, 0x00000101U, 0x00000100U, 0x00000101U,+ 0x00000000U, 0x00000001U, 0x00000000U, 0x00000001U, 0x00000100U, 0x00000101U, 0x00000100U, 0x00000101U,+ 0x00010000U, 0x00010001U, 0x00010000U, 0x00010001U, 0x00010100U, 0x00010101U, 0x00010100U, 0x00010101U,+ 0x00010000U, 0x00010001U, 0x00010000U, 0x00010001U, 0x00010100U, 0x00010101U, 0x00010100U, 0x00010101U,+ 0x01000000U, 0x01000001U, 0x01000000U, 0x01000001U, 0x01000100U, 0x01000101U, 0x01000100U, 0x01000101U,+ 0x01000000U, 0x01000001U, 0x01000000U, 0x01000001U, 0x01000100U, 0x01000101U, 0x01000100U, 0x01000101U,+ 0x01010000U, 0x01010001U, 0x01010000U, 0x01010001U, 0x01010100U, 0x01010101U, 0x01010100U, 0x01010101U,+ 0x01010000U, 0x01010001U, 0x01010000U, 0x01010001U, 0x01010100U, 0x01010101U, 0x01010100U, 0x01010101U,+ 0x01000000U, 0x01000001U, 0x01000000U, 0x01000001U, 0x01000100U, 0x01000101U, 0x01000100U, 0x01000101U,+ 0x01000000U, 0x01000001U, 0x01000000U, 0x01000001U, 0x01000100U, 0x01000101U, 0x01000100U, 0x01000101U,+ 0x01010000U, 0x01010001U, 0x01010000U, 0x01010001U, 0x01010100U, 0x01010101U, 0x01010100U, 0x01010101U,+ 0x01010000U, 0x01010001U, 0x01010000U, 0x01010001U, 0x01010100U, 0x01010101U, 0x01010100U, 0x01010101U,+ 0x00000000U, 0x00000001U, 0x00000000U, 0x00000001U, 0x00000100U, 0x00000101U, 0x00000100U, 0x00000101U,+ 0x00000000U, 0x00000001U, 0x00000000U, 0x00000001U, 0x00000100U, 0x00000101U, 0x00000100U, 0x00000101U,+ 0x00010000U, 0x00010001U, 0x00010000U, 0x00010001U, 0x00010100U, 0x00010101U, 0x00010100U, 0x00010101U,+ 0x00010000U, 0x00010001U, 0x00010000U, 0x00010001U, 0x00010100U, 0x00010101U, 0x00010100U, 0x00010101U,+ 0x00000000U, 0x00000001U, 0x00000000U, 0x00000001U, 0x00000100U, 0x00000101U, 0x00000100U, 0x00000101U,+ 0x00000000U, 0x00000001U, 0x00000000U, 0x00000001U, 0x00000100U, 0x00000101U, 0x00000100U, 0x00000101U,+ 0x00010000U, 0x00010001U, 0x00010000U, 0x00010001U, 0x00010100U, 0x00010101U, 0x00010100U, 0x00010101U,+ 0x00010000U, 0x00010001U, 0x00010000U, 0x00010001U, 0x00010100U, 0x00010101U, 0x00010100U, 0x00010101U,+ 0x01000000U, 0x01000001U, 0x01000000U, 0x01000001U, 0x01000100U, 0x01000101U, 0x01000100U, 0x01000101U,+ 0x01000000U, 0x01000001U, 0x01000000U, 0x01000001U, 0x01000100U, 0x01000101U, 0x01000100U, 0x01000101U,+ 0x01010000U, 0x01010001U, 0x01010000U, 0x01010001U, 0x01010100U, 0x01010101U, 0x01010100U, 0x01010101U,+ 0x01010000U, 0x01010001U, 0x01010000U, 0x01010001U, 0x01010100U, 0x01010101U, 0x01010100U, 0x01010101U,+ 0x01000000U, 0x01000001U, 0x01000000U, 0x01000001U, 0x01000100U, 0x01000101U, 0x01000100U, 0x01000101U,+ 0x01000000U, 0x01000001U, 0x01000000U, 0x01000001U, 0x01000100U, 0x01000101U, 0x01000100U, 0x01000101U,+ 0x01010000U, 0x01010001U, 0x01010000U, 0x01010001U, 0x01010100U, 0x01010101U, 0x01010100U, 0x01010101U,+ 0x01010000U, 0x01010001U, 0x01010000U, 0x01010001U, 0x01010100U, 0x01010101U, 0x01010100U, 0x01010101U,+},+{+ 0x00000000U, 0x00000002U, 0x00000000U, 0x00000002U, 0x00000200U, 0x00000202U, 0x00000200U, 0x00000202U,+ 0x00000000U, 0x00000002U, 0x00000000U, 0x00000002U, 0x00000200U, 0x00000202U, 0x00000200U, 0x00000202U,+ 0x00020000U, 0x00020002U, 0x00020000U, 0x00020002U, 0x00020200U, 0x00020202U, 0x00020200U, 0x00020202U,+ 0x00020000U, 0x00020002U, 0x00020000U, 0x00020002U, 0x00020200U, 0x00020202U, 0x00020200U, 0x00020202U,+ 0x00000000U, 0x00000002U, 0x00000000U, 0x00000002U, 0x00000200U, 0x00000202U, 0x00000200U, 0x00000202U,+ 0x00000000U, 0x00000002U, 0x00000000U, 0x00000002U, 0x00000200U, 0x00000202U, 0x00000200U, 0x00000202U,+ 0x00020000U, 0x00020002U, 0x00020000U, 0x00020002U, 0x00020200U, 0x00020202U, 0x00020200U, 0x00020202U,+ 0x00020000U, 0x00020002U, 0x00020000U, 0x00020002U, 0x00020200U, 0x00020202U, 0x00020200U, 0x00020202U,+ 0x02000000U, 0x02000002U, 0x02000000U, 0x02000002U, 0x02000200U, 0x02000202U, 0x02000200U, 0x02000202U,+ 0x02000000U, 0x02000002U, 0x02000000U, 0x02000002U, 0x02000200U, 0x02000202U, 0x02000200U, 0x02000202U,+ 0x02020000U, 0x02020002U, 0x02020000U, 0x02020002U, 0x02020200U, 0x02020202U, 0x02020200U, 0x02020202U,+ 0x02020000U, 0x02020002U, 0x02020000U, 0x02020002U, 0x02020200U, 0x02020202U, 0x02020200U, 0x02020202U,+ 0x02000000U, 0x02000002U, 0x02000000U, 0x02000002U, 0x02000200U, 0x02000202U, 0x02000200U, 0x02000202U,+ 0x02000000U, 0x02000002U, 0x02000000U, 0x02000002U, 0x02000200U, 0x02000202U, 0x02000200U, 0x02000202U,+ 0x02020000U, 0x02020002U, 0x02020000U, 0x02020002U, 0x02020200U, 0x02020202U, 0x02020200U, 0x02020202U,+ 0x02020000U, 0x02020002U, 0x02020000U, 0x02020002U, 0x02020200U, 0x02020202U, 0x02020200U, 0x02020202U,+ 0x00000000U, 0x00000002U, 0x00000000U, 0x00000002U, 0x00000200U, 0x00000202U, 0x00000200U, 0x00000202U,+ 0x00000000U, 0x00000002U, 0x00000000U, 0x00000002U, 0x00000200U, 0x00000202U, 0x00000200U, 0x00000202U,+ 0x00020000U, 0x00020002U, 0x00020000U, 0x00020002U, 0x00020200U, 0x00020202U, 0x00020200U, 0x00020202U,+ 0x00020000U, 0x00020002U, 0x00020000U, 0x00020002U, 0x00020200U, 0x00020202U, 0x00020200U, 0x00020202U,+ 0x00000000U, 0x00000002U, 0x00000000U, 0x00000002U, 0x00000200U, 0x00000202U, 0x00000200U, 0x00000202U,+ 0x00000000U, 0x00000002U, 0x00000000U, 0x00000002U, 0x00000200U, 0x00000202U, 0x00000200U, 0x00000202U,+ 0x00020000U, 0x00020002U, 0x00020000U, 0x00020002U, 0x00020200U, 0x00020202U, 0x00020200U, 0x00020202U,+ 0x00020000U, 0x00020002U, 0x00020000U, 0x00020002U, 0x00020200U, 0x00020202U, 0x00020200U, 0x00020202U,+ 0x02000000U, 0x02000002U, 0x02000000U, 0x02000002U, 0x02000200U, 0x02000202U, 0x02000200U, 0x02000202U,+ 0x02000000U, 0x02000002U, 0x02000000U, 0x02000002U, 0x02000200U, 0x02000202U, 0x02000200U, 0x02000202U,+ 0x02020000U, 0x02020002U, 0x02020000U, 0x02020002U, 0x02020200U, 0x02020202U, 0x02020200U, 0x02020202U,+ 0x02020000U, 0x02020002U, 0x02020000U, 0x02020002U, 0x02020200U, 0x02020202U, 0x02020200U, 0x02020202U,+ 0x02000000U, 0x02000002U, 0x02000000U, 0x02000002U, 0x02000200U, 0x02000202U, 0x02000200U, 0x02000202U,+ 0x02000000U, 0x02000002U, 0x02000000U, 0x02000002U, 0x02000200U, 0x02000202U, 0x02000200U, 0x02000202U,+ 0x02020000U, 0x02020002U, 0x02020000U, 0x02020002U, 0x02020200U, 0x02020202U, 0x02020200U, 0x02020202U,+ 0x02020000U, 0x02020002U, 0x02020000U, 0x02020002U, 0x02020200U, 0x02020202U, 0x02020200U, 0x02020202U,+},+{+ 0x00000000U, 0x00000004U, 0x00000000U, 0x00000004U, 0x00000400U, 0x00000404U, 0x00000400U, 0x00000404U,+ 0x00000000U, 0x00000004U, 0x00000000U, 0x00000004U, 0x00000400U, 0x00000404U, 0x00000400U, 0x00000404U,+ 0x00040000U, 0x00040004U, 0x00040000U, 0x00040004U, 0x00040400U, 0x00040404U, 0x00040400U, 0x00040404U,+ 0x00040000U, 0x00040004U, 0x00040000U, 0x00040004U, 0x00040400U, 0x00040404U, 0x00040400U, 0x00040404U,+ 0x00000000U, 0x00000004U, 0x00000000U, 0x00000004U, 0x00000400U, 0x00000404U, 0x00000400U, 0x00000404U,+ 0x00000000U, 0x00000004U, 0x00000000U, 0x00000004U, 0x00000400U, 0x00000404U, 0x00000400U, 0x00000404U,+ 0x00040000U, 0x00040004U, 0x00040000U, 0x00040004U, 0x00040400U, 0x00040404U, 0x00040400U, 0x00040404U,+ 0x00040000U, 0x00040004U, 0x00040000U, 0x00040004U, 0x00040400U, 0x00040404U, 0x00040400U, 0x00040404U,+ 0x04000000U, 0x04000004U, 0x04000000U, 0x04000004U, 0x04000400U, 0x04000404U, 0x04000400U, 0x04000404U,+ 0x04000000U, 0x04000004U, 0x04000000U, 0x04000004U, 0x04000400U, 0x04000404U, 0x04000400U, 0x04000404U,+ 0x04040000U, 0x04040004U, 0x04040000U, 0x04040004U, 0x04040400U, 0x04040404U, 0x04040400U, 0x04040404U,+ 0x04040000U, 0x04040004U, 0x04040000U, 0x04040004U, 0x04040400U, 0x04040404U, 0x04040400U, 0x04040404U,+ 0x04000000U, 0x04000004U, 0x04000000U, 0x04000004U, 0x04000400U, 0x04000404U, 0x04000400U, 0x04000404U,+ 0x04000000U, 0x04000004U, 0x04000000U, 0x04000004U, 0x04000400U, 0x04000404U, 0x04000400U, 0x04000404U,+ 0x04040000U, 0x04040004U, 0x04040000U, 0x04040004U, 0x04040400U, 0x04040404U, 0x04040400U, 0x04040404U,+ 0x04040000U, 0x04040004U, 0x04040000U, 0x04040004U, 0x04040400U, 0x04040404U, 0x04040400U, 0x04040404U,+ 0x00000000U, 0x00000004U, 0x00000000U, 0x00000004U, 0x00000400U, 0x00000404U, 0x00000400U, 0x00000404U,+ 0x00000000U, 0x00000004U, 0x00000000U, 0x00000004U, 0x00000400U, 0x00000404U, 0x00000400U, 0x00000404U,+ 0x00040000U, 0x00040004U, 0x00040000U, 0x00040004U, 0x00040400U, 0x00040404U, 0x00040400U, 0x00040404U,+ 0x00040000U, 0x00040004U, 0x00040000U, 0x00040004U, 0x00040400U, 0x00040404U, 0x00040400U, 0x00040404U,+ 0x00000000U, 0x00000004U, 0x00000000U, 0x00000004U, 0x00000400U, 0x00000404U, 0x00000400U, 0x00000404U,+ 0x00000000U, 0x00000004U, 0x00000000U, 0x00000004U, 0x00000400U, 0x00000404U, 0x00000400U, 0x00000404U,+ 0x00040000U, 0x00040004U, 0x00040000U, 0x00040004U, 0x00040400U, 0x00040404U, 0x00040400U, 0x00040404U,+ 0x00040000U, 0x00040004U, 0x00040000U, 0x00040004U, 0x00040400U, 0x00040404U, 0x00040400U, 0x00040404U,+ 0x04000000U, 0x04000004U, 0x04000000U, 0x04000004U, 0x04000400U, 0x04000404U, 0x04000400U, 0x04000404U,+ 0x04000000U, 0x04000004U, 0x04000000U, 0x04000004U, 0x04000400U, 0x04000404U, 0x04000400U, 0x04000404U,+ 0x04040000U, 0x04040004U, 0x04040000U, 0x04040004U, 0x04040400U, 0x04040404U, 0x04040400U, 0x04040404U,+ 0x04040000U, 0x04040004U, 0x04040000U, 0x04040004U, 0x04040400U, 0x04040404U, 0x04040400U, 0x04040404U,+ 0x04000000U, 0x04000004U, 0x04000000U, 0x04000004U, 0x04000400U, 0x04000404U, 0x04000400U, 0x04000404U,+ 0x04000000U, 0x04000004U, 0x04000000U, 0x04000004U, 0x04000400U, 0x04000404U, 0x04000400U, 0x04000404U,+ 0x04040000U, 0x04040004U, 0x04040000U, 0x04040004U, 0x04040400U, 0x04040404U, 0x04040400U, 0x04040404U,+ 0x04040000U, 0x04040004U, 0x04040000U, 0x04040004U, 0x04040400U, 0x04040404U, 0x04040400U, 0x04040404U,+},+{+ 0x00000000U, 0x00000008U, 0x00000000U, 0x00000008U, 0x00000800U, 0x00000808U, 0x00000800U, 0x00000808U,+ 0x00000000U, 0x00000008U, 0x00000000U, 0x00000008U, 0x00000800U, 0x00000808U, 0x00000800U, 0x00000808U,+ 0x00080000U, 0x00080008U, 0x00080000U, 0x00080008U, 0x00080800U, 0x00080808U, 0x00080800U, 0x00080808U,+ 0x00080000U, 0x00080008U, 0x00080000U, 0x00080008U, 0x00080800U, 0x00080808U, 0x00080800U, 0x00080808U,+ 0x00000000U, 0x00000008U, 0x00000000U, 0x00000008U, 0x00000800U, 0x00000808U, 0x00000800U, 0x00000808U,+ 0x00000000U, 0x00000008U, 0x00000000U, 0x00000008U, 0x00000800U, 0x00000808U, 0x00000800U, 0x00000808U,+ 0x00080000U, 0x00080008U, 0x00080000U, 0x00080008U, 0x00080800U, 0x00080808U, 0x00080800U, 0x00080808U,+ 0x00080000U, 0x00080008U, 0x00080000U, 0x00080008U, 0x00080800U, 0x00080808U, 0x00080800U, 0x00080808U,+ 0x08000000U, 0x08000008U, 0x08000000U, 0x08000008U, 0x08000800U, 0x08000808U, 0x08000800U, 0x08000808U,+ 0x08000000U, 0x08000008U, 0x08000000U, 0x08000008U, 0x08000800U, 0x08000808U, 0x08000800U, 0x08000808U,+ 0x08080000U, 0x08080008U, 0x08080000U, 0x08080008U, 0x08080800U, 0x08080808U, 0x08080800U, 0x08080808U,+ 0x08080000U, 0x08080008U, 0x08080000U, 0x08080008U, 0x08080800U, 0x08080808U, 0x08080800U, 0x08080808U,+ 0x08000000U, 0x08000008U, 0x08000000U, 0x08000008U, 0x08000800U, 0x08000808U, 0x08000800U, 0x08000808U,+ 0x08000000U, 0x08000008U, 0x08000000U, 0x08000008U, 0x08000800U, 0x08000808U, 0x08000800U, 0x08000808U,+ 0x08080000U, 0x08080008U, 0x08080000U, 0x08080008U, 0x08080800U, 0x08080808U, 0x08080800U, 0x08080808U,+ 0x08080000U, 0x08080008U, 0x08080000U, 0x08080008U, 0x08080800U, 0x08080808U, 0x08080800U, 0x08080808U,+ 0x00000000U, 0x00000008U, 0x00000000U, 0x00000008U, 0x00000800U, 0x00000808U, 0x00000800U, 0x00000808U,+ 0x00000000U, 0x00000008U, 0x00000000U, 0x00000008U, 0x00000800U, 0x00000808U, 0x00000800U, 0x00000808U,+ 0x00080000U, 0x00080008U, 0x00080000U, 0x00080008U, 0x00080800U, 0x00080808U, 0x00080800U, 0x00080808U,+ 0x00080000U, 0x00080008U, 0x00080000U, 0x00080008U, 0x00080800U, 0x00080808U, 0x00080800U, 0x00080808U,+ 0x00000000U, 0x00000008U, 0x00000000U, 0x00000008U, 0x00000800U, 0x00000808U, 0x00000800U, 0x00000808U,+ 0x00000000U, 0x00000008U, 0x00000000U, 0x00000008U, 0x00000800U, 0x00000808U, 0x00000800U, 0x00000808U,+ 0x00080000U, 0x00080008U, 0x00080000U, 0x00080008U, 0x00080800U, 0x00080808U, 0x00080800U, 0x00080808U,+ 0x00080000U, 0x00080008U, 0x00080000U, 0x00080008U, 0x00080800U, 0x00080808U, 0x00080800U, 0x00080808U,+ 0x08000000U, 0x08000008U, 0x08000000U, 0x08000008U, 0x08000800U, 0x08000808U, 0x08000800U, 0x08000808U,+ 0x08000000U, 0x08000008U, 0x08000000U, 0x08000008U, 0x08000800U, 0x08000808U, 0x08000800U, 0x08000808U,+ 0x08080000U, 0x08080008U, 0x08080000U, 0x08080008U, 0x08080800U, 0x08080808U, 0x08080800U, 0x08080808U,+ 0x08080000U, 0x08080008U, 0x08080000U, 0x08080008U, 0x08080800U, 0x08080808U, 0x08080800U, 0x08080808U,+ 0x08000000U, 0x08000008U, 0x08000000U, 0x08000008U, 0x08000800U, 0x08000808U, 0x08000800U, 0x08000808U,+ 0x08000000U, 0x08000008U, 0x08000000U, 0x08000008U, 0x08000800U, 0x08000808U, 0x08000800U, 0x08000808U,+ 0x08080000U, 0x08080008U, 0x08080000U, 0x08080008U, 0x08080800U, 0x08080808U, 0x08080800U, 0x08080808U,+ 0x08080000U, 0x08080008U, 0x08080000U, 0x08080008U, 0x08080800U, 0x08080808U, 0x08080800U, 0x08080808U,+},+{+ 0x00000000U, 0x00000010U, 0x00000000U, 0x00000010U, 0x00001000U, 0x00001010U, 0x00001000U, 0x00001010U,+ 0x00000000U, 0x00000010U, 0x00000000U, 0x00000010U, 0x00001000U, 0x00001010U, 0x00001000U, 0x00001010U,+ 0x00100000U, 0x00100010U, 0x00100000U, 0x00100010U, 0x00101000U, 0x00101010U, 0x00101000U, 0x00101010U,+ 0x00100000U, 0x00100010U, 0x00100000U, 0x00100010U, 0x00101000U, 0x00101010U, 0x00101000U, 0x00101010U,+ 0x00000000U, 0x00000010U, 0x00000000U, 0x00000010U, 0x00001000U, 0x00001010U, 0x00001000U, 0x00001010U,+ 0x00000000U, 0x00000010U, 0x00000000U, 0x00000010U, 0x00001000U, 0x00001010U, 0x00001000U, 0x00001010U,+ 0x00100000U, 0x00100010U, 0x00100000U, 0x00100010U, 0x00101000U, 0x00101010U, 0x00101000U, 0x00101010U,+ 0x00100000U, 0x00100010U, 0x00100000U, 0x00100010U, 0x00101000U, 0x00101010U, 0x00101000U, 0x00101010U,+ 0x10000000U, 0x10000010U, 0x10000000U, 0x10000010U, 0x10001000U, 0x10001010U, 0x10001000U, 0x10001010U,+ 0x10000000U, 0x10000010U, 0x10000000U, 0x10000010U, 0x10001000U, 0x10001010U, 0x10001000U, 0x10001010U,+ 0x10100000U, 0x10100010U, 0x10100000U, 0x10100010U, 0x10101000U, 0x10101010U, 0x10101000U, 0x10101010U,+ 0x10100000U, 0x10100010U, 0x10100000U, 0x10100010U, 0x10101000U, 0x10101010U, 0x10101000U, 0x10101010U,+ 0x10000000U, 0x10000010U, 0x10000000U, 0x10000010U, 0x10001000U, 0x10001010U, 0x10001000U, 0x10001010U,+ 0x10000000U, 0x10000010U, 0x10000000U, 0x10000010U, 0x10001000U, 0x10001010U, 0x10001000U, 0x10001010U,+ 0x10100000U, 0x10100010U, 0x10100000U, 0x10100010U, 0x10101000U, 0x10101010U, 0x10101000U, 0x10101010U,+ 0x10100000U, 0x10100010U, 0x10100000U, 0x10100010U, 0x10101000U, 0x10101010U, 0x10101000U, 0x10101010U,+ 0x00000000U, 0x00000010U, 0x00000000U, 0x00000010U, 0x00001000U, 0x00001010U, 0x00001000U, 0x00001010U,+ 0x00000000U, 0x00000010U, 0x00000000U, 0x00000010U, 0x00001000U, 0x00001010U, 0x00001000U, 0x00001010U,+ 0x00100000U, 0x00100010U, 0x00100000U, 0x00100010U, 0x00101000U, 0x00101010U, 0x00101000U, 0x00101010U,+ 0x00100000U, 0x00100010U, 0x00100000U, 0x00100010U, 0x00101000U, 0x00101010U, 0x00101000U, 0x00101010U,+ 0x00000000U, 0x00000010U, 0x00000000U, 0x00000010U, 0x00001000U, 0x00001010U, 0x00001000U, 0x00001010U,+ 0x00000000U, 0x00000010U, 0x00000000U, 0x00000010U, 0x00001000U, 0x00001010U, 0x00001000U, 0x00001010U,+ 0x00100000U, 0x00100010U, 0x00100000U, 0x00100010U, 0x00101000U, 0x00101010U, 0x00101000U, 0x00101010U,+ 0x00100000U, 0x00100010U, 0x00100000U, 0x00100010U, 0x00101000U, 0x00101010U, 0x00101000U, 0x00101010U,+ 0x10000000U, 0x10000010U, 0x10000000U, 0x10000010U, 0x10001000U, 0x10001010U, 0x10001000U, 0x10001010U,+ 0x10000000U, 0x10000010U, 0x10000000U, 0x10000010U, 0x10001000U, 0x10001010U, 0x10001000U, 0x10001010U,+ 0x10100000U, 0x10100010U, 0x10100000U, 0x10100010U, 0x10101000U, 0x10101010U, 0x10101000U, 0x10101010U,+ 0x10100000U, 0x10100010U, 0x10100000U, 0x10100010U, 0x10101000U, 0x10101010U, 0x10101000U, 0x10101010U,+ 0x10000000U, 0x10000010U, 0x10000000U, 0x10000010U, 0x10001000U, 0x10001010U, 0x10001000U, 0x10001010U,+ 0x10000000U, 0x10000010U, 0x10000000U, 0x10000010U, 0x10001000U, 0x10001010U, 0x10001000U, 0x10001010U,+ 0x10100000U, 0x10100010U, 0x10100000U, 0x10100010U, 0x10101000U, 0x10101010U, 0x10101000U, 0x10101010U,+ 0x10100000U, 0x10100010U, 0x10100000U, 0x10100010U, 0x10101000U, 0x10101010U, 0x10101000U, 0x10101010U,+},+{+ 0x00000000U, 0x00000020U, 0x00000000U, 0x00000020U, 0x00002000U, 0x00002020U, 0x00002000U, 0x00002020U,+ 0x00000000U, 0x00000020U, 0x00000000U, 0x00000020U, 0x00002000U, 0x00002020U, 0x00002000U, 0x00002020U,+ 0x00200000U, 0x00200020U, 0x00200000U, 0x00200020U, 0x00202000U, 0x00202020U, 0x00202000U, 0x00202020U,+ 0x00200000U, 0x00200020U, 0x00200000U, 0x00200020U, 0x00202000U, 0x00202020U, 0x00202000U, 0x00202020U,+ 0x00000000U, 0x00000020U, 0x00000000U, 0x00000020U, 0x00002000U, 0x00002020U, 0x00002000U, 0x00002020U,+ 0x00000000U, 0x00000020U, 0x00000000U, 0x00000020U, 0x00002000U, 0x00002020U, 0x00002000U, 0x00002020U,+ 0x00200000U, 0x00200020U, 0x00200000U, 0x00200020U, 0x00202000U, 0x00202020U, 0x00202000U, 0x00202020U,+ 0x00200000U, 0x00200020U, 0x00200000U, 0x00200020U, 0x00202000U, 0x00202020U, 0x00202000U, 0x00202020U,+ 0x20000000U, 0x20000020U, 0x20000000U, 0x20000020U, 0x20002000U, 0x20002020U, 0x20002000U, 0x20002020U,+ 0x20000000U, 0x20000020U, 0x20000000U, 0x20000020U, 0x20002000U, 0x20002020U, 0x20002000U, 0x20002020U,+ 0x20200000U, 0x20200020U, 0x20200000U, 0x20200020U, 0x20202000U, 0x20202020U, 0x20202000U, 0x20202020U,+ 0x20200000U, 0x20200020U, 0x20200000U, 0x20200020U, 0x20202000U, 0x20202020U, 0x20202000U, 0x20202020U,+ 0x20000000U, 0x20000020U, 0x20000000U, 0x20000020U, 0x20002000U, 0x20002020U, 0x20002000U, 0x20002020U,+ 0x20000000U, 0x20000020U, 0x20000000U, 0x20000020U, 0x20002000U, 0x20002020U, 0x20002000U, 0x20002020U,+ 0x20200000U, 0x20200020U, 0x20200000U, 0x20200020U, 0x20202000U, 0x20202020U, 0x20202000U, 0x20202020U,+ 0x20200000U, 0x20200020U, 0x20200000U, 0x20200020U, 0x20202000U, 0x20202020U, 0x20202000U, 0x20202020U,+ 0x00000000U, 0x00000020U, 0x00000000U, 0x00000020U, 0x00002000U, 0x00002020U, 0x00002000U, 0x00002020U,+ 0x00000000U, 0x00000020U, 0x00000000U, 0x00000020U, 0x00002000U, 0x00002020U, 0x00002000U, 0x00002020U,+ 0x00200000U, 0x00200020U, 0x00200000U, 0x00200020U, 0x00202000U, 0x00202020U, 0x00202000U, 0x00202020U,+ 0x00200000U, 0x00200020U, 0x00200000U, 0x00200020U, 0x00202000U, 0x00202020U, 0x00202000U, 0x00202020U,+ 0x00000000U, 0x00000020U, 0x00000000U, 0x00000020U, 0x00002000U, 0x00002020U, 0x00002000U, 0x00002020U,+ 0x00000000U, 0x00000020U, 0x00000000U, 0x00000020U, 0x00002000U, 0x00002020U, 0x00002000U, 0x00002020U,+ 0x00200000U, 0x00200020U, 0x00200000U, 0x00200020U, 0x00202000U, 0x00202020U, 0x00202000U, 0x00202020U,+ 0x00200000U, 0x00200020U, 0x00200000U, 0x00200020U, 0x00202000U, 0x00202020U, 0x00202000U, 0x00202020U,+ 0x20000000U, 0x20000020U, 0x20000000U, 0x20000020U, 0x20002000U, 0x20002020U, 0x20002000U, 0x20002020U,+ 0x20000000U, 0x20000020U, 0x20000000U, 0x20000020U, 0x20002000U, 0x20002020U, 0x20002000U, 0x20002020U,+ 0x20200000U, 0x20200020U, 0x20200000U, 0x20200020U, 0x20202000U, 0x20202020U, 0x20202000U, 0x20202020U,+ 0x20200000U, 0x20200020U, 0x20200000U, 0x20200020U, 0x20202000U, 0x20202020U, 0x20202000U, 0x20202020U,+ 0x20000000U, 0x20000020U, 0x20000000U, 0x20000020U, 0x20002000U, 0x20002020U, 0x20002000U, 0x20002020U,+ 0x20000000U, 0x20000020U, 0x20000000U, 0x20000020U, 0x20002000U, 0x20002020U, 0x20002000U, 0x20002020U,+ 0x20200000U, 0x20200020U, 0x20200000U, 0x20200020U, 0x20202000U, 0x20202020U, 0x20202000U, 0x20202020U,+ 0x20200000U, 0x20200020U, 0x20200000U, 0x20200020U, 0x20202000U, 0x20202020U, 0x20202000U, 0x20202020U,+},+{+ 0x00000000U, 0x00000040U, 0x00000000U, 0x00000040U, 0x00004000U, 0x00004040U, 0x00004000U, 0x00004040U,+ 0x00000000U, 0x00000040U, 0x00000000U, 0x00000040U, 0x00004000U, 0x00004040U, 0x00004000U, 0x00004040U,+ 0x00400000U, 0x00400040U, 0x00400000U, 0x00400040U, 0x00404000U, 0x00404040U, 0x00404000U, 0x00404040U,+ 0x00400000U, 0x00400040U, 0x00400000U, 0x00400040U, 0x00404000U, 0x00404040U, 0x00404000U, 0x00404040U,+ 0x00000000U, 0x00000040U, 0x00000000U, 0x00000040U, 0x00004000U, 0x00004040U, 0x00004000U, 0x00004040U,+ 0x00000000U, 0x00000040U, 0x00000000U, 0x00000040U, 0x00004000U, 0x00004040U, 0x00004000U, 0x00004040U,+ 0x00400000U, 0x00400040U, 0x00400000U, 0x00400040U, 0x00404000U, 0x00404040U, 0x00404000U, 0x00404040U,+ 0x00400000U, 0x00400040U, 0x00400000U, 0x00400040U, 0x00404000U, 0x00404040U, 0x00404000U, 0x00404040U,+ 0x40000000U, 0x40000040U, 0x40000000U, 0x40000040U, 0x40004000U, 0x40004040U, 0x40004000U, 0x40004040U,+ 0x40000000U, 0x40000040U, 0x40000000U, 0x40000040U, 0x40004000U, 0x40004040U, 0x40004000U, 0x40004040U,+ 0x40400000U, 0x40400040U, 0x40400000U, 0x40400040U, 0x40404000U, 0x40404040U, 0x40404000U, 0x40404040U,+ 0x40400000U, 0x40400040U, 0x40400000U, 0x40400040U, 0x40404000U, 0x40404040U, 0x40404000U, 0x40404040U,+ 0x40000000U, 0x40000040U, 0x40000000U, 0x40000040U, 0x40004000U, 0x40004040U, 0x40004000U, 0x40004040U,+ 0x40000000U, 0x40000040U, 0x40000000U, 0x40000040U, 0x40004000U, 0x40004040U, 0x40004000U, 0x40004040U,+ 0x40400000U, 0x40400040U, 0x40400000U, 0x40400040U, 0x40404000U, 0x40404040U, 0x40404000U, 0x40404040U,+ 0x40400000U, 0x40400040U, 0x40400000U, 0x40400040U, 0x40404000U, 0x40404040U, 0x40404000U, 0x40404040U,+ 0x00000000U, 0x00000040U, 0x00000000U, 0x00000040U, 0x00004000U, 0x00004040U, 0x00004000U, 0x00004040U,+ 0x00000000U, 0x00000040U, 0x00000000U, 0x00000040U, 0x00004000U, 0x00004040U, 0x00004000U, 0x00004040U,+ 0x00400000U, 0x00400040U, 0x00400000U, 0x00400040U, 0x00404000U, 0x00404040U, 0x00404000U, 0x00404040U,+ 0x00400000U, 0x00400040U, 0x00400000U, 0x00400040U, 0x00404000U, 0x00404040U, 0x00404000U, 0x00404040U,+ 0x00000000U, 0x00000040U, 0x00000000U, 0x00000040U, 0x00004000U, 0x00004040U, 0x00004000U, 0x00004040U,+ 0x00000000U, 0x00000040U, 0x00000000U, 0x00000040U, 0x00004000U, 0x00004040U, 0x00004000U, 0x00004040U,+ 0x00400000U, 0x00400040U, 0x00400000U, 0x00400040U, 0x00404000U, 0x00404040U, 0x00404000U, 0x00404040U,+ 0x00400000U, 0x00400040U, 0x00400000U, 0x00400040U, 0x00404000U, 0x00404040U, 0x00404000U, 0x00404040U,+ 0x40000000U, 0x40000040U, 0x40000000U, 0x40000040U, 0x40004000U, 0x40004040U, 0x40004000U, 0x40004040U,+ 0x40000000U, 0x40000040U, 0x40000000U, 0x40000040U, 0x40004000U, 0x40004040U, 0x40004000U, 0x40004040U,+ 0x40400000U, 0x40400040U, 0x40400000U, 0x40400040U, 0x40404000U, 0x40404040U, 0x40404000U, 0x40404040U,+ 0x40400000U, 0x40400040U, 0x40400000U, 0x40400040U, 0x40404000U, 0x40404040U, 0x40404000U, 0x40404040U,+ 0x40000000U, 0x40000040U, 0x40000000U, 0x40000040U, 0x40004000U, 0x40004040U, 0x40004000U, 0x40004040U,+ 0x40000000U, 0x40000040U, 0x40000000U, 0x40000040U, 0x40004000U, 0x40004040U, 0x40004000U, 0x40004040U,+ 0x40400000U, 0x40400040U, 0x40400000U, 0x40400040U, 0x40404000U, 0x40404040U, 0x40404000U, 0x40404040U,+ 0x40400000U, 0x40400040U, 0x40400000U, 0x40400040U, 0x40404000U, 0x40404040U, 0x40404000U, 0x40404040U,+},+{+ 0x00000000U, 0x00000080U, 0x00000000U, 0x00000080U, 0x00008000U, 0x00008080U, 0x00008000U, 0x00008080U,+ 0x00000000U, 0x00000080U, 0x00000000U, 0x00000080U, 0x00008000U, 0x00008080U, 0x00008000U, 0x00008080U,+ 0x00800000U, 0x00800080U, 0x00800000U, 0x00800080U, 0x00808000U, 0x00808080U, 0x00808000U, 0x00808080U,+ 0x00800000U, 0x00800080U, 0x00800000U, 0x00800080U, 0x00808000U, 0x00808080U, 0x00808000U, 0x00808080U,+ 0x00000000U, 0x00000080U, 0x00000000U, 0x00000080U, 0x00008000U, 0x00008080U, 0x00008000U, 0x00008080U,+ 0x00000000U, 0x00000080U, 0x00000000U, 0x00000080U, 0x00008000U, 0x00008080U, 0x00008000U, 0x00008080U,+ 0x00800000U, 0x00800080U, 0x00800000U, 0x00800080U, 0x00808000U, 0x00808080U, 0x00808000U, 0x00808080U,+ 0x00800000U, 0x00800080U, 0x00800000U, 0x00800080U, 0x00808000U, 0x00808080U, 0x00808000U, 0x00808080U,+ 0x80000000U, 0x80000080U, 0x80000000U, 0x80000080U, 0x80008000U, 0x80008080U, 0x80008000U, 0x80008080U,+ 0x80000000U, 0x80000080U, 0x80000000U, 0x80000080U, 0x80008000U, 0x80008080U, 0x80008000U, 0x80008080U,+ 0x80800000U, 0x80800080U, 0x80800000U, 0x80800080U, 0x80808000U, 0x80808080U, 0x80808000U, 0x80808080U,+ 0x80800000U, 0x80800080U, 0x80800000U, 0x80800080U, 0x80808000U, 0x80808080U, 0x80808000U, 0x80808080U,+ 0x80000000U, 0x80000080U, 0x80000000U, 0x80000080U, 0x80008000U, 0x80008080U, 0x80008000U, 0x80008080U,+ 0x80000000U, 0x80000080U, 0x80000000U, 0x80000080U, 0x80008000U, 0x80008080U, 0x80008000U, 0x80008080U,+ 0x80800000U, 0x80800080U, 0x80800000U, 0x80800080U, 0x80808000U, 0x80808080U, 0x80808000U, 0x80808080U,+ 0x80800000U, 0x80800080U, 0x80800000U, 0x80800080U, 0x80808000U, 0x80808080U, 0x80808000U, 0x80808080U,+ 0x00000000U, 0x00000080U, 0x00000000U, 0x00000080U, 0x00008000U, 0x00008080U, 0x00008000U, 0x00008080U,+ 0x00000000U, 0x00000080U, 0x00000000U, 0x00000080U, 0x00008000U, 0x00008080U, 0x00008000U, 0x00008080U,+ 0x00800000U, 0x00800080U, 0x00800000U, 0x00800080U, 0x00808000U, 0x00808080U, 0x00808000U, 0x00808080U,+ 0x00800000U, 0x00800080U, 0x00800000U, 0x00800080U, 0x00808000U, 0x00808080U, 0x00808000U, 0x00808080U,+ 0x00000000U, 0x00000080U, 0x00000000U, 0x00000080U, 0x00008000U, 0x00008080U, 0x00008000U, 0x00008080U,+ 0x00000000U, 0x00000080U, 0x00000000U, 0x00000080U, 0x00008000U, 0x00008080U, 0x00008000U, 0x00008080U,+ 0x00800000U, 0x00800080U, 0x00800000U, 0x00800080U, 0x00808000U, 0x00808080U, 0x00808000U, 0x00808080U,+ 0x00800000U, 0x00800080U, 0x00800000U, 0x00800080U, 0x00808000U, 0x00808080U, 0x00808000U, 0x00808080U,+ 0x80000000U, 0x80000080U, 0x80000000U, 0x80000080U, 0x80008000U, 0x80008080U, 0x80008000U, 0x80008080U,+ 0x80000000U, 0x80000080U, 0x80000000U, 0x80000080U, 0x80008000U, 0x80008080U, 0x80008000U, 0x80008080U,+ 0x80800000U, 0x80800080U, 0x80800000U, 0x80800080U, 0x80808000U, 0x80808080U, 0x80808000U, 0x80808080U,+ 0x80800000U, 0x80800080U, 0x80800000U, 0x80800080U, 0x80808000U, 0x80808080U, 0x80808000U, 0x80808080U,+ 0x80000000U, 0x80000080U, 0x80000000U, 0x80000080U, 0x80008000U, 0x80008080U, 0x80008000U, 0x80008080U,+ 0x80000000U, 0x80000080U, 0x80000000U, 0x80000080U, 0x80008000U, 0x80008080U, 0x80008000U, 0x80008080U,+ 0x80800000U, 0x80800080U, 0x80800000U, 0x80800080U, 0x80808000U, 0x80808080U, 0x80808000U, 0x80808080U,+ 0x80800000U, 0x80800080U, 0x80800000U, 0x80800080U, 0x80808000U, 0x80808080U, 0x80808000U, 0x80808080U,+},+};++static const uint32_t IPR[8][256] = {+{+ 0x00000000U, 0x00000000U, 0x00000001U, 0x00000001U, 0x00000000U, 0x00000000U, 0x00000001U, 0x00000001U,+ 0x00000100U, 0x00000100U, 0x00000101U, 0x00000101U, 0x00000100U, 0x00000100U, 0x00000101U, 0x00000101U,+ 0x00000000U, 0x00000000U, 0x00000001U, 0x00000001U, 0x00000000U, 0x00000000U, 0x00000001U, 0x00000001U,+ 0x00000100U, 0x00000100U, 0x00000101U, 0x00000101U, 0x00000100U, 0x00000100U, 0x00000101U, 0x00000101U,+ 0x00010000U, 0x00010000U, 0x00010001U, 0x00010001U, 0x00010000U, 0x00010000U, 0x00010001U, 0x00010001U,+ 0x00010100U, 0x00010100U, 0x00010101U, 0x00010101U, 0x00010100U, 0x00010100U, 0x00010101U, 0x00010101U,+ 0x00010000U, 0x00010000U, 0x00010001U, 0x00010001U, 0x00010000U, 0x00010000U, 0x00010001U, 0x00010001U,+ 0x00010100U, 0x00010100U, 0x00010101U, 0x00010101U, 0x00010100U, 0x00010100U, 0x00010101U, 0x00010101U,+ 0x00000000U, 0x00000000U, 0x00000001U, 0x00000001U, 0x00000000U, 0x00000000U, 0x00000001U, 0x00000001U,+ 0x00000100U, 0x00000100U, 0x00000101U, 0x00000101U, 0x00000100U, 0x00000100U, 0x00000101U, 0x00000101U,+ 0x00000000U, 0x00000000U, 0x00000001U, 0x00000001U, 0x00000000U, 0x00000000U, 0x00000001U, 0x00000001U,+ 0x00000100U, 0x00000100U, 0x00000101U, 0x00000101U, 0x00000100U, 0x00000100U, 0x00000101U, 0x00000101U,+ 0x00010000U, 0x00010000U, 0x00010001U, 0x00010001U, 0x00010000U, 0x00010000U, 0x00010001U, 0x00010001U,+ 0x00010100U, 0x00010100U, 0x00010101U, 0x00010101U, 0x00010100U, 0x00010100U, 0x00010101U, 0x00010101U,+ 0x00010000U, 0x00010000U, 0x00010001U, 0x00010001U, 0x00010000U, 0x00010000U, 0x00010001U, 0x00010001U,+ 0x00010100U, 0x00010100U, 0x00010101U, 0x00010101U, 0x00010100U, 0x00010100U, 0x00010101U, 0x00010101U,+ 0x01000000U, 0x01000000U, 0x01000001U, 0x01000001U, 0x01000000U, 0x01000000U, 0x01000001U, 0x01000001U,+ 0x01000100U, 0x01000100U, 0x01000101U, 0x01000101U, 0x01000100U, 0x01000100U, 0x01000101U, 0x01000101U,+ 0x01000000U, 0x01000000U, 0x01000001U, 0x01000001U, 0x01000000U, 0x01000000U, 0x01000001U, 0x01000001U,+ 0x01000100U, 0x01000100U, 0x01000101U, 0x01000101U, 0x01000100U, 0x01000100U, 0x01000101U, 0x01000101U,+ 0x01010000U, 0x01010000U, 0x01010001U, 0x01010001U, 0x01010000U, 0x01010000U, 0x01010001U, 0x01010001U,+ 0x01010100U, 0x01010100U, 0x01010101U, 0x01010101U, 0x01010100U, 0x01010100U, 0x01010101U, 0x01010101U,+ 0x01010000U, 0x01010000U, 0x01010001U, 0x01010001U, 0x01010000U, 0x01010000U, 0x01010001U, 0x01010001U,+ 0x01010100U, 0x01010100U, 0x01010101U, 0x01010101U, 0x01010100U, 0x01010100U, 0x01010101U, 0x01010101U,+ 0x01000000U, 0x01000000U, 0x01000001U, 0x01000001U, 0x01000000U, 0x01000000U, 0x01000001U, 0x01000001U,+ 0x01000100U, 0x01000100U, 0x01000101U, 0x01000101U, 0x01000100U, 0x01000100U, 0x01000101U, 0x01000101U,+ 0x01000000U, 0x01000000U, 0x01000001U, 0x01000001U, 0x01000000U, 0x01000000U, 0x01000001U, 0x01000001U,+ 0x01000100U, 0x01000100U, 0x01000101U, 0x01000101U, 0x01000100U, 0x01000100U, 0x01000101U, 0x01000101U,+ 0x01010000U, 0x01010000U, 0x01010001U, 0x01010001U, 0x01010000U, 0x01010000U, 0x01010001U, 0x01010001U,+ 0x01010100U, 0x01010100U, 0x01010101U, 0x01010101U, 0x01010100U, 0x01010100U, 0x01010101U, 0x01010101U,+ 0x01010000U, 0x01010000U, 0x01010001U, 0x01010001U, 0x01010000U, 0x01010000U, 0x01010001U, 0x01010001U,+ 0x01010100U, 0x01010100U, 0x01010101U, 0x01010101U, 0x01010100U, 0x01010100U, 0x01010101U, 0x01010101U,+},+{+ 0x00000000U, 0x00000000U, 0x00000002U, 0x00000002U, 0x00000000U, 0x00000000U, 0x00000002U, 0x00000002U,+ 0x00000200U, 0x00000200U, 0x00000202U, 0x00000202U, 0x00000200U, 0x00000200U, 0x00000202U, 0x00000202U,+ 0x00000000U, 0x00000000U, 0x00000002U, 0x00000002U, 0x00000000U, 0x00000000U, 0x00000002U, 0x00000002U,+ 0x00000200U, 0x00000200U, 0x00000202U, 0x00000202U, 0x00000200U, 0x00000200U, 0x00000202U, 0x00000202U,+ 0x00020000U, 0x00020000U, 0x00020002U, 0x00020002U, 0x00020000U, 0x00020000U, 0x00020002U, 0x00020002U,+ 0x00020200U, 0x00020200U, 0x00020202U, 0x00020202U, 0x00020200U, 0x00020200U, 0x00020202U, 0x00020202U,+ 0x00020000U, 0x00020000U, 0x00020002U, 0x00020002U, 0x00020000U, 0x00020000U, 0x00020002U, 0x00020002U,+ 0x00020200U, 0x00020200U, 0x00020202U, 0x00020202U, 0x00020200U, 0x00020200U, 0x00020202U, 0x00020202U,+ 0x00000000U, 0x00000000U, 0x00000002U, 0x00000002U, 0x00000000U, 0x00000000U, 0x00000002U, 0x00000002U,+ 0x00000200U, 0x00000200U, 0x00000202U, 0x00000202U, 0x00000200U, 0x00000200U, 0x00000202U, 0x00000202U,+ 0x00000000U, 0x00000000U, 0x00000002U, 0x00000002U, 0x00000000U, 0x00000000U, 0x00000002U, 0x00000002U,+ 0x00000200U, 0x00000200U, 0x00000202U, 0x00000202U, 0x00000200U, 0x00000200U, 0x00000202U, 0x00000202U,+ 0x00020000U, 0x00020000U, 0x00020002U, 0x00020002U, 0x00020000U, 0x00020000U, 0x00020002U, 0x00020002U,+ 0x00020200U, 0x00020200U, 0x00020202U, 0x00020202U, 0x00020200U, 0x00020200U, 0x00020202U, 0x00020202U,+ 0x00020000U, 0x00020000U, 0x00020002U, 0x00020002U, 0x00020000U, 0x00020000U, 0x00020002U, 0x00020002U,+ 0x00020200U, 0x00020200U, 0x00020202U, 0x00020202U, 0x00020200U, 0x00020200U, 0x00020202U, 0x00020202U,+ 0x02000000U, 0x02000000U, 0x02000002U, 0x02000002U, 0x02000000U, 0x02000000U, 0x02000002U, 0x02000002U,+ 0x02000200U, 0x02000200U, 0x02000202U, 0x02000202U, 0x02000200U, 0x02000200U, 0x02000202U, 0x02000202U,+ 0x02000000U, 0x02000000U, 0x02000002U, 0x02000002U, 0x02000000U, 0x02000000U, 0x02000002U, 0x02000002U,+ 0x02000200U, 0x02000200U, 0x02000202U, 0x02000202U, 0x02000200U, 0x02000200U, 0x02000202U, 0x02000202U,+ 0x02020000U, 0x02020000U, 0x02020002U, 0x02020002U, 0x02020000U, 0x02020000U, 0x02020002U, 0x02020002U,+ 0x02020200U, 0x02020200U, 0x02020202U, 0x02020202U, 0x02020200U, 0x02020200U, 0x02020202U, 0x02020202U,+ 0x02020000U, 0x02020000U, 0x02020002U, 0x02020002U, 0x02020000U, 0x02020000U, 0x02020002U, 0x02020002U,+ 0x02020200U, 0x02020200U, 0x02020202U, 0x02020202U, 0x02020200U, 0x02020200U, 0x02020202U, 0x02020202U,+ 0x02000000U, 0x02000000U, 0x02000002U, 0x02000002U, 0x02000000U, 0x02000000U, 0x02000002U, 0x02000002U,+ 0x02000200U, 0x02000200U, 0x02000202U, 0x02000202U, 0x02000200U, 0x02000200U, 0x02000202U, 0x02000202U,+ 0x02000000U, 0x02000000U, 0x02000002U, 0x02000002U, 0x02000000U, 0x02000000U, 0x02000002U, 0x02000002U,+ 0x02000200U, 0x02000200U, 0x02000202U, 0x02000202U, 0x02000200U, 0x02000200U, 0x02000202U, 0x02000202U,+ 0x02020000U, 0x02020000U, 0x02020002U, 0x02020002U, 0x02020000U, 0x02020000U, 0x02020002U, 0x02020002U,+ 0x02020200U, 0x02020200U, 0x02020202U, 0x02020202U, 0x02020200U, 0x02020200U, 0x02020202U, 0x02020202U,+ 0x02020000U, 0x02020000U, 0x02020002U, 0x02020002U, 0x02020000U, 0x02020000U, 0x02020002U, 0x02020002U,+ 0x02020200U, 0x02020200U, 0x02020202U, 0x02020202U, 0x02020200U, 0x02020200U, 0x02020202U, 0x02020202U,+},+{+ 0x00000000U, 0x00000000U, 0x00000004U, 0x00000004U, 0x00000000U, 0x00000000U, 0x00000004U, 0x00000004U,+ 0x00000400U, 0x00000400U, 0x00000404U, 0x00000404U, 0x00000400U, 0x00000400U, 0x00000404U, 0x00000404U,+ 0x00000000U, 0x00000000U, 0x00000004U, 0x00000004U, 0x00000000U, 0x00000000U, 0x00000004U, 0x00000004U,+ 0x00000400U, 0x00000400U, 0x00000404U, 0x00000404U, 0x00000400U, 0x00000400U, 0x00000404U, 0x00000404U,+ 0x00040000U, 0x00040000U, 0x00040004U, 0x00040004U, 0x00040000U, 0x00040000U, 0x00040004U, 0x00040004U,+ 0x00040400U, 0x00040400U, 0x00040404U, 0x00040404U, 0x00040400U, 0x00040400U, 0x00040404U, 0x00040404U,+ 0x00040000U, 0x00040000U, 0x00040004U, 0x00040004U, 0x00040000U, 0x00040000U, 0x00040004U, 0x00040004U,+ 0x00040400U, 0x00040400U, 0x00040404U, 0x00040404U, 0x00040400U, 0x00040400U, 0x00040404U, 0x00040404U,+ 0x00000000U, 0x00000000U, 0x00000004U, 0x00000004U, 0x00000000U, 0x00000000U, 0x00000004U, 0x00000004U,+ 0x00000400U, 0x00000400U, 0x00000404U, 0x00000404U, 0x00000400U, 0x00000400U, 0x00000404U, 0x00000404U,+ 0x00000000U, 0x00000000U, 0x00000004U, 0x00000004U, 0x00000000U, 0x00000000U, 0x00000004U, 0x00000004U,+ 0x00000400U, 0x00000400U, 0x00000404U, 0x00000404U, 0x00000400U, 0x00000400U, 0x00000404U, 0x00000404U,+ 0x00040000U, 0x00040000U, 0x00040004U, 0x00040004U, 0x00040000U, 0x00040000U, 0x00040004U, 0x00040004U,+ 0x00040400U, 0x00040400U, 0x00040404U, 0x00040404U, 0x00040400U, 0x00040400U, 0x00040404U, 0x00040404U,+ 0x00040000U, 0x00040000U, 0x00040004U, 0x00040004U, 0x00040000U, 0x00040000U, 0x00040004U, 0x00040004U,+ 0x00040400U, 0x00040400U, 0x00040404U, 0x00040404U, 0x00040400U, 0x00040400U, 0x00040404U, 0x00040404U,+ 0x04000000U, 0x04000000U, 0x04000004U, 0x04000004U, 0x04000000U, 0x04000000U, 0x04000004U, 0x04000004U,+ 0x04000400U, 0x04000400U, 0x04000404U, 0x04000404U, 0x04000400U, 0x04000400U, 0x04000404U, 0x04000404U,+ 0x04000000U, 0x04000000U, 0x04000004U, 0x04000004U, 0x04000000U, 0x04000000U, 0x04000004U, 0x04000004U,+ 0x04000400U, 0x04000400U, 0x04000404U, 0x04000404U, 0x04000400U, 0x04000400U, 0x04000404U, 0x04000404U,+ 0x04040000U, 0x04040000U, 0x04040004U, 0x04040004U, 0x04040000U, 0x04040000U, 0x04040004U, 0x04040004U,+ 0x04040400U, 0x04040400U, 0x04040404U, 0x04040404U, 0x04040400U, 0x04040400U, 0x04040404U, 0x04040404U,+ 0x04040000U, 0x04040000U, 0x04040004U, 0x04040004U, 0x04040000U, 0x04040000U, 0x04040004U, 0x04040004U,+ 0x04040400U, 0x04040400U, 0x04040404U, 0x04040404U, 0x04040400U, 0x04040400U, 0x04040404U, 0x04040404U,+ 0x04000000U, 0x04000000U, 0x04000004U, 0x04000004U, 0x04000000U, 0x04000000U, 0x04000004U, 0x04000004U,+ 0x04000400U, 0x04000400U, 0x04000404U, 0x04000404U, 0x04000400U, 0x04000400U, 0x04000404U, 0x04000404U,+ 0x04000000U, 0x04000000U, 0x04000004U, 0x04000004U, 0x04000000U, 0x04000000U, 0x04000004U, 0x04000004U,+ 0x04000400U, 0x04000400U, 0x04000404U, 0x04000404U, 0x04000400U, 0x04000400U, 0x04000404U, 0x04000404U,+ 0x04040000U, 0x04040000U, 0x04040004U, 0x04040004U, 0x04040000U, 0x04040000U, 0x04040004U, 0x04040004U,+ 0x04040400U, 0x04040400U, 0x04040404U, 0x04040404U, 0x04040400U, 0x04040400U, 0x04040404U, 0x04040404U,+ 0x04040000U, 0x04040000U, 0x04040004U, 0x04040004U, 0x04040000U, 0x04040000U, 0x04040004U, 0x04040004U,+ 0x04040400U, 0x04040400U, 0x04040404U, 0x04040404U, 0x04040400U, 0x04040400U, 0x04040404U, 0x04040404U,+},+{+ 0x00000000U, 0x00000000U, 0x00000008U, 0x00000008U, 0x00000000U, 0x00000000U, 0x00000008U, 0x00000008U,+ 0x00000800U, 0x00000800U, 0x00000808U, 0x00000808U, 0x00000800U, 0x00000800U, 0x00000808U, 0x00000808U,+ 0x00000000U, 0x00000000U, 0x00000008U, 0x00000008U, 0x00000000U, 0x00000000U, 0x00000008U, 0x00000008U,+ 0x00000800U, 0x00000800U, 0x00000808U, 0x00000808U, 0x00000800U, 0x00000800U, 0x00000808U, 0x00000808U,+ 0x00080000U, 0x00080000U, 0x00080008U, 0x00080008U, 0x00080000U, 0x00080000U, 0x00080008U, 0x00080008U,+ 0x00080800U, 0x00080800U, 0x00080808U, 0x00080808U, 0x00080800U, 0x00080800U, 0x00080808U, 0x00080808U,+ 0x00080000U, 0x00080000U, 0x00080008U, 0x00080008U, 0x00080000U, 0x00080000U, 0x00080008U, 0x00080008U,+ 0x00080800U, 0x00080800U, 0x00080808U, 0x00080808U, 0x00080800U, 0x00080800U, 0x00080808U, 0x00080808U,+ 0x00000000U, 0x00000000U, 0x00000008U, 0x00000008U, 0x00000000U, 0x00000000U, 0x00000008U, 0x00000008U,+ 0x00000800U, 0x00000800U, 0x00000808U, 0x00000808U, 0x00000800U, 0x00000800U, 0x00000808U, 0x00000808U,+ 0x00000000U, 0x00000000U, 0x00000008U, 0x00000008U, 0x00000000U, 0x00000000U, 0x00000008U, 0x00000008U,+ 0x00000800U, 0x00000800U, 0x00000808U, 0x00000808U, 0x00000800U, 0x00000800U, 0x00000808U, 0x00000808U,+ 0x00080000U, 0x00080000U, 0x00080008U, 0x00080008U, 0x00080000U, 0x00080000U, 0x00080008U, 0x00080008U,+ 0x00080800U, 0x00080800U, 0x00080808U, 0x00080808U, 0x00080800U, 0x00080800U, 0x00080808U, 0x00080808U,+ 0x00080000U, 0x00080000U, 0x00080008U, 0x00080008U, 0x00080000U, 0x00080000U, 0x00080008U, 0x00080008U,+ 0x00080800U, 0x00080800U, 0x00080808U, 0x00080808U, 0x00080800U, 0x00080800U, 0x00080808U, 0x00080808U,+ 0x08000000U, 0x08000000U, 0x08000008U, 0x08000008U, 0x08000000U, 0x08000000U, 0x08000008U, 0x08000008U,+ 0x08000800U, 0x08000800U, 0x08000808U, 0x08000808U, 0x08000800U, 0x08000800U, 0x08000808U, 0x08000808U,+ 0x08000000U, 0x08000000U, 0x08000008U, 0x08000008U, 0x08000000U, 0x08000000U, 0x08000008U, 0x08000008U,+ 0x08000800U, 0x08000800U, 0x08000808U, 0x08000808U, 0x08000800U, 0x08000800U, 0x08000808U, 0x08000808U,+ 0x08080000U, 0x08080000U, 0x08080008U, 0x08080008U, 0x08080000U, 0x08080000U, 0x08080008U, 0x08080008U,+ 0x08080800U, 0x08080800U, 0x08080808U, 0x08080808U, 0x08080800U, 0x08080800U, 0x08080808U, 0x08080808U,+ 0x08080000U, 0x08080000U, 0x08080008U, 0x08080008U, 0x08080000U, 0x08080000U, 0x08080008U, 0x08080008U,+ 0x08080800U, 0x08080800U, 0x08080808U, 0x08080808U, 0x08080800U, 0x08080800U, 0x08080808U, 0x08080808U,+ 0x08000000U, 0x08000000U, 0x08000008U, 0x08000008U, 0x08000000U, 0x08000000U, 0x08000008U, 0x08000008U,+ 0x08000800U, 0x08000800U, 0x08000808U, 0x08000808U, 0x08000800U, 0x08000800U, 0x08000808U, 0x08000808U,+ 0x08000000U, 0x08000000U, 0x08000008U, 0x08000008U, 0x08000000U, 0x08000000U, 0x08000008U, 0x08000008U,+ 0x08000800U, 0x08000800U, 0x08000808U, 0x08000808U, 0x08000800U, 0x08000800U, 0x08000808U, 0x08000808U,+ 0x08080000U, 0x08080000U, 0x08080008U, 0x08080008U, 0x08080000U, 0x08080000U, 0x08080008U, 0x08080008U,+ 0x08080800U, 0x08080800U, 0x08080808U, 0x08080808U, 0x08080800U, 0x08080800U, 0x08080808U, 0x08080808U,+ 0x08080000U, 0x08080000U, 0x08080008U, 0x08080008U, 0x08080000U, 0x08080000U, 0x08080008U, 0x08080008U,+ 0x08080800U, 0x08080800U, 0x08080808U, 0x08080808U, 0x08080800U, 0x08080800U, 0x08080808U, 0x08080808U,+},+{+ 0x00000000U, 0x00000000U, 0x00000010U, 0x00000010U, 0x00000000U, 0x00000000U, 0x00000010U, 0x00000010U,+ 0x00001000U, 0x00001000U, 0x00001010U, 0x00001010U, 0x00001000U, 0x00001000U, 0x00001010U, 0x00001010U,+ 0x00000000U, 0x00000000U, 0x00000010U, 0x00000010U, 0x00000000U, 0x00000000U, 0x00000010U, 0x00000010U,+ 0x00001000U, 0x00001000U, 0x00001010U, 0x00001010U, 0x00001000U, 0x00001000U, 0x00001010U, 0x00001010U,+ 0x00100000U, 0x00100000U, 0x00100010U, 0x00100010U, 0x00100000U, 0x00100000U, 0x00100010U, 0x00100010U,+ 0x00101000U, 0x00101000U, 0x00101010U, 0x00101010U, 0x00101000U, 0x00101000U, 0x00101010U, 0x00101010U,+ 0x00100000U, 0x00100000U, 0x00100010U, 0x00100010U, 0x00100000U, 0x00100000U, 0x00100010U, 0x00100010U,+ 0x00101000U, 0x00101000U, 0x00101010U, 0x00101010U, 0x00101000U, 0x00101000U, 0x00101010U, 0x00101010U,+ 0x00000000U, 0x00000000U, 0x00000010U, 0x00000010U, 0x00000000U, 0x00000000U, 0x00000010U, 0x00000010U,+ 0x00001000U, 0x00001000U, 0x00001010U, 0x00001010U, 0x00001000U, 0x00001000U, 0x00001010U, 0x00001010U,+ 0x00000000U, 0x00000000U, 0x00000010U, 0x00000010U, 0x00000000U, 0x00000000U, 0x00000010U, 0x00000010U,+ 0x00001000U, 0x00001000U, 0x00001010U, 0x00001010U, 0x00001000U, 0x00001000U, 0x00001010U, 0x00001010U,+ 0x00100000U, 0x00100000U, 0x00100010U, 0x00100010U, 0x00100000U, 0x00100000U, 0x00100010U, 0x00100010U,+ 0x00101000U, 0x00101000U, 0x00101010U, 0x00101010U, 0x00101000U, 0x00101000U, 0x00101010U, 0x00101010U,+ 0x00100000U, 0x00100000U, 0x00100010U, 0x00100010U, 0x00100000U, 0x00100000U, 0x00100010U, 0x00100010U,+ 0x00101000U, 0x00101000U, 0x00101010U, 0x00101010U, 0x00101000U, 0x00101000U, 0x00101010U, 0x00101010U,+ 0x10000000U, 0x10000000U, 0x10000010U, 0x10000010U, 0x10000000U, 0x10000000U, 0x10000010U, 0x10000010U,+ 0x10001000U, 0x10001000U, 0x10001010U, 0x10001010U, 0x10001000U, 0x10001000U, 0x10001010U, 0x10001010U,+ 0x10000000U, 0x10000000U, 0x10000010U, 0x10000010U, 0x10000000U, 0x10000000U, 0x10000010U, 0x10000010U,+ 0x10001000U, 0x10001000U, 0x10001010U, 0x10001010U, 0x10001000U, 0x10001000U, 0x10001010U, 0x10001010U,+ 0x10100000U, 0x10100000U, 0x10100010U, 0x10100010U, 0x10100000U, 0x10100000U, 0x10100010U, 0x10100010U,+ 0x10101000U, 0x10101000U, 0x10101010U, 0x10101010U, 0x10101000U, 0x10101000U, 0x10101010U, 0x10101010U,+ 0x10100000U, 0x10100000U, 0x10100010U, 0x10100010U, 0x10100000U, 0x10100000U, 0x10100010U, 0x10100010U,+ 0x10101000U, 0x10101000U, 0x10101010U, 0x10101010U, 0x10101000U, 0x10101000U, 0x10101010U, 0x10101010U,+ 0x10000000U, 0x10000000U, 0x10000010U, 0x10000010U, 0x10000000U, 0x10000000U, 0x10000010U, 0x10000010U,+ 0x10001000U, 0x10001000U, 0x10001010U, 0x10001010U, 0x10001000U, 0x10001000U, 0x10001010U, 0x10001010U,+ 0x10000000U, 0x10000000U, 0x10000010U, 0x10000010U, 0x10000000U, 0x10000000U, 0x10000010U, 0x10000010U,+ 0x10001000U, 0x10001000U, 0x10001010U, 0x10001010U, 0x10001000U, 0x10001000U, 0x10001010U, 0x10001010U,+ 0x10100000U, 0x10100000U, 0x10100010U, 0x10100010U, 0x10100000U, 0x10100000U, 0x10100010U, 0x10100010U,+ 0x10101000U, 0x10101000U, 0x10101010U, 0x10101010U, 0x10101000U, 0x10101000U, 0x10101010U, 0x10101010U,+ 0x10100000U, 0x10100000U, 0x10100010U, 0x10100010U, 0x10100000U, 0x10100000U, 0x10100010U, 0x10100010U,+ 0x10101000U, 0x10101000U, 0x10101010U, 0x10101010U, 0x10101000U, 0x10101000U, 0x10101010U, 0x10101010U,+},+{+ 0x00000000U, 0x00000000U, 0x00000020U, 0x00000020U, 0x00000000U, 0x00000000U, 0x00000020U, 0x00000020U,+ 0x00002000U, 0x00002000U, 0x00002020U, 0x00002020U, 0x00002000U, 0x00002000U, 0x00002020U, 0x00002020U,+ 0x00000000U, 0x00000000U, 0x00000020U, 0x00000020U, 0x00000000U, 0x00000000U, 0x00000020U, 0x00000020U,+ 0x00002000U, 0x00002000U, 0x00002020U, 0x00002020U, 0x00002000U, 0x00002000U, 0x00002020U, 0x00002020U,+ 0x00200000U, 0x00200000U, 0x00200020U, 0x00200020U, 0x00200000U, 0x00200000U, 0x00200020U, 0x00200020U,+ 0x00202000U, 0x00202000U, 0x00202020U, 0x00202020U, 0x00202000U, 0x00202000U, 0x00202020U, 0x00202020U,+ 0x00200000U, 0x00200000U, 0x00200020U, 0x00200020U, 0x00200000U, 0x00200000U, 0x00200020U, 0x00200020U,+ 0x00202000U, 0x00202000U, 0x00202020U, 0x00202020U, 0x00202000U, 0x00202000U, 0x00202020U, 0x00202020U,+ 0x00000000U, 0x00000000U, 0x00000020U, 0x00000020U, 0x00000000U, 0x00000000U, 0x00000020U, 0x00000020U,+ 0x00002000U, 0x00002000U, 0x00002020U, 0x00002020U, 0x00002000U, 0x00002000U, 0x00002020U, 0x00002020U,+ 0x00000000U, 0x00000000U, 0x00000020U, 0x00000020U, 0x00000000U, 0x00000000U, 0x00000020U, 0x00000020U,+ 0x00002000U, 0x00002000U, 0x00002020U, 0x00002020U, 0x00002000U, 0x00002000U, 0x00002020U, 0x00002020U,+ 0x00200000U, 0x00200000U, 0x00200020U, 0x00200020U, 0x00200000U, 0x00200000U, 0x00200020U, 0x00200020U,+ 0x00202000U, 0x00202000U, 0x00202020U, 0x00202020U, 0x00202000U, 0x00202000U, 0x00202020U, 0x00202020U,+ 0x00200000U, 0x00200000U, 0x00200020U, 0x00200020U, 0x00200000U, 0x00200000U, 0x00200020U, 0x00200020U,+ 0x00202000U, 0x00202000U, 0x00202020U, 0x00202020U, 0x00202000U, 0x00202000U, 0x00202020U, 0x00202020U,+ 0x20000000U, 0x20000000U, 0x20000020U, 0x20000020U, 0x20000000U, 0x20000000U, 0x20000020U, 0x20000020U,+ 0x20002000U, 0x20002000U, 0x20002020U, 0x20002020U, 0x20002000U, 0x20002000U, 0x20002020U, 0x20002020U,+ 0x20000000U, 0x20000000U, 0x20000020U, 0x20000020U, 0x20000000U, 0x20000000U, 0x20000020U, 0x20000020U,+ 0x20002000U, 0x20002000U, 0x20002020U, 0x20002020U, 0x20002000U, 0x20002000U, 0x20002020U, 0x20002020U,+ 0x20200000U, 0x20200000U, 0x20200020U, 0x20200020U, 0x20200000U, 0x20200000U, 0x20200020U, 0x20200020U,+ 0x20202000U, 0x20202000U, 0x20202020U, 0x20202020U, 0x20202000U, 0x20202000U, 0x20202020U, 0x20202020U,+ 0x20200000U, 0x20200000U, 0x20200020U, 0x20200020U, 0x20200000U, 0x20200000U, 0x20200020U, 0x20200020U,+ 0x20202000U, 0x20202000U, 0x20202020U, 0x20202020U, 0x20202000U, 0x20202000U, 0x20202020U, 0x20202020U,+ 0x20000000U, 0x20000000U, 0x20000020U, 0x20000020U, 0x20000000U, 0x20000000U, 0x20000020U, 0x20000020U,+ 0x20002000U, 0x20002000U, 0x20002020U, 0x20002020U, 0x20002000U, 0x20002000U, 0x20002020U, 0x20002020U,+ 0x20000000U, 0x20000000U, 0x20000020U, 0x20000020U, 0x20000000U, 0x20000000U, 0x20000020U, 0x20000020U,+ 0x20002000U, 0x20002000U, 0x20002020U, 0x20002020U, 0x20002000U, 0x20002000U, 0x20002020U, 0x20002020U,+ 0x20200000U, 0x20200000U, 0x20200020U, 0x20200020U, 0x20200000U, 0x20200000U, 0x20200020U, 0x20200020U,+ 0x20202000U, 0x20202000U, 0x20202020U, 0x20202020U, 0x20202000U, 0x20202000U, 0x20202020U, 0x20202020U,+ 0x20200000U, 0x20200000U, 0x20200020U, 0x20200020U, 0x20200000U, 0x20200000U, 0x20200020U, 0x20200020U,+ 0x20202000U, 0x20202000U, 0x20202020U, 0x20202020U, 0x20202000U, 0x20202000U, 0x20202020U, 0x20202020U,+},+{+ 0x00000000U, 0x00000000U, 0x00000040U, 0x00000040U, 0x00000000U, 0x00000000U, 0x00000040U, 0x00000040U,+ 0x00004000U, 0x00004000U, 0x00004040U, 0x00004040U, 0x00004000U, 0x00004000U, 0x00004040U, 0x00004040U,+ 0x00000000U, 0x00000000U, 0x00000040U, 0x00000040U, 0x00000000U, 0x00000000U, 0x00000040U, 0x00000040U,+ 0x00004000U, 0x00004000U, 0x00004040U, 0x00004040U, 0x00004000U, 0x00004000U, 0x00004040U, 0x00004040U,+ 0x00400000U, 0x00400000U, 0x00400040U, 0x00400040U, 0x00400000U, 0x00400000U, 0x00400040U, 0x00400040U,+ 0x00404000U, 0x00404000U, 0x00404040U, 0x00404040U, 0x00404000U, 0x00404000U, 0x00404040U, 0x00404040U,+ 0x00400000U, 0x00400000U, 0x00400040U, 0x00400040U, 0x00400000U, 0x00400000U, 0x00400040U, 0x00400040U,+ 0x00404000U, 0x00404000U, 0x00404040U, 0x00404040U, 0x00404000U, 0x00404000U, 0x00404040U, 0x00404040U,+ 0x00000000U, 0x00000000U, 0x00000040U, 0x00000040U, 0x00000000U, 0x00000000U, 0x00000040U, 0x00000040U,+ 0x00004000U, 0x00004000U, 0x00004040U, 0x00004040U, 0x00004000U, 0x00004000U, 0x00004040U, 0x00004040U,+ 0x00000000U, 0x00000000U, 0x00000040U, 0x00000040U, 0x00000000U, 0x00000000U, 0x00000040U, 0x00000040U,+ 0x00004000U, 0x00004000U, 0x00004040U, 0x00004040U, 0x00004000U, 0x00004000U, 0x00004040U, 0x00004040U,+ 0x00400000U, 0x00400000U, 0x00400040U, 0x00400040U, 0x00400000U, 0x00400000U, 0x00400040U, 0x00400040U,+ 0x00404000U, 0x00404000U, 0x00404040U, 0x00404040U, 0x00404000U, 0x00404000U, 0x00404040U, 0x00404040U,+ 0x00400000U, 0x00400000U, 0x00400040U, 0x00400040U, 0x00400000U, 0x00400000U, 0x00400040U, 0x00400040U,+ 0x00404000U, 0x00404000U, 0x00404040U, 0x00404040U, 0x00404000U, 0x00404000U, 0x00404040U, 0x00404040U,+ 0x40000000U, 0x40000000U, 0x40000040U, 0x40000040U, 0x40000000U, 0x40000000U, 0x40000040U, 0x40000040U,+ 0x40004000U, 0x40004000U, 0x40004040U, 0x40004040U, 0x40004000U, 0x40004000U, 0x40004040U, 0x40004040U,+ 0x40000000U, 0x40000000U, 0x40000040U, 0x40000040U, 0x40000000U, 0x40000000U, 0x40000040U, 0x40000040U,+ 0x40004000U, 0x40004000U, 0x40004040U, 0x40004040U, 0x40004000U, 0x40004000U, 0x40004040U, 0x40004040U,+ 0x40400000U, 0x40400000U, 0x40400040U, 0x40400040U, 0x40400000U, 0x40400000U, 0x40400040U, 0x40400040U,+ 0x40404000U, 0x40404000U, 0x40404040U, 0x40404040U, 0x40404000U, 0x40404000U, 0x40404040U, 0x40404040U,+ 0x40400000U, 0x40400000U, 0x40400040U, 0x40400040U, 0x40400000U, 0x40400000U, 0x40400040U, 0x40400040U,+ 0x40404000U, 0x40404000U, 0x40404040U, 0x40404040U, 0x40404000U, 0x40404000U, 0x40404040U, 0x40404040U,+ 0x40000000U, 0x40000000U, 0x40000040U, 0x40000040U, 0x40000000U, 0x40000000U, 0x40000040U, 0x40000040U,+ 0x40004000U, 0x40004000U, 0x40004040U, 0x40004040U, 0x40004000U, 0x40004000U, 0x40004040U, 0x40004040U,+ 0x40000000U, 0x40000000U, 0x40000040U, 0x40000040U, 0x40000000U, 0x40000000U, 0x40000040U, 0x40000040U,+ 0x40004000U, 0x40004000U, 0x40004040U, 0x40004040U, 0x40004000U, 0x40004000U, 0x40004040U, 0x40004040U,+ 0x40400000U, 0x40400000U, 0x40400040U, 0x40400040U, 0x40400000U, 0x40400000U, 0x40400040U, 0x40400040U,+ 0x40404000U, 0x40404000U, 0x40404040U, 0x40404040U, 0x40404000U, 0x40404000U, 0x40404040U, 0x40404040U,+ 0x40400000U, 0x40400000U, 0x40400040U, 0x40400040U, 0x40400000U, 0x40400000U, 0x40400040U, 0x40400040U,+ 0x40404000U, 0x40404000U, 0x40404040U, 0x40404040U, 0x40404000U, 0x40404000U, 0x40404040U, 0x40404040U,+},+{+ 0x00000000U, 0x00000000U, 0x00000080U, 0x00000080U, 0x00000000U, 0x00000000U, 0x00000080U, 0x00000080U,+ 0x00008000U, 0x00008000U, 0x00008080U, 0x00008080U, 0x00008000U, 0x00008000U, 0x00008080U, 0x00008080U,+ 0x00000000U, 0x00000000U, 0x00000080U, 0x00000080U, 0x00000000U, 0x00000000U, 0x00000080U, 0x00000080U,+ 0x00008000U, 0x00008000U, 0x00008080U, 0x00008080U, 0x00008000U, 0x00008000U, 0x00008080U, 0x00008080U,+ 0x00800000U, 0x00800000U, 0x00800080U, 0x00800080U, 0x00800000U, 0x00800000U, 0x00800080U, 0x00800080U,+ 0x00808000U, 0x00808000U, 0x00808080U, 0x00808080U, 0x00808000U, 0x00808000U, 0x00808080U, 0x00808080U,+ 0x00800000U, 0x00800000U, 0x00800080U, 0x00800080U, 0x00800000U, 0x00800000U, 0x00800080U, 0x00800080U,+ 0x00808000U, 0x00808000U, 0x00808080U, 0x00808080U, 0x00808000U, 0x00808000U, 0x00808080U, 0x00808080U,+ 0x00000000U, 0x00000000U, 0x00000080U, 0x00000080U, 0x00000000U, 0x00000000U, 0x00000080U, 0x00000080U,+ 0x00008000U, 0x00008000U, 0x00008080U, 0x00008080U, 0x00008000U, 0x00008000U, 0x00008080U, 0x00008080U,+ 0x00000000U, 0x00000000U, 0x00000080U, 0x00000080U, 0x00000000U, 0x00000000U, 0x00000080U, 0x00000080U,+ 0x00008000U, 0x00008000U, 0x00008080U, 0x00008080U, 0x00008000U, 0x00008000U, 0x00008080U, 0x00008080U,+ 0x00800000U, 0x00800000U, 0x00800080U, 0x00800080U, 0x00800000U, 0x00800000U, 0x00800080U, 0x00800080U,+ 0x00808000U, 0x00808000U, 0x00808080U, 0x00808080U, 0x00808000U, 0x00808000U, 0x00808080U, 0x00808080U,+ 0x00800000U, 0x00800000U, 0x00800080U, 0x00800080U, 0x00800000U, 0x00800000U, 0x00800080U, 0x00800080U,+ 0x00808000U, 0x00808000U, 0x00808080U, 0x00808080U, 0x00808000U, 0x00808000U, 0x00808080U, 0x00808080U,+ 0x80000000U, 0x80000000U, 0x80000080U, 0x80000080U, 0x80000000U, 0x80000000U, 0x80000080U, 0x80000080U,+ 0x80008000U, 0x80008000U, 0x80008080U, 0x80008080U, 0x80008000U, 0x80008000U, 0x80008080U, 0x80008080U,+ 0x80000000U, 0x80000000U, 0x80000080U, 0x80000080U, 0x80000000U, 0x80000000U, 0x80000080U, 0x80000080U,+ 0x80008000U, 0x80008000U, 0x80008080U, 0x80008080U, 0x80008000U, 0x80008000U, 0x80008080U, 0x80008080U,+ 0x80800000U, 0x80800000U, 0x80800080U, 0x80800080U, 0x80800000U, 0x80800000U, 0x80800080U, 0x80800080U,+ 0x80808000U, 0x80808000U, 0x80808080U, 0x80808080U, 0x80808000U, 0x80808000U, 0x80808080U, 0x80808080U,+ 0x80800000U, 0x80800000U, 0x80800080U, 0x80800080U, 0x80800000U, 0x80800000U, 0x80800080U, 0x80800080U,+ 0x80808000U, 0x80808000U, 0x80808080U, 0x80808080U, 0x80808000U, 0x80808000U, 0x80808080U, 0x80808080U,+ 0x80000000U, 0x80000000U, 0x80000080U, 0x80000080U, 0x80000000U, 0x80000000U, 0x80000080U, 0x80000080U,+ 0x80008000U, 0x80008000U, 0x80008080U, 0x80008080U, 0x80008000U, 0x80008000U, 0x80008080U, 0x80008080U,+ 0x80000000U, 0x80000000U, 0x80000080U, 0x80000080U, 0x80000000U, 0x80000000U, 0x80000080U, 0x80000080U,+ 0x80008000U, 0x80008000U, 0x80008080U, 0x80008080U, 0x80008000U, 0x80008000U, 0x80008080U, 0x80008080U,+ 0x80800000U, 0x80800000U, 0x80800080U, 0x80800080U, 0x80800000U, 0x80800000U, 0x80800080U, 0x80800080U,+ 0x80808000U, 0x80808000U, 0x80808080U, 0x80808080U, 0x80808000U, 0x80808000U, 0x80808080U, 0x80808080U,+ 0x80800000U, 0x80800000U, 0x80800080U, 0x80800080U, 0x80800000U, 0x80800000U, 0x80800080U, 0x80800080U,+ 0x80808000U, 0x80808000U, 0x80808080U, 0x80808080U, 0x80808000U, 0x80808000U, 0x80808080U, 0x80808080U,+},+};++static const uint32_t FPH[8][256] = {+{+ 0x00000000U, 0x40000000U, 0x00400000U, 0x40400000U, 0x00004000U, 0x40004000U, 0x00404000U, 0x40404000U,+ 0x00000040U, 0x40000040U, 0x00400040U, 0x40400040U, 0x00004040U, 0x40004040U, 0x00404040U, 0x40404040U,+ 0x00000000U, 0x40000000U, 0x00400000U, 0x40400000U, 0x00004000U, 0x40004000U, 0x00404000U, 0x40404000U,+ 0x00000040U, 0x40000040U, 0x00400040U, 0x40400040U, 0x00004040U, 0x40004040U, 0x00404040U, 0x40404040U,+ 0x00000000U, 0x40000000U, 0x00400000U, 0x40400000U, 0x00004000U, 0x40004000U, 0x00404000U, 0x40404000U,+ 0x00000040U, 0x40000040U, 0x00400040U, 0x40400040U, 0x00004040U, 0x40004040U, 0x00404040U, 0x40404040U,+ 0x00000000U, 0x40000000U, 0x00400000U, 0x40400000U, 0x00004000U, 0x40004000U, 0x00404000U, 0x40404000U,+ 0x00000040U, 0x40000040U, 0x00400040U, 0x40400040U, 0x00004040U, 0x40004040U, 0x00404040U, 0x40404040U,+ 0x00000000U, 0x40000000U, 0x00400000U, 0x40400000U, 0x00004000U, 0x40004000U, 0x00404000U, 0x40404000U,+ 0x00000040U, 0x40000040U, 0x00400040U, 0x40400040U, 0x00004040U, 0x40004040U, 0x00404040U, 0x40404040U,+ 0x00000000U, 0x40000000U, 0x00400000U, 0x40400000U, 0x00004000U, 0x40004000U, 0x00404000U, 0x40404000U,+ 0x00000040U, 0x40000040U, 0x00400040U, 0x40400040U, 0x00004040U, 0x40004040U, 0x00404040U, 0x40404040U,+ 0x00000000U, 0x40000000U, 0x00400000U, 0x40400000U, 0x00004000U, 0x40004000U, 0x00404000U, 0x40404000U,+ 0x00000040U, 0x40000040U, 0x00400040U, 0x40400040U, 0x00004040U, 0x40004040U, 0x00404040U, 0x40404040U,+ 0x00000000U, 0x40000000U, 0x00400000U, 0x40400000U, 0x00004000U, 0x40004000U, 0x00404000U, 0x40404000U,+ 0x00000040U, 0x40000040U, 0x00400040U, 0x40400040U, 0x00004040U, 0x40004040U, 0x00404040U, 0x40404040U,+ 0x00000000U, 0x40000000U, 0x00400000U, 0x40400000U, 0x00004000U, 0x40004000U, 0x00404000U, 0x40404000U,+ 0x00000040U, 0x40000040U, 0x00400040U, 0x40400040U, 0x00004040U, 0x40004040U, 0x00404040U, 0x40404040U,+ 0x00000000U, 0x40000000U, 0x00400000U, 0x40400000U, 0x00004000U, 0x40004000U, 0x00404000U, 0x40404000U,+ 0x00000040U, 0x40000040U, 0x00400040U, 0x40400040U, 0x00004040U, 0x40004040U, 0x00404040U, 0x40404040U,+ 0x00000000U, 0x40000000U, 0x00400000U, 0x40400000U, 0x00004000U, 0x40004000U, 0x00404000U, 0x40404000U,+ 0x00000040U, 0x40000040U, 0x00400040U, 0x40400040U, 0x00004040U, 0x40004040U, 0x00404040U, 0x40404040U,+ 0x00000000U, 0x40000000U, 0x00400000U, 0x40400000U, 0x00004000U, 0x40004000U, 0x00404000U, 0x40404000U,+ 0x00000040U, 0x40000040U, 0x00400040U, 0x40400040U, 0x00004040U, 0x40004040U, 0x00404040U, 0x40404040U,+ 0x00000000U, 0x40000000U, 0x00400000U, 0x40400000U, 0x00004000U, 0x40004000U, 0x00404000U, 0x40404000U,+ 0x00000040U, 0x40000040U, 0x00400040U, 0x40400040U, 0x00004040U, 0x40004040U, 0x00404040U, 0x40404040U,+ 0x00000000U, 0x40000000U, 0x00400000U, 0x40400000U, 0x00004000U, 0x40004000U, 0x00404000U, 0x40404000U,+ 0x00000040U, 0x40000040U, 0x00400040U, 0x40400040U, 0x00004040U, 0x40004040U, 0x00404040U, 0x40404040U,+ 0x00000000U, 0x40000000U, 0x00400000U, 0x40400000U, 0x00004000U, 0x40004000U, 0x00404000U, 0x40404000U,+ 0x00000040U, 0x40000040U, 0x00400040U, 0x40400040U, 0x00004040U, 0x40004040U, 0x00404040U, 0x40404040U,+ 0x00000000U, 0x40000000U, 0x00400000U, 0x40400000U, 0x00004000U, 0x40004000U, 0x00404000U, 0x40404000U,+ 0x00000040U, 0x40000040U, 0x00400040U, 0x40400040U, 0x00004040U, 0x40004040U, 0x00404040U, 0x40404040U,+},+{+ 0x00000000U, 0x10000000U, 0x00100000U, 0x10100000U, 0x00001000U, 0x10001000U, 0x00101000U, 0x10101000U,+ 0x00000010U, 0x10000010U, 0x00100010U, 0x10100010U, 0x00001010U, 0x10001010U, 0x00101010U, 0x10101010U,+ 0x00000000U, 0x10000000U, 0x00100000U, 0x10100000U, 0x00001000U, 0x10001000U, 0x00101000U, 0x10101000U,+ 0x00000010U, 0x10000010U, 0x00100010U, 0x10100010U, 0x00001010U, 0x10001010U, 0x00101010U, 0x10101010U,+ 0x00000000U, 0x10000000U, 0x00100000U, 0x10100000U, 0x00001000U, 0x10001000U, 0x00101000U, 0x10101000U,+ 0x00000010U, 0x10000010U, 0x00100010U, 0x10100010U, 0x00001010U, 0x10001010U, 0x00101010U, 0x10101010U,+ 0x00000000U, 0x10000000U, 0x00100000U, 0x10100000U, 0x00001000U, 0x10001000U, 0x00101000U, 0x10101000U,+ 0x00000010U, 0x10000010U, 0x00100010U, 0x10100010U, 0x00001010U, 0x10001010U, 0x00101010U, 0x10101010U,+ 0x00000000U, 0x10000000U, 0x00100000U, 0x10100000U, 0x00001000U, 0x10001000U, 0x00101000U, 0x10101000U,+ 0x00000010U, 0x10000010U, 0x00100010U, 0x10100010U, 0x00001010U, 0x10001010U, 0x00101010U, 0x10101010U,+ 0x00000000U, 0x10000000U, 0x00100000U, 0x10100000U, 0x00001000U, 0x10001000U, 0x00101000U, 0x10101000U,+ 0x00000010U, 0x10000010U, 0x00100010U, 0x10100010U, 0x00001010U, 0x10001010U, 0x00101010U, 0x10101010U,+ 0x00000000U, 0x10000000U, 0x00100000U, 0x10100000U, 0x00001000U, 0x10001000U, 0x00101000U, 0x10101000U,+ 0x00000010U, 0x10000010U, 0x00100010U, 0x10100010U, 0x00001010U, 0x10001010U, 0x00101010U, 0x10101010U,+ 0x00000000U, 0x10000000U, 0x00100000U, 0x10100000U, 0x00001000U, 0x10001000U, 0x00101000U, 0x10101000U,+ 0x00000010U, 0x10000010U, 0x00100010U, 0x10100010U, 0x00001010U, 0x10001010U, 0x00101010U, 0x10101010U,+ 0x00000000U, 0x10000000U, 0x00100000U, 0x10100000U, 0x00001000U, 0x10001000U, 0x00101000U, 0x10101000U,+ 0x00000010U, 0x10000010U, 0x00100010U, 0x10100010U, 0x00001010U, 0x10001010U, 0x00101010U, 0x10101010U,+ 0x00000000U, 0x10000000U, 0x00100000U, 0x10100000U, 0x00001000U, 0x10001000U, 0x00101000U, 0x10101000U,+ 0x00000010U, 0x10000010U, 0x00100010U, 0x10100010U, 0x00001010U, 0x10001010U, 0x00101010U, 0x10101010U,+ 0x00000000U, 0x10000000U, 0x00100000U, 0x10100000U, 0x00001000U, 0x10001000U, 0x00101000U, 0x10101000U,+ 0x00000010U, 0x10000010U, 0x00100010U, 0x10100010U, 0x00001010U, 0x10001010U, 0x00101010U, 0x10101010U,+ 0x00000000U, 0x10000000U, 0x00100000U, 0x10100000U, 0x00001000U, 0x10001000U, 0x00101000U, 0x10101000U,+ 0x00000010U, 0x10000010U, 0x00100010U, 0x10100010U, 0x00001010U, 0x10001010U, 0x00101010U, 0x10101010U,+ 0x00000000U, 0x10000000U, 0x00100000U, 0x10100000U, 0x00001000U, 0x10001000U, 0x00101000U, 0x10101000U,+ 0x00000010U, 0x10000010U, 0x00100010U, 0x10100010U, 0x00001010U, 0x10001010U, 0x00101010U, 0x10101010U,+ 0x00000000U, 0x10000000U, 0x00100000U, 0x10100000U, 0x00001000U, 0x10001000U, 0x00101000U, 0x10101000U,+ 0x00000010U, 0x10000010U, 0x00100010U, 0x10100010U, 0x00001010U, 0x10001010U, 0x00101010U, 0x10101010U,+ 0x00000000U, 0x10000000U, 0x00100000U, 0x10100000U, 0x00001000U, 0x10001000U, 0x00101000U, 0x10101000U,+ 0x00000010U, 0x10000010U, 0x00100010U, 0x10100010U, 0x00001010U, 0x10001010U, 0x00101010U, 0x10101010U,+ 0x00000000U, 0x10000000U, 0x00100000U, 0x10100000U, 0x00001000U, 0x10001000U, 0x00101000U, 0x10101000U,+ 0x00000010U, 0x10000010U, 0x00100010U, 0x10100010U, 0x00001010U, 0x10001010U, 0x00101010U, 0x10101010U,+},+{+ 0x00000000U, 0x04000000U, 0x00040000U, 0x04040000U, 0x00000400U, 0x04000400U, 0x00040400U, 0x04040400U,+ 0x00000004U, 0x04000004U, 0x00040004U, 0x04040004U, 0x00000404U, 0x04000404U, 0x00040404U, 0x04040404U,+ 0x00000000U, 0x04000000U, 0x00040000U, 0x04040000U, 0x00000400U, 0x04000400U, 0x00040400U, 0x04040400U,+ 0x00000004U, 0x04000004U, 0x00040004U, 0x04040004U, 0x00000404U, 0x04000404U, 0x00040404U, 0x04040404U,+ 0x00000000U, 0x04000000U, 0x00040000U, 0x04040000U, 0x00000400U, 0x04000400U, 0x00040400U, 0x04040400U,+ 0x00000004U, 0x04000004U, 0x00040004U, 0x04040004U, 0x00000404U, 0x04000404U, 0x00040404U, 0x04040404U,+ 0x00000000U, 0x04000000U, 0x00040000U, 0x04040000U, 0x00000400U, 0x04000400U, 0x00040400U, 0x04040400U,+ 0x00000004U, 0x04000004U, 0x00040004U, 0x04040004U, 0x00000404U, 0x04000404U, 0x00040404U, 0x04040404U,+ 0x00000000U, 0x04000000U, 0x00040000U, 0x04040000U, 0x00000400U, 0x04000400U, 0x00040400U, 0x04040400U,+ 0x00000004U, 0x04000004U, 0x00040004U, 0x04040004U, 0x00000404U, 0x04000404U, 0x00040404U, 0x04040404U,+ 0x00000000U, 0x04000000U, 0x00040000U, 0x04040000U, 0x00000400U, 0x04000400U, 0x00040400U, 0x04040400U,+ 0x00000004U, 0x04000004U, 0x00040004U, 0x04040004U, 0x00000404U, 0x04000404U, 0x00040404U, 0x04040404U,+ 0x00000000U, 0x04000000U, 0x00040000U, 0x04040000U, 0x00000400U, 0x04000400U, 0x00040400U, 0x04040400U,+ 0x00000004U, 0x04000004U, 0x00040004U, 0x04040004U, 0x00000404U, 0x04000404U, 0x00040404U, 0x04040404U,+ 0x00000000U, 0x04000000U, 0x00040000U, 0x04040000U, 0x00000400U, 0x04000400U, 0x00040400U, 0x04040400U,+ 0x00000004U, 0x04000004U, 0x00040004U, 0x04040004U, 0x00000404U, 0x04000404U, 0x00040404U, 0x04040404U,+ 0x00000000U, 0x04000000U, 0x00040000U, 0x04040000U, 0x00000400U, 0x04000400U, 0x00040400U, 0x04040400U,+ 0x00000004U, 0x04000004U, 0x00040004U, 0x04040004U, 0x00000404U, 0x04000404U, 0x00040404U, 0x04040404U,+ 0x00000000U, 0x04000000U, 0x00040000U, 0x04040000U, 0x00000400U, 0x04000400U, 0x00040400U, 0x04040400U,+ 0x00000004U, 0x04000004U, 0x00040004U, 0x04040004U, 0x00000404U, 0x04000404U, 0x00040404U, 0x04040404U,+ 0x00000000U, 0x04000000U, 0x00040000U, 0x04040000U, 0x00000400U, 0x04000400U, 0x00040400U, 0x04040400U,+ 0x00000004U, 0x04000004U, 0x00040004U, 0x04040004U, 0x00000404U, 0x04000404U, 0x00040404U, 0x04040404U,+ 0x00000000U, 0x04000000U, 0x00040000U, 0x04040000U, 0x00000400U, 0x04000400U, 0x00040400U, 0x04040400U,+ 0x00000004U, 0x04000004U, 0x00040004U, 0x04040004U, 0x00000404U, 0x04000404U, 0x00040404U, 0x04040404U,+ 0x00000000U, 0x04000000U, 0x00040000U, 0x04040000U, 0x00000400U, 0x04000400U, 0x00040400U, 0x04040400U,+ 0x00000004U, 0x04000004U, 0x00040004U, 0x04040004U, 0x00000404U, 0x04000404U, 0x00040404U, 0x04040404U,+ 0x00000000U, 0x04000000U, 0x00040000U, 0x04040000U, 0x00000400U, 0x04000400U, 0x00040400U, 0x04040400U,+ 0x00000004U, 0x04000004U, 0x00040004U, 0x04040004U, 0x00000404U, 0x04000404U, 0x00040404U, 0x04040404U,+ 0x00000000U, 0x04000000U, 0x00040000U, 0x04040000U, 0x00000400U, 0x04000400U, 0x00040400U, 0x04040400U,+ 0x00000004U, 0x04000004U, 0x00040004U, 0x04040004U, 0x00000404U, 0x04000404U, 0x00040404U, 0x04040404U,+ 0x00000000U, 0x04000000U, 0x00040000U, 0x04040000U, 0x00000400U, 0x04000400U, 0x00040400U, 0x04040400U,+ 0x00000004U, 0x04000004U, 0x00040004U, 0x04040004U, 0x00000404U, 0x04000404U, 0x00040404U, 0x04040404U,+},+{+ 0x00000000U, 0x01000000U, 0x00010000U, 0x01010000U, 0x00000100U, 0x01000100U, 0x00010100U, 0x01010100U,+ 0x00000001U, 0x01000001U, 0x00010001U, 0x01010001U, 0x00000101U, 0x01000101U, 0x00010101U, 0x01010101U,+ 0x00000000U, 0x01000000U, 0x00010000U, 0x01010000U, 0x00000100U, 0x01000100U, 0x00010100U, 0x01010100U,+ 0x00000001U, 0x01000001U, 0x00010001U, 0x01010001U, 0x00000101U, 0x01000101U, 0x00010101U, 0x01010101U,+ 0x00000000U, 0x01000000U, 0x00010000U, 0x01010000U, 0x00000100U, 0x01000100U, 0x00010100U, 0x01010100U,+ 0x00000001U, 0x01000001U, 0x00010001U, 0x01010001U, 0x00000101U, 0x01000101U, 0x00010101U, 0x01010101U,+ 0x00000000U, 0x01000000U, 0x00010000U, 0x01010000U, 0x00000100U, 0x01000100U, 0x00010100U, 0x01010100U,+ 0x00000001U, 0x01000001U, 0x00010001U, 0x01010001U, 0x00000101U, 0x01000101U, 0x00010101U, 0x01010101U,+ 0x00000000U, 0x01000000U, 0x00010000U, 0x01010000U, 0x00000100U, 0x01000100U, 0x00010100U, 0x01010100U,+ 0x00000001U, 0x01000001U, 0x00010001U, 0x01010001U, 0x00000101U, 0x01000101U, 0x00010101U, 0x01010101U,+ 0x00000000U, 0x01000000U, 0x00010000U, 0x01010000U, 0x00000100U, 0x01000100U, 0x00010100U, 0x01010100U,+ 0x00000001U, 0x01000001U, 0x00010001U, 0x01010001U, 0x00000101U, 0x01000101U, 0x00010101U, 0x01010101U,+ 0x00000000U, 0x01000000U, 0x00010000U, 0x01010000U, 0x00000100U, 0x01000100U, 0x00010100U, 0x01010100U,+ 0x00000001U, 0x01000001U, 0x00010001U, 0x01010001U, 0x00000101U, 0x01000101U, 0x00010101U, 0x01010101U,+ 0x00000000U, 0x01000000U, 0x00010000U, 0x01010000U, 0x00000100U, 0x01000100U, 0x00010100U, 0x01010100U,+ 0x00000001U, 0x01000001U, 0x00010001U, 0x01010001U, 0x00000101U, 0x01000101U, 0x00010101U, 0x01010101U,+ 0x00000000U, 0x01000000U, 0x00010000U, 0x01010000U, 0x00000100U, 0x01000100U, 0x00010100U, 0x01010100U,+ 0x00000001U, 0x01000001U, 0x00010001U, 0x01010001U, 0x00000101U, 0x01000101U, 0x00010101U, 0x01010101U,+ 0x00000000U, 0x01000000U, 0x00010000U, 0x01010000U, 0x00000100U, 0x01000100U, 0x00010100U, 0x01010100U,+ 0x00000001U, 0x01000001U, 0x00010001U, 0x01010001U, 0x00000101U, 0x01000101U, 0x00010101U, 0x01010101U,+ 0x00000000U, 0x01000000U, 0x00010000U, 0x01010000U, 0x00000100U, 0x01000100U, 0x00010100U, 0x01010100U,+ 0x00000001U, 0x01000001U, 0x00010001U, 0x01010001U, 0x00000101U, 0x01000101U, 0x00010101U, 0x01010101U,+ 0x00000000U, 0x01000000U, 0x00010000U, 0x01010000U, 0x00000100U, 0x01000100U, 0x00010100U, 0x01010100U,+ 0x00000001U, 0x01000001U, 0x00010001U, 0x01010001U, 0x00000101U, 0x01000101U, 0x00010101U, 0x01010101U,+ 0x00000000U, 0x01000000U, 0x00010000U, 0x01010000U, 0x00000100U, 0x01000100U, 0x00010100U, 0x01010100U,+ 0x00000001U, 0x01000001U, 0x00010001U, 0x01010001U, 0x00000101U, 0x01000101U, 0x00010101U, 0x01010101U,+ 0x00000000U, 0x01000000U, 0x00010000U, 0x01010000U, 0x00000100U, 0x01000100U, 0x00010100U, 0x01010100U,+ 0x00000001U, 0x01000001U, 0x00010001U, 0x01010001U, 0x00000101U, 0x01000101U, 0x00010101U, 0x01010101U,+ 0x00000000U, 0x01000000U, 0x00010000U, 0x01010000U, 0x00000100U, 0x01000100U, 0x00010100U, 0x01010100U,+ 0x00000001U, 0x01000001U, 0x00010001U, 0x01010001U, 0x00000101U, 0x01000101U, 0x00010101U, 0x01010101U,+ 0x00000000U, 0x01000000U, 0x00010000U, 0x01010000U, 0x00000100U, 0x01000100U, 0x00010100U, 0x01010100U,+ 0x00000001U, 0x01000001U, 0x00010001U, 0x01010001U, 0x00000101U, 0x01000101U, 0x00010101U, 0x01010101U,+},+{+ 0x00000000U, 0x80000000U, 0x00800000U, 0x80800000U, 0x00008000U, 0x80008000U, 0x00808000U, 0x80808000U,+ 0x00000080U, 0x80000080U, 0x00800080U, 0x80800080U, 0x00008080U, 0x80008080U, 0x00808080U, 0x80808080U,+ 0x00000000U, 0x80000000U, 0x00800000U, 0x80800000U, 0x00008000U, 0x80008000U, 0x00808000U, 0x80808000U,+ 0x00000080U, 0x80000080U, 0x00800080U, 0x80800080U, 0x00008080U, 0x80008080U, 0x00808080U, 0x80808080U,+ 0x00000000U, 0x80000000U, 0x00800000U, 0x80800000U, 0x00008000U, 0x80008000U, 0x00808000U, 0x80808000U,+ 0x00000080U, 0x80000080U, 0x00800080U, 0x80800080U, 0x00008080U, 0x80008080U, 0x00808080U, 0x80808080U,+ 0x00000000U, 0x80000000U, 0x00800000U, 0x80800000U, 0x00008000U, 0x80008000U, 0x00808000U, 0x80808000U,+ 0x00000080U, 0x80000080U, 0x00800080U, 0x80800080U, 0x00008080U, 0x80008080U, 0x00808080U, 0x80808080U,+ 0x00000000U, 0x80000000U, 0x00800000U, 0x80800000U, 0x00008000U, 0x80008000U, 0x00808000U, 0x80808000U,+ 0x00000080U, 0x80000080U, 0x00800080U, 0x80800080U, 0x00008080U, 0x80008080U, 0x00808080U, 0x80808080U,+ 0x00000000U, 0x80000000U, 0x00800000U, 0x80800000U, 0x00008000U, 0x80008000U, 0x00808000U, 0x80808000U,+ 0x00000080U, 0x80000080U, 0x00800080U, 0x80800080U, 0x00008080U, 0x80008080U, 0x00808080U, 0x80808080U,+ 0x00000000U, 0x80000000U, 0x00800000U, 0x80800000U, 0x00008000U, 0x80008000U, 0x00808000U, 0x80808000U,+ 0x00000080U, 0x80000080U, 0x00800080U, 0x80800080U, 0x00008080U, 0x80008080U, 0x00808080U, 0x80808080U,+ 0x00000000U, 0x80000000U, 0x00800000U, 0x80800000U, 0x00008000U, 0x80008000U, 0x00808000U, 0x80808000U,+ 0x00000080U, 0x80000080U, 0x00800080U, 0x80800080U, 0x00008080U, 0x80008080U, 0x00808080U, 0x80808080U,+ 0x00000000U, 0x80000000U, 0x00800000U, 0x80800000U, 0x00008000U, 0x80008000U, 0x00808000U, 0x80808000U,+ 0x00000080U, 0x80000080U, 0x00800080U, 0x80800080U, 0x00008080U, 0x80008080U, 0x00808080U, 0x80808080U,+ 0x00000000U, 0x80000000U, 0x00800000U, 0x80800000U, 0x00008000U, 0x80008000U, 0x00808000U, 0x80808000U,+ 0x00000080U, 0x80000080U, 0x00800080U, 0x80800080U, 0x00008080U, 0x80008080U, 0x00808080U, 0x80808080U,+ 0x00000000U, 0x80000000U, 0x00800000U, 0x80800000U, 0x00008000U, 0x80008000U, 0x00808000U, 0x80808000U,+ 0x00000080U, 0x80000080U, 0x00800080U, 0x80800080U, 0x00008080U, 0x80008080U, 0x00808080U, 0x80808080U,+ 0x00000000U, 0x80000000U, 0x00800000U, 0x80800000U, 0x00008000U, 0x80008000U, 0x00808000U, 0x80808000U,+ 0x00000080U, 0x80000080U, 0x00800080U, 0x80800080U, 0x00008080U, 0x80008080U, 0x00808080U, 0x80808080U,+ 0x00000000U, 0x80000000U, 0x00800000U, 0x80800000U, 0x00008000U, 0x80008000U, 0x00808000U, 0x80808000U,+ 0x00000080U, 0x80000080U, 0x00800080U, 0x80800080U, 0x00008080U, 0x80008080U, 0x00808080U, 0x80808080U,+ 0x00000000U, 0x80000000U, 0x00800000U, 0x80800000U, 0x00008000U, 0x80008000U, 0x00808000U, 0x80808000U,+ 0x00000080U, 0x80000080U, 0x00800080U, 0x80800080U, 0x00008080U, 0x80008080U, 0x00808080U, 0x80808080U,+ 0x00000000U, 0x80000000U, 0x00800000U, 0x80800000U, 0x00008000U, 0x80008000U, 0x00808000U, 0x80808000U,+ 0x00000080U, 0x80000080U, 0x00800080U, 0x80800080U, 0x00008080U, 0x80008080U, 0x00808080U, 0x80808080U,+ 0x00000000U, 0x80000000U, 0x00800000U, 0x80800000U, 0x00008000U, 0x80008000U, 0x00808000U, 0x80808000U,+ 0x00000080U, 0x80000080U, 0x00800080U, 0x80800080U, 0x00008080U, 0x80008080U, 0x00808080U, 0x80808080U,+},+{+ 0x00000000U, 0x20000000U, 0x00200000U, 0x20200000U, 0x00002000U, 0x20002000U, 0x00202000U, 0x20202000U,+ 0x00000020U, 0x20000020U, 0x00200020U, 0x20200020U, 0x00002020U, 0x20002020U, 0x00202020U, 0x20202020U,+ 0x00000000U, 0x20000000U, 0x00200000U, 0x20200000U, 0x00002000U, 0x20002000U, 0x00202000U, 0x20202000U,+ 0x00000020U, 0x20000020U, 0x00200020U, 0x20200020U, 0x00002020U, 0x20002020U, 0x00202020U, 0x20202020U,+ 0x00000000U, 0x20000000U, 0x00200000U, 0x20200000U, 0x00002000U, 0x20002000U, 0x00202000U, 0x20202000U,+ 0x00000020U, 0x20000020U, 0x00200020U, 0x20200020U, 0x00002020U, 0x20002020U, 0x00202020U, 0x20202020U,+ 0x00000000U, 0x20000000U, 0x00200000U, 0x20200000U, 0x00002000U, 0x20002000U, 0x00202000U, 0x20202000U,+ 0x00000020U, 0x20000020U, 0x00200020U, 0x20200020U, 0x00002020U, 0x20002020U, 0x00202020U, 0x20202020U,+ 0x00000000U, 0x20000000U, 0x00200000U, 0x20200000U, 0x00002000U, 0x20002000U, 0x00202000U, 0x20202000U,+ 0x00000020U, 0x20000020U, 0x00200020U, 0x20200020U, 0x00002020U, 0x20002020U, 0x00202020U, 0x20202020U,+ 0x00000000U, 0x20000000U, 0x00200000U, 0x20200000U, 0x00002000U, 0x20002000U, 0x00202000U, 0x20202000U,+ 0x00000020U, 0x20000020U, 0x00200020U, 0x20200020U, 0x00002020U, 0x20002020U, 0x00202020U, 0x20202020U,+ 0x00000000U, 0x20000000U, 0x00200000U, 0x20200000U, 0x00002000U, 0x20002000U, 0x00202000U, 0x20202000U,+ 0x00000020U, 0x20000020U, 0x00200020U, 0x20200020U, 0x00002020U, 0x20002020U, 0x00202020U, 0x20202020U,+ 0x00000000U, 0x20000000U, 0x00200000U, 0x20200000U, 0x00002000U, 0x20002000U, 0x00202000U, 0x20202000U,+ 0x00000020U, 0x20000020U, 0x00200020U, 0x20200020U, 0x00002020U, 0x20002020U, 0x00202020U, 0x20202020U,+ 0x00000000U, 0x20000000U, 0x00200000U, 0x20200000U, 0x00002000U, 0x20002000U, 0x00202000U, 0x20202000U,+ 0x00000020U, 0x20000020U, 0x00200020U, 0x20200020U, 0x00002020U, 0x20002020U, 0x00202020U, 0x20202020U,+ 0x00000000U, 0x20000000U, 0x00200000U, 0x20200000U, 0x00002000U, 0x20002000U, 0x00202000U, 0x20202000U,+ 0x00000020U, 0x20000020U, 0x00200020U, 0x20200020U, 0x00002020U, 0x20002020U, 0x00202020U, 0x20202020U,+ 0x00000000U, 0x20000000U, 0x00200000U, 0x20200000U, 0x00002000U, 0x20002000U, 0x00202000U, 0x20202000U,+ 0x00000020U, 0x20000020U, 0x00200020U, 0x20200020U, 0x00002020U, 0x20002020U, 0x00202020U, 0x20202020U,+ 0x00000000U, 0x20000000U, 0x00200000U, 0x20200000U, 0x00002000U, 0x20002000U, 0x00202000U, 0x20202000U,+ 0x00000020U, 0x20000020U, 0x00200020U, 0x20200020U, 0x00002020U, 0x20002020U, 0x00202020U, 0x20202020U,+ 0x00000000U, 0x20000000U, 0x00200000U, 0x20200000U, 0x00002000U, 0x20002000U, 0x00202000U, 0x20202000U,+ 0x00000020U, 0x20000020U, 0x00200020U, 0x20200020U, 0x00002020U, 0x20002020U, 0x00202020U, 0x20202020U,+ 0x00000000U, 0x20000000U, 0x00200000U, 0x20200000U, 0x00002000U, 0x20002000U, 0x00202000U, 0x20202000U,+ 0x00000020U, 0x20000020U, 0x00200020U, 0x20200020U, 0x00002020U, 0x20002020U, 0x00202020U, 0x20202020U,+ 0x00000000U, 0x20000000U, 0x00200000U, 0x20200000U, 0x00002000U, 0x20002000U, 0x00202000U, 0x20202000U,+ 0x00000020U, 0x20000020U, 0x00200020U, 0x20200020U, 0x00002020U, 0x20002020U, 0x00202020U, 0x20202020U,+ 0x00000000U, 0x20000000U, 0x00200000U, 0x20200000U, 0x00002000U, 0x20002000U, 0x00202000U, 0x20202000U,+ 0x00000020U, 0x20000020U, 0x00200020U, 0x20200020U, 0x00002020U, 0x20002020U, 0x00202020U, 0x20202020U,+},+{+ 0x00000000U, 0x08000000U, 0x00080000U, 0x08080000U, 0x00000800U, 0x08000800U, 0x00080800U, 0x08080800U,+ 0x00000008U, 0x08000008U, 0x00080008U, 0x08080008U, 0x00000808U, 0x08000808U, 0x00080808U, 0x08080808U,+ 0x00000000U, 0x08000000U, 0x00080000U, 0x08080000U, 0x00000800U, 0x08000800U, 0x00080800U, 0x08080800U,+ 0x00000008U, 0x08000008U, 0x00080008U, 0x08080008U, 0x00000808U, 0x08000808U, 0x00080808U, 0x08080808U,+ 0x00000000U, 0x08000000U, 0x00080000U, 0x08080000U, 0x00000800U, 0x08000800U, 0x00080800U, 0x08080800U,+ 0x00000008U, 0x08000008U, 0x00080008U, 0x08080008U, 0x00000808U, 0x08000808U, 0x00080808U, 0x08080808U,+ 0x00000000U, 0x08000000U, 0x00080000U, 0x08080000U, 0x00000800U, 0x08000800U, 0x00080800U, 0x08080800U,+ 0x00000008U, 0x08000008U, 0x00080008U, 0x08080008U, 0x00000808U, 0x08000808U, 0x00080808U, 0x08080808U,+ 0x00000000U, 0x08000000U, 0x00080000U, 0x08080000U, 0x00000800U, 0x08000800U, 0x00080800U, 0x08080800U,+ 0x00000008U, 0x08000008U, 0x00080008U, 0x08080008U, 0x00000808U, 0x08000808U, 0x00080808U, 0x08080808U,+ 0x00000000U, 0x08000000U, 0x00080000U, 0x08080000U, 0x00000800U, 0x08000800U, 0x00080800U, 0x08080800U,+ 0x00000008U, 0x08000008U, 0x00080008U, 0x08080008U, 0x00000808U, 0x08000808U, 0x00080808U, 0x08080808U,+ 0x00000000U, 0x08000000U, 0x00080000U, 0x08080000U, 0x00000800U, 0x08000800U, 0x00080800U, 0x08080800U,+ 0x00000008U, 0x08000008U, 0x00080008U, 0x08080008U, 0x00000808U, 0x08000808U, 0x00080808U, 0x08080808U,+ 0x00000000U, 0x08000000U, 0x00080000U, 0x08080000U, 0x00000800U, 0x08000800U, 0x00080800U, 0x08080800U,+ 0x00000008U, 0x08000008U, 0x00080008U, 0x08080008U, 0x00000808U, 0x08000808U, 0x00080808U, 0x08080808U,+ 0x00000000U, 0x08000000U, 0x00080000U, 0x08080000U, 0x00000800U, 0x08000800U, 0x00080800U, 0x08080800U,+ 0x00000008U, 0x08000008U, 0x00080008U, 0x08080008U, 0x00000808U, 0x08000808U, 0x00080808U, 0x08080808U,+ 0x00000000U, 0x08000000U, 0x00080000U, 0x08080000U, 0x00000800U, 0x08000800U, 0x00080800U, 0x08080800U,+ 0x00000008U, 0x08000008U, 0x00080008U, 0x08080008U, 0x00000808U, 0x08000808U, 0x00080808U, 0x08080808U,+ 0x00000000U, 0x08000000U, 0x00080000U, 0x08080000U, 0x00000800U, 0x08000800U, 0x00080800U, 0x08080800U,+ 0x00000008U, 0x08000008U, 0x00080008U, 0x08080008U, 0x00000808U, 0x08000808U, 0x00080808U, 0x08080808U,+ 0x00000000U, 0x08000000U, 0x00080000U, 0x08080000U, 0x00000800U, 0x08000800U, 0x00080800U, 0x08080800U,+ 0x00000008U, 0x08000008U, 0x00080008U, 0x08080008U, 0x00000808U, 0x08000808U, 0x00080808U, 0x08080808U,+ 0x00000000U, 0x08000000U, 0x00080000U, 0x08080000U, 0x00000800U, 0x08000800U, 0x00080800U, 0x08080800U,+ 0x00000008U, 0x08000008U, 0x00080008U, 0x08080008U, 0x00000808U, 0x08000808U, 0x00080808U, 0x08080808U,+ 0x00000000U, 0x08000000U, 0x00080000U, 0x08080000U, 0x00000800U, 0x08000800U, 0x00080800U, 0x08080800U,+ 0x00000008U, 0x08000008U, 0x00080008U, 0x08080008U, 0x00000808U, 0x08000808U, 0x00080808U, 0x08080808U,+ 0x00000000U, 0x08000000U, 0x00080000U, 0x08080000U, 0x00000800U, 0x08000800U, 0x00080800U, 0x08080800U,+ 0x00000008U, 0x08000008U, 0x00080008U, 0x08080008U, 0x00000808U, 0x08000808U, 0x00080808U, 0x08080808U,+ 0x00000000U, 0x08000000U, 0x00080000U, 0x08080000U, 0x00000800U, 0x08000800U, 0x00080800U, 0x08080800U,+ 0x00000008U, 0x08000008U, 0x00080008U, 0x08080008U, 0x00000808U, 0x08000808U, 0x00080808U, 0x08080808U,+},+{+ 0x00000000U, 0x02000000U, 0x00020000U, 0x02020000U, 0x00000200U, 0x02000200U, 0x00020200U, 0x02020200U,+ 0x00000002U, 0x02000002U, 0x00020002U, 0x02020002U, 0x00000202U, 0x02000202U, 0x00020202U, 0x02020202U,+ 0x00000000U, 0x02000000U, 0x00020000U, 0x02020000U, 0x00000200U, 0x02000200U, 0x00020200U, 0x02020200U,+ 0x00000002U, 0x02000002U, 0x00020002U, 0x02020002U, 0x00000202U, 0x02000202U, 0x00020202U, 0x02020202U,+ 0x00000000U, 0x02000000U, 0x00020000U, 0x02020000U, 0x00000200U, 0x02000200U, 0x00020200U, 0x02020200U,+ 0x00000002U, 0x02000002U, 0x00020002U, 0x02020002U, 0x00000202U, 0x02000202U, 0x00020202U, 0x02020202U,+ 0x00000000U, 0x02000000U, 0x00020000U, 0x02020000U, 0x00000200U, 0x02000200U, 0x00020200U, 0x02020200U,+ 0x00000002U, 0x02000002U, 0x00020002U, 0x02020002U, 0x00000202U, 0x02000202U, 0x00020202U, 0x02020202U,+ 0x00000000U, 0x02000000U, 0x00020000U, 0x02020000U, 0x00000200U, 0x02000200U, 0x00020200U, 0x02020200U,+ 0x00000002U, 0x02000002U, 0x00020002U, 0x02020002U, 0x00000202U, 0x02000202U, 0x00020202U, 0x02020202U,+ 0x00000000U, 0x02000000U, 0x00020000U, 0x02020000U, 0x00000200U, 0x02000200U, 0x00020200U, 0x02020200U,+ 0x00000002U, 0x02000002U, 0x00020002U, 0x02020002U, 0x00000202U, 0x02000202U, 0x00020202U, 0x02020202U,+ 0x00000000U, 0x02000000U, 0x00020000U, 0x02020000U, 0x00000200U, 0x02000200U, 0x00020200U, 0x02020200U,+ 0x00000002U, 0x02000002U, 0x00020002U, 0x02020002U, 0x00000202U, 0x02000202U, 0x00020202U, 0x02020202U,+ 0x00000000U, 0x02000000U, 0x00020000U, 0x02020000U, 0x00000200U, 0x02000200U, 0x00020200U, 0x02020200U,+ 0x00000002U, 0x02000002U, 0x00020002U, 0x02020002U, 0x00000202U, 0x02000202U, 0x00020202U, 0x02020202U,+ 0x00000000U, 0x02000000U, 0x00020000U, 0x02020000U, 0x00000200U, 0x02000200U, 0x00020200U, 0x02020200U,+ 0x00000002U, 0x02000002U, 0x00020002U, 0x02020002U, 0x00000202U, 0x02000202U, 0x00020202U, 0x02020202U,+ 0x00000000U, 0x02000000U, 0x00020000U, 0x02020000U, 0x00000200U, 0x02000200U, 0x00020200U, 0x02020200U,+ 0x00000002U, 0x02000002U, 0x00020002U, 0x02020002U, 0x00000202U, 0x02000202U, 0x00020202U, 0x02020202U,+ 0x00000000U, 0x02000000U, 0x00020000U, 0x02020000U, 0x00000200U, 0x02000200U, 0x00020200U, 0x02020200U,+ 0x00000002U, 0x02000002U, 0x00020002U, 0x02020002U, 0x00000202U, 0x02000202U, 0x00020202U, 0x02020202U,+ 0x00000000U, 0x02000000U, 0x00020000U, 0x02020000U, 0x00000200U, 0x02000200U, 0x00020200U, 0x02020200U,+ 0x00000002U, 0x02000002U, 0x00020002U, 0x02020002U, 0x00000202U, 0x02000202U, 0x00020202U, 0x02020202U,+ 0x00000000U, 0x02000000U, 0x00020000U, 0x02020000U, 0x00000200U, 0x02000200U, 0x00020200U, 0x02020200U,+ 0x00000002U, 0x02000002U, 0x00020002U, 0x02020002U, 0x00000202U, 0x02000202U, 0x00020202U, 0x02020202U,+ 0x00000000U, 0x02000000U, 0x00020000U, 0x02020000U, 0x00000200U, 0x02000200U, 0x00020200U, 0x02020200U,+ 0x00000002U, 0x02000002U, 0x00020002U, 0x02020002U, 0x00000202U, 0x02000202U, 0x00020202U, 0x02020202U,+ 0x00000000U, 0x02000000U, 0x00020000U, 0x02020000U, 0x00000200U, 0x02000200U, 0x00020200U, 0x02020200U,+ 0x00000002U, 0x02000002U, 0x00020002U, 0x02020002U, 0x00000202U, 0x02000202U, 0x00020202U, 0x02020202U,+ 0x00000000U, 0x02000000U, 0x00020000U, 0x02020000U, 0x00000200U, 0x02000200U, 0x00020200U, 0x02020200U,+ 0x00000002U, 0x02000002U, 0x00020002U, 0x02020002U, 0x00000202U, 0x02000202U, 0x00020202U, 0x02020202U,+},+};++static const uint32_t FPL[8][256] = {+{+ 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U,+ 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U,+ 0x40000000U, 0x40000000U, 0x40000000U, 0x40000000U, 0x40000000U, 0x40000000U, 0x40000000U, 0x40000000U,+ 0x40000000U, 0x40000000U, 0x40000000U, 0x40000000U, 0x40000000U, 0x40000000U, 0x40000000U, 0x40000000U,+ 0x00400000U, 0x00400000U, 0x00400000U, 0x00400000U, 0x00400000U, 0x00400000U, 0x00400000U, 0x00400000U,+ 0x00400000U, 0x00400000U, 0x00400000U, 0x00400000U, 0x00400000U, 0x00400000U, 0x00400000U, 0x00400000U,+ 0x40400000U, 0x40400000U, 0x40400000U, 0x40400000U, 0x40400000U, 0x40400000U, 0x40400000U, 0x40400000U,+ 0x40400000U, 0x40400000U, 0x40400000U, 0x40400000U, 0x40400000U, 0x40400000U, 0x40400000U, 0x40400000U,+ 0x00004000U, 0x00004000U, 0x00004000U, 0x00004000U, 0x00004000U, 0x00004000U, 0x00004000U, 0x00004000U,+ 0x00004000U, 0x00004000U, 0x00004000U, 0x00004000U, 0x00004000U, 0x00004000U, 0x00004000U, 0x00004000U,+ 0x40004000U, 0x40004000U, 0x40004000U, 0x40004000U, 0x40004000U, 0x40004000U, 0x40004000U, 0x40004000U,+ 0x40004000U, 0x40004000U, 0x40004000U, 0x40004000U, 0x40004000U, 0x40004000U, 0x40004000U, 0x40004000U,+ 0x00404000U, 0x00404000U, 0x00404000U, 0x00404000U, 0x00404000U, 0x00404000U, 0x00404000U, 0x00404000U,+ 0x00404000U, 0x00404000U, 0x00404000U, 0x00404000U, 0x00404000U, 0x00404000U, 0x00404000U, 0x00404000U,+ 0x40404000U, 0x40404000U, 0x40404000U, 0x40404000U, 0x40404000U, 0x40404000U, 0x40404000U, 0x40404000U,+ 0x40404000U, 0x40404000U, 0x40404000U, 0x40404000U, 0x40404000U, 0x40404000U, 0x40404000U, 0x40404000U,+ 0x00000040U, 0x00000040U, 0x00000040U, 0x00000040U, 0x00000040U, 0x00000040U, 0x00000040U, 0x00000040U,+ 0x00000040U, 0x00000040U, 0x00000040U, 0x00000040U, 0x00000040U, 0x00000040U, 0x00000040U, 0x00000040U,+ 0x40000040U, 0x40000040U, 0x40000040U, 0x40000040U, 0x40000040U, 0x40000040U, 0x40000040U, 0x40000040U,+ 0x40000040U, 0x40000040U, 0x40000040U, 0x40000040U, 0x40000040U, 0x40000040U, 0x40000040U, 0x40000040U,+ 0x00400040U, 0x00400040U, 0x00400040U, 0x00400040U, 0x00400040U, 0x00400040U, 0x00400040U, 0x00400040U,+ 0x00400040U, 0x00400040U, 0x00400040U, 0x00400040U, 0x00400040U, 0x00400040U, 0x00400040U, 0x00400040U,+ 0x40400040U, 0x40400040U, 0x40400040U, 0x40400040U, 0x40400040U, 0x40400040U, 0x40400040U, 0x40400040U,+ 0x40400040U, 0x40400040U, 0x40400040U, 0x40400040U, 0x40400040U, 0x40400040U, 0x40400040U, 0x40400040U,+ 0x00004040U, 0x00004040U, 0x00004040U, 0x00004040U, 0x00004040U, 0x00004040U, 0x00004040U, 0x00004040U,+ 0x00004040U, 0x00004040U, 0x00004040U, 0x00004040U, 0x00004040U, 0x00004040U, 0x00004040U, 0x00004040U,+ 0x40004040U, 0x40004040U, 0x40004040U, 0x40004040U, 0x40004040U, 0x40004040U, 0x40004040U, 0x40004040U,+ 0x40004040U, 0x40004040U, 0x40004040U, 0x40004040U, 0x40004040U, 0x40004040U, 0x40004040U, 0x40004040U,+ 0x00404040U, 0x00404040U, 0x00404040U, 0x00404040U, 0x00404040U, 0x00404040U, 0x00404040U, 0x00404040U,+ 0x00404040U, 0x00404040U, 0x00404040U, 0x00404040U, 0x00404040U, 0x00404040U, 0x00404040U, 0x00404040U,+ 0x40404040U, 0x40404040U, 0x40404040U, 0x40404040U, 0x40404040U, 0x40404040U, 0x40404040U, 0x40404040U,+ 0x40404040U, 0x40404040U, 0x40404040U, 0x40404040U, 0x40404040U, 0x40404040U, 0x40404040U, 0x40404040U,+},+{+ 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U,+ 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U,+ 0x10000000U, 0x10000000U, 0x10000000U, 0x10000000U, 0x10000000U, 0x10000000U, 0x10000000U, 0x10000000U,+ 0x10000000U, 0x10000000U, 0x10000000U, 0x10000000U, 0x10000000U, 0x10000000U, 0x10000000U, 0x10000000U,+ 0x00100000U, 0x00100000U, 0x00100000U, 0x00100000U, 0x00100000U, 0x00100000U, 0x00100000U, 0x00100000U,+ 0x00100000U, 0x00100000U, 0x00100000U, 0x00100000U, 0x00100000U, 0x00100000U, 0x00100000U, 0x00100000U,+ 0x10100000U, 0x10100000U, 0x10100000U, 0x10100000U, 0x10100000U, 0x10100000U, 0x10100000U, 0x10100000U,+ 0x10100000U, 0x10100000U, 0x10100000U, 0x10100000U, 0x10100000U, 0x10100000U, 0x10100000U, 0x10100000U,+ 0x00001000U, 0x00001000U, 0x00001000U, 0x00001000U, 0x00001000U, 0x00001000U, 0x00001000U, 0x00001000U,+ 0x00001000U, 0x00001000U, 0x00001000U, 0x00001000U, 0x00001000U, 0x00001000U, 0x00001000U, 0x00001000U,+ 0x10001000U, 0x10001000U, 0x10001000U, 0x10001000U, 0x10001000U, 0x10001000U, 0x10001000U, 0x10001000U,+ 0x10001000U, 0x10001000U, 0x10001000U, 0x10001000U, 0x10001000U, 0x10001000U, 0x10001000U, 0x10001000U,+ 0x00101000U, 0x00101000U, 0x00101000U, 0x00101000U, 0x00101000U, 0x00101000U, 0x00101000U, 0x00101000U,+ 0x00101000U, 0x00101000U, 0x00101000U, 0x00101000U, 0x00101000U, 0x00101000U, 0x00101000U, 0x00101000U,+ 0x10101000U, 0x10101000U, 0x10101000U, 0x10101000U, 0x10101000U, 0x10101000U, 0x10101000U, 0x10101000U,+ 0x10101000U, 0x10101000U, 0x10101000U, 0x10101000U, 0x10101000U, 0x10101000U, 0x10101000U, 0x10101000U,+ 0x00000010U, 0x00000010U, 0x00000010U, 0x00000010U, 0x00000010U, 0x00000010U, 0x00000010U, 0x00000010U,+ 0x00000010U, 0x00000010U, 0x00000010U, 0x00000010U, 0x00000010U, 0x00000010U, 0x00000010U, 0x00000010U,+ 0x10000010U, 0x10000010U, 0x10000010U, 0x10000010U, 0x10000010U, 0x10000010U, 0x10000010U, 0x10000010U,+ 0x10000010U, 0x10000010U, 0x10000010U, 0x10000010U, 0x10000010U, 0x10000010U, 0x10000010U, 0x10000010U,+ 0x00100010U, 0x00100010U, 0x00100010U, 0x00100010U, 0x00100010U, 0x00100010U, 0x00100010U, 0x00100010U,+ 0x00100010U, 0x00100010U, 0x00100010U, 0x00100010U, 0x00100010U, 0x00100010U, 0x00100010U, 0x00100010U,+ 0x10100010U, 0x10100010U, 0x10100010U, 0x10100010U, 0x10100010U, 0x10100010U, 0x10100010U, 0x10100010U,+ 0x10100010U, 0x10100010U, 0x10100010U, 0x10100010U, 0x10100010U, 0x10100010U, 0x10100010U, 0x10100010U,+ 0x00001010U, 0x00001010U, 0x00001010U, 0x00001010U, 0x00001010U, 0x00001010U, 0x00001010U, 0x00001010U,+ 0x00001010U, 0x00001010U, 0x00001010U, 0x00001010U, 0x00001010U, 0x00001010U, 0x00001010U, 0x00001010U,+ 0x10001010U, 0x10001010U, 0x10001010U, 0x10001010U, 0x10001010U, 0x10001010U, 0x10001010U, 0x10001010U,+ 0x10001010U, 0x10001010U, 0x10001010U, 0x10001010U, 0x10001010U, 0x10001010U, 0x10001010U, 0x10001010U,+ 0x00101010U, 0x00101010U, 0x00101010U, 0x00101010U, 0x00101010U, 0x00101010U, 0x00101010U, 0x00101010U,+ 0x00101010U, 0x00101010U, 0x00101010U, 0x00101010U, 0x00101010U, 0x00101010U, 0x00101010U, 0x00101010U,+ 0x10101010U, 0x10101010U, 0x10101010U, 0x10101010U, 0x10101010U, 0x10101010U, 0x10101010U, 0x10101010U,+ 0x10101010U, 0x10101010U, 0x10101010U, 0x10101010U, 0x10101010U, 0x10101010U, 0x10101010U, 0x10101010U,+},+{+ 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U,+ 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U,+ 0x04000000U, 0x04000000U, 0x04000000U, 0x04000000U, 0x04000000U, 0x04000000U, 0x04000000U, 0x04000000U,+ 0x04000000U, 0x04000000U, 0x04000000U, 0x04000000U, 0x04000000U, 0x04000000U, 0x04000000U, 0x04000000U,+ 0x00040000U, 0x00040000U, 0x00040000U, 0x00040000U, 0x00040000U, 0x00040000U, 0x00040000U, 0x00040000U,+ 0x00040000U, 0x00040000U, 0x00040000U, 0x00040000U, 0x00040000U, 0x00040000U, 0x00040000U, 0x00040000U,+ 0x04040000U, 0x04040000U, 0x04040000U, 0x04040000U, 0x04040000U, 0x04040000U, 0x04040000U, 0x04040000U,+ 0x04040000U, 0x04040000U, 0x04040000U, 0x04040000U, 0x04040000U, 0x04040000U, 0x04040000U, 0x04040000U,+ 0x00000400U, 0x00000400U, 0x00000400U, 0x00000400U, 0x00000400U, 0x00000400U, 0x00000400U, 0x00000400U,+ 0x00000400U, 0x00000400U, 0x00000400U, 0x00000400U, 0x00000400U, 0x00000400U, 0x00000400U, 0x00000400U,+ 0x04000400U, 0x04000400U, 0x04000400U, 0x04000400U, 0x04000400U, 0x04000400U, 0x04000400U, 0x04000400U,+ 0x04000400U, 0x04000400U, 0x04000400U, 0x04000400U, 0x04000400U, 0x04000400U, 0x04000400U, 0x04000400U,+ 0x00040400U, 0x00040400U, 0x00040400U, 0x00040400U, 0x00040400U, 0x00040400U, 0x00040400U, 0x00040400U,+ 0x00040400U, 0x00040400U, 0x00040400U, 0x00040400U, 0x00040400U, 0x00040400U, 0x00040400U, 0x00040400U,+ 0x04040400U, 0x04040400U, 0x04040400U, 0x04040400U, 0x04040400U, 0x04040400U, 0x04040400U, 0x04040400U,+ 0x04040400U, 0x04040400U, 0x04040400U, 0x04040400U, 0x04040400U, 0x04040400U, 0x04040400U, 0x04040400U,+ 0x00000004U, 0x00000004U, 0x00000004U, 0x00000004U, 0x00000004U, 0x00000004U, 0x00000004U, 0x00000004U,+ 0x00000004U, 0x00000004U, 0x00000004U, 0x00000004U, 0x00000004U, 0x00000004U, 0x00000004U, 0x00000004U,+ 0x04000004U, 0x04000004U, 0x04000004U, 0x04000004U, 0x04000004U, 0x04000004U, 0x04000004U, 0x04000004U,+ 0x04000004U, 0x04000004U, 0x04000004U, 0x04000004U, 0x04000004U, 0x04000004U, 0x04000004U, 0x04000004U,+ 0x00040004U, 0x00040004U, 0x00040004U, 0x00040004U, 0x00040004U, 0x00040004U, 0x00040004U, 0x00040004U,+ 0x00040004U, 0x00040004U, 0x00040004U, 0x00040004U, 0x00040004U, 0x00040004U, 0x00040004U, 0x00040004U,+ 0x04040004U, 0x04040004U, 0x04040004U, 0x04040004U, 0x04040004U, 0x04040004U, 0x04040004U, 0x04040004U,+ 0x04040004U, 0x04040004U, 0x04040004U, 0x04040004U, 0x04040004U, 0x04040004U, 0x04040004U, 0x04040004U,+ 0x00000404U, 0x00000404U, 0x00000404U, 0x00000404U, 0x00000404U, 0x00000404U, 0x00000404U, 0x00000404U,+ 0x00000404U, 0x00000404U, 0x00000404U, 0x00000404U, 0x00000404U, 0x00000404U, 0x00000404U, 0x00000404U,+ 0x04000404U, 0x04000404U, 0x04000404U, 0x04000404U, 0x04000404U, 0x04000404U, 0x04000404U, 0x04000404U,+ 0x04000404U, 0x04000404U, 0x04000404U, 0x04000404U, 0x04000404U, 0x04000404U, 0x04000404U, 0x04000404U,+ 0x00040404U, 0x00040404U, 0x00040404U, 0x00040404U, 0x00040404U, 0x00040404U, 0x00040404U, 0x00040404U,+ 0x00040404U, 0x00040404U, 0x00040404U, 0x00040404U, 0x00040404U, 0x00040404U, 0x00040404U, 0x00040404U,+ 0x04040404U, 0x04040404U, 0x04040404U, 0x04040404U, 0x04040404U, 0x04040404U, 0x04040404U, 0x04040404U,+ 0x04040404U, 0x04040404U, 0x04040404U, 0x04040404U, 0x04040404U, 0x04040404U, 0x04040404U, 0x04040404U,+},+{+ 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U,+ 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U,+ 0x01000000U, 0x01000000U, 0x01000000U, 0x01000000U, 0x01000000U, 0x01000000U, 0x01000000U, 0x01000000U,+ 0x01000000U, 0x01000000U, 0x01000000U, 0x01000000U, 0x01000000U, 0x01000000U, 0x01000000U, 0x01000000U,+ 0x00010000U, 0x00010000U, 0x00010000U, 0x00010000U, 0x00010000U, 0x00010000U, 0x00010000U, 0x00010000U,+ 0x00010000U, 0x00010000U, 0x00010000U, 0x00010000U, 0x00010000U, 0x00010000U, 0x00010000U, 0x00010000U,+ 0x01010000U, 0x01010000U, 0x01010000U, 0x01010000U, 0x01010000U, 0x01010000U, 0x01010000U, 0x01010000U,+ 0x01010000U, 0x01010000U, 0x01010000U, 0x01010000U, 0x01010000U, 0x01010000U, 0x01010000U, 0x01010000U,+ 0x00000100U, 0x00000100U, 0x00000100U, 0x00000100U, 0x00000100U, 0x00000100U, 0x00000100U, 0x00000100U,+ 0x00000100U, 0x00000100U, 0x00000100U, 0x00000100U, 0x00000100U, 0x00000100U, 0x00000100U, 0x00000100U,+ 0x01000100U, 0x01000100U, 0x01000100U, 0x01000100U, 0x01000100U, 0x01000100U, 0x01000100U, 0x01000100U,+ 0x01000100U, 0x01000100U, 0x01000100U, 0x01000100U, 0x01000100U, 0x01000100U, 0x01000100U, 0x01000100U,+ 0x00010100U, 0x00010100U, 0x00010100U, 0x00010100U, 0x00010100U, 0x00010100U, 0x00010100U, 0x00010100U,+ 0x00010100U, 0x00010100U, 0x00010100U, 0x00010100U, 0x00010100U, 0x00010100U, 0x00010100U, 0x00010100U,+ 0x01010100U, 0x01010100U, 0x01010100U, 0x01010100U, 0x01010100U, 0x01010100U, 0x01010100U, 0x01010100U,+ 0x01010100U, 0x01010100U, 0x01010100U, 0x01010100U, 0x01010100U, 0x01010100U, 0x01010100U, 0x01010100U,+ 0x00000001U, 0x00000001U, 0x00000001U, 0x00000001U, 0x00000001U, 0x00000001U, 0x00000001U, 0x00000001U,+ 0x00000001U, 0x00000001U, 0x00000001U, 0x00000001U, 0x00000001U, 0x00000001U, 0x00000001U, 0x00000001U,+ 0x01000001U, 0x01000001U, 0x01000001U, 0x01000001U, 0x01000001U, 0x01000001U, 0x01000001U, 0x01000001U,+ 0x01000001U, 0x01000001U, 0x01000001U, 0x01000001U, 0x01000001U, 0x01000001U, 0x01000001U, 0x01000001U,+ 0x00010001U, 0x00010001U, 0x00010001U, 0x00010001U, 0x00010001U, 0x00010001U, 0x00010001U, 0x00010001U,+ 0x00010001U, 0x00010001U, 0x00010001U, 0x00010001U, 0x00010001U, 0x00010001U, 0x00010001U, 0x00010001U,+ 0x01010001U, 0x01010001U, 0x01010001U, 0x01010001U, 0x01010001U, 0x01010001U, 0x01010001U, 0x01010001U,+ 0x01010001U, 0x01010001U, 0x01010001U, 0x01010001U, 0x01010001U, 0x01010001U, 0x01010001U, 0x01010001U,+ 0x00000101U, 0x00000101U, 0x00000101U, 0x00000101U, 0x00000101U, 0x00000101U, 0x00000101U, 0x00000101U,+ 0x00000101U, 0x00000101U, 0x00000101U, 0x00000101U, 0x00000101U, 0x00000101U, 0x00000101U, 0x00000101U,+ 0x01000101U, 0x01000101U, 0x01000101U, 0x01000101U, 0x01000101U, 0x01000101U, 0x01000101U, 0x01000101U,+ 0x01000101U, 0x01000101U, 0x01000101U, 0x01000101U, 0x01000101U, 0x01000101U, 0x01000101U, 0x01000101U,+ 0x00010101U, 0x00010101U, 0x00010101U, 0x00010101U, 0x00010101U, 0x00010101U, 0x00010101U, 0x00010101U,+ 0x00010101U, 0x00010101U, 0x00010101U, 0x00010101U, 0x00010101U, 0x00010101U, 0x00010101U, 0x00010101U,+ 0x01010101U, 0x01010101U, 0x01010101U, 0x01010101U, 0x01010101U, 0x01010101U, 0x01010101U, 0x01010101U,+ 0x01010101U, 0x01010101U, 0x01010101U, 0x01010101U, 0x01010101U, 0x01010101U, 0x01010101U, 0x01010101U,+},+{+ 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U,+ 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U,+ 0x80000000U, 0x80000000U, 0x80000000U, 0x80000000U, 0x80000000U, 0x80000000U, 0x80000000U, 0x80000000U,+ 0x80000000U, 0x80000000U, 0x80000000U, 0x80000000U, 0x80000000U, 0x80000000U, 0x80000000U, 0x80000000U,+ 0x00800000U, 0x00800000U, 0x00800000U, 0x00800000U, 0x00800000U, 0x00800000U, 0x00800000U, 0x00800000U,+ 0x00800000U, 0x00800000U, 0x00800000U, 0x00800000U, 0x00800000U, 0x00800000U, 0x00800000U, 0x00800000U,+ 0x80800000U, 0x80800000U, 0x80800000U, 0x80800000U, 0x80800000U, 0x80800000U, 0x80800000U, 0x80800000U,+ 0x80800000U, 0x80800000U, 0x80800000U, 0x80800000U, 0x80800000U, 0x80800000U, 0x80800000U, 0x80800000U,+ 0x00008000U, 0x00008000U, 0x00008000U, 0x00008000U, 0x00008000U, 0x00008000U, 0x00008000U, 0x00008000U,+ 0x00008000U, 0x00008000U, 0x00008000U, 0x00008000U, 0x00008000U, 0x00008000U, 0x00008000U, 0x00008000U,+ 0x80008000U, 0x80008000U, 0x80008000U, 0x80008000U, 0x80008000U, 0x80008000U, 0x80008000U, 0x80008000U,+ 0x80008000U, 0x80008000U, 0x80008000U, 0x80008000U, 0x80008000U, 0x80008000U, 0x80008000U, 0x80008000U,+ 0x00808000U, 0x00808000U, 0x00808000U, 0x00808000U, 0x00808000U, 0x00808000U, 0x00808000U, 0x00808000U,+ 0x00808000U, 0x00808000U, 0x00808000U, 0x00808000U, 0x00808000U, 0x00808000U, 0x00808000U, 0x00808000U,+ 0x80808000U, 0x80808000U, 0x80808000U, 0x80808000U, 0x80808000U, 0x80808000U, 0x80808000U, 0x80808000U,+ 0x80808000U, 0x80808000U, 0x80808000U, 0x80808000U, 0x80808000U, 0x80808000U, 0x80808000U, 0x80808000U,+ 0x00000080U, 0x00000080U, 0x00000080U, 0x00000080U, 0x00000080U, 0x00000080U, 0x00000080U, 0x00000080U,+ 0x00000080U, 0x00000080U, 0x00000080U, 0x00000080U, 0x00000080U, 0x00000080U, 0x00000080U, 0x00000080U,+ 0x80000080U, 0x80000080U, 0x80000080U, 0x80000080U, 0x80000080U, 0x80000080U, 0x80000080U, 0x80000080U,+ 0x80000080U, 0x80000080U, 0x80000080U, 0x80000080U, 0x80000080U, 0x80000080U, 0x80000080U, 0x80000080U,+ 0x00800080U, 0x00800080U, 0x00800080U, 0x00800080U, 0x00800080U, 0x00800080U, 0x00800080U, 0x00800080U,+ 0x00800080U, 0x00800080U, 0x00800080U, 0x00800080U, 0x00800080U, 0x00800080U, 0x00800080U, 0x00800080U,+ 0x80800080U, 0x80800080U, 0x80800080U, 0x80800080U, 0x80800080U, 0x80800080U, 0x80800080U, 0x80800080U,+ 0x80800080U, 0x80800080U, 0x80800080U, 0x80800080U, 0x80800080U, 0x80800080U, 0x80800080U, 0x80800080U,+ 0x00008080U, 0x00008080U, 0x00008080U, 0x00008080U, 0x00008080U, 0x00008080U, 0x00008080U, 0x00008080U,+ 0x00008080U, 0x00008080U, 0x00008080U, 0x00008080U, 0x00008080U, 0x00008080U, 0x00008080U, 0x00008080U,+ 0x80008080U, 0x80008080U, 0x80008080U, 0x80008080U, 0x80008080U, 0x80008080U, 0x80008080U, 0x80008080U,+ 0x80008080U, 0x80008080U, 0x80008080U, 0x80008080U, 0x80008080U, 0x80008080U, 0x80008080U, 0x80008080U,+ 0x00808080U, 0x00808080U, 0x00808080U, 0x00808080U, 0x00808080U, 0x00808080U, 0x00808080U, 0x00808080U,+ 0x00808080U, 0x00808080U, 0x00808080U, 0x00808080U, 0x00808080U, 0x00808080U, 0x00808080U, 0x00808080U,+ 0x80808080U, 0x80808080U, 0x80808080U, 0x80808080U, 0x80808080U, 0x80808080U, 0x80808080U, 0x80808080U,+ 0x80808080U, 0x80808080U, 0x80808080U, 0x80808080U, 0x80808080U, 0x80808080U, 0x80808080U, 0x80808080U,+},+{+ 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U,+ 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U,+ 0x20000000U, 0x20000000U, 0x20000000U, 0x20000000U, 0x20000000U, 0x20000000U, 0x20000000U, 0x20000000U,+ 0x20000000U, 0x20000000U, 0x20000000U, 0x20000000U, 0x20000000U, 0x20000000U, 0x20000000U, 0x20000000U,+ 0x00200000U, 0x00200000U, 0x00200000U, 0x00200000U, 0x00200000U, 0x00200000U, 0x00200000U, 0x00200000U,+ 0x00200000U, 0x00200000U, 0x00200000U, 0x00200000U, 0x00200000U, 0x00200000U, 0x00200000U, 0x00200000U,+ 0x20200000U, 0x20200000U, 0x20200000U, 0x20200000U, 0x20200000U, 0x20200000U, 0x20200000U, 0x20200000U,+ 0x20200000U, 0x20200000U, 0x20200000U, 0x20200000U, 0x20200000U, 0x20200000U, 0x20200000U, 0x20200000U,+ 0x00002000U, 0x00002000U, 0x00002000U, 0x00002000U, 0x00002000U, 0x00002000U, 0x00002000U, 0x00002000U,+ 0x00002000U, 0x00002000U, 0x00002000U, 0x00002000U, 0x00002000U, 0x00002000U, 0x00002000U, 0x00002000U,+ 0x20002000U, 0x20002000U, 0x20002000U, 0x20002000U, 0x20002000U, 0x20002000U, 0x20002000U, 0x20002000U,+ 0x20002000U, 0x20002000U, 0x20002000U, 0x20002000U, 0x20002000U, 0x20002000U, 0x20002000U, 0x20002000U,+ 0x00202000U, 0x00202000U, 0x00202000U, 0x00202000U, 0x00202000U, 0x00202000U, 0x00202000U, 0x00202000U,+ 0x00202000U, 0x00202000U, 0x00202000U, 0x00202000U, 0x00202000U, 0x00202000U, 0x00202000U, 0x00202000U,+ 0x20202000U, 0x20202000U, 0x20202000U, 0x20202000U, 0x20202000U, 0x20202000U, 0x20202000U, 0x20202000U,+ 0x20202000U, 0x20202000U, 0x20202000U, 0x20202000U, 0x20202000U, 0x20202000U, 0x20202000U, 0x20202000U,+ 0x00000020U, 0x00000020U, 0x00000020U, 0x00000020U, 0x00000020U, 0x00000020U, 0x00000020U, 0x00000020U,+ 0x00000020U, 0x00000020U, 0x00000020U, 0x00000020U, 0x00000020U, 0x00000020U, 0x00000020U, 0x00000020U,+ 0x20000020U, 0x20000020U, 0x20000020U, 0x20000020U, 0x20000020U, 0x20000020U, 0x20000020U, 0x20000020U,+ 0x20000020U, 0x20000020U, 0x20000020U, 0x20000020U, 0x20000020U, 0x20000020U, 0x20000020U, 0x20000020U,+ 0x00200020U, 0x00200020U, 0x00200020U, 0x00200020U, 0x00200020U, 0x00200020U, 0x00200020U, 0x00200020U,+ 0x00200020U, 0x00200020U, 0x00200020U, 0x00200020U, 0x00200020U, 0x00200020U, 0x00200020U, 0x00200020U,+ 0x20200020U, 0x20200020U, 0x20200020U, 0x20200020U, 0x20200020U, 0x20200020U, 0x20200020U, 0x20200020U,+ 0x20200020U, 0x20200020U, 0x20200020U, 0x20200020U, 0x20200020U, 0x20200020U, 0x20200020U, 0x20200020U,+ 0x00002020U, 0x00002020U, 0x00002020U, 0x00002020U, 0x00002020U, 0x00002020U, 0x00002020U, 0x00002020U,+ 0x00002020U, 0x00002020U, 0x00002020U, 0x00002020U, 0x00002020U, 0x00002020U, 0x00002020U, 0x00002020U,+ 0x20002020U, 0x20002020U, 0x20002020U, 0x20002020U, 0x20002020U, 0x20002020U, 0x20002020U, 0x20002020U,+ 0x20002020U, 0x20002020U, 0x20002020U, 0x20002020U, 0x20002020U, 0x20002020U, 0x20002020U, 0x20002020U,+ 0x00202020U, 0x00202020U, 0x00202020U, 0x00202020U, 0x00202020U, 0x00202020U, 0x00202020U, 0x00202020U,+ 0x00202020U, 0x00202020U, 0x00202020U, 0x00202020U, 0x00202020U, 0x00202020U, 0x00202020U, 0x00202020U,+ 0x20202020U, 0x20202020U, 0x20202020U, 0x20202020U, 0x20202020U, 0x20202020U, 0x20202020U, 0x20202020U,+ 0x20202020U, 0x20202020U, 0x20202020U, 0x20202020U, 0x20202020U, 0x20202020U, 0x20202020U, 0x20202020U,+},+{+ 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U,+ 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U,+ 0x08000000U, 0x08000000U, 0x08000000U, 0x08000000U, 0x08000000U, 0x08000000U, 0x08000000U, 0x08000000U,+ 0x08000000U, 0x08000000U, 0x08000000U, 0x08000000U, 0x08000000U, 0x08000000U, 0x08000000U, 0x08000000U,+ 0x00080000U, 0x00080000U, 0x00080000U, 0x00080000U, 0x00080000U, 0x00080000U, 0x00080000U, 0x00080000U,+ 0x00080000U, 0x00080000U, 0x00080000U, 0x00080000U, 0x00080000U, 0x00080000U, 0x00080000U, 0x00080000U,+ 0x08080000U, 0x08080000U, 0x08080000U, 0x08080000U, 0x08080000U, 0x08080000U, 0x08080000U, 0x08080000U,+ 0x08080000U, 0x08080000U, 0x08080000U, 0x08080000U, 0x08080000U, 0x08080000U, 0x08080000U, 0x08080000U,+ 0x00000800U, 0x00000800U, 0x00000800U, 0x00000800U, 0x00000800U, 0x00000800U, 0x00000800U, 0x00000800U,+ 0x00000800U, 0x00000800U, 0x00000800U, 0x00000800U, 0x00000800U, 0x00000800U, 0x00000800U, 0x00000800U,+ 0x08000800U, 0x08000800U, 0x08000800U, 0x08000800U, 0x08000800U, 0x08000800U, 0x08000800U, 0x08000800U,+ 0x08000800U, 0x08000800U, 0x08000800U, 0x08000800U, 0x08000800U, 0x08000800U, 0x08000800U, 0x08000800U,+ 0x00080800U, 0x00080800U, 0x00080800U, 0x00080800U, 0x00080800U, 0x00080800U, 0x00080800U, 0x00080800U,+ 0x00080800U, 0x00080800U, 0x00080800U, 0x00080800U, 0x00080800U, 0x00080800U, 0x00080800U, 0x00080800U,+ 0x08080800U, 0x08080800U, 0x08080800U, 0x08080800U, 0x08080800U, 0x08080800U, 0x08080800U, 0x08080800U,+ 0x08080800U, 0x08080800U, 0x08080800U, 0x08080800U, 0x08080800U, 0x08080800U, 0x08080800U, 0x08080800U,+ 0x00000008U, 0x00000008U, 0x00000008U, 0x00000008U, 0x00000008U, 0x00000008U, 0x00000008U, 0x00000008U,+ 0x00000008U, 0x00000008U, 0x00000008U, 0x00000008U, 0x00000008U, 0x00000008U, 0x00000008U, 0x00000008U,+ 0x08000008U, 0x08000008U, 0x08000008U, 0x08000008U, 0x08000008U, 0x08000008U, 0x08000008U, 0x08000008U,+ 0x08000008U, 0x08000008U, 0x08000008U, 0x08000008U, 0x08000008U, 0x08000008U, 0x08000008U, 0x08000008U,+ 0x00080008U, 0x00080008U, 0x00080008U, 0x00080008U, 0x00080008U, 0x00080008U, 0x00080008U, 0x00080008U,+ 0x00080008U, 0x00080008U, 0x00080008U, 0x00080008U, 0x00080008U, 0x00080008U, 0x00080008U, 0x00080008U,+ 0x08080008U, 0x08080008U, 0x08080008U, 0x08080008U, 0x08080008U, 0x08080008U, 0x08080008U, 0x08080008U,+ 0x08080008U, 0x08080008U, 0x08080008U, 0x08080008U, 0x08080008U, 0x08080008U, 0x08080008U, 0x08080008U,+ 0x00000808U, 0x00000808U, 0x00000808U, 0x00000808U, 0x00000808U, 0x00000808U, 0x00000808U, 0x00000808U,+ 0x00000808U, 0x00000808U, 0x00000808U, 0x00000808U, 0x00000808U, 0x00000808U, 0x00000808U, 0x00000808U,+ 0x08000808U, 0x08000808U, 0x08000808U, 0x08000808U, 0x08000808U, 0x08000808U, 0x08000808U, 0x08000808U,+ 0x08000808U, 0x08000808U, 0x08000808U, 0x08000808U, 0x08000808U, 0x08000808U, 0x08000808U, 0x08000808U,+ 0x00080808U, 0x00080808U, 0x00080808U, 0x00080808U, 0x00080808U, 0x00080808U, 0x00080808U, 0x00080808U,+ 0x00080808U, 0x00080808U, 0x00080808U, 0x00080808U, 0x00080808U, 0x00080808U, 0x00080808U, 0x00080808U,+ 0x08080808U, 0x08080808U, 0x08080808U, 0x08080808U, 0x08080808U, 0x08080808U, 0x08080808U, 0x08080808U,+ 0x08080808U, 0x08080808U, 0x08080808U, 0x08080808U, 0x08080808U, 0x08080808U, 0x08080808U, 0x08080808U,+},+{+ 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U,+ 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U, 0x00000000U,+ 0x02000000U, 0x02000000U, 0x02000000U, 0x02000000U, 0x02000000U, 0x02000000U, 0x02000000U, 0x02000000U,+ 0x02000000U, 0x02000000U, 0x02000000U, 0x02000000U, 0x02000000U, 0x02000000U, 0x02000000U, 0x02000000U,+ 0x00020000U, 0x00020000U, 0x00020000U, 0x00020000U, 0x00020000U, 0x00020000U, 0x00020000U, 0x00020000U,+ 0x00020000U, 0x00020000U, 0x00020000U, 0x00020000U, 0x00020000U, 0x00020000U, 0x00020000U, 0x00020000U,+ 0x02020000U, 0x02020000U, 0x02020000U, 0x02020000U, 0x02020000U, 0x02020000U, 0x02020000U, 0x02020000U,+ 0x02020000U, 0x02020000U, 0x02020000U, 0x02020000U, 0x02020000U, 0x02020000U, 0x02020000U, 0x02020000U,+ 0x00000200U, 0x00000200U, 0x00000200U, 0x00000200U, 0x00000200U, 0x00000200U, 0x00000200U, 0x00000200U,+ 0x00000200U, 0x00000200U, 0x00000200U, 0x00000200U, 0x00000200U, 0x00000200U, 0x00000200U, 0x00000200U,+ 0x02000200U, 0x02000200U, 0x02000200U, 0x02000200U, 0x02000200U, 0x02000200U, 0x02000200U, 0x02000200U,+ 0x02000200U, 0x02000200U, 0x02000200U, 0x02000200U, 0x02000200U, 0x02000200U, 0x02000200U, 0x02000200U,+ 0x00020200U, 0x00020200U, 0x00020200U, 0x00020200U, 0x00020200U, 0x00020200U, 0x00020200U, 0x00020200U,+ 0x00020200U, 0x00020200U, 0x00020200U, 0x00020200U, 0x00020200U, 0x00020200U, 0x00020200U, 0x00020200U,+ 0x02020200U, 0x02020200U, 0x02020200U, 0x02020200U, 0x02020200U, 0x02020200U, 0x02020200U, 0x02020200U,+ 0x02020200U, 0x02020200U, 0x02020200U, 0x02020200U, 0x02020200U, 0x02020200U, 0x02020200U, 0x02020200U,+ 0x00000002U, 0x00000002U, 0x00000002U, 0x00000002U, 0x00000002U, 0x00000002U, 0x00000002U, 0x00000002U,+ 0x00000002U, 0x00000002U, 0x00000002U, 0x00000002U, 0x00000002U, 0x00000002U, 0x00000002U, 0x00000002U,+ 0x02000002U, 0x02000002U, 0x02000002U, 0x02000002U, 0x02000002U, 0x02000002U, 0x02000002U, 0x02000002U,+ 0x02000002U, 0x02000002U, 0x02000002U, 0x02000002U, 0x02000002U, 0x02000002U, 0x02000002U, 0x02000002U,+ 0x00020002U, 0x00020002U, 0x00020002U, 0x00020002U, 0x00020002U, 0x00020002U, 0x00020002U, 0x00020002U,+ 0x00020002U, 0x00020002U, 0x00020002U, 0x00020002U, 0x00020002U, 0x00020002U, 0x00020002U, 0x00020002U,+ 0x02020002U, 0x02020002U, 0x02020002U, 0x02020002U, 0x02020002U, 0x02020002U, 0x02020002U, 0x02020002U,+ 0x02020002U, 0x02020002U, 0x02020002U, 0x02020002U, 0x02020002U, 0x02020002U, 0x02020002U, 0x02020002U,+ 0x00000202U, 0x00000202U, 0x00000202U, 0x00000202U, 0x00000202U, 0x00000202U, 0x00000202U, 0x00000202U,+ 0x00000202U, 0x00000202U, 0x00000202U, 0x00000202U, 0x00000202U, 0x00000202U, 0x00000202U, 0x00000202U,+ 0x02000202U, 0x02000202U, 0x02000202U, 0x02000202U, 0x02000202U, 0x02000202U, 0x02000202U, 0x02000202U,+ 0x02000202U, 0x02000202U, 0x02000202U, 0x02000202U, 0x02000202U, 0x02000202U, 0x02000202U, 0x02000202U,+ 0x00020202U, 0x00020202U, 0x00020202U, 0x00020202U, 0x00020202U, 0x00020202U, 0x00020202U, 0x00020202U,+ 0x00020202U, 0x00020202U, 0x00020202U, 0x00020202U, 0x00020202U, 0x00020202U, 0x00020202U, 0x00020202U,+ 0x02020202U, 0x02020202U, 0x02020202U, 0x02020202U, 0x02020202U, 0x02020202U, 0x02020202U, 0x02020202U,+ 0x02020202U, 0x02020202U, 0x02020202U, 0x02020202U, 0x02020202U, 0x02020202U, 0x02020202U, 0x02020202U,+},+};++#define ROTL32(v, k) (((v) << (k)) | ((v) >> (32 - (k))))++static inline uint64_t load_be64(const uint8_t *p)+{+ return ((uint64_t) p[0] << 56) | ((uint64_t) p[1] << 48)+ | ((uint64_t) p[2] << 40) | ((uint64_t) p[3] << 32)+ | ((uint64_t) p[4] << 24) | ((uint64_t) p[5] << 16)+ | ((uint64_t) p[6] << 8) | ((uint64_t) p[7]);+}++static inline void store_be64(uint8_t *p, uint64_t v)+{+ p[0] = (uint8_t) (v >> 56); p[1] = (uint8_t) (v >> 48);+ p[2] = (uint8_t) (v >> 40); p[3] = (uint8_t) (v >> 32);+ p[4] = (uint8_t) (v >> 24); p[5] = (uint8_t) (v >> 16);+ p[6] = (uint8_t) (v >> 8); p[7] = (uint8_t) v;+}++/* bit i of v, numbered from 1 at the most significant end, as FIPS 46-3 does */+static inline uint32_t bit_of(uint64_t v, int i, int width)+{+ return (uint32_t) ((v >> (width - i)) & 1);+}++static const uint8_t PC1[56] = {+ 57,49,41,33,25,17,9,1,58,50,42,34,26,18,10,2,59,51,43,35,27,+ 19,11,3,60,52,44,36,63,55,47,39,31,23,15,7,62,54,46,38,30,22,+ 14,6,61,53,45,37,29,21,13,5,28,20,12,4+};++static const uint8_t PC2[48] = {+ 14,17,11,24,1,5,3,28,15,6,21,10,23,19,12,4,26,8,16,7,27,20,13,2,+ 41,52,31,37,47,55,30,40,51,45,33,48,44,49,39,56,34,53,46,42,50,36,29,32+};++static const uint8_t SHIFTS[16] = { 1,1,2,2,2,2,2,2,1,2,2,2,2,2,2,1 };++void crypton_des_init(crypton_des_key *ks, const uint8_t *key, int reverse)+{+ uint64_t k = load_be64(key);+ uint64_t cd = 0;+ uint32_t c, d;+ int round, i;++ for (i = 0; i < 56; i++)+ cd = (cd << 1) | bit_of(k, PC1[i], 64);+ c = (uint32_t) (cd >> 28);+ d = (uint32_t) (cd & 0x0fffffffU);++ for (round = 0; round < 16; round++) {+ uint64_t merged;+ uint8_t *sk;+ int s = SHIFTS[round];++ c = ((c << s) | (c >> (28 - s))) & 0x0fffffffU;+ d = ((d << s) | (d >> (28 - s))) & 0x0fffffffU;+ merged = ((uint64_t) c << 28) | d;++ sk = ks->sk + (reverse ? (15 - round) : round) * 8;+ memset(sk, 0, 8);+ for (i = 0; i < 48; i++)+ sk[i / 6] = (uint8_t) ((sk[i / 6] << 1) | bit_of(merged, PC2[i], 56));+ }+}++static inline uint32_t des_f(uint32_t r, const uint8_t *sk)+{+ return SP[0][((ROTL32(r, 31) >> 26) & 0x3f) ^ sk[0]]+ | SP[1][((ROTL32(r, 3) >> 26) & 0x3f) ^ sk[1]]+ | SP[2][((ROTL32(r, 7) >> 26) & 0x3f) ^ sk[2]]+ | SP[3][((ROTL32(r, 11) >> 26) & 0x3f) ^ sk[3]]+ | SP[4][((ROTL32(r, 15) >> 26) & 0x3f) ^ sk[4]]+ | SP[5][((ROTL32(r, 19) >> 26) & 0x3f) ^ sk[5]]+ | SP[6][((ROTL32(r, 23) >> 26) & 0x3f) ^ sk[6]]+ | SP[7][((ROTL32(r, 27) >> 26) & 0x3f) ^ sk[7]];+}++static inline uint64_t des_block(uint64_t b, const uint8_t *sk)+{+ uint32_t l, r;+ int round;++ l = IPL[0][(b >> 56) & 0xff] | IPL[1][(b >> 48) & 0xff]+ | IPL[2][(b >> 40) & 0xff] | IPL[3][(b >> 32) & 0xff]+ | IPL[4][(b >> 24) & 0xff] | IPL[5][(b >> 16) & 0xff]+ | IPL[6][(b >> 8) & 0xff] | IPL[7][ b & 0xff];+ r = IPR[0][(b >> 56) & 0xff] | IPR[1][(b >> 48) & 0xff]+ | IPR[2][(b >> 40) & 0xff] | IPR[3][(b >> 32) & 0xff]+ | IPR[4][(b >> 24) & 0xff] | IPR[5][(b >> 16) & 0xff]+ | IPR[6][(b >> 8) & 0xff] | IPR[7][ b & 0xff];++ for (round = 0; round < 16; round++) {+ uint32_t t = r;+ r = l ^ des_f(r, sk + round * 8);+ l = t;+ }++ {+ /* the preoutput is the halves the other way round */+ uint64_t p = ((uint64_t) r << 32) | l;+ return ((uint64_t) (FPH[0][(p >> 56) & 0xff] | FPH[1][(p >> 48) & 0xff]+ | FPH[2][(p >> 40) & 0xff] | FPH[3][(p >> 32) & 0xff]+ | FPH[4][(p >> 24) & 0xff] | FPH[5][(p >> 16) & 0xff]+ | FPH[6][(p >> 8) & 0xff] | FPH[7][ p & 0xff]) << 32)+ | (FPL[0][(p >> 56) & 0xff] | FPL[1][(p >> 48) & 0xff]+ | FPL[2][(p >> 40) & 0xff] | FPL[3][(p >> 32) & 0xff]+ | FPL[4][(p >> 24) & 0xff] | FPL[5][(p >> 16) & 0xff]+ | FPL[6][(p >> 8) & 0xff] | FPL[7][ p & 0xff]);+ }+}++/* Run every stage of the schedule over each block: one stage is DES, three are+ * EDE or EEE depending on the directions the schedules were built for. */+void crypton_des_ecb(uint8_t *out, const crypton_des_key *ks, uint32_t nkeys,+ const uint8_t *in, uint32_t nblocks)+{+ uint32_t i, s;++ for (i = 0; i < nblocks; i++) {+ uint64_t b = load_be64(in + 8 * i);+ for (s = 0; s < nkeys; s++)+ b = des_block(b, ks[s].sk);+ store_be64(out + 8 * i, b);+ }+}
@@ -0,0 +1,20 @@+#ifndef CRYPTON_DES_H+#define CRYPTON_DES_H++#include <stdint.h>++/* the sixteen round keys, as eight six-bit values each */+typedef struct {+ uint8_t sk[16 * 8];+} crypton_des_key;++/* Build a schedule from an eight byte key. The parity bits are ignored, as+ * FIPS 46-3 says. With reverse set, the rounds come out in the order that+ * decrypts. */+void crypton_des_init(crypton_des_key *ks, const uint8_t *key, int reverse);++/* Apply nkeys schedules in order to each of nblocks eight byte blocks. */+void crypton_des_ecb(uint8_t *out, const crypton_des_key *ks, uint32_t nkeys,+ const uint8_t *in, uint32_t nblocks);++#endif
@@ -0,0 +1,542 @@+/*+ * Scalar multiplication on a curve over a prime field, doing the same work+ * whatever the scalar is.+ *+ * The scalar is walked four bits at a time: four doublings and one addition+ * of a small multiple of the point, taken from a table of sixteen that is+ * read by touching every entry and keeping one of them with a mask. So a+ * window costs the same five operations and the same sixteen reads whatever+ * its bits are, and nothing branches on, or indexes memory with, the scalar.+ *+ * The addition and the doubling are the complete formulas of Renes, Costello+ * and Batina (eprint 2015/1060, algorithms 1 and 3), which answer for every+ * pair of points there is -- the same point twice, a point and its negation,+ * the point at infinity -- without a case to choose between. A formula with+ * cases would need the choice to be made with a mask like everything else+ * here, and would still have to be right about which cases there are; these+ * have none. They cost about half again what the usual Jacobian formulas do,+ * which is the price of that.+ *+ * Points are kept in homogeneous projective coordinates, where the point at+ * infinity is (0 : 1 : 0), and in Montgomery form, so that the only reduction+ * is the one the multiplication does anyway.+ */+#include <stdlib.h>+#include <crypton_bignum.h>+#include <crypton_bzero.h>+#include <crypton_ecc.h>+#include <crypton_powm.h>+#ifdef CRYPTON_S2N_BIGNUM+#include <crypton_ecc_s2n.h>+#endif++/* four bits of scalar per window, so a table of sixteen and no leftover+ * bits: a byte holds exactly two windows */+#define WINDOW_BITS 4+#define TABLE_SIZE (1 << WINDOW_BITS)++/* the field the curve is over, and what it takes to work in it */+typedef struct {+ uint32_t n; /* limbs in a field element */+ limb_t n0; /* -p^-1 mod 2^LIMB_BITS */+ const limb_t *p;+ const limb_t *a; /* the curve's a, in Montgomery form */+ const limb_t *b3; /* three times the curve's b, in Montgomery form */+ const limb_t *zero; /* n limbs of nothing, to subtract from */+ int a_is_zero; /* a is 0 or p-3 for every curve in use, and then */+ int a_is_minus3; /* multiplying by it is additions instead */+ limb_t *t; /* 2n of scratch, for the multiplication */+ limb_t *s; /* n of scratch, for the addition and subtraction */+ limb_t *s2; /* n more, for multiplying by a, which may write over+ * what it is reading */+} field;++static void fe_mul(const field *f, limb_t *r, const limb_t *x, const limb_t *y)+{+ mont_mul(r, x, y, f->p, f->n0, f->n, f->t);+}++static void fe_sqr(const field *f, limb_t *r, const limb_t *x)+{+ mont_sqr(r, x, f->p, f->n0, f->n, f->t);+}++static void fe_add(const field *f, limb_t *r, const limb_t *x, const limb_t *y)+{+ limb_t carry = add_n(r, x, y, f->n);+ limb_t borrow = sub_n(f->s, r, f->p, f->n);++ select_n(r, f->s, r, (carry | (borrow ^ 1)) & 1, f->n);+}++static void fe_sub(const field *f, limb_t *r, const limb_t *x, const limb_t *y)+{+ limb_t borrow = sub_n(r, x, y, f->n);++ add_n(f->s, r, f->p, f->n);+ select_n(r, f->s, r, borrow, f->n);+}++/* r = -x */+static void fe_neg(const field *f, limb_t *r, const limb_t *x)+{+ fe_sub(f, r, f->zero, x);+}++/* r = a * x, where a is the curve's. It is zero or minus three on every+ * curve in use, and then this is additions rather than a multiplication.+ * Which of the three it is comes from the curve, which is public. */+static void fe_mul_a(const field *f, limb_t *r, const limb_t *x)+{+ if (f->a_is_zero) {+ memset(r, 0, f->n * sizeof(limb_t));+ } else if (f->a_is_minus3) {+ /* r and x are the same buffer in places, so this goes through one+ * of its own */+ fe_add(f, f->s2, x, x);+ fe_add(f, f->s2, f->s2, x);+ fe_neg(f, r, f->s2);+ } else {+ fe_mul(f, r, f->a, x);+ }+}++/* Renes-Costello-Batina algorithm 1: r = x + y, for any two points */+static void point_add(const field *f, limb_t *r, const limb_t *x,+ const limb_t *y, limb_t *w)+{+ uint32_t n = f->n;+ const limb_t *x1 = x, *y1 = x + n, *z1 = x + 2 * n;+ const limb_t *x2 = y, *y2 = y + n, *z2 = y + 2 * n;+ limb_t *t0 = w, *t1 = w + n, *t2 = w + 2 * n, *t3 = w + 3 * n;+ limb_t *t4 = w + 4 * n, *t5 = w + 5 * n;+ limb_t *x3 = w + 6 * n, *y3 = w + 7 * n, *z3 = w + 8 * n;++ fe_mul(f, t0, x1, x2);+ fe_mul(f, t1, y1, y2);+ fe_mul(f, t2, z1, z2);+ fe_add(f, t3, x1, y1);+ fe_add(f, t4, x2, y2);+ fe_mul(f, t3, t3, t4);+ fe_add(f, t4, t0, t1);+ fe_sub(f, t3, t3, t4);+ fe_add(f, t4, x1, z1);+ fe_add(f, t5, x2, z2);+ fe_mul(f, t4, t4, t5);+ fe_add(f, t5, t0, t2);+ fe_sub(f, t4, t4, t5);+ fe_add(f, t5, y1, z1);+ fe_add(f, x3, y2, z2);+ fe_mul(f, t5, t5, x3);+ fe_add(f, x3, t1, t2);+ fe_sub(f, t5, t5, x3);+ fe_mul_a(f, z3, t4);+ fe_mul(f, x3, f->b3, t2);+ fe_add(f, z3, x3, z3);+ fe_sub(f, x3, t1, z3);+ fe_add(f, z3, t1, z3);+ fe_mul(f, y3, x3, z3);+ fe_add(f, t1, t0, t0);+ fe_add(f, t1, t1, t0);+ fe_mul_a(f, t2, t2);+ fe_mul(f, t4, f->b3, t4);+ fe_add(f, t1, t1, t2);+ fe_sub(f, t2, t0, t2);+ fe_mul_a(f, t2, t2);+ fe_add(f, t4, t4, t2);+ fe_mul(f, t0, t1, t4);+ fe_add(f, y3, y3, t0);+ fe_mul(f, t0, t5, t4);+ fe_mul(f, x3, t3, x3);+ fe_sub(f, x3, x3, t0);+ fe_mul(f, t0, t3, t1);+ fe_mul(f, t1, t5, z3);+ fe_add(f, z3, t1, t0);++ memcpy(r, x3, n * sizeof(limb_t));+ memcpy(r + n, y3, n * sizeof(limb_t));+ memcpy(r + 2 * n, z3, n * sizeof(limb_t));+}++/* Renes-Costello-Batina algorithm 3: r = x + x, for any point */+static void point_double(const field *f, limb_t *r, const limb_t *x, limb_t *w)+{+ uint32_t n = f->n;+ const limb_t *px = x, *py = x + n, *pz = x + 2 * n;+ limb_t *t0 = w, *t1 = w + n, *t2 = w + 2 * n, *t3 = w + 3 * n;+ limb_t *x3 = w + 6 * n, *y3 = w + 7 * n, *z3 = w + 8 * n;++ fe_sqr(f, t0, px);+ fe_sqr(f, t1, py);+ fe_sqr(f, t2, pz);+ fe_mul(f, t3, px, py);+ fe_add(f, t3, t3, t3);+ fe_mul(f, z3, px, pz);+ fe_add(f, z3, z3, z3);+ fe_mul_a(f, x3, z3);+ fe_mul(f, y3, f->b3, t2);+ fe_add(f, y3, x3, y3);+ fe_sub(f, x3, t1, y3);+ fe_add(f, y3, t1, y3);+ fe_mul(f, y3, x3, y3);+ fe_mul(f, x3, t3, x3);+ fe_mul(f, z3, f->b3, z3);+ fe_mul_a(f, t2, t2);+ fe_sub(f, t3, t0, t2);+ fe_mul_a(f, t3, t3);+ fe_add(f, t3, t3, z3);+ fe_add(f, z3, t0, t0);+ fe_add(f, t0, z3, t0);+ fe_add(f, t0, t0, t2);+ fe_mul(f, t0, t0, t3);+ fe_add(f, y3, y3, t0);+ fe_mul(f, t2, py, pz);+ fe_add(f, t2, t2, t2);+ fe_mul(f, t0, t2, t3);+ fe_sub(f, x3, x3, t0);+ fe_mul(f, z3, t2, t1);+ fe_add(f, z3, z3, z3);+ fe_add(f, z3, z3, z3);++ memcpy(r, x3, n * sizeof(limb_t));+ memcpy(r + n, y3, n * sizeof(limb_t));+ memcpy(r + 2 * n, z3, n * sizeof(limb_t));+}++/* Everything a curve needs, in one allocation: the field, the buffers the+ * formulas work in, and a table of sixteen points. The caller frees it with+ * ctx_free. */+typedef struct {+ field f;+ limb_t *space;+ uint32_t words;+ uint32_t n;+ limb_t *r2; /* R^2 mod p, which is what takes a number to Montgomery form */+ limb_t *one; /* 1, in Montgomery form */+ limb_t *acc; /* a point */+ limb_t *sel; /* a point */+ limb_t *tmp; /* a point */+ limb_t *work; /* 9n, for the formulas */+ limb_t *table; /* sixteen points */+ uint8_t *bytes; /* 2 * plen, for the inversion */+ uint32_t plen;+} curve_ctx;++static void ctx_free(curve_ctx *c)+{+ /* crypton_bzero rather than memset: this memory is freed on the next+ * line, and a store to memory about to die is one an optimizer may+ * drop. Both buffers have held scalars. */+ if (c->space != NULL) {+ crypton_bzero(c->space, c->words * sizeof(limb_t));+ free(c->space);+ }+ if (c->bytes != NULL) {+ crypton_bzero(c->bytes, 2 * c->plen);+ free(c->bytes);+ }+ c->space = NULL;+ c->bytes = NULL;+}++/* r = x, taken into Montgomery form */+static void to_mont(const curve_ctx *c, limb_t *r, const limb_t *x)+{+ mont_mul(r, x, c->r2, c->f.p, c->f.n0, c->n, c->f.t);+}++/* r = x, taken back out of it */+static void from_mont(const curve_ctx *c, limb_t *r, const limb_t *x)+{+ mont_mul(r, x, c->one, c->f.p, c->f.n0, c->n, c->f.t);+}++static int ctx_init(curve_ctx *c, const uint8_t *a, const uint8_t *b,+ const uint8_t *p, uint32_t plen)+{+ uint32_t n = (plen + LIMB_BYTES - 1) / LIMB_BYTES;+ limb_t *mp, *ma, *mb3, *zero, *scratch, *mont_t;+ uint32_t i;++ memset(c, 0, sizeof(*c));+ if (plen == 0 || n == 0 || (p[plen - 1] & 1) == 0)+ return -1;++ /* six single numbers, three points, four of scratch, nine for the+ * formulas, and a table of sixteen points */+ c->n = n;+ c->plen = plen;+ c->words = (6 + 9 + 4 + 9 + 3 * TABLE_SIZE) * n;+ c->space = calloc(c->words, sizeof(limb_t));+ c->bytes = calloc(2, plen);+ if (c->space == NULL || c->bytes == NULL) {+ ctx_free(c);+ return -1;+ }+ mp = c->space;+ ma = mp + n;+ mb3 = ma + n;+ c->r2 = mb3 + n;+ c->one = c->r2 + n;+ zero = c->one + n;+ c->acc = zero + n;+ c->sel = c->acc + 3 * n;+ c->tmp = c->sel + 3 * n;+ scratch = c->tmp + 3 * n;+ mont_t = scratch + 2 * n;+ c->work = mont_t + 2 * n;+ c->table = c->work + 9 * n;++ if (from_be(mp, n, p, plen) != 0) {+ ctx_free(c);+ return -1;+ }+ c->f.n0 = mont_n0(mp[0]);+ mont_r2(c->r2, mp, c->f.n0, n, mont_t);++ c->f.n = n;+ c->f.p = mp;+ c->f.a = ma;+ c->f.b3 = mb3;+ c->f.zero = zero;+ c->f.t = mont_t;+ c->f.s = scratch;+ c->f.s2 = scratch + n;+ c->f.a_is_zero = 0;+ c->f.a_is_minus3 = 0;++ memset(c->one, 0, n * sizeof(limb_t));+ c->one[0] = 1;+ to_mont(c, c->tmp, c->one);+ memcpy(c->one, c->tmp, n * sizeof(limb_t));++ /* the curve's a, and which of the three shapes it has */+ if (from_be(c->tmp, n, a, plen) != 0) {+ ctx_free(c);+ return -1;+ }+ {+ limb_t nonzero = 0, differs = 0;++ for (i = 0; i < n; i++)+ nonzero |= c->tmp[i];+ memset(c->sel, 0, n * sizeof(limb_t));+ c->sel[0] = 3;+ sub_n(c->sel, mp, c->sel, n); /* p - 3 */+ for (i = 0; i < n; i++)+ differs |= c->tmp[i] ^ c->sel[i];+ c->f.a_is_zero = nonzero == 0;+ c->f.a_is_minus3 = differs == 0;+ }+ to_mont(c, ma, c->tmp);++ /* three times the curve's b, which is what the formulas want */+ if (from_be(c->tmp, n, b, plen) != 0) {+ ctx_free(c);+ return -1;+ }+ to_mont(c, mb3, c->tmp);+ fe_add(&c->f, c->tmp, mb3, mb3);+ fe_add(&c->f, mb3, c->tmp, mb3);+ return 0;+}++/* a point, in Montgomery form, from its coordinates */+static int point_from_be(const curve_ctx *c, limb_t *r, const uint8_t *px,+ const uint8_t *py)+{+ uint32_t n = c->n;++ if (from_be(c->tmp, n, px, c->plen) != 0)+ return -1;+ to_mont(c, r, c->tmp);+ if (from_be(c->tmp, n, py, c->plen) != 0)+ return -1;+ to_mont(c, r + n, c->tmp);+ memcpy(r + 2 * n, c->one, n * sizeof(limb_t));+ return 0;+}++/* x = X/Z and y = Y/Z, with the inverse from Fermat, which is the+ * exponentiation that hides its exponent. Returns 1 for the point at+ * infinity, which has no coordinates. */+static int point_to_be(curve_ctx *c, uint8_t *outx, uint8_t *outy,+ const limb_t *pt, const uint8_t *p)+{+ uint32_t n = c->n, plen = c->plen, i;+ limb_t empty = 0;+ uint8_t *zbytes = c->bytes, *pm2 = c->bytes + plen;++ for (i = 0; i < n; i++)+ empty |= pt[2 * n + i];+ if (empty == 0)+ return 1;++ from_mont(c, c->tmp, pt + 2 * n);+ to_be(zbytes, plen, c->tmp, n);+ memset(c->sel, 0, n * sizeof(limb_t));+ c->sel[0] = 2;+ sub_n(c->sel, c->f.p, c->sel, n); /* p - 2 */+ to_be(pm2, plen, c->sel, n);+ if (crypton_powm_sec(zbytes, zbytes, plen, pm2, plen, p, plen) != 0)+ return -1;+ if (from_be(c->tmp, n, zbytes, plen) != 0)+ return -1;+ to_mont(c, c->sel, c->tmp); /* 1/Z, in Montgomery form */++ fe_mul(&c->f, c->tmp, pt, c->sel);+ from_mont(c, c->tmp + n, c->tmp);+ to_be(outx, plen, c->tmp + n, n);++ fe_mul(&c->f, c->tmp, pt + n, c->sel);+ from_mont(c, c->tmp + n, c->tmp);+ to_be(outy, plen, c->tmp + n, n);+ return 0;+}++/* every one of the sixteen entries is read, and a mask keeps the one wanted */+static void table_select(const curve_ctx *c, limb_t *r, const limb_t *table,+ limb_t w)+{+ uint32_t n = c->n, j, l;++ memset(r, 0, 3 * n * sizeof(limb_t));+ for (j = 0; j < TABLE_SIZE; j++) {+ limb_t mask = eq_mask(j, w);++ for (l = 0; l < 3 * n; l++)+ r[l] |= table[3 * j * n + l] & mask;+ }+}++int crypton_ecc_mul(uint8_t *outx, uint8_t *outy,+ const uint8_t *px, const uint8_t *py,+ const uint8_t *k, uint32_t klen,+ const uint8_t *a, const uint8_t *b,+ const uint8_t *p, uint32_t plen)+{+ curve_ctx c;+ uint32_t n, i, j;+ int ret = -1;++#ifdef CRYPTON_S2N_BIGNUM+ /* Two of the curves that reach here have hand-written assembly, six+ * to ten times faster than what follows; see cbits/s2n/README.md.+ * Anything else, including those two named with a different a or b,+ * goes on down. */+ {+ int s2n_ret;++ if (crypton_s2n_ecc_mul(&s2n_ret, outx, outy, px, py, k, klen,+ a, b, p, plen))+ return s2n_ret;+ }+#endif++ if (klen == 0 || ctx_init(&c, a, b, p, plen) != 0)+ return -1;+ n = c.n;++ /* the table: nothing, the point, and its multiples up to fifteen */+ memset(c.table, 0, 3 * n * sizeof(limb_t));+ memcpy(c.table + n, c.one, n * sizeof(limb_t)); /* (0 : 1 : 0) */+ if (point_from_be(&c, c.table + 3 * n, px, py) != 0)+ goto done;+ for (i = 2; i < TABLE_SIZE; i++)+ point_add(&c.f, c.table + 3 * i * n, c.table + 3 * (i - 1) * n,+ c.table + 3 * n, c.work);++ /* four bits at a time, from the top */+ memcpy(c.acc, c.table, 3 * n * sizeof(limb_t));+ for (i = klen * 2; i > 0; i--) {+ uint32_t nib = i - 1;+ limb_t w = (k[klen - 1 - nib / 2] >> (4 * (nib % 2))) & 0xf;++ for (j = 0; j < WINDOW_BITS; j++)+ point_double(&c.f, c.acc, c.acc, c.work);+ table_select(&c, c.sel, c.table, w);+ point_add(&c.f, c.acc, c.acc, c.sel, c.work);+ }+ ret = point_to_be(&c, outx, outy, c.acc, p);++done:+ ctx_free(&c);+ return ret;+}++uint32_t crypton_ecc_table_size(uint32_t plen, uint32_t klen)+{+ uint32_t n = (plen + LIMB_BYTES - 1) / LIMB_BYTES;++ if (plen == 0 || klen == 0 || n == 0)+ return 0;+ return klen * 2 * TABLE_SIZE * 3 * n * (uint32_t) sizeof(limb_t);+}++int crypton_ecc_table_build(uint8_t *tab,+ const uint8_t *gx, const uint8_t *gy,+ uint32_t klen,+ const uint8_t *a, const uint8_t *b,+ const uint8_t *p, uint32_t plen)+{+ curve_ctx c;+ limb_t *t = (limb_t *) (void *) tab;+ uint32_t n, i, j, windows;+ int ret = -1;++ if (klen == 0 || ctx_init(&c, a, b, p, plen) != 0)+ return -1;+ n = c.n;+ windows = klen * 2;++ /* acc walks the powers: at window i it holds 16^i times the point */+ if (point_from_be(&c, c.acc, gx, gy) != 0)+ goto done;+ for (i = 0; i < windows; i++) {+ limb_t *slot = t + (size_t) i * TABLE_SIZE * 3 * n;++ memset(slot, 0, 3 * n * sizeof(limb_t));+ memcpy(slot + n, c.one, n * sizeof(limb_t)); /* (0 : 1 : 0) */+ memcpy(slot + 3 * n, c.acc, 3 * n * sizeof(limb_t));+ for (j = 2; j < TABLE_SIZE; j++)+ point_add(&c.f, slot + 3 * j * n, slot + 3 * (j - 1) * n,+ c.acc, c.work);+ for (j = 0; j < WINDOW_BITS; j++)+ point_double(&c.f, c.acc, c.acc, c.work);+ }+ ret = 0;++done:+ ctx_free(&c);+ return ret;+}++int crypton_ecc_table_mul(uint8_t *outx, uint8_t *outy, const uint8_t *tab,+ const uint8_t *k, uint32_t klen,+ const uint8_t *a, const uint8_t *b,+ const uint8_t *p, uint32_t plen)+{+ curve_ctx c;+ const limb_t *t = (const limb_t *) (const void *) tab;+ uint32_t n, i;+ int ret;++ if (klen == 0 || ctx_init(&c, a, b, p, plen) != 0)+ return -1;+ n = c.n;++ /* nothing to start with, and one addition for every four bits: the+ * multiples the doubling would work out are all in the table */+ memset(c.acc, 0, 3 * n * sizeof(limb_t));+ memcpy(c.acc + n, c.one, n * sizeof(limb_t));+ for (i = 0; i < klen * 2; i++) {+ limb_t w = (k[klen - 1 - i / 2] >> (4 * (i % 2))) & 0xf;++ table_select(&c, c.sel, t + (size_t) i * TABLE_SIZE * 3 * n, w);+ point_add(&c.f, c.acc, c.acc, c.sel, c.work);+ }+ ret = point_to_be(&c, outx, outy, c.acc, p);++ ctx_free(&c);+ return ret;+}
@@ -0,0 +1,57 @@+#ifndef CRYPTON_ECC_H+#define CRYPTON_ECC_H++#include <stdint.h>++/* Multiply a point of a curve over a prime field by a scalar, doing the same+ * work whatever the scalar is.+ *+ * The curve is y^2 = x^3 + a*x + b over the field of p, which has to be an+ * odd prime; the point has to be on it and not the point at infinity, and its+ * coordinates, a and b have to be below p. Every number is a big-endian byte+ * string, and the coordinates, a, b and p are all plen bytes.+ *+ * The scalar is walked four bits at a time over every one of the klen bytes+ * it is given, so its value is hidden but its length is not.+ *+ * Returns 0 with the answer in outx and outy, 1 if the answer is the point at+ * infinity, which has no coordinates, and -1 if the arguments are not ones it+ * can work with or memory ran out.+ */+int crypton_ecc_mul(uint8_t *outx, uint8_t *outy,+ const uint8_t *px, const uint8_t *py,+ const uint8_t *k, uint32_t klen,+ const uint8_t *a, const uint8_t *b,+ const uint8_t *p, uint32_t plen);++/* How many bytes a table for a base point takes, for a prime of plen bytes+ * and scalars of klen. Zero if those sizes are not ones it can work with. */+uint32_t crypton_ecc_table_size(uint32_t plen, uint32_t klen);++/* Fill that many bytes with the multiples of a point that+ * crypton_ecc_table_mul wants: for every four bits of a scalar, the sixteen+ * points that those bits can call for. The buffer has to be aligned as a+ * pointer is, which is what an allocator gives.+ *+ * The point, a, b and p are as for crypton_ecc_mul. Returns 0, or -1 for+ * arguments it cannot work with or memory it could not have.+ */+int crypton_ecc_table_build(uint8_t *table,+ const uint8_t *gx, const uint8_t *gy,+ uint32_t klen,+ const uint8_t *a, const uint8_t *b,+ const uint8_t *p, uint32_t plen);++/* Multiply the point that table was built for by a scalar of klen bytes,+ * which has to be the klen the table was built for. One addition for every+ * four bits and no doublings, since the table holds what the doublings would+ * work out.+ *+ * Returns what crypton_ecc_mul returns.+ */+int crypton_ecc_table_mul(uint8_t *outx, uint8_t *outy, const uint8_t *table,+ const uint8_t *k, uint32_t klen,+ const uint8_t *a, const uint8_t *b,+ const uint8_t *p, uint32_t plen);++#endif
@@ -0,0 +1,204 @@+#include <string.h>++#include "crypton_ecc_s2n.h"+#include "crypton_ecc_s2n_curves.h"+#include "crypton_cpu.h"++/* P-384, Montgomery domain, six words a coordinate */+extern void p384_montjscalarmul(uint64_t *res, const uint64_t *s, const uint64_t *p);+extern void p384_montjscalarmul_alt(uint64_t *res, const uint64_t *s, const uint64_t *p);+extern void bignum_tomont_p384(uint64_t *z, const uint64_t *x);+extern void bignum_tomont_p384_alt(uint64_t *z, const uint64_t *x);+extern void bignum_deamont_p384(uint64_t *z, const uint64_t *x);+extern void bignum_deamont_p384_alt(uint64_t *z, const uint64_t *x);+extern void bignum_montmul_p384(uint64_t *z, const uint64_t *x, const uint64_t *y);+extern void bignum_montmul_p384_alt(uint64_t *z, const uint64_t *x, const uint64_t *y);+extern void bignum_montsqr_p384(uint64_t *z, const uint64_t *x);+extern void bignum_montsqr_p384_alt(uint64_t *z, const uint64_t *x);+extern void bignum_montinv_p384(uint64_t *z, const uint64_t *x);++/* P-521, ordinary values, nine words a coordinate */+extern void p521_jscalarmul(uint64_t *res, const uint64_t *s, const uint64_t *p);+extern void p521_jscalarmul_alt(uint64_t *res, const uint64_t *s, const uint64_t *p);+extern void bignum_mul_p521(uint64_t *z, const uint64_t *x, const uint64_t *y);+extern void bignum_mul_p521_alt(uint64_t *z, const uint64_t *x, const uint64_t *y);+extern void bignum_sqr_p521(uint64_t *z, const uint64_t *x);+extern void bignum_sqr_p521_alt(uint64_t *z, const uint64_t *x);+extern void bignum_inv_p521(uint64_t *z, const uint64_t *x);++/* One flavour or the other, all the way through: on x86-64 the plain form+ * wants MULX, ADCX and ADOX and _alt is the fallback, so mixing them would+ * fault on a machine without those. */+struct p384_asm {+ void (*tomont)(uint64_t *, const uint64_t *);+ void (*deamont)(uint64_t *, const uint64_t *);+ void (*montmul)(uint64_t *, const uint64_t *, const uint64_t *);+ void (*montsqr)(uint64_t *, const uint64_t *);+ void (*jscalarmul)(uint64_t *, const uint64_t *, const uint64_t *);+};+struct p521_asm {+ void (*mul)(uint64_t *, const uint64_t *, const uint64_t *);+ void (*sqr)(uint64_t *, const uint64_t *);+ void (*jscalarmul)(uint64_t *, const uint64_t *, const uint64_t *);+};++static const struct p384_asm p384_std = {+ bignum_tomont_p384, bignum_deamont_p384, bignum_montmul_p384,+ bignum_montsqr_p384, p384_montjscalarmul+};+static const struct p384_asm p384_alt = {+ bignum_tomont_p384_alt, bignum_deamont_p384_alt,+ bignum_montmul_p384_alt, bignum_montsqr_p384_alt,+ p384_montjscalarmul_alt+};+static const struct p521_asm p521_std = {+ bignum_mul_p521, bignum_sqr_p521, p521_jscalarmul+};+static const struct p521_asm p521_alt = {+ bignum_mul_p521_alt, bignum_sqr_p521_alt, p521_jscalarmul_alt+};++/* The same question as for P-256, answered the same way; see+ * cbits/p256/p256_s2n.c. */+static int use_alt(void)+{+#if defined(__aarch64__) || defined(__arm64__)+#ifdef __APPLE__+ return 1;+#else+ return 0;+#endif+#else+ return (crypton_x86_simd_features() & CRYPTON_X86_ADX) == 0;+#endif+}++#define MAXWORDS 9++static void be_to_le64(uint64_t *w, const uint8_t *b, uint32_t len)+{+ uint32_t i;++ for (i = 0; i < MAXWORDS; i++)+ w[i] = 0;+ for (i = 0; i < len; i++) {+ uint32_t pos = len - 1 - i;+ w[pos / 8] |= (uint64_t)b[i] << (8 * (pos % 8));+ }+}++static void le64_to_be(uint8_t *b, uint32_t len, const uint64_t *w)+{+ uint32_t i;++ for (i = 0; i < len; i++)+ b[len - 1 - i] = (uint8_t)(w[i / 8] >> (8 * (i % 8)));+}++static int is_zero(const uint64_t *w, int words)+{+ uint64_t acc = 0;+ int i;++ for (i = 0; i < words; i++)+ acc |= w[i];+ return acc == 0;+}++static int mul_p384(uint8_t *outx, uint8_t *outy, const uint8_t *px,+ const uint8_t *py, const uint8_t *k, uint32_t klen)+{+ const struct p384_asm *f = use_alt() ? &p384_alt : &p384_std;+ uint64_t pt[18], res[18], sc[MAXWORDS], t[MAXWORDS];+ uint64_t zi[6], zi2[6], zi3[6], num[6];+ static const uint64_t one[6] = {1, 0, 0, 0, 0, 0};++ be_to_le64(t, px, P384_PLEN);+ f->tomont(pt, t);+ be_to_le64(t, py, P384_PLEN);+ f->tomont(pt + 6, t);+ f->tomont(pt + 12, one);+ be_to_le64(sc, k, klen);++ f->jscalarmul(res, sc, pt);+ if (is_zero(res + 12, 6))+ return 1;++ /* affine again: x = X/Z^2, y = Y/Z^3, with the inverse taken in the+ * Montgomery domain so that it lands where the rest of these are */+ bignum_montinv_p384(zi, res + 12);+ f->montsqr(zi2, zi);+ f->montmul(zi3, zi2, zi);+ f->montmul(num, res, zi2);+ f->deamont(t, num);+ le64_to_be(outx, P384_PLEN, t);+ f->montmul(num, res + 6, zi3);+ f->deamont(t, num);+ le64_to_be(outy, P384_PLEN, t);+ return 0;+}++static int mul_p521(uint8_t *outx, uint8_t *outy, const uint8_t *px,+ const uint8_t *py, const uint8_t *k, uint32_t klen)+{+ const struct p521_asm *f = use_alt() ? &p521_alt : &p521_std;+ uint64_t pt[27], res[27], sc[MAXWORDS], t[MAXWORDS];+ uint64_t zi[9], zi2[9], zi3[9];+ static const uint64_t one[9] = {1, 0, 0, 0, 0, 0, 0, 0, 0};++ be_to_le64(pt, px, P521_PLEN);+ be_to_le64(pt + 9, py, P521_PLEN);+ memcpy(pt + 18, one, sizeof(one));+ be_to_le64(sc, k, klen);++ f->jscalarmul(res, sc, pt);+ if (is_zero(res + 18, 9))+ return 1;++ bignum_inv_p521(zi, res + 18);+ f->sqr(zi2, zi);+ f->mul(zi3, zi2, zi);+ f->mul(t, res, zi2);+ le64_to_be(outx, P521_PLEN, t);+ f->mul(t, res + 9, zi3);+ le64_to_be(outy, P521_PLEN, t);+ return 0;+}++int crypton_s2n_ecc_mul(int *ret, uint8_t *outx, uint8_t *outy,+ const uint8_t *px, const uint8_t *py,+ const uint8_t *k, uint32_t klen,+ const uint8_t *a, const uint8_t *b,+ const uint8_t *p, uint32_t plen)+{+ int is384;++ if (plen == P384_PLEN && memcmp(p, P384_P, plen) == 0+ && memcmp(a, P384_A, plen) == 0 && memcmp(b, P384_B, plen) == 0)+ is384 = 1;+ else if (plen == P521_PLEN && memcmp(p, P521_P, plen) == 0+ && memcmp(a, P521_A, plen) == 0+ && memcmp(b, P521_B, plen) == 0)+ is384 = 0;+ else+ return 0; /* some other curve; the C answers it */++ /* A scalar longer than the prime is not something the word arrays+ * here hold, and it is not what any caller of these two curves+ * sends, so leave it to the C rather than grow a second path. */+ if (klen == 0 || klen > plen)+ return 0;++ /* The C does not require the coordinates to be reduced -- it takes+ * whatever fits in its limbs and lets the conversion to Montgomery+ * form reduce it. Rather than carry a reduction here to match, hand+ * that case back: nothing sends one, and this way the two cannot+ * disagree about it. (A differential test against the C found this;+ * the first version of this check returned -1 and was wrong.) */+ if (memcmp(px, p, plen) >= 0 || memcmp(py, p, plen) >= 0)+ return 0;++ *ret = is384 ? mul_p384(outx, outy, px, py, k, klen)+ : mul_p521(outx, outy, px, py, k, klen);+ return 1;+}
@@ -0,0 +1,27 @@+/*+ * P-384 and P-521 through the vendored s2n-bignum assembly, for the two+ * curves it knows among the ones crypton_ecc_mul is asked about.+ *+ * s2n-bignum has no affine wrapper for these two -- only a scalar+ * multiplication on Jacobian points, Montgomery-domain for P-384 and plain+ * for P-521 -- so the conversions in and out are built here out of its own+ * field operations. See cbits/s2n/README.md.+ */+#ifndef CRYPTON_ECC_S2N_H+#define CRYPTON_ECC_S2N_H++#include <stdint.h>++/*+ * Returns 1 if this was a curve it knows and it answered, with *ret set to+ * what crypton_ecc_mul should return -- 0 and the point in outx and outy, 1+ * for the point at infinity, or -1 for arguments it will not take. Returns+ * 0 if the curve is not one of its two and nothing was written.+ */+int crypton_s2n_ecc_mul(int *ret, uint8_t *outx, uint8_t *outy,+ const uint8_t *px, const uint8_t *py,+ const uint8_t *k, uint32_t klen,+ const uint8_t *a, const uint8_t *b,+ const uint8_t *p, uint32_t plen);++#endif
@@ -0,0 +1,76 @@+/*+ * p, a and b of the two curves the vendored s2n-bignum assembly knows, as+ * big-endian bytes, which is how crypton_ecc_mul is given a curve. They are+ * here to be compared against, not computed with: a caller naming some other+ * curve with the same sizes has to go to the C.+ *+ * Generated from `openssl ecparam -name secp384r1 -param_enc explicit -text`+ * and the same for secp521r1, rather than transcribed.+ */+#ifndef CRYPTON_ECC_S2N_CURVES_H+#define CRYPTON_ECC_S2N_CURVES_H++#include <stdint.h>++#define P384_PLEN 48+static const uint8_t P384_P[48] = {+ 0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,+ 0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,+ 0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,+ 0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xfe,+ 0xff,0xff,0xff,0xff,0x00,0x00,0x00,0x00,+ 0x00,0x00,0x00,0x00,0xff,0xff,0xff,0xff+};+static const uint8_t P384_A[48] = {+ 0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,+ 0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,+ 0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,+ 0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xfe,+ 0xff,0xff,0xff,0xff,0x00,0x00,0x00,0x00,+ 0x00,0x00,0x00,0x00,0xff,0xff,0xff,0xfc+};+static const uint8_t P384_B[48] = {+ 0xb3,0x31,0x2f,0xa7,0xe2,0x3e,0xe7,0xe4,+ 0x98,0x8e,0x05,0x6b,0xe3,0xf8,0x2d,0x19,+ 0x18,0x1d,0x9c,0x6e,0xfe,0x81,0x41,0x12,+ 0x03,0x14,0x08,0x8f,0x50,0x13,0x87,0x5a,+ 0xc6,0x56,0x39,0x8d,0x8a,0x2e,0xd1,0x9d,+ 0x2a,0x85,0xc8,0xed,0xd3,0xec,0x2a,0xef+};++#define P521_PLEN 66+static const uint8_t P521_P[66] = {+ 0x01,0xff,0xff,0xff,0xff,0xff,0xff,0xff,+ 0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,+ 0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,+ 0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,+ 0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,+ 0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,+ 0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,+ 0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,+ 0xff,0xff+};+static const uint8_t P521_A[66] = {+ 0x01,0xff,0xff,0xff,0xff,0xff,0xff,0xff,+ 0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,+ 0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,+ 0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,+ 0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,+ 0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,+ 0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,+ 0xff,0xff,0xff,0xff,0xff,0xff,0xff,0xff,+ 0xff,0xfc+};+static const uint8_t P521_B[66] = {+ 0x00,0x51,0x95,0x3e,0xb9,0x61,0x8e,0x1c,+ 0x9a,0x1f,0x92,0x9a,0x21,0xa0,0xb6,0x85,+ 0x40,0xee,0xa2,0xda,0x72,0x5b,0x99,0xb3,+ 0x15,0xf3,0xb8,0xb4,0x89,0x91,0x8e,0xf1,+ 0x09,0xe1,0x56,0x19,0x39,0x51,0xec,0x7e,+ 0x93,0x7b,0x16,0x52,0xc0,0xbd,0x3b,0xb1,+ 0xbf,0x07,0x35,0x73,0xdf,0x88,0x3d,0x2c,+ 0x34,0xf1,0xef,0x45,0x1f,0xd4,0x6b,0x50,+ 0x3f,0x00+};++#endif
@@ -0,0 +1,555 @@+/*+ * Arithmetic in a binary field, and the scalar multiplication a curve over+ * one needs, doing the same work whatever the scalar is.+ *+ * A carry-less multiplication is the one thing a binary field needs and+ * ordinary arithmetic does not give. Where the processor has the instruction+ * for it this uses it -- PMULL on aarch64, PCLMULQDQ on x86-64 -- asking the+ * machine at run time where the compiler has not already been told. Where it+ * does not, each operand is split into four groups of every fourth bit, so+ * that the carries of an ordinary multiplication cannot reach the bits that+ * matter, and masked away afterwards. None of the three has a table or a+ * branch that depends on what it is multiplying.+ *+ * Reduction folds what is above the degree back in, which the polynomial+ * being a trinomial or a pentanomial with exponents that are public makes+ * cheap. Inversion is the exponentiation Fermat gives, whose exponent is+ * likewise public.+ *+ * The multiplication itself is Montgomery's ladder: it carries the x+ * coordinates of the multiples of two consecutive numbers, whose difference+ * is therefore the point, and spends one addition and one doubling on every+ * bit of the scalar whichever way the bit goes.+ */+#include <stdint.h>+#include <stdlib.h>+#include <string.h>+#include <crypton_cpu.h>+#include "crypton_armv8_target.h"+#include <crypton_bzero.h>+#include <crypton_f2m.h>++typedef uint64_t limb_t;+#define LIMB_BITS 64+#define LIMB_BYTES 8++/* the four groups, so that no carry of an ordinary multiplication reaches a+ * bit another partial product needs */+static void clmul32(uint32_t x, uint32_t y, limb_t *out)+{+ limb_t x0 = x & 0x11111111u, x1 = x & 0x22222222u;+ limb_t x2 = x & 0x44444444u, x3 = x & 0x88888888u;+ limb_t y0 = y & 0x11111111u, y1 = y & 0x22222222u;+ limb_t y2 = y & 0x44444444u, y3 = y & 0x88888888u;+ limb_t z0 = (x0 * y0) ^ (x1 * y3) ^ (x2 * y2) ^ (x3 * y1);+ limb_t z1 = (x0 * y1) ^ (x1 * y0) ^ (x2 * y3) ^ (x3 * y2);+ limb_t z2 = (x0 * y2) ^ (x1 * y1) ^ (x2 * y0) ^ (x3 * y3);+ limb_t z3 = (x0 * y3) ^ (x1 * y2) ^ (x2 * y1) ^ (x3 * y0);++ *out = (z0 & 0x1111111111111111ULL) | (z1 & 0x2222222222222222ULL)+ | (z2 & 0x4444444444444444ULL) | (z3 & 0x8888888888888888ULL);+}++static inline void clmul(limb_t a, limb_t b, limb_t *lo, limb_t *hi)+{+ limb_t ah = a >> 32, bh = b >> 32, t0, t1, t2;++ clmul32((uint32_t) a, (uint32_t) b, &t0);+ clmul32((uint32_t) ah, (uint32_t) bh, &t1);+ clmul32((uint32_t) (a ^ ah), (uint32_t) (b ^ bh), &t2);+ t2 ^= t0 ^ t1;+ *lo = t0 ^ (t2 << 32);+ *hi = t1 ^ (t2 >> 32);+}++/* t = a * b, over 2n limbs */+static void poly_mul_generic(limb_t *t, const limb_t *a, const limb_t *b,+ uint32_t n)+{+ uint32_t i, j;++ memset(t, 0, 2 * n * sizeof(limb_t));+ for (i = 0; i < n; i++)+ for (j = 0; j < n; j++) {+ limb_t lo, hi;++ clmul(a[i], b[j], &lo, &hi);+ t[i + j] ^= lo;+ t[i + j + 1] ^= hi;+ }+}++#if defined(__aarch64__) && (defined(__GNUC__) || defined(__clang__))+#define HAVE_PMULL 1+#include <arm_neon.h>++/* Where the compiler has been told the machine has the crypto extensions --+ * which it is on every Apple processor -- this needs no attribute and no+ * question. Where it has not, the attribute lets the instruction be emitted+ * in this one function, and the machine is asked before it is called. */+#if defined(__ARM_FEATURE_CRYPTO) || defined(__ARM_FEATURE_AES)+#define PMULL_ATTR+#define PMULL_ALWAYS 1+#else+#define PMULL_ATTR CRYPTON_TARGET_ARMV8_CRYPTO+#define PMULL_ALWAYS 0+#endif++#if !PMULL_ALWAYS+#if defined(__linux__) || defined(__ANDROID__)+#include <asm/hwcap.h>+#include <sys/auxv.h>+#elif defined(__FreeBSD__)+#include <machine/elf.h>+#include <sys/auxv.h>+#elif defined(__APPLE__)+#include <sys/sysctl.h>+#endif+#endif++static int have_pmull(void)+{+#if PMULL_ALWAYS+ return 1;+#elif (defined(__linux__) || defined(__ANDROID__)) && defined(HWCAP_PMULL)+ static int answer = -1;++ if (answer < 0)+ answer = (getauxval(AT_HWCAP) & HWCAP_PMULL) != 0;+ return answer;+#elif defined(__FreeBSD__) && defined(HWCAP_PMULL)+ static int answer = -1;++ if (answer < 0) {+ unsigned long hwcap = 0;++ elf_aux_info(AT_HWCAP, &hwcap, sizeof(hwcap));+ answer = (hwcap & HWCAP_PMULL) != 0;+ }+ return answer;+#elif defined(__APPLE__)+ static int answer = -1;++ if (answer < 0) {+ int has = 0;+ size_t len = sizeof(has);++ answer = sysctlbyname("hw.optional.arm.FEAT_PMULL", &has, &len,+ NULL, 0) == 0+ && has != 0;+ }+ return answer;+#else+ return 0; /* no way to ask, so the four groups it is */+#endif+}++PMULL_ATTR+static void poly_mul_pmull(limb_t *t, const limb_t *a, const limb_t *b,+ uint32_t n)+{+ uint32_t i, j;++ memset(t, 0, 2 * n * sizeof(limb_t));+ for (i = 0; i < n; i++)+ for (j = 0; j < n; j++) {+ uint64x2_t v = vreinterpretq_u64_p128(+ vmull_p64((poly64_t) a[i], (poly64_t) b[j]));++ t[i + j] ^= vgetq_lane_u64(v, 0);+ t[i + j + 1] ^= vgetq_lane_u64(v, 1);+ }+}+#else+#define HAVE_PMULL 0+#endif++#if defined(__x86_64__) && (defined(__GNUC__) || defined(__clang__))+#define HAVE_PCLMUL 1+#include <immintrin.h>++/* The same, with the instruction x86 has for it. The attribute is what lets+ * one file hold both this and the code for a processor without it: the+ * compiler may emit the instruction here and nowhere else, and the caller+ * asks the processor before it comes this way.+ */+__attribute__((target("pclmul,sse2")))+static void poly_mul_pclmul(limb_t *t, const limb_t *a, const limb_t *b,+ uint32_t n)+{+ uint32_t i, j;++ memset(t, 0, 2 * n * sizeof(limb_t));+ for (i = 0; i < n; i++)+ for (j = 0; j < n; j++) {+ __m128i p = _mm_clmulepi64_si128(+ _mm_cvtsi64_si128((long long) a[i]),+ _mm_cvtsi64_si128((long long) b[j]), 0x00);++ t[i + j] ^= (limb_t) _mm_cvtsi128_si64(p);+ t[i + j + 1] ^=+ (limb_t) _mm_cvtsi128_si64(_mm_srli_si128(p, 8));+ }+}+#else+#define HAVE_PCLMUL 0+#endif++static void poly_mul(limb_t *t, const limb_t *a, const limb_t *b, uint32_t n)+{+#if HAVE_PMULL+ /* what the processor has is not what is being multiplied, so asking is+ * not a side channel, and the answer is worked out once */+ if (have_pmull()) {+ poly_mul_pmull(t, a, b, n);+ return;+ }+#endif+#if HAVE_PCLMUL+ /* what the processor has is not what is being multiplied, so asking is+ * not a side channel, and the answer is worked out once */+ if (crypton_x86_simd_features() & CRYPTON_X86_PCLMUL) {+ poly_mul_pclmul(t, a, b, n);+ return;+ }+#endif+ poly_mul_generic(t, a, b, n);+}++/* the bits of a 32-bit half, spread out with a zero between each pair */+static limb_t spread(limb_t x)+{+ x = (x | (x << 16)) & 0x0000ffff0000ffffULL;+ x = (x | (x << 8)) & 0x00ff00ff00ff00ffULL;+ x = (x | (x << 4)) & 0x0f0f0f0f0f0f0f0fULL;+ x = (x | (x << 2)) & 0x3333333333333333ULL;+ x = (x | (x << 1)) & 0x5555555555555555ULL;+ return x;+}++/* t = a * a, which in a binary field is the bits of a spread out */+static void poly_sqr(limb_t *t, const limb_t *a, uint32_t n)+{+ uint32_t i;++ for (i = 0; i < n; i++) {+ t[2 * i] = spread(a[i] & 0xffffffffULL);+ t[2 * i + 1] = spread(a[i] >> 32);+ }+}++/* r = t mod fx, where fx is x^m plus the terms given, which are public+ *+ * Everything above bit m comes back in as those terms, a word at a time, and+ * then what is left above bit m within its own word is folded the same way.+ */+static void poly_reduce(limb_t *r, limb_t *t, uint32_t n, uint32_t m,+ const uint32_t *terms, uint32_t nterms)+{+ uint32_t mw = m / LIMB_BITS, mb = m % LIMB_BITS, i, j, pass;++ for (i = 2 * n; i > mw + 1; i--) {+ limb_t w = t[i - 1];++ t[i - 1] = 0;+ for (j = 0; j < nterms; j++) {+ uint32_t pos = (i - 1) * LIMB_BITS - m + terms[j];+ uint32_t pw = pos / LIMB_BITS, pb = pos % LIMB_BITS;++ t[pw] ^= w << pb;+ if (pb != 0)+ t[pw + 1] ^= w >> (LIMB_BITS - pb);+ }+ }++ /* what is left above bit m sits in the word that holds it; folding it+ * can put a little back, so it is done twice */+ for (pass = 0; pass < 2; pass++) {+ limb_t w;++ if (mb == 0)+ break;+ w = t[mw] >> mb;+ t[mw] &= ((limb_t) 1 << mb) - 1;+ for (j = 0; j < nterms; j++) {+ uint32_t pw = terms[j] / LIMB_BITS, pb = terms[j] % LIMB_BITS;++ t[pw] ^= w << pb;+ if (pb != 0 && pw + 1 <= mw)+ t[pw + 1] ^= w >> (LIMB_BITS - pb);+ }+ }+ memcpy(r, t, n * sizeof(limb_t));+}++/* the field: its polynomial, and scratch for a product */+typedef struct {+ uint32_t n;+ uint32_t m;+ uint32_t terms[8]; /* the polynomial without its leading term */+ uint32_t nterms;+ limb_t *t; /* 2n */+} bfield;++static void fe_mul(const bfield *f, limb_t *r, const limb_t *a, const limb_t *b)+{+ poly_mul(f->t, a, b, f->n);+ poly_reduce(r, f->t, f->n, f->m, f->terms, f->nterms);+}++static void fe_sqr(const bfield *f, limb_t *r, const limb_t *a)+{+ poly_sqr(f->t, a, f->n);+ poly_reduce(r, f->t, f->n, f->m, f->terms, f->nterms);+}++static void fe_add(const bfield *f, limb_t *r, const limb_t *a, const limb_t *b)+{+ uint32_t i;++ for (i = 0; i < f->n; i++)+ r[i] = a[i] ^ b[i];+}++static int fe_is_zero(const bfield *f, const limb_t *a)+{+ limb_t acc = 0;+ uint32_t i;++ for (i = 0; i < f->n; i++)+ acc |= a[i];+ return acc == 0;+}++/* r = 1/a, by Fermat: a to the power 2^m - 2, whose exponent is public */+static void fe_inv(const bfield *f, limb_t *r, const limb_t *a, limb_t *tmp)+{+ uint32_t i;++ memcpy(tmp, a, f->n * sizeof(limb_t));+ for (i = 1; i + 1 < f->m; i++) { /* a to the power 2^(m-1) - 1 */+ fe_sqr(f, tmp, tmp);+ fe_mul(f, tmp, tmp, a);+ }+ fe_sqr(f, r, tmp);+}++/* big-endian bytes into limbs, least significant limb first */+static int from_be(limb_t *r, uint32_t n, const uint8_t *src, uint32_t len)+{+ uint32_t i;++ memset(r, 0, n * sizeof(limb_t));+ for (i = 0; i < len; i++) {+ uint8_t byte = src[len - 1 - i];++ if (i / LIMB_BYTES >= n) {+ if (byte != 0)+ return 1;+ continue;+ }+ r[i / LIMB_BYTES] |= (limb_t) byte << (8 * (i % LIMB_BYTES));+ }+ return 0;+}++static void to_be(uint8_t *dst, uint32_t len, const limb_t *a, uint32_t n)+{+ uint32_t i;++ for (i = 0; i < len; i++) {+ uint32_t pos = len - 1 - i, li = i / LIMB_BYTES;++ dst[pos] = li < n ? (uint8_t) (a[li] >> (8 * (i % LIMB_BYTES))) : 0;+ }+}++/* exchange a and b when swap is one */+static void cswap(limb_t *a, limb_t *b, limb_t swap, uint32_t n)+{+ limb_t mask = (limb_t) 0 - swap;+ uint32_t i;++ for (i = 0; i < n; i++) {+ limb_t t = (a[i] ^ b[i]) & mask;++ a[i] ^= t;+ b[i] ^= t;+ }+}++int crypton_f2m_mul(uint8_t *outx, uint8_t *outy,+ const uint8_t *px, const uint8_t *py,+ const uint8_t *k, uint32_t klen,+ const uint8_t *b, uint32_t flen,+ const uint8_t *fx, uint32_t fxlen)+{+ uint32_t fn = (fxlen + LIMB_BYTES - 1) / LIMB_BYTES;+ uint32_t n, words, i;+ limb_t *space = NULL, *poly, *x, *y, *bb, *x1, *z1, *x2, *z2;+ limb_t *t1, *t2, *t3, *prod;+ bfield f;+ int ret = -1;++ if (flen == 0 || fxlen == 0 || klen == 0 || fn == 0)+ return -1;++ /* the polynomial, and the terms below its leading one */+ {+ limb_t *tmp = calloc(fn, sizeof(limb_t));+ uint32_t m = 0;++ if (tmp == NULL)+ return -1;+ if (from_be(tmp, fn, fx, fxlen) != 0) {+ free(tmp);+ return -1;+ }+ for (i = fn; i > 0 && m == 0; i--)+ if (tmp[i - 1] != 0) {+ limb_t top = tmp[i - 1];++ m = (i - 1) * LIMB_BITS;+ while (top != 0) {+ m++;+ top >>= 1;+ }+ m--; /* the degree is one under the bit count */+ }+ f.m = m;+ f.nterms = 0;+ for (i = 0; i < m; i++)+ if ((tmp[i / LIMB_BITS] >> (i % LIMB_BITS)) & 1) {+ if (f.nterms >= 8) {+ free(tmp);+ return -1; /* more terms than anything in use has */+ }+ f.terms[f.nterms++] = i;+ }+ free(tmp);+ if (m == 0 || f.nterms == 0)+ return -1;+ }++ n = (f.m + LIMB_BITS) / LIMB_BITS; /* room for the degree itself */+ f.n = n;+ words = 12 * n + 2 * n;+ space = calloc(words, sizeof(limb_t));+ if (space == NULL)+ return -1;+ poly = space; /* unused beyond keeping the layout plain */+ x = poly + n;+ y = x + n;+ bb = y + n;+ x1 = bb + n;+ z1 = x1 + n;+ x2 = z1 + n;+ z2 = x2 + n;+ t1 = z2 + n;+ t2 = t1 + n;+ t3 = t2 + n;+ prod = t3 + n; /* 2n, and one n before it is spare */+ f.t = prod;++ if (from_be(x, n, px, flen) != 0 || from_be(y, n, py, flen) != 0+ || from_be(bb, n, b, flen) != 0)+ goto done;+ if (fe_is_zero(&f, x))+ goto done; /* the point with no x is the caller's business */++ /* nothing, and the point next to it */+ memset(x1, 0, n * sizeof(limb_t));+ x1[0] = 1;+ memset(z1, 0, n * sizeof(limb_t));+ memcpy(x2, x, n * sizeof(limb_t));+ memset(z2, 0, n * sizeof(limb_t));+ z2[0] = 1;++ for (i = klen * 8; i > 0; i--) {+ uint32_t bit = i - 1;+ limb_t sel = (k[klen - 1 - bit / 8] >> (bit % 8)) & 1;++ /* whichever way the bit goes, one addition and one doubling: the+ * exchange before and after is what puts them where the bit asks */+ cswap(x1, x2, sel, n);+ cswap(z1, z2, sel, n);++ /* the two added, which their difference being the point allows */+ fe_mul(&f, t1, x1, z2);+ fe_mul(&f, t2, x2, z1);+ fe_add(&f, t3, t1, t2);+ fe_sqr(&f, t3, t3); /* the new z */+ fe_mul(&f, t1, t1, t2);+ fe_mul(&f, t2, x, t3);+ fe_add(&f, t2, t2, t1); /* the new x */++ /* and one of them doubled */+ fe_sqr(&f, x1, x1);+ fe_sqr(&f, z1, z1);+ fe_mul(&f, t1, x1, z1); /* z of the double */+ fe_sqr(&f, x1, x1);+ fe_sqr(&f, z1, z1);+ fe_mul(&f, z1, z1, bb);+ fe_add(&f, x1, x1, z1); /* x of the double */+ memcpy(z1, t1, n * sizeof(limb_t));++ memcpy(x2, t2, n * sizeof(limb_t));+ memcpy(z2, t3, n * sizeof(limb_t));++ cswap(x1, x2, sel, n);+ cswap(z1, z2, sel, n);+ }++ if (fe_is_zero(&f, z1)) {+ ret = 1; /* the multiple is at infinity */+ goto done;+ }+ if (fe_is_zero(&f, z2)) {+ /* the one after it is, so this one is the negation of the point */+ to_be(outx, flen, x, n);+ fe_add(&f, t1, x, y);+ to_be(outy, flen, t1, n);+ ret = 0;+ goto done;+ }++ /* x1/z1 and x2/z2, and the y the ladder does not carry, out of one+ * inversion: 1/(z1 z2 x) gives each of the three */+ fe_mul(&f, t1, z1, z2);+ fe_mul(&f, t1, t1, x);+ fe_inv(&f, t2, t1, t3);+ {+ limb_t *xa = x1, *xb = x2, *u = t1, *v = t3;++ fe_mul(&f, u, z2, x);+ fe_mul(&f, u, u, t2); /* 1/z1 */+ fe_mul(&f, xa, x1, u);+ fe_mul(&f, v, z1, x);+ fe_mul(&f, v, v, t2); /* 1/z2 */+ fe_mul(&f, xb, x2, v);+ fe_mul(&f, u, z1, z2);+ fe_mul(&f, u, u, t2); /* 1/x */++ fe_add(&f, v, xa, x); /* x1 + x */+ fe_add(&f, xb, xb, x); /* x2 + x */+ fe_mul(&f, xb, v, xb); /* (x1 + x)(x2 + x) */+ fe_sqr(&f, t2, x);+ fe_add(&f, xb, xb, t2);+ fe_add(&f, xb, xb, y); /* + x^2 + y */+ fe_mul(&f, xb, v, xb);+ fe_mul(&f, xb, xb, u); /* over x */+ fe_add(&f, xb, xb, y);+ to_be(outx, flen, xa, n);+ to_be(outy, flen, xb, n);+ }+ ret = 0;++done:+ /* freed on the next line, so a plain memset here is a store the+ * optimizer may drop */+ if (space != NULL) {+ crypton_bzero(space, words * sizeof(limb_t));+ free(space);+ }+ return ret;+}
@@ -0,0 +1,31 @@+#ifndef CRYPTON_F2M_H+#define CRYPTON_F2M_H++#include <stdint.h>++/* Multiply a point of a curve over a binary field by a scalar, doing the same+ * work whatever the scalar is.+ *+ * The curve is y^2 + x*y = x^3 + a*x^2 + b over the field of the polynomial+ * fx, and a does not come into it: the ladder carries the x coordinates of+ * two consecutive multiples, and what it takes to add them is b alone. The+ * point has to be on the curve and to have an x that is not zero -- the one+ * point with none is its own negation, and the caller sees to it.+ *+ * Every number is a big-endian byte string. The coordinates and b are flen+ * bytes, and fx is the whole polynomial, x^m included, in fxlen.+ *+ * The scalar is walked over every bit of the klen bytes it is given, so its+ * value is hidden but its length is not.+ *+ * Returns 0 with the answer in outx and outy, 1 if the answer is the point at+ * infinity, and -1 for arguments it cannot work with or memory it could not+ * have.+ */+int crypton_f2m_mul(uint8_t *outx, uint8_t *outy,+ const uint8_t *px, const uint8_t *py,+ const uint8_t *k, uint32_t klen,+ const uint8_t *b, uint32_t flen,+ const uint8_t *fx, uint32_t fxlen);++#endif
@@ -48,16 +48,17 @@ #define K3 0x6ED9EBA1 #define R(a,b,c,d,f,k,s,i) (a = rol32(a + f(b,c,d) + w[i] + k, s)) -static void md4_do_chunk(struct md4_ctx *ctx, uint32_t *buf)+/* The words are read out of the block rather than the block being pointed at+ * as though it were an array of them; see crypton_md5.c. */+static void md4_do_chunk(struct md4_ctx *ctx, const uint8_t *buf) { uint32_t a, b, c, d;-#ifdef ARCH_IS_BIG_ENDIAN uint32_t w[16];- cpu_to_le32_array(w, (uint32_t *) buf, 16);-#else- uint32_t *w = buf;-#endif+ int wi; + for (wi = 0; wi < 16; wi++)+ w[wi] = load_le32(buf + 4 * wi);+ a = ctx->h[0]; b = ctx->h[1]; c = ctx->h[2]; d = ctx->h[3]; R(a, b, c, d, f1, K1, 3, 0);@@ -125,24 +126,15 @@ if (index && len >= to_fill) { memcpy(ctx->buf + index, data, to_fill);- md4_do_chunk(ctx, (uint32_t *) ctx->buf);+ md4_do_chunk(ctx, ctx->buf); len -= to_fill; data += to_fill; index = 0; } - if (need_alignment(data, 4)) {- uint32_t tramp[16];- ASSERT_ALIGNMENT(tramp, 4);- for (; len >= 64; len -= 64, data += 64) {- memcpy(tramp, data, 64);- md4_do_chunk(ctx, tramp);- }- } else {- /* process as much 64-block as possible */- for (; len >= 64; len -= 64, data += 64)- md4_do_chunk(ctx, (uint32_t *) data);- }+ /* No trampoline: load_le32 does not ask for a boundary. */+ for (; len >= 64; len -= 64, data += 64)+ md4_do_chunk(ctx, data); /* append data into buf */ if (len)
@@ -45,15 +45,22 @@ #define f4(x, y, z) (y ^ (x | ~z)) #define R(f, a, b, c, d, i, k, s) a += f(b, c, d) + w[i] + k; a = rol32(a, s); a += b -static void md5_do_chunk(struct md5_ctx *ctx, uint32_t *buf)+/* The sixteen words are read out of the block rather than the block being+ * pointed at as though it were an array of them. A caller's pointer cast to+ * uint32_t * is a pointer the standard says may not exist unless the address+ * is aligned for it, and reading through it is undefined whether or not the+ * machine minds; UndefinedBehaviorSanitizer counted sixty-four of these.+ * load_le32 is a memcpy, which every compiler here turns into the one load+ * the cast used to be, and it takes the endianness with it -- so the two+ * arms this replaces are one. */+static void md5_do_chunk(struct md5_ctx *ctx, const uint8_t *buf) { uint32_t a, b, c, d;-#ifdef ARCH_IS_BIG_ENDIAN uint32_t w[16];- cpu_to_le32_array(w, buf, 16);-#else- uint32_t *w = buf;-#endif+ int wi;++ for (wi = 0; wi < 16; wi++)+ w[wi] = load_le32(buf + 4 * wi); a = ctx->h[0]; b = ctx->h[1]; c = ctx->h[2]; d = ctx->h[3]; R(f1, a, b, c, d, 0, 0xd76aa478, 7);@@ -138,24 +145,16 @@ if (index && len >= to_fill) { memcpy(ctx->buf + index, data, to_fill);- md5_do_chunk(ctx, (uint32_t *) ctx->buf);+ md5_do_chunk(ctx, ctx->buf); len -= to_fill; data += to_fill; index = 0; } - if (need_alignment(data, 4)) {- uint32_t tramp[16];- ASSERT_ALIGNMENT(tramp, 4);- for (; len >= 64; len -= 64, data += 64) {- memcpy(tramp, data, 64);- md5_do_chunk(ctx, tramp);- }- } else {- /* process as much 64-block as possible */- for (; len >= 64; len -= 64, data += 64)- md5_do_chunk(ctx, (uint32_t *) data);- }+ /* No trampoline for a block that is not on a four-byte boundary: the+ * words are read with load_le32 now, which does not ask. */+ for (; len >= 64; len -= 64, data += 64)+ md5_do_chunk(ctx, data); /* append data into buf */ if (len)
@@ -0,0 +1,29 @@+/*+ * dst = a xor b.+ *+ * Data.ByteArray's xor walks a byte at a time through an IO applicative, and+ * that allocates: fifty bytes of heap for every byte exclusive-ored, which in+ * counter mode cost more than the cipher did. This is the same operation in+ * one pass of words.+ */++#include <stdint.h>+#include <string.h>++#include "crypton_memxor.h"++void crypton_memxor(uint8_t *dst, const uint8_t *a, const uint8_t *b, uint32_t len)+{+ uint32_t i = 0;++ for (; i + 8 <= len; i += 8) {+ uint64_t x, y;++ memcpy(&x, a + i, 8);+ memcpy(&y, b + i, 8);+ x ^= y;+ memcpy(dst + i, &x, 8);+ }+ for (; i < len; i++)+ dst[i] = a[i] ^ b[i];+}
@@ -0,0 +1,8 @@+#ifndef CRYPTON_MEMXOR_H+#define CRYPTON_MEMXOR_H++#include <stdint.h>++void crypton_memxor(uint8_t *dst, const uint8_t *a, const uint8_t *b, uint32_t len);++#endif
@@ -0,0 +1,74 @@+/*+ * Inversion modulo an odd number through s2n-bignum, whose routine takes a+ * fixed number of division steps rather than an exponentiation: twenty to+ * thirty times less work than Fermat's little theorem at the sizes here.+ * Measured on an Apple M4, inverting modulo a curve order:+ *+ * Fermat this+ * P-256 6.02 us 0.80+ * P-384 31.3 1.20+ * P-521 63.2 2.05+ *+ * It uses no instruction beyond the base architecture on either x86-64 or+ * AArch64, so unlike the rest of the vendored assembly there is nothing to+ * ask the processor first.+ */+#include <string.h>++#include "crypton_modinv.h"++#ifdef CRYPTON_S2N_BIGNUM++extern void bignum_modinv(uint64_t k, uint64_t *z, const uint64_t *a,+ const uint64_t *b, uint64_t *t);++/* 4096 bits and no more, which covers every modulus that reaches here -- the+ * order of a curve, or a prime factor of an RSA modulus -- and keeps the+ * working space on the stack. Anything larger is handed back. */+#define MODINV_MAXWORDS 64++int crypton_modinv_sec(uint8_t *z, const uint8_t *a, const uint8_t *m,+ uint32_t len)+{+ uint64_t aw[MODINV_MAXWORDS], mw[MODINV_MAXWORDS];+ uint64_t zw[MODINV_MAXWORDS], t[3 * MODINV_MAXWORDS];+ uint32_t k = (len + 7) / 8;+ uint32_t i;++ /* An even modulus is the one case it answers without saying it+ * cannot: it returns a number that is not an inverse rather than+ * failing, so keep it away from here. Every caller's modulus is odd. */+ if (len == 0 || k > MODINV_MAXWORDS || (m[len - 1] & 1) == 0)+ return 1;++ for (i = 0; i < k; i++) {+ aw[i] = 0;+ mw[i] = 0;+ }+ for (i = 0; i < len; i++) {+ uint32_t pos = len - 1 - i;++ aw[pos / 8] |= (uint64_t)a[i] << (8 * (pos % 8));+ mw[pos / 8] |= (uint64_t)m[i] << (8 * (pos % 8));+ }++ bignum_modinv(k, zw, aw, mw, t);++ for (i = 0; i < len; i++)+ z[len - 1 - i] = (uint8_t)(zw[i / 8] >> (8 * (i % 8)));+ return 0;+}++#else++int crypton_modinv_sec(uint8_t *z, const uint8_t *a, const uint8_t *m,+ uint32_t len)+{+ (void)z;+ (void)a;+ (void)m;+ (void)len;+ return 1;+}++#endif
@@ -0,0 +1,22 @@+#ifndef CRYPTON_MODINV_H+#define CRYPTON_MODINV_H++#include <stdint.h>++/* z = a^-1 mod m, all three big-endian byte strings of len bytes.+ *+ * Returns 0 with the answer in z, and 1 without touching z when it will not+ * do this one: the assembly it needs is not built, the modulus is even, or+ * the numbers are larger than it keeps room for. A 1 is not an error, it is+ * "ask something else".+ *+ * The answer is not checked here. When a has no inverse the routine+ * underneath returns something that is not one rather than saying so, so the+ * caller has to multiply out and look -- which is what Crypto.Number.+ * ModArithmetic.inverseSafe already did for the exponentiation this+ * replaces.+ */+int crypton_modinv_sec(uint8_t *z, const uint8_t *a, const uint8_t *m,+ uint32_t len);++#endif
@@ -46,6 +46,7 @@ /* Internal function/type names for hash-specific things. */ #define HMAC_CTX(_name) HMAC_ ## _name ## _ctx+#define DIGEST_FITS(_name) pbkdf2_ ## _name ## _digest_fits_in_a_block #define HMAC_INIT(_name) HMAC_ ## _name ## _init #define HMAC_UPDATE(_name) HMAC_ ## _name ## _update #define HMAC_FINAL(_name) HMAC_ ## _name ## _final@@ -79,6 +80,16 @@ */ #define DECL_PBKDF2(_name, _blocksz, _hashsz, _ctx, \ _init, _update, _xform, _final, _xcpy, _xtract, _xxor) \+ /* HMAC_INIT below shortens a key longer than the block by hashing it, \+ * which writes _hashsz bytes into a buffer of _blocksz. An instantiation \+ * whose digest is larger than its block would overflow that buffer, and \+ * would do it before any check inside the function could say so -- which \+ * is where the check used to be. Refuse such an instantiation here \+ * instead, in front of the person writing it. The three below are \+ * SHA-1, SHA-256 and SHA-512, whose digests are 20, 32 and 64 bytes \+ * against blocks of 64, 64 and 128. */ \+ typedef char DIGEST_FITS(_name)[(_hashsz) <= (_blocksz) ? 1 : -1]; \+ \ typedef struct { \ _ctx inner; \ _ctx outer; \@@ -101,9 +112,6 @@ nkey = _hashsz; \ } \ \- /* Standard doesn't cover case where blocksz < hashsz. */ \- assert(nkey <= _blocksz); \- \ /* Right zero-pad short keys. */ \ if (k != key) \ memcpy(k, key, nkey); \@@ -192,7 +200,16 @@ uint8_t *out, size_t nout) \ { \ assert(iterations); \- assert(out && nout); \+ assert(out); \+ \+ /* Zero bytes of derived key is zero bytes of work. RFC 8018 asks for a \+ * positive dkLen and the loop below would write a block regardless, so \+ * this used to be `assert(out && nout)` -- which aborts the process, in \+ * a library built without NDEBUG, on a length the caller chose. \+ * Crypto.KDF.PBKDF2's own tryGenerate returns an empty result for this, \+ * so return and let the two agree. */ \+ if (nout == 0) \+ return; \ \ /* Starting point for inner loop. */ \ HMAC_CTX(_name) ctx; \
@@ -39,8 +39,52 @@ #include "crypton_bitfn.h" #include "crypton_align.h" ++/*+ * Poly1305 from CRYPTOGAMS, in cbits/asm/poly1305-armv8-*.S and+ * cbits/asm/poly1305-x86_64-*.S, which is the whole of the arithmetic+ * rather than a bulk loop bolted to the side: it keeps its own accumulator+ * -- in base 2^64 while the message is short and base 2^26 once the vector+ * loop has started, switching between the two itself -- and its own powers+ * of r, so what is left here is the buffering of partial blocks.+ *+ * 'padbit' is the high bit above each block, which is set for every block+ * of the message and clear for the padded last one.+ */+#if (defined(WITH_ARMV8_POLY1305_ASM) && !defined(__AARCH64EB__)) \+ || defined(WITH_X86_POLY1305_ASM)+#define POLY1305_ASM 1+#include "crypton_cpu.h"++typedef void (*poly1305_blocks_f)(void *ctx, const uint8_t *inp, size_t len,+ uint32_t padbit);+typedef void (*poly1305_emit_f)(void *ctx, uint8_t mac[16],+ const uint32_t nonce[4]);++int crypton_poly1305_asm_init(void *ctx, const uint8_t key[16], void *func[2]);++/*+ * Initialisation hands back the pair of functions its own dispatch would+ * use, the vector entry points themselves being local to the module. They+ * are the same for every context, so they are kept here rather than in each+ * one; two threads racing to fill them write the same values.+ */+static poly1305_blocks_f asm_blocks;+static poly1305_emit_f asm_emit;+#endif+++#ifdef POLY1305_ASM+ static void poly1305_do_chunk(poly1305_ctx *ctx, uint8_t *data, int blocks, int final) {+ asm_blocks(ctx->st.opaque, data, (size_t) blocks * 16, final ? 0 : 1);+}++#else++static void poly1305_do_chunk(poly1305_ctx *ctx, uint8_t *data, int blocks, int final)+{ /* following is a cleanup copy of code available poly1305-donna */ const uint32_t hibit = (final) ? 0 : (1 << 24); /* 1 << 128 */ uint32_t r0,r1,r2,r3,r4;@@ -49,9 +93,10 @@ uint64_t d0,d1,d2,d3,d4; uint32_t c; + /* load r[i], h[i] */- h0 = ctx->h[0]; h1 = ctx->h[1]; h2 = ctx->h[2]; h3 = ctx->h[3]; h4 = ctx->h[4];- r0 = ctx->r[0]; r1 = ctx->r[1]; r2 = ctx->r[2]; r3 = ctx->r[3]; r4 = ctx->r[4];+ h0 = ctx->st.limb.h[0]; h1 = ctx->st.limb.h[1]; h2 = ctx->st.limb.h[2]; h3 = ctx->st.limb.h[3]; h4 = ctx->st.limb.h[4];+ r0 = ctx->st.limb.r[0]; r1 = ctx->st.limb.r[1]; r2 = ctx->st.limb.r[2]; r3 = ctx->st.limb.r[3]; r4 = ctx->st.limb.r[4]; /* s[i] = r[i] * 5 */ s1 = r1 * 5; s2 = r2 * 5; s3 = r3 * 5; s4 = r4 * 5;@@ -81,21 +126,37 @@ } /* store h[i] */- ctx->h[0] = h0; ctx->h[1] = h1; ctx->h[2] = h2; ctx->h[3] = h3; ctx->h[4] = h4;+ ctx->st.limb.h[0] = h0; ctx->st.limb.h[1] = h1; ctx->st.limb.h[2] = h2; ctx->st.limb.h[3] = h3; ctx->st.limb.h[4] = h4; } +#endif+ void crypton_poly1305_init(poly1305_ctx *ctx, poly1305_key *key) { uint8_t *k = (uint8_t *) key; memset(ctx, 0, sizeof(poly1305_ctx)); - ctx->r[0] = (load_le32(&k[ 0]) ) & 0x3ffffff;- ctx->r[1] = (load_le32(&k[ 3]) >> 2) & 0x3ffff03;- ctx->r[2] = (load_le32(&k[ 6]) >> 4) & 0x3ffc0ff;- ctx->r[3] = (load_le32(&k[ 9]) >> 6) & 0x3f03fff;- ctx->r[4] = (load_le32(&k[12]) >> 8) & 0x00fffff;+#ifdef POLY1305_ASM+ {+ void *func[2]; +#ifdef CRYPTON_X86_ASM+ /* what the module dispatches on, which it reads directly */+ crypton_x86_ia32cap_resolve();+#endif+ crypton_poly1305_asm_init(ctx->st.opaque, k, func);+ asm_blocks = (poly1305_blocks_f) func[0];+ asm_emit = (poly1305_emit_f) func[1];+ }+#else+ ctx->st.limb.r[0] = (load_le32(&k[ 0]) ) & 0x3ffffff;+ ctx->st.limb.r[1] = (load_le32(&k[ 3]) >> 2) & 0x3ffff03;+ ctx->st.limb.r[2] = (load_le32(&k[ 6]) >> 4) & 0x3ffc0ff;+ ctx->st.limb.r[3] = (load_le32(&k[ 9]) >> 6) & 0x3f03fff;+ ctx->st.limb.r[4] = (load_le32(&k[12]) >> 8) & 0x00fffff;+#endif+ ctx->pad[0] = load_le32(&k[16]); ctx->pad[1] = load_le32(&k[20]); ctx->pad[2] = load_le32(&k[24]);@@ -134,11 +195,6 @@ void crypton_poly1305_finalize(poly1305_mac mac8, poly1305_ctx *ctx) {- uint32_t h0,h1,h2,h3,h4,c;- uint32_t g0,g1,g2,g3,g4;- uint64_t f;- uint32_t mask;- uint32_t *mac = (uint32_t *) mac8; int i; if (ctx->index) {@@ -149,10 +205,22 @@ poly1305_do_chunk(ctx, ctx->buf, 1, 1); } +#ifdef POLY1305_ASM+ /* the carry, the reduction and the addition of the second half of+ * the key are the assembly's, since the accumulator is its own */+ asm_emit(ctx->st.opaque, mac8, ctx->pad);+#else+ {+ uint32_t h0,h1,h2,h3,h4,c;+ uint32_t g0,g1,g2,g3,g4;+ uint64_t f;+ uint32_t mask;+ uint32_t *mac = (uint32_t *) mac8;+ /* following is a cleanup copy of code available poly1305-donna */ /* fully carry h */- h0 = ctx->h[0]; h1 = ctx->h[1]; h2 = ctx->h[2]; h3 = ctx->h[3]; h4 = ctx->h[4];+ h0 = ctx->st.limb.h[0]; h1 = ctx->st.limb.h[1]; h2 = ctx->st.limb.h[2]; h3 = ctx->st.limb.h[3]; h4 = ctx->st.limb.h[4]; c = h1 >> 26; h1 = h1 & 0x3ffffff; h2 += c; c = h2 >> 26; h2 = h2 & 0x3ffffff;@@ -200,4 +268,6 @@ f = (uint64_t)h3 + ctx->pad[3] + (f >> 32); mac[3] = cpu_to_le32((uint32_t) f);+ }+#endif }
@@ -30,11 +30,24 @@ #ifndef CRYPTON_POLY1305_H # define CRYPTON_POLY1305_H -/* 8*8+1*16+1*4 = 84 */+/*+ * Either the 26-bit limbs the C implementation works in, or the state the+ * assembly keeps: its accumulator, in whichever base it is using at the+ * time, the clamped key, and the powers of that laid out for the four-way+ * vector loop, which together come to exactly 192 bytes -- OpenSSL allots+ * the same for the same thing.+ *+ * size = 192+16+4+16 = 228, 232 with the alignment the union asks for+ */ typedef struct {- uint32_t r[5];- uint32_t h[5];+ union {+ struct {+ uint32_t r[5];+ uint32_t h[5];+ } limb;+ uint64_t opaque[24];+ } st; uint32_t pad[4]; uint32_t index; uint8_t buf[16]; /* previous partial block */
@@ -0,0 +1,311 @@+/*+ * Modular exponentiation that does the same work whatever the exponent is.+ *+ * The exponent is walked four bits at a time: four squarings and one+ * multiplication by a small power of the base, taken from a table of sixteen.+ * The table is read by touching all sixteen entries and keeping one of them+ * with a mask, so the address stream does not follow the exponent, and the+ * multiplication itself is Montgomery's, whose only conditional step -- the+ * subtraction at the end -- is also done with a mask.+ *+ * So every window costs the same four squarings, the same multiplication and+ * the same sixteen reads, and nothing here branches on, or indexes memory+ * with, anything derived from the exponent.+ *+ * What is still visible is how many bytes the caller passed: the loop runs+ * over every bit of them, so the exponent's value is hidden but its length is+ * not. GMP's mpz_powm_sec, which this replaces on the GHCs that no longer+ * offer it, hides the same amount.+ */+#include <stdlib.h>+#include <crypton_bignum.h>+#include <crypton_powm.h>+#include <crypton_bzero.h>++/*+ * At RSA sizes on x86-64, the Montgomery multiplication below is the whole+ * cost, and s2n-bignum's is twice as fast because the C cannot form the two+ * carry chains ADCX and ADOX give. The window, the table and its masked+ * scan are unchanged: only the multiply and the square are swapped, and+ * only for the sizes s2n-bignum has a Karatsuba multiplication for.+ *+ * Not on AArch64, where the C measures 5% faster than the assembly.+ * See cbits/s2n/README.md.+ */+#if defined(CRYPTON_S2N_BIGNUM) && defined(__x86_64__)+#define CRYPTON_POWM_S2N 1+#include <crypton_cpu.h>++extern void bignum_kmul_16_32(uint64_t *z, const uint64_t *x,+ const uint64_t *y, uint64_t *t);+extern void bignum_ksqr_16_32(uint64_t *z, const uint64_t *x, uint64_t *t);+extern void bignum_kmul_32_64(uint64_t *z, const uint64_t *x,+ const uint64_t *y, uint64_t *t);+extern void bignum_ksqr_32_64(uint64_t *z, const uint64_t *x, uint64_t *t);+extern uint64_t bignum_emontredc_8n(uint64_t k, uint64_t *z,+ const uint64_t *m, uint64_t w);++/* 16 limbs is 1024 bits and 32 is 2048: the halves a CRT exponentiation+ * works in for RSA-2048 and RSA-4096, and the whole thing without CRT. The+ * reduction wants ADX, so the answer is a run-time one. */+static int powm_s2n_usable(uint32_t n)+{+ return (n == 16 || n == 32)+ && (crypton_x86_simd_features() & CRYPTON_X86_ADX) != 0;+}++/* The scratch the widest of them asks for, in multiples of n: kmul_32_64+ * wants 96 limbs for n = 32. */+#define POWM_S2N_SCRATCH 3++/* bignum_emontredc_8n leaves the result in the top half of z with one more+ * bit as its return value, and what is there is under twice the modulus --+ * the same place mont_reduce ends up, and finished the same way. */+static void powm_s2n_finish(limb_t *r, limb_t *z, const limb_t *m,+ limb_t carry, uint32_t n)+{+ limb_t borrow = sub_n(r, z + n, m, n);+ limb_t take = carry | (borrow ^ 1);++ select_n(r, r, z + n, take & 1, n);+}++static void powm_s2n_mul(limb_t *r, const limb_t *a, const limb_t *b,+ const limb_t *m, limb_t n0, uint32_t n, limb_t *z,+ limb_t *scratch)+{+ if (n == 16)+ bignum_kmul_16_32(z, a, b, scratch);+ else+ bignum_kmul_32_64(z, a, b, scratch);+ powm_s2n_finish(r, z, m, bignum_emontredc_8n(n, z, m, n0), n);+}++static void powm_s2n_sqr(limb_t *r, const limb_t *a, const limb_t *m,+ limb_t n0, uint32_t n, limb_t *z, limb_t *scratch)+{+ if (n == 16)+ bignum_ksqr_16_32(z, a, scratch);+ else+ bignum_ksqr_32_64(z, a, scratch);+ powm_s2n_finish(r, z, m, bignum_emontredc_8n(n, z, m, n0), n);+}+#else+#define POWM_S2N_SCRATCH 0+#endif++/* One or the other, decided once per call */+static void powm_mul(limb_t *r, const limb_t *a, const limb_t *b,+ const limb_t *m, limb_t n0, uint32_t n, limb_t *t,+ limb_t *scratch, int s2n)+{+#ifdef CRYPTON_POWM_S2N+ if (s2n) {+ powm_s2n_mul(r, a, b, m, n0, n, t, scratch);+ return;+ }+#else+ (void)scratch;+ (void)s2n;+#endif+ mont_mul(r, a, b, m, n0, n, t);+}++static void powm_sqr(limb_t *r, const limb_t *a, const limb_t *m, limb_t n0,+ uint32_t n, limb_t *t, limb_t *scratch, int s2n)+{+#ifdef CRYPTON_POWM_S2N+ if (s2n) {+ powm_s2n_sqr(r, a, m, n0, n, t, scratch);+ return;+ }+#else+ (void)scratch;+ (void)s2n;+#endif+ mont_sqr(r, a, m, n0, n, t);+}++/* Four bits of exponent per window, so a table of sixteen and no leftover+ * bits: a byte holds exactly two windows.+ *+ * Five was written and measured, and is not here. A wider window saves+ * multiplications -- 205 of them against 256 at 1024 bits, with the same+ * 1024 squarings -- and pays for it in the masked scan of a table twice as+ * long, and which way that comes out depends on the machine and on which+ * multiplication is running: 3.5% better on an Apple M4, about 1% worse on+ * an older x86-64, and 7% worse anywhere s2n-bignum's multiplication is+ * used, since that makes the scan the expensive half. Six measured level+ * with five on the M4 and seven worse. What would make a wider window pay+ * everywhere is a cheaper scan, not a wider window. */+#define WINDOW_BITS 4+#define TABLE_SIZE (1 << WINDOW_BITS)++/* The masked scan of the table: every entry is read and a mask keeps the one+ * wanted, so that the address stream does not follow the exponent. At+ * RSA-2048's CRT size that is two kilobytes read per window, and the window+ * loop runs 256 times per exponentiation, which is why it is worth a vector+ * register: removing the scan altogether measures 11% of an exponentiation+ * where s2n-bignum's multiplication runs, and the AVX2 form below gets+ * essentially all of it. */+static void scan_table(limb_t *sel, const limb_t *table, uint32_t n, limb_t w)+{+ uint32_t k, l;++ memset(sel, 0, n * sizeof(limb_t));+ for (k = 0; k < TABLE_SIZE; k++) {+ limb_t mask = eq_mask(k, w);++ for (l = 0; l < n; l++)+ sel[l] |= table[k * n + l] & mask;+ }+}++#if defined(__x86_64__) && defined(WITH_TARGET_ATTRIBUTES) && LIMB_BITS == 64+#define CRYPTON_POWM_SCAN_AVX2 1+#include <crypton_cpu.h>+#include <immintrin.h>++/* The same scan four limbs at a time. The sixteen masks are worked out+ * once; after that each register of the answer is one pass over the table's+ * column, reading every entry exactly as the scalar form does. */+__attribute__((target("avx2")))+static void scan_table_avx2(limb_t *sel, const limb_t *table, uint32_t n,+ limb_t w)+{+ __m256i masks[TABLE_SIZE];+ uint32_t k, l;++ for (k = 0; k < TABLE_SIZE; k++)+ masks[k] = _mm256_cmpeq_epi64(+ _mm256_set1_epi64x((long long) k),+ _mm256_set1_epi64x((long long) w));++ for (l = 0; l + 4 <= n; l += 4) {+ __m256i acc = _mm256_setzero_si256();++ for (k = 0; k < TABLE_SIZE; k++) {+ __m256i v = _mm256_loadu_si256(+ (const __m256i *) (table + k * n + l));++ acc = _mm256_or_si256(acc,+ _mm256_and_si256(v, masks[k]));+ }+ _mm256_storeu_si256((__m256i *) (sel + l), acc);+ }++ /* a modulus whose limbs do not come in fours ends here */+ for (; l < n; l++) {+ limb_t v = 0;++ for (k = 0; k < TABLE_SIZE; k++)+ v |= table[k * n + l] & eq_mask(k, w);+ sel[l] = v;+ }+}+#endif++int crypton_powm_sec(uint8_t *out,+ const uint8_t *base, uint32_t baselen,+ const uint8_t *exp, uint32_t explen,+ const uint8_t *mod, uint32_t modlen)+{+ uint32_t n = (modlen + LIMB_BYTES - 1) / LIMB_BYTES;+ uint32_t words = (TABLE_SIZE + 7 + POWM_S2N_SCRATCH) * n;+ limb_t *space, *m, *r2, *acc, *sel, *prod, *table, *t, *scratch, n0;+ uint32_t i, j, k;+ int s2n = 0;+#ifdef CRYPTON_POWM_SCAN_AVX2+ int avx2 = (crypton_x86_simd_features() & CRYPTON_X86_AVX2) != 0;+#endif++ if (modlen == 0 || n == 0 || (mod[modlen - 1] & 1) == 0)+ return 1;++ /* the table, five more n-limb numbers and one of 2n */+ space = calloc(words, sizeof(limb_t));+ if (space == NULL)+ return 1;+ m = space;+ r2 = m + n;+ acc = r2 + n;+ sel = acc + n;+ prod = sel + n;+ t = prod + n;+ table = t + 2 * n;+ scratch = table + TABLE_SIZE * n;++#ifdef CRYPTON_POWM_S2N+ s2n = powm_s2n_usable(n);+#endif++ if (from_be(m, n, mod, modlen) != 0)+ goto fail;+ n0 = mont_n0(m[0]);+ mont_r2(r2, m, n0, n, t);++ /* table[k] = base^k in Montgomery form, and table[0] = 1 there */+ memset(table, 0, n * sizeof(limb_t));+ table[0] = 1;+ powm_mul(acc, table, r2, m, n0, n, t, scratch, s2n);+ memcpy(table, acc, n * sizeof(limb_t));++ if (from_be(sel, n, base, baselen) != 0)+ goto fail;+ powm_mul(table + n, sel, r2, m, n0, n, t, scratch, s2n);+ for (k = 2; k < TABLE_SIZE; k++)+ powm_mul(table + k * n, table + (k - 1) * n, table + n, m, n0, n,+ t, scratch, s2n);++ memcpy(acc, table, n * sizeof(limb_t));++ for (i = explen * 2; i > 0; i--) {+ uint32_t nib = i - 1;+ limb_t w = (exp[explen - 1 - nib / 2] >> (4 * (nib % 2))) & 0xf;++ /* squaring into the other buffer and swapping the two saves a+ * copy of the modulus' width every time; which of the three+ * buffers a pointer names is nobody's secret */+ for (j = 0; j < WINDOW_BITS; j++) {+ limb_t *swap;++ powm_sqr(sel, acc, m, n0, n, t, scratch, s2n);+ swap = acc;+ acc = sel;+ sel = swap;+ }++#ifdef CRYPTON_POWM_SCAN_AVX2+ if (avx2)+ scan_table_avx2(sel, table, n, w);+ else+#endif+ scan_table(sel, table, n, w);+ powm_mul(prod, acc, sel, m, n0, n, t, scratch, s2n);+ {+ limb_t *swap = acc;++ acc = prod;+ prod = swap;+ }+ }++ /* out of Montgomery form */+ memset(sel, 0, n * sizeof(limb_t));+ sel[0] = 1;+ powm_mul(prod, acc, sel, m, n0, n, t, scratch, s2n);+ to_be(out, modlen, prod, n);++ /* nothing here is the caller's secret, but the exponent's bits passed+ * through the accumulators. crypton_bzero rather than memset, since+ * this memory is freed on the next line and a store to memory about to+ * die is one an optimizer may drop. */+ crypton_bzero(space, words * sizeof(limb_t));+ free(space);+ return 0;++fail:+ crypton_bzero(space, words * sizeof(limb_t));+ free(space);+ return 1;+}
@@ -0,0 +1,23 @@+#ifndef CRYPTON_POWM_H+#define CRYPTON_POWM_H++#include <stdint.h>++/* Modular exponentiation whose work does not depend on the exponent's bits.+ *+ * All three numbers are big-endian byte strings. The modulus has to be odd+ * and at least one byte, and the base has to be smaller than it: the caller+ * reduces, which it can do in whatever way it likes, because in this library+ * the base is always a public value.+ *+ * The result is written to out, which holds modlen bytes.+ *+ * Returns 0 on success, and nonzero if the modulus is even or memory ran out,+ * in which case out is untouched.+ */+int crypton_powm_sec(uint8_t *out,+ const uint8_t *base, uint32_t baselen,+ const uint8_t *exp, uint32_t explen,+ const uint8_t *mod, uint32_t modlen);++#endif
@@ -57,16 +57,17 @@ #define R(a, b, c, d, e, f, k, i, s) \ a += f(b, c, d) + w[i] + k; a = rol32(a, s) + e; c = rol32(c, 10) -static void ripemd160_do_chunk(struct ripemd160_ctx *ctx, uint32_t *buf)+/* The words are read out of the block rather than the block being pointed at+ * as though it were an array of them; see crypton_md5.c. */+static void ripemd160_do_chunk(struct ripemd160_ctx *ctx, const uint8_t *buf) { uint32_t a1, b1, c1, d1, e1, a2, b2, c2, d2, e2;-#ifdef ARCH_IS_BIG_ENDIAN uint32_t w[16];- cpu_to_le32_array(w, buf, 16);-#else- uint32_t *w = buf;-#endif+ int wi; + for (wi = 0; wi < 16; wi++)+ w[wi] = load_le32(buf + 4 * wi);+ a1 = ctx->h[0]; b1 = ctx->h[1]; c1 = ctx->h[2]; d1 = ctx->h[3]; e1 = ctx->h[4]; a2 = ctx->h[0]; b2 = ctx->h[1]; c2 = ctx->h[2]; d2 = ctx->h[3]; e2 = ctx->h[4]; @@ -260,24 +261,15 @@ ctx->sz += len; if (index && len >= to_fill) { memcpy(ctx->buf + index, data, to_fill);- ripemd160_do_chunk(ctx, (uint32_t *) ctx->buf);+ ripemd160_do_chunk(ctx, ctx->buf); len -= to_fill; data += to_fill; index = 0; } - if (need_alignment(data, 4)) {- uint32_t tramp[16];- ASSERT_ALIGNMENT(tramp, 4);- for (; len >= 64; len -= 64, data += 64) {- memcpy(tramp, data, 64);- ripemd160_do_chunk(ctx, tramp);- }- } else {- /* process as much 64-block as possible */- for (; len >= 64; len -= 64, data += 64)- ripemd160_do_chunk(ctx, (uint32_t *) data);- }+ /* No trampoline: load_le32 does not ask for a boundary. */+ for (; len >= 64; len -= 64, data += 64)+ ripemd160_do_chunk(ctx, data); /* append data into buf */ if (len)
@@ -59,7 +59,7 @@ QR (x15,x12,x13,x14); \ } -static void salsa_core(int rounds, block *out, const crypton_salsa_state *in)+static void salsa_core(int rounds, crypton_salsa_block *out, const crypton_salsa_state *in) { uint32_t x0, x1, x2, x3, x4, x5, x6, x7, x8, x9, x10, x11, x12, x13, x14, x15; int i;@@ -94,7 +94,7 @@ out->d[15] = cpu_to_le32(x15); } -void crypton_salsa_core_xor(int rounds, block *out, block *in)+void crypton_salsa_core_xor(int rounds, crypton_salsa_block *out, crypton_salsa_block *in) { uint32_t x0, x1, x2, x3, x4, x5, x6, x7, x8, x9, x10, x11, x12, x13, x14, x15; int i;@@ -165,7 +165,7 @@ void crypton_salsa_combine(uint8_t *dst, crypton_salsa_context *ctx, const uint8_t *src, uint32_t bytes) {- block out;+ crypton_salsa_block out; crypton_salsa_state *st; int i; @@ -225,7 +225,7 @@ void crypton_salsa_generate(uint8_t *dst, crypton_salsa_context *ctx, uint32_t bytes) { crypton_salsa_state *st;- block out;+ crypton_salsa_block out; int i; if (!bytes)@@ -252,7 +252,7 @@ /* xor new 64-bytes chunks and store the left over if any */ for (; bytes >= 64; bytes -= 64, dst += 64) { /* generate new chunk and update state */- salsa_core(ctx->nb_rounds, (block *) dst, st);+ salsa_core(ctx->nb_rounds, (crypton_salsa_block *) dst, st); st->d[8] += 1; if (st->d[8] == 0) st->d[9] += 1;
@@ -34,9 +34,9 @@ uint64_t q[8]; uint32_t d[16]; uint8_t b[64];-} block;+} crypton_salsa_block; -typedef block crypton_salsa_state;+typedef crypton_salsa_block crypton_salsa_state; typedef struct { crypton_salsa_state st;@@ -47,7 +47,7 @@ } crypton_salsa_context; /* for scrypt */-void crypton_salsa_core_xor(int rounds, block *out, block *in);+void crypton_salsa_core_xor(int rounds, crypton_salsa_block *out, crypton_salsa_block *in); void crypton_salsa_init_core(crypton_salsa_state *st, uint32_t keylen, const uint8_t *key, uint32_t ivlen, const uint8_t *iv); void crypton_salsa_init(crypton_salsa_context *ctx, uint8_t nb_rounds, uint32_t keylen, const uint8_t *key, uint32_t ivlen, const uint8_t *iv);
@@ -37,10 +37,10 @@ array_copy32(X, &in[(2 * r - 1) * 16], 16); for (i = 0; i < 2 * r; i += 2) {- crypton_salsa_core_xor(8, (block *) X, (block *) &in[i*16]);+ crypton_salsa_core_xor(8, (crypton_salsa_block *) X, (crypton_salsa_block *) &in[i*16]); array_copy32(&out[i * 8], X, 16); - crypton_salsa_core_xor(8, (block *) X, (block *) &in[i*16+16]);+ crypton_salsa_core_xor(8, (crypton_salsa_block *) X, (crypton_salsa_block *) &in[i*16+16]); array_copy32(&out[i * 8 + r * 16], X, 16); } }
@@ -26,7 +26,54 @@ #include "crypton_sha1.h" #include "crypton_bitfn.h" #include "crypton_align.h"+/*+ * AArch64 can do four rounds at a time with the SHA-1 instructions; see+ * sha1_armv8.c. They are optional in ARMv8.0, so ask before using them.+ * Two threads racing to answer here both write the same value.+ */+#ifdef WITH_ARMV8_SHA1+extern void crypton_sha1_armv8_do_chunk(uint32_t state[5], const uint8_t buf[64]);+extern void crypton_sha1_armv8_do_chunks(uint32_t state[5], const uint8_t *data,+ uint32_t blocks);+extern int crypton_sha1_armv8_available(void); +#ifdef WITH_ARMV8_SHA1_ASM+/*+ * SHA-1 from CRYPTOGAMS, in cbits/asm/sha1-armv8-*.S. The instructions are+ * the ones the intrinsics beside it use; what the module does with them is+ * schedule the message schedule of the next four rounds against the rounds+ * of this one, which a C function cannot be made to do.+ *+ * The entry point for processors that have the instructions is not+ * exported, so the module's own dispatch is what picks it, and the answer+ * to the question this file already asks goes into the word that dispatch+ * reads.+ */+#define SHA1_ASM 1+#include "crypton_cpu.h"+extern void crypton_sha1_asm_block_data_order(uint32_t state[5],+ const void *data, size_t blocks);+#endif++/* Resolved before there is a second thread; see the constructor in+ * cbits/crypton_aes.c for why. The test below then only ever reads. */+static int sha1_use_armv8 = -1;++__attribute__((constructor))+static void sha1_armv8_ctor(void)+{+ sha1_use_armv8 = crypton_sha1_armv8_available();+#ifdef SHA1_ASM+ if (sha1_use_armv8)+ crypton_armcap_P |= CRYPTON_ARMCAP_SHA1;+#endif+}+#endif++#ifdef WITH_X86_SHA_NI+#include "crypton_cpu.h"+#endif+ void crypton_sha1_init(struct sha1_ctx *ctx) { memset(ctx, 0, sizeof(*ctx));@@ -54,11 +101,13 @@ #define M(i) (w[i & 0x0f] = rol32(w[i & 0x0f] ^ w[(i - 14) & 0x0f] \ ^ w[(i - 8) & 0x0f] ^ w[(i - 3) & 0x0f], 1)) -static inline void sha1_do_chunk(struct sha1_ctx *ctx, uint32_t *buf)+/* The words are read out of the block rather than the block being pointed at+ * as though it were an array of them; see crypton_md5.c. */+static void sha1_do_chunk_generic(struct sha1_ctx *ctx, const uint8_t *buf) { uint32_t a, b, c, d, e; uint32_t w[16];-#define CPY(i) w[i] = be32_to_cpu(buf[i])+#define CPY(i) w[i] = load_be32(buf + 4 * (i)) CPY(0); CPY(1); CPY(2); CPY(3); CPY(4); CPY(5); CPY(6); CPY(7); CPY(8); CPY(9); CPY(10); CPY(11); CPY(12); CPY(13); CPY(14); CPY(15); #undef CPY@@ -156,6 +205,50 @@ ctx->h[4] += e; } +#ifdef WITH_X86_SHA_NI+/*+ * x86 can do four rounds at a time with the SHA extensions; see sha1_x86.c.+ * They arrived long after the x86-64 baseline, so ask before using them.+ * Two threads racing to answer here both write the same value.+ */+extern void crypton_sha1_x86_do_chunk(uint32_t state[5], const uint8_t buf[64]);+extern void crypton_sha1_x86_do_chunks(uint32_t state[5], const uint8_t *data,+ uint32_t blocks);++static int sha1_use_x86 = -1;+#endif++static inline void sha1_do_chunk(struct sha1_ctx *ctx, const uint8_t *buf)+{+#ifdef WITH_ARMV8_SHA1+ if (sha1_use_armv8 < 0) {+ sha1_use_armv8 = crypton_sha1_armv8_available();+#ifdef SHA1_ASM+ if (sha1_use_armv8)+ crypton_armcap_P |= CRYPTON_ARMCAP_SHA1;+#endif+ }+ if (sha1_use_armv8) {+#ifdef SHA1_ASM+ crypton_sha1_asm_block_data_order(ctx->h, buf, 1);+#else+ crypton_sha1_armv8_do_chunk(ctx->h, buf);+#endif+ return;+ }+#endif+#ifdef WITH_X86_SHA_NI+ if (sha1_use_x86 < 0)+ sha1_use_x86 =+ (crypton_x86_simd_features() & CRYPTON_X86_SHA_NI) != 0;+ if (sha1_use_x86) {+ crypton_sha1_x86_do_chunk(ctx->h, buf);+ return;+ }+#endif+ sha1_do_chunk_generic(ctx, buf);+}+ void crypton_sha1_update(struct sha1_ctx *ctx, const uint8_t *data, uint32_t len) { uint32_t index, to_fill;@@ -168,24 +261,54 @@ /* process partial buffer if there's enough data to make a block */ if (index && len >= to_fill) { memcpy(ctx->buf + index, data, to_fill);- sha1_do_chunk(ctx, (uint32_t *) ctx->buf);+ sha1_do_chunk(ctx, ctx->buf); len -= to_fill; data += to_fill; index = 0; } - if (need_alignment(data, 4)) {- uint32_t tramp[16];- ASSERT_ALIGNMENT(tramp, 4);- for (; len >= 64; len -= 64, data += 64) {- memcpy(tramp, data, 64);- sha1_do_chunk(ctx, tramp);- }- } else {- /* process as much 64-block as possible */- for (; len >= 64; len -= 64, data += 64)- sha1_do_chunk(ctx, (uint32_t *) data);+ /*+ * Where there are instructions for this, the whole run of blocks+ * goes over at once: the state then stays in registers from one+ * block to the next, and the message is read as bytes, so neither+ * the alignment nor the copy below is wanted.+ */+#ifdef WITH_ARMV8_SHA1+ if (sha1_use_armv8 < 0) {+ sha1_use_armv8 = crypton_sha1_armv8_available();+#ifdef SHA1_ASM+ if (sha1_use_armv8)+ crypton_armcap_P |= CRYPTON_ARMCAP_SHA1;+#endif }+ if (sha1_use_armv8 && len >= 64) {+ uint32_t blocks = len / 64;++#ifdef SHA1_ASM+ crypton_sha1_asm_block_data_order(ctx->h, data, blocks);+#else+ crypton_sha1_armv8_do_chunks(ctx->h, data, blocks);+#endif+ data += blocks * 64;+ len -= blocks * 64;+ }+#endif+#ifdef WITH_X86_SHA_NI+ if (sha1_use_x86 < 0)+ sha1_use_x86 =+ (crypton_x86_simd_features() & CRYPTON_X86_SHA_NI) != 0;+ if (sha1_use_x86 && len >= 64) {+ uint32_t blocks = len / 64;++ crypton_sha1_x86_do_chunks(ctx->h, data, blocks);+ data += blocks * 64;+ len -= blocks * 64;+ }+#endif++ /* No trampoline: load_be32 does not ask for a boundary. */+ for (; len >= 64; len -= 64, data += 64)+ sha1_do_chunk(ctx, data); /* append data into buf */ if (len)
@@ -26,6 +26,9 @@ #include "crypton_sha256.h" #include "crypton_bitfn.h" #include "crypton_align.h"+#ifdef WITH_X86_SHA_NI+#include "crypton_cpu.h"+#endif void crypton_sha224_init(struct sha224_ctx *ctx) {@@ -75,13 +78,16 @@ #define s0(x) (ror32(x, 7) ^ ror32(x,18) ^ (x >> 3)) #define s1(x) (ror32(x,17) ^ ror32(x,19) ^ (x >> 10)) -static void sha256_do_chunk(struct sha256_ctx *ctx, uint32_t buf[])+/* The sixteen words are read out of the block rather than the block being+ * pointed at as though it were an array of them; see crypton_md5.c. */+static void sha256_do_chunk_generic(struct sha256_ctx *ctx, const uint8_t *buf) { uint32_t a, b, c, d, e, f, g, h, t1, t2; int i; uint32_t w[64]; - cpu_to_be32_array(w, buf, 16);+ for (i = 0; i < 16; i++)+ w[i] = load_be32(buf + 4 * i); for (i = 16; i < 64; i++) w[i] = s1(w[i - 2]) + w[i - 7] + s0(w[i - 15]) + w[i - 16]; @@ -111,6 +117,86 @@ ctx->h[4] += e; ctx->h[5] += f; ctx->h[6] += g; ctx->h[7] += h; } +#ifdef WITH_ARMV8_SHA2+/*+ * AArch64 can do four rounds at a time with the SHA-2 instructions; see+ * sha256_armv8.c. They are optional in ARMv8.0, so ask before using them.+ * Two threads racing to answer here both write the same value.+ */+extern void crypton_sha256_armv8_do_chunk(uint32_t state[8], const uint8_t buf[64]);+extern int crypton_sha256_armv8_available(void);++/* Resolved before there is a second thread; see the constructor in+ * cbits/crypton_aes.c for why. The test below then only ever reads. */+static int sha256_use_armv8 = -1;++__attribute__((constructor))+static void sha256_armv8_ctor(void)+{+ sha256_use_armv8 = crypton_sha256_armv8_available();+}+#endif++#if (defined(WITH_ARMV8_SHA256_ASM) && defined(WITH_ARMV8_SHA2)) \+ || defined(WITH_X86_SHA256_ASM)+/*+ * SHA-256 from CRYPTOGAMS, in cbits/asm/sha256-armv8-*.S and+ * cbits/asm/sha256-x86_64-*.S, which take any number of blocks at once and+ * schedule the instructions across them -- which is where they are ahead+ * of the intrinsics above, the instructions being the same ones. Each+ * picks its own path from the word the processor was asked about, so the+ * answer to the runtime check goes there rather than into a branch here.+ */+#define SHA256_ASM 1+#include "crypton_cpu.h"+extern void crypton_sha256_asm_block_data_order(uint32_t state[8],+ const void *data, size_t blocks);++#ifdef WITH_ARMV8_SHA256_ASM+/* The assembly picks its path from crypton_armcap_P, so the answer to the+ * runtime check has to reach that word rather than the flag above. It is+ * set here, before there is a second thread, for the reason the constructor+ * in cbits/crypton_aes.c gives.+ *+ * This ran from sha256_asm_ready below until 2.1.3, guarded by the flag+ * still being unresolved -- which stopped happening when the constructor+ * above was added, so the bit was never set and the assembly took its+ * generic path. SHA-256 was 5.7 times slower on an Apple M4 for it. */+__attribute__((constructor))+static void sha256_armcap_ctor(void)+{+ if (crypton_sha256_armv8_available())+ crypton_armcap_P |= CRYPTON_ARMCAP_SHA256;+}+#endif++static void sha256_asm_ready(void)+{+#ifndef WITH_ARMV8_SHA256_ASM+ crypton_x86_ia32cap_resolve();+#endif+}+#endif+++static void sha256_do_chunk(struct sha256_ctx *ctx, const uint8_t *buf)+{+#ifdef SHA256_ASM+ sha256_asm_ready();+ crypton_sha256_asm_block_data_order(ctx->h, buf, 1);+ return;+#endif+#if defined(WITH_ARMV8_SHA2) && !defined(SHA256_ASM)+ if (sha256_use_armv8 < 0)+ sha256_use_armv8 = crypton_sha256_armv8_available();+ if (sha256_use_armv8) {+ crypton_sha256_armv8_do_chunk(ctx->h, buf);+ return;+ }+#endif+ sha256_do_chunk_generic(ctx, buf);+}+ void crypton_sha224_update(struct sha224_ctx *ctx, const uint8_t *data, uint32_t len) { return crypton_sha256_update(ctx, data, len);@@ -129,24 +215,29 @@ /* process partial buffer if there's enough data to make a block */ if (index && len >= to_fill) { memcpy(ctx->buf + index, data, to_fill);- sha256_do_chunk(ctx, (uint32_t *) ctx->buf);+ sha256_do_chunk(ctx, ctx->buf); len -= to_fill; data += to_fill; index = 0; } - if (need_alignment(data, 4)) {- uint32_t tramp[16];- ASSERT_ALIGNMENT(tramp, 4);- for (; len >= 64; len -= 64, data += 64) {- memcpy(tramp, data, 64);- sha256_do_chunk(ctx, tramp);- }- } else {- /* process as much 64-block as possible */- for (; len >= 64; len -= 64, data += 64)- sha256_do_chunk(ctx, (uint32_t *) data);+#ifdef SHA256_ASM+ /* the assembly reads the message a byte at a time as far as the+ * machine is concerned, so it wants no alignment and no copy, and+ * it takes the whole run of blocks in one call */+ if (len >= 64) {+ size_t blocks = len / 64;++ sha256_asm_ready();+ crypton_sha256_asm_block_data_order(ctx->h, data, blocks);+ data += blocks * 64;+ len -= (uint32_t) blocks * 64; }+#else+ /* No trampoline: load_be32 does not ask for a boundary. */+ for (; len >= 64; len -= 64, data += 64)+ sha256_do_chunk(ctx, data);+#endif /* append data into buf */ if (len)
@@ -51,6 +51,18 @@ void crypton_sha256_init(struct sha256_ctx *ctx); void crypton_sha256_update(struct sha256_ctx *ctx, const uint8_t *data, uint32_t len);+/* The pointers are all required to be non-null, which is said here so that+ * the compiler knows it too. Both of these write their digest through a+ * loop -- store_be32(out + 4 * i, ...) -- where sha1 and md5 write theirs at+ * constant offsets, and that is the difference that makes gcc's+ * -Wstringop-overflow reason about out being null: with+ * -fsanitize=undefined, UndefinedBehaviorSanitizer inserts a null check+ * before memcpy, because glibc declares memcpy nonnull, and the check puts a+ * null path in front of the warning pass, which then reports writing into+ * "a region of size 0" at "address zero". Saying the pointer is never null+ * removes the path rather than the warning. It costs nothing: compiled as+ * the package compiles it, the assembly is identical with and without. */+__attribute__((nonnull)) void crypton_sha256_finalize(struct sha256_ctx *ctx, uint8_t *out); void crypton_sha256_finalize_prefix(struct sha256_ctx *ctx, const uint8_t *data, uint32_t len, uint32_t n, uint8_t *out);
@@ -50,15 +50,76 @@ static const int keccak_piln[24] = { 10,7,11,17,18,3,5,16,8,21,24,4,15,23,19,13,12,2,20,14,22,9,6,1 }; -static inline void sha3_do_chunk(uint64_t state[25], uint64_t buf[], int bufsz)+/*+ * AArch64 has instructions for this permutation; see sha3_armv8.c. They are+ * an ARMv8.2 extension, so ask before using them. Two threads racing to+ * answer here both write the same value.+ */+#ifdef WITH_ARMV8_SHA3+extern void crypton_sha3_armv8_permute(uint64_t state[25]);+extern int crypton_sha3_armv8_available(void);++static int sha3_use_armv8 = -1;++/* Two threads racing to answer this both write the same value. */+static int sha3_armv8_ok(void) {+ if (sha3_use_armv8 < 0)+ sha3_use_armv8 = crypton_sha3_armv8_available();+ return sha3_use_armv8;+}+#endif++#if defined(WITH_ARMV8_SHA3_ASM) && !defined(__AARCH64EB__)+/*+ * Keccak from CRYPTOGAMS, in cbits/asm/keccak1600-armv8-*.S, which takes a+ * run of blocks rather than one at a time and schedules the instructions+ * across the round it is in and the next. The instructions are the same+ * ones the intrinsics beside it use; the arrangement is what is worth+ * about a tenth here. It reads the message as bytes, so the run wants+ * neither alignment nor a copy.+ */+#define SHA3_ASM 1+/* the runtime question this file already asks decides whether it is used */+#define SHA3_ASM_OPTIONAL 1+extern size_t crypton_keccak_asm_absorb_cext(uint64_t state[25], const void *inp,+ size_t len, size_t bsz);+#define sha3_asm_absorb crypton_keccak_asm_absorb_cext+#endif++#ifdef WITH_X86_SHA3_ASM+/*+ * And the same module for x86-64, where there are no instructions for this+ * and what the assembly has over the C is the arrangement: the twenty-five+ * lanes live in registers across a round, where a compiler given the C+ * below spills them, and the rotations are folded into the operations that+ * consume them. It needs nothing of the processor beyond the baseline, so+ * unlike the AArch64 one it is used wherever it is compiled in.+ */+#define SHA3_ASM 1+extern size_t crypton_keccak_asm_absorb(uint64_t state[25], const void *inp,+ size_t len, size_t bsz);+#define sha3_asm_absorb crypton_keccak_asm_absorb+#endif++/* The words are read out of the block rather than the block being pointed at+ * as though it were an array of them; see crypton_md5.c. */+static inline void sha3_do_chunk(uint64_t state[25], const uint8_t *buf, int bufsz)+{ int i, j, r; uint64_t tmp, bc[5]; /* merge buf with state */ for (i = 0; i < bufsz; i++)- state[i] ^= le64_to_cpu(buf[i]);+ state[i] ^= load_le64(buf + 8 * i); +#ifdef WITH_ARMV8_SHA3+ if (sha3_armv8_ok()) {+ crypton_sha3_armv8_permute(state);+ return;+ }+#endif+ /* run keccak rounds */ for (r = 0; r < KECCAK_NB_ROUNDS; r++) { /* compute the parity of each columns */@@ -121,33 +182,38 @@ to_fill = ctx->bufsz - ctx->bufindex; if (ctx->bufindex == ctx->bufsz) {- sha3_do_chunk(ctx->state, (uint64_t *) ctx->buf, ctx->bufsz / 8);+ sha3_do_chunk(ctx->state, ctx->buf, ctx->bufsz / 8); ctx->bufindex = 0; } /* process partial buffer if there's enough data to make a block */ if (ctx->bufindex && len >= to_fill) { memcpy(ctx->buf + ctx->bufindex, data, to_fill);- sha3_do_chunk(ctx->state, (uint64_t *) ctx->buf, ctx->bufsz / 8);+ sha3_do_chunk(ctx->state, ctx->buf, ctx->bufsz / 8); len -= to_fill; data += to_fill; ctx->bufindex = 0; } - if (need_alignment(data, 8)) {- uint64_t tramp[SHA3_BUF_SIZE_MAX/8];- ASSERT_ALIGNMENT(tramp, 8);- for (; len >= ctx->bufsz; len -= ctx->bufsz, data += ctx->bufsz) {- memcpy(tramp, data, ctx->bufsz);- sha3_do_chunk(ctx->state, tramp, ctx->bufsz / 8);- }- } else {- /* process as much ctx->bufsz-block */- for (; len >= ctx->bufsz; len -= ctx->bufsz, data += ctx->bufsz)- sha3_do_chunk(ctx->state, (uint64_t *) data, ctx->bufsz / 8);+#ifdef SHA3_ASM+ if (len >= ctx->bufsz+#ifdef SHA3_ASM_OPTIONAL+ && sha3_armv8_ok()+#endif+ ) {+ const size_t left = sha3_asm_absorb(ctx->state, data, len,+ ctx->bufsz);++ data += len - left;+ len = (uint32_t) left; }+#endif + /* No trampoline: load_le64 does not ask for a boundary. */+ for (; len >= ctx->bufsz; len -= ctx->bufsz, data += ctx->bufsz)+ sha3_do_chunk(ctx->state, data, ctx->bufsz / 8); + /* append data into buf */ if (len) { memcpy(ctx->buf + ctx->bufindex, data, len);@@ -159,7 +225,7 @@ { /* process full buffer if needed */ if (ctx->bufindex == ctx->bufsz) {- sha3_do_chunk(ctx->state, (uint64_t *) ctx->buf, ctx->bufsz / 8);+ sha3_do_chunk(ctx->state, ctx->buf, ctx->bufsz / 8); ctx->bufindex = 0; } @@ -169,7 +235,7 @@ ctx->buf[ctx->bufsz - 1] |= 0x80; /* process */- sha3_do_chunk(ctx->state, (uint64_t *) ctx->buf, ctx->bufsz / 8);+ sha3_do_chunk(ctx->state, ctx->buf, ctx->bufsz / 8); ctx->bufindex = 0; }
@@ -91,13 +91,16 @@ #define s0(x) (ror64(x, 1) ^ ror64(x, 8) ^ (x >> 7)) #define s1(x) (ror64(x, 19) ^ ror64(x, 61) ^ (x >> 6)) -static void sha512_do_chunk(struct sha512_ctx *ctx, uint64_t *buf)+/* The words are read out of the block rather than the block being pointed at+ * as though it were an array of them; see crypton_md5.c. */+static void sha512_do_chunk_generic(struct sha512_ctx *ctx, const uint8_t *buf) { uint64_t a, b, c, d, e, f, g, h, t1, t2; int i; uint64_t w[80]; - cpu_to_be64_array(w, buf, 16);+ for (i = 0; i < 16; i++)+ w[i] = load_be64(buf + 8 * i); for (i = 16; i < 80; i++) w[i] = s1(w[i - 2]) + w[i - 7] + s0(w[i - 15]) + w[i - 16];@@ -128,6 +131,58 @@ ctx->h[4] += e; ctx->h[5] += f; ctx->h[6] += g; ctx->h[7] += h; } +#ifdef WITH_ARMV8_SHA512+/*+ * AArch64 can do two rounds at a time with the SHA-512 instructions; see+ * sha512_armv8.c. They are an optional ARMv8.2 extension and much less+ * widespread than the SHA-256 ones, so ask before using them. Two threads+ * racing to answer here both write the same value.+ */+extern void crypton_sha512_armv8_do_chunk(uint64_t state[8], const uint8_t buf[128]);+extern int crypton_sha512_armv8_available(void);++/* Resolved before there is a second thread; see the constructor in+ * cbits/crypton_aes.c for why. The test below then only ever reads. */+static int sha512_use_armv8 = -1;++__attribute__((constructor))+static void sha512_armv8_ctor(void)+{+ sha512_use_armv8 = crypton_sha512_armv8_available();+}+#endif+++#ifdef WITH_X86_SHA512_ASM+/*+ * SHA-512 from CRYPTOGAMS, in cbits/asm/sha512-x86_64-*.S, which takes any+ * number of blocks at once and schedules across them, and picks between+ * AVX2, AVX, SSSE3 and plain integer code from crypton_ia32cap_P.+ */+#define SHA512_ASM 1+#include "crypton_cpu.h"+extern void crypton_sha512_asm_block_data_order(uint64_t state[8],+ const void *data, size_t blocks);+#endif++static void sha512_do_chunk(struct sha512_ctx *ctx, const uint8_t *buf)+{+#ifdef SHA512_ASM+ crypton_x86_ia32cap_resolve();+ crypton_sha512_asm_block_data_order(ctx->h, buf, 1);+ return;+#endif+#ifdef WITH_ARMV8_SHA512+ if (sha512_use_armv8 < 0)+ sha512_use_armv8 = crypton_sha512_armv8_available();+ if (sha512_use_armv8) {+ crypton_sha512_armv8_do_chunk(ctx->h, buf);+ return;+ }+#endif+ sha512_do_chunk_generic(ctx, buf);+}+ void crypton_sha384_update(struct sha384_ctx *ctx, const uint8_t *data, uint32_t len) { return crypton_sha512_update(ctx, data, len);@@ -148,24 +203,28 @@ /* process partial buffer if there's enough data to make a block */ if (index && len >= to_fill) { memcpy(ctx->buf + index, data, to_fill);- sha512_do_chunk(ctx, (uint64_t *) ctx->buf);+ sha512_do_chunk(ctx, ctx->buf); len -= to_fill; data += to_fill; index = 0; } - if (need_alignment(data, 8)) {- uint64_t tramp[16];- ASSERT_ALIGNMENT(tramp, 8);- for (; len >= 128; len -= 128, data += 128) {- memcpy(tramp, data, 128);- sha512_do_chunk(ctx, tramp);- }- } else {- /* process as much 128-block as possible */- for (; len >= 128; len -= 128, data += 128)- sha512_do_chunk(ctx, (uint64_t *) data);+#ifdef SHA512_ASM+ /* the assembly reads the message as bytes, so it wants neither the+ * alignment nor the copy, and takes the whole run in one call */+ if (len >= 128) {+ size_t blocks = len / 128;++ crypton_x86_ia32cap_resolve();+ crypton_sha512_asm_block_data_order(ctx->h, data, blocks);+ data += blocks * 128;+ len -= (uint32_t) blocks * 128; }+#else+ /* No trampoline: load_be64 does not ask for a boundary. */+ for (; len >= 128; len -= 128, data += 128)+ sha512_do_chunk(ctx, data);+#endif /* append data into buf */ if (len)@@ -287,7 +346,7 @@ /* re-init the context, otherwise len is changed */ memset(ctx, 0, sizeof(*ctx)); for (i = 0; i < 8; i++)- ctx->h[i] = cpu_to_be64(((uint64_t *) out)[i]);+ ctx->h[i] = load_be64(out + 8 * i); } } }
@@ -50,6 +50,18 @@ void crypton_sha512_init(struct sha512_ctx *ctx); void crypton_sha512_update(struct sha512_ctx *ctx, const uint8_t *data, uint32_t len);+/* The pointers are all required to be non-null, which is said here so that+ * the compiler knows it too. Both of these write their digest through a+ * loop -- store_be32(out + 4 * i, ...) -- where sha1 and md5 write theirs at+ * constant offsets, and that is the difference that makes gcc's+ * -Wstringop-overflow reason about out being null: with+ * -fsanitize=undefined, UndefinedBehaviorSanitizer inserts a null check+ * before memcpy, because glibc declares memcpy nonnull, and the check puts a+ * null path in front of the warning pass, which then reports writing into+ * "a region of size 0" at "address zero". Saying the pointer is never null+ * removes the path rather than the warning. It costs nothing: compiled as+ * the package compiles it, the assembly is identical with and without. */+__attribute__((nonnull)) void crypton_sha512_finalize(struct sha512_ctx *ctx, uint8_t *out); void crypton_sha512_finalize_prefix(struct sha512_ctx *ctx, const uint8_t *data, uint32_t len, uint32_t n, uint8_t *out);
@@ -37,7 +37,9 @@ static const uint8_t K256_6[2] = { 58, 22, }; static const uint8_t K256_7[2] = { 32, 32, }; -static inline void skein256_do_chunk(struct skein256_ctx *ctx, uint64_t *buf, uint32_t len)+/* The four words are read out of the block rather than the block being+ * pointed at as though it were an array of them; see crypton_md5.c. */+static inline void skein256_do_chunk(struct skein256_ctx *ctx, const uint8_t *bufp, uint32_t len) { uint64_t x[4]; uint64_t ts[3];@@ -78,11 +80,18 @@ ROUND(0,3,2,1,K256_7); \ INJECTKEY((i*2) + 2) - x[0] = le64_to_cpu(buf[0]) + ks[0];- x[1] = le64_to_cpu(buf[1]) + ks[1] + ts[0];- x[2] = le64_to_cpu(buf[2]) + ks[2] + ts[1];- x[3] = le64_to_cpu(buf[3]) + ks[3];+ uint64_t buf[4]; + buf[0] = load_le64(bufp);+ buf[1] = load_le64(bufp + 8);+ buf[2] = load_le64(bufp + 16);+ buf[3] = load_le64(bufp + 24);++ x[0] = buf[0] + ks[0];+ x[1] = buf[1] + ks[1] + ts[0];+ x[2] = buf[2] + ks[2] + ts[1];+ x[3] = buf[3] + ks[3];+ /* 9 pass of 8 rounds = 72 rounds */ PASS(0); PASS(1);@@ -98,10 +107,10 @@ ctx->t0 = ts[0]; ctx->t1 = ts[1]; - ctx->h[0] = x[0] ^ cpu_to_le64(buf[0]);- ctx->h[1] = x[1] ^ cpu_to_le64(buf[1]);- ctx->h[2] = x[2] ^ cpu_to_le64(buf[2]);- ctx->h[3] = x[3] ^ cpu_to_le64(buf[3]);+ ctx->h[0] = x[0] ^ buf[0];+ ctx->h[1] = x[1] ^ buf[1];+ ctx->h[2] = x[2] ^ buf[2];+ ctx->h[3] = x[3] ^ buf[3]; } void crypton_skein256_init(struct skein256_ctx *ctx, uint32_t hashlen)@@ -115,7 +124,7 @@ buf[0] = cpu_to_le64((SKEIN_VERSION << 32) | SKEIN_IDSTRING); buf[1] = cpu_to_le64(hashlen); buf[2] = 0; /* tree info, not implemented */- skein256_do_chunk(ctx, buf, 4*8);+ skein256_do_chunk(ctx, (const uint8_t *) buf, 4*8); SET_TYPE(ctx, FLAG_FIRST | FLAG_TYPE(TYPE_MSG)); }@@ -130,7 +139,7 @@ to_fill = 32 - ctx->bufindex; if (ctx->bufindex == 32) {- skein256_do_chunk(ctx, (uint64_t *) ctx->buf, 32);+ skein256_do_chunk(ctx, ctx->buf, 32); ctx->bufindex = 0; } @@ -138,24 +147,17 @@ * and there's without doubt further blocks */ if (ctx->bufindex && len > to_fill) { memcpy(ctx->buf + ctx->bufindex, data, to_fill);- skein256_do_chunk(ctx, (uint64_t *) ctx->buf, 32);+ skein256_do_chunk(ctx, ctx->buf, 32); len -= to_fill; data += to_fill; ctx->bufindex = 0; } - if (need_alignment(data, 8)) {- uint64_t tramp[4];- ASSERT_ALIGNMENT(tramp, 8);- for (; len > 32; len -= 32, data += 32) {- memcpy(tramp, data, 32);- skein256_do_chunk(ctx, tramp, 32);- }- } else {- /* process as much 32-block as possible except the last one in case we finalize */- for (; len > 32; len -= 32, data += 32)- skein256_do_chunk(ctx, (uint64_t *) data, 32);- }+ /* No trampoline for a block that is not on an eight-byte boundary: the+ * words are read with load_le64 now, which does not ask. The last+ * block is left for the finalisation. */+ for (; len > 32; len -= 32, data += 32)+ skein256_do_chunk(ctx, data, 32); /* append data into buf */ if (len) {@@ -174,7 +176,7 @@ /* if buf is not complete pad with 0 bytes */ if (ctx->bufindex < 32) memset(ctx->buf + ctx->bufindex, '\0', 32 - ctx->bufindex);- skein256_do_chunk(ctx, (uint64_t *) ctx->buf, ctx->bufindex);+ skein256_do_chunk(ctx, ctx->buf, ctx->bufindex); memset(ctx->buf, '\0', 32); @@ -187,9 +189,9 @@ /* threefish in counter mode, 0 for 1st 64 bytes, 1 for 2nd 64 bytes, .. */ for (i = 0; i*32 < outsize; i++) { uint64_t w[4];- *((uint64_t *) ctx->buf) = cpu_to_le64(i);+ store_le64(ctx->buf, i); SET_TYPE(ctx, FLAG_FIRST | FLAG_FINAL | FLAG_TYPE(TYPE_OUT));- skein256_do_chunk(ctx, (uint64_t *) ctx->buf, sizeof(uint64_t));+ skein256_do_chunk(ctx, ctx->buf, sizeof(uint64_t)); n = outsize - i * 32; if (n >= 32) n = 32;
@@ -37,8 +37,8 @@ #define SKEIN256_CTX_SIZE sizeof(struct skein256_ctx) -void cryponite_skein256_init(struct skein256_ctx *ctx, uint32_t hashlen);-void cryponite_skein256_update(struct skein256_ctx *ctx, const uint8_t *data, uint32_t len);-void cryponite_skein256_finalize(struct skein256_ctx *ctx, uint32_t hashlen, uint8_t *out);+void crypton_skein256_init(struct skein256_ctx *ctx, uint32_t hashlen);+void crypton_skein256_update(struct skein256_ctx *ctx, const uint8_t *data, uint32_t len);+void crypton_skein256_finalize(struct skein256_ctx *ctx, uint32_t hashlen, uint8_t *out); #endif
@@ -37,7 +37,9 @@ static const uint8_t K512_6[4] = { 25, 29, 39, 43, }; static const uint8_t K512_7[4] = { 8, 35, 56, 22, }; -static inline void skein512_do_chunk(struct skein512_ctx *ctx, uint64_t *buf, uint32_t len)+/* The words are read out of the block rather than the block being pointed at+ * as though it were an array of them; see crypton_md5.c. */+static inline void skein512_do_chunk(struct skein512_ctx *ctx, const uint8_t *bufp, uint32_t len) { uint64_t x[8]; uint64_t ts[3];@@ -88,15 +90,21 @@ ROUND(6,1,0,7,2,5,4,3,K512_7); \ INJECTKEY((i*2) + 2) - x[0] = le64_to_cpu(buf[0]) + ks[0];- x[1] = le64_to_cpu(buf[1]) + ks[1];- x[2] = le64_to_cpu(buf[2]) + ks[2];- x[3] = le64_to_cpu(buf[3]) + ks[3];- x[4] = le64_to_cpu(buf[4]) + ks[4];- x[5] = le64_to_cpu(buf[5]) + ks[5] + ts[0];- x[6] = le64_to_cpu(buf[6]) + ks[6] + ts[1];- x[7] = le64_to_cpu(buf[7]) + ks[7];+ uint64_t buf[8];+ int bi; + for (bi = 0; bi < 8; bi++)+ buf[bi] = load_le64(bufp + 8 * bi);++ x[0] = buf[0] + ks[0];+ x[1] = buf[1] + ks[1];+ x[2] = buf[2] + ks[2];+ x[3] = buf[3] + ks[3];+ x[4] = buf[4] + ks[4];+ x[5] = buf[5] + ks[5] + ts[0];+ x[6] = buf[6] + ks[6] + ts[1];+ x[7] = buf[7] + ks[7];+ /* 9 pass of 8 rounds = 72 rounds */ PASS(0); PASS(1);@@ -112,14 +120,14 @@ ctx->t0 = ts[0]; ctx->t1 = ts[1]; - ctx->h[0] = x[0] ^ cpu_to_le64(buf[0]);- ctx->h[1] = x[1] ^ cpu_to_le64(buf[1]);- ctx->h[2] = x[2] ^ cpu_to_le64(buf[2]);- ctx->h[3] = x[3] ^ cpu_to_le64(buf[3]);- ctx->h[4] = x[4] ^ cpu_to_le64(buf[4]);- ctx->h[5] = x[5] ^ cpu_to_le64(buf[5]);- ctx->h[6] = x[6] ^ cpu_to_le64(buf[6]);- ctx->h[7] = x[7] ^ cpu_to_le64(buf[7]);+ ctx->h[0] = x[0] ^ buf[0];+ ctx->h[1] = x[1] ^ buf[1];+ ctx->h[2] = x[2] ^ buf[2];+ ctx->h[3] = x[3] ^ buf[3];+ ctx->h[4] = x[4] ^ buf[4];+ ctx->h[5] = x[5] ^ buf[5];+ ctx->h[6] = x[6] ^ buf[6];+ ctx->h[7] = x[7] ^ buf[7]; } void crypton_skein512_init(struct skein512_ctx *ctx, uint32_t hashlen)@@ -133,7 +141,7 @@ buf[0] = cpu_to_le64((SKEIN_VERSION << 32) | SKEIN_IDSTRING); buf[1] = cpu_to_le64(hashlen); buf[2] = 0; /* tree info, not implemented */- skein512_do_chunk(ctx, buf, 4*8);+ skein512_do_chunk(ctx, (const uint8_t *) buf, 4*8); SET_TYPE(ctx, FLAG_FIRST | FLAG_TYPE(TYPE_MSG)); }@@ -148,7 +156,7 @@ to_fill = 64 - ctx->bufindex; if (ctx->bufindex == 64) {- skein512_do_chunk(ctx, (uint64_t *) ctx->buf, 64);+ skein512_do_chunk(ctx, ctx->buf, 64); ctx->bufindex = 0; } @@ -156,24 +164,15 @@ * and there's without doubt further blocks */ if (ctx->bufindex && len > to_fill) { memcpy(ctx->buf + ctx->bufindex, data, to_fill);- skein512_do_chunk(ctx, (uint64_t *) ctx->buf, 64);+ skein512_do_chunk(ctx, ctx->buf, 64); len -= to_fill; data += to_fill; ctx->bufindex = 0; } - if (need_alignment(data, 8)) {- uint64_t tramp[8];- ASSERT_ALIGNMENT(tramp, 8);- for (; len > 64; len -= 64, data += 64) {- memcpy(tramp, data, 64);- skein512_do_chunk(ctx, tramp, 64);- }- } else {- /* process as much 64-block as possible except the last one in case we finalize */- for (; len > 64; len -= 64, data += 64)- skein512_do_chunk(ctx, (uint64_t *) data, 64);- }+ /* No trampoline: load_le64 does not ask for a boundary. */+ for (; len > 64; len -= 64, data += 64)+ skein512_do_chunk(ctx, data, 64); /* append data into buf */ if (len) {@@ -192,7 +191,7 @@ /* if buf is not complete pad with 0 bytes */ if (ctx->bufindex < 64) memset(ctx->buf + ctx->bufindex, '\0', 64 - ctx->bufindex);- skein512_do_chunk(ctx, (uint64_t *) ctx->buf, ctx->bufindex);+ skein512_do_chunk(ctx, ctx->buf, ctx->bufindex); memset(ctx->buf, '\0', 64); @@ -205,9 +204,9 @@ /* threefish in counter mode, 0 for 1st 64 bytes, 1 for 2nd 64 bytes, .. */ for (i = 0; i*64 < outsize; i++) { uint64_t w[8];- *((uint64_t *) ctx->buf) = cpu_to_le64(i);+ store_le64(ctx->buf, i); SET_TYPE(ctx, FLAG_FIRST | FLAG_FINAL | FLAG_TYPE(TYPE_OUT));- skein512_do_chunk(ctx, (uint64_t *) ctx->buf, sizeof(uint64_t));+ skein512_do_chunk(ctx, ctx->buf, sizeof(uint64_t)); n = outsize - i * 64; if (n >= 64) n = 64;
@@ -37,8 +37,8 @@ #define SKEIN512_CTX_SIZE sizeof(struct skein512_ctx) -void cryponite_skein512_init(struct skein512_ctx *ctx, uint32_t hashlen);-void cryponite_skein512_update(struct skein512_ctx *ctx, const uint8_t *data, uint32_t len);-void cryponite_skein512_finalize(struct skein512_ctx *ctx, uint32_t hashlen, uint8_t *out);+void crypton_skein512_init(struct skein512_ctx *ctx, uint32_t hashlen);+void crypton_skein512_update(struct skein512_ctx *ctx, const uint8_t *data, uint32_t len);+void crypton_skein512_finalize(struct skein512_ctx *ctx, uint32_t hashlen, uint8_t *out); #endif
@@ -306,7 +306,9 @@ ctx->h[2] = 0xf096a5b4c3b2e187ULL; } -static inline void tiger_do_chunk(struct tiger_ctx *ctx, uint64_t *buf)+/* The words are read out of the block rather than the block being pointed at+ * as though it were an array of them; see crypton_md5.c. */+static inline void tiger_do_chunk(struct tiger_ctx *ctx, const uint8_t *buf) { uint64_t x0, x1, x2, x3, x4, x5, x6, x7; uint64_t a,b,c;@@ -314,8 +316,8 @@ b = ctx->h[1]; c = ctx->h[2]; - x0 = cpu_to_le64(buf[0]); x1 = cpu_to_le64(buf[1]); x2 = cpu_to_le64(buf[2]); x3 = cpu_to_le64(buf[3]);- x4 = cpu_to_le64(buf[4]); x5 = cpu_to_le64(buf[5]); x6 = cpu_to_le64(buf[6]); x7 = cpu_to_le64(buf[7]);+ x0 = load_le64(buf ); x1 = load_le64(buf + 8); x2 = load_le64(buf + 16); x3 = load_le64(buf + 24);+ x4 = load_le64(buf + 32); x5 = load_le64(buf + 40); x6 = load_le64(buf + 48); x7 = load_le64(buf + 56); #define BYTEOF(c, n) ((uint8_t) (c >> ((n * 8)))) @@ -376,24 +378,15 @@ /* process partial buffer if there's enough data to make a block */ if (index && len >= to_fill) { memcpy(ctx->buf + index, data, to_fill);- tiger_do_chunk(ctx, (uint64_t *) ctx->buf);+ tiger_do_chunk(ctx, ctx->buf); len -= to_fill; data += to_fill; index = 0; } - if (need_alignment(data, 8)) {- uint64_t tramp[8];- ASSERT_ALIGNMENT(tramp, 8);- for (; len >= 64; len -= 64, data += 64) {- memcpy(tramp, data, 64);- tiger_do_chunk(ctx, tramp);- }- } else {- /* process as much 64-block as possible */- for (; len >= 64; len -= 64, data += 64)- tiger_do_chunk(ctx, (uint64_t *) data);- }+ /* No trampoline: load_le64 does not ask for a boundary. */+ for (; len >= 64; len -= 64, data += 64)+ tiger_do_chunk(ctx, data); /* append data into buf */ if (len)
@@ -41,7 +41,7 @@ memset(ctx, 0, sizeof(*ctx)); ctx->nb_rounds = nb_rounds; - /* Create initial 512-bit input block:+ /* Create initial 512-bit input crypton_salsa_block: (x0, x5, x10, x15) is the Salsa20 constant (x1, x2, x3, x4, x11, x12, x13, x14) is a 256-bit key (x6, x7, x8, x9) is the first 128 bits of a 192-bit nonce@@ -56,7 +56,7 @@ void crypton_xsalsa_derive(crypton_salsa_context *ctx, uint32_t ivlen, const uint8_t *iv) {- /* Finish creating initial 512-bit input block:+ /* Finish creating initial 512-bit input crypton_salsa_block: (x6, x7, x8, x9) is the first 128 bits of a 192-bit nonce Except iv has been shifted by 64 bits so there are now only 128 bits ahead.@@ -65,15 +65,15 @@ ctx->st.d[ 9] += load_le32(iv + 4); /* Compute (z0, z1, . . . , z15) = doubleround ^(r/2) (x0, x1, . . . , x15) */- block hSalsa;- memset(&hSalsa, 0, sizeof(block));+ crypton_salsa_block hSalsa;+ memset(&hSalsa, 0, sizeof(crypton_salsa_block)); crypton_salsa_core_xor(ctx->nb_rounds, &hSalsa, &ctx->st); - /* Build a new 512-bit input block (x′0, x′1, . . . , x′15):+ /* Build a new 512-bit input crypton_salsa_block (x′0, x′1, . . . , x′15): (x′0, x′5, x′10, x′15) is the Salsa20 constant (x′1,x′2,x′3,x′4,x′11,x′12,x′13,x′14) = (z0,z5,z10,z15,z6,z7,z8,z9) (x′6,x′7) is the last 64 bits of the 192-bit nonce- (x′8, x′9) is a 64-bit block counter.+ (x′8, x′9) is a 64-bit crypton_salsa_block counter. */ ctx->st.d[ 1] = hSalsa.d[ 0] - ctx->st.d[ 0]; ctx->st.d[ 2] = hSalsa.d[ 5] - ctx->st.d[ 5];
@@ -0,0 +1,88 @@+/*+ * X25519 through the vendored s2n-bignum where it is built, and through+ * curve25519-donna where it is not.+ *+ * The fixed-base routine is the interesting half: crypton had none, and asked+ * for the public key by multiplying the base point 9 the general way. With a+ * table it is four to five times less work, and a TLS handshake generates a+ * key every time. Measured on an Apple M4:+ *+ * donna s2n+ * shared secret 18.15 us 12.35+ * key generation 18.15 3.65+ *+ * and on an x86-64, 41.19 to 26.96 and 41.18 to 8.54.+ */+#include <string.h>++#include "curve25519/x25519.h"++void crypton_curve25519_donna(uint8_t *mypublic, const uint8_t *secret,+ const uint8_t *basepoint);++/* The assembly takes four little-endian 64-bit words, which is the same bits+ * as the 32 little-endian bytes RFC 7748 sends, so the two cross by copying+ * -- on a little-endian machine, which is the only kind s2n-bignum is for. */+#if defined(CRYPTON_S2N_BIGNUM) && defined(__BYTE_ORDER__) \+ && __BYTE_ORDER__ == __ORDER_LITTLE_ENDIAN__+#define CRYPTON_X25519_S2N 1+#include "crypton_cpu.h"++extern void curve25519_x25519(uint64_t res[4], const uint64_t scalar[4],+ const uint64_t point[4]);+extern void curve25519_x25519_alt(uint64_t res[4], const uint64_t scalar[4],+ const uint64_t point[4]);+extern void curve25519_x25519base(uint64_t res[4], const uint64_t scalar[4]);+extern void curve25519_x25519base_alt(uint64_t res[4], const uint64_t scalar[4]);++/* The same question as everywhere else in cbits/s2n: a microarchitecture one+ * on ARM that no feature bit answers, and exactly a feature bit on x86-64. */+static int use_alt(void)+{+#if defined(__aarch64__) || defined(__arm64__)+#ifdef __APPLE__+ return 1;+#else+ return 0;+#endif+#else+ return (crypton_x86_simd_features() & CRYPTON_X86_ADX) == 0;+#endif+}+#endif++void crypton_x25519(uint8_t out[32], const uint8_t secret[32],+ const uint8_t point[32])+{+#ifdef CRYPTON_X25519_S2N+ uint64_t r[4], s[4], p[4];++ memcpy(s, secret, 32);+ memcpy(p, point, 32);+ if (use_alt())+ curve25519_x25519_alt(r, s, p);+ else+ curve25519_x25519(r, s, p);+ memcpy(out, r, 32);+#else+ crypton_curve25519_donna(out, secret, point);+#endif+}++void crypton_x25519_base(uint8_t out[32], const uint8_t secret[32])+{+#ifdef CRYPTON_X25519_S2N+ uint64_t r[4], s[4];++ memcpy(s, secret, 32);+ if (use_alt())+ curve25519_x25519base_alt(r, s);+ else+ curve25519_x25519base(r, s);+ memcpy(out, r, 32);+#else+ static const uint8_t nine[32] = {9};++ crypton_curve25519_donna(out, secret, nine);+#endif+}
@@ -0,0 +1,16 @@+#ifndef CRYPTON_X25519_H+#define CRYPTON_X25519_H++#include <stdint.h>++/* out = secret * point, the X25519 of RFC 7748: three 32-byte little-endian+ * strings as they go over the wire. */+void crypton_x25519(uint8_t out[32], const uint8_t secret[32],+ const uint8_t point[32]);++/* out = secret * G, which is the same thing with the base point 9 -- but+ * where the assembly is built this reads a table instead and is four to five+ * times faster, which is what a key generation costs. */+void crypton_x25519_base(uint8_t out[32], const uint8_t secret[32]);++#endif
@@ -1462,7 +1462,10 @@ uint32_t pos = __builtin_ctz((uint32_t)current), odd = (uint32_t)current >> pos; int32_t delta = odd & mask; if (odd & 1<<(table_bits+1)) delta -= (1<<(table_bits+1));- current -= delta << pos;+ /* delta is negative half the time and shifting a negative+ * value left is undefined; current is unsigned, so the shift is+ * done there and means the same thing. */+ current -= (uint64_t)delta << pos; control[position].power = pos + 16*(w-1); control[position].addend = delta; position--;
@@ -68,7 +68,7 @@ * Expand bit 0 of the given uint8_t to a mask_t all 1 or all 0 * The input must be either 0 or 1 */-CRYPTON_DECAF_INLINE mask_t bit_to_mask(uint8_t bit) {+static CRYPTON_DECAF_INLINE mask_t bit_to_mask(uint8_t bit) { #ifdef _MSC_VER #pragma warning ( push) #pragma warning ( disable : 4146)
@@ -12,8 +12,33 @@ #include "ed25519-randombytes.h" #include "ed25519-hash.h" #include "ed25519-crypton-exts.h"+#include "ed25519/ed25519_s2n.h" /*+ The base point multiplied by a scalar, packed. s2n-bignum's assembly+ where it is built -- twice the speed of the table below, and signing+ does this twice -- and ed25519-donna's own table where it is not.+*/+static void+ed25519_base_pack(ed25519_public_key out, const bignum256modm s) {+ ge25519 ALIGN(16) p;++#if defined(CRYPTON_S2N_BIGNUM) && defined(__BYTE_ORDER__) \+ && __BYTE_ORDER__ == __ORDER_LITTLE_ENDIAN__+ unsigned char e[32];++ contract256_modm(e, s);+ if (crypton_ed25519_base_mult(out, e)) {+ memset(e, 0, sizeof e);+ return;+ }+ memset(e, 0, sizeof e);+#endif+ ge25519_scalarmult_base_niels(&p, ge25519_niels_base_multiples, s);+ ge25519_pack(out, &p);+}++/* Generates a (extsk[0..31]) and aExt (extsk[32..63]) */ @@ -38,14 +63,12 @@ void ED25519_FN(ed25519_publickey) (const ed25519_secret_key sk, ed25519_public_key pk) { bignum256modm a;- ge25519 ALIGN(16) A; hash_512bits extsk; /* A = aB */ ed25519_extsk(extsk, sk); expand256_modm(a, extsk, 32);- ge25519_scalarmult_base_niels(&A, ge25519_niels_base_multiples, a);- ge25519_pack(pk, &A);+ ed25519_base_pack(pk, a); } @@ -53,7 +76,6 @@ ED25519_FN(ed25519_sign) (const unsigned char *m, size_t mlen, const ed25519_secret_key sk, const ed25519_public_key pk, ed25519_signature RS) { ed25519_hash_context ctx; bignum256modm r, S, a;- ge25519 ALIGN(16) R; hash_512bits extsk, hashr, hram; ed25519_extsk(extsk, sk);@@ -66,8 +88,7 @@ expand256_modm(r, hashr, 64); /* R = rB */- ge25519_scalarmult_base_niels(&R, ge25519_niels_base_multiples, r);- ge25519_pack(RS, &R);+ ed25519_base_pack(RS, r); /* S = H(R,A,m).. */ ed25519_hram(hram, RS, pk, m, mlen);@@ -89,7 +110,7 @@ ge25519 ALIGN(16) R, A; hash_512bits hash; bignum256modm hram, S;- unsigned char checkR[32];+ unsigned char checkR[32], checkS[32]; if ((RS[63] & 224) || !ge25519_unpack_negative_vartime(&A, pk)) return -1;@@ -100,6 +121,11 @@ /* S */ expand256_modm(S, RS + 32, 32);++ /* check that S is canonical */+ contract256_modm(checkS, S);+ if (!ed25519_verify(RS + 32, checkS, 32))+ return -1; /* SB - H(R,A,m)A */ ge25519_double_scalarmult_vartime(&R, &A, hram, S);
@@ -0,0 +1,65 @@+/*+ * Ed25519's base point multiplication through the vendored s2n-bignum.+ *+ * Signing does this twice: once for the nonce's point R, and once for the+ * public key, which crypton derives from the secret key at every signature+ * rather than trusting the one it is handed. Both go through here, so both+ * halves of a signature move at once.+ *+ * Measured on an Apple M4, one multiplication with the encoding:+ * ed25519-donna 7.5 us against s2n-bignum 3.5.+ */+#include <string.h>++#include "ed25519/ed25519_s2n.h"++#if defined(CRYPTON_S2N_BIGNUM) && defined(__BYTE_ORDER__) \+ && __BYTE_ORDER__ == __ORDER_LITTLE_ENDIAN__+#define CRYPTON_ED25519_S2N 1+#include "crypton_cpu.h"++extern void edwards25519_scalarmulbase(uint64_t res[8],+ const uint64_t scalar[4]);+extern void edwards25519_scalarmulbase_alt(uint64_t res[8],+ const uint64_t scalar[4]);+extern void edwards25519_encode(uint8_t z[32], const uint64_t p[8]);++/* The same question as everywhere else in cbits/s2n: a microarchitecture one+ * on ARM that no feature bit answers, and exactly a feature bit on x86-64. */+static int use_alt(void)+{+#if defined(__aarch64__) || defined(__arm64__)+#ifdef __APPLE__+ return 1;+#else+ return 0;+#endif+#else+ return (crypton_x86_simd_features() & CRYPTON_X86_ADX) == 0;+#endif+}+#endif++int crypton_ed25519_base_mult(uint8_t out[32], const uint8_t scalar[32])+{+#ifdef CRYPTON_ED25519_S2N+ /* the assembly takes four little-endian 64-bit words, which is the+ * same bits as the 32 little-endian bytes the scalar is kept in */+ uint64_t s[4], p[8];++ memcpy(s, scalar, 32);+ if (use_alt())+ edwards25519_scalarmulbase_alt(p, s);+ else+ edwards25519_scalarmulbase(p, s);+ edwards25519_encode(out, p);++ memset(s, 0, sizeof s);+ memset(p, 0, sizeof p);+ return 1;+#else+ (void) out;+ (void) scalar;+ return 0;+#endif+}
@@ -0,0 +1,14 @@+#ifndef CRYPTON_ED25519_S2N_H+#define CRYPTON_ED25519_S2N_H++#include <stdint.h>++/* The base point multiplied by a scalar, packed into Ed25519's 32-byte+ * encoding. The scalar is 32 little-endian bytes, already reduced.+ *+ * Returns 1 when the vendored s2n-bignum did the work and 0 when it is not+ * built, in which case the caller multiplies the base point itself.+ */+int crypton_ed25519_base_mult(uint8_t out[32], const uint8_t scalar[32]);++#endif
@@ -83,16 +83,9 @@ const crypton_p256_int* b, crypton_p256_int* c); -// b := 1 / a % MOD-// MOD best be SECP256r1_n-void crypton_p256_modinv(- const crypton_p256_int* MOD,- const crypton_p256_int* a,- crypton_p256_int* b);--// b := 1 / a % MOD+// b := 1 / a % MOD, in time that depends on a // MOD best be SECP256r1_n-// Faster than crypton_p256_modinv()+// Answers zero for an a that has no inverse, which is zero and MOD void crypton_p256_modinv_vartime( const crypton_p256_int* MOD, const crypton_p256_int* a,@@ -130,13 +123,6 @@ void crypton_p256_base_point_mul(const crypton_p256_int *n, crypton_p256_int *out_x, crypton_p256_int *out_y);--// {out_x,out_y} := n{in_x,in_y}-void crypton_p256_point_mul(const crypton_p256_int *n,- const crypton_p256_int *in_x,- const crypton_p256_int *in_y,- crypton_p256_int *out_x,- crypton_p256_int *out_y); // {out_x,out_y} := n1G + n2{in_x,in_y} void crypton_p256_points_mul_vartime(
@@ -76,6 +76,13 @@ 0 }; static const felem kZero = {0};++/* the curve's b, in Montgomery form, for the complete addition formula */+static const felem kB = {+ 0x13897bbf, 0x9cdf622, 0x43090d8, 0x2e67c4,+ 0x176b5678, 0x2afdc84, 0xd196888, 0xb090e90,+ 0xb8600c3+}; static const felem kP = { 0x1fffffff, 0xfffffff, 0x1fffffff, 0x3ff, 0, 0, 0x200000, 0xf000000,@@ -86,96 +93,99 @@ 0, 0, 0x400000, 0xe000000, 0x1fffffff };-/* kPrecomputed contains precomputed values to aid the calculation of scalar- * multiples of the base point, G. It's actually two, equal length, tables- * concatenated.+/* kPrecomputed holds the multiples of the base point G that the comb in+ * scalar_base_mult reads. Two tables of sixteen affine points, one after the+ * other. *- * The first table contains (x,y) felem pairs for 16 multiples of the base- * point, G.+ * The comb takes five bits of the signed all-bits-set representation at a+ * time, from positions 52 apart, and the two tables are offset from each+ * other by 26: *- * Index | Index (binary) | Value- * 0 | 0000 | 0G (all zeros, omitted)- * 1 | 0001 | G- * 2 | 0010 | 2**64G- * 3 | 0011 | 2**64G + G- * 4 | 0100 | 2**128G- * 5 | 0101 | 2**128G + G- * 6 | 0110 | 2**128G + 2**64G- * 7 | 0111 | 2**128G + 2**64G + G- * 8 | 1000 | 2**192G- * 9 | 1001 | 2**192G + G- * 10 | 1010 | 2**192G + 2**64G- * 11 | 1011 | 2**192G + 2**64G + G- * 12 | 1100 | 2**192G + 2**128G- * 13 | 1101 | 2**192G + 2**128G + G- * 14 | 1110 | 2**192G + 2**128G + 2**64G- * 15 | 1111 | 2**192G + 2**128G + 2**64G + G+ * first table i, 52+i, 104+i, 156+i, 208+i+ * second table 26+i, 78+i, 130+i, 182+i, 234+i *- * The second table follows the same style, but the terms are 2**32G,- * 2**96G, 2**160G, 2**224G.+ * for i from 25 down to 0, which covers all 260 bits between them. *+ * Every digit of that representation is +-1, so a block of five teeth takes+ * one of thirty-two values -- and they come in pairs that differ only by+ * sign. So sixteen entries are enough: the top tooth is taken positive, bit+ * j of the index says that tooth j agrees with it, and where the top tooth is+ * negative the caller negates y, which costs a subtraction. Entry zero is a+ * point like any other here, unlike the unsigned table this replaces, where+ * it stood for the infinity.+ *+ * Index | Index (binary) | Value+ * 0 | 0000 | 2**208G - 2**156G - 2**104G - 2**52G - G+ * 1 | 0001 | 2**208G - 2**156G - 2**104G - 2**52G + G+ * ... | ... | ...+ * 15 | 1111 | 2**208G + 2**156G + 2**104G + 2**52G + G+ * * This is ~2KB of data. */-static const limb kPrecomputed[NLIMBS * 2 * 15 * 2] = {- 0x11522878, 0xe730d41, 0xdb60179, 0x4afe2ff, 0x12883add, 0xcaddd88, 0x119e7edc, 0xd4a6eab, 0x3120bee,- 0x1d2aac15, 0xf25357c, 0x19e45cdd, 0x5c721d0, 0x1992c5a5, 0xa237487, 0x154ba21, 0x14b10bb, 0xae3fe3,- 0xd41a576, 0x922fc51, 0x234994f, 0x60b60d3, 0x164586ae, 0xce95f18, 0x1fe49073, 0x3fa36cc, 0x5ebcd2c,- 0xb402f2f, 0x15c70bf, 0x1561925c, 0x5a26704, 0xda91e90, 0xcdc1c7f, 0x1ea12446, 0xe1ade1e, 0xec91f22,- 0x26f7778, 0x566847e, 0xa0bec9e, 0x234f453, 0x1a31f21a, 0xd85e75c, 0x56c7109, 0xa267a00, 0xb57c050,- 0x98fb57, 0xaa837cc, 0x60c0792, 0xcfa5e19, 0x61bab9e, 0x589e39b, 0xa324c5, 0x7d6dee7, 0x2976e4b,- 0x1fc4124a, 0xa8c244b, 0x1ce86762, 0xcd61c7e, 0x1831c8e0, 0x75774e1, 0x1d96a5a9, 0x843a649, 0xc3ab0fa,- 0x6e2e7d5, 0x7673a2a, 0x178b65e8, 0x4003e9b, 0x1a1f11c2, 0x7816ea, 0xf643e11, 0x58c43df, 0xf423fc2,- 0x19633ffa, 0x891f2b2, 0x123c231c, 0x46add8c, 0x54700dd, 0x59e2b17, 0x172db40f, 0x83e277d, 0xb0dd609,- 0xfd1da12, 0x35c6e52, 0x19ede20c, 0xd19e0c0, 0x97d0f40, 0xb015b19, 0x449e3f5, 0xe10c9e, 0x33ab581,- 0x56a67ab, 0x577734d, 0x1dddc062, 0xc57b10d, 0x149b39d, 0x26a9e7b, 0xc35df9f, 0x48764cd, 0x76dbcca,- 0xca4b366, 0xe9303ab, 0x1a7480e7, 0x57e9e81, 0x1e13eb50, 0xf466cf3, 0x6f16b20, 0x4ba3173, 0xc168c33,- 0x15cb5439, 0x6a38e11, 0x73658bd, 0xb29564f, 0x3f6dc5b, 0x53b97e, 0x1322c4c0, 0x65dd7ff, 0x3a1e4f6,- 0x14e614aa, 0x9246317, 0x1bc83aca, 0xad97eed, 0xd38ce4a, 0xf82b006, 0x341f077, 0xa6add89, 0x4894acd,- 0x9f162d5, 0xf8410ef, 0x1b266a56, 0xd7f223, 0x3e0cb92, 0xe39b672, 0x6a2901a, 0x69a8556, 0x7e7c0,- 0x9b7d8d3, 0x309a80, 0x1ad05f7f, 0xc2fb5dd, 0xcbfd41d, 0x9ceb638, 0x1051825c, 0xda0cf5b, 0x812e881,- 0x6f35669, 0x6a56f2c, 0x1df8d184, 0x345820, 0x1477d477, 0x1645db1, 0xbe80c51, 0xc22be3e, 0xe35e65a,- 0x1aeb7aa0, 0xc375315, 0xf67bc99, 0x7fdd7b9, 0x191fc1be, 0x61235d, 0x2c184e9, 0x1c5a839, 0x47a1e26,- 0xb7cb456, 0x93e225d, 0x14f3c6ed, 0xccc1ac9, 0x17fe37f3, 0x4988989, 0x1a90c502, 0x2f32042, 0xa17769b,- 0xafd8c7c, 0x8191c6e, 0x1dcdb237, 0x16200c0, 0x107b32a1, 0x66c08db, 0x10d06a02, 0x3fc93, 0x5620023,- 0x16722b27, 0x68b5c59, 0x270fcfc, 0xfad0ecc, 0xe5de1c2, 0xeab466b, 0x2fc513c, 0x407f75c, 0xbaab133,- 0x9705fe9, 0xb88b8e7, 0x734c993, 0x1e1ff8f, 0x19156970, 0xabd0f00, 0x10469ea7, 0x3293ac0, 0xcdc98aa,- 0x1d843fd, 0xe14bfe8, 0x15be825f, 0x8b5212, 0xeb3fb67, 0x81cbd29, 0xbc62f16, 0x2b6fcc7, 0xf5a4e29,- 0x13560b66, 0xc0b6ac2, 0x51ae690, 0xd41e271, 0xf3e9bd4, 0x1d70aab, 0x1029f72, 0x73e1c35, 0xee70fbc,- 0xad81baf, 0x9ecc49a, 0x86c741e, 0xfe6be30, 0x176752e7, 0x23d416, 0x1f83de85, 0x27de188, 0x66f70b8,- 0x181cd51f, 0x96b6e4c, 0x188f2335, 0xa5df759, 0x17a77eb6, 0xfeb0e73, 0x154ae914, 0x2f3ec51, 0x3826b59,- 0xb91f17d, 0x1c72949, 0x1362bf0a, 0xe23fddf, 0xa5614b0, 0xf7d8f, 0x79061, 0x823d9d2, 0x8213f39,- 0x1128ae0b, 0xd095d05, 0xb85c0c2, 0x1ecb2ef, 0x24ddc84, 0xe35e901, 0x18411a4a, 0xf5ddc3d, 0x3786689,- 0x52260e8, 0x5ae3564, 0x542b10d, 0x8d93a45, 0x19952aa4, 0x996cc41, 0x1051a729, 0x4be3499, 0x52b23aa,- 0x109f307e, 0x6f5b6bb, 0x1f84e1e7, 0x77a0cfa, 0x10c4df3f, 0x25a02ea, 0xb048035, 0xe31de66, 0xc6ecaa3,- 0x28ea335, 0x2886024, 0x1372f020, 0xf55d35, 0x15e4684c, 0xf2a9e17, 0x1a4a7529, 0xcb7beb1, 0xb2a78a1,- 0x1ab21f1f, 0x6361ccf, 0x6c9179d, 0xb135627, 0x1267b974, 0x4408bad, 0x1cbff658, 0xe3d6511, 0xc7d76f,- 0x1cc7a69, 0xe7ee31b, 0x54fab4f, 0x2b914f, 0x1ad27a30, 0xcd3579e, 0xc50124c, 0x50daa90, 0xb13f72,- 0xb06aa75, 0x70f5cc6, 0x1649e5aa, 0x84a5312, 0x329043c, 0x41c4011, 0x13d32411, 0xb04a838, 0xd760d2d,- 0x1713b532, 0xbaa0c03, 0x84022ab, 0x6bcf5c1, 0x2f45379, 0x18ae070, 0x18c9e11e, 0x20bca9a, 0x66f496b,- 0x3eef294, 0x67500d2, 0xd7f613c, 0x2dbbeb, 0xb741038, 0xe04133f, 0x1582968d, 0xbe985f7, 0x1acbc1a,- 0x1a6a939f, 0x33e50f6, 0xd665ed4, 0xb4b7bd6, 0x1e5a3799, 0x6b33847, 0x17fa56ff, 0x65ef930, 0x21dc4a,- 0x2b37659, 0x450fe17, 0xb357b65, 0xdf5efac, 0x15397bef, 0x9d35a7f, 0x112ac15f, 0x624e62e, 0xa90ae2f,- 0x107eecd2, 0x1f69bbe, 0x77d6bce, 0x5741394, 0x13c684fc, 0x950c910, 0x725522b, 0xdc78583, 0x40eeabb,- 0x1fde328a, 0xbd61d96, 0xd28c387, 0x9e77d89, 0x12550c40, 0x759cb7d, 0x367ef34, 0xae2a960, 0x91b8bdc,- 0x93462a9, 0xf469ef, 0xb2e9aef, 0xd2ca771, 0x54e1f42, 0x7aaa49, 0x6316abb, 0x2413c8e, 0x5425bf9,- 0x1bed3e3a, 0xf272274, 0x1f5e7326, 0x6416517, 0xea27072, 0x9cedea7, 0x6e7633, 0x7c91952, 0xd806dce,- 0x8e2a7e1, 0xe421e1a, 0x418c9e1, 0x1dbc890, 0x1b395c36, 0xa1dc175, 0x1dc4ef73, 0x8956f34, 0xe4b5cf2,- 0x1b0d3a18, 0x3194a36, 0x6c2641f, 0xe44124c, 0xa2f4eaa, 0xa8c25ba, 0xf927ed7, 0x627b614, 0x7371cca,- 0xba16694, 0x417bc03, 0x7c0a7e3, 0x9c35c19, 0x1168a205, 0x8b6b00d, 0x10e3edc9, 0x9c19bf2, 0x5882229,- 0x1b2b4162, 0xa5cef1a, 0x1543622b, 0x9bd433e, 0x364e04d, 0x7480792, 0x5c9b5b3, 0xe85ff25, 0x408ef57,- 0x1814cfa4, 0x121b41b, 0xd248a0f, 0x3b05222, 0x39bb16a, 0xc75966d, 0xa038113, 0xa4a1769, 0x11fbc6c,- 0x917e50e, 0xeec3da8, 0x169d6eac, 0x10c1699, 0xa416153, 0xf724912, 0x15cd60b7, 0x4acbad9, 0x5efc5fa,- 0xf150ed7, 0x122b51, 0x1104b40a, 0xcb7f442, 0xfbb28ff, 0x6ac53ca, 0x196142cc, 0x7bf0fa9, 0x957651,- 0x4e0f215, 0xed439f8, 0x3f46bd5, 0x5ace82f, 0x110916b6, 0x6db078, 0xffd7d57, 0xf2ecaac, 0xca86dec,- 0x15d6b2da, 0x965ecc9, 0x1c92b4c2, 0x1f3811, 0x1cb080f5, 0x2d8b804, 0x19d1c12d, 0xf20bd46, 0x1951fa7,- 0xa3656c3, 0x523a425, 0xfcd0692, 0xd44ddc8, 0x131f0f5b, 0xaf80e4a, 0xcd9fc74, 0x99bb618, 0x2db944c,- 0xa673090, 0x1c210e1, 0x178c8d23, 0x1474383, 0x10b8743d, 0x985a55b, 0x2e74779, 0x576138, 0x9587927,- 0x133130fa, 0xbe05516, 0x9f4d619, 0xbb62570, 0x99ec591, 0xd9468fe, 0x1d07782d, 0xfc72e0b, 0x701b298,- 0x1863863b, 0x85954b8, 0x121a0c36, 0x9e7fedf, 0xf64b429, 0x9b9d71e, 0x14e2f5d8, 0xf858d3a, 0x942eea8,- 0xda5b765, 0x6edafff, 0xa9d18cc, 0xc65e4ba, 0x1c747e86, 0xe4ea915, 0x1981d7a1, 0x8395659, 0x52ed4e2,- 0x87d43b7, 0x37ab11b, 0x19d292ce, 0xf8d4692, 0x18c3053f, 0x8863e13, 0x4c146c0, 0x6bdf55a, 0x4e4457d,- 0x16152289, 0xac78ec2, 0x1a59c5a2, 0x2028b97, 0x71c2d01, 0x295851f, 0x404747b, 0x878558d, 0x7d29aa4,- 0x13d8341f, 0x8daefd7, 0x139c972d, 0x6b7ea75, 0xd4a9dde, 0xff163d8, 0x81d55d7, 0xa5bef68, 0xb7b30d8,- 0xbe73d6f, 0xaa88141, 0xd976c81, 0x7e7a9cc, 0x18beb771, 0xd773cbd, 0x13f51951, 0x9d0c177, 0x1c49a78,+static const limb kPrecomputed[NLIMBS * 2 * 16 * 2] = {+ 0xe01bd76, 0xa0be8b3, 0x8494c1d, 0x609ab3d, 0x1188042f, 0x499c03d, 0x1df7cd26, 0x51b33c5, 0x1fb3bce,+ 0x39cdd45, 0xdc0dd9b, 0xe3053d7, 0x1ffaf46, 0x9ac284a, 0xac051d4, 0x1c09fe1b, 0x8227cbf, 0x5bf049b,+ 0xb9487d, 0x2ecb75f, 0x194825bf, 0xd70cf28, 0x14e528f3, 0x4d8670c, 0x35bbabb, 0x6b692ca, 0xd96d08,+ 0x1db081dc, 0xa87ea7b, 0x190e5549, 0xa4cf420, 0x1e151385, 0xaf3d4bd, 0x4057e9f, 0x5078feb, 0x154519a,+ 0xbf15dea, 0x453fa25, 0x1171c85a, 0x576c824, 0x154e7060, 0x71ede6e, 0x160467a2, 0xdea8a44, 0x81ffcb1,+ 0x61a56fa, 0x76119b9, 0x110bfb9b, 0x3d527ea, 0x1997bdb4, 0xe1d1253, 0x180ce91c, 0x11950ee, 0x53d5938,+ 0x694e7c9, 0xe0cf337, 0x16d8ae50, 0x202517f, 0x4d02e16, 0xd13b5fd, 0xfae97eb, 0xa1c7f60, 0x1206fe8,+ 0x11b1c908, 0xf8a82f, 0x6ab17a0, 0x48058e8, 0x2d0feb, 0xfada550, 0x658edb9, 0xa17567a, 0x8daa44d,+ 0x6361dc9, 0xec00c0e, 0x151a7b1c, 0xb35a683, 0x1643fe02, 0x70155c0, 0x1f131d45, 0x3998068, 0x25beef8,+ 0x88138de, 0x8995ce4, 0x18565c50, 0x60f12b6, 0xc47a656, 0x4a82bd9, 0xb547a17, 0xb333474, 0xe86513a,+ 0x7f8d2bc, 0x7f0f16c, 0x8cde475, 0x1ac3d5c, 0x1a832c9a, 0x6a93e7a, 0x19833281, 0xcec82db, 0x4f08cc,+ 0x72c4394, 0x4686520, 0x1e845ce, 0xfb181a1, 0xf5135a5, 0xa1265d6, 0x6c63ce8, 0xe81797e, 0x5dbcd5a,+ 0x2a5d603, 0x1ad4e91, 0xc86e1b3, 0x793abea, 0x1b8610a4, 0x8d5b975, 0x74dd850, 0xbbca81, 0xd7c35d8,+ 0x1bab7afc, 0x4df749, 0x4acb4ea, 0xfae8c89, 0x14552ae5, 0x6dc1c20, 0x14f629f, 0x2368fe5, 0xe5a9cba,+ 0x67576f7, 0x9c77c50, 0x1d63c92a, 0xbbb9ef8, 0xa7530d4, 0x963335, 0xfa09c54, 0xb6d03e3, 0xed1a022,+ 0x19c59f49, 0x45c823d, 0x17a28df1, 0x80ae516, 0xe2ada82, 0x19b97fb, 0x13a9ebf, 0xf1e7606, 0xde03632,+ 0x14e318b9, 0x7b57b83, 0x51a9a92, 0x3378a17, 0x1cde9289, 0x45956c2, 0x2bbab5e, 0x780b1f5, 0x8356034,+ 0x75eca28, 0x6f39648, 0xf2fdbda, 0x4c65cd9, 0xb3759e7, 0x710c0b1, 0xda24432, 0x8d236aa, 0xf4c449f,+ 0xaf9ba24, 0xb83ea7f, 0x1e46ecc3, 0x3f277b0, 0x6acdecd, 0x1f597d1, 0x8483d72, 0x57d29dd, 0x66d4060,+ 0x167c14af, 0xc142ded, 0xc1689ca, 0x651aa52, 0x8d05768, 0x9709cfa, 0x165fd283, 0x9f6f583, 0x9833b54,+ 0xd17e2d6, 0x2915c32, 0x1970ef24, 0xe8ca45b, 0x11c29e09, 0x8121f26, 0xb5afbe1, 0x10c80ec, 0x834a25c,+ 0xa37f0b0, 0xd6bac16, 0xed484fc, 0x8799206, 0x13db8bb1, 0xdce615c, 0x13320329, 0x79dc25, 0x914e7af,+ 0x860a414, 0xb8a9434, 0x50396ce, 0x902d3e6, 0x15c152f3, 0x3753c64, 0x970c055, 0xba296fb, 0x64bd63b,+ 0x118cbb94, 0x9d4274f, 0x121cdbfe, 0xb9137cf, 0xc8bddf1, 0xa1598d0, 0x446ab41, 0x11f1df2, 0x68115f1,+ 0x2f1f708, 0xee35192, 0x1dfeb3fc, 0x4e1e1a6, 0x1d9adcbb, 0x6688662, 0x63ad21b, 0x9d9a9e1, 0xb0f8b4b,+ 0xcd8bc3a, 0x3577d8, 0x1ffcb97d, 0xd31e8d6, 0x1c776310, 0x95b4ef7, 0x185a82ed, 0xe40bbb0, 0xbca1ea,+ 0x19462e0b, 0x2252179, 0x14f14f09, 0x565e68d, 0x7ba5f37, 0x4cd1858, 0x167941b3, 0x4d1c7a7, 0x7aef1ab,+ 0x1599efe9, 0x658e78d, 0x1ad33917, 0x7e74797, 0x19152edc, 0xdf7dc18, 0xdf677c, 0x9315d96, 0x4ba8eec,+ 0xc45cd82, 0x1acfa03, 0xe265b49, 0xcfa6eb, 0x89d7619, 0x7b05, 0x1ba11068, 0xe1672d3, 0x655622a,+ 0x1607ca6, 0x339049, 0x5454f70, 0x10edd75, 0x1133ceb7, 0xe3eec39, 0xc263156, 0xeb6ddbe, 0x10360a7,+ 0xfd5f981, 0x47dd502, 0x1e66dbbe, 0x8820b63, 0xeee91ef, 0xffde293, 0xb66325d, 0x3a5a2a, 0x6a2acbb,+ 0x11e949bc, 0xf6a6d57, 0x6b2ca24, 0xad903ec, 0x1e55de0, 0xe7074ed, 0xe681934, 0xea44010, 0xaa490cc,+ 0x512324a, 0x6a5eb00, 0xee5e100, 0xd60a02b, 0x1b89c993, 0x5cffb70, 0xa49030c, 0x405aee, 0xe1b27cc,+ 0x2e73a15, 0x9ca8dc0, 0x781dbff, 0x9fd85e0, 0x1884e4c8, 0x40873f6, 0x32d4b69, 0xf42f753, 0xeaf4c38,+ 0x2802a4c, 0x1283dac, 0x759100a, 0xcb3ba75, 0xf3203d3, 0xf89b8aa, 0x7caa59e, 0x384a60a, 0x37f69e7,+ 0x1e56d569, 0x120c552, 0x181734aa, 0x4fcd9b4, 0x7918e4e, 0xcd7938a, 0x2bbd8b2, 0x19f4f97, 0xa7af533,+ 0xbb69948, 0x3eff33e, 0x1f3ac118, 0x3739770, 0x58898fd, 0x623fafb, 0xa7e6d93, 0xd31a676, 0x615192d,+ 0xf543d28, 0xe61ce5a, 0x10ae4b39, 0xcd5a8d7, 0x1b34c6de, 0x81997ad, 0x198e2093, 0xd9b6ff7, 0x9a5954f,+ 0x16782589, 0xed3c1ab, 0x62dc4a5, 0xac12d0c, 0x8bc8b7c, 0x168ec4e, 0x177e11dc, 0x407df09, 0x4056e85,+ 0x9858305, 0xfe445e9, 0x18b1230a, 0x815ee9f, 0xa852eb4, 0x89444e0, 0x1a83481, 0x479359b, 0x2192576,+ 0x16d5e61b, 0xfe480c4, 0x1d60b8b7, 0x1c6e798, 0x1a01310, 0x9998572, 0x7c59f75, 0x49dda87, 0xe75e0b6,+ 0x1b2b7536, 0xb267d8, 0x15443085, 0x45e5924, 0x7fb947f, 0x296915d, 0x38fc56b, 0x4bae39f, 0xb218e7c,+ 0x115c3b16, 0x9d95f0c, 0xa50ac4c, 0xcce8037, 0xc3c7ba3, 0xf02773a, 0xbc2ad54, 0x26914c1, 0x19a4b8d,+ 0x3f8a1a6, 0xa0b8459, 0x1f77a521, 0x7d93297, 0x1dddb4b2, 0x9e4cd1c, 0x6e28403, 0xd7ce413, 0x5575b62,+ 0x12bb7dc1, 0xbdfb15e, 0xa542867, 0xe943d3e, 0x1367fbdd, 0x37b387, 0x14e4d75f, 0xb90b09d, 0xbf6ec28,+ 0xa5182c8, 0x1e5b34e, 0xabe4602, 0x1c13efa, 0x8d1182, 0x1c2947a, 0x1e04e0e3, 0x6caecdb, 0x40f14e8,+ 0x1b845e4a, 0xa9fb149, 0xb34f513, 0x3a0fdbb, 0xfad2335, 0x5bb9342, 0x18c5ad62, 0xc97fdc3, 0x31225c7,+ 0xb28a9ee, 0x585915, 0x1e355da1, 0x26ed08e, 0xc7d06a, 0xa65f219, 0x1fdf45cd, 0xc8323ea, 0x297b9ee,+ 0x1c031098, 0x9c39cf0, 0x1287d79f, 0x69a9e32, 0x10015650, 0x1c3dfc3, 0x12dcd848, 0x3155a59, 0xeff3212,+ 0x1a6ecdd0, 0xd26bd07, 0x18e077ab, 0x442b477, 0x46b735f, 0x495d60c, 0x1b57a6e5, 0x76a368a, 0xf53bd50,+ 0xf12c5e0, 0x80a9b4, 0x15562060, 0x4102113, 0xa144ab5, 0x5fa4e9, 0x1009f5e9, 0xe34343a, 0x26fda5a,+ 0x159e06a4, 0x5fa3aad, 0x10259b5f, 0xa69947d, 0x1190417e, 0x987da3f, 0x14e1e868, 0xdcb7e1e, 0x8890f9f,+ 0x14dece80, 0x94bbf5c, 0x18513a17, 0x4ca31ca, 0xc0a2713, 0xbc46074, 0x1536f6a5, 0x43991aa, 0xb9f8f1c,+ 0x987ba48, 0xb0829ae, 0xd29d324, 0x6339c35, 0x18ace0a4, 0x5d53b55, 0xff829f4, 0xe882ecf, 0xc05164c,+ 0x6bd6bba, 0x9dfc14f, 0x1981cab3, 0xcfebf18, 0x1ddac868, 0x94ec6d4, 0x5abd4b, 0x737fdb3, 0x18f531f,+ 0xd3c2a71, 0x337178f, 0x9f7c32e, 0xd9d7fda, 0x137d191f, 0xdd0757b, 0x14b6d65, 0x179f37a, 0x67b10e1,+ 0x1bfb2cfd, 0xfe4ca43, 0x1fee2930, 0x98c2aa0, 0x9826788, 0xeaf4ceb, 0x17a6be82, 0xc899ed1, 0x500fb01,+ 0x10918e6f, 0x36179ed, 0xbca6643, 0x2d80942, 0xf1ef61, 0xadca21c, 0xbe5b3a7, 0xadae157, 0x12daac,+ 0x1337dda8, 0x6326e5d, 0x2738e1b, 0x5cb5c54, 0x98ec8a0, 0x252647d, 0x1d5c173c, 0xbdf848d, 0x9e5217b,+ 0x1d64f447, 0xb71d1a, 0xb2c2360, 0xccb6bee, 0x1245995e, 0x94a9130, 0x5b93d91, 0x76c57ff, 0xeaa91d1,+ 0x3941881, 0xc2aafbd, 0x1c0540d0, 0x1a938f0, 0x1304b724, 0x8524e10, 0x1bef780f, 0xbc0ea48, 0xbe90dae,+ 0x1d10d5d8, 0xca979e2, 0x10db5cc4, 0x54e2493, 0x44d38f3, 0xbcb73b, 0x12dcff4, 0xd0ab219, 0xde69db2,+ 0x13594366, 0xc30e05f, 0xfc245d4, 0x8c5b52f, 0x81901c7, 0xa9d1e03, 0x11ead62e, 0xb7be89b, 0xc9c8486,+ 0x132a6fa0, 0x56af9b8, 0x41cb561, 0xf74418c, 0x141c461a, 0xbc18514, 0x1d6bbb68, 0x96d43c2, 0x7108696 };
@@ -83,16 +83,9 @@ const crypton_p256_int* b, crypton_p256_int* c); -// b := 1 / a % MOD-// MOD best be SECP256r1_n-void crypton_p256_modinv(- const crypton_p256_int* MOD,- const crypton_p256_int* a,- crypton_p256_int* b);--// b := 1 / a % MOD+// b := 1 / a % MOD, in time that depends on a // MOD best be SECP256r1_n-// Faster than crypton_p256_modinv()+// Answers zero for an a that has no inverse, which is zero and MOD void crypton_p256_modinv_vartime( const crypton_p256_int* MOD, const crypton_p256_int* a,@@ -130,13 +123,6 @@ void crypton_p256_base_point_mul(const crypton_p256_int *n, crypton_p256_int *out_x, crypton_p256_int *out_y);--// {out_x,out_y} := n{in_x,in_y}-void crypton_p256_point_mul(const crypton_p256_int *n,- const crypton_p256_int *in_x,- const crypton_p256_int *in_y,- crypton_p256_int *out_x,- crypton_p256_int *out_y); // {out_x,out_y} := n1G + n2{in_x,in_y} void crypton_p256_points_mul_vartime(
@@ -63,6 +63,38 @@ #define NLIMBS 5 typedef limb felem[NLIMBS]; +/* On AArch64, the three functions that do the field arithmetic are asked to+ * be inlined rather than left for the compiler to decide.+ *+ * felem_mul and felem_square end in felem_reduce_degree, a carry chain the+ * whole width of the number, and that chain is what their latency is: one+ * product feeding the next costs 18.1 ns on an Apple M4, while four+ * independent ones cost 11.4 ns each. The curve arithmetic has independent+ * products to offer -- the two squarings that open a point doubling, the+ * multiplication and the squaring that close it -- but only if the compiler+ * can see one reduction while the other is still going. Left alone it emits+ * felem_reduce_degree once and calls it, and a call is a fence: the two+ * chains cannot overlap. Plain `inline` does not change its mind.+ *+ * Asking costs code: this file's object goes from 30 to 116 kilobytes. That+ * is worth it where there are registers to hold two chains at once and not+ * where there are not, which is the architecture talking rather than the+ * compiler. Measured on a variable-point scalar multiplication:+ *+ * Apple M4, Apple clang 21 1.23x+ * Neoverse, clang 18 1.12x+ * Neoverse, gcc 13 1.05x+ * EPYC 7763, clang 18 0.95x+ * Xeon 8370C, gcc 13 0.82x+ *+ * so x86-64 keeps the compiler's own judgement.+ */+#if defined(__aarch64__) && (defined(__GNUC__) || defined(__clang__))+#define FELEM_INLINE static inline __attribute__((always_inline))+#else+#define FELEM_INLINE static+#endif+ static const limb kBottom51Bits = 0x7ffffffffffff; static const limb kBottom52Bits = 0xfffffffffffff; @@ -72,102 +104,111 @@ 2, 0xfc00000000000, 0x7ffffffffffff, 0xfff7fffffffff, 0x7ffff }; static const felem kZero = {0};++/* the curve's b, in Montgomery form, for the complete addition formula */+static const felem kB = {+ 0x1bec453897bbf, 0x33e210c243627, 0x484bb5ab3c017, 0x41a32d11055fb,+ 0x2e18030ec243a+}; static const felem kP = { 0x7ffffffffffff, 0x1fffffffffff, 0, 0x4000000000, 0x3fffffffc0000 }; static const felem k2P = { 0x7fffffffffffe, 0x3fffffffffff, 0, 0x8000000000, 0x7fffffff80000 };-/* kPrecomputed contains precomputed values to aid the calculation of scalar- * multiples of the base point, G. It's actually two, equal length, tables- * concatenated.+/* kPrecomputed holds the multiples of the base point G that the comb in+ * scalar_base_mult reads. Two tables of sixteen affine points, one after the+ * other. *- * The first table contains (x,y) felem pairs for 16 multiples of the base- * point, G.+ * The comb takes five bits of the signed all-bits-set representation at a+ * time, from positions 52 apart, and the two tables are offset from each+ * other by 26: *- * Index | Index (binary) | Value- * 0 | 0000 | 0G (all zeros, omitted)- * 1 | 0001 | G- * 2 | 0010 | 2**64G- * 3 | 0011 | 2**64G + G- * 4 | 0100 | 2**128G- * 5 | 0101 | 2**128G + G- * 6 | 0110 | 2**128G + 2**64G- * 7 | 0111 | 2**128G + 2**64G + G- * 8 | 1000 | 2**192G- * 9 | 1001 | 2**192G + G- * 10 | 1010 | 2**192G + 2**64G- * 11 | 1011 | 2**192G + 2**64G + G- * 12 | 1100 | 2**192G + 2**128G- * 13 | 1101 | 2**192G + 2**128G + G- * 14 | 1110 | 2**192G + 2**128G + 2**64G- * 15 | 1111 | 2**192G + 2**128G + 2**64G + G+ * first table i, 52+i, 104+i, 156+i, 208+i+ * second table 26+i, 78+i, 130+i, 182+i, 234+i *- * The second table follows the same style, but the terms are 2**32G,- * 2**96G, 2**160G, 2**224G.+ * for i from 25 down to 0, which covers all 260 bits between them. *+ * Every digit of that representation is +-1, so a block of five teeth takes+ * one of thirty-two values -- and they come in pairs that differ only by+ * sign. So sixteen entries are enough: the top tooth is taken positive, bit+ * j of the index says that tooth j agrees with it, and where the top tooth is+ * negative the caller negates y, which costs a subtraction. Entry zero is a+ * point like any other here, unlike the unsigned table this replaces, where+ * it stood for the infinity.+ *+ * Index | Index (binary) | Value+ * 0 | 0000 | 2**208G - 2**156G - 2**104G - 2**52G - G+ * 1 | 0001 | 2**208G - 2**156G - 2**104G - 2**52G + G+ * ... | ... | ...+ * 15 | 1111 | 2**208G + 2**156G + 2**104G + 2**52G + G+ * * This is ~2KB of data. */-static const limb kPrecomputed[NLIMBS * 2 * 15 * 2] = {- 0x661a831522878, 0xf17fb6d805e79, 0x5889441d6ea57, 0xae33cfdb995bb, 0xc482fbb529ba,- 0x4a6af9d2aac15, 0x90e867917377c, 0x487cc962d2ae3, 0xec2a97443446e, 0x2b8ff8c52c42,- 0x45f8a2d41a576, 0xb06988d2653e4, 0x718b22c357305, 0x33fc920e79d2b, 0x17af34b0fe8db,- 0x38e17eb402f2f, 0x3382558649705, 0x47f6d48f482d1, 0x7bd42488d9b83, 0x3b247c8b86b78,- 0x4d08fc26f7778, 0x7a29a82fb2795, 0x75cd18f90d11a, 0xad8e213b0bc, 0x2d5f0142899e8,- 0x506f98098fb57, 0x2f0c98301e4aa, 0x39b30dd5cf67d, 0x9c146498ab13c, 0xa5db92df5b7b,- 0x184897fc4124a, 0xe3f73a19d8aa, 0x4e1c18e47066b, 0x27b2d4b52eaee, 0x30eac3ea10e99,- 0x4e74546e2e7d5, 0x1f4dde2d97a1d, 0x6ead0f88e1200, 0x7dec87c220f02, 0x3d08ff096310f,- 0x23e5659633ffa, 0x6ec648f08c722, 0x3172a3806ea35, 0xf6e5b681eb3c5, 0x2c3758260f89d,- 0x38dca4fd1da12, 0xf06067b78830d, 0x3194be87a068c, 0x78893c7eb602b, 0xcead60438432,- 0x6ee69a56a67ab, 0xd886f77701895, 0x67b0a4d9cee2b, 0x3586bbf3e4d53, 0x1db6f32921d93,- 0x260756ca4b366, 0x4f40e9d2039fa, 0x4f3f09f5a82bf, 0xccde2d641e8cd, 0x305a30cd2e8c5,- 0x471c235cb5439, 0xab279cd962f5a, 0x17e1fb6e2dd94, 0xfe64589800a77, 0xe8793d99775f,- 0x48c62f4e614aa, 0xbf76ef20eb2a4, 0x669c672556c, 0x24683e0eff056, 0x12252b369ab76,- 0x821de9f162d5, 0xf911ec99a95be, 0x6721f065c906b, 0x58d452035c736, 0x1f9f01a6a15,- 0x6135009b7d8d3, 0xdaeeeb417dfc0, 0x63865fea0ee17, 0x6e0a304b939d6, 0x204ba2076833d,- 0x4ade586f35669, 0x2c1077e34611a, 0x5b1a3bea3b81a, 0xf97d018a22c8b, 0x38d7996b08af8,- 0x6ea62baeb7aa0, 0xebdcbd9ef2670, 0x35dc8fe0df3fe, 0xe458309d20c24, 0x11e87898716a0,- 0x7c44bab7cb456, 0xd64d3cf1bb64, 0x189bff1bf9e66, 0xb5218a049311, 0x285dda6cbcc81,- 0x3238dcafd8c7c, 0x607736c8de0, 0xdb83d99508b1, 0x4e1a0d404cd81, 0x1588008c00ff2,- 0x16b8b36722b27, 0x876609c3f3f1a, 0x66b72ef0e17d6, 0x705f8a279d568, 0x2eaac4cd01fdd,- 0x1171ce9705fe9, 0xffc79cd3264ee, 0x700c8ab4b80f0, 0x208d3d4f57a1, 0x337262a8ca4eb,- 0x297fd01d843fd, 0xa90956fa097f8, 0x529759fdb3845, 0x1d78c5e2d0397, 0x3d6938a4adbf3,- 0x16d5853560b66, 0xf138946b9a430, 0x2ab79f4dea6a0, 0xd42053ee43ae1, 0x3b9c3ef1cf870,- 0x598934ad81baf, 0x5f1821b1d07a7, 0x416bb3a973ff3, 0x23f07bd0a047a, 0x19bdc2e09f786,- 0x56dc9981cd51f, 0xfbace23c8cd65, 0x673bd3bf5b52e, 0x46a95d229fd61, 0xe09ad64bcfb1,- 0xe5292b91f17d, 0xfeefcd8afc287, 0x58f52b0a58711, 0x4800f20c201ef, 0x2084fce608f67,- 0x12ba0b128ae0b, 0x5977ae17030b4, 0x101126ee420f6, 0xf70823495c6bd, 0xde19a27d7770,- 0x5c6ac852260e8, 0x9d22950ac4356, 0x441cca955246c, 0x660a34e5332d9, 0x14ac8ea92f8d2,- 0x6b6d7709f307e, 0x67d7e13879db, 0x2ea8626f9fbbd, 0x99609006a4b40, 0x31bb2a8f8c779,- 0x10c04828ea335, 0xae9acdcbc080a, 0x617af2342607a, 0xc7494ea53e553, 0x2ca9e2872defa,- 0x6c399fab21f1f, 0xab139b245e758, 0x3ad933dcba589, 0x4797fecb08811, 0x31f5dbf8f594,- 0x7dc6361cc7a69, 0xc8a7953ead3f9, 0x79ed693d18015, 0x418a024999a6a, 0x2c4fdc9436aa,- 0x1eb98cb06aa75, 0x2989592796a9c, 0x11194821e425, 0xe27a648228388, 0x35d834b6c12a0,- 0x541807713b532, 0x7ae0a1008aaee, 0x7017a29bcb5e, 0x6b193c23c315c, 0x19bd25ac82f2a,- 0x6a01a43eef294, 0xddf5b5fd84f19, 0x33f5ba081c016, 0xdeb052d1bc082, 0x6b2f06afa617,- 0x7ca1eda6a939f, 0xbdeb35997b50c, 0x47f2d1bccda5, 0xc2ff4adfed667, 0x87712997be4,- 0x21fc2e2b37659, 0xf7d62cd5ed951, 0x27fa9cbdf7efa, 0xba25582bf3a6b, 0x2a42b8bd89398,- 0x6d377d07eecd2, 0x9ca1df5af387, 0x1109e3427e2ba, 0xce4aa4572a19, 0x103baaef71e16,- 0x2c3b2dfde328a, 0xbec4b4a30e1ef, 0x37d92a86204f3, 0x806cfde68eb39, 0x246e2f72b8aa5,- 0x68d3de93462a9, 0x53b8acba6bbc3, 0x2492a70fa1696, 0x38c62d5760f55, 0x15096fe4904f2,- 0x4e44e9bed3e3a, 0xb28bfd79cc9bc, 0x6a77513839320, 0x480dcec6739db, 0x3601b739f2465,- 0x43c348e2a7e1, 0xe448106327879, 0x175d9cae1b0ed, 0xd3b89dee743b8, 0x392d73ca255bc,- 0x32946db0d3a18, 0x9261b09907cc, 0x5ba517a755722, 0x51f24fdaf5184, 0x1cdc732989ed8,- 0x2f7806ba16694, 0xae0c9f029f8d0, 0xd8b45102ce1, 0xca1c7db9316d6, 0x162088a67066f,- 0x39de35b2b4162, 0xa19f550d88ae9, 0x7921b27026cde, 0x94b936b66e900, 0x1023bd5fa17fc,- 0x436837814cfa4, 0x29113492283c4, 0x66d1cdd8b51d8, 0xa540702278eb2, 0x47ef1b29285d,- 0x587b50917e50e, 0xb4cda75bab3b, 0x112520b0a9886, 0x66b9ac16fee49, 0x17bf17e92b2eb,- 0x2456a2f150ed7, 0xfa214412d0280, 0x3ca7dd947fe5b, 0xa72c28598d58a, 0x255d945efc3e,- 0x2873f04e0f215, 0x74178fd1af57b, 0x788848b5b2d6, 0xb1ffafaae0db6, 0x32a1b7b3cbb2a,- 0x4bd9935d6b2da, 0x9c08f24ad30a5, 0x4e58407a80f, 0x1b3a3825a5b17, 0x6547e9fc82f5,- 0x47484aa3656c3, 0x6ee43f341a494, 0x64a98f87adea2, 0x619b3f8e95f01, 0xb6e513266ed8,- 0x421c2a673090, 0xa1c1de32348c7, 0x55b85c3a1e8a3, 0xe05ce8ef330b4, 0x2561e49c15d84,- 0x40aa2d33130fa, 0x12b827d35866f, 0xfe4cf62c8ddb, 0x2fa0ef05bb28d, 0x1c06ca63f1cb8,- 0x32a971863863b, 0xff6fc86830da1, 0x71e7b25a14cf3, 0xea9c5ebb1373a, 0x250bbaa3e1634,- 0x5b5ffeda5b765, 0xf25d2a746331b, 0x115e3a3f43632, 0x67303af43c9d5, 0x14bb538a0e559,- 0x75623687d43b7, 0xa349674a4b38d, 0x613c61829ffc6, 0x689828d8110c7, 0x139115f5af7d5,- 0xf1d856152289, 0x45cbe967168ab, 0x51f38e1680901, 0x34808e8f652b0, 0x1f4a6a921e156,- 0x35dfaf3d8341f, 0xf53ace725cb63, 0x3d86a54eef35b, 0xa103aabaffe2c, 0x2decc36296fbd,- 0x510282be73d6f, 0xd4e6365db206a, 0x4bdc5f5bb8bf3, 0xde7ea32a3aee7, 0x71269e274305,+static const limb kPrecomputed[NLIMBS * 2 * 16 * 2] = {+ 0x17d166e01bd76, 0xd59ea12530768, 0x3d8c40217b04, 0x17bef9a4c9338, 0x7ecef3946ccf,+ 0x1bb3639cdd45, 0xd7a338c14f5f7, 0x1d44d614250ff, 0xff813fc37580a, 0x16fc126e089f2,+ 0x596ebe0b9487d, 0x6794652096fcb, 0x70ca729479eb8, 0x286b775769b0c, 0x365b421ada4b,+ 0xfd4f7db081dc, 0x7a1064395526a, 0x4bdf0a89c2d26, 0xac80afd3f5e7a, 0x551466941e3f,+ 0x27f44abf15dea, 0x641245c721691, 0x66eaa738302bb, 0x12c08cf44e3db, 0x207ff2c77aa29,+ 0x42337261a56fa, 0x93f5442fee6dd, 0x253ccbdeda1ea, 0xbb019d239c3a2, 0x14f564e046543,+ 0x19e66e694e7c9, 0x28bfdb62b9438, 0x5fd268170b101, 0x81f5d2fd7a276, 0x481bfa2871fd,+ 0x71505f1b1c908, 0x2c741aac5e803, 0x55001687f5a40, 0xe8cb1db73f5b4, 0x236a913685d59,+ 0x181c6361dc9, 0xd341d469ec73b, 0x5c0b21ff0159a, 0xa3e263a8ae02a, 0x96fbbe0e6601,+ 0x32b9c888138de, 0x895b615971422, 0x3d9623d32b307, 0xd16a8f42e9505, 0x3a1944eacccd1,+ 0x61e2d87f8d2bc, 0x1eae233791d5f, 0x67ad41964d0d6, 0x6f3066502d527, 0x13c23333b20b,+ 0x50ca4072c4394, 0xc0d087a117391, 0x5d67a89ad2fd8, 0xf8d8c79d1424c, 0x176f356ba05e5,+ 0x5a9d222a5d603, 0xd5f5321b86cc6, 0x175dc308523c9, 0x4e9bb0a11ab7, 0x35f0d7602ef2a,+ 0x1bee93bab7afc, 0x464492b2d3a81, 0x420a2a9572fd7, 0x9429ec53edb83, 0x396a72e88da3f,+ 0xef8a067576f7, 0xcf7c758f24aa7, 0x33553a986a5dd, 0x8df4138a812c6, 0x3b46808adb40f,+ 0x39047b9c59f49, 0x728b5e8a37c51, 0x7fb7156d41405, 0x182753d7e3372, 0x3780d8cbc79d8,+ 0x6af7074e318b9, 0xc50b946a6a49e, 0x6c2e6f494499b, 0xd457756bc8b2a, 0x20d580d1e02c7,+ 0x672c9075eca28, 0x2e6cbcbf6f69b, 0xb159bacf3a63, 0xa9b448864e218, 0x3d31127e348da,+ 0x7d4feaf9ba24, 0x3bd8791bb30ee, 0x7d13566f669f9, 0x750907ae43eb2, 0x19b501815f4a7,+ 0x285bdb67c14af, 0xd529305a272b0, 0x4fa4682bb4328, 0xecbfa5072e13, 0x260ced527dbd6,+ 0x22b864d17e2d6, 0x522de5c3bc90a, 0x7268e14f04f46, 0xb16b5f7c30243, 0x20d2897043203,+ 0x57582ca37f0b0, 0xc9033b5213f35, 0x15c9edc5d8c3c, 0x966640653b9cc, 0x24539ebc1e770,+ 0x152868860a414, 0x69f3140e5b3ae, 0x464ae0a979c81, 0xed2e180aa6ea7, 0x192f58eee8a5b,+ 0x284e9f18cbb94, 0x9be7c8736ffa7, 0xd0645eef8dc8, 0xc888d568342b3, 0x1a0457c447c77,+ 0x46a3242f1f708, 0xf0d377facff3b, 0x662ecd6e5da70, 0x84c75a436cd10, 0x2c3e2d2e766a7,+ 0x6aefb0cd8bc3a, 0xf46b7ff2e5f40, 0x6f7e3bb188698, 0xc30b505db2b69, 0x2f287ab902ee,+ 0x4a42f39462e0b, 0xf346d3c53c248, 0x583dd2f9bab2, 0x9ecf2836699a3, 0x1ebbc6ad3471e,+ 0x31cf1b599efe9, 0xa3cbeb4ce45d9, 0x418c8a976e3f3, 0x581becef9befb, 0x12ea3bb24c576,+ 0x59f406c45cd82, 0xd375b8996d246, 0x30544ebb0c867, 0x4f74220d0000f, 0x195588ab859cb,+ 0x6720921607ca6, 0x6eba95153dc00, 0x439899e75b887, 0xf984c62adc7dd, 0x40d829fadb76,+ 0x7baa04fd5f981, 0x5b1f99b6ef91, 0x29377748f7c41, 0xa96cc64bbffbc, 0x1a8ab2ec0e968,+ 0x54daaf1e949bc, 0x81f61acb2893d, 0x4ed0f2aef056c, 0x41cd03269ce0e, 0x2a924333a9100,+ 0x4bd600512324a, 0x5015bb978401a, 0x370dc4e4c9eb0, 0xb94920618b9ff, 0x386c9f301016b,+ 0x151b802e73a15, 0xc2f01e076ffe7, 0x3f6c4272644fe, 0x4c65a96d2810e, 0x3abd30e3d0bdd,+ 0x507b582802a4c, 0xdd3a9d6440284, 0xaa79901e9e59, 0x28f954b3df137, 0xdfda79ce1298,+ 0x418aa5e56d569, 0x6cda605cd2a84, 0x38a3c8c72727e, 0x5c577b1659af2, 0x29ebd4cc67d3e,+ 0x5fe67cbb69948, 0xcbb87ceb0460f, 0x2fb2c44c7e9b9, 0xd94fcdb26c47f, 0x185464b74c699,+ 0x439cb4f543d28, 0xd46bc2b92ce79, 0x7add9a636f66a, 0xdf31c41270332, 0x2696553f66dbf,+ 0x2783576782589, 0x968618b71297b, 0x44e45e45be560, 0x26efc23b82d1d, 0x1015ba1501f7c,+ 0x488bd29858305, 0xf74fe2c48c2bf, 0x4e0542975a40a, 0x6c35069031288, 0x86495d91e4d6,+ 0x4901896d5e61b, 0x73cc7582e2dff, 0x5720d009880e3, 0x1cf8b3eeb3330, 0x39d782d92776a,+ 0x64cfb1b2b7536, 0x2c925510c2142, 0x15d3fdca3fa2f, 0x7c71f8ad652d2, 0x2c8639f12eb8e,+ 0x32be1915c3b16, 0x401ba942b1327, 0x73a61e3dd1e67, 0x57855aa9e04e, 0x6692e349a453,+ 0x1708b23f8a1a6, 0x994bfdde94868, 0x51ceeeda593ec, 0x4cdc508073c99, 0x155d6d8b5f390,+ 0x3f62bd2bb7dc1, 0x1e9f2950a19ef, 0x3879b3fdeef4a, 0x769c9aebe06f6, 0x2fdbb0a2e42c2,+ 0x4b669ca5182c8, 0x9f7d2af918087, 0x47a04688c10e0, 0x6fc09c1c63852, 0x103c53a1b2bb3,+ 0x3f6293b845e4a, 0x7eddacd3d44ea, 0x3427d6919a9d0, 0xf18b5ac4b772, 0xc48971f25ff7,+ 0x30b22ab28a9ee, 0x684778d576841, 0x219063e835137, 0xabfbe8b9b4cbe, 0xa5ee7bb20c8f,+ 0x739e1c031098, 0x4f194a1f5e7e7, 0x7c3800ab2834d, 0x665b9b090387b, 0x3bfcc848c5569,+ 0x4d7a0fa6ecdd0, 0x5a3be381deaf4, 0x60c235b9afa21, 0x2b6af4dca92ba, 0x3d4ef541da8da,+ 0x15368f12c5e0, 0x1089d55881802, 0x4e950a255aa08, 0xea013ebd20bf4, 0x9bf696b8d0d0,+ 0x74755b59e06a4, 0xca3ec0966d7d7, 0x23f8c820bf534, 0x7a9c3d0d130fb, 0x22243e7f72df8,+ 0x177eb94dece80, 0x18e56144e85e5, 0x746051389a65, 0xaaa6ded4b788c, 0x2e7e3c710e646,+ 0x10535c987ba48, 0xce1ab4a74c92c, 0x355c567052319, 0x3dff053e8baa7, 0x30145933a20bb,+ 0x3f829e6bd6bba, 0x5f8c66072ace7, 0x6d4eed643467f, 0xcc0b57a9729d8, 0x63d4c7dcdff6,+ 0x6e2f1ed3c2a71, 0xbfed27df0cb8c, 0x57b9be8c8fece, 0xe8296dacbba0e, 0x19ec43845e7cd,+ 0x499487bfb2cfd, 0x15507fb8a4c3f, 0x4eb4c133c44c6, 0x46f4d7d05d5e9, 0x1403ec072267b,+ 0x42f3db0918e6f, 0x4a12f29990cd, 0x21c078f7b096c, 0x5d7cb674f5b94, 0x4b6ab2b6b85,+ 0x64dcbb337dda8, 0xae2a09ce386d8, 0x47d4c764502e5, 0x37ab82e784a4c, 0x279485eef7e12,+ 0x6e3a35d64f447, 0xb5f72cb08d802, 0x130922ccaf665, 0xfcb727b232952, 0x3aaa4745db15f,+ 0x555f7a3941881, 0x9c78701503430, 0x6109825b920d4, 0x237def01f0a49, 0x2fa436baf03a9,+ 0x52f3c5d10d5d8, 0x1249c36d73132, 0x73b2269c79aa7, 0x6425b9fe81796, 0x379a76cb42ac8,+ 0x61c0bf3594366, 0xda97bf0917530, 0x60340c80e3c62, 0x6e3d5ac5d53a3, 0x3272121adefa2,+ 0x55f37132a6fa0, 0x20c61072d5855, 0x514a0e230d7ba, 0xbad776d17830, 0x1c421a5a5b50f }; @@ -278,7 +319,7 @@ * * On entry: tmp[i] < 2**128 * On exit: out[0,2,...] < 2**52, out[1,3,...] < 2**53 */-static void felem_reduce_degree(felem out, u128 tmp[9]) {+FELEM_INLINE void felem_reduce_degree(felem out, u128 tmp[9]) { /* The following table may be helpful when reading this code: * * Limb number: 0 | 1 | 2 | 3 | 4 | 5 | 6 | 7 | 8 | 9 | 10@@ -468,7 +509,7 @@ * * On entry: in[0,2,...] < 2**52, in[1,3,...] < 2**53. * On exit: out[0,2,...] < 2**52, out[1,3,...] < 2**53. */-static void felem_square(felem out, const felem in) {+FELEM_INLINE void felem_square(felem out, const felem in) { u128 tmp[9], x1x1, x3x3; x1x1 = ((u128) in[1]) * in[1];@@ -496,7 +537,7 @@ * On entry: in[0,2,...] < 2**52, in[1,3,...] < 2**53 and * in2[0,2,...] < 2**52, in2[1,3,...] < 2**53. * On exit: out[0,2,...] < 2**52, out[1,3,...] < 2**53. */-static void felem_mul(felem out, const felem in, const felem in2) {+FELEM_INLINE void felem_mul(felem out, const felem in, const felem in2) { u128 tmp[9], x1y1, x1y3, x3y1, x3y3; x1y1 = ((u128) in[1]) * in2[1];
@@ -0,0 +1,25 @@+/*+ * The vendored s2n-bignum assembly, behind one call that picks the variant+ * the machine wants. See cbits/s2n/README.md for which and why.+ *+ * Only declared in the 64-bit field build: s2n-bignum is x86-64 and AArch64+ * only, and this interface is the four little-endian 64-bit words those+ * architectures give crypton_p256_int anyway.+ */+#ifndef CRYPTON_P256_S2N_H+#define CRYPTON_P256_S2N_H++#include <stdint.h>++/* res = scalar * point, all of them affine and not in Montgomery form:+ * point is x then y, four words each, and res the same. The point at+ * infinity goes in and comes out as (0, 0). */+void crypton_s2n_p256_scalarmul(uint64_t res[8], const uint64_t scalar[4],+ const uint64_t point[8]);++/* res = scalar * G, the same shape out. The table it reads and the window+ * width it was built for are in cbits/p256/p256_base_table.c, which+ * cbits/p256/gen_base_table.py writes. */+void crypton_s2n_p256_scalarmulbase(uint64_t res[8], const uint64_t scalar[4]);++#endif
@@ -0,0 +1,2 @@+834a90d5e846ffa1e1611bd24e160bb2e9b86d35+v2.0.0
@@ -0,0 +1,305 @@+mldsa-native is a fork of the public domain Dilithium reference implementation,+available on https://github.com/pq-crystals/dilithium.++All new files and all files derived from the Dilithium reference+implementation are made available under the Apache-2.0 license OR+the ISC license OR the MIT license. These licenses are+reproduced at the bottom of this file.++Files outside the library itself may carry different terms. Every file+states its own SPDX-License-Identifier, which determines the terms that+apply to it. In particular:++The code in test/notrandombytes/*, and its copies in+examples/*/test_only_rng/* and scripts/notrandombytes, is derived from+https://cr.yp.to/papers.html#surf and licensed under+LicenseRef-PD-hp OR CC0-1.0 OR 0BSD OR MIT-0 OR MIT.+It is only used for testing purposes.++The benchmarking code in test/hal/* carries the+MIT license. It is only used for testing purposes.++The tiny_sha3 code in+examples/bring_your_own_fips202/custom_fips202/tiny_sha3/* and+examples/custom_backend/mldsa_native/src/fips202/native/custom/src/*+carries the MIT license. It is only used to demonstrate custom+FIPS-202 implementations.++The proofs in proofs/* are in part derived from Amazon Web Services+verification infrastructure and licensed under+Apache-2.0 OR ISC OR MIT-0, MIT-0, or MIT-0 AND Apache-2.0. The IACR+document class proofs/isabelle/neon_ntt/document/iacrtrans.cls is+licensed under CC0-1.0. None of this is part of the library.++Documentation is licensed under CC-BY-4.0.++```+Copyright (c) The mldsa-native project authors+Copyright (c) The mlkem-native project authors+Copyright (c) 2020 Dougall Johnson+Copyright (c) 2022 Arm Limited+SPDX-License-Identifier: MIT++Permission is hereby granted, free of charge, to any person obtaining a copy+of this software and associated documentation files (the "Software"), to deal+in the Software without restriction, including without limitation the rights+to use, copy, modify, merge, publish, distribute, sublicense, and/or sell+copies of the Software, and to permit persons to whom the Software is+furnished to do so, subject to the following conditions:++The above copyright notice and this permission notice shall be included in+all copies or substantial portions of the Software.++THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR+IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,+FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE+AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER+LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,+OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE+SOFTWARE.+```++ISC license for mldsa-native content+------------------------------------++Copyright (c) The mldsa-native project authors++Permission to use, copy, modify, and/or distribute this software for any purpose+with or without fee is hereby granted, provided that the above copyright notice+and this permission notice appear in all copies.++THE SOFTWARE IS PROVIDED "AS IS" AND THE AUTHOR DISCLAIMS ALL WARRANTIES WITH+REGARD TO THIS SOFTWARE INCLUDING ALL IMPLIED WARRANTIES OF MERCHANTABILITY AND+FITNESS. IN NO EVENT SHALL THE AUTHOR BE LIABLE FOR ANY SPECIAL, DIRECT,+INDIRECT, OR CONSEQUENTIAL DAMAGES OR ANY DAMAGES WHATSOEVER RESULTING FROM LOSS+OF USE, DATA OR PROFITS, WHETHER IN AN ACTION OF CONTRACT, NEGLIGENCE OR OTHER+TORTIOUS ACTION, ARISING OUT OF OR IN CONNECTION WITH THE USE OR PERFORMANCE OF+THIS SOFTWARE.+++MIT license for mldsa-native content+------------------------------------++Copyright (c) The mldsa-native project authors++Permission is hereby granted, free of charge, to any person obtaining a copy of+this software and associated documentation files (the “Software”), to deal in+the Software without restriction, including without limitation the rights to+use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of+the Software, and to permit persons to whom the Software is furnished to do so,+subject to the following conditions:++The above copyright notice and this permission notice shall be included in all+copies or substantial portions of the Software.++THE SOFTWARE IS PROVIDED “AS IS”, WITHOUT WARRANTY OF ANY KIND, EXPRESS OR+IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS+FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR+COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER+IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN+CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.++Apache-2.0 license for mldsa-native content+-------------------------------------------++ Apache License+ Version 2.0, January 2004+ http://www.apache.org/licenses/++ TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION++ 1. Definitions.++ "License" shall mean the terms and conditions for use, reproduction,+ and distribution as defined by Sections 1 through 9 of this document.++ "Licensor" shall mean the copyright owner or entity authorized by+ the copyright owner that is granting the License.++ "Legal Entity" shall mean the union of the acting entity and all+ other entities that control, are controlled by, or are under common+ control with that entity. For the purposes of this definition,+ "control" means (i) the power, direct or indirect, to cause the+ direction or management of such entity, whether by contract or+ otherwise, or (ii) ownership of fifty percent (50%) or more of the+ outstanding shares, or (iii) beneficial ownership of such entity.++ "You" (or "Your") shall mean an individual or Legal Entity+ exercising permissions granted by this License.++ "Source" form shall mean the preferred form for making modifications,+ including but not limited to software source code, documentation+ source, and configuration files.++ "Object" form shall mean any form resulting from mechanical+ transformation or translation of a Source form, including but+ not limited to compiled object code, generated documentation,+ and conversions to other media types.++ "Work" shall mean the work of authorship, whether in Source or+ Object form, made available under the License, as indicated by a+ copyright notice that is included in or attached to the work+ (an example is provided in the Appendix below).++ "Derivative Works" shall mean any work, whether in Source or Object+ form, that is based on (or derived from) the Work and for which the+ editorial revisions, annotations, elaborations, or other modifications+ represent, as a whole, an original work of authorship. For the purposes+ of this License, Derivative Works shall not include works that remain+ separable from, or merely link (or bind by name) to the interfaces of,+ the Work and Derivative Works thereof.++ "Contribution" shall mean any work of authorship, including+ the original version of the Work and any modifications or additions+ to that Work or Derivative Works thereof, that is intentionally+ submitted to Licensor for inclusion in the Work by the copyright owner+ or by an individual or Legal Entity authorized to submit on behalf of+ the copyright owner. For the purposes of this definition, "submitted"+ means any form of electronic, verbal, or written communication sent+ to the Licensor or its representatives, including but not limited to+ communication on electronic mailing lists, source code control systems,+ and issue tracking systems that are managed by, or on behalf of, the+ Licensor for the purpose of discussing and improving the Work, but+ excluding communication that is conspicuously marked or otherwise+ designated in writing by the copyright owner as "Not a Contribution."++ "Contributor" shall mean Licensor and any individual or Legal Entity+ on behalf of whom a Contribution has been received by Licensor and+ subsequently incorporated within the Work.++ 2. Grant of Copyright License. Subject to the terms and conditions of+ this License, each Contributor hereby grants to You a perpetual,+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable+ copyright license to reproduce, prepare Derivative Works of,+ publicly display, publicly perform, sublicense, and distribute the+ Work and such Derivative Works in Source or Object form.++ 3. Grant of Patent License. Subject to the terms and conditions of+ this License, each Contributor hereby grants to You a perpetual,+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable+ (except as stated in this section) patent license to make, have made,+ use, offer to sell, sell, import, and otherwise transfer the Work,+ where such license applies only to those patent claims licensable+ by such Contributor that are necessarily infringed by their+ Contribution(s) alone or by combination of their Contribution(s)+ with the Work to which such Contribution(s) was submitted. If You+ institute patent litigation against any entity (including a+ cross-claim or counterclaim in a lawsuit) alleging that the Work+ or a Contribution incorporated within the Work constitutes direct+ or contributory patent infringement, then any patent licenses+ granted to You under this License for that Work shall terminate+ as of the date such litigation is filed.++ 4. Redistribution. You may reproduce and distribute copies of the+ Work or Derivative Works thereof in any medium, with or without+ modifications, and in Source or Object form, provided that You+ meet the following conditions:++ (a) You must give any other recipients of the Work or+ Derivative Works a copy of this License; and++ (b) You must cause any modified files to carry prominent notices+ stating that You changed the files; and++ (c) You must retain, in the Source form of any Derivative Works+ that You distribute, all copyright, patent, trademark, and+ attribution notices from the Source form of the Work,+ excluding those notices that do not pertain to any part of+ the Derivative Works; and++ (d) If the Work includes a "NOTICE" text file as part of its+ distribution, then any Derivative Works that You distribute must+ include a readable copy of the attribution notices contained+ within such NOTICE file, excluding those notices that do not+ pertain to any part of the Derivative Works, in at least one+ of the following places: within a NOTICE text file distributed+ as part of the Derivative Works; within the Source form or+ documentation, if provided along with the Derivative Works; or,+ within a display generated by the Derivative Works, if and+ wherever such third-party notices normally appear. The contents+ of the NOTICE file are for informational purposes only and+ do not modify the License. You may add Your own attribution+ notices within Derivative Works that You distribute, alongside+ or as an addendum to the NOTICE text from the Work, provided+ that such additional attribution notices cannot be construed+ as modifying the License.++ You may add Your own copyright statement to Your modifications and+ may provide additional or different license terms and conditions+ for use, reproduction, or distribution of Your modifications, or+ for any such Derivative Works as a whole, provided Your use,+ reproduction, and distribution of the Work otherwise complies with+ the conditions stated in this License.++ 5. Submission of Contributions. Unless You explicitly state otherwise,+ any Contribution intentionally submitted for inclusion in the Work+ by You to the Licensor shall be under the terms and conditions of+ this License, without any additional terms or conditions.+ Notwithstanding the above, nothing herein shall supersede or modify+ the terms of any separate license agreement you may have executed+ with Licensor regarding such Contributions.++ 6. Trademarks. This License does not grant permission to use the trade+ names, trademarks, service marks, or product names of the Licensor,+ except as required for reasonable and customary use in describing the+ origin of the Work and reproducing the content of the NOTICE file.++ 7. Disclaimer of Warranty. Unless required by applicable law or+ agreed to in writing, Licensor provides the Work (and each+ Contributor provides its Contributions) on an "AS IS" BASIS,+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or+ implied, including, without limitation, any warranties or conditions+ of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A+ PARTICULAR PURPOSE. You are solely responsible for determining the+ appropriateness of using or redistributing the Work and assume any+ risks associated with Your exercise of permissions under this License.++ 8. Limitation of Liability. In no event and under no legal theory,+ whether in tort (including negligence), contract, or otherwise,+ unless required by applicable law (such as deliberate and grossly+ negligent acts) or agreed to in writing, shall any Contributor be+ liable to You for damages, including any direct, indirect, special,+ incidental, or consequential damages of any character arising as a+ result of this License or out of the use or inability to use the+ Work (including but not limited to damages for loss of goodwill,+ work stoppage, computer failure or malfunction, or any and all+ other commercial damages or losses), even if such Contributor+ has been advised of the possibility of such damages.++ 9. Accepting Warranty or Additional Liability. While redistributing+ the Work or Derivative Works thereof, You may choose to offer,+ and charge a fee for, acceptance of support, warranty, indemnity,+ or other liability obligations and/or rights consistent with this+ License. However, in accepting such obligations, You may act only+ on Your own behalf and on Your sole responsibility, not on behalf+ of any other Contributor, and only if You agree to indemnify,+ defend, and hold each Contributor harmless for any liability+ incurred by, or claims asserted against, such Contributor by reason+ of your accepting any such warranty or additional liability.++ END OF TERMS AND CONDITIONS++ APPENDIX: How to apply the Apache License to your work.++ To apply the Apache License to your work, attach the following+ boilerplate notice, with the fields enclosed by brackets "[]"+ replaced with your own identifying information. (Don't include+ the brackets!) The text should be enclosed in the appropriate+ comment syntax for the file format. We also recommend that a+ file or class name and description of purpose be included on the+ same "printed page" as the copyright notice for easier+ identification within third-party archives.++ Copyright (c) The mldsa-native project authors++ Licensed under the Apache License, Version 2.0 (the "License");+ you may not use this file except in compliance with the License.+ You may obtain a copy of the License at++ http://www.apache.org/licenses/LICENSE-2.0++ Unless required by applicable law or agreed to in writing, software+ distributed under the License is distributed on an "AS IS" BASIS,+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.+ See the License for the specific language governing permissions and+ limitations under the License.
@@ -0,0 +1,60 @@+# mldsa-native++ML-DSA (FIPS 204) from the PQ Code Package, vendored here and built for all+three parameter sets.++## Where it comes from++<https://github.com/pq-code-package/mldsa-native>. `COMMIT` holds the+revision this tree is at; `import.sh` puts it there and is how the tree is+refreshed. Nothing here is edited by hand -- every choice crypton makes is+made in `crypton_mldsa.h`, `crypton_mldsa.c` and `crypton.cabal`, so that a+re-import is a straight overwrite.++## Licence++`Apache-2.0 OR ISC OR MIT`, the same three-way form as `cbits/s2n` and by+some of the same authors. crypton takes it under ISC, which is already in+the package's `license:` field. `LICENSE` is the upstream file and is+listed in `license-files:`.++## What is taken and what is not++The whole of `mldsa/`, less the backends for architectures crypton does not+build for: the 32-bit+`src/fips202/native/armv81m`. (Unlike mlkem-native, this one ships only the+AArch64 and x86-64 arithmetic backends, so there is nothing else to drop.) Every reference to those is behind an+`MLD_SYS_` guard that cannot be true on the architectures crypton does+build for, so dropping them changes no build and keeps a few dozen files of+unreachable assembly out of the release tarball. To take one back, delete+its line from `import.sh` and add its directory to `extra-source-files:`.++## How it is built++Upstream builds for one parameter set at a time. `crypton_mldsa.c`+includes the amalgamation once per set -- the level-independent half kept by+exactly one of them -- which is how a single crypton offers ML-DSA-44, 65+and 87. `crypton_mldsa_asm.S` does the same for the assembly, which is+level-independent and so is included once.++This is the shape `cbits/aes/armv8.c` already uses for the three AES key+sizes, and it has the same hazard: **cabal does not know that the wrapper+depends on the tree it includes.** After changing anything under+`cbits/mldsa`, touch `crypton_mldsa.c`, or the build keeps the object it+already has and the change is not tested.++The hand-written backends are selected by `CRYPTON_MLDSA_NATIVE_BACKEND`,+which `crypton.cabal` defines on x86-64 and AArch64 other than Windows --+the same exclusion, and for the same reasons, as `cbits/s2n`. Everywhere+else the portable C is built, which is the same code and passes the same+tests.++There is no randomised API: `MLD_CONFIG_NO_RANDOMIZED_API` is set, no+`randombytes()` is needed, and randomness is drawn in Haskell through+`MonadRandom` as it is for every other key crypton generates.++The symbols are `crypton_mldsa44_*`, `crypton_mldsa65_*` and+`crypton_mldsa87_*` rather than upstream's defaults, so that an+application linking another copy of mldsa-native -- through some other+library, or its own -- does not present the linker with two sets of+functions answering to one set of names.
@@ -0,0 +1,31 @@+/*+ * All three ML-DSA parameter sets in one translation unit.+ *+ * mldsa-native is built for one parameter set at a time; a build wanting+ * several includes the amalgamation once per set, with the level-independent+ * half kept by exactly one of them. This is the same shape as+ * cbits/aes/armv8.c, which includes cbits/aes/armv8_impl.c three times for+ * the three AES key sizes.+ *+ * NOTE, as there: cabal does not know that this file depends on the tree it+ * includes. After changing anything under cbits/mldsa, touch this file, or+ * the build will quietly keep the object it already has.+ */+#include "crypton_mldsa.h"++#define MLD_CONFIG_MULTILEVEL_WITH_SHARED 1+#define MLD_CONFIG_MONOBUILD_KEEP_SHARED_HEADERS+#define MLD_CONFIG_PARAMETER_SET 44+#include "mldsa_native.c"+#undef MLD_CONFIG_MULTILEVEL_WITH_SHARED+#undef MLD_CONFIG_PARAMETER_SET++#define MLD_CONFIG_MULTILEVEL_NO_SHARED+#define MLD_CONFIG_PARAMETER_SET 65+#include "mldsa_native.c"+#undef MLD_CONFIG_MONOBUILD_KEEP_SHARED_HEADERS+#undef MLD_CONFIG_PARAMETER_SET++#define MLD_CONFIG_PARAMETER_SET 87+#include "mldsa_native.c"+#undef MLD_CONFIG_PARAMETER_SET
@@ -0,0 +1,41 @@+/*+ * What crypton asks of the vendored mldsa-native, in one place. The tree+ * under cbits/mldsa is upstream's and is overwritten by import.sh, so every+ * choice crypton makes is made here instead of by editing it.+ *+ * Included first by both crypton_mldsa.c and crypton_mldsa_asm.S, so it must+ * hold nothing but preprocessor directives.+ */+#ifndef CRYPTON_MLDSA_H+#define CRYPTON_MLDSA_H++/*+ * The symbols are crypton's own, not the default PQCP_MLDSA_NATIVE_*. An+ * application is free to link another copy of mldsa-native -- through some+ * other library, or its own -- and two copies answering to one set of names+ * is a problem the linker resolves silently and in nobody's favour. With+ * MLD_CONFIG_MULTILEVEL_BUILD the level is appended, so the entry points+ * are crypton_mldsa44_*, crypton_mldsa65_* and crypton_mldsa87_*.+ */+#define MLD_CONFIG_NAMESPACE_PREFIX crypton_mldsa+#define MLD_CONFIG_MULTILEVEL_BUILD++/*+ * No randomised API, so no randombytes() to provide. Randomness is drawn+ * in Haskell through MonadRandom, the way every other key in crypton is+ * generated, and the deterministic entry points are what the FFI calls.+ * That also keeps the C free of any opinion about where entropy comes from.+ */+#define MLD_CONFIG_NO_RANDOMIZED_API++/*+ * The hand-written backends, where crypton.cabal says the architecture has+ * them. Without this the portable C is built, which is correct everywhere+ * and is what every other architecture gets.+ */+#ifdef CRYPTON_MLDSA_NATIVE_BACKEND+#define MLD_CONFIG_USE_NATIVE_BACKEND_ARITH+#define MLD_CONFIG_USE_NATIVE_BACKEND_FIPS202+#endif++#endif /* CRYPTON_MLDSA_H */
@@ -0,0 +1,14 @@+/*+ * The assembly half, which is level-independent: it is included once, with+ * the shared directives kept, and covers all three parameter sets. See the+ * comment at the top of mldsa_native_asm.S.+ *+ * Built only where crypton.cabal defines CRYPTON_MLDSA_NATIVE_BACKEND; on+ * every other architecture this file is not in asm-sources at all.+ */+#include "crypton_mldsa.h"++#define MLD_CONFIG_MULTILEVEL_WITH_SHARED 1+#define MLD_CONFIG_MONOBUILD_KEEP_SHARED_HEADERS+#define MLD_CONFIG_PARAMETER_SET 65+#include "mldsa_native_asm.S"
@@ -0,0 +1,40 @@+#!/bin/sh+# Re-import the vendored parts of the PQ Code Package's mldsa-native.+#+# The files are kept unmodified. Everything crypton decides -- which+# parameter sets exist, what the symbols are called, that there is no+# randomised API -- is decided in cbits/mldsa/crypton_mldsa.c and in+# crypton.cabal, not by editing anything here. Run this from cbits/mldsa:+#+# ./import.sh [tag-or-commit]+#+# and commit the result together with the COMMIT line it writes, so that the+# tree always says which upstream revision it holds.+set -eu++REPO=https://github.com/pq-code-package/mldsa-native+REV=${1:-v2.0.0}+HERE=$(cd "$(dirname "$0")" && pwd)+TMP=$(mktemp -d)+trap 'rm -rf "$TMP"' EXIT++git clone -q "$REPO" "$TMP/u"+git -C "$TMP/u" checkout -q "$REV"++rm -rf "$HERE/src"+cp "$TMP/u/mldsa/mldsa_native.c" "$HERE/"+cp "$TMP/u/mldsa/mldsa_native.h" "$HERE/"+cp "$TMP/u/mldsa/mldsa_native_asm.S" "$HERE/"+cp "$TMP/u/mldsa/mldsa_native_config.h" "$HERE/"+cp -R "$TMP/u/mldsa/src" "$HERE/src"+cp "$TMP/u/LICENSE" "$HERE/LICENSE"++# The 32-bit Armv8.1-M Keccak, for an architecture crypton does not build+# for. Every reference to it is behind an MLD_SYS_ guard that cannot be true+# on the ones it does, so dropping it changes no build and keeps unreachable+# assembly out of the release tarball.+rm -rf "$HERE/src/fips202/native/armv81m"++git -C "$TMP/u" rev-parse HEAD > "$HERE/COMMIT"+git -C "$TMP/u" describe --tags --exact-match 2>/dev/null >> "$HERE/COMMIT" || true+echo "imported $(head -1 "$HERE/COMMIT")"
@@ -0,0 +1,803 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*+ * WARNING: This file is auto-generated from scripts/autogen+ * in the mldsa-native repository.+ * Do not modify it directly.+ */++/******************************************************************************+ *+ * Single compilation unit (SCU) for fixed-level build of mldsa-native+ *+ * This compilation unit bundles together all source files for a build+ * of mldsa-native for a fixed security level (MLDSA-44/65/87).+ *+ * # API+ *+ * The API exposed by this file is described in mldsa_native.h.+ *+ * # Multi-level build+ *+ * If you want an SCU build of mldsa-native with support for multiple security+ * levels, you need to include this file multiple times, and set+ * MLD_CONFIG_MULTILEVEL_WITH_SHARED and MLD_CONFIG_MULTILEVEL_NO_SHARED+ * appropriately. This is exemplified in examples/monolithic_build_multilevel+ * and examples/monolithic_build_multilevel_native.+ *+ * # Configuration+ *+ * The following options from the mldsa-native configuration are relevant:+ *+ * - MLD_CONFIG_FIPS202_CUSTOM_HEADER+ * Set this option if you use a custom FIPS202 implementation.+ *+ * - MLD_CONFIG_USE_NATIVE_BACKEND_ARITH+ * Set this option if you want to include the native arithmetic backends+ * in your build.+ *+ * - MLD_CONFIG_USE_NATIVE_BACKEND_FIPS202+ * Set this option if you want to include the native FIPS202 backends+ * in your build.+ *+ * - MLD_CONFIG_MONOBUILD_KEEP_SHARED_HEADERS+ * Set this option if you want to keep the directives defined in+ * level-independent headers. This is needed for a multi-level build.+ */++/* If parts of the mldsa-native source tree are not used,+ * consider reducing this header via `unifdef`.+ *+ * Example:+ * ```bash+ * unifdef -UMLD_CONFIG_USE_NATIVE_BACKEND_ARITH mldsa_native.c+ * ```+ */++#include "src/common.h"++#include "src/ct.c"+#include "src/debug.c"+#include "src/packing.c"+#include "src/poly.c"+#include "src/poly_kl.c"+#include "src/polyvec.c"+#include "src/polyvec_lazy.c"+#include "src/sign.c"++#if !defined(MLD_CONFIG_FIPS202_CUSTOM_HEADER)+#include "src/fips202/fips202.c"+#include "src/fips202/fips202x4.c"+#include "src/fips202/keccakf1600.c"+#endif++#if defined(MLD_CONFIG_USE_NATIVE_BACKEND_ARITH)+#if defined(MLD_SYS_AARCH64)+#include "src/native/aarch64/src/aarch64_zetas.c"+#include "src/native/aarch64/src/polyz_unpack_table.c"+#include "src/native/aarch64/src/rej_uniform_eta_table.c"+#include "src/native/aarch64/src/rej_uniform_table.c"+#endif /* MLD_SYS_AARCH64 */+#if defined(MLD_SYS_X86_64)+#include "src/native/x86_64/src/consts.c"+#include "src/native/x86_64/src/rej_uniform_table.c"+#endif+#endif /* MLD_CONFIG_USE_NATIVE_BACKEND_ARITH */++#if defined(MLD_CONFIG_USE_NATIVE_BACKEND_FIPS202)+#if defined(MLD_SYS_AARCH64)+#include "src/fips202/native/aarch64/src/keccakf1600_round_constants.c"+#endif+#if defined(MLD_SYS_X86_64)+#include "src/fips202/native/x86_64/src/keccakf1600_constants.c"+#endif+#if defined(MLD_SYS_ARMV81M_MVE)+#include "src/fips202/native/armv81m/src/keccak_f1600_x4_mve.c"+#include "src/fips202/native/armv81m/src/keccakf1600_round_constants.c"+#endif+#endif /* MLD_CONFIG_USE_NATIVE_BACKEND_FIPS202 */++/* Macro #undef's+ *+ * The following undefines macros from headers+ * included by the source files imported above.+ *+ * This is to allow building and linking multiple builds+ * of mldsa-native for varying parameter sets through concatenation+ * of this file, as if the files had been compiled separately.+ * If this is not relevant to you, you may remove the following.+ */++/*+ * Undefine macros from MLD_CONFIG_PARAMETER_SET-specific files+ */+/* mldsa/mldsa_native.h */+#undef MLDSA44_BYTES+#undef MLDSA44_CRHBYTES+#undef MLDSA44_PUBLICKEYBYTES+#undef MLDSA44_RNDBYTES+#undef MLDSA44_SECRETKEYBYTES+#undef MLDSA44_SEEDBYTES+#undef MLDSA44_TRBYTES+#undef MLDSA65_BYTES+#undef MLDSA65_CRHBYTES+#undef MLDSA65_PUBLICKEYBYTES+#undef MLDSA65_RNDBYTES+#undef MLDSA65_SECRETKEYBYTES+#undef MLDSA65_SEEDBYTES+#undef MLDSA65_TRBYTES+#undef MLDSA87_BYTES+#undef MLDSA87_CRHBYTES+#undef MLDSA87_PUBLICKEYBYTES+#undef MLDSA87_RNDBYTES+#undef MLDSA87_SECRETKEYBYTES+#undef MLDSA87_SEEDBYTES+#undef MLDSA87_TRBYTES+#undef MLDSA_BYTES+#undef MLDSA_BYTES_+#undef MLDSA_CRHBYTES+#undef MLDSA_PUBLICKEYBYTES+#undef MLDSA_PUBLICKEYBYTES_+#undef MLDSA_RNDBYTES+#undef MLDSA_SECRETKEYBYTES+#undef MLDSA_SECRETKEYBYTES_+#undef MLDSA_SEEDBYTES+#undef MLDSA_TRBYTES+#undef MLD_API_CONCAT+#undef MLD_API_CONCAT_+#undef MLD_API_CONCAT_UNDERSCORE+#undef MLD_API_MUST_CHECK_RETURN_VALUE+#undef MLD_API_NAMESPACE+#undef MLD_API_NAMESPACE_PREFIX+#undef MLD_API_QUALIFIER+#undef MLD_DOMAIN_SEPARATION_MAX_BYTES+#undef MLD_ERR_FAIL+#undef MLD_ERR_INVALID_ARG+#undef MLD_ERR_INVALID_KEY+#undef MLD_ERR_INVALID_SIGNATURE+#undef MLD_ERR_OUT_OF_MEMORY+#undef MLD_ERR_PCT_FAIL+#undef MLD_ERR_RNG_FAIL+#undef MLD_ERR_SIGNING_PAUSED+#undef MLD_ERR_SIGN_ATTEMPTS_EXHAUSTED+#undef MLD_H+#undef MLD_MAX3_+#undef MLD_MAX4_+#undef MLD_PREHASH_NONE+#undef MLD_PREHASH_SHA2_224+#undef MLD_PREHASH_SHA2_256+#undef MLD_PREHASH_SHA2_384+#undef MLD_PREHASH_SHA2_512+#undef MLD_PREHASH_SHA2_512_224+#undef MLD_PREHASH_SHA2_512_256+#undef MLD_PREHASH_SHA3_224+#undef MLD_PREHASH_SHA3_256+#undef MLD_PREHASH_SHA3_384+#undef MLD_PREHASH_SHA3_512+#undef MLD_PREHASH_SHAKE_128+#undef MLD_PREHASH_SHAKE_256+#undef MLD_TOTAL_ALLOC_44+#undef MLD_TOTAL_ALLOC_44_KEYPAIR+#undef MLD_TOTAL_ALLOC_44_KEYPAIR_NO_PCT+#undef MLD_TOTAL_ALLOC_44_KEYPAIR_PCT+#undef MLD_TOTAL_ALLOC_44_PK_FROM_SK+#undef MLD_TOTAL_ALLOC_44_SIGN+#undef MLD_TOTAL_ALLOC_44_VERIFY+#undef MLD_TOTAL_ALLOC_65+#undef MLD_TOTAL_ALLOC_65_KEYPAIR+#undef MLD_TOTAL_ALLOC_65_KEYPAIR_NO_PCT+#undef MLD_TOTAL_ALLOC_65_KEYPAIR_PCT+#undef MLD_TOTAL_ALLOC_65_PK_FROM_SK+#undef MLD_TOTAL_ALLOC_65_SIGN+#undef MLD_TOTAL_ALLOC_65_VERIFY+#undef MLD_TOTAL_ALLOC_87+#undef MLD_TOTAL_ALLOC_87_KEYPAIR+#undef MLD_TOTAL_ALLOC_87_KEYPAIR_NO_PCT+#undef MLD_TOTAL_ALLOC_87_KEYPAIR_PCT+#undef MLD_TOTAL_ALLOC_87_PK_FROM_SK+#undef MLD_TOTAL_ALLOC_87_SIGN+#undef MLD_TOTAL_ALLOC_87_VERIFY+/* mldsa/src/common.h */+#undef MLD_ADD_PARAM_SET+#undef MLD_ALLOC+#undef MLD_APPLY+#undef MLD_ASM_FN_SIZE+#undef MLD_ASM_FN_SYMBOL+#undef MLD_ASM_NAMESPACE+#undef MLD_BUILD_INTERNAL+#undef MLD_COMMON_H+#undef MLD_CONCAT+#undef MLD_CONCAT_+#undef MLD_EMPTY_CU+#undef MLD_ERR_FAIL+#undef MLD_ERR_INVALID_ARG+#undef MLD_ERR_INVALID_KEY+#undef MLD_ERR_INVALID_SIGNATURE+#undef MLD_ERR_OUT_OF_MEMORY+#undef MLD_ERR_PCT_FAIL+#undef MLD_ERR_RNG_FAIL+#undef MLD_ERR_SIGNING_PAUSED+#undef MLD_ERR_SIGN_ATTEMPTS_EXHAUSTED+#undef MLD_EXTERNAL_API+#undef MLD_FIPS202X4_HEADER_FILE+#undef MLD_FIPS202_HEADER_FILE+#undef MLD_FREE+#undef MLD_INTERNAL_API+#undef MLD_INTERNAL_DATA_DECLARATION+#undef MLD_INTERNAL_DATA_DEFINITION+#undef MLD_MULTILEVEL_BUILD+#undef MLD_NAMESPACE+#undef MLD_NAMESPACE_KL+#undef MLD_NAMESPACE_PREFIX+#undef MLD_NAMESPACE_PREFIX_KL+#undef mld_memcpy+#undef mld_memset+/* mldsa/src/packing.h */+#undef MLD_PACKING_H+#undef mld_pack_sig_c+#undef mld_pack_sig_h+#undef mld_pack_sig_z+#undef mld_pack_sk_rho_key_tr_s2+#undef mld_pack_sk_s1+#undef mld_sig_unpack_hints+#undef mld_unpack_pk_t1+#undef mld_unpack_sk+/* mldsa/src/params.h */+#undef MLDSA_BETA+#undef MLDSA_CRHBYTES+#undef MLDSA_CRYPTO_BYTES+#undef MLDSA_CRYPTO_PUBLICKEYBYTES+#undef MLDSA_CRYPTO_SECRETKEYBYTES+#undef MLDSA_CTILDEBYTES+#undef MLDSA_D+#undef MLDSA_ETA+#undef MLDSA_GAMMA1+#undef MLDSA_GAMMA2+#undef MLDSA_GAMMA2_32+#undef MLDSA_GAMMA2_88+#undef MLDSA_K+#undef MLDSA_L+#undef MLDSA_N+#undef MLDSA_OMEGA+#undef MLDSA_PK_END+#undef MLDSA_PK_RHO_BYTES+#undef MLDSA_PK_RHO_OFFSET+#undef MLDSA_PK_T1_BYTES+#undef MLDSA_PK_T1_OFFSET+#undef MLDSA_POLYETA_PACKEDBYTES+#undef MLDSA_POLYT0_PACKEDBYTES+#undef MLDSA_POLYT1_PACKEDBYTES+#undef MLDSA_POLYVECH_PACKEDBYTES+#undef MLDSA_POLYW1_PACKEDBYTES+#undef MLDSA_POLYW1_PACKEDBYTES_32+#undef MLDSA_POLYW1_PACKEDBYTES_88+#undef MLDSA_POLYZ_PACKEDBYTES+#undef MLDSA_Q+#undef MLDSA_Q_HALF+#undef MLDSA_RNDBYTES+#undef MLDSA_SEEDBYTES+#undef MLDSA_SIG_C_BYTES+#undef MLDSA_SIG_C_OFFSET+#undef MLDSA_SIG_END+#undef MLDSA_SIG_H_BYTES+#undef MLDSA_SIG_H_OFFSET+#undef MLDSA_SIG_Z_BYTES+#undef MLDSA_SIG_Z_OFFSET+#undef MLDSA_SK_END+#undef MLDSA_SK_KEY_BYTES+#undef MLDSA_SK_KEY_OFFSET+#undef MLDSA_SK_RHO_BYTES+#undef MLDSA_SK_RHO_OFFSET+#undef MLDSA_SK_S1_BYTES+#undef MLDSA_SK_S1_OFFSET+#undef MLDSA_SK_S2_BYTES+#undef MLDSA_SK_S2_OFFSET+#undef MLDSA_SK_T0_BYTES+#undef MLDSA_SK_T0_OFFSET+#undef MLDSA_SK_TR_BYTES+#undef MLDSA_SK_TR_OFFSET+#undef MLDSA_TAU+#undef MLDSA_TRBYTES+#undef MLD_MAX_KAPPA+#undef MLD_PARAMS_H+/* mldsa/src/poly_kl.h */+#undef MLD_POLYETA_UNPACK_LOWER_BOUND+#undef MLD_POLY_KL_H+#undef mld_poly_challenge+#undef mld_poly_decompose+#undef mld_poly_uniform_eta+#undef mld_poly_uniform_eta_4x+#undef mld_poly_uniform_gamma1+#undef mld_poly_uniform_gamma1_4x+#undef mld_poly_use_hint+#undef mld_polyeta_pack+#undef mld_polyeta_unpack+#undef mld_polyw1_pack+#undef mld_polyz_pack+#undef mld_polyz_unpack+/* mldsa/src/polyvec.h */+#undef MLD_POLYVEC_H+#undef mld_polyveck+#undef mld_polyveck_caddq+#undef mld_polyveck_chknorm+#undef mld_polyveck_decompose+#undef mld_polyveck_invntt_tomont+#undef mld_polyveck_ntt+#undef mld_polyveck_pack_eta+#undef mld_polyveck_pack_w1+#undef mld_polyveck_reduce+#undef mld_polyveck_unpack_eta+#undef mld_polyvecl+#undef mld_polyvecl_chknorm+#undef mld_polyvecl_ntt+#undef mld_polyvecl_pack_eta+#undef mld_polyvecl_pointwise_acc_montgomery+#undef mld_polyvecl_uniform_gamma1+#undef mld_polyvecl_unpack_eta+#undef mld_polyvecl_unpack_z+/* mldsa/src/polyvec_lazy.h */+#undef MLD_POLYVEC_LAZY_H+#undef mld_poly_permute_bitrev_to_custom_optional+#undef mld_polymat+#undef mld_polymat_eager+#undef mld_polymat_lazy+#undef mld_polyvec_matrix_expand+#undef mld_polyvec_matrix_expand_eager+#undef mld_polyvec_matrix_expand_lazy+#undef mld_polyvec_matrix_pointwise_montgomery+#undef mld_polyvec_matrix_pointwise_montgomery_row+#undef mld_polyvec_matrix_pointwise_montgomery_row_eager+#undef mld_polyvec_matrix_pointwise_montgomery_row_lazy+#undef mld_polyvec_matrix_pointwise_montgomery_yvec+#undef mld_polyvec_matrix_pointwise_montgomery_yvec_eager+#undef mld_polyvec_matrix_pointwise_montgomery_yvec_lazy+#undef mld_sk_s1hat+#undef mld_sk_s1hat_eager+#undef mld_sk_s1hat_get_poly+#undef mld_sk_s1hat_get_poly_eager+#undef mld_sk_s1hat_get_poly_lazy+#undef mld_sk_s1hat_lazy+#undef mld_sk_s2hat+#undef mld_sk_s2hat_eager+#undef mld_sk_s2hat_get_poly+#undef mld_sk_s2hat_get_poly_eager+#undef mld_sk_s2hat_get_poly_lazy+#undef mld_sk_s2hat_lazy+#undef mld_sk_t0hat+#undef mld_sk_t0hat_eager+#undef mld_sk_t0hat_get_poly+#undef mld_sk_t0hat_get_poly_eager+#undef mld_sk_t0hat_get_poly_lazy+#undef mld_sk_t0hat_lazy+#undef mld_unpack_sk_s1hat+#undef mld_unpack_sk_s1hat_eager+#undef mld_unpack_sk_s1hat_lazy+#undef mld_unpack_sk_s2hat+#undef mld_unpack_sk_s2hat_eager+#undef mld_unpack_sk_s2hat_lazy+#undef mld_unpack_sk_t0hat+#undef mld_unpack_sk_t0hat_eager+#undef mld_unpack_sk_t0hat_lazy+#undef mld_yvec+#undef mld_yvec_eager+#undef mld_yvec_get_poly+#undef mld_yvec_get_poly_eager+#undef mld_yvec_get_poly_lazy+#undef mld_yvec_init+#undef mld_yvec_init_eager+#undef mld_yvec_init_lazy+#undef mld_yvec_lazy+/* mldsa/src/rounding.h */+#undef MLD_2_POW_D+#undef MLD_ROUNDING_H+#undef mld_decompose+#undef mld_make_hint+#undef mld_power2round+#undef mld_use_hint+/* mldsa/src/sign.h */+#undef MLD_DOMAIN_SEPARATION_MAX_BYTES+#undef MLD_PREHASH_NONE+#undef MLD_PREHASH_SHA2_224+#undef MLD_PREHASH_SHA2_256+#undef MLD_PREHASH_SHA2_384+#undef MLD_PREHASH_SHA2_512+#undef MLD_PREHASH_SHA2_512_224+#undef MLD_PREHASH_SHA2_512_256+#undef MLD_PREHASH_SHA3_224+#undef MLD_PREHASH_SHA3_256+#undef MLD_PREHASH_SHA3_384+#undef MLD_PREHASH_SHA3_512+#undef MLD_PREHASH_SHAKE_128+#undef MLD_PREHASH_SHAKE_256+#undef MLD_SIGN_H+#undef mld_prepare_domain_separation_prefix+#undef mld_sign_keypair+#undef mld_sign_keypair_internal+#undef mld_sign_pk_from_sk+#undef mld_sign_signature+#undef mld_sign_signature_extmu+#undef mld_sign_signature_internal+#undef mld_sign_signature_pre_hash_internal+#undef mld_sign_signature_pre_hash_shake256+#undef mld_sign_verify+#undef mld_sign_verify_extmu+#undef mld_sign_verify_internal+#undef mld_sign_verify_pre_hash_internal+#undef mld_sign_verify_pre_hash_shake256++#if !defined(MLD_CONFIG_MONOBUILD_KEEP_SHARED_HEADERS)+/*+ * Undefine macros from MLD_CONFIG_PARAMETER_SET-generic files+ */+/* mldsa/src/context.h */+#undef MLD_CONTEXT_H+#undef MLD_CONTEXT_PARAMETERS_0+#undef MLD_CONTEXT_PARAMETERS_1+#undef MLD_CONTEXT_PARAMETERS_2+#undef MLD_CONTEXT_PARAMETERS_3+#undef MLD_CONTEXT_PARAMETERS_4+#undef MLD_CONTEXT_PARAMETERS_5+#undef MLD_CONTEXT_PARAMETERS_6+#undef MLD_CONTEXT_PARAMETERS_7+#undef MLD_CONTEXT_PARAMETERS_8+#undef MLD_CONTEXT_PARAMETERS_9+#undef MLD_CONTEXT_UNUSED+#undef mld_sign_attempt+#undef mld_sign_finish+#undef mld_sign_resume+/* mldsa/src/ct.h */+#undef MLD_CT_H+#undef MLD_USE_ASM_VALUE_BARRIER+#undef mld_ct_opt_blocker_u64+/* mldsa/src/debug.h */+#undef MLD_DEBUG_H+#undef mld_assert+#undef mld_assert_abs_bound+#undef mld_assert_abs_bound_2d+#undef mld_assert_bound+#undef mld_assert_bound_2d+#undef mld_debug_check_assert+#undef mld_debug_check_bounds+/* mldsa/src/poly.h */+#undef MLD_FQMUL_BOUND+#undef MLD_INTT_BOUND+#undef MLD_NTT_BOUND+#undef MLD_POLY_H+#undef mld_poly_add+#undef mld_poly_caddq+#undef mld_poly_chknorm+#undef mld_poly_invntt_tomont+#undef mld_poly_ntt+#undef mld_poly_pointwise_montgomery+#undef mld_poly_power2round+#undef mld_poly_reduce+#undef mld_poly_shiftl+#undef mld_poly_sub+#undef mld_poly_uniform+#undef mld_poly_uniform_4x+#undef mld_polyt0_pack+#undef mld_polyt0_unpack+#undef mld_polyt1_pack+#undef mld_polyt1_unpack+#undef mld_polyw1_pack_32+#undef mld_polyw1_pack_88+/* mldsa/src/randombytes.h */+#undef MLD_RANDOMBYTES_H+/* mldsa/src/reduce.h */+#undef MLD_MONT+#undef MLD_REDUCE32_DOMAIN_MAX+#undef MLD_REDUCE32_RANGE_MAX+#undef MLD_REDUCE_H+/* mldsa/src/symmetric.h */+#undef MLD_STREAM128_BLOCKBYTES+#undef MLD_STREAM256_BLOCKBYTES+#undef MLD_SYMMETRIC_H+#undef mld_xof128_absorb_once+#undef mld_xof128_ctx+#undef mld_xof128_init+#undef mld_xof128_release+#undef mld_xof128_squeezeblocks+#undef mld_xof128_x4_absorb+#undef mld_xof128_x4_ctx+#undef mld_xof128_x4_init+#undef mld_xof128_x4_release+#undef mld_xof128_x4_squeezeblocks+#undef mld_xof256_absorb_once+#undef mld_xof256_ctx+#undef mld_xof256_init+#undef mld_xof256_release+#undef mld_xof256_squeezeblocks+#undef mld_xof256_x4_absorb+#undef mld_xof256_x4_ctx+#undef mld_xof256_x4_init+#undef mld_xof256_x4_release+#undef mld_xof256_x4_squeezeblocks+/* mldsa/src/sys.h */+#undef MLD_ALIGN+#undef MLD_ALIGN_UP+#undef MLD_ALWAYS_INLINE+#undef MLD_CET_ENDBR+#undef MLD_CT_TESTING_DECLASSIFY+#undef MLD_CT_TESTING_SECRET+#undef MLD_DEFAULT_ALIGN+#undef MLD_HAVE_INLINE_ASM+#undef MLD_INLINE+#undef MLD_MUST_CHECK_RETURN_VALUE+#undef MLD_NOINLINE+#undef MLD_RESTRICT+#undef MLD_STATIC_TESTABLE+#undef MLD_SYSV_ABI+#undef MLD_SYSV_ABI_SUPPORTED+#undef MLD_SYS_AARCH64+#undef MLD_SYS_AARCH64_EB+#undef MLD_SYS_AARCH64_NEON+#undef MLD_SYS_APPLE+#undef MLD_SYS_ARMV81M_MVE+#undef MLD_SYS_BIG_ENDIAN+#undef MLD_SYS_H+#undef MLD_SYS_LINUX+#undef MLD_SYS_LITTLE_ENDIAN+#undef MLD_SYS_PPC64LE+#undef MLD_SYS_RISCV32+#undef MLD_SYS_RISCV64+#undef MLD_SYS_RISCV64_RVV+#undef MLD_SYS_WINDOWS+#undef MLD_SYS_X86_64+#undef MLD_SYS_X86_64_AVX2+/* mldsa/src/cbmc.h */+#undef MLD_CBMC_H+#undef __contract__+#undef __loop__++#if !defined(MLD_CONFIG_FIPS202_CUSTOM_HEADER)+/*+ * Undefine macros from FIPS-202 files+ */+/* mldsa/src/fips202/fips202.h */+#undef MLD_FIPS202_FIPS202_H+#undef MLD_KECCAK_LANES+#undef SHA3_256_HASHBYTES+#undef SHA3_256_RATE+#undef SHA3_512_HASHBYTES+#undef SHA3_512_RATE+#undef SHAKE128_RATE+#undef SHAKE256_RATE+#undef mld_shake128_absorb+#undef mld_shake128_finalize+#undef mld_shake128_init+#undef mld_shake128_release+#undef mld_shake128_squeeze+#undef mld_shake256+#undef mld_shake256_absorb+#undef mld_shake256_finalize+#undef mld_shake256_init+#undef mld_shake256_release+#undef mld_shake256_squeeze+/* mldsa/src/fips202/fips202x4.h */+#undef MLD_FIPS202_FIPS202X4_H+#undef mld_shake128x4_absorb_once+#undef mld_shake128x4_init+#undef mld_shake128x4_release+#undef mld_shake128x4_squeezeblocks+#undef mld_shake256x4_absorb_once+#undef mld_shake256x4_init+#undef mld_shake256x4_release+#undef mld_shake256x4_squeezeblocks+/* mldsa/src/fips202/keccakf1600.h */+#undef MLD_FIPS202_KECCAKF1600_H+#undef MLD_KECCAK_LANES+#undef MLD_KECCAK_WAY+#undef mld_keccakf1600_extract_bytes+#undef mld_keccakf1600_permute+#undef mld_keccakf1600_xor_bytes+#undef mld_keccakf1600x4_extract_bytes+#undef mld_keccakf1600x4_permute+#undef mld_keccakf1600x4_xor_bytes+#endif /* !MLD_CONFIG_FIPS202_CUSTOM_HEADER */++#if defined(MLD_CONFIG_USE_NATIVE_BACKEND_FIPS202)+/* mldsa/src/fips202/native/api.h */+#undef MLD_FIPS202_NATIVE_API_H+#undef MLD_NATIVE_FUNC_FALLBACK+#undef MLD_NATIVE_FUNC_SUCCESS+/* mldsa/src/fips202/native/auto.h */+#undef MLD_FIPS202_NATIVE_AUTO_H+#if defined(MLD_SYS_AARCH64)+/*+ * Undefine macros from native code (FIPS202, AArch64)+ */+/* mldsa/src/fips202/native/aarch64/auto.h */+#undef MLD_FIPS202_NATIVE_AARCH64_AUTO_H+/* mldsa/src/fips202/native/aarch64/src/fips202_native_aarch64.h */+#undef MLD_FIPS202_NATIVE_AARCH64_SRC_FIPS202_NATIVE_AARCH64_H+#undef mld_keccak_f1600_x1_scalar_aarch64_asm+#undef mld_keccak_f1600_x1_v84a_aarch64_asm+#undef mld_keccak_f1600_x2_v84a_aarch64_asm+#undef mld_keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm+#undef mld_keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm+#undef mld_keccakf1600_round_constants+/* mldsa/src/fips202/native/aarch64/x1_scalar.h */+#undef MLD_FIPS202_AARCH64_NEED_X1_SCALAR+#undef MLD_FIPS202_NATIVE_AARCH64_X1_SCALAR_H+#undef MLD_USE_NATIVE_FIPS202_X1+/* mldsa/src/fips202/native/aarch64/x1_v84a.h */+#undef MLD_FIPS202_AARCH64_NEED_X1_V84A+#undef MLD_FIPS202_NATIVE_AARCH64_X1_V84A_H+#undef MLD_USE_NATIVE_FIPS202_X1+/* mldsa/src/fips202/native/aarch64/x2_v84a.h */+#undef MLD_FIPS202_AARCH64_NEED_X2_V84A+#undef MLD_FIPS202_NATIVE_AARCH64_X2_V84A_H+#undef MLD_USE_NATIVE_FIPS202_X4+/* mldsa/src/fips202/native/aarch64/x4_v8a_scalar.h */+#undef MLD_FIPS202_AARCH64_NEED_X4_V8A_SCALAR_HYBRID+#undef MLD_FIPS202_NATIVE_AARCH64_X4_V8A_SCALAR_H+#undef MLD_USE_NATIVE_FIPS202_X4+/* mldsa/src/fips202/native/aarch64/x4_v8a_v84a_scalar.h */+#undef MLD_FIPS202_AARCH64_NEED_X4_V8A_V84A_SCALAR_HYBRID+#undef MLD_FIPS202_NATIVE_AARCH64_X4_V8A_V84A_SCALAR_H+#undef MLD_USE_NATIVE_FIPS202_X4+#endif /* MLD_SYS_AARCH64 */+#if defined(MLD_SYS_X86_64)+/*+ * Undefine macros from native code (FIPS202, x86_64)+ */+/* mldsa/src/fips202/native/x86_64/keccak_f1600_x4_avx2.h */+#undef MLD_FIPS202_NATIVE_X86_64_KECCAK_F1600_X4_AVX2_H+#undef MLD_FIPS202_X86_64_NEED_X4_AVX2+#undef MLD_USE_NATIVE_FIPS202_X4+/* mldsa/src/fips202/native/x86_64/src/fips202_native_x86_64.h */+#undef MLD_FIPS202_NATIVE_X86_64_SRC_FIPS202_NATIVE_X86_64_H+#undef mld_keccak_f1600_x4_avx2_asm+#undef mld_keccak_rho56+#undef mld_keccak_rho8+#undef mld_keccakf1600_round_constants+#endif /* MLD_SYS_X86_64 */+#if defined(MLD_SYS_ARMV81M_MVE)+/*+ * Undefine macros from native code (FIPS202, Armv8.1-M)+ */+/* mldsa/src/fips202/native/armv81m/mve.h */+#undef MLD_FIPS202_ARMV81M_NEED_X4+#undef MLD_FIPS202_NATIVE_ARMV81M+#undef MLD_FIPS202_NATIVE_ARMV81M_MVE_H+#undef MLD_USE_NATIVE_FIPS202_X4+#undef MLD_USE_NATIVE_FIPS202_X4_EXTRACT_BYTES+#undef MLD_USE_NATIVE_FIPS202_X4_XOR_BYTES+#undef mld_keccak_f1600_x4_native_impl+/* mldsa/src/fips202/native/armv81m/src/fips202_native_armv81m.h */+#undef MLD_FIPS202_NATIVE_ARMV81M_SRC_FIPS202_NATIVE_ARMV81M_H+#undef mld_keccak_f1600_x4_mve_asm+#undef mld_keccak_f1600_x4_state_extract_bytes_asm+#undef mld_keccak_f1600_x4_state_xor_bytes_asm+#undef mld_keccakf1600_round_constants+#endif /* MLD_SYS_ARMV81M_MVE */+#endif /* MLD_CONFIG_USE_NATIVE_BACKEND_FIPS202 */+#if defined(MLD_CONFIG_USE_NATIVE_BACKEND_ARITH)+/* mldsa/src/native/api.h */+#undef MLD_FQMUL_BOUND+#undef MLD_INTT_BOUND+#undef MLD_NATIVE_API_H+#undef MLD_NATIVE_FUNC_FALLBACK+#undef MLD_NATIVE_FUNC_SUCCESS+#undef MLD_NTT_BOUND+#undef MLD_REDUCE32_RANGE_MAX+/* mldsa/src/native/meta.h */+#undef MLD_NATIVE_META_H+#if defined(MLD_SYS_AARCH64)+/*+ * Undefine macros from native code (Arith, AArch64)+ */+/* mldsa/src/native/aarch64/meta.h */+#undef MLD_ARITH_BACKEND_AARCH64+#undef MLD_NATIVE_AARCH64_META_H+#undef MLD_USE_NATIVE_INTT+#undef MLD_USE_NATIVE_NTT+#undef MLD_USE_NATIVE_POINTWISE_MONTGOMERY+#undef MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L4+#undef MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L5+#undef MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L7+#undef MLD_USE_NATIVE_POLYZ_UNPACK_17+#undef MLD_USE_NATIVE_POLYZ_UNPACK_19+#undef MLD_USE_NATIVE_POLY_CADDQ+#undef MLD_USE_NATIVE_POLY_CHKNORM+#undef MLD_USE_NATIVE_POLY_DECOMPOSE_32+#undef MLD_USE_NATIVE_POLY_DECOMPOSE_88+#undef MLD_USE_NATIVE_POLY_USE_HINT_32+#undef MLD_USE_NATIVE_POLY_USE_HINT_88+#undef MLD_USE_NATIVE_REJ_UNIFORM+#undef MLD_USE_NATIVE_REJ_UNIFORM_ETA2+#undef MLD_USE_NATIVE_REJ_UNIFORM_ETA4+/* mldsa/src/native/aarch64/src/arith_native_aarch64.h */+#undef MLD_AARCH64_REJ_UNIFORM_ETA2_BUFLEN+#undef MLD_AARCH64_REJ_UNIFORM_ETA4_BUFLEN+#undef MLD_NATIVE_AARCH64_SRC_ARITH_NATIVE_AARCH64_H+#undef mld_aarch64_intt_zetas_layer123456+#undef mld_aarch64_intt_zetas_layer78+#undef mld_aarch64_ntt_zetas_layer123456+#undef mld_aarch64_ntt_zetas_layer78+#undef mld_intt_aarch64_asm+#undef mld_ntt_aarch64_asm+#undef mld_poly_caddq_aarch64_asm+#undef mld_poly_chknorm_aarch64_asm+#undef mld_poly_decompose_32_aarch64_asm+#undef mld_poly_decompose_88_aarch64_asm+#undef mld_poly_pointwise_montgomery_aarch64_asm+#undef mld_poly_use_hint_32_aarch64_asm+#undef mld_poly_use_hint_88_aarch64_asm+#undef mld_polyvecl_pointwise_acc_montgomery_l4_aarch64_asm+#undef mld_polyvecl_pointwise_acc_montgomery_l5_aarch64_asm+#undef mld_polyvecl_pointwise_acc_montgomery_l7_aarch64_asm+#undef mld_polyz_unpack_17_aarch64_asm+#undef mld_polyz_unpack_17_indices+#undef mld_polyz_unpack_19_aarch64_asm+#undef mld_polyz_unpack_19_indices+#undef mld_rej_uniform_aarch64_asm+#undef mld_rej_uniform_eta2_aarch64_asm+#undef mld_rej_uniform_eta4_aarch64_asm+#undef mld_rej_uniform_eta_table+#undef mld_rej_uniform_table+#endif /* MLD_SYS_AARCH64 */+#if defined(MLD_SYS_X86_64)+/*+ * Undefine macros from native code (Arith, X86_64)+ */+/* mldsa/src/native/x86_64/meta.h */+#undef MLD_ARITH_BACKEND_X86_64_DEFAULT+#undef MLD_NATIVE_X86_64_META_H+#undef MLD_USE_NATIVE_INTT+#undef MLD_USE_NATIVE_NTT+#undef MLD_USE_NATIVE_NTT_CUSTOM_ORDER+#undef MLD_USE_NATIVE_POINTWISE_MONTGOMERY+#undef MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L4+#undef MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L5+#undef MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L7+#undef MLD_USE_NATIVE_POLYZ_UNPACK_17+#undef MLD_USE_NATIVE_POLYZ_UNPACK_19+#undef MLD_USE_NATIVE_POLY_CADDQ+#undef MLD_USE_NATIVE_POLY_CHKNORM+#undef MLD_USE_NATIVE_POLY_DECOMPOSE_32+#undef MLD_USE_NATIVE_POLY_DECOMPOSE_88+#undef MLD_USE_NATIVE_POLY_USE_HINT_32+#undef MLD_USE_NATIVE_POLY_USE_HINT_88+#undef MLD_USE_NATIVE_REJ_UNIFORM+#undef MLD_USE_NATIVE_REJ_UNIFORM_ETA2+#undef MLD_USE_NATIVE_REJ_UNIFORM_ETA4+/* mldsa/src/native/x86_64/src/arith_native_x86_64.h */+#undef MLD_AVX2_REJ_UNIFORM_BUFLEN+#undef MLD_AVX2_REJ_UNIFORM_ETA2_BUFLEN+#undef MLD_AVX2_REJ_UNIFORM_ETA4_BUFLEN+#undef MLD_NATIVE_X86_64_SRC_ARITH_NATIVE_X86_64_H+#undef mld_invntt_avx2_asm+#undef mld_ntt_avx2_asm+#undef mld_nttunpack_avx2_asm+#undef mld_pointwise_acc_l4_avx2_asm+#undef mld_pointwise_acc_l5_avx2_asm+#undef mld_pointwise_acc_l7_avx2_asm+#undef mld_pointwise_avx2_asm+#undef mld_poly_caddq_avx2_asm+#undef mld_poly_chknorm_avx2_asm+#undef mld_poly_decompose_32_avx2_asm+#undef mld_poly_decompose_88_avx2_asm+#undef mld_poly_use_hint_32_avx2_asm+#undef mld_poly_use_hint_88_avx2_asm+#undef mld_polyz_unpack_17_avx2_asm+#undef mld_polyz_unpack_19_avx2_asm+#undef mld_rej_uniform_avx2_asm+#undef mld_rej_uniform_eta2_avx2_asm+#undef mld_rej_uniform_eta4_avx2_asm+#undef mld_rej_uniform_table+/* mldsa/src/native/x86_64/src/consts.h */+#undef MLD_AVX2_BACKEND_DATA_OFFSET_8XDIV+#undef MLD_AVX2_BACKEND_DATA_OFFSET_8XDIV_QINV+#undef MLD_AVX2_BACKEND_DATA_OFFSET_8XQ+#undef MLD_AVX2_BACKEND_DATA_OFFSET_8XQINV+#undef MLD_AVX2_BACKEND_DATA_OFFSET_ZETAS+#undef MLD_AVX2_BACKEND_DATA_OFFSET_ZETAS_QINV+#undef MLD_NATIVE_X86_64_SRC_CONSTS_H+#undef mld_qdata+#endif /* MLD_SYS_X86_64 */+#endif /* MLD_CONFIG_USE_NATIVE_BACKEND_ARITH */+#endif /* !MLD_CONFIG_MONOBUILD_KEEP_SHARED_HEADERS */
@@ -0,0 +1,956 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS204]+ * FIPS 204 Module-Lattice-Based Digital Signature Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/204/final+ */++#ifndef MLD_H+#define MLD_H++/*+ * Public API for mldsa-native+ *+ * This header defines the public API of a single build of mldsa-native.+ *+ * Make sure the configuration file is in the include path+ * (this is "mldsa_native_config.h" by default, or MLD_CONFIG_FILE if defined).+ *+ * # API conventions+ *+ * Conventions shared by all functions below (return values, pointer validity,+ * output buffers on error) are documented in API-CONVENTIONS.md.+ *+ * # Multi-level builds+ *+ * This header specifies a build of mldsa-native for a fixed security level.+ * If you need multiple security levels, leave the security level unspecified+ * in the configuration file and include this header multiple times, setting+ * MLD_CONFIG_PARAMETER_SET accordingly for each, and #undef'ing the MLD_H+ * guard to allow multiple inclusions.+ */++/******************************* Key sizes ************************************/++/* Sizes of cryptographic material, per parameter set */+/* See mldsa/src/params.h for the arithmetic expressions giving rise to these */+/* check-magic: off */+#define MLDSA44_SECRETKEYBYTES 2560+#define MLDSA44_PUBLICKEYBYTES 1312+#define MLDSA44_BYTES 2420++#define MLDSA65_SECRETKEYBYTES 4032+#define MLDSA65_PUBLICKEYBYTES 1952+#define MLDSA65_BYTES 3309++#define MLDSA87_SECRETKEYBYTES 4896+#define MLDSA87_PUBLICKEYBYTES 2592+#define MLDSA87_BYTES 4627+/* check-magic: on */++/* Size of seed and randomness in bytes (level-independent) */+#define MLDSA_SEEDBYTES 32+#define MLDSA44_SEEDBYTES MLDSA_SEEDBYTES+#define MLDSA65_SEEDBYTES MLDSA_SEEDBYTES+#define MLDSA87_SEEDBYTES MLDSA_SEEDBYTES++/* Size of CRH output in bytes (level-independent) */+#define MLDSA_CRHBYTES 64+#define MLDSA44_CRHBYTES MLDSA_CRHBYTES+#define MLDSA65_CRHBYTES MLDSA_CRHBYTES+#define MLDSA87_CRHBYTES MLDSA_CRHBYTES++/* Size of TR in bytes (level-independent)+ *+ * TR = SHAKE256(pk, 64) is the hash of the public key. Callers of the+ * external-mu API (signature_extmu / verify_extmu) that compute the message+ * representative mu = SHAKE256(TR || M', MLDSA_CRHBYTES) themselves -- e.g. to+ * sign or verify a message that is too large or streamed to hold in memory --+ * need this constant to size the TR buffer. */+#define MLDSA_TRBYTES 64+#define MLDSA44_TRBYTES MLDSA_TRBYTES+#define MLDSA65_TRBYTES MLDSA_TRBYTES+#define MLDSA87_TRBYTES MLDSA_TRBYTES++/* Size of randomness for signing in bytes (level-independent) */+#define MLDSA_RNDBYTES 32+#define MLDSA44_RNDBYTES MLDSA_RNDBYTES+#define MLDSA65_RNDBYTES MLDSA_RNDBYTES+#define MLDSA87_RNDBYTES MLDSA_RNDBYTES++/* Sizes of cryptographic material, as a function of LVL=44,65,87 */+#define MLDSA_SECRETKEYBYTES_(LVL) MLDSA##LVL##_SECRETKEYBYTES+#define MLDSA_PUBLICKEYBYTES_(LVL) MLDSA##LVL##_PUBLICKEYBYTES+#define MLDSA_BYTES_(LVL) MLDSA##LVL##_BYTES+#define MLDSA_SECRETKEYBYTES(LVL) MLDSA_SECRETKEYBYTES_(LVL)+#define MLDSA_PUBLICKEYBYTES(LVL) MLDSA_PUBLICKEYBYTES_(LVL)+#define MLDSA_BYTES(LVL) MLDSA_BYTES_(LVL)++/****************************** Error codes ***********************************/++/* Generic failure condition, reserved for failures not covered by a more+ * specific error code. */+#define MLD_ERR_FAIL (-1)+/* An allocation failed. This can only happen if MLD_CONFIG_CUSTOM_ALLOC_FREE+ * is defined and the provided MLD_CUSTOM_ALLOC can fail. */+#define MLD_ERR_OUT_OF_MEMORY (-2)+/* An RNG failure occurred. Might be due to insufficient entropy or+ * system misconfiguration. */+#define MLD_ERR_RNG_FAIL (-3)+/* The signing rejection-sampling loop exceeded+ * MLD_CONFIG_MAX_SIGNING_ATTEMPTS iterations without producing a valid+ * signature. With a FIPS 204 Appendix C compliant bound (>= 821) this+ * has probability < 2^-256. */+#define MLD_ERR_SIGN_ATTEMPTS_EXHAUSTED (-4)+/* Signing was paused before completing, at the request of a caller-provided+ * MLD_CONFIG_SIGN_HOOK_ATTEMPT hook (see mldsa_native_config.h). The caller+ * resumes by re-invoking signing with the same inputs; the attempt hook,+ * together with MLD_CONFIG_SIGN_HOOK_RESUME, decides where to continue. */+#define MLD_ERR_SIGNING_PAUSED (-5)+/* Signature verification failed: the signature is not valid for the given+ * message and public key. Returned by the verification API. */+#define MLD_ERR_INVALID_SIGNATURE (-6)+/* Secret key validation failed: the secret key is malformed or internally+ * inconsistent. Returned by pk_from_sk. */+#define MLD_ERR_INVALID_KEY (-7)+/* The Pairwise Consistency Test failed. Only possible when+ * MLD_CONFIG_KEYGEN_PCT is enabled; signals that the freshly generated key+ * pair failed its sign/verify self-test. */+#define MLD_ERR_PCT_FAIL (-8)+/* An argument was invalid, e.g. an unsupported pre-hash algorithm or a context+ * string longer than 255 bytes. */+#define MLD_ERR_INVALID_ARG (-9)++/********************* Namespacing and Qualifiers *****************************/++#define MLD_API_CONCAT_(x, y) x##y+#define MLD_API_CONCAT(x, y) MLD_API_CONCAT_(x, y)+#define MLD_API_CONCAT_UNDERSCORE(x, y) MLD_API_CONCAT(MLD_API_CONCAT(x, _), y)++/* You need to make sure the config file is in the include path. */+#if defined(MLD_CONFIG_FILE)+#include MLD_CONFIG_FILE+#else+#include "mldsa_native_config.h"+#endif++/* Namespace prefix for the public API symbols. For multi-level builds, the+ * parameter set is appended to disambiguate the security levels. */+#if defined(MLD_CONFIG_MULTILEVEL_BUILD)+#define MLD_API_NAMESPACE_PREFIX \+ MLD_API_CONCAT(MLD_CONFIG_NAMESPACE_PREFIX, MLD_CONFIG_PARAMETER_SET)+#else+#define MLD_API_NAMESPACE_PREFIX MLD_CONFIG_NAMESPACE_PREFIX+#endif++#define MLD_API_NAMESPACE(sym) \+ MLD_API_CONCAT_UNDERSCORE(MLD_API_NAMESPACE_PREFIX, sym)++#if defined(__GNUC__) || defined(__clang__)+#define MLD_API_MUST_CHECK_RETURN_VALUE __attribute__((warn_unused_result))+#else+#define MLD_API_MUST_CHECK_RETURN_VALUE+#endif++#if defined(MLD_CONFIG_EXTERNAL_API_QUALIFIER)+#define MLD_API_QUALIFIER MLD_CONFIG_EXTERNAL_API_QUALIFIER+#else+#define MLD_API_QUALIFIER+#endif++/* Hash algorithm constants for domain separation */+#define MLD_PREHASH_NONE 0+#define MLD_PREHASH_SHA2_224 1+#define MLD_PREHASH_SHA2_256 2+#define MLD_PREHASH_SHA2_384 3+#define MLD_PREHASH_SHA2_512 4+#define MLD_PREHASH_SHA2_512_224 5+#define MLD_PREHASH_SHA2_512_256 6+#define MLD_PREHASH_SHA3_224 7+#define MLD_PREHASH_SHA3_256 8+#define MLD_PREHASH_SHA3_384 9+#define MLD_PREHASH_SHA3_512 10+#define MLD_PREHASH_SHAKE_128 11+#define MLD_PREHASH_SHAKE_256 12++/* Maximum formatted domain separation message length */+#define MLD_DOMAIN_SEPARATION_MAX_BYTES (2 + 255 + 11 + 64)++/****************************** Function API **********************************/++#if !defined(MLD_CONFIG_CONSTANTS_ONLY)++#include <stddef.h>+#include <stdint.h>++#ifdef __cplusplus+extern "C"+{+#endif++#if !defined(MLD_CONFIG_NO_KEYPAIR_API)+/**+ * Generate a public-private key pair from a seed.+ *+ * When MLD_CONFIG_KEYGEN_PCT is set, performs a Pairwise Consistency Test+ * (PCT) as required by FIPS 140-3 IG.+ *+ * @warning The seed must be generated by a cryptographically secure random+ * number generator.+ *+ * @spec{Implements @[FIPS204, Algorithm 6, ML-DSA.KeyGen_internal].}+ *+ * @param[out] pk Output public key.+ * @param[out] sk Output private key.+ * @param[in] seed Input random seed.+ * @param context Application context. Only present when+ * MLD_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLD_ERR_OUT_OF_MEMORY MLD_CONFIG_CUSTOM_ALLOC_FREE was+ * used and an allocation via+ * MLD_CUSTOM_ALLOC returned NULL.+ * @retval MLD_ERR_RNG_FAIL Random number generation failed+ * during the PCT. Only possible when+ * MLD_CONFIG_KEYGEN_PCT is enabled.+ * @retval MLD_ERR_SIGNING_PAUSED The PCT's signing step was paused by+ * a MLD_CONFIG_SIGN_HOOK_ATTEMPT hook.+ * This should currently never happen:+ * signing hooks require+ * MLD_CONFIG_NO_RANDOMIZED_API, which+ * is incompatible with+ * MLD_CONFIG_KEYGEN_PCT, so the two+ * cannot be enabled simultaneously.+ * @retval MLD_ERR_PCT_FAIL MLD_CONFIG_KEYGEN_PCT is enabled and+ * the PCT check failed.+ */+MLD_API_QUALIFIER+MLD_API_MUST_CHECK_RETURN_VALUE+int MLD_API_NAMESPACE(keypair_internal)(+ uint8_t pk[MLDSA_PUBLICKEYBYTES(MLD_CONFIG_PARAMETER_SET)],+ uint8_t sk[MLDSA_SECRETKEYBYTES(MLD_CONFIG_PARAMETER_SET)],+ const uint8_t seed[MLDSA_SEEDBYTES]+#ifdef MLD_CONFIG_CONTEXT_PARAMETER+ ,+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context+#endif+);++#if !defined(MLD_CONFIG_CORE_API_ONLY)+#if !defined(MLD_CONFIG_NO_RANDOMIZED_API)+/**+ * Generate a public-private key pair.+ *+ * When MLD_CONFIG_KEYGEN_PCT is set, performs a Pairwise Consistency Test+ * (PCT) as required by FIPS 140-3 IG.+ *+ * @spec{Implements @[FIPS204, Algorithm 1, ML-DSA.KeyGen].}+ *+ * @param[out] pk Output public key.+ * @param[out] sk Output private key.+ * @param context Application context. Only present when+ * MLD_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLD_ERR_OUT_OF_MEMORY MLD_CONFIG_CUSTOM_ALLOC_FREE was+ * used and an allocation via+ * MLD_CUSTOM_ALLOC returned NULL.+ * @retval MLD_ERR_RNG_FAIL Random number generation failed.+ * @retval MLD_ERR_SIGNING_PAUSED The PCT's signing step was paused by+ * a MLD_CONFIG_SIGN_HOOK_ATTEMPT hook.+ * This should currently never happen:+ * signing hooks require+ * MLD_CONFIG_NO_RANDOMIZED_API, which+ * is incompatible with+ * MLD_CONFIG_KEYGEN_PCT, so the two+ * cannot be enabled simultaneously.+ * @retval MLD_ERR_PCT_FAIL MLD_CONFIG_KEYGEN_PCT is enabled and+ * the PCT check failed.+ */+MLD_API_QUALIFIER+MLD_API_MUST_CHECK_RETURN_VALUE+int MLD_API_NAMESPACE(keypair)(+ uint8_t pk[MLDSA_PUBLICKEYBYTES(MLD_CONFIG_PARAMETER_SET)],+ uint8_t sk[MLDSA_SECRETKEYBYTES(MLD_CONFIG_PARAMETER_SET)]+#ifdef MLD_CONFIG_CONTEXT_PARAMETER+ ,+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context+#endif+);+#endif /* !MLD_CONFIG_NO_RANDOMIZED_API */+#endif /* !MLD_CONFIG_CORE_API_ONLY */+#endif /* !MLD_CONFIG_NO_KEYPAIR_API */++#if !defined(MLD_CONFIG_NO_SIGN_API)+/**+ * Compute signature using a caller-supplied random seed and prefix.+ *+ * On error (non-zero return value), the signature buffer sig is zeroized.+ *+ * @spec{Implements @[FIPS204, Algorithm 7, ML-DSA.Sign_internal].}+ *+ * @warning This function does not perform secret key validation.+ * Callers importing serialized keys can use pk_from_sk+ * to validate them before signing.+ *+ * @param[out] sig Pointer to buffer to hold the generated signature of+ * MLDSA_BYTES(MLD_CONFIG_PARAMETER_SET) bytes.+ * @param[in] m Pointer to message to be signed (when+ * externalmu == 0), or to a precomputed+ * message representative mu (when externalmu != 0).+ * @param mlen Length of m. Must equal MLDSA_CRHBYTES when+ * externalmu != 0.+ * @param[in] pre Pointer to prefix string. Ignored when+ * externalmu != 0.+ * @param prelen Length of prefix string. Ignored when+ * externalmu != 0.+ * @param[in] rnd Random seed.+ * @param[in] sk Bit-packed secret key; assumed to be valid.+ * @param externalmu 0: m/mlen is the raw message; mu = H(tr, pre, m) is+ * computed internally.+ * non-zero: m points to a precomputed mu of+ * MLDSA_CRHBYTES bytes; pre/prelen unused.+ * @param context Application context. Only present when+ * MLD_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLD_ERR_OUT_OF_MEMORY MLD_CONFIG_CUSTOM_ALLOC_FREE was+ * used and an allocation via+ * MLD_CUSTOM_ALLOC returned NULL.+ * @retval MLD_ERR_SIGN_ATTEMPTS_EXHAUSTED The rejection-sampling loop exceeded+ * MLD_CONFIG_MAX_SIGNING_ATTEMPTS+ * iterations.+ * @retval MLD_ERR_SIGNING_PAUSED A MLD_CONFIG_SIGN_HOOK_ATTEMPT hook+ * paused signing; re-invoke to resume.+ */+MLD_API_QUALIFIER+MLD_API_MUST_CHECK_RETURN_VALUE+int MLD_API_NAMESPACE(signature_internal)(+ uint8_t sig[MLDSA_BYTES(MLD_CONFIG_PARAMETER_SET)], const uint8_t *m,+ size_t mlen, const uint8_t *pre, size_t prelen,+ const uint8_t rnd[MLDSA_RNDBYTES],+ const uint8_t sk[MLDSA_SECRETKEYBYTES(MLD_CONFIG_PARAMETER_SET)],+ int externalmu+#ifdef MLD_CONFIG_CONTEXT_PARAMETER+ ,+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context+#endif+);++#if !defined(MLD_CONFIG_CORE_API_ONLY)+#if !defined(MLD_CONFIG_NO_RANDOMIZED_API)+/**+ * Compute signature. This function implements the randomized variant of+ * ML-DSA. If you require the deterministic variant, use+ * signature_internal directly.+ *+ * @spec{Implements @[FIPS204, Algorithm 2, ML-DSA.Sign].}+ *+ * @warning This function does not perform secret key validation.+ * Callers importing serialized keys can use pk_from_sk+ * to validate them before signing.+ *+ * @param[out] sig Pointer to buffer to hold the generated signature of+ * MLDSA_BYTES(MLD_CONFIG_PARAMETER_SET) bytes.+ * @param[in] m Pointer to message to be signed. May be NULL if+ * mlen == 0.+ * @param mlen Length of message.+ * @param[in] ctx Pointer to context string. May be NULL if ctxlen == 0.+ * @param ctxlen Length of context string. Should be <= 255.+ * @param[in] sk Bit-packed secret key; assumed to be valid.+ * @param context Application context. Only present when+ * MLD_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLD_ERR_OUT_OF_MEMORY MLD_CONFIG_CUSTOM_ALLOC_FREE was+ * used and an allocation via+ * MLD_CUSTOM_ALLOC returned NULL.+ * @retval MLD_ERR_RNG_FAIL Random number generation failed.+ * @retval MLD_ERR_SIGN_ATTEMPTS_EXHAUSTED The rejection-sampling loop exceeded+ * MLD_CONFIG_MAX_SIGNING_ATTEMPTS+ * iterations.+ * @retval MLD_ERR_SIGNING_PAUSED A MLD_CONFIG_SIGN_HOOK_ATTEMPT hook+ * paused signing; re-invoke to resume.+ * @retval MLD_ERR_INVALID_ARG The context string exceeded 255+ * bytes.+ */+MLD_API_QUALIFIER+MLD_API_MUST_CHECK_RETURN_VALUE+int MLD_API_NAMESPACE(signature)(+ uint8_t sig[MLDSA_BYTES(MLD_CONFIG_PARAMETER_SET)], const uint8_t *m,+ size_t mlen, const uint8_t *ctx, size_t ctxlen,+ const uint8_t sk[MLDSA_SECRETKEYBYTES(MLD_CONFIG_PARAMETER_SET)]+#ifdef MLD_CONFIG_CONTEXT_PARAMETER+ ,+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context+#endif+);++/**+ * Compute signature in "external mu" mode: the caller has already computed+ * the message representative mu = SHAKE256(tr || M', 64), where+ * tr = SHAKE256(pk, 64) and M' is the FIPS 204 formatted message (e.g.+ * 0x00 || ctxlen || ctx || msg for pure ML-DSA). This is useful when the+ * message is large or streamed and cannot be held in memory.+ *+ * @spec{Implements @[FIPS204, Algorithm 2, ML-DSA.Sign external mu variant].}+ *+ * @warning This function does not perform secret key validation.+ * Callers importing serialized keys can use pk_from_sk+ * to validate them before signing.+ *+ * @param[out] sig Pointer to buffer to hold the generated signature of+ * MLDSA_BYTES(MLD_CONFIG_PARAMETER_SET) bytes.+ * @param[in] mu Precomputed message representative.+ * @param[in] sk Bit-packed secret key; assumed to be valid.+ * @param context Application context. Only present when+ * MLD_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLD_ERR_OUT_OF_MEMORY MLD_CONFIG_CUSTOM_ALLOC_FREE was+ * used and an allocation via+ * MLD_CUSTOM_ALLOC returned NULL.+ * @retval MLD_ERR_RNG_FAIL Random number generation failed.+ * @retval MLD_ERR_SIGN_ATTEMPTS_EXHAUSTED The rejection-sampling loop exceeded+ * MLD_CONFIG_MAX_SIGNING_ATTEMPTS+ * iterations.+ * @retval MLD_ERR_SIGNING_PAUSED A MLD_CONFIG_SIGN_HOOK_ATTEMPT hook+ * paused signing; re-invoke to resume.+ */+MLD_API_QUALIFIER+MLD_API_MUST_CHECK_RETURN_VALUE+int MLD_API_NAMESPACE(signature_extmu)(+ uint8_t sig[MLDSA_BYTES(MLD_CONFIG_PARAMETER_SET)],+ const uint8_t mu[MLDSA_CRHBYTES],+ const uint8_t sk[MLDSA_SECRETKEYBYTES(MLD_CONFIG_PARAMETER_SET)]+#ifdef MLD_CONFIG_CONTEXT_PARAMETER+ ,+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context+#endif+);++#endif /* !MLD_CONFIG_NO_RANDOMIZED_API */+#endif /* !MLD_CONFIG_CORE_API_ONLY */+#endif /* !MLD_CONFIG_NO_SIGN_API */++#if !defined(MLD_CONFIG_NO_VERIFY_API)+/**+ * Verify signature. Internal API.+ *+ * @spec{Implements @[FIPS204, Algorithm 8, ML-DSA.Verify_internal].}+ *+ * @param[in] sig Pointer to input signature of+ * MLDSA_BYTES(MLD_CONFIG_PARAMETER_SET) bytes.+ * @param[in] m Pointer to message (when externalmu == 0), or to a+ * precomputed message representative mu (when+ * externalmu != 0).+ * @param mlen Length of m. Must equal MLDSA_CRHBYTES when+ * externalmu != 0.+ * @param[in] pre Pointer to prefix string. Ignored when externalmu != 0.+ * @param prelen Length of prefix string. Ignored when externalmu != 0.+ * @param[in] pk Bit-packed public key.+ * @param externalmu 0: m/mlen is the raw message; mu = H(H(pk), pre, m) is+ * computed internally.+ * non-zero: m points to a precomputed mu of+ * MLDSA_CRHBYTES bytes; pre/prelen unused.+ * @param context Application context. Only present when+ * MLD_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLD_ERR_OUT_OF_MEMORY MLD_CONFIG_CUSTOM_ALLOC_FREE was+ * used and an allocation via+ * MLD_CUSTOM_ALLOC returned NULL.+ * @retval MLD_ERR_INVALID_SIGNATURE Signature verification failed.+ */+MLD_API_QUALIFIER+MLD_API_MUST_CHECK_RETURN_VALUE+int MLD_API_NAMESPACE(verify_internal)(+ const uint8_t sig[MLDSA_BYTES(MLD_CONFIG_PARAMETER_SET)], const uint8_t *m,+ size_t mlen, const uint8_t *pre, size_t prelen,+ const uint8_t pk[MLDSA_PUBLICKEYBYTES(MLD_CONFIG_PARAMETER_SET)],+ int externalmu+#ifdef MLD_CONFIG_CONTEXT_PARAMETER+ ,+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context+#endif+);++#if !defined(MLD_CONFIG_CORE_API_ONLY)+/**+ * Verify signature.+ *+ * @spec{Implements @[FIPS204, Algorithm 3, ML-DSA.Verify].}+ *+ * @param[in] sig Pointer to input signature of+ * MLDSA_BYTES(MLD_CONFIG_PARAMETER_SET) bytes.+ * @param[in] m Pointer to message. May be NULL if mlen == 0.+ * @param mlen Length of message.+ * @param[in] ctx Pointer to context string. May be NULL if ctxlen == 0.+ * @param ctxlen Length of context string.+ * @param[in] pk Bit-packed public key.+ * @param context Application context. Only present when+ * MLD_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLD_ERR_OUT_OF_MEMORY MLD_CONFIG_CUSTOM_ALLOC_FREE was+ * used and an allocation via+ * MLD_CUSTOM_ALLOC returned NULL.+ * @retval MLD_ERR_INVALID_SIGNATURE Signature verification failed.+ * @retval MLD_ERR_INVALID_ARG The context string exceeded 255 bytes.+ */+MLD_API_QUALIFIER+MLD_API_MUST_CHECK_RETURN_VALUE+int MLD_API_NAMESPACE(verify)(+ const uint8_t sig[MLDSA_BYTES(MLD_CONFIG_PARAMETER_SET)], const uint8_t *m,+ size_t mlen, const uint8_t *ctx, size_t ctxlen,+ const uint8_t pk[MLDSA_PUBLICKEYBYTES(MLD_CONFIG_PARAMETER_SET)]+#ifdef MLD_CONFIG_CONTEXT_PARAMETER+ ,+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context+#endif+);++/**+ * Verify signature in "external mu" mode: the caller has already computed+ * the message representative mu = SHAKE256(tr || M', 64), where+ * tr = SHAKE256(pk, 64) and M' is the FIPS 204 formatted message (e.g.+ * 0x00 || ctxlen || ctx || msg for pure ML-DSA). The same mu must have+ * been used at signing time.+ *+ * @spec{Implements @[FIPS204, Algorithm 3, ML-DSA.Verify external mu variant].}+ *+ * @param[in] sig Pointer to input signature of+ * MLDSA_BYTES(MLD_CONFIG_PARAMETER_SET) bytes.+ * @param[in] mu Precomputed message representative.+ * @param[in] pk Bit-packed public key.+ * @param context Application context. Only present when+ * MLD_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLD_ERR_OUT_OF_MEMORY MLD_CONFIG_CUSTOM_ALLOC_FREE was+ * used and an allocation via+ * MLD_CUSTOM_ALLOC returned NULL.+ * @retval MLD_ERR_INVALID_SIGNATURE Signature verification failed.+ */+MLD_API_QUALIFIER+MLD_API_MUST_CHECK_RETURN_VALUE+int MLD_API_NAMESPACE(verify_extmu)(+ const uint8_t sig[MLDSA_BYTES(MLD_CONFIG_PARAMETER_SET)],+ const uint8_t mu[MLDSA_CRHBYTES],+ const uint8_t pk[MLDSA_PUBLICKEYBYTES(MLD_CONFIG_PARAMETER_SET)]+#ifdef MLD_CONFIG_CONTEXT_PARAMETER+ ,+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context+#endif+);++#endif /* !MLD_CONFIG_CORE_API_ONLY */+#endif /* !MLD_CONFIG_NO_VERIFY_API */++#if !defined(MLD_CONFIG_CORE_API_ONLY)+#if !defined(MLD_CONFIG_NO_SIGN_API)+/**+ * Compute signature with pre-hashed message.+ *+ * @spec{Implements @[FIPS204, Algorithm 4, HashML-DSA.Sign].}+ *+ * Supported hash algorithm constants:+ * MLD_PREHASH_SHA2_224, MLD_PREHASH_SHA2_256, MLD_PREHASH_SHA2_384,+ * MLD_PREHASH_SHA2_512, MLD_PREHASH_SHA2_512_224, MLD_PREHASH_SHA2_512_256,+ * MLD_PREHASH_SHA3_224, MLD_PREHASH_SHA3_256, MLD_PREHASH_SHA3_384,+ * MLD_PREHASH_SHA3_512, MLD_PREHASH_SHAKE_128, MLD_PREHASH_SHAKE_256.+ *+ * MLD_PREHASH_NONE is rejected by this API.+ *+ * @warning This is an unstable API that may change in the future. If you need+ * a stable API use signature_pre_hash_shake256.+ *+ * @warning This function does not perform secret key validation.+ * Callers importing serialized keys can use pk_from_sk+ * to validate them before signing.+ *+ * @param[out] sig Pointer to buffer to hold the generated signature of+ * MLDSA_BYTES(MLD_CONFIG_PARAMETER_SET) bytes.+ * @param[in] ph Pointer to pre-hashed message.+ * @param phlen Length of pre-hashed message. Must match the output+ * length of hashalg (the digest size for SHA-2/SHA-3,+ * 32 bytes for MLD_PREHASH_SHAKE_128, 64 bytes for+ * MLD_PREHASH_SHAKE_256).+ * @param[in] ctx Pointer to context string. May be NULL if ctxlen == 0.+ * @param ctxlen Length of context string.+ * @param[in] rnd Random seed.+ * @param[in] sk Bit-packed secret key; assumed to be valid.+ * @param hashalg Hash algorithm constant (one of MLD_PREHASH_*).+ * @param context Application context. Only present when+ * MLD_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLD_ERR_OUT_OF_MEMORY MLD_CONFIG_CUSTOM_ALLOC_FREE was+ * used and an allocation via+ * MLD_CUSTOM_ALLOC returned NULL.+ * @retval MLD_ERR_SIGN_ATTEMPTS_EXHAUSTED The rejection-sampling loop exceeded+ * MLD_CONFIG_MAX_SIGNING_ATTEMPTS+ * iterations.+ * @retval MLD_ERR_SIGNING_PAUSED A MLD_CONFIG_SIGN_HOOK_ATTEMPT hook+ * paused signing; re-invoke to resume.+ * @retval MLD_ERR_INVALID_ARG The pre-hash algorithm was+ * MLD_PREHASH_NONE or unsupported,+ * phlen did not match the output+ * length of hashalg, or the context+ * string exceeded 255 bytes.+ */+MLD_API_QUALIFIER+MLD_API_MUST_CHECK_RETURN_VALUE+int MLD_API_NAMESPACE(signature_pre_hash_internal)(+ uint8_t sig[MLDSA_BYTES(MLD_CONFIG_PARAMETER_SET)], const uint8_t *ph,+ size_t phlen, const uint8_t *ctx, size_t ctxlen,+ const uint8_t rnd[MLDSA_RNDBYTES],+ const uint8_t sk[MLDSA_SECRETKEYBYTES(MLD_CONFIG_PARAMETER_SET)],+ int hashalg+#ifdef MLD_CONFIG_CONTEXT_PARAMETER+ ,+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context+#endif+);+#endif /* !MLD_CONFIG_NO_SIGN_API */++#if !defined(MLD_CONFIG_NO_VERIFY_API)+/**+ * Verifies signature with pre-hashed message.+ *+ * @spec{Implements @[FIPS204, Algorithm 5, HashML-DSA.Verify].}+ *+ * Supported hash algorithm constants:+ * MLD_PREHASH_SHA2_224, MLD_PREHASH_SHA2_256, MLD_PREHASH_SHA2_384,+ * MLD_PREHASH_SHA2_512, MLD_PREHASH_SHA2_512_224, MLD_PREHASH_SHA2_512_256,+ * MLD_PREHASH_SHA3_224, MLD_PREHASH_SHA3_256, MLD_PREHASH_SHA3_384,+ * MLD_PREHASH_SHA3_512, MLD_PREHASH_SHAKE_128, MLD_PREHASH_SHAKE_256.+ *+ * MLD_PREHASH_NONE is rejected by this API.+ *+ * @warning This is an unstable API that may change in the future. If you need+ * a stable API use verify_pre_hash_shake256.+ *+ * @param[in] sig Pointer to input signature of+ * MLDSA_BYTES(MLD_CONFIG_PARAMETER_SET) bytes.+ * @param[in] ph Pointer to pre-hashed message.+ * @param phlen Length of pre-hashed message. Must match the output+ * length of hashalg (the digest size for SHA-2/SHA-3,+ * 32 bytes for MLD_PREHASH_SHAKE_128, 64 bytes for+ * MLD_PREHASH_SHAKE_256).+ * @param[in] ctx Pointer to context string. May be NULL if ctxlen == 0.+ * @param ctxlen Length of context string.+ * @param[in] pk Bit-packed public key.+ * @param hashalg Hash algorithm constant (one of MLD_PREHASH_*).+ * @param context Application context. Only present when+ * MLD_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLD_ERR_OUT_OF_MEMORY MLD_CONFIG_CUSTOM_ALLOC_FREE was+ * used and an allocation via+ * MLD_CUSTOM_ALLOC returned NULL.+ * @retval MLD_ERR_INVALID_SIGNATURE Signature verification failed.+ * @retval MLD_ERR_INVALID_ARG The pre-hash algorithm was+ * MLD_PREHASH_NONE or unsupported, phlen+ * did not match the output length of+ * hashalg, or the context string exceeded+ * 255 bytes.+ */+MLD_API_QUALIFIER+MLD_API_MUST_CHECK_RETURN_VALUE+int MLD_API_NAMESPACE(verify_pre_hash_internal)(+ const uint8_t sig[MLDSA_BYTES(MLD_CONFIG_PARAMETER_SET)], const uint8_t *ph,+ size_t phlen, const uint8_t *ctx, size_t ctxlen,+ const uint8_t pk[MLDSA_PUBLICKEYBYTES(MLD_CONFIG_PARAMETER_SET)],+ int hashalg+#ifdef MLD_CONFIG_CONTEXT_PARAMETER+ ,+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context+#endif+);+#endif /* !MLD_CONFIG_NO_VERIFY_API */++#if !defined(MLD_CONFIG_NO_SIGN_API)+/**+ * Compute signature with pre-hashed message using SHAKE256. This function+ * computes the SHAKE256 hash of the message internally.+ *+ * @spec{Implements @[FIPS204, Algorithm 4, HashML-DSA.Sign] with SHAKE256 as+ * the pre-hash.}+ *+ * @warning This function does not perform secret key validation.+ * Callers importing serialized keys can use pk_from_sk+ * to validate them before signing.+ *+ * @param[out] sig Pointer to buffer to hold the generated signature of+ * MLDSA_BYTES(MLD_CONFIG_PARAMETER_SET) bytes.+ * @param[in] m Pointer to message to be hashed and signed. May be+ * NULL if mlen == 0.+ * @param mlen Length of message.+ * @param[in] ctx Pointer to context string. May be NULL if ctxlen == 0.+ * @param ctxlen Length of context string.+ * @param[in] rnd Random seed.+ * @param[in] sk Bit-packed secret key; assumed to be valid.+ * @param context Application context. Only present when+ * MLD_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLD_ERR_OUT_OF_MEMORY MLD_CONFIG_CUSTOM_ALLOC_FREE was+ * used and an allocation via+ * MLD_CUSTOM_ALLOC returned NULL.+ * @retval MLD_ERR_SIGN_ATTEMPTS_EXHAUSTED The rejection-sampling loop exceeded+ * MLD_CONFIG_MAX_SIGNING_ATTEMPTS+ * iterations.+ * @retval MLD_ERR_SIGNING_PAUSED A MLD_CONFIG_SIGN_HOOK_ATTEMPT hook+ * paused signing; re-invoke to resume.+ * @retval MLD_ERR_INVALID_ARG The context string exceeded 255+ * bytes.+ */+MLD_API_QUALIFIER+MLD_API_MUST_CHECK_RETURN_VALUE+int MLD_API_NAMESPACE(signature_pre_hash_shake256)(+ uint8_t sig[MLDSA_BYTES(MLD_CONFIG_PARAMETER_SET)], const uint8_t *m,+ size_t mlen, const uint8_t *ctx, size_t ctxlen,+ const uint8_t rnd[MLDSA_RNDBYTES],+ const uint8_t sk[MLDSA_SECRETKEYBYTES(MLD_CONFIG_PARAMETER_SET)]+#ifdef MLD_CONFIG_CONTEXT_PARAMETER+ ,+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context+#endif+);+#endif /* !MLD_CONFIG_NO_SIGN_API */++#if !defined(MLD_CONFIG_NO_VERIFY_API)+/**+ * Verify signature with pre-hashed message using SHAKE256. This function+ * computes the SHAKE256 hash of the message internally.+ *+ * @spec{Implements @[FIPS204, Algorithm 5, HashML-DSA.Verify] with SHAKE256 as+ * the pre-hash.}+ *+ * @param[in] sig Pointer to input signature of+ * MLDSA_BYTES(MLD_CONFIG_PARAMETER_SET) bytes.+ * @param[in] m Pointer to message to be hashed and verified. May be+ * NULL if mlen == 0.+ * @param mlen Length of message.+ * @param[in] ctx Pointer to context string. May be NULL if ctxlen == 0.+ * @param ctxlen Length of context string.+ * @param[in] pk Bit-packed public key.+ * @param context Application context. Only present when+ * MLD_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLD_ERR_OUT_OF_MEMORY MLD_CONFIG_CUSTOM_ALLOC_FREE was+ * used and an allocation via+ * MLD_CUSTOM_ALLOC returned NULL.+ * @retval MLD_ERR_INVALID_SIGNATURE Signature verification failed.+ * @retval MLD_ERR_INVALID_ARG The context string exceeded 255 bytes.+ */+MLD_API_QUALIFIER+MLD_API_MUST_CHECK_RETURN_VALUE+int MLD_API_NAMESPACE(verify_pre_hash_shake256)(+ const uint8_t sig[MLDSA_BYTES(MLD_CONFIG_PARAMETER_SET)], const uint8_t *m,+ size_t mlen, const uint8_t *ctx, size_t ctxlen,+ const uint8_t pk[MLDSA_PUBLICKEYBYTES(MLD_CONFIG_PARAMETER_SET)]+#ifdef MLD_CONFIG_CONTEXT_PARAMETER+ ,+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context+#endif+);+#endif /* !MLD_CONFIG_NO_VERIFY_API */+#endif /* !MLD_CONFIG_CORE_API_ONLY */++#if !defined(MLD_CONFIG_CORE_API_ONLY)+/**+ * Prepare domain separation prefix for ML-DSA signing.+ *+ * For pure ML-DSA (hashalg == MLD_PREHASH_NONE):+ * Format: 0x00 || ctxlen (1 byte) || ctx.+ *+ * For HashML-DSA (hashalg != MLD_PREHASH_NONE):+ * Format: 0x01 || ctxlen (1 byte) || ctx || oid (11 bytes) || ph.+ *+ * This function is useful for building incremental signing APIs.+ *+ * @spec{For HashML-DSA (hashalg != MLD_PREHASH_NONE), implements+ * @[FIPS204, Algorithm 4, line 23]. For Pure ML-DSA+ * (hashalg == MLD_PREHASH_NONE), implements+ * ```+ * M' <- BytesToBits(IntegerToBytes(0, 1)+ * || IntegerToBytes(|ctx|, 1)+ * || ctx+ * ```+ * which is part of @[FIPS204, Algorithm 2, ML-DSA.Sign, line 10] and+ * @[FIPS204, Algorithm 3, ML-DSA.Verify, line 5].}+ *+ * @param[out] prefix Output domain separation prefix buffer.+ * @param[in] ph Pointer to pre-hashed message (ignored for pure+ * ML-DSA).+ * @param phlen Length of pre-hashed message; must match the output+ * length of hashalg (ignored for pure ML-DSA).+ * @param[in] ctx Pointer to context string. May be NULL if ctxlen == 0.+ * @param ctxlen Length of context string.+ * @param hashalg Hash algorithm constant (MLD_PREHASH_NONE for pure+ * ML-DSA, or MLD_PREHASH_* for HashML-DSA).+ *+ * @return The total length of the formatted prefix, or 0 on error.+ * Errors are:+ * - The context string exceeded 255 bytes.+ * - For HashML-DSA: hashalg was unsupported, ph was NULL, or phlen+ * did not match the output length of hashalg.+ */+MLD_API_QUALIFIER+MLD_API_MUST_CHECK_RETURN_VALUE+size_t MLD_API_NAMESPACE(prepare_domain_separation_prefix)(+ uint8_t prefix[MLD_DOMAIN_SEPARATION_MAX_BYTES], const uint8_t *ph,+ size_t phlen, const uint8_t *ctx, size_t ctxlen, int hashalg);++#if !defined(MLD_CONFIG_NO_KEYPAIR_API)+/**+ * Perform basic validity checks on secret key, and derive public key.+ *+ * Referring to the decoding of the secret key `sk=(rho, K, tr, s1, s2, t0)`+ * (cf. @[FIPS204, Algorithm 25, skDecode]), the following checks are+ * performed:+ * - Check that s1 and s2 have coefficients in [-MLDSA_ETA, MLDSA_ETA].+ * - Check that t0 and tr stored in sk match recomputed values.+ *+ * @note This function leaks whether the secret key is valid or invalid+ * through its return value and timing.+ *+ * @param[out] pk Output public key.+ * @param[in] sk Input secret key.+ * @param context Application context. Only present when+ * MLD_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLD_ERR_OUT_OF_MEMORY MLD_CONFIG_CUSTOM_ALLOC_FREE was used and an+ * allocation via MLD_CUSTOM_ALLOC returned NULL.+ * @retval MLD_ERR_INVALID_KEY Secret key validation failed.+ */+MLD_API_QUALIFIER+MLD_API_MUST_CHECK_RETURN_VALUE+int MLD_API_NAMESPACE(pk_from_sk)(+ uint8_t pk[MLDSA_PUBLICKEYBYTES(MLD_CONFIG_PARAMETER_SET)],+ const uint8_t sk[MLDSA_SECRETKEYBYTES(MLD_CONFIG_PARAMETER_SET)]+#ifdef MLD_CONFIG_CONTEXT_PARAMETER+ ,+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context+#endif+);+#endif /* !MLD_CONFIG_NO_KEYPAIR_API */+#endif /* !MLD_CONFIG_CORE_API_ONLY */++#ifdef __cplusplus+}+#endif++#undef MLD_API_NAMESPACE_PREFIX++#endif /* !MLD_CONFIG_CONSTANTS_ONLY */++/***************************** Memory Usage **********************************/++/*+ * By default mldsa-native performs all memory allocations on the stack.+ * Alternatively, mldsa-native supports custom allocation of large structures+ * through the `MLD_CONFIG_CUSTOM_ALLOC_FREE` configuration option.+ * See mldsa_native_config.h for details.+ *+ * `MLD_TOTAL_ALLOC_{44,65,87}_{KEYPAIR,SIGN,VERIFY}` indicates the maximum+ * (accumulative) allocation via MLD_ALLOC for each parameter set and operation.+ * Note that some stack allocation remains even+ * when using custom allocators, so these values are lower than total stack+ * usage with the default stack-only allocation.+ *+ * These constants may be used to implement custom allocations using a+ * fixed-sized buffer and a simple allocator (e.g., bump allocator).+ */+/* check-magic: off */+#if !defined(MLD_CONFIG_REDUCE_RAM)+#define MLD_TOTAL_ALLOC_44_KEYPAIR_NO_PCT 26912+#define MLD_TOTAL_ALLOC_44_KEYPAIR_PCT 48480+#define MLD_TOTAL_ALLOC_44_PK_FROM_SK 28480+#define MLD_TOTAL_ALLOC_44_SIGN 44704+#define MLD_TOTAL_ALLOC_44_VERIFY 24448+#define MLD_TOTAL_ALLOC_65_KEYPAIR_NO_PCT 44320+#define MLD_TOTAL_ALLOC_65_KEYPAIR_PCT 74624+#define MLD_TOTAL_ALLOC_65_PK_FROM_SK 46720+#define MLD_TOTAL_ALLOC_65_SIGN 69312+#define MLD_TOTAL_ALLOC_65_VERIFY 39872+#define MLD_TOTAL_ALLOC_87_KEYPAIR_NO_PCT 75040+#define MLD_TOTAL_ALLOC_87_KEYPAIR_PCT 115488+#define MLD_TOTAL_ALLOC_87_PK_FROM_SK 78272+#define MLD_TOTAL_ALLOC_87_SIGN 108224+#define MLD_TOTAL_ALLOC_87_VERIFY 68800+#else /* !MLD_CONFIG_REDUCE_RAM */+#define MLD_TOTAL_ALLOC_44_KEYPAIR_NO_PCT 11584+#define MLD_TOTAL_ALLOC_44_KEYPAIR_PCT 16896+#define MLD_TOTAL_ALLOC_44_PK_FROM_SK 13152+#define MLD_TOTAL_ALLOC_44_SIGN 13120+#define MLD_TOTAL_ALLOC_44_VERIFY 9120+#define MLD_TOTAL_ALLOC_65_KEYPAIR_NO_PCT 14656+#define MLD_TOTAL_ALLOC_65_KEYPAIR_PCT 22560+#define MLD_TOTAL_ALLOC_65_PK_FROM_SK 17056+#define MLD_TOTAL_ALLOC_65_SIGN 17248+#define MLD_TOTAL_ALLOC_65_VERIFY 10208+#define MLD_TOTAL_ALLOC_87_KEYPAIR_NO_PCT 18752+#define MLD_TOTAL_ALLOC_87_KEYPAIR_PCT 28608+#define MLD_TOTAL_ALLOC_87_PK_FROM_SK 21984+#define MLD_TOTAL_ALLOC_87_SIGN 21344+#define MLD_TOTAL_ALLOC_87_VERIFY 12512+#endif /* MLD_CONFIG_REDUCE_RAM */+/* check-magic: on */++/*+ * MLD_TOTAL_ALLOC_*_KEYPAIR adapts based on MLD_CONFIG_KEYGEN_PCT.+ */+#if defined(MLD_CONFIG_KEYGEN_PCT)+#define MLD_TOTAL_ALLOC_44_KEYPAIR MLD_TOTAL_ALLOC_44_KEYPAIR_PCT+#define MLD_TOTAL_ALLOC_65_KEYPAIR MLD_TOTAL_ALLOC_65_KEYPAIR_PCT+#define MLD_TOTAL_ALLOC_87_KEYPAIR MLD_TOTAL_ALLOC_87_KEYPAIR_PCT+#else+#define MLD_TOTAL_ALLOC_44_KEYPAIR MLD_TOTAL_ALLOC_44_KEYPAIR_NO_PCT+#define MLD_TOTAL_ALLOC_65_KEYPAIR MLD_TOTAL_ALLOC_65_KEYPAIR_NO_PCT+#define MLD_TOTAL_ALLOC_87_KEYPAIR MLD_TOTAL_ALLOC_87_KEYPAIR_NO_PCT+#endif++#define MLD_MAX3_(a, b, c) \+ ((a) > (b) ? ((a) > (c) ? (a) : (c)) : ((b) > (c) ? (b) : (c)))+#define MLD_MAX4_(a, b, c, d) MLD_MAX3_((a), (b), MLD_MAX3_((c), (d), (d)))++/*+ * `MLD_TOTAL_ALLOC_{44,65,87}` is the maximum across standard API operations+ * (keygen, sign, verify) for each parameter set.+ */+#define MLD_TOTAL_ALLOC_44 \+ MLD_MAX4_(MLD_TOTAL_ALLOC_44_KEYPAIR, MLD_TOTAL_ALLOC_44_PK_FROM_SK, \+ MLD_TOTAL_ALLOC_44_SIGN, MLD_TOTAL_ALLOC_44_VERIFY)+#define MLD_TOTAL_ALLOC_65 \+ MLD_MAX4_(MLD_TOTAL_ALLOC_65_KEYPAIR, MLD_TOTAL_ALLOC_65_PK_FROM_SK, \+ MLD_TOTAL_ALLOC_65_SIGN, MLD_TOTAL_ALLOC_65_VERIFY)+#define MLD_TOTAL_ALLOC_87 \+ MLD_MAX4_(MLD_TOTAL_ALLOC_87_KEYPAIR, MLD_TOTAL_ALLOC_87_PK_FROM_SK, \+ MLD_TOTAL_ALLOC_87_SIGN, MLD_TOTAL_ALLOC_87_VERIFY)++#endif /* !MLD_H */
@@ -0,0 +1,830 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*+ * WARNING: This file is auto-generated from scripts/autogen+ * in the mldsa-native repository.+ * Do not modify it directly.+ */++/******************************************************************************+ *+ * Single assembly unit for fixed-level build of mldsa-native+ *+ * This assembly unit bundles together all assembly files for a build+ * of mldsa-native for a fixed security level (MLDSA-44/65/87).+ *+ * # Multi-level build+ *+ * If you want an SCU build of mldsa-native with support for multiple security+ * levels, you should include this file once with+ * MLD_CONFIG_MULTILEVEL_WITH_SHARED set.+ *+ * (You could also follow the same pattern as for mldsa_native.c+ * and include it for every level, setting MLD_CONFIG_MULTILEVEL_NO_SHARED+ * for all but one. For builds with MLD_CONFIG_MULTILEVEL_NO_SHARED, this+ * file will then be ignored.)+ *+ * # Configuration+ *+ * The following options from the mldsa-native configuration are relevant:+ *+ * - MLD_CONFIG_FIPS202_CUSTOM_HEADER+ * Set this option if you use a custom FIPS202 implementation.+ *+ * - MLD_CONFIG_USE_NATIVE_BACKEND_ARITH+ * Set this option if you want to include the native arithmetic backends+ * in your build.+ *+ * - MLD_CONFIG_USE_NATIVE_BACKEND_FIPS202+ * Set this option if you want to include the native FIPS202 backends+ * in your build.+ *+ * - MLD_CONFIG_MONOBUILD_KEEP_SHARED_HEADERS+ * Set this option if you want to keep the directives defined in+ * level-independent headers. This is needed for a multi-level build.+ */++/* If parts of the mldsa-native source tree are not used,+ * consider reducing this header via `unifdef`.+ *+ * Example:+ * ```bash+ * unifdef -UMLD_CONFIG_USE_NATIVE_BACKEND_ARITH mldsa_native_asm.S+ * ```+ */++#include "src/common.h"++#if defined(MLD_CONFIG_USE_NATIVE_BACKEND_ARITH)+#if defined(MLD_SYS_AARCH64)+#include "src/native/aarch64/src/mldsa_intt_aarch64_asm.S"+#include "src/native/aarch64/src/mldsa_ntt_aarch64_asm.S"+#include "src/native/aarch64/src/mldsa_pointwise_montgomery_aarch64_asm.S"+#include "src/native/aarch64/src/mldsa_poly_caddq_aarch64_asm.S"+#include "src/native/aarch64/src/mldsa_poly_chknorm_aarch64_asm.S"+#include "src/native/aarch64/src/mldsa_poly_decompose_32_aarch64_asm.S"+#include "src/native/aarch64/src/mldsa_poly_decompose_88_aarch64_asm.S"+#include "src/native/aarch64/src/mldsa_poly_use_hint_32_aarch64_asm.S"+#include "src/native/aarch64/src/mldsa_poly_use_hint_88_aarch64_asm.S"+#include "src/native/aarch64/src/mldsa_polyvecl_pointwise_acc_montgomery_l4_aarch64_asm.S"+#include "src/native/aarch64/src/mldsa_polyvecl_pointwise_acc_montgomery_l5_aarch64_asm.S"+#include "src/native/aarch64/src/mldsa_polyvecl_pointwise_acc_montgomery_l7_aarch64_asm.S"+#include "src/native/aarch64/src/mldsa_polyz_unpack_17_aarch64_asm.S"+#include "src/native/aarch64/src/mldsa_polyz_unpack_19_aarch64_asm.S"+#include "src/native/aarch64/src/mldsa_rej_uniform_aarch64_asm.S"+#include "src/native/aarch64/src/mldsa_rej_uniform_eta2_aarch64_asm.S"+#include "src/native/aarch64/src/mldsa_rej_uniform_eta4_aarch64_asm.S"+#endif /* MLD_SYS_AARCH64 */+#if defined(MLD_SYS_X86_64)+#include "src/native/x86_64/src/mldsa_intt_avx2_asm.S"+#include "src/native/x86_64/src/mldsa_ntt_avx2_asm.S"+#include "src/native/x86_64/src/mldsa_nttunpack_avx2_asm.S"+#include "src/native/x86_64/src/mldsa_pointwise_acc_l4_avx2_asm.S"+#include "src/native/x86_64/src/mldsa_pointwise_acc_l5_avx2_asm.S"+#include "src/native/x86_64/src/mldsa_pointwise_acc_l7_avx2_asm.S"+#include "src/native/x86_64/src/mldsa_pointwise_avx2_asm.S"+#include "src/native/x86_64/src/mldsa_poly_caddq_avx2_asm.S"+#include "src/native/x86_64/src/mldsa_poly_chknorm_avx2_asm.S"+#include "src/native/x86_64/src/mldsa_poly_decompose_32_avx2_asm.S"+#include "src/native/x86_64/src/mldsa_poly_decompose_88_avx2_asm.S"+#include "src/native/x86_64/src/mldsa_poly_use_hint_32_avx2_asm.S"+#include "src/native/x86_64/src/mldsa_poly_use_hint_88_avx2_asm.S"+#include "src/native/x86_64/src/mldsa_polyz_unpack_17_avx2_asm.S"+#include "src/native/x86_64/src/mldsa_polyz_unpack_19_avx2_asm.S"+#include "src/native/x86_64/src/mldsa_rej_uniform_avx2_asm.S"+#include "src/native/x86_64/src/mldsa_rej_uniform_eta2_avx2_asm.S"+#include "src/native/x86_64/src/mldsa_rej_uniform_eta4_avx2_asm.S"+#endif /* MLD_SYS_X86_64 */+#endif /* MLD_CONFIG_USE_NATIVE_BACKEND_ARITH */++#if defined(MLD_CONFIG_USE_NATIVE_BACKEND_FIPS202)+#if defined(MLD_SYS_AARCH64)+#include "src/fips202/native/aarch64/src/keccak_f1600_x1_scalar_aarch64_asm.S"+#include "src/fips202/native/aarch64/src/keccak_f1600_x1_v84a_aarch64_asm.S"+#include "src/fips202/native/aarch64/src/keccak_f1600_x2_v84a_aarch64_asm.S"+#include "src/fips202/native/aarch64/src/keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm.S"+#include "src/fips202/native/aarch64/src/keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm.S"+#endif /* MLD_SYS_AARCH64 */+#if defined(MLD_SYS_X86_64)+#include "src/fips202/native/x86_64/src/keccak_f1600_x4_avx2_asm.S"+#endif+#if defined(MLD_SYS_ARMV81M_MVE)+#include "src/fips202/native/armv81m/src/keccak_f1600_x4_mve.S"+#include "src/fips202/native/armv81m/src/keccak_f1600_x4_state_extract_bytes_mve.S"+#include "src/fips202/native/armv81m/src/keccak_f1600_x4_state_xor_bytes_mve.S"+#endif+#endif /* MLD_CONFIG_USE_NATIVE_BACKEND_FIPS202 */+++/* Macro #undef's+ *+ * The following undefines macros from headers+ * included by the source files imported above.+ *+ * This is to allow building and linking multiple builds+ * of mldsa-native for varying parameter sets through concatenation+ * of this file, as if the files had been compiled separately.+ * If this is not relevant to you, you may remove the following.+ *+ * NOTE: This is not needed for the assembly SCU since, at present,+ * there is no need to include it multiple times.+ * We keep it for uniformity with mldsa_native.c only.+ *+ * NOTE: To avoid having to distinguish between which headers are included+ * from the assembly files, we #undef the same set of directives+ * as in mldsa_native.c+ */++/*+ * Undefine macros from MLD_CONFIG_PARAMETER_SET-specific files+ */+/* mldsa/mldsa_native.h */+#undef MLDSA44_BYTES+#undef MLDSA44_CRHBYTES+#undef MLDSA44_PUBLICKEYBYTES+#undef MLDSA44_RNDBYTES+#undef MLDSA44_SECRETKEYBYTES+#undef MLDSA44_SEEDBYTES+#undef MLDSA44_TRBYTES+#undef MLDSA65_BYTES+#undef MLDSA65_CRHBYTES+#undef MLDSA65_PUBLICKEYBYTES+#undef MLDSA65_RNDBYTES+#undef MLDSA65_SECRETKEYBYTES+#undef MLDSA65_SEEDBYTES+#undef MLDSA65_TRBYTES+#undef MLDSA87_BYTES+#undef MLDSA87_CRHBYTES+#undef MLDSA87_PUBLICKEYBYTES+#undef MLDSA87_RNDBYTES+#undef MLDSA87_SECRETKEYBYTES+#undef MLDSA87_SEEDBYTES+#undef MLDSA87_TRBYTES+#undef MLDSA_BYTES+#undef MLDSA_BYTES_+#undef MLDSA_CRHBYTES+#undef MLDSA_PUBLICKEYBYTES+#undef MLDSA_PUBLICKEYBYTES_+#undef MLDSA_RNDBYTES+#undef MLDSA_SECRETKEYBYTES+#undef MLDSA_SECRETKEYBYTES_+#undef MLDSA_SEEDBYTES+#undef MLDSA_TRBYTES+#undef MLD_API_CONCAT+#undef MLD_API_CONCAT_+#undef MLD_API_CONCAT_UNDERSCORE+#undef MLD_API_MUST_CHECK_RETURN_VALUE+#undef MLD_API_NAMESPACE+#undef MLD_API_NAMESPACE_PREFIX+#undef MLD_API_QUALIFIER+#undef MLD_DOMAIN_SEPARATION_MAX_BYTES+#undef MLD_ERR_FAIL+#undef MLD_ERR_INVALID_ARG+#undef MLD_ERR_INVALID_KEY+#undef MLD_ERR_INVALID_SIGNATURE+#undef MLD_ERR_OUT_OF_MEMORY+#undef MLD_ERR_PCT_FAIL+#undef MLD_ERR_RNG_FAIL+#undef MLD_ERR_SIGNING_PAUSED+#undef MLD_ERR_SIGN_ATTEMPTS_EXHAUSTED+#undef MLD_H+#undef MLD_MAX3_+#undef MLD_MAX4_+#undef MLD_PREHASH_NONE+#undef MLD_PREHASH_SHA2_224+#undef MLD_PREHASH_SHA2_256+#undef MLD_PREHASH_SHA2_384+#undef MLD_PREHASH_SHA2_512+#undef MLD_PREHASH_SHA2_512_224+#undef MLD_PREHASH_SHA2_512_256+#undef MLD_PREHASH_SHA3_224+#undef MLD_PREHASH_SHA3_256+#undef MLD_PREHASH_SHA3_384+#undef MLD_PREHASH_SHA3_512+#undef MLD_PREHASH_SHAKE_128+#undef MLD_PREHASH_SHAKE_256+#undef MLD_TOTAL_ALLOC_44+#undef MLD_TOTAL_ALLOC_44_KEYPAIR+#undef MLD_TOTAL_ALLOC_44_KEYPAIR_NO_PCT+#undef MLD_TOTAL_ALLOC_44_KEYPAIR_PCT+#undef MLD_TOTAL_ALLOC_44_PK_FROM_SK+#undef MLD_TOTAL_ALLOC_44_SIGN+#undef MLD_TOTAL_ALLOC_44_VERIFY+#undef MLD_TOTAL_ALLOC_65+#undef MLD_TOTAL_ALLOC_65_KEYPAIR+#undef MLD_TOTAL_ALLOC_65_KEYPAIR_NO_PCT+#undef MLD_TOTAL_ALLOC_65_KEYPAIR_PCT+#undef MLD_TOTAL_ALLOC_65_PK_FROM_SK+#undef MLD_TOTAL_ALLOC_65_SIGN+#undef MLD_TOTAL_ALLOC_65_VERIFY+#undef MLD_TOTAL_ALLOC_87+#undef MLD_TOTAL_ALLOC_87_KEYPAIR+#undef MLD_TOTAL_ALLOC_87_KEYPAIR_NO_PCT+#undef MLD_TOTAL_ALLOC_87_KEYPAIR_PCT+#undef MLD_TOTAL_ALLOC_87_PK_FROM_SK+#undef MLD_TOTAL_ALLOC_87_SIGN+#undef MLD_TOTAL_ALLOC_87_VERIFY+/* mldsa/src/common.h */+#undef MLD_ADD_PARAM_SET+#undef MLD_ALLOC+#undef MLD_APPLY+#undef MLD_ASM_FN_SIZE+#undef MLD_ASM_FN_SYMBOL+#undef MLD_ASM_NAMESPACE+#undef MLD_BUILD_INTERNAL+#undef MLD_COMMON_H+#undef MLD_CONCAT+#undef MLD_CONCAT_+#undef MLD_EMPTY_CU+#undef MLD_ERR_FAIL+#undef MLD_ERR_INVALID_ARG+#undef MLD_ERR_INVALID_KEY+#undef MLD_ERR_INVALID_SIGNATURE+#undef MLD_ERR_OUT_OF_MEMORY+#undef MLD_ERR_PCT_FAIL+#undef MLD_ERR_RNG_FAIL+#undef MLD_ERR_SIGNING_PAUSED+#undef MLD_ERR_SIGN_ATTEMPTS_EXHAUSTED+#undef MLD_EXTERNAL_API+#undef MLD_FIPS202X4_HEADER_FILE+#undef MLD_FIPS202_HEADER_FILE+#undef MLD_FREE+#undef MLD_INTERNAL_API+#undef MLD_INTERNAL_DATA_DECLARATION+#undef MLD_INTERNAL_DATA_DEFINITION+#undef MLD_MULTILEVEL_BUILD+#undef MLD_NAMESPACE+#undef MLD_NAMESPACE_KL+#undef MLD_NAMESPACE_PREFIX+#undef MLD_NAMESPACE_PREFIX_KL+#undef mld_memcpy+#undef mld_memset+/* mldsa/src/packing.h */+#undef MLD_PACKING_H+#undef mld_pack_sig_c+#undef mld_pack_sig_h+#undef mld_pack_sig_z+#undef mld_pack_sk_rho_key_tr_s2+#undef mld_pack_sk_s1+#undef mld_sig_unpack_hints+#undef mld_unpack_pk_t1+#undef mld_unpack_sk+/* mldsa/src/params.h */+#undef MLDSA_BETA+#undef MLDSA_CRHBYTES+#undef MLDSA_CRYPTO_BYTES+#undef MLDSA_CRYPTO_PUBLICKEYBYTES+#undef MLDSA_CRYPTO_SECRETKEYBYTES+#undef MLDSA_CTILDEBYTES+#undef MLDSA_D+#undef MLDSA_ETA+#undef MLDSA_GAMMA1+#undef MLDSA_GAMMA2+#undef MLDSA_GAMMA2_32+#undef MLDSA_GAMMA2_88+#undef MLDSA_K+#undef MLDSA_L+#undef MLDSA_N+#undef MLDSA_OMEGA+#undef MLDSA_PK_END+#undef MLDSA_PK_RHO_BYTES+#undef MLDSA_PK_RHO_OFFSET+#undef MLDSA_PK_T1_BYTES+#undef MLDSA_PK_T1_OFFSET+#undef MLDSA_POLYETA_PACKEDBYTES+#undef MLDSA_POLYT0_PACKEDBYTES+#undef MLDSA_POLYT1_PACKEDBYTES+#undef MLDSA_POLYVECH_PACKEDBYTES+#undef MLDSA_POLYW1_PACKEDBYTES+#undef MLDSA_POLYW1_PACKEDBYTES_32+#undef MLDSA_POLYW1_PACKEDBYTES_88+#undef MLDSA_POLYZ_PACKEDBYTES+#undef MLDSA_Q+#undef MLDSA_Q_HALF+#undef MLDSA_RNDBYTES+#undef MLDSA_SEEDBYTES+#undef MLDSA_SIG_C_BYTES+#undef MLDSA_SIG_C_OFFSET+#undef MLDSA_SIG_END+#undef MLDSA_SIG_H_BYTES+#undef MLDSA_SIG_H_OFFSET+#undef MLDSA_SIG_Z_BYTES+#undef MLDSA_SIG_Z_OFFSET+#undef MLDSA_SK_END+#undef MLDSA_SK_KEY_BYTES+#undef MLDSA_SK_KEY_OFFSET+#undef MLDSA_SK_RHO_BYTES+#undef MLDSA_SK_RHO_OFFSET+#undef MLDSA_SK_S1_BYTES+#undef MLDSA_SK_S1_OFFSET+#undef MLDSA_SK_S2_BYTES+#undef MLDSA_SK_S2_OFFSET+#undef MLDSA_SK_T0_BYTES+#undef MLDSA_SK_T0_OFFSET+#undef MLDSA_SK_TR_BYTES+#undef MLDSA_SK_TR_OFFSET+#undef MLDSA_TAU+#undef MLDSA_TRBYTES+#undef MLD_MAX_KAPPA+#undef MLD_PARAMS_H+/* mldsa/src/poly_kl.h */+#undef MLD_POLYETA_UNPACK_LOWER_BOUND+#undef MLD_POLY_KL_H+#undef mld_poly_challenge+#undef mld_poly_decompose+#undef mld_poly_uniform_eta+#undef mld_poly_uniform_eta_4x+#undef mld_poly_uniform_gamma1+#undef mld_poly_uniform_gamma1_4x+#undef mld_poly_use_hint+#undef mld_polyeta_pack+#undef mld_polyeta_unpack+#undef mld_polyw1_pack+#undef mld_polyz_pack+#undef mld_polyz_unpack+/* mldsa/src/polyvec.h */+#undef MLD_POLYVEC_H+#undef mld_polyveck+#undef mld_polyveck_caddq+#undef mld_polyveck_chknorm+#undef mld_polyveck_decompose+#undef mld_polyveck_invntt_tomont+#undef mld_polyveck_ntt+#undef mld_polyveck_pack_eta+#undef mld_polyveck_pack_w1+#undef mld_polyveck_reduce+#undef mld_polyveck_unpack_eta+#undef mld_polyvecl+#undef mld_polyvecl_chknorm+#undef mld_polyvecl_ntt+#undef mld_polyvecl_pack_eta+#undef mld_polyvecl_pointwise_acc_montgomery+#undef mld_polyvecl_uniform_gamma1+#undef mld_polyvecl_unpack_eta+#undef mld_polyvecl_unpack_z+/* mldsa/src/polyvec_lazy.h */+#undef MLD_POLYVEC_LAZY_H+#undef mld_poly_permute_bitrev_to_custom_optional+#undef mld_polymat+#undef mld_polymat_eager+#undef mld_polymat_lazy+#undef mld_polyvec_matrix_expand+#undef mld_polyvec_matrix_expand_eager+#undef mld_polyvec_matrix_expand_lazy+#undef mld_polyvec_matrix_pointwise_montgomery+#undef mld_polyvec_matrix_pointwise_montgomery_row+#undef mld_polyvec_matrix_pointwise_montgomery_row_eager+#undef mld_polyvec_matrix_pointwise_montgomery_row_lazy+#undef mld_polyvec_matrix_pointwise_montgomery_yvec+#undef mld_polyvec_matrix_pointwise_montgomery_yvec_eager+#undef mld_polyvec_matrix_pointwise_montgomery_yvec_lazy+#undef mld_sk_s1hat+#undef mld_sk_s1hat_eager+#undef mld_sk_s1hat_get_poly+#undef mld_sk_s1hat_get_poly_eager+#undef mld_sk_s1hat_get_poly_lazy+#undef mld_sk_s1hat_lazy+#undef mld_sk_s2hat+#undef mld_sk_s2hat_eager+#undef mld_sk_s2hat_get_poly+#undef mld_sk_s2hat_get_poly_eager+#undef mld_sk_s2hat_get_poly_lazy+#undef mld_sk_s2hat_lazy+#undef mld_sk_t0hat+#undef mld_sk_t0hat_eager+#undef mld_sk_t0hat_get_poly+#undef mld_sk_t0hat_get_poly_eager+#undef mld_sk_t0hat_get_poly_lazy+#undef mld_sk_t0hat_lazy+#undef mld_unpack_sk_s1hat+#undef mld_unpack_sk_s1hat_eager+#undef mld_unpack_sk_s1hat_lazy+#undef mld_unpack_sk_s2hat+#undef mld_unpack_sk_s2hat_eager+#undef mld_unpack_sk_s2hat_lazy+#undef mld_unpack_sk_t0hat+#undef mld_unpack_sk_t0hat_eager+#undef mld_unpack_sk_t0hat_lazy+#undef mld_yvec+#undef mld_yvec_eager+#undef mld_yvec_get_poly+#undef mld_yvec_get_poly_eager+#undef mld_yvec_get_poly_lazy+#undef mld_yvec_init+#undef mld_yvec_init_eager+#undef mld_yvec_init_lazy+#undef mld_yvec_lazy+/* mldsa/src/rounding.h */+#undef MLD_2_POW_D+#undef MLD_ROUNDING_H+#undef mld_decompose+#undef mld_make_hint+#undef mld_power2round+#undef mld_use_hint+/* mldsa/src/sign.h */+#undef MLD_DOMAIN_SEPARATION_MAX_BYTES+#undef MLD_PREHASH_NONE+#undef MLD_PREHASH_SHA2_224+#undef MLD_PREHASH_SHA2_256+#undef MLD_PREHASH_SHA2_384+#undef MLD_PREHASH_SHA2_512+#undef MLD_PREHASH_SHA2_512_224+#undef MLD_PREHASH_SHA2_512_256+#undef MLD_PREHASH_SHA3_224+#undef MLD_PREHASH_SHA3_256+#undef MLD_PREHASH_SHA3_384+#undef MLD_PREHASH_SHA3_512+#undef MLD_PREHASH_SHAKE_128+#undef MLD_PREHASH_SHAKE_256+#undef MLD_SIGN_H+#undef mld_prepare_domain_separation_prefix+#undef mld_sign_keypair+#undef mld_sign_keypair_internal+#undef mld_sign_pk_from_sk+#undef mld_sign_signature+#undef mld_sign_signature_extmu+#undef mld_sign_signature_internal+#undef mld_sign_signature_pre_hash_internal+#undef mld_sign_signature_pre_hash_shake256+#undef mld_sign_verify+#undef mld_sign_verify_extmu+#undef mld_sign_verify_internal+#undef mld_sign_verify_pre_hash_internal+#undef mld_sign_verify_pre_hash_shake256++#if !defined(MLD_CONFIG_MONOBUILD_KEEP_SHARED_HEADERS)+/*+ * Undefine macros from MLD_CONFIG_PARAMETER_SET-generic files+ */+/* mldsa/src/context.h */+#undef MLD_CONTEXT_H+#undef MLD_CONTEXT_PARAMETERS_0+#undef MLD_CONTEXT_PARAMETERS_1+#undef MLD_CONTEXT_PARAMETERS_2+#undef MLD_CONTEXT_PARAMETERS_3+#undef MLD_CONTEXT_PARAMETERS_4+#undef MLD_CONTEXT_PARAMETERS_5+#undef MLD_CONTEXT_PARAMETERS_6+#undef MLD_CONTEXT_PARAMETERS_7+#undef MLD_CONTEXT_PARAMETERS_8+#undef MLD_CONTEXT_PARAMETERS_9+#undef MLD_CONTEXT_UNUSED+#undef mld_sign_attempt+#undef mld_sign_finish+#undef mld_sign_resume+/* mldsa/src/ct.h */+#undef MLD_CT_H+#undef MLD_USE_ASM_VALUE_BARRIER+#undef mld_ct_opt_blocker_u64+/* mldsa/src/debug.h */+#undef MLD_DEBUG_H+#undef mld_assert+#undef mld_assert_abs_bound+#undef mld_assert_abs_bound_2d+#undef mld_assert_bound+#undef mld_assert_bound_2d+#undef mld_debug_check_assert+#undef mld_debug_check_bounds+/* mldsa/src/poly.h */+#undef MLD_FQMUL_BOUND+#undef MLD_INTT_BOUND+#undef MLD_NTT_BOUND+#undef MLD_POLY_H+#undef mld_poly_add+#undef mld_poly_caddq+#undef mld_poly_chknorm+#undef mld_poly_invntt_tomont+#undef mld_poly_ntt+#undef mld_poly_pointwise_montgomery+#undef mld_poly_power2round+#undef mld_poly_reduce+#undef mld_poly_shiftl+#undef mld_poly_sub+#undef mld_poly_uniform+#undef mld_poly_uniform_4x+#undef mld_polyt0_pack+#undef mld_polyt0_unpack+#undef mld_polyt1_pack+#undef mld_polyt1_unpack+#undef mld_polyw1_pack_32+#undef mld_polyw1_pack_88+/* mldsa/src/randombytes.h */+#undef MLD_RANDOMBYTES_H+/* mldsa/src/reduce.h */+#undef MLD_MONT+#undef MLD_REDUCE32_DOMAIN_MAX+#undef MLD_REDUCE32_RANGE_MAX+#undef MLD_REDUCE_H+/* mldsa/src/symmetric.h */+#undef MLD_STREAM128_BLOCKBYTES+#undef MLD_STREAM256_BLOCKBYTES+#undef MLD_SYMMETRIC_H+#undef mld_xof128_absorb_once+#undef mld_xof128_ctx+#undef mld_xof128_init+#undef mld_xof128_release+#undef mld_xof128_squeezeblocks+#undef mld_xof128_x4_absorb+#undef mld_xof128_x4_ctx+#undef mld_xof128_x4_init+#undef mld_xof128_x4_release+#undef mld_xof128_x4_squeezeblocks+#undef mld_xof256_absorb_once+#undef mld_xof256_ctx+#undef mld_xof256_init+#undef mld_xof256_release+#undef mld_xof256_squeezeblocks+#undef mld_xof256_x4_absorb+#undef mld_xof256_x4_ctx+#undef mld_xof256_x4_init+#undef mld_xof256_x4_release+#undef mld_xof256_x4_squeezeblocks+/* mldsa/src/sys.h */+#undef MLD_ALIGN+#undef MLD_ALIGN_UP+#undef MLD_ALWAYS_INLINE+#undef MLD_CET_ENDBR+#undef MLD_CT_TESTING_DECLASSIFY+#undef MLD_CT_TESTING_SECRET+#undef MLD_DEFAULT_ALIGN+#undef MLD_HAVE_INLINE_ASM+#undef MLD_INLINE+#undef MLD_MUST_CHECK_RETURN_VALUE+#undef MLD_NOINLINE+#undef MLD_RESTRICT+#undef MLD_STATIC_TESTABLE+#undef MLD_SYSV_ABI+#undef MLD_SYSV_ABI_SUPPORTED+#undef MLD_SYS_AARCH64+#undef MLD_SYS_AARCH64_EB+#undef MLD_SYS_AARCH64_NEON+#undef MLD_SYS_APPLE+#undef MLD_SYS_ARMV81M_MVE+#undef MLD_SYS_BIG_ENDIAN+#undef MLD_SYS_H+#undef MLD_SYS_LINUX+#undef MLD_SYS_LITTLE_ENDIAN+#undef MLD_SYS_PPC64LE+#undef MLD_SYS_RISCV32+#undef MLD_SYS_RISCV64+#undef MLD_SYS_RISCV64_RVV+#undef MLD_SYS_WINDOWS+#undef MLD_SYS_X86_64+#undef MLD_SYS_X86_64_AVX2+/* mldsa/src/cbmc.h */+#undef MLD_CBMC_H+#undef __contract__+#undef __loop__++#if !defined(MLD_CONFIG_FIPS202_CUSTOM_HEADER)+/*+ * Undefine macros from FIPS-202 files+ */+/* mldsa/src/fips202/fips202.h */+#undef MLD_FIPS202_FIPS202_H+#undef MLD_KECCAK_LANES+#undef SHA3_256_HASHBYTES+#undef SHA3_256_RATE+#undef SHA3_512_HASHBYTES+#undef SHA3_512_RATE+#undef SHAKE128_RATE+#undef SHAKE256_RATE+#undef mld_shake128_absorb+#undef mld_shake128_finalize+#undef mld_shake128_init+#undef mld_shake128_release+#undef mld_shake128_squeeze+#undef mld_shake256+#undef mld_shake256_absorb+#undef mld_shake256_finalize+#undef mld_shake256_init+#undef mld_shake256_release+#undef mld_shake256_squeeze+/* mldsa/src/fips202/fips202x4.h */+#undef MLD_FIPS202_FIPS202X4_H+#undef mld_shake128x4_absorb_once+#undef mld_shake128x4_init+#undef mld_shake128x4_release+#undef mld_shake128x4_squeezeblocks+#undef mld_shake256x4_absorb_once+#undef mld_shake256x4_init+#undef mld_shake256x4_release+#undef mld_shake256x4_squeezeblocks+/* mldsa/src/fips202/keccakf1600.h */+#undef MLD_FIPS202_KECCAKF1600_H+#undef MLD_KECCAK_LANES+#undef MLD_KECCAK_WAY+#undef mld_keccakf1600_extract_bytes+#undef mld_keccakf1600_permute+#undef mld_keccakf1600_xor_bytes+#undef mld_keccakf1600x4_extract_bytes+#undef mld_keccakf1600x4_permute+#undef mld_keccakf1600x4_xor_bytes+#endif /* !MLD_CONFIG_FIPS202_CUSTOM_HEADER */++#if defined(MLD_CONFIG_USE_NATIVE_BACKEND_FIPS202)+/* mldsa/src/fips202/native/api.h */+#undef MLD_FIPS202_NATIVE_API_H+#undef MLD_NATIVE_FUNC_FALLBACK+#undef MLD_NATIVE_FUNC_SUCCESS+/* mldsa/src/fips202/native/auto.h */+#undef MLD_FIPS202_NATIVE_AUTO_H+#if defined(MLD_SYS_AARCH64)+/*+ * Undefine macros from native code (FIPS202, AArch64)+ */+/* mldsa/src/fips202/native/aarch64/auto.h */+#undef MLD_FIPS202_NATIVE_AARCH64_AUTO_H+/* mldsa/src/fips202/native/aarch64/src/fips202_native_aarch64.h */+#undef MLD_FIPS202_NATIVE_AARCH64_SRC_FIPS202_NATIVE_AARCH64_H+#undef mld_keccak_f1600_x1_scalar_aarch64_asm+#undef mld_keccak_f1600_x1_v84a_aarch64_asm+#undef mld_keccak_f1600_x2_v84a_aarch64_asm+#undef mld_keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm+#undef mld_keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm+#undef mld_keccakf1600_round_constants+/* mldsa/src/fips202/native/aarch64/x1_scalar.h */+#undef MLD_FIPS202_AARCH64_NEED_X1_SCALAR+#undef MLD_FIPS202_NATIVE_AARCH64_X1_SCALAR_H+#undef MLD_USE_NATIVE_FIPS202_X1+/* mldsa/src/fips202/native/aarch64/x1_v84a.h */+#undef MLD_FIPS202_AARCH64_NEED_X1_V84A+#undef MLD_FIPS202_NATIVE_AARCH64_X1_V84A_H+#undef MLD_USE_NATIVE_FIPS202_X1+/* mldsa/src/fips202/native/aarch64/x2_v84a.h */+#undef MLD_FIPS202_AARCH64_NEED_X2_V84A+#undef MLD_FIPS202_NATIVE_AARCH64_X2_V84A_H+#undef MLD_USE_NATIVE_FIPS202_X4+/* mldsa/src/fips202/native/aarch64/x4_v8a_scalar.h */+#undef MLD_FIPS202_AARCH64_NEED_X4_V8A_SCALAR_HYBRID+#undef MLD_FIPS202_NATIVE_AARCH64_X4_V8A_SCALAR_H+#undef MLD_USE_NATIVE_FIPS202_X4+/* mldsa/src/fips202/native/aarch64/x4_v8a_v84a_scalar.h */+#undef MLD_FIPS202_AARCH64_NEED_X4_V8A_V84A_SCALAR_HYBRID+#undef MLD_FIPS202_NATIVE_AARCH64_X4_V8A_V84A_SCALAR_H+#undef MLD_USE_NATIVE_FIPS202_X4+#endif /* MLD_SYS_AARCH64 */+#if defined(MLD_SYS_X86_64)+/*+ * Undefine macros from native code (FIPS202, x86_64)+ */+/* mldsa/src/fips202/native/x86_64/keccak_f1600_x4_avx2.h */+#undef MLD_FIPS202_NATIVE_X86_64_KECCAK_F1600_X4_AVX2_H+#undef MLD_FIPS202_X86_64_NEED_X4_AVX2+#undef MLD_USE_NATIVE_FIPS202_X4+/* mldsa/src/fips202/native/x86_64/src/fips202_native_x86_64.h */+#undef MLD_FIPS202_NATIVE_X86_64_SRC_FIPS202_NATIVE_X86_64_H+#undef mld_keccak_f1600_x4_avx2_asm+#undef mld_keccak_rho56+#undef mld_keccak_rho8+#undef mld_keccakf1600_round_constants+#endif /* MLD_SYS_X86_64 */+#if defined(MLD_SYS_ARMV81M_MVE)+/*+ * Undefine macros from native code (FIPS202, Armv8.1-M)+ */+/* mldsa/src/fips202/native/armv81m/mve.h */+#undef MLD_FIPS202_ARMV81M_NEED_X4+#undef MLD_FIPS202_NATIVE_ARMV81M+#undef MLD_FIPS202_NATIVE_ARMV81M_MVE_H+#undef MLD_USE_NATIVE_FIPS202_X4+#undef MLD_USE_NATIVE_FIPS202_X4_EXTRACT_BYTES+#undef MLD_USE_NATIVE_FIPS202_X4_XOR_BYTES+#undef mld_keccak_f1600_x4_native_impl+/* mldsa/src/fips202/native/armv81m/src/fips202_native_armv81m.h */+#undef MLD_FIPS202_NATIVE_ARMV81M_SRC_FIPS202_NATIVE_ARMV81M_H+#undef mld_keccak_f1600_x4_mve_asm+#undef mld_keccak_f1600_x4_state_extract_bytes_asm+#undef mld_keccak_f1600_x4_state_xor_bytes_asm+#undef mld_keccakf1600_round_constants+#endif /* MLD_SYS_ARMV81M_MVE */+#endif /* MLD_CONFIG_USE_NATIVE_BACKEND_FIPS202 */+#if defined(MLD_CONFIG_USE_NATIVE_BACKEND_ARITH)+/* mldsa/src/native/api.h */+#undef MLD_FQMUL_BOUND+#undef MLD_INTT_BOUND+#undef MLD_NATIVE_API_H+#undef MLD_NATIVE_FUNC_FALLBACK+#undef MLD_NATIVE_FUNC_SUCCESS+#undef MLD_NTT_BOUND+#undef MLD_REDUCE32_RANGE_MAX+/* mldsa/src/native/meta.h */+#undef MLD_NATIVE_META_H+#if defined(MLD_SYS_AARCH64)+/*+ * Undefine macros from native code (Arith, AArch64)+ */+/* mldsa/src/native/aarch64/meta.h */+#undef MLD_ARITH_BACKEND_AARCH64+#undef MLD_NATIVE_AARCH64_META_H+#undef MLD_USE_NATIVE_INTT+#undef MLD_USE_NATIVE_NTT+#undef MLD_USE_NATIVE_POINTWISE_MONTGOMERY+#undef MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L4+#undef MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L5+#undef MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L7+#undef MLD_USE_NATIVE_POLYZ_UNPACK_17+#undef MLD_USE_NATIVE_POLYZ_UNPACK_19+#undef MLD_USE_NATIVE_POLY_CADDQ+#undef MLD_USE_NATIVE_POLY_CHKNORM+#undef MLD_USE_NATIVE_POLY_DECOMPOSE_32+#undef MLD_USE_NATIVE_POLY_DECOMPOSE_88+#undef MLD_USE_NATIVE_POLY_USE_HINT_32+#undef MLD_USE_NATIVE_POLY_USE_HINT_88+#undef MLD_USE_NATIVE_REJ_UNIFORM+#undef MLD_USE_NATIVE_REJ_UNIFORM_ETA2+#undef MLD_USE_NATIVE_REJ_UNIFORM_ETA4+/* mldsa/src/native/aarch64/src/arith_native_aarch64.h */+#undef MLD_AARCH64_REJ_UNIFORM_ETA2_BUFLEN+#undef MLD_AARCH64_REJ_UNIFORM_ETA4_BUFLEN+#undef MLD_NATIVE_AARCH64_SRC_ARITH_NATIVE_AARCH64_H+#undef mld_aarch64_intt_zetas_layer123456+#undef mld_aarch64_intt_zetas_layer78+#undef mld_aarch64_ntt_zetas_layer123456+#undef mld_aarch64_ntt_zetas_layer78+#undef mld_intt_aarch64_asm+#undef mld_ntt_aarch64_asm+#undef mld_poly_caddq_aarch64_asm+#undef mld_poly_chknorm_aarch64_asm+#undef mld_poly_decompose_32_aarch64_asm+#undef mld_poly_decompose_88_aarch64_asm+#undef mld_poly_pointwise_montgomery_aarch64_asm+#undef mld_poly_use_hint_32_aarch64_asm+#undef mld_poly_use_hint_88_aarch64_asm+#undef mld_polyvecl_pointwise_acc_montgomery_l4_aarch64_asm+#undef mld_polyvecl_pointwise_acc_montgomery_l5_aarch64_asm+#undef mld_polyvecl_pointwise_acc_montgomery_l7_aarch64_asm+#undef mld_polyz_unpack_17_aarch64_asm+#undef mld_polyz_unpack_17_indices+#undef mld_polyz_unpack_19_aarch64_asm+#undef mld_polyz_unpack_19_indices+#undef mld_rej_uniform_aarch64_asm+#undef mld_rej_uniform_eta2_aarch64_asm+#undef mld_rej_uniform_eta4_aarch64_asm+#undef mld_rej_uniform_eta_table+#undef mld_rej_uniform_table+#endif /* MLD_SYS_AARCH64 */+#if defined(MLD_SYS_X86_64)+/*+ * Undefine macros from native code (Arith, X86_64)+ */+/* mldsa/src/native/x86_64/meta.h */+#undef MLD_ARITH_BACKEND_X86_64_DEFAULT+#undef MLD_NATIVE_X86_64_META_H+#undef MLD_USE_NATIVE_INTT+#undef MLD_USE_NATIVE_NTT+#undef MLD_USE_NATIVE_NTT_CUSTOM_ORDER+#undef MLD_USE_NATIVE_POINTWISE_MONTGOMERY+#undef MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L4+#undef MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L5+#undef MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L7+#undef MLD_USE_NATIVE_POLYZ_UNPACK_17+#undef MLD_USE_NATIVE_POLYZ_UNPACK_19+#undef MLD_USE_NATIVE_POLY_CADDQ+#undef MLD_USE_NATIVE_POLY_CHKNORM+#undef MLD_USE_NATIVE_POLY_DECOMPOSE_32+#undef MLD_USE_NATIVE_POLY_DECOMPOSE_88+#undef MLD_USE_NATIVE_POLY_USE_HINT_32+#undef MLD_USE_NATIVE_POLY_USE_HINT_88+#undef MLD_USE_NATIVE_REJ_UNIFORM+#undef MLD_USE_NATIVE_REJ_UNIFORM_ETA2+#undef MLD_USE_NATIVE_REJ_UNIFORM_ETA4+/* mldsa/src/native/x86_64/src/arith_native_x86_64.h */+#undef MLD_AVX2_REJ_UNIFORM_BUFLEN+#undef MLD_AVX2_REJ_UNIFORM_ETA2_BUFLEN+#undef MLD_AVX2_REJ_UNIFORM_ETA4_BUFLEN+#undef MLD_NATIVE_X86_64_SRC_ARITH_NATIVE_X86_64_H+#undef mld_invntt_avx2_asm+#undef mld_ntt_avx2_asm+#undef mld_nttunpack_avx2_asm+#undef mld_pointwise_acc_l4_avx2_asm+#undef mld_pointwise_acc_l5_avx2_asm+#undef mld_pointwise_acc_l7_avx2_asm+#undef mld_pointwise_avx2_asm+#undef mld_poly_caddq_avx2_asm+#undef mld_poly_chknorm_avx2_asm+#undef mld_poly_decompose_32_avx2_asm+#undef mld_poly_decompose_88_avx2_asm+#undef mld_poly_use_hint_32_avx2_asm+#undef mld_poly_use_hint_88_avx2_asm+#undef mld_polyz_unpack_17_avx2_asm+#undef mld_polyz_unpack_19_avx2_asm+#undef mld_rej_uniform_avx2_asm+#undef mld_rej_uniform_eta2_avx2_asm+#undef mld_rej_uniform_eta4_avx2_asm+#undef mld_rej_uniform_table+/* mldsa/src/native/x86_64/src/consts.h */+#undef MLD_AVX2_BACKEND_DATA_OFFSET_8XDIV+#undef MLD_AVX2_BACKEND_DATA_OFFSET_8XDIV_QINV+#undef MLD_AVX2_BACKEND_DATA_OFFSET_8XQ+#undef MLD_AVX2_BACKEND_DATA_OFFSET_8XQINV+#undef MLD_AVX2_BACKEND_DATA_OFFSET_ZETAS+#undef MLD_AVX2_BACKEND_DATA_OFFSET_ZETAS_QINV+#undef MLD_NATIVE_X86_64_SRC_CONSTS_H+#undef mld_qdata+#endif /* MLD_SYS_X86_64 */+#endif /* MLD_CONFIG_USE_NATIVE_BACKEND_ARITH */+#endif /* !MLD_CONFIG_MONOBUILD_KEEP_SHARED_HEADERS */
@@ -0,0 +1,855 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS140_3_IG]+ * Implementation Guidance for FIPS 140-3 and the Cryptographic Module+ * Validation Program+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/projects/cryptographic-module-validation-program/fips-140-3-ig-announcements+ *+ * - [FIPS204]+ * FIPS 204 Module-Lattice-Based Digital Signature Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/204/final+ *+ * - [FIPS204_UPDATES]+ * FIPS 204 Potential Updates (Errata)+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/files/pubs/fips/204/final/docs/fips-204-potential-updates.xlsx+ */++#ifndef MLD_CONFIG_H+#define MLD_CONFIG_H++/**+ * MLD_CONFIG_PARAMETER_SET+ *+ * Specifies the parameter set for ML-DSA+ * - MLD_CONFIG_PARAMETER_SET=44 corresponds to ML-DSA-44+ * - MLD_CONFIG_PARAMETER_SET=65 corresponds to ML-DSA-65+ * - MLD_CONFIG_PARAMETER_SET=87 corresponds to ML-DSA-87+ *+ * If you want to support multiple parameter sets, build the+ * library multiple times and set MLD_CONFIG_MULTILEVEL_BUILD.+ * See MLD_CONFIG_MULTILEVEL_BUILD for how to do this while+ * minimizing code duplication.+ *+ * This can also be set using CFLAGS.+ */+#ifndef MLD_CONFIG_PARAMETER_SET+#define MLD_CONFIG_PARAMETER_SET \+ 44 /* Change this for different security strengths */+#endif++/**+ * MLD_CONFIG_FILE+ *+ * If defined, this is a header that will be included instead+ * of the default configuration file mldsa/mldsa_native_config.h.+ *+ * When you need to build mldsa-native in multiple configurations,+ * using varying MLD_CONFIG_FILE can be more convenient+ * than configuring everything through CFLAGS.+ *+ * To use, MLD_CONFIG_FILE _must_ be defined prior+ * to the inclusion of any mldsa-native headers. For example,+ * it can be set by passing `-DMLD_CONFIG_FILE="..."`+ * on the command line.+ */+/* #define MLD_CONFIG_FILE "mldsa_native_config.h" */++/**+ * MLD_CONFIG_NAMESPACE_PREFIX+ *+ * The prefix to use to namespace global symbols from mldsa/.+ *+ * In a multi-level build, level-dependent symbols will+ * additionally be prefixed with the parameter set (44/65/87).+ *+ * This can also be set using CFLAGS.+ */+#if !defined(MLD_CONFIG_NAMESPACE_PREFIX)+#define MLD_CONFIG_NAMESPACE_PREFIX MLD_DEFAULT_NAMESPACE_PREFIX+#endif++/**+ * MLD_CONFIG_MULTILEVEL_BUILD+ *+ * Set this if the build is part of a multi-level build supporting+ * multiple parameter sets.+ *+ * If you need only a single parameter set, keep this unset.+ *+ * To build mldsa-native with support for all parameter sets,+ * build it three times -- once per parameter set -- and set the+ * option MLD_CONFIG_MULTILEVEL_WITH_SHARED for exactly one of+ * them, and MLD_CONFIG_MULTILEVEL_NO_SHARED for the others.+ * MLD_CONFIG_MULTILEVEL_BUILD should be set for all of them.+ *+ * See examples/multilevel_build for an example.+ *+ * This can also be set using CFLAGS.+ */+/* #define MLD_CONFIG_MULTILEVEL_BUILD */++/**+ * MLD_CONFIG_EXTERNAL_API_QUALIFIER+ *+ * If set, this option provides an additional function+ * qualifier to be added to declarations of mldsa-native's+ * public API.+ *+ * The primary use case for this option are single-CU builds+ * where the public API exposed by mldsa-native is wrapped by+ * another API in the consuming application. In this case,+ * even mldsa-native's public API can be marked `static`.+ */+/* #define MLD_CONFIG_EXTERNAL_API_QUALIFIER */++/**+ * MLD_CONFIG_NO_KEYPAIR_API+ *+ * By default, mldsa-native includes support for generating key+ * pairs. If you don't need this, set MLD_CONFIG_NO_KEYPAIR_API+ * to exclude keypair, keypair_internal,+ * pk_from_sk, and all internal APIs only needed by+ * those functions.+ */+/* #define MLD_CONFIG_NO_KEYPAIR_API */++/**+ * MLD_CONFIG_NO_SIGN_API+ *+ * By default, mldsa-native includes support for creating+ * signatures. If you don't need this, set MLD_CONFIG_NO_SIGN_API+ * to exclude signature,+ * signature_extmu, signature_internal,+ * signature_pre_hash_internal,+ * signature_pre_hash_shake256, and all internal APIs+ * only needed by those functions.+ */+/* #define MLD_CONFIG_NO_SIGN_API */++/**+ * MLD_CONFIG_NO_VERIFY_API+ *+ * By default, mldsa-native includes support for verifying+ * signatures. If you don't need this, set+ * MLD_CONFIG_NO_VERIFY_API to exclude verify,+ * verify_extmu, verify_internal,+ * verify_pre_hash_internal,+ * verify_pre_hash_shake256, and all internal APIs+ * only needed by those functions.+ */+/* #define MLD_CONFIG_NO_VERIFY_API */++/**+ * MLD_CONFIG_CORE_API_ONLY+ *+ * Set this to remove all public APIs except+ * keypair_internal, signature_internal,+ * and verify_internal.+ */+/* #define MLD_CONFIG_CORE_API_ONLY */++/**+ * MLD_CONFIG_NO_RANDOMIZED_API+ *+ * If this option is set, mldsa-native will be built without the+ * randomized API functions (keypair,+ * signature, and signature_extmu).+ * This allows users to build mldsa-native without providing a+ * randombytes() implementation if they only need the+ * internal deterministic API+ * (keypair_internal, signature_internal).+ *+ * @note This option is incompatible with MLD_CONFIG_KEYGEN_PCT+ * as the current PCT implementation requires+ * signature().+ */+/* #define MLD_CONFIG_NO_RANDOMIZED_API */++/**+ * MLD_CONFIG_CONSTANTS_ONLY+ *+ * If you only need the size constants (MLDSA_PUBLICKEYBYTES, etc.)+ * but no function declarations, set MLD_CONFIG_CONSTANTS_ONLY.+ *+ * This only affects the public header mldsa_native.h, not+ * the implementation.+ */+/* #define MLD_CONFIG_CONSTANTS_ONLY */+/******************************************************************************+ *+ * Build-only configuration options+ *+ * The remaining configurations are build-options only.+ * They do not affect the API described in mldsa_native.h.+ *+ *****************************************************************************/+#if defined(MLD_BUILD_INTERNAL)++/**+ * MLD_CONFIG_MULTILEVEL_WITH_SHARED+ *+ * This is for multi-level builds of mldsa-native only. If you+ * need only a single parameter set, keep this unset.+ *+ * If this is set, all MLD_CONFIG_PARAMETER_SET-independent+ * code will be included in the build, including code needed only+ * for other parameter sets.+ *+ * Example: mld_polyw1_pack_88 is only needed for+ * MLD_CONFIG_PARAMETER_SET == 44. Yet, if this option is set for a+ * build with MLD_CONFIG_PARAMETER_SET == 65/87, it would be included.+ *+ * To build mldsa-native with support for all parameter sets,+ * build it three times -- once per parameter set -- and set the+ * option MLD_CONFIG_MULTILEVEL_WITH_SHARED for exactly one of+ * them, and MLD_CONFIG_MULTILEVEL_NO_SHARED for the others.+ * MLD_CONFIG_MULTILEVEL_BUILD should be set for all of them.+ *+ * See examples/multilevel_build for an example.+ *+ * This can also be set using CFLAGS.+ */+/* #define MLD_CONFIG_MULTILEVEL_WITH_SHARED */++/**+ * MLD_CONFIG_MULTILEVEL_NO_SHARED+ *+ * This is for multi-level builds of mldsa-native only. If you+ * need only a single parameter set, keep this unset.+ *+ * If this is set, no MLD_CONFIG_PARAMETER_SET-independent code+ * will be included in the build.+ *+ * To build mldsa-native with support for all parameter sets,+ * build it three times -- once per parameter set -- and set the+ * option MLD_CONFIG_MULTILEVEL_WITH_SHARED for exactly one of+ * them, and MLD_CONFIG_MULTILEVEL_NO_SHARED for the others.+ * MLD_CONFIG_MULTILEVEL_BUILD should be set for all of them.+ *+ * See examples/multilevel_build for an example.+ *+ * This can also be set using CFLAGS.+ */+/* #define MLD_CONFIG_MULTILEVEL_NO_SHARED */++/**+ * MLD_CONFIG_MONOBUILD_KEEP_SHARED_HEADERS+ *+ * This is only relevant for single compilation unit (SCU)+ * builds of mldsa-native. In this case, it determines whether+ * directives defined in parameter-set-independent headers should+ * be #undef'ined or not at the end of the SCU file. This is+ * needed in multilevel builds.+ *+ * See examples/multilevel_build_native for an example.+ *+ * This can also be set using CFLAGS.+ */+/* #define MLD_CONFIG_MONOBUILD_KEEP_SHARED_HEADERS */++/**+ * MLD_CONFIG_USE_NATIVE_BACKEND_ARITH+ *+ * Determines whether a native arithmetic backend should be used.+ *+ * The arithmetic backend covers performance-critical functions+ * such as the number-theoretic transform (NTT).+ *+ * If this option is unset, the C backend will be used.+ *+ * If this option is set, the arithmetic backend to be used is+ * determined by MLD_CONFIG_ARITH_BACKEND_FILE: If the latter is+ * unset, the default backend for your target architecture+ * will be used. If set, it must be the name of a backend metadata+ * file.+ *+ * This can also be set using CFLAGS.+ */+#if !defined(MLD_CONFIG_USE_NATIVE_BACKEND_ARITH)+/* #define MLD_CONFIG_USE_NATIVE_BACKEND_ARITH */+#endif++/**+ * MLD_CONFIG_ARITH_BACKEND_FILE+ *+ * The arithmetic backend to use.+ *+ * If MLD_CONFIG_USE_NATIVE_BACKEND_ARITH is unset, this option+ * is ignored.+ *+ * If MLD_CONFIG_USE_NATIVE_BACKEND_ARITH is set, this option must+ * either be undefined or the filename of an arithmetic backend.+ * If unset, the default backend will be used.+ *+ * This can be set using CFLAGS.+ */+#if defined(MLD_CONFIG_USE_NATIVE_BACKEND_ARITH) && \+ !defined(MLD_CONFIG_ARITH_BACKEND_FILE)+#define MLD_CONFIG_ARITH_BACKEND_FILE "native/meta.h"+#endif++/**+ * MLD_CONFIG_USE_NATIVE_BACKEND_FIPS202+ *+ * Determines whether a native FIPS202 backend should be used.+ *+ * The FIPS202 backend covers 1x/2x/4x-fold Keccak-f1600, which is+ * the performance bottleneck of SHA3 and SHAKE.+ *+ * If this option is unset, the C backend will be used.+ *+ * If this option is set, the FIPS202 backend to be used is+ * determined by MLD_CONFIG_FIPS202_BACKEND_FILE: If the latter is+ * unset, the default backend for your target architecture+ * will be used. If set, it must be the name of a backend metadata+ * file.+ *+ * This can also be set using CFLAGS.+ */+#if !defined(MLD_CONFIG_USE_NATIVE_BACKEND_FIPS202)+/* #define MLD_CONFIG_USE_NATIVE_BACKEND_FIPS202 */+#endif++/**+ * MLD_CONFIG_FIPS202_BACKEND_FILE+ *+ * The FIPS-202 backend to use.+ *+ * If MLD_CONFIG_USE_NATIVE_BACKEND_FIPS202 is set, this option+ * must either be undefined or the filename of a FIPS202 backend.+ * If unset, the default backend will be used.+ *+ * This can be set using CFLAGS.+ */+#if defined(MLD_CONFIG_USE_NATIVE_BACKEND_FIPS202) && \+ !defined(MLD_CONFIG_FIPS202_BACKEND_FILE)+#define MLD_CONFIG_FIPS202_BACKEND_FILE "fips202/native/auto.h"+#endif++/**+ * MLD_CONFIG_FIPS202_CUSTOM_HEADER+ *+ * Custom header to use for FIPS-202+ *+ * This should only be set if you intend to use a custom+ * FIPS-202 implementation, different from the one shipped+ * with mldsa-native.+ *+ * If set, it must be the name of a file serving as the+ * replacement for mldsa/src/fips202/fips202.h, and exposing+ * the same API (see FIPS202.md).+ */+/* #define MLD_CONFIG_FIPS202_CUSTOM_HEADER "SOME_FILE.h" */++/**+ * MLD_CONFIG_FIPS202X4_CUSTOM_HEADER+ *+ * Custom header to use for FIPS-202-X4+ *+ * This should only be set if you intend to use a custom+ * FIPS-202 implementation, different from the one shipped+ * with mldsa-native.+ *+ * If set, it must be the name of a file serving as the+ * replacement for mldsa/src/fips202/fips202x4.h, and exposing+ * the same API (see FIPS202.md).+ */+/* #define MLD_CONFIG_FIPS202X4_CUSTOM_HEADER "SOME_FILE.h" */++/**+ * MLD_CONFIG_CUSTOM_ZEROIZE+ *+ * In compliance with @[FIPS204, Section 3.6.3], mldsa-native zeroizes+ * intermediate buffers before returning from function calls. By default,+ * those buffers are allocated from the stack; if MLD_CONFIG_CUSTOM_ALLOC_FREE+ * is set, they are (mostly -- few exceptions remain at present) allocated from+ * the configured custom allocator.+ *+ * mldsa-native also zeroizes caller-owned output buffers as needed to uphold+ * the API convention that outputs be either unmodified or zeroized upon+ * failure.+ *+ * Set this option and define `mld_zeroize` if you want to use a custom+ * method to zeroize intermediate and output buffers.+ *+ * The default implementation uses SecureZeroMemory on Windows and a+ * memset + compiler barrier otherwise. If neither of those is available on+ * the target platform, compilation will fail, and you will need to use+ * MLD_CONFIG_CUSTOM_ZEROIZE to provide a custom implementation of+ * `mld_zeroize()`.+ *+ * @warning+ * The zeroization conducted by mldsa-native reduces the likelihood of data+ * leaking on the stack or custom allocators, but it does not eliminate it.+ * For example, the C standard makes no guarantee about where a compiler+ * allocates local structures and whether/where it makes copies of them.+ * Also, in addition to entire structures, there may also be potentially+ * exploitable leakage of individual values on the stack. If you need+ * bullet-proof zeroization of the stack, you need to consider additional+ * measures instead of what this feature provides. In this case, you can+ * set mld_zeroize to a no-op. Note that in this case you are also responsible+ * for zeroizing output buffers upon failure.+ */+/* #define MLD_CONFIG_CUSTOM_ZEROIZE+ #if !defined(__ASSEMBLER__)+ #include <stdint.h>+ #include "src/src.h"+ static MLD_INLINE void mld_zeroize(void *ptr, size_t len)+ {+ ... your implementation ...+ }+ #endif+*/++/**+ * MLD_CONFIG_CUSTOM_RANDOMBYTES+ *+ * mldsa-native does not provide a secure randombytes+ * implementation. Such an implementation has to be provided by+ * the consumer.+ *+ * If this option is not set, mldsa-native expects a function+ * int randombytes(uint8_t *out, size_t outlen).+ *+ * Set this option and define `mld_randombytes` if you want to+ * use a custom method to sample randombytes with a different name+ * or signature.+ */+/* #define MLD_CONFIG_CUSTOM_RANDOMBYTES+ #if !defined(__ASSEMBLER__)+ #include <stdint.h>+ #include "src/src.h"+ static MLD_INLINE int mld_randombytes(uint8_t *ptr, size_t len)+ {+ ... your implementation ...+ return 0;+ }+ #endif+*/++/**+ * MLD_CONFIG_CUSTOM_CAPABILITY_FUNC+ *+ * mldsa-native backends may rely on specific hardware features.+ * Those backends will only be included in an mldsa-native build+ * if support for the respective features is enabled at+ * compile-time. However, when building for a heterogeneous set+ * of CPUs to run the resulting binary/library on, feature+ * detection at _runtime_ is needed to decide whether a backend+ * can be used or not.+ *+ * Set this option and define `mld_sys_check_capability` if you+ * want to use a custom method to dispatch between implementations.+ *+ * Return value 1 indicates that a capability is supported.+ * Return value 0 indicates that a capability is not supported.+ *+ * If this option is not set, mldsa-native uses compile-time+ * feature detection only to decide which backend to use.+ *+ * If you compile mldsa-native on a system with different+ * capabilities than the system that the resulting binary/library+ * will be run on, you must use this option.+ */+/* #define MLD_CONFIG_CUSTOM_CAPABILITY_FUNC+ static MLD_INLINE int mld_sys_check_capability(mld_sys_cap cap)+ {+ ... your implementation ...+ }+*/++/**+ * MLD_CONFIG_CUSTOM_ALLOC_FREE+ *+ * Set this option and define `MLD_CUSTOM_ALLOC` and+ * `MLD_CUSTOM_FREE` if you want to use custom allocation for+ * large local structures or buffers.+ *+ * By default, all buffers/structures are allocated on the stack.+ * If this option is set, most of them will be allocated via+ * MLD_CUSTOM_ALLOC.+ *+ * Parameters to MLD_CUSTOM_ALLOC:+ * - T* v: Target pointer to declare.+ * - T: Type of structure to be allocated+ * - N: Number of elements to be allocated.+ *+ * Parameters to MLD_CUSTOM_FREE:+ * - T* v: Target pointer to free. May be NULL.+ * - T: Type of structure to be freed.+ * - N: Number of elements to be freed.+ *+ * @warning This option is experimental. Its scope, configuration and+ * function/macro signatures may change at any time. We expect a+ * stable API in a future version.+ *+ * @note Even if this option is set, some allocations further down+ * the call stack will still be made from the stack. Those will+ * likely be added to the scope of this option in the future.+ *+ * @note MLD_CUSTOM_ALLOC need not guarantee a successful+ * allocation nor include error handling. Upon failure, the+ * target pointer should simply be set to NULL. The calling+ * code will handle this case and invoke MLD_CUSTOM_FREE.+ */+/* #define MLD_CONFIG_CUSTOM_ALLOC_FREE+ #if !defined(__ASSEMBLER__)+ #include <stdlib.h>+ #define MLD_CUSTOM_ALLOC(v, T, N) \+ T* (v) = (T *)aligned_alloc(MLD_DEFAULT_ALIGN, \+ MLD_ALIGN_UP(sizeof(T) * (N)))+ #define MLD_CUSTOM_FREE(v, T, N) free(v)+ #endif+*/++/**+ * MLD_CONFIG_CUSTOM_MEMCPY+ *+ * Set this option and define `mld_memcpy` if you want to+ * use a custom method to copy memory instead of the standard+ * library memcpy function.+ *+ * The custom implementation must have the same signature and+ * behavior as the standard memcpy function:+ * void *mld_memcpy(void *dest, const void *src, size_t n)+ */+/* #define MLD_CONFIG_CUSTOM_MEMCPY+ #if !defined(__ASSEMBLER__)+ #include <stdint.h>+ #include "src/src.h"+ static MLD_INLINE void *mld_memcpy(void *dest, const void *src, size_t n)+ {+ ... your implementation ...+ }+ #endif+*/++/**+ * MLD_CONFIG_CUSTOM_MEMSET+ *+ * Set this option and define `mld_memset` if you want to+ * use a custom method to set memory instead of the standard+ * library memset function.+ *+ * The custom implementation must have the same signature and+ * behavior as the standard memset function:+ * void *mld_memset(void *s, int c, size_t n)+ */+/* #define MLD_CONFIG_CUSTOM_MEMSET+ #if !defined(__ASSEMBLER__)+ #include <stdint.h>+ #include "src/src.h"+ static MLD_INLINE void *mld_memset(void *s, int c, size_t n)+ {+ ... your implementation ...+ }+ #endif+*/++/**+ * MLD_CONFIG_INTERNAL_API_QUALIFIER+ *+ * If set, this option provides an additional qualifier+ * to be added to declarations of internal API functions and data.+ *+ * The primary use case for this option are single-CU builds,+ * in which case this option can be set to `static`.+ */+/* #define MLD_CONFIG_INTERNAL_API_QUALIFIER */++/**+ * MLD_CONFIG_CT_TESTING_ENABLED+ *+ * If set, mldsa-native annotates data as secret / public using+ * valgrind's annotations VALGRIND_MAKE_MEM_UNDEFINED and+ * VALGRIND_MAKE_MEM_DEFINED, enabling various checks for secret-+ * dependent control flow of variable time execution (depending+ * on the exact version of valgrind installed).+ */+/* #define MLD_CONFIG_CT_TESTING_ENABLED */++/**+ * MLD_CONFIG_NO_ASM+ *+ * If this option is set, mldsa-native will be built without+ * use of native code or inline assembly.+ *+ * By default, inline assembly is used to implement value barriers.+ * Without inline assembly, mldsa-native will use a global volatile+ * 'opt blocker' instead; see ct.h.+ *+ * Inline assembly is also used to implement a secure zeroization+ * function on non-Windows platforms. If this option is set and+ * the target platform is not Windows, you MUST set+ * MLD_CONFIG_CUSTOM_ZEROIZE and provide a custom zeroization+ * function.+ *+ * If this option is set, MLD_CONFIG_USE_NATIVE_BACKEND_FIPS202 and+ * MLD_CONFIG_USE_NATIVE_BACKEND_ARITH will be ignored, and no+ * native backends will be used.+ */+/* #define MLD_CONFIG_NO_ASM */++/**+ * MLD_CONFIG_NO_ASM_VALUE_BARRIER+ *+ * If this option is set, mldsa-native will be built without+ * use of native code or inline assembly for value barriers.+ *+ * By default, inline assembly (if available) is used to implement+ * value barriers.+ * Without inline assembly, mldsa-native will use a global volatile+ * 'opt blocker' instead; see ct.h.+ */+/* #define MLD_CONFIG_NO_ASM_VALUE_BARRIER */++/**+ * MLD_CONFIG_KEYGEN_PCT+ *+ * Compliance with @[FIPS140_3_IG, p.87] requires a+ * Pairwise Consistency Test (PCT) to be carried out on a freshly+ * generated keypair before it can be exported.+ *+ * Set this option if such a check should be implemented.+ * In this case, keypair_internal and+ * keypair will return MLD_ERR_PCT_FAIL if the+ * PCT failed.+ *+ * @note This feature will drastically lower the performance of+ * key generation.+ *+ * @note This option is incompatible with MLD_CONFIG_NO_SIGN_API+ * and MLD_CONFIG_NO_VERIFY_API as the current PCT implementation+ * requires signature() and verify().+ */+/* #define MLD_CONFIG_KEYGEN_PCT */++/**+ * MLD_CONFIG_KEYGEN_PCT_BREAKAGE_TEST+ *+ * If this option is set, the user must provide a runtime+ * function `static inline int mld_break_pct() { ... }` to+ * indicate whether the PCT should be made fail.+ *+ * This option only has an effect if MLD_CONFIG_KEYGEN_PCT is set.+ */+/* #define MLD_CONFIG_KEYGEN_PCT_BREAKAGE_TEST+ #if !defined(__ASSEMBLER__)+ #include "src/src.h"+ static MLD_INLINE int mld_break_pct(void)+ {+ ... return 0/1 depending on whether PCT should be broken ...+ }+ #endif+*/++/**+ * MLD_CONFIG_MAX_SIGNING_ATTEMPTS+ *+ * Upper bound on the number of rejection-sampling iterations+ * performed by ML-DSA signing (@[FIPS204, Algorithm 7]).+ *+ * If a valid signature is not produced within this many+ * attempts, signing returns MLD_ERR_SIGN_ATTEMPTS_EXHAUSTED.+ * This is useful in timing-sensitive environments that+ * require a deterministic worst-case bound on signing time.+ *+ * For FIPS 204 compliance, this value MUST be at least 821,+ * cf. @[FIPS204, Appendix C] and @[FIPS204_UPDATES], which is+ * chosen so that the signing failure rate is < 2^{-256}.+ *+ * Default: Largest possible value before internal counters+ * would overflow. This is larger than the FIPS204 bound.+ *+ * In particular, in the default configuration, the signing+ * failure rate is < 2^{-256}.+ */+/* #define MLD_CONFIG_MAX_SIGNING_ATTEMPTS 821 */++/**+ * MLD_CONFIG_SERIAL_FIPS202_ONLY+ *+ * Set this to use a FIPS202 implementation with global state+ * that supports only one active Keccak computation at a time+ * (e.g. some hardware accelerators).+ *+ * If this option is set, ML-DSA will use FIPS202 operations+ * serially, ensuring that only one SHAKE context is active+ * at any given time.+ *+ * This allows offloading Keccak computations to a hardware+ * accelerator that holds only a single Keccak state locally,+ * rather than requiring support for multiple concurrent+ * Keccak states.+ *+ * @note Depending on the target CPU, this may reduce+ * performance when using software FIPS202 implementations.+ * Only enable this when you have to.+ */+/* #define MLD_CONFIG_SERIAL_FIPS202_ONLY */++/**+ * MLD_CONFIG_CONTEXT_PARAMETER+ *+ * Set this to add a caller-supplied context parameter to the public API+ * functions, which is then forwarded unchanged to the custom callbacks+ * (allocation, and signing hooks below).+ *+ * When this option is set, every public API function gains a trailing+ * parameter+ *+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE context+ *+ * as its last argument; its type is configured via+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE (see below). mldsa-native treats this+ * value as opaque: it never dereferences it and only passes it on to the+ * configurable hook macros. It is meant to carry per-caller state -- e.g. a+ * pointer to a memory pool for the allocation hooks, or the resume state for+ * the signing hooks -- into those hooks.+ *+ * When this option is unset (the default), no extra parameter is added and+ * the hook macros never receive a context argument.+ *+ * The hooks that receive the context are the allocation hooks (see+ * MLD_CONFIG_CUSTOM_ALLOC_FREE) and the signing hooks (see+ * MLD_CONFIG_SIGN_HOOK_RESUME / _ATTEMPT / _FINISH); each is documented with+ * its own option below.+ */+/* #define MLD_CONFIG_CONTEXT_PARAMETER */++/**+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE+ *+ * Set this to define the type of the context parameter added by+ * MLD_CONFIG_CONTEXT_PARAMETER. It can be any C type usable as a function+ * parameter, e.g. `void *` or a pointer to a caller-defined struct such as+ * `struct my_ctx *`.+ *+ * This option must be defined if and only if MLD_CONFIG_CONTEXT_PARAMETER is+ * defined; defining one without the other is a compile-time error.+ */+/* #define MLD_CONFIG_CONTEXT_PARAMETER_TYPE void* */++/**+ * Signing hooks: MLD_CONFIG_SIGN_HOOK_RESUME / _ATTEMPT / _FINISH+ *+ * Three optional, independent hooks into the ML-DSA signing rejection-sampling+ * loop. Each is enabled by defining the matching option, in which case the+ * integration must provide the corresponding function. If a hook needs+ * per-operation state, enable MLD_CONFIG_CONTEXT_PARAMETER; the context is then+ * appended as the last argument.+ *+ * @warning This feature is experimental. Its scope, configuration and+ * function signatures may change at any time, including after v2.+ *+ * Enabling any of the hooks requires MLD_CONFIG_NO_RANDOMIZED_API (restricting+ * the public API to deterministic operations). This is because the restartable+ * signing as enabled by the signing hooks only produces the uninterrupted+ * signature when the randomness is fixed across calls. A logging-only use+ * (attempt always returns 0; resume/finish merely observe) would be safe with+ * the randomized API too, but for now the requirement is imposed uniformly on+ * all three hooks.+ *+ * Note: Randomized signing is a shim wrapper around deterministic signing, and+ * all helper functions you need to build it are exposed publicly. Thus, if you+ * need a restartable, randomized signing operation, you can build your own by+ * replicating the logic and adding the RNG seed to the restart context. In this+ * case, please also consider letting the mldsa-native maintainers know of your+ * need for randomized, restartable signing, so the feature can be appropriately+ * prioritized.+ *+ * - MLD_CONFIG_SIGN_HOOK_ATTEMPT: int mld_sign_hook_attempt(attempt[, ctxt])+ * Called before each attempt. Returns 0 to proceed, or non-zero to pause:+ * signing then returns MLD_ERR_SIGNING_PAUSED with `attempt` as the resume+ * point (needs MLD_CONFIG_SIGN_HOOK_RESUME to resume; otherwise just aborts).+ * Always returning 0 makes it a logging/benchmarking hook.+ *+ * - MLD_CONFIG_SIGN_HOOK_RESUME: uint16_t mld_sign_hook_resume([ctxt])+ * Returns the attempt to resume from (0 for a fresh operation), i.e. the one+ * recorded when a previous call paused.+ *+ * - MLD_CONFIG_SIGN_HOOK_FINISH: void mld_sign_hook_finish(attempt[, ctxt])+ * Called on success with the succeeding attempt. Observe-only.+ *+ * When an option is unset, the hook is a no-op (resume to 0, attempt proceeds),+ * i.e. ordinary one-shot signing.+ *+ * Independent of MLD_CONFIG_MAX_SIGNING_ATTEMPTS, which is a static upper bound+ * on the number of signing attempts.+ *+ * See test/src/test_sign_hook.c for a worked example using all three.+ */+/* #define MLD_CONFIG_SIGN_HOOK_RESUME+ #define MLD_CONFIG_SIGN_HOOK_ATTEMPT+ #define MLD_CONFIG_SIGN_HOOK_FINISH+ #if !defined(__ASSEMBLER__)+ #include <stdint.h>+ #include "src/sys.h"+ static MLD_INLINE uint16_t mld_sign_hook_resume(void)+ {+ ... return the attempt to resume from ...+ }+ static MLD_INLINE int mld_sign_hook_attempt(uint16_t attempt)+ {+ ... return non-zero to pause here; for resume, store attempt ...+ return 0;+ }+ static MLD_INLINE void mld_sign_hook_finish(uint16_t attempt)+ {+ ... mark the operation complete (attempt = successful attempt) ...+ }+ #endif+*/++/**+ * MLD_CONFIG_REDUCE_RAM+ *+ * Set this to reduce RAM usage. This trades memory for performance.+ *+ * For expected memory usage, see the MLD_TOTAL_ALLOC_* constants defined in+ * mldsa_native.h.+ *+ * This option is useful for embedded systems with tight RAM constraints but+ * relaxed performance requirements.+ *+ */+/* #define MLD_CONFIG_REDUCE_RAM */++/************************* Config internals ********************************/++#endif /* MLD_BUILD_INTERNAL */++/* Default namespace+ *+ * Don't change this. If you need a different namespace, re-define+ * MLD_CONFIG_NAMESPACE_PREFIX above instead, and remove the following.+ *+ * The default MLDSA namespace is+ *+ * PQCP_MLDSA_NATIVE_MLDSA<LEVEL>_+ *+ * e.g., PQCP_MLDSA_NATIVE_MLDSA44_+ */++#if defined(MLD_CONFIG_MULTILEVEL_BUILD)+/* In a multi-level build the parameter set is appended by the namespacing+ * machinery, so the default prefix must not embed it. */+#define MLD_DEFAULT_NAMESPACE_PREFIX PQCP_MLDSA_NATIVE_MLDSA+#elif MLD_CONFIG_PARAMETER_SET == 44+#define MLD_DEFAULT_NAMESPACE_PREFIX PQCP_MLDSA_NATIVE_MLDSA44+#elif MLD_CONFIG_PARAMETER_SET == 65+#define MLD_DEFAULT_NAMESPACE_PREFIX PQCP_MLDSA_NATIVE_MLDSA65+#elif MLD_CONFIG_PARAMETER_SET == 87+#define MLD_DEFAULT_NAMESPACE_PREFIX PQCP_MLDSA_NATIVE_MLDSA87+#endif++#endif /* !MLD_CONFIG_H */
@@ -0,0 +1,233 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#ifndef MLD_CBMC_H+#define MLD_CBMC_H++/***************************************************+ * Basic replacements for __CPROVER_XXX contracts+ ***************************************************/+/*+ * The `__contract__` / `__loop__` annotation macros use a+ * leading-double-underscore spelling in line with other CBMC macros.+ * clang-tidy flags these as reserved identifiers; we suppress the diagnostic+ * at each definition site (NOLINT) rather than disabling the check globally,+ * so it stays active for the rest of the tree.+ */+#ifndef CBMC++/* clang-format off */+#define __contract__(x) /* NOLINT(bugprone-reserved-identifier,cert-dcl37-c,cert-dcl51-cpp) */+#define __loop__(x) /* NOLINT(bugprone-reserved-identifier,cert-dcl37-c,cert-dcl51-cpp) */+/* clang-format on */+#define cassert(x)++#else /* !CBMC */+++/* clang-format off */+#define __contract__(x) x /* NOLINT(bugprone-reserved-identifier,cert-dcl37-c,cert-dcl51-cpp) */+#define __loop__(x) x /* NOLINT(bugprone-reserved-identifier,cert-dcl37-c,cert-dcl51-cpp) */+/* clang-format on */++/* Conditionally expand to __VA_ARGS__ depending on MLD_CONFIG_REDUCE_RAM. */+#if defined(MLD_CONFIG_REDUCE_RAM)+#define MLD_IF_REDUCE_RAM(...) __VA_ARGS__+#define MLD_IF_NOT_REDUCE_RAM(...)+#else+#define MLD_IF_REDUCE_RAM(...)+#define MLD_IF_NOT_REDUCE_RAM(...) __VA_ARGS__+#endif++/* https://diffblue.github.io/cbmc/contracts-assigns.html */+#define assigns(...) __CPROVER_assigns(__VA_ARGS__)++/* https://diffblue.github.io/cbmc/contracts-requires-ensures.html */+#define requires(...) __CPROVER_requires(__VA_ARGS__)+#define ensures(...) __CPROVER_ensures(__VA_ARGS__)+/* https://diffblue.github.io/cbmc/contracts-loops.html */+#define invariant(...) __CPROVER_loop_invariant(__VA_ARGS__)+#define decreases(...) __CPROVER_decreases(__VA_ARGS__)+/* cassert to avoid confusion with in-built assert */+#define cassert(x) __CPROVER_assert(x, "cbmc assertion failed")+#define assume(...) __CPROVER_assume(__VA_ARGS__)++/***************************************************+ * Macros for "expression" forms that may appear+ * _inside_ top-level contracts.+ ***************************************************/++/*+ * function return value - useful inside ensures+ * https://diffblue.github.io/cbmc/contracts-functions.html+ */+#define return_value (__CPROVER_return_value)++/*+ * assigns l-value targets+ * https://diffblue.github.io/cbmc/contracts-assigns.html+ */+#define object_whole(...) __CPROVER_object_whole(__VA_ARGS__)+#define memory_slice(...) __CPROVER_object_upto(__VA_ARGS__)+#define same_object(...) __CPROVER_same_object(__VA_ARGS__)++/*+ * Pointer-related predicates+ * https://diffblue.github.io/cbmc/contracts-memory-predicates.html+ */+#define memory_no_alias(...) __CPROVER_is_fresh(__VA_ARGS__)+#define readable(...) __CPROVER_r_ok(__VA_ARGS__)+#define writeable(...) __CPROVER_w_ok(__VA_ARGS__)++/* Maximum supported buffer size+ *+ * Larger buffers may be supported, but due to internal modeling constraints+ * in CBMC, the proofs of memory- and type-safety won't be able to run.+ *+ * If you find yourself in need for a buffer size larger than this,+ * please contact the maintainers, so we can prioritize work to relax+ * this somewhat artificial bound.+ */+#define MLD_MAX_BUFFER_SIZE (SIZE_MAX >> 12)+++/*+ * History variables+ * https://diffblue.github.io/cbmc/contracts-history-variables.html+ */+#define old(...) __CPROVER_old(__VA_ARGS__)+#define loop_entry(...) __CPROVER_loop_entry(__VA_ARGS__)++/*+ * Quantifiers+ * Note that the range on qvar is _exclusive_ between qvar_lb .. qvar_ub+ * https://diffblue.github.io/cbmc/contracts-quantifiers.html+ *+ * The quantified variable is declared as uint32_t, so these macros+ * quantify only over indices in [0, UINT32_MAX). Bounds larger than+ * UINT32_MAX (4 GiB) are NOT supported: the explicit (uint32_t) casts+ * on the bounds will trigger CBMC's conversion check if a wider bound+ * (e.g. a size_t > UINT32_MAX) is passed.+ *+ * Quantifying over size_t (64-bit) was found to blow up SMT proof+ * times, so we deliberately keep the index width at 32 bits. Callers+ * dealing with size_t-typed buffers must add an explicit+ * requires(len <= UINT32_MAX)+ * precondition.+ */++/*+ * Prevent clang-format from corrupting CBMC's special ==> operator+ */+/* clang-format off */+#define forall(qvar, qvar_lb, qvar_ub, predicate) \+ __CPROVER_forall \+ { \+ uint32_t qvar; \+ ((uint32_t) (qvar_lb) <= (qvar) && (qvar) < (uint32_t) (qvar_ub)) ==> (predicate) \+ }++#define exists(qvar, qvar_lb, qvar_ub, predicate) \+ __CPROVER_exists \+ { \+ uint32_t qvar; \+ ((uint32_t) (qvar_lb) <= (qvar) && (qvar) < (uint32_t) (qvar_ub)) && (predicate) \+ }+/* clang-format on */++/***************************************************+ * Convenience macros for common contract patterns+ ***************************************************/+/*+ * Prevent clang-format from corrupting CBMC's special ==> operator+ */+/* clang-format off */+#define CBMC_CONCAT_(left, right) left##right+#define CBMC_CONCAT(left, right) CBMC_CONCAT_(left, right)++#define array_bound_core(qvar, qvar_lb, qvar_ub, array_var, \+ value_lb, value_ub) \+ __CPROVER_forall \+ { \+ uint32_t qvar; \+ ((uint32_t) (qvar_lb) <= (qvar) && (qvar) < (uint32_t) (qvar_ub)) ==> \+ (((int)(value_lb) <= ((array_var)[(qvar)])) && \+ (((array_var)[(qvar)]) < (int)(value_ub))) \+ }++#define array_bound(array_var, qvar_lb, qvar_ub, value_lb, value_ub) \+ array_bound_core(CBMC_CONCAT(_cbmc_idx, __COUNTER__), (qvar_lb), \+ (qvar_ub), (array_var), (value_lb), (value_ub))++#define array_unchanged_core(qvar, qvar_lb, qvar_ub, array_var) \+ __CPROVER_forall \+ { \+ uint32_t qvar; \+ ((uint32_t) (qvar_lb) <= (qvar) && (qvar) < (uint32_t) (qvar_ub)) ==> \+ ((array_var)[(qvar)]) == (old(* (int32_t (*)[(qvar_ub)])(array_var)))[(qvar)] \+ }++#define array_unchanged(array_var, N) \+ array_unchanged_core(CBMC_CONCAT(_cbmc_idx, __COUNTER__), 0, (N), (array_var))++#define array_unchanged_u64_core(qvar, qvar_lb, qvar_ub, array_var) \+ __CPROVER_forall \+ { \+ uint32_t qvar; \+ ((uint32_t) (qvar_lb) <= (qvar) && (qvar) < (uint32_t) (qvar_ub)) ==> \+ ((array_var)[(qvar)]) == (old(* (uint64_t (*)[(qvar_ub)])(array_var)))[(qvar)] \+ }++#define array_unchanged_u64(array_var, N) \+ array_unchanged_u64_core(CBMC_CONCAT(_cbmc_idx, __COUNTER__), 0, (N), (array_var))++#define array_unchanged_u8_core(qvar, qvar_lb, qvar_ub, array_var) \+ __CPROVER_forall \+ { \+ uint32_t qvar; \+ ((uint32_t) (qvar_lb) <= (qvar) && (qvar) < (uint32_t) (qvar_ub)) ==> \+ ((array_var)[(qvar)]) == (old(* (uint8_t (*)[(qvar_ub)])(array_var)))[(qvar)] \+ }++#define array_unchanged_u8(array_var, N) \+ array_unchanged_u8_core(CBMC_CONCAT(_cbmc_idx, __COUNTER__), 0, (N), (array_var))++#define array_zeroized_u8_core(qvar, qvar_lb, qvar_ub, array_var) \+ __CPROVER_forall \+ { \+ uint32_t qvar; \+ ((uint32_t) (qvar_lb) <= (qvar) && (qvar) < (uint32_t) (qvar_ub)) ==> \+ ((array_var)[(qvar)]) == 0 \+ }++#define array_zeroized_u8(array_var, N) \+ array_zeroized_u8_core(CBMC_CONCAT(_cbmc_idx, __COUNTER__), 0, (N), (array_var))+/* clang-format on */++/*+ * Output-buffer discipline on failure, as documented in API-CONVENTIONS.md:+ * when a function fails, each caller-owned output buffer is left either+ * fully unchanged or fully zeroized -- never holding partially computed or+ * stale data that could be mistaken for a valid result.+ *+ * Note the disjunction is over the buffer as a whole: it is not enough for+ * each byte to be individually either unchanged or zero.+ */+#define array_unchanged_or_zeroized_u8(array_var, N) \+ (array_unchanged_u8((array_var), (N)) || array_zeroized_u8((array_var), (N)))++/* Wrapper around array_bound operating on absolute values.+ *+ * The absolute value bound `k` is exclusive.+ *+ * Note that since the lower bound in array_bound is inclusive, we have to+ * raise it by 1 here.+ */+#define array_abs_bound(arr, lb, ub, k) \+ array_bound((arr), (lb), (ub), -((int)(k)) + 1, (k))++#endif /* CBMC */++#endif /* !MLD_CBMC_H */
@@ -0,0 +1,301 @@+/*+ * Copyright (c) The mldsa-native project authors+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+#ifndef MLD_COMMON_H+#define MLD_COMMON_H++#ifndef __ASSEMBLER__+#include <stdint.h>+#endif+++#define MLD_BUILD_INTERNAL++#if defined(MLD_CONFIG_FILE)+#include MLD_CONFIG_FILE+#else+#include "mldsa_native_config.h"+#endif++#include "params.h"+#include "sys.h"++/* Internal and public API have external linkage by default, but+ * this can be overwritten by the user, e.g. for single-CU builds. */+#if !defined(MLD_CONFIG_INTERNAL_API_QUALIFIER)+#define MLD_INTERNAL_API+#define MLD_INTERNAL_DATA_DECLARATION extern+#define MLD_INTERNAL_DATA_DEFINITION+#else+#define MLD_INTERNAL_API MLD_CONFIG_INTERNAL_API_QUALIFIER+#define MLD_INTERNAL_DATA_DECLARATION MLD_CONFIG_INTERNAL_API_QUALIFIER+#define MLD_INTERNAL_DATA_DEFINITION MLD_CONFIG_INTERNAL_API_QUALIFIER+#endif++#if !defined(MLD_CONFIG_EXTERNAL_API_QUALIFIER)+#define MLD_EXTERNAL_API+#else+#define MLD_EXTERNAL_API MLD_CONFIG_EXTERNAL_API_QUALIFIER+#endif++#if defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) || \+ defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED)+#define MLD_MULTILEVEL_BUILD+#endif++#define MLD_CONCAT_(x1, x2) x1##x2+#define MLD_CONCAT(x1, x2) MLD_CONCAT_(x1, x2)++#if defined(MLD_MULTILEVEL_BUILD)+#define MLD_ADD_PARAM_SET(s) MLD_CONCAT(s, MLD_CONFIG_PARAMETER_SET)+#else+#define MLD_ADD_PARAM_SET(s) s+#endif++#define MLD_NAMESPACE_PREFIX MLD_CONCAT(MLD_CONFIG_NAMESPACE_PREFIX, _)+#define MLD_NAMESPACE_PREFIX_KL \+ MLD_CONCAT(MLD_ADD_PARAM_SET(MLD_CONFIG_NAMESPACE_PREFIX), _)++/* Functions are prefixed by MLD_CONFIG_NAMESPACE_PREFIX.+ *+ * If multiple parameter sets are used, functions depending on the parameter+ * set are additionally prefixed with 44/65/87. See mldsa_native_config.h.+ *+ * Example: If MLD_CONFIG_NAMESPACE_PREFIX is PQCP_MLDSA_NATIVE, then+ * MLD_NAMESPACE_KL(keypair) becomes PQCP_MLDSA_NATIVE44_keypair/+ * PQCP_MLDSA_NATIVE65_keypair/PQCP_MLDSA_NATIVE87_keypair.+ */+#define MLD_NAMESPACE(s) MLD_CONCAT(MLD_NAMESPACE_PREFIX, s)+#define MLD_NAMESPACE_KL(s) MLD_CONCAT(MLD_NAMESPACE_PREFIX_KL, s)++/* On Apple platforms, we need to emit leading underscore+ * in front of assembly symbols. We thus introduce a separate+ * namespace wrapper for ASM symbols. */+#if !defined(__APPLE__)+#define MLD_ASM_NAMESPACE(sym) MLD_NAMESPACE(sym)+#else+#define MLD_ASM_NAMESPACE(sym) MLD_CONCAT(_, MLD_NAMESPACE(sym))+#endif++/*+ * On X86_64 if control-flow protections (CET) are enabled (through+ * -fcf-protection=), we add an endbr64 instruction at every global function+ * label. See sys.h for more details+ */+#if defined(MLD_SYS_X86_64)+#define MLD_ASM_FN_SYMBOL(sym) MLD_ASM_NAMESPACE(sym) : MLD_CET_ENDBR+#elif defined(MLD_SYS_ARMV81M_MVE)+/* clang-format off */+#define MLD_ASM_FN_SYMBOL(sym) \+ .type MLD_ASM_NAMESPACE(sym), %function; \+ MLD_ASM_NAMESPACE(sym) :+/* clang-format on */+#else /* !MLD_SYS_X86_64 && MLD_SYS_ARMV81M_MVE */+#define MLD_ASM_FN_SYMBOL(sym) MLD_ASM_NAMESPACE(sym) :+#endif /* !MLD_SYS_X86_64 && !MLD_SYS_ARMV81M_MVE */++/*+ * Output the size of an assembly function.+ */+#if defined(__ELF__)+#define MLD_ASM_FN_SIZE(sym) \+ .size MLD_ASM_NAMESPACE(sym), .- MLD_ASM_NAMESPACE(sym)+#else+#define MLD_ASM_FN_SIZE(sym)+#endif++/* We aim to simplify the user's life by supporting builds where+ * all source files are included, even those that are not needed.+ * Those files are appropriately guarded and will be empty when unneeded.+ * The following is to avoid compilers complaining about this. */+#define MLD_EMPTY_CU(s) extern int MLD_NAMESPACE_KL(empty_cu_##s);++/* MLD_CONFIG_NO_ASM takes precedence over MLD_USE_NATIVE_XXX */+#if defined(MLD_CONFIG_NO_ASM)+#undef MLD_CONFIG_USE_NATIVE_BACKEND_ARITH+#undef MLD_CONFIG_USE_NATIVE_BACKEND_FIPS202+#endif++#if defined(MLD_CONFIG_USE_NATIVE_BACKEND_ARITH) && \+ !defined(MLD_CONFIG_ARITH_BACKEND_FILE)+#error Bad configuration: MLD_CONFIG_USE_NATIVE_BACKEND_ARITH is set, but MLD_CONFIG_ARITH_BACKEND_FILE is not.+#endif++#if defined(MLD_CONFIG_USE_NATIVE_BACKEND_FIPS202) && \+ !defined(MLD_CONFIG_FIPS202_BACKEND_FILE)+#error Bad configuration: MLD_CONFIG_USE_NATIVE_BACKEND_FIPS202 is set, but MLD_CONFIG_FIPS202_BACKEND_FILE is not.+#endif++#if defined(MLD_CONFIG_NO_RANDOMIZED_API) && defined(MLD_CONFIG_KEYGEN_PCT)+#error Bad configuration: MLD_CONFIG_NO_RANDOMIZED_API is incompatible with MLD_CONFIG_KEYGEN_PCT as the current PCT implementation requires signature()+#endif++#if defined(MLD_CONFIG_NO_SIGN_API) && defined(MLD_CONFIG_KEYGEN_PCT)+#error Bad configuration: MLD_CONFIG_NO_SIGN_API is incompatible with MLD_CONFIG_KEYGEN_PCT as the current PCT implementation requires signature()+#endif++#if defined(MLD_CONFIG_NO_VERIFY_API) && defined(MLD_CONFIG_KEYGEN_PCT)+#error Bad configuration: MLD_CONFIG_NO_VERIFY_API is incompatible with MLD_CONFIG_KEYGEN_PCT as the current PCT implementation requires verify()+#endif++#if defined(MLD_CONFIG_CORE_API_ONLY) && defined(MLD_CONFIG_KEYGEN_PCT)+#error Bad configuration: MLD_CONFIG_CORE_API_ONLY is incompatible with MLD_CONFIG_KEYGEN_PCT as the current PCT implementation requires signature() and verify()+#endif++#if defined(MLD_CONFIG_USE_NATIVE_BACKEND_ARITH)+#include MLD_CONFIG_ARITH_BACKEND_FILE+/* Include to enforce consistency of API and implementation,+ * and conduct sanity checks on the backend.+ *+ * Keep this _after_ the inclusion of the backend; otherwise,+ * the sanity checks won't have an effect. */+#if defined(MLD_CHECK_APIS) && !defined(__ASSEMBLER__)+#include "native/api.h"+#endif+#endif /* MLD_CONFIG_USE_NATIVE_BACKEND_ARITH */++#if defined(MLD_CONFIG_USE_NATIVE_BACKEND_FIPS202)+#include MLD_CONFIG_FIPS202_BACKEND_FILE+/* Include to enforce consistency of API and implementation,+ * and conduct sanity checks on the backend.+ *+ * Keep this _after_ the inclusion of the backend; otherwise,+ * the sanity checks won't have an effect. */+#if defined(MLD_CHECK_APIS) && !defined(__ASSEMBLER__)+#include "fips202/native/api.h"+#endif+#endif /* MLD_CONFIG_USE_NATIVE_BACKEND_FIPS202 */++#if !defined(MLD_CONFIG_FIPS202_CUSTOM_HEADER)+#define MLD_FIPS202_HEADER_FILE "fips202/fips202.h"+#else+#define MLD_FIPS202_HEADER_FILE MLD_CONFIG_FIPS202_CUSTOM_HEADER+#endif++#if !defined(MLD_CONFIG_FIPS202X4_CUSTOM_HEADER)+#define MLD_FIPS202X4_HEADER_FILE "fips202/fips202x4.h"+#else+#define MLD_FIPS202X4_HEADER_FILE MLD_CONFIG_FIPS202X4_CUSTOM_HEADER+#endif++/* Standard library function replacements */+#if !defined(__ASSEMBLER__)+#if !defined(MLD_CONFIG_CUSTOM_MEMCPY)+#include <string.h>+#define mld_memcpy memcpy+#endif++#if !defined(MLD_CONFIG_CUSTOM_MEMSET)+#include <string.h>+#define mld_memset memset+#endif++/* Allocation macros for large local structures+ *+ * MLD_ALLOC(v, T, N) declares T *v and attempts to point it to an T[N]+ * MLD_FREE(v, T, N) zeroizes and frees the allocation+ *+ * Default implementation uses stack allocation.+ * Can be overridden by setting the config option MLD_CONFIG_CUSTOM_ALLOC_FREE+ * and defining MLD_CUSTOM_ALLOC and MLD_CUSTOM_FREE.+ */+#if defined(MLD_CONFIG_CUSTOM_ALLOC_FREE) != \+ (defined(MLD_CUSTOM_ALLOC) && defined(MLD_CUSTOM_FREE))+#error Bad configuration: MLD_CONFIG_CUSTOM_ALLOC_FREE must be set together with MLD_CUSTOM_ALLOC and MLD_CUSTOM_FREE+#endif++/* Context-parameter machinery (MLD_CONTEXT_PARAMETERS_n and related config+ * checks). Kept in a separate, level-generic header for readability; included+ * here so it is available to the allocation macros below and to all consumers+ * of common.h. */+#include "context.h"++#if !defined(MLD_CONFIG_CUSTOM_ALLOC_FREE)+/* Default: stack allocation */++/* This is a declaration macro, not an expression macro: T is a type and v is+ * a declarator, neither of which can be wrapped in parentheses. The+ * bugprone-macro-parentheses diagnostic is therefore a false positive here. */+#define MLD_ALLOC(v, T, N, context) \+ MLD_ALIGN T mld_alloc_##v[N]; \+ T *v = mld_alloc_##v /* NOLINT(bugprone-macro-parentheses) */++/* The MLD_FREE macro body references mld_zeroize(), which is declared in+ * ct.h. We deliberately do NOT include ct.h here: doing so would create a+ * circular dependency (ct.h includes common.h), and common.h itself never+ * calls mld_zeroize() -- only the macro expansion does. Each translation+ * unit that uses MLD_FREE therefore includes ct.h directly. */+#define MLD_FREE(v, T, N, context) \+ do \+ { \+ MLD_CONTEXT_UNUSED(context); \+ mld_zeroize(mld_alloc_##v, sizeof(mld_alloc_##v)); \+ (v) = NULL; \+ } while (0)++#else /* !MLD_CONFIG_CUSTOM_ALLOC_FREE */++/* Custom allocation */++/*+ * The indirection here is necessary to use MLD_CONTEXT_PARAMETERS_3 here.+ */+#define MLD_APPLY(f, args) f args++#define MLD_ALLOC(v, T, N, context) \+ MLD_APPLY(MLD_CUSTOM_ALLOC, MLD_CONTEXT_PARAMETERS_3(v, T, N, context))++#define MLD_FREE(v, T, N, context) \+ do \+ { \+ if (v != NULL) \+ { \+ mld_zeroize(v, sizeof(T) * (N)); \+ MLD_APPLY(MLD_CUSTOM_FREE, MLD_CONTEXT_PARAMETERS_3(v, T, N, context)); \+ v = NULL; \+ } \+ } while (0)++#endif /* MLD_CONFIG_CUSTOM_ALLOC_FREE */++/****************************** Error codes ***********************************/++/* Generic failure condition, reserved for failures not covered by a more+ * specific error code. */+#define MLD_ERR_FAIL (-1)+/* An allocation failed. This can only happen if MLD_CONFIG_CUSTOM_ALLOC_FREE+ * is defined and the provided MLD_CUSTOM_ALLOC can fail. */+#define MLD_ERR_OUT_OF_MEMORY (-2)+/* An RNG failure occurred. Might be due to insufficient entropy or+ * system misconfiguration. */+#define MLD_ERR_RNG_FAIL (-3)+/* The signing rejection-sampling loop exceeded+ * MLD_CONFIG_MAX_SIGNING_ATTEMPTS iterations without producing a valid+ * signature. With a FIPS 204 Appendix C compliant bound (>= 821) this+ * has probability < 2^-256. */+#define MLD_ERR_SIGN_ATTEMPTS_EXHAUSTED (-4)+/* Signing was paused before completing, at the request of a caller-provided+ * MLD_CONFIG_SIGN_HOOK_ATTEMPT hook (see mldsa_native_config.h). The caller+ * resumes by re-invoking signing with the same inputs; the attempt hook,+ * together with MLD_CONFIG_SIGN_HOOK_RESUME, decides where to continue. */+#define MLD_ERR_SIGNING_PAUSED (-5)+/* Signature verification failed: the signature is not valid for the given+ * message and public key. Returned by the verification API. */+#define MLD_ERR_INVALID_SIGNATURE (-6)+/* Secret key validation failed: the secret key is malformed or internally+ * inconsistent. Returned by pk_from_sk. */+#define MLD_ERR_INVALID_KEY (-7)+/* The Pairwise Consistency Test failed. Only possible when+ * MLD_CONFIG_KEYGEN_PCT is enabled; signals that the freshly generated key+ * pair failed its sign/verify self-test. */+#define MLD_ERR_PCT_FAIL (-8)+/* An argument was invalid, e.g. an unsupported pre-hash algorithm or a context+ * string longer than 255 bytes. */+#define MLD_ERR_INVALID_ARG (-9)+++#endif /* !__ASSEMBLER__ */++#endif /* !MLD_COMMON_H */
@@ -0,0 +1,152 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+#ifndef MLD_CONTEXT_H+#define MLD_CONTEXT_H++/* This header is included by common.h once the configuration has been pulled+ * in; it is not meant to be included directly. */+#if !defined(__ASSEMBLER__)++#include <stdint.h>+#include "cbmc.h"+#include "sys.h"++/*+ * If the integration wants to provide a context parameter for use in+ * platform-specific hooks, then it should define this parameter.+ *+ * The MLD_CONTEXT_PARAMETERS_n macros are intended to be used with macros+ * defining the function names and expand to either pass or discard the context+ * argument as required by the current build. If there is no context parameter+ * requested then these are removed from the prototypes and from all calls.+ */+#ifdef MLD_CONFIG_CONTEXT_PARAMETER+#define MLD_CONTEXT_PARAMETERS_0(context) (context)+#define MLD_CONTEXT_PARAMETERS_1(arg0, context) (arg0, context)+#define MLD_CONTEXT_PARAMETERS_2(arg0, arg1, context) (arg0, arg1, context)+#define MLD_CONTEXT_PARAMETERS_3(arg0, arg1, arg2, context) \+ (arg0, arg1, arg2, context)+#define MLD_CONTEXT_PARAMETERS_4(arg0, arg1, arg2, arg3, context) \+ (arg0, arg1, arg2, arg3, context)+#define MLD_CONTEXT_PARAMETERS_5(arg0, arg1, arg2, arg3, arg4, context) \+ (arg0, arg1, arg2, arg3, arg4, context)+#define MLD_CONTEXT_PARAMETERS_6(arg0, arg1, arg2, arg3, arg4, arg5, context) \+ (arg0, arg1, arg2, arg3, arg4, arg5, context)+#define MLD_CONTEXT_PARAMETERS_7(arg0, arg1, arg2, arg3, arg4, arg5, arg6, \+ context) \+ (arg0, arg1, arg2, arg3, arg4, arg5, arg6, context)+#define MLD_CONTEXT_PARAMETERS_8(arg0, arg1, arg2, arg3, arg4, arg5, arg6, \+ arg7, context) \+ (arg0, arg1, arg2, arg3, arg4, arg5, arg6, arg7, context)+#define MLD_CONTEXT_PARAMETERS_9(arg0, arg1, arg2, arg3, arg4, arg5, arg6, \+ arg7, arg8, context) \+ (arg0, arg1, arg2, arg3, arg4, arg5, arg6, arg7, arg8, context)+#else /* MLD_CONFIG_CONTEXT_PARAMETER */+#define MLD_CONTEXT_PARAMETERS_0(context) ()+#define MLD_CONTEXT_PARAMETERS_1(arg0, context) (arg0)+#define MLD_CONTEXT_PARAMETERS_2(arg0, arg1, context) (arg0, arg1)+#define MLD_CONTEXT_PARAMETERS_3(arg0, arg1, arg2, context) (arg0, arg1, arg2)+#define MLD_CONTEXT_PARAMETERS_4(arg0, arg1, arg2, arg3, context) \+ (arg0, arg1, arg2, arg3)+#define MLD_CONTEXT_PARAMETERS_5(arg0, arg1, arg2, arg3, arg4, context) \+ (arg0, arg1, arg2, arg3, arg4)+#define MLD_CONTEXT_PARAMETERS_6(arg0, arg1, arg2, arg3, arg4, arg5, context) \+ (arg0, arg1, arg2, arg3, arg4, arg5)+#define MLD_CONTEXT_PARAMETERS_7(arg0, arg1, arg2, arg3, arg4, arg5, arg6, \+ context) \+ (arg0, arg1, arg2, arg3, arg4, arg5, arg6)+#define MLD_CONTEXT_PARAMETERS_8(arg0, arg1, arg2, arg3, arg4, arg5, arg6, \+ arg7, context) \+ (arg0, arg1, arg2, arg3, arg4, arg5, arg6, arg7)+#define MLD_CONTEXT_PARAMETERS_9(arg0, arg1, arg2, arg3, arg4, arg5, arg6, \+ arg7, arg8, context) \+ (arg0, arg1, arg2, arg3, arg4, arg5, arg6, arg7, arg8)+#endif /* !MLD_CONFIG_CONTEXT_PARAMETER */++/* Consume a context parameter carried only for the integration's benefit,+ * avoiding -Wunused-parameter; expands to nothing when no context is+ * configured. */+#if defined(MLD_CONFIG_CONTEXT_PARAMETER)+#define MLD_CONTEXT_UNUSED(context) ((void)(context))+#else+#define MLD_CONTEXT_UNUSED(context) ((void)0)+#endif++#if defined(MLD_CONFIG_CONTEXT_PARAMETER_TYPE) != \+ defined(MLD_CONFIG_CONTEXT_PARAMETER)+#error MLD_CONFIG_CONTEXT_PARAMETER_TYPE must be defined if and only if MLD_CONFIG_CONTEXT_PARAMETER is defined+#endif++/* The signing hooks tie into the rejection-sampling loop. A pausing attempt+ * hook only reproduces the uninterrupted signature if the randomness is fixed+ * across calls, and thus requires the deterministic API.+ * For now we impose that requirement on all three hooks uniformly: enabling any+ * of them requires MLD_CONFIG_NO_RANDOMIZED_API. This also rules out+ * MLD_CONFIG_KEYGEN_PCT (whose PCT needs the randomized signature(), see+ * common.h).+ *+ * A logging-only use (attempt always returns 0; resume/finish merely observe)+ * would be safe with the randomized API too, but the restriction is applied+ * uniformly for now. */+#if (defined(MLD_CONFIG_SIGN_HOOK_RESUME) || \+ defined(MLD_CONFIG_SIGN_HOOK_ATTEMPT) || \+ defined(MLD_CONFIG_SIGN_HOOK_FINISH)) && \+ !defined(MLD_CONFIG_NO_RANDOMIZED_API)+#error Signing hooks (MLD_CONFIG_SIGN_HOOK_RESUME / _ATTEMPT / _FINISH) require MLD_CONFIG_NO_RANDOMIZED_API+#endif /* (MLD_CONFIG_SIGN_HOOK_RESUME || MLD_CONFIG_SIGN_HOOK_ATTEMPT || \+ MLD_CONFIG_SIGN_HOOK_FINISH) && !MLD_CONFIG_NO_RANDOMIZED_API */++/* Signing hooks (MLD_CONFIG_SIGN_HOOK_RESUME / _ATTEMPT / _FINISH; documented+ * in mldsa_native_config.h). The following macros route the call sites to+ * mld_sign_hook_*, appending or dropping the context argument; each unset hook+ * uses the dummy below. */+#define mld_sign_resume mld_sign_hook_resume MLD_CONTEXT_PARAMETERS_0+#define mld_sign_attempt mld_sign_hook_attempt MLD_CONTEXT_PARAMETERS_1+#define mld_sign_finish mld_sign_hook_finish MLD_CONTEXT_PARAMETERS_1++/* We don't use mld_sign_resume here because MLD_CONTEXT_PARAMETERS_0 is+ * unsuitable for function declarations: it misses `void` as the placeholder+ * argument. */+#if !defined(MLD_CONFIG_SIGN_HOOK_RESUME)+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE uint16_t mld_sign_hook_resume(+#if defined(MLD_CONFIG_CONTEXT_PARAMETER)+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context+#else+ void+#endif+)+__contract__(assigns() ensures(1))+{+ MLD_CONTEXT_UNUSED(context);+ return 0;+}+#endif /* !MLD_CONFIG_SIGN_HOOK_RESUME */++#if !defined(MLD_CONFIG_SIGN_HOOK_ATTEMPT)+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_sign_attempt(+ uint16_t attempt, MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+__contract__(assigns() ensures(1))+{+ ((void)attempt);+ MLD_CONTEXT_UNUSED(context);+ return 0;+}+#endif /* !MLD_CONFIG_SIGN_HOOK_ATTEMPT */++#if !defined(MLD_CONFIG_SIGN_HOOK_FINISH)+static MLD_INLINE void mld_sign_finish(+ uint16_t attempt, MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+__contract__(assigns() ensures(1))+{+ ((void)attempt);+ MLD_CONTEXT_UNUSED(context);+}+#endif /* !MLD_CONFIG_SIGN_HOOK_FINISH */++#endif /* !__ASSEMBLER__ */++#endif /* !MLD_CONTEXT_H */
@@ -0,0 +1,21 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+#include "ct.h"++#if !defined(MLD_USE_ASM_VALUE_BARRIER) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)+/*+ * Masking value used in constant-time functions from+ * ct.h to block the compiler's range analysis and+ * thereby reduce the risk of compiler-introduced branches.+ */+volatile uint64_t mld_ct_opt_blocker_u64 = 0;++#else /* !MLD_USE_ASM_VALUE_BARRIER && !MLD_CONFIG_MULTILEVEL_NO_SHARED */++MLD_EMPTY_CU(ct)++#endif /* !(!MLD_USE_ASM_VALUE_BARRIER && !MLD_CONFIG_MULTILEVEL_NO_SHARED) */
@@ -0,0 +1,373 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS204]+ * FIPS 204 Module-Lattice-Based Digital Signature Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/204/final+ *+ * - [libmceliece]+ * libmceliece implementation of Classic McEliece+ * Bernstein, Chou+ * https://lib.mceliece.org/+ *+ * - [optblocker]+ * PQC forum post on opt-blockers using volatile globals+ * Daniel J. Bernstein+ * https://groups.google.com/a/list.nist.gov/g/pqc-forum/c/hqbtIGFKIpU/m/H14H0wOlBgAJ+ */++#ifndef MLD_CT_H+#define MLD_CT_H++#include "cbmc.h"+#include "common.h"++/* Constant-time comparisons and conditional operations++ We reduce the risk for compilation into variable-time code+ through the use of 'value barriers'.++ Functionally, a value barrier is a no-op. To the compiler, however,+ it constitutes an arbitrary modification of its input, and therefore+ harden's value propagation and range analysis.++ We consider two approaches to implement a value barrier:+ - An empty inline asm block which marks the target value as clobbered.+ - XOR'ing with the value of a volatile global that's set to 0;+ see @[optblocker] for a discussion of this idea, and+ @[libmceliece, inttypes/crypto_intN.h] for an implementation.++ The first approach is cheap because it only prevents the compiler+ from reasoning about the value of the variable past the barrier,+ but does not directly generate additional instructions.++ The second approach generates redundant loads and XOR operations+ and therefore comes at a higher runtime cost. However, it appears+ more robust towards optimization, as compilers should never drop+ a volatile load.++ We use the empty-ASM value barrier for GCC and clang, and fall+ back to the global volatile barrier otherwise.++ The global value barrier can be forced by setting+ MLD_CONFIG_NO_ASM_VALUE_BARRIER.++*/++#if defined(MLD_HAVE_INLINE_ASM) && !defined(MLD_CONFIG_NO_ASM_VALUE_BARRIER)+#define MLD_USE_ASM_VALUE_BARRIER+#endif+++#if !defined(MLD_USE_ASM_VALUE_BARRIER)+/*+ * Declaration of global volatile that the global value barrier+ * is loading from and masking with.+ */+#define mld_ct_opt_blocker_u64 MLD_NAMESPACE(ct_opt_blocker_u64)+extern volatile uint64_t mld_ct_opt_blocker_u64;+++/* Helper functions for obtaining global masks of various sizes */++/* This contract is not proved but treated as an axiom.+ *+ * Its validity relies on the assumption that the global opt-blocker+ * constant mld_ct_opt_blocker_u64 is not modified.+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE uint64_t mld_ct_get_optblocker_u64(void)+__contract__(ensures(return_value == 0)) { return mld_ct_opt_blocker_u64; }++MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int64_t mld_ct_get_optblocker_i64(void)+__contract__(ensures(return_value == 0)) { return (int64_t)mld_ct_get_optblocker_u64(); }++MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE uint32_t mld_ct_get_optblocker_u32(void)+__contract__(ensures(return_value == 0)) { return (uint32_t)mld_ct_get_optblocker_u64(); }++MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE uint8_t mld_ct_get_optblocker_u8(void)+__contract__(ensures(return_value == 0)) { return (uint8_t)mld_ct_get_optblocker_u64(); }++/* Opt-blocker based implementation of value barriers */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int64_t mld_value_barrier_i64(int64_t b)+__contract__(ensures(return_value == b)) { return (b ^ mld_ct_get_optblocker_i64()); }++MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE uint32_t mld_value_barrier_u32(uint32_t b)+__contract__(ensures(return_value == b)) { return (b ^ mld_ct_get_optblocker_u32()); }++MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE uint8_t mld_value_barrier_u8(uint8_t b)+__contract__(ensures(return_value == b)) { return (b ^ mld_ct_get_optblocker_u8()); }+++#else /* !MLD_USE_ASM_VALUE_BARRIER */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int64_t mld_value_barrier_i64(int64_t b)+__contract__(ensures(return_value == b))+{+ __asm__ volatile("" : "+r"(b));+ return b;+}++MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE uint32_t mld_value_barrier_u32(uint32_t b)+__contract__(ensures(return_value == b))+{+ __asm__ volatile("" : "+r"(b));+ return b;+}++MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE uint8_t mld_value_barrier_u8(uint8_t b)+__contract__(ensures(return_value == b))+{+ __asm__ volatile("" : "+r"(b));+ return b;+}+#endif /* MLD_USE_ASM_VALUE_BARRIER */++#ifdef CBMC+#pragma CPROVER check push+#pragma CPROVER check disable "conversion"+#endif++/**+ * Cast uint32 value to int32.+ *+ * @param x Input value.+ *+ * @return For uint32_t x, the unique y in int32_t so that x == y mod 2^32.+ * Concretely:+ * - x < 2^31: returns x+ * - x >= 2^31: returns x - 2^32+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_ALWAYS_INLINE int32_t mld_cast_uint32_to_int32(uint32_t x)+{+ /*+ * PORTABILITY: This relies on uint32_t -> int32_t+ * being implemented as the inverse of int32_t -> uint32_t,+ * which is implementation-defined (C99 6.3.1.3 (3))+ * CBMC (correctly) fails to prove this conversion is OK,+ * so we have to suppress that check here+ */+ return (int32_t)x;+}++#ifdef CBMC+#pragma CPROVER check pop+#endif+++/**+ * Cast int64 value to uint32 as per C standard.+ *+ * @param x Input value.+ *+ * @return For int64_t x, the unique y in uint32_t so that x == y mod 2^32.+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_ALWAYS_INLINE uint32_t mld_cast_int64_to_uint32(int64_t x)+{+ return (uint32_t)(x & (int64_t)UINT32_MAX);+}++/**+ * Cast int32 value to uint32 as per C standard.+ *+ * @param x Input value.+ *+ * @return For int32_t x, the unique y in uint32_t so that x == y mod 2^32.+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_ALWAYS_INLINE uint32_t mld_cast_int32_to_uint32(int32_t x)+{+ return mld_cast_int64_to_uint32((int64_t)x);+}++/**+ * Functionally equivalent to cond ? a : b, but implemented with guards against+ * compiler-introduced branches.+ *+ * @param a First alternative.+ * @param b Second alternative.+ * @param cond Condition variable.+ *+ * @return a if cond is 0xFFFFFFFF, b if cond is 0.+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int32_t mld_ct_sel_int32(int32_t a, int32_t b, uint32_t cond)+__contract__(+ requires(cond == 0x0 || cond == 0xFFFFFFFF)+ ensures(return_value == (cond ? a : b))+)+{+ uint32_t au = mld_cast_int32_to_uint32(a);+ uint32_t bu = mld_cast_int32_to_uint32(b);+ uint32_t res = bu ^ (mld_value_barrier_u32(cond) & (au ^ bu));+ return mld_cast_uint32_to_int32(res);+}++/**+ * Return 0 if input is zero, and -1 otherwise.+ *+ * @param x Value to be converted into a mask.+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE uint32_t mld_ct_cmask_nonzero_u32(uint32_t x)+__contract__(ensures(return_value == ((x == 0) ? 0 : 0xFFFFFFFF)))+{+ int64_t tmp = mld_value_barrier_i64(-((int64_t)x));+ /*+ * PORTABILITY: Right-shift on a signed integer is+ * implementation-defined for negative left argument.+ * Here, we assume it's sign-preserving "arithmetic" shift right.+ * See (C99 6.5.7 (5))+ */+ tmp >>= 32;+ return mld_cast_int64_to_uint32(tmp);+}++/**+ * Return 0 if input is zero, and -1 otherwise.+ *+ * @param x Value to be converted into a mask.+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE uint8_t mld_ct_cmask_nonzero_u8(uint8_t x)+__contract__(ensures(return_value == ((x == 0) ? 0 : 0xFF)))+{+ uint32_t mask = mld_ct_cmask_nonzero_u32((uint32_t)x);+ return (uint8_t)(mask & 0xFF);+}++/**+ * Return 0 if input is non-negative, and -1 otherwise.+ *+ * @param x Value to be converted into a mask.+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE uint32_t mld_ct_cmask_neg_i32(int32_t x)+__contract__(+ ensures(return_value == ((x < 0) ? 0xFFFFFFFF : 0))+)+{+ int64_t tmp = mld_value_barrier_i64((int64_t)x);+ /*+ * PORTABILITY: Right-shift on a signed integer is+ * implementation-defined for negative left argument.+ * Here, we assume it's sign-preserving "arithmetic" shift right.+ * See (C99 6.5.7 (5))+ */+ tmp >>= 31;+ return mld_cast_int64_to_uint32(tmp);+}++/**+ * Return -x if x<0, x otherwise.+ *+ * @param x Input value.+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int32_t mld_ct_abs_i32(int32_t x)+__contract__(+ requires(x >= -INT32_MAX)+ ensures(return_value == ((x < 0) ? -x : x))+)+{+ return mld_ct_sel_int32(-x, x, mld_ct_cmask_neg_i32(x));+}++/**+ * Compare two arrays for equality in constant time.+ *+ * @param[in] a Pointer to first byte array.+ * @param[in] b Pointer to second byte array.+ * @param len Length of the byte arrays, upper-bounded to UINT16_MAX to+ * control proof complexity only.+ *+ * @return 0 if the byte arrays are equal, 0xFF otherwise.+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE uint8_t mld_ct_memcmp(const uint8_t *a, const uint8_t *b,+ const size_t len)+__contract__(+ requires(len <= UINT16_MAX)+ requires(memory_no_alias(a, len))+ requires(memory_no_alias(b, len))+ ensures((return_value == 0) || (return_value == 0xFF))+ ensures((return_value == 0) == forall(i, 0, len, (a[i] == b[i]))))+{+ uint8_t r = 0, s = 0;+ unsigned i;++ for (i = 0; i < len; i++)+ __loop__(+ invariant(i <= len)+ invariant((r == 0) == (forall(k, 0, i, (a[k] == b[k]))))+ decreases(len - i))+ {+ r |= a[i] ^ b[i];+ /* s is useless, but prevents the loop from being aborted once r=0xff. */+ s ^= a[i] ^ b[i];+ }++ /*+ * - Convert r into a mask; this may not be necessary, but is an additional+ * safeguard+ * towards leaking information about a and b.+ * - XOR twice with s, separated by a value barrier, to prevent the compile+ * from dropping the s computation in the loop.+ */+ return (mld_value_barrier_u8(mld_ct_cmask_nonzero_u8(r) ^ s) ^ s);+}++/**+ * Force-zeroize a buffer.+ *+ * @[FIPS204, Section 3.6.3] Destruction of intermediate values.+ *+ * @param[out] ptr Pointer to buffer to be zeroed.+ * @param len Amount of bytes to be zeroed.+ */+#if !defined(MLD_CONFIG_CUSTOM_ZEROIZE)+#if defined(MLD_SYS_WINDOWS)+#include <windows.h>+#elif !defined(MLD_HAVE_INLINE_ASM)+#error No plausibly-secure implementation of mld_zeroize available. Please provide your own using MLD_CONFIG_CUSTOM_ZEROIZE.+#endif++static MLD_INLINE void mld_zeroize(void *ptr, size_t len)+__contract__(+ requires(len <= UINT32_MAX)+ requires(memory_no_alias(ptr, len))+ assigns(memory_slice(ptr, len))+ ensures(array_zeroized_u8((uint8_t *)ptr, len)))+{+#if defined(MLD_SYS_WINDOWS)+ SecureZeroMemory(ptr, len);+#else+ mld_memset(ptr, 0, len);+ /* This follows OpenSSL and seems sufficient to prevent the compiler+ * from optimizing away the memset.+ *+ * If there was a reliable way to detect availability of memset_s(),+ * that would be preferred. */+ __asm__ __volatile__("" : : "r"(ptr) : "memory");+#endif /* !MLD_SYS_WINDOWS */+}+#endif /* !MLD_CONFIG_CUSTOM_ZEROIZE */++#endif /* !MLD_CT_H */
@@ -0,0 +1,75 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* NOTE: You can remove this file unless you compile with MLDSA_DEBUG. */++#include "common.h"++#if !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++#if defined(MLDSA_DEBUG)++#include <inttypes.h>+#include <stdio.h>+#include <stdlib.h>+#include "debug.h"++#define MLD_DEBUG_ERROR_HEADER "[ERROR:%s:%04d] "++MLD_INTERNAL_API+void mld_debug_check_assert(const char *file, int line, const int val)+{+ if (val == 0)+ {+ fprintf(stderr, MLD_DEBUG_ERROR_HEADER "Assertion failed (value %d)\n",+ file, line, val);+ exit(1);+ }+}++MLD_INTERNAL_API+void mld_debug_check_bounds(const char *file, int line, const int32_t *ptr,+ unsigned len, int64_t lower_bound_exclusive,+ int64_t upper_bound_exclusive)+{+ int err = 0;+ unsigned i;+ for (i = 0; i < len; i++)+ {+ int32_t val = ptr[i];+ if (!(val > lower_bound_exclusive && val < upper_bound_exclusive))+ {+ fprintf(stderr,+ MLD_DEBUG_ERROR_HEADER+ "Bounds assertion failed: Index %u, value %d out of bounds "+ "(%" PRId64 ",%" PRId64 ")\n",+ file, line, i, (int)val, lower_bound_exclusive,+ upper_bound_exclusive);+ err = 1;+ }+ }++ if (err == 1)+ {+ exit(1);+ }+}++#else /* MLDSA_DEBUG */++MLD_EMPTY_CU(debug)++#endif /* !MLDSA_DEBUG */++#else /* !MLD_CONFIG_MULTILEVEL_NO_SHARED */++MLD_EMPTY_CU(debug)++#endif /* MLD_CONFIG_MULTILEVEL_NO_SHARED */++/* To facilitate single-compilation-unit (SCU) builds, undefine all macros.+ * Don't modify by hand -- this is auto-generated by scripts/autogen. */+#undef MLD_DEBUG_ERROR_HEADER
@@ -0,0 +1,125 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+#ifndef MLD_DEBUG_H+#define MLD_DEBUG_H+#include "common.h"++#if defined(MLDSA_DEBUG)++/**+ * Check debug assertion.+ *+ * Prints an error message to stderr and calls exit(1) if not.+ *+ * @param file Filename.+ * @param line Line number.+ * @param val Value asserted to be non-zero.+ */+#define mld_debug_check_assert MLD_NAMESPACE(mldsa_debug_assert)+MLD_INTERNAL_API+void mld_debug_check_assert(const char *file, int line, const int val);++/**+ * Check whether values in an array of int32_t are within specified bounds.+ *+ * Prints an error message to stderr and calls exit(1) if not.+ *+ * @param file Filename.+ * @param line Line number.+ * @param[in] ptr Base of array to be checked.+ * @param len Number of int32_t in ptr.+ * @param lower_bound_exclusive Exclusive lower bound.+ * @param upper_bound_exclusive Exclusive upper bound.+ */+#define mld_debug_check_bounds MLD_NAMESPACE(mldsa_debug_check_bounds)+MLD_INTERNAL_API+void mld_debug_check_bounds(const char *file, int line, const int32_t *ptr,+ unsigned len, int64_t lower_bound_exclusive,+ int64_t upper_bound_exclusive);++/* Check assertion, calling exit() upon failure+ *+ * val: Value that's asserted to be non-zero+ */+#define mld_assert(val) mld_debug_check_assert(__FILE__, __LINE__, (val))++/* Check bounds in array of int32_t's+ * ptr: Base of int32_t array; will be explicitly cast to int32_t*,+ * so you may pass a byte-compatible type such as mld_poly or mld_polyvec.+ * len: Number of int32_t in array+ * value_lb: Inclusive lower value bound+ * value_ub: Exclusive upper value bound */+#define mld_assert_bound(ptr, len, value_lb, value_ub) \+ mld_debug_check_bounds(__FILE__, __LINE__, (const int32_t *)(ptr), (len), \+ ((int64_t)(value_lb)) - 1, (value_ub))++/* Check absolute bounds in array of int32_t's+ * ptr: Base of array, expression of type int32_t*+ * len: Number of int32_t in array+ * value_abs_bd: Exclusive absolute upper bound */+#define mld_assert_abs_bound(ptr, len, value_abs_bd) \+ mld_assert_bound((ptr), (len), (-((int64_t)(value_abs_bd)) + 1), \+ (value_abs_bd))++/* Version of bounds assertions for 2-dimensional arrays */+#define mld_assert_bound_2d(ptr, len0, len1, value_lb, value_ub) \+ mld_assert_bound((ptr), ((len0) * (len1)), (value_lb), (value_ub))++#define mld_assert_abs_bound_2d(ptr, len0, len1, value_abs_bd) \+ mld_assert_abs_bound((ptr), ((len0) * (len1)), (value_abs_bd))++/* When running CBMC, convert debug assertions into proof obligations */+#elif defined(CBMC)+#include "cbmc.h"++#define mld_assert(val) cassert(val)++#define mld_assert_bound(ptr, len, value_lb, value_ub) \+ cassert(array_bound(((int32_t *)(ptr)), 0, (len), (value_lb), (value_ub)))++#define mld_assert_abs_bound(ptr, len, value_abs_bd) \+ cassert(array_abs_bound(((int32_t *)(ptr)), 0, (len), (value_abs_bd)))++/* Because of https://github.com/diffblue/cbmc/issues/8570, we can't+ * just use a single flattened array_bound(...) here. */+#define mld_assert_bound_2d(ptr, M, N, value_lb, value_ub) \+ cassert(forall(kN, 0, (M), \+ array_bound(&((int32_t (*)[(N)])(ptr))[kN][0], 0, (N), \+ (value_lb), (value_ub))))++#define mld_assert_abs_bound_2d(ptr, M, N, value_abs_bd) \+ cassert(forall(kN, 0, (M), \+ array_abs_bound(&((int32_t (*)[(N)])(ptr))[kN][0], 0, (N), \+ (value_abs_bd))))++#else /* !MLDSA_DEBUG && CBMC */++#define mld_assert(val) \+ do \+ { \+ } while (0)+#define mld_assert_bound(ptr, len, value_lb, value_ub) \+ do \+ { \+ } while (0)+#define mld_assert_abs_bound(ptr, len, value_abs_bd) \+ do \+ { \+ } while (0)++#define mld_assert_bound_2d(ptr, len0, len1, value_lb, value_ub) \+ do \+ { \+ } while (0)++#define mld_assert_abs_bound_2d(ptr, len0, len1, value_abs_bd) \+ do \+ { \+ } while (0)+++#endif /* !MLDSA_DEBUG && !CBMC */+#endif /* !MLD_DEBUG_H */
@@ -0,0 +1,270 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS204]+ * FIPS 204 Module-Lattice-Based Digital Signature Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/204/final+ *+ * - [mupq]+ * Common files for pqm4, pqm3, pqriscv+ * Kannwischer, Petri, Rijneveld, Schwabe, Stoffelen+ * https://github.com/mupq/mupq+ *+ * - [supercop]+ * SUPERCOP benchmarking framework+ * Daniel J. Bernstein+ * http://bench.cr.yp.to/supercop.html+ *+ * - [tweetfips]+ * 'tweetfips202' FIPS202 implementation+ * Van Assche, Bernstein, Schwabe+ * https://keccak.team/2015/tweetfips202.html+ */++/* Based on the CC0 implementation from @[mupq] and the public domain+ * implementation @[supercop, crypto_hash/keccakc512/simple/]+ * by Ronny Van Keer, and the public domain @[tweetfips] implementation. */++#include <stddef.h>++#include "../common.h"+#include "../ct.h"+#include "fips202.h"+#include "keccakf1600.h"+#if !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++/**+ * Initializes the Keccak state.+ *+ * @param[out] s Pointer to Keccak state.+ */+static void keccak_init(uint64_t s[MLD_KECCAK_LANES])+__contract__(+ requires(memory_no_alias(s, sizeof(uint64_t) * MLD_KECCAK_LANES))+ assigns(memory_slice(s, sizeof(uint64_t) * MLD_KECCAK_LANES))+)+{+ mld_memset(s, 0, sizeof(uint64_t) * MLD_KECCAK_LANES);+}++/**+ * Absorb step of Keccak; incremental.+ *+ * @param[in,out] s Pointer to Keccak state.+ * @param pos Position in current block to be absorbed.+ * @param r Rate in bytes (e.g., 168 for SHAKE128).+ * @param[in] in Pointer to input to be absorbed into s.+ * @param inlen Length of input in bytes.+ *+ * @return New position pos in current block.+ */+static unsigned int keccak_absorb(uint64_t s[MLD_KECCAK_LANES],+ unsigned int pos, unsigned int r,+ const uint8_t *in, size_t inlen)+__contract__(+ requires(inlen <= MLD_MAX_BUFFER_SIZE)+ requires(r > 0)+ requires(r < sizeof(uint64_t) * MLD_KECCAK_LANES)+ requires(pos <= r)+ requires(memory_no_alias(s, sizeof(uint64_t) * MLD_KECCAK_LANES))+ requires(memory_no_alias(in, inlen))+ assigns(memory_slice(s, sizeof(uint64_t) * MLD_KECCAK_LANES))+ ensures(return_value < r))+{+ while (inlen >= r - pos)+ __loop__(+ assigns(pos, in, inlen,+ memory_slice(s, sizeof(uint64_t) * MLD_KECCAK_LANES))+ invariant(inlen <= loop_entry(inlen))+ invariant(pos <= r)+ invariant(in == loop_entry(in) + (loop_entry(inlen) - inlen))+ decreases(inlen + pos))+ {+ mld_keccakf1600_xor_bytes(s, in, pos, r - pos);+ inlen -= r - pos;+ in += r - pos;+ mld_keccakf1600_permute(s);+ pos = 0;+ }+ /* Safety: At this point, inlen < r, so the truncation to unsigned is safe. */+ mld_keccakf1600_xor_bytes(s, in, pos, (unsigned)inlen);++ /* Safety: At this point, inlen < r and pos <= r so the truncation to unsigned+ * is safe. */+ return (unsigned)(pos + inlen);+}++/**+ * Finalize absorb step.+ *+ * @param[in,out] s Pointer to Keccak state.+ * @param pos Position in current block to be absorbed.+ * @param r Rate in bytes (e.g., 168 for SHAKE128).+ * @param p Domain separation byte.+ */+static void keccak_finalize(uint64_t s[MLD_KECCAK_LANES], unsigned int pos,+ unsigned int r, uint8_t p)+__contract__(+ requires(pos <= r && r < sizeof(uint64_t) * MLD_KECCAK_LANES)+ requires((r / 8) >= 1)+ requires(memory_no_alias(s, sizeof(uint64_t) * MLD_KECCAK_LANES))+ assigns(memory_slice(s, sizeof(uint64_t) * MLD_KECCAK_LANES))+)+{+ uint8_t b = 0x80;+ mld_keccakf1600_xor_bytes(s, &p, pos, 1);+ mld_keccakf1600_xor_bytes(s, &b, r - 1, 1);+}++/**+ * Squeeze step of Keccak. Squeezes arbitrarily many bytes. Modifies the+ * state. Can be called multiple times to keep squeezing, i.e., is+ * incremental.+ *+ * @param[out] out Pointer to output data.+ * @param outlen Number of bytes to be squeezed (written to out).+ * @param[in,out] s Pointer to input/output Keccak state.+ * @param pos Number of bytes in current block already squeezed.+ * @param r Rate in bytes (e.g., 168 for SHAKE128).+ *+ * @return New position pos in current block.+ */+static unsigned int keccak_squeeze(uint8_t *out, size_t outlen,+ uint64_t s[MLD_KECCAK_LANES],+ unsigned int pos, unsigned int r)+__contract__(+ requires((r == SHAKE128_RATE && pos <= SHAKE128_RATE) ||+ (r == SHAKE256_RATE && pos <= SHAKE256_RATE) ||+ (r == SHA3_512_RATE && pos <= SHA3_512_RATE))+ requires(outlen <= 8 * r /* somewhat arbitrary bound */)+ requires(memory_no_alias(s, sizeof(uint64_t) * MLD_KECCAK_LANES))+ requires(memory_no_alias(out, outlen))+ assigns(memory_slice(s, sizeof(uint64_t) * MLD_KECCAK_LANES))+ assigns(memory_slice(out, outlen))+ ensures(return_value <= r))+{+ unsigned int i;+ size_t out_offset = 0;++ /* Reference: This code is re-factored from the reference implementation+ * to facilitate proof with CBMC and to improve readability.+ *+ * Take a mutable copy of outlen to count down the number of bytes+ * still to squeeze. The initial value of outlen is needed for the CBMC+ * assigns() clauses. */+ size_t bytes_to_go = outlen;++ while (bytes_to_go > 0)+ __loop__(+ assigns(i, bytes_to_go, pos, out_offset, memory_slice(s, sizeof(uint64_t) * MLD_KECCAK_LANES), memory_slice(out, outlen))+ invariant(bytes_to_go <= outlen)+ invariant(out_offset == outlen - bytes_to_go)+ invariant(pos <= r)+ decreases(bytes_to_go)+ )+ {+ if (pos == r)+ {+ mld_keccakf1600_permute(s);+ pos = 0;+ }+ /* Safety: If bytes_to_go < r - pos, truncation to unsigned is safe. */+ i = bytes_to_go < r - pos ? (unsigned)bytes_to_go : r - pos;+ mld_keccakf1600_extract_bytes(s, out + out_offset, pos, i);+ bytes_to_go -= i;+ pos += i;+ out_offset += i;+ }++ return pos;+}++MLD_INTERNAL_API+void mld_shake128_init(mld_shake128ctx *state)+{+ keccak_init(state->s);+ state->pos = 0;+}++MLD_INTERNAL_API+void mld_shake128_absorb(mld_shake128ctx *state, const uint8_t *in,+ size_t inlen)+{+ state->pos = keccak_absorb(state->s, state->pos, SHAKE128_RATE, in, inlen);+}++MLD_INTERNAL_API+void mld_shake128_finalize(mld_shake128ctx *state)+{+ keccak_finalize(state->s, state->pos, SHAKE128_RATE, 0x1F);+ state->pos = SHAKE128_RATE;+}++MLD_INTERNAL_API+void mld_shake128_squeeze(uint8_t *out, size_t outlen, mld_shake128ctx *state)+{+ state->pos = keccak_squeeze(out, outlen, state->s, state->pos, SHAKE128_RATE);+}++MLD_INTERNAL_API+void mld_shake128_release(mld_shake128ctx *state)+{+ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ mld_zeroize(state, sizeof(mld_shake128ctx));+}++MLD_INTERNAL_API+void mld_shake256_init(mld_shake256ctx *state)+{+ keccak_init(state->s);+ state->pos = 0;+}++MLD_INTERNAL_API+void mld_shake256_absorb(mld_shake256ctx *state, const uint8_t *in,+ size_t inlen)+{+ state->pos = keccak_absorb(state->s, state->pos, SHAKE256_RATE, in, inlen);+}++MLD_INTERNAL_API+void mld_shake256_finalize(mld_shake256ctx *state)+{+ keccak_finalize(state->s, state->pos, SHAKE256_RATE, 0x1F);+ state->pos = SHAKE256_RATE;+}++MLD_INTERNAL_API+void mld_shake256_squeeze(uint8_t *out, size_t outlen, mld_shake256ctx *state)+{+ state->pos = keccak_squeeze(out, outlen, state->s, state->pos, SHAKE256_RATE);+}++MLD_INTERNAL_API+void mld_shake256_release(mld_shake256ctx *state)+{+ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ mld_zeroize(state, sizeof(mld_shake256ctx));+}++#if !defined(MLD_CONFIG_NO_KEYPAIR_API) || !defined(MLD_CONFIG_CORE_API_ONLY)+MLD_INTERNAL_API+void mld_shake256(uint8_t *out, size_t outlen, const uint8_t *in, size_t inlen)+{+ mld_shake256ctx state;++ mld_shake256_init(&state);+ mld_shake256_absorb(&state, in, inlen);+ mld_shake256_finalize(&state);+ mld_shake256_squeeze(out, outlen, &state);+ mld_shake256_release(&state);+}+#endif /* !MLD_CONFIG_NO_KEYPAIR_API || !MLD_CONFIG_CORE_API_ONLY */++#endif /* !MLD_CONFIG_MULTILEVEL_NO_SHARED */
@@ -0,0 +1,224 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+#ifndef MLD_FIPS202_FIPS202_H+#define MLD_FIPS202_FIPS202_H++#include <stddef.h>+#include "../cbmc.h"+#include "../common.h"++#define SHAKE128_RATE 168+#define SHAKE256_RATE 136+#define SHA3_256_RATE 136+#define SHA3_512_RATE 72+#define MLD_KECCAK_LANES 25+#define SHA3_256_HASHBYTES 32+#define SHA3_512_HASHBYTES 64+++/** Context for the incremental SHAKE128 XOF. */+typedef struct+{+ uint64_t s[MLD_KECCAK_LANES]; /**< Keccak state. */+ unsigned int pos; /**< Byte position within the current Keccak block. */+} mld_shake128ctx;++/** Context for the incremental SHAKE256 XOF. */+typedef struct+{+ uint64_t s[MLD_KECCAK_LANES]; /**< Keccak state. */+ unsigned int pos; /**< Byte position within the current Keccak block. */+} mld_shake256ctx;++#define mld_shake128_init MLD_NAMESPACE(shake128_init)+/**+ * Initializes state for use as SHAKE128 XOF.+ *+ * @param[out] state Pointer to (uninitialized) state.+ */+MLD_INTERNAL_API+void mld_shake128_init(mld_shake128ctx *state)+__contract__(+ requires(memory_no_alias(state, sizeof(mld_shake128ctx)))+ assigns(memory_slice(state, sizeof(mld_shake128ctx)))+ ensures(state->pos == 0)+);++#define mld_shake128_absorb MLD_NAMESPACE(shake128_absorb)+/**+ * Absorb step of the SHAKE128 XOF. Absorbs arbitrarily many bytes. Can be+ * called multiple times to absorb multiple chunks of data.+ *+ * @param[in,out] state Pointer to (initialized) output state.+ * @param[in] in Pointer to input to be absorbed into s.+ * @param inlen Length of input in bytes.+ */+MLD_INTERNAL_API+void mld_shake128_absorb(mld_shake128ctx *state, const uint8_t *in,+ size_t inlen)+__contract__(+ requires(inlen <= MLD_MAX_BUFFER_SIZE)+ requires(memory_no_alias(state, sizeof(mld_shake128ctx)))+ requires(memory_no_alias(in, inlen))+ requires(state->pos <= SHAKE128_RATE)+ assigns(memory_slice(state, sizeof(mld_shake128ctx)))+ ensures(state->pos <= SHAKE128_RATE)+);++#define mld_shake128_finalize MLD_NAMESPACE(shake128_finalize)+/**+ * Concludes the absorb phase of the SHAKE128 XOF.+ *+ * @param[in,out] state Pointer to state.+ */+MLD_INTERNAL_API+void mld_shake128_finalize(mld_shake128ctx *state)+__contract__(+ requires(memory_no_alias(state, sizeof(mld_shake128ctx)))+ requires(state->pos <= SHAKE128_RATE)+ assigns(memory_slice(state, sizeof(mld_shake128ctx)))+ ensures(state->pos <= SHAKE128_RATE)+);++#define mld_shake128_squeeze MLD_NAMESPACE(shake128_squeeze)+/**+ * Squeeze step of SHAKE128 XOF. Squeezes arbitrarily many bytes. Can be+ * called multiple times to keep squeezing.+ *+ * @param[out] out Pointer to output blocks.+ * @param outlen Number of bytes to be squeezed (written to output).+ * @param[in,out] state Pointer to input/output state.+ */+MLD_INTERNAL_API+void mld_shake128_squeeze(uint8_t *out, size_t outlen, mld_shake128ctx *state)+__contract__(+ requires(outlen <= 8 * SHAKE128_RATE /* somewhat arbitrary bound */)+ requires(memory_no_alias(state, sizeof(mld_shake128ctx)))+ requires(memory_no_alias(out, outlen))+ requires(state->pos <= SHAKE128_RATE)+ assigns(memory_slice(state, sizeof(mld_shake128ctx)))+ assigns(memory_slice(out, outlen))+ ensures(state->pos <= SHAKE128_RATE)+);++#define mld_shake128_release MLD_NAMESPACE(shake128_release)+/**+ * Release and securely zero the SHAKE128 state.+ *+ * @param[in,out] state Pointer to state.+ */+MLD_INTERNAL_API+void mld_shake128_release(mld_shake128ctx *state)+__contract__(+ requires(memory_no_alias(state, sizeof(mld_shake128ctx)))+ assigns(memory_slice(state, sizeof(mld_shake128ctx)))+);++#define mld_shake256_init MLD_NAMESPACE(shake256_init)+/**+ * Initializes state for use as SHAKE256 XOF.+ *+ * @param[out] state Pointer to (uninitialized) state.+ */+MLD_INTERNAL_API+void mld_shake256_init(mld_shake256ctx *state)+__contract__(+ requires(memory_no_alias(state, sizeof(mld_shake256ctx)))+ assigns(memory_slice(state, sizeof(mld_shake256ctx)))+ ensures(state->pos == 0)+);++#define mld_shake256_absorb MLD_NAMESPACE(shake256_absorb)+/**+ * Absorb step of the SHAKE256 XOF. Absorbs arbitrarily many bytes. Can be+ * called multiple times to absorb multiple chunks of data.+ *+ * @param[in,out] state Pointer to (initialized) output state.+ * @param[in] in Pointer to input to be absorbed into s.+ * @param inlen Length of input in bytes.+ */+MLD_INTERNAL_API+void mld_shake256_absorb(mld_shake256ctx *state, const uint8_t *in,+ size_t inlen)+__contract__(+ requires(inlen <= MLD_MAX_BUFFER_SIZE)+ requires(memory_no_alias(state, sizeof(mld_shake256ctx)))+ requires(memory_no_alias(in, inlen))+ requires(state->pos <= SHAKE256_RATE)+ assigns(memory_slice(state, sizeof(mld_shake256ctx)))+ ensures(state->pos <= SHAKE256_RATE)+);++#define mld_shake256_finalize MLD_NAMESPACE(shake256_finalize)+/**+ * Concludes the absorb phase of the SHAKE256 XOF.+ *+ * @param[in,out] state Pointer to state.+ */+MLD_INTERNAL_API+void mld_shake256_finalize(mld_shake256ctx *state)+__contract__(+ requires(memory_no_alias(state, sizeof(mld_shake256ctx)))+ requires(state->pos <= SHAKE256_RATE)+ assigns(memory_slice(state, sizeof(mld_shake256ctx)))+ ensures(state->pos <= SHAKE256_RATE)+);++#define mld_shake256_squeeze MLD_NAMESPACE(shake256_squeeze)+/**+ * Squeeze step of SHAKE256 XOF. Squeezes arbitrarily many bytes. Can be+ * called multiple times to keep squeezing.+ *+ * @param[out] out Pointer to output blocks.+ * @param outlen Number of bytes to be squeezed (written to output).+ * @param[in,out] state Pointer to input/output state.+ */+MLD_INTERNAL_API+void mld_shake256_squeeze(uint8_t *out, size_t outlen, mld_shake256ctx *state)+__contract__(+ requires(outlen <= 8 * SHAKE256_RATE /* somewhat arbitrary bound */)+ requires(memory_no_alias(state, sizeof(mld_shake256ctx)))+ requires(memory_no_alias(out, outlen))+ requires(state->pos <= SHAKE256_RATE)+ assigns(memory_slice(state, sizeof(mld_shake256ctx)))+ assigns(memory_slice(out, outlen))+ ensures(state->pos <= SHAKE256_RATE)+);++#define mld_shake256_release MLD_NAMESPACE(shake256_release)+/**+ * Release and securely zero the SHAKE256 state.+ *+ * @param[in,out] state Pointer to state.+ */+MLD_INTERNAL_API+void mld_shake256_release(mld_shake256ctx *state)+__contract__(+ requires(memory_no_alias(state, sizeof(mld_shake256ctx)))+ assigns(memory_slice(state, sizeof(mld_shake256ctx)))+);++#if !defined(MLD_CONFIG_NO_KEYPAIR_API) || !defined(MLD_CONFIG_CORE_API_ONLY)+#define mld_shake256 MLD_NAMESPACE(shake256)+/**+ * SHAKE256 XOF with non-incremental API.+ *+ * @param[out] out Pointer to output.+ * @param outlen Requested output length in bytes.+ * @param[in] in Pointer to input.+ * @param inlen Length of input in bytes.+ */+MLD_INTERNAL_API+void mld_shake256(uint8_t *out, size_t outlen, const uint8_t *in, size_t inlen)+__contract__(+ requires(inlen <= MLD_MAX_BUFFER_SIZE)+ requires(outlen <= 8 * SHAKE256_RATE /* somewhat arbitrary bound */)+ requires(memory_no_alias(in, inlen))+ requires(memory_no_alias(out, outlen))+ assigns(memory_slice(out, outlen))+);+#endif /* !MLD_CONFIG_NO_KEYPAIR_API || !MLD_CONFIG_CORE_API_ONLY */++#endif /* !MLD_FIPS202_FIPS202_H */
@@ -0,0 +1,187 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS204]+ * FIPS 204 Module-Lattice-Based Digital Signature Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/204/final+ */++#include "../common.h"+#if !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \+ !defined(MLD_CONFIG_SERIAL_FIPS202_ONLY)++#include "../ct.h"+#include "fips202.h"+#include "fips202x4.h"+#include "keccakf1600.h"++#if !defined(MLD_CONFIG_NO_KEYPAIR_API) || !defined(MLD_CONFIG_REDUCE_RAM) || \+ defined(MLD_UNIT_TEST)+static void mld_keccak_absorb_once_x4(uint64_t *s, unsigned r,+ const uint8_t *in0, const uint8_t *in1,+ const uint8_t *in2, const uint8_t *in3,+ size_t inlen, uint8_t p)+__contract__(+ requires(inlen <= MLD_MAX_BUFFER_SIZE)+ requires(memory_no_alias(s, sizeof(uint64_t) * MLD_KECCAK_LANES * MLD_KECCAK_WAY))+ requires(r > 0)+ requires(r <= sizeof(uint64_t) * MLD_KECCAK_LANES)+ requires(memory_no_alias(in0, inlen))+ requires(memory_no_alias(in1, inlen))+ requires(memory_no_alias(in2, inlen))+ requires(memory_no_alias(in3, inlen))+ assigns(memory_slice(s, sizeof(uint64_t) * MLD_KECCAK_LANES * MLD_KECCAK_WAY)))+{+ while (inlen >= r)+ __loop__(+ assigns(inlen, in0, in1, in2, in3, memory_slice(s, sizeof(uint64_t) * MLD_KECCAK_LANES * MLD_KECCAK_WAY))+ invariant(inlen <= loop_entry(inlen))+ invariant(in0 == loop_entry(in0) + (loop_entry(inlen) - inlen))+ invariant(in1 == loop_entry(in1) + (loop_entry(inlen) - inlen))+ invariant(in2 == loop_entry(in2) + (loop_entry(inlen) - inlen))+ invariant(in3 == loop_entry(in3) + (loop_entry(inlen) - inlen))+ decreases(inlen))+ {+ mld_keccakf1600x4_xor_bytes(s, in0, in1, in2, in3, 0, r);+ mld_keccakf1600x4_permute(s);++ in0 += r;+ in1 += r;+ in2 += r;+ in3 += r;+ inlen -= r;+ }++ /* Safety: At this point, inlen < r, so the truncations to unsigned are safe+ * below. */+ if (inlen > 0)+ {+ mld_keccakf1600x4_xor_bytes(s, in0, in1, in2, in3, 0, (unsigned)inlen);+ }++ if (inlen == r - 1)+ {+ p |= 128;+ mld_keccakf1600x4_xor_bytes(s, &p, &p, &p, &p, (unsigned)inlen, 1);+ }+ else+ {+ mld_keccakf1600x4_xor_bytes(s, &p, &p, &p, &p, (unsigned)inlen, 1);+ p = 128;+ mld_keccakf1600x4_xor_bytes(s, &p, &p, &p, &p, r - 1, 1);+ }+}++static void mld_keccak_squeezeblocks_x4(uint8_t *out0, uint8_t *out1,+ uint8_t *out2, uint8_t *out3,+ size_t nblocks, uint64_t *s, unsigned r)+__contract__(+ requires(r <= sizeof(uint64_t) * MLD_KECCAK_LANES)+ requires(nblocks <= 8 /* somewhat arbitrary bound */)+ requires(memory_no_alias(s, sizeof(uint64_t) * MLD_KECCAK_LANES * MLD_KECCAK_WAY))+ requires(memory_no_alias(out0, nblocks * r))+ requires(memory_no_alias(out1, nblocks * r))+ requires(memory_no_alias(out2, nblocks * r))+ requires(memory_no_alias(out3, nblocks * r))+ assigns(memory_slice(s, sizeof(uint64_t) * MLD_KECCAK_LANES * MLD_KECCAK_WAY))+ assigns(memory_slice(out0, nblocks * r))+ assigns(memory_slice(out1, nblocks * r))+ assigns(memory_slice(out2, nblocks * r))+ assigns(memory_slice(out3, nblocks * r)))+{+ size_t current_offset = 0;+ while (nblocks > 0)+ __loop__(+ assigns(nblocks, current_offset,+ memory_slice(s, sizeof(uint64_t) * MLD_KECCAK_LANES * MLD_KECCAK_WAY),+ memory_slice(out0, nblocks * r),+ memory_slice(out1, nblocks * r),+ memory_slice(out2, nblocks * r),+ memory_slice(out3, nblocks * r))+ invariant(nblocks <= loop_entry(nblocks))+ invariant(current_offset == (loop_entry(nblocks) - nblocks) * r)+ decreases(nblocks))+ {+ mld_keccakf1600x4_permute(s);+ mld_keccakf1600x4_extract_bytes(+ s, &out0[current_offset], &out1[current_offset], &out2[current_offset],+ &out3[current_offset], 0, r);+ current_offset += r;+ nblocks--;+ }+}+#endif /* !MLD_CONFIG_NO_KEYPAIR_API || !MLD_CONFIG_REDUCE_RAM || \+ MLD_UNIT_TEST */++#if !defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)+MLD_INTERNAL_API+void mld_shake128x4_absorb_once(mld_shake128x4ctx *state, const uint8_t *in0,+ const uint8_t *in1, const uint8_t *in2,+ const uint8_t *in3, size_t inlen)+{+ mld_memset(state, 0, sizeof(mld_shake128x4ctx));+ mld_keccak_absorb_once_x4(state->ctx, SHAKE128_RATE, in0, in1, in2, in3,+ inlen, 0x1F);+}++MLD_INTERNAL_API+void mld_shake128x4_squeezeblocks(uint8_t *out0, uint8_t *out1, uint8_t *out2,+ uint8_t *out3, size_t nblocks,+ mld_shake128x4ctx *state)+{+ mld_keccak_squeezeblocks_x4(out0, out1, out2, out3, nblocks, state->ctx,+ SHAKE128_RATE);+}++MLD_INTERNAL_API+void mld_shake128x4_init(mld_shake128x4ctx *state) { (void)state; }+MLD_INTERNAL_API+void mld_shake128x4_release(mld_shake128x4ctx *state)+{+ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ mld_zeroize(state, sizeof(mld_shake128x4ctx));+}+#endif /* !MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API) || \+ (!defined(MLD_CONFIG_NO_SIGN_API) && \+ (!defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)))+MLD_INTERNAL_API+void mld_shake256x4_absorb_once(mld_shake256x4ctx *state, const uint8_t *in0,+ const uint8_t *in1, const uint8_t *in2,+ const uint8_t *in3, size_t inlen)+{+ mld_memset(state, 0, sizeof(mld_shake256x4ctx));+ mld_keccak_absorb_once_x4(state->ctx, SHAKE256_RATE, in0, in1, in2, in3,+ inlen, 0x1F);+}++MLD_INTERNAL_API+void mld_shake256x4_squeezeblocks(uint8_t *out0, uint8_t *out1, uint8_t *out2,+ uint8_t *out3, size_t nblocks,+ mld_shake256x4ctx *state)+{+ mld_keccak_squeezeblocks_x4(out0, out1, out2, out3, nblocks, state->ctx,+ SHAKE256_RATE);+}++MLD_INTERNAL_API+void mld_shake256x4_init(mld_shake256x4ctx *state) { (void)state; }+MLD_INTERNAL_API+void mld_shake256x4_release(mld_shake256x4ctx *state)+{+ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ mld_zeroize(state, sizeof(mld_shake256x4ctx));+}+#endif /* !MLD_CONFIG_NO_KEYPAIR_API || (!MLD_CONFIG_NO_SIGN_API && \+ (!MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST)) */++#endif /* !MLD_CONFIG_MULTILEVEL_NO_SHARED && !MLD_CONFIG_SERIAL_FIPS202_ONLY \+ */
@@ -0,0 +1,125 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+#ifndef MLD_FIPS202_FIPS202X4_H+#define MLD_FIPS202_FIPS202X4_H++#include "../common.h"++#if !defined(MLD_CONFIG_SERIAL_FIPS202_ONLY)++#include <stddef.h>++#include "../cbmc.h"+#include "fips202.h"+#include "keccakf1600.h"++/** Context for the non-incremental 4-way SHAKE128 API. */+typedef struct+{+ uint64_t ctx[MLD_KECCAK_LANES *+ MLD_KECCAK_WAY]; /**< 4-way Keccak state, stored sequentially. */+} mld_shake128x4ctx;++/** Context for the 4-way batched SHAKE256 XOF. */+typedef struct+{+ uint64_t ctx[MLD_KECCAK_LANES *+ MLD_KECCAK_WAY]; /**< Interleaved 4-way Keccak state. */+} mld_shake256x4ctx;++#if !defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)+#define mld_shake128x4_absorb_once MLD_NAMESPACE(shake128x4_absorb_once)+MLD_INTERNAL_API+void mld_shake128x4_absorb_once(mld_shake128x4ctx *state, const uint8_t *in0,+ const uint8_t *in1, const uint8_t *in2,+ const uint8_t *in3, size_t inlen)+__contract__(+ requires(inlen <= MLD_MAX_BUFFER_SIZE)+ requires(memory_no_alias(state, sizeof(mld_shake128x4ctx)))+ requires(memory_no_alias(in0, inlen))+ requires(memory_no_alias(in1, inlen))+ requires(memory_no_alias(in2, inlen))+ requires(memory_no_alias(in3, inlen))+ assigns(memory_slice(state, sizeof(mld_shake128x4ctx)))+);++#define mld_shake128x4_squeezeblocks MLD_NAMESPACE(shake128x4_squeezeblocks)+MLD_INTERNAL_API+void mld_shake128x4_squeezeblocks(uint8_t *out0, uint8_t *out1, uint8_t *out2,+ uint8_t *out3, size_t nblocks,+ mld_shake128x4ctx *state)+__contract__(+ requires(nblocks <= 8 /* somewhat arbitrary bound */)+ requires(memory_no_alias(state, sizeof(mld_shake128x4ctx)))+ requires(memory_no_alias(out0, nblocks * SHAKE128_RATE))+ requires(memory_no_alias(out1, nblocks * SHAKE128_RATE))+ requires(memory_no_alias(out2, nblocks * SHAKE128_RATE))+ requires(memory_no_alias(out3, nblocks * SHAKE128_RATE))+ assigns(memory_slice(out0, nblocks * SHAKE128_RATE),+ memory_slice(out1, nblocks * SHAKE128_RATE),+ memory_slice(out2, nblocks * SHAKE128_RATE),+ memory_slice(out3, nblocks * SHAKE128_RATE),+ memory_slice(state, sizeof(mld_shake128x4ctx)))+);++#define mld_shake128x4_init MLD_NAMESPACE(shake128x4_init)+MLD_INTERNAL_API+void mld_shake128x4_init(mld_shake128x4ctx *state);++#define mld_shake128x4_release MLD_NAMESPACE(shake128x4_release)+MLD_INTERNAL_API+void mld_shake128x4_release(mld_shake128x4ctx *state);+#endif /* !MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API) || \+ (!defined(MLD_CONFIG_NO_SIGN_API) && \+ (!defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)))+#define mld_shake256x4_absorb_once MLD_NAMESPACE(shake256x4_absorb_once)+MLD_INTERNAL_API+void mld_shake256x4_absorb_once(mld_shake256x4ctx *state, const uint8_t *in0,+ const uint8_t *in1, const uint8_t *in2,+ const uint8_t *in3, size_t inlen)+__contract__(+ requires(inlen <= MLD_MAX_BUFFER_SIZE)+ requires(memory_no_alias(state, sizeof(mld_shake256x4ctx)))+ requires(memory_no_alias(in0, inlen))+ requires(memory_no_alias(in1, inlen))+ requires(memory_no_alias(in2, inlen))+ requires(memory_no_alias(in3, inlen))+ assigns(memory_slice(state, sizeof(mld_shake256x4ctx)))+);++#define mld_shake256x4_squeezeblocks MLD_NAMESPACE(shake256x4_squeezeblocks)+MLD_INTERNAL_API+void mld_shake256x4_squeezeblocks(uint8_t *out0, uint8_t *out1, uint8_t *out2,+ uint8_t *out3, size_t nblocks,+ mld_shake256x4ctx *state)+__contract__(+ requires(nblocks <= 8 /* somewhat arbitrary bound */)+ requires(memory_no_alias(state, sizeof(mld_shake256x4ctx)))+ requires(memory_no_alias(out0, nblocks * SHAKE256_RATE))+ requires(memory_no_alias(out1, nblocks * SHAKE256_RATE))+ requires(memory_no_alias(out2, nblocks * SHAKE256_RATE))+ requires(memory_no_alias(out3, nblocks * SHAKE256_RATE))+ assigns(memory_slice(out0, nblocks * SHAKE256_RATE),+ memory_slice(out1, nblocks * SHAKE256_RATE),+ memory_slice(out2, nblocks * SHAKE256_RATE),+ memory_slice(out3, nblocks * SHAKE256_RATE),+ memory_slice(state, sizeof(mld_shake256x4ctx)))+);++#define mld_shake256x4_init MLD_NAMESPACE(shake256x4_init)+MLD_INTERNAL_API+void mld_shake256x4_init(mld_shake256x4ctx *state);++#define mld_shake256x4_release MLD_NAMESPACE(shake256x4_release)+MLD_INTERNAL_API+void mld_shake256x4_release(mld_shake256x4ctx *state);+#endif /* !MLD_CONFIG_NO_KEYPAIR_API || (!MLD_CONFIG_NO_SIGN_API && \+ (!MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST)) */++#endif /* !MLD_CONFIG_SERIAL_FIPS202_ONLY */+#endif /* !MLD_FIPS202_FIPS202X4_H */
@@ -0,0 +1,510 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [mupq]+ * Common files for pqm4, pqm3, pqriscv+ * Kannwischer, Petri, Rijneveld, Schwabe, Stoffelen+ * https://github.com/mupq/mupq+ *+ * - [supercop]+ * SUPERCOP benchmarking framework+ * Daniel J. Bernstein+ * http://bench.cr.yp.to/supercop.html+ *+ * - [tweetfips]+ * 'tweetfips202' FIPS202 implementation+ * Van Assche, Bernstein, Schwabe+ * https://keccak.team/2015/tweetfips202.html+ */++/* Based on the CC0 implementation from @[mupq] and the public domain+ * implementation @[supercop, crypto_hash/keccakc512/simple/]+ * by Ronny Van Keer, and the public domain @[tweetfips] implementation. */++#include "keccakf1600.h"+#if !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++#define MLD_KECCAK_NROUNDS 24+#define MLD_KECCAK_ROL(a, offset) (((a) << (offset)) ^ ((a) >> (64 - (offset))))++MLD_INTERNAL_API+void mld_keccakf1600_extract_bytes(uint64_t *state, unsigned char *data,+ unsigned offset, unsigned length)+{+ unsigned i;+#if defined(MLD_SYS_LITTLE_ENDIAN)+ uint8_t *state_ptr = (uint8_t *)state + offset;+ for (i = 0; i < length; i++)+ __loop__(invariant(i <= length)+ decreases(length - i))+ {+ data[i] = state_ptr[i];+ }+#else /* MLD_SYS_LITTLE_ENDIAN */+ /* Portable version */+ for (i = 0; i < length; i++)+ __loop__(invariant(i <= length)+ decreases(length - i))+ {+ data[i] = (state[(offset + i) >> 3] >> (8 * ((offset + i) & 0x07))) & 0xFF;+ }+#endif /* !MLD_SYS_LITTLE_ENDIAN */+}++MLD_INTERNAL_API+void mld_keccakf1600_xor_bytes(uint64_t *state, const unsigned char *data,+ unsigned offset, unsigned length)+{+ unsigned i;+#if defined(MLD_SYS_LITTLE_ENDIAN)+ uint8_t *state_ptr = (uint8_t *)state + offset;+ for (i = 0; i < length; i++)+ __loop__(invariant(i <= length)+ decreases(length - i))+ {+ state_ptr[i] ^= data[i];+ }+#else /* MLD_SYS_LITTLE_ENDIAN */+ /* Portable version */+ for (i = 0; i < length; i++)+ __loop__(invariant(i <= length)+ decreases(length - i))+ {+ state[(offset + i) >> 3] ^= (uint64_t)data[i]+ << (8 * ((offset + i) & 0x07));+ }+#endif /* !MLD_SYS_LITTLE_ENDIAN */+}++#if (!defined(MLD_CONFIG_NO_KEYPAIR_API) || !defined(MLD_CONFIG_REDUCE_RAM) || \+ defined(MLD_UNIT_TEST)) && \+ !defined(MLD_CONFIG_SERIAL_FIPS202_ONLY)+static void mld_keccakf1600x4_extract_bytes_c(uint64_t *state,+ unsigned char *data0,+ unsigned char *data1,+ unsigned char *data2,+ unsigned char *data3,+ unsigned offset, unsigned length)+__contract__(+ requires(0 <= offset && offset <= MLD_KECCAK_LANES * sizeof(uint64_t) &&+ 0 <= length && length <= MLD_KECCAK_LANES * sizeof(uint64_t) - offset)+ requires(memory_no_alias(state, sizeof(uint64_t) * MLD_KECCAK_LANES * MLD_KECCAK_WAY))+ requires(memory_no_alias(data0, length))+ requires(memory_no_alias(data1, length))+ requires(memory_no_alias(data2, length))+ requires(memory_no_alias(data3, length))+ assigns(memory_slice(data0, length))+ assigns(memory_slice(data1, length))+ assigns(memory_slice(data2, length))+ assigns(memory_slice(data3, length))+)+{+ mld_keccakf1600_extract_bytes(state + MLD_KECCAK_LANES * 0, data0, offset,+ length);+ mld_keccakf1600_extract_bytes(state + MLD_KECCAK_LANES * 1, data1, offset,+ length);+ mld_keccakf1600_extract_bytes(state + MLD_KECCAK_LANES * 2, data2, offset,+ length);+ mld_keccakf1600_extract_bytes(state + MLD_KECCAK_LANES * 3, data3, offset,+ length);+}++MLD_INTERNAL_API+void mld_keccakf1600x4_extract_bytes(uint64_t *state, unsigned char *data0,+ unsigned char *data1, unsigned char *data2,+ unsigned char *data3, unsigned offset,+ unsigned length)+{+#if defined(MLD_USE_NATIVE_FIPS202_X4_EXTRACT_BYTES)+ if (mld_keccakf1600_extract_bytes_x4_native(state, data0, data1, data2, data3,+ offset, length) ==+ MLD_NATIVE_FUNC_SUCCESS)+ {+ return;+ }+#endif /* MLD_USE_NATIVE_FIPS202_X4_EXTRACT_BYTES */+ mld_keccakf1600x4_extract_bytes_c(state, data0, data1, data2, data3, offset,+ length);+}++static void mld_keccakf1600x4_xor_bytes_c(uint64_t *state,+ const unsigned char *data0,+ const unsigned char *data1,+ const unsigned char *data2,+ const unsigned char *data3,+ unsigned offset, unsigned length)+__contract__(+ requires(0 <= offset && offset <= MLD_KECCAK_LANES * sizeof(uint64_t) &&+ 0 <= length && length <= MLD_KECCAK_LANES * sizeof(uint64_t) - offset)+ requires(memory_no_alias(state, sizeof(uint64_t) * MLD_KECCAK_LANES * MLD_KECCAK_WAY))+ requires(memory_no_alias(data0, length))+ /* Case 1: all input buffers are distinct; Case 2: All input buffers are the same */+ requires((data0 == data1 &&+ data0 == data2 &&+ data0 == data3) ||+ (memory_no_alias(data1, length) &&+ memory_no_alias(data2, length) &&+ memory_no_alias(data3, length)))+ assigns(memory_slice(state, sizeof(uint64_t) * MLD_KECCAK_LANES * MLD_KECCAK_WAY))+)+{+ mld_keccakf1600_xor_bytes(state + MLD_KECCAK_LANES * 0, data0, offset,+ length);+ mld_keccakf1600_xor_bytes(state + MLD_KECCAK_LANES * 1, data1, offset,+ length);+ mld_keccakf1600_xor_bytes(state + MLD_KECCAK_LANES * 2, data2, offset,+ length);+ mld_keccakf1600_xor_bytes(state + MLD_KECCAK_LANES * 3, data3, offset,+ length);+}++MLD_INTERNAL_API+void mld_keccakf1600x4_xor_bytes(uint64_t *state, const unsigned char *data0,+ const unsigned char *data1,+ const unsigned char *data2,+ const unsigned char *data3, unsigned offset,+ unsigned length)+{+#if defined(MLD_USE_NATIVE_FIPS202_X4_XOR_BYTES)+ if (mld_keccakf1600_xor_bytes_x4_native(state, data0, data1, data2, data3,+ offset,+ length) == MLD_NATIVE_FUNC_SUCCESS)+ {+ return;+ }+#endif /* MLD_USE_NATIVE_FIPS202_X4_XOR_BYTES */+ mld_keccakf1600x4_xor_bytes_c(state, data0, data1, data2, data3, offset,+ length);+}++MLD_INTERNAL_API+void mld_keccakf1600x4_permute(uint64_t *state)+{+#if defined(MLD_USE_NATIVE_FIPS202_X4)+ if (mld_keccak_f1600_x4_native(state) == MLD_NATIVE_FUNC_SUCCESS)+ {+ return;+ }+#endif /* MLD_USE_NATIVE_FIPS202_X4 */+ mld_keccakf1600_permute(state + MLD_KECCAK_LANES * 0);+ mld_keccakf1600_permute(state + MLD_KECCAK_LANES * 1);+ mld_keccakf1600_permute(state + MLD_KECCAK_LANES * 2);+ mld_keccakf1600_permute(state + MLD_KECCAK_LANES * 3);+}+#endif /* (!MLD_CONFIG_NO_KEYPAIR_API || !MLD_CONFIG_REDUCE_RAM || \+ MLD_UNIT_TEST) && !MLD_CONFIG_SERIAL_FIPS202_ONLY */++static const uint64_t mld_KeccakF_RoundConstants[MLD_KECCAK_NROUNDS] = {+ (uint64_t)0x0000000000000001ULL, (uint64_t)0x0000000000008082ULL,+ (uint64_t)0x800000000000808aULL, (uint64_t)0x8000000080008000ULL,+ (uint64_t)0x000000000000808bULL, (uint64_t)0x0000000080000001ULL,+ (uint64_t)0x8000000080008081ULL, (uint64_t)0x8000000000008009ULL,+ (uint64_t)0x000000000000008aULL, (uint64_t)0x0000000000000088ULL,+ (uint64_t)0x0000000080008009ULL, (uint64_t)0x000000008000000aULL,+ (uint64_t)0x000000008000808bULL, (uint64_t)0x800000000000008bULL,+ (uint64_t)0x8000000000008089ULL, (uint64_t)0x8000000000008003ULL,+ (uint64_t)0x8000000000008002ULL, (uint64_t)0x8000000000000080ULL,+ (uint64_t)0x000000000000800aULL, (uint64_t)0x800000008000000aULL,+ (uint64_t)0x8000000080008081ULL, (uint64_t)0x8000000000008080ULL,+ (uint64_t)0x0000000080000001ULL, (uint64_t)0x8000000080008008ULL};++MLD_STATIC_TESTABLE+void mld_keccakf1600_permute_c(uint64_t *state)+__contract__(+ requires(memory_no_alias(state, sizeof(uint64_t) * MLD_KECCAK_LANES))+ assigns(memory_slice(state, sizeof(uint64_t) * MLD_KECCAK_LANES))+)+{+ unsigned round;++ uint64_t Aba, Abe, Abi, Abo, Abu;+ uint64_t Aga, Age, Agi, Ago, Agu;+ uint64_t Aka, Ake, Aki, Ako, Aku;+ uint64_t Ama, Ame, Ami, Amo, Amu;+ uint64_t Asa, Ase, Asi, Aso, Asu;+ uint64_t BCa, BCe, BCi, BCo, BCu;+ uint64_t Da, De, Di, Do, Du;+ uint64_t Eba, Ebe, Ebi, Ebo, Ebu;+ uint64_t Ega, Ege, Egi, Ego, Egu;+ uint64_t Eka, Eke, Eki, Eko, Eku;+ uint64_t Ema, Eme, Emi, Emo, Emu;+ uint64_t Esa, Ese, Esi, Eso, Esu;++ /* copyFromState(A, state) */+ Aba = state[0];+ Abe = state[1];+ Abi = state[2];+ Abo = state[3];+ Abu = state[4];+ Aga = state[5];+ Age = state[6];+ Agi = state[7];+ Ago = state[8];+ Agu = state[9];+ Aka = state[10];+ Ake = state[11];+ Aki = state[12];+ Ako = state[13];+ Aku = state[14];+ Ama = state[15];+ Ame = state[16];+ Ami = state[17];+ Amo = state[18];+ Amu = state[19];+ Asa = state[20];+ Ase = state[21];+ Asi = state[22];+ Aso = state[23];+ Asu = state[24];++ for (round = 0; round < MLD_KECCAK_NROUNDS; round += 2)+ __loop__(invariant(round <= MLD_KECCAK_NROUNDS && round % 2 == 0)+ decreases(MLD_KECCAK_NROUNDS - round))+ {+ /* prepareTheta */+ BCa = Aba ^ Aga ^ Aka ^ Ama ^ Asa;+ BCe = Abe ^ Age ^ Ake ^ Ame ^ Ase;+ BCi = Abi ^ Agi ^ Aki ^ Ami ^ Asi;+ BCo = Abo ^ Ago ^ Ako ^ Amo ^ Aso;+ BCu = Abu ^ Agu ^ Aku ^ Amu ^ Asu;++ /* thetaRhoPiChiIotaPrepareTheta(round, A, E) */+ Da = BCu ^ MLD_KECCAK_ROL(BCe, 1);+ De = BCa ^ MLD_KECCAK_ROL(BCi, 1);+ Di = BCe ^ MLD_KECCAK_ROL(BCo, 1);+ Do = BCi ^ MLD_KECCAK_ROL(BCu, 1);+ Du = BCo ^ MLD_KECCAK_ROL(BCa, 1);++ Aba ^= Da;+ BCa = Aba;+ Age ^= De;+ BCe = MLD_KECCAK_ROL(Age, 44);+ Aki ^= Di;+ BCi = MLD_KECCAK_ROL(Aki, 43);+ Amo ^= Do;+ BCo = MLD_KECCAK_ROL(Amo, 21);+ Asu ^= Du;+ BCu = MLD_KECCAK_ROL(Asu, 14);+ Eba = BCa ^ ((~BCe) & BCi);+ Eba ^= (uint64_t)mld_KeccakF_RoundConstants[round];+ Ebe = BCe ^ ((~BCi) & BCo);+ Ebi = BCi ^ ((~BCo) & BCu);+ Ebo = BCo ^ ((~BCu) & BCa);+ Ebu = BCu ^ ((~BCa) & BCe);++ Abo ^= Do;+ BCa = MLD_KECCAK_ROL(Abo, 28);+ Agu ^= Du;+ BCe = MLD_KECCAK_ROL(Agu, 20);+ Aka ^= Da;+ BCi = MLD_KECCAK_ROL(Aka, 3);+ Ame ^= De;+ BCo = MLD_KECCAK_ROL(Ame, 45);+ Asi ^= Di;+ BCu = MLD_KECCAK_ROL(Asi, 61);+ Ega = BCa ^ ((~BCe) & BCi);+ Ege = BCe ^ ((~BCi) & BCo);+ Egi = BCi ^ ((~BCo) & BCu);+ Ego = BCo ^ ((~BCu) & BCa);+ Egu = BCu ^ ((~BCa) & BCe);++ Abe ^= De;+ BCa = MLD_KECCAK_ROL(Abe, 1);+ Agi ^= Di;+ BCe = MLD_KECCAK_ROL(Agi, 6);+ Ako ^= Do;+ BCi = MLD_KECCAK_ROL(Ako, 25);+ Amu ^= Du;+ BCo = MLD_KECCAK_ROL(Amu, 8);+ Asa ^= Da;+ BCu = MLD_KECCAK_ROL(Asa, 18);+ Eka = BCa ^ ((~BCe) & BCi);+ Eke = BCe ^ ((~BCi) & BCo);+ Eki = BCi ^ ((~BCo) & BCu);+ Eko = BCo ^ ((~BCu) & BCa);+ Eku = BCu ^ ((~BCa) & BCe);++ Abu ^= Du;+ BCa = MLD_KECCAK_ROL(Abu, 27);+ Aga ^= Da;+ BCe = MLD_KECCAK_ROL(Aga, 36);+ Ake ^= De;+ BCi = MLD_KECCAK_ROL(Ake, 10);+ Ami ^= Di;+ BCo = MLD_KECCAK_ROL(Ami, 15);+ Aso ^= Do;+ BCu = MLD_KECCAK_ROL(Aso, 56);+ Ema = BCa ^ ((~BCe) & BCi);+ Eme = BCe ^ ((~BCi) & BCo);+ Emi = BCi ^ ((~BCo) & BCu);+ Emo = BCo ^ ((~BCu) & BCa);+ Emu = BCu ^ ((~BCa) & BCe);++ Abi ^= Di;+ BCa = MLD_KECCAK_ROL(Abi, 62);+ Ago ^= Do;+ BCe = MLD_KECCAK_ROL(Ago, 55);+ Aku ^= Du;+ BCi = MLD_KECCAK_ROL(Aku, 39);+ Ama ^= Da;+ BCo = MLD_KECCAK_ROL(Ama, 41);+ Ase ^= De;+ BCu = MLD_KECCAK_ROL(Ase, 2);+ Esa = BCa ^ ((~BCe) & BCi);+ Ese = BCe ^ ((~BCi) & BCo);+ Esi = BCi ^ ((~BCo) & BCu);+ Eso = BCo ^ ((~BCu) & BCa);+ Esu = BCu ^ ((~BCa) & BCe);++ /* prepareTheta */+ BCa = Eba ^ Ega ^ Eka ^ Ema ^ Esa;+ BCe = Ebe ^ Ege ^ Eke ^ Eme ^ Ese;+ BCi = Ebi ^ Egi ^ Eki ^ Emi ^ Esi;+ BCo = Ebo ^ Ego ^ Eko ^ Emo ^ Eso;+ BCu = Ebu ^ Egu ^ Eku ^ Emu ^ Esu;++ /* thetaRhoPiChiIotaPrepareTheta(round+1, E, A) */+ Da = BCu ^ MLD_KECCAK_ROL(BCe, 1);+ De = BCa ^ MLD_KECCAK_ROL(BCi, 1);+ Di = BCe ^ MLD_KECCAK_ROL(BCo, 1);+ Do = BCi ^ MLD_KECCAK_ROL(BCu, 1);+ Du = BCo ^ MLD_KECCAK_ROL(BCa, 1);++ Eba ^= Da;+ BCa = Eba;+ Ege ^= De;+ BCe = MLD_KECCAK_ROL(Ege, 44);+ Eki ^= Di;+ BCi = MLD_KECCAK_ROL(Eki, 43);+ Emo ^= Do;+ BCo = MLD_KECCAK_ROL(Emo, 21);+ Esu ^= Du;+ BCu = MLD_KECCAK_ROL(Esu, 14);+ Aba = BCa ^ ((~BCe) & BCi);+ Aba ^= (uint64_t)mld_KeccakF_RoundConstants[round + 1];+ Abe = BCe ^ ((~BCi) & BCo);+ Abi = BCi ^ ((~BCo) & BCu);+ Abo = BCo ^ ((~BCu) & BCa);+ Abu = BCu ^ ((~BCa) & BCe);++ Ebo ^= Do;+ BCa = MLD_KECCAK_ROL(Ebo, 28);+ Egu ^= Du;+ BCe = MLD_KECCAK_ROL(Egu, 20);+ Eka ^= Da;+ BCi = MLD_KECCAK_ROL(Eka, 3);+ Eme ^= De;+ BCo = MLD_KECCAK_ROL(Eme, 45);+ Esi ^= Di;+ BCu = MLD_KECCAK_ROL(Esi, 61);+ Aga = BCa ^ ((~BCe) & BCi);+ Age = BCe ^ ((~BCi) & BCo);+ Agi = BCi ^ ((~BCo) & BCu);+ Ago = BCo ^ ((~BCu) & BCa);+ Agu = BCu ^ ((~BCa) & BCe);++ Ebe ^= De;+ BCa = MLD_KECCAK_ROL(Ebe, 1);+ Egi ^= Di;+ BCe = MLD_KECCAK_ROL(Egi, 6);+ Eko ^= Do;+ BCi = MLD_KECCAK_ROL(Eko, 25);+ Emu ^= Du;+ BCo = MLD_KECCAK_ROL(Emu, 8);+ Esa ^= Da;+ BCu = MLD_KECCAK_ROL(Esa, 18);+ Aka = BCa ^ ((~BCe) & BCi);+ Ake = BCe ^ ((~BCi) & BCo);+ Aki = BCi ^ ((~BCo) & BCu);+ Ako = BCo ^ ((~BCu) & BCa);+ Aku = BCu ^ ((~BCa) & BCe);++ Ebu ^= Du;+ BCa = MLD_KECCAK_ROL(Ebu, 27);+ Ega ^= Da;+ BCe = MLD_KECCAK_ROL(Ega, 36);+ Eke ^= De;+ BCi = MLD_KECCAK_ROL(Eke, 10);+ Emi ^= Di;+ BCo = MLD_KECCAK_ROL(Emi, 15);+ Eso ^= Do;+ BCu = MLD_KECCAK_ROL(Eso, 56);+ Ama = BCa ^ ((~BCe) & BCi);+ Ame = BCe ^ ((~BCi) & BCo);+ Ami = BCi ^ ((~BCo) & BCu);+ Amo = BCo ^ ((~BCu) & BCa);+ Amu = BCu ^ ((~BCa) & BCe);++ Ebi ^= Di;+ BCa = MLD_KECCAK_ROL(Ebi, 62);+ Ego ^= Do;+ BCe = MLD_KECCAK_ROL(Ego, 55);+ Eku ^= Du;+ BCi = MLD_KECCAK_ROL(Eku, 39);+ Ema ^= Da;+ BCo = MLD_KECCAK_ROL(Ema, 41);+ Ese ^= De;+ BCu = MLD_KECCAK_ROL(Ese, 2);+ Asa = BCa ^ ((~BCe) & BCi);+ Ase = BCe ^ ((~BCi) & BCo);+ Asi = BCi ^ ((~BCo) & BCu);+ Aso = BCo ^ ((~BCu) & BCa);+ Asu = BCu ^ ((~BCa) & BCe);+ }++ /* copyToState(state, A) */+ state[0] = Aba;+ state[1] = Abe;+ state[2] = Abi;+ state[3] = Abo;+ state[4] = Abu;+ state[5] = Aga;+ state[6] = Age;+ state[7] = Agi;+ state[8] = Ago;+ state[9] = Agu;+ state[10] = Aka;+ state[11] = Ake;+ state[12] = Aki;+ state[13] = Ako;+ state[14] = Aku;+ state[15] = Ama;+ state[16] = Ame;+ state[17] = Ami;+ state[18] = Amo;+ state[19] = Amu;+ state[20] = Asa;+ state[21] = Ase;+ state[22] = Asi;+ state[23] = Aso;+ state[24] = Asu;+}++MLD_INTERNAL_API+void mld_keccakf1600_permute(uint64_t *state)+{+#if defined(MLD_USE_NATIVE_FIPS202_X1)+ if (mld_keccak_f1600_x1_native(state) == MLD_NATIVE_FUNC_SUCCESS)+ {+ return;+ }+#endif /* MLD_USE_NATIVE_FIPS202_X1 */+ mld_keccakf1600_permute_c(state);+}++#else /* !MLD_CONFIG_MULTILEVEL_NO_SHARED */++MLD_EMPTY_CU(keccakf1600)++#endif /* MLD_CONFIG_MULTILEVEL_NO_SHARED */++/* To facilitate single-compilation-unit (SCU) builds, undefine all macros.+ * Don't modify by hand -- this is auto-generated by scripts/autogen. */+#undef MLD_KECCAK_NROUNDS+#undef MLD_KECCAK_ROL
@@ -0,0 +1,110 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+#ifndef MLD_FIPS202_KECCAKF1600_H+#define MLD_FIPS202_KECCAKF1600_H+#include "../cbmc.h"+#include "../common.h"+#include "fips202.h"++#define MLD_KECCAK_LANES 25+#define MLD_KECCAK_WAY 4++/*+ * WARNING:+ * The contents of this structure, including the placement+ * and interleaving of Keccak lanes, are IMPLEMENTATION-DEFINED.+ * The struct is only exposed here to allow its construction on the stack.+ */++#define mld_keccakf1600_extract_bytes MLD_NAMESPACE(keccakf1600_extract_bytes)+MLD_INTERNAL_API+void mld_keccakf1600_extract_bytes(uint64_t *state, unsigned char *data,+ unsigned offset, unsigned length)+__contract__(+ requires(0 <= offset && offset <= MLD_KECCAK_LANES * sizeof(uint64_t) &&+ 0 <= length && length <= MLD_KECCAK_LANES * sizeof(uint64_t) - offset)+ requires(memory_no_alias(state, sizeof(uint64_t) * MLD_KECCAK_LANES))+ requires(memory_no_alias(data, length))+ assigns(memory_slice(data, length))+);++#define mld_keccakf1600_xor_bytes MLD_NAMESPACE(keccakf1600_xor_bytes)+MLD_INTERNAL_API+void mld_keccakf1600_xor_bytes(uint64_t *state, const unsigned char *data,+ unsigned offset, unsigned length)+__contract__(+ requires(0 <= offset && offset <= MLD_KECCAK_LANES * sizeof(uint64_t) &&+ 0 <= length && length <= MLD_KECCAK_LANES * sizeof(uint64_t) - offset)+ requires(memory_no_alias(state, sizeof(uint64_t) * MLD_KECCAK_LANES))+ requires(memory_no_alias(data, length))+ assigns(memory_slice(state, sizeof(uint64_t) * MLD_KECCAK_LANES))+);++#if (!defined(MLD_CONFIG_NO_KEYPAIR_API) || !defined(MLD_CONFIG_REDUCE_RAM) || \+ defined(MLD_UNIT_TEST)) && \+ !defined(MLD_CONFIG_SERIAL_FIPS202_ONLY)+#define mld_keccakf1600x4_extract_bytes \+ MLD_NAMESPACE(keccakf1600x4_extract_bytes)+MLD_INTERNAL_API+void mld_keccakf1600x4_extract_bytes(uint64_t *state, unsigned char *data0,+ unsigned char *data1, unsigned char *data2,+ unsigned char *data3, unsigned offset,+ unsigned length)+__contract__(+ requires(0 <= offset && offset <= MLD_KECCAK_LANES * sizeof(uint64_t) &&+ 0 <= length && length <= MLD_KECCAK_LANES * sizeof(uint64_t) - offset)+ requires(memory_no_alias(state, sizeof(uint64_t) * MLD_KECCAK_LANES * MLD_KECCAK_WAY))+ requires(memory_no_alias(data0, length))+ requires(memory_no_alias(data1, length))+ requires(memory_no_alias(data2, length))+ requires(memory_no_alias(data3, length))+ assigns(memory_slice(data0, length))+ assigns(memory_slice(data1, length))+ assigns(memory_slice(data2, length))+ assigns(memory_slice(data3, length))+);++#define mld_keccakf1600x4_xor_bytes MLD_NAMESPACE(keccakf1600x4_xor_bytes)+MLD_INTERNAL_API+void mld_keccakf1600x4_xor_bytes(uint64_t *state, const unsigned char *data0,+ const unsigned char *data1,+ const unsigned char *data2,+ const unsigned char *data3, unsigned offset,+ unsigned length)+__contract__(+ requires(0 <= offset && offset <= MLD_KECCAK_LANES * sizeof(uint64_t) &&+ 0 <= length && length <= MLD_KECCAK_LANES * sizeof(uint64_t) - offset)+ requires(memory_no_alias(state, sizeof(uint64_t) * MLD_KECCAK_LANES * MLD_KECCAK_WAY))+ requires(memory_no_alias(data0, length))+ /* Case 1: all input buffers are distinct; Case 2: All input buffers are the same */+ requires((data0 == data1 &&+ data0 == data2 &&+ data0 == data3) ||+ (memory_no_alias(data1, length) &&+ memory_no_alias(data2, length) &&+ memory_no_alias(data3, length)))+ assigns(memory_slice(state, sizeof(uint64_t) * MLD_KECCAK_LANES * MLD_KECCAK_WAY))+);++#define mld_keccakf1600x4_permute MLD_NAMESPACE(keccakf1600x4_permute)+MLD_INTERNAL_API+void mld_keccakf1600x4_permute(uint64_t *state)+__contract__(+ requires(memory_no_alias(state, sizeof(uint64_t) * MLD_KECCAK_LANES * MLD_KECCAK_WAY))+ assigns(memory_slice(state, sizeof(uint64_t) * MLD_KECCAK_LANES * MLD_KECCAK_WAY))+);+#endif /* (!MLD_CONFIG_NO_KEYPAIR_API || !MLD_CONFIG_REDUCE_RAM || \+ MLD_UNIT_TEST) && !MLD_CONFIG_SERIAL_FIPS202_ONLY */++#define mld_keccakf1600_permute MLD_NAMESPACE(keccakf1600_permute)+MLD_INTERNAL_API+void mld_keccakf1600_permute(uint64_t *state)+__contract__(+ requires(memory_no_alias(state, sizeof(uint64_t) * MLD_KECCAK_LANES))+ assigns(memory_slice(state, sizeof(uint64_t) * MLD_KECCAK_LANES))+);++#endif /* !MLD_FIPS202_KECCAKF1600_H */
@@ -0,0 +1,85 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [HYBRID]+ * Hybrid scalar/vector implementations of Keccak and SPHINCS+ on AArch64+ * Becker, Kannwischer+ * https://eprint.iacr.org/2022/1243+ */++#ifndef MLD_FIPS202_NATIVE_AARCH64_AUTO_H+#define MLD_FIPS202_NATIVE_AARCH64_AUTO_H+/* Default FIPS202 assembly profile for AArch64 systems */++/*+ * Default logic to decide which implementation to use.+ *+ */++/*+ * Keccak-f1600+ *+ * - On Arm-based Apple CPUs, or CPUs with MLD_SYS_AARCH64_FAST_SHA3 set,+ * we pick a pure Neon implementation.+ * - Otherwise, unless MLD_SYS_AARCH64_SLOW_BARREL_SHIFTER is set,+ * we use lazy-rotation scalar assembly from @[HYBRID].+ * - Otherwise, if MLD_SYS_AARCH64_SLOW_BARREL_SHIFTER is set, we+ * fall back to the standard C implementation.+ */+#if defined(__ARM_FEATURE_SHA3) && \+ (defined(__APPLE__) || defined(MLD_SYS_AARCH64_FAST_SHA3))+#include "x1_v84a.h"+#elif !defined(MLD_SYS_AARCH64_SLOW_BARREL_SHIFTER)+#include "x1_scalar.h"+#endif++#if (!defined(MLD_CONFIG_NO_KEYPAIR_API) || \+ !defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_REDUCE_RAM)) && \+ !defined(MLD_CONFIG_SERIAL_FIPS202_ONLY)+/* Batched, SIMD-based Keccak-f1600 implementations. */+#if defined(MLD_SYS_AARCH64_NEON)++/*+ * Keccak-f1600x2/x4+ *+ * The optimal implementation is highly CPU-specific; see @[HYBRID].+ *+ * For now, if v8.4-A is not implemented, we fall back to Keccak-f1600.+ * If v8.4-A is implemented and we are on an Apple CPU (or a CPU with+ * MLD_SYS_AARCH64_FAST_SHA3 set), we use a plain Neon-based implementation.+ * If v8.4-A is implemented and we are on neither, we use a+ * scalar/Neon/Neon hybrid.+ * The reason for this distinction is that Apple CPUs (and other CPUs flagged+ * via MLD_SYS_AARCH64_FAST_SHA3) appear to implement the SHA3 instructions on+ * all SIMD units, while Arm CPUs prior to Cortex-X4 don't, and ordinary Neon+ * instructions are still needed.+ */+#if defined(__ARM_FEATURE_SHA3)+/*+ * For Apple-M cores (and other cores flagged via MLD_SYS_AARCH64_FAST_SHA3),+ * we use a plain implementation leveraging SHA3 instructions only.+ */+#if defined(__APPLE__) || defined(MLD_SYS_AARCH64_FAST_SHA3)+#include "x2_v84a.h"+#else+#include "x4_v8a_v84a_scalar.h"+#endif++#else /* __ARM_FEATURE_SHA3 */++#include "x4_v8a_scalar.h"++#endif /* !__ARM_FEATURE_SHA3 */++#endif /* MLD_SYS_AARCH64_NEON */++#endif /* (!MLD_CONFIG_NO_KEYPAIR_API || !MLD_CONFIG_NO_SIGN_API || \+ !MLD_CONFIG_REDUCE_RAM) && !MLD_CONFIG_SERIAL_FIPS202_ONLY */++#endif /* !MLD_FIPS202_NATIVE_AARCH64_AUTO_H */
@@ -0,0 +1,69 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+#ifndef MLD_FIPS202_NATIVE_AARCH64_SRC_FIPS202_NATIVE_AARCH64_H+#define MLD_FIPS202_NATIVE_AARCH64_SRC_FIPS202_NATIVE_AARCH64_H+++#include "../../../../cbmc.h"+#include "../../../../common.h"+++#define mld_keccakf1600_round_constants \+ MLD_NAMESPACE(keccakf1600_round_constants)+MLD_INTERNAL_DATA_DECLARATION const uint64_t+ mld_keccakf1600_round_constants[24];++#define mld_keccak_f1600_x1_scalar_aarch64_asm \+ MLD_NAMESPACE(keccak_f1600_x1_scalar_aarch64_asm)+void mld_keccak_f1600_x1_scalar_aarch64_asm(uint64_t state[25],+ const uint64_t rc[24])+__contract__(+ requires(memory_no_alias(state, sizeof(uint64_t) * 25 * 1))+ requires(rc == mld_keccakf1600_round_constants)+ assigns(memory_slice(state, sizeof(uint64_t) * 25 * 1))+);++#define mld_keccak_f1600_x1_v84a_aarch64_asm \+ MLD_NAMESPACE(keccak_f1600_x1_v84a_aarch64_asm)+void mld_keccak_f1600_x1_v84a_aarch64_asm(uint64_t state[25],+ const uint64_t rc[24])+__contract__(+ requires(memory_no_alias(state, sizeof(uint64_t) * 25 * 1))+ requires(rc == mld_keccakf1600_round_constants)+ assigns(memory_slice(state, sizeof(uint64_t) * 25 * 1))+);++#define mld_keccak_f1600_x2_v84a_aarch64_asm \+ MLD_NAMESPACE(keccak_f1600_x2_v84a_aarch64_asm)+void mld_keccak_f1600_x2_v84a_aarch64_asm(uint64_t state[50],+ const uint64_t rc[24])+__contract__(+ requires(memory_no_alias(state, sizeof(uint64_t) * 25 * 2))+ requires(rc == mld_keccakf1600_round_constants)+ assigns(memory_slice(state, sizeof(uint64_t) * 25 * 2))+);++#define mld_keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm \+ MLD_NAMESPACE(keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm)+void mld_keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm(uint64_t state[100],+ const uint64_t rc[24])+__contract__(+ requires(memory_no_alias(state, sizeof(uint64_t) * 25 * 4))+ requires(rc == mld_keccakf1600_round_constants)+ assigns(memory_slice(state, sizeof(uint64_t) * 25 * 4))+);++#define mld_keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm \+ MLD_NAMESPACE(keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm)+void mld_keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm(+ uint64_t state[100], const uint64_t rc[24])+__contract__(+ requires(memory_no_alias(state, sizeof(uint64_t) * 25 * 4))+ requires(rc == mld_keccakf1600_round_constants)+ assigns(memory_slice(state, sizeof(uint64_t) * 25 * 4))+);++#endif /* !MLD_FIPS202_NATIVE_AARCH64_SRC_FIPS202_NATIVE_AARCH64_H */
@@ -0,0 +1,378 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * Copyright (c) 2021-2022 Arm Limited+ * Copyright (c) 2022 Matthias Kannwischer+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++// Author: Hanno Becker <hanno.becker@arm.com>+// Author: Matthias Kannwischer <matthias@kannwischer.eu>++/*yaml+ Name: keccak_f1600_x1_scalar_aarch64_asm+ Description: AArch64 scalar implementation of Keccak-f[1600] permutation for single state+ Signature: void mld_keccak_f1600_x1_scalar_aarch64_asm(uint64_t state[25], const uint64_t rc[24])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ x0:+ type: buffer+ size_bytes: 200+ permissions: read/write+ c_parameter: uint64_t state[25]+ description: Keccak state (25 x uint64_t)+ x1:+ type: buffer+ size_bytes: 192+ permissions: read-only+ c_parameter: uint64_t const *rc+ description: Round constants (24 x uint64_t)+ Stack:+ bytes: 128+ description: register preservation and temporary storage+*/++#include "../../../../common.h"+#if defined(MLD_FIPS202_AARCH64_NEED_X1_SCALAR) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/fips202/aarch64/src/keccak_f1600_x1_scalar_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(keccak_f1600_x1_scalar_aarch64_asm)+MLD_ASM_FN_SYMBOL(keccak_f1600_x1_scalar_aarch64_asm)++ .cfi_startproc+ sub sp, sp, #0x80+ .cfi_adjust_cfa_offset 0x80+ stp x19, x20, [sp, #0x20]+ .cfi_rel_offset x19, 0x20+ .cfi_rel_offset x20, 0x28+ stp x21, x22, [sp, #0x30]+ .cfi_rel_offset x21, 0x30+ .cfi_rel_offset x22, 0x38+ stp x23, x24, [sp, #0x40]+ .cfi_rel_offset x23, 0x40+ .cfi_rel_offset x24, 0x48+ stp x25, x26, [sp, #0x50]+ .cfi_rel_offset x25, 0x50+ .cfi_rel_offset x26, 0x58+ stp x27, x28, [sp, #0x60]+ .cfi_rel_offset x27, 0x60+ .cfi_rel_offset x28, 0x68+ stp x29, x30, [sp, #0x70]+ .cfi_rel_offset x29, 0x70+ .cfi_rel_offset x30, 0x78++Lmld_keccak_f1600_x1_scalar_initial:+ mov x26, x1+ str x1, [sp, #0x8]+ ldp x1, x6, [x0]+ ldp x11, x16, [x0, #0x10]+ ldp x21, x2, [x0, #0x20]+ ldp x7, x12, [x0, #0x30]+ ldp x17, x22, [x0, #0x40]+ ldp x3, x8, [x0, #0x50]+ ldp x13, x28, [x0, #0x60]+ ldp x23, x4, [x0, #0x70]+ ldp x9, x14, [x0, #0x80]+ ldp x19, x24, [x0, #0x90]+ ldp x5, x10, [x0, #0xa0]+ ldp x15, x20, [x0, #0xb0]+ ldr x25, [x0, #0xc0]+ str x0, [sp]+ eor x30, x24, x25+ eor x27, x9, x10+ eor x0, x30, x21+ eor x26, x27, x6+ eor x27, x26, x7+ eor x29, x0, x22+ eor x26, x29, x23+ eor x29, x4, x5+ eor x30, x29, x1+ eor x0, x27, x8+ eor x29, x30, x2+ eor x30, x19, x20+ eor x30, x30, x16+ eor x27, x26, x0, ror #63+ eor x4, x4, x27+ eor x30, x30, x17+ eor x30, x30, x28+ eor x29, x29, x3+ eor x0, x0, x30, ror #63+ eor x30, x30, x29, ror #63+ eor x22, x22, x30+ eor x23, x23, x30+ str x23, [sp, #0x18]+ eor x23, x14, x15+ eor x14, x14, x0+ eor x23, x23, x11+ eor x15, x15, x0+ eor x1, x1, x27+ eor x23, x23, x12+ eor x23, x23, x13+ eor x11, x11, x0+ eor x29, x29, x23, ror #63+ eor x23, x23, x26, ror #63+ eor x26, x13, x0+ eor x13, x28, x23+ eor x28, x24, x30+ eor x24, x16, x23+ eor x16, x21, x30+ eor x21, x25, x30+ eor x30, x19, x23+ eor x19, x20, x23+ eor x20, x17, x23+ eor x17, x12, x0+ eor x0, x2, x27+ eor x2, x6, x29+ eor x6, x8, x29+ bic x8, x28, x13, ror #47+ eor x12, x3, x27+ bic x3, x13, x17, ror #19+ eor x5, x5, x27+ ldr x27, [sp, #0x18]+ bic x25, x17, x2, ror #5+ eor x9, x9, x29+ eor x23, x25, x5, ror #52+ eor x3, x3, x2, ror #24+ eor x8, x8, x17, ror #2+ eor x17, x10, x29+ bic x25, x12, x22, ror #47+ eor x29, x7, x29+ bic x10, x4, x27, ror #2+ bic x7, x5, x28, ror #10+ eor x10, x10, x20, ror #50+ eor x13, x7, x13, ror #57+ bic x7, x2, x5, ror #47+ eor x2, x25, x24, ror #39+ bic x25, x20, x11, ror #57+ bic x5, x17, x4, ror #25+ eor x25, x25, x17, ror #53+ bic x17, x11, x17, ror #60+ eor x28, x7, x28, ror #57+ bic x7, x9, x12, ror #42+ eor x7, x7, x22, ror #25+ bic x22, x22, x24, ror #56+ bic x24, x24, x15, ror #31+ eor x22, x22, x15, ror #23+ bic x20, x27, x20, ror #48+ bic x15, x15, x9, ror #16+ eor x12, x15, x12, ror #58+ eor x15, x5, x27, ror #27+ eor x5, x20, x11, ror #41+ ldr x11, [sp, #0x8]+ eor x20, x17, x4, ror #21+ eor x17, x24, x9, ror #47+ mov x24, #0x1 // =1+ bic x9, x0, x16, ror #9+ str x24, [sp, #0x10]+ bic x24, x29, x1, ror #44+ bic x27, x1, x21, ror #50+ bic x4, x26, x29, ror #63+ eor x1, x1, x4, ror #21+ ldr x11, [x11]+ bic x4, x21, x30, ror #57+ eor x21, x24, x21, ror #30+ eor x24, x9, x19, ror #44+ bic x9, x14, x6, ror #5+ eor x9, x9, x0, ror #43+ bic x0, x6, x0, ror #38+ eor x1, x1, x11+ eor x11, x4, x26, ror #35+ eor x4, x0, x16, ror #47+ bic x0, x16, x19, ror #35+ eor x16, x27, x30, ror #43+ bic x27, x30, x26, ror #42+ bic x26, x19, x14, ror #41+ eor x19, x0, x14, ror #12+ eor x14, x26, x6, ror #46+ eor x6, x27, x29, ror #41++Lmld_keccak_f1600_x1_scalar_loop:+ eor x0, x15, x11, ror #52+ eor x0, x0, x13, ror #48+ eor x26, x8, x9, ror #57+ eor x27, x0, x14, ror #10+ eor x29, x16, x28, ror #63+ eor x26, x26, x6, ror #51+ eor x30, x23, x22, ror #50+ eor x0, x26, x10, ror #31+ eor x29, x29, x19, ror #37+ eor x27, x27, x12, ror #5+ eor x30, x30, x24, ror #34+ eor x0, x0, x7, ror #27+ eor x26, x30, x21, ror #26+ eor x26, x26, x25, ror #15+ ror x30, x27, #0x3e+ eor x30, x30, x26, ror #57+ ror x26, x26, #0x3a+ eor x16, x30, x16+ eor x28, x30, x28, ror #63+ str x28, [sp, #0x18]+ eor x29, x29, x17, ror #36+ eor x28, x1, x2, ror #61+ eor x19, x30, x19, ror #37+ eor x29, x29, x20, ror #2+ eor x28, x28, x4, ror #54+ eor x26, x26, x0, ror #55+ eor x28, x28, x3, ror #39+ eor x28, x28, x5, ror #25+ ror x0, x0, #0x38+ eor x0, x0, x29, ror #63+ eor x27, x28, x27, ror #61+ eor x13, x0, x13, ror #46+ eor x28, x29, x28, ror #63+ eor x29, x30, x20, ror #2+ eor x20, x26, x3, ror #39+ eor x11, x0, x11, ror #50+ eor x25, x28, x25, ror #9+ eor x3, x28, x21, ror #20+ eor x21, x26, x1+ eor x9, x27, x9, ror #49+ eor x24, x28, x24, ror #28+ eor x1, x30, x17, ror #36+ eor x14, x0, x14, ror #8+ eor x22, x28, x22, ror #44+ eor x8, x27, x8, ror #56+ eor x17, x27, x7, ror #19+ eor x15, x0, x15, ror #62+ bic x7, x20, x22, ror #47+ eor x4, x26, x4, ror #54+ eor x0, x0, x12, ror #3+ eor x28, x28, x23, ror #58+ eor x23, x26, x2, ror #61+ eor x26, x26, x5, ror #25+ eor x2, x7, x16, ror #39+ bic x7, x9, x20, ror #42+ bic x30, x15, x9, ror #16+ eor x7, x7, x22, ror #25+ eor x12, x30, x20, ror #58+ bic x20, x22, x16, ror #56+ eor x30, x27, x6, ror #43+ eor x22, x20, x15, ror #23+ bic x6, x19, x13, ror #42+ eor x6, x6, x17, ror #41+ bic x5, x13, x17, ror #63+ eor x5, x21, x5, ror #21+ bic x17, x17, x21, ror #44+ eor x27, x27, x10, ror #23+ bic x21, x21, x25, ror #50+ bic x20, x27, x4, ror #25+ bic x10, x16, x15, ror #31+ eor x16, x21, x19, ror #43+ eor x21, x17, x25, ror #30+ bic x19, x25, x19, ror #57+ ldr x25, [sp, #0x10]+ eor x17, x10, x9, ror #47+ ldr x9, [sp, #0x8]+ eor x15, x20, x28, ror #27+ bic x20, x4, x28, ror #2+ eor x10, x20, x1, ror #50+ bic x20, x11, x27, ror #60+ eor x20, x20, x4, ror #21+ bic x4, x28, x1, ror #48+ bic x1, x1, x11, ror #57+ ldr x28, [x9, x25, lsl #3]+ ldr x9, [sp, #0x18]+ add x25, x25, #0x1+ str x25, [sp, #0x10]+ cmp x25, #0x17+ eor x25, x1, x27, ror #53+ bic x27, x30, x26, ror #47+ eor x1, x5, x28+ eor x5, x4, x11, ror #41+ eor x11, x19, x13, ror #35+ bic x13, x26, x24, ror #10+ eor x28, x27, x24, ror #57+ bic x27, x24, x9, ror #47+ bic x19, x23, x3, ror #9+ bic x4, x29, x14, ror #41+ eor x24, x19, x29, ror #44+ bic x29, x3, x29, ror #35+ eor x13, x13, x9, ror #57+ eor x19, x29, x14, ror #12+ bic x29, x9, x0, ror #19+ bic x14, x14, x8, ror #5+ eor x9, x14, x23, ror #43+ eor x14, x4, x8, ror #46+ bic x23, x8, x23, ror #38+ eor x8, x27, x0, ror #2+ eor x4, x23, x3, ror #47+ bic x3, x0, x30, ror #5+ eor x23, x3, x26, ror #52+ eor x3, x29, x30, ror #24+ b.le Lmld_keccak_f1600_x1_scalar_loop+ ror x6, x6, #0x2b+ ror x11, x11, #0x32+ ror x21, x21, #0x14+ ror x2, x2, #0x3d+ ror x7, x7, #0x13+ ror x12, x12, #0x3+ ror x17, x17, #0x24+ ror x22, x22, #0x2c+ ror x3, x3, #0x27+ ror x8, x8, #0x38+ ror x13, x13, #0x2e+ ror x28, x28, #0x3f+ ror x23, x23, #0x3a+ ror x4, x4, #0x36+ ror x9, x9, #0x31+ ror x14, x14, #0x8+ ror x19, x19, #0x25+ ror x24, x24, #0x1c+ ror x5, x5, #0x19+ ror x10, x10, #0x17+ ror x15, x15, #0x3e+ ror x20, x20, #0x2+ ror x25, x25, #0x9+ ldr x0, [sp]+ stp x1, x6, [x0]+ stp x11, x16, [x0, #0x10]+ stp x21, x2, [x0, #0x20]+ stp x7, x12, [x0, #0x30]+ stp x17, x22, [x0, #0x40]+ stp x3, x8, [x0, #0x50]+ stp x13, x28, [x0, #0x60]+ stp x23, x4, [x0, #0x70]+ stp x9, x14, [x0, #0x80]+ stp x19, x24, [x0, #0x90]+ stp x5, x10, [x0, #0xa0]+ stp x15, x20, [x0, #0xb0]+ str x25, [x0, #0xc0]+ ldp x19, x20, [sp, #0x20]+ .cfi_restore x19+ .cfi_restore x20+ ldp x21, x22, [sp, #0x30]+ .cfi_restore x21+ .cfi_restore x22+ ldp x23, x24, [sp, #0x40]+ .cfi_restore x23+ .cfi_restore x24+ ldp x25, x26, [sp, #0x50]+ .cfi_restore x25+ .cfi_restore x26+ ldp x27, x28, [sp, #0x60]+ .cfi_restore x27+ .cfi_restore x28+ ldp x29, x30, [sp, #0x70]+ .cfi_restore x29+ .cfi_restore x30+ add sp, sp, #0x80+ .cfi_adjust_cfa_offset -0x80+ ret+ .cfi_endproc++MLD_ASM_FN_SIZE(keccak_f1600_x1_scalar_aarch64_asm)++#endif /* MLD_FIPS202_AARCH64_NEED_X1_SCALAR && \+ !MLD_CONFIG_MULTILEVEL_NO_SHARED */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
@@ -0,0 +1,207 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * Copyright (c) 2021-2022 Arm Limited+ * Copyright (c) 2022 Matthias Kannwischer+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [HYBRID]+ * Hybrid scalar/vector implementations of Keccak and SPHINCS+ on AArch64+ * Becker, Kannwischer+ * https://eprint.iacr.org/2022/1243+ */++/*yaml+ Name: keccak_f1600_x1_v84a_aarch64_asm+ Description: AArch64 ARMv8.4-A implementation of Keccak-f[1600] permutation for single state+ Signature: void mld_keccak_f1600_x1_v84a_aarch64_asm(uint64_t state[25], const uint64_t rc[24])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON, SHA3]+ x0:+ type: buffer+ size_bytes: 200+ permissions: read/write+ c_parameter: uint64_t state[25]+ description: Keccak state (25 x uint64_t)+ x1:+ type: buffer+ size_bytes: 192+ permissions: read-only+ c_parameter: uint64_t const *rc+ description: Round constants (24 x uint64_t)+ Stack:+ bytes: 64+ description: register preservation+*/++//+// Author: Hanno Becker <hanno.becker@arm.com>+// Author: Matthias Kannwischer <matthias@kannwischer.eu>+//+// This implementation is essentially from the paper @[HYBRID].+// The only difference is interleaving/deinterleaving of Keccak state+// during load and store, so that the caller need not do this.+//++#include "../../../../common.h"+#if defined(MLD_FIPS202_AARCH64_NEED_X1_V84A) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++#if defined(__ARM_FEATURE_SHA3)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/fips202/aarch64/src/keccak_f1600_x1_v84a_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(keccak_f1600_x1_v84a_aarch64_asm)+MLD_ASM_FN_SYMBOL(keccak_f1600_x1_v84a_aarch64_asm)++ .cfi_startproc+ sub sp, sp, #0x40+ .cfi_adjust_cfa_offset 0x40+ stp d8, d9, [sp]+ .cfi_rel_offset d8, 0x0+ .cfi_rel_offset d9, 0x8+ stp d10, d11, [sp, #0x10]+ .cfi_rel_offset d10, 0x10+ .cfi_rel_offset d11, 0x18+ stp d12, d13, [sp, #0x20]+ .cfi_rel_offset d12, 0x20+ .cfi_rel_offset d13, 0x28+ stp d14, d15, [sp, #0x30]+ .cfi_rel_offset d14, 0x30+ .cfi_rel_offset d15, 0x38+ ldp d0, d1, [x0]+ ldp d2, d3, [x0, #0x10]+ ldp d4, d5, [x0, #0x20]+ ldp d6, d7, [x0, #0x30]+ ldp d8, d9, [x0, #0x40]+ ldp d10, d11, [x0, #0x50]+ ldp d12, d13, [x0, #0x60]+ ldp d14, d15, [x0, #0x70]+ ldp d16, d17, [x0, #0x80]+ ldp d18, d19, [x0, #0x90]+ ldp d20, d21, [x0, #0xa0]+ ldp d22, d23, [x0, #0xb0]+ ldr d24, [x0, #0xc0]+ mov x2, #0x18 // =24++Lmld_keccak_f1600_x1_v84a_loop:+ eor3 v30.16b, v0.16b, v5.16b, v10.16b+ eor3 v29.16b, v1.16b, v6.16b, v11.16b+ eor3 v28.16b, v2.16b, v7.16b, v12.16b+ eor3 v27.16b, v3.16b, v8.16b, v13.16b+ eor3 v26.16b, v4.16b, v9.16b, v14.16b+ eor3 v30.16b, v30.16b, v15.16b, v20.16b+ eor3 v29.16b, v29.16b, v16.16b, v21.16b+ eor3 v28.16b, v28.16b, v17.16b, v22.16b+ eor3 v27.16b, v27.16b, v18.16b, v23.16b+ eor3 v26.16b, v26.16b, v19.16b, v24.16b+ rax1 v25.2d, v30.2d, v28.2d+ rax1 v28.2d, v28.2d, v26.2d+ rax1 v26.2d, v26.2d, v29.2d+ rax1 v29.2d, v29.2d, v27.2d+ rax1 v27.2d, v27.2d, v30.2d+ eor v30.16b, v0.16b, v26.16b+ xar v0.2d, v2.2d, v29.2d, #0x2+ xar v2.2d, v12.2d, v29.2d, #0x15+ xar v12.2d, v13.2d, v28.2d, #0x27+ xar v13.2d, v19.2d, v27.2d, #0x38+ xar v19.2d, v23.2d, v28.2d, #0x8+ xar v23.2d, v15.2d, v26.2d, #0x17+ xar v15.2d, v1.2d, v25.2d, #0x3f+ xar v1.2d, v8.2d, v28.2d, #0x9+ xar v8.2d, v16.2d, v25.2d, #0x13+ xar v16.2d, v7.2d, v29.2d, #0x3a+ xar v7.2d, v10.2d, v26.2d, #0x3d+ xar v10.2d, v3.2d, v28.2d, #0x24+ xar v3.2d, v18.2d, v28.2d, #0x2b+ xar v18.2d, v17.2d, v29.2d, #0x31+ xar v17.2d, v11.2d, v25.2d, #0x36+ xar v11.2d, v9.2d, v27.2d, #0x2c+ xar v9.2d, v22.2d, v29.2d, #0x3+ xar v22.2d, v14.2d, v27.2d, #0x19+ xar v14.2d, v20.2d, v26.2d, #0x2e+ xar v20.2d, v4.2d, v27.2d, #0x25+ xar v4.2d, v24.2d, v27.2d, #0x32+ xar v24.2d, v21.2d, v25.2d, #0x3e+ xar v21.2d, v5.2d, v26.2d, #0x1c+ xar v27.2d, v6.2d, v25.2d, #0x14+ ld1r { v31.2d }, [x1], #8+ bcax v5.16b, v10.16b, v7.16b, v11.16b+ bcax v6.16b, v11.16b, v8.16b, v7.16b+ bcax v7.16b, v7.16b, v9.16b, v8.16b+ bcax v8.16b, v8.16b, v10.16b, v9.16b+ bcax v9.16b, v9.16b, v11.16b, v10.16b+ bcax v10.16b, v15.16b, v12.16b, v16.16b+ bcax v11.16b, v16.16b, v13.16b, v12.16b+ bcax v12.16b, v12.16b, v14.16b, v13.16b+ bcax v13.16b, v13.16b, v15.16b, v14.16b+ bcax v14.16b, v14.16b, v16.16b, v15.16b+ bcax v15.16b, v20.16b, v17.16b, v21.16b+ bcax v16.16b, v21.16b, v18.16b, v17.16b+ bcax v17.16b, v17.16b, v19.16b, v18.16b+ bcax v18.16b, v18.16b, v20.16b, v19.16b+ bcax v19.16b, v19.16b, v21.16b, v20.16b+ bcax v20.16b, v0.16b, v22.16b, v1.16b+ bcax v21.16b, v1.16b, v23.16b, v22.16b+ bcax v22.16b, v22.16b, v24.16b, v23.16b+ bcax v23.16b, v23.16b, v0.16b, v24.16b+ bcax v24.16b, v24.16b, v1.16b, v0.16b+ bcax v0.16b, v30.16b, v2.16b, v27.16b+ bcax v1.16b, v27.16b, v3.16b, v2.16b+ bcax v2.16b, v2.16b, v4.16b, v3.16b+ bcax v3.16b, v3.16b, v30.16b, v4.16b+ bcax v4.16b, v4.16b, v27.16b, v30.16b+ eor v0.16b, v0.16b, v31.16b+ sub x2, x2, #0x1+ cbnz x2, Lmld_keccak_f1600_x1_v84a_loop+ stp d0, d1, [x0]+ stp d2, d3, [x0, #0x10]+ stp d4, d5, [x0, #0x20]+ stp d6, d7, [x0, #0x30]+ stp d8, d9, [x0, #0x40]+ stp d10, d11, [x0, #0x50]+ stp d12, d13, [x0, #0x60]+ stp d14, d15, [x0, #0x70]+ stp d16, d17, [x0, #0x80]+ stp d18, d19, [x0, #0x90]+ stp d20, d21, [x0, #0xa0]+ stp d22, d23, [x0, #0xb0]+ str d24, [x0, #0xc0]+ ldp d8, d9, [sp]+ .cfi_restore d8+ .cfi_restore d9+ ldp d10, d11, [sp, #0x10]+ .cfi_restore d10+ .cfi_restore d11+ ldp d12, d13, [sp, #0x20]+ .cfi_restore d12+ .cfi_restore d13+ ldp d14, d15, [sp, #0x30]+ .cfi_restore d14+ .cfi_restore d15+ add sp, sp, #0x40+ .cfi_adjust_cfa_offset -0x40+ ret+ .cfi_endproc++MLD_ASM_FN_SIZE(keccak_f1600_x1_v84a_aarch64_asm)++#endif /* __ARM_FEATURE_SHA3 */++#endif /* MLD_FIPS202_AARCH64_NEED_X1_V84A && !MLD_CONFIG_MULTILEVEL_NO_SHARED \+ */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
@@ -0,0 +1,262 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * Copyright (c) 2021-2022 Arm Limited+ * Copyright (c) 2022 Matthias Kannwischer+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [HYBRID]+ * Hybrid scalar/vector implementations of Keccak and SPHINCS+ on AArch64+ * Becker, Kannwischer+ * https://eprint.iacr.org/2022/1243+ */++/*yaml+ Name: keccak_f1600_x2_v84a_aarch64_asm+ Description: AArch64 ARMv8.4-A implementation of Keccak-f[1600] permutation for two sequential states+ Signature: void mld_keccak_f1600_x2_v84a_aarch64_asm(uint64_t state[50], const uint64_t rc[24])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON, SHA3]+ x0:+ type: buffer+ size_bytes: 400+ permissions: read/write+ c_parameter: uint64_t state[50]+ description: Two sequential Keccak states (state0[25], state1[25])+ x1:+ type: buffer+ size_bytes: 192+ permissions: read-only+ c_parameter: uint64_t const *rc+ description: Round constants (24 x uint64_t)+ Stack:+ bytes: 64+ description: register preservation+*/++//+// Author: Hanno Becker <hanno.becker@arm.com>+// Author: Matthias Kannwischer <matthias@kannwischer.eu>+//+// This implementation is essentially from the paper @[HYBRID].+// The only difference is interleaving/deinterleaving of Keccak state+// during load and store, so that the caller need not do this.+//++#include "../../../../common.h"+#if defined(MLD_FIPS202_AARCH64_NEED_X2_V84A) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++#if defined(__ARM_FEATURE_SHA3)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/fips202/aarch64/src/keccak_f1600_x2_v84a_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(keccak_f1600_x2_v84a_aarch64_asm)+MLD_ASM_FN_SYMBOL(keccak_f1600_x2_v84a_aarch64_asm)++ .cfi_startproc+ sub sp, sp, #0x40+ .cfi_adjust_cfa_offset 0x40+ stp d8, d9, [sp]+ .cfi_rel_offset d8, 0x0+ .cfi_rel_offset d9, 0x8+ stp d10, d11, [sp, #0x10]+ .cfi_rel_offset d10, 0x10+ .cfi_rel_offset d11, 0x18+ stp d12, d13, [sp, #0x20]+ .cfi_rel_offset d12, 0x20+ .cfi_rel_offset d13, 0x28+ stp d14, d15, [sp, #0x30]+ .cfi_rel_offset d14, 0x30+ .cfi_rel_offset d15, 0x38+ add x2, x0, #0xc8+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x2], #32+ trn1 v0.2d, v25.2d, v27.2d+ trn2 v1.2d, v25.2d, v27.2d+ trn1 v2.2d, v26.2d, v28.2d+ trn2 v3.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x2], #32+ trn1 v4.2d, v25.2d, v27.2d+ trn2 v5.2d, v25.2d, v27.2d+ trn1 v6.2d, v26.2d, v28.2d+ trn2 v7.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x2], #32+ trn1 v8.2d, v25.2d, v27.2d+ trn2 v9.2d, v25.2d, v27.2d+ trn1 v10.2d, v26.2d, v28.2d+ trn2 v11.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x2], #32+ trn1 v12.2d, v25.2d, v27.2d+ trn2 v13.2d, v25.2d, v27.2d+ trn1 v14.2d, v26.2d, v28.2d+ trn2 v15.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x2], #32+ trn1 v16.2d, v25.2d, v27.2d+ trn2 v17.2d, v25.2d, v27.2d+ trn1 v18.2d, v26.2d, v28.2d+ trn2 v19.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x2], #32+ trn1 v20.2d, v25.2d, v27.2d+ trn2 v21.2d, v25.2d, v27.2d+ trn1 v22.2d, v26.2d, v28.2d+ trn2 v23.2d, v26.2d, v28.2d+ ldr d25, [x0]+ ldr d27, [x2]+ trn1 v24.2d, v25.2d, v27.2d+ mov x2, #0x18 // =24++Lmld_keccak_f1600_x2_v84a_loop:+ eor3 v30.16b, v0.16b, v5.16b, v10.16b+ eor3 v29.16b, v1.16b, v6.16b, v11.16b+ eor3 v28.16b, v2.16b, v7.16b, v12.16b+ eor3 v27.16b, v3.16b, v8.16b, v13.16b+ eor3 v26.16b, v4.16b, v9.16b, v14.16b+ eor3 v30.16b, v30.16b, v15.16b, v20.16b+ eor3 v29.16b, v29.16b, v16.16b, v21.16b+ eor3 v28.16b, v28.16b, v17.16b, v22.16b+ eor3 v27.16b, v27.16b, v18.16b, v23.16b+ eor3 v26.16b, v26.16b, v19.16b, v24.16b+ rax1 v25.2d, v30.2d, v28.2d+ rax1 v28.2d, v28.2d, v26.2d+ rax1 v26.2d, v26.2d, v29.2d+ rax1 v29.2d, v29.2d, v27.2d+ rax1 v27.2d, v27.2d, v30.2d+ eor v30.16b, v0.16b, v26.16b+ xar v0.2d, v2.2d, v29.2d, #0x2+ xar v2.2d, v12.2d, v29.2d, #0x15+ xar v12.2d, v13.2d, v28.2d, #0x27+ xar v13.2d, v19.2d, v27.2d, #0x38+ xar v19.2d, v23.2d, v28.2d, #0x8+ xar v23.2d, v15.2d, v26.2d, #0x17+ xar v15.2d, v1.2d, v25.2d, #0x3f+ xar v1.2d, v8.2d, v28.2d, #0x9+ xar v8.2d, v16.2d, v25.2d, #0x13+ xar v16.2d, v7.2d, v29.2d, #0x3a+ xar v7.2d, v10.2d, v26.2d, #0x3d+ xar v10.2d, v3.2d, v28.2d, #0x24+ xar v3.2d, v18.2d, v28.2d, #0x2b+ xar v18.2d, v17.2d, v29.2d, #0x31+ xar v17.2d, v11.2d, v25.2d, #0x36+ xar v11.2d, v9.2d, v27.2d, #0x2c+ xar v9.2d, v22.2d, v29.2d, #0x3+ xar v22.2d, v14.2d, v27.2d, #0x19+ xar v14.2d, v20.2d, v26.2d, #0x2e+ xar v20.2d, v4.2d, v27.2d, #0x25+ xar v4.2d, v24.2d, v27.2d, #0x32+ xar v24.2d, v21.2d, v25.2d, #0x3e+ xar v21.2d, v5.2d, v26.2d, #0x1c+ xar v27.2d, v6.2d, v25.2d, #0x14+ ld1r { v31.2d }, [x1], #8+ bcax v5.16b, v10.16b, v7.16b, v11.16b+ bcax v6.16b, v11.16b, v8.16b, v7.16b+ bcax v7.16b, v7.16b, v9.16b, v8.16b+ bcax v8.16b, v8.16b, v10.16b, v9.16b+ bcax v9.16b, v9.16b, v11.16b, v10.16b+ bcax v10.16b, v15.16b, v12.16b, v16.16b+ bcax v11.16b, v16.16b, v13.16b, v12.16b+ bcax v12.16b, v12.16b, v14.16b, v13.16b+ bcax v13.16b, v13.16b, v15.16b, v14.16b+ bcax v14.16b, v14.16b, v16.16b, v15.16b+ bcax v15.16b, v20.16b, v17.16b, v21.16b+ bcax v16.16b, v21.16b, v18.16b, v17.16b+ bcax v17.16b, v17.16b, v19.16b, v18.16b+ bcax v18.16b, v18.16b, v20.16b, v19.16b+ bcax v19.16b, v19.16b, v21.16b, v20.16b+ bcax v20.16b, v0.16b, v22.16b, v1.16b+ bcax v21.16b, v1.16b, v23.16b, v22.16b+ bcax v22.16b, v22.16b, v24.16b, v23.16b+ bcax v23.16b, v23.16b, v0.16b, v24.16b+ bcax v24.16b, v24.16b, v1.16b, v0.16b+ bcax v0.16b, v30.16b, v2.16b, v27.16b+ bcax v1.16b, v27.16b, v3.16b, v2.16b+ bcax v2.16b, v2.16b, v4.16b, v3.16b+ bcax v3.16b, v3.16b, v30.16b, v4.16b+ bcax v4.16b, v4.16b, v27.16b, v30.16b+ eor v0.16b, v0.16b, v31.16b+ sub x2, x2, #0x1+ cbnz x2, Lmld_keccak_f1600_x2_v84a_loop+ sub x0, x0, #0xc0+ add x2, x0, #0xc8+ trn1 v25.2d, v0.2d, v1.2d+ trn1 v26.2d, v2.2d, v3.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v0.2d, v1.2d+ trn2 v28.2d, v2.2d, v3.2d+ st1 { v27.2d, v28.2d }, [x2], #32+ trn1 v25.2d, v4.2d, v5.2d+ trn1 v26.2d, v6.2d, v7.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v4.2d, v5.2d+ trn2 v28.2d, v6.2d, v7.2d+ st1 { v27.2d, v28.2d }, [x2], #32+ trn1 v25.2d, v8.2d, v9.2d+ trn1 v26.2d, v10.2d, v11.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v8.2d, v9.2d+ trn2 v28.2d, v10.2d, v11.2d+ st1 { v27.2d, v28.2d }, [x2], #32+ trn1 v25.2d, v12.2d, v13.2d+ trn1 v26.2d, v14.2d, v15.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v12.2d, v13.2d+ trn2 v28.2d, v14.2d, v15.2d+ st1 { v27.2d, v28.2d }, [x2], #32+ trn1 v25.2d, v16.2d, v17.2d+ trn1 v26.2d, v18.2d, v19.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v16.2d, v17.2d+ trn2 v28.2d, v18.2d, v19.2d+ st1 { v27.2d, v28.2d }, [x2], #32+ trn1 v25.2d, v20.2d, v21.2d+ trn1 v26.2d, v22.2d, v23.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v20.2d, v21.2d+ trn2 v28.2d, v22.2d, v23.2d+ st1 { v27.2d, v28.2d }, [x2], #32+ str d24, [x0]+ trn2 v25.2d, v24.2d, v24.2d+ str d25, [x2]+ ldp d8, d9, [sp]+ .cfi_restore d8+ .cfi_restore d9+ ldp d10, d11, [sp, #0x10]+ .cfi_restore d10+ .cfi_restore d11+ ldp d12, d13, [sp, #0x20]+ .cfi_restore d12+ .cfi_restore d13+ ldp d14, d15, [sp, #0x30]+ .cfi_restore d14+ .cfi_restore d15+ add sp, sp, #0x40+ .cfi_adjust_cfa_offset -0x40+ ret+ .cfi_endproc++MLD_ASM_FN_SIZE(keccak_f1600_x2_v84a_aarch64_asm)++#endif /* __ARM_FEATURE_SHA3 */++#endif /* MLD_FIPS202_AARCH64_NEED_X2_V84A && !MLD_CONFIG_MULTILEVEL_NO_SHARED \+ */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
@@ -0,0 +1,1080 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * Copyright (c) 2021-2022 Arm Limited+ * Copyright (c) 2022 Matthias Kannwischer+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++// Author: Hanno Becker <hannobecker@posteo.de>+// Author: Matthias Kannwischer <matthias@kannwischer.eu>++/*yaml+ Name: keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm+ Description: AArch64 hybrid scalar/vector implementation of Keccak-f[1600] permutation for four sequential states+ Signature: void mld_keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm(uint64_t state[100], const uint64_t rc[24])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON]+ x0:+ type: buffer+ size_bytes: 800+ permissions: read/write+ c_parameter: uint64_t state[100]+ description: Four sequential Keccak states (state0[25], state1[25], state2[25], state3[25])+ x1:+ type: buffer+ size_bytes: 192+ permissions: read-only+ c_parameter: uint64_t const *rc+ description: Round constants (24 x uint64_t)+ Stack:+ bytes: 224+ description: register preservation and temporary storage+*/++#include "../../../../common.h"+#if defined(MLD_FIPS202_AARCH64_NEED_X4_V8A_SCALAR_HYBRID) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/fips202/aarch64/src/keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm)+MLD_ASM_FN_SYMBOL(keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm)++ .cfi_startproc+ sub sp, sp, #0xe0+ .cfi_adjust_cfa_offset 0xe0+ stp x19, x20, [sp, #0x30]+ .cfi_rel_offset x19, 0x30+ .cfi_rel_offset x20, 0x38+ stp x21, x22, [sp, #0x40]+ .cfi_rel_offset x21, 0x40+ .cfi_rel_offset x22, 0x48+ stp x23, x24, [sp, #0x50]+ .cfi_rel_offset x23, 0x50+ .cfi_rel_offset x24, 0x58+ stp x25, x26, [sp, #0x60]+ .cfi_rel_offset x25, 0x60+ .cfi_rel_offset x26, 0x68+ stp x27, x28, [sp, #0x70]+ .cfi_rel_offset x27, 0x70+ .cfi_rel_offset x28, 0x78+ stp x29, x30, [sp, #0x80]+ .cfi_rel_offset x29, 0x80+ .cfi_rel_offset x30, 0x88+ stp d8, d9, [sp, #0x90]+ .cfi_rel_offset d8, 0x90+ .cfi_rel_offset d9, 0x98+ stp d10, d11, [sp, #0xa0]+ .cfi_rel_offset d10, 0xa0+ .cfi_rel_offset d11, 0xa8+ stp d12, d13, [sp, #0xb0]+ .cfi_rel_offset d12, 0xb0+ .cfi_rel_offset d13, 0xb8+ stp d14, d15, [sp, #0xc0]+ .cfi_rel_offset d14, 0xc0+ .cfi_rel_offset d15, 0xc8+ mov x29, x1+ mov x30, #0x0 // =0+ str x30, [sp, #0x20]+ str x29, [sp, #0x8]+ str x29, [sp, #0x10]+ str x0, [sp]+ add x4, x0, #0xc8+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x4], #32+ trn1 v0.2d, v25.2d, v27.2d+ trn2 v1.2d, v25.2d, v27.2d+ trn1 v2.2d, v26.2d, v28.2d+ trn2 v3.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x4], #32+ trn1 v4.2d, v25.2d, v27.2d+ trn2 v5.2d, v25.2d, v27.2d+ trn1 v6.2d, v26.2d, v28.2d+ trn2 v7.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x4], #32+ trn1 v8.2d, v25.2d, v27.2d+ trn2 v9.2d, v25.2d, v27.2d+ trn1 v10.2d, v26.2d, v28.2d+ trn2 v11.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x4], #32+ trn1 v12.2d, v25.2d, v27.2d+ trn2 v13.2d, v25.2d, v27.2d+ trn1 v14.2d, v26.2d, v28.2d+ trn2 v15.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x4], #32+ trn1 v16.2d, v25.2d, v27.2d+ trn2 v17.2d, v25.2d, v27.2d+ trn1 v18.2d, v26.2d, v28.2d+ trn2 v19.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x4], #32+ trn1 v20.2d, v25.2d, v27.2d+ trn2 v21.2d, v25.2d, v27.2d+ trn1 v22.2d, v26.2d, v28.2d+ trn2 v23.2d, v26.2d, v28.2d+ ldr d25, [x0]+ ldr d27, [x4]+ trn1 v24.2d, v25.2d, v27.2d+ sub x0, x0, #0xc0+ add x0, x0, #0x190+ ldp x1, x6, [x0]+ ldp x11, x16, [x0, #0x10]+ ldp x21, x2, [x0, #0x20]+ ldp x7, x12, [x0, #0x30]+ ldp x17, x22, [x0, #0x40]+ ldp x3, x8, [x0, #0x50]+ ldp x13, x28, [x0, #0x60]+ ldp x23, x4, [x0, #0x70]+ ldp x9, x14, [x0, #0x80]+ ldp x19, x24, [x0, #0x90]+ ldp x5, x10, [x0, #0xa0]+ ldp x15, x20, [x0, #0xb0]+ ldr x25, [x0, #0xc0]+ sub x0, x0, #0x190++Lmld_keccak_f1600_x4_v8a_scalar_hybrid_initial:+ eor x30, x24, x25+ eor x27, x9, x10+ eor v30.16b, v0.16b, v5.16b+ eor v30.16b, v30.16b, v10.16b+ eor x0, x30, x21+ eor v30.16b, v30.16b, v15.16b+ eor x26, x27, x6+ eor x27, x26, x7+ eor v30.16b, v30.16b, v20.16b+ eor x29, x0, x22+ eor v29.16b, v1.16b, v6.16b+ eor x26, x29, x23+ eor v29.16b, v29.16b, v11.16b+ eor x29, x4, x5+ eor x30, x29, x1+ eor v29.16b, v29.16b, v16.16b+ eor x0, x27, x8+ eor v29.16b, v29.16b, v21.16b+ eor x29, x30, x2+ eor v28.16b, v2.16b, v7.16b+ eor x30, x19, x20+ eor x30, x30, x16+ eor v28.16b, v28.16b, v12.16b+ eor x27, x26, x0, ror #63+ eor v28.16b, v28.16b, v17.16b+ eor x4, x4, x27+ eor v28.16b, v28.16b, v22.16b+ eor x30, x30, x17+ eor x30, x30, x28+ eor v27.16b, v3.16b, v8.16b+ eor x29, x29, x3+ eor v27.16b, v27.16b, v13.16b+ eor x0, x0, x30, ror #63+ eor v27.16b, v27.16b, v18.16b+ eor x30, x30, x29, ror #63+ eor x22, x22, x30+ eor v27.16b, v27.16b, v23.16b+ eor x23, x23, x30+ eor v26.16b, v4.16b, v9.16b+ str x23, [sp, #0xd0]+ eor v26.16b, v26.16b, v14.16b+ eor x23, x14, x15+ eor x14, x14, x0+ eor v26.16b, v26.16b, v19.16b+ eor x23, x23, x11+ eor v26.16b, v26.16b, v24.16b+ eor x15, x15, x0+ eor x1, x1, x27+ add v31.2d, v28.2d, v28.2d+ eor x23, x23, x12+ sri v31.2d, v28.2d, #0x3f+ eor x23, x23, x13+ eor v25.16b, v31.16b, v30.16b+ eor x11, x11, x0+ eor x29, x29, x23, ror #63+ add v31.2d, v26.2d, v26.2d+ eor x23, x23, x26, ror #63+ sri v31.2d, v26.2d, #0x3f+ eor x26, x13, x0+ eor v28.16b, v31.16b, v28.16b+ eor x13, x28, x23+ eor x28, x24, x30+ add v31.2d, v29.2d, v29.2d+ eor x24, x16, x23+ sri v31.2d, v29.2d, #0x3f+ eor x16, x21, x30+ eor v26.16b, v31.16b, v26.16b+ eor x21, x25, x30+ eor x30, x19, x23+ add v31.2d, v27.2d, v27.2d+ eor x19, x20, x23+ sri v31.2d, v27.2d, #0x3f+ eor x20, x17, x23+ eor v29.16b, v31.16b, v29.16b+ eor x17, x12, x0+ eor x0, x2, x27+ add v31.2d, v30.2d, v30.2d+ eor x2, x6, x29+ sri v31.2d, v30.2d, #0x3f+ eor x6, x8, x29+ eor v27.16b, v31.16b, v27.16b+ bic x8, x28, x13, ror #47+ eor x12, x3, x27+ eor v30.16b, v0.16b, v26.16b+ bic x3, x13, x17, ror #19+ eor v31.16b, v2.16b, v29.16b+ eor x5, x5, x27+ ldr x27, [sp, #0xd0]+ shl v0.2d, v31.2d, #0x3e+ bic x25, x17, x2, ror #5+ sri v0.2d, v31.2d, #0x2+ eor x9, x9, x29+ eor v31.16b, v12.16b, v29.16b+ eor x23, x25, x5, ror #52+ eor x3, x3, x2, ror #24+ shl v2.2d, v31.2d, #0x2b+ eor x8, x8, x17, ror #2+ sri v2.2d, v31.2d, #0x15+ eor x17, x10, x29+ eor v31.16b, v13.16b, v28.16b+ bic x25, x12, x22, ror #47+ eor x29, x7, x29+ shl v12.2d, v31.2d, #0x19+ bic x10, x4, x27, ror #2+ sri v12.2d, v31.2d, #0x27+ bic x7, x5, x28, ror #10+ eor v31.16b, v19.16b, v27.16b+ eor x10, x10, x20, ror #50+ eor x13, x7, x13, ror #57+ shl v13.2d, v31.2d, #0x8+ bic x7, x2, x5, ror #47+ sri v13.2d, v31.2d, #0x38+ eor x2, x25, x24, ror #39+ eor v31.16b, v23.16b, v28.16b+ bic x25, x20, x11, ror #57+ bic x5, x17, x4, ror #25+ shl v19.2d, v31.2d, #0x38+ eor x25, x25, x17, ror #53+ sri v19.2d, v31.2d, #0x8+ bic x17, x11, x17, ror #60+ eor v31.16b, v15.16b, v26.16b+ eor x28, x7, x28, ror #57+ bic x7, x9, x12, ror #42+ shl v23.2d, v31.2d, #0x29+ eor x7, x7, x22, ror #25+ sri v23.2d, v31.2d, #0x17+ bic x22, x22, x24, ror #56+ bic x24, x24, x15, ror #31+ eor v31.16b, v1.16b, v25.16b+ eor x22, x22, x15, ror #23+ shl v15.2d, v31.2d, #0x1+ bic x20, x27, x20, ror #48+ sri v15.2d, v31.2d, #0x3f+ bic x15, x15, x9, ror #16+ eor x12, x15, x12, ror #58+ eor v31.16b, v8.16b, v28.16b+ eor x15, x5, x27, ror #27+ shl v1.2d, v31.2d, #0x37+ eor x5, x20, x11, ror #41+ sri v1.2d, v31.2d, #0x9+ ldr x11, [sp, #0x8]+ eor x20, x17, x4, ror #21+ eor v31.16b, v16.16b, v25.16b+ eor x17, x24, x9, ror #47+ shl v8.2d, v31.2d, #0x2d+ mov x24, #0x1 // =1+ sri v8.2d, v31.2d, #0x13+ bic x9, x0, x16, ror #9+ str x24, [sp, #0x18]+ eor v31.16b, v7.16b, v29.16b+ bic x24, x29, x1, ror #44+ shl v16.2d, v31.2d, #0x6+ bic x27, x1, x21, ror #50+ sri v16.2d, v31.2d, #0x3a+ bic x4, x26, x29, ror #63+ eor x1, x1, x4, ror #21+ eor v31.16b, v10.16b, v26.16b+ ldr x11, [x11]+ shl v7.2d, v31.2d, #0x3+ bic x4, x21, x30, ror #57+ sri v7.2d, v31.2d, #0x3d+ eor x21, x24, x21, ror #30+ eor x24, x9, x19, ror #44+ eor v31.16b, v3.16b, v28.16b+ bic x9, x14, x6, ror #5+ shl v10.2d, v31.2d, #0x1c+ eor x9, x9, x0, ror #43+ sri v10.2d, v31.2d, #0x24+ bic x0, x6, x0, ror #38+ eor x1, x1, x11+ eor v31.16b, v18.16b, v28.16b+ eor x11, x4, x26, ror #35+ shl v3.2d, v31.2d, #0x15+ eor x4, x0, x16, ror #47+ bic x0, x16, x19, ror #35+ sri v3.2d, v31.2d, #0x2b+ eor x16, x27, x30, ror #43+ eor v31.16b, v17.16b, v29.16b+ bic x27, x30, x26, ror #42+ shl v18.2d, v31.2d, #0xf+ bic x26, x19, x14, ror #41+ eor x19, x0, x14, ror #12+ sri v18.2d, v31.2d, #0x31+ eor x14, x26, x6, ror #46+ eor v31.16b, v11.16b, v25.16b+ eor x6, x27, x29, ror #41+ shl v17.2d, v31.2d, #0xa+ eor x0, x15, x11, ror #52+ eor x0, x0, x13, ror #48+ sri v17.2d, v31.2d, #0x36+ eor x26, x8, x9, ror #57+ eor v31.16b, v9.16b, v27.16b+ eor x27, x0, x14, ror #10+ shl v11.2d, v31.2d, #0x14+ eor x29, x16, x28, ror #63+ eor x26, x26, x6, ror #51+ sri v11.2d, v31.2d, #0x2c+ eor x30, x23, x22, ror #50+ eor v31.16b, v22.16b, v29.16b+ eor x0, x26, x10, ror #31+ shl v9.2d, v31.2d, #0x3d+ eor x29, x29, x19, ror #37+ eor x27, x27, x12, ror #5+ sri v9.2d, v31.2d, #0x3+ eor x30, x30, x24, ror #34+ eor v31.16b, v14.16b, v27.16b+ eor x0, x0, x7, ror #27+ shl v22.2d, v31.2d, #0x27+ eor x26, x30, x21, ror #26+ eor x26, x26, x25, ror #15+ sri v22.2d, v31.2d, #0x19+ ror x30, x27, #0x3e+ eor v31.16b, v20.16b, v26.16b+ eor x30, x30, x26, ror #57+ ror x26, x26, #0x3a+ shl v14.2d, v31.2d, #0x12+ eor x16, x30, x16+ sri v14.2d, v31.2d, #0x2e+ eor x28, x30, x28, ror #63+ eor v31.16b, v4.16b, v27.16b+ str x28, [sp, #0xd0]+ eor x29, x29, x17, ror #36+ shl v20.2d, v31.2d, #0x1b+ eor x28, x1, x2, ror #61+ sri v20.2d, v31.2d, #0x25+ eor x19, x30, x19, ror #37+ eor v31.16b, v24.16b, v27.16b+ eor x29, x29, x20, ror #2+ eor x28, x28, x4, ror #54+ shl v4.2d, v31.2d, #0xe+ eor x26, x26, x0, ror #55+ sri v4.2d, v31.2d, #0x32+ eor x28, x28, x3, ror #39+ eor v31.16b, v21.16b, v25.16b+ eor x28, x28, x5, ror #25+ ror x0, x0, #0x38+ shl v24.2d, v31.2d, #0x2+ eor x0, x0, x29, ror #63+ sri v24.2d, v31.2d, #0x3e+ eor x27, x28, x27, ror #61+ eor v31.16b, v5.16b, v26.16b+ eor x13, x0, x13, ror #46+ eor x28, x29, x28, ror #63+ shl v21.2d, v31.2d, #0x24+ eor x29, x30, x20, ror #2+ sri v21.2d, v31.2d, #0x1c+ eor x20, x26, x3, ror #39+ eor v31.16b, v6.16b, v25.16b+ eor x11, x0, x11, ror #50+ eor x25, x28, x25, ror #9+ shl v27.2d, v31.2d, #0x2c+ eor x3, x28, x21, ror #20+ sri v27.2d, v31.2d, #0x14+ eor x21, x26, x1+ eor x9, x27, x9, ror #49+ bic v31.16b, v7.16b, v11.16b+ eor x24, x28, x24, ror #28+ eor v5.16b, v31.16b, v10.16b+ eor x1, x30, x17, ror #36+ bic v31.16b, v8.16b, v7.16b+ eor x14, x0, x14, ror #8+ eor x22, x28, x22, ror #44+ eor v6.16b, v31.16b, v11.16b+ eor x8, x27, x8, ror #56+ bic v31.16b, v9.16b, v8.16b+ eor x17, x27, x7, ror #19+ eor v7.16b, v31.16b, v7.16b+ eor x15, x0, x15, ror #62+ bic x7, x20, x22, ror #47+ bic v31.16b, v10.16b, v9.16b+ eor x4, x26, x4, ror #54+ eor v8.16b, v31.16b, v8.16b+ eor x0, x0, x12, ror #3+ bic v31.16b, v11.16b, v10.16b+ eor x28, x28, x23, ror #58+ eor x23, x26, x2, ror #61+ eor v9.16b, v31.16b, v9.16b+ eor x26, x26, x5, ror #25+ bic v31.16b, v12.16b, v16.16b+ eor x2, x7, x16, ror #39+ eor v10.16b, v31.16b, v15.16b+ bic x7, x9, x20, ror #42+ bic x30, x15, x9, ror #16+ bic v31.16b, v13.16b, v12.16b+ eor x7, x7, x22, ror #25+ eor v11.16b, v31.16b, v16.16b+ eor x12, x30, x20, ror #58+ bic v31.16b, v14.16b, v13.16b+ bic x20, x22, x16, ror #56+ eor x30, x27, x6, ror #43+ eor v12.16b, v31.16b, v12.16b+ eor x22, x20, x15, ror #23+ bic v31.16b, v15.16b, v14.16b+ bic x6, x19, x13, ror #42+ eor v13.16b, v31.16b, v13.16b+ eor x6, x6, x17, ror #41+ bic x5, x13, x17, ror #63+ bic v31.16b, v16.16b, v15.16b+ eor x5, x21, x5, ror #21+ eor v14.16b, v31.16b, v14.16b+ bic x17, x17, x21, ror #44+ eor x27, x27, x10, ror #23+ bic v31.16b, v17.16b, v21.16b+ bic x21, x21, x25, ror #50+ eor v15.16b, v31.16b, v20.16b+ bic x20, x27, x4, ror #25+ bic v31.16b, v18.16b, v17.16b+ bic x10, x16, x15, ror #31+ eor x16, x21, x19, ror #43+ eor v16.16b, v31.16b, v21.16b+ eor x21, x17, x25, ror #30+ bic v31.16b, v19.16b, v18.16b+ bic x19, x25, x19, ror #57+ eor v17.16b, v31.16b, v17.16b+ ldr x25, [sp, #0x18]+ eor x17, x10, x9, ror #47+ bic v31.16b, v20.16b, v19.16b+ ldr x9, [sp, #0x8]+ eor v18.16b, v31.16b, v18.16b+ eor x15, x20, x28, ror #27+ bic v31.16b, v21.16b, v20.16b+ bic x20, x4, x28, ror #2+ eor x10, x20, x1, ror #50+ eor v19.16b, v31.16b, v19.16b+ bic x20, x11, x27, ror #60+ bic v31.16b, v22.16b, v1.16b+ eor x20, x20, x4, ror #21+ eor v20.16b, v31.16b, v0.16b+ bic x4, x28, x1, ror #48+ bic x1, x1, x11, ror #57+ bic v31.16b, v23.16b, v22.16b+ ldr x28, [x9, x25, lsl #3]+ eor v21.16b, v31.16b, v1.16b+ ldr x9, [sp, #0xd0]+ bic v31.16b, v24.16b, v23.16b+ add x25, x25, #0x1+ str x25, [sp, #0x18]+ eor v22.16b, v31.16b, v22.16b+ cmp x25, #0x17+ bic v31.16b, v0.16b, v24.16b+ eor x25, x1, x27, ror #53+ bic x27, x30, x26, ror #47+ eor v23.16b, v31.16b, v23.16b+ eor x1, x5, x28+ bic v31.16b, v1.16b, v0.16b+ eor x5, x4, x11, ror #41+ eor v24.16b, v31.16b, v24.16b+ eor x11, x19, x13, ror #35+ bic x13, x26, x24, ror #10+ bic v31.16b, v2.16b, v27.16b+ eor x28, x27, x24, ror #57+ eor v0.16b, v31.16b, v30.16b+ bic x27, x24, x9, ror #47+ bic v31.16b, v3.16b, v2.16b+ bic x19, x23, x3, ror #9+ bic x4, x29, x14, ror #41+ eor v1.16b, v31.16b, v27.16b+ eor x24, x19, x29, ror #44+ bic v31.16b, v4.16b, v3.16b+ bic x29, x3, x29, ror #35+ eor v2.16b, v31.16b, v2.16b+ eor x13, x13, x9, ror #57+ eor x19, x29, x14, ror #12+ bic v31.16b, v30.16b, v4.16b+ bic x29, x9, x0, ror #19+ eor v3.16b, v31.16b, v3.16b+ bic x14, x14, x8, ror #5+ bic v31.16b, v27.16b, v30.16b+ eor x9, x14, x23, ror #43+ eor x14, x4, x8, ror #46+ eor v4.16b, v31.16b, v4.16b+ bic x23, x8, x23, ror #38+ eor x8, x27, x0, ror #2+ eor x4, x23, x3, ror #47+ bic x3, x0, x30, ror #5+ eor x23, x3, x26, ror #52+ eor x3, x29, x30, ror #24+ ldr x30, [sp, #0x10]+ ld1r { v28.2d }, [x30], #8+ str x30, [sp, #0x10]+ eor v0.16b, v0.16b, v28.16b++Lmld_keccak_f1600_x4_v8a_scalar_hybrid_loop:+ eor x0, x15, x11, ror #52+ eor x0, x0, x13, ror #48+ eor v30.16b, v0.16b, v5.16b+ eor v30.16b, v30.16b, v10.16b+ eor x26, x8, x9, ror #57+ eor v30.16b, v30.16b, v15.16b+ eor x27, x0, x14, ror #10+ eor x29, x16, x28, ror #63+ eor v30.16b, v30.16b, v20.16b+ eor x26, x26, x6, ror #51+ eor v29.16b, v1.16b, v6.16b+ eor x30, x23, x22, ror #50+ eor v29.16b, v29.16b, v11.16b+ eor x0, x26, x10, ror #31+ eor x29, x29, x19, ror #37+ eor v29.16b, v29.16b, v16.16b+ eor x27, x27, x12, ror #5+ eor v29.16b, v29.16b, v21.16b+ eor x30, x30, x24, ror #34+ eor x0, x0, x7, ror #27+ eor v28.16b, v2.16b, v7.16b+ eor x26, x30, x21, ror #26+ eor v28.16b, v28.16b, v12.16b+ eor x26, x26, x25, ror #15+ eor v28.16b, v28.16b, v17.16b+ ror x30, x27, #0x3e+ eor x30, x30, x26, ror #57+ eor v28.16b, v28.16b, v22.16b+ ror x26, x26, #0x3a+ eor v27.16b, v3.16b, v8.16b+ eor x16, x30, x16+ eor v27.16b, v27.16b, v13.16b+ eor x28, x30, x28, ror #63+ str x28, [sp, #0xd0]+ eor v27.16b, v27.16b, v18.16b+ eor x29, x29, x17, ror #36+ eor v27.16b, v27.16b, v23.16b+ eor x28, x1, x2, ror #61+ eor x19, x30, x19, ror #37+ eor v26.16b, v4.16b, v9.16b+ eor x29, x29, x20, ror #2+ eor v26.16b, v26.16b, v14.16b+ eor x28, x28, x4, ror #54+ eor v26.16b, v26.16b, v19.16b+ eor x26, x26, x0, ror #55+ eor x28, x28, x3, ror #39+ eor v26.16b, v26.16b, v24.16b+ eor x28, x28, x5, ror #25+ add v31.2d, v28.2d, v28.2d+ ror x0, x0, #0x38+ eor x0, x0, x29, ror #63+ sri v31.2d, v28.2d, #0x3f+ eor x27, x28, x27, ror #61+ eor v25.16b, v31.16b, v30.16b+ eor x13, x0, x13, ror #46+ add v31.2d, v26.2d, v26.2d+ eor x28, x29, x28, ror #63+ eor x29, x30, x20, ror #2+ sri v31.2d, v26.2d, #0x3f+ eor x20, x26, x3, ror #39+ eor v28.16b, v31.16b, v28.16b+ eor x11, x0, x11, ror #50+ add v31.2d, v29.2d, v29.2d+ eor x25, x28, x25, ror #9+ eor x3, x28, x21, ror #20+ sri v31.2d, v29.2d, #0x3f+ eor x21, x26, x1+ eor v26.16b, v31.16b, v26.16b+ eor x9, x27, x9, ror #49+ eor x24, x28, x24, ror #28+ add v31.2d, v27.2d, v27.2d+ eor x1, x30, x17, ror #36+ sri v31.2d, v27.2d, #0x3f+ eor x14, x0, x14, ror #8+ eor v29.16b, v31.16b, v29.16b+ eor x22, x28, x22, ror #44+ eor x8, x27, x8, ror #56+ add v31.2d, v30.2d, v30.2d+ eor x17, x27, x7, ror #19+ sri v31.2d, v30.2d, #0x3f+ eor x15, x0, x15, ror #62+ bic x7, x20, x22, ror #47+ eor v27.16b, v31.16b, v27.16b+ eor x4, x26, x4, ror #54+ eor v30.16b, v0.16b, v26.16b+ eor x0, x0, x12, ror #3+ eor v31.16b, v2.16b, v29.16b+ eor x28, x28, x23, ror #58+ eor x23, x26, x2, ror #61+ shl v0.2d, v31.2d, #0x3e+ eor x26, x26, x5, ror #25+ sri v0.2d, v31.2d, #0x2+ eor x2, x7, x16, ror #39+ eor v31.16b, v12.16b, v29.16b+ bic x7, x9, x20, ror #42+ bic x30, x15, x9, ror #16+ shl v2.2d, v31.2d, #0x2b+ eor x7, x7, x22, ror #25+ sri v2.2d, v31.2d, #0x15+ eor x12, x30, x20, ror #58+ bic x20, x22, x16, ror #56+ eor v31.16b, v13.16b, v28.16b+ eor x30, x27, x6, ror #43+ shl v12.2d, v31.2d, #0x19+ eor x22, x20, x15, ror #23+ sri v12.2d, v31.2d, #0x27+ bic x6, x19, x13, ror #42+ eor x6, x6, x17, ror #41+ eor v31.16b, v19.16b, v27.16b+ bic x5, x13, x17, ror #63+ shl v13.2d, v31.2d, #0x8+ eor x5, x21, x5, ror #21+ sri v13.2d, v31.2d, #0x38+ bic x17, x17, x21, ror #44+ eor x27, x27, x10, ror #23+ eor v31.16b, v23.16b, v28.16b+ bic x21, x21, x25, ror #50+ shl v19.2d, v31.2d, #0x38+ bic x20, x27, x4, ror #25+ bic x10, x16, x15, ror #31+ sri v19.2d, v31.2d, #0x8+ eor x16, x21, x19, ror #43+ eor v31.16b, v15.16b, v26.16b+ eor x21, x17, x25, ror #30+ shl v23.2d, v31.2d, #0x29+ bic x19, x25, x19, ror #57+ ldr x25, [sp, #0x18]+ sri v23.2d, v31.2d, #0x17+ eor x17, x10, x9, ror #47+ eor v31.16b, v1.16b, v25.16b+ ldr x9, [sp, #0x8]+ eor x15, x20, x28, ror #27+ shl v15.2d, v31.2d, #0x1+ bic x20, x4, x28, ror #2+ sri v15.2d, v31.2d, #0x3f+ eor x10, x20, x1, ror #50+ eor v31.16b, v8.16b, v28.16b+ bic x20, x11, x27, ror #60+ eor x20, x20, x4, ror #21+ shl v1.2d, v31.2d, #0x37+ bic x4, x28, x1, ror #48+ sri v1.2d, v31.2d, #0x9+ bic x1, x1, x11, ror #57+ eor v31.16b, v16.16b, v25.16b+ ldr x28, [x9, x25, lsl #3]+ ldr x9, [sp, #0xd0]+ shl v8.2d, v31.2d, #0x2d+ add x25, x25, #0x1+ sri v8.2d, v31.2d, #0x13+ str x25, [sp, #0x18]+ cmp x25, #0x17+ eor v31.16b, v7.16b, v29.16b+ eor x25, x1, x27, ror #53+ shl v16.2d, v31.2d, #0x6+ bic x27, x30, x26, ror #47+ sri v16.2d, v31.2d, #0x3a+ eor x1, x5, x28+ eor x5, x4, x11, ror #41+ eor v31.16b, v10.16b, v26.16b+ eor x11, x19, x13, ror #35+ shl v7.2d, v31.2d, #0x3+ bic x13, x26, x24, ror #10+ eor x28, x27, x24, ror #57+ sri v7.2d, v31.2d, #0x3d+ bic x27, x24, x9, ror #47+ eor v31.16b, v3.16b, v28.16b+ bic x19, x23, x3, ror #9+ shl v10.2d, v31.2d, #0x1c+ bic x4, x29, x14, ror #41+ eor x24, x19, x29, ror #44+ sri v10.2d, v31.2d, #0x24+ bic x29, x3, x29, ror #35+ eor v31.16b, v18.16b, v28.16b+ eor x13, x13, x9, ror #57+ shl v3.2d, v31.2d, #0x15+ eor x19, x29, x14, ror #12+ bic x29, x9, x0, ror #19+ sri v3.2d, v31.2d, #0x2b+ bic x14, x14, x8, ror #5+ eor v31.16b, v17.16b, v29.16b+ eor x9, x14, x23, ror #43+ eor x14, x4, x8, ror #46+ shl v18.2d, v31.2d, #0xf+ bic x23, x8, x23, ror #38+ sri v18.2d, v31.2d, #0x31+ eor x8, x27, x0, ror #2+ eor v31.16b, v11.16b, v25.16b+ eor x4, x23, x3, ror #47+ bic x3, x0, x30, ror #5+ shl v17.2d, v31.2d, #0xa+ eor x23, x3, x26, ror #52+ sri v17.2d, v31.2d, #0x36+ eor x3, x29, x30, ror #24+ eor x0, x15, x11, ror #52+ eor v31.16b, v9.16b, v27.16b+ eor x0, x0, x13, ror #48+ shl v11.2d, v31.2d, #0x14+ eor x26, x8, x9, ror #57+ sri v11.2d, v31.2d, #0x2c+ eor x27, x0, x14, ror #10+ eor x29, x16, x28, ror #63+ eor v31.16b, v22.16b, v29.16b+ eor x26, x26, x6, ror #51+ shl v9.2d, v31.2d, #0x3d+ eor x30, x23, x22, ror #50+ sri v9.2d, v31.2d, #0x3+ eor x0, x26, x10, ror #31+ eor x29, x29, x19, ror #37+ eor v31.16b, v14.16b, v27.16b+ eor x27, x27, x12, ror #5+ shl v22.2d, v31.2d, #0x27+ eor x30, x30, x24, ror #34+ eor x0, x0, x7, ror #27+ sri v22.2d, v31.2d, #0x19+ eor x26, x30, x21, ror #26+ eor v31.16b, v20.16b, v26.16b+ eor x26, x26, x25, ror #15+ shl v14.2d, v31.2d, #0x12+ ror x30, x27, #0x3e+ eor x30, x30, x26, ror #57+ sri v14.2d, v31.2d, #0x2e+ ror x26, x26, #0x3a+ eor v31.16b, v4.16b, v27.16b+ eor x16, x30, x16+ shl v20.2d, v31.2d, #0x1b+ eor x28, x30, x28, ror #63+ str x28, [sp, #0xd0]+ sri v20.2d, v31.2d, #0x25+ eor x29, x29, x17, ror #36+ eor v31.16b, v24.16b, v27.16b+ eor x28, x1, x2, ror #61+ eor x19, x30, x19, ror #37+ shl v4.2d, v31.2d, #0xe+ eor x29, x29, x20, ror #2+ sri v4.2d, v31.2d, #0x32+ eor x28, x28, x4, ror #54+ eor v31.16b, v21.16b, v25.16b+ eor x26, x26, x0, ror #55+ eor x28, x28, x3, ror #39+ shl v24.2d, v31.2d, #0x2+ eor x28, x28, x5, ror #25+ sri v24.2d, v31.2d, #0x3e+ ror x0, x0, #0x38+ eor x0, x0, x29, ror #63+ eor v31.16b, v5.16b, v26.16b+ eor x27, x28, x27, ror #61+ shl v21.2d, v31.2d, #0x24+ eor x13, x0, x13, ror #46+ sri v21.2d, v31.2d, #0x1c+ eor x28, x29, x28, ror #63+ eor x29, x30, x20, ror #2+ eor v31.16b, v6.16b, v25.16b+ eor x20, x26, x3, ror #39+ shl v27.2d, v31.2d, #0x2c+ eor x11, x0, x11, ror #50+ sri v27.2d, v31.2d, #0x14+ eor x25, x28, x25, ror #9+ eor x3, x28, x21, ror #20+ bic v31.16b, v7.16b, v11.16b+ eor x21, x26, x1+ eor v5.16b, v31.16b, v10.16b+ eor x9, x27, x9, ror #49+ eor x24, x28, x24, ror #28+ bic v31.16b, v8.16b, v7.16b+ eor x1, x30, x17, ror #36+ eor v6.16b, v31.16b, v11.16b+ eor x14, x0, x14, ror #8+ bic v31.16b, v9.16b, v8.16b+ eor x22, x28, x22, ror #44+ eor x8, x27, x8, ror #56+ eor v7.16b, v31.16b, v7.16b+ eor x17, x27, x7, ror #19+ bic v31.16b, v10.16b, v9.16b+ eor x15, x0, x15, ror #62+ bic x7, x20, x22, ror #47+ eor v8.16b, v31.16b, v8.16b+ eor x4, x26, x4, ror #54+ bic v31.16b, v11.16b, v10.16b+ eor x0, x0, x12, ror #3+ eor v9.16b, v31.16b, v9.16b+ eor x28, x28, x23, ror #58+ eor x23, x26, x2, ror #61+ bic v31.16b, v12.16b, v16.16b+ eor x26, x26, x5, ror #25+ eor v10.16b, v31.16b, v15.16b+ eor x2, x7, x16, ror #39+ bic v31.16b, v13.16b, v12.16b+ bic x7, x9, x20, ror #42+ bic x30, x15, x9, ror #16+ eor v11.16b, v31.16b, v16.16b+ eor x7, x7, x22, ror #25+ bic v31.16b, v14.16b, v13.16b+ eor x12, x30, x20, ror #58+ bic x20, x22, x16, ror #56+ eor v12.16b, v31.16b, v12.16b+ eor x30, x27, x6, ror #43+ bic v31.16b, v15.16b, v14.16b+ eor x22, x20, x15, ror #23+ eor v13.16b, v31.16b, v13.16b+ bic x6, x19, x13, ror #42+ eor x6, x6, x17, ror #41+ bic v31.16b, v16.16b, v15.16b+ bic x5, x13, x17, ror #63+ eor v14.16b, v31.16b, v14.16b+ eor x5, x21, x5, ror #21+ bic v31.16b, v17.16b, v21.16b+ bic x17, x17, x21, ror #44+ eor x27, x27, x10, ror #23+ eor v15.16b, v31.16b, v20.16b+ bic x21, x21, x25, ror #50+ bic v31.16b, v18.16b, v17.16b+ bic x20, x27, x4, ror #25+ bic x10, x16, x15, ror #31+ eor v16.16b, v31.16b, v21.16b+ eor x16, x21, x19, ror #43+ bic v31.16b, v19.16b, v18.16b+ eor x21, x17, x25, ror #30+ eor v17.16b, v31.16b, v17.16b+ bic x19, x25, x19, ror #57+ ldr x25, [sp, #0x18]+ bic v31.16b, v20.16b, v19.16b+ eor x17, x10, x9, ror #47+ eor v18.16b, v31.16b, v18.16b+ ldr x9, [sp, #0x8]+ eor x15, x20, x28, ror #27+ bic v31.16b, v21.16b, v20.16b+ bic x20, x4, x28, ror #2+ eor v19.16b, v31.16b, v19.16b+ eor x10, x20, x1, ror #50+ bic v31.16b, v22.16b, v1.16b+ bic x20, x11, x27, ror #60+ eor x20, x20, x4, ror #21+ eor v20.16b, v31.16b, v0.16b+ bic x4, x28, x1, ror #48+ bic v31.16b, v23.16b, v22.16b+ bic x1, x1, x11, ror #57+ eor v21.16b, v31.16b, v1.16b+ ldr x28, [x9, x25, lsl #3]+ ldr x9, [sp, #0xd0]+ bic v31.16b, v24.16b, v23.16b+ add x25, x25, #0x1+ eor v22.16b, v31.16b, v22.16b+ str x25, [sp, #0x18]+ cmp x25, #0x17+ bic v31.16b, v0.16b, v24.16b+ eor x25, x1, x27, ror #53+ eor v23.16b, v31.16b, v23.16b+ bic x27, x30, x26, ror #47+ bic v31.16b, v1.16b, v0.16b+ eor x1, x5, x28+ eor x5, x4, x11, ror #41+ eor v24.16b, v31.16b, v24.16b+ eor x11, x19, x13, ror #35+ bic v31.16b, v2.16b, v27.16b+ bic x13, x26, x24, ror #10+ eor x28, x27, x24, ror #57+ eor v0.16b, v31.16b, v30.16b+ bic x27, x24, x9, ror #47+ bic v31.16b, v3.16b, v2.16b+ bic x19, x23, x3, ror #9+ eor v1.16b, v31.16b, v27.16b+ bic x4, x29, x14, ror #41+ eor x24, x19, x29, ror #44+ bic v31.16b, v4.16b, v3.16b+ bic x29, x3, x29, ror #35+ eor v2.16b, v31.16b, v2.16b+ eor x13, x13, x9, ror #57+ bic v31.16b, v30.16b, v4.16b+ eor x19, x29, x14, ror #12+ bic x29, x9, x0, ror #19+ eor v3.16b, v31.16b, v3.16b+ bic x14, x14, x8, ror #5+ bic v31.16b, v27.16b, v30.16b+ eor x9, x14, x23, ror #43+ eor x14, x4, x8, ror #46+ eor v4.16b, v31.16b, v4.16b+ bic x23, x8, x23, ror #38+ eor x8, x27, x0, ror #2+ eor x4, x23, x3, ror #47+ bic x3, x0, x30, ror #5+ eor x23, x3, x26, ror #52+ eor x3, x29, x30, ror #24+ ldr x30, [sp, #0x10]+ ld1r { v28.2d }, [x30], #8+ str x30, [sp, #0x10]+ eor v0.16b, v0.16b, v28.16b++Lmld_keccak_f1600_x4_v8a_scalar_hybrid_loop_end:+ b.le Lmld_keccak_f1600_x4_v8a_scalar_hybrid_loop+ ror x2, x2, #0x3d+ ror x3, x3, #0x27+ ror x4, x4, #0x36+ ror x5, x5, #0x19+ ror x6, x6, #0x2b+ ror x7, x7, #0x13+ ror x8, x8, #0x38+ ror x9, x9, #0x31+ ror x10, x10, #0x17+ ror x11, x11, #0x32+ ror x12, x12, #0x3+ ror x13, x13, #0x2e+ ror x14, x14, #0x8+ ror x15, x15, #0x3e+ ror x17, x17, #0x24+ ror x28, x28, #0x3f+ ror x19, x19, #0x25+ ror x20, x20, #0x2+ ror x21, x21, #0x14+ ror x22, x22, #0x2c+ ror x23, x23, #0x3a+ ror x24, x24, #0x1c+ ror x25, x25, #0x9+ ldr x30, [sp, #0x20]+ cmp x30, #0x1+ b.eq Lmld_keccak_f1600_x4_v8a_scalar_hybrid_done+ mov x30, #0x1 // =1+ str x30, [sp, #0x20]+ ldr x0, [sp]+ add x0, x0, #0x190+ stp x1, x6, [x0]+ stp x11, x16, [x0, #0x10]+ stp x21, x2, [x0, #0x20]+ stp x7, x12, [x0, #0x30]+ stp x17, x22, [x0, #0x40]+ stp x3, x8, [x0, #0x50]+ stp x13, x28, [x0, #0x60]+ stp x23, x4, [x0, #0x70]+ stp x9, x14, [x0, #0x80]+ stp x19, x24, [x0, #0x90]+ stp x5, x10, [x0, #0xa0]+ stp x15, x20, [x0, #0xb0]+ str x25, [x0, #0xc0]+ sub x0, x0, #0x190+ add x0, x0, #0x258+ ldp x1, x6, [x0]+ ldp x11, x16, [x0, #0x10]+ ldp x21, x2, [x0, #0x20]+ ldp x7, x12, [x0, #0x30]+ ldp x17, x22, [x0, #0x40]+ ldp x3, x8, [x0, #0x50]+ ldp x13, x28, [x0, #0x60]+ ldp x23, x4, [x0, #0x70]+ ldp x9, x14, [x0, #0x80]+ ldp x19, x24, [x0, #0x90]+ ldp x5, x10, [x0, #0xa0]+ ldp x15, x20, [x0, #0xb0]+ ldr x25, [x0, #0xc0]+ sub x0, x0, #0x258+ b Lmld_keccak_f1600_x4_v8a_scalar_hybrid_initial++Lmld_keccak_f1600_x4_v8a_scalar_hybrid_done:+ ldr x0, [sp]+ add x0, x0, #0x258+ stp x1, x6, [x0]+ stp x11, x16, [x0, #0x10]+ stp x21, x2, [x0, #0x20]+ stp x7, x12, [x0, #0x30]+ stp x17, x22, [x0, #0x40]+ stp x3, x8, [x0, #0x50]+ stp x13, x28, [x0, #0x60]+ stp x23, x4, [x0, #0x70]+ stp x9, x14, [x0, #0x80]+ stp x19, x24, [x0, #0x90]+ stp x5, x10, [x0, #0xa0]+ stp x15, x20, [x0, #0xb0]+ str x25, [x0, #0xc0]+ sub x0, x0, #0x258+ add x4, x0, #0xc8+ trn1 v25.2d, v0.2d, v1.2d+ trn1 v26.2d, v2.2d, v3.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v0.2d, v1.2d+ trn2 v28.2d, v2.2d, v3.2d+ st1 { v27.2d, v28.2d }, [x4], #32+ trn1 v25.2d, v4.2d, v5.2d+ trn1 v26.2d, v6.2d, v7.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v4.2d, v5.2d+ trn2 v28.2d, v6.2d, v7.2d+ st1 { v27.2d, v28.2d }, [x4], #32+ trn1 v25.2d, v8.2d, v9.2d+ trn1 v26.2d, v10.2d, v11.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v8.2d, v9.2d+ trn2 v28.2d, v10.2d, v11.2d+ st1 { v27.2d, v28.2d }, [x4], #32+ trn1 v25.2d, v12.2d, v13.2d+ trn1 v26.2d, v14.2d, v15.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v12.2d, v13.2d+ trn2 v28.2d, v14.2d, v15.2d+ st1 { v27.2d, v28.2d }, [x4], #32+ trn1 v25.2d, v16.2d, v17.2d+ trn1 v26.2d, v18.2d, v19.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v16.2d, v17.2d+ trn2 v28.2d, v18.2d, v19.2d+ st1 { v27.2d, v28.2d }, [x4], #32+ trn1 v25.2d, v20.2d, v21.2d+ trn1 v26.2d, v22.2d, v23.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v20.2d, v21.2d+ trn2 v28.2d, v22.2d, v23.2d+ st1 { v27.2d, v28.2d }, [x4], #32+ str d24, [x0]+ trn2 v25.2d, v24.2d, v24.2d+ str d25, [x4]+ ldp d8, d9, [sp, #0x90]+ .cfi_restore d8+ .cfi_restore d9+ ldp d10, d11, [sp, #0xa0]+ .cfi_restore d10+ .cfi_restore d11+ ldp d12, d13, [sp, #0xb0]+ .cfi_restore d12+ .cfi_restore d13+ ldp d14, d15, [sp, #0xc0]+ .cfi_restore d14+ .cfi_restore d15+ ldp x19, x20, [sp, #0x30]+ .cfi_restore x19+ .cfi_restore x20+ ldp x21, x22, [sp, #0x40]+ .cfi_restore x21+ .cfi_restore x22+ ldp x23, x24, [sp, #0x50]+ .cfi_restore x23+ .cfi_restore x24+ ldp x25, x26, [sp, #0x60]+ .cfi_restore x25+ .cfi_restore x26+ ldp x27, x28, [sp, #0x70]+ .cfi_restore x27+ .cfi_restore x28+ ldp x29, x30, [sp, #0x80]+ .cfi_restore x29+ .cfi_restore x30+ add sp, sp, #0xe0+ .cfi_adjust_cfa_offset -0xe0+ ret+ .cfi_endproc++MLD_ASM_FN_SIZE(keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm)++#endif /* MLD_FIPS202_AARCH64_NEED_X4_V8A_SCALAR_HYBRID && \+ !MLD_CONFIG_MULTILEVEL_NO_SHARED */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
@@ -0,0 +1,990 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * Copyright (c) 2021-2022 Arm Limited+ * Copyright (c) 2022 Matthias Kannwischer+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++// Author: Hanno Becker <hannobecker@posteo.de>+// Author: Matthias Kannwischer <matthias@kannwischer.eu>++/*yaml+ Name: keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm+ Description: AArch64 hybrid scalar/vector implementation of Keccak-f[1600] permutation for four sequential states with ARMv8.4-A optimizations+ Signature: void mld_keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm(uint64_t state[100], const uint64_t rc[24])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON, SHA3]+ x0:+ type: buffer+ size_bytes: 800+ permissions: read/write+ c_parameter: uint64_t state[100]+ description: Four sequential Keccak states (state0[25], state1[25], state2[25], state3[25])+ x1:+ type: buffer+ size_bytes: 192+ permissions: read-only+ c_parameter: uint64_t const *rc+ description: Round constants (24 x uint64_t)+ Stack:+ bytes: 224+ description: register preservation and temporary storage+*/++#include "../../../../common.h"+#if defined(MLD_FIPS202_AARCH64_NEED_X4_V8A_V84A_SCALAR_HYBRID) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++#if defined(__ARM_FEATURE_SHA3)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/fips202/aarch64/src/keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm)+MLD_ASM_FN_SYMBOL(keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm)++ .cfi_startproc+ sub sp, sp, #0xe0+ .cfi_adjust_cfa_offset 0xe0+ stp x19, x20, [sp, #0x30]+ .cfi_rel_offset x19, 0x30+ .cfi_rel_offset x20, 0x38+ stp x21, x22, [sp, #0x40]+ .cfi_rel_offset x21, 0x40+ .cfi_rel_offset x22, 0x48+ stp x23, x24, [sp, #0x50]+ .cfi_rel_offset x23, 0x50+ .cfi_rel_offset x24, 0x58+ stp x25, x26, [sp, #0x60]+ .cfi_rel_offset x25, 0x60+ .cfi_rel_offset x26, 0x68+ stp x27, x28, [sp, #0x70]+ .cfi_rel_offset x27, 0x70+ .cfi_rel_offset x28, 0x78+ stp x29, x30, [sp, #0x80]+ .cfi_rel_offset x29, 0x80+ .cfi_rel_offset x30, 0x88+ stp d8, d9, [sp, #0x90]+ .cfi_rel_offset d8, 0x90+ .cfi_rel_offset d9, 0x98+ stp d10, d11, [sp, #0xa0]+ .cfi_rel_offset d10, 0xa0+ .cfi_rel_offset d11, 0xa8+ stp d12, d13, [sp, #0xb0]+ .cfi_rel_offset d12, 0xb0+ .cfi_rel_offset d13, 0xb8+ stp d14, d15, [sp, #0xc0]+ .cfi_rel_offset d14, 0xc0+ .cfi_rel_offset d15, 0xc8+ mov x29, x1+ mov x30, #0x0 // =0+ str x30, [sp, #0x20]+ str x29, [sp, #0x8]+ str x29, [sp, #0x10]+ str x0, [sp]+ add x4, x0, #0xc8+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x4], #32+ trn1 v0.2d, v25.2d, v27.2d+ trn2 v1.2d, v25.2d, v27.2d+ trn1 v2.2d, v26.2d, v28.2d+ trn2 v3.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x4], #32+ trn1 v4.2d, v25.2d, v27.2d+ trn2 v5.2d, v25.2d, v27.2d+ trn1 v6.2d, v26.2d, v28.2d+ trn2 v7.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x4], #32+ trn1 v8.2d, v25.2d, v27.2d+ trn2 v9.2d, v25.2d, v27.2d+ trn1 v10.2d, v26.2d, v28.2d+ trn2 v11.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x4], #32+ trn1 v12.2d, v25.2d, v27.2d+ trn2 v13.2d, v25.2d, v27.2d+ trn1 v14.2d, v26.2d, v28.2d+ trn2 v15.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x4], #32+ trn1 v16.2d, v25.2d, v27.2d+ trn2 v17.2d, v25.2d, v27.2d+ trn1 v18.2d, v26.2d, v28.2d+ trn2 v19.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x4], #32+ trn1 v20.2d, v25.2d, v27.2d+ trn2 v21.2d, v25.2d, v27.2d+ trn1 v22.2d, v26.2d, v28.2d+ trn2 v23.2d, v26.2d, v28.2d+ ldr d25, [x0]+ ldr d27, [x4]+ trn1 v24.2d, v25.2d, v27.2d+ sub x0, x0, #0xc0+ add x0, x0, #0x190+ ldp x1, x6, [x0]+ ldp x11, x16, [x0, #0x10]+ ldp x21, x2, [x0, #0x20]+ ldp x7, x12, [x0, #0x30]+ ldp x17, x22, [x0, #0x40]+ ldp x3, x8, [x0, #0x50]+ ldp x13, x28, [x0, #0x60]+ ldp x23, x4, [x0, #0x70]+ ldp x9, x14, [x0, #0x80]+ ldp x19, x24, [x0, #0x90]+ ldp x5, x10, [x0, #0xa0]+ ldp x15, x20, [x0, #0xb0]+ ldr x25, [x0, #0xc0]+ sub x0, x0, #0x190++Lmld_keccak_f1600_x4_v8a_v84a_scalar_hybrid_initial:+ eor x30, x24, x25+ eor x27, x9, x10+ eor3 v30.16b, v0.16b, v5.16b, v10.16b+ eor v30.16b, v30.16b, v15.16b+ eor x0, x30, x21+ eor x26, x27, x6+ eor v30.16b, v30.16b, v20.16b+ eor x27, x26, x7+ eor x29, x0, x22+ eor3 v29.16b, v1.16b, v6.16b, v11.16b+ eor x26, x29, x23+ eor x29, x4, x5+ eor v29.16b, v29.16b, v16.16b+ eor x30, x29, x1+ eor x0, x27, x8+ eor v29.16b, v29.16b, v21.16b+ eor x29, x30, x2+ eor x30, x19, x20+ eor3 v28.16b, v2.16b, v7.16b, v12.16b+ eor x30, x30, x16+ eor x27, x26, x0, ror #63+ eor v28.16b, v28.16b, v17.16b+ eor x4, x4, x27+ eor x30, x30, x17+ eor v28.16b, v28.16b, v22.16b+ eor x30, x30, x28+ eor x29, x29, x3+ eor3 v27.16b, v3.16b, v8.16b, v13.16b+ eor x0, x0, x30, ror #63+ eor x30, x30, x29, ror #63+ eor v27.16b, v27.16b, v18.16b+ eor x22, x22, x30+ eor v27.16b, v27.16b, v23.16b+ eor x23, x23, x30+ str x23, [sp, #0xd0]+ eor3 v26.16b, v4.16b, v9.16b, v14.16b+ eor x23, x14, x15+ eor x14, x14, x0+ eor v26.16b, v26.16b, v19.16b+ eor x23, x23, x11+ eor x15, x15, x0+ eor v26.16b, v26.16b, v24.16b+ eor x1, x1, x27+ eor x23, x23, x12+ rax1 v25.2d, v30.2d, v28.2d+ eor x23, x23, x13+ eor x11, x11, x0+ add v31.2d, v26.2d, v26.2d+ eor x29, x29, x23, ror #63+ eor x23, x23, x26, ror #63+ sri v31.2d, v26.2d, #0x3f+ eor x26, x13, x0+ eor x13, x28, x23+ eor v28.16b, v31.16b, v28.16b+ eor x28, x24, x30+ eor x24, x16, x23+ rax1 v26.2d, v26.2d, v29.2d+ eor x16, x21, x30+ eor x21, x25, x30+ add v31.2d, v27.2d, v27.2d+ eor x30, x19, x23+ sri v31.2d, v27.2d, #0x3f+ eor x19, x20, x23+ eor x20, x17, x23+ eor v29.16b, v31.16b, v29.16b+ eor x17, x12, x0+ eor x0, x2, x27+ rax1 v27.2d, v27.2d, v30.2d+ eor x2, x6, x29+ eor x6, x8, x29+ eor v30.16b, v0.16b, v26.16b+ bic x8, x28, x13, ror #47+ eor x12, x3, x27+ eor v31.16b, v2.16b, v29.16b+ bic x3, x13, x17, ror #19+ eor x5, x5, x27+ shl v0.2d, v31.2d, #0x3e+ ldr x27, [sp, #0xd0]+ bic x25, x17, x2, ror #5+ sri v0.2d, v31.2d, #0x2+ eor x9, x9, x29+ eor x23, x25, x5, ror #52+ xar v2.2d, v12.2d, v29.2d, #0x15+ eor x3, x3, x2, ror #24+ eor x8, x8, x17, ror #2+ eor v31.16b, v13.16b, v28.16b+ eor x17, x10, x29+ bic x25, x12, x22, ror #47+ shl v12.2d, v31.2d, #0x19+ eor x29, x7, x29+ bic x10, x4, x27, ror #2+ sri v12.2d, v31.2d, #0x27+ bic x7, x5, x28, ror #10+ xar v13.2d, v19.2d, v27.2d, #0x38+ eor x10, x10, x20, ror #50+ eor x13, x7, x13, ror #57+ eor v31.16b, v23.16b, v28.16b+ bic x7, x2, x5, ror #47+ eor x2, x25, x24, ror #39+ shl v19.2d, v31.2d, #0x38+ bic x25, x20, x11, ror #57+ bic x5, x17, x4, ror #25+ sri v19.2d, v31.2d, #0x8+ eor x25, x25, x17, ror #53+ bic x17, x11, x17, ror #60+ xar v23.2d, v15.2d, v26.2d, #0x17+ eor x28, x7, x28, ror #57+ bic x7, x9, x12, ror #42+ eor v31.16b, v1.16b, v25.16b+ eor x7, x7, x22, ror #25+ bic x22, x22, x24, ror #56+ shl v15.2d, v31.2d, #0x1+ bic x24, x24, x15, ror #31+ eor x22, x22, x15, ror #23+ sri v15.2d, v31.2d, #0x3f+ bic x20, x27, x20, ror #48+ bic x15, x15, x9, ror #16+ xar v1.2d, v8.2d, v28.2d, #0x9+ eor x12, x15, x12, ror #58+ eor x15, x5, x27, ror #27+ eor v31.16b, v16.16b, v25.16b+ eor x5, x20, x11, ror #41+ shl v8.2d, v31.2d, #0x2d+ ldr x11, [sp, #0x8]+ eor x20, x17, x4, ror #21+ sri v8.2d, v31.2d, #0x13+ eor x17, x24, x9, ror #47+ mov x24, #0x1 // =1+ xar v16.2d, v7.2d, v29.2d, #0x3a+ bic x9, x0, x16, ror #9+ str x24, [sp, #0x18]+ eor v31.16b, v10.16b, v26.16b+ bic x24, x29, x1, ror #44+ bic x27, x1, x21, ror #50+ shl v7.2d, v31.2d, #0x3+ bic x4, x26, x29, ror #63+ eor x1, x1, x4, ror #21+ sri v7.2d, v31.2d, #0x3d+ ldr x11, [x11]+ bic x4, x21, x30, ror #57+ xar v10.2d, v3.2d, v28.2d, #0x24+ eor x21, x24, x21, ror #30+ eor x24, x9, x19, ror #44+ eor v31.16b, v18.16b, v28.16b+ bic x9, x14, x6, ror #5+ eor x9, x9, x0, ror #43+ shl v3.2d, v31.2d, #0x15+ bic x0, x6, x0, ror #38+ eor x1, x1, x11+ sri v3.2d, v31.2d, #0x2b+ eor x11, x4, x26, ror #35+ eor x4, x0, x16, ror #47+ xar v18.2d, v17.2d, v29.2d, #0x31+ bic x0, x16, x19, ror #35+ eor v31.16b, v11.16b, v25.16b+ eor x16, x27, x30, ror #43+ bic x27, x30, x26, ror #42+ shl v17.2d, v31.2d, #0xa+ bic x26, x19, x14, ror #41+ eor x19, x0, x14, ror #12+ sri v17.2d, v31.2d, #0x36+ eor x14, x26, x6, ror #46+ eor x6, x27, x29, ror #41+ xar v11.2d, v9.2d, v27.2d, #0x2c+ eor x0, x15, x11, ror #52+ eor x0, x0, x13, ror #48+ eor v31.16b, v22.16b, v29.16b+ eor x26, x8, x9, ror #57+ eor x27, x0, x14, ror #10+ shl v9.2d, v31.2d, #0x3d+ eor x29, x16, x28, ror #63+ eor x26, x26, x6, ror #51+ sri v9.2d, v31.2d, #0x3+ eor x30, x23, x22, ror #50+ eor x0, x26, x10, ror #31+ xar v22.2d, v14.2d, v27.2d, #0x19+ eor x29, x29, x19, ror #37+ eor x27, x27, x12, ror #5+ eor v31.16b, v20.16b, v26.16b+ eor x30, x30, x24, ror #34+ eor x0, x0, x7, ror #27+ shl v14.2d, v31.2d, #0x12+ eor x26, x30, x21, ror #26+ sri v14.2d, v31.2d, #0x2e+ eor x26, x26, x25, ror #15+ ror x30, x27, #0x3e+ xar v20.2d, v4.2d, v27.2d, #0x25+ eor x30, x30, x26, ror #57+ ror x26, x26, #0x3a+ eor v31.16b, v24.16b, v27.16b+ eor x16, x30, x16+ eor x28, x30, x28, ror #63+ shl v4.2d, v31.2d, #0xe+ str x28, [sp, #0xd0]+ eor x29, x29, x17, ror #36+ sri v4.2d, v31.2d, #0x32+ eor x28, x1, x2, ror #61+ eor x19, x30, x19, ror #37+ xar v24.2d, v21.2d, v25.2d, #0x3e+ eor x29, x29, x20, ror #2+ eor x28, x28, x4, ror #54+ eor v31.16b, v5.16b, v26.16b+ eor x26, x26, x0, ror #55+ eor x28, x28, x3, ror #39+ shl v21.2d, v31.2d, #0x24+ eor x28, x28, x5, ror #25+ ror x0, x0, #0x38+ sri v21.2d, v31.2d, #0x1c+ eor x0, x0, x29, ror #63+ eor x27, x28, x27, ror #61+ xar v27.2d, v6.2d, v25.2d, #0x14+ eor x13, x0, x13, ror #46+ eor x28, x29, x28, ror #63+ bic v31.16b, v7.16b, v11.16b+ eor x29, x30, x20, ror #2+ eor v5.16b, v31.16b, v10.16b+ eor x20, x26, x3, ror #39+ eor x11, x0, x11, ror #50+ bcax v6.16b, v11.16b, v8.16b, v7.16b+ eor x25, x28, x25, ror #9+ eor x3, x28, x21, ror #20+ bic v31.16b, v9.16b, v8.16b+ eor x21, x26, x1+ eor x9, x27, x9, ror #49+ eor v7.16b, v31.16b, v7.16b+ eor x24, x28, x24, ror #28+ eor x1, x30, x17, ror #36+ bcax v8.16b, v8.16b, v10.16b, v9.16b+ eor x14, x0, x14, ror #8+ eor x22, x28, x22, ror #44+ bic v31.16b, v11.16b, v10.16b+ eor x8, x27, x8, ror #56+ eor x17, x27, x7, ror #19+ eor v9.16b, v31.16b, v9.16b+ eor x15, x0, x15, ror #62+ bic x7, x20, x22, ror #47+ bcax v10.16b, v15.16b, v12.16b, v16.16b+ eor x4, x26, x4, ror #54+ eor x0, x0, x12, ror #3+ bic v31.16b, v13.16b, v12.16b+ eor x28, x28, x23, ror #58+ eor x23, x26, x2, ror #61+ eor v11.16b, v31.16b, v16.16b+ eor x26, x26, x5, ror #25+ bcax v12.16b, v12.16b, v14.16b, v13.16b+ eor x2, x7, x16, ror #39+ bic x7, x9, x20, ror #42+ bic v31.16b, v15.16b, v14.16b+ bic x30, x15, x9, ror #16+ eor x7, x7, x22, ror #25+ eor v13.16b, v31.16b, v13.16b+ eor x12, x30, x20, ror #58+ bic x20, x22, x16, ror #56+ bic v31.16b, v16.16b, v15.16b+ eor x30, x27, x6, ror #43+ eor x22, x20, x15, ror #23+ eor v14.16b, v31.16b, v14.16b+ bic x6, x19, x13, ror #42+ eor x6, x6, x17, ror #41+ bcax v15.16b, v20.16b, v17.16b, v21.16b+ bic x5, x13, x17, ror #63+ eor x5, x21, x5, ror #21+ bic v31.16b, v18.16b, v17.16b+ bic x17, x17, x21, ror #44+ eor x27, x27, x10, ror #23+ eor v16.16b, v31.16b, v21.16b+ bic x21, x21, x25, ror #50+ bic x20, x27, x4, ror #25+ bcax v17.16b, v17.16b, v19.16b, v18.16b+ bic x10, x16, x15, ror #31+ eor x16, x21, x19, ror #43+ bic v31.16b, v20.16b, v19.16b+ eor x21, x17, x25, ror #30+ bic x19, x25, x19, ror #57+ eor v18.16b, v31.16b, v18.16b+ ldr x25, [sp, #0x18]+ bcax v19.16b, v19.16b, v21.16b, v20.16b+ eor x17, x10, x9, ror #47+ ldr x9, [sp, #0x8]+ bic v31.16b, v22.16b, v1.16b+ eor x15, x20, x28, ror #27+ bic x20, x4, x28, ror #2+ eor v20.16b, v31.16b, v0.16b+ eor x10, x20, x1, ror #50+ bic x20, x11, x27, ror #60+ bcax v21.16b, v1.16b, v23.16b, v22.16b+ eor x20, x20, x4, ror #21+ bic x4, x28, x1, ror #48+ bic v31.16b, v24.16b, v23.16b+ bic x1, x1, x11, ror #57+ ldr x28, [x9, x25, lsl #3]+ eor v22.16b, v31.16b, v22.16b+ ldr x9, [sp, #0xd0]+ add x25, x25, #0x1+ bcax v23.16b, v23.16b, v0.16b, v24.16b+ str x25, [sp, #0x18]+ cmp x25, #0x17+ bic v31.16b, v1.16b, v0.16b+ eor x25, x1, x27, ror #53+ bic x27, x30, x26, ror #47+ eor v24.16b, v31.16b, v24.16b+ eor x1, x5, x28+ eor x5, x4, x11, ror #41+ bcax v0.16b, v30.16b, v2.16b, v27.16b+ eor x11, x19, x13, ror #35+ bic v31.16b, v3.16b, v2.16b+ bic x13, x26, x24, ror #10+ eor x28, x27, x24, ror #57+ eor v1.16b, v31.16b, v27.16b+ bic x27, x24, x9, ror #47+ bic x19, x23, x3, ror #9+ bcax v2.16b, v2.16b, v4.16b, v3.16b+ bic x4, x29, x14, ror #41+ eor x24, x19, x29, ror #44+ bic v31.16b, v30.16b, v4.16b+ bic x29, x3, x29, ror #35+ eor x13, x13, x9, ror #57+ eor v3.16b, v31.16b, v3.16b+ eor x19, x29, x14, ror #12+ bic x29, x9, x0, ror #19+ bcax v4.16b, v4.16b, v27.16b, v30.16b+ bic x14, x14, x8, ror #5+ eor x9, x14, x23, ror #43+ eor x14, x4, x8, ror #46+ bic x23, x8, x23, ror #38+ eor x8, x27, x0, ror #2+ eor x4, x23, x3, ror #47+ bic x3, x0, x30, ror #5+ eor x23, x3, x26, ror #52+ eor x3, x29, x30, ror #24+ ldr x30, [sp, #0x10]+ ld1r { v28.2d }, [x30], #8+ str x30, [sp, #0x10]+ eor v0.16b, v0.16b, v28.16b++Lmld_keccak_f1600_x4_v8a_v84a_scalar_hybrid_loop:+ eor x0, x15, x11, ror #52+ eor x0, x0, x13, ror #48+ eor3 v30.16b, v0.16b, v5.16b, v10.16b+ eor v30.16b, v30.16b, v15.16b+ eor x26, x8, x9, ror #57+ eor x27, x0, x14, ror #10+ eor v30.16b, v30.16b, v20.16b+ eor x29, x16, x28, ror #63+ eor x26, x26, x6, ror #51+ eor3 v29.16b, v1.16b, v6.16b, v11.16b+ eor x30, x23, x22, ror #50+ eor x0, x26, x10, ror #31+ eor v29.16b, v29.16b, v16.16b+ eor x29, x29, x19, ror #37+ eor x27, x27, x12, ror #5+ eor v29.16b, v29.16b, v21.16b+ eor x30, x30, x24, ror #34+ eor x0, x0, x7, ror #27+ eor3 v28.16b, v2.16b, v7.16b, v12.16b+ eor x26, x30, x21, ror #26+ eor x26, x26, x25, ror #15+ eor v28.16b, v28.16b, v17.16b+ ror x30, x27, #0x3e+ eor x30, x30, x26, ror #57+ eor v28.16b, v28.16b, v22.16b+ ror x26, x26, #0x3a+ eor x16, x30, x16+ eor3 v27.16b, v3.16b, v8.16b, v13.16b+ eor x28, x30, x28, ror #63+ str x28, [sp, #0xd0]+ eor v27.16b, v27.16b, v18.16b+ eor x29, x29, x17, ror #36+ eor x28, x1, x2, ror #61+ eor v27.16b, v27.16b, v23.16b+ eor x19, x30, x19, ror #37+ eor x29, x29, x20, ror #2+ eor3 v26.16b, v4.16b, v9.16b, v14.16b+ eor x28, x28, x4, ror #54+ eor x26, x26, x0, ror #55+ eor v26.16b, v26.16b, v19.16b+ eor x28, x28, x3, ror #39+ eor x28, x28, x5, ror #25+ eor v26.16b, v26.16b, v24.16b+ ror x0, x0, #0x38+ eor x0, x0, x29, ror #63+ rax1 v25.2d, v30.2d, v28.2d+ eor x27, x28, x27, ror #61+ eor x13, x0, x13, ror #46+ add v31.2d, v26.2d, v26.2d+ eor x28, x29, x28, ror #63+ eor x29, x30, x20, ror #2+ sri v31.2d, v26.2d, #0x3f+ eor x20, x26, x3, ror #39+ eor x11, x0, x11, ror #50+ eor v28.16b, v31.16b, v28.16b+ eor x25, x28, x25, ror #9+ eor x3, x28, x21, ror #20+ rax1 v26.2d, v26.2d, v29.2d+ eor x21, x26, x1+ add v31.2d, v27.2d, v27.2d+ eor x9, x27, x9, ror #49+ eor x24, x28, x24, ror #28+ sri v31.2d, v27.2d, #0x3f+ eor x1, x30, x17, ror #36+ eor x14, x0, x14, ror #8+ eor v29.16b, v31.16b, v29.16b+ eor x22, x28, x22, ror #44+ eor x8, x27, x8, ror #56+ rax1 v27.2d, v27.2d, v30.2d+ eor x17, x27, x7, ror #19+ eor x15, x0, x15, ror #62+ eor v30.16b, v0.16b, v26.16b+ bic x7, x20, x22, ror #47+ eor x4, x26, x4, ror #54+ eor v31.16b, v2.16b, v29.16b+ eor x0, x0, x12, ror #3+ eor x28, x28, x23, ror #58+ shl v0.2d, v31.2d, #0x3e+ eor x23, x26, x2, ror #61+ eor x26, x26, x5, ror #25+ sri v0.2d, v31.2d, #0x2+ eor x2, x7, x16, ror #39+ bic x7, x9, x20, ror #42+ xar v2.2d, v12.2d, v29.2d, #0x15+ bic x30, x15, x9, ror #16+ eor x7, x7, x22, ror #25+ eor v31.16b, v13.16b, v28.16b+ eor x12, x30, x20, ror #58+ bic x20, x22, x16, ror #56+ shl v12.2d, v31.2d, #0x19+ eor x30, x27, x6, ror #43+ eor x22, x20, x15, ror #23+ sri v12.2d, v31.2d, #0x27+ bic x6, x19, x13, ror #42+ eor x6, x6, x17, ror #41+ xar v13.2d, v19.2d, v27.2d, #0x38+ bic x5, x13, x17, ror #63+ eor x5, x21, x5, ror #21+ eor v31.16b, v23.16b, v28.16b+ bic x17, x17, x21, ror #44+ eor x27, x27, x10, ror #23+ shl v19.2d, v31.2d, #0x38+ bic x21, x21, x25, ror #50+ bic x20, x27, x4, ror #25+ sri v19.2d, v31.2d, #0x8+ bic x10, x16, x15, ror #31+ eor x16, x21, x19, ror #43+ xar v23.2d, v15.2d, v26.2d, #0x17+ eor x21, x17, x25, ror #30+ bic x19, x25, x19, ror #57+ eor v31.16b, v1.16b, v25.16b+ ldr x25, [sp, #0x18]+ eor x17, x10, x9, ror #47+ shl v15.2d, v31.2d, #0x1+ ldr x9, [sp, #0x8]+ sri v15.2d, v31.2d, #0x3f+ eor x15, x20, x28, ror #27+ bic x20, x4, x28, ror #2+ xar v1.2d, v8.2d, v28.2d, #0x9+ eor x10, x20, x1, ror #50+ bic x20, x11, x27, ror #60+ eor v31.16b, v16.16b, v25.16b+ eor x20, x20, x4, ror #21+ bic x4, x28, x1, ror #48+ shl v8.2d, v31.2d, #0x2d+ bic x1, x1, x11, ror #57+ ldr x28, [x9, x25, lsl #3]+ sri v8.2d, v31.2d, #0x13+ ldr x9, [sp, #0xd0]+ add x25, x25, #0x1+ xar v16.2d, v7.2d, v29.2d, #0x3a+ str x25, [sp, #0x18]+ cmp x25, #0x17+ eor v31.16b, v10.16b, v26.16b+ eor x25, x1, x27, ror #53+ bic x27, x30, x26, ror #47+ shl v7.2d, v31.2d, #0x3+ eor x1, x5, x28+ eor x5, x4, x11, ror #41+ sri v7.2d, v31.2d, #0x3d+ eor x11, x19, x13, ror #35+ bic x13, x26, x24, ror #10+ xar v10.2d, v3.2d, v28.2d, #0x24+ eor x28, x27, x24, ror #57+ bic x27, x24, x9, ror #47+ eor v31.16b, v18.16b, v28.16b+ bic x19, x23, x3, ror #9+ bic x4, x29, x14, ror #41+ shl v3.2d, v31.2d, #0x15+ eor x24, x19, x29, ror #44+ bic x29, x3, x29, ror #35+ sri v3.2d, v31.2d, #0x2b+ eor x13, x13, x9, ror #57+ eor x19, x29, x14, ror #12+ xar v18.2d, v17.2d, v29.2d, #0x31+ bic x29, x9, x0, ror #19+ bic x14, x14, x8, ror #5+ eor v31.16b, v11.16b, v25.16b+ eor x9, x14, x23, ror #43+ eor x14, x4, x8, ror #46+ shl v17.2d, v31.2d, #0xa+ bic x23, x8, x23, ror #38+ eor x8, x27, x0, ror #2+ sri v17.2d, v31.2d, #0x36+ eor x4, x23, x3, ror #47+ bic x3, x0, x30, ror #5+ xar v11.2d, v9.2d, v27.2d, #0x2c+ eor x23, x3, x26, ror #52+ eor x3, x29, x30, ror #24+ eor v31.16b, v22.16b, v29.16b+ eor x0, x15, x11, ror #52+ shl v9.2d, v31.2d, #0x3d+ eor x0, x0, x13, ror #48+ eor x26, x8, x9, ror #57+ sri v9.2d, v31.2d, #0x3+ eor x27, x0, x14, ror #10+ eor x29, x16, x28, ror #63+ xar v22.2d, v14.2d, v27.2d, #0x19+ eor x26, x26, x6, ror #51+ eor x30, x23, x22, ror #50+ eor v31.16b, v20.16b, v26.16b+ eor x0, x26, x10, ror #31+ eor x29, x29, x19, ror #37+ shl v14.2d, v31.2d, #0x12+ eor x27, x27, x12, ror #5+ eor x30, x30, x24, ror #34+ sri v14.2d, v31.2d, #0x2e+ eor x0, x0, x7, ror #27+ eor x26, x30, x21, ror #26+ xar v20.2d, v4.2d, v27.2d, #0x25+ eor x26, x26, x25, ror #15+ ror x30, x27, #0x3e+ eor v31.16b, v24.16b, v27.16b+ eor x30, x30, x26, ror #57+ ror x26, x26, #0x3a+ shl v4.2d, v31.2d, #0xe+ eor x16, x30, x16+ eor x28, x30, x28, ror #63+ sri v4.2d, v31.2d, #0x32+ str x28, [sp, #0xd0]+ eor x29, x29, x17, ror #36+ xar v24.2d, v21.2d, v25.2d, #0x3e+ eor x28, x1, x2, ror #61+ eor x19, x30, x19, ror #37+ eor v31.16b, v5.16b, v26.16b+ eor x29, x29, x20, ror #2+ eor x28, x28, x4, ror #54+ shl v21.2d, v31.2d, #0x24+ eor x26, x26, x0, ror #55+ eor x28, x28, x3, ror #39+ sri v21.2d, v31.2d, #0x1c+ eor x28, x28, x5, ror #25+ ror x0, x0, #0x38+ xar v27.2d, v6.2d, v25.2d, #0x14+ eor x0, x0, x29, ror #63+ eor x27, x28, x27, ror #61+ bic v31.16b, v7.16b, v11.16b+ eor x13, x0, x13, ror #46+ eor x28, x29, x28, ror #63+ eor v5.16b, v31.16b, v10.16b+ eor x29, x30, x20, ror #2+ eor x20, x26, x3, ror #39+ bcax v6.16b, v11.16b, v8.16b, v7.16b+ eor x11, x0, x11, ror #50+ eor x25, x28, x25, ror #9+ bic v31.16b, v9.16b, v8.16b+ eor x3, x28, x21, ror #20+ eor v7.16b, v31.16b, v7.16b+ eor x21, x26, x1+ eor x9, x27, x9, ror #49+ bcax v8.16b, v8.16b, v10.16b, v9.16b+ eor x24, x28, x24, ror #28+ eor x1, x30, x17, ror #36+ bic v31.16b, v11.16b, v10.16b+ eor x14, x0, x14, ror #8+ eor x22, x28, x22, ror #44+ eor v9.16b, v31.16b, v9.16b+ eor x8, x27, x8, ror #56+ eor x17, x27, x7, ror #19+ bcax v10.16b, v15.16b, v12.16b, v16.16b+ eor x15, x0, x15, ror #62+ bic x7, x20, x22, ror #47+ bic v31.16b, v13.16b, v12.16b+ eor x4, x26, x4, ror #54+ eor x0, x0, x12, ror #3+ eor v11.16b, v31.16b, v16.16b+ eor x28, x28, x23, ror #58+ eor x23, x26, x2, ror #61+ bcax v12.16b, v12.16b, v14.16b, v13.16b+ eor x26, x26, x5, ror #25+ eor x2, x7, x16, ror #39+ bic v31.16b, v15.16b, v14.16b+ bic x7, x9, x20, ror #42+ bic x30, x15, x9, ror #16+ eor v13.16b, v31.16b, v13.16b+ eor x7, x7, x22, ror #25+ eor x12, x30, x20, ror #58+ bic v31.16b, v16.16b, v15.16b+ bic x20, x22, x16, ror #56+ eor x30, x27, x6, ror #43+ eor v14.16b, v31.16b, v14.16b+ eor x22, x20, x15, ror #23+ bic x6, x19, x13, ror #42+ bcax v15.16b, v20.16b, v17.16b, v21.16b+ eor x6, x6, x17, ror #41+ bic x5, x13, x17, ror #63+ bic v31.16b, v18.16b, v17.16b+ eor x5, x21, x5, ror #21+ bic x17, x17, x21, ror #44+ eor v16.16b, v31.16b, v21.16b+ eor x27, x27, x10, ror #23+ bic x21, x21, x25, ror #50+ bcax v17.16b, v17.16b, v19.16b, v18.16b+ bic x20, x27, x4, ror #25+ bic x10, x16, x15, ror #31+ bic v31.16b, v20.16b, v19.16b+ eor x16, x21, x19, ror #43+ eor x21, x17, x25, ror #30+ eor v18.16b, v31.16b, v18.16b+ bic x19, x25, x19, ror #57+ ldr x25, [sp, #0x18]+ bcax v19.16b, v19.16b, v21.16b, v20.16b+ eor x17, x10, x9, ror #47+ bic v31.16b, v22.16b, v1.16b+ ldr x9, [sp, #0x8]+ eor x15, x20, x28, ror #27+ eor v20.16b, v31.16b, v0.16b+ bic x20, x4, x28, ror #2+ eor x10, x20, x1, ror #50+ bcax v21.16b, v1.16b, v23.16b, v22.16b+ bic x20, x11, x27, ror #60+ eor x20, x20, x4, ror #21+ bic v31.16b, v24.16b, v23.16b+ bic x4, x28, x1, ror #48+ bic x1, x1, x11, ror #57+ eor v22.16b, v31.16b, v22.16b+ ldr x28, [x9, x25, lsl #3]+ ldr x9, [sp, #0xd0]+ bcax v23.16b, v23.16b, v0.16b, v24.16b+ add x25, x25, #0x1+ str x25, [sp, #0x18]+ bic v31.16b, v1.16b, v0.16b+ cmp x25, #0x17+ eor x25, x1, x27, ror #53+ eor v24.16b, v31.16b, v24.16b+ bic x27, x30, x26, ror #47+ eor x1, x5, x28+ bcax v0.16b, v30.16b, v2.16b, v27.16b+ eor x5, x4, x11, ror #41+ eor x11, x19, x13, ror #35+ bic v31.16b, v3.16b, v2.16b+ bic x13, x26, x24, ror #10+ eor x28, x27, x24, ror #57+ eor v1.16b, v31.16b, v27.16b+ bic x27, x24, x9, ror #47+ bic x19, x23, x3, ror #9+ bcax v2.16b, v2.16b, v4.16b, v3.16b+ bic x4, x29, x14, ror #41+ eor x24, x19, x29, ror #44+ bic v31.16b, v30.16b, v4.16b+ bic x29, x3, x29, ror #35+ eor x13, x13, x9, ror #57+ eor v3.16b, v31.16b, v3.16b+ eor x19, x29, x14, ror #12+ bic x29, x9, x0, ror #19+ bcax v4.16b, v4.16b, v27.16b, v30.16b+ bic x14, x14, x8, ror #5+ eor x9, x14, x23, ror #43+ eor x14, x4, x8, ror #46+ bic x23, x8, x23, ror #38+ eor x8, x27, x0, ror #2+ eor x4, x23, x3, ror #47+ bic x3, x0, x30, ror #5+ eor x23, x3, x26, ror #52+ eor x3, x29, x30, ror #24+ ldr x30, [sp, #0x10]+ ld1r { v28.2d }, [x30], #8+ str x30, [sp, #0x10]+ eor v0.16b, v0.16b, v28.16b++Lmld_keccak_f1600_x4_v8a_v84a_scalar_hybrid_loop_end:+ b.le Lmld_keccak_f1600_x4_v8a_v84a_scalar_hybrid_loop+ ror x2, x2, #0x3d+ ror x3, x3, #0x27+ ror x4, x4, #0x36+ ror x5, x5, #0x19+ ror x6, x6, #0x2b+ ror x7, x7, #0x13+ ror x8, x8, #0x38+ ror x9, x9, #0x31+ ror x10, x10, #0x17+ ror x11, x11, #0x32+ ror x12, x12, #0x3+ ror x13, x13, #0x2e+ ror x14, x14, #0x8+ ror x15, x15, #0x3e+ ror x17, x17, #0x24+ ror x28, x28, #0x3f+ ror x19, x19, #0x25+ ror x20, x20, #0x2+ ror x21, x21, #0x14+ ror x22, x22, #0x2c+ ror x23, x23, #0x3a+ ror x24, x24, #0x1c+ ror x25, x25, #0x9+ ldr x30, [sp, #0x20]+ cmp x30, #0x1+ b.eq Lmld_keccak_f1600_x4_v8a_v84a_scalar_hybrid_done+ mov x30, #0x1 // =1+ str x30, [sp, #0x20]+ ldr x0, [sp]+ add x0, x0, #0x190+ stp x1, x6, [x0]+ stp x11, x16, [x0, #0x10]+ stp x21, x2, [x0, #0x20]+ stp x7, x12, [x0, #0x30]+ stp x17, x22, [x0, #0x40]+ stp x3, x8, [x0, #0x50]+ stp x13, x28, [x0, #0x60]+ stp x23, x4, [x0, #0x70]+ stp x9, x14, [x0, #0x80]+ stp x19, x24, [x0, #0x90]+ stp x5, x10, [x0, #0xa0]+ stp x15, x20, [x0, #0xb0]+ str x25, [x0, #0xc0]+ sub x0, x0, #0x190+ add x0, x0, #0x258+ ldp x1, x6, [x0]+ ldp x11, x16, [x0, #0x10]+ ldp x21, x2, [x0, #0x20]+ ldp x7, x12, [x0, #0x30]+ ldp x17, x22, [x0, #0x40]+ ldp x3, x8, [x0, #0x50]+ ldp x13, x28, [x0, #0x60]+ ldp x23, x4, [x0, #0x70]+ ldp x9, x14, [x0, #0x80]+ ldp x19, x24, [x0, #0x90]+ ldp x5, x10, [x0, #0xa0]+ ldp x15, x20, [x0, #0xb0]+ ldr x25, [x0, #0xc0]+ sub x0, x0, #0x258+ b Lmld_keccak_f1600_x4_v8a_v84a_scalar_hybrid_initial++Lmld_keccak_f1600_x4_v8a_v84a_scalar_hybrid_done:+ ldr x0, [sp]+ add x0, x0, #0x258+ stp x1, x6, [x0]+ stp x11, x16, [x0, #0x10]+ stp x21, x2, [x0, #0x20]+ stp x7, x12, [x0, #0x30]+ stp x17, x22, [x0, #0x40]+ stp x3, x8, [x0, #0x50]+ stp x13, x28, [x0, #0x60]+ stp x23, x4, [x0, #0x70]+ stp x9, x14, [x0, #0x80]+ stp x19, x24, [x0, #0x90]+ stp x5, x10, [x0, #0xa0]+ stp x15, x20, [x0, #0xb0]+ str x25, [x0, #0xc0]+ sub x0, x0, #0x258+ add x4, x0, #0xc8+ trn1 v25.2d, v0.2d, v1.2d+ trn1 v26.2d, v2.2d, v3.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v0.2d, v1.2d+ trn2 v28.2d, v2.2d, v3.2d+ st1 { v27.2d, v28.2d }, [x4], #32+ trn1 v25.2d, v4.2d, v5.2d+ trn1 v26.2d, v6.2d, v7.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v4.2d, v5.2d+ trn2 v28.2d, v6.2d, v7.2d+ st1 { v27.2d, v28.2d }, [x4], #32+ trn1 v25.2d, v8.2d, v9.2d+ trn1 v26.2d, v10.2d, v11.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v8.2d, v9.2d+ trn2 v28.2d, v10.2d, v11.2d+ st1 { v27.2d, v28.2d }, [x4], #32+ trn1 v25.2d, v12.2d, v13.2d+ trn1 v26.2d, v14.2d, v15.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v12.2d, v13.2d+ trn2 v28.2d, v14.2d, v15.2d+ st1 { v27.2d, v28.2d }, [x4], #32+ trn1 v25.2d, v16.2d, v17.2d+ trn1 v26.2d, v18.2d, v19.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v16.2d, v17.2d+ trn2 v28.2d, v18.2d, v19.2d+ st1 { v27.2d, v28.2d }, [x4], #32+ trn1 v25.2d, v20.2d, v21.2d+ trn1 v26.2d, v22.2d, v23.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v20.2d, v21.2d+ trn2 v28.2d, v22.2d, v23.2d+ st1 { v27.2d, v28.2d }, [x4], #32+ str d24, [x0]+ trn2 v25.2d, v24.2d, v24.2d+ str d25, [x4]+ ldp d8, d9, [sp, #0x90]+ .cfi_restore d8+ .cfi_restore d9+ ldp d10, d11, [sp, #0xa0]+ .cfi_restore d10+ .cfi_restore d11+ ldp d12, d13, [sp, #0xb0]+ .cfi_restore d12+ .cfi_restore d13+ ldp d14, d15, [sp, #0xc0]+ .cfi_restore d14+ .cfi_restore d15+ ldp x19, x20, [sp, #0x30]+ .cfi_restore x19+ .cfi_restore x20+ ldp x21, x22, [sp, #0x40]+ .cfi_restore x21+ .cfi_restore x22+ ldp x23, x24, [sp, #0x50]+ .cfi_restore x23+ .cfi_restore x24+ ldp x25, x26, [sp, #0x60]+ .cfi_restore x25+ .cfi_restore x26+ ldp x27, x28, [sp, #0x70]+ .cfi_restore x27+ .cfi_restore x28+ ldp x29, x30, [sp, #0x80]+ .cfi_restore x29+ .cfi_restore x30+ add sp, sp, #0xe0+ .cfi_adjust_cfa_offset -0xe0+ ret+ .cfi_endproc++MLD_ASM_FN_SIZE(keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm)++#endif /* __ARM_FEATURE_SHA3 */++#endif /* MLD_FIPS202_AARCH64_NEED_X4_V8A_V84A_SCALAR_HYBRID && \+ !MLD_CONFIG_MULTILEVEL_NO_SHARED */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
@@ -0,0 +1,47 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*+ * WARNING: This file is auto-generated from scripts/autogen+ * in the mldsa-native repository.+ * Do not modify it directly.+ */++#include "../../../../common.h"++#if (defined(MLD_FIPS202_AARCH64_NEED_X1_SCALAR) || \+ defined(MLD_FIPS202_AARCH64_NEED_X1_V84A) || \+ defined(MLD_FIPS202_AARCH64_NEED_X2_V84A) || \+ defined(MLD_FIPS202_AARCH64_NEED_X4_V8A_SCALAR_HYBRID) || \+ defined(MLD_FIPS202_AARCH64_NEED_X4_V8A_V84A_SCALAR_HYBRID)) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++#include "fips202_native_aarch64.h"++MLD_ALIGN MLD_INTERNAL_DATA_DEFINITION const uint64_t+ mld_keccakf1600_round_constants[24] = {+ 0x0000000000000001, 0x0000000000008082, 0x800000000000808a,+ 0x8000000080008000, 0x000000000000808b, 0x0000000080000001,+ 0x8000000080008081, 0x8000000000008009, 0x000000000000008a,+ 0x0000000000000088, 0x0000000080008009, 0x000000008000000a,+ 0x000000008000808b, 0x800000000000008b, 0x8000000000008089,+ 0x8000000000008003, 0x8000000000008002, 0x8000000000000080,+ 0x000000000000800a, 0x800000008000000a, 0x8000000080008081,+ 0x8000000000008080, 0x0000000080000001, 0x8000000080008008,+};++#else /* (MLD_FIPS202_AARCH64_NEED_X1_SCALAR || \+ MLD_FIPS202_AARCH64_NEED_X1_V84A || MLD_FIPS202_AARCH64_NEED_X2_V84A \+ || MLD_FIPS202_AARCH64_NEED_X4_V8A_SCALAR_HYBRID || \+ MLD_FIPS202_AARCH64_NEED_X4_V8A_V84A_SCALAR_HYBRID) && \+ !MLD_CONFIG_MULTILEVEL_NO_SHARED */++MLD_EMPTY_CU(fips202_aarch64_round_constants)++#endif /* !((MLD_FIPS202_AARCH64_NEED_X1_SCALAR || \+ MLD_FIPS202_AARCH64_NEED_X1_V84A || MLD_FIPS202_AARCH64_NEED_X2_V84A \+ || MLD_FIPS202_AARCH64_NEED_X4_V8A_SCALAR_HYBRID || \+ MLD_FIPS202_AARCH64_NEED_X4_V8A_V84A_SCALAR_HYBRID) && \+ !MLD_CONFIG_MULTILEVEL_NO_SHARED) */
@@ -0,0 +1,27 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#ifndef MLD_FIPS202_NATIVE_AARCH64_X1_SCALAR_H+#define MLD_FIPS202_NATIVE_AARCH64_X1_SCALAR_H++/* Part of backend API */+#define MLD_USE_NATIVE_FIPS202_X1+/* Guard for assembly file */+#define MLD_FIPS202_AARCH64_NEED_X1_SCALAR++#if !defined(__ASSEMBLER__)+#include "../api.h"+#include "src/fips202_native_aarch64.h"+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_keccak_f1600_x1_native(uint64_t *state)+{+ mld_keccak_f1600_x1_scalar_aarch64_asm(state,+ mld_keccakf1600_round_constants);+ return MLD_NATIVE_FUNC_SUCCESS;+}+#endif /* !__ASSEMBLER__ */++#endif /* !MLD_FIPS202_NATIVE_AARCH64_X1_SCALAR_H */
@@ -0,0 +1,36 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#ifndef MLD_FIPS202_NATIVE_AARCH64_X1_V84A_H+#define MLD_FIPS202_NATIVE_AARCH64_X1_V84A_H++#if !defined(__ARM_FEATURE_SHA3)+#error This backend can only be used if SHA3 extensions are available.+#endif++/* Part of backend API */+#define MLD_USE_NATIVE_FIPS202_X1+/* Guard for assembly file */+#define MLD_FIPS202_AARCH64_NEED_X1_V84A++#if !defined(__ASSEMBLER__)+#include "../api.h"+#include "src/fips202_native_aarch64.h"+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_keccak_f1600_x1_native(uint64_t *state)+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_AARCH64_NEON) ||+ !mld_sys_check_capability(MLD_SYS_CAP_AARCH64_SHA3))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }++ mld_keccak_f1600_x1_v84a_aarch64_asm(state, mld_keccakf1600_round_constants);+ return MLD_NATIVE_FUNC_SUCCESS;+}+#endif /* !__ASSEMBLER__ */++#endif /* !MLD_FIPS202_NATIVE_AARCH64_X1_V84A_H */
@@ -0,0 +1,40 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#ifndef MLD_FIPS202_NATIVE_AARCH64_X2_V84A_H+#define MLD_FIPS202_NATIVE_AARCH64_X2_V84A_H++#if !defined(__ARM_FEATURE_SHA3)+#error This backend can only be used if SHA3 extensions are available.+#endif++/* Part of backend API */+#define MLD_USE_NATIVE_FIPS202_X4+/* Guard for assembly file */+#define MLD_FIPS202_AARCH64_NEED_X2_V84A++#if !defined(__ASSEMBLER__)+#include "../api.h"+#include "src/fips202_native_aarch64.h"++MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_keccak_f1600_x4_native(uint64_t *state)+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_AARCH64_NEON) ||+ !mld_sys_check_capability(MLD_SYS_CAP_AARCH64_SHA3))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }++ mld_keccak_f1600_x2_v84a_aarch64_asm(state + 0 * 25,+ mld_keccakf1600_round_constants);+ mld_keccak_f1600_x2_v84a_aarch64_asm(state + 2 * 25,+ mld_keccakf1600_round_constants);+ return MLD_NATIVE_FUNC_SUCCESS;+}+#endif /* !__ASSEMBLER__ */++#endif /* !MLD_FIPS202_NATIVE_AARCH64_X2_V84A_H */
@@ -0,0 +1,32 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#ifndef MLD_FIPS202_NATIVE_AARCH64_X4_V8A_SCALAR_H+#define MLD_FIPS202_NATIVE_AARCH64_X4_V8A_SCALAR_H++/* Part of backend API */+#define MLD_USE_NATIVE_FIPS202_X4+/* Guard for assembly file */+#define MLD_FIPS202_AARCH64_NEED_X4_V8A_SCALAR_HYBRID++#if !defined(__ASSEMBLER__)+#include "../api.h"+#include "src/fips202_native_aarch64.h"+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_keccak_f1600_x4_native(uint64_t *state)+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_AARCH64_NEON))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }++ mld_keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm(+ state, mld_keccakf1600_round_constants);+ return MLD_NATIVE_FUNC_SUCCESS;+}+#endif /* !__ASSEMBLER__ */++#endif /* !MLD_FIPS202_NATIVE_AARCH64_X4_V8A_SCALAR_H */
@@ -0,0 +1,37 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#ifndef MLD_FIPS202_NATIVE_AARCH64_X4_V8A_V84A_SCALAR_H+#define MLD_FIPS202_NATIVE_AARCH64_X4_V8A_V84A_SCALAR_H++#if !defined(__ARM_FEATURE_SHA3)+#error This backend can only be used if SHA3 extensions are available.+#endif++/* Part of backend API */+#define MLD_USE_NATIVE_FIPS202_X4+/* Guard for assembly file */+#define MLD_FIPS202_AARCH64_NEED_X4_V8A_V84A_SCALAR_HYBRID++#if !defined(__ASSEMBLER__)+#include "../api.h"+#include "src/fips202_native_aarch64.h"+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_keccak_f1600_x4_native(uint64_t *state)+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_AARCH64_NEON) ||+ !mld_sys_check_capability(MLD_SYS_CAP_AARCH64_SHA3))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }++ mld_keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm(+ state, mld_keccakf1600_round_constants);+ return MLD_NATIVE_FUNC_SUCCESS;+}+#endif /* !__ASSEMBLER__ */++#endif /* !MLD_FIPS202_NATIVE_AARCH64_X4_V8A_V84A_SCALAR_H */
@@ -0,0 +1,129 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#ifndef MLD_FIPS202_NATIVE_API_H+#define MLD_FIPS202_NATIVE_API_H+/*+ * FIPS-202 native interface+ *+ * This header is primarily for documentation purposes.+ * It should not be included by backend implementations.+ */++#include "../../cbmc.h"+#include "../../common.h"++/* Backends must return MLD_NATIVE_FUNC_SUCCESS upon success. */+#define MLD_NATIVE_FUNC_SUCCESS (0)+/* Backends may return MLD_NATIVE_FUNC_FALLBACK to signal to the frontend that+ * the target/parameters are unsupported; typically, this would be because of+ * dependencies on CPU features not detected on the host CPU. In this case,+ * the frontend falls back to the default C implementation.+ *+ * IMPORTANT: Backend implementations must ensure that the decision of whether+ * to fallback (return MLD_NATIVE_FUNC_FALLBACK) or not must never depend on+ * the input data itself. Fallback decisions may only depend on system+ * capabilities (e.g., CPU features) and, where present, length information.+ * This requirement applies to all backend functions to maintain constant-time+ * properties.+ */+#define MLD_NATIVE_FUNC_FALLBACK (-1)++/*+ * This is the C<->native interface allowing for the drop-in+ * of custom Keccak-F1600 implementations.+ *+ * A _backend_ is a specific implementation of parts of this interface.+ *+ * You can replace 1-fold or 4-fold batched Keccak-F1600.+ * To enable, set MLD_USE_NATIVE_FIPS202_X1 or MLD_USE_NATIVE_FIPS202_X4+ * in your backend, and define the inline wrappers mld_keccak_f1600_x1_native()+ * and/or mld_keccak_f1600_x4_native(), respectively, to forward to your+ * implementation.+ */++#if defined(MLD_USE_NATIVE_FIPS202_X1)+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_keccak_f1600_x1_native(uint64_t *state)+__contract__(+ requires(memory_no_alias(state, sizeof(uint64_t) * 25 * 1))+ assigns(memory_slice(state, sizeof(uint64_t) * 25 * 1))+ ensures(return_value == MLD_NATIVE_FUNC_FALLBACK || return_value == MLD_NATIVE_FUNC_SUCCESS)+ ensures((return_value == MLD_NATIVE_FUNC_FALLBACK) ==> array_unchanged_u64(state, 25 * 1))+);+#endif /* MLD_USE_NATIVE_FIPS202_X1 */+#if defined(MLD_USE_NATIVE_FIPS202_X4)+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_keccak_f1600_x4_native(uint64_t *state)+__contract__(+ requires(memory_no_alias(state, sizeof(uint64_t) * 25 * 4))+ assigns(memory_slice(state, sizeof(uint64_t) * 25 * 4))+ ensures(return_value == MLD_NATIVE_FUNC_FALLBACK || return_value == MLD_NATIVE_FUNC_SUCCESS)+ ensures((return_value == MLD_NATIVE_FUNC_FALLBACK) ==> array_unchanged_u64(state, 25 * 4))+);+#endif /* MLD_USE_NATIVE_FIPS202_X4 */++/*+ * Native x4 XOR bytes and extract bytes interface.+ *+ * These functions allow backends to provide optimized implementations for+ * XORing input data into the state and extracting output data from the state.+ * This is particularly useful for backends that use a different internal state+ * representation (e.g., bit-interleaved), as conversion can happen during+ * XOR/extract rather than before/after each permutation.+ *+ * NOTE: We assume that the custom representation of the zero state is the+ * all-zero state.+ *+ * MLD_USE_NATIVE_FIPS202_X4_XOR_BYTES: Backend provides native XOR bytes+ * MLD_USE_NATIVE_FIPS202_X4_EXTRACT_BYTES: Backend provides native extract+ * bytes+ */++#if defined(MLD_USE_NATIVE_FIPS202_X4_XOR_BYTES)+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_keccakf1600_xor_bytes_x4_native(+ uint64_t *state, const unsigned char *data0, const unsigned char *data1,+ const unsigned char *data2, const unsigned char *data3, unsigned offset,+ unsigned length)+__contract__(+ requires(0 <= offset && offset <= 25 * sizeof(uint64_t) &&+ 0 <= length && length <= 25 * sizeof(uint64_t) - offset)+ requires(memory_no_alias(state, sizeof(uint64_t) * 25 * 4))+ requires(memory_no_alias(data0, length))+ requires((data0 == data1 &&+ data0 == data2 &&+ data0 == data3) ||+ (memory_no_alias(data1, length) &&+ memory_no_alias(data2, length) &&+ memory_no_alias(data3, length)))+ assigns(memory_slice(state, sizeof(uint64_t) * 25 * 4))+ ensures(return_value == MLD_NATIVE_FUNC_FALLBACK || return_value == MLD_NATIVE_FUNC_SUCCESS)+ ensures((return_value == MLD_NATIVE_FUNC_FALLBACK) ==> array_unchanged_u64(state, 25 * 4)));+#endif /* MLD_USE_NATIVE_FIPS202_X4_XOR_BYTES */++#if defined(MLD_USE_NATIVE_FIPS202_X4_EXTRACT_BYTES)+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_keccakf1600_extract_bytes_x4_native(+ uint64_t *state, unsigned char *data0, unsigned char *data1,+ unsigned char *data2, unsigned char *data3, unsigned offset,+ unsigned length)+__contract__(+ requires(0 <= offset && offset <= 25 * sizeof(uint64_t) &&+ 0 <= length && length <= 25 * sizeof(uint64_t) - offset)+ requires(memory_no_alias(state, sizeof(uint64_t) * 25 * 4))+ requires(memory_no_alias(data0, length))+ requires(memory_no_alias(data1, length))+ requires(memory_no_alias(data2, length))+ requires(memory_no_alias(data3, length))+ assigns(memory_slice(data0, length))+ assigns(memory_slice(data1, length))+ assigns(memory_slice(data2, length))+ assigns(memory_slice(data3, length))+ ensures(return_value == MLD_NATIVE_FUNC_FALLBACK || return_value == MLD_NATIVE_FUNC_SUCCESS));+#endif /* MLD_USE_NATIVE_FIPS202_X4_EXTRACT_BYTES */++#endif /* !MLD_FIPS202_NATIVE_API_H */
@@ -0,0 +1,35 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#ifndef MLD_FIPS202_NATIVE_AUTO_H+#define MLD_FIPS202_NATIVE_AUTO_H++/*+ * Default FIPS202 backend+ */+#include "../../sys.h"++#if defined(MLD_SYS_AARCH64)+#include "aarch64/auto.h"+#endif++/* The x86_64 backend requires toolchain support for the SysV ABI */+#if defined(MLD_SYS_X86_64_AVX2) && defined(MLD_SYSV_ABI_SUPPORTED) && \+ (!defined(MLD_CONFIG_NO_KEYPAIR_API) || \+ !defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_REDUCE_RAM)) && \+ !defined(MLD_CONFIG_SERIAL_FIPS202_ONLY)+#include "x86_64/keccak_f1600_x4_avx2.h"+#endif /* MLD_SYS_X86_64_AVX2 && MLD_SYSV_ABI_SUPPORTED && \+ (!MLD_CONFIG_NO_KEYPAIR_API || !MLD_CONFIG_NO_SIGN_API || \+ !MLD_CONFIG_REDUCE_RAM) && !MLD_CONFIG_SERIAL_FIPS202_ONLY */++/* We do not yet include the FIPS202 backend for Armv8.1-M+MVE by default+ * as it is still experimental and undergoing review. */+/* #if defined(MLD_SYS_ARMV81M_MVE) */+/* #include "armv81m/mve.h" */+/* #endif */++#endif /* !MLD_FIPS202_NATIVE_AUTO_H */
@@ -0,0 +1,34 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#ifndef MLD_FIPS202_NATIVE_X86_64_KECCAK_F1600_X4_AVX2_H+#define MLD_FIPS202_NATIVE_X86_64_KECCAK_F1600_X4_AVX2_H++#include "../../../common.h"++#define MLD_FIPS202_X86_64_NEED_X4_AVX2++/* Part of backend API */+#define MLD_USE_NATIVE_FIPS202_X4++#if !defined(__ASSEMBLER__)+#include "../api.h"+#include "src/fips202_native_x86_64.h"+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_keccak_f1600_x4_native(uint64_t *state)+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_X86_64_AVX2))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }++ mld_keccak_f1600_x4_avx2_asm(state, mld_keccakf1600_round_constants,+ mld_keccak_rho8, mld_keccak_rho56);+ return MLD_NATIVE_FUNC_SUCCESS;+}+#endif /* !__ASSEMBLER__ */++#endif /* !MLD_FIPS202_NATIVE_X86_64_KECCAK_F1600_X4_AVX2_H */
@@ -0,0 +1,45 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#ifndef MLD_FIPS202_NATIVE_X86_64_SRC_FIPS202_NATIVE_X86_64_H+#define MLD_FIPS202_NATIVE_X86_64_SRC_FIPS202_NATIVE_X86_64_H++#include "../../../../cbmc.h"+#include "../../../../common.h"++/* TODO: Reconsider whether this check is needed -- x86_64 is always+ * little-endian, so the backend selection already implies this. */+#ifndef MLD_SYS_LITTLE_ENDIAN+#error Expecting a little-endian platform+#endif++#define mld_keccakf1600_round_constants \+ MLD_NAMESPACE(keccakf1600_round_constants)+MLD_INTERNAL_DATA_DECLARATION const uint64_t+ mld_keccakf1600_round_constants[24];++#define mld_keccak_rho8 MLD_NAMESPACE(keccak_rho8)+MLD_INTERNAL_DATA_DECLARATION const uint64_t mld_keccak_rho8[4];++#define mld_keccak_rho56 MLD_NAMESPACE(keccak_rho56)+MLD_INTERNAL_DATA_DECLARATION const uint64_t mld_keccak_rho56[4];++#define mld_keccak_f1600_x4_avx2_asm MLD_NAMESPACE(keccak_f1600_x4_avx2_asm)+MLD_SYSV_ABI+void mld_keccak_f1600_x4_avx2_asm(uint64_t states[100], const uint64_t rc[24],+ const uint64_t rho8[4],+ const uint64_t rho56[4])+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/x86_64/proofs/keccak_f1600_x4_avx2_asm.ml */+__contract__(+ requires(memory_no_alias(states, sizeof(uint64_t) * 25 * 4))+ requires(rc == mld_keccakf1600_round_constants)+ requires(rho8 == mld_keccak_rho8)+ requires(rho56 == mld_keccak_rho56)+ assigns(memory_slice(states, sizeof(uint64_t) * 25 * 4))+);++#endif /* !MLD_FIPS202_NATIVE_X86_64_SRC_FIPS202_NATIVE_X86_64_H */
@@ -0,0 +1,488 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+/*yaml+ Name: keccak_f1600_x4_avx2_asm+ Description: x86_64 AVX2 Keccak-f[1600] permutation for four sequential states+ Signature: void mld_keccak_f1600_x4_avx2_asm(uint64_t states[100], const uint64_t rc[24], const uint64_t rho8[4], const uint64_t rho56[4])+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 800+ permissions: read/write+ c_parameter: uint64_t states[100]+ description: Four sequential Keccak states (4 x 25 x uint64_t)+ rsi:+ type: buffer+ size_bytes: 192+ permissions: read-only+ c_parameter: const uint64_t rc[24]+ description: Round constants (24 x uint64_t)+ rdx:+ type: buffer+ size_bytes: 32+ permissions: read-only+ c_parameter: const uint64_t rho8[4]+ description: Rotation constant rho8 (4 x uint64_t)+ rcx:+ type: buffer+ size_bytes: 32+ permissions: read-only+ c_parameter: const uint64_t rho56[4]+ description: Rotation constant rho56 (4 x uint64_t)+*/++#include "../../../../common.h"++#if defined(MLD_FIPS202_X86_64_NEED_X4_AVX2) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/fips202/x86_64/src/keccak_f1600_x4_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(keccak_f1600_x4_avx2_asm)+MLD_ASM_FN_SYMBOL(keccak_f1600_x4_avx2_asm)++ .cfi_startproc+ movq %rsp, %r11+ .cfi_def_cfa_register %r11+ andq $-0x20, %rsp+ subq $0x300, %rsp # imm = 0x300+ vmovdqu (%rdi), %ymm0+ vmovdqu 0xc8(%rdi), %ymm3+ vmovdqu 0x190(%rdi), %ymm1+ vmovdqu 0x258(%rdi), %ymm4+ vpunpcklqdq %ymm3, %ymm0, %ymm2 # ymm2 = ymm0[0],ymm3[0],ymm0[2],ymm3[2]+ vpunpckhqdq %ymm3, %ymm0, %ymm0 # ymm0 = ymm0[1],ymm3[1],ymm0[3],ymm3[3]+ vpunpcklqdq %ymm4, %ymm1, %ymm3 # ymm3 = ymm1[0],ymm4[0],ymm1[2],ymm4[2]+ vperm2i128 $0x20, %ymm3, %ymm2, %ymm7 # ymm7 = ymm2[0,1],ymm3[0,1]+ vpunpckhqdq %ymm4, %ymm1, %ymm1 # ymm1 = ymm1[1],ymm4[1],ymm1[3],ymm4[3]+ vperm2i128 $0x31, %ymm3, %ymm2, %ymm3 # ymm3 = ymm2[2,3],ymm3[2,3]+ vmovdqu 0x278(%rdi), %ymm4+ vmovdqu %ymm3, 0x40(%rsp)+ vperm2i128 $0x31, %ymm1, %ymm0, %ymm3 # ymm3 = ymm0[2,3],ymm1[2,3]+ vmovdqu %ymm7, (%rsp)+ vperm2i128 $0x20, %ymm1, %ymm0, %ymm7 # ymm7 = ymm0[0,1],ymm1[0,1]+ vmovdqu 0x20(%rdi), %ymm0+ vmovdqu 0x1b0(%rdi), %ymm1+ vmovdqu %ymm3, 0x60(%rsp)+ vmovdqu 0xe8(%rdi), %ymm3+ vmovdqu %ymm7, 0x20(%rsp)+ vpunpcklqdq %ymm3, %ymm0, %ymm2 # ymm2 = ymm0[0],ymm3[0],ymm0[2],ymm3[2]+ vpunpckhqdq %ymm3, %ymm0, %ymm0 # ymm0 = ymm0[1],ymm3[1],ymm0[3],ymm3[3]+ vpunpcklqdq %ymm4, %ymm1, %ymm3 # ymm3 = ymm1[0],ymm4[0],ymm1[2],ymm4[2]+ vperm2i128 $0x20, %ymm3, %ymm2, %ymm7 # ymm7 = ymm2[0,1],ymm3[0,1]+ vpunpckhqdq %ymm4, %ymm1, %ymm1 # ymm1 = ymm1[1],ymm4[1],ymm1[3],ymm4[3]+ vperm2i128 $0x31, %ymm3, %ymm2, %ymm3 # ymm3 = ymm2[2,3],ymm3[2,3]+ vmovdqu 0x298(%rdi), %ymm4+ vperm2i128 $0x31, %ymm1, %ymm0, %ymm14 # ymm14 = ymm0[2,3],ymm1[2,3]+ vmovdqu %ymm7, 0x80(%rsp)+ vperm2i128 $0x20, %ymm1, %ymm0, %ymm7 # ymm7 = ymm0[0,1],ymm1[0,1]+ vmovdqu 0x40(%rdi), %ymm0+ vmovdqu 0x1d0(%rdi), %ymm1+ vmovdqu %ymm3, 0xc0(%rsp)+ vmovdqu 0x108(%rdi), %ymm3+ vmovdqu %ymm14, %ymm10+ vmovdqu %ymm7, 0xa0(%rsp)+ vpunpcklqdq %ymm3, %ymm0, %ymm2 # ymm2 = ymm0[0],ymm3[0],ymm0[2],ymm3[2]+ vpunpckhqdq %ymm3, %ymm0, %ymm0 # ymm0 = ymm0[1],ymm3[1],ymm0[3],ymm3[3]+ vpunpcklqdq %ymm4, %ymm1, %ymm3 # ymm3 = ymm1[0],ymm4[0],ymm1[2],ymm4[2]+ vpunpckhqdq %ymm4, %ymm1, %ymm1 # ymm1 = ymm1[1],ymm4[1],ymm1[3],ymm4[3]+ vperm2i128 $0x20, %ymm3, %ymm2, %ymm11 # ymm11 = ymm2[0,1],ymm3[0,1]+ vperm2i128 $0x31, %ymm3, %ymm2, %ymm3 # ymm3 = ymm2[2,3],ymm3[2,3]+ vperm2i128 $0x20, %ymm1, %ymm0, %ymm7 # ymm7 = ymm0[0,1],ymm1[0,1]+ vmovdqu %ymm3, 0x100(%rsp)+ vperm2i128 $0x31, %ymm1, %ymm0, %ymm8 # ymm8 = ymm0[2,3],ymm1[2,3]+ vmovdqu 0x128(%rdi), %ymm3+ vmovdqu 0x60(%rdi), %ymm0+ vmovdqu 0x1f0(%rdi), %ymm1+ vmovdqu %ymm7, 0xe0(%rsp)+ vmovdqu %ymm11, %ymm14+ vmovdqu 0x2b8(%rdi), %ymm4+ vmovdqu 0x2f8(%rdi), %ymm5+ vpunpcklqdq %ymm3, %ymm0, %ymm2 # ymm2 = ymm0[0],ymm3[0],ymm0[2],ymm3[2]+ vpunpckhqdq %ymm3, %ymm0, %ymm0 # ymm0 = ymm0[1],ymm3[1],ymm0[3],ymm3[3]+ vpunpcklqdq %ymm4, %ymm1, %ymm3 # ymm3 = ymm1[0],ymm4[0],ymm1[2],ymm4[2]+ vpunpckhqdq %ymm4, %ymm1, %ymm1 # ymm1 = ymm1[1],ymm4[1],ymm1[3],ymm4[3]+ vmovdqu 0x2d8(%rdi), %ymm4+ vperm2i128 $0x20, %ymm3, %ymm2, %ymm15 # ymm15 = ymm2[0,1],ymm3[0,1]+ vperm2i128 $0x31, %ymm3, %ymm2, %ymm3 # ymm3 = ymm2[2,3],ymm3[2,3]+ vperm2i128 $0x20, %ymm1, %ymm0, %ymm7 # ymm7 = ymm0[0,1],ymm1[0,1]+ vperm2i128 $0x31, %ymm1, %ymm0, %ymm9 # ymm9 = ymm0[2,3],ymm1[2,3]+ vmovdqu %ymm3, 0x140(%rsp)+ vmovdqu 0x80(%rdi), %ymm0+ vmovdqu 0x148(%rdi), %ymm3+ vmovdqu 0x210(%rdi), %ymm1+ vmovdqu %ymm7, 0x120(%rsp)+ vpunpcklqdq %ymm3, %ymm0, %ymm2 # ymm2 = ymm0[0],ymm3[0],ymm0[2],ymm3[2]+ vpunpckhqdq %ymm3, %ymm0, %ymm0 # ymm0 = ymm0[1],ymm3[1],ymm0[3],ymm3[3]+ vpunpcklqdq %ymm4, %ymm1, %ymm3 # ymm3 = ymm1[0],ymm4[0],ymm1[2],ymm4[2]+ vpunpckhqdq %ymm4, %ymm1, %ymm1 # ymm1 = ymm1[1],ymm4[1],ymm1[3],ymm4[3]+ vperm2i128 $0x20, %ymm3, %ymm2, %ymm7 # ymm7 = ymm2[0,1],ymm3[0,1]+ vperm2i128 $0x31, %ymm3, %ymm2, %ymm13 # ymm13 = ymm2[2,3],ymm3[2,3]+ vperm2i128 $0x31, %ymm1, %ymm0, %ymm3 # ymm3 = ymm0[2,3],ymm1[2,3]+ vmovdqu %ymm7, 0x160(%rsp)+ vperm2i128 $0x20, %ymm1, %ymm0, %ymm7 # ymm7 = ymm0[0,1],ymm1[0,1]+ vmovdqu 0xa0(%rdi), %ymm0+ vmovdqu 0x230(%rdi), %ymm1+ vmovdqu %ymm3, 0x1a0(%rsp)+ vmovdqu 0x168(%rdi), %ymm3+ vpunpcklqdq %ymm5, %ymm1, %ymm4 # ymm4 = ymm1[0],ymm5[0],ymm1[2],ymm5[2]+ vpunpckhqdq %ymm5, %ymm1, %ymm1 # ymm1 = ymm1[1],ymm5[1],ymm1[3],ymm5[3]+ vmovdqu %ymm7, 0x180(%rsp)+ vpunpcklqdq %ymm3, %ymm0, %ymm2 # ymm2 = ymm0[0],ymm3[0],ymm0[2],ymm3[2]+ vpunpckhqdq %ymm3, %ymm0, %ymm0 # ymm0 = ymm0[1],ymm3[1],ymm0[3],ymm3[3]+ vperm2i128 $0x20, %ymm4, %ymm2, %ymm12 # ymm12 = ymm2[0,1],ymm4[0,1]+ vperm2i128 $0x20, %ymm1, %ymm0, %ymm3 # ymm3 = ymm0[0,1],ymm1[0,1]+ vperm2i128 $0x31, %ymm4, %ymm2, %ymm7 # ymm7 = ymm2[2,3],ymm4[2,3]+ vperm2i128 $0x31, %ymm1, %ymm0, %ymm4 # ymm4 = ymm0[2,3],ymm1[2,3]+ vmovq 0x250(%rdi), %xmm0+ vmovq 0xc0(%rdi), %xmm1+ vmovdqu %ymm12, 0x1c0(%rsp)+ vmovdqu %ymm4, 0x1e0(%rsp)+ vpinsrq $0x1, 0x318(%rdi), %xmm0, %xmm0+ vpinsrq $0x1, 0x188(%rdi), %xmm1, %xmm1+ vinserti128 $0x1, %xmm0, %ymm1, %ymm2+ movq $0x0, %r10++LLmld_keccak_f1600_x4_avx2_asm:+ vmovdqu 0xa0(%rsp), %ymm4+ vpxor 0x1c0(%rsp), %ymm9, %ymm0+ vmovdqu %ymm9, 0x200(%rsp)+ vmovdqu %ymm10, %ymm9+ vmovdqu 0xc0(%rsp), %ymm11+ vmovdqu 0x160(%rsp), %ymm12+ vmovdqu %ymm3, 0x240(%rsp)+ vpxor 0x100(%rsp), %ymm4, %ymm1+ vmovdqu 0x40(%rsp), %ymm10+ vmovdqu %ymm4, 0x220(%rsp)+ vpxor %ymm3, %ymm12, %ymm12+ vmovdqu 0x20(%rsp), %ymm6+ vmovdqu 0x140(%rsp), %ymm4+ vmovdqu %ymm14, 0x2a0(%rsp)+ vpxor %ymm1, %ymm0, %ymm0+ vpxor %ymm8, %ymm11, %ymm1+ vpxor 0x180(%rsp), %ymm7, %ymm11+ vmovdqu %ymm10, 0x280(%rsp)+ vpxor %ymm1, %ymm12, %ymm12+ vpxor %ymm15, %ymm9, %ymm1+ vmovdqu 0xe0(%rsp), %ymm3+ vmovdqu %ymm8, 0x260(%rsp)+ vpxor %ymm1, %ymm11, %ymm11+ vpxor 0x120(%rsp), %ymm14, %ymm1+ vpxor %ymm6, %ymm12, %ymm12+ vmovdqu 0x60(%rsp), %ymm8+ vpxor %ymm10, %ymm11, %ymm11+ vpxor 0x1e0(%rsp), %ymm13, %ymm10+ vpxor %ymm4, %ymm3, %ymm3+ vmovdqu %ymm4, 0x2c0(%rsp)+ vpsrlq $0x3f, %ymm12, %ymm4+ vpsrlq $0x3f, %ymm11, %ymm5+ vpxor (%rsp), %ymm0, %ymm0+ vpxor %ymm1, %ymm10, %ymm10+ vmovdqu 0x80(%rsp), %ymm1+ vpxor %ymm8, %ymm10, %ymm10+ vmovdqu %ymm1, %ymm14+ vpxor 0x1a0(%rsp), %ymm2, %ymm1+ vmovdqu %ymm14, 0x2e0(%rsp)+ vpxor %ymm3, %ymm1, %ymm1+ vpsllq $0x1, %ymm12, %ymm3+ vpor %ymm4, %ymm3, %ymm3+ vpsllq $0x1, %ymm11, %ymm4+ vpxor %ymm14, %ymm1, %ymm1+ vpor %ymm5, %ymm4, %ymm4+ vpsrlq $0x3f, %ymm10, %ymm14+ vpxor %ymm1, %ymm3, %ymm3+ vpsllq $0x1, %ymm10, %ymm5+ vpxor %ymm0, %ymm4, %ymm4+ vpor %ymm14, %ymm5, %ymm5+ vpxor %ymm6, %ymm4, %ymm6+ vpxor %ymm12, %ymm5, %ymm5+ vpsrlq $0x3f, %ymm1, %ymm12+ vpsllq $0x1, %ymm1, %ymm1+ vpxor %ymm7, %ymm5, %ymm7+ vpxor %ymm9, %ymm5, %ymm9+ vpor %ymm12, %ymm1, %ymm1+ vpxor (%rsp), %ymm3, %ymm12+ vpxor %ymm11, %ymm1, %ymm1+ vpsrlq $0x3f, %ymm0, %ymm11+ vpsllq $0x1, %ymm0, %ymm0+ vpxor %ymm13, %ymm1, %ymm13+ vpxor %ymm8, %ymm1, %ymm8+ vpor %ymm11, %ymm0, %ymm0+ vpxor %ymm10, %ymm0, %ymm0+ vpxor 0xc0(%rsp), %ymm4, %ymm10+ vpxor %ymm2, %ymm0, %ymm2+ vpsrlq $0x14, %ymm10, %ymm11+ vpsllq $0x2c, %ymm10, %ymm10+ vpor %ymm11, %ymm10, %ymm10+ vpxor %ymm15, %ymm5, %ymm11+ vpbroadcastq (%rsi), %ymm15+ vpsrlq $0x15, %ymm11, %ymm14+ vpsllq $0x2b, %ymm11, %ymm11+ vpor %ymm14, %ymm11, %ymm11+ vpandn %ymm11, %ymm10, %ymm14+ vpxor %ymm15, %ymm14, %ymm14+ vpxor %ymm12, %ymm14, %ymm15+ vpsrlq $0x2b, %ymm13, %ymm14+ vpsllq $0x15, %ymm13, %ymm13+ vmovdqu %ymm15, (%rsp)+ vpor %ymm14, %ymm13, %ymm13+ vpandn %ymm13, %ymm11, %ymm14+ vpxor %ymm10, %ymm14, %ymm15+ vpsrlq $0x32, %ymm2, %ymm14+ vpsllq $0xe, %ymm2, %ymm2+ vmovdqu %ymm15, 0x20(%rsp)+ vpor %ymm14, %ymm2, %ymm2+ vpandn %ymm2, %ymm13, %ymm14+ vpxor %ymm11, %ymm14, %ymm11+ vmovdqu %ymm11, 0x40(%rsp)+ vpandn %ymm12, %ymm2, %ymm11+ vpandn %ymm10, %ymm12, %ymm12+ vpxor %ymm13, %ymm11, %ymm11+ vmovdqu %ymm11, 0x60(%rsp)+ vpxor %ymm2, %ymm12, %ymm11+ vpsrlq $0x24, %ymm8, %ymm2+ vpsllq $0x1c, %ymm8, %ymm8+ vmovdqu %ymm11, 0x80(%rsp)+ vpor %ymm2, %ymm8, %ymm8+ vpxor 0xe0(%rsp), %ymm0, %ymm2+ vpsrlq $0x2c, %ymm2, %ymm10+ vpsllq $0x14, %ymm2, %ymm2+ vpor %ymm10, %ymm2, %ymm2+ vpxor 0x100(%rsp), %ymm3, %ymm10+ vpsrlq $0x3d, %ymm10, %ymm11+ vpsllq $0x3, %ymm10, %ymm10+ vpor %ymm11, %ymm10, %ymm10+ vpandn %ymm10, %ymm2, %ymm11+ vpxor %ymm8, %ymm11, %ymm11+ vmovdqu %ymm11, 0xa0(%rsp)+ vpxor 0x160(%rsp), %ymm4, %ymm11+ vpsrlq $0x13, %ymm11, %ymm12+ vpsllq $0x2d, %ymm11, %ymm11+ vpor %ymm12, %ymm11, %ymm11+ vpandn %ymm11, %ymm10, %ymm12+ vpxor %ymm2, %ymm12, %ymm12+ vmovdqu %ymm12, 0xc0(%rsp)+ vpsrlq $0x3, %ymm7, %ymm12+ vpsllq $0x3d, %ymm7, %ymm7+ vpor %ymm12, %ymm7, %ymm7+ vpandn %ymm7, %ymm11, %ymm12+ vpxor %ymm10, %ymm12, %ymm10+ vpandn %ymm8, %ymm7, %ymm12+ vpandn %ymm2, %ymm8, %ymm8+ vpsrlq $0x3f, %ymm6, %ymm2+ vpsllq $0x1, %ymm6, %ymm6+ vpxor %ymm11, %ymm12, %ymm14+ vpor %ymm2, %ymm6, %ymm6+ vpsrlq $0x3a, %ymm9, %ymm2+ vpxor %ymm7, %ymm8, %ymm12+ vpsllq $0x6, %ymm9, %ymm9+ vmovdqu %ymm12, 0xe0(%rsp)+ vpxor 0x1a0(%rsp), %ymm0, %ymm7+ vpor %ymm2, %ymm9, %ymm9+ vpxor 0x120(%rsp), %ymm1, %ymm2+ vpshufb (%rdx), %ymm7, %ymm7+ vpsrlq $0x27, %ymm2, %ymm11+ vpsllq $0x19, %ymm2, %ymm2+ vpor %ymm2, %ymm11, %ymm11+ vpandn %ymm11, %ymm9, %ymm2+ vpandn %ymm7, %ymm11, %ymm8+ vpxor %ymm6, %ymm2, %ymm12+ vpxor 0x1c0(%rsp), %ymm3, %ymm2+ vpxor %ymm9, %ymm8, %ymm8+ vmovdqu %ymm12, 0x100(%rsp)+ vpsrlq $0x2e, %ymm2, %ymm12+ vpsllq $0x12, %ymm2, %ymm2+ vpor %ymm2, %ymm12, %ymm2+ vpandn %ymm2, %ymm7, %ymm12+ vpxor %ymm11, %ymm12, %ymm15+ vpandn %ymm6, %ymm2, %ymm11+ vpandn %ymm9, %ymm6, %ymm6+ vpxor %ymm7, %ymm11, %ymm12+ vmovdqu %ymm12, 0x120(%rsp)+ vpxor %ymm2, %ymm6, %ymm12+ vpxor 0x2e0(%rsp), %ymm0, %ymm6+ vpxor 0x2c0(%rsp), %ymm0, %ymm0+ vmovdqu %ymm12, 0x140(%rsp)+ vpsrlq $0x25, %ymm6, %ymm2+ vpsllq $0x1b, %ymm6, %ymm6+ vpor %ymm6, %ymm2, %ymm2+ vpxor 0x220(%rsp), %ymm3, %ymm6+ vpxor 0x200(%rsp), %ymm3, %ymm3+ vpsrlq $0x1c, %ymm6, %ymm7+ vpsllq $0x24, %ymm6, %ymm6+ vpor %ymm6, %ymm7, %ymm7+ vpxor 0x260(%rsp), %ymm4, %ymm6+ vpxor 0x240(%rsp), %ymm4, %ymm4+ vpsrlq $0x36, %ymm6, %ymm12+ vpsllq $0xa, %ymm6, %ymm6+ vpor %ymm6, %ymm12, %ymm12+ vpxor 0x180(%rsp), %ymm5, %ymm6+ vpxor 0x280(%rsp), %ymm5, %ymm5+ vpandn %ymm12, %ymm7, %ymm9+ vpsrlq $0x31, %ymm6, %ymm11+ vpsllq $0xf, %ymm6, %ymm6+ vpxor %ymm2, %ymm9, %ymm9+ vpor %ymm6, %ymm11, %ymm11+ vpandn %ymm11, %ymm12, %ymm6+ vpxor %ymm7, %ymm6, %ymm6+ vmovdqu %ymm6, 0x160(%rsp)+ vpxor 0x1e0(%rsp), %ymm1, %ymm6+ vpxor 0x2a0(%rsp), %ymm1, %ymm1+ vpshufb (%rcx), %ymm6, %ymm6+ vpandn %ymm6, %ymm11, %ymm13+ vpxor %ymm12, %ymm13, %ymm13+ vmovdqu %ymm13, 0x180(%rsp)+ vpandn %ymm2, %ymm6, %ymm13+ vpandn %ymm7, %ymm2, %ymm2+ vpxor %ymm6, %ymm2, %ymm2+ vpsrlq $0x3e, %ymm4, %ymm6+ vpxor %ymm11, %ymm13, %ymm13+ vmovdqu %ymm2, 0x1a0(%rsp)+ vpsrlq $0x2, %ymm5, %ymm2+ vpsllq $0x3e, %ymm5, %ymm5+ vpor %ymm5, %ymm2, %ymm2+ vpsrlq $0x9, %ymm1, %ymm5+ vpsllq $0x37, %ymm1, %ymm1+ vpsllq $0x2, %ymm4, %ymm4+ vpor %ymm1, %ymm5, %ymm1+ vpsrlq $0x19, %ymm0, %ymm5+ vpor %ymm4, %ymm6, %ymm4+ vpsllq $0x27, %ymm0, %ymm0+ vpor %ymm0, %ymm5, %ymm5+ vpandn %ymm5, %ymm1, %ymm0+ vpxor %ymm2, %ymm0, %ymm0+ vmovdqu %ymm0, 0x1c0(%rsp)+ vpsrlq $0x17, %ymm3, %ymm0+ vpsllq $0x29, %ymm3, %ymm3+ vpor %ymm3, %ymm0, %ymm0+ vpandn %ymm4, %ymm0, %ymm7+ vpandn %ymm0, %ymm5, %ymm3+ vpxor %ymm5, %ymm7, %ymm7+ vpandn %ymm2, %ymm4, %ymm5+ vpandn %ymm1, %ymm2, %ymm2+ vpxor %ymm0, %ymm5, %ymm5+ vpxor %ymm1, %ymm3, %ymm3+ vpxor %ymm4, %ymm2, %ymm2+ vmovdqu %ymm5, 0x1e0(%rsp)+ addq $0x8, %rsi+ addq $0x1, %r10+ cmpq $0x18, %r10+ jne LLmld_keccak_f1600_x4_avx2_asm+ vmovdqu (%rsp), %ymm4+ vmovdqu 0x40(%rsp), %ymm5+ vmovdqu 0x20(%rsp), %ymm0+ vmovdqu 0x60(%rsp), %ymm1+ vmovdqu 0x1c0(%rsp), %ymm12+ vmovdqu %ymm2, 0x1c0(%rsp)+ vpunpcklqdq %ymm0, %ymm4, %ymm2 # ymm2 = ymm4[0],ymm0[0],ymm4[2],ymm0[2]+ vpunpckhqdq %ymm0, %ymm4, %ymm0 # ymm0 = ymm4[1],ymm0[1],ymm4[3],ymm0[3]+ vpunpcklqdq %ymm1, %ymm5, %ymm4 # ymm4 = ymm5[0],ymm1[0],ymm5[2],ymm1[2]+ vpunpckhqdq %ymm1, %ymm5, %ymm1 # ymm1 = ymm5[1],ymm1[1],ymm5[3],ymm1[3]+ vperm2i128 $0x20, %ymm4, %ymm2, %ymm6 # ymm6 = ymm2[0,1],ymm4[0,1]+ vperm2i128 $0x31, %ymm4, %ymm2, %ymm2 # ymm2 = ymm2[2,3],ymm4[2,3]+ vmovdqu 0x80(%rsp), %ymm4+ vperm2i128 $0x20, %ymm1, %ymm0, %ymm5 # ymm5 = ymm0[0,1],ymm1[0,1]+ vperm2i128 $0x31, %ymm1, %ymm0, %ymm0 # ymm0 = ymm0[2,3],ymm1[2,3]+ vmovdqu %ymm6, (%rdi)+ vmovdqu %ymm5, 0xc8(%rdi)+ vmovdqu %ymm2, 0x190(%rdi)+ vmovdqu %ymm0, 0x258(%rdi)+ vmovdqu 0xa0(%rsp), %ymm0+ vpunpcklqdq %ymm0, %ymm4, %ymm2 # ymm2 = ymm4[0],ymm0[0],ymm4[2],ymm0[2]+ vpunpckhqdq %ymm0, %ymm4, %ymm1 # ymm1 = ymm4[1],ymm0[1],ymm4[3],ymm0[3]+ vmovdqu 0xc0(%rsp), %ymm0+ vpunpcklqdq %ymm10, %ymm0, %ymm4 # ymm4 = ymm0[0],ymm10[0],ymm0[2],ymm10[2]+ vpunpckhqdq %ymm10, %ymm0, %ymm0 # ymm0 = ymm0[1],ymm10[1],ymm0[3],ymm10[3]+ vperm2i128 $0x20, %ymm4, %ymm2, %ymm6 # ymm6 = ymm2[0,1],ymm4[0,1]+ vperm2i128 $0x20, %ymm0, %ymm1, %ymm5 # ymm5 = ymm1[0,1],ymm0[0,1]+ vperm2i128 $0x31, %ymm4, %ymm2, %ymm2 # ymm2 = ymm2[2,3],ymm4[2,3]+ vmovdqu 0xe0(%rsp), %ymm4+ vperm2i128 $0x31, %ymm0, %ymm1, %ymm1 # ymm1 = ymm1[2,3],ymm0[2,3]+ vmovdqu 0x100(%rsp), %ymm0+ vmovdqu %ymm2, 0x1b0(%rdi)+ vmovdqu %ymm1, 0x278(%rdi)+ vpunpcklqdq %ymm4, %ymm14, %ymm2 # ymm2 = ymm14[0],ymm4[0],ymm14[2],ymm4[2]+ vpunpckhqdq %ymm4, %ymm14, %ymm1 # ymm1 = ymm14[1],ymm4[1],ymm14[3],ymm4[3]+ vpunpcklqdq %ymm8, %ymm0, %ymm4 # ymm4 = ymm0[0],ymm8[0],ymm0[2],ymm8[2]+ vpunpckhqdq %ymm8, %ymm0, %ymm0 # ymm0 = ymm0[1],ymm8[1],ymm0[3],ymm8[3]+ vmovdqu %ymm6, 0x20(%rdi)+ vmovdqu %ymm5, 0xe8(%rdi)+ vperm2i128 $0x20, %ymm4, %ymm2, %ymm6 # ymm6 = ymm2[0,1],ymm4[0,1]+ vperm2i128 $0x20, %ymm0, %ymm1, %ymm5 # ymm5 = ymm1[0,1],ymm0[0,1]+ vperm2i128 $0x31, %ymm4, %ymm2, %ymm2 # ymm2 = ymm2[2,3],ymm4[2,3]+ vperm2i128 $0x31, %ymm0, %ymm1, %ymm1 # ymm1 = ymm1[2,3],ymm0[2,3]+ vmovdqu 0x120(%rsp), %ymm4+ vmovdqu 0x140(%rsp), %ymm0+ vmovdqu %ymm2, 0x1d0(%rdi)+ vmovdqu %ymm1, 0x298(%rdi)+ vpunpcklqdq %ymm4, %ymm15, %ymm2 # ymm2 = ymm15[0],ymm4[0],ymm15[2],ymm4[2]+ vpunpckhqdq %ymm4, %ymm15, %ymm1 # ymm1 = ymm15[1],ymm4[1],ymm15[3],ymm4[3]+ vpunpcklqdq %ymm9, %ymm0, %ymm4 # ymm4 = ymm0[0],ymm9[0],ymm0[2],ymm9[2]+ vmovdqu %ymm5, 0x108(%rdi)+ vpunpckhqdq %ymm9, %ymm0, %ymm0 # ymm0 = ymm0[1],ymm9[1],ymm0[3],ymm9[3]+ vmovdqu %ymm6, 0x40(%rdi)+ vperm2i128 $0x20, %ymm4, %ymm2, %ymm6 # ymm6 = ymm2[0,1],ymm4[0,1]+ vperm2i128 $0x31, %ymm4, %ymm2, %ymm2 # ymm2 = ymm2[2,3],ymm4[2,3]+ vperm2i128 $0x20, %ymm0, %ymm1, %ymm5 # ymm5 = ymm1[0,1],ymm0[0,1]+ vmovdqu 0x160(%rsp), %ymm4+ vperm2i128 $0x31, %ymm0, %ymm1, %ymm1 # ymm1 = ymm1[2,3],ymm0[2,3]+ vmovdqu 0x180(%rsp), %ymm0+ vmovdqu %ymm5, 0x128(%rdi)+ vmovdqu 0x1a0(%rsp), %ymm5+ vmovdqu %ymm2, 0x1f0(%rdi)+ vpunpcklqdq %ymm0, %ymm4, %ymm2 # ymm2 = ymm4[0],ymm0[0],ymm4[2],ymm0[2]+ vpunpckhqdq %ymm0, %ymm4, %ymm0 # ymm0 = ymm4[1],ymm0[1],ymm4[3],ymm0[3]+ vpunpcklqdq %ymm5, %ymm13, %ymm4 # ymm4 = ymm13[0],ymm5[0],ymm13[2],ymm5[2]+ vmovdqu %ymm6, 0x60(%rdi)+ vperm2i128 $0x20, %ymm4, %ymm2, %ymm6 # ymm6 = ymm2[0,1],ymm4[0,1]+ vmovdqu %ymm1, 0x2b8(%rdi)+ vperm2i128 $0x31, %ymm4, %ymm2, %ymm2 # ymm2 = ymm2[2,3],ymm4[2,3]+ vpunpckhqdq %ymm5, %ymm13, %ymm1 # ymm1 = ymm13[1],ymm5[1],ymm13[3],ymm5[3]+ vmovdqu %ymm6, 0x80(%rdi)+ vmovdqu 0x1e0(%rsp), %ymm4+ vperm2i128 $0x20, %ymm1, %ymm0, %ymm5 # ymm5 = ymm0[0,1],ymm1[0,1]+ vperm2i128 $0x31, %ymm1, %ymm0, %ymm0 # ymm0 = ymm0[2,3],ymm1[2,3]+ vmovdqu %ymm2, 0x210(%rdi)+ vpunpcklqdq %ymm3, %ymm12, %ymm2 # ymm2 = ymm12[0],ymm3[0],ymm12[2],ymm3[2]+ vmovdqu %ymm0, 0x2d8(%rdi)+ vpunpckhqdq %ymm3, %ymm12, %ymm0 # ymm0 = ymm12[1],ymm3[1],ymm12[3],ymm3[3]+ vpunpcklqdq %ymm4, %ymm7, %ymm3 # ymm3 = ymm7[0],ymm4[0],ymm7[2],ymm4[2]+ vpunpckhqdq %ymm4, %ymm7, %ymm1 # ymm1 = ymm7[1],ymm4[1],ymm7[3],ymm4[3]+ vmovdqu %ymm5, 0x148(%rdi)+ vperm2i128 $0x20, %ymm3, %ymm2, %ymm5 # ymm5 = ymm2[0,1],ymm3[0,1]+ vperm2i128 $0x31, %ymm3, %ymm2, %ymm2 # ymm2 = ymm2[2,3],ymm3[2,3]+ vmovdqu 0x1c0(%rsp), %ymm3+ vperm2i128 $0x20, %ymm1, %ymm0, %ymm4 # ymm4 = ymm0[0,1],ymm1[0,1]+ vperm2i128 $0x31, %ymm1, %ymm0, %ymm0 # ymm0 = ymm0[2,3],ymm1[2,3]+ vmovdqu %ymm5, 0xa0(%rdi)+ vextracti128 $0x1, %ymm3, %xmm15+ vmovdqu %ymm4, 0x168(%rdi)+ vmovdqu %ymm2, 0x230(%rdi)+ vmovdqu %ymm0, 0x2f8(%rdi)+ vmovq %xmm3, 0xc0(%rdi)+ vmovhpd %xmm3, 0x188(%rdi)+ vmovq %xmm15, 0x250(%rdi)+ vmovhpd %xmm15, 0x318(%rdi)+ movq %r11, %rsp+ .cfi_def_cfa_register %rsp+ retq+ .cfi_endproc++MLD_ASM_FN_SIZE(keccak_f1600_x4_avx2_asm)++#endif /* MLD_FIPS202_X86_64_NEED_X4_AVX2 && !MLD_CONFIG_MULTILEVEL_NO_SHARED */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
@@ -0,0 +1,52 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*+ * WARNING: This file is auto-generated from scripts/autogen+ * in the mldsa-native repository.+ * Do not modify it directly.+ */++#include "../../../../common.h"+#if defined(MLD_FIPS202_X86_64_NEED_X4_AVX2) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++#include <stdint.h>++#include "fips202_native_x86_64.h"++MLD_ALIGN MLD_INTERNAL_DATA_DEFINITION const uint64_t+ mld_keccakf1600_round_constants[24] = {+ 0x0000000000000001, 0x0000000000008082, 0x800000000000808a,+ 0x8000000080008000, 0x000000000000808b, 0x0000000080000001,+ 0x8000000080008081, 0x8000000000008009, 0x000000000000008a,+ 0x0000000000000088, 0x0000000080008009, 0x000000008000000a,+ 0x000000008000808b, 0x800000000000008b, 0x8000000000008089,+ 0x8000000000008003, 0x8000000000008002, 0x8000000000000080,+ 0x000000000000800a, 0x800000008000000a, 0x8000000080008081,+ 0x8000000000008080, 0x0000000080000001, 0x8000000080008008,+};++MLD_ALIGN MLD_INTERNAL_DATA_DEFINITION const uint64_t mld_keccak_rho8[4] = {+ 0x0605040302010007,+ 0x0e0d0c0b0a09080f,+ 0x1615141312111017,+ 0x1e1d1c1b1a19181f,+};++MLD_ALIGN MLD_INTERNAL_DATA_DEFINITION const uint64_t mld_keccak_rho56[4] = {+ 0x0007060504030201,+ 0x080f0e0d0c0b0a09,+ 0x1017161514131211,+ 0x181f1e1d1c1b1a19,+};++#else /* MLD_FIPS202_X86_64_NEED_X4_AVX2 && !MLD_CONFIG_MULTILEVEL_NO_SHARED \+ */++MLD_EMPTY_CU(fips202_x86_64_constants)++#endif /* !(MLD_FIPS202_X86_64_NEED_X4_AVX2 && \+ !MLD_CONFIG_MULTILEVEL_NO_SHARED) */
@@ -0,0 +1,314 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#ifndef MLD_NATIVE_AARCH64_META_H+#define MLD_NATIVE_AARCH64_META_H++/* Set of primitives that this backend replaces */+#define MLD_USE_NATIVE_NTT+#define MLD_USE_NATIVE_INTT+#define MLD_USE_NATIVE_REJ_UNIFORM+#define MLD_USE_NATIVE_REJ_UNIFORM_ETA2+#define MLD_USE_NATIVE_REJ_UNIFORM_ETA4+#define MLD_USE_NATIVE_POLY_DECOMPOSE_32+#define MLD_USE_NATIVE_POLY_DECOMPOSE_88+#define MLD_USE_NATIVE_POLY_CADDQ+#define MLD_USE_NATIVE_POLY_USE_HINT_32+#define MLD_USE_NATIVE_POLY_USE_HINT_88+#define MLD_USE_NATIVE_POLY_CHKNORM+#define MLD_USE_NATIVE_POLYZ_UNPACK_17+#define MLD_USE_NATIVE_POLYZ_UNPACK_19+#define MLD_USE_NATIVE_POINTWISE_MONTGOMERY+#define MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L4+#define MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L5+#define MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L7++/* Identifier for this backend so that source and assembly files+ * in the build can be appropriately guarded. */+#define MLD_ARITH_BACKEND_AARCH64+++#if !defined(__ASSEMBLER__)+#include "../api.h"+#include "src/arith_native_aarch64.h"++MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_ntt_native(int32_t data[MLDSA_N])+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_AARCH64_NEON))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }+ mld_ntt_aarch64_asm(data, mld_aarch64_ntt_zetas_layer123456,+ mld_aarch64_ntt_zetas_layer78);+ return MLD_NATIVE_FUNC_SUCCESS;+}++MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_intt_native(int32_t data[MLDSA_N])+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_AARCH64_NEON))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }+ mld_intt_aarch64_asm(data, mld_aarch64_intt_zetas_layer78,+ mld_aarch64_intt_zetas_layer123456);+ return MLD_NATIVE_FUNC_SUCCESS;+}++MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_rej_uniform_native(int32_t *r, unsigned len,+ const uint8_t *buf,+ unsigned buflen)+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_AARCH64_NEON) || len != MLDSA_N ||+ buflen % 24 != 0)+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }++ /* Safety: outlen is at most MLDSA_N, hence, this cast is safe. */+ return (int)mld_rej_uniform_aarch64_asm(r, buf, buflen,+ mld_rej_uniform_table);+}++#if !defined(MLD_CONFIG_NO_KEYPAIR_API)+#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLDSA_ETA == 2+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_rej_uniform_eta2_native(int32_t *r, unsigned len,+ const uint8_t *buf,+ unsigned buflen)+{+ uint64_t outlen;+ /* AArch64 implementation assumes specific buffer lengths */+ if (!mld_sys_check_capability(MLD_SYS_CAP_AARCH64_NEON) || len != MLDSA_N ||+ buflen != MLD_AARCH64_REJ_UNIFORM_ETA2_BUFLEN)+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }+ /* Constant time: Inputs and outputs to this function are secret.+ * It is safe to leak which coefficients are accepted/rejected.+ * The assembly implementation must not leak any other information about the+ * accepted coefficients. Constant-time testing cannot cover this, and we+ * hence have to manually verify the assembly.+ * We declassify prior the input data and mark the outputs as secret.+ */+ MLD_CT_TESTING_DECLASSIFY(buf, buflen);+ outlen = mld_rej_uniform_eta2_aarch64_asm(r, buf, buflen,+ mld_rej_uniform_eta_table);+ MLD_CT_TESTING_SECRET(r, sizeof(int32_t) * outlen);+ /* Safety: outlen is at most MLDSA_N and, hence, this cast is safe. */+ return (int)outlen;+}+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLDSA_ETA == 2 */++#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLDSA_ETA == 4+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_rej_uniform_eta4_native(int32_t *r, unsigned len,+ const uint8_t *buf,+ unsigned buflen)+{+ uint64_t outlen;+ /* AArch64 implementation assumes specific buffer lengths */+ if (!mld_sys_check_capability(MLD_SYS_CAP_AARCH64_NEON) || len != MLDSA_N ||+ buflen != MLD_AARCH64_REJ_UNIFORM_ETA4_BUFLEN)+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }+ /* Constant time: Inputs and outputs to this function are secret.+ * It is safe to leak which coefficients are accepted/rejected.+ * The assembly implementation must not leak any other information about the+ * accepted coefficients. Constant-time testing cannot cover this, and we+ * hence have to manually verify the assembly.+ * We declassify prior the input data and mark the outputs as secret.+ */+ MLD_CT_TESTING_DECLASSIFY(buf, buflen);+ outlen = mld_rej_uniform_eta4_aarch64_asm(r, buf, buflen,+ mld_rej_uniform_eta_table);+ MLD_CT_TESTING_SECRET(r, sizeof(int32_t) * outlen);+ /* Safety: outlen is at most MLDSA_N and, hence, this cast is safe. */+ return (int)outlen;+}+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLDSA_ETA == 4 */+#endif /* !MLD_CONFIG_NO_KEYPAIR_API */++#if !defined(MLD_CONFIG_NO_SIGN_API)+#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \+ (MLD_CONFIG_PARAMETER_SET == 65 || MLD_CONFIG_PARAMETER_SET == 87)+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_poly_decompose_32_native(int32_t *a1, int32_t *a0)+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_AARCH64_NEON))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }+ mld_poly_decompose_32_aarch64_asm(a1, a0);+ return MLD_NATIVE_FUNC_SUCCESS;+}+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 65 \+ || MLD_CONFIG_PARAMETER_SET == 87 */++#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLD_CONFIG_PARAMETER_SET == 44+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_poly_decompose_88_native(int32_t *a1, int32_t *a0)+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_AARCH64_NEON))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }+ mld_poly_decompose_88_aarch64_asm(a1, a0);+ return MLD_NATIVE_FUNC_SUCCESS;+}+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 44 \+ */+#endif /* !MLD_CONFIG_NO_SIGN_API */++MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_poly_caddq_native(int32_t a[MLDSA_N])+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_AARCH64_NEON))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }+ mld_poly_caddq_aarch64_asm(a);+ return MLD_NATIVE_FUNC_SUCCESS;+}++#if !defined(MLD_CONFIG_NO_VERIFY_API)+#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \+ (MLD_CONFIG_PARAMETER_SET == 65 || MLD_CONFIG_PARAMETER_SET == 87)+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_poly_use_hint_32_native(int32_t *a, const int32_t *h)+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_AARCH64_NEON))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }+ mld_poly_use_hint_32_aarch64_asm(a, h);+ return MLD_NATIVE_FUNC_SUCCESS;+}+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 65 \+ || MLD_CONFIG_PARAMETER_SET == 87 */++#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLD_CONFIG_PARAMETER_SET == 44+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_poly_use_hint_88_native(int32_t *a, const int32_t *h)+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_AARCH64_NEON))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }+ mld_poly_use_hint_88_aarch64_asm(a, h);+ return MLD_NATIVE_FUNC_SUCCESS;+}+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 44 \+ */+#endif /* !MLD_CONFIG_NO_VERIFY_API */++MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_poly_chknorm_native(const int32_t *a, int32_t B)+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_AARCH64_NEON))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }+ return mld_poly_chknorm_aarch64_asm(a, B);+}++#if !defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API)+#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLD_CONFIG_PARAMETER_SET == 44+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_polyz_unpack_17_native(int32_t *r, const uint8_t *buf)+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_AARCH64_NEON))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }+ mld_polyz_unpack_17_aarch64_asm(r, buf, mld_polyz_unpack_17_indices);+ return MLD_NATIVE_FUNC_SUCCESS;+}+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 44 \+ */++#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \+ (MLD_CONFIG_PARAMETER_SET == 65 || MLD_CONFIG_PARAMETER_SET == 87)+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_polyz_unpack_19_native(int32_t *r, const uint8_t *buf)+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_AARCH64_NEON))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }+ mld_polyz_unpack_19_aarch64_asm(r, buf, mld_polyz_unpack_19_indices);+ return MLD_NATIVE_FUNC_SUCCESS;+}+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 65 \+ || MLD_CONFIG_PARAMETER_SET == 87 */+#endif /* !MLD_CONFIG_NO_SIGN_API || !MLD_CONFIG_NO_VERIFY_API */++#if !defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API) || \+ defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_poly_pointwise_montgomery_native(+ int32_t a[MLDSA_N], const int32_t b[MLDSA_N])+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_AARCH64_NEON))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }+ mld_poly_pointwise_montgomery_aarch64_asm(a, b);+ return MLD_NATIVE_FUNC_SUCCESS;+}+#endif /* !MLD_CONFIG_NO_SIGN_API || !MLD_CONFIG_NO_VERIFY_API || \+ MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST */++#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLDSA_L == 4+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_polyvecl_pointwise_acc_montgomery_l4_native(+ int32_t w[MLDSA_N], const int32_t u[4][MLDSA_N],+ const int32_t v[4][MLDSA_N])+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_AARCH64_NEON))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }+ mld_polyvecl_pointwise_acc_montgomery_l4_aarch64_asm(w, u, v);+ return MLD_NATIVE_FUNC_SUCCESS;+}+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLDSA_L == 4 */++#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLDSA_L == 5+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_polyvecl_pointwise_acc_montgomery_l5_native(+ int32_t w[MLDSA_N], const int32_t u[5][MLDSA_N],+ const int32_t v[5][MLDSA_N])+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_AARCH64_NEON))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }+ mld_polyvecl_pointwise_acc_montgomery_l5_aarch64_asm(w, u, v);+ return MLD_NATIVE_FUNC_SUCCESS;+}+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLDSA_L == 5 */++#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLDSA_L == 7+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_polyvecl_pointwise_acc_montgomery_l7_native(+ int32_t w[MLDSA_N], const int32_t u[7][MLDSA_N],+ const int32_t v[7][MLDSA_N])+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_AARCH64_NEON))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }+ mld_polyvecl_pointwise_acc_montgomery_l7_aarch64_asm(w, u, v);+ return MLD_NATIVE_FUNC_SUCCESS;+}+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLDSA_L == 7 */++#endif /* !__ASSEMBLER__ */+#endif /* !MLD_NATIVE_AARCH64_META_H */
@@ -0,0 +1,248 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*+ * WARNING: This file is auto-generated from scripts/autogen+ * in the mldsa-native repository.+ * Do not modify it directly.+ */++#include "../../../common.h"++#if defined(MLD_ARITH_BACKEND_AARCH64) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++#include "arith_native_aarch64.h"++/*+ * Table of zeta values used in the AArch64 forward NTT+ * See autogen for details.+ */+MLD_ALIGN MLD_INTERNAL_DATA_DEFINITION const int32_t+ mld_aarch64_ntt_zetas_layer123456[144] = {+ -3572223, -915382907, 3765607, 964937599, 3761513, 963888510,+ -3201494, -820383522, -2883726, -738955404, -3145678, -806080660,+ -3201430, -820367122, 0, 0, -601683, -154181397,+ -3370349, -863652652, -4063053, -1041158200, 3602218, 923069133,+ 3182878, 815613168, 2740543, 702264730, -3586446, -919027554,+ 0, 0, 3542485, 907762539, 2663378, 682491182,+ -1674615, -429120452, -3110818, -797147778, 2101410, 538486762,+ 3704823, 949361686, 1159875, 297218217, 0, 0,+ 2682288, 687336873, -3524442, -903139016, -434125, -111244624,+ 394148, 101000509, 928749, 237992130, 1095468, 280713909,+ -3506380, -898510625, 0, 0, 2129892, 545785280,+ 676590, 173376332, -1335936, -342333886, 2071829, 530906624,+ -4018989, -1029866791, 3241972, 830756018, 2156050, 552488273,+ 0, 0, 3764867, 964747974, -3227876, -827143915,+ 1714295, 439288460, 3415069, 875112161, 1759347, 450833045,+ -817536, -209493775, -3574466, -915957677, 0, 0,+ -1005239, -257592709, 2453983, 628833668, 1460718, 374309300,+ 3756790, 962678241, -1935799, -496048908, -1716988, -439978542,+ -3950053, -1012201926, 0, 0, 557458, 142848732,+ -642628, -164673562, -3585098, -918682129, -2897314, -742437332,+ 3192354, 818041395, 556856, 142694469, 3870317, 991769559,+ 0, 0, -1221177, -312926867, 2815639, 721508096,+ 2283733, 585207070, 2917338, 747568486, 1853806, 475038184,+ 3345963, 857403734, 1858416, 476219497, 0, 0,+};++MLD_ALIGN MLD_INTERNAL_DATA_DEFINITION const int32_t+ mld_aarch64_ntt_zetas_layer78[384] = {+ 3073009, 1277625, -2635473, 3852015, 787459213,+ 327391679, -675340520, 987079667, 1753, -2659525,+ 2660408, -59148, 449207, -681503850, 681730119,+ -15156688, -1935420, -1455890, -1780227, 2772600,+ -495951789, -373072124, -456183549, 710479343, 4183372,+ -3222807, -3121440, -274060, 1071989969, -825844983,+ -799869667, -70227934, 1182243, 636927, -3956745,+ -3284915, 302950022, 163212680, -1013916752, -841760171,+ 87208, -3965306, -2296397, -3716946, 22347069,+ -1016110510, -588452222, -952468207, 2508980, 2028118,+ 1937570, -3815725, 642926661, 519705671, 496502727,+ -977780347, -27812, 1009365, -1979497, -3956944,+ -7126831, 258649997, -507246529, -1013967746, 822541,+ -2454145, 1596822, -3759465, 210776307, -628875181,+ 409185979, -963363710, 2811291, -2983781, -1109516,+ 4158088, 720393920, -764594519, -284313712, 1065510939,+ -1685153, 2678278, -3551006, -250446, -431820817,+ 686309310, -909946047, -64176841, -3410568, -3768948,+ 635956, -2455377, -873958779, -965793731, 162963861,+ -629190881, 1528066, 482649, 1148858, -2962264,+ 391567239, 123678909, 294395108, -759080783, -4146264,+ 2192938, 2387513, -268456, -1062481036, 561940831,+ 611800717, -68791907, -1772588, -1727088, -3611750,+ -3180456, -454226054, -442566669, -925511710, -814992530,+ -565603, 169688, 2462444, -3334383, -144935890,+ 43482586, 631001801, -854436357, 3747250, 1239911,+ 3195676, 1254190, 960233614, 317727459, 818892658,+ 321386456, 2296099, -3838479, 2642980, -12417,+ 588375860, -983611064, 677264190, -3181859, -4166425,+ -3488383, 1987814, -3197248, -1067647297, -893898890,+ 509377762, -819295484, 2998219, -89301, -1354892,+ -1310261, 768294260, -22883400, -347191365, -335754661,+ 141835, 2513018, 613238, -2218467, 36345249,+ 643961400, 157142369, -568482643, 1736313, 235407,+ -3250154, 3258457, 444930577, 60323094, -832852657,+ 834980303, -458740, 4040196, 2039144, -818761,+ -117552223, 1035301089, 522531086, -209807681, -1921994,+ -3472069, -1879878, -2178965, -492511373, -889718424,+ -481719139, -558360247, -2579253, 1787943, -2391089,+ -2254727, -660934133, 458160776, -612717067, -577774276,+ -1623354, -2374402, 586241, 527981, -415984810,+ -608441020, 150224382, 135295244, 2105286, -2033807,+ -1179613, -2743411, 539479988, -521163479, -302276083,+ -702999655, 3482206, -4182915, -1300016, -2362063,+ 892316032, -1071872863, -333129378, -605279149, -1476985,+ 2491325, 507927, -724804, -378477722, 638402564,+ 130156402, -185731180, 1994046, -1393159, -1187885,+ -1834526, 510974714, -356997292, -304395785, -470097680,+ -1317678, 2461387, 3035980, 621164, -337655269,+ 630730945, 777970524, 159173408, -3033742, 2647994,+ -2612853, 749577, -777397036, 678549029, -669544140,+ 192079267, -338420, 3009748, 4148469, -4022750,+ -86720197, 771248568, 1063046068, -1030830548, 3901472,+ -1226661, 2925816, 3374250, 999753034, -314332144,+ 749740976, 864652284, 3980599, -1615530, 1665318,+ 1163598, 1020029345, -413979908, 426738094, 298172236,+ 2569011, 1723229, 2028038, -3369273, 658309618,+ 441577800, 519685171, -863376927, 1356448, -2775755,+ 2683270, -2778788, 347590090, -711287812, 687588511,+ -712065019, 3994671, -1370517, 3363542, 545376,+ 1023635298, -351195274, 861908357, 139752717, -11879,+ 3020393, 214880, -770441, -3043996, 773976352,+ 55063046, -197425671, -3467665, 2312838, -653275,+ -459163, -888589898, 592665232, -167401858, -117660617,+ 3105558, 508145, 860144, 140244, 795799901,+ 130212265, 220412084, 35937555, -1103344, -553718,+ 3430436, -1514152, -282732136, -141890356, 879049958,+ -388001774, 348812, -327848, 1011223, -2354215,+ 89383150, -84011120, 259126110, -603268097, -2185084,+ 2358373, -3014420, 2926054, -559928242, 604333585,+ -772445769, 749801963, 3123762, -2193087, -1716814,+ -392707, 800464680, -561979013, -439933955, -100631253,+ -3818627, -1922253, -2236726, 1744507, -978523985,+ -492577742, -573161516, 447030292, -303005, -3974485,+ 1900052, 1054478, -77645096, -1018462631, 486888731,+ 270210213, 3531229, -3773731, -781875, -731434,+ 904878186, -967019376, -200355636, -187430119,+};++MLD_ALIGN MLD_INTERNAL_DATA_DEFINITION const int32_t+ mld_aarch64_intt_zetas_layer78[384] = {+ -1744507, 2236726, 1922253, 3818627, -447030292,+ 573161516, 492577742, 978523985, 731434, 781875,+ 3773731, -3531229, 187430119, 200355636, 967019376,+ -904878186, -1054478, -1900052, 3974485, 303005,+ -270210213, -486888731, 1018462631, 77645096, 2354215,+ -1011223, 327848, -348812, 603268097, -259126110,+ 84011120, -89383150, 392707, 1716814, 2193087,+ -3123762, 100631253, 439933955, 561979013, -800464680,+ -2926054, 3014420, -2358373, 2185084, -749801963,+ 772445769, -604333585, 559928242, 459163, 653275,+ -2312838, 3467665, 117660617, 167401858, -592665232,+ 888589898, 1514152, -3430436, 553718, 1103344,+ 388001774, -879049958, 141890356, 282732136, -140244,+ -860144, -508145, -3105558, -35937555, -220412084,+ -130212265, -795799901, 2778788, -2683270, 2775755,+ -1356448, 712065019, -687588511, 711287812, -347590090,+ 770441, -214880, -3020393, 11879, 197425671,+ -55063046, -773976352, 3043996, -545376, -3363542,+ 1370517, -3994671, -139752717, -861908357, 351195274,+ -1023635298, -3374250, -2925816, 1226661, -3901472,+ -864652284, -749740976, 314332144, -999753034, 3369273,+ -2028038, -1723229, -2569011, 863376927, -519685171,+ -441577800, -658309618, -1163598, -1665318, 1615530,+ -3980599, -298172236, -426738094, 413979908, -1020029345,+ -621164, -3035980, -2461387, 1317678, -159173408,+ -777970524, -630730945, 337655269, 4022750, -4148469,+ -3009748, 338420, 1030830548, -1063046068, -771248568,+ 86720197, -749577, 2612853, -2647994, 3033742,+ -192079267, 669544140, -678549029, 777397036, 2362063,+ 1300016, 4182915, -3482206, 605279149, 333129378,+ 1071872863, -892316032, 1834526, 1187885, 1393159,+ -1994046, 470097680, 304395785, 356997292, -510974714,+ 724804, -507927, -2491325, 1476985, 185731180,+ -130156402, -638402564, 378477722, 2254727, 2391089,+ -1787943, 2579253, 577774276, 612717067, -458160776,+ 660934133, 2743411, 1179613, 2033807, -2105286,+ 702999655, 302276083, 521163479, -539479988, -527981,+ -586241, 2374402, 1623354, -135295244, -150224382,+ 608441020, 415984810, -3258457, 3250154, -235407,+ -1736313, -834980303, 832852657, -60323094, -444930577,+ 2178965, 1879878, 3472069, 1921994, 558360247,+ 481719139, 889718424, 492511373, 818761, -2039144,+ -4040196, 458740, 209807681, -522531086, -1035301089,+ 117552223, 3197248, -1987814, 3488383, 4166425,+ 819295484, -509377762, 893898890, 1067647297, 2218467,+ -613238, -2513018, -141835, 568482643, -157142369,+ -643961400, -36345249, 1310261, 1354892, 89301,+ -2998219, 335754661, 347191365, 22883400, -768294260,+ 3334383, -2462444, -169688, 565603, 854436357,+ -631001801, -43482586, 144935890, 12417, -2642980,+ 3838479, -2296099, 3181859, -677264190, 983611064,+ -588375860, -1254190, -3195676, -1239911, -3747250,+ -321386456, -818892658, -317727459, -960233614, 2962264,+ -1148858, -482649, -1528066, 759080783, -294395108,+ -123678909, -391567239, 3180456, 3611750, 1727088,+ 1772588, 814992530, 925511710, 442566669, 454226054,+ 268456, -2387513, -2192938, 4146264, 68791907,+ -611800717, -561940831, 1062481036, -4158088, 1109516,+ 2983781, -2811291, -1065510939, 284313712, 764594519,+ -720393920, 2455377, -635956, 3768948, 3410568,+ 629190881, -162963861, 965793731, 873958779, 250446,+ 3551006, -2678278, 1685153, 64176841, 909946047,+ -686309310, 431820817, 3815725, -1937570, -2028118,+ -2508980, 977780347, -496502727, -519705671, -642926661,+ 3759465, -1596822, 2454145, -822541, 963363710,+ -409185979, 628875181, -210776307, 3956944, 1979497,+ -1009365, 27812, 1013967746, 507246529, -258649997,+ 7126831, 274060, 3121440, 3222807, -4183372,+ 70227934, 799869667, 825844983, -1071989969, 3716946,+ 2296397, 3965306, -87208, 952468207, 588452222,+ 1016110510, -22347069, 3284915, 3956745, -636927,+ -1182243, 841760171, 1013916752, -163212680, -302950022,+ -3852015, 2635473, -1277625, -3073009, -987079667,+ 675340520, -327391679, -787459213, -2772600, 1780227,+ 1455890, 1935420, -710479343, 456183549, 373072124,+ 495951789, 59148, -2660408, 2659525, -1753,+ 15156688, -681730119, 681503850, -449207,+};++MLD_ALIGN MLD_INTERNAL_DATA_DEFINITION const int32_t+ mld_aarch64_intt_zetas_layer123456[160] = {+ -2283733, -585207070, 0, 0, -1858416, -476219497,+ -3345963, -857403734, -2815639, -721508096, 0, 0,+ -1853806, -475038184, -2917338, -747568486, 3585098, 918682129,+ 0, 0, -3870317, -991769559, -556856, -142694469,+ 642628, 164673562, 0, 0, -3192354, -818041395,+ 2897314, 742437332, -1460718, -374309300, 0, 0,+ 3950053, 1012201926, 1716988, 439978542, -2453983, -628833668,+ 0, 0, 1935799, 496048908, -3756790, -962678241,+ -1714295, -439288460, 0, 0, 3574466, 915957677,+ 817536, 209493775, 3227876, 827143915, 0, 0,+ -1759347, -450833045, -3415069, -875112161, 1335936, 342333886,+ 0, 0, -2156050, -552488273, -3241972, -830756018,+ -676590, -173376332, 0, 0, 4018989, 1029866791,+ -2071829, -530906624, 434125, 111244624, 0, 0,+ 3506380, 898510625, -1095468, -280713909, 3524442, 903139016,+ 0, 0, -928749, -237992130, -394148, -101000509,+ 1674615, 429120452, 0, 0, -1159875, -297218217,+ -3704823, -949361686, -2663378, -682491182, 0, 0,+ -2101410, -538486762, 3110818, 797147778, 4063053, 1041158200,+ 0, 0, 3586446, 919027554, -2740543, -702264730,+ 3370349, 863652652, 0, 0, -3182878, -815613168,+ -3602218, -923069133, -294725, -75523344, -3761513, -963888510,+ -3765607, -964937599, 3201430, 820367122, 3145678, 806080660,+ 2883726, 738955404, 3201494, 820383522, 1221177, 312926867,+ -557458, -142848732, 1005239, 257592709, -3764867, -964747974,+ -2129892, -545785280, -2682288, -687336873, -3542485, -907762539,+ 601683, 154181397, 0, 0,+};++#else /* MLD_ARITH_BACKEND_AARCH64 && !MLD_CONFIG_MULTILEVEL_NO_SHARED */++MLD_EMPTY_CU(aarch64_zetas)++#endif /* !(MLD_ARITH_BACKEND_AARCH64 && !MLD_CONFIG_MULTILEVEL_NO_SHARED) */
@@ -0,0 +1,367 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#ifndef MLD_NATIVE_AARCH64_SRC_ARITH_NATIVE_AARCH64_H+#define MLD_NATIVE_AARCH64_SRC_ARITH_NATIVE_AARCH64_H++#include "../../../cbmc.h"+#include "../../../common.h"++#define mld_aarch64_ntt_zetas_layer123456 \+ MLD_NAMESPACE(aarch64_ntt_zetas_layer123456)+#define mld_aarch64_ntt_zetas_layer78 MLD_NAMESPACE(aarch64_ntt_zetas_layer78)++#define mld_aarch64_intt_zetas_layer78 MLD_NAMESPACE(aarch64_intt_zetas_layer78)+#define mld_aarch64_intt_zetas_layer123456 \+ MLD_NAMESPACE(aarch64_intt_zetas_layer123456)++MLD_INTERNAL_DATA_DECLARATION const int32_t+ mld_aarch64_ntt_zetas_layer123456[144];+MLD_INTERNAL_DATA_DECLARATION const int32_t mld_aarch64_ntt_zetas_layer78[384];++MLD_INTERNAL_DATA_DECLARATION const int32_t mld_aarch64_intt_zetas_layer78[384];+MLD_INTERNAL_DATA_DECLARATION const int32_t+ mld_aarch64_intt_zetas_layer123456[160];++#define mld_rej_uniform_table MLD_NAMESPACE(rej_uniform_table)+MLD_INTERNAL_DATA_DECLARATION const uint8_t mld_rej_uniform_table[256];+#if !defined(MLD_CONFIG_NO_KEYPAIR_API)+#define mld_rej_uniform_eta_table MLD_NAMESPACE(rej_uniform_eta_table)+MLD_INTERNAL_DATA_DECLARATION const uint8_t mld_rej_uniform_eta_table[4096];+#endif++#if !defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API)+#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLD_CONFIG_PARAMETER_SET == 44+#define mld_polyz_unpack_17_indices MLD_NAMESPACE(polyz_unpack_17_indices)+MLD_INTERNAL_DATA_DECLARATION const uint8_t mld_polyz_unpack_17_indices[64];+#endif+#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \+ (MLD_CONFIG_PARAMETER_SET == 65 || MLD_CONFIG_PARAMETER_SET == 87)+#define mld_polyz_unpack_19_indices MLD_NAMESPACE(polyz_unpack_19_indices)+MLD_INTERNAL_DATA_DECLARATION const uint8_t mld_polyz_unpack_19_indices[64];+#endif+#endif /* !MLD_CONFIG_NO_SIGN_API || !MLD_CONFIG_NO_VERIFY_API */+++/*+ * Sampling 256 coefficients mod 15 using rejection sampling from 4 bits.+ * Expected number of required bytes: (256 * (16/15))/2 = 136.5 bytes.+ * We sample 1 block (=136 bytes) of SHAKE256_RATE output initially.+ * Sampling 2 blocks initially results in slightly worse performance.+ */+#define MLD_AARCH64_REJ_UNIFORM_ETA2_BUFLEN (1 * 136)+/*+ * Sampling 256 coefficients mod 9 using rejection sampling from 4 bits.+ * Expected number of required bytes: (256 * (16/9))/2 = 227.5 bytes.+ * We sample 2 blocks (=272 bytes) of SHAKE256_RATE output initially.+ */+#define MLD_AARCH64_REJ_UNIFORM_ETA4_BUFLEN (2 * 136)++#define mld_ntt_aarch64_asm MLD_NAMESPACE(ntt_aarch64_asm)+void mld_ntt_aarch64_asm(int32_t r[MLDSA_N], const int32_t zetas_l123456[144],+ const int32_t zetas_l78[384])+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/aarch64/proofs/mldsa_ntt_aarch64_asm.ml */+__contract__(+ requires(memory_no_alias(r, sizeof(int32_t) * MLDSA_N))+ /* check-magic: 8380417 == MLDSA_Q */+ requires(array_abs_bound(r, 0, MLDSA_N, 8380417))+ requires(zetas_l123456 == mld_aarch64_ntt_zetas_layer123456)+ requires(zetas_l78 == mld_aarch64_ntt_zetas_layer78)+ assigns(memory_slice(r, sizeof(int32_t) * MLDSA_N))+ /* check-magic: off */+ ensures(array_abs_bound(r, 0, MLDSA_N, 75423753))+ /* check-magic: on */+);++#define mld_intt_aarch64_asm MLD_NAMESPACE(intt_aarch64_asm)+void mld_intt_aarch64_asm(int32_t r[MLDSA_N], const int32_t zetas_l78[384],+ const int32_t zetas_l123456[160])+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/aarch64/proofs/mldsa_intt_aarch64_asm.ml */+__contract__(+ requires(memory_no_alias(r, sizeof(int32_t) * MLDSA_N))+ requires(array_abs_bound(r, 0, MLDSA_N, 8380417))+ requires(zetas_l78 == mld_aarch64_intt_zetas_layer78)+ requires(zetas_l123456 == mld_aarch64_intt_zetas_layer123456)+ assigns(memory_slice(r, sizeof(int32_t) * MLDSA_N))+ /* check-magic: off */+ ensures(array_abs_bound(r, 0, MLDSA_N, 8380417))+ /* check-magic: on */+);++#define mld_rej_uniform_aarch64_asm MLD_NAMESPACE(rej_uniform_aarch64_asm)+MLD_MUST_CHECK_RETURN_VALUE+uint64_t mld_rej_uniform_aarch64_asm(int32_t r[MLDSA_N], const uint8_t *buf,+ unsigned buflen, const uint8_t table[256])+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/aarch64/proofs/mldsa_rej_uniform_aarch64_asm.ml. */+__contract__(+ requires(buflen % 24 == 0)+ requires(memory_no_alias(buf, buflen))+ requires(table == mld_rej_uniform_table)+ requires(memory_no_alias(r, sizeof(int32_t) * MLDSA_N))+ assigns(memory_slice(r, sizeof(int32_t) * MLDSA_N))+ ensures(return_value <= MLDSA_N)+ ensures(array_bound(r, 0, (unsigned) return_value, 0, MLDSA_Q))+);++#if !defined(MLD_CONFIG_NO_KEYPAIR_API)+#define mld_rej_uniform_eta2_aarch64_asm \+ MLD_NAMESPACE(rej_uniform_eta2_aarch64_asm)+MLD_MUST_CHECK_RETURN_VALUE+uint64_t mld_rej_uniform_eta2_aarch64_asm(int32_t r[MLDSA_N],+ const uint8_t *buf, unsigned buflen,+ const uint8_t table[4096])+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/aarch64/proofs/mldsa_rej_uniform_eta2_aarch64_asm.ml */+__contract__(+ requires(buflen % 8 == 0)+ requires(buflen >= 8)+ requires(memory_no_alias(r, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(buf, buflen))+ requires(table == mld_rej_uniform_eta_table)+ assigns(memory_slice(r, sizeof(int32_t) * MLDSA_N))+ ensures(return_value <= MLDSA_N)+ /* check-magic: 3 == 2 + 1 (asm is eta=2-specific) */+ ensures(array_abs_bound(r, 0, return_value, 3))+);++#define mld_rej_uniform_eta4_aarch64_asm \+ MLD_NAMESPACE(rej_uniform_eta4_aarch64_asm)+MLD_MUST_CHECK_RETURN_VALUE+uint64_t mld_rej_uniform_eta4_aarch64_asm(int32_t r[MLDSA_N],+ const uint8_t *buf, unsigned buflen,+ const uint8_t table[4096])+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/aarch64/proofs/mldsa_rej_uniform_eta4_aarch64_asm.ml */+__contract__(+ requires(buflen % 8 == 0)+ requires(buflen >= 8)+ requires(memory_no_alias(r, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(buf, buflen))+ requires(table == mld_rej_uniform_eta_table)+ assigns(memory_slice(r, sizeof(int32_t) * MLDSA_N))+ ensures(return_value <= MLDSA_N)+ /* check-magic: 5 == 4 + 1 (asm is eta=4-specific) */+ ensures(array_abs_bound(r, 0, return_value, 5))+);+#endif /* !MLD_CONFIG_NO_KEYPAIR_API */++#if !defined(MLD_CONFIG_NO_SIGN_API)+#define mld_poly_decompose_32_aarch64_asm \+ MLD_NAMESPACE(poly_decompose_32_aarch64_asm)+void mld_poly_decompose_32_aarch64_asm(int32_t a1[MLDSA_N], int32_t a0[MLDSA_N])+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/aarch64/proofs/mldsa_poly_decompose_32_aarch64_asm.ml */+__contract__(+ requires(memory_no_alias(a1, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(a0, sizeof(int32_t) * MLDSA_N))+ requires(array_bound(a0, 0, MLDSA_N, 0, MLDSA_Q))+ assigns(memory_slice(a1, sizeof(int32_t) * MLDSA_N))+ assigns(memory_slice(a0, sizeof(int32_t) * MLDSA_N))+ /* check-magic: 16 == (MLDSA_Q - 1) / (2 * ((MLDSA_Q - 1) / 32)) */+ ensures(array_bound(a1, 0, MLDSA_N, 0, 16))+ /* check-magic: 261889 == (MLDSA_Q - 1) / 32 + 1 */+ ensures(array_abs_bound(a0, 0, MLDSA_N, 261889))+);++#define mld_poly_decompose_88_aarch64_asm \+ MLD_NAMESPACE(poly_decompose_88_aarch64_asm)+void mld_poly_decompose_88_aarch64_asm(int32_t a1[MLDSA_N], int32_t a0[MLDSA_N])+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/aarch64/proofs/mldsa_poly_decompose_88_aarch64_asm.ml */+__contract__(+ requires(memory_no_alias(a1, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(a0, sizeof(int32_t) * MLDSA_N))+ requires(array_bound(a0, 0, MLDSA_N, 0, MLDSA_Q))+ assigns(memory_slice(a1, sizeof(int32_t) * MLDSA_N))+ assigns(memory_slice(a0, sizeof(int32_t) * MLDSA_N))+ /* check-magic: 44 == (MLDSA_Q - 1) / (2 * ((MLDSA_Q - 1) / 88)) */+ ensures(array_bound(a1, 0, MLDSA_N, 0, 44))+ /* check-magic: 95233 == (MLDSA_Q - 1) / 88 + 1 */+ ensures(array_abs_bound(a0, 0, MLDSA_N, 95233))+);+#endif /* !MLD_CONFIG_NO_SIGN_API */++#define mld_poly_caddq_aarch64_asm MLD_NAMESPACE(poly_caddq_aarch64_asm)+void mld_poly_caddq_aarch64_asm(int32_t a[MLDSA_N])+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/aarch64/proofs/mldsa_poly_caddq_aarch64_asm.ml */+__contract__(+ requires(memory_no_alias(a, sizeof(int32_t) * MLDSA_N))+ requires(array_abs_bound(a, 0, MLDSA_N, MLDSA_Q))+ assigns(memory_slice(a, sizeof(int32_t) * MLDSA_N))+ ensures(array_bound(a, 0, MLDSA_N, 0, MLDSA_Q))+);++#if !defined(MLD_CONFIG_NO_VERIFY_API)+#define mld_poly_use_hint_32_aarch64_asm \+ MLD_NAMESPACE(poly_use_hint_32_aarch64_asm)+void mld_poly_use_hint_32_aarch64_asm(int32_t a[MLDSA_N],+ const int32_t h[MLDSA_N])+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/aarch64/proofs/mldsa_poly_use_hint_32_aarch64_asm.ml */+__contract__(+ requires(memory_no_alias(a, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(h, sizeof(int32_t) * MLDSA_N))+ requires(array_bound(a, 0, MLDSA_N, 0, MLDSA_Q))+ requires(array_bound(h, 0, MLDSA_N, 0, 2))+ assigns(memory_slice(a, sizeof(int32_t) * MLDSA_N))+ ensures(array_bound(a, 0, MLDSA_N, 0, 16))+);++#define mld_poly_use_hint_88_aarch64_asm \+ MLD_NAMESPACE(poly_use_hint_88_aarch64_asm)+void mld_poly_use_hint_88_aarch64_asm(int32_t a[MLDSA_N],+ const int32_t h[MLDSA_N])+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/aarch64/proofs/mldsa_poly_use_hint_88_aarch64_asm.ml */+__contract__(+ requires(memory_no_alias(a, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(h, sizeof(int32_t) * MLDSA_N))+ requires(array_bound(a, 0, MLDSA_N, 0, MLDSA_Q))+ requires(array_bound(h, 0, MLDSA_N, 0, 2))+ assigns(memory_slice(a, sizeof(int32_t) * MLDSA_N))+ ensures(array_bound(a, 0, MLDSA_N, 0, 44))+);+#endif /* !MLD_CONFIG_NO_VERIFY_API */++#define mld_poly_chknorm_aarch64_asm MLD_NAMESPACE(poly_chknorm_aarch64_asm)+MLD_MUST_CHECK_RETURN_VALUE+int mld_poly_chknorm_aarch64_asm(const int32_t a[MLDSA_N], int32_t B)+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/aarch64/proofs/mldsa_poly_chknorm_aarch64_asm.ml */+__contract__(+ requires(memory_no_alias(a, sizeof(int32_t) * MLDSA_N))+ /* HOL Light precondition: abs(ival(x i)) < 2^31, i.e., a[i] != INT32_MIN */+ requires(forall(k0, 0, MLDSA_N, a[k0] > INT32_MIN))+ ensures(return_value == 0 || return_value == 1)+ ensures((return_value == 0) == array_abs_bound(a, 0, MLDSA_N, B))+);++#if !defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API)+#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLD_CONFIG_PARAMETER_SET == 44+#define mld_polyz_unpack_17_aarch64_asm \+ MLD_NAMESPACE(polyz_unpack_17_aarch64_asm)+void mld_polyz_unpack_17_aarch64_asm(int32_t r[MLDSA_N], const uint8_t buf[576],+ const uint8_t indices[64])+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/aarch64/proofs/mldsa_polyz_unpack_17_aarch64_asm.ml */+__contract__(+ requires(memory_no_alias(r, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(buf, 576))+ requires(indices == mld_polyz_unpack_17_indices)+ assigns(memory_slice(r, sizeof(int32_t) * MLDSA_N))+ ensures(array_bound(r, 0, MLDSA_N, -((1 << 17) - 1), (1 << 17) + 1))+);+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 44 \+ */++#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \+ (MLD_CONFIG_PARAMETER_SET == 65 || MLD_CONFIG_PARAMETER_SET == 87)+#define mld_polyz_unpack_19_aarch64_asm \+ MLD_NAMESPACE(polyz_unpack_19_aarch64_asm)+void mld_polyz_unpack_19_aarch64_asm(int32_t r[MLDSA_N], const uint8_t buf[640],+ const uint8_t indices[64])+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/aarch64/proofs/mldsa_polyz_unpack_19_aarch64_asm.ml */+__contract__(+ requires(memory_no_alias(r, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(buf, 640))+ requires(indices == mld_polyz_unpack_19_indices)+ assigns(memory_slice(r, sizeof(int32_t) * MLDSA_N))+ ensures(array_bound(r, 0, MLDSA_N, -((1 << 19) - 1), (1 << 19) + 1))+);+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 65 \+ || MLD_CONFIG_PARAMETER_SET == 87 */+#endif /* !MLD_CONFIG_NO_SIGN_API || !MLD_CONFIG_NO_VERIFY_API */++#if !defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API) || \+ defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)+#define mld_poly_pointwise_montgomery_aarch64_asm \+ MLD_NAMESPACE(poly_pointwise_montgomery_aarch64_asm)+void mld_poly_pointwise_montgomery_aarch64_asm(int32_t a[MLDSA_N],+ const int32_t b[MLDSA_N])+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/aarch64/proofs/mldsa_pointwise_montgomery_aarch64_asm.ml+ */+__contract__(+ requires(memory_no_alias(a, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(b, sizeof(int32_t) * MLDSA_N))+ /* Input bound MLD_NTT_BOUND = 9 * MLD_FQMUL_BOUND, the guaranteed bound of+ * any forward NTT implementation. Hardcoded here to keep this header free+ * of poly.h. */+ /* check-magic: 94279698 == 9 * ((5 * MLDSA_Q + 3) / 4) */+ requires(array_abs_bound(a, 0, MLDSA_N, 94279698))+ requires(array_abs_bound(b, 0, MLDSA_N, 94279698))+ assigns(memory_slice(a, sizeof(int32_t) * MLDSA_N))+ ensures(array_abs_bound(a, 0, MLDSA_N, 8380417))+);+#endif /* !MLD_CONFIG_NO_SIGN_API || !MLD_CONFIG_NO_VERIFY_API || \+ MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST */++#define mld_polyvecl_pointwise_acc_montgomery_l4_aarch64_asm \+ MLD_NAMESPACE(polyvecl_pointwise_acc_montgomery_l4_aarch64_asm)+void mld_polyvecl_pointwise_acc_montgomery_l4_aarch64_asm(+ int32_t r[MLDSA_N], const int32_t a[4][MLDSA_N],+ const int32_t b[4][MLDSA_N])+/* This must be kept in sync with the HOL-Light specification+ * in+ * proofs/hol_light/aarch64/proofs/mldsa_polyvecl_pointwise_acc_montgomery_l4_aarch64_asm.ml+ */+__contract__(+ requires(memory_no_alias(r, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(a, sizeof(int32_t) * 4 * MLDSA_N))+ requires(memory_no_alias(b, sizeof(int32_t) * 4 * MLDSA_N))+ requires(forall(l0, 0, 4, array_abs_bound(a[l0], 0, MLDSA_N, 8380417)))+ /* check-magic: 94279698 == 9 * ((5 * MLDSA_Q + 3) / 4) */+ requires(forall(l1, 0, 4, array_abs_bound(b[l1], 0, MLDSA_N, 94279698)))+ assigns(memory_slice(r, sizeof(int32_t) * MLDSA_N))+ ensures(array_abs_bound(r, 0, MLDSA_N, 8380417))+);++#define mld_polyvecl_pointwise_acc_montgomery_l5_aarch64_asm \+ MLD_NAMESPACE(polyvecl_pointwise_acc_montgomery_l5_aarch64_asm)+void mld_polyvecl_pointwise_acc_montgomery_l5_aarch64_asm(+ int32_t r[MLDSA_N], const int32_t a[5][MLDSA_N],+ const int32_t b[5][MLDSA_N])+/* This must be kept in sync with the HOL-Light specification+ * in+ * proofs/hol_light/aarch64/proofs/mldsa_polyvecl_pointwise_acc_montgomery_l5_aarch64_asm.ml+ */+__contract__(+ requires(memory_no_alias(r, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(a, sizeof(int32_t) * 5 * MLDSA_N))+ requires(memory_no_alias(b, sizeof(int32_t) * 5 * MLDSA_N))+ requires(forall(l0, 0, 5, array_abs_bound(a[l0], 0, MLDSA_N, 8380417)))+ /* check-magic: 94279698 == 9 * ((5 * MLDSA_Q + 3) / 4) */+ requires(forall(l1, 0, 5, array_abs_bound(b[l1], 0, MLDSA_N, 94279698)))+ assigns(memory_slice(r, sizeof(int32_t) * MLDSA_N))+ ensures(array_abs_bound(r, 0, MLDSA_N, 8380417))+);++#define mld_polyvecl_pointwise_acc_montgomery_l7_aarch64_asm \+ MLD_NAMESPACE(polyvecl_pointwise_acc_montgomery_l7_aarch64_asm)+void mld_polyvecl_pointwise_acc_montgomery_l7_aarch64_asm(+ int32_t r[MLDSA_N], const int32_t a[7][MLDSA_N],+ const int32_t b[7][MLDSA_N])+/* This must be kept in sync with the HOL-Light specification+ * in+ * proofs/hol_light/aarch64/proofs/mldsa_polyvecl_pointwise_acc_montgomery_l7_aarch64_asm.ml+ */+__contract__(+ requires(memory_no_alias(r, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(a, sizeof(int32_t) * 7 * MLDSA_N))+ requires(memory_no_alias(b, sizeof(int32_t) * 7 * MLDSA_N))+ requires(forall(l0, 0, 7, array_abs_bound(a[l0], 0, MLDSA_N, 8380417)))+ /* check-magic: 94279698 == 9 * ((5 * MLDSA_Q + 3) / 4) */+ requires(forall(l1, 0, 7, array_abs_bound(b[l1], 0, MLDSA_N, 94279698)))+ assigns(memory_slice(r, sizeof(int32_t) * MLDSA_N))+ ensures(array_abs_bound(r, 0, MLDSA_N, 8380417))+);++#endif /* !MLD_NATIVE_AARCH64_SRC_ARITH_NATIVE_AARCH64_H */
@@ -0,0 +1,786 @@+/* Copyright (c) 2022 Arm Limited+ * Copyright (c) 2022 Hanno Becker+ * Copyright (c) 2023 Amin Abdulrahman, Matthias Kannwischer+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [NeonNTT]+ * Neon NTT: Faster Dilithium, Kyber, and Saber on Cortex-A72 and Apple M1+ * Becker, Hwang, Kannwischer, Yang, Yang+ * https://eprint.iacr.org/2021/986+ *+ * - [NeonNTT_Autoformalised]+ * Neon NTT - (Auto)formalised+ * Hanno Becker+ * https://eprint.iacr.org/2026/1223+ *+ * - [SLOTHY_Paper]+ * Fast and Clean: Auditable high-performance assembly via constraint solving+ * Abdulrahman, Becker, Kannwischer, Klein+ * https://eprint.iacr.org/2022/1303+ */++/* AArch64 ML-DSA inverse NTT following @[NeonNTT], @[SLOTHY_Paper], and @[NeonNTT_Autoformalised] */++/*yaml+ Name: intt_aarch64_asm+ Description: AArch64 ML-DSA inverse NTT+ Signature: void mld_intt_aarch64_asm(int32_t r[256], const int32_t zetas_l78[384], const int32_t zetas_l123456[160])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON]+ x0:+ type: buffer+ size_bytes: 1024+ permissions: read/write+ c_parameter: int32_t r[256]+ description: Input/output polynomial (256 x int32_t)+ x1:+ type: buffer+ size_bytes: 1536+ permissions: read-only+ c_parameter: const int32_t zetas_l78[384]+ description: Twiddle factors for layers 7-8 (384 x int32_t)+ x2:+ type: buffer+ size_bytes: 640+ permissions: read-only+ c_parameter: const int32_t zetas_l123456[160]+ description: Twiddle factors for layers 1-6 (160 x int32_t)+*/++#include "../../../common.h"+#if defined(MLD_ARITH_BACKEND_AARCH64) && !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/aarch64_opt/src/mldsa_intt_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(intt_aarch64_asm)+MLD_ASM_FN_SYMBOL(intt_aarch64_asm)++ .cfi_startproc+ sub sp, sp, #0x40+ .cfi_adjust_cfa_offset 0x40+ stp d8, d9, [sp]+ .cfi_rel_offset d8, 0x0+ .cfi_rel_offset d9, 0x8+ stp d10, d11, [sp, #0x10]+ .cfi_rel_offset d10, 0x10+ .cfi_rel_offset d11, 0x18+ stp d12, d13, [sp, #0x20]+ .cfi_rel_offset d12, 0x20+ .cfi_rel_offset d13, 0x28+ stp d14, d15, [sp, #0x30]+ .cfi_rel_offset d14, 0x30+ .cfi_rel_offset d15, 0x38+ mov w5, #0xe001 // =57345+ movk w5, #0x7f, lsl #16+ dup v31.4s, w5+ mov x3, x0+ mov x4, #0x10 // =16+ ldr q27, [x3, #0x10]+ ldr q18, [x3]+ ldr q3, [x3, #0x20]+ ldr q13, [x3, #0x30]+ ldr q2, [x1, #0x30]+ ldr q8, [x3, #0x70]+ ldr q21, [x3, #0x60]+ trn1 v10.4s, v18.4s, v27.4s+ trn2 v23.4s, v18.4s, v27.4s+ trn1 v6.4s, v3.4s, v13.4s+ trn2 v18.4s, v3.4s, v13.4s+ ldr q12, [x1, #0x50]+ ldr q17, [x1, #0x40]+ trn2 v29.2d, v10.2d, v6.2d+ trn2 v14.2d, v23.2d, v18.2d+ trn1 v10.2d, v10.2d, v6.2d+ trn1 v26.2d, v23.2d, v18.2d+ sub v3.4s, v29.4s, v14.4s+ trn1 v13.4s, v21.4s, v8.4s+ ldr q6, [x1, #0x10]+ add v24.4s, v10.4s, v26.4s+ sub v30.4s, v10.4s, v26.4s+ sqrdmulh v12.4s, v3.4s, v12.4s+ ldr q9, [x1, #0x20]+ sqrdmulh v5.4s, v30.4s, v2.4s+ mul v4.4s, v3.4s, v17.4s+ ldr q3, [x3, #0x50]+ add v10.4s, v29.4s, v14.4s+ ldr q26, [x3, #0x40]+ mul v15.4s, v30.4s, v9.4s+ trn2 v30.4s, v21.4s, v8.4s+ sub v21.4s, v24.4s, v10.4s+ add v29.4s, v24.4s, v10.4s+ mls v15.4s, v5.4s, v31.s[0]+ ldr q16, [x1], #0x60+ trn2 v10.4s, v26.4s, v3.4s+ mls v4.4s, v12.4s, v31.s[0]+ trn1 v3.4s, v26.4s, v3.4s+ ldr q0, [x1, #0x50]+ trn2 v25.2d, v10.2d, v30.2d+ sqrdmulh v12.4s, v21.4s, v6.4s+ trn2 v1.2d, v3.2d, v13.2d+ mul v21.4s, v21.4s, v16.4s+ sub v23.4s, v1.4s, v25.4s+ sub v2.4s, v15.4s, v4.4s+ add v20.4s, v15.4s, v4.4s+ sqrdmulh v4.4s, v23.4s, v0.4s+ trn1 v7.2d, v3.2d, v13.2d+ sqrdmulh v3.4s, v2.4s, v6.4s+ trn1 v15.2d, v10.2d, v30.2d+ mls v21.4s, v12.4s, v31.s[0]+ ldr q12, [x1, #0x30]+ sub v13.4s, v7.4s, v15.4s+ mul v10.4s, v2.4s, v16.4s+ ldr q16, [x1, #0x40]+ ldr q5, [x1, #0x20]+ mls v10.4s, v3.4s, v31.s[0]+ trn2 v11.4s, v29.4s, v20.4s+ sqrdmulh v2.4s, v13.4s, v12.4s+ ldr d9, [x2], #0x20+ mul v17.4s, v13.4s, v5.4s+ trn1 v24.4s, v29.4s, v20.4s+ trn1 v12.4s, v21.4s, v10.4s+ trn2 v10.4s, v21.4s, v10.4s+ mul v22.4s, v23.4s, v16.4s+ ldur q26, [x2, #-0x10]+ trn1 v13.2d, v24.2d, v12.2d+ trn1 v3.2d, v11.2d, v10.2d+ mls v17.4s, v2.4s, v31.s[0]+ mls v22.4s, v4.4s, v31.s[0]+ sub v23.4s, v13.4s, v3.4s+ trn2 v10.2d, v11.2d, v10.2d+ sqrdmulh v0.4s, v23.4s, v26.s[1]+ trn2 v21.2d, v24.2d, v12.2d+ mul v29.4s, v23.4s, v26.s[0]+ sub v23.4s, v21.4s, v10.4s+ sqrdmulh v8.4s, v23.4s, v26.s[3]+ add v6.4s, v1.4s, v25.4s+ add v11.4s, v21.4s, v10.4s+ mls v29.4s, v0.4s, v31.s[0]+ add v30.4s, v13.4s, v3.4s+ ldr q2, [x1, #0x10]+ mul v3.4s, v23.4s, v26.s[2]+ add v12.4s, v7.4s, v15.4s+ mls v3.4s, v8.4s, v31.s[0]+ sub v13.4s, v30.4s, v11.4s+ sub v21.4s, v12.4s, v6.4s+ add v26.4s, v17.4s, v22.4s+ sqrdmulh v28.4s, v13.4s, v9.s[1]+ mul v18.4s, v13.4s, v9.s[0]+ sub v8.4s, v29.4s, v3.4s+ sqrdmulh v24.4s, v21.4s, v2.4s+ add v15.4s, v30.4s, v11.4s+ add v10.4s, v29.4s, v3.4s+ add v14.4s, v12.4s, v6.4s+ sqrdmulh v16.4s, v8.4s, v9.s[1]+ sub v3.4s, v17.4s, v22.4s+ mul v8.4s, v8.4s, v9.s[0]+ ldr q17, [x1], #0x60+ sub x4, x4, #0x2++Lmld_intt_layer5678_start:+ ldr d4, [x2], #0x20+ mul v0.4s, v21.4s, v17.4s+ ldr q7, [x3, #0xa0]+ ldr q27, [x3, #0xb0]+ mls v8.4s, v16.4s, v31.s[0]+ ldr q29, [x3, #0x80]+ trn1 v9.4s, v14.4s, v26.4s+ ldr q25, [x3, #0x90]+ mls v0.4s, v24.4s, v31.s[0]+ str q15, [x3], #0x40+ trn1 v1.4s, v7.4s, v27.4s+ ldr q15, [x1, #0x50]+ trn2 v12.4s, v7.4s, v27.4s+ sqrdmulh v16.4s, v3.4s, v2.4s+ trn1 v21.4s, v29.4s, v25.4s+ stur q8, [x3, #-0x10]+ trn2 v19.4s, v29.4s, v25.4s+ mls v18.4s, v28.4s, v31.s[0]+ ldr q22, [x1, #0x30]+ trn2 v13.2d, v21.2d, v1.2d+ trn2 v5.2d, v19.2d, v12.2d+ mul v23.4s, v3.4s, v17.4s+ trn1 v17.2d, v21.2d, v1.2d+ ldr q29, [x1, #0x40]+ mls v23.4s, v16.4s, v31.s[0]+ sub v30.4s, v13.4s, v5.4s+ trn1 v24.2d, v19.2d, v12.2d+ stur q10, [x3, #-0x30]+ sqrdmulh v16.4s, v30.4s, v15.4s+ ldr q10, [x1, #0x20]+ trn2 v12.4s, v14.4s, v26.4s+ stur q18, [x3, #-0x20]+ mul v11.4s, v30.4s, v29.4s+ sub v8.4s, v17.4s, v24.4s+ trn1 v20.4s, v0.4s, v23.4s+ ldr q2, [x1, #0x10]+ trn2 v26.4s, v0.4s, v23.4s+ sqrdmulh v19.4s, v8.4s, v22.4s+ ldur q14, [x2, #-0x10]+ trn1 v21.2d, v9.2d, v20.2d+ mul v8.4s, v8.4s, v10.4s+ trn1 v6.2d, v12.2d, v26.2d+ trn2 v0.2d, v9.2d, v20.2d+ mls v11.4s, v16.4s, v31.s[0]+ sub v7.4s, v21.4s, v6.4s+ trn2 v28.2d, v12.2d, v26.2d+ add v10.4s, v21.4s, v6.4s+ mul v18.4s, v7.4s, v14.s[0]+ add v20.4s, v13.4s, v5.4s+ sqrdmulh v30.4s, v7.4s, v14.s[1]+ sub v21.4s, v0.4s, v28.4s+ add v13.4s, v0.4s, v28.4s+ add v9.4s, v17.4s, v24.4s+ sqrdmulh v1.4s, v21.4s, v14.s[3]+ ldr q17, [x1], #0x60+ mul v14.4s, v21.4s, v14.s[2]+ sub v21.4s, v9.4s, v20.4s+ mls v18.4s, v30.4s, v31.s[0]+ mls v14.4s, v1.4s, v31.s[0]+ mls v8.4s, v19.4s, v31.s[0]+ sub v3.4s, v10.4s, v13.4s+ add v15.4s, v10.4s, v13.4s+ sqrdmulh v28.4s, v3.4s, v4.s[1]+ add v10.4s, v18.4s, v14.4s+ sub v19.4s, v18.4s, v14.4s+ mul v18.4s, v3.4s, v4.s[0]+ sub v3.4s, v8.4s, v11.4s+ add v26.4s, v8.4s, v11.4s+ sqrdmulh v16.4s, v19.4s, v4.s[1]+ add v14.4s, v9.4s, v20.4s+ mul v8.4s, v19.4s, v4.s[0]+ sqrdmulh v24.4s, v21.4s, v2.4s+ subs x4, x4, #0x1+ cbnz x4, Lmld_intt_layer5678_start+ sqrdmulh v19.4s, v3.4s, v2.4s+ trn2 v0.4s, v14.4s, v26.4s+ str q15, [x3], #0x40+ trn1 v30.4s, v14.4s, v26.4s+ mul v26.4s, v21.4s, v17.4s+ stur q10, [x3, #-0x30]+ ldr d29, [x2], #0x20+ mls v26.4s, v24.4s, v31.s[0]+ mul v10.4s, v3.4s, v17.4s+ ldur q14, [x2, #-0x10]+ mls v10.4s, v19.4s, v31.s[0]+ mls v8.4s, v16.4s, v31.s[0]+ trn2 v22.4s, v26.4s, v10.4s+ trn1 v25.4s, v26.4s, v10.4s+ trn1 v1.2d, v30.2d, v25.2d+ trn1 v2.2d, v0.2d, v22.2d+ trn2 v13.2d, v30.2d, v25.2d+ trn2 v7.2d, v0.2d, v22.2d+ mls v18.4s, v28.4s, v31.s[0]+ sub v22.4s, v1.4s, v2.4s+ add v3.4s, v13.4s, v7.4s+ sqrdmulh v27.4s, v22.4s, v14.s[1]+ sub v4.4s, v13.4s, v7.4s+ stur q8, [x3, #-0x10]+ sqrdmulh v0.4s, v4.4s, v14.s[3]+ add v16.4s, v1.4s, v2.4s+ stur q18, [x3, #-0x20]+ mul v23.4s, v4.4s, v14.s[2]+ add v4.4s, v16.4s, v3.4s+ sub v17.4s, v16.4s, v3.4s+ mul v14.4s, v22.4s, v14.s[0]+ str q4, [x3], #0x40+ mls v23.4s, v0.4s, v31.s[0]+ mls v14.4s, v27.4s, v31.s[0]+ mul v11.4s, v17.4s, v29.s[0]+ sqrdmulh v10.4s, v17.4s, v29.s[1]+ sub v0.4s, v14.4s, v23.4s+ add v6.4s, v14.4s, v23.4s+ sqrdmulh v12.4s, v0.4s, v29.s[1]+ mul v0.4s, v0.4s, v29.s[0]+ stur q6, [x3, #-0x30]+ mls v11.4s, v10.4s, v31.s[0]+ mls v0.4s, v12.4s, v31.s[0]+ stur q11, [x3, #-0x20]+ stur q0, [x3, #-0x10]+ mov w5, #0x3ffe // =16382+ dup v29.4s, w5+ mov w5, #0xe03 // =3587+ movk w5, #0x40, lsl #16+ dup v30.4s, w5+ mov x4, #0x4 // =4+ ldr q0, [x2], #0x80+ ldur q1, [x2, #-0x70]+ ldur q2, [x2, #-0x60]+ ldur q3, [x2, #-0x50]+ ldur q4, [x2, #-0x40]+ ldur q5, [x2, #-0x30]+ ldur q6, [x2, #-0x20]+ ldur q7, [x2, #-0x10]+ ldr q8, [x0, #0xc0]+ ldr q27, [x0, #0x80]+ ldr q20, [x0, #0x1c0]+ ldr q23, [x0, #0x180]+ ldr q24, [x0, #0x3c0]+ ldr q28, [x0, #0x40]+ ldr q25, [x0, #0x340]+ ldr q10, [x0, #0x380]+ sub v15.4s, v27.4s, v8.4s+ ldr q26, [x0]+ ldr q18, [x0, #0x300]+ add v9.4s, v23.4s, v20.4s+ mul v11.4s, v15.4s, v4.s[0]+ sub v19.4s, v23.4s, v20.4s+ sub v22.4s, v10.4s, v24.4s+ ldr q13, [x0, #0x240]+ sqrdmulh v12.4s, v15.4s, v4.s[1]+ sub v14.4s, v26.4s, v28.4s+ ldr q17, [x0, #0x200]+ add v23.4s, v18.4s, v25.4s+ sqrdmulh v15.4s, v14.4s, v3.s[3]+ sub v18.4s, v18.4s, v25.4s+ mul v25.4s, v14.4s, v3.s[2]+ sub v16.4s, v17.4s, v13.4s+ mls v11.4s, v12.4s, v31.s[0]+ add v21.4s, v10.4s, v24.4s+ mls v25.4s, v15.4s, v31.s[0]+ mul v12.4s, v16.4s, v5.s[2]+ sub v10.4s, v23.4s, v21.4s+ sqrdmulh v14.4s, v10.4s, v3.s[1]+ add v23.4s, v23.4s, v21.4s+ add v27.4s, v27.4s, v8.4s+ mul v15.4s, v10.4s, v3.s[0]+ add v24.4s, v25.4s, v11.4s+ sub v10.4s, v25.4s, v11.4s+ ldr q25, [x0, #0x100]+ sqrdmulh v8.4s, v18.4s, v6.s[3]+ add v21.4s, v26.4s, v28.4s+ ldr q26, [x0, #0x140]+ sqrdmulh v20.4s, v16.4s, v5.s[3]+ sqrdmulh v28.4s, v22.4s, v7.s[1]+ add v11.4s, v25.4s, v26.4s+ mls v15.4s, v14.4s, v31.s[0]+ add v17.4s, v17.4s, v13.4s+ mul v13.4s, v22.4s, v7.s[0]+ ldr q14, [x0, #0x280]+ ldr q22, [x0, #0x2c0]+ mls v13.4s, v28.4s, v31.s[0]+ sub v16.4s, v25.4s, v26.4s+ sub v28.4s, v21.4s, v27.4s+ mls v12.4s, v20.4s, v31.s[0]+ add v20.4s, v11.4s, v9.4s+ add v26.4s, v14.4s, v22.4s+ sub v14.4s, v14.4s, v22.4s+ sqrdmulh v22.4s, v28.4s, v1.s[3]+ sub v9.4s, v11.4s, v9.4s+ mul v25.4s, v14.4s, v6.s[0]+ sub v11.4s, v17.4s, v26.4s+ add v17.4s, v17.4s, v26.4s+ add v26.4s, v21.4s, v27.4s+ sqrdmulh v27.4s, v11.4s, v2.s[3]+ mul v21.4s, v11.4s, v2.s[2]+ mul v11.4s, v18.4s, v6.s[2]+ mls v11.4s, v8.4s, v31.s[0]+ sqrdmulh v8.4s, v14.4s, v6.s[1]+ add v18.4s, v26.4s, v20.4s+ mls v21.4s, v27.4s, v31.s[0]+ sub v27.4s, v17.4s, v23.4s+ mul v14.4s, v28.4s, v1.s[2]+ sub v28.4s, v26.4s, v20.4s+ add v23.4s, v17.4s, v23.4s+ mls v25.4s, v8.4s, v31.s[0]+ add v26.4s, v21.4s, v15.4s+ sub v15.4s, v21.4s, v15.4s+ mul v8.4s, v19.4s, v5.s[0]+ mls v14.4s, v22.4s, v31.s[0]+ sub v22.4s, v11.4s, v13.4s+ sub v20.4s, v12.4s, v25.4s+ add v21.4s, v12.4s, v25.4s+ sqrdmulh v12.4s, v22.4s, v3.s[1]+ mul v17.4s, v16.4s, v4.s[2]+ sqrdmulh v25.4s, v16.4s, v4.s[3]+ add v16.4s, v11.4s, v13.4s+ mul v11.4s, v22.4s, v3.s[0]+ sqrdmulh v19.4s, v19.4s, v5.s[1]+ sqrdmulh v22.4s, v15.4s, v1.s[1]+ sqrdmulh v13.4s, v20.4s, v2.s[3]+ mul v20.4s, v20.4s, v2.s[2]+ mls v11.4s, v12.4s, v31.s[0]+ mls v20.4s, v13.4s, v31.s[0]+ sub v13.4s, v18.4s, v23.4s+ add v23.4s, v18.4s, v23.4s+ mls v17.4s, v25.4s, v31.s[0]+ mls v8.4s, v19.4s, v31.s[0]+ sub v18.4s, v20.4s, v11.4s+ add v20.4s, v20.4s, v11.4s+ sqrdmulh v12.4s, v9.4s, v2.s[1]+ mul v15.4s, v15.4s, v1.s[0]+ sub v11.4s, v17.4s, v8.4s+ mls v15.4s, v22.4s, v31.s[0]+ add v19.4s, v17.4s, v8.4s+ sqrdmulh v25.4s, v11.4s, v2.s[1]+ add v17.4s, v24.4s, v19.4s+ sub v24.4s, v24.4s, v19.4s+ mul v19.4s, v11.4s, v2.s[0]+ sqrdmulh v8.4s, v24.4s, v0.s[3]+ mls v19.4s, v25.4s, v31.s[0]+ mul v25.4s, v9.4s, v2.s[0]+ sub x4, x4, #0x1++Lmld_intt_layer1234_start:+ sub v22.4s, v21.4s, v16.4s+ mls v25.4s, v12.4s, v31.s[0]+ add v12.4s, v21.4s, v16.4s+ mul v9.4s, v24.4s, v0.s[2]+ mls v9.4s, v8.4s, v31.s[0]+ sub v16.4s, v14.4s, v25.4s+ add v14.4s, v14.4s, v25.4s+ sqrdmulh v24.4s, v22.4s, v1.s[1]+ sub v11.4s, v14.4s, v26.4s+ sqrdmulh v21.4s, v16.4s, v0.s[3]+ add v25.4s, v14.4s, v26.4s+ mul v14.4s, v22.4s, v1.s[0]+ mls v14.4s, v24.4s, v31.s[0]+ mul v16.4s, v16.4s, v0.s[2]+ mls v16.4s, v21.4s, v31.s[0]+ add v24.4s, v9.4s, v14.4s+ sqrdmulh v21.4s, v25.4s, v30.4s+ sqrdmulh v26.4s, v11.4s, v0.s[1]+ mul v8.4s, v25.4s, v29.4s+ mls v8.4s, v21.4s, v31.s[0]+ mul v25.4s, v24.4s, v29.4s+ mul v22.4s, v11.4s, v0.s[0]+ sub v11.4s, v17.4s, v12.4s+ str q8, [x0, #0x80]+ sub v8.4s, v9.4s, v14.4s+ add v14.4s, v16.4s, v15.4s+ sqrdmulh v9.4s, v11.4s, v0.s[1]+ sub v21.4s, v16.4s, v15.4s+ mls v22.4s, v26.4s, v31.s[0]+ mul v15.4s, v11.4s, v0.s[0]+ mls v15.4s, v9.4s, v31.s[0]+ add v11.4s, v17.4s, v12.4s+ mul v16.4s, v10.4s, v1.s[2]+ str q22, [x0, #0x280]+ sqrdmulh v26.4s, v10.4s, v1.s[3]+ str q15, [x0, #0x240]+ sqrdmulh v9.4s, v8.4s, v0.s[1]+ mul v12.4s, v8.4s, v0.s[0]+ mls v16.4s, v26.4s, v31.s[0]+ sqrdmulh v10.4s, v27.4s, v1.s[1]+ sqrdmulh v15.4s, v14.4s, v30.4s+ add v22.4s, v16.4s, v19.4s+ sub v19.4s, v16.4s, v19.4s+ mls v12.4s, v9.4s, v31.s[0]+ add v8.4s, v22.4s, v20.4s+ sub v26.4s, v22.4s, v20.4s+ mul v27.4s, v27.4s, v1.s[0]+ sqrdmulh v17.4s, v11.4s, v30.4s+ mls v27.4s, v10.4s, v31.s[0]+ mul v20.4s, v11.4s, v29.4s+ sqrdmulh v22.4s, v24.4s, v30.4s+ mls v20.4s, v17.4s, v31.s[0]+ mul v14.4s, v14.4s, v29.4s+ sqrdmulh v16.4s, v23.4s, v30.4s+ str q20, [x0, #0x40]+ sqrdmulh v11.4s, v21.4s, v0.s[1]+ sqrdmulh v20.4s, v28.4s, v0.s[3]+ mul v17.4s, v23.4s, v29.4s+ mul v23.4s, v28.4s, v0.s[2]+ mls v17.4s, v16.4s, v31.s[0]+ mul v10.4s, v21.4s, v0.s[0]+ str q12, [x0, #0x340]+ mls v25.4s, v22.4s, v31.s[0]+ str q17, [x0], #0x10+ ldr q17, [x0, #0x2c0]+ mls v14.4s, v15.4s, v31.s[0]+ ldr q12, [x0, #0x200]+ sqrdmulh v16.4s, v13.4s, v0.s[1]+ str q25, [x0, #0x130]+ mul v28.4s, v13.4s, v0.s[0]+ ldr q13, [x0, #0x240]+ str q14, [x0, #0x170]+ mul v14.4s, v18.4s, v1.s[0]+ mls v28.4s, v16.4s, v31.s[0]+ ldr q21, [x0, #0xc0]+ mls v10.4s, v11.4s, v31.s[0]+ mls v23.4s, v20.4s, v31.s[0]+ str q28, [x0, #0x1f0]+ ldr q11, [x0, #0x300]+ sqrdmulh v28.4s, v18.4s, v1.s[1]+ str q10, [x0, #0x370]+ mul v16.4s, v19.4s, v0.s[2]+ sqrdmulh v19.4s, v19.4s, v0.s[3]+ sqrdmulh v10.4s, v26.4s, v0.s[1]+ add v18.4s, v12.4s, v13.4s+ sub v12.4s, v12.4s, v13.4s+ mls v16.4s, v19.4s, v31.s[0]+ ldr q9, [x0, #0x280]+ ldr q20, [x0, #0x380]+ add v24.4s, v23.4s, v27.4s+ sub v25.4s, v23.4s, v27.4s+ mul v19.4s, v26.4s, v0.s[0]+ mls v14.4s, v28.4s, v31.s[0]+ add v13.4s, v9.4s, v17.4s+ sub v28.4s, v18.4s, v13.4s+ mls v19.4s, v10.4s, v31.s[0]+ sqrdmulh v23.4s, v25.4s, v0.s[1]+ mul v15.4s, v25.4s, v0.s[0]+ str q19, [x0, #0x2b0]+ ldr q19, [x0, #0x80]+ sqrdmulh v22.4s, v8.4s, v30.4s+ mls v15.4s, v23.4s, v31.s[0]+ add v26.4s, v16.4s, v14.4s+ mul v10.4s, v8.4s, v29.4s+ add v8.4s, v18.4s, v13.4s+ sqrdmulh v13.4s, v26.4s, v30.4s+ str q15, [x0, #0x2f0]+ sub v18.4s, v9.4s, v17.4s+ mul v27.4s, v26.4s, v29.4s+ sqrdmulh v23.4s, v12.4s, v5.s[3]+ mls v27.4s, v13.4s, v31.s[0]+ mul v25.4s, v18.4s, v6.s[0]+ sqrdmulh v13.4s, v18.4s, v6.s[1]+ str q27, [x0, #0x1b0]+ sub v27.4s, v19.4s, v21.4s+ add v21.4s, v19.4s, v21.4s+ mul v18.4s, v12.4s, v5.s[2]+ sub v9.4s, v16.4s, v14.4s+ mul v26.4s, v24.4s, v29.4s+ ldr q12, [x0, #0x3c0]+ mls v25.4s, v13.4s, v31.s[0]+ sub v16.4s, v20.4s, v12.4s+ sqrdmulh v17.4s, v24.4s, v30.4s+ mul v13.4s, v16.4s, v7.s[0]+ mls v18.4s, v23.4s, v31.s[0]+ mls v26.4s, v17.4s, v31.s[0]+ sqrdmulh v17.4s, v9.4s, v0.s[1]+ sqrdmulh v24.4s, v16.4s, v7.s[1]+ ldr q16, [x0, #0x340]+ str q26, [x0, #0xf0]+ sub v15.4s, v18.4s, v25.4s+ mul v23.4s, v9.4s, v0.s[0]+ ldr q19, [x0, #0x180]+ ldr q14, [x0, #0x1c0]+ sqrdmulh v9.4s, v15.4s, v2.s[3]+ add v20.4s, v20.4s, v12.4s+ add v12.4s, v11.4s, v16.4s+ mul v26.4s, v15.4s, v2.s[2]+ sub v16.4s, v11.4s, v16.4s+ sub v11.4s, v12.4s, v20.4s+ mls v23.4s, v17.4s, v31.s[0]+ add v20.4s, v12.4s, v20.4s+ sub v12.4s, v19.4s, v14.4s+ mul v17.4s, v27.4s, v4.s[0]+ mls v10.4s, v22.4s, v31.s[0]+ add v22.4s, v19.4s, v14.4s+ ldr q14, [x0, #0x40]+ str q23, [x0, #0x3b0]+ sqrdmulh v23.4s, v27.4s, v4.s[1]+ ldr q19, [x0, #0x100]+ ldr q15, [x0, #0x140]+ mls v26.4s, v9.4s, v31.s[0]+ str q10, [x0, #0xb0]+ sqrdmulh v10.4s, v11.4s, v3.s[1]+ sub v27.4s, v8.4s, v20.4s+ mls v13.4s, v24.4s, v31.s[0]+ add v8.4s, v8.4s, v20.4s+ sub v20.4s, v19.4s, v15.4s+ sqrdmulh v9.4s, v16.4s, v6.s[3]+ add v19.4s, v19.4s, v15.4s+ ldr q15, [x0]+ mul v24.4s, v20.4s, v4.s[2]+ mls v17.4s, v23.4s, v31.s[0]+ add v23.4s, v15.4s, v14.4s+ sub v14.4s, v15.4s, v14.4s+ mul v15.4s, v11.4s, v3.s[0]+ sub v11.4s, v23.4s, v21.4s+ mul v16.4s, v16.4s, v6.s[2]+ add v23.4s, v23.4s, v21.4s+ add v21.4s, v18.4s, v25.4s+ sqrdmulh v25.4s, v20.4s, v4.s[3]+ sqrdmulh v18.4s, v12.4s, v5.s[1]+ sqrdmulh v20.4s, v14.4s, v3.s[3]+ mul v12.4s, v12.4s, v5.s[0]+ mls v16.4s, v9.4s, v31.s[0]+ mls v12.4s, v18.4s, v31.s[0]+ mls v24.4s, v25.4s, v31.s[0]+ sub v25.4s, v16.4s, v13.4s+ add v16.4s, v16.4s, v13.4s+ mul v13.4s, v28.4s, v2.s[2]+ sqrdmulh v9.4s, v25.4s, v3.s[1]+ mul v18.4s, v25.4s, v3.s[0]+ sqrdmulh v25.4s, v28.4s, v2.s[3]+ mls v18.4s, v9.4s, v31.s[0]+ mls v15.4s, v10.4s, v31.s[0]+ sub v9.4s, v24.4s, v12.4s+ mul v28.4s, v14.4s, v3.s[2]+ add v12.4s, v24.4s, v12.4s+ mls v28.4s, v20.4s, v31.s[0]+ add v20.4s, v26.4s, v18.4s+ add v24.4s, v19.4s, v22.4s+ mls v13.4s, v25.4s, v31.s[0]+ sub v18.4s, v26.4s, v18.4s+ sqrdmulh v26.4s, v11.4s, v1.s[3]+ add v25.4s, v23.4s, v24.4s+ mul v14.4s, v11.4s, v1.s[2]+ sub v10.4s, v28.4s, v17.4s+ add v17.4s, v28.4s, v17.4s+ sub v28.4s, v23.4s, v24.4s+ sqrdmulh v11.4s, v9.4s, v2.s[1]+ sub v22.4s, v19.4s, v22.4s+ sub v24.4s, v17.4s, v12.4s+ mls v14.4s, v26.4s, v31.s[0]+ add v17.4s, v17.4s, v12.4s+ sqrdmulh v12.4s, v22.4s, v2.s[1]+ sub v23.4s, v13.4s, v15.4s+ add v26.4s, v13.4s, v15.4s+ mul v19.4s, v9.4s, v2.s[0]+ sqrdmulh v9.4s, v23.4s, v1.s[1]+ mul v15.4s, v23.4s, v1.s[0]+ mls v19.4s, v11.4s, v31.s[0]+ sub v13.4s, v25.4s, v8.4s+ mls v15.4s, v9.4s, v31.s[0]+ add v23.4s, v25.4s, v8.4s+ sqrdmulh v8.4s, v24.4s, v0.s[3]+ mul v25.4s, v22.4s, v2.s[0]+ subs x4, x4, #0x1+ cbnz x4, Lmld_intt_layer1234_start+ mul v22.4s, v24.4s, v0.s[2]+ sqrdmulh v9.4s, v23.4s, v30.4s+ mls v25.4s, v12.4s, v31.s[0]+ mul v23.4s, v23.4s, v29.4s+ mls v23.4s, v9.4s, v31.s[0]+ mls v22.4s, v8.4s, v31.s[0]+ sqrdmulh v11.4s, v10.4s, v1.s[3]+ sub v9.4s, v14.4s, v25.4s+ str q23, [x0], #0x10+ mul v23.4s, v10.4s, v1.s[2]+ sqrdmulh v24.4s, v9.4s, v0.s[3]+ mls v23.4s, v11.4s, v31.s[0]+ mul v9.4s, v9.4s, v0.s[2]+ mls v9.4s, v24.4s, v31.s[0]+ sub v24.4s, v23.4s, v19.4s+ add v19.4s, v23.4s, v19.4s+ mul v8.4s, v28.4s, v0.s[2]+ add v23.4s, v14.4s, v25.4s+ mul v11.4s, v13.4s, v0.s[0]+ sub v25.4s, v21.4s, v16.4s+ sub v10.4s, v9.4s, v15.4s+ add v15.4s, v9.4s, v15.4s+ add v16.4s, v21.4s, v16.4s+ sqrdmulh v28.4s, v28.4s, v0.s[3]+ add v9.4s, v23.4s, v26.4s+ add v14.4s, v17.4s, v16.4s+ sqrdmulh v21.4s, v13.4s, v0.s[1]+ sub v13.4s, v17.4s, v16.4s+ mul v16.4s, v14.4s, v29.4s+ sub v12.4s, v23.4s, v26.4s+ mul v17.4s, v13.4s, v0.s[0]+ mls v8.4s, v28.4s, v31.s[0]+ sqrdmulh v28.4s, v14.4s, v30.4s+ sub v14.4s, v19.4s, v20.4s+ add v20.4s, v19.4s, v20.4s+ sqrdmulh v19.4s, v13.4s, v0.s[1]+ sqrdmulh v26.4s, v27.4s, v1.s[1]+ mls v16.4s, v28.4s, v31.s[0]+ sqrdmulh v28.4s, v12.4s, v0.s[1]+ mul v27.4s, v27.4s, v1.s[0]+ str q16, [x0, #0x30]+ mul v12.4s, v12.4s, v0.s[0]+ mls v12.4s, v28.4s, v31.s[0]+ mls v27.4s, v26.4s, v31.s[0]+ sqrdmulh v26.4s, v25.4s, v1.s[1]+ str q12, [x0, #0x270]+ sqrdmulh v12.4s, v14.4s, v0.s[1]+ sub v13.4s, v8.4s, v27.4s+ add v27.4s, v8.4s, v27.4s+ mul v16.4s, v14.4s, v0.s[0]+ mul v8.4s, v20.4s, v29.4s+ mls v16.4s, v12.4s, v31.s[0]+ mul v12.4s, v18.4s, v1.s[0]+ sqrdmulh v14.4s, v20.4s, v30.4s+ str q16, [x0, #0x2b0]+ sqrdmulh v20.4s, v18.4s, v1.s[1]+ mul v25.4s, v25.4s, v1.s[0]+ mls v25.4s, v26.4s, v31.s[0]+ mls v17.4s, v19.4s, v31.s[0]+ mul v18.4s, v9.4s, v29.4s+ add v28.4s, v22.4s, v25.4s+ sqrdmulh v9.4s, v9.4s, v30.4s+ sub v23.4s, v22.4s, v25.4s+ str q17, [x0, #0x230]+ mls v11.4s, v21.4s, v31.s[0]+ mls v8.4s, v14.4s, v31.s[0]+ sqrdmulh v21.4s, v28.4s, v30.4s+ str q11, [x0, #0x1f0]+ mul v17.4s, v28.4s, v29.4s+ str q8, [x0, #0xb0]+ mul v28.4s, v10.4s, v0.s[0]+ mls v17.4s, v21.4s, v31.s[0]+ sqrdmulh v21.4s, v10.4s, v0.s[1]+ mul v8.4s, v15.4s, v29.4s+ str q17, [x0, #0x130]+ sqrdmulh v15.4s, v15.4s, v30.4s+ mul v10.4s, v24.4s, v0.s[2]+ sqrdmulh v24.4s, v24.4s, v0.s[3]+ mls v12.4s, v20.4s, v31.s[0]+ mls v8.4s, v15.4s, v31.s[0]+ mls v10.4s, v24.4s, v31.s[0]+ sqrdmulh v24.4s, v27.4s, v30.4s+ str q8, [x0, #0x170]+ mul v14.4s, v13.4s, v0.s[0]+ sub v16.4s, v10.4s, v12.4s+ sqrdmulh v17.4s, v13.4s, v0.s[1]+ add v10.4s, v10.4s, v12.4s+ mul v13.4s, v27.4s, v29.4s+ sqrdmulh v20.4s, v10.4s, v30.4s+ mls v14.4s, v17.4s, v31.s[0]+ mls v13.4s, v24.4s, v31.s[0]+ sqrdmulh v24.4s, v23.4s, v0.s[1]+ str q14, [x0, #0x2f0]+ sqrdmulh v8.4s, v16.4s, v0.s[1]+ str q13, [x0, #0xf0]+ mls v18.4s, v9.4s, v31.s[0]+ mul v9.4s, v23.4s, v0.s[0]+ mul v14.4s, v16.4s, v0.s[0]+ str q18, [x0, #0x70]+ mul v27.4s, v10.4s, v29.4s+ mls v28.4s, v21.4s, v31.s[0]+ mls v14.4s, v8.4s, v31.s[0]+ mls v27.4s, v20.4s, v31.s[0]+ str q28, [x0, #0x370]+ mls v9.4s, v24.4s, v31.s[0]+ str q14, [x0, #0x3b0]+ str q27, [x0, #0x1b0]+ str q9, [x0, #0x330]+ ldp d8, d9, [sp]+ .cfi_restore d8+ .cfi_restore d9+ ldp d10, d11, [sp, #0x10]+ .cfi_restore d10+ .cfi_restore d11+ ldp d12, d13, [sp, #0x20]+ .cfi_restore d12+ .cfi_restore d13+ ldp d14, d15, [sp, #0x30]+ .cfi_restore d14+ .cfi_restore d15+ add sp, sp, #0x40+ .cfi_adjust_cfa_offset -0x40+ ret+ .cfi_endproc++MLD_ASM_FN_SIZE(intt_aarch64_asm)++#endif /* MLD_ARITH_BACKEND_AARCH64 && !MLD_CONFIG_MULTILEVEL_NO_SHARED */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
@@ -0,0 +1,686 @@+/* Copyright (c) 2022 Arm Limited+ * Copyright (c) 2022 Hanno Becker+ * Copyright (c) 2023 Amin Abdulrahman, Matthias Kannwischer+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [NeonNTT]+ * Neon NTT: Faster Dilithium, Kyber, and Saber on Cortex-A72 and Apple M1+ * Becker, Hwang, Kannwischer, Yang, Yang+ * https://eprint.iacr.org/2021/986+ *+ * - [NeonNTT_Autoformalised]+ * Neon NTT - (Auto)formalised+ * Hanno Becker+ * https://eprint.iacr.org/2026/1223+ *+ * - [SLOTHY_Paper]+ * Fast and Clean: Auditable high-performance assembly via constraint solving+ * Abdulrahman, Becker, Kannwischer, Klein+ * https://eprint.iacr.org/2022/1303+ */++/* AArch64 ML-DSA forward NTT following @[NeonNTT], @[SLOTHY_Paper], and @[NeonNTT_Autoformalised] */++/*yaml+ Name: ntt_aarch64_asm+ Description: AArch64 ML-DSA forward NTT+ Signature: void mld_ntt_aarch64_asm(int32_t r[256], const int32_t zetas_l123456[144], const int32_t zetas_l78[384])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON]+ x0:+ type: buffer+ size_bytes: 1024+ permissions: read/write+ c_parameter: int32_t r[256]+ description: Input/output polynomial (256 x int32_t)+ x1:+ type: buffer+ size_bytes: 576+ permissions: read-only+ c_parameter: const int32_t zetas_l123456[144]+ description: Twiddle factors for layers 1-6 (144 x int32_t)+ x2:+ type: buffer+ size_bytes: 1536+ permissions: read-only+ c_parameter: const int32_t zetas_l78[384]+ description: Twiddle factors for layers 7-8 (384 x int32_t)+*/++#include "../../../common.h"+#if defined(MLD_ARITH_BACKEND_AARCH64) && !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/aarch64_opt/src/mldsa_ntt_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(ntt_aarch64_asm)+MLD_ASM_FN_SYMBOL(ntt_aarch64_asm)++ .cfi_startproc+ sub sp, sp, #0x40+ .cfi_adjust_cfa_offset 0x40+ stp d8, d9, [sp]+ .cfi_rel_offset d8, 0x0+ .cfi_rel_offset d9, 0x8+ stp d10, d11, [sp, #0x10]+ .cfi_rel_offset d10, 0x10+ .cfi_rel_offset d11, 0x18+ stp d12, d13, [sp, #0x20]+ .cfi_rel_offset d12, 0x20+ .cfi_rel_offset d13, 0x28+ stp d14, d15, [sp, #0x30]+ .cfi_rel_offset d14, 0x30+ .cfi_rel_offset d15, 0x38+ mov w5, #0xe001 // =57345+ movk w5, #0x7f, lsl #16+ dup v7.4s, w5+ mov x3, x0+ mov x4, #0x8 // =8+ ldr q0, [x1], #0x40+ ldur q1, [x1, #-0x30]+ ldur q2, [x1, #-0x20]+ ldur q3, [x1, #-0x10]+ ldr q23, [x0, #0x390]+ ldr q13, [x0, #0x380]+ ldr q22, [x0, #0x80]+ ldr q26, [x0, #0x190]+ ldr q8, [x0, #0x280]+ ldr q6, [x0, #0x210]+ mul v10.4s, v13.4s, v0.s[0]+ sqrdmulh v13.4s, v13.4s, v0.s[1]+ mul v12.4s, v8.4s, v0.s[0]+ sqrdmulh v27.4s, v8.4s, v0.s[1]+ mul v4.4s, v6.4s, v0.s[0]+ mls v10.4s, v13.4s, v7.s[0]+ ldr q13, [x0, #0x180]+ sqrdmulh v14.4s, v23.4s, v0.s[1]+ mls v12.4s, v27.4s, v7.s[0]+ add v31.4s, v13.4s, v10.4s+ sub v13.4s, v13.4s, v10.4s+ mul v10.4s, v23.4s, v0.s[0]+ sqrdmulh v8.4s, v13.4s, v1.s[1]+ sub v18.4s, v22.4s, v12.4s+ mls v10.4s, v14.4s, v7.s[0]+ mul v13.4s, v13.4s, v1.s[0]+ mls v13.4s, v8.4s, v7.s[0]+ sub v29.4s, v26.4s, v10.4s+ add v25.4s, v26.4s, v10.4s+ mul v10.4s, v31.4s, v0.s[2]+ mul v14.4s, v25.4s, v0.s[2]+ add v17.4s, v18.4s, v13.4s+ sub v15.4s, v18.4s, v13.4s+ sqrdmulh v13.4s, v31.4s, v0.s[3]+ sqrdmulh v20.4s, v15.4s, v3.s[1]+ sqrdmulh v5.4s, v17.4s, v2.s[3]+ mls v10.4s, v13.4s, v7.s[0]+ ldr q13, [x0, #0x300]+ mul v18.4s, v17.4s, v2.s[2]+ add v31.4s, v22.4s, v12.4s+ mul v23.4s, v15.4s, v3.s[0]+ ldr q17, [x0, #0x90]+ add v19.4s, v31.4s, v10.4s+ sub v16.4s, v31.4s, v10.4s+ mul v10.4s, v13.4s, v0.s[0]+ sqrdmulh v13.4s, v13.4s, v0.s[1]+ sqrdmulh v27.4s, v16.4s, v2.s[1]+ mul v11.4s, v16.4s, v2.s[0]+ mls v10.4s, v13.4s, v7.s[0]+ ldr q13, [x0, #0x290]+ ldr q22, [x0, #0x100]+ mls v11.4s, v27.4s, v7.s[0]+ sqrdmulh v15.4s, v13.4s, v0.s[1]+ sub v12.4s, v22.4s, v10.4s+ add v30.4s, v22.4s, v10.4s+ mul v10.4s, v13.4s, v0.s[0]+ ldr q28, [x0]+ sqrdmulh v13.4s, v25.4s, v0.s[3]+ sqrdmulh v27.4s, v30.4s, v0.s[3]+ mls v10.4s, v15.4s, v7.s[0]+ mls v14.4s, v13.4s, v7.s[0]+ ldr q13, [x0, #0x200]+ sqrdmulh v25.4s, v12.4s, v1.s[1]+ add v24.4s, v17.4s, v10.4s+ sub v21.4s, v17.4s, v10.4s+ sqrdmulh v8.4s, v13.4s, v0.s[1]+ sub v9.4s, v24.4s, v14.4s+ mul v26.4s, v12.4s, v1.s[0]+ mul v13.4s, v13.4s, v0.s[0]+ mls v13.4s, v8.4s, v7.s[0]+ mul v8.4s, v30.4s, v0.s[2]+ mls v8.4s, v27.4s, v7.s[0]+ add v16.4s, v28.4s, v13.4s+ sub v10.4s, v28.4s, v13.4s+ mls v26.4s, v25.4s, v7.s[0]+ sqrdmulh v12.4s, v19.4s, v1.s[3]+ sub v25.4s, v16.4s, v8.4s+ mls v23.4s, v20.4s, v7.s[0]+ sub v22.4s, v25.4s, v11.4s+ sqrdmulh v20.4s, v9.4s, v2.s[1]+ sub v15.4s, v10.4s, v26.4s+ sub x4, x4, #0x2++Lmld_ntt_layer123_start:+ add v31.4s, v10.4s, v26.4s+ mul v17.4s, v19.4s, v1.s[2]+ add v26.4s, v15.4s, v23.4s+ ldr q30, [x0, #0x2a0]+ sub v13.4s, v15.4s, v23.4s+ mul v23.4s, v29.4s, v1.s[0]+ add v25.4s, v25.4s, v11.4s+ str q22, [x0, #0x180]+ mul v11.4s, v9.4s, v2.s[0]+ str q13, [x0, #0x380]+ ldr q28, [x0, #0x10]+ add v10.4s, v16.4s, v8.4s+ mls v17.4s, v12.4s, v7.s[0]+ ldr q13, [x0, #0x3a0]+ str q26, [x0, #0x300]+ sqrdmulh v27.4s, v30.4s, v0.s[1]+ mls v18.4s, v5.4s, v7.s[0]+ ldr q9, [x0, #0x1a0]+ sub v16.4s, v10.4s, v17.4s+ add v15.4s, v10.4s, v17.4s+ sqrdmulh v10.4s, v6.4s, v0.s[1]+ str q16, [x0, #0x80]+ str q15, [x0], #0x10+ sqrdmulh v19.4s, v13.4s, v0.s[1]+ sub v15.4s, v31.4s, v18.4s+ mul v8.4s, v13.4s, v0.s[0]+ add v26.4s, v31.4s, v18.4s+ str q15, [x0, #0x270]+ sqrdmulh v13.4s, v29.4s, v1.s[1]+ str q26, [x0, #0x1f0]+ mls v8.4s, v19.4s, v7.s[0]+ mls v11.4s, v20.4s, v7.s[0]+ mls v23.4s, v13.4s, v7.s[0]+ add v22.4s, v9.4s, v8.4s+ ldr q6, [x0, #0x210]+ sub v29.4s, v9.4s, v8.4s+ mul v17.4s, v30.4s, v0.s[0]+ ldr q9, [x0, #0x300]+ sqrdmulh v13.4s, v22.4s, v0.s[3]+ add v18.4s, v21.4s, v23.4s+ mls v4.4s, v10.4s, v7.s[0]+ sub v31.4s, v21.4s, v23.4s+ sqrdmulh v16.4s, v31.4s, v3.s[1]+ add v19.4s, v24.4s, v14.4s+ mul v14.4s, v22.4s, v0.s[2]+ sub v10.4s, v28.4s, v4.4s+ mls v14.4s, v13.4s, v7.s[0]+ ldr q13, [x0, #0x100]+ sqrdmulh v22.4s, v9.4s, v0.s[1]+ mul v8.4s, v9.4s, v0.s[0]+ mul v23.4s, v31.4s, v3.s[0]+ mls v8.4s, v22.4s, v7.s[0]+ mls v23.4s, v16.4s, v7.s[0]+ add v16.4s, v28.4s, v4.4s+ ldr q22, [x0, #0x90]+ mul v4.4s, v6.4s, v0.s[0]+ mls v17.4s, v27.4s, v7.s[0]+ add v21.4s, v13.4s, v8.4s+ sub v27.4s, v13.4s, v8.4s+ sqrdmulh v31.4s, v21.4s, v0.s[3]+ str q25, [x0, #0xf0]+ mul v8.4s, v21.4s, v0.s[2]+ add v24.4s, v22.4s, v17.4s+ sub v21.4s, v22.4s, v17.4s+ sqrdmulh v5.4s, v18.4s, v2.s[3]+ mls v8.4s, v31.4s, v7.s[0]+ sub v9.4s, v24.4s, v14.4s+ sqrdmulh v20.4s, v27.4s, v1.s[1]+ mul v26.4s, v27.4s, v1.s[0]+ sub v25.4s, v16.4s, v8.4s+ mul v18.4s, v18.4s, v2.s[2]+ sub v22.4s, v25.4s, v11.4s+ mls v26.4s, v20.4s, v7.s[0]+ sqrdmulh v20.4s, v9.4s, v2.s[1]+ sqrdmulh v12.4s, v19.4s, v1.s[3]+ sub v15.4s, v10.4s, v26.4s+ subs x4, x4, #0x1+ cbnz x4, Lmld_ntt_layer123_start+ add v13.4s, v10.4s, v26.4s+ mls v18.4s, v5.4s, v7.s[0]+ str q22, [x0, #0x180]+ add v27.4s, v16.4s, v8.4s+ mul v22.4s, v19.4s, v1.s[2]+ add v26.4s, v24.4s, v14.4s+ ldr q31, [x0, #0x110]+ sub v14.4s, v15.4s, v23.4s+ add v17.4s, v15.4s, v23.4s+ mls v22.4s, v12.4s, v7.s[0]+ add v28.4s, v13.4s, v18.4s+ str q14, [x0, #0x380]+ sqrdmulh v24.4s, v6.4s, v0.s[1]+ add v5.4s, v25.4s, v11.4s+ sub v19.4s, v13.4s, v18.4s+ str q17, [x0, #0x300]+ str q5, [x0, #0x100]+ mul v16.4s, v9.4s, v2.s[0]+ ldr q18, [x0, #0x310]+ str q19, [x0, #0x280]+ mls v16.4s, v20.4s, v7.s[0]+ str q28, [x0, #0x200]+ add v13.4s, v27.4s, v22.4s+ ldr q15, [x0, #0x10]+ sub v10.4s, v27.4s, v22.4s+ mls v4.4s, v24.4s, v7.s[0]+ str q13, [x0], #0x10+ str q10, [x0, #0x70]+ sqrdmulh v12.4s, v29.4s, v1.s[1]+ mul v23.4s, v29.4s, v1.s[0]+ mul v8.4s, v26.4s, v1.s[2]+ add v20.4s, v15.4s, v4.4s+ sub v6.4s, v15.4s, v4.4s+ mls v23.4s, v12.4s, v7.s[0]+ sqrdmulh v22.4s, v18.4s, v0.s[1]+ mul v5.4s, v18.4s, v0.s[0]+ sub v28.4s, v21.4s, v23.4s+ sqrdmulh v10.4s, v26.4s, v1.s[3]+ mls v5.4s, v22.4s, v7.s[0]+ sqrdmulh v30.4s, v28.4s, v3.s[1]+ add v4.4s, v21.4s, v23.4s+ mls v8.4s, v10.4s, v7.s[0]+ add v12.4s, v31.4s, v5.4s+ sub v9.4s, v31.4s, v5.4s+ sqrdmulh v25.4s, v4.4s, v2.s[3]+ sqrdmulh v15.4s, v9.4s, v1.s[1]+ sqrdmulh v31.4s, v12.4s, v0.s[3]+ mul v18.4s, v12.4s, v0.s[2]+ mul v11.4s, v9.4s, v1.s[0]+ mls v18.4s, v31.4s, v7.s[0]+ mul v29.4s, v4.4s, v2.s[2]+ mls v29.4s, v25.4s, v7.s[0]+ add v23.4s, v20.4s, v18.4s+ mls v11.4s, v15.4s, v7.s[0]+ sub v31.4s, v20.4s, v18.4s+ add v17.4s, v23.4s, v8.4s+ add v5.4s, v31.4s, v16.4s+ mul v24.4s, v28.4s, v3.s[0]+ str q17, [x0], #0x10+ sub v19.4s, v31.4s, v16.4s+ mls v24.4s, v30.4s, v7.s[0]+ str q5, [x0, #0xf0]+ add v31.4s, v6.4s, v11.4s+ sub v26.4s, v23.4s, v8.4s+ str q19, [x0, #0x170]+ add v4.4s, v31.4s, v29.4s+ sub v13.4s, v6.4s, v11.4s+ str q26, [x0, #0x70]+ sub v11.4s, v31.4s, v29.4s+ sub v22.4s, v13.4s, v24.4s+ add v23.4s, v13.4s, v24.4s+ str q4, [x0, #0x1f0]+ str q11, [x0, #0x270]+ str q23, [x0, #0x2f0]+ str q22, [x0, #0x370]+ mov x0, x3+ mov x4, #0x8 // =8+ ldr q9, [x0, #0x40]+ ldr q23, [x1], #0x40+ ldr q21, [x2, #0x60]+ ldr q1, [x0, #0x20]+ ldur q14, [x1, #-0x30]+ ldr q13, [x0]+ ldr q11, [x2, #0x50]+ sqrdmulh v16.4s, v9.4s, v23.s[1]+ ldr q17, [x0, #0x50]+ mul v15.4s, v9.4s, v23.s[0]+ ldr q30, [x0, #0x70]+ ldr q27, [x0, #0x60]+ ldr q8, [x2, #0x30]+ sqrdmulh v12.4s, v17.4s, v23.s[1]+ ldr q6, [x0, #0x30]+ mls v15.4s, v16.4s, v7.s[0]+ sqrdmulh v18.4s, v27.4s, v23.s[1]+ sqrdmulh v19.4s, v30.4s, v23.s[1]+ add v5.4s, v13.4s, v15.4s+ mul v25.4s, v27.4s, v23.s[0]+ sub v26.4s, v13.4s, v15.4s+ mls v25.4s, v18.4s, v7.s[0]+ mul v10.4s, v17.4s, v23.s[0]+ mls v10.4s, v12.4s, v7.s[0]+ mul v4.4s, v30.4s, v23.s[0]+ sub v22.4s, v1.4s, v25.4s+ mls v4.4s, v19.4s, v7.s[0]+ add v28.4s, v1.4s, v25.4s+ sqrdmulh v19.4s, v28.4s, v23.s[3]+ sqrdmulh v9.4s, v22.4s, v14.s[1]+ add v2.4s, v6.4s, v4.4s+ mul v0.4s, v28.4s, v23.s[2]+ sqrdmulh v27.4s, v2.4s, v23.s[3]+ sub v17.4s, v6.4s, v4.4s+ mul v3.4s, v2.4s, v23.s[2]+ sqrdmulh v20.4s, v17.4s, v14.s[1]+ ldr q1, [x0, #0x10]+ mls v3.4s, v27.4s, v7.s[0]+ mls v0.4s, v19.4s, v7.s[0]+ ldur q16, [x1, #-0x20]+ add v31.4s, v1.4s, v10.4s+ mul v30.4s, v17.4s, v14.s[0]+ mls v30.4s, v20.4s, v7.s[0]+ add v27.4s, v31.4s, v3.4s+ sub v23.4s, v1.4s, v10.4s+ sub v24.4s, v31.4s, v3.4s+ sqrdmulh v4.4s, v27.4s, v14.s[3]+ sqrdmulh v10.4s, v24.4s, v16.s[1]+ mul v18.4s, v24.4s, v16.s[0]+ add v15.4s, v23.4s, v30.4s+ sub v23.4s, v23.4s, v30.4s+ mul v29.4s, v27.4s, v14.s[2]+ sub v2.4s, v5.4s, v0.4s+ add v12.4s, v5.4s, v0.4s+ mls v18.4s, v10.4s, v7.s[0]+ ldur q3, [x1, #-0x10]+ mls v29.4s, v4.4s, v7.s[0]+ mul v4.4s, v22.4s, v14.s[0]+ add v1.4s, v2.4s, v18.4s+ sub v24.4s, v2.4s, v18.4s+ mls v4.4s, v9.4s, v7.s[0]+ ldr q20, [x2, #0x10]+ add v25.4s, v12.4s, v29.4s+ mul v9.4s, v23.4s, v3.s[0]+ sub v5.4s, v12.4s, v29.4s+ sqrdmulh v31.4s, v23.4s, v3.s[1]+ trn2 v6.4s, v1.4s, v24.4s+ trn2 v10.4s, v25.4s, v5.4s+ sqrdmulh v13.4s, v15.4s, v16.s[3]+ trn2 v30.2d, v10.2d, v6.2d+ ldr q3, [x2], #0xc0+ mul v12.4s, v15.4s, v16.s[2]+ trn1 v27.2d, v10.2d, v6.2d+ mls v9.4s, v31.4s, v7.s[0]+ trn1 v22.4s, v25.4s, v5.4s+ sub v6.4s, v26.4s, v4.4s+ mls v12.4s, v13.4s, v7.s[0]+ trn1 v1.4s, v1.4s, v24.4s+ add v13.4s, v26.4s, v4.4s+ trn2 v10.2d, v22.2d, v1.2d+ mul v28.4s, v30.4s, v3.4s+ sub v31.4s, v6.4s, v9.4s+ sub x4, x4, #0x1++Lmld_ntt_layer45678_start:+ add v2.4s, v13.4s, v12.4s+ sqrdmulh v5.4s, v30.4s, v20.4s+ sub v25.4s, v13.4s, v12.4s+ add v17.4s, v6.4s, v9.4s+ mul v19.4s, v10.4s, v3.4s+ trn2 v4.4s, v2.4s, v25.4s+ ldur q24, [x2, #-0x50]+ trn2 v29.4s, v17.4s, v31.4s+ sqrdmulh v15.4s, v10.4s, v20.4s+ mls v28.4s, v5.4s, v7.s[0]+ trn2 v3.2d, v4.2d, v29.2d+ sqrdmulh v12.4s, v3.4s, v24.4s+ mul v16.4s, v3.4s, v21.4s+ mls v19.4s, v15.4s, v7.s[0]+ ldur q10, [x2, #-0xa0]+ add v13.4s, v27.4s, v28.4s+ mls v16.4s, v12.4s, v7.s[0]+ sqrdmulh v9.4s, v13.4s, v8.4s+ sub v30.4s, v27.4s, v28.4s+ ldr q18, [x1], #0x40+ mul v8.4s, v13.4s, v10.4s+ ldr q10, [x0, #0xd0]+ sqrdmulh v14.4s, v30.4s, v11.4s+ ldr q23, [x0, #0xe0]+ sqrdmulh v13.4s, v10.4s, v18.s[1]+ sqrdmulh v12.4s, v23.4s, v18.s[1]+ ldur q6, [x2, #-0x80]+ mul v3.4s, v10.4s, v18.s[0]+ mls v3.4s, v13.4s, v7.s[0]+ ldr q13, [x0, #0xf0]+ trn1 v27.4s, v2.4s, v25.4s+ mul v2.4s, v30.4s, v6.4s+ trn1 v20.4s, v17.4s, v31.4s+ trn1 v25.2d, v4.2d, v29.2d+ sqrdmulh v10.4s, v13.4s, v18.s[1]+ trn2 v5.2d, v27.2d, v20.2d+ ldur q6, [x2, #-0x10]+ mls v8.4s, v9.4s, v7.s[0]+ sub v15.4s, v25.4s, v16.4s+ sqrdmulh v31.4s, v5.4s, v24.4s+ sqrdmulh v30.4s, v15.4s, v6.4s+ ldur q9, [x2, #-0x30]+ mul v4.4s, v5.4s, v21.4s+ ldur q21, [x2, #-0x40]+ ldur q6, [x2, #-0x20]+ add v5.4s, v25.4s, v16.4s+ mls v4.4s, v31.4s, v7.s[0]+ mul v0.4s, v5.4s, v21.4s+ mul v17.4s, v13.4s, v18.s[0]+ mls v17.4s, v10.4s, v7.s[0]+ ldr q28, [x0, #0xb0]+ sqrdmulh v26.4s, v5.4s, v9.4s+ mul v9.4s, v15.4s, v6.4s+ trn1 v6.2d, v22.2d, v1.2d+ mls v9.4s, v30.4s, v7.s[0]+ add v25.4s, v28.4s, v17.4s+ mls v2.4s, v14.4s, v7.s[0]+ trn1 v5.2d, v27.2d, v20.2d+ ldr q20, [x2, #0x10]+ mul v29.4s, v25.4s, v18.s[2]+ add v15.4s, v6.4s, v19.4s+ ldr q30, [x0, #0xc0]+ sub v19.4s, v6.4s, v19.4s+ add v31.4s, v15.4s, v8.4s+ mls v0.4s, v26.4s, v7.s[0]+ ldur q14, [x1, #-0x30]+ add v21.4s, v19.4s, v2.4s+ sub v24.4s, v19.4s, v2.4s+ sqrdmulh v27.4s, v25.4s, v18.s[3]+ sub v26.4s, v15.4s, v8.4s+ ldr q2, [x0, #0x90]+ mul v16.4s, v30.4s, v18.s[0]+ sub v25.4s, v28.4s, v17.4s+ trn1 v11.4s, v31.4s, v26.4s+ ldr q1, [x0, #0xa0]+ trn1 v6.4s, v21.4s, v24.4s+ sqrdmulh v13.4s, v25.4s, v14.s[1]+ add v8.4s, v2.4s, v3.4s+ trn2 v28.2d, v11.2d, v6.2d+ sqrdmulh v19.4s, v30.4s, v18.s[1]+ sub v10.4s, v5.4s, v4.4s+ ldur q22, [x1, #-0x20]+ str q28, [x0, #0x20]+ mls v29.4s, v27.4s, v7.s[0]+ add v15.4s, v10.4s, v9.4s+ mul v25.4s, v25.4s, v14.s[0]+ ldur q27, [x1, #-0x10]+ trn2 v17.4s, v31.4s, v26.4s+ trn2 v21.4s, v21.4s, v24.4s+ mls v16.4s, v19.4s, v7.s[0]+ sub v24.4s, v8.4s, v29.4s+ sub v10.4s, v10.4s, v9.4s+ mls v25.4s, v13.4s, v7.s[0]+ trn1 v13.2d, v11.2d, v6.2d+ ldr q28, [x0, #0x80]+ sqrdmulh v30.4s, v24.4s, v22.s[1]+ trn2 v19.2d, v17.2d, v21.2d+ trn1 v6.2d, v17.2d, v21.2d+ mul v31.4s, v23.4s, v18.s[0]+ str q13, [x0], #0x80+ stur q6, [x0, #-0x70]+ stur q19, [x0, #-0x50]+ ldr q11, [x2, #0x50]+ mls v31.4s, v12.4s, v7.s[0]+ ldr q21, [x2, #0x60]+ trn1 v9.4s, v15.4s, v10.4s+ trn2 v6.4s, v15.4s, v10.4s+ mul v24.4s, v24.4s, v22.s[0]+ sub v10.4s, v2.4s, v3.4s+ ldr q3, [x2], #0xc0+ mls v24.4s, v30.4s, v7.s[0]+ add v26.4s, v8.4s, v29.4s+ ldur q8, [x2, #-0x90]+ add v17.4s, v5.4s, v4.4s+ sqrdmulh v2.4s, v26.4s, v14.s[3]+ sub v13.4s, v1.4s, v31.4s+ add v30.4s, v1.4s, v31.4s+ add v15.4s, v10.4s, v25.4s+ sqrdmulh v19.4s, v13.4s, v14.s[1]+ sub v25.4s, v10.4s, v25.4s+ mul v29.4s, v13.4s, v14.s[0]+ sub v5.4s, v28.4s, v16.4s+ sqrdmulh v4.4s, v30.4s, v18.s[3]+ sub v23.4s, v17.4s, v0.4s+ add v31.4s, v17.4s, v0.4s+ mul v18.4s, v30.4s, v18.s[2]+ add v1.4s, v28.4s, v16.4s+ trn2 v12.4s, v31.4s, v23.4s+ mls v29.4s, v19.4s, v7.s[0]+ trn1 v13.4s, v31.4s, v23.4s+ trn2 v30.2d, v12.2d, v6.2d+ mls v18.4s, v4.4s, v7.s[0]+ trn2 v10.2d, v13.2d, v9.2d+ trn1 v31.2d, v13.2d, v9.2d+ mul v19.4s, v26.4s, v14.s[2]+ trn1 v12.2d, v12.2d, v6.2d+ sub v6.4s, v5.4s, v29.4s+ mls v19.4s, v2.4s, v7.s[0]+ add v13.4s, v5.4s, v29.4s+ stur q10, [x0, #-0x20]+ sub v10.4s, v1.4s, v18.4s+ add v28.4s, v1.4s, v18.4s+ sqrdmulh v5.4s, v25.4s, v27.s[1]+ stur q31, [x0, #-0x40]+ add v26.4s, v10.4s, v24.4s+ sub v31.4s, v10.4s, v24.4s+ mul v9.4s, v25.4s, v27.s[0]+ stur q12, [x0, #-0x30]+ sub v24.4s, v28.4s, v19.4s+ sqrdmulh v10.4s, v15.4s, v22.s[3]+ trn1 v1.4s, v26.4s, v31.4s+ stur q30, [x0, #-0x10]+ add v30.4s, v28.4s, v19.4s+ mls v9.4s, v5.4s, v7.s[0]+ trn2 v25.4s, v26.4s, v31.4s+ trn2 v14.4s, v30.4s, v24.4s+ mul v12.4s, v15.4s, v22.s[2]+ trn1 v22.4s, v30.4s, v24.4s+ trn1 v27.2d, v14.2d, v25.2d+ mls v12.4s, v10.4s, v7.s[0]+ trn2 v30.2d, v14.2d, v25.2d+ sub v31.4s, v6.4s, v9.4s+ trn2 v10.2d, v22.2d, v1.2d+ mul v28.4s, v30.4s, v3.4s+ subs x4, x4, #0x1+ cbnz x4, Lmld_ntt_layer45678_start+ add v9.4s, v6.4s, v9.4s+ sqrdmulh v6.4s, v30.4s, v20.4s+ ldur q24, [x2, #-0xa0]+ add v25.4s, v13.4s, v12.4s+ sub v15.4s, v13.4s, v12.4s+ mul v19.4s, v10.4s, v3.4s+ trn2 v5.4s, v9.4s, v31.4s+ sqrdmulh v3.4s, v10.4s, v20.4s+ trn2 v10.4s, v25.4s, v15.4s+ mls v28.4s, v6.4s, v7.s[0]+ trn2 v13.2d, v10.2d, v5.2d+ ldur q30, [x2, #-0x50]+ mul v12.4s, v13.4s, v21.4s+ mls v19.4s, v3.4s, v7.s[0]+ add v20.4s, v27.4s, v28.4s+ sqrdmulh v13.4s, v13.4s, v30.4s+ sub v3.4s, v27.4s, v28.4s+ mul v24.4s, v20.4s, v24.4s+ sqrdmulh v6.4s, v3.4s, v11.4s+ ldur q27, [x2, #-0x80]+ mls v12.4s, v13.4s, v7.s[0]+ trn1 v25.4s, v25.4s, v15.4s+ mul v27.4s, v3.4s, v27.4s+ trn1 v31.4s, v9.4s, v31.4s+ trn1 v3.2d, v10.2d, v5.2d+ ldur q13, [x2, #-0x30]+ ldur q15, [x2, #-0x40]+ sqrdmulh v9.4s, v20.4s, v8.4s+ trn2 v20.2d, v25.2d, v31.2d+ ldur q10, [x2, #-0x10]+ mls v27.4s, v6.4s, v7.s[0]+ add v5.4s, v3.4s, v12.4s+ sub v6.4s, v3.4s, v12.4s+ sqrdmulh v3.4s, v20.4s, v30.4s+ trn1 v12.2d, v22.2d, v1.2d+ sqrdmulh v10.4s, v6.4s, v10.4s+ mls v24.4s, v9.4s, v7.s[0]+ sub v9.4s, v12.4s, v19.4s+ trn1 v25.2d, v25.2d, v31.2d+ sqrdmulh v31.4s, v5.4s, v13.4s+ add v30.4s, v9.4s, v27.4s+ add v13.4s, v12.4s, v19.4s+ mul v1.4s, v20.4s, v21.4s+ ldur q12, [x2, #-0x20]+ add v21.4s, v13.4s, v24.4s+ sub v13.4s, v13.4s, v24.4s+ mls v1.4s, v3.4s, v7.s[0]+ sub v3.4s, v9.4s, v27.4s+ mul v9.4s, v6.4s, v12.4s+ trn2 v12.4s, v21.4s, v13.4s+ trn1 v6.4s, v30.4s, v3.4s+ trn2 v30.4s, v30.4s, v3.4s+ mls v9.4s, v10.4s, v7.s[0]+ trn1 v13.4s, v21.4s, v13.4s+ mul v15.4s, v5.4s, v15.4s+ sub v3.4s, v25.4s, v1.4s+ add v5.4s, v25.4s, v1.4s+ mls v15.4s, v31.4s, v7.s[0]+ trn1 v21.2d, v13.2d, v6.2d+ trn2 v6.2d, v13.2d, v6.2d+ add v10.4s, v3.4s, v9.4s+ sub v13.4s, v3.4s, v9.4s+ str q21, [x0], #0x80+ trn1 v3.2d, v12.2d, v30.2d+ trn2 v31.2d, v12.2d, v30.2d+ trn1 v21.4s, v10.4s, v13.4s+ sub v30.4s, v5.4s, v15.4s+ add v12.4s, v5.4s, v15.4s+ stur q3, [x0, #-0x70]+ trn2 v13.4s, v10.4s, v13.4s+ trn1 v19.4s, v12.4s, v30.4s+ trn2 v12.4s, v12.4s, v30.4s+ stur q6, [x0, #-0x60]+ stur q31, [x0, #-0x50]+ trn1 v10.2d, v19.2d, v21.2d+ trn2 v3.2d, v19.2d, v21.2d+ trn1 v21.2d, v12.2d, v13.2d+ trn2 v13.2d, v12.2d, v13.2d+ stur q10, [x0, #-0x40]+ stur q3, [x0, #-0x20]+ stur q13, [x0, #-0x10]+ stur q21, [x0, #-0x30]+ ldp d8, d9, [sp]+ .cfi_restore d8+ .cfi_restore d9+ ldp d10, d11, [sp, #0x10]+ .cfi_restore d10+ .cfi_restore d11+ ldp d12, d13, [sp, #0x20]+ .cfi_restore d12+ .cfi_restore d13+ ldp d14, d15, [sp, #0x30]+ .cfi_restore d14+ .cfi_restore d15+ add sp, sp, #0x40+ .cfi_adjust_cfa_offset -0x40+ ret+ .cfi_endproc++MLD_ASM_FN_SIZE(ntt_aarch64_asm)++#endif /* MLD_ARITH_BACKEND_AARCH64 && !MLD_CONFIG_MULTILEVEL_NO_SHARED */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
@@ -0,0 +1,106 @@+/* Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*yaml+ Name: poly_pointwise_montgomery_aarch64_asm+ Description: AArch64 pointwise Montgomery multiplication of two polynomials+ Signature: void mld_poly_pointwise_montgomery_aarch64_asm(int32_t a[256], const int32_t b[256])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON]+ x0:+ type: buffer+ size_bytes: 1024+ permissions: read/write+ c_parameter: int32_t a[256]+ description: Input/output polynomial (256 x int32_t)+ x1:+ type: buffer+ size_bytes: 1024+ permissions: read-only+ c_parameter: const int32_t b[256]+ description: Input polynomial (256 x int32_t)+*/++#include "../../../common.h"+#if defined(MLD_ARITH_BACKEND_AARCH64) && \+ (!defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API) || \+ defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/aarch64_opt/src/mldsa_pointwise_montgomery_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(poly_pointwise_montgomery_aarch64_asm)+MLD_ASM_FN_SYMBOL(poly_pointwise_montgomery_aarch64_asm)++ .cfi_startproc+ mov w3, #0xe001 // =57345+ movk w3, #0x7f, lsl #16+ dup v0.4s, w3+ mov w3, #0x2001 // =8193+ movk w3, #0x380, lsl #16+ dup v1.4s, w3+ mov x3, #0x40 // =64++Lmld_poly_pointwise_montgomery_loop_start:+ ldr q16, [x0]+ ldr q17, [x0, #0x10]+ ldr q18, [x0, #0x20]+ ldr q19, [x0, #0x30]+ ldr q21, [x1, #0x10]+ ldr q22, [x1, #0x20]+ ldr q23, [x1, #0x30]+ ldr q20, [x1], #0x40+ smull v24.2d, v16.2s, v20.2s+ smull2 v25.2d, v16.4s, v20.4s+ smull v26.2d, v17.2s, v21.2s+ smull2 v27.2d, v17.4s, v21.4s+ smull v28.2d, v18.2s, v22.2s+ smull2 v29.2d, v18.4s, v22.4s+ smull v30.2d, v19.2s, v23.2s+ smull2 v31.2d, v19.4s, v23.4s+ uzp1 v16.4s, v24.4s, v25.4s+ mul v16.4s, v16.4s, v1.4s+ smlsl v24.2d, v16.2s, v0.2s+ smlsl2 v25.2d, v16.4s, v0.4s+ uzp2 v16.4s, v24.4s, v25.4s+ uzp1 v17.4s, v26.4s, v27.4s+ mul v17.4s, v17.4s, v1.4s+ smlsl v26.2d, v17.2s, v0.2s+ smlsl2 v27.2d, v17.4s, v0.4s+ uzp2 v17.4s, v26.4s, v27.4s+ uzp1 v18.4s, v28.4s, v29.4s+ mul v18.4s, v18.4s, v1.4s+ smlsl v28.2d, v18.2s, v0.2s+ smlsl2 v29.2d, v18.4s, v0.4s+ uzp2 v18.4s, v28.4s, v29.4s+ uzp1 v19.4s, v30.4s, v31.4s+ mul v19.4s, v19.4s, v1.4s+ smlsl v30.2d, v19.2s, v0.2s+ smlsl2 v31.2d, v19.4s, v0.4s+ uzp2 v19.4s, v30.4s, v31.4s+ str q17, [x0, #0x10]+ str q18, [x0, #0x20]+ str q19, [x0, #0x30]+ str q16, [x0], #0x40+ subs x3, x3, #0x4+ cbnz x3, Lmld_poly_pointwise_montgomery_loop_start+ ret+ .cfi_endproc++MLD_ASM_FN_SIZE(poly_pointwise_montgomery_aarch64_asm)++#endif /* MLD_ARITH_BACKEND_AARCH64 && (!MLD_CONFIG_NO_SIGN_API || \+ !MLD_CONFIG_NO_VERIFY_API || MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST) \+ && !MLD_CONFIG_MULTILEVEL_NO_SHARED */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
@@ -0,0 +1,69 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+/*yaml+ Name: poly_caddq_aarch64_asm+ Description: AArch64 conditional addition of q to each coefficient+ Signature: void mld_poly_caddq_aarch64_asm(int32_t a[256])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON]+ x0:+ type: buffer+ size_bytes: 1024+ permissions: read/write+ c_parameter: int32_t a[256]+ description: Input/output polynomial (256 x int32_t)+*/++#include "../../../common.h"++#if defined(MLD_ARITH_BACKEND_AARCH64) && !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/aarch64_opt/src/mldsa_poly_caddq_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(poly_caddq_aarch64_asm)+MLD_ASM_FN_SYMBOL(poly_caddq_aarch64_asm)++ .cfi_startproc+ mov w9, #0xe001 // =57345+ movk w9, #0x7f, lsl #16+ dup v4.4s, w9+ mov x1, #0x10 // =16++Lmld_poly_caddq_loop:+ ldr q0, [x0]+ ldr q1, [x0, #0x10]+ ldr q2, [x0, #0x20]+ ldr q3, [x0, #0x30]+ ushr v5.4s, v0.4s, #0x1f+ mla v0.4s, v5.4s, v4.4s+ ushr v5.4s, v1.4s, #0x1f+ mla v1.4s, v5.4s, v4.4s+ ushr v5.4s, v2.4s, #0x1f+ mla v2.4s, v5.4s, v4.4s+ ushr v5.4s, v3.4s, #0x1f+ mla v3.4s, v5.4s, v4.4s+ str q1, [x0, #0x10]+ str q2, [x0, #0x20]+ str q3, [x0, #0x30]+ str q0, [x0], #0x40+ subs x1, x1, #0x1+ b.ne Lmld_poly_caddq_loop+ ret+ .cfi_endproc++MLD_ASM_FN_SIZE(poly_caddq_aarch64_asm)++#endif /* MLD_ARITH_BACKEND_AARCH64 && !MLD_CONFIG_MULTILEVEL_NO_SHARED */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
@@ -0,0 +1,76 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+/*yaml+ Name: poly_chknorm_aarch64_asm+ Description: AArch64 infinity-norm bound check on polynomial coefficients+ Signature: int mld_poly_chknorm_aarch64_asm(const int32_t a[256], int32_t B)+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON]+ x0:+ type: buffer+ size_bytes: 1024+ permissions: read-only+ c_parameter: const int32_t a[256]+ description: Input polynomial (256 x int32_t)+ x1:+ type: scalar+ c_parameter: int32_t B+ description: Norm bound+ test_with: 131072 # representative non-negative bound (1 << 17)+*/++#include "../../../common.h"++#if defined(MLD_ARITH_BACKEND_AARCH64) && !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/aarch64_opt/src/mldsa_poly_chknorm_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(poly_chknorm_aarch64_asm)+MLD_ASM_FN_SYMBOL(poly_chknorm_aarch64_asm)++ .cfi_startproc+ dup v20.4s, w1+ eor v21.16b, v21.16b, v21.16b+ mov x2, #0x10 // =16++Lmld_poly_chknorm_loop:+ ldr q1, [x0, #0x10]+ ldr q2, [x0, #0x20]+ ldr q3, [x0, #0x30]+ ldr q0, [x0], #0x40+ abs v1.4s, v1.4s+ cmge v1.4s, v1.4s, v20.4s+ orr v21.16b, v21.16b, v1.16b+ abs v2.4s, v2.4s+ cmge v2.4s, v2.4s, v20.4s+ orr v21.16b, v21.16b, v2.16b+ abs v3.4s, v3.4s+ cmge v3.4s, v3.4s, v20.4s+ orr v21.16b, v21.16b, v3.16b+ abs v0.4s, v0.4s+ cmge v0.4s, v0.4s, v20.4s+ orr v21.16b, v21.16b, v0.16b+ subs x2, x2, #0x1+ b.ne Lmld_poly_chknorm_loop+ umaxv s21, v21.4s+ fmov w0, s21+ and w0, w0, #0x1+ ret+ .cfi_endproc++MLD_ASM_FN_SIZE(poly_chknorm_aarch64_asm)++#endif /* MLD_ARITH_BACKEND_AARCH64 && !MLD_CONFIG_MULTILEVEL_NO_SHARED */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
@@ -0,0 +1,108 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+/*yaml+ Name: poly_decompose_32_aarch64_asm+ Description: AArch64 coefficient decomposition (alpha = (Q-1)/32)+ Signature: void mld_poly_decompose_32_aarch64_asm(int32_t a1[256], int32_t a0[256])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON]+ x0:+ type: buffer+ size_bytes: 1024+ permissions: write-only+ c_parameter: int32_t a1[256]+ description: Output high-part polynomial (256 x int32_t)+ x1:+ type: buffer+ size_bytes: 1024+ permissions: read/write+ c_parameter: int32_t a0[256]+ description: Input polynomial / output low-part (256 x int32_t)+*/++#include "../../../common.h"++#if defined(MLD_ARITH_BACKEND_AARCH64) && !defined(MLD_CONFIG_NO_SIGN_API) && !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \+ (defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || (MLD_CONFIG_PARAMETER_SET == 65 || MLD_CONFIG_PARAMETER_SET == 87))++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/aarch64_opt/src/mldsa_poly_decompose_32_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(poly_decompose_32_aarch64_asm)+MLD_ASM_FN_SYMBOL(poly_decompose_32_aarch64_asm)++ .cfi_startproc+ mov w4, #0xe001 // =57345+ movk w4, #0x7f, lsl #16+ dup v20.4s, w4+ mov w5, #0xe100 // =57600+ movk w5, #0x7b, lsl #16+ dup v21.4s, w5+ mov w7, #0xfe00 // =65024+ movk w7, #0x7, lsl #16+ dup v22.4s, w7+ mov w11, #0x401 // =1025+ movk w11, #0x4010, lsl #16+ dup v23.4s, w11+ mov x3, #0x10 // =16++Lmld_poly_decompose_32_loop:+ ldr q0, [x1]+ ldr q1, [x1, #0x10]+ ldr q2, [x1, #0x20]+ ldr q3, [x1, #0x30]+ sqdmulh v5.4s, v1.4s, v23.4s+ srshr v5.4s, v5.4s, #0x12+ cmgt v24.4s, v1.4s, v21.4s+ mls v1.4s, v5.4s, v22.4s+ bic v5.16b, v5.16b, v24.16b+ add v1.4s, v1.4s, v24.4s+ sqdmulh v6.4s, v2.4s, v23.4s+ srshr v6.4s, v6.4s, #0x12+ cmgt v24.4s, v2.4s, v21.4s+ mls v2.4s, v6.4s, v22.4s+ bic v6.16b, v6.16b, v24.16b+ add v2.4s, v2.4s, v24.4s+ sqdmulh v7.4s, v3.4s, v23.4s+ srshr v7.4s, v7.4s, #0x12+ cmgt v24.4s, v3.4s, v21.4s+ mls v3.4s, v7.4s, v22.4s+ bic v7.16b, v7.16b, v24.16b+ add v3.4s, v3.4s, v24.4s+ sqdmulh v4.4s, v0.4s, v23.4s+ srshr v4.4s, v4.4s, #0x12+ cmgt v24.4s, v0.4s, v21.4s+ mls v0.4s, v4.4s, v22.4s+ bic v4.16b, v4.16b, v24.16b+ add v0.4s, v0.4s, v24.4s+ str q5, [x0, #0x10]+ str q6, [x0, #0x20]+ str q7, [x0, #0x30]+ str q4, [x0], #0x40+ str q1, [x1, #0x10]+ str q2, [x1, #0x20]+ str q3, [x1, #0x30]+ str q0, [x1], #0x40+ subs x3, x3, #0x1+ b.ne Lmld_poly_decompose_32_loop+ ret+ .cfi_endproc++MLD_ASM_FN_SIZE(poly_decompose_32_aarch64_asm)++#endif /* MLD_ARITH_BACKEND_AARCH64 && !MLD_CONFIG_NO_SIGN_API && \+ !MLD_CONFIG_MULTILEVEL_NO_SHARED && \+ (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 65 \+ || MLD_CONFIG_PARAMETER_SET == 87) */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
@@ -0,0 +1,108 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+/*yaml+ Name: poly_decompose_88_aarch64_asm+ Description: AArch64 coefficient decomposition (alpha = (Q-1)/88)+ Signature: void mld_poly_decompose_88_aarch64_asm(int32_t a1[256], int32_t a0[256])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON]+ x0:+ type: buffer+ size_bytes: 1024+ permissions: write-only+ c_parameter: int32_t a1[256]+ description: Output high-part polynomial (256 x int32_t)+ x1:+ type: buffer+ size_bytes: 1024+ permissions: read/write+ c_parameter: int32_t a0[256]+ description: Input polynomial / output low-part (256 x int32_t)+*/++#include "../../../common.h"++#if defined(MLD_ARITH_BACKEND_AARCH64) && !defined(MLD_CONFIG_NO_SIGN_API) && !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \+ (defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLD_CONFIG_PARAMETER_SET == 44)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/aarch64_opt/src/mldsa_poly_decompose_88_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(poly_decompose_88_aarch64_asm)+MLD_ASM_FN_SYMBOL(poly_decompose_88_aarch64_asm)++ .cfi_startproc+ mov w4, #0xe001 // =57345+ movk w4, #0x7f, lsl #16+ dup v20.4s, w4+ mov w5, #0x6c00 // =27648+ movk w5, #0x7e, lsl #16+ dup v21.4s, w5+ mov w7, #0xe800 // =59392+ movk w7, #0x2, lsl #16+ dup v22.4s, w7+ mov w11, #0x581 // =1409+ movk w11, #0x5816, lsl #16+ dup v23.4s, w11+ mov x3, #0x10 // =16++Lmld_poly_decompose_88_loop:+ ldr q0, [x1]+ ldr q1, [x1, #0x10]+ ldr q2, [x1, #0x20]+ ldr q3, [x1, #0x30]+ sqdmulh v5.4s, v1.4s, v23.4s+ srshr v5.4s, v5.4s, #0x11+ cmgt v24.4s, v1.4s, v21.4s+ mls v1.4s, v5.4s, v22.4s+ bic v5.16b, v5.16b, v24.16b+ add v1.4s, v1.4s, v24.4s+ sqdmulh v6.4s, v2.4s, v23.4s+ srshr v6.4s, v6.4s, #0x11+ cmgt v24.4s, v2.4s, v21.4s+ mls v2.4s, v6.4s, v22.4s+ bic v6.16b, v6.16b, v24.16b+ add v2.4s, v2.4s, v24.4s+ sqdmulh v7.4s, v3.4s, v23.4s+ srshr v7.4s, v7.4s, #0x11+ cmgt v24.4s, v3.4s, v21.4s+ mls v3.4s, v7.4s, v22.4s+ bic v7.16b, v7.16b, v24.16b+ add v3.4s, v3.4s, v24.4s+ sqdmulh v4.4s, v0.4s, v23.4s+ srshr v4.4s, v4.4s, #0x11+ cmgt v24.4s, v0.4s, v21.4s+ mls v0.4s, v4.4s, v22.4s+ bic v4.16b, v4.16b, v24.16b+ add v0.4s, v0.4s, v24.4s+ str q5, [x0, #0x10]+ str q6, [x0, #0x20]+ str q7, [x0, #0x30]+ str q4, [x0], #0x40+ str q1, [x1, #0x10]+ str q2, [x1, #0x20]+ str q3, [x1, #0x30]+ str q0, [x1], #0x40+ subs x3, x3, #0x1+ b.ne Lmld_poly_decompose_88_loop+ ret+ .cfi_endproc++MLD_ASM_FN_SIZE(poly_decompose_88_aarch64_asm)++#endif /* MLD_ARITH_BACKEND_AARCH64 && !MLD_CONFIG_NO_SIGN_API && \+ !MLD_CONFIG_MULTILEVEL_NO_SHARED && \+ (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 44) \+ */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
@@ -0,0 +1,125 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+/*yaml+ Name: poly_use_hint_32_aarch64_asm+ Description: AArch64 hint application (alpha = (Q-1)/32)+ Signature: void mld_poly_use_hint_32_aarch64_asm(int32_t a[256], const int32_t h[256])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON]+ x0:+ type: buffer+ size_bytes: 1024+ permissions: read/write+ c_parameter: int32_t a[256]+ description: Input/output polynomial (256 x int32_t)+ x1:+ type: buffer+ size_bytes: 1024+ permissions: read-only+ c_parameter: const int32_t h[256]+ description: Hint polynomial (256 x int32_t)+*/++#include "../../../common.h"++#if defined(MLD_ARITH_BACKEND_AARCH64) && !defined(MLD_CONFIG_NO_VERIFY_API) && !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \+ (defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || (MLD_CONFIG_PARAMETER_SET == 65 || MLD_CONFIG_PARAMETER_SET == 87))++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/aarch64_opt/src/mldsa_poly_use_hint_32_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(poly_use_hint_32_aarch64_asm)+MLD_ASM_FN_SYMBOL(poly_use_hint_32_aarch64_asm)++ .cfi_startproc+ mov w4, #0xe001 // =57345+ movk w4, #0x7f, lsl #16+ dup v20.4s, w4+ mov w5, #0xe100 // =57600+ movk w5, #0x7b, lsl #16+ dup v21.4s, w5+ mov w7, #0xfe00 // =65024+ movk w7, #0x7, lsl #16+ dup v22.4s, w7+ mov w11, #0x401 // =1025+ movk w11, #0x4010, lsl #16+ dup v23.4s, w11+ movi v24.4s, #0xf+ mov x3, #0x10 // =16++Lmld_poly_use_hint_32_loop:+ ldr q1, [x0, #0x10]+ ldr q2, [x0, #0x20]+ ldr q3, [x0, #0x30]+ ldr q0, [x0]+ ldr q5, [x1, #0x10]+ ldr q6, [x1, #0x20]+ ldr q7, [x1, #0x30]+ ldr q4, [x1], #0x40+ sqdmulh v17.4s, v1.4s, v23.4s+ srshr v17.4s, v17.4s, #0x12+ cmgt v25.4s, v1.4s, v21.4s+ mls v1.4s, v17.4s, v22.4s+ bic v17.16b, v17.16b, v25.16b+ add v1.4s, v1.4s, v25.4s+ cmle v1.4s, v1.4s, #0+ orr v1.4s, #0x1+ mla v17.4s, v1.4s, v5.4s+ and v17.16b, v17.16b, v24.16b+ sqdmulh v18.4s, v2.4s, v23.4s+ srshr v18.4s, v18.4s, #0x12+ cmgt v25.4s, v2.4s, v21.4s+ mls v2.4s, v18.4s, v22.4s+ bic v18.16b, v18.16b, v25.16b+ add v2.4s, v2.4s, v25.4s+ cmle v2.4s, v2.4s, #0+ orr v2.4s, #0x1+ mla v18.4s, v2.4s, v6.4s+ and v18.16b, v18.16b, v24.16b+ sqdmulh v19.4s, v3.4s, v23.4s+ srshr v19.4s, v19.4s, #0x12+ cmgt v25.4s, v3.4s, v21.4s+ mls v3.4s, v19.4s, v22.4s+ bic v19.16b, v19.16b, v25.16b+ add v3.4s, v3.4s, v25.4s+ cmle v3.4s, v3.4s, #0+ orr v3.4s, #0x1+ mla v19.4s, v3.4s, v7.4s+ and v19.16b, v19.16b, v24.16b+ sqdmulh v16.4s, v0.4s, v23.4s+ srshr v16.4s, v16.4s, #0x12+ cmgt v25.4s, v0.4s, v21.4s+ mls v0.4s, v16.4s, v22.4s+ bic v16.16b, v16.16b, v25.16b+ add v0.4s, v0.4s, v25.4s+ cmle v0.4s, v0.4s, #0+ orr v0.4s, #0x1+ mla v16.4s, v0.4s, v4.4s+ and v16.16b, v16.16b, v24.16b+ str q17, [x0, #0x10]+ str q18, [x0, #0x20]+ str q19, [x0, #0x30]+ str q16, [x0], #0x40+ subs x3, x3, #0x1+ b.ne Lmld_poly_use_hint_32_loop+ ret+ .cfi_endproc++MLD_ASM_FN_SIZE(poly_use_hint_32_aarch64_asm)++#endif /* MLD_ARITH_BACKEND_AARCH64 && !MLD_CONFIG_NO_VERIFY_API && \+ !MLD_CONFIG_MULTILEVEL_NO_SHARED && \+ (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 65 \+ || MLD_CONFIG_PARAMETER_SET == 87) */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
@@ -0,0 +1,133 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+/*yaml+ Name: poly_use_hint_88_aarch64_asm+ Description: AArch64 hint application (alpha = (Q-1)/88)+ Signature: void mld_poly_use_hint_88_aarch64_asm(int32_t a[256], const int32_t h[256])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON]+ x0:+ type: buffer+ size_bytes: 1024+ permissions: read/write+ c_parameter: int32_t a[256]+ description: Input/output polynomial (256 x int32_t)+ x1:+ type: buffer+ size_bytes: 1024+ permissions: read-only+ c_parameter: const int32_t h[256]+ description: Hint polynomial (256 x int32_t)+*/++#include "../../../common.h"++#if defined(MLD_ARITH_BACKEND_AARCH64) && !defined(MLD_CONFIG_NO_VERIFY_API) && !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \+ (defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLD_CONFIG_PARAMETER_SET == 44)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/aarch64_opt/src/mldsa_poly_use_hint_88_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(poly_use_hint_88_aarch64_asm)+MLD_ASM_FN_SYMBOL(poly_use_hint_88_aarch64_asm)++ .cfi_startproc+ mov w4, #0xe001 // =57345+ movk w4, #0x7f, lsl #16+ dup v20.4s, w4+ mov w5, #0x6c00 // =27648+ movk w5, #0x7e, lsl #16+ dup v21.4s, w5+ mov w7, #0xe800 // =59392+ movk w7, #0x2, lsl #16+ dup v22.4s, w7+ mov w11, #0x581 // =1409+ movk w11, #0x5816, lsl #16+ dup v23.4s, w11+ movi v24.4s, #0x2b+ mov x3, #0x10 // =16++Lmld_poly_use_hint_88_loop:+ ldr q1, [x0, #0x10]+ ldr q2, [x0, #0x20]+ ldr q3, [x0, #0x30]+ ldr q0, [x0]+ ldr q5, [x1, #0x10]+ ldr q6, [x1, #0x20]+ ldr q7, [x1, #0x30]+ ldr q4, [x1], #0x40+ sqdmulh v17.4s, v1.4s, v23.4s+ srshr v17.4s, v17.4s, #0x11+ cmgt v25.4s, v1.4s, v21.4s+ mls v1.4s, v17.4s, v22.4s+ bic v17.16b, v17.16b, v25.16b+ add v1.4s, v1.4s, v25.4s+ cmle v1.4s, v1.4s, #0+ orr v1.4s, #0x1+ mla v17.4s, v1.4s, v5.4s+ cmgt v25.4s, v17.4s, v24.4s+ bic v17.16b, v17.16b, v25.16b+ umin v17.4s, v17.4s, v24.4s+ sqdmulh v18.4s, v2.4s, v23.4s+ srshr v18.4s, v18.4s, #0x11+ cmgt v25.4s, v2.4s, v21.4s+ mls v2.4s, v18.4s, v22.4s+ bic v18.16b, v18.16b, v25.16b+ add v2.4s, v2.4s, v25.4s+ cmle v2.4s, v2.4s, #0+ orr v2.4s, #0x1+ mla v18.4s, v2.4s, v6.4s+ cmgt v25.4s, v18.4s, v24.4s+ bic v18.16b, v18.16b, v25.16b+ umin v18.4s, v18.4s, v24.4s+ sqdmulh v19.4s, v3.4s, v23.4s+ srshr v19.4s, v19.4s, #0x11+ cmgt v25.4s, v3.4s, v21.4s+ mls v3.4s, v19.4s, v22.4s+ bic v19.16b, v19.16b, v25.16b+ add v3.4s, v3.4s, v25.4s+ cmle v3.4s, v3.4s, #0+ orr v3.4s, #0x1+ mla v19.4s, v3.4s, v7.4s+ cmgt v25.4s, v19.4s, v24.4s+ bic v19.16b, v19.16b, v25.16b+ umin v19.4s, v19.4s, v24.4s+ sqdmulh v16.4s, v0.4s, v23.4s+ srshr v16.4s, v16.4s, #0x11+ cmgt v25.4s, v0.4s, v21.4s+ mls v0.4s, v16.4s, v22.4s+ bic v16.16b, v16.16b, v25.16b+ add v0.4s, v0.4s, v25.4s+ cmle v0.4s, v0.4s, #0+ orr v0.4s, #0x1+ mla v16.4s, v0.4s, v4.4s+ cmgt v25.4s, v16.4s, v24.4s+ bic v16.16b, v16.16b, v25.16b+ umin v16.4s, v16.4s, v24.4s+ str q17, [x0, #0x10]+ str q18, [x0, #0x20]+ str q19, [x0, #0x30]+ str q16, [x0], #0x40+ subs x3, x3, #0x1+ b.ne Lmld_poly_use_hint_88_loop+ ret+ .cfi_endproc++MLD_ASM_FN_SIZE(poly_use_hint_88_aarch64_asm)++#endif /* MLD_ARITH_BACKEND_AARCH64 && !MLD_CONFIG_NO_VERIFY_API && \+ !MLD_CONFIG_MULTILEVEL_NO_SHARED && \+ (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 44) \+ */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
@@ -0,0 +1,157 @@+/* Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*yaml+ Name: polyvecl_pointwise_acc_montgomery_l4_aarch64_asm+ Description: AArch64 pointwise multiply-accumulate of length-4 polynomial vectors+ Signature: void mld_polyvecl_pointwise_acc_montgomery_l4_aarch64_asm(int32_t r[256], const int32_t a[4][256], const int32_t b[4][256])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON]+ x0:+ type: buffer+ size_bytes: 1024+ permissions: write-only+ c_parameter: int32_t r[256]+ description: Output polynomial (256 x int32_t)+ x1:+ type: buffer+ size_bytes: 4096+ permissions: read-only+ c_parameter: const int32_t a[4][256]+ description: Input polynomial vector a (4 x 256 x int32_t)+ x2:+ type: buffer+ size_bytes: 4096+ permissions: read-only+ c_parameter: const int32_t b[4][256]+ description: Input polynomial vector b (4 x 256 x int32_t)+*/++#include "../../../common.h"+#if defined(MLD_ARITH_BACKEND_AARCH64) && !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \+ (defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLDSA_L == 4)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/aarch64_opt/src/mldsa_polyvecl_pointwise_acc_montgomery_l4_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(polyvecl_pointwise_acc_montgomery_l4_aarch64_asm)+MLD_ASM_FN_SYMBOL(polyvecl_pointwise_acc_montgomery_l4_aarch64_asm)++ .cfi_startproc+ mov w3, #0xe001 // =57345+ movk w3, #0x7f, lsl #16+ dup v0.4s, w3+ mov w3, #0x2001 // =8193+ movk w3, #0x380, lsl #16+ dup v1.4s, w3+ mov x3, #0x40 // =64++Lmld_polyvecl_pointwise_acc_montgomery_l4_loop_start:+ ldr q17, [x1, #0x10]+ ldr q18, [x1, #0x20]+ ldr q19, [x1, #0x30]+ ldr q16, [x1], #0x40+ ldr q21, [x2, #0x10]+ ldr q22, [x2, #0x20]+ ldr q23, [x2, #0x30]+ ldr q20, [x2], #0x40+ smull v24.2d, v16.2s, v20.2s+ smull2 v25.2d, v16.4s, v20.4s+ smull v26.2d, v17.2s, v21.2s+ smull2 v27.2d, v17.4s, v21.4s+ smull v28.2d, v18.2s, v22.2s+ smull2 v29.2d, v18.4s, v22.4s+ smull v30.2d, v19.2s, v23.2s+ smull2 v31.2d, v19.4s, v23.4s+ ldr q16, [x1, #0x3c0]+ ldr q17, [x1, #0x3d0]+ ldr q18, [x1, #0x3e0]+ ldr q19, [x1, #0x3f0]+ ldr q20, [x2, #0x3c0]+ ldr q21, [x2, #0x3d0]+ ldr q22, [x2, #0x3e0]+ ldr q23, [x2, #0x3f0]+ smlal v24.2d, v16.2s, v20.2s+ smlal2 v25.2d, v16.4s, v20.4s+ smlal v26.2d, v17.2s, v21.2s+ smlal2 v27.2d, v17.4s, v21.4s+ smlal v28.2d, v18.2s, v22.2s+ smlal2 v29.2d, v18.4s, v22.4s+ smlal v30.2d, v19.2s, v23.2s+ smlal2 v31.2d, v19.4s, v23.4s+ ldr q16, [x1, #0x7c0]+ ldr q17, [x1, #0x7d0]+ ldr q18, [x1, #0x7e0]+ ldr q19, [x1, #0x7f0]+ ldr q20, [x2, #0x7c0]+ ldr q21, [x2, #0x7d0]+ ldr q22, [x2, #0x7e0]+ ldr q23, [x2, #0x7f0]+ smlal v24.2d, v16.2s, v20.2s+ smlal2 v25.2d, v16.4s, v20.4s+ smlal v26.2d, v17.2s, v21.2s+ smlal2 v27.2d, v17.4s, v21.4s+ smlal v28.2d, v18.2s, v22.2s+ smlal2 v29.2d, v18.4s, v22.4s+ smlal v30.2d, v19.2s, v23.2s+ smlal2 v31.2d, v19.4s, v23.4s+ ldr q16, [x1, #0xbc0]+ ldr q17, [x1, #0xbd0]+ ldr q18, [x1, #0xbe0]+ ldr q19, [x1, #0xbf0]+ ldr q20, [x2, #0xbc0]+ ldr q21, [x2, #0xbd0]+ ldr q22, [x2, #0xbe0]+ ldr q23, [x2, #0xbf0]+ smlal v24.2d, v16.2s, v20.2s+ smlal2 v25.2d, v16.4s, v20.4s+ smlal v26.2d, v17.2s, v21.2s+ smlal2 v27.2d, v17.4s, v21.4s+ smlal v28.2d, v18.2s, v22.2s+ smlal2 v29.2d, v18.4s, v22.4s+ smlal v30.2d, v19.2s, v23.2s+ smlal2 v31.2d, v19.4s, v23.4s+ uzp1 v16.4s, v24.4s, v25.4s+ mul v16.4s, v16.4s, v1.4s+ smlsl v24.2d, v16.2s, v0.2s+ smlsl2 v25.2d, v16.4s, v0.4s+ uzp2 v16.4s, v24.4s, v25.4s+ uzp1 v17.4s, v26.4s, v27.4s+ mul v17.4s, v17.4s, v1.4s+ smlsl v26.2d, v17.2s, v0.2s+ smlsl2 v27.2d, v17.4s, v0.4s+ uzp2 v17.4s, v26.4s, v27.4s+ uzp1 v18.4s, v28.4s, v29.4s+ mul v18.4s, v18.4s, v1.4s+ smlsl v28.2d, v18.2s, v0.2s+ smlsl2 v29.2d, v18.4s, v0.4s+ uzp2 v18.4s, v28.4s, v29.4s+ uzp1 v19.4s, v30.4s, v31.4s+ mul v19.4s, v19.4s, v1.4s+ smlsl v30.2d, v19.2s, v0.2s+ smlsl2 v31.2d, v19.4s, v0.4s+ uzp2 v19.4s, v30.4s, v31.4s+ str q17, [x0, #0x10]+ str q18, [x0, #0x20]+ str q19, [x0, #0x30]+ str q16, [x0], #0x40+ subs x3, x3, #0x4+ cbnz x3, Lmld_polyvecl_pointwise_acc_montgomery_l4_loop_start+ ret+ .cfi_endproc++MLD_ASM_FN_SIZE(polyvecl_pointwise_acc_montgomery_l4_aarch64_asm)++#endif /* MLD_ARITH_BACKEND_AARCH64 && !MLD_CONFIG_MULTILEVEL_NO_SHARED && \+ (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLDSA_L == 4) */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
@@ -0,0 +1,173 @@+/* Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*yaml+ Name: polyvecl_pointwise_acc_montgomery_l5_aarch64_asm+ Description: AArch64 pointwise multiply-accumulate of length-5 polynomial vectors+ Signature: void mld_polyvecl_pointwise_acc_montgomery_l5_aarch64_asm(int32_t r[256], const int32_t a[5][256], const int32_t b[5][256])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON]+ x0:+ type: buffer+ size_bytes: 1024+ permissions: write-only+ c_parameter: int32_t r[256]+ description: Output polynomial (256 x int32_t)+ x1:+ type: buffer+ size_bytes: 5120+ permissions: read-only+ c_parameter: const int32_t a[5][256]+ description: Input polynomial vector a (5 x 256 x int32_t)+ x2:+ type: buffer+ size_bytes: 5120+ permissions: read-only+ c_parameter: const int32_t b[5][256]+ description: Input polynomial vector b (5 x 256 x int32_t)+*/++#include "../../../common.h"+#if defined(MLD_ARITH_BACKEND_AARCH64) && !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \+ (defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLDSA_L == 5)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/aarch64_opt/src/mldsa_polyvecl_pointwise_acc_montgomery_l5_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(polyvecl_pointwise_acc_montgomery_l5_aarch64_asm)+MLD_ASM_FN_SYMBOL(polyvecl_pointwise_acc_montgomery_l5_aarch64_asm)++ .cfi_startproc+ mov w3, #0xe001 // =57345+ movk w3, #0x7f, lsl #16+ dup v0.4s, w3+ mov w3, #0x2001 // =8193+ movk w3, #0x380, lsl #16+ dup v1.4s, w3+ mov x3, #0x40 // =64++Lmld_polyvecl_pointwise_acc_montgomery_l5_loop_start:+ ldr q17, [x1, #0x10]+ ldr q18, [x1, #0x20]+ ldr q19, [x1, #0x30]+ ldr q16, [x1], #0x40+ ldr q21, [x2, #0x10]+ ldr q22, [x2, #0x20]+ ldr q23, [x2, #0x30]+ ldr q20, [x2], #0x40+ smull v24.2d, v16.2s, v20.2s+ smull2 v25.2d, v16.4s, v20.4s+ smull v26.2d, v17.2s, v21.2s+ smull2 v27.2d, v17.4s, v21.4s+ smull v28.2d, v18.2s, v22.2s+ smull2 v29.2d, v18.4s, v22.4s+ smull v30.2d, v19.2s, v23.2s+ smull2 v31.2d, v19.4s, v23.4s+ ldr q16, [x1, #0x3c0]+ ldr q17, [x1, #0x3d0]+ ldr q18, [x1, #0x3e0]+ ldr q19, [x1, #0x3f0]+ ldr q20, [x2, #0x3c0]+ ldr q21, [x2, #0x3d0]+ ldr q22, [x2, #0x3e0]+ ldr q23, [x2, #0x3f0]+ smlal v24.2d, v16.2s, v20.2s+ smlal2 v25.2d, v16.4s, v20.4s+ smlal v26.2d, v17.2s, v21.2s+ smlal2 v27.2d, v17.4s, v21.4s+ smlal v28.2d, v18.2s, v22.2s+ smlal2 v29.2d, v18.4s, v22.4s+ smlal v30.2d, v19.2s, v23.2s+ smlal2 v31.2d, v19.4s, v23.4s+ ldr q16, [x1, #0x7c0]+ ldr q17, [x1, #0x7d0]+ ldr q18, [x1, #0x7e0]+ ldr q19, [x1, #0x7f0]+ ldr q20, [x2, #0x7c0]+ ldr q21, [x2, #0x7d0]+ ldr q22, [x2, #0x7e0]+ ldr q23, [x2, #0x7f0]+ smlal v24.2d, v16.2s, v20.2s+ smlal2 v25.2d, v16.4s, v20.4s+ smlal v26.2d, v17.2s, v21.2s+ smlal2 v27.2d, v17.4s, v21.4s+ smlal v28.2d, v18.2s, v22.2s+ smlal2 v29.2d, v18.4s, v22.4s+ smlal v30.2d, v19.2s, v23.2s+ smlal2 v31.2d, v19.4s, v23.4s+ ldr q16, [x1, #0xbc0]+ ldr q17, [x1, #0xbd0]+ ldr q18, [x1, #0xbe0]+ ldr q19, [x1, #0xbf0]+ ldr q20, [x2, #0xbc0]+ ldr q21, [x2, #0xbd0]+ ldr q22, [x2, #0xbe0]+ ldr q23, [x2, #0xbf0]+ smlal v24.2d, v16.2s, v20.2s+ smlal2 v25.2d, v16.4s, v20.4s+ smlal v26.2d, v17.2s, v21.2s+ smlal2 v27.2d, v17.4s, v21.4s+ smlal v28.2d, v18.2s, v22.2s+ smlal2 v29.2d, v18.4s, v22.4s+ smlal v30.2d, v19.2s, v23.2s+ smlal2 v31.2d, v19.4s, v23.4s+ ldr q16, [x1, #0xfc0]+ ldr q17, [x1, #0xfd0]+ ldr q18, [x1, #0xfe0]+ ldr q19, [x1, #0xff0]+ ldr q20, [x2, #0xfc0]+ ldr q21, [x2, #0xfd0]+ ldr q22, [x2, #0xfe0]+ ldr q23, [x2, #0xff0]+ smlal v24.2d, v16.2s, v20.2s+ smlal2 v25.2d, v16.4s, v20.4s+ smlal v26.2d, v17.2s, v21.2s+ smlal2 v27.2d, v17.4s, v21.4s+ smlal v28.2d, v18.2s, v22.2s+ smlal2 v29.2d, v18.4s, v22.4s+ smlal v30.2d, v19.2s, v23.2s+ smlal2 v31.2d, v19.4s, v23.4s+ uzp1 v16.4s, v24.4s, v25.4s+ mul v16.4s, v16.4s, v1.4s+ smlsl v24.2d, v16.2s, v0.2s+ smlsl2 v25.2d, v16.4s, v0.4s+ uzp2 v16.4s, v24.4s, v25.4s+ uzp1 v17.4s, v26.4s, v27.4s+ mul v17.4s, v17.4s, v1.4s+ smlsl v26.2d, v17.2s, v0.2s+ smlsl2 v27.2d, v17.4s, v0.4s+ uzp2 v17.4s, v26.4s, v27.4s+ uzp1 v18.4s, v28.4s, v29.4s+ mul v18.4s, v18.4s, v1.4s+ smlsl v28.2d, v18.2s, v0.2s+ smlsl2 v29.2d, v18.4s, v0.4s+ uzp2 v18.4s, v28.4s, v29.4s+ uzp1 v19.4s, v30.4s, v31.4s+ mul v19.4s, v19.4s, v1.4s+ smlsl v30.2d, v19.2s, v0.2s+ smlsl2 v31.2d, v19.4s, v0.4s+ uzp2 v19.4s, v30.4s, v31.4s+ str q17, [x0, #0x10]+ str q18, [x0, #0x20]+ str q19, [x0, #0x30]+ str q16, [x0], #0x40+ subs x3, x3, #0x4+ cbnz x3, Lmld_polyvecl_pointwise_acc_montgomery_l5_loop_start+ ret+ .cfi_endproc++MLD_ASM_FN_SIZE(polyvecl_pointwise_acc_montgomery_l5_aarch64_asm)++#endif /* MLD_ARITH_BACKEND_AARCH64 && !MLD_CONFIG_MULTILEVEL_NO_SHARED && \+ (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLDSA_L == 5) */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
@@ -0,0 +1,205 @@+/* Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*yaml+ Name: polyvecl_pointwise_acc_montgomery_l7_aarch64_asm+ Description: AArch64 pointwise multiply-accumulate of length-7 polynomial vectors+ Signature: void mld_polyvecl_pointwise_acc_montgomery_l7_aarch64_asm(int32_t r[256], const int32_t a[7][256], const int32_t b[7][256])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON]+ x0:+ type: buffer+ size_bytes: 1024+ permissions: write-only+ c_parameter: int32_t r[256]+ description: Output polynomial (256 x int32_t)+ x1:+ type: buffer+ size_bytes: 7168+ permissions: read-only+ c_parameter: const int32_t a[7][256]+ description: Input polynomial vector a (7 x 256 x int32_t)+ x2:+ type: buffer+ size_bytes: 7168+ permissions: read-only+ c_parameter: const int32_t b[7][256]+ description: Input polynomial vector b (7 x 256 x int32_t)+*/++#include "../../../common.h"+#if defined(MLD_ARITH_BACKEND_AARCH64) && !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \+ (defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLDSA_L == 7)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/aarch64_opt/src/mldsa_polyvecl_pointwise_acc_montgomery_l7_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(polyvecl_pointwise_acc_montgomery_l7_aarch64_asm)+MLD_ASM_FN_SYMBOL(polyvecl_pointwise_acc_montgomery_l7_aarch64_asm)++ .cfi_startproc+ mov w3, #0xe001 // =57345+ movk w3, #0x7f, lsl #16+ dup v0.4s, w3+ mov w3, #0x2001 // =8193+ movk w3, #0x380, lsl #16+ dup v1.4s, w3+ mov x3, #0x40 // =64++Lmld_polyvecl_pointwise_acc_montgomery_l7_loop_start:+ ldr q17, [x1, #0x10]+ ldr q18, [x1, #0x20]+ ldr q19, [x1, #0x30]+ ldr q16, [x1], #0x40+ ldr q21, [x2, #0x10]+ ldr q22, [x2, #0x20]+ ldr q23, [x2, #0x30]+ ldr q20, [x2], #0x40+ smull v24.2d, v16.2s, v20.2s+ smull2 v25.2d, v16.4s, v20.4s+ smull v26.2d, v17.2s, v21.2s+ smull2 v27.2d, v17.4s, v21.4s+ smull v28.2d, v18.2s, v22.2s+ smull2 v29.2d, v18.4s, v22.4s+ smull v30.2d, v19.2s, v23.2s+ smull2 v31.2d, v19.4s, v23.4s+ ldr q16, [x1, #0x3c0]+ ldr q17, [x1, #0x3d0]+ ldr q18, [x1, #0x3e0]+ ldr q19, [x1, #0x3f0]+ ldr q20, [x2, #0x3c0]+ ldr q21, [x2, #0x3d0]+ ldr q22, [x2, #0x3e0]+ ldr q23, [x2, #0x3f0]+ smlal v24.2d, v16.2s, v20.2s+ smlal2 v25.2d, v16.4s, v20.4s+ smlal v26.2d, v17.2s, v21.2s+ smlal2 v27.2d, v17.4s, v21.4s+ smlal v28.2d, v18.2s, v22.2s+ smlal2 v29.2d, v18.4s, v22.4s+ smlal v30.2d, v19.2s, v23.2s+ smlal2 v31.2d, v19.4s, v23.4s+ ldr q16, [x1, #0x7c0]+ ldr q17, [x1, #0x7d0]+ ldr q18, [x1, #0x7e0]+ ldr q19, [x1, #0x7f0]+ ldr q20, [x2, #0x7c0]+ ldr q21, [x2, #0x7d0]+ ldr q22, [x2, #0x7e0]+ ldr q23, [x2, #0x7f0]+ smlal v24.2d, v16.2s, v20.2s+ smlal2 v25.2d, v16.4s, v20.4s+ smlal v26.2d, v17.2s, v21.2s+ smlal2 v27.2d, v17.4s, v21.4s+ smlal v28.2d, v18.2s, v22.2s+ smlal2 v29.2d, v18.4s, v22.4s+ smlal v30.2d, v19.2s, v23.2s+ smlal2 v31.2d, v19.4s, v23.4s+ ldr q16, [x1, #0xbc0]+ ldr q17, [x1, #0xbd0]+ ldr q18, [x1, #0xbe0]+ ldr q19, [x1, #0xbf0]+ ldr q20, [x2, #0xbc0]+ ldr q21, [x2, #0xbd0]+ ldr q22, [x2, #0xbe0]+ ldr q23, [x2, #0xbf0]+ smlal v24.2d, v16.2s, v20.2s+ smlal2 v25.2d, v16.4s, v20.4s+ smlal v26.2d, v17.2s, v21.2s+ smlal2 v27.2d, v17.4s, v21.4s+ smlal v28.2d, v18.2s, v22.2s+ smlal2 v29.2d, v18.4s, v22.4s+ smlal v30.2d, v19.2s, v23.2s+ smlal2 v31.2d, v19.4s, v23.4s+ ldr q16, [x1, #0xfc0]+ ldr q17, [x1, #0xfd0]+ ldr q18, [x1, #0xfe0]+ ldr q19, [x1, #0xff0]+ ldr q20, [x2, #0xfc0]+ ldr q21, [x2, #0xfd0]+ ldr q22, [x2, #0xfe0]+ ldr q23, [x2, #0xff0]+ smlal v24.2d, v16.2s, v20.2s+ smlal2 v25.2d, v16.4s, v20.4s+ smlal v26.2d, v17.2s, v21.2s+ smlal2 v27.2d, v17.4s, v21.4s+ smlal v28.2d, v18.2s, v22.2s+ smlal2 v29.2d, v18.4s, v22.4s+ smlal v30.2d, v19.2s, v23.2s+ smlal2 v31.2d, v19.4s, v23.4s+ ldr q16, [x1, #0x13c0]+ ldr q17, [x1, #0x13d0]+ ldr q18, [x1, #0x13e0]+ ldr q19, [x1, #0x13f0]+ ldr q20, [x2, #0x13c0]+ ldr q21, [x2, #0x13d0]+ ldr q22, [x2, #0x13e0]+ ldr q23, [x2, #0x13f0]+ smlal v24.2d, v16.2s, v20.2s+ smlal2 v25.2d, v16.4s, v20.4s+ smlal v26.2d, v17.2s, v21.2s+ smlal2 v27.2d, v17.4s, v21.4s+ smlal v28.2d, v18.2s, v22.2s+ smlal2 v29.2d, v18.4s, v22.4s+ smlal v30.2d, v19.2s, v23.2s+ smlal2 v31.2d, v19.4s, v23.4s+ ldr q16, [x1, #0x17c0]+ ldr q17, [x1, #0x17d0]+ ldr q18, [x1, #0x17e0]+ ldr q19, [x1, #0x17f0]+ ldr q20, [x2, #0x17c0]+ ldr q21, [x2, #0x17d0]+ ldr q22, [x2, #0x17e0]+ ldr q23, [x2, #0x17f0]+ smlal v24.2d, v16.2s, v20.2s+ smlal2 v25.2d, v16.4s, v20.4s+ smlal v26.2d, v17.2s, v21.2s+ smlal2 v27.2d, v17.4s, v21.4s+ smlal v28.2d, v18.2s, v22.2s+ smlal2 v29.2d, v18.4s, v22.4s+ smlal v30.2d, v19.2s, v23.2s+ smlal2 v31.2d, v19.4s, v23.4s+ uzp1 v16.4s, v24.4s, v25.4s+ mul v16.4s, v16.4s, v1.4s+ smlsl v24.2d, v16.2s, v0.2s+ smlsl2 v25.2d, v16.4s, v0.4s+ uzp2 v16.4s, v24.4s, v25.4s+ uzp1 v17.4s, v26.4s, v27.4s+ mul v17.4s, v17.4s, v1.4s+ smlsl v26.2d, v17.2s, v0.2s+ smlsl2 v27.2d, v17.4s, v0.4s+ uzp2 v17.4s, v26.4s, v27.4s+ uzp1 v18.4s, v28.4s, v29.4s+ mul v18.4s, v18.4s, v1.4s+ smlsl v28.2d, v18.2s, v0.2s+ smlsl2 v29.2d, v18.4s, v0.4s+ uzp2 v18.4s, v28.4s, v29.4s+ uzp1 v19.4s, v30.4s, v31.4s+ mul v19.4s, v19.4s, v1.4s+ smlsl v30.2d, v19.2s, v0.2s+ smlsl2 v31.2d, v19.4s, v0.4s+ uzp2 v19.4s, v30.4s, v31.4s+ str q17, [x0, #0x10]+ str q18, [x0, #0x20]+ str q19, [x0, #0x30]+ str q16, [x0], #0x40+ subs x3, x3, #0x4+ cbnz x3, Lmld_polyvecl_pointwise_acc_montgomery_l7_loop_start+ ret+ .cfi_endproc++MLD_ASM_FN_SIZE(polyvecl_pointwise_acc_montgomery_l7_aarch64_asm)++#endif /* MLD_ARITH_BACKEND_AARCH64 && !MLD_CONFIG_MULTILEVEL_NO_SHARED && \+ (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLDSA_L == 7) */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
@@ -0,0 +1,103 @@+/*+ * Copyright (c) The mldsa-native project authors+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*yaml+ Name: polyz_unpack_17_aarch64_asm+ Description: AArch64 unpacking of 17-bit packed coefficients+ Signature: void mld_polyz_unpack_17_aarch64_asm(int32_t r[256], const uint8_t buf[576], const uint8_t indices[64])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON]+ x0:+ type: buffer+ size_bytes: 1024+ permissions: write-only+ c_parameter: int32_t r[256]+ description: Output polynomial (256 x int32_t)+ x1:+ type: buffer+ size_bytes: 576+ permissions: read-only+ c_parameter: const uint8_t buf[576]+ description: Packed input bytes+ x2:+ type: buffer+ size_bytes: 64+ permissions: read-only+ c_parameter: const uint8_t indices[64]+ description: Permutation index table (64 x uint8_t)+*/++ #include "../../../common.h"+#if defined(MLD_ARITH_BACKEND_AARCH64) && \+ (!defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API)) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \+ (defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLD_CONFIG_PARAMETER_SET == 44)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/aarch64_opt/src/mldsa_polyz_unpack_17_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(polyz_unpack_17_aarch64_asm)+MLD_ASM_FN_SYMBOL(polyz_unpack_17_aarch64_asm)++ .cfi_startproc+ ldr q24, [x2]+ ldr q25, [x2, #0x10]+ ldr q26, [x2, #0x20]+ ldr q27, [x2, #0x30]+ mov x3, #0xfe00000000 // =1090921693184+ mov v28.d[0], x3+ mov x3, #0xfc // =252+ movk x3, #0xfa, lsl #32+ mov v28.d[1], x3+ movi v29.4s, #0x3, msl #16+ movi v30.4s, #0x2, lsl #16+ mov x9, #0x10 // =16++Lmld_polyz_unpack_17_loop:+ ld1 { v0.16b, v1.16b }, [x1]+ add x1, x1, #0x14+ ld1 { v2.16b }, [x1], #16+ tbl v4.16b, { v0.16b }, v24.16b+ tbl v5.16b, { v0.16b, v1.16b }, v25.16b+ tbl v6.16b, { v1.16b }, v26.16b+ tbl v7.16b, { v1.16b, v2.16b }, v27.16b+ ushl v4.4s, v4.4s, v28.4s+ and v4.16b, v4.16b, v29.16b+ sub v4.4s, v30.4s, v4.4s+ ushl v5.4s, v5.4s, v28.4s+ and v5.16b, v5.16b, v29.16b+ sub v5.4s, v30.4s, v5.4s+ ushl v6.4s, v6.4s, v28.4s+ and v6.16b, v6.16b, v29.16b+ sub v6.4s, v30.4s, v6.4s+ ushl v7.4s, v7.4s, v28.4s+ and v7.16b, v7.16b, v29.16b+ sub v7.4s, v30.4s, v7.4s+ str q5, [x0, #0x10]+ str q6, [x0, #0x20]+ str q7, [x0, #0x30]+ str q4, [x0], #0x40+ subs x9, x9, #0x1+ b.ne Lmld_polyz_unpack_17_loop+ ret+ .cfi_endproc++MLD_ASM_FN_SIZE(polyz_unpack_17_aarch64_asm)++#endif /* MLD_ARITH_BACKEND_AARCH64 && (!MLD_CONFIG_NO_SIGN_API || \+ !MLD_CONFIG_NO_VERIFY_API) && !MLD_CONFIG_MULTILEVEL_NO_SHARED && \+ (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 44) \+ */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
@@ -0,0 +1,100 @@+/*+ * Copyright (c) The mldsa-native project authors+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*yaml+ Name: polyz_unpack_19_aarch64_asm+ Description: AArch64 unpacking of 19-bit packed coefficients+ Signature: void mld_polyz_unpack_19_aarch64_asm(int32_t r[256], const uint8_t buf[640], const uint8_t indices[64])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON]+ x0:+ type: buffer+ size_bytes: 1024+ permissions: write-only+ c_parameter: int32_t r[256]+ description: Output polynomial (256 x int32_t)+ x1:+ type: buffer+ size_bytes: 640+ permissions: read-only+ c_parameter: const uint8_t buf[640]+ description: Packed input bytes+ x2:+ type: buffer+ size_bytes: 64+ permissions: read-only+ c_parameter: const uint8_t indices[64]+ description: Permutation index table (64 x uint8_t)+*/++ #include "../../../common.h"+#if defined(MLD_ARITH_BACKEND_AARCH64) && \+ (!defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API)) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \+ (defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || (MLD_CONFIG_PARAMETER_SET == 65 || MLD_CONFIG_PARAMETER_SET == 87))++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/aarch64_opt/src/mldsa_polyz_unpack_19_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(polyz_unpack_19_aarch64_asm)+MLD_ASM_FN_SYMBOL(polyz_unpack_19_aarch64_asm)++ .cfi_startproc+ ldr q24, [x2]+ ldr q25, [x2, #0x10]+ ldr q26, [x2, #0x20]+ ldr q27, [x2, #0x30]+ mov x3, #0xfc00000000 // =1082331758592+ dup v28.2d, x3+ movi v29.4s, #0xf, msl #16+ movi v30.4s, #0x8, lsl #16+ mov x9, #0x10 // =16++Lmld_polyz_unpack_19_loop:+ ld1 { v0.16b, v1.16b }, [x1]+ add x1, x1, #0x18+ ld1 { v2.16b }, [x1], #16+ tbl v4.16b, { v0.16b }, v24.16b+ tbl v5.16b, { v0.16b, v1.16b }, v25.16b+ tbl v6.16b, { v1.16b }, v26.16b+ tbl v7.16b, { v1.16b, v2.16b }, v27.16b+ ushl v4.4s, v4.4s, v28.4s+ and v4.16b, v4.16b, v29.16b+ sub v4.4s, v30.4s, v4.4s+ ushl v5.4s, v5.4s, v28.4s+ and v5.16b, v5.16b, v29.16b+ sub v5.4s, v30.4s, v5.4s+ ushl v6.4s, v6.4s, v28.4s+ and v6.16b, v6.16b, v29.16b+ sub v6.4s, v30.4s, v6.4s+ ushl v7.4s, v7.4s, v28.4s+ and v7.16b, v7.16b, v29.16b+ sub v7.4s, v30.4s, v7.4s+ str q5, [x0, #0x10]+ str q6, [x0, #0x20]+ str q7, [x0, #0x30]+ str q4, [x0], #0x40+ subs x9, x9, #0x1+ b.ne Lmld_polyz_unpack_19_loop+ ret+ .cfi_endproc++MLD_ASM_FN_SIZE(polyz_unpack_19_aarch64_asm)++#endif /* MLD_ARITH_BACKEND_AARCH64 && (!MLD_CONFIG_NO_SIGN_API || \+ !MLD_CONFIG_NO_VERIFY_API) && !MLD_CONFIG_MULTILEVEL_NO_SHARED && \+ (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 65 \+ || MLD_CONFIG_PARAMETER_SET == 87) */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
@@ -0,0 +1,222 @@+/*+ * Copyright (c) The mldsa-native project authors+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*yaml+ Name: rej_uniform_aarch64_asm+ Description: AArch64 rejection sampling of uniform coefficients mod q+ Signature: uint64_t mld_rej_uniform_aarch64_asm(int32_t r[256], const uint8_t *buf, unsigned buflen, const uint8_t table[256])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON]+ x0:+ type: buffer+ size_bytes: 1024+ permissions: write-only+ c_parameter: int32_t r[256]+ description: Output buffer (256 x int32_t)+ x1:+ type: buffer+ size_bytes: x2+ permissions: read-only+ c_parameter: const uint8_t *buf+ description: Input buffer+ x2:+ type: scalar+ c_parameter: unsigned buflen+ description: Length of input buffer (must be a multiple of 24)+ test_with: 840 # MLD_POLY_UNIFORM_NBLOCKS * SHAKE128_RATE = 5 * 168+ x3:+ type: buffer+ size_bytes: 256+ permissions: read-only+ c_parameter: const uint8_t table[256]+ description: Lookup table (256 x uint8_t)+*/++ #include "../../../common.h"+#if defined(MLD_ARITH_BACKEND_AARCH64) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/aarch64_opt/src/mldsa_rej_uniform_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(rej_uniform_aarch64_asm)+MLD_ASM_FN_SYMBOL(rej_uniform_aarch64_asm)++ .cfi_startproc+ sub sp, sp, #0x440+ .cfi_adjust_cfa_offset 0x440+ mov x7, #0x1 // =1+ movk x7, #0x2, lsl #32+ mov v31.d[0], x7+ mov x7, #0x4 // =4+ movk x7, #0x8, lsl #32+ mov v31.d[1], x7+ mov w7, #0xe001 // =57345+ movk w7, #0x7f, lsl #16+ dup v30.4s, w7+ mov x8, sp+ mov x7, x8+ mov x11, #0x0 // =0+ eor v16.16b, v16.16b, v16.16b++Lmld_rej_uniform_initial_zero:+ str q16, [x7], #0x40+ stur q16, [x7, #-0x30]+ stur q16, [x7, #-0x20]+ stur q16, [x7, #-0x10]+ add x11, x11, #0x10+ cmp x11, #0x100+ b.lt Lmld_rej_uniform_initial_zero+ mov x7, x8+ mov x9, #0x0 // =0+ mov x4, #0x100 // =256+ cmp x2, #0x30+ b.lo Lmld_rej_uniform_loop48_end++Lmld_rej_uniform_loop48:+ cmp x9, x4+ b.hs Lmld_rej_uniform_memory_copy+ sub x2, x2, #0x30+ ld3 { v0.16b, v1.16b, v2.16b }, [x1], #48+ movi v4.16b, #0x80+ bic v2.16b, v2.16b, v4.16b+ zip1 v4.16b, v0.16b, v1.16b+ zip2 v5.16b, v0.16b, v1.16b+ ushll v6.8h, v2.8b, #0x0+ ushll2 v7.8h, v2.16b, #0x0+ zip1 v16.8h, v4.8h, v6.8h+ zip2 v17.8h, v4.8h, v6.8h+ zip1 v18.8h, v5.8h, v7.8h+ zip2 v19.8h, v5.8h, v7.8h+ cmhi v4.4s, v30.4s, v16.4s+ cmhi v5.4s, v30.4s, v17.4s+ cmhi v6.4s, v30.4s, v18.4s+ cmhi v7.4s, v30.4s, v19.4s+ and v4.16b, v4.16b, v31.16b+ and v5.16b, v5.16b, v31.16b+ and v6.16b, v6.16b, v31.16b+ and v7.16b, v7.16b, v31.16b+ uaddlv d20, v4.4s+ uaddlv d21, v5.4s+ uaddlv d22, v6.4s+ uaddlv d23, v7.4s+ fmov x12, d20+ fmov x13, d21+ fmov x14, d22+ fmov x15, d23+ ldr q24, [x3, x12, lsl #4]+ ldr q25, [x3, x13, lsl #4]+ ldr q26, [x3, x14, lsl #4]+ ldr q27, [x3, x15, lsl #4]+ cnt v4.16b, v4.16b+ cnt v5.16b, v5.16b+ cnt v6.16b, v6.16b+ cnt v7.16b, v7.16b+ uaddlv d20, v4.4s+ uaddlv d21, v5.4s+ uaddlv d22, v6.4s+ uaddlv d23, v7.4s+ fmov x12, d20+ fmov x13, d21+ fmov x14, d22+ fmov x15, d23+ tbl v16.16b, { v16.16b }, v24.16b+ tbl v17.16b, { v17.16b }, v25.16b+ tbl v18.16b, { v18.16b }, v26.16b+ tbl v19.16b, { v19.16b }, v27.16b+ st1 { v16.4s }, [x7]+ add x7, x7, x12, lsl #2+ st1 { v17.4s }, [x7]+ add x7, x7, x13, lsl #2+ st1 { v18.4s }, [x7]+ add x7, x7, x14, lsl #2+ st1 { v19.4s }, [x7]+ add x7, x7, x15, lsl #2+ add x12, x12, x13+ add x14, x14, x15+ add x9, x9, x12+ add x9, x9, x14+ cmp x2, #0x30+ b.hs Lmld_rej_uniform_loop48++Lmld_rej_uniform_loop48_end:+ cmp x9, x4+ b.hs Lmld_rej_uniform_memory_copy+ cmp x2, #0x18+ b.lo Lmld_rej_uniform_memory_copy+ sub x2, x2, #0x18+ ld3 { v0.8b, v1.8b, v2.8b }, [x1], #24+ movi v4.16b, #0x80+ bic v2.16b, v2.16b, v4.16b+ zip1 v4.16b, v0.16b, v1.16b+ ushll v6.8h, v2.8b, #0x0+ zip1 v16.8h, v4.8h, v6.8h+ zip2 v17.8h, v4.8h, v6.8h+ cmhi v4.4s, v30.4s, v16.4s+ cmhi v5.4s, v30.4s, v17.4s+ and v4.16b, v4.16b, v31.16b+ and v5.16b, v5.16b, v31.16b+ uaddlv d20, v4.4s+ uaddlv d21, v5.4s+ fmov x12, d20+ fmov x13, d21+ ldr q24, [x3, x12, lsl #4]+ ldr q25, [x3, x13, lsl #4]+ cnt v4.16b, v4.16b+ cnt v5.16b, v5.16b+ uaddlv d20, v4.4s+ uaddlv d21, v5.4s+ fmov x12, d20+ fmov x13, d21+ tbl v16.16b, { v16.16b }, v24.16b+ tbl v17.16b, { v17.16b }, v25.16b+ st1 { v16.4s }, [x7]+ add x7, x7, x12, lsl #2+ st1 { v17.4s }, [x7]+ add x7, x7, x13, lsl #2+ add x9, x9, x12+ add x9, x9, x13++Lmld_rej_uniform_memory_copy:+ cmp x9, x4+ csel x9, x9, x4, lo+ mov x11, #0x0 // =0+ mov x7, x8++Lmld_rej_uniform_final_copy:+ ldr q16, [x7], #0x40+ ldur q17, [x7, #-0x30]+ ldur q18, [x7, #-0x20]+ ldur q19, [x7, #-0x10]+ str q16, [x0], #0x40+ stur q17, [x0, #-0x30]+ stur q18, [x0, #-0x20]+ stur q19, [x0, #-0x10]+ add x11, x11, #0x10+ cmp x11, #0x100+ b.lt Lmld_rej_uniform_final_copy+ mov x0, x9+ b Lmld_rej_uniform_return++Lmld_rej_uniform_return:+ add sp, sp, #0x440+ .cfi_adjust_cfa_offset -0x440+ ret+ .cfi_endproc++MLD_ASM_FN_SIZE(rej_uniform_aarch64_asm)++#endif /* MLD_ARITH_BACKEND_AARCH64 && !MLD_CONFIG_MULTILEVEL_NO_SHARED */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
@@ -0,0 +1,170 @@+/*+ * Copyright (c) The mldsa-native project authors+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*yaml+ Name: rej_uniform_eta2_aarch64_asm+ Description: AArch64 rejection sampling of eta=2 secret coefficients+ Signature: uint64_t mld_rej_uniform_eta2_aarch64_asm(int32_t r[256], const uint8_t *buf, unsigned buflen, const uint8_t table[4096])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON]+ x0:+ type: buffer+ size_bytes: 1024+ permissions: write-only+ c_parameter: int32_t r[256]+ description: Output buffer (256 x int32_t)+ x1:+ type: buffer+ size_bytes: x2+ permissions: read-only+ c_parameter: const uint8_t *buf+ description: Input buffer+ x2:+ type: scalar+ c_parameter: unsigned buflen+ description: Length of input buffer (must be a multiple of 8)+ test_with: 136 # MLD_AARCH64_REJ_UNIFORM_ETA2_BUFLEN+ x3:+ type: buffer+ size_bytes: 4096+ permissions: read-only+ c_parameter: const uint8_t table[4096]+ description: Lookup table (4096 x uint8_t)+*/++#include "../../../common.h"+#if defined(MLD_ARITH_BACKEND_AARCH64) && \+ !defined(MLD_CONFIG_NO_KEYPAIR_API) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \+ (defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLDSA_ETA == 2)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/aarch64_opt/src/mldsa_rej_uniform_eta2_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(rej_uniform_eta2_aarch64_asm)+MLD_ASM_FN_SYMBOL(rej_uniform_eta2_aarch64_asm)++ .cfi_startproc+ sub sp, sp, #0x240+ .cfi_adjust_cfa_offset 0x240+ mov x7, #0x1 // =1+ movk x7, #0x2, lsl #16+ movk x7, #0x4, lsl #32+ movk x7, #0x8, lsl #48+ mov v31.d[0], x7+ mov x7, #0x10 // =16+ movk x7, #0x20, lsl #16+ movk x7, #0x40, lsl #32+ movk x7, #0x80, lsl #48+ mov v31.d[1], x7+ movi v30.8h, #0xf+ mov x8, sp+ mov x7, x8+ mov x11, #0x0 // =0+ eor v16.16b, v16.16b, v16.16b++Lmld_rej_uniform_eta2_initial_zero:+ str q16, [x7], #0x40+ stur q16, [x7, #-0x30]+ stur q16, [x7, #-0x20]+ stur q16, [x7, #-0x10]+ add x11, x11, #0x20+ cmp x11, #0x100+ b.lt Lmld_rej_uniform_eta2_initial_zero+ mov x7, x8+ mov x9, #0x0 // =0+ mov x4, #0x100 // =256++Lmld_rej_uniform_eta2_loop8:+ cmp x9, x4+ b.hs Lmld_rej_uniform_eta2_memory_copy+ sub x2, x2, #0x8+ ld1 { v0.8b }, [x1], #8+ movi v26.8b, #0xf+ and v27.8b, v0.8b, v26.8b+ ushr v28.8b, v0.8b, #0x4+ zip1 v26.8b, v27.8b, v28.8b+ zip2 v29.8b, v27.8b, v28.8b+ ushll v16.8h, v26.8b, #0x0+ ushll v17.8h, v29.8b, #0x0+ cmhi v4.8h, v30.8h, v16.8h+ cmhi v5.8h, v30.8h, v17.8h+ and v4.16b, v4.16b, v31.16b+ and v5.16b, v5.16b, v31.16b+ uaddlv s20, v4.8h+ uaddlv s21, v5.8h+ fmov w12, s20+ fmov w13, s21+ ldr q24, [x3, x12, lsl #4]+ ldr q25, [x3, x13, lsl #4]+ cnt v4.16b, v4.16b+ cnt v5.16b, v5.16b+ uaddlv s20, v4.8h+ uaddlv s21, v5.8h+ fmov w12, s20+ fmov w13, s21+ tbl v16.16b, { v16.16b }, v24.16b+ tbl v17.16b, { v17.16b }, v25.16b+ st1 { v16.8h }, [x7]+ add x7, x7, x12, lsl #1+ st1 { v17.8h }, [x7]+ add x7, x7, x13, lsl #1+ add x12, x12, x13+ add x9, x9, x12+ cmp x2, #0x8+ b.hs Lmld_rej_uniform_eta2_loop8++Lmld_rej_uniform_eta2_memory_copy:+ cmp x9, x4+ csel x9, x9, x4, lo+ mov w7, #0x199a // =6554+ dup v26.8h, w7+ movi v27.8h, #0x5+ movi v7.8h, #0x2+ mov x11, #0x0 // =0+ mov x7, x8++Lmld_rej_uniform_eta2_final_copy:+ ldr q16, [x7], #0x20+ ldur q18, [x7, #-0x10]+ sqdmulh v28.8h, v16.8h, v26.8h+ mls v16.8h, v28.8h, v27.8h+ sqdmulh v28.8h, v18.8h, v26.8h+ mls v18.8h, v28.8h, v27.8h+ sub v16.8h, v7.8h, v16.8h+ sub v18.8h, v7.8h, v18.8h+ sshll2 v17.4s, v16.8h, #0x0+ sshll v16.4s, v16.4h, #0x0+ sshll2 v19.4s, v18.8h, #0x0+ sshll v18.4s, v18.4h, #0x0+ str q16, [x0], #0x40+ stur q17, [x0, #-0x30]+ stur q18, [x0, #-0x20]+ stur q19, [x0, #-0x10]+ add x11, x11, #0x10+ cmp x11, #0x100+ b.lt Lmld_rej_uniform_eta2_final_copy+ mov x0, x9+ add sp, sp, #0x240+ .cfi_adjust_cfa_offset -0x240+ ret+ .cfi_endproc++MLD_ASM_FN_SIZE(rej_uniform_eta2_aarch64_asm)++#endif /* MLD_ARITH_BACKEND_AARCH64 && !MLD_CONFIG_NO_KEYPAIR_API && \+ !MLD_CONFIG_MULTILEVEL_NO_SHARED && \+ (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLDSA_ETA == 2) */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
@@ -0,0 +1,163 @@+/*+ * Copyright (c) The mldsa-native project authors+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*yaml+ Name: rej_uniform_eta4_aarch64_asm+ Description: AArch64 rejection sampling of eta=4 secret coefficients+ Signature: uint64_t mld_rej_uniform_eta4_aarch64_asm(int32_t r[256], const uint8_t *buf, unsigned buflen, const uint8_t table[4096])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON]+ x0:+ type: buffer+ size_bytes: 1024+ permissions: write-only+ c_parameter: int32_t r[256]+ description: Output buffer (256 x int32_t)+ x1:+ type: buffer+ size_bytes: x2+ permissions: read-only+ c_parameter: const uint8_t *buf+ description: Input buffer+ x2:+ type: scalar+ c_parameter: unsigned buflen+ description: Length of input buffer (must be a multiple of 8)+ test_with: 272 # MLD_AARCH64_REJ_UNIFORM_ETA4_BUFLEN+ x3:+ type: buffer+ size_bytes: 4096+ permissions: read-only+ c_parameter: const uint8_t table[4096]+ description: Lookup table (4096 x uint8_t)+*/++#include "../../../common.h"+#if defined(MLD_ARITH_BACKEND_AARCH64) && \+ !defined(MLD_CONFIG_NO_KEYPAIR_API) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \+ (defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLDSA_ETA == 4)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/aarch64_opt/src/mldsa_rej_uniform_eta4_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(rej_uniform_eta4_aarch64_asm)+MLD_ASM_FN_SYMBOL(rej_uniform_eta4_aarch64_asm)++ .cfi_startproc+ sub sp, sp, #0x240+ .cfi_adjust_cfa_offset 0x240+ mov x7, #0x1 // =1+ movk x7, #0x2, lsl #16+ movk x7, #0x4, lsl #32+ movk x7, #0x8, lsl #48+ mov v31.d[0], x7+ mov x7, #0x10 // =16+ movk x7, #0x20, lsl #16+ movk x7, #0x40, lsl #32+ movk x7, #0x80, lsl #48+ mov v31.d[1], x7+ movi v30.8h, #0x9+ movi v7.8h, #0x4+ mov x8, sp+ mov x7, x8+ mov x11, #0x0 // =0+ eor v16.16b, v16.16b, v16.16b++Lmld_rej_uniform_eta4_initial_zero:+ str q16, [x7], #0x40+ stur q16, [x7, #-0x30]+ stur q16, [x7, #-0x20]+ stur q16, [x7, #-0x10]+ add x11, x11, #0x20+ cmp x11, #0x100+ b.lt Lmld_rej_uniform_eta4_initial_zero+ mov x7, x8+ mov x9, #0x0 // =0+ mov x4, #0x100 // =256++Lmld_rej_uniform_eta4_loop8:+ cmp x9, x4+ b.hs Lmld_rej_uniform_eta4_memory_copy+ sub x2, x2, #0x8+ ld1 { v0.8b }, [x1], #8+ movi v26.8b, #0xf+ and v27.8b, v0.8b, v26.8b+ ushr v28.8b, v0.8b, #0x4+ zip1 v26.8b, v27.8b, v28.8b+ zip2 v29.8b, v27.8b, v28.8b+ ushll v16.8h, v26.8b, #0x0+ ushll v17.8h, v29.8b, #0x0+ cmhi v4.8h, v30.8h, v16.8h+ cmhi v5.8h, v30.8h, v17.8h+ and v4.16b, v4.16b, v31.16b+ and v5.16b, v5.16b, v31.16b+ uaddlv s20, v4.8h+ uaddlv s21, v5.8h+ fmov w12, s20+ fmov w13, s21+ ldr q24, [x3, x12, lsl #4]+ ldr q25, [x3, x13, lsl #4]+ cnt v4.16b, v4.16b+ cnt v5.16b, v5.16b+ uaddlv s20, v4.8h+ uaddlv s21, v5.8h+ fmov w12, s20+ fmov w13, s21+ tbl v16.16b, { v16.16b }, v24.16b+ tbl v17.16b, { v17.16b }, v25.16b+ st1 { v16.8h }, [x7]+ add x7, x7, x12, lsl #1+ st1 { v17.8h }, [x7]+ add x7, x7, x13, lsl #1+ add x12, x12, x13+ add x9, x9, x12+ cmp x2, #0x8+ b.hs Lmld_rej_uniform_eta4_loop8++Lmld_rej_uniform_eta4_memory_copy:+ cmp x9, x4+ csel x9, x9, x4, lo+ mov x11, #0x0 // =0+ mov x7, x8++Lmld_rej_uniform_eta4_final_copy:+ ldr q16, [x7], #0x20+ ldur q18, [x7, #-0x10]+ sub v16.8h, v7.8h, v16.8h+ sub v18.8h, v7.8h, v18.8h+ sshll2 v17.4s, v16.8h, #0x0+ sshll v16.4s, v16.4h, #0x0+ sshll2 v19.4s, v18.8h, #0x0+ sshll v18.4s, v18.4h, #0x0+ str q16, [x0], #0x40+ stur q17, [x0, #-0x30]+ stur q18, [x0, #-0x20]+ stur q19, [x0, #-0x10]+ add x11, x11, #0x10+ cmp x11, #0x100+ b.lt Lmld_rej_uniform_eta4_final_copy+ mov x0, x9+ add sp, sp, #0x240+ .cfi_adjust_cfa_offset -0x240+ ret+ .cfi_endproc++MLD_ASM_FN_SIZE(rej_uniform_eta4_aarch64_asm)++#endif /* MLD_ARITH_BACKEND_AARCH64 && !MLD_CONFIG_NO_KEYPAIR_API && \+ !MLD_CONFIG_MULTILEVEL_NO_SHARED && \+ (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLDSA_ETA == 4) */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
@@ -0,0 +1,52 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*+ * WARNING: This file is auto-generated from scripts/autogen+ * in the mldsa-native repository.+ * Do not modify it directly.+ */++#include "../../../common.h"++#if defined(MLD_ARITH_BACKEND_AARCH64) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++#include "arith_native_aarch64.h"++#if !defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API)+/* Table of indices used for tbl instructions in polyz_unpack_{17,19}.+ * See autogen for details. */++#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLD_CONFIG_PARAMETER_SET == 44+MLD_ALIGN MLD_INTERNAL_DATA_DEFINITION const uint8_t+ mld_polyz_unpack_17_indices[64] = {+ 0, 1, 2, 255, 2, 3, 4, 255, 4, 5, 6, 255, 6, 7, 8, 255,+ 9, 10, 11, 255, 11, 12, 13, 255, 13, 14, 15, 255, 15, 16, 17, 255,+ 2, 3, 4, 255, 4, 5, 6, 255, 6, 7, 8, 255, 8, 9, 10, 255,+ 11, 12, 13, 255, 13, 14, 15, 255, 15, 28, 29, 255, 29, 30, 31, 255,+};+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 44 \+ */++#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \+ (MLD_CONFIG_PARAMETER_SET == 65 || MLD_CONFIG_PARAMETER_SET == 87)+MLD_ALIGN MLD_INTERNAL_DATA_DEFINITION const uint8_t+ mld_polyz_unpack_19_indices[64] = {+ 0, 1, 2, 255, 2, 3, 4, 255, 5, 6, 7, 255, 7, 8, 9, 255,+ 10, 11, 12, 255, 12, 13, 14, 255, 15, 16, 17, 255, 17, 18, 19, 255,+ 4, 5, 6, 255, 6, 7, 8, 255, 9, 10, 11, 255, 11, 12, 13, 255,+ 14, 15, 24, 255, 24, 25, 26, 255, 27, 28, 29, 255, 29, 30, 31, 255,+};+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 65 \+ || MLD_CONFIG_PARAMETER_SET == 87 */++#endif /* !MLD_CONFIG_NO_SIGN_API || !MLD_CONFIG_NO_VERIFY_API */++#else /* MLD_ARITH_BACKEND_AARCH64 && !MLD_CONFIG_MULTILEVEL_NO_SHARED */++MLD_EMPTY_CU(aarch64_polyz_unpack_table)++#endif /* !(MLD_ARITH_BACKEND_AARCH64 && !MLD_CONFIG_MULTILEVEL_NO_SHARED) */
@@ -0,0 +1,547 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*+ * WARNING: This file is auto-generated from scripts/autogen+ * in the mldsa-native repository.+ * Do not modify it directly.+ */++#include "../../../common.h"++#if defined(MLD_ARITH_BACKEND_AARCH64) && \+ !defined(MLD_CONFIG_NO_KEYPAIR_API) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++#include "arith_native_aarch64.h"++/*+ * Lookup table used by 16-bit rejection sampling (rej_eta).+ * Adapted from ML-KEM for ML-DSA eta rejection sampling.+ * See autogen for details.+ */+MLD_ALIGN MLD_INTERNAL_DATA_DEFINITION const uint8_t+ mld_rej_uniform_eta_table[4096] = {+ 255, 255, 255, 255, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 0 */,+ 0, 1, 255, 255, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 1 */,+ 2, 3, 255, 255, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 2 */,+ 0, 1, 2, 3, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 3 */,+ 4, 5, 255, 255, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 4 */,+ 0, 1, 4, 5, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 5 */,+ 2, 3, 4, 5, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 6 */,+ 0, 1, 2, 3, 4, 5, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 7 */,+ 6, 7, 255, 255, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 8 */,+ 0, 1, 6, 7, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 9 */,+ 2, 3, 6, 7, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 10 */,+ 0, 1, 2, 3, 6, 7, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 11 */,+ 4, 5, 6, 7, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 12 */,+ 0, 1, 4, 5, 6, 7, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 13 */,+ 2, 3, 4, 5, 6, 7, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 14 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 15 */,+ 8, 9, 255, 255, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 16 */,+ 0, 1, 8, 9, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 17 */,+ 2, 3, 8, 9, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 18 */,+ 0, 1, 2, 3, 8, 9, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 19 */,+ 4, 5, 8, 9, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 20 */,+ 0, 1, 4, 5, 8, 9, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 21 */,+ 2, 3, 4, 5, 8, 9, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 22 */,+ 0, 1, 2, 3, 4, 5, 8, 9,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 23 */,+ 6, 7, 8, 9, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 24 */,+ 0, 1, 6, 7, 8, 9, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 25 */,+ 2, 3, 6, 7, 8, 9, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 26 */,+ 0, 1, 2, 3, 6, 7, 8, 9,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 27 */,+ 4, 5, 6, 7, 8, 9, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 28 */,+ 0, 1, 4, 5, 6, 7, 8, 9,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 29 */,+ 2, 3, 4, 5, 6, 7, 8, 9,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 30 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 8, 9, 255, 255, 255, 255, 255, 255 /* 31 */,+ 10, 11, 255, 255, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 32 */,+ 0, 1, 10, 11, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 33 */,+ 2, 3, 10, 11, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 34 */,+ 0, 1, 2, 3, 10, 11, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 35 */,+ 4, 5, 10, 11, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 36 */,+ 0, 1, 4, 5, 10, 11, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 37 */,+ 2, 3, 4, 5, 10, 11, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 38 */,+ 0, 1, 2, 3, 4, 5, 10, 11,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 39 */,+ 6, 7, 10, 11, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 40 */,+ 0, 1, 6, 7, 10, 11, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 41 */,+ 2, 3, 6, 7, 10, 11, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 42 */,+ 0, 1, 2, 3, 6, 7, 10, 11,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 43 */,+ 4, 5, 6, 7, 10, 11, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 44 */,+ 0, 1, 4, 5, 6, 7, 10, 11,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 45 */,+ 2, 3, 4, 5, 6, 7, 10, 11,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 46 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 10, 11, 255, 255, 255, 255, 255, 255 /* 47 */,+ 8, 9, 10, 11, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 48 */,+ 0, 1, 8, 9, 10, 11, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 49 */,+ 2, 3, 8, 9, 10, 11, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 50 */,+ 0, 1, 2, 3, 8, 9, 10, 11,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 51 */,+ 4, 5, 8, 9, 10, 11, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 52 */,+ 0, 1, 4, 5, 8, 9, 10, 11,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 53 */,+ 2, 3, 4, 5, 8, 9, 10, 11,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 54 */,+ 0, 1, 2, 3, 4, 5, 8, 9,+ 10, 11, 255, 255, 255, 255, 255, 255 /* 55 */,+ 6, 7, 8, 9, 10, 11, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 56 */,+ 0, 1, 6, 7, 8, 9, 10, 11,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 57 */,+ 2, 3, 6, 7, 8, 9, 10, 11,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 58 */,+ 0, 1, 2, 3, 6, 7, 8, 9,+ 10, 11, 255, 255, 255, 255, 255, 255 /* 59 */,+ 4, 5, 6, 7, 8, 9, 10, 11,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 60 */,+ 0, 1, 4, 5, 6, 7, 8, 9,+ 10, 11, 255, 255, 255, 255, 255, 255 /* 61 */,+ 2, 3, 4, 5, 6, 7, 8, 9,+ 10, 11, 255, 255, 255, 255, 255, 255 /* 62 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 8, 9, 10, 11, 255, 255, 255, 255 /* 63 */,+ 12, 13, 255, 255, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 64 */,+ 0, 1, 12, 13, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 65 */,+ 2, 3, 12, 13, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 66 */,+ 0, 1, 2, 3, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 67 */,+ 4, 5, 12, 13, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 68 */,+ 0, 1, 4, 5, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 69 */,+ 2, 3, 4, 5, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 70 */,+ 0, 1, 2, 3, 4, 5, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 71 */,+ 6, 7, 12, 13, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 72 */,+ 0, 1, 6, 7, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 73 */,+ 2, 3, 6, 7, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 74 */,+ 0, 1, 2, 3, 6, 7, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 75 */,+ 4, 5, 6, 7, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 76 */,+ 0, 1, 4, 5, 6, 7, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 77 */,+ 2, 3, 4, 5, 6, 7, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 78 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 79 */,+ 8, 9, 12, 13, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 80 */,+ 0, 1, 8, 9, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 81 */,+ 2, 3, 8, 9, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 82 */,+ 0, 1, 2, 3, 8, 9, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 83 */,+ 4, 5, 8, 9, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 84 */,+ 0, 1, 4, 5, 8, 9, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 85 */,+ 2, 3, 4, 5, 8, 9, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 86 */,+ 0, 1, 2, 3, 4, 5, 8, 9,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 87 */,+ 6, 7, 8, 9, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 88 */,+ 0, 1, 6, 7, 8, 9, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 89 */,+ 2, 3, 6, 7, 8, 9, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 90 */,+ 0, 1, 2, 3, 6, 7, 8, 9,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 91 */,+ 4, 5, 6, 7, 8, 9, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 92 */,+ 0, 1, 4, 5, 6, 7, 8, 9,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 93 */,+ 2, 3, 4, 5, 6, 7, 8, 9,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 94 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 8, 9, 12, 13, 255, 255, 255, 255 /* 95 */,+ 10, 11, 12, 13, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 96 */,+ 0, 1, 10, 11, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 97 */,+ 2, 3, 10, 11, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 98 */,+ 0, 1, 2, 3, 10, 11, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 99 */,+ 4, 5, 10, 11, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 100 */,+ 0, 1, 4, 5, 10, 11, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 101 */,+ 2, 3, 4, 5, 10, 11, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 102 */,+ 0, 1, 2, 3, 4, 5, 10, 11,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 103 */,+ 6, 7, 10, 11, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 104 */,+ 0, 1, 6, 7, 10, 11, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 105 */,+ 2, 3, 6, 7, 10, 11, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 106 */,+ 0, 1, 2, 3, 6, 7, 10, 11,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 107 */,+ 4, 5, 6, 7, 10, 11, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 108 */,+ 0, 1, 4, 5, 6, 7, 10, 11,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 109 */,+ 2, 3, 4, 5, 6, 7, 10, 11,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 110 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 10, 11, 12, 13, 255, 255, 255, 255 /* 111 */,+ 8, 9, 10, 11, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 112 */,+ 0, 1, 8, 9, 10, 11, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 113 */,+ 2, 3, 8, 9, 10, 11, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 114 */,+ 0, 1, 2, 3, 8, 9, 10, 11,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 115 */,+ 4, 5, 8, 9, 10, 11, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 116 */,+ 0, 1, 4, 5, 8, 9, 10, 11,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 117 */,+ 2, 3, 4, 5, 8, 9, 10, 11,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 118 */,+ 0, 1, 2, 3, 4, 5, 8, 9,+ 10, 11, 12, 13, 255, 255, 255, 255 /* 119 */,+ 6, 7, 8, 9, 10, 11, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 120 */,+ 0, 1, 6, 7, 8, 9, 10, 11,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 121 */,+ 2, 3, 6, 7, 8, 9, 10, 11,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 122 */,+ 0, 1, 2, 3, 6, 7, 8, 9,+ 10, 11, 12, 13, 255, 255, 255, 255 /* 123 */,+ 4, 5, 6, 7, 8, 9, 10, 11,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 124 */,+ 0, 1, 4, 5, 6, 7, 8, 9,+ 10, 11, 12, 13, 255, 255, 255, 255 /* 125 */,+ 2, 3, 4, 5, 6, 7, 8, 9,+ 10, 11, 12, 13, 255, 255, 255, 255 /* 126 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 8, 9, 10, 11, 12, 13, 255, 255 /* 127 */,+ 14, 15, 255, 255, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 128 */,+ 0, 1, 14, 15, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 129 */,+ 2, 3, 14, 15, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 130 */,+ 0, 1, 2, 3, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 131 */,+ 4, 5, 14, 15, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 132 */,+ 0, 1, 4, 5, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 133 */,+ 2, 3, 4, 5, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 134 */,+ 0, 1, 2, 3, 4, 5, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 135 */,+ 6, 7, 14, 15, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 136 */,+ 0, 1, 6, 7, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 137 */,+ 2, 3, 6, 7, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 138 */,+ 0, 1, 2, 3, 6, 7, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 139 */,+ 4, 5, 6, 7, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 140 */,+ 0, 1, 4, 5, 6, 7, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 141 */,+ 2, 3, 4, 5, 6, 7, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 142 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 143 */,+ 8, 9, 14, 15, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 144 */,+ 0, 1, 8, 9, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 145 */,+ 2, 3, 8, 9, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 146 */,+ 0, 1, 2, 3, 8, 9, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 147 */,+ 4, 5, 8, 9, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 148 */,+ 0, 1, 4, 5, 8, 9, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 149 */,+ 2, 3, 4, 5, 8, 9, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 150 */,+ 0, 1, 2, 3, 4, 5, 8, 9,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 151 */,+ 6, 7, 8, 9, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 152 */,+ 0, 1, 6, 7, 8, 9, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 153 */,+ 2, 3, 6, 7, 8, 9, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 154 */,+ 0, 1, 2, 3, 6, 7, 8, 9,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 155 */,+ 4, 5, 6, 7, 8, 9, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 156 */,+ 0, 1, 4, 5, 6, 7, 8, 9,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 157 */,+ 2, 3, 4, 5, 6, 7, 8, 9,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 158 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 8, 9, 14, 15, 255, 255, 255, 255 /* 159 */,+ 10, 11, 14, 15, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 160 */,+ 0, 1, 10, 11, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 161 */,+ 2, 3, 10, 11, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 162 */,+ 0, 1, 2, 3, 10, 11, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 163 */,+ 4, 5, 10, 11, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 164 */,+ 0, 1, 4, 5, 10, 11, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 165 */,+ 2, 3, 4, 5, 10, 11, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 166 */,+ 0, 1, 2, 3, 4, 5, 10, 11,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 167 */,+ 6, 7, 10, 11, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 168 */,+ 0, 1, 6, 7, 10, 11, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 169 */,+ 2, 3, 6, 7, 10, 11, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 170 */,+ 0, 1, 2, 3, 6, 7, 10, 11,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 171 */,+ 4, 5, 6, 7, 10, 11, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 172 */,+ 0, 1, 4, 5, 6, 7, 10, 11,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 173 */,+ 2, 3, 4, 5, 6, 7, 10, 11,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 174 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 10, 11, 14, 15, 255, 255, 255, 255 /* 175 */,+ 8, 9, 10, 11, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 176 */,+ 0, 1, 8, 9, 10, 11, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 177 */,+ 2, 3, 8, 9, 10, 11, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 178 */,+ 0, 1, 2, 3, 8, 9, 10, 11,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 179 */,+ 4, 5, 8, 9, 10, 11, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 180 */,+ 0, 1, 4, 5, 8, 9, 10, 11,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 181 */,+ 2, 3, 4, 5, 8, 9, 10, 11,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 182 */,+ 0, 1, 2, 3, 4, 5, 8, 9,+ 10, 11, 14, 15, 255, 255, 255, 255 /* 183 */,+ 6, 7, 8, 9, 10, 11, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 184 */,+ 0, 1, 6, 7, 8, 9, 10, 11,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 185 */,+ 2, 3, 6, 7, 8, 9, 10, 11,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 186 */,+ 0, 1, 2, 3, 6, 7, 8, 9,+ 10, 11, 14, 15, 255, 255, 255, 255 /* 187 */,+ 4, 5, 6, 7, 8, 9, 10, 11,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 188 */,+ 0, 1, 4, 5, 6, 7, 8, 9,+ 10, 11, 14, 15, 255, 255, 255, 255 /* 189 */,+ 2, 3, 4, 5, 6, 7, 8, 9,+ 10, 11, 14, 15, 255, 255, 255, 255 /* 190 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 8, 9, 10, 11, 14, 15, 255, 255 /* 191 */,+ 12, 13, 14, 15, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 192 */,+ 0, 1, 12, 13, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 193 */,+ 2, 3, 12, 13, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 194 */,+ 0, 1, 2, 3, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 195 */,+ 4, 5, 12, 13, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 196 */,+ 0, 1, 4, 5, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 197 */,+ 2, 3, 4, 5, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 198 */,+ 0, 1, 2, 3, 4, 5, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 199 */,+ 6, 7, 12, 13, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 200 */,+ 0, 1, 6, 7, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 201 */,+ 2, 3, 6, 7, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 202 */,+ 0, 1, 2, 3, 6, 7, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 203 */,+ 4, 5, 6, 7, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 204 */,+ 0, 1, 4, 5, 6, 7, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 205 */,+ 2, 3, 4, 5, 6, 7, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 206 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 207 */,+ 8, 9, 12, 13, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 208 */,+ 0, 1, 8, 9, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 209 */,+ 2, 3, 8, 9, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 210 */,+ 0, 1, 2, 3, 8, 9, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 211 */,+ 4, 5, 8, 9, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 212 */,+ 0, 1, 4, 5, 8, 9, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 213 */,+ 2, 3, 4, 5, 8, 9, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 214 */,+ 0, 1, 2, 3, 4, 5, 8, 9,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 215 */,+ 6, 7, 8, 9, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 216 */,+ 0, 1, 6, 7, 8, 9, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 217 */,+ 2, 3, 6, 7, 8, 9, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 218 */,+ 0, 1, 2, 3, 6, 7, 8, 9,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 219 */,+ 4, 5, 6, 7, 8, 9, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 220 */,+ 0, 1, 4, 5, 6, 7, 8, 9,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 221 */,+ 2, 3, 4, 5, 6, 7, 8, 9,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 222 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 8, 9, 12, 13, 14, 15, 255, 255 /* 223 */,+ 10, 11, 12, 13, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 224 */,+ 0, 1, 10, 11, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 225 */,+ 2, 3, 10, 11, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 226 */,+ 0, 1, 2, 3, 10, 11, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 227 */,+ 4, 5, 10, 11, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 228 */,+ 0, 1, 4, 5, 10, 11, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 229 */,+ 2, 3, 4, 5, 10, 11, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 230 */,+ 0, 1, 2, 3, 4, 5, 10, 11,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 231 */,+ 6, 7, 10, 11, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 232 */,+ 0, 1, 6, 7, 10, 11, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 233 */,+ 2, 3, 6, 7, 10, 11, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 234 */,+ 0, 1, 2, 3, 6, 7, 10, 11,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 235 */,+ 4, 5, 6, 7, 10, 11, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 236 */,+ 0, 1, 4, 5, 6, 7, 10, 11,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 237 */,+ 2, 3, 4, 5, 6, 7, 10, 11,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 238 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 10, 11, 12, 13, 14, 15, 255, 255 /* 239 */,+ 8, 9, 10, 11, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 240 */,+ 0, 1, 8, 9, 10, 11, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 241 */,+ 2, 3, 8, 9, 10, 11, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 242 */,+ 0, 1, 2, 3, 8, 9, 10, 11,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 243 */,+ 4, 5, 8, 9, 10, 11, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 244 */,+ 0, 1, 4, 5, 8, 9, 10, 11,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 245 */,+ 2, 3, 4, 5, 8, 9, 10, 11,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 246 */,+ 0, 1, 2, 3, 4, 5, 8, 9,+ 10, 11, 12, 13, 14, 15, 255, 255 /* 247 */,+ 6, 7, 8, 9, 10, 11, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 248 */,+ 0, 1, 6, 7, 8, 9, 10, 11,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 249 */,+ 2, 3, 6, 7, 8, 9, 10, 11,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 250 */,+ 0, 1, 2, 3, 6, 7, 8, 9,+ 10, 11, 12, 13, 14, 15, 255, 255 /* 251 */,+ 4, 5, 6, 7, 8, 9, 10, 11,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 252 */,+ 0, 1, 4, 5, 6, 7, 8, 9,+ 10, 11, 12, 13, 14, 15, 255, 255 /* 253 */,+ 2, 3, 4, 5, 6, 7, 8, 9,+ 10, 11, 12, 13, 14, 15, 255, 255 /* 254 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 8, 9, 10, 11, 12, 13, 14, 15 /* 255 */,+};++#else /* MLD_ARITH_BACKEND_AARCH64 && !MLD_CONFIG_NO_KEYPAIR_API && \+ !MLD_CONFIG_MULTILEVEL_NO_SHARED */++MLD_EMPTY_CU(aarch64_rej_uniform_eta_table)++#endif /* !(MLD_ARITH_BACKEND_AARCH64 && !MLD_CONFIG_NO_KEYPAIR_API && \+ !MLD_CONFIG_MULTILEVEL_NO_SHARED) */
@@ -0,0 +1,63 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*+ * WARNING: This file is auto-generated from scripts/autogen+ * in the mldsa-native repository.+ * Do not modify it directly.+ */++#include "../../../common.h"++#if defined(MLD_ARITH_BACKEND_AARCH64) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++#include "arith_native_aarch64.h"++/*+ * Lookup table used by rejection sampling of the public matrix.+ * See autogen for details.+ */+MLD_ALIGN MLD_INTERNAL_DATA_DEFINITION const uint8_t+ mld_rej_uniform_table[256] = {+ 255, 255, 255, 255, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 0 */,+ 0, 1, 2, 3, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 1 */,+ 4, 5, 6, 7, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 2 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 3 */,+ 8, 9, 10, 11, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 4 */,+ 0, 1, 2, 3, 8, 9, 10, 11,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 5 */,+ 4, 5, 6, 7, 8, 9, 10, 11,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 6 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 8, 9, 10, 11, 255, 255, 255, 255 /* 7 */,+ 12, 13, 14, 15, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 8 */,+ 0, 1, 2, 3, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 9 */,+ 4, 5, 6, 7, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 10 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 11 */,+ 8, 9, 10, 11, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 12 */,+ 0, 1, 2, 3, 8, 9, 10, 11,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 13 */,+ 4, 5, 6, 7, 8, 9, 10, 11,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 14 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 8, 9, 10, 11, 12, 13, 14, 15 /* 15 */,+};++#else /* MLD_ARITH_BACKEND_AARCH64 && !MLD_CONFIG_MULTILEVEL_NO_SHARED */++MLD_EMPTY_CU(aarch64_rej_uniform_table)++#endif /* !(MLD_ARITH_BACKEND_AARCH64 && !MLD_CONFIG_MULTILEVEL_NO_SHARED) */
@@ -0,0 +1,617 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#ifndef MLD_NATIVE_API_H+#define MLD_NATIVE_API_H+/*+ * Native arithmetic interface+ *+ * This header is primarily for documentation purposes.+ * It should not be included by backend implementations.+ *+ * To ensure consistency with backends, the header will be+ * included automatically after inclusion of the active+ * backend, to ensure consistency of function signatures,+ * and run sanity checks.+ */++#include "../cbmc.h"+#include "../common.h"++/* Backends must return MLD_NATIVE_FUNC_SUCCESS upon success. */+#define MLD_NATIVE_FUNC_SUCCESS (0)+/* Backends may return MLD_NATIVE_FUNC_FALLBACK to signal to the frontend that+ * the target/parameters are unsupported; typically, this would be because of+ * dependencies on CPU features not detected on the host CPU. In this case,+ * the frontend falls back to the default C implementation.+ *+ * IMPORTANT: Backend implementations must ensure that the decision of whether+ * to fallback (return MLD_NATIVE_FUNC_FALLBACK) or not must never depend on+ * the input data itself. Fallback decisions may only depend on system+ * capabilities (e.g., CPU features) and, where present, length information.+ * This requirement applies to all backend functions to maintain constant-time+ * properties.+ */+#define MLD_NATIVE_FUNC_FALLBACK (-1)++/* Absolute exclusive upper bound for the output of fqmul.+ *+ * NOTE: This is the same bound as in poly.h and has to be kept+ * in sync. */+#define MLD_FQMUL_BOUND ((5 * MLDSA_Q + 3) / 4)++/* Bound on absolute value of coefficients after NTT.+ *+ * NOTE: This is the same bound as in poly.h and has to be kept+ * in sync. */+#define MLD_NTT_BOUND (9 * MLD_FQMUL_BOUND)++/* Absolute exclusive upper bound for the output of the inverse NTT+ *+ * NOTE: This is the same bound as in poly.h and has to be kept+ * in sync. */+#define MLD_INTT_BOUND MLDSA_Q++/* Absolute bound for range of mld_reduce32()+ *+ * NOTE: This is the same bound as in reduce.h and has to be kept+ * in sync. */+/* check-magic: 6283009 == (MLD_REDUCE32_DOMAIN_MAX - 255 * MLDSA_Q + 1) */+#define MLD_REDUCE32_RANGE_MAX 6283009+/*+ * This is the C<->native interface allowing for the drop-in of+ * native code for performance-critical arithmetic components of ML-DSA.+ *+ * A _backend_ is a specific implementation of (part of) this interface.+ *+ * To add a function to a backend, define MLD_USE_NATIVE_XXX and+ * implement `static inline xxx(...)` in the profile header.+ */++/*+ * Those functions are meant to be trivial wrappers around the chosen native+ * implementation. The are static inline to avoid unnecessary calls.+ * The macro before each declaration controls whether a native+ * implementation is present.+ */++#if defined(MLD_USE_NATIVE_NTT)+/**+ * Computes negacyclic number-theoretic transform (NTT) of a polynomial+ * in place.+ *+ * The input polynomial is assumed to be in normal order. The output+ * polynomial is in bitreversed order.+ *+ * @param[in,out] p Pointer to in/output polynomial.+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_ntt_native(int32_t p[MLDSA_N])+__contract__(+ requires(memory_no_alias(p, sizeof(int32_t) * MLDSA_N))+ requires(array_abs_bound(p, 0, MLDSA_N, MLDSA_Q))+ assigns(memory_slice(p, sizeof(int32_t) * MLDSA_N))+ ensures(return_value == MLD_NATIVE_FUNC_FALLBACK || return_value == MLD_NATIVE_FUNC_SUCCESS)+ ensures((return_value == MLD_NATIVE_FUNC_SUCCESS) ==> array_abs_bound(p, 0, MLDSA_N, MLD_NTT_BOUND))+ ensures((return_value == MLD_NATIVE_FUNC_FALLBACK) ==> array_abs_bound(p, 0, MLDSA_N, MLDSA_Q))+ ensures((return_value == MLD_NATIVE_FUNC_FALLBACK) ==> array_unchanged(p, MLDSA_N))+);+#endif /* MLD_USE_NATIVE_NTT */+++#if defined(MLD_USE_NATIVE_NTT_CUSTOM_ORDER)+/*+ * This must only be set if NTT and INTT have native implementations+ * that are adapted to the custom order.+ */+#if !defined(MLD_USE_NATIVE_NTT) || !defined(MLD_USE_NATIVE_INTT)+#error \+ "Invalid native profile: MLD_USE_NATIVE_NTT_CUSTOM_ORDER can only be \+set if there are native implementations for NTT and INTT."+#endif++/**+ * When MLD_USE_NATIVE_NTT_CUSTOM_ORDER is defined, convert a polynomial in+ * NTT domain from bitreversed order to the custom order output by the native+ * NTT.+ *+ * This must only be defined if there is native code for both the NTT and+ * INTT.+ *+ * @param[in,out] p Pointer to in/output polynomial.+ */+static MLD_INLINE void mld_poly_permute_bitrev_to_custom(int32_t p[MLDSA_N])+__contract__(+ /* We don't specify that this should be a permutation, but only+ * that it does not change the bound established at the end of+ * mld_polyvec_matrix_expand.+ */+ requires(memory_no_alias(p, sizeof(int32_t) * MLDSA_N))+ requires(array_bound(p, 0, MLDSA_N, 0, MLDSA_Q))+ assigns(memory_slice(p, sizeof(int32_t) * MLDSA_N))+ ensures(array_bound(p, 0, MLDSA_N, 0, MLDSA_Q)));+#endif /* MLD_USE_NATIVE_NTT_CUSTOM_ORDER */+++#if defined(MLD_USE_NATIVE_INTT)+/**+ * Computes inverse of negacyclic number-theoretic transform (NTT) of a+ * polynomial in place.+ *+ * The input polynomial is in bitreversed order. The output polynomial is+ * assumed to be in normal order.+ *+ * @param[in,out] p Pointer to in/output polynomial.+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_intt_native(int32_t p[MLDSA_N])+__contract__(+ requires(memory_no_alias(p, sizeof(int32_t) * MLDSA_N))+ requires(array_abs_bound(p, 0, MLDSA_N, MLDSA_Q))+ assigns(memory_slice(p, sizeof(int32_t) * MLDSA_N))+ ensures(return_value == MLD_NATIVE_FUNC_FALLBACK || return_value == MLD_NATIVE_FUNC_SUCCESS)+ ensures((return_value == MLD_NATIVE_FUNC_SUCCESS) ==> array_abs_bound(p, 0, MLDSA_N, MLD_INTT_BOUND))+ ensures((return_value == MLD_NATIVE_FUNC_FALLBACK) ==> array_abs_bound(p, 0, MLDSA_N, MLDSA_Q))+ ensures((return_value == MLD_NATIVE_FUNC_FALLBACK) ==> array_unchanged(p, MLDSA_N))+);+#endif /* MLD_USE_NATIVE_INTT */++#if defined(MLD_USE_NATIVE_REJ_UNIFORM)+/**+ * Run rejection sampling on uniform random bytes to generate uniform random+ * integers in [0, MLDSA_Q-1].+ *+ * @param[out] r Pointer to output buffer.+ * @param len Requested number of 32-bit integers (uniform mod+ * MLDSA_Q).+ * @param[in] buf Pointer to input buffer (assumed to be uniform random+ * bytes).+ * @param buflen Length of input buffer in bytes.+ *+ * @return - MLD_NATIVE_FUNC_FALLBACK if the native implementation does not+ * support the input lengths.+ * - Otherwise, the non-negative number of sampled 32-bit integers+ * (at most len).+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_rej_uniform_native(int32_t *r, unsigned len,+ const uint8_t *buf,+ unsigned buflen)+__contract__(+ requires(len <= MLDSA_N)+ requires(buflen <= ( 5 * 168) && buflen % 3 == 0)+ requires(memory_no_alias(r, sizeof(int32_t) * len))+ requires(memory_no_alias(buf, buflen))+ assigns(memory_slice(r, sizeof(int32_t) * len))+ ensures(return_value == MLD_NATIVE_FUNC_FALLBACK || (0 <= return_value && return_value <= len))+ ensures((return_value != MLD_NATIVE_FUNC_FALLBACK) ==> array_bound(r, 0, (unsigned) return_value, 0, MLDSA_Q))+);+#endif /* MLD_USE_NATIVE_REJ_UNIFORM */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API)+#if defined(MLD_USE_NATIVE_REJ_UNIFORM_ETA2)+#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLDSA_ETA == 2+/**+ * Run rejection sampling on uniform random bytes to generate uniform random+ * integers in [-2, +2].+ *+ * @param[out] r Pointer to output buffer.+ * @param len Requested number of 32-bit integers (uniform in+ * [-2, +2]).+ * @param[in] buf Pointer to input buffer (assumed to be uniform random+ * bytes).+ * @param buflen Length of input buffer in bytes.+ *+ * @return - MLD_NATIVE_FUNC_FALLBACK if the native implementation does not+ * support the input lengths.+ * - Otherwise, the non-negative number of sampled 32-bit integers+ * (at most len).+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_rej_uniform_eta2_native(int32_t *r, unsigned len,+ const uint8_t *buf,+ unsigned buflen)+__contract__(+ requires(len <= MLDSA_N)+ requires(buflen <= (2 * 136))+ requires(memory_no_alias(r, sizeof(int32_t) * len))+ requires(memory_no_alias(buf, buflen))+ assigns(memory_slice(r, sizeof(int32_t) * len))+ ensures(return_value == MLD_NATIVE_FUNC_FALLBACK || (0 <= return_value && return_value <= len))+ /* check-magic: 3 == 2 + 1 (decl gated on MLDSA_ETA == 2) */+ ensures((return_value != MLD_NATIVE_FUNC_FALLBACK) ==> (array_abs_bound(r, 0, return_value, 3)))+);+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLDSA_ETA == 2 */+#endif /* MLD_USE_NATIVE_REJ_UNIFORM_ETA2 */++#if defined(MLD_USE_NATIVE_REJ_UNIFORM_ETA4)+#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLDSA_ETA == 4+/**+ * Run rejection sampling on uniform random bytes to generate uniform random+ * integers in [-4, +4].+ *+ * @param[out] r Pointer to output buffer.+ * @param len Requested number of 32-bit integers (uniform in+ * [-4, +4]).+ * @param[in] buf Pointer to input buffer (assumed to be uniform random+ * bytes).+ * @param buflen Length of input buffer in bytes.+ *+ * @return - MLD_NATIVE_FUNC_FALLBACK if the native implementation does not+ * support the input lengths.+ * - Otherwise, the non-negative number of sampled 32-bit integers+ * (at most len).+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_rej_uniform_eta4_native(int32_t *r, unsigned len,+ const uint8_t *buf,+ unsigned buflen)+__contract__(+ requires(len <= MLDSA_N)+ requires(buflen <= (2 * 136))+ requires(memory_no_alias(r, sizeof(int32_t) * len))+ requires(memory_no_alias(buf, buflen))+ assigns(memory_slice(r, sizeof(int32_t) * len))+ ensures(return_value == MLD_NATIVE_FUNC_FALLBACK || (0 <= return_value && return_value <= len))+ /* check-magic: 5 == 4 + 1 (decl gated on MLDSA_ETA == 4) */+ ensures((return_value != MLD_NATIVE_FUNC_FALLBACK) ==> (array_abs_bound(r, 0, return_value, 5)))+);+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLDSA_ETA == 4 */+#endif /* MLD_USE_NATIVE_REJ_UNIFORM_ETA4 */+#endif /* !MLD_CONFIG_NO_KEYPAIR_API */++#if !defined(MLD_CONFIG_NO_SIGN_API)+#if defined(MLD_USE_NATIVE_POLY_DECOMPOSE_32)+#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \+ (MLD_CONFIG_PARAMETER_SET == 65 || MLD_CONFIG_PARAMETER_SET == 87)+/**+ * Native implementation of poly_decompose for GAMMA2 = (MLDSA_Q-1)/32.+ *+ * For all coefficients c of the input polynomial, compute high and low bits+ * c0, c1 such c mod MLDSA_Q = c1*(2*GAMMA2) + c0 with+ * -(2*GAMMA2)/2 < c0 <= (2*GAMMA2)/2 except c1 = (MLDSA_Q-1)/(2*GAMMA2) where+ * we set c1 = 0 and -(2*GAMMA2)/2 <= c0 = c mod MLDSA_Q - MLDSA_Q < 0.+ * Assumes coefficients to be standard representatives.+ *+ * @param[out] a1 Output polynomial with coefficients c1.+ * @param[in,out] a0 Input/output polynomial. Output has coefficients c0.+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_poly_decompose_32_native(int32_t *a1, int32_t *a0)+__contract__(+ requires(memory_no_alias(a1, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(a0, sizeof(int32_t) * MLDSA_N))+ requires(array_bound(a0, 0, MLDSA_N, 0, MLDSA_Q))+ assigns(memory_slice(a1, sizeof(int32_t) * MLDSA_N))+ assigns(memory_slice(a0, sizeof(int32_t) * MLDSA_N))+ ensures(return_value == MLD_NATIVE_FUNC_FALLBACK || return_value == MLD_NATIVE_FUNC_SUCCESS)+ ensures((return_value == MLD_NATIVE_FUNC_SUCCESS) ==> array_bound(a1, 0, MLDSA_N, 0, (MLDSA_Q-1)/(2*MLDSA_GAMMA2)))+ ensures((return_value == MLD_NATIVE_FUNC_SUCCESS) ==> array_abs_bound(a0, 0, MLDSA_N, MLDSA_GAMMA2+1))+ ensures((return_value == MLD_NATIVE_FUNC_FALLBACK) ==> array_bound(a0, 0, MLDSA_N, 0, MLDSA_Q))+ ensures((return_value == MLD_NATIVE_FUNC_FALLBACK) ==> array_unchanged(a0, MLDSA_N))+);+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 65 \+ || MLD_CONFIG_PARAMETER_SET == 87 */+#endif /* MLD_USE_NATIVE_POLY_DECOMPOSE_32 */++#if defined(MLD_USE_NATIVE_POLY_DECOMPOSE_88)+#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLD_CONFIG_PARAMETER_SET == 44+/**+ * Native implementation of poly_decompose for GAMMA2 = (MLDSA_Q-1)/88.+ *+ * For all coefficients c of the input polynomial, compute high and low bits+ * c0, c1 such c mod MLDSA_Q = c1*(2*GAMMA2) + c0 with+ * -(2*GAMMA2)/2 < c0 <= (2*GAMMA2)/2 except c1 = (MLDSA_Q-1)/(2*GAMMA2) where+ * we set c1 = 0 and -(2*GAMMA2)/2 <= c0 = c mod MLDSA_Q - MLDSA_Q < 0.+ * Assumes coefficients to be standard representatives.+ *+ * @param[out] a1 Output polynomial with coefficients c1.+ * @param[in,out] a0 Input/output polynomial. Output has coefficients c0.+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_poly_decompose_88_native(int32_t *a1, int32_t *a0)+__contract__(+ requires(memory_no_alias(a1, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(a0, sizeof(int32_t) * MLDSA_N))+ requires(array_bound(a0, 0, MLDSA_N, 0, MLDSA_Q))+ assigns(memory_slice(a1, sizeof(int32_t) * MLDSA_N))+ assigns(memory_slice(a0, sizeof(int32_t) * MLDSA_N))+ ensures(return_value == MLD_NATIVE_FUNC_FALLBACK || return_value == MLD_NATIVE_FUNC_SUCCESS)+ ensures((return_value == MLD_NATIVE_FUNC_SUCCESS) ==> array_bound(a1, 0, MLDSA_N, 0, (MLDSA_Q-1)/(2*MLDSA_GAMMA2)))+ ensures((return_value == MLD_NATIVE_FUNC_SUCCESS) ==> array_abs_bound(a0, 0, MLDSA_N, MLDSA_GAMMA2+1))+ ensures((return_value == MLD_NATIVE_FUNC_FALLBACK) ==> array_bound(a0, 0, MLDSA_N, 0, MLDSA_Q))+ ensures((return_value == MLD_NATIVE_FUNC_FALLBACK) ==> array_unchanged(a0, MLDSA_N))+);+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 44 \+ */+#endif /* MLD_USE_NATIVE_POLY_DECOMPOSE_88 */+#endif /* !MLD_CONFIG_NO_SIGN_API */++#if defined(MLD_USE_NATIVE_POLY_CADDQ)+/**+ * For all coefficients of in/out polynomial add Q if coefficient is negative.+ *+ * @param[in,out] a Pointer to input/output polynomial.+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_poly_caddq_native(int32_t a[MLDSA_N])+__contract__(+ requires(memory_no_alias(a, sizeof(int32_t) * MLDSA_N))+ requires(array_abs_bound(a, 0, MLDSA_N, MLDSA_Q))+ assigns(memory_slice(a, sizeof(int32_t) * MLDSA_N))+ ensures(return_value == MLD_NATIVE_FUNC_FALLBACK || return_value == MLD_NATIVE_FUNC_SUCCESS)+ ensures((return_value == MLD_NATIVE_FUNC_SUCCESS) ==> array_bound(a, 0, MLDSA_N, 0, MLDSA_Q))+ ensures((return_value == MLD_NATIVE_FUNC_FALLBACK) ==> array_abs_bound(a, 0, MLDSA_N, MLDSA_Q))+ ensures((return_value == MLD_NATIVE_FUNC_FALLBACK) ==> array_unchanged(a, MLDSA_N))+);+#endif /* MLD_USE_NATIVE_POLY_CADDQ */++#if !defined(MLD_CONFIG_NO_VERIFY_API)+#if defined(MLD_USE_NATIVE_POLY_USE_HINT_32)+#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \+ (MLD_CONFIG_PARAMETER_SET == 65 || MLD_CONFIG_PARAMETER_SET == 87)+/**+ * Native implementation of poly_use_hint for GAMMA2 = (MLDSA_Q-1)/32.+ *+ * Use hint h to correct the high bits of a in-place.+ *+ * @param[in,out] a Input/output polynomial.+ * @param[in] h Hint polynomial.+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_poly_use_hint_32_native(int32_t *a, const int32_t *h)+__contract__(+ requires(memory_no_alias(a, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(h, sizeof(int32_t) * MLDSA_N))+ requires(array_bound(a, 0, MLDSA_N, 0, MLDSA_Q))+ requires(array_bound(h, 0, MLDSA_N, 0, 2))+ assigns(memory_slice(a, sizeof(int32_t) * MLDSA_N))+ ensures(return_value == MLD_NATIVE_FUNC_FALLBACK || return_value == MLD_NATIVE_FUNC_SUCCESS)+ ensures((return_value == MLD_NATIVE_FUNC_SUCCESS) ==> array_bound(a, 0, MLDSA_N, 0, (MLDSA_Q-1)/(2*MLDSA_GAMMA2)))+ ensures((return_value == MLD_NATIVE_FUNC_FALLBACK) ==> array_unchanged(a, MLDSA_N))+);+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 65 \+ || MLD_CONFIG_PARAMETER_SET == 87 */+#endif /* MLD_USE_NATIVE_POLY_USE_HINT_32 */++#if defined(MLD_USE_NATIVE_POLY_USE_HINT_88)+#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLD_CONFIG_PARAMETER_SET == 44+/**+ * Native implementation of poly_use_hint for GAMMA2 = (MLDSA_Q-1)/88.+ *+ * Use hint h to correct the high bits of a in-place.+ *+ * @param[in,out] a Input/output polynomial.+ * @param[in] h Hint polynomial.+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_poly_use_hint_88_native(int32_t *a, const int32_t *h)+__contract__(+ requires(memory_no_alias(a, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(h, sizeof(int32_t) * MLDSA_N))+ requires(array_bound(a, 0, MLDSA_N, 0, MLDSA_Q))+ requires(array_bound(h, 0, MLDSA_N, 0, 2))+ assigns(memory_slice(a, sizeof(int32_t) * MLDSA_N))+ ensures(return_value == MLD_NATIVE_FUNC_FALLBACK || return_value == MLD_NATIVE_FUNC_SUCCESS)+ ensures((return_value == MLD_NATIVE_FUNC_SUCCESS) ==> array_bound(a, 0, MLDSA_N, 0, (MLDSA_Q-1)/(2*MLDSA_GAMMA2)))+ ensures((return_value == MLD_NATIVE_FUNC_FALLBACK) ==> array_unchanged(a, MLDSA_N))+);+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 44 \+ */+#endif /* MLD_USE_NATIVE_POLY_USE_HINT_88 */+#endif /* !MLD_CONFIG_NO_VERIFY_API */++#if defined(MLD_USE_NATIVE_POLY_CHKNORM)+/**+ * Check infinity norm of polynomial against given bound. Assumes input+ * coefficients were reduced by mld_reduce32().+ *+ * @param[in] a Pointer to polynomial.+ * @param B Norm bound, which must be in the range+ * 0 .. MLDSA_Q - MLD_REDUCE32_RANGE_MAX inclusive.+ *+ * @return - MLD_NATIVE_FUNC_FALLBACK if the target CPU cannot support a+ * native implementation of this function.+ * - MLD_NATIVE_FUNC_SUCCESS if the infinity norm is strictly smaller+ * than B.+ * - 1 otherwise.+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_poly_chknorm_native(const int32_t *a, int32_t B)+__contract__(+ requires(memory_no_alias(a, sizeof(int32_t) * MLDSA_N))+ requires(0 <= B && B <= MLDSA_Q - MLD_REDUCE32_RANGE_MAX)+ requires(array_bound(a, 0, MLDSA_N, -MLD_REDUCE32_RANGE_MAX, MLD_REDUCE32_RANGE_MAX))+ ensures(return_value == MLD_NATIVE_FUNC_FALLBACK || return_value == 0 ||+ return_value == 1)+ ensures((return_value != MLD_NATIVE_FUNC_FALLBACK) ==>+ ((return_value == 0) == array_abs_bound(a, 0, MLDSA_N, B)))+);+#endif /* MLD_USE_NATIVE_POLY_CHKNORM */++#if !defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API)+#if defined(MLD_USE_NATIVE_POLYZ_UNPACK_17)+#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLD_CONFIG_PARAMETER_SET == 44+/**+ * Native implementation of polyz_unpack for GAMMA1 = 2^17.+ *+ * Unpack polynomial z with coefficients in+ * [-(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1].+ *+ * @param[out] r Pointer to output polynomial.+ * @param[in] a Byte array with bit-packed polynomial.+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_polyz_unpack_17_native(int32_t *r, const uint8_t *a)+__contract__(+ requires(memory_no_alias(r, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(a, MLDSA_POLYZ_PACKEDBYTES))+ assigns(memory_slice(r, sizeof(int32_t) * MLDSA_N))+ ensures(return_value == MLD_NATIVE_FUNC_FALLBACK || return_value == MLD_NATIVE_FUNC_SUCCESS)+ ensures((return_value == MLD_NATIVE_FUNC_SUCCESS) ==> array_bound(r, 0, MLDSA_N, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1))+ ensures((return_value == MLD_NATIVE_FUNC_FALLBACK) ==> array_unchanged(r, MLDSA_N))+);+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 44 \+ */+#endif /* MLD_USE_NATIVE_POLYZ_UNPACK_17 */++#if defined(MLD_USE_NATIVE_POLYZ_UNPACK_19)+#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \+ (MLD_CONFIG_PARAMETER_SET == 65 || MLD_CONFIG_PARAMETER_SET == 87)+/**+ * Native implementation of polyz_unpack for GAMMA1 = 2^19.+ *+ * Unpack polynomial z with coefficients in+ * [-(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1].+ *+ * @param[out] r Pointer to output polynomial.+ * @param[in] a Byte array with bit-packed polynomial.+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_polyz_unpack_19_native(int32_t *r, const uint8_t *a)+__contract__(+ requires(memory_no_alias(r, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(a, MLDSA_POLYZ_PACKEDBYTES))+ assigns(memory_slice(r, sizeof(int32_t) * MLDSA_N))+ ensures(return_value == MLD_NATIVE_FUNC_FALLBACK || return_value == MLD_NATIVE_FUNC_SUCCESS)+ ensures((return_value == MLD_NATIVE_FUNC_SUCCESS) ==> array_bound(r, 0, MLDSA_N, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1))+ ensures((return_value == MLD_NATIVE_FUNC_FALLBACK) ==> array_unchanged(r, MLDSA_N))+);+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 65 \+ || MLD_CONFIG_PARAMETER_SET == 87 */+#endif /* MLD_USE_NATIVE_POLYZ_UNPACK_19 */+#endif /* !MLD_CONFIG_NO_SIGN_API || !MLD_CONFIG_NO_VERIFY_API */++#if !defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API) || \+ defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)+#if defined(MLD_USE_NATIVE_POINTWISE_MONTGOMERY)+/**+ * Pointwise multiplication of polynomials in NTT domain with Montgomery+ * reduction. Destructive in the first argument.+ *+ * Computes a[i] = a[i] * b[i] * R^(-1) mod MLDSA_Q for all i, where R = 2^32.+ *+ * @param[in,out] a First input/output polynomial.+ * @param[in] b Second input polynomial.+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_poly_pointwise_montgomery_native(+ int32_t a[MLDSA_N], const int32_t b[MLDSA_N])+__contract__(+ requires(memory_no_alias(a, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(b, sizeof(int32_t) * MLDSA_N))+ requires(array_abs_bound(a, 0, MLDSA_N, MLD_NTT_BOUND))+ requires(array_abs_bound(b, 0, MLDSA_N, MLD_NTT_BOUND))+ assigns(memory_slice(a, sizeof(int32_t) * MLDSA_N))+ ensures(return_value == MLD_NATIVE_FUNC_FALLBACK || return_value == MLD_NATIVE_FUNC_SUCCESS)+ ensures((return_value == MLD_NATIVE_FUNC_SUCCESS) ==> array_abs_bound(a, 0, MLDSA_N, MLDSA_Q))+ ensures((return_value == MLD_NATIVE_FUNC_FALLBACK) ==> array_abs_bound(a, 0, MLDSA_N, MLD_NTT_BOUND))+ ensures((return_value == MLD_NATIVE_FUNC_FALLBACK) ==> array_abs_bound(b, 0, MLDSA_N, MLD_NTT_BOUND))+);+#endif /* MLD_USE_NATIVE_POINTWISE_MONTGOMERY */+#endif /* !MLD_CONFIG_NO_SIGN_API || !MLD_CONFIG_NO_VERIFY_API || \+ MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST */++#if defined(MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L4)+#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLDSA_L == 4+/**+ * Native implementation of polyvecl_pointwise_acc_montgomery for MLDSA_L = 4.+ *+ * Pointwise multiply vectors of polynomials of length MLDSA_L, multiply+ * resulting vector by 2^{-32} and add (accumulate) polynomials in it.+ * Input/output vectors are in NTT domain representation.+ *+ * @param[out] w Output polynomial.+ * @param[in] u First input vector.+ * @param[in] v Second input vector.+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_polyvecl_pointwise_acc_montgomery_l4_native(+ int32_t w[MLDSA_N], const int32_t u[4][MLDSA_N],+ const int32_t v[4][MLDSA_N])+__contract__(+ requires(memory_no_alias(w, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(u, sizeof(int32_t) * 4 * MLDSA_N))+ requires(memory_no_alias(v, sizeof(int32_t) * 4 * MLDSA_N))+ requires(forall(l0, 0, 4,+ array_bound(u[l0], 0, MLDSA_N, 0, MLDSA_Q)))+ requires(forall(l1, 0, 4,+ array_abs_bound(v[l1], 0, MLDSA_N, MLD_NTT_BOUND)))+ assigns(memory_slice(w, sizeof(int32_t) * MLDSA_N))+ ensures(return_value == MLD_NATIVE_FUNC_FALLBACK || return_value == MLD_NATIVE_FUNC_SUCCESS)+ ensures((return_value == MLD_NATIVE_FUNC_SUCCESS) ==> array_abs_bound(w, 0, MLDSA_N, MLDSA_Q))+ ensures((return_value == MLD_NATIVE_FUNC_FALLBACK) ==> array_unchanged(w, MLDSA_N))+);+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLDSA_L == 4 */+#endif /* MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L4 */++#if defined(MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L5)+#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLDSA_L == 5+/**+ * Native implementation of polyvecl_pointwise_acc_montgomery for MLDSA_L = 5.+ *+ * Pointwise multiply vectors of polynomials of length MLDSA_L, multiply+ * resulting vector by 2^{-32} and add (accumulate) polynomials in it.+ * Input/output vectors are in NTT domain representation.+ *+ * @param[out] w Output polynomial.+ * @param[in] u First input vector.+ * @param[in] v Second input vector.+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_polyvecl_pointwise_acc_montgomery_l5_native(+ int32_t w[MLDSA_N], const int32_t u[5][MLDSA_N],+ const int32_t v[5][MLDSA_N])+__contract__(+ requires(memory_no_alias(w, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(u, sizeof(int32_t) * 5 * MLDSA_N))+ requires(memory_no_alias(v, sizeof(int32_t) * 5 * MLDSA_N))+ requires(forall(l0, 0, 5,+ array_bound(u[l0], 0, MLDSA_N, 0, MLDSA_Q)))+ requires(forall(l1, 0, 5,+ array_abs_bound(v[l1], 0, MLDSA_N, MLD_NTT_BOUND)))+ assigns(memory_slice(w, sizeof(int32_t) * MLDSA_N))+ ensures(return_value == MLD_NATIVE_FUNC_FALLBACK || return_value == MLD_NATIVE_FUNC_SUCCESS)+ ensures((return_value == MLD_NATIVE_FUNC_SUCCESS) ==> array_abs_bound(w, 0, MLDSA_N, MLDSA_Q))+ ensures((return_value == MLD_NATIVE_FUNC_FALLBACK) ==> array_unchanged(w, MLDSA_N))+);+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLDSA_L == 5 */+#endif /* MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L5 */++#if defined(MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L7)+#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLDSA_L == 7+/**+ * Native implementation of polyvecl_pointwise_acc_montgomery for MLDSA_L = 7.+ *+ * Pointwise multiply vectors of polynomials of length MLDSA_L, multiply+ * resulting vector by 2^{-32} and add (accumulate) polynomials in it.+ * Input/output vectors are in NTT domain representation.+ *+ * @param[out] w Output polynomial.+ * @param[in] u First input vector.+ * @param[in] v Second input vector.+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_polyvecl_pointwise_acc_montgomery_l7_native(+ int32_t w[MLDSA_N], const int32_t u[7][MLDSA_N],+ const int32_t v[7][MLDSA_N])+__contract__(+ requires(memory_no_alias(w, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(u, sizeof(int32_t) * 7 * MLDSA_N))+ requires(memory_no_alias(v, sizeof(int32_t) * 7 * MLDSA_N))+ requires(forall(l0, 0, 7,+ array_bound(u[l0], 0, MLDSA_N, 0, MLDSA_Q)))+ requires(forall(l1, 0, 7,+ array_abs_bound(v[l1], 0, MLDSA_N, MLD_NTT_BOUND)))+ assigns(memory_slice(w, sizeof(int32_t) * MLDSA_N))+ ensures(return_value == MLD_NATIVE_FUNC_FALLBACK || return_value == MLD_NATIVE_FUNC_SUCCESS)+ ensures((return_value == MLD_NATIVE_FUNC_SUCCESS) ==> array_abs_bound(w, 0, MLDSA_N, MLDSA_Q))+ ensures((return_value == MLD_NATIVE_FUNC_FALLBACK) ==> array_unchanged(w, MLDSA_N))+);+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLDSA_L == 7 */+#endif /* MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L7 */++#endif /* !MLD_NATIVE_API_H */
@@ -0,0 +1,24 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#ifndef MLD_NATIVE_META_H+#define MLD_NATIVE_META_H++/*+ * Default arithmetic backend+ */+#include "../sys.h"++#ifdef MLD_SYS_AARCH64_NEON+#include "aarch64/meta.h"+#endif++/* The x86_64 backend requires toolchain support for the SysV ABI */+#if defined(MLD_SYS_X86_64_AVX2) && defined(MLD_SYSV_ABI_SUPPORTED)+#include "x86_64/meta.h"+#endif++#endif /* !MLD_NATIVE_META_H */
@@ -0,0 +1,323 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#ifndef MLD_NATIVE_X86_64_META_H+#define MLD_NATIVE_X86_64_META_H++/* Identifier for this backend so that source and assembly files+ * in the build can be appropriately guarded. */+#define MLD_ARITH_BACKEND_X86_64_DEFAULT++#define MLD_USE_NATIVE_NTT_CUSTOM_ORDER+#define MLD_USE_NATIVE_NTT+#define MLD_USE_NATIVE_INTT+#define MLD_USE_NATIVE_REJ_UNIFORM+#define MLD_USE_NATIVE_REJ_UNIFORM_ETA2+#define MLD_USE_NATIVE_REJ_UNIFORM_ETA4+#define MLD_USE_NATIVE_POLY_DECOMPOSE_32+#define MLD_USE_NATIVE_POLY_DECOMPOSE_88+#define MLD_USE_NATIVE_POLY_CADDQ+#define MLD_USE_NATIVE_POLY_USE_HINT_32+#define MLD_USE_NATIVE_POLY_USE_HINT_88+#define MLD_USE_NATIVE_POLY_CHKNORM+#define MLD_USE_NATIVE_POLYZ_UNPACK_17+#define MLD_USE_NATIVE_POLYZ_UNPACK_19+#define MLD_USE_NATIVE_POINTWISE_MONTGOMERY+#define MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L4+#define MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L5+#define MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L7++#if !defined(__ASSEMBLER__)+#include <string.h>+#include "../../common.h"+#include "../api.h"+#include "src/arith_native_x86_64.h"++static MLD_INLINE void mld_poly_permute_bitrev_to_custom(int32_t data[MLDSA_N])+{+ if (mld_sys_check_capability(MLD_SYS_CAP_X86_64_AVX2))+ {+ mld_nttunpack_avx2_asm(data);+ }+}++MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_ntt_native(int32_t data[MLDSA_N])+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_X86_64_AVX2))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }++ mld_ntt_avx2_asm(data, mld_qdata);+ return MLD_NATIVE_FUNC_SUCCESS;+}++MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_intt_native(int32_t data[MLDSA_N])+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_X86_64_AVX2))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }+ mld_invntt_avx2_asm(data, mld_qdata);+ return MLD_NATIVE_FUNC_SUCCESS;+}++MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_rej_uniform_native(int32_t *r, unsigned len,+ const uint8_t *buf,+ unsigned buflen)+{+ /* AVX2 implementation assumes specific buffer lengths */+ if (!mld_sys_check_capability(MLD_SYS_CAP_X86_64_AVX2) || len != MLDSA_N ||+ buflen != MLD_AVX2_REJ_UNIFORM_BUFLEN)+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }++ /* Safety: outlen is at most MLDSA_N and, hence, this cast is safe. */+ return (int)mld_rej_uniform_avx2_asm(r, buf, mld_rej_uniform_table);+}++#if !defined(MLD_CONFIG_NO_KEYPAIR_API)+#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLDSA_ETA == 2+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_rej_uniform_eta2_native(int32_t *r, unsigned len,+ const uint8_t *buf,+ unsigned buflen)+{+ unsigned int outlen;+ /* AVX2 implementation assumes specific buffer lengths */+ if (!mld_sys_check_capability(MLD_SYS_CAP_X86_64_AVX2) || len != MLDSA_N ||+ buflen != MLD_AVX2_REJ_UNIFORM_ETA2_BUFLEN)+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }++ /* Constant time: Inputs and outputs to this function are secret.+ * It is safe to leak which coefficients are accepted/rejected.+ * The assembly implementation must not leak any other information about the+ * accepted coefficients. Constant-time testing cannot cover this, and we+ * hence have to manually verify the assembly.+ * We declassify prior the input data and mark the outputs as secret.+ */+ MLD_CT_TESTING_DECLASSIFY(buf, buflen);+ outlen = mld_rej_uniform_eta2_avx2_asm(r, buf, mld_rej_uniform_table);+ MLD_CT_TESTING_SECRET(r, sizeof(int32_t) * outlen);+ /* Safety: outlen is at most MLDSA_N and, hence, this cast is safe. */+ return (int)outlen;+}+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLDSA_ETA == 2 */++#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLDSA_ETA == 4+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_rej_uniform_eta4_native(int32_t *r, unsigned len,+ const uint8_t *buf,+ unsigned buflen)+{+ unsigned int outlen;+ /* AVX2 implementation assumes specific buffer lengths */+ if (!mld_sys_check_capability(MLD_SYS_CAP_X86_64_AVX2) || len != MLDSA_N ||+ buflen != MLD_AVX2_REJ_UNIFORM_ETA4_BUFLEN)+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }++ /* Constant time: Inputs and outputs to this function are secret.+ * It is safe to leak which coefficients are accepted/rejected.+ * The assembly implementation must not leak any other information about the+ * accepted coefficients. Constant-time testing cannot cover this, and we+ * hence have to manually verify the assembly.+ * We declassify prior the input data and mark the outputs as secret.+ */+ MLD_CT_TESTING_DECLASSIFY(buf, buflen);+ outlen = mld_rej_uniform_eta4_avx2_asm(r, buf, mld_rej_uniform_table);+ MLD_CT_TESTING_SECRET(r, sizeof(int32_t) * outlen);+ /* Safety: outlen is at most MLDSA_N and, hence, this cast is safe. */+ return (int)outlen;+}+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLDSA_ETA == 4 */+#endif /* !MLD_CONFIG_NO_KEYPAIR_API */++#if !defined(MLD_CONFIG_NO_SIGN_API)+#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \+ (MLD_CONFIG_PARAMETER_SET == 65 || MLD_CONFIG_PARAMETER_SET == 87)+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_poly_decompose_32_native(int32_t *a1, int32_t *a0)+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_X86_64_AVX2))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }+ mld_poly_decompose_32_avx2_asm(a1, a0);+ return MLD_NATIVE_FUNC_SUCCESS;+}+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 65 \+ || MLD_CONFIG_PARAMETER_SET == 87 */++#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLD_CONFIG_PARAMETER_SET == 44+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_poly_decompose_88_native(int32_t *a1, int32_t *a0)+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_X86_64_AVX2))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }+ mld_poly_decompose_88_avx2_asm(a1, a0);+ return MLD_NATIVE_FUNC_SUCCESS;+}+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 44 \+ */+#endif /* !MLD_CONFIG_NO_SIGN_API */++MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_poly_caddq_native(int32_t a[MLDSA_N])+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_X86_64_AVX2))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }+ mld_poly_caddq_avx2_asm(a);+ return MLD_NATIVE_FUNC_SUCCESS;+}++#if !defined(MLD_CONFIG_NO_VERIFY_API)+#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \+ (MLD_CONFIG_PARAMETER_SET == 65 || MLD_CONFIG_PARAMETER_SET == 87)+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_poly_use_hint_32_native(int32_t *a, const int32_t *h)+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_X86_64_AVX2))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }+ mld_poly_use_hint_32_avx2_asm(a, h);+ return MLD_NATIVE_FUNC_SUCCESS;+}+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 65 \+ || MLD_CONFIG_PARAMETER_SET == 87 */++#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLD_CONFIG_PARAMETER_SET == 44+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_poly_use_hint_88_native(int32_t *a, const int32_t *h)+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_X86_64_AVX2))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }+ mld_poly_use_hint_88_avx2_asm(a, h);+ return MLD_NATIVE_FUNC_SUCCESS;+}+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 44 \+ */+#endif /* !MLD_CONFIG_NO_VERIFY_API */++MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_poly_chknorm_native(const int32_t *a, int32_t B)+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_X86_64_AVX2))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }+ return mld_poly_chknorm_avx2_asm(a, B);+}++#if !defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API)+#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLD_CONFIG_PARAMETER_SET == 44+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_polyz_unpack_17_native(int32_t *r, const uint8_t *a)+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_X86_64_AVX2))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }+ mld_polyz_unpack_17_avx2_asm(r, a);+ return MLD_NATIVE_FUNC_SUCCESS;+}+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 44 \+ */++#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \+ (MLD_CONFIG_PARAMETER_SET == 65 || MLD_CONFIG_PARAMETER_SET == 87)+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_polyz_unpack_19_native(int32_t *r, const uint8_t *a)+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_X86_64_AVX2))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }+ mld_polyz_unpack_19_avx2_asm(r, a);+ return MLD_NATIVE_FUNC_SUCCESS;+}+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 65 \+ || MLD_CONFIG_PARAMETER_SET == 87 */+#endif /* !MLD_CONFIG_NO_SIGN_API || !MLD_CONFIG_NO_VERIFY_API */++#if !defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API) || \+ defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_poly_pointwise_montgomery_native(+ int32_t a[MLDSA_N], const int32_t b[MLDSA_N])+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_X86_64_AVX2))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }+ mld_pointwise_avx2_asm(a, b, mld_qdata);+ return MLD_NATIVE_FUNC_SUCCESS;+}+#endif /* !MLD_CONFIG_NO_SIGN_API || !MLD_CONFIG_NO_VERIFY_API || \+ MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST */++#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLDSA_L == 4+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_polyvecl_pointwise_acc_montgomery_l4_native(+ int32_t w[MLDSA_N], const int32_t u[4][MLDSA_N],+ const int32_t v[4][MLDSA_N])+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_X86_64_AVX2))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }+ mld_pointwise_acc_l4_avx2_asm(w, u, v, mld_qdata);+ return MLD_NATIVE_FUNC_SUCCESS;+}+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLDSA_L == 4 */++#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLDSA_L == 5+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_polyvecl_pointwise_acc_montgomery_l5_native(+ int32_t w[MLDSA_N], const int32_t u[5][MLDSA_N],+ const int32_t v[5][MLDSA_N])+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_X86_64_AVX2))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }+ mld_pointwise_acc_l5_avx2_asm(w, u, v, mld_qdata);+ return MLD_NATIVE_FUNC_SUCCESS;+}+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLDSA_L == 5 */++#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLDSA_L == 7+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_polyvecl_pointwise_acc_montgomery_l7_native(+ int32_t w[MLDSA_N], const int32_t u[7][MLDSA_N],+ const int32_t v[7][MLDSA_N])+{+ if (!mld_sys_check_capability(MLD_SYS_CAP_X86_64_AVX2))+ {+ return MLD_NATIVE_FUNC_FALLBACK;+ }+ mld_pointwise_acc_l7_avx2_asm(w, u, v, mld_qdata);+ return MLD_NATIVE_FUNC_SUCCESS;+}+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLDSA_L == 7 */++#endif /* !__ASSEMBLER__ */++#endif /* !MLD_NATIVE_X86_64_META_H */
@@ -0,0 +1,330 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+#ifndef MLD_NATIVE_X86_64_SRC_ARITH_NATIVE_X86_64_H+#define MLD_NATIVE_X86_64_SRC_ARITH_NATIVE_X86_64_H+#include "../../../common.h"++#include "consts.h"++#define MLD_AVX2_REJ_UNIFORM_BUFLEN \+ (5 * 168) /* REJ_UNIFORM_NBLOCKS * SHAKE128_RATE */+++/*+ * Sampling 256 coefficients mod 15 using rejection sampling from 4 bits.+ * Expected number of required bytes: (256 * (16/15))/2 = 136.5 bytes.+ * We sample 1 block (=136 bytes) of SHAKE256_RATE output initially.+ * Sampling 2 blocks initially results in slightly worse performance.+ */+#define MLD_AVX2_REJ_UNIFORM_ETA2_BUFLEN (1 * 136)++/*+ * Sampling 256 coefficients mod 9 using rejection sampling from 4 bits.+ * Expected number of required bytes: (256 * (16/9))/2 = 227.5 bytes.+ * We sample 2 blocks (=272 bytes) of SHAKE256_RATE output initially.+ */+#define MLD_AVX2_REJ_UNIFORM_ETA4_BUFLEN (2 * 136)++#define mld_rej_uniform_table MLD_NAMESPACE(mld_rej_uniform_table)+MLD_INTERNAL_DATA_DECLARATION const uint8_t mld_rej_uniform_table[256][8];++#define mld_ntt_avx2_asm MLD_NAMESPACE(ntt_avx2_asm)+MLD_SYSV_ABI+void mld_ntt_avx2_asm(int32_t *r, const int32_t *qdata)+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/x86_64/proofs/mldsa_ntt_avx2_asm.ml */+__contract__(+ requires(memory_no_alias(r, sizeof(int32_t) * MLDSA_N))+ /* check-magic: 8380417 == MLDSA_Q */+ requires(array_abs_bound(r, 0, MLDSA_N, 8380417))+ requires(qdata == mld_qdata)+ assigns(memory_slice(r, sizeof(int32_t) * MLDSA_N))+ /* check-magic: off */+ ensures(array_abs_bound(r, 0, MLDSA_N, 42035262))+ /* check-magic: on */+);++#define mld_invntt_avx2_asm MLD_NAMESPACE(invntt_avx2_asm)+MLD_SYSV_ABI+void mld_invntt_avx2_asm(int32_t *r, const int32_t *qdata)+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/x86_64/proofs/mldsa_intt_avx2_asm.ml */+__contract__(+ requires(memory_no_alias(r, sizeof(int32_t) * MLDSA_N))+ requires(array_abs_bound(r, 0, MLDSA_N, 8380417))+ requires(qdata == mld_qdata)+ assigns(memory_slice(r, sizeof(int32_t) * MLDSA_N))+ /* check-magic: off */+ ensures(array_abs_bound(r, 0, MLDSA_N, 6285313))+ /* check-magic: on */+);++#define mld_nttunpack_avx2_asm MLD_NAMESPACE(nttunpack_avx2_asm)+MLD_SYSV_ABI+void mld_nttunpack_avx2_asm(int32_t *r)+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/x86_64/proofs/mldsa_nttunpack_avx2_asm.ml */+__contract__(+ requires(memory_no_alias(r, sizeof(int32_t) * MLDSA_N))+ requires(array_abs_bound(r, 0, MLDSA_N, 8380417))+ assigns(memory_slice(r, sizeof(int32_t) * MLDSA_N))+ /* Output is a permutation of input: every output coefficient+ * is some input coefficient */+ ensures(forall(i, 0, MLDSA_N, exists(j, 0, MLDSA_N,+ r[i] == old(*(int32_t (*)[MLDSA_N])r)[j])))+);++#define mld_rej_uniform_avx2_asm MLD_NAMESPACE(rej_uniform_avx2_asm)+/* This contract must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/x86_64/proofs/mldsa_rej_uniform_avx2_asm.ml */+MLD_MUST_CHECK_RETURN_VALUE MLD_SYSV_ABI+unsigned mld_rej_uniform_avx2_asm(+ int32_t *r, const uint8_t buf[MLD_AVX2_REJ_UNIFORM_BUFLEN],+ const uint8_t table[256][8])+__contract__(+ requires(memory_no_alias(r, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(buf, 840))+ requires(table == mld_rej_uniform_table)+ assigns(memory_slice(r, sizeof(int32_t) * MLDSA_N))+ ensures(return_value <= MLDSA_N)+ ensures(array_bound(r, 0, return_value, 0, MLDSA_Q))+);++#if !defined(MLD_CONFIG_NO_KEYPAIR_API)+#define mld_rej_uniform_eta2_avx2_asm MLD_NAMESPACE(rej_uniform_eta2_avx2_asm)+/* This contract must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/x86_64/proofs/mldsa_rej_uniform_eta2_avx2_asm.ml */+MLD_MUST_CHECK_RETURN_VALUE MLD_SYSV_ABI+unsigned mld_rej_uniform_eta2_avx2_asm(+ int32_t *r, const uint8_t buf[MLD_AVX2_REJ_UNIFORM_ETA2_BUFLEN],+ const uint8_t table[256][8])+__contract__(+ requires(memory_no_alias(r, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(buf, MLD_AVX2_REJ_UNIFORM_ETA2_BUFLEN))+ requires(table == mld_rej_uniform_table)+ assigns(memory_slice(r, sizeof(int32_t) * MLDSA_N))+ ensures(return_value <= MLDSA_N)+ ensures(array_abs_bound(r, 0, return_value, 3))+);++#define mld_rej_uniform_eta4_avx2_asm MLD_NAMESPACE(rej_uniform_eta4_avx2_asm)+/* This contract must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/x86_64/proofs/mldsa_rej_uniform_eta4_avx2_asm.ml */+MLD_MUST_CHECK_RETURN_VALUE MLD_SYSV_ABI+unsigned mld_rej_uniform_eta4_avx2_asm(+ int32_t *r, const uint8_t buf[MLD_AVX2_REJ_UNIFORM_ETA4_BUFLEN],+ const uint8_t table[256][8])+__contract__(+ requires(memory_no_alias(r, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(buf, MLD_AVX2_REJ_UNIFORM_ETA4_BUFLEN))+ requires(table == mld_rej_uniform_table)+ assigns(memory_slice(r, sizeof(int32_t) * MLDSA_N))+ ensures(return_value <= MLDSA_N)+ ensures(array_abs_bound(r, 0, return_value, 5))+);+#endif /* !MLD_CONFIG_NO_KEYPAIR_API */++#if !defined(MLD_CONFIG_NO_SIGN_API)+#define mld_poly_decompose_32_avx2_asm MLD_NAMESPACE(poly_decompose_32_avx2_asm)+MLD_SYSV_ABI+void mld_poly_decompose_32_avx2_asm(int32_t *a1, int32_t *a0)+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/x86_64/proofs/mldsa_poly_decompose_32_avx2_asm.ml */+__contract__(+ requires(memory_no_alias(a1, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(a0, sizeof(int32_t) * MLDSA_N))+ requires(array_bound(a0, 0, MLDSA_N, 0, MLDSA_Q))+ assigns(memory_slice(a1, sizeof(int32_t) * MLDSA_N))+ assigns(memory_slice(a0, sizeof(int32_t) * MLDSA_N))+ /* check-magic: 16 == (MLDSA_Q - 1) / (2 * ((MLDSA_Q - 1) / 32)) */+ ensures(array_bound(a1, 0, MLDSA_N, 0, 16))+ /* check-magic: 261889 == (MLDSA_Q - 1) / 32 + 1 */+ ensures(array_abs_bound(a0, 0, MLDSA_N, 261889))+);++#define mld_poly_decompose_88_avx2_asm MLD_NAMESPACE(poly_decompose_88_avx2_asm)+MLD_SYSV_ABI+void mld_poly_decompose_88_avx2_asm(int32_t *a1, int32_t *a0)+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/x86_64/proofs/mldsa_poly_decompose_88_avx2_asm.ml */+__contract__(+ requires(memory_no_alias(a1, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(a0, sizeof(int32_t) * MLDSA_N))+ requires(array_bound(a0, 0, MLDSA_N, 0, MLDSA_Q))+ assigns(memory_slice(a1, sizeof(int32_t) * MLDSA_N))+ assigns(memory_slice(a0, sizeof(int32_t) * MLDSA_N))+ /* check-magic: 44 == (MLDSA_Q - 1) / (2 * ((MLDSA_Q - 1) / 88)) */+ ensures(array_bound(a1, 0, MLDSA_N, 0, 44))+ /* check-magic: 95233 == (MLDSA_Q - 1) / 88 + 1 */+ ensures(array_abs_bound(a0, 0, MLDSA_N, 95233))+);+#endif /* !MLD_CONFIG_NO_SIGN_API */++#define mld_poly_caddq_avx2_asm MLD_NAMESPACE(poly_caddq_avx2_asm)+MLD_SYSV_ABI+void mld_poly_caddq_avx2_asm(int32_t *r)+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/x86_64/proofs/mldsa_poly_caddq_avx2_asm.ml */+__contract__(+ requires(memory_no_alias(r, sizeof(int32_t) * MLDSA_N))+ requires(array_abs_bound(r, 0, MLDSA_N, MLDSA_Q))+ assigns(memory_slice(r, sizeof(int32_t) * MLDSA_N))+ ensures(array_bound(r, 0, MLDSA_N, 0, MLDSA_Q))+);++#if !defined(MLD_CONFIG_NO_VERIFY_API)+#define mld_poly_use_hint_32_avx2_asm MLD_NAMESPACE(poly_use_hint_32_avx2_asm)+MLD_SYSV_ABI+void mld_poly_use_hint_32_avx2_asm(int32_t *a, const int32_t *h)+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/x86_64/proofs/mldsa_poly_use_hint_32_avx2_asm.ml */+__contract__(+ requires(memory_no_alias(a, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(h, sizeof(int32_t) * MLDSA_N))+ requires(array_bound(a, 0, MLDSA_N, 0, MLDSA_Q))+ requires(array_bound(h, 0, MLDSA_N, 0, 2))+ assigns(memory_slice(a, sizeof(int32_t) * MLDSA_N))+ /* check-magic: 16 == (MLDSA_Q - 1) / (2 * ((MLDSA_Q - 1) / 32)) */+ ensures(array_bound(a, 0, MLDSA_N, 0, 16))+);++#define mld_poly_use_hint_88_avx2_asm MLD_NAMESPACE(poly_use_hint_88_avx2_asm)+MLD_SYSV_ABI+void mld_poly_use_hint_88_avx2_asm(int32_t *a, const int32_t *h)+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/x86_64/proofs/mldsa_poly_use_hint_88_avx2_asm.ml */+__contract__(+ requires(memory_no_alias(a, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(h, sizeof(int32_t) * MLDSA_N))+ requires(array_bound(a, 0, MLDSA_N, 0, MLDSA_Q))+ requires(array_bound(h, 0, MLDSA_N, 0, 2))+ assigns(memory_slice(a, sizeof(int32_t) * MLDSA_N))+ /* check-magic: 44 == (MLDSA_Q - 1) / (2 * ((MLDSA_Q - 1) / 88)) */+ ensures(array_bound(a, 0, MLDSA_N, 0, 44))+);+#endif /* !MLD_CONFIG_NO_VERIFY_API */++#define mld_poly_chknorm_avx2_asm MLD_NAMESPACE(poly_chknorm_avx2_asm)+MLD_MUST_CHECK_RETURN_VALUE MLD_SYSV_ABI+int mld_poly_chknorm_avx2_asm(const int32_t *a, int32_t B)+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/x86_64/proofs/mldsa_poly_chknorm_avx2_asm.ml */+__contract__(+ requires(memory_no_alias(a, sizeof(int32_t) * MLDSA_N))+ /* HOL Light precondition: abs(ival(x i)) < 2^31, i.e., a[i] != INT32_MIN */+ requires(forall(k0, 0, MLDSA_N, a[k0] > INT32_MIN))+ /* HOL Light precondition: 0 <= ival bound (asm computes B-1 internally) */+ requires(B >= 0)+ ensures(return_value == 0 || return_value == 1)+ ensures((return_value == 0) == array_abs_bound(a, 0, MLDSA_N, B))+);++#if !defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API)+#define mld_polyz_unpack_17_avx2_asm MLD_NAMESPACE(polyz_unpack_17_avx2_asm)+MLD_SYSV_ABI+void mld_polyz_unpack_17_avx2_asm(int32_t *r, const uint8_t *a)+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/x86_64/proofs/mldsa_polyz_unpack_17_avx2_asm.ml */+__contract__(+ requires(memory_no_alias(r, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(a, 576))+ assigns(memory_slice(r, sizeof(int32_t) * MLDSA_N))+ ensures(array_bound(r, 0, MLDSA_N, -((1 << 17) - 1), (1 << 17) + 1))+);++#define mld_polyz_unpack_19_avx2_asm MLD_NAMESPACE(polyz_unpack_19_avx2_asm)+MLD_SYSV_ABI+void mld_polyz_unpack_19_avx2_asm(int32_t *r, const uint8_t *a)+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/x86_64/proofs/mldsa_polyz_unpack_19_avx2_asm.ml */+__contract__(+ requires(memory_no_alias(r, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(a, 640))+ assigns(memory_slice(r, sizeof(int32_t) * MLDSA_N))+ ensures(array_bound(r, 0, MLDSA_N, -((1 << 19) - 1), (1 << 19) + 1))+);+#endif /* !MLD_CONFIG_NO_SIGN_API || !MLD_CONFIG_NO_VERIFY_API */++#define mld_pointwise_avx2_asm MLD_NAMESPACE(pointwise_avx2_asm)+MLD_SYSV_ABI+void mld_pointwise_avx2_asm(int32_t *a, const int32_t *b, const int32_t *qdata)+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/x86_64/proofs/mldsa_pointwise_avx2_asm.ml */+__contract__(+ requires(memory_no_alias(a, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(b, sizeof(int32_t) * MLDSA_N))+ /* Input bound MLD_NTT_BOUND = 9 * MLD_FQMUL_BOUND, the guaranteed bound of+ * any forward NTT implementation. Hardcoded here to keep this header free+ * of poly.h. */+ /* check-magic: 94279698 == 9 * ((5 * MLDSA_Q + 3) / 4) */+ requires(array_abs_bound(a, 0, MLDSA_N, 94279698))+ requires(array_abs_bound(b, 0, MLDSA_N, 94279698))+ requires(qdata == mld_qdata)+ assigns(memory_slice(a, sizeof(int32_t) * MLDSA_N))+ ensures(array_abs_bound(a, 0, MLDSA_N, 8380417))+);++#define mld_pointwise_acc_l4_avx2_asm MLD_NAMESPACE(pointwise_acc_l4_avx2_asm)+MLD_SYSV_ABI+void mld_pointwise_acc_l4_avx2_asm(int32_t c[MLDSA_N],+ const int32_t a[4][MLDSA_N],+ const int32_t b[4][MLDSA_N],+ const int32_t *qdata)+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/x86_64/proofs/mldsa_pointwise_acc_l4_avx2_asm.ml */+__contract__(+ requires(memory_no_alias(c, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(a, sizeof(int32_t) * 4 * MLDSA_N))+ requires(memory_no_alias(b, sizeof(int32_t) * 4 * MLDSA_N))+ requires(forall(l0, 0, 4, array_abs_bound(a[l0], 0, MLDSA_N, 8380417)))+ /* check-magic: 94279698 == 9 * ((5 * MLDSA_Q + 3) / 4) */+ requires(forall(l1, 0, 4, array_abs_bound(b[l1], 0, MLDSA_N, 94279698)))+ requires(qdata == mld_qdata)+ assigns(memory_slice(c, sizeof(int32_t) * MLDSA_N))+ ensures(array_abs_bound(c, 0, MLDSA_N, 8380417))+);++#define mld_pointwise_acc_l5_avx2_asm MLD_NAMESPACE(pointwise_acc_l5_avx2_asm)+MLD_SYSV_ABI+void mld_pointwise_acc_l5_avx2_asm(int32_t c[MLDSA_N],+ const int32_t a[5][MLDSA_N],+ const int32_t b[5][MLDSA_N],+ const int32_t *qdata)+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/x86_64/proofs/mldsa_pointwise_acc_l5_avx2_asm.ml */+__contract__(+ requires(memory_no_alias(c, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(a, sizeof(int32_t) * 5 * MLDSA_N))+ requires(memory_no_alias(b, sizeof(int32_t) * 5 * MLDSA_N))+ requires(forall(l0, 0, 5, array_abs_bound(a[l0], 0, MLDSA_N, 8380417)))+ /* check-magic: 94279698 == 9 * ((5 * MLDSA_Q + 3) / 4) */+ requires(forall(l1, 0, 5, array_abs_bound(b[l1], 0, MLDSA_N, 94279698)))+ requires(qdata == mld_qdata)+ assigns(memory_slice(c, sizeof(int32_t) * MLDSA_N))+ ensures(array_abs_bound(c, 0, MLDSA_N, 8380417))+);++#define mld_pointwise_acc_l7_avx2_asm MLD_NAMESPACE(pointwise_acc_l7_avx2_asm)+MLD_SYSV_ABI+void mld_pointwise_acc_l7_avx2_asm(int32_t c[MLDSA_N],+ const int32_t a[7][MLDSA_N],+ const int32_t b[7][MLDSA_N],+ const int32_t *qdata)+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/x86_64/proofs/mldsa_pointwise_acc_l7_avx2_asm.ml */+__contract__(+ requires(memory_no_alias(c, sizeof(int32_t) * MLDSA_N))+ requires(memory_no_alias(a, sizeof(int32_t) * 7 * MLDSA_N))+ requires(memory_no_alias(b, sizeof(int32_t) * 7 * MLDSA_N))+ requires(forall(l0, 0, 7, array_abs_bound(a[l0], 0, MLDSA_N, 8380417)))+ /* check-magic: 94279698 == 9 * ((5 * MLDSA_Q + 3) / 4) */+ requires(forall(l1, 0, 7, array_abs_bound(b[l1], 0, MLDSA_N, 94279698)))+ requires(qdata == mld_qdata)+ assigns(memory_slice(c, sizeof(int32_t) * MLDSA_N))+ ensures(array_abs_bound(c, 0, MLDSA_N, 8380417))+);++#endif /* !MLD_NATIVE_X86_64_SRC_ARITH_NATIVE_X86_64_H */
@@ -0,0 +1,157 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*+ * WARNING: This file is auto-generated from scripts/autogen+ * in the mldsa-native repository.+ * Do not modify it directly.+ */++#include "../../../common.h"++#if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++#include "consts.h"++/*+ * Table of zeta values used in the AVX2 forward and inverse NTT+ * See autogen for details.+ */+MLD_ALIGN MLD_INTERNAL_DATA_DEFINITION const int32_t mld_qdata[624] = {+ 8380417, 8380417, 8380417, 8380417, 8380417,+ 8380417, 8380417, 8380417, 58728449, 58728449,+ 58728449, 58728449, 58728449, 58728449, 58728449,+ 58728449, -8395782, -8395782, -8395782, -8395782,+ -8395782, -8395782, -8395782, -8395782, 41978,+ 41978, 41978, 41978, 41978, 41978,+ 41978, 41978, -151046689, 1830765815, -1929875198,+ -1927777021, 1640767044, 1477910808, 1612161320, 1640734244,+ 308362795, 308362795, 308362795, 308362795, -1815525077,+ -1815525077, -1815525077, -1815525077, -1374673747, -1374673747,+ -1374673747, -1374673747, -1091570561, -1091570561, -1091570561,+ -1091570561, -1929495947, -1929495947, -1929495947, -1929495947,+ 515185417, 515185417, 515185417, 515185417, -285697463,+ -285697463, -285697463, -285697463, 625853735, 625853735,+ 625853735, 625853735, 1727305304, 1727305304, 2082316400,+ 2082316400, -1364982364, -1364982364, 858240904, 858240904,+ 1806278032, 1806278032, 222489248, 222489248, -346752664,+ -346752664, 684667771, 684667771, 1654287830, 1654287830,+ -878576921, -878576921, -1257667337, -1257667337, -748618600,+ -748618600, 329347125, 329347125, 1837364258, 1837364258,+ -1443016191, -1443016191, -1170414139, -1170414139, -1846138265,+ -1631226336, -1404529459, 1838055109, 1594295555, -1076973524,+ -1898723372, -594436433, -202001019, -475984260, -561427818,+ 1797021249, -1061813248, 2059733581, -1661512036, -1104976547,+ -1750224323, -901666090, 418987550, 1831915353, -1925356481,+ 992097815, 879957084, 2024403852, 1484874664, -1636082790,+ -285388938, -1983539117, -1495136972, -950076368, -1714807468,+ -952438995, -1574918427, 1350681039, -2143979939, 1599739335,+ -1285853323, -993005454, -1440787840, 568627424, -783134478,+ -588790216, 289871779, -1262003603, 2135294594, -1018755525,+ -889861155, 1665705315, 1321868265, 1225434135, -1784632064,+ 666258756, 675310538, -1555941048, -1999506068, -1499481951,+ -695180180, -1375177022, 1777179795, 334803717, -178766299,+ -518252220, 1957047970, 1146323031, -654783359, -1974159335,+ 1651689966, 140455867, -1039411342, 1955560694, 1529189038,+ -2131021878, -247357819, 1518161567, -86965173, 1708872713,+ 1787797779, 1638590967, -120646188, -1669960606, -916321552,+ 1155548552, 2143745726, 1210558298, -1261461890, -318346816,+ 628664287, -1729304568, 1422575624, 1424130038, -1185330464,+ 235321234, 168022240, 1206536194, 985155484, -894060583,+ -898413, -1363460238, -605900043, 2027833504, 14253662,+ 1014493059, 863641633, 1819892093, 2124962073, -1223601433,+ -1920467227, -1637785316, -1536588520, 694382729, 235104446,+ -1045062172, 831969619, -300448763, 756955444, -260312805,+ 1554794072, 1339088280, -2040058690, -853476187, -2047270596,+ -1723816713, -1591599803, -440824168, 1119856484, 1544891539,+ 155290192, -973777462, 991903578, 912367099, -44694137,+ 1176904444, -421552614, -818371958, 1747917558, -325927722,+ 908452108, 1851023419, -1176751719, -1354528380, -72690498,+ -314284737, 985022747, 963438279, -1078959975, 604552167,+ -1021949428, 608791570, 173440395, -2126092136, -1316619236,+ -1039370342, 6087993, -110126092, 565464272, -1758099917,+ -1600929361, 879867909, -1809756372, 400711272, 1363007700,+ 30313375, -326425360, 1683520342, -517299994, 2027935492,+ -1372618620, 128353682, -1123881663, 137583815, -635454918,+ -642772911, 45766801, 671509323, -2070602178, 419615363,+ 1216882040, -270590488, -1276805128, 371462360, -1357098057,+ -384158533, 827959816, -596344473, 702390549, -279505433,+ -260424530, -71875110, -1208667171, -1499603926, 2036925262,+ -540420426, 746144248, -1420958686, 2032221021, 1904936414,+ 1257750362, 1926727420, 1931587462, 1258381762, 885133339,+ 1629985060, 1967222129, 6363718, -1287922800, 1136965286,+ 1779436847, 1116720494, 1042326957, 1405999311, 713994583,+ 940195359, -1542497137, 2061661095, -883155599, 1726753853,+ -1547952704, 394851342, 283780712, 776003547, 1123958025,+ 201262505, 1934038751, 374860238, -3975713, 25847,+ -2608894, -518909, 237124, -777960, -876248,+ 466468, 1826347, 1826347, 1826347, 1826347,+ 2353451, 2353451, 2353451, 2353451, -359251,+ -359251, -359251, -359251, -2091905, -2091905,+ -2091905, -2091905, 3119733, 3119733, 3119733,+ 3119733, -2884855, -2884855, -2884855, -2884855,+ 3111497, 3111497, 3111497, 3111497, 2680103,+ 2680103, 2680103, 2680103, 2725464, 2725464,+ 1024112, 1024112, -1079900, -1079900, 3585928,+ 3585928, -549488, -549488, -1119584, -1119584,+ 2619752, 2619752, -2108549, -2108549, -2118186,+ -2118186, -3859737, -3859737, -1399561, -1399561,+ -3277672, -3277672, 1757237, 1757237, -19422,+ -19422, 4010497, 4010497, 280005, 280005,+ 2706023, 95776, 3077325, 3530437, -1661693,+ -3592148, -2537516, 3915439, -3861115, -3043716,+ 3574422, -2867647, 3539968, -300467, 2348700,+ -539299, -1699267, -1643818, 3505694, -3821735,+ 3507263, -2140649, -1600420, 3699596, 811944,+ 531354, 954230, 3881043, 3900724, -2556880,+ 2071892, -2797779, -3930395, -3677745, -1452451,+ 2176455, -1257611, -4083598, -3190144, -3632928,+ 3412210, 2147896, -2967645, -411027, -671102,+ -22981, -381987, 1852771, -3343383, 508951,+ 44288, 904516, -3724342, 1653064, 2389356,+ 759969, 189548, 3159746, -2409325, 1315589,+ 1285669, -812732, -3019102, -3628969, -1528703,+ -3041255, 3475950, -1585221, 1939314, -1000202,+ -3157330, 126922, -983419, 2715295, -3693493,+ -2477047, -1228525, -1308169, 1349076, -1430430,+ 264944, 3097992, -1100098, 3958618, -8578,+ -3249728, -210977, -1316856, -3553272, -1851402,+ -177440, 1341330, -1584928, -1439742, -3881060,+ 3839961, 2091667, -3342478, 266997, -3520352,+ 900702, 495491, -655327, -3556995, 342297,+ 3437287, 2842341, 4055324, -3767016, -2994039,+ -1333058, -451100, -1279661, 1500165, -542412,+ -2584293, -2013608, 1957272, -3183426, 810149,+ -3038916, 2213111, -426683, -1667432, -2939036,+ 183443, -554416, 3937738, 3407706, 2244091,+ 2434439, -3759364, 1859098, -1613174, -3122442,+ -525098, 286988, -3342277, 2691481, 1247620,+ 1250494, 1869119, 1237275, 1312455, 1917081,+ 777191, -2831860, -3724270, 2432395, 3369112,+ 162844, 1652634, 3523897, -975884, 1723600,+ -1104333, -2235985, -976891, 3919660, 1400424,+ 2316500, -2446433, -1235728, -1197226, 909542,+ -43260, 2031748, -768622, -2437823, 1735879,+ -2590150, 2486353, 2635921, 1903435, -3318210,+ 3306115, -2546312, 2235880, -1671176, 594136,+ 2454455, 185531, 1616392, -3694233, 3866901,+ 1717735, -1803090, -260646, -420899, 1612842,+ -48306, -846154, 3817976, -3562462, 3513181,+ -3193378, 819034, -522500, 3207046, -3595838,+ 4108315, 203044, 1265009, 1595974, -3548272,+ -1050970, -1430225, -1962642, -1374803, 3406031,+ -1846953, -3776993, -164721, -1207385, 3014001,+ -1799107, 269760, 472078, 1910376, -3833893,+ -2286327, -3545687, -1362209, 1976782,+};++#else /* MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_MULTILEVEL_NO_SHARED \+ */++MLD_EMPTY_CU(avx2_consts)++#endif /* !(MLD_ARITH_BACKEND_X86_64_DEFAULT && \+ !MLD_CONFIG_MULTILEVEL_NO_SHARED) */
@@ -0,0 +1,27 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*+ * WARNING: This file is auto-generated from scripts/autogen+ * in the mldsa-native repository.+ * Do not modify it directly.+ */++#ifndef MLD_NATIVE_X86_64_SRC_CONSTS_H+#define MLD_NATIVE_X86_64_SRC_CONSTS_H+#include "../../../common.h"+#define MLD_AVX2_BACKEND_DATA_OFFSET_8XQ 0+#define MLD_AVX2_BACKEND_DATA_OFFSET_8XQINV 8+#define MLD_AVX2_BACKEND_DATA_OFFSET_8XDIV_QINV 16+#define MLD_AVX2_BACKEND_DATA_OFFSET_8XDIV 24+#define MLD_AVX2_BACKEND_DATA_OFFSET_ZETAS_QINV 32+#define MLD_AVX2_BACKEND_DATA_OFFSET_ZETAS 328++#ifndef __ASSEMBLER__+#define mld_qdata MLD_NAMESPACE(qdata)+MLD_INTERNAL_DATA_DECLARATION const int32_t mld_qdata[624];+#endif++#endif /* !MLD_NATIVE_X86_64_SRC_CONSTS_H */
@@ -0,0 +1,2333 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [REF_AVX2]+ * CRYSTALS-Dilithium optimized AVX2 implementation+ * Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/dilithium/tree/master/avx2+ */++ /*+ * This file is derived from the public domain+ * AVX2 Dilithium implementation @[REF_AVX2].+ */++/*yaml+ Name: invntt_avx2_asm+ Description: x86_64 AVX2 inverse NTT+ Signature: void mld_invntt_avx2_asm(int32_t *r, const int32_t *qdata)+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 1024+ permissions: read/write+ c_parameter: int32_t *r+ description: Input/output polynomial (256 x int32_t)+ rsi:+ type: buffer+ size_bytes: 2496+ permissions: read-only+ c_parameter: const int32_t *qdata+ description: Precomputed constants (624 x int32_t)+*/++#include "../../../common.h"+#if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/x86_64/src/mldsa_intt_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(invntt_avx2_asm)+MLD_ASM_FN_SYMBOL(invntt_avx2_asm)++ .cfi_startproc+ vmovdqa (%rsi), %ymm0+ vmovdqa (%rdi), %ymm4+ vmovdqa 0x20(%rdi), %ymm5+ vmovdqa 0x40(%rdi), %ymm6+ vmovdqa 0x60(%rdi), %ymm7+ vmovdqa 0x80(%rdi), %ymm8+ vmovdqa 0xa0(%rdi), %ymm9+ vmovdqa 0xc0(%rdi), %ymm10+ vmovdqa 0xe0(%rdi), %ymm11+ vpermq $0x1b, 0x500(%rsi), %ymm3 # ymm3 = mem[3,2,1,0]+ vpermq $0x1b, 0x9a0(%rsi), %ymm15 # ymm15 = mem[3,2,1,0]+ vmovshdup %ymm3, %ymm1 # ymm1 = ymm3[1,1,3,3,5,5,7,7]+ vmovshdup %ymm15, %ymm2 # ymm2 = ymm15[1,1,3,3,5,5,7,7]+ vpsubd %ymm4, %ymm5, %ymm12+ vpaddd %ymm5, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm5 # ymm5 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm5, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm5, %ymm5+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm5, %ymm5+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm5, %ymm12, %ymm5 # ymm5 = ymm12[0],ymm5[1],ymm12[2],ymm5[3],ymm12[4],ymm5[5],ymm12[6],ymm5[7]+ vpermq $0x1b, 0x480(%rsi), %ymm3 # ymm3 = mem[3,2,1,0]+ vpermq $0x1b, 0x920(%rsi), %ymm15 # ymm15 = mem[3,2,1,0]+ vmovshdup %ymm3, %ymm1 # ymm1 = ymm3[1,1,3,3,5,5,7,7]+ vmovshdup %ymm15, %ymm2 # ymm2 = ymm15[1,1,3,3,5,5,7,7]+ vpsubd %ymm6, %ymm7, %ymm12+ vpaddd %ymm7, %ymm6, %ymm6+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm7 # ymm7 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm7, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm7, %ymm7+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm7, %ymm7+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm7, %ymm12, %ymm7 # ymm7 = ymm12[0],ymm7[1],ymm12[2],ymm7[3],ymm12[4],ymm7[5],ymm12[6],ymm7[7]+ vpermq $0x1b, 0x400(%rsi), %ymm3 # ymm3 = mem[3,2,1,0]+ vpermq $0x1b, 0x8a0(%rsi), %ymm15 # ymm15 = mem[3,2,1,0]+ vmovshdup %ymm3, %ymm1 # ymm1 = ymm3[1,1,3,3,5,5,7,7]+ vmovshdup %ymm15, %ymm2 # ymm2 = ymm15[1,1,3,3,5,5,7,7]+ vpsubd %ymm8, %ymm9, %ymm12+ vpaddd %ymm9, %ymm8, %ymm8+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm9 # ymm9 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm9, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm9, %ymm9+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm9, %ymm9+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm9, %ymm12, %ymm9 # ymm9 = ymm12[0],ymm9[1],ymm12[2],ymm9[3],ymm12[4],ymm9[5],ymm12[6],ymm9[7]+ vpermq $0x1b, 0x380(%rsi), %ymm3 # ymm3 = mem[3,2,1,0]+ vpermq $0x1b, 0x820(%rsi), %ymm15 # ymm15 = mem[3,2,1,0]+ vmovshdup %ymm3, %ymm1 # ymm1 = ymm3[1,1,3,3,5,5,7,7]+ vmovshdup %ymm15, %ymm2 # ymm2 = ymm15[1,1,3,3,5,5,7,7]+ vpsubd %ymm10, %ymm11, %ymm12+ vpaddd %ymm11, %ymm10, %ymm10+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vpermq $0x1b, 0x300(%rsi), %ymm3 # ymm3 = mem[3,2,1,0]+ vpermq $0x1b, 0x7a0(%rsi), %ymm15 # ymm15 = mem[3,2,1,0]+ vmovshdup %ymm3, %ymm1 # ymm1 = ymm3[1,1,3,3,5,5,7,7]+ vmovshdup %ymm15, %ymm2 # ymm2 = ymm15[1,1,3,3,5,5,7,7]+ vpsubd %ymm4, %ymm6, %ymm12+ vpaddd %ymm6, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm6 # ymm6 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm6, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm6, %ymm6+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm6, %ymm6+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm6, %ymm12, %ymm6 # ymm6 = ymm12[0],ymm6[1],ymm12[2],ymm6[3],ymm12[4],ymm6[5],ymm12[6],ymm6[7]+ vpsubd %ymm5, %ymm7, %ymm12+ vpaddd %ymm7, %ymm5, %ymm5+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm7 # ymm7 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm7, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm7, %ymm7+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm7, %ymm7+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm7, %ymm12, %ymm7 # ymm7 = ymm12[0],ymm7[1],ymm12[2],ymm7[3],ymm12[4],ymm7[5],ymm12[6],ymm7[7]+ vpermq $0x1b, 0x280(%rsi), %ymm3 # ymm3 = mem[3,2,1,0]+ vpermq $0x1b, 0x720(%rsi), %ymm15 # ymm15 = mem[3,2,1,0]+ vmovshdup %ymm3, %ymm1 # ymm1 = ymm3[1,1,3,3,5,5,7,7]+ vmovshdup %ymm15, %ymm2 # ymm2 = ymm15[1,1,3,3,5,5,7,7]+ vpsubd %ymm8, %ymm10, %ymm12+ vpaddd %ymm10, %ymm8, %ymm8+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm10 # ymm10 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm10, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm10, %ymm10+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm10, %ymm10+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm10, %ymm12, %ymm10 # ymm10 = ymm12[0],ymm10[1],ymm12[2],ymm10[3],ymm12[4],ymm10[5],ymm12[6],ymm10[7]+ vpsubd %ymm9, %ymm11, %ymm12+ vpaddd %ymm11, %ymm9, %ymm9+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vpermq $0x1b, 0x200(%rsi), %ymm3 # ymm3 = mem[3,2,1,0]+ vpermq $0x1b, 0x6a0(%rsi), %ymm15 # ymm15 = mem[3,2,1,0]+ vmovshdup %ymm3, %ymm1 # ymm1 = ymm3[1,1,3,3,5,5,7,7]+ vmovshdup %ymm15, %ymm2 # ymm2 = ymm15[1,1,3,3,5,5,7,7]+ vpsubd %ymm4, %ymm8, %ymm12+ vpaddd %ymm4, %ymm8, %ymm4+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm8 # ymm8 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm8, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm8, %ymm8+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm8, %ymm8+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm8, %ymm12, %ymm8 # ymm8 = ymm12[0],ymm8[1],ymm12[2],ymm8[3],ymm12[4],ymm8[5],ymm12[6],ymm8[7]+ vpsubd %ymm5, %ymm9, %ymm12+ vpaddd %ymm5, %ymm9, %ymm5+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm9 # ymm9 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm9, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm9, %ymm9+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm9, %ymm9+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm9, %ymm12, %ymm9 # ymm9 = ymm12[0],ymm9[1],ymm12[2],ymm9[3],ymm12[4],ymm9[5],ymm12[6],ymm9[7]+ vpsubd %ymm6, %ymm10, %ymm12+ vpaddd %ymm6, %ymm10, %ymm6+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm10 # ymm10 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm10, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm10, %ymm10+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm10, %ymm10+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm10, %ymm12, %ymm10 # ymm10 = ymm12[0],ymm10[1],ymm12[2],ymm10[3],ymm12[4],ymm10[5],ymm12[6],ymm10[7]+ vpsubd %ymm7, %ymm11, %ymm12+ vpaddd %ymm7, %ymm11, %ymm7+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vmovsldup %ymm5, %ymm3 # ymm3 = ymm5[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm3, %ymm4, %ymm3 # ymm3 = ymm4[0],ymm3[1],ymm4[2],ymm3[3],ymm4[4],ymm3[5],ymm4[6],ymm3[7]+ vpsrlq $0x20, %ymm4, %ymm4+ vpblendd $0xaa, %ymm5, %ymm4, %ymm5 # ymm5 = ymm4[0],ymm5[1],ymm4[2],ymm5[3],ymm4[4],ymm5[5],ymm4[6],ymm5[7]+ vmovsldup %ymm7, %ymm4 # ymm4 = ymm7[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm4, %ymm6, %ymm4 # ymm4 = ymm6[0],ymm4[1],ymm6[2],ymm4[3],ymm6[4],ymm4[5],ymm6[6],ymm4[7]+ vpsrlq $0x20, %ymm6, %ymm6+ vpblendd $0xaa, %ymm7, %ymm6, %ymm7 # ymm7 = ymm6[0],ymm7[1],ymm6[2],ymm7[3],ymm6[4],ymm7[5],ymm6[6],ymm7[7]+ vmovsldup %ymm9, %ymm6 # ymm6 = ymm9[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm6, %ymm8, %ymm6 # ymm6 = ymm8[0],ymm6[1],ymm8[2],ymm6[3],ymm8[4],ymm6[5],ymm8[6],ymm6[7]+ vpsrlq $0x20, %ymm8, %ymm8+ vpblendd $0xaa, %ymm9, %ymm8, %ymm9 # ymm9 = ymm8[0],ymm9[1],ymm8[2],ymm9[3],ymm8[4],ymm9[5],ymm8[6],ymm9[7]+ vmovsldup %ymm11, %ymm8 # ymm8 = ymm11[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm8, %ymm10, %ymm8 # ymm8 = ymm10[0],ymm8[1],ymm10[2],ymm8[3],ymm10[4],ymm8[5],ymm10[6],ymm8[7]+ vpsrlq $0x20, %ymm10, %ymm10+ vpblendd $0xaa, %ymm11, %ymm10, %ymm11 # ymm11 = ymm10[0],ymm11[1],ymm10[2],ymm11[3],ymm10[4],ymm11[5],ymm10[6],ymm11[7]+ vpermq $0x1b, 0x180(%rsi), %ymm1 # ymm1 = mem[3,2,1,0]+ vpermq $0x1b, 0x620(%rsi), %ymm2 # ymm2 = mem[3,2,1,0]+ vpsubd %ymm3, %ymm5, %ymm12+ vpaddd %ymm5, %ymm3, %ymm3+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm5 # ymm5 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm5, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm5, %ymm5+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm5, %ymm5+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm5, %ymm12, %ymm5 # ymm5 = ymm12[0],ymm5[1],ymm12[2],ymm5[3],ymm12[4],ymm5[5],ymm12[6],ymm5[7]+ vpsubd %ymm4, %ymm7, %ymm12+ vpaddd %ymm7, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm7 # ymm7 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm7, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm7, %ymm7+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm7, %ymm7+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm7, %ymm12, %ymm7 # ymm7 = ymm12[0],ymm7[1],ymm12[2],ymm7[3],ymm12[4],ymm7[5],ymm12[6],ymm7[7]+ vpsubd %ymm6, %ymm9, %ymm12+ vpaddd %ymm6, %ymm9, %ymm6+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm9 # ymm9 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm9, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm9, %ymm9+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm9, %ymm9+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm9, %ymm12, %ymm9 # ymm9 = ymm12[0],ymm9[1],ymm12[2],ymm9[3],ymm12[4],ymm9[5],ymm12[6],ymm9[7]+ vpsubd %ymm8, %ymm11, %ymm12+ vpaddd %ymm11, %ymm8, %ymm8+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vpunpcklqdq %ymm4, %ymm3, %ymm10 # ymm10 = ymm3[0],ymm4[0],ymm3[2],ymm4[2]+ vpunpckhqdq %ymm4, %ymm3, %ymm4 # ymm4 = ymm3[1],ymm4[1],ymm3[3],ymm4[3]+ vpunpcklqdq %ymm8, %ymm6, %ymm3 # ymm3 = ymm6[0],ymm8[0],ymm6[2],ymm8[2]+ vpunpckhqdq %ymm8, %ymm6, %ymm8 # ymm8 = ymm6[1],ymm8[1],ymm6[3],ymm8[3]+ vpunpcklqdq %ymm7, %ymm5, %ymm6 # ymm6 = ymm5[0],ymm7[0],ymm5[2],ymm7[2]+ vpunpckhqdq %ymm7, %ymm5, %ymm7 # ymm7 = ymm5[1],ymm7[1],ymm5[3],ymm7[3]+ vpunpcklqdq %ymm11, %ymm9, %ymm5 # ymm5 = ymm9[0],ymm11[0],ymm9[2],ymm11[2]+ vpunpckhqdq %ymm11, %ymm9, %ymm11 # ymm11 = ymm9[1],ymm11[1],ymm9[3],ymm11[3]+ vpermq $0x1b, 0x100(%rsi), %ymm1 # ymm1 = mem[3,2,1,0]+ vpermq $0x1b, 0x5a0(%rsi), %ymm2 # ymm2 = mem[3,2,1,0]+ vpsubd %ymm10, %ymm4, %ymm12+ vpaddd %ymm4, %ymm10, %ymm10+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm4 # ymm4 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm4, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm4, %ymm4+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm4, %ymm4+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm4, %ymm12, %ymm4 # ymm4 = ymm12[0],ymm4[1],ymm12[2],ymm4[3],ymm12[4],ymm4[5],ymm12[6],ymm4[7]+ vpsubd %ymm3, %ymm8, %ymm12+ vpaddd %ymm3, %ymm8, %ymm3+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm8 # ymm8 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm8, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm8, %ymm8+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm8, %ymm12, %ymm8 # ymm8 = ymm12[0],ymm8[1],ymm12[2],ymm8[3],ymm12[4],ymm8[5],ymm12[6],ymm8[7]+ vpsubd %ymm6, %ymm7, %ymm12+ vpaddd %ymm7, %ymm6, %ymm6+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm7 # ymm7 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm7, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm7, %ymm7+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm7, %ymm7+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm7, %ymm12, %ymm7 # ymm7 = ymm12[0],ymm7[1],ymm12[2],ymm7[3],ymm12[4],ymm7[5],ymm12[6],ymm7[7]+ vpsubd %ymm5, %ymm11, %ymm12+ vpaddd %ymm5, %ymm11, %ymm5+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vperm2i128 $0x20, %ymm3, %ymm10, %ymm9 # ymm9 = ymm10[0,1],ymm3[0,1]+ vperm2i128 $0x31, %ymm3, %ymm10, %ymm3 # ymm3 = ymm10[2,3],ymm3[2,3]+ vperm2i128 $0x20, %ymm5, %ymm6, %ymm10 # ymm10 = ymm6[0,1],ymm5[0,1]+ vperm2i128 $0x31, %ymm5, %ymm6, %ymm5 # ymm5 = ymm6[2,3],ymm5[2,3]+ vperm2i128 $0x20, %ymm8, %ymm4, %ymm6 # ymm6 = ymm4[0,1],ymm8[0,1]+ vperm2i128 $0x31, %ymm8, %ymm4, %ymm8 # ymm8 = ymm4[2,3],ymm8[2,3]+ vperm2i128 $0x20, %ymm11, %ymm7, %ymm4 # ymm4 = ymm7[0,1],ymm11[0,1]+ vperm2i128 $0x31, %ymm11, %ymm7, %ymm11 # ymm11 = ymm7[2,3],ymm11[2,3]+ vpbroadcastd 0x9c(%rsi), %ymm1+ vpbroadcastd 0x53c(%rsi), %ymm2+ vpsubd %ymm9, %ymm3, %ymm12+ vpaddd %ymm3, %ymm9, %ymm9+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm3 # ymm3 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm3, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm3, %ymm3+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm3, %ymm3+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm3, %ymm12, %ymm3 # ymm3 = ymm12[0],ymm3[1],ymm12[2],ymm3[3],ymm12[4],ymm3[5],ymm12[6],ymm3[7]+ vpsubd %ymm10, %ymm5, %ymm12+ vpaddd %ymm5, %ymm10, %ymm10+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm5 # ymm5 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm5, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm5, %ymm5+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm5, %ymm5+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm5, %ymm12, %ymm5 # ymm5 = ymm12[0],ymm5[1],ymm12[2],ymm5[3],ymm12[4],ymm5[5],ymm12[6],ymm5[7]+ vpsubd %ymm6, %ymm8, %ymm12+ vpaddd %ymm6, %ymm8, %ymm6+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm8 # ymm8 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm8, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm8, %ymm8+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm8, %ymm12, %ymm8 # ymm8 = ymm12[0],ymm8[1],ymm12[2],ymm8[3],ymm12[4],ymm8[5],ymm12[6],ymm8[7]+ vpsubd %ymm4, %ymm11, %ymm12+ vpaddd %ymm4, %ymm11, %ymm4+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vmovdqa %ymm9, (%rdi)+ vmovdqa %ymm10, 0x20(%rdi)+ vmovdqa %ymm6, 0x40(%rdi)+ vmovdqa %ymm4, 0x60(%rdi)+ vmovdqa %ymm3, 0x80(%rdi)+ vmovdqa %ymm5, 0xa0(%rdi)+ vmovdqa %ymm8, 0xc0(%rdi)+ vmovdqa %ymm11, 0xe0(%rdi)+ vmovdqa 0x100(%rdi), %ymm4+ vmovdqa 0x120(%rdi), %ymm5+ vmovdqa 0x140(%rdi), %ymm6+ vmovdqa 0x160(%rdi), %ymm7+ vmovdqa 0x180(%rdi), %ymm8+ vmovdqa 0x1a0(%rdi), %ymm9+ vmovdqa 0x1c0(%rdi), %ymm10+ vmovdqa 0x1e0(%rdi), %ymm11+ vpermq $0x1b, 0x4e0(%rsi), %ymm3 # ymm3 = mem[3,2,1,0]+ vpermq $0x1b, 0x980(%rsi), %ymm15 # ymm15 = mem[3,2,1,0]+ vmovshdup %ymm3, %ymm1 # ymm1 = ymm3[1,1,3,3,5,5,7,7]+ vmovshdup %ymm15, %ymm2 # ymm2 = ymm15[1,1,3,3,5,5,7,7]+ vpsubd %ymm4, %ymm5, %ymm12+ vpaddd %ymm5, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm5 # ymm5 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm5, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm5, %ymm5+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm5, %ymm5+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm5, %ymm12, %ymm5 # ymm5 = ymm12[0],ymm5[1],ymm12[2],ymm5[3],ymm12[4],ymm5[5],ymm12[6],ymm5[7]+ vpermq $0x1b, 0x460(%rsi), %ymm3 # ymm3 = mem[3,2,1,0]+ vpermq $0x1b, 0x900(%rsi), %ymm15 # ymm15 = mem[3,2,1,0]+ vmovshdup %ymm3, %ymm1 # ymm1 = ymm3[1,1,3,3,5,5,7,7]+ vmovshdup %ymm15, %ymm2 # ymm2 = ymm15[1,1,3,3,5,5,7,7]+ vpsubd %ymm6, %ymm7, %ymm12+ vpaddd %ymm7, %ymm6, %ymm6+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm7 # ymm7 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm7, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm7, %ymm7+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm7, %ymm7+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm7, %ymm12, %ymm7 # ymm7 = ymm12[0],ymm7[1],ymm12[2],ymm7[3],ymm12[4],ymm7[5],ymm12[6],ymm7[7]+ vpermq $0x1b, 0x3e0(%rsi), %ymm3 # ymm3 = mem[3,2,1,0]+ vpermq $0x1b, 0x880(%rsi), %ymm15 # ymm15 = mem[3,2,1,0]+ vmovshdup %ymm3, %ymm1 # ymm1 = ymm3[1,1,3,3,5,5,7,7]+ vmovshdup %ymm15, %ymm2 # ymm2 = ymm15[1,1,3,3,5,5,7,7]+ vpsubd %ymm8, %ymm9, %ymm12+ vpaddd %ymm9, %ymm8, %ymm8+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm9 # ymm9 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm9, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm9, %ymm9+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm9, %ymm9+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm9, %ymm12, %ymm9 # ymm9 = ymm12[0],ymm9[1],ymm12[2],ymm9[3],ymm12[4],ymm9[5],ymm12[6],ymm9[7]+ vpermq $0x1b, 0x360(%rsi), %ymm3 # ymm3 = mem[3,2,1,0]+ vpermq $0x1b, 0x800(%rsi), %ymm15 # ymm15 = mem[3,2,1,0]+ vmovshdup %ymm3, %ymm1 # ymm1 = ymm3[1,1,3,3,5,5,7,7]+ vmovshdup %ymm15, %ymm2 # ymm2 = ymm15[1,1,3,3,5,5,7,7]+ vpsubd %ymm10, %ymm11, %ymm12+ vpaddd %ymm11, %ymm10, %ymm10+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vpermq $0x1b, 0x2e0(%rsi), %ymm3 # ymm3 = mem[3,2,1,0]+ vpermq $0x1b, 0x780(%rsi), %ymm15 # ymm15 = mem[3,2,1,0]+ vmovshdup %ymm3, %ymm1 # ymm1 = ymm3[1,1,3,3,5,5,7,7]+ vmovshdup %ymm15, %ymm2 # ymm2 = ymm15[1,1,3,3,5,5,7,7]+ vpsubd %ymm4, %ymm6, %ymm12+ vpaddd %ymm6, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm6 # ymm6 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm6, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm6, %ymm6+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm6, %ymm6+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm6, %ymm12, %ymm6 # ymm6 = ymm12[0],ymm6[1],ymm12[2],ymm6[3],ymm12[4],ymm6[5],ymm12[6],ymm6[7]+ vpsubd %ymm5, %ymm7, %ymm12+ vpaddd %ymm7, %ymm5, %ymm5+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm7 # ymm7 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm7, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm7, %ymm7+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm7, %ymm7+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm7, %ymm12, %ymm7 # ymm7 = ymm12[0],ymm7[1],ymm12[2],ymm7[3],ymm12[4],ymm7[5],ymm12[6],ymm7[7]+ vpermq $0x1b, 0x260(%rsi), %ymm3 # ymm3 = mem[3,2,1,0]+ vpermq $0x1b, 0x700(%rsi), %ymm15 # ymm15 = mem[3,2,1,0]+ vmovshdup %ymm3, %ymm1 # ymm1 = ymm3[1,1,3,3,5,5,7,7]+ vmovshdup %ymm15, %ymm2 # ymm2 = ymm15[1,1,3,3,5,5,7,7]+ vpsubd %ymm8, %ymm10, %ymm12+ vpaddd %ymm10, %ymm8, %ymm8+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm10 # ymm10 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm10, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm10, %ymm10+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm10, %ymm10+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm10, %ymm12, %ymm10 # ymm10 = ymm12[0],ymm10[1],ymm12[2],ymm10[3],ymm12[4],ymm10[5],ymm12[6],ymm10[7]+ vpsubd %ymm9, %ymm11, %ymm12+ vpaddd %ymm11, %ymm9, %ymm9+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vpermq $0x1b, 0x1e0(%rsi), %ymm3 # ymm3 = mem[3,2,1,0]+ vpermq $0x1b, 0x680(%rsi), %ymm15 # ymm15 = mem[3,2,1,0]+ vmovshdup %ymm3, %ymm1 # ymm1 = ymm3[1,1,3,3,5,5,7,7]+ vmovshdup %ymm15, %ymm2 # ymm2 = ymm15[1,1,3,3,5,5,7,7]+ vpsubd %ymm4, %ymm8, %ymm12+ vpaddd %ymm4, %ymm8, %ymm4+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm8 # ymm8 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm8, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm8, %ymm8+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm8, %ymm8+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm8, %ymm12, %ymm8 # ymm8 = ymm12[0],ymm8[1],ymm12[2],ymm8[3],ymm12[4],ymm8[5],ymm12[6],ymm8[7]+ vpsubd %ymm5, %ymm9, %ymm12+ vpaddd %ymm5, %ymm9, %ymm5+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm9 # ymm9 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm9, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm9, %ymm9+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm9, %ymm9+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm9, %ymm12, %ymm9 # ymm9 = ymm12[0],ymm9[1],ymm12[2],ymm9[3],ymm12[4],ymm9[5],ymm12[6],ymm9[7]+ vpsubd %ymm6, %ymm10, %ymm12+ vpaddd %ymm6, %ymm10, %ymm6+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm10 # ymm10 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm10, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm10, %ymm10+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm10, %ymm10+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm10, %ymm12, %ymm10 # ymm10 = ymm12[0],ymm10[1],ymm12[2],ymm10[3],ymm12[4],ymm10[5],ymm12[6],ymm10[7]+ vpsubd %ymm7, %ymm11, %ymm12+ vpaddd %ymm7, %ymm11, %ymm7+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vmovsldup %ymm5, %ymm3 # ymm3 = ymm5[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm3, %ymm4, %ymm3 # ymm3 = ymm4[0],ymm3[1],ymm4[2],ymm3[3],ymm4[4],ymm3[5],ymm4[6],ymm3[7]+ vpsrlq $0x20, %ymm4, %ymm4+ vpblendd $0xaa, %ymm5, %ymm4, %ymm5 # ymm5 = ymm4[0],ymm5[1],ymm4[2],ymm5[3],ymm4[4],ymm5[5],ymm4[6],ymm5[7]+ vmovsldup %ymm7, %ymm4 # ymm4 = ymm7[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm4, %ymm6, %ymm4 # ymm4 = ymm6[0],ymm4[1],ymm6[2],ymm4[3],ymm6[4],ymm4[5],ymm6[6],ymm4[7]+ vpsrlq $0x20, %ymm6, %ymm6+ vpblendd $0xaa, %ymm7, %ymm6, %ymm7 # ymm7 = ymm6[0],ymm7[1],ymm6[2],ymm7[3],ymm6[4],ymm7[5],ymm6[6],ymm7[7]+ vmovsldup %ymm9, %ymm6 # ymm6 = ymm9[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm6, %ymm8, %ymm6 # ymm6 = ymm8[0],ymm6[1],ymm8[2],ymm6[3],ymm8[4],ymm6[5],ymm8[6],ymm6[7]+ vpsrlq $0x20, %ymm8, %ymm8+ vpblendd $0xaa, %ymm9, %ymm8, %ymm9 # ymm9 = ymm8[0],ymm9[1],ymm8[2],ymm9[3],ymm8[4],ymm9[5],ymm8[6],ymm9[7]+ vmovsldup %ymm11, %ymm8 # ymm8 = ymm11[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm8, %ymm10, %ymm8 # ymm8 = ymm10[0],ymm8[1],ymm10[2],ymm8[3],ymm10[4],ymm8[5],ymm10[6],ymm8[7]+ vpsrlq $0x20, %ymm10, %ymm10+ vpblendd $0xaa, %ymm11, %ymm10, %ymm11 # ymm11 = ymm10[0],ymm11[1],ymm10[2],ymm11[3],ymm10[4],ymm11[5],ymm10[6],ymm11[7]+ vpermq $0x1b, 0x160(%rsi), %ymm1 # ymm1 = mem[3,2,1,0]+ vpermq $0x1b, 0x600(%rsi), %ymm2 # ymm2 = mem[3,2,1,0]+ vpsubd %ymm3, %ymm5, %ymm12+ vpaddd %ymm5, %ymm3, %ymm3+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm5 # ymm5 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm5, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm5, %ymm5+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm5, %ymm5+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm5, %ymm12, %ymm5 # ymm5 = ymm12[0],ymm5[1],ymm12[2],ymm5[3],ymm12[4],ymm5[5],ymm12[6],ymm5[7]+ vpsubd %ymm4, %ymm7, %ymm12+ vpaddd %ymm7, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm7 # ymm7 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm7, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm7, %ymm7+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm7, %ymm7+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm7, %ymm12, %ymm7 # ymm7 = ymm12[0],ymm7[1],ymm12[2],ymm7[3],ymm12[4],ymm7[5],ymm12[6],ymm7[7]+ vpsubd %ymm6, %ymm9, %ymm12+ vpaddd %ymm6, %ymm9, %ymm6+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm9 # ymm9 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm9, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm9, %ymm9+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm9, %ymm9+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm9, %ymm12, %ymm9 # ymm9 = ymm12[0],ymm9[1],ymm12[2],ymm9[3],ymm12[4],ymm9[5],ymm12[6],ymm9[7]+ vpsubd %ymm8, %ymm11, %ymm12+ vpaddd %ymm11, %ymm8, %ymm8+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vpunpcklqdq %ymm4, %ymm3, %ymm10 # ymm10 = ymm3[0],ymm4[0],ymm3[2],ymm4[2]+ vpunpckhqdq %ymm4, %ymm3, %ymm4 # ymm4 = ymm3[1],ymm4[1],ymm3[3],ymm4[3]+ vpunpcklqdq %ymm8, %ymm6, %ymm3 # ymm3 = ymm6[0],ymm8[0],ymm6[2],ymm8[2]+ vpunpckhqdq %ymm8, %ymm6, %ymm8 # ymm8 = ymm6[1],ymm8[1],ymm6[3],ymm8[3]+ vpunpcklqdq %ymm7, %ymm5, %ymm6 # ymm6 = ymm5[0],ymm7[0],ymm5[2],ymm7[2]+ vpunpckhqdq %ymm7, %ymm5, %ymm7 # ymm7 = ymm5[1],ymm7[1],ymm5[3],ymm7[3]+ vpunpcklqdq %ymm11, %ymm9, %ymm5 # ymm5 = ymm9[0],ymm11[0],ymm9[2],ymm11[2]+ vpunpckhqdq %ymm11, %ymm9, %ymm11 # ymm11 = ymm9[1],ymm11[1],ymm9[3],ymm11[3]+ vpermq $0x1b, 0xe0(%rsi), %ymm1 # ymm1 = mem[3,2,1,0]+ vpermq $0x1b, 0x580(%rsi), %ymm2 # ymm2 = mem[3,2,1,0]+ vpsubd %ymm10, %ymm4, %ymm12+ vpaddd %ymm4, %ymm10, %ymm10+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm4 # ymm4 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm4, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm4, %ymm4+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm4, %ymm4+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm4, %ymm12, %ymm4 # ymm4 = ymm12[0],ymm4[1],ymm12[2],ymm4[3],ymm12[4],ymm4[5],ymm12[6],ymm4[7]+ vpsubd %ymm3, %ymm8, %ymm12+ vpaddd %ymm3, %ymm8, %ymm3+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm8 # ymm8 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm8, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm8, %ymm8+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm8, %ymm12, %ymm8 # ymm8 = ymm12[0],ymm8[1],ymm12[2],ymm8[3],ymm12[4],ymm8[5],ymm12[6],ymm8[7]+ vpsubd %ymm6, %ymm7, %ymm12+ vpaddd %ymm7, %ymm6, %ymm6+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm7 # ymm7 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm7, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm7, %ymm7+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm7, %ymm7+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm7, %ymm12, %ymm7 # ymm7 = ymm12[0],ymm7[1],ymm12[2],ymm7[3],ymm12[4],ymm7[5],ymm12[6],ymm7[7]+ vpsubd %ymm5, %ymm11, %ymm12+ vpaddd %ymm5, %ymm11, %ymm5+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vperm2i128 $0x20, %ymm3, %ymm10, %ymm9 # ymm9 = ymm10[0,1],ymm3[0,1]+ vperm2i128 $0x31, %ymm3, %ymm10, %ymm3 # ymm3 = ymm10[2,3],ymm3[2,3]+ vperm2i128 $0x20, %ymm5, %ymm6, %ymm10 # ymm10 = ymm6[0,1],ymm5[0,1]+ vperm2i128 $0x31, %ymm5, %ymm6, %ymm5 # ymm5 = ymm6[2,3],ymm5[2,3]+ vperm2i128 $0x20, %ymm8, %ymm4, %ymm6 # ymm6 = ymm4[0,1],ymm8[0,1]+ vperm2i128 $0x31, %ymm8, %ymm4, %ymm8 # ymm8 = ymm4[2,3],ymm8[2,3]+ vperm2i128 $0x20, %ymm11, %ymm7, %ymm4 # ymm4 = ymm7[0,1],ymm11[0,1]+ vperm2i128 $0x31, %ymm11, %ymm7, %ymm11 # ymm11 = ymm7[2,3],ymm11[2,3]+ vpbroadcastd 0x98(%rsi), %ymm1+ vpbroadcastd 0x538(%rsi), %ymm2+ vpsubd %ymm9, %ymm3, %ymm12+ vpaddd %ymm3, %ymm9, %ymm9+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm3 # ymm3 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm3, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm3, %ymm3+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm3, %ymm3+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm3, %ymm12, %ymm3 # ymm3 = ymm12[0],ymm3[1],ymm12[2],ymm3[3],ymm12[4],ymm3[5],ymm12[6],ymm3[7]+ vpsubd %ymm10, %ymm5, %ymm12+ vpaddd %ymm5, %ymm10, %ymm10+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm5 # ymm5 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm5, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm5, %ymm5+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm5, %ymm5+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm5, %ymm12, %ymm5 # ymm5 = ymm12[0],ymm5[1],ymm12[2],ymm5[3],ymm12[4],ymm5[5],ymm12[6],ymm5[7]+ vpsubd %ymm6, %ymm8, %ymm12+ vpaddd %ymm6, %ymm8, %ymm6+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm8 # ymm8 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm8, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm8, %ymm8+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm8, %ymm12, %ymm8 # ymm8 = ymm12[0],ymm8[1],ymm12[2],ymm8[3],ymm12[4],ymm8[5],ymm12[6],ymm8[7]+ vpsubd %ymm4, %ymm11, %ymm12+ vpaddd %ymm4, %ymm11, %ymm4+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vmovdqa %ymm9, 0x100(%rdi)+ vmovdqa %ymm10, 0x120(%rdi)+ vmovdqa %ymm6, 0x140(%rdi)+ vmovdqa %ymm4, 0x160(%rdi)+ vmovdqa %ymm3, 0x180(%rdi)+ vmovdqa %ymm5, 0x1a0(%rdi)+ vmovdqa %ymm8, 0x1c0(%rdi)+ vmovdqa %ymm11, 0x1e0(%rdi)+ vmovdqa 0x200(%rdi), %ymm4+ vmovdqa 0x220(%rdi), %ymm5+ vmovdqa 0x240(%rdi), %ymm6+ vmovdqa 0x260(%rdi), %ymm7+ vmovdqa 0x280(%rdi), %ymm8+ vmovdqa 0x2a0(%rdi), %ymm9+ vmovdqa 0x2c0(%rdi), %ymm10+ vmovdqa 0x2e0(%rdi), %ymm11+ vpermq $0x1b, 0x4c0(%rsi), %ymm3 # ymm3 = mem[3,2,1,0]+ vpermq $0x1b, 0x960(%rsi), %ymm15 # ymm15 = mem[3,2,1,0]+ vmovshdup %ymm3, %ymm1 # ymm1 = ymm3[1,1,3,3,5,5,7,7]+ vmovshdup %ymm15, %ymm2 # ymm2 = ymm15[1,1,3,3,5,5,7,7]+ vpsubd %ymm4, %ymm5, %ymm12+ vpaddd %ymm5, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm5 # ymm5 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm5, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm5, %ymm5+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm5, %ymm5+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm5, %ymm12, %ymm5 # ymm5 = ymm12[0],ymm5[1],ymm12[2],ymm5[3],ymm12[4],ymm5[5],ymm12[6],ymm5[7]+ vpermq $0x1b, 0x440(%rsi), %ymm3 # ymm3 = mem[3,2,1,0]+ vpermq $0x1b, 0x8e0(%rsi), %ymm15 # ymm15 = mem[3,2,1,0]+ vmovshdup %ymm3, %ymm1 # ymm1 = ymm3[1,1,3,3,5,5,7,7]+ vmovshdup %ymm15, %ymm2 # ymm2 = ymm15[1,1,3,3,5,5,7,7]+ vpsubd %ymm6, %ymm7, %ymm12+ vpaddd %ymm7, %ymm6, %ymm6+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm7 # ymm7 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm7, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm7, %ymm7+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm7, %ymm7+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm7, %ymm12, %ymm7 # ymm7 = ymm12[0],ymm7[1],ymm12[2],ymm7[3],ymm12[4],ymm7[5],ymm12[6],ymm7[7]+ vpermq $0x1b, 0x3c0(%rsi), %ymm3 # ymm3 = mem[3,2,1,0]+ vpermq $0x1b, 0x860(%rsi), %ymm15 # ymm15 = mem[3,2,1,0]+ vmovshdup %ymm3, %ymm1 # ymm1 = ymm3[1,1,3,3,5,5,7,7]+ vmovshdup %ymm15, %ymm2 # ymm2 = ymm15[1,1,3,3,5,5,7,7]+ vpsubd %ymm8, %ymm9, %ymm12+ vpaddd %ymm9, %ymm8, %ymm8+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm9 # ymm9 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm9, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm9, %ymm9+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm9, %ymm9+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm9, %ymm12, %ymm9 # ymm9 = ymm12[0],ymm9[1],ymm12[2],ymm9[3],ymm12[4],ymm9[5],ymm12[6],ymm9[7]+ vpermq $0x1b, 0x340(%rsi), %ymm3 # ymm3 = mem[3,2,1,0]+ vpermq $0x1b, 0x7e0(%rsi), %ymm15 # ymm15 = mem[3,2,1,0]+ vmovshdup %ymm3, %ymm1 # ymm1 = ymm3[1,1,3,3,5,5,7,7]+ vmovshdup %ymm15, %ymm2 # ymm2 = ymm15[1,1,3,3,5,5,7,7]+ vpsubd %ymm10, %ymm11, %ymm12+ vpaddd %ymm11, %ymm10, %ymm10+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vpermq $0x1b, 0x2c0(%rsi), %ymm3 # ymm3 = mem[3,2,1,0]+ vpermq $0x1b, 0x760(%rsi), %ymm15 # ymm15 = mem[3,2,1,0]+ vmovshdup %ymm3, %ymm1 # ymm1 = ymm3[1,1,3,3,5,5,7,7]+ vmovshdup %ymm15, %ymm2 # ymm2 = ymm15[1,1,3,3,5,5,7,7]+ vpsubd %ymm4, %ymm6, %ymm12+ vpaddd %ymm6, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm6 # ymm6 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm6, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm6, %ymm6+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm6, %ymm6+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm6, %ymm12, %ymm6 # ymm6 = ymm12[0],ymm6[1],ymm12[2],ymm6[3],ymm12[4],ymm6[5],ymm12[6],ymm6[7]+ vpsubd %ymm5, %ymm7, %ymm12+ vpaddd %ymm7, %ymm5, %ymm5+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm7 # ymm7 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm7, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm7, %ymm7+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm7, %ymm7+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm7, %ymm12, %ymm7 # ymm7 = ymm12[0],ymm7[1],ymm12[2],ymm7[3],ymm12[4],ymm7[5],ymm12[6],ymm7[7]+ vpermq $0x1b, 0x240(%rsi), %ymm3 # ymm3 = mem[3,2,1,0]+ vpermq $0x1b, 0x6e0(%rsi), %ymm15 # ymm15 = mem[3,2,1,0]+ vmovshdup %ymm3, %ymm1 # ymm1 = ymm3[1,1,3,3,5,5,7,7]+ vmovshdup %ymm15, %ymm2 # ymm2 = ymm15[1,1,3,3,5,5,7,7]+ vpsubd %ymm8, %ymm10, %ymm12+ vpaddd %ymm10, %ymm8, %ymm8+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm10 # ymm10 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm10, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm10, %ymm10+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm10, %ymm10+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm10, %ymm12, %ymm10 # ymm10 = ymm12[0],ymm10[1],ymm12[2],ymm10[3],ymm12[4],ymm10[5],ymm12[6],ymm10[7]+ vpsubd %ymm9, %ymm11, %ymm12+ vpaddd %ymm11, %ymm9, %ymm9+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vpermq $0x1b, 0x1c0(%rsi), %ymm3 # ymm3 = mem[3,2,1,0]+ vpermq $0x1b, 0x660(%rsi), %ymm15 # ymm15 = mem[3,2,1,0]+ vmovshdup %ymm3, %ymm1 # ymm1 = ymm3[1,1,3,3,5,5,7,7]+ vmovshdup %ymm15, %ymm2 # ymm2 = ymm15[1,1,3,3,5,5,7,7]+ vpsubd %ymm4, %ymm8, %ymm12+ vpaddd %ymm4, %ymm8, %ymm4+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm8 # ymm8 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm8, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm8, %ymm8+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm8, %ymm8+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm8, %ymm12, %ymm8 # ymm8 = ymm12[0],ymm8[1],ymm12[2],ymm8[3],ymm12[4],ymm8[5],ymm12[6],ymm8[7]+ vpsubd %ymm5, %ymm9, %ymm12+ vpaddd %ymm5, %ymm9, %ymm5+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm9 # ymm9 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm9, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm9, %ymm9+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm9, %ymm9+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm9, %ymm12, %ymm9 # ymm9 = ymm12[0],ymm9[1],ymm12[2],ymm9[3],ymm12[4],ymm9[5],ymm12[6],ymm9[7]+ vpsubd %ymm6, %ymm10, %ymm12+ vpaddd %ymm6, %ymm10, %ymm6+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm10 # ymm10 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm10, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm10, %ymm10+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm10, %ymm10+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm10, %ymm12, %ymm10 # ymm10 = ymm12[0],ymm10[1],ymm12[2],ymm10[3],ymm12[4],ymm10[5],ymm12[6],ymm10[7]+ vpsubd %ymm7, %ymm11, %ymm12+ vpaddd %ymm7, %ymm11, %ymm7+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vmovsldup %ymm5, %ymm3 # ymm3 = ymm5[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm3, %ymm4, %ymm3 # ymm3 = ymm4[0],ymm3[1],ymm4[2],ymm3[3],ymm4[4],ymm3[5],ymm4[6],ymm3[7]+ vpsrlq $0x20, %ymm4, %ymm4+ vpblendd $0xaa, %ymm5, %ymm4, %ymm5 # ymm5 = ymm4[0],ymm5[1],ymm4[2],ymm5[3],ymm4[4],ymm5[5],ymm4[6],ymm5[7]+ vmovsldup %ymm7, %ymm4 # ymm4 = ymm7[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm4, %ymm6, %ymm4 # ymm4 = ymm6[0],ymm4[1],ymm6[2],ymm4[3],ymm6[4],ymm4[5],ymm6[6],ymm4[7]+ vpsrlq $0x20, %ymm6, %ymm6+ vpblendd $0xaa, %ymm7, %ymm6, %ymm7 # ymm7 = ymm6[0],ymm7[1],ymm6[2],ymm7[3],ymm6[4],ymm7[5],ymm6[6],ymm7[7]+ vmovsldup %ymm9, %ymm6 # ymm6 = ymm9[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm6, %ymm8, %ymm6 # ymm6 = ymm8[0],ymm6[1],ymm8[2],ymm6[3],ymm8[4],ymm6[5],ymm8[6],ymm6[7]+ vpsrlq $0x20, %ymm8, %ymm8+ vpblendd $0xaa, %ymm9, %ymm8, %ymm9 # ymm9 = ymm8[0],ymm9[1],ymm8[2],ymm9[3],ymm8[4],ymm9[5],ymm8[6],ymm9[7]+ vmovsldup %ymm11, %ymm8 # ymm8 = ymm11[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm8, %ymm10, %ymm8 # ymm8 = ymm10[0],ymm8[1],ymm10[2],ymm8[3],ymm10[4],ymm8[5],ymm10[6],ymm8[7]+ vpsrlq $0x20, %ymm10, %ymm10+ vpblendd $0xaa, %ymm11, %ymm10, %ymm11 # ymm11 = ymm10[0],ymm11[1],ymm10[2],ymm11[3],ymm10[4],ymm11[5],ymm10[6],ymm11[7]+ vpermq $0x1b, 0x140(%rsi), %ymm1 # ymm1 = mem[3,2,1,0]+ vpermq $0x1b, 0x5e0(%rsi), %ymm2 # ymm2 = mem[3,2,1,0]+ vpsubd %ymm3, %ymm5, %ymm12+ vpaddd %ymm5, %ymm3, %ymm3+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm5 # ymm5 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm5, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm5, %ymm5+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm5, %ymm5+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm5, %ymm12, %ymm5 # ymm5 = ymm12[0],ymm5[1],ymm12[2],ymm5[3],ymm12[4],ymm5[5],ymm12[6],ymm5[7]+ vpsubd %ymm4, %ymm7, %ymm12+ vpaddd %ymm7, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm7 # ymm7 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm7, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm7, %ymm7+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm7, %ymm7+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm7, %ymm12, %ymm7 # ymm7 = ymm12[0],ymm7[1],ymm12[2],ymm7[3],ymm12[4],ymm7[5],ymm12[6],ymm7[7]+ vpsubd %ymm6, %ymm9, %ymm12+ vpaddd %ymm6, %ymm9, %ymm6+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm9 # ymm9 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm9, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm9, %ymm9+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm9, %ymm9+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm9, %ymm12, %ymm9 # ymm9 = ymm12[0],ymm9[1],ymm12[2],ymm9[3],ymm12[4],ymm9[5],ymm12[6],ymm9[7]+ vpsubd %ymm8, %ymm11, %ymm12+ vpaddd %ymm11, %ymm8, %ymm8+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vpunpcklqdq %ymm4, %ymm3, %ymm10 # ymm10 = ymm3[0],ymm4[0],ymm3[2],ymm4[2]+ vpunpckhqdq %ymm4, %ymm3, %ymm4 # ymm4 = ymm3[1],ymm4[1],ymm3[3],ymm4[3]+ vpunpcklqdq %ymm8, %ymm6, %ymm3 # ymm3 = ymm6[0],ymm8[0],ymm6[2],ymm8[2]+ vpunpckhqdq %ymm8, %ymm6, %ymm8 # ymm8 = ymm6[1],ymm8[1],ymm6[3],ymm8[3]+ vpunpcklqdq %ymm7, %ymm5, %ymm6 # ymm6 = ymm5[0],ymm7[0],ymm5[2],ymm7[2]+ vpunpckhqdq %ymm7, %ymm5, %ymm7 # ymm7 = ymm5[1],ymm7[1],ymm5[3],ymm7[3]+ vpunpcklqdq %ymm11, %ymm9, %ymm5 # ymm5 = ymm9[0],ymm11[0],ymm9[2],ymm11[2]+ vpunpckhqdq %ymm11, %ymm9, %ymm11 # ymm11 = ymm9[1],ymm11[1],ymm9[3],ymm11[3]+ vpermq $0x1b, 0xc0(%rsi), %ymm1 # ymm1 = mem[3,2,1,0]+ vpermq $0x1b, 0x560(%rsi), %ymm2 # ymm2 = mem[3,2,1,0]+ vpsubd %ymm10, %ymm4, %ymm12+ vpaddd %ymm4, %ymm10, %ymm10+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm4 # ymm4 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm4, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm4, %ymm4+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm4, %ymm4+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm4, %ymm12, %ymm4 # ymm4 = ymm12[0],ymm4[1],ymm12[2],ymm4[3],ymm12[4],ymm4[5],ymm12[6],ymm4[7]+ vpsubd %ymm3, %ymm8, %ymm12+ vpaddd %ymm3, %ymm8, %ymm3+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm8 # ymm8 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm8, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm8, %ymm8+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm8, %ymm12, %ymm8 # ymm8 = ymm12[0],ymm8[1],ymm12[2],ymm8[3],ymm12[4],ymm8[5],ymm12[6],ymm8[7]+ vpsubd %ymm6, %ymm7, %ymm12+ vpaddd %ymm7, %ymm6, %ymm6+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm7 # ymm7 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm7, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm7, %ymm7+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm7, %ymm7+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm7, %ymm12, %ymm7 # ymm7 = ymm12[0],ymm7[1],ymm12[2],ymm7[3],ymm12[4],ymm7[5],ymm12[6],ymm7[7]+ vpsubd %ymm5, %ymm11, %ymm12+ vpaddd %ymm5, %ymm11, %ymm5+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vperm2i128 $0x20, %ymm3, %ymm10, %ymm9 # ymm9 = ymm10[0,1],ymm3[0,1]+ vperm2i128 $0x31, %ymm3, %ymm10, %ymm3 # ymm3 = ymm10[2,3],ymm3[2,3]+ vperm2i128 $0x20, %ymm5, %ymm6, %ymm10 # ymm10 = ymm6[0,1],ymm5[0,1]+ vperm2i128 $0x31, %ymm5, %ymm6, %ymm5 # ymm5 = ymm6[2,3],ymm5[2,3]+ vperm2i128 $0x20, %ymm8, %ymm4, %ymm6 # ymm6 = ymm4[0,1],ymm8[0,1]+ vperm2i128 $0x31, %ymm8, %ymm4, %ymm8 # ymm8 = ymm4[2,3],ymm8[2,3]+ vperm2i128 $0x20, %ymm11, %ymm7, %ymm4 # ymm4 = ymm7[0,1],ymm11[0,1]+ vperm2i128 $0x31, %ymm11, %ymm7, %ymm11 # ymm11 = ymm7[2,3],ymm11[2,3]+ vpbroadcastd 0x94(%rsi), %ymm1+ vpbroadcastd 0x534(%rsi), %ymm2+ vpsubd %ymm9, %ymm3, %ymm12+ vpaddd %ymm3, %ymm9, %ymm9+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm3 # ymm3 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm3, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm3, %ymm3+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm3, %ymm3+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm3, %ymm12, %ymm3 # ymm3 = ymm12[0],ymm3[1],ymm12[2],ymm3[3],ymm12[4],ymm3[5],ymm12[6],ymm3[7]+ vpsubd %ymm10, %ymm5, %ymm12+ vpaddd %ymm5, %ymm10, %ymm10+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm5 # ymm5 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm5, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm5, %ymm5+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm5, %ymm5+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm5, %ymm12, %ymm5 # ymm5 = ymm12[0],ymm5[1],ymm12[2],ymm5[3],ymm12[4],ymm5[5],ymm12[6],ymm5[7]+ vpsubd %ymm6, %ymm8, %ymm12+ vpaddd %ymm6, %ymm8, %ymm6+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm8 # ymm8 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm8, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm8, %ymm8+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm8, %ymm12, %ymm8 # ymm8 = ymm12[0],ymm8[1],ymm12[2],ymm8[3],ymm12[4],ymm8[5],ymm12[6],ymm8[7]+ vpsubd %ymm4, %ymm11, %ymm12+ vpaddd %ymm4, %ymm11, %ymm4+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vmovdqa %ymm9, 0x200(%rdi)+ vmovdqa %ymm10, 0x220(%rdi)+ vmovdqa %ymm6, 0x240(%rdi)+ vmovdqa %ymm4, 0x260(%rdi)+ vmovdqa %ymm3, 0x280(%rdi)+ vmovdqa %ymm5, 0x2a0(%rdi)+ vmovdqa %ymm8, 0x2c0(%rdi)+ vmovdqa %ymm11, 0x2e0(%rdi)+ vmovdqa 0x300(%rdi), %ymm4+ vmovdqa 0x320(%rdi), %ymm5+ vmovdqa 0x340(%rdi), %ymm6+ vmovdqa 0x360(%rdi), %ymm7+ vmovdqa 0x380(%rdi), %ymm8+ vmovdqa 0x3a0(%rdi), %ymm9+ vmovdqa 0x3c0(%rdi), %ymm10+ vmovdqa 0x3e0(%rdi), %ymm11+ vpermq $0x1b, 0x4a0(%rsi), %ymm3 # ymm3 = mem[3,2,1,0]+ vpermq $0x1b, 0x940(%rsi), %ymm15 # ymm15 = mem[3,2,1,0]+ vmovshdup %ymm3, %ymm1 # ymm1 = ymm3[1,1,3,3,5,5,7,7]+ vmovshdup %ymm15, %ymm2 # ymm2 = ymm15[1,1,3,3,5,5,7,7]+ vpsubd %ymm4, %ymm5, %ymm12+ vpaddd %ymm5, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm5 # ymm5 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm5, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm5, %ymm5+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm5, %ymm5+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm5, %ymm12, %ymm5 # ymm5 = ymm12[0],ymm5[1],ymm12[2],ymm5[3],ymm12[4],ymm5[5],ymm12[6],ymm5[7]+ vpermq $0x1b, 0x420(%rsi), %ymm3 # ymm3 = mem[3,2,1,0]+ vpermq $0x1b, 0x8c0(%rsi), %ymm15 # ymm15 = mem[3,2,1,0]+ vmovshdup %ymm3, %ymm1 # ymm1 = ymm3[1,1,3,3,5,5,7,7]+ vmovshdup %ymm15, %ymm2 # ymm2 = ymm15[1,1,3,3,5,5,7,7]+ vpsubd %ymm6, %ymm7, %ymm12+ vpaddd %ymm7, %ymm6, %ymm6+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm7 # ymm7 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm7, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm7, %ymm7+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm7, %ymm7+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm7, %ymm12, %ymm7 # ymm7 = ymm12[0],ymm7[1],ymm12[2],ymm7[3],ymm12[4],ymm7[5],ymm12[6],ymm7[7]+ vpermq $0x1b, 0x3a0(%rsi), %ymm3 # ymm3 = mem[3,2,1,0]+ vpermq $0x1b, 0x840(%rsi), %ymm15 # ymm15 = mem[3,2,1,0]+ vmovshdup %ymm3, %ymm1 # ymm1 = ymm3[1,1,3,3,5,5,7,7]+ vmovshdup %ymm15, %ymm2 # ymm2 = ymm15[1,1,3,3,5,5,7,7]+ vpsubd %ymm8, %ymm9, %ymm12+ vpaddd %ymm9, %ymm8, %ymm8+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm9 # ymm9 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm9, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm9, %ymm9+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm9, %ymm9+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm9, %ymm12, %ymm9 # ymm9 = ymm12[0],ymm9[1],ymm12[2],ymm9[3],ymm12[4],ymm9[5],ymm12[6],ymm9[7]+ vpermq $0x1b, 0x320(%rsi), %ymm3 # ymm3 = mem[3,2,1,0]+ vpermq $0x1b, 0x7c0(%rsi), %ymm15 # ymm15 = mem[3,2,1,0]+ vmovshdup %ymm3, %ymm1 # ymm1 = ymm3[1,1,3,3,5,5,7,7]+ vmovshdup %ymm15, %ymm2 # ymm2 = ymm15[1,1,3,3,5,5,7,7]+ vpsubd %ymm10, %ymm11, %ymm12+ vpaddd %ymm11, %ymm10, %ymm10+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vpermq $0x1b, 0x2a0(%rsi), %ymm3 # ymm3 = mem[3,2,1,0]+ vpermq $0x1b, 0x740(%rsi), %ymm15 # ymm15 = mem[3,2,1,0]+ vmovshdup %ymm3, %ymm1 # ymm1 = ymm3[1,1,3,3,5,5,7,7]+ vmovshdup %ymm15, %ymm2 # ymm2 = ymm15[1,1,3,3,5,5,7,7]+ vpsubd %ymm4, %ymm6, %ymm12+ vpaddd %ymm6, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm6 # ymm6 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm6, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm6, %ymm6+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm6, %ymm6+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm6, %ymm12, %ymm6 # ymm6 = ymm12[0],ymm6[1],ymm12[2],ymm6[3],ymm12[4],ymm6[5],ymm12[6],ymm6[7]+ vpsubd %ymm5, %ymm7, %ymm12+ vpaddd %ymm7, %ymm5, %ymm5+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm7 # ymm7 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm7, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm7, %ymm7+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm7, %ymm7+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm7, %ymm12, %ymm7 # ymm7 = ymm12[0],ymm7[1],ymm12[2],ymm7[3],ymm12[4],ymm7[5],ymm12[6],ymm7[7]+ vpermq $0x1b, 0x220(%rsi), %ymm3 # ymm3 = mem[3,2,1,0]+ vpermq $0x1b, 0x6c0(%rsi), %ymm15 # ymm15 = mem[3,2,1,0]+ vmovshdup %ymm3, %ymm1 # ymm1 = ymm3[1,1,3,3,5,5,7,7]+ vmovshdup %ymm15, %ymm2 # ymm2 = ymm15[1,1,3,3,5,5,7,7]+ vpsubd %ymm8, %ymm10, %ymm12+ vpaddd %ymm10, %ymm8, %ymm8+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm10 # ymm10 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm10, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm10, %ymm10+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm10, %ymm10+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm10, %ymm12, %ymm10 # ymm10 = ymm12[0],ymm10[1],ymm12[2],ymm10[3],ymm12[4],ymm10[5],ymm12[6],ymm10[7]+ vpsubd %ymm9, %ymm11, %ymm12+ vpaddd %ymm11, %ymm9, %ymm9+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vpermq $0x1b, 0x1a0(%rsi), %ymm3 # ymm3 = mem[3,2,1,0]+ vpermq $0x1b, 0x640(%rsi), %ymm15 # ymm15 = mem[3,2,1,0]+ vmovshdup %ymm3, %ymm1 # ymm1 = ymm3[1,1,3,3,5,5,7,7]+ vmovshdup %ymm15, %ymm2 # ymm2 = ymm15[1,1,3,3,5,5,7,7]+ vpsubd %ymm4, %ymm8, %ymm12+ vpaddd %ymm4, %ymm8, %ymm4+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm8 # ymm8 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm8, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm8, %ymm8+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm8, %ymm8+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm8, %ymm12, %ymm8 # ymm8 = ymm12[0],ymm8[1],ymm12[2],ymm8[3],ymm12[4],ymm8[5],ymm12[6],ymm8[7]+ vpsubd %ymm5, %ymm9, %ymm12+ vpaddd %ymm5, %ymm9, %ymm5+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm9 # ymm9 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm9, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm9, %ymm9+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm9, %ymm9+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm9, %ymm12, %ymm9 # ymm9 = ymm12[0],ymm9[1],ymm12[2],ymm9[3],ymm12[4],ymm9[5],ymm12[6],ymm9[7]+ vpsubd %ymm6, %ymm10, %ymm12+ vpaddd %ymm6, %ymm10, %ymm6+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm10 # ymm10 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm10, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm10, %ymm10+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm10, %ymm10+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm10, %ymm12, %ymm10 # ymm10 = ymm12[0],ymm10[1],ymm12[2],ymm10[3],ymm12[4],ymm10[5],ymm12[6],ymm10[7]+ vpsubd %ymm7, %ymm11, %ymm12+ vpaddd %ymm7, %ymm11, %ymm7+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm3, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm15, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vmovsldup %ymm5, %ymm3 # ymm3 = ymm5[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm3, %ymm4, %ymm3 # ymm3 = ymm4[0],ymm3[1],ymm4[2],ymm3[3],ymm4[4],ymm3[5],ymm4[6],ymm3[7]+ vpsrlq $0x20, %ymm4, %ymm4+ vpblendd $0xaa, %ymm5, %ymm4, %ymm5 # ymm5 = ymm4[0],ymm5[1],ymm4[2],ymm5[3],ymm4[4],ymm5[5],ymm4[6],ymm5[7]+ vmovsldup %ymm7, %ymm4 # ymm4 = ymm7[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm4, %ymm6, %ymm4 # ymm4 = ymm6[0],ymm4[1],ymm6[2],ymm4[3],ymm6[4],ymm4[5],ymm6[6],ymm4[7]+ vpsrlq $0x20, %ymm6, %ymm6+ vpblendd $0xaa, %ymm7, %ymm6, %ymm7 # ymm7 = ymm6[0],ymm7[1],ymm6[2],ymm7[3],ymm6[4],ymm7[5],ymm6[6],ymm7[7]+ vmovsldup %ymm9, %ymm6 # ymm6 = ymm9[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm6, %ymm8, %ymm6 # ymm6 = ymm8[0],ymm6[1],ymm8[2],ymm6[3],ymm8[4],ymm6[5],ymm8[6],ymm6[7]+ vpsrlq $0x20, %ymm8, %ymm8+ vpblendd $0xaa, %ymm9, %ymm8, %ymm9 # ymm9 = ymm8[0],ymm9[1],ymm8[2],ymm9[3],ymm8[4],ymm9[5],ymm8[6],ymm9[7]+ vmovsldup %ymm11, %ymm8 # ymm8 = ymm11[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm8, %ymm10, %ymm8 # ymm8 = ymm10[0],ymm8[1],ymm10[2],ymm8[3],ymm10[4],ymm8[5],ymm10[6],ymm8[7]+ vpsrlq $0x20, %ymm10, %ymm10+ vpblendd $0xaa, %ymm11, %ymm10, %ymm11 # ymm11 = ymm10[0],ymm11[1],ymm10[2],ymm11[3],ymm10[4],ymm11[5],ymm10[6],ymm11[7]+ vpermq $0x1b, 0x120(%rsi), %ymm1 # ymm1 = mem[3,2,1,0]+ vpermq $0x1b, 0x5c0(%rsi), %ymm2 # ymm2 = mem[3,2,1,0]+ vpsubd %ymm3, %ymm5, %ymm12+ vpaddd %ymm5, %ymm3, %ymm3+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm5 # ymm5 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm5, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm5, %ymm5+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm5, %ymm5+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm5, %ymm12, %ymm5 # ymm5 = ymm12[0],ymm5[1],ymm12[2],ymm5[3],ymm12[4],ymm5[5],ymm12[6],ymm5[7]+ vpsubd %ymm4, %ymm7, %ymm12+ vpaddd %ymm7, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm7 # ymm7 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm7, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm7, %ymm7+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm7, %ymm7+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm7, %ymm12, %ymm7 # ymm7 = ymm12[0],ymm7[1],ymm12[2],ymm7[3],ymm12[4],ymm7[5],ymm12[6],ymm7[7]+ vpsubd %ymm6, %ymm9, %ymm12+ vpaddd %ymm6, %ymm9, %ymm6+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm9 # ymm9 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm9, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm9, %ymm9+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm9, %ymm9+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm9, %ymm12, %ymm9 # ymm9 = ymm12[0],ymm9[1],ymm12[2],ymm9[3],ymm12[4],ymm9[5],ymm12[6],ymm9[7]+ vpsubd %ymm8, %ymm11, %ymm12+ vpaddd %ymm11, %ymm8, %ymm8+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vpunpcklqdq %ymm4, %ymm3, %ymm10 # ymm10 = ymm3[0],ymm4[0],ymm3[2],ymm4[2]+ vpunpckhqdq %ymm4, %ymm3, %ymm4 # ymm4 = ymm3[1],ymm4[1],ymm3[3],ymm4[3]+ vpunpcklqdq %ymm8, %ymm6, %ymm3 # ymm3 = ymm6[0],ymm8[0],ymm6[2],ymm8[2]+ vpunpckhqdq %ymm8, %ymm6, %ymm8 # ymm8 = ymm6[1],ymm8[1],ymm6[3],ymm8[3]+ vpunpcklqdq %ymm7, %ymm5, %ymm6 # ymm6 = ymm5[0],ymm7[0],ymm5[2],ymm7[2]+ vpunpckhqdq %ymm7, %ymm5, %ymm7 # ymm7 = ymm5[1],ymm7[1],ymm5[3],ymm7[3]+ vpunpcklqdq %ymm11, %ymm9, %ymm5 # ymm5 = ymm9[0],ymm11[0],ymm9[2],ymm11[2]+ vpunpckhqdq %ymm11, %ymm9, %ymm11 # ymm11 = ymm9[1],ymm11[1],ymm9[3],ymm11[3]+ vpermq $0x1b, 0xa0(%rsi), %ymm1 # ymm1 = mem[3,2,1,0]+ vpermq $0x1b, 0x540(%rsi), %ymm2 # ymm2 = mem[3,2,1,0]+ vpsubd %ymm10, %ymm4, %ymm12+ vpaddd %ymm4, %ymm10, %ymm10+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm4 # ymm4 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm4, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm4, %ymm4+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm4, %ymm4+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm4, %ymm12, %ymm4 # ymm4 = ymm12[0],ymm4[1],ymm12[2],ymm4[3],ymm12[4],ymm4[5],ymm12[6],ymm4[7]+ vpsubd %ymm3, %ymm8, %ymm12+ vpaddd %ymm3, %ymm8, %ymm3+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm8 # ymm8 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm8, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm8, %ymm8+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm8, %ymm12, %ymm8 # ymm8 = ymm12[0],ymm8[1],ymm12[2],ymm8[3],ymm12[4],ymm8[5],ymm12[6],ymm8[7]+ vpsubd %ymm6, %ymm7, %ymm12+ vpaddd %ymm7, %ymm6, %ymm6+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm7 # ymm7 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm7, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm7, %ymm7+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm7, %ymm7+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm7, %ymm12, %ymm7 # ymm7 = ymm12[0],ymm7[1],ymm12[2],ymm7[3],ymm12[4],ymm7[5],ymm12[6],ymm7[7]+ vpsubd %ymm5, %ymm11, %ymm12+ vpaddd %ymm5, %ymm11, %ymm5+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vperm2i128 $0x20, %ymm3, %ymm10, %ymm9 # ymm9 = ymm10[0,1],ymm3[0,1]+ vperm2i128 $0x31, %ymm3, %ymm10, %ymm3 # ymm3 = ymm10[2,3],ymm3[2,3]+ vperm2i128 $0x20, %ymm5, %ymm6, %ymm10 # ymm10 = ymm6[0,1],ymm5[0,1]+ vperm2i128 $0x31, %ymm5, %ymm6, %ymm5 # ymm5 = ymm6[2,3],ymm5[2,3]+ vperm2i128 $0x20, %ymm8, %ymm4, %ymm6 # ymm6 = ymm4[0,1],ymm8[0,1]+ vperm2i128 $0x31, %ymm8, %ymm4, %ymm8 # ymm8 = ymm4[2,3],ymm8[2,3]+ vperm2i128 $0x20, %ymm11, %ymm7, %ymm4 # ymm4 = ymm7[0,1],ymm11[0,1]+ vperm2i128 $0x31, %ymm11, %ymm7, %ymm11 # ymm11 = ymm7[2,3],ymm11[2,3]+ vpbroadcastd 0x90(%rsi), %ymm1+ vpbroadcastd 0x530(%rsi), %ymm2+ vpsubd %ymm9, %ymm3, %ymm12+ vpaddd %ymm3, %ymm9, %ymm9+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm3 # ymm3 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm3, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm3, %ymm3+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm3, %ymm3+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm3, %ymm12, %ymm3 # ymm3 = ymm12[0],ymm3[1],ymm12[2],ymm3[3],ymm12[4],ymm3[5],ymm12[6],ymm3[7]+ vpsubd %ymm10, %ymm5, %ymm12+ vpaddd %ymm5, %ymm10, %ymm10+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm5 # ymm5 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm5, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm5, %ymm5+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm5, %ymm5+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm5, %ymm12, %ymm5 # ymm5 = ymm12[0],ymm5[1],ymm12[2],ymm5[3],ymm12[4],ymm5[5],ymm12[6],ymm5[7]+ vpsubd %ymm6, %ymm8, %ymm12+ vpaddd %ymm6, %ymm8, %ymm6+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm8 # ymm8 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm8, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm8, %ymm8+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm8, %ymm12, %ymm8 # ymm8 = ymm12[0],ymm8[1],ymm12[2],ymm8[3],ymm12[4],ymm8[5],ymm12[6],ymm8[7]+ vpsubd %ymm4, %ymm11, %ymm12+ vpaddd %ymm4, %ymm11, %ymm4+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vmovdqa %ymm9, 0x300(%rdi)+ vmovdqa %ymm10, 0x320(%rdi)+ vmovdqa %ymm6, 0x340(%rdi)+ vmovdqa %ymm4, 0x360(%rdi)+ vmovdqa %ymm3, 0x380(%rdi)+ vmovdqa %ymm5, 0x3a0(%rdi)+ vmovdqa %ymm8, 0x3c0(%rdi)+ vmovdqa %ymm11, 0x3e0(%rdi)+ vmovdqa (%rdi), %ymm4+ vmovdqa 0x80(%rdi), %ymm5+ vmovdqa 0x100(%rdi), %ymm6+ vmovdqa 0x180(%rdi), %ymm7+ vmovdqa 0x200(%rdi), %ymm8+ vmovdqa 0x280(%rdi), %ymm9+ vmovdqa 0x300(%rdi), %ymm10+ vmovdqa 0x380(%rdi), %ymm11+ vpbroadcastd 0x8c(%rsi), %ymm1+ vpbroadcastd 0x52c(%rsi), %ymm2+ vpsubd %ymm4, %ymm6, %ymm12+ vpaddd %ymm6, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm6 # ymm6 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm6, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm6, %ymm6+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm6, %ymm6+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm6, %ymm12, %ymm6 # ymm6 = ymm12[0],ymm6[1],ymm12[2],ymm6[3],ymm12[4],ymm6[5],ymm12[6],ymm6[7]+ vpsubd %ymm5, %ymm7, %ymm12+ vpaddd %ymm7, %ymm5, %ymm5+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm7 # ymm7 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm7, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm7, %ymm7+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm7, %ymm7+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm7, %ymm12, %ymm7 # ymm7 = ymm12[0],ymm7[1],ymm12[2],ymm7[3],ymm12[4],ymm7[5],ymm12[6],ymm7[7]+ vpbroadcastd 0x88(%rsi), %ymm1+ vpbroadcastd 0x528(%rsi), %ymm2+ vpsubd %ymm8, %ymm10, %ymm12+ vpaddd %ymm10, %ymm8, %ymm8+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm10 # ymm10 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm10, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm10, %ymm10+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm10, %ymm10+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm10, %ymm12, %ymm10 # ymm10 = ymm12[0],ymm10[1],ymm12[2],ymm10[3],ymm12[4],ymm10[5],ymm12[6],ymm10[7]+ vpsubd %ymm9, %ymm11, %ymm12+ vpaddd %ymm11, %ymm9, %ymm9+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vpbroadcastd 0x80(%rsi), %ymm1+ vpbroadcastd 0x520(%rsi), %ymm2+ vpsubd %ymm4, %ymm8, %ymm12+ vpaddd %ymm4, %ymm8, %ymm4+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm8 # ymm8 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm8, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm8, %ymm8+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm8, %ymm12, %ymm8 # ymm8 = ymm12[0],ymm8[1],ymm12[2],ymm8[3],ymm12[4],ymm8[5],ymm12[6],ymm8[7]+ vpsubd %ymm5, %ymm9, %ymm12+ vpaddd %ymm5, %ymm9, %ymm5+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm9 # ymm9 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm9, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm9, %ymm9+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm9, %ymm9+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm9, %ymm12, %ymm9 # ymm9 = ymm12[0],ymm9[1],ymm12[2],ymm9[3],ymm12[4],ymm9[5],ymm12[6],ymm9[7]+ vpsubd %ymm6, %ymm10, %ymm12+ vpaddd %ymm6, %ymm10, %ymm6+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm10 # ymm10 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm10, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm10, %ymm10+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm10, %ymm10+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm10, %ymm12, %ymm10 # ymm10 = ymm12[0],ymm10[1],ymm12[2],ymm10[3],ymm12[4],ymm10[5],ymm12[6],ymm10[7]+ vpsubd %ymm7, %ymm11, %ymm12+ vpaddd %ymm7, %ymm11, %ymm7+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vmovdqa %ymm8, 0x200(%rdi)+ vmovdqa %ymm9, 0x280(%rdi)+ vmovdqa %ymm10, 0x300(%rdi)+ vmovdqa %ymm11, 0x380(%rdi)+ vmovdqa 0x40(%rsi), %ymm1+ vmovdqa 0x60(%rsi), %ymm2+ vpmuldq %ymm1, %ymm4, %ymm12+ vpmuldq %ymm1, %ymm5, %ymm13+ vmovshdup %ymm4, %ymm8 # ymm8 = ymm4[1,1,3,3,5,5,7,7]+ vmovshdup %ymm5, %ymm9 # ymm9 = ymm5[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm8, %ymm14+ vpmuldq %ymm1, %ymm9, %ymm15+ vpmuldq %ymm2, %ymm4, %ymm4+ vpmuldq %ymm2, %ymm5, %ymm5+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm2, %ymm9, %ymm9+ vpmuldq %ymm0, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpmuldq %ymm0, %ymm15, %ymm15+ vpsubd %ymm12, %ymm4, %ymm4+ vpsubd %ymm13, %ymm5, %ymm5+ vpsubd %ymm14, %ymm8, %ymm8+ vpsubd %ymm15, %ymm9, %ymm9+ vmovshdup %ymm4, %ymm4 # ymm4 = ymm4[1,1,3,3,5,5,7,7]+ vmovshdup %ymm5, %ymm5 # ymm5 = ymm5[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm8, %ymm4, %ymm4 # ymm4 = ymm4[0],ymm8[1],ymm4[2],ymm8[3],ymm4[4],ymm8[5],ymm4[6],ymm8[7]+ vpblendd $0xaa, %ymm9, %ymm5, %ymm5 # ymm5 = ymm5[0],ymm9[1],ymm5[2],ymm9[3],ymm5[4],ymm9[5],ymm5[6],ymm9[7]+ vpmuldq %ymm1, %ymm6, %ymm12+ vpmuldq %ymm1, %ymm7, %ymm13+ vmovshdup %ymm6, %ymm8 # ymm8 = ymm6[1,1,3,3,5,5,7,7]+ vmovshdup %ymm7, %ymm9 # ymm9 = ymm7[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm8, %ymm14+ vpmuldq %ymm1, %ymm9, %ymm15+ vpmuldq %ymm2, %ymm6, %ymm6+ vpmuldq %ymm2, %ymm7, %ymm7+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm2, %ymm9, %ymm9+ vpmuldq %ymm0, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpmuldq %ymm0, %ymm15, %ymm15+ vpsubd %ymm12, %ymm6, %ymm6+ vpsubd %ymm13, %ymm7, %ymm7+ vpsubd %ymm14, %ymm8, %ymm8+ vpsubd %ymm15, %ymm9, %ymm9+ vmovshdup %ymm6, %ymm6 # ymm6 = ymm6[1,1,3,3,5,5,7,7]+ vmovshdup %ymm7, %ymm7 # ymm7 = ymm7[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm8, %ymm6, %ymm6 # ymm6 = ymm6[0],ymm8[1],ymm6[2],ymm8[3],ymm6[4],ymm8[5],ymm6[6],ymm8[7]+ vpblendd $0xaa, %ymm9, %ymm7, %ymm7 # ymm7 = ymm7[0],ymm9[1],ymm7[2],ymm9[3],ymm7[4],ymm9[5],ymm7[6],ymm9[7]+ vmovdqa %ymm4, (%rdi)+ vmovdqa %ymm5, 0x80(%rdi)+ vmovdqa %ymm6, 0x100(%rdi)+ vmovdqa %ymm7, 0x180(%rdi)+ vmovdqa 0x20(%rdi), %ymm4+ vmovdqa 0xa0(%rdi), %ymm5+ vmovdqa 0x120(%rdi), %ymm6+ vmovdqa 0x1a0(%rdi), %ymm7+ vmovdqa 0x220(%rdi), %ymm8+ vmovdqa 0x2a0(%rdi), %ymm9+ vmovdqa 0x320(%rdi), %ymm10+ vmovdqa 0x3a0(%rdi), %ymm11+ vpbroadcastd 0x8c(%rsi), %ymm1+ vpbroadcastd 0x52c(%rsi), %ymm2+ vpsubd %ymm4, %ymm6, %ymm12+ vpaddd %ymm6, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm6 # ymm6 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm6, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm6, %ymm6+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm6, %ymm6+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm6, %ymm12, %ymm6 # ymm6 = ymm12[0],ymm6[1],ymm12[2],ymm6[3],ymm12[4],ymm6[5],ymm12[6],ymm6[7]+ vpsubd %ymm5, %ymm7, %ymm12+ vpaddd %ymm7, %ymm5, %ymm5+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm7 # ymm7 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm7, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm7, %ymm7+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm7, %ymm7+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm7, %ymm12, %ymm7 # ymm7 = ymm12[0],ymm7[1],ymm12[2],ymm7[3],ymm12[4],ymm7[5],ymm12[6],ymm7[7]+ vpbroadcastd 0x88(%rsi), %ymm1+ vpbroadcastd 0x528(%rsi), %ymm2+ vpsubd %ymm8, %ymm10, %ymm12+ vpaddd %ymm10, %ymm8, %ymm8+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm10 # ymm10 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm10, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm10, %ymm10+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm10, %ymm10+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm10, %ymm12, %ymm10 # ymm10 = ymm12[0],ymm10[1],ymm12[2],ymm10[3],ymm12[4],ymm10[5],ymm12[6],ymm10[7]+ vpsubd %ymm9, %ymm11, %ymm12+ vpaddd %ymm11, %ymm9, %ymm9+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vpbroadcastd 0x80(%rsi), %ymm1+ vpbroadcastd 0x520(%rsi), %ymm2+ vpsubd %ymm4, %ymm8, %ymm12+ vpaddd %ymm4, %ymm8, %ymm4+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm8 # ymm8 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm8, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm8, %ymm8+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm8, %ymm12, %ymm8 # ymm8 = ymm12[0],ymm8[1],ymm12[2],ymm8[3],ymm12[4],ymm8[5],ymm12[6],ymm8[7]+ vpsubd %ymm5, %ymm9, %ymm12+ vpaddd %ymm5, %ymm9, %ymm5+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm9 # ymm9 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm9, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm9, %ymm9+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm9, %ymm9+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm9, %ymm12, %ymm9 # ymm9 = ymm12[0],ymm9[1],ymm12[2],ymm9[3],ymm12[4],ymm9[5],ymm12[6],ymm9[7]+ vpsubd %ymm6, %ymm10, %ymm12+ vpaddd %ymm6, %ymm10, %ymm6+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm10 # ymm10 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm10, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm10, %ymm10+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm10, %ymm10+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm10, %ymm12, %ymm10 # ymm10 = ymm12[0],ymm10[1],ymm12[2],ymm10[3],ymm12[4],ymm10[5],ymm12[6],ymm10[7]+ vpsubd %ymm7, %ymm11, %ymm12+ vpaddd %ymm7, %ymm11, %ymm7+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vmovdqa %ymm8, 0x220(%rdi)+ vmovdqa %ymm9, 0x2a0(%rdi)+ vmovdqa %ymm10, 0x320(%rdi)+ vmovdqa %ymm11, 0x3a0(%rdi)+ vmovdqa 0x40(%rsi), %ymm1+ vmovdqa 0x60(%rsi), %ymm2+ vpmuldq %ymm1, %ymm4, %ymm12+ vpmuldq %ymm1, %ymm5, %ymm13+ vmovshdup %ymm4, %ymm8 # ymm8 = ymm4[1,1,3,3,5,5,7,7]+ vmovshdup %ymm5, %ymm9 # ymm9 = ymm5[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm8, %ymm14+ vpmuldq %ymm1, %ymm9, %ymm15+ vpmuldq %ymm2, %ymm4, %ymm4+ vpmuldq %ymm2, %ymm5, %ymm5+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm2, %ymm9, %ymm9+ vpmuldq %ymm0, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpmuldq %ymm0, %ymm15, %ymm15+ vpsubd %ymm12, %ymm4, %ymm4+ vpsubd %ymm13, %ymm5, %ymm5+ vpsubd %ymm14, %ymm8, %ymm8+ vpsubd %ymm15, %ymm9, %ymm9+ vmovshdup %ymm4, %ymm4 # ymm4 = ymm4[1,1,3,3,5,5,7,7]+ vmovshdup %ymm5, %ymm5 # ymm5 = ymm5[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm8, %ymm4, %ymm4 # ymm4 = ymm4[0],ymm8[1],ymm4[2],ymm8[3],ymm4[4],ymm8[5],ymm4[6],ymm8[7]+ vpblendd $0xaa, %ymm9, %ymm5, %ymm5 # ymm5 = ymm5[0],ymm9[1],ymm5[2],ymm9[3],ymm5[4],ymm9[5],ymm5[6],ymm9[7]+ vpmuldq %ymm1, %ymm6, %ymm12+ vpmuldq %ymm1, %ymm7, %ymm13+ vmovshdup %ymm6, %ymm8 # ymm8 = ymm6[1,1,3,3,5,5,7,7]+ vmovshdup %ymm7, %ymm9 # ymm9 = ymm7[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm8, %ymm14+ vpmuldq %ymm1, %ymm9, %ymm15+ vpmuldq %ymm2, %ymm6, %ymm6+ vpmuldq %ymm2, %ymm7, %ymm7+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm2, %ymm9, %ymm9+ vpmuldq %ymm0, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpmuldq %ymm0, %ymm15, %ymm15+ vpsubd %ymm12, %ymm6, %ymm6+ vpsubd %ymm13, %ymm7, %ymm7+ vpsubd %ymm14, %ymm8, %ymm8+ vpsubd %ymm15, %ymm9, %ymm9+ vmovshdup %ymm6, %ymm6 # ymm6 = ymm6[1,1,3,3,5,5,7,7]+ vmovshdup %ymm7, %ymm7 # ymm7 = ymm7[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm8, %ymm6, %ymm6 # ymm6 = ymm6[0],ymm8[1],ymm6[2],ymm8[3],ymm6[4],ymm8[5],ymm6[6],ymm8[7]+ vpblendd $0xaa, %ymm9, %ymm7, %ymm7 # ymm7 = ymm7[0],ymm9[1],ymm7[2],ymm9[3],ymm7[4],ymm9[5],ymm7[6],ymm9[7]+ vmovdqa %ymm4, 0x20(%rdi)+ vmovdqa %ymm5, 0xa0(%rdi)+ vmovdqa %ymm6, 0x120(%rdi)+ vmovdqa %ymm7, 0x1a0(%rdi)+ vmovdqa 0x40(%rdi), %ymm4+ vmovdqa 0xc0(%rdi), %ymm5+ vmovdqa 0x140(%rdi), %ymm6+ vmovdqa 0x1c0(%rdi), %ymm7+ vmovdqa 0x240(%rdi), %ymm8+ vmovdqa 0x2c0(%rdi), %ymm9+ vmovdqa 0x340(%rdi), %ymm10+ vmovdqa 0x3c0(%rdi), %ymm11+ vpbroadcastd 0x8c(%rsi), %ymm1+ vpbroadcastd 0x52c(%rsi), %ymm2+ vpsubd %ymm4, %ymm6, %ymm12+ vpaddd %ymm6, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm6 # ymm6 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm6, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm6, %ymm6+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm6, %ymm6+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm6, %ymm12, %ymm6 # ymm6 = ymm12[0],ymm6[1],ymm12[2],ymm6[3],ymm12[4],ymm6[5],ymm12[6],ymm6[7]+ vpsubd %ymm5, %ymm7, %ymm12+ vpaddd %ymm7, %ymm5, %ymm5+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm7 # ymm7 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm7, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm7, %ymm7+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm7, %ymm7+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm7, %ymm12, %ymm7 # ymm7 = ymm12[0],ymm7[1],ymm12[2],ymm7[3],ymm12[4],ymm7[5],ymm12[6],ymm7[7]+ vpbroadcastd 0x88(%rsi), %ymm1+ vpbroadcastd 0x528(%rsi), %ymm2+ vpsubd %ymm8, %ymm10, %ymm12+ vpaddd %ymm10, %ymm8, %ymm8+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm10 # ymm10 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm10, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm10, %ymm10+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm10, %ymm10+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm10, %ymm12, %ymm10 # ymm10 = ymm12[0],ymm10[1],ymm12[2],ymm10[3],ymm12[4],ymm10[5],ymm12[6],ymm10[7]+ vpsubd %ymm9, %ymm11, %ymm12+ vpaddd %ymm11, %ymm9, %ymm9+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vpbroadcastd 0x80(%rsi), %ymm1+ vpbroadcastd 0x520(%rsi), %ymm2+ vpsubd %ymm4, %ymm8, %ymm12+ vpaddd %ymm4, %ymm8, %ymm4+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm8 # ymm8 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm8, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm8, %ymm8+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm8, %ymm12, %ymm8 # ymm8 = ymm12[0],ymm8[1],ymm12[2],ymm8[3],ymm12[4],ymm8[5],ymm12[6],ymm8[7]+ vpsubd %ymm5, %ymm9, %ymm12+ vpaddd %ymm5, %ymm9, %ymm5+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm9 # ymm9 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm9, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm9, %ymm9+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm9, %ymm9+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm9, %ymm12, %ymm9 # ymm9 = ymm12[0],ymm9[1],ymm12[2],ymm9[3],ymm12[4],ymm9[5],ymm12[6],ymm9[7]+ vpsubd %ymm6, %ymm10, %ymm12+ vpaddd %ymm6, %ymm10, %ymm6+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm10 # ymm10 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm10, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm10, %ymm10+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm10, %ymm10+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm10, %ymm12, %ymm10 # ymm10 = ymm12[0],ymm10[1],ymm12[2],ymm10[3],ymm12[4],ymm10[5],ymm12[6],ymm10[7]+ vpsubd %ymm7, %ymm11, %ymm12+ vpaddd %ymm7, %ymm11, %ymm7+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vmovdqa %ymm8, 0x240(%rdi)+ vmovdqa %ymm9, 0x2c0(%rdi)+ vmovdqa %ymm10, 0x340(%rdi)+ vmovdqa %ymm11, 0x3c0(%rdi)+ vmovdqa 0x40(%rsi), %ymm1+ vmovdqa 0x60(%rsi), %ymm2+ vpmuldq %ymm1, %ymm4, %ymm12+ vpmuldq %ymm1, %ymm5, %ymm13+ vmovshdup %ymm4, %ymm8 # ymm8 = ymm4[1,1,3,3,5,5,7,7]+ vmovshdup %ymm5, %ymm9 # ymm9 = ymm5[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm8, %ymm14+ vpmuldq %ymm1, %ymm9, %ymm15+ vpmuldq %ymm2, %ymm4, %ymm4+ vpmuldq %ymm2, %ymm5, %ymm5+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm2, %ymm9, %ymm9+ vpmuldq %ymm0, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpmuldq %ymm0, %ymm15, %ymm15+ vpsubd %ymm12, %ymm4, %ymm4+ vpsubd %ymm13, %ymm5, %ymm5+ vpsubd %ymm14, %ymm8, %ymm8+ vpsubd %ymm15, %ymm9, %ymm9+ vmovshdup %ymm4, %ymm4 # ymm4 = ymm4[1,1,3,3,5,5,7,7]+ vmovshdup %ymm5, %ymm5 # ymm5 = ymm5[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm8, %ymm4, %ymm4 # ymm4 = ymm4[0],ymm8[1],ymm4[2],ymm8[3],ymm4[4],ymm8[5],ymm4[6],ymm8[7]+ vpblendd $0xaa, %ymm9, %ymm5, %ymm5 # ymm5 = ymm5[0],ymm9[1],ymm5[2],ymm9[3],ymm5[4],ymm9[5],ymm5[6],ymm9[7]+ vpmuldq %ymm1, %ymm6, %ymm12+ vpmuldq %ymm1, %ymm7, %ymm13+ vmovshdup %ymm6, %ymm8 # ymm8 = ymm6[1,1,3,3,5,5,7,7]+ vmovshdup %ymm7, %ymm9 # ymm9 = ymm7[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm8, %ymm14+ vpmuldq %ymm1, %ymm9, %ymm15+ vpmuldq %ymm2, %ymm6, %ymm6+ vpmuldq %ymm2, %ymm7, %ymm7+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm2, %ymm9, %ymm9+ vpmuldq %ymm0, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpmuldq %ymm0, %ymm15, %ymm15+ vpsubd %ymm12, %ymm6, %ymm6+ vpsubd %ymm13, %ymm7, %ymm7+ vpsubd %ymm14, %ymm8, %ymm8+ vpsubd %ymm15, %ymm9, %ymm9+ vmovshdup %ymm6, %ymm6 # ymm6 = ymm6[1,1,3,3,5,5,7,7]+ vmovshdup %ymm7, %ymm7 # ymm7 = ymm7[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm8, %ymm6, %ymm6 # ymm6 = ymm6[0],ymm8[1],ymm6[2],ymm8[3],ymm6[4],ymm8[5],ymm6[6],ymm8[7]+ vpblendd $0xaa, %ymm9, %ymm7, %ymm7 # ymm7 = ymm7[0],ymm9[1],ymm7[2],ymm9[3],ymm7[4],ymm9[5],ymm7[6],ymm9[7]+ vmovdqa %ymm4, 0x40(%rdi)+ vmovdqa %ymm5, 0xc0(%rdi)+ vmovdqa %ymm6, 0x140(%rdi)+ vmovdqa %ymm7, 0x1c0(%rdi)+ vmovdqa 0x60(%rdi), %ymm4+ vmovdqa 0xe0(%rdi), %ymm5+ vmovdqa 0x160(%rdi), %ymm6+ vmovdqa 0x1e0(%rdi), %ymm7+ vmovdqa 0x260(%rdi), %ymm8+ vmovdqa 0x2e0(%rdi), %ymm9+ vmovdqa 0x360(%rdi), %ymm10+ vmovdqa 0x3e0(%rdi), %ymm11+ vpbroadcastd 0x8c(%rsi), %ymm1+ vpbroadcastd 0x52c(%rsi), %ymm2+ vpsubd %ymm4, %ymm6, %ymm12+ vpaddd %ymm6, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm6 # ymm6 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm6, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm6, %ymm6+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm6, %ymm6+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm6, %ymm12, %ymm6 # ymm6 = ymm12[0],ymm6[1],ymm12[2],ymm6[3],ymm12[4],ymm6[5],ymm12[6],ymm6[7]+ vpsubd %ymm5, %ymm7, %ymm12+ vpaddd %ymm7, %ymm5, %ymm5+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm7 # ymm7 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm7, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm7, %ymm7+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm7, %ymm7+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm7, %ymm12, %ymm7 # ymm7 = ymm12[0],ymm7[1],ymm12[2],ymm7[3],ymm12[4],ymm7[5],ymm12[6],ymm7[7]+ vpbroadcastd 0x88(%rsi), %ymm1+ vpbroadcastd 0x528(%rsi), %ymm2+ vpsubd %ymm8, %ymm10, %ymm12+ vpaddd %ymm10, %ymm8, %ymm8+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm10 # ymm10 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm10, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm10, %ymm10+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm10, %ymm10+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm10, %ymm12, %ymm10 # ymm10 = ymm12[0],ymm10[1],ymm12[2],ymm10[3],ymm12[4],ymm10[5],ymm12[6],ymm10[7]+ vpsubd %ymm9, %ymm11, %ymm12+ vpaddd %ymm11, %ymm9, %ymm9+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vpbroadcastd 0x80(%rsi), %ymm1+ vpbroadcastd 0x520(%rsi), %ymm2+ vpsubd %ymm4, %ymm8, %ymm12+ vpaddd %ymm4, %ymm8, %ymm4+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm8 # ymm8 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm8, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm8, %ymm8+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm8, %ymm12, %ymm8 # ymm8 = ymm12[0],ymm8[1],ymm12[2],ymm8[3],ymm12[4],ymm8[5],ymm12[6],ymm8[7]+ vpsubd %ymm5, %ymm9, %ymm12+ vpaddd %ymm5, %ymm9, %ymm5+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm9 # ymm9 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm9, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm9, %ymm9+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm9, %ymm9+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm9, %ymm12, %ymm9 # ymm9 = ymm12[0],ymm9[1],ymm12[2],ymm9[3],ymm12[4],ymm9[5],ymm12[6],ymm9[7]+ vpsubd %ymm6, %ymm10, %ymm12+ vpaddd %ymm6, %ymm10, %ymm6+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm10 # ymm10 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm10, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm10, %ymm10+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm10, %ymm10+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm10, %ymm12, %ymm10 # ymm10 = ymm12[0],ymm10[1],ymm12[2],ymm10[3],ymm12[4],ymm10[5],ymm12[6],ymm10[7]+ vpsubd %ymm7, %ymm11, %ymm12+ vpaddd %ymm7, %ymm11, %ymm7+ vpmuldq %ymm1, %ymm12, %ymm13+ vmovshdup %ymm12, %ymm11 # ymm11 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm11, %ymm14+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpsubd %ymm13, %ymm12, %ymm12+ vpsubd %ymm14, %ymm11, %ymm11+ vmovshdup %ymm12, %ymm12 # ymm12 = ymm12[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm11, %ymm12, %ymm11 # ymm11 = ymm12[0],ymm11[1],ymm12[2],ymm11[3],ymm12[4],ymm11[5],ymm12[6],ymm11[7]+ vmovdqa %ymm8, 0x260(%rdi)+ vmovdqa %ymm9, 0x2e0(%rdi)+ vmovdqa %ymm10, 0x360(%rdi)+ vmovdqa %ymm11, 0x3e0(%rdi)+ vmovdqa 0x40(%rsi), %ymm1+ vmovdqa 0x60(%rsi), %ymm2+ vpmuldq %ymm1, %ymm4, %ymm12+ vpmuldq %ymm1, %ymm5, %ymm13+ vmovshdup %ymm4, %ymm8 # ymm8 = ymm4[1,1,3,3,5,5,7,7]+ vmovshdup %ymm5, %ymm9 # ymm9 = ymm5[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm8, %ymm14+ vpmuldq %ymm1, %ymm9, %ymm15+ vpmuldq %ymm2, %ymm4, %ymm4+ vpmuldq %ymm2, %ymm5, %ymm5+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm2, %ymm9, %ymm9+ vpmuldq %ymm0, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpmuldq %ymm0, %ymm15, %ymm15+ vpsubd %ymm12, %ymm4, %ymm4+ vpsubd %ymm13, %ymm5, %ymm5+ vpsubd %ymm14, %ymm8, %ymm8+ vpsubd %ymm15, %ymm9, %ymm9+ vmovshdup %ymm4, %ymm4 # ymm4 = ymm4[1,1,3,3,5,5,7,7]+ vmovshdup %ymm5, %ymm5 # ymm5 = ymm5[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm8, %ymm4, %ymm4 # ymm4 = ymm4[0],ymm8[1],ymm4[2],ymm8[3],ymm4[4],ymm8[5],ymm4[6],ymm8[7]+ vpblendd $0xaa, %ymm9, %ymm5, %ymm5 # ymm5 = ymm5[0],ymm9[1],ymm5[2],ymm9[3],ymm5[4],ymm9[5],ymm5[6],ymm9[7]+ vpmuldq %ymm1, %ymm6, %ymm12+ vpmuldq %ymm1, %ymm7, %ymm13+ vmovshdup %ymm6, %ymm8 # ymm8 = ymm6[1,1,3,3,5,5,7,7]+ vmovshdup %ymm7, %ymm9 # ymm9 = ymm7[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm8, %ymm14+ vpmuldq %ymm1, %ymm9, %ymm15+ vpmuldq %ymm2, %ymm6, %ymm6+ vpmuldq %ymm2, %ymm7, %ymm7+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm2, %ymm9, %ymm9+ vpmuldq %ymm0, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vpmuldq %ymm0, %ymm15, %ymm15+ vpsubd %ymm12, %ymm6, %ymm6+ vpsubd %ymm13, %ymm7, %ymm7+ vpsubd %ymm14, %ymm8, %ymm8+ vpsubd %ymm15, %ymm9, %ymm9+ vmovshdup %ymm6, %ymm6 # ymm6 = ymm6[1,1,3,3,5,5,7,7]+ vmovshdup %ymm7, %ymm7 # ymm7 = ymm7[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm8, %ymm6, %ymm6 # ymm6 = ymm6[0],ymm8[1],ymm6[2],ymm8[3],ymm6[4],ymm8[5],ymm6[6],ymm8[7]+ vpblendd $0xaa, %ymm9, %ymm7, %ymm7 # ymm7 = ymm7[0],ymm9[1],ymm7[2],ymm9[3],ymm7[4],ymm9[5],ymm7[6],ymm9[7]+ vmovdqa %ymm4, 0x60(%rdi)+ vmovdqa %ymm5, 0xe0(%rdi)+ vmovdqa %ymm6, 0x160(%rdi)+ vmovdqa %ymm7, 0x1e0(%rdi)+ retq+ .cfi_endproc++MLD_ASM_FN_SIZE(invntt_avx2_asm)++#endif /* MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_MULTILEVEL_NO_SHARED \+ */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
@@ -0,0 +1,2405 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [REF_AVX2]+ * CRYSTALS-Dilithium optimized AVX2 implementation+ * Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/dilithium/tree/master/avx2+ */++ /*+ * This file is derived from the public domain+ * AVX2 Dilithium implementation @[REF_AVX2].+ */++/*yaml+ Name: ntt_avx2_asm+ Description: x86_64 AVX2 forward NTT+ Signature: void mld_ntt_avx2_asm(int32_t *r, const int32_t *qdata)+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 1024+ permissions: read/write+ c_parameter: int32_t *r+ description: Input/output polynomial (256 x int32_t)+ rsi:+ type: buffer+ size_bytes: 2496+ permissions: read-only+ c_parameter: const int32_t *qdata+ description: Precomputed constants (624 x int32_t)+*/++#include "../../../common.h"+#if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/x86_64/src/mldsa_ntt_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(ntt_avx2_asm)+MLD_ASM_FN_SYMBOL(ntt_avx2_asm)++ .cfi_startproc+ vmovdqa (%rsi), %ymm0+ vpbroadcastd 0x84(%rsi), %ymm1+ vpbroadcastd 0x524(%rsi), %ymm2+ vmovdqa (%rdi), %ymm4+ vmovdqa 0x80(%rdi), %ymm5+ vmovdqa 0x100(%rdi), %ymm6+ vmovdqa 0x180(%rdi), %ymm7+ vmovdqa 0x200(%rdi), %ymm8+ vmovdqa 0x280(%rdi), %ymm9+ vmovdqa 0x300(%rdi), %ymm10+ vmovdqa 0x380(%rdi), %ymm11+ vpmuldq %ymm1, %ymm8, %ymm13+ vmovshdup %ymm8, %ymm12 # ymm12 = ymm8[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm8, %ymm8 # ymm8 = ymm8[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm8, %ymm8 # ymm8 = ymm8[0],ymm12[1],ymm8[2],ymm12[3],ymm8[4],ymm12[5],ymm8[6],ymm12[7]+ vpsubd %ymm8, %ymm4, %ymm12+ vpaddd %ymm4, %ymm8, %ymm4+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm8+ vpsubd %ymm13, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm9, %ymm13+ vmovshdup %ymm9, %ymm12 # ymm12 = ymm9[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm9, %ymm9+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm9, %ymm9 # ymm9 = ymm9[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm9, %ymm9 # ymm9 = ymm9[0],ymm12[1],ymm9[2],ymm12[3],ymm9[4],ymm12[5],ymm9[6],ymm12[7]+ vpsubd %ymm9, %ymm5, %ymm12+ vpaddd %ymm5, %ymm9, %ymm5+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm9+ vpsubd %ymm13, %ymm5, %ymm5+ vpmuldq %ymm1, %ymm10, %ymm13+ vmovshdup %ymm10, %ymm12 # ymm12 = ymm10[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm10, %ymm10+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm10, %ymm10 # ymm10 = ymm10[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm10, %ymm10 # ymm10 = ymm10[0],ymm12[1],ymm10[2],ymm12[3],ymm10[4],ymm12[5],ymm10[6],ymm12[7]+ vpsubd %ymm10, %ymm6, %ymm12+ vpaddd %ymm6, %ymm10, %ymm6+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm10+ vpsubd %ymm13, %ymm6, %ymm6+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm7, %ymm12+ vpaddd %ymm7, %ymm11, %ymm7+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm7, %ymm7+ vpbroadcastd 0x88(%rsi), %ymm1+ vpbroadcastd 0x528(%rsi), %ymm2+ vpmuldq %ymm1, %ymm6, %ymm13+ vmovshdup %ymm6, %ymm12 # ymm12 = ymm6[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm6, %ymm6+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm6, %ymm6 # ymm6 = ymm6[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm6, %ymm6 # ymm6 = ymm6[0],ymm12[1],ymm6[2],ymm12[3],ymm6[4],ymm12[5],ymm6[6],ymm12[7]+ vpsubd %ymm6, %ymm4, %ymm12+ vpaddd %ymm6, %ymm4, %ymm4+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm6+ vpsubd %ymm13, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm7, %ymm13+ vmovshdup %ymm7, %ymm12 # ymm12 = ymm7[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm7, %ymm7+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm7, %ymm7 # ymm7 = ymm7[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm7, %ymm7 # ymm7 = ymm7[0],ymm12[1],ymm7[2],ymm12[3],ymm7[4],ymm12[5],ymm7[6],ymm12[7]+ vpsubd %ymm7, %ymm5, %ymm12+ vpaddd %ymm7, %ymm5, %ymm5+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm7+ vpsubd %ymm13, %ymm5, %ymm5+ vpbroadcastd 0x8c(%rsi), %ymm1+ vpbroadcastd 0x52c(%rsi), %ymm2+ vpmuldq %ymm1, %ymm10, %ymm13+ vmovshdup %ymm10, %ymm12 # ymm12 = ymm10[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm10, %ymm10+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm10, %ymm10 # ymm10 = ymm10[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm10, %ymm10 # ymm10 = ymm10[0],ymm12[1],ymm10[2],ymm12[3],ymm10[4],ymm12[5],ymm10[6],ymm12[7]+ vpsubd %ymm10, %ymm8, %ymm12+ vpaddd %ymm10, %ymm8, %ymm8+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm10+ vpsubd %ymm13, %ymm8, %ymm8+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm9, %ymm12+ vpaddd %ymm11, %ymm9, %ymm9+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm9, %ymm9+ vmovdqa %ymm4, (%rdi)+ vmovdqa %ymm5, 0x80(%rdi)+ vmovdqa %ymm6, 0x100(%rdi)+ vmovdqa %ymm7, 0x180(%rdi)+ vmovdqa %ymm8, 0x200(%rdi)+ vmovdqa %ymm9, 0x280(%rdi)+ vmovdqa %ymm10, 0x300(%rdi)+ vmovdqa %ymm11, 0x380(%rdi)+ vpbroadcastd 0x84(%rsi), %ymm1+ vpbroadcastd 0x524(%rsi), %ymm2+ vmovdqa 0x20(%rdi), %ymm4+ vmovdqa 0xa0(%rdi), %ymm5+ vmovdqa 0x120(%rdi), %ymm6+ vmovdqa 0x1a0(%rdi), %ymm7+ vmovdqa 0x220(%rdi), %ymm8+ vmovdqa 0x2a0(%rdi), %ymm9+ vmovdqa 0x320(%rdi), %ymm10+ vmovdqa 0x3a0(%rdi), %ymm11+ vpmuldq %ymm1, %ymm8, %ymm13+ vmovshdup %ymm8, %ymm12 # ymm12 = ymm8[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm8, %ymm8 # ymm8 = ymm8[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm8, %ymm8 # ymm8 = ymm8[0],ymm12[1],ymm8[2],ymm12[3],ymm8[4],ymm12[5],ymm8[6],ymm12[7]+ vpsubd %ymm8, %ymm4, %ymm12+ vpaddd %ymm4, %ymm8, %ymm4+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm8+ vpsubd %ymm13, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm9, %ymm13+ vmovshdup %ymm9, %ymm12 # ymm12 = ymm9[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm9, %ymm9+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm9, %ymm9 # ymm9 = ymm9[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm9, %ymm9 # ymm9 = ymm9[0],ymm12[1],ymm9[2],ymm12[3],ymm9[4],ymm12[5],ymm9[6],ymm12[7]+ vpsubd %ymm9, %ymm5, %ymm12+ vpaddd %ymm5, %ymm9, %ymm5+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm9+ vpsubd %ymm13, %ymm5, %ymm5+ vpmuldq %ymm1, %ymm10, %ymm13+ vmovshdup %ymm10, %ymm12 # ymm12 = ymm10[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm10, %ymm10+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm10, %ymm10 # ymm10 = ymm10[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm10, %ymm10 # ymm10 = ymm10[0],ymm12[1],ymm10[2],ymm12[3],ymm10[4],ymm12[5],ymm10[6],ymm12[7]+ vpsubd %ymm10, %ymm6, %ymm12+ vpaddd %ymm6, %ymm10, %ymm6+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm10+ vpsubd %ymm13, %ymm6, %ymm6+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm7, %ymm12+ vpaddd %ymm7, %ymm11, %ymm7+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm7, %ymm7+ vpbroadcastd 0x88(%rsi), %ymm1+ vpbroadcastd 0x528(%rsi), %ymm2+ vpmuldq %ymm1, %ymm6, %ymm13+ vmovshdup %ymm6, %ymm12 # ymm12 = ymm6[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm6, %ymm6+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm6, %ymm6 # ymm6 = ymm6[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm6, %ymm6 # ymm6 = ymm6[0],ymm12[1],ymm6[2],ymm12[3],ymm6[4],ymm12[5],ymm6[6],ymm12[7]+ vpsubd %ymm6, %ymm4, %ymm12+ vpaddd %ymm6, %ymm4, %ymm4+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm6+ vpsubd %ymm13, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm7, %ymm13+ vmovshdup %ymm7, %ymm12 # ymm12 = ymm7[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm7, %ymm7+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm7, %ymm7 # ymm7 = ymm7[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm7, %ymm7 # ymm7 = ymm7[0],ymm12[1],ymm7[2],ymm12[3],ymm7[4],ymm12[5],ymm7[6],ymm12[7]+ vpsubd %ymm7, %ymm5, %ymm12+ vpaddd %ymm7, %ymm5, %ymm5+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm7+ vpsubd %ymm13, %ymm5, %ymm5+ vpbroadcastd 0x8c(%rsi), %ymm1+ vpbroadcastd 0x52c(%rsi), %ymm2+ vpmuldq %ymm1, %ymm10, %ymm13+ vmovshdup %ymm10, %ymm12 # ymm12 = ymm10[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm10, %ymm10+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm10, %ymm10 # ymm10 = ymm10[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm10, %ymm10 # ymm10 = ymm10[0],ymm12[1],ymm10[2],ymm12[3],ymm10[4],ymm12[5],ymm10[6],ymm12[7]+ vpsubd %ymm10, %ymm8, %ymm12+ vpaddd %ymm10, %ymm8, %ymm8+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm10+ vpsubd %ymm13, %ymm8, %ymm8+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm9, %ymm12+ vpaddd %ymm11, %ymm9, %ymm9+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm9, %ymm9+ vmovdqa %ymm4, 0x20(%rdi)+ vmovdqa %ymm5, 0xa0(%rdi)+ vmovdqa %ymm6, 0x120(%rdi)+ vmovdqa %ymm7, 0x1a0(%rdi)+ vmovdqa %ymm8, 0x220(%rdi)+ vmovdqa %ymm9, 0x2a0(%rdi)+ vmovdqa %ymm10, 0x320(%rdi)+ vmovdqa %ymm11, 0x3a0(%rdi)+ vpbroadcastd 0x84(%rsi), %ymm1+ vpbroadcastd 0x524(%rsi), %ymm2+ vmovdqa 0x40(%rdi), %ymm4+ vmovdqa 0xc0(%rdi), %ymm5+ vmovdqa 0x140(%rdi), %ymm6+ vmovdqa 0x1c0(%rdi), %ymm7+ vmovdqa 0x240(%rdi), %ymm8+ vmovdqa 0x2c0(%rdi), %ymm9+ vmovdqa 0x340(%rdi), %ymm10+ vmovdqa 0x3c0(%rdi), %ymm11+ vpmuldq %ymm1, %ymm8, %ymm13+ vmovshdup %ymm8, %ymm12 # ymm12 = ymm8[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm8, %ymm8 # ymm8 = ymm8[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm8, %ymm8 # ymm8 = ymm8[0],ymm12[1],ymm8[2],ymm12[3],ymm8[4],ymm12[5],ymm8[6],ymm12[7]+ vpsubd %ymm8, %ymm4, %ymm12+ vpaddd %ymm4, %ymm8, %ymm4+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm8+ vpsubd %ymm13, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm9, %ymm13+ vmovshdup %ymm9, %ymm12 # ymm12 = ymm9[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm9, %ymm9+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm9, %ymm9 # ymm9 = ymm9[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm9, %ymm9 # ymm9 = ymm9[0],ymm12[1],ymm9[2],ymm12[3],ymm9[4],ymm12[5],ymm9[6],ymm12[7]+ vpsubd %ymm9, %ymm5, %ymm12+ vpaddd %ymm5, %ymm9, %ymm5+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm9+ vpsubd %ymm13, %ymm5, %ymm5+ vpmuldq %ymm1, %ymm10, %ymm13+ vmovshdup %ymm10, %ymm12 # ymm12 = ymm10[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm10, %ymm10+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm10, %ymm10 # ymm10 = ymm10[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm10, %ymm10 # ymm10 = ymm10[0],ymm12[1],ymm10[2],ymm12[3],ymm10[4],ymm12[5],ymm10[6],ymm12[7]+ vpsubd %ymm10, %ymm6, %ymm12+ vpaddd %ymm6, %ymm10, %ymm6+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm10+ vpsubd %ymm13, %ymm6, %ymm6+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm7, %ymm12+ vpaddd %ymm7, %ymm11, %ymm7+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm7, %ymm7+ vpbroadcastd 0x88(%rsi), %ymm1+ vpbroadcastd 0x528(%rsi), %ymm2+ vpmuldq %ymm1, %ymm6, %ymm13+ vmovshdup %ymm6, %ymm12 # ymm12 = ymm6[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm6, %ymm6+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm6, %ymm6 # ymm6 = ymm6[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm6, %ymm6 # ymm6 = ymm6[0],ymm12[1],ymm6[2],ymm12[3],ymm6[4],ymm12[5],ymm6[6],ymm12[7]+ vpsubd %ymm6, %ymm4, %ymm12+ vpaddd %ymm6, %ymm4, %ymm4+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm6+ vpsubd %ymm13, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm7, %ymm13+ vmovshdup %ymm7, %ymm12 # ymm12 = ymm7[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm7, %ymm7+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm7, %ymm7 # ymm7 = ymm7[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm7, %ymm7 # ymm7 = ymm7[0],ymm12[1],ymm7[2],ymm12[3],ymm7[4],ymm12[5],ymm7[6],ymm12[7]+ vpsubd %ymm7, %ymm5, %ymm12+ vpaddd %ymm7, %ymm5, %ymm5+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm7+ vpsubd %ymm13, %ymm5, %ymm5+ vpbroadcastd 0x8c(%rsi), %ymm1+ vpbroadcastd 0x52c(%rsi), %ymm2+ vpmuldq %ymm1, %ymm10, %ymm13+ vmovshdup %ymm10, %ymm12 # ymm12 = ymm10[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm10, %ymm10+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm10, %ymm10 # ymm10 = ymm10[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm10, %ymm10 # ymm10 = ymm10[0],ymm12[1],ymm10[2],ymm12[3],ymm10[4],ymm12[5],ymm10[6],ymm12[7]+ vpsubd %ymm10, %ymm8, %ymm12+ vpaddd %ymm10, %ymm8, %ymm8+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm10+ vpsubd %ymm13, %ymm8, %ymm8+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm9, %ymm12+ vpaddd %ymm11, %ymm9, %ymm9+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm9, %ymm9+ vmovdqa %ymm4, 0x40(%rdi)+ vmovdqa %ymm5, 0xc0(%rdi)+ vmovdqa %ymm6, 0x140(%rdi)+ vmovdqa %ymm7, 0x1c0(%rdi)+ vmovdqa %ymm8, 0x240(%rdi)+ vmovdqa %ymm9, 0x2c0(%rdi)+ vmovdqa %ymm10, 0x340(%rdi)+ vmovdqa %ymm11, 0x3c0(%rdi)+ vpbroadcastd 0x84(%rsi), %ymm1+ vpbroadcastd 0x524(%rsi), %ymm2+ vmovdqa 0x60(%rdi), %ymm4+ vmovdqa 0xe0(%rdi), %ymm5+ vmovdqa 0x160(%rdi), %ymm6+ vmovdqa 0x1e0(%rdi), %ymm7+ vmovdqa 0x260(%rdi), %ymm8+ vmovdqa 0x2e0(%rdi), %ymm9+ vmovdqa 0x360(%rdi), %ymm10+ vmovdqa 0x3e0(%rdi), %ymm11+ vpmuldq %ymm1, %ymm8, %ymm13+ vmovshdup %ymm8, %ymm12 # ymm12 = ymm8[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm8, %ymm8 # ymm8 = ymm8[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm8, %ymm8 # ymm8 = ymm8[0],ymm12[1],ymm8[2],ymm12[3],ymm8[4],ymm12[5],ymm8[6],ymm12[7]+ vpsubd %ymm8, %ymm4, %ymm12+ vpaddd %ymm4, %ymm8, %ymm4+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm8+ vpsubd %ymm13, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm9, %ymm13+ vmovshdup %ymm9, %ymm12 # ymm12 = ymm9[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm9, %ymm9+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm9, %ymm9 # ymm9 = ymm9[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm9, %ymm9 # ymm9 = ymm9[0],ymm12[1],ymm9[2],ymm12[3],ymm9[4],ymm12[5],ymm9[6],ymm12[7]+ vpsubd %ymm9, %ymm5, %ymm12+ vpaddd %ymm5, %ymm9, %ymm5+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm9+ vpsubd %ymm13, %ymm5, %ymm5+ vpmuldq %ymm1, %ymm10, %ymm13+ vmovshdup %ymm10, %ymm12 # ymm12 = ymm10[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm10, %ymm10+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm10, %ymm10 # ymm10 = ymm10[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm10, %ymm10 # ymm10 = ymm10[0],ymm12[1],ymm10[2],ymm12[3],ymm10[4],ymm12[5],ymm10[6],ymm12[7]+ vpsubd %ymm10, %ymm6, %ymm12+ vpaddd %ymm6, %ymm10, %ymm6+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm10+ vpsubd %ymm13, %ymm6, %ymm6+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm7, %ymm12+ vpaddd %ymm7, %ymm11, %ymm7+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm7, %ymm7+ vpbroadcastd 0x88(%rsi), %ymm1+ vpbroadcastd 0x528(%rsi), %ymm2+ vpmuldq %ymm1, %ymm6, %ymm13+ vmovshdup %ymm6, %ymm12 # ymm12 = ymm6[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm6, %ymm6+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm6, %ymm6 # ymm6 = ymm6[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm6, %ymm6 # ymm6 = ymm6[0],ymm12[1],ymm6[2],ymm12[3],ymm6[4],ymm12[5],ymm6[6],ymm12[7]+ vpsubd %ymm6, %ymm4, %ymm12+ vpaddd %ymm6, %ymm4, %ymm4+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm6+ vpsubd %ymm13, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm7, %ymm13+ vmovshdup %ymm7, %ymm12 # ymm12 = ymm7[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm7, %ymm7+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm7, %ymm7 # ymm7 = ymm7[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm7, %ymm7 # ymm7 = ymm7[0],ymm12[1],ymm7[2],ymm12[3],ymm7[4],ymm12[5],ymm7[6],ymm12[7]+ vpsubd %ymm7, %ymm5, %ymm12+ vpaddd %ymm7, %ymm5, %ymm5+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm7+ vpsubd %ymm13, %ymm5, %ymm5+ vpbroadcastd 0x8c(%rsi), %ymm1+ vpbroadcastd 0x52c(%rsi), %ymm2+ vpmuldq %ymm1, %ymm10, %ymm13+ vmovshdup %ymm10, %ymm12 # ymm12 = ymm10[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm10, %ymm10+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm10, %ymm10 # ymm10 = ymm10[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm10, %ymm10 # ymm10 = ymm10[0],ymm12[1],ymm10[2],ymm12[3],ymm10[4],ymm12[5],ymm10[6],ymm12[7]+ vpsubd %ymm10, %ymm8, %ymm12+ vpaddd %ymm10, %ymm8, %ymm8+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm10+ vpsubd %ymm13, %ymm8, %ymm8+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm9, %ymm12+ vpaddd %ymm11, %ymm9, %ymm9+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm9, %ymm9+ vmovdqa %ymm4, 0x60(%rdi)+ vmovdqa %ymm5, 0xe0(%rdi)+ vmovdqa %ymm6, 0x160(%rdi)+ vmovdqa %ymm7, 0x1e0(%rdi)+ vmovdqa %ymm8, 0x260(%rdi)+ vmovdqa %ymm9, 0x2e0(%rdi)+ vmovdqa %ymm10, 0x360(%rdi)+ vmovdqa %ymm11, 0x3e0(%rdi)+ vmovdqa (%rdi), %ymm4+ vmovdqa 0x20(%rdi), %ymm5+ vmovdqa 0x40(%rdi), %ymm6+ vmovdqa 0x60(%rdi), %ymm7+ vmovdqa 0x80(%rdi), %ymm8+ vmovdqa 0xa0(%rdi), %ymm9+ vmovdqa 0xc0(%rdi), %ymm10+ vmovdqa 0xe0(%rdi), %ymm11+ vpbroadcastd 0x90(%rsi), %ymm1+ vpbroadcastd 0x530(%rsi), %ymm2+ vpmuldq %ymm1, %ymm8, %ymm13+ vmovshdup %ymm8, %ymm12 # ymm12 = ymm8[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm8, %ymm8 # ymm8 = ymm8[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm8, %ymm8 # ymm8 = ymm8[0],ymm12[1],ymm8[2],ymm12[3],ymm8[4],ymm12[5],ymm8[6],ymm12[7]+ vpsubd %ymm8, %ymm4, %ymm12+ vpaddd %ymm4, %ymm8, %ymm4+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm8+ vpsubd %ymm13, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm9, %ymm13+ vmovshdup %ymm9, %ymm12 # ymm12 = ymm9[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm9, %ymm9+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm9, %ymm9 # ymm9 = ymm9[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm9, %ymm9 # ymm9 = ymm9[0],ymm12[1],ymm9[2],ymm12[3],ymm9[4],ymm12[5],ymm9[6],ymm12[7]+ vpsubd %ymm9, %ymm5, %ymm12+ vpaddd %ymm5, %ymm9, %ymm5+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm9+ vpsubd %ymm13, %ymm5, %ymm5+ vpmuldq %ymm1, %ymm10, %ymm13+ vmovshdup %ymm10, %ymm12 # ymm12 = ymm10[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm10, %ymm10+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm10, %ymm10 # ymm10 = ymm10[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm10, %ymm10 # ymm10 = ymm10[0],ymm12[1],ymm10[2],ymm12[3],ymm10[4],ymm12[5],ymm10[6],ymm12[7]+ vpsubd %ymm10, %ymm6, %ymm12+ vpaddd %ymm6, %ymm10, %ymm6+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm10+ vpsubd %ymm13, %ymm6, %ymm6+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm7, %ymm12+ vpaddd %ymm7, %ymm11, %ymm7+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm7, %ymm7+ vperm2i128 $0x20, %ymm8, %ymm4, %ymm3 # ymm3 = ymm4[0,1],ymm8[0,1]+ vperm2i128 $0x31, %ymm8, %ymm4, %ymm8 # ymm8 = ymm4[2,3],ymm8[2,3]+ vperm2i128 $0x20, %ymm9, %ymm5, %ymm4 # ymm4 = ymm5[0,1],ymm9[0,1]+ vperm2i128 $0x31, %ymm9, %ymm5, %ymm9 # ymm9 = ymm5[2,3],ymm9[2,3]+ vperm2i128 $0x20, %ymm10, %ymm6, %ymm5 # ymm5 = ymm6[0,1],ymm10[0,1]+ vperm2i128 $0x31, %ymm10, %ymm6, %ymm10 # ymm10 = ymm6[2,3],ymm10[2,3]+ vperm2i128 $0x20, %ymm11, %ymm7, %ymm6 # ymm6 = ymm7[0,1],ymm11[0,1]+ vperm2i128 $0x31, %ymm11, %ymm7, %ymm11 # ymm11 = ymm7[2,3],ymm11[2,3]+ vmovdqa 0xa0(%rsi), %ymm1+ vmovdqa 0x540(%rsi), %ymm2+ vpmuldq %ymm1, %ymm5, %ymm13+ vmovshdup %ymm5, %ymm12 # ymm12 = ymm5[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm5, %ymm5+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm5, %ymm5 # ymm5 = ymm5[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm5, %ymm5 # ymm5 = ymm5[0],ymm12[1],ymm5[2],ymm12[3],ymm5[4],ymm12[5],ymm5[6],ymm12[7]+ vpsubd %ymm5, %ymm3, %ymm12+ vpaddd %ymm5, %ymm3, %ymm3+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm5+ vpsubd %ymm13, %ymm3, %ymm3+ vpmuldq %ymm1, %ymm10, %ymm13+ vmovshdup %ymm10, %ymm12 # ymm12 = ymm10[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm10, %ymm10+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm10, %ymm10 # ymm10 = ymm10[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm10, %ymm10 # ymm10 = ymm10[0],ymm12[1],ymm10[2],ymm12[3],ymm10[4],ymm12[5],ymm10[6],ymm12[7]+ vpsubd %ymm10, %ymm8, %ymm12+ vpaddd %ymm10, %ymm8, %ymm8+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm10+ vpsubd %ymm13, %ymm8, %ymm8+ vpmuldq %ymm1, %ymm6, %ymm13+ vmovshdup %ymm6, %ymm12 # ymm12 = ymm6[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm6, %ymm6+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm6, %ymm6 # ymm6 = ymm6[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm6, %ymm6 # ymm6 = ymm6[0],ymm12[1],ymm6[2],ymm12[3],ymm6[4],ymm12[5],ymm6[6],ymm12[7]+ vpsubd %ymm6, %ymm4, %ymm12+ vpaddd %ymm6, %ymm4, %ymm4+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm6+ vpsubd %ymm13, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm9, %ymm12+ vpaddd %ymm11, %ymm9, %ymm9+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm9, %ymm9+ vpunpcklqdq %ymm5, %ymm3, %ymm7 # ymm7 = ymm3[0],ymm5[0],ymm3[2],ymm5[2]+ vpunpckhqdq %ymm5, %ymm3, %ymm5 # ymm5 = ymm3[1],ymm5[1],ymm3[3],ymm5[3]+ vpunpcklqdq %ymm10, %ymm8, %ymm3 # ymm3 = ymm8[0],ymm10[0],ymm8[2],ymm10[2]+ vpunpckhqdq %ymm10, %ymm8, %ymm10 # ymm10 = ymm8[1],ymm10[1],ymm8[3],ymm10[3]+ vpunpcklqdq %ymm6, %ymm4, %ymm8 # ymm8 = ymm4[0],ymm6[0],ymm4[2],ymm6[2]+ vpunpckhqdq %ymm6, %ymm4, %ymm6 # ymm6 = ymm4[1],ymm6[1],ymm4[3],ymm6[3]+ vpunpcklqdq %ymm11, %ymm9, %ymm4 # ymm4 = ymm9[0],ymm11[0],ymm9[2],ymm11[2]+ vpunpckhqdq %ymm11, %ymm9, %ymm11 # ymm11 = ymm9[1],ymm11[1],ymm9[3],ymm11[3]+ vmovdqa 0x120(%rsi), %ymm1+ vmovdqa 0x5c0(%rsi), %ymm2+ vpmuldq %ymm1, %ymm8, %ymm13+ vmovshdup %ymm8, %ymm12 # ymm12 = ymm8[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm8, %ymm8 # ymm8 = ymm8[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm8, %ymm8 # ymm8 = ymm8[0],ymm12[1],ymm8[2],ymm12[3],ymm8[4],ymm12[5],ymm8[6],ymm12[7]+ vpsubd %ymm8, %ymm7, %ymm12+ vpaddd %ymm7, %ymm8, %ymm7+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm8+ vpsubd %ymm13, %ymm7, %ymm7+ vpmuldq %ymm1, %ymm6, %ymm13+ vmovshdup %ymm6, %ymm12 # ymm12 = ymm6[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm6, %ymm6+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm6, %ymm6 # ymm6 = ymm6[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm6, %ymm6 # ymm6 = ymm6[0],ymm12[1],ymm6[2],ymm12[3],ymm6[4],ymm12[5],ymm6[6],ymm12[7]+ vpsubd %ymm6, %ymm5, %ymm12+ vpaddd %ymm6, %ymm5, %ymm5+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm6+ vpsubd %ymm13, %ymm5, %ymm5+ vpmuldq %ymm1, %ymm4, %ymm13+ vmovshdup %ymm4, %ymm12 # ymm12 = ymm4[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm4, %ymm4+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm4, %ymm4 # ymm4 = ymm4[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm4, %ymm4 # ymm4 = ymm4[0],ymm12[1],ymm4[2],ymm12[3],ymm4[4],ymm12[5],ymm4[6],ymm12[7]+ vpsubd %ymm4, %ymm3, %ymm12+ vpaddd %ymm4, %ymm3, %ymm3+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm4+ vpsubd %ymm13, %ymm3, %ymm3+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm10, %ymm12+ vpaddd %ymm11, %ymm10, %ymm10+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm10, %ymm10+ vmovsldup %ymm8, %ymm9 # ymm9 = ymm8[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm9, %ymm7, %ymm9 # ymm9 = ymm7[0],ymm9[1],ymm7[2],ymm9[3],ymm7[4],ymm9[5],ymm7[6],ymm9[7]+ vpsrlq $0x20, %ymm7, %ymm7+ vpblendd $0xaa, %ymm8, %ymm7, %ymm8 # ymm8 = ymm7[0],ymm8[1],ymm7[2],ymm8[3],ymm7[4],ymm8[5],ymm7[6],ymm8[7]+ vmovsldup %ymm6, %ymm7 # ymm7 = ymm6[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm7, %ymm5, %ymm7 # ymm7 = ymm5[0],ymm7[1],ymm5[2],ymm7[3],ymm5[4],ymm7[5],ymm5[6],ymm7[7]+ vpsrlq $0x20, %ymm5, %ymm5+ vpblendd $0xaa, %ymm6, %ymm5, %ymm6 # ymm6 = ymm5[0],ymm6[1],ymm5[2],ymm6[3],ymm5[4],ymm6[5],ymm5[6],ymm6[7]+ vmovsldup %ymm4, %ymm5 # ymm5 = ymm4[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm5, %ymm3, %ymm5 # ymm5 = ymm3[0],ymm5[1],ymm3[2],ymm5[3],ymm3[4],ymm5[5],ymm3[6],ymm5[7]+ vpsrlq $0x20, %ymm3, %ymm3+ vpblendd $0xaa, %ymm4, %ymm3, %ymm4 # ymm4 = ymm3[0],ymm4[1],ymm3[2],ymm4[3],ymm3[4],ymm4[5],ymm3[6],ymm4[7]+ vmovsldup %ymm11, %ymm3 # ymm3 = ymm11[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm3, %ymm10, %ymm3 # ymm3 = ymm10[0],ymm3[1],ymm10[2],ymm3[3],ymm10[4],ymm3[5],ymm10[6],ymm3[7]+ vpsrlq $0x20, %ymm10, %ymm10+ vpblendd $0xaa, %ymm11, %ymm10, %ymm11 # ymm11 = ymm10[0],ymm11[1],ymm10[2],ymm11[3],ymm10[4],ymm11[5],ymm10[6],ymm11[7]+ vmovdqa 0x1a0(%rsi), %ymm1+ vmovdqa 0x640(%rsi), %ymm2+ vpsrlq $0x20, %ymm1, %ymm10+ vmovshdup %ymm2, %ymm15 # ymm15 = ymm2[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm5, %ymm13+ vmovshdup %ymm5, %ymm12 # ymm12 = ymm5[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm5, %ymm5+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm5, %ymm5 # ymm5 = ymm5[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm5, %ymm5 # ymm5 = ymm5[0],ymm12[1],ymm5[2],ymm12[3],ymm5[4],ymm12[5],ymm5[6],ymm12[7]+ vpsubd %ymm5, %ymm9, %ymm12+ vpaddd %ymm5, %ymm9, %ymm9+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm5+ vpsubd %ymm13, %ymm9, %ymm9+ vpmuldq %ymm1, %ymm4, %ymm13+ vmovshdup %ymm4, %ymm12 # ymm12 = ymm4[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm4, %ymm4+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm4, %ymm4 # ymm4 = ymm4[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm4, %ymm4 # ymm4 = ymm4[0],ymm12[1],ymm4[2],ymm12[3],ymm4[4],ymm12[5],ymm4[6],ymm12[7]+ vpsubd %ymm4, %ymm8, %ymm12+ vpaddd %ymm4, %ymm8, %ymm8+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm4+ vpsubd %ymm13, %ymm8, %ymm8+ vpmuldq %ymm1, %ymm3, %ymm13+ vmovshdup %ymm3, %ymm12 # ymm12 = ymm3[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm3, %ymm3+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm3, %ymm3 # ymm3 = ymm3[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm3, %ymm3 # ymm3 = ymm3[0],ymm12[1],ymm3[2],ymm12[3],ymm3[4],ymm12[5],ymm3[6],ymm12[7]+ vpsubd %ymm3, %ymm7, %ymm12+ vpaddd %ymm3, %ymm7, %ymm7+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm3+ vpsubd %ymm13, %ymm7, %ymm7+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm6, %ymm12+ vpaddd %ymm6, %ymm11, %ymm6+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm6, %ymm6+ vmovdqa 0x220(%rsi), %ymm1+ vmovdqa 0x6c0(%rsi), %ymm2+ vpsrlq $0x20, %ymm1, %ymm10+ vmovshdup %ymm2, %ymm15 # ymm15 = ymm2[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm7, %ymm13+ vmovshdup %ymm7, %ymm12 # ymm12 = ymm7[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm7, %ymm7+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm7, %ymm7 # ymm7 = ymm7[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm7, %ymm7 # ymm7 = ymm7[0],ymm12[1],ymm7[2],ymm12[3],ymm7[4],ymm12[5],ymm7[6],ymm12[7]+ vpsubd %ymm7, %ymm9, %ymm12+ vpaddd %ymm7, %ymm9, %ymm9+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm7+ vpsubd %ymm13, %ymm9, %ymm9+ vpmuldq %ymm1, %ymm6, %ymm13+ vmovshdup %ymm6, %ymm12 # ymm12 = ymm6[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm6, %ymm6+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm6, %ymm6 # ymm6 = ymm6[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm6, %ymm6 # ymm6 = ymm6[0],ymm12[1],ymm6[2],ymm12[3],ymm6[4],ymm12[5],ymm6[6],ymm12[7]+ vpsubd %ymm6, %ymm8, %ymm12+ vpaddd %ymm6, %ymm8, %ymm8+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm6+ vpsubd %ymm13, %ymm8, %ymm8+ vmovdqa 0x2a0(%rsi), %ymm1+ vmovdqa 0x740(%rsi), %ymm2+ vpsrlq $0x20, %ymm1, %ymm10+ vmovshdup %ymm2, %ymm15 # ymm15 = ymm2[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm3, %ymm13+ vmovshdup %ymm3, %ymm12 # ymm12 = ymm3[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm3, %ymm3+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm3, %ymm3 # ymm3 = ymm3[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm3, %ymm3 # ymm3 = ymm3[0],ymm12[1],ymm3[2],ymm12[3],ymm3[4],ymm12[5],ymm3[6],ymm12[7]+ vpsubd %ymm3, %ymm5, %ymm12+ vpaddd %ymm3, %ymm5, %ymm5+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm3+ vpsubd %ymm13, %ymm5, %ymm5+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm4, %ymm12+ vpaddd %ymm4, %ymm11, %ymm4+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm4, %ymm4+ vmovdqa 0x320(%rsi), %ymm1+ vmovdqa 0x7c0(%rsi), %ymm2+ vpsrlq $0x20, %ymm1, %ymm10+ vmovshdup %ymm2, %ymm15 # ymm15 = ymm2[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm8, %ymm13+ vmovshdup %ymm8, %ymm12 # ymm12 = ymm8[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm8, %ymm8 # ymm8 = ymm8[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm8, %ymm8 # ymm8 = ymm8[0],ymm12[1],ymm8[2],ymm12[3],ymm8[4],ymm12[5],ymm8[6],ymm12[7]+ vpsubd %ymm8, %ymm9, %ymm12+ vpaddd %ymm8, %ymm9, %ymm9+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm8+ vpsubd %ymm13, %ymm9, %ymm9+ vmovdqa 0x3a0(%rsi), %ymm1+ vmovdqa 0x840(%rsi), %ymm2+ vpsrlq $0x20, %ymm1, %ymm10+ vmovshdup %ymm2, %ymm15 # ymm15 = ymm2[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm6, %ymm13+ vmovshdup %ymm6, %ymm12 # ymm12 = ymm6[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm6, %ymm6+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm6, %ymm6 # ymm6 = ymm6[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm6, %ymm6 # ymm6 = ymm6[0],ymm12[1],ymm6[2],ymm12[3],ymm6[4],ymm12[5],ymm6[6],ymm12[7]+ vpsubd %ymm6, %ymm7, %ymm12+ vpaddd %ymm6, %ymm7, %ymm7+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm6+ vpsubd %ymm13, %ymm7, %ymm7+ vmovdqa 0x420(%rsi), %ymm1+ vmovdqa 0x8c0(%rsi), %ymm2+ vpsrlq $0x20, %ymm1, %ymm10+ vmovshdup %ymm2, %ymm15 # ymm15 = ymm2[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm4, %ymm13+ vmovshdup %ymm4, %ymm12 # ymm12 = ymm4[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm4, %ymm4+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm4, %ymm4 # ymm4 = ymm4[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm4, %ymm4 # ymm4 = ymm4[0],ymm12[1],ymm4[2],ymm12[3],ymm4[4],ymm12[5],ymm4[6],ymm12[7]+ vpsubd %ymm4, %ymm5, %ymm12+ vpaddd %ymm4, %ymm5, %ymm5+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm4+ vpsubd %ymm13, %ymm5, %ymm5+ vmovdqa 0x4a0(%rsi), %ymm1+ vmovdqa 0x940(%rsi), %ymm2+ vpsrlq $0x20, %ymm1, %ymm10+ vmovshdup %ymm2, %ymm15 # ymm15 = ymm2[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm3, %ymm12+ vpaddd %ymm3, %ymm11, %ymm3+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm3, %ymm3+ vmovdqa %ymm9, (%rdi)+ vmovdqa %ymm8, 0x20(%rdi)+ vmovdqa %ymm7, 0x40(%rdi)+ vmovdqa %ymm6, 0x60(%rdi)+ vmovdqa %ymm5, 0x80(%rdi)+ vmovdqa %ymm4, 0xa0(%rdi)+ vmovdqa %ymm3, 0xc0(%rdi)+ vmovdqa %ymm11, 0xe0(%rdi)+ vmovdqa 0x100(%rdi), %ymm4+ vmovdqa 0x120(%rdi), %ymm5+ vmovdqa 0x140(%rdi), %ymm6+ vmovdqa 0x160(%rdi), %ymm7+ vmovdqa 0x180(%rdi), %ymm8+ vmovdqa 0x1a0(%rdi), %ymm9+ vmovdqa 0x1c0(%rdi), %ymm10+ vmovdqa 0x1e0(%rdi), %ymm11+ vpbroadcastd 0x94(%rsi), %ymm1+ vpbroadcastd 0x534(%rsi), %ymm2+ vpmuldq %ymm1, %ymm8, %ymm13+ vmovshdup %ymm8, %ymm12 # ymm12 = ymm8[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm8, %ymm8 # ymm8 = ymm8[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm8, %ymm8 # ymm8 = ymm8[0],ymm12[1],ymm8[2],ymm12[3],ymm8[4],ymm12[5],ymm8[6],ymm12[7]+ vpsubd %ymm8, %ymm4, %ymm12+ vpaddd %ymm4, %ymm8, %ymm4+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm8+ vpsubd %ymm13, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm9, %ymm13+ vmovshdup %ymm9, %ymm12 # ymm12 = ymm9[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm9, %ymm9+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm9, %ymm9 # ymm9 = ymm9[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm9, %ymm9 # ymm9 = ymm9[0],ymm12[1],ymm9[2],ymm12[3],ymm9[4],ymm12[5],ymm9[6],ymm12[7]+ vpsubd %ymm9, %ymm5, %ymm12+ vpaddd %ymm5, %ymm9, %ymm5+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm9+ vpsubd %ymm13, %ymm5, %ymm5+ vpmuldq %ymm1, %ymm10, %ymm13+ vmovshdup %ymm10, %ymm12 # ymm12 = ymm10[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm10, %ymm10+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm10, %ymm10 # ymm10 = ymm10[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm10, %ymm10 # ymm10 = ymm10[0],ymm12[1],ymm10[2],ymm12[3],ymm10[4],ymm12[5],ymm10[6],ymm12[7]+ vpsubd %ymm10, %ymm6, %ymm12+ vpaddd %ymm6, %ymm10, %ymm6+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm10+ vpsubd %ymm13, %ymm6, %ymm6+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm7, %ymm12+ vpaddd %ymm7, %ymm11, %ymm7+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm7, %ymm7+ vperm2i128 $0x20, %ymm8, %ymm4, %ymm3 # ymm3 = ymm4[0,1],ymm8[0,1]+ vperm2i128 $0x31, %ymm8, %ymm4, %ymm8 # ymm8 = ymm4[2,3],ymm8[2,3]+ vperm2i128 $0x20, %ymm9, %ymm5, %ymm4 # ymm4 = ymm5[0,1],ymm9[0,1]+ vperm2i128 $0x31, %ymm9, %ymm5, %ymm9 # ymm9 = ymm5[2,3],ymm9[2,3]+ vperm2i128 $0x20, %ymm10, %ymm6, %ymm5 # ymm5 = ymm6[0,1],ymm10[0,1]+ vperm2i128 $0x31, %ymm10, %ymm6, %ymm10 # ymm10 = ymm6[2,3],ymm10[2,3]+ vperm2i128 $0x20, %ymm11, %ymm7, %ymm6 # ymm6 = ymm7[0,1],ymm11[0,1]+ vperm2i128 $0x31, %ymm11, %ymm7, %ymm11 # ymm11 = ymm7[2,3],ymm11[2,3]+ vmovdqa 0xc0(%rsi), %ymm1+ vmovdqa 0x560(%rsi), %ymm2+ vpmuldq %ymm1, %ymm5, %ymm13+ vmovshdup %ymm5, %ymm12 # ymm12 = ymm5[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm5, %ymm5+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm5, %ymm5 # ymm5 = ymm5[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm5, %ymm5 # ymm5 = ymm5[0],ymm12[1],ymm5[2],ymm12[3],ymm5[4],ymm12[5],ymm5[6],ymm12[7]+ vpsubd %ymm5, %ymm3, %ymm12+ vpaddd %ymm5, %ymm3, %ymm3+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm5+ vpsubd %ymm13, %ymm3, %ymm3+ vpmuldq %ymm1, %ymm10, %ymm13+ vmovshdup %ymm10, %ymm12 # ymm12 = ymm10[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm10, %ymm10+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm10, %ymm10 # ymm10 = ymm10[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm10, %ymm10 # ymm10 = ymm10[0],ymm12[1],ymm10[2],ymm12[3],ymm10[4],ymm12[5],ymm10[6],ymm12[7]+ vpsubd %ymm10, %ymm8, %ymm12+ vpaddd %ymm10, %ymm8, %ymm8+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm10+ vpsubd %ymm13, %ymm8, %ymm8+ vpmuldq %ymm1, %ymm6, %ymm13+ vmovshdup %ymm6, %ymm12 # ymm12 = ymm6[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm6, %ymm6+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm6, %ymm6 # ymm6 = ymm6[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm6, %ymm6 # ymm6 = ymm6[0],ymm12[1],ymm6[2],ymm12[3],ymm6[4],ymm12[5],ymm6[6],ymm12[7]+ vpsubd %ymm6, %ymm4, %ymm12+ vpaddd %ymm6, %ymm4, %ymm4+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm6+ vpsubd %ymm13, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm9, %ymm12+ vpaddd %ymm11, %ymm9, %ymm9+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm9, %ymm9+ vpunpcklqdq %ymm5, %ymm3, %ymm7 # ymm7 = ymm3[0],ymm5[0],ymm3[2],ymm5[2]+ vpunpckhqdq %ymm5, %ymm3, %ymm5 # ymm5 = ymm3[1],ymm5[1],ymm3[3],ymm5[3]+ vpunpcklqdq %ymm10, %ymm8, %ymm3 # ymm3 = ymm8[0],ymm10[0],ymm8[2],ymm10[2]+ vpunpckhqdq %ymm10, %ymm8, %ymm10 # ymm10 = ymm8[1],ymm10[1],ymm8[3],ymm10[3]+ vpunpcklqdq %ymm6, %ymm4, %ymm8 # ymm8 = ymm4[0],ymm6[0],ymm4[2],ymm6[2]+ vpunpckhqdq %ymm6, %ymm4, %ymm6 # ymm6 = ymm4[1],ymm6[1],ymm4[3],ymm6[3]+ vpunpcklqdq %ymm11, %ymm9, %ymm4 # ymm4 = ymm9[0],ymm11[0],ymm9[2],ymm11[2]+ vpunpckhqdq %ymm11, %ymm9, %ymm11 # ymm11 = ymm9[1],ymm11[1],ymm9[3],ymm11[3]+ vmovdqa 0x140(%rsi), %ymm1+ vmovdqa 0x5e0(%rsi), %ymm2+ vpmuldq %ymm1, %ymm8, %ymm13+ vmovshdup %ymm8, %ymm12 # ymm12 = ymm8[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm8, %ymm8 # ymm8 = ymm8[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm8, %ymm8 # ymm8 = ymm8[0],ymm12[1],ymm8[2],ymm12[3],ymm8[4],ymm12[5],ymm8[6],ymm12[7]+ vpsubd %ymm8, %ymm7, %ymm12+ vpaddd %ymm7, %ymm8, %ymm7+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm8+ vpsubd %ymm13, %ymm7, %ymm7+ vpmuldq %ymm1, %ymm6, %ymm13+ vmovshdup %ymm6, %ymm12 # ymm12 = ymm6[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm6, %ymm6+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm6, %ymm6 # ymm6 = ymm6[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm6, %ymm6 # ymm6 = ymm6[0],ymm12[1],ymm6[2],ymm12[3],ymm6[4],ymm12[5],ymm6[6],ymm12[7]+ vpsubd %ymm6, %ymm5, %ymm12+ vpaddd %ymm6, %ymm5, %ymm5+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm6+ vpsubd %ymm13, %ymm5, %ymm5+ vpmuldq %ymm1, %ymm4, %ymm13+ vmovshdup %ymm4, %ymm12 # ymm12 = ymm4[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm4, %ymm4+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm4, %ymm4 # ymm4 = ymm4[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm4, %ymm4 # ymm4 = ymm4[0],ymm12[1],ymm4[2],ymm12[3],ymm4[4],ymm12[5],ymm4[6],ymm12[7]+ vpsubd %ymm4, %ymm3, %ymm12+ vpaddd %ymm4, %ymm3, %ymm3+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm4+ vpsubd %ymm13, %ymm3, %ymm3+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm10, %ymm12+ vpaddd %ymm11, %ymm10, %ymm10+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm10, %ymm10+ vmovsldup %ymm8, %ymm9 # ymm9 = ymm8[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm9, %ymm7, %ymm9 # ymm9 = ymm7[0],ymm9[1],ymm7[2],ymm9[3],ymm7[4],ymm9[5],ymm7[6],ymm9[7]+ vpsrlq $0x20, %ymm7, %ymm7+ vpblendd $0xaa, %ymm8, %ymm7, %ymm8 # ymm8 = ymm7[0],ymm8[1],ymm7[2],ymm8[3],ymm7[4],ymm8[5],ymm7[6],ymm8[7]+ vmovsldup %ymm6, %ymm7 # ymm7 = ymm6[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm7, %ymm5, %ymm7 # ymm7 = ymm5[0],ymm7[1],ymm5[2],ymm7[3],ymm5[4],ymm7[5],ymm5[6],ymm7[7]+ vpsrlq $0x20, %ymm5, %ymm5+ vpblendd $0xaa, %ymm6, %ymm5, %ymm6 # ymm6 = ymm5[0],ymm6[1],ymm5[2],ymm6[3],ymm5[4],ymm6[5],ymm5[6],ymm6[7]+ vmovsldup %ymm4, %ymm5 # ymm5 = ymm4[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm5, %ymm3, %ymm5 # ymm5 = ymm3[0],ymm5[1],ymm3[2],ymm5[3],ymm3[4],ymm5[5],ymm3[6],ymm5[7]+ vpsrlq $0x20, %ymm3, %ymm3+ vpblendd $0xaa, %ymm4, %ymm3, %ymm4 # ymm4 = ymm3[0],ymm4[1],ymm3[2],ymm4[3],ymm3[4],ymm4[5],ymm3[6],ymm4[7]+ vmovsldup %ymm11, %ymm3 # ymm3 = ymm11[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm3, %ymm10, %ymm3 # ymm3 = ymm10[0],ymm3[1],ymm10[2],ymm3[3],ymm10[4],ymm3[5],ymm10[6],ymm3[7]+ vpsrlq $0x20, %ymm10, %ymm10+ vpblendd $0xaa, %ymm11, %ymm10, %ymm11 # ymm11 = ymm10[0],ymm11[1],ymm10[2],ymm11[3],ymm10[4],ymm11[5],ymm10[6],ymm11[7]+ vmovdqa 0x1c0(%rsi), %ymm1+ vmovdqa 0x660(%rsi), %ymm2+ vpsrlq $0x20, %ymm1, %ymm10+ vmovshdup %ymm2, %ymm15 # ymm15 = ymm2[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm5, %ymm13+ vmovshdup %ymm5, %ymm12 # ymm12 = ymm5[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm5, %ymm5+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm5, %ymm5 # ymm5 = ymm5[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm5, %ymm5 # ymm5 = ymm5[0],ymm12[1],ymm5[2],ymm12[3],ymm5[4],ymm12[5],ymm5[6],ymm12[7]+ vpsubd %ymm5, %ymm9, %ymm12+ vpaddd %ymm5, %ymm9, %ymm9+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm5+ vpsubd %ymm13, %ymm9, %ymm9+ vpmuldq %ymm1, %ymm4, %ymm13+ vmovshdup %ymm4, %ymm12 # ymm12 = ymm4[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm4, %ymm4+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm4, %ymm4 # ymm4 = ymm4[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm4, %ymm4 # ymm4 = ymm4[0],ymm12[1],ymm4[2],ymm12[3],ymm4[4],ymm12[5],ymm4[6],ymm12[7]+ vpsubd %ymm4, %ymm8, %ymm12+ vpaddd %ymm4, %ymm8, %ymm8+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm4+ vpsubd %ymm13, %ymm8, %ymm8+ vpmuldq %ymm1, %ymm3, %ymm13+ vmovshdup %ymm3, %ymm12 # ymm12 = ymm3[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm3, %ymm3+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm3, %ymm3 # ymm3 = ymm3[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm3, %ymm3 # ymm3 = ymm3[0],ymm12[1],ymm3[2],ymm12[3],ymm3[4],ymm12[5],ymm3[6],ymm12[7]+ vpsubd %ymm3, %ymm7, %ymm12+ vpaddd %ymm3, %ymm7, %ymm7+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm3+ vpsubd %ymm13, %ymm7, %ymm7+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm6, %ymm12+ vpaddd %ymm6, %ymm11, %ymm6+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm6, %ymm6+ vmovdqa 0x240(%rsi), %ymm1+ vmovdqa 0x6e0(%rsi), %ymm2+ vpsrlq $0x20, %ymm1, %ymm10+ vmovshdup %ymm2, %ymm15 # ymm15 = ymm2[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm7, %ymm13+ vmovshdup %ymm7, %ymm12 # ymm12 = ymm7[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm7, %ymm7+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm7, %ymm7 # ymm7 = ymm7[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm7, %ymm7 # ymm7 = ymm7[0],ymm12[1],ymm7[2],ymm12[3],ymm7[4],ymm12[5],ymm7[6],ymm12[7]+ vpsubd %ymm7, %ymm9, %ymm12+ vpaddd %ymm7, %ymm9, %ymm9+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm7+ vpsubd %ymm13, %ymm9, %ymm9+ vpmuldq %ymm1, %ymm6, %ymm13+ vmovshdup %ymm6, %ymm12 # ymm12 = ymm6[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm6, %ymm6+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm6, %ymm6 # ymm6 = ymm6[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm6, %ymm6 # ymm6 = ymm6[0],ymm12[1],ymm6[2],ymm12[3],ymm6[4],ymm12[5],ymm6[6],ymm12[7]+ vpsubd %ymm6, %ymm8, %ymm12+ vpaddd %ymm6, %ymm8, %ymm8+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm6+ vpsubd %ymm13, %ymm8, %ymm8+ vmovdqa 0x2c0(%rsi), %ymm1+ vmovdqa 0x760(%rsi), %ymm2+ vpsrlq $0x20, %ymm1, %ymm10+ vmovshdup %ymm2, %ymm15 # ymm15 = ymm2[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm3, %ymm13+ vmovshdup %ymm3, %ymm12 # ymm12 = ymm3[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm3, %ymm3+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm3, %ymm3 # ymm3 = ymm3[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm3, %ymm3 # ymm3 = ymm3[0],ymm12[1],ymm3[2],ymm12[3],ymm3[4],ymm12[5],ymm3[6],ymm12[7]+ vpsubd %ymm3, %ymm5, %ymm12+ vpaddd %ymm3, %ymm5, %ymm5+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm3+ vpsubd %ymm13, %ymm5, %ymm5+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm4, %ymm12+ vpaddd %ymm4, %ymm11, %ymm4+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm4, %ymm4+ vmovdqa 0x340(%rsi), %ymm1+ vmovdqa 0x7e0(%rsi), %ymm2+ vpsrlq $0x20, %ymm1, %ymm10+ vmovshdup %ymm2, %ymm15 # ymm15 = ymm2[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm8, %ymm13+ vmovshdup %ymm8, %ymm12 # ymm12 = ymm8[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm8, %ymm8 # ymm8 = ymm8[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm8, %ymm8 # ymm8 = ymm8[0],ymm12[1],ymm8[2],ymm12[3],ymm8[4],ymm12[5],ymm8[6],ymm12[7]+ vpsubd %ymm8, %ymm9, %ymm12+ vpaddd %ymm8, %ymm9, %ymm9+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm8+ vpsubd %ymm13, %ymm9, %ymm9+ vmovdqa 0x3c0(%rsi), %ymm1+ vmovdqa 0x860(%rsi), %ymm2+ vpsrlq $0x20, %ymm1, %ymm10+ vmovshdup %ymm2, %ymm15 # ymm15 = ymm2[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm6, %ymm13+ vmovshdup %ymm6, %ymm12 # ymm12 = ymm6[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm6, %ymm6+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm6, %ymm6 # ymm6 = ymm6[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm6, %ymm6 # ymm6 = ymm6[0],ymm12[1],ymm6[2],ymm12[3],ymm6[4],ymm12[5],ymm6[6],ymm12[7]+ vpsubd %ymm6, %ymm7, %ymm12+ vpaddd %ymm6, %ymm7, %ymm7+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm6+ vpsubd %ymm13, %ymm7, %ymm7+ vmovdqa 0x440(%rsi), %ymm1+ vmovdqa 0x8e0(%rsi), %ymm2+ vpsrlq $0x20, %ymm1, %ymm10+ vmovshdup %ymm2, %ymm15 # ymm15 = ymm2[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm4, %ymm13+ vmovshdup %ymm4, %ymm12 # ymm12 = ymm4[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm4, %ymm4+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm4, %ymm4 # ymm4 = ymm4[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm4, %ymm4 # ymm4 = ymm4[0],ymm12[1],ymm4[2],ymm12[3],ymm4[4],ymm12[5],ymm4[6],ymm12[7]+ vpsubd %ymm4, %ymm5, %ymm12+ vpaddd %ymm4, %ymm5, %ymm5+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm4+ vpsubd %ymm13, %ymm5, %ymm5+ vmovdqa 0x4c0(%rsi), %ymm1+ vmovdqa 0x960(%rsi), %ymm2+ vpsrlq $0x20, %ymm1, %ymm10+ vmovshdup %ymm2, %ymm15 # ymm15 = ymm2[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm3, %ymm12+ vpaddd %ymm3, %ymm11, %ymm3+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm3, %ymm3+ vmovdqa %ymm9, 0x100(%rdi)+ vmovdqa %ymm8, 0x120(%rdi)+ vmovdqa %ymm7, 0x140(%rdi)+ vmovdqa %ymm6, 0x160(%rdi)+ vmovdqa %ymm5, 0x180(%rdi)+ vmovdqa %ymm4, 0x1a0(%rdi)+ vmovdqa %ymm3, 0x1c0(%rdi)+ vmovdqa %ymm11, 0x1e0(%rdi)+ vmovdqa 0x200(%rdi), %ymm4+ vmovdqa 0x220(%rdi), %ymm5+ vmovdqa 0x240(%rdi), %ymm6+ vmovdqa 0x260(%rdi), %ymm7+ vmovdqa 0x280(%rdi), %ymm8+ vmovdqa 0x2a0(%rdi), %ymm9+ vmovdqa 0x2c0(%rdi), %ymm10+ vmovdqa 0x2e0(%rdi), %ymm11+ vpbroadcastd 0x98(%rsi), %ymm1+ vpbroadcastd 0x538(%rsi), %ymm2+ vpmuldq %ymm1, %ymm8, %ymm13+ vmovshdup %ymm8, %ymm12 # ymm12 = ymm8[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm8, %ymm8 # ymm8 = ymm8[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm8, %ymm8 # ymm8 = ymm8[0],ymm12[1],ymm8[2],ymm12[3],ymm8[4],ymm12[5],ymm8[6],ymm12[7]+ vpsubd %ymm8, %ymm4, %ymm12+ vpaddd %ymm4, %ymm8, %ymm4+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm8+ vpsubd %ymm13, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm9, %ymm13+ vmovshdup %ymm9, %ymm12 # ymm12 = ymm9[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm9, %ymm9+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm9, %ymm9 # ymm9 = ymm9[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm9, %ymm9 # ymm9 = ymm9[0],ymm12[1],ymm9[2],ymm12[3],ymm9[4],ymm12[5],ymm9[6],ymm12[7]+ vpsubd %ymm9, %ymm5, %ymm12+ vpaddd %ymm5, %ymm9, %ymm5+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm9+ vpsubd %ymm13, %ymm5, %ymm5+ vpmuldq %ymm1, %ymm10, %ymm13+ vmovshdup %ymm10, %ymm12 # ymm12 = ymm10[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm10, %ymm10+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm10, %ymm10 # ymm10 = ymm10[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm10, %ymm10 # ymm10 = ymm10[0],ymm12[1],ymm10[2],ymm12[3],ymm10[4],ymm12[5],ymm10[6],ymm12[7]+ vpsubd %ymm10, %ymm6, %ymm12+ vpaddd %ymm6, %ymm10, %ymm6+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm10+ vpsubd %ymm13, %ymm6, %ymm6+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm7, %ymm12+ vpaddd %ymm7, %ymm11, %ymm7+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm7, %ymm7+ vperm2i128 $0x20, %ymm8, %ymm4, %ymm3 # ymm3 = ymm4[0,1],ymm8[0,1]+ vperm2i128 $0x31, %ymm8, %ymm4, %ymm8 # ymm8 = ymm4[2,3],ymm8[2,3]+ vperm2i128 $0x20, %ymm9, %ymm5, %ymm4 # ymm4 = ymm5[0,1],ymm9[0,1]+ vperm2i128 $0x31, %ymm9, %ymm5, %ymm9 # ymm9 = ymm5[2,3],ymm9[2,3]+ vperm2i128 $0x20, %ymm10, %ymm6, %ymm5 # ymm5 = ymm6[0,1],ymm10[0,1]+ vperm2i128 $0x31, %ymm10, %ymm6, %ymm10 # ymm10 = ymm6[2,3],ymm10[2,3]+ vperm2i128 $0x20, %ymm11, %ymm7, %ymm6 # ymm6 = ymm7[0,1],ymm11[0,1]+ vperm2i128 $0x31, %ymm11, %ymm7, %ymm11 # ymm11 = ymm7[2,3],ymm11[2,3]+ vmovdqa 0xe0(%rsi), %ymm1+ vmovdqa 0x580(%rsi), %ymm2+ vpmuldq %ymm1, %ymm5, %ymm13+ vmovshdup %ymm5, %ymm12 # ymm12 = ymm5[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm5, %ymm5+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm5, %ymm5 # ymm5 = ymm5[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm5, %ymm5 # ymm5 = ymm5[0],ymm12[1],ymm5[2],ymm12[3],ymm5[4],ymm12[5],ymm5[6],ymm12[7]+ vpsubd %ymm5, %ymm3, %ymm12+ vpaddd %ymm5, %ymm3, %ymm3+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm5+ vpsubd %ymm13, %ymm3, %ymm3+ vpmuldq %ymm1, %ymm10, %ymm13+ vmovshdup %ymm10, %ymm12 # ymm12 = ymm10[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm10, %ymm10+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm10, %ymm10 # ymm10 = ymm10[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm10, %ymm10 # ymm10 = ymm10[0],ymm12[1],ymm10[2],ymm12[3],ymm10[4],ymm12[5],ymm10[6],ymm12[7]+ vpsubd %ymm10, %ymm8, %ymm12+ vpaddd %ymm10, %ymm8, %ymm8+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm10+ vpsubd %ymm13, %ymm8, %ymm8+ vpmuldq %ymm1, %ymm6, %ymm13+ vmovshdup %ymm6, %ymm12 # ymm12 = ymm6[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm6, %ymm6+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm6, %ymm6 # ymm6 = ymm6[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm6, %ymm6 # ymm6 = ymm6[0],ymm12[1],ymm6[2],ymm12[3],ymm6[4],ymm12[5],ymm6[6],ymm12[7]+ vpsubd %ymm6, %ymm4, %ymm12+ vpaddd %ymm6, %ymm4, %ymm4+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm6+ vpsubd %ymm13, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm9, %ymm12+ vpaddd %ymm11, %ymm9, %ymm9+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm9, %ymm9+ vpunpcklqdq %ymm5, %ymm3, %ymm7 # ymm7 = ymm3[0],ymm5[0],ymm3[2],ymm5[2]+ vpunpckhqdq %ymm5, %ymm3, %ymm5 # ymm5 = ymm3[1],ymm5[1],ymm3[3],ymm5[3]+ vpunpcklqdq %ymm10, %ymm8, %ymm3 # ymm3 = ymm8[0],ymm10[0],ymm8[2],ymm10[2]+ vpunpckhqdq %ymm10, %ymm8, %ymm10 # ymm10 = ymm8[1],ymm10[1],ymm8[3],ymm10[3]+ vpunpcklqdq %ymm6, %ymm4, %ymm8 # ymm8 = ymm4[0],ymm6[0],ymm4[2],ymm6[2]+ vpunpckhqdq %ymm6, %ymm4, %ymm6 # ymm6 = ymm4[1],ymm6[1],ymm4[3],ymm6[3]+ vpunpcklqdq %ymm11, %ymm9, %ymm4 # ymm4 = ymm9[0],ymm11[0],ymm9[2],ymm11[2]+ vpunpckhqdq %ymm11, %ymm9, %ymm11 # ymm11 = ymm9[1],ymm11[1],ymm9[3],ymm11[3]+ vmovdqa 0x160(%rsi), %ymm1+ vmovdqa 0x600(%rsi), %ymm2+ vpmuldq %ymm1, %ymm8, %ymm13+ vmovshdup %ymm8, %ymm12 # ymm12 = ymm8[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm8, %ymm8 # ymm8 = ymm8[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm8, %ymm8 # ymm8 = ymm8[0],ymm12[1],ymm8[2],ymm12[3],ymm8[4],ymm12[5],ymm8[6],ymm12[7]+ vpsubd %ymm8, %ymm7, %ymm12+ vpaddd %ymm7, %ymm8, %ymm7+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm8+ vpsubd %ymm13, %ymm7, %ymm7+ vpmuldq %ymm1, %ymm6, %ymm13+ vmovshdup %ymm6, %ymm12 # ymm12 = ymm6[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm6, %ymm6+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm6, %ymm6 # ymm6 = ymm6[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm6, %ymm6 # ymm6 = ymm6[0],ymm12[1],ymm6[2],ymm12[3],ymm6[4],ymm12[5],ymm6[6],ymm12[7]+ vpsubd %ymm6, %ymm5, %ymm12+ vpaddd %ymm6, %ymm5, %ymm5+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm6+ vpsubd %ymm13, %ymm5, %ymm5+ vpmuldq %ymm1, %ymm4, %ymm13+ vmovshdup %ymm4, %ymm12 # ymm12 = ymm4[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm4, %ymm4+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm4, %ymm4 # ymm4 = ymm4[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm4, %ymm4 # ymm4 = ymm4[0],ymm12[1],ymm4[2],ymm12[3],ymm4[4],ymm12[5],ymm4[6],ymm12[7]+ vpsubd %ymm4, %ymm3, %ymm12+ vpaddd %ymm4, %ymm3, %ymm3+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm4+ vpsubd %ymm13, %ymm3, %ymm3+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm10, %ymm12+ vpaddd %ymm11, %ymm10, %ymm10+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm10, %ymm10+ vmovsldup %ymm8, %ymm9 # ymm9 = ymm8[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm9, %ymm7, %ymm9 # ymm9 = ymm7[0],ymm9[1],ymm7[2],ymm9[3],ymm7[4],ymm9[5],ymm7[6],ymm9[7]+ vpsrlq $0x20, %ymm7, %ymm7+ vpblendd $0xaa, %ymm8, %ymm7, %ymm8 # ymm8 = ymm7[0],ymm8[1],ymm7[2],ymm8[3],ymm7[4],ymm8[5],ymm7[6],ymm8[7]+ vmovsldup %ymm6, %ymm7 # ymm7 = ymm6[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm7, %ymm5, %ymm7 # ymm7 = ymm5[0],ymm7[1],ymm5[2],ymm7[3],ymm5[4],ymm7[5],ymm5[6],ymm7[7]+ vpsrlq $0x20, %ymm5, %ymm5+ vpblendd $0xaa, %ymm6, %ymm5, %ymm6 # ymm6 = ymm5[0],ymm6[1],ymm5[2],ymm6[3],ymm5[4],ymm6[5],ymm5[6],ymm6[7]+ vmovsldup %ymm4, %ymm5 # ymm5 = ymm4[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm5, %ymm3, %ymm5 # ymm5 = ymm3[0],ymm5[1],ymm3[2],ymm5[3],ymm3[4],ymm5[5],ymm3[6],ymm5[7]+ vpsrlq $0x20, %ymm3, %ymm3+ vpblendd $0xaa, %ymm4, %ymm3, %ymm4 # ymm4 = ymm3[0],ymm4[1],ymm3[2],ymm4[3],ymm3[4],ymm4[5],ymm3[6],ymm4[7]+ vmovsldup %ymm11, %ymm3 # ymm3 = ymm11[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm3, %ymm10, %ymm3 # ymm3 = ymm10[0],ymm3[1],ymm10[2],ymm3[3],ymm10[4],ymm3[5],ymm10[6],ymm3[7]+ vpsrlq $0x20, %ymm10, %ymm10+ vpblendd $0xaa, %ymm11, %ymm10, %ymm11 # ymm11 = ymm10[0],ymm11[1],ymm10[2],ymm11[3],ymm10[4],ymm11[5],ymm10[6],ymm11[7]+ vmovdqa 0x1e0(%rsi), %ymm1+ vmovdqa 0x680(%rsi), %ymm2+ vpsrlq $0x20, %ymm1, %ymm10+ vmovshdup %ymm2, %ymm15 # ymm15 = ymm2[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm5, %ymm13+ vmovshdup %ymm5, %ymm12 # ymm12 = ymm5[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm5, %ymm5+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm5, %ymm5 # ymm5 = ymm5[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm5, %ymm5 # ymm5 = ymm5[0],ymm12[1],ymm5[2],ymm12[3],ymm5[4],ymm12[5],ymm5[6],ymm12[7]+ vpsubd %ymm5, %ymm9, %ymm12+ vpaddd %ymm5, %ymm9, %ymm9+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm5+ vpsubd %ymm13, %ymm9, %ymm9+ vpmuldq %ymm1, %ymm4, %ymm13+ vmovshdup %ymm4, %ymm12 # ymm12 = ymm4[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm4, %ymm4+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm4, %ymm4 # ymm4 = ymm4[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm4, %ymm4 # ymm4 = ymm4[0],ymm12[1],ymm4[2],ymm12[3],ymm4[4],ymm12[5],ymm4[6],ymm12[7]+ vpsubd %ymm4, %ymm8, %ymm12+ vpaddd %ymm4, %ymm8, %ymm8+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm4+ vpsubd %ymm13, %ymm8, %ymm8+ vpmuldq %ymm1, %ymm3, %ymm13+ vmovshdup %ymm3, %ymm12 # ymm12 = ymm3[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm3, %ymm3+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm3, %ymm3 # ymm3 = ymm3[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm3, %ymm3 # ymm3 = ymm3[0],ymm12[1],ymm3[2],ymm12[3],ymm3[4],ymm12[5],ymm3[6],ymm12[7]+ vpsubd %ymm3, %ymm7, %ymm12+ vpaddd %ymm3, %ymm7, %ymm7+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm3+ vpsubd %ymm13, %ymm7, %ymm7+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm6, %ymm12+ vpaddd %ymm6, %ymm11, %ymm6+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm6, %ymm6+ vmovdqa 0x260(%rsi), %ymm1+ vmovdqa 0x700(%rsi), %ymm2+ vpsrlq $0x20, %ymm1, %ymm10+ vmovshdup %ymm2, %ymm15 # ymm15 = ymm2[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm7, %ymm13+ vmovshdup %ymm7, %ymm12 # ymm12 = ymm7[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm7, %ymm7+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm7, %ymm7 # ymm7 = ymm7[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm7, %ymm7 # ymm7 = ymm7[0],ymm12[1],ymm7[2],ymm12[3],ymm7[4],ymm12[5],ymm7[6],ymm12[7]+ vpsubd %ymm7, %ymm9, %ymm12+ vpaddd %ymm7, %ymm9, %ymm9+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm7+ vpsubd %ymm13, %ymm9, %ymm9+ vpmuldq %ymm1, %ymm6, %ymm13+ vmovshdup %ymm6, %ymm12 # ymm12 = ymm6[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm6, %ymm6+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm6, %ymm6 # ymm6 = ymm6[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm6, %ymm6 # ymm6 = ymm6[0],ymm12[1],ymm6[2],ymm12[3],ymm6[4],ymm12[5],ymm6[6],ymm12[7]+ vpsubd %ymm6, %ymm8, %ymm12+ vpaddd %ymm6, %ymm8, %ymm8+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm6+ vpsubd %ymm13, %ymm8, %ymm8+ vmovdqa 0x2e0(%rsi), %ymm1+ vmovdqa 0x780(%rsi), %ymm2+ vpsrlq $0x20, %ymm1, %ymm10+ vmovshdup %ymm2, %ymm15 # ymm15 = ymm2[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm3, %ymm13+ vmovshdup %ymm3, %ymm12 # ymm12 = ymm3[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm3, %ymm3+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm3, %ymm3 # ymm3 = ymm3[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm3, %ymm3 # ymm3 = ymm3[0],ymm12[1],ymm3[2],ymm12[3],ymm3[4],ymm12[5],ymm3[6],ymm12[7]+ vpsubd %ymm3, %ymm5, %ymm12+ vpaddd %ymm3, %ymm5, %ymm5+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm3+ vpsubd %ymm13, %ymm5, %ymm5+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm4, %ymm12+ vpaddd %ymm4, %ymm11, %ymm4+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm4, %ymm4+ vmovdqa 0x360(%rsi), %ymm1+ vmovdqa 0x800(%rsi), %ymm2+ vpsrlq $0x20, %ymm1, %ymm10+ vmovshdup %ymm2, %ymm15 # ymm15 = ymm2[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm8, %ymm13+ vmovshdup %ymm8, %ymm12 # ymm12 = ymm8[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm8, %ymm8 # ymm8 = ymm8[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm8, %ymm8 # ymm8 = ymm8[0],ymm12[1],ymm8[2],ymm12[3],ymm8[4],ymm12[5],ymm8[6],ymm12[7]+ vpsubd %ymm8, %ymm9, %ymm12+ vpaddd %ymm8, %ymm9, %ymm9+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm8+ vpsubd %ymm13, %ymm9, %ymm9+ vmovdqa 0x3e0(%rsi), %ymm1+ vmovdqa 0x880(%rsi), %ymm2+ vpsrlq $0x20, %ymm1, %ymm10+ vmovshdup %ymm2, %ymm15 # ymm15 = ymm2[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm6, %ymm13+ vmovshdup %ymm6, %ymm12 # ymm12 = ymm6[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm6, %ymm6+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm6, %ymm6 # ymm6 = ymm6[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm6, %ymm6 # ymm6 = ymm6[0],ymm12[1],ymm6[2],ymm12[3],ymm6[4],ymm12[5],ymm6[6],ymm12[7]+ vpsubd %ymm6, %ymm7, %ymm12+ vpaddd %ymm6, %ymm7, %ymm7+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm6+ vpsubd %ymm13, %ymm7, %ymm7+ vmovdqa 0x460(%rsi), %ymm1+ vmovdqa 0x900(%rsi), %ymm2+ vpsrlq $0x20, %ymm1, %ymm10+ vmovshdup %ymm2, %ymm15 # ymm15 = ymm2[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm4, %ymm13+ vmovshdup %ymm4, %ymm12 # ymm12 = ymm4[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm4, %ymm4+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm4, %ymm4 # ymm4 = ymm4[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm4, %ymm4 # ymm4 = ymm4[0],ymm12[1],ymm4[2],ymm12[3],ymm4[4],ymm12[5],ymm4[6],ymm12[7]+ vpsubd %ymm4, %ymm5, %ymm12+ vpaddd %ymm4, %ymm5, %ymm5+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm4+ vpsubd %ymm13, %ymm5, %ymm5+ vmovdqa 0x4e0(%rsi), %ymm1+ vmovdqa 0x980(%rsi), %ymm2+ vpsrlq $0x20, %ymm1, %ymm10+ vmovshdup %ymm2, %ymm15 # ymm15 = ymm2[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm3, %ymm12+ vpaddd %ymm3, %ymm11, %ymm3+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm3, %ymm3+ vmovdqa %ymm9, 0x200(%rdi)+ vmovdqa %ymm8, 0x220(%rdi)+ vmovdqa %ymm7, 0x240(%rdi)+ vmovdqa %ymm6, 0x260(%rdi)+ vmovdqa %ymm5, 0x280(%rdi)+ vmovdqa %ymm4, 0x2a0(%rdi)+ vmovdqa %ymm3, 0x2c0(%rdi)+ vmovdqa %ymm11, 0x2e0(%rdi)+ vmovdqa 0x300(%rdi), %ymm4+ vmovdqa 0x320(%rdi), %ymm5+ vmovdqa 0x340(%rdi), %ymm6+ vmovdqa 0x360(%rdi), %ymm7+ vmovdqa 0x380(%rdi), %ymm8+ vmovdqa 0x3a0(%rdi), %ymm9+ vmovdqa 0x3c0(%rdi), %ymm10+ vmovdqa 0x3e0(%rdi), %ymm11+ vpbroadcastd 0x9c(%rsi), %ymm1+ vpbroadcastd 0x53c(%rsi), %ymm2+ vpmuldq %ymm1, %ymm8, %ymm13+ vmovshdup %ymm8, %ymm12 # ymm12 = ymm8[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm8, %ymm8 # ymm8 = ymm8[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm8, %ymm8 # ymm8 = ymm8[0],ymm12[1],ymm8[2],ymm12[3],ymm8[4],ymm12[5],ymm8[6],ymm12[7]+ vpsubd %ymm8, %ymm4, %ymm12+ vpaddd %ymm4, %ymm8, %ymm4+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm8+ vpsubd %ymm13, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm9, %ymm13+ vmovshdup %ymm9, %ymm12 # ymm12 = ymm9[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm9, %ymm9+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm9, %ymm9 # ymm9 = ymm9[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm9, %ymm9 # ymm9 = ymm9[0],ymm12[1],ymm9[2],ymm12[3],ymm9[4],ymm12[5],ymm9[6],ymm12[7]+ vpsubd %ymm9, %ymm5, %ymm12+ vpaddd %ymm5, %ymm9, %ymm5+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm9+ vpsubd %ymm13, %ymm5, %ymm5+ vpmuldq %ymm1, %ymm10, %ymm13+ vmovshdup %ymm10, %ymm12 # ymm12 = ymm10[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm10, %ymm10+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm10, %ymm10 # ymm10 = ymm10[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm10, %ymm10 # ymm10 = ymm10[0],ymm12[1],ymm10[2],ymm12[3],ymm10[4],ymm12[5],ymm10[6],ymm12[7]+ vpsubd %ymm10, %ymm6, %ymm12+ vpaddd %ymm6, %ymm10, %ymm6+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm10+ vpsubd %ymm13, %ymm6, %ymm6+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm7, %ymm12+ vpaddd %ymm7, %ymm11, %ymm7+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm7, %ymm7+ vperm2i128 $0x20, %ymm8, %ymm4, %ymm3 # ymm3 = ymm4[0,1],ymm8[0,1]+ vperm2i128 $0x31, %ymm8, %ymm4, %ymm8 # ymm8 = ymm4[2,3],ymm8[2,3]+ vperm2i128 $0x20, %ymm9, %ymm5, %ymm4 # ymm4 = ymm5[0,1],ymm9[0,1]+ vperm2i128 $0x31, %ymm9, %ymm5, %ymm9 # ymm9 = ymm5[2,3],ymm9[2,3]+ vperm2i128 $0x20, %ymm10, %ymm6, %ymm5 # ymm5 = ymm6[0,1],ymm10[0,1]+ vperm2i128 $0x31, %ymm10, %ymm6, %ymm10 # ymm10 = ymm6[2,3],ymm10[2,3]+ vperm2i128 $0x20, %ymm11, %ymm7, %ymm6 # ymm6 = ymm7[0,1],ymm11[0,1]+ vperm2i128 $0x31, %ymm11, %ymm7, %ymm11 # ymm11 = ymm7[2,3],ymm11[2,3]+ vmovdqa 0x100(%rsi), %ymm1+ vmovdqa 0x5a0(%rsi), %ymm2+ vpmuldq %ymm1, %ymm5, %ymm13+ vmovshdup %ymm5, %ymm12 # ymm12 = ymm5[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm5, %ymm5+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm5, %ymm5 # ymm5 = ymm5[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm5, %ymm5 # ymm5 = ymm5[0],ymm12[1],ymm5[2],ymm12[3],ymm5[4],ymm12[5],ymm5[6],ymm12[7]+ vpsubd %ymm5, %ymm3, %ymm12+ vpaddd %ymm5, %ymm3, %ymm3+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm5+ vpsubd %ymm13, %ymm3, %ymm3+ vpmuldq %ymm1, %ymm10, %ymm13+ vmovshdup %ymm10, %ymm12 # ymm12 = ymm10[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm10, %ymm10+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm10, %ymm10 # ymm10 = ymm10[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm10, %ymm10 # ymm10 = ymm10[0],ymm12[1],ymm10[2],ymm12[3],ymm10[4],ymm12[5],ymm10[6],ymm12[7]+ vpsubd %ymm10, %ymm8, %ymm12+ vpaddd %ymm10, %ymm8, %ymm8+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm10+ vpsubd %ymm13, %ymm8, %ymm8+ vpmuldq %ymm1, %ymm6, %ymm13+ vmovshdup %ymm6, %ymm12 # ymm12 = ymm6[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm6, %ymm6+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm6, %ymm6 # ymm6 = ymm6[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm6, %ymm6 # ymm6 = ymm6[0],ymm12[1],ymm6[2],ymm12[3],ymm6[4],ymm12[5],ymm6[6],ymm12[7]+ vpsubd %ymm6, %ymm4, %ymm12+ vpaddd %ymm6, %ymm4, %ymm4+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm6+ vpsubd %ymm13, %ymm4, %ymm4+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm9, %ymm12+ vpaddd %ymm11, %ymm9, %ymm9+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm9, %ymm9+ vpunpcklqdq %ymm5, %ymm3, %ymm7 # ymm7 = ymm3[0],ymm5[0],ymm3[2],ymm5[2]+ vpunpckhqdq %ymm5, %ymm3, %ymm5 # ymm5 = ymm3[1],ymm5[1],ymm3[3],ymm5[3]+ vpunpcklqdq %ymm10, %ymm8, %ymm3 # ymm3 = ymm8[0],ymm10[0],ymm8[2],ymm10[2]+ vpunpckhqdq %ymm10, %ymm8, %ymm10 # ymm10 = ymm8[1],ymm10[1],ymm8[3],ymm10[3]+ vpunpcklqdq %ymm6, %ymm4, %ymm8 # ymm8 = ymm4[0],ymm6[0],ymm4[2],ymm6[2]+ vpunpckhqdq %ymm6, %ymm4, %ymm6 # ymm6 = ymm4[1],ymm6[1],ymm4[3],ymm6[3]+ vpunpcklqdq %ymm11, %ymm9, %ymm4 # ymm4 = ymm9[0],ymm11[0],ymm9[2],ymm11[2]+ vpunpckhqdq %ymm11, %ymm9, %ymm11 # ymm11 = ymm9[1],ymm11[1],ymm9[3],ymm11[3]+ vmovdqa 0x180(%rsi), %ymm1+ vmovdqa 0x620(%rsi), %ymm2+ vpmuldq %ymm1, %ymm8, %ymm13+ vmovshdup %ymm8, %ymm12 # ymm12 = ymm8[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm8, %ymm8 # ymm8 = ymm8[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm8, %ymm8 # ymm8 = ymm8[0],ymm12[1],ymm8[2],ymm12[3],ymm8[4],ymm12[5],ymm8[6],ymm12[7]+ vpsubd %ymm8, %ymm7, %ymm12+ vpaddd %ymm7, %ymm8, %ymm7+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm8+ vpsubd %ymm13, %ymm7, %ymm7+ vpmuldq %ymm1, %ymm6, %ymm13+ vmovshdup %ymm6, %ymm12 # ymm12 = ymm6[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm6, %ymm6+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm6, %ymm6 # ymm6 = ymm6[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm6, %ymm6 # ymm6 = ymm6[0],ymm12[1],ymm6[2],ymm12[3],ymm6[4],ymm12[5],ymm6[6],ymm12[7]+ vpsubd %ymm6, %ymm5, %ymm12+ vpaddd %ymm6, %ymm5, %ymm5+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm6+ vpsubd %ymm13, %ymm5, %ymm5+ vpmuldq %ymm1, %ymm4, %ymm13+ vmovshdup %ymm4, %ymm12 # ymm12 = ymm4[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm4, %ymm4+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm4, %ymm4 # ymm4 = ymm4[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm4, %ymm4 # ymm4 = ymm4[0],ymm12[1],ymm4[2],ymm12[3],ymm4[4],ymm12[5],ymm4[6],ymm12[7]+ vpsubd %ymm4, %ymm3, %ymm12+ vpaddd %ymm4, %ymm3, %ymm3+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm4+ vpsubd %ymm13, %ymm3, %ymm3+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm2, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm10, %ymm12+ vpaddd %ymm11, %ymm10, %ymm10+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm10, %ymm10+ vmovsldup %ymm8, %ymm9 # ymm9 = ymm8[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm9, %ymm7, %ymm9 # ymm9 = ymm7[0],ymm9[1],ymm7[2],ymm9[3],ymm7[4],ymm9[5],ymm7[6],ymm9[7]+ vpsrlq $0x20, %ymm7, %ymm7+ vpblendd $0xaa, %ymm8, %ymm7, %ymm8 # ymm8 = ymm7[0],ymm8[1],ymm7[2],ymm8[3],ymm7[4],ymm8[5],ymm7[6],ymm8[7]+ vmovsldup %ymm6, %ymm7 # ymm7 = ymm6[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm7, %ymm5, %ymm7 # ymm7 = ymm5[0],ymm7[1],ymm5[2],ymm7[3],ymm5[4],ymm7[5],ymm5[6],ymm7[7]+ vpsrlq $0x20, %ymm5, %ymm5+ vpblendd $0xaa, %ymm6, %ymm5, %ymm6 # ymm6 = ymm5[0],ymm6[1],ymm5[2],ymm6[3],ymm5[4],ymm6[5],ymm5[6],ymm6[7]+ vmovsldup %ymm4, %ymm5 # ymm5 = ymm4[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm5, %ymm3, %ymm5 # ymm5 = ymm3[0],ymm5[1],ymm3[2],ymm5[3],ymm3[4],ymm5[5],ymm3[6],ymm5[7]+ vpsrlq $0x20, %ymm3, %ymm3+ vpblendd $0xaa, %ymm4, %ymm3, %ymm4 # ymm4 = ymm3[0],ymm4[1],ymm3[2],ymm4[3],ymm3[4],ymm4[5],ymm3[6],ymm4[7]+ vmovsldup %ymm11, %ymm3 # ymm3 = ymm11[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm3, %ymm10, %ymm3 # ymm3 = ymm10[0],ymm3[1],ymm10[2],ymm3[3],ymm10[4],ymm3[5],ymm10[6],ymm3[7]+ vpsrlq $0x20, %ymm10, %ymm10+ vpblendd $0xaa, %ymm11, %ymm10, %ymm11 # ymm11 = ymm10[0],ymm11[1],ymm10[2],ymm11[3],ymm10[4],ymm11[5],ymm10[6],ymm11[7]+ vmovdqa 0x200(%rsi), %ymm1+ vmovdqa 0x6a0(%rsi), %ymm2+ vpsrlq $0x20, %ymm1, %ymm10+ vmovshdup %ymm2, %ymm15 # ymm15 = ymm2[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm5, %ymm13+ vmovshdup %ymm5, %ymm12 # ymm12 = ymm5[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm5, %ymm5+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm5, %ymm5 # ymm5 = ymm5[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm5, %ymm5 # ymm5 = ymm5[0],ymm12[1],ymm5[2],ymm12[3],ymm5[4],ymm12[5],ymm5[6],ymm12[7]+ vpsubd %ymm5, %ymm9, %ymm12+ vpaddd %ymm5, %ymm9, %ymm9+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm5+ vpsubd %ymm13, %ymm9, %ymm9+ vpmuldq %ymm1, %ymm4, %ymm13+ vmovshdup %ymm4, %ymm12 # ymm12 = ymm4[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm4, %ymm4+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm4, %ymm4 # ymm4 = ymm4[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm4, %ymm4 # ymm4 = ymm4[0],ymm12[1],ymm4[2],ymm12[3],ymm4[4],ymm12[5],ymm4[6],ymm12[7]+ vpsubd %ymm4, %ymm8, %ymm12+ vpaddd %ymm4, %ymm8, %ymm8+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm4+ vpsubd %ymm13, %ymm8, %ymm8+ vpmuldq %ymm1, %ymm3, %ymm13+ vmovshdup %ymm3, %ymm12 # ymm12 = ymm3[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm3, %ymm3+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm3, %ymm3 # ymm3 = ymm3[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm3, %ymm3 # ymm3 = ymm3[0],ymm12[1],ymm3[2],ymm12[3],ymm3[4],ymm12[5],ymm3[6],ymm12[7]+ vpsubd %ymm3, %ymm7, %ymm12+ vpaddd %ymm3, %ymm7, %ymm7+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm3+ vpsubd %ymm13, %ymm7, %ymm7+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm6, %ymm12+ vpaddd %ymm6, %ymm11, %ymm6+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm6, %ymm6+ vmovdqa 0x280(%rsi), %ymm1+ vmovdqa 0x720(%rsi), %ymm2+ vpsrlq $0x20, %ymm1, %ymm10+ vmovshdup %ymm2, %ymm15 # ymm15 = ymm2[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm7, %ymm13+ vmovshdup %ymm7, %ymm12 # ymm12 = ymm7[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm7, %ymm7+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm7, %ymm7 # ymm7 = ymm7[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm7, %ymm7 # ymm7 = ymm7[0],ymm12[1],ymm7[2],ymm12[3],ymm7[4],ymm12[5],ymm7[6],ymm12[7]+ vpsubd %ymm7, %ymm9, %ymm12+ vpaddd %ymm7, %ymm9, %ymm9+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm7+ vpsubd %ymm13, %ymm9, %ymm9+ vpmuldq %ymm1, %ymm6, %ymm13+ vmovshdup %ymm6, %ymm12 # ymm12 = ymm6[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm6, %ymm6+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm6, %ymm6 # ymm6 = ymm6[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm6, %ymm6 # ymm6 = ymm6[0],ymm12[1],ymm6[2],ymm12[3],ymm6[4],ymm12[5],ymm6[6],ymm12[7]+ vpsubd %ymm6, %ymm8, %ymm12+ vpaddd %ymm6, %ymm8, %ymm8+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm6+ vpsubd %ymm13, %ymm8, %ymm8+ vmovdqa 0x300(%rsi), %ymm1+ vmovdqa 0x7a0(%rsi), %ymm2+ vpsrlq $0x20, %ymm1, %ymm10+ vmovshdup %ymm2, %ymm15 # ymm15 = ymm2[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm3, %ymm13+ vmovshdup %ymm3, %ymm12 # ymm12 = ymm3[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm3, %ymm3+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm3, %ymm3 # ymm3 = ymm3[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm3, %ymm3 # ymm3 = ymm3[0],ymm12[1],ymm3[2],ymm12[3],ymm3[4],ymm12[5],ymm3[6],ymm12[7]+ vpsubd %ymm3, %ymm5, %ymm12+ vpaddd %ymm3, %ymm5, %ymm5+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm3+ vpsubd %ymm13, %ymm5, %ymm5+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm4, %ymm12+ vpaddd %ymm4, %ymm11, %ymm4+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm4, %ymm4+ vmovdqa 0x380(%rsi), %ymm1+ vmovdqa 0x820(%rsi), %ymm2+ vpsrlq $0x20, %ymm1, %ymm10+ vmovshdup %ymm2, %ymm15 # ymm15 = ymm2[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm8, %ymm13+ vmovshdup %ymm8, %ymm12 # ymm12 = ymm8[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm8, %ymm8+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm8, %ymm8 # ymm8 = ymm8[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm8, %ymm8 # ymm8 = ymm8[0],ymm12[1],ymm8[2],ymm12[3],ymm8[4],ymm12[5],ymm8[6],ymm12[7]+ vpsubd %ymm8, %ymm9, %ymm12+ vpaddd %ymm8, %ymm9, %ymm9+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm8+ vpsubd %ymm13, %ymm9, %ymm9+ vmovdqa 0x400(%rsi), %ymm1+ vmovdqa 0x8a0(%rsi), %ymm2+ vpsrlq $0x20, %ymm1, %ymm10+ vmovshdup %ymm2, %ymm15 # ymm15 = ymm2[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm6, %ymm13+ vmovshdup %ymm6, %ymm12 # ymm12 = ymm6[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm6, %ymm6+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm6, %ymm6 # ymm6 = ymm6[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm6, %ymm6 # ymm6 = ymm6[0],ymm12[1],ymm6[2],ymm12[3],ymm6[4],ymm12[5],ymm6[6],ymm12[7]+ vpsubd %ymm6, %ymm7, %ymm12+ vpaddd %ymm6, %ymm7, %ymm7+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm6+ vpsubd %ymm13, %ymm7, %ymm7+ vmovdqa 0x480(%rsi), %ymm1+ vmovdqa 0x920(%rsi), %ymm2+ vpsrlq $0x20, %ymm1, %ymm10+ vmovshdup %ymm2, %ymm15 # ymm15 = ymm2[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm4, %ymm13+ vmovshdup %ymm4, %ymm12 # ymm12 = ymm4[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm4, %ymm4+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm4, %ymm4 # ymm4 = ymm4[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm4, %ymm4 # ymm4 = ymm4[0],ymm12[1],ymm4[2],ymm12[3],ymm4[4],ymm12[5],ymm4[6],ymm12[7]+ vpsubd %ymm4, %ymm5, %ymm12+ vpaddd %ymm4, %ymm5, %ymm5+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm4+ vpsubd %ymm13, %ymm5, %ymm5+ vmovdqa 0x500(%rsi), %ymm1+ vmovdqa 0x9a0(%rsi), %ymm2+ vpsrlq $0x20, %ymm1, %ymm10+ vmovshdup %ymm2, %ymm15 # ymm15 = ymm2[1,1,3,3,5,5,7,7]+ vpmuldq %ymm1, %ymm11, %ymm13+ vmovshdup %ymm11, %ymm12 # ymm12 = ymm11[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm12, %ymm14+ vpmuldq %ymm2, %ymm11, %ymm11+ vpmuldq %ymm15, %ymm12, %ymm12+ vpmuldq %ymm0, %ymm13, %ymm13+ vpmuldq %ymm0, %ymm14, %ymm14+ vmovshdup %ymm11, %ymm11 # ymm11 = ymm11[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm12, %ymm11, %ymm11 # ymm11 = ymm11[0],ymm12[1],ymm11[2],ymm12[3],ymm11[4],ymm12[5],ymm11[6],ymm12[7]+ vpsubd %ymm11, %ymm3, %ymm12+ vpaddd %ymm3, %ymm11, %ymm3+ vmovshdup %ymm13, %ymm13 # ymm13 = ymm13[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm14, %ymm13, %ymm13 # ymm13 = ymm13[0],ymm14[1],ymm13[2],ymm14[3],ymm13[4],ymm14[5],ymm13[6],ymm14[7]+ vpaddd %ymm13, %ymm12, %ymm11+ vpsubd %ymm13, %ymm3, %ymm3+ vmovdqa %ymm9, 0x300(%rdi)+ vmovdqa %ymm8, 0x320(%rdi)+ vmovdqa %ymm7, 0x340(%rdi)+ vmovdqa %ymm6, 0x360(%rdi)+ vmovdqa %ymm5, 0x380(%rdi)+ vmovdqa %ymm4, 0x3a0(%rdi)+ vmovdqa %ymm3, 0x3c0(%rdi)+ vmovdqa %ymm11, 0x3e0(%rdi)+ retq+ .cfi_endproc++MLD_ASM_FN_SIZE(ntt_avx2_asm)++#endif /* MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_MULTILEVEL_NO_SHARED \+ */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
@@ -0,0 +1,254 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [REF_AVX2]+ * CRYSTALS-Dilithium optimized AVX2 implementation+ * Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/dilithium/tree/master/avx2+ */++/*+ * This file is derived from the public domain+ * AVX2 Dilithium implementation @[REF_AVX2].+ */++/*yaml+ Name: nttunpack_avx2_asm+ Description: x86_64 AVX2 NTT coefficient unpacking/permutation+ Signature: void mld_nttunpack_avx2_asm(int32_t *r)+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 1024+ permissions: read/write+ c_parameter: int32_t *r+ description: Input/output polynomial (256 x int32_t)+*/++#include "../../../common.h"+#if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/x86_64/src/mldsa_nttunpack_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(nttunpack_avx2_asm)+MLD_ASM_FN_SYMBOL(nttunpack_avx2_asm)++ .cfi_startproc+ vmovdqa (%rdi), %ymm4+ vmovdqa 0x20(%rdi), %ymm5+ vmovdqa 0x40(%rdi), %ymm6+ vmovdqa 0x60(%rdi), %ymm7+ vmovdqa 0x80(%rdi), %ymm8+ vmovdqa 0xa0(%rdi), %ymm9+ vmovdqa 0xc0(%rdi), %ymm10+ vmovdqa 0xe0(%rdi), %ymm11+ vperm2i128 $0x20, %ymm8, %ymm4, %ymm3 # ymm3 = ymm4[0,1],ymm8[0,1]+ vperm2i128 $0x31, %ymm8, %ymm4, %ymm8 # ymm8 = ymm4[2,3],ymm8[2,3]+ vperm2i128 $0x20, %ymm9, %ymm5, %ymm4 # ymm4 = ymm5[0,1],ymm9[0,1]+ vperm2i128 $0x31, %ymm9, %ymm5, %ymm9 # ymm9 = ymm5[2,3],ymm9[2,3]+ vperm2i128 $0x20, %ymm10, %ymm6, %ymm5 # ymm5 = ymm6[0,1],ymm10[0,1]+ vperm2i128 $0x31, %ymm10, %ymm6, %ymm10 # ymm10 = ymm6[2,3],ymm10[2,3]+ vperm2i128 $0x20, %ymm11, %ymm7, %ymm6 # ymm6 = ymm7[0,1],ymm11[0,1]+ vperm2i128 $0x31, %ymm11, %ymm7, %ymm11 # ymm11 = ymm7[2,3],ymm11[2,3]+ vpunpcklqdq %ymm5, %ymm3, %ymm7 # ymm7 = ymm3[0],ymm5[0],ymm3[2],ymm5[2]+ vpunpckhqdq %ymm5, %ymm3, %ymm5 # ymm5 = ymm3[1],ymm5[1],ymm3[3],ymm5[3]+ vpunpcklqdq %ymm10, %ymm8, %ymm3 # ymm3 = ymm8[0],ymm10[0],ymm8[2],ymm10[2]+ vpunpckhqdq %ymm10, %ymm8, %ymm10 # ymm10 = ymm8[1],ymm10[1],ymm8[3],ymm10[3]+ vpunpcklqdq %ymm6, %ymm4, %ymm8 # ymm8 = ymm4[0],ymm6[0],ymm4[2],ymm6[2]+ vpunpckhqdq %ymm6, %ymm4, %ymm6 # ymm6 = ymm4[1],ymm6[1],ymm4[3],ymm6[3]+ vpunpcklqdq %ymm11, %ymm9, %ymm4 # ymm4 = ymm9[0],ymm11[0],ymm9[2],ymm11[2]+ vpunpckhqdq %ymm11, %ymm9, %ymm11 # ymm11 = ymm9[1],ymm11[1],ymm9[3],ymm11[3]+ vmovsldup %ymm8, %ymm9 # ymm9 = ymm8[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm9, %ymm7, %ymm9 # ymm9 = ymm7[0],ymm9[1],ymm7[2],ymm9[3],ymm7[4],ymm9[5],ymm7[6],ymm9[7]+ vpsrlq $0x20, %ymm7, %ymm7+ vpblendd $0xaa, %ymm8, %ymm7, %ymm8 # ymm8 = ymm7[0],ymm8[1],ymm7[2],ymm8[3],ymm7[4],ymm8[5],ymm7[6],ymm8[7]+ vmovsldup %ymm6, %ymm7 # ymm7 = ymm6[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm7, %ymm5, %ymm7 # ymm7 = ymm5[0],ymm7[1],ymm5[2],ymm7[3],ymm5[4],ymm7[5],ymm5[6],ymm7[7]+ vpsrlq $0x20, %ymm5, %ymm5+ vpblendd $0xaa, %ymm6, %ymm5, %ymm6 # ymm6 = ymm5[0],ymm6[1],ymm5[2],ymm6[3],ymm5[4],ymm6[5],ymm5[6],ymm6[7]+ vmovsldup %ymm4, %ymm5 # ymm5 = ymm4[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm5, %ymm3, %ymm5 # ymm5 = ymm3[0],ymm5[1],ymm3[2],ymm5[3],ymm3[4],ymm5[5],ymm3[6],ymm5[7]+ vpsrlq $0x20, %ymm3, %ymm3+ vpblendd $0xaa, %ymm4, %ymm3, %ymm4 # ymm4 = ymm3[0],ymm4[1],ymm3[2],ymm4[3],ymm3[4],ymm4[5],ymm3[6],ymm4[7]+ vmovsldup %ymm11, %ymm3 # ymm3 = ymm11[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm3, %ymm10, %ymm3 # ymm3 = ymm10[0],ymm3[1],ymm10[2],ymm3[3],ymm10[4],ymm3[5],ymm10[6],ymm3[7]+ vpsrlq $0x20, %ymm10, %ymm10+ vpblendd $0xaa, %ymm11, %ymm10, %ymm11 # ymm11 = ymm10[0],ymm11[1],ymm10[2],ymm11[3],ymm10[4],ymm11[5],ymm10[6],ymm11[7]+ vmovdqa %ymm9, (%rdi)+ vmovdqa %ymm8, 0x20(%rdi)+ vmovdqa %ymm7, 0x40(%rdi)+ vmovdqa %ymm6, 0x60(%rdi)+ vmovdqa %ymm5, 0x80(%rdi)+ vmovdqa %ymm4, 0xa0(%rdi)+ vmovdqa %ymm3, 0xc0(%rdi)+ vmovdqa %ymm11, 0xe0(%rdi)+ vmovdqa 0x100(%rdi), %ymm4+ vmovdqa 0x120(%rdi), %ymm5+ vmovdqa 0x140(%rdi), %ymm6+ vmovdqa 0x160(%rdi), %ymm7+ vmovdqa 0x180(%rdi), %ymm8+ vmovdqa 0x1a0(%rdi), %ymm9+ vmovdqa 0x1c0(%rdi), %ymm10+ vmovdqa 0x1e0(%rdi), %ymm11+ vperm2i128 $0x20, %ymm8, %ymm4, %ymm3 # ymm3 = ymm4[0,1],ymm8[0,1]+ vperm2i128 $0x31, %ymm8, %ymm4, %ymm8 # ymm8 = ymm4[2,3],ymm8[2,3]+ vperm2i128 $0x20, %ymm9, %ymm5, %ymm4 # ymm4 = ymm5[0,1],ymm9[0,1]+ vperm2i128 $0x31, %ymm9, %ymm5, %ymm9 # ymm9 = ymm5[2,3],ymm9[2,3]+ vperm2i128 $0x20, %ymm10, %ymm6, %ymm5 # ymm5 = ymm6[0,1],ymm10[0,1]+ vperm2i128 $0x31, %ymm10, %ymm6, %ymm10 # ymm10 = ymm6[2,3],ymm10[2,3]+ vperm2i128 $0x20, %ymm11, %ymm7, %ymm6 # ymm6 = ymm7[0,1],ymm11[0,1]+ vperm2i128 $0x31, %ymm11, %ymm7, %ymm11 # ymm11 = ymm7[2,3],ymm11[2,3]+ vpunpcklqdq %ymm5, %ymm3, %ymm7 # ymm7 = ymm3[0],ymm5[0],ymm3[2],ymm5[2]+ vpunpckhqdq %ymm5, %ymm3, %ymm5 # ymm5 = ymm3[1],ymm5[1],ymm3[3],ymm5[3]+ vpunpcklqdq %ymm10, %ymm8, %ymm3 # ymm3 = ymm8[0],ymm10[0],ymm8[2],ymm10[2]+ vpunpckhqdq %ymm10, %ymm8, %ymm10 # ymm10 = ymm8[1],ymm10[1],ymm8[3],ymm10[3]+ vpunpcklqdq %ymm6, %ymm4, %ymm8 # ymm8 = ymm4[0],ymm6[0],ymm4[2],ymm6[2]+ vpunpckhqdq %ymm6, %ymm4, %ymm6 # ymm6 = ymm4[1],ymm6[1],ymm4[3],ymm6[3]+ vpunpcklqdq %ymm11, %ymm9, %ymm4 # ymm4 = ymm9[0],ymm11[0],ymm9[2],ymm11[2]+ vpunpckhqdq %ymm11, %ymm9, %ymm11 # ymm11 = ymm9[1],ymm11[1],ymm9[3],ymm11[3]+ vmovsldup %ymm8, %ymm9 # ymm9 = ymm8[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm9, %ymm7, %ymm9 # ymm9 = ymm7[0],ymm9[1],ymm7[2],ymm9[3],ymm7[4],ymm9[5],ymm7[6],ymm9[7]+ vpsrlq $0x20, %ymm7, %ymm7+ vpblendd $0xaa, %ymm8, %ymm7, %ymm8 # ymm8 = ymm7[0],ymm8[1],ymm7[2],ymm8[3],ymm7[4],ymm8[5],ymm7[6],ymm8[7]+ vmovsldup %ymm6, %ymm7 # ymm7 = ymm6[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm7, %ymm5, %ymm7 # ymm7 = ymm5[0],ymm7[1],ymm5[2],ymm7[3],ymm5[4],ymm7[5],ymm5[6],ymm7[7]+ vpsrlq $0x20, %ymm5, %ymm5+ vpblendd $0xaa, %ymm6, %ymm5, %ymm6 # ymm6 = ymm5[0],ymm6[1],ymm5[2],ymm6[3],ymm5[4],ymm6[5],ymm5[6],ymm6[7]+ vmovsldup %ymm4, %ymm5 # ymm5 = ymm4[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm5, %ymm3, %ymm5 # ymm5 = ymm3[0],ymm5[1],ymm3[2],ymm5[3],ymm3[4],ymm5[5],ymm3[6],ymm5[7]+ vpsrlq $0x20, %ymm3, %ymm3+ vpblendd $0xaa, %ymm4, %ymm3, %ymm4 # ymm4 = ymm3[0],ymm4[1],ymm3[2],ymm4[3],ymm3[4],ymm4[5],ymm3[6],ymm4[7]+ vmovsldup %ymm11, %ymm3 # ymm3 = ymm11[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm3, %ymm10, %ymm3 # ymm3 = ymm10[0],ymm3[1],ymm10[2],ymm3[3],ymm10[4],ymm3[5],ymm10[6],ymm3[7]+ vpsrlq $0x20, %ymm10, %ymm10+ vpblendd $0xaa, %ymm11, %ymm10, %ymm11 # ymm11 = ymm10[0],ymm11[1],ymm10[2],ymm11[3],ymm10[4],ymm11[5],ymm10[6],ymm11[7]+ vmovdqa %ymm9, 0x100(%rdi)+ vmovdqa %ymm8, 0x120(%rdi)+ vmovdqa %ymm7, 0x140(%rdi)+ vmovdqa %ymm6, 0x160(%rdi)+ vmovdqa %ymm5, 0x180(%rdi)+ vmovdqa %ymm4, 0x1a0(%rdi)+ vmovdqa %ymm3, 0x1c0(%rdi)+ vmovdqa %ymm11, 0x1e0(%rdi)+ vmovdqa 0x200(%rdi), %ymm4+ vmovdqa 0x220(%rdi), %ymm5+ vmovdqa 0x240(%rdi), %ymm6+ vmovdqa 0x260(%rdi), %ymm7+ vmovdqa 0x280(%rdi), %ymm8+ vmovdqa 0x2a0(%rdi), %ymm9+ vmovdqa 0x2c0(%rdi), %ymm10+ vmovdqa 0x2e0(%rdi), %ymm11+ vperm2i128 $0x20, %ymm8, %ymm4, %ymm3 # ymm3 = ymm4[0,1],ymm8[0,1]+ vperm2i128 $0x31, %ymm8, %ymm4, %ymm8 # ymm8 = ymm4[2,3],ymm8[2,3]+ vperm2i128 $0x20, %ymm9, %ymm5, %ymm4 # ymm4 = ymm5[0,1],ymm9[0,1]+ vperm2i128 $0x31, %ymm9, %ymm5, %ymm9 # ymm9 = ymm5[2,3],ymm9[2,3]+ vperm2i128 $0x20, %ymm10, %ymm6, %ymm5 # ymm5 = ymm6[0,1],ymm10[0,1]+ vperm2i128 $0x31, %ymm10, %ymm6, %ymm10 # ymm10 = ymm6[2,3],ymm10[2,3]+ vperm2i128 $0x20, %ymm11, %ymm7, %ymm6 # ymm6 = ymm7[0,1],ymm11[0,1]+ vperm2i128 $0x31, %ymm11, %ymm7, %ymm11 # ymm11 = ymm7[2,3],ymm11[2,3]+ vpunpcklqdq %ymm5, %ymm3, %ymm7 # ymm7 = ymm3[0],ymm5[0],ymm3[2],ymm5[2]+ vpunpckhqdq %ymm5, %ymm3, %ymm5 # ymm5 = ymm3[1],ymm5[1],ymm3[3],ymm5[3]+ vpunpcklqdq %ymm10, %ymm8, %ymm3 # ymm3 = ymm8[0],ymm10[0],ymm8[2],ymm10[2]+ vpunpckhqdq %ymm10, %ymm8, %ymm10 # ymm10 = ymm8[1],ymm10[1],ymm8[3],ymm10[3]+ vpunpcklqdq %ymm6, %ymm4, %ymm8 # ymm8 = ymm4[0],ymm6[0],ymm4[2],ymm6[2]+ vpunpckhqdq %ymm6, %ymm4, %ymm6 # ymm6 = ymm4[1],ymm6[1],ymm4[3],ymm6[3]+ vpunpcklqdq %ymm11, %ymm9, %ymm4 # ymm4 = ymm9[0],ymm11[0],ymm9[2],ymm11[2]+ vpunpckhqdq %ymm11, %ymm9, %ymm11 # ymm11 = ymm9[1],ymm11[1],ymm9[3],ymm11[3]+ vmovsldup %ymm8, %ymm9 # ymm9 = ymm8[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm9, %ymm7, %ymm9 # ymm9 = ymm7[0],ymm9[1],ymm7[2],ymm9[3],ymm7[4],ymm9[5],ymm7[6],ymm9[7]+ vpsrlq $0x20, %ymm7, %ymm7+ vpblendd $0xaa, %ymm8, %ymm7, %ymm8 # ymm8 = ymm7[0],ymm8[1],ymm7[2],ymm8[3],ymm7[4],ymm8[5],ymm7[6],ymm8[7]+ vmovsldup %ymm6, %ymm7 # ymm7 = ymm6[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm7, %ymm5, %ymm7 # ymm7 = ymm5[0],ymm7[1],ymm5[2],ymm7[3],ymm5[4],ymm7[5],ymm5[6],ymm7[7]+ vpsrlq $0x20, %ymm5, %ymm5+ vpblendd $0xaa, %ymm6, %ymm5, %ymm6 # ymm6 = ymm5[0],ymm6[1],ymm5[2],ymm6[3],ymm5[4],ymm6[5],ymm5[6],ymm6[7]+ vmovsldup %ymm4, %ymm5 # ymm5 = ymm4[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm5, %ymm3, %ymm5 # ymm5 = ymm3[0],ymm5[1],ymm3[2],ymm5[3],ymm3[4],ymm5[5],ymm3[6],ymm5[7]+ vpsrlq $0x20, %ymm3, %ymm3+ vpblendd $0xaa, %ymm4, %ymm3, %ymm4 # ymm4 = ymm3[0],ymm4[1],ymm3[2],ymm4[3],ymm3[4],ymm4[5],ymm3[6],ymm4[7]+ vmovsldup %ymm11, %ymm3 # ymm3 = ymm11[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm3, %ymm10, %ymm3 # ymm3 = ymm10[0],ymm3[1],ymm10[2],ymm3[3],ymm10[4],ymm3[5],ymm10[6],ymm3[7]+ vpsrlq $0x20, %ymm10, %ymm10+ vpblendd $0xaa, %ymm11, %ymm10, %ymm11 # ymm11 = ymm10[0],ymm11[1],ymm10[2],ymm11[3],ymm10[4],ymm11[5],ymm10[6],ymm11[7]+ vmovdqa %ymm9, 0x200(%rdi)+ vmovdqa %ymm8, 0x220(%rdi)+ vmovdqa %ymm7, 0x240(%rdi)+ vmovdqa %ymm6, 0x260(%rdi)+ vmovdqa %ymm5, 0x280(%rdi)+ vmovdqa %ymm4, 0x2a0(%rdi)+ vmovdqa %ymm3, 0x2c0(%rdi)+ vmovdqa %ymm11, 0x2e0(%rdi)+ vmovdqa 0x300(%rdi), %ymm4+ vmovdqa 0x320(%rdi), %ymm5+ vmovdqa 0x340(%rdi), %ymm6+ vmovdqa 0x360(%rdi), %ymm7+ vmovdqa 0x380(%rdi), %ymm8+ vmovdqa 0x3a0(%rdi), %ymm9+ vmovdqa 0x3c0(%rdi), %ymm10+ vmovdqa 0x3e0(%rdi), %ymm11+ vperm2i128 $0x20, %ymm8, %ymm4, %ymm3 # ymm3 = ymm4[0,1],ymm8[0,1]+ vperm2i128 $0x31, %ymm8, %ymm4, %ymm8 # ymm8 = ymm4[2,3],ymm8[2,3]+ vperm2i128 $0x20, %ymm9, %ymm5, %ymm4 # ymm4 = ymm5[0,1],ymm9[0,1]+ vperm2i128 $0x31, %ymm9, %ymm5, %ymm9 # ymm9 = ymm5[2,3],ymm9[2,3]+ vperm2i128 $0x20, %ymm10, %ymm6, %ymm5 # ymm5 = ymm6[0,1],ymm10[0,1]+ vperm2i128 $0x31, %ymm10, %ymm6, %ymm10 # ymm10 = ymm6[2,3],ymm10[2,3]+ vperm2i128 $0x20, %ymm11, %ymm7, %ymm6 # ymm6 = ymm7[0,1],ymm11[0,1]+ vperm2i128 $0x31, %ymm11, %ymm7, %ymm11 # ymm11 = ymm7[2,3],ymm11[2,3]+ vpunpcklqdq %ymm5, %ymm3, %ymm7 # ymm7 = ymm3[0],ymm5[0],ymm3[2],ymm5[2]+ vpunpckhqdq %ymm5, %ymm3, %ymm5 # ymm5 = ymm3[1],ymm5[1],ymm3[3],ymm5[3]+ vpunpcklqdq %ymm10, %ymm8, %ymm3 # ymm3 = ymm8[0],ymm10[0],ymm8[2],ymm10[2]+ vpunpckhqdq %ymm10, %ymm8, %ymm10 # ymm10 = ymm8[1],ymm10[1],ymm8[3],ymm10[3]+ vpunpcklqdq %ymm6, %ymm4, %ymm8 # ymm8 = ymm4[0],ymm6[0],ymm4[2],ymm6[2]+ vpunpckhqdq %ymm6, %ymm4, %ymm6 # ymm6 = ymm4[1],ymm6[1],ymm4[3],ymm6[3]+ vpunpcklqdq %ymm11, %ymm9, %ymm4 # ymm4 = ymm9[0],ymm11[0],ymm9[2],ymm11[2]+ vpunpckhqdq %ymm11, %ymm9, %ymm11 # ymm11 = ymm9[1],ymm11[1],ymm9[3],ymm11[3]+ vmovsldup %ymm8, %ymm9 # ymm9 = ymm8[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm9, %ymm7, %ymm9 # ymm9 = ymm7[0],ymm9[1],ymm7[2],ymm9[3],ymm7[4],ymm9[5],ymm7[6],ymm9[7]+ vpsrlq $0x20, %ymm7, %ymm7+ vpblendd $0xaa, %ymm8, %ymm7, %ymm8 # ymm8 = ymm7[0],ymm8[1],ymm7[2],ymm8[3],ymm7[4],ymm8[5],ymm7[6],ymm8[7]+ vmovsldup %ymm6, %ymm7 # ymm7 = ymm6[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm7, %ymm5, %ymm7 # ymm7 = ymm5[0],ymm7[1],ymm5[2],ymm7[3],ymm5[4],ymm7[5],ymm5[6],ymm7[7]+ vpsrlq $0x20, %ymm5, %ymm5+ vpblendd $0xaa, %ymm6, %ymm5, %ymm6 # ymm6 = ymm5[0],ymm6[1],ymm5[2],ymm6[3],ymm5[4],ymm6[5],ymm5[6],ymm6[7]+ vmovsldup %ymm4, %ymm5 # ymm5 = ymm4[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm5, %ymm3, %ymm5 # ymm5 = ymm3[0],ymm5[1],ymm3[2],ymm5[3],ymm3[4],ymm5[5],ymm3[6],ymm5[7]+ vpsrlq $0x20, %ymm3, %ymm3+ vpblendd $0xaa, %ymm4, %ymm3, %ymm4 # ymm4 = ymm3[0],ymm4[1],ymm3[2],ymm4[3],ymm3[4],ymm4[5],ymm3[6],ymm4[7]+ vmovsldup %ymm11, %ymm3 # ymm3 = ymm11[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm3, %ymm10, %ymm3 # ymm3 = ymm10[0],ymm3[1],ymm10[2],ymm3[3],ymm10[4],ymm3[5],ymm10[6],ymm3[7]+ vpsrlq $0x20, %ymm10, %ymm10+ vpblendd $0xaa, %ymm11, %ymm10, %ymm11 # ymm11 = ymm10[0],ymm11[1],ymm10[2],ymm11[3],ymm10[4],ymm11[5],ymm10[6],ymm11[7]+ vmovdqa %ymm9, 0x300(%rdi)+ vmovdqa %ymm8, 0x320(%rdi)+ vmovdqa %ymm7, 0x340(%rdi)+ vmovdqa %ymm6, 0x360(%rdi)+ vmovdqa %ymm5, 0x380(%rdi)+ vmovdqa %ymm4, 0x3a0(%rdi)+ vmovdqa %ymm3, 0x3c0(%rdi)+ vmovdqa %ymm11, 0x3e0(%rdi)+ retq+ .cfi_endproc++MLD_ASM_FN_SIZE(nttunpack_avx2_asm)++#endif /* MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_MULTILEVEL_NO_SHARED \+ */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
@@ -0,0 +1,173 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [REF_AVX2]+ * CRYSTALS-Dilithium optimized AVX2 implementation+ * Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/dilithium/tree/master/avx2+ */++/*+ * This file is derived from the public domain+ * AVX2 Dilithium implementation @[REF_AVX2].+ */++/*yaml+ Name: pointwise_acc_l4_avx2_asm+ Description: x86_64 AVX2 pointwise multiply-accumulate of length-4 polynomial vectors+ Signature: void mld_pointwise_acc_l4_avx2_asm(int32_t *c, const int32_t a[4][256], const int32_t b[4][256], const int32_t *qdata)+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 1024+ permissions: write-only+ c_parameter: int32_t *c+ description: Output polynomial (256 x int32_t)+ rsi:+ type: buffer+ size_bytes: 4096+ permissions: read-only+ c_parameter: const int32_t a[4][256]+ description: Input polynomial vector a (4 x 256 x int32_t)+ rdx:+ type: buffer+ size_bytes: 4096+ permissions: read-only+ c_parameter: const int32_t b[4][256]+ description: Input polynomial vector b (4 x 256 x int32_t)+ rcx:+ type: buffer+ size_bytes: 2496+ permissions: read-only+ c_parameter: const int32_t *qdata+ description: Precomputed constants (624 x int32_t)+*/++#include "../../../common.h"+#if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \+ (defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLDSA_L == 4)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/x86_64/src/mldsa_pointwise_acc_l4_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(pointwise_acc_l4_avx2_asm)+MLD_ASM_FN_SYMBOL(pointwise_acc_l4_avx2_asm)++ .cfi_startproc+ vmovdqa 0x20(%rcx), %ymm0+ vmovdqa (%rcx), %ymm1+ xorl %eax, %eax++Lmld_pointwise_acc_l4_avx2_looptop2:+ vmovdqa (%rsi), %ymm6+ vmovdqa 0x20(%rsi), %ymm8+ vmovdqa (%rdx), %ymm10+ vmovdqa 0x20(%rdx), %ymm12+ vpsrlq $0x20, %ymm6, %ymm7+ vpsrlq $0x20, %ymm8, %ymm9+ vmovshdup %ymm10, %ymm11 # ymm11 = ymm10[1,1,3,3,5,5,7,7]+ vmovshdup %ymm12, %ymm13 # ymm13 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm6, %ymm6+ vpmuldq %ymm11, %ymm7, %ymm7+ vpmuldq %ymm12, %ymm8, %ymm8+ vpmuldq %ymm13, %ymm9, %ymm9+ vmovdqa %ymm6, %ymm2+ vmovdqa %ymm7, %ymm3+ vmovdqa %ymm8, %ymm4+ vmovdqa %ymm9, %ymm5+ vmovdqa 0x400(%rsi), %ymm6+ vmovdqa 0x420(%rsi), %ymm8+ vmovdqa 0x400(%rdx), %ymm10+ vmovdqa 0x420(%rdx), %ymm12+ vpsrlq $0x20, %ymm6, %ymm7+ vpsrlq $0x20, %ymm8, %ymm9+ vmovshdup %ymm10, %ymm11 # ymm11 = ymm10[1,1,3,3,5,5,7,7]+ vmovshdup %ymm12, %ymm13 # ymm13 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm6, %ymm6+ vpmuldq %ymm11, %ymm7, %ymm7+ vpmuldq %ymm12, %ymm8, %ymm8+ vpmuldq %ymm13, %ymm9, %ymm9+ vpaddq %ymm2, %ymm6, %ymm2+ vpaddq %ymm3, %ymm7, %ymm3+ vpaddq %ymm4, %ymm8, %ymm4+ vpaddq %ymm5, %ymm9, %ymm5+ vmovdqa 0x800(%rsi), %ymm6+ vmovdqa 0x820(%rsi), %ymm8+ vmovdqa 0x800(%rdx), %ymm10+ vmovdqa 0x820(%rdx), %ymm12+ vpsrlq $0x20, %ymm6, %ymm7+ vpsrlq $0x20, %ymm8, %ymm9+ vmovshdup %ymm10, %ymm11 # ymm11 = ymm10[1,1,3,3,5,5,7,7]+ vmovshdup %ymm12, %ymm13 # ymm13 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm6, %ymm6+ vpmuldq %ymm11, %ymm7, %ymm7+ vpmuldq %ymm12, %ymm8, %ymm8+ vpmuldq %ymm13, %ymm9, %ymm9+ vpaddq %ymm2, %ymm6, %ymm2+ vpaddq %ymm3, %ymm7, %ymm3+ vpaddq %ymm4, %ymm8, %ymm4+ vpaddq %ymm5, %ymm9, %ymm5+ vmovdqa 0xc00(%rsi), %ymm6+ vmovdqa 0xc20(%rsi), %ymm8+ vmovdqa 0xc00(%rdx), %ymm10+ vmovdqa 0xc20(%rdx), %ymm12+ vpsrlq $0x20, %ymm6, %ymm7+ vpsrlq $0x20, %ymm8, %ymm9+ vmovshdup %ymm10, %ymm11 # ymm11 = ymm10[1,1,3,3,5,5,7,7]+ vmovshdup %ymm12, %ymm13 # ymm13 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm6, %ymm6+ vpmuldq %ymm11, %ymm7, %ymm7+ vpmuldq %ymm12, %ymm8, %ymm8+ vpmuldq %ymm13, %ymm9, %ymm9+ vpaddq %ymm2, %ymm6, %ymm2+ vpaddq %ymm3, %ymm7, %ymm3+ vpaddq %ymm4, %ymm8, %ymm4+ vpaddq %ymm5, %ymm9, %ymm5+ vpmuldq %ymm2, %ymm0, %ymm6+ vpmuldq %ymm3, %ymm0, %ymm7+ vpmuldq %ymm4, %ymm0, %ymm8+ vpmuldq %ymm5, %ymm0, %ymm9+ vpmuldq %ymm6, %ymm1, %ymm6+ vpmuldq %ymm7, %ymm1, %ymm7+ vpmuldq %ymm8, %ymm1, %ymm8+ vpmuldq %ymm9, %ymm1, %ymm9+ vpsubq %ymm6, %ymm2, %ymm2+ vpsubq %ymm7, %ymm3, %ymm3+ vpsubq %ymm8, %ymm4, %ymm4+ vpsubq %ymm9, %ymm5, %ymm5+ vpsrlq $0x20, %ymm2, %ymm2+ vmovshdup %ymm4, %ymm4 # ymm4 = ymm4[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm3, %ymm2, %ymm2 # ymm2 = ymm2[0],ymm3[1],ymm2[2],ymm3[3],ymm2[4],ymm3[5],ymm2[6],ymm3[7]+ vpblendd $0xaa, %ymm5, %ymm4, %ymm4 # ymm4 = ymm4[0],ymm5[1],ymm4[2],ymm5[3],ymm4[4],ymm5[5],ymm4[6],ymm5[7]+ vmovdqa %ymm2, (%rdi)+ vmovdqa %ymm4, 0x20(%rdi)+ addq $0x40, %rsi+ addq $0x40, %rdx+ addq $0x40, %rdi+ addl $0x1, %eax+ cmpl $0x10, %eax+ jb Lmld_pointwise_acc_l4_avx2_looptop2+ retq+ .cfi_endproc++MLD_ASM_FN_SIZE(pointwise_acc_l4_avx2_asm)++#endif /* MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_MULTILEVEL_NO_SHARED \+ && (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLDSA_L == 4) */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
@@ -0,0 +1,189 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [REF_AVX2]+ * CRYSTALS-Dilithium optimized AVX2 implementation+ * Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/dilithium/tree/master/avx2+ */++/*+ * This file is derived from the public domain+ * AVX2 Dilithium implementation @[REF_AVX2].+ */++/*yaml+ Name: pointwise_acc_l5_avx2_asm+ Description: x86_64 AVX2 pointwise multiply-accumulate of length-5 polynomial vectors+ Signature: void mld_pointwise_acc_l5_avx2_asm(int32_t *c, const int32_t a[5][256], const int32_t b[5][256], const int32_t *qdata)+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 1024+ permissions: write-only+ c_parameter: int32_t *c+ description: Output polynomial (256 x int32_t)+ rsi:+ type: buffer+ size_bytes: 5120+ permissions: read-only+ c_parameter: const int32_t a[5][256]+ description: Input polynomial vector a (5 x 256 x int32_t)+ rdx:+ type: buffer+ size_bytes: 5120+ permissions: read-only+ c_parameter: const int32_t b[5][256]+ description: Input polynomial vector b (5 x 256 x int32_t)+ rcx:+ type: buffer+ size_bytes: 2496+ permissions: read-only+ c_parameter: const int32_t *qdata+ description: Precomputed constants (624 x int32_t)+*/++#include "../../../common.h"+#if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \+ (defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLDSA_L == 5)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/x86_64/src/mldsa_pointwise_acc_l5_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(pointwise_acc_l5_avx2_asm)+MLD_ASM_FN_SYMBOL(pointwise_acc_l5_avx2_asm)++ .cfi_startproc+ vmovdqa 0x20(%rcx), %ymm0+ vmovdqa (%rcx), %ymm1+ xorl %eax, %eax++Lmld_pointwise_acc_l5_avx2_looptop2:+ vmovdqa (%rsi), %ymm6+ vmovdqa 0x20(%rsi), %ymm8+ vmovdqa (%rdx), %ymm10+ vmovdqa 0x20(%rdx), %ymm12+ vpsrlq $0x20, %ymm6, %ymm7+ vpsrlq $0x20, %ymm8, %ymm9+ vmovshdup %ymm10, %ymm11 # ymm11 = ymm10[1,1,3,3,5,5,7,7]+ vmovshdup %ymm12, %ymm13 # ymm13 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm6, %ymm6+ vpmuldq %ymm11, %ymm7, %ymm7+ vpmuldq %ymm12, %ymm8, %ymm8+ vpmuldq %ymm13, %ymm9, %ymm9+ vmovdqa %ymm6, %ymm2+ vmovdqa %ymm7, %ymm3+ vmovdqa %ymm8, %ymm4+ vmovdqa %ymm9, %ymm5+ vmovdqa 0x400(%rsi), %ymm6+ vmovdqa 0x420(%rsi), %ymm8+ vmovdqa 0x400(%rdx), %ymm10+ vmovdqa 0x420(%rdx), %ymm12+ vpsrlq $0x20, %ymm6, %ymm7+ vpsrlq $0x20, %ymm8, %ymm9+ vmovshdup %ymm10, %ymm11 # ymm11 = ymm10[1,1,3,3,5,5,7,7]+ vmovshdup %ymm12, %ymm13 # ymm13 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm6, %ymm6+ vpmuldq %ymm11, %ymm7, %ymm7+ vpmuldq %ymm12, %ymm8, %ymm8+ vpmuldq %ymm13, %ymm9, %ymm9+ vpaddq %ymm2, %ymm6, %ymm2+ vpaddq %ymm3, %ymm7, %ymm3+ vpaddq %ymm4, %ymm8, %ymm4+ vpaddq %ymm5, %ymm9, %ymm5+ vmovdqa 0x800(%rsi), %ymm6+ vmovdqa 0x820(%rsi), %ymm8+ vmovdqa 0x800(%rdx), %ymm10+ vmovdqa 0x820(%rdx), %ymm12+ vpsrlq $0x20, %ymm6, %ymm7+ vpsrlq $0x20, %ymm8, %ymm9+ vmovshdup %ymm10, %ymm11 # ymm11 = ymm10[1,1,3,3,5,5,7,7]+ vmovshdup %ymm12, %ymm13 # ymm13 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm6, %ymm6+ vpmuldq %ymm11, %ymm7, %ymm7+ vpmuldq %ymm12, %ymm8, %ymm8+ vpmuldq %ymm13, %ymm9, %ymm9+ vpaddq %ymm2, %ymm6, %ymm2+ vpaddq %ymm3, %ymm7, %ymm3+ vpaddq %ymm4, %ymm8, %ymm4+ vpaddq %ymm5, %ymm9, %ymm5+ vmovdqa 0xc00(%rsi), %ymm6+ vmovdqa 0xc20(%rsi), %ymm8+ vmovdqa 0xc00(%rdx), %ymm10+ vmovdqa 0xc20(%rdx), %ymm12+ vpsrlq $0x20, %ymm6, %ymm7+ vpsrlq $0x20, %ymm8, %ymm9+ vmovshdup %ymm10, %ymm11 # ymm11 = ymm10[1,1,3,3,5,5,7,7]+ vmovshdup %ymm12, %ymm13 # ymm13 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm6, %ymm6+ vpmuldq %ymm11, %ymm7, %ymm7+ vpmuldq %ymm12, %ymm8, %ymm8+ vpmuldq %ymm13, %ymm9, %ymm9+ vpaddq %ymm2, %ymm6, %ymm2+ vpaddq %ymm3, %ymm7, %ymm3+ vpaddq %ymm4, %ymm8, %ymm4+ vpaddq %ymm5, %ymm9, %ymm5+ vmovdqa 0x1000(%rsi), %ymm6+ vmovdqa 0x1020(%rsi), %ymm8+ vmovdqa 0x1000(%rdx), %ymm10+ vmovdqa 0x1020(%rdx), %ymm12+ vpsrlq $0x20, %ymm6, %ymm7+ vpsrlq $0x20, %ymm8, %ymm9+ vmovshdup %ymm10, %ymm11 # ymm11 = ymm10[1,1,3,3,5,5,7,7]+ vmovshdup %ymm12, %ymm13 # ymm13 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm6, %ymm6+ vpmuldq %ymm11, %ymm7, %ymm7+ vpmuldq %ymm12, %ymm8, %ymm8+ vpmuldq %ymm13, %ymm9, %ymm9+ vpaddq %ymm2, %ymm6, %ymm2+ vpaddq %ymm3, %ymm7, %ymm3+ vpaddq %ymm4, %ymm8, %ymm4+ vpaddq %ymm5, %ymm9, %ymm5+ vpmuldq %ymm2, %ymm0, %ymm6+ vpmuldq %ymm3, %ymm0, %ymm7+ vpmuldq %ymm4, %ymm0, %ymm8+ vpmuldq %ymm5, %ymm0, %ymm9+ vpmuldq %ymm6, %ymm1, %ymm6+ vpmuldq %ymm7, %ymm1, %ymm7+ vpmuldq %ymm8, %ymm1, %ymm8+ vpmuldq %ymm9, %ymm1, %ymm9+ vpsubq %ymm6, %ymm2, %ymm2+ vpsubq %ymm7, %ymm3, %ymm3+ vpsubq %ymm8, %ymm4, %ymm4+ vpsubq %ymm9, %ymm5, %ymm5+ vpsrlq $0x20, %ymm2, %ymm2+ vmovshdup %ymm4, %ymm4 # ymm4 = ymm4[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm3, %ymm2, %ymm2 # ymm2 = ymm2[0],ymm3[1],ymm2[2],ymm3[3],ymm2[4],ymm3[5],ymm2[6],ymm3[7]+ vpblendd $0xaa, %ymm5, %ymm4, %ymm4 # ymm4 = ymm4[0],ymm5[1],ymm4[2],ymm5[3],ymm4[4],ymm5[5],ymm4[6],ymm5[7]+ vmovdqa %ymm2, (%rdi)+ vmovdqa %ymm4, 0x20(%rdi)+ addq $0x40, %rsi+ addq $0x40, %rdx+ addq $0x40, %rdi+ addl $0x1, %eax+ cmpl $0x10, %eax+ jb Lmld_pointwise_acc_l5_avx2_looptop2+ retq+ .cfi_endproc++MLD_ASM_FN_SIZE(pointwise_acc_l5_avx2_asm)++#endif /* MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_MULTILEVEL_NO_SHARED \+ && (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLDSA_L == 5) */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
@@ -0,0 +1,221 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [REF_AVX2]+ * CRYSTALS-Dilithium optimized AVX2 implementation+ * Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/dilithium/tree/master/avx2+ */++/*+ * This file is derived from the public domain+ * AVX2 Dilithium implementation @[REF_AVX2].+ */++/*yaml+ Name: pointwise_acc_l7_avx2_asm+ Description: x86_64 AVX2 pointwise multiply-accumulate of length-7 polynomial vectors+ Signature: void mld_pointwise_acc_l7_avx2_asm(int32_t *c, const int32_t a[7][256], const int32_t b[7][256], const int32_t *qdata)+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 1024+ permissions: write-only+ c_parameter: int32_t *c+ description: Output polynomial (256 x int32_t)+ rsi:+ type: buffer+ size_bytes: 7168+ permissions: read-only+ c_parameter: const int32_t a[7][256]+ description: Input polynomial vector a (7 x 256 x int32_t)+ rdx:+ type: buffer+ size_bytes: 7168+ permissions: read-only+ c_parameter: const int32_t b[7][256]+ description: Input polynomial vector b (7 x 256 x int32_t)+ rcx:+ type: buffer+ size_bytes: 2496+ permissions: read-only+ c_parameter: const int32_t *qdata+ description: Precomputed constants (624 x int32_t)+*/++#include "../../../common.h"+#if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \+ (defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLDSA_L == 7)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/x86_64/src/mldsa_pointwise_acc_l7_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(pointwise_acc_l7_avx2_asm)+MLD_ASM_FN_SYMBOL(pointwise_acc_l7_avx2_asm)++ .cfi_startproc+ vmovdqa 0x20(%rcx), %ymm0+ vmovdqa (%rcx), %ymm1+ xorl %eax, %eax++Lmld_pointwise_acc_l7_avx2_looptop2:+ vmovdqa (%rsi), %ymm6+ vmovdqa 0x20(%rsi), %ymm8+ vmovdqa (%rdx), %ymm10+ vmovdqa 0x20(%rdx), %ymm12+ vpsrlq $0x20, %ymm6, %ymm7+ vpsrlq $0x20, %ymm8, %ymm9+ vmovshdup %ymm10, %ymm11 # ymm11 = ymm10[1,1,3,3,5,5,7,7]+ vmovshdup %ymm12, %ymm13 # ymm13 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm6, %ymm6+ vpmuldq %ymm11, %ymm7, %ymm7+ vpmuldq %ymm12, %ymm8, %ymm8+ vpmuldq %ymm13, %ymm9, %ymm9+ vmovdqa %ymm6, %ymm2+ vmovdqa %ymm7, %ymm3+ vmovdqa %ymm8, %ymm4+ vmovdqa %ymm9, %ymm5+ vmovdqa 0x400(%rsi), %ymm6+ vmovdqa 0x420(%rsi), %ymm8+ vmovdqa 0x400(%rdx), %ymm10+ vmovdqa 0x420(%rdx), %ymm12+ vpsrlq $0x20, %ymm6, %ymm7+ vpsrlq $0x20, %ymm8, %ymm9+ vmovshdup %ymm10, %ymm11 # ymm11 = ymm10[1,1,3,3,5,5,7,7]+ vmovshdup %ymm12, %ymm13 # ymm13 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm6, %ymm6+ vpmuldq %ymm11, %ymm7, %ymm7+ vpmuldq %ymm12, %ymm8, %ymm8+ vpmuldq %ymm13, %ymm9, %ymm9+ vpaddq %ymm2, %ymm6, %ymm2+ vpaddq %ymm3, %ymm7, %ymm3+ vpaddq %ymm4, %ymm8, %ymm4+ vpaddq %ymm5, %ymm9, %ymm5+ vmovdqa 0x800(%rsi), %ymm6+ vmovdqa 0x820(%rsi), %ymm8+ vmovdqa 0x800(%rdx), %ymm10+ vmovdqa 0x820(%rdx), %ymm12+ vpsrlq $0x20, %ymm6, %ymm7+ vpsrlq $0x20, %ymm8, %ymm9+ vmovshdup %ymm10, %ymm11 # ymm11 = ymm10[1,1,3,3,5,5,7,7]+ vmovshdup %ymm12, %ymm13 # ymm13 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm6, %ymm6+ vpmuldq %ymm11, %ymm7, %ymm7+ vpmuldq %ymm12, %ymm8, %ymm8+ vpmuldq %ymm13, %ymm9, %ymm9+ vpaddq %ymm2, %ymm6, %ymm2+ vpaddq %ymm3, %ymm7, %ymm3+ vpaddq %ymm4, %ymm8, %ymm4+ vpaddq %ymm5, %ymm9, %ymm5+ vmovdqa 0xc00(%rsi), %ymm6+ vmovdqa 0xc20(%rsi), %ymm8+ vmovdqa 0xc00(%rdx), %ymm10+ vmovdqa 0xc20(%rdx), %ymm12+ vpsrlq $0x20, %ymm6, %ymm7+ vpsrlq $0x20, %ymm8, %ymm9+ vmovshdup %ymm10, %ymm11 # ymm11 = ymm10[1,1,3,3,5,5,7,7]+ vmovshdup %ymm12, %ymm13 # ymm13 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm6, %ymm6+ vpmuldq %ymm11, %ymm7, %ymm7+ vpmuldq %ymm12, %ymm8, %ymm8+ vpmuldq %ymm13, %ymm9, %ymm9+ vpaddq %ymm2, %ymm6, %ymm2+ vpaddq %ymm3, %ymm7, %ymm3+ vpaddq %ymm4, %ymm8, %ymm4+ vpaddq %ymm5, %ymm9, %ymm5+ vmovdqa 0x1000(%rsi), %ymm6+ vmovdqa 0x1020(%rsi), %ymm8+ vmovdqa 0x1000(%rdx), %ymm10+ vmovdqa 0x1020(%rdx), %ymm12+ vpsrlq $0x20, %ymm6, %ymm7+ vpsrlq $0x20, %ymm8, %ymm9+ vmovshdup %ymm10, %ymm11 # ymm11 = ymm10[1,1,3,3,5,5,7,7]+ vmovshdup %ymm12, %ymm13 # ymm13 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm6, %ymm6+ vpmuldq %ymm11, %ymm7, %ymm7+ vpmuldq %ymm12, %ymm8, %ymm8+ vpmuldq %ymm13, %ymm9, %ymm9+ vpaddq %ymm2, %ymm6, %ymm2+ vpaddq %ymm3, %ymm7, %ymm3+ vpaddq %ymm4, %ymm8, %ymm4+ vpaddq %ymm5, %ymm9, %ymm5+ vmovdqa 0x1400(%rsi), %ymm6+ vmovdqa 0x1420(%rsi), %ymm8+ vmovdqa 0x1400(%rdx), %ymm10+ vmovdqa 0x1420(%rdx), %ymm12+ vpsrlq $0x20, %ymm6, %ymm7+ vpsrlq $0x20, %ymm8, %ymm9+ vmovshdup %ymm10, %ymm11 # ymm11 = ymm10[1,1,3,3,5,5,7,7]+ vmovshdup %ymm12, %ymm13 # ymm13 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm6, %ymm6+ vpmuldq %ymm11, %ymm7, %ymm7+ vpmuldq %ymm12, %ymm8, %ymm8+ vpmuldq %ymm13, %ymm9, %ymm9+ vpaddq %ymm2, %ymm6, %ymm2+ vpaddq %ymm3, %ymm7, %ymm3+ vpaddq %ymm4, %ymm8, %ymm4+ vpaddq %ymm5, %ymm9, %ymm5+ vmovdqa 0x1800(%rsi), %ymm6+ vmovdqa 0x1820(%rsi), %ymm8+ vmovdqa 0x1800(%rdx), %ymm10+ vmovdqa 0x1820(%rdx), %ymm12+ vpsrlq $0x20, %ymm6, %ymm7+ vpsrlq $0x20, %ymm8, %ymm9+ vmovshdup %ymm10, %ymm11 # ymm11 = ymm10[1,1,3,3,5,5,7,7]+ vmovshdup %ymm12, %ymm13 # ymm13 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm6, %ymm6+ vpmuldq %ymm11, %ymm7, %ymm7+ vpmuldq %ymm12, %ymm8, %ymm8+ vpmuldq %ymm13, %ymm9, %ymm9+ vpaddq %ymm2, %ymm6, %ymm2+ vpaddq %ymm3, %ymm7, %ymm3+ vpaddq %ymm4, %ymm8, %ymm4+ vpaddq %ymm5, %ymm9, %ymm5+ vpmuldq %ymm2, %ymm0, %ymm6+ vpmuldq %ymm3, %ymm0, %ymm7+ vpmuldq %ymm4, %ymm0, %ymm8+ vpmuldq %ymm5, %ymm0, %ymm9+ vpmuldq %ymm6, %ymm1, %ymm6+ vpmuldq %ymm7, %ymm1, %ymm7+ vpmuldq %ymm8, %ymm1, %ymm8+ vpmuldq %ymm9, %ymm1, %ymm9+ vpsubq %ymm6, %ymm2, %ymm2+ vpsubq %ymm7, %ymm3, %ymm3+ vpsubq %ymm8, %ymm4, %ymm4+ vpsubq %ymm9, %ymm5, %ymm5+ vpsrlq $0x20, %ymm2, %ymm2+ vmovshdup %ymm4, %ymm4 # ymm4 = ymm4[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm3, %ymm2, %ymm2 # ymm2 = ymm2[0],ymm3[1],ymm2[2],ymm3[3],ymm2[4],ymm3[5],ymm2[6],ymm3[7]+ vpblendd $0xaa, %ymm5, %ymm4, %ymm4 # ymm4 = ymm4[0],ymm5[1],ymm4[2],ymm5[3],ymm4[4],ymm5[5],ymm4[6],ymm5[7]+ vmovdqa %ymm2, (%rdi)+ vmovdqa %ymm4, 0x20(%rdi)+ addq $0x40, %rsi+ addq $0x40, %rdx+ addq $0x40, %rdi+ addl $0x1, %eax+ cmpl $0x10, %eax+ jb Lmld_pointwise_acc_l7_avx2_looptop2+ retq+ .cfi_endproc++MLD_ASM_FN_SIZE(pointwise_acc_l7_avx2_asm)++#endif /* MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_MULTILEVEL_NO_SHARED \+ && (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLDSA_L == 7) */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
@@ -0,0 +1,158 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [REF_AVX2]+ * CRYSTALS-Dilithium optimized AVX2 implementation+ * Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/dilithium/tree/master/avx2+ */++/*+ * This file is derived from the public domain+ * AVX2 Dilithium implementation @[REF_AVX2].+ */++/*yaml+ Name: pointwise_avx2_asm+ Description: x86_64 AVX2 pointwise Montgomery multiplication+ Signature: void mld_pointwise_avx2_asm(int32_t *a, const int32_t *b, const int32_t *qdata)+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 1024+ permissions: read/write+ c_parameter: int32_t *a+ description: Input/output polynomial (256 x int32_t)+ rsi:+ type: buffer+ size_bytes: 1024+ permissions: read-only+ c_parameter: const int32_t *b+ description: Input polynomial (256 x int32_t)+ rdx:+ type: buffer+ size_bytes: 2496+ permissions: read-only+ c_parameter: const int32_t *qdata+ description: Precomputed constants (624 x int32_t)+*/++#include "../../../common.h"+#if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/x86_64/src/mldsa_pointwise_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(pointwise_avx2_asm)+MLD_ASM_FN_SYMBOL(pointwise_avx2_asm)++ .cfi_startproc+ vmovdqa 0x20(%rdx), %ymm0+ vmovdqa (%rdx), %ymm1+ xorl %eax, %eax++Lmld_pointwise_avx2_looptop1:+ vmovdqa (%rdi), %ymm2+ vmovdqa 0x20(%rdi), %ymm4+ vmovdqa 0x40(%rdi), %ymm6+ vmovdqa (%rsi), %ymm10+ vmovdqa 0x20(%rsi), %ymm12+ vmovdqa 0x40(%rsi), %ymm14+ vpsrlq $0x20, %ymm2, %ymm3+ vpsrlq $0x20, %ymm4, %ymm5+ vmovshdup %ymm6, %ymm7 # ymm7 = ymm6[1,1,3,3,5,5,7,7]+ vpsrlq $0x20, %ymm10, %ymm11+ vpsrlq $0x20, %ymm12, %ymm13+ vmovshdup %ymm14, %ymm15 # ymm15 = ymm14[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm2, %ymm2+ vpmuldq %ymm11, %ymm3, %ymm3+ vpmuldq %ymm12, %ymm4, %ymm4+ vpmuldq %ymm13, %ymm5, %ymm5+ vpmuldq %ymm14, %ymm6, %ymm6+ vpmuldq %ymm15, %ymm7, %ymm7+ vpmuldq %ymm2, %ymm0, %ymm10+ vpmuldq %ymm3, %ymm0, %ymm11+ vpmuldq %ymm4, %ymm0, %ymm12+ vpmuldq %ymm5, %ymm0, %ymm13+ vpmuldq %ymm6, %ymm0, %ymm14+ vpmuldq %ymm7, %ymm0, %ymm15+ vpmuldq %ymm10, %ymm1, %ymm10+ vpmuldq %ymm11, %ymm1, %ymm11+ vpmuldq %ymm12, %ymm1, %ymm12+ vpmuldq %ymm13, %ymm1, %ymm13+ vpmuldq %ymm14, %ymm1, %ymm14+ vpmuldq %ymm15, %ymm1, %ymm15+ vpsubq %ymm10, %ymm2, %ymm2+ vpsubq %ymm11, %ymm3, %ymm3+ vpsubq %ymm12, %ymm4, %ymm4+ vpsubq %ymm13, %ymm5, %ymm5+ vpsubq %ymm14, %ymm6, %ymm6+ vpsubq %ymm15, %ymm7, %ymm7+ vpsrlq $0x20, %ymm2, %ymm2+ vpsrlq $0x20, %ymm4, %ymm4+ vmovshdup %ymm6, %ymm6 # ymm6 = ymm6[1,1,3,3,5,5,7,7]+ vpblendd $0xaa, %ymm3, %ymm2, %ymm2 # ymm2 = ymm2[0],ymm3[1],ymm2[2],ymm3[3],ymm2[4],ymm3[5],ymm2[6],ymm3[7]+ vpblendd $0xaa, %ymm5, %ymm4, %ymm4 # ymm4 = ymm4[0],ymm5[1],ymm4[2],ymm5[3],ymm4[4],ymm5[5],ymm4[6],ymm5[7]+ vpblendd $0xaa, %ymm7, %ymm6, %ymm6 # ymm6 = ymm6[0],ymm7[1],ymm6[2],ymm7[3],ymm6[4],ymm7[5],ymm6[6],ymm7[7]+ vmovdqa %ymm2, (%rdi)+ vmovdqa %ymm4, 0x20(%rdi)+ vmovdqa %ymm6, 0x40(%rdi)+ addq $0x60, %rdi+ addq $0x60, %rsi+ addl $0x1, %eax+ cmpl $0xa, %eax+ jb Lmld_pointwise_avx2_looptop1+ vmovdqa (%rdi), %ymm2+ vmovdqa 0x20(%rdi), %ymm4+ vmovdqa (%rsi), %ymm10+ vmovdqa 0x20(%rsi), %ymm12+ vpsrlq $0x20, %ymm2, %ymm3+ vpsrlq $0x20, %ymm4, %ymm5+ vmovshdup %ymm10, %ymm11 # ymm11 = ymm10[1,1,3,3,5,5,7,7]+ vmovshdup %ymm12, %ymm13 # ymm13 = ymm12[1,1,3,3,5,5,7,7]+ vpmuldq %ymm10, %ymm2, %ymm2+ vpmuldq %ymm11, %ymm3, %ymm3+ vpmuldq %ymm12, %ymm4, %ymm4+ vpmuldq %ymm13, %ymm5, %ymm5+ vpmuldq %ymm2, %ymm0, %ymm10+ vpmuldq %ymm3, %ymm0, %ymm11+ vpmuldq %ymm4, %ymm0, %ymm12+ vpmuldq %ymm5, %ymm0, %ymm13+ vpmuldq %ymm10, %ymm1, %ymm10+ vpmuldq %ymm11, %ymm1, %ymm11+ vpmuldq %ymm12, %ymm1, %ymm12+ vpmuldq %ymm13, %ymm1, %ymm13+ vpsubq %ymm10, %ymm2, %ymm2+ vpsubq %ymm11, %ymm3, %ymm3+ vpsubq %ymm12, %ymm4, %ymm4+ vpsubq %ymm13, %ymm5, %ymm5+ vpsrlq $0x20, %ymm2, %ymm2+ vmovshdup %ymm4, %ymm4 # ymm4 = ymm4[1,1,3,3,5,5,7,7]+ vpblendd $0x55, %ymm2, %ymm3, %ymm2 # ymm2 = ymm2[0],ymm3[1],ymm2[2],ymm3[3],ymm2[4],ymm3[5],ymm2[6],ymm3[7]+ vpblendd $0x55, %ymm4, %ymm5, %ymm4 # ymm4 = ymm4[0],ymm5[1],ymm4[2],ymm5[3],ymm4[4],ymm5[5],ymm4[6],ymm5[7]+ vmovdqa %ymm2, (%rdi)+ vmovdqa %ymm4, 0x20(%rdi)+ retq+ .cfi_endproc++MLD_ASM_FN_SIZE(pointwise_avx2_asm)++#endif /* MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_MULTILEVEL_NO_SHARED \+ */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
@@ -0,0 +1,199 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [REF_AVX2]+ * CRYSTALS-Dilithium optimized AVX2 implementation+ * Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/dilithium/tree/master/avx2+ */++/*+ * This file is derived from the public domain+ * AVX2 Dilithium implementation @[REF_AVX2].+ */+++/*yaml+ Name: poly_caddq_avx2_asm+ Description: x86_64 AVX2 conditional addition of q to each coefficient.+ For all coefficients of the in/out polynomial, add Q if the coefficient+ is negative.+ Signature: void mld_poly_caddq_avx2_asm(int32_t *r)+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 1024+ permissions: read/write+ c_parameter: int32_t *r+ description: Input/output polynomial (256 x int32_t)+*/++#include "../../../common.h"++#if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)+++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/x86_64/src/mldsa_poly_caddq_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(poly_caddq_avx2_asm)+MLD_ASM_FN_SYMBOL(poly_caddq_avx2_asm)++ .cfi_startproc+ vpxor %xmm2, %xmm2, %xmm2+ movl $0x7fe001, %eax # imm = 0x7FE001+ vmovd %eax, %xmm1+ vpbroadcastd %xmm1, %ymm1+ vpcmpgtd (%rdi), %ymm2, %ymm0+ vpand %ymm1, %ymm0, %ymm0+ vpaddd (%rdi), %ymm0, %ymm0+ vmovdqa %ymm0, (%rdi)+ vpcmpgtd 0x20(%rdi), %ymm2, %ymm3+ vpand %ymm1, %ymm3, %ymm3+ vpaddd 0x20(%rdi), %ymm3, %ymm3+ vmovdqa %ymm3, 0x20(%rdi)+ vpcmpgtd 0x40(%rdi), %ymm2, %ymm4+ vpand %ymm1, %ymm4, %ymm4+ vpaddd 0x40(%rdi), %ymm4, %ymm4+ vmovdqa %ymm4, 0x40(%rdi)+ vpcmpgtd 0x60(%rdi), %ymm2, %ymm5+ vpand %ymm1, %ymm5, %ymm5+ vpaddd 0x60(%rdi), %ymm5, %ymm5+ vmovdqa %ymm5, 0x60(%rdi)+ vpcmpgtd 0x80(%rdi), %ymm2, %ymm0+ vpand %ymm1, %ymm0, %ymm0+ vpaddd 0x80(%rdi), %ymm0, %ymm0+ vmovdqa %ymm0, 0x80(%rdi)+ vpcmpgtd 0xa0(%rdi), %ymm2, %ymm3+ vpand %ymm1, %ymm3, %ymm3+ vpaddd 0xa0(%rdi), %ymm3, %ymm3+ vmovdqa %ymm3, 0xa0(%rdi)+ vpcmpgtd 0xc0(%rdi), %ymm2, %ymm4+ vpand %ymm1, %ymm4, %ymm4+ vpaddd 0xc0(%rdi), %ymm4, %ymm4+ vmovdqa %ymm4, 0xc0(%rdi)+ vpcmpgtd 0xe0(%rdi), %ymm2, %ymm5+ vpand %ymm1, %ymm5, %ymm5+ vpaddd 0xe0(%rdi), %ymm5, %ymm5+ vmovdqa %ymm5, 0xe0(%rdi)+ vpcmpgtd 0x100(%rdi), %ymm2, %ymm0+ vpand %ymm1, %ymm0, %ymm0+ vpaddd 0x100(%rdi), %ymm0, %ymm0+ vmovdqa %ymm0, 0x100(%rdi)+ vpcmpgtd 0x120(%rdi), %ymm2, %ymm3+ vpand %ymm1, %ymm3, %ymm3+ vpaddd 0x120(%rdi), %ymm3, %ymm3+ vmovdqa %ymm3, 0x120(%rdi)+ vpcmpgtd 0x140(%rdi), %ymm2, %ymm4+ vpand %ymm1, %ymm4, %ymm4+ vpaddd 0x140(%rdi), %ymm4, %ymm4+ vmovdqa %ymm4, 0x140(%rdi)+ vpcmpgtd 0x160(%rdi), %ymm2, %ymm5+ vpand %ymm1, %ymm5, %ymm5+ vpaddd 0x160(%rdi), %ymm5, %ymm5+ vmovdqa %ymm5, 0x160(%rdi)+ vpcmpgtd 0x180(%rdi), %ymm2, %ymm0+ vpand %ymm1, %ymm0, %ymm0+ vpaddd 0x180(%rdi), %ymm0, %ymm0+ vmovdqa %ymm0, 0x180(%rdi)+ vpcmpgtd 0x1a0(%rdi), %ymm2, %ymm3+ vpand %ymm1, %ymm3, %ymm3+ vpaddd 0x1a0(%rdi), %ymm3, %ymm3+ vmovdqa %ymm3, 0x1a0(%rdi)+ vpcmpgtd 0x1c0(%rdi), %ymm2, %ymm4+ vpand %ymm1, %ymm4, %ymm4+ vpaddd 0x1c0(%rdi), %ymm4, %ymm4+ vmovdqa %ymm4, 0x1c0(%rdi)+ vpcmpgtd 0x1e0(%rdi), %ymm2, %ymm5+ vpand %ymm1, %ymm5, %ymm5+ vpaddd 0x1e0(%rdi), %ymm5, %ymm5+ vmovdqa %ymm5, 0x1e0(%rdi)+ vpcmpgtd 0x200(%rdi), %ymm2, %ymm0+ vpand %ymm1, %ymm0, %ymm0+ vpaddd 0x200(%rdi), %ymm0, %ymm0+ vmovdqa %ymm0, 0x200(%rdi)+ vpcmpgtd 0x220(%rdi), %ymm2, %ymm3+ vpand %ymm1, %ymm3, %ymm3+ vpaddd 0x220(%rdi), %ymm3, %ymm3+ vmovdqa %ymm3, 0x220(%rdi)+ vpcmpgtd 0x240(%rdi), %ymm2, %ymm4+ vpand %ymm1, %ymm4, %ymm4+ vpaddd 0x240(%rdi), %ymm4, %ymm4+ vmovdqa %ymm4, 0x240(%rdi)+ vpcmpgtd 0x260(%rdi), %ymm2, %ymm5+ vpand %ymm1, %ymm5, %ymm5+ vpaddd 0x260(%rdi), %ymm5, %ymm5+ vmovdqa %ymm5, 0x260(%rdi)+ vpcmpgtd 0x280(%rdi), %ymm2, %ymm0+ vpand %ymm1, %ymm0, %ymm0+ vpaddd 0x280(%rdi), %ymm0, %ymm0+ vmovdqa %ymm0, 0x280(%rdi)+ vpcmpgtd 0x2a0(%rdi), %ymm2, %ymm3+ vpand %ymm1, %ymm3, %ymm3+ vpaddd 0x2a0(%rdi), %ymm3, %ymm3+ vmovdqa %ymm3, 0x2a0(%rdi)+ vpcmpgtd 0x2c0(%rdi), %ymm2, %ymm4+ vpand %ymm1, %ymm4, %ymm4+ vpaddd 0x2c0(%rdi), %ymm4, %ymm4+ vmovdqa %ymm4, 0x2c0(%rdi)+ vpcmpgtd 0x2e0(%rdi), %ymm2, %ymm5+ vpand %ymm1, %ymm5, %ymm5+ vpaddd 0x2e0(%rdi), %ymm5, %ymm5+ vmovdqa %ymm5, 0x2e0(%rdi)+ vpcmpgtd 0x300(%rdi), %ymm2, %ymm0+ vpand %ymm1, %ymm0, %ymm0+ vpaddd 0x300(%rdi), %ymm0, %ymm0+ vmovdqa %ymm0, 0x300(%rdi)+ vpcmpgtd 0x320(%rdi), %ymm2, %ymm3+ vpand %ymm1, %ymm3, %ymm3+ vpaddd 0x320(%rdi), %ymm3, %ymm3+ vmovdqa %ymm3, 0x320(%rdi)+ vpcmpgtd 0x340(%rdi), %ymm2, %ymm4+ vpand %ymm1, %ymm4, %ymm4+ vpaddd 0x340(%rdi), %ymm4, %ymm4+ vmovdqa %ymm4, 0x340(%rdi)+ vpcmpgtd 0x360(%rdi), %ymm2, %ymm5+ vpand %ymm1, %ymm5, %ymm5+ vpaddd 0x360(%rdi), %ymm5, %ymm5+ vmovdqa %ymm5, 0x360(%rdi)+ vpcmpgtd 0x380(%rdi), %ymm2, %ymm0+ vpand %ymm1, %ymm0, %ymm0+ vpaddd 0x380(%rdi), %ymm0, %ymm0+ vmovdqa %ymm0, 0x380(%rdi)+ vpcmpgtd 0x3a0(%rdi), %ymm2, %ymm3+ vpand %ymm1, %ymm3, %ymm3+ vpaddd 0x3a0(%rdi), %ymm3, %ymm3+ vmovdqa %ymm3, 0x3a0(%rdi)+ vpcmpgtd 0x3c0(%rdi), %ymm2, %ymm4+ vpand %ymm1, %ymm4, %ymm4+ vpaddd 0x3c0(%rdi), %ymm4, %ymm4+ vmovdqa %ymm4, 0x3c0(%rdi)+ vpcmpgtd 0x3e0(%rdi), %ymm2, %ymm5+ vpand %ymm1, %ymm5, %ymm5+ vpaddd 0x3e0(%rdi), %ymm5, %ymm5+ vmovdqa %ymm5, 0x3e0(%rdi)+ retq+ .cfi_endproc++MLD_ASM_FN_SIZE(poly_caddq_avx2_asm)+++#endif /* MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_MULTILEVEL_NO_SHARED \+ */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
@@ -0,0 +1,176 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [REF_AVX2]+ * CRYSTALS-Dilithium optimized AVX2 implementation+ * Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/dilithium/tree/master/avx2+ */++/*+ * This file is derived from the public domain+ * AVX2 Dilithium implementation @[REF_AVX2].+ */+++/*yaml+ Name: poly_chknorm_avx2_asm+ Description: x86_64 AVX2 infinity-norm bound check on polynomial coefficients.+ Check the infinity norm of the polynomial against the given bound B.+ Returns 0 if the norm is strictly smaller than B; otherwise returns 1+ (i.e. returns 1 if any |coefficient| >= B, 0 otherwise).+ Signature: int mld_poly_chknorm_avx2_asm(const int32_t *a, int32_t B)+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 1024+ permissions: read-only+ c_parameter: const int32_t *a+ description: Input polynomial (256 x int32_t)+ rsi:+ type: scalar+ c_parameter: int32_t B+ description: Norm bound (must be non-negative)+ test_with: 131072 # representative non-negative bound (1 << 17)+*/++#include "../../../common.h"++#if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)+++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/x86_64/src/mldsa_poly_chknorm_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(poly_chknorm_avx2_asm)+MLD_ASM_FN_SYMBOL(poly_chknorm_avx2_asm)++ .cfi_startproc+ subl $0x1, %esi+ vpxor %xmm1, %xmm1, %xmm1+ vmovd %esi, %xmm2+ vpbroadcastd %xmm2, %ymm2+ vpabsd (%rdi), %ymm0+ vpcmpgtd %ymm2, %ymm0, %ymm0+ vpor %ymm0, %ymm1, %ymm1+ vpabsd 0x20(%rdi), %ymm3+ vpcmpgtd %ymm2, %ymm3, %ymm3+ vpor %ymm3, %ymm1, %ymm1+ vpabsd 0x40(%rdi), %ymm4+ vpcmpgtd %ymm2, %ymm4, %ymm4+ vpor %ymm4, %ymm1, %ymm1+ vpabsd 0x60(%rdi), %ymm5+ vpcmpgtd %ymm2, %ymm5, %ymm5+ vpor %ymm5, %ymm1, %ymm1+ vpabsd 0x80(%rdi), %ymm0+ vpcmpgtd %ymm2, %ymm0, %ymm0+ vpor %ymm0, %ymm1, %ymm1+ vpabsd 0xa0(%rdi), %ymm3+ vpcmpgtd %ymm2, %ymm3, %ymm3+ vpor %ymm3, %ymm1, %ymm1+ vpabsd 0xc0(%rdi), %ymm4+ vpcmpgtd %ymm2, %ymm4, %ymm4+ vpor %ymm4, %ymm1, %ymm1+ vpabsd 0xe0(%rdi), %ymm5+ vpcmpgtd %ymm2, %ymm5, %ymm5+ vpor %ymm5, %ymm1, %ymm1+ vpabsd 0x100(%rdi), %ymm0+ vpcmpgtd %ymm2, %ymm0, %ymm0+ vpor %ymm0, %ymm1, %ymm1+ vpabsd 0x120(%rdi), %ymm3+ vpcmpgtd %ymm2, %ymm3, %ymm3+ vpor %ymm3, %ymm1, %ymm1+ vpabsd 0x140(%rdi), %ymm4+ vpcmpgtd %ymm2, %ymm4, %ymm4+ vpor %ymm4, %ymm1, %ymm1+ vpabsd 0x160(%rdi), %ymm5+ vpcmpgtd %ymm2, %ymm5, %ymm5+ vpor %ymm5, %ymm1, %ymm1+ vpabsd 0x180(%rdi), %ymm0+ vpcmpgtd %ymm2, %ymm0, %ymm0+ vpor %ymm0, %ymm1, %ymm1+ vpabsd 0x1a0(%rdi), %ymm3+ vpcmpgtd %ymm2, %ymm3, %ymm3+ vpor %ymm3, %ymm1, %ymm1+ vpabsd 0x1c0(%rdi), %ymm4+ vpcmpgtd %ymm2, %ymm4, %ymm4+ vpor %ymm4, %ymm1, %ymm1+ vpabsd 0x1e0(%rdi), %ymm5+ vpcmpgtd %ymm2, %ymm5, %ymm5+ vpor %ymm5, %ymm1, %ymm1+ vpabsd 0x200(%rdi), %ymm0+ vpcmpgtd %ymm2, %ymm0, %ymm0+ vpor %ymm0, %ymm1, %ymm1+ vpabsd 0x220(%rdi), %ymm3+ vpcmpgtd %ymm2, %ymm3, %ymm3+ vpor %ymm3, %ymm1, %ymm1+ vpabsd 0x240(%rdi), %ymm4+ vpcmpgtd %ymm2, %ymm4, %ymm4+ vpor %ymm4, %ymm1, %ymm1+ vpabsd 0x260(%rdi), %ymm5+ vpcmpgtd %ymm2, %ymm5, %ymm5+ vpor %ymm5, %ymm1, %ymm1+ vpabsd 0x280(%rdi), %ymm0+ vpcmpgtd %ymm2, %ymm0, %ymm0+ vpor %ymm0, %ymm1, %ymm1+ vpabsd 0x2a0(%rdi), %ymm3+ vpcmpgtd %ymm2, %ymm3, %ymm3+ vpor %ymm3, %ymm1, %ymm1+ vpabsd 0x2c0(%rdi), %ymm4+ vpcmpgtd %ymm2, %ymm4, %ymm4+ vpor %ymm4, %ymm1, %ymm1+ vpabsd 0x2e0(%rdi), %ymm5+ vpcmpgtd %ymm2, %ymm5, %ymm5+ vpor %ymm5, %ymm1, %ymm1+ vpabsd 0x300(%rdi), %ymm0+ vpcmpgtd %ymm2, %ymm0, %ymm0+ vpor %ymm0, %ymm1, %ymm1+ vpabsd 0x320(%rdi), %ymm3+ vpcmpgtd %ymm2, %ymm3, %ymm3+ vpor %ymm3, %ymm1, %ymm1+ vpabsd 0x340(%rdi), %ymm4+ vpcmpgtd %ymm2, %ymm4, %ymm4+ vpor %ymm4, %ymm1, %ymm1+ vpabsd 0x360(%rdi), %ymm5+ vpcmpgtd %ymm2, %ymm5, %ymm5+ vpor %ymm5, %ymm1, %ymm1+ vpabsd 0x380(%rdi), %ymm0+ vpcmpgtd %ymm2, %ymm0, %ymm0+ vpor %ymm0, %ymm1, %ymm1+ vpabsd 0x3a0(%rdi), %ymm3+ vpcmpgtd %ymm2, %ymm3, %ymm3+ vpor %ymm3, %ymm1, %ymm1+ vpabsd 0x3c0(%rdi), %ymm4+ vpcmpgtd %ymm2, %ymm4, %ymm4+ vpor %ymm4, %ymm1, %ymm1+ vpabsd 0x3e0(%rdi), %ymm5+ vpcmpgtd %ymm2, %ymm5, %ymm5+ vpor %ymm5, %ymm1, %ymm1+ xorl %eax, %eax+ vptest %ymm1, %ymm1+ setne %al+ retq+ .cfi_endproc++MLD_ASM_FN_SIZE(poly_chknorm_avx2_asm)+++#endif /* MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_MULTILEVEL_NO_SHARED \+ */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
@@ -0,0 +1,490 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [REF_AVX2]+ * CRYSTALS-Dilithium optimized AVX2 implementation+ * Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/dilithium/tree/master/avx2+ */++/*+ * This file is derived from the public domain+ * AVX2 Dilithium implementation @[REF_AVX2].+ *+ * The algorithm for Decompose(r) (more specifically the handling for the+ * wrap-around cases) is modified. See the AVX2 intrinsics version+ * (poly_decompose_32_avx2.c, predecessor of this file) for a more detailed+ * comparison.+ */+++/*yaml+ Name: poly_decompose_32_avx2_asm+ Description: x86_64 AVX2 coefficient decomposition (alpha = 2*(Q-1)/32).+ For all coefficients c of the input polynomial, compute high and low bits+ c0, c1 such that c = c1*ALPHA + c0 with -ALPHA/2 < c0 <= ALPHA/2, except if+ c1 = (Q-1)/ALPHA where we set c1 = 0 and -ALPHA/2 <= c0 = c - Q < 0.+ Assumes coefficients to be standard (unsigned canonical) representatives,+ i.e. 0 <= c < Q. For ML-DSA-65 / ML-DSA-87 (gamma2 = (Q-1)/32,+ alpha = 2*gamma2).+ Signature: void mld_poly_decompose_32_avx2_asm(int32_t *a1, int32_t *a0)+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 1024+ permissions: write-only+ c_parameter: int32_t *a1+ description: Output high-part polynomial (256 x int32_t)+ rsi:+ type: buffer+ size_bytes: 1024+ permissions: read/write+ c_parameter: int32_t *a0+ description: Input polynomial / output low-part (256 x int32_t)+*/++#include "../../../common.h"++#if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \+ !defined(MLD_CONFIG_NO_SIGN_API) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \+ (defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \+ (MLD_CONFIG_PARAMETER_SET == 65 || MLD_CONFIG_PARAMETER_SET == 87))+++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/x86_64/src/mldsa_poly_decompose_32_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(poly_decompose_32_avx2_asm)+MLD_ASM_FN_SYMBOL(poly_decompose_32_avx2_asm)++ .cfi_startproc+ movl $0x7f, %eax+ vmovd %eax, %xmm10+ vpbroadcastd %xmm10, %ymm10+ movl $0x401, %eax # imm = 0x401+ vmovd %eax, %xmm11+ vpbroadcastd %xmm11, %ymm11+ movl $0x200, %eax # imm = 0x200+ vmovd %eax, %xmm12+ vpbroadcastd %xmm12, %ymm12+ movl $0x7be100, %eax # imm = 0x7BE100+ vmovd %eax, %xmm13+ vpbroadcastd %xmm13, %ymm13+ movl $0x7fe00, %eax # imm = 0x7FE00+ vmovd %eax, %xmm14+ vpbroadcastd %xmm14, %ymm14+ vmovdqa (%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, (%rdi)+ vmovdqa %ymm2, (%rsi)+ vmovdqa 0x20(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x20(%rdi)+ vmovdqa %ymm2, 0x20(%rsi)+ vmovdqa 0x40(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x40(%rdi)+ vmovdqa %ymm2, 0x40(%rsi)+ vmovdqa 0x60(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x60(%rdi)+ vmovdqa %ymm2, 0x60(%rsi)+ vmovdqa 0x80(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x80(%rdi)+ vmovdqa %ymm2, 0x80(%rsi)+ vmovdqa 0xa0(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0xa0(%rdi)+ vmovdqa %ymm2, 0xa0(%rsi)+ vmovdqa 0xc0(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0xc0(%rdi)+ vmovdqa %ymm2, 0xc0(%rsi)+ vmovdqa 0xe0(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0xe0(%rdi)+ vmovdqa %ymm2, 0xe0(%rsi)+ vmovdqa 0x100(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x100(%rdi)+ vmovdqa %ymm2, 0x100(%rsi)+ vmovdqa 0x120(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x120(%rdi)+ vmovdqa %ymm2, 0x120(%rsi)+ vmovdqa 0x140(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x140(%rdi)+ vmovdqa %ymm2, 0x140(%rsi)+ vmovdqa 0x160(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x160(%rdi)+ vmovdqa %ymm2, 0x160(%rsi)+ vmovdqa 0x180(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x180(%rdi)+ vmovdqa %ymm2, 0x180(%rsi)+ vmovdqa 0x1a0(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x1a0(%rdi)+ vmovdqa %ymm2, 0x1a0(%rsi)+ vmovdqa 0x1c0(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x1c0(%rdi)+ vmovdqa %ymm2, 0x1c0(%rsi)+ vmovdqa 0x1e0(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x1e0(%rdi)+ vmovdqa %ymm2, 0x1e0(%rsi)+ vmovdqa 0x200(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x200(%rdi)+ vmovdqa %ymm2, 0x200(%rsi)+ vmovdqa 0x220(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x220(%rdi)+ vmovdqa %ymm2, 0x220(%rsi)+ vmovdqa 0x240(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x240(%rdi)+ vmovdqa %ymm2, 0x240(%rsi)+ vmovdqa 0x260(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x260(%rdi)+ vmovdqa %ymm2, 0x260(%rsi)+ vmovdqa 0x280(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x280(%rdi)+ vmovdqa %ymm2, 0x280(%rsi)+ vmovdqa 0x2a0(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x2a0(%rdi)+ vmovdqa %ymm2, 0x2a0(%rsi)+ vmovdqa 0x2c0(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x2c0(%rdi)+ vmovdqa %ymm2, 0x2c0(%rsi)+ vmovdqa 0x2e0(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x2e0(%rdi)+ vmovdqa %ymm2, 0x2e0(%rsi)+ vmovdqa 0x300(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x300(%rdi)+ vmovdqa %ymm2, 0x300(%rsi)+ vmovdqa 0x320(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x320(%rdi)+ vmovdqa %ymm2, 0x320(%rsi)+ vmovdqa 0x340(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x340(%rdi)+ vmovdqa %ymm2, 0x340(%rsi)+ vmovdqa 0x360(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x360(%rdi)+ vmovdqa %ymm2, 0x360(%rsi)+ vmovdqa 0x380(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x380(%rdi)+ vmovdqa %ymm2, 0x380(%rsi)+ vmovdqa 0x3a0(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x3a0(%rdi)+ vmovdqa %ymm2, 0x3a0(%rsi)+ vmovdqa 0x3c0(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x3c0(%rdi)+ vmovdqa %ymm2, 0x3c0(%rsi)+ vmovdqa 0x3e0(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x3e0(%rdi)+ vmovdqa %ymm2, 0x3e0(%rsi)+ retq+ .cfi_endproc++MLD_ASM_FN_SIZE(poly_decompose_32_avx2_asm)+++#endif /* MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_NO_SIGN_API && \+ !MLD_CONFIG_MULTILEVEL_NO_SHARED && \+ (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 65 \+ || MLD_CONFIG_PARAMETER_SET == 87) */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
@@ -0,0 +1,489 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [REF_AVX2]+ * CRYSTALS-Dilithium optimized AVX2 implementation+ * Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/dilithium/tree/master/avx2+ */++/*+ * This file is derived from the public domain+ * AVX2 Dilithium implementation @[REF_AVX2].+ *+ * The algorithm for Decompose(r) (more specifically the handling for the+ * wrap-around cases) is modified. See the AVX2 intrinsics version+ * (poly_decompose_88_avx2.c, predecessor of this file) for a more detailed+ * comparison.+ */+++/*yaml+ Name: poly_decompose_88_avx2_asm+ Description: x86_64 AVX2 coefficient decomposition (alpha = 2*(Q-1)/88).+ For all coefficients c of the input polynomial, compute high and low bits+ c0, c1 such that c = c1*ALPHA + c0 with -ALPHA/2 < c0 <= ALPHA/2, except if+ c1 = (Q-1)/ALPHA where we set c1 = 0 and -ALPHA/2 <= c0 = c - Q < 0.+ Assumes coefficients to be standard (unsigned canonical) representatives,+ i.e. 0 <= c < Q. For ML-DSA-44 (gamma2 = (Q-1)/88, alpha = 2*gamma2).+ Signature: void mld_poly_decompose_88_avx2_asm(int32_t *a1, int32_t *a0)+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 1024+ permissions: write-only+ c_parameter: int32_t *a1+ description: Output high-part polynomial (256 x int32_t)+ rsi:+ type: buffer+ size_bytes: 1024+ permissions: read/write+ c_parameter: int32_t *a0+ description: Input polynomial / output low-part (256 x int32_t)+*/++#include "../../../common.h"++#if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \+ !defined(MLD_CONFIG_NO_SIGN_API) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \+ (defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \+ MLD_CONFIG_PARAMETER_SET == 44)+++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/x86_64/src/mldsa_poly_decompose_88_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(poly_decompose_88_avx2_asm)+MLD_ASM_FN_SYMBOL(poly_decompose_88_avx2_asm)++ .cfi_startproc+ movl $0x7f, %eax+ vmovd %eax, %xmm10+ vpbroadcastd %xmm10, %ymm10+ movl $0x2c0b, %eax # imm = 0x2C0B+ vmovd %eax, %xmm11+ vpbroadcastd %xmm11, %ymm11+ movl $0x80, %eax+ vmovd %eax, %xmm12+ vpbroadcastd %xmm12, %ymm12+ movl $0x7e6c00, %eax # imm = 0x7E6C00+ vmovd %eax, %xmm13+ vpbroadcastd %xmm13, %ymm13+ movl $0x2e800, %eax # imm = 0x2E800+ vmovd %eax, %xmm14+ vpbroadcastd %xmm14, %ymm14+ vmovdqa (%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, (%rdi)+ vmovdqa %ymm2, (%rsi)+ vmovdqa 0x20(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x20(%rdi)+ vmovdqa %ymm2, 0x20(%rsi)+ vmovdqa 0x40(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x40(%rdi)+ vmovdqa %ymm2, 0x40(%rsi)+ vmovdqa 0x60(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x60(%rdi)+ vmovdqa %ymm2, 0x60(%rsi)+ vmovdqa 0x80(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x80(%rdi)+ vmovdqa %ymm2, 0x80(%rsi)+ vmovdqa 0xa0(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0xa0(%rdi)+ vmovdqa %ymm2, 0xa0(%rsi)+ vmovdqa 0xc0(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0xc0(%rdi)+ vmovdqa %ymm2, 0xc0(%rsi)+ vmovdqa 0xe0(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0xe0(%rdi)+ vmovdqa %ymm2, 0xe0(%rsi)+ vmovdqa 0x100(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x100(%rdi)+ vmovdqa %ymm2, 0x100(%rsi)+ vmovdqa 0x120(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x120(%rdi)+ vmovdqa %ymm2, 0x120(%rsi)+ vmovdqa 0x140(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x140(%rdi)+ vmovdqa %ymm2, 0x140(%rsi)+ vmovdqa 0x160(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x160(%rdi)+ vmovdqa %ymm2, 0x160(%rsi)+ vmovdqa 0x180(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x180(%rdi)+ vmovdqa %ymm2, 0x180(%rsi)+ vmovdqa 0x1a0(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x1a0(%rdi)+ vmovdqa %ymm2, 0x1a0(%rsi)+ vmovdqa 0x1c0(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x1c0(%rdi)+ vmovdqa %ymm2, 0x1c0(%rsi)+ vmovdqa 0x1e0(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x1e0(%rdi)+ vmovdqa %ymm2, 0x1e0(%rsi)+ vmovdqa 0x200(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x200(%rdi)+ vmovdqa %ymm2, 0x200(%rsi)+ vmovdqa 0x220(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x220(%rdi)+ vmovdqa %ymm2, 0x220(%rsi)+ vmovdqa 0x240(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x240(%rdi)+ vmovdqa %ymm2, 0x240(%rsi)+ vmovdqa 0x260(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x260(%rdi)+ vmovdqa %ymm2, 0x260(%rsi)+ vmovdqa 0x280(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x280(%rdi)+ vmovdqa %ymm2, 0x280(%rsi)+ vmovdqa 0x2a0(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x2a0(%rdi)+ vmovdqa %ymm2, 0x2a0(%rsi)+ vmovdqa 0x2c0(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x2c0(%rdi)+ vmovdqa %ymm2, 0x2c0(%rsi)+ vmovdqa 0x2e0(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x2e0(%rdi)+ vmovdqa %ymm2, 0x2e0(%rsi)+ vmovdqa 0x300(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x300(%rdi)+ vmovdqa %ymm2, 0x300(%rsi)+ vmovdqa 0x320(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x320(%rdi)+ vmovdqa %ymm2, 0x320(%rsi)+ vmovdqa 0x340(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x340(%rdi)+ vmovdqa %ymm2, 0x340(%rsi)+ vmovdqa 0x360(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x360(%rdi)+ vmovdqa %ymm2, 0x360(%rsi)+ vmovdqa 0x380(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x380(%rdi)+ vmovdqa %ymm2, 0x380(%rsi)+ vmovdqa 0x3a0(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x3a0(%rdi)+ vmovdqa %ymm2, 0x3a0(%rsi)+ vmovdqa 0x3c0(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x3c0(%rdi)+ vmovdqa %ymm2, 0x3c0(%rsi)+ vmovdqa 0x3e0(%rsi), %ymm0+ vpaddd %ymm10, %ymm0, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm11, %ymm1, %ymm1+ vpmulhrsw %ymm12, %ymm1, %ymm1+ vpcmpgtd %ymm13, %ymm0, %ymm3+ vpmulld %ymm14, %ymm1, %ymm2+ vpsubd %ymm2, %ymm0, %ymm2+ vpandn %ymm1, %ymm3, %ymm1+ vpaddd %ymm3, %ymm2, %ymm2+ vmovdqa %ymm1, 0x3e0(%rdi)+ vmovdqa %ymm2, 0x3e0(%rsi)+ retq+ .cfi_endproc++MLD_ASM_FN_SIZE(poly_decompose_88_avx2_asm)+++#endif /* MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_NO_SIGN_API && \+ !MLD_CONFIG_MULTILEVEL_NO_SHARED && \+ (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 44) \+ */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
@@ -0,0 +1,123 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [REF_AVX2]+ * CRYSTALS-Dilithium optimized AVX2 implementation+ * Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/dilithium/tree/master/avx2+ */++/*+ * This file is derived from the public domain+ * AVX2 Dilithium implementation @[REF_AVX2].+ */+++/*yaml+ Name: poly_use_hint_32_avx2_asm+ Description: x86_64 AVX2 hint application (alpha = (Q-1)/32).+ Use the hint polynomial h to correct the high bits of the polynomial a,+ in place. Variant for parameter sets ML-DSA-65 and ML-DSA-87+ (GAMMA2 = (Q-1)/32).+ Signature: void mld_poly_use_hint_32_avx2_asm(int32_t *a, const int32_t *h)+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 1024+ permissions: read/write+ c_parameter: int32_t *a+ description: Input/output polynomial (256 x int32_t)+ rsi:+ type: buffer+ size_bytes: 1024+ permissions: read-only+ c_parameter: const int32_t *h+ description: Hint polynomial (256 x int32_t)+*/++#include "../../../common.h"++#if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \+ !defined(MLD_CONFIG_NO_VERIFY_API) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \+ (defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \+ (MLD_CONFIG_PARAMETER_SET == 65 || MLD_CONFIG_PARAMETER_SET == 87))+++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/x86_64/src/mldsa_poly_use_hint_32_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(poly_use_hint_32_avx2_asm)+MLD_ASM_FN_SYMBOL(poly_use_hint_32_avx2_asm)++ .cfi_startproc+ movl $0x7f, %ecx+ movl $0x401, %r8d # imm = 0x401+ vmovd %r8d, %xmm8+ vpbroadcastd %xmm8, %ymm8+ xorl %eax, %eax+ vpxor %xmm6, %xmm6, %xmm6+ vmovd %ecx, %xmm5+ movl $0x7be100, %ecx # imm = 0x7BE100+ movl $0x200, %r9d # imm = 0x200+ vmovd %r9d, %xmm7+ vpbroadcastd %xmm7, %ymm7+ vmovd %ecx, %xmm4+ movl $0xf, %ecx+ vpbroadcastd %xmm5, %ymm5+ vmovd %ecx, %xmm3+ vpbroadcastd %xmm4, %ymm4+ vpbroadcastd %xmm3, %ymm3++Lmld_poly_use_hint_32_avx2_asm_loop:+ vmovdqa (%rdi), %ymm0+ vmovdqa (%rsi), %ymm2+ vpaddd %ymm0, %ymm5, %ymm1+ vpsrld $0x7, %ymm1, %ymm1+ vpmulhuw %ymm8, %ymm1, %ymm1+ vpmulhrsw %ymm7, %ymm1, %ymm1+ vpcmpgtd %ymm4, %ymm0, %ymm11+ vpandn %ymm1, %ymm11, %ymm9+ vpslld $0xa, %ymm1, %ymm10+ vpsubd %ymm1, %ymm10, %ymm1+ vpslld $0x9, %ymm1, %ymm1+ vpsubd %ymm1, %ymm0, %ymm0+ vpaddd %ymm11, %ymm0, %ymm0+ vpcmpgtd %ymm6, %ymm0, %ymm0+ vpandn %ymm2, %ymm0, %ymm0+ vpslld $0x1, %ymm0, %ymm0+ vpsubd %ymm0, %ymm2, %ymm2+ vpaddd %ymm9, %ymm2, %ymm2+ vpand %ymm3, %ymm2, %ymm2+ vmovdqa %ymm2, (%rdi)+ addq $0x20, %rdi+ addq $0x20, %rsi+ addq $0x20, %rax+ cmpq $0x400, %rax # imm = 0x400+ jne Lmld_poly_use_hint_32_avx2_asm_loop+ retq+ .cfi_endproc++MLD_ASM_FN_SIZE(poly_use_hint_32_avx2_asm)+++#endif /* MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_NO_VERIFY_API && \+ !MLD_CONFIG_MULTILEVEL_NO_SHARED && \+ (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 65 \+ || MLD_CONFIG_PARAMETER_SET == 87) */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
@@ -0,0 +1,125 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [REF_AVX2]+ * CRYSTALS-Dilithium optimized AVX2 implementation+ * Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/dilithium/tree/master/avx2+ */++/*+ * This file is derived from the public domain+ * AVX2 Dilithium implementation @[REF_AVX2].+ */+++/*yaml+ Name: poly_use_hint_88_avx2_asm+ Description: x86_64 AVX2 hint application (alpha = (Q-1)/88).+ Use the hint polynomial h to correct the high bits of the polynomial a,+ in place. Variant for parameter set ML-DSA-44 (GAMMA2 = (Q-1)/88).+ Signature: void mld_poly_use_hint_88_avx2_asm(int32_t *a, const int32_t *h)+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 1024+ permissions: read/write+ c_parameter: int32_t *a+ description: Input/output polynomial (256 x int32_t)+ rsi:+ type: buffer+ size_bytes: 1024+ permissions: read-only+ c_parameter: const int32_t *h+ description: Hint polynomial (256 x int32_t)+*/++#include "../../../common.h"++#if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \+ !defined(MLD_CONFIG_NO_VERIFY_API) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \+ (defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \+ MLD_CONFIG_PARAMETER_SET == 44)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/x86_64/src/mldsa_poly_use_hint_88_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(poly_use_hint_88_avx2_asm)+MLD_ASM_FN_SYMBOL(poly_use_hint_88_avx2_asm)++ .cfi_startproc+ movl $0x7f, %ecx+ xorl %eax, %eax+ vpxor %xmm5, %xmm5, %xmm5+ movl $0x2c0b, %r8d # imm = 0x2C0B+ vmovd %r8d, %xmm8+ vpbroadcastd %xmm8, %ymm8+ vmovd %ecx, %xmm4+ movl $0x7e6c00, %ecx # imm = 0x7E6C00+ movl $0x80, %r9d+ vmovd %r9d, %xmm7+ vpbroadcastd %xmm7, %ymm7+ movl $0x2b, %r10d+ vmovd %r10d, %xmm6+ vpbroadcastd %xmm6, %ymm6+ vmovd %ecx, %xmm3+ vpbroadcastd %xmm4, %ymm4+ vpbroadcastd %xmm3, %ymm3++Lmld_poly_use_hint_88_avx2_asm_loop:+ vmovdqa (%rdi), %ymm0+ vmovdqa (%rsi), %ymm1+ vpaddd %ymm0, %ymm4, %ymm10+ vpsrld $0x7, %ymm10, %ymm10+ vpmulhuw %ymm8, %ymm10, %ymm10+ vpmulhrsw %ymm7, %ymm10, %ymm10+ vpcmpgtd %ymm3, %ymm0, %ymm11+ vpandn %ymm10, %ymm11, %ymm9+ vpslld $0x1, %ymm10, %ymm12+ vpaddd %ymm10, %ymm12, %ymm12+ vpslld $0x5, %ymm12, %ymm10+ vpsubd %ymm12, %ymm10, %ymm10+ vpslld $0xb, %ymm10, %ymm10+ vpsubd %ymm10, %ymm0, %ymm0+ vpaddd %ymm11, %ymm0, %ymm0+ vpcmpgtd %ymm5, %ymm0, %ymm0+ vpandn %ymm1, %ymm0, %ymm0+ vpslld $0x1, %ymm0, %ymm0+ vpsubd %ymm0, %ymm1, %ymm0+ vpaddd %ymm9, %ymm0, %ymm0+ vpblendvb %ymm0, %ymm6, %ymm0, %ymm0+ vpcmpgtd %ymm6, %ymm0, %ymm1+ vpandn %ymm0, %ymm1, %ymm0+ vmovdqa %ymm0, (%rdi)+ addq $0x20, %rdi+ addq $0x20, %rsi+ addq $0x20, %rax+ cmpq $0x400, %rax # imm = 0x400+ jne Lmld_poly_use_hint_88_avx2_asm_loop+ retq+ .cfi_endproc++MLD_ASM_FN_SIZE(poly_use_hint_88_avx2_asm)+++#endif /* MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_NO_VERIFY_API && \+ !MLD_CONFIG_MULTILEVEL_NO_SHARED && \+ (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 44) \+ */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
@@ -0,0 +1,355 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [REF_AVX2]+ * CRYSTALS-Dilithium optimized AVX2 implementation+ * Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/dilithium/tree/master/avx2+ */++/*+ * This file is derived from the public domain+ * AVX2 Dilithium implementation @[REF_AVX2].+ */+++/*yaml+ Name: polyz_unpack_17_avx2_asm+ Description: x86_64 AVX2 unpacking of 17-bit packed coefficients.+ Unpack polynomial z with 18-bit packed coefficients (GAMMA1 = 2^17),+ mapping packed [0, 2^18-1] to signed [-(2^17-1), 2^17] via GAMMA1 - x.+ Signature: void mld_polyz_unpack_17_avx2_asm(int32_t *r, const uint8_t *a)+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 1024+ permissions: write-only+ c_parameter: int32_t *r+ description: Output polynomial (256 x int32_t)+ rsi:+ type: buffer+ size_bytes: 576+ permissions: read-only+ c_parameter: const uint8_t *a+ description: Packed input bytes+*/++#include "../../../common.h"++#if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \+ (!defined(MLD_CONFIG_NO_SIGN_API) || \+ !defined(MLD_CONFIG_NO_VERIFY_API)) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \+ (defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \+ MLD_CONFIG_PARAMETER_SET == 44)+++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/x86_64/src/mldsa_polyz_unpack_17_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(polyz_unpack_17_avx2_asm)+MLD_ASM_FN_SYMBOL(polyz_unpack_17_avx2_asm)++ .cfi_startproc+ movabsq $-0xfbfcfd00fdff00, %rax # imm = 0xFF040302FF020100+ vmovq %rax, %xmm1+ movabsq $-0xf7f8f900f9fafc, %rax # imm = 0xFF080706FF060504+ vpinsrq $0x1, %rax, %xmm1, %xmm1+ movabsq $-0xe4e5e600e6e7e9, %rax # imm = 0xFF1B1A19FF191817+ vmovq %rax, %xmm5+ movabsq $-0xe0e1e200e2e3e5, %rax # imm = 0xFF1F1E1DFF1D1C1B+ vpinsrq $0x1, %rax, %xmm5, %xmm5+ vinserti128 $0x1, %xmm5, %ymm1, %ymm1+ movabsq $0x200000000, %rax # imm = 0x200000000+ vmovq %rax, %xmm2+ movabsq $0x600000004, %rax # imm = 0x600000004+ vpinsrq $0x1, %rax, %xmm2, %xmm2+ vinserti128 $0x1, %xmm2, %ymm2, %ymm2+ movl $0x3ffff, %eax # imm = 0x3FFFF+ vmovd %eax, %xmm3+ vpbroadcastd %xmm3, %ymm3+ movl $0x20000, %eax # imm = 0x20000+ vmovd %eax, %xmm4+ vpbroadcastd %xmm4, %ymm4+ vmovdqu (%rsi), %xmm0+ vmovdqu 0x2(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, (%rdi)+ vmovdqu 0x12(%rsi), %xmm0+ vmovdqu 0x14(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x20(%rdi)+ vmovdqu 0x24(%rsi), %xmm0+ vmovdqu 0x26(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x40(%rdi)+ vmovdqu 0x36(%rsi), %xmm0+ vmovdqu 0x38(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x60(%rdi)+ vmovdqu 0x48(%rsi), %xmm0+ vmovdqu 0x4a(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x80(%rdi)+ vmovdqu 0x5a(%rsi), %xmm0+ vmovdqu 0x5c(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0xa0(%rdi)+ vmovdqu 0x6c(%rsi), %xmm0+ vmovdqu 0x6e(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0xc0(%rdi)+ vmovdqu 0x7e(%rsi), %xmm0+ vmovdqu 0x80(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0xe0(%rdi)+ vmovdqu 0x90(%rsi), %xmm0+ vmovdqu 0x92(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x100(%rdi)+ vmovdqu 0xa2(%rsi), %xmm0+ vmovdqu 0xa4(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x120(%rdi)+ vmovdqu 0xb4(%rsi), %xmm0+ vmovdqu 0xb6(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x140(%rdi)+ vmovdqu 0xc6(%rsi), %xmm0+ vmovdqu 0xc8(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x160(%rdi)+ vmovdqu 0xd8(%rsi), %xmm0+ vmovdqu 0xda(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x180(%rdi)+ vmovdqu 0xea(%rsi), %xmm0+ vmovdqu 0xec(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x1a0(%rdi)+ vmovdqu 0xfc(%rsi), %xmm0+ vmovdqu 0xfe(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x1c0(%rdi)+ vmovdqu 0x10e(%rsi), %xmm0+ vmovdqu 0x110(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x1e0(%rdi)+ vmovdqu 0x120(%rsi), %xmm0+ vmovdqu 0x122(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x200(%rdi)+ vmovdqu 0x132(%rsi), %xmm0+ vmovdqu 0x134(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x220(%rdi)+ vmovdqu 0x144(%rsi), %xmm0+ vmovdqu 0x146(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x240(%rdi)+ vmovdqu 0x156(%rsi), %xmm0+ vmovdqu 0x158(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x260(%rdi)+ vmovdqu 0x168(%rsi), %xmm0+ vmovdqu 0x16a(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x280(%rdi)+ vmovdqu 0x17a(%rsi), %xmm0+ vmovdqu 0x17c(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x2a0(%rdi)+ vmovdqu 0x18c(%rsi), %xmm0+ vmovdqu 0x18e(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x2c0(%rdi)+ vmovdqu 0x19e(%rsi), %xmm0+ vmovdqu 0x1a0(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x2e0(%rdi)+ vmovdqu 0x1b0(%rsi), %xmm0+ vmovdqu 0x1b2(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x300(%rdi)+ vmovdqu 0x1c2(%rsi), %xmm0+ vmovdqu 0x1c4(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x320(%rdi)+ vmovdqu 0x1d4(%rsi), %xmm0+ vmovdqu 0x1d6(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x340(%rdi)+ vmovdqu 0x1e6(%rsi), %xmm0+ vmovdqu 0x1e8(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x360(%rdi)+ vmovdqu 0x1f8(%rsi), %xmm0+ vmovdqu 0x1fa(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x380(%rdi)+ vmovdqu 0x20a(%rsi), %xmm0+ vmovdqu 0x20c(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x3a0(%rdi)+ vmovdqu 0x21c(%rsi), %xmm0+ vmovdqu 0x21e(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x3c0(%rdi)+ vmovdqu 0x22e(%rsi), %xmm0+ vmovdqu 0x230(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x3e0(%rdi)+ retq+ .cfi_endproc++MLD_ASM_FN_SIZE(polyz_unpack_17_avx2_asm)+++#endif /* MLD_ARITH_BACKEND_X86_64_DEFAULT && (!MLD_CONFIG_NO_SIGN_API || \+ !MLD_CONFIG_NO_VERIFY_API) && !MLD_CONFIG_MULTILEVEL_NO_SHARED && \+ (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 44) \+ */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
@@ -0,0 +1,355 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [REF_AVX2]+ * CRYSTALS-Dilithium optimized AVX2 implementation+ * Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/dilithium/tree/master/avx2+ */++/*+ * This file is derived from the public domain+ * AVX2 Dilithium implementation @[REF_AVX2].+ */+++/*yaml+ Name: polyz_unpack_19_avx2_asm+ Description: x86_64 AVX2 unpacking of 19-bit packed coefficients.+ Unpack polynomial z with 20-bit packed coefficients (GAMMA1 = 2^19),+ mapping packed [0, 2^20-1] to signed [-(2^19-1), 2^19] via GAMMA1 - x.+ Signature: void mld_polyz_unpack_19_avx2_asm(int32_t *r, const uint8_t *a)+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 1024+ permissions: write-only+ c_parameter: int32_t *r+ description: Output polynomial (256 x int32_t)+ rsi:+ type: buffer+ size_bytes: 640+ permissions: read-only+ c_parameter: const uint8_t *a+ description: Packed input bytes+*/++#include "../../../common.h"++#if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \+ (!defined(MLD_CONFIG_NO_SIGN_API) || \+ !defined(MLD_CONFIG_NO_VERIFY_API)) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED) && \+ (defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \+ (MLD_CONFIG_PARAMETER_SET == 65 || MLD_CONFIG_PARAMETER_SET == 87))+++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/x86_64/src/mldsa_polyz_unpack_19_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(polyz_unpack_19_avx2_asm)+MLD_ASM_FN_SYMBOL(polyz_unpack_19_avx2_asm)++ .cfi_startproc+ movabsq $-0xfbfcfd00fdff00, %rax # imm = 0xFF040302FF020100+ vmovq %rax, %xmm1+ movabsq $-0xf6f7f800f8f9fb, %rax # imm = 0xFF090807FF070605+ vpinsrq $0x1, %rax, %xmm1, %xmm1+ movabsq $-0xe5e6e700e7e8ea, %rax # imm = 0xFF1A1918FF181716+ vmovq %rax, %xmm5+ movabsq $-0xe0e1e200e2e3e5, %rax # imm = 0xFF1F1E1DFF1D1C1B+ vpinsrq $0x1, %rax, %xmm5, %xmm5+ vinserti128 $0x1, %xmm5, %ymm1, %ymm1+ movabsq $0x400000000, %rax # imm = 0x400000000+ vmovq %rax, %xmm2+ movabsq $0x400000000, %rax # imm = 0x400000000+ vpinsrq $0x1, %rax, %xmm2, %xmm2+ vinserti128 $0x1, %xmm2, %ymm2, %ymm2+ movl $0xfffff, %eax # imm = 0xFFFFF+ vmovd %eax, %xmm3+ vpbroadcastd %xmm3, %ymm3+ movl $0x80000, %eax # imm = 0x80000+ vmovd %eax, %xmm4+ vpbroadcastd %xmm4, %ymm4+ vmovdqu (%rsi), %xmm0+ vmovdqu 0x4(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, (%rdi)+ vmovdqu 0x14(%rsi), %xmm0+ vmovdqu 0x18(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x20(%rdi)+ vmovdqu 0x28(%rsi), %xmm0+ vmovdqu 0x2c(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x40(%rdi)+ vmovdqu 0x3c(%rsi), %xmm0+ vmovdqu 0x40(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x60(%rdi)+ vmovdqu 0x50(%rsi), %xmm0+ vmovdqu 0x54(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x80(%rdi)+ vmovdqu 0x64(%rsi), %xmm0+ vmovdqu 0x68(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0xa0(%rdi)+ vmovdqu 0x78(%rsi), %xmm0+ vmovdqu 0x7c(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0xc0(%rdi)+ vmovdqu 0x8c(%rsi), %xmm0+ vmovdqu 0x90(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0xe0(%rdi)+ vmovdqu 0xa0(%rsi), %xmm0+ vmovdqu 0xa4(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x100(%rdi)+ vmovdqu 0xb4(%rsi), %xmm0+ vmovdqu 0xb8(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x120(%rdi)+ vmovdqu 0xc8(%rsi), %xmm0+ vmovdqu 0xcc(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x140(%rdi)+ vmovdqu 0xdc(%rsi), %xmm0+ vmovdqu 0xe0(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x160(%rdi)+ vmovdqu 0xf0(%rsi), %xmm0+ vmovdqu 0xf4(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x180(%rdi)+ vmovdqu 0x104(%rsi), %xmm0+ vmovdqu 0x108(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x1a0(%rdi)+ vmovdqu 0x118(%rsi), %xmm0+ vmovdqu 0x11c(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x1c0(%rdi)+ vmovdqu 0x12c(%rsi), %xmm0+ vmovdqu 0x130(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x1e0(%rdi)+ vmovdqu 0x140(%rsi), %xmm0+ vmovdqu 0x144(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x200(%rdi)+ vmovdqu 0x154(%rsi), %xmm0+ vmovdqu 0x158(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x220(%rdi)+ vmovdqu 0x168(%rsi), %xmm0+ vmovdqu 0x16c(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x240(%rdi)+ vmovdqu 0x17c(%rsi), %xmm0+ vmovdqu 0x180(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x260(%rdi)+ vmovdqu 0x190(%rsi), %xmm0+ vmovdqu 0x194(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x280(%rdi)+ vmovdqu 0x1a4(%rsi), %xmm0+ vmovdqu 0x1a8(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x2a0(%rdi)+ vmovdqu 0x1b8(%rsi), %xmm0+ vmovdqu 0x1bc(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x2c0(%rdi)+ vmovdqu 0x1cc(%rsi), %xmm0+ vmovdqu 0x1d0(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x2e0(%rdi)+ vmovdqu 0x1e0(%rsi), %xmm0+ vmovdqu 0x1e4(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x300(%rdi)+ vmovdqu 0x1f4(%rsi), %xmm0+ vmovdqu 0x1f8(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x320(%rdi)+ vmovdqu 0x208(%rsi), %xmm0+ vmovdqu 0x20c(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x340(%rdi)+ vmovdqu 0x21c(%rsi), %xmm0+ vmovdqu 0x220(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x360(%rdi)+ vmovdqu 0x230(%rsi), %xmm0+ vmovdqu 0x234(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x380(%rdi)+ vmovdqu 0x244(%rsi), %xmm0+ vmovdqu 0x248(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x3a0(%rdi)+ vmovdqu 0x258(%rsi), %xmm0+ vmovdqu 0x25c(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x3c0(%rdi)+ vmovdqu 0x26c(%rsi), %xmm0+ vmovdqu 0x270(%rsi), %xmm5+ vinserti128 $0x1, %xmm5, %ymm0, %ymm0+ vpshufb %ymm1, %ymm0, %ymm0+ vpsrlvd %ymm2, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubd %ymm0, %ymm4, %ymm0+ vmovdqa %ymm0, 0x3e0(%rdi)+ retq+ .cfi_endproc++MLD_ASM_FN_SIZE(polyz_unpack_19_avx2_asm)+++#endif /* MLD_ARITH_BACKEND_X86_64_DEFAULT && (!MLD_CONFIG_NO_SIGN_API || \+ !MLD_CONFIG_NO_VERIFY_API) && !MLD_CONFIG_MULTILEVEL_NO_SHARED && \+ (MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 65 \+ || MLD_CONFIG_PARAMETER_SET == 87) */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
@@ -0,0 +1,132 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [REF_AVX2]+ * CRYSTALS-Dilithium optimized AVX2 implementation+ * Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/dilithium/tree/master/avx2+ */++/*+ * This file is derived from the public domain+ * AVX2 Dilithium implementation @[REF_AVX2].+ */++/*yaml+ Name: rej_uniform_avx2_asm+ Description: x86_64 AVX2 rejection sampling of uniform coefficients mod q.+ Extract 23-bit values from the input byte buffer and accept those that are+ < MLDSA_Q, writing accepted coefficients to the output buffer. Returns the+ number of valid coefficients written (at most 256).+ Signature: unsigned mld_rej_uniform_avx2_asm(int32_t *r, const uint8_t buf[840], const uint8_t table[256][8])+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 1024+ permissions: write-only+ c_parameter: int32_t *r+ description: Output buffer (256 x int32_t)+ rsi:+ type: buffer+ size_bytes: 840+ permissions: read-only+ c_parameter: const uint8_t buf[840]+ description: Input buffer (MLD_POLY_UNIFORM_NBLOCKS * SHAKE128_RATE = 5 * 168)+ rdx:+ type: buffer+ size_bytes: 2048+ permissions: read-only+ c_parameter: const uint8_t table[256][8]+ description: Lookup table (256 x 8 x uint8_t)+*/++#include "../../../common.h"+#if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/x86_64/src/mldsa_rej_uniform_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(rej_uniform_avx2_asm)+MLD_ASM_FN_SYMBOL(rej_uniform_avx2_asm)++ .cfi_startproc+ movabsq $-0xfafbfc00fdff00, %r10 # imm = 0xFF050403FF020100+ vmovq %r10, %xmm0+ movabsq $-0xf4f5f600f7f8fa, %r10 # imm = 0xFF0B0A09FF080706+ vpinsrq $0x1, %r10, %xmm0, %xmm0+ movabsq $-0xf6f7f800f9fafc, %r10 # imm = 0xFF090807FF060504+ vmovq %r10, %xmm3+ movabsq $-0xf0f1f200f3f4f6, %r10 # imm = 0xFF0F0E0DFF0C0B0A+ vpinsrq $0x1, %r10, %xmm3, %xmm3+ vinserti128 $0x1, %xmm3, %ymm0, %ymm0+ movl $0x7fffff, %r8d # imm = 0x7FFFFF+ vmovd %r8d, %xmm1+ vpbroadcastd %xmm1, %ymm1+ movl $0x7fe001, %r8d # imm = 0x7FE001+ vmovd %r8d, %xmm2+ vpbroadcastd %xmm2, %ymm2+ xorl %eax, %eax+ xorl %ecx, %ecx++Lmld_rej_uniform_avx2_asm_loop:+ cmpl $0xf8, %eax+ ja Lmld_rej_uniform_avx2_asm_scalar+ cmpl $0x328, %ecx # imm = 0x328+ ja Lmld_rej_uniform_avx2_asm_scalar+ vmovdqu (%rsi,%rcx), %ymm3+ addl $0x18, %ecx+ vpermq $0x94, %ymm3, %ymm3 # ymm3 = ymm3[0,1,1,2]+ vpshufb %ymm0, %ymm3, %ymm3+ vpand %ymm1, %ymm3, %ymm3+ vpsubd %ymm2, %ymm3, %ymm4+ vmovmskps %ymm4, %r8d+ popcntl %r8d, %r9d+ vmovq (%rdx,%r8,8), %xmm4+ vpmovzxbd %xmm4, %ymm4 # ymm4 = xmm4[0],zero,zero,zero,xmm4[1],zero,zero,zero,xmm4[2],zero,zero,zero,xmm4[3],zero,zero,zero,xmm4[4],zero,zero,zero,xmm4[5],zero,zero,zero,xmm4[6],zero,zero,zero,xmm4[7],zero,zero,zero+ vpermd %ymm3, %ymm4, %ymm3+ vmovdqu %ymm3, (%rdi,%rax,4)+ addl %r9d, %eax+ jmp Lmld_rej_uniform_avx2_asm_loop++Lmld_rej_uniform_avx2_asm_scalar:+ cmpl $0x100, %eax # imm = 0x100+ jae Lmld_rej_uniform_avx2_asm_done+ cmpl $0x345, %ecx # imm = 0x345+ ja Lmld_rej_uniform_avx2_asm_done+ movzwl (%rsi,%rcx), %r8d+ movzbl 0x2(%rsi,%rcx), %r9d+ shll $0x10, %r9d+ orl %r9d, %r8d+ andl $0x7fffff, %r8d # imm = 0x7FFFFF+ addl $0x3, %ecx+ cmpl $0x7fe001, %r8d # imm = 0x7FE001+ jae Lmld_rej_uniform_avx2_asm_scalar+ movl %r8d, (%rdi,%rax,4)+ addl $0x1, %eax+ jmp Lmld_rej_uniform_avx2_asm_scalar++Lmld_rej_uniform_avx2_asm_done:+ retq+ .cfi_endproc++MLD_ASM_FN_SIZE(rej_uniform_avx2_asm)++#endif /* MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_MULTILEVEL_NO_SHARED \+ */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
@@ -0,0 +1,205 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [REF_AVX2]+ * CRYSTALS-Dilithium optimized AVX2 implementation+ * Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/dilithium/tree/master/avx2+ */++/*+ * This file is derived from the public domain+ * AVX2 Dilithium implementation @[REF_AVX2].+ */++/*yaml+ Name: rej_uniform_eta2_avx2_asm+ Description: x86_64 AVX2 rejection sampling of eta=2 secret coefficients.+ Extracts 4-bit nibbles from the input byte buffer, accepts those < 15,+ and maps an accepted nibble n to the coefficient 2 - (n mod 5) via a+ centered modulo-5 reduction, producing values in [-2, 2].+ Signature: unsigned mld_rej_uniform_eta2_avx2_asm(int32_t *r, const uint8_t *buf, const uint8_t table[256][8])+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 1024+ permissions: write-only+ c_parameter: int32_t *r+ description: Output coefficient buffer (256 x int32_t)+ rsi:+ type: buffer+ size_bytes: 136+ permissions: read-only+ c_parameter: const uint8_t *buf+ description: Input byte buffer (MLD_AVX2_REJ_UNIFORM_ETA2_BUFLEN = 136)+ rdx:+ type: buffer+ size_bytes: 2048+ permissions: read-only+ c_parameter: const uint8_t *table+ description: Lookup table (256 x 8 uint8_t = mld_rej_uniform_table)+*/++#include "../../../common.h"+#if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/x86_64/src/mldsa_rej_uniform_eta2_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(rej_uniform_eta2_avx2_asm)+MLD_ASM_FN_SYMBOL(rej_uniform_eta2_avx2_asm)++ .cfi_startproc+ movl $0xf0f0f0f, %r8d # imm = 0xF0F0F0F+ vmovd %r8d, %xmm3+ vpbroadcastd %xmm3, %ymm3+ movl $0x2020202, %r8d # imm = 0x2020202+ vmovd %r8d, %xmm4+ vpbroadcastd %xmm4, %ymm4+ movl $0xf0f0f0f, %r8d # imm = 0xF0F0F0F+ vmovd %r8d, %xmm5+ vpbroadcastd %xmm5, %ymm5+ movl $0xffffe660, %r8d # imm = 0xFFFFE660+ vpinsrw $0x0, %r8d, %xmm6, %xmm6+ vpbroadcastw %xmm6, %ymm6+ movl $0x5, %r8d+ vpinsrw $0x0, %r8d, %xmm7, %xmm7+ vpbroadcastw %xmm7, %ymm7+ xorl %eax, %eax+ xorl %ecx, %ecx++Lmld_rej_uniform_eta2_avx2_asm_loop:+ cmpl $0xf8, %eax+ ja Lmld_rej_uniform_eta2_avx2_asm_scalar+ cmpl $0x78, %ecx+ ja Lmld_rej_uniform_eta2_avx2_asm_scalar+ vpmovzxbw (%rsi,%rcx), %ymm0+ vpsllw $0x4, %ymm0, %ymm1+ vpor %ymm1, %ymm0, %ymm0+ vpand %ymm3, %ymm0, %ymm0+ vpsubb %ymm5, %ymm0, %ymm1+ vpsubb %ymm0, %ymm4, %ymm0+ vpmovmskb %ymm1, %r8d+ vextracti128 $0x0, %ymm0, %xmm8+ movzbl %r8b, %r10d+ vmovq (%rdx,%r10,8), %xmm9+ vpshufb %xmm9, %xmm8, %xmm9+ vpmovsxbd %xmm9, %ymm1+ vpmulhrsw %ymm6, %ymm1, %ymm2+ vpmullw %ymm7, %ymm2, %ymm2+ vpaddd %ymm2, %ymm1, %ymm1+ vmovdqu %ymm1, (%rdi,%rax,4)+ popcntl %r10d, %r9d+ addl %r9d, %eax+ shrl $0x8, %r8d+ addl $0x4, %ecx+ cmpl $0xf8, %eax+ ja Lmld_rej_uniform_eta2_avx2_asm_scalar+ vpsrldq $0x8, %xmm8, %xmm8 # xmm8 = xmm8[8,9,10,11,12,13,14,15],zero,zero,zero,zero,zero,zero,zero,zero+ movzbl %r8b, %r10d+ vmovq (%rdx,%r10,8), %xmm9+ vpshufb %xmm9, %xmm8, %xmm9+ vpmovsxbd %xmm9, %ymm1+ vpmulhrsw %ymm6, %ymm1, %ymm2+ vpmullw %ymm7, %ymm2, %ymm2+ vpaddd %ymm2, %ymm1, %ymm1+ vmovdqu %ymm1, (%rdi,%rax,4)+ popcntl %r10d, %r9d+ addl %r9d, %eax+ shrl $0x8, %r8d+ addl $0x4, %ecx+ cmpl $0xf8, %eax+ ja Lmld_rej_uniform_eta2_avx2_asm_scalar+ vextracti128 $0x1, %ymm0, %xmm8+ movzbl %r8b, %r10d+ vmovq (%rdx,%r10,8), %xmm9+ vpshufb %xmm9, %xmm8, %xmm9+ vpmovsxbd %xmm9, %ymm1+ vpmulhrsw %ymm6, %ymm1, %ymm2+ vpmullw %ymm7, %ymm2, %ymm2+ vpaddd %ymm2, %ymm1, %ymm1+ vmovdqu %ymm1, (%rdi,%rax,4)+ popcntl %r10d, %r9d+ addl %r9d, %eax+ shrl $0x8, %r8d+ addl $0x4, %ecx+ cmpl $0xf8, %eax+ ja Lmld_rej_uniform_eta2_avx2_asm_scalar+ vpsrldq $0x8, %xmm8, %xmm8 # xmm8 = xmm8[8,9,10,11,12,13,14,15],zero,zero,zero,zero,zero,zero,zero,zero+ movzbl %r8b, %r10d+ vmovq (%rdx,%r10,8), %xmm9+ vpshufb %xmm9, %xmm8, %xmm9+ vpmovsxbd %xmm9, %ymm1+ vpmulhrsw %ymm6, %ymm1, %ymm2+ vpmullw %ymm7, %ymm2, %ymm2+ vpaddd %ymm2, %ymm1, %ymm1+ vmovdqu %ymm1, (%rdi,%rax,4)+ popcntl %r10d, %r9d+ addl %r9d, %eax+ addl $0x4, %ecx+ jmp Lmld_rej_uniform_eta2_avx2_asm_loop++Lmld_rej_uniform_eta2_avx2_asm_scalar:+ cmpl $0x100, %eax # imm = 0x100+ jae Lmld_rej_uniform_eta2_avx2_asm_done+ cmpl $0x88, %ecx+ jae Lmld_rej_uniform_eta2_avx2_asm_done+ movzbl (%rsi,%rcx), %r11d+ incl %ecx+ movl %r11d, %r10d+ andl $0xf, %r10d+ cmpl $0xf, %r10d+ jae Lmld_rej_uniform_eta2_avx2_asm_high_nibble+ movl %r10d, %r11d+ imull $0xcd, %r11d, %r11d+ shrl $0xa, %r11d+ imull $0x5, %r11d, %r11d+ subl %r11d, %r10d+ movl $0x2, %r11d+ subl %r10d, %r11d+ movl %r11d, (%rdi,%rax,4)+ incl %eax+ cmpl $0x100, %eax # imm = 0x100+ jae Lmld_rej_uniform_eta2_avx2_asm_done++Lmld_rej_uniform_eta2_avx2_asm_high_nibble:+ movzbl -0x1(%rsi,%rcx), %r11d+ shrl $0x4, %r11d+ andl $0xf, %r11d+ cmpl $0xf, %r11d+ jae Lmld_rej_uniform_eta2_avx2_asm_scalar+ movl %r11d, %r10d+ imull $0xcd, %r10d, %r10d+ shrl $0xa, %r10d+ imull $0x5, %r10d, %r10d+ subl %r10d, %r11d+ movl $0x2, %r10d+ subl %r11d, %r10d+ movl %r10d, (%rdi,%rax,4)+ incl %eax+ jmp Lmld_rej_uniform_eta2_avx2_asm_scalar++Lmld_rej_uniform_eta2_avx2_asm_done:+ retq+ .cfi_endproc++MLD_ASM_FN_SIZE(rej_uniform_eta2_avx2_asm)++#endif /* MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_MULTILEVEL_NO_SHARED \+ */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
@@ -0,0 +1,176 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [REF_AVX2]+ * CRYSTALS-Dilithium optimized AVX2 implementation+ * Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/dilithium/tree/master/avx2+ */++/*+ * This file is derived from the public domain+ * AVX2 Dilithium implementation @[REF_AVX2].+ */++/*yaml+ Name: rej_uniform_eta4_avx2_asm+ Description: x86_64 AVX2 rejection sampling of eta=4 secret coefficients.+ Extracts 4-bit nibbles from the input byte buffer, accepts those < 9,+ and maps an accepted nibble n to the coefficient 4 - n, producing+ values in [-4, 4].+ Signature: unsigned mld_rej_uniform_eta4_avx2_asm(int32_t *r, const uint8_t *buf, const uint8_t table[256][8])+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 1024+ permissions: write-only+ c_parameter: int32_t *r+ description: Output coefficient buffer (256 x int32_t)+ rsi:+ type: buffer+ size_bytes: 272+ permissions: read-only+ c_parameter: const uint8_t *buf+ description: Input byte buffer (MLD_AVX2_REJ_UNIFORM_ETA4_BUFLEN = 272)+ rdx:+ type: buffer+ size_bytes: 2048+ permissions: read-only+ c_parameter: const uint8_t *table+ description: Lookup table (256 x 8 uint8_t = mld_rej_uniform_table)+*/++#include "../../../common.h"+#if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++/*+ * WARNING: This file is auto-derived from the mldsa-native source file+ * dev/x86_64/src/mldsa_rej_uniform_eta4_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLD_ASM_NAMESPACE(rej_uniform_eta4_avx2_asm)+MLD_ASM_FN_SYMBOL(rej_uniform_eta4_avx2_asm)++ .cfi_startproc+ movl $0xf0f0f0f, %r8d # imm = 0xF0F0F0F+ vmovd %r8d, %xmm2+ vpbroadcastd %xmm2, %ymm2+ movl $0x4040404, %r8d # imm = 0x4040404+ vmovd %r8d, %xmm3+ vpbroadcastd %xmm3, %ymm3+ movl $0x9090909, %r8d # imm = 0x9090909+ vmovd %r8d, %xmm4+ vpbroadcastd %xmm4, %ymm4+ xorl %eax, %eax+ xorl %ecx, %ecx++Lmld_rej_uniform_eta4_avx2_asm_loop:+ cmpl $0xf8, %eax+ ja Lmld_rej_uniform_eta4_avx2_asm_scalar+ cmpl $0x100, %ecx # imm = 0x100+ ja Lmld_rej_uniform_eta4_avx2_asm_scalar+ vpmovzxbw (%rsi,%rcx), %ymm0+ vpsllw $0x4, %ymm0, %ymm1+ vpor %ymm1, %ymm0, %ymm0+ vpand %ymm2, %ymm0, %ymm0+ vpsubb %ymm4, %ymm0, %ymm1+ vpsubb %ymm0, %ymm3, %ymm0+ vpmovmskb %ymm1, %r8d+ vextracti128 $0x0, %ymm0, %xmm5+ movzbl %r8b, %r10d+ vmovq (%rdx,%r10,8), %xmm6+ vpshufb %xmm6, %xmm5, %xmm6+ vpmovsxbd %xmm6, %ymm1+ vmovdqu %ymm1, (%rdi,%rax,4)+ popcntl %r10d, %r9d+ addl %r9d, %eax+ shrl $0x8, %r8d+ addl $0x4, %ecx+ cmpl $0xf8, %eax+ ja Lmld_rej_uniform_eta4_avx2_asm_scalar+ vpsrldq $0x8, %xmm5, %xmm5 # xmm5 = xmm5[8,9,10,11,12,13,14,15],zero,zero,zero,zero,zero,zero,zero,zero+ movzbl %r8b, %r10d+ vmovq (%rdx,%r10,8), %xmm6+ vpshufb %xmm6, %xmm5, %xmm6+ vpmovsxbd %xmm6, %ymm1+ vmovdqu %ymm1, (%rdi,%rax,4)+ popcntl %r10d, %r9d+ addl %r9d, %eax+ shrl $0x8, %r8d+ addl $0x4, %ecx+ cmpl $0xf8, %eax+ ja Lmld_rej_uniform_eta4_avx2_asm_scalar+ vextracti128 $0x1, %ymm0, %xmm5+ movzbl %r8b, %r10d+ vmovq (%rdx,%r10,8), %xmm6+ vpshufb %xmm6, %xmm5, %xmm6+ vpmovsxbd %xmm6, %ymm1+ vmovdqu %ymm1, (%rdi,%rax,4)+ popcntl %r10d, %r9d+ addl %r9d, %eax+ shrl $0x8, %r8d+ addl $0x4, %ecx+ cmpl $0xf8, %eax+ ja Lmld_rej_uniform_eta4_avx2_asm_scalar+ vpsrldq $0x8, %xmm5, %xmm5 # xmm5 = xmm5[8,9,10,11,12,13,14,15],zero,zero,zero,zero,zero,zero,zero,zero+ movzbl %r8b, %r10d+ vmovq (%rdx,%r10,8), %xmm6+ vpshufb %xmm6, %xmm5, %xmm6+ vpmovsxbd %xmm6, %ymm1+ vmovdqu %ymm1, (%rdi,%rax,4)+ popcntl %r10d, %r9d+ addl %r9d, %eax+ addl $0x4, %ecx+ jmp Lmld_rej_uniform_eta4_avx2_asm_loop++Lmld_rej_uniform_eta4_avx2_asm_scalar:+ cmpl $0x100, %eax # imm = 0x100+ jae Lmld_rej_uniform_eta4_avx2_asm_done+ cmpl $0x110, %ecx # imm = 0x110+ jae Lmld_rej_uniform_eta4_avx2_asm_done+ movzbl (%rsi,%rcx), %r11d+ incl %ecx+ movl %r11d, %r10d+ andl $0xf, %r10d+ cmpl $0x9, %r10d+ jae Lmld_rej_uniform_eta4_avx2_asm_high_nibble+ movl $0x4, %r9d+ subl %r10d, %r9d+ movl %r9d, (%rdi,%rax,4)+ incl %eax+ cmpl $0x100, %eax # imm = 0x100+ jae Lmld_rej_uniform_eta4_avx2_asm_done++Lmld_rej_uniform_eta4_avx2_asm_high_nibble:+ shrl $0x4, %r11d+ andl $0xf, %r11d+ cmpl $0x9, %r11d+ jae Lmld_rej_uniform_eta4_avx2_asm_scalar+ movl $0x4, %r10d+ subl %r11d, %r10d+ movl %r10d, (%rdi,%rax,4)+ incl %eax+ jmp Lmld_rej_uniform_eta4_avx2_asm_scalar++Lmld_rej_uniform_eta4_avx2_asm_done:+ retq+ .cfi_endproc++MLD_ASM_FN_SIZE(rej_uniform_eta4_avx2_asm)++#endif /* MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_MULTILEVEL_NO_SHARED \+ */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
@@ -0,0 +1,161 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*+ * WARNING: This file is auto-generated from scripts/autogen+ * in the mldsa-native repository.+ * Do not modify it directly.+ */++#include "../../../common.h"++#if defined(MLD_ARITH_BACKEND_X86_64_DEFAULT) && \+ !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)++#include "arith_native_x86_64.h"++/*+ * Lookup table used by rejection sampling.+ * See autogen for details.+ */+MLD_ALIGN MLD_INTERNAL_DATA_DEFINITION const uint8_t+ mld_rej_uniform_table[256][8] = {+ {0, 0, 0, 0, 0, 0, 0, 0}, {0, 0, 0, 0, 0, 0, 0, 0},+ {1, 0, 0, 0, 0, 0, 0, 0}, {0, 1, 0, 0, 0, 0, 0, 0},+ {2, 0, 0, 0, 0, 0, 0, 0}, {0, 2, 0, 0, 0, 0, 0, 0},+ {1, 2, 0, 0, 0, 0, 0, 0}, {0, 1, 2, 0, 0, 0, 0, 0},+ {3, 0, 0, 0, 0, 0, 0, 0}, {0, 3, 0, 0, 0, 0, 0, 0},+ {1, 3, 0, 0, 0, 0, 0, 0}, {0, 1, 3, 0, 0, 0, 0, 0},+ {2, 3, 0, 0, 0, 0, 0, 0}, {0, 2, 3, 0, 0, 0, 0, 0},+ {1, 2, 3, 0, 0, 0, 0, 0}, {0, 1, 2, 3, 0, 0, 0, 0},+ {4, 0, 0, 0, 0, 0, 0, 0}, {0, 4, 0, 0, 0, 0, 0, 0},+ {1, 4, 0, 0, 0, 0, 0, 0}, {0, 1, 4, 0, 0, 0, 0, 0},+ {2, 4, 0, 0, 0, 0, 0, 0}, {0, 2, 4, 0, 0, 0, 0, 0},+ {1, 2, 4, 0, 0, 0, 0, 0}, {0, 1, 2, 4, 0, 0, 0, 0},+ {3, 4, 0, 0, 0, 0, 0, 0}, {0, 3, 4, 0, 0, 0, 0, 0},+ {1, 3, 4, 0, 0, 0, 0, 0}, {0, 1, 3, 4, 0, 0, 0, 0},+ {2, 3, 4, 0, 0, 0, 0, 0}, {0, 2, 3, 4, 0, 0, 0, 0},+ {1, 2, 3, 4, 0, 0, 0, 0}, {0, 1, 2, 3, 4, 0, 0, 0},+ {5, 0, 0, 0, 0, 0, 0, 0}, {0, 5, 0, 0, 0, 0, 0, 0},+ {1, 5, 0, 0, 0, 0, 0, 0}, {0, 1, 5, 0, 0, 0, 0, 0},+ {2, 5, 0, 0, 0, 0, 0, 0}, {0, 2, 5, 0, 0, 0, 0, 0},+ {1, 2, 5, 0, 0, 0, 0, 0}, {0, 1, 2, 5, 0, 0, 0, 0},+ {3, 5, 0, 0, 0, 0, 0, 0}, {0, 3, 5, 0, 0, 0, 0, 0},+ {1, 3, 5, 0, 0, 0, 0, 0}, {0, 1, 3, 5, 0, 0, 0, 0},+ {2, 3, 5, 0, 0, 0, 0, 0}, {0, 2, 3, 5, 0, 0, 0, 0},+ {1, 2, 3, 5, 0, 0, 0, 0}, {0, 1, 2, 3, 5, 0, 0, 0},+ {4, 5, 0, 0, 0, 0, 0, 0}, {0, 4, 5, 0, 0, 0, 0, 0},+ {1, 4, 5, 0, 0, 0, 0, 0}, {0, 1, 4, 5, 0, 0, 0, 0},+ {2, 4, 5, 0, 0, 0, 0, 0}, {0, 2, 4, 5, 0, 0, 0, 0},+ {1, 2, 4, 5, 0, 0, 0, 0}, {0, 1, 2, 4, 5, 0, 0, 0},+ {3, 4, 5, 0, 0, 0, 0, 0}, {0, 3, 4, 5, 0, 0, 0, 0},+ {1, 3, 4, 5, 0, 0, 0, 0}, {0, 1, 3, 4, 5, 0, 0, 0},+ {2, 3, 4, 5, 0, 0, 0, 0}, {0, 2, 3, 4, 5, 0, 0, 0},+ {1, 2, 3, 4, 5, 0, 0, 0}, {0, 1, 2, 3, 4, 5, 0, 0},+ {6, 0, 0, 0, 0, 0, 0, 0}, {0, 6, 0, 0, 0, 0, 0, 0},+ {1, 6, 0, 0, 0, 0, 0, 0}, {0, 1, 6, 0, 0, 0, 0, 0},+ {2, 6, 0, 0, 0, 0, 0, 0}, {0, 2, 6, 0, 0, 0, 0, 0},+ {1, 2, 6, 0, 0, 0, 0, 0}, {0, 1, 2, 6, 0, 0, 0, 0},+ {3, 6, 0, 0, 0, 0, 0, 0}, {0, 3, 6, 0, 0, 0, 0, 0},+ {1, 3, 6, 0, 0, 0, 0, 0}, {0, 1, 3, 6, 0, 0, 0, 0},+ {2, 3, 6, 0, 0, 0, 0, 0}, {0, 2, 3, 6, 0, 0, 0, 0},+ {1, 2, 3, 6, 0, 0, 0, 0}, {0, 1, 2, 3, 6, 0, 0, 0},+ {4, 6, 0, 0, 0, 0, 0, 0}, {0, 4, 6, 0, 0, 0, 0, 0},+ {1, 4, 6, 0, 0, 0, 0, 0}, {0, 1, 4, 6, 0, 0, 0, 0},+ {2, 4, 6, 0, 0, 0, 0, 0}, {0, 2, 4, 6, 0, 0, 0, 0},+ {1, 2, 4, 6, 0, 0, 0, 0}, {0, 1, 2, 4, 6, 0, 0, 0},+ {3, 4, 6, 0, 0, 0, 0, 0}, {0, 3, 4, 6, 0, 0, 0, 0},+ {1, 3, 4, 6, 0, 0, 0, 0}, {0, 1, 3, 4, 6, 0, 0, 0},+ {2, 3, 4, 6, 0, 0, 0, 0}, {0, 2, 3, 4, 6, 0, 0, 0},+ {1, 2, 3, 4, 6, 0, 0, 0}, {0, 1, 2, 3, 4, 6, 0, 0},+ {5, 6, 0, 0, 0, 0, 0, 0}, {0, 5, 6, 0, 0, 0, 0, 0},+ {1, 5, 6, 0, 0, 0, 0, 0}, {0, 1, 5, 6, 0, 0, 0, 0},+ {2, 5, 6, 0, 0, 0, 0, 0}, {0, 2, 5, 6, 0, 0, 0, 0},+ {1, 2, 5, 6, 0, 0, 0, 0}, {0, 1, 2, 5, 6, 0, 0, 0},+ {3, 5, 6, 0, 0, 0, 0, 0}, {0, 3, 5, 6, 0, 0, 0, 0},+ {1, 3, 5, 6, 0, 0, 0, 0}, {0, 1, 3, 5, 6, 0, 0, 0},+ {2, 3, 5, 6, 0, 0, 0, 0}, {0, 2, 3, 5, 6, 0, 0, 0},+ {1, 2, 3, 5, 6, 0, 0, 0}, {0, 1, 2, 3, 5, 6, 0, 0},+ {4, 5, 6, 0, 0, 0, 0, 0}, {0, 4, 5, 6, 0, 0, 0, 0},+ {1, 4, 5, 6, 0, 0, 0, 0}, {0, 1, 4, 5, 6, 0, 0, 0},+ {2, 4, 5, 6, 0, 0, 0, 0}, {0, 2, 4, 5, 6, 0, 0, 0},+ {1, 2, 4, 5, 6, 0, 0, 0}, {0, 1, 2, 4, 5, 6, 0, 0},+ {3, 4, 5, 6, 0, 0, 0, 0}, {0, 3, 4, 5, 6, 0, 0, 0},+ {1, 3, 4, 5, 6, 0, 0, 0}, {0, 1, 3, 4, 5, 6, 0, 0},+ {2, 3, 4, 5, 6, 0, 0, 0}, {0, 2, 3, 4, 5, 6, 0, 0},+ {1, 2, 3, 4, 5, 6, 0, 0}, {0, 1, 2, 3, 4, 5, 6, 0},+ {7, 0, 0, 0, 0, 0, 0, 0}, {0, 7, 0, 0, 0, 0, 0, 0},+ {1, 7, 0, 0, 0, 0, 0, 0}, {0, 1, 7, 0, 0, 0, 0, 0},+ {2, 7, 0, 0, 0, 0, 0, 0}, {0, 2, 7, 0, 0, 0, 0, 0},+ {1, 2, 7, 0, 0, 0, 0, 0}, {0, 1, 2, 7, 0, 0, 0, 0},+ {3, 7, 0, 0, 0, 0, 0, 0}, {0, 3, 7, 0, 0, 0, 0, 0},+ {1, 3, 7, 0, 0, 0, 0, 0}, {0, 1, 3, 7, 0, 0, 0, 0},+ {2, 3, 7, 0, 0, 0, 0, 0}, {0, 2, 3, 7, 0, 0, 0, 0},+ {1, 2, 3, 7, 0, 0, 0, 0}, {0, 1, 2, 3, 7, 0, 0, 0},+ {4, 7, 0, 0, 0, 0, 0, 0}, {0, 4, 7, 0, 0, 0, 0, 0},+ {1, 4, 7, 0, 0, 0, 0, 0}, {0, 1, 4, 7, 0, 0, 0, 0},+ {2, 4, 7, 0, 0, 0, 0, 0}, {0, 2, 4, 7, 0, 0, 0, 0},+ {1, 2, 4, 7, 0, 0, 0, 0}, {0, 1, 2, 4, 7, 0, 0, 0},+ {3, 4, 7, 0, 0, 0, 0, 0}, {0, 3, 4, 7, 0, 0, 0, 0},+ {1, 3, 4, 7, 0, 0, 0, 0}, {0, 1, 3, 4, 7, 0, 0, 0},+ {2, 3, 4, 7, 0, 0, 0, 0}, {0, 2, 3, 4, 7, 0, 0, 0},+ {1, 2, 3, 4, 7, 0, 0, 0}, {0, 1, 2, 3, 4, 7, 0, 0},+ {5, 7, 0, 0, 0, 0, 0, 0}, {0, 5, 7, 0, 0, 0, 0, 0},+ {1, 5, 7, 0, 0, 0, 0, 0}, {0, 1, 5, 7, 0, 0, 0, 0},+ {2, 5, 7, 0, 0, 0, 0, 0}, {0, 2, 5, 7, 0, 0, 0, 0},+ {1, 2, 5, 7, 0, 0, 0, 0}, {0, 1, 2, 5, 7, 0, 0, 0},+ {3, 5, 7, 0, 0, 0, 0, 0}, {0, 3, 5, 7, 0, 0, 0, 0},+ {1, 3, 5, 7, 0, 0, 0, 0}, {0, 1, 3, 5, 7, 0, 0, 0},+ {2, 3, 5, 7, 0, 0, 0, 0}, {0, 2, 3, 5, 7, 0, 0, 0},+ {1, 2, 3, 5, 7, 0, 0, 0}, {0, 1, 2, 3, 5, 7, 0, 0},+ {4, 5, 7, 0, 0, 0, 0, 0}, {0, 4, 5, 7, 0, 0, 0, 0},+ {1, 4, 5, 7, 0, 0, 0, 0}, {0, 1, 4, 5, 7, 0, 0, 0},+ {2, 4, 5, 7, 0, 0, 0, 0}, {0, 2, 4, 5, 7, 0, 0, 0},+ {1, 2, 4, 5, 7, 0, 0, 0}, {0, 1, 2, 4, 5, 7, 0, 0},+ {3, 4, 5, 7, 0, 0, 0, 0}, {0, 3, 4, 5, 7, 0, 0, 0},+ {1, 3, 4, 5, 7, 0, 0, 0}, {0, 1, 3, 4, 5, 7, 0, 0},+ {2, 3, 4, 5, 7, 0, 0, 0}, {0, 2, 3, 4, 5, 7, 0, 0},+ {1, 2, 3, 4, 5, 7, 0, 0}, {0, 1, 2, 3, 4, 5, 7, 0},+ {6, 7, 0, 0, 0, 0, 0, 0}, {0, 6, 7, 0, 0, 0, 0, 0},+ {1, 6, 7, 0, 0, 0, 0, 0}, {0, 1, 6, 7, 0, 0, 0, 0},+ {2, 6, 7, 0, 0, 0, 0, 0}, {0, 2, 6, 7, 0, 0, 0, 0},+ {1, 2, 6, 7, 0, 0, 0, 0}, {0, 1, 2, 6, 7, 0, 0, 0},+ {3, 6, 7, 0, 0, 0, 0, 0}, {0, 3, 6, 7, 0, 0, 0, 0},+ {1, 3, 6, 7, 0, 0, 0, 0}, {0, 1, 3, 6, 7, 0, 0, 0},+ {2, 3, 6, 7, 0, 0, 0, 0}, {0, 2, 3, 6, 7, 0, 0, 0},+ {1, 2, 3, 6, 7, 0, 0, 0}, {0, 1, 2, 3, 6, 7, 0, 0},+ {4, 6, 7, 0, 0, 0, 0, 0}, {0, 4, 6, 7, 0, 0, 0, 0},+ {1, 4, 6, 7, 0, 0, 0, 0}, {0, 1, 4, 6, 7, 0, 0, 0},+ {2, 4, 6, 7, 0, 0, 0, 0}, {0, 2, 4, 6, 7, 0, 0, 0},+ {1, 2, 4, 6, 7, 0, 0, 0}, {0, 1, 2, 4, 6, 7, 0, 0},+ {3, 4, 6, 7, 0, 0, 0, 0}, {0, 3, 4, 6, 7, 0, 0, 0},+ {1, 3, 4, 6, 7, 0, 0, 0}, {0, 1, 3, 4, 6, 7, 0, 0},+ {2, 3, 4, 6, 7, 0, 0, 0}, {0, 2, 3, 4, 6, 7, 0, 0},+ {1, 2, 3, 4, 6, 7, 0, 0}, {0, 1, 2, 3, 4, 6, 7, 0},+ {5, 6, 7, 0, 0, 0, 0, 0}, {0, 5, 6, 7, 0, 0, 0, 0},+ {1, 5, 6, 7, 0, 0, 0, 0}, {0, 1, 5, 6, 7, 0, 0, 0},+ {2, 5, 6, 7, 0, 0, 0, 0}, {0, 2, 5, 6, 7, 0, 0, 0},+ {1, 2, 5, 6, 7, 0, 0, 0}, {0, 1, 2, 5, 6, 7, 0, 0},+ {3, 5, 6, 7, 0, 0, 0, 0}, {0, 3, 5, 6, 7, 0, 0, 0},+ {1, 3, 5, 6, 7, 0, 0, 0}, {0, 1, 3, 5, 6, 7, 0, 0},+ {2, 3, 5, 6, 7, 0, 0, 0}, {0, 2, 3, 5, 6, 7, 0, 0},+ {1, 2, 3, 5, 6, 7, 0, 0}, {0, 1, 2, 3, 5, 6, 7, 0},+ {4, 5, 6, 7, 0, 0, 0, 0}, {0, 4, 5, 6, 7, 0, 0, 0},+ {1, 4, 5, 6, 7, 0, 0, 0}, {0, 1, 4, 5, 6, 7, 0, 0},+ {2, 4, 5, 6, 7, 0, 0, 0}, {0, 2, 4, 5, 6, 7, 0, 0},+ {1, 2, 4, 5, 6, 7, 0, 0}, {0, 1, 2, 4, 5, 6, 7, 0},+ {3, 4, 5, 6, 7, 0, 0, 0}, {0, 3, 4, 5, 6, 7, 0, 0},+ {1, 3, 4, 5, 6, 7, 0, 0}, {0, 1, 3, 4, 5, 6, 7, 0},+ {2, 3, 4, 5, 6, 7, 0, 0}, {0, 2, 3, 4, 5, 6, 7, 0},+ {1, 2, 3, 4, 5, 6, 7, 0}, {0, 1, 2, 3, 4, 5, 6, 7},+};++#else /* MLD_ARITH_BACKEND_X86_64_DEFAULT && !MLD_CONFIG_MULTILEVEL_NO_SHARED \+ */++MLD_EMPTY_CU(avx2_rej_uniform_table)++#endif /* !(MLD_ARITH_BACKEND_X86_64_DEFAULT && \+ !MLD_CONFIG_MULTILEVEL_NO_SHARED) */
@@ -0,0 +1,213 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+#include <string.h>++#include "common.h"+#include "packing.h"+#include "poly.h"+#include "polyvec.h"+#include "rounding.h"++/* Parameter set namespacing+ * This is to facilitate building multiple instances+ * of mldsa-native (e.g. with varying parameter sets)+ * within a single compilation unit. */+/* End of parameter set namespacing */++#if !defined(MLD_CONFIG_NO_VERIFY_API)+MLD_INTERNAL_API+void mld_unpack_pk_t1(mld_poly *t1,+ const uint8_t pk[MLDSA_CRYPTO_PUBLICKEYBYTES],+ unsigned int i)+{+ mld_polyt1_unpack(t1, pk + MLDSA_PK_T1_OFFSET + i * MLDSA_POLYT1_PACKEDBYTES);+}+#endif /* !MLD_CONFIG_NO_VERIFY_API */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API)+MLD_INTERNAL_API+void mld_pack_sk_s1(uint8_t sk[MLDSA_CRYPTO_SECRETKEYBYTES],+ const mld_polyvecl *s1)+{+ mld_polyvecl_pack_eta(sk + MLDSA_SK_S1_OFFSET, s1);+}++MLD_INTERNAL_API+void mld_pack_sk_rho_key_tr_s2(uint8_t sk[MLDSA_CRYPTO_SECRETKEYBYTES],+ const uint8_t rho[MLDSA_SEEDBYTES],+ const uint8_t tr[MLDSA_TRBYTES],+ const uint8_t key[MLDSA_SEEDBYTES],+ const mld_polyveck *s2)+{+ mld_memcpy(sk + MLDSA_SK_RHO_OFFSET, rho, MLDSA_SEEDBYTES);+ mld_memcpy(sk + MLDSA_SK_KEY_OFFSET, key, MLDSA_SEEDBYTES);+ mld_memcpy(sk + MLDSA_SK_TR_OFFSET, tr, MLDSA_TRBYTES);+ /* s1 already packed via mld_pack_sk_s1 */+ mld_polyveck_pack_eta(sk + MLDSA_SK_S2_OFFSET, s2);+ /* t0 already packed via mld_compute_pack_t0_t1 */+}+#endif /* !MLD_CONFIG_NO_KEYPAIR_API */++#if !defined(MLD_CONFIG_NO_SIGN_API)+MLD_INTERNAL_API+void mld_unpack_sk(uint8_t rho[MLDSA_SEEDBYTES], uint8_t tr[MLDSA_TRBYTES],+ uint8_t key[MLDSA_SEEDBYTES], mld_sk_t0hat *t0,+ mld_sk_s1hat *s1, mld_sk_s2hat *s2,+ const uint8_t sk[MLDSA_CRYPTO_SECRETKEYBYTES])+{+ mld_memcpy(rho, sk + MLDSA_SK_RHO_OFFSET, MLDSA_SEEDBYTES);+ mld_memcpy(key, sk + MLDSA_SK_KEY_OFFSET, MLDSA_SEEDBYTES);+ mld_memcpy(tr, sk + MLDSA_SK_TR_OFFSET, MLDSA_TRBYTES);+ mld_unpack_sk_s1hat(s1, sk + MLDSA_SK_S1_OFFSET);+ mld_unpack_sk_s2hat(s2, sk + MLDSA_SK_S2_OFFSET);+ mld_unpack_sk_t0hat(t0, sk + MLDSA_SK_T0_OFFSET);+}++MLD_INTERNAL_API+void mld_pack_sig_c(uint8_t sig[MLDSA_CRYPTO_BYTES],+ const uint8_t c[MLDSA_CTILDEBYTES])+{+ mld_memcpy(sig, c, MLDSA_CTILDEBYTES);+}++MLD_INTERNAL_API+int mld_pack_sig_h(uint8_t sig[MLDSA_CRYPTO_BYTES], const mld_polyveck *w0,+ const mld_polyveck *w1)+{+ unsigned int j, k, n;++ /* The hint section of sig[] is MLDSA_POLYVECH_PACKEDBYTES long, where+ * MLDSA_POLYVECH_PACKEDBYTES = MLDSA_OMEGA + MLDSA_K.+ *+ * The first OMEGA bytes record the index numbers of the coefficients+ * that are not equal to 0.+ *+ * The final K bytes record a running tally of the number of hints+ * coming from each of the K polynomials. */+ uint8_t *sig_h = sig + MLDSA_SIG_H_OFFSET;++ mld_memset(sig_h, 0, MLDSA_POLYVECH_PACKEDBYTES);+ n = 0;++ /* For each coefficient of each polynomial, compute its hint bit and, if+ * non-zero, record the index in the hint section of sig. If recording the+ * hint would overflow the OMEGA-sized index array, abort early and return+ * MLD_ERR_FAIL. The caller is expected to reject the signature in that case.+ *+ * Constant time: At this point w0/w1 are public (see comment in sign.c+ * before the call), so a data-dependent early return is fine. */+ for (k = 0; k < MLDSA_K; k++)+ __loop__(+ assigns(k, j, n, memory_slice(sig_h, MLDSA_POLYVECH_PACKEDBYTES))+ invariant(k <= MLDSA_K && n <= MLDSA_OMEGA)+ decreases(MLDSA_K - k)+ )+ {+ for (j = 0; j < MLDSA_N; j++)+ __loop__(+ assigns(j, n, memory_slice(sig_h, MLDSA_POLYVECH_PACKEDBYTES))+ invariant(j <= MLDSA_N && n <= MLDSA_OMEGA)+ decreases(MLDSA_N - j)+ )+ {+ const unsigned int hint_bit =+ mld_make_hint(w0->vec[k].coeffs[j], w1->vec[k].coeffs[j]);+ if (hint_bit)+ {+ if (n == MLDSA_OMEGA)+ {+ return MLD_ERR_FAIL;+ }+ /* Safety: branch above ensures n < MLDSA_OMEGA so n is a valid index+ * into the OMEGA-sized index array; j < MLDSA_N <= 256 fits in+ * uint8_t. */+ sig_h[n] = (uint8_t)j;+ n++;+ }+ }+ /* Record the running tally into the correct slot for this polynomial.+ * Safety: k < MLDSA_K, so MLDSA_OMEGA + k is a valid index into the+ * K-byte tally tail; n <= MLDSA_OMEGA fits in uint8_t. */+ sig_h[MLDSA_OMEGA + k] = (uint8_t)n;+ }+ return 0;+}++MLD_INTERNAL_API+void mld_pack_sig_z(uint8_t sig[MLDSA_CRYPTO_BYTES], const mld_poly *zi,+ unsigned i)+{+ mld_polyz_pack(sig + MLDSA_SIG_Z_OFFSET + i * MLDSA_POLYZ_PACKEDBYTES, zi);+}+#endif /* !MLD_CONFIG_NO_SIGN_API */++#if !defined(MLD_CONFIG_NO_VERIFY_API)+MLD_INTERNAL_API+int mld_sig_unpack_hints(mld_poly *h, const uint8_t sig[MLDSA_CRYPTO_BYTES],+ unsigned int i)+{+ const uint8_t *packed_hints = sig + MLDSA_SIG_H_OFFSET;+ const unsigned int old_hint_count =+ (i == 0) ? 0 : packed_hints[MLDSA_OMEGA + i - 1];+ const unsigned int new_hint_count = packed_hints[MLDSA_OMEGA + i];+ unsigned int j;++ if (new_hint_count < old_hint_count || new_hint_count > MLDSA_OMEGA)+ {+ return MLD_ERR_FAIL;+ }++ mld_memset(h, 0, sizeof(mld_poly));++ for (j = old_hint_count; j < new_hint_count; ++j)+ __loop__(+ invariant(j >= old_hint_count && j <= new_hint_count &&+ new_hint_count <= MLDSA_OMEGA)+ invariant(array_bound(h->coeffs, 0, MLDSA_N, 0, 2))+ invariant(forall(p, 0, MLDSA_N,+ (h->coeffs[p] == 1) ==+ exists(hj, old_hint_count, j, packed_hints[hj] == p)))+ decreases(new_hint_count - j)+ )+ {+ if (j > old_hint_count && packed_hints[j] <= packed_hints[j - 1])+ {+ return MLD_ERR_FAIL;+ }+ /* Safety: packed_hints[j] is uint8_t (<= 255) and MLDSA_N == 256. */+ h->coeffs[packed_hints[j]] = 1;+ }++ /* On the last row, also verify that the trailing index slots are zero. */+ if (i == MLDSA_K - 1)+ {+ for (j = new_hint_count; j < MLDSA_OMEGA; ++j)+ __loop__(+ invariant(j <= MLDSA_OMEGA)+ decreases(MLDSA_OMEGA - j)+ )+ {+ if (packed_hints[j] != 0)+ {+ return MLD_ERR_FAIL;+ }+ }+ }++ /* On success, h->coeffs[p] is 1 exactly for the hint indices decoded for this+ * row, i.e. packed_hints[old_hint_count, new_hint_count). Asserted here+ * rather than posted as a contract to not unnecessarily increase proof+ * complexity at callers that do not need the functional description. */+ cassert(forall(+ p, 0, MLDSA_N,+ (h->coeffs[p] == 1) ==+ exists(hj, old_hint_count, new_hint_count, packed_hints[hj] == p)));++ return 0;+}+#endif /* !MLD_CONFIG_NO_VERIFY_API */++/* To facilitate single-compilation-unit (SCU) builds, undefine all macros.+ * Don't modify by hand -- this is auto-generated by scripts/autogen. */
@@ -0,0 +1,277 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS204]+ * FIPS 204 Module-Lattice-Based Digital Signature Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/204/final+ */++#ifndef MLD_PACKING_H+#define MLD_PACKING_H++#include "polyvec.h"+#include "polyvec_lazy.h"++#if !defined(MLD_CONFIG_NO_KEYPAIR_API)+#define mld_pack_sk_s1 MLD_NAMESPACE_KL(pack_sk_s1)+/**+ * Bit-pack the s1 component into the secret key.+ *+ * @spec{Partially implements @[FIPS204, Algorithm 24, skEncode] (s1+ * component).}+ *+ * @param[out] sk Output byte array.+ * @param[in] s1 Pointer to vector s1.+ */+MLD_INTERNAL_API+void mld_pack_sk_s1(uint8_t sk[MLDSA_CRYPTO_SECRETKEYBYTES],+ const mld_polyvecl *s1)+__contract__(+ requires(memory_no_alias(sk, MLDSA_CRYPTO_SECRETKEYBYTES))+ requires(memory_no_alias(s1, sizeof(mld_polyvecl)))+ requires(forall(k1, 0, MLDSA_L,+ array_abs_bound(s1->vec[k1].coeffs, 0, MLDSA_N, MLDSA_ETA + 1)))+ assigns(memory_slice(sk, MLDSA_CRYPTO_SECRETKEYBYTES))+);++#define mld_pack_sk_rho_key_tr_s2 MLD_NAMESPACE_KL(pack_sk_rho_key_tr_s2)+/**+ * Bit-pack rho, key, tr, s2 into the secret key.+ *+ * s1 must already be packed via mld_pack_sk_s1, and t0 via+ * mld_compute_pack_t0_t1.+ *+ * @spec{Partially implements @[FIPS204, Algorithm 24, skEncode] (rho, key, tr,+ * s2 components).}+ *+ * @param[out] sk Output byte array.+ * @param[in] rho Byte array containing rho.+ * @param[in] tr Byte array containing tr.+ * @param[in] key Byte array containing key.+ * @param[in] s2 Pointer to vector s2.+ */+MLD_INTERNAL_API+void mld_pack_sk_rho_key_tr_s2(uint8_t sk[MLDSA_CRYPTO_SECRETKEYBYTES],+ const uint8_t rho[MLDSA_SEEDBYTES],+ const uint8_t tr[MLDSA_TRBYTES],+ const uint8_t key[MLDSA_SEEDBYTES],+ const mld_polyveck *s2)+__contract__(+ requires(memory_no_alias(sk, MLDSA_CRYPTO_SECRETKEYBYTES))+ requires(memory_no_alias(rho, MLDSA_SEEDBYTES))+ requires(memory_no_alias(tr, MLDSA_TRBYTES))+ requires(memory_no_alias(key, MLDSA_SEEDBYTES))+ requires(memory_no_alias(s2, sizeof(mld_polyveck)))+ requires(forall(k2, 0, MLDSA_K,+ array_abs_bound(s2->vec[k2].coeffs, 0, MLDSA_N, MLDSA_ETA + 1)))+ assigns(memory_slice(sk, MLDSA_CRYPTO_SECRETKEYBYTES))+);+#endif /* !MLD_CONFIG_NO_KEYPAIR_API */+++#if !defined(MLD_CONFIG_NO_SIGN_API)+#define mld_pack_sig_c MLD_NAMESPACE_KL(pack_sig_c)+/**+ * Bit-pack challenge c into sig = (c, z, h).+ *+ * @spec{Partially implements @[FIPS204, Algorithm 26, sigEncode] (c+ * component).}+ *+ * @param[out] sig Output byte array.+ * @param[in] c Pointer to challenge hash.+ */+MLD_INTERNAL_API+void mld_pack_sig_c(uint8_t sig[MLDSA_CRYPTO_BYTES],+ const uint8_t c[MLDSA_CTILDEBYTES])+__contract__(+ requires(memory_no_alias(sig, MLDSA_CRYPTO_BYTES))+ requires(memory_no_alias(c, MLDSA_CTILDEBYTES))+ assigns(memory_slice(sig, MLDSA_CRYPTO_BYTES))+);++#define mld_pack_sig_h MLD_NAMESPACE_KL(pack_sig_h)+/**+ * Compute hints from (w0, w1) and pack them into the hint section of sig.+ *+ * @spec{Combines the hint computation @[FIPS204, Algorithm 39, MakeHint] with+ * the packing @[FIPS204, Algorithm 20, HintBitPack] (the h component of+ * @[FIPS204, Algorithm 26, sigEncode]): it computes the hint vector h from+ * (w0, w1) and packs it, rather than receiving a ready-made h as HintBitPack+ * does. The hints are computed via mld_make_hint (rounding.h), a specialized+ * MakeHint valid only for the values arising during signing; see the block+ * comment in mld_attempt_signature_generation (sign.c).}+ *+ * @param[in,out] sig Byte array containing signature.+ * @param[in] w0 Pointer to low part of input vector.+ * @param[in] w1 Pointer to high part of input vector.+ *+ * @retval 0 Success.+ * @retval MLD_ERR_FAIL The total number of hints exceeds MLDSA_OMEGA. In this+ * case the hint section of sig is left in a+ * partially-written state and the caller must reject the+ * signature.+ */+MLD_INTERNAL_API+MLD_MUST_CHECK_RETURN_VALUE+int mld_pack_sig_h(uint8_t sig[MLDSA_CRYPTO_BYTES], const mld_polyveck *w0,+ const mld_polyveck *w1)+__contract__(+ requires(memory_no_alias(sig, MLDSA_CRYPTO_BYTES))+ requires(memory_no_alias(w0, sizeof(mld_polyveck)))+ requires(memory_no_alias(w1, sizeof(mld_polyveck)))+ assigns(memory_slice(sig + MLDSA_SIG_H_OFFSET, MLDSA_POLYVECH_PACKEDBYTES))+ ensures(return_value == 0 || return_value == MLD_ERR_FAIL)+);++#define mld_pack_sig_z MLD_NAMESPACE_KL(pack_sig_z)+/**+ * Bit-pack single polynomial of z component of sig = (c, z, h).+ *+ * The c and h components are packed separately using mld_pack_sig_c and+ * mld_pack_sig_h.+ *+ * @spec{Partially implements @[FIPS204, Algorithm 26, sigEncode] (one+ * polynomial of the z component).}+ *+ * @param[in,out] sig Output byte array.+ * @param[in] zi Pointer to a single polynomial in z.+ * @param i Index of zi in vector z.+ */+MLD_INTERNAL_API+void mld_pack_sig_z(uint8_t sig[MLDSA_CRYPTO_BYTES], const mld_poly *zi,+ unsigned i)+__contract__(+ requires(memory_no_alias(sig, MLDSA_CRYPTO_BYTES))+ requires(memory_no_alias(zi, sizeof(mld_poly)))+ requires(i < MLDSA_L)+ requires(array_bound(zi->coeffs, 0, MLDSA_N, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1))+ assigns(memory_slice(sig, MLDSA_CRYPTO_BYTES))+);+#endif /* !MLD_CONFIG_NO_SIGN_API */++#if !defined(MLD_CONFIG_NO_VERIFY_API)+#define mld_unpack_pk_t1 MLD_NAMESPACE_KL(unpack_pk_t1)+/**+ * Unpack a single polynomial of the t1 component of a public key+ * pk = (rho, t1).+ *+ * @spec{Partially implements @[FIPS204, Algorithm 23, pkDecode] (one polynomial+ * of t1).}+ *+ * @param[out] t1 Pointer to output polynomial t1[i].+ * @param[in] pk Byte array containing bit-packed pk.+ * @param i Row index, must be < MLDSA_K.+ */+MLD_INTERNAL_API+void mld_unpack_pk_t1(mld_poly *t1,+ const uint8_t pk[MLDSA_CRYPTO_PUBLICKEYBYTES],+ unsigned int i)+__contract__(+ requires(memory_no_alias(pk, MLDSA_CRYPTO_PUBLICKEYBYTES))+ requires(memory_no_alias(t1, sizeof(mld_poly)))+ requires(i < MLDSA_K)+ assigns(memory_slice(t1, sizeof(mld_poly)))+ ensures(array_bound(t1->coeffs, 0, MLDSA_N, 0, 1 << 10))+);+#endif /* !MLD_CONFIG_NO_VERIFY_API */++#if !defined(MLD_CONFIG_NO_SIGN_API)+#define mld_unpack_sk MLD_NAMESPACE_KL(unpack_sk)+/**+ * Unpack secret key sk = (rho, tr, key, t0, s1, s2).+ *+ * NOTE: In REDUCE_RAM mode, s1/s2/t0 borrow from sk rather than copying.+ *+ * @spec{Implements @[FIPS204, Algorithm 25, skDecode].}+ *+ * @param[out] rho Output byte array for rho.+ * @param[out] tr Output byte array for tr.+ * @param[out] key Output byte array for key.+ * @param[out] t0 Pointer to output vector t0.+ * @param[out] s1 Pointer to output vector s1.+ * @param[out] s2 Pointer to output vector s2.+ * @param[in] sk Byte array containing bit-packed sk.+ */+MLD_INTERNAL_API+void mld_unpack_sk(uint8_t rho[MLDSA_SEEDBYTES], uint8_t tr[MLDSA_TRBYTES],+ uint8_t key[MLDSA_SEEDBYTES], mld_sk_t0hat *t0,+ mld_sk_s1hat *s1, mld_sk_s2hat *s2,+ const uint8_t sk[MLDSA_CRYPTO_SECRETKEYBYTES])+__contract__(+ requires(memory_no_alias(rho, MLDSA_SEEDBYTES))+ requires(memory_no_alias(tr, MLDSA_TRBYTES))+ requires(memory_no_alias(key, MLDSA_SEEDBYTES))+ requires(memory_no_alias(t0, sizeof(mld_sk_t0hat)))+ requires(memory_no_alias(s1, sizeof(mld_sk_s1hat)))+ requires(memory_no_alias(s2, sizeof(mld_sk_s2hat)))+ requires(memory_no_alias(sk, MLDSA_CRYPTO_SECRETKEYBYTES))+ assigns(memory_slice(rho, MLDSA_SEEDBYTES))+ assigns(memory_slice(tr, MLDSA_TRBYTES))+ assigns(memory_slice(key, MLDSA_SEEDBYTES))+ assigns(memory_slice(t0, sizeof(mld_sk_t0hat)))+ assigns(memory_slice(s1, sizeof(mld_sk_s1hat)))+ assigns(memory_slice(s2, sizeof(mld_sk_s2hat)))+ MLD_IF_NOT_REDUCE_RAM(+ ensures(forall(k0, 0, MLDSA_K,+ array_abs_bound(t0->vec.vec[k0].coeffs, 0, MLDSA_N, MLD_NTT_BOUND)))+ ensures(forall(k1, 0, MLDSA_L,+ array_abs_bound(s1->vec.vec[k1].coeffs, 0, MLDSA_N, MLD_NTT_BOUND)))+ ensures(forall(k2, 0, MLDSA_K,+ array_abs_bound(s2->vec.vec[k2].coeffs, 0, MLDSA_N, MLD_NTT_BOUND)))+ )+ MLD_IF_REDUCE_RAM(+ ensures(s1->packed == old(sk) + MLDSA_SK_S1_OFFSET)+ ensures(s2->packed == old(sk) + MLDSA_SK_S2_OFFSET)+ ensures(t0->packed == old(sk) + MLDSA_SK_T0_OFFSET)+ )+);+#endif /* !MLD_CONFIG_NO_SIGN_API */++#if !defined(MLD_CONFIG_NO_VERIFY_API)+#define mld_sig_unpack_hints MLD_NAMESPACE_KL(sig_unpack_hints)+/**+ * Decode and validate a single row of the hint vector h from a signature+ * buffer.+ *+ * The hint encoding is shared across all rows (a count array followed by a+ * single index list), so this function performs the validation relevant to+ * row i:+ * - the i'th hint count is non-decreasing and bounded by MLDSA_OMEGA;+ * - the indices for row i are strictly ascending;+ * - on i == MLDSA_K - 1, the trailing index slots are zero.+ *+ * Callers must invoke this for every i in [0, 1, .., MLDSA_K - 1]; if any+ * call returns MLD_ERR_FAIL the encoding is malformed and the signature must+ * be rejected.+ *+ * @spec{Implements @[FIPS204, Algorithm 21, HintBitUnpack] (one row; part of+ * @[FIPS204, Algorithm 27, sigDecode]).}+ *+ * @param[out] h Pointer to output polynomial h[i].+ * @param[in] sig Signature buffer.+ * @param i Row index, must be < MLDSA_K.+ *+ * @retval 0 Hints were decoded successfully.+ * @retval MLD_ERR_FAIL Hints are malformed.+ */+MLD_INTERNAL_API+MLD_MUST_CHECK_RETURN_VALUE+int mld_sig_unpack_hints(mld_poly *h, const uint8_t sig[MLDSA_CRYPTO_BYTES],+ unsigned int i)+__contract__(+ requires(memory_no_alias(sig, MLDSA_CRYPTO_BYTES))+ requires(memory_no_alias(h, sizeof(mld_poly)))+ requires(i < MLDSA_K)+ assigns(memory_slice(h, sizeof(mld_poly)))+ ensures(return_value == 0 || return_value == MLD_ERR_FAIL)+ ensures(return_value == 0 ==> array_bound(h->coeffs, 0, MLDSA_N, 0, 2))+);+#endif /* !MLD_CONFIG_NO_VERIFY_API */++#endif /* !MLD_PACKING_H */
@@ -0,0 +1,153 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+#ifndef MLD_PARAMS_H+#define MLD_PARAMS_H++#define MLDSA_SEEDBYTES 32+#define MLDSA_CRHBYTES 64+#define MLDSA_TRBYTES 64+#define MLDSA_RNDBYTES 32+#define MLDSA_N 256+#define MLDSA_Q 8380417+#define MLDSA_Q_HALF ((MLDSA_Q + 1) / 2)+#define MLDSA_D 13++#define MLDSA_GAMMA2_88 ((MLDSA_Q - 1) / 88)+#define MLDSA_GAMMA2_32 ((MLDSA_Q - 1) / 32)+#define MLDSA_POLYW1_PACKEDBYTES_88 192+#define MLDSA_POLYW1_PACKEDBYTES_32 128++#if MLD_CONFIG_PARAMETER_SET == 44++#define MLDSA_K 4+#define MLDSA_L 4+#define MLDSA_ETA 2+#define MLDSA_TAU 39+#define MLDSA_BETA 78+#define MLDSA_GAMMA1 ((int32_t)1 << 17)+#define MLDSA_GAMMA2 MLDSA_GAMMA2_88+#define MLDSA_OMEGA 80+#define MLDSA_CTILDEBYTES 32+#define MLDSA_POLYZ_PACKEDBYTES 576+#define MLDSA_POLYW1_PACKEDBYTES MLDSA_POLYW1_PACKEDBYTES_88+#define MLDSA_POLYETA_PACKEDBYTES 96++#elif MLD_CONFIG_PARAMETER_SET == 65++#define MLDSA_K 6+#define MLDSA_L 5+#define MLDSA_ETA 4+#define MLDSA_TAU 49+#define MLDSA_BETA 196+#define MLDSA_GAMMA1 ((int32_t)1 << 19)+#define MLDSA_GAMMA2 MLDSA_GAMMA2_32+#define MLDSA_OMEGA 55+#define MLDSA_CTILDEBYTES 48+#define MLDSA_POLYZ_PACKEDBYTES 640+#define MLDSA_POLYW1_PACKEDBYTES MLDSA_POLYW1_PACKEDBYTES_32+#define MLDSA_POLYETA_PACKEDBYTES 128++#elif MLD_CONFIG_PARAMETER_SET == 87++#define MLDSA_K 8+#define MLDSA_L 7+#define MLDSA_ETA 2+#define MLDSA_TAU 60+#define MLDSA_BETA 120+#define MLDSA_GAMMA1 ((int32_t)1 << 19)+#define MLDSA_GAMMA2 MLDSA_GAMMA2_32+#define MLDSA_OMEGA 75+#define MLDSA_CTILDEBYTES 64+#define MLDSA_POLYZ_PACKEDBYTES 640+#define MLDSA_POLYW1_PACKEDBYTES MLDSA_POLYW1_PACKEDBYTES_32+#define MLDSA_POLYETA_PACKEDBYTES 96++#endif /* MLD_CONFIG_PARAMETER_SET == 87 */++#define MLDSA_POLYT1_PACKEDBYTES 320+#define MLDSA_POLYT0_PACKEDBYTES 416+#define MLDSA_POLYVECH_PACKEDBYTES (MLDSA_OMEGA + MLDSA_K)++/* Sampling y from counter kappa uses nonces kappa, ..., kappa+L-1, which fit in+ * uint16_t iff kappa <= UINT16_MAX - MLDSA_L. With kappa = attempt*MLDSA_L this+ * bounds the number of signing attempts by MLD_MAX_KAPPA / MLDSA_L; see+ * MLD_MAX_SIGNING_ATTEMPTS in sign.c. */+#define MLD_MAX_KAPPA (UINT16_MAX - MLDSA_L)++/* Layout of the packed public key pk[MLDSA_CRYPTO_PUBLICKEYBYTES] = (rho, t1):+ *+ * +-------------+--------------------------++ * | rho | t1 |+ * +-------------+--------------------------++ * | SEEDBYTES | K * POLYT1_PACKEDBYTES |+ * +-------------+--------------------------++ */+#define MLDSA_PK_RHO_OFFSET 0+#define MLDSA_PK_RHO_BYTES MLDSA_SEEDBYTES++#define MLDSA_PK_T1_OFFSET (MLDSA_PK_RHO_OFFSET + MLDSA_PK_RHO_BYTES)+#define MLDSA_PK_T1_BYTES (MLDSA_K * MLDSA_POLYT1_PACKEDBYTES)++#define MLDSA_PK_END (MLDSA_PK_T1_OFFSET + MLDSA_PK_T1_BYTES)++#define MLDSA_CRYPTO_PUBLICKEYBYTES MLDSA_PK_END++/* Layout of the packed secret key+ * sk[MLDSA_CRYPTO_SECRETKEYBYTES] = (rho, key, tr, s1, s2, t0):+ *+ * +-----------+-----------+-----------+-----------+-----------+-----------++ * | rho | key | tr | s1 | s2 | t0 |+ * +-----------+-----------+-----------+-----------+-----------+-----------++ * | SEEDBYTES | SEEDBYTES | TRBYTES | L * | K * | K * |+ * | | | | POLYETA_ | POLYETA_ | POLYT0_ |+ * | | | | PACKED- | PACKED- | PACKED- |+ * | | | | BYTES | BYTES | BYTES |+ * +-----------+-----------+-----------+-----------+-----------+-----------++ */+#define MLDSA_SK_RHO_OFFSET 0+#define MLDSA_SK_RHO_BYTES MLDSA_SEEDBYTES++#define MLDSA_SK_KEY_OFFSET (MLDSA_SK_RHO_OFFSET + MLDSA_SK_RHO_BYTES)+#define MLDSA_SK_KEY_BYTES MLDSA_SEEDBYTES++#define MLDSA_SK_TR_OFFSET (MLDSA_SK_KEY_OFFSET + MLDSA_SK_KEY_BYTES)+#define MLDSA_SK_TR_BYTES MLDSA_TRBYTES++#define MLDSA_SK_S1_OFFSET (MLDSA_SK_TR_OFFSET + MLDSA_SK_TR_BYTES)+#define MLDSA_SK_S1_BYTES (MLDSA_L * MLDSA_POLYETA_PACKEDBYTES)++#define MLDSA_SK_S2_OFFSET (MLDSA_SK_S1_OFFSET + MLDSA_SK_S1_BYTES)+#define MLDSA_SK_S2_BYTES (MLDSA_K * MLDSA_POLYETA_PACKEDBYTES)++#define MLDSA_SK_T0_OFFSET (MLDSA_SK_S2_OFFSET + MLDSA_SK_S2_BYTES)+#define MLDSA_SK_T0_BYTES (MLDSA_K * MLDSA_POLYT0_PACKEDBYTES)++#define MLDSA_SK_END (MLDSA_SK_T0_OFFSET + MLDSA_SK_T0_BYTES)++#define MLDSA_CRYPTO_SECRETKEYBYTES MLDSA_SK_END++/* Layout of the packed signature sig[MLDSA_CRYPTO_BYTES] = (c, z, h):+ *+ * +----------------+-------------------+----------------------++ * | c (challenge) | z | h (hints) |+ * +----------------+-------------------+----------------------++ * | CTILDEBYTES | L * | POLYVECH_PACKEDBYTES |+ * | | POLYZ_PACKEDBYTES | (= OMEGA + K) |+ * +----------------+-------------------+----------------------++ */+#define MLDSA_SIG_C_OFFSET 0+#define MLDSA_SIG_C_BYTES MLDSA_CTILDEBYTES++#define MLDSA_SIG_Z_OFFSET (MLDSA_SIG_C_OFFSET + MLDSA_SIG_C_BYTES)+#define MLDSA_SIG_Z_BYTES (MLDSA_L * MLDSA_POLYZ_PACKEDBYTES)++#define MLDSA_SIG_H_OFFSET (MLDSA_SIG_Z_OFFSET + MLDSA_SIG_Z_BYTES)+#define MLDSA_SIG_H_BYTES MLDSA_POLYVECH_PACKEDBYTES++#define MLDSA_SIG_END (MLDSA_SIG_H_OFFSET + MLDSA_SIG_H_BYTES)++#define MLDSA_CRYPTO_BYTES MLDSA_SIG_END++#endif /* !MLD_PARAMS_H */
@@ -0,0 +1,1066 @@+/*+ * Copyright (c) The mldsa-native project authors+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS204]+ * FIPS 204 Module-Lattice-Based Digital Signature Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/204/final+ *+ * - [REF]+ * CRYSTALS-Dilithium reference implementation+ * Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/dilithium/tree/master/ref+ */++#include "poly.h"++#include "common.h"+#include "ct.h"+#include "debug.h"+#include "reduce.h"+#include "rounding.h"+#include "symmetric.h"++#if !defined(MLD_CONFIG_MULTILEVEL_NO_SHARED)+#include "zetas.inc"++MLD_INTERNAL_API+void mld_poly_reduce(mld_poly *a)+{+ unsigned int i;+ mld_assert_bound(a->coeffs, MLDSA_N, INT32_MIN, MLD_REDUCE32_DOMAIN_MAX);++ for (i = 0; i < MLDSA_N; ++i)+ __loop__(+ invariant(i <= MLDSA_N)+ invariant(forall(k0, i, MLDSA_N, a->coeffs[k0] == loop_entry(*a).coeffs[k0]))+ invariant(array_bound(a->coeffs, 0, i, -MLD_REDUCE32_RANGE_MAX, MLD_REDUCE32_RANGE_MAX))+ decreases(MLDSA_N - i))+ {+ a->coeffs[i] = mld_reduce32(a->coeffs[i]);+ }++ mld_assert_bound(a->coeffs, MLDSA_N, -MLD_REDUCE32_RANGE_MAX,+ MLD_REDUCE32_RANGE_MAX);+}++MLD_STATIC_TESTABLE void mld_poly_caddq_c(mld_poly *a)+__contract__(+ requires(memory_no_alias(a, sizeof(mld_poly)))+ requires(array_abs_bound(a->coeffs, 0, MLDSA_N, MLDSA_Q))+ assigns(memory_slice(a, sizeof(mld_poly)))+ ensures(array_bound(a->coeffs, 0, MLDSA_N, 0, MLDSA_Q))+)+{+ unsigned int i;+ mld_assert_abs_bound(a->coeffs, MLDSA_N, MLDSA_Q);++ for (i = 0; i < MLDSA_N; ++i)+ __loop__(+ invariant(i <= MLDSA_N)+ invariant(forall(k0, i, MLDSA_N, a->coeffs[k0] == loop_entry(*a).coeffs[k0]))+ invariant(array_bound(a->coeffs, 0, i, 0, MLDSA_Q))+ decreases(MLDSA_N - i)+ )+ {+ a->coeffs[i] = mld_caddq(a->coeffs[i]);+ }++ mld_assert_bound(a->coeffs, MLDSA_N, 0, MLDSA_Q);+}++MLD_INTERNAL_API+void mld_poly_caddq(mld_poly *a)+{+#if defined(MLD_USE_NATIVE_POLY_CADDQ)+ int ret;+ mld_assert_abs_bound(a->coeffs, MLDSA_N, MLDSA_Q);+ ret = mld_poly_caddq_native(a->coeffs);+ if (ret == MLD_NATIVE_FUNC_SUCCESS)+ {+ mld_assert_bound(a->coeffs, MLDSA_N, 0, MLDSA_Q);+ return;+ }+#endif /* MLD_USE_NATIVE_POLY_CADDQ */+ mld_poly_caddq_c(a);+}++#if !defined(MLD_CONFIG_NO_KEYPAIR_API) || !defined(MLD_CONFIG_NO_SIGN_API) || \+ defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)+/* Reference: We use destructive version (output=first input) to avoid+ * reasoning about aliasing in the CBMC specification */+MLD_INTERNAL_API+void mld_poly_add(mld_poly *r, const mld_poly *b)+{+ unsigned int i;+ for (i = 0; i < MLDSA_N; ++i)+ __loop__(+ assigns(i, memory_slice(r, sizeof(mld_poly)))+ invariant(i <= MLDSA_N)+ invariant(forall(k0, i, MLDSA_N, r->coeffs[k0] == loop_entry(*r).coeffs[k0]))+ invariant(forall(k1, 0, i, r->coeffs[k1] == loop_entry(*r).coeffs[k1] + b->coeffs[k1]))+ invariant(forall(k2, 0, i, r->coeffs[k2] < MLD_REDUCE32_DOMAIN_MAX))+ invariant(forall(k2, 0, i, r->coeffs[k2] >= INT32_MIN))+ decreases(MLDSA_N - i)+ )+ {+ r->coeffs[i] = r->coeffs[i] + b->coeffs[i];+ }+}+#endif /* !MLD_CONFIG_NO_KEYPAIR_API || !MLD_CONFIG_NO_SIGN_API || \+ MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST */++#if !defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API)+/* Reference: We use destructive version (output=first input) to avoid+ * reasoning about aliasing in the CBMC specification */+MLD_INTERNAL_API+void mld_poly_sub(mld_poly *r, const mld_poly *b)+{+ unsigned int i;+ mld_assert_abs_bound(b->coeffs, MLDSA_N, MLDSA_Q);+ mld_assert_abs_bound(r->coeffs, MLDSA_N, MLDSA_Q);++ for (i = 0; i < MLDSA_N; ++i)+ __loop__(+ invariant(i <= MLDSA_N)+ invariant(array_bound(r->coeffs, 0, i, INT32_MIN, MLD_REDUCE32_DOMAIN_MAX))+ invariant(forall(k0, i, MLDSA_N, r->coeffs[k0] == loop_entry(*r).coeffs[k0]))+ decreases(MLDSA_N - i)+ )+ {+ r->coeffs[i] = r->coeffs[i] - b->coeffs[i];+ }++ mld_assert_bound(r->coeffs, MLDSA_N, INT32_MIN, MLD_REDUCE32_DOMAIN_MAX);+}+#endif /* !MLD_CONFIG_NO_SIGN_API || !MLD_CONFIG_NO_VERIFY_API */++#if !defined(MLD_CONFIG_NO_VERIFY_API)+MLD_INTERNAL_API+void mld_poly_shiftl(mld_poly *a)+{+ unsigned int i;+ mld_assert_bound(a->coeffs, MLDSA_N, 0, 1 << 10);++ for (i = 0; i < MLDSA_N; i++)+ __loop__(+ invariant(i <= MLDSA_N)+ invariant(array_bound(a->coeffs, 0, i, 0, MLDSA_Q))+ invariant(forall(k0, i, MLDSA_N, a->coeffs[k0] == loop_entry(*a).coeffs[k0]))+ decreases(MLDSA_N - i))+ {+ /* Reference: uses a left shift by MLDSA_D which is undefined behaviour in+ * C90/C99+ */+ a->coeffs[i] *= (1 << MLDSA_D);+ }+ mld_assert_bound(a->coeffs, MLDSA_N, 0, MLDSA_Q);+}+#endif /* !MLD_CONFIG_NO_VERIFY_API */++static MLD_INLINE int32_t mld_fqmul(int32_t a, int32_t b)+__contract__(+ requires(b > -MLDSA_Q_HALF && b < MLDSA_Q_HALF)+ ensures(return_value > -MLD_FQMUL_BOUND && return_value < MLD_FQMUL_BOUND)+)+{+ /* Bounds: We argue in mld_montgomery_reduce() that the result+ * of Montgomery reduction is < MLDSA_Q if the input is smaller+ * than 2^31 * MLDSA_Q in absolute value. Indeed, we have:+ *+ * |a * b| = |a| * |b|+ * < 2^31 * MLDSA_Q_HALF+ * < 2^31 * MLDSA_Q+ *+ * So the output is < MLDSA_Q < MLD_FQMUL_BOUND.+ */+ return mld_montgomery_reduce((int64_t)a * (int64_t)b);+}++/* mld_ntt_butterfly_block()+ *+ * Computes a block CT butterflies with a fixed twiddle factor,+ * using Montgomery multiplication.+ *+ * Parameters:+ * - r: Pointer to base of polynomial (_not_ the base of butterfly block)+ * - zeta: Twiddle factor to use for the butterfly. This must be in+ * Montgomery form and signed canonical.+ * - start: Offset to the beginning of the butterfly block+ * - len: Index difference between coefficients subject to a butterfly+ * - bound: Ghost variable describing coefficient bound: Prior to `start`,+ * coefficients must be bound by `bound + MLDSA_Q`. Post `start`,+ * they must be bound by `bound`.+ * When this function returns, output coefficients in the index range+ * [start, start+2*len) have bound bumped to `bound + MLDSA_Q`.+ * Example:+ * - start=8, len=4+ * This would compute the following four butterflies+ * 8 -- 12+ * 9 -- 13+ * 10 -- 14+ * 11 -- 15+ * - start=4, len=2+ * This would compute the following two butterflies+ * 4 -- 6+ * 5 -- 7+ */++/* Reference: Embedded in `ntt()` in the reference implementation @[REF]. */+static MLD_INLINE void mld_ntt_butterfly_block(int32_t r[MLDSA_N],+ const int32_t zeta,+ const unsigned start,+ const unsigned len,+ const uint32_t bound)+__contract__(+ requires(start < MLDSA_N)+ requires(1 <= len && len <= MLDSA_N / 2 && start + 2 * len <= MLDSA_N)+ requires(0 <= bound && bound < INT32_MAX - MLD_FQMUL_BOUND)+ requires(-MLDSA_Q_HALF < zeta && zeta < MLDSA_Q_HALF)+ requires(memory_no_alias(r, sizeof(int32_t) * MLDSA_N))+ requires(array_abs_bound(r, 0, start, bound + MLD_FQMUL_BOUND))+ requires(array_abs_bound(r, start, MLDSA_N, bound))+ assigns(memory_slice(r, sizeof(int32_t) * MLDSA_N))+ ensures(array_abs_bound(r, 0, start + 2*len, bound + MLD_FQMUL_BOUND))+ ensures(array_abs_bound(r, start + 2 * len, MLDSA_N, bound)))+{+ /* `bound` is a ghost variable only needed in the CBMC specification */+ unsigned j;+ ((void)bound);+ for (j = start; j < start + len; j++)+ __loop__(+ invariant(start <= j && j <= start + len)+ /*+ * Coefficients are updated in strided pairs, so the bounds for the+ * intermediate states alternate twice between the old and new bound+ */+ invariant(array_abs_bound(r, 0, j, bound + MLD_FQMUL_BOUND))+ invariant(array_abs_bound(r, j, start + len, bound))+ invariant(array_abs_bound(r, start + len, j + len, bound + MLD_FQMUL_BOUND))+ invariant(array_abs_bound(r, j + len, MLDSA_N, bound))+ decreases(start + len - j))+ {+ int32_t t;+ t = mld_fqmul(r[j + len], zeta);+ r[j + len] = r[j] - t;+ r[j] = r[j] + t;+ }+}++/* mld_ntt_layer()+ *+ * Compute one layer of forward NTT+ *+ * Parameters:+ * - r: Pointer to base of polynomial+ * - layer: Indicates which layer is being applied.+ */++/* Reference: Embedded in `ntt()` in the reference implementation @[REF]. */+static MLD_INLINE void mld_ntt_layer(int32_t r[MLDSA_N], const unsigned layer)+__contract__(+ requires(memory_no_alias(r, sizeof(int32_t) * MLDSA_N))+ requires(1 <= layer && layer <= 8)+ requires(array_abs_bound(r, 0, MLDSA_N, layer * MLD_FQMUL_BOUND))+ assigns(memory_slice(r, sizeof(int32_t) * MLDSA_N))+ ensures(array_abs_bound(r, 0, MLDSA_N, (layer + 1) * MLD_FQMUL_BOUND)))+{+ unsigned start, k, len;+ /* Twiddle factors for layer n are at indices 2^(n-1)..2^n-1. */+ k = 1u << (layer - 1);+ len = (unsigned)MLDSA_N >> layer;+ for (start = 0; start < MLDSA_N; start += 2 * len)+ __loop__(+ invariant(start < MLDSA_N + 2 * len)+ invariant(k <= MLDSA_N)+ invariant(2 * len * k == start + MLDSA_N)+ invariant(array_abs_bound(r, 0, start, layer * MLD_FQMUL_BOUND + MLD_FQMUL_BOUND))+ invariant(array_abs_bound(r, start, MLDSA_N, layer * MLD_FQMUL_BOUND))+ decreases(MLDSA_N - start))+ {+ int32_t zeta = mld_zetas[k++];+ mld_ntt_butterfly_block(r, zeta, start, len, layer * MLD_FQMUL_BOUND);+ }+}++MLD_STATIC_TESTABLE void mld_poly_ntt_c(mld_poly *a)+__contract__(+ requires(memory_no_alias(a, sizeof(mld_poly)))+ requires(array_abs_bound(a->coeffs, 0, MLDSA_N, MLDSA_Q))+ assigns(memory_slice(a, sizeof(mld_poly)))+ ensures(array_abs_bound(a->coeffs, 0, MLDSA_N, MLD_NTT_BOUND))+)+{+ unsigned int layer;+ int32_t *r;+++ mld_assert_abs_bound(a->coeffs, MLDSA_N, MLDSA_Q);+ r = a->coeffs;++ for (layer = 1; layer < 9; layer++)+ __loop__(+ invariant(1 <= layer && layer <= 9)+ invariant(array_abs_bound(r, 0, MLDSA_N, layer * MLD_FQMUL_BOUND))+ decreases(9 - layer)+ )+ {+ mld_ntt_layer(r, layer);+ }++ mld_assert_abs_bound(a->coeffs, MLDSA_N, MLD_NTT_BOUND);+}++MLD_INTERNAL_API+void mld_poly_ntt(mld_poly *a)+{+#if defined(MLD_USE_NATIVE_NTT)+ int ret;+ mld_assert_abs_bound(a->coeffs, MLDSA_N, MLDSA_Q);+ ret = mld_ntt_native(a->coeffs);+ if (ret == MLD_NATIVE_FUNC_SUCCESS)+ {+ mld_assert_abs_bound(a->coeffs, MLDSA_N, MLD_NTT_BOUND);+ return;+ }+#endif /* MLD_USE_NATIVE_NTT */+ mld_poly_ntt_c(a);+}++/**+ * Scale a field element by mont/256, i.e., perform Montgomery multiplication+ * by mont^2/256.+ *+ * Input is expected to have absolute value smaller than 256 * MLDSA_Q. Output+ * has absolute value smaller than MLD_INTT_BOUND.+ *+ * @param a Field element to be scaled.+ */+static MLD_INLINE int32_t mld_fqscale(int32_t a)+__contract__(+ requires(a > -256*MLDSA_Q && a < 256*MLDSA_Q)+ ensures(return_value > -MLD_INTT_BOUND && return_value < MLD_INTT_BOUND)+)+{+ /* check-magic: 41978 == pow(2,64-8,MLDSA_Q) */+ const int32_t f = 41978;+ /* Bounds: MLD_INTT_BOUND is MLDSA_Q, so the bounds reasoning is just+ * a special case of that in mld_fqmul(). */+ return mld_montgomery_reduce((int64_t)a * f);+}++/* Reference: Embedded into `invntt_tomont()` in the reference implementation+ * @[REF] */+static MLD_INLINE void mld_invntt_layer(int32_t r[MLDSA_N], unsigned layer)+__contract__(+ requires(memory_no_alias(r, sizeof(int32_t) * MLDSA_N))+ requires(1 <= layer && layer <= 8)+ requires(array_abs_bound(r, 0, MLDSA_N, (MLDSA_N >> layer) * MLDSA_Q))+ assigns(memory_slice(r, sizeof(int32_t) * MLDSA_N))+ ensures(array_abs_bound(r, 0, MLDSA_N, (MLDSA_N >> (layer - 1)) * MLDSA_Q)))+{+ unsigned start, k, len;+ len = (unsigned)MLDSA_N >> layer;+ k = (1u << layer) - 1;+ for (start = 0; start < MLDSA_N; start += 2 * len)+ __loop__(+ invariant(start <= MLDSA_N && k <= 255)+ invariant(2 * len * k + start == 2 * MLDSA_N - 2 * len)+ invariant(array_abs_bound(r, 0, start, (MLDSA_N >> (layer - 1)) * MLDSA_Q))+ invariant(array_abs_bound(r, start, MLDSA_N, (MLDSA_N >> layer) * MLDSA_Q))+ decreases(MLDSA_N - start))+ {+ unsigned j;+ int32_t zeta = -mld_zetas[k--];++ /* The bound `(MLDSA_N >> (layer - 1)) * MLDSA_Q` is loose enough to+ * cover both the input bound `(MLDSA_N >> layer) * MLDSA_Q`+ * (for layers >= 1) and the fqmul output bound `MLD_FQMUL_BOUND`+ * (which is < 2 * MLDSA_Q <= (MLDSA_N >> (layer - 1)) * MLDSA_Q). */+ for (j = start; j < start + len; j++)+ __loop__(+ invariant(start <= j && j <= start + len)+ invariant(array_abs_bound(r, 0, start, (MLDSA_N >> (layer - 1)) * MLDSA_Q))+ invariant(array_abs_bound(r, start, j, (MLDSA_N >> (layer - 1)) * MLDSA_Q))+ invariant(array_abs_bound(r, j, start + len, (MLDSA_N >> layer) * MLDSA_Q))+ invariant(array_abs_bound(r, start + len, j + len, (MLDSA_N >> (layer - 1)) * MLDSA_Q))+ invariant(array_abs_bound(r, j + len, MLDSA_N, (MLDSA_N >> layer) * MLDSA_Q))+ decreases(start + len - j))+ {+ int32_t t = r[j];+ r[j] = t + r[j + len];+ r[j + len] = t - r[j + len];+ r[j + len] = mld_fqmul(r[j + len], zeta);+ }+ }+}++MLD_STATIC_TESTABLE void mld_poly_invntt_tomont_c(mld_poly *a)+__contract__(+ requires(memory_no_alias(a, sizeof(mld_poly)))+ requires(array_abs_bound(a->coeffs, 0, MLDSA_N, MLDSA_Q))+ assigns(memory_slice(a, sizeof(mld_poly)))+ ensures(array_abs_bound(a->coeffs, 0, MLDSA_N, MLD_INTT_BOUND))+)+{+ unsigned int layer, j;+ int32_t *r;++ mld_assert_abs_bound(a->coeffs, MLDSA_N, MLDSA_Q);++ r = a->coeffs;+ for (layer = 8; layer >= 1; layer--)+ __loop__(+ invariant(layer <= 8)+ /* Absolute bounds increase from 1Q before layer 8 */+ /* up to 256Q after layer 1 */+ invariant(array_abs_bound(r, 0, MLDSA_N, (MLDSA_N >> layer) * MLDSA_Q))+ decreases(layer))+ {+ mld_invntt_layer(r, layer);+ }++ /* Coefficient bounds are now at 256Q. We now scale by mont / 256,+ * i.e., compute the Montgomery multiplication by mont^2 / 256.+ * mont corrects the mont^-1 factor introduced in the basemul.+ * 1/256 performs that scaling of the inverse NTT.+ * The reduced value is bounded by MLD_INTT_BOUND in absolute+ * value.*/+ for (j = 0; j < MLDSA_N; ++j)+ __loop__(+ invariant(j <= MLDSA_N)+ invariant(array_abs_bound(r, 0, j, MLD_INTT_BOUND))+ invariant(array_abs_bound(r, j, MLDSA_N, MLDSA_N * MLDSA_Q))+ decreases(MLDSA_N - j)+ )+ {+ r[j] = mld_fqscale(r[j]);+ }++ mld_assert_abs_bound(a->coeffs, MLDSA_N, MLD_INTT_BOUND);+}+++MLD_INTERNAL_API+void mld_poly_invntt_tomont(mld_poly *a)+{+#if defined(MLD_USE_NATIVE_INTT)+ int ret;+ mld_assert_abs_bound(a->coeffs, MLDSA_N, MLDSA_Q);+ ret = mld_intt_native(a->coeffs);+ if (ret == MLD_NATIVE_FUNC_SUCCESS)+ {+ mld_assert_abs_bound(a->coeffs, MLDSA_N, MLD_INTT_BOUND);+ return;+ }+#endif /* MLD_USE_NATIVE_INTT */+ mld_poly_invntt_tomont_c(a);+}++#if !defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API) || \+ defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)+MLD_STATIC_TESTABLE void mld_poly_pointwise_montgomery_c(mld_poly *a,+ const mld_poly *b)+__contract__(+ requires(memory_no_alias(a, sizeof(mld_poly)))+ requires(memory_no_alias(b, sizeof(mld_poly)))+ requires(array_abs_bound(a->coeffs, 0, MLDSA_N, MLD_NTT_BOUND))+ requires(array_abs_bound(b->coeffs, 0, MLDSA_N, MLD_NTT_BOUND))+ assigns(memory_slice(a, sizeof(mld_poly)))+ ensures(array_abs_bound(a->coeffs, 0, MLDSA_N, MLDSA_Q))+)+{+ unsigned int i;+ mld_assert_abs_bound(a->coeffs, MLDSA_N, MLD_NTT_BOUND);+ mld_assert_abs_bound(b->coeffs, MLDSA_N, MLD_NTT_BOUND);++ for (i = 0; i < MLDSA_N; ++i)+ __loop__(+ invariant(i <= MLDSA_N)+ invariant(array_abs_bound(a->coeffs, 0, i, MLDSA_Q))+ invariant(array_abs_bound(a->coeffs, i, MLDSA_N, MLD_NTT_BOUND))+ decreases(MLDSA_N - i)+ )+ {+ a->coeffs[i] = mld_montgomery_reduce((int64_t)a->coeffs[i] * b->coeffs[i]);+ }+ mld_assert_abs_bound(a->coeffs, MLDSA_N, MLDSA_Q);+}++MLD_INTERNAL_API+void mld_poly_pointwise_montgomery(mld_poly *a, const mld_poly *b)+{+#if defined(MLD_USE_NATIVE_POINTWISE_MONTGOMERY)+ int ret;+ mld_assert_abs_bound(a->coeffs, MLDSA_N, MLD_NTT_BOUND);+ mld_assert_abs_bound(b->coeffs, MLDSA_N, MLD_NTT_BOUND);+ ret = mld_poly_pointwise_montgomery_native(a->coeffs, b->coeffs);+ if (ret == MLD_NATIVE_FUNC_SUCCESS)+ {+ mld_assert_abs_bound(a->coeffs, MLDSA_N, MLDSA_Q);+ return;+ }+#endif /* MLD_USE_NATIVE_POINTWISE_MONTGOMERY */+ mld_poly_pointwise_montgomery_c(a, b);+}+#endif /* !MLD_CONFIG_NO_SIGN_API || !MLD_CONFIG_NO_VERIFY_API || \+ MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API)+MLD_INTERNAL_API+void mld_poly_power2round(mld_poly *a1, mld_poly *a0, const mld_poly *a)+{+ unsigned int i;+ mld_assert_bound(a->coeffs, MLDSA_N, 0, MLDSA_Q);++ for (i = 0; i < MLDSA_N; ++i)+ __loop__(+ assigns(i, memory_slice(a0, sizeof(mld_poly)), memory_slice(a1, sizeof(mld_poly)))+ invariant(i <= MLDSA_N)+ invariant(forall(k0, i, MLDSA_N, a->coeffs[k0] == loop_entry(*a).coeffs[k0]))+ invariant(array_bound(a0->coeffs, 0, i, -(MLD_2_POW_D/2)+1, (MLD_2_POW_D/2)+1))+ invariant(array_bound(a1->coeffs, 0, i, 0, ((MLDSA_Q - 1) / MLD_2_POW_D) + 1))+ decreases(MLDSA_N - i)+ )+ {+ mld_power2round(&a0->coeffs[i], &a1->coeffs[i], a->coeffs[i]);+ }++ mld_assert_bound(a0->coeffs, MLDSA_N, -(MLD_2_POW_D / 2) + 1,+ (MLD_2_POW_D / 2) + 1);+ mld_assert_bound(a1->coeffs, MLDSA_N, 0, ((MLDSA_Q - 1) / MLD_2_POW_D) + 1);+}+#endif /* !MLD_CONFIG_NO_KEYPAIR_API */++#ifndef MLD_POLY_UNIFORM_NBLOCKS+#define MLD_POLY_UNIFORM_NBLOCKS \+ ((768 + MLD_STREAM128_BLOCKBYTES - 1) / MLD_STREAM128_BLOCKBYTES)+#endif+/* Reference: `mld_rej_uniform()` in the reference implementation @[REF].+ * - Our signature differs from the reference implementation+ * in that it adds the offset and always expects the base of the+ * target buffer. This avoids shifting the buffer base in the+ * caller, which appears tricky to reason about. */+MLD_STATIC_TESTABLE unsigned int mld_rej_uniform_c(int32_t *a,+ unsigned int target,+ unsigned int offset,+ const uint8_t *buf,+ unsigned int buflen)+__contract__(+ requires(offset <= target && target <= MLDSA_N)+ requires(buflen <= (MLD_POLY_UNIFORM_NBLOCKS * MLD_STREAM128_BLOCKBYTES) && buflen % 3 == 0)+ requires(memory_no_alias(a, sizeof(int32_t) * target))+ requires(memory_no_alias(buf, buflen))+ requires(array_bound(a, 0, offset, 0, MLDSA_Q))+ assigns(memory_slice(a, sizeof(int32_t) * target))+ ensures(offset <= return_value && return_value <= target)+ ensures(array_bound(a, 0, return_value, 0, MLDSA_Q))+)+{+ unsigned int ctr, pos;+ uint32_t t;+ mld_assert_bound(a, offset, 0, MLDSA_Q);++ ctr = offset;+ pos = 0;+ /* pos + 3 cannot overflow due to the assumption+ buflen <= (MLD_POLY_UNIFORM_NBLOCKS * MLD_STREAM128_BLOCKBYTES) */+ while (ctr < target && pos + 3 <= buflen)+ __loop__(+ invariant(offset <= ctr && ctr <= target && pos <= buflen)+ invariant(array_bound(a, 0, ctr, 0, MLDSA_Q))+ decreases(buflen - pos))+ {+ t = buf[pos++];+ t |= (uint32_t)buf[pos++] << 8;+ t |= (uint32_t)buf[pos++] << 16;+ t &= 0x7FFFFF;++ if (t < MLDSA_Q)+ {+ a[ctr++] = (int32_t)t;+ }+ }++ mld_assert_bound(a, ctr, 0, MLDSA_Q);++ return ctr;+}+/**+ * Sample uniformly random coefficients in [0, MLDSA_Q-1] by performing+ * rejection sampling on an array of random bytes.+ *+ * @param[out] a Pointer to output array (allocated).+ * @param target Requested number of coefficients to sample.+ * @param offset Number of coefficients already sampled.+ * @param[in] buf Array of random bytes to sample from.+ * @param buflen Length of array of random bytes (must be multiple of 3).+ *+ * @return Number of sampled coefficients. Can be smaller than len if not+ * enough random bytes were given.+ */++/* Reference: `mld_rej_uniform()` in the reference implementation @[REF].+ * - Our signature differs from the reference implementation+ * in that it adds the offset and always expects the base of the+ * target buffer. This avoids shifting the buffer base in the+ * caller, which appears tricky to reason about. */+static unsigned int mld_rej_uniform(int32_t *a, unsigned int target,+ unsigned int offset, const uint8_t *buf,+ unsigned int buflen)+__contract__(+ requires(offset <= target && target <= MLDSA_N)+ requires(buflen <= (MLD_POLY_UNIFORM_NBLOCKS * MLD_STREAM128_BLOCKBYTES) && buflen % 3 == 0)+ requires(memory_no_alias(a, sizeof(int32_t) * target))+ requires(memory_no_alias(buf, buflen))+ requires(array_bound(a, 0, offset, 0, MLDSA_Q))+ assigns(memory_slice(a, sizeof(int32_t) * target))+ ensures(offset <= return_value && return_value <= target)+ ensures(array_bound(a, 0, return_value, 0, MLDSA_Q))+)+{+#if defined(MLD_USE_NATIVE_REJ_UNIFORM)+ int ret;+ mld_assert_bound(a, offset, 0, MLDSA_Q);+ if (offset == 0)+ {+ ret = mld_rej_uniform_native(a, target, buf, buflen);+ if (ret != MLD_NATIVE_FUNC_FALLBACK)+ {+ unsigned res = (unsigned)ret;+ mld_assert_bound(a, res, 0, MLDSA_Q);+ return res;+ }+ }+#endif /* MLD_USE_NATIVE_REJ_UNIFORM */++ return mld_rej_uniform_c(a, target, offset, buf, buflen);+}++/* Reference: poly_uniform() in the reference implementation @[REF].+ * - Simplified from reference by removing buffer tail handling+ * since buflen % 3 = 0 always holds true (MLD_STREAM128_BLOCKBYTES+ * = 168).+ * - Modified rej_uniform interface to track offset directly.+ * - Pass nonce packed in the extended seed array instead of a third+ * argument.+ * */+MLD_INTERNAL_API+void mld_poly_uniform(mld_poly *a, const uint8_t seed[MLDSA_SEEDBYTES + 2])+{+ unsigned int ctr;+ unsigned int buflen = MLD_POLY_UNIFORM_NBLOCKS * MLD_STREAM128_BLOCKBYTES;+ MLD_ALIGN uint8_t buf[MLD_POLY_UNIFORM_NBLOCKS * MLD_STREAM128_BLOCKBYTES];+ mld_xof128_ctx state;++ mld_xof128_init(&state);+ mld_xof128_absorb_once(&state, seed, MLDSA_SEEDBYTES + 2);+ mld_xof128_squeezeblocks(buf, MLD_POLY_UNIFORM_NBLOCKS, &state);++ ctr = mld_rej_uniform(a->coeffs, MLDSA_N, 0, buf, buflen);+ buflen = MLD_STREAM128_BLOCKBYTES;+ while (ctr < MLDSA_N)+ __loop__(+ assigns(ctr, state, memory_slice(a, sizeof(mld_poly)), object_whole(buf))+ invariant(ctr <= MLDSA_N)+ invariant(array_bound(a->coeffs, 0, ctr, 0, MLDSA_Q))+ invariant(state.pos <= SHAKE128_RATE)+ )+ {+ mld_xof128_squeezeblocks(buf, 1, &state);+ ctr = mld_rej_uniform(a->coeffs, MLDSA_N, ctr, buf, buflen);+ }+ mld_xof128_release(&state);+ mld_assert_bound(a->coeffs, MLDSA_N, 0, MLDSA_Q);++ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ mld_zeroize(buf, sizeof(buf));+}++#if !defined(MLD_CONFIG_SERIAL_FIPS202_ONLY) && \+ (!defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST))+MLD_INTERNAL_API+void mld_poly_uniform_4x(mld_poly *vec0, mld_poly *vec1, mld_poly *vec2,+ mld_poly *vec3,+ uint8_t seed[4][MLD_ALIGN_UP(MLDSA_SEEDBYTES + 2)])+{+ /* Temporary buffers for XOF output before rejection sampling */+ MLD_ALIGN uint8_t+ buf[4][MLD_ALIGN_UP(MLD_POLY_UNIFORM_NBLOCKS * MLD_STREAM128_BLOCKBYTES)];++ /* Tracks the number of coefficients we have already sampled */+ unsigned ctr[4];+ mld_xof128_x4_ctx state;+ unsigned buflen;++ mld_xof128_x4_init(&state);+ mld_xof128_x4_absorb(&state, seed, MLDSA_SEEDBYTES + 2);++ /*+ * Initially, squeeze heuristic number of MLD_POLY_UNIFORM_NBLOCKS.+ * This should generate the matrix entries with high probability.+ */++ mld_xof128_x4_squeezeblocks(buf, MLD_POLY_UNIFORM_NBLOCKS, &state);+ buflen = MLD_POLY_UNIFORM_NBLOCKS * MLD_STREAM128_BLOCKBYTES;+ ctr[0] = mld_rej_uniform(vec0->coeffs, MLDSA_N, 0, buf[0], buflen);+ ctr[1] = mld_rej_uniform(vec1->coeffs, MLDSA_N, 0, buf[1], buflen);+ ctr[2] = mld_rej_uniform(vec2->coeffs, MLDSA_N, 0, buf[2], buflen);+ ctr[3] = mld_rej_uniform(vec3->coeffs, MLDSA_N, 0, buf[3], buflen);++ /*+ * So long as not all matrix entries have been generated, squeeze+ * one more block a time until we're done.+ */+ buflen = MLD_STREAM128_BLOCKBYTES;+ while (ctr[0] < MLDSA_N || ctr[1] < MLDSA_N || ctr[2] < MLDSA_N ||+ ctr[3] < MLDSA_N)+ __loop__(+ assigns(ctr, state, object_whole(buf),+ memory_slice(vec0, sizeof(mld_poly)), memory_slice(vec1, sizeof(mld_poly)),+ memory_slice(vec2, sizeof(mld_poly)), memory_slice(vec3, sizeof(mld_poly)))+ invariant(ctr[0] <= MLDSA_N && ctr[1] <= MLDSA_N)+ invariant(ctr[2] <= MLDSA_N && ctr[3] <= MLDSA_N)+ invariant(array_bound(vec0->coeffs, 0, ctr[0], 0, MLDSA_Q))+ invariant(array_bound(vec1->coeffs, 0, ctr[1], 0, MLDSA_Q))+ invariant(array_bound(vec2->coeffs, 0, ctr[2], 0, MLDSA_Q))+ invariant(array_bound(vec3->coeffs, 0, ctr[3], 0, MLDSA_Q)))+ {+ mld_xof128_x4_squeezeblocks(buf, 1, &state);+ ctr[0] = mld_rej_uniform(vec0->coeffs, MLDSA_N, ctr[0], buf[0], buflen);+ ctr[1] = mld_rej_uniform(vec1->coeffs, MLDSA_N, ctr[1], buf[1], buflen);+ ctr[2] = mld_rej_uniform(vec2->coeffs, MLDSA_N, ctr[2], buf[2], buflen);+ ctr[3] = mld_rej_uniform(vec3->coeffs, MLDSA_N, ctr[3], buf[3], buflen);+ }+ mld_xof128_x4_release(&state);++ mld_assert_bound(vec0->coeffs, MLDSA_N, 0, MLDSA_Q);+ mld_assert_bound(vec1->coeffs, MLDSA_N, 0, MLDSA_Q);+ mld_assert_bound(vec2->coeffs, MLDSA_N, 0, MLDSA_Q);+ mld_assert_bound(vec3->coeffs, MLDSA_N, 0, MLDSA_Q);++ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ mld_zeroize(buf, sizeof(buf));+}++#endif /* !MLD_CONFIG_SERIAL_FIPS202_ONLY && (!MLD_CONFIG_REDUCE_RAM || \+ MLD_UNIT_TEST) */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API)+MLD_INTERNAL_API+void mld_polyt1_pack(uint8_t r[MLDSA_POLYT1_PACKEDBYTES], const mld_poly *a)+{+ unsigned int i;+ mld_assert_bound(a->coeffs, MLDSA_N, 0, 1 << 10);++ for (i = 0; i < MLDSA_N / 4; ++i)+ __loop__(+ invariant(i <= MLDSA_N/4)+ decreases(MLDSA_N / 4 - i))+ {+ r[5 * i + 0] = (uint8_t)((a->coeffs[4 * i + 0] >> 0) & 0xFF);+ r[5 * i + 1] =+ (uint8_t)(((a->coeffs[4 * i + 0] >> 8) | (a->coeffs[4 * i + 1] << 2)) &+ 0xFF);+ r[5 * i + 2] =+ (uint8_t)(((a->coeffs[4 * i + 1] >> 6) | (a->coeffs[4 * i + 2] << 4)) &+ 0xFF);+ r[5 * i + 3] =+ (uint8_t)(((a->coeffs[4 * i + 2] >> 4) | (a->coeffs[4 * i + 3] << 6)) &+ 0xFF);+ r[5 * i + 4] = (uint8_t)((a->coeffs[4 * i + 3] >> 2) & 0xFF);+ }+}+#endif /* !MLD_CONFIG_NO_KEYPAIR_API */++#if !defined(MLD_CONFIG_NO_VERIFY_API)+MLD_INTERNAL_API+void mld_polyt1_unpack(mld_poly *r, const uint8_t a[MLDSA_POLYT1_PACKEDBYTES])+{+ unsigned int i;++ for (i = 0; i < MLDSA_N / 4; ++i)+ __loop__(+ invariant(i <= MLDSA_N/4)+ invariant(array_bound(r->coeffs, 0, i*4, 0, 1 << 10))+ decreases(MLDSA_N / 4 - i))+ {+ r->coeffs[4 * i + 0] =+ ((a[5 * i + 0] >> 0) | ((int32_t)a[5 * i + 1] << 8)) & 0x3FF;+ r->coeffs[4 * i + 1] =+ ((a[5 * i + 1] >> 2) | ((int32_t)a[5 * i + 2] << 6)) & 0x3FF;+ r->coeffs[4 * i + 2] =+ ((a[5 * i + 2] >> 4) | ((int32_t)a[5 * i + 3] << 4)) & 0x3FF;+ r->coeffs[4 * i + 3] =+ ((a[5 * i + 3] >> 6) | ((int32_t)a[5 * i + 4] << 2)) & 0x3FF;+ }++ mld_assert_bound(r->coeffs, MLDSA_N, 0, 1 << 10);+}+#endif /* !MLD_CONFIG_NO_VERIFY_API */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API)+MLD_INTERNAL_API+void mld_polyt0_pack(uint8_t r[MLDSA_POLYT0_PACKEDBYTES], const mld_poly *a)+{+ unsigned int i;+ uint32_t t[8];++ mld_assert_bound(a->coeffs, MLDSA_N, -(1 << (MLDSA_D - 1)) + 1,+ (1 << (MLDSA_D - 1)) + 1);++ for (i = 0; i < MLDSA_N / 8; ++i)+ __loop__(+ invariant(i <= MLDSA_N/8)+ decreases(MLDSA_N / 8 - i))+ {+ /* Safety: a->coeffs[i] <= (1 << (MLDSA_D - 1) as they are output of+ * power2round, hence, these casts are safe. */+ t[0] = (uint32_t)((1 << (MLDSA_D - 1)) - a->coeffs[8 * i + 0]);+ t[1] = (uint32_t)((1 << (MLDSA_D - 1)) - a->coeffs[8 * i + 1]);+ t[2] = (uint32_t)((1 << (MLDSA_D - 1)) - a->coeffs[8 * i + 2]);+ t[3] = (uint32_t)((1 << (MLDSA_D - 1)) - a->coeffs[8 * i + 3]);+ t[4] = (uint32_t)((1 << (MLDSA_D - 1)) - a->coeffs[8 * i + 4]);+ t[5] = (uint32_t)((1 << (MLDSA_D - 1)) - a->coeffs[8 * i + 5]);+ t[6] = (uint32_t)((1 << (MLDSA_D - 1)) - a->coeffs[8 * i + 6]);+ t[7] = (uint32_t)((1 << (MLDSA_D - 1)) - a->coeffs[8 * i + 7]);++ r[13 * i + 0] = (uint8_t)((t[0]) & 0xFF);+ r[13 * i + 1] = (uint8_t)((t[0] >> 8) & 0xFF);+ r[13 * i + 1] |= (uint8_t)((t[1] << 5) & 0xFF);+ r[13 * i + 2] = (uint8_t)((t[1] >> 3) & 0xFF);+ r[13 * i + 3] = (uint8_t)((t[1] >> 11) & 0xFF);+ r[13 * i + 3] |= (uint8_t)((t[2] << 2) & 0xFF);+ r[13 * i + 4] = (uint8_t)((t[2] >> 6) & 0xFF);+ r[13 * i + 4] |= (uint8_t)((t[3] << 7) & 0xFF);+ r[13 * i + 5] = (uint8_t)((t[3] >> 1) & 0xFF);+ r[13 * i + 6] = (uint8_t)((t[3] >> 9) & 0xFF);+ r[13 * i + 6] |= (uint8_t)((t[4] << 4) & 0xFF);+ r[13 * i + 7] = (uint8_t)((t[4] >> 4) & 0xFF);+ r[13 * i + 8] = (uint8_t)((t[4] >> 12) & 0xFF);+ r[13 * i + 8] |= (uint8_t)((t[5] << 1) & 0xFF);+ r[13 * i + 9] = (uint8_t)((t[5] >> 7) & 0xFF);+ r[13 * i + 9] |= (uint8_t)((t[6] << 6) & 0xFF);+ r[13 * i + 10] = (uint8_t)((t[6] >> 2) & 0xFF);+ r[13 * i + 11] = (uint8_t)((t[6] >> 10) & 0xFF);+ r[13 * i + 11] |= (uint8_t)((t[7] << 3) & 0xFF);+ r[13 * i + 12] = (uint8_t)((t[7] >> 5) & 0xFF);+ }+}+#endif /* !MLD_CONFIG_NO_KEYPAIR_API */++#if !defined(MLD_CONFIG_NO_SIGN_API) || defined(MLD_UNIT_TEST)+MLD_INTERNAL_API+void mld_polyt0_unpack(mld_poly *r, const uint8_t a[MLDSA_POLYT0_PACKEDBYTES])+{+ unsigned int i;++ for (i = 0; i < MLDSA_N / 8; ++i)+ __loop__(+ invariant(i <= MLDSA_N/8)+ invariant(array_bound(r->coeffs, 0, i*8, -(1<<(MLDSA_D-1)) + 1, (1<<(MLDSA_D-1)) + 1))+ decreases(MLDSA_N / 8 - i))+ {+ r->coeffs[8 * i + 0] = a[13 * i + 0];+ r->coeffs[8 * i + 0] |= (int32_t)a[13 * i + 1] << 8;+ r->coeffs[8 * i + 0] &= 0x1FFF;++ r->coeffs[8 * i + 1] = a[13 * i + 1] >> 5;+ r->coeffs[8 * i + 1] |= (int32_t)a[13 * i + 2] << 3;+ r->coeffs[8 * i + 1] |= (int32_t)a[13 * i + 3] << 11;+ r->coeffs[8 * i + 1] &= 0x1FFF;++ r->coeffs[8 * i + 2] = a[13 * i + 3] >> 2;+ r->coeffs[8 * i + 2] |= (int32_t)a[13 * i + 4] << 6;+ r->coeffs[8 * i + 2] &= 0x1FFF;++ r->coeffs[8 * i + 3] = a[13 * i + 4] >> 7;+ r->coeffs[8 * i + 3] |= (int32_t)a[13 * i + 5] << 1;+ r->coeffs[8 * i + 3] |= (int32_t)a[13 * i + 6] << 9;+ r->coeffs[8 * i + 3] &= 0x1FFF;++ r->coeffs[8 * i + 4] = a[13 * i + 6] >> 4;+ r->coeffs[8 * i + 4] |= (int32_t)a[13 * i + 7] << 4;+ r->coeffs[8 * i + 4] |= (int32_t)a[13 * i + 8] << 12;+ r->coeffs[8 * i + 4] &= 0x1FFF;++ r->coeffs[8 * i + 5] = a[13 * i + 8] >> 1;+ r->coeffs[8 * i + 5] |= (int32_t)a[13 * i + 9] << 7;+ r->coeffs[8 * i + 5] &= 0x1FFF;++ r->coeffs[8 * i + 6] = a[13 * i + 9] >> 6;+ r->coeffs[8 * i + 6] |= (int32_t)a[13 * i + 10] << 2;+ r->coeffs[8 * i + 6] |= (int32_t)a[13 * i + 11] << 10;+ r->coeffs[8 * i + 6] &= 0x1FFF;++ r->coeffs[8 * i + 7] = a[13 * i + 11] >> 3;+ r->coeffs[8 * i + 7] |= (int32_t)a[13 * i + 12] << 5;+ r->coeffs[8 * i + 7] &= 0x1FFF;++ r->coeffs[8 * i + 0] = (1 << (MLDSA_D - 1)) - r->coeffs[8 * i + 0];+ r->coeffs[8 * i + 1] = (1 << (MLDSA_D - 1)) - r->coeffs[8 * i + 1];+ r->coeffs[8 * i + 2] = (1 << (MLDSA_D - 1)) - r->coeffs[8 * i + 2];+ r->coeffs[8 * i + 3] = (1 << (MLDSA_D - 1)) - r->coeffs[8 * i + 3];+ r->coeffs[8 * i + 4] = (1 << (MLDSA_D - 1)) - r->coeffs[8 * i + 4];+ r->coeffs[8 * i + 5] = (1 << (MLDSA_D - 1)) - r->coeffs[8 * i + 5];+ r->coeffs[8 * i + 6] = (1 << (MLDSA_D - 1)) - r->coeffs[8 * i + 6];+ r->coeffs[8 * i + 7] = (1 << (MLDSA_D - 1)) - r->coeffs[8 * i + 7];+ }++ mld_assert_bound(r->coeffs, MLDSA_N, -(1 << (MLDSA_D - 1)) + 1,+ (1 << (MLDSA_D - 1)) + 1);+}+#endif /* !MLD_CONFIG_NO_SIGN_API || MLD_UNIT_TEST */++MLD_STATIC_TESTABLE uint32_t mld_poly_chknorm_c(const mld_poly *a, int32_t B)+__contract__(+ requires(memory_no_alias(a, sizeof(mld_poly)))+ requires(0 <= B && B <= MLDSA_Q - MLD_REDUCE32_RANGE_MAX)+ requires(array_bound(a->coeffs, 0, MLDSA_N, -MLD_REDUCE32_RANGE_MAX, MLD_REDUCE32_RANGE_MAX))+ ensures(return_value == 0 || return_value == 0xFFFFFFFF)+ ensures((return_value == 0) == array_abs_bound(a->coeffs, 0, MLDSA_N, B))+)+{+ unsigned int i;+ uint32_t t = 0;+ mld_assert_bound(a->coeffs, MLDSA_N, -MLD_REDUCE32_RANGE_MAX,+ MLD_REDUCE32_RANGE_MAX);+ for (i = 0; i < MLDSA_N; ++i)+ __loop__(+ invariant(i <= MLDSA_N)+ invariant(t == 0 || t == 0xFFFFFFFF)+ invariant((t == 0) == array_abs_bound(a->coeffs, 0, i, B))+ decreases(MLDSA_N - i)+ )+ {+ /*+ * Since we know that -MLD_REDUCE32_RANGE_MAX <= a < MLD_REDUCE32_RANGE_MAX,+ * and B <= MLDSA_Q - MLD_REDUCE32_RANGE_MAX, to check if+ * -B < (a mod± MLDSA_Q) < B, it suffices to check if -B < a < B.+ *+ * We prove this to be true using the following CBMC assertions.+ * a ==> b expressed as !a || b to also allow run-time assertion.+ */+ mld_assert(a->coeffs[i] < B || a->coeffs[i] - MLDSA_Q <= -B);+ mld_assert(a->coeffs[i] > -B || a->coeffs[i] + MLDSA_Q >= B);++ /* Reference: Leaks which coefficient violates the bound via a conditional.+ * We are more conservative to reduce the number of declassifications in+ * constant-time testing.+ */++ /* if (abs(a[i]) >= B) */+ t |= mld_ct_cmask_neg_i32(B - 1 - mld_ct_abs_i32(a->coeffs[i]));+ }++ return t;+}++/* Reference: explicitly checks the bound B to be <= (MLDSA_Q - 1) / 8).+ * This is unnecessary as it's always a compile-time constant.+ * We instead model it as a precondition.+ * Checking the bound is performed using a conditional arguing+ * that it is okay to leak which coefficient violates the bound (while the+ * coefficient itself must remain secret).+ * We instead perform everything in constant-time.+ * Also it is sufficient to check that it is smaller than+ * MLDSA_Q - MLD_REDUCE32_RANGE_MAX > (MLDSA_Q - 1) / 8).+ */+MLD_INTERNAL_API+uint32_t mld_poly_chknorm(const mld_poly *a, int32_t B)+{+#if defined(MLD_USE_NATIVE_POLY_CHKNORM)+ int ret;+ int success;+ mld_assert_bound(a->coeffs, MLDSA_N, -MLD_REDUCE32_RANGE_MAX,+ MLD_REDUCE32_RANGE_MAX);+ /* The native backend returns 0 if all coefficients are within the bound,+ * 1 if at least one coefficient exceeds the bound, and+ * -1 (MLD_NATIVE_FUNC_FALLBACK) if the platform does not have the+ * required capabilities to run the native function.+ */+ ret = mld_poly_chknorm_native(a->coeffs, B);++ success = (ret != MLD_NATIVE_FUNC_FALLBACK);+ /* Constant-time: It would be fine to leak the return value of chknorm+ * entirely (as it is fine to leak if any coefficient exceeded the bound or+ * not). However, it is cleaner to perform declassification in sign.c.+ * Hence, here we only declassify if the native function returned+ * MLD_NATIVE_FUNC_FALLBACK or not (which solely depends on system+ * capabilities).+ */+ MLD_CT_TESTING_DECLASSIFY(&success, sizeof(int));+ if (success)+ {+ /* Convert 0 / 1 to 0 / 0xFFFFFFFF here */+ return mld_ct_cmask_nonzero_u32((uint32_t)ret);+ }+#endif /* MLD_USE_NATIVE_POLY_CHKNORM */+ return mld_poly_chknorm_c(a, B);+}++#if !defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API)+#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \+ MLD_CONFIG_PARAMETER_SET == 44+MLD_INTERNAL_API+void mld_polyw1_pack_88(uint8_t r[MLDSA_POLYW1_PACKEDBYTES_88],+ const mld_poly *a)+{+ unsigned int i;++ mld_assert_bound(a->coeffs, MLDSA_N, 0,+ (MLDSA_Q - 1) / (2 * MLDSA_GAMMA2_88));++ for (i = 0; i < MLDSA_N / 4; ++i)+ __loop__(+ invariant(i <= MLDSA_N/4)+ decreases(MLDSA_N / 4 - i))+ {+ r[3 * i + 0] = (uint8_t)((a->coeffs[4 * i + 0]) & 0xFF);+ r[3 * i + 0] |= (uint8_t)((a->coeffs[4 * i + 1] << 6) & 0xFF);+ r[3 * i + 1] = (uint8_t)((a->coeffs[4 * i + 1] >> 2) & 0xFF);+ r[3 * i + 1] |= (uint8_t)((a->coeffs[4 * i + 2] << 4) & 0xFF);+ r[3 * i + 2] = (uint8_t)((a->coeffs[4 * i + 2] >> 4) & 0xFF);+ r[3 * i + 2] |= (uint8_t)((a->coeffs[4 * i + 3] << 2) & 0xFF);+ }+}+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 44 \+ */++#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \+ (MLD_CONFIG_PARAMETER_SET == 65 || MLD_CONFIG_PARAMETER_SET == 87)+MLD_INTERNAL_API+void mld_polyw1_pack_32(uint8_t r[MLDSA_POLYW1_PACKEDBYTES_32],+ const mld_poly *a)+{+ unsigned int i;++ mld_assert_bound(a->coeffs, MLDSA_N, 0,+ (MLDSA_Q - 1) / (2 * MLDSA_GAMMA2_32));++ for (i = 0; i < MLDSA_N / 2; ++i)+ __loop__(+ invariant(i <= MLDSA_N/2)+ decreases(MLDSA_N / 2 - i))+ {+ r[i] =+ (uint8_t)((a->coeffs[2 * i + 0] | (a->coeffs[2 * i + 1] << 4)) & 0xFF);+ }+}+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 65 \+ || MLD_CONFIG_PARAMETER_SET == 87 */+#endif /* !MLD_CONFIG_NO_SIGN_API || !MLD_CONFIG_NO_VERIFY_API */++#else /* !MLD_CONFIG_MULTILEVEL_NO_SHARED */+MLD_EMPTY_CU(mld_poly)+#endif /* MLD_CONFIG_MULTILEVEL_NO_SHARED */++/* To facilitate single-compilation-unit (SCU) builds, undefine all macros.+ * Don't modify by hand -- this is auto-generated by scripts/autogen. */+#undef MLD_POLY_UNIFORM_NBLOCKS
@@ -0,0 +1,464 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS204]+ * FIPS 204 Module-Lattice-Based Digital Signature Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/204/final+ */++#ifndef MLD_POLY_H+#define MLD_POLY_H++#include "cbmc.h"+#include "common.h"+#include "reduce.h"+#include "rounding.h"++/* Absolute exclusive upper bound for the output of fqmul */+#define MLD_FQMUL_BOUND ((5 * MLDSA_Q + 3) / 4)+/* Absolute exclusive upper bound for the output of the forward NTT */+#define MLD_NTT_BOUND (9 * MLD_FQMUL_BOUND)+/* Absolute exclusive upper bound for the output of the inverse NTT*/+#define MLD_INTT_BOUND MLDSA_Q++/**+ * Element of R_q = Z_q[X]/(X^n + 1). Represents polynomial+ * coeffs[0] + X*coeffs[1] + X^2*coeffs[2] + ... + X^{n-1}*coeffs[n-1].+ */+typedef struct+{+ int32_t coeffs[MLDSA_N]; /**< Polynomial coefficients. */+} MLD_ALIGN mld_poly;++#define mld_poly_reduce MLD_NAMESPACE(poly_reduce)+/**+ * In-place reduction of all coefficients of polynomial to representative in+ * [-MLD_REDUCE32_RANGE_MAX, MLD_REDUCE32_RANGE_MAX].+ *+ * @param[in,out] a Pointer to input/output polynomial.+ */+MLD_INTERNAL_API+void mld_poly_reduce(mld_poly *a)+__contract__(+ requires(memory_no_alias(a, sizeof(mld_poly)))+ requires(array_bound(a->coeffs, 0, MLDSA_N, INT32_MIN, MLD_REDUCE32_DOMAIN_MAX))+ assigns(memory_slice(a, sizeof(mld_poly)))+ ensures(array_bound(a->coeffs, 0, MLDSA_N, -MLD_REDUCE32_RANGE_MAX, MLD_REDUCE32_RANGE_MAX))+);++#define mld_poly_caddq MLD_NAMESPACE(poly_caddq)+/**+ * For all coefficients of in/out polynomial add MLDSA_Q if coefficient is+ * negative.+ *+ * @param[in,out] a Pointer to input/output polynomial.+ */+MLD_INTERNAL_API+void mld_poly_caddq(mld_poly *a)+__contract__(+ requires(memory_no_alias(a, sizeof(mld_poly)))+ requires(array_abs_bound(a->coeffs, 0, MLDSA_N, MLDSA_Q))+ assigns(memory_slice(a, sizeof(mld_poly)))+ ensures(array_bound(a->coeffs, 0, MLDSA_N, 0, MLDSA_Q))+);++#if !defined(MLD_CONFIG_NO_KEYPAIR_API) || !defined(MLD_CONFIG_NO_SIGN_API) || \+ defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)+#define mld_poly_add MLD_NAMESPACE(poly_add)+/**+ * Add polynomials. No modular reduction is performed.+ *+ * @spec{Implements @[FIPS204, Algorithm 44, AddNTT] (coefficientwise+ * polynomial addition; also used for addition in the normal domain).}+ *+ * @param[in,out] r Pointer to input-output polynomial to be added to.+ * @param[in] b Pointer to input polynomial that should be added to r.+ * Must be disjoint from r.+ */++/*+ * NOTE: The reference implementation uses a 3-argument poly_add.+ * We specialize to the accumulator form to avoid reasoning about aliasing.+ */+MLD_INTERNAL_API+void mld_poly_add(mld_poly *r, const mld_poly *b)+__contract__(+ requires(memory_no_alias(b, sizeof(mld_poly)))+ requires(memory_no_alias(r, sizeof(mld_poly)))+ requires(forall(k0, 0, MLDSA_N, (int64_t) r->coeffs[k0] + b->coeffs[k0] < MLD_REDUCE32_DOMAIN_MAX))+ requires(forall(k1, 0, MLDSA_N, (int64_t) r->coeffs[k1] + b->coeffs[k1] >= INT32_MIN))+ assigns(memory_slice(r, sizeof(mld_poly)))+ ensures(forall(k2, 0, MLDSA_N, r->coeffs[k2] == old(*r).coeffs[k2] + b->coeffs[k2]))+ ensures(forall(k3, 0, MLDSA_N, r->coeffs[k3] < MLD_REDUCE32_DOMAIN_MAX))+ ensures(forall(k4, 0, MLDSA_N, r->coeffs[k4] >= INT32_MIN))+);+#endif /* !MLD_CONFIG_NO_KEYPAIR_API || !MLD_CONFIG_NO_SIGN_API || \+ MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST */++#if !defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API)+#define mld_poly_sub MLD_NAMESPACE(poly_sub)+/**+ * Subtract polynomials. No modular reduction is performed.+ *+ * @param[in,out] r Pointer to input-output polynomial.+ * @param[in] b Pointer to input polynomial that should be subtracted from+ * r. Must be disjoint from r.+ */+/*+ * NOTE: The reference implementation uses a 3-argument poly_sub.+ * We specialize to the accumulator form to avoid reasoning about aliasing.+ */+MLD_INTERNAL_API+void mld_poly_sub(mld_poly *r, const mld_poly *b)+__contract__(+ requires(memory_no_alias(b, sizeof(mld_poly)))+ requires(memory_no_alias(r, sizeof(mld_poly)))+ requires(array_abs_bound(r->coeffs, 0, MLDSA_N, MLDSA_Q))+ requires(array_abs_bound(b->coeffs, 0, MLDSA_N, MLDSA_Q))+ assigns(memory_slice(r, sizeof(mld_poly)))+ ensures(array_bound(r->coeffs, 0, MLDSA_N, INT32_MIN, MLD_REDUCE32_DOMAIN_MAX))+);+#endif /* !MLD_CONFIG_NO_SIGN_API || !MLD_CONFIG_NO_VERIFY_API */++#if !defined(MLD_CONFIG_NO_VERIFY_API)+#define mld_poly_shiftl MLD_NAMESPACE(poly_shiftl)+/**+ * Multiply polynomial by 2^MLDSA_D without modular reduction. Assumes input+ * coefficients to be less than 2^{31-MLDSA_D} in absolute value.+ *+ * @param[in,out] a Pointer to input/output polynomial.+ */+MLD_INTERNAL_API+void mld_poly_shiftl(mld_poly *a)+__contract__(+ requires(memory_no_alias(a, sizeof(mld_poly)))+ requires(array_bound(a->coeffs, 0, MLDSA_N, 0, 1 << 10))+ assigns(memory_slice(a, sizeof(mld_poly)))+ ensures(array_bound(a->coeffs, 0, MLDSA_N, 0, MLDSA_Q))+);+#endif /* !MLD_CONFIG_NO_VERIFY_API */++#define mld_poly_ntt MLD_NAMESPACE(poly_ntt)+/**+ * In-place forward NTT. Output coefficients are bounded by MLD_NTT_BOUND in+ * absolute value.+ *+ * @spec{Implements @[FIPS204, Algorithm 41, NTT].}+ *+ * @param[in,out] a Pointer to input/output polynomial.+ */+MLD_INTERNAL_API+void mld_poly_ntt(mld_poly *a)+__contract__(+ requires(memory_no_alias(a, sizeof(mld_poly)))+ requires(array_abs_bound(a->coeffs, 0, MLDSA_N, MLDSA_Q))+ assigns(memory_slice(a, sizeof(mld_poly)))+ ensures(array_abs_bound(a->coeffs, 0, MLDSA_N, MLD_NTT_BOUND))+);+++#define mld_poly_invntt_tomont MLD_NAMESPACE(poly_invntt_tomont)+/**+ * In-place inverse NTT.+ *+ * Input coefficients need to be less than MLDSA_Q in absolute value and+ * output coefficients are bounded by MLD_INTT_BOUND.+ *+ * @spec{Implements @[FIPS204, Algorithm 42, NTT^{-1}] up to scaling:+ * The input is scaled by 2^{-32} as a result of the Montgomery base+ * multiplication. The output is in normal domain. In other words, this+ * function implements `NTT^{-1} o mult(2^32)`.}+ *+ * @param[in,out] a Pointer to input/output polynomial.+ */+MLD_INTERNAL_API+void mld_poly_invntt_tomont(mld_poly *a)+__contract__(+ requires(memory_no_alias(a, sizeof(mld_poly)))+ requires(array_abs_bound(a->coeffs, 0, MLDSA_N, MLDSA_Q))+ assigns(memory_slice(a, sizeof(mld_poly)))+ ensures(array_abs_bound(a->coeffs, 0, MLDSA_N, MLD_INTT_BOUND))+);++#if !defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API) || \+ defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)+#define mld_poly_pointwise_montgomery MLD_NAMESPACE(poly_pointwise_montgomery)+/**+ * Pointwise multiplication of polynomials. Destructive in the first argument.+ *+ * @spec{Implements @[FIPS204, Algorithm 45, MultiplyNTT], up to scaling: The+ * input is in normal domain, the output is scaled by 2^{-32} as a result of+ * the use of Montgomery multiplication. In other words, this function+ * implements `mult(2^{-32}) o MultiplyNTT`.}+ *+ * @param[in,out] a Pointer to first input/output polynomial. On entry, holds+ * the first multiplicand; on exit, holds the product+ * a * b * 2^{-32}.+ * @param[in] b Pointer to second input polynomial.+ */+MLD_INTERNAL_API+void mld_poly_pointwise_montgomery(mld_poly *a, const mld_poly *b)+__contract__(+ requires(memory_no_alias(a, sizeof(mld_poly)))+ requires(memory_no_alias(b, sizeof(mld_poly)))+ requires(array_abs_bound(a->coeffs, 0, MLDSA_N, MLD_NTT_BOUND))+ requires(array_abs_bound(b->coeffs, 0, MLDSA_N, MLD_NTT_BOUND))+ assigns(memory_slice(a, sizeof(mld_poly)))+ ensures(array_abs_bound(a->coeffs, 0, MLDSA_N, MLDSA_Q))+);+#endif /* !MLD_CONFIG_NO_SIGN_API || !MLD_CONFIG_NO_VERIFY_API || \+ MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API)+#define mld_poly_power2round MLD_NAMESPACE(poly_power2round)+/**+ * For all coefficients c of the input polynomial, compute c0, c1 such that+ * c mod MLDSA_Q = c1*2^MLDSA_D + c0 with -2^{MLDSA_D-1} < c0 <= 2^{MLDSA_D-1}.+ * Assumes coefficients to be standard representatives.+ *+ * @param[out] a1 Pointer to output polynomial with coefficients c1.+ * @param[out] a0 Pointer to output polynomial with coefficients c0; may alias+ * the input polynomial a.+ * @param[in] a Pointer to input polynomial.+ */+MLD_INTERNAL_API+void mld_poly_power2round(mld_poly *a1, mld_poly *a0, const mld_poly *a)+__contract__(+ requires(memory_no_alias(a0, sizeof(mld_poly)))+ requires(memory_no_alias(a1, sizeof(mld_poly)))+ /* The implementation does not require a0 == a, but the single call site+ * aliases them and asserting equality simplifies the proof. */+ requires(a0 == a)+ requires(array_bound(a->coeffs, 0, MLDSA_N, 0, MLDSA_Q))+ assigns(memory_slice(a1, sizeof(mld_poly)))+ assigns(memory_slice(a0, sizeof(mld_poly)))+ ensures(array_bound(a0->coeffs, 0, MLDSA_N, -(MLD_2_POW_D/2)+1, (MLD_2_POW_D/2)+1))+ ensures(array_bound(a1->coeffs, 0, MLDSA_N, 0, ((MLDSA_Q - 1) / MLD_2_POW_D) + 1))+);+#endif /* !MLD_CONFIG_NO_KEYPAIR_API */++#define mld_poly_uniform MLD_NAMESPACE(poly_uniform)+/**+ * Sample polynomial with uniformly random coefficients in [0, MLDSA_Q-1] by+ * performing rejection sampling on the output stream of SHAKE128(seed|nonce).+ *+ * @spec{Implements @[FIPS204, Algorithm 30, RejNTTPoly].}+ *+ * @param[out] a Pointer to output polynomial.+ * @param[in] seed Byte array with seed of length MLDSA_SEEDBYTES and the+ * packed 2-byte nonce.+ */+MLD_INTERNAL_API+void mld_poly_uniform(mld_poly *a, const uint8_t seed[MLDSA_SEEDBYTES + 2])+__contract__(+ requires(memory_no_alias(a, sizeof(mld_poly)))+ requires(memory_no_alias(seed, MLDSA_SEEDBYTES + 2))+ assigns(memory_slice(a, sizeof(mld_poly)))+ ensures(array_bound(a->coeffs, 0, MLDSA_N, 0, MLDSA_Q))+);++#if !defined(MLD_CONFIG_SERIAL_FIPS202_ONLY) && \+ (!defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST))+#define mld_poly_uniform_4x MLD_NAMESPACE(poly_uniform_4x)+/**+ * Generate four polynomials using rejection sampling on (pseudo-)uniformly+ * random bytes sampled from a seed.+ *+ * @spec{Implements @[FIPS204, Algorithm 30, RejNTTPoly] (four-way batched).}+ *+ * @param[out] vec0 Pointer to first polynomial to be sampled.+ * @param[out] vec1 Pointer to second polynomial to be sampled.+ * @param[out] vec2 Pointer to third polynomial to be sampled.+ * @param[out] vec3 Pointer to fourth polynomial to be sampled.+ * @param[in] seed Pointer to consecutive array of seed buffers of size+ * MLDSA_SEEDBYTES + 2 each, plus padding for alignment.+ */+MLD_INTERNAL_API+void mld_poly_uniform_4x(mld_poly *vec0, mld_poly *vec1, mld_poly *vec2,+ mld_poly *vec3,+ uint8_t seed[4][MLD_ALIGN_UP(MLDSA_SEEDBYTES + 2)])+__contract__(+ requires(memory_no_alias(vec0, sizeof(mld_poly)))+ requires(memory_no_alias(vec1, sizeof(mld_poly)))+ requires(memory_no_alias(vec2, sizeof(mld_poly)))+ requires(memory_no_alias(vec3, sizeof(mld_poly)))+ requires(memory_no_alias(seed, 4 * MLD_ALIGN_UP(MLDSA_SEEDBYTES + 2)))+ assigns(memory_slice(vec0, sizeof(mld_poly)))+ assigns(memory_slice(vec1, sizeof(mld_poly)))+ assigns(memory_slice(vec2, sizeof(mld_poly)))+ assigns(memory_slice(vec3, sizeof(mld_poly)))+ ensures(array_bound(vec0->coeffs, 0, MLDSA_N, 0, MLDSA_Q))+ ensures(array_bound(vec1->coeffs, 0, MLDSA_N, 0, MLDSA_Q))+ ensures(array_bound(vec2->coeffs, 0, MLDSA_N, 0, MLDSA_Q))+ ensures(array_bound(vec3->coeffs, 0, MLDSA_N, 0, MLDSA_Q))+);+#endif /* !MLD_CONFIG_SERIAL_FIPS202_ONLY && (!MLD_CONFIG_REDUCE_RAM || \+ MLD_UNIT_TEST) */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API)+#define mld_polyt1_pack MLD_NAMESPACE(polyt1_pack)+/**+ * Bit-pack polynomial t1 with coefficients fitting in 10 bits. Input+ * coefficients are assumed to be standard representatives.+ *+ * @spec{Implements @[FIPS204, Algorithm 16, SimpleBitPack].}+ *+ * @param[out] r Pointer to output byte array with at least+ * MLDSA_POLYT1_PACKEDBYTES bytes.+ * @param[in] a Pointer to input polynomial.+ */+MLD_INTERNAL_API+void mld_polyt1_pack(uint8_t r[MLDSA_POLYT1_PACKEDBYTES], const mld_poly *a)+__contract__(+ requires(memory_no_alias(r, MLDSA_POLYT1_PACKEDBYTES))+ requires(memory_no_alias(a, sizeof(mld_poly)))+ requires(array_bound(a->coeffs, 0, MLDSA_N, 0, 1 << 10))+ assigns(memory_slice(r, MLDSA_POLYT1_PACKEDBYTES))+);+#endif /* !MLD_CONFIG_NO_KEYPAIR_API */++#if !defined(MLD_CONFIG_NO_VERIFY_API)+#define mld_polyt1_unpack MLD_NAMESPACE(polyt1_unpack)+/**+ * Unpack polynomial t1 with 10-bit coefficients. Output coefficients are+ * standard representatives.+ *+ * @spec{Implements @[FIPS204, Algorithm 18, SimpleBitUnpack].}+ *+ * @param[out] r Pointer to output polynomial.+ * @param[in] a Byte array with bit-packed polynomial.+ */+MLD_INTERNAL_API+void mld_polyt1_unpack(mld_poly *r, const uint8_t a[MLDSA_POLYT1_PACKEDBYTES])+__contract__(+ requires(memory_no_alias(r, sizeof(mld_poly)))+ requires(memory_no_alias(a, MLDSA_POLYT1_PACKEDBYTES))+ assigns(memory_slice(r, sizeof(mld_poly)))+ ensures(array_bound(r->coeffs, 0, MLDSA_N, 0, 1 << 10))+);+#endif /* !MLD_CONFIG_NO_VERIFY_API */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API)+#define mld_polyt0_pack MLD_NAMESPACE(polyt0_pack)+/**+ * Bit-pack polynomial t0 with coefficients in ]-2^{MLDSA_D-1}, 2^{MLDSA_D-1}].+ *+ * @spec{Implements @[FIPS204, Algorithm 17, BitPack].}+ *+ * @param[out] r Pointer to output byte array with at least+ * MLDSA_POLYT0_PACKEDBYTES bytes.+ * @param[in] a Pointer to input polynomial.+ */+MLD_INTERNAL_API+void mld_polyt0_pack(uint8_t r[MLDSA_POLYT0_PACKEDBYTES], const mld_poly *a)+__contract__(+ requires(memory_no_alias(r, MLDSA_POLYT0_PACKEDBYTES))+ requires(memory_no_alias(a, sizeof(mld_poly)))+ requires(array_bound(a->coeffs, 0, MLDSA_N, -(1<<(MLDSA_D-1)) + 1, (1<<(MLDSA_D-1)) + 1))+ assigns(memory_slice(r, MLDSA_POLYT0_PACKEDBYTES))+);+#endif /* !MLD_CONFIG_NO_KEYPAIR_API */++#if !defined(MLD_CONFIG_NO_SIGN_API) || defined(MLD_UNIT_TEST)+#define mld_polyt0_unpack MLD_NAMESPACE(polyt0_unpack)+/**+ * Unpack polynomial t0 with coefficients in ]-2^{MLDSA_D-1}, 2^{MLDSA_D-1}].+ *+ * @spec{Implements @[FIPS204, Algorithm 19, BitUnpack].}+ *+ * @param[out] r Pointer to output polynomial.+ * @param[in] a Byte array with bit-packed polynomial.+ */+MLD_INTERNAL_API+void mld_polyt0_unpack(mld_poly *r, const uint8_t a[MLDSA_POLYT0_PACKEDBYTES])+__contract__(+ requires(memory_no_alias(r, sizeof(mld_poly)))+ requires(memory_no_alias(a, MLDSA_POLYT0_PACKEDBYTES))+ assigns(memory_slice(r, sizeof(mld_poly)))+ ensures(array_bound(r->coeffs, 0, MLDSA_N, -(1<<(MLDSA_D-1)) + 1, (1<<(MLDSA_D-1)) + 1))+);+#endif /* !MLD_CONFIG_NO_SIGN_API || MLD_UNIT_TEST */++#define mld_poly_chknorm MLD_NAMESPACE(poly_chknorm)+/**+ * Check infinity norm of polynomial against given bound. Assumes input+ * coefficients were reduced by mld_reduce32().+ *+ * @spec{@[FIPS204] defines the infinity norm via signed canonical reduction+ * (mod± MLDSA_Q) prior to applying the bounds check. However,+ * `-B < (a mod± MLDSA_Q) < B` is equivalent to `-B < a < B` under the+ * assumption that `B <= MLDSA_Q - MLD_REDUCE32_RANGE_MAX` (cf. the assertion in+ * the code). Hence, this contract and implementation are correct without+ * reduction.}+ *+ * @param[in] a Pointer to polynomial.+ * @param B Norm bound.+ *+ * @return 0 if norm is strictly smaller than+ * B <= (MLDSA_Q - MLD_REDUCE32_RANGE_MAX) and 0xFFFFFFFF otherwise.+ */+MLD_INTERNAL_API+MLD_MUST_CHECK_RETURN_VALUE+uint32_t mld_poly_chknorm(const mld_poly *a, int32_t B)+__contract__(+ requires(memory_no_alias(a, sizeof(mld_poly)))+ requires(0 <= B && B <= MLDSA_Q - MLD_REDUCE32_RANGE_MAX)+ requires(array_bound(a->coeffs, 0, MLDSA_N, -MLD_REDUCE32_RANGE_MAX, MLD_REDUCE32_RANGE_MAX))+ ensures(return_value == 0 || return_value == 0xFFFFFFFF)+ ensures((return_value == 0) == array_abs_bound(a->coeffs, 0, MLDSA_N, B))+);++#if !defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API)+#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || MLD_CONFIG_PARAMETER_SET == 44+#define mld_polyw1_pack_88 MLD_NAMESPACE(polyw1_pack_88)+/**+ * Bit-pack polynomial w1, using 6 bits per coefficient.+ * This is the variant for parameter sets with MLDSA_GAMMA2 = (MLDSA_Q-1)/88+ * (ML-DSA-44), for which w1 coefficients lie in [0, 43].+ *+ * @param[out] r Pointer to output byte array (MLDSA_POLYW1_PACKEDBYTES_88).+ * @param[in] a Pointer to input polynomial.+ */+MLD_INTERNAL_API+void mld_polyw1_pack_88(uint8_t r[MLDSA_POLYW1_PACKEDBYTES_88],+ const mld_poly *a)+__contract__(+ requires(memory_no_alias(r, MLDSA_POLYW1_PACKEDBYTES_88))+ requires(memory_no_alias(a, sizeof(mld_poly)))+ requires(array_bound(a->coeffs, 0, MLDSA_N, 0, (MLDSA_Q-1)/(2*MLDSA_GAMMA2_88)))+ assigns(memory_slice(r, MLDSA_POLYW1_PACKEDBYTES_88))+);+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 44 \+ */++#if defined(MLD_CONFIG_MULTILEVEL_WITH_SHARED) || \+ (MLD_CONFIG_PARAMETER_SET == 65 || MLD_CONFIG_PARAMETER_SET == 87)+#define mld_polyw1_pack_32 MLD_NAMESPACE(polyw1_pack_32)+/**+ * Bit-pack polynomial w1, using 4 bits per coefficient.+ * This is the variant for parameter sets with MLDSA_GAMMA2 = (MLDSA_Q-1)/32+ * (ML-DSA-65 and ML-DSA-87), for which w1 coefficients lie in [0, 15].+ *+ * @param[out] r Pointer to output byte array (MLDSA_POLYW1_PACKEDBYTES_32).+ * @param[in] a Pointer to input polynomial.+ */+MLD_INTERNAL_API+void mld_polyw1_pack_32(uint8_t r[MLDSA_POLYW1_PACKEDBYTES_32],+ const mld_poly *a)+__contract__(+ requires(memory_no_alias(r, MLDSA_POLYW1_PACKEDBYTES_32))+ requires(memory_no_alias(a, sizeof(mld_poly)))+ requires(array_bound(a->coeffs, 0, MLDSA_N, 0, (MLDSA_Q-1)/(2*MLDSA_GAMMA2_32)))+ assigns(memory_slice(r, MLDSA_POLYW1_PACKEDBYTES_32))+);+#endif /* MLD_CONFIG_MULTILEVEL_WITH_SHARED || MLD_CONFIG_PARAMETER_SET == 65 \+ || MLD_CONFIG_PARAMETER_SET == 87 */+#endif /* !MLD_CONFIG_NO_SIGN_API || !MLD_CONFIG_NO_VERIFY_API */++#endif /* !MLD_POLY_H */
@@ -0,0 +1,910 @@+/*+ * Copyright (c) The mldsa-native project authors+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS204]+ * FIPS 204 Module-Lattice-Based Digital Signature Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/204/final+ *+ * - [REF]+ * CRYSTALS-Dilithium reference implementation+ * Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/dilithium/tree/master/ref+ */++#include "poly_kl.h"++#include "ct.h"+#include "debug.h"+#include "rounding.h"+#include "symmetric.h"++/* Parameter set namespacing+ * This is to facilitate building multiple instances+ * of mldsa-native (e.g. with varying parameter sets)+ * within a single compilation unit. */+#define mld_rej_eta MLD_ADD_PARAM_SET(mld_rej_eta)+#define mld_rej_eta_c MLD_ADD_PARAM_SET(mld_rej_eta_c)+#define mld_poly_decompose_c MLD_ADD_PARAM_SET(mld_poly_decompose_c)+#define mld_poly_use_hint_c MLD_ADD_PARAM_SET(mld_poly_use_hint_c)+#define mld_polyz_unpack_c MLD_ADD_PARAM_SET(mld_polyz_unpack_c)+/* End of parameter set namespacing */+++#if !defined(MLD_CONFIG_NO_SIGN_API)+MLD_STATIC_TESTABLE+void mld_poly_decompose_c(mld_poly *a1, mld_poly *a0)+__contract__(+ requires(memory_no_alias(a1, sizeof(mld_poly)))+ requires(memory_no_alias(a0, sizeof(mld_poly)))+ requires(array_bound(a0->coeffs, 0, MLDSA_N, 0, MLDSA_Q))+ assigns(memory_slice(a1, sizeof(mld_poly)))+ assigns(memory_slice(a0, sizeof(mld_poly)))+ ensures(array_bound(a1->coeffs, 0, MLDSA_N, 0, (MLDSA_Q-1)/(2*MLDSA_GAMMA2)))+ ensures(array_abs_bound(a0->coeffs, 0, MLDSA_N, MLDSA_GAMMA2+1))+)+{+ unsigned int i;+ mld_assert_bound(a0->coeffs, MLDSA_N, 0, MLDSA_Q);+ for (i = 0; i < MLDSA_N; ++i)+ __loop__(+ assigns(i, memory_slice(a0, sizeof(mld_poly)), memory_slice(a1, sizeof(mld_poly)))+ invariant(i <= MLDSA_N)+ invariant(array_bound(a0->coeffs, i, MLDSA_N, 0, MLDSA_Q))+ invariant(array_bound(a1->coeffs, 0, i, 0, (MLDSA_Q-1)/(2*MLDSA_GAMMA2)))+ invariant(array_abs_bound(a0->coeffs, 0, i, MLDSA_GAMMA2+1))+ decreases(MLDSA_N - i)+ )+ {+ mld_decompose(&a0->coeffs[i], &a1->coeffs[i], a0->coeffs[i]);+ }++ mld_assert_abs_bound(a0->coeffs, MLDSA_N, MLDSA_GAMMA2 + 1);+ mld_assert_bound(a1->coeffs, MLDSA_N, 0, (MLDSA_Q - 1) / (2 * MLDSA_GAMMA2));+}++MLD_INTERNAL_API+void mld_poly_decompose(mld_poly *a1, mld_poly *a0)+{+#if defined(MLD_USE_NATIVE_POLY_DECOMPOSE_88) && MLD_CONFIG_PARAMETER_SET == 44+ int ret;+ mld_assert_bound(a0->coeffs, MLDSA_N, 0, MLDSA_Q);+ ret = mld_poly_decompose_88_native(a1->coeffs, a0->coeffs);+ if (ret == MLD_NATIVE_FUNC_SUCCESS)+ {+ mld_assert_abs_bound(a0->coeffs, MLDSA_N, MLDSA_GAMMA2 + 1);+ mld_assert_bound(a1->coeffs, MLDSA_N, 0,+ (MLDSA_Q - 1) / (2 * MLDSA_GAMMA2));+ return;+ }+#elif defined(MLD_USE_NATIVE_POLY_DECOMPOSE_32) && \+ (MLD_CONFIG_PARAMETER_SET == 65 || MLD_CONFIG_PARAMETER_SET == 87)+ int ret;+ mld_assert_bound(a0->coeffs, MLDSA_N, 0, MLDSA_Q);+ ret = mld_poly_decompose_32_native(a1->coeffs, a0->coeffs);+ if (ret == MLD_NATIVE_FUNC_SUCCESS)+ {+ mld_assert_abs_bound(a0->coeffs, MLDSA_N, MLDSA_GAMMA2 + 1);+ mld_assert_bound(a1->coeffs, MLDSA_N, 0,+ (MLDSA_Q - 1) / (2 * MLDSA_GAMMA2));+ return;+ }+#endif /* !(MLD_USE_NATIVE_POLY_DECOMPOSE_88 && MLD_CONFIG_PARAMETER_SET == \+ 44) && MLD_USE_NATIVE_POLY_DECOMPOSE_32 && (MLD_CONFIG_PARAMETER_SET \+ == 65 || MLD_CONFIG_PARAMETER_SET == 87) */+ mld_poly_decompose_c(a1, a0);+}++#endif /* !MLD_CONFIG_NO_SIGN_API */++#if !defined(MLD_CONFIG_NO_VERIFY_API)+MLD_STATIC_TESTABLE void mld_poly_use_hint_c(mld_poly *a, const mld_poly *h)+__contract__(+ requires(memory_no_alias(a, sizeof(mld_poly)))+ requires(memory_no_alias(h, sizeof(mld_poly)))+ requires(array_bound(a->coeffs, 0, MLDSA_N, 0, MLDSA_Q))+ requires(array_bound(h->coeffs, 0, MLDSA_N, 0, 2))+ assigns(memory_slice(a, sizeof(mld_poly)))+ ensures(array_bound(a->coeffs, 0, MLDSA_N, 0, (MLDSA_Q-1)/(2*MLDSA_GAMMA2)))+)+{+ unsigned int i;+ mld_assert_bound(a->coeffs, MLDSA_N, 0, MLDSA_Q);+ mld_assert_bound(h->coeffs, MLDSA_N, 0, 2);++ for (i = 0; i < MLDSA_N; ++i)+ __loop__(+ invariant(i <= MLDSA_N)+ invariant(array_bound(a->coeffs, 0, i, 0, (MLDSA_Q-1)/(2*MLDSA_GAMMA2)))+ invariant(array_bound(a->coeffs, i, MLDSA_N, 0, MLDSA_Q))+ decreases(MLDSA_N - i)+ )+ {+ a->coeffs[i] = mld_use_hint(a->coeffs[i], h->coeffs[i]);+ }+ mld_assert_bound(a->coeffs, MLDSA_N, 0, (MLDSA_Q - 1) / (2 * MLDSA_GAMMA2));+}++MLD_INTERNAL_API+void mld_poly_use_hint(mld_poly *a, const mld_poly *h)+{+#if defined(MLD_USE_NATIVE_POLY_USE_HINT_88) && MLD_CONFIG_PARAMETER_SET == 44+ int ret;+ mld_assert_bound(a->coeffs, MLDSA_N, 0, MLDSA_Q);+ mld_assert_bound(h->coeffs, MLDSA_N, 0, 2);+ ret = mld_poly_use_hint_88_native(a->coeffs, h->coeffs);+ if (ret == MLD_NATIVE_FUNC_SUCCESS)+ {+ mld_assert_bound(a->coeffs, MLDSA_N, 0, (MLDSA_Q - 1) / (2 * MLDSA_GAMMA2));+ return;+ }+#elif defined(MLD_USE_NATIVE_POLY_USE_HINT_32) && \+ (MLD_CONFIG_PARAMETER_SET == 65 || MLD_CONFIG_PARAMETER_SET == 87)+ int ret;+ mld_assert_bound(a->coeffs, MLDSA_N, 0, MLDSA_Q);+ mld_assert_bound(h->coeffs, MLDSA_N, 0, 2);+ ret = mld_poly_use_hint_32_native(a->coeffs, h->coeffs);+ if (ret == MLD_NATIVE_FUNC_SUCCESS)+ {+ mld_assert_bound(a->coeffs, MLDSA_N, 0, (MLDSA_Q - 1) / (2 * MLDSA_GAMMA2));+ return;+ }+#endif /* !(MLD_USE_NATIVE_POLY_USE_HINT_88 && MLD_CONFIG_PARAMETER_SET == 44) \+ && MLD_USE_NATIVE_POLY_USE_HINT_32 && (MLD_CONFIG_PARAMETER_SET == \+ 65 || MLD_CONFIG_PARAMETER_SET == 87) */+ mld_poly_use_hint_c(a, h);+}+#endif /* !MLD_CONFIG_NO_VERIFY_API */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API)+/**+ * Sample uniformly random coefficients in [-MLDSA_ETA, MLDSA_ETA] by+ * performing rejection sampling on an array of random bytes.+ *+ * @param[out] a Pointer to output array (allocated).+ * @param target Requested number of coefficients to sample.+ * @param offset Number of coefficients already sampled.+ * @param[in] buf Array of random bytes to sample from.+ * @param buflen Length of array of random bytes.+ *+ * @return Number of sampled coefficients. Can be smaller than target if not+ * enough random bytes were given.+ */++/* Reference: `mld_rej_eta()` in the reference implementation @[REF].+ * - Our signature differs from the reference implementation+ * in that it adds the offset and always expects the base of the+ * target buffer. This avoids shifting the buffer base in the+ * caller, which appears tricky to reason about. */+#if MLDSA_ETA == 2+/*+ * Sampling 256 coefficients mod 15 using rejection sampling from 4 bits.+ * Expected number of required bytes: (256 * (16/15))/2 = 136.5 bytes.+ * We sample 1 block (=136 bytes) of SHAKE256_RATE output initially.+ * Sampling 2 blocks initially results in slightly worse performance.+ */+#define MLD_POLY_UNIFORM_ETA_NBLOCKS 1+#elif MLDSA_ETA == 4+/*+ * Sampling 256 coefficients mod 9 using rejection sampling from 4 bits.+ * Expected number of required bytes: (256 * (16/9))/2 = 227.5 bytes.+ * We sample 2 blocks (=272 bytes) of SHAKE256_RATE output initially.+ */+#define MLD_POLY_UNIFORM_ETA_NBLOCKS 2+#else /* MLDSA_ETA == 4 */+#error "Invalid value of MLDSA_ETA"+#endif /* MLDSA_ETA != 2 && MLDSA_ETA != 4 */++MLD_STATIC_TESTABLE unsigned int mld_rej_eta_c(int32_t *a, unsigned int target,+ unsigned int offset,+ const uint8_t *buf,+ unsigned int buflen)+__contract__(+ requires(offset <= target && target <= MLDSA_N)+ requires(buflen <= (MLD_POLY_UNIFORM_ETA_NBLOCKS * MLD_STREAM256_BLOCKBYTES))+ requires(memory_no_alias(a, sizeof(int32_t) * target))+ requires(memory_no_alias(buf, buflen))+ requires(array_abs_bound(a, 0, offset, MLDSA_ETA + 1))+ assigns(memory_slice(a, sizeof(int32_t) * target))+ ensures(offset <= return_value && return_value <= target)+ ensures(array_abs_bound(a, 0, return_value, MLDSA_ETA + 1))+)+{+ unsigned int ctr, pos;+ int t_valid;+ uint32_t t0, t1;+ mld_assert_abs_bound(a, offset, MLDSA_ETA + 1);+ ctr = offset;+ pos = 0;+ while (ctr < target && pos < buflen)+ __loop__(+ invariant(offset <= ctr && ctr <= target && pos <= buflen)+ invariant(array_abs_bound(a, 0, ctr, MLDSA_ETA + 1))+ decreases(buflen - pos)+ )+ {+ t0 = buf[pos] & 0x0F;+ t1 = buf[pos++] >> 4;++ /* Constant time: The inputs and outputs to the rejection sampling are+ * secret. However, it is fine to leak which coefficients have been+ * rejected. For constant-time testing, we declassify the result of+ * the comparison.+ */+#if MLDSA_ETA == 2+ t_valid = t0 < 15;+ MLD_CT_TESTING_DECLASSIFY(&t_valid, sizeof(int));+ if (t_valid) /* t0 < 15 */+ {+ t0 = t0 - (205 * t0 >> 10) * 5;+ a[ctr++] = 2 - (int32_t)t0;+ }+ t_valid = t1 < 15;+ MLD_CT_TESTING_DECLASSIFY(&t_valid, sizeof(int));+ if (t_valid && ctr < target) /* t1 < 15 */+ {+ t1 = t1 - (205 * t1 >> 10) * 5;+ a[ctr++] = 2 - (int32_t)t1;+ }+#elif MLDSA_ETA == 4+ t_valid = t0 < 9;+ MLD_CT_TESTING_DECLASSIFY(&t_valid, sizeof(int));+ if (t_valid) /* t0 < 9 */+ {+ a[ctr++] = 4 - (int32_t)t0;+ }+ t_valid = t1 < 9; /* t1 < 9 */+ MLD_CT_TESTING_DECLASSIFY(&t_valid, sizeof(int));+ if (t_valid && ctr < target)+ {+ a[ctr++] = 4 - (int32_t)t1;+ }+#else /* MLDSA_ETA == 4 */+#error "Invalid value of MLDSA_ETA"+#endif /* MLDSA_ETA != 2 && MLDSA_ETA != 4 */+ }++ mld_assert_abs_bound(a, ctr, MLDSA_ETA + 1);++ return ctr;+}++static unsigned int mld_rej_eta(int32_t *a, unsigned int target,+ unsigned int offset, const uint8_t *buf,+ unsigned int buflen)+__contract__(+ requires(offset <= target && target <= MLDSA_N)+ requires(buflen <= (MLD_POLY_UNIFORM_ETA_NBLOCKS * MLD_STREAM256_BLOCKBYTES))+ requires(memory_no_alias(a, sizeof(int32_t) * target))+ requires(memory_no_alias(buf, buflen))+ requires(array_abs_bound(a, 0, offset, MLDSA_ETA + 1))+ assigns(memory_slice(a, sizeof(int32_t) * target))+ ensures(offset <= return_value && return_value <= target)+ ensures(array_abs_bound(a, 0, return_value, MLDSA_ETA + 1))+)+{+#if MLDSA_ETA == 2 && defined(MLD_USE_NATIVE_REJ_UNIFORM_ETA2)+ int ret;+ mld_assert_abs_bound(a, offset, MLDSA_ETA + 1);+ if (offset == 0)+ {+ ret = mld_rej_uniform_eta2_native(a, target, buf, buflen);+ if (ret != MLD_NATIVE_FUNC_FALLBACK)+ {+ unsigned res = (unsigned)ret;+ mld_assert_abs_bound(a, res, MLDSA_ETA + 1);+ return res;+ }+ }+#elif MLDSA_ETA == 4 && defined(MLD_USE_NATIVE_REJ_UNIFORM_ETA4)+ int ret;+ mld_assert_abs_bound(a, offset, MLDSA_ETA + 1);+ if (offset == 0)+ {+ ret = mld_rej_uniform_eta4_native(a, target, buf, buflen);+ if (ret != MLD_NATIVE_FUNC_FALLBACK)+ {+ unsigned res = (unsigned)ret;+ mld_assert_abs_bound(a, res, MLDSA_ETA + 1);+ return res;+ }+ }+#endif /* !(MLDSA_ETA == 2 && MLD_USE_NATIVE_REJ_UNIFORM_ETA2) && MLDSA_ETA == \+ 4 && MLD_USE_NATIVE_REJ_UNIFORM_ETA4 */++ return mld_rej_eta_c(a, target, offset, buf, buflen);+}++#if !defined(MLD_CONFIG_SERIAL_FIPS202_ONLY)+MLD_INTERNAL_API+void mld_poly_uniform_eta_4x(mld_poly *r0, mld_poly *r1, mld_poly *r2,+ mld_poly *r3, const uint8_t seed[MLDSA_CRHBYTES],+ uint8_t nonce0, uint8_t nonce1, uint8_t nonce2,+ uint8_t nonce3)+{+ /* Temporary buffers for XOF output before rejection sampling */+ MLD_ALIGN uint8_t buf[4][MLD_ALIGN_UP(MLD_POLY_UNIFORM_ETA_NBLOCKS *+ MLD_STREAM256_BLOCKBYTES)];++ MLD_ALIGN uint8_t extseed[4][MLD_ALIGN_UP(MLDSA_CRHBYTES + 2)];++ /* Tracks the number of coefficients we have already sampled */+ unsigned ctr[4];+ mld_xof256_x4_ctx state;+ unsigned buflen;++ mld_memcpy(extseed[0], seed, MLDSA_CRHBYTES);+ mld_memcpy(extseed[1], seed, MLDSA_CRHBYTES);+ mld_memcpy(extseed[2], seed, MLDSA_CRHBYTES);+ mld_memcpy(extseed[3], seed, MLDSA_CRHBYTES);+ extseed[0][MLDSA_CRHBYTES] = nonce0;+ extseed[1][MLDSA_CRHBYTES] = nonce1;+ extseed[2][MLDSA_CRHBYTES] = nonce2;+ extseed[3][MLDSA_CRHBYTES] = nonce3;+ extseed[0][MLDSA_CRHBYTES + 1] = 0;+ extseed[1][MLDSA_CRHBYTES + 1] = 0;+ extseed[2][MLDSA_CRHBYTES + 1] = 0;+ extseed[3][MLDSA_CRHBYTES + 1] = 0;++ mld_xof256_x4_init(&state);+ mld_xof256_x4_absorb(&state, extseed, MLDSA_CRHBYTES + 2);++ /*+ * Initially, squeeze heuristic number of MLD_POLY_UNIFORM_ETA_NBLOCKS.+ * This should generate the coefficients with high probability.+ */+ mld_xof256_x4_squeezeblocks(buf, MLD_POLY_UNIFORM_ETA_NBLOCKS, &state);+ buflen = MLD_POLY_UNIFORM_ETA_NBLOCKS * MLD_STREAM256_BLOCKBYTES;++ ctr[0] = mld_rej_eta(r0->coeffs, MLDSA_N, 0, buf[0], buflen);+ ctr[1] = mld_rej_eta(r1->coeffs, MLDSA_N, 0, buf[1], buflen);+ ctr[2] = mld_rej_eta(r2->coeffs, MLDSA_N, 0, buf[2], buflen);+ ctr[3] = mld_rej_eta(r3->coeffs, MLDSA_N, 0, buf[3], buflen);++ /*+ * So long as not all entries have been generated, squeeze+ * one more block at a time until we're done.+ */+ buflen = MLD_STREAM256_BLOCKBYTES;+ while (ctr[0] < MLDSA_N || ctr[1] < MLDSA_N || ctr[2] < MLDSA_N ||+ ctr[3] < MLDSA_N)+ __loop__(+ assigns(ctr, state, memory_slice(r0, sizeof(mld_poly)),+ memory_slice(r1, sizeof(mld_poly)), memory_slice(r2, sizeof(mld_poly)),+ memory_slice(r3, sizeof(mld_poly)), object_whole(buf[0]),+ object_whole(buf[1]), object_whole(buf[2]),+ object_whole(buf[3]))+ invariant(ctr[0] <= MLDSA_N && ctr[1] <= MLDSA_N)+ invariant(ctr[2] <= MLDSA_N && ctr[3] <= MLDSA_N)+ invariant(array_abs_bound(r0->coeffs, 0, ctr[0], MLDSA_ETA + 1))+ invariant(array_abs_bound(r1->coeffs, 0, ctr[1], MLDSA_ETA + 1))+ invariant(array_abs_bound(r2->coeffs, 0, ctr[2], MLDSA_ETA + 1))+ invariant(array_abs_bound(r3->coeffs, 0, ctr[3], MLDSA_ETA + 1)))+ {+ mld_xof256_x4_squeezeblocks(buf, 1, &state);+ ctr[0] = mld_rej_eta(r0->coeffs, MLDSA_N, ctr[0], buf[0], buflen);+ ctr[1] = mld_rej_eta(r1->coeffs, MLDSA_N, ctr[1], buf[1], buflen);+ ctr[2] = mld_rej_eta(r2->coeffs, MLDSA_N, ctr[2], buf[2], buflen);+ ctr[3] = mld_rej_eta(r3->coeffs, MLDSA_N, ctr[3], buf[3], buflen);+ }++ mld_xof256_x4_release(&state);++ mld_assert_abs_bound(r0->coeffs, MLDSA_N, MLDSA_ETA + 1);+ mld_assert_abs_bound(r1->coeffs, MLDSA_N, MLDSA_ETA + 1);+ mld_assert_abs_bound(r2->coeffs, MLDSA_N, MLDSA_ETA + 1);+ mld_assert_abs_bound(r3->coeffs, MLDSA_N, MLDSA_ETA + 1);++ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ mld_zeroize(buf, sizeof(buf));+ mld_zeroize(extseed, sizeof(extseed));+}+#else /* !MLD_CONFIG_SERIAL_FIPS202_ONLY */++MLD_INTERNAL_API+void mld_poly_uniform_eta(mld_poly *r, const uint8_t seed[MLDSA_CRHBYTES],+ uint8_t nonce)+{+ /* Temporary buffer for XOF output before rejection sampling */+ MLD_ALIGN uint8_t+ buf[MLD_POLY_UNIFORM_ETA_NBLOCKS * MLD_STREAM256_BLOCKBYTES];+ MLD_ALIGN uint8_t extseed[MLDSA_CRHBYTES + 2];++ /* Tracks the number of coefficients we have already sampled */+ unsigned ctr;+ mld_xof256_ctx state;+ unsigned buflen;++ mld_memcpy(extseed, seed, MLDSA_CRHBYTES);+ extseed[MLDSA_CRHBYTES] = nonce;+ extseed[MLDSA_CRHBYTES + 1] = 0;++ mld_xof256_init(&state);+ mld_xof256_absorb_once(&state, extseed, MLDSA_CRHBYTES + 2);++ /*+ * Initially, squeeze heuristic number of MLD_POLY_UNIFORM_ETA_NBLOCKS.+ * This should generate the coefficients with high probability.+ */+ mld_xof256_squeezeblocks(buf, MLD_POLY_UNIFORM_ETA_NBLOCKS, &state);+ buflen = MLD_POLY_UNIFORM_ETA_NBLOCKS * MLD_STREAM256_BLOCKBYTES;++ ctr = mld_rej_eta(r->coeffs, MLDSA_N, 0, buf, buflen);++ /*+ * So long as not all entries have been generated, squeeze+ * one more block at a time until we're done.+ */+ buflen = MLD_STREAM256_BLOCKBYTES;+ while (ctr < MLDSA_N)+ __loop__(+ assigns(ctr, object_whole(&state),+ object_whole(buf), memory_slice(r, sizeof(mld_poly)))+ invariant(ctr <= MLDSA_N)+ invariant(state.pos <= SHAKE256_RATE)+ invariant(array_abs_bound(r->coeffs, 0, ctr, MLDSA_ETA + 1)))+ {+ mld_xof256_squeezeblocks(buf, 1, &state);+ ctr = mld_rej_eta(r->coeffs, MLDSA_N, ctr, buf, buflen);+ }++ mld_xof256_release(&state);++ mld_assert_abs_bound(r->coeffs, MLDSA_N, MLDSA_ETA + 1);++ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ mld_zeroize(buf, sizeof(buf));+ mld_zeroize(extseed, sizeof(extseed));+}+#endif /* MLD_CONFIG_SERIAL_FIPS202_ONLY */+#endif /* !MLD_CONFIG_NO_KEYPAIR_API */++#if !defined(MLD_CONFIG_NO_SIGN_API)+#define MLD_POLY_UNIFORM_GAMMA1_NBLOCKS \+ ((MLDSA_POLYZ_PACKEDBYTES + MLD_STREAM256_BLOCKBYTES - 1) / \+ MLD_STREAM256_BLOCKBYTES)++#if MLD_CONFIG_PARAMETER_SET == 65 || \+ defined(MLD_CONFIG_SERIAL_FIPS202_ONLY) || \+ defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)+MLD_INTERNAL_API+void mld_poly_uniform_gamma1(mld_poly *a, const uint8_t seed[MLDSA_CRHBYTES],+ uint16_t nonce)+{+ MLD_ALIGN uint8_t+ buf[MLD_POLY_UNIFORM_GAMMA1_NBLOCKS * MLD_STREAM256_BLOCKBYTES];+ MLD_ALIGN uint8_t extseed[MLDSA_CRHBYTES + 2];+ mld_xof256_ctx state;++ mld_memcpy(extseed, seed, MLDSA_CRHBYTES);+ extseed[MLDSA_CRHBYTES] = (uint8_t)(nonce & 0xFF);+ extseed[MLDSA_CRHBYTES + 1] = (uint8_t)(nonce >> 8);++ mld_xof256_init(&state);+ mld_xof256_absorb_once(&state, extseed, MLDSA_CRHBYTES + 2);++ mld_xof256_squeezeblocks(buf, MLD_POLY_UNIFORM_GAMMA1_NBLOCKS, &state);+ mld_polyz_unpack(a, buf);++ mld_xof256_release(&state);++ mld_assert_bound(a->coeffs, MLDSA_N, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1);++ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ mld_zeroize(buf, sizeof(buf));+ mld_zeroize(extseed, sizeof(extseed));+}+#endif /* MLD_CONFIG_PARAMETER_SET == 65 || MLD_CONFIG_SERIAL_FIPS202_ONLY || \+ MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST */+++#if !defined(MLD_CONFIG_SERIAL_FIPS202_ONLY) && \+ (!defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST))+MLD_INTERNAL_API+void mld_poly_uniform_gamma1_4x(mld_poly *r0, mld_poly *r1, mld_poly *r2,+ mld_poly *r3,+ const uint8_t seed[MLDSA_CRHBYTES],+ uint16_t nonce0, uint16_t nonce1,+ uint16_t nonce2, uint16_t nonce3)+{+ /* Temporary buffers for XOF output before rejection sampling */+ MLD_ALIGN uint8_t buf[4][MLD_ALIGN_UP(MLD_POLY_UNIFORM_GAMMA1_NBLOCKS *+ MLD_STREAM256_BLOCKBYTES)];++ MLD_ALIGN uint8_t extseed[4][MLD_ALIGN_UP(MLDSA_CRHBYTES + 2)];++ /* Tracks the number of coefficients we have already sampled */+ mld_xof256_x4_ctx state;++ mld_memcpy(extseed[0], seed, MLDSA_CRHBYTES);+ mld_memcpy(extseed[1], seed, MLDSA_CRHBYTES);+ mld_memcpy(extseed[2], seed, MLDSA_CRHBYTES);+ mld_memcpy(extseed[3], seed, MLDSA_CRHBYTES);+ extseed[0][MLDSA_CRHBYTES] = (uint8_t)(nonce0 & 0xFF);+ extseed[1][MLDSA_CRHBYTES] = (uint8_t)(nonce1 & 0xFF);+ extseed[2][MLDSA_CRHBYTES] = (uint8_t)(nonce2 & 0xFF);+ extseed[3][MLDSA_CRHBYTES] = (uint8_t)(nonce3 & 0xFF);+ extseed[0][MLDSA_CRHBYTES + 1] = (uint8_t)(nonce0 >> 8);+ extseed[1][MLDSA_CRHBYTES + 1] = (uint8_t)(nonce1 >> 8);+ extseed[2][MLDSA_CRHBYTES + 1] = (uint8_t)(nonce2 >> 8);+ extseed[3][MLDSA_CRHBYTES + 1] = (uint8_t)(nonce3 >> 8);++ mld_xof256_x4_init(&state);+ mld_xof256_x4_absorb(&state, extseed, MLDSA_CRHBYTES + 2);+ mld_xof256_x4_squeezeblocks(buf, MLD_POLY_UNIFORM_GAMMA1_NBLOCKS, &state);++ mld_polyz_unpack(r0, buf[0]);+ mld_polyz_unpack(r1, buf[1]);+ mld_polyz_unpack(r2, buf[2]);+ mld_polyz_unpack(r3, buf[3]);+ mld_xof256_x4_release(&state);++ mld_assert_bound(r0->coeffs, MLDSA_N, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1);+ mld_assert_bound(r1->coeffs, MLDSA_N, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1);+ mld_assert_bound(r2->coeffs, MLDSA_N, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1);+ mld_assert_bound(r3->coeffs, MLDSA_N, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1);++ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ mld_zeroize(buf, sizeof(buf));+ mld_zeroize(extseed, sizeof(extseed));+}+#endif /* !MLD_CONFIG_SERIAL_FIPS202_ONLY && (!MLD_CONFIG_REDUCE_RAM || \+ MLD_UNIT_TEST) */+#endif /* !MLD_CONFIG_NO_SIGN_API */++#if !defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API)+MLD_INTERNAL_API+void mld_poly_challenge(mld_poly *c, const uint8_t seed[MLDSA_CTILDEBYTES])+{+ unsigned int i, j, pos;+ uint64_t signs;+ uint64_t offset;+ MLD_ALIGN uint8_t buf[SHAKE256_RATE];+ mld_shake256ctx state;++ mld_shake256_init(&state);+ mld_shake256_absorb(&state, seed, MLDSA_CTILDEBYTES);+ mld_shake256_finalize(&state);+ mld_shake256_squeeze(buf, SHAKE256_RATE, &state);++ /* Convert the first 8 bytes of buf[] into an unsigned 64-bit value. */+ /* Each bit of that dictates the sign of the resulting challenge value */+ signs = 0;+ for (i = 0; i < 8; ++i)+ __loop__(+ assigns(i, signs)+ invariant(i <= 8)+ decreases(8 - i)+ )+ {+ signs |= (uint64_t)buf[i] << 8 * i;+ }+ pos = 8;++ mld_memset(c, 0, sizeof(mld_poly));++ for (i = MLDSA_N - MLDSA_TAU; i < MLDSA_N; ++i)+ __loop__(+ assigns(i, j, object_whole(buf), state, pos, memory_slice(c, sizeof(mld_poly)), signs)+ invariant(i >= MLDSA_N - MLDSA_TAU)+ invariant(i <= MLDSA_N)+ invariant(pos >= 1)+ invariant(pos <= SHAKE256_RATE)+ invariant(array_bound(c->coeffs, 0, MLDSA_N, -1, 2))+ invariant(state.pos <= SHAKE256_RATE)+ decreases(MLDSA_N - i)+ )+ {+ /* This loop terminates only probabilistically, hence no decreases+ * clause. */+ do+ __loop__(+ assigns(j, object_whole(buf), state, pos)+ invariant(state.pos <= SHAKE256_RATE)+ )+ {+ if (pos >= SHAKE256_RATE)+ {+ mld_shake256_squeeze(buf, SHAKE256_RATE, &state);+ pos = 0;+ }+ j = buf[pos++];+ } while (j > i);++ c->coeffs[i] = c->coeffs[j];++ /* Reference: Compute coefficient value here in two steps to */+ /* avoid mixing unsigned and signed arithmetic with implicit */+ /* conversions, and so that CBMC can keep track of ranges */+ /* to complete type-safety proof here. */++ /* The least-significant bit of signs tells us if we want -1 or +1 */+ offset = 2 * (signs & 1);++ /* offset has value 0 or 2 here, so (1 - (int32_t) offset) has+ * value -1 or +1 */+ c->coeffs[j] = 1 - (int32_t)offset;++ /* Move to the next bit of signs for next time */+ signs >>= 1;+ }++ mld_assert_bound(c->coeffs, MLDSA_N, -1, 2);+ mld_shake256_release(&state);++ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ mld_zeroize(buf, sizeof(buf));+ mld_zeroize(&signs, sizeof(signs));+}+#endif /* !MLD_CONFIG_NO_SIGN_API || !MLD_CONFIG_NO_VERIFY_API */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API)+MLD_INTERNAL_API+void mld_polyeta_pack(uint8_t r[MLDSA_POLYETA_PACKEDBYTES], const mld_poly *a)+{+ unsigned int i;+ uint8_t t[8];++ mld_assert_abs_bound(a->coeffs, MLDSA_N, MLDSA_ETA + 1);++#if MLDSA_ETA == 2+ for (i = 0; i < MLDSA_N / 8; ++i)+ __loop__(+ invariant(i <= MLDSA_N/8)+ decreases(MLDSA_N / 8 - i))+ {+ /* The casts are safe since we assume that the coefficients+ * of a are <= MLDSA_ETA in absolute value. */+ t[0] = (uint8_t)(MLDSA_ETA - a->coeffs[8 * i + 0]);+ t[1] = (uint8_t)(MLDSA_ETA - a->coeffs[8 * i + 1]);+ t[2] = (uint8_t)(MLDSA_ETA - a->coeffs[8 * i + 2]);+ t[3] = (uint8_t)(MLDSA_ETA - a->coeffs[8 * i + 3]);+ t[4] = (uint8_t)(MLDSA_ETA - a->coeffs[8 * i + 4]);+ t[5] = (uint8_t)(MLDSA_ETA - a->coeffs[8 * i + 5]);+ t[6] = (uint8_t)(MLDSA_ETA - a->coeffs[8 * i + 6]);+ t[7] = (uint8_t)(MLDSA_ETA - a->coeffs[8 * i + 7]);++ r[3 * i + 0] = (uint8_t)(((t[0] >> 0) | (t[1] << 3) | (t[2] << 6)) & 0xFF);+ r[3 * i + 1] =+ (uint8_t)(((t[2] >> 2) | (t[3] << 1) | (t[4] << 4) | (t[5] << 7)) &+ 0xFF);+ r[3 * i + 2] = (uint8_t)(((t[5] >> 1) | (t[6] << 2) | (t[7] << 5)) & 0xFF);+ }+#elif MLDSA_ETA == 4+ for (i = 0; i < MLDSA_N / 2; ++i)+ __loop__(+ invariant(i <= MLDSA_N/2)+ decreases(MLDSA_N / 2 - i))+ {+ /* The casts are safe since we assume that the coefficients+ * of a are <= MLDSA_ETA in absolute value. */+ t[0] = (uint8_t)(MLDSA_ETA - a->coeffs[2 * i + 0]);+ t[1] = (uint8_t)(MLDSA_ETA - a->coeffs[2 * i + 1]);+ r[i] = (uint8_t)(t[0] | (t[1] << 4));+ }+#else /* MLDSA_ETA == 4 */+#error "Invalid value of MLDSA_ETA"+#endif /* MLDSA_ETA != 2 && MLDSA_ETA != 4 */+}+#endif /* !MLD_CONFIG_NO_KEYPAIR_API */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API) || !defined(MLD_CONFIG_NO_SIGN_API)+MLD_INTERNAL_API+void mld_polyeta_unpack(mld_poly *r, const uint8_t a[MLDSA_POLYETA_PACKEDBYTES])+{+ unsigned int i;++#if MLDSA_ETA == 2+ for (i = 0; i < MLDSA_N / 8; ++i)+ __loop__(+ invariant(i <= MLDSA_N/8)+ invariant(array_bound(r->coeffs, 0, i*8, -5, MLDSA_ETA + 1))+ decreases(MLDSA_N / 8 - i))+ {+ r->coeffs[8 * i + 0] = (a[3 * i + 0] >> 0) & 7;+ r->coeffs[8 * i + 1] = (a[3 * i + 0] >> 3) & 7;+ r->coeffs[8 * i + 2] = ((a[3 * i + 0] >> 6) | (a[3 * i + 1] << 2)) & 7;+ r->coeffs[8 * i + 3] = (a[3 * i + 1] >> 1) & 7;+ r->coeffs[8 * i + 4] = (a[3 * i + 1] >> 4) & 7;+ r->coeffs[8 * i + 5] = ((a[3 * i + 1] >> 7) | (a[3 * i + 2] << 1)) & 7;+ r->coeffs[8 * i + 6] = (a[3 * i + 2] >> 2) & 7;+ r->coeffs[8 * i + 7] = (a[3 * i + 2] >> 5) & 7;++ r->coeffs[8 * i + 0] = MLDSA_ETA - r->coeffs[8 * i + 0];+ r->coeffs[8 * i + 1] = MLDSA_ETA - r->coeffs[8 * i + 1];+ r->coeffs[8 * i + 2] = MLDSA_ETA - r->coeffs[8 * i + 2];+ r->coeffs[8 * i + 3] = MLDSA_ETA - r->coeffs[8 * i + 3];+ r->coeffs[8 * i + 4] = MLDSA_ETA - r->coeffs[8 * i + 4];+ r->coeffs[8 * i + 5] = MLDSA_ETA - r->coeffs[8 * i + 5];+ r->coeffs[8 * i + 6] = MLDSA_ETA - r->coeffs[8 * i + 6];+ r->coeffs[8 * i + 7] = MLDSA_ETA - r->coeffs[8 * i + 7];+ }+#elif MLDSA_ETA == 4+ for (i = 0; i < MLDSA_N / 2; ++i)+ __loop__(+ invariant(i <= MLDSA_N/2)+ invariant(array_bound(r->coeffs, 0, i*2, -11, MLDSA_ETA + 1))+ decreases(MLDSA_N / 2 - i))+ {+ r->coeffs[2 * i + 0] = a[i] & 0x0F;+ r->coeffs[2 * i + 1] = a[i] >> 4;+ r->coeffs[2 * i + 0] = MLDSA_ETA - r->coeffs[2 * i + 0];+ r->coeffs[2 * i + 1] = MLDSA_ETA - r->coeffs[2 * i + 1];+ }+#else /* MLDSA_ETA == 4 */+#error "Invalid value of MLDSA_ETA"+#endif /* MLDSA_ETA != 2 && MLDSA_ETA != 4 */++ mld_assert_bound(r->coeffs, MLDSA_N, MLD_POLYETA_UNPACK_LOWER_BOUND,+ MLDSA_ETA + 1);+}+#endif /* !MLD_CONFIG_NO_KEYPAIR_API || !MLD_CONFIG_NO_SIGN_API */++#if !defined(MLD_CONFIG_NO_SIGN_API)+MLD_INTERNAL_API+void mld_polyz_pack(uint8_t r[MLDSA_POLYZ_PACKEDBYTES], const mld_poly *a)+{+ unsigned int i;+ uint32_t t[4];++ mld_assert_bound(a->coeffs, MLDSA_N, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1);++#if MLD_CONFIG_PARAMETER_SET == 44+ for (i = 0; i < MLDSA_N / 4; ++i)+ __loop__(+ invariant(i <= MLDSA_N/4)+ decreases(MLDSA_N / 4 - i))+ {+ /* Safety: a->coeffs[i] <= MLDSA_GAMMA1, hence, these casts are safe. */+ t[0] = (uint32_t)(MLDSA_GAMMA1 - a->coeffs[4 * i + 0]);+ t[1] = (uint32_t)(MLDSA_GAMMA1 - a->coeffs[4 * i + 1]);+ t[2] = (uint32_t)(MLDSA_GAMMA1 - a->coeffs[4 * i + 2]);+ t[3] = (uint32_t)(MLDSA_GAMMA1 - a->coeffs[4 * i + 3]);++ r[9 * i + 0] = (uint8_t)((t[0]) & 0xFF);+ r[9 * i + 1] = (uint8_t)((t[0] >> 8) & 0xFF);+ r[9 * i + 2] = (uint8_t)((t[0] >> 16) & 0xFF);+ r[9 * i + 2] |= (uint8_t)((t[1] << 2) & 0xFF);+ r[9 * i + 3] = (uint8_t)((t[1] >> 6) & 0xFF);+ r[9 * i + 4] = (uint8_t)((t[1] >> 14) & 0xFF);+ r[9 * i + 4] |= (uint8_t)((t[2] << 4) & 0xFF);+ r[9 * i + 5] = (uint8_t)((t[2] >> 4) & 0xFF);+ r[9 * i + 6] = (uint8_t)((t[2] >> 12) & 0xFF);+ r[9 * i + 6] |= (uint8_t)((t[3] << 6) & 0xFF);+ r[9 * i + 7] = (uint8_t)((t[3] >> 2) & 0xFF);+ r[9 * i + 8] = (uint8_t)((t[3] >> 10) & 0xFF);+ }+#else /* MLD_CONFIG_PARAMETER_SET == 44 */+ for (i = 0; i < MLDSA_N / 2; ++i)+ __loop__(+ invariant(i <= MLDSA_N/2)+ decreases(MLDSA_N / 2 - i))+ {+ /* Safety: a->coeffs[i] <= MLDSA_GAMMA1, hence, these casts are safe. */+ t[0] = (uint32_t)(MLDSA_GAMMA1 - a->coeffs[2 * i + 0]);+ t[1] = (uint32_t)(MLDSA_GAMMA1 - a->coeffs[2 * i + 1]);++ r[5 * i + 0] = (uint8_t)((t[0]) & 0xFF);+ r[5 * i + 1] = (uint8_t)((t[0] >> 8) & 0xFF);+ r[5 * i + 2] = (uint8_t)((t[0] >> 16) & 0xFF);+ r[5 * i + 2] |= (uint8_t)((t[1] << 4) & 0xFF);+ r[5 * i + 3] = (uint8_t)((t[1] >> 4) & 0xFF);+ r[5 * i + 4] = (uint8_t)((t[1] >> 12) & 0xFF);+ }+#endif /* MLD_CONFIG_PARAMETER_SET != 44 */+}+#endif /* !MLD_CONFIG_NO_SIGN_API */++#if !defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API)+MLD_STATIC_TESTABLE void mld_polyz_unpack_c(+ mld_poly *r, const uint8_t a[MLDSA_POLYZ_PACKEDBYTES])+__contract__(+ requires(memory_no_alias(r, sizeof(mld_poly)))+ requires(memory_no_alias(a, MLDSA_POLYZ_PACKEDBYTES))+ assigns(memory_slice(r, sizeof(mld_poly)))+ ensures(array_bound(r->coeffs, 0, MLDSA_N, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1))+)+{+ unsigned int i;+#if MLD_CONFIG_PARAMETER_SET == 44+ for (i = 0; i < MLDSA_N / 4; ++i)+ __loop__(+ invariant(i <= MLDSA_N/4)+ invariant(array_bound(r->coeffs, 0, i*4, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1))+ decreases(MLDSA_N / 4 - i))+ {+ r->coeffs[4 * i + 0] = a[9 * i + 0];+ r->coeffs[4 * i + 0] |= (int32_t)a[9 * i + 1] << 8;+ r->coeffs[4 * i + 0] |= (int32_t)a[9 * i + 2] << 16;+ r->coeffs[4 * i + 0] &= 0x3FFFF;++ r->coeffs[4 * i + 1] = a[9 * i + 2] >> 2;+ r->coeffs[4 * i + 1] |= (int32_t)a[9 * i + 3] << 6;+ r->coeffs[4 * i + 1] |= (int32_t)a[9 * i + 4] << 14;+ r->coeffs[4 * i + 1] &= 0x3FFFF;++ r->coeffs[4 * i + 2] = a[9 * i + 4] >> 4;+ r->coeffs[4 * i + 2] |= (int32_t)a[9 * i + 5] << 4;+ r->coeffs[4 * i + 2] |= (int32_t)a[9 * i + 6] << 12;+ r->coeffs[4 * i + 2] &= 0x3FFFF;++ r->coeffs[4 * i + 3] = a[9 * i + 6] >> 6;+ r->coeffs[4 * i + 3] |= (int32_t)a[9 * i + 7] << 2;+ r->coeffs[4 * i + 3] |= (int32_t)a[9 * i + 8] << 10;+ r->coeffs[4 * i + 3] &= 0x3FFFF;++ r->coeffs[4 * i + 0] = MLDSA_GAMMA1 - r->coeffs[4 * i + 0];+ r->coeffs[4 * i + 1] = MLDSA_GAMMA1 - r->coeffs[4 * i + 1];+ r->coeffs[4 * i + 2] = MLDSA_GAMMA1 - r->coeffs[4 * i + 2];+ r->coeffs[4 * i + 3] = MLDSA_GAMMA1 - r->coeffs[4 * i + 3];+ }+#else /* MLD_CONFIG_PARAMETER_SET == 44 */+ for (i = 0; i < MLDSA_N / 2; ++i)+ __loop__(+ invariant(i <= MLDSA_N/2)+ invariant(array_bound(r->coeffs, 0, i*2, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1))+ decreases(MLDSA_N / 2 - i))+ {+ r->coeffs[2 * i + 0] = a[5 * i + 0];+ r->coeffs[2 * i + 0] |= (int32_t)a[5 * i + 1] << 8;+ r->coeffs[2 * i + 0] |= (int32_t)a[5 * i + 2] << 16;+ r->coeffs[2 * i + 0] &= 0xFFFFF;++ r->coeffs[2 * i + 1] = a[5 * i + 2] >> 4;+ r->coeffs[2 * i + 1] |= (int32_t)a[5 * i + 3] << 4;+ r->coeffs[2 * i + 1] |= (int32_t)a[5 * i + 4] << 12;+ /* r->coeffs[2*i+1] &= 0xFFFFF; */ /* No effect, since we're anyway at 20+ bits */++ r->coeffs[2 * i + 0] = MLDSA_GAMMA1 - r->coeffs[2 * i + 0];+ r->coeffs[2 * i + 1] = MLDSA_GAMMA1 - r->coeffs[2 * i + 1];+ }+#endif /* MLD_CONFIG_PARAMETER_SET != 44 */+ mld_assert_bound(r->coeffs, MLDSA_N, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1);+}++MLD_INTERNAL_API+void mld_polyz_unpack(mld_poly *r, const uint8_t a[MLDSA_POLYZ_PACKEDBYTES])+{+#if defined(MLD_USE_NATIVE_POLYZ_UNPACK_17) && MLD_CONFIG_PARAMETER_SET == 44+ int ret;+ ret = mld_polyz_unpack_17_native(r->coeffs, a);+ if (ret == MLD_NATIVE_FUNC_SUCCESS)+ {+ mld_assert_bound(r->coeffs, MLDSA_N, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1);+ return;+ }+#elif defined(MLD_USE_NATIVE_POLYZ_UNPACK_19) && \+ (MLD_CONFIG_PARAMETER_SET == 65 || MLD_CONFIG_PARAMETER_SET == 87)+ int ret;+ ret = mld_polyz_unpack_19_native(r->coeffs, a);+ if (ret == MLD_NATIVE_FUNC_SUCCESS)+ {+ mld_assert_bound(r->coeffs, MLDSA_N, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1);+ return;+ }+#endif /* !(MLD_USE_NATIVE_POLYZ_UNPACK_17 && MLD_CONFIG_PARAMETER_SET == 44) \+ && MLD_USE_NATIVE_POLYZ_UNPACK_19 && (MLD_CONFIG_PARAMETER_SET == 65 \+ || MLD_CONFIG_PARAMETER_SET == 87) */++ mld_polyz_unpack_c(r, a);+}+#endif /* !MLD_CONFIG_NO_SIGN_API || !MLD_CONFIG_NO_VERIFY_API */++/* To facilitate single-compilation-unit (SCU) builds, undefine all macros. */++/* To facilitate single-compilation-unit (SCU) builds, undefine all macros.+ * Don't modify by hand -- this is auto-generated by scripts/autogen. */+#undef mld_rej_eta+#undef mld_rej_eta_c+#undef mld_poly_decompose_c+#undef mld_poly_use_hint_c+#undef mld_polyz_unpack_c+#undef MLD_POLY_UNIFORM_ETA_NBLOCKS+#undef MLD_POLY_UNIFORM_GAMMA1_NBLOCKS
@@ -0,0 +1,367 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS204]+ * FIPS 204 Module-Lattice-Based Digital Signature Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/204/final+ */++#ifndef MLD_POLY_KL_H+#define MLD_POLY_KL_H++#include "cbmc.h"+#include "common.h"+#include "poly.h"++#if !defined(MLD_CONFIG_NO_SIGN_API)+#define mld_poly_decompose MLD_NAMESPACE_KL(poly_decompose)+/**+ * For all coefficients c of the input polynomial, compute high and low bits+ * c0, c1 such c mod MLDSA_Q = c1*ALPHA + c0 with -ALPHA/2 < c0 <= ALPHA/2+ * except c1 = (MLDSA_Q-1)/ALPHA where we set c1 = 0 and+ * -ALPHA/2 <= c0 = c mod MLDSA_Q - MLDSA_Q < 0. Assumes coefficients to be+ * standard representatives.+ *+ * @reference{The reference implementation has the input polynomial as a+ * separate argument that may be aliased with either of the outputs. Removing+ * the aliasing eases CBMC proofs.}+ *+ * @param[out] a1 Pointer to output polynomial with coefficients c1.+ * @param[in,out] a0 Pointer to input/output polynomial. Output polynomial has+ * coefficients c0.+ */+MLD_INTERNAL_API+void mld_poly_decompose(mld_poly *a1, mld_poly *a0)+__contract__(+ requires(memory_no_alias(a1, sizeof(mld_poly)))+ requires(memory_no_alias(a0, sizeof(mld_poly)))+ requires(array_bound(a0->coeffs, 0, MLDSA_N, 0, MLDSA_Q))+ assigns(memory_slice(a1, sizeof(mld_poly)))+ assigns(memory_slice(a0, sizeof(mld_poly)))+ ensures(array_bound(a1->coeffs, 0, MLDSA_N, 0, (MLDSA_Q-1)/(2*MLDSA_GAMMA2)))+ ensures(array_abs_bound(a0->coeffs, 0, MLDSA_N, MLDSA_GAMMA2+1))+);++#endif /* !MLD_CONFIG_NO_SIGN_API */++#if !defined(MLD_CONFIG_NO_VERIFY_API)+#define mld_poly_use_hint MLD_NAMESPACE_KL(poly_use_hint)+/**+ * Use hint polynomial h to correct the high bits of a in-place.+ *+ * @param[in,out] a Input/output polynomial.+ * @param[in] h Hint polynomial.+ */+MLD_INTERNAL_API+void mld_poly_use_hint(mld_poly *a, const mld_poly *h)+__contract__(+ requires(memory_no_alias(a, sizeof(mld_poly)))+ requires(memory_no_alias(h, sizeof(mld_poly)))+ requires(array_bound(a->coeffs, 0, MLDSA_N, 0, MLDSA_Q))+ requires(array_bound(h->coeffs, 0, MLDSA_N, 0, 2))+ assigns(memory_slice(a, sizeof(mld_poly)))+ ensures(array_bound(a->coeffs, 0, MLDSA_N, 0, (MLDSA_Q-1)/(2*MLDSA_GAMMA2)))+);+#endif /* !MLD_CONFIG_NO_VERIFY_API */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API)+#if !defined(MLD_CONFIG_SERIAL_FIPS202_ONLY)+#define mld_poly_uniform_eta_4x MLD_NAMESPACE_KL(poly_uniform_eta_4x)+/**+ * Sample four polynomials with uniformly random coefficients in+ * [-MLDSA_ETA, MLDSA_ETA] by performing rejection sampling on the output+ * stream from SHAKE256(seed|nonce_i).+ *+ * @spec{Implements @[FIPS204, Algorithm 31, RejBoundedPoly] (four-way+ * batched).}+ *+ * @param[out] r0 Pointer to first output polynomial.+ * @param[out] r1 Pointer to second output polynomial.+ * @param[out] r2 Pointer to third output polynomial.+ * @param[out] r3 Pointer to fourth output polynomial.+ * @param[in] seed Byte array with seed of length MLDSA_CRHBYTES.+ * @param nonce0 First nonce.+ * @param nonce1 Second nonce.+ * @param nonce2 Third nonce.+ * @param nonce3 Fourth nonce.+ */+MLD_INTERNAL_API+void mld_poly_uniform_eta_4x(mld_poly *r0, mld_poly *r1, mld_poly *r2,+ mld_poly *r3, const uint8_t seed[MLDSA_CRHBYTES],+ uint8_t nonce0, uint8_t nonce1, uint8_t nonce2,+ uint8_t nonce3)+__contract__(+ requires(memory_no_alias(r0, sizeof(mld_poly)))+ requires(memory_no_alias(r1, sizeof(mld_poly)))+ requires(memory_no_alias(r2, sizeof(mld_poly)))+ requires(memory_no_alias(r3, sizeof(mld_poly)))+ requires(memory_no_alias(seed, MLDSA_CRHBYTES))+ assigns(memory_slice(r0, sizeof(mld_poly)))+ assigns(memory_slice(r1, sizeof(mld_poly)))+ assigns(memory_slice(r2, sizeof(mld_poly)))+ assigns(memory_slice(r3, sizeof(mld_poly)))+ ensures(array_abs_bound(r0->coeffs, 0, MLDSA_N, MLDSA_ETA + 1))+ ensures(array_abs_bound(r1->coeffs, 0, MLDSA_N, MLDSA_ETA + 1))+ ensures(array_abs_bound(r2->coeffs, 0, MLDSA_N, MLDSA_ETA + 1))+ ensures(array_abs_bound(r3->coeffs, 0, MLDSA_N, MLDSA_ETA + 1))+);+#endif /* !MLD_CONFIG_SERIAL_FIPS202_ONLY */++#if defined(MLD_CONFIG_SERIAL_FIPS202_ONLY)+#define mld_poly_uniform_eta MLD_NAMESPACE_KL(poly_uniform_eta)+/**+ * Sample polynomial with uniformly random coefficients in+ * [-MLDSA_ETA, MLDSA_ETA] by performing rejection sampling on the output+ * stream from SHAKE256(seed|nonce).+ *+ * @spec{Implements @[FIPS204, Algorithm 31, RejBoundedPoly].}+ *+ * @param[out] r Pointer to output polynomial.+ * @param[in] seed Byte array with seed of length MLDSA_CRHBYTES.+ * @param nonce Nonce.+ */+MLD_INTERNAL_API+void mld_poly_uniform_eta(mld_poly *r, const uint8_t seed[MLDSA_CRHBYTES],+ uint8_t nonce)+__contract__(+ requires(memory_no_alias(r, sizeof(mld_poly)))+ requires(memory_no_alias(seed, MLDSA_CRHBYTES))+ assigns(memory_slice(r, sizeof(mld_poly)))+ ensures(array_abs_bound(r->coeffs, 0, MLDSA_N, MLDSA_ETA + 1))+);+#endif /* MLD_CONFIG_SERIAL_FIPS202_ONLY */+#endif /* !MLD_CONFIG_NO_KEYPAIR_API */++#if !defined(MLD_CONFIG_NO_SIGN_API)+#if MLD_CONFIG_PARAMETER_SET == 65 || \+ defined(MLD_CONFIG_SERIAL_FIPS202_ONLY) || \+ defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)+#define mld_poly_uniform_gamma1 MLD_NAMESPACE_KL(poly_uniform_gamma1)+/**+ * Sample polynomial with uniformly random coefficients in+ * [-(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1] by unpacking output stream of+ * SHAKE256(seed|nonce).+ *+ * @spec{Partially implements @[FIPS204, Algorithm 34, ExpandMask] (one+ * polynomial, i.e. the loop body of lines 3-5).}+ *+ * @param[out] a Pointer to output polynomial.+ * @param[in] seed Byte array with seed of length MLDSA_CRHBYTES.+ * @param nonce 16-bit nonce.+ */+MLD_INTERNAL_API+void mld_poly_uniform_gamma1(mld_poly *a, const uint8_t seed[MLDSA_CRHBYTES],+ uint16_t nonce)+__contract__(+ requires(memory_no_alias(a, sizeof(mld_poly)))+ requires(memory_no_alias(seed, MLDSA_CRHBYTES))+ assigns(memory_slice(a, sizeof(mld_poly)))+ ensures(array_bound(a->coeffs, 0, MLDSA_N, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1))+);+#endif /* MLD_CONFIG_PARAMETER_SET == 65 || MLD_CONFIG_SERIAL_FIPS202_ONLY || \+ MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST */++#if !defined(MLD_CONFIG_SERIAL_FIPS202_ONLY) && \+ (!defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST))+#define mld_poly_uniform_gamma1_4x MLD_NAMESPACE_KL(poly_uniform_gamma1_4x)+/**+ * Sample four polynomials with uniformly random coefficients in+ * [-(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1] by unpacking output streams of+ * SHAKE256(seed|nonce_i).+ *+ * @spec{Partially implements @[FIPS204, Algorithm 34, ExpandMask] (four-way+ * batched, i.e. four iterations of the loop body of lines 3-5).}+ *+ * @param[out] r0 Pointer to first output polynomial.+ * @param[out] r1 Pointer to second output polynomial.+ * @param[out] r2 Pointer to third output polynomial.+ * @param[out] r3 Pointer to fourth output polynomial.+ * @param[in] seed Byte array with seed of length MLDSA_CRHBYTES.+ * @param nonce0 First 16-bit nonce.+ * @param nonce1 Second 16-bit nonce.+ * @param nonce2 Third 16-bit nonce.+ * @param nonce3 Fourth 16-bit nonce.+ */+MLD_INTERNAL_API+void mld_poly_uniform_gamma1_4x(mld_poly *r0, mld_poly *r1, mld_poly *r2,+ mld_poly *r3,+ const uint8_t seed[MLDSA_CRHBYTES],+ uint16_t nonce0, uint16_t nonce1,+ uint16_t nonce2, uint16_t nonce3)+__contract__(+ requires(memory_no_alias(r0, sizeof(mld_poly)))+ requires(memory_no_alias(r1, sizeof(mld_poly)))+ requires(memory_no_alias(r2, sizeof(mld_poly)))+ requires(memory_no_alias(r3, sizeof(mld_poly)))+ requires(memory_no_alias(seed, MLDSA_CRHBYTES))+ assigns(memory_slice(r0, sizeof(mld_poly)))+ assigns(memory_slice(r1, sizeof(mld_poly)))+ assigns(memory_slice(r2, sizeof(mld_poly)))+ assigns(memory_slice(r3, sizeof(mld_poly)))+ ensures(array_bound(r0->coeffs, 0, MLDSA_N, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1))+ ensures(array_bound(r1->coeffs, 0, MLDSA_N, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1))+ ensures(array_bound(r2->coeffs, 0, MLDSA_N, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1))+ ensures(array_bound(r3->coeffs, 0, MLDSA_N, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1))+);+#endif /* !MLD_CONFIG_SERIAL_FIPS202_ONLY && (!MLD_CONFIG_REDUCE_RAM || \+ MLD_UNIT_TEST) */+#endif /* !MLD_CONFIG_NO_SIGN_API */++#if !defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API)+#define mld_poly_challenge MLD_NAMESPACE_KL(poly_challenge)+/**+ * Samples polynomial with MLDSA_TAU nonzero coefficients in {-1, 1} using the+ * output stream of SHAKE256(seed).+ *+ * @spec{Implements @[FIPS204, Algorithm 29, SampleInBall].}+ *+ * @param[out] c Pointer to output polynomial.+ * @param[in] seed Byte array containing seed of length MLDSA_CTILDEBYTES.+ */+MLD_INTERNAL_API+void mld_poly_challenge(mld_poly *c, const uint8_t seed[MLDSA_CTILDEBYTES])+__contract__(+ requires(memory_no_alias(c, sizeof(mld_poly)))+ requires(memory_no_alias(seed, MLDSA_CTILDEBYTES))+ assigns(memory_slice(c, sizeof(mld_poly)))+ /* All coefficients of c are -1, 0 or +1 */+ ensures(array_bound(c->coeffs, 0, MLDSA_N, -1, 2))+);+#endif /* !MLD_CONFIG_NO_SIGN_API || !MLD_CONFIG_NO_VERIFY_API */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API)+#define mld_polyeta_pack MLD_NAMESPACE_KL(polyeta_pack)+/**+ * Bit-pack polynomial with coefficients in [-MLDSA_ETA, MLDSA_ETA].+ *+ * @spec{Implements @[FIPS204, Algorithm 17, BitPack].}+ *+ * @param[out] r Pointer to output byte array with at least+ * MLDSA_POLYETA_PACKEDBYTES bytes.+ * @param[in] a Pointer to input polynomial.+ */+MLD_INTERNAL_API+void mld_polyeta_pack(uint8_t r[MLDSA_POLYETA_PACKEDBYTES], const mld_poly *a)+__contract__(+ requires(memory_no_alias(r, MLDSA_POLYETA_PACKEDBYTES))+ requires(memory_no_alias(a, sizeof(mld_poly)))+ requires(array_abs_bound(a->coeffs, 0, MLDSA_N, MLDSA_ETA + 1))+ assigns(memory_slice(r, MLDSA_POLYETA_PACKEDBYTES))+);+#endif /* !MLD_CONFIG_NO_KEYPAIR_API */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API) || !defined(MLD_CONFIG_NO_SIGN_API)+/*+ * polyeta_unpack produces coefficients in [-MLDSA_ETA, MLDSA_ETA] for+ * well-formed inputs (i.e., those produced by polyeta_pack).+ * However, when passed an arbitrary byte array, it may produce smaller values,+ * i.e., values in [MLD_POLYETA_UNPACK_LOWER_BOUND, MLDSA_ETA].+ * Even though this should never happen, we use use the bound for arbitrary+ * inputs in the CBMC proofs.+ */+#if MLDSA_ETA == 2+#define MLD_POLYETA_UNPACK_LOWER_BOUND (-5)+#elif MLDSA_ETA == 4+#define MLD_POLYETA_UNPACK_LOWER_BOUND (-11)+#else+#error "Invalid value of MLDSA_ETA"+#endif++#define mld_polyeta_unpack MLD_NAMESPACE_KL(polyeta_unpack)+/**+ * Unpack polynomial with coefficients in [-MLDSA_ETA, MLDSA_ETA].+ *+ * @spec{Implements @[FIPS204, Algorithm 19, BitUnpack].}+ *+ * @param[out] r Pointer to output polynomial.+ * @param[in] a Byte array with bit-packed polynomial.+ */+MLD_INTERNAL_API+void mld_polyeta_unpack(mld_poly *r, const uint8_t a[MLDSA_POLYETA_PACKEDBYTES])+__contract__(+ requires(memory_no_alias(r, sizeof(mld_poly)))+ requires(memory_no_alias(a, MLDSA_POLYETA_PACKEDBYTES))+ assigns(memory_slice(r, sizeof(mld_poly)))+ ensures(array_bound(r->coeffs, 0, MLDSA_N, MLD_POLYETA_UNPACK_LOWER_BOUND, MLDSA_ETA + 1))+);+#endif /* !MLD_CONFIG_NO_KEYPAIR_API || !MLD_CONFIG_NO_SIGN_API */++#if !defined(MLD_CONFIG_NO_SIGN_API)+#define mld_polyz_pack MLD_NAMESPACE_KL(polyz_pack)+/**+ * Bit-pack polynomial with coefficients in+ * [-(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1].+ *+ * @spec{Implements @[FIPS204, Algorithm 17, BitPack].}+ *+ * @param[out] r Pointer to output byte array with at least+ * MLDSA_POLYZ_PACKEDBYTES bytes.+ * @param[in] a Pointer to input polynomial.+ */+MLD_INTERNAL_API+void mld_polyz_pack(uint8_t r[MLDSA_POLYZ_PACKEDBYTES], const mld_poly *a)+__contract__(+ requires(memory_no_alias(r, MLDSA_POLYZ_PACKEDBYTES))+ requires(memory_no_alias(a, sizeof(mld_poly)))+ requires(array_bound(a->coeffs, 0, MLDSA_N, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1))+ assigns(memory_slice(r, MLDSA_POLYZ_PACKEDBYTES))+);+#endif /* !MLD_CONFIG_NO_SIGN_API */++#if !defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API)+#define mld_polyz_unpack MLD_NAMESPACE_KL(polyz_unpack)+/**+ * Unpack polynomial z with coefficients in+ * [-(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1].+ *+ * @spec{Implements @[FIPS204, Algorithm 19, BitUnpack].}+ *+ * @param[out] r Pointer to output polynomial.+ * @param[in] a Byte array with bit-packed polynomial.+ */+MLD_INTERNAL_API+void mld_polyz_unpack(mld_poly *r, const uint8_t a[MLDSA_POLYZ_PACKEDBYTES])+__contract__(+ requires(memory_no_alias(r, sizeof(mld_poly)))+ requires(memory_no_alias(a, MLDSA_POLYZ_PACKEDBYTES))+ assigns(memory_slice(r, sizeof(mld_poly)))+ ensures(array_bound(r->coeffs, 0, MLDSA_N, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1))+);++#define mld_polyw1_pack MLD_NAMESPACE_KL(polyw1_pack)+/**+ * Bit-pack polynomial w1. Input coefficients must be in+ * [0, (MLDSA_Q-1)/(2*MLDSA_GAMMA2)), i.e. [0, 43] for ML-DSA-44 and [0, 15]+ * for ML-DSA-65/87. Dispatches to the value-specialized variant for the+ * selected parameter set.+ *+ * @spec{Implements @[FIPS204, Algorithm 16, SimpleBitPack].}+ *+ * @param[out] r Pointer to output byte array with at least+ * MLDSA_POLYW1_PACKEDBYTES bytes.+ * @param[in] a Pointer to input polynomial.+ */+static MLD_INLINE void mld_polyw1_pack(uint8_t r[MLDSA_POLYW1_PACKEDBYTES],+ const mld_poly *a)+__contract__(+ requires(memory_no_alias(r, MLDSA_POLYW1_PACKEDBYTES))+ requires(memory_no_alias(a, sizeof(mld_poly)))+ requires(array_bound(a->coeffs, 0, MLDSA_N, 0, (MLDSA_Q-1)/(2*MLDSA_GAMMA2)))+ assigns(memory_slice(r, MLDSA_POLYW1_PACKEDBYTES))+)+{+#if MLD_CONFIG_PARAMETER_SET == 44+ mld_polyw1_pack_88(r, a);+#else+ mld_polyw1_pack_32(r, a);+#endif+}+#endif /* !MLD_CONFIG_NO_SIGN_API || !MLD_CONFIG_NO_VERIFY_API */++#endif /* !MLD_POLY_KL_H */
@@ -0,0 +1,509 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS204]+ * FIPS 204 Module-Lattice-Based Digital Signature Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/204/final+ */++#include "polyvec.h"++#include "debug.h"+#include "polyvec_lazy.h"++/* This namespacing is not done at the top to avoid a naming conflict+ * with native backends, which are currently not yet namespaced. */+#define mld_polyvecl_pointwise_acc_montgomery_c \+ MLD_ADD_PARAM_SET(mld_polyvecl_pointwise_acc_montgomery_c)++/**************************************************************/+/************ Vectors of polynomials of length MLDSA_L **************/+/**************************************************************/+#if !defined(MLD_CONFIG_NO_SIGN_API) && \+ (!defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST))+MLD_INTERNAL_API+void mld_polyvecl_uniform_gamma1(mld_polyvecl *v,+ const uint8_t seed[MLDSA_CRHBYTES],+ uint16_t kappa)+{+#if defined(MLD_CONFIG_SERIAL_FIPS202_ONLY)+ int i;+#endif++ /* The caller passes the base counter kappa; component i is sampled from+ * kappa + i. Safety: kappa <= MLD_MAX_KAPPA and i < MLDSA_L, so the+ * casts below are safe. See MLD_MAX_KAPPA comment in params.h. */+#if defined(MLD_CONFIG_SERIAL_FIPS202_ONLY)+ for (i = 0; i < MLDSA_L; i++)+ {+ mld_poly_uniform_gamma1(&v->vec[i], seed, (uint16_t)(kappa + i));+ }+#else /* MLD_CONFIG_SERIAL_FIPS202_ONLY */+#if MLDSA_L == 4+ mld_poly_uniform_gamma1_4x(&v->vec[0], &v->vec[1], &v->vec[2], &v->vec[3],+ seed, kappa, (uint16_t)(kappa + 1),+ (uint16_t)(kappa + 2), (uint16_t)(kappa + 3));+#elif MLDSA_L == 5+ mld_poly_uniform_gamma1_4x(&v->vec[0], &v->vec[1], &v->vec[2], &v->vec[3],+ seed, kappa, (uint16_t)(kappa + 1),+ (uint16_t)(kappa + 2), (uint16_t)(kappa + 3));+ mld_poly_uniform_gamma1(&v->vec[4], seed, (uint16_t)(kappa + 4));+#elif MLDSA_L == 7+ mld_poly_uniform_gamma1_4x(&v->vec[0], &v->vec[1], &v->vec[2],+ &v->vec[3 /* irrelevant */], seed, kappa,+ (uint16_t)(kappa + 1), (uint16_t)(kappa + 2),+ 0xFF /* irrelevant */);+ mld_poly_uniform_gamma1_4x(&v->vec[3], &v->vec[4], &v->vec[5], &v->vec[6],+ seed, (uint16_t)(kappa + 3), (uint16_t)(kappa + 4),+ (uint16_t)(kappa + 5), (uint16_t)(kappa + 6));+#endif /* MLDSA_L == 7 */+#endif /* !MLD_CONFIG_SERIAL_FIPS202_ONLY */++ mld_assert_bound_2d(v->vec, MLDSA_L, MLDSA_N, -(MLDSA_GAMMA1 - 1),+ MLDSA_GAMMA1 + 1);+}+#endif /* !MLD_CONFIG_NO_SIGN_API && (!MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST) \+ */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API) || \+ !defined(MLD_CONFIG_NO_VERIFY_API) || \+ (!defined(MLD_CONFIG_NO_SIGN_API) && \+ (!defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)))+MLD_INTERNAL_API+void mld_polyvecl_ntt(mld_polyvecl *v)+{+ unsigned int i;+ mld_assert_abs_bound_2d(v->vec, MLDSA_L, MLDSA_N, MLDSA_Q);++ for (i = 0; i < MLDSA_L; ++i)+ __loop__(+ assigns(i, memory_slice(v, sizeof(mld_polyvecl)))+ invariant(i <= MLDSA_L)+ invariant(forall(k0, i, MLDSA_L, forall(k1, 0, MLDSA_N, v->vec[k0].coeffs[k1] == loop_entry(*v).vec[k0].coeffs[k1])))+ invariant(forall(k1, 0, i, array_abs_bound(v->vec[k1].coeffs, 0, MLDSA_N, MLD_NTT_BOUND)))+ decreases(MLDSA_L - i))+ {+ mld_poly_ntt(&v->vec[i]);+ }++ mld_assert_abs_bound_2d(v->vec, MLDSA_L, MLDSA_N, MLD_NTT_BOUND);+}+#endif /* !MLD_CONFIG_NO_KEYPAIR_API || !MLD_CONFIG_NO_VERIFY_API || \+ (!MLD_CONFIG_NO_SIGN_API && (!MLD_CONFIG_REDUCE_RAM || \+ MLD_UNIT_TEST)) */++#if !defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)+MLD_STATIC_TESTABLE void mld_polyvecl_pointwise_acc_montgomery_c(+ mld_poly *w, const mld_polyvecl *u, const mld_polyvecl *v)+__contract__(+ requires(memory_no_alias(w, sizeof(mld_poly)))+ requires(memory_no_alias(u, sizeof(mld_polyvecl)))+ requires(memory_no_alias(v, sizeof(mld_polyvecl)))+ requires(forall(l0, 0, MLDSA_L,+ array_bound(u->vec[l0].coeffs, 0, MLDSA_N, 0, MLDSA_Q)))+ requires(forall(l1, 0, MLDSA_L,+ array_abs_bound(v->vec[l1].coeffs, 0, MLDSA_N, MLD_NTT_BOUND)))+ assigns(memory_slice(w, sizeof(mld_poly)))+ ensures(array_abs_bound(w->coeffs, 0, MLDSA_N, MLDSA_Q))+)+{+ unsigned int i, j;+ mld_assert_bound_2d(u->vec, MLDSA_L, MLDSA_N, 0, MLDSA_Q);+ mld_assert_abs_bound_2d(v->vec, MLDSA_L, MLDSA_N, MLD_NTT_BOUND);+ for (i = 0; i < MLDSA_N; i++)+ __loop__(+ assigns(i, j, memory_slice(w, sizeof(mld_poly)))+ invariant(i <= MLDSA_N)+ invariant(array_abs_bound(w->coeffs, 0, i, MLDSA_Q))+ decreases(MLDSA_N - i)+ )+ {+ int64_t t = 0;+ int32_t r;+ for (j = 0; j < MLDSA_L; j++)+ __loop__(+ assigns(j, t)+ invariant(j <= MLDSA_L)+ invariant(t >= -(int64_t)j*(MLDSA_Q - 1)*(MLD_NTT_BOUND - 1))+ invariant(t <= (int64_t)j*(MLDSA_Q - 1)*(MLD_NTT_BOUND - 1))+ decreases(MLDSA_L - j)+ )+ {+ t += (int64_t)u->vec[j].coeffs[i] * v->vec[j].coeffs[i];+ }++ r = mld_montgomery_reduce(t);+ w->coeffs[i] = r;+ }++ mld_assert_abs_bound(w->coeffs, MLDSA_N, MLDSA_Q);+}++MLD_INTERNAL_API+void mld_polyvecl_pointwise_acc_montgomery(mld_poly *w, const mld_polyvecl *u,+ const mld_polyvecl *v)+{+#if defined(MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L4) && \+ MLD_CONFIG_PARAMETER_SET == 44+ int ret;+ mld_assert_bound_2d(u->vec, MLDSA_L, MLDSA_N, 0, MLDSA_Q);+ mld_assert_abs_bound_2d(v->vec, MLDSA_L, MLDSA_N, MLD_NTT_BOUND);+ ret = mld_polyvecl_pointwise_acc_montgomery_l4_native(+ w->coeffs, (const int32_t (*)[MLDSA_N])u->vec,+ (const int32_t (*)[MLDSA_N])v->vec);+ if (ret == MLD_NATIVE_FUNC_SUCCESS)+ {+ mld_assert_abs_bound(w->coeffs, MLDSA_N, MLDSA_Q);+ return;+ }+#elif defined(MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L5) && \+ MLD_CONFIG_PARAMETER_SET == 65+ int ret;+ mld_assert_bound_2d(u->vec, MLDSA_L, MLDSA_N, 0, MLDSA_Q);+ mld_assert_abs_bound_2d(v->vec, MLDSA_L, MLDSA_N, MLD_NTT_BOUND);+ ret = mld_polyvecl_pointwise_acc_montgomery_l5_native(+ w->coeffs, (const int32_t (*)[MLDSA_N])u->vec,+ (const int32_t (*)[MLDSA_N])v->vec);+ if (ret == MLD_NATIVE_FUNC_SUCCESS)+ {+ mld_assert_abs_bound(w->coeffs, MLDSA_N, MLDSA_Q);+ return;+ }+#elif defined(MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L7) && \+ MLD_CONFIG_PARAMETER_SET == 87+ int ret;+ mld_assert_bound_2d(u->vec, MLDSA_L, MLDSA_N, 0, MLDSA_Q);+ mld_assert_abs_bound_2d(v->vec, MLDSA_L, MLDSA_N, MLD_NTT_BOUND);+ ret = mld_polyvecl_pointwise_acc_montgomery_l7_native(+ w->coeffs, (const int32_t (*)[MLDSA_N])u->vec,+ (const int32_t (*)[MLDSA_N])v->vec);+ if (ret == MLD_NATIVE_FUNC_SUCCESS)+ {+ mld_assert_abs_bound(w->coeffs, MLDSA_N, MLDSA_Q);+ return;+ }+#endif /* !(MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L4 && \+ MLD_CONFIG_PARAMETER_SET == 44) && \+ !(MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L5 && \+ MLD_CONFIG_PARAMETER_SET == 65) && \+ MLD_USE_NATIVE_POLYVECL_POINTWISE_ACC_MONTGOMERY_L7 && \+ MLD_CONFIG_PARAMETER_SET == 87 */+ /* The first input is bounded by [0, MLDSA_Q-1] inclusive.+ * The second input is bounded by [-(MLD_NTT_BOUND-1), MLD_NTT_BOUND-1].+ * Hence, we can safely accumulate in 64-bits without intermediate reductions+ * as MLDSA_L * (MLD_NTT_BOUND-1) * (MLDSA_Q-1) < INT64_MAX.+ *+ * The worst case is ML-DSA-87: 7 * (MLD_NTT_BOUND-1) * (MLDSA_Q-1) < 2**53+ * (and likewise for negative values).+ */+ mld_polyvecl_pointwise_acc_montgomery_c(w, u, v);+}+#endif /* !MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API) || !defined(MLD_CONFIG_NO_VERIFY_API) || \+ defined(MLD_UNIT_TEST)+MLD_INTERNAL_API+uint32_t mld_polyvecl_chknorm(const mld_polyvecl *v, int32_t bound)+{+ unsigned int i;+ uint32_t t = 0;+ mld_assert_bound_2d(v->vec, MLDSA_L, MLDSA_N, -MLD_REDUCE32_RANGE_MAX,+ MLD_REDUCE32_RANGE_MAX);++ for (i = 0; i < MLDSA_L; ++i)+ __loop__(+ invariant(i <= MLDSA_L)+ invariant(t == 0 || t == 0xFFFFFFFF)+ invariant((t == 0) == forall(k1, 0, i, array_abs_bound(v->vec[k1].coeffs, 0, MLDSA_N, bound)))+ decreases(MLDSA_L - i)+ )+ {+ /* Reference: Leaks which polynomial violates the bound via a conditional.+ * We are more conservative to reduce the number of declassifications in+ * constant-time testing.+ */+ t |= mld_poly_chknorm(&v->vec[i], bound);+ }+ return t;+}+#endif /* !MLD_CONFIG_NO_KEYPAIR_API || !MLD_CONFIG_NO_VERIFY_API || \+ MLD_UNIT_TEST */++/**************************************************************/+/************ Vectors of polynomials of length MLDSA_K **************/+/**************************************************************/+#if (!defined(MLD_CONFIG_NO_SIGN_API) && \+ defined(MLD_CONFIG_REDUCE_RAM)) || \+ defined(MLD_UNIT_TEST)+MLD_INTERNAL_API+void mld_polyveck_reduce(mld_polyveck *v)+{+ unsigned int i;+ mld_assert_bound_2d(v->vec, MLDSA_K, MLDSA_N, INT32_MIN,+ MLD_REDUCE32_DOMAIN_MAX);++ for (i = 0; i < MLDSA_K; ++i)+ __loop__(+ assigns(i, memory_slice(v, sizeof(mld_polyveck)))+ invariant(i <= MLDSA_K)+ invariant(forall(k0, i, MLDSA_K, forall(k1, 0, MLDSA_N, v->vec[k0].coeffs[k1] == loop_entry(*v).vec[k0].coeffs[k1])))+ invariant(forall(k2, 0, i,+ array_bound(v->vec[k2].coeffs, 0, MLDSA_N, -MLD_REDUCE32_RANGE_MAX, MLD_REDUCE32_RANGE_MAX)))+ decreases(MLDSA_K - i)+ )+ {+ mld_poly_reduce(&v->vec[i]);+ }++ mld_assert_bound_2d(v->vec, MLDSA_K, MLDSA_N, -MLD_REDUCE32_RANGE_MAX,+ MLD_REDUCE32_RANGE_MAX);+}+#endif /* (!MLD_CONFIG_NO_SIGN_API && MLD_CONFIG_REDUCE_RAM) || MLD_UNIT_TEST \+ */++#if !defined(MLD_CONFIG_NO_SIGN_API) || defined(MLD_UNIT_TEST)+MLD_INTERNAL_API+void mld_polyveck_caddq(mld_polyveck *v)+{+ unsigned int i;+ mld_assert_abs_bound_2d(v->vec, MLDSA_K, MLDSA_N, MLDSA_Q);++ for (i = 0; i < MLDSA_K; ++i)+ __loop__(+ assigns(i, memory_slice(v, sizeof(mld_polyveck)))+ invariant(i <= MLDSA_K)+ invariant(forall(k0, i, MLDSA_K, forall(k1, 0, MLDSA_N, v->vec[k0].coeffs[k1] == loop_entry(*v).vec[k0].coeffs[k1])))+ invariant(forall(k1, 0, i, array_bound(v->vec[k1].coeffs, 0, MLDSA_N, 0, MLDSA_Q)))+ decreases(MLDSA_K - i))+ {+ mld_poly_caddq(&v->vec[i]);+ }++ mld_assert_bound_2d(v->vec, MLDSA_K, MLDSA_N, 0, MLDSA_Q);+}+#endif /* !MLD_CONFIG_NO_SIGN_API || MLD_UNIT_TEST */++#if (!defined(MLD_CONFIG_NO_SIGN_API) || defined(MLD_UNIT_TEST)) && \+ (!defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST))+MLD_INTERNAL_API+void mld_polyveck_ntt(mld_polyveck *v)+{+ unsigned int i;+ mld_assert_abs_bound_2d(v->vec, MLDSA_K, MLDSA_N, MLDSA_Q);++ for (i = 0; i < MLDSA_K; ++i)+ __loop__(+ assigns(i, memory_slice(v, sizeof(mld_polyveck)))+ invariant(i <= MLDSA_K)+ invariant(forall(k0, i, MLDSA_K, forall(k1, 0, MLDSA_N, v->vec[k0].coeffs[k1] == loop_entry(*v).vec[k0].coeffs[k1])))+ invariant(forall(k1, 0, i, array_abs_bound(v->vec[k1].coeffs, 0, MLDSA_N, MLD_NTT_BOUND)))+ decreases(MLDSA_K - i))+ {+ mld_poly_ntt(&v->vec[i]);+ }+ mld_assert_abs_bound_2d(v->vec, MLDSA_K, MLDSA_N, MLD_NTT_BOUND);+}+#endif /* (!MLD_CONFIG_NO_SIGN_API || MLD_UNIT_TEST) && \+ (!MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST) */++#if !defined(MLD_CONFIG_NO_SIGN_API) || defined(MLD_UNIT_TEST)+MLD_INTERNAL_API+void mld_polyveck_invntt_tomont(mld_polyveck *v)+{+ unsigned int i;+ mld_assert_abs_bound_2d(v->vec, MLDSA_K, MLDSA_N, MLDSA_Q);++ for (i = 0; i < MLDSA_K; ++i)+ __loop__(+ assigns(i, memory_slice(v, sizeof(mld_polyveck)))+ invariant(i <= MLDSA_K)+ invariant(forall(k0, i, MLDSA_K, forall(k1, 0, MLDSA_N, v->vec[k0].coeffs[k1] == loop_entry(*v).vec[k0].coeffs[k1])))+ invariant(forall(k1, 0, i, array_abs_bound(v->vec[k1].coeffs, 0, MLDSA_N, MLD_INTT_BOUND)))+ decreases(MLDSA_K - i))+ {+ mld_poly_invntt_tomont(&v->vec[i]);+ }++ mld_assert_abs_bound_2d(v->vec, MLDSA_K, MLDSA_N, MLD_INTT_BOUND);+}+#endif /* !MLD_CONFIG_NO_SIGN_API || MLD_UNIT_TEST */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API)+MLD_INTERNAL_API+uint32_t mld_polyveck_chknorm(const mld_polyveck *v, int32_t bound)+{+ unsigned int i;+ uint32_t t = 0;+ mld_assert_bound_2d(v->vec, MLDSA_K, MLDSA_N, -MLD_REDUCE32_RANGE_MAX,+ MLD_REDUCE32_RANGE_MAX);++ for (i = 0; i < MLDSA_K; ++i)+ __loop__(+ invariant(i <= MLDSA_K)+ invariant(t == 0 || t == 0xFFFFFFFF)+ invariant((t == 0) == forall(k1, 0, i, array_abs_bound(v->vec[k1].coeffs, 0, MLDSA_N, bound)))+ decreases(MLDSA_K - i)+ )+ {+ /* Reference: Leaks which polynomial violates the bound via a conditional.+ * We are more conservative to reduce the number of declassifications in+ * constant-time testing.+ */+ t |= mld_poly_chknorm(&v->vec[i], bound);+ }++ return t;+}++#endif /* !MLD_CONFIG_NO_KEYPAIR_API */++#if !defined(MLD_CONFIG_NO_SIGN_API)+MLD_INTERNAL_API+void mld_polyveck_decompose(mld_polyveck *v1, mld_polyveck *v0)+{+ unsigned int i;+ mld_assert_bound_2d(v0->vec, MLDSA_K, MLDSA_N, 0, MLDSA_Q);++ for (i = 0; i < MLDSA_K; ++i)+ __loop__(+ assigns(i, memory_slice(v0, sizeof(mld_polyveck)), memory_slice(v1, sizeof(mld_polyveck)))+ invariant(i <= MLDSA_K)+ invariant(forall(k1, 0, i,+ array_bound(v1->vec[k1].coeffs, 0, MLDSA_N, 0, (MLDSA_Q-1)/(2*MLDSA_GAMMA2))))+ invariant(forall(k2, 0, i,+ array_abs_bound(v0->vec[k2].coeffs, 0, MLDSA_N, MLDSA_GAMMA2+1)))+ invariant(forall(k3, i, MLDSA_K,+ array_bound(v0->vec[k3].coeffs, 0, MLDSA_N, 0, MLDSA_Q)))+ decreases(MLDSA_K - i)+ )+ {+ mld_poly_decompose(&v1->vec[i], &v0->vec[i]);+ }++ mld_assert_bound_2d(v1->vec, MLDSA_K, MLDSA_N, 0,+ (MLDSA_Q - 1) / (2 * MLDSA_GAMMA2));+ mld_assert_abs_bound_2d(v0->vec, MLDSA_K, MLDSA_N, MLDSA_GAMMA2 + 1);+}+#endif /* !MLD_CONFIG_NO_SIGN_API */++#if !defined(MLD_CONFIG_NO_SIGN_API)+MLD_INTERNAL_API+void mld_polyveck_pack_w1(uint8_t r[MLDSA_K * MLDSA_POLYW1_PACKEDBYTES],+ const mld_polyveck *w1)+{+ unsigned int i;+ mld_assert_bound_2d(w1->vec, MLDSA_K, MLDSA_N, 0,+ (MLDSA_Q - 1) / (2 * MLDSA_GAMMA2));++ for (i = 0; i < MLDSA_K; ++i)+ __loop__(+ assigns(i, memory_slice(r, MLDSA_K * MLDSA_POLYW1_PACKEDBYTES))+ invariant(i <= MLDSA_K)+ decreases(MLDSA_K - i)+ )+ {+ mld_polyw1_pack(&r[i * MLDSA_POLYW1_PACKEDBYTES], &w1->vec[i]);+ }+}+#endif /* !MLD_CONFIG_NO_SIGN_API */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API)+MLD_INTERNAL_API+void mld_polyveck_pack_eta(uint8_t r[MLDSA_K * MLDSA_POLYETA_PACKEDBYTES],+ const mld_polyveck *p)+{+ unsigned int i;+ mld_assert_abs_bound_2d(p->vec, MLDSA_K, MLDSA_N, MLDSA_ETA + 1);+ for (i = 0; i < MLDSA_K; ++i)+ __loop__(+ assigns(i, memory_slice(r, MLDSA_K * MLDSA_POLYETA_PACKEDBYTES))+ invariant(i <= MLDSA_K)+ decreases(MLDSA_K - i)+ )+ {+ mld_polyeta_pack(&r[i * MLDSA_POLYETA_PACKEDBYTES], &p->vec[i]);+ }+}++MLD_INTERNAL_API+void mld_polyvecl_pack_eta(uint8_t r[MLDSA_L * MLDSA_POLYETA_PACKEDBYTES],+ const mld_polyvecl *p)+{+ unsigned int i;+ mld_assert_abs_bound_2d(p->vec, MLDSA_L, MLDSA_N, MLDSA_ETA + 1);+ for (i = 0; i < MLDSA_L; ++i)+ __loop__(+ assigns(i, memory_slice(r, MLDSA_L * MLDSA_POLYETA_PACKEDBYTES))+ invariant(i <= MLDSA_L)+ decreases(MLDSA_L - i)+ )+ {+ mld_polyeta_pack(&r[i * MLDSA_POLYETA_PACKEDBYTES], &p->vec[i]);+ }+}++#endif /* !MLD_CONFIG_NO_KEYPAIR_API */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API) || \+ (!defined(MLD_CONFIG_NO_SIGN_API) && \+ (!defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)))+MLD_INTERNAL_API+void mld_polyvecl_unpack_eta(+ mld_polyvecl *p, const uint8_t r[MLDSA_L * MLDSA_POLYETA_PACKEDBYTES])+{+ unsigned int i;+ for (i = 0; i < MLDSA_L; ++i)+ {+ mld_polyeta_unpack(&p->vec[i], r + i * MLDSA_POLYETA_PACKEDBYTES);+ }++ mld_assert_bound_2d(p->vec, MLDSA_L, MLDSA_N, MLD_POLYETA_UNPACK_LOWER_BOUND,+ MLDSA_ETA + 1);+}+#endif /* !MLD_CONFIG_NO_KEYPAIR_API || (!MLD_CONFIG_NO_SIGN_API && \+ (!MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST)) */++#if !defined(MLD_CONFIG_NO_VERIFY_API)+MLD_INTERNAL_API+void mld_polyvecl_unpack_z(mld_polyvecl *z,+ const uint8_t r[MLDSA_L * MLDSA_POLYZ_PACKEDBYTES])+{+ unsigned int i;+ for (i = 0; i < MLDSA_L; ++i)+ {+ mld_polyz_unpack(&z->vec[i], r + i * MLDSA_POLYZ_PACKEDBYTES);+ }++ mld_assert_bound_2d(z->vec, MLDSA_L, MLDSA_N, -(MLDSA_GAMMA1 - 1),+ MLDSA_GAMMA1 + 1);+}+#endif /* !MLD_CONFIG_NO_VERIFY_API */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API) || \+ (!defined(MLD_CONFIG_NO_SIGN_API) && \+ (!defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)))+MLD_INTERNAL_API+void mld_polyveck_unpack_eta(+ mld_polyveck *p, const uint8_t r[MLDSA_K * MLDSA_POLYETA_PACKEDBYTES])+{+ unsigned int i;+ for (i = 0; i < MLDSA_K; ++i)+ {+ mld_polyeta_unpack(&p->vec[i], r + i * MLDSA_POLYETA_PACKEDBYTES);+ }++ mld_assert_bound_2d(p->vec, MLDSA_K, MLDSA_N, MLD_POLYETA_UNPACK_LOWER_BOUND,+ MLDSA_ETA + 1);+}+#endif /* !MLD_CONFIG_NO_KEYPAIR_API || (!MLD_CONFIG_NO_SIGN_API && \+ (!MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST)) */++/* To facilitate single-compilation-unit (SCU) builds, undefine all macros.+ * Don't modify by hand -- this is auto-generated by scripts/autogen. */+#undef mld_polyvecl_pointwise_acc_montgomery_c
@@ -0,0 +1,435 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS204]+ * FIPS 204 Module-Lattice-Based Digital Signature Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/204/final+ */++#ifndef MLD_POLYVEC_H+#define MLD_POLYVEC_H++#include "cbmc.h"+#include "common.h"+#include "poly.h"+#include "poly_kl.h"++/* Parameter set namespacing+ * This is to facilitate building multiple instances+ * of mldsa-native (e.g. with varying parameter sets)+ * within a single compilation unit. */+#define mld_polyvecl MLD_ADD_PARAM_SET(mld_polyvecl)+#define mld_polyveck MLD_ADD_PARAM_SET(mld_polyveck)+/* End of parameter set namespacing */++/** Vector of MLDSA_L polynomials. */+typedef struct+{+ mld_poly vec[MLDSA_L]; /**< Component polynomials. */+} mld_polyvecl;+++#if !defined(MLD_CONFIG_NO_SIGN_API) && \+ (!defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST))+#define mld_polyvecl_uniform_gamma1 MLD_NAMESPACE_KL(polyvecl_uniform_gamma1)+/**+ * Sample vector of polynomials with uniformly random coefficients in+ * [-(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1] by unpacking output stream of+ * SHAKE256(seed|kappa+i) for component i.+ *+ * @spec{Implements @[FIPS204, Algorithm 34, ExpandMask].}+ *+ * @param[out] v Pointer to output vector.+ * @param[in] seed Byte array with seed of length MLDSA_CRHBYTES.+ * @param kappa Base counter; component i uses kappa + i.+ */+MLD_INTERNAL_API+void mld_polyvecl_uniform_gamma1(mld_polyvecl *v,+ const uint8_t seed[MLDSA_CRHBYTES],+ uint16_t kappa)+__contract__(+ requires(memory_no_alias(v, sizeof(mld_polyvecl)))+ requires(memory_no_alias(seed, MLDSA_CRHBYTES))+ requires(kappa <= MLD_MAX_KAPPA)+ assigns(memory_slice(v, sizeof(mld_polyvecl)))+ ensures(forall(k0, 0, MLDSA_L,+ array_bound(v->vec[k0].coeffs, 0, MLDSA_N, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1)))+);+#endif /* !MLD_CONFIG_NO_SIGN_API && (!MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST) \+ */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API) || \+ !defined(MLD_CONFIG_NO_VERIFY_API) || \+ (!defined(MLD_CONFIG_NO_SIGN_API) && \+ (!defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)))+#define mld_polyvecl_ntt MLD_NAMESPACE_KL(polyvecl_ntt)+/**+ * Forward NTT of all polynomials in vector of length MLDSA_L. Output+ * coefficients are bounded by MLD_NTT_BOUND in absolute value.+ *+ * @param[in,out] v Pointer to input/output vector.+ */+MLD_INTERNAL_API+void mld_polyvecl_ntt(mld_polyvecl *v)+__contract__(+ requires(memory_no_alias(v, sizeof(mld_polyvecl)))+ requires(forall(k0, 0, MLDSA_L, array_abs_bound(v->vec[k0].coeffs, 0, MLDSA_N, MLDSA_Q)))+ assigns(memory_slice(v, sizeof(mld_polyvecl)))+ ensures(forall(k1, 0, MLDSA_L, array_abs_bound(v->vec[k1].coeffs, 0, MLDSA_N, MLD_NTT_BOUND)))+);+#endif /* !MLD_CONFIG_NO_KEYPAIR_API || !MLD_CONFIG_NO_VERIFY_API || \+ (!MLD_CONFIG_NO_SIGN_API && (!MLD_CONFIG_REDUCE_RAM || \+ MLD_UNIT_TEST)) */++#if !defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)+#define mld_polyvecl_pointwise_acc_montgomery \+ MLD_NAMESPACE_KL(polyvecl_pointwise_acc_montgomery)+/**+ * Pointwise multiply vectors of polynomials of length MLDSA_L, multiply+ * resulting vector by 2^{-32} and add (accumulate) polynomials in it.+ * Input/output vectors are in NTT domain representation.+ *+ * The first input "u" must be the output of polyvec_matrix_expand() and so+ * have coefficients in [0, MLDSA_Q-1] inclusive.+ *+ * The second input "v" is assumed to be output of an NTT, and hence must have+ * coefficients bounded by [-(MLD_NTT_BOUND-1), MLD_NTT_BOUND-1] inclusive.+ *+ * @spec{Partially implements @[FIPS204, Algorithm 48, MatrixVectorNTT]+ * (one output polynomial; multiply-accumulate of two NTT-domain vectors).}+ *+ * @param[out] w Output polynomial.+ * @param[in] u Pointer to first input vector.+ * @param[in] v Pointer to second input vector.+ */+MLD_INTERNAL_API+void mld_polyvecl_pointwise_acc_montgomery(mld_poly *w, const mld_polyvecl *u,+ const mld_polyvecl *v)+__contract__(+ requires(memory_no_alias(w, sizeof(mld_poly)))+ requires(memory_no_alias(u, sizeof(mld_polyvecl)))+ requires(memory_no_alias(v, sizeof(mld_polyvecl)))+ requires(forall(l0, 0, MLDSA_L,+ array_bound(u->vec[l0].coeffs, 0, MLDSA_N, 0, MLDSA_Q)))+ requires(forall(l1, 0, MLDSA_L,+ array_abs_bound(v->vec[l1].coeffs, 0, MLDSA_N, MLD_NTT_BOUND)))+ assigns(memory_slice(w, sizeof(mld_poly)))+ ensures(array_abs_bound(w->coeffs, 0, MLDSA_N, MLDSA_Q))+);+#endif /* !MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API) || !defined(MLD_CONFIG_NO_VERIFY_API)+#define mld_polyvecl_chknorm MLD_NAMESPACE_KL(polyvecl_chknorm)+/**+ * Check infinity norm of polynomials in vector of length MLDSA_L. Assumes+ * input mld_polyvecl to be reduced by polyvecl_reduce().+ *+ * @param[in] v Pointer to vector.+ * @param B Norm bound.+ *+ * @return 0 if norm of all polynomials is strictly smaller than+ * B <= (MLDSA_Q-1)/8 and 0xFFFFFFFF otherwise.+ */+MLD_INTERNAL_API+MLD_MUST_CHECK_RETURN_VALUE+uint32_t mld_polyvecl_chknorm(const mld_polyvecl *v, int32_t B)+__contract__(+ requires(memory_no_alias(v, sizeof(mld_polyvecl)))+ requires(0 <= B && B <= (MLDSA_Q - 1) / 8)+ requires(forall(k0, 0, MLDSA_L,+ array_bound(v->vec[k0].coeffs, 0, MLDSA_N, -MLD_REDUCE32_RANGE_MAX, MLD_REDUCE32_RANGE_MAX)))+ ensures(return_value == 0 || return_value == 0xFFFFFFFF)+ ensures((return_value == 0) == forall(k1, 0, MLDSA_L, array_abs_bound(v->vec[k1].coeffs, 0, MLDSA_N, B)))+);+#endif /* !MLD_CONFIG_NO_KEYPAIR_API || !MLD_CONFIG_NO_VERIFY_API */++/** Vector of MLDSA_K polynomials. */+typedef struct+{+ mld_poly vec[MLDSA_K]; /**< Component polynomials. */+} mld_polyveck;++#if (!defined(MLD_CONFIG_NO_SIGN_API) && defined(MLD_CONFIG_REDUCE_RAM)) || \+ defined(MLD_UNIT_TEST)+#define mld_polyveck_reduce MLD_NAMESPACE_KL(polyveck_reduce)+/**+ * Reduce coefficients of polynomials in vector of length MLDSA_K to+ * representatives in [-MLD_REDUCE32_RANGE_MAX, MLD_REDUCE32_RANGE_MAX].+ *+ * @param[in,out] v Pointer to input/output vector.+ */+MLD_INTERNAL_API+void mld_polyveck_reduce(mld_polyveck *v)+__contract__(+ requires(memory_no_alias(v, sizeof(mld_polyveck)))+ requires(forall(k0, 0, MLDSA_K,+ array_bound(v->vec[k0].coeffs, 0, MLDSA_N, INT32_MIN, MLD_REDUCE32_DOMAIN_MAX)))+ assigns(memory_slice(v, sizeof(mld_polyveck)))+ ensures(forall(k1, 0, MLDSA_K,+ array_bound(v->vec[k1].coeffs, 0, MLDSA_N, -MLD_REDUCE32_RANGE_MAX, MLD_REDUCE32_RANGE_MAX)))+);+#endif /* (!MLD_CONFIG_NO_SIGN_API && MLD_CONFIG_REDUCE_RAM) || MLD_UNIT_TEST \+ */++#if !defined(MLD_CONFIG_NO_SIGN_API) || defined(MLD_UNIT_TEST)+#define mld_polyveck_caddq MLD_NAMESPACE_KL(polyveck_caddq)+/**+ * For all coefficients of polynomials in vector of length MLDSA_K add MLDSA_Q+ * if coefficient is negative.+ *+ * @param[in,out] v Pointer to input/output vector.+ */+MLD_INTERNAL_API+void mld_polyveck_caddq(mld_polyveck *v)+__contract__(+ requires(memory_no_alias(v, sizeof(mld_polyveck)))+ requires(forall(k0, 0, MLDSA_K,+ array_abs_bound(v->vec[k0].coeffs, 0, MLDSA_N, MLDSA_Q)))+ assigns(memory_slice(v, sizeof(mld_polyveck)))+ ensures(forall(k1, 0, MLDSA_K,+ array_bound(v->vec[k1].coeffs, 0, MLDSA_N, 0, MLDSA_Q)))+);+#endif /* !MLD_CONFIG_NO_SIGN_API || MLD_UNIT_TEST */++#if (!defined(MLD_CONFIG_NO_SIGN_API) || defined(MLD_UNIT_TEST)) && \+ (!defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST))+#define mld_polyveck_ntt MLD_NAMESPACE_KL(polyveck_ntt)+/**+ * Forward NTT of all polynomials in vector of length MLDSA_K. Output+ * coefficients are bounded by MLD_NTT_BOUND in absolute value.+ *+ * @param[in,out] v Pointer to input/output vector.+ */+MLD_INTERNAL_API+void mld_polyveck_ntt(mld_polyveck *v)+__contract__(+ requires(memory_no_alias(v, sizeof(mld_polyveck)))+ requires(forall(k0, 0, MLDSA_K, array_abs_bound(v->vec[k0].coeffs, 0, MLDSA_N, MLDSA_Q)))+ assigns(memory_slice(v, sizeof(mld_polyveck)))+ ensures(forall(k1, 0, MLDSA_K, array_abs_bound(v->vec[k1].coeffs, 0, MLDSA_N, MLD_NTT_BOUND)))+);+#endif /* (!MLD_CONFIG_NO_SIGN_API || MLD_UNIT_TEST) && \+ (!MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST) */++#if !defined(MLD_CONFIG_NO_SIGN_API) || defined(MLD_UNIT_TEST)+#define mld_polyveck_invntt_tomont MLD_NAMESPACE_KL(polyveck_invntt_tomont)+/**+ * Inverse NTT and multiplication by 2^{32} of polynomials in vector of+ * length MLDSA_K.+ *+ * Input coefficients need to be less than MLDSA_Q, and output coefficients+ * are bounded by MLD_INTT_BOUND.+ *+ * @param[in,out] v Pointer to input/output vector.+ */+MLD_INTERNAL_API+void mld_polyveck_invntt_tomont(mld_polyveck *v)+__contract__(+ requires(memory_no_alias(v, sizeof(mld_polyveck)))+ requires(forall(k0, 0, MLDSA_K, array_abs_bound(v->vec[k0].coeffs, 0, MLDSA_N, MLDSA_Q)))+ assigns(memory_slice(v, sizeof(mld_polyveck)))+ ensures(forall(k1, 0, MLDSA_K, array_abs_bound(v->vec[k1].coeffs, 0, MLDSA_N, MLD_INTT_BOUND)))+);+#endif /* !MLD_CONFIG_NO_SIGN_API || MLD_UNIT_TEST */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API)+#define mld_polyveck_chknorm MLD_NAMESPACE_KL(polyveck_chknorm)+/**+ * Check infinity norm of polynomials in vector of length MLDSA_K. Assumes+ * input mld_polyveck to be reduced by polyveck_reduce().+ *+ * @param[in] v Pointer to vector.+ * @param B Norm bound.+ *+ * @return 0 if norm of all polynomials are strictly smaller than+ * B <= (MLDSA_Q-1)/8 and 0xFFFFFFFF otherwise.+ */+MLD_INTERNAL_API+MLD_MUST_CHECK_RETURN_VALUE+uint32_t mld_polyveck_chknorm(const mld_polyveck *v, int32_t B)+__contract__(+ requires(memory_no_alias(v, sizeof(mld_polyveck)))+ requires(0 <= B && B <= (MLDSA_Q - 1) / 8)+ requires(forall(k0, 0, MLDSA_K,+ array_bound(v->vec[k0].coeffs, 0, MLDSA_N,+ -MLD_REDUCE32_RANGE_MAX, MLD_REDUCE32_RANGE_MAX)))+ ensures(return_value == 0 || return_value == 0xFFFFFFFF)+ ensures((return_value == 0) == forall(k1, 0, MLDSA_K, array_abs_bound(v->vec[k1].coeffs, 0, MLDSA_N, B)))+);++#endif /* !MLD_CONFIG_NO_KEYPAIR_API */++#if !defined(MLD_CONFIG_NO_SIGN_API)+#define mld_polyveck_decompose MLD_NAMESPACE_KL(polyveck_decompose)+/**+ * For all coefficients a of polynomials in vector of length MLDSA_K, compute+ * high and low bits a0, a1 such a mod^+ MLDSA_Q = a1*ALPHA + a0 with+ * -ALPHA/2 < a0 <= ALPHA/2 except a1 = (MLDSA_Q-1)/ALPHA where we set+ * a1 = 0 and -ALPHA/2 <= a0 = a mod MLDSA_Q - MLDSA_Q < 0. Assumes+ * coefficients to be standard representatives.+ *+ * @reference{The reference implementation has the input polynomial as a+ * separate argument that may be aliased with either of the outputs. Removing+ * the aliasing eases CBMC proofs.}+ *+ * @param[out] v1 Pointer to output vector of polynomials with+ * coefficients a1.+ * @param[in,out] v0 Pointer to input/output vector of polynomials. Output+ * polynomial has coefficients a0.+ */+MLD_INTERNAL_API+void mld_polyveck_decompose(mld_polyveck *v1, mld_polyveck *v0)+__contract__(+ requires(memory_no_alias(v1, sizeof(mld_polyveck)))+ requires(memory_no_alias(v0, sizeof(mld_polyveck)))+ requires(forall(k0, 0, MLDSA_K,+ array_bound(v0->vec[k0].coeffs, 0, MLDSA_N, 0, MLDSA_Q)))+ assigns(memory_slice(v1, sizeof(mld_polyveck)))+ assigns(memory_slice(v0, sizeof(mld_polyveck)))+ ensures(forall(k1, 0, MLDSA_K,+ array_bound(v1->vec[k1].coeffs, 0, MLDSA_N, 0, (MLDSA_Q-1)/(2*MLDSA_GAMMA2))))+ ensures(forall(k2, 0, MLDSA_K,+ array_abs_bound(v0->vec[k2].coeffs, 0, MLDSA_N, MLDSA_GAMMA2+1)))+);+#endif /* !MLD_CONFIG_NO_SIGN_API */++#if !defined(MLD_CONFIG_NO_SIGN_API)+#define mld_polyveck_pack_w1 MLD_NAMESPACE_KL(polyveck_pack_w1)+/**+ * Bit-pack polynomial vector w1 with coefficients in [0, 15] or [0, 43]. Input+ * coefficients are assumed to be standard representatives.+ *+ * @spec{Implements @[FIPS204, Algorithm 28, w1Encode].}+ *+ * @param[out] r Pointer to output byte array with at least+ * MLDSA_K * MLDSA_POLYW1_PACKEDBYTES bytes.+ * @param[in] w1 Pointer to input polynomial vector.+ */+MLD_INTERNAL_API+void mld_polyveck_pack_w1(uint8_t r[MLDSA_K * MLDSA_POLYW1_PACKEDBYTES],+ const mld_polyveck *w1)+__contract__(+ requires(memory_no_alias(r, MLDSA_K * MLDSA_POLYW1_PACKEDBYTES))+ requires(memory_no_alias(w1, sizeof(mld_polyveck)))+ requires(forall(k1, 0, MLDSA_K,+ array_bound(w1->vec[k1].coeffs, 0, MLDSA_N, 0, (MLDSA_Q-1)/(2*MLDSA_GAMMA2))))+ assigns(memory_slice(r, MLDSA_K * MLDSA_POLYW1_PACKEDBYTES))+);+#endif /* !MLD_CONFIG_NO_SIGN_API */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API)+#define mld_polyveck_pack_eta MLD_NAMESPACE_KL(polyveck_pack_eta)+/**+ * Bit-pack polynomial vector with coefficients in [-MLDSA_ETA, MLDSA_ETA].+ *+ * @param[out] r Pointer to output byte array with+ * MLDSA_K * MLDSA_POLYETA_PACKEDBYTES bytes.+ * @param[in] p Pointer to input polynomial vector.+ */+MLD_INTERNAL_API+void mld_polyveck_pack_eta(uint8_t r[MLDSA_K * MLDSA_POLYETA_PACKEDBYTES],+ const mld_polyveck *p)+__contract__(+ requires(memory_no_alias(r, MLDSA_K * MLDSA_POLYETA_PACKEDBYTES))+ requires(memory_no_alias(p, sizeof(mld_polyveck)))+ requires(forall(k1, 0, MLDSA_K,+ array_abs_bound(p->vec[k1].coeffs, 0, MLDSA_N, MLDSA_ETA + 1)))+ assigns(memory_slice(r, MLDSA_K * MLDSA_POLYETA_PACKEDBYTES))+);++#define mld_polyvecl_pack_eta MLD_NAMESPACE_KL(polyvecl_pack_eta)+/**+ * Bit-pack polynomial vector with coefficients in [-MLDSA_ETA, MLDSA_ETA].+ *+ * @param[out] r Pointer to output byte array with+ * MLDSA_L * MLDSA_POLYETA_PACKEDBYTES bytes.+ * @param[in] p Pointer to input polynomial vector.+ */+MLD_INTERNAL_API+void mld_polyvecl_pack_eta(uint8_t r[MLDSA_L * MLDSA_POLYETA_PACKEDBYTES],+ const mld_polyvecl *p)+__contract__(+ requires(memory_no_alias(r, MLDSA_L * MLDSA_POLYETA_PACKEDBYTES))+ requires(memory_no_alias(p, sizeof(mld_polyvecl)))+ requires(forall(k1, 0, MLDSA_L,+ array_abs_bound(p->vec[k1].coeffs, 0, MLDSA_N, MLDSA_ETA + 1)))+ assigns(memory_slice(r, MLDSA_L * MLDSA_POLYETA_PACKEDBYTES))+);++#endif /* !MLD_CONFIG_NO_KEYPAIR_API */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API) || \+ (!defined(MLD_CONFIG_NO_SIGN_API) && \+ (!defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)))+#define mld_polyvecl_unpack_eta MLD_NAMESPACE_KL(polyvecl_unpack_eta)+/**+ * Unpack polynomial vector with coefficients in [-MLDSA_ETA, MLDSA_ETA].+ *+ * @param[out] p Pointer to output polynomial vector.+ * @param[in] r Input byte array with bit-packed polynomial vector.+ */+MLD_INTERNAL_API+void mld_polyvecl_unpack_eta(+ mld_polyvecl *p, const uint8_t r[MLDSA_L * MLDSA_POLYETA_PACKEDBYTES])+__contract__(+ requires(memory_no_alias(r, MLDSA_L * MLDSA_POLYETA_PACKEDBYTES))+ requires(memory_no_alias(p, sizeof(mld_polyvecl)))+ assigns(memory_slice(p, sizeof(mld_polyvecl)))+ ensures(forall(k1, 0, MLDSA_L,+ array_bound(p->vec[k1].coeffs, 0, MLDSA_N, MLD_POLYETA_UNPACK_LOWER_BOUND, MLDSA_ETA + 1)))+);+#endif /* !MLD_CONFIG_NO_KEYPAIR_API || (!MLD_CONFIG_NO_SIGN_API && \+ (!MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST)) */++#if !defined(MLD_CONFIG_NO_VERIFY_API)+#define mld_polyvecl_unpack_z MLD_NAMESPACE_KL(polyvecl_unpack_z)+/**+ * Unpack polynomial vector with coefficients in+ * [-(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1].+ *+ * @param[out] z Pointer to output polynomial vector.+ * @param[in] r Input byte array with bit-packed polynomial vector.+ */+MLD_INTERNAL_API+void mld_polyvecl_unpack_z(mld_polyvecl *z,+ const uint8_t r[MLDSA_L * MLDSA_POLYZ_PACKEDBYTES])+__contract__(+ requires(memory_no_alias(r, MLDSA_L * MLDSA_POLYZ_PACKEDBYTES))+ requires(memory_no_alias(z, sizeof(mld_polyvecl)))+ assigns(memory_slice(z, sizeof(mld_polyvecl)))+ ensures(forall(k1, 0, MLDSA_L,+ array_bound(z->vec[k1].coeffs, 0, MLDSA_N, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1)))+);+#endif /* !MLD_CONFIG_NO_VERIFY_API */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API) || \+ (!defined(MLD_CONFIG_NO_SIGN_API) && \+ (!defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)))+#define mld_polyveck_unpack_eta MLD_NAMESPACE_KL(polyveck_unpack_eta)+/**+ * Unpack polynomial vector with coefficients in [-MLDSA_ETA, MLDSA_ETA].+ *+ * @param[out] p Pointer to output polynomial vector.+ * @param[in] r Input byte array with bit-packed polynomial vector.+ */+MLD_INTERNAL_API+void mld_polyveck_unpack_eta(+ mld_polyveck *p, const uint8_t r[MLDSA_K * MLDSA_POLYETA_PACKEDBYTES])+__contract__(+ requires(memory_no_alias(r, MLDSA_K * MLDSA_POLYETA_PACKEDBYTES))+ requires(memory_no_alias(p, sizeof(mld_polyveck)))+ assigns(memory_slice(p, sizeof(mld_polyveck)))+ ensures(forall(k1, 0, MLDSA_K,+ array_bound(p->vec[k1].coeffs, 0, MLDSA_N, MLD_POLYETA_UNPACK_LOWER_BOUND, MLDSA_ETA + 1)))+);+#endif /* !MLD_CONFIG_NO_KEYPAIR_API || (!MLD_CONFIG_NO_SIGN_API && \+ (!MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST)) */+++#endif /* !MLD_POLYVEC_H */
@@ -0,0 +1,311 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS204]+ * FIPS 204 Module-Lattice-Based Digital Signature Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/204/final+ */++#include "polyvec_lazy.h"++#include "debug.h"++/* This namespacing is not done at the top to avoid a naming conflict+ * with native backends, which are currently not yet namespaced. */+#define mld_polymat_expand_entry MLD_ADD_PARAM_SET(mld_polymat_expand_entry)++/**+ * Sample a single matrix entry A[k][l] of ExpandA(rho) by rejection sampling+ * from SHAKE128(rho|l|k), and apply the custom-order permutation when a+ * native NTT backend is in use.+ *+ * The caller is expected to have copied rho into the first MLDSA_SEEDBYTES+ * of seed_ext. This function writes the domain-separation bytes+ * seed_ext[SEEDBYTES..+2] = {l, k} before sampling.+ *+ * @spec{Partially implements @[FIPS204, Algorithm 32, ExpandA] (samples one+ * matrix entry via @[FIPS204, Algorithm 30, RejNTTPoly]).}+ *+ * @param[out] p Pointer to output polynomial.+ * @param[in,out] seed_ext Seed buffer pre-filled with rho in the first+ * MLDSA_SEEDBYTES; the final two bytes are+ * overwritten.+ * @param l Column index (inner, aka nonce low byte).+ * @param k Row index (outer, aka nonce high byte).+ */+static MLD_INLINE void mld_polymat_expand_entry(+ mld_poly *p, uint8_t seed_ext[MLD_ALIGN_UP(MLDSA_SEEDBYTES + 2)], uint8_t l,+ uint8_t k)+__contract__(+ requires(memory_no_alias(p, sizeof(mld_poly)))+ requires(memory_no_alias(seed_ext, MLD_ALIGN_UP(MLDSA_SEEDBYTES + 2)))+ assigns(memory_slice(p, sizeof(mld_poly)))+ assigns(memory_slice(seed_ext, MLD_ALIGN_UP(MLDSA_SEEDBYTES + 2)))+ ensures(array_bound(p->coeffs, 0, MLDSA_N, 0, MLDSA_Q))+)+{+ seed_ext[MLDSA_SEEDBYTES + 0] = l;+ seed_ext[MLDSA_SEEDBYTES + 1] = k;+ mld_poly_uniform(p, seed_ext);+ mld_poly_permute_bitrev_to_custom_optional(p);+}++#if !defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)++MLD_INTERNAL_API+void mld_polyvec_matrix_expand_eager(mld_polymat_eager *mat,+ const uint8_t rho[MLDSA_SEEDBYTES])+{+ unsigned int i, j;+ MLD_ALIGN uint8_t seed_ext[4][MLD_ALIGN_UP(MLDSA_SEEDBYTES + 2)];++ for (j = 0; j < 4; j++)+ __loop__(+ assigns(j, object_whole(seed_ext))+ invariant(j <= 4)+ decreases(4 - j)+ )+ {+ mld_memcpy(seed_ext[j], rho, MLDSA_SEEDBYTES);+ }++#if !defined(MLD_CONFIG_SERIAL_FIPS202_ONLY)+ /* Sample 4 matrix entries a time. */+ for (i = 0; i < (MLDSA_K * MLDSA_L / 4) * 4; i += 4)+ __loop__(+ assigns(i, j, object_whole(seed_ext), memory_slice(mat, sizeof(mld_polymat_eager)))+ invariant(i <= (MLDSA_K * MLDSA_L / 4) * 4 && i % 4 == 0)+ /* vectors 0 .. i / MLDSA_L are completely sampled */+ invariant(forall(k1, 0, i / MLDSA_L, forall(l1, 0, MLDSA_L,+ array_bound(mat->vec[k1].vec[l1].coeffs, 0, MLDSA_N, 0, MLDSA_Q))))+ /* last vector is sampled up to i % MLDSA_L */+ invariant(forall(k2, i / MLDSA_L, i / MLDSA_L + 1, forall(l2, 0, i % MLDSA_L,+ array_bound(mat->vec[k2].vec[l2].coeffs, 0, MLDSA_N, 0, MLDSA_Q))))+ decreases((MLDSA_K * MLDSA_L / 4) * 4 - i)+ )+ {+ for (j = 0; j < 4; j++)+ __loop__(+ assigns(j, object_whole(seed_ext))+ invariant(j <= 4)+ decreases(4 - j)+ )+ {+ uint8_t x = (uint8_t)((i + j) / MLDSA_L);+ uint8_t y = (uint8_t)((i + j) % MLDSA_L);++ seed_ext[j][MLDSA_SEEDBYTES + 0] = y;+ seed_ext[j][MLDSA_SEEDBYTES + 1] = x;+ }++ mld_poly_uniform_4x(&mat->vec[i / MLDSA_L].vec[i % MLDSA_L],+ &mat->vec[(i + 1) / MLDSA_L].vec[(i + 1) % MLDSA_L],+ &mat->vec[(i + 2) / MLDSA_L].vec[(i + 2) % MLDSA_L],+ &mat->vec[(i + 3) / MLDSA_L].vec[(i + 3) % MLDSA_L],+ seed_ext);+ mld_poly_permute_bitrev_to_custom_optional(+ &mat->vec[i / MLDSA_L].vec[i % MLDSA_L]);+ mld_poly_permute_bitrev_to_custom_optional(+ &mat->vec[(i + 1) / MLDSA_L].vec[(i + 1) % MLDSA_L]);+ mld_poly_permute_bitrev_to_custom_optional(+ &mat->vec[(i + 2) / MLDSA_L].vec[(i + 2) % MLDSA_L]);+ mld_poly_permute_bitrev_to_custom_optional(+ &mat->vec[(i + 3) / MLDSA_L].vec[(i + 3) % MLDSA_L]);+ }+#else /* !MLD_CONFIG_SERIAL_FIPS202_ONLY */+ i = 0;+#endif /* MLD_CONFIG_SERIAL_FIPS202_ONLY */++ /* Entries omitted by the batch-sampling are sampled individually. */+ while (i < MLDSA_K * MLDSA_L)+ __loop__(+ assigns(i, object_whole(seed_ext), memory_slice(mat, sizeof(mld_polymat_eager)))+ invariant(i <= MLDSA_K * MLDSA_L)+ /* vectors 0 .. i / MLDSA_L are completely sampled */+ invariant(forall(k1, 0, i / MLDSA_L, forall(l1, 0, MLDSA_L,+ array_bound(mat->vec[k1].vec[l1].coeffs, 0, MLDSA_N, 0, MLDSA_Q))))+ /* last vector is sampled up to i % MLDSA_L */+ invariant(forall(k2, i / MLDSA_L, i / MLDSA_L + 1, forall(l2, 0, i % MLDSA_L,+ array_bound(mat->vec[k2].vec[l2].coeffs, 0, MLDSA_N, 0, MLDSA_Q))))+ decreases(MLDSA_K * MLDSA_L - i)+ )+ {+ uint8_t x = (uint8_t)(i / MLDSA_L);+ uint8_t y = (uint8_t)(i % MLDSA_L);+ mld_polymat_expand_entry(&mat->vec[x].vec[y], seed_ext[0], y, x);+ i++;+ }++ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ mld_zeroize(seed_ext, sizeof(seed_ext));+}++MLD_INTERNAL_API+void mld_polyvec_matrix_pointwise_montgomery_row_eager(mld_poly *t_row,+ mld_polymat_eager *mat,+ const mld_polyvecl *v,+ unsigned int i)+{+ mld_polyvecl_pointwise_acc_montgomery(t_row, &mat->vec[i], v);+}++#if !defined(MLD_CONFIG_NO_SIGN_API)+MLD_INTERNAL_API+void mld_polyvec_matrix_pointwise_montgomery_yvec_eager(mld_polyveck *w,+ mld_polymat_eager *mat,+ const mld_yvec_eager *y,+ mld_polyvecl *scratch)+{+ unsigned int i;+ *scratch = y->vec;+ mld_polyvecl_ntt(scratch);++ for (i = 0; i < MLDSA_K; ++i)+ __loop__(+ assigns(i, memory_slice(w, sizeof(mld_polyveck)))+ invariant(i <= MLDSA_K)+ invariant(forall(k0, 0, i,+ array_abs_bound(w->vec[k0].coeffs, 0, MLDSA_N, MLDSA_Q)))+ decreases(MLDSA_K - i)+ )+ {+ mld_polyvec_matrix_pointwise_montgomery_row_eager(&w->vec[i], mat, scratch,+ i);+ }++ mld_polyveck_invntt_tomont(w);+}+#endif /* !MLD_CONFIG_NO_SIGN_API */++#endif /* !MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST */++#if defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)++MLD_INTERNAL_API+void mld_polyvec_matrix_expand_lazy(mld_polymat_lazy *mat,+ const uint8_t rho[MLDSA_SEEDBYTES])+{+ mld_memcpy(mat->rho, rho, MLDSA_SEEDBYTES);+}++#if !defined(MLD_CONFIG_NO_KEYPAIR_API) || !defined(MLD_CONFIG_NO_VERIFY_API)+MLD_INTERNAL_API+void mld_polyvec_matrix_pointwise_montgomery_row_lazy(mld_poly *t_row,+ mld_polymat_lazy *mat,+ const mld_polyvecl *v,+ unsigned int i)+{+ unsigned int l;+ MLD_ALIGN uint8_t seed_ext[MLD_ALIGN_UP(MLDSA_SEEDBYTES + 2)];+ mld_memcpy(seed_ext, mat->rho, MLDSA_SEEDBYTES);++ mld_polymat_expand_entry(t_row, seed_ext, 0, (uint8_t)i);+ mld_poly_pointwise_montgomery(t_row, &v->vec[0]);++ for (l = 1; l < MLDSA_L; ++l)+ __loop__(+ assigns(l, object_whole(seed_ext),+ memory_slice(t_row, sizeof(mld_poly)),+ memory_slice(mat, sizeof(mld_polymat_lazy)))+ invariant(l >= 1 && l <= MLDSA_L)+ invariant(array_abs_bound(t_row->coeffs, 0, MLDSA_N, l * MLDSA_Q))+ decreases(MLDSA_L - l)+ )+ {+ mld_polymat_expand_entry(&mat->cur, seed_ext, (uint8_t)l, (uint8_t)i);+ mld_poly_pointwise_montgomery(&mat->cur, &v->vec[l]);+ mld_poly_add(t_row, &mat->cur);+ }+ mld_poly_reduce(t_row);++ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ mld_zeroize(seed_ext, sizeof(seed_ext));+}+#endif /* !MLD_CONFIG_NO_KEYPAIR_API || !MLD_CONFIG_NO_VERIFY_API */++#if !defined(MLD_CONFIG_NO_SIGN_API)+MLD_INTERNAL_API+void mld_polyvec_matrix_pointwise_montgomery_yvec_lazy(mld_polyveck *w,+ mld_polymat_lazy *mat,+ const mld_yvec_lazy *y,+ mld_polyvecl *scratch)+{+ unsigned int k, l;+ MLD_ALIGN uint8_t seed_ext[MLD_ALIGN_UP(MLDSA_SEEDBYTES + 2)];+ /* Only the first poly of the polyvecl scratch is used. The polyvecl type+ * matches the eager variant for API uniformity; in REDUCE_RAM mode the+ * polyvecl storage is provided "for free" by the caller's polyveck/polyvecl+ * union. */+ mld_poly *y_ntt = &scratch->vec[0];++ mld_memcpy(seed_ext, mat->rho, MLDSA_SEEDBYTES);++ /* Column-by-column: sample y[l], NTT, accumulate column l of A into w. */+ for (l = 0; l < MLDSA_L; l++)+ __loop__(+ assigns(k, l, object_whole(seed_ext),+ memory_slice(w, sizeof(mld_polyveck)),+ memory_slice(mat, sizeof(mld_polymat_lazy)),+ memory_slice(scratch, sizeof(mld_polyvecl)))+ invariant(l <= MLDSA_L)+ invariant(l == 0 ||+ forall(k0, 0, MLDSA_K,+ array_abs_bound(w->vec[k0].coeffs, 0, MLDSA_N,+ (int)l * MLDSA_Q)))+ decreases(MLDSA_L - l)+ )+ {+ mld_yvec_get_poly_lazy(y_ntt, y, l);+ mld_poly_ntt(y_ntt);+ for (k = 0; k < MLDSA_K; k++)+ __loop__(+ assigns(k, object_whole(seed_ext),+ memory_slice(w, sizeof(mld_polyveck)),+ memory_slice(mat, sizeof(mld_polymat_lazy)))+ invariant(k <= MLDSA_K)+ invariant(l != 0 ||+ forall(k1, 0, k,+ array_abs_bound(w->vec[k1].coeffs, 0, MLDSA_N, MLDSA_Q)))+ invariant(l == 0 ||+ forall(k2, 0, k,+ array_abs_bound(w->vec[k2].coeffs, 0, MLDSA_N,+ ((int)l + 1) * MLDSA_Q)))+ invariant(l == 0 ||+ forall(k3, k, MLDSA_K,+ array_abs_bound(w->vec[k3].coeffs, 0, MLDSA_N,+ (int)l * MLDSA_Q)))+ decreases(MLDSA_K - k)+ )+ {+ if (l == 0)+ {+ mld_polymat_expand_entry(&w->vec[k], seed_ext, 0, (uint8_t)k);+ mld_poly_pointwise_montgomery(&w->vec[k], y_ntt);+ }+ else+ {+ mld_polymat_expand_entry(&mat->cur, seed_ext, (uint8_t)l, (uint8_t)k);+ mld_poly_pointwise_montgomery(&mat->cur, y_ntt);+ mld_poly_add(&w->vec[k], &mat->cur);+ }+ }+ }++ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ mld_zeroize(seed_ext, sizeof(seed_ext));+ mld_polyveck_reduce(w);+ mld_polyveck_invntt_tomont(w);+}+#endif /* !MLD_CONFIG_NO_SIGN_API */++#endif /* MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST */++/* To facilitate single-compilation-unit (SCU) builds, undefine all macros.+ * Don't modify by hand -- this is auto-generated by scripts/autogen. */+#undef mld_polymat_expand_entry
@@ -0,0 +1,652 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS204]+ * FIPS 204 Module-Lattice-Based Digital Signature Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/204/final+ */++/*+ * Eager and lazy variants of polynomial vector types.+ *+ * In eager mode, full vectors are precomputed and stored in memory.+ * In lazy mode, data is stored in packed form and expanded on demand,+ * trading computation for reduced memory usage.+ *+ * MLD_CONFIG_REDUCE_RAM selects which variant is used.+ */++#ifndef MLD_POLYVEC_LAZY_H+#define MLD_POLYVEC_LAZY_H++#if !defined(MLD_CONFIG_NO_KEYPAIR_API) || !defined(MLD_CONFIG_NO_SIGN_API) || \+ !defined(MLD_CONFIG_NO_VERIFY_API)++#include "poly.h"+#include "poly_kl.h"+#include "polyvec.h"++/* Parameter set namespacing */+#define mld_sk_s1hat_eager MLD_ADD_PARAM_SET(mld_sk_s1hat_eager)+#define mld_sk_s1hat_lazy MLD_ADD_PARAM_SET(mld_sk_s1hat_lazy)+#define mld_sk_s1hat MLD_ADD_PARAM_SET(mld_sk_s1hat)+#define mld_unpack_sk_s1hat_eager MLD_ADD_PARAM_SET(mld_unpack_sk_s1hat_eager)+#define mld_unpack_sk_s1hat_lazy MLD_ADD_PARAM_SET(mld_unpack_sk_s1hat_lazy)+#define mld_sk_s1hat_get_poly_eager \+ MLD_ADD_PARAM_SET(mld_sk_s1hat_get_poly_eager)+#define mld_sk_s1hat_get_poly_lazy MLD_ADD_PARAM_SET(mld_sk_s1hat_get_poly_lazy)+#define mld_sk_s2hat_eager MLD_ADD_PARAM_SET(mld_sk_s2hat_eager)+#define mld_sk_s2hat_lazy MLD_ADD_PARAM_SET(mld_sk_s2hat_lazy)+#define mld_sk_s2hat MLD_ADD_PARAM_SET(mld_sk_s2hat)+#define mld_unpack_sk_s2hat_eager MLD_ADD_PARAM_SET(mld_unpack_sk_s2hat_eager)+#define mld_unpack_sk_s2hat_lazy MLD_ADD_PARAM_SET(mld_unpack_sk_s2hat_lazy)+#define mld_sk_s2hat_get_poly_eager \+ MLD_ADD_PARAM_SET(mld_sk_s2hat_get_poly_eager)+#define mld_sk_s2hat_get_poly_lazy MLD_ADD_PARAM_SET(mld_sk_s2hat_get_poly_lazy)+#define mld_sk_t0hat_eager MLD_ADD_PARAM_SET(mld_sk_t0hat_eager)+#define mld_sk_t0hat_lazy MLD_ADD_PARAM_SET(mld_sk_t0hat_lazy)+#define mld_sk_t0hat MLD_ADD_PARAM_SET(mld_sk_t0hat)+#define mld_unpack_sk_t0hat_eager MLD_ADD_PARAM_SET(mld_unpack_sk_t0hat_eager)+#define mld_unpack_sk_t0hat_lazy MLD_ADD_PARAM_SET(mld_unpack_sk_t0hat_lazy)+#define mld_sk_t0hat_get_poly_eager \+ MLD_ADD_PARAM_SET(mld_sk_t0hat_get_poly_eager)+#define mld_sk_t0hat_get_poly_lazy MLD_ADD_PARAM_SET(mld_sk_t0hat_get_poly_lazy)+#define mld_polymat MLD_ADD_PARAM_SET(mld_polymat)+#define mld_polymat_eager MLD_ADD_PARAM_SET(mld_polymat_eager)+#define mld_polymat_lazy MLD_ADD_PARAM_SET(mld_polymat_lazy)+#define mld_poly_permute_bitrev_to_custom_optional \+ MLD_ADD_PARAM_SET(mld_poly_permute_bitrev_to_custom_optional)+#define mld_polyvec_matrix_expand_eager \+ MLD_NAMESPACE_KL(polyvec_matrix_expand_eager)+#define mld_polyvec_matrix_expand_lazy \+ MLD_NAMESPACE_KL(polyvec_matrix_expand_lazy)+#define mld_polyvec_matrix_pointwise_montgomery \+ MLD_NAMESPACE_KL(polyvec_matrix_pointwise_montgomery)+#define mld_polyvec_matrix_pointwise_montgomery_row_eager \+ MLD_NAMESPACE_KL(polyvec_matrix_pointwise_montgomery_row_eager)+#define mld_polyvec_matrix_pointwise_montgomery_row_lazy \+ MLD_NAMESPACE_KL(polyvec_matrix_pointwise_montgomery_row_lazy)+#define mld_polyvec_matrix_pointwise_montgomery_yvec_eager \+ MLD_NAMESPACE_KL(polyvec_matrix_pointwise_montgomery_yvec_eager)+#define mld_polyvec_matrix_pointwise_montgomery_yvec_lazy \+ MLD_NAMESPACE_KL(polyvec_matrix_pointwise_montgomery_yvec_lazy)+#define mld_yvec_eager MLD_ADD_PARAM_SET(mld_yvec_eager)+#define mld_yvec_lazy MLD_ADD_PARAM_SET(mld_yvec_lazy)+#define mld_yvec MLD_ADD_PARAM_SET(mld_yvec)+#define mld_yvec_init_eager MLD_ADD_PARAM_SET(mld_yvec_init_eager)+#define mld_yvec_init_lazy MLD_ADD_PARAM_SET(mld_yvec_init_lazy)+#define mld_yvec_get_poly_eager MLD_ADD_PARAM_SET(mld_yvec_get_poly_eager)+#define mld_yvec_get_poly_lazy MLD_ADD_PARAM_SET(mld_yvec_get_poly_lazy)+/* End of parameter set namespacing */++/** Eager s1hat: precomputed s1 vector in NTT domain. */+typedef struct+{+ mld_polyvecl vec; /**< s1 vector in NTT domain. */+} mld_sk_s1hat_eager;++/** Eager s2hat: precomputed s2 vector in NTT domain. */+typedef struct+{+ mld_polyveck vec; /**< s2 vector in NTT domain. */+} mld_sk_s2hat_eager;++/** Eager t0hat: precomputed t0 vector in NTT domain. */+typedef struct+{+ mld_polyveck vec; /**< t0 vector in NTT domain. */+} mld_sk_t0hat_eager;++/** Lazy s1hat: borrow packed s1, unpack and convert to NTT domain on demand. */+typedef struct+{+ const uint8_t *packed; /**< Pointer to packed s1 in the secret key. */+} mld_sk_s1hat_lazy;++/** Lazy s2hat: borrow packed s2, unpack and convert to NTT domain on demand. */+typedef struct+{+ const uint8_t *packed; /**< Pointer to packed s2 in the secret key. */+} mld_sk_s2hat_lazy;++/** Lazy t0hat: borrow packed t0, unpack and convert to NTT domain on demand. */+typedef struct+{+ const uint8_t *packed; /**< Pointer to packed t0 in the secret key. */+} mld_sk_t0hat_lazy;++/** Eager yvec: precomputed and stored full signing masking vector y. */+typedef struct+{+ mld_polyvecl vec; /**< Masking vector y. */+} mld_yvec_eager;++/** Lazy yvec: store seed and base counter kappa, regenerate y[i] on demand. */+typedef struct+{+ const uint8_t *rhoprime; /**< Pointer to seed used to derive y. */+ uint16_t kappa; /**< Base counter; component i uses kappa + i. */+} mld_yvec_lazy;++#if !defined(MLD_CONFIG_NO_KEYPAIR_API) || !defined(MLD_CONFIG_NO_SIGN_API)+/* s1vec */++#if !defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)+static MLD_INLINE void mld_unpack_sk_s1hat_eager(+ mld_sk_s1hat_eager *s1,+ const uint8_t packed_s1[MLDSA_L * MLDSA_POLYETA_PACKEDBYTES])+__contract__(+ requires(memory_no_alias(s1, sizeof(mld_sk_s1hat_eager)))+ requires(memory_no_alias(packed_s1, MLDSA_L * MLDSA_POLYETA_PACKEDBYTES))+ assigns(memory_slice(s1, sizeof(mld_sk_s1hat_eager)))+ ensures(forall(k1, 0, MLDSA_L,+ array_abs_bound(s1->vec.vec[k1].coeffs, 0, MLDSA_N, MLD_NTT_BOUND)))+)+{+ mld_polyvecl_unpack_eta(&s1->vec, packed_s1);+ mld_polyvecl_ntt(&s1->vec);+}++#if !defined(MLD_CONFIG_NO_SIGN_API)+static MLD_INLINE void mld_sk_s1hat_get_poly_eager(mld_poly *buf,+ const mld_sk_s1hat_eager *s1,+ unsigned int i)+__contract__(+ requires(memory_no_alias(buf, sizeof(mld_poly)))+ requires(memory_no_alias(s1, sizeof(mld_sk_s1hat_eager)))+ requires(i < MLDSA_L)+ requires(array_abs_bound(s1->vec.vec[i].coeffs, 0, MLDSA_N, MLD_NTT_BOUND))+ assigns(memory_slice(buf, sizeof(mld_poly)))+ ensures(array_abs_bound(buf->coeffs, 0, MLDSA_N, MLD_NTT_BOUND))+) { *buf = s1->vec.vec[i]; }+#endif /* !MLD_CONFIG_NO_SIGN_API */+#endif /* !MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST */+#if defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)+static MLD_INLINE void mld_unpack_sk_s1hat_lazy(+ mld_sk_s1hat_lazy *s1,+ const uint8_t packed_s1[MLDSA_L * MLDSA_POLYETA_PACKEDBYTES])+__contract__(+ requires(memory_no_alias(s1, sizeof(mld_sk_s1hat_lazy)))+ assigns(memory_slice(s1, sizeof(mld_sk_s1hat_lazy)))+ ensures(s1->packed == old(packed_s1))+) { s1->packed = packed_s1; }++#if !defined(MLD_CONFIG_NO_SIGN_API)+static MLD_INLINE void mld_sk_s1hat_get_poly_lazy(mld_poly *buf,+ const mld_sk_s1hat_lazy *s1,+ unsigned int i)+__contract__(+ requires(memory_no_alias(buf, sizeof(mld_poly)))+ requires(memory_no_alias(s1, sizeof(mld_sk_s1hat_lazy)))+ requires(i < MLDSA_L)+ requires(memory_no_alias(s1->packed, MLDSA_L * MLDSA_POLYETA_PACKEDBYTES))+ assigns(memory_slice(buf, sizeof(mld_poly)))+ ensures(array_abs_bound(buf->coeffs, 0, MLDSA_N, MLD_NTT_BOUND))+)+{+ mld_polyeta_unpack(buf, s1->packed + i * MLDSA_POLYETA_PACKEDBYTES);+ mld_poly_ntt(buf);+}+#endif /* !MLD_CONFIG_NO_SIGN_API */+#endif /* MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST */++/* s2vec */++#if (!defined(MLD_CONFIG_NO_SIGN_API) || defined(MLD_UNIT_TEST)) && \+ (!defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST))+static MLD_INLINE void mld_unpack_sk_s2hat_eager(+ mld_sk_s2hat_eager *s2,+ const uint8_t packed_s2[MLDSA_K * MLDSA_POLYETA_PACKEDBYTES])+__contract__(+ requires(memory_no_alias(s2, sizeof(mld_sk_s2hat_eager)))+ requires(memory_no_alias(packed_s2, MLDSA_K * MLDSA_POLYETA_PACKEDBYTES))+ assigns(memory_slice(s2, sizeof(mld_sk_s2hat_eager)))+ ensures(forall(k1, 0, MLDSA_K,+ array_abs_bound(s2->vec.vec[k1].coeffs, 0, MLDSA_N, MLD_NTT_BOUND)))+)+{+ mld_polyveck_unpack_eta(&s2->vec, packed_s2);+ mld_polyveck_ntt(&s2->vec);+}++#if !defined(MLD_CONFIG_NO_SIGN_API)+static MLD_INLINE void mld_sk_s2hat_get_poly_eager(mld_poly *buf,+ const mld_sk_s2hat_eager *s2,+ unsigned int i)+__contract__(+ requires(memory_no_alias(buf, sizeof(mld_poly)))+ requires(memory_no_alias(s2, sizeof(mld_sk_s2hat_eager)))+ requires(i < MLDSA_K)+ requires(array_abs_bound(s2->vec.vec[i].coeffs, 0, MLDSA_N, MLD_NTT_BOUND))+ assigns(memory_slice(buf, sizeof(mld_poly)))+ ensures(array_abs_bound(buf->coeffs, 0, MLDSA_N, MLD_NTT_BOUND))+) { *buf = s2->vec.vec[i]; }+#endif /* !MLD_CONFIG_NO_SIGN_API */+#endif /* (!MLD_CONFIG_NO_SIGN_API || MLD_UNIT_TEST) && \+ (!MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST) */+#if defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)+static MLD_INLINE void mld_unpack_sk_s2hat_lazy(+ mld_sk_s2hat_lazy *s2,+ const uint8_t packed_s2[MLDSA_K * MLDSA_POLYETA_PACKEDBYTES])+__contract__(+ requires(memory_no_alias(s2, sizeof(mld_sk_s2hat_lazy)))+ assigns(memory_slice(s2, sizeof(mld_sk_s2hat_lazy)))+ ensures(s2->packed == old(packed_s2))+) { s2->packed = packed_s2; }++#if !defined(MLD_CONFIG_NO_SIGN_API)+static MLD_INLINE void mld_sk_s2hat_get_poly_lazy(mld_poly *buf,+ const mld_sk_s2hat_lazy *s2,+ unsigned int i)+__contract__(+ requires(memory_no_alias(buf, sizeof(mld_poly)))+ requires(memory_no_alias(s2, sizeof(mld_sk_s2hat_lazy)))+ requires(i < MLDSA_K)+ requires(memory_no_alias(s2->packed, MLDSA_K * MLDSA_POLYETA_PACKEDBYTES))+ assigns(memory_slice(buf, sizeof(mld_poly)))+ ensures(array_abs_bound(buf->coeffs, 0, MLDSA_N, MLD_NTT_BOUND))+)+{+ mld_polyeta_unpack(buf, s2->packed + i * MLDSA_POLYETA_PACKEDBYTES);+ mld_poly_ntt(buf);+}+#endif /* !MLD_CONFIG_NO_SIGN_API */+#endif /* MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST */++/* t0vec */++#if (!defined(MLD_CONFIG_NO_SIGN_API) || defined(MLD_UNIT_TEST)) && \+ (!defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST))+static MLD_INLINE void mld_unpack_sk_t0hat_eager(+ mld_sk_t0hat_eager *t0,+ const uint8_t packed_t0[MLDSA_K * MLDSA_POLYT0_PACKEDBYTES])+__contract__(+ requires(memory_no_alias(t0, sizeof(mld_sk_t0hat_eager)))+ requires(memory_no_alias(packed_t0, MLDSA_K * MLDSA_POLYT0_PACKEDBYTES))+ assigns(memory_slice(t0, sizeof(mld_sk_t0hat_eager)))+ ensures(forall(k1, 0, MLDSA_K,+ array_abs_bound(t0->vec.vec[k1].coeffs, 0, MLDSA_N, MLD_NTT_BOUND)))+)+{+ unsigned int i;+ for (i = 0; i < MLDSA_K; ++i)+ __loop__(+ assigns(i, memory_slice(t0, sizeof(mld_sk_t0hat_eager)))+ invariant(i <= MLDSA_K)+ invariant(forall(k0, 0, i,+ array_bound(t0->vec.vec[k0].coeffs, 0, MLDSA_N,+ -(1 << (MLDSA_D - 1)) + 1, (1 << (MLDSA_D - 1)) + 1)))+ decreases(MLDSA_K - i)+ )+ {+ mld_polyt0_unpack(&t0->vec.vec[i],+ packed_t0 + i * MLDSA_POLYT0_PACKEDBYTES);+ }+ mld_polyveck_ntt(&t0->vec);+}++#if !defined(MLD_CONFIG_NO_SIGN_API)+static MLD_INLINE void mld_sk_t0hat_get_poly_eager(mld_poly *buf,+ const mld_sk_t0hat_eager *t0,+ unsigned int i)+__contract__(+ requires(memory_no_alias(buf, sizeof(mld_poly)))+ requires(memory_no_alias(t0, sizeof(mld_sk_t0hat_eager)))+ requires(i < MLDSA_K)+ requires(array_abs_bound(t0->vec.vec[i].coeffs, 0, MLDSA_N, MLD_NTT_BOUND))+ assigns(memory_slice(buf, sizeof(mld_poly)))+ ensures(array_abs_bound(buf->coeffs, 0, MLDSA_N, MLD_NTT_BOUND))+) { *buf = t0->vec.vec[i]; }+#endif /* !MLD_CONFIG_NO_SIGN_API */+#endif /* (!MLD_CONFIG_NO_SIGN_API || MLD_UNIT_TEST) && \+ (!MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST) */+#if defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)+static MLD_INLINE void mld_unpack_sk_t0hat_lazy(+ mld_sk_t0hat_lazy *t0,+ const uint8_t packed_t0[MLDSA_K * MLDSA_POLYT0_PACKEDBYTES])+__contract__(+ requires(memory_no_alias(t0, sizeof(mld_sk_t0hat_lazy)))+ assigns(memory_slice(t0, sizeof(mld_sk_t0hat_lazy)))+ ensures(t0->packed == old(packed_t0))+) { t0->packed = packed_t0; }++#if !defined(MLD_CONFIG_NO_SIGN_API)+static MLD_INLINE void mld_sk_t0hat_get_poly_lazy(mld_poly *buf,+ const mld_sk_t0hat_lazy *t0,+ unsigned int i)+__contract__(+ requires(memory_no_alias(buf, sizeof(mld_poly)))+ requires(memory_no_alias(t0, sizeof(mld_sk_t0hat_lazy)))+ requires(i < MLDSA_K)+ requires(memory_no_alias(t0->packed, MLDSA_K * MLDSA_POLYT0_PACKEDBYTES))+ assigns(memory_slice(buf, sizeof(mld_poly)))+ ensures(array_abs_bound(buf->coeffs, 0, MLDSA_N, MLD_NTT_BOUND))+)+{+ mld_polyt0_unpack(buf, t0->packed + i * MLDSA_POLYT0_PACKEDBYTES);+ mld_poly_ntt(buf);+}+#endif /* !MLD_CONFIG_NO_SIGN_API */+#endif /* MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST */+#endif /* !MLD_CONFIG_NO_KEYPAIR_API || !MLD_CONFIG_NO_SIGN_API */++/* yvec */++#if !defined(MLD_CONFIG_NO_SIGN_API) && \+ (!defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST))+static MLD_INLINE void mld_yvec_init_eager(+ mld_yvec_eager *y, const uint8_t rhoprime[MLDSA_CRHBYTES], uint16_t kappa)+__contract__(+ requires(memory_no_alias(y, sizeof(mld_yvec_eager)))+ requires(memory_no_alias(rhoprime, MLDSA_CRHBYTES))+ requires(kappa <= MLD_MAX_KAPPA)+ assigns(memory_slice(y, sizeof(mld_yvec_eager)))+ ensures(forall(k1, 0, MLDSA_L,+ array_bound(y->vec.vec[k1].coeffs, 0, MLDSA_N, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1)))+)+{+ mld_polyvecl_uniform_gamma1(&y->vec, rhoprime, kappa);+}++static MLD_INLINE void mld_yvec_get_poly_eager(mld_poly *buf,+ const mld_yvec_eager *y,+ unsigned int i)+__contract__(+ requires(memory_no_alias(buf, sizeof(mld_poly)))+ requires(memory_no_alias(y, sizeof(mld_yvec_eager)))+ requires(i < MLDSA_L)+ requires(array_bound(y->vec.vec[i].coeffs, 0, MLDSA_N, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1))+ assigns(memory_slice(buf, sizeof(mld_poly)))+ ensures(array_bound(buf->coeffs, 0, MLDSA_N, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1))+) { *buf = y->vec.vec[i]; }+#endif /* !MLD_CONFIG_NO_SIGN_API && (!MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST) \+ */+#if !defined(MLD_CONFIG_NO_SIGN_API) && \+ (defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST))+static MLD_INLINE void mld_yvec_init_lazy(+ mld_yvec_lazy *y, const uint8_t rhoprime[MLDSA_CRHBYTES], uint16_t kappa)+__contract__(+ requires(memory_no_alias(y, sizeof(mld_yvec_lazy)))+ assigns(memory_slice(y, sizeof(mld_yvec_lazy)))+ ensures(y->rhoprime == old(rhoprime))+ ensures(y->kappa == old(kappa))+)+{+ y->rhoprime = rhoprime;+ y->kappa = kappa;+}++static MLD_INLINE void mld_yvec_get_poly_lazy(mld_poly *buf,+ const mld_yvec_lazy *y,+ unsigned int i)+__contract__(+ requires(memory_no_alias(buf, sizeof(mld_poly)))+ requires(memory_no_alias(y, sizeof(mld_yvec_lazy)))+ requires(i < MLDSA_L)+ requires(memory_no_alias(y->rhoprime, MLDSA_CRHBYTES))+ requires(y->kappa <= MLD_MAX_KAPPA)+ assigns(memory_slice(buf, sizeof(mld_poly)))+ ensures(array_bound(buf->coeffs, 0, MLDSA_N, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1))+)+{+ /* Safety: y->kappa <= MLD_MAX_KAPPA and i < MLDSA_L, so y->kappa + i+ * fits in uint16_t. See MLD_MAX_KAPPA comment in params.h. */+ mld_poly_uniform_gamma1(buf, y->rhoprime, (uint16_t)(y->kappa + i));+}+#endif /* !MLD_CONFIG_NO_SIGN_API && (MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST) \+ */++/* polymat */++#if !defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)+/** Eager polymat: precomputed and stored full MLDSA_K x MLDSA_L matrix. */+typedef struct+{+ mld_polyvecl vec[MLDSA_K]; /**< Rows of the matrix. */+} mld_polymat_eager;+#endif /* !MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST */++/** Lazy polymat: store seed, sample elements A[k][l] on demand. */+typedef struct+{+ mld_poly cur; /**< On-demand sampled matrix element A[k][l]. */+ uint8_t rho[MLDSA_SEEDBYTES]; /**< Public seed used to expand A. */+} mld_polymat_lazy;++static MLD_INLINE void mld_poly_permute_bitrev_to_custom_optional(mld_poly *p)+__contract__(+ /* We don't specify that this is a permutation, only that it preserves+ * the bounds.+ * When the native NTT backend does not use the custom order, this is a no-op. */+ requires(memory_no_alias(p, sizeof(mld_poly)))+ requires(array_bound(p->coeffs, 0, MLDSA_N, 0, MLDSA_Q))+ assigns(memory_slice(p, sizeof(mld_poly)))+ ensures(array_bound(p->coeffs, 0, MLDSA_N, 0, MLDSA_Q))+)+{+#if defined(MLD_USE_NATIVE_NTT_CUSTOM_ORDER)+ mld_poly_permute_bitrev_to_custom(p->coeffs);+#else+ (void)p;+#endif+}++#if !defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)+/**+ * Generates matrix A with uniformly random coefficients a_{i,j} by performing+ * rejection sampling on the output stream of SHAKE128(rho|j|i).+ *+ * @spec{Implements @[FIPS204, Algorithm 32, ExpandA].}+ *+ * @param[out] mat Pointer to output matrix.+ * @param[in] rho Byte array containing seed rho.+ */+MLD_INTERNAL_API+void mld_polyvec_matrix_expand_eager(mld_polymat_eager *mat,+ const uint8_t rho[MLDSA_SEEDBYTES])+__contract__(+ requires(memory_no_alias(mat, sizeof(mld_polymat_eager)))+ requires(memory_no_alias(rho, MLDSA_SEEDBYTES))+ assigns(memory_slice(mat, sizeof(mld_polymat_eager)))+ ensures(forall(k1, 0, MLDSA_K, forall(l1, 0, MLDSA_L,+ array_bound(mat->vec[k1].vec[l1].coeffs, 0, MLDSA_N, 0, MLDSA_Q))))+);++/**+ * Compute row i of matrix-vector multiplication in NTT domain with pointwise+ * multiplication and multiplication by 2^{-32}.+ *+ * Input matrix and vector must be in NTT domain representation. Output+ * coefficients are bounded by MLDSA_Q in absolute value.+ *+ * @param[out] t_row Pointer to output row polynomial.+ * @param[in] mat Pointer to input matrix.+ * @param[in] v Pointer to input vector v.+ * @param i Row index, 0 <= i < MLDSA_K.+ */+MLD_INTERNAL_API+void mld_polyvec_matrix_pointwise_montgomery_row_eager(mld_poly *t_row,+ mld_polymat_eager *mat,+ const mld_polyvecl *v,+ unsigned int i)+__contract__(+ requires(memory_no_alias(t_row, sizeof(mld_poly)))+ requires(memory_no_alias(mat, sizeof(mld_polymat_eager)))+ requires(memory_no_alias(v, sizeof(mld_polyvecl)))+ requires(i < MLDSA_K)+ requires(forall(l1, 0, MLDSA_L,+ array_bound(mat->vec[i].vec[l1].coeffs, 0, MLDSA_N, 0, MLDSA_Q)))+ requires(forall(l2, 0, MLDSA_L,+ array_abs_bound(v->vec[l2].coeffs, 0, MLDSA_N, MLD_NTT_BOUND)))+ assigns(memory_slice(t_row, sizeof(mld_poly)))+ ensures(array_abs_bound(t_row->coeffs, 0, MLDSA_N, MLDSA_Q))+);++#if !defined(MLD_CONFIG_NO_SIGN_API)+/**+ * Compute w = invNTT(A * NTT(y)) for the signing y vector.+ *+ * The eager variant copies y into the scratch polyvecl, NTTs it in place,+ * calls the standard matrix-vector multiply, and finally inverse-NTTs the+ * result into w.+ *+ * @param[out] w Pointer to output vector.+ * @param[in] mat Pointer to input matrix.+ * @param[in] y Pointer to (non-NTT) y vector.+ * @param[out] scratch Scratch polyvecl for NTT'd copy of y.+ */+MLD_INTERNAL_API+void mld_polyvec_matrix_pointwise_montgomery_yvec_eager(mld_polyveck *w,+ mld_polymat_eager *mat,+ const mld_yvec_eager *y,+ mld_polyvecl *scratch)+__contract__(+ requires(memory_no_alias(w, sizeof(mld_polyveck)))+ requires(memory_no_alias(mat, sizeof(mld_polymat_eager)))+ requires(memory_no_alias(y, sizeof(mld_yvec_eager)))+ requires(memory_no_alias(scratch, sizeof(mld_polyvecl)))+ requires(forall(k1, 0, MLDSA_K, forall(l1, 0, MLDSA_L,+ array_bound(mat->vec[k1].vec[l1].coeffs, 0, MLDSA_N, 0, MLDSA_Q))))+ requires(forall(l2, 0, MLDSA_L,+ array_bound(y->vec.vec[l2].coeffs, 0, MLDSA_N, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1)))+ assigns(memory_slice(w, sizeof(mld_polyveck)))+ assigns(memory_slice(scratch, sizeof(mld_polyvecl)))+ ensures(forall(k0, 0, MLDSA_K,+ array_abs_bound(w->vec[k0].coeffs, 0, MLDSA_N, MLD_INTT_BOUND)))+);+#endif /* !MLD_CONFIG_NO_SIGN_API */+#endif /* !MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST */++#if defined(MLD_CONFIG_REDUCE_RAM) || defined(MLD_UNIT_TEST)+MLD_INTERNAL_API+void mld_polyvec_matrix_expand_lazy(mld_polymat_lazy *mat,+ const uint8_t rho[MLDSA_SEEDBYTES])+__contract__(+ requires(memory_no_alias(mat, sizeof(mld_polymat_lazy)))+ requires(memory_no_alias(rho, MLDSA_SEEDBYTES))+ assigns(memory_slice(mat, sizeof(mld_polymat_lazy)))+);++#if !defined(MLD_CONFIG_NO_KEYPAIR_API) || !defined(MLD_CONFIG_NO_VERIFY_API)+/**+ * Compute row i of matrix-vector multiplication in NTT domain with pointwise+ * multiplication and multiplication by 2^{-32}.+ *+ * Input vector must be in NTT domain representation; the matrix entries are+ * sampled on demand from the seed stored in mat->rho, using mat->cur as+ * scratch. Output coefficients are bounded by MLDSA_Q in absolute value.+ *+ * @param[out] t_row Pointer to output row polynomial.+ * @param[in,out] mat Pointer to input matrix (seed + scratch).+ * @param[in] v Pointer to input vector v.+ * @param i Row index, 0 <= i < MLDSA_K.+ */+MLD_INTERNAL_API+void mld_polyvec_matrix_pointwise_montgomery_row_lazy(mld_poly *t_row,+ mld_polymat_lazy *mat,+ const mld_polyvecl *v,+ unsigned int i)+__contract__(+ requires(memory_no_alias(t_row, sizeof(mld_poly)))+ requires(memory_no_alias(mat, sizeof(mld_polymat_lazy)))+ requires(memory_no_alias(v, sizeof(mld_polyvecl)))+ requires(i < MLDSA_K)+ requires(forall(l1, 0, MLDSA_L,+ array_abs_bound(v->vec[l1].coeffs, 0, MLDSA_N, MLD_NTT_BOUND)))+ assigns(memory_slice(t_row, sizeof(mld_poly)))+ assigns(memory_slice(mat, sizeof(mld_polymat_lazy)))+ ensures(array_abs_bound(t_row->coeffs, 0, MLDSA_N, MLDSA_Q))+);+#endif /* !MLD_CONFIG_NO_KEYPAIR_API || !MLD_CONFIG_NO_VERIFY_API */++#if !defined(MLD_CONFIG_NO_SIGN_API)+/**+ * Compute w = invNTT(A * NTT(y)) for the signing y vector.+ *+ * The lazy variant samples one column of y at a time, NTTs it into+ * &scratch->vec[0], and accumulates the matrix-vector product+ * column-by-column with on-demand sampling of A[k][l]. Only the first poly of+ * the polyvecl scratch is used; the polyvecl type is shared with the eager+ * variant for API uniformity (the storage is provided "for free" by the+ * caller's polyveck/polyvecl union in REDUCE_RAM mode).+ *+ * @param[out] w Pointer to output vector.+ * @param[in,out] mat Pointer to input matrix.+ * @param[in] y Pointer to y seed/kappa.+ * @param[out] scratch Scratch (only &scratch->vec[0] used).+ */+MLD_INTERNAL_API+void mld_polyvec_matrix_pointwise_montgomery_yvec_lazy(mld_polyveck *w,+ mld_polymat_lazy *mat,+ const mld_yvec_lazy *y,+ mld_polyvecl *scratch)+__contract__(+ requires(memory_no_alias(w, sizeof(mld_polyveck)))+ requires(memory_no_alias(mat, sizeof(mld_polymat_lazy)))+ requires(memory_no_alias(y, sizeof(mld_yvec_lazy)))+ requires(memory_no_alias(scratch, sizeof(mld_polyvecl)))+ requires(memory_no_alias(y->rhoprime, MLDSA_CRHBYTES))+ requires(y->kappa <= MLD_MAX_KAPPA)+ assigns(memory_slice(w, sizeof(mld_polyveck)))+ assigns(memory_slice(mat, sizeof(mld_polymat_lazy)))+ assigns(memory_slice(scratch, sizeof(mld_polyvecl)))+ ensures(forall(k0, 0, MLDSA_K,+ array_abs_bound(w->vec[k0].coeffs, 0, MLDSA_N, MLD_INTT_BOUND)))+);+#endif /* !MLD_CONFIG_NO_SIGN_API */+#endif /* MLD_CONFIG_REDUCE_RAM || MLD_UNIT_TEST */++/* Dispatch: typedef and define based on MLD_CONFIG_REDUCE_RAM */+#if defined(MLD_CONFIG_REDUCE_RAM)+typedef mld_sk_s1hat_lazy mld_sk_s1hat;+typedef mld_sk_s2hat_lazy mld_sk_s2hat;+typedef mld_sk_t0hat_lazy mld_sk_t0hat;+typedef mld_polymat_lazy mld_polymat;+typedef mld_yvec_lazy mld_yvec;+#define mld_unpack_sk_s1hat mld_unpack_sk_s1hat_lazy+#define mld_unpack_sk_s2hat mld_unpack_sk_s2hat_lazy+#define mld_unpack_sk_t0hat mld_unpack_sk_t0hat_lazy+#if !defined(MLD_CONFIG_NO_SIGN_API)+#define mld_sk_s1hat_get_poly mld_sk_s1hat_get_poly_lazy+#define mld_sk_s2hat_get_poly mld_sk_s2hat_get_poly_lazy+#define mld_sk_t0hat_get_poly mld_sk_t0hat_get_poly_lazy+#endif+#define mld_polyvec_matrix_expand mld_polyvec_matrix_expand_lazy+#define mld_polyvec_matrix_pointwise_montgomery_row \+ mld_polyvec_matrix_pointwise_montgomery_row_lazy+#define mld_yvec_init mld_yvec_init_lazy+#define mld_yvec_get_poly mld_yvec_get_poly_lazy+#define mld_polyvec_matrix_pointwise_montgomery_yvec \+ mld_polyvec_matrix_pointwise_montgomery_yvec_lazy+#else /* MLD_CONFIG_REDUCE_RAM */+typedef mld_sk_s1hat_eager mld_sk_s1hat;+typedef mld_sk_s2hat_eager mld_sk_s2hat;+typedef mld_sk_t0hat_eager mld_sk_t0hat;+typedef mld_polymat_eager mld_polymat;+typedef mld_yvec_eager mld_yvec;+#define mld_unpack_sk_s1hat mld_unpack_sk_s1hat_eager+#define mld_unpack_sk_s2hat mld_unpack_sk_s2hat_eager+#define mld_unpack_sk_t0hat mld_unpack_sk_t0hat_eager+#if !defined(MLD_CONFIG_NO_SIGN_API)+#define mld_sk_s2hat_get_poly mld_sk_s2hat_get_poly_eager+#define mld_sk_s1hat_get_poly mld_sk_s1hat_get_poly_eager+#define mld_sk_t0hat_get_poly mld_sk_t0hat_get_poly_eager+#endif+#define mld_polyvec_matrix_expand mld_polyvec_matrix_expand_eager+#define mld_polyvec_matrix_pointwise_montgomery_row \+ mld_polyvec_matrix_pointwise_montgomery_row_eager+#define mld_yvec_init mld_yvec_init_eager+#define mld_yvec_get_poly mld_yvec_get_poly_eager+#define mld_polyvec_matrix_pointwise_montgomery_yvec \+ mld_polyvec_matrix_pointwise_montgomery_yvec_eager+#endif /* !MLD_CONFIG_REDUCE_RAM */++#endif /* !MLD_CONFIG_NO_KEYPAIR_API || !MLD_CONFIG_NO_SIGN_API || \+ !MLD_CONFIG_NO_VERIFY_API */+#endif /* !MLD_POLYVEC_LAZY_H */
@@ -0,0 +1,26 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+#ifndef MLD_RANDOMBYTES_H+#define MLD_RANDOMBYTES_H++#include <stddef.h>++#include "cbmc.h"+#include "common.h"++#if !defined(MLD_CONFIG_NO_RANDOMIZED_API)+#if !defined(MLD_CONFIG_CUSTOM_RANDOMBYTES)+MLD_MUST_CHECK_RETURN_VALUE+int randombytes(uint8_t *out, size_t outlen);++MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_randombytes(uint8_t *out, size_t outlen)+__contract__(+ requires(memory_no_alias(out, outlen))+ assigns(memory_slice(out, outlen))+) { return randombytes(out, outlen); }+#endif /* !MLD_CONFIG_CUSTOM_RANDOMBYTES */+#endif /* !MLD_CONFIG_NO_RANDOMIZED_API */+#endif /* !MLD_RANDOMBYTES_H */
@@ -0,0 +1,144 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS204]+ * FIPS 204 Module-Lattice-Based Digital Signature Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/204/final+ */++#ifndef MLD_REDUCE_H+#define MLD_REDUCE_H++#include "cbmc.h"+#include "common.h"+#include "ct.h"+#include "debug.h"++/* check-magic: -4186625 == pow(2,32,MLDSA_Q) */+#define MLD_MONT (-4186625)++/* Upper bound for domain of mld_reduce32() */+#define MLD_REDUCE32_DOMAIN_MAX (INT32_MAX - ((int32_t)1 << 22))++/* Absolute bound for range of mld_reduce32() */+/* check-magic: 6283009 == (MLD_REDUCE32_DOMAIN_MAX - 255 * MLDSA_Q + 1) */+#define MLD_REDUCE32_RANGE_MAX 6283009++/**+ * Generic Montgomery reduction; given a 64-bit integer a, computes a 32-bit+ * integer congruent to a * R^-1 mod MLDSA_Q, where R=2^32.+ *+ * @spec{Implements @[FIPS204, Algorithm 49, MontgomeryReduce].}+ *+ * @param a Input integer to be reduced, of absolute value smaller or equal+ * to INT64_MAX - 2^31 * MLDSA_Q.+ *+ * @return Integer congruent to a * R^-1 modulo MLDSA_Q, with absolute value+ * <= |a| / 2^32 + MLDSA_Q / 2.+ * In particular, if |a| < 2^31 * MLDSA_Q, the absolute value of the+ * return value is < MLDSA_Q.+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int32_t mld_montgomery_reduce(int64_t a)+__contract__(+ /* We don't attempt to express an input-dependent output bound+ * as the post-condition here, as all call-sites satisfy the+ * absolute input bound 2^31 * MLDSA_Q and higher-level+ * reasoning can be conducted using |return_value| < MLDSA_Q. */+ requires(a > -(((int64_t)1 << 31) * MLDSA_Q) &&+ a < (((int64_t)1 << 31) * MLDSA_Q))+ ensures(return_value > -MLDSA_Q && return_value < MLDSA_Q)+)+{+ /* check-magic: 58728449 == unsigned_mod(pow(MLDSA_Q, -1, 2^32), 2^32) */+ const uint64_t QINV = 58728449;++ /* Compute a*q^{-1} mod 2^32 in unsigned representatives */+ const uint32_t a_reduced = mld_cast_int64_to_uint32(a);+ const uint32_t a_inverted = (a_reduced * QINV) & UINT32_MAX;++ /* Lift to signed canonical representative mod 2^32. */+ const int32_t t = mld_cast_uint32_to_int32(a_inverted);++ int64_t r;++ mld_assert(a < +(INT64_MAX - (((int64_t)1 << 31) * MLDSA_Q)) &&+ a > -(INT64_MAX - (((int64_t)1 << 31) * MLDSA_Q)));++ r = a - (int64_t)t * MLDSA_Q;++ /*+ * PORTABILITY: Right-shift on a signed integer is, strictly-speaking,+ * implementation-defined for negative left argument. Here,+ * we assume it's sign-preserving "arithmetic" shift right. (C99 6.5.7 (5))+ */+ r = r >> 32;++ /* Bounds:+ *+ * By construction of the Montgomery multiplication, by the time we+ * compute r >> 32, r is divisible by 2^32, and hence+ *+ * |r >> 32| = |r| / 2^32+ * <= |a| / 2^32 + MLDSA_Q / 2+ *+ * (In general, we would only have |x >> n| <= ceil(|x| / 2^n)).+ *+ * In particular, if |a| < 2^31 * MLDSA_Q, then |return_value| < MLDSA_Q.+ */+ return (int32_t)r;+}++/**+ * For finite field element a with a <= 2^{31} - 2^{22} - 1, compute+ * r congruent to a (mod MLDSA_Q) such that+ * -MLD_REDUCE32_RANGE_MAX <= r < MLD_REDUCE32_RANGE_MAX.+ *+ * @param a Finite field element.+ *+ * @return r.+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int32_t mld_reduce32(int32_t a)+__contract__(+ requires(a <= MLD_REDUCE32_DOMAIN_MAX)+ ensures(return_value >= -MLD_REDUCE32_RANGE_MAX)+ ensures(return_value < MLD_REDUCE32_RANGE_MAX)+)+{+ int32_t t;++ t = (a + ((int32_t)1 << 22)) >> 23;+ t = a - t * MLDSA_Q;+ mld_assert((t - a) % MLDSA_Q == 0);+ return t;+}++/**+ * Add MLDSA_Q if input coefficient is negative.+ *+ * @param a Finite field element.+ *+ * @return r.+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int32_t mld_caddq(int32_t a)+__contract__(+ requires(a > -MLDSA_Q)+ requires(a < MLDSA_Q)+ ensures(return_value >= 0)+ ensures(return_value < MLDSA_Q)+ ensures(return_value == ((a >= 0) ? a : (a + MLDSA_Q)))+)+{+ return mld_ct_sel_int32(a + MLDSA_Q, a, mld_ct_cmask_neg_i32(a));+}+++#endif /* !MLD_REDUCE_H */
@@ -0,0 +1,265 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS204]+ * FIPS 204 Module-Lattice-Based Digital Signature Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/204/final+ */++#ifndef MLD_ROUNDING_H+#define MLD_ROUNDING_H++#include "cbmc.h"+#include "common.h"+#include "ct.h"+#include "debug.h"++/* Parameter set namespacing+ * This is to facilitate building multiple instances+ * of mldsa-native (e.g. with varying parameter sets)+ * within a single compilation unit. */+#define mld_power2round MLD_ADD_PARAM_SET(mld_power2round)+#define mld_decompose MLD_ADD_PARAM_SET(mld_decompose)+#define mld_make_hint MLD_ADD_PARAM_SET(mld_make_hint)+#define mld_use_hint MLD_ADD_PARAM_SET(mld_use_hint)+/* End of parameter set namespacing */++#define MLD_2_POW_D (1 << MLDSA_D)++/**+ * For finite field element a, compute a0, a1 such that+ * a mod^+ MLDSA_Q = a1*2^MLDSA_D + a0 with+ * -2^{MLDSA_D-1} < a0 <= 2^{MLDSA_D-1}. Assumes a to be standard+ * representative.+ *+ * @spec{Implements @[FIPS204, Algorithm 35, Power2Round].}+ *+ * @reference{In the reference implementation, a1 is passed as a return value+ * instead.}+ *+ * @param[out] a0 Pointer to output element a0.+ * @param[out] a1 Pointer to output element a1.+ * @param a Input element.+ */+static MLD_INLINE void mld_power2round(int32_t *a0, int32_t *a1, int32_t a)+__contract__(+ requires(memory_no_alias(a0, sizeof(int32_t)))+ requires(memory_no_alias(a1, sizeof(int32_t)))+ requires(a >= 0 && a < MLDSA_Q)+ assigns(memory_slice(a0, sizeof(int32_t)))+ assigns(memory_slice(a1, sizeof(int32_t)))+ ensures(*a0 > -(MLD_2_POW_D/2) && *a0 <= (MLD_2_POW_D/2))+ ensures(*a1 >= 0 && *a1 <= (MLDSA_Q - 1) / MLD_2_POW_D)+ ensures((*a1 * MLD_2_POW_D + *a0 - a) % MLDSA_Q == 0)+)+{+ *a1 = (a + (1 << (MLDSA_D - 1)) - 1) >> MLDSA_D;+ *a0 = a - (*a1 << MLDSA_D);+}++/**+ * For finite field element a, compute high and low bits a0, a1 such that+ * a mod^+ MLDSA_Q = a1 * 2 * MLDSA_GAMMA2 + a0 with+ * -MLDSA_GAMMA2 < a0 <= MLDSA_GAMMA2 except if+ * a1 = (MLDSA_Q-1)/(MLDSA_GAMMA2*2) where we set a1 = 0 and+ * -MLDSA_GAMMA2 <= a0 = a mod^+ MLDSA_Q - MLDSA_Q < 0. Assumes a to be+ * standard representative.+ *+ * @spec{Implements @[FIPS204, Algorithm 36, Decompose].}+ *+ * @reference{In the reference implementation, a1 is passed as a return value+ * instead.}+ *+ * @param[out] a0 Pointer to output element a0.+ * @param[out] a1 Pointer to output element a1.+ * @param a Input element.+ */+static MLD_INLINE void mld_decompose(int32_t *a0, int32_t *a1, int32_t a)+__contract__(+ requires(memory_no_alias(a0, sizeof(int32_t)))+ requires(memory_no_alias(a1, sizeof(int32_t)))+ requires(a >= 0 && a < MLDSA_Q)+ assigns(memory_slice(a0, sizeof(int32_t)))+ assigns(memory_slice(a1, sizeof(int32_t)))+ /* a0 = -MLDSA_GAMMA2 occurs exactly when a = MLDSA_Q - MLDSA_GAMMA2: the+ * border case of Decompose where a1 = (MLDSA_Q-1)/(2*MLDSA_GAMMA2) is+ * wrapped to 0 and a0 = a - MLDSA_Q (@[FIPS204, Algorithm 36, Decompose]) */+ ensures(*a0 >= -MLDSA_GAMMA2 && *a0 <= MLDSA_GAMMA2)+ ensures(*a1 >= 0 && *a1 < (MLDSA_Q-1)/(2*MLDSA_GAMMA2))+ ensures((*a1 * 2 * MLDSA_GAMMA2 + *a0 - a) % MLDSA_Q == 0)+)+{+ /*+ * The goal is to compute f1 = round-(f / (2*GAMMA2)), which can be computed+ * alternatively as round-(f / (128B)) = round-(ceil(f / 128) / B) where+ * B = 2*GAMMA2 / 128. Here round-() denotes "round half down".+ *+ * The equality round-(f / (128B)) = round-(ceil(f / 128) / B) can deduced+ * as follows. Since changing f to align-up(f, 128) can move f onto but not+ * across a rounding boundary for division by 128*B (note that we need B to be+ * even for this to work), and round- rounds down on the boundary, we have+ *+ * round-(f / (128B)) = round-(align-up(f, 128) / (128B))+ * = round-((align-up(f, 128) / 128) / B)+ * = round-(ceil(f / 128) / B).+ */+ *a1 = (a + 127) >> 7;+ /* We know a >= 0 and a < MLDSA_Q, so... */+ /* check-magic: 65472 == round((MLDSA_Q-1)/128) */+ mld_assert(*a1 >= 0 && *a1 <= 65472);++#if MLD_CONFIG_PARAMETER_SET == 44+ /* check-magic: 1488 == 2 * intdiv(intdiv(MLDSA_Q - 1, 88), 128) */+ /* check-magic: 11275 == floor(2**24 / 1488) */+ /* check-magic: 1560281088 == 1 / (1 / 1488 - 11275 / 2**24) */+ /*+ * Compute f1 = round-(f1' / B) ≈ round(f1' * 11275 / 2^24). This is exact for+ * 0 <= f1' < 2^16.+ *+ * To see this, consider the (signed) error f1' * (1 / B - 11275 / 2^24)+ * between f1' / B and the (under-)approximation f1' * 11275 / 2^24. Because+ * eps := 1 / B - 11275 / 2^24 is 1 / 1560281088 ≈ 2^(-30.54) < 2^(-30), we+ * have 0 <= f1' * eps < 2^16 * 2^(-30) = 1 / 2^14 < 1 / 2^11 < 1 / B (note+ * that f1' is non-negative).+ *+ * On the other hand, 1 / B is the spacing between the integral multiples+ * of 1 / B, which includes all rounding boundaries n + 0.5 (since B is even).+ * Hence, if f1' / B is not of the form n + 0.5, then it is at least 1 / B+ * away from the nearest rounding boundary, so moving from f1' / B to+ * f1' * 11275 / 2^24 does not affect the rounding result, no matter the type+ * of rounding used in either side. In particular, we have round-(f1' / B) =+ * round(f1' * 11275 / 2^24) as claimed.+ *+ * As for the remaining case where f1' / B _is_ of the form n + 0.5, because+ * f1' * 11275 / 2^24 is slightly but strictly below f1' / B = n + 0.5 (note+ * that f1' and thus the error f1' * eps cannot be 0 here), it is always+ * rounded down to n. More precisely, we have round-(f1' / B) =+ * round(f1' * 11275 / 2^24), where the round-down on the LHS is essential,+ * and on the RHS the type of rounding again does not matter. This concludes+ * the proof.+ *+ * See proofs/isabelle/compress for a formalization of the above argument.+ */+ *a1 = (*a1 * 11275 + ((int32_t)1 << 23)) >> 24;+ mld_assert(*a1 >= 0 && *a1 <= 44);++ *a1 = mld_ct_sel_int32(0, *a1, mld_ct_cmask_neg_i32(43 - *a1));+ mld_assert(*a1 >= 0 && *a1 <= 43);+#else /* MLD_CONFIG_PARAMETER_SET == 44 */+ /* check-magic: 4092 == 2 * intdiv(intdiv(MLDSA_Q - 1, 32), 128) */+ /* check-magic: 1025 == floor(2**22 / 4092) */+ /* check-magic: 4290772992 == 1 / (1 / 4092 - 1025 / 2**22) */+ /*+ * Compute f1 = round-(f1' / B) ≈ round(f1' * 1025 / 2^22). This is exact for+ * 0 <= f1' < 2^16. Following the same argument above, it suffices to show+ * that f1' * eps < 1 / B, where eps := 1 / B - 1025 / 2^22. Indeed, we have+ * eps = 1 / 4290772992 ≈ 2^(-31.99) < 2^(-31), therefore f1' * eps <+ * 2^16 * 2^(-31) = 1 / 2^15 < 1 / 2^12 < 1 / B.+ */+ *a1 = (*a1 * 1025 + ((int32_t)1 << 21)) >> 22;+ mld_assert(*a1 >= 0 && *a1 <= 16);++ *a1 &= 15;+ mld_assert(*a1 >= 0 && *a1 <= 15);++#endif /* MLD_CONFIG_PARAMETER_SET != 44 */++ *a0 = a - *a1 * 2 * MLDSA_GAMMA2;+ *a0 = mld_ct_sel_int32(*a0 - MLDSA_Q, *a0,+ mld_ct_cmask_neg_i32((MLDSA_Q - 1) / 2 - *a0));+}++/**+ * Decide a single hint bit from the low part a0 and high part a1 of a+ * coefficient: return 1 unless a0 lies in the range (-GAMMA2, GAMMA2] that+ * LowBits would produce, with the boundary value -GAMMA2 also admitted when+ * a1 == 0 (the Decompose border case).+ *+ * @note This is not a line-for-line implementation of FIPS 204's MakeHint(z, r)+ * (@[FIPS204, Algorithm 39, MakeHint]), which takes two ring elements and+ * returns [[HighBits(r) != HighBits(r + z)]]. Instead, it takes the already+ * decomposed low/high parts (a0, a1) of a coefficient and decides the hint bit+ * from them directly. As explained in the block comment of+ * mld_attempt_signature_generation (sign.c), for the specific values that arise+ * during signing -- a0 = w0 - cs2 + ct0 and a1 = w1 = HighBits(w) -- this is+ * equivalent to the spec's MakeHint(-ct0, w - cs2 + ct0) coefficient-wise.+ * Because it consumes (a0, a1) rather than (z, r), it relies on the caller+ * having computed a compatible decomposition.+ *+ * @param a0 Low bits of input element.+ * @param a1 High bits of input element.+ *+ * @return 1 if overflow, 0 otherwise.+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE unsigned int mld_make_hint(int32_t a0, int32_t a1)+__contract__(+ ensures(return_value >= 0 && return_value <= 1)+ ensures(return_value == (a0 > MLDSA_GAMMA2 || a0 < -MLDSA_GAMMA2 ||+ (a0 == -MLDSA_GAMMA2 && a1 != 0)))+)+{+ if (a0 > MLDSA_GAMMA2 || a0 < -MLDSA_GAMMA2 ||+ (a0 == -MLDSA_GAMMA2 && a1 != 0))+ {+ return 1;+ }++ return 0;+}++/**+ * Correct high bits according to hint.+ *+ * @spec{Implements @[FIPS204, Algorithm 40, UseHint].}+ *+ * @param a Input element.+ * @param hint Hint bit.+ *+ * @return Corrected high bits.+ */+MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int32_t mld_use_hint(int32_t a, int32_t hint)+__contract__(+ requires(hint >= 0 && hint <= 1)+ requires(a >= 0 && a < MLDSA_Q)+ ensures(return_value >= 0 && return_value < (MLDSA_Q-1)/(2*MLDSA_GAMMA2))+)+{+ int32_t a0, a1;++ mld_decompose(&a0, &a1, a);+ if (hint == 0)+ {+ return a1;+ }++#if MLD_CONFIG_PARAMETER_SET == 44+ if (a0 > 0)+ {+ return (a1 == 43) ? 0 : a1 + 1;+ }+ else+ {+ return (a1 == 0) ? 43 : a1 - 1;+ }+#else /* MLD_CONFIG_PARAMETER_SET == 44 */+ if (a0 > 0)+ {+ return (a1 + 1) & 15;+ }+ else+ {+ return (a1 - 1) & 15;+ }+#endif /* MLD_CONFIG_PARAMETER_SET != 44 */+}+++#endif /* !MLD_ROUNDING_H */
@@ -0,0 +1,1720 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS140_3_IG]+ * Implementation Guidance for FIPS 140-3 and the Cryptographic Module+ * Validation Program+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/projects/cryptographic-module-validation-program/fips-140-3-ig-announcements+ *+ * - [FIPS204]+ * FIPS 204 Module-Lattice-Based Digital Signature Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/204/final+ *+ * - [FIPS204_UPDATES]+ * FIPS 204 Potential Updates (Errata)+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/files/pubs/fips/204/final/docs/fips-204-potential-updates.xlsx+ *+ * - [Round3_Spec]+ * CRYSTALS-Dilithium Algorithm Specifications and Supporting Documentation+ * (Version 3.1)+ * Bai, Ducas, Kiltz, Lepoint, Lyubashevsky, Schwabe, Seiler, Stehlé+ * https://pq-crystals.org/dilithium/data/dilithium-specification-round3-20210208.pdf+ */++#include "sign.h"++#include "cbmc.h"+#include "ct.h"+#include "debug.h"+#include "packing.h"+#include "poly.h"+#include "poly_kl.h"+#include "polyvec.h"+#include "randombytes.h"+#include "symmetric.h"++/* Parameter set namespacing+ * This is to facilitate building multiple instances+ * of mldsa-native (e.g. with varying parameter sets)+ * within a single compilation unit. */+#define mld_check_pct MLD_ADD_PARAM_SET(mld_check_pct) MLD_CONTEXT_PARAMETERS_2+#define mld_sample_s1_s2 MLD_ADD_PARAM_SET(mld_sample_s1_s2)+#define mld_validate_hash_length MLD_ADD_PARAM_SET(mld_validate_hash_length)+#define mld_get_hash_oid MLD_ADD_PARAM_SET(mld_get_hash_oid)+#define mld_H MLD_ADD_PARAM_SET(mld_H)+#define mld_compute_pack_z MLD_ADD_PARAM_SET(mld_compute_pack_z)+#define mld_attempt_signature_generation \+ MLD_ADD_PARAM_SET(mld_attempt_signature_generation) MLD_CONTEXT_PARAMETERS_8+#define mld_compute_pack_t0_t1 \+ MLD_ADD_PARAM_SET(mld_compute_pack_t0_t1) MLD_CONTEXT_PARAMETERS_5+#define mld_get_max_signing_attempts \+ MLD_ADD_PARAM_SET(mld_get_max_signing_attempts)++#if !defined(MLD_CONFIG_NO_KEYPAIR_API)+static int mld_check_pct(uint8_t const pk[MLDSA_CRYPTO_PUBLICKEYBYTES],+ uint8_t const sk[MLDSA_CRYPTO_SECRETKEYBYTES],+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+__contract__(+ requires(memory_no_alias(pk, MLDSA_CRYPTO_PUBLICKEYBYTES))+ requires(memory_no_alias(sk, MLDSA_CRYPTO_SECRETKEYBYTES))+ ensures(return_value == 0 || return_value == MLD_ERR_OUT_OF_MEMORY ||+ return_value == MLD_ERR_RNG_FAIL ||+ return_value == MLD_ERR_SIGNING_PAUSED ||+ return_value == MLD_ERR_PCT_FAIL)+);++#if defined(MLD_CONFIG_KEYGEN_PCT)+/**+ * Pair-wise Consistency Test (PCT) for DSA keypairs.+ *+ * @[FIPS140_3_IG] TE10.35.02+ * (https://csrc.nist.gov/csrc/media/Projects/cryptographic-module-validation-program/documents/fips%20140-3/FIPS%20140-3%20IG.pdf).+ *+ * Validates that a generated public/private key pair can correctly sign and+ * verify data. Performs signature generation using the private key (sk),+ * followed by signature verification using the public key (pk).+ *+ * @note @[FIPS204] requires that public/private key pairs are to be used+ * only for the calculation and/or verification of digital signatures.+ *+ * @param[in] pk Public key.+ * @param[in] sk Secret key.+ * @param context Application context. Only present when+ * MLD_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLD_ERR_OUT_OF_MEMORY MLD_CONFIG_CUSTOM_ALLOC_FREE was used and an+ * allocation via MLD_CUSTOM_ALLOC returned NULL.+ * @retval MLD_ERR_RNG_FAIL Random number generation failed.+ * @retval MLD_ERR_SIGNING_PAUSED The PCT's signing step was paused by a+ * MLD_CONFIG_SIGN_HOOK_ATTEMPT hook.+ * @retval MLD_ERR_PCT_FAIL The consistency check failed.+ */+static int mld_check_pct(uint8_t const pk[MLDSA_CRYPTO_PUBLICKEYBYTES],+ uint8_t const sk[MLDSA_CRYPTO_SECRETKEYBYTES],+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+{+ MLD_ALIGN uint8_t message[1] = {0};+ int ret;+ MLD_ALLOC(signature, uint8_t, MLDSA_CRYPTO_BYTES, context);+ MLD_ALLOC(pk_test, uint8_t, MLDSA_CRYPTO_PUBLICKEYBYTES, context);++ if (signature == NULL || pk_test == NULL)+ {+ ret = MLD_ERR_OUT_OF_MEMORY;+ goto cleanup;+ }++ /* Copy public key for testing */+ mld_memcpy(pk_test, pk, MLDSA_CRYPTO_PUBLICKEYBYTES);++ /* Sign a test message using the original secret key */+ ret = mld_sign_signature(signature, message, sizeof(message), NULL, 0, sk,+ context);+ if (ret != 0)+ {+ goto cleanup;+ }++#if defined(MLD_CONFIG_KEYGEN_PCT_BREAKAGE_TEST)+ /* Deliberately break public key for testing purposes */+ if (mld_break_pct())+ {+ pk_test[0] = ~pk_test[0];+ }+#endif /* MLD_CONFIG_KEYGEN_PCT_BREAKAGE_TEST */++ /* Verify the signature using the (potentially corrupted) public key. */+ ret = mld_sign_verify(signature, message, sizeof(message), NULL, 0, pk_test,+ context);+ if (ret != 0)+ {+ goto cleanup;+ }++cleanup:+ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ MLD_FREE(pk_test, uint8_t, MLDSA_CRYPTO_PUBLICKEYBYTES, context);+ MLD_FREE(signature, uint8_t, MLDSA_CRYPTO_BYTES, context);++ /* A failed signing operation or an invalid signature hint at a faulty+ * implementation and map to a dedicated error code for PCT failure. */+ if (ret == MLD_ERR_INVALID_SIGNATURE ||+ ret == MLD_ERR_SIGN_ATTEMPTS_EXHAUSTED)+ {+ ret = MLD_ERR_PCT_FAIL;+ }++ /* Other error codes, e.g. platform failures like out of memory or+ * randomness failure, are passed on unmodified. */++ return ret;+}+#else /* MLD_CONFIG_KEYGEN_PCT */+static int mld_check_pct(uint8_t const pk[MLDSA_CRYPTO_PUBLICKEYBYTES],+ uint8_t const sk[MLDSA_CRYPTO_SECRETKEYBYTES],+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+{+ /* Skip PCT */+ ((void)pk);+ ((void)sk);+ MLD_CONTEXT_UNUSED(context);+ return 0;+}+#endif /* !MLD_CONFIG_KEYGEN_PCT */++/**+ * Sample the short secret vectors s1 (length MLDSA_L) and s2 (length MLDSA_K)+ * with coefficients in [-MLDSA_ETA, MLDSA_ETA] from the seed.+ *+ * @spec{Implements @[FIPS204, Algorithm 33, ExpandS].}+ *+ * @param[out] s1 Output vector s1.+ * @param[out] s2 Output vector s2.+ * @param[in] seed Byte array with seed of length MLDSA_CRHBYTES.+ */+static void mld_sample_s1_s2(mld_polyvecl *s1, mld_polyveck *s2,+ const uint8_t seed[MLDSA_CRHBYTES])+__contract__(+ requires(memory_no_alias(s1, sizeof(mld_polyvecl)))+ requires(memory_no_alias(s2, sizeof(mld_polyveck)))+ requires(memory_no_alias(seed, MLDSA_CRHBYTES))+ assigns(object_whole(s1), object_whole(s2))+ ensures(forall(l0, 0, MLDSA_L, array_abs_bound(s1->vec[l0].coeffs, 0, MLDSA_N, MLDSA_ETA + 1)))+ ensures(forall(k0, 0, MLDSA_K, array_abs_bound(s2->vec[k0].coeffs, 0, MLDSA_N, MLDSA_ETA + 1)))+)+{+/* Sample short vectors s1 and s2 */+#if defined(MLD_CONFIG_SERIAL_FIPS202_ONLY)+ int i;+ uint16_t nonce = 0;+ /* Safety: The nonces are at most 14 (MLDSA_L + MLDSA_K - 1), and, hence, the+ * casts are safe. */+ for (i = 0; i < MLDSA_L; i++)+ {+ mld_poly_uniform_eta(&s1->vec[i], seed, (uint8_t)(nonce + i));+ }+ for (i = 0; i < MLDSA_K; i++)+ {+ mld_poly_uniform_eta(&s2->vec[i], seed, (uint8_t)(nonce + MLDSA_L + i));+ }+#else /* MLD_CONFIG_SERIAL_FIPS202_ONLY */+#if MLD_CONFIG_PARAMETER_SET == 44+ mld_poly_uniform_eta_4x(&s1->vec[0], &s1->vec[1], &s1->vec[2], &s1->vec[3],+ seed, 0, 1, 2, 3);+ mld_poly_uniform_eta_4x(&s2->vec[0], &s2->vec[1], &s2->vec[2], &s2->vec[3],+ seed, 4, 5, 6, 7);+#elif MLD_CONFIG_PARAMETER_SET == 65+ mld_poly_uniform_eta_4x(&s1->vec[0], &s1->vec[1], &s1->vec[2], &s1->vec[3],+ seed, 0, 1, 2, 3);+ mld_poly_uniform_eta_4x(&s1->vec[4], &s2->vec[0], &s2->vec[1],+ &s2->vec[2] /* irrelevant */, seed, 4, 5, 6,+ 0xFF /* irrelevant */);+ mld_poly_uniform_eta_4x(&s2->vec[2], &s2->vec[3], &s2->vec[4], &s2->vec[5],+ seed, 7, 8, 9, 10);+#elif MLD_CONFIG_PARAMETER_SET == 87+ mld_poly_uniform_eta_4x(&s1->vec[0], &s1->vec[1], &s1->vec[2], &s1->vec[3],+ seed, 0, 1, 2, 3);+ mld_poly_uniform_eta_4x(&s1->vec[4], &s1->vec[5], &s1->vec[6],+ &s2->vec[0] /* irrelevant */, seed, 4, 5, 6,+ 0xFF /* irrelevant */);+ mld_poly_uniform_eta_4x(&s2->vec[0], &s2->vec[1], &s2->vec[2], &s2->vec[3],+ seed, 7, 8, 9, 10);+ mld_poly_uniform_eta_4x(&s2->vec[4], &s2->vec[5], &s2->vec[6], &s2->vec[7],+ seed, 11, 12, 13, 14);+#endif /* MLD_CONFIG_PARAMETER_SET == 87 */+#endif /* !MLD_CONFIG_SERIAL_FIPS202_ONLY */+}++/**+ * Compute t = A*s1hat + s2 row by row, decompose each row into t0[k] and+ * t1[k] via power2round, and bit-pack t1[k] into pk_t1 and t0[k] into the+ * t0_packed buffer. Used by both keygen and pk_from_sk.+ *+ * @spec{Partially implements @[FIPS204, Algorithm 22, pkEncode] (t1) and+ * @[FIPS204, Algorithm 24, skEncode] (t0).}+ *+ * @param[out] pk_t1 Output buffer for packed t1 (size+ * MLDSA_K * MLDSA_POLYT1_PACKEDBYTES; i.e. the t1+ * region of pk).+ * @param[out] t0_packed Output buffer for packed t0 (size+ * MLDSA_K * MLDSA_POLYT0_PACKEDBYTES).+ * @param[in] s1hat s1 in NTT domain.+ * @param[in] s2 s2.+ * @param[in] rho Byte array containing seed rho.+ * @param context Application context. Only present when+ * MLD_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @return - 0: Success.+ * - MLD_ERR_OUT_OF_MEMORY: If MLD_CONFIG_CUSTOM_ALLOC_FREE is used and+ * an allocation via MLD_CUSTOM_ALLOC returned NULL.+ */+MLD_MUST_CHECK_RETURN_VALUE+static int mld_compute_pack_t0_t1(+ uint8_t pk_t1[MLDSA_K * MLDSA_POLYT1_PACKEDBYTES],+ uint8_t t0_packed[MLDSA_K * MLDSA_POLYT0_PACKEDBYTES],+ const mld_polyvecl *s1hat, const mld_polyveck *s2,+ const uint8_t rho[MLDSA_SEEDBYTES],+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+__contract__(+ requires(memory_no_alias(pk_t1, MLDSA_K * MLDSA_POLYT1_PACKEDBYTES))+ requires(memory_no_alias(t0_packed, MLDSA_K * MLDSA_POLYT0_PACKEDBYTES))+ requires(memory_no_alias(s1hat, sizeof(mld_polyvecl)))+ requires(memory_no_alias(s2, sizeof(mld_polyveck)))+ requires(memory_no_alias(rho, MLDSA_SEEDBYTES))+ requires(forall(l1, 0, MLDSA_L,+ array_abs_bound(s1hat->vec[l1].coeffs, 0, MLDSA_N, MLD_NTT_BOUND)))+ requires(forall(k2, 0, MLDSA_K,+ array_bound(s2->vec[k2].coeffs, 0, MLDSA_N,+ MLD_POLYETA_UNPACK_LOWER_BOUND, MLDSA_ETA + 1)))+ assigns(memory_slice(pk_t1, MLDSA_K * MLDSA_POLYT1_PACKEDBYTES))+ assigns(memory_slice(t0_packed, MLDSA_K * MLDSA_POLYT0_PACKEDBYTES))+ ensures(return_value == 0 || return_value == MLD_ERR_OUT_OF_MEMORY))+{+ unsigned int k;+ int ret;+ MLD_ALLOC(mat, mld_polymat, 1, context);+ MLD_ALLOC(t0k, mld_poly, 1, context);+ MLD_ALLOC(t1k, mld_poly, 1, context);++ if (mat == NULL || t0k == NULL || t1k == NULL)+ {+ ret = MLD_ERR_OUT_OF_MEMORY;+ goto cleanup;+ }++ /* Expand matrix */+ mld_polyvec_matrix_expand(mat, rho);++ for (k = 0; k < MLDSA_K; k++)+ __loop__(+ assigns(k, memory_slice(pk_t1, MLDSA_K * MLDSA_POLYT1_PACKEDBYTES),+ memory_slice(t0_packed, MLDSA_K * MLDSA_POLYT0_PACKEDBYTES),+ memory_slice(t0k, sizeof(mld_poly)),+ memory_slice(t1k, sizeof(mld_poly))+ MLD_IF_REDUCE_RAM(, memory_slice(mat, sizeof(mld_polymat))))+ invariant(k <= MLDSA_K)+ decreases(MLDSA_K - k)+ )+ {+ /* t0k = (A * s1hat)_k in NTT domain */+ mld_polyvec_matrix_pointwise_montgomery_row(t0k, mat, s1hat, k);++ /* t0k = invNTT(t0k) */+ mld_poly_invntt_tomont(t0k);++ /* t0k += s2[k] */+ mld_poly_add(t0k, &s2->vec[k]);++ /* Reference: The following reduction is not present in the reference+ * implementation. Omitting this reduction requires the output+ * of the invntt to be small enough such that the addition of+ * s2 does not result in absolute values >= MLDSA_Q. While our+ * C, x86_64, and AArch64 invntt implementations produce small+ * enough values for this to work out, it complicates the+ * bounds reasoning. We instead add an additional reduction,+ * and can consequently, relax the bounds requirements for the+ * invntt.+ */+ mld_poly_reduce(t0k);++ /* Decompose into t1[k] and t0[k] (in place into t0k). */+ mld_poly_caddq(t0k);+ mld_poly_power2round(t1k, t0k, t0k);++ /* Pack t1[k] into pk and t0[k] into the t0 output buffer. */+ mld_polyt1_pack(pk_t1 + k * MLDSA_POLYT1_PACKEDBYTES, t1k);+ mld_polyt0_pack(t0_packed + k * MLDSA_POLYT0_PACKEDBYTES, t0k);+ }++ ret = 0;+cleanup:+ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ MLD_FREE(t1k, mld_poly, 1, context);+ MLD_FREE(t0k, mld_poly, 1, context);+ MLD_FREE(mat, mld_polymat, 1, context);+ return ret;+}++MLD_MUST_CHECK_RETURN_VALUE+MLD_EXTERNAL_API+int mld_sign_keypair_internal(uint8_t pk[MLDSA_CRYPTO_PUBLICKEYBYTES],+ uint8_t sk[MLDSA_CRYPTO_SECRETKEYBYTES],+ const uint8_t seed[MLDSA_SEEDBYTES],+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+{+ int ret;+ const uint8_t *rho, *rhoprime, *key;++ MLD_ALLOC(seedbuf, uint8_t, 2 * MLDSA_SEEDBYTES + MLDSA_CRHBYTES, context);+ MLD_ALLOC(inbuf, uint8_t, MLDSA_SEEDBYTES + 2, context);+ MLD_ALLOC(tr, uint8_t, MLDSA_TRBYTES, context);+ MLD_ALLOC(s1, mld_polyvecl, 1, context);+ MLD_ALLOC(s2, mld_polyveck, 1, context);++ if (seedbuf == NULL || inbuf == NULL || tr == NULL || s1 == NULL ||+ s2 == NULL)+ {+ ret = MLD_ERR_OUT_OF_MEMORY;+ goto cleanup;+ }++ /* Get randomness for rho, rhoprime and key */+ mld_memcpy(inbuf, seed, MLDSA_SEEDBYTES);+ inbuf[MLDSA_SEEDBYTES + 0] = MLDSA_K;+ inbuf[MLDSA_SEEDBYTES + 1] = MLDSA_L;+ mld_shake256(seedbuf, 2 * MLDSA_SEEDBYTES + MLDSA_CRHBYTES, inbuf,+ MLDSA_SEEDBYTES + 2);+ rho = seedbuf;+ rhoprime = rho + MLDSA_SEEDBYTES;+ key = rhoprime + MLDSA_CRHBYTES;++ /* Constant time: rho is part of the public key and, hence, public. */+ MLD_CT_TESTING_DECLASSIFY(rho, MLDSA_SEEDBYTES);++ /* Sample s1 and s2 */+ mld_sample_s1_s2(s1, s2, rhoprime);++ /* Pack s1 into sk before NTT */+ mld_pack_sk_s1(sk, s1);++ /* NTT s1 in place to use as s1hat */+ mld_polyvecl_ntt(s1);++ /* Pack rho into pk */+ mld_memcpy(pk + MLDSA_PK_RHO_OFFSET, rho, MLDSA_SEEDBYTES);++ /* Compute t = A*s1hat + s2 row by row, decompose into t1/t0, and pack+ * t1 into pk and t0 directly into the t0 region of sk. */+ ret = mld_compute_pack_t0_t1(pk + MLDSA_PK_T1_OFFSET, sk + MLDSA_SK_T0_OFFSET,+ s1, s2, rho, context);+ if (ret != 0)+ {+ goto cleanup;+ }++ /* Compute tr = H(pk) */+ mld_shake256(tr, MLDSA_TRBYTES, pk, MLDSA_CRYPTO_PUBLICKEYBYTES);++ /* Pack remaining secret key components (s1 and t0 already packed) */+ mld_pack_sk_rho_key_tr_s2(sk, rho, tr, key, s2);++ /* Constant time: pk is the public key, inherently public data */+ MLD_CT_TESTING_DECLASSIFY(pk, MLDSA_CRYPTO_PUBLICKEYBYTES);++cleanup:+ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ MLD_FREE(s2, mld_polyveck, 1, context);+ MLD_FREE(s1, mld_polyvecl, 1, context);+ MLD_FREE(tr, uint8_t, MLDSA_TRBYTES, context);+ MLD_FREE(inbuf, uint8_t, MLDSA_SEEDBYTES + 2, context);+ MLD_FREE(seedbuf, uint8_t, 2 * MLDSA_SEEDBYTES + MLDSA_CRHBYTES, context);++ /* Pairwise Consistency Test (PCT) @[FIPS140_3_IG, p.87] */+ /* Do this after freeing all temporaries. */+ if (ret == 0)+ {+ ret = mld_check_pct(pk, sk, context);+ }++ if (ret != 0)+ {+ /* Clear caller outputs on failure. */+ mld_zeroize(pk, MLDSA_CRYPTO_PUBLICKEYBYTES);+ mld_zeroize(sk, MLDSA_CRYPTO_SECRETKEYBYTES);+ }++ return ret;+}++#if !defined(MLD_CONFIG_CORE_API_ONLY)+#if !defined(MLD_CONFIG_NO_RANDOMIZED_API)+MLD_MUST_CHECK_RETURN_VALUE+MLD_EXTERNAL_API+int mld_sign_keypair(uint8_t pk[MLDSA_CRYPTO_PUBLICKEYBYTES],+ uint8_t sk[MLDSA_CRYPTO_SECRETKEYBYTES],+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+{+ int ret;+ MLD_ALLOC(seed, uint8_t, MLDSA_SEEDBYTES, context);++ if (seed == NULL)+ {+ ret = MLD_ERR_OUT_OF_MEMORY;+ goto cleanup;+ }++ if (mld_randombytes(seed, MLDSA_SEEDBYTES) != 0)+ {+ ret = MLD_ERR_RNG_FAIL;+ goto cleanup;+ }+ MLD_CT_TESTING_SECRET(seed, MLDSA_SEEDBYTES);+ ret = mld_sign_keypair_internal(pk, sk, seed, context);++cleanup:+ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ MLD_FREE(seed, uint8_t, MLDSA_SEEDBYTES, context);+ return ret;+}+#endif /* !MLD_CONFIG_NO_RANDOMIZED_API */+#endif /* !MLD_CONFIG_CORE_API_ONLY */+#endif /* !MLD_CONFIG_NO_KEYPAIR_API */++#if !defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API)+/**+ * Abstracts application of SHAKE256 to one, two or three blocks of data,+ * yielding a user-requested size of output.+ *+ * @param[out] out Pointer to output.+ * @param outlen Requested output length in bytes.+ * @param[in] in1 Pointer to input block 1. Must NOT be NULL.+ * @param in1len Length of input in1 in bytes.+ * @param[in] in2 Pointer to input block 2. May be NULL if in2len == 0,+ * in which case this block is ignored.+ * @param in2len Length of input in2 in bytes.+ * @param[in] in3 Pointer to input block 3. May be NULL if in3len == 0,+ * in which case this block is ignored.+ * @param in3len Length of input in3 in bytes.+ */+static void mld_H(uint8_t *out, size_t outlen, const uint8_t *in1,+ size_t in1len, const uint8_t *in2, size_t in2len,+ const uint8_t *in3, size_t in3len)+__contract__(+ requires(in1len <= MLD_MAX_BUFFER_SIZE)+ requires(in2len <= MLD_MAX_BUFFER_SIZE)+ requires(in3len <= MLD_MAX_BUFFER_SIZE)+ requires(outlen <= 8 * SHAKE256_RATE /* somewhat arbitrary bound */)+ requires(memory_no_alias(in1, in1len))+ requires(in2len == 0 || memory_no_alias(in2, in2len))+ requires(in3len == 0 || memory_no_alias(in3, in3len))+ requires(memory_no_alias(out, outlen))+ assigns(memory_slice(out, outlen))+)+{+ mld_shake256ctx state;+ mld_shake256_init(&state);+ mld_shake256_absorb(&state, in1, in1len);+ if (in2len != 0)+ {+ mld_shake256_absorb(&state, in2, in2len);+ }+ if (in3len != 0)+ {+ mld_shake256_absorb(&state, in3, in3len);+ }+ mld_shake256_finalize(&state);+ mld_shake256_squeeze(out, outlen, &state);+ mld_shake256_release(&state);++ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ mld_zeroize(&state, sizeof(state));+}+#endif /* !MLD_CONFIG_NO_SIGN_API || !MLD_CONFIG_NO_VERIFY_API */++#if !defined(MLD_CONFIG_NO_SIGN_API)+/* MLD_MAX_KAPPA (see params.h) bounds the rejection-sampling counter kappa;+ * MLD_MAX_SIGNING_ATTEMPTS below turns that into a bound on attempts. */++/**+ * Compute z = y + s1*c, check that z has coefficients smaller than+ * MLDSA_GAMMA1 - MLDSA_BETA, and pack z into the signature buffer.+ *+ * @reference{This function is inlined into mld_sign_signature in the+ * reference implementation.}+ *+ * @param[in,out] sig Output signature.+ * @param[in] cp Challenge polynomial.+ * @param[in] s1hat Secret vector s1 in NTT domain.+ * @param[in] y Masking vector y (or seed in REDUCE_RAM mode).+ * @param[out] z Scratch polynomial for z computation.+ * @param[out] tmp Scratch polynomial.+ *+ * @return - 0: Success (z has coefficients smaller than+ * MLDSA_GAMMA1 - MLDSA_BETA).+ * - MLD_ERR_FAIL: z rejected (norm check failed).+ * - MLD_ERR_OUT_OF_MEMORY: If MLD_CONFIG_CUSTOM_ALLOC_FREE is used and+ * an allocation via MLD_CUSTOM_ALLOC returned NULL.+ */+MLD_MUST_CHECK_RETURN_VALUE+static int mld_compute_pack_z(uint8_t sig[MLDSA_CRYPTO_BYTES],+ const mld_poly *cp, const mld_sk_s1hat *s1hat,+ const mld_yvec *y, mld_poly *z, mld_poly *tmp)+__contract__(+ requires(memory_no_alias(sig, MLDSA_CRYPTO_BYTES))+ requires(memory_no_alias(cp, sizeof(mld_poly)))+ requires(memory_no_alias(s1hat, sizeof(mld_sk_s1hat)))+ requires(memory_no_alias(y, sizeof(mld_yvec)))+ requires(memory_no_alias(z, sizeof(mld_poly)))+ requires(memory_no_alias(tmp, sizeof(mld_poly)))+ requires(array_abs_bound(cp->coeffs, 0, MLDSA_N, MLD_NTT_BOUND))+ MLD_IF_NOT_REDUCE_RAM(+ requires(forall(k0, 0, MLDSA_L,+ array_bound(y->vec.vec[k0].coeffs, 0, MLDSA_N, -(MLDSA_GAMMA1 - 1), MLDSA_GAMMA1 + 1)))+ requires(forall(k1, 0, MLDSA_L, array_abs_bound(s1hat->vec.vec[k1].coeffs, 0, MLDSA_N, MLD_NTT_BOUND)))+ )+ MLD_IF_REDUCE_RAM(+ requires(memory_no_alias(s1hat->packed, MLDSA_L * MLDSA_POLYETA_PACKEDBYTES))+ requires(memory_no_alias(y->rhoprime, MLDSA_CRHBYTES))+ requires(y->kappa <= MLD_MAX_KAPPA)+ )+ assigns(memory_slice(sig, MLDSA_CRYPTO_BYTES))+ assigns(memory_slice(z, sizeof(mld_poly)))+ assigns(memory_slice(tmp, sizeof(mld_poly)))+ ensures(return_value == 0 || return_value == MLD_ERR_FAIL ||+ return_value == MLD_ERR_OUT_OF_MEMORY)+)+{+ unsigned int i;+ uint32_t z_invalid;+ for (i = 0; i < MLDSA_L; i++)+ __loop__(+ assigns(i, memory_slice(z, sizeof(mld_poly)),+ memory_slice(tmp, sizeof(mld_poly)),+ memory_slice(sig, MLDSA_CRYPTO_BYTES))+ invariant(i <= MLDSA_L)+ decreases(MLDSA_L - i)+ )+ {+ mld_sk_s1hat_get_poly(z, s1hat, i);+ mld_poly_pointwise_montgomery(z, cp);+ mld_poly_invntt_tomont(z);+ mld_yvec_get_poly(tmp, y, i);+ mld_poly_add(z, tmp);+ mld_poly_reduce(z);++ z_invalid = mld_poly_chknorm(z, MLDSA_GAMMA1 - MLDSA_BETA);+ /* Constant time: It is fine (and prohibitively expensive to avoid)+ * to leak the result of the norm check and which polynomial in z caused a+ * rejection. It would even be okay to leak which coefficient led to+ * rejection as the candidate signature will be discarded anyway.+ * See Section 5.5 of @[Round3_Spec]. */+ MLD_CT_TESTING_DECLASSIFY(&z_invalid, sizeof(uint32_t));+ if (z_invalid)+ {+ return MLD_ERR_FAIL; /* reject */+ }+ /* If z is valid, then its coefficients are bounded by+ * MLDSA_GAMMA1 - MLDSA_BETA. This will be needed below+ * to prove the pre-condition of pack_sig_z() */+ mld_assert_abs_bound(z, MLDSA_N, (MLDSA_GAMMA1 - MLDSA_BETA));++ /* After the norm check, the distribution of each coefficient of z is+ * independent of the secret key and it can, hence, be considered+ * public. It is, hence, okay to immediately pack it into the user-provided+ * signature buffer. */+ mld_pack_sig_z(sig, z, i);+ }+ return 0;+}++/* Effective bound on signing attempts: the configured bound+ * MLD_CONFIG_MAX_SIGNING_ATTEMPTS (see mldsa_native_config.h) if set, otherwise+ * the hard type-safety bound MLD_MAX_KAPPA / MLDSA_L (see MLD_MAX_KAPPA in+ * params.h). */+#if defined(MLD_CONFIG_MAX_SIGNING_ATTEMPTS)++#if !defined(MLD_ALLOW_NONCOMPLIANT_SIGNING_BOUND) && \+ MLD_CONFIG_MAX_SIGNING_ATTEMPTS < 821+#error Bad configuration: MLD_CONFIG_MAX_SIGNING_ATTEMPTS must be >= 821 for FIPS 204 compliance @[FIPS204, Appendix C] @[FIPS204_UPDATES]+#endif++#if MLD_CONFIG_MAX_SIGNING_ATTEMPTS < 1+#error Bad configuration: MLD_CONFIG_MAX_SIGNING_ATTEMPTS must be >= 1+#endif++#if MLD_CONFIG_MAX_SIGNING_ATTEMPTS > MLD_MAX_KAPPA / MLDSA_L+#error Bad configuration: MLD_CONFIG_MAX_SIGNING_ATTEMPTS exceeds the maximum allowed value.+#endif++#define MLD_MAX_SIGNING_ATTEMPTS MLD_CONFIG_MAX_SIGNING_ATTEMPTS+#else /* MLD_CONFIG_MAX_SIGNING_ATTEMPTS */+#define MLD_MAX_SIGNING_ATTEMPTS (MLD_MAX_KAPPA / MLDSA_L)+#endif /* !MLD_CONFIG_MAX_SIGNING_ATTEMPTS */++MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE uint16_t mld_get_max_signing_attempts(void)+__contract__(+ ensures(return_value >= 1)+ ensures(return_value <= MLD_MAX_KAPPA / MLDSA_L)+)+{+ /* cassert(0) ensures CBMC uses the contract rather than inlining the body,+ * keeping proofs agnostic of the configured value. */+ cassert(0);+ return MLD_MAX_SIGNING_ATTEMPTS;+}++/**+ * Attempt to generate a single signature: one iteration of the+ * ML-DSA.Sign_internal rejection-sampling loop.+ *+ * @spec{Implements one iteration of the rejection-sampling loop body of+ * @[FIPS204, Algorithm 7, ML-DSA.Sign_internal] (lines 11-30) plus, on success,+ * the sigEncode step (line 33). The per-signature setup (Algorithm 7 lines 1-7:+ * skDecode, NTT of s1/s2/t0, ExpandA, and computation of mu and rhoprime) and+ * the loop itself (lines 8-10, 31-32) live in the caller+ * mld_sign_signature_internal; kappa is this iteration's counter, used to+ * sample y.}+ *+ * @reference{This code differs from the reference implementation in that it+ * factors out the core signature generation step into a distinct function+ * here in order to improve efficiency of CBMC proof.}+ *+ * @param[out] sig Pointer to output signature.+ * @param[in] mu Pointer to message or hash of exactly MLDSA_CRHBYTES+ * bytes.+ * @param[in] rhoprime Pointer to randomness seed.+ * @param kappa Counter for this iteration (= attempt*MLDSA_L).+ * @param[in] mat Expanded matrix.+ * @param[in] s1hat Secret vector s1 in NTT domain.+ * @param[in] s2hat Secret vector s2 in NTT domain.+ * @param[in] t0hat Vector t0 in NTT domain.+ * @param context Application context. Only present when+ * MLD_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @return - 0: Signature generation succeeded.+ * - MLD_ERR_FAIL: Signature rejected (norm check failed).+ * - MLD_ERR_OUT_OF_MEMORY: If MLD_CONFIG_CUSTOM_ALLOC_FREE is used and+ * an allocation via MLD_CUSTOM_ALLOC returned NULL.+ */+MLD_MUST_CHECK_RETURN_VALUE+/* NOLINTNEXTLINE(readability-function-cognitive-complexity) */+static int mld_attempt_signature_generation(+ uint8_t sig[MLDSA_CRYPTO_BYTES], const uint8_t *mu,+ const uint8_t rhoprime[MLDSA_CRHBYTES], uint16_t kappa, mld_polymat *mat,+ const mld_sk_s1hat *s1hat, const mld_sk_s2hat *s2hat,+ const mld_sk_t0hat *t0hat, MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+__contract__(+ requires(memory_no_alias(sig, MLDSA_CRYPTO_BYTES))+ requires(memory_no_alias(mu, MLDSA_CRHBYTES))+ requires(memory_no_alias(rhoprime, MLDSA_CRHBYTES))+ requires(memory_no_alias(mat, sizeof(mld_polymat)))+ requires(memory_no_alias(s1hat, sizeof(mld_sk_s1hat)))+ requires(memory_no_alias(s2hat, sizeof(mld_sk_s2hat)))+ requires(memory_no_alias(t0hat, sizeof(mld_sk_t0hat)))+ requires(kappa <= MLD_MAX_KAPPA)+ MLD_IF_NOT_REDUCE_RAM(+ requires(forall(k1, 0, MLDSA_K, forall(l1, 0, MLDSA_L,+ array_bound(mat->vec[k1].vec[l1].coeffs, 0, MLDSA_N, 0, MLDSA_Q))))+ requires(forall(k2, 0, MLDSA_K, array_abs_bound(t0hat->vec.vec[k2].coeffs, 0, MLDSA_N, MLD_NTT_BOUND)))+ requires(forall(k3, 0, MLDSA_L, array_abs_bound(s1hat->vec.vec[k3].coeffs, 0, MLDSA_N, MLD_NTT_BOUND)))+ requires(forall(k4, 0, MLDSA_K, array_abs_bound(s2hat->vec.vec[k4].coeffs, 0, MLDSA_N, MLD_NTT_BOUND)))+ )+ MLD_IF_REDUCE_RAM(+ requires(memory_no_alias(s1hat->packed, MLDSA_L * MLDSA_POLYETA_PACKEDBYTES))+ requires(memory_no_alias(s2hat->packed, MLDSA_K * MLDSA_POLYETA_PACKEDBYTES))+ requires(memory_no_alias(t0hat->packed, MLDSA_K * MLDSA_POLYT0_PACKEDBYTES))+ )+ assigns(memory_slice(sig, MLDSA_CRYPTO_BYTES))+ MLD_IF_REDUCE_RAM(+ assigns(memory_slice(mat, sizeof(mld_polymat)))+ )+ ensures(return_value == 0 || return_value == MLD_ERR_FAIL ||+ return_value == MLD_ERR_OUT_OF_MEMORY)+)+{+ unsigned int k;+ uint32_t w0_invalid, h_invalid;+ int ret;++ typedef union+ {+ mld_polyveck w1;+ mld_polyvecl tmp;+ } w1tmp_u;+ mld_polyveck *w1;+ mld_polyvecl *tmp;++ MLD_ALLOC(challenge_bytes, uint8_t, MLDSA_CTILDEBYTES, context);+ MLD_ALLOC(y, mld_yvec, 1, context);+ MLD_ALLOC(z, mld_poly, 1, context);+ MLD_ALLOC(w1tmp, w1tmp_u, 1, context);+ MLD_ALLOC(w0, mld_polyveck, 1, context);+ MLD_ALLOC(cp, mld_poly, 1, context);+ MLD_ALLOC(t, mld_poly, 1, context);++ if (challenge_bytes == NULL || y == NULL || z == NULL || w1tmp == NULL ||+ w0 == NULL || cp == NULL || t == NULL)+ {+ ret = MLD_ERR_OUT_OF_MEMORY;+ goto cleanup;+ }+ w1 = &w1tmp->w1;+ tmp = &w1tmp->tmp;++ /* @[FIPS204, Algorithm 7, line 11] y <- ExpandMask(rhoprime, kappa). */+ mld_yvec_init(y, rhoprime, kappa);++ /* @[FIPS204, Algorithm 7, line 12] w <- invNTT(A_hat o NTT(y)). This call+ * performs the whole line: it NTTs y, accumulates the pointwise product with+ * A_hat, and applies the inverse NTT. In REDUCE_RAM mode the y sampling is+ * fused into the same pass. */+ mld_polyvec_matrix_pointwise_montgomery_yvec(w0, mat, y, tmp);++ /* @[FIPS204, Algorithm 7, line 13] w1 <- HighBits(w), here together with the+ * low part: Decompose yields w = 2*GAMMA2*w1 + w0, keeping both w1 and w0+ * (w0 is reused below in the line-21/26 alternative, see further down). */+ mld_polyveck_caddq(w0);+ mld_polyveck_decompose(w1, w0);++ /* @[FIPS204, Algorithm 7, line 15] ctilde <- H(mu || w1Encode(w1), lambda/4).+ * w1Encode(w1) is packed into the w1 region of sig (mld_polyveck_pack_w1),+ * then absorbed by H together with mu. */+ mld_polyveck_pack_w1(sig, w1);++ mld_H(challenge_bytes, MLDSA_CTILDEBYTES, mu, MLDSA_CRHBYTES, sig,+ MLDSA_K * MLDSA_POLYW1_PACKEDBYTES, NULL, 0);+ /* Constant time: Leaking challenge_bytes does not reveal any information+ * about the secret key as H() is modelled as random oracle.+ * This also applies to challenges for rejected signatures.+ * See Section 5.5 of @[Round3_Spec]. */+ MLD_CT_TESTING_DECLASSIFY(challenge_bytes, MLDSA_CTILDEBYTES);+ /* @[FIPS204, Algorithm 7, line 16] c <- SampleInBall(ctilde) and+ * @[FIPS204, Algorithm 7, line 17] c_hat <- NTT(c). */+ mld_poly_challenge(cp, challenge_bytes);+ mld_poly_ntt(cp);++ /* @[FIPS204, Algorithm 7, lines 18+20] cs1 <- invNTT(c_hat o s1_hat) and+ * z <- y + cs1, followed by the line-23 norm check ||z||_inf >= GAMMA1 -+ * BETA. mld_compute_pack_z fuses all three per polynomial and, on success,+ * packs z into sig; it returns MLD_ERR_FAIL if the norm check rejects z. */+ ret = mld_compute_pack_z(sig, cp, s1hat, y, t, z);+ if (ret != 0)+ {+ goto cleanup;+ }++ /* The remaining steps realize @[FIPS204, Algorithm 7, lines 21-28] (the+ * low-bits norm check and the hint h) via the faster alternative formulation+ * of @[Round3_Spec, Section 5.1]. @[FIPS204] explicitly permits this: the+ * note accompanying Algorithm 7 states that the validity checks on z and the+ * computation of h may instead be implemented "as described in Section 5.1 of+ * [6]", and that reference is @[Round3_Spec, Section 5.1].+ *+ * The loop below builds w0 - cs2 + ct0 in place in w0; w1 is unmodified, and+ * is HighBits(w) from line 13. Those are the inputs to the streamlined+ * computation of MakeHint explained below.+ *+ * Low-bits norm check:+ * @[FIPS204, Algorithm 7, line 21] computes r0 = LowBits(w - cs2) and line+ * 23 rejects when ||r0||_inf >= GAMMA2 - BETA. By @[Round3_Spec, Section+ * 5.1] (Lemma 3), this line-23 check on r0 = LowBits(w - cs2) is implied by+ * ||w0 - cs2||_inf < GAMMA2 - BETA, where w0 is the low part of w. In our+ * context, w0 already holds the low part of w from the line-13 Decompose;+ * after subtracting cs2 from it in place, the mld_poly_chknorm(w0, GAMMA2 -+ * BETA) call below is exactly that check.+ *+ * Hint:+ * @[FIPS204, Algorithm 7, line 26] sets h = MakeHint(-ct0, w - cs2 + ct0),+ * and line 28 rejects when ||ct0||_inf >= GAMMA2 or h has more than OMEGA+ * nonzero coefficients. @[Round3_Spec, Section 5.1] provides the following+ * alternative description for MakeHint(-ct0, w - cs2 + ct0): a hint bit is+ * zero exactly when the coefficient of w0 - cs2 + ct0 lies in+ * (-GAMMA2, GAMMA2], or equals -GAMMA2 while the matching w1 coefficient is+ * zero (the Decompose border case), and is set otherwise. This equivalence+ * is precisely what mld_pack_sig_h -> mld_make_hint compute from w0+ * (= w0 - cs2 + ct0) and w1. The line-28 ||ct0||_inf >= GAMMA2 check is the+ * mld_poly_chknorm(z, GAMMA2) call on ct0 below; the weight bound is+ * enforced by mld_pack_sig_h.+ *+ * Building w0 per-component and checking norms incrementally also avoids+ * allocating a full polyveck for h. */+ for (k = 0; k < MLDSA_K; k++)+ __loop__(+ assigns(k,+ object_whole(z),+ object_whole(w0))+ invariant(k <= MLDSA_K)+ invariant(forall(k0, k, MLDSA_K,+ array_abs_bound(w0->vec[k0].coeffs, 0, MLDSA_N, MLDSA_GAMMA2 + 1)))+ decreases(MLDSA_K - k)+ )+ {+ /* @[FIPS204, Algorithm 7, line 19] cs2[k] <- invNTT(c_hat o s2_hat)[k],+ * then subtract from w0[k] to form (w0 - cs2)[k]. */+ mld_sk_s2hat_get_poly(z, s2hat, k);+ mld_poly_pointwise_montgomery(z, cp);+ mld_poly_invntt_tomont(z);++ mld_poly_sub(&w0->vec[k], z);+ mld_poly_reduce(&w0->vec[k]);++ /* Low-bits norm check (see block comment above): the line-23 check on+ * r0 = LowBits(w - cs2) holds via ||w0 - cs2||_inf < GAMMA2 - BETA. */+ w0_invalid = mld_poly_chknorm(&w0->vec[k], MLDSA_GAMMA2 - MLDSA_BETA);+ /* Constant time: w0_invalid may be leaked - see comment for z_invalid. */+ MLD_CT_TESTING_DECLASSIFY(&w0_invalid, sizeof(uint32_t));+ if (w0_invalid)+ {+ ret = MLD_ERR_FAIL; /* reject */+ goto cleanup;+ }++ /* @[FIPS204, Algorithm 7, line 25] ct0[k] <- invNTT(c_hat o t0_hat)[k]. */+ mld_sk_t0hat_get_poly(z, t0hat, k);+ mld_poly_pointwise_montgomery(z, cp);+ mld_poly_invntt_tomont(z);+ mld_poly_reduce(z);++ /* @[FIPS204, Algorithm 7, line 28] reject when ||ct0||_inf >= GAMMA2 (the+ * second part, the OMEGA weight bound, is enforced by mld_pack_sig_h). */+ h_invalid = mld_poly_chknorm(z, MLDSA_GAMMA2);+ /* Constant time: h_invalid may be leaked - see comment for z_invalid. */+ MLD_CT_TESTING_DECLASSIFY(&h_invalid, sizeof(uint32_t));+ if (h_invalid)+ {+ ret = MLD_ERR_FAIL; /* reject */+ goto cleanup;+ }++ /* Add ct0[k] to (w0 - cs2)[k], leaving (w0 - cs2 + ct0)[k] in w0[k] -- the+ * MakeHint input prepared for mld_pack_sig_h (see block comment above). */+ mld_poly_add(&w0->vec[k], z);+ }++ /* Constant time: At this point all norm checks have passed and we, hence,+ * know that the signature does not leak any secret information.+ * Consequently, any value that can be computed from the signature and public+ * key is considered public.+ * w0 and w1 are public as they can be computed from Az - ct = \alpha w1 + w0.+ * h=c*t0 is public as both c and t0 are considered public.+ * While t0 is not part of the public key, it can be reconstructed from+ * a small number of signatures and need not be regarded as secret+ * (see @[FIPS204, Section 6.1]).+ */+ MLD_CT_TESTING_DECLASSIFY(w0, sizeof(*w0));+ MLD_CT_TESTING_DECLASSIFY(w1, sizeof(*w1));++ /* @[FIPS204, Algorithm 7, line 33] sigEncode(ctilde, z mod+/- q, h) is split+ * across three calls: z was already packed by mld_compute_pack_z, this call+ * packs ctilde, and mld_pack_sig_h below packs the hint h. */+ mld_pack_sig_c(sig, challenge_bytes);++ /* @[FIPS204, Algorithm 7, line 26] h <- MakeHint(-ct0, w - cs2 + ct0),+ * computed from (w0 = w0 - cs2 + ct0, w1) as described in the block comment+ * above, and packed as the h component of the line-33 sigEncode. Returns+ * MLD_ERR_FAIL if h would exceed OMEGA nonzero coefficients (the remaining+ * part of the line-28 check), in which case we reject. */+ ret = mld_pack_sig_h(sig, w0, w1);+ if (ret != 0)+ {+ goto cleanup;+ }++ /* Constant time: At this point it is clear that the signature is valid - it+ * can, hence, be considered public. */+ MLD_CT_TESTING_DECLASSIFY(sig, MLDSA_CRYPTO_BYTES);+ ret = 0; /* success */++cleanup:+ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ MLD_FREE(t, mld_poly, 1, context);+ MLD_FREE(cp, mld_poly, 1, context);+ MLD_FREE(w0, mld_polyveck, 1, context);+ MLD_FREE(w1tmp, w1tmp_u, 1, context);+ MLD_FREE(z, mld_poly, 1, context);+ MLD_FREE(y, mld_yvec, 1, context);+ MLD_FREE(challenge_bytes, uint8_t, MLDSA_CTILDEBYTES, context);++ return ret;+}+MLD_MUST_CHECK_RETURN_VALUE+MLD_EXTERNAL_API+int mld_sign_signature_internal(uint8_t sig[MLDSA_CRYPTO_BYTES],+ const uint8_t *m, size_t mlen,+ const uint8_t *pre, size_t prelen,+ const uint8_t rnd[MLDSA_RNDBYTES],+ const uint8_t sk[MLDSA_CRYPTO_SECRETKEYBYTES],+ int externalmu,+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+{+ int ret;+ uint8_t *rho, *tr, *key, *mu, *rhoprime;+ uint16_t attempt;+ const uint16_t max_signing_attempts = mld_get_max_signing_attempts();+ MLD_ALLOC(seedbuf, uint8_t,+ 2 * MLDSA_SEEDBYTES + MLDSA_TRBYTES + 2 * MLDSA_CRHBYTES, context);+ MLD_ALLOC(mat, mld_polymat, 1, context);+ MLD_ALLOC(s1hat, mld_sk_s1hat, 1, context);+ MLD_ALLOC(t0hat, mld_sk_t0hat, 1, context);+ MLD_ALLOC(s2hat, mld_sk_s2hat, 1, context);++ if (seedbuf == NULL || mat == NULL || s1hat == NULL || t0hat == NULL ||+ s2hat == NULL)+ {+ ret = MLD_ERR_OUT_OF_MEMORY;+ goto cleanup;+ }++ /* If a resume hook is configured via MLD_CONFIG_SIGN_HOOK_RESUME, it provides+ * the attempt to resume from after an earlier pause. Otherwise, we start at+ * 0. Clamp to max_signing_attempts. */+ attempt = mld_sign_resume(context);+ if (attempt > max_signing_attempts)+ {+ attempt = max_signing_attempts;+ }++ rho = seedbuf;+ tr = rho + MLDSA_SEEDBYTES;+ key = tr + MLDSA_TRBYTES;+ mu = key + MLDSA_SEEDBYTES;+ rhoprime = mu + MLDSA_CRHBYTES;+ /* @[FIPS204, Algorithm 7, line 1] (rho, K, tr, s1, s2, t0) <- skDecode(sk)+ * and @[FIPS204, Algorithm 7, lines 2-4] s1_hat/s2_hat/t0_hat <- NTT(...):+ * mld_unpack_sk returns s1hat, s2hat, t0hat already in NTT domain. The spec's+ * private random seed K is held in the local variable key. */+ mld_unpack_sk(rho, tr, key, t0hat, s1hat, s2hat, sk);++ if (!externalmu)+ {+ /* @[FIPS204, Algorithm 7, line 6] mu <- H(BytesToBits(tr) || M', 64). */+ mld_H(mu, MLDSA_CRHBYTES, tr, MLDSA_TRBYTES, pre, prelen, m, mlen);+ }+ else+ {+ /* mu has been provided directly (external-mu variant; line 6 done by the+ * caller in a separate cryptographic module). */+ mld_memcpy(mu, m, MLDSA_CRHBYTES);+ }++ /* @[FIPS204, Algorithm 7, line 7] rhoprime <- H(K || rnd || mu, 64). */+ mld_H(rhoprime, MLDSA_CRHBYTES, key, MLDSA_SEEDBYTES, rnd, MLDSA_RNDBYTES, mu,+ MLDSA_CRHBYTES);++ /* Constant time: rho is part of the public key and, hence, public. */+ MLD_CT_TESTING_DECLASSIFY(rho, MLDSA_SEEDBYTES);+ /* @[FIPS204, Algorithm 7, line 5] A_hat <- ExpandA(rho). */+ mld_polyvec_matrix_expand(mat, rho);++ /* @[FIPS204, Algorithm 7, lines 8-10 and 31-32] the rejection-sampling loop,+ * tracked by attempt (kappa = attempt*MLDSA_L). Each iteration's body (lines+ * 11-30) plus, on success, the line-33 sigEncode are performed by+ * mld_attempt_signature_generation. */++ /* Reference: the reference loops unboundedly; we instead iterate over the+ * bounded range [0, max_signing_attempts) for predictable termination.+ * A success or fatal error exits via goto cleanup; running to completion+ * means every attempt was rejected; with a FIPS compliant choice of+ * MLD_CONFIG_MAX_SIGNING_ATTEMPTS, this should never happen. */+ for (; attempt < max_signing_attempts; attempt++)+ __loop__(+ MLD_IF_NOT_REDUCE_RAM(+ assigns(attempt, ret, memory_slice(sig, MLDSA_CRYPTO_BYTES))+ )+ MLD_IF_REDUCE_RAM(+ assigns(attempt, ret, memory_slice(sig, MLDSA_CRYPTO_BYTES),+ memory_slice(mat, sizeof(mld_polymat)))+ )+ invariant(attempt <= max_signing_attempts)++ /* t0, s1, s2, and mat are initialized above and are NOT changed by this */+ /* loop. We can therefore re-assert their bounds here as part of the */+ /* loop invariant. This makes proof noticeably faster with CBMC */+ MLD_IF_NOT_REDUCE_RAM(+ invariant(forall(k1, 0, MLDSA_K, forall(l1, 0, MLDSA_L,+ array_bound(mat->vec[k1].vec[l1].coeffs, 0, MLDSA_N, 0, MLDSA_Q))))+ invariant(forall(k2, 0, MLDSA_K, array_abs_bound(t0hat->vec.vec[k2].coeffs, 0, MLDSA_N, MLD_NTT_BOUND)))+ invariant(forall(k3, 0, MLDSA_L, array_abs_bound(s1hat->vec.vec[k3].coeffs, 0, MLDSA_N, MLD_NTT_BOUND)))+ invariant(forall(k4, 0, MLDSA_K, array_abs_bound(s2hat->vec.vec[k4].coeffs, 0, MLDSA_N, MLD_NTT_BOUND)))+ )+ decreases(max_signing_attempts - attempt)+ )+ {+ /* Safety: attempt < max_signing_attempts <= MLD_MAX_KAPPA / MLDSA_L, so+ * kappa <= MLD_MAX_KAPPA and the cast is safe. */+ const uint16_t kappa = (uint16_t)(attempt * MLDSA_L);++ /* Query configurable signing hook whether signing should be paused.+ * This is skipped by default and only used if the user sets the+ * configuration option MLD_CONFIG_SIGN_HOOK_ATTEMPT. */+ if (mld_sign_attempt(attempt, context) != 0)+ {+ ret = MLD_ERR_SIGNING_PAUSED;+ goto cleanup;+ }++ ret = mld_attempt_signature_generation(sig, mu, rhoprime, kappa, mat, s1hat,+ s2hat, t0hat, context);++ /* Decide whether to keep trying based on the return value:+ * - ret == 0: a valid signature was produced; we are done.+ * - ret == MLD_ERR_FAIL: this candidate was rejected by one of the norm+ * or hint checks. We continue the loop and try again with the next+ * nonce.+ * - any other value (e.g. MLD_ERR_OUT_OF_MEMORY): an unrecoverable error+ * occurred, so we propagate it to the caller. */+ if (ret == 0)+ {+ /* Signing succeeded: record the attempt that succeeded. No-op in the+ * default build. */+ mld_sign_finish(attempt, context);+ goto cleanup;+ }+ if (ret != MLD_ERR_FAIL)+ {+ goto cleanup;+ }+ }++ /* Loop ran to completion: all attempts rejected, budget exhausted.+ * This should never happen with a FIPS compliant choice of+ * MLD_CONFIG_MAX_SIGNING_ATTEMPTS. */+ ret = MLD_ERR_SIGN_ATTEMPTS_EXHAUSTED;++cleanup:++ if (ret != 0)+ {+ /* To be on the safe-side, we zeroize the signature buffer. */+ mld_zeroize(sig, MLDSA_CRYPTO_BYTES);+ }++ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ MLD_FREE(s2hat, mld_sk_s2hat, 1, context);+ MLD_FREE(t0hat, mld_sk_t0hat, 1, context);+ MLD_FREE(s1hat, mld_sk_s1hat, 1, context);+ MLD_FREE(mat, mld_polymat, 1, context);+ MLD_FREE(seedbuf, uint8_t,+ 2 * MLDSA_SEEDBYTES + MLDSA_TRBYTES + 2 * MLDSA_CRHBYTES, context);+ return ret;+}++#if !defined(MLD_CONFIG_CORE_API_ONLY)+#if !defined(MLD_CONFIG_NO_RANDOMIZED_API)+MLD_MUST_CHECK_RETURN_VALUE+MLD_EXTERNAL_API+int mld_sign_signature(uint8_t sig[MLDSA_CRYPTO_BYTES], const uint8_t *m,+ size_t mlen, const uint8_t *ctx, size_t ctxlen,+ const uint8_t sk[MLDSA_CRYPTO_SECRETKEYBYTES],+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+{+ size_t pre_len;+ int ret;+ MLD_ALLOC(pre, uint8_t, MLD_DOMAIN_SEPARATION_MAX_BYTES, context);+ MLD_ALLOC(rnd, uint8_t, MLDSA_RNDBYTES, context);++ if (pre == NULL || rnd == NULL)+ {+ ret = MLD_ERR_OUT_OF_MEMORY;+ goto cleanup;+ }++ /* Prepare domain separation prefix for pure ML-DSA */+ pre_len = mld_prepare_domain_separation_prefix(pre, NULL, 0, ctx, ctxlen,+ MLD_PREHASH_NONE);+ if (pre_len == 0)+ {+ ret = MLD_ERR_INVALID_ARG;+ goto cleanup;+ }++ /* Randomized variant of ML-DSA. If you need the deterministic variant,+ * call mld_sign_signature_internal directly with all-zero rnd. */+ if (mld_randombytes(rnd, MLDSA_RNDBYTES) != 0)+ {+ ret = MLD_ERR_RNG_FAIL;+ goto cleanup;+ }+ MLD_CT_TESTING_SECRET(rnd, MLDSA_RNDBYTES);++ ret = mld_sign_signature_internal(sig, m, mlen, pre, pre_len, rnd, sk, 0,+ context);++cleanup:+ if (ret != 0)+ {+ /* To be on the safe-side, make sure sig has a well-defined value, even in+ * the case of error.+ *+ * If we come from mld_sign_signature_internal, this is redundant, but the+ * error case should not be the norm, and the added cost of the zeroization+ * insignificant. */+ mld_zeroize(sig, MLDSA_CRYPTO_BYTES);+ }++ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ MLD_FREE(rnd, uint8_t, MLDSA_RNDBYTES, context);+ MLD_FREE(pre, uint8_t, MLD_DOMAIN_SEPARATION_MAX_BYTES, context);++ return ret;+}+#endif /* !MLD_CONFIG_NO_RANDOMIZED_API */++#if !defined(MLD_CONFIG_NO_RANDOMIZED_API)+MLD_MUST_CHECK_RETURN_VALUE+MLD_EXTERNAL_API+int mld_sign_signature_extmu(uint8_t sig[MLDSA_CRYPTO_BYTES],+ const uint8_t mu[MLDSA_CRHBYTES],+ const uint8_t sk[MLDSA_CRYPTO_SECRETKEYBYTES],+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+{+ int ret;+ MLD_ALLOC(rnd, uint8_t, MLDSA_RNDBYTES, context);++ if (rnd == NULL)+ {+ ret = MLD_ERR_OUT_OF_MEMORY;+ goto cleanup;+ }++ /* Randomized variant of ML-DSA. If you need the deterministic variant,+ * call mld_sign_signature_internal directly with all-zero rnd. */+ if (mld_randombytes(rnd, MLDSA_RNDBYTES) != 0)+ {+ ret = MLD_ERR_RNG_FAIL;+ goto cleanup;+ }+ MLD_CT_TESTING_SECRET(rnd, MLDSA_RNDBYTES);++ ret = mld_sign_signature_internal(sig, mu, MLDSA_CRHBYTES, NULL, 0, rnd, sk,+ 1, context);++cleanup:+ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ MLD_FREE(rnd, uint8_t, MLDSA_RNDBYTES, context);++ return ret;+}+#endif /* !MLD_CONFIG_NO_RANDOMIZED_API */+#endif /* !MLD_CONFIG_CORE_API_ONLY */+#endif /* !MLD_CONFIG_NO_SIGN_API */++#if !defined(MLD_CONFIG_NO_VERIFY_API)+MLD_MUST_CHECK_RETURN_VALUE+MLD_EXTERNAL_API+int mld_sign_verify_internal(const uint8_t sig[MLDSA_CRYPTO_BYTES],+ const uint8_t *m, size_t mlen, const uint8_t *pre,+ size_t prelen,+ const uint8_t pk[MLDSA_CRYPTO_PUBLICKEYBYTES],+ int externalmu,+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+{+ int ret, cmp;+ unsigned int i;++ MLD_ALLOC(buf, uint8_t, (MLDSA_K * MLDSA_POLYW1_PACKEDBYTES), context);+ MLD_ALLOC(mu, uint8_t, MLDSA_CRHBYTES, context);+ MLD_ALLOC(c, uint8_t, MLDSA_CTILDEBYTES, context);+ MLD_ALLOC(c2, uint8_t, MLDSA_CTILDEBYTES, context);+ MLD_ALLOC(z, mld_polyvecl, 1, context);+ MLD_ALLOC(cp, mld_poly, 1, context);+ MLD_ALLOC(mat, mld_polymat, 1, context);+ MLD_ALLOC(w1, mld_poly, 1, context);+ MLD_ALLOC(tmp, mld_poly, 1, context);++ if (buf == NULL || mu == NULL || c == NULL || c2 == NULL || z == NULL ||+ cp == NULL || mat == NULL || w1 == NULL || tmp == NULL)+ {+ ret = MLD_ERR_OUT_OF_MEMORY;+ goto cleanup;+ }++ mld_memcpy(c, sig, MLDSA_CTILDEBYTES);+ mld_polyvecl_unpack_z(z, sig + MLDSA_SIG_Z_OFFSET);++ /* mld_polyvecl_chknorm signals failure through a single non-zero error code+ * that's not yet aligned with MLD_ERR_XXX. A norm-check failure here means+ * the signature is invalid, so map it to MLD_ERR_INVALID_SIGNATURE. */+ if (mld_polyvecl_chknorm(z, MLDSA_GAMMA1 - MLDSA_BETA))+ {+ ret = MLD_ERR_INVALID_SIGNATURE;+ goto cleanup;+ }++ if (!externalmu)+ {+ /* Compute CRH(H(rho, t1), pre, msg) */+ MLD_ALIGN uint8_t hpk[MLDSA_CRHBYTES];+ mld_H(hpk, MLDSA_TRBYTES, pk, MLDSA_CRYPTO_PUBLICKEYBYTES, NULL, 0, NULL,+ 0);+ mld_H(mu, MLDSA_CRHBYTES, hpk, MLDSA_TRBYTES, pre, prelen, m, mlen);++ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ mld_zeroize(hpk, sizeof(hpk));+ }+ else+ {+ /* mu has been provided directly */+ mld_memcpy(mu, m, MLDSA_CRHBYTES);+ }++ /* Matrix-vector multiplication and per-row reconstruction of w1. */+ mld_polyvecl_ntt(z);+ mld_polyvec_matrix_expand(mat, pk);+ mld_poly_challenge(cp, c);+ mld_poly_ntt(cp);++ for (i = 0; i < MLDSA_K; ++i)+ __loop__(+ assigns(MLD_IF_REDUCE_RAM(memory_slice(mat, sizeof(mld_polymat)),)+ i, ret,+ memory_slice(w1, sizeof(mld_poly)),+ memory_slice(tmp, sizeof(mld_poly)),+ memory_slice(buf, MLDSA_K * MLDSA_POLYW1_PACKEDBYTES)+ )+ invariant(i <= MLDSA_K)+ decreases(MLDSA_K - i)+ )+ {+ /* w1 = (A * z)_i in NTT domain */+ mld_polyvec_matrix_pointwise_montgomery_row(w1, mat, z, i);++ /* tmp = c * t1_i * 2^d in NTT domain */+ mld_unpack_pk_t1(tmp, pk, i);+ mld_poly_shiftl(tmp);+ mld_poly_ntt(tmp);+ mld_poly_pointwise_montgomery(tmp, cp);++ /* w1 = invNTT(w1 - c * t1_i * 2^d) */+ mld_poly_sub(w1, tmp);+ mld_poly_reduce(w1);+ mld_poly_invntt_tomont(w1);+ mld_poly_caddq(w1);++ /* tmp = h_i (decoded and validated from signature). A non-zero return+ * means the hint encoding is malformed, i.e. the signature is invalid. */+ if (mld_sig_unpack_hints(tmp, sig, i) != 0)+ {+ ret = MLD_ERR_INVALID_SIGNATURE;+ goto cleanup;+ }++ /* w1 = use_hint(w1, tmp), then pack into buf[i] */+ mld_poly_use_hint(w1, tmp);+ mld_polyw1_pack(buf + i * MLDSA_POLYW1_PACKEDBYTES, w1);+ }++ /* Call random oracle and verify challenge */+ mld_H(c2, MLDSA_CTILDEBYTES, mu, MLDSA_CRHBYTES, buf,+ MLDSA_K * MLDSA_POLYW1_PACKEDBYTES, NULL, 0);++ cmp = mld_ct_memcmp(c, c2, MLDSA_CTILDEBYTES);++ /* Declassify the result of the verification. */+ MLD_CT_TESTING_DECLASSIFY(&cmp, sizeof(cmp));++ ret = cmp == 0 ? 0 : MLD_ERR_INVALID_SIGNATURE;++cleanup:+ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ MLD_FREE(tmp, mld_poly, 1, context);+ MLD_FREE(w1, mld_poly, 1, context);+ MLD_FREE(mat, mld_polymat, 1, context);+ MLD_FREE(cp, mld_poly, 1, context);+ MLD_FREE(z, mld_polyvecl, 1, context);+ MLD_FREE(c2, uint8_t, MLDSA_CTILDEBYTES, context);+ MLD_FREE(c, uint8_t, MLDSA_CTILDEBYTES, context);+ MLD_FREE(mu, uint8_t, MLDSA_CRHBYTES, context);+ MLD_FREE(buf, uint8_t, (MLDSA_K * MLDSA_POLYW1_PACKEDBYTES), context);+ return ret;+}++#if !defined(MLD_CONFIG_CORE_API_ONLY)+MLD_MUST_CHECK_RETURN_VALUE+MLD_EXTERNAL_API+int mld_sign_verify(const uint8_t sig[MLDSA_CRYPTO_BYTES], const uint8_t *m,+ size_t mlen, const uint8_t *ctx, size_t ctxlen,+ const uint8_t pk[MLDSA_CRYPTO_PUBLICKEYBYTES],+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+{+ MLD_ALIGN uint8_t pre[MLD_DOMAIN_SEPARATION_MAX_BYTES];+ size_t pre_len;+ int ret;++ pre_len = mld_prepare_domain_separation_prefix(pre, NULL, 0, ctx, ctxlen,+ MLD_PREHASH_NONE);+ if (pre_len == 0)+ {+ ret = MLD_ERR_INVALID_ARG;+ goto cleanup;+ }++ ret = mld_sign_verify_internal(sig, m, mlen, pre, pre_len, pk, 0, context);++cleanup:+ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ mld_zeroize(pre, sizeof(pre));++ return ret;+}++MLD_MUST_CHECK_RETURN_VALUE+MLD_EXTERNAL_API+int mld_sign_verify_extmu(const uint8_t sig[MLDSA_CRYPTO_BYTES],+ const uint8_t mu[MLDSA_CRHBYTES],+ const uint8_t pk[MLDSA_CRYPTO_PUBLICKEYBYTES],+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+{+ return mld_sign_verify_internal(sig, mu, MLDSA_CRHBYTES, NULL, 0, pk, 1,+ context);+}+#endif /* !MLD_CONFIG_CORE_API_ONLY */+#endif /* !MLD_CONFIG_NO_VERIFY_API */++#if !defined(MLD_CONFIG_CORE_API_ONLY)+#if !defined(MLD_CONFIG_NO_SIGN_API)+MLD_MUST_CHECK_RETURN_VALUE+MLD_EXTERNAL_API+int mld_sign_signature_pre_hash_internal(+ uint8_t sig[MLDSA_CRYPTO_BYTES], const uint8_t *ph, size_t phlen,+ const uint8_t *ctx, size_t ctxlen, const uint8_t rnd[MLDSA_RNDBYTES],+ const uint8_t sk[MLDSA_CRYPTO_SECRETKEYBYTES], int hashalg,+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+{+ MLD_ALIGN uint8_t pre[MLD_DOMAIN_SEPARATION_MAX_BYTES];+ size_t pre_len;+ int ret;++ if (hashalg == MLD_PREHASH_NONE)+ {+ ret = MLD_ERR_INVALID_ARG;+ goto cleanup;+ }++ pre_len = mld_prepare_domain_separation_prefix(pre, ph, phlen, ctx, ctxlen,+ hashalg);+ if (pre_len == 0)+ {+ ret = MLD_ERR_INVALID_ARG;+ goto cleanup;+ }++ ret = mld_sign_signature_internal(sig, pre, pre_len, NULL, 0, rnd, sk, 0,+ context);+cleanup:+ if (ret != 0)+ {+ /* To be on the safe-side, make sure sig has a well-defined value, even in+ * the case of error.+ *+ * If we come from mld_sign_signature_internal, this is redundant, but the+ * error case should not be the norm, and the added cost of the zeroization+ * insignificant. */+ mld_zeroize(sig, MLDSA_CRYPTO_BYTES);+ }++ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ mld_zeroize(pre, sizeof(pre));+ return ret;+}+#endif /* !MLD_CONFIG_NO_SIGN_API */++#if !defined(MLD_CONFIG_NO_VERIFY_API)+MLD_MUST_CHECK_RETURN_VALUE+MLD_EXTERNAL_API+int mld_sign_verify_pre_hash_internal(+ const uint8_t sig[MLDSA_CRYPTO_BYTES], const uint8_t *ph, size_t phlen,+ const uint8_t *ctx, size_t ctxlen,+ const uint8_t pk[MLDSA_CRYPTO_PUBLICKEYBYTES], int hashalg,+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+{+ MLD_ALIGN uint8_t pre[MLD_DOMAIN_SEPARATION_MAX_BYTES];+ size_t pre_len;+ int ret;++ if (hashalg == MLD_PREHASH_NONE)+ {+ ret = MLD_ERR_INVALID_ARG;+ goto cleanup;+ }++ pre_len = mld_prepare_domain_separation_prefix(pre, ph, phlen, ctx, ctxlen,+ hashalg);+ if (pre_len == 0)+ {+ ret = MLD_ERR_INVALID_ARG;+ goto cleanup;+ }++ ret = mld_sign_verify_internal(sig, pre, pre_len, NULL, 0, pk, 0, context);++cleanup:+ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ mld_zeroize(pre, sizeof(pre));+ return ret;+}+#endif /* !MLD_CONFIG_NO_VERIFY_API */++#if !defined(MLD_CONFIG_NO_SIGN_API)+MLD_MUST_CHECK_RETURN_VALUE+MLD_EXTERNAL_API+int mld_sign_signature_pre_hash_shake256(+ uint8_t sig[MLDSA_CRYPTO_BYTES], const uint8_t *m, size_t mlen,+ const uint8_t *ctx, size_t ctxlen, const uint8_t rnd[MLDSA_RNDBYTES],+ const uint8_t sk[MLDSA_CRYPTO_SECRETKEYBYTES],+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+{+ MLD_ALIGN uint8_t ph[64];+ int ret;+ mld_shake256(ph, sizeof(ph), m, mlen);+ ret = mld_sign_signature_pre_hash_internal(sig, ph, sizeof(ph), ctx, ctxlen,+ rnd, sk, MLD_PREHASH_SHAKE_256,+ context);+ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ mld_zeroize(ph, sizeof(ph));+ return ret;+}+#endif /* !MLD_CONFIG_NO_SIGN_API */++#if !defined(MLD_CONFIG_NO_VERIFY_API)+MLD_MUST_CHECK_RETURN_VALUE+MLD_EXTERNAL_API+int mld_sign_verify_pre_hash_shake256(+ const uint8_t sig[MLDSA_CRYPTO_BYTES], const uint8_t *m, size_t mlen,+ const uint8_t *ctx, size_t ctxlen,+ const uint8_t pk[MLDSA_CRYPTO_PUBLICKEYBYTES],+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+{+ MLD_ALIGN uint8_t ph[64];+ int ret;+ mld_shake256(ph, sizeof(ph), m, mlen);+ ret = mld_sign_verify_pre_hash_internal(sig, ph, sizeof(ph), ctx, ctxlen, pk,+ MLD_PREHASH_SHAKE_256, context);+ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ mld_zeroize(ph, sizeof(ph));+ return ret;+}+#endif /* !MLD_CONFIG_NO_VERIFY_API */++#if !defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API)+#define MLD_PRE_HASH_OID_LEN 11++/**+ * Return the OID of a given SHA-2/SHA-3 hash function.+ *+ * @param[out] oid Pointer to output OID.+ * @param hashalg Hash algorithm constant (MLD_PREHASH_*).+ */+static void mld_get_hash_oid(uint8_t oid[MLD_PRE_HASH_OID_LEN], int hashalg)+{+ unsigned int i;+ static const struct+ {+ int alg;+ uint8_t oid[MLD_PRE_HASH_OID_LEN];+ } oid_map[] = {+ {MLD_PREHASH_SHA2_224,+ {0x06, 0x09, 0x60, 0x86, 0x48, 0x01, 0x65, 0x03, 0x04, 0x02, 0x04}},+ {MLD_PREHASH_SHA2_256,+ {0x06, 0x09, 0x60, 0x86, 0x48, 0x01, 0x65, 0x03, 0x04, 0x02, 0x01}},+ {MLD_PREHASH_SHA2_384,+ {0x06, 0x09, 0x60, 0x86, 0x48, 0x01, 0x65, 0x03, 0x04, 0x02, 0x02}},+ {MLD_PREHASH_SHA2_512,+ {0x06, 0x09, 0x60, 0x86, 0x48, 0x01, 0x65, 0x03, 0x04, 0x02, 0x03}},+ {MLD_PREHASH_SHA2_512_224,+ {0x06, 0x09, 0x60, 0x86, 0x48, 0x01, 0x65, 0x03, 0x04, 0x02, 0x05}},+ {MLD_PREHASH_SHA2_512_256,+ {0x06, 0x09, 0x60, 0x86, 0x48, 0x01, 0x65, 0x03, 0x04, 0x02, 0x06}},+ {MLD_PREHASH_SHA3_224,+ {0x06, 0x09, 0x60, 0x86, 0x48, 0x01, 0x65, 0x03, 0x04, 0x02, 0x07}},+ {MLD_PREHASH_SHA3_256,+ {0x06, 0x09, 0x60, 0x86, 0x48, 0x01, 0x65, 0x03, 0x04, 0x02, 0x08}},+ {MLD_PREHASH_SHA3_384,+ {0x06, 0x09, 0x60, 0x86, 0x48, 0x01, 0x65, 0x03, 0x04, 0x02, 0x09}},+ {MLD_PREHASH_SHA3_512,+ {0x06, 0x09, 0x60, 0x86, 0x48, 0x01, 0x65, 0x03, 0x04, 0x02, 0x0A}},+ {MLD_PREHASH_SHAKE_128,+ {0x06, 0x09, 0x60, 0x86, 0x48, 0x01, 0x65, 0x03, 0x04, 0x02, 0x0B}},+ {MLD_PREHASH_SHAKE_256,+ {0x06, 0x09, 0x60, 0x86, 0x48, 0x01, 0x65, 0x03, 0x04, 0x02, 0x0C}}};++ for (i = 0; i < sizeof(oid_map) / sizeof(oid_map[0]); i++)+ __loop__(+ invariant(i <= sizeof(oid_map) / sizeof(oid_map[0]))+ decreases(sizeof(oid_map) / sizeof(oid_map[0]) - i)+ )+ {+ if (oid_map[i].alg == hashalg)+ {+ mld_memcpy(oid, oid_map[i].oid, MLD_PRE_HASH_OID_LEN);+ return;+ }+ }+}++static int mld_validate_hash_length(int hashalg, size_t len)+{+ switch (hashalg)+ {+ case MLD_PREHASH_SHA2_224:+ return (len == 224 / 8) ? 0 : -1;+ case MLD_PREHASH_SHA2_256:+ return (len == 256 / 8) ? 0 : -1;+ case MLD_PREHASH_SHA2_384:+ return (len == 384 / 8) ? 0 : -1;+ case MLD_PREHASH_SHA2_512:+ return (len == 512 / 8) ? 0 : -1;+ case MLD_PREHASH_SHA2_512_224:+ return (len == 224 / 8) ? 0 : -1;+ case MLD_PREHASH_SHA2_512_256:+ return (len == 256 / 8) ? 0 : -1;+ case MLD_PREHASH_SHA3_224:+ return (len == 224 / 8) ? 0 : -1;+ case MLD_PREHASH_SHA3_256:+ return (len == 256 / 8) ? 0 : -1;+ case MLD_PREHASH_SHA3_384:+ return (len == 384 / 8) ? 0 : -1;+ case MLD_PREHASH_SHA3_512:+ return (len == 512 / 8) ? 0 : -1;+ case MLD_PREHASH_SHAKE_128:+ return (len == 256 / 8) ? 0 : -1;+ case MLD_PREHASH_SHAKE_256:+ return (len == 512 / 8) ? 0 : -1;+ default:+ return -1;+ }+}++MLD_EXTERNAL_API+size_t mld_prepare_domain_separation_prefix(+ uint8_t prefix[MLD_DOMAIN_SEPARATION_MAX_BYTES], const uint8_t *ph,+ size_t phlen, const uint8_t *ctx, size_t ctxlen, int hashalg)+{+ if (ctxlen > 255)+ {+ return 0;+ }++ if (hashalg != MLD_PREHASH_NONE)+ {+ if (ph == NULL || mld_validate_hash_length(hashalg, phlen) != 0)+ {+ return 0;+ }+ }++ /* Common prefix: 0x00/0x01 || ctxlen || ctx */+ prefix[0] = (hashalg == MLD_PREHASH_NONE) ? 0 : 1;+ prefix[1] = (uint8_t)ctxlen;+ if (ctxlen > 0)+ {+ mld_memcpy(prefix + 2, ctx, ctxlen);+ }++ if (hashalg == MLD_PREHASH_NONE)+ {+ return 2 + ctxlen;+ }++ /* HashML-DSA: append oid || ph */+ mld_get_hash_oid(prefix + 2 + ctxlen, hashalg);+ mld_memcpy(prefix + 2 + ctxlen + MLD_PRE_HASH_OID_LEN, ph, phlen);+ return 2 + ctxlen + MLD_PRE_HASH_OID_LEN + phlen;+}+#endif /* !MLD_CONFIG_NO_SIGN_API || !MLD_CONFIG_NO_VERIFY_API */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API)+MLD_EXTERNAL_API+int mld_sign_pk_from_sk(uint8_t pk[MLDSA_CRYPTO_PUBLICKEYBYTES],+ const uint8_t sk[MLDSA_CRYPTO_SECRETKEYBYTES],+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+{+ uint8_t check, cmp0, cmp1, chk1, chk2;+ int ret;+ MLD_ALLOC(rho, uint8_t, MLDSA_SEEDBYTES, context);+ MLD_ALLOC(tr, uint8_t, MLDSA_TRBYTES, context);+ MLD_ALLOC(tr_computed, uint8_t, MLDSA_TRBYTES, context);+ MLD_ALLOC(key, uint8_t, MLDSA_SEEDBYTES, context);+ MLD_ALLOC(s1, mld_polyvecl, 1, context);+ MLD_ALLOC(s2, mld_polyveck, 1, context);+ MLD_ALLOC(t0_packed, uint8_t, MLDSA_K *MLDSA_POLYT0_PACKEDBYTES, context);++ if (rho == NULL || tr == NULL || tr_computed == NULL || key == NULL ||+ s1 == NULL || s2 == NULL || t0_packed == NULL)+ {+ ret = MLD_ERR_OUT_OF_MEMORY;+ goto cleanup;+ }++ /* Inline unpack_sk: mld_unpack_sk uses lazy types for s1/s2/t0 which+ * we cannot use here. t0 stays in packed form -- we compare it against+ * the recomputed value below. */+ mld_memcpy(rho, sk + MLDSA_SK_RHO_OFFSET, MLDSA_SEEDBYTES);+ mld_memcpy(key, sk + MLDSA_SK_KEY_OFFSET, MLDSA_SEEDBYTES);+ mld_memcpy(tr, sk + MLDSA_SK_TR_OFFSET, MLDSA_TRBYTES);+ mld_polyvecl_unpack_eta(s1, sk + MLDSA_SK_S1_OFFSET);+ mld_polyveck_unpack_eta(s2, sk + MLDSA_SK_S2_OFFSET);++ /* Validate s1 and s2 coefficients are within [-MLDSA_ETA, MLDSA_ETA] */+ chk1 = mld_polyvecl_chknorm(s1, MLDSA_ETA + 1) & 0xFF;+ chk2 = mld_polyveck_chknorm(s2, MLDSA_ETA + 1) & 0xFF;++ /* NTT s1 in place to use as s1hat */+ mld_polyvecl_ntt(s1);++ /* Pack rho into pk */+ mld_memcpy(pk + MLDSA_PK_RHO_OFFSET, rho, MLDSA_SEEDBYTES);++ /* Recompute t row by row, decompose, and pack t1 into pk and t0 into+ * t0_packed. */+ ret = mld_compute_pack_t0_t1(pk + MLDSA_PK_T1_OFFSET, t0_packed, s1, s2, rho,+ context);+ if (ret != 0)+ {+ goto cleanup;+ }++ /* Compare recomputed packed t0 against the t0 region of sk. */+ cmp0 = mld_ct_memcmp(t0_packed, sk + MLDSA_SK_T0_OFFSET,+ MLDSA_K * MLDSA_POLYT0_PACKEDBYTES);++ /* Compute tr_computed = H(pk) and compare to the stored tr */+ mld_shake256(tr_computed, MLDSA_TRBYTES, pk, MLDSA_CRYPTO_PUBLICKEYBYTES);+ cmp1 = mld_ct_memcmp((const uint8_t *)tr, (const uint8_t *)tr_computed,+ MLDSA_TRBYTES);+ check = mld_value_barrier_u8(cmp0 | cmp1 | chk1 | chk2);++ /* Declassify the final result of the validity check. */+ MLD_CT_TESTING_DECLASSIFY(&check, sizeof(check));+ ret = (check != 0) ? MLD_ERR_INVALID_KEY : 0;++cleanup:++ if (ret != 0)+ {+ mld_zeroize(pk, MLDSA_CRYPTO_PUBLICKEYBYTES);+ }++ /* Constant time: pk is either the valid public key or zeroed on error */+ MLD_CT_TESTING_DECLASSIFY(pk, MLDSA_CRYPTO_PUBLICKEYBYTES);++ /* @[FIPS204, Section 3.6.3] Destruction of intermediate values. */+ MLD_FREE(t0_packed, uint8_t, MLDSA_K *MLDSA_POLYT0_PACKEDBYTES, context);+ MLD_FREE(s2, mld_polyveck, 1, context);+ MLD_FREE(s1, mld_polyvecl, 1, context);+ MLD_FREE(key, uint8_t, MLDSA_SEEDBYTES, context);+ MLD_FREE(tr_computed, uint8_t, MLDSA_TRBYTES, context);+ MLD_FREE(tr, uint8_t, MLDSA_TRBYTES, context);+ MLD_FREE(rho, uint8_t, MLDSA_SEEDBYTES, context);++ return ret;+}+#endif /* !MLD_CONFIG_NO_KEYPAIR_API */+#endif /* !MLD_CONFIG_CORE_API_ONLY */++/* To facilitate single-compilation-unit (SCU) builds, undefine all macros.+ * Don't modify by hand -- this is auto-generated by scripts/autogen. */+#undef mld_check_pct+#undef mld_sample_s1_s2+#undef mld_validate_hash_length+#undef mld_get_hash_oid+#undef mld_H+#undef mld_compute_pack_z+#undef mld_attempt_signature_generation+#undef mld_compute_pack_t0_t1+#undef mld_get_max_signing_attempts+#undef MLD_MAX_SIGNING_ATTEMPTS+#undef MLD_PRE_HASH_OID_LEN
@@ -0,0 +1,850 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS204]+ * FIPS 204 Module-Lattice-Based Digital Signature Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/204/final+ */++#ifndef MLD_SIGN_H+#define MLD_SIGN_H++#include <stddef.h>+#include "cbmc.h"+#include "common.h"+#include "poly.h"+#include "polyvec.h"+#include "sys.h"++#if defined(MLD_CHECK_APIS)+/* Include to ensure consistency between internal sign.h+ * and external mldsa_native.h. */+#include "mldsa_native.h"++#if MLDSA_CRYPTO_SECRETKEYBYTES != \+ MLDSA_SECRETKEYBYTES(MLD_CONFIG_PARAMETER_SET)+#error Mismatch for SECRETKEYBYTES between sign.h and mldsa_native.h+#endif++#if MLDSA_CRYPTO_PUBLICKEYBYTES != \+ MLDSA_PUBLICKEYBYTES(MLD_CONFIG_PARAMETER_SET)+#error Mismatch for PUBLICKEYBYTES between sign.h and mldsa_native.h+#endif++#if MLDSA_CRYPTO_BYTES != MLDSA_BYTES(MLD_CONFIG_PARAMETER_SET)+#error Mismatch for BYTES between sign.h and mldsa_native.h+#endif++#endif /* MLD_CHECK_APIS */++#define mld_sign_keypair_internal \+ MLD_NAMESPACE_KL(keypair_internal) MLD_CONTEXT_PARAMETERS_3+#define mld_sign_keypair MLD_NAMESPACE_KL(keypair) MLD_CONTEXT_PARAMETERS_2+#define mld_sign_signature_internal \+ MLD_NAMESPACE_KL(signature_internal) MLD_CONTEXT_PARAMETERS_8+#define mld_sign_signature MLD_NAMESPACE_KL(signature) MLD_CONTEXT_PARAMETERS_6+#define mld_sign_signature_extmu \+ MLD_NAMESPACE_KL(signature_extmu) MLD_CONTEXT_PARAMETERS_3+#define mld_sign_verify_internal \+ MLD_NAMESPACE_KL(verify_internal) MLD_CONTEXT_PARAMETERS_7+#define mld_sign_verify MLD_NAMESPACE_KL(verify) MLD_CONTEXT_PARAMETERS_6+#define mld_sign_verify_extmu \+ MLD_NAMESPACE_KL(verify_extmu) MLD_CONTEXT_PARAMETERS_3+#define mld_sign_signature_pre_hash_internal \+ MLD_NAMESPACE_KL(signature_pre_hash_internal) MLD_CONTEXT_PARAMETERS_8+#define mld_sign_verify_pre_hash_internal \+ MLD_NAMESPACE_KL(verify_pre_hash_internal) MLD_CONTEXT_PARAMETERS_7+#define mld_sign_signature_pre_hash_shake256 \+ MLD_NAMESPACE_KL(signature_pre_hash_shake256) MLD_CONTEXT_PARAMETERS_7+#define mld_sign_verify_pre_hash_shake256 \+ MLD_NAMESPACE_KL(verify_pre_hash_shake256) MLD_CONTEXT_PARAMETERS_6+#define mld_prepare_domain_separation_prefix \+ MLD_NAMESPACE_KL(prepare_domain_separation_prefix)+#define mld_sign_pk_from_sk \+ MLD_NAMESPACE_KL(pk_from_sk) MLD_CONTEXT_PARAMETERS_2++/* Hash algorithm constants for domain separation */+#define MLD_PREHASH_NONE 0+#define MLD_PREHASH_SHA2_224 1+#define MLD_PREHASH_SHA2_256 2+#define MLD_PREHASH_SHA2_384 3+#define MLD_PREHASH_SHA2_512 4+#define MLD_PREHASH_SHA2_512_224 5+#define MLD_PREHASH_SHA2_512_256 6+#define MLD_PREHASH_SHA3_224 7+#define MLD_PREHASH_SHA3_256 8+#define MLD_PREHASH_SHA3_384 9+#define MLD_PREHASH_SHA3_512 10+#define MLD_PREHASH_SHAKE_128 11+#define MLD_PREHASH_SHAKE_256 12++#if !defined(MLD_CONFIG_NO_KEYPAIR_API)+/**+ * Generate a public-private key pair from a seed.+ *+ * When MLD_CONFIG_KEYGEN_PCT is set, performs a Pairwise Consistency Test+ * (PCT) as required by FIPS 140-3 IG.+ *+ * @spec{Implements @[FIPS204, Algorithm 6, ML-DSA.KeyGen_internal].}+ *+ * @param[out] pk Output public key.+ * @param[out] sk Output private key.+ * @param[in] seed Input random seed.+ * @param context Application context. Only present when+ * MLD_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLD_ERR_OUT_OF_MEMORY MLD_CONFIG_CUSTOM_ALLOC_FREE was+ * used and an allocation via+ * MLD_CUSTOM_ALLOC returned NULL.+ * @retval MLD_ERR_RNG_FAIL Random number generation failed+ * during the PCT. Only possible when+ * MLD_CONFIG_KEYGEN_PCT is enabled.+ * @retval MLD_ERR_SIGNING_PAUSED The PCT's signing step was paused by+ * a MLD_CONFIG_SIGN_HOOK_ATTEMPT hook.+ * This should currently never happen:+ * signing hooks require+ * MLD_CONFIG_NO_RANDOMIZED_API, which+ * is incompatible with+ * MLD_CONFIG_KEYGEN_PCT, so the two+ * cannot be enabled simultaneously.+ * @retval MLD_ERR_PCT_FAIL MLD_CONFIG_KEYGEN_PCT is enabled and+ * the PCT check failed.+ */+MLD_MUST_CHECK_RETURN_VALUE+MLD_EXTERNAL_API+int mld_sign_keypair_internal(uint8_t pk[MLDSA_CRYPTO_PUBLICKEYBYTES],+ uint8_t sk[MLDSA_CRYPTO_SECRETKEYBYTES],+ const uint8_t seed[MLDSA_SEEDBYTES],+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+__contract__(+ requires(memory_no_alias(pk, MLDSA_CRYPTO_PUBLICKEYBYTES))+ requires(memory_no_alias(sk, MLDSA_CRYPTO_SECRETKEYBYTES))+ requires(memory_no_alias(seed, MLDSA_SEEDBYTES))+ assigns(object_whole(pk))+ assigns(object_whole(sk))+ ensures(return_value == 0 || return_value == MLD_ERR_OUT_OF_MEMORY ||+ return_value == MLD_ERR_RNG_FAIL ||+ return_value == MLD_ERR_SIGNING_PAUSED ||+ return_value == MLD_ERR_PCT_FAIL)+ /* Output buffers on error, per API-CONVENTIONS.md */+ ensures(return_value != 0 ==>+ array_unchanged_or_zeroized_u8(pk, MLDSA_CRYPTO_PUBLICKEYBYTES))+ ensures(return_value != 0 ==>+ array_unchanged_or_zeroized_u8(sk, MLDSA_CRYPTO_SECRETKEYBYTES))+);++#if !defined(MLD_CONFIG_CORE_API_ONLY)+#if !defined(MLD_CONFIG_NO_RANDOMIZED_API)+/**+ * Generate a public-private key pair.+ *+ * When MLD_CONFIG_KEYGEN_PCT is set, performs a Pairwise Consistency Test+ * (PCT) as required by FIPS 140-3 IG.+ *+ * @spec{Implements @[FIPS204, Algorithm 1, ML-DSA.KeyGen].}+ *+ * @param[out] pk Output public key.+ * @param[out] sk Output private key.+ * @param context Application context. Only present when+ * MLD_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLD_ERR_OUT_OF_MEMORY MLD_CONFIG_CUSTOM_ALLOC_FREE was+ * used and an allocation via+ * MLD_CUSTOM_ALLOC returned NULL.+ * @retval MLD_ERR_RNG_FAIL Random number generation failed.+ * @retval MLD_ERR_SIGNING_PAUSED The PCT's signing step was paused by+ * a MLD_CONFIG_SIGN_HOOK_ATTEMPT hook.+ * This should currently never happen:+ * signing hooks require+ * MLD_CONFIG_NO_RANDOMIZED_API, which+ * is incompatible with+ * MLD_CONFIG_KEYGEN_PCT, so the two+ * cannot be enabled simultaneously.+ * @retval MLD_ERR_PCT_FAIL MLD_CONFIG_KEYGEN_PCT is enabled and+ * the PCT check failed.+ */+MLD_MUST_CHECK_RETURN_VALUE+MLD_EXTERNAL_API+int mld_sign_keypair(uint8_t pk[MLDSA_CRYPTO_PUBLICKEYBYTES],+ uint8_t sk[MLDSA_CRYPTO_SECRETKEYBYTES],+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+__contract__(+ requires(memory_no_alias(pk, MLDSA_CRYPTO_PUBLICKEYBYTES))+ requires(memory_no_alias(sk, MLDSA_CRYPTO_SECRETKEYBYTES))+ assigns(object_whole(pk))+ assigns(object_whole(sk))+ ensures(return_value == 0 || return_value == MLD_ERR_OUT_OF_MEMORY ||+ return_value == MLD_ERR_RNG_FAIL ||+ return_value == MLD_ERR_SIGNING_PAUSED ||+ return_value == MLD_ERR_PCT_FAIL)+ /* Output buffers on error, per API-CONVENTIONS.md */+ ensures(return_value != 0 ==>+ array_unchanged_or_zeroized_u8(pk, MLDSA_CRYPTO_PUBLICKEYBYTES))+ ensures(return_value != 0 ==>+ array_unchanged_or_zeroized_u8(sk, MLDSA_CRYPTO_SECRETKEYBYTES))+);+#endif /* !MLD_CONFIG_NO_RANDOMIZED_API */+#endif /* !MLD_CONFIG_CORE_API_ONLY */+#endif /* !MLD_CONFIG_NO_KEYPAIR_API */++#if !defined(MLD_CONFIG_NO_SIGN_API)+/**+ * Compute signature using a caller-supplied random seed and prefix.+ *+ * On error (non-zero return value), the signature buffer sig is zeroized.+ *+ * @spec{Implements @[FIPS204, Algorithm 7, ML-DSA.Sign_internal].}+ *+ * @warning This function does not perform secret key validation.+ * Callers importing serialized keys can use mld_sign_pk_from_sk+ * to validate them before signing.+ *+ * @param[out] sig Pointer to buffer to hold the generated signature of+ * MLDSA_CRYPTO_BYTES bytes.+ * @param[in] m Pointer to message to be signed (when+ * externalmu == 0), or to a precomputed+ * message representative mu (when externalmu != 0).+ * @param mlen Length of m. Must equal MLDSA_CRHBYTES when+ * externalmu != 0.+ * @param[in] pre Pointer to prefix string. Ignored when+ * externalmu != 0.+ * @param prelen Length of prefix string. Ignored when+ * externalmu != 0.+ * @param[in] rnd Random seed.+ * @param[in] sk Bit-packed secret key; assumed to be valid.+ * @param externalmu 0: m/mlen is the raw message; mu = H(tr, pre, m) is+ * computed internally.+ * non-zero: m points to a precomputed mu of+ * MLDSA_CRHBYTES bytes; pre/prelen unused.+ * @param context Application context. Only present when+ * MLD_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLD_ERR_OUT_OF_MEMORY MLD_CONFIG_CUSTOM_ALLOC_FREE was+ * used and an allocation via+ * MLD_CUSTOM_ALLOC returned NULL.+ * @retval MLD_ERR_SIGN_ATTEMPTS_EXHAUSTED The rejection-sampling loop exceeded+ * MLD_CONFIG_MAX_SIGNING_ATTEMPTS+ * iterations.+ * @retval MLD_ERR_SIGNING_PAUSED A MLD_CONFIG_SIGN_HOOK_ATTEMPT hook+ * paused signing; re-invoke to resume.+ */+MLD_MUST_CHECK_RETURN_VALUE+MLD_EXTERNAL_API+int mld_sign_signature_internal(uint8_t sig[MLDSA_CRYPTO_BYTES],+ const uint8_t *m, size_t mlen,+ const uint8_t *pre, size_t prelen,+ const uint8_t rnd[MLDSA_RNDBYTES],+ const uint8_t sk[MLDSA_CRYPTO_SECRETKEYBYTES],+ int externalmu,+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+__contract__(+ requires(mlen <= MLD_MAX_BUFFER_SIZE)+ requires(prelen <= MLD_MAX_BUFFER_SIZE)+ requires(memory_no_alias(sig, MLDSA_CRYPTO_BYTES))+ requires(memory_no_alias(m, mlen))+ requires(memory_no_alias(rnd, MLDSA_RNDBYTES))+ requires(memory_no_alias(sk, MLDSA_CRYPTO_SECRETKEYBYTES))+ requires((externalmu == 0) ==> ((prelen == 0) || memory_no_alias(pre, prelen)))+ requires((externalmu != 0) ==> (mlen == MLDSA_CRHBYTES))+ assigns(memory_slice(sig, MLDSA_CRYPTO_BYTES))+ ensures(return_value == 0 ||+ return_value == MLD_ERR_OUT_OF_MEMORY ||+ return_value == MLD_ERR_SIGN_ATTEMPTS_EXHAUSTED ||+ return_value == MLD_ERR_SIGNING_PAUSED)+ /* Output buffers on error, per API-CONVENTIONS.md */+ ensures(return_value != 0 ==>+ array_unchanged_or_zeroized_u8(sig, MLDSA_CRYPTO_BYTES)));++#if !defined(MLD_CONFIG_CORE_API_ONLY)+#if !defined(MLD_CONFIG_NO_RANDOMIZED_API)+/**+ * Compute signature. This function implements the randomized variant of+ * ML-DSA. If you require the deterministic variant, use+ * mld_sign_signature_internal directly.+ *+ * @spec{Implements @[FIPS204, Algorithm 2, ML-DSA.Sign].}+ *+ * @warning This function does not perform secret key validation.+ * Callers importing serialized keys can use mld_sign_pk_from_sk+ * to validate them before signing.+ *+ * @param[out] sig Output signature.+ * @param[in] m Pointer to message to be signed.+ * @param mlen Length of message.+ * @param[in] ctx Pointer to context string. May be NULL if ctxlen == 0.+ * @param ctxlen Length of context string. Should be <= 255.+ * @param[in] sk Bit-packed secret key; assumed to be valid.+ * @param context Application context. Only present when+ * MLD_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLD_ERR_OUT_OF_MEMORY MLD_CONFIG_CUSTOM_ALLOC_FREE was+ * used and an allocation via+ * MLD_CUSTOM_ALLOC returned NULL.+ * @retval MLD_ERR_RNG_FAIL Random number generation failed.+ * @retval MLD_ERR_SIGN_ATTEMPTS_EXHAUSTED The rejection-sampling loop exceeded+ * MLD_CONFIG_MAX_SIGNING_ATTEMPTS+ * iterations.+ * @retval MLD_ERR_SIGNING_PAUSED A MLD_CONFIG_SIGN_HOOK_ATTEMPT hook+ * paused signing; re-invoke to resume.+ * @retval MLD_ERR_INVALID_ARG The context string exceeded 255+ * bytes.+ */+MLD_MUST_CHECK_RETURN_VALUE+MLD_EXTERNAL_API+int mld_sign_signature(uint8_t sig[MLDSA_CRYPTO_BYTES], const uint8_t *m,+ size_t mlen, const uint8_t *ctx, size_t ctxlen,+ const uint8_t sk[MLDSA_CRYPTO_SECRETKEYBYTES],+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+__contract__(+ requires(mlen <= MLD_MAX_BUFFER_SIZE)+ requires(memory_no_alias(sig, MLDSA_CRYPTO_BYTES))+ requires(memory_no_alias(m, mlen))+ requires(ctxlen <= MLD_MAX_BUFFER_SIZE)+ requires(ctxlen == 0 || memory_no_alias(ctx, ctxlen))+ requires(memory_no_alias(sk, MLDSA_CRYPTO_SECRETKEYBYTES))+ assigns(memory_slice(sig, MLDSA_CRYPTO_BYTES))+ ensures(return_value == 0 || return_value == MLD_ERR_OUT_OF_MEMORY ||+ return_value == MLD_ERR_RNG_FAIL ||+ return_value == MLD_ERR_SIGN_ATTEMPTS_EXHAUSTED ||+ return_value == MLD_ERR_SIGNING_PAUSED ||+ return_value == MLD_ERR_INVALID_ARG)+ ensures((return_value == MLD_ERR_INVALID_ARG) ==> (ctxlen > 255))+ /* Output buffers on error, per API-CONVENTIONS.md */+ ensures(return_value != 0 ==>+ array_unchanged_or_zeroized_u8(sig, MLDSA_CRYPTO_BYTES))+);++/**+ * Compute signature in "external mu" mode: the caller has already computed+ * the message representative mu = SHAKE256(tr || M', 64), where+ * tr = SHAKE256(pk, 64) and M' is the FIPS 204 formatted message (e.g.+ * 0x00 || ctxlen || ctx || msg for pure ML-DSA). This is the randomized+ * variant; for the deterministic variant, use mld_sign_signature_internal+ * directly with externalmu set to non-zero and an all-zero rnd.+ *+ * @spec{Implements @[FIPS204, Algorithm 2, ML-DSA.Sign external mu variant].}+ *+ * @warning This function does not perform secret key validation.+ * Callers importing serialized keys can use mld_sign_pk_from_sk+ * to validate them before signing.+ *+ * @param[out] sig Output signature.+ * @param[in] mu Precomputed message representative.+ * @param[in] sk Bit-packed secret key; assumed to be valid.+ * @param context Application context. Only present when+ * MLD_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLD_ERR_OUT_OF_MEMORY MLD_CONFIG_CUSTOM_ALLOC_FREE was+ * used and an allocation via+ * MLD_CUSTOM_ALLOC returned NULL.+ * @retval MLD_ERR_RNG_FAIL Random number generation failed.+ * @retval MLD_ERR_SIGN_ATTEMPTS_EXHAUSTED The rejection-sampling loop exceeded+ * MLD_CONFIG_MAX_SIGNING_ATTEMPTS+ * iterations.+ * @retval MLD_ERR_SIGNING_PAUSED A MLD_CONFIG_SIGN_HOOK_ATTEMPT hook+ * paused signing; re-invoke to resume.+ */+MLD_MUST_CHECK_RETURN_VALUE+MLD_EXTERNAL_API+int mld_sign_signature_extmu(uint8_t sig[MLDSA_CRYPTO_BYTES],+ const uint8_t mu[MLDSA_CRHBYTES],+ const uint8_t sk[MLDSA_CRYPTO_SECRETKEYBYTES],+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+__contract__(+ requires(memory_no_alias(sig, MLDSA_CRYPTO_BYTES))+ requires(memory_no_alias(mu, MLDSA_CRHBYTES))+ requires(memory_no_alias(sk, MLDSA_CRYPTO_SECRETKEYBYTES))+ assigns(memory_slice(sig, MLDSA_CRYPTO_BYTES))+ ensures(return_value == 0 || return_value == MLD_ERR_OUT_OF_MEMORY ||+ return_value == MLD_ERR_RNG_FAIL ||+ return_value == MLD_ERR_SIGN_ATTEMPTS_EXHAUSTED ||+ return_value == MLD_ERR_SIGNING_PAUSED)+ /* Output buffers on error, per API-CONVENTIONS.md */+ ensures(return_value != 0 ==>+ array_unchanged_or_zeroized_u8(sig, MLDSA_CRYPTO_BYTES))+);++#endif /* !MLD_CONFIG_NO_RANDOMIZED_API */+#endif /* !MLD_CONFIG_CORE_API_ONLY */+#endif /* !MLD_CONFIG_NO_SIGN_API */++#if !defined(MLD_CONFIG_NO_VERIFY_API)+/**+ * Verify signature.+ *+ * @spec{Implements @[FIPS204, Algorithm 8, ML-DSA.Verify_internal].}+ *+ * @param[in] sig Pointer to input signature of+ * MLDSA_CRYPTO_BYTES bytes.+ * @param[in] m Pointer to message (when externalmu == 0), or to a+ * precomputed message representative mu (when+ * externalmu != 0).+ * @param mlen Length of m. Must equal MLDSA_CRHBYTES when+ * externalmu != 0.+ * @param[in] pre Pointer to prefix string. Ignored when externalmu != 0.+ * @param prelen Length of prefix string. Ignored when externalmu != 0.+ * @param[in] pk Bit-packed public key.+ * @param externalmu 0: m/mlen is the raw message; mu = H(H(pk), pre, m) is+ * computed internally.+ * non-zero: m points to a precomputed mu of+ * MLDSA_CRHBYTES bytes; pre/prelen unused.+ * @param context Application context. Only present when+ * MLD_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLD_ERR_OUT_OF_MEMORY MLD_CONFIG_CUSTOM_ALLOC_FREE was+ * used and an allocation via+ * MLD_CUSTOM_ALLOC returned NULL.+ * @retval MLD_ERR_INVALID_SIGNATURE Signature verification failed.+ */+MLD_MUST_CHECK_RETURN_VALUE+MLD_EXTERNAL_API+int mld_sign_verify_internal(const uint8_t sig[MLDSA_CRYPTO_BYTES],+ const uint8_t *m, size_t mlen, const uint8_t *pre,+ size_t prelen,+ const uint8_t pk[MLDSA_CRYPTO_PUBLICKEYBYTES],+ int externalmu,+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+__contract__(+ requires(prelen <= MLD_MAX_BUFFER_SIZE)+ requires(mlen <= MLD_MAX_BUFFER_SIZE)+ requires(memory_no_alias(sig, MLDSA_CRYPTO_BYTES))+ requires(memory_no_alias(m, mlen))+ requires((externalmu == 0) ==> ((prelen == 0) || memory_no_alias(pre, prelen)))+ requires((externalmu != 0) ==> (mlen == MLDSA_CRHBYTES))+ requires(memory_no_alias(pk, MLDSA_CRYPTO_PUBLICKEYBYTES))+ ensures(return_value == 0 || return_value == MLD_ERR_INVALID_SIGNATURE || return_value == MLD_ERR_OUT_OF_MEMORY)+);++#if !defined(MLD_CONFIG_CORE_API_ONLY)+/**+ * Verify signature.+ *+ * @spec{Implements @[FIPS204, Algorithm 3, ML-DSA.Verify].}+ *+ * @param[in] sig Pointer to input signature.+ * @param[in] m Pointer to message.+ * @param mlen Length of message.+ * @param[in] ctx Pointer to context string. May be NULL if ctxlen == 0.+ * @param ctxlen Length of context string.+ * @param[in] pk Bit-packed public key.+ * @param context Application context. Only present when+ * MLD_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLD_ERR_OUT_OF_MEMORY MLD_CONFIG_CUSTOM_ALLOC_FREE was+ * used and an allocation via+ * MLD_CUSTOM_ALLOC returned NULL.+ * @retval MLD_ERR_INVALID_SIGNATURE Signature verification failed.+ * @retval MLD_ERR_INVALID_ARG The context string exceeded 255 bytes.+ */+MLD_MUST_CHECK_RETURN_VALUE+MLD_EXTERNAL_API+int mld_sign_verify(const uint8_t sig[MLDSA_CRYPTO_BYTES], const uint8_t *m,+ size_t mlen, const uint8_t *ctx, size_t ctxlen,+ const uint8_t pk[MLDSA_CRYPTO_PUBLICKEYBYTES],+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+__contract__(+ requires(mlen <= MLD_MAX_BUFFER_SIZE)+ requires(ctxlen <= MLD_MAX_BUFFER_SIZE)+ requires(memory_no_alias(sig, MLDSA_CRYPTO_BYTES))+ requires(memory_no_alias(m, mlen))+ requires(ctxlen == 0 || memory_no_alias(ctx, ctxlen))+ requires(memory_no_alias(pk, MLDSA_CRYPTO_PUBLICKEYBYTES))+ ensures(return_value == 0 || return_value == MLD_ERR_INVALID_SIGNATURE || return_value == MLD_ERR_INVALID_ARG || return_value == MLD_ERR_OUT_OF_MEMORY)+ ensures((return_value == MLD_ERR_INVALID_ARG) ==> (ctxlen > 255))+);++/**+ * Verify signature in "external mu" mode: the caller has already computed+ * the message representative mu = SHAKE256(tr || M', 64), where+ * tr = SHAKE256(pk, 64) and M' is the FIPS 204 formatted message (e.g.+ * 0x00 || ctxlen || ctx || msg for pure ML-DSA). The same mu must have been+ * used at signing time.+ *+ * @spec{Implements @[FIPS204, Algorithm 3, ML-DSA.Verify external mu variant].}+ *+ * @param[in] sig Pointer to input signature.+ * @param[in] mu Precomputed message representative.+ * @param[in] pk Bit-packed public key.+ * @param context Application context. Only present when+ * MLD_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLD_ERR_OUT_OF_MEMORY MLD_CONFIG_CUSTOM_ALLOC_FREE was+ * used and an allocation via+ * MLD_CUSTOM_ALLOC returned NULL.+ * @retval MLD_ERR_INVALID_SIGNATURE Signature verification failed.+ */+MLD_MUST_CHECK_RETURN_VALUE+MLD_EXTERNAL_API+int mld_sign_verify_extmu(const uint8_t sig[MLDSA_CRYPTO_BYTES],+ const uint8_t mu[MLDSA_CRHBYTES],+ const uint8_t pk[MLDSA_CRYPTO_PUBLICKEYBYTES],+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+__contract__( requires(memory_no_alias(sig, MLDSA_CRYPTO_BYTES))+ requires(memory_no_alias(mu, MLDSA_CRHBYTES))+ requires(memory_no_alias(pk, MLDSA_CRYPTO_PUBLICKEYBYTES))+ ensures(return_value == 0 || return_value == MLD_ERR_INVALID_SIGNATURE || return_value == MLD_ERR_OUT_OF_MEMORY)+);++#endif /* !MLD_CONFIG_CORE_API_ONLY */+#endif /* !MLD_CONFIG_NO_VERIFY_API */++#if !defined(MLD_CONFIG_CORE_API_ONLY)+#if !defined(MLD_CONFIG_NO_SIGN_API)+/**+ * Compute signature with pre-hashed message.+ *+ * @spec{Implements @[FIPS204, Algorithm 4, HashML-DSA.Sign].}+ *+ * Supported hash algorithm constants:+ * MLD_PREHASH_SHA2_224, MLD_PREHASH_SHA2_256, MLD_PREHASH_SHA2_384,+ * MLD_PREHASH_SHA2_512, MLD_PREHASH_SHA2_512_224, MLD_PREHASH_SHA2_512_256,+ * MLD_PREHASH_SHA3_224, MLD_PREHASH_SHA3_256, MLD_PREHASH_SHA3_384,+ * MLD_PREHASH_SHA3_512, MLD_PREHASH_SHAKE_128, MLD_PREHASH_SHAKE_256.+ *+ * MLD_PREHASH_NONE is rejected by this API.+ *+ * @warning This is an unstable API that may change in the future. If you need+ * a stable API use mld_sign_signature_pre_hash_shake256.+ *+ * @warning This function does not perform secret key validation.+ * Callers importing serialized keys can use mld_sign_pk_from_sk+ * to validate them before signing.+ *+ * @param[out] sig Output signature.+ * @param[in] ph Pointer to pre-hashed message.+ * @param phlen Length of pre-hashed message. Must match the output+ * length of hashalg (the digest size for SHA-2/SHA-3,+ * 32 bytes for MLD_PREHASH_SHAKE_128, 64 bytes for+ * MLD_PREHASH_SHAKE_256).+ * @param[in] ctx Pointer to context string. May be NULL if ctxlen == 0.+ * @param ctxlen Length of context string.+ * @param[in] rnd Random seed.+ * @param[in] sk Bit-packed secret key; assumed to be valid.+ * @param hashalg Hash algorithm constant (one of MLD_PREHASH_*).+ * @param context Application context. Only present when+ * MLD_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLD_ERR_OUT_OF_MEMORY MLD_CONFIG_CUSTOM_ALLOC_FREE was+ * used and an allocation via+ * MLD_CUSTOM_ALLOC returned NULL.+ * @retval MLD_ERR_SIGN_ATTEMPTS_EXHAUSTED The rejection-sampling loop exceeded+ * MLD_CONFIG_MAX_SIGNING_ATTEMPTS+ * iterations.+ * @retval MLD_ERR_SIGNING_PAUSED A MLD_CONFIG_SIGN_HOOK_ATTEMPT hook+ * paused signing; re-invoke to resume.+ * @retval MLD_ERR_INVALID_ARG The pre-hash algorithm was+ * MLD_PREHASH_NONE or unsupported,+ * phlen did not match the output+ * length of hashalg, or the context+ * string exceeded 255 bytes.+ */+MLD_MUST_CHECK_RETURN_VALUE+MLD_EXTERNAL_API+int mld_sign_signature_pre_hash_internal(+ uint8_t sig[MLDSA_CRYPTO_BYTES], const uint8_t *ph, size_t phlen,+ const uint8_t *ctx, size_t ctxlen, const uint8_t rnd[MLDSA_RNDBYTES],+ const uint8_t sk[MLDSA_CRYPTO_SECRETKEYBYTES], int hashalg,+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+__contract__(+ requires(ctxlen <= MLD_MAX_BUFFER_SIZE)+ requires(phlen <= MLD_MAX_BUFFER_SIZE)+ requires(memory_no_alias(sig, MLDSA_CRYPTO_BYTES))+ requires(memory_no_alias(ph, phlen))+ requires(ctxlen == 0 || memory_no_alias(ctx, ctxlen))+ requires(memory_no_alias(rnd, MLDSA_RNDBYTES))+ requires(memory_no_alias(sk, MLDSA_CRYPTO_SECRETKEYBYTES))+ assigns(memory_slice(sig, MLDSA_CRYPTO_BYTES))+ ensures(return_value == 0 || return_value == MLD_ERR_OUT_OF_MEMORY || return_value == MLD_ERR_SIGN_ATTEMPTS_EXHAUSTED || return_value == MLD_ERR_SIGNING_PAUSED || return_value == MLD_ERR_INVALID_ARG)+ /* Output buffers on error, per API-CONVENTIONS.md */+ ensures(return_value != 0 ==>+ array_unchanged_or_zeroized_u8(sig, MLDSA_CRYPTO_BYTES))+);+#endif /* !MLD_CONFIG_NO_SIGN_API */++#if !defined(MLD_CONFIG_NO_VERIFY_API)+/**+ * Verify signature with pre-hashed message.+ *+ * @spec{Implements @[FIPS204, Algorithm 5, HashML-DSA.Verify].}+ *+ * Supported hash algorithm constants:+ * MLD_PREHASH_SHA2_224, MLD_PREHASH_SHA2_256, MLD_PREHASH_SHA2_384,+ * MLD_PREHASH_SHA2_512, MLD_PREHASH_SHA2_512_224, MLD_PREHASH_SHA2_512_256,+ * MLD_PREHASH_SHA3_224, MLD_PREHASH_SHA3_256, MLD_PREHASH_SHA3_384,+ * MLD_PREHASH_SHA3_512, MLD_PREHASH_SHAKE_128, MLD_PREHASH_SHAKE_256.+ *+ * MLD_PREHASH_NONE is rejected by this API.+ *+ * @warning This is an unstable API that may change in the future. If you need+ * a stable API use mld_sign_verify_pre_hash_shake256.+ *+ * @param[in] sig Pointer to input signature.+ * @param[in] ph Pointer to pre-hashed message.+ * @param phlen Length of pre-hashed message. Must match the output+ * length of hashalg (the digest size for SHA-2/SHA-3,+ * 32 bytes for MLD_PREHASH_SHAKE_128, 64 bytes for+ * MLD_PREHASH_SHAKE_256).+ * @param[in] ctx Pointer to context string. May be NULL if ctxlen == 0.+ * @param ctxlen Length of context string.+ * @param[in] pk Bit-packed public key.+ * @param hashalg Hash algorithm constant (one of MLD_PREHASH_*).+ * @param context Application context. Only present when+ * MLD_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLD_ERR_OUT_OF_MEMORY MLD_CONFIG_CUSTOM_ALLOC_FREE was+ * used and an allocation via+ * MLD_CUSTOM_ALLOC returned NULL.+ * @retval MLD_ERR_INVALID_SIGNATURE Signature verification failed.+ * @retval MLD_ERR_INVALID_ARG The pre-hash algorithm was+ * MLD_PREHASH_NONE or unsupported, phlen+ * did not match the output length of+ * hashalg, or the context string exceeded+ * 255 bytes.+ */+MLD_MUST_CHECK_RETURN_VALUE+MLD_EXTERNAL_API+int mld_sign_verify_pre_hash_internal(+ const uint8_t sig[MLDSA_CRYPTO_BYTES], const uint8_t *ph, size_t phlen,+ const uint8_t *ctx, size_t ctxlen,+ const uint8_t pk[MLDSA_CRYPTO_PUBLICKEYBYTES], int hashalg,+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+__contract__(+ requires(phlen <= MLD_MAX_BUFFER_SIZE)+ requires(ctxlen <= MLD_MAX_BUFFER_SIZE - 77)+ requires(memory_no_alias(sig, MLDSA_CRYPTO_BYTES))+ requires(memory_no_alias(ph, phlen))+ requires(ctxlen == 0 || memory_no_alias(ctx, ctxlen))+ requires(memory_no_alias(pk, MLDSA_CRYPTO_PUBLICKEYBYTES))+ ensures(return_value == 0 || return_value == MLD_ERR_INVALID_SIGNATURE || return_value == MLD_ERR_INVALID_ARG || return_value == MLD_ERR_OUT_OF_MEMORY)+);+#endif /* !MLD_CONFIG_NO_VERIFY_API */++#if !defined(MLD_CONFIG_NO_SIGN_API)+/**+ * Compute signature with pre-hashed message using SHAKE256. This function+ * computes the SHAKE256 hash of the message internally.+ *+ * @spec{Implements @[FIPS204, Algorithm 4, HashML-DSA.Sign] with SHAKE256 as+ * the pre-hash.}+ *+ * @warning This function does not perform secret key validation.+ * Callers importing serialized keys can use mld_sign_pk_from_sk+ * to validate them before signing.+ *+ * @param[out] sig Output signature.+ * @param[in] m Pointer to message to be hashed and signed.+ * @param mlen Length of message.+ * @param[in] ctx Pointer to context string. May be NULL if ctxlen == 0.+ * @param ctxlen Length of context string.+ * @param[in] rnd Random seed.+ * @param[in] sk Bit-packed secret key; assumed to be valid.+ * @param context Application context. Only present when+ * MLD_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLD_ERR_OUT_OF_MEMORY MLD_CONFIG_CUSTOM_ALLOC_FREE was+ * used and an allocation via+ * MLD_CUSTOM_ALLOC returned NULL.+ * @retval MLD_ERR_SIGN_ATTEMPTS_EXHAUSTED The rejection-sampling loop exceeded+ * MLD_CONFIG_MAX_SIGNING_ATTEMPTS+ * iterations.+ * @retval MLD_ERR_SIGNING_PAUSED A MLD_CONFIG_SIGN_HOOK_ATTEMPT hook+ * paused signing; re-invoke to resume.+ * @retval MLD_ERR_INVALID_ARG The context string exceeded 255+ * bytes.+ */+MLD_MUST_CHECK_RETURN_VALUE+MLD_EXTERNAL_API+int mld_sign_signature_pre_hash_shake256(+ uint8_t sig[MLDSA_CRYPTO_BYTES], const uint8_t *m, size_t mlen,+ const uint8_t *ctx, size_t ctxlen, const uint8_t rnd[MLDSA_RNDBYTES],+ const uint8_t sk[MLDSA_CRYPTO_SECRETKEYBYTES],+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+__contract__(+ requires(mlen <= MLD_MAX_BUFFER_SIZE)+ requires(ctxlen <= MLD_MAX_BUFFER_SIZE)+ requires(memory_no_alias(sig, MLDSA_CRYPTO_BYTES))+ requires(memory_no_alias(m, mlen))+ requires(ctxlen == 0 || memory_no_alias(ctx, ctxlen))+ requires(memory_no_alias(rnd, MLDSA_RNDBYTES))+ requires(memory_no_alias(sk, MLDSA_CRYPTO_SECRETKEYBYTES))+ assigns(memory_slice(sig, MLDSA_CRYPTO_BYTES))+ ensures(return_value == 0 || return_value == MLD_ERR_OUT_OF_MEMORY || return_value == MLD_ERR_SIGN_ATTEMPTS_EXHAUSTED || return_value == MLD_ERR_SIGNING_PAUSED || return_value == MLD_ERR_INVALID_ARG)+ /* Output buffers on error, per API-CONVENTIONS.md */+ ensures(return_value != 0 ==>+ array_unchanged_or_zeroized_u8(sig, MLDSA_CRYPTO_BYTES))+);+#endif /* !MLD_CONFIG_NO_SIGN_API */++#if !defined(MLD_CONFIG_NO_VERIFY_API)+/**+ * Verify signature with pre-hashed message using SHAKE256. This function+ * computes the SHAKE256 hash of the message internally.+ *+ * @spec{Implements @[FIPS204, Algorithm 5, HashML-DSA.Verify] with SHAKE256 as+ * the pre-hash.}+ *+ * @param[in] sig Pointer to input signature.+ * @param[in] m Pointer to message to be hashed and verified.+ * @param mlen Length of message.+ * @param[in] ctx Pointer to context string. May be NULL if ctxlen == 0.+ * @param ctxlen Length of context string.+ * @param[in] pk Bit-packed public key.+ * @param context Application context. Only present when+ * MLD_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLD_ERR_OUT_OF_MEMORY MLD_CONFIG_CUSTOM_ALLOC_FREE was+ * used and an allocation via+ * MLD_CUSTOM_ALLOC returned NULL.+ * @retval MLD_ERR_INVALID_SIGNATURE Signature verification failed.+ * @retval MLD_ERR_INVALID_ARG The context string exceeded 255 bytes.+ */+MLD_MUST_CHECK_RETURN_VALUE+MLD_EXTERNAL_API+int mld_sign_verify_pre_hash_shake256(+ const uint8_t sig[MLDSA_CRYPTO_BYTES], const uint8_t *m, size_t mlen,+ const uint8_t *ctx, size_t ctxlen,+ const uint8_t pk[MLDSA_CRYPTO_PUBLICKEYBYTES],+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+__contract__(+ requires(mlen <= MLD_MAX_BUFFER_SIZE)+ requires(ctxlen <= MLD_MAX_BUFFER_SIZE - 77)+ requires(memory_no_alias(sig, MLDSA_CRYPTO_BYTES))+ requires(memory_no_alias(m, mlen))+ requires(ctxlen == 0 || memory_no_alias(ctx, ctxlen))+ requires(memory_no_alias(pk, MLDSA_CRYPTO_PUBLICKEYBYTES))+ ensures(return_value == 0 || return_value == MLD_ERR_INVALID_SIGNATURE || return_value == MLD_ERR_INVALID_ARG || return_value == MLD_ERR_OUT_OF_MEMORY)+);+#endif /* !MLD_CONFIG_NO_VERIFY_API */++#if !defined(MLD_CONFIG_NO_SIGN_API) || !defined(MLD_CONFIG_NO_VERIFY_API)+/* Maximum formatted domain separation message length:+ * - Pure ML-DSA: 0x00 || ctxlen || ctx (max 255)+ * - HashML-DSA: 0x01 || ctxlen || ctx (max 255) || oid (11) || ph (max 64) */+#define MLD_DOMAIN_SEPARATION_MAX_BYTES (2 + 255 + 11 + 64)++/**+ * Prepare domain separation prefix for ML-DSA signing.+ *+ * For pure ML-DSA (hashalg == MLD_PREHASH_NONE):+ * Format: 0x00 || ctxlen (1 byte) || ctx.+ *+ * For HashML-DSA (hashalg != MLD_PREHASH_NONE):+ * Format: 0x01 || ctxlen (1 byte) || ctx || oid (11 bytes) || ph.+ *+ * This function is useful for building incremental signing APIs.+ *+ * @spec{For HashML-DSA (hashalg != MLD_PREHASH_NONE), implements+ * @[FIPS204, Algorithm 4, line 23]. For Pure ML-DSA+ * (hashalg == MLD_PREHASH_NONE), implements+ * ```+ * M' <- BytesToBits(IntegerToBytes(0, 1)+ * || IntegerToBytes(|ctx|, 1)+ * || ctx+ * ```+ * which is part of @[FIPS204, Algorithm 2, ML-DSA.Sign, line 10] and+ * @[FIPS204, Algorithm 3, ML-DSA.Verify, line 5].}+ *+ * @param[out] prefix Output domain separation prefix buffer.+ * @param[in] ph Pointer to pre-hashed message (ignored for pure+ * ML-DSA).+ * @param phlen Length of pre-hashed message; must match the output+ * length of hashalg (ignored for pure ML-DSA).+ * @param[in] ctx Pointer to context string. May be NULL if ctxlen == 0.+ * @param ctxlen Length of context string.+ * @param hashalg Hash algorithm constant (MLD_PREHASH_NONE for pure+ * ML-DSA, or MLD_PREHASH_* for HashML-DSA).+ *+ * @return The total length of the formatted prefix, or 0 on error.+ * Errors are:+ * - The context string exceeded 255 bytes.+ * - For HashML-DSA: hashalg was unsupported, ph was NULL, or phlen+ * did not match the output length of hashalg.+ */+MLD_MUST_CHECK_RETURN_VALUE+MLD_EXTERNAL_API+size_t mld_prepare_domain_separation_prefix(+ uint8_t prefix[MLD_DOMAIN_SEPARATION_MAX_BYTES], const uint8_t *ph,+ size_t phlen, const uint8_t *ctx, size_t ctxlen, int hashalg)+__contract__(+ requires(ctxlen <= 255)+ requires(phlen <= MLD_MAX_BUFFER_SIZE)+ requires(ctxlen == 0 || memory_no_alias(ctx, ctxlen))+ requires(hashalg == MLD_PREHASH_NONE || memory_no_alias(ph, phlen))+ requires(memory_no_alias(prefix, MLD_DOMAIN_SEPARATION_MAX_BYTES))+ assigns(memory_slice(prefix, MLD_DOMAIN_SEPARATION_MAX_BYTES))+ ensures(return_value <= MLD_DOMAIN_SEPARATION_MAX_BYTES)+);+#endif /* !MLD_CONFIG_NO_SIGN_API || !MLD_CONFIG_NO_VERIFY_API */++#if !defined(MLD_CONFIG_NO_KEYPAIR_API)+/**+ * Perform basic validity checks on secret key, and derive public key.+ *+ * Referring to the decoding of the secret key `sk=(rho, K, tr, s1, s2, t0)`+ * (cf. @[FIPS204, Algorithm 25, skDecode]), the following checks are+ * performed:+ * - Check that s1 and s2 have coefficients in [-MLDSA_ETA, MLDSA_ETA].+ * - Check that t0 and tr stored in sk match recomputed values.+ *+ * @note This function leaks whether the secret key is valid or invalid+ * through its return value and timing.+ *+ * @param[out] pk Output public key.+ * @param[in] sk Input secret key.+ * @param context Application context. Only present when+ * MLD_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLD_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLD_ERR_OUT_OF_MEMORY MLD_CONFIG_CUSTOM_ALLOC_FREE was used and an+ * allocation via MLD_CUSTOM_ALLOC returned NULL.+ * @retval MLD_ERR_INVALID_KEY Secret key validation failed.+ */+MLD_MUST_CHECK_RETURN_VALUE+MLD_EXTERNAL_API+int mld_sign_pk_from_sk(uint8_t pk[MLDSA_CRYPTO_PUBLICKEYBYTES],+ const uint8_t sk[MLDSA_CRYPTO_SECRETKEYBYTES],+ MLD_CONFIG_CONTEXT_PARAMETER_TYPE context)+__contract__(+ requires(memory_no_alias(pk, MLDSA_CRYPTO_PUBLICKEYBYTES))+ requires(memory_no_alias(sk, MLDSA_CRYPTO_SECRETKEYBYTES))+ assigns(memory_slice(pk, MLDSA_CRYPTO_PUBLICKEYBYTES))+ ensures(return_value == 0 || return_value == MLD_ERR_INVALID_KEY || return_value == MLD_ERR_OUT_OF_MEMORY)+ /* Output buffers on error, per API-CONVENTIONS.md */+ ensures(return_value != 0 ==>+ array_unchanged_or_zeroized_u8(pk, MLDSA_CRYPTO_PUBLICKEYBYTES))+);+#endif /* !MLD_CONFIG_NO_KEYPAIR_API */+#endif /* !MLD_CONFIG_CORE_API_ONLY */++#endif /* !MLD_SIGN_H */
@@ -0,0 +1,68 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+#ifndef MLD_SYMMETRIC_H+#define MLD_SYMMETRIC_H++#include "cbmc.h"+#include "common.h"++#include MLD_FIPS202_HEADER_FILE+#if !defined(MLD_CONFIG_SERIAL_FIPS202_ONLY)+#include MLD_FIPS202X4_HEADER_FILE+#endif++#define MLD_STREAM128_BLOCKBYTES SHAKE128_RATE+#define MLD_STREAM256_BLOCKBYTES SHAKE256_RATE++#define mld_xof256_ctx mld_shake256ctx+#define mld_xof256_init(CTX) mld_shake256_init(CTX)++#define mld_xof256_absorb_once(CTX, IN, INBYTES) \+ do \+ { \+ mld_shake256_absorb(CTX, IN, INBYTES); \+ mld_shake256_finalize(CTX); \+ } while (0)+++#define mld_xof256_release(CTX) mld_shake256_release(CTX)+#define mld_xof256_squeezeblocks(OUT, OUTBLOCKS, STATE) \+ mld_shake256_squeeze(OUT, (OUTBLOCKS) * SHAKE256_RATE, STATE)++#define mld_xof128_ctx mld_shake128ctx+#define mld_xof128_init(CTX) mld_shake128_init(CTX)++#define mld_xof128_absorb_once(CTX, IN, INBYTES) \+ do \+ { \+ mld_shake128_absorb(CTX, IN, INBYTES); \+ mld_shake128_finalize(CTX); \+ } while (0)++#define mld_xof128_release(CTX) mld_shake128_release(CTX)+#define mld_xof128_squeezeblocks(OUT, OUTBLOCKS, STATE) \+ mld_shake128_squeeze(OUT, (OUTBLOCKS) * SHAKE128_RATE, STATE)++#define mld_xof256_x4_ctx mld_shake256x4ctx+#define mld_xof256_x4_init(CTX) mld_shake256x4_init((CTX))+#define mld_xof256_x4_absorb(CTX, IN, INBYTES) \+ mld_shake256x4_absorb_once((CTX), (IN)[0], (IN)[1], (IN)[2], (IN)[3], \+ (INBYTES))+#define mld_xof256_x4_squeezeblocks(BUF, NBLOCKS, CTX) \+ mld_shake256x4_squeezeblocks((BUF)[0], (BUF)[1], (BUF)[2], (BUF)[3], \+ (NBLOCKS), (CTX))+#define mld_xof256_x4_release(CTX) mld_shake256x4_release((CTX))++#define mld_xof128_x4_ctx mld_shake128x4ctx+#define mld_xof128_x4_init(CTX) mld_shake128x4_init((CTX))+#define mld_xof128_x4_absorb(CTX, IN, INBYTES) \+ mld_shake128x4_absorb_once((CTX), (IN)[0], (IN)[1], (IN)[2], (IN)[3], \+ (INBYTES))+#define mld_xof128_x4_squeezeblocks(BUF, NBLOCKS, CTX) \+ mld_shake128x4_squeezeblocks((BUF)[0], (BUF)[1], (BUF)[2], (BUF)[3], \+ (NBLOCKS), (CTX))+#define mld_xof128_x4_release(CTX) mld_shake128x4_release((CTX))++#endif /* !MLD_SYMMETRIC_H */
@@ -0,0 +1,327 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#ifndef MLD_SYS_H+#define MLD_SYS_H++#if !defined(MLD_CONFIG_NO_ASM) && (defined(__GNUC__) || defined(__clang__))+#define MLD_HAVE_INLINE_ASM+#endif++/* Try to find endianness, if not forced through CFLAGS already */+#if !defined(MLD_SYS_LITTLE_ENDIAN) && !defined(MLD_SYS_BIG_ENDIAN)+#if defined(__BYTE_ORDER__)+#if __BYTE_ORDER__ == __ORDER_LITTLE_ENDIAN__+#define MLD_SYS_LITTLE_ENDIAN+#elif __BYTE_ORDER__ == __ORDER_BIG_ENDIAN__+#define MLD_SYS_BIG_ENDIAN+#else+#error "__BYTE_ORDER__ defined, but don't recognize value."+#endif+#endif /* __BYTE_ORDER__ */++/* MSVC does not define __BYTE_ORDER__. However, MSVC only supports+ * little endian x86, x86_64, and AArch64. It is, hence, safe to assume+ * little endian. */+#if defined(_MSC_VER) && (defined(_M_X64) || defined(_M_AMD64) || \+ defined(_M_IX86) || defined(_M_ARM64))+#define MLD_SYS_LITTLE_ENDIAN+#endif++#endif /* !MLD_SYS_LITTLE_ENDIAN && !MLD_SYS_BIG_ENDIAN */++/* Check if we're running on an AArch64 little endian system. _M_ARM64 is set by+ * MSVC. */+#if defined(__AARCH64EL__) || defined(_M_ARM64)+#define MLD_SYS_AARCH64+#endif++/* Check if the AArch64 compilation target supports NEON (Advanced SIMD).+ *+ * Some compilers also define __ARM_NEON__, but __ARM_NEON is the most reliable+ * signal. Specifically, clang on Apple appears to keep __ARM_NEON__ set even if+ * -march=armv8-a+nosimd is set.+ *+ * gcc 4.8 -- the first gcc version introducing Neon support -- sets neither+ * __ARM_NEON nor __ARM_NEON__; in fact, there is no preprocessor signal that+ * Neon is enabled. If you use gcc 4.8 and need Neon, you should set+ * MLD_SYS_AARCH64_NEON manually. gcc 4.9 onwards do set __ARM_NEON.+ */+#if defined(MLD_SYS_AARCH64) && defined(__ARM_NEON)+#define MLD_SYS_AARCH64_NEON+#endif++/* Check if we're running on an AArch64 big endian system. */+#if defined(__AARCH64EB__)+#define MLD_SYS_AARCH64_EB+#endif++/* Check if we're running on an Armv8.1-M system with MVE */+#if defined(__ARM_ARCH_8_1M_MAIN__) || defined(__ARM_FEATURE_MVE)+#define MLD_SYS_ARMV81M_MVE+#endif++/* Check if we're running on an x86_64 system. */+#if defined(__x86_64__) || defined(_M_X64) || defined(_M_AMD64)+#define MLD_SYS_X86_64+#if defined(__AVX2__)+#define MLD_SYS_X86_64_AVX2+#endif+#endif /* __x86_64__ || _M_X64 || _M_AMD64 */++#if defined(MLD_SYS_LITTLE_ENDIAN) && defined(__powerpc64__)+#define MLD_SYS_PPC64LE+#endif++#if defined(__riscv) && defined(__riscv_xlen) && __riscv_xlen == 64+#define MLD_SYS_RISCV64+#endif++#if defined(MLD_SYS_RISCV64) && defined(__riscv_vector) && \+ defined(__riscv_v_intrinsic)+#define MLD_SYS_RISCV64_RVV+#endif++#if defined(__riscv) && defined(__riscv_xlen) && __riscv_xlen == 32+#define MLD_SYS_RISCV32+#endif++#if defined(_WIN64) || defined(_WIN32)+#define MLD_SYS_WINDOWS+#endif++#if defined(__linux__)+#define MLD_SYS_LINUX+#endif++#if defined(__APPLE__)+#define MLD_SYS_APPLE+#endif++/* If MLD_FORCE_AARCH64 is set, assert that we're indeed on an AArch64 system.+ */+#if defined(MLD_FORCE_AARCH64) && !defined(MLD_SYS_AARCH64)+#error "MLD_FORCE_AARCH64 is set, but we don't seem to be on an AArch64 system."+#endif++/* If MLD_FORCE_AARCH64_EB is set, assert that we're indeed on a big endian+ * AArch64 system. */+#if defined(MLD_FORCE_AARCH64_EB) && !defined(MLD_SYS_AARCH64_EB)+#error \+ "MLD_FORCE_AARCH64_EB is set, but we don't seem to be on an AArch64 system."+#endif++/* If MLD_FORCE_X86_64 is set, assert that we're indeed on an X86_64 system. */+#if defined(MLD_FORCE_X86_64) && !defined(MLD_SYS_X86_64)+#error "MLD_FORCE_X86_64 is set, but we don't seem to be on an X86_64 system."+#endif++#if defined(MLD_FORCE_PPC64LE) && !defined(MLD_SYS_PPC64LE)+#error "MLD_FORCE_PPC64LE is set, but we don't seem to be on a PPC64LE system."+#endif++#if defined(MLD_FORCE_RISCV64) && !defined(MLD_SYS_RISCV64)+#error "MLD_FORCE_RISCV64 is set, but we don't seem to be on a RISCV64 system."+#endif++#if defined(MLD_FORCE_RISCV32) && !defined(MLD_SYS_RISCV32)+#error "MLD_FORCE_RISCV32 is set, but we don't seem to be on a RISCV32 system."+#endif++/*+ * MLD_INLINE: Hint for inlining.+ * - MSVC: __inline+ * - C99+: inline+ * - GCC/Clang C90: __attribute__((unused)) to silence warnings+ * - Other C90: empty+ */+#if !defined(MLD_INLINE)+#if defined(_MSC_VER)+#define MLD_INLINE __inline+#elif defined(inline) || \+ (defined(__STDC_VERSION__) && __STDC_VERSION__ >= 199901L)+#define MLD_INLINE inline+#elif defined(__GNUC__) || defined(__clang__)+#define MLD_INLINE __attribute__((unused))+#else+#define MLD_INLINE+#endif+#endif /* !MLD_INLINE */++/*+ * MLD_ALWAYS_INLINE: Force inlining.+ * - MSVC: __forceinline+ * - GCC/Clang C99+: MLD_INLINE __attribute__((always_inline))+ * - Other: MLD_INLINE (no forced inlining)+ */+#if !defined(MLD_ALWAYS_INLINE)+#if defined(_MSC_VER)+#define MLD_ALWAYS_INLINE __forceinline+#elif (defined(__GNUC__) || defined(__clang__)) && \+ (defined(inline) || \+ (defined(__STDC_VERSION__) && __STDC_VERSION__ >= 199901L))+#define MLD_ALWAYS_INLINE MLD_INLINE __attribute__((always_inline))+#else+#define MLD_ALWAYS_INLINE MLD_INLINE+#endif+#endif /* !MLD_ALWAYS_INLINE */++/*+ * MLD_NOINLINE: Prevent inlining.+ * - MSVC: __declspec(noinline)+ * - GCC/Clang: __attribute__((noinline))+ * - Other: empty+ */+#if !defined(MLD_NOINLINE)+#if defined(_MSC_VER)+#define MLD_NOINLINE __declspec(noinline)+#elif defined(__GNUC__) || defined(__clang__)+#define MLD_NOINLINE __attribute__((noinline))+#else+#define MLD_NOINLINE+#endif+#endif /* !MLD_NOINLINE */++#ifndef MLD_STATIC_TESTABLE+#define MLD_STATIC_TESTABLE static+#endif++/*+ * C90 does not have the restrict compiler directive yet.+ * We don't use it in C90 builds.+ */+#if !defined(restrict)+#if defined(__STDC_VERSION__) && __STDC_VERSION__ >= 199901L+#define MLD_RESTRICT restrict+#else+#define MLD_RESTRICT+#endif++#else /* !restrict */++#define MLD_RESTRICT restrict+#endif /* restrict */++#define MLD_DEFAULT_ALIGN 32+#define MLD_ALIGN_UP(N) \+ ((((N) + (MLD_DEFAULT_ALIGN - 1)) / MLD_DEFAULT_ALIGN) * MLD_DEFAULT_ALIGN)+#if defined(__GNUC__)+#define MLD_ALIGN __attribute__((aligned(MLD_DEFAULT_ALIGN)))+#elif defined(_MSC_VER)+#define MLD_ALIGN __declspec(align(MLD_DEFAULT_ALIGN))+#else+#define MLD_ALIGN /* No known support for alignment constraints */+#endif+++/* New X86_64 CPUs support control-flow protection using the CET instructions.+ * When enabled (through -fcf-protection=), all compilation units (including+ * empty ones) need to support CET for this to work.+ * For assembly, this means that source files need to signal support for+ * CET by setting the appropriate note.gnu.property section.+ * This can be achieved by including the <cet.h> header in all assembly file.+ * This file also provides the _CET_ENDBR macro which needs to be placed at+ * every potential target of an indirect branch.+ * If CET is enabled _CET_ENDBR maps to the endbr64 instruction, otherwise+ * it is empty.+ * In case the compiler does not support CET (e.g., <gcc8, <clang11),+ * the __CET__ macro is not set and we default to nothing.+ * Note that we only issue _CET_ENDBR instructions through the MLD_ASM_FN_SYMBOL+ * macro as the global symbols are the only possible targets of indirect+ * branches in our code.+ */+#if defined(MLD_SYS_X86_64)+#if defined(__CET__)+#include <cet.h>+#define MLD_CET_ENDBR _CET_ENDBR+#else+#define MLD_CET_ENDBR+#endif+#endif /* MLD_SYS_X86_64 */++#if defined(MLD_CONFIG_CT_TESTING_ENABLED) && !defined(__ASSEMBLER__)+#include <valgrind/memcheck.h>+#define MLD_CT_TESTING_SECRET(ptr, len) \+ VALGRIND_MAKE_MEM_UNDEFINED((ptr), (len))+#define MLD_CT_TESTING_DECLASSIFY(ptr, len) \+ VALGRIND_MAKE_MEM_DEFINED((ptr), (len))+#else /* MLD_CONFIG_CT_TESTING_ENABLED && !__ASSEMBLER__ */+#define MLD_CT_TESTING_SECRET(ptr, len) \+ do \+ { \+ } while (0)+#define MLD_CT_TESTING_DECLASSIFY(ptr, len) \+ do \+ { \+ } while (0)+#endif /* !(MLD_CONFIG_CT_TESTING_ENABLED && !__ASSEMBLER__) */++#if defined(__GNUC__) || defined(__clang__)+#define MLD_MUST_CHECK_RETURN_VALUE __attribute__((warn_unused_result))+#else+#define MLD_MUST_CHECK_RETURN_VALUE+#endif++/* The x86_64 assembly backend uses the SysV calling convention. On Windows,+ * where the Microsoft x64 calling convention is the default, it can still be+ * used with compilers that allow choosing the calling convention per+ * function: GCC and Clang support __attribute__((sysv_abi)), which makes+ * calls to the annotated function follow the SysV calling convention.+ *+ * MLD_SYSV_ABI_SUPPORTED signals that the toolchain can call SysV assembly+ * routines; the x86_64 assembly backend is only enabled if it is defined.+ * MLD_SYSV_ABI is the attribute carried by declarations of x86_64 assembly+ * routines. Both macros can be set externally for toolchains offering an+ * equivalent mechanism that is not recognized here. */+#if defined(MLD_SYS_X86_64) && !defined(MLD_SYSV_ABI_SUPPORTED)+#if !defined(MLD_SYS_WINDOWS) || defined(__GNUC__) || defined(__clang__)+#define MLD_SYSV_ABI_SUPPORTED+#endif+#endif++#if !defined(MLD_SYSV_ABI)+#if defined(MLD_SYS_WINDOWS) && defined(MLD_SYSV_ABI_SUPPORTED)+#define MLD_SYSV_ABI __attribute__((sysv_abi))+#else+#define MLD_SYSV_ABI+#endif+#endif /* !MLD_SYSV_ABI */++#if !defined(__ASSEMBLER__)+/* System capability enumeration */+typedef enum+{+ /* x86_64 */+ MLD_SYS_CAP_X86_64_AVX2,+ /* AArch64 */+ MLD_SYS_CAP_AARCH64_NEON,+ MLD_SYS_CAP_AARCH64_SHA3,+ /* Armv8.1-M */+ MLD_SYS_CAP_ARMV81M_MVE+} mld_sys_cap;++#if !defined(MLD_CONFIG_CUSTOM_CAPABILITY_FUNC)+#include "cbmc.h"++MLD_MUST_CHECK_RETURN_VALUE+static MLD_INLINE int mld_sys_check_capability(mld_sys_cap cap)+__contract__(+ ensures(return_value == 0 || return_value == 1)+)+{+ /* By default, we rely on compile-time feature detection/specification:+ * If a feature is enabled at compile-time, we assume it is supported by+ * the host that the resulting library/binary will be built on.+ * If this assumption is not true, you MUST overwrite this function.+ * See the documentation of MLD_CONFIG_CUSTOM_CAPABILITY_FUNC in+ * mldsa_native_config.h for more information. */+ (void)cap;+ return 1;+}+#endif /* !MLD_CONFIG_CUSTOM_CAPABILITY_FUNC */+#endif /* !__ASSEMBLER__ */++#endif /* !MLD_SYS_H */
@@ -0,0 +1,55 @@+/*+ * Copyright (c) The mldsa-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*+ * WARNING: This file is auto-generated from scripts/autogen+ * in the mldsa-native repository.+ * Do not modify it directly.+ */+++/*+ * Table of zeta values used in the reference NTT and inverse NTT.+ * See autogen for details.+ */+static const int32_t mld_zetas[MLDSA_N] = {+ 0, 25847, -2608894, -518909, 237124, -777960, -876248,+ 466468, 1826347, 2353451, -359251, -2091905, 3119733, -2884855,+ 3111497, 2680103, 2725464, 1024112, -1079900, 3585928, -549488,+ -1119584, 2619752, -2108549, -2118186, -3859737, -1399561, -3277672,+ 1757237, -19422, 4010497, 280005, 2706023, 95776, 3077325,+ 3530437, -1661693, -3592148, -2537516, 3915439, -3861115, -3043716,+ 3574422, -2867647, 3539968, -300467, 2348700, -539299, -1699267,+ -1643818, 3505694, -3821735, 3507263, -2140649, -1600420, 3699596,+ 811944, 531354, 954230, 3881043, 3900724, -2556880, 2071892,+ -2797779, -3930395, -1528703, -3677745, -3041255, -1452451, 3475950,+ 2176455, -1585221, -1257611, 1939314, -4083598, -1000202, -3190144,+ -3157330, -3632928, 126922, 3412210, -983419, 2147896, 2715295,+ -2967645, -3693493, -411027, -2477047, -671102, -1228525, -22981,+ -1308169, -381987, 1349076, 1852771, -1430430, -3343383, 264944,+ 508951, 3097992, 44288, -1100098, 904516, 3958618, -3724342,+ -8578, 1653064, -3249728, 2389356, -210977, 759969, -1316856,+ 189548, -3553272, 3159746, -1851402, -2409325, -177440, 1315589,+ 1341330, 1285669, -1584928, -812732, -1439742, -3019102, -3881060,+ -3628969, 3839961, 2091667, 3407706, 2316500, 3817976, -3342478,+ 2244091, -2446433, -3562462, 266997, 2434439, -1235728, 3513181,+ -3520352, -3759364, -1197226, -3193378, 900702, 1859098, 909542,+ 819034, 495491, -1613174, -43260, -522500, -655327, -3122442,+ 2031748, 3207046, -3556995, -525098, -768622, -3595838, 342297,+ 286988, -2437823, 4108315, 3437287, -3342277, 1735879, 203044,+ 2842341, 2691481, -2590150, 1265009, 4055324, 1247620, 2486353,+ 1595974, -3767016, 1250494, 2635921, -3548272, -2994039, 1869119,+ 1903435, -1050970, -1333058, 1237275, -3318210, -1430225, -451100,+ 1312455, 3306115, -1962642, -1279661, 1917081, -2546312, -1374803,+ 1500165, 777191, 2235880, 3406031, -542412, -2831860, -1671176,+ -1846953, -2584293, -3724270, 594136, -3776993, -2013608, 2432395,+ 2454455, -164721, 1957272, 3369112, 185531, -1207385, -3183426,+ 162844, 1616392, 3014001, 810149, 1652634, -3694233, -1799107,+ -3038916, 3523897, 3866901, 269760, 2213111, -975884, 1717735,+ 472078, -426683, 1723600, -1803090, 1910376, -1667432, -1104333,+ -260646, -3833893, -2939036, -2235985, -420899, -2286327, 183443,+ -976891, 1612842, -3545687, -554416, 3919660, -48306, -1362209,+ 3937738, 1400424, -846154, 1976782,+};
@@ -0,0 +1,2 @@+d1b2fe782888bdb761a50336012923180be7f502+v2.0.0
@@ -0,0 +1,312 @@+mlkem-native is a fork of the public domain Kyber reference+implementation, available on https://github.com/pq-crystals/kyber.++All source code in mlkem/* and dev/* is licensed under your choice+of the Apache-2.0 license OR the ISC license OR the MIT license.+These licenses are reproduced at the bottom of this file.+The copyright holders are indicated at the top of each file.++Files outside the library itself may carry different terms. Every file+states its own SPDX-License-Identifier, which determines the terms that+apply to it. In particular:++The code in test/notrandombytes/*, and its copies in+examples/*/test_only_rng/* and scripts/notrandombytes, is derived from+https://cr.yp.to/papers.html#surf and licensed under+LicenseRef-PD-hp OR CC0-1.0 OR 0BSD OR MIT-0 OR MIT.+It is only used for testing purposes.++The benchmarking code in test/hal/* carries the+MIT license. It is only used for testing purposes.++The tiny_sha3 code in+examples/bring_your_own_fips202/custom_fips202/tiny_sha3/* and+examples/custom_backend/mlkem_native/src/fips202/native/custom/src/*+carries the MIT license. It is only used to demonstrate custom+FIPS-202 implementations.++The proofs in proofs/* are in part derived from Amazon Web Services+verification infrastructure. Individual files carry, among others,+Apache-2.0 OR ISC OR MIT-0, MIT-0, and MIT-0 AND Apache-2.0. None of the+proofs are part of the library.++Documentation is licensed under CC-BY-4.0.++```+Copyright (c) The mlkem-native project authors+Copyright (c) 2020 Dougall Johnson+Copyright (c) 2022 Arm Limited+SPDX-License-Identifier: MIT++Permission is hereby granted, free of charge, to any person obtaining a copy+of this software and associated documentation files (the "Software"), to deal+in the Software without restriction, including without limitation the rights+to use, copy, modify, merge, publish, distribute, sublicense, and/or sell+copies of the Software, and to permit persons to whom the Software is+furnished to do so, subject to the following conditions:++The above copyright notice and this permission notice shall be included in+all copies or substantial portions of the Software.++THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR+IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,+FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE+AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER+LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,+OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE+SOFTWARE.+```++The tiny_sha3 implementation in examples/custom_backend/mlkem_native/src/fips202/native/custom/src/+carries the MIT license. It is only used for testing purposes.++```+SPDX-License-Identifier: MIT++19-Nov-11 Markku-Juhani O. Saarinen <mjos@iki.fi>+```++ISC license+-----------++Copyright <YEAR> <OWNER>++Permission to use, copy, modify, and/or distribute this software for any purpose+with or without fee is hereby granted, provided that the above copyright notice+and this permission notice appear in all copies.++THE SOFTWARE IS PROVIDED "AS IS" AND THE AUTHOR DISCLAIMS ALL WARRANTIES WITH+REGARD TO THIS SOFTWARE INCLUDING ALL IMPLIED WARRANTIES OF MERCHANTABILITY AND+FITNESS. IN NO EVENT SHALL THE AUTHOR BE LIABLE FOR ANY SPECIAL, DIRECT,+INDIRECT, OR CONSEQUENTIAL DAMAGES OR ANY DAMAGES WHATSOEVER RESULTING FROM LOSS+OF USE, DATA OR PROFITS, WHETHER IN AN ACTION OF CONTRACT, NEGLIGENCE OR OTHER+TORTIOUS ACTION, ARISING OUT OF OR IN CONNECTION WITH THE USE OR PERFORMANCE OF+THIS SOFTWARE.+++MIT license+-----------++Copyright <YEAR> <COPYRIGHT HOLDER>++Permission is hereby granted, free of charge, to any person obtaining a copy of+this software and associated documentation files (the “Software”), to deal in+the Software without restriction, including without limitation the rights to+use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of+the Software, and to permit persons to whom the Software is furnished to do so,+subject to the following conditions:++The above copyright notice and this permission notice shall be included in all+copies or substantial portions of the Software.++THE SOFTWARE IS PROVIDED “AS IS”, WITHOUT WARRANTY OF ANY KIND, EXPRESS OR+IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS+FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR+COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER+IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN+CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.++Apache-2.0 license+------------------++ Apache License+ Version 2.0, January 2004+ http://www.apache.org/licenses/++ TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION++ 1. Definitions.++ "License" shall mean the terms and conditions for use, reproduction,+ and distribution as defined by Sections 1 through 9 of this document.++ "Licensor" shall mean the copyright owner or entity authorized by+ the copyright owner that is granting the License.++ "Legal Entity" shall mean the union of the acting entity and all+ other entities that control, are controlled by, or are under common+ control with that entity. For the purposes of this definition,+ "control" means (i) the power, direct or indirect, to cause the+ direction or management of such entity, whether by contract or+ otherwise, or (ii) ownership of fifty percent (50%) or more of the+ outstanding shares, or (iii) beneficial ownership of such entity.++ "You" (or "Your") shall mean an individual or Legal Entity+ exercising permissions granted by this License.++ "Source" form shall mean the preferred form for making modifications,+ including but not limited to software source code, documentation+ source, and configuration files.++ "Object" form shall mean any form resulting from mechanical+ transformation or translation of a Source form, including but+ not limited to compiled object code, generated documentation,+ and conversions to other media types.++ "Work" shall mean the work of authorship, whether in Source or+ Object form, made available under the License, as indicated by a+ copyright notice that is included in or attached to the work+ (an example is provided in the Appendix below).++ "Derivative Works" shall mean any work, whether in Source or Object+ form, that is based on (or derived from) the Work and for which the+ editorial revisions, annotations, elaborations, or other modifications+ represent, as a whole, an original work of authorship. For the purposes+ of this License, Derivative Works shall not include works that remain+ separable from, or merely link (or bind by name) to the interfaces of,+ the Work and Derivative Works thereof.++ "Contribution" shall mean any work of authorship, including+ the original version of the Work and any modifications or additions+ to that Work or Derivative Works thereof, that is intentionally+ submitted to Licensor for inclusion in the Work by the copyright owner+ or by an individual or Legal Entity authorized to submit on behalf of+ the copyright owner. For the purposes of this definition, "submitted"+ means any form of electronic, verbal, or written communication sent+ to the Licensor or its representatives, including but not limited to+ communication on electronic mailing lists, source code control systems,+ and issue tracking systems that are managed by, or on behalf of, the+ Licensor for the purpose of discussing and improving the Work, but+ excluding communication that is conspicuously marked or otherwise+ designated in writing by the copyright owner as "Not a Contribution."++ "Contributor" shall mean Licensor and any individual or Legal Entity+ on behalf of whom a Contribution has been received by Licensor and+ subsequently incorporated within the Work.++ 2. Grant of Copyright License. Subject to the terms and conditions of+ this License, each Contributor hereby grants to You a perpetual,+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable+ copyright license to reproduce, prepare Derivative Works of,+ publicly display, publicly perform, sublicense, and distribute the+ Work and such Derivative Works in Source or Object form.++ 3. Grant of Patent License. Subject to the terms and conditions of+ this License, each Contributor hereby grants to You a perpetual,+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable+ (except as stated in this section) patent license to make, have made,+ use, offer to sell, sell, import, and otherwise transfer the Work,+ where such license applies only to those patent claims licensable+ by such Contributor that are necessarily infringed by their+ Contribution(s) alone or by combination of their Contribution(s)+ with the Work to which such Contribution(s) was submitted. If You+ institute patent litigation against any entity (including a+ cross-claim or counterclaim in a lawsuit) alleging that the Work+ or a Contribution incorporated within the Work constitutes direct+ or contributory patent infringement, then any patent licenses+ granted to You under this License for that Work shall terminate+ as of the date such litigation is filed.++ 4. Redistribution. You may reproduce and distribute copies of the+ Work or Derivative Works thereof in any medium, with or without+ modifications, and in Source or Object form, provided that You+ meet the following conditions:++ (a) You must give any other recipients of the Work or+ Derivative Works a copy of this License; and++ (b) You must cause any modified files to carry prominent notices+ stating that You changed the files; and++ (c) You must retain, in the Source form of any Derivative Works+ that You distribute, all copyright, patent, trademark, and+ attribution notices from the Source form of the Work,+ excluding those notices that do not pertain to any part of+ the Derivative Works; and++ (d) If the Work includes a "NOTICE" text file as part of its+ distribution, then any Derivative Works that You distribute must+ include a readable copy of the attribution notices contained+ within such NOTICE file, excluding those notices that do not+ pertain to any part of the Derivative Works, in at least one+ of the following places: within a NOTICE text file distributed+ as part of the Derivative Works; within the Source form or+ documentation, if provided along with the Derivative Works; or,+ within a display generated by the Derivative Works, if and+ wherever such third-party notices normally appear. The contents+ of the NOTICE file are for informational purposes only and+ do not modify the License. You may add Your own attribution+ notices within Derivative Works that You distribute, alongside+ or as an addendum to the NOTICE text from the Work, provided+ that such additional attribution notices cannot be construed+ as modifying the License.++ You may add Your own copyright statement to Your modifications and+ may provide additional or different license terms and conditions+ for use, reproduction, or distribution of Your modifications, or+ for any such Derivative Works as a whole, provided Your use,+ reproduction, and distribution of the Work otherwise complies with+ the conditions stated in this License.++ 5. Submission of Contributions. Unless You explicitly state otherwise,+ any Contribution intentionally submitted for inclusion in the Work+ by You to the Licensor shall be under the terms and conditions of+ this License, without any additional terms or conditions.+ Notwithstanding the above, nothing herein shall supersede or modify+ the terms of any separate license agreement you may have executed+ with Licensor regarding such Contributions.++ 6. Trademarks. This License does not grant permission to use the trade+ names, trademarks, service marks, or product names of the Licensor,+ except as required for reasonable and customary use in describing the+ origin of the Work and reproducing the content of the NOTICE file.++ 7. Disclaimer of Warranty. Unless required by applicable law or+ agreed to in writing, Licensor provides the Work (and each+ Contributor provides its Contributions) on an "AS IS" BASIS,+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or+ implied, including, without limitation, any warranties or conditions+ of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A+ PARTICULAR PURPOSE. You are solely responsible for determining the+ appropriateness of using or redistributing the Work and assume any+ risks associated with Your exercise of permissions under this License.++ 8. Limitation of Liability. In no event and under no legal theory,+ whether in tort (including negligence), contract, or otherwise,+ unless required by applicable law (such as deliberate and grossly+ negligent acts) or agreed to in writing, shall any Contributor be+ liable to You for damages, including any direct, indirect, special,+ incidental, or consequential damages of any character arising as a+ result of this License or out of the use or inability to use the+ Work (including but not limited to damages for loss of goodwill,+ work stoppage, computer failure or malfunction, or any and all+ other commercial damages or losses), even if such Contributor+ has been advised of the possibility of such damages.++ 9. Accepting Warranty or Additional Liability. While redistributing+ the Work or Derivative Works thereof, You may choose to offer,+ and charge a fee for, acceptance of support, warranty, indemnity,+ or other liability obligations and/or rights consistent with this+ License. However, in accepting such obligations, You may act only+ on Your own behalf and on Your sole responsibility, not on behalf+ of any other Contributor, and only if You agree to indemnify,+ defend, and hold each Contributor harmless for any liability+ incurred by, or claims asserted against, such Contributor by reason+ of your accepting any such warranty or additional liability.++ END OF TERMS AND CONDITIONS++ APPENDIX: How to apply the Apache License to your work.++ To apply the Apache License to your work, attach the following+ boilerplate notice, with the fields enclosed by brackets "[]"+ replaced with your own identifying information. (Don't include+ the brackets!) The text should be enclosed in the appropriate+ comment syntax for the file format. We also recommend that a+ file or class name and description of purpose be included on the+ same "printed page" as the copyright notice for easier+ identification within third-party archives.++ Copyright [yyyy] [name of copyright owner]++ Licensed under the Apache License, Version 2.0 (the "License");+ you may not use this file except in compliance with the License.+ You may obtain a copy of the License at++ http://www.apache.org/licenses/LICENSE-2.0++ Unless required by applicable law or agreed to in writing, software+ distributed under the License is distributed on an "AS IS" BASIS,+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.+ See the License for the specific language governing permissions and+ limitations under the License.
@@ -0,0 +1,59 @@+# mlkem-native++ML-KEM (FIPS 203) from the PQ Code Package, vendored here and built for all+three parameter sets.++## Where it comes from++<https://github.com/pq-code-package/mlkem-native>. `COMMIT` holds the+revision this tree is at; `import.sh` puts it there and is how the tree is+refreshed. Nothing here is edited by hand -- every choice crypton makes is+made in `crypton_mlkem.h`, `crypton_mlkem.c` and `crypton.cabal`, so that a+re-import is a straight overwrite.++## Licence++`Apache-2.0 OR ISC OR MIT`, the same three-way form as `cbits/s2n` and by+some of the same authors. crypton takes it under ISC, which is already in+the package's `license:` field. `LICENSE` is the upstream file and is+listed in `license-files:`.++## What is taken and what is not++The whole of `mlkem/`, less the backends for architectures crypton does not+build for: `src/native/riscv64`, `src/native/ppc64le` and the 32-bit+`src/fips202/native/armv81m`. Every reference to those is behind an+`MLK_SYS_` guard that cannot be true on the architectures crypton does+build for, so dropping them changes no build and keeps a few dozen files of+unreachable assembly out of the release tarball. To take one back, delete+its line from `import.sh` and add its directory to `extra-source-files:`.++## How it is built++Upstream builds for one parameter set at a time. `crypton_mlkem.c`+includes the amalgamation once per set -- the level-independent half kept by+exactly one of them -- which is how a single crypton offers ML-KEM-512, 768+and 1024. `crypton_mlkem_asm.S` does the same for the assembly, which is+level-independent and so is included once.++This is the shape `cbits/aes/armv8.c` already uses for the three AES key+sizes, and it has the same hazard: **cabal does not know that the wrapper+depends on the tree it includes.** After changing anything under+`cbits/mlkem`, touch `crypton_mlkem.c`, or the build keeps the object it+already has and the change is not tested.++The hand-written backends are selected by `CRYPTON_MLKEM_NATIVE_BACKEND`,+which `crypton.cabal` defines on x86-64 and AArch64 other than Windows --+the same exclusion, and for the same reasons, as `cbits/s2n`. Everywhere+else the portable C is built, which is the same code and passes the same+tests.++There is no randomised API: `MLK_CONFIG_NO_RANDOMIZED_API` is set, no+`randombytes()` is needed, and randomness is drawn in Haskell through+`MonadRandom` as it is for every other key crypton generates.++The symbols are `crypton_mlkem512_*`, `crypton_mlkem768_*` and+`crypton_mlkem1024_*` rather than upstream's defaults, so that an+application linking another copy of mlkem-native -- through some other+library, or its own -- does not present the linker with two sets of+functions answering to one set of names.
@@ -0,0 +1,31 @@+/*+ * All three ML-KEM parameter sets in one translation unit.+ *+ * mlkem-native is built for one parameter set at a time; a build wanting+ * several includes the amalgamation once per set, with the level-independent+ * half kept by exactly one of them. This is the same shape as+ * cbits/aes/armv8.c, which includes cbits/aes/armv8_impl.c three times for+ * the three AES key sizes.+ *+ * NOTE, as there: cabal does not know that this file depends on the tree it+ * includes. After changing anything under cbits/mlkem, touch this file, or+ * the build will quietly keep the object it already has.+ */+#include "crypton_mlkem.h"++#define MLK_CONFIG_MULTILEVEL_WITH_SHARED 1+#define MLK_CONFIG_MONOBUILD_KEEP_SHARED_HEADERS+#define MLK_CONFIG_PARAMETER_SET 512+#include "mlkem_native.c"+#undef MLK_CONFIG_MULTILEVEL_WITH_SHARED+#undef MLK_CONFIG_PARAMETER_SET++#define MLK_CONFIG_MULTILEVEL_NO_SHARED+#define MLK_CONFIG_PARAMETER_SET 768+#include "mlkem_native.c"+#undef MLK_CONFIG_MONOBUILD_KEEP_SHARED_HEADERS+#undef MLK_CONFIG_PARAMETER_SET++#define MLK_CONFIG_PARAMETER_SET 1024+#include "mlkem_native.c"+#undef MLK_CONFIG_PARAMETER_SET
@@ -0,0 +1,41 @@+/*+ * What crypton asks of the vendored mlkem-native, in one place. The tree+ * under cbits/mlkem is upstream's and is overwritten by import.sh, so every+ * choice crypton makes is made here instead of by editing it.+ *+ * Included first by both crypton_mlkem.c and crypton_mlkem_asm.S, so it must+ * hold nothing but preprocessor directives.+ */+#ifndef CRYPTON_MLKEM_H+#define CRYPTON_MLKEM_H++/*+ * The symbols are crypton's own, not the default PQCP_MLKEM_NATIVE_*. An+ * application is free to link another copy of mlkem-native -- through some+ * other library, or its own -- and two copies answering to one set of names+ * is a problem the linker resolves silently and in nobody's favour. With+ * MLK_CONFIG_MULTILEVEL_BUILD the level is appended, so the entry points+ * are crypton_mlkem512_*, crypton_mlkem768_* and crypton_mlkem1024_*.+ */+#define MLK_CONFIG_NAMESPACE_PREFIX crypton_mlkem+#define MLK_CONFIG_MULTILEVEL_BUILD++/*+ * No randomised API, so no randombytes() to provide. Randomness is drawn+ * in Haskell through MonadRandom, the way every other key in crypton is+ * generated, and the deterministic entry points are what the FFI calls.+ * That also keeps the C free of any opinion about where entropy comes from.+ */+#define MLK_CONFIG_NO_RANDOMIZED_API++/*+ * The hand-written backends, where crypton.cabal says the architecture has+ * them. Without this the portable C is built, which is correct everywhere+ * and is what every other architecture gets.+ */+#ifdef CRYPTON_MLKEM_NATIVE_BACKEND+#define MLK_CONFIG_USE_NATIVE_BACKEND_ARITH+#define MLK_CONFIG_USE_NATIVE_BACKEND_FIPS202+#endif++#endif /* CRYPTON_MLKEM_H */
@@ -0,0 +1,14 @@+/*+ * The assembly half, which is level-independent: it is included once, with+ * the shared directives kept, and covers all three parameter sets. See the+ * comment at the top of mlkem_native_asm.S.+ *+ * Built only where crypton.cabal defines CRYPTON_MLKEM_NATIVE_BACKEND; on+ * every other architecture this file is not in asm-sources at all.+ */+#include "crypton_mlkem.h"++#define MLK_CONFIG_MULTILEVEL_WITH_SHARED 1+#define MLK_CONFIG_MONOBUILD_KEEP_SHARED_HEADERS+#define MLK_CONFIG_PARAMETER_SET 768+#include "mlkem_native_asm.S"
@@ -0,0 +1,42 @@+#!/bin/sh+# Re-import the vendored parts of the PQ Code Package's mlkem-native.+#+# The files are kept unmodified. Everything crypton decides -- which+# parameter sets exist, what the symbols are called, that there is no+# randomised API -- is decided in cbits/mlkem/crypton_mlkem.c and in+# crypton.cabal, not by editing anything here. Run this from cbits/mlkem:+#+# ./import.sh [tag-or-commit]+#+# and commit the result together with the COMMIT line it writes, so that the+# tree always says which upstream revision it holds.+set -eu++REPO=https://github.com/pq-code-package/mlkem-native+REV=${1:-v2.0.0}+HERE=$(cd "$(dirname "$0")" && pwd)+TMP=$(mktemp -d)+trap 'rm -rf "$TMP"' EXIT++git clone -q "$REPO" "$TMP/u"+git -C "$TMP/u" checkout -q "$REV"++rm -rf "$HERE/src"+cp "$TMP/u/mlkem/mlkem_native.c" "$HERE/"+cp "$TMP/u/mlkem/mlkem_native.h" "$HERE/"+cp "$TMP/u/mlkem/mlkem_native_asm.S" "$HERE/"+cp "$TMP/u/mlkem/mlkem_native_config.h" "$HERE/"+cp -R "$TMP/u/mlkem/src" "$HERE/src"+cp "$TMP/u/LICENSE" "$HERE/LICENSE"++# The backends for architectures crypton does not build for. Every+# reference to them is behind an MLK_SYS_ guard that cannot be true on the+# two it does, so dropping them changes no build and keeps a few dozen files+# of unreachable assembly out of the release tarball. If crypton ever+# wants one of them, delete its line here rather than patching anything.+rm -rf "$HERE/src/native/riscv64" "$HERE/src/native/ppc64le"+rm -rf "$HERE/src/fips202/native/armv81m"++git -C "$TMP/u" rev-parse HEAD > "$HERE/COMMIT"+git -C "$TMP/u" describe --tags --exact-match 2>/dev/null >> "$HERE/COMMIT" || true+echo "imported $(head -1 "$HERE/COMMIT")"
@@ -0,0 +1,692 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*+ * WARNING: This file is auto-generated from scripts/autogen+ * in the mlkem-native repository.+ * Do not modify it directly.+ */++/******************************************************************************+ *+ * Single compilation unit (SCU) for fixed-level build of mlkem-native+ *+ * This compilation unit bundles together all source files for a build+ * of mlkem-native for a fixed security level (MLKEM-512/768/1024).+ *+ * # API+ *+ * The API exposed by this file is described in mlkem_native.h.+ *+ * # Multi-level build+ *+ * If you want an SCU build of mlkem-native with support for multiple security+ * levels, you need to include this file multiple times, and set+ * MLK_CONFIG_MULTILEVEL_WITH_SHARED and MLK_CONFIG_MULTILEVEL_NO_SHARED+ * appropriately. This is exemplified in examples/monolithic_build_multilevel+ * and examples/monolithic_build_multilevel_native.+ *+ * # Configuration+ *+ * The following options from the mlkem-native configuration are relevant:+ *+ * - MLK_CONFIG_FIPS202_CUSTOM_HEADER+ * Set this option if you use a custom FIPS202 implementation.+ *+ * - MLK_CONFIG_USE_NATIVE_BACKEND_ARITH+ * Set this option if you want to include the native arithmetic backends+ * in your build.+ *+ * - MLK_CONFIG_USE_NATIVE_BACKEND_FIPS202+ * Set this option if you want to include the native FIPS202 backends+ * in your build.+ *+ * - MLK_CONFIG_MONOBUILD_KEEP_SHARED_HEADERS+ * Set this option if you want to keep the directives defined in+ * level-independent headers. This is needed for a multi-level build.+ */++/* If parts of the mlkem-native source tree are not used,+ * consider reducing this header via `unifdef`.+ *+ * Example:+ * ```bash+ * unifdef -UMLK_CONFIG_USE_NATIVE_BACKEND_ARITH mlkem_native.c+ * ```+ */++#include "src/common.h"++#include "src/compress.c"+#include "src/debug.c"+#include "src/indcpa.c"+#include "src/kem.c"+#include "src/poly.c"+#include "src/poly_k.c"+#include "src/sampling.c"+#include "src/verify.c"++#if !defined(MLK_CONFIG_FIPS202_CUSTOM_HEADER)+#include "src/fips202/fips202.c"+#include "src/fips202/fips202x4.c"+#include "src/fips202/keccakf1600.c"+#endif++#if defined(MLK_CONFIG_USE_NATIVE_BACKEND_ARITH)+#if defined(MLK_SYS_AARCH64)+#include "src/native/aarch64/src/aarch64_zetas.c"+#include "src/native/aarch64/src/rej_uniform_table.c"+#endif+#if defined(MLK_SYS_X86_64)+#include "src/native/x86_64/src/compress_consts.c"+#include "src/native/x86_64/src/consts.c"+#include "src/native/x86_64/src/rej_uniform_table.c"+#endif+#if defined(MLK_SYS_RISCV64)+#include "src/native/riscv64/src/rv64v_debug.c"+#include "src/native/riscv64/src/rv64v_poly.c"+#endif+#if defined(MLK_SYS_PPC64LE)+#include "src/native/ppc64le/src/consts.c"+#endif+#endif /* MLK_CONFIG_USE_NATIVE_BACKEND_ARITH */++#if defined(MLK_CONFIG_USE_NATIVE_BACKEND_FIPS202)+#if defined(MLK_SYS_AARCH64)+#include "src/fips202/native/aarch64/src/keccakf1600_round_constants.c"+#endif+#if defined(MLK_SYS_X86_64)+#include "src/fips202/native/x86_64/src/keccakf1600_constants.c"+#endif+#if defined(MLK_SYS_ARMV81M_MVE)+#include "src/fips202/native/armv81m/src/keccak_f1600_x4_mve.c"+#include "src/fips202/native/armv81m/src/keccakf1600_round_constants.c"+#endif+#endif /* MLK_CONFIG_USE_NATIVE_BACKEND_FIPS202 */++/* Macro #undef's+ *+ * The following undefines macros from headers+ * included by the source files imported above.+ *+ * This is to allow building and linking multiple builds+ * of mlkem-native for varying parameter sets through concatenation+ * of this file, as if the files had been compiled separately.+ * If this is not relevant to you, you may remove the following.+ */++/*+ * Undefine macros from MLK_CONFIG_PARAMETER_SET-specific files+ */+/* mlkem/mlkem_native.h */+#undef MLKEM1024_BYTES+#undef MLKEM1024_CIPHERTEXTBYTES+#undef MLKEM1024_PUBLICKEYBYTES+#undef MLKEM1024_SECRETKEYBYTES+#undef MLKEM1024_SYMBYTES+#undef MLKEM512_BYTES+#undef MLKEM512_CIPHERTEXTBYTES+#undef MLKEM512_PUBLICKEYBYTES+#undef MLKEM512_SECRETKEYBYTES+#undef MLKEM512_SYMBYTES+#undef MLKEM768_BYTES+#undef MLKEM768_CIPHERTEXTBYTES+#undef MLKEM768_PUBLICKEYBYTES+#undef MLKEM768_SECRETKEYBYTES+#undef MLKEM768_SYMBYTES+#undef MLKEM_BYTES+#undef MLKEM_CIPHERTEXTBYTES+#undef MLKEM_CIPHERTEXTBYTES_+#undef MLKEM_PUBLICKEYBYTES+#undef MLKEM_PUBLICKEYBYTES_+#undef MLKEM_SECRETKEYBYTES+#undef MLKEM_SECRETKEYBYTES_+#undef MLKEM_SYMBYTES+#undef MLK_API_CONCAT+#undef MLK_API_CONCAT_+#undef MLK_API_CONCAT_UNDERSCORE+#undef MLK_API_MUST_CHECK_RETURN_VALUE+#undef MLK_API_NAMESPACE+#undef MLK_API_NAMESPACE_PREFIX+#undef MLK_API_QUALIFIER+#undef MLK_ERR_FAIL+#undef MLK_ERR_INVALID_PK+#undef MLK_ERR_INVALID_SK+#undef MLK_ERR_OUT_OF_MEMORY+#undef MLK_ERR_PCT_FAIL+#undef MLK_ERR_RNG_FAIL+#undef MLK_H+#undef MLK_MAX3_+#undef MLK_TOTAL_ALLOC_1024+#undef MLK_TOTAL_ALLOC_1024_DECAPS+#undef MLK_TOTAL_ALLOC_1024_ENCAPS+#undef MLK_TOTAL_ALLOC_1024_KEYPAIR+#undef MLK_TOTAL_ALLOC_1024_KEYPAIR_NO_PCT+#undef MLK_TOTAL_ALLOC_1024_KEYPAIR_PCT+#undef MLK_TOTAL_ALLOC_512+#undef MLK_TOTAL_ALLOC_512_DECAPS+#undef MLK_TOTAL_ALLOC_512_ENCAPS+#undef MLK_TOTAL_ALLOC_512_KEYPAIR+#undef MLK_TOTAL_ALLOC_512_KEYPAIR_NO_PCT+#undef MLK_TOTAL_ALLOC_512_KEYPAIR_PCT+#undef MLK_TOTAL_ALLOC_768+#undef MLK_TOTAL_ALLOC_768_DECAPS+#undef MLK_TOTAL_ALLOC_768_ENCAPS+#undef MLK_TOTAL_ALLOC_768_KEYPAIR+#undef MLK_TOTAL_ALLOC_768_KEYPAIR_NO_PCT+#undef MLK_TOTAL_ALLOC_768_KEYPAIR_PCT+/* mlkem/src/common.h */+#undef MLK_ADD_PARAM_SET+#undef MLK_ALLOC+#undef MLK_APPLY+#undef MLK_ASM_FN_SIZE+#undef MLK_ASM_FN_SYMBOL+#undef MLK_ASM_NAMESPACE+#undef MLK_BUILD_INTERNAL+#undef MLK_COMMON_H+#undef MLK_CONCAT+#undef MLK_CONCAT_+#undef MLK_EMPTY_CU+#undef MLK_ERR_FAIL+#undef MLK_ERR_INVALID_PK+#undef MLK_ERR_INVALID_SK+#undef MLK_ERR_OUT_OF_MEMORY+#undef MLK_ERR_PCT_FAIL+#undef MLK_ERR_RNG_FAIL+#undef MLK_EXTERNAL_API+#undef MLK_FIPS202X4_HEADER_FILE+#undef MLK_FIPS202_HEADER_FILE+#undef MLK_FREE+#undef MLK_INTERNAL_API+#undef MLK_INTERNAL_DATA_DECLARATION+#undef MLK_INTERNAL_DATA_DEFINITION+#undef MLK_NAMESPACE+#undef MLK_NAMESPACE_K+#undef MLK_NAMESPACE_PREFIX+#undef MLK_NAMESPACE_PREFIX_K+#undef mlk_memcpy+#undef mlk_memset+/* mlkem/src/indcpa.h */+#undef MLK_INDCPA_H+#undef mlk_gen_matrix+#undef mlk_indcpa_dec+#undef mlk_indcpa_enc+#undef mlk_indcpa_keypair_derand+/* mlkem/src/kem.h */+#undef MLK_KEM_H+#undef mlk_kem_check_pk+#undef mlk_kem_check_sk+#undef mlk_kem_dec+#undef mlk_kem_enc+#undef mlk_kem_enc_derand+#undef mlk_kem_keypair+#undef mlk_kem_keypair_derand+/* mlkem/src/params.h */+#undef MLKEM_DU+#undef MLKEM_DV+#undef MLKEM_ETA1+#undef MLKEM_ETA2+#undef MLKEM_INDCCA_CIPHERTEXTBYTES+#undef MLKEM_INDCCA_PUBLICKEYBYTES+#undef MLKEM_INDCCA_SECRETKEYBYTES+#undef MLKEM_INDCPA_BYTES+#undef MLKEM_INDCPA_MSGBYTES+#undef MLKEM_INDCPA_PUBLICKEYBYTES+#undef MLKEM_INDCPA_SECRETKEYBYTES+#undef MLKEM_K+#undef MLKEM_N+#undef MLKEM_POLYBYTES+#undef MLKEM_POLYCOMPRESSEDBYTES_D10+#undef MLKEM_POLYCOMPRESSEDBYTES_D11+#undef MLKEM_POLYCOMPRESSEDBYTES_D4+#undef MLKEM_POLYCOMPRESSEDBYTES_D5+#undef MLKEM_POLYCOMPRESSEDBYTES_DU+#undef MLKEM_POLYCOMPRESSEDBYTES_DV+#undef MLKEM_POLYVECBYTES+#undef MLKEM_POLYVECCOMPRESSEDBYTES_DU+#undef MLKEM_Q+#undef MLKEM_Q_HALF+#undef MLKEM_SSBYTES+#undef MLKEM_SYMBYTES+#undef MLKEM_UINT12_LIMIT+#undef MLK_PARAMS_H+/* mlkem/src/poly_k.h */+#undef MLK_POLY_K_H+#undef mlk_poly_compress_du+#undef mlk_poly_compress_dv+#undef mlk_poly_decompress_du+#undef mlk_poly_decompress_dv+#undef mlk_poly_getnoise_eta1122_4x+#undef mlk_poly_getnoise_eta1_4x+#undef mlk_poly_getnoise_eta2+#undef mlk_poly_getnoise_eta2_4x+#undef mlk_polymat+#undef mlk_polyvec+#undef mlk_polyvec_add+#undef mlk_polyvec_basemul_acc_montgomery_cached+#undef mlk_polyvec_compress_du+#undef mlk_polyvec_decompress_du+#undef mlk_polyvec_frombytes+#undef mlk_polyvec_invntt_tomont+#undef mlk_polyvec_mulcache+#undef mlk_polyvec_mulcache_compute+#undef mlk_polyvec_ntt+#undef mlk_polyvec_reduce+#undef mlk_polyvec_tobytes+#undef mlk_polyvec_tomont++#if !defined(MLK_CONFIG_MONOBUILD_KEEP_SHARED_HEADERS)+/*+ * Undefine macros from MLK_CONFIG_PARAMETER_SET-generic files+ */+/* mlkem/src/compress.h */+#undef MLK_COMPRESS_H+#undef mlk_poly_compress_d10+#undef mlk_poly_compress_d11+#undef mlk_poly_compress_d4+#undef mlk_poly_compress_d5+#undef mlk_poly_decompress_d10+#undef mlk_poly_decompress_d11+#undef mlk_poly_decompress_d4+#undef mlk_poly_decompress_d5+#undef mlk_poly_frombytes+#undef mlk_poly_frommsg+#undef mlk_poly_tobytes+#undef mlk_poly_tomsg+/* mlkem/src/context.h */+#undef MLK_CONTEXT_H+#undef MLK_CONTEXT_PARAMETERS_0+#undef MLK_CONTEXT_PARAMETERS_1+#undef MLK_CONTEXT_PARAMETERS_2+#undef MLK_CONTEXT_PARAMETERS_3+#undef MLK_CONTEXT_PARAMETERS_4+#undef MLK_CONTEXT_UNUSED+/* mlkem/src/debug.h */+#undef MLK_DEBUG_H+#undef mlk_assert+#undef mlk_assert_abs_bound+#undef mlk_assert_abs_bound_2d+#undef mlk_assert_bound+#undef mlk_assert_bound_2d+#undef mlk_debug_check_assert+#undef mlk_debug_check_bounds+/* mlkem/src/poly.h */+#undef MLK_INVNTT_BOUND+#undef MLK_NTT_BOUND+#undef MLK_POLY_H+#undef mlk_poly_add+#undef mlk_poly_invntt_tomont+#undef mlk_poly_mulcache_compute+#undef mlk_poly_ntt+#undef mlk_poly_reduce+#undef mlk_poly_sub+#undef mlk_poly_tomont+/* mlkem/src/randombytes.h */+#undef MLK_RANDOMBYTES_H+/* mlkem/src/sampling.h */+#undef MLK_SAMPLING_H+#undef mlk_poly_cbd2+#undef mlk_poly_cbd3+#undef mlk_poly_rej_uniform+#undef mlk_poly_rej_uniform_x4+/* mlkem/src/symmetric.h */+#undef MLK_SYMMETRIC_H+#undef MLK_XOF_RATE+#undef mlk_hash_g+#undef mlk_hash_h+#undef mlk_hash_j+#undef mlk_prf_eta+#undef mlk_prf_eta1+#undef mlk_prf_eta1_x4+#undef mlk_prf_eta2+#undef mlk_xof_absorb+#undef mlk_xof_ctx+#undef mlk_xof_init+#undef mlk_xof_release+#undef mlk_xof_squeezeblocks+#undef mlk_xof_x4_absorb+#undef mlk_xof_x4_ctx+#undef mlk_xof_x4_init+#undef mlk_xof_x4_release+#undef mlk_xof_x4_squeezeblocks+/* mlkem/src/sys.h */+#undef MLK_ALIGN+#undef MLK_ALIGN_UP+#undef MLK_ALWAYS_INLINE+#undef MLK_CET_ENDBR+#undef MLK_CT_TESTING_DECLASSIFY+#undef MLK_CT_TESTING_SECRET+#undef MLK_DEFAULT_ALIGN+#undef MLK_HAVE_INLINE_ASM+#undef MLK_INLINE+#undef MLK_MUST_CHECK_RETURN_VALUE+#undef MLK_NOINLINE+#undef MLK_RESTRICT+#undef MLK_STATIC_TESTABLE+#undef MLK_SYSV_ABI+#undef MLK_SYSV_ABI_SUPPORTED+#undef MLK_SYS_AARCH64+#undef MLK_SYS_AARCH64_EB+#undef MLK_SYS_AARCH64_NEON+#undef MLK_SYS_APPLE+#undef MLK_SYS_ARMV81M_MVE+#undef MLK_SYS_BIG_ENDIAN+#undef MLK_SYS_H+#undef MLK_SYS_LINUX+#undef MLK_SYS_LITTLE_ENDIAN+#undef MLK_SYS_PPC64LE+#undef MLK_SYS_RISCV32+#undef MLK_SYS_RISCV64+#undef MLK_SYS_RISCV64_RVV+#undef MLK_SYS_WINDOWS+#undef MLK_SYS_X86_64+#undef MLK_SYS_X86_64_AVX2+/* mlkem/src/verify.h */+#undef MLK_USE_ASM_VALUE_BARRIER+#undef MLK_VERIFY_H+#undef mlk_ct_opt_blocker_u64+/* mlkem/src/cbmc.h */+#undef MLK_CBMC_H+#undef __contract__+#undef __loop__++#if !defined(MLK_CONFIG_FIPS202_CUSTOM_HEADER)+/*+ * Undefine macros from FIPS-202 files+ */+/* mlkem/src/fips202/fips202.h */+#undef FIPS202_X4_DEFAULT_IMPLEMENTATION+#undef MLK_FIPS202_FIPS202_H+#undef SHA3_256_HASHBYTES+#undef SHA3_256_RATE+#undef SHA3_384_RATE+#undef SHA3_512_HASHBYTES+#undef SHA3_512_RATE+#undef SHAKE128_RATE+#undef SHAKE256_RATE+#undef mlk_sha3_256+#undef mlk_sha3_512+#undef mlk_shake128_absorb_once+#undef mlk_shake128_init+#undef mlk_shake128_release+#undef mlk_shake128_squeezeblocks+#undef mlk_shake256+/* mlkem/src/fips202/fips202x4.h */+#undef MLK_FIPS202_FIPS202X4_H+#undef mlk_shake128x4_absorb_once+#undef mlk_shake128x4_init+#undef mlk_shake128x4_release+#undef mlk_shake128x4_squeezeblocks+#undef mlk_shake256x4+/* mlkem/src/fips202/keccakf1600.h */+#undef MLK_FIPS202_KECCAKF1600_H+#undef MLK_KECCAK_LANES+#undef MLK_KECCAK_WAY+#undef mlk_keccakf1600_extract_bytes+#undef mlk_keccakf1600_permute+#undef mlk_keccakf1600_xor_bytes+#undef mlk_keccakf1600x4_extract_bytes+#undef mlk_keccakf1600x4_permute+#undef mlk_keccakf1600x4_xor_bytes+#endif /* !MLK_CONFIG_FIPS202_CUSTOM_HEADER */++#if defined(MLK_CONFIG_USE_NATIVE_BACKEND_FIPS202)+/* mlkem/src/fips202/native/api.h */+#undef MLK_FIPS202_NATIVE_API_H+#undef MLK_NATIVE_FUNC_FALLBACK+#undef MLK_NATIVE_FUNC_SUCCESS+/* mlkem/src/fips202/native/auto.h */+#undef MLK_FIPS202_NATIVE_AUTO_H+#if defined(MLK_SYS_AARCH64)+/*+ * Undefine macros from native code (FIPS202, AArch64)+ */+/* mlkem/src/fips202/native/aarch64/auto.h */+#undef MLK_FIPS202_NATIVE_AARCH64_AUTO_H+/* mlkem/src/fips202/native/aarch64/src/fips202_native_aarch64.h */+#undef MLK_FIPS202_NATIVE_AARCH64_SRC_FIPS202_NATIVE_AARCH64_H+#undef mlk_keccak_f1600_x1_scalar_aarch64_asm+#undef mlk_keccak_f1600_x1_v84a_aarch64_asm+#undef mlk_keccak_f1600_x2_v84a_aarch64_asm+#undef mlk_keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm+#undef mlk_keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm+#undef mlk_keccakf1600_round_constants+/* mlkem/src/fips202/native/aarch64/x1_scalar.h */+#undef MLK_FIPS202_AARCH64_NEED_X1_SCALAR+#undef MLK_FIPS202_NATIVE_AARCH64_X1_SCALAR_H+#undef MLK_USE_NATIVE_FIPS202_X1+/* mlkem/src/fips202/native/aarch64/x1_v84a.h */+#undef MLK_FIPS202_AARCH64_NEED_X1_V84A+#undef MLK_FIPS202_NATIVE_AARCH64_X1_V84A_H+#undef MLK_USE_NATIVE_FIPS202_X1+/* mlkem/src/fips202/native/aarch64/x2_v84a.h */+#undef MLK_FIPS202_AARCH64_NEED_X2_V84A+#undef MLK_FIPS202_NATIVE_AARCH64_X2_V84A_H+#undef MLK_USE_NATIVE_FIPS202_X4+/* mlkem/src/fips202/native/aarch64/x4_v8a_scalar.h */+#undef MLK_FIPS202_AARCH64_NEED_X4_V8A_SCALAR_HYBRID+#undef MLK_FIPS202_NATIVE_AARCH64_X4_V8A_SCALAR_H+#undef MLK_USE_NATIVE_FIPS202_X4+/* mlkem/src/fips202/native/aarch64/x4_v8a_v84a_scalar.h */+#undef MLK_FIPS202_AARCH64_NEED_X4_V8A_V84A_SCALAR_HYBRID+#undef MLK_FIPS202_NATIVE_AARCH64_X4_V8A_V84A_SCALAR_H+#undef MLK_USE_NATIVE_FIPS202_X4+#endif /* MLK_SYS_AARCH64 */+#if defined(MLK_SYS_X86_64)+/*+ * Undefine macros from native code (FIPS202, x86_64)+ */+/* mlkem/src/fips202/native/x86_64/keccak_f1600_x4_avx2.h */+#undef MLK_FIPS202_NATIVE_X86_64_KECCAK_F1600_X4_AVX2_H+#undef MLK_FIPS202_X86_64_NEED_X4_AVX2+#undef MLK_USE_NATIVE_FIPS202_X4+/* mlkem/src/fips202/native/x86_64/src/fips202_native_x86_64.h */+#undef MLK_FIPS202_NATIVE_X86_64_SRC_FIPS202_NATIVE_X86_64_H+#undef mlk_keccak_f1600_x4_avx2_asm+#undef mlk_keccak_rho56+#undef mlk_keccak_rho8+#undef mlk_keccakf1600_round_constants+#endif /* MLK_SYS_X86_64 */+#if defined(MLK_SYS_ARMV81M_MVE)+/*+ * Undefine macros from native code (FIPS202, Armv8.1-M)+ */+/* mlkem/src/fips202/native/armv81m/mve.h */+#undef MLK_FIPS202_ARMV81M_NEED_X4+#undef MLK_FIPS202_NATIVE_ARMV81M+#undef MLK_FIPS202_NATIVE_ARMV81M_MVE_H+#undef MLK_USE_NATIVE_FIPS202_X4+#undef MLK_USE_NATIVE_FIPS202_X4_EXTRACT_BYTES+#undef MLK_USE_NATIVE_FIPS202_X4_XOR_BYTES+#undef mlk_keccak_f1600_x4_native_impl+/* mlkem/src/fips202/native/armv81m/src/fips202_native_armv81m.h */+#undef MLK_FIPS202_NATIVE_ARMV81M_SRC_FIPS202_NATIVE_ARMV81M_H+#undef mlk_keccak_f1600_x4_mve_asm+#undef mlk_keccak_f1600_x4_state_extract_bytes_asm+#undef mlk_keccak_f1600_x4_state_xor_bytes_asm+#undef mlk_keccakf1600_round_constants+#endif /* MLK_SYS_ARMV81M_MVE */+#endif /* MLK_CONFIG_USE_NATIVE_BACKEND_FIPS202 */+#if defined(MLK_CONFIG_USE_NATIVE_BACKEND_ARITH)+/* mlkem/src/native/api.h */+#undef MLK_INVNTT_BOUND+#undef MLK_NATIVE_API_H+#undef MLK_NATIVE_FUNC_FALLBACK+#undef MLK_NATIVE_FUNC_SUCCESS+#undef MLK_NTT_BOUND+/* mlkem/src/native/meta.h */+#undef MLK_NATIVE_META_H+#if defined(MLK_SYS_AARCH64)+/*+ * Undefine macros from native code (Arith, AArch64)+ */+/* mlkem/src/native/aarch64/meta.h */+#undef MLK_ARITH_BACKEND_AARCH64+#undef MLK_NATIVE_AARCH64_META_H+#undef MLK_USE_NATIVE_INTT+#undef MLK_USE_NATIVE_NTT+#undef MLK_USE_NATIVE_POLYVEC_BASEMUL_ACC_MONTGOMERY_CACHED+#undef MLK_USE_NATIVE_POLY_MULCACHE_COMPUTE+#undef MLK_USE_NATIVE_POLY_REDUCE+#undef MLK_USE_NATIVE_POLY_TOBYTES+#undef MLK_USE_NATIVE_POLY_TOMONT+#undef MLK_USE_NATIVE_REJ_UNIFORM+/* mlkem/src/native/aarch64/src/arith_native_aarch64.h */+#undef MLK_NATIVE_AARCH64_SRC_ARITH_NATIVE_AARCH64_H+#undef mlk_aarch64_invntt_zetas_layer12345+#undef mlk_aarch64_invntt_zetas_layer67+#undef mlk_aarch64_ntt_zetas_layer12345+#undef mlk_aarch64_ntt_zetas_layer67+#undef mlk_aarch64_zetas_mulcache_native+#undef mlk_aarch64_zetas_mulcache_twisted_native+#undef mlk_intt_aarch64_asm+#undef mlk_ntt_aarch64_asm+#undef mlk_poly_mulcache_compute_aarch64_asm+#undef mlk_poly_reduce_aarch64_asm+#undef mlk_poly_tobytes_aarch64_asm+#undef mlk_poly_tomont_aarch64_asm+#undef mlk_polyvec_basemul_acc_montgomery_cached_k2_aarch64_asm+#undef mlk_polyvec_basemul_acc_montgomery_cached_k3_aarch64_asm+#undef mlk_polyvec_basemul_acc_montgomery_cached_k4_aarch64_asm+#undef mlk_rej_uniform_aarch64_asm+#undef mlk_rej_uniform_table+#endif /* MLK_SYS_AARCH64 */+#if defined(MLK_SYS_X86_64)+/*+ * Undefine macros from native code (Arith, X86_64)+ */+/* mlkem/src/native/x86_64/meta.h */+#undef MLK_ARITH_BACKEND_X86_64_DEFAULT+#undef MLK_NATIVE_X86_64_META_H+#undef MLK_USE_NATIVE_INTT+#undef MLK_USE_NATIVE_NTT+#undef MLK_USE_NATIVE_NTT_CUSTOM_ORDER+#undef MLK_USE_NATIVE_POLYVEC_BASEMUL_ACC_MONTGOMERY_CACHED+#undef MLK_USE_NATIVE_POLY_COMPRESS_D10+#undef MLK_USE_NATIVE_POLY_COMPRESS_D11+#undef MLK_USE_NATIVE_POLY_COMPRESS_D4+#undef MLK_USE_NATIVE_POLY_COMPRESS_D5+#undef MLK_USE_NATIVE_POLY_DECOMPRESS_D10+#undef MLK_USE_NATIVE_POLY_DECOMPRESS_D11+#undef MLK_USE_NATIVE_POLY_DECOMPRESS_D4+#undef MLK_USE_NATIVE_POLY_DECOMPRESS_D5+#undef MLK_USE_NATIVE_POLY_FROMBYTES+#undef MLK_USE_NATIVE_POLY_MULCACHE_COMPUTE+#undef MLK_USE_NATIVE_POLY_REDUCE+#undef MLK_USE_NATIVE_POLY_TOBYTES+#undef MLK_USE_NATIVE_POLY_TOMONT+#undef MLK_USE_NATIVE_REJ_UNIFORM+/* mlkem/src/native/x86_64/src/arith_native_x86_64.h */+#undef MLK_AVX2_REJ_UNIFORM_BUFLEN+#undef MLK_NATIVE_X86_64_SRC_ARITH_NATIVE_X86_64_H+#undef mlk_invntt_avx2_asm+#undef mlk_ntt_avx2_asm+#undef mlk_nttfrombytes_avx2_asm+#undef mlk_ntttobytes_avx2_asm+#undef mlk_nttunpack_avx2_asm+#undef mlk_poly_compress_d10_avx2_asm+#undef mlk_poly_compress_d11_avx2_asm+#undef mlk_poly_compress_d4_avx2_asm+#undef mlk_poly_compress_d5_avx2_asm+#undef mlk_poly_decompress_d10_avx2_asm+#undef mlk_poly_decompress_d11_avx2_asm+#undef mlk_poly_decompress_d4_avx2_asm+#undef mlk_poly_decompress_d5_avx2_asm+#undef mlk_poly_mulcache_compute_avx2_asm+#undef mlk_polyvec_basemul_acc_montgomery_cached_k2_avx2_asm+#undef mlk_polyvec_basemul_acc_montgomery_cached_k3_avx2_asm+#undef mlk_polyvec_basemul_acc_montgomery_cached_k4_avx2_asm+#undef mlk_reduce_avx2_asm+#undef mlk_rej_uniform_avx2_asm+#undef mlk_rej_uniform_table+#undef mlk_tomont_avx2_asm+/* mlkem/src/native/x86_64/src/compress_consts.h */+#undef MLK_NATIVE_X86_64_SRC_COMPRESS_CONSTS_H+#undef mlk_compress_d10_data+#undef mlk_compress_d11_data+#undef mlk_compress_d4_data+#undef mlk_compress_d5_data+#undef mlk_decompress_d10_data+#undef mlk_decompress_d11_data+#undef mlk_decompress_d4_data+#undef mlk_decompress_d5_data+/* mlkem/src/native/x86_64/src/consts.h */+#undef MLK_AVX2_BACKEND_DATA_OFFSET_MULCACHE_TWIDDLES+#undef MLK_AVX2_BACKEND_DATA_OFFSET_REVIDXB+#undef MLK_AVX2_BACKEND_DATA_OFFSET_REVIDXD+#undef MLK_AVX2_BACKEND_DATA_OFFSET_ZETAS_EXP+#undef MLK_NATIVE_X86_64_SRC_CONSTS_H+#undef mlk_qdata+#endif /* MLK_SYS_X86_64 */+#if defined(MLK_SYS_RISCV64)+/*+ * Undefine macros from native code (Arith, RISC-V 64)+ */+/* mlkem/src/native/riscv64/meta.h */+#undef MLK_ARITH_BACKEND_RISCV64+#undef MLK_NATIVE_RISCV64_META_H+#undef MLK_USE_NATIVE_INTT+#undef MLK_USE_NATIVE_NTT+#undef MLK_USE_NATIVE_POLYVEC_BASEMUL_ACC_MONTGOMERY_CACHED+#undef MLK_USE_NATIVE_POLY_MULCACHE_COMPUTE+#undef MLK_USE_NATIVE_POLY_REDUCE+#undef MLK_USE_NATIVE_POLY_TOMONT+#undef MLK_USE_NATIVE_REJ_UNIFORM+/* mlkem/src/native/riscv64/src/arith_native_riscv64.h */+#undef MLK_NATIVE_RISCV64_SRC_ARITH_NATIVE_RISCV64_H+#undef mlk_rv64v_poly_add+#undef mlk_rv64v_poly_basemul_mont_add_k2+#undef mlk_rv64v_poly_basemul_mont_add_k3+#undef mlk_rv64v_poly_basemul_mont_add_k4+#undef mlk_rv64v_poly_invntt_tomont+#undef mlk_rv64v_poly_ntt+#undef mlk_rv64v_poly_reduce+#undef mlk_rv64v_poly_sub+#undef mlk_rv64v_poly_tomont+#undef mlk_rv64v_rej_uniform+/* mlkem/src/native/riscv64/src/rv64v_debug.h */+#undef MLK_NATIVE_RISCV64_SRC_RV64V_DEBUG_H+#undef mlk_assert_abs_bound_int16m1+#undef mlk_assert_abs_bound_int16m2+#undef mlk_assert_bound_int16m1+#undef mlk_assert_bound_int16m2+#undef mlk_debug_check_bounds_int16m1+#undef mlk_debug_check_bounds_int16m2+#endif /* MLK_SYS_RISCV64 */+#if defined(MLK_SYS_PPC64LE)+/*+ * Undefine macros from native code (Arith, PPC64LE)+ */+/* mlkem/src/native/ppc64le/meta.h */+#undef MLK_ARITH_BACKEND_NAME+#undef MLK_ARITH_BACKEND_PPC64LE_DEFAULT+#undef MLK_NATIVE_PPC64LE_META_H+#undef MLK_USE_NATIVE_INTT+#undef MLK_USE_NATIVE_NTT+#undef MLK_USE_NATIVE_POLY_REDUCE+#undef MLK_USE_NATIVE_POLY_TOMONT+/* mlkem/src/native/ppc64le/src/arith_native_ppc64le.h */+#undef MLK_NATIVE_PPC64LE_SRC_ARITH_NATIVE_PPC64LE_H+#undef mlk_intt_ppc_asm+#undef mlk_ntt_ppc_asm+#undef mlk_poly_tomont_ppc_asm+#undef mlk_reduce_ppc_asm+/* mlkem/src/native/ppc64le/src/consts.h */+#undef MLK_NATIVE_PPC64LE_SRC_CONSTS_H+#undef MLK_PPC_C20159_OFFSET+#undef MLK_PPC_NQ_OFFSET+#undef MLK_PPC_N_INV_OFFSET+#undef MLK_PPC_N_INV_TW_OFFSET+#undef MLK_PPC_Q_OFFSET+#undef MLK_PPC_TOMONT_OFFSET+#undef MLK_PPC_TOMONT_TW_OFFSET+#undef MLK_PPC_ZETA_INTT_OFFSET+#undef MLK_PPC_ZETA_INTT_TW_OFFSET+#undef MLK_PPC_ZETA_NTT_OFFSET+#undef MLK_PPC_ZETA_NTT_TW_OFFSET+#undef mlk_ppc_qdata+#endif /* MLK_SYS_PPC64LE */+#endif /* MLK_CONFIG_USE_NATIVE_BACKEND_ARITH */+#endif /* !MLK_CONFIG_MONOBUILD_KEEP_SHARED_HEADERS */
@@ -0,0 +1,464 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS140_3_IG]+ * Implementation Guidance for FIPS 140-3 and the Cryptographic Module+ * Validation Program+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/projects/cryptographic-module-validation-program/fips-140-3-ig-announcements+ *+ * - [FIPS203]+ * FIPS 203 Module-Lattice-Based Key-Encapsulation Mechanism Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/203/final+ */++#ifndef MLK_H+#define MLK_H++/*+ * Public API for mlkem-native.+ *+ * This header defines the public API of a single build of mlkem-native.+ *+ * Make sure the configuration file is in the include path+ * (this is "mlkem_native_config.h" by default, or MLK_CONFIG_FILE if defined).+ *+ * # API conventions+ *+ * Conventions shared by all functions below (return values, pointer validity,+ * output buffers on error) are documented in API-CONVENTIONS.md.+ *+ * # Multi-level builds+ *+ * This header specifies a build of mlkem-native for a fixed security level.+ * If you need multiple security levels, leave the security level unspecified+ * in the configuration file and include this header multiple times, setting+ * MLK_CONFIG_PARAMETER_SET accordingly for each, and #undef'ing the MLK_H+ * guard to allow multiple inclusions.+ *+ * In this case, the configuration file must also set+ * MLK_CONFIG_MULTILEVEL_BUILD. Without it, the parameter set is not appended+ * to the namespace prefix and all inclusions declare the same symbol names.+ */++/******************************* Key sizes ************************************/++/* Sizes of cryptographic material, per parameter set */+/* See mlkem/src/params.h for the arithmetic expressions giving rise to these */+/* check-magic: off */+#define MLKEM512_SECRETKEYBYTES 1632+#define MLKEM512_PUBLICKEYBYTES 800+#define MLKEM512_CIPHERTEXTBYTES 768++#define MLKEM768_SECRETKEYBYTES 2400+#define MLKEM768_PUBLICKEYBYTES 1184+#define MLKEM768_CIPHERTEXTBYTES 1088++#define MLKEM1024_SECRETKEYBYTES 3168+#define MLKEM1024_PUBLICKEYBYTES 1568+#define MLKEM1024_CIPHERTEXTBYTES 1568+/* check-magic: on */++/* Size of randomness coins in bytes (level-independent) */+#define MLKEM_SYMBYTES 32+#define MLKEM512_SYMBYTES MLKEM_SYMBYTES+#define MLKEM768_SYMBYTES MLKEM_SYMBYTES+#define MLKEM1024_SYMBYTES MLKEM_SYMBYTES+/* Size of shared secret in bytes (level-independent) */+#define MLKEM_BYTES 32+#define MLKEM512_BYTES MLKEM_BYTES+#define MLKEM768_BYTES MLKEM_BYTES+#define MLKEM1024_BYTES MLKEM_BYTES++/* Sizes of cryptographic material, as a function of LVL=512,768,1024 */+#define MLKEM_SECRETKEYBYTES_(LVL) MLKEM##LVL##_SECRETKEYBYTES+#define MLKEM_PUBLICKEYBYTES_(LVL) MLKEM##LVL##_PUBLICKEYBYTES+#define MLKEM_CIPHERTEXTBYTES_(LVL) MLKEM##LVL##_CIPHERTEXTBYTES+#define MLKEM_SECRETKEYBYTES(LVL) MLKEM_SECRETKEYBYTES_(LVL)+#define MLKEM_PUBLICKEYBYTES(LVL) MLKEM_PUBLICKEYBYTES_(LVL)+#define MLKEM_CIPHERTEXTBYTES(LVL) MLKEM_CIPHERTEXTBYTES_(LVL)++/****************************** Error codes ***********************************/++/* Generic failure condition. Currently not returned by any function;+ * reserved for failures that no more specific code covers. */+#define MLK_ERR_FAIL (-1)+/* An allocation failed. This can only happen if MLK_CONFIG_CUSTOM_ALLOC_FREE+ * is defined and the provided MLK_CUSTOM_ALLOC can fail. */+#define MLK_ERR_OUT_OF_MEMORY (-2)+/* An RNG failure occurred. Might be due to insufficient entropy or+ * system misconfiguration. */+#define MLK_ERR_RNG_FAIL (-3)+/* Public key validation failed: the @[FIPS203, Section 7.2, 'modulus check']+ * found a coefficient outside [0,q-1]. Returned by check_pk and by the+ * encapsulation API. */+#define MLK_ERR_INVALID_PK (-4)+/* Secret key validation failed: the @[FIPS203, Section 7.3, 'hash check']+ * found the embedded public key hash inconsistent. Returned by check_sk and+ * by the decapsulation API. */+#define MLK_ERR_INVALID_SK (-5)+/* The 'Pairwise Consistency Test' @[FIPS140_3_IG, p.87] and+ * @[FIPS203, Section 7.1, Pairwise Consistency] failed. Only possible when+ * MLK_CONFIG_KEYGEN_PCT is enabled; signals that the freshly generated key+ * pair failed its encaps/decaps self-test. */+#define MLK_ERR_PCT_FAIL (-6)++/********************* Namespacing and Qualifiers *****************************/++#define MLK_API_CONCAT_(x, y) x##y+#define MLK_API_CONCAT(x, y) MLK_API_CONCAT_(x, y)+#define MLK_API_CONCAT_UNDERSCORE(x, y) MLK_API_CONCAT(MLK_API_CONCAT(x, _), y)++/* You need to make sure the config file is in the include path. */+#if defined(MLK_CONFIG_FILE)+#include MLK_CONFIG_FILE+#else+#include "mlkem_native_config.h"+#endif++/* Namespace prefix for the public API symbols. For multi-level builds, the+ * parameter set is appended to disambiguate the security levels. */+#if defined(MLK_CONFIG_MULTILEVEL_BUILD)+#define MLK_API_NAMESPACE_PREFIX \+ MLK_API_CONCAT(MLK_CONFIG_NAMESPACE_PREFIX, MLK_CONFIG_PARAMETER_SET)+#else+#define MLK_API_NAMESPACE_PREFIX MLK_CONFIG_NAMESPACE_PREFIX+#endif++#define MLK_API_NAMESPACE(sym) \+ MLK_API_CONCAT_UNDERSCORE(MLK_API_NAMESPACE_PREFIX, sym)++#if defined(__GNUC__) || defined(__clang__)+#define MLK_API_MUST_CHECK_RETURN_VALUE __attribute__((warn_unused_result))+#else+#define MLK_API_MUST_CHECK_RETURN_VALUE+#endif++#if defined(MLK_CONFIG_EXTERNAL_API_QUALIFIER)+#define MLK_API_QUALIFIER MLK_CONFIG_EXTERNAL_API_QUALIFIER+#else+#define MLK_API_QUALIFIER+#endif++/****************************** Function API **********************************/++#if !defined(MLK_CONFIG_CONSTANTS_ONLY)++#include <stdint.h>++#ifdef __cplusplus+extern "C"+{+#endif++#if !defined(MLK_CONFIG_NO_KEYPAIR_API)+/**+ * Generate a public/private keypair for the ML-KEM key encapsulation mechanism.+ *+ * @spec{Implements @[FIPS203, Algorithm 16, ML-KEM.KeyGen_Internal].}+ *+ * @param[out] pk Output public key, an array of+ * MLKEM{512,768,1024}_PUBLICKEYBYTES bytes.+ * @param[out] sk Output private key, an array of+ * MLKEM{512,768,1024}_SECRETKEYBYTES bytes.+ * @param[in] coins Input randomness, an array of 2*MLKEM_SYMBYTES uniformly+ * random bytes.+ * @param context Application context. Only present when+ * MLK_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLK_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLK_ERR_PCT_FAIL MLK_CONFIG_KEYGEN_PCT enabled and PCT failed.+ * @retval MLK_ERR_OUT_OF_MEMORY MLK_CONFIG_CUSTOM_ALLOC_FREE was used and+ * MLK_CUSTOM_ALLOC returned NULL.+ * @retval MLK_ERR_RNG_FAIL MLK_CONFIG_KEYGEN_PCT enabled and random+ * number generation failed within the PCT.+ */+MLK_API_QUALIFIER+MLK_API_MUST_CHECK_RETURN_VALUE+int MLK_API_NAMESPACE(keypair_derand)(+ uint8_t pk[MLKEM_PUBLICKEYBYTES(MLK_CONFIG_PARAMETER_SET)],+ uint8_t sk[MLKEM_SECRETKEYBYTES(MLK_CONFIG_PARAMETER_SET)],+ const uint8_t coins[2 * MLKEM_SYMBYTES]+#ifdef MLK_CONFIG_CONTEXT_PARAMETER+ ,+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context+#endif+);+++#if !defined(MLK_CONFIG_NO_RANDOMIZED_API)+/**+ * Generate a public/private keypair for the ML-KEM key encapsulation mechanism.+ *+ * @spec{Implements @[FIPS203, Algorithm 19, ML-KEM.KeyGen].}+ *+ * @param[out] pk Output public key, an array of+ * MLKEM{512,768,1024}_PUBLICKEYBYTES bytes.+ * @param[out] sk Output private key, an array of+ * MLKEM{512,768,1024}_SECRETKEYBYTES bytes.+ * @param context Application context. Only present when+ * MLK_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLK_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLK_ERR_PCT_FAIL MLK_CONFIG_KEYGEN_PCT enabled and PCT failed.+ * @retval MLK_ERR_OUT_OF_MEMORY MLK_CONFIG_CUSTOM_ALLOC_FREE was used and+ * MLK_CUSTOM_ALLOC returned NULL.+ * @retval MLK_ERR_RNG_FAIL Random number generation failed.+ */+MLK_API_QUALIFIER+MLK_API_MUST_CHECK_RETURN_VALUE+int MLK_API_NAMESPACE(keypair)(+ uint8_t pk[MLKEM_PUBLICKEYBYTES(MLK_CONFIG_PARAMETER_SET)],+ uint8_t sk[MLKEM_SECRETKEYBYTES(MLK_CONFIG_PARAMETER_SET)]+#ifdef MLK_CONFIG_CONTEXT_PARAMETER+ ,+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context+#endif+);+#endif /* !MLK_CONFIG_NO_RANDOMIZED_API */+#endif /* !MLK_CONFIG_NO_KEYPAIR_API */++#if !defined(MLK_CONFIG_NO_ENCAPS_API)+/**+ * Generate ciphertext and shared secret for a given public key.+ *+ * @spec{Implements @[FIPS203, Algorithm 17, ML-KEM.Encaps_Internal].}+ *+ * @param[out] ct Output ciphertext, an array of+ * MLKEM{512,768,1024}_CIPHERTEXTBYTES bytes.+ * @param[out] ss Output shared secret, an array of MLKEM_BYTES bytes.+ * @param[in] pk Input public key, an array of+ * MLKEM{512,768,1024}_PUBLICKEYBYTES bytes.+ * @param[in] coins Input randomness, an array of MLKEM_SYMBYTES bytes.+ * @param context Application context. Only present when+ * MLK_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLK_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLK_ERR_INVALID_PK The 'modulus check' @[FIPS203, Section 7.2]+ * for the public key failed.+ * @retval MLK_ERR_OUT_OF_MEMORY MLK_CONFIG_CUSTOM_ALLOC_FREE was used and+ * MLK_CUSTOM_ALLOC returned NULL.+ */+MLK_API_QUALIFIER+MLK_API_MUST_CHECK_RETURN_VALUE+int MLK_API_NAMESPACE(enc_derand)(+ uint8_t ct[MLKEM_CIPHERTEXTBYTES(MLK_CONFIG_PARAMETER_SET)],+ uint8_t ss[MLKEM_BYTES],+ const uint8_t pk[MLKEM_PUBLICKEYBYTES(MLK_CONFIG_PARAMETER_SET)],+ const uint8_t coins[MLKEM_SYMBYTES]+#ifdef MLK_CONFIG_CONTEXT_PARAMETER+ ,+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context+#endif+);++#if !defined(MLK_CONFIG_NO_RANDOMIZED_API)+/**+ * Generate ciphertext and shared secret for a given public key.+ *+ * @spec{Implements @[FIPS203, Algorithm 20, ML-KEM.Encaps].}+ *+ * @param[out] ct Output ciphertext, an array of+ * MLKEM{512,768,1024}_CIPHERTEXTBYTES bytes.+ * @param[out] ss Output shared secret, an array of MLKEM_BYTES bytes.+ * @param[in] pk Input public key, an array of+ * MLKEM{512,768,1024}_PUBLICKEYBYTES bytes.+ * @param context Application context. Only present when+ * MLK_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLK_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLK_ERR_INVALID_PK The 'modulus check' @[FIPS203, Section 7.2]+ * for the public key failed.+ * @retval MLK_ERR_OUT_OF_MEMORY MLK_CONFIG_CUSTOM_ALLOC_FREE was used and+ * MLK_CUSTOM_ALLOC returned NULL.+ * @retval MLK_ERR_RNG_FAIL Random number generation failed.+ */+MLK_API_QUALIFIER+MLK_API_MUST_CHECK_RETURN_VALUE+int MLK_API_NAMESPACE(enc)(+ uint8_t ct[MLKEM_CIPHERTEXTBYTES(MLK_CONFIG_PARAMETER_SET)],+ uint8_t ss[MLKEM_BYTES],+ const uint8_t pk[MLKEM_PUBLICKEYBYTES(MLK_CONFIG_PARAMETER_SET)]+#ifdef MLK_CONFIG_CONTEXT_PARAMETER+ ,+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context+#endif+);+#endif /* !MLK_CONFIG_NO_RANDOMIZED_API */+#endif /* !MLK_CONFIG_NO_ENCAPS_API */++#if !defined(MLK_CONFIG_NO_DECAPS_API)+/**+ * Generate shared secret for a given ciphertext and private key.+ *+ * @spec{Implements @[FIPS203, Algorithm 21, ML-KEM.Decaps].}+ *+ * @param[out] ss Output shared secret, an array of MLKEM_BYTES bytes.+ * @param[in] ct Input ciphertext, an array of+ * MLKEM{512,768,1024}_CIPHERTEXTBYTES bytes.+ * @param[in] sk Input private key, an array of+ * MLKEM{512,768,1024}_SECRETKEYBYTES bytes.+ * @param context Application context. Only present when+ * MLK_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLK_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLK_ERR_INVALID_SK The 'hash check' @[FIPS203, Section 7.3]+ * for the secret key failed.+ * @retval MLK_ERR_OUT_OF_MEMORY MLK_CONFIG_CUSTOM_ALLOC_FREE was used and+ * MLK_CUSTOM_ALLOC returned NULL.+ */+MLK_API_QUALIFIER+MLK_API_MUST_CHECK_RETURN_VALUE+int MLK_API_NAMESPACE(dec)(+ uint8_t ss[MLKEM_BYTES],+ const uint8_t ct[MLKEM_CIPHERTEXTBYTES(MLK_CONFIG_PARAMETER_SET)],+ const uint8_t sk[MLKEM_SECRETKEYBYTES(MLK_CONFIG_PARAMETER_SET)]+#ifdef MLK_CONFIG_CONTEXT_PARAMETER+ ,+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context+#endif+);+#endif /* !MLK_CONFIG_NO_DECAPS_API */+++/**+ * Implements modulus check mandated by FIPS 203, i.e., ensures that+ * coefficients are in [0,q-1].+ *+ * @spec{Implements @[FIPS203, Section 7.2, 'modulus check'].}+ *+ * @param[in] pk Input public key, an array of+ * MLKEM{512,768,1024}_PUBLICKEYBYTES bytes.+ * @param context Application context. Only present when+ * MLK_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLK_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLK_ERR_INVALID_PK Modulus check failed.+ * @retval MLK_ERR_OUT_OF_MEMORY MLK_CONFIG_CUSTOM_ALLOC_FREE was used and+ * MLK_CUSTOM_ALLOC returned NULL.+ */+#if !defined(MLK_CONFIG_NO_ENCAPS_API)+MLK_API_QUALIFIER+MLK_API_MUST_CHECK_RETURN_VALUE+int MLK_API_NAMESPACE(check_pk)(+ const uint8_t pk[MLKEM_PUBLICKEYBYTES(MLK_CONFIG_PARAMETER_SET)]+#ifdef MLK_CONFIG_CONTEXT_PARAMETER+ ,+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context+#endif+);+#endif /* !MLK_CONFIG_NO_ENCAPS_API */++/**+ * Implements public key hash check mandated by FIPS 203, i.e., ensures that+ * sk[768𝑘+32 ∶ 768𝑘+64] = H(pk) = H(sk[384𝑘 : 768𝑘+32]).+ *+ * @spec{Implements @[FIPS203, Section 7.3, 'hash check'].}+ *+ * @param[in] sk Input private key, an array of+ * MLKEM{512,768,1024}_SECRETKEYBYTES bytes.+ * @param context Application context. Only present when+ * MLK_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLK_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLK_ERR_INVALID_SK Public key hash check failed.+ * @retval MLK_ERR_OUT_OF_MEMORY MLK_CONFIG_CUSTOM_ALLOC_FREE was used and+ * MLK_CUSTOM_ALLOC returned NULL.+ */+#if !defined(MLK_CONFIG_NO_DECAPS_API)+MLK_API_QUALIFIER+MLK_API_MUST_CHECK_RETURN_VALUE+int MLK_API_NAMESPACE(check_sk)(+ const uint8_t sk[MLKEM_SECRETKEYBYTES(MLK_CONFIG_PARAMETER_SET)]+#ifdef MLK_CONFIG_CONTEXT_PARAMETER+ ,+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context+#endif+);+#endif /* !MLK_CONFIG_NO_DECAPS_API */++#ifdef __cplusplus+}+#endif++#undef MLK_API_NAMESPACE_PREFIX++#endif /* !MLK_CONFIG_CONSTANTS_ONLY */+++/***************************** Memory Usage **********************************/++/*+ * By default mlkem-native performs all memory allocations on the stack.+ * Alternatively, mlkem-native supports custom allocation of large structures+ * through the `MLK_CONFIG_CUSTOM_ALLOC_FREE` configuration option.+ * See mlkem_native_config.h for details.+ *+ * `MLK_TOTAL_ALLOC_{512,768,1024}_{KEYPAIR,ENCAPS,DECAPS}` indicates the+ * maximum (accumulative) allocation via MLK_ALLOC for each parameter set and+ * operation. Note that some stack allocation remains even when using custom+ * allocators, so these values are lower than total stack usage with the default+ * stack-only allocation.+ *+ * These constants may be used to implement custom allocations using a+ * fixed-sized buffer and a simple allocator (e.g., bump allocator).+ */+/* check-magic: off */+#define MLK_TOTAL_ALLOC_512_KEYPAIR_NO_PCT 5824+#define MLK_TOTAL_ALLOC_512_KEYPAIR_PCT 10048+#define MLK_TOTAL_ALLOC_512_ENCAPS 8384+#define MLK_TOTAL_ALLOC_512_DECAPS 9152+#define MLK_TOTAL_ALLOC_768_KEYPAIR_NO_PCT 10176+#define MLK_TOTAL_ALLOC_768_KEYPAIR_PCT 15552+#define MLK_TOTAL_ALLOC_768_ENCAPS 13248+#define MLK_TOTAL_ALLOC_768_DECAPS 14336+#define MLK_TOTAL_ALLOC_1024_KEYPAIR_NO_PCT 15552+#define MLK_TOTAL_ALLOC_1024_KEYPAIR_PCT 22400+#define MLK_TOTAL_ALLOC_1024_ENCAPS 19136+#define MLK_TOTAL_ALLOC_1024_DECAPS 20704+/* check-magic: on */++/*+ * MLK_TOTAL_ALLOC_*_KEYPAIR adapts based on MLK_CONFIG_KEYGEN_PCT.+ */+#if defined(MLK_CONFIG_KEYGEN_PCT)+#define MLK_TOTAL_ALLOC_512_KEYPAIR MLK_TOTAL_ALLOC_512_KEYPAIR_PCT+#define MLK_TOTAL_ALLOC_768_KEYPAIR MLK_TOTAL_ALLOC_768_KEYPAIR_PCT+#define MLK_TOTAL_ALLOC_1024_KEYPAIR MLK_TOTAL_ALLOC_1024_KEYPAIR_PCT+#else+#define MLK_TOTAL_ALLOC_512_KEYPAIR MLK_TOTAL_ALLOC_512_KEYPAIR_NO_PCT+#define MLK_TOTAL_ALLOC_768_KEYPAIR MLK_TOTAL_ALLOC_768_KEYPAIR_NO_PCT+#define MLK_TOTAL_ALLOC_1024_KEYPAIR MLK_TOTAL_ALLOC_1024_KEYPAIR_NO_PCT+#endif++#define MLK_MAX3_(a, b, c) \+ ((a) > (b) ? ((a) > (c) ? (a) : (c)) : ((b) > (c) ? (b) : (c)))++/*+ * `MLK_TOTAL_ALLOC_{512,768,1024}` is the maximum across all operations for+ * each parameter set.+ */+#define MLK_TOTAL_ALLOC_512 \+ MLK_MAX3_(MLK_TOTAL_ALLOC_512_KEYPAIR, MLK_TOTAL_ALLOC_512_ENCAPS, \+ MLK_TOTAL_ALLOC_512_DECAPS)+#define MLK_TOTAL_ALLOC_768 \+ MLK_MAX3_(MLK_TOTAL_ALLOC_768_KEYPAIR, MLK_TOTAL_ALLOC_768_ENCAPS, \+ MLK_TOTAL_ALLOC_768_DECAPS)+#define MLK_TOTAL_ALLOC_1024 \+ MLK_MAX3_(MLK_TOTAL_ALLOC_1024_KEYPAIR, MLK_TOTAL_ALLOC_1024_ENCAPS, \+ MLK_TOTAL_ALLOC_1024_DECAPS)++#endif /* !MLK_H */
@@ -0,0 +1,716 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*+ * WARNING: This file is auto-generated from scripts/autogen+ * in the mlkem-native repository.+ * Do not modify it directly.+ */++/******************************************************************************+ *+ * Single assembly unit for fixed-level build of mlkem-native+ *+ * This assembly unit bundles together all assembly files for a build+ * of mlkem-native for a fixed security level (MLKEM-512/768/1024).+ *+ * # Multi-level build+ *+ * If you want an SCU build of mlkem-native with support for multiple security+ * levels, you should include this file once with+ * MLK_CONFIG_MULTILEVEL_WITH_SHARED set.+ *+ * (You could also follow the same pattern as for mlkem_native.c+ * and include it for every level, setting MLK_CONFIG_MULTILEVEL_NO_SHARED+ * for all but one. For builds with MLK_CONFIG_MULTILEVEL_NO_SHARED, this+ * file will then be ignored.)+ *+ * # Configuration+ *+ * The following options from the mlkem-native configuration are relevant:+ *+ * - MLK_CONFIG_FIPS202_CUSTOM_HEADER+ * Set this option if you use a custom FIPS202 implementation.+ *+ * - MLK_CONFIG_USE_NATIVE_BACKEND_ARITH+ * Set this option if you want to include the native arithmetic backends+ * in your build.+ *+ * - MLK_CONFIG_USE_NATIVE_BACKEND_FIPS202+ * Set this option if you want to include the native FIPS202 backends+ * in your build.+ *+ * - MLK_CONFIG_MONOBUILD_KEEP_SHARED_HEADERS+ * Set this option if you want to keep the directives defined in+ * level-independent headers. This is needed for a multi-level build.+ */++/* If parts of the mlkem-native source tree are not used,+ * consider reducing this header via `unifdef`.+ *+ * Example:+ * ```bash+ * unifdef -UMLK_CONFIG_USE_NATIVE_BACKEND_ARITH mlkem_native_asm.S+ * ```+ */++#include "src/common.h"++#if defined(MLK_CONFIG_USE_NATIVE_BACKEND_ARITH)+#if defined(MLK_SYS_AARCH64)+#include "src/native/aarch64/src/mlkem_intt_aarch64_asm.S"+#include "src/native/aarch64/src/mlkem_ntt_aarch64_asm.S"+#include "src/native/aarch64/src/mlkem_poly_mulcache_compute_aarch64_asm.S"+#include "src/native/aarch64/src/mlkem_poly_reduce_aarch64_asm.S"+#include "src/native/aarch64/src/mlkem_poly_tobytes_aarch64_asm.S"+#include "src/native/aarch64/src/mlkem_poly_tomont_aarch64_asm.S"+#include "src/native/aarch64/src/mlkem_polyvec_basemul_acc_montgomery_cached_k2_aarch64_asm.S"+#include "src/native/aarch64/src/mlkem_polyvec_basemul_acc_montgomery_cached_k3_aarch64_asm.S"+#include "src/native/aarch64/src/mlkem_polyvec_basemul_acc_montgomery_cached_k4_aarch64_asm.S"+#include "src/native/aarch64/src/mlkem_rej_uniform_aarch64_asm.S"+#endif /* MLK_SYS_AARCH64 */+#if defined(MLK_SYS_X86_64)+#include "src/native/x86_64/src/mlkem_intt_avx2_asm.S"+#include "src/native/x86_64/src/mlkem_ntt_avx2_asm.S"+#include "src/native/x86_64/src/mlkem_nttfrombytes_avx2_asm.S"+#include "src/native/x86_64/src/mlkem_ntttobytes_avx2_asm.S"+#include "src/native/x86_64/src/mlkem_nttunpack_avx2_asm.S"+#include "src/native/x86_64/src/mlkem_poly_compress_d10_avx2_asm.S"+#include "src/native/x86_64/src/mlkem_poly_compress_d11_avx2_asm.S"+#include "src/native/x86_64/src/mlkem_poly_compress_d4_avx2_asm.S"+#include "src/native/x86_64/src/mlkem_poly_compress_d5_avx2_asm.S"+#include "src/native/x86_64/src/mlkem_poly_decompress_d10_avx2_asm.S"+#include "src/native/x86_64/src/mlkem_poly_decompress_d11_avx2_asm.S"+#include "src/native/x86_64/src/mlkem_poly_decompress_d4_avx2_asm.S"+#include "src/native/x86_64/src/mlkem_poly_decompress_d5_avx2_asm.S"+#include "src/native/x86_64/src/mlkem_poly_mulcache_compute_avx2_asm.S"+#include "src/native/x86_64/src/mlkem_polyvec_basemul_acc_montgomery_cached_k2_avx2_asm.S"+#include "src/native/x86_64/src/mlkem_polyvec_basemul_acc_montgomery_cached_k3_avx2_asm.S"+#include "src/native/x86_64/src/mlkem_polyvec_basemul_acc_montgomery_cached_k4_avx2_asm.S"+#include "src/native/x86_64/src/mlkem_reduce_avx2_asm.S"+#include "src/native/x86_64/src/mlkem_rej_uniform_avx2_asm.S"+#include "src/native/x86_64/src/mlkem_tomont_avx2_asm.S"+#endif /* MLK_SYS_X86_64 */+#if defined(MLK_SYS_RISCV64)+#endif+#if defined(MLK_SYS_PPC64LE)+#include "src/native/ppc64le/src/mlkem_intt_ppc_asm.S"+#include "src/native/ppc64le/src/mlkem_ntt_ppc_asm.S"+#include "src/native/ppc64le/src/mlkem_poly_tomont_ppc_asm.S"+#include "src/native/ppc64le/src/mlkem_reduce_ppc_asm.S"+#endif /* MLK_SYS_PPC64LE */+#endif /* MLK_CONFIG_USE_NATIVE_BACKEND_ARITH */++#if defined(MLK_CONFIG_USE_NATIVE_BACKEND_FIPS202)+#if defined(MLK_SYS_AARCH64)+#include "src/fips202/native/aarch64/src/keccak_f1600_x1_scalar_aarch64_asm.S"+#include "src/fips202/native/aarch64/src/keccak_f1600_x1_v84a_aarch64_asm.S"+#include "src/fips202/native/aarch64/src/keccak_f1600_x2_v84a_aarch64_asm.S"+#include "src/fips202/native/aarch64/src/keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm.S"+#include "src/fips202/native/aarch64/src/keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm.S"+#endif /* MLK_SYS_AARCH64 */+#if defined(MLK_SYS_X86_64)+#include "src/fips202/native/x86_64/src/keccak_f1600_x4_avx2_asm.S"+#endif+#if defined(MLK_SYS_ARMV81M_MVE)+#include "src/fips202/native/armv81m/src/keccak_f1600_x4_mve.S"+#include "src/fips202/native/armv81m/src/keccak_f1600_x4_state_extract_bytes_mve.S"+#include "src/fips202/native/armv81m/src/keccak_f1600_x4_state_xor_bytes_mve.S"+#endif+#endif /* MLK_CONFIG_USE_NATIVE_BACKEND_FIPS202 */+++/* Macro #undef's+ *+ * The following undefines macros from headers+ * included by the source files imported above.+ *+ * This is to allow building and linking multiple builds+ * of mlkem-native for varying parameter sets through concatenation+ * of this file, as if the files had been compiled separately.+ * If this is not relevant to you, you may remove the following.+ *+ * NOTE: This is not needed for the assembly SCU since, at present,+ * there is no need to include it multiple times.+ * We keep it for uniformity with mlkem_native.c only.+ *+ * NOTE: To avoid having to distinguish between which headers are included+ * from the assembly files, we #undef the same set of directives+ * as in mlkem_native.c+ */++/*+ * Undefine macros from MLK_CONFIG_PARAMETER_SET-specific files+ */+/* mlkem/mlkem_native.h */+#undef MLKEM1024_BYTES+#undef MLKEM1024_CIPHERTEXTBYTES+#undef MLKEM1024_PUBLICKEYBYTES+#undef MLKEM1024_SECRETKEYBYTES+#undef MLKEM1024_SYMBYTES+#undef MLKEM512_BYTES+#undef MLKEM512_CIPHERTEXTBYTES+#undef MLKEM512_PUBLICKEYBYTES+#undef MLKEM512_SECRETKEYBYTES+#undef MLKEM512_SYMBYTES+#undef MLKEM768_BYTES+#undef MLKEM768_CIPHERTEXTBYTES+#undef MLKEM768_PUBLICKEYBYTES+#undef MLKEM768_SECRETKEYBYTES+#undef MLKEM768_SYMBYTES+#undef MLKEM_BYTES+#undef MLKEM_CIPHERTEXTBYTES+#undef MLKEM_CIPHERTEXTBYTES_+#undef MLKEM_PUBLICKEYBYTES+#undef MLKEM_PUBLICKEYBYTES_+#undef MLKEM_SECRETKEYBYTES+#undef MLKEM_SECRETKEYBYTES_+#undef MLKEM_SYMBYTES+#undef MLK_API_CONCAT+#undef MLK_API_CONCAT_+#undef MLK_API_CONCAT_UNDERSCORE+#undef MLK_API_MUST_CHECK_RETURN_VALUE+#undef MLK_API_NAMESPACE+#undef MLK_API_NAMESPACE_PREFIX+#undef MLK_API_QUALIFIER+#undef MLK_ERR_FAIL+#undef MLK_ERR_INVALID_PK+#undef MLK_ERR_INVALID_SK+#undef MLK_ERR_OUT_OF_MEMORY+#undef MLK_ERR_PCT_FAIL+#undef MLK_ERR_RNG_FAIL+#undef MLK_H+#undef MLK_MAX3_+#undef MLK_TOTAL_ALLOC_1024+#undef MLK_TOTAL_ALLOC_1024_DECAPS+#undef MLK_TOTAL_ALLOC_1024_ENCAPS+#undef MLK_TOTAL_ALLOC_1024_KEYPAIR+#undef MLK_TOTAL_ALLOC_1024_KEYPAIR_NO_PCT+#undef MLK_TOTAL_ALLOC_1024_KEYPAIR_PCT+#undef MLK_TOTAL_ALLOC_512+#undef MLK_TOTAL_ALLOC_512_DECAPS+#undef MLK_TOTAL_ALLOC_512_ENCAPS+#undef MLK_TOTAL_ALLOC_512_KEYPAIR+#undef MLK_TOTAL_ALLOC_512_KEYPAIR_NO_PCT+#undef MLK_TOTAL_ALLOC_512_KEYPAIR_PCT+#undef MLK_TOTAL_ALLOC_768+#undef MLK_TOTAL_ALLOC_768_DECAPS+#undef MLK_TOTAL_ALLOC_768_ENCAPS+#undef MLK_TOTAL_ALLOC_768_KEYPAIR+#undef MLK_TOTAL_ALLOC_768_KEYPAIR_NO_PCT+#undef MLK_TOTAL_ALLOC_768_KEYPAIR_PCT+/* mlkem/src/common.h */+#undef MLK_ADD_PARAM_SET+#undef MLK_ALLOC+#undef MLK_APPLY+#undef MLK_ASM_FN_SIZE+#undef MLK_ASM_FN_SYMBOL+#undef MLK_ASM_NAMESPACE+#undef MLK_BUILD_INTERNAL+#undef MLK_COMMON_H+#undef MLK_CONCAT+#undef MLK_CONCAT_+#undef MLK_EMPTY_CU+#undef MLK_ERR_FAIL+#undef MLK_ERR_INVALID_PK+#undef MLK_ERR_INVALID_SK+#undef MLK_ERR_OUT_OF_MEMORY+#undef MLK_ERR_PCT_FAIL+#undef MLK_ERR_RNG_FAIL+#undef MLK_EXTERNAL_API+#undef MLK_FIPS202X4_HEADER_FILE+#undef MLK_FIPS202_HEADER_FILE+#undef MLK_FREE+#undef MLK_INTERNAL_API+#undef MLK_INTERNAL_DATA_DECLARATION+#undef MLK_INTERNAL_DATA_DEFINITION+#undef MLK_NAMESPACE+#undef MLK_NAMESPACE_K+#undef MLK_NAMESPACE_PREFIX+#undef MLK_NAMESPACE_PREFIX_K+#undef mlk_memcpy+#undef mlk_memset+/* mlkem/src/indcpa.h */+#undef MLK_INDCPA_H+#undef mlk_gen_matrix+#undef mlk_indcpa_dec+#undef mlk_indcpa_enc+#undef mlk_indcpa_keypair_derand+/* mlkem/src/kem.h */+#undef MLK_KEM_H+#undef mlk_kem_check_pk+#undef mlk_kem_check_sk+#undef mlk_kem_dec+#undef mlk_kem_enc+#undef mlk_kem_enc_derand+#undef mlk_kem_keypair+#undef mlk_kem_keypair_derand+/* mlkem/src/params.h */+#undef MLKEM_DU+#undef MLKEM_DV+#undef MLKEM_ETA1+#undef MLKEM_ETA2+#undef MLKEM_INDCCA_CIPHERTEXTBYTES+#undef MLKEM_INDCCA_PUBLICKEYBYTES+#undef MLKEM_INDCCA_SECRETKEYBYTES+#undef MLKEM_INDCPA_BYTES+#undef MLKEM_INDCPA_MSGBYTES+#undef MLKEM_INDCPA_PUBLICKEYBYTES+#undef MLKEM_INDCPA_SECRETKEYBYTES+#undef MLKEM_K+#undef MLKEM_N+#undef MLKEM_POLYBYTES+#undef MLKEM_POLYCOMPRESSEDBYTES_D10+#undef MLKEM_POLYCOMPRESSEDBYTES_D11+#undef MLKEM_POLYCOMPRESSEDBYTES_D4+#undef MLKEM_POLYCOMPRESSEDBYTES_D5+#undef MLKEM_POLYCOMPRESSEDBYTES_DU+#undef MLKEM_POLYCOMPRESSEDBYTES_DV+#undef MLKEM_POLYVECBYTES+#undef MLKEM_POLYVECCOMPRESSEDBYTES_DU+#undef MLKEM_Q+#undef MLKEM_Q_HALF+#undef MLKEM_SSBYTES+#undef MLKEM_SYMBYTES+#undef MLKEM_UINT12_LIMIT+#undef MLK_PARAMS_H+/* mlkem/src/poly_k.h */+#undef MLK_POLY_K_H+#undef mlk_poly_compress_du+#undef mlk_poly_compress_dv+#undef mlk_poly_decompress_du+#undef mlk_poly_decompress_dv+#undef mlk_poly_getnoise_eta1122_4x+#undef mlk_poly_getnoise_eta1_4x+#undef mlk_poly_getnoise_eta2+#undef mlk_poly_getnoise_eta2_4x+#undef mlk_polymat+#undef mlk_polyvec+#undef mlk_polyvec_add+#undef mlk_polyvec_basemul_acc_montgomery_cached+#undef mlk_polyvec_compress_du+#undef mlk_polyvec_decompress_du+#undef mlk_polyvec_frombytes+#undef mlk_polyvec_invntt_tomont+#undef mlk_polyvec_mulcache+#undef mlk_polyvec_mulcache_compute+#undef mlk_polyvec_ntt+#undef mlk_polyvec_reduce+#undef mlk_polyvec_tobytes+#undef mlk_polyvec_tomont++#if !defined(MLK_CONFIG_MONOBUILD_KEEP_SHARED_HEADERS)+/*+ * Undefine macros from MLK_CONFIG_PARAMETER_SET-generic files+ */+/* mlkem/src/compress.h */+#undef MLK_COMPRESS_H+#undef mlk_poly_compress_d10+#undef mlk_poly_compress_d11+#undef mlk_poly_compress_d4+#undef mlk_poly_compress_d5+#undef mlk_poly_decompress_d10+#undef mlk_poly_decompress_d11+#undef mlk_poly_decompress_d4+#undef mlk_poly_decompress_d5+#undef mlk_poly_frombytes+#undef mlk_poly_frommsg+#undef mlk_poly_tobytes+#undef mlk_poly_tomsg+/* mlkem/src/context.h */+#undef MLK_CONTEXT_H+#undef MLK_CONTEXT_PARAMETERS_0+#undef MLK_CONTEXT_PARAMETERS_1+#undef MLK_CONTEXT_PARAMETERS_2+#undef MLK_CONTEXT_PARAMETERS_3+#undef MLK_CONTEXT_PARAMETERS_4+#undef MLK_CONTEXT_UNUSED+/* mlkem/src/debug.h */+#undef MLK_DEBUG_H+#undef mlk_assert+#undef mlk_assert_abs_bound+#undef mlk_assert_abs_bound_2d+#undef mlk_assert_bound+#undef mlk_assert_bound_2d+#undef mlk_debug_check_assert+#undef mlk_debug_check_bounds+/* mlkem/src/poly.h */+#undef MLK_INVNTT_BOUND+#undef MLK_NTT_BOUND+#undef MLK_POLY_H+#undef mlk_poly_add+#undef mlk_poly_invntt_tomont+#undef mlk_poly_mulcache_compute+#undef mlk_poly_ntt+#undef mlk_poly_reduce+#undef mlk_poly_sub+#undef mlk_poly_tomont+/* mlkem/src/randombytes.h */+#undef MLK_RANDOMBYTES_H+/* mlkem/src/sampling.h */+#undef MLK_SAMPLING_H+#undef mlk_poly_cbd2+#undef mlk_poly_cbd3+#undef mlk_poly_rej_uniform+#undef mlk_poly_rej_uniform_x4+/* mlkem/src/symmetric.h */+#undef MLK_SYMMETRIC_H+#undef MLK_XOF_RATE+#undef mlk_hash_g+#undef mlk_hash_h+#undef mlk_hash_j+#undef mlk_prf_eta+#undef mlk_prf_eta1+#undef mlk_prf_eta1_x4+#undef mlk_prf_eta2+#undef mlk_xof_absorb+#undef mlk_xof_ctx+#undef mlk_xof_init+#undef mlk_xof_release+#undef mlk_xof_squeezeblocks+#undef mlk_xof_x4_absorb+#undef mlk_xof_x4_ctx+#undef mlk_xof_x4_init+#undef mlk_xof_x4_release+#undef mlk_xof_x4_squeezeblocks+/* mlkem/src/sys.h */+#undef MLK_ALIGN+#undef MLK_ALIGN_UP+#undef MLK_ALWAYS_INLINE+#undef MLK_CET_ENDBR+#undef MLK_CT_TESTING_DECLASSIFY+#undef MLK_CT_TESTING_SECRET+#undef MLK_DEFAULT_ALIGN+#undef MLK_HAVE_INLINE_ASM+#undef MLK_INLINE+#undef MLK_MUST_CHECK_RETURN_VALUE+#undef MLK_NOINLINE+#undef MLK_RESTRICT+#undef MLK_STATIC_TESTABLE+#undef MLK_SYSV_ABI+#undef MLK_SYSV_ABI_SUPPORTED+#undef MLK_SYS_AARCH64+#undef MLK_SYS_AARCH64_EB+#undef MLK_SYS_AARCH64_NEON+#undef MLK_SYS_APPLE+#undef MLK_SYS_ARMV81M_MVE+#undef MLK_SYS_BIG_ENDIAN+#undef MLK_SYS_H+#undef MLK_SYS_LINUX+#undef MLK_SYS_LITTLE_ENDIAN+#undef MLK_SYS_PPC64LE+#undef MLK_SYS_RISCV32+#undef MLK_SYS_RISCV64+#undef MLK_SYS_RISCV64_RVV+#undef MLK_SYS_WINDOWS+#undef MLK_SYS_X86_64+#undef MLK_SYS_X86_64_AVX2+/* mlkem/src/verify.h */+#undef MLK_USE_ASM_VALUE_BARRIER+#undef MLK_VERIFY_H+#undef mlk_ct_opt_blocker_u64+/* mlkem/src/cbmc.h */+#undef MLK_CBMC_H+#undef __contract__+#undef __loop__++#if !defined(MLK_CONFIG_FIPS202_CUSTOM_HEADER)+/*+ * Undefine macros from FIPS-202 files+ */+/* mlkem/src/fips202/fips202.h */+#undef FIPS202_X4_DEFAULT_IMPLEMENTATION+#undef MLK_FIPS202_FIPS202_H+#undef SHA3_256_HASHBYTES+#undef SHA3_256_RATE+#undef SHA3_384_RATE+#undef SHA3_512_HASHBYTES+#undef SHA3_512_RATE+#undef SHAKE128_RATE+#undef SHAKE256_RATE+#undef mlk_sha3_256+#undef mlk_sha3_512+#undef mlk_shake128_absorb_once+#undef mlk_shake128_init+#undef mlk_shake128_release+#undef mlk_shake128_squeezeblocks+#undef mlk_shake256+/* mlkem/src/fips202/fips202x4.h */+#undef MLK_FIPS202_FIPS202X4_H+#undef mlk_shake128x4_absorb_once+#undef mlk_shake128x4_init+#undef mlk_shake128x4_release+#undef mlk_shake128x4_squeezeblocks+#undef mlk_shake256x4+/* mlkem/src/fips202/keccakf1600.h */+#undef MLK_FIPS202_KECCAKF1600_H+#undef MLK_KECCAK_LANES+#undef MLK_KECCAK_WAY+#undef mlk_keccakf1600_extract_bytes+#undef mlk_keccakf1600_permute+#undef mlk_keccakf1600_xor_bytes+#undef mlk_keccakf1600x4_extract_bytes+#undef mlk_keccakf1600x4_permute+#undef mlk_keccakf1600x4_xor_bytes+#endif /* !MLK_CONFIG_FIPS202_CUSTOM_HEADER */++#if defined(MLK_CONFIG_USE_NATIVE_BACKEND_FIPS202)+/* mlkem/src/fips202/native/api.h */+#undef MLK_FIPS202_NATIVE_API_H+#undef MLK_NATIVE_FUNC_FALLBACK+#undef MLK_NATIVE_FUNC_SUCCESS+/* mlkem/src/fips202/native/auto.h */+#undef MLK_FIPS202_NATIVE_AUTO_H+#if defined(MLK_SYS_AARCH64)+/*+ * Undefine macros from native code (FIPS202, AArch64)+ */+/* mlkem/src/fips202/native/aarch64/auto.h */+#undef MLK_FIPS202_NATIVE_AARCH64_AUTO_H+/* mlkem/src/fips202/native/aarch64/src/fips202_native_aarch64.h */+#undef MLK_FIPS202_NATIVE_AARCH64_SRC_FIPS202_NATIVE_AARCH64_H+#undef mlk_keccak_f1600_x1_scalar_aarch64_asm+#undef mlk_keccak_f1600_x1_v84a_aarch64_asm+#undef mlk_keccak_f1600_x2_v84a_aarch64_asm+#undef mlk_keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm+#undef mlk_keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm+#undef mlk_keccakf1600_round_constants+/* mlkem/src/fips202/native/aarch64/x1_scalar.h */+#undef MLK_FIPS202_AARCH64_NEED_X1_SCALAR+#undef MLK_FIPS202_NATIVE_AARCH64_X1_SCALAR_H+#undef MLK_USE_NATIVE_FIPS202_X1+/* mlkem/src/fips202/native/aarch64/x1_v84a.h */+#undef MLK_FIPS202_AARCH64_NEED_X1_V84A+#undef MLK_FIPS202_NATIVE_AARCH64_X1_V84A_H+#undef MLK_USE_NATIVE_FIPS202_X1+/* mlkem/src/fips202/native/aarch64/x2_v84a.h */+#undef MLK_FIPS202_AARCH64_NEED_X2_V84A+#undef MLK_FIPS202_NATIVE_AARCH64_X2_V84A_H+#undef MLK_USE_NATIVE_FIPS202_X4+/* mlkem/src/fips202/native/aarch64/x4_v8a_scalar.h */+#undef MLK_FIPS202_AARCH64_NEED_X4_V8A_SCALAR_HYBRID+#undef MLK_FIPS202_NATIVE_AARCH64_X4_V8A_SCALAR_H+#undef MLK_USE_NATIVE_FIPS202_X4+/* mlkem/src/fips202/native/aarch64/x4_v8a_v84a_scalar.h */+#undef MLK_FIPS202_AARCH64_NEED_X4_V8A_V84A_SCALAR_HYBRID+#undef MLK_FIPS202_NATIVE_AARCH64_X4_V8A_V84A_SCALAR_H+#undef MLK_USE_NATIVE_FIPS202_X4+#endif /* MLK_SYS_AARCH64 */+#if defined(MLK_SYS_X86_64)+/*+ * Undefine macros from native code (FIPS202, x86_64)+ */+/* mlkem/src/fips202/native/x86_64/keccak_f1600_x4_avx2.h */+#undef MLK_FIPS202_NATIVE_X86_64_KECCAK_F1600_X4_AVX2_H+#undef MLK_FIPS202_X86_64_NEED_X4_AVX2+#undef MLK_USE_NATIVE_FIPS202_X4+/* mlkem/src/fips202/native/x86_64/src/fips202_native_x86_64.h */+#undef MLK_FIPS202_NATIVE_X86_64_SRC_FIPS202_NATIVE_X86_64_H+#undef mlk_keccak_f1600_x4_avx2_asm+#undef mlk_keccak_rho56+#undef mlk_keccak_rho8+#undef mlk_keccakf1600_round_constants+#endif /* MLK_SYS_X86_64 */+#if defined(MLK_SYS_ARMV81M_MVE)+/*+ * Undefine macros from native code (FIPS202, Armv8.1-M)+ */+/* mlkem/src/fips202/native/armv81m/mve.h */+#undef MLK_FIPS202_ARMV81M_NEED_X4+#undef MLK_FIPS202_NATIVE_ARMV81M+#undef MLK_FIPS202_NATIVE_ARMV81M_MVE_H+#undef MLK_USE_NATIVE_FIPS202_X4+#undef MLK_USE_NATIVE_FIPS202_X4_EXTRACT_BYTES+#undef MLK_USE_NATIVE_FIPS202_X4_XOR_BYTES+#undef mlk_keccak_f1600_x4_native_impl+/* mlkem/src/fips202/native/armv81m/src/fips202_native_armv81m.h */+#undef MLK_FIPS202_NATIVE_ARMV81M_SRC_FIPS202_NATIVE_ARMV81M_H+#undef mlk_keccak_f1600_x4_mve_asm+#undef mlk_keccak_f1600_x4_state_extract_bytes_asm+#undef mlk_keccak_f1600_x4_state_xor_bytes_asm+#undef mlk_keccakf1600_round_constants+#endif /* MLK_SYS_ARMV81M_MVE */+#endif /* MLK_CONFIG_USE_NATIVE_BACKEND_FIPS202 */+#if defined(MLK_CONFIG_USE_NATIVE_BACKEND_ARITH)+/* mlkem/src/native/api.h */+#undef MLK_INVNTT_BOUND+#undef MLK_NATIVE_API_H+#undef MLK_NATIVE_FUNC_FALLBACK+#undef MLK_NATIVE_FUNC_SUCCESS+#undef MLK_NTT_BOUND+/* mlkem/src/native/meta.h */+#undef MLK_NATIVE_META_H+#if defined(MLK_SYS_AARCH64)+/*+ * Undefine macros from native code (Arith, AArch64)+ */+/* mlkem/src/native/aarch64/meta.h */+#undef MLK_ARITH_BACKEND_AARCH64+#undef MLK_NATIVE_AARCH64_META_H+#undef MLK_USE_NATIVE_INTT+#undef MLK_USE_NATIVE_NTT+#undef MLK_USE_NATIVE_POLYVEC_BASEMUL_ACC_MONTGOMERY_CACHED+#undef MLK_USE_NATIVE_POLY_MULCACHE_COMPUTE+#undef MLK_USE_NATIVE_POLY_REDUCE+#undef MLK_USE_NATIVE_POLY_TOBYTES+#undef MLK_USE_NATIVE_POLY_TOMONT+#undef MLK_USE_NATIVE_REJ_UNIFORM+/* mlkem/src/native/aarch64/src/arith_native_aarch64.h */+#undef MLK_NATIVE_AARCH64_SRC_ARITH_NATIVE_AARCH64_H+#undef mlk_aarch64_invntt_zetas_layer12345+#undef mlk_aarch64_invntt_zetas_layer67+#undef mlk_aarch64_ntt_zetas_layer12345+#undef mlk_aarch64_ntt_zetas_layer67+#undef mlk_aarch64_zetas_mulcache_native+#undef mlk_aarch64_zetas_mulcache_twisted_native+#undef mlk_intt_aarch64_asm+#undef mlk_ntt_aarch64_asm+#undef mlk_poly_mulcache_compute_aarch64_asm+#undef mlk_poly_reduce_aarch64_asm+#undef mlk_poly_tobytes_aarch64_asm+#undef mlk_poly_tomont_aarch64_asm+#undef mlk_polyvec_basemul_acc_montgomery_cached_k2_aarch64_asm+#undef mlk_polyvec_basemul_acc_montgomery_cached_k3_aarch64_asm+#undef mlk_polyvec_basemul_acc_montgomery_cached_k4_aarch64_asm+#undef mlk_rej_uniform_aarch64_asm+#undef mlk_rej_uniform_table+#endif /* MLK_SYS_AARCH64 */+#if defined(MLK_SYS_X86_64)+/*+ * Undefine macros from native code (Arith, X86_64)+ */+/* mlkem/src/native/x86_64/meta.h */+#undef MLK_ARITH_BACKEND_X86_64_DEFAULT+#undef MLK_NATIVE_X86_64_META_H+#undef MLK_USE_NATIVE_INTT+#undef MLK_USE_NATIVE_NTT+#undef MLK_USE_NATIVE_NTT_CUSTOM_ORDER+#undef MLK_USE_NATIVE_POLYVEC_BASEMUL_ACC_MONTGOMERY_CACHED+#undef MLK_USE_NATIVE_POLY_COMPRESS_D10+#undef MLK_USE_NATIVE_POLY_COMPRESS_D11+#undef MLK_USE_NATIVE_POLY_COMPRESS_D4+#undef MLK_USE_NATIVE_POLY_COMPRESS_D5+#undef MLK_USE_NATIVE_POLY_DECOMPRESS_D10+#undef MLK_USE_NATIVE_POLY_DECOMPRESS_D11+#undef MLK_USE_NATIVE_POLY_DECOMPRESS_D4+#undef MLK_USE_NATIVE_POLY_DECOMPRESS_D5+#undef MLK_USE_NATIVE_POLY_FROMBYTES+#undef MLK_USE_NATIVE_POLY_MULCACHE_COMPUTE+#undef MLK_USE_NATIVE_POLY_REDUCE+#undef MLK_USE_NATIVE_POLY_TOBYTES+#undef MLK_USE_NATIVE_POLY_TOMONT+#undef MLK_USE_NATIVE_REJ_UNIFORM+/* mlkem/src/native/x86_64/src/arith_native_x86_64.h */+#undef MLK_AVX2_REJ_UNIFORM_BUFLEN+#undef MLK_NATIVE_X86_64_SRC_ARITH_NATIVE_X86_64_H+#undef mlk_invntt_avx2_asm+#undef mlk_ntt_avx2_asm+#undef mlk_nttfrombytes_avx2_asm+#undef mlk_ntttobytes_avx2_asm+#undef mlk_nttunpack_avx2_asm+#undef mlk_poly_compress_d10_avx2_asm+#undef mlk_poly_compress_d11_avx2_asm+#undef mlk_poly_compress_d4_avx2_asm+#undef mlk_poly_compress_d5_avx2_asm+#undef mlk_poly_decompress_d10_avx2_asm+#undef mlk_poly_decompress_d11_avx2_asm+#undef mlk_poly_decompress_d4_avx2_asm+#undef mlk_poly_decompress_d5_avx2_asm+#undef mlk_poly_mulcache_compute_avx2_asm+#undef mlk_polyvec_basemul_acc_montgomery_cached_k2_avx2_asm+#undef mlk_polyvec_basemul_acc_montgomery_cached_k3_avx2_asm+#undef mlk_polyvec_basemul_acc_montgomery_cached_k4_avx2_asm+#undef mlk_reduce_avx2_asm+#undef mlk_rej_uniform_avx2_asm+#undef mlk_rej_uniform_table+#undef mlk_tomont_avx2_asm+/* mlkem/src/native/x86_64/src/compress_consts.h */+#undef MLK_NATIVE_X86_64_SRC_COMPRESS_CONSTS_H+#undef mlk_compress_d10_data+#undef mlk_compress_d11_data+#undef mlk_compress_d4_data+#undef mlk_compress_d5_data+#undef mlk_decompress_d10_data+#undef mlk_decompress_d11_data+#undef mlk_decompress_d4_data+#undef mlk_decompress_d5_data+/* mlkem/src/native/x86_64/src/consts.h */+#undef MLK_AVX2_BACKEND_DATA_OFFSET_MULCACHE_TWIDDLES+#undef MLK_AVX2_BACKEND_DATA_OFFSET_REVIDXB+#undef MLK_AVX2_BACKEND_DATA_OFFSET_REVIDXD+#undef MLK_AVX2_BACKEND_DATA_OFFSET_ZETAS_EXP+#undef MLK_NATIVE_X86_64_SRC_CONSTS_H+#undef mlk_qdata+#endif /* MLK_SYS_X86_64 */+#if defined(MLK_SYS_RISCV64)+/*+ * Undefine macros from native code (Arith, RISC-V 64)+ */+/* mlkem/src/native/riscv64/meta.h */+#undef MLK_ARITH_BACKEND_RISCV64+#undef MLK_NATIVE_RISCV64_META_H+#undef MLK_USE_NATIVE_INTT+#undef MLK_USE_NATIVE_NTT+#undef MLK_USE_NATIVE_POLYVEC_BASEMUL_ACC_MONTGOMERY_CACHED+#undef MLK_USE_NATIVE_POLY_MULCACHE_COMPUTE+#undef MLK_USE_NATIVE_POLY_REDUCE+#undef MLK_USE_NATIVE_POLY_TOMONT+#undef MLK_USE_NATIVE_REJ_UNIFORM+/* mlkem/src/native/riscv64/src/arith_native_riscv64.h */+#undef MLK_NATIVE_RISCV64_SRC_ARITH_NATIVE_RISCV64_H+#undef mlk_rv64v_poly_add+#undef mlk_rv64v_poly_basemul_mont_add_k2+#undef mlk_rv64v_poly_basemul_mont_add_k3+#undef mlk_rv64v_poly_basemul_mont_add_k4+#undef mlk_rv64v_poly_invntt_tomont+#undef mlk_rv64v_poly_ntt+#undef mlk_rv64v_poly_reduce+#undef mlk_rv64v_poly_sub+#undef mlk_rv64v_poly_tomont+#undef mlk_rv64v_rej_uniform+/* mlkem/src/native/riscv64/src/rv64v_debug.h */+#undef MLK_NATIVE_RISCV64_SRC_RV64V_DEBUG_H+#undef mlk_assert_abs_bound_int16m1+#undef mlk_assert_abs_bound_int16m2+#undef mlk_assert_bound_int16m1+#undef mlk_assert_bound_int16m2+#undef mlk_debug_check_bounds_int16m1+#undef mlk_debug_check_bounds_int16m2+#endif /* MLK_SYS_RISCV64 */+#if defined(MLK_SYS_PPC64LE)+/*+ * Undefine macros from native code (Arith, PPC64LE)+ */+/* mlkem/src/native/ppc64le/meta.h */+#undef MLK_ARITH_BACKEND_NAME+#undef MLK_ARITH_BACKEND_PPC64LE_DEFAULT+#undef MLK_NATIVE_PPC64LE_META_H+#undef MLK_USE_NATIVE_INTT+#undef MLK_USE_NATIVE_NTT+#undef MLK_USE_NATIVE_POLY_REDUCE+#undef MLK_USE_NATIVE_POLY_TOMONT+/* mlkem/src/native/ppc64le/src/arith_native_ppc64le.h */+#undef MLK_NATIVE_PPC64LE_SRC_ARITH_NATIVE_PPC64LE_H+#undef mlk_intt_ppc_asm+#undef mlk_ntt_ppc_asm+#undef mlk_poly_tomont_ppc_asm+#undef mlk_reduce_ppc_asm+/* mlkem/src/native/ppc64le/src/consts.h */+#undef MLK_NATIVE_PPC64LE_SRC_CONSTS_H+#undef MLK_PPC_C20159_OFFSET+#undef MLK_PPC_NQ_OFFSET+#undef MLK_PPC_N_INV_OFFSET+#undef MLK_PPC_N_INV_TW_OFFSET+#undef MLK_PPC_Q_OFFSET+#undef MLK_PPC_TOMONT_OFFSET+#undef MLK_PPC_TOMONT_TW_OFFSET+#undef MLK_PPC_ZETA_INTT_OFFSET+#undef MLK_PPC_ZETA_INTT_TW_OFFSET+#undef MLK_PPC_ZETA_NTT_OFFSET+#undef MLK_PPC_ZETA_NTT_TW_OFFSET+#undef mlk_ppc_qdata+#endif /* MLK_SYS_PPC64LE */+#endif /* MLK_CONFIG_USE_NATIVE_BACKEND_ARITH */+#endif /* !MLK_CONFIG_MONOBUILD_KEEP_SHARED_HEADERS */
@@ -0,0 +1,683 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS140_3_IG]+ * Implementation Guidance for FIPS 140-3 and the Cryptographic Module+ * Validation Program+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/projects/cryptographic-module-validation-program/fips-140-3-ig-announcements+ *+ * - [FIPS203]+ * FIPS 203 Module-Lattice-Based Key-Encapsulation Mechanism Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/203/final+ */++#ifndef MLK_CONFIG_H+#define MLK_CONFIG_H++/**+ * Specifies the parameter set for ML-KEM:+ * - MLK_CONFIG_PARAMETER_SET=512 corresponds to ML-KEM-512+ * - MLK_CONFIG_PARAMETER_SET=768 corresponds to ML-KEM-768+ * - MLK_CONFIG_PARAMETER_SET=1024 corresponds to ML-KEM-1024+ *+ * If you want to support multiple parameter sets, build the library multiple+ * times and set MLK_CONFIG_MULTILEVEL_BUILD. See MLK_CONFIG_MULTILEVEL_BUILD+ * for how to do this while minimizing code duplication.+ *+ * This can also be set using CFLAGS.+ */+#ifndef MLK_CONFIG_PARAMETER_SET+#define MLK_CONFIG_PARAMETER_SET \+ 768 /* Change this for different security strengths */+#endif++/**+ * MLK_CONFIG_FILE+ *+ * If defined, this is a header that will be included instead of the default+ * configuration file mlkem/mlkem_native_config.h.+ *+ * When you need to build mlkem-native in multiple configurations, using+ * varying MLK_CONFIG_FILE can be more convenient than configuring everything+ * through CFLAGS.+ *+ * To use, MLK_CONFIG_FILE _must_ be defined prior to the inclusion of any+ * mlkem-native headers. For example, it can be set by passing+ * `-DMLK_CONFIG_FILE="..."` on the command line.+ */+/* #define MLK_CONFIG_FILE "mlkem_native_config.h" */++/**+ * The prefix to use to namespace global symbols from mlkem/.+ *+ * In a multi-level build, level-dependent symbols will additionally be+ * prefixed with the parameter set (512/768/1024).+ *+ * This can also be set using CFLAGS.+ */+#if !defined(MLK_CONFIG_NAMESPACE_PREFIX)+#define MLK_CONFIG_NAMESPACE_PREFIX MLK_DEFAULT_NAMESPACE_PREFIX+#endif++/**+ * MLK_CONFIG_MULTILEVEL_BUILD+ *+ * Set this if the build is part of a multi-level build supporting multiple+ * parameter sets.+ *+ * If you need only a single parameter set, keep this unset.+ *+ * To build mlkem-native with support for all parameter sets, build it three+ * times -- once per parameter set -- and set the option+ * MLK_CONFIG_MULTILEVEL_WITH_SHARED for exactly one of them, and+ * MLK_CONFIG_MULTILEVEL_NO_SHARED for the others.+ * MLK_CONFIG_MULTILEVEL_BUILD should be set for all of them.+ *+ * See examples/multilevel_build for an example.+ *+ * This can also be set using CFLAGS.+ */+/* #define MLK_CONFIG_MULTILEVEL_BUILD */++/**+ * MLK_CONFIG_EXTERNAL_API_QUALIFIER+ *+ * If set, this option provides an additional function qualifier to be added+ * to declarations of mlkem-native's public API.+ *+ * The primary use case for this option are single-CU builds where the public+ * API exposed by mlkem-native is wrapped by another API in the consuming+ * application. In this case, even mlkem-native's public API can be marked+ * `static`.+ */+/* #define MLK_CONFIG_EXTERNAL_API_QUALIFIER */++/**+ * MLK_CONFIG_NO_KEYPAIR_API+ *+ * By default, mlkem-native includes support for generating key pairs.+ * If you don't need this, set MLK_CONFIG_NO_KEYPAIR_API to exclude+ * keypair and keypair_derand, and all internal+ * APIs only needed by those functions.+ */+/* #define MLK_CONFIG_NO_KEYPAIR_API */++/**+ * MLK_CONFIG_NO_ENCAPS_API+ *+ * By default, mlkem-native includes support for encapsulation. If you+ * don't need this, set MLK_CONFIG_NO_ENCAPS_API to exclude+ * enc, enc_derand, check_pk, and+ * all internal APIs only needed by those functions.+ *+ * @note Setting this option is incompatible with MLK_CONFIG_KEYGEN_PCT+ * as the current PCT implementation requires enc().+ */+/* #define MLK_CONFIG_NO_ENCAPS_API */++/**+ * MLK_CONFIG_NO_DECAPS_API+ *+ * By default, mlkem-native includes support for decapsulation. If you+ * don't need this, set MLK_CONFIG_NO_DECAPS_API to exclude+ * dec, check_sk, and all internal APIs only+ * needed by those functions.+ *+ * @note Setting this option is incompatible with MLK_CONFIG_KEYGEN_PCT+ * as the current PCT implementation requires dec().+ */+/* #define MLK_CONFIG_NO_DECAPS_API */++/**+ * MLK_CONFIG_NO_RANDOMIZED_API+ *+ * If this option is set, mlkem-native will be built without the randomized+ * API functions (keypair and enc). This allows users+ * to build mlkem-native without providing a randombytes() implementation+ * if they only need the deterministic API (keypair_derand,+ * enc_derand, dec).+ *+ * @note This option is incompatible with MLK_CONFIG_KEYGEN_PCT as the+ * current PCT implementation requires enc().+ */+/* #define MLK_CONFIG_NO_RANDOMIZED_API */++/**+ * MLK_CONFIG_CONSTANTS_ONLY+ *+ * If you only need the size constants (MLKEM_PUBLICKEYBYTES, etc.) but no+ * function declarations, set MLK_CONFIG_CONSTANTS_ONLY.+ *+ * This only affects the public header mlkem_native.h, not the+ * implementation.+ */+/* #define MLK_CONFIG_CONSTANTS_ONLY */++/******************************************************************************+ *+ * Build-only configuration options+ *+ * The remaining configurations are build-options only.+ * They do not affect the API described in mlkem_native.h.+ *+ *****************************************************************************/++#if defined(MLK_BUILD_INTERNAL)+/**+ * MLK_CONFIG_MULTILEVEL_WITH_SHARED+ *+ * This is for multi-level builds of mlkem-native only. If you need only a+ * single parameter set, keep this unset.+ *+ * If this is set, all MLK_CONFIG_PARAMETER_SET-independent code will be+ * included in the build, including code needed only for other parameter+ * sets.+ *+ * Example: mlk_poly_cbd3 is only needed for MLK_CONFIG_PARAMETER_SET == 512.+ * Yet, if this option is set for a build with+ * MLK_CONFIG_PARAMETER_SET == 768/1024, it would be included.+ *+ * To build mlkem-native with support for all parameter sets, build it three+ * times -- once per parameter set -- and set the option+ * MLK_CONFIG_MULTILEVEL_WITH_SHARED for exactly one of them, and+ * MLK_CONFIG_MULTILEVEL_NO_SHARED for the others.+ * MLK_CONFIG_MULTILEVEL_BUILD should be set for all of them.+ *+ * See examples/multilevel_build for an example.+ *+ * This can also be set using CFLAGS.+ */+/* #define MLK_CONFIG_MULTILEVEL_WITH_SHARED */++/**+ * MLK_CONFIG_MULTILEVEL_NO_SHARED+ *+ * This is for multi-level builds of mlkem-native only. If you need only a+ * single parameter set, keep this unset.+ *+ * If this is set, no MLK_CONFIG_PARAMETER_SET-independent code will be+ * included in the build.+ *+ * To build mlkem-native with support for all parameter sets, build it three+ * times -- once per parameter set -- and set the option+ * MLK_CONFIG_MULTILEVEL_WITH_SHARED for exactly one of them, and+ * MLK_CONFIG_MULTILEVEL_NO_SHARED for the others.+ * MLK_CONFIG_MULTILEVEL_BUILD should be set for all of them.+ *+ * See examples/multilevel_build for an example.+ *+ * This can also be set using CFLAGS.+ */+/* #define MLK_CONFIG_MULTILEVEL_NO_SHARED */++/**+ * MLK_CONFIG_MONOBUILD_KEEP_SHARED_HEADERS+ *+ * This is only relevant for single compilation unit (SCU) builds of+ * mlkem-native. In this case, it determines whether directives defined in+ * parameter-set-independent headers should be #undef'ined or not at the+ * end of the SCU file. This is needed in multilevel builds.+ *+ * See examples/multilevel_build_native for an example.+ *+ * This can also be set using CFLAGS.+ */+/* #define MLK_CONFIG_MONOBUILD_KEEP_SHARED_HEADERS */++/**+ * MLK_CONFIG_USE_NATIVE_BACKEND_ARITH+ *+ * Determines whether a native arithmetic backend should be used.+ *+ * The arithmetic backend covers performance-critical functions such as the+ * number-theoretic transform (NTT).+ *+ * If this option is unset, the C backend will be used.+ *+ * If this option is set, the arithmetic backend to be used is determined+ * by MLK_CONFIG_ARITH_BACKEND_FILE: if the latter is unset, the default+ * backend for the target architecture will be used. If set, it must be the+ * name of a backend metadata file.+ *+ * This can also be set using CFLAGS.+ */+#if !defined(MLK_CONFIG_USE_NATIVE_BACKEND_ARITH)+/* #define MLK_CONFIG_USE_NATIVE_BACKEND_ARITH */+#endif++/**+ * MLK_CONFIG_ARITH_BACKEND_FILE+ *+ * The arithmetic backend to use.+ *+ * If MLK_CONFIG_USE_NATIVE_BACKEND_ARITH is unset, this option is ignored.+ *+ * If MLK_CONFIG_USE_NATIVE_BACKEND_ARITH is set, this option must either+ * be undefined or the filename of an arithmetic backend. If unset, the+ * default backend will be used.+ *+ * This can be set using CFLAGS.+ */+#if defined(MLK_CONFIG_USE_NATIVE_BACKEND_ARITH) && \+ !defined(MLK_CONFIG_ARITH_BACKEND_FILE)+#define MLK_CONFIG_ARITH_BACKEND_FILE "native/meta.h"+#endif++/**+ * MLK_CONFIG_USE_NATIVE_BACKEND_FIPS202+ *+ * Determines whether a native FIPS202 backend should be used.+ *+ * The FIPS202 backend covers 1x/2x/4x-fold Keccak-f1600, which is the+ * performance bottleneck of SHA3 and SHAKE.+ *+ * If this option is unset, the C backend will be used.+ *+ * If this option is set, the FIPS202 backend to be used is determined by+ * MLK_CONFIG_FIPS202_BACKEND_FILE: if the latter is unset, the default+ * backend for the target architecture will be used. If set, it must be+ * the name of a backend metadata file.+ *+ * This can also be set using CFLAGS.+ */+#if !defined(MLK_CONFIG_USE_NATIVE_BACKEND_FIPS202)+/* #define MLK_CONFIG_USE_NATIVE_BACKEND_FIPS202 */+#endif++/**+ * MLK_CONFIG_FIPS202_BACKEND_FILE+ *+ * The FIPS-202 backend to use.+ *+ * If MLK_CONFIG_USE_NATIVE_BACKEND_FIPS202 is set, this option must either+ * be undefined or the filename of a FIPS202 backend. If unset, the default+ * backend will be used.+ *+ * This can be set using CFLAGS.+ */+#if defined(MLK_CONFIG_USE_NATIVE_BACKEND_FIPS202) && \+ !defined(MLK_CONFIG_FIPS202_BACKEND_FILE)+#define MLK_CONFIG_FIPS202_BACKEND_FILE "fips202/native/auto.h"+#endif++/**+ * MLK_CONFIG_FIPS202_CUSTOM_HEADER+ *+ * Custom header to use for FIPS-202.+ *+ * This should only be set if you intend to use a custom FIPS-202+ * implementation, different from the one shipped with mlkem-native.+ *+ * If set, it must be the name of a file serving as the replacement for+ * mlkem/src/fips202/fips202.h, and exposing the same API (see FIPS202.md).+ */+/* #define MLK_CONFIG_FIPS202_CUSTOM_HEADER "SOME_FILE.h" */++/**+ * MLK_CONFIG_FIPS202X4_CUSTOM_HEADER+ *+ * Custom header to use for FIPS-202-X4.+ *+ * This should only be set if you intend to use a custom FIPS-202+ * implementation, different from the one shipped with mlkem-native.+ *+ * If set, it must be the name of a file serving as the replacement for+ * mlkem/src/fips202/fips202x4.h, and exposing the same API (see FIPS202.md).+ */+/* #define MLK_CONFIG_FIPS202X4_CUSTOM_HEADER "SOME_FILE.h" */++/**+ * MLK_CONFIG_CUSTOM_ZEROIZE+ *+ * In compliance with @[FIPS203, Section 3.3], mlkem-native zeroizes+ * intermediate buffers before returning from function calls. By default,+ * those buffers are allocated from the stack; if MLK_CONFIG_CUSTOM_ALLOC_FREE+ * is set, they are (mostly -- few exceptions remain at present) allocated from+ * the configured custom allocator.+ *+ * mlkem-native also zeroizes caller-owned output buffers as needed to uphold+ * the API convention that outputs be either unmodified or zeroized upon+ * failure.+ *+ * Set this option and define `mlk_zeroize` if you want to use a custom+ * method to zeroize intermediate and output buffers.+ *+ * The default implementation uses SecureZeroMemory on Windows and a+ * memset + compiler barrier otherwise. If neither of those is available on+ * the target platform, compilation will fail, and you will need to use+ * MLK_CONFIG_CUSTOM_ZEROIZE to provide a custom implementation of+ * `mlk_zeroize()`.+ *+ * @warning+ * The zeroization conducted by mlkem-native reduces the likelihood of data+ * leaking on the stack or custom allocators, but it does not eliminate it.+ * For example, the C standard makes no guarantee about where a compiler+ * allocates local structures and whether/where it makes copies of them.+ * Also, in addition to entire structures, there may also be potentially+ * exploitable leakage of individual values on the stack. If you need+ * bullet-proof zeroization of the stack, you need to consider additional+ * measures instead of what this feature provides. In this case, you can+ * set mlk_zeroize to a no-op. Note that in this case you are also responsible+ * for zeroizing output buffers upon failure.+ */+/* #define MLK_CONFIG_CUSTOM_ZEROIZE+ #if !defined(__ASSEMBLER__)+ #include <stdint.h>+ #include "src/sys.h"+ static MLK_INLINE void mlk_zeroize(void *ptr, size_t len)+ {+ ... your implementation ...+ }+ #endif+*/++/**+ * MLK_CONFIG_CUSTOM_RANDOMBYTES+ *+ * mlkem-native does not provide a secure randombytes implementation. Such+ * an implementation has to be provided by the consumer.+ *+ * If this option is not set, mlkem-native expects a function+ * int randombytes(uint8_t *out, size_t outlen). It is expected to return+ * zero on success, and non-zero on failure. In case of failure, the+ * top-level APIs will return an MLK_ERR_RNG_FAIL error code.+ *+ * Set this option and define `mlk_randombytes` (with the same signature+ * and behaviour) if you want to use a custom method to sample randombytes+ * with a different name or signature.+ */+/* #define MLK_CONFIG_CUSTOM_RANDOMBYTES+ #if !defined(__ASSEMBLER__)+ #include <stdint.h>+ #include "src/sys.h"+ static MLK_INLINE int mlk_randombytes(uint8_t *ptr, size_t len)+ {+ ... your implementation ...+ return 0;+ }+ #endif+*/++/**+ * MLK_CONFIG_CUSTOM_CAPABILITY_FUNC+ *+ * mlkem-native backends may rely on specific hardware features. Those+ * backends will only be included in an mlkem-native build if support for+ * the respective features is enabled at compile-time. However, when+ * building for a heterogeneous set of CPUs to run the resulting+ * binary/library on, feature detection at _runtime_ is needed to decide+ * whether a backend can be used or not.+ *+ * Set this option and define `mlk_sys_check_capability` if you want to+ * use a custom method to dispatch between implementations.+ *+ * If this option is not set, mlkem-native uses compile-time feature+ * detection only to decide which backend to use.+ *+ * If you compile mlkem-native on a system with different capabilities+ * than the system that the resulting binary/library will be run on, you+ * must use this option.+ */+/* #define MLK_CONFIG_CUSTOM_CAPABILITY_FUNC+ static MLK_INLINE int mlk_sys_check_capability(mlk_sys_cap cap)+ __contract__(+ ensures(return_value == 0 || return_value == 1)+ )+ {+ ... your implementation ...+ }+*/++/**+ * MLK_CONFIG_CUSTOM_ALLOC_FREE [EXPERIMENTAL]+ *+ * Set this option and define `MLK_CUSTOM_ALLOC` and `MLK_CUSTOM_FREE` if+ * you want to use custom allocation for large local structures or buffers.+ *+ * By default, all buffers/structures are allocated on the stack. If this+ * option is set, most of them will be allocated via MLK_CUSTOM_ALLOC.+ *+ * Parameters to MLK_CUSTOM_ALLOC:+ * - T* v: Target pointer to declare.+ * - T: Type of structure to be allocated.+ * - N: Number of elements to be allocated.+ *+ * Parameters to MLK_CUSTOM_FREE:+ * - T* v: Target pointer to free. May be NULL.+ * - T: Type of structure to be freed.+ * - N: Number of elements to be freed.+ *+ * @warning This option is experimental. Its scope, configuration and+ * function/macro signatures may change at any time. We expect a+ * stable API in a future version.+ *+ * @note Even if this option is set, some allocations further down the call+ * stack will still be made from the stack, consuming up to 3KB of+ * stack space. Those will likely be added to the scope of this+ * option in the future.+ *+ * @note MLK_CUSTOM_ALLOC need not guarantee a successful allocation nor+ * include error handling. Upon failure, the target pointer should+ * simply be set to NULL. The calling code will handle this case and+ * invoke MLK_CUSTOM_FREE.+ */+/* #define MLK_CONFIG_CUSTOM_ALLOC_FREE+ #if !defined(__ASSEMBLER__)+ #include <stdlib.h>+ #define MLK_CUSTOM_ALLOC(v, T, N) \+ T* (v) = (T *)aligned_alloc(MLK_DEFAULT_ALIGN, \+ MLK_ALIGN_UP(sizeof(T) * (N)))+ #define MLK_CUSTOM_FREE(v, T, N) free(v)+ #endif+*/++/**+ * MLK_CONFIG_CUSTOM_MEMCPY+ *+ * Set this option and define `mlk_memcpy` if you want to use a custom+ * method to copy memory instead of the standard library memcpy function.+ *+ * The custom implementation must have the same signature and behavior as+ * the standard memcpy function:+ * void *mlk_memcpy(void *dest, const void *src, size_t n)+ */+/* #define MLK_CONFIG_CUSTOM_MEMCPY+ #if !defined(__ASSEMBLER__)+ #include <stdint.h>+ #include "src/sys.h"+ static MLK_INLINE void *mlk_memcpy(void *dest, const void *src, size_t n)+ {+ ... your implementation ...+ }+ #endif+*/++/**+ * MLK_CONFIG_CUSTOM_MEMSET+ *+ * Set this option and define `mlk_memset` if you want to use a custom+ * method to set memory instead of the standard library memset function.+ *+ * The custom implementation must have the same signature and behavior as+ * the standard memset function:+ * void *mlk_memset(void *s, int c, size_t n)+ */+/* #define MLK_CONFIG_CUSTOM_MEMSET+ #if !defined(__ASSEMBLER__)+ #include <stdint.h>+ #include "src/sys.h"+ static MLK_INLINE void *mlk_memset(void *s, int c, size_t n)+ {+ ... your implementation ...+ }+ #endif+*/++/**+ * MLK_CONFIG_INTERNAL_API_QUALIFIER+ *+ * If set, this option provides an additional qualifier to be added to+ * declarations of internal API functions and data.+ *+ * The primary use case for this option are single-CU builds, in which case+ * this option can be set to `static`.+ */+/* #define MLK_CONFIG_INTERNAL_API_QUALIFIER */++/**+ * MLK_CONFIG_CT_TESTING_ENABLED+ *+ * If set, mlkem-native annotates data as secret/public using valgrind's+ * annotations VALGRIND_MAKE_MEM_UNDEFINED and VALGRIND_MAKE_MEM_DEFINED,+ * enabling various checks for secret-dependent control flow or+ * variable-time execution (depending on the exact version of valgrind+ * installed).+ */+/* #define MLK_CONFIG_CT_TESTING_ENABLED */++/**+ * MLK_CONFIG_NO_ASM+ *+ * If this option is set, mlkem-native will be built without use of native+ * code or inline assembly.+ *+ * By default, inline assembly is used to implement value barriers. Without+ * inline assembly, mlkem-native will use a global volatile 'opt blocker'+ * instead; see verify.h.+ *+ * Inline assembly is also used to implement a secure zeroization function+ * on non-Windows platforms. If this option is set and the target platform+ * is not Windows, you MUST set MLK_CONFIG_CUSTOM_ZEROIZE and provide a+ * custom zeroization function.+ *+ * If this option is set, MLK_CONFIG_USE_NATIVE_BACKEND_FIPS202 and+ * MLK_CONFIG_USE_NATIVE_BACKEND_ARITH will be ignored, and no native+ * backends will be used.+ */+/* #define MLK_CONFIG_NO_ASM */++/**+ * MLK_CONFIG_NO_ASM_VALUE_BARRIER+ *+ * If this option is set, mlkem-native will be built without use of native+ * code or inline assembly for value barriers.+ *+ * By default, inline assembly (if available) is used to implement value+ * barriers. Without inline assembly, mlkem-native will use a global+ * volatile 'opt blocker' instead; see verify.h.+ */+/* #define MLK_CONFIG_NO_ASM_VALUE_BARRIER */++/**+ * MLK_CONFIG_KEYGEN_PCT+ *+ * Compliance with @[FIPS140_3_IG, p.87] requires a Pairwise Consistency+ * Test (PCT) to be carried out on a freshly generated keypair before it+ * can be exported.+ *+ * Set this option if such a check should be implemented. In this case,+ * keypair_derand and keypair will return+ * MLK_ERR_PCT_FAIL if the PCT failed.+ *+ * @note This feature will drastically lower the performance of key+ * generation.+ */+/* #define MLK_CONFIG_KEYGEN_PCT */++/**+ * MLK_CONFIG_KEYGEN_PCT_BREAKAGE_TEST+ *+ * If this option is set, the user must provide a runtime function+ * `static inline int mlk_break_pct() { ... }` to indicate whether the PCT+ * should be made to fail.+ *+ * This option only has an effect if MLK_CONFIG_KEYGEN_PCT is set.+ */+/* #define MLK_CONFIG_KEYGEN_PCT_BREAKAGE_TEST+ #if !defined(__ASSEMBLER__)+ #include "src/sys.h"+ static MLK_INLINE int mlk_break_pct(void)+ {+ ... return 0/1 depending on whether PCT should be broken ...+ }+ #endif+*/++/**+ * MLK_CONFIG_SERIAL_FIPS202_ONLY+ *+ * Set this to use a FIPS202 implementation with global state that supports+ * only one active Keccak computation at a time (e.g. some hardware+ * accelerators).+ *+ * If this option is set, batched Keccak operations are disabled for+ * rejection sampling during matrix generation. Instead, matrix entries+ * will be generated one at a time.+ *+ * This allows offloading Keccak computations to a hardware accelerator+ * that holds only a single Keccak state locally, rather than requiring+ * support for batched (4x) Keccak states.+ *+ * @note Depending on the target CPU, disabling batched Keccak may reduce+ * performance when using software FIPS202 implementations. Only+ * enable this when you have to.+ */+/* #define MLK_CONFIG_SERIAL_FIPS202_ONLY */++/**+ * MLK_CONFIG_CONTEXT_PARAMETER+ *+ * Set this to add a context parameter that is provided to public API+ * functions and is then available in custom callbacks.+ *+ * The type of the context parameter is configured via+ * MLK_CONFIG_CONTEXT_PARAMETER_TYPE.+ */+/* #define MLK_CONFIG_CONTEXT_PARAMETER */++/**+ * MLK_CONFIG_CONTEXT_PARAMETER_TYPE+ *+ * Set this to define the type for the context parameter used by+ * MLK_CONFIG_CONTEXT_PARAMETER.+ *+ * This is only relevant if MLK_CONFIG_CONTEXT_PARAMETER is set.+ */+/* #define MLK_CONFIG_CONTEXT_PARAMETER_TYPE void* */++/************************* Config internals ********************************/++#endif /* MLK_BUILD_INTERNAL */++/* Default namespace+ *+ * Don't change this. If you need a different namespace, re-define+ * MLK_CONFIG_NAMESPACE_PREFIX above instead, and remove the following.+ *+ * The default MLKEM namespace is+ *+ * PQCP_MLKEM_NATIVE_MLKEM<LEVEL>_+ *+ * e.g., PQCP_MLKEM_NATIVE_MLKEM512_+ */++#if defined(MLK_CONFIG_MULTILEVEL_BUILD)+/* In a multi-level build the parameter set is appended by the namespacing+ * machinery, so the default prefix must not embed it. */+#define MLK_DEFAULT_NAMESPACE_PREFIX PQCP_MLKEM_NATIVE_MLKEM+#elif MLK_CONFIG_PARAMETER_SET == 512+#define MLK_DEFAULT_NAMESPACE_PREFIX PQCP_MLKEM_NATIVE_MLKEM512+#elif MLK_CONFIG_PARAMETER_SET == 768+#define MLK_DEFAULT_NAMESPACE_PREFIX PQCP_MLKEM_NATIVE_MLKEM768+#elif MLK_CONFIG_PARAMETER_SET == 1024+#define MLK_DEFAULT_NAMESPACE_PREFIX PQCP_MLKEM_NATIVE_MLKEM1024+#endif++#endif /* !MLK_CONFIG_H */
@@ -0,0 +1,222 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#ifndef MLK_CBMC_H+#define MLK_CBMC_H+/***************************************************+ * Basic replacements for __CPROVER_XXX contracts+ ***************************************************/+/*+ * The `__contract__` / `__loop__` annotation macros use a+ * leading-double-underscore spelling in line with other CBMC macros.+ * clang-tidy flags these as reserved identifiers; we suppress the diagnostic+ * at each definition site (NOLINT) rather than disabling the check globally,+ * so it stays active for the rest of the tree.+ */+#ifndef CBMC++/* clang-format off */+#define __contract__(x) /* NOLINT(bugprone-reserved-identifier,cert-dcl37-c,cert-dcl51-cpp) */+#define __loop__(x) /* NOLINT(bugprone-reserved-identifier,cert-dcl37-c,cert-dcl51-cpp) */+/* clang-format on */++#else /* !CBMC */+++/* clang-format off */+#define __contract__(x) x /* NOLINT(bugprone-reserved-identifier,cert-dcl37-c,cert-dcl51-cpp) */+#define __loop__(x) x /* NOLINT(bugprone-reserved-identifier,cert-dcl37-c,cert-dcl51-cpp) */+/* clang-format on */++/* https://diffblue.github.io/cbmc/contracts-assigns.html */+#define assigns(...) __CPROVER_assigns(__VA_ARGS__)++/* https://diffblue.github.io/cbmc/contracts-requires-ensures.html */+#define requires(...) __CPROVER_requires(__VA_ARGS__)+#define ensures(...) __CPROVER_ensures(__VA_ARGS__)+/* https://diffblue.github.io/cbmc/contracts-loops.html */+#define invariant(...) __CPROVER_loop_invariant(__VA_ARGS__)+#define decreases(...) __CPROVER_decreases(__VA_ARGS__)+/* cassert to avoid confusion with in-built assert */+#define cassert(x) __CPROVER_assert(x, "cbmc assertion failed")+#define assume(...) __CPROVER_assume(__VA_ARGS__)++/***************************************************+ * Macros for "expression" forms that may appear+ * _inside_ top-level contracts.+ ***************************************************/++/*+ * function return value - useful inside ensures+ * https://diffblue.github.io/cbmc/contracts-functions.html+ */+#define return_value (__CPROVER_return_value)++/*+ * assigns l-value targets+ * https://diffblue.github.io/cbmc/contracts-assigns.html+ */+#define object_whole(...) __CPROVER_object_whole(__VA_ARGS__)+#define memory_slice(...) __CPROVER_object_upto(__VA_ARGS__)++/*+ * Pointer-related predicates+ * https://diffblue.github.io/cbmc/contracts-memory-predicates.html+ */+#define memory_no_alias(...) __CPROVER_is_fresh(__VA_ARGS__)+#define readable(...) __CPROVER_r_ok(__VA_ARGS__)+#define writeable(...) __CPROVER_w_ok(__VA_ARGS__)++/* Maximum supported buffer size+ *+ * Larger buffers may be supported, but due to internal modeling constraints+ * in CBMC, the proofs of memory- and type-safety won't be able to run.+ *+ * If you find yourself in need for a buffer size larger than this,+ * please contact the maintainers, so we can prioritize work to relax+ * this somewhat artificial bound.+ */+#define MLK_MAX_BUFFER_SIZE (SIZE_MAX >> 12)++/*+ * History variables+ * https://diffblue.github.io/cbmc/contracts-history-variables.html+ */+#define old(...) __CPROVER_old(__VA_ARGS__)+#define loop_entry(...) __CPROVER_loop_entry(__VA_ARGS__)++/*+ * Quantifiers+ * Note that the range on qvar is _exclusive_ between qvar_lb .. qvar_ub+ * https://diffblue.github.io/cbmc/contracts-quantifiers.html+ *+ * The quantified variable is declared as uint32_t, so these macros+ * quantify only over indices in [0, UINT32_MAX). Bounds larger than+ * UINT32_MAX (4 GiB) are NOT supported: the explicit (uint32_t) casts+ * on the bounds will trigger CBMC's conversion check if a wider bound+ * (e.g. a size_t > UINT32_MAX) is passed.+ *+ * Quantifying over size_t (64-bit) was found to blow up SMT proof+ * times, so we deliberately keep the index width at 32 bits. Callers+ * dealing with size_t-typed buffers must add an explicit+ * requires(len <= UINT32_MAX)+ * precondition.+ */++/*+ * Prevent clang-format from corrupting CBMC's special ==> operator+ */+/* clang-format off */+#define forall(qvar, qvar_lb, qvar_ub, predicate) \+ __CPROVER_forall \+ { \+ uint32_t qvar; \+ ((uint32_t) (qvar_lb) <= (qvar) && (qvar) < (uint32_t) (qvar_ub)) ==> (predicate) \+ }++#define exists(qvar, qvar_lb, qvar_ub, predicate) \+ __CPROVER_exists \+ { \+ uint32_t qvar; \+ ((uint32_t) (qvar_lb) <= (qvar) && (qvar) < (uint32_t) (qvar_ub)) && (predicate) \+ }+/* clang-format on */++/***************************************************+ * Convenience macros for common contract patterns+ ***************************************************/++/*+ * Boolean-value predidate that asserts that "all values of array_var are in+ * range value_lb (inclusive) .. value_ub (exclusive)"+ * Example:+ * array_bound(a->coeffs, 0, MLKEM_N, 0, MLKEM_Q)+ * expands to+ * __CPROVER_forall { int k; (0 <= k && k <= MLKEM_N-1) ==> (+ * 0 <= a->coeffs[k]) && a->coeffs[k] < MLKEM_Q)) }+ */++/*+ * Prevent clang-format from corrupting CBMC's special ==> operator+ */+/* clang-format off */+#define CBMC_CONCAT_(left, right) left##right+#define CBMC_CONCAT(left, right) CBMC_CONCAT_(left, right)++#define array_bound_core(qvar, qvar_lb, qvar_ub, array_var, \+ value_lb, value_ub) \+ __CPROVER_forall \+ { \+ uint32_t qvar; \+ ((uint32_t) (qvar_lb) <= (qvar) && (qvar) < (uint32_t) (qvar_ub)) ==> \+ (((int)(value_lb) <= ((array_var)[(qvar)])) && \+ (((array_var)[(qvar)]) < (int)(value_ub))) \+ }++#define array_bound(array_var, qvar_lb, qvar_ub, value_lb, value_ub) \+ array_bound_core(CBMC_CONCAT(_cbmc_idx, __COUNTER__), (qvar_lb), \+ (qvar_ub), (array_var), (value_lb), (value_ub))++#define array_unchanged_core(qvar, qvar_lb, qvar_ub, array_var) \+ __CPROVER_forall \+ { \+ uint32_t qvar; \+ ((uint32_t) (qvar_lb) <= (qvar) && (qvar) < (uint32_t) (qvar_ub)) ==> \+ ((array_var)[(qvar)]) == (old(* (int16_t (*)[(qvar_ub)])(array_var)))[(qvar)] \+ }++#define array_unchanged(array_var, N) \+ array_unchanged_core(CBMC_CONCAT(_cbmc_idx, __COUNTER__), 0, (N), (array_var))++#define array_unchanged_u64_core(qvar, qvar_lb, qvar_ub, array_var) \+ __CPROVER_forall \+ { \+ uint32_t qvar; \+ ((uint32_t) (qvar_lb) <= (qvar) && (qvar) < (uint32_t) (qvar_ub)) ==> \+ ((array_var)[(qvar)]) == (old(* (uint64_t (*)[(qvar_ub)])(array_var)))[(qvar)] \+ }++#define array_unchanged_u64(array_var, N) \+ array_unchanged_u64_core(CBMC_CONCAT(_cbmc_idx, __COUNTER__), 0, (N), (array_var))++#define array_unchanged_u8_core(qvar, array_var, N) \+ forall(qvar, 0, (N), \+ ((array_var)[(qvar)]) == (old(* (uint8_t (*)[(N)])(array_var)))[(qvar)])++#define array_unchanged_u8(array_var, N) \+ array_unchanged_u8_core(CBMC_CONCAT(_cbmc_idx, __COUNTER__), (array_var), (N))++#define array_zeroized_u8_core(qvar, array_var, N) \+ forall(qvar, 0, (N), ((array_var)[(qvar)]) == 0)++#define array_zeroized_u8(array_var, N) \+ array_zeroized_u8_core(CBMC_CONCAT(_cbmc_idx, __COUNTER__), (array_var), (N))+/* clang-format on */++/*+ * Output-buffer discipline on failure, as documented in API-CONVENTIONS.md:+ * when a function fails, each caller-owned output buffer is left either+ * fully unchanged or fully zeroized -- never holding partially computed or+ * stale data that could be mistaken for a valid result.+ *+ * Note the disjunction is over the buffer as a whole: it is not enough for+ * each byte to be individually either unchanged or zero.+ */+#define array_unchanged_or_zeroized_u8(array_var, N) \+ (array_unchanged_u8((array_var), (N)) || array_zeroized_u8((array_var), (N)))++/* Wrapper around array_bound operating on absolute values.+ *+ * The absolute value bound `k` is exclusive.+ *+ * Note that since the lower bound in array_bound is inclusive, we have to+ * raise it by 1 here.+ */+#define array_abs_bound(arr, lb, ub, k) \+ array_bound((arr), (lb), (ub), -((int)(k)) + 1, (k))++#endif /* CBMC */++#endif /* !MLK_CBMC_H */
@@ -0,0 +1,296 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS140_3_IG]+ * Implementation Guidance for FIPS 140-3 and the Cryptographic Module+ * Validation Program+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/projects/cryptographic-module-validation-program/fips-140-3-ig-announcements+ *+ * - [FIPS203]+ * FIPS 203 Module-Lattice-Based Key-Encapsulation Mechanism Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/203/final+ */++#ifndef MLK_COMMON_H+#define MLK_COMMON_H++#ifndef __ASSEMBLER__+#include <stdint.h>+#endif++#define MLK_BUILD_INTERNAL++#if defined(MLK_CONFIG_FILE)+#include MLK_CONFIG_FILE+#else+#include "mlkem_native_config.h"+#endif++#include "params.h"+#include "sys.h"++/* Internal and public API have external linkage by default, but+ * this can be overwritten by the user, e.g. for single-CU builds. */+#if !defined(MLK_CONFIG_INTERNAL_API_QUALIFIER)+#define MLK_INTERNAL_API+#define MLK_INTERNAL_DATA_DECLARATION extern+#define MLK_INTERNAL_DATA_DEFINITION+#else+#define MLK_INTERNAL_API MLK_CONFIG_INTERNAL_API_QUALIFIER+#define MLK_INTERNAL_DATA_DECLARATION MLK_CONFIG_INTERNAL_API_QUALIFIER+#define MLK_INTERNAL_DATA_DEFINITION MLK_CONFIG_INTERNAL_API_QUALIFIER+#endif++#if !defined(MLK_CONFIG_EXTERNAL_API_QUALIFIER)+#define MLK_EXTERNAL_API+#else+#define MLK_EXTERNAL_API MLK_CONFIG_EXTERNAL_API_QUALIFIER+#endif++#define MLK_CONCAT_(x1, x2) x1##x2+#define MLK_CONCAT(x1, x2) MLK_CONCAT_(x1, x2)++#if (defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || \+ defined(MLK_CONFIG_MULTILEVEL_NO_SHARED))+#define MLK_ADD_PARAM_SET(s) MLK_CONCAT(s, MLK_CONFIG_PARAMETER_SET)+#else+#define MLK_ADD_PARAM_SET(s) s+#endif++#define MLK_NAMESPACE_PREFIX MLK_CONCAT(MLK_CONFIG_NAMESPACE_PREFIX, _)+#define MLK_NAMESPACE_PREFIX_K \+ MLK_CONCAT(MLK_ADD_PARAM_SET(MLK_CONFIG_NAMESPACE_PREFIX), _)++/* Functions are prefixed by MLK_CONFIG_NAMESPACE_PREFIX.+ *+ * If multiple parameter sets are used, functions depending on the parameter+ * set are additionally prefixed with 512/768/1024. See mlkem_native_config.h.+ *+ * Example: If MLK_CONFIG_NAMESPACE_PREFIX is mlkem, then+ * MLK_NAMESPACE_K(enc) becomes mlkem512_enc/mlkem768_enc/mlkem1024_enc.+ */+#define MLK_NAMESPACE(s) MLK_CONCAT(MLK_NAMESPACE_PREFIX, s)+#define MLK_NAMESPACE_K(s) MLK_CONCAT(MLK_NAMESPACE_PREFIX_K, s)++/* On Apple platforms, we need to emit leading underscore+ * in front of assembly symbols. We thus introduce a separate+ * namespace wrapper for ASM symbols. */+#if !defined(__APPLE__)+#define MLK_ASM_NAMESPACE(sym) MLK_NAMESPACE(sym)+#else+#define MLK_ASM_NAMESPACE(sym) MLK_CONCAT(_, MLK_NAMESPACE(sym))+#endif++/*+ * On X86_64 if control-flow protections (CET) are enabled (through+ * -fcf-protection=), we add an endbr64 instruction at every global function+ * label. See sys.h for more details+ */+#if defined(MLK_SYS_X86_64)+#define MLK_ASM_FN_SYMBOL(sym) MLK_ASM_NAMESPACE(sym) : MLK_CET_ENDBR+#elif defined(MLK_SYS_ARMV81M_MVE)+/* clang-format off */+#define MLK_ASM_FN_SYMBOL(sym) \+ .type MLK_ASM_NAMESPACE(sym), %function; \+ MLK_ASM_NAMESPACE(sym) :+/* clang-format on */+#else /* !MLK_SYS_X86_64 && MLK_SYS_ARMV81M_MVE */+#define MLK_ASM_FN_SYMBOL(sym) MLK_ASM_NAMESPACE(sym) :+#endif /* !MLK_SYS_X86_64 && !MLK_SYS_ARMV81M_MVE */++/*+ * Output the size of an assembly function.+ */+#if defined(__ELF__)+#define MLK_ASM_FN_SIZE(sym) \+ .size MLK_ASM_NAMESPACE(sym), .- MLK_ASM_NAMESPACE(sym)+#else+#define MLK_ASM_FN_SIZE(sym)+#endif++/* We aim to simplify the user's life by supporting builds where+ * all source files are included, even those that are not needed.+ * Those files are appropriately guarded and will be empty when unneeded.+ * The following is to avoid compilers complaining about this. */+#define MLK_EMPTY_CU(s) extern int MLK_NAMESPACE_K(empty_cu_##s);++/* MLK_CONFIG_NO_ASM takes precedence over MLK_USE_NATIVE_XXX */+#if defined(MLK_CONFIG_NO_ASM)+#undef MLK_CONFIG_USE_NATIVE_BACKEND_ARITH+#undef MLK_CONFIG_USE_NATIVE_BACKEND_FIPS202+#endif++#if defined(MLK_CONFIG_USE_NATIVE_BACKEND_ARITH) && \+ !defined(MLK_CONFIG_ARITH_BACKEND_FILE)+#error Bad configuration: MLK_CONFIG_USE_NATIVE_BACKEND_ARITH is set, but MLK_CONFIG_ARITH_BACKEND_FILE is not.+#endif++#if defined(MLK_CONFIG_USE_NATIVE_BACKEND_FIPS202) && \+ !defined(MLK_CONFIG_FIPS202_BACKEND_FILE)+#error Bad configuration: MLK_CONFIG_USE_NATIVE_BACKEND_FIPS202 is set, but MLK_CONFIG_FIPS202_BACKEND_FILE is not.+#endif++#if defined(MLK_CONFIG_NO_RANDOMIZED_API) && defined(MLK_CONFIG_KEYGEN_PCT)+#error Bad configuration: MLK_CONFIG_NO_RANDOMIZED_API is incompatible with MLK_CONFIG_KEYGEN_PCT as the current PCT implementation requires enc()+#endif++#if defined(MLK_CONFIG_NO_ENCAPS_API) && defined(MLK_CONFIG_KEYGEN_PCT)+#error Bad configuration: MLK_CONFIG_NO_ENCAPS_API is incompatible with MLK_CONFIG_KEYGEN_PCT as the current PCT implementation requires enc()+#endif++#if defined(MLK_CONFIG_NO_DECAPS_API) && defined(MLK_CONFIG_KEYGEN_PCT)+#error Bad configuration: MLK_CONFIG_NO_DECAPS_API is incompatible with MLK_CONFIG_KEYGEN_PCT as the current PCT implementation requires dec()+#endif++#if defined(MLK_CONFIG_USE_NATIVE_BACKEND_ARITH)+#include MLK_CONFIG_ARITH_BACKEND_FILE+/* Include to enforce consistency of API and implementation,+ * and conduct sanity checks on the backend.+ *+ * Keep this _after_ the inclusion of the backend; otherwise,+ * the sanity checks won't have an effect. */+#if defined(MLK_CHECK_APIS) && !defined(__ASSEMBLER__)+#include "native/api.h"+#endif+#endif /* MLK_CONFIG_USE_NATIVE_BACKEND_ARITH */++#if defined(MLK_CONFIG_USE_NATIVE_BACKEND_FIPS202)+#include MLK_CONFIG_FIPS202_BACKEND_FILE+/* Include to enforce consistency of API and implementation,+ * and conduct sanity checks on the backend.+ *+ * Keep this _after_ the inclusion of the backend; otherwise,+ * the sanity checks won't have an effect. */+#if defined(MLK_CHECK_APIS) && !defined(__ASSEMBLER__)+#include "fips202/native/api.h"+#endif+#endif /* MLK_CONFIG_USE_NATIVE_BACKEND_FIPS202 */++#if !defined(MLK_CONFIG_FIPS202_CUSTOM_HEADER)+#define MLK_FIPS202_HEADER_FILE "fips202/fips202.h"+#else+#define MLK_FIPS202_HEADER_FILE MLK_CONFIG_FIPS202_CUSTOM_HEADER+#endif++#if !defined(MLK_CONFIG_FIPS202X4_CUSTOM_HEADER)+#define MLK_FIPS202X4_HEADER_FILE "fips202/fips202x4.h"+#else+#define MLK_FIPS202X4_HEADER_FILE MLK_CONFIG_FIPS202X4_CUSTOM_HEADER+#endif++/* Standard library function replacements */+#if !defined(__ASSEMBLER__)+#if !defined(MLK_CONFIG_CUSTOM_MEMCPY)+#include <string.h>+#define mlk_memcpy memcpy+#endif++#if !defined(MLK_CONFIG_CUSTOM_MEMSET)+#include <string.h>+#define mlk_memset memset+#endif+++/* Allocation macros for large local structures+ *+ * MLK_ALLOC(v, T, N) declares T *v and attempts to point it to an T[N]+ * MLK_FREE(v, T, N) zeroizes and frees the allocation+ *+ * Default implementation uses stack allocation.+ * Can be overridden by setting the config option MLK_CONFIG_CUSTOM_ALLOC_FREE+ * and defining MLK_CUSTOM_ALLOC and MLK_CUSTOM_FREE.+ */+#if defined(MLK_CONFIG_CUSTOM_ALLOC_FREE) != \+ (defined(MLK_CUSTOM_ALLOC) && defined(MLK_CUSTOM_FREE))+#error Bad configuration: MLK_CONFIG_CUSTOM_ALLOC_FREE must be set together with MLK_CUSTOM_ALLOC and MLK_CUSTOM_FREE+#endif++/* Context-parameter machinery (MLK_CONTEXT_PARAMETERS_n and related config+ * checks). Kept in a separate, level-generic header for readability; included+ * here so it is available to the allocation macros below and to all consumers+ * of common.h. */+#include "context.h"++#if !defined(MLK_CONFIG_CUSTOM_ALLOC_FREE)+/* Default: stack allocation */++/* This is a declaration macro, not an expression macro: T is a type and v is+ * a declarator, neither of which can be wrapped in parentheses. The+ * bugprone-macro-parentheses diagnostic is therefore a false positive here. */+#define MLK_ALLOC(v, T, N, context) \+ MLK_ALIGN T mlk_alloc_##v[N]; \+ T *v = mlk_alloc_##v /* NOLINT(bugprone-macro-parentheses) */++/* The MLK_FREE macro body references mlk_zeroize(), which is declared in+ * verify.h. We deliberately do NOT include verify.h here: doing so would+ * create a circular dependency (verify.h includes common.h), and common.h+ * itself never calls mlk_zeroize() -- only the macro expansion does. Each+ * translation unit that uses MLK_FREE therefore includes verify.h directly. */+#define MLK_FREE(v, T, N, context) \+ do \+ { \+ MLK_CONTEXT_UNUSED(context); \+ mlk_zeroize(mlk_alloc_##v, sizeof(mlk_alloc_##v)); \+ (v) = NULL; \+ } while (0)++#else /* !MLK_CONFIG_CUSTOM_ALLOC_FREE */++/* Custom allocation */++/*+ * The indirection here is necessary to use MLK_CONTEXT_PARAMETERS_3 here.+ */+#define MLK_APPLY(f, args) f args++#define MLK_ALLOC(v, T, N, context) \+ MLK_APPLY(MLK_CUSTOM_ALLOC, MLK_CONTEXT_PARAMETERS_3(v, T, N, context))++#define MLK_FREE(v, T, N, context) \+ do \+ { \+ if (v != NULL) \+ { \+ mlk_zeroize(v, sizeof(T) * (N)); \+ MLK_APPLY(MLK_CUSTOM_FREE, MLK_CONTEXT_PARAMETERS_3(v, T, N, context)); \+ v = NULL; \+ } \+ } while (0)++#endif /* MLK_CONFIG_CUSTOM_ALLOC_FREE */++/****************************** Error codes ***********************************/++/* Generic failure condition. Currently not returned by any function;+ * reserved for failures that no more specific code covers. */+#define MLK_ERR_FAIL (-1)+/* An allocation failed. This can only happen if MLK_CONFIG_CUSTOM_ALLOC_FREE+ * is defined and the provided MLK_CUSTOM_ALLOC can fail. */+#define MLK_ERR_OUT_OF_MEMORY (-2)+/* An RNG failure occurred. Might be due to insufficient entropy or+ * system misconfiguration. */+#define MLK_ERR_RNG_FAIL (-3)+/* Public key validation failed: the @[FIPS203, Section 7.2, 'modulus check']+ * found a coefficient outside [0,q-1]. Returned by check_pk and by the+ * encapsulation API. */+#define MLK_ERR_INVALID_PK (-4)+/* Secret key validation failed: the @[FIPS203, Section 7.3, 'hash check']+ * found the embedded public key hash inconsistent. Returned by check_sk and+ * by the decapsulation API. */+#define MLK_ERR_INVALID_SK (-5)+/* The 'Pairwise Consistency Test' @[FIPS140_3_IG, p.87] and+ * @[FIPS203, Section 7.1, Pairwise Consistency] failed. Only possible when+ * MLK_CONFIG_KEYGEN_PCT is enabled; signals that the freshly generated key+ * pair failed its encaps/decaps self-test. */+#define MLK_ERR_PCT_FAIL (-6)++#endif /* !__ASSEMBLER__ */++#endif /* !MLK_COMMON_H */
@@ -0,0 +1,763 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS203]+ * FIPS 203 Module-Lattice-Based Key-Encapsulation Mechanism Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/203/final+ *+ * - [REF]+ * CRYSTALS-Kyber C reference implementation+ * Bos, Ducas, Kiltz, Lepoint, Lyubashevsky, Schanck, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/kyber/tree/main/ref+ */++#include "common.h"+#if !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED)+++#include "cbmc.h"+#include "compress.h"+#include "debug.h"+#include "verify.h"++#if (!defined(MLK_CONFIG_NO_ENCAPS_API) || \+ !defined(MLK_CONFIG_NO_DECAPS_API)) && \+ (defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_K == 2 || \+ MLKEM_K == 3)+/* Reference: `poly_compress()` in the reference implementation @[REF],+ * for ML-KEM-{512,768}.+ * - In contrast to the reference implementation, we assume+ * unsigned canonical coefficients here.+ * The reference implementation works with coefficients+ * in the range [-(MLKEM_Q-1), MLKEM_Q-1]. */+MLK_STATIC_TESTABLE void mlk_poly_compress_d4_c(+ uint8_t r[MLKEM_POLYCOMPRESSEDBYTES_D4], const mlk_poly *a)+__contract__(+ requires(memory_no_alias(r, MLKEM_POLYCOMPRESSEDBYTES_D4))+ requires(memory_no_alias(a, sizeof(mlk_poly)))+ requires(array_bound(a->coeffs, 0, MLKEM_N, 0, MLKEM_Q))+ assigns(memory_slice(r, MLKEM_POLYCOMPRESSEDBYTES_D4))+)+{+ unsigned i;+ mlk_assert_bound(a, MLKEM_N, 0, MLKEM_Q);++ for (i = 0; i < MLKEM_N / 8; i++)+ __loop__(invariant(i <= MLKEM_N / 8)+ decreases(MLKEM_N / 8 - i))+ {+ unsigned j;+ uint8_t t[8] = {0};+ for (j = 0; j < 8; j++)+ __loop__(+ invariant(i <= MLKEM_N / 8 && j <= 8)+ invariant(array_bound(t, 0, j, 0, 16))+ decreases(8 - j))+ {+ t[j] = mlk_scalar_compress_d4(a->coeffs[8 * i + j]);+ }++ /* All t[i] are 4-bit wide, so the truncations don't alter the value. */+ r[i * 4] = (uint8_t)(t[0] | (t[1] << 4));+ r[i * 4 + 1] = (uint8_t)(t[2] | (t[3] << 4));+ r[i * 4 + 2] = (uint8_t)(t[4] | (t[5] << 4));+ r[i * 4 + 3] = (uint8_t)(t[6] | (t[7] << 4));+ }+}++MLK_INTERNAL_API+void mlk_poly_compress_d4(uint8_t r[MLKEM_POLYCOMPRESSEDBYTES_D4],+ const mlk_poly *a)+__contract__(+ requires(memory_no_alias(r, MLKEM_POLYCOMPRESSEDBYTES_D4))+ requires(memory_no_alias(a, sizeof(mlk_poly)))+ requires(array_bound(a->coeffs, 0, MLKEM_N, 0, MLKEM_Q))+ assigns(memory_slice(r, MLKEM_POLYCOMPRESSEDBYTES_D4))+)+{+#if defined(MLK_USE_NATIVE_POLY_COMPRESS_D4)+ int ret;+ mlk_assert_bound(a, MLKEM_N, 0, MLKEM_Q);+ ret = mlk_poly_compress_d4_native(r, a->coeffs);+ if (ret == MLK_NATIVE_FUNC_SUCCESS)+ {+ return;+ }+#endif /* MLK_USE_NATIVE_POLY_COMPRESS_D4 */++ mlk_poly_compress_d4_c(r, a);+}++/* Reference: Embedded into `polyvec_compress()` in the+ * reference implementation, for ML-KEM-{512,768}.+ * - In contrast to the reference implementation, we assume+ * unsigned canonical coefficients here.+ * The reference implementation works with coefficients+ * in the range [-(MLKEM_Q-1), MLKEM_Q-1]. */+MLK_STATIC_TESTABLE void mlk_poly_compress_d10_c(+ uint8_t r[MLKEM_POLYCOMPRESSEDBYTES_D10], const mlk_poly *a)+__contract__(+ requires(memory_no_alias(r, MLKEM_POLYCOMPRESSEDBYTES_D10))+ requires(memory_no_alias(a, sizeof(mlk_poly)))+ requires(array_bound(a->coeffs, 0, MLKEM_N, 0, MLKEM_Q))+ assigns(memory_slice(r, MLKEM_POLYCOMPRESSEDBYTES_D10))+)+{+ unsigned j;+ mlk_assert_bound(a, MLKEM_N, 0, MLKEM_Q);+ for (j = 0; j < MLKEM_N / 4; j++)+ __loop__(invariant(j <= MLKEM_N / 4)+ decreases(MLKEM_N / 4 - j))+ {+ unsigned k;+ uint16_t t[4];+ for (k = 0; k < 4; k++)+ __loop__(+ invariant(k <= 4)+ invariant(forall(r, 0, k, t[r] < (1u << 10)))+ decreases(4 - k))+ {+ t[k] = mlk_scalar_compress_d10(a->coeffs[4 * j + k]);+ }++ /*+ * Make all implicit truncation explicit. No data is being+ * truncated for the LHS's since each t[i] is 10-bit in size.+ */+ r[5 * j + 0] = (uint8_t)((t[0] >> 0) & 0xFF);+ r[5 * j + 1] = (uint8_t)((t[0] >> 8) | ((t[1] << 2) & 0xFF));+ r[5 * j + 2] = (uint8_t)((t[1] >> 6) | ((t[2] << 4) & 0xFF));+ r[5 * j + 3] = (uint8_t)((t[2] >> 4) | ((t[3] << 6) & 0xFF));+ r[5 * j + 4] = (uint8_t)(t[3] >> 2);+ }+}++MLK_INTERNAL_API+void mlk_poly_compress_d10(uint8_t r[MLKEM_POLYCOMPRESSEDBYTES_D10],+ const mlk_poly *a)+__contract__(+ requires(memory_no_alias(r, MLKEM_POLYCOMPRESSEDBYTES_D10))+ requires(memory_no_alias(a, sizeof(mlk_poly)))+ requires(array_bound(a->coeffs, 0, MLKEM_N, 0, MLKEM_Q))+ assigns(memory_slice(r, MLKEM_POLYCOMPRESSEDBYTES_D10))+)+{+#if defined(MLK_USE_NATIVE_POLY_COMPRESS_D10)+ int ret;+ mlk_assert_bound(a, MLKEM_N, 0, MLKEM_Q);+ ret = mlk_poly_compress_d10_native(r, a->coeffs);+ if (ret == MLK_NATIVE_FUNC_SUCCESS)+ {+ return;+ }+#endif /* MLK_USE_NATIVE_POLY_COMPRESS_D10 */++ mlk_poly_compress_d10_c(r, a);+}++#if !defined(MLK_CONFIG_NO_DECAPS_API)+/* Reference: `poly_decompress()` in the reference implementation @[REF],+ * for ML-KEM-{512,768}. */+MLK_STATIC_TESTABLE void mlk_poly_decompress_d4_c(+ mlk_poly *r, const uint8_t a[MLKEM_POLYCOMPRESSEDBYTES_D4])+__contract__(+ requires(memory_no_alias(r, sizeof(mlk_poly)))+ requires(memory_no_alias(a, MLKEM_POLYCOMPRESSEDBYTES_D4))+ assigns(memory_slice(r, sizeof(mlk_poly)))+ ensures(array_bound(r->coeffs, 0, MLKEM_N, 0, MLKEM_Q))+)+{+ unsigned i;+ for (i = 0; i < MLKEM_N / 2; i++)+ __loop__(+ invariant(i <= MLKEM_N / 2)+ invariant(array_bound(r->coeffs, 0, 2 * i, 0, MLKEM_Q))+ decreases(MLKEM_N / 2 - i))+ {+ r->coeffs[2 * i + 0] = mlk_scalar_decompress_d4((a[i] >> 0) & 0xF);+ r->coeffs[2 * i + 1] = mlk_scalar_decompress_d4((a[i] >> 4) & 0xF);+ }++ mlk_assert_bound(r, MLKEM_N, 0, MLKEM_Q);+}++MLK_INTERNAL_API+void mlk_poly_decompress_d4(mlk_poly *r,+ const uint8_t a[MLKEM_POLYCOMPRESSEDBYTES_D4])+__contract__(+ requires(memory_no_alias(r, sizeof(mlk_poly)))+ requires(memory_no_alias(a, MLKEM_POLYCOMPRESSEDBYTES_D4))+ assigns(memory_slice(r, sizeof(mlk_poly)))+ ensures(array_bound(r->coeffs, 0, MLKEM_N, 0, MLKEM_Q))+)+{+#if defined(MLK_USE_NATIVE_POLY_DECOMPRESS_D4)+ int ret;+ ret = mlk_poly_decompress_d4_native(r->coeffs, a);+ if (ret == MLK_NATIVE_FUNC_SUCCESS)+ {+ mlk_assert_bound(r, MLKEM_N, 0, MLKEM_Q);+ return;+ }+#endif /* MLK_USE_NATIVE_POLY_DECOMPRESS_D4 */++ mlk_poly_decompress_d4_c(r, a);+}++/* Reference: Embedded into `polyvec_decompress()` in the+ * reference implementation, for ML-KEM-{512,768}. */+MLK_STATIC_TESTABLE void mlk_poly_decompress_d10_c(+ mlk_poly *r, const uint8_t a[MLKEM_POLYCOMPRESSEDBYTES_D10])+__contract__(+ requires(memory_no_alias(r, sizeof(mlk_poly)))+ requires(memory_no_alias(a, MLKEM_POLYCOMPRESSEDBYTES_D10))+ assigns(memory_slice(r, sizeof(mlk_poly)))+ ensures(array_bound(r->coeffs, 0, MLKEM_N, 0, MLKEM_Q))+)+{+ unsigned j;+ for (j = 0; j < MLKEM_N / 4; j++)+ __loop__(+ invariant(j <= MLKEM_N / 4)+ invariant(array_bound(r->coeffs, 0, 4 * j, 0, MLKEM_Q))+ decreases(MLKEM_N / 4 - j))+ {+ unsigned k;+ uint16_t t[4];+ uint8_t const *base = &a[5 * j];++ t[0] = 0x3FF & ((base[0] >> 0) | ((uint16_t)base[1] << 8));+ t[1] = 0x3FF & ((base[1] >> 2) | ((uint16_t)base[2] << 6));+ t[2] = 0x3FF & ((base[2] >> 4) | ((uint16_t)base[3] << 4));+ t[3] = 0x3FF & ((base[3] >> 6) | ((uint16_t)base[4] << 2));++ for (k = 0; k < 4; k++)+ __loop__(+ invariant(k <= 4)+ invariant(array_bound(r->coeffs, 0, 4 * j + k, 0, MLKEM_Q))+ decreases(4 - k))+ {+ r->coeffs[4 * j + k] = mlk_scalar_decompress_d10(t[k]);+ }+ }++ mlk_assert_bound(r, MLKEM_N, 0, MLKEM_Q);+}++MLK_INTERNAL_API+void mlk_poly_decompress_d10(mlk_poly *r,+ const uint8_t a[MLKEM_POLYCOMPRESSEDBYTES_D10])+__contract__(+ requires(memory_no_alias(r, sizeof(mlk_poly)))+ requires(memory_no_alias(a, MLKEM_POLYCOMPRESSEDBYTES_D10))+ assigns(memory_slice(r, sizeof(mlk_poly)))+ ensures(array_bound(r->coeffs, 0, MLKEM_N, 0, MLKEM_Q))+)+{+#if defined(MLK_USE_NATIVE_POLY_DECOMPRESS_D10)+ int ret;+ ret = mlk_poly_decompress_d10_native(r->coeffs, a);+ if (ret == MLK_NATIVE_FUNC_SUCCESS)+ {+ mlk_assert_bound(r, MLKEM_N, 0, MLKEM_Q);+ return;+ }+#endif /* MLK_USE_NATIVE_POLY_DECOMPRESS_D10 */++ mlk_poly_decompress_d10_c(r, a);+}+#endif /* !MLK_CONFIG_NO_DECAPS_API */+#endif /* (!MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API) && \+ (MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 2 || MLKEM_K == 3) \+ */++#if (!defined(MLK_CONFIG_NO_ENCAPS_API) || \+ !defined(MLK_CONFIG_NO_DECAPS_API)) && \+ (defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_K == 4)+/* Reference: `poly_compress()` in the reference implementation @[REF],+ * for ML-KEM-1024.+ * - In contrast to the reference implementation, we assume+ * unsigned canonical coefficients here.+ * The reference implementation works with coefficients+ * in the range [-(MLKEM_Q-1), MLKEM_Q-1]. */+MLK_STATIC_TESTABLE void mlk_poly_compress_d5_c(+ uint8_t r[MLKEM_POLYCOMPRESSEDBYTES_D5], const mlk_poly *a)+__contract__(+ requires(memory_no_alias(r, MLKEM_POLYCOMPRESSEDBYTES_D5))+ requires(memory_no_alias(a, sizeof(mlk_poly)))+ requires(array_bound(a->coeffs, 0, MLKEM_N, 0, MLKEM_Q))+ assigns(memory_slice(r, MLKEM_POLYCOMPRESSEDBYTES_D5))+)+{+ unsigned i;+ mlk_assert_bound(a, MLKEM_N, 0, MLKEM_Q);++ for (i = 0; i < MLKEM_N / 8; i++)+ __loop__(invariant(i <= MLKEM_N / 8)+ decreases(MLKEM_N / 8 - i))+ {+ unsigned j;+ uint8_t t[8] = {0};+ for (j = 0; j < 8; j++)+ __loop__(+ invariant(i <= MLKEM_N / 8 && j <= 8)+ invariant(array_bound(t, 0, j, 0, 32))+ decreases(8 - j))+ {+ t[j] = mlk_scalar_compress_d5(a->coeffs[8 * i + j]);+ }++ r[i * 5] = (uint8_t)(0xFF & ((t[0] >> 0) | (t[1] << 5)));+ r[i * 5 + 1] = (uint8_t)(0xFF & ((t[1] >> 3) | (t[2] << 2) | (t[3] << 7)));+ r[i * 5 + 2] = (uint8_t)(0xFF & ((t[3] >> 1) | (t[4] << 4)));+ r[i * 5 + 3] = (uint8_t)(0xFF & ((t[4] >> 4) | (t[5] << 1) | (t[6] << 6)));+ r[i * 5 + 4] = (uint8_t)(0xFF & ((t[6] >> 2) | (t[7] << 3)));+ }+}++MLK_INTERNAL_API+void mlk_poly_compress_d5(uint8_t r[MLKEM_POLYCOMPRESSEDBYTES_D5],+ const mlk_poly *a)+__contract__(+ requires(memory_no_alias(r, MLKEM_POLYCOMPRESSEDBYTES_D5))+ requires(memory_no_alias(a, sizeof(mlk_poly)))+ requires(array_bound(a->coeffs, 0, MLKEM_N, 0, MLKEM_Q))+ assigns(memory_slice(r, MLKEM_POLYCOMPRESSEDBYTES_D5))+)+{+#if defined(MLK_USE_NATIVE_POLY_COMPRESS_D5)+ int ret;+ mlk_assert_bound(a, MLKEM_N, 0, MLKEM_Q);+ ret = mlk_poly_compress_d5_native(r, a->coeffs);+ if (ret == MLK_NATIVE_FUNC_SUCCESS)+ {+ return;+ }+#endif /* MLK_USE_NATIVE_POLY_COMPRESS_D5 */++ mlk_poly_compress_d5_c(r, a);+}++/* Reference: Embedded into `polyvec_compress()` in the+ * reference implementation, for ML-KEM-1024.+ * - In contrast to the reference implementation, we assume+ * unsigned canonical coefficients here.+ * The reference implementation works with coefficients+ * in the range [-(MLKEM_Q-1), MLKEM_Q-1]. */+MLK_STATIC_TESTABLE void mlk_poly_compress_d11_c(+ uint8_t r[MLKEM_POLYCOMPRESSEDBYTES_D11], const mlk_poly *a)+__contract__(+ requires(memory_no_alias(r, MLKEM_POLYCOMPRESSEDBYTES_D11))+ requires(memory_no_alias(a, sizeof(mlk_poly)))+ requires(array_bound(a->coeffs, 0, MLKEM_N, 0, MLKEM_Q))+ assigns(memory_slice(r, MLKEM_POLYCOMPRESSEDBYTES_D11))+)+{+ unsigned j;+ mlk_assert_bound(a, MLKEM_N, 0, MLKEM_Q);++ for (j = 0; j < MLKEM_N / 8; j++)+ __loop__(invariant(j <= MLKEM_N / 8)+ decreases(MLKEM_N / 8 - j))+ {+ unsigned k;+ uint16_t t[8];+ for (k = 0; k < 8; k++)+ __loop__(+ invariant(k <= 8)+ invariant(forall(r, 0, k, t[r] < (1u << 11)))+ decreases(8 - k))+ {+ t[k] = mlk_scalar_compress_d11(a->coeffs[8 * j + k]);+ }++ /*+ * Make all implicit truncation explicit. No data is being+ * truncated for the LHS's since each t[i] is 11-bit in size.+ */+ r[11 * j + 0] = (uint8_t)((t[0] >> 0) & 0xFF);+ r[11 * j + 1] = (uint8_t)((t[0] >> 8) | ((t[1] << 3) & 0xFF));+ r[11 * j + 2] = (uint8_t)((t[1] >> 5) | ((t[2] << 6) & 0xFF));+ r[11 * j + 3] = (uint8_t)((t[2] >> 2) & 0xFF);+ r[11 * j + 4] = (uint8_t)((t[2] >> 10) | ((t[3] << 1) & 0xFF));+ r[11 * j + 5] = (uint8_t)((t[3] >> 7) | ((t[4] << 4) & 0xFF));+ r[11 * j + 6] = (uint8_t)((t[4] >> 4) | ((t[5] << 7) & 0xFF));+ r[11 * j + 7] = (uint8_t)((t[5] >> 1) & 0xFF);+ r[11 * j + 8] = (uint8_t)((t[5] >> 9) | ((t[6] << 2) & 0xFF));+ r[11 * j + 9] = (uint8_t)((t[6] >> 6) | ((t[7] << 5) & 0xFF));+ r[11 * j + 10] = (uint8_t)(t[7] >> 3);+ }+}++MLK_INTERNAL_API+void mlk_poly_compress_d11(uint8_t r[MLKEM_POLYCOMPRESSEDBYTES_D11],+ const mlk_poly *a)+__contract__(+ requires(memory_no_alias(r, MLKEM_POLYCOMPRESSEDBYTES_D11))+ requires(memory_no_alias(a, sizeof(mlk_poly)))+ requires(array_bound(a->coeffs, 0, MLKEM_N, 0, MLKEM_Q))+ assigns(memory_slice(r, MLKEM_POLYCOMPRESSEDBYTES_D11))+)+{+#if defined(MLK_USE_NATIVE_POLY_COMPRESS_D11)+ int ret;+ mlk_assert_bound(a, MLKEM_N, 0, MLKEM_Q);+ ret = mlk_poly_compress_d11_native(r, a->coeffs);+ if (ret == MLK_NATIVE_FUNC_SUCCESS)+ {+ return;+ }+#endif /* MLK_USE_NATIVE_POLY_COMPRESS_D11 */++ mlk_poly_compress_d11_c(r, a);+}++#if !defined(MLK_CONFIG_NO_DECAPS_API)+/* Reference: `poly_decompress()` in the reference implementation @[REF],+ * for ML-KEM-1024. */+MLK_STATIC_TESTABLE void mlk_poly_decompress_d5_c(+ mlk_poly *r, const uint8_t a[MLKEM_POLYCOMPRESSEDBYTES_D5])+__contract__(+ requires(memory_no_alias(r, sizeof(mlk_poly)))+ requires(memory_no_alias(a, MLKEM_POLYCOMPRESSEDBYTES_D5))+ assigns(memory_slice(r, sizeof(mlk_poly)))+ ensures(array_bound(r->coeffs, 0, MLKEM_N, 0, MLKEM_Q))+)+{+ unsigned i;+ for (i = 0; i < MLKEM_N / 8; i++)+ __loop__(+ invariant(i <= MLKEM_N / 8)+ invariant(array_bound(r->coeffs, 0, 8 * i, 0, MLKEM_Q))+ decreases(MLKEM_N / 8 - i))+ {+ unsigned j;+ uint8_t t[8];+ const unsigned offset = i * 5;+ /*+ * Explicitly truncate to avoid warning about+ * implicit truncation in CBMC and unwind loop for ease+ * of proof.+ */++ /*+ * Decompress 5 8-bit bytes (so 40 bits) into+ * 8 5-bit values stored in t[]+ */+ t[0] = 0x1F & (a[offset + 0] >> 0);+ t[1] = 0x1F & ((a[offset + 0] >> 5) | (a[offset + 1] << 3));+ t[2] = 0x1F & (a[offset + 1] >> 2);+ t[3] = 0x1F & ((a[offset + 1] >> 7) | (a[offset + 2] << 1));+ t[4] = 0x1F & ((a[offset + 2] >> 4) | (a[offset + 3] << 4));+ t[5] = 0x1F & (a[offset + 3] >> 1);+ t[6] = 0x1F & ((a[offset + 3] >> 6) | (a[offset + 4] << 2));+ t[7] = 0x1F & (a[offset + 4] >> 3);++ /* and copy to the correct slice in r[] */+ for (j = 0; j < 8; j++)+ __loop__(+ invariant(j <= 8 && i <= MLKEM_N / 8)+ invariant(array_bound(r->coeffs, 0, 8 * i + j, 0, MLKEM_Q))+ decreases(8 - j))+ {+ r->coeffs[8 * i + j] = mlk_scalar_decompress_d5(t[j]);+ }+ }++ mlk_assert_bound(r, MLKEM_N, 0, MLKEM_Q);+}++MLK_INTERNAL_API+void mlk_poly_decompress_d5(mlk_poly *r,+ const uint8_t a[MLKEM_POLYCOMPRESSEDBYTES_D5])+__contract__(+ requires(memory_no_alias(r, sizeof(mlk_poly)))+ requires(memory_no_alias(a, MLKEM_POLYCOMPRESSEDBYTES_D5))+ assigns(memory_slice(r, sizeof(mlk_poly)))+ ensures(array_bound(r->coeffs, 0, MLKEM_N, 0, MLKEM_Q))+)+{+#if defined(MLK_USE_NATIVE_POLY_DECOMPRESS_D5)+ int ret;+ ret = mlk_poly_decompress_d5_native(r->coeffs, a);+ if (ret == MLK_NATIVE_FUNC_SUCCESS)+ {+ mlk_assert_bound(r, MLKEM_N, 0, MLKEM_Q);+ return;+ }+#endif /* MLK_USE_NATIVE_POLY_DECOMPRESS_D5 */++ mlk_poly_decompress_d5_c(r, a);+}++/* Reference: Embedded into `polyvec_decompress()` in the+ * reference implementation, for ML-KEM-1024. */+MLK_STATIC_TESTABLE void mlk_poly_decompress_d11_c(+ mlk_poly *r, const uint8_t a[MLKEM_POLYCOMPRESSEDBYTES_D11])+__contract__(+ requires(memory_no_alias(r, sizeof(mlk_poly)))+ requires(memory_no_alias(a, MLKEM_POLYCOMPRESSEDBYTES_D11))+ assigns(memory_slice(r, sizeof(mlk_poly)))+ ensures(array_bound(r->coeffs, 0, MLKEM_N, 0, MLKEM_Q))+)+{+ unsigned j;+ for (j = 0; j < MLKEM_N / 8; j++)+ __loop__(+ invariant(j <= MLKEM_N / 8)+ invariant(array_bound(r->coeffs, 0, 8 * j, 0, MLKEM_Q))+ decreases(MLKEM_N / 8 - j))+ {+ unsigned k;+ uint16_t t[8];+ uint8_t const *base = &a[11 * j];+ t[0] = 0x7FF & ((base[0] >> 0) | ((uint16_t)base[1] << 8));+ t[1] = 0x7FF & ((base[1] >> 3) | ((uint16_t)base[2] << 5));+ t[2] = 0x7FF & ((base[2] >> 6) | ((uint16_t)base[3] << 2) |+ ((uint16_t)base[4] << 10));+ t[3] = 0x7FF & ((base[4] >> 1) | ((uint16_t)base[5] << 7));+ t[4] = 0x7FF & ((base[5] >> 4) | ((uint16_t)base[6] << 4));+ t[5] = 0x7FF & ((base[6] >> 7) | ((uint16_t)base[7] << 1) |+ ((uint16_t)base[8] << 9));+ t[6] = 0x7FF & ((base[8] >> 2) | ((uint16_t)base[9] << 6));+ t[7] = 0x7FF & ((base[9] >> 5) | ((uint16_t)base[10] << 3));++ for (k = 0; k < 8; k++)+ __loop__(+ invariant(k <= 8)+ invariant(array_bound(r->coeffs, 0, 8 * j + k, 0, MLKEM_Q))+ decreases(8 - k))+ {+ r->coeffs[8 * j + k] = mlk_scalar_decompress_d11(t[k]);+ }+ }++ mlk_assert_bound(r, MLKEM_N, 0, MLKEM_Q);+}++MLK_INTERNAL_API+void mlk_poly_decompress_d11(mlk_poly *r,+ const uint8_t a[MLKEM_POLYCOMPRESSEDBYTES_D11])+__contract__(+ requires(memory_no_alias(r, sizeof(mlk_poly)))+ requires(memory_no_alias(a, MLKEM_POLYCOMPRESSEDBYTES_D11))+ assigns(memory_slice(r, sizeof(mlk_poly)))+ ensures(array_bound(r->coeffs, 0, MLKEM_N, 0, MLKEM_Q))+)+{+#if defined(MLK_USE_NATIVE_POLY_DECOMPRESS_D11)+ int ret;+ ret = mlk_poly_decompress_d11_native(r->coeffs, a);+ if (ret == MLK_NATIVE_FUNC_SUCCESS)+ {+ mlk_assert_bound(r, MLKEM_N, 0, MLKEM_Q);+ return;+ }+#endif /* MLK_USE_NATIVE_POLY_DECOMPRESS_D11 */++ mlk_poly_decompress_d11_c(r, a);+}++#endif /* !MLK_CONFIG_NO_DECAPS_API */+#endif /* (!MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API) && \+ (MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 4) */++#if !defined(MLK_CONFIG_NO_KEYPAIR_API) || !defined(MLK_CONFIG_NO_ENCAPS_API)+/* Reference: `poly_tobytes()` in the reference implementation @[REF].+ * - In contrast to the reference implementation, we assume+ * unsigned canonical coefficients here.+ * The reference implementation works with coefficients+ * in the range [-(MLKEM_Q-1), MLKEM_Q-1]. */+MLK_STATIC_TESTABLE void mlk_poly_tobytes_c(uint8_t r[MLKEM_POLYBYTES],+ const mlk_poly *a)+__contract__(+ requires(memory_no_alias(r, MLKEM_POLYBYTES))+ requires(memory_no_alias(a, sizeof(mlk_poly)))+ requires(array_bound(a->coeffs, 0, MLKEM_N, 0, MLKEM_Q))+ assigns(memory_slice(r, MLKEM_POLYBYTES))+)+{+ unsigned i;+ mlk_assert_bound(a, MLKEM_N, 0, MLKEM_Q);++ for (i = 0; i < MLKEM_N / 2; i++)+ __loop__(invariant(i <= MLKEM_N / 2)+ decreases(MLKEM_N / 2 - i))+ {+ /* The conversion to uint16_t is safe since we assume that+ * the coefficients of `a` are non-negative. */+ const uint16_t t0 = (uint16_t)a->coeffs[2 * i];+ const uint16_t t1 = (uint16_t)a->coeffs[2 * i + 1];+ /*+ * t0 and t1 are both < MLKEM_Q, so contain at most 12 bits each of+ * significant data, so these can be packed into 24 bits or exactly+ * 3 bytes, as follows.+ */++ /* Least significant bits 0 - 7 of t0. */+ r[3 * i + 0] = (uint8_t)(t0 & 0xFF);++ /*+ * Most significant bits 8 - 11 of t0 become the least significant+ * nibble of the second byte. The least significant 4 bits+ * of t1 become the upper nibble of the second byte.+ *+ * The conversion to uint8_t does not alter the value.+ */+ r[3 * i + 1] = (uint8_t)((t0 >> 8) | ((t1 << 4) & 0xF0));++ /* Bits 4 - 11 of t1 become the third byte. The conversion to uint8_t+ * does not alter the value because t1 is 12-bit wide. */+ r[3 * i + 2] = (uint8_t)(t1 >> 4);+ }+}++MLK_INTERNAL_API+void mlk_poly_tobytes(uint8_t r[MLKEM_POLYBYTES], const mlk_poly *a)+{+#if defined(MLK_USE_NATIVE_POLY_TOBYTES)+ int ret;+ mlk_assert_bound(a, MLKEM_N, 0, MLKEM_Q);+ ret = mlk_poly_tobytes_native(r, a->coeffs);+ if (ret == MLK_NATIVE_FUNC_SUCCESS)+ {+ return;+ }+#endif /* MLK_USE_NATIVE_POLY_TOBYTES */++ mlk_poly_tobytes_c(r, a);+}+#endif /* !MLK_CONFIG_NO_KEYPAIR_API || !MLK_CONFIG_NO_ENCAPS_API */++#if !defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API)+/* Reference: `poly_frombytes()` in the reference implementation @[REF]. */+MLK_STATIC_TESTABLE void mlk_poly_frombytes_c(mlk_poly *r,+ const uint8_t a[MLKEM_POLYBYTES])+__contract__(+ requires(memory_no_alias(a, MLKEM_POLYBYTES))+ requires(memory_no_alias(r, sizeof(mlk_poly)))+ assigns(memory_slice(r, sizeof(mlk_poly)))+ ensures(array_bound(r->coeffs, 0, MLKEM_N, 0, MLKEM_UINT12_LIMIT))+)+{+ unsigned i;+ for (i = 0; i < MLKEM_N / 2; i++)+ __loop__(+ invariant(i <= MLKEM_N / 2)+ invariant(array_bound(r->coeffs, 0, 2 * i, 0, MLKEM_UINT12_LIMIT))+ decreases(MLKEM_N / 2 - i))+ {+ const uint8_t t0 = a[3 * i + 0];+ const uint8_t t1 = a[3 * i + 1];+ const uint8_t t2 = a[3 * i + 2];+ /* Safety:+ * - The explicit cast to uint16_t ensures that << 8 does+ * not signed-overflow even on a 16-bit system.+ * - The cast to int16_t is safe due to the explicit 0xFFF truncation.+ */+ r->coeffs[2 * i + 0] = (int16_t)(t0 | (((uint16_t)t1 << 8) & 0xFFF));+ r->coeffs[2 * i + 1] = (int16_t)((t1 >> 4) | (t2 << 4));+ }++ /* Note that the coefficients are not canonical */+ mlk_assert_bound(r, MLKEM_N, 0, MLKEM_UINT12_LIMIT);+}++MLK_INTERNAL_API+void mlk_poly_frombytes(mlk_poly *r, const uint8_t a[MLKEM_POLYBYTES])+{+#if defined(MLK_USE_NATIVE_POLY_FROMBYTES)+ int ret;+ ret = mlk_poly_frombytes_native(r->coeffs, a);+ if (ret == MLK_NATIVE_FUNC_SUCCESS)+ {+ return;+ }+#endif /* MLK_USE_NATIVE_POLY_FROMBYTES */++ mlk_poly_frombytes_c(r, a);+}+#endif /* !MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API */++#if !defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API)+/* Reference: `poly_frommsg()` in the reference implementation @[REF].+ * - We use a value barrier around the bit-selection mask to+ * reduce the risk of compiler-introduced branches.+ * The reference implementation contains the expression+ * `(msg[i] >> j) & 1` which the compiler can reason must+ * be either 0 or 1. */+MLK_INTERNAL_API+void mlk_poly_frommsg(mlk_poly *r, const uint8_t msg[MLKEM_INDCPA_MSGBYTES])+{+ unsigned i;+#if (MLKEM_INDCPA_MSGBYTES != MLKEM_N / 8)+#error "MLKEM_INDCPA_MSGBYTES must be equal to MLKEM_N/8 bytes!"+#endif++ for (i = 0; i < MLKEM_N / 8; i++)+ __loop__(+ invariant(i <= MLKEM_N / 8)+ invariant(array_bound(r->coeffs, 0, 8 * i, 0, MLKEM_Q))+ decreases(MLKEM_N / 8 - i))+ {+ unsigned j;+ for (j = 0; j < 8; j++)+ __loop__(+ invariant(i < MLKEM_N / 8 && j <= 8)+ invariant(array_bound(r->coeffs, 0, 8 * i + j, 0, MLKEM_Q))+ decreases(8 - j))+ {+ /* mlk_ct_sel_int16(MLKEM_Q_HALF, 0, b) is `Decompress_1(b != 0)`+ * as per @[FIPS203, Eq (4.8)]. */++ /* Prevent the compiler from recognizing this as a bit selection */+ uint8_t mask = mlk_value_barrier_u8((uint8_t)(1u << j));+ r->coeffs[8 * i + j] = mlk_ct_sel_int16(MLKEM_Q_HALF, 0, msg[i] & mask);+ }+ }+ mlk_assert_abs_bound(r, MLKEM_N, MLKEM_Q);+}+#endif /* !MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API */++#if !defined(MLK_CONFIG_NO_DECAPS_API)+/* Reference: `poly_tomsg()` in the reference implementation @[REF].+ * - In contrast to the reference implementation, we assume+ * unsigned canonical coefficients here.+ * The reference implementation works with coefficients+ * in the range [-(MLKEM_Q-1), MLKEM_Q-1].+ */+MLK_INTERNAL_API+void mlk_poly_tomsg(uint8_t msg[MLKEM_INDCPA_MSGBYTES], const mlk_poly *r)+{+ unsigned i;+ mlk_assert_bound(r, MLKEM_N, 0, MLKEM_Q);++ for (i = 0; i < MLKEM_N / 8; i++)+ __loop__(invariant(i <= MLKEM_N / 8)+ decreases(MLKEM_N / 8 - i))+ {+ unsigned j;+ msg[i] = 0;+ for (j = 0; j < 8; j++)+ __loop__(+ invariant(i <= MLKEM_N / 8 && j <= 8)+ decreases(8 - j))+ {+ uint32_t t = mlk_scalar_compress_d1(r->coeffs[8 * i + j]);+ msg[i] |= (uint8_t)(t << j);+ }+ }+}+#endif /* !MLK_CONFIG_NO_DECAPS_API */++#else /* !MLK_CONFIG_MULTILEVEL_NO_SHARED */++MLK_EMPTY_CU(compress)++#endif /* MLK_CONFIG_MULTILEVEL_NO_SHARED */
@@ -0,0 +1,613 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS203]+ * FIPS 203 Module-Lattice-Based Key-Encapsulation Mechanism Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/203/final+ *+ * - [REF]+ * CRYSTALS-Kyber C reference implementation+ * Bos, Ducas, Kiltz, Lepoint, Lyubashevsky, Schanck, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/kyber/tree/main/ref+ */++#ifndef MLK_COMPRESS_H+#define MLK_COMPRESS_H+++#include "cbmc.h"+#include "common.h"+#include "debug.h"+#include "poly.h"+#include "verify.h"++/*+ * The multiplication in this routine will exceed UINT32_MAX+ * and wrap around for large values of u. This is expected and required.+ */+#ifdef CBMC+#pragma CPROVER check push+#pragma CPROVER check disable "unsigned-overflow"+#endif++/**+ * Compute round(u * 2 / MLKEM_Q).+ *+ * @spec{Compress_1 from @[FIPS203, Eq (4.7)].}+ *+ * @reference{Part of poly_tomsg() in the reference implementation @[REF].}+ *+ * @param u Unsigned canonical modulus modulo MLKEM_Q to be compressed.+ *+ * @return Compressed value.+ */+static MLK_INLINE uint8_t mlk_scalar_compress_d1(int16_t u)+__contract__(+ requires(0 <= u && u <= MLKEM_Q - 1)+ ensures(return_value < 2)+ ensures(return_value == (((uint32_t)u * 2 + MLKEM_Q / 2) / MLKEM_Q) % 2) )+{+ /* Compute as follows:+ * ```+ * round(u * 2 / MLKEM_Q)+ * = round(u * 2 * (2^31 / MLKEM_Q) / 2^31)+ * ~= round(u * 2 * round(2^31 / MLKEM_Q) / 2^31)+ * ```+ */+ /* check-magic: 1290168 == 2*round(2^31 / MLKEM_Q) */+ uint32_t d0 = (uint32_t)u * 1290168;+ /* Unsigned shifting by 31 positions leaves only the top bit. */+ return (uint8_t)((d0 + ((uint32_t)1u << 30)) >> 31);+}+#ifdef CBMC+#pragma CPROVER check pop+#endif++/*+ * The multiplication in this routine will exceed UINT32_MAX+ * and wrap around for large values of u. This is expected and required.+ */+#ifdef CBMC+#pragma CPROVER check push+#pragma CPROVER check disable "unsigned-overflow"+#endif++/**+ * Compute round(u * 16 / MLKEM_Q) % 16.+ *+ * @spec{Compress_4 from @[FIPS203, Eq (4.7)].}+ *+ * @reference{Embedded into `poly_compress()` in the reference+ * implementation @[REF].}+ *+ * @param u Unsigned canonical modulus modulo MLKEM_Q to be compressed.+ *+ * @return Compressed value.+ */+static MLK_INLINE uint8_t mlk_scalar_compress_d4(int16_t u)+__contract__(+ requires(0 <= u && u <= MLKEM_Q - 1)+ ensures(return_value < 16)+ ensures(return_value == (((uint32_t)u * 16 + MLKEM_Q / 2) / MLKEM_Q) % 16))+{+ /* Compute as follows:+ * ```+ * round(u * 16 / MLKEM_Q)+ * = round(u * 16 * (2^28 / MLKEM_Q) / 2^28)+ * ~= round(u * 16 * round(2^28 / MLKEM_Q) / 2^28)+ * ```+ */+ /* check-magic: 1290160 == 16 * round(2^28 / MLKEM_Q) */+ uint32_t d0 = (uint32_t)u * 1290160;+ /* The return value is < 16, so not altered by the conversion to uint8_t. */+ return (uint8_t)((d0 + ((uint32_t)1u << 27)) >> 28); /* round(d0/2^28) */+}+#ifdef CBMC+#pragma CPROVER check pop+#endif++/**+ * Compute round(u * MLKEM_Q / 16).+ *+ * @spec{Decompress_4 from @[FIPS203, Eq (4.8)].}+ *+ * @reference{Embedded into `poly_decompress()` in the reference+ * implementation @[REF].}+ *+ * @param u Unsigned canonical modulus modulo 16 to be decompressed.+ *+ * @return Decompressed value.+ */+static MLK_INLINE int16_t mlk_scalar_decompress_d4(uint8_t u)+__contract__(+ requires(0 <= u && u < 16)+ ensures(return_value <= (MLKEM_Q - 1))+)+{+ /* The return value is in 0..MLKEM_Q-1, hence not altered by the+ * conversion to int16_t. */+ return (int16_t)((((uint32_t)u * MLKEM_Q) + 8) >> 4);+}++/*+ * The multiplication in this routine will exceed UINT32_MAX+ * and wrap around for large values of u. This is expected and required.+ */+#ifdef CBMC+#pragma CPROVER check push+#pragma CPROVER check disable "unsigned-overflow"+#endif++/**+ * Compute round(u * 32 / MLKEM_Q) % 32.+ *+ * @spec{Compress_5 from @[FIPS203, Eq (4.7)].}+ *+ * @reference{Embedded into `poly_compress()` in the reference+ * implementation @[REF].}+ *+ * @param u Unsigned canonical modulus modulo MLKEM_Q to be compressed.+ *+ * @return Compressed value.+ */+static MLK_INLINE uint8_t mlk_scalar_compress_d5(int16_t u)+__contract__(+ requires(0 <= u && u <= MLKEM_Q - 1)+ ensures(return_value < 32)+ ensures(return_value == (((uint32_t)u * 32 + MLKEM_Q / 2) / MLKEM_Q) % 32) )+{+ /* Compute as follows:+ * ```+ * round(u * 32 / MLKEM_Q)+ * = round(u * 32 * (2^27 / MLKEM_Q) / 2^27)+ * ~= round(u * 32 * round(2^27 / MLKEM_Q) / 2^27)+ * ```+ */+ /* check-magic: 1290176 == 2^5 * round(2^27 / MLKEM_Q) */+ uint32_t d0 = (uint32_t)u * 1290176;+ /* The return value is < 32, so not altered by the conversion to uint8_t. */+ return (uint8_t)((d0 + ((uint32_t)1u << 26)) >> 27); /* round(d0/2^27) */+}+#ifdef CBMC+#pragma CPROVER check pop+#endif++/**+ * Compute round(u * MLKEM_Q / 32).+ *+ * @spec{Decompress_5 from @[FIPS203, Eq (4.8)].}+ *+ * @reference{Embedded into `poly_decompress()` in the reference+ * implementation @[REF].}+ *+ * @param u Unsigned canonical modulus modulo 32 to be decompressed.+ *+ * @return Decompressed value.+ */+static MLK_INLINE int16_t mlk_scalar_decompress_d5(uint8_t u)+__contract__(+ requires(0 <= u && u < 32)+ ensures(0 <= return_value && return_value <= MLKEM_Q - 1)+)+{+ /* The return value is in 0..MLKEM_Q-1, hence not altered by the+ * conversion to int16_t. */+ return (int16_t)((((uint32_t)u * MLKEM_Q) + 16) >> 5);+}++/*+ * The multiplication in this routine will exceed UINT32_MAX+ * and wrap around for large values of u. This is expected and required.+ */+#ifdef CBMC+#pragma CPROVER check push+#pragma CPROVER check disable "unsigned-overflow"+#endif++/**+ * Compute round(u * 2**10 / MLKEM_Q) % 2**10.+ *+ * @spec{Compress_10 from @[FIPS203, Eq (4.7)].}+ *+ * @reference{Embedded into `polyvec_compress()` in the reference+ * implementation @[REF].}+ *+ * @param u Unsigned canonical modulus modulo MLKEM_Q to be compressed.+ *+ * @return Compressed value.+ */+static MLK_INLINE uint16_t mlk_scalar_compress_d10(int16_t u)+__contract__(+ requires(0 <= u && u <= MLKEM_Q - 1)+ ensures(return_value < (1u << 10))+ ensures(return_value == (((uint32_t)u * (1u << 10) + MLKEM_Q / 2) / MLKEM_Q) % (1 << 10)))+{+ /* Compute as follows:+ * ```+ * round(u * 1024 / MLKEM_Q)+ * = round(u * 1024 * (2^33 / MLKEM_Q) / 2^33)+ * ~= round(u * 1024 * round(2^33 / MLKEM_Q) / 2^33)+ * ```+ */+ /* check-magic: 2642263040 == 2^10 * round(2^33 / MLKEM_Q) */+ uint64_t d0 = (uint64_t)u * 2642263040;+ d0 = (d0 + ((uint64_t)1u << 32)) >> 33; /* round(d0/2^33) */+ return (d0 & 0x3FF);+}+#ifdef CBMC+#pragma CPROVER check pop+#endif++/**+ * Compute round(u * MLKEM_Q / 1024).+ *+ * @spec{Decompress_10 from @[FIPS203, Eq (4.8)].}+ *+ * @reference{Embedded into `polyvec_decompress()` in the reference+ * implementation @[REF].}+ *+ * @param u Unsigned canonical modulus modulo 1024 to be decompressed.+ *+ * @return Decompressed value.+ */+static MLK_INLINE int16_t mlk_scalar_decompress_d10(uint16_t u)+__contract__(+ requires(0 <= u && u < 1024)+ ensures(0 <= return_value && return_value <= (MLKEM_Q - 1))+)+{+ /* The return value is in 0..MLKEM_Q-1, hence not altered by the+ * conversion to int16_t. */+ return (int16_t)((((uint32_t)u * MLKEM_Q) + 512) >> 10);+}++/*+ * The multiplication in this routine will exceed UINT32_MAX+ * and wrap around for large values of u. This is expected and required.+ */+#ifdef CBMC+#pragma CPROVER check push+#pragma CPROVER check disable "unsigned-overflow"+#endif++/**+ * Compute round(u * 2**11 / MLKEM_Q) % 2**11.+ *+ * @spec{Compress_11 from @[FIPS203, Eq (4.7)].}+ *+ * @reference{Embedded into `polyvec_compress()` in the reference+ * implementation @[REF].}+ *+ * @param u Unsigned canonical modulus modulo MLKEM_Q to be compressed.+ *+ * @return Compressed value.+ */+static MLK_INLINE uint16_t mlk_scalar_compress_d11(int16_t u)+__contract__(+ requires(0 <= u && u <= MLKEM_Q - 1)+ ensures(return_value < (1u << 11))+ ensures(return_value == (((uint32_t)u * (1u << 11) + MLKEM_Q / 2) / MLKEM_Q) % (1 << 11)))+{+ /* Compute as follows:+ * ```+ * round(u * 2048 / MLKEM_Q)+ * = round(u * 2048 * (2^33 / MLKEM_Q) / 2^33)+ * ~= round(u * 2048 * round(2^33 / MLKEM_Q) / 2^33)+ * ```+ */+ /* check-magic: 5284526080 == 2^11 * round(2^33 / MLKEM_Q) */+ uint64_t d0 = (uint64_t)u * 5284526080;+ d0 = (d0 + ((uint64_t)1u << 32)) >> 33; /* round(d0/2^33) */+ return (d0 & 0x7FF);+}+#ifdef CBMC+#pragma CPROVER check pop+#endif++/**+ * Compute round(u * MLKEM_Q / 2048).+ *+ * @spec{Decompress_11 from @[FIPS203, Eq (4.8)].}+ *+ * @reference{Embedded into `polyvec_decompress()` in the reference+ * implementation @[REF].}+ *+ * @param u Unsigned canonical modulus modulo 2048 to be decompressed.+ *+ * @return Decompressed value.+ */+static MLK_INLINE int16_t mlk_scalar_decompress_d11(uint16_t u)+__contract__(+ requires(0 <= u && u < 2048)+ ensures(0 <= return_value && return_value <= (MLKEM_Q - 1))+)+{+ /* The return value is in 0..MLKEM_Q-1, hence not altered by the+ * conversion to int16_t. */+ return (int16_t)((((uint32_t)u * MLKEM_Q) + 1024) >> 11);+}++#if !defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API)+#if defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || (MLKEM_K == 2 || MLKEM_K == 3)+#define mlk_poly_compress_d4 MLK_NAMESPACE(poly_compress_d4)+/**+ * Compression (4 bits) and subsequent serialization of a polynomial.+ *+ * @spec{Implements `ByteEncode_4 (Compress_4 (a))`: ByteEncode_d+ * @[FIPS203, Algorithm 5], Compress_d @[FIPS203, Eq (4.7)], extended to+ * vectors as per @[FIPS203, 2.4.8 Applying Algorithms to Arrays].+ * `ByteEncode_{d_v} (Compress_{d_v} (v))` appears in @[FIPS203, Algorithm+ * 14 (K-PKE.Encrypt), L23], where `d_v=4` for ML-KEM-{512,768} @[FIPS203,+ * Table 2].}+ *+ * @param[out] r Output byte array (of length MLKEM_POLYCOMPRESSEDBYTES_D4+ * bytes).+ * @param[in] a Input polynomial. Coefficients must be unsigned canonical,+ * i.e. in [0,1,..,MLKEM_Q-1].+ */+MLK_INTERNAL_API+void mlk_poly_compress_d4(uint8_t r[MLKEM_POLYCOMPRESSEDBYTES_D4],+ const mlk_poly *a);++#define mlk_poly_compress_d10 MLK_NAMESPACE(poly_compress_d10)+/**+ * Compression (10 bits) and subsequent serialization of a polynomial.+ *+ * @spec{Implements `ByteEncode_10 (Compress_10 (a))`: ByteEncode_d+ * @[FIPS203, Algorithm 5], Compress_d @[FIPS203, Eq (4.7)], extended to+ * vectors as per @[FIPS203, 2.4.8 Applying Algorithms to Arrays].+ * `ByteEncode_{d_u} (Compress_{d_u} (u))` appears in @[FIPS203, Algorithm+ * 14 (K-PKE.Encrypt), L22], where `d_u=10` for ML-KEM-{512,768} @[FIPS203,+ * Table 2].}+ *+ * @param[out] r Output byte array (of length MLKEM_POLYCOMPRESSEDBYTES_D10+ * bytes).+ * @param[in] a Input polynomial. Coefficients must be unsigned canonical,+ * i.e. in [0,1,..,MLKEM_Q-1].+ */+MLK_INTERNAL_API+void mlk_poly_compress_d10(uint8_t r[MLKEM_POLYCOMPRESSEDBYTES_D10],+ const mlk_poly *a);++#if !defined(MLK_CONFIG_NO_DECAPS_API)+#define mlk_poly_decompress_d4 MLK_NAMESPACE(poly_decompress_d4)+/**+ * De-serialization and subsequent decompression (4 bits) of a polynomial;+ * approximate inverse of mlk_poly_compress_d4.+ *+ * Upon return, the coefficients of the output polynomial are+ * unsigned-canonical (non-negative and smaller than MLKEM_Q).+ *+ * @spec{Implements `Decompress_4 (ByteDecode_4 (a))`: ByteDecode_d+ * @[FIPS203, Algorithm 6], Decompress_d @[FIPS203, Eq (4.8)], extended to+ * vectors as per @[FIPS203, 2.4.8 Applying Algorithms to Arrays].+ * `Decompress_{d_v} (ByteDecode_{d_v} (v))` appears in @[FIPS203, Algorithm+ * 15 (K-PKE.Decrypt), L4], where `d_v=4` for ML-KEM-{512,768} @[FIPS203,+ * Table 2].}+ *+ * @param[out] r Output polynomial.+ * @param[in] a Input byte array (of length MLKEM_POLYCOMPRESSEDBYTES_D4+ * bytes).+ */+MLK_INTERNAL_API+void mlk_poly_decompress_d4(mlk_poly *r,+ const uint8_t a[MLKEM_POLYCOMPRESSEDBYTES_D4]);++#define mlk_poly_decompress_d10 MLK_NAMESPACE(poly_decompress_d10)+/**+ * De-serialization and subsequent decompression (10 bits) of a polynomial;+ * approximate inverse of mlk_poly_compress_d10.+ *+ * Upon return, the coefficients of the output polynomial are+ * unsigned-canonical (non-negative and smaller than MLKEM_Q).+ *+ * @spec{Implements `Decompress_10 (ByteDecode_10 (a))`: ByteDecode_d+ * @[FIPS203, Algorithm 6], Decompress_d @[FIPS203, Eq (4.8)], extended to+ * vectors as per @[FIPS203, 2.4.8 Applying Algorithms to Arrays].+ * `Decompress_{d_u} (ByteDecode_{d_u} (u))` appears in @[FIPS203, Algorithm+ * 15 (K-PKE.Decrypt), L3], where `d_u=10` for ML-KEM-{512,768} @[FIPS203,+ * Table 2].}+ *+ * @param[out] r Output polynomial.+ * @param[in] a Input byte array (of length MLKEM_POLYCOMPRESSEDBYTES_D10+ * bytes).+ */+MLK_INTERNAL_API+void mlk_poly_decompress_d10(mlk_poly *r,+ const uint8_t a[MLKEM_POLYCOMPRESSEDBYTES_D10]);+#endif /* !MLK_CONFIG_NO_DECAPS_API */+#endif /* MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 2 || MLKEM_K == 3 */++#if defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_K == 4+#define mlk_poly_compress_d5 MLK_NAMESPACE(poly_compress_d5)+/**+ * Compression (5 bits) and subsequent serialization of a polynomial.+ *+ * @spec{Implements `ByteEncode_5 (Compress_5 (a))`: ByteEncode_d+ * @[FIPS203, Algorithm 5], Compress_d @[FIPS203, Eq (4.7)], extended to+ * vectors as per @[FIPS203, 2.4.8 Applying Algorithms to Arrays].+ * `ByteEncode_{d_v} (Compress_{d_v} (v))` appears in @[FIPS203, Algorithm+ * 14 (K-PKE.Encrypt), L23], where `d_v=5` for ML-KEM-1024 @[FIPS203,+ * Table 2].}+ *+ * @param[out] r Output byte array (of length MLKEM_POLYCOMPRESSEDBYTES_D5+ * bytes).+ * @param[in] a Input polynomial. Coefficients must be unsigned canonical,+ * i.e. in [0,1,..,MLKEM_Q-1].+ */+MLK_INTERNAL_API+void mlk_poly_compress_d5(uint8_t r[MLKEM_POLYCOMPRESSEDBYTES_D5],+ const mlk_poly *a);++#define mlk_poly_compress_d11 MLK_NAMESPACE(poly_compress_d11)+/**+ * Compression (11 bits) and subsequent serialization of a polynomial.+ *+ * @spec{`ByteEncode_11 (Compress_11 (a))`: ByteEncode_d @[FIPS203,+ * Algorithm 5], Compress_d @[FIPS203, Eq (4.7)], extended to vectors as+ * per @[FIPS203, 2.4.8 Applying Algorithms to Arrays].+ * `ByteEncode_{d_u} (Compress_{d_u} (u))` appears in @[FIPS203, Algorithm+ * 14 (K-PKE.Encrypt), L22], where `d_u=11` for ML-KEM-1024 @[FIPS203,+ * Table 2].}+ *+ * @param[out] r Output byte array (of length MLKEM_POLYCOMPRESSEDBYTES_D11+ * bytes).+ * @param[in] a Input polynomial. Coefficients must be unsigned canonical,+ * i.e. in [0,1,..,MLKEM_Q-1].+ */+MLK_INTERNAL_API+void mlk_poly_compress_d11(uint8_t r[MLKEM_POLYCOMPRESSEDBYTES_D11],+ const mlk_poly *a);++#if !defined(MLK_CONFIG_NO_DECAPS_API)+#define mlk_poly_decompress_d5 MLK_NAMESPACE(poly_decompress_d5)+/**+ * De-serialization and subsequent decompression (5 bits) of a polynomial;+ * approximate inverse of mlk_poly_compress_d5.+ *+ * Upon return, the coefficients of the output polynomial are+ * unsigned-canonical (non-negative and smaller than MLKEM_Q).+ *+ * @spec{Implements `Decompress_5 (ByteDecode_5 (a))`: ByteDecode_d+ * @[FIPS203, Algorithm 6], Decompress_d @[FIPS203, Eq (4.8)], extended to+ * vectors as per @[FIPS203, 2.4.8 Applying Algorithms to Arrays].+ * `Decompress_{d_v} (ByteDecode_{d_v} (v))` appears in @[FIPS203, Algorithm+ * 15 (K-PKE.Decrypt), L4], where `d_v=5` for ML-KEM-1024 @[FIPS203,+ * Table 2].}+ *+ * @param[out] r Output polynomial.+ * @param[in] a Input byte array (of length MLKEM_POLYCOMPRESSEDBYTES_D5+ * bytes).+ */+MLK_INTERNAL_API+void mlk_poly_decompress_d5(mlk_poly *r,+ const uint8_t a[MLKEM_POLYCOMPRESSEDBYTES_D5]);++#define mlk_poly_decompress_d11 MLK_NAMESPACE(poly_decompress_d11)+/**+ * De-serialization and subsequent decompression (11 bits) of a polynomial;+ * approximate inverse of mlk_poly_compress_d11.+ *+ * Upon return, the coefficients of the output polynomial are+ * unsigned-canonical (non-negative and smaller than MLKEM_Q).+ *+ * @spec{Implements `Decompress_11 (ByteDecode_11 (a))`: ByteDecode_d+ * @[FIPS203, Algorithm 6], Decompress_d @[FIPS203, Eq (4.8)], extended to+ * vectors as per @[FIPS203, 2.4.8 Applying Algorithms to Arrays].+ * `Decompress_{d_u} (ByteDecode_{d_u} (u))` appears in @[FIPS203, Algorithm+ * 15 (K-PKE.Decrypt), L3], where `d_u=11` for ML-KEM-1024 @[FIPS203,+ * Table 2].}+ *+ * @param[out] r Output polynomial.+ * @param[in] a Input byte array (of length MLKEM_POLYCOMPRESSEDBYTES_D11+ * bytes).+ */+MLK_INTERNAL_API+void mlk_poly_decompress_d11(mlk_poly *r,+ const uint8_t a[MLKEM_POLYCOMPRESSEDBYTES_D11]);+#endif /* !MLK_CONFIG_NO_DECAPS_API */+#endif /* MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 4 */+#endif /* !MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API */++#if !defined(MLK_CONFIG_NO_KEYPAIR_API) || !defined(MLK_CONFIG_NO_ENCAPS_API)+#define mlk_poly_tobytes MLK_NAMESPACE(poly_tobytes)+/**+ * Serialization of a polynomial. Signed coefficients are converted to+ * unsigned form before serialization.+ *+ * @spec{Implements ByteEncode_12 @[FIPS203, Algorithm 5]. Extended to+ * vectors as per @[FIPS203, 2.4.8 Applying Algorithms to Arrays].}+ *+ * @param[out] r Output byte array (of MLKEM_POLYBYTES bytes).+ * @param[in] a Input polynomial, with each coefficient in the range+ * [0,1,..,MLKEM_Q-1].+ */+MLK_INTERNAL_API+void mlk_poly_tobytes(uint8_t r[MLKEM_POLYBYTES], const mlk_poly *a)+__contract__(+ requires(memory_no_alias(r, MLKEM_POLYBYTES))+ requires(memory_no_alias(a, sizeof(mlk_poly)))+ requires(array_bound(a->coeffs, 0, MLKEM_N, 0, MLKEM_Q))+ assigns(memory_slice(r, MLKEM_POLYBYTES))+);+#endif /* !MLK_CONFIG_NO_KEYPAIR_API || !MLK_CONFIG_NO_ENCAPS_API */+++#if !defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API)+#define mlk_poly_frombytes MLK_NAMESPACE(poly_frombytes)+/**+ * De-serialization of a polynomial.+ *+ * @spec{Implements ByteDecode_12 @[FIPS203, Algorithm 6]. Extended to+ * vectors as per @[FIPS203, 2.4.8 Applying Algorithms to Arrays].}+ *+ * @param[out] r Output polynomial, with each coefficient unsigned and in+ * the range 0..4095.+ * @param[in] a Input byte array (of MLKEM_POLYBYTES bytes).+ */+MLK_INTERNAL_API+void mlk_poly_frombytes(mlk_poly *r, const uint8_t a[MLKEM_POLYBYTES])+__contract__(+ requires(memory_no_alias(a, MLKEM_POLYBYTES))+ requires(memory_no_alias(r, sizeof(mlk_poly)))+ assigns(memory_slice(r, sizeof(mlk_poly)))+ ensures(array_bound(r->coeffs, 0, MLKEM_N, 0, MLKEM_UINT12_LIMIT))+);+#endif /* !MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API */+++#if !defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API)+#define mlk_poly_frommsg MLK_NAMESPACE(poly_frommsg)+/**+ * Convert a 32-byte message to a polynomial.+ *+ * @spec{Implements `Decompress_1 (ByteDecode_1 (a))`: ByteDecode_d+ * @[FIPS203, Algorithm 6], Decompress_d @[FIPS203, Eq (4.8)], extended to+ * vectors as per @[FIPS203, 2.4.8 Applying Algorithms to Arrays].+ * `Decompress_1 (ByteDecode_1 (w))` appears in @[FIPS203, Algorithm 15+ * (K-PKE.Encrypt), L20].}+ *+ * @param[out] r Output polynomial.+ * @param[in] msg Input message.+ */+MLK_INTERNAL_API+void mlk_poly_frommsg(mlk_poly *r, const uint8_t msg[MLKEM_INDCPA_MSGBYTES])+__contract__(+ requires(memory_no_alias(msg, MLKEM_INDCPA_MSGBYTES))+ requires(memory_no_alias(r, sizeof(mlk_poly)))+ assigns(memory_slice(r, sizeof(mlk_poly)))+ ensures(array_bound(r->coeffs, 0, MLKEM_N, 0, MLKEM_Q))+);+#endif /* !MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API */++#if !defined(MLK_CONFIG_NO_DECAPS_API)+#define mlk_poly_tomsg MLK_NAMESPACE(poly_tomsg)+/**+ * Convert a polynomial to a 32-byte message.+ *+ * @spec{Implements `ByteEncode_1 (Compress_1 (a))`: ByteEncode_d+ * @[FIPS203, Algorithm 5], Compress_d @[FIPS203, Eq (4.7)], extended to+ * vectors as per @[FIPS203, 2.4.8 Applying Algorithms to Arrays].+ * `ByteEncode_1 (Compress_1 (w))` appears in @[FIPS203, Algorithm 14+ * (K-PKE.Decrypt), L7].}+ *+ * @param[out] msg Output message.+ * @param[in] r Input polynomial. Coefficients must be unsigned canonical.+ */+MLK_INTERNAL_API+void mlk_poly_tomsg(uint8_t msg[MLKEM_INDCPA_MSGBYTES], const mlk_poly *r)+__contract__(+ requires(memory_no_alias(msg, MLKEM_INDCPA_MSGBYTES))+ requires(memory_no_alias(r, sizeof(mlk_poly)))+ requires(array_bound(r->coeffs, 0, MLKEM_N, 0, MLKEM_Q))+ assigns(memory_slice(msg, MLKEM_INDCPA_MSGBYTES))+);+#endif /* !MLK_CONFIG_NO_DECAPS_API */++#endif /* !MLK_COMPRESS_H */
@@ -0,0 +1,51 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+#ifndef MLK_CONTEXT_H+#define MLK_CONTEXT_H++/* This header is included by common.h once the configuration has been pulled+ * in; it is not meant to be included directly. */++/*+ * If the integration wants to provide a context parameter for use in+ * platform-specific hooks, then it should define this parameter.+ *+ * The MLK_CONTEXT_PARAMETERS_n macros are intended to be used with macros+ * defining the function names and expand to either pass or discard the context+ * argument as required by the current build. If there is no context parameter+ * requested then these are removed from the prototypes and from all calls.+ */+#ifdef MLK_CONFIG_CONTEXT_PARAMETER+#define MLK_CONTEXT_PARAMETERS_0(context) (context)+#define MLK_CONTEXT_PARAMETERS_1(arg0, context) (arg0, context)+#define MLK_CONTEXT_PARAMETERS_2(arg0, arg1, context) (arg0, arg1, context)+#define MLK_CONTEXT_PARAMETERS_3(arg0, arg1, arg2, context) \+ (arg0, arg1, arg2, context)+#define MLK_CONTEXT_PARAMETERS_4(arg0, arg1, arg2, arg3, context) \+ (arg0, arg1, arg2, arg3, context)+#else /* MLK_CONFIG_CONTEXT_PARAMETER */+#define MLK_CONTEXT_PARAMETERS_0(context) ()+#define MLK_CONTEXT_PARAMETERS_1(arg0, context) (arg0)+#define MLK_CONTEXT_PARAMETERS_2(arg0, arg1, context) (arg0, arg1)+#define MLK_CONTEXT_PARAMETERS_3(arg0, arg1, arg2, context) (arg0, arg1, arg2)+#define MLK_CONTEXT_PARAMETERS_4(arg0, arg1, arg2, arg3, context) \+ (arg0, arg1, arg2, arg3)+#endif /* !MLK_CONFIG_CONTEXT_PARAMETER */++/* Consume a context parameter carried only for the integration's benefit,+ * avoiding -Wunused-parameter; expands to nothing when no context is+ * configured. */+#if defined(MLK_CONFIG_CONTEXT_PARAMETER)+#define MLK_CONTEXT_UNUSED(context) ((void)(context))+#else+#define MLK_CONTEXT_UNUSED(context) ((void)0)+#endif++#if defined(MLK_CONFIG_CONTEXT_PARAMETER_TYPE) != \+ defined(MLK_CONFIG_CONTEXT_PARAMETER)+#error MLK_CONFIG_CONTEXT_PARAMETER_TYPE must be defined if and only if MLK_CONFIG_CONTEXT_PARAMETER is defined+#endif++#endif /* !MLK_CONTEXT_H */
@@ -0,0 +1,64 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* NOTE: You can remove this file unless you compile with MLKEM_DEBUG. */++#include "common.h"++#if !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED) && defined(MLKEM_DEBUG)+++#include <stdio.h>+#include <stdlib.h>+#include "debug.h"++#define MLK_DEBUG_ERROR_HEADER "[ERROR:%s:%04d] "++void mlk_debug_check_assert(const char *file, int line, const int val)+{+ if (val == 0)+ {+ fprintf(stderr, MLK_DEBUG_ERROR_HEADER "Assertion failed (value %d)\n",+ file, line, val);+ exit(1);+ }+}++void mlk_debug_check_bounds(const char *file, int line, const int16_t *ptr,+ unsigned len, int lower_bound_exclusive,+ int upper_bound_exclusive)+{+ int err = 0;+ unsigned i;+ for (i = 0; i < len; i++)+ {+ int16_t val = ptr[i];+ if (!(val > lower_bound_exclusive && val < upper_bound_exclusive))+ {+ fprintf(+ stderr,+ MLK_DEBUG_ERROR_HEADER+ "Bounds assertion failed: Index %u, value %d out of bounds (%d,%d)\n",+ file, line, i, (int)val, lower_bound_exclusive,+ upper_bound_exclusive);+ err = 1;+ }+ }++ if (err == 1)+ {+ exit(1);+ }+}++#else /* !MLK_CONFIG_MULTILEVEL_NO_SHARED && MLKEM_DEBUG */++MLK_EMPTY_CU(debug)++#endif /* !(!MLK_CONFIG_MULTILEVEL_NO_SHARED && MLKEM_DEBUG) */++/* To facilitate single-compilation-unit (SCU) builds, undefine all macros.+ * Don't modify by hand -- this is auto-generated by scripts/autogen. */+#undef MLK_DEBUG_ERROR_HEADER
@@ -0,0 +1,121 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+#ifndef MLK_DEBUG_H+#define MLK_DEBUG_H+#include "common.h"++#if defined(MLKEM_DEBUG)++/**+ * Check debug assertion.+ *+ * Prints an error message to stderr and calls exit(1) on failure.+ *+ * @param[in] file Filename.+ * @param line Line number.+ * @param val Value asserted to be non-zero.+ */+#define mlk_debug_check_assert MLK_NAMESPACE(mlkem_debug_assert)+void mlk_debug_check_assert(const char *file, int line, const int val);++/**+ * Check whether values in an array of int16_t are within specified bounds.+ *+ * Prints an error message to stderr and calls exit(1) on failure.+ *+ * @param[in] file Filename.+ * @param line Line number.+ * @param[in] ptr Base of array to be checked.+ * @param len Number of int16_t in @p ptr.+ * @param lower_bound_exclusive Exclusive lower bound.+ * @param upper_bound_exclusive Exclusive upper bound.+ */+#define mlk_debug_check_bounds MLK_NAMESPACE(mlkem_debug_check_bounds)+void mlk_debug_check_bounds(const char *file, int line, const int16_t *ptr,+ unsigned len, int lower_bound_exclusive,+ int upper_bound_exclusive);++/* Check assertion, calling exit() upon failure+ *+ * val: Value that's asserted to be non-zero+ */+#define mlk_assert(val) mlk_debug_check_assert(__FILE__, __LINE__, (val))++/* Check bounds in array of int16_t's+ * ptr: Base of int16_t array; will be explicitly cast to int16_t*,+ * so you may pass a byte-compatible type such as mlk_poly or mlk_polyvec.+ * len: Number of int16_t in array+ * value_lb: Inclusive lower value bound+ * value_ub: Exclusive upper value bound */+#define mlk_assert_bound(ptr, len, value_lb, value_ub) \+ mlk_debug_check_bounds(__FILE__, __LINE__, (const int16_t *)(ptr), (len), \+ (value_lb) - 1, (value_ub))++/* Check absolute bounds in array of int16_t's+ * ptr: Base of array, expression of type int16_t*+ * len: Number of int16_t in array+ * value_abs_bd: Exclusive absolute upper bound */+#define mlk_assert_abs_bound(ptr, len, value_abs_bd) \+ mlk_assert_bound((ptr), (len), (-(value_abs_bd) + 1), (value_abs_bd))++/* Version of bounds assertions for 2-dimensional arrays */+#define mlk_assert_bound_2d(ptr, len0, len1, value_lb, value_ub) \+ mlk_assert_bound((ptr), ((len0) * (len1)), (value_lb), (value_ub))++#define mlk_assert_abs_bound_2d(ptr, len0, len1, value_abs_bd) \+ mlk_assert_abs_bound((ptr), ((len0) * (len1)), (value_abs_bd))++/* When running CBMC, convert debug assertions into proof obligations */+#elif defined(CBMC)+#include "cbmc.h"++#define mlk_assert(val) cassert(val)++#define mlk_assert_bound(ptr, len, value_lb, value_ub) \+ cassert(array_bound(((int16_t *)(ptr)), 0, (len), (value_lb), (value_ub)))++#define mlk_assert_abs_bound(ptr, len, value_abs_bd) \+ cassert(array_abs_bound(((int16_t *)(ptr)), 0, (len), (value_abs_bd)))++/* Because of https://github.com/diffblue/cbmc/issues/8570, we can't+ * just use a single flattened array_bound(...) here. */+#define mlk_assert_bound_2d(ptr, M, N, value_lb, value_ub) \+ cassert(forall(kN, 0, (M), \+ array_bound(&((int16_t (*)[(N)])(ptr))[kN][0], 0, (N), \+ (value_lb), (value_ub))))++#define mlk_assert_abs_bound_2d(ptr, M, N, value_abs_bd) \+ cassert(forall(kN, 0, (M), \+ array_abs_bound(&((int16_t (*)[(N)])(ptr))[kN][0], 0, (N), \+ (value_abs_bd))))++#else /* !MLKEM_DEBUG && CBMC */++#define mlk_assert(val) \+ do \+ { \+ } while (0)+#define mlk_assert_bound(ptr, len, value_lb, value_ub) \+ do \+ { \+ } while (0)+#define mlk_assert_abs_bound(ptr, len, value_abs_bd) \+ do \+ { \+ } while (0)++#define mlk_assert_bound_2d(ptr, len0, len1, value_lb, value_ub) \+ do \+ { \+ } while (0)++#define mlk_assert_abs_bound_2d(ptr, len0, len1, value_abs_bd) \+ do \+ { \+ } while (0)+++#endif /* !MLKEM_DEBUG && !CBMC */+#endif /* !MLK_DEBUG_H */
@@ -0,0 +1,249 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS203]+ * FIPS 203 Module-Lattice-Based Key-Encapsulation Mechanism Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/203/final+ *+ * - [mupq]+ * Common files for pqm4, pqm3, pqriscv+ * Kannwischer, Petri, Rijneveld, Schwabe, Stoffelen+ * https://github.com/mupq/mupq+ *+ * - [supercop]+ * SUPERCOP benchmarking framework+ * Daniel J. Bernstein+ * http://bench.cr.yp.to/supercop.html+ *+ * - [tweetfips]+ * 'tweetfips202' FIPS202 implementation+ * Van Assche, Bernstein, Schwabe+ * https://keccak.team/2015/tweetfips202.html+ */++/* Based on the CC0 implementation from @[mupq] and the public domain+ * implementation @[supercop, crypto_hash/keccakc512/simple/]+ * by Ronny Van Keer, and the public domain @[tweetfips] implementation. */++#include "../common.h"+#if !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED)+++#include "../verify.h"+#include "fips202.h"+#include "keccakf1600.h"++/**+ * Absorb step of Keccak; non-incremental, starts by zeroeing the state.+ *+ * @warning Must only be called once.+ *+ * @param[out] s Pointer to (uninitialized) output Keccak state.+ * @param r Rate in bytes (e.g., 168 for SHAKE128).+ * @param[in] m Input to be absorbed into @p s.+ * @param mlen Length of input in bytes.+ * @param p Domain-separation byte for different Keccak-derived+ * functions.+ */+static void mlk_keccak_absorb_once(uint64_t *s, unsigned r, const uint8_t *m,+ size_t mlen, uint8_t p)+__contract__(+ requires(mlen <= MLK_MAX_BUFFER_SIZE)+ requires(r > 0)+ requires(r <= sizeof(uint64_t) * MLK_KECCAK_LANES)+ requires(memory_no_alias(s, sizeof(uint64_t) * MLK_KECCAK_LANES))+ requires(memory_no_alias(m, mlen))+ assigns(memory_slice(s, sizeof(uint64_t) * MLK_KECCAK_LANES)))+{+ /* Initialize state */+ size_t i;+ for (i = 0; i < 25; ++i)+ __loop__(invariant(i <= 25)+ decreases(25 - i))+ {+ s[i] = 0;+ }++ while (mlen >= r)+ __loop__(+ assigns(mlen, m, memory_slice(s, sizeof(uint64_t) * MLK_KECCAK_LANES))+ invariant(mlen <= loop_entry(mlen))+ invariant(m == loop_entry(m) + (loop_entry(mlen) - mlen))+ decreases(mlen))+ {+ mlk_keccakf1600_xor_bytes(s, m, 0, r);+ mlk_keccakf1600_permute(s);+ mlen -= r;+ m += r;+ }++ /* At this point, mlen < r, so the truncations to unsigned are safe below. */++ if (mlen > 0)+ {+ mlk_keccakf1600_xor_bytes(s, m, 0, (unsigned int)mlen);+ }++ if (mlen == r - 1)+ {+ p |= 128;+ mlk_keccakf1600_xor_bytes(s, &p, (unsigned int)mlen, 1);+ }+ else+ {+ mlk_keccakf1600_xor_bytes(s, &p, (unsigned int)mlen, 1);+ p = 128;+ mlk_keccakf1600_xor_bytes(s, &p, r - 1, 1);+ }+}++/**+ * Block-level Keccak squeeze.+ *+ * @param[out] h Output bytes.+ * @param nblocks Number of blocks to be squeezed.+ * @param[in,out] s Input/output state.+ * @param r Rate in bytes (e.g., 168 for SHAKE128).+ */+static void mlk_keccak_squeezeblocks(uint8_t *h, size_t nblocks, uint64_t *s,+ unsigned r)+__contract__(+ requires(r <= sizeof(uint64_t) * MLK_KECCAK_LANES)+ requires(nblocks <= 8 /* somewhat arbitrary bound */)+ requires(memory_no_alias(s, sizeof(uint64_t) * MLK_KECCAK_LANES))+ requires(memory_no_alias(h, nblocks * r))+ assigns(memory_slice(s, sizeof(uint64_t) * MLK_KECCAK_LANES))+ assigns(memory_slice(h, nblocks * r)))+{+ while (nblocks > 0)+ __loop__(+ assigns(h, nblocks,+ memory_slice(s, sizeof(uint64_t) * MLK_KECCAK_LANES),+ memory_slice(h, nblocks * r))+ invariant(nblocks <= loop_entry(nblocks) &&+ h == loop_entry(h) + r * (loop_entry(nblocks) - nblocks))+ decreases(nblocks))+ {+ mlk_keccakf1600_permute(s);+ mlk_keccakf1600_extract_bytes(s, h, 0, r);+ h += r;+ nblocks--;+ }+}++/**+ * Keccak squeeze; can be called on byte-level.+ *+ * @warning Must only be called once.+ *+ * @param[out] h Output bytes.+ * @param outlen Number of bytes to be squeezed.+ * @param[in,out] s Keccak state.+ * @param r Rate in bytes (e.g., 168 for SHAKE128).+ */+static void mlk_keccak_squeeze_once(uint8_t *h, size_t outlen, uint64_t *s,+ unsigned r)+__contract__(+ requires(outlen <= MLK_MAX_BUFFER_SIZE)+ requires(r > 0)+ requires(r <= sizeof(uint64_t) * MLK_KECCAK_LANES)+ requires(memory_no_alias(s, sizeof(uint64_t) * MLK_KECCAK_LANES))+ requires(memory_no_alias(h, outlen))+ assigns(memory_slice(s, sizeof(uint64_t) * MLK_KECCAK_LANES))+ assigns(memory_slice(h, outlen)))+{+ size_t len;+ while (outlen > 0)+ __loop__(+ assigns(len, h, outlen,+ memory_slice(s, sizeof(uint64_t) * MLK_KECCAK_LANES),+ memory_slice(h, outlen))+ invariant(outlen <= loop_entry(outlen) &&+ h == loop_entry(h) + (loop_entry(outlen) - outlen))+ decreases(outlen))+ {+ mlk_keccakf1600_permute(s);++ if (outlen < r)+ {+ len = outlen;+ }+ else+ {+ len = r;+ }+ mlk_keccakf1600_extract_bytes(s, h, 0, (unsigned int)len);+ h += len;+ outlen -= len;+ }+}++void mlk_shake128_absorb_once(mlk_shake128ctx *state, const uint8_t *input,+ size_t inlen)+{+ mlk_keccak_absorb_once(state->ctx, SHAKE128_RATE, input, inlen, 0x1F);+}++void mlk_shake128_squeezeblocks(uint8_t *output, size_t nblocks,+ mlk_shake128ctx *state)+{+ mlk_keccak_squeezeblocks(output, nblocks, state->ctx, SHAKE128_RATE);+}++void mlk_shake128_init(mlk_shake128ctx *state) { (void)state; }+void mlk_shake128_release(mlk_shake128ctx *state)+{+ /* Specification: Partially implements+ * @[FIPS203, Section 3.3, Destruction of intermediate values] */+ mlk_zeroize(state, sizeof(mlk_shake128ctx));+}++typedef mlk_shake128ctx mlk_shake256ctx;+void mlk_shake256(uint8_t *output, size_t outlen, const uint8_t *input,+ size_t inlen)+{+ mlk_shake256ctx state;+ /* Absorb input */+ mlk_keccak_absorb_once(state.ctx, SHAKE256_RATE, input, inlen, 0x1F);+ /* Squeeze output */+ mlk_keccak_squeeze_once(output, outlen, state.ctx, SHAKE256_RATE);+ /* Specification: Partially implements+ * @[FIPS203, Section 3.3, Destruction of intermediate values] */+ mlk_zeroize(&state, sizeof(state));+}++void mlk_sha3_256(uint8_t *output, const uint8_t *input, size_t inlen)+{+ uint64_t ctx[25];+ /* Absorb input */+ mlk_keccak_absorb_once(ctx, SHA3_256_RATE, input, inlen, 0x06);+ /* Squeeze output */+ mlk_keccak_squeeze_once(output, 32, ctx, SHA3_256_RATE);+ /* Specification: Partially implements+ * @[FIPS203, Section 3.3, Destruction of intermediate values] */+ mlk_zeroize(ctx, sizeof(ctx));+}++void mlk_sha3_512(uint8_t *output, const uint8_t *input, size_t inlen)+{+ uint64_t ctx[25];+ /* Absorb input */+ mlk_keccak_absorb_once(ctx, SHA3_512_RATE, input, inlen, 0x06);+ /* Squeeze output */+ mlk_keccak_squeeze_once(output, 64, ctx, SHA3_512_RATE);+ /* Specification: Partially implements+ * @[FIPS203, Section 3.3, Destruction of intermediate values] */+ mlk_zeroize(ctx, sizeof(ctx));+}++#else /* !MLK_CONFIG_MULTILEVEL_NO_SHARED */++MLK_EMPTY_CU(fips202)++#endif /* MLK_CONFIG_MULTILEVEL_NO_SHARED */
@@ -0,0 +1,144 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+#ifndef MLK_FIPS202_FIPS202_H+#define MLK_FIPS202_FIPS202_H++#include "../cbmc.h"+#include "../common.h"++#define SHAKE128_RATE 168+#define SHAKE256_RATE 136+#define SHA3_256_RATE 136+#define SHA3_384_RATE 104+#define SHA3_512_RATE 72++/** Context for the non-incremental SHAKE128 API. */+typedef struct+{+ uint64_t ctx[25]; /**< Keccak state. */+} MLK_ALIGN mlk_shake128ctx;++#define mlk_shake128_absorb_once MLK_NAMESPACE(shake128_absorb_once)+/**+ * One-shot absorb step of the SHAKE128 XOF.+ *+ * For call-sites (in mlkem-native):+ * - This function MUST ONLY be called straight after mlk_shake128_init().+ * - This function MUST ONLY be called once.+ *+ * Consequently, for providers of custom FIPS202 code to be used with+ * mlkem-native:+ * - You may assume that the input context is freshly initialized via+ * mlk_shake128_init().+ * - You may assume that this function is called exactly once.+ *+ * @param[in,out] state SHAKE128 context.+ * @param[in] input Input to be absorbed into the state.+ * @param inlen Length of input in bytes.+ */+void mlk_shake128_absorb_once(mlk_shake128ctx *state, const uint8_t *input,+ size_t inlen)+__contract__(+ requires(inlen <= MLK_MAX_BUFFER_SIZE)+ requires(memory_no_alias(state, sizeof(mlk_shake128ctx)))+ requires(memory_no_alias(input, inlen))+ assigns(memory_slice(state, sizeof(mlk_shake128ctx)))+);++#define mlk_shake128_squeezeblocks MLK_NAMESPACE(shake128_squeezeblocks)+/**+ * Squeeze step of SHAKE128 XOF. Squeezes full blocks of SHAKE128_RATE bytes+ * each. Modifies the state. Can be called multiple times to keep squeezing,+ * i.e., is incremental.+ *+ * @param[out] output Output blocks.+ * @param nblocks Number of blocks to be squeezed (written to output).+ * @param[in,out] state Keccak state.+ */+void mlk_shake128_squeezeblocks(uint8_t *output, size_t nblocks,+ mlk_shake128ctx *state)+__contract__(+ requires(nblocks <= 8 /* somewhat arbitrary bound */)+ requires(memory_no_alias(state, sizeof(mlk_shake128ctx)))+ requires(memory_no_alias(output, nblocks * SHAKE128_RATE))+ assigns(memory_slice(output, nblocks * SHAKE128_RATE), memory_slice(state, sizeof(mlk_shake128ctx)))+);++#define mlk_shake128_init MLK_NAMESPACE(shake128_init)+void mlk_shake128_init(mlk_shake128ctx *state);++#define mlk_shake128_release MLK_NAMESPACE(shake128_release)+void mlk_shake128_release(mlk_shake128ctx *state);++/* One-stop SHAKE256 call. Aliasing between input and+ * output is not permitted */+#define mlk_shake256 MLK_NAMESPACE(shake256)+/**+ * SHAKE256 XOF with non-incremental API.+ *+ * @param[out] output Output buffer.+ * @param outlen Requested output length in bytes.+ * @param[in] input Input buffer.+ * @param inlen Length of input in bytes.+ */+void mlk_shake256(uint8_t *output, size_t outlen, const uint8_t *input,+ size_t inlen)+__contract__(+ requires(inlen <= MLK_MAX_BUFFER_SIZE)+ requires(outlen <= MLK_MAX_BUFFER_SIZE)+ requires(memory_no_alias(input, inlen))+ requires(memory_no_alias(output, outlen))+ assigns(memory_slice(output, outlen))+);++/* One-stop SHA3_256 call. Aliasing between input and+ * output is not permitted */+#define SHA3_256_HASHBYTES 32+#define mlk_sha3_256 MLK_NAMESPACE(sha3_256)+/**+ * SHA3-256 with non-incremental API.+ *+ * @param[out] output Output buffer.+ * @param[in] input Input buffer.+ * @param inlen Length of input in bytes.+ */+void mlk_sha3_256(uint8_t *output, const uint8_t *input, size_t inlen)+__contract__(+ requires(inlen <= MLK_MAX_BUFFER_SIZE)+ requires(memory_no_alias(input, inlen))+ requires(memory_no_alias(output, SHA3_256_HASHBYTES))+ assigns(memory_slice(output, SHA3_256_HASHBYTES))+);++/* One-stop SHA3_512 call. Aliasing between input and+ * output is not permitted */+#define SHA3_512_HASHBYTES 64+#define mlk_sha3_512 MLK_NAMESPACE(sha3_512)+/**+ * SHA3-512 with non-incremental API.+ *+ * @param[out] output Output buffer.+ * @param[in] input Input buffer.+ * @param inlen Length of input in bytes.+ */+void mlk_sha3_512(uint8_t *output, const uint8_t *input, size_t inlen)+__contract__(+ requires(inlen <= MLK_MAX_BUFFER_SIZE)+ requires(memory_no_alias(input, inlen))+ requires(memory_no_alias(output, SHA3_512_HASHBYTES))+ assigns(memory_slice(output, SHA3_512_HASHBYTES))+);++#if !defined(MLK_CONFIG_USE_NATIVE_BACKEND_FIPS202) || \+ !defined(MLK_USE_NATIVE_FIPS202_X4)+/* If you provide your own FIPS-202 implementation where the x4-+ * Keccak-f1600-x4 implementation falls back to 4-fold Keccak-f1600,+ * set this to gain a small speedup. */+#define FIPS202_X4_DEFAULT_IMPLEMENTATION+#endif /* !MLK_CONFIG_USE_NATIVE_BACKEND_FIPS202 || !MLK_USE_NATIVE_FIPS202_X4 \+ */+++#endif /* !MLK_FIPS202_FIPS202_H */
@@ -0,0 +1,207 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS203]+ * FIPS 203 Module-Lattice-Based Key-Encapsulation Mechanism Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/203/final+ */++#include "../common.h"+#if !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED)++#include "../verify.h"+#include "fips202.h"+#include "fips202x4.h"+#include "keccakf1600.h"++typedef mlk_shake128x4ctx mlk_shake256x4_ctx;++static void mlk_keccak_absorb_once_x4(uint64_t *s, unsigned r,+ const uint8_t *in0, const uint8_t *in1,+ const uint8_t *in2, const uint8_t *in3,+ size_t inlen, uint8_t p)+__contract__(+ requires(inlen <= MLK_MAX_BUFFER_SIZE)+ requires(memory_no_alias(s, sizeof(uint64_t) * MLK_KECCAK_LANES * MLK_KECCAK_WAY))+ requires(r > 0)+ requires(r <= sizeof(uint64_t) * MLK_KECCAK_LANES)+ requires(memory_no_alias(in0, inlen))+ requires(memory_no_alias(in1, inlen))+ requires(memory_no_alias(in2, inlen))+ requires(memory_no_alias(in3, inlen))+ assigns(memory_slice(s, sizeof(uint64_t) * MLK_KECCAK_LANES * MLK_KECCAK_WAY)))+{+ while (inlen >= r)+ __loop__(+ assigns(inlen, in0, in1, in2, in3, memory_slice(s, sizeof(uint64_t) * MLK_KECCAK_LANES * MLK_KECCAK_WAY))+ invariant(inlen <= loop_entry(inlen))+ invariant(in0 == loop_entry(in0) + (loop_entry(inlen) - inlen))+ invariant(in1 == loop_entry(in1) + (loop_entry(inlen) - inlen))+ invariant(in2 == loop_entry(in2) + (loop_entry(inlen) - inlen))+ invariant(in3 == loop_entry(in3) + (loop_entry(inlen) - inlen))+ decreases(inlen))+ {+ mlk_keccakf1600x4_xor_bytes(s, in0, in1, in2, in3, 0, r);+ mlk_keccakf1600x4_permute(s);++ in0 += r;+ in1 += r;+ in2 += r;+ in3 += r;+ inlen -= r;+ }++ /* At this point, inlen < r, so the truncations to unsigned are safe below. */++ if (inlen > 0)+ {+ mlk_keccakf1600x4_xor_bytes(s, in0, in1, in2, in3, 0, (unsigned int)inlen);+ }++ if (inlen == r - 1)+ {+ p |= 128;+ mlk_keccakf1600x4_xor_bytes(s, &p, &p, &p, &p, (unsigned int)inlen, 1);+ }+ else+ {+ mlk_keccakf1600x4_xor_bytes(s, &p, &p, &p, &p, (unsigned int)inlen, 1);+ p = 128;+ mlk_keccakf1600x4_xor_bytes(s, &p, &p, &p, &p, r - 1, 1);+ }+}++static void mlk_keccak_squeezeblocks_x4(uint8_t *out0, uint8_t *out1,+ uint8_t *out2, uint8_t *out3,+ size_t nblocks, uint64_t *s, unsigned r)+__contract__(+ requires(r <= sizeof(uint64_t) * MLK_KECCAK_LANES)+ requires(r == SHAKE128_RATE || r == SHAKE256_RATE)+ requires(nblocks <= (MLK_MAX_BUFFER_SIZE / SHAKE256_RATE))+ requires(memory_no_alias(s, sizeof(uint64_t) * MLK_KECCAK_LANES * MLK_KECCAK_WAY))+ requires(memory_no_alias(out0, nblocks * r))+ requires(memory_no_alias(out1, nblocks * r))+ requires(memory_no_alias(out2, nblocks * r))+ requires(memory_no_alias(out3, nblocks * r))+ assigns(memory_slice(s, sizeof(uint64_t) * MLK_KECCAK_LANES * MLK_KECCAK_WAY))+ assigns(memory_slice(out0, nblocks * r))+ assigns(memory_slice(out1, nblocks * r))+ assigns(memory_slice(out2, nblocks * r))+ assigns(memory_slice(out3, nblocks * r)))+{+ size_t current_offset = 0;+ while (nblocks > 0)+ __loop__(+ assigns(nblocks, current_offset,+ memory_slice(s, sizeof(uint64_t) * MLK_KECCAK_LANES * MLK_KECCAK_WAY),+ memory_slice(out0, nblocks * r),+ memory_slice(out1, nblocks * r),+ memory_slice(out2, nblocks * r),+ memory_slice(out3, nblocks * r))+ invariant(nblocks <= loop_entry(nblocks))+ invariant(current_offset == (loop_entry(nblocks) - nblocks) * r)+ decreases(nblocks))+ {+ mlk_keccakf1600x4_permute(s);+ mlk_keccakf1600x4_extract_bytes(+ s, &out0[current_offset], &out1[current_offset], &out2[current_offset],+ &out3[current_offset], 0, r);+ current_offset += r;+ nblocks--;+ }+}++void mlk_shake128x4_absorb_once(mlk_shake128x4ctx *state, const uint8_t *in0,+ const uint8_t *in1, const uint8_t *in2,+ const uint8_t *in3, size_t inlen)+{+ mlk_memset(state, 0, sizeof(mlk_shake128x4ctx));+ mlk_keccak_absorb_once_x4(state->ctx, SHAKE128_RATE, in0, in1, in2, in3,+ inlen, 0x1F);+}++void mlk_shake128x4_squeezeblocks(uint8_t *out0, uint8_t *out1, uint8_t *out2,+ uint8_t *out3, size_t nblocks,+ mlk_shake128x4ctx *state)+{+ mlk_keccak_squeezeblocks_x4(out0, out1, out2, out3, nblocks, state->ctx,+ SHAKE128_RATE);+}++void mlk_shake128x4_init(mlk_shake128x4ctx *state) { (void)state; }+void mlk_shake128x4_release(mlk_shake128x4ctx *state)+{+ /* Specification: Partially implements+ * @[FIPS203, Section 3.3, Destruction of intermediate values] */+ mlk_zeroize(state, sizeof(mlk_shake128x4ctx));+}++static void mlk_shake256x4_absorb_once(mlk_shake256x4_ctx *state,+ const uint8_t *in0, const uint8_t *in1,+ const uint8_t *in2, const uint8_t *in3,+ size_t inlen)+{+ mlk_memset(state, 0, sizeof(mlk_shake128x4ctx));+ mlk_keccak_absorb_once_x4(state->ctx, SHAKE256_RATE, in0, in1, in2, in3,+ inlen, 0x1F);+}++static void mlk_shake256x4_squeezeblocks(uint8_t *out0, uint8_t *out1,+ uint8_t *out2, uint8_t *out3,+ size_t nblocks,+ mlk_shake256x4_ctx *state)+{+ mlk_keccak_squeezeblocks_x4(out0, out1, out2, out3, nblocks, state->ctx,+ SHAKE256_RATE);+}++void mlk_shake256x4(uint8_t *out0, uint8_t *out1, uint8_t *out2, uint8_t *out3,+ size_t outlen, const uint8_t *in0, const uint8_t *in1,+ const uint8_t *in2, const uint8_t *in3, size_t inlen)+{+ mlk_shake256x4_ctx statex;+ size_t nblocks = outlen / SHAKE256_RATE;+ uint8_t tmp0[SHAKE256_RATE];+ uint8_t tmp1[SHAKE256_RATE];+ uint8_t tmp2[SHAKE256_RATE];+ uint8_t tmp3[SHAKE256_RATE];++ mlk_shake256x4_absorb_once(&statex, in0, in1, in2, in3, inlen);+ mlk_shake256x4_squeezeblocks(out0, out1, out2, out3, nblocks, &statex);++ out0 += nblocks * SHAKE256_RATE;+ out1 += nblocks * SHAKE256_RATE;+ out2 += nblocks * SHAKE256_RATE;+ out3 += nblocks * SHAKE256_RATE;++ outlen -= nblocks * SHAKE256_RATE;++ if (outlen)+ {+ mlk_shake256x4_squeezeblocks(tmp0, tmp1, tmp2, tmp3, 1, &statex);+ mlk_memcpy(out0, tmp0, outlen);+ mlk_memcpy(out1, tmp1, outlen);+ mlk_memcpy(out2, tmp2, outlen);+ mlk_memcpy(out3, tmp3, outlen);+ }++ /* Specification: Partially implements+ * @[FIPS203, Section 3.3, Destruction of intermediate values] */+ mlk_zeroize(&statex, sizeof(statex));+ mlk_zeroize(tmp0, sizeof(tmp0));+ mlk_zeroize(tmp1, sizeof(tmp1));+ mlk_zeroize(tmp2, sizeof(tmp2));+ mlk_zeroize(tmp3, sizeof(tmp3));+}++#else /* !MLK_CONFIG_MULTILEVEL_NO_SHARED */++MLK_EMPTY_CU(fips202x4)++#endif /* MLK_CONFIG_MULTILEVEL_NO_SHARED */
@@ -0,0 +1,81 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+#ifndef MLK_FIPS202_FIPS202X4_H+#define MLK_FIPS202_FIPS202X4_H+++#include "../cbmc.h"+#include "../common.h"++#include "fips202.h"+#include "keccakf1600.h"++/** Context for the non-incremental 4-way SHAKE128 API. */+typedef struct+{+ uint64_t ctx[MLK_KECCAK_LANES *+ MLK_KECCAK_WAY]; /**< 4-way Keccak state, stored sequentially. */+} MLK_ALIGN mlk_shake128x4ctx;++#define mlk_shake128x4_absorb_once MLK_NAMESPACE(shake128x4_absorb_once)+void mlk_shake128x4_absorb_once(mlk_shake128x4ctx *state, const uint8_t *in0,+ const uint8_t *in1, const uint8_t *in2,+ const uint8_t *in3, size_t inlen)+__contract__(+ requires(inlen <= MLK_MAX_BUFFER_SIZE)+ requires(memory_no_alias(state, sizeof(mlk_shake128x4ctx)))+ requires(memory_no_alias(in0, inlen))+ requires(memory_no_alias(in1, inlen))+ requires(memory_no_alias(in2, inlen))+ requires(memory_no_alias(in3, inlen))+ assigns(memory_slice(state, sizeof(mlk_shake128x4ctx)))+);++#define mlk_shake128x4_squeezeblocks MLK_NAMESPACE(shake128x4_squeezeblocks)+void mlk_shake128x4_squeezeblocks(uint8_t *out0, uint8_t *out1, uint8_t *out2,+ uint8_t *out3, size_t nblocks,+ mlk_shake128x4ctx *state)+__contract__(+ requires(nblocks <= 8 /* somewhat arbitrary bound */)+ requires(memory_no_alias(state, sizeof(mlk_shake128x4ctx)))+ requires(memory_no_alias(out0, nblocks * SHAKE128_RATE))+ requires(memory_no_alias(out1, nblocks * SHAKE128_RATE))+ requires(memory_no_alias(out2, nblocks * SHAKE128_RATE))+ requires(memory_no_alias(out3, nblocks * SHAKE128_RATE))+ assigns(memory_slice(out0, nblocks * SHAKE128_RATE),+ memory_slice(out1, nblocks * SHAKE128_RATE),+ memory_slice(out2, nblocks * SHAKE128_RATE),+ memory_slice(out3, nblocks * SHAKE128_RATE),+ memory_slice(state, sizeof(mlk_shake128x4ctx)))+);++#define mlk_shake128x4_init MLK_NAMESPACE(shake128x4_init)+void mlk_shake128x4_init(mlk_shake128x4ctx *state);++#define mlk_shake128x4_release MLK_NAMESPACE(shake128x4_release)+void mlk_shake128x4_release(mlk_shake128x4ctx *state);++#define mlk_shake256x4 MLK_NAMESPACE(shake256x4)+void mlk_shake256x4(uint8_t *out0, uint8_t *out1, uint8_t *out2, uint8_t *out3,+ size_t outlen, const uint8_t *in0, const uint8_t *in1,+ const uint8_t *in2, const uint8_t *in3, size_t inlen)+__contract__(+ requires(inlen <= MLK_MAX_BUFFER_SIZE)+ requires(outlen <= MLK_MAX_BUFFER_SIZE)+ requires(memory_no_alias(in0, inlen))+ requires(memory_no_alias(in1, inlen))+ requires(memory_no_alias(in2, inlen))+ requires(memory_no_alias(in3, inlen))+ requires(memory_no_alias(out0, outlen))+ requires(memory_no_alias(out1, outlen))+ requires(memory_no_alias(out2, outlen))+ requires(memory_no_alias(out3, outlen))+ assigns(memory_slice(out0, outlen))+ assigns(memory_slice(out1, outlen))+ assigns(memory_slice(out2, outlen))+ assigns(memory_slice(out3, outlen))+);++#endif /* !MLK_FIPS202_FIPS202X4_H */
@@ -0,0 +1,499 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [mupq]+ * Common files for pqm4, pqm3, pqriscv+ * Kannwischer, Petri, Rijneveld, Schwabe, Stoffelen+ * https://github.com/mupq/mupq+ *+ * - [supercop]+ * SUPERCOP benchmarking framework+ * Daniel J. Bernstein+ * http://bench.cr.yp.to/supercop.html+ *+ * - [tweetfips]+ * 'tweetfips202' FIPS202 implementation+ * Van Assche, Bernstein, Schwabe+ * https://keccak.team/2015/tweetfips202.html+ */++/* Based on the CC0 implementation from @[mupq] and the public domain+ * implementation @[supercop, crypto_hash/keccakc512/simple/]+ * by Ronny Van Keer, and the public domain @[tweetfips] implementation. */+++#include "keccakf1600.h"+#if !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED)++#define MLK_KECCAK_NROUNDS 24+#define MLK_KECCAK_ROL(a, offset) (((a) << (offset)) ^ ((a) >> (64 - (offset))))++void mlk_keccakf1600_extract_bytes(uint64_t *state, unsigned char *data,+ unsigned offset, unsigned length)+{+ unsigned i;+#if defined(MLK_SYS_LITTLE_ENDIAN)+ uint8_t *state_ptr = (uint8_t *)state + offset;+ for (i = 0; i < length; i++)+ __loop__(invariant(i <= length)+ decreases(length - i))+ {+ data[i] = state_ptr[i];+ }+#else /* MLK_SYS_LITTLE_ENDIAN */+ /* Portable version */+ for (i = 0; i < length; i++)+ __loop__(invariant(i <= length)+ decreases(length - i))+ {+ data[i] = (state[(offset + i) >> 3] >> (8 * ((offset + i) & 0x07))) & 0xFF;+ }+#endif /* !MLK_SYS_LITTLE_ENDIAN */+}++void mlk_keccakf1600_xor_bytes(uint64_t *state, const unsigned char *data,+ unsigned offset, unsigned length)+{+ unsigned i;+#if defined(MLK_SYS_LITTLE_ENDIAN)+ uint8_t *state_ptr = (uint8_t *)state + offset;+ for (i = 0; i < length; i++)+ __loop__(invariant(i <= length)+ decreases(length - i))+ {+ state_ptr[i] ^= data[i];+ }+#else /* MLK_SYS_LITTLE_ENDIAN */+ /* Portable version */+ for (i = 0; i < length; i++)+ __loop__(invariant(i <= length)+ decreases(length - i))+ {+ state[(offset + i) >> 3] ^= (uint64_t)data[i]+ << (8 * ((offset + i) & 0x07));+ }+#endif /* !MLK_SYS_LITTLE_ENDIAN */+}++static void mlk_keccakf1600x4_extract_bytes_c(uint64_t *state,+ unsigned char *data0,+ unsigned char *data1,+ unsigned char *data2,+ unsigned char *data3,+ unsigned offset, unsigned length)+__contract__(+ requires(0 <= offset && offset <= MLK_KECCAK_LANES * sizeof(uint64_t) &&+ 0 <= length && length <= MLK_KECCAK_LANES * sizeof(uint64_t) - offset)+ requires(memory_no_alias(state, sizeof(uint64_t) * MLK_KECCAK_LANES * MLK_KECCAK_WAY))+ requires(memory_no_alias(data0, length))+ requires(memory_no_alias(data1, length))+ requires(memory_no_alias(data2, length))+ requires(memory_no_alias(data3, length))+ assigns(memory_slice(data0, length))+ assigns(memory_slice(data1, length))+ assigns(memory_slice(data2, length))+ assigns(memory_slice(data3, length))+)+{+ mlk_keccakf1600_extract_bytes(state + MLK_KECCAK_LANES * 0, data0, offset,+ length);+ mlk_keccakf1600_extract_bytes(state + MLK_KECCAK_LANES * 1, data1, offset,+ length);+ mlk_keccakf1600_extract_bytes(state + MLK_KECCAK_LANES * 2, data2, offset,+ length);+ mlk_keccakf1600_extract_bytes(state + MLK_KECCAK_LANES * 3, data3, offset,+ length);+}++void mlk_keccakf1600x4_extract_bytes(uint64_t *state, unsigned char *data0,+ unsigned char *data1, unsigned char *data2,+ unsigned char *data3, unsigned offset,+ unsigned length)+{+#if defined(MLK_USE_NATIVE_FIPS202_X4_EXTRACT_BYTES)+ if (mlk_keccakf1600_extract_bytes_x4_native(state, data0, data1, data2, data3,+ offset, length) ==+ MLK_NATIVE_FUNC_SUCCESS)+ {+ return;+ }+#endif /* MLK_USE_NATIVE_FIPS202_X4_EXTRACT_BYTES */+ mlk_keccakf1600x4_extract_bytes_c(state, data0, data1, data2, data3, offset,+ length);+}++static void mlk_keccakf1600x4_xor_bytes_c(uint64_t *state,+ const unsigned char *data0,+ const unsigned char *data1,+ const unsigned char *data2,+ const unsigned char *data3,+ unsigned offset, unsigned length)+__contract__(+ requires(0 <= offset && offset <= MLK_KECCAK_LANES * sizeof(uint64_t) &&+ 0 <= length && length <= MLK_KECCAK_LANES * sizeof(uint64_t) - offset)+ requires(memory_no_alias(state, sizeof(uint64_t) * MLK_KECCAK_LANES * MLK_KECCAK_WAY))+ requires(memory_no_alias(data0, length))+ /* Case 1: all input buffers are distinct; Case 2: All input buffers are the same */+ requires((data0 == data1 &&+ data0 == data2 &&+ data0 == data3) ||+ (memory_no_alias(data1, length) &&+ memory_no_alias(data2, length) &&+ memory_no_alias(data3, length)))+ assigns(memory_slice(state, sizeof(uint64_t) * MLK_KECCAK_LANES * MLK_KECCAK_WAY))+)+{+ mlk_keccakf1600_xor_bytes(state + MLK_KECCAK_LANES * 0, data0, offset,+ length);+ mlk_keccakf1600_xor_bytes(state + MLK_KECCAK_LANES * 1, data1, offset,+ length);+ mlk_keccakf1600_xor_bytes(state + MLK_KECCAK_LANES * 2, data2, offset,+ length);+ mlk_keccakf1600_xor_bytes(state + MLK_KECCAK_LANES * 3, data3, offset,+ length);+}++void mlk_keccakf1600x4_xor_bytes(uint64_t *state, const unsigned char *data0,+ const unsigned char *data1,+ const unsigned char *data2,+ const unsigned char *data3, unsigned offset,+ unsigned length)+{+#if defined(MLK_USE_NATIVE_FIPS202_X4_XOR_BYTES)+ if (mlk_keccakf1600_xor_bytes_x4_native(state, data0, data1, data2, data3,+ offset,+ length) == MLK_NATIVE_FUNC_SUCCESS)+ {+ return;+ }+#endif /* MLK_USE_NATIVE_FIPS202_X4_XOR_BYTES */+ mlk_keccakf1600x4_xor_bytes_c(state, data0, data1, data2, data3, offset,+ length);+}++void mlk_keccakf1600x4_permute(uint64_t *state)+{+#if defined(MLK_USE_NATIVE_FIPS202_X4)+ if (mlk_keccak_f1600_x4_native(state) == MLK_NATIVE_FUNC_SUCCESS)+ {+ return;+ }+#endif /* MLK_USE_NATIVE_FIPS202_X4 */+ mlk_keccakf1600_permute(state + MLK_KECCAK_LANES * 0);+ mlk_keccakf1600_permute(state + MLK_KECCAK_LANES * 1);+ mlk_keccakf1600_permute(state + MLK_KECCAK_LANES * 2);+ mlk_keccakf1600_permute(state + MLK_KECCAK_LANES * 3);+}++static const uint64_t mlk_KeccakF_RoundConstants[MLK_KECCAK_NROUNDS] = {+ (uint64_t)0x0000000000000001ULL, (uint64_t)0x0000000000008082ULL,+ (uint64_t)0x800000000000808aULL, (uint64_t)0x8000000080008000ULL,+ (uint64_t)0x000000000000808bULL, (uint64_t)0x0000000080000001ULL,+ (uint64_t)0x8000000080008081ULL, (uint64_t)0x8000000000008009ULL,+ (uint64_t)0x000000000000008aULL, (uint64_t)0x0000000000000088ULL,+ (uint64_t)0x0000000080008009ULL, (uint64_t)0x000000008000000aULL,+ (uint64_t)0x000000008000808bULL, (uint64_t)0x800000000000008bULL,+ (uint64_t)0x8000000000008089ULL, (uint64_t)0x8000000000008003ULL,+ (uint64_t)0x8000000000008002ULL, (uint64_t)0x8000000000000080ULL,+ (uint64_t)0x000000000000800aULL, (uint64_t)0x800000008000000aULL,+ (uint64_t)0x8000000080008081ULL, (uint64_t)0x8000000000008080ULL,+ (uint64_t)0x0000000080000001ULL, (uint64_t)0x8000000080008008ULL};++MLK_STATIC_TESTABLE+void mlk_keccakf1600_permute_c(uint64_t *state)+__contract__(+ requires(memory_no_alias(state, sizeof(uint64_t) * MLK_KECCAK_LANES))+ assigns(memory_slice(state, sizeof(uint64_t) * MLK_KECCAK_LANES))+)+{+ unsigned round;++ uint64_t Aba, Abe, Abi, Abo, Abu;+ uint64_t Aga, Age, Agi, Ago, Agu;+ uint64_t Aka, Ake, Aki, Ako, Aku;+ uint64_t Ama, Ame, Ami, Amo, Amu;+ uint64_t Asa, Ase, Asi, Aso, Asu;+ uint64_t BCa, BCe, BCi, BCo, BCu;+ uint64_t Da, De, Di, Do, Du;+ uint64_t Eba, Ebe, Ebi, Ebo, Ebu;+ uint64_t Ega, Ege, Egi, Ego, Egu;+ uint64_t Eka, Eke, Eki, Eko, Eku;+ uint64_t Ema, Eme, Emi, Emo, Emu;+ uint64_t Esa, Ese, Esi, Eso, Esu;++ /* copyFromState(A, state) */+ Aba = state[0];+ Abe = state[1];+ Abi = state[2];+ Abo = state[3];+ Abu = state[4];+ Aga = state[5];+ Age = state[6];+ Agi = state[7];+ Ago = state[8];+ Agu = state[9];+ Aka = state[10];+ Ake = state[11];+ Aki = state[12];+ Ako = state[13];+ Aku = state[14];+ Ama = state[15];+ Ame = state[16];+ Ami = state[17];+ Amo = state[18];+ Amu = state[19];+ Asa = state[20];+ Ase = state[21];+ Asi = state[22];+ Aso = state[23];+ Asu = state[24];++ for (round = 0; round < MLK_KECCAK_NROUNDS; round += 2)+ __loop__(invariant(round <= MLK_KECCAK_NROUNDS && round % 2 == 0)+ decreases(MLK_KECCAK_NROUNDS - round))+ {+ /* prepareTheta */+ BCa = Aba ^ Aga ^ Aka ^ Ama ^ Asa;+ BCe = Abe ^ Age ^ Ake ^ Ame ^ Ase;+ BCi = Abi ^ Agi ^ Aki ^ Ami ^ Asi;+ BCo = Abo ^ Ago ^ Ako ^ Amo ^ Aso;+ BCu = Abu ^ Agu ^ Aku ^ Amu ^ Asu;++ /* thetaRhoPiChiIotaPrepareTheta(round, A, E) */+ Da = BCu ^ MLK_KECCAK_ROL(BCe, 1);+ De = BCa ^ MLK_KECCAK_ROL(BCi, 1);+ Di = BCe ^ MLK_KECCAK_ROL(BCo, 1);+ Do = BCi ^ MLK_KECCAK_ROL(BCu, 1);+ Du = BCo ^ MLK_KECCAK_ROL(BCa, 1);++ Aba ^= Da;+ BCa = Aba;+ Age ^= De;+ BCe = MLK_KECCAK_ROL(Age, 44);+ Aki ^= Di;+ BCi = MLK_KECCAK_ROL(Aki, 43);+ Amo ^= Do;+ BCo = MLK_KECCAK_ROL(Amo, 21);+ Asu ^= Du;+ BCu = MLK_KECCAK_ROL(Asu, 14);+ Eba = BCa ^ ((~BCe) & BCi);+ Eba ^= (uint64_t)mlk_KeccakF_RoundConstants[round];+ Ebe = BCe ^ ((~BCi) & BCo);+ Ebi = BCi ^ ((~BCo) & BCu);+ Ebo = BCo ^ ((~BCu) & BCa);+ Ebu = BCu ^ ((~BCa) & BCe);++ Abo ^= Do;+ BCa = MLK_KECCAK_ROL(Abo, 28);+ Agu ^= Du;+ BCe = MLK_KECCAK_ROL(Agu, 20);+ Aka ^= Da;+ BCi = MLK_KECCAK_ROL(Aka, 3);+ Ame ^= De;+ BCo = MLK_KECCAK_ROL(Ame, 45);+ Asi ^= Di;+ BCu = MLK_KECCAK_ROL(Asi, 61);+ Ega = BCa ^ ((~BCe) & BCi);+ Ege = BCe ^ ((~BCi) & BCo);+ Egi = BCi ^ ((~BCo) & BCu);+ Ego = BCo ^ ((~BCu) & BCa);+ Egu = BCu ^ ((~BCa) & BCe);++ Abe ^= De;+ BCa = MLK_KECCAK_ROL(Abe, 1);+ Agi ^= Di;+ BCe = MLK_KECCAK_ROL(Agi, 6);+ Ako ^= Do;+ BCi = MLK_KECCAK_ROL(Ako, 25);+ Amu ^= Du;+ BCo = MLK_KECCAK_ROL(Amu, 8);+ Asa ^= Da;+ BCu = MLK_KECCAK_ROL(Asa, 18);+ Eka = BCa ^ ((~BCe) & BCi);+ Eke = BCe ^ ((~BCi) & BCo);+ Eki = BCi ^ ((~BCo) & BCu);+ Eko = BCo ^ ((~BCu) & BCa);+ Eku = BCu ^ ((~BCa) & BCe);++ Abu ^= Du;+ BCa = MLK_KECCAK_ROL(Abu, 27);+ Aga ^= Da;+ BCe = MLK_KECCAK_ROL(Aga, 36);+ Ake ^= De;+ BCi = MLK_KECCAK_ROL(Ake, 10);+ Ami ^= Di;+ BCo = MLK_KECCAK_ROL(Ami, 15);+ Aso ^= Do;+ BCu = MLK_KECCAK_ROL(Aso, 56);+ Ema = BCa ^ ((~BCe) & BCi);+ Eme = BCe ^ ((~BCi) & BCo);+ Emi = BCi ^ ((~BCo) & BCu);+ Emo = BCo ^ ((~BCu) & BCa);+ Emu = BCu ^ ((~BCa) & BCe);++ Abi ^= Di;+ BCa = MLK_KECCAK_ROL(Abi, 62);+ Ago ^= Do;+ BCe = MLK_KECCAK_ROL(Ago, 55);+ Aku ^= Du;+ BCi = MLK_KECCAK_ROL(Aku, 39);+ Ama ^= Da;+ BCo = MLK_KECCAK_ROL(Ama, 41);+ Ase ^= De;+ BCu = MLK_KECCAK_ROL(Ase, 2);+ Esa = BCa ^ ((~BCe) & BCi);+ Ese = BCe ^ ((~BCi) & BCo);+ Esi = BCi ^ ((~BCo) & BCu);+ Eso = BCo ^ ((~BCu) & BCa);+ Esu = BCu ^ ((~BCa) & BCe);++ /* prepareTheta */+ BCa = Eba ^ Ega ^ Eka ^ Ema ^ Esa;+ BCe = Ebe ^ Ege ^ Eke ^ Eme ^ Ese;+ BCi = Ebi ^ Egi ^ Eki ^ Emi ^ Esi;+ BCo = Ebo ^ Ego ^ Eko ^ Emo ^ Eso;+ BCu = Ebu ^ Egu ^ Eku ^ Emu ^ Esu;++ /* thetaRhoPiChiIotaPrepareTheta(round+1, E, A) */+ Da = BCu ^ MLK_KECCAK_ROL(BCe, 1);+ De = BCa ^ MLK_KECCAK_ROL(BCi, 1);+ Di = BCe ^ MLK_KECCAK_ROL(BCo, 1);+ Do = BCi ^ MLK_KECCAK_ROL(BCu, 1);+ Du = BCo ^ MLK_KECCAK_ROL(BCa, 1);++ Eba ^= Da;+ BCa = Eba;+ Ege ^= De;+ BCe = MLK_KECCAK_ROL(Ege, 44);+ Eki ^= Di;+ BCi = MLK_KECCAK_ROL(Eki, 43);+ Emo ^= Do;+ BCo = MLK_KECCAK_ROL(Emo, 21);+ Esu ^= Du;+ BCu = MLK_KECCAK_ROL(Esu, 14);+ Aba = BCa ^ ((~BCe) & BCi);+ Aba ^= (uint64_t)mlk_KeccakF_RoundConstants[round + 1];+ Abe = BCe ^ ((~BCi) & BCo);+ Abi = BCi ^ ((~BCo) & BCu);+ Abo = BCo ^ ((~BCu) & BCa);+ Abu = BCu ^ ((~BCa) & BCe);++ Ebo ^= Do;+ BCa = MLK_KECCAK_ROL(Ebo, 28);+ Egu ^= Du;+ BCe = MLK_KECCAK_ROL(Egu, 20);+ Eka ^= Da;+ BCi = MLK_KECCAK_ROL(Eka, 3);+ Eme ^= De;+ BCo = MLK_KECCAK_ROL(Eme, 45);+ Esi ^= Di;+ BCu = MLK_KECCAK_ROL(Esi, 61);+ Aga = BCa ^ ((~BCe) & BCi);+ Age = BCe ^ ((~BCi) & BCo);+ Agi = BCi ^ ((~BCo) & BCu);+ Ago = BCo ^ ((~BCu) & BCa);+ Agu = BCu ^ ((~BCa) & BCe);++ Ebe ^= De;+ BCa = MLK_KECCAK_ROL(Ebe, 1);+ Egi ^= Di;+ BCe = MLK_KECCAK_ROL(Egi, 6);+ Eko ^= Do;+ BCi = MLK_KECCAK_ROL(Eko, 25);+ Emu ^= Du;+ BCo = MLK_KECCAK_ROL(Emu, 8);+ Esa ^= Da;+ BCu = MLK_KECCAK_ROL(Esa, 18);+ Aka = BCa ^ ((~BCe) & BCi);+ Ake = BCe ^ ((~BCi) & BCo);+ Aki = BCi ^ ((~BCo) & BCu);+ Ako = BCo ^ ((~BCu) & BCa);+ Aku = BCu ^ ((~BCa) & BCe);++ Ebu ^= Du;+ BCa = MLK_KECCAK_ROL(Ebu, 27);+ Ega ^= Da;+ BCe = MLK_KECCAK_ROL(Ega, 36);+ Eke ^= De;+ BCi = MLK_KECCAK_ROL(Eke, 10);+ Emi ^= Di;+ BCo = MLK_KECCAK_ROL(Emi, 15);+ Eso ^= Do;+ BCu = MLK_KECCAK_ROL(Eso, 56);+ Ama = BCa ^ ((~BCe) & BCi);+ Ame = BCe ^ ((~BCi) & BCo);+ Ami = BCi ^ ((~BCo) & BCu);+ Amo = BCo ^ ((~BCu) & BCa);+ Amu = BCu ^ ((~BCa) & BCe);++ Ebi ^= Di;+ BCa = MLK_KECCAK_ROL(Ebi, 62);+ Ego ^= Do;+ BCe = MLK_KECCAK_ROL(Ego, 55);+ Eku ^= Du;+ BCi = MLK_KECCAK_ROL(Eku, 39);+ Ema ^= Da;+ BCo = MLK_KECCAK_ROL(Ema, 41);+ Ese ^= De;+ BCu = MLK_KECCAK_ROL(Ese, 2);+ Asa = BCa ^ ((~BCe) & BCi);+ Ase = BCe ^ ((~BCi) & BCo);+ Asi = BCi ^ ((~BCo) & BCu);+ Aso = BCo ^ ((~BCu) & BCa);+ Asu = BCu ^ ((~BCa) & BCe);+ }++ /* copyToState(state, A) */+ state[0] = Aba;+ state[1] = Abe;+ state[2] = Abi;+ state[3] = Abo;+ state[4] = Abu;+ state[5] = Aga;+ state[6] = Age;+ state[7] = Agi;+ state[8] = Ago;+ state[9] = Agu;+ state[10] = Aka;+ state[11] = Ake;+ state[12] = Aki;+ state[13] = Ako;+ state[14] = Aku;+ state[15] = Ama;+ state[16] = Ame;+ state[17] = Ami;+ state[18] = Amo;+ state[19] = Amu;+ state[20] = Asa;+ state[21] = Ase;+ state[22] = Asi;+ state[23] = Aso;+ state[24] = Asu;+}++void mlk_keccakf1600_permute(uint64_t *state)+{+#if defined(MLK_USE_NATIVE_FIPS202_X1)+ if (mlk_keccak_f1600_x1_native(state) == MLK_NATIVE_FUNC_SUCCESS)+ {+ return;+ }+#endif /* MLK_USE_NATIVE_FIPS202_X1 */+ mlk_keccakf1600_permute_c(state);+}++#else /* !MLK_CONFIG_MULTILEVEL_NO_SHARED */++MLK_EMPTY_CU(keccakf1600)++#endif /* MLK_CONFIG_MULTILEVEL_NO_SHARED */++/* To facilitate single-compilation-unit (SCU) builds, undefine all macros.+ * Don't modify by hand -- this is auto-generated by scripts/autogen. */+#undef MLK_KECCAK_NROUNDS+#undef MLK_KECCAK_ROL
@@ -0,0 +1,98 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+#ifndef MLK_FIPS202_KECCAKF1600_H+#define MLK_FIPS202_KECCAKF1600_H+#include "../cbmc.h"+#include "../common.h"++#define MLK_KECCAK_LANES 25+#define MLK_KECCAK_WAY 4++/*+ * WARNING:+ * The contents of this structure, including the placement+ * and interleaving of Keccak lanes, are IMPLEMENTATION-DEFINED.+ * The struct is only exposed here to allow its construction on the stack.+ */++#define mlk_keccakf1600_extract_bytes MLK_NAMESPACE(keccakf1600_extract_bytes)+void mlk_keccakf1600_extract_bytes(uint64_t *state, unsigned char *data,+ unsigned offset, unsigned length)+__contract__(+ requires(0 <= offset && offset <= MLK_KECCAK_LANES * sizeof(uint64_t) &&+ 0 <= length && length <= MLK_KECCAK_LANES * sizeof(uint64_t) - offset)+ requires(memory_no_alias(state, sizeof(uint64_t) * MLK_KECCAK_LANES))+ requires(memory_no_alias(data, length))+ assigns(memory_slice(data, length))+);++#define mlk_keccakf1600_xor_bytes MLK_NAMESPACE(keccakf1600_xor_bytes)+void mlk_keccakf1600_xor_bytes(uint64_t *state, const unsigned char *data,+ unsigned offset, unsigned length)+__contract__(+ requires(0 <= offset && offset <= MLK_KECCAK_LANES * sizeof(uint64_t) &&+ 0 <= length && length <= MLK_KECCAK_LANES * sizeof(uint64_t) - offset)+ requires(memory_no_alias(state, sizeof(uint64_t) * MLK_KECCAK_LANES))+ requires(memory_no_alias(data, length))+ assigns(memory_slice(state, sizeof(uint64_t) * MLK_KECCAK_LANES))+);++#define mlk_keccakf1600x4_extract_bytes \+ MLK_NAMESPACE(keccakf1600x4_extract_bytes)+void mlk_keccakf1600x4_extract_bytes(uint64_t *state, unsigned char *data0,+ unsigned char *data1, unsigned char *data2,+ unsigned char *data3, unsigned offset,+ unsigned length)+__contract__(+ requires(0 <= offset && offset <= MLK_KECCAK_LANES * sizeof(uint64_t) &&+ 0 <= length && length <= MLK_KECCAK_LANES * sizeof(uint64_t) - offset)+ requires(memory_no_alias(state, sizeof(uint64_t) * MLK_KECCAK_LANES * MLK_KECCAK_WAY))+ requires(memory_no_alias(data0, length))+ requires(memory_no_alias(data1, length))+ requires(memory_no_alias(data2, length))+ requires(memory_no_alias(data3, length))+ assigns(memory_slice(data0, length))+ assigns(memory_slice(data1, length))+ assigns(memory_slice(data2, length))+ assigns(memory_slice(data3, length))+);++#define mlk_keccakf1600x4_xor_bytes MLK_NAMESPACE(keccakf1600x4_xor_bytes)+void mlk_keccakf1600x4_xor_bytes(uint64_t *state, const unsigned char *data0,+ const unsigned char *data1,+ const unsigned char *data2,+ const unsigned char *data3, unsigned offset,+ unsigned length)+__contract__(+ requires(0 <= offset && offset <= MLK_KECCAK_LANES * sizeof(uint64_t) &&+ 0 <= length && length <= MLK_KECCAK_LANES * sizeof(uint64_t) - offset)+ requires(memory_no_alias(state, sizeof(uint64_t) * MLK_KECCAK_LANES * MLK_KECCAK_WAY))+ requires(memory_no_alias(data0, length))+ /* Case 1: all input buffers are distinct; Case 2: All input buffers are the same */+ requires((data0 == data1 &&+ data0 == data2 &&+ data0 == data3) ||+ (memory_no_alias(data1, length) &&+ memory_no_alias(data2, length) &&+ memory_no_alias(data3, length)))+ assigns(memory_slice(state, sizeof(uint64_t) * MLK_KECCAK_LANES * MLK_KECCAK_WAY))+);+++#define mlk_keccakf1600x4_permute MLK_NAMESPACE(keccakf1600x4_permute)+void mlk_keccakf1600x4_permute(uint64_t *state)+__contract__(+ requires(memory_no_alias(state, sizeof(uint64_t) * MLK_KECCAK_LANES * MLK_KECCAK_WAY))+ assigns(memory_slice(state, sizeof(uint64_t) * MLK_KECCAK_LANES * MLK_KECCAK_WAY))+);++#define mlk_keccakf1600_permute MLK_NAMESPACE(keccakf1600_permute)+void mlk_keccakf1600_permute(uint64_t *state)+__contract__(+ requires(memory_no_alias(state, sizeof(uint64_t) * MLK_KECCAK_LANES))+ assigns(memory_slice(state, sizeof(uint64_t) * MLK_KECCAK_LANES))+);++#endif /* !MLK_FIPS202_KECCAKF1600_H */
@@ -0,0 +1,78 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [HYBRID]+ * Hybrid scalar/vector implementations of Keccak and SPHINCS+ on AArch64+ * Becker, Kannwischer+ * https://eprint.iacr.org/2022/1243+ */++#ifndef MLK_FIPS202_NATIVE_AARCH64_AUTO_H+#define MLK_FIPS202_NATIVE_AARCH64_AUTO_H+/* Default FIPS202 assembly profile for AArch64 systems */++/*+ * Default logic to decide which implementation to use.+ *+ */++/*+ * Keccak-f1600+ *+ * - On Arm-based Apple CPUs, or if MLK_SYS_AARCH64_FAST_SHA3 is set,+ * we pick a pure Neon implementation.+ * - Otherwise, unless MLK_SYS_AARCH64_SLOW_BARREL_SHIFTER is set,+ * we use lazy-rotation scalar assembly from @[HYBRID].+ * - Otherwise, if MLK_SYS_AARCH64_SLOW_BARREL_SHIFTER is set, we+ * fall back to the standard C implementation.+ */+#if defined(__ARM_FEATURE_SHA3) && \+ (defined(__APPLE__) || defined(MLK_SYS_AARCH64_FAST_SHA3))+#include "x1_v84a.h"+#elif !defined(MLK_SYS_AARCH64_SLOW_BARREL_SHIFTER)+#include "x1_scalar.h"+#endif++/* Batched, SIMD-based Keccak-f1600 implementations. */+#if defined(MLK_SYS_AARCH64_NEON)++/*+ * Keccak-f1600x2/x4+ *+ * The optimal implementation is highly CPU-specific; see @[HYBRID].+ *+ * For now, if v8.4-A is not implemented, we fall back to Keccak-f1600.+ * If v8.4-A is implemented and we are on an Apple CPU or+ * MLK_SYS_AARCH64_FAST_SHA3 is set, we use a plain Neon-based+ * implementation.+ * Otherwise, if v8.4-A is implemented, we use a scalar/Neon/Neon hybrid.+ * The reason for this distinction is that Apple CPUs (and CPUs flagged with+ * MLK_SYS_AARCH64_FAST_SHA3) implement the SHA3 instructions on all SIMD+ * units, while Arm CPUs prior to Cortex-X4 don't, and ordinary Neon+ * instructions are still needed.+ */+#if defined(__ARM_FEATURE_SHA3)+/*+ * For Apple-M cores (and CPUs flagged with MLK_SYS_AARCH64_FAST_SHA3), we+ * use a plain implementation leveraging SHA3 instructions only.+ */+#if defined(__APPLE__) || defined(MLK_SYS_AARCH64_FAST_SHA3)+#include "x2_v84a.h"+#else+#include "x4_v8a_v84a_scalar.h"+#endif++#else /* __ARM_FEATURE_SHA3 */++#include "x4_v8a_scalar.h"++#endif /* !__ARM_FEATURE_SHA3 */++#endif /* MLK_SYS_AARCH64_NEON */++#endif /* !MLK_FIPS202_NATIVE_AARCH64_AUTO_H */
@@ -0,0 +1,80 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+#ifndef MLK_FIPS202_NATIVE_AARCH64_SRC_FIPS202_NATIVE_AARCH64_H+#define MLK_FIPS202_NATIVE_AARCH64_SRC_FIPS202_NATIVE_AARCH64_H++#include "../../../../cbmc.h"+#include "../../../../common.h"++#define mlk_keccakf1600_round_constants \+ MLK_NAMESPACE(keccakf1600_round_constants)+MLK_INTERNAL_DATA_DECLARATION const uint64_t+ mlk_keccakf1600_round_constants[24];++#define mlk_keccak_f1600_x1_scalar_aarch64_asm \+ MLK_NAMESPACE(keccak_f1600_x1_scalar_aarch64_asm)+void mlk_keccak_f1600_x1_scalar_aarch64_asm(uint64_t state[25],+ const uint64_t rc[24])+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/aarch64/proofs/keccak_f1600_x1_scalar_aarch64_asm.ml */+__contract__(+ requires(memory_no_alias(state, sizeof(uint64_t) * 25 * 1))+ requires(rc == mlk_keccakf1600_round_constants)+ assigns(memory_slice(state, sizeof(uint64_t) * 25 * 1))+);++#define mlk_keccak_f1600_x1_v84a_aarch64_asm \+ MLK_NAMESPACE(keccak_f1600_x1_v84a_aarch64_asm)+void mlk_keccak_f1600_x1_v84a_aarch64_asm(uint64_t state[25],+ const uint64_t rc[24])+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/aarch64/proofs/keccak_f1600_x1_v84a_aarch64_asm.ml */+__contract__(+ requires(memory_no_alias(state, sizeof(uint64_t) * 25 * 1))+ requires(rc == mlk_keccakf1600_round_constants)+ assigns(memory_slice(state, sizeof(uint64_t) * 25 * 1))+);++#define mlk_keccak_f1600_x2_v84a_aarch64_asm \+ MLK_NAMESPACE(keccak_f1600_x2_v84a_aarch64_asm)+void mlk_keccak_f1600_x2_v84a_aarch64_asm(uint64_t state[50],+ const uint64_t rc[24])+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/aarch64/proofs/keccak_f1600_x2_v84a_aarch64_asm.ml */+__contract__(+ requires(memory_no_alias(state, sizeof(uint64_t) * 25 * 2))+ requires(rc == mlk_keccakf1600_round_constants)+ assigns(memory_slice(state, sizeof(uint64_t) * 25 * 2))+);++#define mlk_keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm \+ MLK_NAMESPACE(keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm)+void mlk_keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm(uint64_t state[100],+ const uint64_t rc[24])+/* This must be kept in sync with the HOL-Light specification+ * in+ * proofs/hol_light/aarch64/proofs/keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm.ml+ */+__contract__(+ requires(memory_no_alias(state, sizeof(uint64_t) * 25 * 4))+ requires(rc == mlk_keccakf1600_round_constants)+ assigns(memory_slice(state, sizeof(uint64_t) * 25 * 4))+);++#define mlk_keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm \+ MLK_NAMESPACE(keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm)+void mlk_keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm(+ uint64_t state[100], const uint64_t rc[24])+/* This must be kept in sync with the HOL-Light specification+ * in+ * proofs/hol_light/aarch64/proofs/keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm.ml+ */+__contract__(+ requires(memory_no_alias(state, sizeof(uint64_t) * 25 * 4))+ requires(rc == mlk_keccakf1600_round_constants)+ assigns(memory_slice(state, sizeof(uint64_t) * 25 * 4))+);++#endif /* !MLK_FIPS202_NATIVE_AARCH64_SRC_FIPS202_NATIVE_AARCH64_H */
@@ -0,0 +1,377 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) 2021-2022 Arm Limited+ * Copyright (c) 2022 Matthias Kannwischer+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++// Author: Hanno Becker <hanno.becker@arm.com>+// Author: Matthias Kannwischer <matthias@kannwischer.eu>++/*yaml+ Name: keccak_f1600_x1_scalar_aarch64_asm+ Description: AArch64 scalar implementation of Keccak-f[1600] permutation for single state+ Signature: void mlk_keccak_f1600_x1_scalar_aarch64_asm(uint64_t state[25], const uint64_t rc[24])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ x0:+ type: buffer+ size_bytes: 200+ permissions: read/write+ c_parameter: uint64_t state[25]+ description: Keccak state (25 x uint64_t)+ x1:+ type: buffer+ size_bytes: 192+ permissions: read-only+ c_parameter: uint64_t const *rc+ description: Round constants (24 x uint64_t)+ Stack:+ bytes: 128+ description: register preservation and temporary storage+*/++#include "../../../../common.h"+#if defined(MLK_FIPS202_AARCH64_NEED_X1_SCALAR) && \+ !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED)++/*+ * WARNING: This file is auto-derived from the mlkem-native source file+ * dev/fips202/aarch64/src/keccak_f1600_x1_scalar_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLK_ASM_NAMESPACE(keccak_f1600_x1_scalar_aarch64_asm)+MLK_ASM_FN_SYMBOL(keccak_f1600_x1_scalar_aarch64_asm)++ .cfi_startproc+ sub sp, sp, #0x80+ .cfi_adjust_cfa_offset 0x80+ stp x19, x20, [sp, #0x20]+ .cfi_rel_offset x19, 0x20+ .cfi_rel_offset x20, 0x28+ stp x21, x22, [sp, #0x30]+ .cfi_rel_offset x21, 0x30+ .cfi_rel_offset x22, 0x38+ stp x23, x24, [sp, #0x40]+ .cfi_rel_offset x23, 0x40+ .cfi_rel_offset x24, 0x48+ stp x25, x26, [sp, #0x50]+ .cfi_rel_offset x25, 0x50+ .cfi_rel_offset x26, 0x58+ stp x27, x28, [sp, #0x60]+ .cfi_rel_offset x27, 0x60+ .cfi_rel_offset x28, 0x68+ stp x29, x30, [sp, #0x70]+ .cfi_rel_offset x29, 0x70+ .cfi_rel_offset x30, 0x78++Lmlk_keccak_f1600_x1_scalar_initial:+ mov x26, x1+ str x1, [sp, #0x8]+ ldp x1, x6, [x0]+ ldp x11, x16, [x0, #0x10]+ ldp x21, x2, [x0, #0x20]+ ldp x7, x12, [x0, #0x30]+ ldp x17, x22, [x0, #0x40]+ ldp x3, x8, [x0, #0x50]+ ldp x13, x28, [x0, #0x60]+ ldp x23, x4, [x0, #0x70]+ ldp x9, x14, [x0, #0x80]+ ldp x19, x24, [x0, #0x90]+ ldp x5, x10, [x0, #0xa0]+ ldp x15, x20, [x0, #0xb0]+ ldr x25, [x0, #0xc0]+ str x0, [sp]+ eor x30, x24, x25+ eor x27, x9, x10+ eor x0, x30, x21+ eor x26, x27, x6+ eor x27, x26, x7+ eor x29, x0, x22+ eor x26, x29, x23+ eor x29, x4, x5+ eor x30, x29, x1+ eor x0, x27, x8+ eor x29, x30, x2+ eor x30, x19, x20+ eor x30, x30, x16+ eor x27, x26, x0, ror #63+ eor x4, x4, x27+ eor x30, x30, x17+ eor x30, x30, x28+ eor x29, x29, x3+ eor x0, x0, x30, ror #63+ eor x30, x30, x29, ror #63+ eor x22, x22, x30+ eor x23, x23, x30+ str x23, [sp, #0x18]+ eor x23, x14, x15+ eor x14, x14, x0+ eor x23, x23, x11+ eor x15, x15, x0+ eor x1, x1, x27+ eor x23, x23, x12+ eor x23, x23, x13+ eor x11, x11, x0+ eor x29, x29, x23, ror #63+ eor x23, x23, x26, ror #63+ eor x26, x13, x0+ eor x13, x28, x23+ eor x28, x24, x30+ eor x24, x16, x23+ eor x16, x21, x30+ eor x21, x25, x30+ eor x30, x19, x23+ eor x19, x20, x23+ eor x20, x17, x23+ eor x17, x12, x0+ eor x0, x2, x27+ eor x2, x6, x29+ eor x6, x8, x29+ bic x8, x28, x13, ror #47+ eor x12, x3, x27+ bic x3, x13, x17, ror #19+ eor x5, x5, x27+ ldr x27, [sp, #0x18]+ bic x25, x17, x2, ror #5+ eor x9, x9, x29+ eor x23, x25, x5, ror #52+ eor x3, x3, x2, ror #24+ eor x8, x8, x17, ror #2+ eor x17, x10, x29+ bic x25, x12, x22, ror #47+ eor x29, x7, x29+ bic x10, x4, x27, ror #2+ bic x7, x5, x28, ror #10+ eor x10, x10, x20, ror #50+ eor x13, x7, x13, ror #57+ bic x7, x2, x5, ror #47+ eor x2, x25, x24, ror #39+ bic x25, x20, x11, ror #57+ bic x5, x17, x4, ror #25+ eor x25, x25, x17, ror #53+ bic x17, x11, x17, ror #60+ eor x28, x7, x28, ror #57+ bic x7, x9, x12, ror #42+ eor x7, x7, x22, ror #25+ bic x22, x22, x24, ror #56+ bic x24, x24, x15, ror #31+ eor x22, x22, x15, ror #23+ bic x20, x27, x20, ror #48+ bic x15, x15, x9, ror #16+ eor x12, x15, x12, ror #58+ eor x15, x5, x27, ror #27+ eor x5, x20, x11, ror #41+ ldr x11, [sp, #0x8]+ eor x20, x17, x4, ror #21+ eor x17, x24, x9, ror #47+ mov x24, #0x1 // =1+ bic x9, x0, x16, ror #9+ str x24, [sp, #0x10]+ bic x24, x29, x1, ror #44+ bic x27, x1, x21, ror #50+ bic x4, x26, x29, ror #63+ eor x1, x1, x4, ror #21+ ldr x11, [x11]+ bic x4, x21, x30, ror #57+ eor x21, x24, x21, ror #30+ eor x24, x9, x19, ror #44+ bic x9, x14, x6, ror #5+ eor x9, x9, x0, ror #43+ bic x0, x6, x0, ror #38+ eor x1, x1, x11+ eor x11, x4, x26, ror #35+ eor x4, x0, x16, ror #47+ bic x0, x16, x19, ror #35+ eor x16, x27, x30, ror #43+ bic x27, x30, x26, ror #42+ bic x26, x19, x14, ror #41+ eor x19, x0, x14, ror #12+ eor x14, x26, x6, ror #46+ eor x6, x27, x29, ror #41++Lmlk_keccak_f1600_x1_scalar_loop:+ eor x0, x15, x11, ror #52+ eor x0, x0, x13, ror #48+ eor x26, x8, x9, ror #57+ eor x27, x0, x14, ror #10+ eor x29, x16, x28, ror #63+ eor x26, x26, x6, ror #51+ eor x30, x23, x22, ror #50+ eor x0, x26, x10, ror #31+ eor x29, x29, x19, ror #37+ eor x27, x27, x12, ror #5+ eor x30, x30, x24, ror #34+ eor x0, x0, x7, ror #27+ eor x26, x30, x21, ror #26+ eor x26, x26, x25, ror #15+ ror x30, x27, #0x3e+ eor x30, x30, x26, ror #57+ ror x26, x26, #0x3a+ eor x16, x30, x16+ eor x28, x30, x28, ror #63+ str x28, [sp, #0x18]+ eor x29, x29, x17, ror #36+ eor x28, x1, x2, ror #61+ eor x19, x30, x19, ror #37+ eor x29, x29, x20, ror #2+ eor x28, x28, x4, ror #54+ eor x26, x26, x0, ror #55+ eor x28, x28, x3, ror #39+ eor x28, x28, x5, ror #25+ ror x0, x0, #0x38+ eor x0, x0, x29, ror #63+ eor x27, x28, x27, ror #61+ eor x13, x0, x13, ror #46+ eor x28, x29, x28, ror #63+ eor x29, x30, x20, ror #2+ eor x20, x26, x3, ror #39+ eor x11, x0, x11, ror #50+ eor x25, x28, x25, ror #9+ eor x3, x28, x21, ror #20+ eor x21, x26, x1+ eor x9, x27, x9, ror #49+ eor x24, x28, x24, ror #28+ eor x1, x30, x17, ror #36+ eor x14, x0, x14, ror #8+ eor x22, x28, x22, ror #44+ eor x8, x27, x8, ror #56+ eor x17, x27, x7, ror #19+ eor x15, x0, x15, ror #62+ bic x7, x20, x22, ror #47+ eor x4, x26, x4, ror #54+ eor x0, x0, x12, ror #3+ eor x28, x28, x23, ror #58+ eor x23, x26, x2, ror #61+ eor x26, x26, x5, ror #25+ eor x2, x7, x16, ror #39+ bic x7, x9, x20, ror #42+ bic x30, x15, x9, ror #16+ eor x7, x7, x22, ror #25+ eor x12, x30, x20, ror #58+ bic x20, x22, x16, ror #56+ eor x30, x27, x6, ror #43+ eor x22, x20, x15, ror #23+ bic x6, x19, x13, ror #42+ eor x6, x6, x17, ror #41+ bic x5, x13, x17, ror #63+ eor x5, x21, x5, ror #21+ bic x17, x17, x21, ror #44+ eor x27, x27, x10, ror #23+ bic x21, x21, x25, ror #50+ bic x20, x27, x4, ror #25+ bic x10, x16, x15, ror #31+ eor x16, x21, x19, ror #43+ eor x21, x17, x25, ror #30+ bic x19, x25, x19, ror #57+ ldr x25, [sp, #0x10]+ eor x17, x10, x9, ror #47+ ldr x9, [sp, #0x8]+ eor x15, x20, x28, ror #27+ bic x20, x4, x28, ror #2+ eor x10, x20, x1, ror #50+ bic x20, x11, x27, ror #60+ eor x20, x20, x4, ror #21+ bic x4, x28, x1, ror #48+ bic x1, x1, x11, ror #57+ ldr x28, [x9, x25, lsl #3]+ ldr x9, [sp, #0x18]+ add x25, x25, #0x1+ str x25, [sp, #0x10]+ cmp x25, #0x17+ eor x25, x1, x27, ror #53+ bic x27, x30, x26, ror #47+ eor x1, x5, x28+ eor x5, x4, x11, ror #41+ eor x11, x19, x13, ror #35+ bic x13, x26, x24, ror #10+ eor x28, x27, x24, ror #57+ bic x27, x24, x9, ror #47+ bic x19, x23, x3, ror #9+ bic x4, x29, x14, ror #41+ eor x24, x19, x29, ror #44+ bic x29, x3, x29, ror #35+ eor x13, x13, x9, ror #57+ eor x19, x29, x14, ror #12+ bic x29, x9, x0, ror #19+ bic x14, x14, x8, ror #5+ eor x9, x14, x23, ror #43+ eor x14, x4, x8, ror #46+ bic x23, x8, x23, ror #38+ eor x8, x27, x0, ror #2+ eor x4, x23, x3, ror #47+ bic x3, x0, x30, ror #5+ eor x23, x3, x26, ror #52+ eor x3, x29, x30, ror #24+ b.le Lmlk_keccak_f1600_x1_scalar_loop+ ror x6, x6, #0x2b+ ror x11, x11, #0x32+ ror x21, x21, #0x14+ ror x2, x2, #0x3d+ ror x7, x7, #0x13+ ror x12, x12, #0x3+ ror x17, x17, #0x24+ ror x22, x22, #0x2c+ ror x3, x3, #0x27+ ror x8, x8, #0x38+ ror x13, x13, #0x2e+ ror x28, x28, #0x3f+ ror x23, x23, #0x3a+ ror x4, x4, #0x36+ ror x9, x9, #0x31+ ror x14, x14, #0x8+ ror x19, x19, #0x25+ ror x24, x24, #0x1c+ ror x5, x5, #0x19+ ror x10, x10, #0x17+ ror x15, x15, #0x3e+ ror x20, x20, #0x2+ ror x25, x25, #0x9+ ldr x0, [sp]+ stp x1, x6, [x0]+ stp x11, x16, [x0, #0x10]+ stp x21, x2, [x0, #0x20]+ stp x7, x12, [x0, #0x30]+ stp x17, x22, [x0, #0x40]+ stp x3, x8, [x0, #0x50]+ stp x13, x28, [x0, #0x60]+ stp x23, x4, [x0, #0x70]+ stp x9, x14, [x0, #0x80]+ stp x19, x24, [x0, #0x90]+ stp x5, x10, [x0, #0xa0]+ stp x15, x20, [x0, #0xb0]+ str x25, [x0, #0xc0]+ ldp x19, x20, [sp, #0x20]+ .cfi_restore x19+ .cfi_restore x20+ ldp x21, x22, [sp, #0x30]+ .cfi_restore x21+ .cfi_restore x22+ ldp x23, x24, [sp, #0x40]+ .cfi_restore x23+ .cfi_restore x24+ ldp x25, x26, [sp, #0x50]+ .cfi_restore x25+ .cfi_restore x26+ ldp x27, x28, [sp, #0x60]+ .cfi_restore x27+ .cfi_restore x28+ ldp x29, x30, [sp, #0x70]+ .cfi_restore x29+ .cfi_restore x30+ add sp, sp, #0x80+ .cfi_adjust_cfa_offset -0x80+ ret+ .cfi_endproc++MLK_ASM_FN_SIZE(keccak_f1600_x1_scalar_aarch64_asm)++#endif /* MLK_FIPS202_AARCH64_NEED_X1_SCALAR && \+ !MLK_CONFIG_MULTILEVEL_NO_SHARED */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
@@ -0,0 +1,206 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) 2021-2022 Arm Limited+ * Copyright (c) 2022 Matthias Kannwischer+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [HYBRID]+ * Hybrid scalar/vector implementations of Keccak and SPHINCS+ on AArch64+ * Becker, Kannwischer+ * https://eprint.iacr.org/2022/1243+ */++/*yaml+ Name: keccak_f1600_x1_v84a_aarch64_asm+ Description: AArch64 ARMv8.4-A implementation of Keccak-f[1600] permutation for single state+ Signature: void mlk_keccak_f1600_x1_v84a_aarch64_asm(uint64_t state[25], const uint64_t rc[24])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON, SHA3]+ x0:+ type: buffer+ size_bytes: 200+ permissions: read/write+ c_parameter: uint64_t state[25]+ description: Keccak state (25 x uint64_t)+ x1:+ type: buffer+ size_bytes: 192+ permissions: read-only+ c_parameter: const uint64_t rc[24]+ description: Round constants (24 x uint64_t)+ Stack:+ bytes: 64+ description: register preservation+*/++//+// Author: Hanno Becker <hanno.becker@arm.com>+// Author: Matthias Kannwischer <matthias@kannwischer.eu>+//+// This implementation is essentially from the paper @[HYBRID].+// The only difference is interleaving/deinterleaving of Keccak state+// during load and store, so that the caller need not do this.+//++#include "../../../../common.h"+#if defined(MLK_FIPS202_AARCH64_NEED_X1_V84A) && \+ !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED)++#if defined(__ARM_FEATURE_SHA3)++/*+ * WARNING: This file is auto-derived from the mlkem-native source file+ * dev/fips202/aarch64/src/keccak_f1600_x1_v84a_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLK_ASM_NAMESPACE(keccak_f1600_x1_v84a_aarch64_asm)+MLK_ASM_FN_SYMBOL(keccak_f1600_x1_v84a_aarch64_asm)++ .cfi_startproc+ sub sp, sp, #0x40+ .cfi_adjust_cfa_offset 0x40+ stp d8, d9, [sp]+ .cfi_rel_offset d8, 0x0+ .cfi_rel_offset d9, 0x8+ stp d10, d11, [sp, #0x10]+ .cfi_rel_offset d10, 0x10+ .cfi_rel_offset d11, 0x18+ stp d12, d13, [sp, #0x20]+ .cfi_rel_offset d12, 0x20+ .cfi_rel_offset d13, 0x28+ stp d14, d15, [sp, #0x30]+ .cfi_rel_offset d14, 0x30+ .cfi_rel_offset d15, 0x38+ ldp d0, d1, [x0]+ ldp d2, d3, [x0, #0x10]+ ldp d4, d5, [x0, #0x20]+ ldp d6, d7, [x0, #0x30]+ ldp d8, d9, [x0, #0x40]+ ldp d10, d11, [x0, #0x50]+ ldp d12, d13, [x0, #0x60]+ ldp d14, d15, [x0, #0x70]+ ldp d16, d17, [x0, #0x80]+ ldp d18, d19, [x0, #0x90]+ ldp d20, d21, [x0, #0xa0]+ ldp d22, d23, [x0, #0xb0]+ ldr d24, [x0, #0xc0]+ mov x2, #0x18 // =24++Lmlk_keccak_f1600_x1_v84a_loop:+ eor3 v30.16b, v0.16b, v5.16b, v10.16b+ eor3 v29.16b, v1.16b, v6.16b, v11.16b+ eor3 v28.16b, v2.16b, v7.16b, v12.16b+ eor3 v27.16b, v3.16b, v8.16b, v13.16b+ eor3 v26.16b, v4.16b, v9.16b, v14.16b+ eor3 v30.16b, v30.16b, v15.16b, v20.16b+ eor3 v29.16b, v29.16b, v16.16b, v21.16b+ eor3 v28.16b, v28.16b, v17.16b, v22.16b+ eor3 v27.16b, v27.16b, v18.16b, v23.16b+ eor3 v26.16b, v26.16b, v19.16b, v24.16b+ rax1 v25.2d, v30.2d, v28.2d+ rax1 v28.2d, v28.2d, v26.2d+ rax1 v26.2d, v26.2d, v29.2d+ rax1 v29.2d, v29.2d, v27.2d+ rax1 v27.2d, v27.2d, v30.2d+ eor v30.16b, v0.16b, v26.16b+ xar v0.2d, v2.2d, v29.2d, #0x2+ xar v2.2d, v12.2d, v29.2d, #0x15+ xar v12.2d, v13.2d, v28.2d, #0x27+ xar v13.2d, v19.2d, v27.2d, #0x38+ xar v19.2d, v23.2d, v28.2d, #0x8+ xar v23.2d, v15.2d, v26.2d, #0x17+ xar v15.2d, v1.2d, v25.2d, #0x3f+ xar v1.2d, v8.2d, v28.2d, #0x9+ xar v8.2d, v16.2d, v25.2d, #0x13+ xar v16.2d, v7.2d, v29.2d, #0x3a+ xar v7.2d, v10.2d, v26.2d, #0x3d+ xar v10.2d, v3.2d, v28.2d, #0x24+ xar v3.2d, v18.2d, v28.2d, #0x2b+ xar v18.2d, v17.2d, v29.2d, #0x31+ xar v17.2d, v11.2d, v25.2d, #0x36+ xar v11.2d, v9.2d, v27.2d, #0x2c+ xar v9.2d, v22.2d, v29.2d, #0x3+ xar v22.2d, v14.2d, v27.2d, #0x19+ xar v14.2d, v20.2d, v26.2d, #0x2e+ xar v20.2d, v4.2d, v27.2d, #0x25+ xar v4.2d, v24.2d, v27.2d, #0x32+ xar v24.2d, v21.2d, v25.2d, #0x3e+ xar v21.2d, v5.2d, v26.2d, #0x1c+ xar v27.2d, v6.2d, v25.2d, #0x14+ ld1r { v31.2d }, [x1], #8+ bcax v5.16b, v10.16b, v7.16b, v11.16b+ bcax v6.16b, v11.16b, v8.16b, v7.16b+ bcax v7.16b, v7.16b, v9.16b, v8.16b+ bcax v8.16b, v8.16b, v10.16b, v9.16b+ bcax v9.16b, v9.16b, v11.16b, v10.16b+ bcax v10.16b, v15.16b, v12.16b, v16.16b+ bcax v11.16b, v16.16b, v13.16b, v12.16b+ bcax v12.16b, v12.16b, v14.16b, v13.16b+ bcax v13.16b, v13.16b, v15.16b, v14.16b+ bcax v14.16b, v14.16b, v16.16b, v15.16b+ bcax v15.16b, v20.16b, v17.16b, v21.16b+ bcax v16.16b, v21.16b, v18.16b, v17.16b+ bcax v17.16b, v17.16b, v19.16b, v18.16b+ bcax v18.16b, v18.16b, v20.16b, v19.16b+ bcax v19.16b, v19.16b, v21.16b, v20.16b+ bcax v20.16b, v0.16b, v22.16b, v1.16b+ bcax v21.16b, v1.16b, v23.16b, v22.16b+ bcax v22.16b, v22.16b, v24.16b, v23.16b+ bcax v23.16b, v23.16b, v0.16b, v24.16b+ bcax v24.16b, v24.16b, v1.16b, v0.16b+ bcax v0.16b, v30.16b, v2.16b, v27.16b+ bcax v1.16b, v27.16b, v3.16b, v2.16b+ bcax v2.16b, v2.16b, v4.16b, v3.16b+ bcax v3.16b, v3.16b, v30.16b, v4.16b+ bcax v4.16b, v4.16b, v27.16b, v30.16b+ eor v0.16b, v0.16b, v31.16b+ sub x2, x2, #0x1+ cbnz x2, Lmlk_keccak_f1600_x1_v84a_loop+ stp d0, d1, [x0]+ stp d2, d3, [x0, #0x10]+ stp d4, d5, [x0, #0x20]+ stp d6, d7, [x0, #0x30]+ stp d8, d9, [x0, #0x40]+ stp d10, d11, [x0, #0x50]+ stp d12, d13, [x0, #0x60]+ stp d14, d15, [x0, #0x70]+ stp d16, d17, [x0, #0x80]+ stp d18, d19, [x0, #0x90]+ stp d20, d21, [x0, #0xa0]+ stp d22, d23, [x0, #0xb0]+ str d24, [x0, #0xc0]+ ldp d8, d9, [sp]+ .cfi_restore d8+ .cfi_restore d9+ ldp d10, d11, [sp, #0x10]+ .cfi_restore d10+ .cfi_restore d11+ ldp d12, d13, [sp, #0x20]+ .cfi_restore d12+ .cfi_restore d13+ ldp d14, d15, [sp, #0x30]+ .cfi_restore d14+ .cfi_restore d15+ add sp, sp, #0x40+ .cfi_adjust_cfa_offset -0x40+ ret+ .cfi_endproc++MLK_ASM_FN_SIZE(keccak_f1600_x1_v84a_aarch64_asm)++#endif /* __ARM_FEATURE_SHA3 */++#endif /* MLK_FIPS202_AARCH64_NEED_X1_V84A && !MLK_CONFIG_MULTILEVEL_NO_SHARED \+ */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
@@ -0,0 +1,261 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) 2021-2022 Arm Limited+ * Copyright (c) 2022 Matthias Kannwischer+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [HYBRID]+ * Hybrid scalar/vector implementations of Keccak and SPHINCS+ on AArch64+ * Becker, Kannwischer+ * https://eprint.iacr.org/2022/1243+ */++/*yaml+ Name: keccak_f1600_x2_v84a_aarch64_asm+ Description: AArch64 ARMv8.4-A implementation of Keccak-f[1600] permutation for two sequential states+ Signature: void mlk_keccak_f1600_x2_v84a_aarch64_asm(uint64_t state[50], const uint64_t rc[24])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON, SHA3]+ x0:+ type: buffer+ size_bytes: 400+ permissions: read/write+ c_parameter: uint64_t state[50]+ description: Two sequential Keccak states (state0[25], state1[25])+ x1:+ type: buffer+ size_bytes: 192+ permissions: read-only+ c_parameter: const uint64_t rc[24]+ description: Round constants (24 x uint64_t)+ Stack:+ bytes: 64+ description: register preservation+*/++//+// Author: Hanno Becker <hanno.becker@arm.com>+// Author: Matthias Kannwischer <matthias@kannwischer.eu>+//+// This implementation is essentially from the paper @[HYBRID].+// The only difference is interleaving/deinterleaving of Keccak state+// during load and store, so that the caller need not do this.+//++#include "../../../../common.h"+#if defined(MLK_FIPS202_AARCH64_NEED_X2_V84A) && \+ !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED)++#if defined(__ARM_FEATURE_SHA3)++/*+ * WARNING: This file is auto-derived from the mlkem-native source file+ * dev/fips202/aarch64/src/keccak_f1600_x2_v84a_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLK_ASM_NAMESPACE(keccak_f1600_x2_v84a_aarch64_asm)+MLK_ASM_FN_SYMBOL(keccak_f1600_x2_v84a_aarch64_asm)++ .cfi_startproc+ sub sp, sp, #0x40+ .cfi_adjust_cfa_offset 0x40+ stp d8, d9, [sp]+ .cfi_rel_offset d8, 0x0+ .cfi_rel_offset d9, 0x8+ stp d10, d11, [sp, #0x10]+ .cfi_rel_offset d10, 0x10+ .cfi_rel_offset d11, 0x18+ stp d12, d13, [sp, #0x20]+ .cfi_rel_offset d12, 0x20+ .cfi_rel_offset d13, 0x28+ stp d14, d15, [sp, #0x30]+ .cfi_rel_offset d14, 0x30+ .cfi_rel_offset d15, 0x38+ add x2, x0, #0xc8+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x2], #32+ trn1 v0.2d, v25.2d, v27.2d+ trn2 v1.2d, v25.2d, v27.2d+ trn1 v2.2d, v26.2d, v28.2d+ trn2 v3.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x2], #32+ trn1 v4.2d, v25.2d, v27.2d+ trn2 v5.2d, v25.2d, v27.2d+ trn1 v6.2d, v26.2d, v28.2d+ trn2 v7.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x2], #32+ trn1 v8.2d, v25.2d, v27.2d+ trn2 v9.2d, v25.2d, v27.2d+ trn1 v10.2d, v26.2d, v28.2d+ trn2 v11.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x2], #32+ trn1 v12.2d, v25.2d, v27.2d+ trn2 v13.2d, v25.2d, v27.2d+ trn1 v14.2d, v26.2d, v28.2d+ trn2 v15.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x2], #32+ trn1 v16.2d, v25.2d, v27.2d+ trn2 v17.2d, v25.2d, v27.2d+ trn1 v18.2d, v26.2d, v28.2d+ trn2 v19.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x2], #32+ trn1 v20.2d, v25.2d, v27.2d+ trn2 v21.2d, v25.2d, v27.2d+ trn1 v22.2d, v26.2d, v28.2d+ trn2 v23.2d, v26.2d, v28.2d+ ldr d25, [x0]+ ldr d27, [x2]+ trn1 v24.2d, v25.2d, v27.2d+ mov x2, #0x18 // =24++Lmlk_keccak_f1600_x2_v84a_loop:+ eor3 v30.16b, v0.16b, v5.16b, v10.16b+ eor3 v29.16b, v1.16b, v6.16b, v11.16b+ eor3 v28.16b, v2.16b, v7.16b, v12.16b+ eor3 v27.16b, v3.16b, v8.16b, v13.16b+ eor3 v26.16b, v4.16b, v9.16b, v14.16b+ eor3 v30.16b, v30.16b, v15.16b, v20.16b+ eor3 v29.16b, v29.16b, v16.16b, v21.16b+ eor3 v28.16b, v28.16b, v17.16b, v22.16b+ eor3 v27.16b, v27.16b, v18.16b, v23.16b+ eor3 v26.16b, v26.16b, v19.16b, v24.16b+ rax1 v25.2d, v30.2d, v28.2d+ rax1 v28.2d, v28.2d, v26.2d+ rax1 v26.2d, v26.2d, v29.2d+ rax1 v29.2d, v29.2d, v27.2d+ rax1 v27.2d, v27.2d, v30.2d+ eor v30.16b, v0.16b, v26.16b+ xar v0.2d, v2.2d, v29.2d, #0x2+ xar v2.2d, v12.2d, v29.2d, #0x15+ xar v12.2d, v13.2d, v28.2d, #0x27+ xar v13.2d, v19.2d, v27.2d, #0x38+ xar v19.2d, v23.2d, v28.2d, #0x8+ xar v23.2d, v15.2d, v26.2d, #0x17+ xar v15.2d, v1.2d, v25.2d, #0x3f+ xar v1.2d, v8.2d, v28.2d, #0x9+ xar v8.2d, v16.2d, v25.2d, #0x13+ xar v16.2d, v7.2d, v29.2d, #0x3a+ xar v7.2d, v10.2d, v26.2d, #0x3d+ xar v10.2d, v3.2d, v28.2d, #0x24+ xar v3.2d, v18.2d, v28.2d, #0x2b+ xar v18.2d, v17.2d, v29.2d, #0x31+ xar v17.2d, v11.2d, v25.2d, #0x36+ xar v11.2d, v9.2d, v27.2d, #0x2c+ xar v9.2d, v22.2d, v29.2d, #0x3+ xar v22.2d, v14.2d, v27.2d, #0x19+ xar v14.2d, v20.2d, v26.2d, #0x2e+ xar v20.2d, v4.2d, v27.2d, #0x25+ xar v4.2d, v24.2d, v27.2d, #0x32+ xar v24.2d, v21.2d, v25.2d, #0x3e+ xar v21.2d, v5.2d, v26.2d, #0x1c+ xar v27.2d, v6.2d, v25.2d, #0x14+ ld1r { v31.2d }, [x1], #8+ bcax v5.16b, v10.16b, v7.16b, v11.16b+ bcax v6.16b, v11.16b, v8.16b, v7.16b+ bcax v7.16b, v7.16b, v9.16b, v8.16b+ bcax v8.16b, v8.16b, v10.16b, v9.16b+ bcax v9.16b, v9.16b, v11.16b, v10.16b+ bcax v10.16b, v15.16b, v12.16b, v16.16b+ bcax v11.16b, v16.16b, v13.16b, v12.16b+ bcax v12.16b, v12.16b, v14.16b, v13.16b+ bcax v13.16b, v13.16b, v15.16b, v14.16b+ bcax v14.16b, v14.16b, v16.16b, v15.16b+ bcax v15.16b, v20.16b, v17.16b, v21.16b+ bcax v16.16b, v21.16b, v18.16b, v17.16b+ bcax v17.16b, v17.16b, v19.16b, v18.16b+ bcax v18.16b, v18.16b, v20.16b, v19.16b+ bcax v19.16b, v19.16b, v21.16b, v20.16b+ bcax v20.16b, v0.16b, v22.16b, v1.16b+ bcax v21.16b, v1.16b, v23.16b, v22.16b+ bcax v22.16b, v22.16b, v24.16b, v23.16b+ bcax v23.16b, v23.16b, v0.16b, v24.16b+ bcax v24.16b, v24.16b, v1.16b, v0.16b+ bcax v0.16b, v30.16b, v2.16b, v27.16b+ bcax v1.16b, v27.16b, v3.16b, v2.16b+ bcax v2.16b, v2.16b, v4.16b, v3.16b+ bcax v3.16b, v3.16b, v30.16b, v4.16b+ bcax v4.16b, v4.16b, v27.16b, v30.16b+ eor v0.16b, v0.16b, v31.16b+ sub x2, x2, #0x1+ cbnz x2, Lmlk_keccak_f1600_x2_v84a_loop+ sub x0, x0, #0xc0+ add x2, x0, #0xc8+ trn1 v25.2d, v0.2d, v1.2d+ trn1 v26.2d, v2.2d, v3.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v0.2d, v1.2d+ trn2 v28.2d, v2.2d, v3.2d+ st1 { v27.2d, v28.2d }, [x2], #32+ trn1 v25.2d, v4.2d, v5.2d+ trn1 v26.2d, v6.2d, v7.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v4.2d, v5.2d+ trn2 v28.2d, v6.2d, v7.2d+ st1 { v27.2d, v28.2d }, [x2], #32+ trn1 v25.2d, v8.2d, v9.2d+ trn1 v26.2d, v10.2d, v11.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v8.2d, v9.2d+ trn2 v28.2d, v10.2d, v11.2d+ st1 { v27.2d, v28.2d }, [x2], #32+ trn1 v25.2d, v12.2d, v13.2d+ trn1 v26.2d, v14.2d, v15.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v12.2d, v13.2d+ trn2 v28.2d, v14.2d, v15.2d+ st1 { v27.2d, v28.2d }, [x2], #32+ trn1 v25.2d, v16.2d, v17.2d+ trn1 v26.2d, v18.2d, v19.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v16.2d, v17.2d+ trn2 v28.2d, v18.2d, v19.2d+ st1 { v27.2d, v28.2d }, [x2], #32+ trn1 v25.2d, v20.2d, v21.2d+ trn1 v26.2d, v22.2d, v23.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v20.2d, v21.2d+ trn2 v28.2d, v22.2d, v23.2d+ st1 { v27.2d, v28.2d }, [x2], #32+ str d24, [x0]+ trn2 v25.2d, v24.2d, v24.2d+ str d25, [x2]+ ldp d8, d9, [sp]+ .cfi_restore d8+ .cfi_restore d9+ ldp d10, d11, [sp, #0x10]+ .cfi_restore d10+ .cfi_restore d11+ ldp d12, d13, [sp, #0x20]+ .cfi_restore d12+ .cfi_restore d13+ ldp d14, d15, [sp, #0x30]+ .cfi_restore d14+ .cfi_restore d15+ add sp, sp, #0x40+ .cfi_adjust_cfa_offset -0x40+ ret+ .cfi_endproc++MLK_ASM_FN_SIZE(keccak_f1600_x2_v84a_aarch64_asm)++#endif /* __ARM_FEATURE_SHA3 */++#endif /* MLK_FIPS202_AARCH64_NEED_X2_V84A && !MLK_CONFIG_MULTILEVEL_NO_SHARED \+ */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
@@ -0,0 +1,1079 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) 2021-2022 Arm Limited+ * Copyright (c) 2022 Matthias Kannwischer+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++// Author: Hanno Becker <hannobecker@posteo.de>+// Author: Matthias Kannwischer <matthias@kannwischer.eu>++/*yaml+ Name: keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm+ Description: AArch64 hybrid scalar/vector implementation of Keccak-f[1600] permutation for four sequential states+ Signature: void mlk_keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm(uint64_t state[100], const uint64_t rc[24])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON]+ x0:+ type: buffer+ size_bytes: 800+ permissions: read/write+ c_parameter: uint64_t state[100]+ description: Four sequential Keccak states (state0[25], state1[25], state2[25], state3[25])+ x1:+ type: buffer+ size_bytes: 192+ permissions: read-only+ c_parameter: const uint64_t rc[24]+ description: Round constants (24 x uint64_t)+ Stack:+ bytes: 224+ description: register preservation and temporary storage+*/++#include "../../../../common.h"+#if defined(MLK_FIPS202_AARCH64_NEED_X4_V8A_SCALAR_HYBRID) && \+ !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED)++/*+ * WARNING: This file is auto-derived from the mlkem-native source file+ * dev/fips202/aarch64/src/keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLK_ASM_NAMESPACE(keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm)+MLK_ASM_FN_SYMBOL(keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm)++ .cfi_startproc+ sub sp, sp, #0xe0+ .cfi_adjust_cfa_offset 0xe0+ stp x19, x20, [sp, #0x30]+ .cfi_rel_offset x19, 0x30+ .cfi_rel_offset x20, 0x38+ stp x21, x22, [sp, #0x40]+ .cfi_rel_offset x21, 0x40+ .cfi_rel_offset x22, 0x48+ stp x23, x24, [sp, #0x50]+ .cfi_rel_offset x23, 0x50+ .cfi_rel_offset x24, 0x58+ stp x25, x26, [sp, #0x60]+ .cfi_rel_offset x25, 0x60+ .cfi_rel_offset x26, 0x68+ stp x27, x28, [sp, #0x70]+ .cfi_rel_offset x27, 0x70+ .cfi_rel_offset x28, 0x78+ stp x29, x30, [sp, #0x80]+ .cfi_rel_offset x29, 0x80+ .cfi_rel_offset x30, 0x88+ stp d8, d9, [sp, #0x90]+ .cfi_rel_offset d8, 0x90+ .cfi_rel_offset d9, 0x98+ stp d10, d11, [sp, #0xa0]+ .cfi_rel_offset d10, 0xa0+ .cfi_rel_offset d11, 0xa8+ stp d12, d13, [sp, #0xb0]+ .cfi_rel_offset d12, 0xb0+ .cfi_rel_offset d13, 0xb8+ stp d14, d15, [sp, #0xc0]+ .cfi_rel_offset d14, 0xc0+ .cfi_rel_offset d15, 0xc8+ mov x29, x1+ mov x30, #0x0 // =0+ str x30, [sp, #0x20]+ str x29, [sp, #0x8]+ str x29, [sp, #0x10]+ str x0, [sp]+ add x4, x0, #0xc8+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x4], #32+ trn1 v0.2d, v25.2d, v27.2d+ trn2 v1.2d, v25.2d, v27.2d+ trn1 v2.2d, v26.2d, v28.2d+ trn2 v3.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x4], #32+ trn1 v4.2d, v25.2d, v27.2d+ trn2 v5.2d, v25.2d, v27.2d+ trn1 v6.2d, v26.2d, v28.2d+ trn2 v7.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x4], #32+ trn1 v8.2d, v25.2d, v27.2d+ trn2 v9.2d, v25.2d, v27.2d+ trn1 v10.2d, v26.2d, v28.2d+ trn2 v11.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x4], #32+ trn1 v12.2d, v25.2d, v27.2d+ trn2 v13.2d, v25.2d, v27.2d+ trn1 v14.2d, v26.2d, v28.2d+ trn2 v15.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x4], #32+ trn1 v16.2d, v25.2d, v27.2d+ trn2 v17.2d, v25.2d, v27.2d+ trn1 v18.2d, v26.2d, v28.2d+ trn2 v19.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x4], #32+ trn1 v20.2d, v25.2d, v27.2d+ trn2 v21.2d, v25.2d, v27.2d+ trn1 v22.2d, v26.2d, v28.2d+ trn2 v23.2d, v26.2d, v28.2d+ ldr d25, [x0]+ ldr d27, [x4]+ trn1 v24.2d, v25.2d, v27.2d+ sub x0, x0, #0xc0+ add x0, x0, #0x190+ ldp x1, x6, [x0]+ ldp x11, x16, [x0, #0x10]+ ldp x21, x2, [x0, #0x20]+ ldp x7, x12, [x0, #0x30]+ ldp x17, x22, [x0, #0x40]+ ldp x3, x8, [x0, #0x50]+ ldp x13, x28, [x0, #0x60]+ ldp x23, x4, [x0, #0x70]+ ldp x9, x14, [x0, #0x80]+ ldp x19, x24, [x0, #0x90]+ ldp x5, x10, [x0, #0xa0]+ ldp x15, x20, [x0, #0xb0]+ ldr x25, [x0, #0xc0]+ sub x0, x0, #0x190++Lmlk_keccak_f1600_x4_v8a_scalar_hybrid_initial:+ eor x30, x24, x25+ eor x27, x9, x10+ eor v30.16b, v0.16b, v5.16b+ eor v30.16b, v30.16b, v10.16b+ eor x0, x30, x21+ eor v30.16b, v30.16b, v15.16b+ eor x26, x27, x6+ eor x27, x26, x7+ eor v30.16b, v30.16b, v20.16b+ eor x29, x0, x22+ eor v29.16b, v1.16b, v6.16b+ eor x26, x29, x23+ eor v29.16b, v29.16b, v11.16b+ eor x29, x4, x5+ eor x30, x29, x1+ eor v29.16b, v29.16b, v16.16b+ eor x0, x27, x8+ eor v29.16b, v29.16b, v21.16b+ eor x29, x30, x2+ eor v28.16b, v2.16b, v7.16b+ eor x30, x19, x20+ eor x30, x30, x16+ eor v28.16b, v28.16b, v12.16b+ eor x27, x26, x0, ror #63+ eor v28.16b, v28.16b, v17.16b+ eor x4, x4, x27+ eor v28.16b, v28.16b, v22.16b+ eor x30, x30, x17+ eor x30, x30, x28+ eor v27.16b, v3.16b, v8.16b+ eor x29, x29, x3+ eor v27.16b, v27.16b, v13.16b+ eor x0, x0, x30, ror #63+ eor v27.16b, v27.16b, v18.16b+ eor x30, x30, x29, ror #63+ eor x22, x22, x30+ eor v27.16b, v27.16b, v23.16b+ eor x23, x23, x30+ eor v26.16b, v4.16b, v9.16b+ str x23, [sp, #0xd0]+ eor v26.16b, v26.16b, v14.16b+ eor x23, x14, x15+ eor x14, x14, x0+ eor v26.16b, v26.16b, v19.16b+ eor x23, x23, x11+ eor v26.16b, v26.16b, v24.16b+ eor x15, x15, x0+ eor x1, x1, x27+ add v31.2d, v28.2d, v28.2d+ eor x23, x23, x12+ sri v31.2d, v28.2d, #0x3f+ eor x23, x23, x13+ eor v25.16b, v31.16b, v30.16b+ eor x11, x11, x0+ eor x29, x29, x23, ror #63+ add v31.2d, v26.2d, v26.2d+ eor x23, x23, x26, ror #63+ sri v31.2d, v26.2d, #0x3f+ eor x26, x13, x0+ eor v28.16b, v31.16b, v28.16b+ eor x13, x28, x23+ eor x28, x24, x30+ add v31.2d, v29.2d, v29.2d+ eor x24, x16, x23+ sri v31.2d, v29.2d, #0x3f+ eor x16, x21, x30+ eor v26.16b, v31.16b, v26.16b+ eor x21, x25, x30+ eor x30, x19, x23+ add v31.2d, v27.2d, v27.2d+ eor x19, x20, x23+ sri v31.2d, v27.2d, #0x3f+ eor x20, x17, x23+ eor v29.16b, v31.16b, v29.16b+ eor x17, x12, x0+ eor x0, x2, x27+ add v31.2d, v30.2d, v30.2d+ eor x2, x6, x29+ sri v31.2d, v30.2d, #0x3f+ eor x6, x8, x29+ eor v27.16b, v31.16b, v27.16b+ bic x8, x28, x13, ror #47+ eor x12, x3, x27+ eor v30.16b, v0.16b, v26.16b+ bic x3, x13, x17, ror #19+ eor v31.16b, v2.16b, v29.16b+ eor x5, x5, x27+ ldr x27, [sp, #0xd0]+ shl v0.2d, v31.2d, #0x3e+ bic x25, x17, x2, ror #5+ sri v0.2d, v31.2d, #0x2+ eor x9, x9, x29+ eor v31.16b, v12.16b, v29.16b+ eor x23, x25, x5, ror #52+ eor x3, x3, x2, ror #24+ shl v2.2d, v31.2d, #0x2b+ eor x8, x8, x17, ror #2+ sri v2.2d, v31.2d, #0x15+ eor x17, x10, x29+ eor v31.16b, v13.16b, v28.16b+ bic x25, x12, x22, ror #47+ eor x29, x7, x29+ shl v12.2d, v31.2d, #0x19+ bic x10, x4, x27, ror #2+ sri v12.2d, v31.2d, #0x27+ bic x7, x5, x28, ror #10+ eor v31.16b, v19.16b, v27.16b+ eor x10, x10, x20, ror #50+ eor x13, x7, x13, ror #57+ shl v13.2d, v31.2d, #0x8+ bic x7, x2, x5, ror #47+ sri v13.2d, v31.2d, #0x38+ eor x2, x25, x24, ror #39+ eor v31.16b, v23.16b, v28.16b+ bic x25, x20, x11, ror #57+ bic x5, x17, x4, ror #25+ shl v19.2d, v31.2d, #0x38+ eor x25, x25, x17, ror #53+ sri v19.2d, v31.2d, #0x8+ bic x17, x11, x17, ror #60+ eor v31.16b, v15.16b, v26.16b+ eor x28, x7, x28, ror #57+ bic x7, x9, x12, ror #42+ shl v23.2d, v31.2d, #0x29+ eor x7, x7, x22, ror #25+ sri v23.2d, v31.2d, #0x17+ bic x22, x22, x24, ror #56+ bic x24, x24, x15, ror #31+ eor v31.16b, v1.16b, v25.16b+ eor x22, x22, x15, ror #23+ shl v15.2d, v31.2d, #0x1+ bic x20, x27, x20, ror #48+ sri v15.2d, v31.2d, #0x3f+ bic x15, x15, x9, ror #16+ eor x12, x15, x12, ror #58+ eor v31.16b, v8.16b, v28.16b+ eor x15, x5, x27, ror #27+ shl v1.2d, v31.2d, #0x37+ eor x5, x20, x11, ror #41+ sri v1.2d, v31.2d, #0x9+ ldr x11, [sp, #0x8]+ eor x20, x17, x4, ror #21+ eor v31.16b, v16.16b, v25.16b+ eor x17, x24, x9, ror #47+ shl v8.2d, v31.2d, #0x2d+ mov x24, #0x1 // =1+ sri v8.2d, v31.2d, #0x13+ bic x9, x0, x16, ror #9+ str x24, [sp, #0x18]+ eor v31.16b, v7.16b, v29.16b+ bic x24, x29, x1, ror #44+ shl v16.2d, v31.2d, #0x6+ bic x27, x1, x21, ror #50+ sri v16.2d, v31.2d, #0x3a+ bic x4, x26, x29, ror #63+ eor x1, x1, x4, ror #21+ eor v31.16b, v10.16b, v26.16b+ ldr x11, [x11]+ shl v7.2d, v31.2d, #0x3+ bic x4, x21, x30, ror #57+ sri v7.2d, v31.2d, #0x3d+ eor x21, x24, x21, ror #30+ eor x24, x9, x19, ror #44+ eor v31.16b, v3.16b, v28.16b+ bic x9, x14, x6, ror #5+ shl v10.2d, v31.2d, #0x1c+ eor x9, x9, x0, ror #43+ sri v10.2d, v31.2d, #0x24+ bic x0, x6, x0, ror #38+ eor x1, x1, x11+ eor v31.16b, v18.16b, v28.16b+ eor x11, x4, x26, ror #35+ shl v3.2d, v31.2d, #0x15+ eor x4, x0, x16, ror #47+ bic x0, x16, x19, ror #35+ sri v3.2d, v31.2d, #0x2b+ eor x16, x27, x30, ror #43+ eor v31.16b, v17.16b, v29.16b+ bic x27, x30, x26, ror #42+ shl v18.2d, v31.2d, #0xf+ bic x26, x19, x14, ror #41+ eor x19, x0, x14, ror #12+ sri v18.2d, v31.2d, #0x31+ eor x14, x26, x6, ror #46+ eor v31.16b, v11.16b, v25.16b+ eor x6, x27, x29, ror #41+ shl v17.2d, v31.2d, #0xa+ eor x0, x15, x11, ror #52+ eor x0, x0, x13, ror #48+ sri v17.2d, v31.2d, #0x36+ eor x26, x8, x9, ror #57+ eor v31.16b, v9.16b, v27.16b+ eor x27, x0, x14, ror #10+ shl v11.2d, v31.2d, #0x14+ eor x29, x16, x28, ror #63+ eor x26, x26, x6, ror #51+ sri v11.2d, v31.2d, #0x2c+ eor x30, x23, x22, ror #50+ eor v31.16b, v22.16b, v29.16b+ eor x0, x26, x10, ror #31+ shl v9.2d, v31.2d, #0x3d+ eor x29, x29, x19, ror #37+ eor x27, x27, x12, ror #5+ sri v9.2d, v31.2d, #0x3+ eor x30, x30, x24, ror #34+ eor v31.16b, v14.16b, v27.16b+ eor x0, x0, x7, ror #27+ shl v22.2d, v31.2d, #0x27+ eor x26, x30, x21, ror #26+ eor x26, x26, x25, ror #15+ sri v22.2d, v31.2d, #0x19+ ror x30, x27, #0x3e+ eor v31.16b, v20.16b, v26.16b+ eor x30, x30, x26, ror #57+ ror x26, x26, #0x3a+ shl v14.2d, v31.2d, #0x12+ eor x16, x30, x16+ sri v14.2d, v31.2d, #0x2e+ eor x28, x30, x28, ror #63+ eor v31.16b, v4.16b, v27.16b+ str x28, [sp, #0xd0]+ eor x29, x29, x17, ror #36+ shl v20.2d, v31.2d, #0x1b+ eor x28, x1, x2, ror #61+ sri v20.2d, v31.2d, #0x25+ eor x19, x30, x19, ror #37+ eor v31.16b, v24.16b, v27.16b+ eor x29, x29, x20, ror #2+ eor x28, x28, x4, ror #54+ shl v4.2d, v31.2d, #0xe+ eor x26, x26, x0, ror #55+ sri v4.2d, v31.2d, #0x32+ eor x28, x28, x3, ror #39+ eor v31.16b, v21.16b, v25.16b+ eor x28, x28, x5, ror #25+ ror x0, x0, #0x38+ shl v24.2d, v31.2d, #0x2+ eor x0, x0, x29, ror #63+ sri v24.2d, v31.2d, #0x3e+ eor x27, x28, x27, ror #61+ eor v31.16b, v5.16b, v26.16b+ eor x13, x0, x13, ror #46+ eor x28, x29, x28, ror #63+ shl v21.2d, v31.2d, #0x24+ eor x29, x30, x20, ror #2+ sri v21.2d, v31.2d, #0x1c+ eor x20, x26, x3, ror #39+ eor v31.16b, v6.16b, v25.16b+ eor x11, x0, x11, ror #50+ eor x25, x28, x25, ror #9+ shl v27.2d, v31.2d, #0x2c+ eor x3, x28, x21, ror #20+ sri v27.2d, v31.2d, #0x14+ eor x21, x26, x1+ eor x9, x27, x9, ror #49+ bic v31.16b, v7.16b, v11.16b+ eor x24, x28, x24, ror #28+ eor v5.16b, v31.16b, v10.16b+ eor x1, x30, x17, ror #36+ bic v31.16b, v8.16b, v7.16b+ eor x14, x0, x14, ror #8+ eor x22, x28, x22, ror #44+ eor v6.16b, v31.16b, v11.16b+ eor x8, x27, x8, ror #56+ bic v31.16b, v9.16b, v8.16b+ eor x17, x27, x7, ror #19+ eor v7.16b, v31.16b, v7.16b+ eor x15, x0, x15, ror #62+ bic x7, x20, x22, ror #47+ bic v31.16b, v10.16b, v9.16b+ eor x4, x26, x4, ror #54+ eor v8.16b, v31.16b, v8.16b+ eor x0, x0, x12, ror #3+ bic v31.16b, v11.16b, v10.16b+ eor x28, x28, x23, ror #58+ eor x23, x26, x2, ror #61+ eor v9.16b, v31.16b, v9.16b+ eor x26, x26, x5, ror #25+ bic v31.16b, v12.16b, v16.16b+ eor x2, x7, x16, ror #39+ eor v10.16b, v31.16b, v15.16b+ bic x7, x9, x20, ror #42+ bic x30, x15, x9, ror #16+ bic v31.16b, v13.16b, v12.16b+ eor x7, x7, x22, ror #25+ eor v11.16b, v31.16b, v16.16b+ eor x12, x30, x20, ror #58+ bic v31.16b, v14.16b, v13.16b+ bic x20, x22, x16, ror #56+ eor x30, x27, x6, ror #43+ eor v12.16b, v31.16b, v12.16b+ eor x22, x20, x15, ror #23+ bic v31.16b, v15.16b, v14.16b+ bic x6, x19, x13, ror #42+ eor v13.16b, v31.16b, v13.16b+ eor x6, x6, x17, ror #41+ bic x5, x13, x17, ror #63+ bic v31.16b, v16.16b, v15.16b+ eor x5, x21, x5, ror #21+ eor v14.16b, v31.16b, v14.16b+ bic x17, x17, x21, ror #44+ eor x27, x27, x10, ror #23+ bic v31.16b, v17.16b, v21.16b+ bic x21, x21, x25, ror #50+ eor v15.16b, v31.16b, v20.16b+ bic x20, x27, x4, ror #25+ bic v31.16b, v18.16b, v17.16b+ bic x10, x16, x15, ror #31+ eor x16, x21, x19, ror #43+ eor v16.16b, v31.16b, v21.16b+ eor x21, x17, x25, ror #30+ bic v31.16b, v19.16b, v18.16b+ bic x19, x25, x19, ror #57+ eor v17.16b, v31.16b, v17.16b+ ldr x25, [sp, #0x18]+ eor x17, x10, x9, ror #47+ bic v31.16b, v20.16b, v19.16b+ ldr x9, [sp, #0x8]+ eor v18.16b, v31.16b, v18.16b+ eor x15, x20, x28, ror #27+ bic v31.16b, v21.16b, v20.16b+ bic x20, x4, x28, ror #2+ eor x10, x20, x1, ror #50+ eor v19.16b, v31.16b, v19.16b+ bic x20, x11, x27, ror #60+ bic v31.16b, v22.16b, v1.16b+ eor x20, x20, x4, ror #21+ eor v20.16b, v31.16b, v0.16b+ bic x4, x28, x1, ror #48+ bic x1, x1, x11, ror #57+ bic v31.16b, v23.16b, v22.16b+ ldr x28, [x9, x25, lsl #3]+ eor v21.16b, v31.16b, v1.16b+ ldr x9, [sp, #0xd0]+ bic v31.16b, v24.16b, v23.16b+ add x25, x25, #0x1+ str x25, [sp, #0x18]+ eor v22.16b, v31.16b, v22.16b+ cmp x25, #0x17+ bic v31.16b, v0.16b, v24.16b+ eor x25, x1, x27, ror #53+ bic x27, x30, x26, ror #47+ eor v23.16b, v31.16b, v23.16b+ eor x1, x5, x28+ bic v31.16b, v1.16b, v0.16b+ eor x5, x4, x11, ror #41+ eor v24.16b, v31.16b, v24.16b+ eor x11, x19, x13, ror #35+ bic x13, x26, x24, ror #10+ bic v31.16b, v2.16b, v27.16b+ eor x28, x27, x24, ror #57+ eor v0.16b, v31.16b, v30.16b+ bic x27, x24, x9, ror #47+ bic v31.16b, v3.16b, v2.16b+ bic x19, x23, x3, ror #9+ bic x4, x29, x14, ror #41+ eor v1.16b, v31.16b, v27.16b+ eor x24, x19, x29, ror #44+ bic v31.16b, v4.16b, v3.16b+ bic x29, x3, x29, ror #35+ eor v2.16b, v31.16b, v2.16b+ eor x13, x13, x9, ror #57+ eor x19, x29, x14, ror #12+ bic v31.16b, v30.16b, v4.16b+ bic x29, x9, x0, ror #19+ eor v3.16b, v31.16b, v3.16b+ bic x14, x14, x8, ror #5+ bic v31.16b, v27.16b, v30.16b+ eor x9, x14, x23, ror #43+ eor x14, x4, x8, ror #46+ eor v4.16b, v31.16b, v4.16b+ bic x23, x8, x23, ror #38+ eor x8, x27, x0, ror #2+ eor x4, x23, x3, ror #47+ bic x3, x0, x30, ror #5+ eor x23, x3, x26, ror #52+ eor x3, x29, x30, ror #24+ ldr x30, [sp, #0x10]+ ld1r { v28.2d }, [x30], #8+ str x30, [sp, #0x10]+ eor v0.16b, v0.16b, v28.16b++Lmlk_keccak_f1600_x4_v8a_scalar_hybrid_loop:+ eor x0, x15, x11, ror #52+ eor x0, x0, x13, ror #48+ eor v30.16b, v0.16b, v5.16b+ eor v30.16b, v30.16b, v10.16b+ eor x26, x8, x9, ror #57+ eor v30.16b, v30.16b, v15.16b+ eor x27, x0, x14, ror #10+ eor x29, x16, x28, ror #63+ eor v30.16b, v30.16b, v20.16b+ eor x26, x26, x6, ror #51+ eor v29.16b, v1.16b, v6.16b+ eor x30, x23, x22, ror #50+ eor v29.16b, v29.16b, v11.16b+ eor x0, x26, x10, ror #31+ eor x29, x29, x19, ror #37+ eor v29.16b, v29.16b, v16.16b+ eor x27, x27, x12, ror #5+ eor v29.16b, v29.16b, v21.16b+ eor x30, x30, x24, ror #34+ eor x0, x0, x7, ror #27+ eor v28.16b, v2.16b, v7.16b+ eor x26, x30, x21, ror #26+ eor v28.16b, v28.16b, v12.16b+ eor x26, x26, x25, ror #15+ eor v28.16b, v28.16b, v17.16b+ ror x30, x27, #0x3e+ eor x30, x30, x26, ror #57+ eor v28.16b, v28.16b, v22.16b+ ror x26, x26, #0x3a+ eor v27.16b, v3.16b, v8.16b+ eor x16, x30, x16+ eor v27.16b, v27.16b, v13.16b+ eor x28, x30, x28, ror #63+ str x28, [sp, #0xd0]+ eor v27.16b, v27.16b, v18.16b+ eor x29, x29, x17, ror #36+ eor v27.16b, v27.16b, v23.16b+ eor x28, x1, x2, ror #61+ eor x19, x30, x19, ror #37+ eor v26.16b, v4.16b, v9.16b+ eor x29, x29, x20, ror #2+ eor v26.16b, v26.16b, v14.16b+ eor x28, x28, x4, ror #54+ eor v26.16b, v26.16b, v19.16b+ eor x26, x26, x0, ror #55+ eor x28, x28, x3, ror #39+ eor v26.16b, v26.16b, v24.16b+ eor x28, x28, x5, ror #25+ add v31.2d, v28.2d, v28.2d+ ror x0, x0, #0x38+ eor x0, x0, x29, ror #63+ sri v31.2d, v28.2d, #0x3f+ eor x27, x28, x27, ror #61+ eor v25.16b, v31.16b, v30.16b+ eor x13, x0, x13, ror #46+ add v31.2d, v26.2d, v26.2d+ eor x28, x29, x28, ror #63+ eor x29, x30, x20, ror #2+ sri v31.2d, v26.2d, #0x3f+ eor x20, x26, x3, ror #39+ eor v28.16b, v31.16b, v28.16b+ eor x11, x0, x11, ror #50+ add v31.2d, v29.2d, v29.2d+ eor x25, x28, x25, ror #9+ eor x3, x28, x21, ror #20+ sri v31.2d, v29.2d, #0x3f+ eor x21, x26, x1+ eor v26.16b, v31.16b, v26.16b+ eor x9, x27, x9, ror #49+ eor x24, x28, x24, ror #28+ add v31.2d, v27.2d, v27.2d+ eor x1, x30, x17, ror #36+ sri v31.2d, v27.2d, #0x3f+ eor x14, x0, x14, ror #8+ eor v29.16b, v31.16b, v29.16b+ eor x22, x28, x22, ror #44+ eor x8, x27, x8, ror #56+ add v31.2d, v30.2d, v30.2d+ eor x17, x27, x7, ror #19+ sri v31.2d, v30.2d, #0x3f+ eor x15, x0, x15, ror #62+ bic x7, x20, x22, ror #47+ eor v27.16b, v31.16b, v27.16b+ eor x4, x26, x4, ror #54+ eor v30.16b, v0.16b, v26.16b+ eor x0, x0, x12, ror #3+ eor v31.16b, v2.16b, v29.16b+ eor x28, x28, x23, ror #58+ eor x23, x26, x2, ror #61+ shl v0.2d, v31.2d, #0x3e+ eor x26, x26, x5, ror #25+ sri v0.2d, v31.2d, #0x2+ eor x2, x7, x16, ror #39+ eor v31.16b, v12.16b, v29.16b+ bic x7, x9, x20, ror #42+ bic x30, x15, x9, ror #16+ shl v2.2d, v31.2d, #0x2b+ eor x7, x7, x22, ror #25+ sri v2.2d, v31.2d, #0x15+ eor x12, x30, x20, ror #58+ bic x20, x22, x16, ror #56+ eor v31.16b, v13.16b, v28.16b+ eor x30, x27, x6, ror #43+ shl v12.2d, v31.2d, #0x19+ eor x22, x20, x15, ror #23+ sri v12.2d, v31.2d, #0x27+ bic x6, x19, x13, ror #42+ eor x6, x6, x17, ror #41+ eor v31.16b, v19.16b, v27.16b+ bic x5, x13, x17, ror #63+ shl v13.2d, v31.2d, #0x8+ eor x5, x21, x5, ror #21+ sri v13.2d, v31.2d, #0x38+ bic x17, x17, x21, ror #44+ eor x27, x27, x10, ror #23+ eor v31.16b, v23.16b, v28.16b+ bic x21, x21, x25, ror #50+ shl v19.2d, v31.2d, #0x38+ bic x20, x27, x4, ror #25+ bic x10, x16, x15, ror #31+ sri v19.2d, v31.2d, #0x8+ eor x16, x21, x19, ror #43+ eor v31.16b, v15.16b, v26.16b+ eor x21, x17, x25, ror #30+ shl v23.2d, v31.2d, #0x29+ bic x19, x25, x19, ror #57+ ldr x25, [sp, #0x18]+ sri v23.2d, v31.2d, #0x17+ eor x17, x10, x9, ror #47+ eor v31.16b, v1.16b, v25.16b+ ldr x9, [sp, #0x8]+ eor x15, x20, x28, ror #27+ shl v15.2d, v31.2d, #0x1+ bic x20, x4, x28, ror #2+ sri v15.2d, v31.2d, #0x3f+ eor x10, x20, x1, ror #50+ eor v31.16b, v8.16b, v28.16b+ bic x20, x11, x27, ror #60+ eor x20, x20, x4, ror #21+ shl v1.2d, v31.2d, #0x37+ bic x4, x28, x1, ror #48+ sri v1.2d, v31.2d, #0x9+ bic x1, x1, x11, ror #57+ eor v31.16b, v16.16b, v25.16b+ ldr x28, [x9, x25, lsl #3]+ ldr x9, [sp, #0xd0]+ shl v8.2d, v31.2d, #0x2d+ add x25, x25, #0x1+ sri v8.2d, v31.2d, #0x13+ str x25, [sp, #0x18]+ cmp x25, #0x17+ eor v31.16b, v7.16b, v29.16b+ eor x25, x1, x27, ror #53+ shl v16.2d, v31.2d, #0x6+ bic x27, x30, x26, ror #47+ sri v16.2d, v31.2d, #0x3a+ eor x1, x5, x28+ eor x5, x4, x11, ror #41+ eor v31.16b, v10.16b, v26.16b+ eor x11, x19, x13, ror #35+ shl v7.2d, v31.2d, #0x3+ bic x13, x26, x24, ror #10+ eor x28, x27, x24, ror #57+ sri v7.2d, v31.2d, #0x3d+ bic x27, x24, x9, ror #47+ eor v31.16b, v3.16b, v28.16b+ bic x19, x23, x3, ror #9+ shl v10.2d, v31.2d, #0x1c+ bic x4, x29, x14, ror #41+ eor x24, x19, x29, ror #44+ sri v10.2d, v31.2d, #0x24+ bic x29, x3, x29, ror #35+ eor v31.16b, v18.16b, v28.16b+ eor x13, x13, x9, ror #57+ shl v3.2d, v31.2d, #0x15+ eor x19, x29, x14, ror #12+ bic x29, x9, x0, ror #19+ sri v3.2d, v31.2d, #0x2b+ bic x14, x14, x8, ror #5+ eor v31.16b, v17.16b, v29.16b+ eor x9, x14, x23, ror #43+ eor x14, x4, x8, ror #46+ shl v18.2d, v31.2d, #0xf+ bic x23, x8, x23, ror #38+ sri v18.2d, v31.2d, #0x31+ eor x8, x27, x0, ror #2+ eor v31.16b, v11.16b, v25.16b+ eor x4, x23, x3, ror #47+ bic x3, x0, x30, ror #5+ shl v17.2d, v31.2d, #0xa+ eor x23, x3, x26, ror #52+ sri v17.2d, v31.2d, #0x36+ eor x3, x29, x30, ror #24+ eor x0, x15, x11, ror #52+ eor v31.16b, v9.16b, v27.16b+ eor x0, x0, x13, ror #48+ shl v11.2d, v31.2d, #0x14+ eor x26, x8, x9, ror #57+ sri v11.2d, v31.2d, #0x2c+ eor x27, x0, x14, ror #10+ eor x29, x16, x28, ror #63+ eor v31.16b, v22.16b, v29.16b+ eor x26, x26, x6, ror #51+ shl v9.2d, v31.2d, #0x3d+ eor x30, x23, x22, ror #50+ sri v9.2d, v31.2d, #0x3+ eor x0, x26, x10, ror #31+ eor x29, x29, x19, ror #37+ eor v31.16b, v14.16b, v27.16b+ eor x27, x27, x12, ror #5+ shl v22.2d, v31.2d, #0x27+ eor x30, x30, x24, ror #34+ eor x0, x0, x7, ror #27+ sri v22.2d, v31.2d, #0x19+ eor x26, x30, x21, ror #26+ eor v31.16b, v20.16b, v26.16b+ eor x26, x26, x25, ror #15+ shl v14.2d, v31.2d, #0x12+ ror x30, x27, #0x3e+ eor x30, x30, x26, ror #57+ sri v14.2d, v31.2d, #0x2e+ ror x26, x26, #0x3a+ eor v31.16b, v4.16b, v27.16b+ eor x16, x30, x16+ shl v20.2d, v31.2d, #0x1b+ eor x28, x30, x28, ror #63+ str x28, [sp, #0xd0]+ sri v20.2d, v31.2d, #0x25+ eor x29, x29, x17, ror #36+ eor v31.16b, v24.16b, v27.16b+ eor x28, x1, x2, ror #61+ eor x19, x30, x19, ror #37+ shl v4.2d, v31.2d, #0xe+ eor x29, x29, x20, ror #2+ sri v4.2d, v31.2d, #0x32+ eor x28, x28, x4, ror #54+ eor v31.16b, v21.16b, v25.16b+ eor x26, x26, x0, ror #55+ eor x28, x28, x3, ror #39+ shl v24.2d, v31.2d, #0x2+ eor x28, x28, x5, ror #25+ sri v24.2d, v31.2d, #0x3e+ ror x0, x0, #0x38+ eor x0, x0, x29, ror #63+ eor v31.16b, v5.16b, v26.16b+ eor x27, x28, x27, ror #61+ shl v21.2d, v31.2d, #0x24+ eor x13, x0, x13, ror #46+ sri v21.2d, v31.2d, #0x1c+ eor x28, x29, x28, ror #63+ eor x29, x30, x20, ror #2+ eor v31.16b, v6.16b, v25.16b+ eor x20, x26, x3, ror #39+ shl v27.2d, v31.2d, #0x2c+ eor x11, x0, x11, ror #50+ sri v27.2d, v31.2d, #0x14+ eor x25, x28, x25, ror #9+ eor x3, x28, x21, ror #20+ bic v31.16b, v7.16b, v11.16b+ eor x21, x26, x1+ eor v5.16b, v31.16b, v10.16b+ eor x9, x27, x9, ror #49+ eor x24, x28, x24, ror #28+ bic v31.16b, v8.16b, v7.16b+ eor x1, x30, x17, ror #36+ eor v6.16b, v31.16b, v11.16b+ eor x14, x0, x14, ror #8+ bic v31.16b, v9.16b, v8.16b+ eor x22, x28, x22, ror #44+ eor x8, x27, x8, ror #56+ eor v7.16b, v31.16b, v7.16b+ eor x17, x27, x7, ror #19+ bic v31.16b, v10.16b, v9.16b+ eor x15, x0, x15, ror #62+ bic x7, x20, x22, ror #47+ eor v8.16b, v31.16b, v8.16b+ eor x4, x26, x4, ror #54+ bic v31.16b, v11.16b, v10.16b+ eor x0, x0, x12, ror #3+ eor v9.16b, v31.16b, v9.16b+ eor x28, x28, x23, ror #58+ eor x23, x26, x2, ror #61+ bic v31.16b, v12.16b, v16.16b+ eor x26, x26, x5, ror #25+ eor v10.16b, v31.16b, v15.16b+ eor x2, x7, x16, ror #39+ bic v31.16b, v13.16b, v12.16b+ bic x7, x9, x20, ror #42+ bic x30, x15, x9, ror #16+ eor v11.16b, v31.16b, v16.16b+ eor x7, x7, x22, ror #25+ bic v31.16b, v14.16b, v13.16b+ eor x12, x30, x20, ror #58+ bic x20, x22, x16, ror #56+ eor v12.16b, v31.16b, v12.16b+ eor x30, x27, x6, ror #43+ bic v31.16b, v15.16b, v14.16b+ eor x22, x20, x15, ror #23+ eor v13.16b, v31.16b, v13.16b+ bic x6, x19, x13, ror #42+ eor x6, x6, x17, ror #41+ bic v31.16b, v16.16b, v15.16b+ bic x5, x13, x17, ror #63+ eor v14.16b, v31.16b, v14.16b+ eor x5, x21, x5, ror #21+ bic v31.16b, v17.16b, v21.16b+ bic x17, x17, x21, ror #44+ eor x27, x27, x10, ror #23+ eor v15.16b, v31.16b, v20.16b+ bic x21, x21, x25, ror #50+ bic v31.16b, v18.16b, v17.16b+ bic x20, x27, x4, ror #25+ bic x10, x16, x15, ror #31+ eor v16.16b, v31.16b, v21.16b+ eor x16, x21, x19, ror #43+ bic v31.16b, v19.16b, v18.16b+ eor x21, x17, x25, ror #30+ eor v17.16b, v31.16b, v17.16b+ bic x19, x25, x19, ror #57+ ldr x25, [sp, #0x18]+ bic v31.16b, v20.16b, v19.16b+ eor x17, x10, x9, ror #47+ eor v18.16b, v31.16b, v18.16b+ ldr x9, [sp, #0x8]+ eor x15, x20, x28, ror #27+ bic v31.16b, v21.16b, v20.16b+ bic x20, x4, x28, ror #2+ eor v19.16b, v31.16b, v19.16b+ eor x10, x20, x1, ror #50+ bic v31.16b, v22.16b, v1.16b+ bic x20, x11, x27, ror #60+ eor x20, x20, x4, ror #21+ eor v20.16b, v31.16b, v0.16b+ bic x4, x28, x1, ror #48+ bic v31.16b, v23.16b, v22.16b+ bic x1, x1, x11, ror #57+ eor v21.16b, v31.16b, v1.16b+ ldr x28, [x9, x25, lsl #3]+ ldr x9, [sp, #0xd0]+ bic v31.16b, v24.16b, v23.16b+ add x25, x25, #0x1+ eor v22.16b, v31.16b, v22.16b+ str x25, [sp, #0x18]+ cmp x25, #0x17+ bic v31.16b, v0.16b, v24.16b+ eor x25, x1, x27, ror #53+ eor v23.16b, v31.16b, v23.16b+ bic x27, x30, x26, ror #47+ bic v31.16b, v1.16b, v0.16b+ eor x1, x5, x28+ eor x5, x4, x11, ror #41+ eor v24.16b, v31.16b, v24.16b+ eor x11, x19, x13, ror #35+ bic v31.16b, v2.16b, v27.16b+ bic x13, x26, x24, ror #10+ eor x28, x27, x24, ror #57+ eor v0.16b, v31.16b, v30.16b+ bic x27, x24, x9, ror #47+ bic v31.16b, v3.16b, v2.16b+ bic x19, x23, x3, ror #9+ eor v1.16b, v31.16b, v27.16b+ bic x4, x29, x14, ror #41+ eor x24, x19, x29, ror #44+ bic v31.16b, v4.16b, v3.16b+ bic x29, x3, x29, ror #35+ eor v2.16b, v31.16b, v2.16b+ eor x13, x13, x9, ror #57+ bic v31.16b, v30.16b, v4.16b+ eor x19, x29, x14, ror #12+ bic x29, x9, x0, ror #19+ eor v3.16b, v31.16b, v3.16b+ bic x14, x14, x8, ror #5+ bic v31.16b, v27.16b, v30.16b+ eor x9, x14, x23, ror #43+ eor x14, x4, x8, ror #46+ eor v4.16b, v31.16b, v4.16b+ bic x23, x8, x23, ror #38+ eor x8, x27, x0, ror #2+ eor x4, x23, x3, ror #47+ bic x3, x0, x30, ror #5+ eor x23, x3, x26, ror #52+ eor x3, x29, x30, ror #24+ ldr x30, [sp, #0x10]+ ld1r { v28.2d }, [x30], #8+ str x30, [sp, #0x10]+ eor v0.16b, v0.16b, v28.16b++Lmlk_keccak_f1600_x4_v8a_scalar_hybrid_loop_end:+ b.le Lmlk_keccak_f1600_x4_v8a_scalar_hybrid_loop+ ror x2, x2, #0x3d+ ror x3, x3, #0x27+ ror x4, x4, #0x36+ ror x5, x5, #0x19+ ror x6, x6, #0x2b+ ror x7, x7, #0x13+ ror x8, x8, #0x38+ ror x9, x9, #0x31+ ror x10, x10, #0x17+ ror x11, x11, #0x32+ ror x12, x12, #0x3+ ror x13, x13, #0x2e+ ror x14, x14, #0x8+ ror x15, x15, #0x3e+ ror x17, x17, #0x24+ ror x28, x28, #0x3f+ ror x19, x19, #0x25+ ror x20, x20, #0x2+ ror x21, x21, #0x14+ ror x22, x22, #0x2c+ ror x23, x23, #0x3a+ ror x24, x24, #0x1c+ ror x25, x25, #0x9+ ldr x30, [sp, #0x20]+ cmp x30, #0x1+ b.eq Lmlk_keccak_f1600_x4_v8a_scalar_hybrid_done+ mov x30, #0x1 // =1+ str x30, [sp, #0x20]+ ldr x0, [sp]+ add x0, x0, #0x190+ stp x1, x6, [x0]+ stp x11, x16, [x0, #0x10]+ stp x21, x2, [x0, #0x20]+ stp x7, x12, [x0, #0x30]+ stp x17, x22, [x0, #0x40]+ stp x3, x8, [x0, #0x50]+ stp x13, x28, [x0, #0x60]+ stp x23, x4, [x0, #0x70]+ stp x9, x14, [x0, #0x80]+ stp x19, x24, [x0, #0x90]+ stp x5, x10, [x0, #0xa0]+ stp x15, x20, [x0, #0xb0]+ str x25, [x0, #0xc0]+ sub x0, x0, #0x190+ add x0, x0, #0x258+ ldp x1, x6, [x0]+ ldp x11, x16, [x0, #0x10]+ ldp x21, x2, [x0, #0x20]+ ldp x7, x12, [x0, #0x30]+ ldp x17, x22, [x0, #0x40]+ ldp x3, x8, [x0, #0x50]+ ldp x13, x28, [x0, #0x60]+ ldp x23, x4, [x0, #0x70]+ ldp x9, x14, [x0, #0x80]+ ldp x19, x24, [x0, #0x90]+ ldp x5, x10, [x0, #0xa0]+ ldp x15, x20, [x0, #0xb0]+ ldr x25, [x0, #0xc0]+ sub x0, x0, #0x258+ b Lmlk_keccak_f1600_x4_v8a_scalar_hybrid_initial++Lmlk_keccak_f1600_x4_v8a_scalar_hybrid_done:+ ldr x0, [sp]+ add x0, x0, #0x258+ stp x1, x6, [x0]+ stp x11, x16, [x0, #0x10]+ stp x21, x2, [x0, #0x20]+ stp x7, x12, [x0, #0x30]+ stp x17, x22, [x0, #0x40]+ stp x3, x8, [x0, #0x50]+ stp x13, x28, [x0, #0x60]+ stp x23, x4, [x0, #0x70]+ stp x9, x14, [x0, #0x80]+ stp x19, x24, [x0, #0x90]+ stp x5, x10, [x0, #0xa0]+ stp x15, x20, [x0, #0xb0]+ str x25, [x0, #0xc0]+ sub x0, x0, #0x258+ add x4, x0, #0xc8+ trn1 v25.2d, v0.2d, v1.2d+ trn1 v26.2d, v2.2d, v3.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v0.2d, v1.2d+ trn2 v28.2d, v2.2d, v3.2d+ st1 { v27.2d, v28.2d }, [x4], #32+ trn1 v25.2d, v4.2d, v5.2d+ trn1 v26.2d, v6.2d, v7.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v4.2d, v5.2d+ trn2 v28.2d, v6.2d, v7.2d+ st1 { v27.2d, v28.2d }, [x4], #32+ trn1 v25.2d, v8.2d, v9.2d+ trn1 v26.2d, v10.2d, v11.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v8.2d, v9.2d+ trn2 v28.2d, v10.2d, v11.2d+ st1 { v27.2d, v28.2d }, [x4], #32+ trn1 v25.2d, v12.2d, v13.2d+ trn1 v26.2d, v14.2d, v15.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v12.2d, v13.2d+ trn2 v28.2d, v14.2d, v15.2d+ st1 { v27.2d, v28.2d }, [x4], #32+ trn1 v25.2d, v16.2d, v17.2d+ trn1 v26.2d, v18.2d, v19.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v16.2d, v17.2d+ trn2 v28.2d, v18.2d, v19.2d+ st1 { v27.2d, v28.2d }, [x4], #32+ trn1 v25.2d, v20.2d, v21.2d+ trn1 v26.2d, v22.2d, v23.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v20.2d, v21.2d+ trn2 v28.2d, v22.2d, v23.2d+ st1 { v27.2d, v28.2d }, [x4], #32+ str d24, [x0]+ trn2 v25.2d, v24.2d, v24.2d+ str d25, [x4]+ ldp d8, d9, [sp, #0x90]+ .cfi_restore d8+ .cfi_restore d9+ ldp d10, d11, [sp, #0xa0]+ .cfi_restore d10+ .cfi_restore d11+ ldp d12, d13, [sp, #0xb0]+ .cfi_restore d12+ .cfi_restore d13+ ldp d14, d15, [sp, #0xc0]+ .cfi_restore d14+ .cfi_restore d15+ ldp x19, x20, [sp, #0x30]+ .cfi_restore x19+ .cfi_restore x20+ ldp x21, x22, [sp, #0x40]+ .cfi_restore x21+ .cfi_restore x22+ ldp x23, x24, [sp, #0x50]+ .cfi_restore x23+ .cfi_restore x24+ ldp x25, x26, [sp, #0x60]+ .cfi_restore x25+ .cfi_restore x26+ ldp x27, x28, [sp, #0x70]+ .cfi_restore x27+ .cfi_restore x28+ ldp x29, x30, [sp, #0x80]+ .cfi_restore x29+ .cfi_restore x30+ add sp, sp, #0xe0+ .cfi_adjust_cfa_offset -0xe0+ ret+ .cfi_endproc++MLK_ASM_FN_SIZE(keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm)++#endif /* MLK_FIPS202_AARCH64_NEED_X4_V8A_SCALAR_HYBRID && \+ !MLK_CONFIG_MULTILEVEL_NO_SHARED */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
@@ -0,0 +1,989 @@+/*+ * Copyright (c) The mlkem-native project authors+ * Copyright (c) 2021-2022 Arm Limited+ * Copyright (c) 2022 Matthias Kannwischer+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++// Author: Hanno Becker <hannobecker@posteo.de>+// Author: Matthias Kannwischer <matthias@kannwischer.eu>++/*yaml+ Name: keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm+ Description: AArch64 hybrid scalar/vector implementation of Keccak-f[1600] permutation for four sequential states with ARMv8.4-A optimizations+ Signature: void mlk_keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm(uint64_t state[100], const uint64_t rc[24])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON, SHA3]+ x0:+ type: buffer+ size_bytes: 800+ permissions: read/write+ c_parameter: uint64_t state[100]+ description: Four sequential Keccak states (state0[25], state1[25], state2[25], state3[25])+ x1:+ type: buffer+ size_bytes: 192+ permissions: read-only+ c_parameter: const uint64_t rc[24]+ description: Round constants (24 x uint64_t)+ Stack:+ bytes: 224+ description: register preservation and temporary storage+*/++#include "../../../../common.h"+#if defined(MLK_FIPS202_AARCH64_NEED_X4_V8A_V84A_SCALAR_HYBRID) && \+ !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED)++#if defined(__ARM_FEATURE_SHA3)++/*+ * WARNING: This file is auto-derived from the mlkem-native source file+ * dev/fips202/aarch64/src/keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLK_ASM_NAMESPACE(keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm)+MLK_ASM_FN_SYMBOL(keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm)++ .cfi_startproc+ sub sp, sp, #0xe0+ .cfi_adjust_cfa_offset 0xe0+ stp x19, x20, [sp, #0x30]+ .cfi_rel_offset x19, 0x30+ .cfi_rel_offset x20, 0x38+ stp x21, x22, [sp, #0x40]+ .cfi_rel_offset x21, 0x40+ .cfi_rel_offset x22, 0x48+ stp x23, x24, [sp, #0x50]+ .cfi_rel_offset x23, 0x50+ .cfi_rel_offset x24, 0x58+ stp x25, x26, [sp, #0x60]+ .cfi_rel_offset x25, 0x60+ .cfi_rel_offset x26, 0x68+ stp x27, x28, [sp, #0x70]+ .cfi_rel_offset x27, 0x70+ .cfi_rel_offset x28, 0x78+ stp x29, x30, [sp, #0x80]+ .cfi_rel_offset x29, 0x80+ .cfi_rel_offset x30, 0x88+ stp d8, d9, [sp, #0x90]+ .cfi_rel_offset d8, 0x90+ .cfi_rel_offset d9, 0x98+ stp d10, d11, [sp, #0xa0]+ .cfi_rel_offset d10, 0xa0+ .cfi_rel_offset d11, 0xa8+ stp d12, d13, [sp, #0xb0]+ .cfi_rel_offset d12, 0xb0+ .cfi_rel_offset d13, 0xb8+ stp d14, d15, [sp, #0xc0]+ .cfi_rel_offset d14, 0xc0+ .cfi_rel_offset d15, 0xc8+ mov x29, x1+ mov x30, #0x0 // =0+ str x30, [sp, #0x20]+ str x29, [sp, #0x8]+ str x29, [sp, #0x10]+ str x0, [sp]+ add x4, x0, #0xc8+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x4], #32+ trn1 v0.2d, v25.2d, v27.2d+ trn2 v1.2d, v25.2d, v27.2d+ trn1 v2.2d, v26.2d, v28.2d+ trn2 v3.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x4], #32+ trn1 v4.2d, v25.2d, v27.2d+ trn2 v5.2d, v25.2d, v27.2d+ trn1 v6.2d, v26.2d, v28.2d+ trn2 v7.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x4], #32+ trn1 v8.2d, v25.2d, v27.2d+ trn2 v9.2d, v25.2d, v27.2d+ trn1 v10.2d, v26.2d, v28.2d+ trn2 v11.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x4], #32+ trn1 v12.2d, v25.2d, v27.2d+ trn2 v13.2d, v25.2d, v27.2d+ trn1 v14.2d, v26.2d, v28.2d+ trn2 v15.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x4], #32+ trn1 v16.2d, v25.2d, v27.2d+ trn2 v17.2d, v25.2d, v27.2d+ trn1 v18.2d, v26.2d, v28.2d+ trn2 v19.2d, v26.2d, v28.2d+ ldp q25, q26, [x0], #0x20+ ld1 { v27.2d, v28.2d }, [x4], #32+ trn1 v20.2d, v25.2d, v27.2d+ trn2 v21.2d, v25.2d, v27.2d+ trn1 v22.2d, v26.2d, v28.2d+ trn2 v23.2d, v26.2d, v28.2d+ ldr d25, [x0]+ ldr d27, [x4]+ trn1 v24.2d, v25.2d, v27.2d+ sub x0, x0, #0xc0+ add x0, x0, #0x190+ ldp x1, x6, [x0]+ ldp x11, x16, [x0, #0x10]+ ldp x21, x2, [x0, #0x20]+ ldp x7, x12, [x0, #0x30]+ ldp x17, x22, [x0, #0x40]+ ldp x3, x8, [x0, #0x50]+ ldp x13, x28, [x0, #0x60]+ ldp x23, x4, [x0, #0x70]+ ldp x9, x14, [x0, #0x80]+ ldp x19, x24, [x0, #0x90]+ ldp x5, x10, [x0, #0xa0]+ ldp x15, x20, [x0, #0xb0]+ ldr x25, [x0, #0xc0]+ sub x0, x0, #0x190++Lmlk_keccak_f1600_x4_v8a_v84a_scalar_hybrid_initial:+ eor x30, x24, x25+ eor x27, x9, x10+ eor3 v30.16b, v0.16b, v5.16b, v10.16b+ eor v30.16b, v30.16b, v15.16b+ eor x0, x30, x21+ eor x26, x27, x6+ eor v30.16b, v30.16b, v20.16b+ eor x27, x26, x7+ eor x29, x0, x22+ eor3 v29.16b, v1.16b, v6.16b, v11.16b+ eor x26, x29, x23+ eor x29, x4, x5+ eor v29.16b, v29.16b, v16.16b+ eor x30, x29, x1+ eor x0, x27, x8+ eor v29.16b, v29.16b, v21.16b+ eor x29, x30, x2+ eor x30, x19, x20+ eor3 v28.16b, v2.16b, v7.16b, v12.16b+ eor x30, x30, x16+ eor x27, x26, x0, ror #63+ eor v28.16b, v28.16b, v17.16b+ eor x4, x4, x27+ eor x30, x30, x17+ eor v28.16b, v28.16b, v22.16b+ eor x30, x30, x28+ eor x29, x29, x3+ eor3 v27.16b, v3.16b, v8.16b, v13.16b+ eor x0, x0, x30, ror #63+ eor x30, x30, x29, ror #63+ eor v27.16b, v27.16b, v18.16b+ eor x22, x22, x30+ eor v27.16b, v27.16b, v23.16b+ eor x23, x23, x30+ str x23, [sp, #0xd0]+ eor3 v26.16b, v4.16b, v9.16b, v14.16b+ eor x23, x14, x15+ eor x14, x14, x0+ eor v26.16b, v26.16b, v19.16b+ eor x23, x23, x11+ eor x15, x15, x0+ eor v26.16b, v26.16b, v24.16b+ eor x1, x1, x27+ eor x23, x23, x12+ rax1 v25.2d, v30.2d, v28.2d+ eor x23, x23, x13+ eor x11, x11, x0+ add v31.2d, v26.2d, v26.2d+ eor x29, x29, x23, ror #63+ eor x23, x23, x26, ror #63+ sri v31.2d, v26.2d, #0x3f+ eor x26, x13, x0+ eor x13, x28, x23+ eor v28.16b, v31.16b, v28.16b+ eor x28, x24, x30+ eor x24, x16, x23+ rax1 v26.2d, v26.2d, v29.2d+ eor x16, x21, x30+ eor x21, x25, x30+ add v31.2d, v27.2d, v27.2d+ eor x30, x19, x23+ sri v31.2d, v27.2d, #0x3f+ eor x19, x20, x23+ eor x20, x17, x23+ eor v29.16b, v31.16b, v29.16b+ eor x17, x12, x0+ eor x0, x2, x27+ rax1 v27.2d, v27.2d, v30.2d+ eor x2, x6, x29+ eor x6, x8, x29+ eor v30.16b, v0.16b, v26.16b+ bic x8, x28, x13, ror #47+ eor x12, x3, x27+ eor v31.16b, v2.16b, v29.16b+ bic x3, x13, x17, ror #19+ eor x5, x5, x27+ shl v0.2d, v31.2d, #0x3e+ ldr x27, [sp, #0xd0]+ bic x25, x17, x2, ror #5+ sri v0.2d, v31.2d, #0x2+ eor x9, x9, x29+ eor x23, x25, x5, ror #52+ xar v2.2d, v12.2d, v29.2d, #0x15+ eor x3, x3, x2, ror #24+ eor x8, x8, x17, ror #2+ eor v31.16b, v13.16b, v28.16b+ eor x17, x10, x29+ bic x25, x12, x22, ror #47+ shl v12.2d, v31.2d, #0x19+ eor x29, x7, x29+ bic x10, x4, x27, ror #2+ sri v12.2d, v31.2d, #0x27+ bic x7, x5, x28, ror #10+ xar v13.2d, v19.2d, v27.2d, #0x38+ eor x10, x10, x20, ror #50+ eor x13, x7, x13, ror #57+ eor v31.16b, v23.16b, v28.16b+ bic x7, x2, x5, ror #47+ eor x2, x25, x24, ror #39+ shl v19.2d, v31.2d, #0x38+ bic x25, x20, x11, ror #57+ bic x5, x17, x4, ror #25+ sri v19.2d, v31.2d, #0x8+ eor x25, x25, x17, ror #53+ bic x17, x11, x17, ror #60+ xar v23.2d, v15.2d, v26.2d, #0x17+ eor x28, x7, x28, ror #57+ bic x7, x9, x12, ror #42+ eor v31.16b, v1.16b, v25.16b+ eor x7, x7, x22, ror #25+ bic x22, x22, x24, ror #56+ shl v15.2d, v31.2d, #0x1+ bic x24, x24, x15, ror #31+ eor x22, x22, x15, ror #23+ sri v15.2d, v31.2d, #0x3f+ bic x20, x27, x20, ror #48+ bic x15, x15, x9, ror #16+ xar v1.2d, v8.2d, v28.2d, #0x9+ eor x12, x15, x12, ror #58+ eor x15, x5, x27, ror #27+ eor v31.16b, v16.16b, v25.16b+ eor x5, x20, x11, ror #41+ shl v8.2d, v31.2d, #0x2d+ ldr x11, [sp, #0x8]+ eor x20, x17, x4, ror #21+ sri v8.2d, v31.2d, #0x13+ eor x17, x24, x9, ror #47+ mov x24, #0x1 // =1+ xar v16.2d, v7.2d, v29.2d, #0x3a+ bic x9, x0, x16, ror #9+ str x24, [sp, #0x18]+ eor v31.16b, v10.16b, v26.16b+ bic x24, x29, x1, ror #44+ bic x27, x1, x21, ror #50+ shl v7.2d, v31.2d, #0x3+ bic x4, x26, x29, ror #63+ eor x1, x1, x4, ror #21+ sri v7.2d, v31.2d, #0x3d+ ldr x11, [x11]+ bic x4, x21, x30, ror #57+ xar v10.2d, v3.2d, v28.2d, #0x24+ eor x21, x24, x21, ror #30+ eor x24, x9, x19, ror #44+ eor v31.16b, v18.16b, v28.16b+ bic x9, x14, x6, ror #5+ eor x9, x9, x0, ror #43+ shl v3.2d, v31.2d, #0x15+ bic x0, x6, x0, ror #38+ eor x1, x1, x11+ sri v3.2d, v31.2d, #0x2b+ eor x11, x4, x26, ror #35+ eor x4, x0, x16, ror #47+ xar v18.2d, v17.2d, v29.2d, #0x31+ bic x0, x16, x19, ror #35+ eor v31.16b, v11.16b, v25.16b+ eor x16, x27, x30, ror #43+ bic x27, x30, x26, ror #42+ shl v17.2d, v31.2d, #0xa+ bic x26, x19, x14, ror #41+ eor x19, x0, x14, ror #12+ sri v17.2d, v31.2d, #0x36+ eor x14, x26, x6, ror #46+ eor x6, x27, x29, ror #41+ xar v11.2d, v9.2d, v27.2d, #0x2c+ eor x0, x15, x11, ror #52+ eor x0, x0, x13, ror #48+ eor v31.16b, v22.16b, v29.16b+ eor x26, x8, x9, ror #57+ eor x27, x0, x14, ror #10+ shl v9.2d, v31.2d, #0x3d+ eor x29, x16, x28, ror #63+ eor x26, x26, x6, ror #51+ sri v9.2d, v31.2d, #0x3+ eor x30, x23, x22, ror #50+ eor x0, x26, x10, ror #31+ xar v22.2d, v14.2d, v27.2d, #0x19+ eor x29, x29, x19, ror #37+ eor x27, x27, x12, ror #5+ eor v31.16b, v20.16b, v26.16b+ eor x30, x30, x24, ror #34+ eor x0, x0, x7, ror #27+ shl v14.2d, v31.2d, #0x12+ eor x26, x30, x21, ror #26+ sri v14.2d, v31.2d, #0x2e+ eor x26, x26, x25, ror #15+ ror x30, x27, #0x3e+ xar v20.2d, v4.2d, v27.2d, #0x25+ eor x30, x30, x26, ror #57+ ror x26, x26, #0x3a+ eor v31.16b, v24.16b, v27.16b+ eor x16, x30, x16+ eor x28, x30, x28, ror #63+ shl v4.2d, v31.2d, #0xe+ str x28, [sp, #0xd0]+ eor x29, x29, x17, ror #36+ sri v4.2d, v31.2d, #0x32+ eor x28, x1, x2, ror #61+ eor x19, x30, x19, ror #37+ xar v24.2d, v21.2d, v25.2d, #0x3e+ eor x29, x29, x20, ror #2+ eor x28, x28, x4, ror #54+ eor v31.16b, v5.16b, v26.16b+ eor x26, x26, x0, ror #55+ eor x28, x28, x3, ror #39+ shl v21.2d, v31.2d, #0x24+ eor x28, x28, x5, ror #25+ ror x0, x0, #0x38+ sri v21.2d, v31.2d, #0x1c+ eor x0, x0, x29, ror #63+ eor x27, x28, x27, ror #61+ xar v27.2d, v6.2d, v25.2d, #0x14+ eor x13, x0, x13, ror #46+ eor x28, x29, x28, ror #63+ bic v31.16b, v7.16b, v11.16b+ eor x29, x30, x20, ror #2+ eor v5.16b, v31.16b, v10.16b+ eor x20, x26, x3, ror #39+ eor x11, x0, x11, ror #50+ bcax v6.16b, v11.16b, v8.16b, v7.16b+ eor x25, x28, x25, ror #9+ eor x3, x28, x21, ror #20+ bic v31.16b, v9.16b, v8.16b+ eor x21, x26, x1+ eor x9, x27, x9, ror #49+ eor v7.16b, v31.16b, v7.16b+ eor x24, x28, x24, ror #28+ eor x1, x30, x17, ror #36+ bcax v8.16b, v8.16b, v10.16b, v9.16b+ eor x14, x0, x14, ror #8+ eor x22, x28, x22, ror #44+ bic v31.16b, v11.16b, v10.16b+ eor x8, x27, x8, ror #56+ eor x17, x27, x7, ror #19+ eor v9.16b, v31.16b, v9.16b+ eor x15, x0, x15, ror #62+ bic x7, x20, x22, ror #47+ bcax v10.16b, v15.16b, v12.16b, v16.16b+ eor x4, x26, x4, ror #54+ eor x0, x0, x12, ror #3+ bic v31.16b, v13.16b, v12.16b+ eor x28, x28, x23, ror #58+ eor x23, x26, x2, ror #61+ eor v11.16b, v31.16b, v16.16b+ eor x26, x26, x5, ror #25+ bcax v12.16b, v12.16b, v14.16b, v13.16b+ eor x2, x7, x16, ror #39+ bic x7, x9, x20, ror #42+ bic v31.16b, v15.16b, v14.16b+ bic x30, x15, x9, ror #16+ eor x7, x7, x22, ror #25+ eor v13.16b, v31.16b, v13.16b+ eor x12, x30, x20, ror #58+ bic x20, x22, x16, ror #56+ bic v31.16b, v16.16b, v15.16b+ eor x30, x27, x6, ror #43+ eor x22, x20, x15, ror #23+ eor v14.16b, v31.16b, v14.16b+ bic x6, x19, x13, ror #42+ eor x6, x6, x17, ror #41+ bcax v15.16b, v20.16b, v17.16b, v21.16b+ bic x5, x13, x17, ror #63+ eor x5, x21, x5, ror #21+ bic v31.16b, v18.16b, v17.16b+ bic x17, x17, x21, ror #44+ eor x27, x27, x10, ror #23+ eor v16.16b, v31.16b, v21.16b+ bic x21, x21, x25, ror #50+ bic x20, x27, x4, ror #25+ bcax v17.16b, v17.16b, v19.16b, v18.16b+ bic x10, x16, x15, ror #31+ eor x16, x21, x19, ror #43+ bic v31.16b, v20.16b, v19.16b+ eor x21, x17, x25, ror #30+ bic x19, x25, x19, ror #57+ eor v18.16b, v31.16b, v18.16b+ ldr x25, [sp, #0x18]+ bcax v19.16b, v19.16b, v21.16b, v20.16b+ eor x17, x10, x9, ror #47+ ldr x9, [sp, #0x8]+ bic v31.16b, v22.16b, v1.16b+ eor x15, x20, x28, ror #27+ bic x20, x4, x28, ror #2+ eor v20.16b, v31.16b, v0.16b+ eor x10, x20, x1, ror #50+ bic x20, x11, x27, ror #60+ bcax v21.16b, v1.16b, v23.16b, v22.16b+ eor x20, x20, x4, ror #21+ bic x4, x28, x1, ror #48+ bic v31.16b, v24.16b, v23.16b+ bic x1, x1, x11, ror #57+ ldr x28, [x9, x25, lsl #3]+ eor v22.16b, v31.16b, v22.16b+ ldr x9, [sp, #0xd0]+ add x25, x25, #0x1+ bcax v23.16b, v23.16b, v0.16b, v24.16b+ str x25, [sp, #0x18]+ cmp x25, #0x17+ bic v31.16b, v1.16b, v0.16b+ eor x25, x1, x27, ror #53+ bic x27, x30, x26, ror #47+ eor v24.16b, v31.16b, v24.16b+ eor x1, x5, x28+ eor x5, x4, x11, ror #41+ bcax v0.16b, v30.16b, v2.16b, v27.16b+ eor x11, x19, x13, ror #35+ bic v31.16b, v3.16b, v2.16b+ bic x13, x26, x24, ror #10+ eor x28, x27, x24, ror #57+ eor v1.16b, v31.16b, v27.16b+ bic x27, x24, x9, ror #47+ bic x19, x23, x3, ror #9+ bcax v2.16b, v2.16b, v4.16b, v3.16b+ bic x4, x29, x14, ror #41+ eor x24, x19, x29, ror #44+ bic v31.16b, v30.16b, v4.16b+ bic x29, x3, x29, ror #35+ eor x13, x13, x9, ror #57+ eor v3.16b, v31.16b, v3.16b+ eor x19, x29, x14, ror #12+ bic x29, x9, x0, ror #19+ bcax v4.16b, v4.16b, v27.16b, v30.16b+ bic x14, x14, x8, ror #5+ eor x9, x14, x23, ror #43+ eor x14, x4, x8, ror #46+ bic x23, x8, x23, ror #38+ eor x8, x27, x0, ror #2+ eor x4, x23, x3, ror #47+ bic x3, x0, x30, ror #5+ eor x23, x3, x26, ror #52+ eor x3, x29, x30, ror #24+ ldr x30, [sp, #0x10]+ ld1r { v28.2d }, [x30], #8+ str x30, [sp, #0x10]+ eor v0.16b, v0.16b, v28.16b++Lmlk_keccak_f1600_x4_v8a_v84a_scalar_hybrid_loop:+ eor x0, x15, x11, ror #52+ eor x0, x0, x13, ror #48+ eor3 v30.16b, v0.16b, v5.16b, v10.16b+ eor v30.16b, v30.16b, v15.16b+ eor x26, x8, x9, ror #57+ eor x27, x0, x14, ror #10+ eor v30.16b, v30.16b, v20.16b+ eor x29, x16, x28, ror #63+ eor x26, x26, x6, ror #51+ eor3 v29.16b, v1.16b, v6.16b, v11.16b+ eor x30, x23, x22, ror #50+ eor x0, x26, x10, ror #31+ eor v29.16b, v29.16b, v16.16b+ eor x29, x29, x19, ror #37+ eor x27, x27, x12, ror #5+ eor v29.16b, v29.16b, v21.16b+ eor x30, x30, x24, ror #34+ eor x0, x0, x7, ror #27+ eor3 v28.16b, v2.16b, v7.16b, v12.16b+ eor x26, x30, x21, ror #26+ eor x26, x26, x25, ror #15+ eor v28.16b, v28.16b, v17.16b+ ror x30, x27, #0x3e+ eor x30, x30, x26, ror #57+ eor v28.16b, v28.16b, v22.16b+ ror x26, x26, #0x3a+ eor x16, x30, x16+ eor3 v27.16b, v3.16b, v8.16b, v13.16b+ eor x28, x30, x28, ror #63+ str x28, [sp, #0xd0]+ eor v27.16b, v27.16b, v18.16b+ eor x29, x29, x17, ror #36+ eor x28, x1, x2, ror #61+ eor v27.16b, v27.16b, v23.16b+ eor x19, x30, x19, ror #37+ eor x29, x29, x20, ror #2+ eor3 v26.16b, v4.16b, v9.16b, v14.16b+ eor x28, x28, x4, ror #54+ eor x26, x26, x0, ror #55+ eor v26.16b, v26.16b, v19.16b+ eor x28, x28, x3, ror #39+ eor x28, x28, x5, ror #25+ eor v26.16b, v26.16b, v24.16b+ ror x0, x0, #0x38+ eor x0, x0, x29, ror #63+ rax1 v25.2d, v30.2d, v28.2d+ eor x27, x28, x27, ror #61+ eor x13, x0, x13, ror #46+ add v31.2d, v26.2d, v26.2d+ eor x28, x29, x28, ror #63+ eor x29, x30, x20, ror #2+ sri v31.2d, v26.2d, #0x3f+ eor x20, x26, x3, ror #39+ eor x11, x0, x11, ror #50+ eor v28.16b, v31.16b, v28.16b+ eor x25, x28, x25, ror #9+ eor x3, x28, x21, ror #20+ rax1 v26.2d, v26.2d, v29.2d+ eor x21, x26, x1+ add v31.2d, v27.2d, v27.2d+ eor x9, x27, x9, ror #49+ eor x24, x28, x24, ror #28+ sri v31.2d, v27.2d, #0x3f+ eor x1, x30, x17, ror #36+ eor x14, x0, x14, ror #8+ eor v29.16b, v31.16b, v29.16b+ eor x22, x28, x22, ror #44+ eor x8, x27, x8, ror #56+ rax1 v27.2d, v27.2d, v30.2d+ eor x17, x27, x7, ror #19+ eor x15, x0, x15, ror #62+ eor v30.16b, v0.16b, v26.16b+ bic x7, x20, x22, ror #47+ eor x4, x26, x4, ror #54+ eor v31.16b, v2.16b, v29.16b+ eor x0, x0, x12, ror #3+ eor x28, x28, x23, ror #58+ shl v0.2d, v31.2d, #0x3e+ eor x23, x26, x2, ror #61+ eor x26, x26, x5, ror #25+ sri v0.2d, v31.2d, #0x2+ eor x2, x7, x16, ror #39+ bic x7, x9, x20, ror #42+ xar v2.2d, v12.2d, v29.2d, #0x15+ bic x30, x15, x9, ror #16+ eor x7, x7, x22, ror #25+ eor v31.16b, v13.16b, v28.16b+ eor x12, x30, x20, ror #58+ bic x20, x22, x16, ror #56+ shl v12.2d, v31.2d, #0x19+ eor x30, x27, x6, ror #43+ eor x22, x20, x15, ror #23+ sri v12.2d, v31.2d, #0x27+ bic x6, x19, x13, ror #42+ eor x6, x6, x17, ror #41+ xar v13.2d, v19.2d, v27.2d, #0x38+ bic x5, x13, x17, ror #63+ eor x5, x21, x5, ror #21+ eor v31.16b, v23.16b, v28.16b+ bic x17, x17, x21, ror #44+ eor x27, x27, x10, ror #23+ shl v19.2d, v31.2d, #0x38+ bic x21, x21, x25, ror #50+ bic x20, x27, x4, ror #25+ sri v19.2d, v31.2d, #0x8+ bic x10, x16, x15, ror #31+ eor x16, x21, x19, ror #43+ xar v23.2d, v15.2d, v26.2d, #0x17+ eor x21, x17, x25, ror #30+ bic x19, x25, x19, ror #57+ eor v31.16b, v1.16b, v25.16b+ ldr x25, [sp, #0x18]+ eor x17, x10, x9, ror #47+ shl v15.2d, v31.2d, #0x1+ ldr x9, [sp, #0x8]+ sri v15.2d, v31.2d, #0x3f+ eor x15, x20, x28, ror #27+ bic x20, x4, x28, ror #2+ xar v1.2d, v8.2d, v28.2d, #0x9+ eor x10, x20, x1, ror #50+ bic x20, x11, x27, ror #60+ eor v31.16b, v16.16b, v25.16b+ eor x20, x20, x4, ror #21+ bic x4, x28, x1, ror #48+ shl v8.2d, v31.2d, #0x2d+ bic x1, x1, x11, ror #57+ ldr x28, [x9, x25, lsl #3]+ sri v8.2d, v31.2d, #0x13+ ldr x9, [sp, #0xd0]+ add x25, x25, #0x1+ xar v16.2d, v7.2d, v29.2d, #0x3a+ str x25, [sp, #0x18]+ cmp x25, #0x17+ eor v31.16b, v10.16b, v26.16b+ eor x25, x1, x27, ror #53+ bic x27, x30, x26, ror #47+ shl v7.2d, v31.2d, #0x3+ eor x1, x5, x28+ eor x5, x4, x11, ror #41+ sri v7.2d, v31.2d, #0x3d+ eor x11, x19, x13, ror #35+ bic x13, x26, x24, ror #10+ xar v10.2d, v3.2d, v28.2d, #0x24+ eor x28, x27, x24, ror #57+ bic x27, x24, x9, ror #47+ eor v31.16b, v18.16b, v28.16b+ bic x19, x23, x3, ror #9+ bic x4, x29, x14, ror #41+ shl v3.2d, v31.2d, #0x15+ eor x24, x19, x29, ror #44+ bic x29, x3, x29, ror #35+ sri v3.2d, v31.2d, #0x2b+ eor x13, x13, x9, ror #57+ eor x19, x29, x14, ror #12+ xar v18.2d, v17.2d, v29.2d, #0x31+ bic x29, x9, x0, ror #19+ bic x14, x14, x8, ror #5+ eor v31.16b, v11.16b, v25.16b+ eor x9, x14, x23, ror #43+ eor x14, x4, x8, ror #46+ shl v17.2d, v31.2d, #0xa+ bic x23, x8, x23, ror #38+ eor x8, x27, x0, ror #2+ sri v17.2d, v31.2d, #0x36+ eor x4, x23, x3, ror #47+ bic x3, x0, x30, ror #5+ xar v11.2d, v9.2d, v27.2d, #0x2c+ eor x23, x3, x26, ror #52+ eor x3, x29, x30, ror #24+ eor v31.16b, v22.16b, v29.16b+ eor x0, x15, x11, ror #52+ shl v9.2d, v31.2d, #0x3d+ eor x0, x0, x13, ror #48+ eor x26, x8, x9, ror #57+ sri v9.2d, v31.2d, #0x3+ eor x27, x0, x14, ror #10+ eor x29, x16, x28, ror #63+ xar v22.2d, v14.2d, v27.2d, #0x19+ eor x26, x26, x6, ror #51+ eor x30, x23, x22, ror #50+ eor v31.16b, v20.16b, v26.16b+ eor x0, x26, x10, ror #31+ eor x29, x29, x19, ror #37+ shl v14.2d, v31.2d, #0x12+ eor x27, x27, x12, ror #5+ eor x30, x30, x24, ror #34+ sri v14.2d, v31.2d, #0x2e+ eor x0, x0, x7, ror #27+ eor x26, x30, x21, ror #26+ xar v20.2d, v4.2d, v27.2d, #0x25+ eor x26, x26, x25, ror #15+ ror x30, x27, #0x3e+ eor v31.16b, v24.16b, v27.16b+ eor x30, x30, x26, ror #57+ ror x26, x26, #0x3a+ shl v4.2d, v31.2d, #0xe+ eor x16, x30, x16+ eor x28, x30, x28, ror #63+ sri v4.2d, v31.2d, #0x32+ str x28, [sp, #0xd0]+ eor x29, x29, x17, ror #36+ xar v24.2d, v21.2d, v25.2d, #0x3e+ eor x28, x1, x2, ror #61+ eor x19, x30, x19, ror #37+ eor v31.16b, v5.16b, v26.16b+ eor x29, x29, x20, ror #2+ eor x28, x28, x4, ror #54+ shl v21.2d, v31.2d, #0x24+ eor x26, x26, x0, ror #55+ eor x28, x28, x3, ror #39+ sri v21.2d, v31.2d, #0x1c+ eor x28, x28, x5, ror #25+ ror x0, x0, #0x38+ xar v27.2d, v6.2d, v25.2d, #0x14+ eor x0, x0, x29, ror #63+ eor x27, x28, x27, ror #61+ bic v31.16b, v7.16b, v11.16b+ eor x13, x0, x13, ror #46+ eor x28, x29, x28, ror #63+ eor v5.16b, v31.16b, v10.16b+ eor x29, x30, x20, ror #2+ eor x20, x26, x3, ror #39+ bcax v6.16b, v11.16b, v8.16b, v7.16b+ eor x11, x0, x11, ror #50+ eor x25, x28, x25, ror #9+ bic v31.16b, v9.16b, v8.16b+ eor x3, x28, x21, ror #20+ eor v7.16b, v31.16b, v7.16b+ eor x21, x26, x1+ eor x9, x27, x9, ror #49+ bcax v8.16b, v8.16b, v10.16b, v9.16b+ eor x24, x28, x24, ror #28+ eor x1, x30, x17, ror #36+ bic v31.16b, v11.16b, v10.16b+ eor x14, x0, x14, ror #8+ eor x22, x28, x22, ror #44+ eor v9.16b, v31.16b, v9.16b+ eor x8, x27, x8, ror #56+ eor x17, x27, x7, ror #19+ bcax v10.16b, v15.16b, v12.16b, v16.16b+ eor x15, x0, x15, ror #62+ bic x7, x20, x22, ror #47+ bic v31.16b, v13.16b, v12.16b+ eor x4, x26, x4, ror #54+ eor x0, x0, x12, ror #3+ eor v11.16b, v31.16b, v16.16b+ eor x28, x28, x23, ror #58+ eor x23, x26, x2, ror #61+ bcax v12.16b, v12.16b, v14.16b, v13.16b+ eor x26, x26, x5, ror #25+ eor x2, x7, x16, ror #39+ bic v31.16b, v15.16b, v14.16b+ bic x7, x9, x20, ror #42+ bic x30, x15, x9, ror #16+ eor v13.16b, v31.16b, v13.16b+ eor x7, x7, x22, ror #25+ eor x12, x30, x20, ror #58+ bic v31.16b, v16.16b, v15.16b+ bic x20, x22, x16, ror #56+ eor x30, x27, x6, ror #43+ eor v14.16b, v31.16b, v14.16b+ eor x22, x20, x15, ror #23+ bic x6, x19, x13, ror #42+ bcax v15.16b, v20.16b, v17.16b, v21.16b+ eor x6, x6, x17, ror #41+ bic x5, x13, x17, ror #63+ bic v31.16b, v18.16b, v17.16b+ eor x5, x21, x5, ror #21+ bic x17, x17, x21, ror #44+ eor v16.16b, v31.16b, v21.16b+ eor x27, x27, x10, ror #23+ bic x21, x21, x25, ror #50+ bcax v17.16b, v17.16b, v19.16b, v18.16b+ bic x20, x27, x4, ror #25+ bic x10, x16, x15, ror #31+ bic v31.16b, v20.16b, v19.16b+ eor x16, x21, x19, ror #43+ eor x21, x17, x25, ror #30+ eor v18.16b, v31.16b, v18.16b+ bic x19, x25, x19, ror #57+ ldr x25, [sp, #0x18]+ bcax v19.16b, v19.16b, v21.16b, v20.16b+ eor x17, x10, x9, ror #47+ bic v31.16b, v22.16b, v1.16b+ ldr x9, [sp, #0x8]+ eor x15, x20, x28, ror #27+ eor v20.16b, v31.16b, v0.16b+ bic x20, x4, x28, ror #2+ eor x10, x20, x1, ror #50+ bcax v21.16b, v1.16b, v23.16b, v22.16b+ bic x20, x11, x27, ror #60+ eor x20, x20, x4, ror #21+ bic v31.16b, v24.16b, v23.16b+ bic x4, x28, x1, ror #48+ bic x1, x1, x11, ror #57+ eor v22.16b, v31.16b, v22.16b+ ldr x28, [x9, x25, lsl #3]+ ldr x9, [sp, #0xd0]+ bcax v23.16b, v23.16b, v0.16b, v24.16b+ add x25, x25, #0x1+ str x25, [sp, #0x18]+ bic v31.16b, v1.16b, v0.16b+ cmp x25, #0x17+ eor x25, x1, x27, ror #53+ eor v24.16b, v31.16b, v24.16b+ bic x27, x30, x26, ror #47+ eor x1, x5, x28+ bcax v0.16b, v30.16b, v2.16b, v27.16b+ eor x5, x4, x11, ror #41+ eor x11, x19, x13, ror #35+ bic v31.16b, v3.16b, v2.16b+ bic x13, x26, x24, ror #10+ eor x28, x27, x24, ror #57+ eor v1.16b, v31.16b, v27.16b+ bic x27, x24, x9, ror #47+ bic x19, x23, x3, ror #9+ bcax v2.16b, v2.16b, v4.16b, v3.16b+ bic x4, x29, x14, ror #41+ eor x24, x19, x29, ror #44+ bic v31.16b, v30.16b, v4.16b+ bic x29, x3, x29, ror #35+ eor x13, x13, x9, ror #57+ eor v3.16b, v31.16b, v3.16b+ eor x19, x29, x14, ror #12+ bic x29, x9, x0, ror #19+ bcax v4.16b, v4.16b, v27.16b, v30.16b+ bic x14, x14, x8, ror #5+ eor x9, x14, x23, ror #43+ eor x14, x4, x8, ror #46+ bic x23, x8, x23, ror #38+ eor x8, x27, x0, ror #2+ eor x4, x23, x3, ror #47+ bic x3, x0, x30, ror #5+ eor x23, x3, x26, ror #52+ eor x3, x29, x30, ror #24+ ldr x30, [sp, #0x10]+ ld1r { v28.2d }, [x30], #8+ str x30, [sp, #0x10]+ eor v0.16b, v0.16b, v28.16b++Lmlk_keccak_f1600_x4_v8a_v84a_scalar_hybrid_loop_end:+ b.le Lmlk_keccak_f1600_x4_v8a_v84a_scalar_hybrid_loop+ ror x2, x2, #0x3d+ ror x3, x3, #0x27+ ror x4, x4, #0x36+ ror x5, x5, #0x19+ ror x6, x6, #0x2b+ ror x7, x7, #0x13+ ror x8, x8, #0x38+ ror x9, x9, #0x31+ ror x10, x10, #0x17+ ror x11, x11, #0x32+ ror x12, x12, #0x3+ ror x13, x13, #0x2e+ ror x14, x14, #0x8+ ror x15, x15, #0x3e+ ror x17, x17, #0x24+ ror x28, x28, #0x3f+ ror x19, x19, #0x25+ ror x20, x20, #0x2+ ror x21, x21, #0x14+ ror x22, x22, #0x2c+ ror x23, x23, #0x3a+ ror x24, x24, #0x1c+ ror x25, x25, #0x9+ ldr x30, [sp, #0x20]+ cmp x30, #0x1+ b.eq Lmlk_keccak_f1600_x4_v8a_v84a_scalar_hybrid_done+ mov x30, #0x1 // =1+ str x30, [sp, #0x20]+ ldr x0, [sp]+ add x0, x0, #0x190+ stp x1, x6, [x0]+ stp x11, x16, [x0, #0x10]+ stp x21, x2, [x0, #0x20]+ stp x7, x12, [x0, #0x30]+ stp x17, x22, [x0, #0x40]+ stp x3, x8, [x0, #0x50]+ stp x13, x28, [x0, #0x60]+ stp x23, x4, [x0, #0x70]+ stp x9, x14, [x0, #0x80]+ stp x19, x24, [x0, #0x90]+ stp x5, x10, [x0, #0xa0]+ stp x15, x20, [x0, #0xb0]+ str x25, [x0, #0xc0]+ sub x0, x0, #0x190+ add x0, x0, #0x258+ ldp x1, x6, [x0]+ ldp x11, x16, [x0, #0x10]+ ldp x21, x2, [x0, #0x20]+ ldp x7, x12, [x0, #0x30]+ ldp x17, x22, [x0, #0x40]+ ldp x3, x8, [x0, #0x50]+ ldp x13, x28, [x0, #0x60]+ ldp x23, x4, [x0, #0x70]+ ldp x9, x14, [x0, #0x80]+ ldp x19, x24, [x0, #0x90]+ ldp x5, x10, [x0, #0xa0]+ ldp x15, x20, [x0, #0xb0]+ ldr x25, [x0, #0xc0]+ sub x0, x0, #0x258+ b Lmlk_keccak_f1600_x4_v8a_v84a_scalar_hybrid_initial++Lmlk_keccak_f1600_x4_v8a_v84a_scalar_hybrid_done:+ ldr x0, [sp]+ add x0, x0, #0x258+ stp x1, x6, [x0]+ stp x11, x16, [x0, #0x10]+ stp x21, x2, [x0, #0x20]+ stp x7, x12, [x0, #0x30]+ stp x17, x22, [x0, #0x40]+ stp x3, x8, [x0, #0x50]+ stp x13, x28, [x0, #0x60]+ stp x23, x4, [x0, #0x70]+ stp x9, x14, [x0, #0x80]+ stp x19, x24, [x0, #0x90]+ stp x5, x10, [x0, #0xa0]+ stp x15, x20, [x0, #0xb0]+ str x25, [x0, #0xc0]+ sub x0, x0, #0x258+ add x4, x0, #0xc8+ trn1 v25.2d, v0.2d, v1.2d+ trn1 v26.2d, v2.2d, v3.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v0.2d, v1.2d+ trn2 v28.2d, v2.2d, v3.2d+ st1 { v27.2d, v28.2d }, [x4], #32+ trn1 v25.2d, v4.2d, v5.2d+ trn1 v26.2d, v6.2d, v7.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v4.2d, v5.2d+ trn2 v28.2d, v6.2d, v7.2d+ st1 { v27.2d, v28.2d }, [x4], #32+ trn1 v25.2d, v8.2d, v9.2d+ trn1 v26.2d, v10.2d, v11.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v8.2d, v9.2d+ trn2 v28.2d, v10.2d, v11.2d+ st1 { v27.2d, v28.2d }, [x4], #32+ trn1 v25.2d, v12.2d, v13.2d+ trn1 v26.2d, v14.2d, v15.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v12.2d, v13.2d+ trn2 v28.2d, v14.2d, v15.2d+ st1 { v27.2d, v28.2d }, [x4], #32+ trn1 v25.2d, v16.2d, v17.2d+ trn1 v26.2d, v18.2d, v19.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v16.2d, v17.2d+ trn2 v28.2d, v18.2d, v19.2d+ st1 { v27.2d, v28.2d }, [x4], #32+ trn1 v25.2d, v20.2d, v21.2d+ trn1 v26.2d, v22.2d, v23.2d+ stp q25, q26, [x0], #0x20+ trn2 v27.2d, v20.2d, v21.2d+ trn2 v28.2d, v22.2d, v23.2d+ st1 { v27.2d, v28.2d }, [x4], #32+ str d24, [x0]+ trn2 v25.2d, v24.2d, v24.2d+ str d25, [x4]+ ldp d8, d9, [sp, #0x90]+ .cfi_restore d8+ .cfi_restore d9+ ldp d10, d11, [sp, #0xa0]+ .cfi_restore d10+ .cfi_restore d11+ ldp d12, d13, [sp, #0xb0]+ .cfi_restore d12+ .cfi_restore d13+ ldp d14, d15, [sp, #0xc0]+ .cfi_restore d14+ .cfi_restore d15+ ldp x19, x20, [sp, #0x30]+ .cfi_restore x19+ .cfi_restore x20+ ldp x21, x22, [sp, #0x40]+ .cfi_restore x21+ .cfi_restore x22+ ldp x23, x24, [sp, #0x50]+ .cfi_restore x23+ .cfi_restore x24+ ldp x25, x26, [sp, #0x60]+ .cfi_restore x25+ .cfi_restore x26+ ldp x27, x28, [sp, #0x70]+ .cfi_restore x27+ .cfi_restore x28+ ldp x29, x30, [sp, #0x80]+ .cfi_restore x29+ .cfi_restore x30+ add sp, sp, #0xe0+ .cfi_adjust_cfa_offset -0xe0+ ret+ .cfi_endproc++MLK_ASM_FN_SIZE(keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm)++#endif /* __ARM_FEATURE_SHA3 */++#endif /* MLK_FIPS202_AARCH64_NEED_X4_V8A_V84A_SCALAR_HYBRID && \+ !MLK_CONFIG_MULTILEVEL_NO_SHARED */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
@@ -0,0 +1,47 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*+ * WARNING: This file is auto-generated from scripts/autogen+ * in the mlkem-native repository.+ * Do not modify it directly.+ */++#include "../../../../common.h"++#if (defined(MLK_FIPS202_AARCH64_NEED_X1_SCALAR) || \+ defined(MLK_FIPS202_AARCH64_NEED_X1_V84A) || \+ defined(MLK_FIPS202_AARCH64_NEED_X2_V84A) || \+ defined(MLK_FIPS202_AARCH64_NEED_X4_V8A_SCALAR_HYBRID) || \+ defined(MLK_FIPS202_AARCH64_NEED_X4_V8A_V84A_SCALAR_HYBRID)) && \+ !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED)++#include "fips202_native_aarch64.h"++MLK_ALIGN MLK_INTERNAL_DATA_DEFINITION const uint64_t+ mlk_keccakf1600_round_constants[24] = {+ 0x0000000000000001, 0x0000000000008082, 0x800000000000808a,+ 0x8000000080008000, 0x000000000000808b, 0x0000000080000001,+ 0x8000000080008081, 0x8000000000008009, 0x000000000000008a,+ 0x0000000000000088, 0x0000000080008009, 0x000000008000000a,+ 0x000000008000808b, 0x800000000000008b, 0x8000000000008089,+ 0x8000000000008003, 0x8000000000008002, 0x8000000000000080,+ 0x000000000000800a, 0x800000008000000a, 0x8000000080008081,+ 0x8000000000008080, 0x0000000080000001, 0x8000000080008008,+};++#else /* (MLK_FIPS202_AARCH64_NEED_X1_SCALAR || \+ MLK_FIPS202_AARCH64_NEED_X1_V84A || MLK_FIPS202_AARCH64_NEED_X2_V84A \+ || MLK_FIPS202_AARCH64_NEED_X4_V8A_SCALAR_HYBRID || \+ MLK_FIPS202_AARCH64_NEED_X4_V8A_V84A_SCALAR_HYBRID) && \+ !MLK_CONFIG_MULTILEVEL_NO_SHARED */++MLK_EMPTY_CU(fips202_aarch64_round_constants)++#endif /* !((MLK_FIPS202_AARCH64_NEED_X1_SCALAR || \+ MLK_FIPS202_AARCH64_NEED_X1_V84A || MLK_FIPS202_AARCH64_NEED_X2_V84A \+ || MLK_FIPS202_AARCH64_NEED_X4_V8A_SCALAR_HYBRID || \+ MLK_FIPS202_AARCH64_NEED_X4_V8A_V84A_SCALAR_HYBRID) && \+ !MLK_CONFIG_MULTILEVEL_NO_SHARED) */
@@ -0,0 +1,26 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#ifndef MLK_FIPS202_NATIVE_AARCH64_X1_SCALAR_H+#define MLK_FIPS202_NATIVE_AARCH64_X1_SCALAR_H++/* Part of backend API */+#define MLK_USE_NATIVE_FIPS202_X1+/* Guard for assembly file */+#define MLK_FIPS202_AARCH64_NEED_X1_SCALAR++#if !defined(__ASSEMBLER__)+#include "../api.h"+#include "src/fips202_native_aarch64.h"+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_keccak_f1600_x1_native(uint64_t *state)+{+ mlk_keccak_f1600_x1_scalar_aarch64_asm(state,+ mlk_keccakf1600_round_constants);+ return MLK_NATIVE_FUNC_SUCCESS;+}+#endif /* !__ASSEMBLER__ */++#endif /* !MLK_FIPS202_NATIVE_AARCH64_X1_SCALAR_H */
@@ -0,0 +1,35 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#ifndef MLK_FIPS202_NATIVE_AARCH64_X1_V84A_H+#define MLK_FIPS202_NATIVE_AARCH64_X1_V84A_H++#if !defined(__ARM_FEATURE_SHA3)+#error This backend can only be used if SHA3 extensions are available.+#endif++/* Part of backend API */+#define MLK_USE_NATIVE_FIPS202_X1+/* Guard for assembly file */+#define MLK_FIPS202_AARCH64_NEED_X1_V84A++#if !defined(__ASSEMBLER__)+#include "../api.h"+#include "src/fips202_native_aarch64.h"+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_keccak_f1600_x1_native(uint64_t *state)+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_AARCH64_NEON) ||+ !mlk_sys_check_capability(MLK_SYS_CAP_AARCH64_SHA3))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }++ mlk_keccak_f1600_x1_v84a_aarch64_asm(state, mlk_keccakf1600_round_constants);+ return MLK_NATIVE_FUNC_SUCCESS;+}+#endif /* !__ASSEMBLER__ */++#endif /* !MLK_FIPS202_NATIVE_AARCH64_X1_V84A_H */
@@ -0,0 +1,38 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#ifndef MLK_FIPS202_NATIVE_AARCH64_X2_V84A_H+#define MLK_FIPS202_NATIVE_AARCH64_X2_V84A_H++#if !defined(__ARM_FEATURE_SHA3)+#error This backend can only be used if SHA3 extensions are available.+#endif++/* Part of backend API */+#define MLK_USE_NATIVE_FIPS202_X4+/* Guard for assembly file */+#define MLK_FIPS202_AARCH64_NEED_X2_V84A++#if !defined(__ASSEMBLER__)+#include "../api.h"+#include "src/fips202_native_aarch64.h"+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_keccak_f1600_x4_native(uint64_t *state)+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_AARCH64_NEON) ||+ !mlk_sys_check_capability(MLK_SYS_CAP_AARCH64_SHA3))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }++ mlk_keccak_f1600_x2_v84a_aarch64_asm(state + 0 * 25,+ mlk_keccakf1600_round_constants);+ mlk_keccak_f1600_x2_v84a_aarch64_asm(state + 2 * 25,+ mlk_keccakf1600_round_constants);+ return MLK_NATIVE_FUNC_SUCCESS;+}+#endif /* !__ASSEMBLER__ */++#endif /* !MLK_FIPS202_NATIVE_AARCH64_X2_V84A_H */
@@ -0,0 +1,31 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#ifndef MLK_FIPS202_NATIVE_AARCH64_X4_V8A_SCALAR_H+#define MLK_FIPS202_NATIVE_AARCH64_X4_V8A_SCALAR_H++/* Part of backend API */+#define MLK_USE_NATIVE_FIPS202_X4+/* Guard for assembly file */+#define MLK_FIPS202_AARCH64_NEED_X4_V8A_SCALAR_HYBRID++#if !defined(__ASSEMBLER__)+#include "../api.h"+#include "src/fips202_native_aarch64.h"+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_keccak_f1600_x4_native(uint64_t *state)+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_AARCH64_NEON))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }++ mlk_keccak_f1600_x4_v8a_scalar_hybrid_aarch64_asm(+ state, mlk_keccakf1600_round_constants);+ return MLK_NATIVE_FUNC_SUCCESS;+}+#endif /* !__ASSEMBLER__ */++#endif /* !MLK_FIPS202_NATIVE_AARCH64_X4_V8A_SCALAR_H */
@@ -0,0 +1,36 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#ifndef MLK_FIPS202_NATIVE_AARCH64_X4_V8A_V84A_SCALAR_H+#define MLK_FIPS202_NATIVE_AARCH64_X4_V8A_V84A_SCALAR_H++#if !defined(__ARM_FEATURE_SHA3)+#error This backend can only be used if SHA3 extensions are available.+#endif++/* Part of backend API */+#define MLK_USE_NATIVE_FIPS202_X4+/* Guard for assembly file */+#define MLK_FIPS202_AARCH64_NEED_X4_V8A_V84A_SCALAR_HYBRID++#if !defined(__ASSEMBLER__)+#include "../api.h"+#include "src/fips202_native_aarch64.h"+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_keccak_f1600_x4_native(uint64_t *state)+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_AARCH64_NEON) ||+ !mlk_sys_check_capability(MLK_SYS_CAP_AARCH64_SHA3))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }++ mlk_keccak_f1600_x4_v8a_v84a_scalar_hybrid_aarch64_asm(+ state, mlk_keccakf1600_round_constants);+ return MLK_NATIVE_FUNC_SUCCESS;+}+#endif /* !__ASSEMBLER__ */++#endif /* !MLK_FIPS202_NATIVE_AARCH64_X4_V8A_V84A_SCALAR_H */
@@ -0,0 +1,117 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#ifndef MLK_FIPS202_NATIVE_API_H+#define MLK_FIPS202_NATIVE_API_H+/*+ * FIPS-202 native interface+ *+ * This header is primarily for documentation purposes.+ * It should not be included by backend implementations.+ */++#include "../../cbmc.h"++/* Backends must return MLK_NATIVE_FUNC_SUCCESS upon success. */+#define MLK_NATIVE_FUNC_SUCCESS (0)+/* Backends may return MLK_NATIVE_FUNC_FALLBACK to signal to the frontend that+ * the target/parameters are unsupported; typically, this would be because of+ * dependencies on CPU features not detected on the host CPU. In this case,+ * the frontend falls back to the default C implementation. */+#define MLK_NATIVE_FUNC_FALLBACK (-1)++/*+ * This is the C<->native interface allowing for the drop-in+ * of custom Keccak-F1600 implementations.+ *+ * A _backend_ is a specific implementation of parts of this interface.+ *+ * You can replace 1-fold or 4-fold batched Keccak-F1600.+ * To enable, set MLK_USE_NATIVE_FIPS202_X1 or MLK_USE_NATIVE_FIPS202_X4+ * in your backend, and define the inline wrappers mlk_keccak_f1600_x1_native()+ * and/or mlk_keccak_f1600_x4_native(), respectively, to forward to your+ * implementation.+ */++#if defined(MLK_USE_NATIVE_FIPS202_X1)+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_keccak_f1600_x1_native(uint64_t *state)+__contract__(+ requires(memory_no_alias(state, sizeof(uint64_t) * 25 * 1))+ assigns(memory_slice(state, sizeof(uint64_t) * 25 * 1))+ ensures(return_value == MLK_NATIVE_FUNC_FALLBACK || return_value == MLK_NATIVE_FUNC_SUCCESS)+ ensures((return_value == MLK_NATIVE_FUNC_FALLBACK) ==> array_unchanged_u64(state, 25 * 1)));+#endif /* MLK_USE_NATIVE_FIPS202_X1 */+#if defined(MLK_USE_NATIVE_FIPS202_X4)+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_keccak_f1600_x4_native(uint64_t *state)+__contract__(+ requires(memory_no_alias(state, sizeof(uint64_t) * 25 * 4))+ assigns(memory_slice(state, sizeof(uint64_t) * 25 * 4))+ ensures(return_value == MLK_NATIVE_FUNC_FALLBACK || return_value == MLK_NATIVE_FUNC_SUCCESS)+ ensures((return_value == MLK_NATIVE_FUNC_FALLBACK) ==> array_unchanged_u64(state, 25 * 4)));+#endif /* MLK_USE_NATIVE_FIPS202_X4 */++/*+ * Native x4 XOR bytes and extract bytes interface.+ *+ * These functions allow backends to provide optimized implementations for+ * XORing input data into the state and extracting output data from the state.+ * This is particularly useful for backends that use a different internal state+ * representation (e.g., bit-interleaved), as conversion can happen during+ * XOR/extract rather than before/after each permutation.+ *+ * NOTE: We assume that the custom representation of the zero state is the+ * all-zero state.+ *+ * MLK_USE_NATIVE_FIPS202_X4_XOR_BYTES: Backend provides native XOR bytes+ * MLK_USE_NATIVE_FIPS202_X4_EXTRACT_BYTES: Backend provides native extract+ * bytes+ */++#if defined(MLK_USE_NATIVE_FIPS202_X4_XOR_BYTES)+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_keccakf1600_xor_bytes_x4_native(+ uint64_t *state, const unsigned char *data0, const unsigned char *data1,+ const unsigned char *data2, const unsigned char *data3, unsigned offset,+ unsigned length)+__contract__(+ requires(0 <= offset && offset <= 25 * sizeof(uint64_t) &&+ 0 <= length && length <= 25 * sizeof(uint64_t) - offset)+ requires(memory_no_alias(state, sizeof(uint64_t) * 25 * 4))+ requires(memory_no_alias(data0, length))+ requires((data0 == data1 &&+ data0 == data2 &&+ data0 == data3) ||+ (memory_no_alias(data1, length) &&+ memory_no_alias(data2, length) &&+ memory_no_alias(data3, length)))+ assigns(memory_slice(state, sizeof(uint64_t) * 25 * 4))+ ensures(return_value == MLK_NATIVE_FUNC_FALLBACK || return_value == MLK_NATIVE_FUNC_SUCCESS)+ ensures((return_value == MLK_NATIVE_FUNC_FALLBACK) ==> array_unchanged_u64(state, 25 * 4)));+#endif /* MLK_USE_NATIVE_FIPS202_X4_XOR_BYTES */++#if defined(MLK_USE_NATIVE_FIPS202_X4_EXTRACT_BYTES)+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_keccakf1600_extract_bytes_x4_native(+ uint64_t *state, unsigned char *data0, unsigned char *data1,+ unsigned char *data2, unsigned char *data3, unsigned offset,+ unsigned length)+__contract__(+ requires(0 <= offset && offset <= 25 * sizeof(uint64_t) &&+ 0 <= length && length <= 25 * sizeof(uint64_t) - offset)+ requires(memory_no_alias(state, sizeof(uint64_t) * 25 * 4))+ requires(memory_no_alias(data0, length))+ requires(memory_no_alias(data1, length))+ requires(memory_no_alias(data2, length))+ requires(memory_no_alias(data3, length))+ assigns(memory_slice(data0, length))+ assigns(memory_slice(data1, length))+ assigns(memory_slice(data2, length))+ assigns(memory_slice(data3, length))+ ensures(return_value == MLK_NATIVE_FUNC_FALLBACK || return_value == MLK_NATIVE_FUNC_SUCCESS));+#endif /* MLK_USE_NATIVE_FIPS202_X4_EXTRACT_BYTES */++#endif /* !MLK_FIPS202_NATIVE_API_H */
@@ -0,0 +1,29 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#ifndef MLK_FIPS202_NATIVE_AUTO_H+#define MLK_FIPS202_NATIVE_AUTO_H++/*+ * Default FIPS202 backend+ */+#include "../../sys.h"++#if defined(MLK_SYS_AARCH64)+#include "aarch64/auto.h"+#endif++/* The x86_64 backend requires toolchain support for the SysV ABI */+#if defined(MLK_SYS_X86_64_AVX2) && defined(MLK_SYSV_ABI_SUPPORTED)+#include "x86_64/keccak_f1600_x4_avx2.h"+#endif++/* We do not yet include the FIPS202 backend for Armv8.1-M+MVE by default+ * as it is still experimental and undergoing review. */+/* #if defined(MLK_SYS_ARMV81M_MVE) */+/* #include "armv81m/mve.h" */+/* #endif */++#endif /* !MLK_FIPS202_NATIVE_AUTO_H */
@@ -0,0 +1,33 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#ifndef MLK_FIPS202_NATIVE_X86_64_KECCAK_F1600_X4_AVX2_H+#define MLK_FIPS202_NATIVE_X86_64_KECCAK_F1600_X4_AVX2_H++#include "../../../common.h"++#define MLK_FIPS202_X86_64_NEED_X4_AVX2++/* Part of backend API */+#define MLK_USE_NATIVE_FIPS202_X4++#if !defined(__ASSEMBLER__)+#include "../api.h"+#include "src/fips202_native_x86_64.h"+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_keccak_f1600_x4_native(uint64_t *state)+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_X86_64_AVX2))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }++ mlk_keccak_f1600_x4_avx2_asm(state, mlk_keccakf1600_round_constants,+ mlk_keccak_rho8, mlk_keccak_rho56);+ return MLK_NATIVE_FUNC_SUCCESS;+}+#endif /* !__ASSEMBLER__ */++#endif /* !MLK_FIPS202_NATIVE_X86_64_KECCAK_F1600_X4_AVX2_H */
@@ -0,0 +1,44 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#ifndef MLK_FIPS202_NATIVE_X86_64_SRC_FIPS202_NATIVE_X86_64_H+#define MLK_FIPS202_NATIVE_X86_64_SRC_FIPS202_NATIVE_X86_64_H++#include "../../../../cbmc.h"+#include "../../../../common.h"++/* TODO: Reconsider whether this check is needed -- x86_64 is always+ * little-endian, so the backend selection already implies this. */+#ifndef MLK_SYS_LITTLE_ENDIAN+#error Expecting a little-endian platform+#endif++#define mlk_keccakf1600_round_constants \+ MLK_NAMESPACE(keccakf1600_round_constants)+MLK_INTERNAL_DATA_DECLARATION const uint64_t+ mlk_keccakf1600_round_constants[24];++#define mlk_keccak_rho8 MLK_NAMESPACE(keccak_rho8)+MLK_INTERNAL_DATA_DECLARATION const uint64_t mlk_keccak_rho8[4];++#define mlk_keccak_rho56 MLK_NAMESPACE(keccak_rho56)+MLK_INTERNAL_DATA_DECLARATION const uint64_t mlk_keccak_rho56[4];++#define mlk_keccak_f1600_x4_avx2_asm MLK_NAMESPACE(keccak_f1600_x4_avx2_asm)+MLK_SYSV_ABI+void mlk_keccak_f1600_x4_avx2_asm(uint64_t states[100], const uint64_t rc[24],+ const uint64_t rho8[4],+ const uint64_t rho56[4])+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/x86_64/proofs/keccak_f1600_x4_avx2_asm.ml */+__contract__(+ requires(memory_no_alias(states, sizeof(uint64_t) * 25 * 4))+ requires(rc == mlk_keccakf1600_round_constants)+ requires(rho8 == mlk_keccak_rho8)+ requires(rho56 == mlk_keccak_rho56)+ assigns(memory_slice(states, sizeof(uint64_t) * 25 * 4))+);++#endif /* !MLK_FIPS202_NATIVE_X86_64_SRC_FIPS202_NATIVE_X86_64_H */
@@ -0,0 +1,487 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+#include "../../../../common.h"++#if defined(MLK_FIPS202_X86_64_NEED_X4_AVX2) && \+ !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED)+/*yaml+ Name: keccak_f1600_x4_avx2_asm+ Description: x86_64 AVX2 Keccak-f[1600] permutation for four sequential states+ Signature: void mlk_keccak_f1600_x4_avx2_asm(uint64_t states[100], const uint64_t rc[24], const uint64_t rho8[4], const uint64_t rho56[4])+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 800+ permissions: read/write+ c_parameter: uint64_t states[100]+ description: Four sequential Keccak states (4 x 25 x uint64_t)+ rsi:+ type: buffer+ size_bytes: 192+ permissions: read-only+ c_parameter: const uint64_t rc[24]+ description: Round constants (24 x uint64_t)+ rdx:+ type: buffer+ size_bytes: 32+ permissions: read-only+ c_parameter: const uint64_t rho8[4]+ description: Rotation constant rho8 (4 x uint64_t)+ rcx:+ type: buffer+ size_bytes: 32+ permissions: read-only+ c_parameter: const uint64_t rho56[4]+ description: Rotation constant rho56 (4 x uint64_t)+*/+++/*+ * WARNING: This file is auto-derived from the mlkem-native source file+ * dev/fips202/x86_64/src/keccak_f1600_x4_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLK_ASM_NAMESPACE(keccak_f1600_x4_avx2_asm)+MLK_ASM_FN_SYMBOL(keccak_f1600_x4_avx2_asm)++ .cfi_startproc+ movq %rsp, %r11+ .cfi_def_cfa_register %r11+ andq $-0x20, %rsp+ subq $0x300, %rsp # imm = 0x300+ vmovdqu (%rdi), %ymm0+ vmovdqu 0xc8(%rdi), %ymm3+ vmovdqu 0x190(%rdi), %ymm1+ vmovdqu 0x258(%rdi), %ymm4+ vpunpcklqdq %ymm3, %ymm0, %ymm2 # ymm2 = ymm0[0],ymm3[0],ymm0[2],ymm3[2]+ vpunpckhqdq %ymm3, %ymm0, %ymm0 # ymm0 = ymm0[1],ymm3[1],ymm0[3],ymm3[3]+ vpunpcklqdq %ymm4, %ymm1, %ymm3 # ymm3 = ymm1[0],ymm4[0],ymm1[2],ymm4[2]+ vperm2i128 $0x20, %ymm3, %ymm2, %ymm7 # ymm7 = ymm2[0,1],ymm3[0,1]+ vpunpckhqdq %ymm4, %ymm1, %ymm1 # ymm1 = ymm1[1],ymm4[1],ymm1[3],ymm4[3]+ vperm2i128 $0x31, %ymm3, %ymm2, %ymm3 # ymm3 = ymm2[2,3],ymm3[2,3]+ vmovdqu 0x278(%rdi), %ymm4+ vmovdqu %ymm3, 0x40(%rsp)+ vperm2i128 $0x31, %ymm1, %ymm0, %ymm3 # ymm3 = ymm0[2,3],ymm1[2,3]+ vmovdqu %ymm7, (%rsp)+ vperm2i128 $0x20, %ymm1, %ymm0, %ymm7 # ymm7 = ymm0[0,1],ymm1[0,1]+ vmovdqu 0x20(%rdi), %ymm0+ vmovdqu 0x1b0(%rdi), %ymm1+ vmovdqu %ymm3, 0x60(%rsp)+ vmovdqu 0xe8(%rdi), %ymm3+ vmovdqu %ymm7, 0x20(%rsp)+ vpunpcklqdq %ymm3, %ymm0, %ymm2 # ymm2 = ymm0[0],ymm3[0],ymm0[2],ymm3[2]+ vpunpckhqdq %ymm3, %ymm0, %ymm0 # ymm0 = ymm0[1],ymm3[1],ymm0[3],ymm3[3]+ vpunpcklqdq %ymm4, %ymm1, %ymm3 # ymm3 = ymm1[0],ymm4[0],ymm1[2],ymm4[2]+ vperm2i128 $0x20, %ymm3, %ymm2, %ymm7 # ymm7 = ymm2[0,1],ymm3[0,1]+ vpunpckhqdq %ymm4, %ymm1, %ymm1 # ymm1 = ymm1[1],ymm4[1],ymm1[3],ymm4[3]+ vperm2i128 $0x31, %ymm3, %ymm2, %ymm3 # ymm3 = ymm2[2,3],ymm3[2,3]+ vmovdqu 0x298(%rdi), %ymm4+ vperm2i128 $0x31, %ymm1, %ymm0, %ymm14 # ymm14 = ymm0[2,3],ymm1[2,3]+ vmovdqu %ymm7, 0x80(%rsp)+ vperm2i128 $0x20, %ymm1, %ymm0, %ymm7 # ymm7 = ymm0[0,1],ymm1[0,1]+ vmovdqu 0x40(%rdi), %ymm0+ vmovdqu 0x1d0(%rdi), %ymm1+ vmovdqu %ymm3, 0xc0(%rsp)+ vmovdqu 0x108(%rdi), %ymm3+ vmovdqu %ymm14, %ymm10+ vmovdqu %ymm7, 0xa0(%rsp)+ vpunpcklqdq %ymm3, %ymm0, %ymm2 # ymm2 = ymm0[0],ymm3[0],ymm0[2],ymm3[2]+ vpunpckhqdq %ymm3, %ymm0, %ymm0 # ymm0 = ymm0[1],ymm3[1],ymm0[3],ymm3[3]+ vpunpcklqdq %ymm4, %ymm1, %ymm3 # ymm3 = ymm1[0],ymm4[0],ymm1[2],ymm4[2]+ vpunpckhqdq %ymm4, %ymm1, %ymm1 # ymm1 = ymm1[1],ymm4[1],ymm1[3],ymm4[3]+ vperm2i128 $0x20, %ymm3, %ymm2, %ymm11 # ymm11 = ymm2[0,1],ymm3[0,1]+ vperm2i128 $0x31, %ymm3, %ymm2, %ymm3 # ymm3 = ymm2[2,3],ymm3[2,3]+ vperm2i128 $0x20, %ymm1, %ymm0, %ymm7 # ymm7 = ymm0[0,1],ymm1[0,1]+ vmovdqu %ymm3, 0x100(%rsp)+ vperm2i128 $0x31, %ymm1, %ymm0, %ymm8 # ymm8 = ymm0[2,3],ymm1[2,3]+ vmovdqu 0x128(%rdi), %ymm3+ vmovdqu 0x60(%rdi), %ymm0+ vmovdqu 0x1f0(%rdi), %ymm1+ vmovdqu %ymm7, 0xe0(%rsp)+ vmovdqu %ymm11, %ymm14+ vmovdqu 0x2b8(%rdi), %ymm4+ vmovdqu 0x2f8(%rdi), %ymm5+ vpunpcklqdq %ymm3, %ymm0, %ymm2 # ymm2 = ymm0[0],ymm3[0],ymm0[2],ymm3[2]+ vpunpckhqdq %ymm3, %ymm0, %ymm0 # ymm0 = ymm0[1],ymm3[1],ymm0[3],ymm3[3]+ vpunpcklqdq %ymm4, %ymm1, %ymm3 # ymm3 = ymm1[0],ymm4[0],ymm1[2],ymm4[2]+ vpunpckhqdq %ymm4, %ymm1, %ymm1 # ymm1 = ymm1[1],ymm4[1],ymm1[3],ymm4[3]+ vmovdqu 0x2d8(%rdi), %ymm4+ vperm2i128 $0x20, %ymm3, %ymm2, %ymm15 # ymm15 = ymm2[0,1],ymm3[0,1]+ vperm2i128 $0x31, %ymm3, %ymm2, %ymm3 # ymm3 = ymm2[2,3],ymm3[2,3]+ vperm2i128 $0x20, %ymm1, %ymm0, %ymm7 # ymm7 = ymm0[0,1],ymm1[0,1]+ vperm2i128 $0x31, %ymm1, %ymm0, %ymm9 # ymm9 = ymm0[2,3],ymm1[2,3]+ vmovdqu %ymm3, 0x140(%rsp)+ vmovdqu 0x80(%rdi), %ymm0+ vmovdqu 0x148(%rdi), %ymm3+ vmovdqu 0x210(%rdi), %ymm1+ vmovdqu %ymm7, 0x120(%rsp)+ vpunpcklqdq %ymm3, %ymm0, %ymm2 # ymm2 = ymm0[0],ymm3[0],ymm0[2],ymm3[2]+ vpunpckhqdq %ymm3, %ymm0, %ymm0 # ymm0 = ymm0[1],ymm3[1],ymm0[3],ymm3[3]+ vpunpcklqdq %ymm4, %ymm1, %ymm3 # ymm3 = ymm1[0],ymm4[0],ymm1[2],ymm4[2]+ vpunpckhqdq %ymm4, %ymm1, %ymm1 # ymm1 = ymm1[1],ymm4[1],ymm1[3],ymm4[3]+ vperm2i128 $0x20, %ymm3, %ymm2, %ymm7 # ymm7 = ymm2[0,1],ymm3[0,1]+ vperm2i128 $0x31, %ymm3, %ymm2, %ymm13 # ymm13 = ymm2[2,3],ymm3[2,3]+ vperm2i128 $0x31, %ymm1, %ymm0, %ymm3 # ymm3 = ymm0[2,3],ymm1[2,3]+ vmovdqu %ymm7, 0x160(%rsp)+ vperm2i128 $0x20, %ymm1, %ymm0, %ymm7 # ymm7 = ymm0[0,1],ymm1[0,1]+ vmovdqu 0xa0(%rdi), %ymm0+ vmovdqu 0x230(%rdi), %ymm1+ vmovdqu %ymm3, 0x1a0(%rsp)+ vmovdqu 0x168(%rdi), %ymm3+ vpunpcklqdq %ymm5, %ymm1, %ymm4 # ymm4 = ymm1[0],ymm5[0],ymm1[2],ymm5[2]+ vpunpckhqdq %ymm5, %ymm1, %ymm1 # ymm1 = ymm1[1],ymm5[1],ymm1[3],ymm5[3]+ vmovdqu %ymm7, 0x180(%rsp)+ vpunpcklqdq %ymm3, %ymm0, %ymm2 # ymm2 = ymm0[0],ymm3[0],ymm0[2],ymm3[2]+ vpunpckhqdq %ymm3, %ymm0, %ymm0 # ymm0 = ymm0[1],ymm3[1],ymm0[3],ymm3[3]+ vperm2i128 $0x20, %ymm4, %ymm2, %ymm12 # ymm12 = ymm2[0,1],ymm4[0,1]+ vperm2i128 $0x20, %ymm1, %ymm0, %ymm3 # ymm3 = ymm0[0,1],ymm1[0,1]+ vperm2i128 $0x31, %ymm4, %ymm2, %ymm7 # ymm7 = ymm2[2,3],ymm4[2,3]+ vperm2i128 $0x31, %ymm1, %ymm0, %ymm4 # ymm4 = ymm0[2,3],ymm1[2,3]+ vmovq 0x250(%rdi), %xmm0+ vmovq 0xc0(%rdi), %xmm1+ vmovdqu %ymm12, 0x1c0(%rsp)+ vmovdqu %ymm4, 0x1e0(%rsp)+ vpinsrq $0x1, 0x318(%rdi), %xmm0, %xmm0+ vpinsrq $0x1, 0x188(%rdi), %xmm1, %xmm1+ vinserti128 $0x1, %xmm0, %ymm1, %ymm2+ movq $0x0, %r10++LLmlk_keccak_f1600_x4_avx2:+ vmovdqu 0xa0(%rsp), %ymm4+ vpxor 0x1c0(%rsp), %ymm9, %ymm0+ vmovdqu %ymm9, 0x200(%rsp)+ vmovdqu %ymm10, %ymm9+ vmovdqu 0xc0(%rsp), %ymm11+ vmovdqu 0x160(%rsp), %ymm12+ vmovdqu %ymm3, 0x240(%rsp)+ vpxor 0x100(%rsp), %ymm4, %ymm1+ vmovdqu 0x40(%rsp), %ymm10+ vmovdqu %ymm4, 0x220(%rsp)+ vpxor %ymm3, %ymm12, %ymm12+ vmovdqu 0x20(%rsp), %ymm6+ vmovdqu 0x140(%rsp), %ymm4+ vmovdqu %ymm14, 0x2a0(%rsp)+ vpxor %ymm1, %ymm0, %ymm0+ vpxor %ymm8, %ymm11, %ymm1+ vpxor 0x180(%rsp), %ymm7, %ymm11+ vmovdqu %ymm10, 0x280(%rsp)+ vpxor %ymm1, %ymm12, %ymm12+ vpxor %ymm15, %ymm9, %ymm1+ vmovdqu 0xe0(%rsp), %ymm3+ vmovdqu %ymm8, 0x260(%rsp)+ vpxor %ymm1, %ymm11, %ymm11+ vpxor 0x120(%rsp), %ymm14, %ymm1+ vpxor %ymm6, %ymm12, %ymm12+ vmovdqu 0x60(%rsp), %ymm8+ vpxor %ymm10, %ymm11, %ymm11+ vpxor 0x1e0(%rsp), %ymm13, %ymm10+ vpxor %ymm4, %ymm3, %ymm3+ vmovdqu %ymm4, 0x2c0(%rsp)+ vpsrlq $0x3f, %ymm12, %ymm4+ vpsrlq $0x3f, %ymm11, %ymm5+ vpxor (%rsp), %ymm0, %ymm0+ vpxor %ymm1, %ymm10, %ymm10+ vmovdqu 0x80(%rsp), %ymm1+ vpxor %ymm8, %ymm10, %ymm10+ vmovdqu %ymm1, %ymm14+ vpxor 0x1a0(%rsp), %ymm2, %ymm1+ vmovdqu %ymm14, 0x2e0(%rsp)+ vpxor %ymm3, %ymm1, %ymm1+ vpsllq $0x1, %ymm12, %ymm3+ vpor %ymm4, %ymm3, %ymm3+ vpsllq $0x1, %ymm11, %ymm4+ vpxor %ymm14, %ymm1, %ymm1+ vpor %ymm5, %ymm4, %ymm4+ vpsrlq $0x3f, %ymm10, %ymm14+ vpxor %ymm1, %ymm3, %ymm3+ vpsllq $0x1, %ymm10, %ymm5+ vpxor %ymm0, %ymm4, %ymm4+ vpor %ymm14, %ymm5, %ymm5+ vpxor %ymm6, %ymm4, %ymm6+ vpxor %ymm12, %ymm5, %ymm5+ vpsrlq $0x3f, %ymm1, %ymm12+ vpsllq $0x1, %ymm1, %ymm1+ vpxor %ymm7, %ymm5, %ymm7+ vpxor %ymm9, %ymm5, %ymm9+ vpor %ymm12, %ymm1, %ymm1+ vpxor (%rsp), %ymm3, %ymm12+ vpxor %ymm11, %ymm1, %ymm1+ vpsrlq $0x3f, %ymm0, %ymm11+ vpsllq $0x1, %ymm0, %ymm0+ vpxor %ymm13, %ymm1, %ymm13+ vpxor %ymm8, %ymm1, %ymm8+ vpor %ymm11, %ymm0, %ymm0+ vpxor %ymm10, %ymm0, %ymm0+ vpxor 0xc0(%rsp), %ymm4, %ymm10+ vpxor %ymm2, %ymm0, %ymm2+ vpsrlq $0x14, %ymm10, %ymm11+ vpsllq $0x2c, %ymm10, %ymm10+ vpor %ymm11, %ymm10, %ymm10+ vpxor %ymm15, %ymm5, %ymm11+ vpbroadcastq (%rsi), %ymm15+ vpsrlq $0x15, %ymm11, %ymm14+ vpsllq $0x2b, %ymm11, %ymm11+ vpor %ymm14, %ymm11, %ymm11+ vpandn %ymm11, %ymm10, %ymm14+ vpxor %ymm15, %ymm14, %ymm14+ vpxor %ymm12, %ymm14, %ymm15+ vpsrlq $0x2b, %ymm13, %ymm14+ vpsllq $0x15, %ymm13, %ymm13+ vmovdqu %ymm15, (%rsp)+ vpor %ymm14, %ymm13, %ymm13+ vpandn %ymm13, %ymm11, %ymm14+ vpxor %ymm10, %ymm14, %ymm15+ vpsrlq $0x32, %ymm2, %ymm14+ vpsllq $0xe, %ymm2, %ymm2+ vmovdqu %ymm15, 0x20(%rsp)+ vpor %ymm14, %ymm2, %ymm2+ vpandn %ymm2, %ymm13, %ymm14+ vpxor %ymm11, %ymm14, %ymm11+ vmovdqu %ymm11, 0x40(%rsp)+ vpandn %ymm12, %ymm2, %ymm11+ vpandn %ymm10, %ymm12, %ymm12+ vpxor %ymm13, %ymm11, %ymm11+ vmovdqu %ymm11, 0x60(%rsp)+ vpxor %ymm2, %ymm12, %ymm11+ vpsrlq $0x24, %ymm8, %ymm2+ vpsllq $0x1c, %ymm8, %ymm8+ vmovdqu %ymm11, 0x80(%rsp)+ vpor %ymm2, %ymm8, %ymm8+ vpxor 0xe0(%rsp), %ymm0, %ymm2+ vpsrlq $0x2c, %ymm2, %ymm10+ vpsllq $0x14, %ymm2, %ymm2+ vpor %ymm10, %ymm2, %ymm2+ vpxor 0x100(%rsp), %ymm3, %ymm10+ vpsrlq $0x3d, %ymm10, %ymm11+ vpsllq $0x3, %ymm10, %ymm10+ vpor %ymm11, %ymm10, %ymm10+ vpandn %ymm10, %ymm2, %ymm11+ vpxor %ymm8, %ymm11, %ymm11+ vmovdqu %ymm11, 0xa0(%rsp)+ vpxor 0x160(%rsp), %ymm4, %ymm11+ vpsrlq $0x13, %ymm11, %ymm12+ vpsllq $0x2d, %ymm11, %ymm11+ vpor %ymm12, %ymm11, %ymm11+ vpandn %ymm11, %ymm10, %ymm12+ vpxor %ymm2, %ymm12, %ymm12+ vmovdqu %ymm12, 0xc0(%rsp)+ vpsrlq $0x3, %ymm7, %ymm12+ vpsllq $0x3d, %ymm7, %ymm7+ vpor %ymm12, %ymm7, %ymm7+ vpandn %ymm7, %ymm11, %ymm12+ vpxor %ymm10, %ymm12, %ymm10+ vpandn %ymm8, %ymm7, %ymm12+ vpandn %ymm2, %ymm8, %ymm8+ vpsrlq $0x3f, %ymm6, %ymm2+ vpsllq $0x1, %ymm6, %ymm6+ vpxor %ymm11, %ymm12, %ymm14+ vpor %ymm2, %ymm6, %ymm6+ vpsrlq $0x3a, %ymm9, %ymm2+ vpxor %ymm7, %ymm8, %ymm12+ vpsllq $0x6, %ymm9, %ymm9+ vmovdqu %ymm12, 0xe0(%rsp)+ vpxor 0x1a0(%rsp), %ymm0, %ymm7+ vpor %ymm2, %ymm9, %ymm9+ vpxor 0x120(%rsp), %ymm1, %ymm2+ vpshufb (%rdx), %ymm7, %ymm7+ vpsrlq $0x27, %ymm2, %ymm11+ vpsllq $0x19, %ymm2, %ymm2+ vpor %ymm2, %ymm11, %ymm11+ vpandn %ymm11, %ymm9, %ymm2+ vpandn %ymm7, %ymm11, %ymm8+ vpxor %ymm6, %ymm2, %ymm12+ vpxor 0x1c0(%rsp), %ymm3, %ymm2+ vpxor %ymm9, %ymm8, %ymm8+ vmovdqu %ymm12, 0x100(%rsp)+ vpsrlq $0x2e, %ymm2, %ymm12+ vpsllq $0x12, %ymm2, %ymm2+ vpor %ymm2, %ymm12, %ymm2+ vpandn %ymm2, %ymm7, %ymm12+ vpxor %ymm11, %ymm12, %ymm15+ vpandn %ymm6, %ymm2, %ymm11+ vpandn %ymm9, %ymm6, %ymm6+ vpxor %ymm7, %ymm11, %ymm12+ vmovdqu %ymm12, 0x120(%rsp)+ vpxor %ymm2, %ymm6, %ymm12+ vpxor 0x2e0(%rsp), %ymm0, %ymm6+ vpxor 0x2c0(%rsp), %ymm0, %ymm0+ vmovdqu %ymm12, 0x140(%rsp)+ vpsrlq $0x25, %ymm6, %ymm2+ vpsllq $0x1b, %ymm6, %ymm6+ vpor %ymm6, %ymm2, %ymm2+ vpxor 0x220(%rsp), %ymm3, %ymm6+ vpxor 0x200(%rsp), %ymm3, %ymm3+ vpsrlq $0x1c, %ymm6, %ymm7+ vpsllq $0x24, %ymm6, %ymm6+ vpor %ymm6, %ymm7, %ymm7+ vpxor 0x260(%rsp), %ymm4, %ymm6+ vpxor 0x240(%rsp), %ymm4, %ymm4+ vpsrlq $0x36, %ymm6, %ymm12+ vpsllq $0xa, %ymm6, %ymm6+ vpor %ymm6, %ymm12, %ymm12+ vpxor 0x180(%rsp), %ymm5, %ymm6+ vpxor 0x280(%rsp), %ymm5, %ymm5+ vpandn %ymm12, %ymm7, %ymm9+ vpsrlq $0x31, %ymm6, %ymm11+ vpsllq $0xf, %ymm6, %ymm6+ vpxor %ymm2, %ymm9, %ymm9+ vpor %ymm6, %ymm11, %ymm11+ vpandn %ymm11, %ymm12, %ymm6+ vpxor %ymm7, %ymm6, %ymm6+ vmovdqu %ymm6, 0x160(%rsp)+ vpxor 0x1e0(%rsp), %ymm1, %ymm6+ vpxor 0x2a0(%rsp), %ymm1, %ymm1+ vpshufb (%rcx), %ymm6, %ymm6+ vpandn %ymm6, %ymm11, %ymm13+ vpxor %ymm12, %ymm13, %ymm13+ vmovdqu %ymm13, 0x180(%rsp)+ vpandn %ymm2, %ymm6, %ymm13+ vpandn %ymm7, %ymm2, %ymm2+ vpxor %ymm6, %ymm2, %ymm2+ vpsrlq $0x3e, %ymm4, %ymm6+ vpxor %ymm11, %ymm13, %ymm13+ vmovdqu %ymm2, 0x1a0(%rsp)+ vpsrlq $0x2, %ymm5, %ymm2+ vpsllq $0x3e, %ymm5, %ymm5+ vpor %ymm5, %ymm2, %ymm2+ vpsrlq $0x9, %ymm1, %ymm5+ vpsllq $0x37, %ymm1, %ymm1+ vpsllq $0x2, %ymm4, %ymm4+ vpor %ymm1, %ymm5, %ymm1+ vpsrlq $0x19, %ymm0, %ymm5+ vpor %ymm4, %ymm6, %ymm4+ vpsllq $0x27, %ymm0, %ymm0+ vpor %ymm0, %ymm5, %ymm5+ vpandn %ymm5, %ymm1, %ymm0+ vpxor %ymm2, %ymm0, %ymm0+ vmovdqu %ymm0, 0x1c0(%rsp)+ vpsrlq $0x17, %ymm3, %ymm0+ vpsllq $0x29, %ymm3, %ymm3+ vpor %ymm3, %ymm0, %ymm0+ vpandn %ymm4, %ymm0, %ymm7+ vpandn %ymm0, %ymm5, %ymm3+ vpxor %ymm5, %ymm7, %ymm7+ vpandn %ymm2, %ymm4, %ymm5+ vpandn %ymm1, %ymm2, %ymm2+ vpxor %ymm0, %ymm5, %ymm5+ vpxor %ymm1, %ymm3, %ymm3+ vpxor %ymm4, %ymm2, %ymm2+ vmovdqu %ymm5, 0x1e0(%rsp)+ addq $0x8, %rsi+ addq $0x1, %r10+ cmpq $0x18, %r10+ jne LLmlk_keccak_f1600_x4_avx2+ vmovdqu (%rsp), %ymm4+ vmovdqu 0x40(%rsp), %ymm5+ vmovdqu 0x20(%rsp), %ymm0+ vmovdqu 0x60(%rsp), %ymm1+ vmovdqu 0x1c0(%rsp), %ymm12+ vmovdqu %ymm2, 0x1c0(%rsp)+ vpunpcklqdq %ymm0, %ymm4, %ymm2 # ymm2 = ymm4[0],ymm0[0],ymm4[2],ymm0[2]+ vpunpckhqdq %ymm0, %ymm4, %ymm0 # ymm0 = ymm4[1],ymm0[1],ymm4[3],ymm0[3]+ vpunpcklqdq %ymm1, %ymm5, %ymm4 # ymm4 = ymm5[0],ymm1[0],ymm5[2],ymm1[2]+ vpunpckhqdq %ymm1, %ymm5, %ymm1 # ymm1 = ymm5[1],ymm1[1],ymm5[3],ymm1[3]+ vperm2i128 $0x20, %ymm4, %ymm2, %ymm6 # ymm6 = ymm2[0,1],ymm4[0,1]+ vperm2i128 $0x31, %ymm4, %ymm2, %ymm2 # ymm2 = ymm2[2,3],ymm4[2,3]+ vmovdqu 0x80(%rsp), %ymm4+ vperm2i128 $0x20, %ymm1, %ymm0, %ymm5 # ymm5 = ymm0[0,1],ymm1[0,1]+ vperm2i128 $0x31, %ymm1, %ymm0, %ymm0 # ymm0 = ymm0[2,3],ymm1[2,3]+ vmovdqu %ymm6, (%rdi)+ vmovdqu %ymm5, 0xc8(%rdi)+ vmovdqu %ymm2, 0x190(%rdi)+ vmovdqu %ymm0, 0x258(%rdi)+ vmovdqu 0xa0(%rsp), %ymm0+ vpunpcklqdq %ymm0, %ymm4, %ymm2 # ymm2 = ymm4[0],ymm0[0],ymm4[2],ymm0[2]+ vpunpckhqdq %ymm0, %ymm4, %ymm1 # ymm1 = ymm4[1],ymm0[1],ymm4[3],ymm0[3]+ vmovdqu 0xc0(%rsp), %ymm0+ vpunpcklqdq %ymm10, %ymm0, %ymm4 # ymm4 = ymm0[0],ymm10[0],ymm0[2],ymm10[2]+ vpunpckhqdq %ymm10, %ymm0, %ymm0 # ymm0 = ymm0[1],ymm10[1],ymm0[3],ymm10[3]+ vperm2i128 $0x20, %ymm4, %ymm2, %ymm6 # ymm6 = ymm2[0,1],ymm4[0,1]+ vperm2i128 $0x20, %ymm0, %ymm1, %ymm5 # ymm5 = ymm1[0,1],ymm0[0,1]+ vperm2i128 $0x31, %ymm4, %ymm2, %ymm2 # ymm2 = ymm2[2,3],ymm4[2,3]+ vmovdqu 0xe0(%rsp), %ymm4+ vperm2i128 $0x31, %ymm0, %ymm1, %ymm1 # ymm1 = ymm1[2,3],ymm0[2,3]+ vmovdqu 0x100(%rsp), %ymm0+ vmovdqu %ymm2, 0x1b0(%rdi)+ vmovdqu %ymm1, 0x278(%rdi)+ vpunpcklqdq %ymm4, %ymm14, %ymm2 # ymm2 = ymm14[0],ymm4[0],ymm14[2],ymm4[2]+ vpunpckhqdq %ymm4, %ymm14, %ymm1 # ymm1 = ymm14[1],ymm4[1],ymm14[3],ymm4[3]+ vpunpcklqdq %ymm8, %ymm0, %ymm4 # ymm4 = ymm0[0],ymm8[0],ymm0[2],ymm8[2]+ vpunpckhqdq %ymm8, %ymm0, %ymm0 # ymm0 = ymm0[1],ymm8[1],ymm0[3],ymm8[3]+ vmovdqu %ymm6, 0x20(%rdi)+ vmovdqu %ymm5, 0xe8(%rdi)+ vperm2i128 $0x20, %ymm4, %ymm2, %ymm6 # ymm6 = ymm2[0,1],ymm4[0,1]+ vperm2i128 $0x20, %ymm0, %ymm1, %ymm5 # ymm5 = ymm1[0,1],ymm0[0,1]+ vperm2i128 $0x31, %ymm4, %ymm2, %ymm2 # ymm2 = ymm2[2,3],ymm4[2,3]+ vperm2i128 $0x31, %ymm0, %ymm1, %ymm1 # ymm1 = ymm1[2,3],ymm0[2,3]+ vmovdqu 0x120(%rsp), %ymm4+ vmovdqu 0x140(%rsp), %ymm0+ vmovdqu %ymm2, 0x1d0(%rdi)+ vmovdqu %ymm1, 0x298(%rdi)+ vpunpcklqdq %ymm4, %ymm15, %ymm2 # ymm2 = ymm15[0],ymm4[0],ymm15[2],ymm4[2]+ vpunpckhqdq %ymm4, %ymm15, %ymm1 # ymm1 = ymm15[1],ymm4[1],ymm15[3],ymm4[3]+ vpunpcklqdq %ymm9, %ymm0, %ymm4 # ymm4 = ymm0[0],ymm9[0],ymm0[2],ymm9[2]+ vmovdqu %ymm5, 0x108(%rdi)+ vpunpckhqdq %ymm9, %ymm0, %ymm0 # ymm0 = ymm0[1],ymm9[1],ymm0[3],ymm9[3]+ vmovdqu %ymm6, 0x40(%rdi)+ vperm2i128 $0x20, %ymm4, %ymm2, %ymm6 # ymm6 = ymm2[0,1],ymm4[0,1]+ vperm2i128 $0x31, %ymm4, %ymm2, %ymm2 # ymm2 = ymm2[2,3],ymm4[2,3]+ vperm2i128 $0x20, %ymm0, %ymm1, %ymm5 # ymm5 = ymm1[0,1],ymm0[0,1]+ vmovdqu 0x160(%rsp), %ymm4+ vperm2i128 $0x31, %ymm0, %ymm1, %ymm1 # ymm1 = ymm1[2,3],ymm0[2,3]+ vmovdqu 0x180(%rsp), %ymm0+ vmovdqu %ymm5, 0x128(%rdi)+ vmovdqu 0x1a0(%rsp), %ymm5+ vmovdqu %ymm2, 0x1f0(%rdi)+ vpunpcklqdq %ymm0, %ymm4, %ymm2 # ymm2 = ymm4[0],ymm0[0],ymm4[2],ymm0[2]+ vpunpckhqdq %ymm0, %ymm4, %ymm0 # ymm0 = ymm4[1],ymm0[1],ymm4[3],ymm0[3]+ vpunpcklqdq %ymm5, %ymm13, %ymm4 # ymm4 = ymm13[0],ymm5[0],ymm13[2],ymm5[2]+ vmovdqu %ymm6, 0x60(%rdi)+ vperm2i128 $0x20, %ymm4, %ymm2, %ymm6 # ymm6 = ymm2[0,1],ymm4[0,1]+ vmovdqu %ymm1, 0x2b8(%rdi)+ vperm2i128 $0x31, %ymm4, %ymm2, %ymm2 # ymm2 = ymm2[2,3],ymm4[2,3]+ vpunpckhqdq %ymm5, %ymm13, %ymm1 # ymm1 = ymm13[1],ymm5[1],ymm13[3],ymm5[3]+ vmovdqu %ymm6, 0x80(%rdi)+ vmovdqu 0x1e0(%rsp), %ymm4+ vperm2i128 $0x20, %ymm1, %ymm0, %ymm5 # ymm5 = ymm0[0,1],ymm1[0,1]+ vperm2i128 $0x31, %ymm1, %ymm0, %ymm0 # ymm0 = ymm0[2,3],ymm1[2,3]+ vmovdqu %ymm2, 0x210(%rdi)+ vpunpcklqdq %ymm3, %ymm12, %ymm2 # ymm2 = ymm12[0],ymm3[0],ymm12[2],ymm3[2]+ vmovdqu %ymm0, 0x2d8(%rdi)+ vpunpckhqdq %ymm3, %ymm12, %ymm0 # ymm0 = ymm12[1],ymm3[1],ymm12[3],ymm3[3]+ vpunpcklqdq %ymm4, %ymm7, %ymm3 # ymm3 = ymm7[0],ymm4[0],ymm7[2],ymm4[2]+ vpunpckhqdq %ymm4, %ymm7, %ymm1 # ymm1 = ymm7[1],ymm4[1],ymm7[3],ymm4[3]+ vmovdqu %ymm5, 0x148(%rdi)+ vperm2i128 $0x20, %ymm3, %ymm2, %ymm5 # ymm5 = ymm2[0,1],ymm3[0,1]+ vperm2i128 $0x31, %ymm3, %ymm2, %ymm2 # ymm2 = ymm2[2,3],ymm3[2,3]+ vmovdqu 0x1c0(%rsp), %ymm3+ vperm2i128 $0x20, %ymm1, %ymm0, %ymm4 # ymm4 = ymm0[0,1],ymm1[0,1]+ vperm2i128 $0x31, %ymm1, %ymm0, %ymm0 # ymm0 = ymm0[2,3],ymm1[2,3]+ vmovdqu %ymm5, 0xa0(%rdi)+ vextracti128 $0x1, %ymm3, %xmm15+ vmovdqu %ymm4, 0x168(%rdi)+ vmovdqu %ymm2, 0x230(%rdi)+ vmovdqu %ymm0, 0x2f8(%rdi)+ vmovq %xmm3, 0xc0(%rdi)+ vmovhpd %xmm3, 0x188(%rdi)+ vmovq %xmm15, 0x250(%rdi)+ vmovhpd %xmm15, 0x318(%rdi)+ movq %r11, %rsp+ .cfi_def_cfa_register %rsp+ retq+ .cfi_endproc++MLK_ASM_FN_SIZE(keccak_f1600_x4_avx2_asm)++#endif /* MLK_FIPS202_X86_64_NEED_X4_AVX2 && !MLK_CONFIG_MULTILEVEL_NO_SHARED */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
@@ -0,0 +1,52 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*+ * WARNING: This file is auto-generated from scripts/autogen+ * in the mlkem-native repository.+ * Do not modify it directly.+ */++#include "../../../../common.h"+#if defined(MLK_FIPS202_X86_64_NEED_X4_AVX2) && \+ !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED)++#include <stdint.h>++#include "fips202_native_x86_64.h"++MLK_ALIGN MLK_INTERNAL_DATA_DEFINITION const uint64_t+ mlk_keccakf1600_round_constants[24] = {+ 0x0000000000000001, 0x0000000000008082, 0x800000000000808a,+ 0x8000000080008000, 0x000000000000808b, 0x0000000080000001,+ 0x8000000080008081, 0x8000000000008009, 0x000000000000008a,+ 0x0000000000000088, 0x0000000080008009, 0x000000008000000a,+ 0x000000008000808b, 0x800000000000008b, 0x8000000000008089,+ 0x8000000000008003, 0x8000000000008002, 0x8000000000000080,+ 0x000000000000800a, 0x800000008000000a, 0x8000000080008081,+ 0x8000000000008080, 0x0000000080000001, 0x8000000080008008,+};++MLK_ALIGN MLK_INTERNAL_DATA_DEFINITION const uint64_t mlk_keccak_rho8[4] = {+ 0x0605040302010007,+ 0x0e0d0c0b0a09080f,+ 0x1615141312111017,+ 0x1e1d1c1b1a19181f,+};++MLK_ALIGN MLK_INTERNAL_DATA_DEFINITION const uint64_t mlk_keccak_rho56[4] = {+ 0x0007060504030201,+ 0x080f0e0d0c0b0a09,+ 0x1017161514131211,+ 0x181f1e1d1c1b1a19,+};++#else /* MLK_FIPS202_X86_64_NEED_X4_AVX2 && !MLK_CONFIG_MULTILEVEL_NO_SHARED \+ */++MLK_EMPTY_CU(fips202_x86_64_constants)++#endif /* !(MLK_FIPS202_X86_64_NEED_X4_AVX2 && \+ !MLK_CONFIG_MULTILEVEL_NO_SHARED) */
@@ -0,0 +1,675 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS203]+ * FIPS 203 Module-Lattice-Based Key-Encapsulation Mechanism Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/203/final+ *+ * - [REF]+ * CRYSTALS-Kyber C reference implementation+ * Bos, Ducas, Kiltz, Lepoint, Lyubashevsky, Schanck, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/kyber/tree/main/ref+ */++#include "indcpa.h"++#include "debug.h"+#include "randombytes.h"+#include "sampling.h"+#include "symmetric.h"+#include "verify.h"++/* Parameter set namespacing+ * This is to facilitate building multiple instances+ * of mlkem-native (e.g. with varying parameter sets)+ * within a single compilation unit. */+#define mlk_pack_pk MLK_ADD_PARAM_SET(mlk_pack_pk)+#define mlk_unpack_pk MLK_ADD_PARAM_SET(mlk_unpack_pk)+#define mlk_pack_sk MLK_ADD_PARAM_SET(mlk_pack_sk)+#define mlk_unpack_sk MLK_ADD_PARAM_SET(mlk_unpack_sk)+#define mlk_pack_ciphertext MLK_ADD_PARAM_SET(mlk_pack_ciphertext)+#define mlk_unpack_ciphertext MLK_ADD_PARAM_SET(mlk_unpack_ciphertext)+#define mlk_matvec_mul MLK_ADD_PARAM_SET(mlk_matvec_mul)+#define mlk_polyvec_permute_bitrev_to_custom \+ MLK_ADD_PARAM_SET(mlk_polyvec_permute_bitrev_to_custom)+#define mlk_polymat_permute_bitrev_to_custom \+ MLK_ADD_PARAM_SET(mlk_polymat_permute_bitrev_to_custom)+#define mlk_keypair_getnoise_eta1 MLK_ADD_PARAM_SET(mlk_keypair_getnoise_eta1)+#define mlk_enc_getnoise_eta1_eta2 MLK_ADD_PARAM_SET(mlk_enc_getnoise_eta1_eta2)+/* End of parameter set namespacing */++/**+ * Serialize the public key as the concatenation of the serialized vector of+ * polynomials pk and the public seed used to generate the matrix A.+ *+ * @spec{Implements @[FIPS203, Algorithm 13 (K-PKE.KeyGen), L19].}+ *+ * @param[out] r Output serialized public key.+ * @param[in] pk Input public-key polyvec. Must have coefficients within+ * [0,..,MLKEM_Q-1].+ * @param[in] seed Input public seed.+ */+#if !defined(MLK_CONFIG_NO_KEYPAIR_API)+static void mlk_pack_pk(uint8_t r[MLKEM_INDCPA_PUBLICKEYBYTES],+ const mlk_polyvec *pk,+ const uint8_t seed[MLKEM_SYMBYTES])+{+ mlk_assert_bound_2d(pk->vec, MLKEM_K, MLKEM_N, 0, MLKEM_Q);+ mlk_polyvec_tobytes(r, pk);+ mlk_memcpy(r + MLKEM_POLYVECBYTES, seed, MLKEM_SYMBYTES);+}+#endif /* !MLK_CONFIG_NO_KEYPAIR_API */++/**+ * De-serialize public key from a byte array; approximate inverse of+ * mlk_pack_pk.+ *+ * @spec{Implements @[FIPS203, Algorithm 14 (K-PKE.Encrypt), L2-3].}+ *+ * @param[out] pk Output public-key polynomial vector. Coefficients+ * will be normalized to [0,1,..,MLKEM_Q-1].+ * @param[out] seed Output seed to generate matrix A.+ * @param[in] packedpk Input serialized public key.+ */+#if !defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API)+static void mlk_unpack_pk(mlk_polyvec *pk, uint8_t seed[MLKEM_SYMBYTES],+ const uint8_t packedpk[MLKEM_INDCPA_PUBLICKEYBYTES])+{+ mlk_polyvec_frombytes(pk, packedpk);+ mlk_memcpy(seed, packedpk + MLKEM_POLYVECBYTES, MLKEM_SYMBYTES);++ /* NOTE: If a modulus check was conducted on the PK, we know at this+ * point that the coefficients of `pk` are unsigned canonical. The+ * specifications and proofs, however, do _not_ assume this, and instead+ * work with the easily provable bound by MLKEM_UINT12_LIMIT. */+}+#endif /* !MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API */++/**+ * Serialize the secret key.+ *+ * @spec{Implements @[FIPS203, Algorithm 13 (K-PKE.KeyGen), L20].}+ *+ * @param[out] r Output serialized secret key.+ * @param[in] sk Input vector of polynomials (secret key).+ */+#if !defined(MLK_CONFIG_NO_KEYPAIR_API)+static void mlk_pack_sk(uint8_t r[MLKEM_INDCPA_SECRETKEYBYTES],+ const mlk_polyvec *sk)+{+ mlk_assert_bound_2d(sk->vec, MLKEM_K, MLKEM_N, 0, MLKEM_Q);+ mlk_polyvec_tobytes(r, sk);+}+#endif /* !MLK_CONFIG_NO_KEYPAIR_API */++/**+ * De-serialize the secret key; inverse of mlk_pack_sk.+ *+ * @spec{Implements @[FIPS203, Algorithm 15 (K-PKE.Decrypt), L5].}+ *+ * @param[out] sk Output vector of polynomials (secret key).+ * @param[in] packedsk Input serialized secret key.+ */+#if !defined(MLK_CONFIG_NO_DECAPS_API)+static void mlk_unpack_sk(mlk_polyvec *sk,+ const uint8_t packedsk[MLKEM_INDCPA_SECRETKEYBYTES])+{+ mlk_polyvec_frombytes(sk, packedsk);+}+#endif /* !MLK_CONFIG_NO_DECAPS_API */++/**+ * Serialize the ciphertext as the concatenation of the compressed and+ * serialized vector of polynomials b and the compressed and serialized+ * polynomial v.+ *+ * @spec{Implements @[FIPS203, Algorithm 14 (K-PKE.Encrypt), L22-23].}+ *+ * @param[out] r Output serialized ciphertext.+ * @param[in] b Input vector of polynomials b.+ * @param[in] v Input polynomial v.+ */+#if !defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API)+static void mlk_pack_ciphertext(uint8_t r[MLKEM_INDCPA_BYTES],+ const mlk_polyvec *b, mlk_poly *v)+{+ mlk_polyvec_compress_du(r, b);+ mlk_poly_compress_dv(r + MLKEM_POLYVECCOMPRESSEDBYTES_DU, v);+}+#endif /* !MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API */++/**+ * De-serialize and decompress ciphertext from a byte array; approximate+ * inverse of mlk_pack_ciphertext.+ *+ * @spec{Implements @[FIPS203, Algorithm 15 (K-PKE.Decrypt), L1-4].}+ *+ * @param[out] b Output vector of polynomials b.+ * @param[out] v Output polynomial v.+ * @param[in] c Input serialized ciphertext.+ */+#if !defined(MLK_CONFIG_NO_DECAPS_API)+static void mlk_unpack_ciphertext(mlk_polyvec *b, mlk_poly *v,+ const uint8_t c[MLKEM_INDCPA_BYTES])+{+ mlk_polyvec_decompress_du(b, c);+ mlk_poly_decompress_dv(v, c + MLKEM_POLYVECCOMPRESSEDBYTES_DU);+}+#endif /* !MLK_CONFIG_NO_DECAPS_API */++/* Helper function to ensure that the polynomial entries in the output+ * of gen_matrix use the standard (bitreversed) ordering of coefficients.+ * No-op unless a native backend with a custom ordering is used.+ *+ * We don't inline this into gen_matrix to avoid having to split the CBMC+ * proof for gen_matrix based on MLK_USE_NATIVE_NTT_CUSTOM_ORDER. */+static void mlk_polyvec_permute_bitrev_to_custom(mlk_polyvec *v)+__contract__(+ /* We don't specify that this should be a permutation, but only+ * that it does not change the bound established at the end of mlk_gen_matrix. */+ requires(memory_no_alias(v, sizeof(mlk_polyvec)))+ requires(forall(x, 0, MLKEM_K,+ array_bound(v->vec[x].coeffs, 0, MLKEM_N, 0, MLKEM_Q)))+ assigns(memory_slice(v, sizeof(mlk_polyvec)))+ ensures(forall(x, 0, MLKEM_K,+ array_bound(v->vec[x].coeffs, 0, MLKEM_N, 0, MLKEM_Q))))+{+#if defined(MLK_USE_NATIVE_NTT_CUSTOM_ORDER)+ unsigned i;+ for (i = 0; i < MLKEM_K; i++)+ __loop__(+ assigns(i, memory_slice(v, sizeof(mlk_polyvec)))+ invariant(i <= MLKEM_K)+ invariant(forall(x, 0, MLKEM_K,+ array_bound(v->vec[x].coeffs, 0, MLKEM_N, 0, MLKEM_Q)))+ decreases(MLKEM_K - i))+ {+ mlk_poly_permute_bitrev_to_custom(v->vec[i].coeffs);+ }+#else /* MLK_USE_NATIVE_NTT_CUSTOM_ORDER */+ /* Nothing to do */+ (void)v;+#endif /* !MLK_USE_NATIVE_NTT_CUSTOM_ORDER */+}++static void mlk_polymat_permute_bitrev_to_custom(mlk_polymat *a)+__contract__(+ /* We don't specify that this should be a permutation, but only+ * that it does not change the bound established at the end of mlk_gen_matrix. */+ requires(memory_no_alias(a, sizeof(mlk_polymat)))+ requires(forall(x, 0, MLKEM_K, forall(y, 0, MLKEM_K,+ array_bound(a->vec[x].vec[y].coeffs, 0, MLKEM_N, 0, MLKEM_Q))))+ assigns(memory_slice(a, sizeof(mlk_polymat)))+ ensures(forall(x, 0, MLKEM_K, forall(y, 0, MLKEM_K,+ array_bound(a->vec[x].vec[y].coeffs, 0, MLKEM_N, 0, MLKEM_Q)))))+{+ unsigned i;+ for (i = 0; i < MLKEM_K; i++)+ __loop__(+ assigns(i, memory_slice(a, sizeof(mlk_polymat)))+ invariant(i <= MLKEM_K)+ invariant(forall(x, 0, MLKEM_K, forall(y, 0, MLKEM_K,+ array_bound(a->vec[x].vec[y].coeffs, 0, MLKEM_N, 0, MLKEM_Q))))+ decreases(MLKEM_K - i))+ {+ mlk_polyvec_permute_bitrev_to_custom(&a->vec[i]);+ }+}++/* Reference: `gen_matrix()` in the reference implementation @[REF].+ * - We use a special subroutine to generate 4 polynomials+ * at a time, to be able to leverage batched Keccak-f1600+ * implementations. The reference implementation generates+ * one matrix entry a time.+ *+ * Not static for benchmarking */+MLK_INTERNAL_API+void mlk_gen_matrix(mlk_polymat *a, const uint8_t seed[MLKEM_SYMBYTES],+ int transposed)+{+ unsigned i, j;+ MLK_ALIGN uint8_t seed_ext[4][MLK_ALIGN_UP(MLKEM_SYMBYTES + 2)];++ for (j = 0; j < 4; j++)+ {+ mlk_memcpy(seed_ext[j], seed, MLKEM_SYMBYTES);+ }++#if !defined(MLK_CONFIG_SERIAL_FIPS202_ONLY)+ /* Sample 4 matrix entries a time. */+ for (i = 0; i < (MLKEM_K * MLKEM_K / 4) * 4; i += 4)+ {+ for (j = 0; j < 4; j++)+ {+ uint8_t x, y;+ /* MLKEM_K <= 4, so the values fit in uint8_t. */+ x = (uint8_t)((i + j) / MLKEM_K);+ y = (uint8_t)((i + j) % MLKEM_K);+ if (transposed)+ {+ seed_ext[j][MLKEM_SYMBYTES + 0] = x;+ seed_ext[j][MLKEM_SYMBYTES + 1] = y;+ }+ else+ {+ seed_ext[j][MLKEM_SYMBYTES + 0] = y;+ seed_ext[j][MLKEM_SYMBYTES + 1] = x;+ }+ }++ mlk_poly_rej_uniform_x4(&a->vec[i / MLKEM_K].vec[i % MLKEM_K],+ &a->vec[(i + 1) / MLKEM_K].vec[(i + 1) % MLKEM_K],+ &a->vec[(i + 2) / MLKEM_K].vec[(i + 2) % MLKEM_K],+ &a->vec[(i + 3) / MLKEM_K].vec[(i + 3) % MLKEM_K],+ seed_ext);+ }+#else /* !MLK_CONFIG_SERIAL_FIPS202_ONLY */+ /* When using serial FIPS202, sample all entries individually. */+ i = 0;+#endif /* MLK_CONFIG_SERIAL_FIPS202_ONLY */++ /* For MLKEM_K == 3, sample the last entry individually.+ * When MLK_CONFIG_SERIAL_FIPS202_ONLY is set, sample all entries+ * individually. */+ for (; i < MLKEM_K * MLKEM_K; i++)+ {+ uint8_t x, y;+ /* MLKEM_K <= 4, so the values fit in uint8_t. */+ x = (uint8_t)(i / MLKEM_K);+ y = (uint8_t)(i % MLKEM_K);++ if (transposed)+ {+ seed_ext[0][MLKEM_SYMBYTES + 0] = x;+ seed_ext[0][MLKEM_SYMBYTES + 1] = y;+ }+ else+ {+ seed_ext[0][MLKEM_SYMBYTES + 0] = y;+ seed_ext[0][MLKEM_SYMBYTES + 1] = x;+ }++ mlk_poly_rej_uniform(&a->vec[i / MLKEM_K].vec[i % MLKEM_K], seed_ext[0]);+ }++ mlk_assert(i == MLKEM_K * MLKEM_K);++ /*+ * The public matrix is generated in NTT domain. If the native backend+ * uses a custom order in NTT domain, permute A accordingly.+ */+ mlk_polymat_permute_bitrev_to_custom(a);++ /* Specification: Partially implements+ * @[FIPS203, Section 3.3, Destruction of intermediate values] */+ mlk_zeroize(seed_ext, sizeof(seed_ext));+}++/**+ * Compute matrix-vector product in NTT domain, via Montgomery multiplication.+ *+ * @spec{Implements @[FIPS203, Section 2.4.7, Eq (2.12), (2.13)].}+ *+ * @param[out] out Output polynomial vector.+ * @param[in] a Input matrix. Must be in NTT domain and have coefficients+ * of absolute value < 4096.+ * @param[in] v Input polynomial vector. Must be in NTT domain.+ * @param[in] vc Mulcache for @p v, computed via+ * mlk_polyvec_mulcache_compute().+ */+static void mlk_matvec_mul(mlk_polyvec *out, const mlk_polymat *a,+ const mlk_polyvec *v, const mlk_polyvec_mulcache *vc)+__contract__(+ requires(memory_no_alias(out, sizeof(mlk_polyvec)))+ requires(memory_no_alias(a, sizeof(mlk_polymat)))+ requires(memory_no_alias(v, sizeof(mlk_polyvec)))+ requires(memory_no_alias(vc, sizeof(mlk_polyvec_mulcache)))+ requires(forall(k0, 0, MLKEM_K,+ forall(k1, 0, MLKEM_K,+ array_bound(a->vec[k0].vec[k1].coeffs, 0, MLKEM_N, 0, MLKEM_UINT12_LIMIT))))+ assigns(memory_slice(out, sizeof(mlk_polyvec))))+{+ unsigned i;+ for (i = 0; i < MLKEM_K; i++)+ __loop__(+ assigns(i, memory_slice(out, sizeof(mlk_polyvec)))+ invariant(i <= MLKEM_K)+ decreases(MLKEM_K - i))+ {+ mlk_polyvec_basemul_acc_montgomery_cached(&out->vec[i], &a->vec[i], v, vc);+ }+}++/**+ * Compute and fill the pv and e polyvec structures needed by+ * mlk_keypair_derand(). Uses x4-batched versions of `poly_getnoise` to+ * leverage batched Keccak-f1600.+ *+ * @spec{Implements @[FIPS203, Algorithm 13 (K-PKE.KeyGen)] steps 8-15.}+ *+ * @param[out] pv Output polynomial vector.+ * @param[out] e Output polynomial vector.+ * @param[in] seed Seed bytes for sampling.+ */+#if !defined(MLK_CONFIG_NO_KEYPAIR_API)+static void mlk_keypair_getnoise_eta1(mlk_polyvec *pv, mlk_polyvec *e,+ const uint8_t seed[MLKEM_SYMBYTES])+__contract__(+ requires(memory_no_alias(pv, sizeof(mlk_polyvec)))+ requires(memory_no_alias(e, sizeof(mlk_polyvec)))+ requires(memory_no_alias(seed, MLKEM_SYMBYTES))+ assigns(memory_slice(pv, sizeof(mlk_polyvec)))+ assigns(memory_slice(e, sizeof(mlk_polyvec)))+ ensures(forall(k0, 0, MLKEM_K, array_abs_bound(pv->vec[k0].coeffs, 0, MLKEM_N, MLKEM_ETA1 + 1)))+ ensures(forall(k1, 0, MLKEM_K, array_abs_bound(e->vec[k1].coeffs, 0, MLKEM_N, MLKEM_ETA1 + 1)))+)+{+#if MLKEM_K == 2+ mlk_poly_getnoise_eta1_4x(&pv->vec[0], &pv->vec[1], /* Fill elements of pv */+ &e->vec[0], &e->vec[1], /* and two elements of e */+ seed, 0, 1, 2, 3);+#elif MLKEM_K == 3+ /*+ * Only the first three output buffers are needed, so we pass NULL as+ * the fourth parameter, and 0xFF as its dummy nonce.+ */+ mlk_poly_getnoise_eta1_4x(&pv->vec[0], &pv->vec[1], &pv->vec[2], NULL, seed,+ 0, 1, 2, 0xFF);+ /* Same here */+ mlk_poly_getnoise_eta1_4x(&e->vec[0], &e->vec[1], &e->vec[2], NULL, seed, 3,+ 4, 5, 0xFF);+#elif MLKEM_K == 4+ mlk_poly_getnoise_eta1_4x(&pv->vec[0], &pv->vec[1], &pv->vec[2], &pv->vec[3],+ seed, 0, 1, 2, 3);+ mlk_poly_getnoise_eta1_4x(&e->vec[0], &e->vec[1], &e->vec[2], &e->vec[3],+ seed, 4, 5, 6, 7);+#endif /* MLKEM_K == 4 */+}+#endif /* !MLK_CONFIG_NO_KEYPAIR_API */++/**+ * Compute and fill the sp, ep, and epp polynomial structures needed by+ * mlk_indcpa_enc(). Uses x4-batched versions of `poly_getnoise` to leverage+ * batched Keccak-f1600.+ *+ * @spec{Implements @[FIPS203, Algorithm 14 (K-PKE.Encrypt)] steps 9-16.}+ *+ * @param[out] sp Output polynomial vector.+ * @param[out] ep Output polynomial vector.+ * @param[out] epp Output polynomial.+ * @param[in] coins Seed bytes for sampling.+ */+#if !defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API)+static void mlk_enc_getnoise_eta1_eta2(mlk_polyvec *sp, mlk_polyvec *ep,+ mlk_poly *epp,+ const uint8_t coins[MLKEM_SYMBYTES])+__contract__(+ requires(memory_no_alias(sp, sizeof(mlk_polyvec)))+ requires(memory_no_alias(ep, sizeof(mlk_polyvec)))+ requires(memory_no_alias(epp, sizeof(mlk_poly)))+ requires(memory_no_alias(coins, MLKEM_SYMBYTES))+ assigns(memory_slice(sp, sizeof(mlk_polyvec)))+ assigns(memory_slice(ep, sizeof(mlk_polyvec)))+ assigns(memory_slice(epp, sizeof(mlk_poly)))+ ensures(forall(k0, 0, MLKEM_K, array_abs_bound(sp->vec[k0].coeffs, 0, MLKEM_N, MLKEM_ETA1 + 1)))+ ensures(forall(k1, 0, MLKEM_K, array_abs_bound(ep->vec[k1].coeffs, 0, MLKEM_N, MLKEM_ETA2 + 1)))+ ensures(array_abs_bound(epp->coeffs, 0, MLKEM_N, MLKEM_ETA2 + 1))+)+{+#if MLKEM_K == 2+ mlk_poly_getnoise_eta1122_4x(&sp->vec[0], &sp->vec[1], &ep->vec[0],+ &ep->vec[1], coins, 0, 1, 2, 3);+ mlk_poly_getnoise_eta2(epp, coins, 4);+#elif MLKEM_K == 3+ /*+ * In this call, only the first three output buffers are needed,+ * so we pass NULL as the fourth parameter, and 0xFF as its dummy nonce.+ */+ mlk_poly_getnoise_eta1_4x(&sp->vec[0], &sp->vec[1], &sp->vec[2], NULL, coins,+ 0, 1, 2, 0xFF /* irrelevant */);+ /* The fourth output buffer in this call _is_ used. */+ mlk_poly_getnoise_eta2_4x(&ep->vec[0], &ep->vec[1], &ep->vec[2], epp, coins,+ 3, 4, 5, 6);+#elif MLKEM_K == 4+ mlk_poly_getnoise_eta1_4x(&sp->vec[0], &sp->vec[1], &sp->vec[2], &sp->vec[3],+ coins, 0, 1, 2, 3);+ mlk_poly_getnoise_eta2_4x(&ep->vec[0], &ep->vec[1], &ep->vec[2], &ep->vec[3],+ coins, 4, 5, 6, 7);+ mlk_poly_getnoise_eta2(epp, coins, 8);+#endif /* MLKEM_K == 4 */+}+#endif /* !MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API */+++/* Reference: `indcpa_keypair_derand()` in the reference implementation @[REF].+ * - We use a different implementation of `gen_matrix()` which+ * uses x4-batched Keccak-f1600 (see `mlk_gen_matrix()` above).+ * - We use a mulcache to speed up matrix-vector multiplication.+ * - We include buffer zeroization.+ */+#if !defined(MLK_CONFIG_NO_KEYPAIR_API)+MLK_INTERNAL_API+int mlk_indcpa_keypair_derand(uint8_t pk[MLKEM_INDCPA_PUBLICKEYBYTES],+ uint8_t sk[MLKEM_INDCPA_SECRETKEYBYTES],+ const uint8_t coins[MLKEM_SYMBYTES],+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context)+{+ int ret = 0;+ const uint8_t *publicseed;+ const uint8_t *noiseseed;+ MLK_ALLOC(buf, uint8_t, 2 * MLKEM_SYMBYTES, context);+ MLK_ALLOC(coins_with_domain_separator, uint8_t, MLKEM_SYMBYTES + 1, context);+ MLK_ALLOC(a, mlk_polymat, 1, context);+ MLK_ALLOC(e, mlk_polyvec, 1, context);+ MLK_ALLOC(pkpv, mlk_polyvec, 1, context);+ MLK_ALLOC(skpv, mlk_polyvec, 1, context);+ MLK_ALLOC(skpv_cache, mlk_polyvec_mulcache, 1, context);++ if (buf == NULL || coins_with_domain_separator == NULL || a == NULL ||+ e == NULL || pkpv == NULL || skpv == NULL || skpv_cache == NULL)+ {+ ret = MLK_ERR_OUT_OF_MEMORY;+ goto cleanup;+ }++ publicseed = buf;+ noiseseed = buf + MLKEM_SYMBYTES;++ /* Concatenate coins with MLKEM_K for domain separation of security levels */+ mlk_memcpy(coins_with_domain_separator, coins, MLKEM_SYMBYTES);+ coins_with_domain_separator[MLKEM_SYMBYTES] = MLKEM_K;++ mlk_hash_g(buf, coins_with_domain_separator, MLKEM_SYMBYTES + 1);++ /*+ * Declassify the public seed.+ * Required to use it in conditional-branches in rejection sampling.+ * This is needed because all output of randombytes is marked as secret+ * (=undefined)+ */+ MLK_CT_TESTING_DECLASSIFY(publicseed, MLKEM_SYMBYTES);++ mlk_gen_matrix(a, publicseed, 0 /* no transpose */);++ mlk_keypair_getnoise_eta1(skpv, e, noiseseed);++ mlk_polyvec_ntt(skpv);+ mlk_polyvec_ntt(e);++ mlk_polyvec_mulcache_compute(skpv_cache, skpv);+ mlk_matvec_mul(pkpv, a, skpv, skpv_cache);+ mlk_polyvec_tomont(pkpv);++ mlk_polyvec_add(pkpv, e);+ mlk_polyvec_reduce(pkpv);+ mlk_polyvec_reduce(skpv);++ mlk_pack_sk(sk, skpv);+ mlk_pack_pk(pk, pkpv, publicseed);++cleanup:+ /* Specification: Partially implements+ * @[FIPS203, Section 3.3, Destruction of intermediate values] */+ MLK_FREE(skpv_cache, mlk_polyvec_mulcache, 1, context);+ MLK_FREE(skpv, mlk_polyvec, 1, context);+ MLK_FREE(pkpv, mlk_polyvec, 1, context);+ MLK_FREE(e, mlk_polyvec, 1, context);+ MLK_FREE(a, mlk_polymat, 1, context);+ MLK_FREE(coins_with_domain_separator, uint8_t, MLKEM_SYMBYTES + 1, context);+ MLK_FREE(buf, uint8_t, 2 * MLKEM_SYMBYTES, context);+ return ret;+}+#endif /* !MLK_CONFIG_NO_KEYPAIR_API */++/* Reference: `indcpa_enc()` in the reference implementation @[REF].+ * - We use x4-batched versions of `poly_getnoise` to leverage+ * batched x4-batched Keccak-f1600.+ * - We use a different implementation of `gen_matrix()` which+ * uses x4-batched Keccak-f1600 (see `mlk_gen_matrix()` above).+ * - We use a mulcache to speed up matrix-vector multiplication.+ * - We include buffer zeroization.+ */+#if !defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API)+MLK_INTERNAL_API+int mlk_indcpa_enc(uint8_t c[MLKEM_INDCPA_BYTES],+ const uint8_t m[MLKEM_INDCPA_MSGBYTES],+ const uint8_t pk[MLKEM_INDCPA_PUBLICKEYBYTES],+ const uint8_t coins[MLKEM_SYMBYTES],+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context)+{+ int ret = 0;+ MLK_ALLOC(seed, uint8_t, MLKEM_SYMBYTES, context);+ MLK_ALLOC(at, mlk_polymat, 1, context);+ MLK_ALLOC(sp, mlk_polyvec, 1, context);+ MLK_ALLOC(pkpv, mlk_polyvec, 1, context);+ MLK_ALLOC(ep, mlk_polyvec, 1, context);+ MLK_ALLOC(b, mlk_polyvec, 1, context);+ MLK_ALLOC(v, mlk_poly, 1, context);+ MLK_ALLOC(k, mlk_poly, 1, context);+ MLK_ALLOC(epp, mlk_poly, 1, context);+ MLK_ALLOC(sp_cache, mlk_polyvec_mulcache, 1, context);++ if (seed == NULL || at == NULL || sp == NULL || pkpv == NULL || ep == NULL ||+ b == NULL || v == NULL || k == NULL || epp == NULL || sp_cache == NULL)+ {+ ret = MLK_ERR_OUT_OF_MEMORY;+ goto cleanup;+ }++ mlk_unpack_pk(pkpv, seed, pk);+ mlk_poly_frommsg(k, m);++ /*+ * Declassify the public seed.+ * Required to use it in conditional-branches in rejection sampling.+ * This is needed because in re-encryption the publicseed originated from sk+ * which is marked undefined.+ */+ MLK_CT_TESTING_DECLASSIFY(seed, MLKEM_SYMBYTES);++ mlk_gen_matrix(at, seed, 1 /* transpose */);++ mlk_enc_getnoise_eta1_eta2(sp, ep, epp, coins);++ mlk_polyvec_ntt(sp);++ mlk_polyvec_mulcache_compute(sp_cache, sp);+ mlk_matvec_mul(b, at, sp, sp_cache);+ mlk_polyvec_basemul_acc_montgomery_cached(v, pkpv, sp, sp_cache);++ mlk_polyvec_invntt_tomont(b);+ mlk_poly_invntt_tomont(v);++ mlk_polyvec_add(b, ep);+ mlk_poly_add(v, epp);+ mlk_poly_add(v, k);++ mlk_polyvec_reduce(b);+ mlk_poly_reduce(v);++ mlk_pack_ciphertext(c, b, v);++cleanup:+ /* Specification: Partially implements+ * @[FIPS203, Section 3.3, Destruction of intermediate values] */+ MLK_FREE(sp_cache, mlk_polyvec_mulcache, 1, context);+ MLK_FREE(epp, mlk_poly, 1, context);+ MLK_FREE(k, mlk_poly, 1, context);+ MLK_FREE(v, mlk_poly, 1, context);+ MLK_FREE(b, mlk_polyvec, 1, context);+ MLK_FREE(ep, mlk_polyvec, 1, context);+ MLK_FREE(pkpv, mlk_polyvec, 1, context);+ MLK_FREE(sp, mlk_polyvec, 1, context);+ MLK_FREE(at, mlk_polymat, 1, context);+ MLK_FREE(seed, uint8_t, MLKEM_SYMBYTES, context);+ return ret;+}+#endif /* !MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API */++/* Reference: `indcpa_dec()` in the reference implementation @[REF].+ * - We use a mulcache for the scalar product.+ * - We include buffer zeroization. */+#if !defined(MLK_CONFIG_NO_DECAPS_API)+MLK_INTERNAL_API+int mlk_indcpa_dec(uint8_t m[MLKEM_INDCPA_MSGBYTES],+ const uint8_t c[MLKEM_INDCPA_BYTES],+ const uint8_t sk[MLKEM_INDCPA_SECRETKEYBYTES],+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context)+{+ int ret = 0;+ MLK_ALLOC(b, mlk_polyvec, 1, context);+ MLK_ALLOC(skpv, mlk_polyvec, 1, context);+ MLK_ALLOC(v, mlk_poly, 1, context);+ MLK_ALLOC(sb, mlk_poly, 1, context);+ MLK_ALLOC(b_cache, mlk_polyvec_mulcache, 1, context);++ if (b == NULL || skpv == NULL || v == NULL || sb == NULL || b_cache == NULL)+ {+ ret = MLK_ERR_OUT_OF_MEMORY;+ goto cleanup;+ }++ mlk_unpack_ciphertext(b, v, c);+ mlk_unpack_sk(skpv, sk);++ mlk_polyvec_ntt(b);+ mlk_polyvec_mulcache_compute(b_cache, b);+ mlk_polyvec_basemul_acc_montgomery_cached(sb, skpv, b, b_cache);+ mlk_poly_invntt_tomont(sb);++ mlk_poly_sub(v, sb);+ mlk_poly_reduce(v);++ mlk_poly_tomsg(m, v);++cleanup:+ /* Specification: Partially implements+ * @[FIPS203, Section 3.3, Destruction of intermediate values] */+ MLK_FREE(b_cache, mlk_polyvec_mulcache, 1, context);+ MLK_FREE(sb, mlk_poly, 1, context);+ MLK_FREE(v, mlk_poly, 1, context);+ MLK_FREE(skpv, mlk_polyvec, 1, context);+ MLK_FREE(b, mlk_polyvec, 1, context);+ return ret;+}+#endif /* !MLK_CONFIG_NO_DECAPS_API */++/* To facilitate single-compilation-unit (SCU) builds, undefine all macros.+ * Don't modify by hand -- this is auto-generated by scripts/autogen. */+#undef mlk_pack_pk+#undef mlk_unpack_pk+#undef mlk_pack_sk+#undef mlk_unpack_sk+#undef mlk_pack_ciphertext+#undef mlk_unpack_ciphertext+#undef mlk_matvec_mul+#undef mlk_polyvec_permute_bitrev_to_custom+#undef mlk_polymat_permute_bitrev_to_custom+#undef mlk_keypair_getnoise_eta1+#undef mlk_enc_getnoise_eta1_eta2
@@ -0,0 +1,172 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS203]+ * FIPS 203 Module-Lattice-Based Key-Encapsulation Mechanism Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/203/final+ */++#ifndef MLK_INDCPA_H+#define MLK_INDCPA_H++#include "cbmc.h"+#include "common.h"+#include "poly_k.h"++#define mlk_gen_matrix MLK_NAMESPACE_K(gen_matrix)+/**+ * Deterministically generate matrix A (or the transpose of A) from a seed.+ * Entries of the matrix are polynomials that look uniformly random.+ * Performs rejection sampling on the output of an XOF.+ *+ * @spec{Implements @[FIPS203, Algorithm 13 (K-PKE.KeyGen), L3-7] and+ * @[FIPS203, Algorithm 14 (K-PKE.Encrypt), L4-8]. The @p transposed+ * parameter only affects internal presentation.}+ *+ * @param[out] a Output matrix A.+ * @param[in] seed Input seed.+ * @param transposed Boolean deciding whether A or A^T is generated.+ */+MLK_INTERNAL_API+void mlk_gen_matrix(mlk_polymat *a, const uint8_t seed[MLKEM_SYMBYTES],+ int transposed)+__contract__(+ requires(memory_no_alias(a, sizeof(mlk_polymat)))+ requires(memory_no_alias(seed, MLKEM_SYMBYTES))+ requires(transposed == 0 || transposed == 1)+ assigns(memory_slice(a, sizeof(mlk_polymat)))+ ensures(forall(x, 0, MLKEM_K, forall(y, 0, MLKEM_K,+ array_bound(a->vec[x].vec[y].coeffs, 0, MLKEM_N, 0, MLKEM_Q))))+);++#if !defined(MLK_CONFIG_NO_KEYPAIR_API)+#define mlk_indcpa_keypair_derand \+ MLK_NAMESPACE_K(indcpa_keypair_derand) MLK_CONTEXT_PARAMETERS_3+/**+ * Generate public and private key for the CPA-secure public-key encryption+ * scheme underlying ML-KEM.+ *+ * @spec{Implements @[FIPS203, Algorithm 13 (K-PKE.KeyGen)].}+ *+ * @param[out] pk Output public key+ * (length MLKEM_INDCPA_PUBLICKEYBYTES bytes).+ * @param[out] sk Output private key+ * (length MLKEM_INDCPA_SECRETKEYBYTES bytes).+ * @param[in] coins Input randomness (length MLKEM_SYMBYTES bytes).+ * @param context Application context. Only present when+ * MLK_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLK_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLK_ERR_OUT_OF_MEMORY MLK_CONFIG_CUSTOM_ALLOC_FREE was used and+ * MLK_CUSTOM_ALLOC returned NULL.+ */+MLK_INTERNAL_API+MLK_MUST_CHECK_RETURN_VALUE+int mlk_indcpa_keypair_derand(uint8_t pk[MLKEM_INDCPA_PUBLICKEYBYTES],+ uint8_t sk[MLKEM_INDCPA_SECRETKEYBYTES],+ const uint8_t coins[MLKEM_SYMBYTES],+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context)+__contract__(+ requires(memory_no_alias(pk, MLKEM_INDCPA_PUBLICKEYBYTES))+ requires(memory_no_alias(sk, MLKEM_INDCPA_SECRETKEYBYTES))+ requires(memory_no_alias(coins, MLKEM_SYMBYTES))+ assigns(memory_slice(pk, MLKEM_INDCPA_PUBLICKEYBYTES))+ assigns(memory_slice(sk, MLKEM_INDCPA_SECRETKEYBYTES))+ ensures(return_value == 0 || return_value == MLK_ERR_OUT_OF_MEMORY)+ /* Output buffers on error, per API-CONVENTIONS.md */+ ensures(return_value != 0 ==>+ array_unchanged_or_zeroized_u8(pk, MLKEM_INDCPA_PUBLICKEYBYTES))+ ensures(return_value != 0 ==>+ array_unchanged_or_zeroized_u8(sk, MLKEM_INDCPA_SECRETKEYBYTES))+);+#endif /* !MLK_CONFIG_NO_KEYPAIR_API */++#if !defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API)+#define mlk_indcpa_enc MLK_NAMESPACE_K(indcpa_enc) MLK_CONTEXT_PARAMETERS_4+/**+ * Encryption function of the CPA-secure public-key encryption scheme+ * underlying ML-KEM.+ *+ * @spec{Implements @[FIPS203, Algorithm 14 (K-PKE.Encrypt)].}+ *+ * @param[out] c Output ciphertext (length MLKEM_INDCPA_BYTES bytes).+ * @param[in] m Input message (length MLKEM_INDCPA_MSGBYTES bytes).+ * @param[in] pk Input public key+ * (length MLKEM_INDCPA_PUBLICKEYBYTES bytes).+ * @param[in] coins Input random coins used as seed (length MLKEM_SYMBYTES+ * bytes) to deterministically generate all randomness.+ * @param context Application context. Only present when+ * MLK_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLK_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLK_ERR_OUT_OF_MEMORY MLK_CONFIG_CUSTOM_ALLOC_FREE was used and+ * MLK_CUSTOM_ALLOC returned NULL.+ */+MLK_INTERNAL_API+MLK_MUST_CHECK_RETURN_VALUE+int mlk_indcpa_enc(uint8_t c[MLKEM_INDCPA_BYTES],+ const uint8_t m[MLKEM_INDCPA_MSGBYTES],+ const uint8_t pk[MLKEM_INDCPA_PUBLICKEYBYTES],+ const uint8_t coins[MLKEM_SYMBYTES],+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context)+__contract__(+ requires(memory_no_alias(c, MLKEM_INDCPA_BYTES))+ requires(memory_no_alias(m, MLKEM_INDCPA_MSGBYTES))+ requires(memory_no_alias(pk, MLKEM_INDCPA_PUBLICKEYBYTES))+ requires(memory_no_alias(coins, MLKEM_SYMBYTES))+ assigns(memory_slice(c, MLKEM_INDCPA_BYTES))+ ensures(return_value == 0 || return_value == MLK_ERR_OUT_OF_MEMORY)+ /* Output buffers on error, per API-CONVENTIONS.md */+ ensures(return_value != 0 ==>+ array_unchanged_or_zeroized_u8(c, MLKEM_INDCPA_BYTES))+);+#endif /* !MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API */++#if !defined(MLK_CONFIG_NO_DECAPS_API)+#define mlk_indcpa_dec MLK_NAMESPACE_K(indcpa_dec) MLK_CONTEXT_PARAMETERS_3+/**+ * Decryption function of the CPA-secure public-key encryption scheme+ * underlying ML-KEM.+ *+ * @spec{Implements @[FIPS203, Algorithm 15 (K-PKE.Decrypt)].}+ *+ * @param[out] m Output decrypted message+ * (length MLKEM_INDCPA_MSGBYTES bytes).+ * @param[in] c Input ciphertext (length MLKEM_INDCPA_BYTES bytes).+ * @param[in] sk Input secret key+ * (length MLKEM_INDCPA_SECRETKEYBYTES bytes).+ * @param context Application context. Only present when+ * MLK_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLK_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLK_ERR_OUT_OF_MEMORY MLK_CONFIG_CUSTOM_ALLOC_FREE was used and+ * MLK_CUSTOM_ALLOC returned NULL.+ */+MLK_INTERNAL_API+MLK_MUST_CHECK_RETURN_VALUE+int mlk_indcpa_dec(uint8_t m[MLKEM_INDCPA_MSGBYTES],+ const uint8_t c[MLKEM_INDCPA_BYTES],+ const uint8_t sk[MLKEM_INDCPA_SECRETKEYBYTES],+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context)+__contract__(+ requires(memory_no_alias(c, MLKEM_INDCPA_BYTES))+ requires(memory_no_alias(m, MLKEM_INDCPA_MSGBYTES))+ requires(memory_no_alias(sk, MLKEM_INDCPA_SECRETKEYBYTES))+ assigns(memory_slice(m, MLKEM_INDCPA_MSGBYTES))+ ensures(return_value == 0 || return_value == MLK_ERR_OUT_OF_MEMORY)+ /* Output buffers on error, per API-CONVENTIONS.md */+ ensures(return_value != 0 ==>+ array_unchanged_or_zeroized_u8(m, MLKEM_INDCPA_MSGBYTES))+);+#endif /* !MLK_CONFIG_NO_DECAPS_API */++#endif /* !MLK_INDCPA_H */
@@ -0,0 +1,474 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS140_3_IG]+ * Implementation Guidance for FIPS 140-3 and the Cryptographic Module+ * Validation Program+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/projects/cryptographic-module-validation-program/fips-140-3-ig-announcements+ *+ * - [FIPS203]+ * FIPS 203 Module-Lattice-Based Key-Encapsulation Mechanism Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/203/final+ *+ * - [REF]+ * CRYSTALS-Kyber C reference implementation+ * Bos, Ducas, Kiltz, Lepoint, Lyubashevsky, Schanck, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/kyber/tree/main/ref+ */++#include "kem.h"++#include "indcpa.h"+#include "randombytes.h"+#include "symmetric.h"+#include "verify.h"++/* Parameter set namespacing+ * This is to facilitate building multiple instances+ * of mlkem-native (e.g. with varying security levels)+ * within a single compilation unit. */+#define mlk_check_pct MLK_ADD_PARAM_SET(mlk_check_pct) MLK_CONTEXT_PARAMETERS_2+/* End of parameter set namespacing */++#if !defined(MLK_CONFIG_NO_ENCAPS_API)+/* Reference: Not implemented in the reference implementation @[REF]. */+MLK_EXTERNAL_API+MLK_MUST_CHECK_RETURN_VALUE+int mlk_kem_check_pk(const uint8_t pk[MLKEM_INDCCA_PUBLICKEYBYTES],+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context)+{+ int ret = 0;+ MLK_ALLOC(p, mlk_polyvec, 1, context);+ MLK_ALLOC(p_reencoded, uint8_t, MLKEM_POLYVECBYTES, context);++ if (p == NULL || p_reencoded == NULL)+ {+ ret = MLK_ERR_OUT_OF_MEMORY;+ goto cleanup;+ }++ mlk_polyvec_frombytes(p, pk);+ mlk_polyvec_reduce(p);+ mlk_polyvec_tobytes(p_reencoded, p);++ /* We use a constant-time memcmp here to avoid having to+ * declassify the PK before the PCT has succeeded. */+ ret = mlk_ct_memcmp(pk, p_reencoded, MLKEM_POLYVECBYTES) ? MLK_ERR_INVALID_PK+ : 0;++cleanup:+ /* Specification: Partially implements+ * @[FIPS203, Section 3.3, Destruction of intermediate values] */+ MLK_FREE(p_reencoded, uint8_t, MLKEM_POLYVECBYTES, context);+ MLK_FREE(p, mlk_polyvec, 1, context);+ return ret;+}+#endif /* !MLK_CONFIG_NO_ENCAPS_API */+++#if !defined(MLK_CONFIG_NO_DECAPS_API)+/* Reference: Not implemented in the reference implementation @[REF]. */+MLK_EXTERNAL_API+MLK_MUST_CHECK_RETURN_VALUE+int mlk_kem_check_sk(const uint8_t sk[MLKEM_INDCCA_SECRETKEYBYTES],+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context)+{+ int ret = 0;+ MLK_ALLOC(test, uint8_t, MLKEM_SYMBYTES, context);++ if (test == NULL)+ {+ ret = MLK_ERR_OUT_OF_MEMORY;+ goto cleanup;+ }++ /*+ * The parts of `sk` being hashed and compared here are public, so+ * no public information is leaked through the runtime or the return value+ * of this function.+ */++ /* Declassify the public part of the secret key */+ MLK_CT_TESTING_DECLASSIFY(sk + MLKEM_INDCPA_SECRETKEYBYTES,+ MLKEM_INDCCA_PUBLICKEYBYTES);+ MLK_CT_TESTING_DECLASSIFY(+ sk + MLKEM_INDCCA_SECRETKEYBYTES - 2 * MLKEM_SYMBYTES, MLKEM_SYMBYTES);++ mlk_hash_h(test, sk + MLKEM_INDCPA_SECRETKEYBYTES,+ MLKEM_INDCCA_PUBLICKEYBYTES);+ /* This doesn't have to be a constant-time memcmp, but it's the only place+ * in the library where a normal memcmp would be used otherwise, so for sake+ * of minimizing stdlib dependency, we use our constant-time one anyway. */+ ret = mlk_ct_memcmp(sk + MLKEM_INDCCA_SECRETKEYBYTES - 2 * MLKEM_SYMBYTES,+ test, MLKEM_SYMBYTES)+ ? MLK_ERR_INVALID_SK+ : 0;++cleanup:+ /* Specification: Partially implements+ * @[FIPS203, Section 3.3, Destruction of intermediate values] */+ MLK_FREE(test, uint8_t, MLKEM_SYMBYTES, context);+ return ret;+}+#endif /* !MLK_CONFIG_NO_DECAPS_API */++#if !defined(MLK_CONFIG_NO_KEYPAIR_API)+MLK_MUST_CHECK_RETURN_VALUE+static int mlk_check_pct(uint8_t const pk[MLKEM_INDCCA_PUBLICKEYBYTES],+ uint8_t const sk[MLKEM_INDCCA_SECRETKEYBYTES],+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context)+__contract__(+ requires(memory_no_alias(pk, MLKEM_INDCCA_PUBLICKEYBYTES))+ requires(memory_no_alias(sk, MLKEM_INDCCA_SECRETKEYBYTES))+ ensures(return_value == 0 || return_value == MLK_ERR_OUT_OF_MEMORY ||+ return_value == MLK_ERR_RNG_FAIL ||+ return_value == MLK_ERR_PCT_FAIL)+);++#if defined(MLK_CONFIG_KEYGEN_PCT)+/* Specification:+ * Partially implements 'Pairwise Consistency Test' @[FIPS140_3_IG, p.87] and+ * @[FIPS203, Section 7.1, Pairwise Consistency]. */++/* Reference: Not implemented in the reference implementation @[REF].+ *+ * @retval 0 Success.+ * @retval MLK_ERR_OUT_OF_MEMORY MLK_CONFIG_CUSTOM_ALLOC_FREE was used and+ * MLK_CUSTOM_ALLOC returned NULL.+ * @retval MLK_ERR_RNG_FAIL Random number generation failed.+ * @retval MLK_ERR_PCT_FAIL The consistency check failed. */+MLK_MUST_CHECK_RETURN_VALUE+static int mlk_check_pct(uint8_t const pk[MLKEM_INDCCA_PUBLICKEYBYTES],+ uint8_t const sk[MLKEM_INDCCA_SECRETKEYBYTES],+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context)+{+ int ret = 0;+ MLK_ALLOC(ct, uint8_t, MLKEM_INDCCA_CIPHERTEXTBYTES, context);+ MLK_ALLOC(ss_enc, uint8_t, MLKEM_SSBYTES, context);+ MLK_ALLOC(ss_dec, uint8_t, MLKEM_SSBYTES, context);++ if (ct == NULL || ss_enc == NULL || ss_dec == NULL)+ {+ ret = MLK_ERR_OUT_OF_MEMORY;+ goto cleanup;+ }++ ret = mlk_kem_enc(ct, ss_enc, pk, context);+ if (ret != 0)+ {+ goto cleanup;+ }++ ret = mlk_kem_dec(ss_dec, ct, sk, context);+ if (ret != 0)+ {+ goto cleanup;+ }++#if defined(MLK_CONFIG_KEYGEN_PCT_BREAKAGE_TEST)+ /* Deliberately break PCT for testing purposes */+ if (mlk_break_pct())+ {+ ss_enc[0] = ~ss_enc[0];+ }+#endif /* MLK_CONFIG_KEYGEN_PCT_BREAKAGE_TEST */++ ret = mlk_ct_memcmp(ss_enc, ss_dec, MLKEM_SSBYTES);+ /* The result of the PCT is public. */+ MLK_CT_TESTING_DECLASSIFY(&ret, sizeof(ret));++ if (ret != 0)+ {+ ret = MLK_ERR_PCT_FAIL;+ }++cleanup:++ /* Specification: Partially implements+ * @[FIPS203, Section 3.3, Destruction of intermediate values] */+ MLK_FREE(ss_dec, uint8_t, MLKEM_SSBYTES, context);+ MLK_FREE(ss_enc, uint8_t, MLKEM_SSBYTES, context);+ MLK_FREE(ct, uint8_t, MLKEM_INDCCA_CIPHERTEXTBYTES, context);++ /* The key pair being tested was just generated by this library, so a key+ * check rejecting it hints at a faulty implementation rather than at bad+ * input. Report it as a PCT failure. */+ if (ret == MLK_ERR_INVALID_PK || ret == MLK_ERR_INVALID_SK)+ {+ ret = MLK_ERR_PCT_FAIL;+ }++ /* Other error codes, e.g. platform failures like out of memory or+ * randomness failure, are passed on unmodified. */++ return ret;+}+#else /* MLK_CONFIG_KEYGEN_PCT */+MLK_MUST_CHECK_RETURN_VALUE+static int mlk_check_pct(uint8_t const pk[MLKEM_INDCCA_PUBLICKEYBYTES],+ uint8_t const sk[MLKEM_INDCCA_SECRETKEYBYTES],+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context)+{+ /* Skip PCT */+ ((void)pk);+ ((void)sk);+ MLK_CONTEXT_UNUSED(context);+ return 0;+}+#endif /* !MLK_CONFIG_KEYGEN_PCT */++/* Reference: `crypto_kem_keypair_derand()` in the reference implementation+ * @[REF].+ * - We optionally include PCT which is not present in+ * the reference code. */+MLK_EXTERNAL_API+MLK_MUST_CHECK_RETURN_VALUE+int mlk_kem_keypair_derand(uint8_t pk[MLKEM_INDCCA_PUBLICKEYBYTES],+ uint8_t sk[MLKEM_INDCCA_SECRETKEYBYTES],+ const uint8_t coins[2 * MLKEM_SYMBYTES],+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context)+{+ int ret;++ ret = mlk_indcpa_keypair_derand(pk, sk, coins, context);+ if (ret != 0)+ {+ goto cleanup;+ }++ mlk_memcpy(sk + MLKEM_INDCPA_SECRETKEYBYTES, pk, MLKEM_INDCCA_PUBLICKEYBYTES);+ mlk_hash_h(sk + MLKEM_INDCCA_SECRETKEYBYTES - 2 * MLKEM_SYMBYTES, pk,+ MLKEM_INDCCA_PUBLICKEYBYTES);+ /* Value z for pseudo-random output on reject */+ mlk_memcpy(sk + MLKEM_INDCCA_SECRETKEYBYTES - MLKEM_SYMBYTES,+ coins + MLKEM_SYMBYTES, MLKEM_SYMBYTES);++ /* Declassify public key */+ MLK_CT_TESTING_DECLASSIFY(pk, MLKEM_INDCCA_PUBLICKEYBYTES);++ /* Pairwise Consistency Test (PCT) @[FIPS140_3_IG, p.87] */+ ret = mlk_check_pct(pk, sk, context);+ if (ret != 0)+ {+ goto cleanup;+ }++cleanup:+ if (ret != 0)+ {+ mlk_zeroize(pk, MLKEM_INDCCA_PUBLICKEYBYTES);+ mlk_zeroize(sk, MLKEM_INDCCA_SECRETKEYBYTES);+ }++ return ret;+}++#if !defined(MLK_CONFIG_NO_RANDOMIZED_API)+/* Reference: `crypto_kem_keypair()` in the reference implementation @[REF]+ * - We zeroize the stack buffer */+MLK_EXTERNAL_API+int mlk_kem_keypair(uint8_t pk[MLKEM_INDCCA_PUBLICKEYBYTES],+ uint8_t sk[MLKEM_INDCCA_SECRETKEYBYTES],+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context)+{+ int ret = 0;+ MLK_ALLOC(coins, uint8_t, 2 * MLKEM_SYMBYTES, context);++ if (coins == NULL)+ {+ ret = MLK_ERR_OUT_OF_MEMORY;+ goto cleanup;+ }++ /* Acquire necessary randomness, and mark it as secret. */+ if (mlk_randombytes(coins, 2 * MLKEM_SYMBYTES) != 0)+ {+ ret = MLK_ERR_RNG_FAIL;+ goto cleanup;+ }++ MLK_CT_TESTING_SECRET(coins, 2 * MLKEM_SYMBYTES);++ ret = mlk_kem_keypair_derand(pk, sk, coins, context);++cleanup:+ /* Specification: Partially implements+ * @[FIPS203, Section 3.3, Destruction of intermediate values] */+ MLK_FREE(coins, uint8_t, 2 * MLKEM_SYMBYTES, context);+ return ret;+}+#endif /* !MLK_CONFIG_NO_RANDOMIZED_API */+#endif /* !MLK_CONFIG_NO_KEYPAIR_API */++#if !defined(MLK_CONFIG_NO_ENCAPS_API)+/* Reference: `crypto_kem_enc_derand()` in the reference implementation @[REF]+ * - We include public key check+ * - We include stack buffer zeroization */+MLK_EXTERNAL_API+int mlk_kem_enc_derand(uint8_t ct[MLKEM_INDCCA_CIPHERTEXTBYTES],+ uint8_t ss[MLKEM_SSBYTES],+ const uint8_t pk[MLKEM_INDCCA_PUBLICKEYBYTES],+ const uint8_t coins[MLKEM_SYMBYTES],+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context)+{+ int ret = 0;+ MLK_ALLOC(buf, uint8_t, 2 * MLKEM_SYMBYTES, context);+ MLK_ALLOC(kr, uint8_t, 2 * MLKEM_SYMBYTES, context);++ if (buf == NULL || kr == NULL)+ {+ ret = MLK_ERR_OUT_OF_MEMORY;+ goto cleanup;+ }++ /* Specification: Implements @[FIPS203, Section 7.2, Modulus check] */+ ret = mlk_kem_check_pk(pk, context);+ if (ret != 0)+ {+ goto cleanup;+ }++ mlk_memcpy(buf, coins, MLKEM_SYMBYTES);++ /* Multitarget countermeasure for coins + contributory KEM */+ mlk_hash_h(buf + MLKEM_SYMBYTES, pk, MLKEM_INDCCA_PUBLICKEYBYTES);+ mlk_hash_g(kr, buf, 2 * MLKEM_SYMBYTES);++ /* coins are in kr+MLKEM_SYMBYTES */+ ret = mlk_indcpa_enc(ct, buf, pk, kr + MLKEM_SYMBYTES, context);+ if (ret != 0)+ {+ goto cleanup;+ }++ mlk_memcpy(ss, kr, MLKEM_SYMBYTES);++cleanup:+ /* Specification: Partially implements+ * @[FIPS203, Section 3.3, Destruction of intermediate values] */+ MLK_FREE(kr, uint8_t, 2 * MLKEM_SYMBYTES, context);+ MLK_FREE(buf, uint8_t, 2 * MLKEM_SYMBYTES, context);+ return ret;+}++#if !defined(MLK_CONFIG_NO_RANDOMIZED_API)+/* Reference: `crypto_kem_enc()` in the reference implementation @[REF]+ * - We include stack buffer zeroization */+MLK_EXTERNAL_API+int mlk_kem_enc(uint8_t ct[MLKEM_INDCCA_CIPHERTEXTBYTES],+ uint8_t ss[MLKEM_SSBYTES],+ const uint8_t pk[MLKEM_INDCCA_PUBLICKEYBYTES],+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context)+{+ int ret = 0;+ MLK_ALLOC(coins, uint8_t, MLKEM_SYMBYTES, context);++ if (coins == NULL)+ {+ ret = MLK_ERR_OUT_OF_MEMORY;+ goto cleanup;+ }++ if (mlk_randombytes(coins, MLKEM_SYMBYTES) != 0)+ {+ ret = MLK_ERR_RNG_FAIL;+ goto cleanup;+ }++ MLK_CT_TESTING_SECRET(coins, MLKEM_SYMBYTES);++ ret = mlk_kem_enc_derand(ct, ss, pk, coins, context);++cleanup:+ /* Specification: Partially implements+ * @[FIPS203, Section 3.3, Destruction of intermediate values] */+ MLK_FREE(coins, uint8_t, MLKEM_SYMBYTES, context);+ return ret;+}+#endif /* !MLK_CONFIG_NO_RANDOMIZED_API */+#endif /* !MLK_CONFIG_NO_ENCAPS_API */++#if !defined(MLK_CONFIG_NO_DECAPS_API)+/* Reference: `crypto_kem_dec()` in the reference implementation @[REF]+ * - We include secret key check+ * - We include stack buffer zeroization */+MLK_EXTERNAL_API+int mlk_kem_dec(uint8_t ss[MLKEM_SSBYTES],+ const uint8_t ct[MLKEM_INDCCA_CIPHERTEXTBYTES],+ const uint8_t sk[MLKEM_INDCCA_SECRETKEYBYTES],+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context)+{+ int ret = 0;+ uint8_t fail;+ const uint8_t *pk = sk + MLKEM_INDCPA_SECRETKEYBYTES;+ MLK_ALLOC(buf, uint8_t, 2 * MLKEM_SYMBYTES, context);+ MLK_ALLOC(kr, uint8_t, 2 * MLKEM_SYMBYTES, context);+ MLK_ALLOC(tmp, uint8_t, MLKEM_SYMBYTES + MLKEM_INDCCA_CIPHERTEXTBYTES,+ context);++ if (buf == NULL || kr == NULL || tmp == NULL)+ {+ ret = MLK_ERR_OUT_OF_MEMORY;+ goto cleanup;+ }++ /* Specification: Implements @[FIPS203, Section 7.3, Hash check] */+ ret = mlk_kem_check_sk(sk, context);+ if (ret != 0)+ {+ goto cleanup;+ }++ ret = mlk_indcpa_dec(buf, ct, sk, context);+ if (ret != 0)+ {+ goto cleanup;+ }++ /* Multitarget countermeasure for coins + contributory KEM */+ mlk_memcpy(buf + MLKEM_SYMBYTES,+ sk + MLKEM_INDCCA_SECRETKEYBYTES - 2 * MLKEM_SYMBYTES,+ MLKEM_SYMBYTES);+ mlk_hash_g(kr, buf, 2 * MLKEM_SYMBYTES);++ /* Recompute and compare ciphertext */+ /* coins are in kr+MLKEM_SYMBYTES */+ ret = mlk_indcpa_enc(tmp, buf, pk, kr + MLKEM_SYMBYTES, context);+ if (ret != 0)+ {+ goto cleanup;+ }++ fail = mlk_ct_memcmp(ct, tmp, MLKEM_INDCCA_CIPHERTEXTBYTES);++ /* Compute rejection key */+ mlk_memcpy(tmp, sk + MLKEM_INDCCA_SECRETKEYBYTES - MLKEM_SYMBYTES,+ MLKEM_SYMBYTES);+ mlk_memcpy(tmp + MLKEM_SYMBYTES, ct, MLKEM_INDCCA_CIPHERTEXTBYTES);+ mlk_hash_j(ss, tmp, MLKEM_SYMBYTES + MLKEM_INDCCA_CIPHERTEXTBYTES);++ /* Copy true key to return buffer if fail is 0 */+ mlk_ct_cmov_zero(ss, kr, MLKEM_SYMBYTES, fail);++cleanup:+ /* Specification: Partially implements+ * @[FIPS203, Section 3.3, Destruction of intermediate values] */+ MLK_FREE(tmp, uint8_t, MLKEM_SYMBYTES + MLKEM_INDCCA_CIPHERTEXTBYTES,+ context);+ MLK_FREE(kr, uint8_t, 2 * MLKEM_SYMBYTES, context);+ MLK_FREE(buf, uint8_t, 2 * MLKEM_SYMBYTES, context);++ return ret;+}+#endif /* !MLK_CONFIG_NO_DECAPS_API */++/* To facilitate single-compilation-unit (SCU) builds, undefine all macros.+ * Don't modify by hand -- this is auto-generated by scripts/autogen. */+#undef mlk_check_pct
@@ -0,0 +1,353 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS203]+ * FIPS 203 Module-Lattice-Based Key-Encapsulation Mechanism Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/203/final+ *+ * - [REF]+ * CRYSTALS-Kyber C reference implementation+ * Bos, Ducas, Kiltz, Lepoint, Lyubashevsky, Schanck, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/kyber/tree/main/ref+ */++#ifndef MLK_KEM_H+#define MLK_KEM_H++#include "cbmc.h"+#include "common.h"+#include "sys.h"++#if defined(MLK_CHECK_APIS)+/* Include to ensure consistency between internal kem.h+ * and external mlkem_native.h. */+#include "mlkem_native.h"++#if MLKEM_INDCCA_SECRETKEYBYTES != \+ MLKEM_SECRETKEYBYTES(MLK_CONFIG_PARAMETER_SET)+#error Mismatch for SECRETKEYBYTES between kem.h and mlkem_native.h+#endif++#if MLKEM_INDCCA_PUBLICKEYBYTES != \+ MLKEM_PUBLICKEYBYTES(MLK_CONFIG_PARAMETER_SET)+#error Mismatch for PUBLICKEYBYTES between kem.h and mlkem_native.h+#endif++#if MLKEM_INDCCA_CIPHERTEXTBYTES != \+ MLKEM_CIPHERTEXTBYTES(MLK_CONFIG_PARAMETER_SET)+#error Mismatch for CIPHERTEXTBYTES between kem.h and mlkem_native.h+#endif++#endif /* MLK_CHECK_APIS */++#if !defined(MLK_CONFIG_NO_KEYPAIR_API)+#define mlk_kem_keypair_derand \+ MLK_NAMESPACE_K(keypair_derand) MLK_CONTEXT_PARAMETERS_3+#if !defined(MLK_CONFIG_NO_RANDOMIZED_API)+#define mlk_kem_keypair MLK_NAMESPACE_K(keypair) MLK_CONTEXT_PARAMETERS_2+#endif+#endif /* !MLK_CONFIG_NO_KEYPAIR_API */+#if !defined(MLK_CONFIG_NO_ENCAPS_API)+#define mlk_kem_enc_derand MLK_NAMESPACE_K(enc_derand) MLK_CONTEXT_PARAMETERS_4+#if !defined(MLK_CONFIG_NO_RANDOMIZED_API)+#define mlk_kem_enc MLK_NAMESPACE_K(enc) MLK_CONTEXT_PARAMETERS_3+#endif+#define mlk_kem_check_pk MLK_NAMESPACE_K(check_pk) MLK_CONTEXT_PARAMETERS_1+#endif /* !MLK_CONFIG_NO_ENCAPS_API */+#if !defined(MLK_CONFIG_NO_DECAPS_API)+#define mlk_kem_dec MLK_NAMESPACE_K(dec) MLK_CONTEXT_PARAMETERS_3+#define mlk_kem_check_sk MLK_NAMESPACE_K(check_sk) MLK_CONTEXT_PARAMETERS_1+#endif++/**+ * Implements modulus check mandated by FIPS 203, i.e., ensures that+ * coefficients are in [0,q-1].+ *+ * @spec{Implements @[FIPS203, Section 7.2, 'modulus check'].}+ *+ * @reference{Not implemented in the reference implementation @[REF].}+ *+ * @param[in] pk Input public key (an already allocated array of+ * MLKEM_INDCCA_PUBLICKEYBYTES bytes).+ * @param context Application context. Only present when+ * MLK_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLK_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLK_ERR_INVALID_PK Modulus check failed.+ * @retval MLK_ERR_OUT_OF_MEMORY MLK_CONFIG_CUSTOM_ALLOC_FREE was used and+ * MLK_CUSTOM_ALLOC returned NULL.+ */+#if !defined(MLK_CONFIG_NO_ENCAPS_API)+MLK_EXTERNAL_API+MLK_MUST_CHECK_RETURN_VALUE+int mlk_kem_check_pk(const uint8_t pk[MLKEM_INDCCA_PUBLICKEYBYTES],+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context)+__contract__(+ requires(memory_no_alias(pk, MLKEM_INDCCA_PUBLICKEYBYTES))+ ensures(return_value == 0 || return_value == MLK_ERR_INVALID_PK ||+ return_value == MLK_ERR_OUT_OF_MEMORY)+);+#endif /* !MLK_CONFIG_NO_ENCAPS_API */+++/**+ * Implements public key hash check mandated by FIPS 203, i.e., ensures that+ * sk[768𝑘+32 ∶ 768𝑘+64] = H(pk) = H(sk[384𝑘 : 768𝑘+32]).+ *+ * @spec{Implements @[FIPS203, Section 7.3, 'hash check'].}+ *+ * @reference{Not implemented in the reference implementation @[REF].}+ *+ * @param[in] sk Input private key (an already allocated array of+ * MLKEM_INDCCA_SECRETKEYBYTES bytes).+ * @param context Application context. Only present when+ * MLK_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLK_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLK_ERR_INVALID_SK Public key hash check failed.+ * @retval MLK_ERR_OUT_OF_MEMORY MLK_CONFIG_CUSTOM_ALLOC_FREE was used and+ * MLK_CUSTOM_ALLOC returned NULL.+ */+#if !defined(MLK_CONFIG_NO_DECAPS_API)+MLK_EXTERNAL_API+MLK_MUST_CHECK_RETURN_VALUE+int mlk_kem_check_sk(const uint8_t sk[MLKEM_INDCCA_SECRETKEYBYTES],+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context)+__contract__(+ requires(memory_no_alias(sk, MLKEM_INDCCA_SECRETKEYBYTES))+ ensures(return_value == 0 || return_value == MLK_ERR_INVALID_SK ||+ return_value == MLK_ERR_OUT_OF_MEMORY)+);+#endif /* !MLK_CONFIG_NO_DECAPS_API */++#if !defined(MLK_CONFIG_NO_KEYPAIR_API)+/**+ * Generate a public/private keypair for the ML-KEM key encapsulation mechanism.+ *+ * @spec{Implements @[FIPS203, Algorithm 16, ML-KEM.KeyGen_Internal].}+ *+ * @param[out] pk Output public key (an already allocated array of+ * MLKEM_INDCCA_PUBLICKEYBYTES bytes).+ * @param[out] sk Output private key (an already allocated array of+ * MLKEM_INDCCA_SECRETKEYBYTES bytes).+ * @param[in] coins Input randomness (an already allocated array filled+ * with 2*MLKEM_SYMBYTES random bytes).+ * @param context Application context. Only present when+ * MLK_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLK_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLK_ERR_PCT_FAIL MLK_CONFIG_KEYGEN_PCT enabled and PCT failed.+ * @retval MLK_ERR_OUT_OF_MEMORY MLK_CONFIG_CUSTOM_ALLOC_FREE was used and+ * MLK_CUSTOM_ALLOC returned NULL.+ * @retval MLK_ERR_RNG_FAIL MLK_CONFIG_KEYGEN_PCT enabled and random+ * number generation failed within the PCT.+ */+MLK_EXTERNAL_API+MLK_MUST_CHECK_RETURN_VALUE+int mlk_kem_keypair_derand(uint8_t pk[MLKEM_INDCCA_PUBLICKEYBYTES],+ uint8_t sk[MLKEM_INDCCA_SECRETKEYBYTES],+ const uint8_t coins[2 * MLKEM_SYMBYTES],+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context)+__contract__(+ requires(memory_no_alias(pk, MLKEM_INDCCA_PUBLICKEYBYTES))+ requires(memory_no_alias(sk, MLKEM_INDCCA_SECRETKEYBYTES))+ requires(memory_no_alias(coins, 2 * MLKEM_SYMBYTES))+ assigns(memory_slice(pk, MLKEM_INDCCA_PUBLICKEYBYTES))+ assigns(memory_slice(sk, MLKEM_INDCCA_SECRETKEYBYTES))+ ensures(return_value == 0 || return_value == MLK_ERR_PCT_FAIL ||+ return_value == MLK_ERR_OUT_OF_MEMORY ||+ return_value == MLK_ERR_RNG_FAIL)+ /* Output buffers on error, per API-CONVENTIONS.md */+ ensures(return_value != 0 ==>+ array_unchanged_or_zeroized_u8(pk, MLKEM_INDCCA_PUBLICKEYBYTES))+ ensures(return_value != 0 ==>+ array_unchanged_or_zeroized_u8(sk, MLKEM_INDCCA_SECRETKEYBYTES))+);++#if !defined(MLK_CONFIG_NO_RANDOMIZED_API)+/**+ * Generate a public/private keypair for the ML-KEM key encapsulation mechanism.+ *+ * @spec{Implements @[FIPS203, Algorithm 19, ML-KEM.KeyGen].}+ *+ * @param[out] pk Output public key (an already allocated array of+ * MLKEM_INDCCA_PUBLICKEYBYTES bytes).+ * @param[out] sk Output private key (an already allocated array of+ * MLKEM_INDCCA_SECRETKEYBYTES bytes).+ * @param context Application context. Only present when+ * MLK_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLK_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLK_ERR_OUT_OF_MEMORY MLK_CONFIG_CUSTOM_ALLOC_FREE was used and+ * MLK_CUSTOM_ALLOC returned NULL.+ * @retval MLK_ERR_RNG_FAIL Random number generation failed.+ * @retval MLK_ERR_PCT_FAIL MLK_CONFIG_KEYGEN_PCT enabled and PCT failed.+ */+MLK_EXTERNAL_API+MLK_MUST_CHECK_RETURN_VALUE+int mlk_kem_keypair(uint8_t pk[MLKEM_INDCCA_PUBLICKEYBYTES],+ uint8_t sk[MLKEM_INDCCA_SECRETKEYBYTES],+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context)+__contract__(+ requires(memory_no_alias(pk, MLKEM_INDCCA_PUBLICKEYBYTES))+ requires(memory_no_alias(sk, MLKEM_INDCCA_SECRETKEYBYTES))+ assigns(memory_slice(pk, MLKEM_INDCCA_PUBLICKEYBYTES))+ assigns(memory_slice(sk, MLKEM_INDCCA_SECRETKEYBYTES))+ ensures(return_value == 0 || return_value == MLK_ERR_PCT_FAIL ||+ return_value == MLK_ERR_OUT_OF_MEMORY ||+ return_value == MLK_ERR_RNG_FAIL)+ /* Output buffers on error, per API-CONVENTIONS.md */+ ensures(return_value != 0 ==>+ array_unchanged_or_zeroized_u8(pk, MLKEM_INDCCA_PUBLICKEYBYTES))+ ensures(return_value != 0 ==>+ array_unchanged_or_zeroized_u8(sk, MLKEM_INDCCA_SECRETKEYBYTES))+);+#endif /* !MLK_CONFIG_NO_RANDOMIZED_API */+#endif /* !MLK_CONFIG_NO_KEYPAIR_API */++/**+ * Generate ciphertext and shared secret for a given public key.+ *+ * @spec{Implements @[FIPS203, Algorithm 17, ML-KEM.Encaps_Internal].}+ *+ * @param[out] ct Output ciphertext (an already allocated array of+ * MLKEM_INDCCA_CIPHERTEXTBYTES bytes).+ * @param[out] ss Output shared secret (an already allocated array of+ * MLKEM_SSBYTES bytes).+ * @param[in] pk Input public key (an already allocated array of+ * MLKEM_INDCCA_PUBLICKEYBYTES bytes).+ * @param[in] coins Input randomness (an already allocated array filled+ * with MLKEM_SYMBYTES random bytes).+ * @param context Application context. Only present when+ * MLK_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLK_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLK_ERR_INVALID_PK The 'modulus check' @[FIPS203, Section 7.2]+ * for the public key failed.+ * @retval MLK_ERR_OUT_OF_MEMORY MLK_CONFIG_CUSTOM_ALLOC_FREE was used and+ * MLK_CUSTOM_ALLOC returned NULL.+ */+#if !defined(MLK_CONFIG_NO_ENCAPS_API)+MLK_EXTERNAL_API+MLK_MUST_CHECK_RETURN_VALUE+int mlk_kem_enc_derand(uint8_t ct[MLKEM_INDCCA_CIPHERTEXTBYTES],+ uint8_t ss[MLKEM_SSBYTES],+ const uint8_t pk[MLKEM_INDCCA_PUBLICKEYBYTES],+ const uint8_t coins[MLKEM_SYMBYTES],+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context)+__contract__(+ requires(memory_no_alias(ct, MLKEM_INDCCA_CIPHERTEXTBYTES))+ requires(memory_no_alias(ss, MLKEM_SSBYTES))+ requires(memory_no_alias(pk, MLKEM_INDCCA_PUBLICKEYBYTES))+ requires(memory_no_alias(coins, MLKEM_SYMBYTES))+ assigns(memory_slice(ct, MLKEM_INDCCA_CIPHERTEXTBYTES))+ assigns(memory_slice(ss, MLKEM_SSBYTES))+ ensures(return_value == 0 || return_value == MLK_ERR_INVALID_PK ||+ return_value == MLK_ERR_OUT_OF_MEMORY)+ /* Output buffers on error, per API-CONVENTIONS.md */+ ensures(return_value != 0 ==>+ array_unchanged_or_zeroized_u8(ct, MLKEM_INDCCA_CIPHERTEXTBYTES))+ ensures(return_value != 0 ==>+ array_unchanged_or_zeroized_u8(ss, MLKEM_SSBYTES))+);++#if !defined(MLK_CONFIG_NO_RANDOMIZED_API)+/**+ * Generate ciphertext and shared secret for a given public key.+ *+ * @spec{Implements @[FIPS203, Algorithm 20, ML-KEM.Encaps].}+ *+ * @param[out] ct Output ciphertext (an already allocated array of+ * MLKEM_INDCCA_CIPHERTEXTBYTES bytes).+ * @param[out] ss Output shared secret (an already allocated array of+ * MLKEM_SSBYTES bytes).+ * @param[in] pk Input public key (an already allocated array of+ * MLKEM_INDCCA_PUBLICKEYBYTES bytes).+ * @param context Application context. Only present when+ * MLK_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLK_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLK_ERR_OUT_OF_MEMORY MLK_CONFIG_CUSTOM_ALLOC_FREE was used and+ * MLK_CUSTOM_ALLOC returned NULL.+ * @retval MLK_ERR_RNG_FAIL Random number generation failed.+ * @retval MLK_ERR_INVALID_PK The 'modulus check' @[FIPS203, Section 7.2]+ * for the public key failed.+ */+MLK_EXTERNAL_API+MLK_MUST_CHECK_RETURN_VALUE+int mlk_kem_enc(uint8_t ct[MLKEM_INDCCA_CIPHERTEXTBYTES],+ uint8_t ss[MLKEM_SSBYTES],+ const uint8_t pk[MLKEM_INDCCA_PUBLICKEYBYTES],+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context)+__contract__(+ requires(memory_no_alias(ct, MLKEM_INDCCA_CIPHERTEXTBYTES))+ requires(memory_no_alias(ss, MLKEM_SSBYTES))+ requires(memory_no_alias(pk, MLKEM_INDCCA_PUBLICKEYBYTES))+ assigns(memory_slice(ct, MLKEM_INDCCA_CIPHERTEXTBYTES))+ assigns(memory_slice(ss, MLKEM_SSBYTES))+ ensures(return_value == 0 || return_value == MLK_ERR_INVALID_PK ||+ return_value == MLK_ERR_OUT_OF_MEMORY ||+ return_value == MLK_ERR_RNG_FAIL)+ /* Output buffers on error, per API-CONVENTIONS.md */+ ensures(return_value != 0 ==>+ array_unchanged_or_zeroized_u8(ct, MLKEM_INDCCA_CIPHERTEXTBYTES))+ ensures(return_value != 0 ==>+ array_unchanged_or_zeroized_u8(ss, MLKEM_SSBYTES))+);+#endif /* !MLK_CONFIG_NO_RANDOMIZED_API */+#endif /* !MLK_CONFIG_NO_ENCAPS_API */++/**+ * Generate shared secret for a given ciphertext and private key.+ *+ * @spec{Implements @[FIPS203, Algorithm 21, ML-KEM.Decaps].}+ *+ * @param[out] ss Output shared secret (an already allocated array of+ * MLKEM_SSBYTES bytes).+ * @param[in] ct Input ciphertext (an already allocated array of+ * MLKEM_INDCCA_CIPHERTEXTBYTES bytes).+ * @param[in] sk Input private key (an already allocated array of+ * MLKEM_INDCCA_SECRETKEYBYTES bytes).+ * @param context Application context. Only present when+ * MLK_CONFIG_CONTEXT_PARAMETER is defined; type set by+ * MLK_CONFIG_CONTEXT_PARAMETER_TYPE.+ *+ * @retval 0 Success.+ * @retval MLK_ERR_INVALID_SK The 'hash check' @[FIPS203, Section 7.3]+ * for the secret key failed.+ * @retval MLK_ERR_OUT_OF_MEMORY MLK_CONFIG_CUSTOM_ALLOC_FREE was used and+ * MLK_CUSTOM_ALLOC returned NULL.+ */+#if !defined(MLK_CONFIG_NO_DECAPS_API)+MLK_EXTERNAL_API+MLK_MUST_CHECK_RETURN_VALUE+int mlk_kem_dec(uint8_t ss[MLKEM_SSBYTES],+ const uint8_t ct[MLKEM_INDCCA_CIPHERTEXTBYTES],+ const uint8_t sk[MLKEM_INDCCA_SECRETKEYBYTES],+ MLK_CONFIG_CONTEXT_PARAMETER_TYPE context)+__contract__(+ requires(memory_no_alias(ss, MLKEM_SSBYTES))+ requires(memory_no_alias(ct, MLKEM_INDCCA_CIPHERTEXTBYTES))+ requires(memory_no_alias(sk, MLKEM_INDCCA_SECRETKEYBYTES))+ assigns(memory_slice(ss, MLKEM_SSBYTES))+ ensures(return_value == 0 || return_value == MLK_ERR_INVALID_SK ||+ return_value == MLK_ERR_OUT_OF_MEMORY)+ /* Output buffers on error, per API-CONVENTIONS.md */+ ensures(return_value != 0 ==>+ array_unchanged_or_zeroized_u8(ss, MLKEM_SSBYTES))+);+#endif /* !MLK_CONFIG_NO_DECAPS_API */++#endif /* !MLK_KEM_H */
@@ -0,0 +1,16 @@+[//]: # (SPDX-License-Identifier: CC-BY-4.0)++# AArch64 backend (little endian)++This directory contains a native backend for little endian AArch64 systems. It is derived from [^NeonNTT] [^SLOTHY_Paper].++The code in this directory is auto-generated from the 'clean' assembly in [dev/aarch64_clean](../../../../dev/aarch64_clean)+in a two-step fashion: First, it is superoptimized using the [SLOTHY](https://github.com/slothy-optimizer/slothy) superoptimizer,+giving the assembly in [dev/aarch64_opt](../../../../dev/aarch64_opt). Then, it is stripped of remaining register aliases, macros+and most preprocessor directives by [`scripts/simpasm`](../../../../scripts/simpasm).++If you want to understand how the assembly works, and/or make changes to it, consult [dev/](../../../../dev).++<!--- bibliography --->+[^NeonNTT]: Becker, Hwang, Kannwischer, Yang, Yang: Neon NTT: Faster Dilithium, Kyber, and Saber on Cortex-A72 and Apple M1, [https://eprint.iacr.org/2021/986](https://eprint.iacr.org/2021/986)+[^SLOTHY_Paper]: Abdulrahman, Becker, Kannwischer, Klein: Fast and Clean: Auditable high-performance assembly via constraint solving, [https://eprint.iacr.org/2022/1303](https://eprint.iacr.org/2022/1303)
@@ -0,0 +1,166 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#ifndef MLK_NATIVE_AARCH64_META_H+#define MLK_NATIVE_AARCH64_META_H++/* Set of primitives that this backend replaces */+#define MLK_USE_NATIVE_NTT+#define MLK_USE_NATIVE_INTT+#define MLK_USE_NATIVE_POLY_REDUCE+#define MLK_USE_NATIVE_POLY_TOMONT+#define MLK_USE_NATIVE_POLY_MULCACHE_COMPUTE+#define MLK_USE_NATIVE_POLYVEC_BASEMUL_ACC_MONTGOMERY_CACHED+#define MLK_USE_NATIVE_POLY_TOBYTES+#define MLK_USE_NATIVE_REJ_UNIFORM++/* Identifier for this backend so that source and assembly files+ * in the build can be appropriately guarded. */+#define MLK_ARITH_BACKEND_AARCH64+++#if !defined(__ASSEMBLER__)+#include "../api.h"+#include "src/arith_native_aarch64.h"++MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_ntt_native(int16_t data[MLKEM_N])+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_AARCH64_NEON))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }+ mlk_ntt_aarch64_asm(data, mlk_aarch64_ntt_zetas_layer12345,+ mlk_aarch64_ntt_zetas_layer67);+ return MLK_NATIVE_FUNC_SUCCESS;+}++#if !defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API)+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_intt_native(int16_t data[MLKEM_N])+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_AARCH64_NEON))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }+ mlk_intt_aarch64_asm(data, mlk_aarch64_invntt_zetas_layer12345,+ mlk_aarch64_invntt_zetas_layer67);+ return MLK_NATIVE_FUNC_SUCCESS;+}+#endif /* !MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API */++MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_reduce_native(int16_t data[MLKEM_N])+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_AARCH64_NEON))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }+ mlk_poly_reduce_aarch64_asm(data);+ return MLK_NATIVE_FUNC_SUCCESS;+}++#if !defined(MLK_CONFIG_NO_KEYPAIR_API)+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_tomont_native(int16_t data[MLKEM_N])+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_AARCH64_NEON))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }+ mlk_poly_tomont_aarch64_asm(data);+ return MLK_NATIVE_FUNC_SUCCESS;+}+#endif /* !MLK_CONFIG_NO_KEYPAIR_API */++MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_mulcache_compute_native(int16_t x[MLKEM_N / 2],+ const int16_t y[MLKEM_N])+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_AARCH64_NEON))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }+ mlk_poly_mulcache_compute_aarch64_asm(+ x, y, mlk_aarch64_zetas_mulcache_native,+ mlk_aarch64_zetas_mulcache_twisted_native);+ return MLK_NATIVE_FUNC_SUCCESS;+}++#if defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_K == 2+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_polyvec_basemul_acc_montgomery_cached_k2_native(+ int16_t r[MLKEM_N], const int16_t a[2 * MLKEM_N],+ const int16_t b[2 * MLKEM_N], const int16_t b_cache[2 * (MLKEM_N / 2)])+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_AARCH64_NEON))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }+ mlk_polyvec_basemul_acc_montgomery_cached_k2_aarch64_asm(r, a, b, b_cache);+ return MLK_NATIVE_FUNC_SUCCESS;+}+#endif /* MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 2 */++#if defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_K == 3+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_polyvec_basemul_acc_montgomery_cached_k3_native(+ int16_t r[MLKEM_N], const int16_t a[3 * MLKEM_N],+ const int16_t b[3 * MLKEM_N], const int16_t b_cache[3 * (MLKEM_N / 2)])+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_AARCH64_NEON))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }+ mlk_polyvec_basemul_acc_montgomery_cached_k3_aarch64_asm(r, a, b, b_cache);+ return MLK_NATIVE_FUNC_SUCCESS;+}+#endif /* MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 3 */++#if defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_K == 4+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_polyvec_basemul_acc_montgomery_cached_k4_native(+ int16_t r[MLKEM_N], const int16_t a[4 * MLKEM_N],+ const int16_t b[4 * MLKEM_N], const int16_t b_cache[4 * (MLKEM_N / 2)])+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_AARCH64_NEON))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }+ mlk_polyvec_basemul_acc_montgomery_cached_k4_aarch64_asm(r, a, b, b_cache);+ return MLK_NATIVE_FUNC_SUCCESS;+}+#endif /* MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 4 */++#if !defined(MLK_CONFIG_NO_KEYPAIR_API) || !defined(MLK_CONFIG_NO_ENCAPS_API)+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_tobytes_native(uint8_t r[MLKEM_POLYBYTES],+ const int16_t a[MLKEM_N])+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_AARCH64_NEON))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }+ mlk_poly_tobytes_aarch64_asm(r, a);+ return MLK_NATIVE_FUNC_SUCCESS;+}+#endif /* !MLK_CONFIG_NO_KEYPAIR_API || !MLK_CONFIG_NO_ENCAPS_API */++MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_rej_uniform_native(int16_t *r, unsigned len,+ const uint8_t *buf,+ unsigned buflen)+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_AARCH64_NEON) || len != MLKEM_N ||+ buflen % 24 != 0)+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }+ return (int)mlk_rej_uniform_aarch64_asm(r, buf, buflen,+ mlk_rej_uniform_table);+}+#endif /* !__ASSEMBLER__ */++#endif /* !MLK_NATIVE_AARCH64_META_H */
@@ -0,0 +1,184 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*+ * WARNING: This file is auto-generated from scripts/autogen+ * in the mlkem-native repository.+ * Do not modify it directly.+ */++#include "../../../common.h"++#if defined(MLK_ARITH_BACKEND_AARCH64) && \+ !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED)++#include "arith_native_aarch64.h"++/*+ * Table of zeta values used in the AArch64 forward NTT+ * See autogen for details.+ */+MLK_ALIGN MLK_INTERNAL_DATA_DEFINITION const int16_t+ mlk_aarch64_ntt_zetas_layer12345[80] = {+ -1600, -15749, -749, -7373, -40, -394, -687, -6762, 630, 6201,+ -1432, -14095, 848, 8347, 0, 0, 1062, 10453, 296, 2914,+ -882, -8682, 0, 0, -1410, -13879, 1339, 13180, 1476, 14529,+ 0, 0, 193, 1900, -283, -2786, 56, 551, 0, 0,+ 797, 7845, -1089, -10719, 1333, 13121, 0, 0, -543, -5345,+ 1426, 14036, -1235, -12156, 0, 0, -69, -679, 535, 5266,+ -447, -4400, 0, 0, 569, 5601, -936, -9213, -450, -4429,+ 0, 0, -1583, -15582, -1355, -13338, 821, 8081, 0, 0,+};++MLK_ALIGN MLK_INTERNAL_DATA_DEFINITION const int16_t+ mlk_aarch64_ntt_zetas_layer67[384] = {+ 289, 289, 331, 331, -76, -76, -1573, -1573, 2845,+ 2845, 3258, 3258, -748, -748, -15483, -15483, 17, 17,+ 583, 583, 1637, 1637, -1041, -1041, 167, 167, 5739,+ 5739, 16113, 16113, -10247, -10247, -568, -568, -680, -680,+ 723, 723, 1100, 1100, -5591, -5591, -6693, -6693, 7117,+ 7117, 10828, 10828, 1197, 1197, -1025, -1025, -1052, -1052,+ -1274, -1274, 11782, 11782, -10089, -10089, -10355, -10355, -12540,+ -12540, 1409, 1409, -48, -48, 756, 756, -314, -314,+ 13869, 13869, -472, -472, 7441, 7441, -3091, -3091, -667,+ -667, 233, 233, -1173, -1173, -279, -279, -6565, -6565,+ 2293, 2293, -11546, -11546, -2746, -2746, 650, 650, -1352,+ -1352, -816, -816, 632, 632, 6398, 6398, -13308, -13308,+ -8032, -8032, 6221, 6221, -1626, -1626, -540, -540, -1482,+ -1482, 1461, 1461, -16005, -16005, -5315, -5315, -14588, -14588,+ 14381, 14381, 1651, 1651, -1540, -1540, 952, 952, -642,+ -642, 16251, 16251, -15159, -15159, 9371, 9371, -6319, -6319,+ -464, -464, 33, 33, 1320, 1320, -1414, -1414, -4567,+ -4567, 325, 325, 12993, 12993, -13918, -13918, 939, 939,+ -892, -892, 733, 733, 268, 268, 9243, 9243, -8780,+ -8780, 7215, 7215, 2638, 2638, -1021, -1021, -941, -941,+ -992, -992, 641, 641, -10050, -10050, -9262, -9262, -9764,+ -9764, 6309, 6309, -1010, -1010, 1435, 1435, 807, 807,+ 452, 452, -9942, -9942, 14125, 14125, 7943, 7943, 4449,+ 4449, 1584, 1584, -1292, -1292, 375, 375, -1239, -1239,+ 15592, 15592, -12717, -12717, 3691, 3691, -12196, -12196, -1031,+ -1031, -109, -109, -780, -780, 1645, 1645, -10148, -10148,+ -1073, -1073, -7678, -7678, 16192, 16192, 1438, 1438, -461,+ -461, 1534, 1534, -927, -927, 14155, 14155, -4538, -4538,+ 15099, 15099, -9125, -9125, 1063, 1063, -556, -556, -1230,+ -1230, -863, -863, 10463, 10463, -5473, -5473, -12107, -12107,+ -8495, -8495, 319, 319, 757, 757, 561, 561, -735,+ -735, 3140, 3140, 7451, 7451, 5522, 5522, -7235, -7235,+ -682, -682, -712, -712, 1481, 1481, 648, 648, -6713,+ -6713, -7008, -7008, 14578, 14578, 6378, 6378, -525, -525,+ 403, 403, 1143, 1143, -554, -554, -5168, -5168, 3967,+ 3967, 11251, 11251, -5453, -5453, 1092, 1092, 1026, 1026,+ -1179, -1179, 886, 886, 10749, 10749, 10099, 10099, -11605,+ -11605, 8721, 8721, -855, -855, -219, -219, 1227, 1227,+ 910, 910, -8416, -8416, -2156, -2156, 12078, 12078, 8957,+ 8957, -1607, -1607, -1455, -1455, -1219, -1219, 885, 885,+ -15818, -15818, -14322, -14322, -11999, -11999, 8711, 8711, 1212,+ 1212, 1029, 1029, -394, -394, -1175, -1175, 11930, 11930,+ 10129, 10129, -3878, -3878, -11566, -11566,+};++#if !defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API)+MLK_ALIGN MLK_INTERNAL_DATA_DEFINITION const int16_t+ mlk_aarch64_invntt_zetas_layer12345[80] = {+ 1583, 15582, -821, -8081, 1355, 13338, 0, 0, -569,+ -5601, 450, 4429, 936, 9213, 0, 0, 69, 679,+ 447, 4400, -535, -5266, 0, 0, 543, 5345, 1235,+ 12156, -1426, -14036, 0, 0, -797, -7845, -1333, -13121,+ 1089, 10719, 0, 0, -193, -1900, -56, -551, 283,+ 2786, 0, 0, 1410, 13879, -1476, -14529, -1339, -13180,+ 0, 0, -1062, -10453, 882, 8682, -296, -2914, 0,+ 0, 1600, 15749, 40, 394, 749, 7373, -848, -8347,+ 1432, 14095, -630, -6201, 687, 6762, 0, 0,+};++MLK_ALIGN MLK_INTERNAL_DATA_DEFINITION const int16_t+ mlk_aarch64_invntt_zetas_layer67[384] = {+ -910, -910, -1227, -1227, 219, 219, 855, 855, -8957,+ -8957, -12078, -12078, 2156, 2156, 8416, 8416, 1175, 1175,+ 394, 394, -1029, -1029, -1212, -1212, 11566, 11566, 3878,+ 3878, -10129, -10129, -11930, -11930, -885, -885, 1219, 1219,+ 1455, 1455, 1607, 1607, -8711, -8711, 11999, 11999, 14322,+ 14322, 15818, 15818, -648, -648, -1481, -1481, 712, 712,+ 682, 682, -6378, -6378, -14578, -14578, 7008, 7008, 6713,+ 6713, -886, -886, 1179, 1179, -1026, -1026, -1092, -1092,+ -8721, -8721, 11605, 11605, -10099, -10099, -10749, -10749, 554,+ 554, -1143, -1143, -403, -403, 525, 525, 5453, 5453,+ -11251, -11251, -3967, -3967, 5168, 5168, 927, 927, -1534,+ -1534, 461, 461, -1438, -1438, 9125, 9125, -15099, -15099,+ 4538, 4538, -14155, -14155, 735, 735, -561, -561, -757,+ -757, -319, -319, 7235, 7235, -5522, -5522, -7451, -7451,+ -3140, -3140, 863, 863, 1230, 1230, 556, 556, -1063,+ -1063, 8495, 8495, 12107, 12107, 5473, 5473, -10463, -10463,+ -452, -452, -807, -807, -1435, -1435, 1010, 1010, -4449,+ -4449, -7943, -7943, -14125, -14125, 9942, 9942, -1645, -1645,+ 780, 780, 109, 109, 1031, 1031, -16192, -16192, 7678,+ 7678, 1073, 1073, 10148, 10148, 1239, 1239, -375, -375,+ 1292, 1292, -1584, -1584, 12196, 12196, -3691, -3691, 12717,+ 12717, -15592, -15592, 1414, 1414, -1320, -1320, -33, -33,+ 464, 464, 13918, 13918, -12993, -12993, -325, -325, 4567,+ 4567, -641, -641, 992, 992, 941, 941, 1021, 1021,+ -6309, -6309, 9764, 9764, 9262, 9262, 10050, 10050, -268,+ -268, -733, -733, 892, 892, -939, -939, -2638, -2638,+ -7215, -7215, 8780, 8780, -9243, -9243, -632, -632, 816,+ 816, 1352, 1352, -650, -650, -6221, -6221, 8032, 8032,+ 13308, 13308, -6398, -6398, 642, 642, -952, -952, 1540,+ 1540, -1651, -1651, 6319, 6319, -9371, -9371, 15159, 15159,+ -16251, -16251, -1461, -1461, 1482, 1482, 540, 540, 1626,+ 1626, -14381, -14381, 14588, 14588, 5315, 5315, 16005, 16005,+ 1274, 1274, 1052, 1052, 1025, 1025, -1197, -1197, 12540,+ 12540, 10355, 10355, 10089, 10089, -11782, -11782, 279, 279,+ 1173, 1173, -233, -233, 667, 667, 2746, 2746, 11546,+ 11546, -2293, -2293, 6565, 6565, 314, 314, -756, -756,+ 48, 48, -1409, -1409, 3091, 3091, -7441, -7441, 472,+ 472, -13869, -13869, 1573, 1573, 76, 76, -331, -331,+ -289, -289, 15483, 15483, 748, 748, -3258, -3258, -2845,+ -2845, -1100, -1100, -723, -723, 680, 680, 568, 568,+ -10828, -10828, -7117, -7117, 6693, 6693, 5591, 5591, 1041,+ 1041, -1637, -1637, -583, -583, -17, -17, 10247, 10247,+ -16113, -16113, -5739, -5739, -167, -167,+};+#endif /* !MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API */++MLK_ALIGN MLK_INTERNAL_DATA_DEFINITION const int16_t+ mlk_aarch64_zetas_mulcache_native[128] = {+ 17, -17, -568, 568, 583, -583, -680, 680, 1637, -1637,+ 723, -723, -1041, 1041, 1100, -1100, 1409, -1409, -667, 667,+ -48, 48, 233, -233, 756, -756, -1173, 1173, -314, 314,+ -279, 279, -1626, 1626, 1651, -1651, -540, 540, -1540, 1540,+ -1482, 1482, 952, -952, 1461, -1461, -642, 642, 939, -939,+ -1021, 1021, -892, 892, -941, 941, 733, -733, -992, 992,+ 268, -268, 641, -641, 1584, -1584, -1031, 1031, -1292, 1292,+ -109, 109, 375, -375, -780, 780, -1239, 1239, 1645, -1645,+ 1063, -1063, 319, -319, -556, 556, 757, -757, -1230, 1230,+ 561, -561, -863, 863, -735, 735, -525, 525, 1092, -1092,+ 403, -403, 1026, -1026, 1143, -1143, -1179, 1179, -554, 554,+ 886, -886, -1607, 1607, 1212, -1212, -1455, 1455, 1029, -1029,+ -1219, 1219, -394, 394, 885, -885, -1175, 1175,+};++MLK_ALIGN MLK_INTERNAL_DATA_DEFINITION const int16_t+ mlk_aarch64_zetas_mulcache_twisted_native[128] = {+ 167, -167, -5591, 5591, 5739, -5739, -6693, 6693, 16113,+ -16113, 7117, -7117, -10247, 10247, 10828, -10828, 13869, -13869,+ -6565, 6565, -472, 472, 2293, -2293, 7441, -7441, -11546,+ 11546, -3091, 3091, -2746, 2746, -16005, 16005, 16251, -16251,+ -5315, 5315, -15159, 15159, -14588, 14588, 9371, -9371, 14381,+ -14381, -6319, 6319, 9243, -9243, -10050, 10050, -8780, 8780,+ -9262, 9262, 7215, -7215, -9764, 9764, 2638, -2638, 6309,+ -6309, 15592, -15592, -10148, 10148, -12717, 12717, -1073, 1073,+ 3691, -3691, -7678, 7678, -12196, 12196, 16192, -16192, 10463,+ -10463, 3140, -3140, -5473, 5473, 7451, -7451, -12107, 12107,+ 5522, -5522, -8495, 8495, -7235, 7235, -5168, 5168, 10749,+ -10749, 3967, -3967, 10099, -10099, 11251, -11251, -11605, 11605,+ -5453, 5453, 8721, -8721, -15818, 15818, 11930, -11930, -14322,+ 14322, 10129, -10129, -11999, 11999, -3878, 3878, 8711, -8711,+ -11566, 11566,+};++#else /* MLK_ARITH_BACKEND_AARCH64 && !MLK_CONFIG_MULTILEVEL_NO_SHARED */++MLK_EMPTY_CU(aarch64_zetas)++#endif /* !(MLK_ARITH_BACKEND_AARCH64 && !MLK_CONFIG_MULTILEVEL_NO_SHARED) */
@@ -0,0 +1,184 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+#ifndef MLK_NATIVE_AARCH64_SRC_ARITH_NATIVE_AARCH64_H+#define MLK_NATIVE_AARCH64_SRC_ARITH_NATIVE_AARCH64_H++#include "../../../cbmc.h"+#include "../../../common.h"++#define mlk_aarch64_ntt_zetas_layer12345 \+ MLK_NAMESPACE(aarch64_ntt_zetas_layer12345)+#define mlk_aarch64_ntt_zetas_layer67 MLK_NAMESPACE(aarch64_ntt_zetas_layer67)+#define mlk_aarch64_invntt_zetas_layer12345 \+ MLK_NAMESPACE(aarch64_invntt_zetas_layer12345)+#define mlk_aarch64_invntt_zetas_layer67 \+ MLK_NAMESPACE(aarch64_invntt_zetas_layer67)+#define mlk_aarch64_zetas_mulcache_native \+ MLK_NAMESPACE(aarch64_zetas_mulcache_native)+#define mlk_aarch64_zetas_mulcache_twisted_native \+ MLK_NAMESPACE(aarch64_zetas_mulcache_twisted_native)+#define mlk_rej_uniform_table MLK_NAMESPACE(rej_uniform_table)++MLK_INTERNAL_DATA_DECLARATION const int16_t+ mlk_aarch64_ntt_zetas_layer12345[80];+MLK_INTERNAL_DATA_DECLARATION const int16_t mlk_aarch64_ntt_zetas_layer67[384];+MLK_INTERNAL_DATA_DECLARATION const int16_t+ mlk_aarch64_invntt_zetas_layer12345[80];+MLK_INTERNAL_DATA_DECLARATION const int16_t+ mlk_aarch64_invntt_zetas_layer67[384];+MLK_INTERNAL_DATA_DECLARATION const int16_t+ mlk_aarch64_zetas_mulcache_native[128];+MLK_INTERNAL_DATA_DECLARATION const int16_t+ mlk_aarch64_zetas_mulcache_twisted_native[128];+MLK_INTERNAL_DATA_DECLARATION const uint8_t mlk_rej_uniform_table[4096];++#define mlk_ntt_aarch64_asm MLK_NAMESPACE(ntt_aarch64_asm)+void mlk_ntt_aarch64_asm(int16_t p[256], const int16_t twiddles12345[80],+ const int16_t twiddles56[384])+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/aarch64/proofs/mlkem_ntt_aarch64_asm.ml */+__contract__(+ requires(memory_no_alias(p, sizeof(int16_t) * MLKEM_N))+ requires(array_abs_bound(p, 0, MLKEM_N, 8192))+ requires(twiddles12345 == mlk_aarch64_ntt_zetas_layer12345)+ requires(twiddles56 == mlk_aarch64_ntt_zetas_layer67)+ assigns(memory_slice(p, sizeof(int16_t) * MLKEM_N))+ /* check-magic: off */+ ensures(array_abs_bound(p, 0, MLKEM_N, 23595))+ /* check-magic: on */+);++#define mlk_intt_aarch64_asm MLK_NAMESPACE(intt_aarch64_asm)+void mlk_intt_aarch64_asm(int16_t p[256], const int16_t twiddles12345[80],+ const int16_t twiddles56[384])+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/aarch64/proofs/mlkem_intt_aarch64_asm.ml */+__contract__(+ requires(memory_no_alias(p, sizeof(int16_t) * MLKEM_N))+ requires(twiddles12345 == mlk_aarch64_invntt_zetas_layer12345)+ requires(twiddles56 == mlk_aarch64_invntt_zetas_layer67)+ assigns(memory_slice(p, sizeof(int16_t) * MLKEM_N))+ /* check-magic: off */+ ensures(array_abs_bound(p, 0, MLKEM_N, 26625))+ /* check-magic: on */+);++#define mlk_poly_reduce_aarch64_asm MLK_NAMESPACE(poly_reduce_aarch64_asm)+void mlk_poly_reduce_aarch64_asm(int16_t p[256])+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/aarch64/proofs/mlkem_poly_reduce_aarch64_asm.ml */+__contract__(+ requires(memory_no_alias(p, sizeof(int16_t) * MLKEM_N))+ assigns(memory_slice(p, sizeof(int16_t) * MLKEM_N))+ ensures(array_bound(p, 0, MLKEM_N, 0, MLKEM_Q))+);++#define mlk_poly_tomont_aarch64_asm MLK_NAMESPACE(poly_tomont_aarch64_asm)+void mlk_poly_tomont_aarch64_asm(int16_t p[256])+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/aarch64/proofs/mlkem_poly_tomont_aarch64_asm.ml */+__contract__(+ requires(memory_no_alias(p, sizeof(int16_t) * MLKEM_N))+ assigns(memory_slice(p, sizeof(int16_t) * MLKEM_N))+ ensures(array_abs_bound(p, 0, MLKEM_N, MLKEM_Q))+);++#define mlk_poly_mulcache_compute_aarch64_asm \+ MLK_NAMESPACE(poly_mulcache_compute_aarch64_asm)+void mlk_poly_mulcache_compute_aarch64_asm(int16_t cache[128],+ const int16_t mlk_poly[256],+ const int16_t zetas[128],+ const int16_t zetas_twisted[128])+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/aarch64/proofs/mlkem_poly_mulcache_compute_aarch64_asm.ml+ */+__contract__(+ requires(memory_no_alias(cache, sizeof(int16_t) * (MLKEM_N / 2)))+ requires(memory_no_alias(mlk_poly, sizeof(int16_t) * MLKEM_N))+ requires(zetas == mlk_aarch64_zetas_mulcache_native)+ requires(zetas_twisted == mlk_aarch64_zetas_mulcache_twisted_native)+ assigns(memory_slice(cache, sizeof(int16_t) * (MLKEM_N / 2)))+ ensures(array_abs_bound(cache, 0, MLKEM_N/2, MLKEM_Q))+);++#define mlk_poly_tobytes_aarch64_asm MLK_NAMESPACE(poly_tobytes_aarch64_asm)+void mlk_poly_tobytes_aarch64_asm(uint8_t r[384], const int16_t a[256])+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/aarch64/proofs/mlkem_poly_tobytes_aarch64_asm.ml */+__contract__(+ requires(memory_no_alias(r, MLKEM_POLYBYTES))+ requires(memory_no_alias(a, sizeof(int16_t) * MLKEM_N))+ requires(array_bound(a, 0, MLKEM_N, 0, MLKEM_UINT12_LIMIT))+ assigns(memory_slice(r, MLKEM_POLYBYTES))+);++#define mlk_polyvec_basemul_acc_montgomery_cached_k2_aarch64_asm \+ MLK_NAMESPACE(polyvec_basemul_acc_montgomery_cached_k2_aarch64_asm)+void mlk_polyvec_basemul_acc_montgomery_cached_k2_aarch64_asm(+ int16_t r[256], const int16_t a[512], const int16_t b[512],+ const int16_t b_cache[256])+/* This must be kept in sync with the HOL-Light specification in+ * proofs/hol_light/aarch64/proofs/mlkem_polyvec_basemul_acc_montgomery_cached_k2_aarch64_asm.ml.+ */+__contract__(+ requires(memory_no_alias(r, sizeof(int16_t) * MLKEM_N))+ requires(memory_no_alias(a, sizeof(int16_t) * 2 * MLKEM_N))+ requires(memory_no_alias(b, sizeof(int16_t) * 2 * MLKEM_N))+ requires(memory_no_alias(b_cache, sizeof(int16_t) * 2 * (MLKEM_N / 2)))+ requires(array_abs_bound(a, 0, 2 * MLKEM_N, MLKEM_UINT12_LIMIT + 1))+ assigns(memory_slice(r, sizeof(int16_t) * MLKEM_N))+);++#define mlk_polyvec_basemul_acc_montgomery_cached_k3_aarch64_asm \+ MLK_NAMESPACE(polyvec_basemul_acc_montgomery_cached_k3_aarch64_asm)+void mlk_polyvec_basemul_acc_montgomery_cached_k3_aarch64_asm(+ int16_t r[256], const int16_t a[768], const int16_t b[768],+ const int16_t b_cache[384])+/* This must be kept in sync with the HOL-Light specification in+ * proofs/hol_light/aarch64/proofs/mlkem_polyvec_basemul_acc_montgomery_cached_k3_aarch64_asm.ml.+ */+__contract__(+ requires(memory_no_alias(r, sizeof(int16_t) * MLKEM_N))+ requires(memory_no_alias(a, sizeof(int16_t) * 3 * MLKEM_N))+ requires(memory_no_alias(b, sizeof(int16_t) * 3 * MLKEM_N))+ requires(memory_no_alias(b_cache, sizeof(int16_t) * 3 * (MLKEM_N / 2)))+ requires(array_abs_bound(a, 0, 3 * MLKEM_N, MLKEM_UINT12_LIMIT + 1))+ assigns(memory_slice(r, sizeof(int16_t) * MLKEM_N))+);++#define mlk_polyvec_basemul_acc_montgomery_cached_k4_aarch64_asm \+ MLK_NAMESPACE(polyvec_basemul_acc_montgomery_cached_k4_aarch64_asm)+void mlk_polyvec_basemul_acc_montgomery_cached_k4_aarch64_asm(+ int16_t r[256], const int16_t a[1024], const int16_t b[1024],+ const int16_t b_cache[512])+/* This must be kept in sync with the HOL-Light specification in+ * proofs/hol_light/aarch64/proofs/mlkem_polyvec_basemul_acc_montgomery_cached_k4_aarch64_asm.ml.+ */+__contract__(+ requires(memory_no_alias(r, sizeof(int16_t) * MLKEM_N))+ requires(memory_no_alias(a, sizeof(int16_t) * 4 * MLKEM_N))+ requires(memory_no_alias(b, sizeof(int16_t) * 4 * MLKEM_N))+ requires(memory_no_alias(b_cache, sizeof(int16_t) * 4 * (MLKEM_N / 2)))+ requires(array_abs_bound(a, 0, 4 * MLKEM_N, MLKEM_UINT12_LIMIT + 1))+ assigns(memory_slice(r, sizeof(int16_t) * MLKEM_N))+);++#define mlk_rej_uniform_aarch64_asm MLK_NAMESPACE(rej_uniform_aarch64_asm)+MLK_MUST_CHECK_RETURN_VALUE+uint64_t mlk_rej_uniform_aarch64_asm(int16_t r[256], const uint8_t *buf,+ unsigned buflen, const uint8_t table[4096])+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/aarch64/proofs/mlkem_rej_uniform_aarch64_asm.ml. */+__contract__(+ requires(buflen % 24 == 0)+ requires(memory_no_alias(buf, buflen))+ requires(table == mlk_rej_uniform_table)+ requires(memory_no_alias(r, sizeof(int16_t) * MLKEM_N))+ assigns(memory_slice(r, sizeof(int16_t) * MLKEM_N))+ ensures(return_value <= MLKEM_N)+ ensures(array_bound(r, 0, (unsigned) return_value, 0, MLKEM_Q))+);++#endif /* !MLK_NATIVE_AARCH64_SRC_ARITH_NATIVE_AARCH64_H */
@@ -0,0 +1,635 @@+/* Copyright (c) 2022 Arm Limited+ * Copyright (c) 2022 Hanno Becker+ * Copyright (c) 2023 Amin Abdulrahman, Matthias Kannwischer+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [NeonNTT]+ * Neon NTT: Faster Dilithium, Kyber, and Saber on Cortex-A72 and Apple M1+ * Becker, Hwang, Kannwischer, Yang, Yang+ * https://eprint.iacr.org/2021/986+ *+ * - [SLOTHY_Paper]+ * Fast and Clean: Auditable high-performance assembly via constraint solving+ * Abdulrahman, Becker, Kannwischer, Klein+ * https://eprint.iacr.org/2022/1303+ */++/*yaml+ Name: intt_aarch64_asm+ Description: AArch64 ML-KEM inverse NTT following @[NeonNTT] and @[SLOTHY_Paper]+ Signature: void mlk_intt_aarch64_asm(int16_t p[256], const int16_t twiddles12345[80], const int16_t twiddles56[384])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON]+ x0:+ type: buffer+ size_bytes: 512+ permissions: read/write+ c_parameter: int16_t p[256]+ description: Input/output polynomial+ x1:+ type: buffer+ size_bytes: 160+ permissions: read-only+ c_parameter: const int16_t twiddles12345[80]+ description: Twiddle factors for layers 1-5+ x2:+ type: buffer+ size_bytes: 768+ permissions: read-only+ c_parameter: const int16_t twiddles56[384]+ description: Twiddle factors for layers 6-7+ Stack:+ bytes: 64+ description: saving callee-saved Neon registers+*/++#include "../../../common.h"+#if defined(MLK_ARITH_BACKEND_AARCH64) && \+ !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED) && \+ (!defined(MLK_CONFIG_NO_ENCAPS_API) || \+ !defined(MLK_CONFIG_NO_DECAPS_API))++/*+ * WARNING: This file is auto-derived from the mlkem-native source file+ * dev/aarch64_opt/src/mlkem_intt_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLK_ASM_NAMESPACE(intt_aarch64_asm)+MLK_ASM_FN_SYMBOL(intt_aarch64_asm)++ .cfi_startproc+ sub sp, sp, #0x40+ .cfi_adjust_cfa_offset 0x40+ stp d8, d9, [sp]+ .cfi_rel_offset d8, 0x0+ .cfi_rel_offset d9, 0x8+ stp d10, d11, [sp, #0x10]+ .cfi_rel_offset d10, 0x10+ .cfi_rel_offset d11, 0x18+ stp d12, d13, [sp, #0x20]+ .cfi_rel_offset d12, 0x20+ .cfi_rel_offset d13, 0x28+ stp d14, d15, [sp, #0x30]+ .cfi_rel_offset d14, 0x30+ .cfi_rel_offset d15, 0x38+ mov w5, #0xd01 // =3329+ mov v7.h[0], w5+ mov w5, #0x4ebf // =20159+ mov v7.h[1], w5+ mov w5, #0x200 // =512+ dup v29.8h, w5+ mov w5, #0x13b0 // =5040+ dup v30.8h, w5+ mov x3, x0+ mov x4, #0x8 // =8+ ldr q13, [x3, #0x20]+ ldr q8, [x3, #0x30]+ ldr q6, [x3]+ ldr q16, [x3, #0x10]+ ldr q4, [x3, #0x50]+ ldr q11, [x3, #0x40]+ ldr q3, [x3, #0x70]+ trn1 v23.4s, v13.4s, v8.4s+ ldr q0, [x3, #0x60]+ trn2 v19.4s, v6.4s, v16.4s+ trn2 v21.4s, v13.4s, v8.4s+ trn1 v6.4s, v6.4s, v16.4s+ ldr q24, [x2, #0x20]+ trn1 v10.2d, v19.2d, v21.2d+ ldr q16, [x2], #0x60+ trn1 v5.2d, v6.2d, v23.2d+ trn1 v28.4s, v0.4s, v3.4s+ trn2 v18.2d, v6.2d, v23.2d+ mul v31.8h, v10.8h, v29.8h+ trn2 v13.4s, v0.4s, v3.4s+ ldur q14, [x2, #-0x50]+ sqrdmulh v26.8h, v18.8h, v30.8h+ ldur q20, [x2, #-0x20]+ mul v17.8h, v18.8h, v29.8h+ trn2 v18.2d, v19.2d, v21.2d+ mul v9.8h, v18.8h, v29.8h+ trn1 v12.4s, v11.4s, v4.4s+ sqrdmulh v22.8h, v18.8h, v30.8h+ sqrdmulh v3.8h, v10.8h, v30.8h+ sqrdmulh v25.8h, v5.8h, v30.8h+ mls v9.8h, v22.8h, v7.h[0]+ mls v17.8h, v26.8h, v7.h[0]+ trn2 v26.4s, v11.4s, v4.4s+ mul v8.8h, v5.8h, v29.8h+ trn1 v10.2d, v26.2d, v13.2d+ ldur q11, [x2, #-0x10]+ mls v31.8h, v3.8h, v7.h[0]+ trn1 v6.2d, v12.2d, v28.2d+ trn2 v3.2d, v26.2d, v13.2d+ ldur q4, [x2, #-0x30]+ mls v8.8h, v25.8h, v7.h[0]+ sub v19.8h, v17.8h, v9.8h+ trn2 v13.2d, v12.2d, v28.2d+ sqrdmulh v1.8h, v3.8h, v30.8h+ add v9.8h, v17.8h, v9.8h+ mul v18.8h, v19.8h, v20.8h+ add v28.8h, v8.8h, v31.8h+ sqrdmulh v20.8h, v19.8h, v11.8h+ sub v12.8h, v28.8h, v9.8h+ sub v23.8h, v8.8h, v31.8h+ sqrdmulh v11.8h, v13.8h, v30.8h+ sqrdmulh v5.8h, v23.8h, v4.8h+ mul v0.8h, v23.8h, v24.8h+ mul v2.8h, v13.8h, v29.8h+ mls v0.8h, v5.8h, v7.h[0]+ add v24.8h, v28.8h, v9.8h+ mls v18.8h, v20.8h, v7.h[0]+ sqrdmulh v15.8h, v6.8h, v30.8h+ sqrdmulh v25.8h, v12.8h, v14.8h+ mul v21.8h, v12.8h, v16.8h+ sub v23.8h, v0.8h, v18.8h+ sqrdmulh v8.8h, v23.8h, v14.8h+ mul v23.8h, v23.8h, v16.8h+ mls v21.8h, v25.8h, v7.h[0]+ mls v23.8h, v8.8h, v7.h[0]+ mul v14.8h, v3.8h, v29.8h+ add v3.8h, v0.8h, v18.8h+ trn2 v4.4s, v24.4s, v3.4s+ mls v14.8h, v1.8h, v7.h[0]+ trn1 v9.4s, v24.4s, v3.4s+ trn2 v12.4s, v21.4s, v23.4s+ mls v2.8h, v11.8h, v7.h[0]+ trn1 v28.4s, v21.4s, v23.4s+ ldr q11, [x1], #0x10+ mul v31.8h, v10.8h, v29.8h+ trn1 v25.2d, v4.2d, v12.2d+ trn1 v20.2d, v9.2d, v28.2d+ ldr q23, [x2, #0x50]+ trn2 v13.2d, v4.2d, v12.2d+ sqrdmulh v21.8h, v10.8h, v30.8h+ trn2 v4.2d, v9.2d, v28.2d+ ldr q9, [x2, #0x40]+ mul v27.8h, v6.8h, v29.8h+ add v26.8h, v20.8h, v25.8h+ sub v3.8h, v2.8h, v14.8h+ sqdmulh v12.8h, v26.8h, v7.h[1]+ add v5.8h, v4.8h, v13.8h+ sub v8.8h, v4.8h, v13.8h+ add v10.8h, v2.8h, v14.8h+ sqdmulh v6.8h, v5.8h, v7.h[1]+ ldr q2, [x2, #0x10]+ mls v27.8h, v15.8h, v7.h[0]+ ldr q15, [x2, #0x20]+ srshr v12.8h, v12.8h, #0xb+ mls v31.8h, v21.8h, v7.h[0]+ srshr v6.8h, v6.8h, #0xb+ sqrdmulh v23.8h, v3.8h, v23.8h+ mls v26.8h, v12.8h, v7.h[0]+ add v21.8h, v27.8h, v31.8h+ mls v5.8h, v6.8h, v7.h[0]+ sub v6.8h, v27.8h, v31.8h+ sub v14.8h, v21.8h, v10.8h+ ldr q27, [x2], #0x60+ mul v3.8h, v3.8h, v9.8h+ mls v3.8h, v23.8h, v7.h[0]+ ldur q13, [x2, #-0x30]+ sub v12.8h, v26.8h, v5.8h+ add v5.8h, v26.8h, v5.8h+ sqrdmulh v31.8h, v8.8h, v11.h[5]+ sqrdmulh v19.8h, v12.8h, v11.h[1]+ mul v24.8h, v12.8h, v11.h[0]+ sqrdmulh v13.8h, v6.8h, v13.8h+ mls v24.8h, v19.8h, v7.h[0]+ sub x4, x4, #0x2++Lmlk_intt_layer4567_start:+ add v16.8h, v21.8h, v10.8h+ mul v18.8h, v6.8h, v15.8h+ sub v19.8h, v20.8h, v25.8h+ ldr q21, [x3, #0xa0]+ str q5, [x3], #0x40+ mls v18.8h, v13.8h, v7.h[0]+ sqrdmulh v15.8h, v14.8h, v2.8h+ ldr q10, [x3, #0x50]+ ldr q12, [x3, #0x40]+ stur q24, [x3, #-0x20]+ mul v5.8h, v8.8h, v11.h[4]+ sub v0.8h, v18.8h, v3.8h+ ldr q24, [x3, #0x70]+ mls v5.8h, v31.8h, v7.h[0]+ ldr q26, [x2, #0x50]+ trn2 v1.4s, v12.4s, v10.4s+ add v6.8h, v18.8h, v3.8h+ sqrdmulh v20.8h, v0.8h, v2.8h+ trn1 v13.4s, v12.4s, v10.4s+ trn1 v18.4s, v16.4s, v6.4s+ mul v22.8h, v0.8h, v27.8h+ trn1 v17.4s, v21.4s, v24.4s+ sqrdmulh v0.8h, v19.8h, v11.h[3]+ trn1 v25.2d, v13.2d, v17.2d+ mls v22.8h, v20.8h, v7.h[0]+ trn2 v21.4s, v21.4s, v24.4s+ mul v24.8h, v25.8h, v29.8h+ trn2 v28.2d, v13.2d, v17.2d+ sqrdmulh v4.8h, v25.8h, v30.8h+ trn2 v3.2d, v1.2d, v21.2d+ mul v17.8h, v28.8h, v29.8h+ sqrdmulh v31.8h, v28.8h, v30.8h+ ldr q2, [x2, #0x10]+ mls v24.8h, v4.8h, v7.h[0]+ mul v4.8h, v19.8h, v11.h[2]+ ldr q19, [x2, #0x40]+ mls v4.8h, v0.8h, v7.h[0]+ mul v0.8h, v14.8h, v27.8h+ mls v0.8h, v15.8h, v7.h[0]+ sub v8.8h, v4.8h, v5.8h+ mul v12.8h, v3.8h, v29.8h+ mul v23.8h, v8.8h, v11.h[0]+ trn2 v28.4s, v16.4s, v6.4s+ sqrdmulh v10.8h, v8.8h, v11.h[1]+ trn1 v9.4s, v0.4s, v22.4s+ trn2 v22.4s, v0.4s, v22.4s+ ldr q11, [x1], #0x10+ mls v17.8h, v31.8h, v7.h[0]+ trn1 v20.2d, v18.2d, v9.2d+ trn2 v14.2d, v18.2d, v9.2d+ ldr q15, [x2, #0x20]+ trn1 v6.2d, v1.2d, v21.2d+ sqrdmulh v9.8h, v3.8h, v30.8h+ trn1 v25.2d, v28.2d, v22.2d+ trn2 v16.2d, v28.2d, v22.2d+ mls v23.8h, v10.8h, v7.h[0]+ add v1.8h, v20.8h, v25.8h+ sqrdmulh v21.8h, v6.8h, v30.8h+ add v8.8h, v14.8h, v16.8h+ ldr q27, [x2], #0x60+ sqdmulh v28.8h, v8.8h, v7.h[1]+ mls v12.8h, v9.8h, v7.h[0]+ sqdmulh v31.8h, v1.8h, v7.h[1]+ mul v0.8h, v6.8h, v29.8h+ sub v10.8h, v17.8h, v12.8h+ mls v0.8h, v21.8h, v7.h[0]+ srshr v21.8h, v28.8h, #0xb+ srshr v13.8h, v31.8h, #0xb+ sqrdmulh v22.8h, v10.8h, v26.8h+ mls v8.8h, v21.8h, v7.h[0]+ mls v1.8h, v13.8h, v7.h[0]+ add v21.8h, v24.8h, v0.8h+ stur q23, [x3, #-0x10]+ sub v6.8h, v24.8h, v0.8h+ mul v3.8h, v10.8h, v19.8h+ add v0.8h, v4.8h, v5.8h+ sqdmulh v13.8h, v0.8h, v7.h[1]+ ldur q10, [x2, #-0x30]+ add v5.8h, v1.8h, v8.8h+ mls v3.8h, v22.8h, v7.h[0]+ sub v8.8h, v1.8h, v8.8h+ mul v24.8h, v8.8h, v11.h[0]+ sqrdmulh v8.8h, v8.8h, v11.h[1]+ srshr v1.8h, v13.8h, #0xb+ sqrdmulh v13.8h, v6.8h, v10.8h+ mls v0.8h, v1.8h, v7.h[0]+ add v10.8h, v17.8h, v12.8h+ mls v24.8h, v8.8h, v7.h[0]+ sub v8.8h, v14.8h, v16.8h+ sqrdmulh v31.8h, v8.8h, v11.h[5]+ sub v14.8h, v21.8h, v10.8h+ stur q0, [x3, #-0x30]+ sub x4, x4, #0x1+ cbnz x4, Lmlk_intt_layer4567_start+ mul v15.8h, v6.8h, v15.8h+ sub v22.8h, v20.8h, v25.8h+ add v4.8h, v21.8h, v10.8h+ str q24, [x3, #0x20]+ mls v15.8h, v13.8h, v7.h[0]+ str q5, [x3], #0x40+ ldr q9, [x1], #0x10+ sqrdmulh v28.8h, v14.8h, v2.8h+ mul v16.8h, v14.8h, v27.8h+ sub v18.8h, v15.8h, v3.8h+ add v15.8h, v15.8h, v3.8h+ sqrdmulh v0.8h, v18.8h, v2.8h+ trn2 v24.4s, v4.4s, v15.4s+ trn1 v2.4s, v4.4s, v15.4s+ mul v18.8h, v18.8h, v27.8h+ mls v16.8h, v28.8h, v7.h[0]+ mls v18.8h, v0.8h, v7.h[0]+ mul v23.8h, v8.8h, v11.h[4]+ sqrdmulh v12.8h, v22.8h, v11.h[3]+ trn1 v17.4s, v16.4s, v18.4s+ trn2 v4.4s, v16.4s, v18.4s+ mls v23.8h, v31.8h, v7.h[0]+ trn2 v3.2d, v2.2d, v17.2d+ trn2 v6.2d, v24.2d, v4.2d+ mul v26.8h, v22.8h, v11.h[2]+ trn1 v28.2d, v2.2d, v17.2d+ mls v26.8h, v12.8h, v7.h[0]+ add v25.8h, v3.8h, v6.8h+ sub v18.8h, v3.8h, v6.8h+ trn1 v24.2d, v24.2d, v4.2d+ sqdmulh v1.8h, v25.8h, v7.h[1]+ sub v27.8h, v28.8h, v24.8h+ sqrdmulh v2.8h, v18.8h, v9.h[5]+ add v28.8h, v28.8h, v24.8h+ mul v24.8h, v27.8h, v9.h[2]+ sqdmulh v12.8h, v28.8h, v7.h[1]+ mul v20.8h, v18.8h, v9.h[4]+ mls v20.8h, v2.8h, v7.h[0]+ srshr v1.8h, v1.8h, #0xb+ sqrdmulh v19.8h, v27.8h, v9.h[3]+ srshr v15.8h, v12.8h, #0xb+ mls v25.8h, v1.8h, v7.h[0]+ add v8.8h, v26.8h, v23.8h+ sub v4.8h, v26.8h, v23.8h+ mls v28.8h, v15.8h, v7.h[0]+ mls v24.8h, v19.8h, v7.h[0]+ mul v2.8h, v4.8h, v11.h[0]+ sub v19.8h, v28.8h, v25.8h+ sqrdmulh v15.8h, v4.8h, v11.h[1]+ add v25.8h, v28.8h, v25.8h+ sub v10.8h, v24.8h, v20.8h+ str q25, [x3], #0x40+ sqrdmulh v22.8h, v19.8h, v9.h[1]+ add v28.8h, v24.8h, v20.8h+ sqrdmulh v25.8h, v10.8h, v9.h[1]+ mul v27.8h, v19.8h, v9.h[0]+ mul v26.8h, v10.8h, v9.h[0]+ sqdmulh v20.8h, v28.8h, v7.h[1]+ sqdmulh v16.8h, v8.8h, v7.h[1]+ mls v26.8h, v25.8h, v7.h[0]+ mls v2.8h, v15.8h, v7.h[0]+ srshr v15.8h, v20.8h, #0xb+ srshr v1.8h, v16.8h, #0xb+ mls v27.8h, v22.8h, v7.h[0]+ mls v28.8h, v15.8h, v7.h[0]+ mls v8.8h, v1.8h, v7.h[0]+ stur q27, [x3, #-0x20]+ stur q2, [x3, #-0x50]+ stur q28, [x3, #-0x30]+ stur q26, [x3, #-0x10]+ stur q8, [x3, #-0x70]+ mov x4, #0x4 // =4+ ldr q0, [x1], #0x20+ ldur q1, [x1, #-0x10]+ ldr q26, [x0]+ ldr q13, [x0, #0x40]+ ldr q28, [x0, #0xc0]+ ldr q2, [x0, #0x140]+ ldr q6, [x0, #0x80]+ ldr q9, [x0, #0x100]+ ldr q29, [x0, #0x1c0]+ ldr q23, [x0, #0x180]+ sub v17.8h, v26.8h, v13.8h+ add v4.8h, v26.8h, v13.8h+ ldr q25, [x0, #0xd0]+ ldr q24, [x0, #0x50]+ add v5.8h, v6.8h, v28.8h+ mul v19.8h, v17.8h, v0.h[6]+ sub v10.8h, v6.8h, v28.8h+ ldr q30, [x0, #0x150]+ sqrdmulh v12.8h, v17.8h, v0.h[7]+ add v17.8h, v9.8h, v2.8h+ sub v28.8h, v9.8h, v2.8h+ ldr q2, [x0, #0x90]+ sub v26.8h, v23.8h, v29.8h+ sqrdmulh v31.8h, v10.8h, v1.h[1]+ add v22.8h, v23.8h, v29.8h+ ldr q3, [x0, #0x110]+ sqrdmulh v9.8h, v28.8h, v1.h[3]+ sub v20.8h, v4.8h, v5.8h+ sub v27.8h, v17.8h, v22.8h+ ldr q29, [x0, #0x10]+ add v16.8h, v4.8h, v5.8h+ sqrdmulh v4.8h, v26.8h, v1.h[5]+ add v6.8h, v17.8h, v22.8h+ ldr q22, [x0, #0x1d0]+ mul v8.8h, v28.8h, v1.h[2]+ sub v21.8h, v2.8h, v25.8h+ sub v5.8h, v16.8h, v6.8h+ mul v17.8h, v26.8h, v1.h[4]+ mul v26.8h, v10.8h, v1.h[0]+ mls v26.8h, v31.8h, v7.h[0]+ mls v17.8h, v4.8h, v7.h[0]+ mls v19.8h, v12.8h, v7.h[0]+ mls v8.8h, v9.8h, v7.h[0]+ sqrdmulh v10.8h, v27.8h, v0.h[5]+ sub v12.8h, v19.8h, v26.8h+ add v9.8h, v19.8h, v26.8h+ sqrdmulh v26.8h, v20.8h, v0.h[3]+ sub v11.8h, v8.8h, v17.8h+ add v14.8h, v8.8h, v17.8h+ sqrdmulh v13.8h, v12.8h, v0.h[3]+ add v23.8h, v9.8h, v14.8h+ sqrdmulh v28.8h, v11.8h, v0.h[5]+ sub v19.8h, v9.8h, v14.8h+ mul v17.8h, v27.8h, v0.h[4]+ str q23, [x0, #0x40]+ mul v14.8h, v20.8h, v0.h[2]+ mul v8.8h, v11.8h, v0.h[4]+ mul v4.8h, v12.8h, v0.h[2]+ mls v14.8h, v26.8h, v7.h[0]+ mls v17.8h, v10.8h, v7.h[0]+ mls v8.8h, v28.8h, v7.h[0]+ mls v4.8h, v13.8h, v7.h[0]+ sub v10.8h, v14.8h, v17.8h+ add v20.8h, v14.8h, v17.8h+ sqrdmulh v28.8h, v5.8h, v0.h[1]+ mul v18.8h, v5.8h, v0.h[0]+ str q20, [x0, #0x80]+ sub v13.8h, v4.8h, v8.8h+ mul v23.8h, v10.8h, v0.h[0]+ mul v17.8h, v19.8h, v0.h[0]+ sqrdmulh v9.8h, v13.8h, v0.h[1]+ mls v18.8h, v28.8h, v7.h[0]+ sqrdmulh v10.8h, v10.8h, v0.h[1]+ sub x4, x4, #0x2++Lmlk_intt_layer123_start:+ sub v12.8h, v3.8h, v30.8h+ mul v11.8h, v21.8h, v1.h[0]+ add v28.8h, v4.8h, v8.8h+ ldr q20, [x0, #0x190]+ add v27.8h, v16.8h, v6.8h+ sqrdmulh v8.8h, v12.8h, v1.h[3]+ add v16.8h, v29.8h, v24.8h+ str q28, [x0, #0xc0]+ mls v23.8h, v10.8h, v7.h[0]+ str q27, [x0], #0x10+ add v15.8h, v20.8h, v22.8h+ str q18, [x0, #0xf0]+ mul v14.8h, v13.8h, v0.h[0]+ add v2.8h, v2.8h, v25.8h+ sub v26.8h, v20.8h, v22.8h+ mul v4.8h, v12.8h, v1.h[2]+ sub v5.8h, v16.8h, v2.8h+ str q23, [x0, #0x170]+ add v20.8h, v3.8h, v30.8h+ sqrdmulh v27.8h, v26.8h, v1.h[5]+ add v16.8h, v16.8h, v2.8h+ mul v18.8h, v26.8h, v1.h[4]+ sub v31.8h, v20.8h, v15.8h+ mls v4.8h, v8.8h, v7.h[0]+ sub v28.8h, v29.8h, v24.8h+ mls v18.8h, v27.8h, v7.h[0]+ ldr q22, [x0, #0x1d0]+ mul v26.8h, v28.8h, v0.h[6]+ mul v2.8h, v5.8h, v0.h[2]+ sub v12.8h, v4.8h, v18.8h+ sqrdmulh v24.8h, v28.8h, v0.h[7]+ mls v14.8h, v9.8h, v7.h[0]+ sqrdmulh v10.8h, v12.8h, v0.h[5]+ mls v26.8h, v24.8h, v7.h[0]+ ldr q24, [x0, #0x50]+ mul v8.8h, v12.8h, v0.h[4]+ str q14, [x0, #0x1b0]+ add v28.8h, v4.8h, v18.8h+ sqrdmulh v5.8h, v5.8h, v0.h[3]+ add v6.8h, v20.8h, v15.8h+ sqrdmulh v3.8h, v19.8h, v0.h[1]+ sub v13.8h, v16.8h, v6.8h+ sqrdmulh v12.8h, v21.8h, v1.h[1]+ sqrdmulh v21.8h, v13.8h, v0.h[1]+ sqrdmulh v27.8h, v31.8h, v0.h[5]+ ldr q25, [x0, #0xd0]+ mls v11.8h, v12.8h, v7.h[0]+ mul v23.8h, v31.8h, v0.h[4]+ mul v18.8h, v13.8h, v0.h[0]+ add v30.8h, v26.8h, v11.8h+ sub v13.8h, v26.8h, v11.8h+ mls v23.8h, v27.8h, v7.h[0]+ add v12.8h, v30.8h, v28.8h+ sub v19.8h, v30.8h, v28.8h+ mls v2.8h, v5.8h, v7.h[0]+ str q12, [x0, #0x40]+ sqrdmulh v26.8h, v13.8h, v0.h[3]+ mls v8.8h, v10.8h, v7.h[0]+ ldr q30, [x0, #0x150]+ sub v20.8h, v2.8h, v23.8h+ mul v4.8h, v13.8h, v0.h[2]+ add v13.8h, v2.8h, v23.8h+ mls v4.8h, v26.8h, v7.h[0]+ ldr q2, [x0, #0x90]+ mul v23.8h, v20.8h, v0.h[0]+ ldr q29, [x0, #0x10]+ sqrdmulh v10.8h, v20.8h, v0.h[1]+ str q13, [x0, #0x80]+ sub v13.8h, v4.8h, v8.8h+ mls v17.8h, v3.8h, v7.h[0]+ ldr q3, [x0, #0x110]+ mls v18.8h, v21.8h, v7.h[0]+ sub v21.8h, v2.8h, v25.8h+ sqrdmulh v9.8h, v13.8h, v0.h[1]+ str q17, [x0, #0x130]+ mul v17.8h, v19.8h, v0.h[0]+ sub x4, x4, #0x1+ cbnz x4, Lmlk_intt_layer123_start+ mls v23.8h, v10.8h, v7.h[0]+ ldr q11, [x0, #0x190]+ str q18, [x0, #0x100]+ add v27.8h, v3.8h, v30.8h+ mul v13.8h, v13.8h, v0.h[0]+ sub v5.8h, v29.8h, v24.8h+ add v14.8h, v16.8h, v6.8h+ mls v13.8h, v9.8h, v7.h[0]+ add v10.8h, v11.8h, v22.8h+ str q23, [x0, #0x180]+ sub v20.8h, v11.8h, v22.8h+ sub v23.8h, v27.8h, v10.8h+ sqrdmulh v16.8h, v21.8h, v1.h[1]+ sqrdmulh v31.8h, v23.8h, v0.h[5]+ str q13, [x0, #0x1c0]+ add v13.8h, v4.8h, v8.8h+ mul v18.8h, v21.8h, v1.h[0]+ str q13, [x0, #0xc0]+ sqrdmulh v13.8h, v19.8h, v0.h[1]+ sqrdmulh v28.8h, v20.8h, v1.h[5]+ str q14, [x0], #0x10+ mul v4.8h, v20.8h, v1.h[4]+ mls v17.8h, v13.8h, v7.h[0]+ sub v13.8h, v3.8h, v30.8h+ sqrdmulh v8.8h, v13.8h, v1.h[3]+ mul v12.8h, v13.8h, v1.h[2]+ mls v4.8h, v28.8h, v7.h[0]+ mls v12.8h, v8.8h, v7.h[0]+ mls v18.8h, v16.8h, v7.h[0]+ str q17, [x0, #0x130]+ sqrdmulh v15.8h, v5.8h, v0.h[7]+ add v11.8h, v27.8h, v10.8h+ mul v16.8h, v5.8h, v0.h[6]+ sub v8.8h, v12.8h, v4.8h+ sqrdmulh v28.8h, v8.8h, v0.h[5]+ add v13.8h, v2.8h, v25.8h+ mls v16.8h, v15.8h, v7.h[0]+ add v26.8h, v12.8h, v4.8h+ mul v8.8h, v8.8h, v0.h[4]+ add v4.8h, v29.8h, v24.8h+ mls v8.8h, v28.8h, v7.h[0]+ sub v20.8h, v4.8h, v13.8h+ add v14.8h, v4.8h, v13.8h+ add v12.8h, v16.8h, v18.8h+ sqrdmulh v22.8h, v20.8h, v0.h[3]+ add v27.8h, v14.8h, v11.8h+ sub v13.8h, v16.8h, v18.8h+ mul v4.8h, v20.8h, v0.h[2]+ str q27, [x0], #0x10+ sub v24.8h, v12.8h, v26.8h+ sqrdmulh v3.8h, v13.8h, v0.h[3]+ mul v13.8h, v13.8h, v0.h[2]+ sqrdmulh v27.8h, v24.8h, v0.h[1]+ mls v13.8h, v3.8h, v7.h[0]+ mul v9.8h, v24.8h, v0.h[0]+ mls v9.8h, v27.8h, v7.h[0]+ add v30.8h, v13.8h, v8.8h+ sub v13.8h, v13.8h, v8.8h+ mls v4.8h, v22.8h, v7.h[0]+ str q30, [x0, #0xb0]+ sqrdmulh v16.8h, v13.8h, v0.h[1]+ str q9, [x0, #0x130]+ mul v9.8h, v13.8h, v0.h[0]+ add v13.8h, v12.8h, v26.8h+ str q13, [x0, #0x30]+ mul v13.8h, v23.8h, v0.h[4]+ sub v23.8h, v14.8h, v11.8h+ mls v13.8h, v31.8h, v7.h[0]+ mls v9.8h, v16.8h, v7.h[0]+ mul v30.8h, v23.8h, v0.h[0]+ sub v24.8h, v4.8h, v13.8h+ add v13.8h, v4.8h, v13.8h+ sqrdmulh v23.8h, v23.8h, v0.h[1]+ str q9, [x0, #0x1b0]+ str q13, [x0, #0x70]+ sqrdmulh v13.8h, v24.8h, v0.h[1]+ mul v21.8h, v24.8h, v0.h[0]+ mls v30.8h, v23.8h, v7.h[0]+ mls v21.8h, v13.8h, v7.h[0]+ str q30, [x0, #0xf0]+ str q21, [x0, #0x170]+ ldp d8, d9, [sp]+ .cfi_restore d8+ .cfi_restore d9+ ldp d10, d11, [sp, #0x10]+ .cfi_restore d10+ .cfi_restore d11+ ldp d12, d13, [sp, #0x20]+ .cfi_restore d12+ .cfi_restore d13+ ldp d14, d15, [sp, #0x30]+ .cfi_restore d14+ .cfi_restore d15+ add sp, sp, #0x40+ .cfi_adjust_cfa_offset -0x40+ ret+ .cfi_endproc++MLK_ASM_FN_SIZE(intt_aarch64_asm)++#endif /* MLK_ARITH_BACKEND_AARCH64 && !MLK_CONFIG_MULTILEVEL_NO_SHARED && \+ (!MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API) */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
@@ -0,0 +1,565 @@+/* Copyright (c) 2022 Arm Limited+ * Copyright (c) 2022 Hanno Becker+ * Copyright (c) 2023 Amin Abdulrahman, Matthias Kannwischer+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [NeonNTT]+ * Neon NTT: Faster Dilithium, Kyber, and Saber on Cortex-A72 and Apple M1+ * Becker, Hwang, Kannwischer, Yang, Yang+ * https://eprint.iacr.org/2021/986+ *+ * - [SLOTHY_Paper]+ * Fast and Clean: Auditable high-performance assembly via constraint solving+ * Abdulrahman, Becker, Kannwischer, Klein+ * https://eprint.iacr.org/2022/1303+ */++/*yaml+ Name: ntt_aarch64_asm+ Description: AArch64 ML-KEM forward NTT following @[NeonNTT] and @[SLOTHY_Paper]+ Signature: void mlk_ntt_aarch64_asm(int16_t p[256], const int16_t twiddles12345[80], const int16_t twiddles56[384])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON]+ x0:+ type: buffer+ size_bytes: 512+ permissions: read/write+ c_parameter: int16_t p[256]+ description: Input/output polynomial+ x1:+ type: buffer+ size_bytes: 160+ permissions: read-only+ c_parameter: const int16_t twiddles12345[80]+ description: Twiddle factors for layers 1-5+ x2:+ type: buffer+ size_bytes: 768+ permissions: read-only+ c_parameter: const int16_t twiddles56[384]+ description: Twiddle factors for layers 6-7+ Stack:+ bytes: 64+ description: saving callee-saved Neon registers+*/++#include "../../../common.h"+#if defined(MLK_ARITH_BACKEND_AARCH64) && !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED)++/*+ * WARNING: This file is auto-derived from the mlkem-native source file+ * dev/aarch64_opt/src/mlkem_ntt_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLK_ASM_NAMESPACE(ntt_aarch64_asm)+MLK_ASM_FN_SYMBOL(ntt_aarch64_asm)++ .cfi_startproc+ sub sp, sp, #0x40+ .cfi_adjust_cfa_offset 0x40+ stp d8, d9, [sp]+ .cfi_rel_offset d8, 0x0+ .cfi_rel_offset d9, 0x8+ stp d10, d11, [sp, #0x10]+ .cfi_rel_offset d10, 0x10+ .cfi_rel_offset d11, 0x18+ stp d12, d13, [sp, #0x20]+ .cfi_rel_offset d12, 0x20+ .cfi_rel_offset d13, 0x28+ stp d14, d15, [sp, #0x30]+ .cfi_rel_offset d14, 0x30+ .cfi_rel_offset d15, 0x38+ mov w5, #0xd01 // =3329+ mov v7.h[0], w5+ mov w5, #0x4ebf // =20159+ mov v7.h[1], w5+ mov x3, x0+ mov x4, #0x4 // =4+ ldr q0, [x1], #0x20+ ldur q1, [x1, #-0x10]+ ldr q21, [x0, #0x40]+ ldr q5, [x0, #0x1c0]+ ldr q30, [x0, #0x110]+ ldr q24, [x0, #0x140]+ ldr q12, [x0, #0x80]+ sqrdmulh v9.8h, v5.8h, v0.h[1]+ mul v23.8h, v5.8h, v0.h[0]+ sqrdmulh v17.8h, v24.8h, v0.h[1]+ ldr q13, [x0, #0xc0]+ mls v23.8h, v9.8h, v7.h[0]+ mul v8.8h, v24.8h, v0.h[0]+ mls v8.8h, v17.8h, v7.h[0]+ add v9.8h, v13.8h, v23.8h+ sub v10.8h, v13.8h, v23.8h+ mul v11.8h, v30.8h, v0.h[0]+ ldr q13, [x0, #0x180]+ sqrdmulh v28.8h, v9.8h, v0.h[3]+ sub v29.8h, v21.8h, v8.8h+ mul v26.8h, v9.8h, v0.h[2]+ add v8.8h, v21.8h, v8.8h+ mul v2.8h, v13.8h, v0.h[0]+ mls v26.8h, v28.8h, v7.h[0]+ mul v28.8h, v10.8h, v0.h[4]+ sqrdmulh v23.8h, v10.8h, v0.h[5]+ add v22.8h, v8.8h, v26.8h+ sqrdmulh v10.8h, v13.8h, v0.h[1]+ sqrdmulh v21.8h, v22.8h, v0.h[7]+ ldr q13, [x0, #0x100]+ mul v16.8h, v22.8h, v0.h[6]+ mls v28.8h, v23.8h, v7.h[0]+ mls v2.8h, v10.8h, v7.h[0]+ sqrdmulh v23.8h, v13.8h, v0.h[1]+ sub v10.8h, v29.8h, v28.8h+ add v17.8h, v29.8h, v28.8h+ mls v16.8h, v21.8h, v7.h[0]+ sub v18.8h, v12.8h, v2.8h+ ldr q29, [x0]+ sqrdmulh v14.8h, v17.8h, v1.h[3]+ add v22.8h, v12.8h, v2.8h+ sqrdmulh v9.8h, v18.8h, v0.h[5]+ mul v21.8h, v13.8h, v0.h[0]+ ldr q13, [x0, #0x150]+ mul v5.8h, v18.8h, v0.h[4]+ mls v5.8h, v9.8h, v7.h[0]+ mul v18.8h, v13.8h, v0.h[0]+ mls v21.8h, v23.8h, v7.h[0]+ sqrdmulh v2.8h, v13.8h, v0.h[1]+ mul v13.8h, v17.8h, v1.h[2]+ sub v4.8h, v29.8h, v21.8h+ mls v13.8h, v14.8h, v7.h[0]+ add v25.8h, v29.8h, v21.8h+ add v6.8h, v4.8h, v5.8h+ sqrdmulh v15.8h, v22.8h, v0.h[3]+ sub v21.8h, v4.8h, v5.8h+ sub v5.8h, v8.8h, v26.8h+ mul v23.8h, v22.8h, v0.h[2]+ add v28.8h, v6.8h, v13.8h+ sub v13.8h, v6.8h, v13.8h+ mul v4.8h, v5.8h, v1.h[0]+ sub x4, x4, #0x2++Lmlk_ntt_layer123_start:+ mls v23.8h, v15.8h, v7.h[0]+ ldr q6, [x0, #0x190]+ ldr q15, [x0, #0x90]+ ldr q19, [x0, #0x10]+ mul v22.8h, v10.8h, v1.h[4]+ ldr q24, [x0, #0x50]+ str q13, [x0, #0x140]+ sqrdmulh v13.8h, v6.8h, v0.h[1]+ sub v20.8h, v25.8h, v23.8h+ sqrdmulh v3.8h, v30.8h, v0.h[1]+ str q28, [x0, #0x100]+ ldr q30, [x0, #0x120]+ mul v8.8h, v6.8h, v0.h[0]+ sqrdmulh v27.8h, v10.8h, v1.h[5]+ mls v11.8h, v3.8h, v7.h[0]+ mls v18.8h, v2.8h, v7.h[0]+ ldr q31, [x0, #0x160]+ sqrdmulh v10.8h, v5.8h, v1.h[1]+ mls v8.8h, v13.8h, v7.h[0]+ ldr q13, [x0, #0x1d0]+ sub v14.8h, v24.8h, v18.8h+ add v9.8h, v24.8h, v18.8h+ sqrdmulh v2.8h, v31.8h, v0.h[1]+ mls v4.8h, v10.8h, v7.h[0]+ add v10.8h, v25.8h, v23.8h+ sub v24.8h, v19.8h, v11.8h+ add v25.8h, v19.8h, v11.8h+ sqrdmulh v28.8h, v13.8h, v0.h[1]+ mul v11.8h, v30.8h, v0.h[0]+ mul v17.8h, v13.8h, v0.h[0]+ sub v13.8h, v10.8h, v16.8h+ sub v6.8h, v15.8h, v8.8h+ mls v17.8h, v28.8h, v7.h[0]+ str q13, [x0, #0x40]+ mls v22.8h, v27.8h, v7.h[0]+ ldr q13, [x0, #0xd0]+ add v26.8h, v20.8h, v4.8h+ mul v18.8h, v31.8h, v0.h[0]+ add v27.8h, v10.8h, v16.8h+ str q26, [x0, #0x80]+ sqrdmulh v31.8h, v6.8h, v0.h[5]+ add v3.8h, v21.8h, v22.8h+ str q27, [x0], #0x10+ mul v26.8h, v6.8h, v0.h[4]+ add v6.8h, v13.8h, v17.8h+ sub v5.8h, v13.8h, v17.8h+ str q3, [x0, #0x170]+ sub v17.8h, v21.8h, v22.8h+ sqrdmulh v10.8h, v6.8h, v0.h[3]+ sub v13.8h, v20.8h, v4.8h+ add v20.8h, v15.8h, v8.8h+ sqrdmulh v12.8h, v5.8h, v0.h[5]+ str q13, [x0, #0xb0]+ mul v8.8h, v6.8h, v0.h[2]+ str q17, [x0, #0x1b0]+ mls v8.8h, v10.8h, v7.h[0]+ mul v29.8h, v5.8h, v0.h[4]+ mls v29.8h, v12.8h, v7.h[0]+ sub v5.8h, v9.8h, v8.8h+ add v3.8h, v9.8h, v8.8h+ sqrdmulh v15.8h, v20.8h, v0.h[3]+ mul v4.8h, v5.8h, v1.h[0]+ add v6.8h, v14.8h, v29.8h+ sqrdmulh v9.8h, v3.8h, v0.h[7]+ sqrdmulh v12.8h, v6.8h, v1.h[3]+ sub v10.8h, v14.8h, v29.8h+ mul v23.8h, v6.8h, v1.h[2]+ mls v26.8h, v31.8h, v7.h[0]+ mls v23.8h, v12.8h, v7.h[0]+ mul v16.8h, v3.8h, v0.h[6]+ add v13.8h, v24.8h, v26.8h+ sub v21.8h, v24.8h, v26.8h+ mls v16.8h, v9.8h, v7.h[0]+ add v28.8h, v13.8h, v23.8h+ sub v13.8h, v13.8h, v23.8h+ mul v23.8h, v20.8h, v0.h[2]+ sub x4, x4, #0x1+ cbnz x4, Lmlk_ntt_layer123_start+ sqrdmulh v3.8h, v5.8h, v1.h[1]+ mls v23.8h, v15.8h, v7.h[0]+ ldr q5, [x0, #0x190]+ mul v29.8h, v10.8h, v1.h[4]+ mls v4.8h, v3.8h, v7.h[0]+ sub v19.8h, v25.8h, v23.8h+ sqrdmulh v31.8h, v5.8h, v0.h[1]+ sqrdmulh v6.8h, v30.8h, v0.h[1]+ sub v3.8h, v19.8h, v4.8h+ mul v5.8h, v5.8h, v0.h[0]+ str q3, [x0, #0xc0]+ sqrdmulh v12.8h, v10.8h, v1.h[5]+ mls v18.8h, v2.8h, v7.h[0]+ ldr q3, [x0, #0x1d0]+ mls v5.8h, v31.8h, v7.h[0]+ sqrdmulh v10.8h, v3.8h, v0.h[1]+ mls v11.8h, v6.8h, v7.h[0]+ ldr q31, [x0, #0x90]+ mul v30.8h, v3.8h, v0.h[0]+ mls v30.8h, v10.8h, v7.h[0]+ sub v10.8h, v31.8h, v5.8h+ mls v29.8h, v12.8h, v7.h[0]+ ldr q6, [x0, #0xd0]+ sqrdmulh v15.8h, v10.8h, v0.h[5]+ mul v17.8h, v10.8h, v0.h[4]+ add v10.8h, v6.8h, v30.8h+ sub v6.8h, v6.8h, v30.8h+ sqrdmulh v12.8h, v10.8h, v0.h[3]+ sub v27.8h, v21.8h, v29.8h+ sqrdmulh v3.8h, v6.8h, v0.h[5]+ mul v10.8h, v10.8h, v0.h[2]+ ldr q20, [x0, #0x50]+ mls v10.8h, v12.8h, v7.h[0]+ mul v2.8h, v6.8h, v0.h[4]+ add v6.8h, v20.8h, v18.8h+ add v5.8h, v31.8h, v5.8h+ mls v2.8h, v3.8h, v7.h[0]+ sub v31.8h, v6.8h, v10.8h+ sqrdmulh v12.8h, v5.8h, v0.h[3]+ sub v22.8h, v20.8h, v18.8h+ add v6.8h, v6.8h, v10.8h+ mul v20.8h, v31.8h, v1.h[0]+ add v30.8h, v22.8h, v2.8h+ sqrdmulh v3.8h, v6.8h, v0.h[7]+ sqrdmulh v10.8h, v30.8h, v1.h[3]+ mul v9.8h, v30.8h, v1.h[2]+ ldr q30, [x0, #0x10]+ mls v17.8h, v15.8h, v7.h[0]+ mls v9.8h, v10.8h, v7.h[0]+ mul v15.8h, v6.8h, v0.h[6]+ add v24.8h, v30.8h, v11.8h+ sub v10.8h, v22.8h, v2.8h+ mls v15.8h, v3.8h, v7.h[0]+ add v6.8h, v19.8h, v4.8h+ add v22.8h, v25.8h, v23.8h+ sqrdmulh v3.8h, v10.8h, v1.h[5]+ str q13, [x0, #0x140]+ sub v19.8h, v30.8h, v11.8h+ add v25.8h, v22.8h, v16.8h+ mul v5.8h, v5.8h, v0.h[2]+ sub v13.8h, v22.8h, v16.8h+ str q28, [x0, #0x100]+ mls v5.8h, v12.8h, v7.h[0]+ str q13, [x0, #0x40]+ str q6, [x0, #0x80]+ add v21.8h, v21.8h, v29.8h+ sqrdmulh v13.8h, v31.8h, v1.h[1]+ str q25, [x0], #0x10+ add v12.8h, v19.8h, v17.8h+ sub v31.8h, v19.8h, v17.8h+ mul v30.8h, v10.8h, v1.h[4]+ str q21, [x0, #0x170]+ add v21.8h, v24.8h, v5.8h+ add v6.8h, v12.8h, v9.8h+ mls v30.8h, v3.8h, v7.h[0]+ str q27, [x0, #0x1b0]+ sub v10.8h, v21.8h, v15.8h+ sub v12.8h, v12.8h, v9.8h+ mls v20.8h, v13.8h, v7.h[0]+ str q6, [x0, #0x100]+ str q10, [x0, #0x40]+ sub v13.8h, v24.8h, v5.8h+ add v3.8h, v21.8h, v15.8h+ str q12, [x0, #0x140]+ sub v10.8h, v31.8h, v30.8h+ add v21.8h, v31.8h, v30.8h+ str q3, [x0], #0x10+ add v12.8h, v13.8h, v20.8h+ sub v13.8h, v13.8h, v20.8h+ str q21, [x0, #0x170]+ str q10, [x0, #0x1b0]+ str q12, [x0, #0x70]+ str q13, [x0, #0xb0]+ mov x0, x3+ mov x4, #0x8 // =8+ ldr q2, [x0, #0x20]+ ldr q13, [x1], #0x10+ ldr q30, [x0, #0x30]+ ldr q25, [x2, #0x40]+ ldr q5, [x0]+ ldr q18, [x0, #0x60]+ ldr q12, [x0, #0x70]+ sqrdmulh v17.8h, v2.8h, v13.h[1]+ ldr q4, [x1], #0x10+ ldr q23, [x0, #0x10]+ sqrdmulh v21.8h, v30.8h, v13.h[1]+ ldr q24, [x2, #0x20]+ ldr q9, [x2], #0x60+ mul v10.8h, v30.8h, v13.h[0]+ mul v11.8h, v2.8h, v13.h[0]+ mls v10.8h, v21.8h, v7.h[0]+ sqrdmulh v29.8h, v12.8h, v4.h[1]+ mul v1.8h, v12.8h, v4.h[0]+ add v21.8h, v23.8h, v10.8h+ sub v10.8h, v23.8h, v10.8h+ mul v8.8h, v18.8h, v4.h[0]+ sqrdmulh v23.8h, v21.8h, v13.h[3]+ mul v2.8h, v21.8h, v13.h[2]+ mls v1.8h, v29.8h, v7.h[0]+ mls v2.8h, v23.8h, v7.h[0]+ ldur q15, [x2, #-0x50]+ sqrdmulh v0.8h, v10.8h, v13.h[5]+ mls v11.8h, v17.8h, v7.h[0]+ ldr q29, [x0, #0x50]+ mul v23.8h, v10.8h, v13.h[4]+ mls v23.8h, v0.8h, v7.h[0]+ sub v16.8h, v29.8h, v1.8h+ add v3.8h, v5.8h, v11.8h+ sub v31.8h, v5.8h, v11.8h+ sqrdmulh v22.8h, v16.8h, v4.h[5]+ add v30.8h, v3.8h, v2.8h+ sub v0.8h, v3.8h, v2.8h+ sqrdmulh v28.8h, v18.8h, v4.h[1]+ add v21.8h, v31.8h, v23.8h+ sub v19.8h, v31.8h, v23.8h+ mul v26.8h, v16.8h, v4.h[4]+ trn2 v3.4s, v30.4s, v0.4s+ ldur q23, [x2, #-0x10]+ trn2 v18.4s, v21.4s, v19.4s+ mls v26.8h, v22.8h, v7.h[0]+ trn1 v13.4s, v30.4s, v0.4s+ mls v8.8h, v28.8h, v7.h[0]+ trn2 v31.2d, v3.2d, v18.2d+ trn1 v11.4s, v21.4s, v19.4s+ add v27.8h, v29.8h, v1.8h+ sqrdmulh v6.8h, v31.8h, v15.8h+ trn1 v2.2d, v13.2d, v11.2d+ trn2 v13.2d, v13.2d, v11.2d+ mul v1.8h, v31.8h, v9.8h+ ldr q11, [x0, #0x40]+ sqrdmulh v29.8h, v13.8h, v15.8h+ mls v1.8h, v6.8h, v7.h[0]+ trn1 v6.2d, v3.2d, v18.2d+ mul v17.8h, v13.8h, v9.8h+ sub v13.8h, v11.8h, v8.8h+ sqrdmulh v10.8h, v27.8h, v4.h[3]+ sub v12.8h, v13.8h, v26.8h+ sub v18.8h, v6.8h, v1.8h+ mls v17.8h, v29.8h, v7.h[0]+ add v30.8h, v6.8h, v1.8h+ add v6.8h, v13.8h, v26.8h+ ldur q13, [x2, #-0x30]+ sqrdmulh v16.8h, v18.8h, v23.8h+ trn1 v28.4s, v6.4s, v12.4s+ mul v23.8h, v18.8h, v25.8h+ ldr q25, [x2, #0x10]+ add v20.8h, v2.8h, v17.8h+ mul v0.8h, v30.8h, v24.8h+ sqrdmulh v29.8h, v30.8h, v13.8h+ sub v30.8h, v2.8h, v17.8h+ mls v23.8h, v16.8h, v7.h[0]+ sub x4, x4, #0x2++Lmlk_ntt_layer4567_start:+ ldr q19, [x2, #0x50]+ sub v31.8h, v30.8h, v23.8h+ mls v0.8h, v29.8h, v7.h[0]+ add v16.8h, v11.8h, v8.8h+ ldr q18, [x0, #0xa0]+ trn2 v14.4s, v6.4s, v12.4s+ mul v26.8h, v27.8h, v4.h[2]+ ldr q4, [x1], #0x10+ ldr q24, [x2, #0x40]+ ldr q21, [x0, #0xb0]+ mls v26.8h, v10.8h, v7.h[0]+ add v23.8h, v30.8h, v23.8h+ sub v15.8h, v20.8h, v0.8h+ ldr q9, [x0, #0x90]+ add v10.8h, v20.8h, v0.8h+ mul v8.8h, v18.8h, v4.h[0]+ ldr q1, [x2], #0x60+ trn1 v27.4s, v23.4s, v31.4s+ sqrdmulh v12.8h, v18.8h, v4.h[1]+ trn1 v5.4s, v10.4s, v15.4s+ sub v30.8h, v16.8h, v26.8h+ trn2 v13.2d, v5.2d, v27.2d+ sqrdmulh v2.8h, v21.8h, v4.h[1]+ add v29.8h, v16.8h, v26.8h+ mul v0.8h, v21.8h, v4.h[0]+ str q13, [x0, #0x20]+ trn1 v11.4s, v29.4s, v30.4s+ mls v8.8h, v12.8h, v7.h[0]+ trn2 v26.4s, v29.4s, v30.4s+ trn2 v6.2d, v11.2d, v28.2d+ mls v0.8h, v2.8h, v7.h[0]+ trn2 v16.2d, v26.2d, v14.2d+ trn1 v26.2d, v26.2d, v14.2d+ trn1 v20.2d, v5.2d, v27.2d+ sqrdmulh v29.8h, v6.8h, v25.8h+ trn2 v15.4s, v10.4s, v15.4s+ sqrdmulh v13.8h, v16.8h, v25.8h+ str q20, [x0], #0x40+ sub v30.8h, v9.8h, v0.8h+ add v27.8h, v9.8h, v0.8h+ mul v17.8h, v6.8h, v1.8h+ sqrdmulh v22.8h, v30.8h, v4.h[5]+ mul v18.8h, v16.8h, v1.8h+ mls v18.8h, v13.8h, v7.h[0]+ mul v2.8h, v30.8h, v4.h[4]+ mls v2.8h, v22.8h, v7.h[0]+ trn2 v22.4s, v23.4s, v31.4s+ sub v3.8h, v26.8h, v18.8h+ ldur q25, [x2, #-0x30]+ mls v17.8h, v29.8h, v7.h[0]+ trn2 v31.2d, v15.2d, v22.2d+ trn1 v20.2d, v15.2d, v22.2d+ add v16.8h, v26.8h, v18.8h+ sqrdmulh v26.8h, v3.8h, v19.8h+ trn1 v21.2d, v11.2d, v28.2d+ ldr q11, [x0, #0x40]+ sqrdmulh v29.8h, v16.8h, v25.8h+ stur q20, [x0, #-0x30]+ add v20.8h, v21.8h, v17.8h+ stur q31, [x0, #-0x10]+ mul v23.8h, v3.8h, v24.8h+ ldr q25, [x2, #0x10]+ sub v13.8h, v11.8h, v8.8h+ mls v23.8h, v26.8h, v7.h[0]+ ldur q1, [x2, #-0x40]+ sub v12.8h, v13.8h, v2.8h+ add v6.8h, v13.8h, v2.8h+ sqrdmulh v10.8h, v27.8h, v4.h[3]+ sub v30.8h, v21.8h, v17.8h+ mul v0.8h, v16.8h, v1.8h+ trn1 v28.4s, v6.4s, v12.4s+ sub x4, x4, #0x1+ cbnz x4, Lmlk_ntt_layer4567_start+ add v22.8h, v11.8h, v8.8h+ mul v27.8h, v27.8h, v4.h[2]+ trn2 v17.4s, v6.4s, v12.4s+ ldr q15, [x2], #0x60+ mls v27.8h, v10.8h, v7.h[0]+ add v4.8h, v30.8h, v23.8h+ sub v18.8h, v30.8h, v23.8h+ ldur q6, [x2, #-0x30]+ mls v0.8h, v29.8h, v7.h[0]+ ldur q12, [x2, #-0x40]+ ldur q24, [x2, #-0x20]+ ldur q2, [x2, #-0x10]+ trn1 v9.4s, v4.4s, v18.4s+ add v10.8h, v22.8h, v27.8h+ sub v13.8h, v22.8h, v27.8h+ sub v1.8h, v20.8h, v0.8h+ trn2 v21.4s, v10.4s, v13.4s+ add v27.8h, v20.8h, v0.8h+ trn2 v3.2d, v21.2d, v17.2d+ trn1 v13.4s, v10.4s, v13.4s+ trn1 v31.4s, v27.4s, v1.4s+ sqrdmulh v10.8h, v3.8h, v25.8h+ trn2 v5.2d, v13.2d, v28.2d+ trn1 v13.2d, v13.2d, v28.2d+ trn1 v21.2d, v21.2d, v17.2d+ sqrdmulh v17.8h, v5.8h, v25.8h+ trn2 v30.2d, v31.2d, v9.2d+ mul v25.8h, v3.8h, v15.8h+ str q30, [x0, #0x20]+ trn2 v30.4s, v4.4s, v18.4s+ mls v25.8h, v10.8h, v7.h[0]+ trn2 v3.4s, v27.4s, v1.4s+ mul v20.8h, v5.8h, v15.8h+ trn2 v10.2d, v3.2d, v30.2d+ mls v20.8h, v17.8h, v7.h[0]+ str q10, [x0, #0x30]+ sub v18.8h, v21.8h, v25.8h+ add v10.8h, v21.8h, v25.8h+ trn1 v3.2d, v3.2d, v30.2d+ sqrdmulh v30.8h, v18.8h, v2.8h+ mul v12.8h, v10.8h, v12.8h+ sqrdmulh v6.8h, v10.8h, v6.8h+ str q3, [x0, #0x10]+ add v21.8h, v13.8h, v20.8h+ mul v10.8h, v18.8h, v24.8h+ sub v13.8h, v13.8h, v20.8h+ mls v10.8h, v30.8h, v7.h[0]+ mls v12.8h, v6.8h, v7.h[0]+ trn1 v30.2d, v31.2d, v9.2d+ sub v3.8h, v13.8h, v10.8h+ add v6.8h, v13.8h, v10.8h+ add v10.8h, v21.8h, v12.8h+ sub v21.8h, v21.8h, v12.8h+ trn2 v13.4s, v6.4s, v3.4s+ trn1 v12.4s, v10.4s, v21.4s+ trn2 v21.4s, v10.4s, v21.4s+ trn1 v3.4s, v6.4s, v3.4s+ str q30, [x0], #0x40+ trn2 v10.2d, v21.2d, v13.2d+ trn1 v13.2d, v21.2d, v13.2d+ trn2 v21.2d, v12.2d, v3.2d+ trn1 v3.2d, v12.2d, v3.2d+ str q10, [x0, #0x30]+ str q13, [x0, #0x10]+ str q3, [x0], #0x40+ stur q21, [x0, #-0x20]+ ldp d8, d9, [sp]+ .cfi_restore d8+ .cfi_restore d9+ ldp d10, d11, [sp, #0x10]+ .cfi_restore d10+ .cfi_restore d11+ ldp d12, d13, [sp, #0x20]+ .cfi_restore d12+ .cfi_restore d13+ ldp d14, d15, [sp, #0x30]+ .cfi_restore d14+ .cfi_restore d15+ add sp, sp, #0x40+ .cfi_adjust_cfa_offset -0x40+ ret+ .cfi_endproc++MLK_ASM_FN_SIZE(ntt_aarch64_asm)++#endif /* MLK_ARITH_BACKEND_AARCH64 && !MLK_CONFIG_MULTILEVEL_NO_SHARED */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
@@ -0,0 +1,130 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*yaml+ Name: poly_mulcache_compute_aarch64_asm+ Description: Compute multiplication cache for polynomial+ Signature: void mlk_poly_mulcache_compute_aarch64_asm(int16_t cache[128], const int16_t mlk_poly[256], const int16_t zetas[128], const int16_t zetas_twisted[128])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON]+ x0:+ type: buffer+ size_bytes: 256+ permissions: write-only+ c_parameter: int16_t cache[128]+ description: Output cache+ x1:+ type: buffer+ size_bytes: 512+ permissions: read-only+ c_parameter: const int16_t mlk_poly[256]+ description: Input polynomial+ x2:+ type: buffer+ size_bytes: 256+ permissions: read-only+ c_parameter: const int16_t zetas[128]+ description: Zeta values+ x3:+ type: buffer+ size_bytes: 256+ permissions: read-only+ c_parameter: const int16_t zetas_twisted[128]+ description: Twisted zeta values+ Stack:+ bytes: 0+*/++#include "../../../common.h"+#if defined(MLK_ARITH_BACKEND_AARCH64) && !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED)++/*+ * WARNING: This file is auto-derived from the mlkem-native source file+ * dev/aarch64_opt/src/mlkem_poly_mulcache_compute_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLK_ASM_NAMESPACE(poly_mulcache_compute_aarch64_asm)+MLK_ASM_FN_SYMBOL(poly_mulcache_compute_aarch64_asm)++ .cfi_startproc+ mov w5, #0xd01 // =3329+ dup v6.8h, w5+ mov w5, #0x4ebf // =20159+ dup v7.8h, w5+ mov x4, #0x10 // =16+ ldr q0, [x1], #0x20+ ldur q2, [x1, #-0x10]+ ldr q19, [x1], #0x20+ ldr q29, [x3], #0x10+ ldur q16, [x1, #-0x10]+ ldr q18, [x2], #0x10+ ldr q26, [x1], #0x20+ ldr q25, [x2], #0x10+ uzp2 v5.8h, v0.8h, v2.8h+ ldr q28, [x3], #0x10+ ldur q7, [x1, #-0x10]+ ldr q2, [x1], #0x20+ uzp2 v27.8h, v19.8h, v16.8h+ sqrdmulh v16.8h, v5.8h, v29.8h+ ldr q17, [x3], #0x10+ ldr q19, [x3], #0x10+ mul v5.8h, v5.8h, v18.8h+ uzp2 v29.8h, v26.8h, v7.8h+ mul v26.8h, v27.8h, v25.8h+ sqrdmulh v4.8h, v27.8h, v28.8h+ mls v5.8h, v16.8h, v6.h[0]+ lsr x4, x4, #1+ sub x4, x4, #0x2++Lmlk_poly_mulcache_compute_loop_start:+ str q5, [x0], #0x10+ sqrdmulh v22.8h, v29.8h, v17.8h+ ldr q28, [x2], #0x10+ ldur q24, [x1, #-0x10]+ ldr q0, [x1], #0x20+ mls v26.8h, v4.8h, v6.h[0]+ ldur q16, [x1, #-0x10]+ ldr q17, [x3], #0x10+ mul v5.8h, v29.8h, v28.8h+ uzp2 v23.8h, v2.8h, v24.8h+ ldr q18, [x2], #0x10+ mls v5.8h, v22.8h, v6.h[0]+ uzp2 v29.8h, v0.8h, v16.8h+ sqrdmulh v4.8h, v23.8h, v19.8h+ ldr q2, [x1], #0x20+ ldr q19, [x3], #0x10+ str q26, [x0], #0x10+ mul v26.8h, v23.8h, v18.8h+ subs x4, x4, #0x1+ cbnz x4, Lmlk_poly_mulcache_compute_loop_start+ mls v26.8h, v4.8h, v6.h[0]+ str q5, [x0], #0x10+ ldr q5, [x2], #0x10+ ldur q4, [x1, #-0x10]+ sqrdmulh v16.8h, v29.8h, v17.8h+ ldr q0, [x2], #0x10+ mul v29.8h, v29.8h, v5.8h+ uzp2 v18.8h, v2.8h, v4.8h+ str q26, [x0], #0x10+ sqrdmulh v17.8h, v18.8h, v19.8h+ mls v29.8h, v16.8h, v6.h[0]+ mul v26.8h, v18.8h, v0.8h+ mls v26.8h, v17.8h, v6.h[0]+ str q29, [x0], #0x10+ str q26, [x0], #0x10+ ret+ .cfi_endproc++MLK_ASM_FN_SIZE(poly_mulcache_compute_aarch64_asm)++#endif /* MLK_ARITH_BACKEND_AARCH64 && !MLK_CONFIG_MULTILEVEL_NO_SHARED */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
@@ -0,0 +1,153 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*yaml+ Name: poly_reduce_aarch64_asm+ Description: Barrett reduction of polynomial coefficients+ Signature: void mlk_poly_reduce_aarch64_asm(int16_t p[256])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON]+ x0:+ type: buffer+ size_bytes: 512+ permissions: read/write+ c_parameter: int16_t p[256]+ description: Input/output polynomial+ Stack:+ bytes: 0+*/++#include "../../../common.h"+#if defined(MLK_ARITH_BACKEND_AARCH64) && !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED)++/*+ * WARNING: This file is auto-derived from the mlkem-native source file+ * dev/aarch64_opt/src/mlkem_poly_reduce_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLK_ASM_NAMESPACE(poly_reduce_aarch64_asm)+MLK_ASM_FN_SYMBOL(poly_reduce_aarch64_asm)++ .cfi_startproc+ mov w2, #0xd01 // =3329+ dup v3.8h, w2+ mov w2, #0x4ebf // =20159+ dup v4.8h, w2+ mov x1, #0x8 // =8+ ldr q21, [x0], #0x40+ ldur q18, [x0, #-0x20]+ ldur q0, [x0, #-0x30]+ ldur q5, [x0, #-0x10]+ ldr q26, [x0], #0x40+ sqdmulh v17.8h, v21.8h, v4.h[0]+ sqdmulh v27.8h, v18.8h, v4.h[0]+ sqdmulh v22.8h, v0.8h, v4.h[0]+ srshr v17.8h, v17.8h, #0xb+ sqdmulh v23.8h, v5.8h, v4.h[0]+ srshr v29.8h, v27.8h, #0xb+ mls v21.8h, v17.8h, v3.h[0]+ srshr v17.8h, v22.8h, #0xb+ mls v18.8h, v29.8h, v3.h[0]+ srshr v22.8h, v23.8h, #0xb+ mls v0.8h, v17.8h, v3.h[0]+ sshr v2.8h, v21.8h, #0xf+ mls v5.8h, v22.8h, v3.h[0]+ sshr v29.8h, v18.8h, #0xf+ and v19.16b, v3.16b, v2.16b+ sqdmulh v2.8h, v26.8h, v4.h[0]+ sshr v31.8h, v0.8h, #0xf+ add v17.8h, v21.8h, v19.8h+ and v21.16b, v3.16b, v29.16b+ and v31.16b, v3.16b, v31.16b+ sub x1, x1, #0x2++Lmlk_poly_reduce_loop_start:+ add v21.8h, v18.8h, v21.8h+ ldur q18, [x0, #-0x20]+ add v25.8h, v0.8h, v31.8h+ ldur q0, [x0, #-0x30]+ stur q21, [x0, #-0x60]+ sshr v28.8h, v5.8h, #0xf+ stur q17, [x0, #-0x80]+ srshr v23.8h, v2.8h, #0xb+ sqdmulh v30.8h, v18.8h, v4.h[0]+ stur q25, [x0, #-0x70]+ and v22.16b, v3.16b, v28.16b+ sqdmulh v7.8h, v0.8h, v4.h[0]+ add v16.8h, v5.8h, v22.8h+ ldur q5, [x0, #-0x10]+ mls v26.8h, v23.8h, v3.h[0]+ stur q16, [x0, #-0x50]+ srshr v6.8h, v30.8h, #0xb+ srshr v1.8h, v7.8h, #0xb+ sqdmulh v19.8h, v5.8h, v4.h[0]+ mls v18.8h, v6.8h, v3.h[0]+ sshr v24.8h, v26.8h, #0xf+ mls v0.8h, v1.8h, v3.h[0]+ and v27.16b, v3.16b, v24.16b+ srshr v29.8h, v19.8h, #0xb+ add v17.8h, v26.8h, v27.8h+ ldr q26, [x0], #0x40+ sshr v1.8h, v18.8h, #0xf+ mls v5.8h, v29.8h, v3.h[0]+ sshr v20.8h, v0.8h, #0xf+ and v21.16b, v3.16b, v1.16b+ and v31.16b, v3.16b, v20.16b+ sqdmulh v2.8h, v26.8h, v4.h[0]+ subs x1, x1, #0x1+ cbnz x1, Lmlk_poly_reduce_loop_start+ add v28.8h, v0.8h, v31.8h+ ldur q29, [x0, #-0x10]+ add v21.8h, v18.8h, v21.8h+ srshr v18.8h, v2.8h, #0xb+ sshr v2.8h, v5.8h, #0xf+ ldur q16, [x0, #-0x20]+ stur q17, [x0, #-0x80]+ ldur q0, [x0, #-0x30]+ and v2.16b, v3.16b, v2.16b+ sqdmulh v24.8h, v29.8h, v4.h[0]+ stur q28, [x0, #-0x70]+ stur q21, [x0, #-0x60]+ add v31.8h, v5.8h, v2.8h+ sqdmulh v6.8h, v16.8h, v4.h[0]+ stur q31, [x0, #-0x50]+ sqdmulh v17.8h, v0.8h, v4.h[0]+ srshr v22.8h, v24.8h, #0xb+ mls v26.8h, v18.8h, v3.h[0]+ srshr v31.8h, v6.8h, #0xb+ mls v29.8h, v22.8h, v3.h[0]+ srshr v19.8h, v17.8h, #0xb+ mls v16.8h, v31.8h, v3.h[0]+ sshr v7.8h, v26.8h, #0xf+ mls v0.8h, v19.8h, v3.h[0]+ and v5.16b, v3.16b, v7.16b+ sshr v22.8h, v29.8h, #0xf+ add v27.8h, v26.8h, v5.8h+ and v26.16b, v3.16b, v22.16b+ sshr v20.8h, v16.8h, #0xf+ stur q27, [x0, #-0x40]+ and v2.16b, v3.16b, v20.16b+ sshr v23.8h, v0.8h, #0xf+ add v18.8h, v29.8h, v26.8h+ add v31.8h, v16.8h, v2.8h+ and v29.16b, v3.16b, v23.16b+ stur q18, [x0, #-0x10]+ add v25.8h, v0.8h, v29.8h+ stur q31, [x0, #-0x20]+ stur q25, [x0, #-0x30]+ ret+ .cfi_endproc++MLK_ASM_FN_SIZE(poly_reduce_aarch64_asm)++#endif /* MLK_ARITH_BACKEND_AARCH64 && !MLK_CONFIG_MULTILEVEL_NO_SHARED */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
@@ -0,0 +1,124 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*yaml+ Name: poly_tobytes_aarch64_asm+ Description: Convert polynomial to byte representation+ Signature: void mlk_poly_tobytes_aarch64_asm(uint8_t r[384], const int16_t a[256])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON]+ x0:+ type: buffer+ size_bytes: 384+ permissions: write-only+ c_parameter: uint8_t r[384]+ description: Output byte array+ x1:+ type: buffer+ size_bytes: 512+ permissions: read-only+ c_parameter: const int16_t a[256]+ description: Input polynomial+ Stack:+ bytes: 0+*/++#include "../../../common.h"+#if defined(MLK_ARITH_BACKEND_AARCH64) && \+ !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED) && \+ (!defined(MLK_CONFIG_NO_KEYPAIR_API) || \+ !defined(MLK_CONFIG_NO_ENCAPS_API))++/*+ * WARNING: This file is auto-derived from the mlkem-native source file+ * dev/aarch64_opt/src/mlkem_poly_tobytes_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLK_ASM_NAMESPACE(poly_tobytes_aarch64_asm)+MLK_ASM_FN_SYMBOL(poly_tobytes_aarch64_asm)++ .cfi_startproc+ mov x2, #0x10 // =16+ ldr q5, [x1, #0x10]+ ldr q3, [x1], #0x20+ ldr q29, [x1], #0x20+ ldur q2, [x1, #-0x10]+ ldr q27, [x1, #0x10]+ ldr q23, [x1, #0x30]+ ldr q17, [x1], #0x20+ ldr q16, [x1], #0x20+ uzp2 v26.8h, v3.8h, v5.8h+ uzp1 v19.8h, v3.8h, v5.8h+ uzp2 v0.8h, v29.8h, v2.8h+ uzp1 v1.8h, v29.8h, v2.8h+ xtn v5.8b, v26.8h+ shrn v3.8b, v19.8h, #0x8+ shrn v4.8b, v26.8h, #0x4+ xtn v18.8b, v0.8h+ shrn v30.8b, v0.8h, #0x4+ xtn v28.8b, v1.8h+ shrn v29.8b, v1.8h, #0x8+ sli v3.8b, v5.8b, #0x4+ xtn v2.8b, v19.8h+ sli v29.8b, v18.8b, #0x4+ lsr x2, x2, #1+ sub x2, x2, #0x2++Lmlk_poly_tobytes_loop_start:+ uzp1 v25.8h, v17.8h, v27.8h+ uzp2 v31.8h, v17.8h, v27.8h+ uzp1 v24.8h, v16.8h, v23.8h+ uzp2 v6.8h, v16.8h, v23.8h+ st3 { v2.8b, v3.8b, v4.8b }, [x0], #24+ shrn v3.8b, v25.8h, #0x8+ ldr q17, [x1], #0x20+ shrn v4.8b, v31.8h, #0x4+ xtn v21.8b, v6.8h+ ldr q23, [x1, #0x10]+ st3 { v28.8b, v29.8b, v30.8b }, [x0], #24+ shrn v29.8b, v24.8h, #0x8+ ldur q27, [x1, #-0x10]+ xtn v20.8b, v31.8h+ ldr q16, [x1], #0x20+ sli v29.8b, v21.8b, #0x4+ xtn v2.8b, v25.8h+ sli v3.8b, v20.8b, #0x4+ xtn v28.8b, v24.8h+ shrn v30.8b, v6.8h, #0x4+ subs x2, x2, #0x1+ cbnz x2, Lmlk_poly_tobytes_loop_start+ uzp2 v7.8h, v17.8h, v27.8h+ uzp1 v25.8h, v17.8h, v27.8h+ uzp2 v0.8h, v16.8h, v23.8h+ st3 { v2.8b, v3.8b, v4.8b }, [x0], #24+ st3 { v28.8b, v29.8b, v30.8b }, [x0], #24+ shrn v21.8b, v25.8h, #0x8+ uzp1 v2.8h, v16.8h, v23.8h+ shrn v22.8b, v7.8h, #0x4+ shrn v4.8b, v0.8h, #0x4+ xtn v28.8b, v7.8h+ xtn v27.8b, v0.8h+ shrn v3.8b, v2.8h, #0x8+ sli v21.8b, v28.8b, #0x4+ xtn v2.8b, v2.8h+ sli v3.8b, v27.8b, #0x4+ xtn v20.8b, v25.8h+ st3 { v20.8b, v21.8b, v22.8b }, [x0], #24+ st3 { v2.8b, v3.8b, v4.8b }, [x0], #24+ ret+ .cfi_endproc++MLK_ASM_FN_SIZE(poly_tobytes_aarch64_asm)++#endif /* MLK_ARITH_BACKEND_AARCH64 && !MLK_CONFIG_MULTILEVEL_NO_SHARED && \+ (!MLK_CONFIG_NO_KEYPAIR_API || !MLK_CONFIG_NO_ENCAPS_API) */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
@@ -0,0 +1,102 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*yaml+ Name: poly_tomont_aarch64_asm+ Description: Convert polynomial to Montgomery domain+ Signature: void mlk_poly_tomont_aarch64_asm(int16_t p[256])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON]+ x0:+ type: buffer+ size_bytes: 512+ permissions: read/write+ c_parameter: int16_t p[256]+ description: Input/output polynomial+ Stack:+ bytes: 0+*/++#include "../../../common.h"+#if defined(MLK_ARITH_BACKEND_AARCH64) && \+ !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED) && \+ !defined(MLK_CONFIG_NO_KEYPAIR_API)++/*+ * WARNING: This file is auto-derived from the mlkem-native source file+ * dev/aarch64_opt/src/mlkem_poly_tomont_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLK_ASM_NAMESPACE(poly_tomont_aarch64_asm)+MLK_ASM_FN_SYMBOL(poly_tomont_aarch64_asm)++ .cfi_startproc+ mov w2, #0xd01 // =3329+ dup v4.8h, w2+ mov w2, #-0x414 // =-1044+ dup v2.8h, w2+ mov w2, #-0x2824 // =-10276+ dup v3.8h, w2+ mov x1, #0x8 // =8+ ldr q18, [x0, #0x20]+ ldr q0, [x0, #0x10]+ ldr q16, [x0], #0x40+ sqrdmulh v23.8h, v0.8h, v3.8h+ mul v26.8h, v0.8h, v2.8h+ sqrdmulh v19.8h, v16.8h, v3.8h+ mls v26.8h, v23.8h, v4.h[0]+ mul v29.8h, v16.8h, v2.8h+ ldur q16, [x0, #-0x10]+ mls v29.8h, v19.8h, v4.h[0]+ stur q26, [x0, #-0x30]+ sqrdmulh v26.8h, v18.8h, v3.8h+ mul v18.8h, v18.8h, v2.8h+ stur q29, [x0, #-0x40]+ sqrdmulh v29.8h, v16.8h, v3.8h+ mls v18.8h, v26.8h, v4.h[0]+ sub x1, x1, #0x1++Lmlk_poly_tomont_loop:+ ldr q19, [x0, #0x10]+ mul v26.8h, v16.8h, v2.8h+ ldr q23, [x0, #0x20]+ ldr q17, [x0], #0x40+ mls v26.8h, v29.8h, v4.h[0]+ ldur q16, [x0, #-0x10]+ sqrdmulh v28.8h, v19.8h, v3.8h+ stur q18, [x0, #-0x60]+ mul v0.8h, v19.8h, v2.8h+ stur q26, [x0, #-0x50]+ sqrdmulh v24.8h, v23.8h, v3.8h+ mul v18.8h, v23.8h, v2.8h+ sqrdmulh v22.8h, v17.8h, v3.8h+ mul v26.8h, v17.8h, v2.8h+ mls v0.8h, v28.8h, v4.h[0]+ mls v26.8h, v22.8h, v4.h[0]+ sqrdmulh v29.8h, v16.8h, v3.8h+ stur q0, [x0, #-0x30]+ mls v18.8h, v24.8h, v4.h[0]+ stur q26, [x0, #-0x40]+ sub x1, x1, #0x1+ cbnz x1, Lmlk_poly_tomont_loop+ mul v16.8h, v16.8h, v2.8h+ stur q18, [x0, #-0x20]+ mls v16.8h, v29.8h, v4.h[0]+ stur q16, [x0, #-0x10]+ ret+ .cfi_endproc++MLK_ASM_FN_SIZE(poly_tomont_aarch64_asm)++#endif /* MLK_ARITH_BACKEND_AARCH64 && !MLK_CONFIG_MULTILEVEL_NO_SHARED && \+ !MLK_CONFIG_NO_KEYPAIR_API */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
@@ -0,0 +1,264 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [NeonNTT]+ * Neon NTT: Faster Dilithium, Kyber, and Saber on Cortex-A72 and Apple M1+ * Becker, Hwang, Kannwischer, Yang, Yang+ * https://eprint.iacr.org/2021/986+ */++/*yaml+ Name: polyvec_basemul_acc_montgomery_cached_k2_aarch64_asm+ Description: Re-implementation of asymmetric base multiplication following @[NeonNTT] for k=2+ Signature: void mlk_polyvec_basemul_acc_montgomery_cached_k2_aarch64_asm(int16_t r[256], const int16_t a[512], const int16_t b[512], const int16_t b_cache[256])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON]+ x0:+ type: buffer+ size_bytes: 512+ permissions: write-only+ c_parameter: int16_t r[256]+ description: Output polynomial+ x1:+ type: buffer+ size_bytes: 1024+ permissions: read-only+ c_parameter: const int16_t a[512]+ description: Input polynomial vector a+ x2:+ type: buffer+ size_bytes: 1024+ permissions: read-only+ c_parameter: const int16_t b[512]+ description: Input polynomial vector b+ x3:+ type: buffer+ size_bytes: 512+ permissions: read-only+ c_parameter: const int16_t b_cache[256]+ description: Cached values for b+ Stack:+ bytes: 64+ description: saving callee-saved Neon registers+*/++/* Re-implementation of asymmetric base multiplication following @[NeonNTT] */++#include "../../../common.h"+#if defined(MLK_ARITH_BACKEND_AARCH64) && !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED) && (defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_K == 2)++/*+ * WARNING: This file is auto-derived from the mlkem-native source file+ * dev/aarch64_opt/src/mlkem_polyvec_basemul_acc_montgomery_cached_k2_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLK_ASM_NAMESPACE(polyvec_basemul_acc_montgomery_cached_k2_aarch64_asm)+MLK_ASM_FN_SYMBOL(polyvec_basemul_acc_montgomery_cached_k2_aarch64_asm)++ .cfi_startproc+ sub sp, sp, #0x40+ .cfi_adjust_cfa_offset 0x40+ stp d8, d9, [sp]+ .cfi_rel_offset d8, 0x0+ .cfi_rel_offset d9, 0x8+ stp d10, d11, [sp, #0x10]+ .cfi_rel_offset d10, 0x10+ .cfi_rel_offset d11, 0x18+ stp d12, d13, [sp, #0x20]+ .cfi_rel_offset d12, 0x20+ .cfi_rel_offset d13, 0x28+ stp d14, d15, [sp, #0x30]+ .cfi_rel_offset d14, 0x30+ .cfi_rel_offset d15, 0x38+ mov w14, #0xd01 // =3329+ dup v0.8h, w14+ mov w14, #0xcff // =3327+ dup v2.8h, w14+ add x4, x1, #0x200+ add x5, x2, #0x200+ add x6, x3, #0x100+ mov x13, #0x10 // =16+ ldr q12, [x1], #0x20+ ldur q9, [x1, #-0x10]+ ldr q22, [x2], #0x20+ ldur q30, [x2, #-0x10]+ ldr q6, [x5], #0x20+ ldr q7, [x4, #0x10]+ ldr q8, [x4], #0x20+ ldur q23, [x5, #-0x10]+ uzp1 v16.8h, v12.8h, v9.8h+ uzp2 v14.8h, v12.8h, v9.8h+ uzp2 v13.8h, v22.8h, v30.8h+ uzp1 v18.8h, v22.8h, v30.8h+ ld1 { v27.8h }, [x3], #16+ ld1 { v17.8h }, [x6], #16+ smull2 v4.4s, v16.8h, v18.8h+ ldr q31, [x1, #0x10]+ smull v19.4s, v16.4h, v13.4h+ ldr q24, [x1], #0x20+ smlal v19.4s, v14.4h, v18.4h+ ldr q22, [x2], #0x20+ smlal2 v4.4s, v14.8h, v27.8h+ uzp2 v5.8h, v6.8h, v23.8h+ smull2 v29.4s, v16.8h, v13.8h+ uzp2 v26.8h, v8.8h, v7.8h+ smlal2 v29.4s, v14.8h, v18.8h+ uzp1 v30.8h, v24.8h, v31.8h+ uzp1 v8.8h, v8.8h, v7.8h+ smull v11.4s, v16.4h, v18.4h+ smlal v11.4s, v14.4h, v27.4h+ ldur q1, [x2, #-0x10]+ uzp1 v28.8h, v6.8h, v23.8h+ smlal2 v29.4s, v8.8h, v5.8h+ ldr q25, [x5], #0x20+ smlal v19.4s, v8.4h, v5.4h+ ldr q3, [x4, #0x10]+ smlal2 v29.4s, v26.8h, v28.8h+ uzp1 v27.8h, v22.8h, v1.8h+ smlal v19.4s, v26.4h, v28.4h+ ldr q12, [x4], #0x20+ smlal2 v4.4s, v8.8h, v28.8h+ ldur q21, [x5, #-0x10]+ smlal2 v4.4s, v26.8h, v17.8h+ smlal v11.4s, v8.4h, v28.4h+ ld1 { v15.8h }, [x6], #16+ smlal v11.4s, v26.4h, v17.4h+ ld1 { v20.8h }, [x3], #16+ uzp1 v28.8h, v19.8h, v29.8h+ smull2 v23.4s, v30.8h, v27.8h+ smull v26.4s, v30.4h, v27.4h+ uzp2 v16.8h, v22.8h, v1.8h+ mul v28.8h, v28.8h, v2.8h+ uzp1 v10.8h, v11.8h, v4.8h+ smull2 v8.4s, v30.8h, v16.8h+ mul v13.8h, v10.8h, v2.8h+ smlal v19.4s, v28.4h, v0.4h+ smlal2 v29.4s, v28.8h, v0.8h+ smull v18.4s, v30.4h, v16.4h+ uzp1 v30.8h, v25.8h, v21.8h+ smlal v11.4s, v13.4h, v0.4h+ uzp2 v6.8h, v24.8h, v31.8h+ uzp1 v16.8h, v12.8h, v3.8h+ smlal2 v4.4s, v13.8h, v0.8h+ uzp2 v17.8h, v25.8h, v21.8h+ smlal2 v8.4s, v6.8h, v27.8h+ uzp2 v12.8h, v12.8h, v3.8h+ smlal v18.4s, v6.4h, v27.4h+ uzp2 v9.8h, v19.8h, v29.8h+ smlal2 v8.4s, v16.8h, v17.8h+ smlal2 v8.4s, v12.8h, v30.8h+ uzp2 v19.8h, v11.8h, v4.8h+ sub x13, x13, #0x2++Lmlk_polyvec_basemul_acc_montgomery_cached_k2_loop_start:+ smlal v18.4s, v16.4h, v17.4h+ ldr q7, [x4], #0x20+ ldr q10, [x2, #0x10]+ smlal v18.4s, v12.4h, v30.4h+ smlal2 v23.4s, v6.8h, v20.8h+ ldr q14, [x2], #0x20+ smlal2 v23.4s, v16.8h, v30.8h+ zip1 v25.8h, v19.8h, v9.8h+ zip2 v3.8h, v19.8h, v9.8h+ smlal2 v23.4s, v12.8h, v15.8h+ smlal v26.4s, v6.4h, v20.4h+ uzp1 v5.8h, v18.8h, v8.8h+ uzp2 v21.8h, v14.8h, v10.8h+ smlal v26.4s, v16.4h, v30.4h+ str q25, [x0], #0x20+ mul v29.8h, v5.8h, v2.8h+ uzp1 v24.8h, v14.8h, v10.8h+ stur q3, [x0, #-0x10]+ smlal v26.4s, v12.4h, v15.4h+ ld1 { v15.8h }, [x6], #16+ ldr q28, [x1, #0x10]+ ldr q11, [x1], #0x20+ ldr q13, [x5], #0x20+ ldur q27, [x4, #-0x10]+ smlal2 v8.4s, v29.8h, v0.8h+ ldur q22, [x5, #-0x10]+ smlal v18.4s, v29.4h, v0.4h+ uzp1 v4.8h, v26.8h, v23.8h+ uzp1 v1.8h, v11.8h, v28.8h+ uzp2 v6.8h, v11.8h, v28.8h+ uzp1 v16.8h, v7.8h, v27.8h+ mul v31.8h, v4.8h, v2.8h+ uzp2 v17.8h, v13.8h, v22.8h+ ld1 { v20.8h }, [x3], #16+ uzp2 v9.8h, v18.8h, v8.8h+ smull2 v8.4s, v1.8h, v21.8h+ uzp1 v30.8h, v13.8h, v22.8h+ smlal2 v8.4s, v6.8h, v24.8h+ smlal2 v8.4s, v16.8h, v17.8h+ uzp2 v12.8h, v7.8h, v27.8h+ smlal v26.4s, v31.4h, v0.4h+ smlal2 v23.4s, v31.8h, v0.8h+ smull v18.4s, v1.4h, v21.4h+ smlal v18.4s, v6.4h, v24.4h+ smlal2 v8.4s, v12.8h, v30.8h+ uzp2 v19.8h, v26.8h, v23.8h+ smull2 v23.4s, v1.8h, v24.8h+ smull v26.4s, v1.4h, v24.4h+ subs x13, x13, #0x1+ cbnz x13, Lmlk_polyvec_basemul_acc_montgomery_cached_k2_loop_start+ smlal v26.4s, v6.4h, v20.4h+ smlal2 v23.4s, v6.8h, v20.8h+ smlal v26.4s, v16.4h, v30.4h+ smlal2 v23.4s, v16.8h, v30.8h+ smlal v26.4s, v12.4h, v15.4h+ smlal2 v23.4s, v12.8h, v15.8h+ smlal v18.4s, v16.4h, v17.4h+ smlal v18.4s, v12.4h, v30.4h+ zip1 v12.8h, v19.8h, v9.8h+ str q12, [x0], #0x20+ uzp1 v12.8h, v26.8h, v23.8h+ mul v6.8h, v12.8h, v2.8h+ uzp1 v12.8h, v18.8h, v8.8h+ mul v12.8h, v12.8h, v2.8h+ smlal v26.4s, v6.4h, v0.4h+ smlal2 v23.4s, v6.8h, v0.8h+ smlal2 v8.4s, v12.8h, v0.8h+ smlal v18.4s, v12.4h, v0.4h+ zip2 v12.8h, v19.8h, v9.8h+ uzp2 v6.8h, v26.8h, v23.8h+ stur q12, [x0, #-0x10]+ uzp2 v12.8h, v18.8h, v8.8h+ zip2 v1.8h, v6.8h, v12.8h+ zip1 v12.8h, v6.8h, v12.8h+ str q1, [x0, #0x10]+ str q12, [x0], #0x20+ ldp d8, d9, [sp]+ .cfi_restore d8+ .cfi_restore d9+ ldp d10, d11, [sp, #0x10]+ .cfi_restore d10+ .cfi_restore d11+ ldp d12, d13, [sp, #0x20]+ .cfi_restore d12+ .cfi_restore d13+ ldp d14, d15, [sp, #0x30]+ .cfi_restore d14+ .cfi_restore d15+ add sp, sp, #0x40+ .cfi_adjust_cfa_offset -0x40+ ret+ .cfi_endproc++MLK_ASM_FN_SIZE(polyvec_basemul_acc_montgomery_cached_k2_aarch64_asm)+++#endif /* MLK_ARITH_BACKEND_AARCH64 && !MLK_CONFIG_MULTILEVEL_NO_SHARED && \+ (MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 2) */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
@@ -0,0 +1,317 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [NeonNTT]+ * Neon NTT: Faster Dilithium, Kyber, and Saber on Cortex-A72 and Apple M1+ * Becker, Hwang, Kannwischer, Yang, Yang+ * https://eprint.iacr.org/2021/986+ */++/*yaml+ Name: polyvec_basemul_acc_montgomery_cached_k3_aarch64_asm+ Description: Re-implementation of asymmetric base multiplication following @[NeonNTT] for k=3+ Signature: void mlk_polyvec_basemul_acc_montgomery_cached_k3_aarch64_asm(int16_t r[256], const int16_t a[768], const int16_t b[768], const int16_t b_cache[384])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON]+ x0:+ type: buffer+ size_bytes: 512+ permissions: write-only+ c_parameter: int16_t r[256]+ description: Output polynomial+ x1:+ type: buffer+ size_bytes: 1536+ permissions: read-only+ c_parameter: const int16_t a[768]+ description: Input polynomial vector a+ x2:+ type: buffer+ size_bytes: 1536+ permissions: read-only+ c_parameter: const int16_t b[768]+ description: Input polynomial vector b+ x3:+ type: buffer+ size_bytes: 768+ permissions: read-only+ c_parameter: const int16_t b_cache[384]+ description: Cached values for b+ Stack:+ bytes: 64+ description: saving callee-saved Neon registers+*/++/* Re-implementation of asymmetric base multiplication following @[NeonNTT] */++#include "../../../common.h"+#if defined(MLK_ARITH_BACKEND_AARCH64) && !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED) && (defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_K == 3)++/*+ * WARNING: This file is auto-derived from the mlkem-native source file+ * dev/aarch64_opt/src/mlkem_polyvec_basemul_acc_montgomery_cached_k3_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLK_ASM_NAMESPACE(polyvec_basemul_acc_montgomery_cached_k3_aarch64_asm)+MLK_ASM_FN_SYMBOL(polyvec_basemul_acc_montgomery_cached_k3_aarch64_asm)++ .cfi_startproc+ sub sp, sp, #0x40+ .cfi_adjust_cfa_offset 0x40+ stp d8, d9, [sp]+ .cfi_rel_offset d8, 0x0+ .cfi_rel_offset d9, 0x8+ stp d10, d11, [sp, #0x10]+ .cfi_rel_offset d10, 0x10+ .cfi_rel_offset d11, 0x18+ stp d12, d13, [sp, #0x20]+ .cfi_rel_offset d12, 0x20+ .cfi_rel_offset d13, 0x28+ stp d14, d15, [sp, #0x30]+ .cfi_rel_offset d14, 0x30+ .cfi_rel_offset d15, 0x38+ mov w14, #0xd01 // =3329+ dup v0.8h, w14+ mov w14, #0xcff // =3327+ dup v2.8h, w14+ add x4, x1, #0x200+ add x5, x2, #0x200+ add x6, x3, #0x100+ add x7, x1, #0x400+ add x8, x2, #0x400+ add x9, x3, #0x200+ mov x13, #0x10 // =16+ ldr q6, [x7], #0x20+ ldr q19, [x2, #0x10]+ ldr q23, [x1], #0x20+ ldur q14, [x1, #-0x10]+ ldr q17, [x2], #0x20+ ldr q11, [x4, #0x10]+ ldur q28, [x7, #-0x10]+ ld1 { v30.8h }, [x3], #16+ ldr q26, [x4], #0x20+ ldr q16, [x8, #0x10]+ uzp1 v8.8h, v23.8h, v14.8h+ ldr q22, [x5, #0x10]+ ldr q18, [x5], #0x20+ uzp1 v20.8h, v17.8h, v19.8h+ uzp2 v24.8h, v23.8h, v14.8h+ ldr q31, [x8], #0x20+ smull2 v4.4s, v8.8h, v20.8h+ uzp1 v25.8h, v26.8h, v11.8h+ smull v13.4s, v8.4h, v20.4h+ ld1 { v23.8h }, [x6], #16+ uzp1 v1.8h, v18.8h, v22.8h+ smlal v13.4s, v24.4h, v30.4h+ smlal2 v4.4s, v24.8h, v30.8h+ uzp2 v5.8h, v26.8h, v11.8h+ smlal2 v4.4s, v25.8h, v1.8h+ uzp1 v29.8h, v6.8h, v28.8h+ smlal2 v4.4s, v5.8h, v23.8h+ ld1 { v7.8h }, [x9], #16+ smlal v13.4s, v25.4h, v1.4h+ uzp2 v17.8h, v17.8h, v19.8h+ uzp1 v27.8h, v31.8h, v16.8h+ smlal v13.4s, v5.4h, v23.4h+ uzp2 v22.8h, v18.8h, v22.8h+ smull v18.4s, v8.4h, v17.4h+ uzp2 v28.8h, v6.8h, v28.8h+ smlal v13.4s, v29.4h, v27.4h+ smlal2 v4.4s, v29.8h, v27.8h+ uzp2 v26.8h, v31.8h, v16.8h+ smlal2 v4.4s, v28.8h, v7.8h+ ldr q3, [x7, #0x10]+ smlal v13.4s, v28.4h, v7.4h+ ldr q7, [x1], #0x20+ smlal v18.4s, v24.4h, v20.4h+ ldr q15, [x2], #0x20+ smlal v18.4s, v25.4h, v22.4h+ smull2 v8.4s, v8.8h, v17.8h+ ldur q17, [x1, #-0x10]+ uzp1 v23.8h, v13.8h, v4.8h+ smlal v18.4s, v5.4h, v1.4h+ smlal2 v8.4s, v24.8h, v20.8h+ ld1 { v16.8h }, [x3], #16+ mul v23.8h, v23.8h, v2.8h+ ldr q19, [x5, #0x10]+ ldr q14, [x4, #0x10]+ ldr q11, [x4], #0x20+ ldur q20, [x2, #-0x10]+ smlal2 v8.4s, v25.8h, v22.8h+ smlal2 v8.4s, v5.8h, v1.8h+ ldr q22, [x5], #0x20+ uzp1 v1.8h, v7.8h, v17.8h+ smlal v18.4s, v29.4h, v26.4h+ smlal v13.4s, v23.4h, v0.4h+ uzp2 v31.8h, v11.8h, v14.8h+ uzp1 v21.8h, v15.8h, v20.8h+ smlal2 v4.4s, v23.8h, v0.8h+ ld1 { v9.8h }, [x6], #16+ smlal v18.4s, v28.4h, v27.4h+ smlal2 v8.4s, v29.8h, v26.8h+ ldr q25, [x7], #0x20+ smull v26.4s, v1.4h, v21.4h+ uzp1 v24.8h, v22.8h, v19.8h+ smlal2 v8.4s, v28.8h, v27.8h+ uzp2 v28.8h, v7.8h, v17.8h+ uzp1 v29.8h, v11.8h, v14.8h+ smull2 v23.4s, v1.8h, v21.8h+ ldr q27, [x8], #0x20+ smlal2 v23.4s, v28.8h, v16.8h+ ldur q11, [x8, #-0x10]+ smlal2 v23.4s, v29.8h, v24.8h+ uzp2 v7.8h, v13.8h, v4.8h+ uzp2 v19.8h, v22.8h, v19.8h+ ld1 { v4.8h }, [x9], #16+ smlal2 v23.4s, v31.8h, v9.8h+ uzp1 v13.8h, v25.8h, v3.8h+ uzp1 v14.8h, v18.8h, v8.8h+ smlal v26.4s, v28.4h, v16.4h+ uzp2 v17.8h, v27.8h, v11.8h+ uzp2 v20.8h, v15.8h, v20.8h+ mul v14.8h, v14.8h, v2.8h+ sub x13, x13, #0x2++Lmlk_polyvec_basemul_acc_montgomery_cached_k3_loop_start:+ uzp1 v6.8h, v27.8h, v11.8h+ smlal v26.4s, v29.4h, v24.4h+ uzp2 v16.8h, v25.8h, v3.8h+ smlal v26.4s, v31.4h, v9.4h+ ldr q3, [x7, #0x10]+ smlal v26.4s, v13.4h, v6.4h+ smlal2 v8.4s, v14.8h, v0.8h+ ldr q27, [x8], #0x20+ smlal v18.4s, v14.4h, v0.4h+ ldr q25, [x7], #0x20+ smlal2 v23.4s, v13.8h, v6.8h+ ldr q11, [x1], #0x20+ smlal2 v23.4s, v16.8h, v4.8h+ smlal v26.4s, v16.4h, v4.4h+ ldur q22, [x1, #-0x10]+ uzp2 v30.8h, v18.8h, v8.8h+ smull v18.4s, v1.4h, v20.4h+ smlal v18.4s, v28.4h, v21.4h+ ldr q14, [x2], #0x20+ smlal v18.4s, v29.4h, v19.4h+ zip1 v5.8h, v7.8h, v30.8h+ uzp1 v4.8h, v26.8h, v23.8h+ smull2 v8.4s, v1.8h, v20.8h+ zip2 v10.8h, v7.8h, v30.8h+ smlal v18.4s, v31.4h, v24.4h+ mul v12.8h, v4.8h, v2.8h+ ldr q4, [x5, #0x10]+ ldr q20, [x4, #0x10]+ ldr q1, [x4], #0x20+ ldur q30, [x2, #-0x10]+ smlal2 v8.4s, v28.8h, v21.8h+ smlal2 v8.4s, v29.8h, v19.8h+ ldr q19, [x5], #0x20+ smlal2 v8.4s, v31.8h, v24.8h+ ld1 { v15.8h }, [x3], #16+ uzp2 v31.8h, v1.8h, v20.8h+ smlal v26.4s, v12.4h, v0.4h+ smlal2 v23.4s, v12.8h, v0.8h+ uzp1 v21.8h, v14.8h, v30.8h+ uzp1 v29.8h, v1.8h, v20.8h+ uzp1 v1.8h, v11.8h, v22.8h+ smlal2 v8.4s, v13.8h, v17.8h+ ld1 { v9.8h }, [x6], #16+ smlal v18.4s, v13.4h, v17.4h+ uzp1 v24.8h, v19.8h, v4.8h+ uzp2 v7.8h, v26.8h, v23.8h+ smull v26.4s, v1.4h, v21.4h+ smlal v18.4s, v16.4h, v6.4h+ uzp2 v19.8h, v19.8h, v4.8h+ smlal2 v8.4s, v16.8h, v6.8h+ uzp2 v28.8h, v11.8h, v22.8h+ smull2 v23.4s, v1.8h, v21.8h+ uzp1 v13.8h, v25.8h, v3.8h+ smlal2 v23.4s, v28.8h, v15.8h+ ldur q11, [x8, #-0x10]+ smlal2 v23.4s, v29.8h, v24.8h+ ld1 { v4.8h }, [x9], #16+ smlal2 v23.4s, v31.8h, v9.8h+ uzp1 v12.8h, v18.8h, v8.8h+ uzp2 v20.8h, v14.8h, v30.8h+ smlal v26.4s, v28.4h, v15.4h+ str q5, [x0], #0x20+ mul v14.8h, v12.8h, v2.8h+ stur q10, [x0, #-0x10]+ uzp2 v17.8h, v27.8h, v11.8h+ subs x13, x13, #0x1+ cbnz x13, Lmlk_polyvec_basemul_acc_montgomery_cached_k3_loop_start+ uzp2 v3.8h, v25.8h, v3.8h+ smull2 v16.4s, v1.8h, v20.8h+ smull v25.4s, v1.4h, v20.4h+ uzp1 v22.8h, v27.8h, v11.8h+ smlal2 v16.4s, v28.8h, v21.8h+ smlal v25.4s, v28.4h, v21.4h+ smlal2 v16.4s, v29.8h, v19.8h+ smlal v25.4s, v29.4h, v19.4h+ smlal2 v16.4s, v31.8h, v24.8h+ smlal v25.4s, v31.4h, v24.4h+ smlal v25.4s, v13.4h, v17.4h+ smlal2 v16.4s, v13.8h, v17.8h+ smlal2 v16.4s, v3.8h, v22.8h+ smlal v25.4s, v3.4h, v22.4h+ smlal2 v23.4s, v13.8h, v22.8h+ smlal v26.4s, v29.4h, v24.4h+ smlal v26.4s, v31.4h, v9.4h+ smlal v26.4s, v13.4h, v22.4h+ uzp1 v10.8h, v25.8h, v16.8h+ smlal2 v23.4s, v3.8h, v4.8h+ smlal v26.4s, v3.4h, v4.4h+ mul v13.8h, v10.8h, v2.8h+ smlal v18.4s, v14.4h, v0.4h+ smlal2 v8.4s, v14.8h, v0.8h+ uzp1 v3.8h, v26.8h, v23.8h+ mul v24.8h, v3.8h, v2.8h+ uzp2 v17.8h, v18.8h, v8.8h+ smlal v25.4s, v13.4h, v0.4h+ smlal2 v16.4s, v13.8h, v0.8h+ zip1 v21.8h, v7.8h, v17.8h+ zip2 v20.8h, v7.8h, v17.8h+ smlal2 v23.4s, v24.8h, v0.8h+ str q21, [x0], #0x20+ smlal v26.4s, v24.4h, v0.4h+ uzp2 v13.8h, v25.8h, v16.8h+ stur q20, [x0, #-0x10]+ uzp2 v23.8h, v26.8h, v23.8h+ zip1 v18.8h, v23.8h, v13.8h+ zip2 v13.8h, v23.8h, v13.8h+ str q18, [x0], #0x20+ stur q13, [x0, #-0x10]+ ldp d8, d9, [sp]+ .cfi_restore d8+ .cfi_restore d9+ ldp d10, d11, [sp, #0x10]+ .cfi_restore d10+ .cfi_restore d11+ ldp d12, d13, [sp, #0x20]+ .cfi_restore d12+ .cfi_restore d13+ ldp d14, d15, [sp, #0x30]+ .cfi_restore d14+ .cfi_restore d15+ add sp, sp, #0x40+ .cfi_adjust_cfa_offset -0x40+ ret+ .cfi_endproc++MLK_ASM_FN_SIZE(polyvec_basemul_acc_montgomery_cached_k3_aarch64_asm)++#endif /* MLK_ARITH_BACKEND_AARCH64 && !MLK_CONFIG_MULTILEVEL_NO_SHARED && \+ (MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 3) */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
@@ -0,0 +1,371 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [NeonNTT]+ * Neon NTT: Faster Dilithium, Kyber, and Saber on Cortex-A72 and Apple M1+ * Becker, Hwang, Kannwischer, Yang, Yang+ * https://eprint.iacr.org/2021/986+ */++/*yaml+ Name: polyvec_basemul_acc_montgomery_cached_k4_aarch64_asm+ Description: Re-implementation of asymmetric base multiplication following @[NeonNTT] for k=4+ Signature: void mlk_polyvec_basemul_acc_montgomery_cached_k4_aarch64_asm(int16_t r[256], const int16_t a[1024], const int16_t b[1024], const int16_t b_cache[512])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON]+ x0:+ type: buffer+ size_bytes: 512+ permissions: write-only+ c_parameter: int16_t r[256]+ description: Output polynomial+ x1:+ type: buffer+ size_bytes: 2048+ permissions: read-only+ c_parameter: const int16_t a[1024]+ description: Input polynomial vector a+ x2:+ type: buffer+ size_bytes: 2048+ permissions: read-only+ c_parameter: const int16_t b[1024]+ description: Input polynomial vector b+ x3:+ type: buffer+ size_bytes: 1024+ permissions: read-only+ c_parameter: const int16_t b_cache[512]+ description: Cached values for b+ Stack:+ bytes: 64+ description: saving callee-saved Neon registers+*/++/* Re-implementation of asymmetric base multiplication following @[NeonNTT] */++#include "../../../common.h"+#if defined(MLK_ARITH_BACKEND_AARCH64) && !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED) && (defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_K == 4)++/*+ * WARNING: This file is auto-derived from the mlkem-native source file+ * dev/aarch64_opt/src/mlkem_polyvec_basemul_acc_montgomery_cached_k4_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLK_ASM_NAMESPACE(polyvec_basemul_acc_montgomery_cached_k4_aarch64_asm)+MLK_ASM_FN_SYMBOL(polyvec_basemul_acc_montgomery_cached_k4_aarch64_asm)++ .cfi_startproc+ sub sp, sp, #0x40+ .cfi_adjust_cfa_offset 0x40+ stp d8, d9, [sp]+ .cfi_rel_offset d8, 0x0+ .cfi_rel_offset d9, 0x8+ stp d10, d11, [sp, #0x10]+ .cfi_rel_offset d10, 0x10+ .cfi_rel_offset d11, 0x18+ stp d12, d13, [sp, #0x20]+ .cfi_rel_offset d12, 0x20+ .cfi_rel_offset d13, 0x28+ stp d14, d15, [sp, #0x30]+ .cfi_rel_offset d14, 0x30+ .cfi_rel_offset d15, 0x38+ mov w14, #0xd01 // =3329+ dup v0.8h, w14+ mov w14, #0xcff // =3327+ dup v2.8h, w14+ add x4, x1, #0x200+ add x5, x2, #0x200+ add x6, x3, #0x100+ add x7, x1, #0x400+ add x8, x2, #0x400+ add x9, x3, #0x200+ add x10, x1, #0x600+ add x11, x2, #0x600+ add x12, x3, #0x300+ mov x13, #0x10 // =16+ ldr q28, [x1], #0x20+ ldur q5, [x1, #-0x10]+ ldr q31, [x2], #0x20+ ldur q27, [x2, #-0x10]+ ldr q7, [x5], #0x20+ ldr q10, [x4], #0x20+ ldur q18, [x5, #-0x10]+ ldur q9, [x4, #-0x10]+ uzp1 v11.8h, v28.8h, v5.8h+ uzp2 v19.8h, v28.8h, v5.8h+ uzp2 v4.8h, v31.8h, v27.8h+ uzp1 v1.8h, v31.8h, v27.8h+ ldr q29, [x7], #0x20+ ldr q28, [x8, #0x10]+ uzp1 v24.8h, v10.8h, v9.8h+ uzp1 v17.8h, v7.8h, v18.8h+ uzp2 v7.8h, v7.8h, v18.8h+ ldr q21, [x8], #0x20+ uzp2 v27.8h, v10.8h, v9.8h+ ldur q6, [x7, #-0x10]+ smull v18.4s, v11.4h, v4.4h+ ld1 { v9.8h }, [x3], #16+ smull2 v8.4s, v11.8h, v4.8h+ ldr q16, [x11], #0x20+ smlal2 v8.4s, v19.8h, v1.8h+ ldur q14, [x11, #-0x10]+ smlal v18.4s, v19.4h, v1.4h+ uzp1 v10.8h, v21.8h, v28.8h+ smlal v18.4s, v24.4h, v7.4h+ ldr q4, [x10], #0x20+ smlal2 v8.4s, v24.8h, v7.8h+ ld1 { v12.8h }, [x6], #16+ smull2 v23.4s, v11.8h, v1.8h+ uzp2 v13.8h, v29.8h, v6.8h+ smull v26.4s, v11.4h, v1.4h+ uzp1 v29.8h, v29.8h, v6.8h+ smlal v26.4s, v19.4h, v9.4h+ ldur q15, [x10, #-0x10]+ smlal2 v23.4s, v19.8h, v9.8h+ uzp2 v9.8h, v21.8h, v28.8h+ smlal v18.4s, v27.4h, v17.4h+ uzp2 v6.8h, v16.8h, v14.8h+ uzp1 v21.8h, v16.8h, v14.8h+ smlal2 v8.4s, v27.8h, v17.8h+ smlal2 v8.4s, v29.8h, v9.8h+ uzp1 v30.8h, v4.8h, v15.8h+ uzp2 v16.8h, v4.8h, v15.8h+ smlal v18.4s, v29.4h, v9.4h+ smlal2 v8.4s, v13.8h, v10.8h+ ld1 { v15.8h }, [x9], #16+ smlal v18.4s, v13.4h, v10.4h+ ldr q11, [x4], #0x20+ smlal v18.4s, v30.4h, v6.4h+ ldr q7, [x2], #0x20+ smlal2 v8.4s, v30.8h, v6.8h+ ld1 { v9.8h }, [x12], #16+ smlal2 v23.4s, v24.8h, v17.8h+ ldur q4, [x2, #-0x10]+ smlal v26.4s, v24.4h, v17.4h+ ldur q25, [x4, #-0x10]+ smlal2 v8.4s, v16.8h, v21.8h+ ldr q5, [x5], #0x20+ smlal v18.4s, v16.4h, v21.4h+ ldur q22, [x5, #-0x10]+ smlal v26.4s, v27.4h, v12.4h+ ldr q19, [x1, #0x10]+ smlal v26.4s, v29.4h, v10.4h+ ld1 { v20.8h }, [x3], #16+ smlal v26.4s, v13.4h, v15.4h+ uzp1 v24.8h, v7.8h, v4.8h+ smlal2 v23.4s, v27.8h, v12.8h+ uzp1 v28.8h, v18.8h, v8.8h+ smlal v26.4s, v30.4h, v21.4h+ uzp2 v27.8h, v11.8h, v25.8h+ smlal2 v23.4s, v29.8h, v10.8h+ uzp2 v31.8h, v7.8h, v4.8h+ smlal2 v23.4s, v13.8h, v15.8h+ uzp1 v14.8h, v5.8h, v22.8h+ uzp1 v17.8h, v11.8h, v25.8h+ smlal v26.4s, v16.4h, v9.4h+ mul v29.8h, v28.8h, v2.8h+ sub x13, x13, #0x2++Lmlk_polyvec_basemul_acc_montgomery_cached_k4_loop_start:+ smlal2 v23.4s, v30.8h, v21.8h+ ldr q11, [x1], #0x20+ uzp2 v15.8h, v5.8h, v22.8h+ smlal v18.4s, v29.4h, v0.4h+ ldr q12, [x7], #0x20+ smlal2 v8.4s, v29.8h, v0.8h+ ldur q3, [x7, #-0x10]+ ldr q21, [x8], #0x20+ uzp1 v29.8h, v11.8h, v19.8h+ ldur q13, [x8, #-0x10]+ uzp2 v5.8h, v11.8h, v19.8h+ smlal2 v23.4s, v16.8h, v9.8h+ uzp2 v28.8h, v18.8h, v8.8h+ smull2 v8.4s, v29.8h, v31.8h+ smlal2 v8.4s, v5.8h, v24.8h+ uzp1 v7.8h, v12.8h, v3.8h+ smlal2 v8.4s, v17.8h, v15.8h+ uzp2 v11.8h, v21.8h, v13.8h+ uzp1 v4.8h, v26.8h, v23.8h+ smlal2 v8.4s, v27.8h, v14.8h+ smlal2 v8.4s, v7.8h, v11.8h+ mul v6.8h, v4.8h, v2.8h+ ldr q19, [x11], #0x20+ uzp2 v25.8h, v12.8h, v3.8h+ ldr q12, [x10], #0x20+ smull v18.4s, v29.4h, v31.4h+ ldur q3, [x10, #-0x10]+ smlal v18.4s, v5.4h, v24.4h+ uzp1 v4.8h, v21.8h, v13.8h+ smlal v18.4s, v17.4h, v15.4h+ ldur q13, [x11, #-0x10]+ ld1 { v1.8h }, [x6], #16+ smlal v26.4s, v6.4h, v0.4h+ smlal2 v23.4s, v6.8h, v0.8h+ ld1 { v10.8h }, [x9], #16+ smlal v18.4s, v27.4h, v14.4h+ uzp1 v30.8h, v12.8h, v3.8h+ smlal2 v8.4s, v25.8h, v4.8h+ uzp2 v31.8h, v19.8h, v13.8h+ smlal v18.4s, v7.4h, v11.4h+ ld1 { v9.8h }, [x12], #16+ smlal v18.4s, v25.4h, v4.4h+ uzp1 v21.8h, v19.8h, v13.8h+ uzp2 v16.8h, v12.8h, v3.8h+ smlal v18.4s, v30.4h, v31.4h+ smlal2 v8.4s, v30.8h, v31.8h+ uzp2 v31.8h, v26.8h, v23.8h+ smlal2 v8.4s, v16.8h, v21.8h+ smlal v18.4s, v16.4h, v21.4h+ zip1 v15.8h, v31.8h, v28.8h+ ldr q19, [x1, #0x10]+ smull2 v23.4s, v29.8h, v24.8h+ smull v26.4s, v29.4h, v24.4h+ ldr q3, [x2, #0x10]+ smlal v26.4s, v5.4h, v20.4h+ ldr q11, [x2], #0x20+ uzp1 v6.8h, v18.8h, v8.8h+ smlal v26.4s, v17.4h, v14.4h+ smlal v26.4s, v27.4h, v1.4h+ zip2 v13.8h, v31.8h, v28.8h+ smlal v26.4s, v7.4h, v4.4h+ str q15, [x0], #0x20+ smlal v26.4s, v25.4h, v10.4h+ stur q13, [x0, #-0x10]+ mul v29.8h, v6.8h, v2.8h+ uzp1 v24.8h, v11.8h, v3.8h+ uzp2 v31.8h, v11.8h, v3.8h+ ldr q11, [x4], #0x20+ smlal2 v23.4s, v5.8h, v20.8h+ ldur q28, [x4, #-0x10]+ smlal2 v23.4s, v17.8h, v14.8h+ ldr q5, [x5], #0x20+ smlal2 v23.4s, v27.8h, v1.8h+ ldur q22, [x5, #-0x10]+ smlal v26.4s, v30.4h, v21.4h+ ld1 { v20.8h }, [x3], #16+ smlal v26.4s, v16.4h, v9.4h+ uzp1 v17.8h, v11.8h, v28.8h+ smlal2 v23.4s, v7.8h, v4.8h+ uzp2 v27.8h, v11.8h, v28.8h+ smlal2 v23.4s, v25.8h, v10.8h+ uzp1 v14.8h, v5.8h, v22.8h+ subs x13, x13, #0x1+ cbnz x13, Lmlk_polyvec_basemul_acc_montgomery_cached_k4_loop_start+ smlal v18.4s, v29.4h, v0.4h+ ldr q11, [x1], #0x20+ uzp2 v28.8h, v5.8h, v22.8h+ smlal2 v23.4s, v30.8h, v21.8h+ smlal2 v8.4s, v29.8h, v0.8h+ ldr q15, [x8, #0x10]+ smlal2 v23.4s, v16.8h, v9.8h+ ldr q21, [x8], #0x20+ uzp1 v22.8h, v11.8h, v19.8h+ uzp2 v12.8h, v11.8h, v19.8h+ ldr q1, [x7, #0x10]+ ld1 { v6.8h }, [x6], #16+ uzp2 v3.8h, v18.8h, v8.8h+ smull v9.4s, v22.4h, v31.4h+ smull2 v18.4s, v22.8h, v31.8h+ ldr q16, [x7], #0x20+ smull v19.4s, v22.4h, v24.4h+ uzp1 v30.8h, v21.8h, v15.8h+ uzp2 v25.8h, v21.8h, v15.8h+ smull2 v8.4s, v22.8h, v24.8h+ smlal v19.4s, v12.4h, v20.4h+ ldr q13, [x10, #0x10]+ smlal2 v8.4s, v12.8h, v20.8h+ uzp1 v29.8h, v16.8h, v1.8h+ smlal2 v18.4s, v12.8h, v24.8h+ ldr q5, [x10], #0x20+ smlal v9.4s, v12.4h, v24.4h+ ldr q4, [x11], #0x20+ smlal v9.4s, v17.4h, v28.4h+ ldur q22, [x11, #-0x10]+ smlal2 v18.4s, v17.8h, v28.8h+ uzp2 v16.8h, v16.8h, v1.8h+ smlal v19.4s, v17.4h, v14.4h+ ld1 { v28.8h }, [x9], #16+ smlal2 v8.4s, v17.8h, v14.8h+ uzp1 v7.8h, v5.8h, v13.8h+ smlal v9.4s, v27.4h, v14.4h+ uzp1 v17.8h, v4.8h, v22.8h+ smlal2 v18.4s, v27.8h, v14.8h+ uzp2 v12.8h, v5.8h, v13.8h+ uzp2 v21.8h, v4.8h, v22.8h+ smlal v19.4s, v27.4h, v6.4h+ smlal2 v8.4s, v27.8h, v6.8h+ ld1 { v15.8h }, [x12], #16+ smlal v19.4s, v29.4h, v30.4h+ uzp1 v20.8h, v26.8h, v23.8h+ smlal v9.4s, v29.4h, v25.4h+ smlal2 v18.4s, v29.8h, v25.8h+ smlal2 v8.4s, v29.8h, v30.8h+ smlal v19.4s, v16.4h, v28.4h+ smlal2 v8.4s, v16.8h, v28.8h+ smlal2 v18.4s, v16.8h, v30.8h+ smlal v9.4s, v16.4h, v30.4h+ smlal v9.4s, v7.4h, v21.4h+ smlal2 v18.4s, v7.8h, v21.8h+ smlal2 v8.4s, v7.8h, v17.8h+ smlal v19.4s, v7.4h, v17.4h+ smlal v19.4s, v12.4h, v15.4h+ smlal2 v8.4s, v12.8h, v15.8h+ smlal2 v18.4s, v12.8h, v17.8h+ smlal v9.4s, v12.4h, v17.4h+ mul v6.8h, v20.8h, v2.8h+ uzp1 v4.8h, v19.8h, v8.8h+ mul v17.8h, v4.8h, v2.8h+ uzp1 v12.8h, v9.8h, v18.8h+ smlal v26.4s, v6.4h, v0.4h+ mul v21.8h, v12.8h, v2.8h+ smlal2 v23.4s, v6.8h, v0.8h+ smlal2 v8.4s, v17.8h, v0.8h+ smlal v19.4s, v17.4h, v0.4h+ smlal2 v18.4s, v21.8h, v0.8h+ uzp2 v23.8h, v26.8h, v23.8h+ smlal v9.4s, v21.4h, v0.4h+ zip2 v12.8h, v23.8h, v3.8h+ zip1 v22.8h, v23.8h, v3.8h+ uzp2 v14.8h, v19.8h, v8.8h+ uzp2 v18.8h, v9.8h, v18.8h+ str q12, [x0, #0x10]+ str q22, [x0], #0x20+ zip2 v24.8h, v14.8h, v18.8h+ zip1 v21.8h, v14.8h, v18.8h+ str q24, [x0, #0x10]+ str q21, [x0], #0x20+ ldp d8, d9, [sp]+ .cfi_restore d8+ .cfi_restore d9+ ldp d10, d11, [sp, #0x10]+ .cfi_restore d10+ .cfi_restore d11+ ldp d12, d13, [sp, #0x20]+ .cfi_restore d12+ .cfi_restore d13+ ldp d14, d15, [sp, #0x30]+ .cfi_restore d14+ .cfi_restore d15+ add sp, sp, #0x40+ .cfi_adjust_cfa_offset -0x40+ ret+ .cfi_endproc++MLK_ASM_FN_SIZE(polyvec_basemul_acc_montgomery_cached_k4_aarch64_asm)++#endif /* MLK_ARITH_BACKEND_AARCH64 && !MLK_CONFIG_MULTILEVEL_NO_SHARED && \+ (MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 4) */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
@@ -0,0 +1,226 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*yaml+ Name: rej_uniform_aarch64_asm+ Description: Run rejection sampling on uniform random bytes to generate uniform random integers mod q+ Signature: uint64_t mlk_rej_uniform_aarch64_asm(int16_t r[256], const uint8_t *buf, unsigned buflen, const uint8_t table[4096])+ ABI:+ Architecture: aarch64+ CallingConvention: AAPCS64+ Features: [NEON]+ x0:+ type: buffer+ size_bytes: 512+ permissions: write-only+ c_parameter: int16_t r[256]+ description: Output buffer+ x1:+ type: buffer+ size_bytes: x2+ permissions: read-only+ c_parameter: const uint8_t *buf+ description: Input buffer+ x2:+ type: scalar+ c_parameter: unsigned buflen+ description: Length of input buffer (must be multiple of 24)+ test_with: 504 # MLKEM_GEN_MATRIX_NBLOCKS * MLK_XOF_RATE+ x3:+ type: buffer+ size_bytes: 4096+ permissions: read-only+ c_parameter: const uint8_t table[4096]+ description: Lookup table+ Stack:+ bytes: 576+ description: register preservation and temporary storage+*/++#include "../../../common.h"+#if defined(MLK_ARITH_BACKEND_AARCH64) && \+ !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED)++/*+ * WARNING: This file is auto-derived from the mlkem-native source file+ * dev/aarch64_opt/src/mlkem_rej_uniform_aarch64_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLK_ASM_NAMESPACE(rej_uniform_aarch64_asm)+MLK_ASM_FN_SYMBOL(rej_uniform_aarch64_asm)++ .cfi_startproc+ sub sp, sp, #0x240+ .cfi_adjust_cfa_offset 0x240+ mov x7, #0x1 // =1+ movk x7, #0x2, lsl #16+ movk x7, #0x4, lsl #32+ movk x7, #0x8, lsl #48+ mov v31.d[0], x7+ mov x7, #0x10 // =16+ movk x7, #0x20, lsl #16+ movk x7, #0x40, lsl #32+ movk x7, #0x80, lsl #48+ mov v31.d[1], x7+ mov w11, #0xd01 // =3329+ dup v30.8h, w11+ mov x8, sp+ mov x7, x8+ mov x11, #0x0 // =0+ eor v16.16b, v16.16b, v16.16b++Lmlk_rej_uniform_initial_zero:+ str q16, [x7], #0x40+ stur q16, [x7, #-0x30]+ stur q16, [x7, #-0x20]+ stur q16, [x7, #-0x10]+ add x11, x11, #0x20+ cmp x11, #0x100+ b.lt Lmlk_rej_uniform_initial_zero+ mov x7, x8+ mov x9, #0x0 // =0+ mov x4, #0x100 // =256+ cmp x2, #0x30+ b.lo Lmlk_rej_uniform_loop48_end++Lmlk_rej_uniform_loop48:+ cmp x9, x4+ b.hs Lmlk_rej_uniform_memory_copy+ sub x2, x2, #0x30+ ld3 { v0.16b, v1.16b, v2.16b }, [x1], #48+ zip1 v4.16b, v0.16b, v1.16b+ zip2 v5.16b, v0.16b, v1.16b+ zip1 v6.16b, v1.16b, v2.16b+ zip2 v7.16b, v1.16b, v2.16b+ bic v4.8h, #0xf0, lsl #8+ bic v5.8h, #0xf0, lsl #8+ ushr v6.8h, v6.8h, #0x4+ ushr v7.8h, v7.8h, #0x4+ zip1 v16.8h, v4.8h, v6.8h+ zip2 v17.8h, v4.8h, v6.8h+ zip1 v18.8h, v5.8h, v7.8h+ zip2 v19.8h, v5.8h, v7.8h+ cmhi v4.8h, v30.8h, v16.8h+ cmhi v5.8h, v30.8h, v17.8h+ cmhi v6.8h, v30.8h, v18.8h+ cmhi v7.8h, v30.8h, v19.8h+ and v4.16b, v4.16b, v31.16b+ and v5.16b, v5.16b, v31.16b+ and v6.16b, v6.16b, v31.16b+ and v7.16b, v7.16b, v31.16b+ uaddlv s20, v4.8h+ uaddlv s21, v5.8h+ uaddlv s22, v6.8h+ uaddlv s23, v7.8h+ fmov w12, s20+ fmov w13, s21+ fmov w14, s22+ fmov w15, s23+ ldr q24, [x3, x12, lsl #4]+ ldr q25, [x3, x13, lsl #4]+ ldr q26, [x3, x14, lsl #4]+ ldr q27, [x3, x15, lsl #4]+ cnt v4.16b, v4.16b+ cnt v5.16b, v5.16b+ cnt v6.16b, v6.16b+ cnt v7.16b, v7.16b+ uaddlv s20, v4.8h+ uaddlv s21, v5.8h+ uaddlv s22, v6.8h+ uaddlv s23, v7.8h+ fmov w12, s20+ fmov w13, s21+ fmov w14, s22+ fmov w15, s23+ tbl v16.16b, { v16.16b }, v24.16b+ tbl v17.16b, { v17.16b }, v25.16b+ tbl v18.16b, { v18.16b }, v26.16b+ tbl v19.16b, { v19.16b }, v27.16b+ st1 { v16.8h }, [x7]+ add x7, x7, x12, lsl #1+ st1 { v17.8h }, [x7]+ add x7, x7, x13, lsl #1+ st1 { v18.8h }, [x7]+ add x7, x7, x14, lsl #1+ st1 { v19.8h }, [x7]+ add x7, x7, x15, lsl #1+ add x12, x12, x13+ add x14, x14, x15+ add x9, x9, x12+ add x9, x9, x14+ cmp x2, #0x30+ b.hs Lmlk_rej_uniform_loop48++Lmlk_rej_uniform_loop48_end:+ cmp x9, x4+ b.hs Lmlk_rej_uniform_memory_copy+ cmp x2, #0x18+ b.lo Lmlk_rej_uniform_memory_copy+ ld3 { v0.8b, v1.8b, v2.8b }, [x1], #24+ zip1 v4.16b, v0.16b, v1.16b+ zip1 v5.16b, v1.16b, v2.16b+ bic v4.8h, #0xf0, lsl #8+ ushr v5.8h, v5.8h, #0x4+ zip1 v16.8h, v4.8h, v5.8h+ zip2 v17.8h, v4.8h, v5.8h+ cmhi v4.8h, v30.8h, v16.8h+ cmhi v5.8h, v30.8h, v17.8h+ and v4.16b, v4.16b, v31.16b+ and v5.16b, v5.16b, v31.16b+ uaddlv s20, v4.8h+ uaddlv s21, v5.8h+ fmov w12, s20+ fmov w13, s21+ ldr q24, [x3, x12, lsl #4]+ ldr q25, [x3, x13, lsl #4]+ cnt v4.16b, v4.16b+ cnt v5.16b, v5.16b+ uaddlv s20, v4.8h+ uaddlv s21, v5.8h+ fmov w12, s20+ fmov w13, s21+ tbl v16.16b, { v16.16b }, v24.16b+ tbl v17.16b, { v17.16b }, v25.16b+ st1 { v16.8h }, [x7]+ add x7, x7, x12, lsl #1+ st1 { v17.8h }, [x7]+ add x9, x9, x12+ add x9, x9, x13++Lmlk_rej_uniform_memory_copy:+ cmp x9, x4+ csel x9, x9, x4, lo+ mov x11, #0x0 // =0+ mov x7, x8++Lmlk_rej_uniform_final_copy:+ ldr q16, [x7], #0x40+ ldur q17, [x7, #-0x30]+ ldur q18, [x7, #-0x20]+ ldur q19, [x7, #-0x10]+ str q16, [x0], #0x40+ stur q17, [x0, #-0x30]+ stur q18, [x0, #-0x20]+ stur q19, [x0, #-0x10]+ add x11, x11, #0x20+ cmp x11, #0x100+ b.lt Lmlk_rej_uniform_final_copy+ mov x0, x9++Lmlk_rej_uniform_return:+ add sp, sp, #0x240+ .cfi_adjust_cfa_offset -0x240+ ret+ .cfi_endproc++MLK_ASM_FN_SIZE(rej_uniform_aarch64_asm)++#endif /* MLK_ARITH_BACKEND_AARCH64 && !MLK_CONFIG_MULTILEVEL_NO_SHARED */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
@@ -0,0 +1,543 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*+ * WARNING: This file is auto-generated from scripts/autogen+ * in the mlkem-native repository.+ * Do not modify it directly.+ */++#include "../../../common.h"++#if defined(MLK_ARITH_BACKEND_AARCH64) && \+ !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED)++#include "arith_native_aarch64.h"++/*+ * Lookup table used by rejection sampling of the public matrix.+ * See autogen for details.+ */+MLK_ALIGN MLK_INTERNAL_DATA_DEFINITION const uint8_t+ mlk_rej_uniform_table[4096] = {+ 255, 255, 255, 255, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 0 */,+ 0, 1, 255, 255, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 1 */,+ 2, 3, 255, 255, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 2 */,+ 0, 1, 2, 3, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 3 */,+ 4, 5, 255, 255, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 4 */,+ 0, 1, 4, 5, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 5 */,+ 2, 3, 4, 5, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 6 */,+ 0, 1, 2, 3, 4, 5, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 7 */,+ 6, 7, 255, 255, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 8 */,+ 0, 1, 6, 7, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 9 */,+ 2, 3, 6, 7, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 10 */,+ 0, 1, 2, 3, 6, 7, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 11 */,+ 4, 5, 6, 7, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 12 */,+ 0, 1, 4, 5, 6, 7, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 13 */,+ 2, 3, 4, 5, 6, 7, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 14 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 15 */,+ 8, 9, 255, 255, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 16 */,+ 0, 1, 8, 9, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 17 */,+ 2, 3, 8, 9, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 18 */,+ 0, 1, 2, 3, 8, 9, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 19 */,+ 4, 5, 8, 9, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 20 */,+ 0, 1, 4, 5, 8, 9, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 21 */,+ 2, 3, 4, 5, 8, 9, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 22 */,+ 0, 1, 2, 3, 4, 5, 8, 9,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 23 */,+ 6, 7, 8, 9, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 24 */,+ 0, 1, 6, 7, 8, 9, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 25 */,+ 2, 3, 6, 7, 8, 9, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 26 */,+ 0, 1, 2, 3, 6, 7, 8, 9,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 27 */,+ 4, 5, 6, 7, 8, 9, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 28 */,+ 0, 1, 4, 5, 6, 7, 8, 9,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 29 */,+ 2, 3, 4, 5, 6, 7, 8, 9,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 30 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 8, 9, 255, 255, 255, 255, 255, 255 /* 31 */,+ 10, 11, 255, 255, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 32 */,+ 0, 1, 10, 11, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 33 */,+ 2, 3, 10, 11, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 34 */,+ 0, 1, 2, 3, 10, 11, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 35 */,+ 4, 5, 10, 11, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 36 */,+ 0, 1, 4, 5, 10, 11, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 37 */,+ 2, 3, 4, 5, 10, 11, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 38 */,+ 0, 1, 2, 3, 4, 5, 10, 11,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 39 */,+ 6, 7, 10, 11, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 40 */,+ 0, 1, 6, 7, 10, 11, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 41 */,+ 2, 3, 6, 7, 10, 11, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 42 */,+ 0, 1, 2, 3, 6, 7, 10, 11,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 43 */,+ 4, 5, 6, 7, 10, 11, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 44 */,+ 0, 1, 4, 5, 6, 7, 10, 11,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 45 */,+ 2, 3, 4, 5, 6, 7, 10, 11,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 46 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 10, 11, 255, 255, 255, 255, 255, 255 /* 47 */,+ 8, 9, 10, 11, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 48 */,+ 0, 1, 8, 9, 10, 11, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 49 */,+ 2, 3, 8, 9, 10, 11, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 50 */,+ 0, 1, 2, 3, 8, 9, 10, 11,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 51 */,+ 4, 5, 8, 9, 10, 11, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 52 */,+ 0, 1, 4, 5, 8, 9, 10, 11,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 53 */,+ 2, 3, 4, 5, 8, 9, 10, 11,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 54 */,+ 0, 1, 2, 3, 4, 5, 8, 9,+ 10, 11, 255, 255, 255, 255, 255, 255 /* 55 */,+ 6, 7, 8, 9, 10, 11, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 56 */,+ 0, 1, 6, 7, 8, 9, 10, 11,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 57 */,+ 2, 3, 6, 7, 8, 9, 10, 11,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 58 */,+ 0, 1, 2, 3, 6, 7, 8, 9,+ 10, 11, 255, 255, 255, 255, 255, 255 /* 59 */,+ 4, 5, 6, 7, 8, 9, 10, 11,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 60 */,+ 0, 1, 4, 5, 6, 7, 8, 9,+ 10, 11, 255, 255, 255, 255, 255, 255 /* 61 */,+ 2, 3, 4, 5, 6, 7, 8, 9,+ 10, 11, 255, 255, 255, 255, 255, 255 /* 62 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 8, 9, 10, 11, 255, 255, 255, 255 /* 63 */,+ 12, 13, 255, 255, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 64 */,+ 0, 1, 12, 13, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 65 */,+ 2, 3, 12, 13, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 66 */,+ 0, 1, 2, 3, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 67 */,+ 4, 5, 12, 13, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 68 */,+ 0, 1, 4, 5, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 69 */,+ 2, 3, 4, 5, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 70 */,+ 0, 1, 2, 3, 4, 5, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 71 */,+ 6, 7, 12, 13, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 72 */,+ 0, 1, 6, 7, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 73 */,+ 2, 3, 6, 7, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 74 */,+ 0, 1, 2, 3, 6, 7, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 75 */,+ 4, 5, 6, 7, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 76 */,+ 0, 1, 4, 5, 6, 7, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 77 */,+ 2, 3, 4, 5, 6, 7, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 78 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 79 */,+ 8, 9, 12, 13, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 80 */,+ 0, 1, 8, 9, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 81 */,+ 2, 3, 8, 9, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 82 */,+ 0, 1, 2, 3, 8, 9, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 83 */,+ 4, 5, 8, 9, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 84 */,+ 0, 1, 4, 5, 8, 9, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 85 */,+ 2, 3, 4, 5, 8, 9, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 86 */,+ 0, 1, 2, 3, 4, 5, 8, 9,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 87 */,+ 6, 7, 8, 9, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 88 */,+ 0, 1, 6, 7, 8, 9, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 89 */,+ 2, 3, 6, 7, 8, 9, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 90 */,+ 0, 1, 2, 3, 6, 7, 8, 9,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 91 */,+ 4, 5, 6, 7, 8, 9, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 92 */,+ 0, 1, 4, 5, 6, 7, 8, 9,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 93 */,+ 2, 3, 4, 5, 6, 7, 8, 9,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 94 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 8, 9, 12, 13, 255, 255, 255, 255 /* 95 */,+ 10, 11, 12, 13, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 96 */,+ 0, 1, 10, 11, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 97 */,+ 2, 3, 10, 11, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 98 */,+ 0, 1, 2, 3, 10, 11, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 99 */,+ 4, 5, 10, 11, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 100 */,+ 0, 1, 4, 5, 10, 11, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 101 */,+ 2, 3, 4, 5, 10, 11, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 102 */,+ 0, 1, 2, 3, 4, 5, 10, 11,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 103 */,+ 6, 7, 10, 11, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 104 */,+ 0, 1, 6, 7, 10, 11, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 105 */,+ 2, 3, 6, 7, 10, 11, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 106 */,+ 0, 1, 2, 3, 6, 7, 10, 11,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 107 */,+ 4, 5, 6, 7, 10, 11, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 108 */,+ 0, 1, 4, 5, 6, 7, 10, 11,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 109 */,+ 2, 3, 4, 5, 6, 7, 10, 11,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 110 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 10, 11, 12, 13, 255, 255, 255, 255 /* 111 */,+ 8, 9, 10, 11, 12, 13, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 112 */,+ 0, 1, 8, 9, 10, 11, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 113 */,+ 2, 3, 8, 9, 10, 11, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 114 */,+ 0, 1, 2, 3, 8, 9, 10, 11,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 115 */,+ 4, 5, 8, 9, 10, 11, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 116 */,+ 0, 1, 4, 5, 8, 9, 10, 11,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 117 */,+ 2, 3, 4, 5, 8, 9, 10, 11,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 118 */,+ 0, 1, 2, 3, 4, 5, 8, 9,+ 10, 11, 12, 13, 255, 255, 255, 255 /* 119 */,+ 6, 7, 8, 9, 10, 11, 12, 13,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 120 */,+ 0, 1, 6, 7, 8, 9, 10, 11,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 121 */,+ 2, 3, 6, 7, 8, 9, 10, 11,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 122 */,+ 0, 1, 2, 3, 6, 7, 8, 9,+ 10, 11, 12, 13, 255, 255, 255, 255 /* 123 */,+ 4, 5, 6, 7, 8, 9, 10, 11,+ 12, 13, 255, 255, 255, 255, 255, 255 /* 124 */,+ 0, 1, 4, 5, 6, 7, 8, 9,+ 10, 11, 12, 13, 255, 255, 255, 255 /* 125 */,+ 2, 3, 4, 5, 6, 7, 8, 9,+ 10, 11, 12, 13, 255, 255, 255, 255 /* 126 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 8, 9, 10, 11, 12, 13, 255, 255 /* 127 */,+ 14, 15, 255, 255, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 128 */,+ 0, 1, 14, 15, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 129 */,+ 2, 3, 14, 15, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 130 */,+ 0, 1, 2, 3, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 131 */,+ 4, 5, 14, 15, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 132 */,+ 0, 1, 4, 5, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 133 */,+ 2, 3, 4, 5, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 134 */,+ 0, 1, 2, 3, 4, 5, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 135 */,+ 6, 7, 14, 15, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 136 */,+ 0, 1, 6, 7, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 137 */,+ 2, 3, 6, 7, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 138 */,+ 0, 1, 2, 3, 6, 7, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 139 */,+ 4, 5, 6, 7, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 140 */,+ 0, 1, 4, 5, 6, 7, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 141 */,+ 2, 3, 4, 5, 6, 7, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 142 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 143 */,+ 8, 9, 14, 15, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 144 */,+ 0, 1, 8, 9, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 145 */,+ 2, 3, 8, 9, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 146 */,+ 0, 1, 2, 3, 8, 9, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 147 */,+ 4, 5, 8, 9, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 148 */,+ 0, 1, 4, 5, 8, 9, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 149 */,+ 2, 3, 4, 5, 8, 9, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 150 */,+ 0, 1, 2, 3, 4, 5, 8, 9,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 151 */,+ 6, 7, 8, 9, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 152 */,+ 0, 1, 6, 7, 8, 9, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 153 */,+ 2, 3, 6, 7, 8, 9, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 154 */,+ 0, 1, 2, 3, 6, 7, 8, 9,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 155 */,+ 4, 5, 6, 7, 8, 9, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 156 */,+ 0, 1, 4, 5, 6, 7, 8, 9,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 157 */,+ 2, 3, 4, 5, 6, 7, 8, 9,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 158 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 8, 9, 14, 15, 255, 255, 255, 255 /* 159 */,+ 10, 11, 14, 15, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 160 */,+ 0, 1, 10, 11, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 161 */,+ 2, 3, 10, 11, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 162 */,+ 0, 1, 2, 3, 10, 11, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 163 */,+ 4, 5, 10, 11, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 164 */,+ 0, 1, 4, 5, 10, 11, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 165 */,+ 2, 3, 4, 5, 10, 11, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 166 */,+ 0, 1, 2, 3, 4, 5, 10, 11,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 167 */,+ 6, 7, 10, 11, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 168 */,+ 0, 1, 6, 7, 10, 11, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 169 */,+ 2, 3, 6, 7, 10, 11, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 170 */,+ 0, 1, 2, 3, 6, 7, 10, 11,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 171 */,+ 4, 5, 6, 7, 10, 11, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 172 */,+ 0, 1, 4, 5, 6, 7, 10, 11,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 173 */,+ 2, 3, 4, 5, 6, 7, 10, 11,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 174 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 10, 11, 14, 15, 255, 255, 255, 255 /* 175 */,+ 8, 9, 10, 11, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 176 */,+ 0, 1, 8, 9, 10, 11, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 177 */,+ 2, 3, 8, 9, 10, 11, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 178 */,+ 0, 1, 2, 3, 8, 9, 10, 11,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 179 */,+ 4, 5, 8, 9, 10, 11, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 180 */,+ 0, 1, 4, 5, 8, 9, 10, 11,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 181 */,+ 2, 3, 4, 5, 8, 9, 10, 11,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 182 */,+ 0, 1, 2, 3, 4, 5, 8, 9,+ 10, 11, 14, 15, 255, 255, 255, 255 /* 183 */,+ 6, 7, 8, 9, 10, 11, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 184 */,+ 0, 1, 6, 7, 8, 9, 10, 11,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 185 */,+ 2, 3, 6, 7, 8, 9, 10, 11,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 186 */,+ 0, 1, 2, 3, 6, 7, 8, 9,+ 10, 11, 14, 15, 255, 255, 255, 255 /* 187 */,+ 4, 5, 6, 7, 8, 9, 10, 11,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 188 */,+ 0, 1, 4, 5, 6, 7, 8, 9,+ 10, 11, 14, 15, 255, 255, 255, 255 /* 189 */,+ 2, 3, 4, 5, 6, 7, 8, 9,+ 10, 11, 14, 15, 255, 255, 255, 255 /* 190 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 8, 9, 10, 11, 14, 15, 255, 255 /* 191 */,+ 12, 13, 14, 15, 255, 255, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 192 */,+ 0, 1, 12, 13, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 193 */,+ 2, 3, 12, 13, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 194 */,+ 0, 1, 2, 3, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 195 */,+ 4, 5, 12, 13, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 196 */,+ 0, 1, 4, 5, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 197 */,+ 2, 3, 4, 5, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 198 */,+ 0, 1, 2, 3, 4, 5, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 199 */,+ 6, 7, 12, 13, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 200 */,+ 0, 1, 6, 7, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 201 */,+ 2, 3, 6, 7, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 202 */,+ 0, 1, 2, 3, 6, 7, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 203 */,+ 4, 5, 6, 7, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 204 */,+ 0, 1, 4, 5, 6, 7, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 205 */,+ 2, 3, 4, 5, 6, 7, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 206 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 207 */,+ 8, 9, 12, 13, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 208 */,+ 0, 1, 8, 9, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 209 */,+ 2, 3, 8, 9, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 210 */,+ 0, 1, 2, 3, 8, 9, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 211 */,+ 4, 5, 8, 9, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 212 */,+ 0, 1, 4, 5, 8, 9, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 213 */,+ 2, 3, 4, 5, 8, 9, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 214 */,+ 0, 1, 2, 3, 4, 5, 8, 9,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 215 */,+ 6, 7, 8, 9, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 216 */,+ 0, 1, 6, 7, 8, 9, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 217 */,+ 2, 3, 6, 7, 8, 9, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 218 */,+ 0, 1, 2, 3, 6, 7, 8, 9,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 219 */,+ 4, 5, 6, 7, 8, 9, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 220 */,+ 0, 1, 4, 5, 6, 7, 8, 9,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 221 */,+ 2, 3, 4, 5, 6, 7, 8, 9,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 222 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 8, 9, 12, 13, 14, 15, 255, 255 /* 223 */,+ 10, 11, 12, 13, 14, 15, 255, 255,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 224 */,+ 0, 1, 10, 11, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 225 */,+ 2, 3, 10, 11, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 226 */,+ 0, 1, 2, 3, 10, 11, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 227 */,+ 4, 5, 10, 11, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 228 */,+ 0, 1, 4, 5, 10, 11, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 229 */,+ 2, 3, 4, 5, 10, 11, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 230 */,+ 0, 1, 2, 3, 4, 5, 10, 11,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 231 */,+ 6, 7, 10, 11, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 232 */,+ 0, 1, 6, 7, 10, 11, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 233 */,+ 2, 3, 6, 7, 10, 11, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 234 */,+ 0, 1, 2, 3, 6, 7, 10, 11,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 235 */,+ 4, 5, 6, 7, 10, 11, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 236 */,+ 0, 1, 4, 5, 6, 7, 10, 11,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 237 */,+ 2, 3, 4, 5, 6, 7, 10, 11,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 238 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 10, 11, 12, 13, 14, 15, 255, 255 /* 239 */,+ 8, 9, 10, 11, 12, 13, 14, 15,+ 255, 255, 255, 255, 255, 255, 255, 255 /* 240 */,+ 0, 1, 8, 9, 10, 11, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 241 */,+ 2, 3, 8, 9, 10, 11, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 242 */,+ 0, 1, 2, 3, 8, 9, 10, 11,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 243 */,+ 4, 5, 8, 9, 10, 11, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 244 */,+ 0, 1, 4, 5, 8, 9, 10, 11,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 245 */,+ 2, 3, 4, 5, 8, 9, 10, 11,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 246 */,+ 0, 1, 2, 3, 4, 5, 8, 9,+ 10, 11, 12, 13, 14, 15, 255, 255 /* 247 */,+ 6, 7, 8, 9, 10, 11, 12, 13,+ 14, 15, 255, 255, 255, 255, 255, 255 /* 248 */,+ 0, 1, 6, 7, 8, 9, 10, 11,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 249 */,+ 2, 3, 6, 7, 8, 9, 10, 11,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 250 */,+ 0, 1, 2, 3, 6, 7, 8, 9,+ 10, 11, 12, 13, 14, 15, 255, 255 /* 251 */,+ 4, 5, 6, 7, 8, 9, 10, 11,+ 12, 13, 14, 15, 255, 255, 255, 255 /* 252 */,+ 0, 1, 4, 5, 6, 7, 8, 9,+ 10, 11, 12, 13, 14, 15, 255, 255 /* 253 */,+ 2, 3, 4, 5, 6, 7, 8, 9,+ 10, 11, 12, 13, 14, 15, 255, 255 /* 254 */,+ 0, 1, 2, 3, 4, 5, 6, 7,+ 8, 9, 10, 11, 12, 13, 14, 15 /* 255 */,+};++#else /* MLK_ARITH_BACKEND_AARCH64 && !MLK_CONFIG_MULTILEVEL_NO_SHARED */++MLK_EMPTY_CU(aarch64_rej_uniform_table)++#endif /* !(MLK_ARITH_BACKEND_AARCH64 && !MLK_CONFIG_MULTILEVEL_NO_SHARED) */
@@ -0,0 +1,651 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [FIPS203]+ * FIPS 203 Module-Lattice-Based Key-Encapsulation Mechanism Standard+ * National Institute of Standards and Technology+ * https://csrc.nist.gov/pubs/fips/203/final+ */++#ifndef MLK_NATIVE_API_H+#define MLK_NATIVE_API_H+/*+ * Native arithmetic interface+ *+ * This header is primarily for documentation purposes.+ * It should not be included by backend implementations.+ *+ * To ensure consistency with backends, the header will be+ * included automatically after inclusion of the active+ * backend, to ensure consistency of function signatures,+ * and run sanity checks.+ */++#include "../cbmc.h"+#include "../common.h"++/* Backends must return MLK_NATIVE_FUNC_SUCCESS upon success. */+#define MLK_NATIVE_FUNC_SUCCESS (0)+/* Backends may return MLK_NATIVE_FUNC_FALLBACK to signal to the frontend that+ * the target/parameters are unsupported; typically, this would be because of+ * dependencies on CPU features not detected on the host CPU. In this case,+ * the frontend falls back to the default C implementation. */+#define MLK_NATIVE_FUNC_FALLBACK (-1)+++/* Absolute exclusive upper bound for the output of the inverse NTT+ *+ * NOTE: This is the same bound as in poly.h and has to be kept+ * in sync. */+#define MLK_INVNTT_BOUND (8 * MLKEM_Q)++/* Absolute exclusive upper bound for the output of the forward NTT+ *+ * NOTE: This is the same bound as in poly.h and has to be kept+ * in sync. */+#define MLK_NTT_BOUND (8 * MLKEM_Q)++/*+ * This is the C<->native interface allowing for the drop-in of+ * native code for performance critical arithmetic components of ML-KEM.+ *+ * A _backend_ is a specific implementation of (part of) this interface.+ *+ * To add a function to a backend, define MLK_USE_NATIVE_XXX and+ * implement `static inline xxx(...)` in the profile header.+ *+ * The only exception is MLK_USE_NATIVE_NTT_CUSTOM_ORDER. This option can+ * be set if there are native implementations for all of NTT, invNTT, and+ * base multiplication, and allows the native implementation to use a+ * custom order of polynomial coefficients in NTT domain -- the use of such+ * custom order is not an implementation-detail since the public matrix+ * is generated in NTT domain. In this case, a permutation function+ * mlk_poly_permute_bitrev_to_custom() needs to be provided that permutes+ * polynomials in NTT domain from bitreversed to the custom order.+ */++/*+ * Those functions are meant to be trivial wrappers around the chosen native+ * implementation. The are static inline to avoid unnecessary calls.+ * The macro before each declaration controls whether a native+ * implementation is present.+ */++#if defined(MLK_USE_NATIVE_NTT)+/**+ * Compute the negacyclic number-theoretic transform (NTT) of a polynomial+ * in place.+ *+ * The input polynomial is assumed to be in normal order. The output+ * polynomial is in bitreversed order, or of a custom order if+ * MLK_USE_NATIVE_NTT_CUSTOM_ORDER is set. See the documentation of+ * MLK_USE_NATIVE_NTT_CUSTOM_ORDER for more information.+ *+ * @param[in,out] p Input/output polynomial.+ *+ * @retval MLK_NATIVE_FUNC_SUCCESS Operation succeeded.+ * @retval MLK_NATIVE_FUNC_FALLBACK Backend declined; caller should fall back.+ */+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_ntt_native(int16_t p[MLKEM_N])+__contract__(+ requires(memory_no_alias(p, sizeof(int16_t) * MLKEM_N))+ requires(array_abs_bound(p, 0, MLKEM_N, MLKEM_Q))+ assigns(memory_slice(p, sizeof(int16_t) * MLKEM_N))+ ensures(return_value == MLK_NATIVE_FUNC_FALLBACK || return_value == MLK_NATIVE_FUNC_SUCCESS)+ ensures((return_value == MLK_NATIVE_FUNC_SUCCESS) ==> array_abs_bound(p, 0, MLKEM_N, MLK_NTT_BOUND))+ ensures((return_value == MLK_NATIVE_FUNC_FALLBACK) ==> array_abs_bound(p, 0, MLKEM_N, MLKEM_Q))+ ensures((return_value == MLK_NATIVE_FUNC_FALLBACK) ==> array_unchanged(p, MLKEM_N))+);+#endif /* MLK_USE_NATIVE_NTT */++#if defined(MLK_USE_NATIVE_NTT_CUSTOM_ORDER)+/*+ * This must only be set if NTT, invNTT, basemul, mulcache, and+ * to/from byte stream conversions all have native implementations+ * that are adapted to the custom order.+ */+#if !defined(MLK_USE_NATIVE_NTT) || !defined(MLK_USE_NATIVE_INTT) || \+ !defined(MLK_USE_NATIVE_POLY_MULCACHE_COMPUTE) || \+ !defined(MLK_USE_NATIVE_POLYVEC_BASEMUL_ACC_MONTGOMERY_CACHED) || \+ !defined(MLK_USE_NATIVE_POLY_TOBYTES) || \+ !defined(MLK_USE_NATIVE_POLY_FROMBYTES)+#error \+ "Invalid native profile: MLK_USE_NATIVE_NTT_CUSTOM_ORDER can only be \+set if there are native implementations for NTT, invNTT, mulcache, basemul, \+and to/from bytes conversions."+#endif /* !MLK_USE_NATIVE_NTT || !MLK_USE_NATIVE_INTT || \+ !MLK_USE_NATIVE_POLY_MULCACHE_COMPUTE || \+ !MLK_USE_NATIVE_POLYVEC_BASEMUL_ACC_MONTGOMERY_CACHED || \+ !MLK_USE_NATIVE_POLY_TOBYTES || !MLK_USE_NATIVE_POLY_FROMBYTES */++/**+ * When MLK_USE_NATIVE_NTT_CUSTOM_ORDER is defined, convert a polynomial in+ * NTT domain from bitreversed order to the custom order output by the+ * native NTT.+ *+ * This must only be defined if there is native code for all of (a) NTT,+ * (b) invNTT, (c) basemul, (d) mulcache.+ *+ * @param[in,out] p Input/output polynomial.+ */+static MLK_INLINE void mlk_poly_permute_bitrev_to_custom(int16_t p[MLKEM_N])+__contract__(+ /* We don't specify that this should be a permutation, but only+ * that it does not change the bound established at the end of mlk_gen_matrix. */+ requires(memory_no_alias(p, sizeof(int16_t) * MLKEM_N))+ requires(array_bound(p, 0, MLKEM_N, 0, MLKEM_Q))+ assigns(memory_slice(p, sizeof(int16_t) * MLKEM_N))+ ensures(array_bound(p, 0, MLKEM_N, 0, MLKEM_Q)));+#endif /* MLK_USE_NATIVE_NTT_CUSTOM_ORDER */++#if defined(MLK_USE_NATIVE_INTT) && \+ (!defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API))+/**+ * Compute the inverse negacyclic number-theoretic transform (NTT) of a+ * polynomial in place.+ *+ * The input polynomial is in bitreversed order, or of a custom order if+ * MLK_USE_NATIVE_NTT_CUSTOM_ORDER is set. See the documentation of+ * MLK_USE_NATIVE_NTT_CUSTOM_ORDER for more information. The output+ * polynomial is assumed to be in normal order.+ *+ * @param[in,out] p Input/output polynomial.+ *+ * @retval MLK_NATIVE_FUNC_SUCCESS Operation succeeded.+ * @retval MLK_NATIVE_FUNC_FALLBACK Backend declined; caller should fall back.+ */+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_intt_native(int16_t p[MLKEM_N])+__contract__(+ requires(memory_no_alias(p, sizeof(int16_t) * MLKEM_N))+ assigns(memory_slice(p, sizeof(int16_t) * MLKEM_N))+ ensures(return_value == MLK_NATIVE_FUNC_FALLBACK || return_value == MLK_NATIVE_FUNC_SUCCESS)+ ensures((return_value == MLK_NATIVE_FUNC_SUCCESS) ==> array_abs_bound(p, 0, MLKEM_N, MLK_INVNTT_BOUND))+ ensures((return_value == MLK_NATIVE_FUNC_FALLBACK) ==> array_unchanged(p, MLKEM_N))+);+#endif /* MLK_USE_NATIVE_INTT && (!MLK_CONFIG_NO_ENCAPS_API || \+ !MLK_CONFIG_NO_DECAPS_API) */++#if defined(MLK_USE_NATIVE_POLY_REDUCE)+/**+ * Apply modular reduction to all coefficients of a polynomial, mapping them+ * to unsigned canonical representatives in [0,..,MLKEM_Q-1].+ *+ * @param[in,out] p Input/output polynomial.+ *+ * @retval MLK_NATIVE_FUNC_SUCCESS Operation succeeded.+ * @retval MLK_NATIVE_FUNC_FALLBACK Backend declined; caller should fall back.+ */+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_reduce_native(int16_t p[MLKEM_N])+__contract__(+ requires(memory_no_alias(p, sizeof(int16_t) * MLKEM_N))+ assigns(memory_slice(p, sizeof(int16_t) * MLKEM_N))+ ensures(return_value == MLK_NATIVE_FUNC_FALLBACK || return_value == MLK_NATIVE_FUNC_SUCCESS)+ ensures((return_value == MLK_NATIVE_FUNC_SUCCESS) ==> array_bound(p, 0, MLKEM_N, 0, MLKEM_Q))+ ensures((return_value == MLK_NATIVE_FUNC_FALLBACK) ==> array_unchanged(p, MLKEM_N))+);+#endif /* MLK_USE_NATIVE_POLY_REDUCE */++#if defined(MLK_USE_NATIVE_POLY_TOMONT) && !defined(MLK_CONFIG_NO_KEYPAIR_API)+/**+ * In-place conversion of all coefficients of a polynomial from the normal+ * domain to the Montgomery domain.+ *+ * @param[in,out] p Input/output polynomial.+ *+ * @retval MLK_NATIVE_FUNC_SUCCESS Operation succeeded.+ * @retval MLK_NATIVE_FUNC_FALLBACK Backend declined; caller should fall back.+ */+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_tomont_native(int16_t p[MLKEM_N])+__contract__(+ requires(memory_no_alias(p, sizeof(int16_t) * MLKEM_N))+ assigns(memory_slice(p, sizeof(int16_t) * MLKEM_N))+ ensures(return_value == MLK_NATIVE_FUNC_FALLBACK || return_value == MLK_NATIVE_FUNC_SUCCESS)+ ensures((return_value == MLK_NATIVE_FUNC_SUCCESS) ==> array_abs_bound(p, 0, MLKEM_N, MLKEM_Q))+ ensures((return_value == MLK_NATIVE_FUNC_FALLBACK) ==> array_unchanged(p, MLKEM_N))+);+#endif /* MLK_USE_NATIVE_POLY_TOMONT && !MLK_CONFIG_NO_KEYPAIR_API */++#if defined(MLK_USE_NATIVE_POLY_MULCACHE_COMPUTE)+/**+ * Compute multiplication cache for a polynomial in NTT domain.+ *+ * The purpose of the multiplication cache is to cache repeated computations+ * required during a base multiplication of polynomials in NTT domain. The+ * structure of the multiplication-cache is implementation defined.+ *+ * @param[out] cache Multiplication cache.+ * @param[in] mlk_poly Input polynomial. Must be in NTT domain and in+ * bitreversed order, or of a custom order if+ * MLK_USE_NATIVE_NTT_CUSTOM_ORDER is set. See the+ * documentation of MLK_USE_NATIVE_NTT_CUSTOM_ORDER for+ * more information.+ *+ * @retval MLK_NATIVE_FUNC_SUCCESS Operation succeeded.+ * @retval MLK_NATIVE_FUNC_FALLBACK Backend declined; caller should fall back.+ */+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_mulcache_compute_native(+ int16_t cache[MLKEM_N / 2], const int16_t mlk_poly[MLKEM_N])+__contract__(+ requires(memory_no_alias(cache, sizeof(int16_t) * (MLKEM_N / 2)))+ requires(memory_no_alias(mlk_poly, sizeof(int16_t) * MLKEM_N))+ assigns(memory_slice(cache, sizeof(int16_t) * (MLKEM_N / 2)))+ ensures(return_value == MLK_NATIVE_FUNC_FALLBACK || return_value == MLK_NATIVE_FUNC_SUCCESS)+ ensures((return_value == MLK_NATIVE_FUNC_SUCCESS) ==> array_abs_bound(cache, 0, MLKEM_N/2, MLKEM_Q))+);+#endif /* MLK_USE_NATIVE_POLY_MULCACHE_COMPUTE */++#if defined(MLK_USE_NATIVE_POLYVEC_BASEMUL_ACC_MONTGOMERY_CACHED)+#if defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_K == 2+/**+ * Compute scalar product of length-2 polynomial vectors in NTT domain.+ *+ * @param[out] r Result of the scalar product. Again in NTT domain, of+ * the same ordering as @p a and @p b.+ * @param[in] a First polynomial vector operand. Must be in NTT+ * domain and in bitreversed order, or of a custom order+ * if MLK_USE_NATIVE_NTT_CUSTOM_ORDER is set. See the+ * documentation of MLK_USE_NATIVE_NTT_CUSTOM_ORDER for+ * more information.+ * @param[in] b Second polynomial vector operand. As for @p a.+ * @param[in] b_cache Multiplication-cache for @p b.+ *+ * @retval MLK_NATIVE_FUNC_SUCCESS Operation succeeded.+ * @retval MLK_NATIVE_FUNC_FALLBACK Backend declined; caller should fall back.+ */+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_polyvec_basemul_acc_montgomery_cached_k2_native(+ int16_t r[MLKEM_N], const int16_t a[2 * MLKEM_N],+ const int16_t b[2 * MLKEM_N], const int16_t b_cache[2 * (MLKEM_N / 2)])+__contract__(+ requires(memory_no_alias(r, sizeof(int16_t) * MLKEM_N))+ requires(memory_no_alias(a, sizeof(int16_t) * 2 * MLKEM_N))+ requires(memory_no_alias(b, sizeof(int16_t) * 2 * MLKEM_N))+ requires(memory_no_alias(b_cache, sizeof(int16_t) * 2 * (MLKEM_N / 2)))+ requires(array_bound(a, 0, 2 * MLKEM_N, 0, MLKEM_UINT12_LIMIT))+ assigns(memory_slice(r, sizeof(int16_t) * MLKEM_N))+ ensures(return_value == MLK_NATIVE_FUNC_FALLBACK || return_value == MLK_NATIVE_FUNC_SUCCESS)+);+#endif /* MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 2 */++#if defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_K == 3+/**+ * Compute scalar product of length-3 polynomial vectors in NTT domain.+ *+ * @param[out] r Result of the scalar product. Again in NTT domain, of+ * the same ordering as @p a and @p b.+ * @param[in] a First polynomial vector operand. Must be in NTT+ * domain and in bitreversed order, or of a custom order+ * if MLK_USE_NATIVE_NTT_CUSTOM_ORDER is set. See the+ * documentation of MLK_USE_NATIVE_NTT_CUSTOM_ORDER for+ * more information.+ * @param[in] b Second polynomial vector operand. As for @p a.+ * @param[in] b_cache Multiplication-cache for @p b.+ *+ * @retval MLK_NATIVE_FUNC_SUCCESS Operation succeeded.+ * @retval MLK_NATIVE_FUNC_FALLBACK Backend declined; caller should fall back.+ */+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_polyvec_basemul_acc_montgomery_cached_k3_native(+ int16_t r[MLKEM_N], const int16_t a[3 * MLKEM_N],+ const int16_t b[3 * MLKEM_N], const int16_t b_cache[3 * (MLKEM_N / 2)])+__contract__(+ requires(memory_no_alias(r, sizeof(int16_t) * MLKEM_N))+ requires(memory_no_alias(a, sizeof(int16_t) * 3 * MLKEM_N))+ requires(memory_no_alias(b, sizeof(int16_t) * 3 * MLKEM_N))+ requires(memory_no_alias(b_cache, sizeof(int16_t) * 3 * (MLKEM_N / 2)))+ requires(array_bound(a, 0, 3 * MLKEM_N, 0, MLKEM_UINT12_LIMIT))+ assigns(memory_slice(r, sizeof(int16_t) * MLKEM_N))+ ensures(return_value == MLK_NATIVE_FUNC_FALLBACK || return_value == MLK_NATIVE_FUNC_SUCCESS)+);+#endif /* MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 3 */++#if defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_K == 4+/**+ * Compute scalar product of length-4 polynomial vectors in NTT domain.+ *+ * @param[out] r Result of the scalar product. Again in NTT domain, of+ * the same ordering as @p a and @p b.+ * @param[in] a First polynomial vector operand. Must be in NTT+ * domain and in bitreversed order, or of a custom order+ * if MLK_USE_NATIVE_NTT_CUSTOM_ORDER is set. See the+ * documentation of MLK_USE_NATIVE_NTT_CUSTOM_ORDER for+ * more information.+ * @param[in] b Second polynomial vector operand. As for @p a.+ * @param[in] b_cache Multiplication-cache for @p b.+ *+ * @retval MLK_NATIVE_FUNC_SUCCESS Operation succeeded.+ * @retval MLK_NATIVE_FUNC_FALLBACK Backend declined; caller should fall back.+ */+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_polyvec_basemul_acc_montgomery_cached_k4_native(+ int16_t r[MLKEM_N], const int16_t a[4 * MLKEM_N],+ const int16_t b[4 * MLKEM_N], const int16_t b_cache[4 * (MLKEM_N / 2)])+__contract__(+ requires(memory_no_alias(r, sizeof(int16_t) * MLKEM_N))+ requires(memory_no_alias(a, sizeof(int16_t) * 4 * MLKEM_N))+ requires(memory_no_alias(b, sizeof(int16_t) * 4 * MLKEM_N))+ requires(memory_no_alias(b_cache, sizeof(int16_t) * 4 * (MLKEM_N / 2)))+ requires(array_bound(a, 0, 4 * MLKEM_N, 0, MLKEM_UINT12_LIMIT))+ assigns(memory_slice(r, sizeof(int16_t) * MLKEM_N))+ ensures(return_value == MLK_NATIVE_FUNC_FALLBACK || return_value == MLK_NATIVE_FUNC_SUCCESS)+);+#endif /* MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 4 */+#endif /* MLK_USE_NATIVE_POLYVEC_BASEMUL_ACC_MONTGOMERY_CACHED */++#if defined(MLK_USE_NATIVE_POLY_TOBYTES) && \+ (!defined(MLK_CONFIG_NO_KEYPAIR_API) || \+ !defined(MLK_CONFIG_NO_ENCAPS_API))+/**+ * Serialization of a polynomial with unsigned canonical coefficients.+ *+ * @spec{Implements ByteEncode_12 from @[FIPS203, Algorithm 5].}+ *+ * @param[out] r Output byte array (of MLKEM_POLYBYTES bytes).+ * @param[in] a Input polynomial, with each coefficient in [0,..,MLKEM_Q-1].+ *+ * @retval MLK_NATIVE_FUNC_SUCCESS Operation succeeded.+ * @retval MLK_NATIVE_FUNC_FALLBACK Backend declined; caller should fall back.+ */+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_tobytes_native(uint8_t r[MLKEM_POLYBYTES],+ const int16_t a[MLKEM_N])+__contract__(+ requires(memory_no_alias(r, MLKEM_POLYBYTES))+ requires(memory_no_alias(a, sizeof(int16_t) * MLKEM_N))+ requires(array_bound(a, 0, MLKEM_N, 0, MLKEM_Q))+ assigns(memory_slice(r, MLKEM_POLYBYTES))+ ensures(return_value == MLK_NATIVE_FUNC_SUCCESS || return_value == MLK_NATIVE_FUNC_FALLBACK)+);+#endif /* MLK_USE_NATIVE_POLY_TOBYTES && (!MLK_CONFIG_NO_KEYPAIR_API || \+ !MLK_CONFIG_NO_ENCAPS_API) */++#if defined(MLK_USE_NATIVE_POLY_FROMBYTES) && \+ (!defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API))+/**+ * Deserialization of a polynomial.+ *+ * @spec{Implements ByteDecode_12 from @[FIPS203, Algorithm 6].}+ *+ * @param[out] a Output polynomial in NTT domain.+ * @param[in] r Input byte array (of MLKEM_POLYBYTES bytes).+ *+ * @retval MLK_NATIVE_FUNC_SUCCESS Operation succeeded.+ * @retval MLK_NATIVE_FUNC_FALLBACK Backend declined; caller should fall back.+ */+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_frombytes_native(+ int16_t a[MLKEM_N], const uint8_t r[MLKEM_POLYBYTES])+__contract__(+ requires(memory_no_alias(r, MLKEM_POLYBYTES))+ requires(memory_no_alias(a, sizeof(int16_t) * MLKEM_N))+ assigns(memory_slice(a, sizeof(int16_t) * MLKEM_N))+ ensures(return_value == MLK_NATIVE_FUNC_SUCCESS || return_value == MLK_NATIVE_FUNC_FALLBACK)+ ensures((return_value == MLK_NATIVE_FUNC_SUCCESS) ==> array_bound(a, 0, MLKEM_N, 0, MLKEM_UINT12_LIMIT))+);+#endif /* MLK_USE_NATIVE_POLY_FROMBYTES && (!MLK_CONFIG_NO_ENCAPS_API || \+ !MLK_CONFIG_NO_DECAPS_API) */++#if defined(MLK_USE_NATIVE_REJ_UNIFORM)+/**+ * Run rejection sampling on uniformly random bytes to generate uniformly random+ * integers mod MLKEM_Q, represented in [0,..,MLKEM_Q-1].+ *+ * @param[out] r Output buffer.+ * @param len Requested number of 16-bit integers (uniform mod MLKEM_Q).+ * @param[in] buf Input buffer (assumed to be uniform random bytes).+ * @param buflen Length of input buffer in bytes.+ *+ * @retval MLK_NATIVE_FUNC_FALLBACK Native implementation does not support+ * the input lengths.+ * @retval other Non-negative number of sampled 16-bit+ * integers (at most @p len).+ */+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_rej_uniform_native(int16_t *r, unsigned len,+ const uint8_t *buf,+ unsigned buflen)+__contract__(+ requires(len <= 4096 && buflen <= 4096 && buflen % 3 == 0)+ requires(memory_no_alias(r, sizeof(int16_t) * len))+ requires(memory_no_alias(buf, buflen))+ assigns(memory_slice(r, sizeof(int16_t) * len))+ ensures(return_value != MLK_NATIVE_FUNC_FALLBACK+ ==> (0 <= return_value && return_value <= len))+ ensures(return_value != MLK_NATIVE_FUNC_FALLBACK+ ==> array_bound(r, 0, (unsigned) return_value, 0, MLKEM_Q))+);+#endif /* MLK_USE_NATIVE_REJ_UNIFORM */++#if !defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API)+#if defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || (MLKEM_K == 2 || MLKEM_K == 3)+#if defined(MLK_USE_NATIVE_POLY_COMPRESS_D4)+/**+ * Compression (4 bits) and subsequent serialization of a polynomial.+ *+ * @spec{Compress_4 from @[FIPS203, Eq (4.7)].}+ *+ * @param[out] r Output byte array (of length MLKEM_POLYCOMPRESSEDBYTES_D4+ * bytes).+ * @param[in] a Input polynomial. Coefficients must be unsigned canonical,+ * i.e. in [0,1,..,MLKEM_Q-1].+ *+ * @retval MLK_NATIVE_FUNC_SUCCESS Operation succeeded.+ * @retval MLK_NATIVE_FUNC_FALLBACK Backend declined; caller should fall back.+ */+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_compress_d4_native(+ uint8_t r[MLKEM_POLYCOMPRESSEDBYTES_D4], const int16_t a[MLKEM_N])+__contract__(+ requires(memory_no_alias(r, MLKEM_POLYCOMPRESSEDBYTES_D4))+ requires(memory_no_alias(a, sizeof(int16_t) * MLKEM_N))+ requires(array_bound(a, 0, MLKEM_N, 0, MLKEM_Q))+ assigns(memory_slice(r, MLKEM_POLYCOMPRESSEDBYTES_D4))+ ensures(return_value == MLK_NATIVE_FUNC_SUCCESS || return_value == MLK_NATIVE_FUNC_FALLBACK));+#endif /* MLK_USE_NATIVE_POLY_COMPRESS_D4 */++#if defined(MLK_USE_NATIVE_POLY_COMPRESS_D10)+/**+ * Compression (10 bits) and subsequent serialization of a polynomial.+ *+ * @spec{Compress_10 from @[FIPS203, Eq (4.7)].}+ *+ * @param[out] r Output byte array (of length MLKEM_POLYCOMPRESSEDBYTES_D10+ * bytes).+ * @param[in] a Input polynomial. Coefficients must be unsigned canonical,+ * i.e. in [0,1,..,MLKEM_Q-1].+ *+ * @retval MLK_NATIVE_FUNC_SUCCESS Operation succeeded.+ * @retval MLK_NATIVE_FUNC_FALLBACK Backend declined; caller should fall back.+ */+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_compress_d10_native(+ uint8_t r[MLKEM_POLYCOMPRESSEDBYTES_D10], const int16_t a[MLKEM_N])+__contract__(+ requires(memory_no_alias(r, MLKEM_POLYCOMPRESSEDBYTES_D10))+ requires(memory_no_alias(a, sizeof(int16_t) * MLKEM_N))+ requires(array_bound(a, 0, MLKEM_N, 0, MLKEM_Q))+ assigns(memory_slice(r, MLKEM_POLYCOMPRESSEDBYTES_D10))+ ensures(return_value == MLK_NATIVE_FUNC_SUCCESS || return_value == MLK_NATIVE_FUNC_FALLBACK));+#endif /* MLK_USE_NATIVE_POLY_COMPRESS_D10 */++#if !defined(MLK_CONFIG_NO_DECAPS_API)+#if defined(MLK_USE_NATIVE_POLY_DECOMPRESS_D4)+/**+ * De-serialization and subsequent decompression (4 bits) of a polynomial;+ * approximate inverse of mlk_poly_compress_d4.+ *+ * @spec{Decompress_4 from @[FIPS203, Eq (4.8)].}+ *+ * Upon return, the coefficients of the output polynomial are+ * unsigned-canonical (non-negative and smaller than MLKEM_Q).+ *+ * @param[out] r Output polynomial.+ * @param[in] a Input byte array (of length MLKEM_POLYCOMPRESSEDBYTES_D4+ * bytes).+ *+ * @retval MLK_NATIVE_FUNC_SUCCESS Operation succeeded.+ * @retval MLK_NATIVE_FUNC_FALLBACK Backend declined; caller should fall back.+ */+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_decompress_d4_native(+ int16_t r[MLKEM_N], const uint8_t a[MLKEM_POLYCOMPRESSEDBYTES_D4])+__contract__(+ requires(memory_no_alias(r, sizeof(int16_t) * MLKEM_N))+ requires(memory_no_alias(a, MLKEM_POLYCOMPRESSEDBYTES_D4))+ assigns(memory_slice(r, sizeof(int16_t) * MLKEM_N))+ ensures(return_value == MLK_NATIVE_FUNC_SUCCESS || return_value == MLK_NATIVE_FUNC_FALLBACK)+ ensures((return_value == MLK_NATIVE_FUNC_SUCCESS) ==> array_bound(r, 0, MLKEM_N, 0, MLKEM_Q)));+#endif /* MLK_USE_NATIVE_POLY_DECOMPRESS_D4 */++#if defined(MLK_USE_NATIVE_POLY_DECOMPRESS_D10)+/**+ * De-serialization and subsequent decompression (10 bits) of a polynomial;+ * approximate inverse of mlk_poly_compress_d10.+ *+ * @spec{Decompress_10 from @[FIPS203, Eq (4.8)].}+ *+ * Upon return, the coefficients of the output polynomial are+ * unsigned-canonical (non-negative and smaller than MLKEM_Q).+ *+ * @param[out] r Output polynomial.+ * @param[in] a Input byte array (of length MLKEM_POLYCOMPRESSEDBYTES_D10+ * bytes).+ *+ * @retval MLK_NATIVE_FUNC_SUCCESS Operation succeeded.+ * @retval MLK_NATIVE_FUNC_FALLBACK Backend declined; caller should fall back.+ */+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_decompress_d10_native(+ int16_t r[MLKEM_N], const uint8_t a[MLKEM_POLYCOMPRESSEDBYTES_D10])+__contract__(+ requires(memory_no_alias(r, sizeof(int16_t) * MLKEM_N))+ requires(memory_no_alias(a, MLKEM_POLYCOMPRESSEDBYTES_D10))+ assigns(memory_slice(r, sizeof(int16_t) * MLKEM_N))+ ensures(return_value == MLK_NATIVE_FUNC_SUCCESS || return_value == MLK_NATIVE_FUNC_FALLBACK)+ ensures((return_value == MLK_NATIVE_FUNC_SUCCESS) ==> array_bound(r, 0, MLKEM_N, 0, MLKEM_Q)));+#endif /* MLK_USE_NATIVE_POLY_DECOMPRESS_D10 */+#endif /* !MLK_CONFIG_NO_DECAPS_API */+#endif /* MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 2 || MLKEM_K == 3 */++#if defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_K == 4+#if defined(MLK_USE_NATIVE_POLY_COMPRESS_D5)+/**+ * Compression (5 bits) and subsequent serialization of a polynomial.+ *+ * @spec{Compress_5 from @[FIPS203, Eq (4.7)].}+ *+ * @param[out] r Output byte array (of length MLKEM_POLYCOMPRESSEDBYTES_D5+ * bytes).+ * @param[in] a Input polynomial. Coefficients must be unsigned canonical,+ * i.e. in [0,1,..,MLKEM_Q-1].+ *+ * @retval MLK_NATIVE_FUNC_SUCCESS Operation succeeded.+ * @retval MLK_NATIVE_FUNC_FALLBACK Backend declined; caller should fall back.+ */+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_compress_d5_native(+ uint8_t r[MLKEM_POLYCOMPRESSEDBYTES_D5], const int16_t a[MLKEM_N])+__contract__(+ requires(memory_no_alias(r, MLKEM_POLYCOMPRESSEDBYTES_D5))+ requires(memory_no_alias(a, sizeof(int16_t) * MLKEM_N))+ requires(array_bound(a, 0, MLKEM_N, 0, MLKEM_Q))+ assigns(memory_slice(r, MLKEM_POLYCOMPRESSEDBYTES_D5))+ ensures(return_value == MLK_NATIVE_FUNC_SUCCESS || return_value == MLK_NATIVE_FUNC_FALLBACK));+#endif /* MLK_USE_NATIVE_POLY_COMPRESS_D5 */++#if defined(MLK_USE_NATIVE_POLY_COMPRESS_D11)+/**+ * Compression (11 bits) and subsequent serialization of a polynomial.+ *+ * @spec{Compress_11 from @[FIPS203, Eq (4.7)].}+ *+ * @param[out] r Output byte array (of length MLKEM_POLYCOMPRESSEDBYTES_D11+ * bytes).+ * @param[in] a Input polynomial. Coefficients must be unsigned canonical,+ * i.e. in [0,1,..,MLKEM_Q-1].+ *+ * @retval MLK_NATIVE_FUNC_SUCCESS Operation succeeded.+ * @retval MLK_NATIVE_FUNC_FALLBACK Backend declined; caller should fall back.+ */+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_compress_d11_native(+ uint8_t r[MLKEM_POLYCOMPRESSEDBYTES_D11], const int16_t a[MLKEM_N])+__contract__(+ requires(memory_no_alias(r, MLKEM_POLYCOMPRESSEDBYTES_D11))+ requires(memory_no_alias(a, sizeof(int16_t) * MLKEM_N))+ requires(array_bound(a, 0, MLKEM_N, 0, MLKEM_Q))+ assigns(memory_slice(r, MLKEM_POLYCOMPRESSEDBYTES_D11))+ ensures(return_value == MLK_NATIVE_FUNC_SUCCESS || return_value == MLK_NATIVE_FUNC_FALLBACK));+#endif /* MLK_USE_NATIVE_POLY_COMPRESS_D11 */++#if !defined(MLK_CONFIG_NO_DECAPS_API)+#if defined(MLK_USE_NATIVE_POLY_DECOMPRESS_D5)+/**+ * De-serialization and subsequent decompression (5 bits) of a polynomial;+ * approximate inverse of mlk_poly_compress_d5.+ *+ * @spec{Decompress_5 from @[FIPS203, Eq (4.8)].}+ *+ * Upon return, the coefficients of the output polynomial are+ * unsigned-canonical (non-negative and smaller than MLKEM_Q).+ *+ * @param[out] r Output polynomial.+ * @param[in] a Input byte array (of length MLKEM_POLYCOMPRESSEDBYTES_D5+ * bytes).+ *+ * @retval MLK_NATIVE_FUNC_SUCCESS Operation succeeded.+ * @retval MLK_NATIVE_FUNC_FALLBACK Backend declined; caller should fall back.+ */+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_decompress_d5_native(+ int16_t r[MLKEM_N], const uint8_t a[MLKEM_POLYCOMPRESSEDBYTES_D5])+__contract__(+ requires(memory_no_alias(r, sizeof(int16_t) * MLKEM_N))+ requires(memory_no_alias(a, MLKEM_POLYCOMPRESSEDBYTES_D5))+ assigns(memory_slice(r, sizeof(int16_t) * MLKEM_N))+ ensures(return_value == MLK_NATIVE_FUNC_SUCCESS || return_value == MLK_NATIVE_FUNC_FALLBACK)+ ensures((return_value == MLK_NATIVE_FUNC_SUCCESS) ==> array_bound(r, 0, MLKEM_N, 0, MLKEM_Q)));+#endif /* MLK_USE_NATIVE_POLY_DECOMPRESS_D5 */++#if defined(MLK_USE_NATIVE_POLY_DECOMPRESS_D11)+/**+ * De-serialization and subsequent decompression (11 bits) of a polynomial;+ * approximate inverse of mlk_poly_compress_d11.+ *+ * @spec{Decompress_11 from @[FIPS203, Eq (4.8)].}+ *+ * Upon return, the coefficients of the output polynomial are+ * unsigned-canonical (non-negative and smaller than MLKEM_Q).+ *+ * @param[out] r Output polynomial.+ * @param[in] a Input byte array (of length MLKEM_POLYCOMPRESSEDBYTES_D11+ * bytes).+ *+ * @retval MLK_NATIVE_FUNC_SUCCESS Operation succeeded.+ * @retval MLK_NATIVE_FUNC_FALLBACK Backend declined; caller should fall back.+ */+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_decompress_d11_native(+ int16_t r[MLKEM_N], const uint8_t a[MLKEM_POLYCOMPRESSEDBYTES_D11])+__contract__(+ requires(memory_no_alias(r, sizeof(int16_t) * MLKEM_N))+ requires(memory_no_alias(a, MLKEM_POLYCOMPRESSEDBYTES_D11))+ assigns(memory_slice(r, sizeof(int16_t) * MLKEM_N))+ ensures(return_value == MLK_NATIVE_FUNC_SUCCESS || return_value == MLK_NATIVE_FUNC_FALLBACK)+ ensures((return_value == MLK_NATIVE_FUNC_SUCCESS) ==> array_bound(r, 0, MLKEM_N, 0, MLKEM_Q)));+#endif /* MLK_USE_NATIVE_POLY_DECOMPRESS_D11 */+#endif /* !MLK_CONFIG_NO_DECAPS_API */+#endif /* MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 4 */+#endif /* !MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API */++#endif /* !MLK_NATIVE_API_H */
@@ -0,0 +1,30 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+#ifndef MLK_NATIVE_META_H+#define MLK_NATIVE_META_H++/*+ * Default arithmetic backend+ */+#include "../sys.h"++#ifdef MLK_SYS_AARCH64_NEON+#include "aarch64/meta.h"+#endif++/* The x86_64 backend requires toolchain support for the SysV ABI */+#if defined(MLK_SYS_X86_64_AVX2) && defined(MLK_SYSV_ABI_SUPPORTED)+#include "x86_64/meta.h"+#endif++#if defined(MLK_SYS_RISCV64_RVV)+#include "riscv64/meta.h"+#endif++#ifdef MLK_SYS_PPC64LE+#include "ppc64le/meta.h"+#endif++#endif /* !MLK_NATIVE_META_H */
@@ -0,0 +1,4 @@+[//]: # (SPDX-License-Identifier: CC-BY-4.0)++This directory contains the native x86_64 arithmetic backend for ML-KEM provided by the official [AVX2+implementation](https://github.com/pq-crystals/kyber/tree/main/avx2) of the Kyber team.
@@ -0,0 +1,324 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++#ifndef MLK_NATIVE_X86_64_META_H+#define MLK_NATIVE_X86_64_META_H++/* Identifier for this backend so that source and assembly files+ * in the build can be appropriately guarded. */+#define MLK_ARITH_BACKEND_X86_64_DEFAULT++#define MLK_USE_NATIVE_NTT_CUSTOM_ORDER+#define MLK_USE_NATIVE_REJ_UNIFORM+#define MLK_USE_NATIVE_NTT+#define MLK_USE_NATIVE_INTT+#define MLK_USE_NATIVE_POLY_REDUCE+#define MLK_USE_NATIVE_POLY_TOMONT+#define MLK_USE_NATIVE_POLYVEC_BASEMUL_ACC_MONTGOMERY_CACHED+#define MLK_USE_NATIVE_POLY_MULCACHE_COMPUTE+#define MLK_USE_NATIVE_POLY_TOBYTES+#define MLK_USE_NATIVE_POLY_FROMBYTES+#define MLK_USE_NATIVE_POLY_COMPRESS_D4+#define MLK_USE_NATIVE_POLY_COMPRESS_D5+#define MLK_USE_NATIVE_POLY_COMPRESS_D10+#define MLK_USE_NATIVE_POLY_COMPRESS_D11+#define MLK_USE_NATIVE_POLY_DECOMPRESS_D4+#define MLK_USE_NATIVE_POLY_DECOMPRESS_D5+#define MLK_USE_NATIVE_POLY_DECOMPRESS_D10+#define MLK_USE_NATIVE_POLY_DECOMPRESS_D11++#if !defined(__ASSEMBLER__)+#include "../../common.h"+#include "../api.h"+#include "src/arith_native_x86_64.h"+#include "src/compress_consts.h"++static MLK_INLINE void mlk_poly_permute_bitrev_to_custom(int16_t data[MLKEM_N])+{+ if (mlk_sys_check_capability(MLK_SYS_CAP_X86_64_AVX2))+ {+ mlk_nttunpack_avx2_asm(data);+ }+}++MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_rej_uniform_native(int16_t *r, unsigned len,+ const uint8_t *buf,+ unsigned buflen)+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_X86_64_AVX2) || len != MLKEM_N ||+ buflen % 12 != 0)+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }+ return (int)mlk_rej_uniform_avx2_asm(r, buf, buflen, mlk_rej_uniform_table);+}++MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_ntt_native(int16_t data[MLKEM_N])+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_X86_64_AVX2))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }++ mlk_ntt_avx2_asm(data, mlk_qdata);+ return MLK_NATIVE_FUNC_SUCCESS;+}++#if !defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API)+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_intt_native(int16_t data[MLKEM_N])+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_X86_64_AVX2))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }++ mlk_invntt_avx2_asm(data, mlk_qdata);+ return MLK_NATIVE_FUNC_SUCCESS;+}+#endif /* !MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API */++MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_reduce_native(int16_t data[MLKEM_N])+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_X86_64_AVX2))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }++ mlk_reduce_avx2_asm(data);+ return MLK_NATIVE_FUNC_SUCCESS;+}++#if !defined(MLK_CONFIG_NO_KEYPAIR_API)+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_tomont_native(int16_t data[MLKEM_N])+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_X86_64_AVX2))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }++ mlk_tomont_avx2_asm(data);+ return MLK_NATIVE_FUNC_SUCCESS;+}+#endif /* !MLK_CONFIG_NO_KEYPAIR_API */++MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_mulcache_compute_native(int16_t x[MLKEM_N / 2],+ const int16_t y[MLKEM_N])+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_X86_64_AVX2))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }++ mlk_poly_mulcache_compute_avx2_asm(x, y, mlk_qdata);+ return MLK_NATIVE_FUNC_SUCCESS;+}++#if defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_K == 2+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_polyvec_basemul_acc_montgomery_cached_k2_native(+ int16_t r[MLKEM_N], const int16_t a[2 * MLKEM_N],+ const int16_t b[2 * MLKEM_N], const int16_t b_cache[2 * (MLKEM_N / 2)])+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_X86_64_AVX2))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }++ mlk_polyvec_basemul_acc_montgomery_cached_k2_avx2_asm(r, a, b, b_cache);+ return MLK_NATIVE_FUNC_SUCCESS;+}+#endif /* MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 2 */++#if defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_K == 3+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_polyvec_basemul_acc_montgomery_cached_k3_native(+ int16_t r[MLKEM_N], const int16_t a[3 * MLKEM_N],+ const int16_t b[3 * MLKEM_N], const int16_t b_cache[3 * (MLKEM_N / 2)])+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_X86_64_AVX2))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }++ mlk_polyvec_basemul_acc_montgomery_cached_k3_avx2_asm(r, a, b, b_cache);+ return MLK_NATIVE_FUNC_SUCCESS;+}+#endif /* MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 3 */++#if defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_K == 4+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_polyvec_basemul_acc_montgomery_cached_k4_native(+ int16_t r[MLKEM_N], const int16_t a[4 * MLKEM_N],+ const int16_t b[4 * MLKEM_N], const int16_t b_cache[4 * (MLKEM_N / 2)])+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_X86_64_AVX2))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }++ mlk_polyvec_basemul_acc_montgomery_cached_k4_avx2_asm(r, a, b, b_cache);+ return MLK_NATIVE_FUNC_SUCCESS;+}+#endif /* MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 4 */++#if !defined(MLK_CONFIG_NO_KEYPAIR_API) || !defined(MLK_CONFIG_NO_ENCAPS_API)+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_tobytes_native(uint8_t r[MLKEM_POLYBYTES],+ const int16_t a[MLKEM_N])+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_X86_64_AVX2))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }++ mlk_ntttobytes_avx2_asm(r, a);+ return MLK_NATIVE_FUNC_SUCCESS;+}+#endif /* !MLK_CONFIG_NO_KEYPAIR_API || !MLK_CONFIG_NO_ENCAPS_API */++#if !defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API)+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_frombytes_native(+ int16_t r[MLKEM_N], const uint8_t a[MLKEM_POLYBYTES])+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_X86_64_AVX2))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }++ mlk_nttfrombytes_avx2_asm(r, a);+ return MLK_NATIVE_FUNC_SUCCESS;+}+#endif /* !MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API */++#if (!defined(MLK_CONFIG_NO_ENCAPS_API) || \+ !defined(MLK_CONFIG_NO_DECAPS_API)) && \+ (defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_K == 2 || \+ MLKEM_K == 3)+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_compress_d4_native(+ uint8_t r[MLKEM_POLYCOMPRESSEDBYTES_D4], const int16_t a[MLKEM_N])+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_X86_64_AVX2))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }++ mlk_poly_compress_d4_avx2_asm(r, a, mlk_compress_d4_data);+ return MLK_NATIVE_FUNC_SUCCESS;+}++MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_compress_d10_native(+ uint8_t r[MLKEM_POLYCOMPRESSEDBYTES_D10], const int16_t a[MLKEM_N])+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_X86_64_AVX2))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }++ mlk_poly_compress_d10_avx2_asm(r, a, mlk_compress_d10_data);+ return MLK_NATIVE_FUNC_SUCCESS;+}++#if !defined(MLK_CONFIG_NO_DECAPS_API)+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_decompress_d4_native(+ int16_t r[MLKEM_N], const uint8_t a[MLKEM_POLYCOMPRESSEDBYTES_D4])+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_X86_64_AVX2))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }++ mlk_poly_decompress_d4_avx2_asm(r, a, mlk_decompress_d4_data);+ return MLK_NATIVE_FUNC_SUCCESS;+}++MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_decompress_d10_native(+ int16_t r[MLKEM_N], const uint8_t a[MLKEM_POLYCOMPRESSEDBYTES_D10])+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_X86_64_AVX2))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }++ mlk_poly_decompress_d10_avx2_asm(r, a, mlk_decompress_d10_data);+ return MLK_NATIVE_FUNC_SUCCESS;+}+#endif /* !MLK_CONFIG_NO_DECAPS_API */+#endif /* (!MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API) && \+ (MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 2 || MLKEM_K == 3) \+ */++#if (!defined(MLK_CONFIG_NO_ENCAPS_API) || \+ !defined(MLK_CONFIG_NO_DECAPS_API)) && \+ (defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_K == 4)+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_compress_d5_native(+ uint8_t r[MLKEM_POLYCOMPRESSEDBYTES_D5], const int16_t a[MLKEM_N])+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_X86_64_AVX2))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }++ mlk_poly_compress_d5_avx2_asm(r, a, mlk_compress_d5_data);+ return MLK_NATIVE_FUNC_SUCCESS;+}++MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_compress_d11_native(+ uint8_t r[MLKEM_POLYCOMPRESSEDBYTES_D11], const int16_t a[MLKEM_N])+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_X86_64_AVX2))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }++ mlk_poly_compress_d11_avx2_asm(r, a, mlk_compress_d11_data);+ return MLK_NATIVE_FUNC_SUCCESS;+}++#if !defined(MLK_CONFIG_NO_DECAPS_API)+MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_decompress_d5_native(+ int16_t r[MLKEM_N], const uint8_t a[MLKEM_POLYCOMPRESSEDBYTES_D5])+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_X86_64_AVX2))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }++ mlk_poly_decompress_d5_avx2_asm(r, a, mlk_decompress_d5_data);+ return MLK_NATIVE_FUNC_SUCCESS;+}++MLK_MUST_CHECK_RETURN_VALUE+static MLK_INLINE int mlk_poly_decompress_d11_native(+ int16_t r[MLKEM_N], const uint8_t a[MLKEM_POLYCOMPRESSEDBYTES_D11])+{+ if (!mlk_sys_check_capability(MLK_SYS_CAP_X86_64_AVX2))+ {+ return MLK_NATIVE_FUNC_FALLBACK;+ }++ mlk_poly_decompress_d11_avx2_asm(r, a, mlk_decompress_d11_data);+ return MLK_NATIVE_FUNC_SUCCESS;+}+#endif /* !MLK_CONFIG_NO_DECAPS_API */+#endif /* (!MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API) && \+ (MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 4) */++#endif /* !__ASSEMBLER__ */++#endif /* !MLK_NATIVE_X86_64_META_H */
@@ -0,0 +1,327 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */+#ifndef MLK_NATIVE_X86_64_SRC_ARITH_NATIVE_X86_64_H+#define MLK_NATIVE_X86_64_SRC_ARITH_NATIVE_X86_64_H++#include "../../../common.h"++#include <stdint.h>+#include "compress_consts.h"+#include "consts.h"++#define MLK_AVX2_REJ_UNIFORM_BUFLEN \+ (3 * 168) /* REJ_UNIFORM_NBLOCKS * SHAKE128_RATE */++#define mlk_rej_uniform_table MLK_NAMESPACE(rej_uniform_table)+MLK_INTERNAL_DATA_DECLARATION const uint8_t mlk_rej_uniform_table[4096];++#define mlk_rej_uniform_avx2_asm MLK_NAMESPACE(rej_uniform_avx2_asm)+MLK_MUST_CHECK_RETURN_VALUE MLK_SYSV_ABI+uint64_t mlk_rej_uniform_avx2_asm(int16_t *r, const uint8_t *buf,+ unsigned buflen, const uint8_t *table)+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/x86_64/proofs/mlkem_rej_uniform_avx2_asm.ml. */+__contract__(+ requires(buflen % 12 == 0)+ requires(memory_no_alias(buf, buflen))+ requires(table == mlk_rej_uniform_table)+ requires(memory_no_alias(r, sizeof(int16_t) * MLKEM_N))+ assigns(memory_slice(r, sizeof(int16_t) * MLKEM_N))+ ensures(return_value <= MLKEM_N)+ ensures(array_bound(r, 0, (unsigned) return_value, 0, MLKEM_Q))+);++#define mlk_ntt_avx2_asm MLK_NAMESPACE(ntt_avx2_asm)+MLK_SYSV_ABI+void mlk_ntt_avx2_asm(int16_t *r, const int16_t *qdata)+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/x86_64/proofs/mlkem_ntt_avx2_asm.ml */+__contract__(+ requires(memory_no_alias(r, sizeof(int16_t) * MLKEM_N))+ requires(array_abs_bound(r, 0, MLKEM_N, 8192))+ requires(qdata == mlk_qdata)+ assigns(memory_slice(r, sizeof(int16_t) * MLKEM_N))+ /* check-magic: off */+ ensures(array_abs_bound(r, 0, MLKEM_N, 23595))+ /* check-magic: on */+);++#define mlk_invntt_avx2_asm MLK_NAMESPACE(invntt_avx2_asm)+MLK_SYSV_ABI+void mlk_invntt_avx2_asm(int16_t *r, const int16_t *qdata)+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/x86_64/proofs/mlkem_intt_avx2_asm.ml */+__contract__(+ requires(memory_no_alias(r, sizeof(int16_t) * MLKEM_N))+ requires(qdata == mlk_qdata)+ assigns(memory_slice(r, sizeof(int16_t) * MLKEM_N))+ /* check-magic: off */+ ensures(array_abs_bound(r, 0, MLKEM_N, 26632))+ /* check-magic: on */+);++#define mlk_nttunpack_avx2_asm MLK_NAMESPACE(nttunpack_avx2_asm)+MLK_SYSV_ABI+void mlk_nttunpack_avx2_asm(int16_t *r)+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/x86_64/proofs/mlkem_nttunpack_avx2_asm.ml */+__contract__(+ requires(memory_no_alias(r, sizeof(int16_t) * MLKEM_N))+ requires(array_bound(r, 0, MLKEM_N, 0, MLKEM_Q))+ assigns(memory_slice(r, sizeof(int16_t) * MLKEM_N))+ /* Output is a permutation of input: every output coefficient+ * is some input coefficient */+ ensures(forall(i, 0, MLKEM_N, exists(j, 0, MLKEM_N,+ r[i] == old(*(int16_t (*)[MLKEM_N])r)[j])))+);++#define mlk_reduce_avx2_asm MLK_NAMESPACE(reduce_avx2_asm)+MLK_SYSV_ABI+void mlk_reduce_avx2_asm(int16_t *r)+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/x86_64/proofs/mlkem_reduce_avx2_asm.ml */+__contract__(+ requires(memory_no_alias(r, sizeof(int16_t) * MLKEM_N))+ assigns(memory_slice(r, sizeof(int16_t) * MLKEM_N))+ ensures(array_bound(r, 0, MLKEM_N, 0, MLKEM_Q))+);++#define mlk_poly_mulcache_compute_avx2_asm \+ MLK_NAMESPACE(poly_mulcache_compute_avx2_asm)+MLK_SYSV_ABI+void mlk_poly_mulcache_compute_avx2_asm(int16_t *out, const int16_t *in,+ const int16_t *qdata)+/* This must be kept in sync with the HOL-Light specification+ * in proofs/hol_light/x86_64/proofs/mlkem_poly_mulcache_compute_avx2_asm.ml */+__contract__(+ requires(memory_no_alias(out, sizeof(int16_t) * (MLKEM_N / 2)))+ requires(memory_no_alias(in, sizeof(int16_t) * MLKEM_N))+ requires(qdata == mlk_qdata)+ assigns(memory_slice(out, sizeof(int16_t) * (MLKEM_N / 2)))+ ensures(array_abs_bound(out, 0, MLKEM_N/2, MLKEM_Q))+);++#define mlk_polyvec_basemul_acc_montgomery_cached_k2_avx2_asm \+ MLK_NAMESPACE(polyvec_basemul_acc_montgomery_cached_k2_avx2_asm)+MLK_SYSV_ABI+void mlk_polyvec_basemul_acc_montgomery_cached_k2_avx2_asm(+ int16_t *r, const int16_t *a, const int16_t *b, const int16_t *b_cache)+/* This must be kept in sync with the HOL-Light specification in+ * proofs/hol_light/x86_64/proofs/mlkem_polyvec_basemul_acc_montgomery_cached_k2_avx2_asm.ml.+ */+__contract__(+ requires(memory_no_alias(r, sizeof(int16_t) * MLKEM_N))+ requires(memory_no_alias(a, sizeof(int16_t) * 2 * MLKEM_N))+ requires(memory_no_alias(b, sizeof(int16_t) * 2 * MLKEM_N))+ requires(memory_no_alias(b_cache, sizeof(int16_t) * 2 * (MLKEM_N / 2)))+ requires(array_abs_bound(a, 0, 2 * MLKEM_N, MLKEM_UINT12_LIMIT + 1))+ assigns(memory_slice(r, sizeof(int16_t) * MLKEM_N))+);++#define mlk_polyvec_basemul_acc_montgomery_cached_k3_avx2_asm \+ MLK_NAMESPACE(polyvec_basemul_acc_montgomery_cached_k3_avx2_asm)+MLK_SYSV_ABI+void mlk_polyvec_basemul_acc_montgomery_cached_k3_avx2_asm(+ int16_t *r, const int16_t *a, const int16_t *b, const int16_t *b_cache)+/* This must be kept in sync with the HOL-Light specification in+ * proofs/hol_light/x86_64/proofs/mlkem_polyvec_basemul_acc_montgomery_cached_k3_avx2_asm.ml.+ */+__contract__(+ requires(memory_no_alias(r, sizeof(int16_t) * MLKEM_N))+ requires(memory_no_alias(a, sizeof(int16_t) * 3 * MLKEM_N))+ requires(memory_no_alias(b, sizeof(int16_t) * 3 * MLKEM_N))+ requires(memory_no_alias(b_cache, sizeof(int16_t) * 3 * (MLKEM_N / 2)))+ requires(array_abs_bound(a, 0, 3 * MLKEM_N, MLKEM_UINT12_LIMIT + 1))+ assigns(memory_slice(r, sizeof(int16_t) * MLKEM_N))+);++#define mlk_polyvec_basemul_acc_montgomery_cached_k4_avx2_asm \+ MLK_NAMESPACE(polyvec_basemul_acc_montgomery_cached_k4_avx2_asm)+MLK_SYSV_ABI+void mlk_polyvec_basemul_acc_montgomery_cached_k4_avx2_asm(+ int16_t *r, const int16_t *a, const int16_t *b, const int16_t *b_cache)+/* This must be kept in sync with the HOL-Light specification in+ * proofs/hol_light/x86_64/proofs/mlkem_polyvec_basemul_acc_montgomery_cached_k4_avx2_asm.ml.+ */+__contract__(+ requires(memory_no_alias(r, sizeof(int16_t) * MLKEM_N))+ requires(memory_no_alias(a, sizeof(int16_t) * 4 * MLKEM_N))+ requires(memory_no_alias(b, sizeof(int16_t) * 4 * MLKEM_N))+ requires(memory_no_alias(b_cache, sizeof(int16_t) * 4 * (MLKEM_N / 2)))+ requires(array_abs_bound(a, 0, 4 * MLKEM_N, MLKEM_UINT12_LIMIT + 1))+ assigns(memory_slice(r, sizeof(int16_t) * MLKEM_N))+);++#define mlk_ntttobytes_avx2_asm MLK_NAMESPACE(ntttobytes_avx2_asm)+MLK_SYSV_ABI+void mlk_ntttobytes_avx2_asm(uint8_t *r, const int16_t *a)+/* This must be kept in sync with the HOL-Light specification in+ * proofs/hol_light/x86_64/proofs/mlkem_ntttobytes_avx2_asm.ml.+ */+__contract__(+ requires(memory_no_alias(r, MLKEM_POLYBYTES))+ requires(memory_no_alias(a, sizeof(int16_t) * MLKEM_N))+ requires(array_bound(a, 0, MLKEM_N, 0, MLKEM_Q))+ assigns(memory_slice(r, MLKEM_POLYBYTES))+);++#define mlk_nttfrombytes_avx2_asm MLK_NAMESPACE(nttfrombytes_avx2_asm)+MLK_SYSV_ABI+void mlk_nttfrombytes_avx2_asm(int16_t *r, const uint8_t *a)+/* This must be kept in sync with the HOL-Light specification in+ * proofs/hol_light/x86_64/proofs/mlkem_nttfrombytes_avx2_asm.ml.+ */+__contract__(+ requires(memory_no_alias(a, MLKEM_POLYBYTES))+ requires(memory_no_alias(r, sizeof(int16_t) * MLKEM_N))+ assigns(memory_slice(r, sizeof(int16_t) * MLKEM_N))+ ensures(array_bound(r, 0, MLKEM_N, 0, MLKEM_UINT12_LIMIT))+);++#define mlk_tomont_avx2_asm MLK_NAMESPACE(tomont_avx2_asm)+MLK_SYSV_ABI+void mlk_tomont_avx2_asm(int16_t *r)+/* This must be kept in sync with the HOL-Light specification in+ * proofs/hol_light/x86_64/proofs/mlkem_tomont_avx2_asm.ml.+ */+__contract__(+ requires(memory_no_alias(r, sizeof(int16_t) * MLKEM_N))+ assigns(memory_slice(r, sizeof(int16_t) * MLKEM_N))+ ensures(array_abs_bound(r, 0, MLKEM_N, MLKEM_Q))+);++#define mlk_poly_compress_d4_avx2_asm MLK_NAMESPACE(poly_compress_d4_avx2_asm)+MLK_SYSV_ABI+void mlk_poly_compress_d4_avx2_asm(uint8_t r[MLKEM_POLYCOMPRESSEDBYTES_D4],+ const int16_t *MLK_RESTRICT a,+ const uint8_t *data)+/* This must be kept in sync with the HOL-Light specification in+ * proofs/hol_light/x86_64/proofs/mlkem_poly_compress_d4_avx2_asm.ml.+ */+__contract__(+ requires(memory_no_alias(r, MLKEM_POLYCOMPRESSEDBYTES_D4))+ requires(memory_no_alias(a, sizeof(int16_t) * MLKEM_N))+ requires(array_bound(a, 0, MLKEM_N, 0, MLKEM_Q))+ requires(data == mlk_compress_d4_data)+ assigns(memory_slice(r, MLKEM_POLYCOMPRESSEDBYTES_D4))+);++#define mlk_poly_decompress_d4_avx2_asm \+ MLK_NAMESPACE(poly_decompress_d4_avx2_asm)+MLK_SYSV_ABI+void mlk_poly_decompress_d4_avx2_asm(+ int16_t *MLK_RESTRICT r, const uint8_t a[MLKEM_POLYCOMPRESSEDBYTES_D4],+ const uint8_t *data)+/* This must be kept in sync with the HOL-Light specification in+ * proofs/hol_light/x86_64/proofs/mlkem_poly_decompress_d4_avx2_asm.ml.+ */+__contract__(+ requires(memory_no_alias(r, sizeof(int16_t) * MLKEM_N))+ requires(memory_no_alias(a, MLKEM_POLYCOMPRESSEDBYTES_D4))+ requires(data == mlk_decompress_d4_data)+ assigns(memory_slice(r, sizeof(int16_t) * MLKEM_N))+ ensures(array_bound(r, 0, MLKEM_N, 0, MLKEM_Q))+);++#define mlk_poly_compress_d10_avx2_asm MLK_NAMESPACE(poly_compress_d10_avx2_asm)+MLK_SYSV_ABI+void mlk_poly_compress_d10_avx2_asm(uint8_t r[MLKEM_POLYCOMPRESSEDBYTES_D10],+ const int16_t *MLK_RESTRICT a,+ const uint8_t *data)+/* This must be kept in sync with the HOL-Light specification in+ * proofs/hol_light/x86_64/proofs/mlkem_poly_compress_d10_avx2_asm.ml.+ */+__contract__(+ requires(memory_no_alias(r, MLKEM_POLYCOMPRESSEDBYTES_D10))+ requires(memory_no_alias(a, sizeof(int16_t) * MLKEM_N))+ requires(array_bound(a, 0, MLKEM_N, 0, MLKEM_Q))+ requires(data == mlk_compress_d10_data)+ assigns(memory_slice(r, MLKEM_POLYCOMPRESSEDBYTES_D10))+);++#define mlk_poly_decompress_d10_avx2_asm \+ MLK_NAMESPACE(poly_decompress_d10_avx2_asm)+MLK_SYSV_ABI+void mlk_poly_decompress_d10_avx2_asm(+ int16_t *MLK_RESTRICT r, const uint8_t a[MLKEM_POLYCOMPRESSEDBYTES_D10],+ const uint8_t *data)+/* This must be kept in sync with the HOL-Light specification in+ * proofs/hol_light/x86_64/proofs/mlkem_poly_decompress_d10_avx2_asm.ml.+ */+__contract__(+ requires(memory_no_alias(r, sizeof(int16_t) * MLKEM_N))+ requires(memory_no_alias(a, MLKEM_POLYCOMPRESSEDBYTES_D10))+ requires(data == mlk_decompress_d10_data)+ assigns(memory_slice(r, sizeof(int16_t) * MLKEM_N))+ ensures(array_bound(r, 0, MLKEM_N, 0, MLKEM_Q))+);++#define mlk_poly_compress_d5_avx2_asm MLK_NAMESPACE(poly_compress_d5_avx2_asm)+MLK_SYSV_ABI+void mlk_poly_compress_d5_avx2_asm(uint8_t r[MLKEM_POLYCOMPRESSEDBYTES_D5],+ const int16_t *MLK_RESTRICT a,+ const uint8_t *data)+/* This must be kept in sync with the HOL-Light specification in+ * proofs/hol_light/x86_64/proofs/mlkem_poly_compress_d5_avx2_asm.ml.+ */+__contract__(+ requires(memory_no_alias(r, MLKEM_POLYCOMPRESSEDBYTES_D5))+ requires(memory_no_alias(a, sizeof(int16_t) * MLKEM_N))+ requires(array_bound(a, 0, MLKEM_N, 0, MLKEM_Q))+ requires(data == mlk_compress_d5_data)+ assigns(memory_slice(r, MLKEM_POLYCOMPRESSEDBYTES_D5))+);++#define mlk_poly_decompress_d5_avx2_asm \+ MLK_NAMESPACE(poly_decompress_d5_avx2_asm)+MLK_SYSV_ABI+void mlk_poly_decompress_d5_avx2_asm(+ int16_t *MLK_RESTRICT r, const uint8_t a[MLKEM_POLYCOMPRESSEDBYTES_D5],+ const uint8_t *data)+/* This must be kept in sync with the HOL-Light specification in+ * proofs/hol_light/x86_64/proofs/mlkem_poly_decompress_d5_avx2_asm.ml.+ */+__contract__(+ requires(memory_no_alias(r, sizeof(int16_t) * MLKEM_N))+ requires(memory_no_alias(a, MLKEM_POLYCOMPRESSEDBYTES_D5))+ requires(data == mlk_decompress_d5_data)+ assigns(memory_slice(r, sizeof(int16_t) * MLKEM_N))+ ensures(array_bound(r, 0, MLKEM_N, 0, MLKEM_Q))+);++#define mlk_poly_compress_d11_avx2_asm MLK_NAMESPACE(poly_compress_d11_avx2_asm)+MLK_SYSV_ABI+void mlk_poly_compress_d11_avx2_asm(uint8_t r[MLKEM_POLYCOMPRESSEDBYTES_D11],+ const int16_t *MLK_RESTRICT a,+ const uint8_t *data)+/* This must be kept in sync with the HOL-Light specification in+ * proofs/hol_light/x86_64/proofs/mlkem_poly_compress_d11_avx2_asm.ml.+ */+__contract__(+ requires(memory_no_alias(r, MLKEM_POLYCOMPRESSEDBYTES_D11))+ requires(memory_no_alias(a, sizeof(int16_t) * MLKEM_N))+ requires(array_bound(a, 0, MLKEM_N, 0, MLKEM_Q))+ requires(data == mlk_compress_d11_data)+ assigns(memory_slice(r, MLKEM_POLYCOMPRESSEDBYTES_D11))+);++#define mlk_poly_decompress_d11_avx2_asm \+ MLK_NAMESPACE(poly_decompress_d11_avx2_asm)+MLK_SYSV_ABI+void mlk_poly_decompress_d11_avx2_asm(+ int16_t *MLK_RESTRICT r, const uint8_t a[MLKEM_POLYCOMPRESSEDBYTES_D11],+ const uint8_t *data)+/* This must be kept in sync with the HOL-Light specification in+ * proofs/hol_light/x86_64/proofs/mlkem_poly_decompress_d11_avx2_asm.ml.+ */+__contract__(+ requires(memory_no_alias(r, sizeof(int16_t) * MLKEM_N))+ requires(memory_no_alias(a, MLKEM_POLYCOMPRESSEDBYTES_D11))+ requires(data == mlk_decompress_d11_data)+ assigns(memory_slice(r, sizeof(int16_t) * MLKEM_N))+ ensures(array_bound(r, 0, MLKEM_N, 0, MLKEM_Q))+);++#endif /* !MLK_NATIVE_X86_64_SRC_ARITH_NATIVE_X86_64_H */
@@ -0,0 +1,115 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*+ * WARNING: This file is auto-generated from scripts/autogen+ * in the mlkem-native repository.+ * Do not modify it directly.+ */++#include "../../../common.h"++#if defined(MLK_ARITH_BACKEND_X86_64_DEFAULT) && \+ (!defined(MLK_CONFIG_NO_ENCAPS_API) || !defined(MLK_CONFIG_NO_DECAPS_API))++#include "compress_consts.h"++#if !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED) && \+ (defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_K == 2 || \+ MLKEM_K == 3)++MLK_ALIGN MLK_INTERNAL_DATA_DEFINITION const uint8_t+ mlk_compress_d4_data[32] = {+ 0, 0, 0, 0, 4, 0, 0, 0, 1, 0, 0, 0, 5, 0, 0, 0,+ 2, 0, 0, 0, 6, 0, 0, 0, 3, 0, 0, 0, 7, 0, 0, 0, /* permdidx */+};++#if !defined(MLK_CONFIG_NO_DECAPS_API)+MLK_ALIGN MLK_INTERNAL_DATA_DEFINITION const uint8_t+ mlk_decompress_d4_data[32] = {+ 0, 0, 0, 0, 1, 1, 1, 1, 2, 2, 2, 2, 3, 3, 3, 3,+ 4, 4, 4, 4, 5, 5, 5, 5, 6, 6, 6, 6, 7, 7, 7, 7, /* shufbidx */+};+#endif /* !MLK_CONFIG_NO_DECAPS_API */++MLK_ALIGN MLK_INTERNAL_DATA_DEFINITION const uint8_t+ mlk_compress_d10_data[32] = {+ 0, 1, 2, 3, 4, 8, 9, 10, 11, 12, 255,+ 255, 255, 255, 255, 255, 9, 10, 11, 12, 255, 255,+ 255, 255, 255, 255, 0, 1, 2, 3, 4, 8, /* shufbidx */+};++#if !defined(MLK_CONFIG_NO_DECAPS_API)+MLK_ALIGN MLK_INTERNAL_DATA_DEFINITION const uint8_t+ mlk_decompress_d10_data[32] = {+ 0, 1, 1, 2, 2, 3, 3, 4, 5, 6, 6, 7, 7, 8, 8, 9,+ 2, 3, 3, 4, 4, 5, 5, 6, 7, 8, 8, 9, 9, 10, 10, 11, /* shufbidx */+};+#endif /* !MLK_CONFIG_NO_DECAPS_API */++#endif /* !MLK_CONFIG_MULTILEVEL_NO_SHARED && \+ (MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 2 || MLKEM_K == 3) \+ */++#if !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED) && \+ (defined(MLK_CONFIG_MULTILEVEL_WITH_SHARED) || MLKEM_K == 4)++MLK_ALIGN MLK_INTERNAL_DATA_DEFINITION const uint8_t+ mlk_compress_d5_data[32] = {+ 0, 1, 2, 3, 4, 255, 255, 255, 255, 255, 8,+ 9, 10, 11, 12, 255, 9, 10, 11, 12, 255, 0,+ 1, 2, 3, 4, 255, 255, 255, 255, 255, 8, /* shufbidx */+};++#if !defined(MLK_CONFIG_NO_DECAPS_API)+/* shufbidx[0:32], mask[32:64], shift[64:96] */+MLK_ALIGN MLK_INTERNAL_DATA_DEFINITION const uint8_t+ mlk_decompress_d5_data[96] = {+ 0, 0, 0, 1, 1, 1, 1, 2, 2, 3, 3, 3, 3, 4, 4, 4, 5, 5,+ 5, 6, 6, 6, 6, 7, 7, 8, 8, 8, 8, 9, 9, 9, /* shufbidx */+ 31, 0, 224, 3, 124, 0, 128, 15, 240, 1, 62, 0, 192, 7, 248, 0, 31, 0,+ 224, 3, 124, 0, 128, 15, 240, 1, 62, 0, 192, 7, 248, 0, /* mask */+ 0, 4, 32, 0, 0, 1, 8, 0, 64, 0, 0, 2, 16, 0, 128, 0, 0, 4,+ 32, 0, 0, 1, 8, 0, 64, 0, 0, 2, 16, 0, 128, 0, /* shift */+};+#endif /* !MLK_CONFIG_NO_DECAPS_API */++/* srlvqidx[0:32], shufbidx[32:64] */+MLK_ALIGN MLK_INTERNAL_DATA_DEFINITION const uint8_t+ mlk_compress_d11_data[64] =+ {+ 10, 0, 0, 0, 0, 0, 0, 0, 30, 0, 0,+ 0, 0, 0, 0, 0, 10, 0, 0, 0, 0, 0,+ 0, 0, 30, 0, 0, 0, 0, 0, 0, 0, /* srlvqidx */+ 0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10,+ 255, 255, 255, 255, 255, 5, 6, 7, 8, 9, 10,+ 255, 255, 255, 255, 0, 0, 1, 2, 3, 4, /* shufbidx */+};++#if !defined(MLK_CONFIG_NO_DECAPS_API)+/* shufbidx[0:32], srlvdidx[32:64], srlvqidx[64:96], shift[96:128] */+MLK_ALIGN MLK_INTERNAL_DATA_DEFINITION const uint8_t+ mlk_decompress_d11_data[128] = {+ 0, 1, 1, 2, 2, 3, 4, 5, 5, 6, 6, 7, 8, 9, 9, 10,+ 3, 4, 4, 5, 5, 6, 7, 8, 8, 9, 9, 10, 11, 12, 12, 13, /* shufbidx */+ 0, 0, 0, 0, 1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,+ 0, 0, 0, 0, 1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, /* srlvdidx */+ 0, 0, 0, 0, 0, 0, 0, 0, 2, 0, 0, 0, 0, 0, 0, 0,+ 0, 0, 0, 0, 0, 0, 0, 0, 2, 0, 0, 0, 0, 0, 0, 0, /* srlvqidx */+ 32, 0, 4, 0, 1, 0, 32, 0, 8, 0, 1, 0, 32, 0, 4, 0,+ 32, 0, 4, 0, 1, 0, 32, 0, 8, 0, 1, 0, 32, 0, 4, 0, /* shift */+};+#endif /* !MLK_CONFIG_NO_DECAPS_API */++#endif /* !MLK_CONFIG_MULTILEVEL_NO_SHARED && \+ (MLK_CONFIG_MULTILEVEL_WITH_SHARED || MLKEM_K == 4) */++#else /* MLK_ARITH_BACKEND_X86_64_DEFAULT && (!MLK_CONFIG_NO_ENCAPS_API || \+ !MLK_CONFIG_NO_DECAPS_API) */++MLK_EMPTY_CU(avx2_compress_consts)++#endif /* !(MLK_ARITH_BACKEND_X86_64_DEFAULT && (!MLK_CONFIG_NO_ENCAPS_API || \+ !MLK_CONFIG_NO_DECAPS_API)) */
@@ -0,0 +1,53 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*+ * WARNING: This file is auto-generated from scripts/autogen+ * in the mlkem-native repository.+ * Do not modify it directly.+ */++#ifndef MLK_NATIVE_X86_64_SRC_COMPRESS_CONSTS_H+#define MLK_NATIVE_X86_64_SRC_COMPRESS_CONSTS_H++#include "../../../common.h"++#ifndef __ASSEMBLER__++#define mlk_compress_d4_data MLK_NAMESPACE(compress_d4_data)+MLK_INTERNAL_DATA_DECLARATION const uint8_t mlk_compress_d4_data[32];++#if !defined(MLK_CONFIG_NO_DECAPS_API)+#define mlk_decompress_d4_data MLK_NAMESPACE(decompress_d4_data)+MLK_INTERNAL_DATA_DECLARATION const uint8_t mlk_decompress_d4_data[32];+#endif++#define mlk_compress_d10_data MLK_NAMESPACE(compress_d10_data)+MLK_INTERNAL_DATA_DECLARATION const uint8_t mlk_compress_d10_data[32];++#if !defined(MLK_CONFIG_NO_DECAPS_API)+#define mlk_decompress_d10_data MLK_NAMESPACE(decompress_d10_data)+MLK_INTERNAL_DATA_DECLARATION const uint8_t mlk_decompress_d10_data[32];+#endif++#define mlk_compress_d5_data MLK_NAMESPACE(compress_d5_data)+MLK_INTERNAL_DATA_DECLARATION const uint8_t mlk_compress_d5_data[32];++#if !defined(MLK_CONFIG_NO_DECAPS_API)+#define mlk_decompress_d5_data MLK_NAMESPACE(decompress_d5_data)+MLK_INTERNAL_DATA_DECLARATION const uint8_t mlk_decompress_d5_data[96];+#endif++#define mlk_compress_d11_data MLK_NAMESPACE(compress_d11_data)+MLK_INTERNAL_DATA_DECLARATION const uint8_t mlk_compress_d11_data[64];++#if !defined(MLK_CONFIG_NO_DECAPS_API)+#define mlk_decompress_d11_data MLK_NAMESPACE(decompress_d11_data)+MLK_INTERNAL_DATA_DECLARATION const uint8_t mlk_decompress_d11_data[128];+#endif++#endif /* !__ASSEMBLER__ */++#endif /* !MLK_NATIVE_X86_64_SRC_COMPRESS_CONSTS_H */
@@ -0,0 +1,102 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*+ * WARNING: This file is auto-generated from scripts/autogen+ * in the mlkem-native repository.+ * Do not modify it directly.+ */++#include "../../../common.h"++#if defined(MLK_ARITH_BACKEND_X86_64_DEFAULT) && \+ !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED)++#include "consts.h"++/*+ * Table of zeta values used in the AVX2 NTTs+ * See autogen for details.+ */+MLK_ALIGN MLK_INTERNAL_DATA_DEFINITION const int16_t mlk_qdata[624] = {+ 3854, 3340, 2826, 2312, 1798, 1284, 770, 256, 3854,+ 3340, 2826, 2312, 1798, 1284, 770, 256, 7, 0,+ 6, 0, 5, 0, 4, 0, 3, 0, 2,+ 0, 1, 0, 0, 0, 31498, 31498, 31498, 31498,+ -758, -758, -758, -758, 0, 0, 0, 0, 0,+ 0, 0, 0, 14745, 14745, 14745, 14745, 14745, 14745,+ 14745, 14745, 14745, 14745, 14745, 14745, 14745, 14745, 14745,+ 14745, -359, -359, -359, -359, -359, -359, -359, -359,+ -359, -359, -359, -359, -359, -359, -359, -359, 13525,+ 13525, 13525, 13525, 13525, 13525, 13525, 13525, -12402, -12402,+ -12402, -12402, -12402, -12402, -12402, -12402, 1493, 1493, 1493,+ 1493, 1493, 1493, 1493, 1493, 1422, 1422, 1422, 1422,+ 1422, 1422, 1422, 1422, -20907, -20907, -20907, -20907, 27758,+ 27758, 27758, 27758, -3799, -3799, -3799, -3799, -15690, -15690,+ -15690, -15690, -171, -171, -171, -171, 622, 622, 622,+ 622, 1577, 1577, 1577, 1577, 182, 182, 182, 182,+ -5827, -5827, 17363, 17363, -26360, -26360, -29057, -29057, 5571,+ 5571, -1102, -1102, 21438, 21438, -26242, -26242, 573, 573,+ -1325, -1325, 264, 264, 383, 383, -829, -829, 1458,+ 1458, -1602, -1602, -130, -130, -5689, -6516, 1496, 30967,+ -23565, 20179, 20710, 25080, -12796, 26616, 16064, -12442, 9134,+ -650, -25986, 27837, 1223, 652, -552, 1015, -1293, 1491,+ -282, -1544, 516, -8, -320, -666, -1618, -1162, 126,+ 1469, -335, -11477, -32227, 20494, -27738, 945, -14883, 6182,+ 32010, 10631, 29175, -28762, -18486, 17560, -14430, -5276, -1103,+ 555, -1251, 1550, 422, 177, -291, 1574, -246, 1159,+ -777, -602, -1590, -872, 418, -156, 11182, 13387, -14233,+ -21655, 13131, -4587, 23092, 5493, -32502, 30317, -18741, 12639,+ 20100, 18525, 19529, -12619, 430, 843, 871, 105, 587,+ -235, -460, 1653, 778, -147, 1483, 1119, 644, 349,+ 329, -75, 787, 787, 787, 787, 787, 787, 787,+ 787, 787, 787, 787, 787, 787, 787, 787, 787,+ -1517, -1517, -1517, -1517, -1517, -1517, -1517, -1517, -1517,+ -1517, -1517, -1517, -1517, -1517, -1517, -1517, 28191, 28191,+ 28191, 28191, 28191, 28191, 28191, 28191, -16694, -16694, -16694,+ -16694, -16694, -16694, -16694, -16694, 287, 287, 287, 287,+ 287, 287, 287, 287, 202, 202, 202, 202, 202,+ 202, 202, 202, 10690, 10690, 10690, 10690, 1358, 1358,+ 1358, 1358, -11202, -11202, -11202, -11202, 31164, 31164, 31164,+ 31164, 962, 962, 962, 962, -1202, -1202, -1202, -1202,+ -1474, -1474, -1474, -1474, 1468, 1468, 1468, 1468, -28073,+ -28073, 24313, 24313, -10532, -10532, 8800, 8800, 18426, 18426,+ 8859, 8859, 26675, 26675, -16163, -16163, -681, -681, 1017,+ 1017, 732, 732, 608, 608, -1542, -1542, 411, 411,+ -205, -205, -1571, -1571, 19883, -28250, -15887, -8898, -28309,+ 9075, -30199, 18249, 13426, 14017, -29156, -12757, 16832, 4311,+ -24155, -17915, -853, -90, -271, 830, 107, -1421, -247,+ -951, -398, 961, -1508, -725, 448, -1065, 677, -1275,+ -31183, 25435, -7382, 24391, -20927, 10946, 24214, 16989, 10335,+ -7934, -22502, 10906, 31636, 28644, 23998, -17422, 817, 603,+ 1322, -1465, -1215, 1218, -874, -1187, -1185, -1278, -1510,+ -870, -108, 996, 958, 1522, 20297, 2146, 15355, -32384,+ -6280, -14903, -11044, 14469, -21498, -20198, 23210, -17442, -23860,+ -20257, 7756, 23132, 1097, 610, -1285, 384, -136, -1335,+ 220, -1659, -1530, 794, -854, 478, -308, 991, -1460,+ 1628, -1103, 555, -1251, 1550, 422, 177, -291, 1574,+ -246, 1159, -777, -602, -1590, -872, 418, -156, 430,+ 843, 871, 105, 587, -235, -460, 1653, 778, -147,+ 1483, 1119, 644, 349, 329, -75, 817, 603, 1322,+ -1465, -1215, 1218, -874, -1187, -1185, -1278, -1510, -870,+ -108, 996, 958, 1522, 1097, 610, -1285, 384, -136,+ -1335, 220, -1659, -1530, 794, -854, 478, -308, 991,+ -1460, 1628, -335, -11477, -32227, 20494, -27738, 945, -14883,+ 6182, 32010, 10631, 29175, -28762, -18486, 17560, -14430, -5276,+ 11182, 13387, -14233, -21655, 13131, -4587, 23092, 5493, -32502,+ 30317, -18741, 12639, 20100, 18525, 19529, -12619, -31183, 25435,+ -7382, 24391, -20927, 10946, 24214, 16989, 10335, -7934, -22502,+ 10906, 31636, 28644, 23998, -17422, 20297, 2146, 15355, -32384,+ -6280, -14903, -11044, 14469, -21498, -20198, 23210, -17442, -23860,+ -20257, 7756, 23132,+};++#else /* MLK_ARITH_BACKEND_X86_64_DEFAULT && !MLK_CONFIG_MULTILEVEL_NO_SHARED \+ */++MLK_EMPTY_CU(avx2_consts)++#endif /* !(MLK_ARITH_BACKEND_X86_64_DEFAULT && \+ !MLK_CONFIG_MULTILEVEL_NO_SHARED) */
@@ -0,0 +1,25 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/*+ * WARNING: This file is auto-generated from scripts/autogen+ * in the mlkem-native repository.+ * Do not modify it directly.+ */++#ifndef MLK_NATIVE_X86_64_SRC_CONSTS_H+#define MLK_NATIVE_X86_64_SRC_CONSTS_H+#include "../../../common.h"+#define MLK_AVX2_BACKEND_DATA_OFFSET_REVIDXB 0+#define MLK_AVX2_BACKEND_DATA_OFFSET_REVIDXD 16+#define MLK_AVX2_BACKEND_DATA_OFFSET_ZETAS_EXP 32+#define MLK_AVX2_BACKEND_DATA_OFFSET_MULCACHE_TWIDDLES 496++#ifndef __ASSEMBLER__+#define mlk_qdata MLK_NAMESPACE(qdata)+MLK_INTERNAL_DATA_DECLARATION const int16_t mlk_qdata[624];+#endif++#endif /* !MLK_NATIVE_X86_64_SRC_CONSTS_H */
@@ -0,0 +1,743 @@+/*+ * Copyright (c) The mlkem-native project authors+ * SPDX-License-Identifier: Apache-2.0 OR ISC OR MIT+ */++/* References+ * ==========+ *+ * - [AVX2_NTT]+ * Faster AVX2 optimized NTT multiplication for Ring-LWE lattice cryptography.+ * Gregor Seiler+ * https://eprint.iacr.org/2018/039+ *+ * - [REF_AVX2]+ * CRYSTALS-Kyber optimized AVX2 implementation+ * Bos, Ducas, Kiltz, Lepoint, Lyubashevsky, Schanck, Schwabe, Seiler, Stehlé+ * https://github.com/pq-crystals/kyber/tree/main/avx2+ */++/*+ * This file is derived from the public domain+ * AVX2 Kyber implementation @[REF_AVX2].+ *+ * The core ideas behind the implementation are described in @[AVX2_NTT].+ *+ * Changes:+ * - Different placement of modular reductions to simplify+ * reasoning of non-overflow+ */++#include "../../../common.h"+#if defined(MLK_ARITH_BACKEND_X86_64_DEFAULT) && \+ !defined(MLK_CONFIG_MULTILEVEL_NO_SHARED) && \+ (!defined(MLK_CONFIG_NO_ENCAPS_API) || \+ !defined(MLK_CONFIG_NO_DECAPS_API))+/*yaml+ Name: invntt_avx2_asm+ Description: x86_64 AVX2 inverse NTT+ Signature: void mlk_invntt_avx2_asm(int16_t *r, const int16_t *qdata)+ ABI:+ Architecture: x86_64+ CallingConvention: SysV+ Features: [AVX2]+ rdi:+ type: buffer+ size_bytes: 512+ permissions: read/write+ c_parameter: int16_t *r+ description: Input/output polynomial (256 x int16_t)+ rsi:+ type: buffer+ size_bytes: 1248+ permissions: read-only+ c_parameter: const int16_t *qdata+ description: Precomputed constants (624 x int16_t)+*/+++/*+ * WARNING: This file is auto-derived from the mlkem-native source file+ * dev/x86_64/src/mlkem_intt_avx2_asm.S using scripts/simpasm. Do not modify it directly.+ */++.text+.balign 4+.global MLK_ASM_NAMESPACE(invntt_avx2_asm)+MLK_ASM_FN_SYMBOL(invntt_avx2_asm)++ .cfi_startproc+ movl $0xd010d01, %eax # imm = 0xD010D01+ vmovd %eax, %xmm0+ vpbroadcastd %xmm0, %ymm0+ movl $0xd8a1d8a1, %eax # imm = 0xD8A1D8A1+ vmovd %eax, %xmm2+ vpbroadcastd %xmm2, %ymm2+ movl $0x5a105a1, %eax # imm = 0x5A105A1+ vmovd %eax, %xmm3+ vpbroadcastd %xmm3, %ymm3+ vmovdqa (%rdi), %ymm4+ vmovdqa 0x40(%rdi), %ymm6+ vmovdqa 0x20(%rdi), %ymm5+ vmovdqa 0x60(%rdi), %ymm7+ vpmullw %ymm2, %ymm4, %ymm12+ vpmulhw %ymm3, %ymm4, %ymm4+ vpmulhw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm4, %ymm4+ vpmullw %ymm2, %ymm6, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm6+ vpmulhw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm6, %ymm6+ vpmullw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm5, %ymm5+ vpmulhw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm5, %ymm5+ vpmullw %ymm2, %ymm7, %ymm12+ vpmulhw %ymm3, %ymm7, %ymm7+ vpmulhw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm7, %ymm7+ vmovdqa 0x80(%rdi), %ymm8+ vmovdqa 0xc0(%rdi), %ymm10+ vmovdqa 0xa0(%rdi), %ymm9+ vmovdqa 0xe0(%rdi), %ymm11+ vpmullw %ymm2, %ymm8, %ymm12+ vpmulhw %ymm3, %ymm8, %ymm8+ vpmulhw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm8, %ymm8+ vpmullw %ymm2, %ymm10, %ymm12+ vpmulhw %ymm3, %ymm10, %ymm10+ vpmulhw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm10, %ymm10+ vpmullw %ymm2, %ymm9, %ymm12+ vpmulhw %ymm3, %ymm9, %ymm9+ vpmulhw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm9, %ymm9+ vpmullw %ymm2, %ymm11, %ymm12+ vpmulhw %ymm3, %ymm11, %ymm11+ vpmulhw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm11, %ymm11+ vpermq $0x4e, 0x3a0(%rsi), %ymm15 # ymm15 = mem[2,3,0,1]+ vpermq $0x4e, 0x360(%rsi), %ymm1 # ymm1 = mem[2,3,0,1]+ vpermq $0x4e, 0x3c0(%rsi), %ymm2 # ymm2 = mem[2,3,0,1]+ vpermq $0x4e, 0x380(%rsi), %ymm3 # ymm3 = mem[2,3,0,1]+ vmovdqa (%rsi), %ymm12+ vpshufb %ymm12, %ymm15, %ymm15+ vpshufb %ymm12, %ymm1, %ymm1+ vpshufb %ymm12, %ymm2, %ymm2+ vpshufb %ymm12, %ymm3, %ymm3+ vpsubw %ymm4, %ymm6, %ymm12+ vpaddw %ymm6, %ymm4, %ymm4+ vpsubw %ymm5, %ymm7, %ymm13+ vpmullw %ymm15, %ymm12, %ymm6+ vpaddw %ymm7, %ymm5, %ymm5+ vpsubw %ymm8, %ymm10, %ymm14+ vpmullw %ymm15, %ymm13, %ymm7+ vpaddw %ymm10, %ymm8, %ymm8+ vpsubw %ymm9, %ymm11, %ymm15+ vpmullw %ymm1, %ymm14, %ymm10+ vpaddw %ymm11, %ymm9, %ymm9+ vpmullw %ymm1, %ymm15, %ymm11+ vpmulhw %ymm2, %ymm12, %ymm12+ vpmulhw %ymm2, %ymm13, %ymm13+ vpmulhw %ymm3, %ymm14, %ymm14+ vpmulhw %ymm3, %ymm15, %ymm15+ vpmulhw %ymm0, %ymm6, %ymm6+ vpmulhw %ymm0, %ymm7, %ymm7+ vpmulhw %ymm0, %ymm10, %ymm10+ vpmulhw %ymm0, %ymm11, %ymm11+ vpsubw %ymm6, %ymm12, %ymm6+ vpsubw %ymm7, %ymm13, %ymm7+ vpsubw %ymm10, %ymm14, %ymm10+ vpsubw %ymm11, %ymm15, %ymm11+ vpermq $0x4e, 0x320(%rsi), %ymm2 # ymm2 = mem[2,3,0,1]+ vpermq $0x4e, 0x340(%rsi), %ymm3 # ymm3 = mem[2,3,0,1]+ vmovdqa (%rsi), %ymm1+ vpshufb %ymm1, %ymm2, %ymm2+ vpshufb %ymm1, %ymm3, %ymm3+ vpsubw %ymm4, %ymm8, %ymm12+ vpaddw %ymm8, %ymm4, %ymm4+ vpsubw %ymm5, %ymm9, %ymm13+ vpmullw %ymm2, %ymm12, %ymm8+ vpaddw %ymm9, %ymm5, %ymm5+ vpsubw %ymm6, %ymm10, %ymm14+ vpmullw %ymm2, %ymm13, %ymm9+ vpaddw %ymm10, %ymm6, %ymm6+ vpsubw %ymm7, %ymm11, %ymm15+ vpmullw %ymm2, %ymm14, %ymm10+ vpaddw %ymm11, %ymm7, %ymm7+ vpmullw %ymm2, %ymm15, %ymm11+ vpmulhw %ymm3, %ymm12, %ymm12+ vpmulhw %ymm3, %ymm13, %ymm13+ vpmulhw %ymm3, %ymm14, %ymm14+ vpmulhw %ymm3, %ymm15, %ymm15+ vpmulhw %ymm0, %ymm8, %ymm8+ vpmulhw %ymm0, %ymm9, %ymm9+ vpmulhw %ymm0, %ymm10, %ymm10+ vpmulhw %ymm0, %ymm11, %ymm11+ vpsubw %ymm8, %ymm12, %ymm8+ vpsubw %ymm9, %ymm13, %ymm9+ vpsubw %ymm10, %ymm14, %ymm10+ vpsubw %ymm11, %ymm15, %ymm11+ vpslld $0x10, %ymm5, %ymm3+ vpblendw $0xaa, %ymm3, %ymm4, %ymm3 # ymm3 = ymm4[0],ymm3[1],ymm4[2],ymm3[3],ymm4[4],ymm3[5],ymm4[6],ymm3[7],ymm4[8],ymm3[9],ymm4[10],ymm3[11],ymm4[12],ymm3[13],ymm4[14],ymm3[15]+ vpsrld $0x10, %ymm4, %ymm4+ vpblendw $0xaa, %ymm5, %ymm4, %ymm5 # ymm5 = ymm4[0],ymm5[1],ymm4[2],ymm5[3],ymm4[4],ymm5[5],ymm4[6],ymm5[7],ymm4[8],ymm5[9],ymm4[10],ymm5[11],ymm4[12],ymm5[13],ymm4[14],ymm5[15]+ vpslld $0x10, %ymm7, %ymm4+ vpblendw $0xaa, %ymm4, %ymm6, %ymm4 # ymm4 = ymm6[0],ymm4[1],ymm6[2],ymm4[3],ymm6[4],ymm4[5],ymm6[6],ymm4[7],ymm6[8],ymm4[9],ymm6[10],ymm4[11],ymm6[12],ymm4[13],ymm6[14],ymm4[15]+ vpsrld $0x10, %ymm6, %ymm6+ vpblendw $0xaa, %ymm7, %ymm6, %ymm7 # ymm7 = ymm6[0],ymm7[1],ymm6[2],ymm7[3],ymm6[4],ymm7[5],ymm6[6],ymm7[7],ymm6[8],ymm7[9],ymm6[10],ymm7[11],ymm6[12],ymm7[13],ymm6[14],ymm7[15]+ vpslld $0x10, %ymm9, %ymm6+ vpblendw $0xaa, %ymm6, %ymm8, %ymm6 # ymm6 = ymm8[0],ymm6[1],ymm8[2],ymm6[3],ymm8[4],ymm6[5],ymm8[6],ymm6[7],ymm8[8],ymm6[9],ymm8[10],ymm6[11],ymm8[12],ymm6[13],ymm8[14],ymm6[15]+ vpsrld $0x10, %ymm8, %ymm8+ vpblendw $0xaa, %ymm9, %ymm8, %ymm9 # ymm9 = ymm8[0],ymm9[1],ymm8[2],ymm9[3],ymm8[4],ymm9[5],ymm8[6],ymm9[7],ymm8[8],ymm9[9],ymm8[10],ymm9[11],ymm8[12],ymm9[13],ymm8[14],ymm9[15]+ vpslld $0x10, %ymm11, %ymm8+ vpblendw $0xaa, %ymm8, %ymm10, %ymm8 # ymm8 = ymm10[0],ymm8[1],ymm10[2],ymm8[3],ymm10[4],ymm8[5],ymm10[6],ymm8[7],ymm10[8],ymm8[9],ymm10[10],ymm8[11],ymm10[12],ymm8[13],ymm10[14],ymm8[15]+ vpsrld $0x10, %ymm10, %ymm10+ vpblendw $0xaa, %ymm11, %ymm10, %ymm11 # ymm11 = ymm10[0],ymm11[1],ymm10[2],ymm11[3],ymm10[4],ymm11[5],ymm10[6],ymm11[7],ymm10[8],ymm11[9],ymm10[10],ymm11[11],ymm10[12],ymm11[13],ymm10[14],ymm11[15]+ vmovdqa 0x20(%rsi), %ymm12+ vpermd 0x2e0(%rsi), %ymm12, %ymm2+ vpermd 0x300(%rsi), %ymm12, %ymm10+ vpsubw %ymm3, %ymm5, %ymm12+ vpaddw %ymm5, %ymm3, %ymm3+ vpsubw %ymm4, %ymm7, %ymm13+ vpmullw %ymm2, %ymm12, %ymm5+ vpaddw %ymm7, %ymm4, %ymm4+ vpsubw %ymm6, %ymm9, %ymm14+ vpmullw %ymm2, %ymm13, %ymm7+ vpaddw %ymm9, %ymm6, %ymm6+ vpsubw %ymm8, %ymm11, %ymm15+ vpmullw %ymm2, %ymm14, %ymm9+ vpaddw %ymm11, %ymm8, %ymm8+ vpmullw %ymm2, %ymm15, %ymm11+ vpmulhw %ymm10, %ymm12, %ymm12+ vpmulhw %ymm10, %ymm13, %ymm13+ vpmulhw %ymm10, %ymm14, %ymm14+ vpmulhw %ymm10, %ymm15, %ymm15+ vpmulhw %ymm0, %ymm5, %ymm5+ vpmulhw %ymm0, %ymm7, %ymm7+ vpmulhw %ymm0, %ymm9, %ymm9+ vpmulhw %ymm0, %ymm11, %ymm11+ vpsubw %ymm5, %ymm12, %ymm5+ vpsubw %ymm7, %ymm13, %ymm7+ vpsubw %ymm9, %ymm14, %ymm9+ vpsubw %ymm11, %ymm15, %ymm11+ movl $0x4ebf4ebf, %eax # imm = 0x4EBF4EBF+ vmovd %eax, %xmm1+ vpbroadcastd %xmm1, %ymm1+ vpmulhw %ymm1, %ymm3, %ymm12+ vpsraw $0xa, %ymm12, %ymm12+ vpmullw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm3, %ymm3+ vmovsldup %ymm4, %ymm10 # ymm10 = ymm4[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm10, %ymm3, %ymm10 # ymm10 = ymm3[0],ymm10[1],ymm3[2],ymm10[3],ymm3[4],ymm10[5],ymm3[6],ymm10[7]+ vpsrlq $0x20, %ymm3, %ymm3+ vpblendd $0xaa, %ymm4, %ymm3, %ymm4 # ymm4 = ymm3[0],ymm4[1],ymm3[2],ymm4[3],ymm3[4],ymm4[5],ymm3[6],ymm4[7]+ vmovsldup %ymm8, %ymm3 # ymm3 = ymm8[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm3, %ymm6, %ymm3 # ymm3 = ymm6[0],ymm3[1],ymm6[2],ymm3[3],ymm6[4],ymm3[5],ymm6[6],ymm3[7]+ vpsrlq $0x20, %ymm6, %ymm6+ vpblendd $0xaa, %ymm8, %ymm6, %ymm8 # ymm8 = ymm6[0],ymm8[1],ymm6[2],ymm8[3],ymm6[4],ymm8[5],ymm6[6],ymm8[7]+ vmovsldup %ymm7, %ymm6 # ymm6 = ymm7[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm6, %ymm5, %ymm6 # ymm6 = ymm5[0],ymm6[1],ymm5[2],ymm6[3],ymm5[4],ymm6[5],ymm5[6],ymm6[7]+ vpsrlq $0x20, %ymm5, %ymm5+ vpblendd $0xaa, %ymm7, %ymm5, %ymm7 # ymm7 = ymm5[0],ymm7[1],ymm5[2],ymm7[3],ymm5[4],ymm7[5],ymm5[6],ymm7[7]+ vmovsldup %ymm11, %ymm5 # ymm5 = ymm11[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm5, %ymm9, %ymm5 # ymm5 = ymm9[0],ymm5[1],ymm9[2],ymm5[3],ymm9[4],ymm5[5],ymm9[6],ymm5[7]+ vpsrlq $0x20, %ymm9, %ymm9+ vpblendd $0xaa, %ymm11, %ymm9, %ymm11 # ymm11 = ymm9[0],ymm11[1],ymm9[2],ymm11[3],ymm9[4],ymm11[5],ymm9[6],ymm11[7]+ vpermq $0x1b, 0x2a0(%rsi), %ymm2 # ymm2 = mem[3,2,1,0]+ vpermq $0x1b, 0x2c0(%rsi), %ymm9 # ymm9 = mem[3,2,1,0]+ vpsubw %ymm10, %ymm4, %ymm12+ vpaddw %ymm4, %ymm10, %ymm10+ vpsubw %ymm3, %ymm8, %ymm13+ vpmullw %ymm2, %ymm12, %ymm4+ vpaddw %ymm8, %ymm3, %ymm3+ vpsubw %ymm6, %ymm7, %ymm14+ vpmullw %ymm2, %ymm13, %ymm8+ vpaddw %ymm7, %ymm6, %ymm6+ vpsubw %ymm5, %ymm11, %ymm15+ vpmullw %ymm2, %ymm14, %ymm7+ vpaddw %ymm11, %ymm5, %ymm5+ vpmullw %ymm2, %ymm15, %ymm11+ vpmulhw %ymm9, %ymm12, %ymm12+ vpmulhw %ymm9, %ymm13, %ymm13+ vpmulhw %ymm9, %ymm14, %ymm14+ vpmulhw %ymm9, %ymm15, %ymm15+ vpmulhw %ymm0, %ymm4, %ymm4+ vpmulhw %ymm0, %ymm8, %ymm8+ vpmulhw %ymm0, %ymm7, %ymm7+ vpmulhw %ymm0, %ymm11, %ymm11+ vpsubw %ymm4, %ymm12, %ymm4+ vpsubw %ymm8, %ymm13, %ymm8+ vpsubw %ymm7, %ymm14, %ymm7+ vpsubw %ymm11, %ymm15, %ymm11+ vpmulhw %ymm1, %ymm10, %ymm12+ vpsraw $0xa, %ymm12, %ymm12+ vpmullw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm10, %ymm10+ vpunpcklqdq %ymm3, %ymm10, %ymm9 # ymm9 = ymm10[0],ymm3[0],ymm10[2],ymm3[2]+ vpunpckhqdq %ymm3, %ymm10, %ymm3 # ymm3 = ymm10[1],ymm3[1],ymm10[3],ymm3[3]+ vpunpcklqdq %ymm5, %ymm6, %ymm10 # ymm10 = ymm6[0],ymm5[0],ymm6[2],ymm5[2]+ vpunpckhqdq %ymm5, %ymm6, %ymm5 # ymm5 = ymm6[1],ymm5[1],ymm6[3],ymm5[3]+ vpunpcklqdq %ymm8, %ymm4, %ymm6 # ymm6 = ymm4[0],ymm8[0],ymm4[2],ymm8[2]+ vpunpckhqdq %ymm8, %ymm4, %ymm8 # ymm8 = ymm4[1],ymm8[1],ymm4[3],ymm8[3]+ vpunpcklqdq %ymm11, %ymm7, %ymm4 # ymm4 = ymm7[0],ymm11[0],ymm7[2],ymm11[2]+ vpunpckhqdq %ymm11, %ymm7, %ymm11 # ymm11 = ymm7[1],ymm11[1],ymm7[3],ymm11[3]+ vpermq $0x4e, 0x260(%rsi), %ymm2 # ymm2 = mem[2,3,0,1]+ vpermq $0x4e, 0x280(%rsi), %ymm7 # ymm7 = mem[2,3,0,1]+ vpsubw %ymm9, %ymm3, %ymm12+ vpaddw %ymm3, %ymm9, %ymm9+ vpsubw %ymm10, %ymm5, %ymm13+ vpmullw %ymm2, %ymm12, %ymm3+ vpaddw %ymm5, %ymm10, %ymm10+ vpsubw %ymm6, %ymm8, %ymm14+ vpmullw %ymm2, %ymm13, %ymm5+ vpaddw %ymm8, %ymm6, %ymm6+ vpsubw %ymm4, %ymm11, %ymm15+ vpmullw %ymm2, %ymm14, %ymm8+ vpaddw %ymm11, %ymm4, %ymm4+ vpmullw %ymm2, %ymm15, %ymm11+ vpmulhw %ymm7, %ymm12, %ymm12+ vpmulhw %ymm7, %ymm13, %ymm13+ vpmulhw %ymm7, %ymm14, %ymm14+ vpmulhw %ymm7, %ymm15, %ymm15+ vpmulhw %ymm0, %ymm3, %ymm3+ vpmulhw %ymm0, %ymm5, %ymm5+ vpmulhw %ymm0, %ymm8, %ymm8+ vpmulhw %ymm0, %ymm11, %ymm11+ vpsubw %ymm3, %ymm12, %ymm3+ vpsubw %ymm5, %ymm13, %ymm5+ vpsubw %ymm8, %ymm14, %ymm8+ vpsubw %ymm11, %ymm15, %ymm11+ vpmulhw %ymm1, %ymm9, %ymm12+ vpsraw $0xa, %ymm12, %ymm12+ vpmullw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm9, %ymm9+ vperm2i128 $0x20, %ymm10, %ymm9, %ymm7 # ymm7 = ymm9[0,1],ymm10[0,1]+ vperm2i128 $0x31, %ymm10, %ymm9, %ymm10 # ymm10 = ymm9[2,3],ymm10[2,3]+ vperm2i128 $0x20, %ymm4, %ymm6, %ymm9 # ymm9 = ymm6[0,1],ymm4[0,1]+ vperm2i128 $0x31, %ymm4, %ymm6, %ymm4 # ymm4 = ymm6[2,3],ymm4[2,3]+ vperm2i128 $0x20, %ymm5, %ymm3, %ymm6 # ymm6 = ymm3[0,1],ymm5[0,1]+ vperm2i128 $0x31, %ymm5, %ymm3, %ymm5 # ymm5 = ymm3[2,3],ymm5[2,3]+ vperm2i128 $0x20, %ymm11, %ymm8, %ymm3 # ymm3 = ymm8[0,1],ymm11[0,1]+ vperm2i128 $0x31, %ymm11, %ymm8, %ymm11 # ymm11 = ymm8[2,3],ymm11[2,3]+ vmovdqa 0x220(%rsi), %ymm2+ vmovdqa 0x240(%rsi), %ymm8+ vpsubw %ymm7, %ymm10, %ymm12+ vpaddw %ymm10, %ymm7, %ymm7+ vpsubw %ymm9, %ymm4, %ymm13+ vpmullw %ymm2, %ymm12, %ymm10+ vpaddw %ymm4, %ymm9, %ymm9+ vpsubw %ymm6, %ymm5, %ymm14+ vpmullw %ymm2, %ymm13, %ymm4+ vpaddw %ymm5, %ymm6, %ymm6+ vpsubw %ymm3, %ymm11, %ymm15+ vpmullw %ymm2, %ymm14, %ymm5+ vpaddw %ymm11, %ymm3, %ymm3+ vpmullw %ymm2, %ymm15, %ymm11+ vpmulhw %ymm8, %ymm12, %ymm12+ vpmulhw %ymm8, %ymm13, %ymm13+ vpmulhw %ymm8, %ymm14, %ymm14+ vpmulhw %ymm8, %ymm15, %ymm15+ vpmulhw %ymm0, %ymm10, %ymm10+ vpmulhw %ymm0, %ymm4, %ymm4+ vpmulhw %ymm0, %ymm5, %ymm5+ vpmulhw %ymm0, %ymm11, %ymm11+ vpsubw %ymm10, %ymm12, %ymm10+ vpsubw %ymm4, %ymm13, %ymm4+ vpsubw %ymm5, %ymm14, %ymm5+ vpsubw %ymm11, %ymm15, %ymm11+ vpmulhw %ymm1, %ymm7, %ymm12+ vpsraw $0xa, %ymm12, %ymm12+ vpmullw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm7, %ymm7+ vmovdqa %ymm7, (%rdi)+ vmovdqa %ymm9, 0x20(%rdi)+ vmovdqa %ymm6, 0x40(%rdi)+ vmovdqa %ymm3, 0x60(%rdi)+ vmovdqa %ymm10, 0x80(%rdi)+ vmovdqa %ymm4, 0xa0(%rdi)+ vmovdqa %ymm5, 0xc0(%rdi)+ vmovdqa %ymm11, 0xe0(%rdi)+ movl $0xd8a1d8a1, %eax # imm = 0xD8A1D8A1+ vmovd %eax, %xmm2+ vpbroadcastd %xmm2, %ymm2+ movl $0x5a105a1, %eax # imm = 0x5A105A1+ vmovd %eax, %xmm3+ vpbroadcastd %xmm3, %ymm3+ vmovdqa 0x100(%rdi), %ymm4+ vmovdqa 0x140(%rdi), %ymm6+ vmovdqa 0x120(%rdi), %ymm5+ vmovdqa 0x160(%rdi), %ymm7+ vpmullw %ymm2, %ymm4, %ymm12+ vpmulhw %ymm3, %ymm4, %ymm4+ vpmulhw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm4, %ymm4+ vpmullw %ymm2, %ymm6, %ymm12+ vpmulhw %ymm3, %ymm6, %ymm6+ vpmulhw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm6, %ymm6+ vpmullw %ymm2, %ymm5, %ymm12+ vpmulhw %ymm3, %ymm5, %ymm5+ vpmulhw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm5, %ymm5+ vpmullw %ymm2, %ymm7, %ymm12+ vpmulhw %ymm3, %ymm7, %ymm7+ vpmulhw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm7, %ymm7+ vmovdqa 0x180(%rdi), %ymm8+ vmovdqa 0x1c0(%rdi), %ymm10+ vmovdqa 0x1a0(%rdi), %ymm9+ vmovdqa 0x1e0(%rdi), %ymm11+ vpmullw %ymm2, %ymm8, %ymm12+ vpmulhw %ymm3, %ymm8, %ymm8+ vpmulhw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm8, %ymm8+ vpmullw %ymm2, %ymm10, %ymm12+ vpmulhw %ymm3, %ymm10, %ymm10+ vpmulhw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm10, %ymm10+ vpmullw %ymm2, %ymm9, %ymm12+ vpmulhw %ymm3, %ymm9, %ymm9+ vpmulhw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm9, %ymm9+ vpmullw %ymm2, %ymm11, %ymm12+ vpmulhw %ymm3, %ymm11, %ymm11+ vpmulhw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm11, %ymm11+ vpermq $0x4e, 0x1e0(%rsi), %ymm15 # ymm15 = mem[2,3,0,1]+ vpermq $0x4e, 0x1a0(%rsi), %ymm1 # ymm1 = mem[2,3,0,1]+ vpermq $0x4e, 0x200(%rsi), %ymm2 # ymm2 = mem[2,3,0,1]+ vpermq $0x4e, 0x1c0(%rsi), %ymm3 # ymm3 = mem[2,3,0,1]+ vmovdqa (%rsi), %ymm12+ vpshufb %ymm12, %ymm15, %ymm15+ vpshufb %ymm12, %ymm1, %ymm1+ vpshufb %ymm12, %ymm2, %ymm2+ vpshufb %ymm12, %ymm3, %ymm3+ vpsubw %ymm4, %ymm6, %ymm12+ vpaddw %ymm6, %ymm4, %ymm4+ vpsubw %ymm5, %ymm7, %ymm13+ vpmullw %ymm15, %ymm12, %ymm6+ vpaddw %ymm7, %ymm5, %ymm5+ vpsubw %ymm8, %ymm10, %ymm14+ vpmullw %ymm15, %ymm13, %ymm7+ vpaddw %ymm10, %ymm8, %ymm8+ vpsubw %ymm9, %ymm11, %ymm15+ vpmullw %ymm1, %ymm14, %ymm10+ vpaddw %ymm11, %ymm9, %ymm9+ vpmullw %ymm1, %ymm15, %ymm11+ vpmulhw %ymm2, %ymm12, %ymm12+ vpmulhw %ymm2, %ymm13, %ymm13+ vpmulhw %ymm3, %ymm14, %ymm14+ vpmulhw %ymm3, %ymm15, %ymm15+ vpmulhw %ymm0, %ymm6, %ymm6+ vpmulhw %ymm0, %ymm7, %ymm7+ vpmulhw %ymm0, %ymm10, %ymm10+ vpmulhw %ymm0, %ymm11, %ymm11+ vpsubw %ymm6, %ymm12, %ymm6+ vpsubw %ymm7, %ymm13, %ymm7+ vpsubw %ymm10, %ymm14, %ymm10+ vpsubw %ymm11, %ymm15, %ymm11+ vpermq $0x4e, 0x160(%rsi), %ymm2 # ymm2 = mem[2,3,0,1]+ vpermq $0x4e, 0x180(%rsi), %ymm3 # ymm3 = mem[2,3,0,1]+ vmovdqa (%rsi), %ymm1+ vpshufb %ymm1, %ymm2, %ymm2+ vpshufb %ymm1, %ymm3, %ymm3+ vpsubw %ymm4, %ymm8, %ymm12+ vpaddw %ymm8, %ymm4, %ymm4+ vpsubw %ymm5, %ymm9, %ymm13+ vpmullw %ymm2, %ymm12, %ymm8+ vpaddw %ymm9, %ymm5, %ymm5+ vpsubw %ymm6, %ymm10, %ymm14+ vpmullw %ymm2, %ymm13, %ymm9+ vpaddw %ymm10, %ymm6, %ymm6+ vpsubw %ymm7, %ymm11, %ymm15+ vpmullw %ymm2, %ymm14, %ymm10+ vpaddw %ymm11, %ymm7, %ymm7+ vpmullw %ymm2, %ymm15, %ymm11+ vpmulhw %ymm3, %ymm12, %ymm12+ vpmulhw %ymm3, %ymm13, %ymm13+ vpmulhw %ymm3, %ymm14, %ymm14+ vpmulhw %ymm3, %ymm15, %ymm15+ vpmulhw %ymm0, %ymm8, %ymm8+ vpmulhw %ymm0, %ymm9, %ymm9+ vpmulhw %ymm0, %ymm10, %ymm10+ vpmulhw %ymm0, %ymm11, %ymm11+ vpsubw %ymm8, %ymm12, %ymm8+ vpsubw %ymm9, %ymm13, %ymm9+ vpsubw %ymm10, %ymm14, %ymm10+ vpsubw %ymm11, %ymm15, %ymm11+ vpslld $0x10, %ymm5, %ymm3+ vpblendw $0xaa, %ymm3, %ymm4, %ymm3 # ymm3 = ymm4[0],ymm3[1],ymm4[2],ymm3[3],ymm4[4],ymm3[5],ymm4[6],ymm3[7],ymm4[8],ymm3[9],ymm4[10],ymm3[11],ymm4[12],ymm3[13],ymm4[14],ymm3[15]+ vpsrld $0x10, %ymm4, %ymm4+ vpblendw $0xaa, %ymm5, %ymm4, %ymm5 # ymm5 = ymm4[0],ymm5[1],ymm4[2],ymm5[3],ymm4[4],ymm5[5],ymm4[6],ymm5[7],ymm4[8],ymm5[9],ymm4[10],ymm5[11],ymm4[12],ymm5[13],ymm4[14],ymm5[15]+ vpslld $0x10, %ymm7, %ymm4+ vpblendw $0xaa, %ymm4, %ymm6, %ymm4 # ymm4 = ymm6[0],ymm4[1],ymm6[2],ymm4[3],ymm6[4],ymm4[5],ymm6[6],ymm4[7],ymm6[8],ymm4[9],ymm6[10],ymm4[11],ymm6[12],ymm4[13],ymm6[14],ymm4[15]+ vpsrld $0x10, %ymm6, %ymm6+ vpblendw $0xaa, %ymm7, %ymm6, %ymm7 # ymm7 = ymm6[0],ymm7[1],ymm6[2],ymm7[3],ymm6[4],ymm7[5],ymm6[6],ymm7[7],ymm6[8],ymm7[9],ymm6[10],ymm7[11],ymm6[12],ymm7[13],ymm6[14],ymm7[15]+ vpslld $0x10, %ymm9, %ymm6+ vpblendw $0xaa, %ymm6, %ymm8, %ymm6 # ymm6 = ymm8[0],ymm6[1],ymm8[2],ymm6[3],ymm8[4],ymm6[5],ymm8[6],ymm6[7],ymm8[8],ymm6[9],ymm8[10],ymm6[11],ymm8[12],ymm6[13],ymm8[14],ymm6[15]+ vpsrld $0x10, %ymm8, %ymm8+ vpblendw $0xaa, %ymm9, %ymm8, %ymm9 # ymm9 = ymm8[0],ymm9[1],ymm8[2],ymm9[3],ymm8[4],ymm9[5],ymm8[6],ymm9[7],ymm8[8],ymm9[9],ymm8[10],ymm9[11],ymm8[12],ymm9[13],ymm8[14],ymm9[15]+ vpslld $0x10, %ymm11, %ymm8+ vpblendw $0xaa, %ymm8, %ymm10, %ymm8 # ymm8 = ymm10[0],ymm8[1],ymm10[2],ymm8[3],ymm10[4],ymm8[5],ymm10[6],ymm8[7],ymm10[8],ymm8[9],ymm10[10],ymm8[11],ymm10[12],ymm8[13],ymm10[14],ymm8[15]+ vpsrld $0x10, %ymm10, %ymm10+ vpblendw $0xaa, %ymm11, %ymm10, %ymm11 # ymm11 = ymm10[0],ymm11[1],ymm10[2],ymm11[3],ymm10[4],ymm11[5],ymm10[6],ymm11[7],ymm10[8],ymm11[9],ymm10[10],ymm11[11],ymm10[12],ymm11[13],ymm10[14],ymm11[15]+ vmovdqa 0x20(%rsi), %ymm12+ vpermd 0x120(%rsi), %ymm12, %ymm2+ vpermd 0x140(%rsi), %ymm12, %ymm10+ vpsubw %ymm3, %ymm5, %ymm12+ vpaddw %ymm5, %ymm3, %ymm3+ vpsubw %ymm4, %ymm7, %ymm13+ vpmullw %ymm2, %ymm12, %ymm5+ vpaddw %ymm7, %ymm4, %ymm4+ vpsubw %ymm6, %ymm9, %ymm14+ vpmullw %ymm2, %ymm13, %ymm7+ vpaddw %ymm9, %ymm6, %ymm6+ vpsubw %ymm8, %ymm11, %ymm15+ vpmullw %ymm2, %ymm14, %ymm9+ vpaddw %ymm11, %ymm8, %ymm8+ vpmullw %ymm2, %ymm15, %ymm11+ vpmulhw %ymm10, %ymm12, %ymm12+ vpmulhw %ymm10, %ymm13, %ymm13+ vpmulhw %ymm10, %ymm14, %ymm14+ vpmulhw %ymm10, %ymm15, %ymm15+ vpmulhw %ymm0, %ymm5, %ymm5+ vpmulhw %ymm0, %ymm7, %ymm7+ vpmulhw %ymm0, %ymm9, %ymm9+ vpmulhw %ymm0, %ymm11, %ymm11+ vpsubw %ymm5, %ymm12, %ymm5+ vpsubw %ymm7, %ymm13, %ymm7+ vpsubw %ymm9, %ymm14, %ymm9+ vpsubw %ymm11, %ymm15, %ymm11+ movl $0x4ebf4ebf, %eax # imm = 0x4EBF4EBF+ vmovd %eax, %xmm1+ vpbroadcastd %xmm1, %ymm1+ vpmulhw %ymm1, %ymm3, %ymm12+ vpsraw $0xa, %ymm12, %ymm12+ vpmullw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm3, %ymm3+ vmovsldup %ymm4, %ymm10 # ymm10 = ymm4[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm10, %ymm3, %ymm10 # ymm10 = ymm3[0],ymm10[1],ymm3[2],ymm10[3],ymm3[4],ymm10[5],ymm3[6],ymm10[7]+ vpsrlq $0x20, %ymm3, %ymm3+ vpblendd $0xaa, %ymm4, %ymm3, %ymm4 # ymm4 = ymm3[0],ymm4[1],ymm3[2],ymm4[3],ymm3[4],ymm4[5],ymm3[6],ymm4[7]+ vmovsldup %ymm8, %ymm3 # ymm3 = ymm8[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm3, %ymm6, %ymm3 # ymm3 = ymm6[0],ymm3[1],ymm6[2],ymm3[3],ymm6[4],ymm3[5],ymm6[6],ymm3[7]+ vpsrlq $0x20, %ymm6, %ymm6+ vpblendd $0xaa, %ymm8, %ymm6, %ymm8 # ymm8 = ymm6[0],ymm8[1],ymm6[2],ymm8[3],ymm6[4],ymm8[5],ymm6[6],ymm8[7]+ vmovsldup %ymm7, %ymm6 # ymm6 = ymm7[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm6, %ymm5, %ymm6 # ymm6 = ymm5[0],ymm6[1],ymm5[2],ymm6[3],ymm5[4],ymm6[5],ymm5[6],ymm6[7]+ vpsrlq $0x20, %ymm5, %ymm5+ vpblendd $0xaa, %ymm7, %ymm5, %ymm7 # ymm7 = ymm5[0],ymm7[1],ymm5[2],ymm7[3],ymm5[4],ymm7[5],ymm5[6],ymm7[7]+ vmovsldup %ymm11, %ymm5 # ymm5 = ymm11[0,0,2,2,4,4,6,6]+ vpblendd $0xaa, %ymm5, %ymm9, %ymm5 # ymm5 = ymm9[0],ymm5[1],ymm9[2],ymm5[3],ymm9[4],ymm5[5],ymm9[6],ymm5[7]+ vpsrlq $0x20, %ymm9, %ymm9+ vpblendd $0xaa, %ymm11, %ymm9, %ymm11 # ymm11 = ymm9[0],ymm11[1],ymm9[2],ymm11[3],ymm9[4],ymm11[5],ymm9[6],ymm11[7]+ vpermq $0x1b, 0xe0(%rsi), %ymm2 # ymm2 = mem[3,2,1,0]+ vpermq $0x1b, 0x100(%rsi), %ymm9 # ymm9 = mem[3,2,1,0]+ vpsubw %ymm10, %ymm4, %ymm12+ vpaddw %ymm4, %ymm10, %ymm10+ vpsubw %ymm3, %ymm8, %ymm13+ vpmullw %ymm2, %ymm12, %ymm4+ vpaddw %ymm8, %ymm3, %ymm3+ vpsubw %ymm6, %ymm7, %ymm14+ vpmullw %ymm2, %ymm13, %ymm8+ vpaddw %ymm7, %ymm6, %ymm6+ vpsubw %ymm5, %ymm11, %ymm15+ vpmullw %ymm2, %ymm14, %ymm7+ vpaddw %ymm11, %ymm5, %ymm5+ vpmullw %ymm2, %ymm15, %ymm11+ vpmulhw %ymm9, %ymm12, %ymm12+ vpmulhw %ymm9, %ymm13, %ymm13+ vpmulhw %ymm9, %ymm14, %ymm14+ vpmulhw %ymm9, %ymm15, %ymm15+ vpmulhw %ymm0, %ymm4, %ymm4+ vpmulhw %ymm0, %ymm8, %ymm8+ vpmulhw %ymm0, %ymm7, %ymm7+ vpmulhw %ymm0, %ymm11, %ymm11+ vpsubw %ymm4, %ymm12, %ymm4+ vpsubw %ymm8, %ymm13, %ymm8+ vpsubw %ymm7, %ymm14, %ymm7+ vpsubw %ymm11, %ymm15, %ymm11+ vpmulhw %ymm1, %ymm10, %ymm12+ vpsraw $0xa, %ymm12, %ymm12+ vpmullw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm10, %ymm10+ vpunpcklqdq %ymm3, %ymm10, %ymm9 # ymm9 = ymm10[0],ymm3[0],ymm10[2],ymm3[2]+ vpunpckhqdq %ymm3, %ymm10, %ymm3 # ymm3 = ymm10[1],ymm3[1],ymm10[3],ymm3[3]+ vpunpcklqdq %ymm5, %ymm6, %ymm10 # ymm10 = ymm6[0],ymm5[0],ymm6[2],ymm5[2]+ vpunpckhqdq %ymm5, %ymm6, %ymm5 # ymm5 = ymm6[1],ymm5[1],ymm6[3],ymm5[3]+ vpunpcklqdq %ymm8, %ymm4, %ymm6 # ymm6 = ymm4[0],ymm8[0],ymm4[2],ymm8[2]+ vpunpckhqdq %ymm8, %ymm4, %ymm8 # ymm8 = ymm4[1],ymm8[1],ymm4[3],ymm8[3]+ vpunpcklqdq %ymm11, %ymm7, %ymm4 # ymm4 = ymm7[0],ymm11[0],ymm7[2],ymm11[2]+ vpunpckhqdq %ymm11, %ymm7, %ymm11 # ymm11 = ymm7[1],ymm11[1],ymm7[3],ymm11[3]+ vpermq $0x4e, 0xa0(%rsi), %ymm2 # ymm2 = mem[2,3,0,1]+ vpermq $0x4e, 0xc0(%rsi), %ymm7 # ymm7 = mem[2,3,0,1]+ vpsubw %ymm9, %ymm3, %ymm12+ vpaddw %ymm3, %ymm9, %ymm9+ vpsubw %ymm10, %ymm5, %ymm13+ vpmullw %ymm2, %ymm12, %ymm3+ vpaddw %ymm5, %ymm10, %ymm10+ vpsubw %ymm6, %ymm8, %ymm14+ vpmullw %ymm2, %ymm13, %ymm5+ vpaddw %ymm8, %ymm6, %ymm6+ vpsubw %ymm4, %ymm11, %ymm15+ vpmullw %ymm2, %ymm14, %ymm8+ vpaddw %ymm11, %ymm4, %ymm4+ vpmullw %ymm2, %ymm15, %ymm11+ vpmulhw %ymm7, %ymm12, %ymm12+ vpmulhw %ymm7, %ymm13, %ymm13+ vpmulhw %ymm7, %ymm14, %ymm14+ vpmulhw %ymm7, %ymm15, %ymm15+ vpmulhw %ymm0, %ymm3, %ymm3+ vpmulhw %ymm0, %ymm5, %ymm5+ vpmulhw %ymm0, %ymm8, %ymm8+ vpmulhw %ymm0, %ymm11, %ymm11+ vpsubw %ymm3, %ymm12, %ymm3+ vpsubw %ymm5, %ymm13, %ymm5+ vpsubw %ymm8, %ymm14, %ymm8+ vpsubw %ymm11, %ymm15, %ymm11+ vpmulhw %ymm1, %ymm9, %ymm12+ vpsraw $0xa, %ymm12, %ymm12+ vpmullw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm9, %ymm9+ vperm2i128 $0x20, %ymm10, %ymm9, %ymm7 # ymm7 = ymm9[0,1],ymm10[0,1]+ vperm2i128 $0x31, %ymm10, %ymm9, %ymm10 # ymm10 = ymm9[2,3],ymm10[2,3]+ vperm2i128 $0x20, %ymm4, %ymm6, %ymm9 # ymm9 = ymm6[0,1],ymm4[0,1]+ vperm2i128 $0x31, %ymm4, %ymm6, %ymm4 # ymm4 = ymm6[2,3],ymm4[2,3]+ vperm2i128 $0x20, %ymm5, %ymm3, %ymm6 # ymm6 = ymm3[0,1],ymm5[0,1]+ vperm2i128 $0x31, %ymm5, %ymm3, %ymm5 # ymm5 = ymm3[2,3],ymm5[2,3]+ vperm2i128 $0x20, %ymm11, %ymm8, %ymm3 # ymm3 = ymm8[0,1],ymm11[0,1]+ vperm2i128 $0x31, %ymm11, %ymm8, %ymm11 # ymm11 = ymm8[2,3],ymm11[2,3]+ vmovdqa 0x60(%rsi), %ymm2+ vmovdqa 0x80(%rsi), %ymm8+ vpsubw %ymm7, %ymm10, %ymm12+ vpaddw %ymm10, %ymm7, %ymm7+ vpsubw %ymm9, %ymm4, %ymm13+ vpmullw %ymm2, %ymm12, %ymm10+ vpaddw %ymm4, %ymm9, %ymm9+ vpsubw %ymm6, %ymm5, %ymm14+ vpmullw %ymm2, %ymm13, %ymm4+ vpaddw %ymm5, %ymm6, %ymm6+ vpsubw %ymm3, %ymm11, %ymm15+ vpmullw %ymm2, %ymm14, %ymm5+ vpaddw %ymm11, %ymm3, %ymm3+ vpmullw %ymm2, %ymm15, %ymm11+ vpmulhw %ymm8, %ymm12, %ymm12+ vpmulhw %ymm8, %ymm13, %ymm13+ vpmulhw %ymm8, %ymm14, %ymm14+ vpmulhw %ymm8, %ymm15, %ymm15+ vpmulhw %ymm0, %ymm10, %ymm10+ vpmulhw %ymm0, %ymm4, %ymm4+ vpmulhw %ymm0, %ymm5, %ymm5+ vpmulhw %ymm0, %ymm11, %ymm11+ vpsubw %ymm10, %ymm12, %ymm10+ vpsubw %ymm4, %ymm13, %ymm4+ vpsubw %ymm5, %ymm14, %ymm5+ vpsubw %ymm11, %ymm15, %ymm11+ vpmulhw %ymm1, %ymm7, %ymm12+ vpsraw $0xa, %ymm12, %ymm12+ vpmullw %ymm0, %ymm12, %ymm12+ vpsubw %ymm12, %ymm7, %ymm7+ vmovdqa %ymm7, 0x100(%rdi)+ vmovdqa %ymm9, 0x120(%rdi)+ vmovdqa %ymm6, 0x140(%rdi)+ vmovdqa %ymm3, 0x160(%rdi)+ vmovdqa %ymm10, 0x180(%rdi)+ vmovdqa %ymm4, 0x1a0(%rdi)+ vmovdqa %ymm5, 0x1c0(%rdi)+ vmovdqa %ymm11, 0x1e0(%rdi)+ vmovdqa (%rdi), %ymm4+ vmovdqa 0x100(%rdi), %ymm8+ vmovdqa 0x20(%rdi), %ymm5+ vmovdqa 0x120(%rdi), %ymm9+ vpbroadcastq 0x40(%rsi), %ymm2+ vmovdqa 0x40(%rdi), %ymm6+ vmovdqa 0x140(%rdi), %ymm10+ vmovdqa 0x60(%rdi), %ymm7+ vmovdqa 0x160(%rdi), %ymm11+ vpbroadcastq 0x48(%rsi), %ymm3+ vpsubw %ymm4, %ymm8, %ymm12+ vpaddw %ymm8, %ymm4, %ymm4+ vpsubw %ymm5, %ymm9, %ymm13+ vpmullw %ymm2, %ymm12, %ymm8+ vpaddw %ymm9, %ymm5, %ymm5+ vpsubw %ymm6, %ymm10, %ymm14+ vpmullw %ymm2, %ymm13, %ymm9+ vpaddw %ymm10, %ymm6, %ymm6+ vpsubw %ymm7, %ymm11, %ymm15+ vpmullw %ymm2, %ymm14, %ymm10+ vpaddw %ymm11, %ymm7, %ymm7+ vpmullw %ymm2, %ymm15, %ymm11+ vpmulhw %ymm3, %ymm12, %ymm12+ vpmulhw %ymm3, %ymm13, %ymm13+ vpmulhw %ymm3, %ymm14, %ymm14+ vpmulhw %ymm3, %ymm15, %ymm15+ vpmulhw %ymm0, %ymm8, %ymm8+ vpmulhw %ymm0, %ymm9, %ymm9+ vpmulhw %ymm0, %ymm10, %ymm10+ vpmulhw %ymm0, %ymm11, %ymm11+ vpsubw %ymm8, %ymm12, %ymm8+ vpsubw %ymm9, %ymm13, %ymm9+ vpsubw %ymm10, %ymm14, %ymm10+ vpsubw %ymm11, %ymm15, %ymm11+ vmovdqa %ymm4, (%rdi)+ vmovdqa %ymm5, 0x20(%rdi)+ vmovdqa %ymm6, 0x40(%rdi)+ vmovdqa %ymm7, 0x60(%rdi)+ vmovdqa %ymm8, 0x100(%rdi)+ vmovdqa %ymm9, 0x120(%rdi)+ vmovdqa %ymm10, 0x140(%rdi)+ vmovdqa %ymm11, 0x160(%rdi)+ vmovdqa 0x80(%rdi), %ymm4+ vmovdqa 0x180(%rdi), %ymm8+ vmovdqa 0xa0(%rdi), %ymm5+ vmovdqa 0x1a0(%rdi), %ymm9+ vpbroadcastq 0x40(%rsi), %ymm2+ vmovdqa 0xc0(%rdi), %ymm6+ vmovdqa 0x1c0(%rdi), %ymm10+ vmovdqa 0xe0(%rdi), %ymm7+ vmovdqa 0x1e0(%rdi), %ymm11+ vpbroadcastq 0x48(%rsi), %ymm3+ vpsubw %ymm4, %ymm8, %ymm12+ vpaddw %ymm8, %ymm4, %ymm4+ vpsubw %ymm5, %ymm9, %ymm13+ vpmullw %ymm2, %ymm12, %ymm8+ vpaddw %ymm9, %ymm5, %ymm5+ vpsubw %ymm6, %ymm10, %ymm14+ vpmullw %ymm2, %ymm13, %ymm9+ vpaddw %ymm10, %ymm6, %ymm6+ vpsubw %ymm7, %ymm11, %ymm15+ vpmullw %ymm2, %ymm14, %ymm10+ vpaddw %ymm11, %ymm7, %ymm7+ vpmullw %ymm2, %ymm15, %ymm11+ vpmulhw %ymm3, %ymm12, %ymm12+ vpmulhw %ymm3, %ymm13, %ymm13+ vpmulhw %ymm3, %ymm14, %ymm14+ vpmulhw %ymm3, %ymm15, %ymm15+ vpmulhw %ymm0, %ymm8, %ymm8+ vpmulhw %ymm0, %ymm9, %ymm9+ vpmulhw %ymm0, %ymm10, %ymm10+ vpmulhw %ymm0, %ymm11, %ymm11+ vpsubw %ymm8, %ymm12, %ymm8+ vpsubw %ymm9, %ymm13, %ymm9+ vpsubw %ymm10, %ymm14, %ymm10+ vpsubw %ymm11, %ymm15, %ymm11+ vmovdqa %ymm4, 0x80(%rdi)+ vmovdqa %ymm5, 0xa0(%rdi)+ vmovdqa %ymm6, 0xc0(%rdi)+ vmovdqa %ymm7, 0xe0(%rdi)+ vmovdqa %ymm8, 0x180(%rdi)+ vmovdqa %ymm9, 0x1a0(%rdi)+ vmovdqa %ymm10, 0x1c0(%rdi)+ vmovdqa %ymm11, 0x1e0(%rdi)+ retq+ .cfi_endproc++MLK_ASM_FN_SIZE(invntt_avx2_asm)++#endif /* MLK_ARITH_BACKEND_X86_64_DEFAULT && !MLK_CONFIG_MULTILEVEL_NO_SHARED \+ && (!MLK_CONFIG_NO_ENCAPS_API || !MLK_CONFIG_NO_DECAPS_API) */++#if defined(__ELF__)+.section .note.GNU-stack,"",%progbits+#endif
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff
file too large to diff